@ngockhoale/ukit 2.2.14 → 2.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +69 -0
- package/manifests/platform.full.yaml +11 -13
- package/package.json +1 -1
- package/src/core/executionContracts.js +130 -0
- package/src/core/runtimeConfig.js +13 -50
- package/src/index/taskRouting.js +10 -103
- package/templates/.claude/agents/ukit-vision-analyst.md +32 -21
- package/templates/.claude/hooks/context-hardcap-gate.sh +2 -2
- package/templates/.claude/hooks/protect-files.sh +1 -0
- package/templates/.claude/hooks/sensitive-data-guard.sh +269 -0
- package/templates/.claude/hooks/skill-router.sh +6 -0
- package/templates/.claude/hooks/vision-router.sh +67 -47
- package/templates/.claude/settings.json +17 -6
- package/templates/.claude/ukit/index/extract-image.mjs +18 -8
- package/templates/.claude/ukit/index/provision-worktree.mjs +1 -1
- package/templates/.claude/ukit/index/route-task.mjs +53 -2
- package/templates/.claude/ukit/index/unic-gateway.mjs +1 -1
- package/templates/.claude/ukit/runtime/execution-ledger.mjs +120 -21
- package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +8 -5
- package/templates/.omp/README.md +3 -3
- package/templates/.omp/hooks/pre/ukit-bridge.js +15 -7
- package/templates/AGENTS.md +2 -2
- package/templates/CLAUDE.md +2 -2
- package/templates/ukit/storage/config.json +8 -7
- package/src/core/router/advisor.js +0 -42
- package/src/core/router/router.js +0 -180
- package/src/core/validation/confidence.js +0 -89
- package/src/core/validation/validator.js +0 -165
- package/templates/.claude/hooks/vision-gate.sh +0 -230
- package/templates/docs/INSTALL.md +0 -115
- package/templates/docs/STATUS.md +0 -81
- package/templates/docs/TASKS.md +0 -79
- package/templates/docs/UKIT_USAGE_GUIDE.md +0 -163
|
@@ -579,6 +579,9 @@ function printRouteState(state) {
|
|
|
579
579
|
if (state.routeSummary?.executionContract?.modelTier) {
|
|
580
580
|
console.log(`modelTier: ${state.routeSummary.executionContract.modelTier}`);
|
|
581
581
|
}
|
|
582
|
+
if (state.routeSummary?.tierLane?.instruction) {
|
|
583
|
+
console.log(`tierLane: ${state.routeSummary.tierLane.instruction}`);
|
|
584
|
+
}
|
|
582
585
|
if (state.routeSummary?.escalatedTier) {
|
|
583
586
|
console.log(`escalatedTier: ${state.routeSummary.escalatedTier}`);
|
|
584
587
|
if (state.routeSummary.escalationReason) {
|
|
@@ -1475,6 +1478,15 @@ function buildRouteSummary({
|
|
|
1475
1478
|
const primaryTargets = summarizeCompactList(preview.primaryTargets ?? [], 2);
|
|
1476
1479
|
const relatedTests = summarizeCompactList(preview.relatedTests ?? [], 2);
|
|
1477
1480
|
const styleFiles = summarizeCompactList(preview.styleFiles ?? [], 1);
|
|
1481
|
+
// WS-C target-aware evidence: the completion ledger checks impact-evidence against these
|
|
1482
|
+
// files, so carry the full resolver list (target + primaryTargets + relatedTests), bounded
|
|
1483
|
+
// for compact route state. Absent when the route has no indexed context — the ledger then
|
|
1484
|
+
// falls back to its legacy any-read behaviour.
|
|
1485
|
+
const expectedSourceFiles = unique([
|
|
1486
|
+
...(routingContext.targetFile ? [routingContext.targetFile] : []),
|
|
1487
|
+
...(preview.primaryTargets ?? []),
|
|
1488
|
+
...(preview.relatedTests ?? []),
|
|
1489
|
+
].filter((filePath) => typeof filePath === 'string' && filePath.trim())).slice(0, 8);
|
|
1478
1490
|
const primaryCommands = unique(verificationRecommendation?.commands ?? []);
|
|
1479
1491
|
const fallbackCommands = unique(verificationRecommendation?.fallbackCommands ?? []);
|
|
1480
1492
|
const preferredOrder = unique(
|
|
@@ -1497,6 +1509,7 @@ function buildRouteSummary({
|
|
|
1497
1509
|
executionCandidates,
|
|
1498
1510
|
});
|
|
1499
1511
|
const executionContract = buildExecutionContract(executionMode);
|
|
1512
|
+
const tierLane = buildTierLane({ modelTier: executionContract?.modelTier ?? null });
|
|
1500
1513
|
const completionState = buildCompletionState({
|
|
1501
1514
|
executionMode,
|
|
1502
1515
|
verificationRecommendation,
|
|
@@ -1539,8 +1552,10 @@ function buildRouteSummary({
|
|
|
1539
1552
|
executionCandidates,
|
|
1540
1553
|
approachSelector,
|
|
1541
1554
|
executionContract,
|
|
1555
|
+
tierLane,
|
|
1542
1556
|
completionState,
|
|
1543
1557
|
continuationState,
|
|
1558
|
+
...(expectedSourceFiles.length > 0 ? { expectedSourceFiles } : {}),
|
|
1544
1559
|
intentMode: routingContext.intentMode ?? null,
|
|
1545
1560
|
handoffFile,
|
|
1546
1561
|
delegateHint: delegationRecommendation?.hint ?? null,
|
|
@@ -1894,9 +1909,10 @@ function buildExecutionContract(executionMode = null) {
|
|
|
1894
1909
|
maxReadPasses: 2,
|
|
1895
1910
|
maxContextPulls: 1,
|
|
1896
1911
|
verificationPolicy: 'targeted-if-covered',
|
|
1897
|
-
completionRule: 'require-write',
|
|
1912
|
+
completionRule: 'require-write-and-verification',
|
|
1898
1913
|
delegationPolicy: 'disallow-by-default',
|
|
1899
|
-
completionEvidence: ['write-evidence'],
|
|
1914
|
+
completionEvidence: ['write-evidence', 'verification-evidence'],
|
|
1915
|
+
postEditReviewPolicy: 'sidecar-non-blocking',
|
|
1900
1916
|
},
|
|
1901
1917
|
'find-cause': {
|
|
1902
1918
|
modelTier: 'code',
|
|
@@ -1915,6 +1931,7 @@ function buildExecutionContract(executionMode = null) {
|
|
|
1915
1931
|
delegationPolicy: 'allow-qualified-sidecar',
|
|
1916
1932
|
completionEvidence: ['write-evidence', 'verification-evidence'],
|
|
1917
1933
|
mirrorConsistencyRequired: true,
|
|
1934
|
+
postEditReviewPolicy: 'sidecar-non-blocking',
|
|
1918
1935
|
},
|
|
1919
1936
|
'map-impact': {
|
|
1920
1937
|
modelTier: 'code',
|
|
@@ -2182,10 +2199,36 @@ function applyEscalationToRouteSummary(routeSummary = null, { config = null, tar
|
|
|
2182
2199
|
delete routeSummary.escalatedTier;
|
|
2183
2200
|
delete routeSummary.escalationReason;
|
|
2184
2201
|
}
|
|
2202
|
+
// Recomputed here (not only in buildRouteSummary) so the lane always reflects the final
|
|
2203
|
+
// tier after escalation, on every path that emits a route summary.
|
|
2204
|
+
routeSummary.tierLane = buildTierLane({
|
|
2205
|
+
modelTier: routeSummary.executionContract?.modelTier ?? null,
|
|
2206
|
+
escalatedTier: escalation?.escalatedTier ?? null,
|
|
2207
|
+
});
|
|
2185
2208
|
|
|
2186
2209
|
return routeSummary;
|
|
2187
2210
|
}
|
|
2188
2211
|
|
|
2212
|
+
// WS-E tier-role wiring: a contract tier only takes effect when the executing session hands
|
|
2213
|
+
// the work to an agent whose definition binds that tier — otherwise every session behaves as
|
|
2214
|
+
// if it only had one model. `code` is the DEFAULT implementation lane, so a code-tier contract
|
|
2215
|
+
// produces no instruction (inline execution on a code session is already correct, and silence
|
|
2216
|
+
// keeps the hook flow smooth). Labels are stable tier roles (lite/code/smart), never provider
|
|
2217
|
+
// or model identity. Vision never appears here — it is a capability lane, not a cost tier.
|
|
2218
|
+
function buildTierLane({ modelTier = null, escalatedTier = null } = {}) {
|
|
2219
|
+
const tier = escalatedTier ?? modelTier;
|
|
2220
|
+
if (!tier || tier === 'vision') {
|
|
2221
|
+
return null;
|
|
2222
|
+
}
|
|
2223
|
+
return {
|
|
2224
|
+
tier,
|
|
2225
|
+
defaultLane: tier === 'code',
|
|
2226
|
+
instruction: tier === 'code'
|
|
2227
|
+
? null
|
|
2228
|
+
: `Hand this task to an agent bound to the ${tier} tier instead of doing it inline, unless this session already runs on the ${tier} tier.`,
|
|
2229
|
+
};
|
|
2230
|
+
}
|
|
2231
|
+
|
|
2189
2232
|
function advanceContinuationState(continuationState = null, previousContinuationState = null, thresholds = null) {
|
|
2190
2233
|
if (!continuationState || typeof continuationState !== 'object') {
|
|
2191
2234
|
return continuationState;
|
|
@@ -2875,6 +2918,14 @@ function compactRouteSummary(routeSummary = null) {
|
|
|
2875
2918
|
completionState: routeSummary.completionState ?? null,
|
|
2876
2919
|
continuationState: routeSummary.continuationState ?? null,
|
|
2877
2920
|
delegateHint: routeSummary.delegateHint ?? null,
|
|
2921
|
+
// WS-E tier-role binding: consumers need the hand-off instruction, not just the raw tier.
|
|
2922
|
+
...(routeSummary.tierLane ? { tierLane: routeSummary.tierLane } : {}),
|
|
2923
|
+
// WS-C target-aware evidence: the completion ledger reads these from the PERSISTED route
|
|
2924
|
+
// state — without this spread the CLI path never delivers them and the gate falls back
|
|
2925
|
+
// to legacy any-read behaviour.
|
|
2926
|
+
...(Array.isArray(routeSummary.expectedSourceFiles) && routeSummary.expectedSourceFiles.length > 0 ? {
|
|
2927
|
+
expectedSourceFiles: routeSummary.expectedSourceFiles,
|
|
2928
|
+
} : {}),
|
|
2878
2929
|
nextActionType: routeSummary.nextActionType ?? null,
|
|
2879
2930
|
nextActionCommand: routeSummary.nextActionCommand ?? null,
|
|
2880
2931
|
helperHint: routeSummary.helperHint ?? null,
|
|
@@ -22,7 +22,7 @@
|
|
|
22
22
|
* Every probe here is scoped to what changes the CALLING
|
|
23
23
|
* runtime's OWN outbound endpoint — same rule that already governs probes 1-3. Cross-tool
|
|
24
24
|
* configs are NOT probed. This module is only ever invoked from Claude Code / omp hooks
|
|
25
|
-
* (vision-router.sh,
|
|
25
|
+
* (vision-router.sh, route-task.mjs, and the omp bridge) to gate what THIS
|
|
26
26
|
* session does — a Codex `config.toml` `base_url`, a Kilo `secrets.json` endpoint, or an
|
|
27
27
|
* `OPENAI_BASE_URL` env var describe a completely different tool's outbound endpoint and say
|
|
28
28
|
* nothing about where Claude Code or omp itself is sending requests. Treating them as evidence
|
|
@@ -8,6 +8,7 @@ import { fileURLToPath } from 'node:url';
|
|
|
8
8
|
|
|
9
9
|
const LEDGER_VERSION = 1;
|
|
10
10
|
const MAX_RECEIPTS = 24;
|
|
11
|
+
const MAX_SOURCE_FILES = 16;
|
|
11
12
|
const MAX_CONTINUATIONS = 6;
|
|
12
13
|
const IMPLEMENT_MODES = new Set([
|
|
13
14
|
'tiny-fix',
|
|
@@ -130,7 +131,7 @@ function compactReceipt(receipt) {
|
|
|
130
131
|
kind: receipt.kind,
|
|
131
132
|
success: receipt.success,
|
|
132
133
|
};
|
|
133
|
-
for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'error']) {
|
|
134
|
+
for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error']) {
|
|
134
135
|
if (receipt[key] !== undefined && receipt[key] !== null && receipt[key] !== '') {
|
|
135
136
|
compact[key] = receipt[key];
|
|
136
137
|
}
|
|
@@ -138,6 +139,38 @@ function compactReceipt(receipt) {
|
|
|
138
139
|
return compact;
|
|
139
140
|
}
|
|
140
141
|
|
|
142
|
+
/**
|
|
143
|
+
* Target-aware matching between a receipt's file and a routed expected file. Both sides
|
|
144
|
+
* may disagree about absolute vs relative spelling (hook payloads carry absolute paths,
|
|
145
|
+
* the router carries repo-relative ones), so exact equality, path-suffix, and bare-name
|
|
146
|
+
* matches all count.
|
|
147
|
+
*/
|
|
148
|
+
function fileMatchesExpected(receiptFile, expectedFile) {
|
|
149
|
+
if (!receiptFile || !expectedFile) return false;
|
|
150
|
+
const receipt = String(receiptFile).replace(/\\/g, '/').replace(/^\.\//, '');
|
|
151
|
+
const expected = String(expectedFile).replace(/\\/g, '/').replace(/^\.\//, '');
|
|
152
|
+
if (receipt === expected) return true;
|
|
153
|
+
if (receipt.endsWith(`/${expected}`) || expected.endsWith(`/${receipt}`)) return true;
|
|
154
|
+
if (!expected.includes('/')) {
|
|
155
|
+
const base = receipt.split('/').pop();
|
|
156
|
+
return base === expected;
|
|
157
|
+
}
|
|
158
|
+
return false;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
function matchesRoutedCommand(command, routedCommand) {
|
|
162
|
+
const receipt = String(command || '').trim();
|
|
163
|
+
const routed = String(routedCommand || '').trim();
|
|
164
|
+
if (!receipt || !routed) return false;
|
|
165
|
+
return receipt === routed || receipt.startsWith(routed) || routed.startsWith(receipt);
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
function routedVerificationCommands(routeSummary = {}) {
|
|
169
|
+
const preferred = Array.isArray(routeSummary?.preferredOrder) ? routeSummary.preferredOrder : [];
|
|
170
|
+
const primary = Array.isArray(routeSummary?.primaryCommands) ? routeSummary.primaryCommands : [];
|
|
171
|
+
return preferred.length > 0 ? preferred : primary;
|
|
172
|
+
}
|
|
173
|
+
|
|
141
174
|
function appendReceipt(receipts, receipt) {
|
|
142
175
|
return [...(receipts || []), compactReceipt(receipt)].slice(-MAX_RECEIPTS);
|
|
143
176
|
}
|
|
@@ -152,6 +185,7 @@ function freshLedger(payload, routeState, harness) {
|
|
|
152
185
|
requestKey: routeState?.requestKey || null,
|
|
153
186
|
routeFingerprint: routeState?.fingerprint || null,
|
|
154
187
|
sourceSucceeded: false,
|
|
188
|
+
sourceFiles: [],
|
|
155
189
|
writeAttempted: false,
|
|
156
190
|
writeSucceeded: false,
|
|
157
191
|
verificationAttempted: false,
|
|
@@ -194,6 +228,9 @@ export async function recordExecutionReceipt({
|
|
|
194
228
|
receipt.kind = 'source';
|
|
195
229
|
receipt.file = toolInput.file_path || toolInput.path || null;
|
|
196
230
|
ledger.sourceSucceeded ||= success;
|
|
231
|
+
if (success && receipt.file) {
|
|
232
|
+
ledger.sourceFiles = [...new Set([...(ledger.sourceFiles || []), receipt.file])].slice(-MAX_SOURCE_FILES);
|
|
233
|
+
}
|
|
197
234
|
} else if (toolName === 'Edit' || toolName === 'Write') {
|
|
198
235
|
receipt.kind = 'write';
|
|
199
236
|
receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
|
|
@@ -202,6 +239,17 @@ export async function recordExecutionReceipt({
|
|
|
202
239
|
} else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
|
|
203
240
|
receipt.kind = 'verification';
|
|
204
241
|
receipt.command = String(toolInput.command || '').trim();
|
|
242
|
+
// WS-C routed-verification receipt: a command counts as "targeted" when it matches the
|
|
243
|
+
// routed plan (preferredOrder / primaryCommands). Off-plan verification still records
|
|
244
|
+
// as broad — counted only when the route carried no commands at all.
|
|
245
|
+
const routedCommands = routedVerificationCommands(routeState?.routeSummary);
|
|
246
|
+
if (routedCommands.length > 0) {
|
|
247
|
+
const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
|
|
248
|
+
receipt.scope = matched ? 'targeted' : 'broad';
|
|
249
|
+
if (receipt.scope === 'targeted') {
|
|
250
|
+
ledger.targetedVerificationSucceeded ||= success;
|
|
251
|
+
}
|
|
252
|
+
}
|
|
205
253
|
ledger.verificationAttempted = true;
|
|
206
254
|
ledger.verificationSucceeded ||= success;
|
|
207
255
|
ledger.verificationFailed ||= !success;
|
|
@@ -224,30 +272,66 @@ function requiredEvidence(state = {}) {
|
|
|
224
272
|
return [...new Set(routeSummary.completionState?.missingEvidence || [])];
|
|
225
273
|
}
|
|
226
274
|
|
|
227
|
-
function evidenceSatisfied(evidence, ledger = {}) {
|
|
275
|
+
function evidenceSatisfied(evidence, ledger = {}, routeSummary = {}) {
|
|
228
276
|
if (evidence === 'write-evidence') return ledger.writeSucceeded === true;
|
|
229
|
-
if (evidence === 'verification-evidence')
|
|
230
|
-
|
|
277
|
+
if (evidence === 'verification-evidence') {
|
|
278
|
+
// WS-C: when the route names concrete verification commands, only a receipt that ran
|
|
279
|
+
// one of them counts — an unrelated `yarn test` no longer satisfies the gate.
|
|
280
|
+
const routedCommands = routedVerificationCommands(routeSummary);
|
|
281
|
+
if (routedCommands.length > 0) {
|
|
282
|
+
return ledger.targetedVerificationSucceeded === true;
|
|
283
|
+
}
|
|
284
|
+
return ledger.verificationSucceeded === true;
|
|
285
|
+
}
|
|
286
|
+
if (evidence === 'impact-evidence') {
|
|
287
|
+
// WS-C: when the route names expected source files, a read must cover one of them —
|
|
288
|
+
// ANY read (e.g. docs) no longer satisfies the impact-evidence gate.
|
|
289
|
+
const expectedFiles = Array.isArray(routeSummary?.expectedSourceFiles)
|
|
290
|
+
? routeSummary.expectedSourceFiles
|
|
291
|
+
: [];
|
|
292
|
+
if (expectedFiles.length > 0) {
|
|
293
|
+
const readFiles = Array.isArray(ledger.sourceFiles) ? ledger.sourceFiles : [];
|
|
294
|
+
return readFiles.some((file) => expectedFiles.some((expected) => fileMatchesExpected(file, expected)));
|
|
295
|
+
}
|
|
296
|
+
return ledger.sourceSucceeded === true;
|
|
297
|
+
}
|
|
231
298
|
return false;
|
|
232
299
|
}
|
|
233
300
|
|
|
234
|
-
function recoveryInstruction(missingEvidence, ledger = {}) {
|
|
301
|
+
function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}) {
|
|
302
|
+
let instruction = null;
|
|
235
303
|
if (missingEvidence.includes('write-evidence')) {
|
|
236
304
|
if (!ledger.sourceSucceeded) {
|
|
237
|
-
|
|
305
|
+
instruction = 'Pull one bounded indexed source slice, then make the requested Edit/Write in this continuation.';
|
|
306
|
+
} else if (ledger.writeAttempted && !ledger.writeSucceeded) {
|
|
307
|
+
instruction = 'Recover from the failed mutation using the current error, then retry the smallest correct Edit/Write.';
|
|
308
|
+
} else {
|
|
309
|
+
instruction = 'Make the smallest correct Edit/Write now; do not end after more read-only analysis.';
|
|
238
310
|
}
|
|
239
|
-
|
|
240
|
-
|
|
311
|
+
} else if (missingEvidence.includes('verification-evidence')) {
|
|
312
|
+
if (ledger.verificationFailed) {
|
|
313
|
+
instruction = 'Use the latest failed verification output, fix the failure, and rerun targeted verification.';
|
|
314
|
+
} else {
|
|
315
|
+
const routedCommands = routedVerificationCommands(routeSummary);
|
|
316
|
+
if (routedCommands.length > 0 && ledger.verificationSucceeded && !ledger.targetedVerificationSucceeded) {
|
|
317
|
+
instruction = `Verification ran but did not match the routed plan (${routedCommands.slice(0, 3).join('; ')}). Run one of those and inspect its result before stopping.`;
|
|
318
|
+
} else {
|
|
319
|
+
instruction = 'Run the routed targeted verification now and inspect its result before stopping.';
|
|
320
|
+
}
|
|
241
321
|
}
|
|
242
|
-
return 'Make the smallest correct Edit/Write now; do not end after more read-only analysis.';
|
|
243
322
|
}
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
323
|
+
// WS-C: the target hint travels with the reason whenever impact evidence is missing and
|
|
324
|
+
// the route names expected files — it must survive branch precedence above.
|
|
325
|
+
if (missingEvidence.includes('impact-evidence')) {
|
|
326
|
+
const expectedFiles = Array.isArray(routeSummary?.expectedSourceFiles) ? routeSummary.expectedSourceFiles : [];
|
|
327
|
+
if (expectedFiles.length > 0 && !expectedFiles.some(
|
|
328
|
+
(expected) => (ledger.sourceFiles || []).some((file) => fileMatchesExpected(file, expected))
|
|
329
|
+
)) {
|
|
330
|
+
const hint = `Read the routed target files (${expectedFiles.slice(0, 3).join(', ')}) — reads of unrelated files do not count as impact evidence.`;
|
|
331
|
+
instruction = instruction ? `${instruction} ${hint}` : hint;
|
|
247
332
|
}
|
|
248
|
-
return 'Run the routed targeted verification now and inspect its result before stopping.';
|
|
249
333
|
}
|
|
250
|
-
return 'Complete the current routed milestone before stopping.';
|
|
334
|
+
return instruction ?? 'Complete the current routed milestone before stopping.';
|
|
251
335
|
}
|
|
252
336
|
|
|
253
337
|
export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
@@ -263,7 +347,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
263
347
|
|
|
264
348
|
const sameRequest = !ledger?.requestKey || !state?.requestKey || ledger.requestKey === state.requestKey;
|
|
265
349
|
const effectiveLedger = sameRequest ? ledger : {};
|
|
266
|
-
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger));
|
|
350
|
+
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, routeSummary));
|
|
267
351
|
if (missingEvidence.length === 0) {
|
|
268
352
|
return { continue: false, notify: false, missingEvidence: [] };
|
|
269
353
|
}
|
|
@@ -278,7 +362,15 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
278
362
|
};
|
|
279
363
|
}
|
|
280
364
|
|
|
281
|
-
|
|
365
|
+
// Continuation attempts only make sense within the request that minted them: a new
|
|
366
|
+
// routed request in the same session must start with a fresh budget, otherwise a cap
|
|
367
|
+
// exhausted on task A suppresses recovery for task B.
|
|
368
|
+
const staleContinuations = effectiveLedger?.continuationRequestKey
|
|
369
|
+
&& state?.requestKey
|
|
370
|
+
&& effectiveLedger.continuationRequestKey !== state.requestKey;
|
|
371
|
+
const continuationCount = staleContinuations
|
|
372
|
+
? 0
|
|
373
|
+
: Number(effectiveLedger?.continuationCount || 0);
|
|
282
374
|
if (continuationCount >= MAX_CONTINUATIONS) {
|
|
283
375
|
if (effectiveLedger?.notified === true) {
|
|
284
376
|
return {
|
|
@@ -300,7 +392,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
300
392
|
}
|
|
301
393
|
|
|
302
394
|
const finalAttempt = continuationCount === MAX_CONTINUATIONS - 1;
|
|
303
|
-
const instruction = recoveryInstruction(missingEvidence, effectiveLedger);
|
|
395
|
+
const instruction = recoveryInstruction(missingEvidence, effectiveLedger, routeSummary);
|
|
304
396
|
return {
|
|
305
397
|
continue: true,
|
|
306
398
|
missingEvidence,
|
|
@@ -312,11 +404,18 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
312
404
|
};
|
|
313
405
|
}
|
|
314
406
|
|
|
315
|
-
export async function incrementContinuation(projectRoot, payload = {}, ledger = null) {
|
|
407
|
+
export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null) {
|
|
316
408
|
const current = ledger || await readExecutionLedger(projectRoot, payload) || freshLedger(payload, null, 'unknown');
|
|
409
|
+
// Mirror evaluateCompletion's staleness rule: a count minted by an earlier request must
|
|
410
|
+
// not be carried into the new request's budget, or the cap fires early (evaluate says 0,
|
|
411
|
+
// persist says 7) and the next request inherits a nearly exhausted budget.
|
|
412
|
+
const stale = requestKey
|
|
413
|
+
&& current?.continuationRequestKey
|
|
414
|
+
&& current.continuationRequestKey !== requestKey;
|
|
317
415
|
const next = {
|
|
318
416
|
...current,
|
|
319
|
-
continuationCount: Number(current.continuationCount || 0) + 1,
|
|
417
|
+
continuationCount: (stale ? 0 : Number(current.continuationCount || 0)) + 1,
|
|
418
|
+
...(requestKey ? { continuationRequestKey: requestKey } : {}),
|
|
320
419
|
lastContinuationAt: Date.now(),
|
|
321
420
|
updatedAt: Date.now(),
|
|
322
421
|
};
|
|
@@ -356,9 +455,9 @@ async function main() {
|
|
|
356
455
|
const result = evaluateCompletion({ state, ledger });
|
|
357
456
|
if (result.continue) {
|
|
358
457
|
if (result.finalNotice) await markNotified(projectRoot, payload, ledger);
|
|
359
|
-
else await incrementContinuation(projectRoot, payload, ledger);
|
|
458
|
+
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
|
|
360
459
|
process.stdout.write(`${JSON.stringify({ decision: 'block', reason: result.reason })}\n`);
|
|
361
|
-
} else if (result.capped) {
|
|
460
|
+
} else if (result.capped || result.notify) {
|
|
362
461
|
process.stderr.write(`[ukit-completion] ${result.reason}\n`);
|
|
363
462
|
}
|
|
364
463
|
}
|
|
@@ -8,7 +8,6 @@ const FAIL_CLOSED_SCRIPTS = new Set([
|
|
|
8
8
|
'protect-files.sh',
|
|
9
9
|
'stale-spec-guard.sh',
|
|
10
10
|
'handoff-model-guard.sh',
|
|
11
|
-
'vision-gate.sh',
|
|
12
11
|
'context-hardcap-gate.sh',
|
|
13
12
|
'block-dangerous.sh',
|
|
14
13
|
]);
|
|
@@ -39,7 +38,11 @@ function run(payloadText, scriptPaths) {
|
|
|
39
38
|
? path.resolve(path.dirname(firstScript), '../..')
|
|
40
39
|
: (payload.cwd || process.cwd());
|
|
41
40
|
const startedAt = Date.now();
|
|
42
|
-
|
|
41
|
+
// The chain budget must always be able to run EVERY child at its full per-child
|
|
42
|
+
// budget — a fixed total silently starves later scripts once a chain grows
|
|
43
|
+
// (UserPromptSubmit now carries 4 hooks). The floor keeps short chains at 10s.
|
|
44
|
+
const totalBudgetMs = Math.max(TOTAL_BUDGET_MS, scriptPaths.length * CHILD_BUDGET_MS);
|
|
45
|
+
const deadline = startedAt + totalBudgetMs;
|
|
43
46
|
const results = [];
|
|
44
47
|
|
|
45
48
|
for (const scriptPath of scriptPaths) {
|
|
@@ -50,7 +53,7 @@ function run(payloadText, scriptPaths) {
|
|
|
50
53
|
scriptName,
|
|
51
54
|
code: 1,
|
|
52
55
|
stdout: '',
|
|
53
|
-
stderr: `hook chain exceeded its ${
|
|
56
|
+
stderr: `hook chain exceeded its ${totalBudgetMs}ms total budget`,
|
|
54
57
|
killed: true,
|
|
55
58
|
elapsedMs: 0,
|
|
56
59
|
});
|
|
@@ -90,7 +93,7 @@ function run(payloadText, scriptPaths) {
|
|
|
90
93
|
toolName: payload?.tool_name || null,
|
|
91
94
|
toolUseId: payload?.tool_use_id || null,
|
|
92
95
|
elapsedMs,
|
|
93
|
-
budgetMs:
|
|
96
|
+
budgetMs: totalBudgetMs,
|
|
94
97
|
scripts: results.map(({ scriptName, code, killed, elapsedMs: scriptElapsedMs }) => ({
|
|
95
98
|
scriptName,
|
|
96
99
|
code,
|
|
@@ -99,7 +102,7 @@ function run(payloadText, scriptPaths) {
|
|
|
99
102
|
})),
|
|
100
103
|
});
|
|
101
104
|
|
|
102
|
-
return { results, elapsedMs, budgetMs:
|
|
105
|
+
return { results, elapsedMs, budgetMs: totalBudgetMs };
|
|
103
106
|
}
|
|
104
107
|
|
|
105
108
|
try {
|
package/templates/.omp/README.md
CHANGED
|
@@ -62,14 +62,14 @@ One source of truth, two runtimes — fix a hook once and both runtimes get the
|
|
|
62
62
|
Two things worth knowing:
|
|
63
63
|
|
|
64
64
|
**Tool names are mapped explicitly, never guessed.** omp's write surface is `edit`, `write` *and*
|
|
65
|
-
`ast_edit`; all three map to the `Edit` group, or `ast_edit` would slip past `protect-files.sh
|
|
66
|
-
`
|
|
65
|
+
`ast_edit`; all three map to the `Edit` group, or `ast_edit` would slip past `protect-files.sh`.
|
|
66
|
+
`eval` maps to `Bash` so it still hits `block-dangerous.sh`. Anything not in the
|
|
67
67
|
table maps to nothing and runs zero scripts — it never falls back to `Bash` or `Edit`.
|
|
68
68
|
|
|
69
69
|
**Failure direction is per-script, transcribed from each script's own header — not a blanket rule.**
|
|
70
70
|
|
|
71
71
|
- *Fail closed* (a crash or non-zero exit blocks the action): `protect-files.sh`,
|
|
72
|
-
`stale-spec-guard.sh`, `handoff-model-guard.sh`, `
|
|
72
|
+
`stale-spec-guard.sh`, `handoff-model-guard.sh`, `context-hardcap-gate.sh`,
|
|
73
73
|
`block-dangerous.sh`, `verification-guard.sh`. These are gates; a broken gate must not open.
|
|
74
74
|
- *Fail open* (a crash logs a warning and the action proceeds): the advisory scripts —
|
|
75
75
|
routing, backups, output compression, context reinjection, pressure reset, handoff resume.
|
|
@@ -20,19 +20,19 @@ import {
|
|
|
20
20
|
|
|
21
21
|
export const HOOK_EVENT_MAP = {
|
|
22
22
|
tool_call: {
|
|
23
|
-
'Read|Grep|Glob': [],
|
|
23
|
+
'Read|Grep|Glob': ['sensitive-data-guard.sh'],
|
|
24
24
|
'Edit|Write': [
|
|
25
25
|
'protect-files.sh',
|
|
26
26
|
'stale-spec-guard.sh',
|
|
27
27
|
'pre-edit-backup.sh',
|
|
28
28
|
'skill-router.sh',
|
|
29
29
|
'handoff-model-guard.sh',
|
|
30
|
-
'vision-gate.sh',
|
|
31
30
|
'context-hardcap-gate.sh',
|
|
32
31
|
],
|
|
33
32
|
Bash: [
|
|
34
33
|
'auto-allow-bash.sh',
|
|
35
34
|
'block-dangerous.sh',
|
|
35
|
+
'sensitive-data-guard.sh',
|
|
36
36
|
'handoff-model-guard.sh',
|
|
37
37
|
'context-hardcap-gate.sh',
|
|
38
38
|
],
|
|
@@ -42,7 +42,7 @@ export const HOOK_EVENT_MAP = {
|
|
|
42
42
|
'Edit|Write': ['post-edit-verify.sh', 'record-execution.sh'],
|
|
43
43
|
Bash: ['compress-output.sh', 'record-execution.sh'],
|
|
44
44
|
},
|
|
45
|
-
before_agent_start: ['skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
|
|
45
|
+
before_agent_start: ['sensitive-data-guard.sh', 'skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
|
|
46
46
|
'session.compacting': ['reinject-context.sh'],
|
|
47
47
|
session_start: ['auto-prune-bash.sh', 'reset-compact-pressure.sh', 'handoff-resume.sh'],
|
|
48
48
|
};
|
|
@@ -83,9 +83,9 @@ export const FAIL_CLOSED_SCRIPTS = new Set([
|
|
|
83
83
|
'protect-files.sh',
|
|
84
84
|
'stale-spec-guard.sh',
|
|
85
85
|
'handoff-model-guard.sh',
|
|
86
|
-
'vision-gate.sh',
|
|
87
86
|
'context-hardcap-gate.sh',
|
|
88
87
|
'block-dangerous.sh',
|
|
88
|
+
'sensitive-data-guard.sh',
|
|
89
89
|
]);
|
|
90
90
|
|
|
91
91
|
export const ADVISORY_SCRIPTS = new Set([
|
|
@@ -170,7 +170,15 @@ function translateExecResult(scriptName, execResult) {
|
|
|
170
170
|
|
|
171
171
|
export { translateExecResult };
|
|
172
172
|
|
|
173
|
-
|
|
173
|
+
// Mirrors hook-chain-runner.mjs's TOTAL/CHILD budget constants: the exec timeout must
|
|
174
|
+
// exceed the runner's own chain budget (max(10s, scripts×4s)) plus parse margin, or the
|
|
175
|
+
// bridge orphans the runner mid-chain once a chain grows past 2 scripts.
|
|
176
|
+
const HOOK_CHAIN_BASE_BUDGET_MS = 10000;
|
|
177
|
+
const HOOK_CHAIN_CHILD_BUDGET_MS = 4000;
|
|
178
|
+
function chainExecTimeoutMs(scriptCount) {
|
|
179
|
+
const budget = Math.max(HOOK_CHAIN_BASE_BUDGET_MS, scriptCount * HOOK_CHAIN_CHILD_BUDGET_MS);
|
|
180
|
+
return Math.min(30000, budget + 2000);
|
|
181
|
+
}
|
|
174
182
|
|
|
175
183
|
function recordHookErrorDiagnostic(projectRoot, sessionId, diagnostic) {
|
|
176
184
|
try {
|
|
@@ -228,7 +236,7 @@ export async function runScriptChain(
|
|
|
228
236
|
execResult = await pi.exec(
|
|
229
237
|
nodeExecutable,
|
|
230
238
|
[runnerPath, JSON.stringify(payload), ...scriptPaths],
|
|
231
|
-
{ cwd: projectRoot, timeout:
|
|
239
|
+
{ cwd: projectRoot, timeout: chainExecTimeoutMs(scripts.length) },
|
|
232
240
|
);
|
|
233
241
|
} catch (error) {
|
|
234
242
|
execResult = { code: 1, stdout: '', stderr: error?.message ?? String(error), killed: false };
|
|
@@ -517,7 +525,7 @@ export async function runSessionStop(
|
|
|
517
525
|
if (suppliedLedger === undefined) {
|
|
518
526
|
try {
|
|
519
527
|
if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger);
|
|
520
|
-
else await incrementContinuation(projectRoot, payload, ledger);
|
|
528
|
+
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
|
|
521
529
|
} catch (error) {
|
|
522
530
|
pi.logger?.warn?.(`[UKit] continuation bookkeeping failed open: ${error?.message || error}`);
|
|
523
531
|
}
|
package/templates/AGENTS.md
CHANGED
|
@@ -206,10 +206,10 @@ This is internal orchestration — end users do not need to know about tiers, th
|
|
|
206
206
|
|
|
207
207
|
### Vision lane (capability, not a cost tier)
|
|
208
208
|
|
|
209
|
-
`unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table.
|
|
209
|
+
`unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table. Whether a mapping can read images is a capability fact, not a provider fact: a mapping that has not **verified** native vision must never guess at image contents — choose native-first when verified, otherwise route to the specialist.
|
|
210
210
|
|
|
211
211
|
- **Gateway detection**: UNIC routing for a Claude Code session is decided ONLY by what changes Claude Code's own outbound endpoint — `ANTHROPIC_BASE_URL` (env var, or the `env.ANTHROPIC_BASE_URL` key in project/home `.claude/settings.json`) containing `unicjsc.com`. Other tools' configs — Codex `config.toml`, Kilo `secrets.json`, or an `OPENAI_BASE_URL` env var — describe a different tool's endpoint entirely and never decide this session's routing.
|
|
212
|
-
- **
|
|
212
|
+
- **Advisory routing (no hard block)**: when an image reaches the prompt, the vision router reminds the session to have `ukit-vision-analyst` analyse it before relying on its contents. Edits are **never blocked** — correctness relies on the model routing images to the analyst instead of guessing.
|
|
213
213
|
- This is internal orchestration — end users never invoke a vision command directly; `ukit install` plus natural language remains the whole surface. No new commands.
|
|
214
214
|
|
|
215
215
|
## Session Start — OpenCode
|
package/templates/CLAUDE.md
CHANGED
|
@@ -206,10 +206,10 @@ This is internal orchestration — end users do not need to know about tiers, th
|
|
|
206
206
|
|
|
207
207
|
### Vision lane (capability, not a cost tier)
|
|
208
208
|
|
|
209
|
-
`unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table.
|
|
209
|
+
`unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table. Whether a mapping can read images is a capability fact, not a provider fact: a mapping that has not **verified** native vision must never guess at image contents — choose native-first when verified, otherwise route to the specialist.
|
|
210
210
|
|
|
211
211
|
- **Gateway detection**: UNIC routing for a Claude Code session is decided ONLY by what changes Claude Code's own outbound endpoint — `ANTHROPIC_BASE_URL` (env var, or the `env.ANTHROPIC_BASE_URL` key in project/home `.claude/settings.json`) containing `unicjsc.com`. Other tools' configs — Codex `config.toml`, Kilo `secrets.json`, or an `OPENAI_BASE_URL` env var — describe a different tool's endpoint entirely and never decide this session's routing.
|
|
212
|
-
- **
|
|
212
|
+
- **Advisory routing (no hard block)**: when an image reaches the prompt, the vision router reminds the session to have `ukit-vision-analyst` analyse it before relying on its contents. Edits are **never blocked** — correctness relies on the model routing images to the analyst instead of guessing.
|
|
213
213
|
- This is internal orchestration — end users never invoke a vision command directly; `ukit install` plus natural language remains the whole surface. No new commands.
|
|
214
214
|
|
|
215
215
|
## Skills
|
|
@@ -6,6 +6,10 @@
|
|
|
6
6
|
"affectVerification": true,
|
|
7
7
|
"affectDelegation": true
|
|
8
8
|
},
|
|
9
|
+
"security": {
|
|
10
|
+
"sensitiveDataGate": true,
|
|
11
|
+
"allowlistPath": ".ukit/storage/security/allowlist.json"
|
|
12
|
+
},
|
|
9
13
|
"compact": {
|
|
10
14
|
"enabled": true,
|
|
11
15
|
"tokenThreshold": 150000,
|
|
@@ -91,7 +95,6 @@
|
|
|
91
95
|
"advisorEnabled": true,
|
|
92
96
|
"contracts": {
|
|
93
97
|
"tiny-fix": {
|
|
94
|
-
"modelTier": "lite",
|
|
95
98
|
"maxReadPasses": 0,
|
|
96
99
|
"maxContextPulls": 0,
|
|
97
100
|
"verificationPolicy": "minimal-or-targeted",
|
|
@@ -99,7 +102,6 @@
|
|
|
99
102
|
"delegationPolicy": "disallow"
|
|
100
103
|
},
|
|
101
104
|
"local-fix": {
|
|
102
|
-
"modelTier": "code",
|
|
103
105
|
"maxReadPasses": 1,
|
|
104
106
|
"maxContextPulls": 1,
|
|
105
107
|
"verificationPolicy": "targeted-if-covered",
|
|
@@ -107,7 +109,6 @@
|
|
|
107
109
|
"delegationPolicy": "disallow"
|
|
108
110
|
},
|
|
109
111
|
"local-build": {
|
|
110
|
-
"modelTier": "code",
|
|
111
112
|
"maxReadPasses": 2,
|
|
112
113
|
"maxContextPulls": 1,
|
|
113
114
|
"verificationPolicy": "targeted-if-covered",
|
|
@@ -116,14 +117,12 @@
|
|
|
116
117
|
"postEditReviewPolicy": "sidecar-non-blocking"
|
|
117
118
|
},
|
|
118
119
|
"find-cause": {
|
|
119
|
-
"modelTier": "code",
|
|
120
120
|
"maxReadPassesBeforeReassess": 3,
|
|
121
121
|
"verificationPolicy": "root-cause-then-targeted",
|
|
122
122
|
"completionRule": "never-claim-fixed-without-write-and-verification",
|
|
123
123
|
"delegationPolicy": "allow-specialized-debug-lane"
|
|
124
124
|
},
|
|
125
125
|
"shared-edit": {
|
|
126
|
-
"modelTier": "code",
|
|
127
126
|
"maxReadPasses": 2,
|
|
128
127
|
"maxContextPulls": 2,
|
|
129
128
|
"verificationPolicy": "targeted-then-widen-on-risk",
|
|
@@ -132,7 +131,6 @@
|
|
|
132
131
|
"postEditReviewPolicy": "sidecar-non-blocking"
|
|
133
132
|
},
|
|
134
133
|
"map-impact": {
|
|
135
|
-
"modelTier": "code",
|
|
136
134
|
"maxReadPasses": 3,
|
|
137
135
|
"maxContextPulls": 3,
|
|
138
136
|
"verificationPolicy": "impact-first-then-targeted-then-widen-on-risk",
|
|
@@ -140,7 +138,6 @@
|
|
|
140
138
|
"delegationPolicy": "allow-impact-sidecar"
|
|
141
139
|
},
|
|
142
140
|
"review-release": {
|
|
143
|
-
"modelTier": "smart",
|
|
144
141
|
"verificationPolicy": "evidence-first",
|
|
145
142
|
"completionRule": "report-findings-not-implementation",
|
|
146
143
|
"delegationPolicy": "allow-review-sidecar"
|
|
@@ -392,6 +389,10 @@
|
|
|
392
389
|
"affectVerification": "Nếu true, autonomy.level ảnh hưởng hành vi verification plan.",
|
|
393
390
|
"affectDelegation": "Nếu true, autonomy.level ảnh hưởng ngưỡng delegation."
|
|
394
391
|
},
|
|
392
|
+
"security": {
|
|
393
|
+
"sensitiveDataGate": "Bật sensitive-data gate: chặn key/private data/secret (đọc file .env, *.pem, id_rsa; lệnh bash dump secret; prompt chứa token) trước khi chúng tới AI. Chặn cực gắt theo yêu cầu: chỉ cần nghi ngờ là chặn và hỏi user. Tắt chỉ khi debug gate này.",
|
|
394
|
+
"allowlistPath": "File allowlist JSON do USER tự tạo để phê duyệt tường minh giá trị secret (sha256) hoặc đường dẫn file được phép gửi. File này được protect-files.sh chặn AI tự sửa."
|
|
395
|
+
},
|
|
395
396
|
"compact": {
|
|
396
397
|
"enabled": "Bật/tắt toàn bộ helper compact của UKit.",
|
|
397
398
|
"tokenThreshold": "Ngưỡng token chung cho runtime compact dùng chung.",
|
|
@@ -1,42 +0,0 @@
|
|
|
1
|
-
export function shouldEscalate(trigger) {
|
|
2
|
-
if (!trigger || typeof trigger !== 'object') {
|
|
3
|
-
return false;
|
|
4
|
-
}
|
|
5
|
-
|
|
6
|
-
if (trigger.type === 'user_requested') {
|
|
7
|
-
return true;
|
|
8
|
-
}
|
|
9
|
-
|
|
10
|
-
if (trigger.type === 'complexity_detected') {
|
|
11
|
-
return true;
|
|
12
|
-
}
|
|
13
|
-
|
|
14
|
-
if (trigger.type === 'retry_exceeded') {
|
|
15
|
-
return (trigger.attempts ?? 0) >= 2;
|
|
16
|
-
}
|
|
17
|
-
|
|
18
|
-
if (trigger.type === 'validation_failed') {
|
|
19
|
-
return (trigger.attempts ?? 0) >= 1;
|
|
20
|
-
}
|
|
21
|
-
|
|
22
|
-
return false;
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
export async function askAdvisor(request) {
|
|
26
|
-
const strategy = request?.question
|
|
27
|
-
? `Break the task into verifiable steps, focusing on: ${request.question}`
|
|
28
|
-
: 'Break the task into smaller verifiable steps.';
|
|
29
|
-
|
|
30
|
-
return {
|
|
31
|
-
strategy,
|
|
32
|
-
steps: [
|
|
33
|
-
'Restate the goal in one short sentence.',
|
|
34
|
-
'List the smallest changes that can be verified locally.',
|
|
35
|
-
'Run focused checks before expanding the scope.',
|
|
36
|
-
],
|
|
37
|
-
warnings: [
|
|
38
|
-
'Fallback advisor response used because no external advisor integration is wired in this local runtime.',
|
|
39
|
-
],
|
|
40
|
-
confidence: 55,
|
|
41
|
-
};
|
|
42
|
-
}
|