@ngockhoale/ukit 2.2.16 → 2.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,9 +6,10 @@
6
6
  * blocks, so pasted images are invisible unless this script first decodes them
7
7
  * to disk. This is the single home of the sha256 key: `pending-<sha>` markers
8
8
  * (written by --mark-pending) and `<sha>.<ext>` image files (written by a normal
9
- * extract) MUST agree exactly, or the TASK-008 gate deadlocks forever. Both
10
- * paths call the same hashPayload() function on the exact base64 string as it
11
- * appears in the transcript — no trimming, no normalising, no decoding first.
9
+ * extract) MUST agree exactly, or analysis receipts can never be matched back to
10
+ * their markers. Both paths call the same hashPayload() function on the exact
11
+ * base64 string as it appears in the transcript — no trimming, no normalising,
12
+ * no decoding first.
12
13
  *
13
14
  * Usage:
14
15
  * node extract-image.mjs [options]
@@ -323,14 +324,23 @@ function main() {
323
324
  // A sha with an existing analyzed-<sha>.json receipt is a closed, content-addressed case:
324
325
  // identical bytes already produced that analysis, so skipping it loses no information.
325
326
  const isAlreadyAnalyzed = (sha) => fs.existsSync(path.join(markerDir, `analyzed-${sha}.json`));
326
- const selected = selectImages(blocks, opts.limit).filter((img) => !isAlreadyAnalyzed(img.sha));
327
- const refs = resolveRefs(opts.refs).filter((ref) => !isAlreadyAnalyzed(ref.sha));
327
+ const allSelected = selectImages(blocks, opts.limit);
328
+ const allRefs = resolveRefs(opts.refs);
329
+ const selected = allSelected.filter((img) => !isAlreadyAnalyzed(img.sha));
330
+ const refs = allRefs.filter((ref) => !isAlreadyAnalyzed(ref.sha));
331
+ // Closed cases reported back so callers (vision-router.sh) can stay silent on them:
332
+ // a sha with an analyzed-<sha>.json receipt needs no new dispatch, and without this
333
+ // list the caller cannot tell "already analyzed" from "never resolved".
334
+ const alreadyAnalyzed = [
335
+ ...allSelected.filter((img) => isAlreadyAnalyzed(img.sha)).map((img) => ({ sha: img.sha, source: 'pasted' })),
336
+ ...allRefs.filter((ref) => isAlreadyAnalyzed(ref.sha)).map((ref) => ({ sha: ref.sha, source: ref.source, ref: ref.value })),
337
+ ];
328
338
 
329
339
  // --detect is report-only, so it must not short-circuit --mark-pending when both are
330
340
  // passed together. vision-router.sh combines them; short-circuiting here would arm
331
- // nothing and leave the TASK-008 write gate permanently open.
341
+ // nothing and leave every image unmarked.
332
342
  if (opts.detect && !opts.markPending) {
333
- printResult(opts, { imageCount: selected.length + refs.length, sessionId, images: [] });
343
+ printResult(opts, { imageCount: selected.length + refs.length, sessionId, images: [], alreadyAnalyzed });
334
344
  return;
335
345
  }
336
346
 
@@ -374,7 +384,7 @@ function main() {
374
384
  }
375
385
  }
376
386
  pruneMarkerDirs(outDir);
377
- printResult(opts, { imageCount: written.length, sessionId, images: written });
387
+ printResult(opts, { imageCount: written.length, sessionId, images: written, alreadyAnalyzed });
378
388
  return;
379
389
  }
380
390
 
@@ -579,6 +579,9 @@ function printRouteState(state) {
579
579
  if (state.routeSummary?.executionContract?.modelTier) {
580
580
  console.log(`modelTier: ${state.routeSummary.executionContract.modelTier}`);
581
581
  }
582
+ if (state.routeSummary?.tierLane?.instruction) {
583
+ console.log(`tierLane: ${state.routeSummary.tierLane.instruction}`);
584
+ }
582
585
  if (state.routeSummary?.escalatedTier) {
583
586
  console.log(`escalatedTier: ${state.routeSummary.escalatedTier}`);
584
587
  if (state.routeSummary.escalationReason) {
@@ -1475,6 +1478,15 @@ function buildRouteSummary({
1475
1478
  const primaryTargets = summarizeCompactList(preview.primaryTargets ?? [], 2);
1476
1479
  const relatedTests = summarizeCompactList(preview.relatedTests ?? [], 2);
1477
1480
  const styleFiles = summarizeCompactList(preview.styleFiles ?? [], 1);
1481
+ // WS-C target-aware evidence: the completion ledger checks impact-evidence against these
1482
+ // files, so carry the full resolver list (target + primaryTargets + relatedTests), bounded
1483
+ // for compact route state. Absent when the route has no indexed context — the ledger then
1484
+ // falls back to its legacy any-read behaviour.
1485
+ const expectedSourceFiles = unique([
1486
+ ...(routingContext.targetFile ? [routingContext.targetFile] : []),
1487
+ ...(preview.primaryTargets ?? []),
1488
+ ...(preview.relatedTests ?? []),
1489
+ ].filter((filePath) => typeof filePath === 'string' && filePath.trim())).slice(0, 8);
1478
1490
  const primaryCommands = unique(verificationRecommendation?.commands ?? []);
1479
1491
  const fallbackCommands = unique(verificationRecommendation?.fallbackCommands ?? []);
1480
1492
  const preferredOrder = unique(
@@ -1497,6 +1509,7 @@ function buildRouteSummary({
1497
1509
  executionCandidates,
1498
1510
  });
1499
1511
  const executionContract = buildExecutionContract(executionMode);
1512
+ const tierLane = buildTierLane({ modelTier: executionContract?.modelTier ?? null });
1500
1513
  const completionState = buildCompletionState({
1501
1514
  executionMode,
1502
1515
  verificationRecommendation,
@@ -1539,8 +1552,10 @@ function buildRouteSummary({
1539
1552
  executionCandidates,
1540
1553
  approachSelector,
1541
1554
  executionContract,
1555
+ tierLane,
1542
1556
  completionState,
1543
1557
  continuationState,
1558
+ ...(expectedSourceFiles.length > 0 ? { expectedSourceFiles } : {}),
1544
1559
  intentMode: routingContext.intentMode ?? null,
1545
1560
  handoffFile,
1546
1561
  delegateHint: delegationRecommendation?.hint ?? null,
@@ -1894,9 +1909,10 @@ function buildExecutionContract(executionMode = null) {
1894
1909
  maxReadPasses: 2,
1895
1910
  maxContextPulls: 1,
1896
1911
  verificationPolicy: 'targeted-if-covered',
1897
- completionRule: 'require-write',
1912
+ completionRule: 'require-write-and-verification',
1898
1913
  delegationPolicy: 'disallow-by-default',
1899
- completionEvidence: ['write-evidence'],
1914
+ completionEvidence: ['write-evidence', 'verification-evidence'],
1915
+ postEditReviewPolicy: 'sidecar-non-blocking',
1900
1916
  },
1901
1917
  'find-cause': {
1902
1918
  modelTier: 'code',
@@ -1915,6 +1931,7 @@ function buildExecutionContract(executionMode = null) {
1915
1931
  delegationPolicy: 'allow-qualified-sidecar',
1916
1932
  completionEvidence: ['write-evidence', 'verification-evidence'],
1917
1933
  mirrorConsistencyRequired: true,
1934
+ postEditReviewPolicy: 'sidecar-non-blocking',
1918
1935
  },
1919
1936
  'map-impact': {
1920
1937
  modelTier: 'code',
@@ -2182,10 +2199,36 @@ function applyEscalationToRouteSummary(routeSummary = null, { config = null, tar
2182
2199
  delete routeSummary.escalatedTier;
2183
2200
  delete routeSummary.escalationReason;
2184
2201
  }
2202
+ // Recomputed here (not only in buildRouteSummary) so the lane always reflects the final
2203
+ // tier after escalation, on every path that emits a route summary.
2204
+ routeSummary.tierLane = buildTierLane({
2205
+ modelTier: routeSummary.executionContract?.modelTier ?? null,
2206
+ escalatedTier: escalation?.escalatedTier ?? null,
2207
+ });
2185
2208
 
2186
2209
  return routeSummary;
2187
2210
  }
2188
2211
 
2212
+ // WS-E tier-role wiring: a contract tier only takes effect when the executing session hands
2213
+ // the work to an agent whose definition binds that tier — otherwise every session behaves as
2214
+ // if it only had one model. `code` is the DEFAULT implementation lane, so a code-tier contract
2215
+ // produces no instruction (inline execution on a code session is already correct, and silence
2216
+ // keeps the hook flow smooth). Labels are stable tier roles (lite/code/smart), never provider
2217
+ // or model identity. Vision never appears here — it is a capability lane, not a cost tier.
2218
+ function buildTierLane({ modelTier = null, escalatedTier = null } = {}) {
2219
+ const tier = escalatedTier ?? modelTier;
2220
+ if (!tier || tier === 'vision') {
2221
+ return null;
2222
+ }
2223
+ return {
2224
+ tier,
2225
+ defaultLane: tier === 'code',
2226
+ instruction: tier === 'code'
2227
+ ? null
2228
+ : `Hand this task to an agent bound to the ${tier} tier instead of doing it inline, unless this session already runs on the ${tier} tier.`,
2229
+ };
2230
+ }
2231
+
2189
2232
  function advanceContinuationState(continuationState = null, previousContinuationState = null, thresholds = null) {
2190
2233
  if (!continuationState || typeof continuationState !== 'object') {
2191
2234
  return continuationState;
@@ -2875,6 +2918,14 @@ function compactRouteSummary(routeSummary = null) {
2875
2918
  completionState: routeSummary.completionState ?? null,
2876
2919
  continuationState: routeSummary.continuationState ?? null,
2877
2920
  delegateHint: routeSummary.delegateHint ?? null,
2921
+ // WS-E tier-role binding: consumers need the hand-off instruction, not just the raw tier.
2922
+ ...(routeSummary.tierLane ? { tierLane: routeSummary.tierLane } : {}),
2923
+ // WS-C target-aware evidence: the completion ledger reads these from the PERSISTED route
2924
+ // state — without this spread the CLI path never delivers them and the gate falls back
2925
+ // to legacy any-read behaviour.
2926
+ ...(Array.isArray(routeSummary.expectedSourceFiles) && routeSummary.expectedSourceFiles.length > 0 ? {
2927
+ expectedSourceFiles: routeSummary.expectedSourceFiles,
2928
+ } : {}),
2878
2929
  nextActionType: routeSummary.nextActionType ?? null,
2879
2930
  nextActionCommand: routeSummary.nextActionCommand ?? null,
2880
2931
  helperHint: routeSummary.helperHint ?? null,
@@ -8,6 +8,7 @@ import { fileURLToPath } from 'node:url';
8
8
 
9
9
  const LEDGER_VERSION = 1;
10
10
  const MAX_RECEIPTS = 24;
11
+ const MAX_SOURCE_FILES = 16;
11
12
  const MAX_CONTINUATIONS = 6;
12
13
  const IMPLEMENT_MODES = new Set([
13
14
  'tiny-fix',
@@ -130,7 +131,7 @@ function compactReceipt(receipt) {
130
131
  kind: receipt.kind,
131
132
  success: receipt.success,
132
133
  };
133
- for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'error']) {
134
+ for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error']) {
134
135
  if (receipt[key] !== undefined && receipt[key] !== null && receipt[key] !== '') {
135
136
  compact[key] = receipt[key];
136
137
  }
@@ -138,10 +139,81 @@ function compactReceipt(receipt) {
138
139
  return compact;
139
140
  }
140
141
 
142
+ /**
143
+ * Target-aware matching between a receipt's file and a routed expected file. Both sides
144
+ * may disagree about absolute vs relative spelling (hook payloads carry absolute paths,
145
+ * the router carries repo-relative ones), so exact equality, path-suffix, and bare-name
146
+ * matches all count.
147
+ */
148
+ function fileMatchesExpected(receiptFile, expectedFile) {
149
+ if (!receiptFile || !expectedFile) return false;
150
+ const receipt = String(receiptFile).replace(/\\/g, '/').replace(/^\.\//, '');
151
+ const expected = String(expectedFile).replace(/\\/g, '/').replace(/^\.\//, '');
152
+ if (receipt === expected) return true;
153
+ if (receipt.endsWith(`/${expected}`) || expected.endsWith(`/${receipt}`)) return true;
154
+ if (!expected.includes('/')) {
155
+ const base = receipt.split('/').pop();
156
+ return base === expected;
157
+ }
158
+ return false;
159
+ }
160
+
161
+ function matchesRoutedCommand(command, routedCommand) {
162
+ const receipt = String(command || '').trim();
163
+ const routed = String(routedCommand || '').trim();
164
+ if (!receipt || !routed) return false;
165
+ return receipt === routed || receipt.startsWith(routed) || routed.startsWith(receipt);
166
+ }
167
+
168
+ function routedVerificationCommands(routeSummary = {}) {
169
+ const preferred = Array.isArray(routeSummary?.preferredOrder) ? routeSummary.preferredOrder : [];
170
+ const primary = Array.isArray(routeSummary?.primaryCommands) ? routeSummary.primaryCommands : [];
171
+ return preferred.length > 0 ? preferred : primary;
172
+ }
173
+
141
174
  function appendReceipt(receipts, receipt) {
142
175
  return [...(receipts || []), compactReceipt(receipt)].slice(-MAX_RECEIPTS);
143
176
  }
144
177
 
178
+ // A re-key within the same logical request must not drop the evidence the request
179
+ // already produced — that turned one finished task into a cap-exhausted forced stop.
180
+ // A genuinely different prompt keeps a clean slate so old work never satisfies a new
181
+ // request's completion gate.
182
+ function carriedEvidenceLedger(fresh, current) {
183
+ if (!current || !fresh.promptKey || !current.promptKey || current.promptKey !== fresh.promptKey) {
184
+ return fresh;
185
+ }
186
+ return {
187
+ ...fresh,
188
+ sourceSucceeded: fresh.sourceSucceeded || current.sourceSucceeded === true,
189
+ // targeted-verification evidence (2.3.0) must survive the same re-key carry as the
190
+ // coarse flags, or the stricter gates re-demand evidence mid-request — the exact
191
+ // stall this carry exists to prevent.
192
+ targetedVerificationSucceeded: fresh.targetedVerificationSucceeded
193
+ || current.targetedVerificationSucceeded === true,
194
+ sourceFiles: [...new Set([...(current.sourceFiles || []), ...(fresh.sourceFiles || [])])]
195
+ .slice(-MAX_SOURCE_FILES),
196
+ writeAttempted: fresh.writeAttempted || current.writeAttempted === true,
197
+ writeSucceeded: fresh.writeSucceeded || current.writeSucceeded === true,
198
+ verificationAttempted: fresh.verificationAttempted || current.verificationAttempted === true,
199
+ verificationSucceeded: fresh.verificationSucceeded || current.verificationSucceeded === true,
200
+ verificationFailed: fresh.verificationFailed || current.verificationFailed === true,
201
+ receipts: [...(current.receipts || [])].slice(-MAX_RECEIPTS),
202
+ };
203
+ }
204
+
205
+ // The router rebuilds requestKey per tool call (commandText/targetFile are part of the
206
+ // key), so requestKey changes mid-request on every verification command — and a subagent's
207
+ // Edit can additionally re-classify executionMode/taskType for the same user prompt. The
208
+ // logical request identity is therefore the prompt text alone: hash it so completion
209
+ // evidence survives any re-key/re-route within one request without leaking across requests.
210
+ // States with no recorded prompt text get no identity and never carry evidence.
211
+ function evidencePromptKey(routeState) {
212
+ const promptText = String(routeState?.routingContext?.lastExplicitUserPromptText || '').trim();
213
+ if (!promptText) return null;
214
+ return `prompt-${crypto.createHash('sha256').update(promptText).digest('hex').slice(0, 20)}`;
215
+ }
216
+
145
217
  function freshLedger(payload, routeState, harness) {
146
218
  return {
147
219
  version: LEDGER_VERSION,
@@ -150,8 +222,10 @@ function freshLedger(payload, routeState, harness) {
150
222
  transcriptPath: payload.transcript_path || payload.transcriptPath || null,
151
223
  harness,
152
224
  requestKey: routeState?.requestKey || null,
225
+ promptKey: evidencePromptKey(routeState),
153
226
  routeFingerprint: routeState?.fingerprint || null,
154
227
  sourceSucceeded: false,
228
+ sourceFiles: [],
155
229
  writeAttempted: false,
156
230
  writeSucceeded: false,
157
231
  verificationAttempted: false,
@@ -175,7 +249,7 @@ export async function recordExecutionReceipt({
175
249
  const routeState = await readRouteState(projectRoot);
176
250
  const current = await readExecutionLedger(projectRoot, payload);
177
251
  const ledger = !current || current.requestKey !== (routeState?.requestKey || null)
178
- ? freshLedger(payload, routeState, harness)
252
+ ? carriedEvidenceLedger(freshLedger(payload, routeState, harness), current)
179
253
  : { ...current, harness: current.harness || harness };
180
254
 
181
255
  const failed = explicitError(payload);
@@ -194,6 +268,9 @@ export async function recordExecutionReceipt({
194
268
  receipt.kind = 'source';
195
269
  receipt.file = toolInput.file_path || toolInput.path || null;
196
270
  ledger.sourceSucceeded ||= success;
271
+ if (success && receipt.file) {
272
+ ledger.sourceFiles = [...new Set([...(ledger.sourceFiles || []), receipt.file])].slice(-MAX_SOURCE_FILES);
273
+ }
197
274
  } else if (toolName === 'Edit' || toolName === 'Write') {
198
275
  receipt.kind = 'write';
199
276
  receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
@@ -202,6 +279,17 @@ export async function recordExecutionReceipt({
202
279
  } else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
203
280
  receipt.kind = 'verification';
204
281
  receipt.command = String(toolInput.command || '').trim();
282
+ // WS-C routed-verification receipt: a command counts as "targeted" when it matches the
283
+ // routed plan (preferredOrder / primaryCommands). Off-plan verification still records
284
+ // as broad — counted only when the route carried no commands at all.
285
+ const routedCommands = routedVerificationCommands(routeState?.routeSummary);
286
+ if (routedCommands.length > 0) {
287
+ const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
288
+ receipt.scope = matched ? 'targeted' : 'broad';
289
+ if (receipt.scope === 'targeted') {
290
+ ledger.targetedVerificationSucceeded ||= success;
291
+ }
292
+ }
205
293
  ledger.verificationAttempted = true;
206
294
  ledger.verificationSucceeded ||= success;
207
295
  ledger.verificationFailed ||= !success;
@@ -224,30 +312,66 @@ function requiredEvidence(state = {}) {
224
312
  return [...new Set(routeSummary.completionState?.missingEvidence || [])];
225
313
  }
226
314
 
227
- function evidenceSatisfied(evidence, ledger = {}) {
315
+ function evidenceSatisfied(evidence, ledger = {}, routeSummary = {}) {
228
316
  if (evidence === 'write-evidence') return ledger.writeSucceeded === true;
229
- if (evidence === 'verification-evidence') return ledger.verificationSucceeded === true;
230
- if (evidence === 'impact-evidence') return ledger.sourceSucceeded === true;
317
+ if (evidence === 'verification-evidence') {
318
+ // WS-C: when the route names concrete verification commands, only a receipt that ran
319
+ // one of them counts — an unrelated `yarn test` no longer satisfies the gate.
320
+ const routedCommands = routedVerificationCommands(routeSummary);
321
+ if (routedCommands.length > 0) {
322
+ return ledger.targetedVerificationSucceeded === true;
323
+ }
324
+ return ledger.verificationSucceeded === true;
325
+ }
326
+ if (evidence === 'impact-evidence') {
327
+ // WS-C: when the route names expected source files, a read must cover one of them —
328
+ // ANY read (e.g. docs) no longer satisfies the impact-evidence gate.
329
+ const expectedFiles = Array.isArray(routeSummary?.expectedSourceFiles)
330
+ ? routeSummary.expectedSourceFiles
331
+ : [];
332
+ if (expectedFiles.length > 0) {
333
+ const readFiles = Array.isArray(ledger.sourceFiles) ? ledger.sourceFiles : [];
334
+ return readFiles.some((file) => expectedFiles.some((expected) => fileMatchesExpected(file, expected)));
335
+ }
336
+ return ledger.sourceSucceeded === true;
337
+ }
231
338
  return false;
232
339
  }
233
340
 
234
- function recoveryInstruction(missingEvidence, ledger = {}) {
341
+ function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}) {
342
+ let instruction = null;
235
343
  if (missingEvidence.includes('write-evidence')) {
236
344
  if (!ledger.sourceSucceeded) {
237
- return 'Pull one bounded indexed source slice, then make the requested Edit/Write in this continuation.';
345
+ instruction = 'Pull one bounded indexed source slice, then make the requested Edit/Write in this continuation.';
346
+ } else if (ledger.writeAttempted && !ledger.writeSucceeded) {
347
+ instruction = 'Recover from the failed mutation using the current error, then retry the smallest correct Edit/Write.';
348
+ } else {
349
+ instruction = 'Make the smallest correct Edit/Write now; do not end after more read-only analysis.';
238
350
  }
239
- if (ledger.writeAttempted && !ledger.writeSucceeded) {
240
- return 'Recover from the failed mutation using the current error, then retry the smallest correct Edit/Write.';
351
+ } else if (missingEvidence.includes('verification-evidence')) {
352
+ if (ledger.verificationFailed) {
353
+ instruction = 'Use the latest failed verification output, fix the failure, and rerun targeted verification.';
354
+ } else {
355
+ const routedCommands = routedVerificationCommands(routeSummary);
356
+ if (routedCommands.length > 0 && ledger.verificationSucceeded && !ledger.targetedVerificationSucceeded) {
357
+ instruction = `Verification ran but did not match the routed plan (${routedCommands.slice(0, 3).join('; ')}). Run one of those and inspect its result before stopping.`;
358
+ } else {
359
+ instruction = 'Run the routed targeted verification now and inspect its result before stopping.';
360
+ }
241
361
  }
242
- return 'Make the smallest correct Edit/Write now; do not end after more read-only analysis.';
243
362
  }
244
- if (missingEvidence.includes('verification-evidence')) {
245
- if (ledger.verificationFailed) {
246
- return 'Use the latest failed verification output, fix the failure, and rerun targeted verification.';
363
+ // WS-C: the target hint travels with the reason whenever impact evidence is missing and
364
+ // the route names expected files — it must survive branch precedence above.
365
+ if (missingEvidence.includes('impact-evidence')) {
366
+ const expectedFiles = Array.isArray(routeSummary?.expectedSourceFiles) ? routeSummary.expectedSourceFiles : [];
367
+ if (expectedFiles.length > 0 && !expectedFiles.some(
368
+ (expected) => (ledger.sourceFiles || []).some((file) => fileMatchesExpected(file, expected))
369
+ )) {
370
+ const hint = `Read the routed target files (${expectedFiles.slice(0, 3).join(', ')}) — reads of unrelated files do not count as impact evidence.`;
371
+ instruction = instruction ? `${instruction} ${hint}` : hint;
247
372
  }
248
- return 'Run the routed targeted verification now and inspect its result before stopping.';
249
373
  }
250
- return 'Complete the current routed milestone before stopping.';
374
+ return instruction ?? 'Complete the current routed milestone before stopping.';
251
375
  }
252
376
 
253
377
  export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
@@ -263,7 +387,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
263
387
 
264
388
  const sameRequest = !ledger?.requestKey || !state?.requestKey || ledger.requestKey === state.requestKey;
265
389
  const effectiveLedger = sameRequest ? ledger : {};
266
- const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger));
390
+ const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, routeSummary));
267
391
  if (missingEvidence.length === 0) {
268
392
  return { continue: false, notify: false, missingEvidence: [] };
269
393
  }
@@ -278,7 +402,15 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
278
402
  };
279
403
  }
280
404
 
281
- const continuationCount = Number(effectiveLedger?.continuationCount || 0);
405
+ // Continuation attempts only make sense within the request that minted them: a new
406
+ // routed request in the same session must start with a fresh budget, otherwise a cap
407
+ // exhausted on task A suppresses recovery for task B.
408
+ const staleContinuations = effectiveLedger?.continuationRequestKey
409
+ && state?.requestKey
410
+ && effectiveLedger.continuationRequestKey !== state.requestKey;
411
+ const continuationCount = staleContinuations
412
+ ? 0
413
+ : Number(effectiveLedger?.continuationCount || 0);
282
414
  if (continuationCount >= MAX_CONTINUATIONS) {
283
415
  if (effectiveLedger?.notified === true) {
284
416
  return {
@@ -300,7 +432,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
300
432
  }
301
433
 
302
434
  const finalAttempt = continuationCount === MAX_CONTINUATIONS - 1;
303
- const instruction = recoveryInstruction(missingEvidence, effectiveLedger);
435
+ const instruction = recoveryInstruction(missingEvidence, effectiveLedger, routeSummary);
304
436
  return {
305
437
  continue: true,
306
438
  missingEvidence,
@@ -312,11 +444,18 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
312
444
  };
313
445
  }
314
446
 
315
- export async function incrementContinuation(projectRoot, payload = {}, ledger = null) {
447
+ export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null) {
316
448
  const current = ledger || await readExecutionLedger(projectRoot, payload) || freshLedger(payload, null, 'unknown');
449
+ // Mirror evaluateCompletion's staleness rule: a count minted by an earlier request must
450
+ // not be carried into the new request's budget, or the cap fires early (evaluate says 0,
451
+ // persist says 7) and the next request inherits a nearly exhausted budget.
452
+ const stale = requestKey
453
+ && current?.continuationRequestKey
454
+ && current.continuationRequestKey !== requestKey;
317
455
  const next = {
318
456
  ...current,
319
- continuationCount: Number(current.continuationCount || 0) + 1,
457
+ continuationCount: (stale ? 0 : Number(current.continuationCount || 0)) + 1,
458
+ ...(requestKey ? { continuationRequestKey: requestKey } : {}),
320
459
  lastContinuationAt: Date.now(),
321
460
  updatedAt: Date.now(),
322
461
  };
@@ -354,12 +493,31 @@ async function main() {
354
493
  const state = await readRouteState(projectRoot);
355
494
  const ledger = await readExecutionLedger(projectRoot, payload) || {};
356
495
  const result = evaluateCompletion({ state, ledger });
496
+
497
+ // Claude Code invokes Stop again after a Stop hook blocks the first stop. Re-blocking
498
+ // that recovery turn creates a self-sustaining loop, so let it end normally instead.
499
+ // If work still lacks evidence, surface the recovery reason to the user rather than
500
+ // silently ending after the automatic continuation.
501
+ if (payload.stop_hook_active === true) {
502
+ if (result.continue || result.capped || result.notify) {
503
+ const recoveryReason = result.reason
504
+ || 'UKit completion gate reached its continuation limit; unfinished work was not retried again.';
505
+ process.stdout.write(`${JSON.stringify({
506
+ systemMessage: `UKit stopped automatic recovery after one continuation: ${recoveryReason}`,
507
+ })}\n`);
508
+ }
509
+ return;
510
+ }
511
+
357
512
  if (result.continue) {
358
513
  if (result.finalNotice) await markNotified(projectRoot, payload, ledger);
359
- else await incrementContinuation(projectRoot, payload, ledger);
514
+ else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
360
515
  process.stdout.write(`${JSON.stringify({ decision: 'block', reason: result.reason })}\n`);
361
- } else if (result.capped) {
516
+ } else if (result.capped || result.notify) {
517
+ // Non-blocking endings (non-gated modes, or cap reached after the final notice) must
518
+ // still tell the user what is unfinished — a silent end is indistinguishable from a stall.
362
519
  process.stderr.write(`[ukit-completion] ${result.reason}\n`);
520
+ process.stdout.write(`${JSON.stringify({ systemMessage: result.reason })}\n`);
363
521
  }
364
522
  }
365
523
  }
@@ -38,7 +38,11 @@ function run(payloadText, scriptPaths) {
38
38
  ? path.resolve(path.dirname(firstScript), '../..')
39
39
  : (payload.cwd || process.cwd());
40
40
  const startedAt = Date.now();
41
- const deadline = startedAt + TOTAL_BUDGET_MS;
41
+ // The chain budget must always be able to run EVERY child at its full per-child
42
+ // budget — a fixed total silently starves later scripts once a chain grows
43
+ // (UserPromptSubmit now carries 4 hooks). The floor keeps short chains at 10s.
44
+ const totalBudgetMs = Math.max(TOTAL_BUDGET_MS, scriptPaths.length * CHILD_BUDGET_MS);
45
+ const deadline = startedAt + totalBudgetMs;
42
46
  const results = [];
43
47
 
44
48
  for (const scriptPath of scriptPaths) {
@@ -49,7 +53,7 @@ function run(payloadText, scriptPaths) {
49
53
  scriptName,
50
54
  code: 1,
51
55
  stdout: '',
52
- stderr: `hook chain exceeded its ${TOTAL_BUDGET_MS}ms total budget`,
56
+ stderr: `hook chain exceeded its ${totalBudgetMs}ms total budget`,
53
57
  killed: true,
54
58
  elapsedMs: 0,
55
59
  });
@@ -89,7 +93,7 @@ function run(payloadText, scriptPaths) {
89
93
  toolName: payload?.tool_name || null,
90
94
  toolUseId: payload?.tool_use_id || null,
91
95
  elapsedMs,
92
- budgetMs: TOTAL_BUDGET_MS,
96
+ budgetMs: totalBudgetMs,
93
97
  scripts: results.map(({ scriptName, code, killed, elapsedMs: scriptElapsedMs }) => ({
94
98
  scriptName,
95
99
  code,
@@ -98,7 +102,7 @@ function run(payloadText, scriptPaths) {
98
102
  })),
99
103
  });
100
104
 
101
- return { results, elapsedMs, budgetMs: TOTAL_BUDGET_MS };
105
+ return { results, elapsedMs, budgetMs: totalBudgetMs };
102
106
  }
103
107
 
104
108
  try {
@@ -20,7 +20,7 @@ import {
20
20
 
21
21
  export const HOOK_EVENT_MAP = {
22
22
  tool_call: {
23
- 'Read|Grep|Glob': [],
23
+ 'Read|Grep|Glob': ['sensitive-data-guard.sh'],
24
24
  'Edit|Write': [
25
25
  'protect-files.sh',
26
26
  'stale-spec-guard.sh',
@@ -32,6 +32,7 @@ export const HOOK_EVENT_MAP = {
32
32
  Bash: [
33
33
  'auto-allow-bash.sh',
34
34
  'block-dangerous.sh',
35
+ 'sensitive-data-guard.sh',
35
36
  'handoff-model-guard.sh',
36
37
  'context-hardcap-gate.sh',
37
38
  ],
@@ -41,7 +42,7 @@ export const HOOK_EVENT_MAP = {
41
42
  'Edit|Write': ['post-edit-verify.sh', 'record-execution.sh'],
42
43
  Bash: ['compress-output.sh', 'record-execution.sh'],
43
44
  },
44
- before_agent_start: ['skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
45
+ before_agent_start: ['sensitive-data-guard.sh', 'skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
45
46
  'session.compacting': ['reinject-context.sh'],
46
47
  session_start: ['auto-prune-bash.sh', 'reset-compact-pressure.sh', 'handoff-resume.sh'],
47
48
  };
@@ -84,6 +85,7 @@ export const FAIL_CLOSED_SCRIPTS = new Set([
84
85
  'handoff-model-guard.sh',
85
86
  'context-hardcap-gate.sh',
86
87
  'block-dangerous.sh',
88
+ 'sensitive-data-guard.sh',
87
89
  ]);
88
90
 
89
91
  export const ADVISORY_SCRIPTS = new Set([
@@ -168,7 +170,15 @@ function translateExecResult(scriptName, execResult) {
168
170
 
169
171
  export { translateExecResult };
170
172
 
171
- const HOOK_CHAIN_TIMEOUT_MS = 12000;
173
+ // Mirrors hook-chain-runner.mjs's TOTAL/CHILD budget constants: the exec timeout must
174
+ // exceed the runner's own chain budget (max(10s, scripts×4s)) plus parse margin, or the
175
+ // bridge orphans the runner mid-chain once a chain grows past 2 scripts.
176
+ const HOOK_CHAIN_BASE_BUDGET_MS = 10000;
177
+ const HOOK_CHAIN_CHILD_BUDGET_MS = 4000;
178
+ function chainExecTimeoutMs(scriptCount) {
179
+ const budget = Math.max(HOOK_CHAIN_BASE_BUDGET_MS, scriptCount * HOOK_CHAIN_CHILD_BUDGET_MS);
180
+ return Math.min(30000, budget + 2000);
181
+ }
172
182
 
173
183
  function recordHookErrorDiagnostic(projectRoot, sessionId, diagnostic) {
174
184
  try {
@@ -226,7 +236,7 @@ export async function runScriptChain(
226
236
  execResult = await pi.exec(
227
237
  nodeExecutable,
228
238
  [runnerPath, JSON.stringify(payload), ...scriptPaths],
229
- { cwd: projectRoot, timeout: HOOK_CHAIN_TIMEOUT_MS },
239
+ { cwd: projectRoot, timeout: chainExecTimeoutMs(scripts.length) },
230
240
  );
231
241
  } catch (error) {
232
242
  execResult = { code: 1, stdout: '', stderr: error?.message ?? String(error), killed: false };
@@ -515,7 +525,7 @@ export async function runSessionStop(
515
525
  if (suppliedLedger === undefined) {
516
526
  try {
517
527
  if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger);
518
- else await incrementContinuation(projectRoot, payload, ledger);
528
+ else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
519
529
  } catch (error) {
520
530
  pi.logger?.warn?.(`[UKit] continuation bookkeeping failed open: ${error?.message || error}`);
521
531
  }
@@ -206,10 +206,10 @@ This is internal orchestration — end users do not need to know about tiers, th
206
206
 
207
207
  ### Vision lane (capability, not a cost tier)
208
208
 
209
- `unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table. `unic-code` and `unic-smart` cannot read images on the UNIC gateway; they must never guess at image contents.
209
+ `unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table. Whether a mapping can read images is a capability fact, not a provider fact: a mapping that has not **verified** native vision must never guess at image contents — choose native-first when verified, otherwise route to the specialist.
210
210
 
211
211
  - **Gateway detection**: UNIC routing for a Claude Code session is decided ONLY by what changes Claude Code's own outbound endpoint — `ANTHROPIC_BASE_URL` (env var, or the `env.ANTHROPIC_BASE_URL` key in project/home `.claude/settings.json`) containing `unicjsc.com`. Other tools' configs — Codex `config.toml`, Kilo `secrets.json`, or an `OPENAI_BASE_URL` env var — describe a different tool's endpoint entirely and never decide this session's routing.
212
- - **Enforcement**: every image (pasted, local file path, or URL) must be analysed by the `ukit-vision-analyst` agent running on the vision lane before any related edit happens. `Edit`/`Write` are **hard-blocked** until an analysis receipt exists for every pending image; `Read`/`Grep`/`Glob`/`Bash` stay unblocked so the analyst itself can see the image and write its receipt.
212
+ - **Advisory routing (no hard block)**: when an image reaches the prompt, the vision router reminds the session to have `ukit-vision-analyst` analyse it before relying on its contents. Edits are **never blocked** — correctness relies on the model routing images to the analyst instead of guessing.
213
213
  - This is internal orchestration — end users never invoke a vision command directly; `ukit install` plus natural language remains the whole surface. No new commands.
214
214
 
215
215
  ## Session Start — OpenCode
@@ -206,7 +206,7 @@ This is internal orchestration — end users do not need to know about tiers, th
206
206
 
207
207
  ### Vision lane (capability, not a cost tier)
208
208
 
209
- `unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table. `unic-code` and `unic-smart` cannot read images on the UNIC gateway; they must never guess at image contents.
209
+ `unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table. Whether a mapping can read images is a capability fact, not a provider fact: a mapping that has not **verified** native vision must never guess at image contents — choose native-first when verified, otherwise route to the specialist.
210
210
 
211
211
  - **Gateway detection**: UNIC routing for a Claude Code session is decided ONLY by what changes Claude Code's own outbound endpoint — `ANTHROPIC_BASE_URL` (env var, or the `env.ANTHROPIC_BASE_URL` key in project/home `.claude/settings.json`) containing `unicjsc.com`. Other tools' configs — Codex `config.toml`, Kilo `secrets.json`, or an `OPENAI_BASE_URL` env var — describe a different tool's endpoint entirely and never decide this session's routing.
212
212
  - **Advisory routing (no hard block)**: when an image reaches the prompt, the vision router reminds the session to have `ukit-vision-analyst` analyse it before relying on its contents. Edits are **never blocked** — correctness relies on the model routing images to the analyst instead of guessing.