@ngockhoale/ukit 3.4.1 → 3.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/CHANGELOG.md +23 -0
  2. package/package.json +1 -1
  3. package/src/cli/commands/code.js +29 -5
  4. package/src/cli/commands/decision.js +18 -4
  5. package/src/cli/commands/doctor.js +7 -3
  6. package/src/cli/commands/install.js +29 -4
  7. package/src/cli/commands/memory.js +25 -5
  8. package/src/cli/commands/telemetry.js +18 -1
  9. package/src/cli/commands/vm.js +7 -1
  10. package/src/context/detectProjectContext.js +7 -2
  11. package/src/core/agentRuntime/contract.js +5 -1
  12. package/src/core/agentRuntime/eventStore.js +54 -7
  13. package/src/core/agentRuntime/recovery.js +22 -15
  14. package/src/core/agentRuntime/supervisor.js +71 -13
  15. package/src/core/applyPlan.js +11 -1
  16. package/src/core/codeintel/compiler.js +51 -8
  17. package/src/core/codeintel/diagnostics.js +124 -33
  18. package/src/core/codeintel/freshness.js +25 -12
  19. package/src/core/codeintel/invalidation.js +11 -3
  20. package/src/core/codeintel/retriever.js +53 -20
  21. package/src/core/codeintel/router.js +19 -9
  22. package/src/core/codeintel/summaries.js +4 -3
  23. package/src/core/codeintel/vectorProvider.js +30 -4
  24. package/src/core/compact/index.js +24 -7
  25. package/src/core/compact/threshold.js +49 -14
  26. package/src/core/diffPlan.js +51 -23
  27. package/src/core/ensureGitignore.js +19 -2
  28. package/src/core/fileOps.js +61 -0
  29. package/src/core/memory/hygiene.js +51 -1
  30. package/src/core/memory/migrate.js +41 -21
  31. package/src/core/memory/store.js +96 -61
  32. package/src/core/metadata.js +37 -2
  33. package/src/core/observability/adapters/ingest.js +30 -2
  34. package/src/core/observability/emit/config.js +19 -4
  35. package/src/core/observability/emit/crash.js +3 -1
  36. package/src/core/observability/emit/recorder.js +15 -7
  37. package/src/core/observability/privacy/sanitizeObserved.js +3 -1
  38. package/src/core/observability/segments/internal.js +36 -8
  39. package/src/core/observability/segments/retention.js +11 -0
  40. package/src/core/observability/support/import.js +27 -1
  41. package/src/core/output/index.js +16 -1
  42. package/src/core/permissionDoctor.js +72 -9
  43. package/src/core/repairBrokenHooks.js +15 -2
  44. package/src/core/reviewPanelAggregate.js +26 -10
  45. package/src/core/runInstallPipeline.js +71 -22
  46. package/src/core/runtimeConfig.js +2 -0
  47. package/src/core/status.js +2 -0
  48. package/src/core/taskBudgetValidator.js +7 -1
  49. package/src/core/taskProgressGuard.js +11 -1
  50. package/src/core/unattendedDoctor.js +36 -5
  51. package/src/core/uninstall.js +52 -12
  52. package/src/core/update.js +5 -1
  53. package/src/decision/client.js +158 -27
  54. package/src/decision/reviewVerdict.js +23 -7
  55. package/src/diagnostics/failurePatterns.js +1 -1
  56. package/src/diagnostics/feedbackEvents.js +1 -1
  57. package/src/diagnostics/routeOutcomes.js +42 -4
  58. package/src/diagnostics/skillAccuracy.js +35 -4
  59. package/src/index/buildIndex.js +123 -26
  60. package/src/index/fixLoopEscalation.js +3 -0
  61. package/src/index/gitHooks.js +99 -29
  62. package/src/index/importResolution.js +7 -1
  63. package/src/index/playbookRegistry.js +15 -11
  64. package/src/index/queryIndex.js +28 -10
  65. package/src/index/routeResolver.js +8 -3
  66. package/src/index/taskRouting.js +37 -2
  67. package/src/learning/codeProposals.js +24 -5
  68. package/src/learning/selfImprove.js +29 -5
  69. package/src/learning/tunedOverlay.js +18 -7
  70. package/src/learning/tuning.js +10 -4
  71. package/src/skill/auditSkill.js +46 -7
  72. package/template_project/.claude/commands/ukit/handoff-review.md +4 -1
  73. package/template_project/.claude/hooks/auto-allow-bash.sh +10 -1
  74. package/template_project/.claude/hooks/block-dangerous.mjs +10 -2
  75. package/template_project/.claude/hooks/handoff-model-guard.sh +46 -16
  76. package/template_project/.claude/hooks/reset-compact-pressure.sh +128 -72
  77. package/template_project/.claude/hooks/sensitive-data-guard.mjs +394 -11
  78. package/template_project/.claude/hooks/session-episode.sh +60 -28
  79. package/template_project/.claude/hooks/verification-guard.sh +26 -15
  80. package/template_project/.claude/skills/pptx/scripts/thumbnail.py +6 -1
  81. package/template_project/.claude/ukit/index/lib/index-core.mjs +156 -39
  82. package/template_project/.claude/ukit/index/playbook-registry.mjs +15 -11
  83. package/template_project/.claude/ukit/index/post-edit-verify.mjs +25 -4
  84. package/template_project/.claude/ukit/index/pre-edit-backup.mjs +4 -0
  85. package/template_project/.claude/ukit/index/provision-worktree.mjs +15 -10
  86. package/template_project/.claude/ukit/index/query-index.mjs +13 -6
  87. package/template_project/.claude/ukit/index/reset-auto-permissions.mjs +127 -25
  88. package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +36 -14
  89. package/template_project/.claude/ukit/index/review-verdict.mjs +93 -19
  90. package/template_project/.claude/ukit/index/route-resolver.mjs +8 -3
  91. package/template_project/.claude/ukit/index/route-task.mjs +15 -0
  92. package/template_project/.claude/ukit/index/safe-patch.mjs +4 -1
  93. package/template_project/.claude/ukit/index/sidecar-decision.mjs +43 -10
  94. package/template_project/.claude/ukit/index/stale-spec-check.mjs +13 -3
  95. package/template_project/.claude/ukit/index/task-budget-validator.mjs +7 -1
  96. package/template_project/.claude/ukit/index/unic-decision.mjs +179 -28
  97. package/template_project/.claude/ukit/index/unic-gateway.mjs +33 -8
  98. package/template_project/.claude/ukit/index/verify-context.mjs +9 -2
  99. package/template_project/.claude/ukit/index/worktree-sweep.mjs +89 -31
  100. package/template_project/.claude/ukit/runtime/compact-threshold.mjs +47 -15
  101. package/template_project/.claude/ukit/runtime/execution-ledger.mjs +63 -30
  102. package/template_project/.claude/ukit/runtime/hook-field-salvage.mjs +49 -13
  103. package/template_project/.claude/ukit/runtime/hook-input.sh +48 -13
  104. package/template_project/.claude/ukit/runtime/hook-telemetry.mjs +92 -7
  105. package/template_project/.claude/ukit/runtime/memory-freshness.mjs +17 -0
  106. package/template_project/.claude/ukit/runtime/observability-emit.mjs +38 -10
  107. package/template_project/.claude/ukit/runtime/output-compression.mjs +11 -0
  108. package/template_project/.claude/ukit/runtime/reinject-context.mjs +24 -3
  109. package/template_project/.claude/ukit/runtime/resumable-run.mjs +62 -32
  110. package/template_project/.claude/ukit/runtime/token-utils.mjs +57 -14
@@ -34,6 +34,16 @@ function safeItems(artifact) {
34
34
  return artifact.items.filter((entry) => entry && typeof entry === 'object');
35
35
  }
36
36
 
37
+ // C92-B-03 consumer symmetry: an artifact written under a different
38
+ // INDEX_SCHEMA_VERSION (older OR newer — a package downgrade leaves newer
39
+ // artifacts on disk) is treated as absent, exactly like getFileOutline's
40
+ // schema gate on symbols.json. Without it, queryCodeIndex scores the
41
+ // mismatched set and looks healthy while outline: silently disappears.
42
+ function safeSchemaArtifact(artifact) {
43
+ const normalized = ensureArtifact(artifact);
44
+ return normalized.schemaVersion === INDEX_SCHEMA_VERSION ? normalized : { items: [] };
45
+ }
46
+
37
47
  export async function queryCodeIndex({ rootDir = process.cwd(), query, limit = 5 } = {}) {
38
48
  const normalizedQuery = String(query ?? '').trim();
39
49
  if (!normalizedQuery) {
@@ -67,7 +77,7 @@ export async function queryCodeIndex({ rootDir = process.cwd(), query, limit = 5
67
77
 
68
78
  for (const file of safeItems(files)) {
69
79
  const filePath = file.filePath;
70
- if (isLikelyTestFilePath(filePath) && !includeLikelyTestFiles) {
80
+ if (typeof filePath !== 'string' || !filePath || (isLikelyTestFilePath(filePath) && !includeLikelyTestFiles)) {
71
81
  continue;
72
82
  }
73
83
  const searchDescriptor = searchDescriptorsByFile.get(filePath) ?? createFileSearchDescriptor({ filePath, symbols: [] });
@@ -235,8 +245,10 @@ async function loadQuerySearchBundle(rootDir) {
235
245
  readArtifact(absoluteRoot, INDEX_ARTIFACTS.symbols),
236
246
  ])
237
247
  .then(([files, symbols]) => {
238
- const safeFiles = ensureArtifact(files);
239
- const safeSymbols = ensureArtifact(symbols);
248
+ // C92-B-03: same schema contract getFileOutline applies — a mismatched
249
+ // artifact reads as empty instead of feeding ranking.
250
+ const safeFiles = safeSchemaArtifact(files);
251
+ const safeSymbols = safeSchemaArtifact(symbols);
240
252
  const symbolMap = groupBy(safeItems(safeSymbols), (item) => item.filePath);
241
253
  return {
242
254
  files: safeFiles,
@@ -269,11 +281,11 @@ async function loadQuerySupportBundle(rootDir) {
269
281
  readArtifact(absoluteRoot, INDEX_ARTIFACTS.archetypes),
270
282
  ]))
271
283
  .then(([files, imports, testsMap, hotspots, archetypes]) => {
272
- const safeFiles = ensureArtifact(files);
273
- const safeImports = ensureArtifact(imports);
274
- const safeTestsMap = ensureArtifact(testsMap);
275
- const safeHotspots = ensureArtifact(hotspots);
276
- const safeArchetypes = ensureArtifact(archetypes);
284
+ const safeFiles = safeSchemaArtifact(files);
285
+ const safeImports = safeSchemaArtifact(imports);
286
+ const safeTestsMap = safeSchemaArtifact(testsMap);
287
+ const safeHotspots = safeSchemaArtifact(hotspots);
288
+ const safeArchetypes = safeSchemaArtifact(archetypes);
277
289
  const indexedFileSet = new Set(safeItems(safeFiles).map((item) => item.filePath));
278
290
  const aliasContextPromise = importsNeedAliasContext(safeItems(safeImports))
279
291
  ? loadImportAliasContext({ rootDir: absoluteRoot })
@@ -566,7 +578,7 @@ function buildResolvedImportGraphs(rootDir, imports, indexedFileSet, importAlias
566
578
  const importersByTarget = new Map();
567
579
 
568
580
  for (const edge of imports) {
569
- if (!edge?.from || !edge?.to) {
581
+ if (typeof edge?.from !== 'string' || typeof edge?.to !== 'string' || !edge.from || !edge.to) {
570
582
  continue;
571
583
  }
572
584
 
@@ -700,6 +712,12 @@ function ensureArtifact(artifact) {
700
712
  };
701
713
  }
702
714
 
715
+ // The cache keys matched against this escaped value are produced by
716
+ // JSON.stringify, which escapes control characters (\n, \b, \u00XX, lone
717
+ // surrogates) on top of backslash/quote. Hand-escaping only \\ and \" means a
718
+ // rootDir containing control characters never matches its own cache keys, so
719
+ // clearIndexArtifactCache leaves stale results behind. Reuse JSON.stringify
720
+ // itself and strip the surrounding quotes to stay in lockstep.
703
721
  function escapeJsonString(value) {
704
- return String(value).replaceAll('\\', '\\\\').replaceAll('"', '\\"');
722
+ return JSON.stringify(String(value)).slice(1, -1);
705
723
  }
@@ -312,7 +312,7 @@ export function deriveRiskFloor({
312
312
  codes.push('shared-impact');
313
313
  }
314
314
  if (
315
- /(^|\/)(package\.json|manifests\/|.*\.d\.ts$|(^|\/)api\/|openapi|swagger)/i.test(target)
315
+ /(^|\/)(package\.json|manifests\/|.*\.d\.ts$|api\/|openapi|swagger)/i.test(target)
316
316
  || /\b(public api|breaking change|api contract|semver)\b/i.test(signalText)
317
317
  ) {
318
318
  codes.push('public-contract');
@@ -417,14 +417,19 @@ export function deriveFastPath({
417
417
  return null;
418
418
  }
419
419
  const reasons = (riskFloor.codes ?? []).filter((code) => FAST_PATH_REASON_CODES.has(code));
420
- const hasLocalTarget = Boolean(targetFile) || contextPreview?.primaryTargets?.length === 1;
420
+ // C91-022 (W3-TR2): the shared-impact veto must consult the same target that
421
+ // granted locality — the explicit targetFile, else the single primaryTarget.
422
+ const localTarget = targetFile || (contextPreview?.primaryTargets?.length === 1
423
+ ? contextPreview.primaryTargets[0]
424
+ : null);
425
+ const hasLocalTarget = Boolean(localTarget);
421
426
  const hasBoundedVerification = (verificationRecommendation?.commands?.length ?? 0) > 0
422
427
  || buildExecutionContract(executionMode)?.verificationPolicy === 'minimal-or-targeted';
423
428
  const eligible = FAST_PATH_MODES.has(executionMode)
424
429
  && hasLocalTarget
425
430
  && riskFloor.floor === 'none'
426
431
  && hasBoundedVerification
427
- && !isSharedImpactFile(targetFile);
432
+ && !isSharedImpactFile(localTarget);
428
433
  return {
429
434
  eligible,
430
435
  suppressed: eligible ? [...FAST_PATH_SUPPRESSED] : [],
@@ -2452,6 +2452,29 @@ function normalizeRelativeFile(rootDir, rawFilePath) {
2452
2452
  return trimmed.replace(/^\.\/+/, '').replaceAll('\\', '/');
2453
2453
  }
2454
2454
 
2455
+ // C91-022 (W3-TR3): the probe stays a health check but must not re-parse ~23MB
2456
+ // of index artifacts on every routed prompt. Results are cached per index-dir
2457
+ // fingerprint (entry names + mtime + size of each *.json artifact), so an
2458
+ // unchanged index costs O(#files) stats and zero parses; any artifact change
2459
+ // shifts the fingerprint and re-probes. The cache is bounded so long-running
2460
+ // hosts probing many roots never grow it without limit.
2461
+ const INDEX_PROBE_CACHE_MAX = 64;
2462
+ const indexProbeCache = new Map();
2463
+
2464
+ async function fingerprintIndexDir(indexDir, entries) {
2465
+ const jsonEntries = entries.filter((entry) => entry.endsWith('.json')).sort();
2466
+ const parts = [];
2467
+ for (const entry of jsonEntries) {
2468
+ try {
2469
+ const stat = await fs.stat(path.join(indexDir, entry));
2470
+ parts.push(`${entry}:${stat.mtimeMs}:${stat.size}`);
2471
+ } catch {
2472
+ parts.push(`${entry}:missing`);
2473
+ }
2474
+ }
2475
+ return parts.join('|');
2476
+ }
2477
+
2455
2478
  async function findUnreadableIndexArtifact(rootDir) {
2456
2479
  const indexDir = path.join(rootDir, '.cache', 'index');
2457
2480
  let entries;
@@ -2461,6 +2484,13 @@ async function findUnreadableIndexArtifact(rootDir) {
2461
2484
  return null;
2462
2485
  }
2463
2486
 
2487
+ const fingerprint = await fingerprintIndexDir(indexDir, entries);
2488
+ const cacheKey = `${indexDir}\0${fingerprint}`;
2489
+ if (indexProbeCache.has(cacheKey)) {
2490
+ return indexProbeCache.get(cacheKey);
2491
+ }
2492
+
2493
+ let unreadable = null;
2464
2494
  for (const entry of entries) {
2465
2495
  if (!entry.endsWith('.json')) {
2466
2496
  continue;
@@ -2469,11 +2499,16 @@ async function findUnreadableIndexArtifact(rootDir) {
2469
2499
  try {
2470
2500
  JSON.parse(await fs.readFile(path.join(indexDir, entry), 'utf8'));
2471
2501
  } catch (error) {
2472
- return new Error(`${entry}: ${error?.message ?? String(error)}`);
2502
+ unreadable = new Error(`${entry}: ${error?.message ?? String(error)}`);
2503
+ break;
2473
2504
  }
2474
2505
  }
2475
2506
 
2476
- return null;
2507
+ if (indexProbeCache.size >= INDEX_PROBE_CACHE_MAX) {
2508
+ indexProbeCache.clear();
2509
+ }
2510
+ indexProbeCache.set(cacheKey, unreadable);
2511
+ return unreadable;
2477
2512
  }
2478
2513
 
2479
2514
  async function pathExists(filePath) {
@@ -28,7 +28,7 @@ const TASKS_DIR_REL = path.join('docs', 'AI_HANDOFF', 'tasks');
28
28
  const ARCHIVE_DIR_REL = path.join('docs', 'AI_HANDOFF', 'archive');
29
29
  const ARTIFACT_REL = path.join('.ukit', 'storage', 'cache', 'failure-patterns.json');
30
30
  const PLAN_REL = 'docs/plans/SELF_IMPROVE_CODE_LANE.md';
31
- const SIGNATURE_FIELD_RE = /signature["']?\s*:\s*["'`]?([^\n"'`,}]+)/gi;
31
+ const SIGNATURE_FIELD_RE = /signature["']?\s*:\s*(?:`([^`\n]*)`|"((?:[^"\\\n]|\\.)*)"|'([^'\n]*)')/gi;
32
32
  const MIN_SESSIONS = 2;
33
33
  const AUTO_NAME_RE = /^AUTO-(\d+)\.md$/;
34
34
 
@@ -94,11 +94,19 @@ export function mapPatternToFixTarget(pattern) {
94
94
  return null;
95
95
  }
96
96
 
97
+ // W3-F4: strip ./ and resolve ../ so syntactic variants can't slip past the
98
+ // denylist. On case-insensitive filesystems (win32/darwin) a case variant
99
+ // names the same denied file, so the patterns are folded there — Linux keeps
100
+ // case-sensitive matching (denylist semantics unchanged, planner note).
101
+ const DENY_TARGET_RES = (process.platform === 'win32' || process.platform === 'darwin')
102
+ ? DENIED_TARGET_RES.map((re) => (re.ignoreCase ? re : new RegExp(re.source, `${re.flags}i`)))
103
+ : DENIED_TARGET_RES;
104
+
97
105
  function normalizeTargets(pattern) {
98
106
  const files = Array.isArray(pattern?.files)
99
107
  ? pattern.files.filter((f) => typeof f === 'string' && f.trim())
100
108
  : [];
101
- return files.map((f) => f.replace(/\\/g, '/'));
109
+ return files.map((f) => path.posix.normalize(f.replace(/\\/g, '/')));
102
110
  }
103
111
 
104
112
  async function loadPatterns(projectRoot) {
@@ -157,8 +165,19 @@ async function scanHandoffDirs(projectRoot) {
157
165
  }
158
166
  SIGNATURE_FIELD_RE.lastIndex = 0;
159
167
  for (const match of text.matchAll(SIGNATURE_FIELD_RE)) {
160
- const sig = match[1].trim();
161
- if (sig) signatures.add(sig);
168
+ const raw = (match[1] ?? match[2] ?? match[3] ?? '').trim();
169
+ if (!raw) continue;
170
+ // JSON "…" captures may carry escapes; decode so the scanned signature
171
+ // equals the raw pattern.signature stored in the evidence block.
172
+ let sig = raw;
173
+ if (match[2] !== undefined) {
174
+ try {
175
+ sig = JSON.parse(`"${match[2]}"`);
176
+ } catch {
177
+ // keep raw
178
+ }
179
+ }
180
+ signatures.add(sig);
162
181
  }
163
182
  }
164
183
  }
@@ -300,7 +319,7 @@ export async function proposeCodeFixTasks(projectRoot, {
300
319
  entry.fixClass = fixClass.id;
301
320
 
302
321
  const targets = normalizeTargets(pattern);
303
- if (targets.some((f) => DENIED_TARGET_RES.some((re) => re.test(f)))) {
322
+ if (targets.some((f) => DENY_TARGET_RES.some((re) => re.test(f)))) {
304
323
  entry.reason = 'denied-target';
305
324
  continue;
306
325
  }
@@ -37,6 +37,7 @@ import path from 'node:path';
37
37
  import { inspectRuntimeConfig, resolveDecisionRuntimeStage } from '../core/runtimeConfig.js';
38
38
  import { withFileLock } from '../core/fileOps.js';
39
39
  import { detectProjectContext } from '../context/detectProjectContext.js';
40
+ import { buildUserPaths } from '../core/userPaths.js';
40
41
  import { backfillEpisodes } from '../core/memory/episodes.js';
41
42
  import { mineFailurePatterns } from '../diagnostics/failurePatterns.js';
42
43
  import { collectFeedbackEvents } from '../diagnostics/feedbackEvents.js';
@@ -68,13 +69,27 @@ async function readJson(filePath) {
68
69
  }
69
70
 
70
71
  async function writeJsonAtomic(filePath, value) {
71
- const tmp = `${filePath}.tmp-${process.pid}`;
72
+ const tmp = `${filePath}.tmp-${process.pid}-${Math.random().toString(16).slice(2)}`;
72
73
  try {
73
74
  await fs.mkdir(path.dirname(filePath), { recursive: true });
74
75
  await fs.writeFile(tmp, `${JSON.stringify(value, null, 2)}\n`);
75
76
  await fs.rename(tmp, filePath);
76
- } catch {
77
+ return null;
78
+ } catch (error) {
77
79
  await fs.rm(tmp, { force: true }).catch(() => {});
80
+ return error?.message ?? String(error);
81
+ }
82
+ }
83
+
84
+ // W3-F3: read the user-level config layer raw so its tuning opt-out can
85
+ // override project config (resolveApplyMode treats user 'manual'|'off'|disabled
86
+ // as the more restrictive mode that wins).
87
+ async function readUserRawConfig(homeDir) {
88
+ const configPath = buildUserPaths({ homeDir }).configPath;
89
+ try {
90
+ return JSON.parse(await fs.readFile(configPath, 'utf8'));
91
+ } catch {
92
+ return null;
78
93
  }
79
94
  }
80
95
 
@@ -167,7 +182,11 @@ export async function runSelfImprove(projectRoot, {
167
182
  }
168
183
  const stamp = await readJson(stampPath);
169
184
  const lastRunAt = Date.parse(stamp?.lastRunAt ?? '');
170
- if (!force && Number.isFinite(lastRunAt) && now - lastRunAt < minIntervalMs) {
185
+ // W3-F5: a future lastRunAt (clock skew, hand-edit) would keep
186
+ // `now - lastRunAt` negative forever — treat it as stale so the pass
187
+ // re-stamps the clock-corrected value.
188
+ const stampIsFuture = Number.isFinite(lastRunAt) && lastRunAt > now;
189
+ if (!force && Number.isFinite(lastRunAt) && !stampIsFuture && now - lastRunAt < minIntervalMs) {
171
190
  return { status: 'skipped', reason: 'min_interval' };
172
191
  }
173
192
 
@@ -227,7 +246,7 @@ export async function runSelfImprove(projectRoot, {
227
246
  }));
228
247
  steps.push(await step('tuning', async () => {
229
248
  const result = await computeTuningSuggestions(root);
230
- const mode = resolveApplyMode(rawConfig ?? {});
249
+ const mode = resolveApplyMode(rawConfig ?? {}, await readUserRawConfig(homeDir));
231
250
  const applied = mode === 'auto' ? await applyTuning(root, result.suggestions, { now }) : [];
232
251
  return { status: 'ok', mode, suggestions: result.suggestions.length, applied };
233
252
  }));
@@ -256,10 +275,15 @@ export async function runSelfImprove(projectRoot, {
256
275
  };
257
276
  }));
258
277
 
259
- await writeJsonAtomic(stampPath, {
278
+ const stampError = await writeJsonAtomic(stampPath, {
260
279
  lastRunAt: new Date(now).toISOString(),
261
280
  steps,
262
281
  });
282
+ // W3-F7: surface the failure — a swallowed stamp write leaves lastRunAt
283
+ // stale and the next trigger re-runs the full pass invisibly.
284
+ if (stampError !== null) {
285
+ steps.push({ name: 'stamp', status: 'failed', reason: stampError });
286
+ }
263
287
  return steps;
264
288
  }, { maxWaitMs: 500 });
265
289
 
@@ -67,7 +67,7 @@ export async function readTunedDocument(projectRoot) {
67
67
 
68
68
  export async function writeTunedDocument(projectRoot, doc) {
69
69
  const filePath = path.join(projectRoot, TUNED_REL);
70
- const tmp = `${filePath}.tmp-${process.pid}`;
70
+ const tmp = `${filePath}.tmp-${process.pid}-${Math.random().toString(16).slice(2)}`;
71
71
  try {
72
72
  await fs.mkdir(path.dirname(filePath), { recursive: true });
73
73
  await fs.writeFile(tmp, `${JSON.stringify({ ...doc, version: TUNED_VERSION }, null, 2)}\n`);
@@ -79,13 +79,24 @@ export async function writeTunedDocument(projectRoot, doc) {
79
79
  }
80
80
  }
81
81
 
82
- // Resolves the effective apply mode from a raw (pre-default) config object:
83
- // absent → 'auto' (the shipped default).
84
- export function resolveApplyMode(rawConfig) {
82
+ // Resolves the effective apply mode from raw (pre-default) config objects:
83
+ // absent → 'auto' (the shipped default). When the user-level layer is passed,
84
+ // the more restrictive mode wins so a user opt-out ('manual'|'off'|disabled)
85
+ // can never be overridden back to auto-apply by project config.
86
+ const APPLY_MODE_RANK = { off: 0, manual: 1, auto: 2 };
87
+ export function resolveApplyMode(rawConfig, userConfig) {
85
88
  const tuning = rawConfig?.learning?.tuning;
86
- if (tuning?.enabled === false) return 'off';
87
- const mode = tuning?.applyMode;
88
- return mode === 'manual' || mode === 'off' || mode === 'auto' ? mode : 'auto';
89
+ const enabled = tuning?.enabled !== false
90
+ && (!userConfig || userConfig?.learning?.tuning?.enabled !== false);
91
+ if (!enabled) return 'off';
92
+ const modeOf = (cfg) => {
93
+ const mode = cfg?.learning?.tuning?.applyMode;
94
+ return mode === 'manual' || mode === 'off' || mode === 'auto' ? mode : 'auto';
95
+ };
96
+ const mode = modeOf(rawConfig);
97
+ if (userConfig === undefined) return mode;
98
+ const userMode = modeOf(userConfig);
99
+ return APPLY_MODE_RANK[userMode] < APPLY_MODE_RANK[mode] ? userMode : mode;
89
100
  }
90
101
 
91
102
  // Returns a copy of `config` with `overrides` set at their dotted paths.
@@ -24,6 +24,8 @@
24
24
  import fs from 'node:fs/promises';
25
25
  import path from 'node:path';
26
26
 
27
+ import { mergeTunedOverlay } from './tunedOverlay.js';
28
+
27
29
  const LEARNING_DIR_REL = path.join('.ukit', 'storage', 'learning');
28
30
  const SUGGESTIONS_REL = path.join(LEARNING_DIR_REL, 'suggestions.json');
29
31
  const FEEDBACK_EVENTS_REL = path.join(LEARNING_DIR_REL, 'feedback-events.json');
@@ -63,7 +65,7 @@ async function readJson(filePath) {
63
65
 
64
66
  async function writeArtifact(filePath, result) {
65
67
  const dir = path.dirname(filePath);
66
- const tmp = path.join(dir, `.suggestions-${process.pid}.tmp`);
68
+ const tmp = path.join(dir, `.suggestions-${process.pid}-${Math.random().toString(16).slice(2)}.tmp`);
67
69
  try {
68
70
  await fs.mkdir(dir, { recursive: true });
69
71
  await fs.writeFile(tmp, JSON.stringify(result, null, 2));
@@ -197,10 +199,14 @@ export async function computeTuningSuggestions(projectRoot) {
197
199
  result.skipped.push({ target: 'skills.*', reason: 'no skill-accuracy.json artifact' });
198
200
  }
199
201
 
200
- const weights = isObject(config?.codeIntel?.retriever?.weights)
201
- ? config.codeIntel.retriever.weights
202
+ // W3-F2: suggestions must ratchet off the effective (overlay-applied)
203
+ // value — merging tuned.json mirrors what inspectRuntimeConfig feeds the
204
+ // runtime. In non-'auto' modes mergeTunedOverlay returns the raw config.
205
+ const effectiveConfig = await mergeTunedOverlay(projectRoot, config ?? {});
206
+ const weights = isObject(effectiveConfig?.codeIntel?.retriever?.weights)
207
+ ? effectiveConfig.codeIntel.retriever.weights
202
208
  : DEFAULT_WEIGHTS;
203
- const debugLoopThreshold = config?.orchestration?.escalation?.debugLoopThreshold;
209
+ const debugLoopThreshold = effectiveConfig?.orchestration?.escalation?.debugLoopThreshold;
204
210
 
205
211
  laneWeightSuggestions(laneStats, weights, result.suggestions, result.skipped);
206
212
  escalationSuggestion(feedbackEvents, debugLoopThreshold, result.suggestions, result.skipped);
@@ -1,5 +1,6 @@
1
1
  import fs from 'node:fs/promises';
2
2
  import path from 'node:path';
3
+ import { writeFileAtomic } from '../core/fileOps.js';
3
4
 
4
5
  const TEMPLATE_FILES = [
5
6
  'pressure-scenario-template.md',
@@ -7,7 +8,21 @@ const TEMPLATE_FILES = [
7
8
  'trigger-accuracy-template.md',
8
9
  ];
9
10
 
11
+ // C91-022 (W3-skill): skillName becomes a path segment — reject traversal and
12
+ // separators before path.join so `../../x` can never escape docs/skill-audits.
13
+ function isValidSkillName(skillName) {
14
+ return typeof skillName === 'string'
15
+ && skillName.trim().length > 0
16
+ && skillName !== '.'
17
+ && !skillName.includes('..')
18
+ && !skillName.includes('/')
19
+ && !skillName.includes('\\');
20
+ }
21
+
10
22
  export function getAuditDir(rootDir, skillName) {
23
+ if (!isValidSkillName(skillName)) {
24
+ throw new Error(`invalid skill name: ${JSON.stringify(skillName)} — expected a plain directory name without path separators or ".."`);
25
+ }
11
26
  return path.join(rootDir, 'docs', 'skill-audits', skillName);
12
27
  }
13
28
 
@@ -64,7 +79,7 @@ export async function recordAuditEntry({ rootDir = process.cwd(), skillName, ent
64
79
  };
65
80
 
66
81
  history.push(record);
67
- await fs.writeFile(historyPath, `${JSON.stringify(history, null, 2)}\n`, 'utf8');
82
+ await writeFileAtomic(historyPath, `${JSON.stringify(history, null, 2)}\n`);
68
83
 
69
84
  return record;
70
85
  }
@@ -88,14 +103,38 @@ export async function statusSkillAudit({ rootDir = process.cwd(), skillName } =
88
103
  }
89
104
 
90
105
  async function readHistory(historyPath) {
91
- const raw = await fs.readFile(historyPath, 'utf8').catch(() => null);
92
- if (!raw) {
93
- return [];
106
+ // C92-H-04: "absent" and "unreadable" are NOT the same. Only ENOENT yields a
107
+ // fresh array — any other failure to read or parse an existing history must
108
+ // fail loudly, or recordAuditEntry would overwrite the whole audit trail
109
+ // with a single row and report it as success.
110
+ let raw;
111
+ try {
112
+ raw = await fs.readFile(historyPath, 'utf8');
113
+ } catch (error) {
114
+ if (error?.code === 'ENOENT') {
115
+ return [];
116
+ }
117
+ const wrapped = new Error(
118
+ `cannot read skill-audit history at ${historyPath}: ${error?.message ?? error}. ` +
119
+ 'Fix the file permissions or remove it to start a fresh audit trail.',
120
+ { cause: error },
121
+ );
122
+ wrapped.code = 'ERR_SKILL_AUDIT_HISTORY_UNREADABLE';
123
+ throw wrapped;
94
124
  }
95
125
  try {
96
126
  const parsed = JSON.parse(raw);
97
- return Array.isArray(parsed) ? parsed : [];
98
- } catch {
99
- return [];
127
+ if (!Array.isArray(parsed)) {
128
+ throw new Error(`expected a JSON array at the top level, got ${Array.isArray(parsed) ? 'array' : typeof parsed}`);
129
+ }
130
+ return parsed;
131
+ } catch (error) {
132
+ const wrapped = new Error(
133
+ `malformed skill-audit history at ${historyPath}: ${error?.message ?? error}. ` +
134
+ 'The file was left untouched — repair or remove it before recording a new entry.',
135
+ { cause: error },
136
+ );
137
+ wrapped.code = 'ERR_SKILL_AUDIT_HISTORY_MALFORMED';
138
+ throw wrapped;
100
139
  }
101
140
  }
@@ -129,7 +129,10 @@ node .claude/ukit/index/review-panel-aggregate.mjs docs/AI_HANDOFF/tasks/TASK-xx
129
129
 
130
130
  The script emits an Agreement Map (finding → set of panel members reporting it) and a
131
131
  consensus verdict (`approved` when ≥2 members approve and no critical;
132
- `changes_requested` when ≥1 critical or ≥2 request changes; else `approved_minor`).
132
+ `changes_requested` on ≥1 critical verdict/finding, ≥1 member requesting
133
+ changes, or zero parseable verdict blocks — it fails closed; else
134
+ `approved_minor`). Only the latest review round counts: `## Reviewer Verdict
135
+ (round N)` blocks supersede earlier rounds.
133
136
  `consensus≥2 identical findings = high signal` — a finding reported by two or more
134
137
  members is high-signal. The consensus verdict, not any single member's vote, drives
135
138
  `NEXT_STATUS_FOR_INDEX`.
@@ -107,7 +107,16 @@ fi
107
107
 
108
108
  BIN_BASENAME=$(basename "$BIN")
109
109
 
110
- if [[ "$BIN" =~ ^(rm|sudo|dd|mkfs|shutdown|reboot|halt|poweroff|chown|killall|pkill)$ ]]; then
110
+ if [[ "$BIN_BASENAME" =~ ^(rm|sudo|dd|mkfs|shutdown|reboot|halt|poweroff|chown|killall|pkill)$ ]]; then
111
+ exit 0
112
+ fi
113
+
114
+ # W2-D02: wrapper executables launch an arbitrary operand (a command string or
115
+ # another binary), so minting Bash(<wrapper>:*) allows anything they can run.
116
+ # They are NOT in the token-skip list above: skipping would resolve BIN to the
117
+ # operand token (`timeout 30 ls` → BIN=30, `ssh host ls` → BIN=host) and mint a
118
+ # nonsense wildcard rule. Deny outright and fall back to the permission prompt.
119
+ if [[ "$BIN_BASENAME" =~ ^(eval|nice|timeout|xargs|ssh|su|chroot|docker|sudo|doas|watch|stdbuf|noglob)$ ]]; then
111
120
  exit 0
112
121
  fi
113
122
 
@@ -128,6 +128,14 @@ const HAS_FORCE_RE = new RegExp(`(^|${SP})--force|(^|${SP})-[a-zA-Z]*f`, 'm');
128
128
 
129
129
  const basename = (token) => token.split('/').pop();
130
130
 
131
+ // BUG-C90-04 (FR-001): a shell word is formed AFTER quote removal and
132
+ // backslash escapes, so `r''m`, `r""m`, `r'm'`, `\rm` are all the single word
133
+ // `rm`. Whitespace-splitting the raw text leaves those splices inside token[0]
134
+ // and the literal 'rm' comparison below would silently pass them. Minimal
135
+ // normalization only (strip ' " \, then basename) — this is not a shell
136
+ // parser; every other token keeps its raw bytes.
137
+ const normalizeExecWord = (token) => basename(token.replace(/['"\\]/g, ''));
138
+
131
139
  // ukit_canonical_path parity: realpath -m semantics — resolve the deepest
132
140
  // existing ancestor, append the lexical remainder; unresolvable -> '' (the
133
141
  // caller fails closed, never silently allows). UKIT_REALPATH_MISSING=1 (set by
@@ -196,7 +204,7 @@ async function evaluate({ command, projectRoot, env, deadlineMs, signal }) {
196
204
  // its output cannot reach a shell — `echo "rm -rf /" | bash` executes the
197
205
  // quoted text, so it must never be scrubbed.
198
206
  const head = (command.trimStart().split(/\s+/)[0] || '');
199
- const cmdHead = basename(head);
207
+ const cmdHead = normalizeExecWord(head);
200
208
  let scan = command;
201
209
  if (SAFE_TEXT_HEADS.has(cmdHead) && !PIPE_TO_SHELL_RE.test(command)) {
202
210
  scan = command.replace(/'[^']*'/g, "'Q'").replace(/"[^"]*"/g, '"Q"');
@@ -220,7 +228,7 @@ async function evaluate({ command, projectRoot, env, deadlineMs, signal }) {
220
228
  const segment = rawSegment;
221
229
  if (!segment) continue;
222
230
  const tokens = segment.trimStart().split(/\s+/).filter(Boolean);
223
- if (basename(tokens[0] || '') !== 'rm') continue;
231
+ if (normalizeExecWord(tokens[0] || '') !== 'rm') continue;
224
232
 
225
233
  // flags: leading -x tokens until `--` or the first non-flag word.
226
234
  let hasR = false;
@@ -161,6 +161,42 @@ function isPlaceholderValue(value) {
161
161
  if (!raw || /^[<(]/.test(raw)) return true;
162
162
  return ['-','--','tbd','todo','n/a','na','none','unknown','pending'].includes(raw.toLowerCase());
163
163
  }
164
+ // W2-D07 fix: `OVERRIDE_APPROVED_BY: human` was plain text inside gated files —
165
+ // self-authorizable by the gated model. The in-file `## Model-Tier Guard Override`
166
+ // declaration stays (it is the audit trail), but it only counts when a marker file
167
+ // `.ukit/storage/tier-override` — created by the human/operator OUTSIDE the gated
168
+ // paths — exists and scopes this target:
169
+ // `all` → every target in this cycle
170
+ // `plan` → plan-level checks (PLANNER_MODEL gate)
171
+ // `TASK-<id>` → that task only (e.g. TASK-C91-001, TASK-042)
172
+ // One scope token per line; `#` comments allowed. Empty/scopeless marker = no
173
+ // override (fail closed). UKIT_TIER_OVERRIDE and other env vars are NOT anchors —
174
+ // tool-run env is agent-settable; the marker is a distinct filesystem action.
175
+ const overrideScopeSet = async () => {
176
+ const text = await readTextSafe(path.join(projectRoot, '.ukit/storage/tier-override'));
177
+ const scopes = new Set();
178
+ for (const line of text.split(/\r?\n/)) {
179
+ const t = line.trim();
180
+ if (!t || t.startsWith('#')) continue;
181
+ for (const tok of t.split(/\s+/)) {
182
+ const m = tok.toUpperCase().match(/^(ALL|PLAN|TASK-(?:C\d+-)?\d+)$/);
183
+ if (m) scopes.add(m[1]);
184
+ }
185
+ }
186
+ return scopes;
187
+ };
188
+ const declaresOverride = (text) =>
189
+ /## Model-Tier Guard Override/.test(text) && /OVERRIDE_APPROVED_BY:\s*human/i.test(text);
190
+ const planHasHumanOverride = async (text) => {
191
+ if (!declaresOverride(text)) return false;
192
+ const s = await overrideScopeSet();
193
+ return s.has('ALL') || s.has('PLAN');
194
+ };
195
+ const taskHasHumanOverride = async (text, taskId) => {
196
+ if (!declaresOverride(text)) return false;
197
+ const s = await overrideScopeSet();
198
+ return s.has('ALL') || s.has(String(taskId).toUpperCase());
199
+ };
164
200
 
165
201
  if (toolName === 'Write' || toolName === 'Edit') {
166
202
  const filePath = String(input.file_path || '');
@@ -168,7 +204,7 @@ if (toolName === 'Write' || toolName === 'Edit') {
168
204
 
169
205
  const relPath = path.relative(projectRoot, filePath).replace(/\\/g, '/');
170
206
  const isPlan = relPath === 'docs/AI_HANDOFF/PLAN.md';
171
- const taskMatch = relPath.match(/^docs\/AI_HANDOFF\/tasks\/(TASK-\d+)\.md$/);
207
+ const taskMatch = relPath.match(/^docs\/AI_HANDOFF\/tasks\/(TASK-(?:C\d+-)?\d+)\.md$/);
172
208
  if (!isPlan && !taskMatch) process.exit(0);
173
209
 
174
210
  const fileExists = await pathExists(filePath);
@@ -189,15 +225,9 @@ if (toolName === 'Write' || toolName === 'Edit') {
189
225
  const newVisible = stripHtmlComments(newContent);
190
226
  const currentVisible = stripHtmlComments(currentContent);
191
227
 
192
- // A documented, human-approved override closes the tier contract for this
193
- // cycle — e.g. a gateway with no smart-tier model where the human approved
194
- // planning on the available tier. Same semantics as the push-time check.
195
- const planHasHumanOverride = (text) =>
196
- /## Model-Tier Guard Override/.test(text) && /OVERRIDE_APPROVED_BY:\s*human/i.test(text);
197
-
198
228
  if (isPlan && /## Planner Report/.test(newVisible)) {
199
229
  const plannerModel = extractField(newVisible, 'PLANNER_MODEL');
200
- if (!planHasHumanOverride(newVisible)) {
230
+ if (!(await planHasHumanOverride(newVisible))) {
201
231
  if (!plannerModel || /^unknown$/i.test(plannerModel)) {
202
232
  block('PLANNER_MODEL missing/unknown in PLAN.md. Planning must run via Agent tool subagent_type: "handoff-planner" (opus/unic-smart) and self-report its model.');
203
233
  }
@@ -221,7 +251,7 @@ if (toolName === 'Write' || toolName === 'Edit') {
221
251
  const planContent = (await pathExists(planPath)) ? await readTextSafe(planPath) : '';
222
252
  const planVisible = stripHtmlComments(planContent);
223
253
  const plannerModel = extractField(planVisible, 'PLANNER_MODEL');
224
- if (!planHasHumanOverride(planVisible) && (!plannerModel || tierOf(plannerModel) !== 'smart')) {
254
+ if (!(await planHasHumanOverride(planVisible)) && (!plannerModel || tierOf(plannerModel) !== 'smart')) {
225
255
  block(`Cannot create ${taskId}.md — PLAN.md has no valid smart-tier PLANNER_MODEL yet. Run planning via Agent tool subagent_type: "handoff-planner" (opus/unic-smart) first.`);
226
256
  }
227
257
 
@@ -297,7 +327,8 @@ if (toolName === 'Write' || toolName === 'Edit') {
297
327
  const executorModel = extractField(currentVisible, 'EXECUTOR_MODEL') || extractField(newVisible, 'EXECUTOR_MODEL');
298
328
  const reviewerModel = extractField(newVisible, 'REVIEWER_MODEL');
299
329
  const planContentForReview = (await pathExists(planPath)) ? await readTextSafe(planPath) : '';
300
- const reviewOverride = planHasHumanOverride(stripHtmlComments(planContentForReview));
330
+ const reviewOverride = await taskHasHumanOverride(newVisible, taskId)
331
+ || await planHasHumanOverride(stripHtmlComments(planContentForReview));
301
332
  if (isPlaceholderValue(reviewerModel) || /^unknown$/i.test(reviewerModel)) {
302
333
  if (!isFreshTaskFile) {
303
334
  block(`${taskId}: REVIEWER_MODEL missing/placeholder. Review must run via Agent tool subagent_type: "code-reviewer" (opus/unic-smart) and self-report its model.`);
@@ -345,7 +376,7 @@ if (toolName === 'Bash') {
345
376
  const indexPath = path.join(projectRoot, 'docs/AI_HANDOFF/INDEX.md');
346
377
  if (!(await pathExists(activePath)) || !(await pathExists(indexPath))) process.exit(0); // no handoff cycle here
347
378
 
348
- const taskIds = [...(await fsp.readFile(indexPath, 'utf8')).matchAll(/TASK-\d+/g)]
379
+ const taskIds = [...(await fsp.readFile(indexPath, 'utf8')).matchAll(/TASK-(?:C\d+-)?\d+/g)]
349
380
  .map((m) => m[0])
350
381
  .filter((v, i, a) => a.indexOf(v) === i);
351
382
 
@@ -356,11 +387,10 @@ if (toolName === 'Bash') {
356
387
  const text = stripHtmlComments(await fsp.readFile(taskPath, 'utf8'));
357
388
  if (!text.includes('## Reviewer Verdict')) continue; // not yet reviewed, not this push's concern
358
389
 
359
- // A documented, human-approved exception (e.g. an opus-tier executor escalation left no
360
- // second opus identity free to review it) is a closed, one-time decision — it must not
361
- // re-block every future unrelated push forever just because the task file still contains
362
- // the exempted model string.
363
- if (/## Model-Tier Guard Override/.test(text) && /OVERRIDE_APPROVED_BY:\s*human/i.test(text)) continue;
390
+ // A declared + externally-evidenced human override (see planHasHumanOverride above)
391
+ // is a closed decision — it must not re-block every future unrelated push forever
392
+ // just because the task file still contains the exempted model string.
393
+ if (await taskHasHumanOverride(text, taskId)) continue;
364
394
 
365
395
  const executorModel = extractField(text, 'EXECUTOR_MODEL');
366
396
  const reviewerModel = extractField(text, 'REVIEWER_MODEL');