@ngockhoale/ukit 3.3.1 → 3.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,17 @@
2
2
 
3
3
  All notable changes to UKit are documented here.
4
4
 
5
+ ## 3.3.2 - 2026-09-27
6
+
7
+ **Agent VM driver + self-improve code lane.** CLI (list/show/status/shadow-run/evidence), driver producing promotion evidence into , self-improve step (one shadow run per reference plan per pass), and step (AUTO-<id>.md handoff task files mined from recurring failure patterns — whitelist fix classes, denylist protected lanes, ≤3/pass, never commits code). See docs/plans/AGENT_VM_DRIVER_PLAN.md + SELF_IMPROVE_CODE_LANE.md.
8
+
9
+
10
+ ## Unreleased
11
+ - **Self-improve `codeProposals` step — the code lane (SELF_IMPROVE_CODE_LANE plan).** After telemetry, each rate-limited pass turns mined, recurring failure patterns into **AUTO handoff task files** (`docs/AI_HANDOFF/tasks/AUTO-<seq>.md`) on the standard `_TEMPLATE.md` skeleton — frozen pattern evidence, `ready`/`S`, executor + reviewer + tests stay the quality gate. Whitelist fix classes only (timeout/hang, missing-fallback, empty-output, stale-doc); denylist targets (`src/decision/`, `src/core/agentRuntime/`, runtimeConfig, auth/security code) are skipped, never proposed; ≤3 AUTO tasks per pass; signatures already carried by `tasks/` or `archive/` files dedupe to `duplicate`. `learning.selfImprove.codeProposals` (default `true`) gates the step; `learning.selfImprove.codeProposalsCommit` is hard-false — writing a task file is the ceiling, self-improve never writes code or commits. `ukit self-improve --proposals` prints the proposal list; `--dry-run` lists would-be AUTO tasks without writing.
12
+ - **`ukit vm` driver surface + shadow-run + evidence (agent-VM driver D-01..D-03).** New `src/cli/commands/vm.js` exposes `list`/`show <planId>`/`status`/`shadow-run <planId>`/`evidence <key>` over the plan library; all subcommands refuse `stage_off` (exit 2, zero writes) when `decisionRuntime.vm` is `'off'`. New `src/core/agentRuntime/shadowRun.js` `runShadowPlan(planId, {projectRoot, config, plansDir?, engineDir?, evidence?})` replays a pinned plan through a real `createVmEngine` rooted at `.ukit/storage/agent-runtime/shadow/<planId>/` with a stub `startFn` — deterministic synthetic events (RUN/WAIT_EVENT → `operation.completed`, BRANCH → first declared choice) until terminal, never throws, returns `{transitions, nodeStates, wallMs, irHash, planVersion}`. Each run appends one record to `evidence/vm.jsonl` (256 KiB rotation; `qualityDelta`/`wallP95Ratio`/`costRatio` stay `null` so shadow evidence can never self-promote); `ukit vm evidence vm` prints the `promote()` verdict verbatim without writing config.
13
+
14
+ - **Self-improve `vmShadow` step (agent-VM driver D-04).** After diagnostics, each rate-limited `runSelfImprove` pass now runs one shadow-run per `planLibrary` reference plan via `agentRuntime/shadowRun.js` when `decisionRuntime.vm.stage` is `'shadow'` or higher — the first live-path caller that accumulates evidence for `PROMOTION_CRITERIA`. `off` → step `skipped` (`stage_off`), zero writes under `.ukit/storage/agent-runtime/`; driver missing, refusing, or throwing → step status `error`, the pass never throws and later steps still run.
15
+
5
16
  ## 3.3.1 - 2026-09-27
6
17
 
7
18
  - **Republish of 3.3.0, content-identical.** npm staged 3.3.0 then returned E409 "previously staged version" on retry — same staged-limbo pattern as 3.0.11→3.0.12. Version bump only; parity verified below.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ngockhoale/ukit",
3
- "version": "3.3.1",
3
+ "version": "3.3.2",
4
4
  "description": "Install/update an index-first AI workspace for Claude Code, OpenAI Codex and omp (Oh My Pi).",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -6,17 +6,19 @@
6
6
  import { runSelfImprove } from '../../learning/selfImprove.js';
7
7
 
8
8
  const HELP_FLAGS = new Set(['--help', '-h', 'help']);
9
- const KNOWN_FLAGS = new Set(['--if-due', '--json', '--quiet']);
9
+ const KNOWN_FLAGS = new Set(['--if-due', '--json', '--quiet', '--proposals', '--dry-run']);
10
10
 
11
11
  function printUsage() {
12
- console.log('Usage: ukit self-improve [--if-due] [--json] [--quiet]');
12
+ console.log('Usage: ukit self-improve [--if-due] [--proposals] [--dry-run] [--json] [--quiet]');
13
13
  console.log('');
14
14
  console.log('Runs one self-improve pass: episode backfill, diagnostics, auto-tuning,');
15
- console.log('pattern proposals, telemetry collect. Hooks run it automatically.');
15
+ console.log('pattern proposals, telemetry collect, code proposals. Hooks run it automatically.');
16
16
  console.log('');
17
- console.log(' --if-due Skip when the last pass ran < 10 minutes ago (hook mode)');
18
- console.log(' --json JSON output');
19
- console.log(' --quiet No output (hook mode)');
17
+ console.log(' --if-due Skip when the last pass ran < 10 minutes ago (hook mode)');
18
+ console.log(' --proposals Print the AUTO handoff-task proposals from the codeProposals step');
19
+ console.log(' --dry-run List would-be AUTO tasks without writing files');
20
+ console.log(' --json JSON output');
21
+ console.log(' --quiet No output (hook mode)');
20
22
  }
21
23
 
22
24
  function describeStep(stepResult) {
@@ -40,7 +42,11 @@ export async function runSelfImproveCommand({ projectRoot, argv = [] }) {
40
42
  process.exitCode = 1;
41
43
  return;
42
44
  }
43
- const result = await runSelfImprove(projectRoot, { force: !argv.includes('--if-due') });
45
+ const dryRun = argv.includes('--dry-run');
46
+ const result = await runSelfImprove(projectRoot, {
47
+ force: !argv.includes('--if-due'),
48
+ codeProposalsOpts: { dryRun },
49
+ });
44
50
  if (argv.includes('--quiet')) return;
45
51
  if (argv.includes('--json')) {
46
52
  console.log(JSON.stringify(result, null, 2));
@@ -52,4 +58,15 @@ export async function runSelfImproveCommand({ projectRoot, argv = [] }) {
52
58
  }
53
59
  console.log('[UKit] self-improve: ran');
54
60
  for (const stepResult of result.steps) console.log(describeStep(stepResult));
61
+ if (argv.includes('--proposals') || dryRun) {
62
+ const codeStep = result.steps.find((s) => s.name === 'codeProposals');
63
+ const entries = Array.isArray(codeStep?.results) ? codeStep.results : [];
64
+ console.log(`[UKit] code proposals${dryRun ? ' (dry-run — no files written)' : ''}:`);
65
+ for (const entry of entries) {
66
+ const target = entry.taskFile ?? '-';
67
+ const detail = entry.reason ? ` (${entry.reason})` : '';
68
+ console.log(` ${entry.status}: ${entry.signature} → ${target}${detail}`);
69
+ }
70
+ if (entries.length === 0) console.log(' (no mined patterns)');
71
+ }
55
72
  }
@@ -0,0 +1,242 @@
1
+ // `ukit vm` — agent-VM live-path driver surface (AGENT_VM_DRIVER_PLAN D-01,
2
+ // D-02, D-03). Discovery + shadow-evidence commands over the
3
+ // language-compiled runtime; nothing here executes real operations
4
+ // (shadow startFn is a stub) and nothing mutates config.
5
+ //
6
+ // list Reference plans (planId, planVersion, irHash)
7
+ // show <planId> One plan's compiled node ledger
8
+ // status Stage + evidence counts + shadow dir
9
+ // shadow-run <planId> Deterministic replay through a real VmEngine;
10
+ // appends one evidence record per run (D-03)
11
+ // evidence <key> Read .ukit/storage/agent-runtime/evidence/<key>.jsonl
12
+ // and print the promote() verdict verbatim (G-D3)
13
+ //
14
+ // Every subcommand is stage-gated on `decisionRuntime.vm`: 'off' → typed
15
+ // refuse `stage_off`, exit 2, zero writes under
16
+ // .ukit/storage/agent-runtime/ (G-D1). Exit codes: 0 success,
17
+ // 2 typed failure, 1 usage error.
18
+
19
+ import fs from 'node:fs/promises';
20
+ import os from 'node:os';
21
+ import path from 'node:path';
22
+
23
+ import {
24
+ inspectRuntimeConfig,
25
+ resolveDecisionRuntimeStage,
26
+ } from '../../core/runtimeConfig.js';
27
+ import { loadPlan, loadPlanLibrary } from '../../core/agentRuntime/planLibrary.js';
28
+ import { promote, promotionCriteriaFor } from '../../core/agentRuntime/promotion.js';
29
+ import {
30
+ runShadowPlan,
31
+ readEvidence,
32
+ } from '../../core/agentRuntime/shadowRun.js';
33
+
34
+ const HELP_FLAGS = new Set(['--help', '-h', 'help']);
35
+ const SUBCOMMANDS = new Set(['list', 'show', 'status', 'shadow-run', 'evidence']);
36
+ const OPERAND_SUBCOMMANDS = new Set(['show', 'shadow-run', 'evidence']);
37
+
38
+ const RUNTIME_REL = path.join('.ukit', 'storage', 'agent-runtime');
39
+
40
+ function printUsage() {
41
+ console.log('Usage: ukit vm <command> [options]');
42
+ console.log('');
43
+ console.log('Commands:');
44
+ console.log(' list List pinned reference plans');
45
+ console.log(' show <planId> Show one plan\'s compiled node ledger');
46
+ console.log(' status Show decisionRuntime.vm stage + evidence summary');
47
+ console.log(' shadow-run <planId> Run one deterministic shadow pass (stub startFn)');
48
+ console.log(' evidence <key> Print the promote() verdict over stored runs');
49
+ console.log('');
50
+ console.log('Options:');
51
+ console.log(' --json JSON output (status, evidence)');
52
+ console.log(' -h, --help Show this help');
53
+ console.log('');
54
+ console.log('All commands require decisionRuntime.vm.stage != "off" (G-D1).');
55
+ }
56
+
57
+ // Same fail-safe posture as telemetry: unreadable config resolves {} → 'off'.
58
+ async function loadConfig(projectRoot) {
59
+ try {
60
+ const inspection = await inspectRuntimeConfig(projectRoot, { homeDir: os.homedir() });
61
+ return inspection && inspection.config && typeof inspection.config === 'object'
62
+ ? inspection.config
63
+ : {};
64
+ } catch {
65
+ return {};
66
+ }
67
+ }
68
+
69
+ function refuseStageOff() {
70
+ console.error('[UKit] vm: refused — decisionRuntime.vm.stage is off (stage_off).');
71
+ console.error('[UKit] Set decisionRuntime.vm.stage to "shadow" (or higher) in .ukit/storage/config.json to enable.');
72
+ process.exitCode = 2;
73
+ }
74
+
75
+ // --- list -----------------------------------------------------------------
76
+
77
+ async function listCommand({ config }) {
78
+ const catalog = loadPlanLibrary({ config });
79
+ const ids = Object.keys(catalog).sort();
80
+ if (ids.length === 0) {
81
+ console.log('vm: no plans in library');
82
+ return;
83
+ }
84
+ for (const id of ids) {
85
+ const entry = catalog[id];
86
+ if (entry.ok) {
87
+ console.log(`${id} planVersion=${entry.planVersion} irHash=${entry.irHash}`);
88
+ if (entry.description) console.log(` ${entry.description}`);
89
+ } else {
90
+ const code = entry.errors?.[0]?.code ?? 'load_error';
91
+ console.log(`${id} ERROR ${code}`);
92
+ }
93
+ }
94
+ }
95
+
96
+ // --- show -----------------------------------------------------------------
97
+
98
+ function showCommand({ config, planId }) {
99
+ const loaded = loadPlan(planId, { config });
100
+ if (!loaded.ok) {
101
+ for (const err of loaded.errors ?? []) {
102
+ console.error(`[UKit] vm show: ${err.code ?? 'load_error'} ${err.planId ?? planId}`);
103
+ }
104
+ process.exitCode = 2;
105
+ return;
106
+ }
107
+ console.log(`plan ${loaded.planId}`);
108
+ console.log(` planVersion ${loaded.planVersion} irHash ${loaded.irHash}`);
109
+ if (loaded.description) console.log(` ${loaded.description}`);
110
+ for (const id of loaded.plan.order) {
111
+ const node = loaded.plan.nodes.get(id);
112
+ const deps = node.deps.length > 0 ? ` deps=[${node.deps.join(',')}]` : '';
113
+ const sel = node.selector?.eventType ? ` selector=${node.selector.eventType}` : '';
114
+ const branches = node.op === 'BRANCH'
115
+ ? ` branches=[${Object.keys(node.branches ?? {}).join(',')}]`
116
+ : '';
117
+ const timeout = node.timeoutMs != null ? ` timeoutMs=${node.timeoutMs}` : '';
118
+ console.log(` ${id}: ${node.op}${deps}${sel}${branches}${timeout}`);
119
+ }
120
+ }
121
+
122
+ // --- status ---------------------------------------------------------------
123
+
124
+ async function statusCommand({ projectRoot, config, json }) {
125
+ const stage = resolveDecisionRuntimeStage(config, 'vm');
126
+ const catalog = loadPlanLibrary({ config });
127
+ const planIds = Object.keys(catalog).sort();
128
+ const brokenPlans = planIds.filter((id) => !catalog[id].ok);
129
+ const evidence = await readEvidence(projectRoot, { key: 'vm' });
130
+ const runs = evidence.ok ? evidence.runs.length : null;
131
+ let shadowDirs = 0;
132
+ try {
133
+ shadowDirs = (await fs.readdir(path.join(projectRoot, RUNTIME_REL, 'shadow'), { withFileTypes: true }))
134
+ .filter((e) => e.isDirectory()).length;
135
+ } catch { /* no shadow dir yet */ }
136
+ if (json) {
137
+ console.log(JSON.stringify({
138
+ stage,
139
+ plans: planIds,
140
+ brokenPlans,
141
+ evidenceRuns: runs,
142
+ shadowDirs,
143
+ }));
144
+ return;
145
+ }
146
+ console.log(`vm stage: ${stage}`);
147
+ console.log(`plans: ${planIds.length} (${planIds.join(', ') || 'none'})`);
148
+ if (brokenPlans.length > 0) console.log(`broken plans: ${brokenPlans.join(', ')}`);
149
+ console.log(`evidence runs: ${runs === null ? 'unknown' : runs}`);
150
+ console.log(`shadow instances: ${shadowDirs} plan dir(s)`);
151
+ }
152
+
153
+ // --- shadow-run -----------------------------------------------------------
154
+
155
+ async function shadowRunCommand({ projectRoot, config, planId }) {
156
+ const res = await runShadowPlan(planId, { projectRoot, config });
157
+ if (!res.ok) {
158
+ console.error(`[UKit] vm shadow-run: ${res.code} — ${res.reason}`);
159
+ process.exitCode = 2;
160
+ return;
161
+ }
162
+ console.log(`shadow-run ${res.planId}: ${res.status} (${res.code})`);
163
+ console.log(` planInstance ${res.planInstanceId}`);
164
+ console.log(` planVersion ${res.planVersion} irHash ${res.irHash}`);
165
+ console.log(` transitions ${res.transitions.length} wallMs ${res.wallMs}`);
166
+ for (const t of res.transitions) {
167
+ console.log(` ${t.node} -> ${t.to}${t.branch !== undefined ? ` (${t.branch})` : ''}`);
168
+ }
169
+ }
170
+
171
+ // --- evidence -------------------------------------------------------------
172
+
173
+ async function evidenceCommand({ projectRoot, config, key, json }) {
174
+ const res = await readEvidence(projectRoot, { key });
175
+ if (!res.ok) {
176
+ console.error(`[UKit] vm evidence: ${res.code} — ${res.reason}`);
177
+ process.exitCode = 2;
178
+ return;
179
+ }
180
+ const verdict = promote(config, { key, runs: res.runs });
181
+ const criteria = promotionCriteriaFor(key);
182
+ if (json) {
183
+ console.log(JSON.stringify({ key, runs: res.runs.length, dropped: res.dropped, criteria, verdict }));
184
+ return;
185
+ }
186
+ console.log(`evidence ${key}: ${res.runs.length} run(s)${res.dropped > 0 ? ` (${res.dropped} malformed dropped)` : ''}`);
187
+ // G-D3: verdict is advisory — printed verbatim, never written back.
188
+ console.log(`promote ${key}: stage=${verdict.stage} promoted=${verdict.promoted} reason=${verdict.reason}`);
189
+ }
190
+
191
+ // --- dispatch -------------------------------------------------------------
192
+
193
+ export async function runVm({ projectRoot, argv = [] }) {
194
+ const args = Array.isArray(argv) ? argv : [];
195
+ const sub = (args[0] ?? '').toLowerCase();
196
+
197
+ if (args.length === 0 || HELP_FLAGS.has(sub)) {
198
+ printUsage();
199
+ return;
200
+ }
201
+ if (!SUBCOMMANDS.has(sub)) {
202
+ console.error(`[UKit] Unknown vm subcommand: ${sub}`);
203
+ printUsage();
204
+ process.exitCode = 1;
205
+ return;
206
+ }
207
+
208
+ const rest = args.slice(1);
209
+ if (rest.some((a) => HELP_FLAGS.has(a))) {
210
+ printUsage();
211
+ return;
212
+ }
213
+ const positional = rest.filter((a) => !a.startsWith('-'));
214
+ const flags = rest.filter((a) => a.startsWith('-'));
215
+ const allowed = (sub === 'status' || sub === 'evidence') ? new Set(['--json']) : new Set();
216
+ const unknownFlags = flags.filter((f) => !allowed.has(f));
217
+ if (unknownFlags.length > 0) {
218
+ console.error(`[UKit] Unknown vm flag(s): ${unknownFlags.join(', ')}`);
219
+ printUsage();
220
+ process.exitCode = 1;
221
+ return;
222
+ }
223
+ if (positional.length > 1 || (OPERAND_SUBCOMMANDS.has(sub) && positional.length === 0)) {
224
+ printUsage();
225
+ process.exitCode = 1;
226
+ return;
227
+ }
228
+
229
+ // G-D1 gate: 'off' refuses every subcommand with zero writes.
230
+ const config = await loadConfig(projectRoot);
231
+ if (resolveDecisionRuntimeStage(config, 'vm') === 'off') {
232
+ refuseStageOff();
233
+ return;
234
+ }
235
+
236
+ const operand = positional[0];
237
+ if (sub === 'list') return listCommand({ config });
238
+ if (sub === 'show') return showCommand({ config, planId: operand });
239
+ if (sub === 'status') return statusCommand({ projectRoot, config, json: flags.includes('--json') });
240
+ if (sub === 'shadow-run') return shadowRunCommand({ projectRoot, config, planId: operand });
241
+ return evidenceCommand({ projectRoot, config, key: operand, json: flags.includes('--json') });
242
+ }
package/src/cli/index.js CHANGED
@@ -13,6 +13,7 @@ import { runFeedback } from './commands/feedback.js';
13
13
  import { runPlaybook } from './commands/playbook.js';
14
14
  import { runDecision } from './commands/decision.js';
15
15
  import { runSelfImproveCommand } from './commands/selfImprove.js';
16
+ import { runVm } from './commands/vm.js';
16
17
  const GLOBAL_FLAGS = new Set(['--help', '-h', '--version', '-v']);
17
18
 
18
19
  export async function runCli({ argv, packageRoot, projectRoot, packageVersion }) {
@@ -105,6 +106,11 @@ export async function runCli({ argv, packageRoot, projectRoot, packageVersion })
105
106
  return;
106
107
  }
107
108
 
109
+ if (command === 'vm') {
110
+ await runVm({ projectRoot, argv: commandArgv });
111
+ return;
112
+ }
113
+
108
114
  if (command === 'build' && (commandArgv[0] ?? '').toLowerCase() === 'index') {
109
115
  await runIndexTools({ projectRoot, argv: ['build', ...commandArgv.slice(1)] });
110
116
  return;
@@ -138,6 +144,7 @@ export async function runCli({ argv, packageRoot, projectRoot, packageVersion })
138
144
  console.log(' telemetry Flight recorder (collect/status/digest/export-support/import/evaluate)');
139
145
  console.log(' feedback Record or list wrong-route feedback labels');
140
146
  console.log(' self-improve Run the automatic learn/tune/collect pass now (hooks run it)');
147
+ console.log(' vm Agent VM driver (list/show/status/shadow-run/evidence)');
141
148
  console.log(' update Upgrade the global UKit CLI to the latest version');
142
149
  console.log(' version Show UKit version');
143
150
  console.log('');
@@ -0,0 +1,384 @@
1
+ /**
2
+ * agentRuntime/shadowRun.js — agent-VM live-path driver (D-02/D-03).
3
+ *
4
+ * The shadow driver makes `decisionRuntime.vm` ≥ 'shadow' actually produce
5
+ * evidence: it loads a pinned reference plan via `planLibrary.loadPlan`,
6
+ * runs it inside a real `createVmEngine` rooted at
7
+ * `.ukit/storage/agent-runtime/shadow/<planId>/`, and feeds each activated
8
+ * node a deterministic synthetic event:
9
+ *
10
+ * RUN / WAIT_EVENT → <node.selector.eventType> (default
11
+ * 'operation.completed') with
12
+ * safePayload { outcome:'completed', ...selector.fields }
13
+ * BRANCH → <node.selector.eventType> (default 'branch.selected')
14
+ * with safePayload { [selector.field]: <first declared
15
+ * choice> }
16
+ * COMPLETE → resolved by the engine itself (no external event)
17
+ *
18
+ * `startFn` is a noop — shadow runs measure VM/decision-plane behaviour
19
+ * (journals, continuations, classify verdicts on unclassifiable events),
20
+ * never real operation side effects. The run is deterministic modulo the
21
+ * decision model: selector-matched events never call classifyFn; an
22
+ * unroutable node degrades to the deterministic escalation lane
23
+ * (recovery_required) exactly as the V-05 contract requires.
24
+ *
25
+ * Never throws: every failure resolves to a typed
26
+ * `{ ok:false, code, reason }` — `stage_off` (decisionRuntime.vm 'off' →
27
+ * zero filesystem writes, G-D1), plan-load error codes verbatim from
28
+ * loadPlan, `malformed_plan`, `unknown_plan_id`, `engine_error`,
29
+ * `unterminated` (bounded-loop guard tripped).
30
+ *
31
+ * Evidence (D-03): each run appends one JSONL record to
32
+ * `.ukit/storage/agent-runtime/evidence/<key>.jsonl`
33
+ * `{planId, irHash, planVersion, planInstanceId, status, transitions,
34
+ * wallMs, qualityDelta, wallP95Ratio, costRatio, forbidden, at}` —
35
+ * qualityDelta/wallP95Ratio/costRatio stay `null` until a measured
36
+ * baseline exists (promotion.js treats missing metrics as non-passing,
37
+ * so shadow evidence can never self-promote). The file rotates to
38
+ * `<key>.jsonl.1` at 256 KiB (single generation, bounded). Callers may
39
+ * pass `evidence:false` to skip the append (tests, dry probing).
40
+ */
41
+
42
+ import { promises as fs } from 'node:fs';
43
+ import path from 'node:path';
44
+
45
+ import { resolveDecisionRuntimeStage } from '../runtimeConfig.js';
46
+ import { isTerminal, CONTRACT_VERSION } from './contract.js';
47
+ import { createVmEngine } from './vmEngine.js';
48
+ import { loadPlan } from './planLibrary.js';
49
+
50
+ const RUNTIME_REL = path.join('.ukit', 'storage', 'agent-runtime');
51
+ const EVIDENCE_REL = path.join('evidence');
52
+ const EVIDENCE_MAX_BYTES = 256 * 1024;
53
+
54
+ /** Bounded driver loop: every step delivers ≥1 event, so a correct engine
55
+ * always terminates well inside this bound — tripping it means a driver or
56
+ * engine fault, surfaced as `unterminated` rather than an infinite await. */
57
+ const maxSteps = (nodeCount) => nodeCount * 8 + 16;
58
+
59
+ const KEY_RE = /^[a-z0-9][a-z0-9-]*$/;
60
+
61
+ function refuse(code, reason) {
62
+ return Object.freeze({ ok: false, code, reason });
63
+ }
64
+
65
+ function runtimeDir(projectRoot) {
66
+ return path.join(projectRoot, RUNTIME_REL);
67
+ }
68
+
69
+ export function evidenceFilePath(projectRoot, key = 'vm') {
70
+ return path.join(runtimeDir(projectRoot), EVIDENCE_REL, `${key}.jsonl`);
71
+ }
72
+
73
+ /**
74
+ * Append one evidence record (single JSON line) to the bounded store.
75
+ * Rotation: one generation (`<key>.jsonl.1`), renamed aside before append
76
+ * when the file would exceed EVIDENCE_MAX_BYTES. Never throws.
77
+ *
78
+ * @param {string} projectRoot
79
+ * @param {object} record JSON-serializable run record
80
+ * @param {{key?: string}} [opts]
81
+ * @returns {Promise<{ok:true, bytes:number}|{ok:false, code:string, reason:string}>}
82
+ */
83
+ export async function appendEvidence(projectRoot, record, { key = 'vm' } = {}) {
84
+ if (typeof projectRoot !== 'string' || projectRoot === '') {
85
+ return refuse('invalid_root', 'projectRoot required');
86
+ }
87
+ if (!KEY_RE.test(key)) return refuse('invalid_key', `evidence key ${JSON.stringify(key)}`);
88
+ if (record === null || typeof record !== 'object') {
89
+ return refuse('malformed_record', 'record must be an object');
90
+ }
91
+ let line;
92
+ try {
93
+ line = `${JSON.stringify(record)}\n`;
94
+ } catch {
95
+ return refuse('malformed_record', 'record is not JSON-serializable');
96
+ }
97
+ const file = evidenceFilePath(projectRoot, key);
98
+ try {
99
+ let size = 0;
100
+ try {
101
+ size = (await fs.stat(file)).size;
102
+ } catch (err) {
103
+ if (err?.code !== 'ENOENT') throw err;
104
+ }
105
+ if (size > 0 && size + Buffer.byteLength(line, 'utf8') > EVIDENCE_MAX_BYTES) {
106
+ await fs.rename(file, `${file}.1`).catch((err) => {
107
+ if (err?.code !== 'ENOENT') throw err;
108
+ });
109
+ }
110
+ await fs.mkdir(path.dirname(file), { recursive: true });
111
+ const fh = await fs.open(file, 'a');
112
+ try {
113
+ await fh.writeFile(line, 'utf8');
114
+ await fh.sync();
115
+ } finally {
116
+ await fh.close();
117
+ }
118
+ return { ok: true, bytes: Buffer.byteLength(line, 'utf8') };
119
+ } catch (err) {
120
+ return refuse('evidence_write_error', err?.code ?? String(err));
121
+ }
122
+ }
123
+
124
+ /**
125
+ * Read evidence runs for one family key. Missing file → zero runs; a
126
+ * crash-torn final line is dropped; malformed mid-file lines are dropped
127
+ * too (evidence is advisory — one bad row must not break the read).
128
+ *
129
+ * @param {string} projectRoot
130
+ * @param {{key?: string}} [opts]
131
+ * @returns {Promise<{ok:true, key:string, runs:object[], dropped:number}
132
+ * |{ok:false, code:string, reason:string}>}
133
+ */
134
+ export async function readEvidence(projectRoot, { key = 'vm' } = {}) {
135
+ if (typeof projectRoot !== 'string' || projectRoot === '') {
136
+ return refuse('invalid_root', 'projectRoot required');
137
+ }
138
+ if (!KEY_RE.test(key)) return refuse('invalid_key', `evidence key ${JSON.stringify(key)}`);
139
+ const file = evidenceFilePath(projectRoot, key);
140
+ let raw;
141
+ try {
142
+ raw = await fs.readFile(file, 'utf8');
143
+ } catch (err) {
144
+ if (err?.code === 'ENOENT') return { ok: true, key, runs: [], dropped: 0 };
145
+ return refuse('evidence_read_error', err?.code ?? String(err));
146
+ }
147
+ const runs = [];
148
+ let dropped = 0;
149
+ for (const line of raw.split('\n')) {
150
+ if (line === '') continue;
151
+ try {
152
+ const parsed = JSON.parse(line);
153
+ if (parsed !== null && typeof parsed === 'object' && !Array.isArray(parsed)) {
154
+ runs.push(parsed);
155
+ } else {
156
+ dropped += 1;
157
+ }
158
+ } catch {
159
+ dropped += 1;
160
+ }
161
+ }
162
+ return { ok: true, key, runs, dropped };
163
+ }
164
+
165
+ /** Synthetic SemanticEvent for one running node (deterministic payload). */
166
+ function syntheticEvent(planInstanceId, node, seq, observedAt) {
167
+ // selector.fields is validated as a name-list by the compiler; only a
168
+ // caller-supplied object (test fixtures) is merged into the payload.
169
+ const payload = (node.selector?.fields !== null && typeof node.selector?.fields === 'object'
170
+ && !Array.isArray(node.selector.fields)) ? { ...node.selector.fields } : {};
171
+ let eventType = node.selector?.eventType;
172
+ if (node.op === 'BRANCH') {
173
+ const choice = Object.keys(node.branches ?? {})[0];
174
+ const field = node.selector?.field ?? 'branch';
175
+ payload[field] = choice;
176
+ eventType = eventType ?? 'branch.selected';
177
+ return {
178
+ choice: { field, value: choice, target: node.branches?.[choice] },
179
+ record: {
180
+ eventId: `shadow-${planInstanceId}-${node.id}-${seq}`,
181
+ operationId: `${planInstanceId}:${node.id}`,
182
+ seq,
183
+ eventType,
184
+ observedAt,
185
+ producerVersion: 'ukit-vm-shadow',
186
+ contractVersion: CONTRACT_VERSION,
187
+ privacyClass: 'internal',
188
+ artifactRefs: [],
189
+ safePayload: payload,
190
+ },
191
+ };
192
+ }
193
+ payload.outcome = 'completed';
194
+ eventType = eventType ?? 'operation.completed';
195
+ return {
196
+ choice: null,
197
+ record: {
198
+ eventId: `shadow-${planInstanceId}-${node.id}-${seq}`,
199
+ operationId: `${planInstanceId}:${node.id}`,
200
+ seq,
201
+ eventType,
202
+ observedAt,
203
+ producerVersion: 'ukit-vm-shadow',
204
+ contractVersion: CONTRACT_VERSION,
205
+ privacyClass: 'internal',
206
+ artifactRefs: [],
207
+ safePayload: payload,
208
+ },
209
+ };
210
+ }
211
+
212
+ /**
213
+ * Run one reference plan under a shadow engine and return its transition
214
+ * ledger. Deterministic: RUN/WAIT_EVENT resolve 'completed' on their first
215
+ * delivery; BRANCH takes its first declared choice; the loop stops as soon
216
+ * as the plan reaches a terminal status.
217
+ *
218
+ * @param {string} planId
219
+ * @param {{projectRoot?: string, config?: object, plansDir?: string,
220
+ * engineDir?: string, now?: () => Date, classifyFn?: Function,
221
+ * decisionClient?: object, decider?: object, evidence?: boolean,
222
+ * maxTransitions?: number}} [opts]
223
+ * `plansDir`/`engineDir` exist for tests and non-repo fixtures; the CLI
224
+ * never passes them. `classifyFn`/`decisionClient`/`decider` are
225
+ * forwarded to createVmEngine (test fault injection — the live path
226
+ * leaves them unset so the V-05 stage-gated adapter applies).
227
+ * @returns {Promise<object>} frozen result — see module header.
228
+ */
229
+ export async function runShadowPlan(planId, opts = {}) {
230
+ const { projectRoot, config = null } = opts;
231
+ try {
232
+ if (resolveDecisionRuntimeStage(config, 'vm') === 'off') {
233
+ return refuse('stage_off', 'decisionRuntime.vm.stage is off — zero writes');
234
+ }
235
+
236
+ const loaded = loadPlan(planId, {
237
+ ...(opts.plansDir ? { plansDir: opts.plansDir } : {}),
238
+ config,
239
+ });
240
+ if (!loaded.ok) {
241
+ const first = loaded.errors?.[0] ?? {};
242
+ return refuse(
243
+ typeof first.code === 'string' ? first.code : 'plan_not_found',
244
+ `${planId}: ${JSON.stringify(first)}`,
245
+ );
246
+ }
247
+ const plan = loaded.plan;
248
+ if (!plan || !(plan.nodes instanceof Map) || plan.nodes.size === 0) {
249
+ return refuse('malformed_plan', `${planId}: compiled plan has no nodes`);
250
+ }
251
+
252
+ const dir = typeof opts.engineDir === 'string' && opts.engineDir !== ''
253
+ ? opts.engineDir
254
+ : (typeof projectRoot === 'string' && projectRoot !== ''
255
+ ? path.join(runtimeDir(projectRoot), 'shadow', planId)
256
+ : null);
257
+ if (dir == null) return refuse('invalid_root', 'projectRoot or engineDir required');
258
+
259
+ const started = Date.now();
260
+ const engine = createVmEngine({
261
+ dir,
262
+ ...(typeof opts.now === 'function' ? { now: opts.now } : {}),
263
+ ...(typeof opts.classifyFn === 'function' ? { classifyFn: opts.classifyFn } : {}),
264
+ ...(opts.decisionClient != null ? { decisionClient: opts.decisionClient } : {}),
265
+ ...(opts.decider != null ? { decider: opts.decider } : {}),
266
+ startFn: async () => {}, // stub: shadow runs measure the VM, not real ops
267
+ config,
268
+ });
269
+
270
+ let startRes;
271
+ try {
272
+ startRes = await engine.start(plan);
273
+ } catch (err) {
274
+ return refuse('engine_error', err?.code ?? String(err));
275
+ }
276
+ if (!startRes || startRes.unsupported || typeof startRes.planInstanceId !== 'string') {
277
+ return refuse('malformed_plan', `${planId}: engine refused start (${startRes?.code ?? 'no instance'})`);
278
+ }
279
+ const pi = startRes.planInstanceId;
280
+
281
+ const transitions = [];
282
+ let status = 'running';
283
+ let steps = 0;
284
+ try {
285
+ const bound = maxSteps(plan.nodes.size);
286
+ while (steps < bound) {
287
+ const snapshot = await engine.state(pi);
288
+ status = snapshot.status;
289
+ if (isTerminal(status)) break;
290
+ const nodeId = plan.order.find(
291
+ (id) => snapshot.nodes[id]?.state === 'running'
292
+ && ['RUN', 'WAIT_EVENT', 'BRANCH'].includes(plan.nodes.get(id)?.op),
293
+ );
294
+ if (nodeId == null) {
295
+ // Nothing the driver can feed (e.g. only wrapper/non-terminal
296
+ // leftovers) — report honestly rather than spinning.
297
+ break;
298
+ }
299
+ const node = plan.nodes.get(nodeId);
300
+ const seq = (snapshot.cursor?.[nodeId] ?? 0) + 1;
301
+ const { record, choice } = syntheticEvent(
302
+ pi, node, seq, (typeof opts.now === 'function' ? opts.now() : new Date()).toISOString(),
303
+ );
304
+ const res = await engine.deliver(record);
305
+ steps += 1;
306
+ for (const t of res.transitions ?? []) {
307
+ transitions.push(
308
+ choice != null && t.node === nodeId && t.branch !== undefined
309
+ ? { node: t.node, to: t.to, branch: t.branch }
310
+ : { node: t.node, to: t.to },
311
+ );
312
+ }
313
+ if (!res.consumed) {
314
+ return refuse('event_not_consumed',
315
+ `${planId}/${nodeId}: deliver rejected (${res.code ?? 'unknown'})`);
316
+ }
317
+ }
318
+ } catch (err) {
319
+ return refuse('engine_error', err?.code ?? String(err));
320
+ } finally {
321
+ try { await engine.close(); } catch { /* persist-best-effort; run result stands */ }
322
+ }
323
+
324
+ const final = await engineStateSafe(engine, pi);
325
+ if (final) status = final.status;
326
+ const wallMs = Date.now() - started;
327
+ if (!isTerminal(status)) {
328
+ return refuse('unterminated', `${planId}: no terminal status after ${steps} deliveries`);
329
+ }
330
+
331
+ const nodeStates = {};
332
+ for (const [id, n] of Object.entries(final?.nodes ?? {})) {
333
+ nodeStates[id] = { state: n.state, attempt: n.attempt };
334
+ }
335
+ const result = Object.freeze({
336
+ ok: true,
337
+ code: status,
338
+ planId,
339
+ planInstanceId: pi,
340
+ status,
341
+ transitions: Object.freeze(transitions),
342
+ nodeStates: Object.freeze(nodeStates),
343
+ wallMs,
344
+ irHash: loaded.irHash,
345
+ planVersion: loaded.planVersion,
346
+ });
347
+
348
+ // D-03 evidence append (caller may opt out with evidence:false).
349
+ if (opts.evidence !== false && typeof projectRoot === 'string' && projectRoot !== '') {
350
+ const append = await appendEvidence(projectRoot, {
351
+ planId,
352
+ irHash: loaded.irHash,
353
+ planVersion: loaded.planVersion,
354
+ planInstanceId: pi,
355
+ status,
356
+ transitions: transitions.length,
357
+ wallMs,
358
+ qualityDelta: null,
359
+ wallP95Ratio: null,
360
+ costRatio: null,
361
+ // 'cancelled' is a normal terminal for plans whose untaken BRANCH
362
+ // path cancels dependents — only real failure lanes are forbidden.
363
+ forbidden: status === 'failed',
364
+ at: new Date().toISOString(),
365
+ }, { key: 'vm' });
366
+ if (!append.ok) {
367
+ return refuse('evidence_write_error', `${planId}: ${append.reason}`);
368
+ }
369
+ }
370
+ return result;
371
+ } catch (err) {
372
+ // Outer never-throws guard: an unexpected fault (e.g. engine internals
373
+ // throwing a non-typed error) still resolves to a typed refuse.
374
+ return refuse('engine_error', err?.code ?? String(err?.message ?? err));
375
+ }
376
+ }
377
+
378
+ async function engineStateSafe(engine, pi) {
379
+ try {
380
+ return await engine.state(pi);
381
+ } catch {
382
+ return null;
383
+ }
384
+ }
@@ -229,7 +229,7 @@ function pushStageError(errors, node, label) {
229
229
  export function buildDefaultRuntimeConfig(overrides = {}) {
230
230
  const safeOverrides = isPlainObject(overrides) ? overrides : {};
231
231
 
232
- return mergeObjects({
232
+ const config = mergeObjects({
233
233
  version: PACKAGE_VERSION,
234
234
  agent: 'claude-code',
235
235
  autonomy: {
@@ -451,7 +451,10 @@ export function buildDefaultRuntimeConfig(overrides = {}) {
451
451
  episodes: { autoWrite: true },
452
452
  tuning: { enabled: true, applyMode: 'auto' },
453
453
  // 3.3.0: automatic collect → learn → apply pass (src/learning/selfImprove.js).
454
- selfImprove: { enabled: true },
454
+ // codeProposals: AUTO task-file lane (SELF_IMPROVE_CODE_LANE plan).
455
+ // codeProposalsCommit is a reserved hard-false — writing a task file is
456
+ // the ceiling; self-improve never commits code (clamped below).
457
+ selfImprove: { enabled: true, codeProposals: true, codeProposalsCommit: false },
455
458
  // C52 M04.2 stage keys (SPEC §5 FR-016–FR-018). candidates: repeated
456
459
  // corrections/suppressions/escalations promote to a LearningCandidate
457
460
  // only after minOccurrences + cross-session evidence (status
@@ -575,6 +578,15 @@ export function buildDefaultRuntimeConfig(overrides = {}) {
575
578
  },
576
579
  },
577
580
  }, safeOverrides);
581
+
582
+ // learning.selfImprove.codeProposalsCommit is hard-false forever: writing an
583
+ // AUTO task file is the ceiling of the code lane — applying/committing a
584
+ // patch is always the executor's job under test + review. A user-provided
585
+ // `true` is silently clamped, never honored.
586
+ if (isPlainObject(config.learning?.selfImprove)) {
587
+ config.learning.selfImprove.codeProposalsCommit = false;
588
+ }
589
+ return config;
578
590
  }
579
591
 
580
592
  export function validateRuntimeConfig(config) {
@@ -1114,6 +1126,8 @@ export function validateRuntimeConfig(config) {
1114
1126
  errors.push('learning.selfImprove must be an object.');
1115
1127
  } else {
1116
1128
  pushBooleanError(errors, learning.selfImprove.enabled, 'learning.selfImprove.enabled');
1129
+ pushBooleanError(errors, learning.selfImprove.codeProposals, 'learning.selfImprove.codeProposals');
1130
+ pushBooleanError(errors, learning.selfImprove.codeProposalsCommit, 'learning.selfImprove.codeProposalsCommit');
1117
1131
  }
1118
1132
  }
1119
1133
  // C52 M04.2 stage keys — same optional-present contract as routing.*.
@@ -0,0 +1,349 @@
1
+ // codeProposals.js — self-improve code lane (SELF_IMPROVE_CODE_LANE plan).
2
+ //
3
+ // Turns mined, recurring failure patterns into HANDOFF TASK FILES
4
+ // (`docs/AI_HANDOFF/tasks/AUTO-<seq>.md`) — proposals for the normal
5
+ // executor + reviewer + tests chain, never code writes or commits from
6
+ // self-improve itself. Whitelist-only fix classes; anything outside the
7
+ // deterministic mapping is `skipped`.
8
+ //
9
+ // Contracts:
10
+ // * NEVER THROWS — missing dirs, unreadable files, write failures →
11
+ // partial result, no exception escapes.
12
+ // * Eligible: count >= minCount (default 3) AND sessions >= 2.
13
+ // * Whitelist fix classes only (timeout/hang, missing-fallback,
14
+ // empty-output, stale-doc); unknown shapes → status 'skipped'.
15
+ // * Denylist: targets under src/decision/, src/core/agentRuntime/,
16
+ // runtimeConfig, auth/security/secret/credential/token code are never
17
+ // proposed.
18
+ // * Dedupe: a signature already carried by any `.md` in
19
+ // docs/AI_HANDOFF/tasks/ or docs/AI_HANDOFF/archive/ → 'duplicate'.
20
+ // * Cap: at most maxTasks (default 3) AUTO files per pass.
21
+ // * Writes are confined to docs/AI_HANDOFF/tasks/AUTO-<seq>.md via
22
+ // exclusive-create (`wx`); `dryRun` writes nothing.
23
+
24
+ import fs from 'node:fs/promises';
25
+ import path from 'node:path';
26
+
27
+ const TASKS_DIR_REL = path.join('docs', 'AI_HANDOFF', 'tasks');
28
+ const ARCHIVE_DIR_REL = path.join('docs', 'AI_HANDOFF', 'archive');
29
+ const ARTIFACT_REL = path.join('.ukit', 'storage', 'cache', 'failure-patterns.json');
30
+ const PLAN_REL = 'docs/plans/SELF_IMPROVE_CODE_LANE.md';
31
+ const SIGNATURE_FIELD_RE = /signature["']?\s*:\s*["'`]?([^\n"'`,}]+)/gi;
32
+ const MIN_SESSIONS = 2;
33
+ const AUTO_NAME_RE = /^AUTO-(\d+)\.md$/;
34
+
35
+ // Protected lanes — the same areas keepMainModelFor guards. A mined pattern
36
+ // pointing here is skipped, never proposed.
37
+ const DENIED_TARGET_RES = [
38
+ /^src\/decision\//,
39
+ /^src\/core\/agentRuntime\//,
40
+ /^src\/core\/runtimeConfig\.js$/,
41
+ /^src\/security\//,
42
+ /(^|\/)auth/i,
43
+ /secret|credential|token/i,
44
+ ];
45
+
46
+ // Whitelist fix classes — bounded deterministic mapping only.
47
+ const FIX_CLASSES = [
48
+ {
49
+ id: 'timeout-hang',
50
+ matches: (hay) => /timeout|timed?\s*out|hang|hung/.test(hay),
51
+ change: 'Increase the timeout, add an explicit deadline, and retry once before giving up.',
52
+ },
53
+ {
54
+ id: 'missing-fallback',
55
+ matches: (hay) => /missing-fallback/.test(hay)
56
+ || (/fallback/.test(hay) && /decision|decisionruntime/.test(hay)),
57
+ change: 'Add a `fallbackCode` path plus a deterministic default so the caller never blocks on the decision call.',
58
+ },
59
+ {
60
+ id: 'empty-output',
61
+ matches: (hay) => /empty[-_ ]?output|no output|empty stdout/.test(hay),
62
+ change: 'Keep the chain alive on empty output (`|| true` or equivalent) and log stderr for diagnosis.',
63
+ },
64
+ {
65
+ id: 'stale-doc',
66
+ matches: (hay) => /stale[-_ ]?doc|doc drift|docs?\s+drift|out[- ]?of[- ]?date\s+doc/.test(hay),
67
+ change: 'Re-render or refresh the flagged documentation file to match current source.',
68
+ },
69
+ ];
70
+
71
+ function isObject(value) {
72
+ return Boolean(value) && typeof value === 'object' && !Array.isArray(value);
73
+ }
74
+
75
+ // Evidence haystack: signature + example commands + files, lowercased.
76
+ function haystack(pattern) {
77
+ return [
78
+ pattern.signature,
79
+ ...(Array.isArray(pattern.commands) ? pattern.commands : []),
80
+ ...(Array.isArray(pattern.files) ? pattern.files : []),
81
+ ].join(' ').toLowerCase();
82
+ }
83
+
84
+ /**
85
+ * Map a mined pattern to a whitelisted fix class.
86
+ * @returns {{id: string, change: string}|null}
87
+ */
88
+ export function mapPatternToFixTarget(pattern) {
89
+ if (!isObject(pattern)) return null;
90
+ const hay = haystack(pattern);
91
+ for (const fixClass of FIX_CLASSES) {
92
+ if (fixClass.matches(hay)) return { id: fixClass.id, change: fixClass.change };
93
+ }
94
+ return null;
95
+ }
96
+
97
+ function normalizeTargets(pattern) {
98
+ const files = Array.isArray(pattern?.files)
99
+ ? pattern.files.filter((f) => typeof f === 'string' && f.trim())
100
+ : [];
101
+ return files.map((f) => f.replace(/\\/g, '/'));
102
+ }
103
+
104
+ async function loadPatterns(projectRoot) {
105
+ try {
106
+ const raw = await fs.readFile(path.join(projectRoot, ARTIFACT_REL), 'utf8');
107
+ const parsed = JSON.parse(raw);
108
+ if (parsed && Array.isArray(parsed.patterns)) return parsed.patterns;
109
+ } catch {
110
+ // Artifact absent/corrupt → lazy re-mine below.
111
+ }
112
+ try {
113
+ const mod = await import('../diagnostics/failurePatterns.js');
114
+ if (typeof mod?.mineFailurePatterns !== 'function') return [];
115
+ const mined = await mod.mineFailurePatterns(projectRoot, { minCount: 1 });
116
+ return Array.isArray(mined?.patterns) ? mined.patterns : [];
117
+ } catch {
118
+ return [];
119
+ }
120
+ }
121
+
122
+ // Bounded recursive .md listing; missing dirs → [].
123
+ async function listMarkdownFiles(dir, depth = 0) {
124
+ if (depth > 4) return [];
125
+ let entries;
126
+ try {
127
+ entries = await fs.readdir(dir, { withFileTypes: true });
128
+ } catch {
129
+ return [];
130
+ }
131
+ const files = [];
132
+ for (const entry of entries) {
133
+ const full = path.join(dir, entry.name);
134
+ if (entry.isDirectory()) {
135
+ files.push(...await listMarkdownFiles(full, depth + 1));
136
+ } else if (entry.isFile() && entry.name.endsWith('.md')) {
137
+ files.push(full);
138
+ }
139
+ }
140
+ return files;
141
+ }
142
+
143
+ // Signatures already claimed by a task/archived file + highest AUTO seq.
144
+ async function scanHandoffDirs(projectRoot) {
145
+ const signatures = new Set();
146
+ let maxSeq = 0;
147
+ for (const rel of [TASKS_DIR_REL, ARCHIVE_DIR_REL]) {
148
+ const dir = path.join(projectRoot, rel);
149
+ for (const file of await listMarkdownFiles(dir)) {
150
+ const seqMatch = AUTO_NAME_RE.exec(path.basename(file));
151
+ if (seqMatch) maxSeq = Math.max(maxSeq, Number(seqMatch[1]));
152
+ let text;
153
+ try {
154
+ text = await fs.readFile(file, 'utf8');
155
+ } catch {
156
+ continue;
157
+ }
158
+ SIGNATURE_FIELD_RE.lastIndex = 0;
159
+ for (const match of text.matchAll(SIGNATURE_FIELD_RE)) {
160
+ const sig = match[1].trim();
161
+ if (sig) signatures.add(sig);
162
+ }
163
+ }
164
+ }
165
+ return { signatures, maxSeq };
166
+ }
167
+
168
+ function renderTaskFile({ seq, pattern, fixClass, targets, now }) {
169
+ const id = `AUTO-${String(seq).padStart(3, '0')}`;
170
+ const signature = pattern.signature;
171
+ const files = targets.length > 0 ? targets : ['(determine from evidence block)'];
172
+ const evidence = JSON.stringify(pattern, null, 2);
173
+ return `# ${id} — self-improve: ${fixClass.id} fix for "${signature}"
174
+
175
+ - Status: \`ready\`
176
+ - Size: \`S\`
177
+ - Owner: \`-\`
178
+ - Reviewer: \`-\`
179
+ - Parent plan: \`${PLAN_REL}\`
180
+ - Spec references: \`docs/AI_HANDOFF/SPEC.md\` §self-improve
181
+ - Supersedes: \`(none)\`
182
+ - Superseded by: \`(none)\`
183
+
184
+ ## Goal
185
+
186
+ Recurring verification failure pattern \`"${signature}"\` — signature: \`${signature}\` —
187
+ seen ${pattern.count}× across ${pattern.sessions} sessions. Apply the whitelisted
188
+ **${fixClass.id}** fix: ${fixClass.change}
189
+
190
+ ## Target Files
191
+
192
+ ${files.map((f) => `- \`${f}\` — apply the ${fixClass.id} fix`).join('\n')}
193
+
194
+ ## Evidence (frozen — do not re-derive)
195
+
196
+ \`\`\`json
197
+ ${evidence}
198
+ \`\`\`
199
+
200
+ ## Progress
201
+
202
+ ## Test Cases (REQUIRED — TDD)
203
+
204
+ | # | Loại | Tên test | Expected | Pre-state / Fixture |
205
+ |---|------|----------|----------|---------------------|
206
+ | 1 | unit | pattern signature "${signature}" repro | failing path exercised | fixture from Evidence block |
207
+ | 2 | regression | fix holds | GREEN after patch | same fixture |
208
+
209
+ ## Test Files
210
+
211
+ - (executor picks the closest existing test file for the target)
212
+
213
+ ## Verification Commands
214
+
215
+ \`\`\`bash
216
+ yarn test <targeted test file>
217
+ \`\`\`
218
+
219
+ ## Acceptance Criteria
220
+
221
+ - [ ] Targeted test green (RED → GREEN evidence in Executor Report).
222
+ - [ ] No regression in the related suite.
223
+ - [ ] Reviewer verdict APPROVED or APPROVED-WITH-MINOR.
224
+
225
+ ## Dependencies
226
+
227
+ - (none)
228
+
229
+ ## Interfaces
230
+
231
+ - Consumes: (none)
232
+ - Produces: (none)
233
+
234
+ ---
235
+
236
+ ## Discussion
237
+
238
+ ### ${new Date(now).toISOString()} · planner · ukit self-improve
239
+ Auto-generated by \`ukit self-improve\` at ${new Date(now).toISOString()} from mined failure-patterns artifact.
240
+
241
+ <!-- auto-proposal: do-not-approve-without-human-review -->
242
+ `;
243
+ }
244
+
245
+ /**
246
+ * Propose AUTO handoff task files from mined failure patterns.
247
+ *
248
+ * @param {string} projectRoot
249
+ * @param {{ minCount?: number, maxTasks?: number, dryRun?: boolean,
250
+ * now?: number }} [options]
251
+ * @returns {Promise<{generatedAt: string, dryRun: boolean,
252
+ * patternsScanned: number, eligible: number, created: number,
253
+ * results: Array<{signature: string, fixClass: string|null,
254
+ * status: 'proposed'|'duplicate'|'skipped', reason?: string,
255
+ * taskFile: string|null, written: boolean}>, error?: string}>}
256
+ */
257
+ export async function proposeCodeFixTasks(projectRoot, {
258
+ minCount = 3,
259
+ maxTasks = 3,
260
+ dryRun = true,
261
+ now = Date.now(),
262
+ } = {}) {
263
+ const result = {
264
+ generatedAt: new Date(now).toISOString(),
265
+ dryRun,
266
+ patternsScanned: 0,
267
+ eligible: 0,
268
+ created: 0,
269
+ results: [],
270
+ };
271
+
272
+ try {
273
+ const patterns = await loadPatterns(projectRoot);
274
+ result.patternsScanned = patterns.length;
275
+ const known = await scanHandoffDirs(projectRoot);
276
+ let seq = known.maxSeq;
277
+ let emitted = 0;
278
+
279
+ for (const pattern of patterns) {
280
+ const signature = typeof pattern?.signature === 'string' ? pattern.signature : null;
281
+ if (!signature) continue;
282
+ const entry = {
283
+ signature, fixClass: null, status: 'skipped', taskFile: null, written: false,
284
+ };
285
+ result.results.push(entry);
286
+
287
+ const count = Number(pattern?.count) || 0;
288
+ const sessions = Number(pattern?.sessions) || 0;
289
+ if (count < minCount || sessions < MIN_SESSIONS) {
290
+ entry.reason = 'ineligible';
291
+ continue;
292
+ }
293
+ result.eligible += 1;
294
+
295
+ const fixClass = mapPatternToFixTarget(pattern);
296
+ if (!fixClass) {
297
+ entry.reason = 'no-fix-class';
298
+ continue;
299
+ }
300
+ entry.fixClass = fixClass.id;
301
+
302
+ const targets = normalizeTargets(pattern);
303
+ if (targets.some((f) => DENIED_TARGET_RES.some((re) => re.test(f)))) {
304
+ entry.reason = 'denied-target';
305
+ continue;
306
+ }
307
+
308
+ if (known.signatures.has(signature)) {
309
+ entry.status = 'duplicate';
310
+ continue;
311
+ }
312
+
313
+ if (emitted >= maxTasks) {
314
+ entry.reason = 'cap';
315
+ continue;
316
+ }
317
+
318
+ seq += 1;
319
+ const fileName = `AUTO-${String(seq).padStart(3, '0')}.md`;
320
+ entry.taskFile = path.join(TASKS_DIR_REL, fileName);
321
+ entry.status = 'proposed';
322
+ emitted += 1;
323
+ result.created += 1;
324
+ known.signatures.add(signature);
325
+
326
+ if (dryRun) continue;
327
+
328
+ try {
329
+ await fs.mkdir(path.join(projectRoot, TASKS_DIR_REL), { recursive: true });
330
+ await fs.writeFile(
331
+ path.join(projectRoot, TASKS_DIR_REL, fileName),
332
+ renderTaskFile({ seq, pattern, fixClass, targets, now }),
333
+ { flag: 'wx' },
334
+ );
335
+ entry.written = true;
336
+ } catch (err) {
337
+ entry.status = 'skipped';
338
+ entry.reason = `write-failed: ${err?.message ?? String(err)}`;
339
+ entry.written = false;
340
+ result.created -= 1;
341
+ emitted -= 1;
342
+ }
343
+ }
344
+ } catch (err) {
345
+ result.error = err?.message ?? String(err);
346
+ }
347
+
348
+ return result;
349
+ }
@@ -9,14 +9,22 @@
9
9
  // Steps (each isolated: one failing step never skips the others):
10
10
  // 1. episodes — backfill an episode record for every idle exec-ledger
11
11
  // 2. diagnostics — failure patterns, feedback events, skill accuracy
12
- // 3. tuning — compute suggestions; applyMode 'auto' applies them as
12
+ // 3. vmShadow — when decisionRuntime.vm stage ≥ 'shadow', one shadow-run
13
+ // per reference plan via the agent-VM driver (D-04); off →
14
+ // skipped, any failure → status 'error', never throws
15
+ // 4. tuning — compute suggestions; applyMode 'auto' applies them as
13
16
  // clamped one-step changes in learning/tuned.json, with a
14
17
  // per-key cooldown so one noisy window cannot ratchet a
15
18
  // value across its whole range
16
- // 4. proposals — repeated failure patterns become PENDING memory
19
+ // 5. proposals — repeated failure patterns become PENDING memory
17
20
  // candidates (rules still need `ukit memory approve`)
18
- // 5. telemetry — ingest stored hook/ledger telemetry, flush, refresh the
21
+ // 6. telemetry — ingest stored hook/ledger telemetry, flush, refresh the
19
22
  // support view
23
+ // 7. codeProposals — eligible mined patterns become AUTO-<seq> handoff task
24
+ // files (docs/AI_HANDOFF/tasks/); whitelist fix classes
25
+ // only, ≤3/pass, signature-deduped, denylisted targets
26
+ // skipped. Proposals are task FILES — self-improve never
27
+ // writes code or commits (SELF_IMPROVE_CODE_LANE plan)
20
28
  //
21
29
  // Guards: `.ukit/storage/learning/self-improve.json` stamp rate-limits the
22
30
  // pass (minIntervalMs, default 10 min) and a file lock keeps two triggers from
@@ -26,7 +34,7 @@ import fs from 'node:fs/promises';
26
34
  import os from 'node:os';
27
35
  import path from 'node:path';
28
36
 
29
- import { inspectRuntimeConfig } from '../core/runtimeConfig.js';
37
+ import { inspectRuntimeConfig, resolveDecisionRuntimeStage } from '../core/runtimeConfig.js';
30
38
  import { withFileLock } from '../core/fileOps.js';
31
39
  import { detectProjectContext } from '../context/detectProjectContext.js';
32
40
  import { backfillEpisodes } from '../core/memory/episodes.js';
@@ -35,6 +43,7 @@ import { collectFeedbackEvents } from '../diagnostics/feedbackEvents.js';
35
43
  import { collectSkillAccuracy } from '../diagnostics/skillAccuracy.js';
36
44
  import { computeTuningSuggestions } from './tuning.js';
37
45
  import { proposeFromPatterns } from './patternProposals.js';
46
+ import { proposeCodeFixTasks } from './codeProposals.js';
38
47
  import {
39
48
  TUNABLE_KEYS,
40
49
  clampTunable,
@@ -133,7 +142,12 @@ async function collectTelemetry(projectRoot, config) {
133
142
  /**
134
143
  * Run one self-improve pass.
135
144
  * @param {string} projectRoot
136
- * @param {{ force?: boolean, homeDir?: string, now?: number, minIntervalMs?: number }} [opts]
145
+ * @param {{ force?: boolean, homeDir?: string, now?: number, minIntervalMs?: number,
146
+ * runShadowPlan?: Function, codeProposalsOpts?: { dryRun?: boolean } }} [opts] —
147
+ * runShadowPlan overrides the agentRuntime shadow-run driver (tests); absent →
148
+ * dynamic import of agentRuntime/shadowRun.js (import failure → vmShadow step
149
+ * 'error'). codeProposalsOpts.dryRun:false lets the codeProposals step write
150
+ * AUTO task files; anything else keeps it dry-run.
137
151
  * @returns {Promise<{status: 'ran'|'skipped', reason?: string, steps?: object[]}>}
138
152
  */
139
153
  export async function runSelfImprove(projectRoot, {
@@ -141,6 +155,8 @@ export async function runSelfImprove(projectRoot, {
141
155
  homeDir = os.homedir(),
142
156
  now = Date.now(),
143
157
  minIntervalMs = DEFAULT_MIN_INTERVAL_MS,
158
+ runShadowPlan,
159
+ codeProposalsOpts,
144
160
  } = {}) {
145
161
  const root = path.resolve(projectRoot);
146
162
  const stampPath = path.join(root, STAMP_REL);
@@ -172,6 +188,43 @@ export async function runSelfImprove(projectRoot, {
172
188
  skills: Object.keys(skills?.skills ?? {}).length,
173
189
  };
174
190
  }));
191
+ steps.push(await step('vmShadow', async () => {
192
+ if (resolveDecisionRuntimeStage(config, 'vm') === 'off') {
193
+ return { status: 'skipped', reason: 'stage_off' };
194
+ }
195
+ try {
196
+ const runner = runShadowPlan
197
+ ?? (await import('../core/agentRuntime/shadowRun.js')).runShadowPlan;
198
+ const { listPlanIds } = await import('../core/agentRuntime/planLibrary.js');
199
+ const runs = [];
200
+ for (const planId of listPlanIds()) {
201
+ let res;
202
+ try {
203
+ res = await runner(planId, { projectRoot: root, config });
204
+ } catch (error) {
205
+ res = { ok: false, code: 'driver_threw', reason: error?.message ?? String(error) };
206
+ }
207
+ runs.push({
208
+ planId,
209
+ ok: res?.ok === true,
210
+ code: res?.code ?? null,
211
+ transitions: Array.isArray(res?.transitions) ? res.transitions.length : 0,
212
+ wallMs: res?.wallMs ?? null,
213
+ irHash: res?.irHash ?? null,
214
+ });
215
+ }
216
+ const failed = runs.filter((r) => !r.ok).length;
217
+ return {
218
+ status: failed === 0 ? 'ok' : 'error',
219
+ ran: runs.length - failed,
220
+ failed,
221
+ runs,
222
+ };
223
+ } catch (error) {
224
+ // Driver not built yet or plansDir unreadable: report, never throw.
225
+ return { status: 'error', reason: error?.message ?? String(error) };
226
+ }
227
+ }));
175
228
  steps.push(await step('tuning', async () => {
176
229
  const result = await computeTuningSuggestions(root);
177
230
  const mode = resolveApplyMode(rawConfig ?? {});
@@ -188,7 +241,20 @@ export async function runSelfImprove(projectRoot, {
188
241
  reason: res.error,
189
242
  };
190
243
  }));
244
+
191
245
  steps.push(await step('telemetry', () => collectTelemetry(root, config)));
246
+ steps.push(await step('codeProposals', async () => {
247
+ if (config?.learning?.selfImprove?.codeProposals === false) {
248
+ return { status: 'skipped', reason: 'disabled' };
249
+ }
250
+ const res = await proposeCodeFixTasks(root, { dryRun: codeProposalsOpts?.dryRun !== false });
251
+ return {
252
+ status: res.error ? 'failed' : 'ok',
253
+ created: res.created,
254
+ results: res.results,
255
+ reason: res.error,
256
+ };
257
+ }));
192
258
 
193
259
  await writeJsonAtomic(stampPath, {
194
260
  lastRunAt: new Date(now).toISOString(),