@ngockhoale/ukit 3.3.0 → 3.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/package.json +1 -1
- package/src/cli/commands/selfImprove.js +24 -7
- package/src/cli/commands/vm.js +242 -0
- package/src/cli/index.js +7 -0
- package/src/core/agentRuntime/shadowRun.js +384 -0
- package/src/core/runtimeConfig.js +16 -2
- package/src/learning/codeProposals.js +349 -0
- package/src/learning/selfImprove.js +71 -5
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,21 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to UKit are documented here.
|
|
4
4
|
|
|
5
|
+
## 3.3.2 - 2026-09-27
|
|
6
|
+
|
|
7
|
+
**Agent VM driver + self-improve code lane.** CLI (list/show/status/shadow-run/evidence), driver producing promotion evidence into , self-improve step (one shadow run per reference plan per pass), and step (AUTO-<id>.md handoff task files mined from recurring failure patterns — whitelist fix classes, denylist protected lanes, ≤3/pass, never commits code). See docs/plans/AGENT_VM_DRIVER_PLAN.md + SELF_IMPROVE_CODE_LANE.md.
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
## Unreleased
|
|
11
|
+
- **Self-improve `codeProposals` step — the code lane (SELF_IMPROVE_CODE_LANE plan).** After telemetry, each rate-limited pass turns mined, recurring failure patterns into **AUTO handoff task files** (`docs/AI_HANDOFF/tasks/AUTO-<seq>.md`) on the standard `_TEMPLATE.md` skeleton — frozen pattern evidence, `ready`/`S`, executor + reviewer + tests stay the quality gate. Whitelist fix classes only (timeout/hang, missing-fallback, empty-output, stale-doc); denylist targets (`src/decision/`, `src/core/agentRuntime/`, runtimeConfig, auth/security code) are skipped, never proposed; ≤3 AUTO tasks per pass; signatures already carried by `tasks/` or `archive/` files dedupe to `duplicate`. `learning.selfImprove.codeProposals` (default `true`) gates the step; `learning.selfImprove.codeProposalsCommit` is hard-false — writing a task file is the ceiling, self-improve never writes code or commits. `ukit self-improve --proposals` prints the proposal list; `--dry-run` lists would-be AUTO tasks without writing.
|
|
12
|
+
- **`ukit vm` driver surface + shadow-run + evidence (agent-VM driver D-01..D-03).** New `src/cli/commands/vm.js` exposes `list`/`show <planId>`/`status`/`shadow-run <planId>`/`evidence <key>` over the plan library; all subcommands refuse `stage_off` (exit 2, zero writes) when `decisionRuntime.vm` is `'off'`. New `src/core/agentRuntime/shadowRun.js` `runShadowPlan(planId, {projectRoot, config, plansDir?, engineDir?, evidence?})` replays a pinned plan through a real `createVmEngine` rooted at `.ukit/storage/agent-runtime/shadow/<planId>/` with a stub `startFn` — deterministic synthetic events (RUN/WAIT_EVENT → `operation.completed`, BRANCH → first declared choice) until terminal, never throws, returns `{transitions, nodeStates, wallMs, irHash, planVersion}`. Each run appends one record to `evidence/vm.jsonl` (256 KiB rotation; `qualityDelta`/`wallP95Ratio`/`costRatio` stay `null` so shadow evidence can never self-promote); `ukit vm evidence vm` prints the `promote()` verdict verbatim without writing config.
|
|
13
|
+
|
|
14
|
+
- **Self-improve `vmShadow` step (agent-VM driver D-04).** After diagnostics, each rate-limited `runSelfImprove` pass now runs one shadow-run per `planLibrary` reference plan via `agentRuntime/shadowRun.js` when `decisionRuntime.vm.stage` is `'shadow'` or higher — the first live-path caller that accumulates evidence for `PROMOTION_CRITERIA`. `off` → step `skipped` (`stage_off`), zero writes under `.ukit/storage/agent-runtime/`; driver missing, refusing, or throwing → step status `error`, the pass never throws and later steps still run.
|
|
15
|
+
|
|
16
|
+
## 3.3.1 - 2026-09-27
|
|
17
|
+
|
|
18
|
+
- **Republish of 3.3.0, content-identical.** npm staged 3.3.0 then returned E409 "previously staged version" on retry — same staged-limbo pattern as 3.0.11→3.0.12. Version bump only; parity verified below.
|
|
19
|
+
|
|
5
20
|
## 3.3.0 - 2026-09-27
|
|
6
21
|
|
|
7
22
|
**Zero-config self-improvement: every feature stage ships on, UKit collects and learns from its own data.** Fixes the gap where 2.6.8–3.2.0 built the machinery (flight recorder, memory v2, learning, decision plane) but nothing on the live hook path fed it — measured 1 telemetry record and 0 memory-v2 records after days of use.
|
package/package.json
CHANGED
|
@@ -6,17 +6,19 @@
|
|
|
6
6
|
import { runSelfImprove } from '../../learning/selfImprove.js';
|
|
7
7
|
|
|
8
8
|
const HELP_FLAGS = new Set(['--help', '-h', 'help']);
|
|
9
|
-
const KNOWN_FLAGS = new Set(['--if-due', '--json', '--quiet']);
|
|
9
|
+
const KNOWN_FLAGS = new Set(['--if-due', '--json', '--quiet', '--proposals', '--dry-run']);
|
|
10
10
|
|
|
11
11
|
function printUsage() {
|
|
12
|
-
console.log('Usage: ukit self-improve [--if-due] [--json] [--quiet]');
|
|
12
|
+
console.log('Usage: ukit self-improve [--if-due] [--proposals] [--dry-run] [--json] [--quiet]');
|
|
13
13
|
console.log('');
|
|
14
14
|
console.log('Runs one self-improve pass: episode backfill, diagnostics, auto-tuning,');
|
|
15
|
-
console.log('pattern proposals, telemetry collect. Hooks run it automatically.');
|
|
15
|
+
console.log('pattern proposals, telemetry collect, code proposals. Hooks run it automatically.');
|
|
16
16
|
console.log('');
|
|
17
|
-
console.log(' --if-due
|
|
18
|
-
console.log(' --
|
|
19
|
-
console.log(' --
|
|
17
|
+
console.log(' --if-due Skip when the last pass ran < 10 minutes ago (hook mode)');
|
|
18
|
+
console.log(' --proposals Print the AUTO handoff-task proposals from the codeProposals step');
|
|
19
|
+
console.log(' --dry-run List would-be AUTO tasks without writing files');
|
|
20
|
+
console.log(' --json JSON output');
|
|
21
|
+
console.log(' --quiet No output (hook mode)');
|
|
20
22
|
}
|
|
21
23
|
|
|
22
24
|
function describeStep(stepResult) {
|
|
@@ -40,7 +42,11 @@ export async function runSelfImproveCommand({ projectRoot, argv = [] }) {
|
|
|
40
42
|
process.exitCode = 1;
|
|
41
43
|
return;
|
|
42
44
|
}
|
|
43
|
-
const
|
|
45
|
+
const dryRun = argv.includes('--dry-run');
|
|
46
|
+
const result = await runSelfImprove(projectRoot, {
|
|
47
|
+
force: !argv.includes('--if-due'),
|
|
48
|
+
codeProposalsOpts: { dryRun },
|
|
49
|
+
});
|
|
44
50
|
if (argv.includes('--quiet')) return;
|
|
45
51
|
if (argv.includes('--json')) {
|
|
46
52
|
console.log(JSON.stringify(result, null, 2));
|
|
@@ -52,4 +58,15 @@ export async function runSelfImproveCommand({ projectRoot, argv = [] }) {
|
|
|
52
58
|
}
|
|
53
59
|
console.log('[UKit] self-improve: ran');
|
|
54
60
|
for (const stepResult of result.steps) console.log(describeStep(stepResult));
|
|
61
|
+
if (argv.includes('--proposals') || dryRun) {
|
|
62
|
+
const codeStep = result.steps.find((s) => s.name === 'codeProposals');
|
|
63
|
+
const entries = Array.isArray(codeStep?.results) ? codeStep.results : [];
|
|
64
|
+
console.log(`[UKit] code proposals${dryRun ? ' (dry-run — no files written)' : ''}:`);
|
|
65
|
+
for (const entry of entries) {
|
|
66
|
+
const target = entry.taskFile ?? '-';
|
|
67
|
+
const detail = entry.reason ? ` (${entry.reason})` : '';
|
|
68
|
+
console.log(` ${entry.status}: ${entry.signature} → ${target}${detail}`);
|
|
69
|
+
}
|
|
70
|
+
if (entries.length === 0) console.log(' (no mined patterns)');
|
|
71
|
+
}
|
|
55
72
|
}
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
// `ukit vm` — agent-VM live-path driver surface (AGENT_VM_DRIVER_PLAN D-01,
|
|
2
|
+
// D-02, D-03). Discovery + shadow-evidence commands over the
|
|
3
|
+
// language-compiled runtime; nothing here executes real operations
|
|
4
|
+
// (shadow startFn is a stub) and nothing mutates config.
|
|
5
|
+
//
|
|
6
|
+
// list Reference plans (planId, planVersion, irHash)
|
|
7
|
+
// show <planId> One plan's compiled node ledger
|
|
8
|
+
// status Stage + evidence counts + shadow dir
|
|
9
|
+
// shadow-run <planId> Deterministic replay through a real VmEngine;
|
|
10
|
+
// appends one evidence record per run (D-03)
|
|
11
|
+
// evidence <key> Read .ukit/storage/agent-runtime/evidence/<key>.jsonl
|
|
12
|
+
// and print the promote() verdict verbatim (G-D3)
|
|
13
|
+
//
|
|
14
|
+
// Every subcommand is stage-gated on `decisionRuntime.vm`: 'off' → typed
|
|
15
|
+
// refuse `stage_off`, exit 2, zero writes under
|
|
16
|
+
// .ukit/storage/agent-runtime/ (G-D1). Exit codes: 0 success,
|
|
17
|
+
// 2 typed failure, 1 usage error.
|
|
18
|
+
|
|
19
|
+
import fs from 'node:fs/promises';
|
|
20
|
+
import os from 'node:os';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
|
|
23
|
+
import {
|
|
24
|
+
inspectRuntimeConfig,
|
|
25
|
+
resolveDecisionRuntimeStage,
|
|
26
|
+
} from '../../core/runtimeConfig.js';
|
|
27
|
+
import { loadPlan, loadPlanLibrary } from '../../core/agentRuntime/planLibrary.js';
|
|
28
|
+
import { promote, promotionCriteriaFor } from '../../core/agentRuntime/promotion.js';
|
|
29
|
+
import {
|
|
30
|
+
runShadowPlan,
|
|
31
|
+
readEvidence,
|
|
32
|
+
} from '../../core/agentRuntime/shadowRun.js';
|
|
33
|
+
|
|
34
|
+
const HELP_FLAGS = new Set(['--help', '-h', 'help']);
|
|
35
|
+
const SUBCOMMANDS = new Set(['list', 'show', 'status', 'shadow-run', 'evidence']);
|
|
36
|
+
const OPERAND_SUBCOMMANDS = new Set(['show', 'shadow-run', 'evidence']);
|
|
37
|
+
|
|
38
|
+
const RUNTIME_REL = path.join('.ukit', 'storage', 'agent-runtime');
|
|
39
|
+
|
|
40
|
+
function printUsage() {
|
|
41
|
+
console.log('Usage: ukit vm <command> [options]');
|
|
42
|
+
console.log('');
|
|
43
|
+
console.log('Commands:');
|
|
44
|
+
console.log(' list List pinned reference plans');
|
|
45
|
+
console.log(' show <planId> Show one plan\'s compiled node ledger');
|
|
46
|
+
console.log(' status Show decisionRuntime.vm stage + evidence summary');
|
|
47
|
+
console.log(' shadow-run <planId> Run one deterministic shadow pass (stub startFn)');
|
|
48
|
+
console.log(' evidence <key> Print the promote() verdict over stored runs');
|
|
49
|
+
console.log('');
|
|
50
|
+
console.log('Options:');
|
|
51
|
+
console.log(' --json JSON output (status, evidence)');
|
|
52
|
+
console.log(' -h, --help Show this help');
|
|
53
|
+
console.log('');
|
|
54
|
+
console.log('All commands require decisionRuntime.vm.stage != "off" (G-D1).');
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
// Same fail-safe posture as telemetry: unreadable config resolves {} → 'off'.
|
|
58
|
+
async function loadConfig(projectRoot) {
|
|
59
|
+
try {
|
|
60
|
+
const inspection = await inspectRuntimeConfig(projectRoot, { homeDir: os.homedir() });
|
|
61
|
+
return inspection && inspection.config && typeof inspection.config === 'object'
|
|
62
|
+
? inspection.config
|
|
63
|
+
: {};
|
|
64
|
+
} catch {
|
|
65
|
+
return {};
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function refuseStageOff() {
|
|
70
|
+
console.error('[UKit] vm: refused — decisionRuntime.vm.stage is off (stage_off).');
|
|
71
|
+
console.error('[UKit] Set decisionRuntime.vm.stage to "shadow" (or higher) in .ukit/storage/config.json to enable.');
|
|
72
|
+
process.exitCode = 2;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// --- list -----------------------------------------------------------------
|
|
76
|
+
|
|
77
|
+
async function listCommand({ config }) {
|
|
78
|
+
const catalog = loadPlanLibrary({ config });
|
|
79
|
+
const ids = Object.keys(catalog).sort();
|
|
80
|
+
if (ids.length === 0) {
|
|
81
|
+
console.log('vm: no plans in library');
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
84
|
+
for (const id of ids) {
|
|
85
|
+
const entry = catalog[id];
|
|
86
|
+
if (entry.ok) {
|
|
87
|
+
console.log(`${id} planVersion=${entry.planVersion} irHash=${entry.irHash}`);
|
|
88
|
+
if (entry.description) console.log(` ${entry.description}`);
|
|
89
|
+
} else {
|
|
90
|
+
const code = entry.errors?.[0]?.code ?? 'load_error';
|
|
91
|
+
console.log(`${id} ERROR ${code}`);
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// --- show -----------------------------------------------------------------
|
|
97
|
+
|
|
98
|
+
function showCommand({ config, planId }) {
|
|
99
|
+
const loaded = loadPlan(planId, { config });
|
|
100
|
+
if (!loaded.ok) {
|
|
101
|
+
for (const err of loaded.errors ?? []) {
|
|
102
|
+
console.error(`[UKit] vm show: ${err.code ?? 'load_error'} ${err.planId ?? planId}`);
|
|
103
|
+
}
|
|
104
|
+
process.exitCode = 2;
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
console.log(`plan ${loaded.planId}`);
|
|
108
|
+
console.log(` planVersion ${loaded.planVersion} irHash ${loaded.irHash}`);
|
|
109
|
+
if (loaded.description) console.log(` ${loaded.description}`);
|
|
110
|
+
for (const id of loaded.plan.order) {
|
|
111
|
+
const node = loaded.plan.nodes.get(id);
|
|
112
|
+
const deps = node.deps.length > 0 ? ` deps=[${node.deps.join(',')}]` : '';
|
|
113
|
+
const sel = node.selector?.eventType ? ` selector=${node.selector.eventType}` : '';
|
|
114
|
+
const branches = node.op === 'BRANCH'
|
|
115
|
+
? ` branches=[${Object.keys(node.branches ?? {}).join(',')}]`
|
|
116
|
+
: '';
|
|
117
|
+
const timeout = node.timeoutMs != null ? ` timeoutMs=${node.timeoutMs}` : '';
|
|
118
|
+
console.log(` ${id}: ${node.op}${deps}${sel}${branches}${timeout}`);
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// --- status ---------------------------------------------------------------
|
|
123
|
+
|
|
124
|
+
async function statusCommand({ projectRoot, config, json }) {
|
|
125
|
+
const stage = resolveDecisionRuntimeStage(config, 'vm');
|
|
126
|
+
const catalog = loadPlanLibrary({ config });
|
|
127
|
+
const planIds = Object.keys(catalog).sort();
|
|
128
|
+
const brokenPlans = planIds.filter((id) => !catalog[id].ok);
|
|
129
|
+
const evidence = await readEvidence(projectRoot, { key: 'vm' });
|
|
130
|
+
const runs = evidence.ok ? evidence.runs.length : null;
|
|
131
|
+
let shadowDirs = 0;
|
|
132
|
+
try {
|
|
133
|
+
shadowDirs = (await fs.readdir(path.join(projectRoot, RUNTIME_REL, 'shadow'), { withFileTypes: true }))
|
|
134
|
+
.filter((e) => e.isDirectory()).length;
|
|
135
|
+
} catch { /* no shadow dir yet */ }
|
|
136
|
+
if (json) {
|
|
137
|
+
console.log(JSON.stringify({
|
|
138
|
+
stage,
|
|
139
|
+
plans: planIds,
|
|
140
|
+
brokenPlans,
|
|
141
|
+
evidenceRuns: runs,
|
|
142
|
+
shadowDirs,
|
|
143
|
+
}));
|
|
144
|
+
return;
|
|
145
|
+
}
|
|
146
|
+
console.log(`vm stage: ${stage}`);
|
|
147
|
+
console.log(`plans: ${planIds.length} (${planIds.join(', ') || 'none'})`);
|
|
148
|
+
if (brokenPlans.length > 0) console.log(`broken plans: ${brokenPlans.join(', ')}`);
|
|
149
|
+
console.log(`evidence runs: ${runs === null ? 'unknown' : runs}`);
|
|
150
|
+
console.log(`shadow instances: ${shadowDirs} plan dir(s)`);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// --- shadow-run -----------------------------------------------------------
|
|
154
|
+
|
|
155
|
+
async function shadowRunCommand({ projectRoot, config, planId }) {
|
|
156
|
+
const res = await runShadowPlan(planId, { projectRoot, config });
|
|
157
|
+
if (!res.ok) {
|
|
158
|
+
console.error(`[UKit] vm shadow-run: ${res.code} — ${res.reason}`);
|
|
159
|
+
process.exitCode = 2;
|
|
160
|
+
return;
|
|
161
|
+
}
|
|
162
|
+
console.log(`shadow-run ${res.planId}: ${res.status} (${res.code})`);
|
|
163
|
+
console.log(` planInstance ${res.planInstanceId}`);
|
|
164
|
+
console.log(` planVersion ${res.planVersion} irHash ${res.irHash}`);
|
|
165
|
+
console.log(` transitions ${res.transitions.length} wallMs ${res.wallMs}`);
|
|
166
|
+
for (const t of res.transitions) {
|
|
167
|
+
console.log(` ${t.node} -> ${t.to}${t.branch !== undefined ? ` (${t.branch})` : ''}`);
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
// --- evidence -------------------------------------------------------------
|
|
172
|
+
|
|
173
|
+
async function evidenceCommand({ projectRoot, config, key, json }) {
|
|
174
|
+
const res = await readEvidence(projectRoot, { key });
|
|
175
|
+
if (!res.ok) {
|
|
176
|
+
console.error(`[UKit] vm evidence: ${res.code} — ${res.reason}`);
|
|
177
|
+
process.exitCode = 2;
|
|
178
|
+
return;
|
|
179
|
+
}
|
|
180
|
+
const verdict = promote(config, { key, runs: res.runs });
|
|
181
|
+
const criteria = promotionCriteriaFor(key);
|
|
182
|
+
if (json) {
|
|
183
|
+
console.log(JSON.stringify({ key, runs: res.runs.length, dropped: res.dropped, criteria, verdict }));
|
|
184
|
+
return;
|
|
185
|
+
}
|
|
186
|
+
console.log(`evidence ${key}: ${res.runs.length} run(s)${res.dropped > 0 ? ` (${res.dropped} malformed dropped)` : ''}`);
|
|
187
|
+
// G-D3: verdict is advisory — printed verbatim, never written back.
|
|
188
|
+
console.log(`promote ${key}: stage=${verdict.stage} promoted=${verdict.promoted} reason=${verdict.reason}`);
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
// --- dispatch -------------------------------------------------------------
|
|
192
|
+
|
|
193
|
+
export async function runVm({ projectRoot, argv = [] }) {
|
|
194
|
+
const args = Array.isArray(argv) ? argv : [];
|
|
195
|
+
const sub = (args[0] ?? '').toLowerCase();
|
|
196
|
+
|
|
197
|
+
if (args.length === 0 || HELP_FLAGS.has(sub)) {
|
|
198
|
+
printUsage();
|
|
199
|
+
return;
|
|
200
|
+
}
|
|
201
|
+
if (!SUBCOMMANDS.has(sub)) {
|
|
202
|
+
console.error(`[UKit] Unknown vm subcommand: ${sub}`);
|
|
203
|
+
printUsage();
|
|
204
|
+
process.exitCode = 1;
|
|
205
|
+
return;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
const rest = args.slice(1);
|
|
209
|
+
if (rest.some((a) => HELP_FLAGS.has(a))) {
|
|
210
|
+
printUsage();
|
|
211
|
+
return;
|
|
212
|
+
}
|
|
213
|
+
const positional = rest.filter((a) => !a.startsWith('-'));
|
|
214
|
+
const flags = rest.filter((a) => a.startsWith('-'));
|
|
215
|
+
const allowed = (sub === 'status' || sub === 'evidence') ? new Set(['--json']) : new Set();
|
|
216
|
+
const unknownFlags = flags.filter((f) => !allowed.has(f));
|
|
217
|
+
if (unknownFlags.length > 0) {
|
|
218
|
+
console.error(`[UKit] Unknown vm flag(s): ${unknownFlags.join(', ')}`);
|
|
219
|
+
printUsage();
|
|
220
|
+
process.exitCode = 1;
|
|
221
|
+
return;
|
|
222
|
+
}
|
|
223
|
+
if (positional.length > 1 || (OPERAND_SUBCOMMANDS.has(sub) && positional.length === 0)) {
|
|
224
|
+
printUsage();
|
|
225
|
+
process.exitCode = 1;
|
|
226
|
+
return;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
// G-D1 gate: 'off' refuses every subcommand with zero writes.
|
|
230
|
+
const config = await loadConfig(projectRoot);
|
|
231
|
+
if (resolveDecisionRuntimeStage(config, 'vm') === 'off') {
|
|
232
|
+
refuseStageOff();
|
|
233
|
+
return;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
const operand = positional[0];
|
|
237
|
+
if (sub === 'list') return listCommand({ config });
|
|
238
|
+
if (sub === 'show') return showCommand({ config, planId: operand });
|
|
239
|
+
if (sub === 'status') return statusCommand({ projectRoot, config, json: flags.includes('--json') });
|
|
240
|
+
if (sub === 'shadow-run') return shadowRunCommand({ projectRoot, config, planId: operand });
|
|
241
|
+
return evidenceCommand({ projectRoot, config, key: operand, json: flags.includes('--json') });
|
|
242
|
+
}
|
package/src/cli/index.js
CHANGED
|
@@ -13,6 +13,7 @@ import { runFeedback } from './commands/feedback.js';
|
|
|
13
13
|
import { runPlaybook } from './commands/playbook.js';
|
|
14
14
|
import { runDecision } from './commands/decision.js';
|
|
15
15
|
import { runSelfImproveCommand } from './commands/selfImprove.js';
|
|
16
|
+
import { runVm } from './commands/vm.js';
|
|
16
17
|
const GLOBAL_FLAGS = new Set(['--help', '-h', '--version', '-v']);
|
|
17
18
|
|
|
18
19
|
export async function runCli({ argv, packageRoot, projectRoot, packageVersion }) {
|
|
@@ -105,6 +106,11 @@ export async function runCli({ argv, packageRoot, projectRoot, packageVersion })
|
|
|
105
106
|
return;
|
|
106
107
|
}
|
|
107
108
|
|
|
109
|
+
if (command === 'vm') {
|
|
110
|
+
await runVm({ projectRoot, argv: commandArgv });
|
|
111
|
+
return;
|
|
112
|
+
}
|
|
113
|
+
|
|
108
114
|
if (command === 'build' && (commandArgv[0] ?? '').toLowerCase() === 'index') {
|
|
109
115
|
await runIndexTools({ projectRoot, argv: ['build', ...commandArgv.slice(1)] });
|
|
110
116
|
return;
|
|
@@ -138,6 +144,7 @@ export async function runCli({ argv, packageRoot, projectRoot, packageVersion })
|
|
|
138
144
|
console.log(' telemetry Flight recorder (collect/status/digest/export-support/import/evaluate)');
|
|
139
145
|
console.log(' feedback Record or list wrong-route feedback labels');
|
|
140
146
|
console.log(' self-improve Run the automatic learn/tune/collect pass now (hooks run it)');
|
|
147
|
+
console.log(' vm Agent VM driver (list/show/status/shadow-run/evidence)');
|
|
141
148
|
console.log(' update Upgrade the global UKit CLI to the latest version');
|
|
142
149
|
console.log(' version Show UKit version');
|
|
143
150
|
console.log('');
|
|
@@ -0,0 +1,384 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* agentRuntime/shadowRun.js — agent-VM live-path driver (D-02/D-03).
|
|
3
|
+
*
|
|
4
|
+
* The shadow driver makes `decisionRuntime.vm` ≥ 'shadow' actually produce
|
|
5
|
+
* evidence: it loads a pinned reference plan via `planLibrary.loadPlan`,
|
|
6
|
+
* runs it inside a real `createVmEngine` rooted at
|
|
7
|
+
* `.ukit/storage/agent-runtime/shadow/<planId>/`, and feeds each activated
|
|
8
|
+
* node a deterministic synthetic event:
|
|
9
|
+
*
|
|
10
|
+
* RUN / WAIT_EVENT → <node.selector.eventType> (default
|
|
11
|
+
* 'operation.completed') with
|
|
12
|
+
* safePayload { outcome:'completed', ...selector.fields }
|
|
13
|
+
* BRANCH → <node.selector.eventType> (default 'branch.selected')
|
|
14
|
+
* with safePayload { [selector.field]: <first declared
|
|
15
|
+
* choice> }
|
|
16
|
+
* COMPLETE → resolved by the engine itself (no external event)
|
|
17
|
+
*
|
|
18
|
+
* `startFn` is a noop — shadow runs measure VM/decision-plane behaviour
|
|
19
|
+
* (journals, continuations, classify verdicts on unclassifiable events),
|
|
20
|
+
* never real operation side effects. The run is deterministic modulo the
|
|
21
|
+
* decision model: selector-matched events never call classifyFn; an
|
|
22
|
+
* unroutable node degrades to the deterministic escalation lane
|
|
23
|
+
* (recovery_required) exactly as the V-05 contract requires.
|
|
24
|
+
*
|
|
25
|
+
* Never throws: every failure resolves to a typed
|
|
26
|
+
* `{ ok:false, code, reason }` — `stage_off` (decisionRuntime.vm 'off' →
|
|
27
|
+
* zero filesystem writes, G-D1), plan-load error codes verbatim from
|
|
28
|
+
* loadPlan, `malformed_plan`, `unknown_plan_id`, `engine_error`,
|
|
29
|
+
* `unterminated` (bounded-loop guard tripped).
|
|
30
|
+
*
|
|
31
|
+
* Evidence (D-03): each run appends one JSONL record to
|
|
32
|
+
* `.ukit/storage/agent-runtime/evidence/<key>.jsonl`
|
|
33
|
+
* `{planId, irHash, planVersion, planInstanceId, status, transitions,
|
|
34
|
+
* wallMs, qualityDelta, wallP95Ratio, costRatio, forbidden, at}` —
|
|
35
|
+
* qualityDelta/wallP95Ratio/costRatio stay `null` until a measured
|
|
36
|
+
* baseline exists (promotion.js treats missing metrics as non-passing,
|
|
37
|
+
* so shadow evidence can never self-promote). The file rotates to
|
|
38
|
+
* `<key>.jsonl.1` at 256 KiB (single generation, bounded). Callers may
|
|
39
|
+
* pass `evidence:false` to skip the append (tests, dry probing).
|
|
40
|
+
*/
|
|
41
|
+
|
|
42
|
+
import { promises as fs } from 'node:fs';
|
|
43
|
+
import path from 'node:path';
|
|
44
|
+
|
|
45
|
+
import { resolveDecisionRuntimeStage } from '../runtimeConfig.js';
|
|
46
|
+
import { isTerminal, CONTRACT_VERSION } from './contract.js';
|
|
47
|
+
import { createVmEngine } from './vmEngine.js';
|
|
48
|
+
import { loadPlan } from './planLibrary.js';
|
|
49
|
+
|
|
50
|
+
const RUNTIME_REL = path.join('.ukit', 'storage', 'agent-runtime');
|
|
51
|
+
const EVIDENCE_REL = path.join('evidence');
|
|
52
|
+
const EVIDENCE_MAX_BYTES = 256 * 1024;
|
|
53
|
+
|
|
54
|
+
/** Bounded driver loop: every step delivers ≥1 event, so a correct engine
|
|
55
|
+
* always terminates well inside this bound — tripping it means a driver or
|
|
56
|
+
* engine fault, surfaced as `unterminated` rather than an infinite await. */
|
|
57
|
+
const maxSteps = (nodeCount) => nodeCount * 8 + 16;
|
|
58
|
+
|
|
59
|
+
const KEY_RE = /^[a-z0-9][a-z0-9-]*$/;
|
|
60
|
+
|
|
61
|
+
function refuse(code, reason) {
|
|
62
|
+
return Object.freeze({ ok: false, code, reason });
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function runtimeDir(projectRoot) {
|
|
66
|
+
return path.join(projectRoot, RUNTIME_REL);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export function evidenceFilePath(projectRoot, key = 'vm') {
|
|
70
|
+
return path.join(runtimeDir(projectRoot), EVIDENCE_REL, `${key}.jsonl`);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Append one evidence record (single JSON line) to the bounded store.
|
|
75
|
+
* Rotation: one generation (`<key>.jsonl.1`), renamed aside before append
|
|
76
|
+
* when the file would exceed EVIDENCE_MAX_BYTES. Never throws.
|
|
77
|
+
*
|
|
78
|
+
* @param {string} projectRoot
|
|
79
|
+
* @param {object} record JSON-serializable run record
|
|
80
|
+
* @param {{key?: string}} [opts]
|
|
81
|
+
* @returns {Promise<{ok:true, bytes:number}|{ok:false, code:string, reason:string}>}
|
|
82
|
+
*/
|
|
83
|
+
export async function appendEvidence(projectRoot, record, { key = 'vm' } = {}) {
|
|
84
|
+
if (typeof projectRoot !== 'string' || projectRoot === '') {
|
|
85
|
+
return refuse('invalid_root', 'projectRoot required');
|
|
86
|
+
}
|
|
87
|
+
if (!KEY_RE.test(key)) return refuse('invalid_key', `evidence key ${JSON.stringify(key)}`);
|
|
88
|
+
if (record === null || typeof record !== 'object') {
|
|
89
|
+
return refuse('malformed_record', 'record must be an object');
|
|
90
|
+
}
|
|
91
|
+
let line;
|
|
92
|
+
try {
|
|
93
|
+
line = `${JSON.stringify(record)}\n`;
|
|
94
|
+
} catch {
|
|
95
|
+
return refuse('malformed_record', 'record is not JSON-serializable');
|
|
96
|
+
}
|
|
97
|
+
const file = evidenceFilePath(projectRoot, key);
|
|
98
|
+
try {
|
|
99
|
+
let size = 0;
|
|
100
|
+
try {
|
|
101
|
+
size = (await fs.stat(file)).size;
|
|
102
|
+
} catch (err) {
|
|
103
|
+
if (err?.code !== 'ENOENT') throw err;
|
|
104
|
+
}
|
|
105
|
+
if (size > 0 && size + Buffer.byteLength(line, 'utf8') > EVIDENCE_MAX_BYTES) {
|
|
106
|
+
await fs.rename(file, `${file}.1`).catch((err) => {
|
|
107
|
+
if (err?.code !== 'ENOENT') throw err;
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
await fs.mkdir(path.dirname(file), { recursive: true });
|
|
111
|
+
const fh = await fs.open(file, 'a');
|
|
112
|
+
try {
|
|
113
|
+
await fh.writeFile(line, 'utf8');
|
|
114
|
+
await fh.sync();
|
|
115
|
+
} finally {
|
|
116
|
+
await fh.close();
|
|
117
|
+
}
|
|
118
|
+
return { ok: true, bytes: Buffer.byteLength(line, 'utf8') };
|
|
119
|
+
} catch (err) {
|
|
120
|
+
return refuse('evidence_write_error', err?.code ?? String(err));
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Read evidence runs for one family key. Missing file → zero runs; a
|
|
126
|
+
* crash-torn final line is dropped; malformed mid-file lines are dropped
|
|
127
|
+
* too (evidence is advisory — one bad row must not break the read).
|
|
128
|
+
*
|
|
129
|
+
* @param {string} projectRoot
|
|
130
|
+
* @param {{key?: string}} [opts]
|
|
131
|
+
* @returns {Promise<{ok:true, key:string, runs:object[], dropped:number}
|
|
132
|
+
* |{ok:false, code:string, reason:string}>}
|
|
133
|
+
*/
|
|
134
|
+
export async function readEvidence(projectRoot, { key = 'vm' } = {}) {
|
|
135
|
+
if (typeof projectRoot !== 'string' || projectRoot === '') {
|
|
136
|
+
return refuse('invalid_root', 'projectRoot required');
|
|
137
|
+
}
|
|
138
|
+
if (!KEY_RE.test(key)) return refuse('invalid_key', `evidence key ${JSON.stringify(key)}`);
|
|
139
|
+
const file = evidenceFilePath(projectRoot, key);
|
|
140
|
+
let raw;
|
|
141
|
+
try {
|
|
142
|
+
raw = await fs.readFile(file, 'utf8');
|
|
143
|
+
} catch (err) {
|
|
144
|
+
if (err?.code === 'ENOENT') return { ok: true, key, runs: [], dropped: 0 };
|
|
145
|
+
return refuse('evidence_read_error', err?.code ?? String(err));
|
|
146
|
+
}
|
|
147
|
+
const runs = [];
|
|
148
|
+
let dropped = 0;
|
|
149
|
+
for (const line of raw.split('\n')) {
|
|
150
|
+
if (line === '') continue;
|
|
151
|
+
try {
|
|
152
|
+
const parsed = JSON.parse(line);
|
|
153
|
+
if (parsed !== null && typeof parsed === 'object' && !Array.isArray(parsed)) {
|
|
154
|
+
runs.push(parsed);
|
|
155
|
+
} else {
|
|
156
|
+
dropped += 1;
|
|
157
|
+
}
|
|
158
|
+
} catch {
|
|
159
|
+
dropped += 1;
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
return { ok: true, key, runs, dropped };
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/** Synthetic SemanticEvent for one running node (deterministic payload). */
|
|
166
|
+
function syntheticEvent(planInstanceId, node, seq, observedAt) {
|
|
167
|
+
// selector.fields is validated as a name-list by the compiler; only a
|
|
168
|
+
// caller-supplied object (test fixtures) is merged into the payload.
|
|
169
|
+
const payload = (node.selector?.fields !== null && typeof node.selector?.fields === 'object'
|
|
170
|
+
&& !Array.isArray(node.selector.fields)) ? { ...node.selector.fields } : {};
|
|
171
|
+
let eventType = node.selector?.eventType;
|
|
172
|
+
if (node.op === 'BRANCH') {
|
|
173
|
+
const choice = Object.keys(node.branches ?? {})[0];
|
|
174
|
+
const field = node.selector?.field ?? 'branch';
|
|
175
|
+
payload[field] = choice;
|
|
176
|
+
eventType = eventType ?? 'branch.selected';
|
|
177
|
+
return {
|
|
178
|
+
choice: { field, value: choice, target: node.branches?.[choice] },
|
|
179
|
+
record: {
|
|
180
|
+
eventId: `shadow-${planInstanceId}-${node.id}-${seq}`,
|
|
181
|
+
operationId: `${planInstanceId}:${node.id}`,
|
|
182
|
+
seq,
|
|
183
|
+
eventType,
|
|
184
|
+
observedAt,
|
|
185
|
+
producerVersion: 'ukit-vm-shadow',
|
|
186
|
+
contractVersion: CONTRACT_VERSION,
|
|
187
|
+
privacyClass: 'internal',
|
|
188
|
+
artifactRefs: [],
|
|
189
|
+
safePayload: payload,
|
|
190
|
+
},
|
|
191
|
+
};
|
|
192
|
+
}
|
|
193
|
+
payload.outcome = 'completed';
|
|
194
|
+
eventType = eventType ?? 'operation.completed';
|
|
195
|
+
return {
|
|
196
|
+
choice: null,
|
|
197
|
+
record: {
|
|
198
|
+
eventId: `shadow-${planInstanceId}-${node.id}-${seq}`,
|
|
199
|
+
operationId: `${planInstanceId}:${node.id}`,
|
|
200
|
+
seq,
|
|
201
|
+
eventType,
|
|
202
|
+
observedAt,
|
|
203
|
+
producerVersion: 'ukit-vm-shadow',
|
|
204
|
+
contractVersion: CONTRACT_VERSION,
|
|
205
|
+
privacyClass: 'internal',
|
|
206
|
+
artifactRefs: [],
|
|
207
|
+
safePayload: payload,
|
|
208
|
+
},
|
|
209
|
+
};
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* Run one reference plan under a shadow engine and return its transition
|
|
214
|
+
* ledger. Deterministic: RUN/WAIT_EVENT resolve 'completed' on their first
|
|
215
|
+
* delivery; BRANCH takes its first declared choice; the loop stops as soon
|
|
216
|
+
* as the plan reaches a terminal status.
|
|
217
|
+
*
|
|
218
|
+
* @param {string} planId
|
|
219
|
+
* @param {{projectRoot?: string, config?: object, plansDir?: string,
|
|
220
|
+
* engineDir?: string, now?: () => Date, classifyFn?: Function,
|
|
221
|
+
* decisionClient?: object, decider?: object, evidence?: boolean,
|
|
222
|
+
* maxTransitions?: number}} [opts]
|
|
223
|
+
* `plansDir`/`engineDir` exist for tests and non-repo fixtures; the CLI
|
|
224
|
+
* never passes them. `classifyFn`/`decisionClient`/`decider` are
|
|
225
|
+
* forwarded to createVmEngine (test fault injection — the live path
|
|
226
|
+
* leaves them unset so the V-05 stage-gated adapter applies).
|
|
227
|
+
* @returns {Promise<object>} frozen result — see module header.
|
|
228
|
+
*/
|
|
229
|
+
export async function runShadowPlan(planId, opts = {}) {
|
|
230
|
+
const { projectRoot, config = null } = opts;
|
|
231
|
+
try {
|
|
232
|
+
if (resolveDecisionRuntimeStage(config, 'vm') === 'off') {
|
|
233
|
+
return refuse('stage_off', 'decisionRuntime.vm.stage is off — zero writes');
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
const loaded = loadPlan(planId, {
|
|
237
|
+
...(opts.plansDir ? { plansDir: opts.plansDir } : {}),
|
|
238
|
+
config,
|
|
239
|
+
});
|
|
240
|
+
if (!loaded.ok) {
|
|
241
|
+
const first = loaded.errors?.[0] ?? {};
|
|
242
|
+
return refuse(
|
|
243
|
+
typeof first.code === 'string' ? first.code : 'plan_not_found',
|
|
244
|
+
`${planId}: ${JSON.stringify(first)}`,
|
|
245
|
+
);
|
|
246
|
+
}
|
|
247
|
+
const plan = loaded.plan;
|
|
248
|
+
if (!plan || !(plan.nodes instanceof Map) || plan.nodes.size === 0) {
|
|
249
|
+
return refuse('malformed_plan', `${planId}: compiled plan has no nodes`);
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
const dir = typeof opts.engineDir === 'string' && opts.engineDir !== ''
|
|
253
|
+
? opts.engineDir
|
|
254
|
+
: (typeof projectRoot === 'string' && projectRoot !== ''
|
|
255
|
+
? path.join(runtimeDir(projectRoot), 'shadow', planId)
|
|
256
|
+
: null);
|
|
257
|
+
if (dir == null) return refuse('invalid_root', 'projectRoot or engineDir required');
|
|
258
|
+
|
|
259
|
+
const started = Date.now();
|
|
260
|
+
const engine = createVmEngine({
|
|
261
|
+
dir,
|
|
262
|
+
...(typeof opts.now === 'function' ? { now: opts.now } : {}),
|
|
263
|
+
...(typeof opts.classifyFn === 'function' ? { classifyFn: opts.classifyFn } : {}),
|
|
264
|
+
...(opts.decisionClient != null ? { decisionClient: opts.decisionClient } : {}),
|
|
265
|
+
...(opts.decider != null ? { decider: opts.decider } : {}),
|
|
266
|
+
startFn: async () => {}, // stub: shadow runs measure the VM, not real ops
|
|
267
|
+
config,
|
|
268
|
+
});
|
|
269
|
+
|
|
270
|
+
let startRes;
|
|
271
|
+
try {
|
|
272
|
+
startRes = await engine.start(plan);
|
|
273
|
+
} catch (err) {
|
|
274
|
+
return refuse('engine_error', err?.code ?? String(err));
|
|
275
|
+
}
|
|
276
|
+
if (!startRes || startRes.unsupported || typeof startRes.planInstanceId !== 'string') {
|
|
277
|
+
return refuse('malformed_plan', `${planId}: engine refused start (${startRes?.code ?? 'no instance'})`);
|
|
278
|
+
}
|
|
279
|
+
const pi = startRes.planInstanceId;
|
|
280
|
+
|
|
281
|
+
const transitions = [];
|
|
282
|
+
let status = 'running';
|
|
283
|
+
let steps = 0;
|
|
284
|
+
try {
|
|
285
|
+
const bound = maxSteps(plan.nodes.size);
|
|
286
|
+
while (steps < bound) {
|
|
287
|
+
const snapshot = await engine.state(pi);
|
|
288
|
+
status = snapshot.status;
|
|
289
|
+
if (isTerminal(status)) break;
|
|
290
|
+
const nodeId = plan.order.find(
|
|
291
|
+
(id) => snapshot.nodes[id]?.state === 'running'
|
|
292
|
+
&& ['RUN', 'WAIT_EVENT', 'BRANCH'].includes(plan.nodes.get(id)?.op),
|
|
293
|
+
);
|
|
294
|
+
if (nodeId == null) {
|
|
295
|
+
// Nothing the driver can feed (e.g. only wrapper/non-terminal
|
|
296
|
+
// leftovers) — report honestly rather than spinning.
|
|
297
|
+
break;
|
|
298
|
+
}
|
|
299
|
+
const node = plan.nodes.get(nodeId);
|
|
300
|
+
const seq = (snapshot.cursor?.[nodeId] ?? 0) + 1;
|
|
301
|
+
const { record, choice } = syntheticEvent(
|
|
302
|
+
pi, node, seq, (typeof opts.now === 'function' ? opts.now() : new Date()).toISOString(),
|
|
303
|
+
);
|
|
304
|
+
const res = await engine.deliver(record);
|
|
305
|
+
steps += 1;
|
|
306
|
+
for (const t of res.transitions ?? []) {
|
|
307
|
+
transitions.push(
|
|
308
|
+
choice != null && t.node === nodeId && t.branch !== undefined
|
|
309
|
+
? { node: t.node, to: t.to, branch: t.branch }
|
|
310
|
+
: { node: t.node, to: t.to },
|
|
311
|
+
);
|
|
312
|
+
}
|
|
313
|
+
if (!res.consumed) {
|
|
314
|
+
return refuse('event_not_consumed',
|
|
315
|
+
`${planId}/${nodeId}: deliver rejected (${res.code ?? 'unknown'})`);
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
} catch (err) {
|
|
319
|
+
return refuse('engine_error', err?.code ?? String(err));
|
|
320
|
+
} finally {
|
|
321
|
+
try { await engine.close(); } catch { /* persist-best-effort; run result stands */ }
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
const final = await engineStateSafe(engine, pi);
|
|
325
|
+
if (final) status = final.status;
|
|
326
|
+
const wallMs = Date.now() - started;
|
|
327
|
+
if (!isTerminal(status)) {
|
|
328
|
+
return refuse('unterminated', `${planId}: no terminal status after ${steps} deliveries`);
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
const nodeStates = {};
|
|
332
|
+
for (const [id, n] of Object.entries(final?.nodes ?? {})) {
|
|
333
|
+
nodeStates[id] = { state: n.state, attempt: n.attempt };
|
|
334
|
+
}
|
|
335
|
+
const result = Object.freeze({
|
|
336
|
+
ok: true,
|
|
337
|
+
code: status,
|
|
338
|
+
planId,
|
|
339
|
+
planInstanceId: pi,
|
|
340
|
+
status,
|
|
341
|
+
transitions: Object.freeze(transitions),
|
|
342
|
+
nodeStates: Object.freeze(nodeStates),
|
|
343
|
+
wallMs,
|
|
344
|
+
irHash: loaded.irHash,
|
|
345
|
+
planVersion: loaded.planVersion,
|
|
346
|
+
});
|
|
347
|
+
|
|
348
|
+
// D-03 evidence append (caller may opt out with evidence:false).
|
|
349
|
+
if (opts.evidence !== false && typeof projectRoot === 'string' && projectRoot !== '') {
|
|
350
|
+
const append = await appendEvidence(projectRoot, {
|
|
351
|
+
planId,
|
|
352
|
+
irHash: loaded.irHash,
|
|
353
|
+
planVersion: loaded.planVersion,
|
|
354
|
+
planInstanceId: pi,
|
|
355
|
+
status,
|
|
356
|
+
transitions: transitions.length,
|
|
357
|
+
wallMs,
|
|
358
|
+
qualityDelta: null,
|
|
359
|
+
wallP95Ratio: null,
|
|
360
|
+
costRatio: null,
|
|
361
|
+
// 'cancelled' is a normal terminal for plans whose untaken BRANCH
|
|
362
|
+
// path cancels dependents — only real failure lanes are forbidden.
|
|
363
|
+
forbidden: status === 'failed',
|
|
364
|
+
at: new Date().toISOString(),
|
|
365
|
+
}, { key: 'vm' });
|
|
366
|
+
if (!append.ok) {
|
|
367
|
+
return refuse('evidence_write_error', `${planId}: ${append.reason}`);
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
return result;
|
|
371
|
+
} catch (err) {
|
|
372
|
+
// Outer never-throws guard: an unexpected fault (e.g. engine internals
|
|
373
|
+
// throwing a non-typed error) still resolves to a typed refuse.
|
|
374
|
+
return refuse('engine_error', err?.code ?? String(err?.message ?? err));
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
async function engineStateSafe(engine, pi) {
|
|
379
|
+
try {
|
|
380
|
+
return await engine.state(pi);
|
|
381
|
+
} catch {
|
|
382
|
+
return null;
|
|
383
|
+
}
|
|
384
|
+
}
|
|
@@ -229,7 +229,7 @@ function pushStageError(errors, node, label) {
|
|
|
229
229
|
export function buildDefaultRuntimeConfig(overrides = {}) {
|
|
230
230
|
const safeOverrides = isPlainObject(overrides) ? overrides : {};
|
|
231
231
|
|
|
232
|
-
|
|
232
|
+
const config = mergeObjects({
|
|
233
233
|
version: PACKAGE_VERSION,
|
|
234
234
|
agent: 'claude-code',
|
|
235
235
|
autonomy: {
|
|
@@ -451,7 +451,10 @@ export function buildDefaultRuntimeConfig(overrides = {}) {
|
|
|
451
451
|
episodes: { autoWrite: true },
|
|
452
452
|
tuning: { enabled: true, applyMode: 'auto' },
|
|
453
453
|
// 3.3.0: automatic collect → learn → apply pass (src/learning/selfImprove.js).
|
|
454
|
-
|
|
454
|
+
// codeProposals: AUTO task-file lane (SELF_IMPROVE_CODE_LANE plan).
|
|
455
|
+
// codeProposalsCommit is a reserved hard-false — writing a task file is
|
|
456
|
+
// the ceiling; self-improve never commits code (clamped below).
|
|
457
|
+
selfImprove: { enabled: true, codeProposals: true, codeProposalsCommit: false },
|
|
455
458
|
// C52 M04.2 stage keys (SPEC §5 FR-016–FR-018). candidates: repeated
|
|
456
459
|
// corrections/suppressions/escalations promote to a LearningCandidate
|
|
457
460
|
// only after minOccurrences + cross-session evidence (status
|
|
@@ -575,6 +578,15 @@ export function buildDefaultRuntimeConfig(overrides = {}) {
|
|
|
575
578
|
},
|
|
576
579
|
},
|
|
577
580
|
}, safeOverrides);
|
|
581
|
+
|
|
582
|
+
// learning.selfImprove.codeProposalsCommit is hard-false forever: writing an
|
|
583
|
+
// AUTO task file is the ceiling of the code lane — applying/committing a
|
|
584
|
+
// patch is always the executor's job under test + review. A user-provided
|
|
585
|
+
// `true` is silently clamped, never honored.
|
|
586
|
+
if (isPlainObject(config.learning?.selfImprove)) {
|
|
587
|
+
config.learning.selfImprove.codeProposalsCommit = false;
|
|
588
|
+
}
|
|
589
|
+
return config;
|
|
578
590
|
}
|
|
579
591
|
|
|
580
592
|
export function validateRuntimeConfig(config) {
|
|
@@ -1114,6 +1126,8 @@ export function validateRuntimeConfig(config) {
|
|
|
1114
1126
|
errors.push('learning.selfImprove must be an object.');
|
|
1115
1127
|
} else {
|
|
1116
1128
|
pushBooleanError(errors, learning.selfImprove.enabled, 'learning.selfImprove.enabled');
|
|
1129
|
+
pushBooleanError(errors, learning.selfImprove.codeProposals, 'learning.selfImprove.codeProposals');
|
|
1130
|
+
pushBooleanError(errors, learning.selfImprove.codeProposalsCommit, 'learning.selfImprove.codeProposalsCommit');
|
|
1117
1131
|
}
|
|
1118
1132
|
}
|
|
1119
1133
|
// C52 M04.2 stage keys — same optional-present contract as routing.*.
|
|
@@ -0,0 +1,349 @@
|
|
|
1
|
+
// codeProposals.js — self-improve code lane (SELF_IMPROVE_CODE_LANE plan).
|
|
2
|
+
//
|
|
3
|
+
// Turns mined, recurring failure patterns into HANDOFF TASK FILES
|
|
4
|
+
// (`docs/AI_HANDOFF/tasks/AUTO-<seq>.md`) — proposals for the normal
|
|
5
|
+
// executor + reviewer + tests chain, never code writes or commits from
|
|
6
|
+
// self-improve itself. Whitelist-only fix classes; anything outside the
|
|
7
|
+
// deterministic mapping is `skipped`.
|
|
8
|
+
//
|
|
9
|
+
// Contracts:
|
|
10
|
+
// * NEVER THROWS — missing dirs, unreadable files, write failures →
|
|
11
|
+
// partial result, no exception escapes.
|
|
12
|
+
// * Eligible: count >= minCount (default 3) AND sessions >= 2.
|
|
13
|
+
// * Whitelist fix classes only (timeout/hang, missing-fallback,
|
|
14
|
+
// empty-output, stale-doc); unknown shapes → status 'skipped'.
|
|
15
|
+
// * Denylist: targets under src/decision/, src/core/agentRuntime/,
|
|
16
|
+
// runtimeConfig, auth/security/secret/credential/token code are never
|
|
17
|
+
// proposed.
|
|
18
|
+
// * Dedupe: a signature already carried by any `.md` in
|
|
19
|
+
// docs/AI_HANDOFF/tasks/ or docs/AI_HANDOFF/archive/ → 'duplicate'.
|
|
20
|
+
// * Cap: at most maxTasks (default 3) AUTO files per pass.
|
|
21
|
+
// * Writes are confined to docs/AI_HANDOFF/tasks/AUTO-<seq>.md via
|
|
22
|
+
// exclusive-create (`wx`); `dryRun` writes nothing.
|
|
23
|
+
|
|
24
|
+
import fs from 'node:fs/promises';
|
|
25
|
+
import path from 'node:path';
|
|
26
|
+
|
|
27
|
+
const TASKS_DIR_REL = path.join('docs', 'AI_HANDOFF', 'tasks');
|
|
28
|
+
const ARCHIVE_DIR_REL = path.join('docs', 'AI_HANDOFF', 'archive');
|
|
29
|
+
const ARTIFACT_REL = path.join('.ukit', 'storage', 'cache', 'failure-patterns.json');
|
|
30
|
+
const PLAN_REL = 'docs/plans/SELF_IMPROVE_CODE_LANE.md';
|
|
31
|
+
const SIGNATURE_FIELD_RE = /signature["']?\s*:\s*["'`]?([^\n"'`,}]+)/gi;
|
|
32
|
+
const MIN_SESSIONS = 2;
|
|
33
|
+
const AUTO_NAME_RE = /^AUTO-(\d+)\.md$/;
|
|
34
|
+
|
|
35
|
+
// Protected lanes — the same areas keepMainModelFor guards. A mined pattern
|
|
36
|
+
// pointing here is skipped, never proposed.
|
|
37
|
+
const DENIED_TARGET_RES = [
|
|
38
|
+
/^src\/decision\//,
|
|
39
|
+
/^src\/core\/agentRuntime\//,
|
|
40
|
+
/^src\/core\/runtimeConfig\.js$/,
|
|
41
|
+
/^src\/security\//,
|
|
42
|
+
/(^|\/)auth/i,
|
|
43
|
+
/secret|credential|token/i,
|
|
44
|
+
];
|
|
45
|
+
|
|
46
|
+
// Whitelist fix classes — bounded deterministic mapping only.
|
|
47
|
+
const FIX_CLASSES = [
|
|
48
|
+
{
|
|
49
|
+
id: 'timeout-hang',
|
|
50
|
+
matches: (hay) => /timeout|timed?\s*out|hang|hung/.test(hay),
|
|
51
|
+
change: 'Increase the timeout, add an explicit deadline, and retry once before giving up.',
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
id: 'missing-fallback',
|
|
55
|
+
matches: (hay) => /missing-fallback/.test(hay)
|
|
56
|
+
|| (/fallback/.test(hay) && /decision|decisionruntime/.test(hay)),
|
|
57
|
+
change: 'Add a `fallbackCode` path plus a deterministic default so the caller never blocks on the decision call.',
|
|
58
|
+
},
|
|
59
|
+
{
|
|
60
|
+
id: 'empty-output',
|
|
61
|
+
matches: (hay) => /empty[-_ ]?output|no output|empty stdout/.test(hay),
|
|
62
|
+
change: 'Keep the chain alive on empty output (`|| true` or equivalent) and log stderr for diagnosis.',
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
id: 'stale-doc',
|
|
66
|
+
matches: (hay) => /stale[-_ ]?doc|doc drift|docs?\s+drift|out[- ]?of[- ]?date\s+doc/.test(hay),
|
|
67
|
+
change: 'Re-render or refresh the flagged documentation file to match current source.',
|
|
68
|
+
},
|
|
69
|
+
];
|
|
70
|
+
|
|
71
|
+
function isObject(value) {
|
|
72
|
+
return Boolean(value) && typeof value === 'object' && !Array.isArray(value);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// Evidence haystack: signature + example commands + files, lowercased.
|
|
76
|
+
function haystack(pattern) {
|
|
77
|
+
return [
|
|
78
|
+
pattern.signature,
|
|
79
|
+
...(Array.isArray(pattern.commands) ? pattern.commands : []),
|
|
80
|
+
...(Array.isArray(pattern.files) ? pattern.files : []),
|
|
81
|
+
].join(' ').toLowerCase();
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Map a mined pattern to a whitelisted fix class.
|
|
86
|
+
* @returns {{id: string, change: string}|null}
|
|
87
|
+
*/
|
|
88
|
+
export function mapPatternToFixTarget(pattern) {
|
|
89
|
+
if (!isObject(pattern)) return null;
|
|
90
|
+
const hay = haystack(pattern);
|
|
91
|
+
for (const fixClass of FIX_CLASSES) {
|
|
92
|
+
if (fixClass.matches(hay)) return { id: fixClass.id, change: fixClass.change };
|
|
93
|
+
}
|
|
94
|
+
return null;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function normalizeTargets(pattern) {
|
|
98
|
+
const files = Array.isArray(pattern?.files)
|
|
99
|
+
? pattern.files.filter((f) => typeof f === 'string' && f.trim())
|
|
100
|
+
: [];
|
|
101
|
+
return files.map((f) => f.replace(/\\/g, '/'));
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
async function loadPatterns(projectRoot) {
|
|
105
|
+
try {
|
|
106
|
+
const raw = await fs.readFile(path.join(projectRoot, ARTIFACT_REL), 'utf8');
|
|
107
|
+
const parsed = JSON.parse(raw);
|
|
108
|
+
if (parsed && Array.isArray(parsed.patterns)) return parsed.patterns;
|
|
109
|
+
} catch {
|
|
110
|
+
// Artifact absent/corrupt → lazy re-mine below.
|
|
111
|
+
}
|
|
112
|
+
try {
|
|
113
|
+
const mod = await import('../diagnostics/failurePatterns.js');
|
|
114
|
+
if (typeof mod?.mineFailurePatterns !== 'function') return [];
|
|
115
|
+
const mined = await mod.mineFailurePatterns(projectRoot, { minCount: 1 });
|
|
116
|
+
return Array.isArray(mined?.patterns) ? mined.patterns : [];
|
|
117
|
+
} catch {
|
|
118
|
+
return [];
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// Bounded recursive .md listing; missing dirs → [].
|
|
123
|
+
async function listMarkdownFiles(dir, depth = 0) {
|
|
124
|
+
if (depth > 4) return [];
|
|
125
|
+
let entries;
|
|
126
|
+
try {
|
|
127
|
+
entries = await fs.readdir(dir, { withFileTypes: true });
|
|
128
|
+
} catch {
|
|
129
|
+
return [];
|
|
130
|
+
}
|
|
131
|
+
const files = [];
|
|
132
|
+
for (const entry of entries) {
|
|
133
|
+
const full = path.join(dir, entry.name);
|
|
134
|
+
if (entry.isDirectory()) {
|
|
135
|
+
files.push(...await listMarkdownFiles(full, depth + 1));
|
|
136
|
+
} else if (entry.isFile() && entry.name.endsWith('.md')) {
|
|
137
|
+
files.push(full);
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
return files;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// Signatures already claimed by a task/archived file + highest AUTO seq.
|
|
144
|
+
async function scanHandoffDirs(projectRoot) {
|
|
145
|
+
const signatures = new Set();
|
|
146
|
+
let maxSeq = 0;
|
|
147
|
+
for (const rel of [TASKS_DIR_REL, ARCHIVE_DIR_REL]) {
|
|
148
|
+
const dir = path.join(projectRoot, rel);
|
|
149
|
+
for (const file of await listMarkdownFiles(dir)) {
|
|
150
|
+
const seqMatch = AUTO_NAME_RE.exec(path.basename(file));
|
|
151
|
+
if (seqMatch) maxSeq = Math.max(maxSeq, Number(seqMatch[1]));
|
|
152
|
+
let text;
|
|
153
|
+
try {
|
|
154
|
+
text = await fs.readFile(file, 'utf8');
|
|
155
|
+
} catch {
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
SIGNATURE_FIELD_RE.lastIndex = 0;
|
|
159
|
+
for (const match of text.matchAll(SIGNATURE_FIELD_RE)) {
|
|
160
|
+
const sig = match[1].trim();
|
|
161
|
+
if (sig) signatures.add(sig);
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
return { signatures, maxSeq };
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
function renderTaskFile({ seq, pattern, fixClass, targets, now }) {
|
|
169
|
+
const id = `AUTO-${String(seq).padStart(3, '0')}`;
|
|
170
|
+
const signature = pattern.signature;
|
|
171
|
+
const files = targets.length > 0 ? targets : ['(determine from evidence block)'];
|
|
172
|
+
const evidence = JSON.stringify(pattern, null, 2);
|
|
173
|
+
return `# ${id} — self-improve: ${fixClass.id} fix for "${signature}"
|
|
174
|
+
|
|
175
|
+
- Status: \`ready\`
|
|
176
|
+
- Size: \`S\`
|
|
177
|
+
- Owner: \`-\`
|
|
178
|
+
- Reviewer: \`-\`
|
|
179
|
+
- Parent plan: \`${PLAN_REL}\`
|
|
180
|
+
- Spec references: \`docs/AI_HANDOFF/SPEC.md\` §self-improve
|
|
181
|
+
- Supersedes: \`(none)\`
|
|
182
|
+
- Superseded by: \`(none)\`
|
|
183
|
+
|
|
184
|
+
## Goal
|
|
185
|
+
|
|
186
|
+
Recurring verification failure pattern \`"${signature}"\` — signature: \`${signature}\` —
|
|
187
|
+
seen ${pattern.count}× across ${pattern.sessions} sessions. Apply the whitelisted
|
|
188
|
+
**${fixClass.id}** fix: ${fixClass.change}
|
|
189
|
+
|
|
190
|
+
## Target Files
|
|
191
|
+
|
|
192
|
+
${files.map((f) => `- \`${f}\` — apply the ${fixClass.id} fix`).join('\n')}
|
|
193
|
+
|
|
194
|
+
## Evidence (frozen — do not re-derive)
|
|
195
|
+
|
|
196
|
+
\`\`\`json
|
|
197
|
+
${evidence}
|
|
198
|
+
\`\`\`
|
|
199
|
+
|
|
200
|
+
## Progress
|
|
201
|
+
|
|
202
|
+
## Test Cases (REQUIRED — TDD)
|
|
203
|
+
|
|
204
|
+
| # | Loại | Tên test | Expected | Pre-state / Fixture |
|
|
205
|
+
|---|------|----------|----------|---------------------|
|
|
206
|
+
| 1 | unit | pattern signature "${signature}" repro | failing path exercised | fixture from Evidence block |
|
|
207
|
+
| 2 | regression | fix holds | GREEN after patch | same fixture |
|
|
208
|
+
|
|
209
|
+
## Test Files
|
|
210
|
+
|
|
211
|
+
- (executor picks the closest existing test file for the target)
|
|
212
|
+
|
|
213
|
+
## Verification Commands
|
|
214
|
+
|
|
215
|
+
\`\`\`bash
|
|
216
|
+
yarn test <targeted test file>
|
|
217
|
+
\`\`\`
|
|
218
|
+
|
|
219
|
+
## Acceptance Criteria
|
|
220
|
+
|
|
221
|
+
- [ ] Targeted test green (RED → GREEN evidence in Executor Report).
|
|
222
|
+
- [ ] No regression in the related suite.
|
|
223
|
+
- [ ] Reviewer verdict APPROVED or APPROVED-WITH-MINOR.
|
|
224
|
+
|
|
225
|
+
## Dependencies
|
|
226
|
+
|
|
227
|
+
- (none)
|
|
228
|
+
|
|
229
|
+
## Interfaces
|
|
230
|
+
|
|
231
|
+
- Consumes: (none)
|
|
232
|
+
- Produces: (none)
|
|
233
|
+
|
|
234
|
+
---
|
|
235
|
+
|
|
236
|
+
## Discussion
|
|
237
|
+
|
|
238
|
+
### ${new Date(now).toISOString()} · planner · ukit self-improve
|
|
239
|
+
Auto-generated by \`ukit self-improve\` at ${new Date(now).toISOString()} from mined failure-patterns artifact.
|
|
240
|
+
|
|
241
|
+
<!-- auto-proposal: do-not-approve-without-human-review -->
|
|
242
|
+
`;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
/**
|
|
246
|
+
* Propose AUTO handoff task files from mined failure patterns.
|
|
247
|
+
*
|
|
248
|
+
* @param {string} projectRoot
|
|
249
|
+
* @param {{ minCount?: number, maxTasks?: number, dryRun?: boolean,
|
|
250
|
+
* now?: number }} [options]
|
|
251
|
+
* @returns {Promise<{generatedAt: string, dryRun: boolean,
|
|
252
|
+
* patternsScanned: number, eligible: number, created: number,
|
|
253
|
+
* results: Array<{signature: string, fixClass: string|null,
|
|
254
|
+
* status: 'proposed'|'duplicate'|'skipped', reason?: string,
|
|
255
|
+
* taskFile: string|null, written: boolean}>, error?: string}>}
|
|
256
|
+
*/
|
|
257
|
+
export async function proposeCodeFixTasks(projectRoot, {
|
|
258
|
+
minCount = 3,
|
|
259
|
+
maxTasks = 3,
|
|
260
|
+
dryRun = true,
|
|
261
|
+
now = Date.now(),
|
|
262
|
+
} = {}) {
|
|
263
|
+
const result = {
|
|
264
|
+
generatedAt: new Date(now).toISOString(),
|
|
265
|
+
dryRun,
|
|
266
|
+
patternsScanned: 0,
|
|
267
|
+
eligible: 0,
|
|
268
|
+
created: 0,
|
|
269
|
+
results: [],
|
|
270
|
+
};
|
|
271
|
+
|
|
272
|
+
try {
|
|
273
|
+
const patterns = await loadPatterns(projectRoot);
|
|
274
|
+
result.patternsScanned = patterns.length;
|
|
275
|
+
const known = await scanHandoffDirs(projectRoot);
|
|
276
|
+
let seq = known.maxSeq;
|
|
277
|
+
let emitted = 0;
|
|
278
|
+
|
|
279
|
+
for (const pattern of patterns) {
|
|
280
|
+
const signature = typeof pattern?.signature === 'string' ? pattern.signature : null;
|
|
281
|
+
if (!signature) continue;
|
|
282
|
+
const entry = {
|
|
283
|
+
signature, fixClass: null, status: 'skipped', taskFile: null, written: false,
|
|
284
|
+
};
|
|
285
|
+
result.results.push(entry);
|
|
286
|
+
|
|
287
|
+
const count = Number(pattern?.count) || 0;
|
|
288
|
+
const sessions = Number(pattern?.sessions) || 0;
|
|
289
|
+
if (count < minCount || sessions < MIN_SESSIONS) {
|
|
290
|
+
entry.reason = 'ineligible';
|
|
291
|
+
continue;
|
|
292
|
+
}
|
|
293
|
+
result.eligible += 1;
|
|
294
|
+
|
|
295
|
+
const fixClass = mapPatternToFixTarget(pattern);
|
|
296
|
+
if (!fixClass) {
|
|
297
|
+
entry.reason = 'no-fix-class';
|
|
298
|
+
continue;
|
|
299
|
+
}
|
|
300
|
+
entry.fixClass = fixClass.id;
|
|
301
|
+
|
|
302
|
+
const targets = normalizeTargets(pattern);
|
|
303
|
+
if (targets.some((f) => DENIED_TARGET_RES.some((re) => re.test(f)))) {
|
|
304
|
+
entry.reason = 'denied-target';
|
|
305
|
+
continue;
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
if (known.signatures.has(signature)) {
|
|
309
|
+
entry.status = 'duplicate';
|
|
310
|
+
continue;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
if (emitted >= maxTasks) {
|
|
314
|
+
entry.reason = 'cap';
|
|
315
|
+
continue;
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
seq += 1;
|
|
319
|
+
const fileName = `AUTO-${String(seq).padStart(3, '0')}.md`;
|
|
320
|
+
entry.taskFile = path.join(TASKS_DIR_REL, fileName);
|
|
321
|
+
entry.status = 'proposed';
|
|
322
|
+
emitted += 1;
|
|
323
|
+
result.created += 1;
|
|
324
|
+
known.signatures.add(signature);
|
|
325
|
+
|
|
326
|
+
if (dryRun) continue;
|
|
327
|
+
|
|
328
|
+
try {
|
|
329
|
+
await fs.mkdir(path.join(projectRoot, TASKS_DIR_REL), { recursive: true });
|
|
330
|
+
await fs.writeFile(
|
|
331
|
+
path.join(projectRoot, TASKS_DIR_REL, fileName),
|
|
332
|
+
renderTaskFile({ seq, pattern, fixClass, targets, now }),
|
|
333
|
+
{ flag: 'wx' },
|
|
334
|
+
);
|
|
335
|
+
entry.written = true;
|
|
336
|
+
} catch (err) {
|
|
337
|
+
entry.status = 'skipped';
|
|
338
|
+
entry.reason = `write-failed: ${err?.message ?? String(err)}`;
|
|
339
|
+
entry.written = false;
|
|
340
|
+
result.created -= 1;
|
|
341
|
+
emitted -= 1;
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
} catch (err) {
|
|
345
|
+
result.error = err?.message ?? String(err);
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
return result;
|
|
349
|
+
}
|
|
@@ -9,14 +9,22 @@
|
|
|
9
9
|
// Steps (each isolated: one failing step never skips the others):
|
|
10
10
|
// 1. episodes — backfill an episode record for every idle exec-ledger
|
|
11
11
|
// 2. diagnostics — failure patterns, feedback events, skill accuracy
|
|
12
|
-
// 3.
|
|
12
|
+
// 3. vmShadow — when decisionRuntime.vm stage ≥ 'shadow', one shadow-run
|
|
13
|
+
// per reference plan via the agent-VM driver (D-04); off →
|
|
14
|
+
// skipped, any failure → status 'error', never throws
|
|
15
|
+
// 4. tuning — compute suggestions; applyMode 'auto' applies them as
|
|
13
16
|
// clamped one-step changes in learning/tuned.json, with a
|
|
14
17
|
// per-key cooldown so one noisy window cannot ratchet a
|
|
15
18
|
// value across its whole range
|
|
16
|
-
//
|
|
19
|
+
// 5. proposals — repeated failure patterns become PENDING memory
|
|
17
20
|
// candidates (rules still need `ukit memory approve`)
|
|
18
|
-
//
|
|
21
|
+
// 6. telemetry — ingest stored hook/ledger telemetry, flush, refresh the
|
|
19
22
|
// support view
|
|
23
|
+
// 7. codeProposals — eligible mined patterns become AUTO-<seq> handoff task
|
|
24
|
+
// files (docs/AI_HANDOFF/tasks/); whitelist fix classes
|
|
25
|
+
// only, ≤3/pass, signature-deduped, denylisted targets
|
|
26
|
+
// skipped. Proposals are task FILES — self-improve never
|
|
27
|
+
// writes code or commits (SELF_IMPROVE_CODE_LANE plan)
|
|
20
28
|
//
|
|
21
29
|
// Guards: `.ukit/storage/learning/self-improve.json` stamp rate-limits the
|
|
22
30
|
// pass (minIntervalMs, default 10 min) and a file lock keeps two triggers from
|
|
@@ -26,7 +34,7 @@ import fs from 'node:fs/promises';
|
|
|
26
34
|
import os from 'node:os';
|
|
27
35
|
import path from 'node:path';
|
|
28
36
|
|
|
29
|
-
import { inspectRuntimeConfig } from '../core/runtimeConfig.js';
|
|
37
|
+
import { inspectRuntimeConfig, resolveDecisionRuntimeStage } from '../core/runtimeConfig.js';
|
|
30
38
|
import { withFileLock } from '../core/fileOps.js';
|
|
31
39
|
import { detectProjectContext } from '../context/detectProjectContext.js';
|
|
32
40
|
import { backfillEpisodes } from '../core/memory/episodes.js';
|
|
@@ -35,6 +43,7 @@ import { collectFeedbackEvents } from '../diagnostics/feedbackEvents.js';
|
|
|
35
43
|
import { collectSkillAccuracy } from '../diagnostics/skillAccuracy.js';
|
|
36
44
|
import { computeTuningSuggestions } from './tuning.js';
|
|
37
45
|
import { proposeFromPatterns } from './patternProposals.js';
|
|
46
|
+
import { proposeCodeFixTasks } from './codeProposals.js';
|
|
38
47
|
import {
|
|
39
48
|
TUNABLE_KEYS,
|
|
40
49
|
clampTunable,
|
|
@@ -133,7 +142,12 @@ async function collectTelemetry(projectRoot, config) {
|
|
|
133
142
|
/**
|
|
134
143
|
* Run one self-improve pass.
|
|
135
144
|
* @param {string} projectRoot
|
|
136
|
-
* @param {{ force?: boolean, homeDir?: string, now?: number, minIntervalMs?: number
|
|
145
|
+
* @param {{ force?: boolean, homeDir?: string, now?: number, minIntervalMs?: number,
|
|
146
|
+
* runShadowPlan?: Function, codeProposalsOpts?: { dryRun?: boolean } }} [opts] —
|
|
147
|
+
* runShadowPlan overrides the agentRuntime shadow-run driver (tests); absent →
|
|
148
|
+
* dynamic import of agentRuntime/shadowRun.js (import failure → vmShadow step
|
|
149
|
+
* 'error'). codeProposalsOpts.dryRun:false lets the codeProposals step write
|
|
150
|
+
* AUTO task files; anything else keeps it dry-run.
|
|
137
151
|
* @returns {Promise<{status: 'ran'|'skipped', reason?: string, steps?: object[]}>}
|
|
138
152
|
*/
|
|
139
153
|
export async function runSelfImprove(projectRoot, {
|
|
@@ -141,6 +155,8 @@ export async function runSelfImprove(projectRoot, {
|
|
|
141
155
|
homeDir = os.homedir(),
|
|
142
156
|
now = Date.now(),
|
|
143
157
|
minIntervalMs = DEFAULT_MIN_INTERVAL_MS,
|
|
158
|
+
runShadowPlan,
|
|
159
|
+
codeProposalsOpts,
|
|
144
160
|
} = {}) {
|
|
145
161
|
const root = path.resolve(projectRoot);
|
|
146
162
|
const stampPath = path.join(root, STAMP_REL);
|
|
@@ -172,6 +188,43 @@ export async function runSelfImprove(projectRoot, {
|
|
|
172
188
|
skills: Object.keys(skills?.skills ?? {}).length,
|
|
173
189
|
};
|
|
174
190
|
}));
|
|
191
|
+
steps.push(await step('vmShadow', async () => {
|
|
192
|
+
if (resolveDecisionRuntimeStage(config, 'vm') === 'off') {
|
|
193
|
+
return { status: 'skipped', reason: 'stage_off' };
|
|
194
|
+
}
|
|
195
|
+
try {
|
|
196
|
+
const runner = runShadowPlan
|
|
197
|
+
?? (await import('../core/agentRuntime/shadowRun.js')).runShadowPlan;
|
|
198
|
+
const { listPlanIds } = await import('../core/agentRuntime/planLibrary.js');
|
|
199
|
+
const runs = [];
|
|
200
|
+
for (const planId of listPlanIds()) {
|
|
201
|
+
let res;
|
|
202
|
+
try {
|
|
203
|
+
res = await runner(planId, { projectRoot: root, config });
|
|
204
|
+
} catch (error) {
|
|
205
|
+
res = { ok: false, code: 'driver_threw', reason: error?.message ?? String(error) };
|
|
206
|
+
}
|
|
207
|
+
runs.push({
|
|
208
|
+
planId,
|
|
209
|
+
ok: res?.ok === true,
|
|
210
|
+
code: res?.code ?? null,
|
|
211
|
+
transitions: Array.isArray(res?.transitions) ? res.transitions.length : 0,
|
|
212
|
+
wallMs: res?.wallMs ?? null,
|
|
213
|
+
irHash: res?.irHash ?? null,
|
|
214
|
+
});
|
|
215
|
+
}
|
|
216
|
+
const failed = runs.filter((r) => !r.ok).length;
|
|
217
|
+
return {
|
|
218
|
+
status: failed === 0 ? 'ok' : 'error',
|
|
219
|
+
ran: runs.length - failed,
|
|
220
|
+
failed,
|
|
221
|
+
runs,
|
|
222
|
+
};
|
|
223
|
+
} catch (error) {
|
|
224
|
+
// Driver not built yet or plansDir unreadable: report, never throw.
|
|
225
|
+
return { status: 'error', reason: error?.message ?? String(error) };
|
|
226
|
+
}
|
|
227
|
+
}));
|
|
175
228
|
steps.push(await step('tuning', async () => {
|
|
176
229
|
const result = await computeTuningSuggestions(root);
|
|
177
230
|
const mode = resolveApplyMode(rawConfig ?? {});
|
|
@@ -188,7 +241,20 @@ export async function runSelfImprove(projectRoot, {
|
|
|
188
241
|
reason: res.error,
|
|
189
242
|
};
|
|
190
243
|
}));
|
|
244
|
+
|
|
191
245
|
steps.push(await step('telemetry', () => collectTelemetry(root, config)));
|
|
246
|
+
steps.push(await step('codeProposals', async () => {
|
|
247
|
+
if (config?.learning?.selfImprove?.codeProposals === false) {
|
|
248
|
+
return { status: 'skipped', reason: 'disabled' };
|
|
249
|
+
}
|
|
250
|
+
const res = await proposeCodeFixTasks(root, { dryRun: codeProposalsOpts?.dryRun !== false });
|
|
251
|
+
return {
|
|
252
|
+
status: res.error ? 'failed' : 'ok',
|
|
253
|
+
created: res.created,
|
|
254
|
+
results: res.results,
|
|
255
|
+
reason: res.error,
|
|
256
|
+
};
|
|
257
|
+
}));
|
|
192
258
|
|
|
193
259
|
await writeJsonAtomic(stampPath, {
|
|
194
260
|
lastRunAt: new Date(now).toISOString(),
|