@ngockhoale/ukit 3.1.9 → 3.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/package.json +1 -1
- package/src/cli/commands/memory.js +11 -87
- package/src/cli/commands/selfImprove.js +55 -0
- package/src/cli/commands/telemetry.js +45 -3
- package/src/cli/index.js +7 -0
- package/src/core/agentRuntime/adapters.js +177 -0
- package/src/core/agentRuntime/contract.js +77 -0
- package/src/core/agentRuntime/diagnostics.js +343 -42
- package/src/core/agentRuntime/eventStore.js +139 -0
- package/src/core/agentRuntime/planCompiler.js +45 -6
- package/src/core/agentRuntime/planLibrary.js +43 -7
- package/src/core/agentRuntime/plans/bugfix-loop.json +1 -0
- package/src/core/agentRuntime/plans/flag-promotion.json +138 -0
- package/src/core/agentRuntime/plans/handoff-review-batch.json +1 -0
- package/src/core/agentRuntime/plans/release-check.json +2 -1
- package/src/core/agentRuntime/promotion.js +147 -9
- package/src/core/agentRuntime/runtimeSupport.js +18 -0
- package/src/core/agentRuntime/supervisor.js +44 -2
- package/src/core/agentRuntime/telemetry.js +123 -0
- package/src/core/agentRuntime/vmEngine.js +51 -10
- package/src/core/memory/episodes.js +168 -0
- package/src/core/metadata.js +21 -0
- package/src/core/runInstallPipeline.js +6 -1
- package/src/core/runtimeConfig.js +71 -40
- package/src/decision/client.js +4 -5
- package/src/decision/runtimeDecide.js +183 -5
- package/src/learning/selfImprove.js +205 -0
- package/src/learning/tunedOverlay.js +112 -0
- package/template_project/.claude/hooks/session-episode.sh +35 -15
- package/template_project/.claude/ukit/index/route-task.mjs +13 -0
- package/template_project/.claude/ukit/index/unic-decision.mjs +1 -2
- package/template_project/.claude/ukit/runtime/self-improve-trigger.mjs +98 -0
- package/template_project/ukit/storage/config.json +68 -17
package/src/decision/client.js
CHANGED
|
@@ -49,9 +49,9 @@ const DEFAULT_CHECKPOINTS = {
|
|
|
49
49
|
english: 'unic-decision',
|
|
50
50
|
unknownLanguage: 'unic-decision',
|
|
51
51
|
};
|
|
52
|
-
// Credential-safe alias: every UNIC decision credential binds the bare name
|
|
53
|
-
//
|
|
54
|
-
// the
|
|
52
|
+
// Credential-safe alias: every UNIC decision credential binds the bare name.
|
|
53
|
+
// On model_not_found the client falls back here once per batch — the bare
|
|
54
|
+
// name is also the only checkpoint name UKit ever sends.
|
|
55
55
|
const FALLBACK_CHECKPOINT = 'unic-decision';
|
|
56
56
|
|
|
57
57
|
const DEFAULT_TIMEOUT_MS = 5000;
|
|
@@ -360,8 +360,7 @@ export function createDecisionClient({
|
|
|
360
360
|
|
|
361
361
|
const status = res?.status ?? (res?.ok === false ? 500 : 200);
|
|
362
362
|
if (res?.ok === false || status < 200 || status >= 300) {
|
|
363
|
-
// A
|
|
364
|
-
// unic-decision-multilingual before provider binding) returns a 4xx
|
|
363
|
+
// A checkpoint without bound credentials returns a 4xx
|
|
365
364
|
// model_not_found body — retry once on the bare `unic-decision`
|
|
366
365
|
// model, the credential-safe alias the owner provisions first.
|
|
367
366
|
if (checkpoint !== FALLBACK_CHECKPOINT) {
|
|
@@ -83,6 +83,46 @@ export const RUNTIME_DECISIONS = Object.freeze([
|
|
|
83
83
|
'A runtime event may warrant a retry but no deterministic policy covers it. Decide conservatively.',
|
|
84
84
|
candidates: ['retry', 'no_retry', 'escalate'],
|
|
85
85
|
},
|
|
86
|
+
// V-05: VM decision nodes — the bounded route/classify questions a
|
|
87
|
+
// vmEngine classifyFn may ask when the plan IR cannot route an event.
|
|
88
|
+
// Owner is the deterministic escalation lane: every failure mode resolves
|
|
89
|
+
// to escalateUnclassified, never a hard failure.
|
|
90
|
+
{
|
|
91
|
+
decisionKey: 'runtime.node_route.v1',
|
|
92
|
+
schemaVersion: 1,
|
|
93
|
+
family: 'runtime',
|
|
94
|
+
owner: 'vmEngine',
|
|
95
|
+
kind: 'choice',
|
|
96
|
+
description: 'Outcome for a VM node event the plan IR could not route — bounded route/escalate answer only.',
|
|
97
|
+
candidatePolicy: 'node-outcomes-plus-escalate',
|
|
98
|
+
hardConstraints: ['data-only-no-tool-dispatch'],
|
|
99
|
+
probabilityPolicy: 'raw-label',
|
|
100
|
+
fallbackPolicy: 'frozen-conservative-fallback',
|
|
101
|
+
telemetryClass: 'decision',
|
|
102
|
+
rolloutStage: 'off',
|
|
103
|
+
cacheSensitivity: 'none',
|
|
104
|
+
instruction:
|
|
105
|
+
'A VM node event did not match any plan selector. Choose the node outcome, or escalate to the deterministic lane.',
|
|
106
|
+
candidates: ['completed', 'failed', 'cancelled', 'escalate'],
|
|
107
|
+
},
|
|
108
|
+
{
|
|
109
|
+
decisionKey: 'runtime.node_classify.v1',
|
|
110
|
+
schemaVersion: 1,
|
|
111
|
+
family: 'runtime',
|
|
112
|
+
owner: 'vmEngine',
|
|
113
|
+
kind: 'choice',
|
|
114
|
+
description: 'Bounded node-declared classification for an unclassifiable VM event; candidates come from the node question block.',
|
|
115
|
+
candidatePolicy: 'node-declared-candidates',
|
|
116
|
+
hardConstraints: ['data-only-no-tool-dispatch'],
|
|
117
|
+
probabilityPolicy: 'raw-label',
|
|
118
|
+
fallbackPolicy: 'frozen-conservative-fallback',
|
|
119
|
+
telemetryClass: 'decision',
|
|
120
|
+
rolloutStage: 'off',
|
|
121
|
+
cacheSensitivity: 'none',
|
|
122
|
+
instruction:
|
|
123
|
+
'Classify this unclassifiable VM node event into exactly one of the node-declared candidates.',
|
|
124
|
+
candidates: ['completed', 'failed', 'escalate'],
|
|
125
|
+
},
|
|
86
126
|
]);
|
|
87
127
|
|
|
88
128
|
// Decision→fallback action map (frozen, SPEC §4). Fallbacks are conservative:
|
|
@@ -92,6 +132,9 @@ const FALLBACK_ACTIONS = Object.freeze({
|
|
|
92
132
|
'runtime.stall_action.v1': 'continue_observe',
|
|
93
133
|
'runtime.wake_needed.v1': 'escalate',
|
|
94
134
|
'runtime.safe_retry.v1': 'no_retry',
|
|
135
|
+
// VM decision nodes fail closed onto the deterministic escalation lane.
|
|
136
|
+
'runtime.node_route.v1': 'escalate',
|
|
137
|
+
'runtime.node_classify.v1': 'escalate',
|
|
95
138
|
});
|
|
96
139
|
|
|
97
140
|
// Batch-level statuses → fallback codes. 'abstained' stays distinct so
|
|
@@ -135,22 +178,37 @@ function buildStatePacket(item) {
|
|
|
135
178
|
decisionKey: item.decisionKey ?? null,
|
|
136
179
|
reason: item.reason ?? null,
|
|
137
180
|
eventType: item.event?.eventType ?? item.eventType ?? null,
|
|
181
|
+
nodeId: typeof item.nodeId === 'string' ? item.nodeId : null,
|
|
182
|
+
planInstanceId:
|
|
183
|
+
typeof item.planInstanceId === 'string' ? item.planInstanceId : null,
|
|
138
184
|
};
|
|
139
185
|
}
|
|
140
186
|
|
|
141
187
|
function buildQuestion(item, entry) {
|
|
188
|
+
// A node-declared question block may override the registered instruction/
|
|
189
|
+
// candidates — it is still a bounded choice against one registered key.
|
|
190
|
+
const instruction =
|
|
191
|
+
typeof item.instruction === 'string' && item.instruction.length > 0
|
|
192
|
+
? item.instruction
|
|
193
|
+
: typeof entry.instruction === 'string' && entry.instruction.length > 0
|
|
194
|
+
? entry.instruction
|
|
195
|
+
: `Decide ${entry.decisionKey} for the described runtime event.`;
|
|
196
|
+
const candidates =
|
|
197
|
+
Array.isArray(item.candidates) && item.candidates.length > 0
|
|
198
|
+
? item.candidates
|
|
199
|
+
: Array.isArray(entry.candidates)
|
|
200
|
+
? entry.candidates
|
|
201
|
+
: [];
|
|
142
202
|
return {
|
|
143
203
|
decisionKey: entry.decisionKey,
|
|
144
204
|
schemaVersion: entry.schemaVersion ?? 1,
|
|
145
205
|
kind: entry.kind,
|
|
146
|
-
instruction
|
|
147
|
-
|
|
148
|
-
? entry.instruction
|
|
149
|
-
: `Decide ${entry.decisionKey} for the described runtime event.`,
|
|
150
|
-
candidates: Array.isArray(entry.candidates) ? entry.candidates : [],
|
|
206
|
+
instruction,
|
|
207
|
+
candidates,
|
|
151
208
|
};
|
|
152
209
|
}
|
|
153
210
|
|
|
211
|
+
|
|
154
212
|
/**
|
|
155
213
|
* @param {{client: {requestBatch: Function}, registry?: object,
|
|
156
214
|
* confidenceFloor?: number|object, now?: Function}} options
|
|
@@ -240,3 +298,123 @@ export function createRuntimeDecider({
|
|
|
240
298
|
|
|
241
299
|
return { decide };
|
|
242
300
|
}
|
|
301
|
+
|
|
302
|
+
// ---------------------------------------------------------------------------
|
|
303
|
+
// V-05 — VM decision nodes (agent-vm-runtime V5): a classifyFn for vmEngine
|
|
304
|
+
// that asks one bounded runtime.node_route.v1 / runtime.node_classify.v1
|
|
305
|
+
// question per unclassifiable event. The model answer is data only — the
|
|
306
|
+
// client's tool_calls envelope is consumed as an answer list, NEVER
|
|
307
|
+
// dispatched. Every failure mode maps to the deterministic escalation lane
|
|
308
|
+
// with a fallbackCode; this adapter never throws.
|
|
309
|
+
|
|
310
|
+
// Registry keys a VM decision node may ask — anything else resolves to the
|
|
311
|
+
// deterministic escalation lane without touching the gateway.
|
|
312
|
+
export const VM_NODE_DECISION_KEYS = Object.freeze([
|
|
313
|
+
'runtime.node_route.v1',
|
|
314
|
+
'runtime.node_classify.v1',
|
|
315
|
+
]);
|
|
316
|
+
const DEFAULT_NODE_DECISION_KEY = VM_NODE_DECISION_KEYS[0];
|
|
317
|
+
|
|
318
|
+
// Node question-block bounds — keeps a bounded choice small enough to stay
|
|
319
|
+
// inside the state/question budget.
|
|
320
|
+
const MAX_NODE_CANDIDATES = 8;
|
|
321
|
+
const MAX_NODE_QUESTION_TEXT = 240;
|
|
322
|
+
|
|
323
|
+
function isShortString(value) {
|
|
324
|
+
return typeof value === 'string'
|
|
325
|
+
&& value.length > 0
|
|
326
|
+
&& value.length <= MAX_NODE_QUESTION_TEXT;
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
// A node question block is valid only when every declared field is
|
|
330
|
+
// well-formed; a malformed block is rejected wholesale (never partially
|
|
331
|
+
// honored) so a bad plan cannot smuggle an unbounded question.
|
|
332
|
+
function validNodeQuestion(question) {
|
|
333
|
+
return isPlainObject(question)
|
|
334
|
+
&& (question.decisionKey === undefined || isShortString(question.decisionKey))
|
|
335
|
+
&& (question.instruction === undefined || isShortString(question.instruction))
|
|
336
|
+
&& (question.candidates === undefined
|
|
337
|
+
|| (Array.isArray(question.candidates)
|
|
338
|
+
&& question.candidates.length > 0
|
|
339
|
+
&& question.candidates.length <= MAX_NODE_CANDIDATES
|
|
340
|
+
&& question.candidates.every(isShortString)));
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
/**
|
|
344
|
+
* Build a vmEngine-compatible `classifyFn(event, ctx)` backed by the decision
|
|
345
|
+
* plane. Returns only `{action:'route', outcome}` / `{action:'escalate',
|
|
346
|
+
* fallbackCode?}` verdicts — model unavailability, timeouts, malformed
|
|
347
|
+
* answers, and adapter faults all resolve to the deterministic escalation
|
|
348
|
+
* lane with a `fallbackCode`, and the function NEVER throws.
|
|
349
|
+
*
|
|
350
|
+
* @param {{client?: {requestBatch: Function}, decider?: {decide: Function},
|
|
351
|
+
* registry?: object, confidenceFloor?: number|object}} options
|
|
352
|
+
* `decider` wins over `client` (a custom decider composes its own client).
|
|
353
|
+
* Neither present → every call escalates 'unavailable' with zero I/O.
|
|
354
|
+
*/
|
|
355
|
+
export function createVmClassifyFn({
|
|
356
|
+
client,
|
|
357
|
+
decider,
|
|
358
|
+
registry,
|
|
359
|
+
confidenceFloor,
|
|
360
|
+
} = {}) {
|
|
361
|
+
const resolvedDecider =
|
|
362
|
+
decider
|
|
363
|
+
?? (client != null
|
|
364
|
+
? createRuntimeDecider({ client, registry, confidenceFloor })
|
|
365
|
+
: null);
|
|
366
|
+
|
|
367
|
+
return async function vmClassify(event, ctx = {}) {
|
|
368
|
+
try {
|
|
369
|
+
const node = isPlainObject(ctx?.node) ? ctx.node : null;
|
|
370
|
+
const question = node?.question;
|
|
371
|
+
if (question !== undefined && !validNodeQuestion(question)) {
|
|
372
|
+
return { action: 'escalate', fallbackCode: 'malformed-question' };
|
|
373
|
+
}
|
|
374
|
+
const decisionKey = question?.decisionKey ?? DEFAULT_NODE_DECISION_KEY;
|
|
375
|
+
if (!VM_NODE_DECISION_KEYS.includes(decisionKey)) {
|
|
376
|
+
return { action: 'escalate', fallbackCode: 'unregistered-key' };
|
|
377
|
+
}
|
|
378
|
+
if (!resolvedDecider || typeof resolvedDecider.decide !== 'function') {
|
|
379
|
+
return { action: 'escalate', fallbackCode: 'unavailable' };
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
const decision = await resolvedDecider.decide({
|
|
383
|
+
action: 'model_decision',
|
|
384
|
+
decisionKey,
|
|
385
|
+
reason: `vm_unrouted:${String(event?.eventType ?? 'unknown')}`,
|
|
386
|
+
event: isPlainObject(event) ? event : null,
|
|
387
|
+
nodeId: typeof ctx?.nodeId === 'string' ? ctx.nodeId : null,
|
|
388
|
+
planInstanceId:
|
|
389
|
+
typeof ctx?.planInstanceId === 'string' ? ctx.planInstanceId : null,
|
|
390
|
+
...(isShortString(question?.instruction)
|
|
391
|
+
? { instruction: question.instruction }
|
|
392
|
+
: {}),
|
|
393
|
+
...(Array.isArray(question?.candidates)
|
|
394
|
+
? { candidates: question.candidates }
|
|
395
|
+
: {}),
|
|
396
|
+
});
|
|
397
|
+
|
|
398
|
+
if (!isPlainObject(decision)) {
|
|
399
|
+
return { action: 'escalate', fallbackCode: 'invalid' };
|
|
400
|
+
}
|
|
401
|
+
if (decision.decidedBy === 'fallback') {
|
|
402
|
+
return { action: 'escalate', fallbackCode: decision.code ?? 'invalid' };
|
|
403
|
+
}
|
|
404
|
+
const outcome = decision.action ?? decision.value;
|
|
405
|
+
if (outcome === 'escalate') {
|
|
406
|
+
return { action: 'escalate', decisionKey };
|
|
407
|
+
}
|
|
408
|
+
if (typeof outcome !== 'string' || outcome.length === 0) {
|
|
409
|
+
return { action: 'escalate', fallbackCode: 'invalid' };
|
|
410
|
+
}
|
|
411
|
+
// The engine's transition table still gates the outcome — an undeclared
|
|
412
|
+
// value fails closed to recovery_required downstream, never silently.
|
|
413
|
+
return { action: 'route', outcome, decisionKey };
|
|
414
|
+
} catch {
|
|
415
|
+
// Any adapter fault (bad ctx shape, throwing decider impl) resolves to
|
|
416
|
+
// the same deterministic lane — decision nodes never hard-fail the VM.
|
|
417
|
+
return { action: 'escalate', fallbackCode: 'adapter-error' };
|
|
418
|
+
}
|
|
419
|
+
};
|
|
420
|
+
}
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
// selfImprove.js — the automatic collect → learn → apply cycle (3.3.0).
|
|
2
|
+
//
|
|
3
|
+
// One bounded pass that turns what hooks already record into memory,
|
|
4
|
+
// diagnostics, tuning, and the flight recorder — no human command needed.
|
|
5
|
+
// Triggered detached by `.claude/ukit/runtime/self-improve-trigger.mjs` from
|
|
6
|
+
// Claude SessionEnd, omp session_stop, and route-task.mjs (the Codex helper
|
|
7
|
+
// path), and runnable by hand via `ukit self-improve`.
|
|
8
|
+
//
|
|
9
|
+
// Steps (each isolated: one failing step never skips the others):
|
|
10
|
+
// 1. episodes — backfill an episode record for every idle exec-ledger
|
|
11
|
+
// 2. diagnostics — failure patterns, feedback events, skill accuracy
|
|
12
|
+
// 3. tuning — compute suggestions; applyMode 'auto' applies them as
|
|
13
|
+
// clamped one-step changes in learning/tuned.json, with a
|
|
14
|
+
// per-key cooldown so one noisy window cannot ratchet a
|
|
15
|
+
// value across its whole range
|
|
16
|
+
// 4. proposals — repeated failure patterns become PENDING memory
|
|
17
|
+
// candidates (rules still need `ukit memory approve`)
|
|
18
|
+
// 5. telemetry — ingest stored hook/ledger telemetry, flush, refresh the
|
|
19
|
+
// support view
|
|
20
|
+
//
|
|
21
|
+
// Guards: `.ukit/storage/learning/self-improve.json` stamp rate-limits the
|
|
22
|
+
// pass (minIntervalMs, default 10 min) and a file lock keeps two triggers from
|
|
23
|
+
// overlapping. Never throws.
|
|
24
|
+
|
|
25
|
+
import fs from 'node:fs/promises';
|
|
26
|
+
import os from 'node:os';
|
|
27
|
+
import path from 'node:path';
|
|
28
|
+
|
|
29
|
+
import { inspectRuntimeConfig } from '../core/runtimeConfig.js';
|
|
30
|
+
import { withFileLock } from '../core/fileOps.js';
|
|
31
|
+
import { detectProjectContext } from '../context/detectProjectContext.js';
|
|
32
|
+
import { backfillEpisodes } from '../core/memory/episodes.js';
|
|
33
|
+
import { mineFailurePatterns } from '../diagnostics/failurePatterns.js';
|
|
34
|
+
import { collectFeedbackEvents } from '../diagnostics/feedbackEvents.js';
|
|
35
|
+
import { collectSkillAccuracy } from '../diagnostics/skillAccuracy.js';
|
|
36
|
+
import { computeTuningSuggestions } from './tuning.js';
|
|
37
|
+
import { proposeFromPatterns } from './patternProposals.js';
|
|
38
|
+
import {
|
|
39
|
+
TUNABLE_KEYS,
|
|
40
|
+
clampTunable,
|
|
41
|
+
readTunedDocument,
|
|
42
|
+
resolveApplyMode,
|
|
43
|
+
writeTunedDocument,
|
|
44
|
+
} from './tunedOverlay.js';
|
|
45
|
+
|
|
46
|
+
export const STAMP_REL = path.join('.ukit', 'storage', 'learning', 'self-improve.json');
|
|
47
|
+
export const DEFAULT_MIN_INTERVAL_MS = 10 * 60 * 1000;
|
|
48
|
+
// A tuned key moves at most one step per cooldown window.
|
|
49
|
+
export const TUNING_COOLDOWN_MS = 24 * 60 * 60 * 1000;
|
|
50
|
+
const TELEMETRY_DEADLINE_MS = 30_000;
|
|
51
|
+
const FLUSH_DEADLINE_MS = 10_000;
|
|
52
|
+
|
|
53
|
+
async function readJson(filePath) {
|
|
54
|
+
try {
|
|
55
|
+
return JSON.parse(await fs.readFile(filePath, 'utf8'));
|
|
56
|
+
} catch {
|
|
57
|
+
return null;
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
async function writeJsonAtomic(filePath, value) {
|
|
62
|
+
const tmp = `${filePath}.tmp-${process.pid}`;
|
|
63
|
+
try {
|
|
64
|
+
await fs.mkdir(path.dirname(filePath), { recursive: true });
|
|
65
|
+
await fs.writeFile(tmp, `${JSON.stringify(value, null, 2)}\n`);
|
|
66
|
+
await fs.rename(tmp, filePath);
|
|
67
|
+
} catch {
|
|
68
|
+
await fs.rm(tmp, { force: true }).catch(() => {});
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
async function step(name, fn) {
|
|
73
|
+
try {
|
|
74
|
+
return { name, ...(await fn()) };
|
|
75
|
+
} catch (error) {
|
|
76
|
+
return { name, status: 'failed', reason: error?.message ?? String(error) };
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
// Applies tuning suggestions to the overlay. Pure over its inputs apart from
|
|
81
|
+
// the overlay write, so it is testable without the rest of the cycle.
|
|
82
|
+
export async function applyTuning(projectRoot, suggestions, { now = Date.now() } = {}) {
|
|
83
|
+
const doc = (await readTunedDocument(projectRoot))
|
|
84
|
+
?? { overrides: {}, lastAppliedAt: {}, history: [] };
|
|
85
|
+
const applied = [];
|
|
86
|
+
for (const suggestion of Array.isArray(suggestions) ? suggestions : []) {
|
|
87
|
+
const key = suggestion?.target;
|
|
88
|
+
if (!TUNABLE_KEYS[key]) continue;
|
|
89
|
+
const last = Number(doc.lastAppliedAt[key]);
|
|
90
|
+
if (Number.isFinite(last) && now - last < TUNING_COOLDOWN_MS) continue;
|
|
91
|
+
const value = clampTunable(key, suggestion.suggested);
|
|
92
|
+
if (value === null || doc.overrides[key] === value) continue;
|
|
93
|
+
const from = doc.overrides[key] ?? suggestion.current ?? null;
|
|
94
|
+
doc.overrides[key] = value;
|
|
95
|
+
doc.lastAppliedAt[key] = now;
|
|
96
|
+
doc.history.push({
|
|
97
|
+
at: new Date(now).toISOString(),
|
|
98
|
+
key,
|
|
99
|
+
from,
|
|
100
|
+
to: value,
|
|
101
|
+
reason: suggestion?.evidence?.reason ?? null,
|
|
102
|
+
});
|
|
103
|
+
applied.push({ key, from, to: value });
|
|
104
|
+
}
|
|
105
|
+
if (applied.length > 0) {
|
|
106
|
+
doc.history = doc.history.slice(-50);
|
|
107
|
+
await writeTunedDocument(projectRoot, doc);
|
|
108
|
+
}
|
|
109
|
+
return applied;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
async function collectTelemetry(projectRoot, config) {
|
|
113
|
+
const { resolveStage } = await import('../core/observability/emit/config.js');
|
|
114
|
+
if (resolveStage(config) === 'off') return { status: 'skipped', reason: 'stage_off' };
|
|
115
|
+
const { getRecorder, segmentsRoot } = await import('../core/observability/emit/lifecycle.js');
|
|
116
|
+
const { ingestStoredTelemetry } = await import('../core/observability/adapters/ingest.js');
|
|
117
|
+
const { maybeRefreshSupport } = await import('../core/observability/support/schedule.js');
|
|
118
|
+
await fs.mkdir(segmentsRoot(projectRoot), { recursive: true });
|
|
119
|
+
const recorder = getRecorder({ projectRoot, config });
|
|
120
|
+
const ingest = await ingestStoredTelemetry({ projectRoot, recorder, deadlineMs: TELEMETRY_DEADLINE_MS });
|
|
121
|
+
const flush = await recorder.flush({ deadlineMs: FLUSH_DEADLINE_MS });
|
|
122
|
+
const support = await maybeRefreshSupport({ projectRoot, config, recorder });
|
|
123
|
+
const emitted = Object.values(ingest?.coverage ?? {})
|
|
124
|
+
.reduce((sum, cov) => sum + (Number(cov?.emitted) || 0), 0);
|
|
125
|
+
return {
|
|
126
|
+
status: ingest?.ok === true ? ingest.status : 'failed',
|
|
127
|
+
emitted,
|
|
128
|
+
written: flush?.written ?? 0,
|
|
129
|
+
support: support?.status ?? 'unknown',
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Run one self-improve pass.
|
|
135
|
+
* @param {string} projectRoot
|
|
136
|
+
* @param {{ force?: boolean, homeDir?: string, now?: number, minIntervalMs?: number }} [opts]
|
|
137
|
+
* @returns {Promise<{status: 'ran'|'skipped', reason?: string, steps?: object[]}>}
|
|
138
|
+
*/
|
|
139
|
+
export async function runSelfImprove(projectRoot, {
|
|
140
|
+
force = false,
|
|
141
|
+
homeDir = os.homedir(),
|
|
142
|
+
now = Date.now(),
|
|
143
|
+
minIntervalMs = DEFAULT_MIN_INTERVAL_MS,
|
|
144
|
+
} = {}) {
|
|
145
|
+
const root = path.resolve(projectRoot);
|
|
146
|
+
const stampPath = path.join(root, STAMP_REL);
|
|
147
|
+
try {
|
|
148
|
+
const { config, rawConfig } = await inspectRuntimeConfig(root, { homeDir });
|
|
149
|
+
if (config?.learning?.selfImprove?.enabled === false) {
|
|
150
|
+
return { status: 'skipped', reason: 'disabled' };
|
|
151
|
+
}
|
|
152
|
+
const stamp = await readJson(stampPath);
|
|
153
|
+
const lastRunAt = Date.parse(stamp?.lastRunAt ?? '');
|
|
154
|
+
if (!force && Number.isFinite(lastRunAt) && now - lastRunAt < minIntervalMs) {
|
|
155
|
+
return { status: 'skipped', reason: 'min_interval' };
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
const outcome = await withFileLock(stampPath, async () => {
|
|
159
|
+
const steps = [];
|
|
160
|
+
steps.push(await step('episodes', async () => {
|
|
161
|
+
const res = await backfillEpisodes(root, { config, homeDir, now });
|
|
162
|
+
return { status: res.reason ? 'skipped' : 'ok', ...res };
|
|
163
|
+
}));
|
|
164
|
+
steps.push(await step('diagnostics', async () => {
|
|
165
|
+
const patterns = await mineFailurePatterns(root);
|
|
166
|
+
const feedback = await collectFeedbackEvents(root);
|
|
167
|
+
const skills = await collectSkillAccuracy(root);
|
|
168
|
+
return {
|
|
169
|
+
status: 'ok',
|
|
170
|
+
patterns: Array.isArray(patterns?.patterns) ? patterns.patterns.length : 0,
|
|
171
|
+
feedbackEvents: Array.isArray(feedback?.events) ? feedback.events.length : 0,
|
|
172
|
+
skills: Object.keys(skills?.skills ?? {}).length,
|
|
173
|
+
};
|
|
174
|
+
}));
|
|
175
|
+
steps.push(await step('tuning', async () => {
|
|
176
|
+
const result = await computeTuningSuggestions(root);
|
|
177
|
+
const mode = resolveApplyMode(rawConfig ?? {});
|
|
178
|
+
const applied = mode === 'auto' ? await applyTuning(root, result.suggestions, { now }) : [];
|
|
179
|
+
return { status: 'ok', mode, suggestions: result.suggestions.length, applied };
|
|
180
|
+
}));
|
|
181
|
+
steps.push(await step('proposals', async () => {
|
|
182
|
+
const projectId = (await detectProjectContext(root, { homeDir })).project.name;
|
|
183
|
+
const minCount = Number(config?.learning?.proposals?.minCount) || undefined;
|
|
184
|
+
const res = await proposeFromPatterns(root, projectId, { minCount, dryRun: false });
|
|
185
|
+
return {
|
|
186
|
+
status: res.error ? 'failed' : 'ok',
|
|
187
|
+
proposed: res.proposed.filter((p) => p.status === 'proposed').length,
|
|
188
|
+
reason: res.error,
|
|
189
|
+
};
|
|
190
|
+
}));
|
|
191
|
+
steps.push(await step('telemetry', () => collectTelemetry(root, config)));
|
|
192
|
+
|
|
193
|
+
await writeJsonAtomic(stampPath, {
|
|
194
|
+
lastRunAt: new Date(now).toISOString(),
|
|
195
|
+
steps,
|
|
196
|
+
});
|
|
197
|
+
return steps;
|
|
198
|
+
}, { maxWaitMs: 500 });
|
|
199
|
+
|
|
200
|
+
if (outcome === undefined) return { status: 'skipped', reason: 'locked' };
|
|
201
|
+
return { status: 'ran', steps: outcome };
|
|
202
|
+
} catch (error) {
|
|
203
|
+
return { status: 'skipped', reason: error?.message ?? String(error) };
|
|
204
|
+
}
|
|
205
|
+
}
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
// tunedOverlay.js — auto-applied tuning overlay (learning.tuning.applyMode 'auto').
|
|
2
|
+
//
|
|
3
|
+
// The self-improve cycle writes clamped one-step tuning changes to
|
|
4
|
+
// `.ukit/storage/learning/tuned.json`; runtime config readers merge the
|
|
5
|
+
// overlay over the project config. The file lives outside the install-managed
|
|
6
|
+
// `config.json`, so a reinstall never erases learned values, and deleting the
|
|
7
|
+
// file (or setting applyMode 'manual'|'off') is a complete rollback.
|
|
8
|
+
//
|
|
9
|
+
// Contract:
|
|
10
|
+
// * Only TUNABLE_KEYS are honored; every value is clamped to its bounds on
|
|
11
|
+
// read, so a hand-edited or corrupt overlay can never push a key outside
|
|
12
|
+
// the range the tuner itself is allowed to reach.
|
|
13
|
+
// * Never throws: missing/corrupt overlay → no overrides.
|
|
14
|
+
// * Merge point: inspectRuntimeConfig (src/core/runtimeConfig.js) applies
|
|
15
|
+
// the overlay after the user+project merge, before defaults/validation.
|
|
16
|
+
|
|
17
|
+
import fs from 'node:fs/promises';
|
|
18
|
+
import path from 'node:path';
|
|
19
|
+
|
|
20
|
+
export const TUNED_REL = path.join('.ukit', 'storage', 'learning', 'tuned.json');
|
|
21
|
+
export const TUNED_VERSION = 1;
|
|
22
|
+
|
|
23
|
+
export const TUNABLE_KEYS = Object.freeze({
|
|
24
|
+
'codeIntel.retriever.weights.exact': { min: 0.1, max: 2.0 },
|
|
25
|
+
'codeIntel.retriever.weights.symbol': { min: 0.1, max: 2.0 },
|
|
26
|
+
'codeIntel.retriever.weights.bm25': { min: 0.1, max: 2.0 },
|
|
27
|
+
'codeIntel.retriever.weights.semantic': { min: 0.1, max: 2.0 },
|
|
28
|
+
'codeIntel.retriever.weights.vector': { min: 0.1, max: 2.0 },
|
|
29
|
+
'orchestration.escalation.debugLoopThreshold': { min: 1, max: 5, integer: true },
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
function isPlainObject(value) {
|
|
33
|
+
return Boolean(value) && typeof value === 'object' && !Array.isArray(value);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export function clampTunable(key, value) {
|
|
37
|
+
const spec = TUNABLE_KEYS[key];
|
|
38
|
+
if (!spec || typeof value !== 'number' || !Number.isFinite(value)) return null;
|
|
39
|
+
const clamped = Math.min(spec.max, Math.max(spec.min, value));
|
|
40
|
+
return spec.integer ? Math.round(clamped) : Math.round(clamped * 100) / 100;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export function sanitizeOverrides(raw) {
|
|
44
|
+
const out = {};
|
|
45
|
+
if (!isPlainObject(raw)) return out;
|
|
46
|
+
for (const [key, value] of Object.entries(raw)) {
|
|
47
|
+
const clamped = clampTunable(key, value);
|
|
48
|
+
if (clamped !== null) out[key] = clamped;
|
|
49
|
+
}
|
|
50
|
+
return out;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export async function readTunedDocument(projectRoot) {
|
|
54
|
+
try {
|
|
55
|
+
const doc = JSON.parse(await fs.readFile(path.join(projectRoot, TUNED_REL), 'utf8'));
|
|
56
|
+
if (!isPlainObject(doc)) return null;
|
|
57
|
+
return {
|
|
58
|
+
version: TUNED_VERSION,
|
|
59
|
+
overrides: sanitizeOverrides(doc.overrides),
|
|
60
|
+
lastAppliedAt: isPlainObject(doc.lastAppliedAt) ? doc.lastAppliedAt : {},
|
|
61
|
+
history: Array.isArray(doc.history) ? doc.history.slice(-50) : [],
|
|
62
|
+
};
|
|
63
|
+
} catch {
|
|
64
|
+
return null;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export async function writeTunedDocument(projectRoot, doc) {
|
|
69
|
+
const filePath = path.join(projectRoot, TUNED_REL);
|
|
70
|
+
const tmp = `${filePath}.tmp-${process.pid}`;
|
|
71
|
+
try {
|
|
72
|
+
await fs.mkdir(path.dirname(filePath), { recursive: true });
|
|
73
|
+
await fs.writeFile(tmp, `${JSON.stringify({ ...doc, version: TUNED_VERSION }, null, 2)}\n`);
|
|
74
|
+
await fs.rename(tmp, filePath);
|
|
75
|
+
return true;
|
|
76
|
+
} catch {
|
|
77
|
+
await fs.rm(tmp, { force: true }).catch(() => {});
|
|
78
|
+
return false;
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// Resolves the effective apply mode from a raw (pre-default) config object:
|
|
83
|
+
// absent → 'auto' (the shipped default).
|
|
84
|
+
export function resolveApplyMode(rawConfig) {
|
|
85
|
+
const tuning = rawConfig?.learning?.tuning;
|
|
86
|
+
if (tuning?.enabled === false) return 'off';
|
|
87
|
+
const mode = tuning?.applyMode;
|
|
88
|
+
return mode === 'manual' || mode === 'off' || mode === 'auto' ? mode : 'auto';
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
// Returns a copy of `config` with `overrides` set at their dotted paths.
|
|
92
|
+
export function applyOverrides(config, overrides) {
|
|
93
|
+
if (!isPlainObject(overrides) || Object.keys(overrides).length === 0) return config;
|
|
94
|
+
const next = structuredClone(isPlainObject(config) ? config : {});
|
|
95
|
+
for (const [key, value] of Object.entries(overrides)) {
|
|
96
|
+
const parts = key.split('.');
|
|
97
|
+
let node = next;
|
|
98
|
+
for (const part of parts.slice(0, -1)) {
|
|
99
|
+
if (!isPlainObject(node[part])) node[part] = {};
|
|
100
|
+
node = node[part];
|
|
101
|
+
}
|
|
102
|
+
node[parts[parts.length - 1]] = value;
|
|
103
|
+
}
|
|
104
|
+
return next;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// Merge helper for config readers: overlay applies only in 'auto' mode.
|
|
108
|
+
export async function mergeTunedOverlay(projectRoot, rawConfig) {
|
|
109
|
+
if (resolveApplyMode(rawConfig) !== 'auto') return rawConfig;
|
|
110
|
+
const doc = await readTunedDocument(projectRoot);
|
|
111
|
+
return doc ? applyOverrides(rawConfig, doc.overrides) : rawConfig;
|
|
112
|
+
}
|
|
@@ -112,25 +112,45 @@ try {
|
|
|
112
112
|
const config = readJson(configPath) || {};
|
|
113
113
|
if (config?.learning?.episodes?.autoWrite !== true) process.exit(0);
|
|
114
114
|
|
|
115
|
-
const probe = spawnSync("ukit", ["--version"], {
|
|
116
|
-
shell: true, stdio: "ignore", timeout: 2000,
|
|
117
|
-
});
|
|
118
|
-
if (probe.error || probe.status === null || probe.status === undefined) process.exit(0);
|
|
119
|
-
|
|
120
115
|
const sessionId = typeof payload.session_id === "string" && payload.session_id.trim()
|
|
121
116
|
? payload.session_id.trim()
|
|
122
117
|
: null;
|
|
123
118
|
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
}
|
|
133
|
-
|
|
119
|
+
(async () => {
|
|
120
|
+
// 3.3.0: prefer the absolute node + CLI recorded by `ukit install`
|
|
121
|
+
// (.ukit/storage/cli.json, validated by resolveCliLocator) — GUI-launched
|
|
122
|
+
// hosts often lack nvm/volta on PATH, which made the bare `ukit` spawn fail
|
|
123
|
+
// silently. PATH lookup stays the fallback for pre-locator installs.
|
|
124
|
+
let trigger = null;
|
|
125
|
+
try {
|
|
126
|
+
trigger = await import(require("url").pathToFileURL(process.argv[4]).href);
|
|
127
|
+
} catch {}
|
|
128
|
+
const located = trigger?.resolveCliLocator?.(projectRoot) ?? null;
|
|
129
|
+
const cli = located
|
|
130
|
+
? { cmd: located.node, pre: [located.bin], shell: false }
|
|
131
|
+
: { cmd: "ukit", pre: [], shell: true };
|
|
132
|
+
|
|
133
|
+
const probe = spawnSync(cli.cmd, [...cli.pre, "--version"], {
|
|
134
|
+
shell: cli.shell, stdio: "ignore", timeout: 2000,
|
|
135
|
+
});
|
|
136
|
+
if (probe.error || probe.status === null || probe.status === undefined) return;
|
|
137
|
+
|
|
138
|
+
spawnSync(cli.cmd, [...cli.pre, "memory", "episode"], {
|
|
139
|
+
shell: cli.shell,
|
|
140
|
+
stdio: "ignore",
|
|
141
|
+
timeout: 4000,
|
|
142
|
+
cwd: projectRoot,
|
|
143
|
+
env: sessionId
|
|
144
|
+
? { ...process.env, UKIT_SESSION_ID: sessionId }
|
|
145
|
+
: process.env,
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
// Detached self-improve pass (episode backfill for sessions whose stop id
|
|
149
|
+
// did not match their ledger, diagnostics, auto-tuning, telemetry
|
|
150
|
+
// collect). Returns in ms; rate-limited inside the trigger.
|
|
151
|
+
trigger?.triggerSelfImprove?.({ projectRoot });
|
|
152
|
+
})().catch(() => {}).finally(() => process.exit(0));
|
|
153
|
+
' "$UKIT_INPUT_FILE" "$CONFIG_FILE" "$PROJECT_ROOT" "$SCRIPT_DIR/../ukit/runtime/self-improve-trigger.mjs" 2>/dev/null || true
|
|
134
154
|
|
|
135
155
|
rm -f "$UKIT_INPUT_FILE" 2>/dev/null || true
|
|
136
156
|
exit 0
|
|
@@ -1722,6 +1722,19 @@ async function main() {
|
|
|
1722
1722
|
}
|
|
1723
1723
|
|
|
1724
1724
|
printRouteState(sharedState);
|
|
1725
|
+
|
|
1726
|
+
// 3.3.0: Codex has no hook lifecycle — route-task is the one helper it runs
|
|
1727
|
+
// per task, so it doubles as the self-improve trigger there. Fire-and-forget:
|
|
1728
|
+
// detached child, rate-limited inside the trigger, never throws. Opt out
|
|
1729
|
+
// with UKIT_SELF_IMPROVE=0 (tests/CI) or learning.selfImprove.enabled:false.
|
|
1730
|
+
if (process.env.UKIT_SELF_IMPROVE !== '0') {
|
|
1731
|
+
try {
|
|
1732
|
+
const { triggerSelfImprove } = await import('../runtime/self-improve-trigger.mjs');
|
|
1733
|
+
triggerSelfImprove({ projectRoot: rootDir });
|
|
1734
|
+
} catch {
|
|
1735
|
+
// background work only — a missing/broken trigger never affects routing
|
|
1736
|
+
}
|
|
1737
|
+
}
|
|
1725
1738
|
}
|
|
1726
1739
|
|
|
1727
1740
|
async function ensureFreshIndex({ rootDir, logPrefix, signal = null, deadlineMs = null, discoverySnapshot = null } = {}) {
|
|
@@ -718,8 +718,7 @@ export async function requestBatch(batch, {
|
|
|
718
718
|
|
|
719
719
|
const status = res?.status ?? (res?.ok === false ? 500 : 200);
|
|
720
720
|
if (res?.ok === false || status < 200 || status >= 300) {
|
|
721
|
-
// A
|
|
722
|
-
// unic-decision-multilingual before provider binding) returns a 4xx
|
|
721
|
+
// A checkpoint without bound credentials returns a 4xx
|
|
723
722
|
// model_not_found body — retry once on the bare `unic-decision` model.
|
|
724
723
|
if (checkpoint !== 'unic-decision') {
|
|
725
724
|
let notFound = status === 404;
|