@ngockhoale/ukit 2.3.15 → 2.3.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +122 -19
- package/manifests/platform.full.yaml +26 -1
- package/package.json +1 -1
- package/src/cli/commands/doctor.js +49 -2
- package/src/cli/commands/install.js +5 -0
- package/src/core/applyPlan.js +1 -1
- package/src/core/diffPlan.js +80 -0
- package/src/core/gatewayProbe.js +441 -0
- package/src/core/gatewayResilienceEnv.js +292 -0
- package/src/core/runInstallPipeline.js +11 -0
- package/src/core/taskBudgetValidator.js +241 -0
- package/src/core/taskProgressGuard.js +157 -0
- package/src/manifest/validateManifest.js +1 -1
- package/templates/.claude/agents/feature-implementer.md +30 -0
- package/templates/.claude/agents/handoff-planner.md +12 -0
- package/templates/.claude/commands/ukit/handoff-fullstack.md +4 -1
- package/templates/.claude/hooks/task-watchdog.sh +221 -0
- package/templates/.claude/settings.json +10 -0
- package/templates/.claude/ukit/index/task-budget-validator.mjs +236 -0
- package/templates/.claude/ukit/runtime/task-watchdog.mjs +299 -0
- package/templates/.gitignore +5 -4
- package/templates/.omp/agents/feature-implementer.md +30 -0
- package/templates/.omp/agents/handoff-planner.md +12 -0
- package/templates/.omp/hooks/pre/ukit-bridge.js +2 -1
- package/templates/docs/AI_HANDOFF/tasks/_TEMPLATE.md +17 -0
- package/templates/ukit/storage/config.json +8 -1
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// task-budget-validator.mjs — TASK-014 / Cycle C12 (shipped CLI twin)
|
|
3
|
+
//
|
|
4
|
+
// Self-contained twin of src/core/taskBudgetValidator.js for user installs (no `src/`).
|
|
5
|
+
// The logic below is a literal port — keep the two files behavior-identical when editing.
|
|
6
|
+
//
|
|
7
|
+
// Usage:
|
|
8
|
+
// node .claude/ukit/index/task-budget-validator.mjs <path-to-TASK-xxx.md> [criteria JSON]
|
|
9
|
+
//
|
|
10
|
+
// Output:
|
|
11
|
+
// First stdout line: `VERDICT: ok` or `VERDICT: needs_breakdown`
|
|
12
|
+
// Then one `REASON: <criterion>: <detail>` line per reason.
|
|
13
|
+
// ALWAYS exits 0 — advisory gate, must never break a pipeline.
|
|
14
|
+
|
|
15
|
+
import fs from 'node:fs';
|
|
16
|
+
import process from 'node:process';
|
|
17
|
+
import { pathToFileURL } from 'node:url';
|
|
18
|
+
|
|
19
|
+
const DEFAULT_CRITERIA = Object.freeze({
|
|
20
|
+
maxTargetFiles: 3,
|
|
21
|
+
maxTestCases: 8,
|
|
22
|
+
maxVerificationMinutes: 10,
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
const VERIFICATION_MINUTE_TABLE = Object.freeze({
|
|
26
|
+
'yarn test:release-core': 3,
|
|
27
|
+
'node scripts/release/verify-release.mjs': 2,
|
|
28
|
+
});
|
|
29
|
+
|
|
30
|
+
function estimateVerificationMinutes(commands) {
|
|
31
|
+
if (!Array.isArray(commands)) return 0;
|
|
32
|
+
let total = 0;
|
|
33
|
+
for (const raw of commands) {
|
|
34
|
+
const cmd = String(raw || '').trim();
|
|
35
|
+
if (!cmd) continue;
|
|
36
|
+
if (Object.prototype.hasOwnProperty.call(VERIFICATION_MINUTE_TABLE, cmd)) {
|
|
37
|
+
total += VERIFICATION_MINUTE_TABLE[cmd];
|
|
38
|
+
continue;
|
|
39
|
+
}
|
|
40
|
+
if (/^yarn\s+vitest(\s+run)?\b/.test(cmd)) {
|
|
41
|
+
const tokens = cmd.split(/\s+/);
|
|
42
|
+
const runIdx = tokens.indexOf('run');
|
|
43
|
+
const tail = runIdx >= 0 ? tokens.slice(runIdx + 1) : tokens.slice(2);
|
|
44
|
+
const fileArgs = tail.filter((t) => !t.startsWith('-') && t.length > 0);
|
|
45
|
+
total += fileArgs.length * 0.5;
|
|
46
|
+
continue;
|
|
47
|
+
}
|
|
48
|
+
total += 1;
|
|
49
|
+
}
|
|
50
|
+
return Math.round(total * 10) / 10;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
function splitSections(markdown) {
|
|
54
|
+
const lines = String(markdown || '').split('\n');
|
|
55
|
+
const sections = new Map();
|
|
56
|
+
let currentHeading = null;
|
|
57
|
+
let buffer = [];
|
|
58
|
+
for (const line of lines) {
|
|
59
|
+
const m = /^##\s+(.+?)\s*$/.exec(line);
|
|
60
|
+
if (m) {
|
|
61
|
+
if (currentHeading !== null) sections.set(currentHeading, buffer);
|
|
62
|
+
currentHeading = m[1].trim();
|
|
63
|
+
buffer = [];
|
|
64
|
+
} else if (currentHeading !== null) {
|
|
65
|
+
buffer.push(line);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
if (currentHeading !== null) sections.set(currentHeading, buffer);
|
|
69
|
+
return sections;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function getSectionBody(sections, name) {
|
|
73
|
+
if (!sections.has(name)) return null;
|
|
74
|
+
return sections.get(name).join('\n');
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function countTargetFiles(body) {
|
|
78
|
+
if (body == null) return 0;
|
|
79
|
+
let count = 0;
|
|
80
|
+
for (const line of body.split('\n')) {
|
|
81
|
+
if (/^\s*-\s+\S/.test(line)) count += 1;
|
|
82
|
+
}
|
|
83
|
+
return count;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function countTestCases(body) {
|
|
87
|
+
if (body == null) return 0;
|
|
88
|
+
let count = 0;
|
|
89
|
+
let sawHeader = false;
|
|
90
|
+
let sawSeparator = false;
|
|
91
|
+
for (const line of body.split('\n')) {
|
|
92
|
+
const trimmed = line.trim();
|
|
93
|
+
if (!trimmed.startsWith('|')) continue;
|
|
94
|
+
if (!sawHeader) {
|
|
95
|
+
sawHeader = true;
|
|
96
|
+
continue;
|
|
97
|
+
}
|
|
98
|
+
if (!sawSeparator && /^\|[-\s|]+\|\s*$/.test(trimmed)) {
|
|
99
|
+
sawSeparator = true;
|
|
100
|
+
continue;
|
|
101
|
+
}
|
|
102
|
+
count += 1;
|
|
103
|
+
}
|
|
104
|
+
return count;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function extractVerificationCommands(body) {
|
|
108
|
+
if (body == null) return [];
|
|
109
|
+
const lines = body.split('\n');
|
|
110
|
+
const fenceStart = /^```\s*bash\s*$/;
|
|
111
|
+
const fenceEnd = /^```\s*$/;
|
|
112
|
+
let inFence = false;
|
|
113
|
+
const out = [];
|
|
114
|
+
for (const line of lines) {
|
|
115
|
+
if (!inFence) {
|
|
116
|
+
if (fenceStart.test(line.trim())) inFence = true;
|
|
117
|
+
continue;
|
|
118
|
+
}
|
|
119
|
+
if (fenceEnd.test(line.trim())) break;
|
|
120
|
+
out.push(line);
|
|
121
|
+
}
|
|
122
|
+
return out.map((l) => l.trim()).filter(Boolean);
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
function goalText(body) {
|
|
126
|
+
if (body == null) return '';
|
|
127
|
+
return body
|
|
128
|
+
.split('\n')
|
|
129
|
+
.map((l) => l.replace(/^#+\s*/, '').trim())
|
|
130
|
+
.filter(Boolean)
|
|
131
|
+
.join(' ');
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
export function validateTaskFile(markdown, criteria = DEFAULT_CRITERIA) {
|
|
135
|
+
const reasons = [];
|
|
136
|
+
const text = typeof markdown === 'string' ? markdown : '';
|
|
137
|
+
const sections = splitSections(text);
|
|
138
|
+
|
|
139
|
+
const goalBody = getSectionBody(sections, 'Goal');
|
|
140
|
+
const targetBody = getSectionBody(sections, 'Target Files');
|
|
141
|
+
const testBody = getSectionBody(sections, 'Test Cases');
|
|
142
|
+
const verifyBody = getSectionBody(sections, 'Verification Commands');
|
|
143
|
+
|
|
144
|
+
const requiredSections = [
|
|
145
|
+
['Goal', goalBody],
|
|
146
|
+
['Test Cases', testBody],
|
|
147
|
+
['Verification Commands', verifyBody],
|
|
148
|
+
];
|
|
149
|
+
for (const [name, body] of requiredSections) {
|
|
150
|
+
if (body == null || body.trim() === '') {
|
|
151
|
+
reasons.push({
|
|
152
|
+
criterion: 'missing-field',
|
|
153
|
+
detail: `required section \`## ${name}\` is missing or empty`,
|
|
154
|
+
});
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
if (targetBody != null) {
|
|
159
|
+
const n = countTargetFiles(targetBody);
|
|
160
|
+
if (n > criteria.maxTargetFiles) {
|
|
161
|
+
reasons.push({
|
|
162
|
+
criterion: 'target-file-count',
|
|
163
|
+
detail: `${n} target files exceeds maxTargetFiles=${criteria.maxTargetFiles}`,
|
|
164
|
+
});
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
if (testBody != null) {
|
|
169
|
+
const n = countTestCases(testBody);
|
|
170
|
+
if (n > criteria.maxTestCases) {
|
|
171
|
+
reasons.push({
|
|
172
|
+
criterion: 'test-case-count',
|
|
173
|
+
detail: `${n} test cases exceeds maxTestCases=${criteria.maxTestCases}`,
|
|
174
|
+
});
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
if (verifyBody != null) {
|
|
179
|
+
const commands = extractVerificationCommands(verifyBody);
|
|
180
|
+
if (commands.length > 0) {
|
|
181
|
+
const minutes = estimateVerificationMinutes(commands);
|
|
182
|
+
if (minutes > criteria.maxVerificationMinutes) {
|
|
183
|
+
reasons.push({
|
|
184
|
+
criterion: 'verification-minutes',
|
|
185
|
+
detail: `estimated ${minutes} verification minutes exceeds maxVerificationMinutes=${criteria.maxVerificationMinutes}`,
|
|
186
|
+
});
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
if (goalBody != null) {
|
|
192
|
+
const prose = goalText(goalBody);
|
|
193
|
+
if (/\binvestigate\b/i.test(prose) && /\bfirst\b/i.test(prose)) {
|
|
194
|
+
reasons.push({
|
|
195
|
+
criterion: 'investigate-first',
|
|
196
|
+
detail: 'Goal contains both "investigate" and "first" — split the task into a spike + a build step',
|
|
197
|
+
});
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
return {
|
|
202
|
+
verdict: reasons.length === 0 ? 'ok' : 'needs_breakdown',
|
|
203
|
+
reasons,
|
|
204
|
+
};
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function main() {
|
|
208
|
+
const [, , targetPath, criteriaArg] = process.argv;
|
|
209
|
+
let criteria = DEFAULT_CRITERIA;
|
|
210
|
+
if (criteriaArg) {
|
|
211
|
+
try {
|
|
212
|
+
const overrides = JSON.parse(criteriaArg);
|
|
213
|
+
criteria = { ...DEFAULT_CRITERIA, ...overrides };
|
|
214
|
+
} catch {
|
|
215
|
+
// malformed override — fall back to defaults (advisory, never fail)
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
let markdown = '';
|
|
219
|
+
try {
|
|
220
|
+
markdown = fs.readFileSync(targetPath, 'utf8');
|
|
221
|
+
} catch (err) {
|
|
222
|
+
console.log('VERDICT: needs_breakdown');
|
|
223
|
+
console.log(`REASON: missing-field: cannot read ${targetPath} (${err && err.code ? err.code : 'error'})`);
|
|
224
|
+
process.exit(0);
|
|
225
|
+
}
|
|
226
|
+
const { verdict, reasons } = validateTaskFile(markdown, criteria);
|
|
227
|
+
console.log(`VERDICT: ${verdict}`);
|
|
228
|
+
for (const r of reasons) {
|
|
229
|
+
console.log(`REASON: ${r.criterion}: ${r.detail}`);
|
|
230
|
+
}
|
|
231
|
+
process.exit(0);
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
if (process.argv[1] && pathToFileURL(process.argv[1]).href === import.meta.url) {
|
|
235
|
+
main();
|
|
236
|
+
}
|
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* task-watchdog.mjs — wall-clock watchdog runtime for handoff tasks.
|
|
4
|
+
*
|
|
5
|
+
* Self-contained runtime module. No imports from `src/`, no sibling imports
|
|
6
|
+
* beyond `node:` builtins (the `token-utils` file lock is intentionally NOT
|
|
7
|
+
* used — file-locking watchdog state under a Stop hook with a 4s budget is a
|
|
8
|
+
* recipe for getting killed mid-write by its own chain).
|
|
9
|
+
*
|
|
10
|
+
* Exports:
|
|
11
|
+
* DEFAULT_CONFIG — fail-open defaults; identical to the
|
|
12
|
+
* `handoff.taskBudgets` / `handoff.milestoneIntervalMin`
|
|
13
|
+
* keys seeded into templates/ukit/storage/config.json.
|
|
14
|
+
* parseTaskFile(markdown) — minimal twin of src/core/taskProgressGuard's
|
|
15
|
+
* parser; reads id, status, size, lastProgressIso.
|
|
16
|
+
* evaluateBudgets(...) — per in_progress task returns
|
|
17
|
+
* { id, phase: 'ok'|'soft'|'hard', elapsedMin, lastGreenMin }.
|
|
18
|
+
* elapsed = now - state[id].firstSeen; Progress
|
|
19
|
+
* appends do NOT reset the clock.
|
|
20
|
+
* loadConfig(configPath) — deep-merges .ukit/storage/config.json's
|
|
21
|
+
* `handoff` key over DEFAULT_CONFIG; any read/parse
|
|
22
|
+
* error → DEFAULT_CONFIG (fail-open).
|
|
23
|
+
* readState(statePath) / writeState(statePath, value) —
|
|
24
|
+
* `.ukit/storage/cache/task-watchdog/state.json`,
|
|
25
|
+
* shape { firstSeen, hardBlocks }; any I/O error
|
|
26
|
+
* → treat as empty, never throw.
|
|
27
|
+
*
|
|
28
|
+
* The hard clock does not reset on Progress appends on purpose: trickling entries
|
|
29
|
+
* every 4 minutes must not buy an unlimited task. The newest Progress timestamp is
|
|
30
|
+
* used only to name the last-green milestone in user-facing messages.
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
import fs from 'node:fs/promises';
|
|
34
|
+
import path from 'node:path';
|
|
35
|
+
|
|
36
|
+
// ─── Defaults ─────────────────────────────────────────────────────────────
|
|
37
|
+
export const DEFAULT_CONFIG = Object.freeze({
|
|
38
|
+
taskBudgets: {
|
|
39
|
+
S: { softMin: 8, hardMin: 15 },
|
|
40
|
+
M: { softMin: 15, hardMin: 30 },
|
|
41
|
+
L: { softMin: 25, hardMin: 45 },
|
|
42
|
+
hardPolicy: 'split',
|
|
43
|
+
},
|
|
44
|
+
milestoneIntervalMin: 5,
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
// Cap on consecutive hard-trip emissions before degrading to advisory —
|
|
48
|
+
// after the third evaluation the hook stops blocking and the run is allowed
|
|
49
|
+
// to end so the orchestrator can hand back to the user. An honest failure of
|
|
50
|
+
// enforcement is better than a wedged pipeline.
|
|
51
|
+
export const HARD_BLOCK_CAP = 2;
|
|
52
|
+
|
|
53
|
+
// ─── Helpers ──────────────────────────────────────────────────────────────
|
|
54
|
+
function isPlainObject(value) {
|
|
55
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function deepFreeze(value) {
|
|
59
|
+
if (!isPlainObject(value)) return value;
|
|
60
|
+
for (const key of Object.keys(value)) deepFreeze(value[key]);
|
|
61
|
+
return Object.freeze(value);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
function mergeBudgets(into, from) {
|
|
65
|
+
if (!isPlainObject(from)) return into;
|
|
66
|
+
for (const [size, bucket] of Object.entries(from)) {
|
|
67
|
+
if (!isPlainObject(bucket)) continue;
|
|
68
|
+
const prev = isPlainObject(into[size]) ? into[size] : {};
|
|
69
|
+
into[size] = { ...prev, ...bucket };
|
|
70
|
+
}
|
|
71
|
+
return into;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// ─── Config ───────────────────────────────────────────────────────────────
|
|
75
|
+
export async function loadConfig(configPath) {
|
|
76
|
+
const fallback = deepFreeze(JSON.parse(JSON.stringify(DEFAULT_CONFIG)));
|
|
77
|
+
if (!configPath) return fallback;
|
|
78
|
+
let raw;
|
|
79
|
+
try {
|
|
80
|
+
raw = await fs.readFile(configPath, 'utf8');
|
|
81
|
+
} catch {
|
|
82
|
+
return fallback;
|
|
83
|
+
}
|
|
84
|
+
let parsed;
|
|
85
|
+
try {
|
|
86
|
+
parsed = JSON.parse(raw);
|
|
87
|
+
} catch {
|
|
88
|
+
return fallback;
|
|
89
|
+
}
|
|
90
|
+
const handoff = parsed?.handoff;
|
|
91
|
+
if (!isPlainObject(handoff)) return fallback;
|
|
92
|
+
|
|
93
|
+
const merged = {
|
|
94
|
+
taskBudgets: {},
|
|
95
|
+
hardPolicy: typeof handoff.taskBudgets?.hardPolicy === 'string'
|
|
96
|
+
? handoff.taskBudgets.hardPolicy
|
|
97
|
+
: DEFAULT_CONFIG.taskBudgets.hardPolicy,
|
|
98
|
+
milestoneIntervalMin: Number.isFinite(handoff.milestoneIntervalMin)
|
|
99
|
+
? handoff.milestoneIntervalMin
|
|
100
|
+
: DEFAULT_CONFIG.milestoneIntervalMin,
|
|
101
|
+
};
|
|
102
|
+
mergeBudgets(merged.taskBudgets, DEFAULT_CONFIG.taskBudgets);
|
|
103
|
+
mergeBudgets(merged.taskBudgets, handoff.taskBudgets);
|
|
104
|
+
|
|
105
|
+
// Keep the legacy key visible (matches the seeded config.json shape).
|
|
106
|
+
merged.taskBudgets.hardPolicy = merged.hardPolicy;
|
|
107
|
+
return deepFreeze(merged);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
// ─── State ────────────────────────────────────────────────────────────────
|
|
111
|
+
const EMPTY_STATE = Object.freeze({ firstSeen: Object.freeze({}), hardBlocks: Object.freeze({}) });
|
|
112
|
+
|
|
113
|
+
export async function readState(statePath) {
|
|
114
|
+
if (!statePath) return { firstSeen: {}, hardBlocks: {} };
|
|
115
|
+
try {
|
|
116
|
+
const raw = await fs.readFile(statePath, 'utf8');
|
|
117
|
+
const parsed = JSON.parse(raw);
|
|
118
|
+
return {
|
|
119
|
+
firstSeen: isPlainObject(parsed?.firstSeen) ? { ...parsed.firstSeen } : {},
|
|
120
|
+
hardBlocks: isPlainObject(parsed?.hardBlocks) ? { ...parsed.hardBlocks } : {},
|
|
121
|
+
};
|
|
122
|
+
} catch {
|
|
123
|
+
return { firstSeen: {}, hardBlocks: {} };
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
export async function writeState(statePath, value) {
|
|
128
|
+
if (!statePath) return;
|
|
129
|
+
try {
|
|
130
|
+
await fs.mkdir(path.dirname(statePath), { recursive: true });
|
|
131
|
+
// temp + rename to avoid leaving a half-written state file if the hook is killed.
|
|
132
|
+
const tempPath = `${statePath}.${process.pid}-${Date.now()}-${Math.random().toString(16).slice(2)}.tmp`;
|
|
133
|
+
await fs.writeFile(tempPath, `${JSON.stringify(value, null, 1)}\n`, 'utf8');
|
|
134
|
+
await fs.rename(tempPath, statePath);
|
|
135
|
+
} catch {
|
|
136
|
+
// Fail-open: never throw from state writes.
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
// ─── Task parser ──────────────────────────────────────────────────────────
|
|
141
|
+
const TASK_ID_REGEX = /^#\s+TASK-([0-9]+[A-Za-z0-9._-]*)\b/m;
|
|
142
|
+
const STATUS_REGEX = /^-\s*Status:\s*`?([a-z_]+)`?/m;
|
|
143
|
+
const SIZE_REGEX = /^-\s*Size:\s*`?([SML])`?/m;
|
|
144
|
+
|
|
145
|
+
// Progress entries follow the format fixed by src/core/taskProgressGuard.js:
|
|
146
|
+
// - <ISO-8601> · milestone: <name> · ...
|
|
147
|
+
const PROGRESS_ENTRY_REGEX = /^-\s+(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2})?(?:[+-]\d{2}:?\d{2}|Z)?)\s+·/gm;
|
|
148
|
+
|
|
149
|
+
export function parseTaskFile(markdown) {
|
|
150
|
+
if (typeof markdown !== 'string') {
|
|
151
|
+
return { id: null, status: null, size: null, lastProgressIso: null };
|
|
152
|
+
}
|
|
153
|
+
const idMatch = markdown.match(TASK_ID_REGEX);
|
|
154
|
+
const statusMatch = markdown.match(STATUS_REGEX);
|
|
155
|
+
const sizeMatch = markdown.match(SIZE_REGEX);
|
|
156
|
+
let lastProgressIso = null;
|
|
157
|
+
PROGRESS_ENTRY_REGEX.lastIndex = 0;
|
|
158
|
+
for (const m of markdown.matchAll(PROGRESS_ENTRY_REGEX)) {
|
|
159
|
+
const candidate = m[1];
|
|
160
|
+
if (!lastProgressIso || candidate > lastProgressIso) lastProgressIso = candidate;
|
|
161
|
+
}
|
|
162
|
+
return {
|
|
163
|
+
id: idMatch ? `TASK-${idMatch[1]}` : null,
|
|
164
|
+
status: statusMatch ? statusMatch[1] : null,
|
|
165
|
+
size: sizeMatch ? sizeMatch[1] : null,
|
|
166
|
+
lastProgressIso,
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
function parseIsoToMs(iso) {
|
|
171
|
+
if (typeof iso !== 'string') return NaN;
|
|
172
|
+
const ms = Date.parse(iso);
|
|
173
|
+
return Number.isFinite(ms) ? ms : NaN;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// ─── Discovery ────────────────────────────────────────────────────────────
|
|
177
|
+
export async function listInProgressTasks(projectRoot) {
|
|
178
|
+
const tasksDir = path.join(projectRoot, 'docs', 'AI_HANDOFF', 'tasks');
|
|
179
|
+
let entries;
|
|
180
|
+
try {
|
|
181
|
+
entries = await fs.readdir(tasksDir);
|
|
182
|
+
} catch {
|
|
183
|
+
return [];
|
|
184
|
+
}
|
|
185
|
+
const out = [];
|
|
186
|
+
for (const entry of entries) {
|
|
187
|
+
if (!/^TASK-.*\.md$/i.test(entry)) continue;
|
|
188
|
+
let text;
|
|
189
|
+
try {
|
|
190
|
+
text = await fs.readFile(path.join(tasksDir, entry), 'utf8');
|
|
191
|
+
} catch {
|
|
192
|
+
continue;
|
|
193
|
+
}
|
|
194
|
+
const parsed = parseTaskFile(text);
|
|
195
|
+
if (!parsed.id) continue;
|
|
196
|
+
if (parsed.status !== 'in_progress') continue;
|
|
197
|
+
if (!['S', 'M', 'L'].includes(parsed.size)) parsed.size = 'M';
|
|
198
|
+
out.push(parsed);
|
|
199
|
+
}
|
|
200
|
+
return out;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
// ─── Budget evaluation ────────────────────────────────────────────────────
|
|
204
|
+
function bucketFor(size, config) {
|
|
205
|
+
const key = ['S', 'M', 'L'].includes(size) ? size : 'M';
|
|
206
|
+
const budgets = config?.taskBudgets || DEFAULT_CONFIG.taskBudgets;
|
|
207
|
+
return budgets[key] || DEFAULT_CONFIG.taskBudgets[key];
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* Evaluate every in_progress task against its wall-clock budget.
|
|
212
|
+
*
|
|
213
|
+
* `state` carries `firstSeen[id]` (epochMs) and `hardBlocks[id]` (count).
|
|
214
|
+
* The hard clock does NOT reset on Progress appends — that would let a task
|
|
215
|
+
* trickle entries every 4 minutes and never trip. Progress is used only to
|
|
216
|
+
* name the last-green milestone in user-facing messages.
|
|
217
|
+
*/
|
|
218
|
+
export function evaluateBudgets({ tasks = [], state = {}, now, config = DEFAULT_CONFIG } = {}) {
|
|
219
|
+
const out = [];
|
|
220
|
+
for (const task of tasks) {
|
|
221
|
+
const id = task.id;
|
|
222
|
+
const bucket = bucketFor(task.size, config);
|
|
223
|
+
const softMin = Number(bucket?.softMin) || 0;
|
|
224
|
+
const hardMin = Number(bucket?.hardMin) || 0;
|
|
225
|
+
const firstSeen = Number(state?.firstSeen?.[id]) || now;
|
|
226
|
+
const elapsedMs = Math.max(0, now - firstSeen);
|
|
227
|
+
const elapsedMin = elapsedMs / 60_000;
|
|
228
|
+
let phase = 'ok';
|
|
229
|
+
if (elapsedMin > hardMin) phase = 'hard';
|
|
230
|
+
else if (elapsedMin >= softMin) phase = 'soft';
|
|
231
|
+
let lastGreenMin = null;
|
|
232
|
+
if (task.lastProgressIso) {
|
|
233
|
+
const ms = parseIsoToMs(task.lastProgressIso);
|
|
234
|
+
if (Number.isFinite(ms)) lastGreenMin = Math.max(0, (now - ms) / 60_000);
|
|
235
|
+
}
|
|
236
|
+
out.push({ id, phase, elapsedMin, lastGreenMin, size: task.size, softMin, hardMin });
|
|
237
|
+
}
|
|
238
|
+
return out;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
// ─── Hook-driver helpers ──────────────────────────────────────────────────
|
|
242
|
+
export function describeSoft(result, id) {
|
|
243
|
+
return [
|
|
244
|
+
`[ukit-watchdog] ${id} is in its soft overrun window (elapsed ${result.elapsedMin.toFixed(1)}min; budget ${result.softMin}/${result.hardMin}min).`,
|
|
245
|
+
`Checkpoint a milestone now via \`## Progress\` and split the remaining Acceptance Criteria into ${id}-b before crossing hardMin.`,
|
|
246
|
+
].join(' ');
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
// Hard-phase advisory for advisory-only contexts (PostToolUse, degraded paths):
|
|
250
|
+
// same split instruction as the split-policy Stop block, but never a decision.
|
|
251
|
+
export function describeHardAdvisory(result, id) {
|
|
252
|
+
return [
|
|
253
|
+
`[ukit-watchdog] ${id} has crossed its hard budget (elapsed ${result.elapsedMin.toFixed(1)}min > hardMin ${result.hardMin}min).`,
|
|
254
|
+
`Checkpoint a milestone via \`## Progress\` and split the remaining Acceptance Criteria into ${id}-b now.`,
|
|
255
|
+
].join(' ');
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
export function describeSplitReason({ id, result, hardBlocks, lastGreen }) {
|
|
259
|
+
const lastGreenLine = lastGreen?.iso
|
|
260
|
+
? `last-green: ${lastGreen.iso} (${lastGreen.label || 'milestone'})`
|
|
261
|
+
: `last-green: none recorded`;
|
|
262
|
+
return [
|
|
263
|
+
`UKit watchdog: ${id} crossed its hard budget (elapsed ${result.elapsedMin.toFixed(1)}min > hardMin ${result.hardMin}min).`,
|
|
264
|
+
`Hard-trip count for this task: ${hardBlocks}/${HARD_BLOCK_CAP}.`,
|
|
265
|
+
`Action: split the remaining Acceptance Criteria into ${id}-b right now and continue from ${lastGreenLine}.`,
|
|
266
|
+
`Do not extend or skip — create ${id}-b carrying the unresolved Acceptance Criteria, run TDD there, and report status when both halves are green.`,
|
|
267
|
+
].join(' ');
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
export function describePause({ id, result }) {
|
|
271
|
+
return [
|
|
272
|
+
`[ukit-watchdog] ${id} crossed its hard budget (elapsed ${result.elapsedMin.toFixed(1)}min > hardMin ${result.hardMin}min) but hardPolicy="pause".`,
|
|
273
|
+
`STOP and hand back to the user — do not auto-split.`,
|
|
274
|
+
].join(' ');
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
export function describeDegraded({ id, hardBlocks }) {
|
|
278
|
+
return [
|
|
279
|
+
`[ukit-watchdog] ${id} has tripped its hard budget ${hardBlocks} times already; degrading to advisory so the orchestrator is not wedged.`,
|
|
280
|
+
`Hand back to the user now with the running progress, last green, and a concrete blocker.`,
|
|
281
|
+
].join(' ');
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
// ─── State bookkeeping ────────────────────────────────────────────────────
|
|
285
|
+
export async function ensureFirstSeen(state, id, now) {
|
|
286
|
+
if (!state.firstSeen[id]) state.firstSeen[id] = now;
|
|
287
|
+
return state.firstSeen[id];
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
export async function bumpHardBlocks(state, id) {
|
|
291
|
+
const next = Number(state.hardBlocks[id] || 0) + 1;
|
|
292
|
+
state.hardBlocks[id] = next;
|
|
293
|
+
return next;
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
export function pickLastGreen(task) {
|
|
297
|
+
if (!task?.lastProgressIso) return null;
|
|
298
|
+
return { iso: task.lastProgressIso, label: 'milestone' };
|
|
299
|
+
}
|
package/templates/.gitignore
CHANGED
|
@@ -48,10 +48,11 @@ Thumbs.db
|
|
|
48
48
|
.cache/
|
|
49
49
|
# legacy: Antigravity adapter removed in v2.2.0; leftovers must stay ignored
|
|
50
50
|
.antigravity/
|
|
51
|
-
.claude/
|
|
52
|
-
.
|
|
53
|
-
|
|
54
|
-
|
|
51
|
+
# NOTE (C12): runtime dirs (.claude/ .codex/ .omp/ .ukit/) are deliberately NOT
|
|
52
|
+
# ignored here. This file is a per-directory .gitignore for templates/, so an
|
|
53
|
+
# unanchored `.claude/` silently hid NEW template files (it swallowed
|
|
54
|
+
# templates/.claude/ukit/index/task-budget-validator.mjs). The root .gitignore
|
|
55
|
+
# already carries the anchored /.claude/ /.codex/ /.omp/ /.ukit/ exclusions.
|
|
55
56
|
opencode.json
|
|
56
57
|
AGENTS.md
|
|
57
58
|
CLAUDE.md
|
|
@@ -116,6 +116,36 @@ what lets the cycle reach the end.
|
|
|
116
116
|
- The next AI session (any tool, model from `handoff.reviewer.model`, MUST differ from executor) will pick `pending_review` task and run review.
|
|
117
117
|
- Do NOT dispatch reviewer in-process unless your host explicitly supports it AND can guarantee a different model — file-based handoff is the default.
|
|
118
118
|
|
|
119
|
+
## Milestone landing (mandatory — Handoff mode)
|
|
120
|
+
|
|
121
|
+
Long runs must land work to disk continuously, so a killed turn loses at most one milestone
|
|
122
|
+
and a respawn picks up exactly where the previous executor stopped. This applies in Handoff
|
|
123
|
+
mode only; daily flow does not get a Progress section.
|
|
124
|
+
|
|
125
|
+
- **Cadence.** Every `handoff.milestoneIntervalMin` (default 5) wall-clock minutes OR at
|
|
126
|
+
every RED → GREEN → verify milestone (whichever comes first), append a Progress entry to
|
|
127
|
+
the task file and make a milestone commit on YOUR worktree branch.
|
|
128
|
+
- **Entry format (fixed — `src/core/taskProgressGuard.js` parses it).**
|
|
129
|
+
|
|
130
|
+
`- <ISO-8601 local timestamp> · milestone: <name> · last-green: <what passed> · files: <comma-separated paths> · drift: none|<one line why>`
|
|
131
|
+
|
|
132
|
+
Re-state every file changed since the previous entry. If a file is not under
|
|
133
|
+
`## Target Files`, the `drift:` field MUST carry a one-line reason; otherwise the guard
|
|
134
|
+
reports `undeclared-drift`.
|
|
135
|
+
- **Milestone commit.** The commit lives on your worktree branch only (e.g.
|
|
136
|
+
`handoff/task-xxx`). Use a `-m "milestone: <name>"` message. Never commit outside your
|
|
137
|
+
worktree, never push. The orchestrator's copy-back step (`git diff --name-only` against
|
|
138
|
+
base) diffs the working tree, so intermediate commits on the worktree branch are free —
|
|
139
|
+
**copy-back is unaffected** by milestone commits.
|
|
140
|
+
- **Resume.** When a previous run aborted, you will be respawned. Read `## Progress`,
|
|
141
|
+
jump straight to the last entry whose `last-green:` is not `none`, and continue from
|
|
142
|
+
there. Do NOT re-plan, do NOT restart from RED, do NOT re-write tests that already
|
|
143
|
+
passed. The progress section is the resume contract.
|
|
144
|
+
- **Staleness.** If your newest entry is older than `2 × handoff.milestoneIntervalMin`,
|
|
145
|
+
the guard reports `stale-milestone`; the boundary is inclusive at exactly `2 ×` and
|
|
146
|
+
exclusive one minute beyond. Keep entries fresh — the cadence rule above is what
|
|
147
|
+
guarantees that.
|
|
148
|
+
|
|
119
149
|
## Rules
|
|
120
150
|
|
|
121
151
|
- **Iron law (Handoff mode):** no `DONE` without fresh PASS output in the current turn.
|
|
@@ -115,6 +115,18 @@ Missing any field → `needs_breakdown`. Never mark incomplete tasks `ready`.
|
|
|
115
115
|
- Chain: A → B → C runs as 3 sequential waves (1 task each, no parallelism)
|
|
116
116
|
- Independent: A, B, C (all `none`) runs as 1 wave, all parallel
|
|
117
117
|
|
|
118
|
+
### Task-budget validator — gate before any task becomes `ready`
|
|
119
|
+
|
|
120
|
+
Before marking a task `ready`, run the task-budget validator on the file you just wrote:
|
|
121
|
+
|
|
122
|
+
```
|
|
123
|
+
node .claude/ukit/index/task-budget-validator.mjs docs/AI_HANDOFF/tasks/TASK-xxx.md
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
The validator's first stdout line is `VERDICT: ok` or `VERDICT: needs_breakdown`, followed by `REASON:` lines. It always exits 0 (advisory). When the verdict is `needs_breakdown`, either split the task into smaller TASK files until each one passes, or record a one-line dismissal in that task's `## Discussion` thread explaining why the violation is acceptable. A `ready` task whose validator verdict is `needs_breakdown` is not actually ready — Phase 3 (state file write) must follow a `VERDICT: ok` (or a documented dismissal).
|
|
127
|
+
|
|
128
|
+
Numeric thresholds and the minute table live in `src/core/taskBudgetValidator.js` and the shipped CLI twin. Do not restate them here — `tests/consistency/configDocsSync.test.js` greps this file for stale numbers and will fail if a threshold is hard-coded. Reference the validator / PLAN §3 instead.
|
|
129
|
+
|
|
118
130
|
### Maximize wave width — dependencies are expensive
|
|
119
131
|
|
|
120
132
|
Wave width is the single biggest lever on how long a cycle takes: a wave of 6 finishes in
|
|
@@ -41,7 +41,7 @@ export const HOOK_EVENT_MAP = {
|
|
|
41
41
|
},
|
|
42
42
|
tool_result: {
|
|
43
43
|
'Read|Grep|Glob': ['record-execution.sh'],
|
|
44
|
-
'Edit|Write': ['post-edit-verify.sh', 'record-execution.sh'],
|
|
44
|
+
'Edit|Write': ['post-edit-verify.sh', 'record-execution.sh', 'task-watchdog.sh'],
|
|
45
45
|
Bash: ['compress-output.sh', 'record-execution.sh'],
|
|
46
46
|
},
|
|
47
47
|
before_agent_start: ['sensitive-data-guard.sh', 'skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
|
|
@@ -98,6 +98,7 @@ export const ADVISORY_SCRIPTS = new Set([
|
|
|
98
98
|
'vision-router.sh',
|
|
99
99
|
'context-window-guard.sh',
|
|
100
100
|
'post-edit-verify.sh',
|
|
101
|
+
'task-watchdog.sh',
|
|
101
102
|
'compress-output.sh',
|
|
102
103
|
'reinject-context.sh',
|
|
103
104
|
'auto-prune-bash.sh',
|
|
@@ -7,6 +7,7 @@ Mọi AI (planner / executor / reviewer) đọc và ghi vào file NÀY. Không t
|
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
- Status: `ready` <!-- ready | in_progress | pending_review | changes_requested | critical_block | approved | approved_minor | blocked | done -->
|
|
10
|
+
- Size: `M` <!-- S | M | L — runtime budgets key off this field; defaults to M when missing -->
|
|
10
11
|
- Owner: `-` <!-- tool đang giữ task -->
|
|
11
12
|
- Reviewer: `-` <!-- model name reviewer dùng, set ở Phase 4 -->
|
|
12
13
|
- Parent plan: `docs/AI_HANDOFF/PLAN.md` §<section>
|
|
@@ -19,6 +20,22 @@ Mọi AI (planner / executor / reviewer) đọc và ghi vào file NÀY. Không t
|
|
|
19
20
|
|
|
20
21
|
- `<path/to/source.js>` — <what changes>
|
|
21
22
|
|
|
23
|
+
## Progress
|
|
24
|
+
|
|
25
|
+
<!--
|
|
26
|
+
Milestone landing (mandatory): append a Progress entry every `handoff.milestoneIntervalMin`
|
|
27
|
+
(default 5) wall-clock minutes AND at each RED → GREEN → verify milestone. Format is fixed so
|
|
28
|
+
`src/core/taskProgressGuard.js` can audit it:
|
|
29
|
+
|
|
30
|
+
- <ISO-8601 local timestamp, e.g. 2026-09-12T10:30+07:00> · milestone: <name> · last-green: <what passed, e.g. tests/core/x.test.js 6/6> · files: <comma-separated paths> · drift: none|<one line why>
|
|
31
|
+
|
|
32
|
+
Each entry must re-state the files changed since the last entry. If a file appears that is NOT
|
|
33
|
+
under `## Target Files`, the `drift:` field MUST carry a one-line reason — otherwise the guard
|
|
34
|
+
reports `undeclared-drift`. The newest entry (max parsed timestamp) must be within
|
|
35
|
+
`2 × handoff.milestoneIntervalMin` of `now`; one minute beyond → `stale-milestone`.
|
|
36
|
+
A respawned executor reads this section and resumes from the last green milestone.
|
|
37
|
+
-->
|
|
38
|
+
|
|
22
39
|
## Test Cases (REQUIRED — TDD)
|
|
23
40
|
|
|
24
41
|
| # | Loại | Tên test | Expected | Pre-state / Fixture |
|
|
@@ -240,7 +240,14 @@
|
|
|
240
240
|
"approved_minor",
|
|
241
241
|
"blocked",
|
|
242
242
|
"done"
|
|
243
|
-
]
|
|
243
|
+
],
|
|
244
|
+
"taskBudgets": {
|
|
245
|
+
"S": { "softMin": 8, "hardMin": 15 },
|
|
246
|
+
"M": { "softMin": 15, "hardMin": 30 },
|
|
247
|
+
"L": { "softMin": 25, "hardMin": 45 },
|
|
248
|
+
"hardPolicy": "split"
|
|
249
|
+
},
|
|
250
|
+
"milestoneIntervalMin": 5
|
|
244
251
|
},
|
|
245
252
|
"subagents": {
|
|
246
253
|
"enabled": true,
|