@ngockhoale/ukit 2.3.15 → 2.3.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,242 @@
1
+ #!/usr/bin/env node
2
+ // task-budget-validator.mjs — TASK-014 / Cycle C12 (shipped CLI twin)
3
+ //
4
+ // Self-contained twin of src/core/taskBudgetValidator.js for user installs (no `src/`).
5
+ // The logic below is a literal port — keep the two files behavior-identical when editing.
6
+ //
7
+ // Usage:
8
+ // node .claude/ukit/index/task-budget-validator.mjs <path-to-TASK-xxx.md> [criteria JSON]
9
+ //
10
+ // Output:
11
+ // First stdout line: `VERDICT: ok` or `VERDICT: needs_breakdown`
12
+ // Then one `REASON: <criterion>: <detail>` line per reason.
13
+ // ALWAYS exits 0 — advisory gate, must never break a pipeline.
14
+
15
+ import fs from 'node:fs';
16
+ import process from 'node:process';
17
+ import { pathToFileURL } from 'node:url';
18
+
19
+ const DEFAULT_CRITERIA = Object.freeze({
20
+ maxTargetFiles: 3,
21
+ maxTestCases: 8,
22
+ maxVerificationMinutes: 10,
23
+ });
24
+
25
+ const VERIFICATION_MINUTE_TABLE = Object.freeze({
26
+ 'yarn test:release-core': 3,
27
+ 'node scripts/release/verify-release.mjs': 2,
28
+ });
29
+
30
+ function estimateVerificationMinutes(commands) {
31
+ if (!Array.isArray(commands)) return 0;
32
+ let total = 0;
33
+ for (const raw of commands) {
34
+ const cmd = String(raw || '').trim();
35
+ if (!cmd) continue;
36
+ if (Object.prototype.hasOwnProperty.call(VERIFICATION_MINUTE_TABLE, cmd)) {
37
+ total += VERIFICATION_MINUTE_TABLE[cmd];
38
+ continue;
39
+ }
40
+ if (/^yarn\s+vitest(\s+run)?\b/.test(cmd)) {
41
+ const tokens = cmd.split(/\s+/);
42
+ const runIdx = tokens.indexOf('run');
43
+ const tail = runIdx >= 0 ? tokens.slice(runIdx + 1) : tokens.slice(2);
44
+ const fileArgs = tail.filter((t) => !t.startsWith('-') && t.length > 0);
45
+ total += fileArgs.length * 0.5;
46
+ continue;
47
+ }
48
+ total += 1;
49
+ }
50
+ return Math.round(total * 10) / 10;
51
+ }
52
+
53
+ function splitSections(markdown) {
54
+ const lines = String(markdown || '').split('\n');
55
+ const sections = new Map();
56
+ let currentHeading = null;
57
+ let buffer = [];
58
+ for (const line of lines) {
59
+ const m = /^##\s+(.+?)\s*$/.exec(line);
60
+ if (m) {
61
+ if (currentHeading !== null) sections.set(currentHeading, buffer);
62
+ currentHeading = m[1].trim();
63
+ buffer = [];
64
+ } else if (currentHeading !== null) {
65
+ buffer.push(line);
66
+ }
67
+ }
68
+ if (currentHeading !== null) sections.set(currentHeading, buffer);
69
+ return sections;
70
+ }
71
+
72
+ function getSectionBody(sections, name) {
73
+ if (sections.has(name)) return sections.get(name).join('\n');
74
+ const normalizedName = name.replace(/\s*\([^)]*\)\s*$/, '');
75
+ for (const [heading, body] of sections) {
76
+ if (heading.replace(/\s*\([^)]*\)\s*$/, '') === normalizedName) {
77
+ return body.join('\n');
78
+ }
79
+ }
80
+ return null;
81
+ }
82
+
83
+ function countTargetFiles(body) {
84
+ if (body == null) return 0;
85
+ let count = 0;
86
+ for (const line of body.split('\n')) {
87
+ if (/^-\s+\S/.test(line)) count += 1;
88
+ }
89
+ return count;
90
+ }
91
+
92
+ function countTestCases(body) {
93
+ if (body == null) return 0;
94
+ let count = 0;
95
+ let sawHeader = false;
96
+ let sawSeparator = false;
97
+ for (const line of body.split('\n')) {
98
+ const trimmed = line.trim();
99
+ if (!trimmed.startsWith('|')) continue;
100
+ if (!sawHeader) {
101
+ sawHeader = true;
102
+ continue;
103
+ }
104
+ if (!sawSeparator && /^\|[-\s|]+\|\s*$/.test(trimmed)) {
105
+ sawSeparator = true;
106
+ continue;
107
+ }
108
+ count += 1;
109
+ }
110
+ return count;
111
+ }
112
+
113
+ function extractVerificationCommands(body) {
114
+ if (body == null) return [];
115
+ const lines = body.split('\n');
116
+ const fenceStart = /^```\s*bash\s*$/;
117
+ const fenceEnd = /^```\s*$/;
118
+ let inFence = false;
119
+ const out = [];
120
+ for (const line of lines) {
121
+ if (!inFence) {
122
+ if (fenceStart.test(line.trim())) inFence = true;
123
+ continue;
124
+ }
125
+ if (fenceEnd.test(line.trim())) break;
126
+ out.push(line);
127
+ }
128
+ return out.map((l) => l.trim()).filter(Boolean);
129
+ }
130
+
131
+ function goalText(body) {
132
+ if (body == null) return '';
133
+ return body
134
+ .split('\n')
135
+ .map((l) => l.replace(/^#+\s*/, '').trim())
136
+ .filter(Boolean)
137
+ .join(' ');
138
+ }
139
+
140
+ export function validateTaskFile(markdown, criteria = DEFAULT_CRITERIA) {
141
+ const reasons = [];
142
+ const text = typeof markdown === 'string' ? markdown : '';
143
+ const sections = splitSections(text);
144
+
145
+ const goalBody = getSectionBody(sections, 'Goal');
146
+ const targetBody = getSectionBody(sections, 'Target Files');
147
+ const testBody = getSectionBody(sections, 'Test Cases');
148
+ const verifyBody = getSectionBody(sections, 'Verification Commands');
149
+
150
+ const requiredSections = [
151
+ ['Goal', goalBody],
152
+ ['Test Cases', testBody],
153
+ ['Verification Commands', verifyBody],
154
+ ];
155
+ for (const [name, body] of requiredSections) {
156
+ if (body == null || body.trim() === '') {
157
+ reasons.push({
158
+ criterion: 'missing-field',
159
+ detail: `required section \`## ${name}\` is missing or empty`,
160
+ });
161
+ }
162
+ }
163
+
164
+ if (targetBody != null) {
165
+ const n = countTargetFiles(targetBody);
166
+ if (n > criteria.maxTargetFiles) {
167
+ reasons.push({
168
+ criterion: 'target-file-count',
169
+ detail: `${n} target files exceeds maxTargetFiles=${criteria.maxTargetFiles}`,
170
+ });
171
+ }
172
+ }
173
+
174
+ if (testBody != null) {
175
+ const n = countTestCases(testBody);
176
+ if (n > criteria.maxTestCases) {
177
+ reasons.push({
178
+ criterion: 'test-case-count',
179
+ detail: `${n} test cases exceeds maxTestCases=${criteria.maxTestCases}`,
180
+ });
181
+ }
182
+ }
183
+
184
+ if (verifyBody != null) {
185
+ const commands = extractVerificationCommands(verifyBody);
186
+ if (commands.length > 0) {
187
+ const minutes = estimateVerificationMinutes(commands);
188
+ if (minutes > criteria.maxVerificationMinutes) {
189
+ reasons.push({
190
+ criterion: 'verification-minutes',
191
+ detail: `estimated ${minutes} verification minutes exceeds maxVerificationMinutes=${criteria.maxVerificationMinutes}`,
192
+ });
193
+ }
194
+ }
195
+ }
196
+
197
+ if (goalBody != null) {
198
+ const prose = goalText(goalBody);
199
+ if (/\binvestigate\b/i.test(prose) && /\bfirst\b/i.test(prose)) {
200
+ reasons.push({
201
+ criterion: 'investigate-first',
202
+ detail: 'Goal contains both "investigate" and "first" — split the task into a spike + a build step',
203
+ });
204
+ }
205
+ }
206
+
207
+ return {
208
+ verdict: reasons.length === 0 ? 'ok' : 'needs_breakdown',
209
+ reasons,
210
+ };
211
+ }
212
+
213
+ function main() {
214
+ const [, , targetPath, criteriaArg] = process.argv;
215
+ let criteria = DEFAULT_CRITERIA;
216
+ if (criteriaArg) {
217
+ try {
218
+ const overrides = JSON.parse(criteriaArg);
219
+ criteria = { ...DEFAULT_CRITERIA, ...overrides };
220
+ } catch {
221
+ // malformed override — fall back to defaults (advisory, never fail)
222
+ }
223
+ }
224
+ let markdown = '';
225
+ try {
226
+ markdown = fs.readFileSync(targetPath, 'utf8');
227
+ } catch (err) {
228
+ console.log('VERDICT: needs_breakdown');
229
+ console.log(`REASON: missing-field: cannot read ${targetPath} (${err && err.code ? err.code : 'error'})`);
230
+ process.exit(0);
231
+ }
232
+ const { verdict, reasons } = validateTaskFile(markdown, criteria);
233
+ console.log(`VERDICT: ${verdict}`);
234
+ for (const r of reasons) {
235
+ console.log(`REASON: ${r.criterion}: ${r.detail}`);
236
+ }
237
+ process.exit(0);
238
+ }
239
+
240
+ if (process.argv[1] && pathToFileURL(process.argv[1]).href === import.meta.url) {
241
+ main();
242
+ }
@@ -938,23 +938,12 @@ async function main() {
938
938
  return;
939
939
  }
940
940
 
941
- // Claude Code invokes Stop again after a Stop hook blocks the first stop. Re-blocking
942
- // that recovery turn creates a self-sustaining loop, so let it end normally instead.
943
- // If work still lacks evidence, surface the recovery reason to the user rather than
944
- // silently ending after the automatic continuation.
945
- // Vibecode autonomy keeps pushing through the reentrant Stop as well: the recovery turn
946
- // after a block is another chance to finish the work, not a release valve. Block again
947
- // with the reason; the loop still ends the moment evidence lands or a blocker appears.
948
- if (payload.stop_hook_active === true && state?.routeSummary?.autonomyLevel !== 'vibecode') {
949
- if (result.continue || result.capped || result.notify) {
950
- const recoveryReason = result.reason
951
- || 'UKit completion gate reached its continuation limit; unfinished work was not retried again.';
952
- process.stdout.write(`${JSON.stringify({
953
- systemMessage: `UKit stopped automatic recovery after one continuation: ${recoveryReason}`,
954
- })}\n`);
955
- }
956
- return;
957
- }
941
+ // A reentrant Stop (Claude Code re-fires Stop after this hook already blocked once) is
942
+ // gated exactly like any other Stop: while evidence is still missing it blocks again
943
+ // with the actionable reason instead of silently releasing the recovery turn. Loop
944
+ // termination is guaranteed elsewhere — non-vibecode routes end at the continuation cap
945
+ // (final-notice block -> capped visible release), and vibecode ends on the visible
946
+ // verification-loop blocker.
958
947
 
959
948
  if (result.continue) {
960
949
  if (result.finalNotice) await markNotified(projectRoot, payload, ledger);
@@ -0,0 +1,299 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * task-watchdog.mjs — wall-clock watchdog runtime for handoff tasks.
4
+ *
5
+ * Self-contained runtime module. No imports from `src/`, no sibling imports
6
+ * beyond `node:` builtins (the `token-utils` file lock is intentionally NOT
7
+ * used — file-locking watchdog state under a Stop hook with a 4s budget is a
8
+ * recipe for getting killed mid-write by its own chain).
9
+ *
10
+ * Exports:
11
+ * DEFAULT_CONFIG — fail-open defaults; identical to the
12
+ * `handoff.taskBudgets` / `handoff.milestoneIntervalMin`
13
+ * keys seeded into templates/ukit/storage/config.json.
14
+ * parseTaskFile(markdown) — minimal twin of src/core/taskProgressGuard's
15
+ * parser; reads id, status, size, lastProgressIso.
16
+ * evaluateBudgets(...) — per in_progress task returns
17
+ * { id, phase: 'ok'|'soft'|'hard', elapsedMin, lastGreenMin }.
18
+ * elapsed = now - state[id].firstSeen; Progress
19
+ * appends do NOT reset the clock.
20
+ * loadConfig(configPath) — deep-merges .ukit/storage/config.json's
21
+ * `handoff` key over DEFAULT_CONFIG; any read/parse
22
+ * error → DEFAULT_CONFIG (fail-open).
23
+ * readState(statePath) / writeState(statePath, value) —
24
+ * `.ukit/storage/cache/task-watchdog/state.json`,
25
+ * shape { firstSeen, hardBlocks }; any I/O error
26
+ * → treat as empty, never throw.
27
+ *
28
+ * The hard clock does not reset on Progress appends on purpose: trickling entries
29
+ * every 4 minutes must not buy an unlimited task. The newest Progress timestamp is
30
+ * used only to name the last-green milestone in user-facing messages.
31
+ */
32
+
33
+ import fs from 'node:fs/promises';
34
+ import path from 'node:path';
35
+
36
+ // ─── Defaults ─────────────────────────────────────────────────────────────
37
+ export const DEFAULT_CONFIG = Object.freeze({
38
+ taskBudgets: {
39
+ S: { softMin: 8, hardMin: 15 },
40
+ M: { softMin: 15, hardMin: 30 },
41
+ L: { softMin: 25, hardMin: 45 },
42
+ hardPolicy: 'split',
43
+ },
44
+ milestoneIntervalMin: 5,
45
+ });
46
+
47
+ // Cap on consecutive hard-trip emissions before degrading to advisory —
48
+ // after the third evaluation the hook stops blocking and the run is allowed
49
+ // to end so the orchestrator can hand back to the user. An honest failure of
50
+ // enforcement is better than a wedged pipeline.
51
+ export const HARD_BLOCK_CAP = 2;
52
+
53
+ // ─── Helpers ──────────────────────────────────────────────────────────────
54
+ function isPlainObject(value) {
55
+ return value !== null && typeof value === 'object' && !Array.isArray(value);
56
+ }
57
+
58
+ function deepFreeze(value) {
59
+ if (!isPlainObject(value)) return value;
60
+ for (const key of Object.keys(value)) deepFreeze(value[key]);
61
+ return Object.freeze(value);
62
+ }
63
+
64
+ function mergeBudgets(into, from) {
65
+ if (!isPlainObject(from)) return into;
66
+ for (const [size, bucket] of Object.entries(from)) {
67
+ if (!isPlainObject(bucket)) continue;
68
+ const prev = isPlainObject(into[size]) ? into[size] : {};
69
+ into[size] = { ...prev, ...bucket };
70
+ }
71
+ return into;
72
+ }
73
+
74
+ // ─── Config ───────────────────────────────────────────────────────────────
75
+ export async function loadConfig(configPath) {
76
+ const fallback = deepFreeze(JSON.parse(JSON.stringify(DEFAULT_CONFIG)));
77
+ if (!configPath) return fallback;
78
+ let raw;
79
+ try {
80
+ raw = await fs.readFile(configPath, 'utf8');
81
+ } catch {
82
+ return fallback;
83
+ }
84
+ let parsed;
85
+ try {
86
+ parsed = JSON.parse(raw);
87
+ } catch {
88
+ return fallback;
89
+ }
90
+ const handoff = parsed?.handoff;
91
+ if (!isPlainObject(handoff)) return fallback;
92
+
93
+ const merged = {
94
+ taskBudgets: {},
95
+ hardPolicy: typeof handoff.taskBudgets?.hardPolicy === 'string'
96
+ ? handoff.taskBudgets.hardPolicy
97
+ : DEFAULT_CONFIG.taskBudgets.hardPolicy,
98
+ milestoneIntervalMin: Number.isFinite(handoff.milestoneIntervalMin)
99
+ ? handoff.milestoneIntervalMin
100
+ : DEFAULT_CONFIG.milestoneIntervalMin,
101
+ };
102
+ mergeBudgets(merged.taskBudgets, DEFAULT_CONFIG.taskBudgets);
103
+ mergeBudgets(merged.taskBudgets, handoff.taskBudgets);
104
+
105
+ // Keep the legacy key visible (matches the seeded config.json shape).
106
+ merged.taskBudgets.hardPolicy = merged.hardPolicy;
107
+ return deepFreeze(merged);
108
+ }
109
+
110
+ // ─── State ────────────────────────────────────────────────────────────────
111
+ const EMPTY_STATE = Object.freeze({ firstSeen: Object.freeze({}), hardBlocks: Object.freeze({}) });
112
+
113
+ export async function readState(statePath) {
114
+ if (!statePath) return { firstSeen: {}, hardBlocks: {} };
115
+ try {
116
+ const raw = await fs.readFile(statePath, 'utf8');
117
+ const parsed = JSON.parse(raw);
118
+ return {
119
+ firstSeen: isPlainObject(parsed?.firstSeen) ? { ...parsed.firstSeen } : {},
120
+ hardBlocks: isPlainObject(parsed?.hardBlocks) ? { ...parsed.hardBlocks } : {},
121
+ };
122
+ } catch {
123
+ return { firstSeen: {}, hardBlocks: {} };
124
+ }
125
+ }
126
+
127
+ export async function writeState(statePath, value) {
128
+ if (!statePath) return;
129
+ try {
130
+ await fs.mkdir(path.dirname(statePath), { recursive: true });
131
+ // temp + rename to avoid leaving a half-written state file if the hook is killed.
132
+ const tempPath = `${statePath}.${process.pid}-${Date.now()}-${Math.random().toString(16).slice(2)}.tmp`;
133
+ await fs.writeFile(tempPath, `${JSON.stringify(value, null, 1)}\n`, 'utf8');
134
+ await fs.rename(tempPath, statePath);
135
+ } catch {
136
+ // Fail-open: never throw from state writes.
137
+ }
138
+ }
139
+
140
+ // ─── Task parser ──────────────────────────────────────────────────────────
141
+ const TASK_ID_REGEX = /^#\s+TASK-([0-9]+[A-Za-z0-9._-]*)\b/m;
142
+ const STATUS_REGEX = /^-\s*Status:\s*`?([a-z_]+)`?/m;
143
+ const SIZE_REGEX = /^-\s*Size:\s*`?([SML])`?/m;
144
+
145
+ // Progress entries follow the format fixed by src/core/taskProgressGuard.js:
146
+ // - <ISO-8601> · milestone: <name> · ...
147
+ const PROGRESS_ENTRY_REGEX = /^-\s+(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2})?(?:[+-]\d{2}:?\d{2}|Z)?)\s+·/gm;
148
+
149
+ export function parseTaskFile(markdown) {
150
+ if (typeof markdown !== 'string') {
151
+ return { id: null, status: null, size: null, lastProgressIso: null };
152
+ }
153
+ const idMatch = markdown.match(TASK_ID_REGEX);
154
+ const statusMatch = markdown.match(STATUS_REGEX);
155
+ const sizeMatch = markdown.match(SIZE_REGEX);
156
+ let lastProgressIso = null;
157
+ PROGRESS_ENTRY_REGEX.lastIndex = 0;
158
+ for (const m of markdown.matchAll(PROGRESS_ENTRY_REGEX)) {
159
+ const candidate = m[1];
160
+ if (!lastProgressIso || candidate > lastProgressIso) lastProgressIso = candidate;
161
+ }
162
+ return {
163
+ id: idMatch ? `TASK-${idMatch[1]}` : null,
164
+ status: statusMatch ? statusMatch[1] : null,
165
+ size: sizeMatch ? sizeMatch[1] : null,
166
+ lastProgressIso,
167
+ };
168
+ }
169
+
170
+ function parseIsoToMs(iso) {
171
+ if (typeof iso !== 'string') return NaN;
172
+ const ms = Date.parse(iso);
173
+ return Number.isFinite(ms) ? ms : NaN;
174
+ }
175
+
176
+ // ─── Discovery ────────────────────────────────────────────────────────────
177
+ export async function listInProgressTasks(projectRoot) {
178
+ const tasksDir = path.join(projectRoot, 'docs', 'AI_HANDOFF', 'tasks');
179
+ let entries;
180
+ try {
181
+ entries = await fs.readdir(tasksDir);
182
+ } catch {
183
+ return [];
184
+ }
185
+ const out = [];
186
+ for (const entry of entries) {
187
+ if (!/^TASK-.*\.md$/i.test(entry)) continue;
188
+ let text;
189
+ try {
190
+ text = await fs.readFile(path.join(tasksDir, entry), 'utf8');
191
+ } catch {
192
+ continue;
193
+ }
194
+ const parsed = parseTaskFile(text);
195
+ if (!parsed.id) continue;
196
+ if (parsed.status !== 'in_progress') continue;
197
+ if (!['S', 'M', 'L'].includes(parsed.size)) parsed.size = 'M';
198
+ out.push(parsed);
199
+ }
200
+ return out;
201
+ }
202
+
203
+ // ─── Budget evaluation ────────────────────────────────────────────────────
204
+ function bucketFor(size, config) {
205
+ const key = ['S', 'M', 'L'].includes(size) ? size : 'M';
206
+ const budgets = config?.taskBudgets || DEFAULT_CONFIG.taskBudgets;
207
+ return budgets[key] || DEFAULT_CONFIG.taskBudgets[key];
208
+ }
209
+
210
+ /**
211
+ * Evaluate every in_progress task against its wall-clock budget.
212
+ *
213
+ * `state` carries `firstSeen[id]` (epochMs) and `hardBlocks[id]` (count).
214
+ * The hard clock does NOT reset on Progress appends — that would let a task
215
+ * trickle entries every 4 minutes and never trip. Progress is used only to
216
+ * name the last-green milestone in user-facing messages.
217
+ */
218
+ export function evaluateBudgets({ tasks = [], state = {}, now, config = DEFAULT_CONFIG } = {}) {
219
+ const out = [];
220
+ for (const task of tasks) {
221
+ const id = task.id;
222
+ const bucket = bucketFor(task.size, config);
223
+ const softMin = Number(bucket?.softMin) || 0;
224
+ const hardMin = Number(bucket?.hardMin) || 0;
225
+ const firstSeen = Number(state?.firstSeen?.[id]) || now;
226
+ const elapsedMs = Math.max(0, now - firstSeen);
227
+ const elapsedMin = elapsedMs / 60_000;
228
+ let phase = 'ok';
229
+ if (elapsedMin > hardMin) phase = 'hard';
230
+ else if (elapsedMin >= softMin) phase = 'soft';
231
+ let lastGreenMin = null;
232
+ if (task.lastProgressIso) {
233
+ const ms = parseIsoToMs(task.lastProgressIso);
234
+ if (Number.isFinite(ms)) lastGreenMin = Math.max(0, (now - ms) / 60_000);
235
+ }
236
+ out.push({ id, phase, elapsedMin, lastGreenMin, size: task.size, softMin, hardMin });
237
+ }
238
+ return out;
239
+ }
240
+
241
+ // ─── Hook-driver helpers ──────────────────────────────────────────────────
242
+ export function describeSoft(result, id) {
243
+ return [
244
+ `[ukit-watchdog] ${id} is in its soft overrun window (elapsed ${result.elapsedMin.toFixed(1)}min; budget ${result.softMin}/${result.hardMin}min).`,
245
+ `Checkpoint a milestone now via \`## Progress\` and split the remaining Acceptance Criteria into ${id}-b before crossing hardMin.`,
246
+ ].join(' ');
247
+ }
248
+
249
+ // Hard-phase advisory for advisory-only contexts (PostToolUse, degraded paths):
250
+ // same split instruction as the split-policy Stop block, but never a decision.
251
+ export function describeHardAdvisory(result, id) {
252
+ return [
253
+ `[ukit-watchdog] ${id} has crossed its hard budget (elapsed ${result.elapsedMin.toFixed(1)}min > hardMin ${result.hardMin}min).`,
254
+ `Checkpoint a milestone via \`## Progress\` and split the remaining Acceptance Criteria into ${id}-b now.`,
255
+ ].join(' ');
256
+ }
257
+
258
+ export function describeSplitReason({ id, result, hardBlocks, lastGreen }) {
259
+ const lastGreenLine = lastGreen?.iso
260
+ ? `last-green: ${lastGreen.iso} (${lastGreen.label || 'milestone'})`
261
+ : `last-green: none recorded`;
262
+ return [
263
+ `UKit watchdog: ${id} crossed its hard budget (elapsed ${result.elapsedMin.toFixed(1)}min > hardMin ${result.hardMin}min).`,
264
+ `Hard-trip count for this task: ${hardBlocks}/${HARD_BLOCK_CAP}.`,
265
+ `Action: split the remaining Acceptance Criteria into ${id}-b right now and continue from ${lastGreenLine}.`,
266
+ `Do not extend or skip — create ${id}-b carrying the unresolved Acceptance Criteria, run TDD there, and report status when both halves are green.`,
267
+ ].join(' ');
268
+ }
269
+
270
+ export function describePause({ id, result }) {
271
+ return [
272
+ `[ukit-watchdog] ${id} crossed its hard budget (elapsed ${result.elapsedMin.toFixed(1)}min > hardMin ${result.hardMin}min) but hardPolicy="pause".`,
273
+ `STOP and hand back to the user — do not auto-split.`,
274
+ ].join(' ');
275
+ }
276
+
277
+ export function describeDegraded({ id, hardBlocks }) {
278
+ return [
279
+ `[ukit-watchdog] ${id} has tripped its hard budget ${hardBlocks} times already; degrading to advisory so the orchestrator is not wedged.`,
280
+ `Hand back to the user now with the running progress, last green, and a concrete blocker.`,
281
+ ].join(' ');
282
+ }
283
+
284
+ // ─── State bookkeeping ────────────────────────────────────────────────────
285
+ export async function ensureFirstSeen(state, id, now) {
286
+ if (!state.firstSeen[id]) state.firstSeen[id] = now;
287
+ return state.firstSeen[id];
288
+ }
289
+
290
+ export async function bumpHardBlocks(state, id) {
291
+ const next = Number(state.hardBlocks[id] || 0) + 1;
292
+ state.hardBlocks[id] = next;
293
+ return next;
294
+ }
295
+
296
+ export function pickLastGreen(task) {
297
+ if (!task?.lastProgressIso) return null;
298
+ return { iso: task.lastProgressIso, label: 'milestone' };
299
+ }
@@ -48,10 +48,11 @@ Thumbs.db
48
48
  .cache/
49
49
  # legacy: Antigravity adapter removed in v2.2.0; leftovers must stay ignored
50
50
  .antigravity/
51
- .claude/
52
- .codex/
53
- .omp/
54
- .ukit/
51
+ # NOTE (C12): runtime dirs (.claude/ .codex/ .omp/ .ukit/) are deliberately NOT
52
+ # ignored here. This file is a per-directory .gitignore for templates/, so an
53
+ # unanchored `.claude/` silently hid NEW template files (it swallowed
54
+ # templates/.claude/ukit/index/task-budget-validator.mjs). The root .gitignore
55
+ # already carries the anchored /.claude/ /.codex/ /.omp/ /.ukit/ exclusions.
55
56
  opencode.json
56
57
  AGENTS.md
57
58
  CLAUDE.md
@@ -116,6 +116,36 @@ what lets the cycle reach the end.
116
116
  - The next AI session (any tool, model from `handoff.reviewer.model`, MUST differ from executor) will pick `pending_review` task and run review.
117
117
  - Do NOT dispatch reviewer in-process unless your host explicitly supports it AND can guarantee a different model — file-based handoff is the default.
118
118
 
119
+ ## Milestone landing (mandatory — Handoff mode)
120
+
121
+ Long runs must land work to disk continuously, so a killed turn loses at most one milestone
122
+ and a respawn picks up exactly where the previous executor stopped. This applies in Handoff
123
+ mode only; daily flow does not get a Progress section.
124
+
125
+ - **Cadence.** Every `handoff.milestoneIntervalMin` (default 5) wall-clock minutes OR at
126
+ every RED → GREEN → verify milestone (whichever comes first), append a Progress entry to
127
+ the task file and make a milestone commit on YOUR worktree branch.
128
+ - **Entry format (fixed — `src/core/taskProgressGuard.js` parses it).**
129
+
130
+ `- <ISO-8601 local timestamp> · milestone: <name> · last-green: <what passed> · files: <comma-separated paths> · drift: none|<one line why>`
131
+
132
+ Re-state every file changed since the previous entry. If a file is not under
133
+ `## Target Files`, the `drift:` field MUST carry a one-line reason; otherwise the guard
134
+ reports `undeclared-drift`.
135
+ - **Milestone commit.** The commit lives on your worktree branch only (e.g.
136
+ `handoff/task-xxx`). Use a `-m "milestone: <name>"` message. Never commit outside your
137
+ worktree, never push. The orchestrator's copy-back step (`git diff --name-only` against
138
+ base) diffs the working tree, so intermediate commits on the worktree branch are free —
139
+ **copy-back is unaffected** by milestone commits.
140
+ - **Resume.** When a previous run aborted, you will be respawned. Read `## Progress`,
141
+ jump straight to the last entry whose `last-green:` is not `none`, and continue from
142
+ there. Do NOT re-plan, do NOT restart from RED, do NOT re-write tests that already
143
+ passed. The progress section is the resume contract.
144
+ - **Staleness.** If your newest entry is older than `2 × handoff.milestoneIntervalMin`,
145
+ the guard reports `stale-milestone`; the boundary is inclusive at exactly `2 ×` and
146
+ exclusive one minute beyond. Keep entries fresh — the cadence rule above is what
147
+ guarantees that.
148
+
119
149
  ## Rules
120
150
 
121
151
  - **Iron law (Handoff mode):** no `DONE` without fresh PASS output in the current turn.
@@ -115,6 +115,18 @@ Missing any field → `needs_breakdown`. Never mark incomplete tasks `ready`.
115
115
  - Chain: A → B → C runs as 3 sequential waves (1 task each, no parallelism)
116
116
  - Independent: A, B, C (all `none`) runs as 1 wave, all parallel
117
117
 
118
+ ### Task-budget validator — gate before any task becomes `ready`
119
+
120
+ Before marking a task `ready`, run the task-budget validator on the file you just wrote:
121
+
122
+ ```
123
+ node .claude/ukit/index/task-budget-validator.mjs docs/AI_HANDOFF/tasks/TASK-xxx.md
124
+ ```
125
+
126
+ The validator's first stdout line is `VERDICT: ok` or `VERDICT: needs_breakdown`, followed by `REASON:` lines. It always exits 0 (advisory). When the verdict is `needs_breakdown`, either split the task into smaller TASK files until each one passes, or record a one-line dismissal in that task's `## Discussion` thread explaining why the violation is acceptable. A `ready` task whose validator verdict is `needs_breakdown` is not actually ready — Phase 3 (state file write) must follow a `VERDICT: ok` (or a documented dismissal).
127
+
128
+ Numeric thresholds and the minute table live in `src/core/taskBudgetValidator.js` and the shipped CLI twin. Do not restate them here — `tests/consistency/configDocsSync.test.js` greps this file for stale numbers and will fail if a threshold is hard-coded. Reference the validator / PLAN §3 instead.
129
+
118
130
  ### Maximize wave width — dependencies are expensive
119
131
 
120
132
  Wave width is the single biggest lever on how long a cycle takes: a wave of 6 finishes in
@@ -41,7 +41,7 @@ export const HOOK_EVENT_MAP = {
41
41
  },
42
42
  tool_result: {
43
43
  'Read|Grep|Glob': ['record-execution.sh'],
44
- 'Edit|Write': ['post-edit-verify.sh', 'record-execution.sh'],
44
+ 'Edit|Write': ['post-edit-verify.sh', 'record-execution.sh', 'task-watchdog.sh'],
45
45
  Bash: ['compress-output.sh', 'record-execution.sh'],
46
46
  },
47
47
  before_agent_start: ['sensitive-data-guard.sh', 'skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
@@ -98,6 +98,7 @@ export const ADVISORY_SCRIPTS = new Set([
98
98
  'vision-router.sh',
99
99
  'context-window-guard.sh',
100
100
  'post-edit-verify.sh',
101
+ 'task-watchdog.sh',
101
102
  'compress-output.sh',
102
103
  'reinject-context.sh',
103
104
  'auto-prune-bash.sh',