@rasensio/aidlc 1.66.0 → 1.72.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/dist/autopilot/driver.d.ts +35 -2
- package/dist/autopilot/driver.d.ts.map +1 -1
- package/dist/autopilot/driver.js +650 -27
- package/dist/autopilot/driver.js.map +1 -1
- package/dist/autopilot/presets.d.ts.map +1 -1
- package/dist/autopilot/presets.js +5 -0
- package/dist/autopilot/presets.js.map +1 -1
- package/dist/autopilot/result.d.ts +133 -0
- package/dist/autopilot/result.d.ts.map +1 -0
- package/dist/autopilot/result.js +431 -0
- package/dist/autopilot/result.js.map +1 -0
- package/dist/autopilot/run.d.ts +2 -0
- package/dist/autopilot/run.d.ts.map +1 -1
- package/dist/autopilot/run.js +33 -6
- package/dist/autopilot/run.js.map +1 -1
- package/dist/autopilot/state.d.ts +51 -0
- package/dist/autopilot/state.d.ts.map +1 -1
- package/dist/autopilot/state.js.map +1 -1
- package/dist/autopilot/typed.d.ts +98 -0
- package/dist/autopilot/typed.d.ts.map +1 -0
- package/dist/autopilot/typed.js +122 -0
- package/dist/autopilot/typed.js.map +1 -0
- package/dist/cli.d.ts.map +1 -1
- package/dist/cli.js +2 -0
- package/dist/cli.js.map +1 -1
- package/dist/commands/hook.d.ts +37 -0
- package/dist/commands/hook.d.ts.map +1 -0
- package/dist/commands/hook.js +208 -0
- package/dist/commands/hook.js.map +1 -0
- package/dist/commands/init.d.ts +18 -0
- package/dist/commands/init.d.ts.map +1 -1
- package/dist/commands/init.js +66 -3
- package/dist/commands/init.js.map +1 -1
- package/dist/commands/knowledge.d.ts.map +1 -1
- package/dist/commands/knowledge.js +8 -4
- package/dist/commands/knowledge.js.map +1 -1
- package/dist/commands/roadmap.d.ts +20 -3
- package/dist/commands/roadmap.d.ts.map +1 -1
- package/dist/commands/roadmap.js +60 -15
- package/dist/commands/roadmap.js.map +1 -1
- package/dist/commands/update.d.ts.map +1 -1
- package/dist/commands/update.js +8 -2
- package/dist/commands/update.js.map +1 -1
- package/dist/compile/adapters/claude-code.d.ts.map +1 -1
- package/dist/compile/adapters/claude-code.js +35 -4
- package/dist/compile/adapters/claude-code.js.map +1 -1
- package/dist/compile/adapters/codex.d.ts +12 -0
- package/dist/compile/adapters/codex.d.ts.map +1 -1
- package/dist/compile/adapters/codex.js +58 -8
- package/dist/compile/adapters/codex.js.map +1 -1
- package/dist/compile/loaders.d.ts.map +1 -1
- package/dist/compile/loaders.js +2 -0
- package/dist/compile/loaders.js.map +1 -1
- package/dist/compile/skill-metadata.d.ts +82 -0
- package/dist/compile/skill-metadata.d.ts.map +1 -0
- package/dist/compile/skill-metadata.js +156 -0
- package/dist/compile/skill-metadata.js.map +1 -0
- package/dist/concurrency/claim-store.d.ts +17 -0
- package/dist/concurrency/claim-store.d.ts.map +1 -1
- package/dist/concurrency/claim-store.js +42 -2
- package/dist/concurrency/claim-store.js.map +1 -1
- package/dist/concurrency/common-dir.d.ts +15 -0
- package/dist/concurrency/common-dir.d.ts.map +1 -1
- package/dist/concurrency/common-dir.js +39 -0
- package/dist/concurrency/common-dir.js.map +1 -1
- package/dist/core/types.d.ts +17 -0
- package/dist/core/types.d.ts.map +1 -1
- package/dist/cost/hooks/arm.d.ts +9 -1
- package/dist/cost/hooks/arm.d.ts.map +1 -1
- package/dist/cost/hooks/arm.js +16 -2
- package/dist/cost/hooks/arm.js.map +1 -1
- package/dist/cost/providers/codex.d.ts +13 -0
- package/dist/cost/providers/codex.d.ts.map +1 -1
- package/dist/cost/providers/codex.js +74 -2
- package/dist/cost/providers/codex.js.map +1 -1
- package/dist/doctor/migrations/index.d.ts.map +1 -1
- package/dist/doctor/migrations/index.js +5 -0
- package/dist/doctor/migrations/index.js.map +1 -1
- package/dist/doctor/migrations/roadmap-mermaid-graphs.d.ts +22 -0
- package/dist/doctor/migrations/roadmap-mermaid-graphs.d.ts.map +1 -0
- package/dist/doctor/migrations/roadmap-mermaid-graphs.js +74 -0
- package/dist/doctor/migrations/roadmap-mermaid-graphs.js.map +1 -0
- package/dist/hooks/guard.d.ts +41 -0
- package/dist/hooks/guard.d.ts.map +1 -0
- package/dist/hooks/guard.js +108 -0
- package/dist/hooks/guard.js.map +1 -0
- package/dist/hooks/install.d.ts +78 -0
- package/dist/hooks/install.d.ts.map +1 -0
- package/dist/hooks/install.js +207 -0
- package/dist/hooks/install.js.map +1 -0
- package/dist/hooks/lifecycle-context.d.ts +65 -0
- package/dist/hooks/lifecycle-context.d.ts.map +1 -0
- package/dist/hooks/lifecycle-context.js +212 -0
- package/dist/hooks/lifecycle-context.js.map +1 -0
- package/dist/knowledge/engine.d.ts +19 -2
- package/dist/knowledge/engine.d.ts.map +1 -1
- package/dist/knowledge/engine.js +35 -11
- package/dist/knowledge/engine.js.map +1 -1
- package/dist/knowledge/index/knowledge-index.d.ts +28 -2
- package/dist/knowledge/index/knowledge-index.d.ts.map +1 -1
- package/dist/knowledge/index/sqlite-index.d.ts +19 -1
- package/dist/knowledge/index/sqlite-index.d.ts.map +1 -1
- package/dist/knowledge/index/sqlite-index.js +66 -5
- package/dist/knowledge/index/sqlite-index.js.map +1 -1
- package/dist/knowledge/source-fingerprint.d.ts +71 -0
- package/dist/knowledge/source-fingerprint.d.ts.map +1 -0
- package/dist/knowledge/source-fingerprint.js +105 -0
- package/dist/knowledge/source-fingerprint.js.map +1 -0
- package/dist/knowledge/sync.d.ts +12 -1
- package/dist/knowledge/sync.d.ts.map +1 -1
- package/dist/knowledge/sync.js +23 -3
- package/dist/knowledge/sync.js.map +1 -1
- package/dist/knowledge/types.d.ts +10 -1
- package/dist/knowledge/types.d.ts.map +1 -1
- package/dist/knowledge/types.js +10 -1
- package/dist/knowledge/types.js.map +1 -1
- package/dist/menu/roster.d.ts.map +1 -1
- package/dist/menu/roster.js +2 -0
- package/dist/menu/roster.js.map +1 -1
- package/dist/roadmap/graph.d.ts +73 -28
- package/dist/roadmap/graph.d.ts.map +1 -1
- package/dist/roadmap/graph.js +184 -125
- package/dist/roadmap/graph.js.map +1 -1
- package/dist/roadmap/next.d.ts +5 -0
- package/dist/roadmap/next.d.ts.map +1 -1
- package/dist/roadmap/next.js +7 -1
- package/dist/roadmap/next.js.map +1 -1
- package/dist/roadmap/open-browser.d.ts +34 -0
- package/dist/roadmap/open-browser.d.ts.map +1 -0
- package/dist/roadmap/open-browser.js +40 -0
- package/dist/roadmap/open-browser.js.map +1 -0
- package/dist/roadmap/paths.d.ts +5 -1
- package/dist/roadmap/paths.d.ts.map +1 -1
- package/dist/roadmap/paths.js +5 -9
- package/dist/roadmap/paths.js.map +1 -1
- package/dist/roadmap/view-assets.d.ts +18 -0
- package/dist/roadmap/view-assets.d.ts.map +1 -0
- package/dist/roadmap/view-assets.js +301 -0
- package/dist/roadmap/view-assets.js.map +1 -0
- package/dist/roadmap/view-html.d.ts +37 -0
- package/dist/roadmap/view-html.d.ts.map +1 -0
- package/dist/roadmap/view-html.js +317 -0
- package/dist/roadmap/view-html.js.map +1 -0
- package/dist/roadmap/view-model.d.ts +93 -0
- package/dist/roadmap/view-model.d.ts.map +1 -0
- package/dist/roadmap/view-model.js +194 -0
- package/dist/roadmap/view-model.js.map +1 -0
- package/package.json +2 -2
package/dist/autopilot/driver.js
CHANGED
|
@@ -15,7 +15,12 @@
|
|
|
15
15
|
*
|
|
16
16
|
* @module
|
|
17
17
|
*/
|
|
18
|
+
import { mapUnits } from '../cost/providers/codex.js';
|
|
19
|
+
import { microToUsd, priceUnits } from '../cost/rates.js';
|
|
20
|
+
import { MODEL_ID_RE } from '../cost/types.js';
|
|
21
|
+
import { REASON_MAX, RESET_WAIT_MAX_MS, codexEnd, emptyCodexProgress, firstLine, readClaudeResult, readCodexEvents, refusalSuffix, resetAt, stoppedBy, summaryLines, } from './result.js';
|
|
18
22
|
import { LEGACY_STOPS, pauseEnv, pendingReview, stopsInForce } from './stops.js';
|
|
23
|
+
import { OUTCOME_SCHEMA, UUID_RE, budgetCents, resumePrompt, typedArgv, typedHarness, withOutcomeContract, } from './typed.js';
|
|
19
24
|
const RATE_LIMIT_RE = /rate limit|\b429\b|secondary rate/i;
|
|
20
25
|
const BACKOFF_FIRST_MS = 5 * 60 * 1000;
|
|
21
26
|
const BACKOFF_MAX_MS = 2 * 60 * 60 * 1000;
|
|
@@ -39,16 +44,27 @@ function makeIO(ctx) {
|
|
|
39
44
|
ctx.say(`[dry-run] ${opts.cwd ? `(in ${opts.cwd}) ` : ''}${[cmd, ...args].join(' ')}`);
|
|
40
45
|
return { status: 0, stdout: '', stderr: '', dry: true };
|
|
41
46
|
}
|
|
42
|
-
const r = ctx.run(cmd, args, opts);
|
|
43
|
-
if (RATE_LIMIT_RE.test(`${r.stdout}\n${r.stderr}`)) {
|
|
47
|
+
const r = ctx.run(cmd, args, opts.cwd === undefined ? {} : { cwd: opts.cwd });
|
|
48
|
+
if (opts.echoes !== true && RATE_LIMIT_RE.test(`${r.stdout}\n${r.stderr}`)) {
|
|
44
49
|
throw new RateLimited(`${cmd} ${args.slice(0, 2).join(' ')}: ${(r.stderr || r.stdout).trim().split('\n')[0]}`);
|
|
45
50
|
}
|
|
46
51
|
return r;
|
|
47
52
|
};
|
|
48
|
-
const cli = (args) => call(ctx.cli[0], [...ctx.cli.slice(1), ...args], { cwd: ctx.repo });
|
|
53
|
+
const cli = (args, opts = {}) => call(ctx.cli[0], [...ctx.cli.slice(1), ...args], { cwd: ctx.repo, ...opts });
|
|
49
54
|
return { call, cli };
|
|
50
55
|
}
|
|
51
56
|
const fail = (r) => (r.stderr || r.stdout || `exit ${r.status}`).trim().split('\n').slice(-1)[0];
|
|
57
|
+
/**
|
|
58
|
+
* Git's whole message on one line, at most 400 characters. `fail` keeps the last
|
|
59
|
+
* line, and a refused merge's last line is `Aborting` — the files it would
|
|
60
|
+
* overwrite are the lines before (autopilot-merge-refusal-is-not-a-conflict/AC-7).
|
|
61
|
+
*/
|
|
62
|
+
const gitMessage = (r) => (r.stderr.trim() ? r.stderr : r.stdout || `exit ${r.status}`)
|
|
63
|
+
.split('\n')
|
|
64
|
+
.map((l) => l.trim())
|
|
65
|
+
.filter(Boolean)
|
|
66
|
+
.join(' ')
|
|
67
|
+
.slice(0, 400);
|
|
52
68
|
/**
|
|
53
69
|
* Why `aidlc transition --json` refused: each violation, where `fail` alone would keep
|
|
54
70
|
* the last line — a "next step" naming no cause, which is all a halted queue logged
|
|
@@ -158,16 +174,13 @@ const itemPauses = (item) => pauseEnv(itemStops(item), item.approved ?? []);
|
|
|
158
174
|
* fails later prints a banner first, which names nothing (codex-preset-removed-flag/AC-6, AC-10).
|
|
159
175
|
*/
|
|
160
176
|
export const LAUNCH_FAILURE_MS = 30_000;
|
|
161
|
-
const REASON_MAX = 200;
|
|
162
177
|
/**
|
|
163
178
|
* The first non-blank line of `text`, terminal escapes and control characters removed,
|
|
164
179
|
* at most 200 characters; null when there is none (codex-preset-removed-flag/AC-12).
|
|
165
180
|
* The first line, not `fail`'s last: for an argument error the last is the usage line.
|
|
166
181
|
*/
|
|
167
182
|
export function firstOutputLine(text) {
|
|
168
|
-
|
|
169
|
-
const line = plain.split('\n').map((l) => l.trim()).find((l) => l !== '');
|
|
170
|
-
return line === undefined ? null : line.slice(0, REASON_MAX);
|
|
183
|
+
return firstLine(text, REASON_MAX);
|
|
171
184
|
}
|
|
172
185
|
/**
|
|
173
186
|
* The park reason for an agent that failed to launch, or null when it did not or the
|
|
@@ -183,9 +196,92 @@ function launchFailure(ctx, item) {
|
|
|
183
196
|
const line = firstOutputLine(exit.output);
|
|
184
197
|
return line === null ? `agent failed to launch (${how}) with no output` : `agent failed to launch (${how}): ${line}`;
|
|
185
198
|
}
|
|
186
|
-
|
|
199
|
+
/**
|
|
200
|
+
* The reset time of the latest session-limit wait still in force, or null. While there is
|
|
201
|
+
* one, no agent is launched for any item: the limit is the account's, so every launch
|
|
202
|
+
* would hit it (autopilot-reads-agent-result/AC-26).
|
|
203
|
+
*/
|
|
204
|
+
export function launchBlocked(state, now) {
|
|
205
|
+
let latest = null;
|
|
206
|
+
for (const i of state.items) {
|
|
207
|
+
const w = i.resume_wait;
|
|
208
|
+
const until = w?.kind === 'session-limit' ? Date.parse(w.until) : Number.NaN;
|
|
209
|
+
if (until > now && (latest === null || until > latest))
|
|
210
|
+
latest = until;
|
|
211
|
+
}
|
|
212
|
+
// Re-rendered, not repeated: the stored text came from a file that can be edited.
|
|
213
|
+
return latest === null ? null : new Date(latest).toISOString();
|
|
214
|
+
}
|
|
215
|
+
/** The park reason when the item's spend has used up its cap (autopilot-reads-agent-result/AC-7, AC-18, AC-20). */
|
|
216
|
+
const capReached = (cap, spentMicro) => `cost cap ${cap} USD reached — spent ${microToUsd(spentMicro)}`;
|
|
217
|
+
const logFile = (ctx, item) => `${ctx.paths.logs}/${item.name}.log`;
|
|
218
|
+
/**
|
|
219
|
+
* Launch the item's agent. False when nothing was launched: a session-limit wait is in
|
|
220
|
+
* force — the net under each caller's own check (design D7) — or the item was parked,
|
|
221
|
+
* for a budget under a cent (autopilot-reads-agent-result/AC-18) or a session id that is
|
|
222
|
+
* not a UUID (AC-33).
|
|
223
|
+
*
|
|
224
|
+
* A preset command is launched typed (AC-1, AC-2): the outcome contract in the prompt
|
|
225
|
+
* (AC-5), the flags, the budget (AC-17), and the launch recorded on the item with where
|
|
226
|
+
* its output starts, so a pass in any process reads its result (AC-32). A custom
|
|
227
|
+
* command is launched exactly as configured, and clears any earlier mark (AC-3, AC-31).
|
|
228
|
+
*
|
|
229
|
+
* @param budget - the run's item budget, which a park here must reach.
|
|
230
|
+
*/
|
|
231
|
+
function launchAgent(ctx, io, state, item, prompt, budget, resume) {
|
|
232
|
+
if (launchBlocked(state, ctx.now()) !== null)
|
|
233
|
+
return false;
|
|
234
|
+
const cap = ctx.config?.cost_cap_usd;
|
|
235
|
+
const spent = item.spent_micro_usd ?? 0;
|
|
236
|
+
const cents = typeof cap === 'number' ? budgetCents(cap, spent) : null;
|
|
237
|
+
if (typeof cap === 'number' && (cents ?? 0) < 1) {
|
|
238
|
+
park(ctx, io, state, item, capReached(cap, spent), budget);
|
|
239
|
+
return false;
|
|
240
|
+
}
|
|
187
241
|
const command = ctx.config?.agent_command ?? [];
|
|
188
|
-
const
|
|
242
|
+
const harness = typedHarness(command);
|
|
243
|
+
const log = logFile(ctx, item);
|
|
244
|
+
let argv;
|
|
245
|
+
let launch;
|
|
246
|
+
if (harness === null) {
|
|
247
|
+
argv = command.map((a) => (a === '{prompt}' ? prompt : a));
|
|
248
|
+
}
|
|
249
|
+
else {
|
|
250
|
+
const outcomeFile = `${ctx.paths.dir}/outcomes/${item.name}.json`;
|
|
251
|
+
const schemaFile = `${ctx.paths.dir}/outcome-schema.json`;
|
|
252
|
+
const typed = typedArgv(command, {
|
|
253
|
+
harness,
|
|
254
|
+
prompt: withOutcomeContract(prompt),
|
|
255
|
+
schemaFile,
|
|
256
|
+
outcomeFile,
|
|
257
|
+
maxBudgetCents: harness === 'claude' ? cents : null,
|
|
258
|
+
resume: resume?.session ?? null,
|
|
259
|
+
});
|
|
260
|
+
if (typed === null) {
|
|
261
|
+
park(ctx, io, state, item, `${resume?.label ?? 'session'} — no usable session id`, budget);
|
|
262
|
+
return false;
|
|
263
|
+
}
|
|
264
|
+
argv = typed;
|
|
265
|
+
launch = {
|
|
266
|
+
harness,
|
|
267
|
+
at: new Date(ctx.now()).toISOString(),
|
|
268
|
+
...(typeof cap === 'number' ? { cap_usd: cap } : {}),
|
|
269
|
+
log_offset: 0,
|
|
270
|
+
...(harness === 'codex' ? { outcome_file: outcomeFile, progress: emptyCodexProgress() } : {}),
|
|
271
|
+
...(resume === undefined ? {} : { resumes: resume.session }),
|
|
272
|
+
};
|
|
273
|
+
// A resume continues the thread's rollout, which sits in its first launch's day
|
|
274
|
+
// directory, and Codex reports the model there once (design D10).
|
|
275
|
+
if (resume?.from?.harness === 'codex') {
|
|
276
|
+
if (resume.from.model !== undefined)
|
|
277
|
+
launch.model = resume.from.model;
|
|
278
|
+
launch.rollout_at = resume.from.rollout_at ?? resume.from.at;
|
|
279
|
+
}
|
|
280
|
+
if (harness === 'codex' && !ctx.dryRun) {
|
|
281
|
+
ctx.fs.write(schemaFile, `${JSON.stringify(OUTCOME_SCHEMA, null, 2)}\n`);
|
|
282
|
+
ctx.fs.write(outcomeFile, '');
|
|
283
|
+
}
|
|
284
|
+
}
|
|
189
285
|
const env = {
|
|
190
286
|
AIDLC_UNATTENDED: '1',
|
|
191
287
|
AIDLC_AGENT: ctx.config?.agent_platform ?? 'claude-code',
|
|
@@ -193,10 +289,21 @@ function launchAgent(ctx, item, prompt) {
|
|
|
193
289
|
AIDLC_PAUSE_AFTER: itemPauses(item),
|
|
194
290
|
};
|
|
195
291
|
if (ctx.dryRun) {
|
|
196
|
-
ctx.say(`[dry-run] (in ${item.worktree}) launch agent: ${argv[0]} … (${argv.length} args)`);
|
|
197
|
-
return;
|
|
292
|
+
ctx.say(`[dry-run] (in ${item.worktree}) launch agent: ${argv[0]} … (${argv.length} args${harness === null ? '' : `, typed ${harness}`})`);
|
|
293
|
+
return true;
|
|
294
|
+
}
|
|
295
|
+
if (launch === undefined) {
|
|
296
|
+
delete item.launch;
|
|
198
297
|
}
|
|
199
|
-
|
|
298
|
+
else {
|
|
299
|
+
// Read before the spawn, not after: only this item's agent writes its log, and it is
|
|
300
|
+
// not running yet, so nothing can write in between.
|
|
301
|
+
launch.log_offset = ctx.fs.size?.(log) ?? 0;
|
|
302
|
+
if (launch.harness === 'codex')
|
|
303
|
+
launch.read_offset = launch.log_offset;
|
|
304
|
+
item.launch = launch;
|
|
305
|
+
}
|
|
306
|
+
item.pid = ctx.spawnDetached(argv, { cwd: item.worktree, env, logFile: log });
|
|
200
307
|
item.pid_started_at = new Date(ctx.now()).toISOString();
|
|
201
308
|
const id = ctx.identify?.(item.pid);
|
|
202
309
|
if (id) {
|
|
@@ -205,6 +312,7 @@ function launchAgent(ctx, item, prompt) {
|
|
|
205
312
|
item.pid_ns = id.ns;
|
|
206
313
|
}
|
|
207
314
|
tell(ctx, item, `agent started (pid ${item.pid})`);
|
|
315
|
+
return true;
|
|
208
316
|
}
|
|
209
317
|
function drop(state, item, budget) {
|
|
210
318
|
state.items = state.items.filter((i) => i !== item);
|
|
@@ -212,7 +320,7 @@ function drop(state, item, budget) {
|
|
|
212
320
|
budget?.onFinish(item.name);
|
|
213
321
|
}
|
|
214
322
|
function park(ctx, io, state, item, reason, budget) {
|
|
215
|
-
const r = io.cli(['roadmap', 'park', item.name, '--reason', reason]);
|
|
323
|
+
const r = io.cli(['roadmap', 'park', item.name, '--reason', reason], { echoes: true });
|
|
216
324
|
if (r.status !== 0) {
|
|
217
325
|
failItem(ctx, state, item, `park failed: ${fail(r)}`);
|
|
218
326
|
tell(ctx, item, `could not park — ${item.error}`);
|
|
@@ -223,10 +331,16 @@ function park(ctx, io, state, item, reason, budget) {
|
|
|
223
331
|
notice(ctx, 'item parked', `${item.name}: ${reason}`);
|
|
224
332
|
drop(state, item, budget);
|
|
225
333
|
}
|
|
226
|
-
/**
|
|
334
|
+
/**
|
|
335
|
+
* The instance's `current_phase` in its worktree, or null when unreadable. Shared with the
|
|
336
|
+
* view. Only a phase-shaped name: the agent writes this file, and the phase reaches park
|
|
337
|
+
* reasons committed to the roadmap (autopilot-reads-agent-result/AC-29).
|
|
338
|
+
*/
|
|
227
339
|
export function instancePhase(read, item) {
|
|
228
340
|
const yaml = read(`${item.worktree}/.aidlc/state/${item.name}/instance.yaml`) ?? '';
|
|
229
|
-
|
|
341
|
+
// A name of word characters and hyphens ending at a space or the line's end: what `\S+`
|
|
342
|
+
// read for every real phase — a trailing comment too — without its escapes.
|
|
343
|
+
return /^current_phase:[ \t]*([A-Za-z0-9_-]{1,32})(?=\s|$)/m.exec(yaml)?.[1] ?? null;
|
|
230
344
|
}
|
|
231
345
|
function currentPhase(ctx, item) {
|
|
232
346
|
return instancePhase(ctx.fs.read, item);
|
|
@@ -252,10 +366,225 @@ function awaitIfPaused(ctx, item) {
|
|
|
252
366
|
notice(ctx, 'review needed', `${item.name}: paused after ${phase} — aidlc autopilot approve ${item.name}`);
|
|
253
367
|
return true;
|
|
254
368
|
}
|
|
369
|
+
/** How much of a log one read takes, and one pass at most (design, "While the agent runs"). */
|
|
370
|
+
const CHUNK_BYTES = 1024 * 1024;
|
|
371
|
+
const PASS_READ_BYTES = 8 * CHUNK_BYTES;
|
|
372
|
+
/**
|
|
373
|
+
* Fold what a Codex launch appended to the item's log since the last pass into its
|
|
374
|
+
* progress, at most 8 MiB a pass (autopilot-reads-agent-result/AC-20). Offsets are bytes,
|
|
375
|
+
* and a chunk is decoded only up to its last newline byte, so a character split across
|
|
376
|
+
* two chunks is never miscounted. A line still being written waits for the next pass —
|
|
377
|
+
* unless `final`, when the agent has exited and the line will never be finished. A line
|
|
378
|
+
* longer than a chunk is skipped; its tail fails to parse and is counted unreadable.
|
|
379
|
+
*
|
|
380
|
+
* @returns whether the whole log has been folded.
|
|
381
|
+
*/
|
|
382
|
+
function followCodex(ctx, item, launch, final) {
|
|
383
|
+
const readRange = ctx.fs.readRange;
|
|
384
|
+
if (readRange === undefined)
|
|
385
|
+
return true;
|
|
386
|
+
const progress = (launch.progress ??= emptyCodexProgress());
|
|
387
|
+
const log = logFile(ctx, item);
|
|
388
|
+
let offset = launch.read_offset ?? launch.log_offset;
|
|
389
|
+
for (let left = PASS_READ_BYTES; left > 0;) {
|
|
390
|
+
const want = Math.min(CHUNK_BYTES, left);
|
|
391
|
+
const chunk = readRange(log, offset, want);
|
|
392
|
+
if (chunk.length === 0)
|
|
393
|
+
break;
|
|
394
|
+
const newline = chunk.lastIndexOf(0x0a);
|
|
395
|
+
if (newline === -1) {
|
|
396
|
+
if (chunk.length === CHUNK_BYTES) {
|
|
397
|
+
offset += chunk.length;
|
|
398
|
+
left -= chunk.length;
|
|
399
|
+
continue;
|
|
400
|
+
}
|
|
401
|
+
// The end of the log, mid-line: finished only when the agent will write no more.
|
|
402
|
+
if (final && chunk.length < want) {
|
|
403
|
+
readCodexEvents(chunk.toString('utf8'), progress);
|
|
404
|
+
offset += chunk.length;
|
|
405
|
+
}
|
|
406
|
+
break;
|
|
407
|
+
}
|
|
408
|
+
readCodexEvents(chunk.toString('utf8', 0, newline + 1), progress);
|
|
409
|
+
offset += newline + 1;
|
|
410
|
+
left -= newline + 1;
|
|
411
|
+
}
|
|
412
|
+
launch.read_offset = offset;
|
|
413
|
+
return offset >= (ctx.fs.size?.(log) ?? offset);
|
|
414
|
+
}
|
|
415
|
+
/** Rollout lookups for a thread's model before its launch is treated as unpriced (design D10). */
|
|
416
|
+
const MODEL_LOOKUPS = 3;
|
|
417
|
+
/** The model a Codex launch used, looked up at most three times, kept once found. */
|
|
418
|
+
function codexModelOf(ctx, launch) {
|
|
419
|
+
if (launch.model !== undefined)
|
|
420
|
+
return launch.model;
|
|
421
|
+
const thread = launch.progress?.thread_id ?? null;
|
|
422
|
+
if (thread === null || ctx.codexModel === undefined || (launch.model_lookups ?? 0) >= MODEL_LOOKUPS)
|
|
423
|
+
return null;
|
|
424
|
+
launch.model_lookups = (launch.model_lookups ?? 0) + 1;
|
|
425
|
+
const model = ctx.codexModel(thread, launch.rollout_at ?? launch.at);
|
|
426
|
+
// The shape the cost ledger accepts, and nothing else reaches state or a message (AC-21).
|
|
427
|
+
if (model !== null && MODEL_ID_RE.test(model))
|
|
428
|
+
launch.model = model;
|
|
429
|
+
return launch.model ?? null;
|
|
430
|
+
}
|
|
431
|
+
/**
|
|
432
|
+
* Whether the model lookup has given up for good, which is when "cannot be read" is said
|
|
433
|
+
* (AC-21) — after three tries, or at once when a turn completed with no usable thread id,
|
|
434
|
+
* since `thread.started` comes first or never.
|
|
435
|
+
*/
|
|
436
|
+
const modelLookupGaveUp = (launch) => launch.model === undefined &&
|
|
437
|
+
((launch.model_lookups ?? 0) >= MODEL_LOOKUPS || (launch.progress?.thread_id == null && (launch.progress?.turns ?? 0) > 0));
|
|
438
|
+
/**
|
|
439
|
+
* What a Codex launch's summed `turn.completed` usage costs, in micro-USD, priced as the
|
|
440
|
+
* Codex cost provider prices a turn — cached input counted inside input, not on top of
|
|
441
|
+
* it — at the effective rate for its model; null when it cannot be priced
|
|
442
|
+
* (autopilot-reads-agent-result/AC-20, AC-21). The one place the reading of Codex's
|
|
443
|
+
* usage lives: if the release run finds it a running total, this changes and nothing
|
|
444
|
+
* else does (design, Risk 1).
|
|
445
|
+
*/
|
|
446
|
+
export function launchUsageMicro(ctx, launch) {
|
|
447
|
+
const usage = launch.progress?.usage;
|
|
448
|
+
if (usage === undefined)
|
|
449
|
+
return null;
|
|
450
|
+
const model = codexModelOf(ctx, launch);
|
|
451
|
+
const rates = ctx.rates?.();
|
|
452
|
+
if (model === null || rates === undefined)
|
|
453
|
+
return null;
|
|
454
|
+
const units = mapUnits({
|
|
455
|
+
input_tokens: usage.input,
|
|
456
|
+
cached_input_tokens: usage.cached,
|
|
457
|
+
cache_write_input_tokens: 0,
|
|
458
|
+
output_tokens: usage.output,
|
|
459
|
+
reasoning_output_tokens: 0,
|
|
460
|
+
total_tokens: usage.input + usage.output,
|
|
461
|
+
}, model);
|
|
462
|
+
const micro = priceUnits(units, rates);
|
|
463
|
+
// A figure the state file cannot hold is no figure: JSON writes Infinity as null, and the
|
|
464
|
+
// next load would read the spend as 0.
|
|
465
|
+
return micro !== null && Number.isSafeInteger(micro) ? micro : null;
|
|
466
|
+
}
|
|
467
|
+
/** What a running launch's session amounts to so far: a Codex launch's folded events; nothing for Claude. */
|
|
468
|
+
function sessionSoFar(launch, costMicro) {
|
|
469
|
+
return launch.harness === 'codex' && launch.progress !== undefined ? codexEnd(launch.progress, null, costMicro) : null;
|
|
470
|
+
}
|
|
471
|
+
/**
|
|
472
|
+
* Stop a running Codex launch whose priced usage, on top of the item's recorded spend,
|
|
473
|
+
* has reached the cap less the 10% margin (autopilot-reads-agent-result/AC-20). Codex has
|
|
474
|
+
* no budget flag, so this is the cap inside its run. Usage that cannot be priced is said
|
|
475
|
+
* once and stops nothing (AC-21).
|
|
476
|
+
*
|
|
477
|
+
* @returns whether it stopped and parked the item.
|
|
478
|
+
*/
|
|
479
|
+
function stoppedAtCap(ctx, io, state, item, launch, cap, budget) {
|
|
480
|
+
if ((launch.progress?.turns ?? 0) === 0)
|
|
481
|
+
return false;
|
|
482
|
+
const priced = launchUsageMicro(ctx, launch);
|
|
483
|
+
if (priced === null) {
|
|
484
|
+
noteUnpriced(ctx, item, launch);
|
|
485
|
+
return false;
|
|
486
|
+
}
|
|
487
|
+
const spent = (item.spent_micro_usd ?? 0) + priced;
|
|
488
|
+
if (spent < Math.round(cap * 1e6) - Math.round(cap * 1e5))
|
|
489
|
+
return false;
|
|
490
|
+
if (!ctx.dryRun && item.pid !== null)
|
|
491
|
+
ctx.kill(item.pid);
|
|
492
|
+
appendSummary(ctx, item, summaryLines(sessionSoFar(launch, priced), 'stopped by autopilot at the cost cap'));
|
|
493
|
+
park(ctx, io, state, item, capReached(cap, spent), budget);
|
|
494
|
+
return true;
|
|
495
|
+
}
|
|
496
|
+
/**
|
|
497
|
+
* Say once for the item that its Codex spend cannot be priced, and where a rate goes
|
|
498
|
+
* (autopilot-reads-agent-result/AC-21). Only once it is known why: the model is read and
|
|
499
|
+
* the rate table has nothing for it, or the lookup has given up — not while the rollout
|
|
500
|
+
* may still turn up.
|
|
501
|
+
*/
|
|
502
|
+
function noteUnpriced(ctx, item, launch) {
|
|
503
|
+
if (item.unpriced_noted === true)
|
|
504
|
+
return;
|
|
505
|
+
const why = launch.model !== undefined ? `no rate for ${launch.model}` : modelLookupGaveUp(launch) ? "the session's model cannot be read" : null;
|
|
506
|
+
if (why === null)
|
|
507
|
+
return;
|
|
508
|
+
item.unpriced_noted = true;
|
|
509
|
+
tell(ctx, item, `Codex spend cannot be priced: ${why}; a rate for its model goes under cost.rates in .aidlc/config.yaml — until then this run is not stopped for cost`);
|
|
510
|
+
}
|
|
511
|
+
/** How much of a Claude launch's segment is read for its result, and of a Codex outcome file. */
|
|
512
|
+
const RESULT_TAIL_BYTES = 1024 * 1024;
|
|
513
|
+
const OUTCOME_FILE_BYTES = 64 * 1024;
|
|
514
|
+
/**
|
|
515
|
+
* Read how a typed launch's session ended, once: store it on the launch, add its cost to
|
|
516
|
+
* the item's spend, and append the summary to the item's log. Read from the offset the
|
|
517
|
+
* launch recorded, so it is this launch's result whichever process saw it exit
|
|
518
|
+
* (autopilot-reads-agent-result/AC-6, AC-19, AC-28, AC-32).
|
|
519
|
+
*
|
|
520
|
+
* @returns false while a Codex launch's log is still being folded: its end is decided
|
|
521
|
+
* only once everything it wrote has been read, over several passes if need be.
|
|
522
|
+
*/
|
|
523
|
+
function endSession(ctx, item, launch) {
|
|
524
|
+
const { readRange, size } = ctx.fs;
|
|
525
|
+
if (readRange === undefined || size === undefined) {
|
|
526
|
+
launch.end = null;
|
|
527
|
+
return true;
|
|
528
|
+
}
|
|
529
|
+
const log = logFile(ctx, item);
|
|
530
|
+
let end;
|
|
531
|
+
if (launch.harness === 'claude') {
|
|
532
|
+
const total = size(log);
|
|
533
|
+
const start = Math.max(launch.log_offset, total - RESULT_TAIL_BYTES);
|
|
534
|
+
end = readClaudeResult(readRange(log, start, Math.max(0, total - start)).toString('utf8'));
|
|
535
|
+
}
|
|
536
|
+
else {
|
|
537
|
+
if (!followCodex(ctx, item, launch, true))
|
|
538
|
+
return false;
|
|
539
|
+
const outcome = launch.outcome_file === undefined ? null : readRange(launch.outcome_file, 0, OUTCOME_FILE_BYTES).toString('utf8');
|
|
540
|
+
end = codexEnd(launch.progress ?? emptyCodexProgress(), outcome, launchUsageMicro(ctx, launch));
|
|
541
|
+
}
|
|
542
|
+
launch.end = end;
|
|
543
|
+
if (end !== null)
|
|
544
|
+
recordSpend(item, end);
|
|
545
|
+
appendSummary(ctx, item, summaryLines(end));
|
|
546
|
+
return true;
|
|
547
|
+
}
|
|
548
|
+
/**
|
|
549
|
+
* Add a session's cost to the item's spend (autopilot-reads-agent-result/AC-19). A
|
|
550
|
+
* session already on record adds only its increase. Claude reports a session's cost;
|
|
551
|
+
* a resumed Codex launch's usage counts only its own turns, so a Codex session's cost
|
|
552
|
+
* is what was recorded plus this launch's (design, Risk 1). Unpriced adds nothing.
|
|
553
|
+
*/
|
|
554
|
+
function recordSpend(item, end) {
|
|
555
|
+
if (end.cost_micro === null)
|
|
556
|
+
return;
|
|
557
|
+
const costs = item.session_costs ?? {};
|
|
558
|
+
const id = end.session_id;
|
|
559
|
+
const previous = id === null ? 0 : costs[id] ?? 0;
|
|
560
|
+
const sessionCost = end.harness === 'codex' ? previous + end.cost_micro : end.cost_micro;
|
|
561
|
+
item.spent_micro_usd = (item.spent_micro_usd ?? 0) + Math.max(0, sessionCost - previous);
|
|
562
|
+
if (id !== null) {
|
|
563
|
+
costs[id] = Math.max(previous, sessionCost);
|
|
564
|
+
item.session_costs = costs;
|
|
565
|
+
}
|
|
566
|
+
}
|
|
567
|
+
/** Append lines to the item's log, on a line of their own. Nothing in a dry run. */
|
|
568
|
+
function appendSummary(ctx, item, lines) {
|
|
569
|
+
const { append, readRange, size } = ctx.fs;
|
|
570
|
+
if (ctx.dryRun || append === undefined)
|
|
571
|
+
return;
|
|
572
|
+
const log = logFile(ctx, item);
|
|
573
|
+
const total = size?.(log) ?? 0;
|
|
574
|
+
const last = total > 0 ? readRange?.(log, total - 1, 1) : undefined;
|
|
575
|
+
const lead = last !== undefined && last.length === 1 && last[0] !== 0x0a ? '\n' : '';
|
|
576
|
+
append(log, `${lead}${lines.join('\n')}\n`);
|
|
577
|
+
}
|
|
255
578
|
function advanceBuilding(ctx, io, state, item, budget) {
|
|
256
579
|
if (item.pid && ctx.isAlive(item.pid)) {
|
|
580
|
+
const launch = item.launch;
|
|
581
|
+
if (launch?.harness === 'codex')
|
|
582
|
+
followCodex(ctx, item, launch, false);
|
|
257
583
|
const cap = ctx.config?.cost_cap_usd;
|
|
258
584
|
if (typeof cap === 'number') {
|
|
585
|
+
if (launch?.harness === 'codex' && stoppedAtCap(ctx, io, state, item, launch, cap, budget))
|
|
586
|
+
return;
|
|
587
|
+
// Every agent, typed or not: the ledger's view of the cap stands (autopilot-reads-agent-result/AC-22).
|
|
259
588
|
const r = io.cli(['cost', item.name, '--json']);
|
|
260
589
|
let micro = 0;
|
|
261
590
|
try {
|
|
@@ -267,19 +596,44 @@ function advanceBuilding(ctx, io, state, item, budget) {
|
|
|
267
596
|
if (micro > cap * 1e6) {
|
|
268
597
|
if (!ctx.dryRun)
|
|
269
598
|
ctx.kill(item.pid);
|
|
599
|
+
if (launch !== undefined) {
|
|
600
|
+
const priced = launch.harness === 'codex' ? launchUsageMicro(ctx, launch) : null;
|
|
601
|
+
appendSummary(ctx, item, summaryLines(sessionSoFar(launch, priced), 'stopped by autopilot: aidlc cost passed the cap'));
|
|
602
|
+
}
|
|
270
603
|
park(ctx, io, state, item, `cost cap ${cap} USD exceeded`, budget);
|
|
271
604
|
}
|
|
272
605
|
}
|
|
273
606
|
return;
|
|
274
607
|
}
|
|
608
|
+
// First, once per launch: every session that ended gets its summary, one that parked
|
|
609
|
+
// itself included, and a paused item's spend is on record before an approval works
|
|
610
|
+
// out the relaunch's budget (autopilot-reads-agent-result/AC-19, AC-28).
|
|
611
|
+
if (item.launch !== undefined && item.launch.end === undefined && !endSession(ctx, item, item.launch))
|
|
612
|
+
return;
|
|
275
613
|
if (ctx.fs.exists(`${ctx.repo}/.aidlc/roadmap/hold/${item.file}`)) {
|
|
276
614
|
tell(ctx, item, 'the agent parked it');
|
|
277
615
|
notice(ctx, 'item parked', `${item.name}: parked by its agent — the reason is in the item's history`);
|
|
278
616
|
drop(state, item, budget);
|
|
279
617
|
return;
|
|
280
618
|
}
|
|
281
|
-
|
|
619
|
+
// After hold/: a person who parked a waiting item by hand has the last word.
|
|
620
|
+
if (item.resume_wait !== undefined) {
|
|
621
|
+
resumeWhenDue(ctx, io, state, item, item.resume_wait, budget);
|
|
622
|
+
return;
|
|
623
|
+
}
|
|
624
|
+
if (awaitIfPaused(ctx, item)) {
|
|
625
|
+
// The result led to a pause; after the approval it must not decide again (design D8).
|
|
626
|
+
delete item.launch;
|
|
627
|
+
return;
|
|
628
|
+
}
|
|
629
|
+
// Before the launch-failure check: a typed launch that exited inside 30 seconds but
|
|
630
|
+
// wrote a result is decided by it (autopilot-reads-agent-result/AC-6, AC-14).
|
|
631
|
+
const end = item.launch?.end;
|
|
632
|
+
if (item.launch !== undefined && end !== undefined && end !== null) {
|
|
633
|
+
decideFromResult(ctx, io, state, item, item.launch, end, budget);
|
|
282
634
|
return;
|
|
635
|
+
}
|
|
636
|
+
// Untyped, or no readable result: as before (autopilot-reads-agent-result/AC-15, AC-31).
|
|
283
637
|
// Before the phase: a relaunch that never started would otherwise push and open
|
|
284
638
|
// the pull request as if it had finished (codex-preset-removed-flag/AC-9).
|
|
285
639
|
const launch = launchFailure(ctx, item);
|
|
@@ -287,19 +641,197 @@ function advanceBuilding(ctx, io, state, item, budget) {
|
|
|
287
641
|
park(ctx, io, state, item, launch, budget);
|
|
288
642
|
return;
|
|
289
643
|
}
|
|
644
|
+
byPhase(ctx, io, state, item, budget);
|
|
645
|
+
}
|
|
646
|
+
const stopLabel = (kind) => kind === 'connection' ? 'connection dropped' : kind === 'ended-early' ? 'ended before the release boundary' : 'session limit';
|
|
647
|
+
const waitingLine = (until, session) => `waiting for the session limit to reset at ${until}, then resuming session ${session}`;
|
|
648
|
+
/**
|
|
649
|
+
* A session that stopped on a dropped connection or a session limit is resumed once with
|
|
650
|
+
* its own session id — at the next pass, or once the limit resets — and parked when it
|
|
651
|
+
* cannot be (autopilot-reads-agent-result/AC-23..AC-27, AC-33). The item stays `building`
|
|
652
|
+
* with no agent while it waits (AC-26).
|
|
653
|
+
*/
|
|
654
|
+
function decideResume(ctx, io, state, item, launch, end, error, stop, budget) {
|
|
655
|
+
const label = stopLabel(stop);
|
|
656
|
+
const line = firstLine(error, REASON_MAX) ?? 'error';
|
|
657
|
+
// A resume that stops again is AC-27's whether or not it reported its id again: a
|
|
658
|
+
// resumed Codex thread writes no second `thread.started`.
|
|
659
|
+
const resumed = launch.resumes !== undefined && UUID_RE.test(launch.resumes) ? launch.resumes : null;
|
|
660
|
+
const id = end.session_id ?? resumed;
|
|
661
|
+
if (launch.resumes !== undefined && id !== null) {
|
|
662
|
+
park(ctx, io, state, item, `${label} — resumed once, stopped again: ${line}; session ${id}`, budget);
|
|
663
|
+
return;
|
|
664
|
+
}
|
|
665
|
+
if (id === null) {
|
|
666
|
+
park(ctx, io, state, item, `${label} — no usable session id: ${line}`, budget);
|
|
667
|
+
return;
|
|
668
|
+
}
|
|
669
|
+
const now = ctx.now();
|
|
670
|
+
item.pid = null;
|
|
671
|
+
if (stop === 'connection') {
|
|
672
|
+
item.resume_wait = { kind: 'connection', until: new Date(now).toISOString(), session: id };
|
|
673
|
+
tell(ctx, item, `connection dropped — resuming session ${id} at the next pass`);
|
|
674
|
+
return;
|
|
675
|
+
}
|
|
676
|
+
const until = resetAt(error, now);
|
|
677
|
+
const at = new Date(until).toISOString();
|
|
678
|
+
// A weekly limit: hours of a build slot held, and every other launch blocked, for one resume (AC-25).
|
|
679
|
+
if (until - now > RESET_WAIT_MAX_MS) {
|
|
680
|
+
park(ctx, io, state, item, `session limit — resets ${at}; session ${id}`, budget);
|
|
681
|
+
return;
|
|
682
|
+
}
|
|
683
|
+
item.resume_wait = { kind: 'session-limit', until: at, session: id };
|
|
684
|
+
tell(ctx, item, waitingLine(at, id));
|
|
685
|
+
}
|
|
686
|
+
/**
|
|
687
|
+
* A waiting item, each pass: wait on until its time, then resume its session with the
|
|
688
|
+
* same preset flags and a budget worked out again (autopilot-reads-agent-result/AC-23,
|
|
689
|
+
* AC-24, AC-26). The id has been in `state.json` since it was read, so it is checked
|
|
690
|
+
* again here (AC-33); a resume needs the harness that started the session.
|
|
691
|
+
*/
|
|
692
|
+
function resumeWhenDue(ctx, io, state, item, wait, budget) {
|
|
693
|
+
const label = stopLabel(wait.kind);
|
|
694
|
+
// Before the id is printed anywhere: it came back from a file that can be edited.
|
|
695
|
+
if (!UUID_RE.test(wait.session)) {
|
|
696
|
+
delete item.resume_wait;
|
|
697
|
+
park(ctx, io, state, item, `${label} — no usable session id`, budget);
|
|
698
|
+
return;
|
|
699
|
+
}
|
|
700
|
+
const until = Date.parse(wait.until);
|
|
701
|
+
if (ctx.now() < until) {
|
|
702
|
+
// The same text keeps the time it was first said, so the view shows how long it has waited.
|
|
703
|
+
tell(ctx, item, waitingLine(new Date(until).toISOString(), wait.session));
|
|
704
|
+
return;
|
|
705
|
+
}
|
|
706
|
+
const blocked = launchBlocked(state, ctx.now());
|
|
707
|
+
if (blocked !== null) {
|
|
708
|
+
tell(ctx, item, `${label} — resuming session ${wait.session} waits for the session limit to reset at ${blocked}`);
|
|
709
|
+
return;
|
|
710
|
+
}
|
|
711
|
+
delete item.resume_wait;
|
|
712
|
+
const from = item.launch;
|
|
713
|
+
if (from === undefined || typedHarness(ctx.config?.agent_command ?? []) !== from.harness) {
|
|
714
|
+
park(ctx, io, state, item, `${label} — the agent command changed since the session stopped; session ${wait.session}`, budget);
|
|
715
|
+
return;
|
|
716
|
+
}
|
|
717
|
+
tell(ctx, item, `${label} — resuming session ${wait.session}`);
|
|
718
|
+
// The stopped session's claim outlives it by the 60-minute staleness, and a resume may
|
|
719
|
+
// run under another id, whose `aidlc claim` it would refuse. From the worktree: an
|
|
720
|
+
// in-flight instance's state is on its branch, and `release` refuses an instance it
|
|
721
|
+
// cannot find. `echoes`: it prints the instance name (unattended-agent-ends-turn-to-wait/AC-7).
|
|
722
|
+
const released = io.call(ctx.cli[0], [...ctx.cli.slice(1), 'release', item.name], { cwd: item.worktree, echoes: true });
|
|
723
|
+
if (released.status !== 0) {
|
|
724
|
+
const said = `${label} — could not release the claim: ${fail(released)}; resuming anyway`;
|
|
725
|
+
tell(ctx, item, said);
|
|
726
|
+
// The item's own log too: it is where the resumed session's summary lands next.
|
|
727
|
+
appendSummary(ctx, item, [`[autopilot] ${said}`]);
|
|
728
|
+
}
|
|
729
|
+
launchAgent(ctx, io, state, item, resumePrompt(item, label), budget, { session: wait.session, label, from });
|
|
730
|
+
}
|
|
731
|
+
/** At deployment, the pull request; short of it, parked. `suffix` follows the park reason. */
|
|
732
|
+
function byPhase(ctx, io, state, item, budget, suffix = '') {
|
|
290
733
|
const phase = currentPhase(ctx, item);
|
|
291
734
|
if (phase !== 'deployment') {
|
|
292
|
-
park(ctx, io, state, item, `agent exited before the release boundary (phase ${phase ?? 'unknown'})`, budget);
|
|
735
|
+
park(ctx, io, state, item, `agent exited before the release boundary (phase ${phase ?? 'unknown'})${suffix}`, budget);
|
|
293
736
|
return;
|
|
294
737
|
}
|
|
295
738
|
openPullRequest(ctx, io, state, item);
|
|
296
739
|
}
|
|
740
|
+
/**
|
|
741
|
+
* What a typed session's own result says happens next — first match wins
|
|
742
|
+
* (autopilot-reads-agent-result/AC-7..AC-13). Every park decided here ends with the
|
|
743
|
+
* refused tool calls, after any other suffix (AC-16). Only the pull request keeps the
|
|
744
|
+
* launch's end: a failed push resumed by `aidlc autopilot run` decides the same way
|
|
745
|
+
* again; every other row parks the item or waits to resume it (design D8).
|
|
746
|
+
*/
|
|
747
|
+
function decideFromResult(ctx, io, state, item, launch, end, budget) {
|
|
748
|
+
const refusals = refusalSuffix(end);
|
|
749
|
+
const parkWith = (reason) => park(ctx, io, state, item, `${reason}${refusals}`, budget);
|
|
750
|
+
const spent = item.spent_micro_usd ?? 0;
|
|
751
|
+
if (end.limit === 'budget') {
|
|
752
|
+
// The cap the budget came from, so a config edited mid-run cannot leave the reason without one.
|
|
753
|
+
parkWith(launch.cap_usd === undefined ? `budget reached — spent ${microToUsd(spent)}` : capReached(launch.cap_usd, spent));
|
|
754
|
+
return;
|
|
755
|
+
}
|
|
756
|
+
if (end.limit === 'turns') {
|
|
757
|
+
parkWith(`turn limit reached after ${end.turns ?? 'an unknown number of'} turns`);
|
|
758
|
+
return;
|
|
759
|
+
}
|
|
760
|
+
if (end.error !== null) {
|
|
761
|
+
const stop = stoppedBy(end.error);
|
|
762
|
+
if (stop !== null) {
|
|
763
|
+
decideResume(ctx, io, state, item, launch, end, end.error, stop, budget);
|
|
764
|
+
return;
|
|
765
|
+
}
|
|
766
|
+
parkWith(`agent session failed: ${firstLine(end.error, REASON_MAX) ?? 'error'}`);
|
|
767
|
+
return;
|
|
768
|
+
}
|
|
769
|
+
const early = (reason) => endedEarly(ctx, io, state, item, launch, end, reason, budget);
|
|
770
|
+
if (end.outcome === null) {
|
|
771
|
+
// `byPhase`, with the early ending where it would park.
|
|
772
|
+
const phase = currentPhase(ctx, item);
|
|
773
|
+
if (phase === 'deployment')
|
|
774
|
+
openPullRequest(ctx, io, state, item);
|
|
775
|
+
else
|
|
776
|
+
early(`agent exited before the release boundary (phase ${phase ?? 'unknown'}) — the agent reported no outcome`);
|
|
777
|
+
return;
|
|
778
|
+
}
|
|
779
|
+
const { outcome, reason } = end.outcome;
|
|
780
|
+
const phase = currentPhase(ctx, item) ?? 'unknown';
|
|
781
|
+
if (outcome === 'parked') {
|
|
782
|
+
parkWith(`agent parked at ${phase}: ${reason}`);
|
|
783
|
+
return;
|
|
784
|
+
}
|
|
785
|
+
if (outcome === 'blocked') {
|
|
786
|
+
early(`agent blocked at ${phase}: ${reason}`);
|
|
787
|
+
return;
|
|
788
|
+
}
|
|
789
|
+
if (outcome === 'shipped' && phase === 'deployment') {
|
|
790
|
+
openPullRequest(ctx, io, state, item);
|
|
791
|
+
return;
|
|
792
|
+
}
|
|
793
|
+
// A `paused` that awaits review was handled before this; one that does not is a stop
|
|
794
|
+
// the phase files do not bear out (AC-11).
|
|
795
|
+
early(`agent reported ${outcome}, but the instance is at ${phase}`);
|
|
796
|
+
}
|
|
797
|
+
/**
|
|
798
|
+
* A session that ended before the release boundary with no stop autopilot honours — no
|
|
799
|
+
* outcome short of deployment, `blocked`, or a `shipped` or `paused` its phase does not
|
|
800
|
+
* bear out. Most likely it ended its turn to wait for a command, which under `claude -p`
|
|
801
|
+
* ends the process and the command with it. Resumed once at the next pass, like a dropped
|
|
802
|
+
* connection; parked with `reason` — today's text — when it cannot be
|
|
803
|
+
* (unattended-agent-ends-turn-to-wait/AC-3, AC-5, AC-6).
|
|
804
|
+
*/
|
|
805
|
+
function endedEarly(ctx, io, state, item, launch, end, reason, budget) {
|
|
806
|
+
const refusals = refusalSuffix(end);
|
|
807
|
+
// A CI-fix or conflict relaunch: its pull request is open, so it is past the release
|
|
808
|
+
// boundary, and the early-ending prompt would be wrong about it. Parked as before.
|
|
809
|
+
if (item.pr) {
|
|
810
|
+
park(ctx, io, state, item, `${reason}${refusals}`, budget);
|
|
811
|
+
return;
|
|
812
|
+
}
|
|
813
|
+
// As `decideResume` reads it: a resumed Codex thread writes no second `thread.started`.
|
|
814
|
+
const resumed = launch.resumes !== undefined && UUID_RE.test(launch.resumes) ? launch.resumes : null;
|
|
815
|
+
const id = end.session_id ?? resumed;
|
|
816
|
+
// A resume of any kind: a session is resumed once (AC-5).
|
|
817
|
+
if (launch.resumes !== undefined) {
|
|
818
|
+
park(ctx, io, state, item, `${reason}; resumed once${id === null ? '' : `, session ${id}`}${refusals}`, budget);
|
|
819
|
+
return;
|
|
820
|
+
}
|
|
821
|
+
if (id === null) {
|
|
822
|
+
park(ctx, io, state, item, `${reason}${refusals}`, budget);
|
|
823
|
+
return;
|
|
824
|
+
}
|
|
825
|
+
item.pid = null;
|
|
826
|
+
item.resume_wait = { kind: 'ended-early', until: new Date(ctx.now()).toISOString(), session: id };
|
|
827
|
+
tell(ctx, item, `ended before the release boundary — resuming session ${id} at the next pass`);
|
|
828
|
+
}
|
|
297
829
|
/**
|
|
298
830
|
* The pass after `aidlc autopilot approve`: wait again at the next completed pause,
|
|
299
831
|
* open the pull request when the agent is already at deployment, or relaunch it to
|
|
300
832
|
* carry on (autopilot-stop-controls/AC-18).
|
|
301
833
|
*/
|
|
302
|
-
function advanceApproved(ctx, io, state, item) {
|
|
834
|
+
function advanceApproved(ctx, io, state, item, budget) {
|
|
303
835
|
if (awaitIfPaused(ctx, item))
|
|
304
836
|
return;
|
|
305
837
|
if (currentPhase(ctx, item) === 'deployment') {
|
|
@@ -311,9 +843,13 @@ function advanceApproved(ctx, io, state, item) {
|
|
|
311
843
|
// since this pass serves it before `startBuilds` runs.
|
|
312
844
|
if (building(state) >= buildSlots(ctx))
|
|
313
845
|
return;
|
|
846
|
+
// And for the session limit, still `approved` (autopilot-reads-agent-result/AC-26).
|
|
847
|
+
if (launchBlocked(state, ctx.now()) !== null)
|
|
848
|
+
return;
|
|
314
849
|
item.step = 'building';
|
|
315
850
|
tell(ctx, item, 'approved — relaunching the agent to carry on');
|
|
316
|
-
|
|
851
|
+
// The run's budget: a park for the cap here must reach `run.finished`, or `--items n` never ends.
|
|
852
|
+
launchAgent(ctx, io, state, item, agentPrompt(item, itemPauses(item)), budget);
|
|
317
853
|
}
|
|
318
854
|
function openPullRequest(ctx, io, state, item) {
|
|
319
855
|
const push = io.call('git', ['push', '-u', 'origin', item.branch], { cwd: item.worktree });
|
|
@@ -445,11 +981,55 @@ function handleConflict(ctx, io, state, item, budget) {
|
|
|
445
981
|
park(ctx, io, state, item, `conflicts with main after ${CONFLICT_FIX_LIMIT} fix attempts`, budget);
|
|
446
982
|
return;
|
|
447
983
|
}
|
|
984
|
+
const blocked = launchBlocked(state, ctx.now());
|
|
985
|
+
if (blocked !== null) {
|
|
986
|
+
// No attempt is made, so none is counted; out of the merge queue, so the items
|
|
987
|
+
// behind it can merge meanwhile (autopilot-reads-agent-result/AC-26, design D7).
|
|
988
|
+
dequeue(state, item);
|
|
989
|
+
item.step = 'pr-open';
|
|
990
|
+
tell(ctx, item, `conflicts with main — the fix waits for the session limit to reset at ${blocked}`);
|
|
991
|
+
return;
|
|
992
|
+
}
|
|
448
993
|
item.conflict_fixes = tries + 1;
|
|
449
994
|
dequeue(state, item);
|
|
450
995
|
item.step = 'building';
|
|
451
996
|
tell(ctx, item, `conflicts with main — relaunching the agent to resolve it (attempt ${tries + 1} of ${CONFLICT_FIX_LIMIT})`);
|
|
452
|
-
launchAgent(ctx, item, conflictPrompt(item));
|
|
997
|
+
launchAgent(ctx, io, state, item, conflictPrompt(item), budget);
|
|
998
|
+
}
|
|
999
|
+
/**
|
|
1000
|
+
* The two `merge=union` cost ledgers a worktree collects between commits. `name` is
|
|
1001
|
+
* the one `aidlc start` created and the agent claimed, so it holds no glob or
|
|
1002
|
+
* option character (autopilot-merge-refusal-is-not-a-conflict/AC-1).
|
|
1003
|
+
*/
|
|
1004
|
+
const ledgerPaths = (name) => ['.aidlc/cost/unattributed.ndjson', `.aidlc/state/${name}/costs.ndjson`];
|
|
1005
|
+
const LEDGER_COMMIT_MESSAGE = 'chore(aidlc): commit cost ledgers before updating from main (autopilot)';
|
|
1006
|
+
/**
|
|
1007
|
+
* Commit the worktree's uncommitted cost ledgers, and nothing else, so git does not
|
|
1008
|
+
* refuse the update from main over a ledger both sides appended to — PR #125 on
|
|
1009
|
+
* 2026-10-06 (autopilot-merge-refusal-is-not-a-conflict/AC-1..AC-4). Null when
|
|
1010
|
+
* nothing needed committing or the commit was made; otherwise git's failure.
|
|
1011
|
+
*/
|
|
1012
|
+
function commitLedgers(ctx, io, item) {
|
|
1013
|
+
const wt = { cwd: item.worktree };
|
|
1014
|
+
const candidates = ledgerPaths(item.name);
|
|
1015
|
+
const status = io.call('git', ['status', '--porcelain=v1', '--untracked-files=all', '--', ...candidates], wt);
|
|
1016
|
+
if (status.status !== 0)
|
|
1017
|
+
return status;
|
|
1018
|
+
// Exact matches only, so a quoted or renamed path fails safe; a deleted ledger is
|
|
1019
|
+
// never committed as a deletion (AC-2).
|
|
1020
|
+
const lines = status.stdout.split('\n');
|
|
1021
|
+
const dirty = candidates.filter((p) => lines.some((l) => l.slice(3) === p && !l.slice(0, 2).includes('D')));
|
|
1022
|
+
if (dirty.length === 0)
|
|
1023
|
+
return null;
|
|
1024
|
+
const add = io.call('git', ['add', '--', ...dirty], wt);
|
|
1025
|
+
if (add.status !== 0)
|
|
1026
|
+
return add;
|
|
1027
|
+
// By pathspec: another staged change stays staged and outside this commit (AC-2).
|
|
1028
|
+
const commit = io.call('git', ['commit', '-q', '-m', LEDGER_COMMIT_MESSAGE, '--', ...dirty], wt);
|
|
1029
|
+
if (commit.status !== 0)
|
|
1030
|
+
return commit;
|
|
1031
|
+
tell(ctx, item, 'committed its cost ledgers before updating from main');
|
|
1032
|
+
return null;
|
|
453
1033
|
}
|
|
454
1034
|
/**
|
|
455
1035
|
* Bring the item's branch up to date with main in its own worktree, where the
|
|
@@ -469,16 +1049,35 @@ function upToDateWithMain(ctx, io, state, item, resume, budget) {
|
|
|
469
1049
|
return false;
|
|
470
1050
|
}
|
|
471
1051
|
const contains = io.call('git', ['merge-base', '--is-ancestor', 'origin/main', 'HEAD'], wt);
|
|
1052
|
+
// Nothing is committed on this path: the head CI passed on is the head merged (AC-3).
|
|
472
1053
|
if (contains.status === 0)
|
|
473
1054
|
return true;
|
|
474
1055
|
if (contains.status !== 1) {
|
|
475
1056
|
failItem(ctx, state, item, `update from main failed: ${fail(contains)}`, resume);
|
|
476
1057
|
return false;
|
|
477
1058
|
}
|
|
1059
|
+
// Here, not at the end of the build: the cost hook can write after the agent's
|
|
1060
|
+
// last commit, and the head is about to move anyway (design D1).
|
|
1061
|
+
const ledgers = commitLedgers(ctx, io, item);
|
|
1062
|
+
if (ledgers !== null) {
|
|
1063
|
+
failItem(ctx, state, item, `update from main failed: could not commit the cost ledgers: ${gitMessage(ledgers)}`, resume);
|
|
1064
|
+
return false;
|
|
1065
|
+
}
|
|
478
1066
|
const merge = io.call('git', ['merge', '--no-edit', 'origin/main'], wt);
|
|
479
1067
|
if (merge.status !== 0) {
|
|
1068
|
+
// Read before the abort clears them. A merge git refused — local changes it would
|
|
1069
|
+
// overwrite — leaves none, and an agent relaunched over it found "a clean merge"
|
|
1070
|
+
// seven minutes later (autopilot-merge-refusal-is-not-a-conflict/AC-5, AC-8).
|
|
1071
|
+
const unmerged = io.call('git', ['diff', '--name-only', '--diff-filter=U'], wt);
|
|
1072
|
+
// Fails harmlessly when the merge never started.
|
|
480
1073
|
io.call('git', ['merge', '--abort'], wt);
|
|
481
|
-
|
|
1074
|
+
if (unmerged.status === 0 && unmerged.stdout.trim() !== '') {
|
|
1075
|
+
handleConflict(ctx, io, state, item, budget);
|
|
1076
|
+
}
|
|
1077
|
+
else {
|
|
1078
|
+
// A check that cannot answer is no conflict either: an agent cannot mend git (AC-6).
|
|
1079
|
+
failItem(ctx, state, item, `update from main failed: ${gitMessage(merge)}`, resume);
|
|
1080
|
+
}
|
|
482
1081
|
return false;
|
|
483
1082
|
}
|
|
484
1083
|
dequeue(state, item);
|
|
@@ -538,14 +1137,23 @@ function advancePr(ctx, io, state, item, budget) {
|
|
|
538
1137
|
return;
|
|
539
1138
|
}
|
|
540
1139
|
item.ci = 'failed';
|
|
541
|
-
|
|
542
|
-
if (
|
|
1140
|
+
const failures = (item.ci_failures ?? 0) + 1;
|
|
1141
|
+
if (failures >= CI_FAILURE_LIMIT) {
|
|
1142
|
+
item.ci_failures = failures;
|
|
543
1143
|
park(ctx, io, state, item, `CI failed twice: ${failing.join(', ')}`, budget);
|
|
544
1144
|
return;
|
|
545
1145
|
}
|
|
1146
|
+
const blocked = launchBlocked(state, ctx.now());
|
|
1147
|
+
if (blocked !== null) {
|
|
1148
|
+
// Counted when the fix is launched, not before: the next pass reads the checks again
|
|
1149
|
+
// (autopilot-reads-agent-result/AC-26, design D7).
|
|
1150
|
+
tell(ctx, item, `CI failed (${failing.join(', ')}) — the fix waits for the session limit to reset at ${blocked}`);
|
|
1151
|
+
return;
|
|
1152
|
+
}
|
|
1153
|
+
item.ci_failures = failures;
|
|
546
1154
|
tell(ctx, item, `CI failed (${failing.join(', ')}), relaunching the agent once`);
|
|
547
1155
|
item.step = 'building';
|
|
548
|
-
launchAgent(ctx, item, fixPrompt(item, failing));
|
|
1156
|
+
launchAgent(ctx, io, state, item, fixPrompt(item, failing), budget);
|
|
549
1157
|
}
|
|
550
1158
|
/**
|
|
551
1159
|
* Mark the deployment artifact complete so `aidlc transition` can close the
|
|
@@ -646,7 +1254,8 @@ function serveQueue(ctx, io, state, budget) {
|
|
|
646
1254
|
*/
|
|
647
1255
|
function parkUnstartable(ctx, io, file, failure) {
|
|
648
1256
|
const error = failure.trim();
|
|
649
|
-
|
|
1257
|
+
// The start failure was read for a rate limit already, when `aidlc start` printed it.
|
|
1258
|
+
const r = io.cli(['roadmap', 'park', '--item', file, '--reason', `could not start: ${error}`], { echoes: true });
|
|
650
1259
|
if (r.status !== 0) {
|
|
651
1260
|
ctx.say(`could not start ${file}: ${error} — and could not park it: ${fail(r)}`);
|
|
652
1261
|
return false;
|
|
@@ -661,6 +1270,18 @@ function startBuilds(ctx, io, state, budget) {
|
|
|
661
1270
|
ctx.say('not starting builds: no agent command configured — run `aidlc autopilot setup`');
|
|
662
1271
|
return;
|
|
663
1272
|
}
|
|
1273
|
+
// Before `aidlc start`, not at launch: a new item has spent nothing, so a cap this small
|
|
1274
|
+
// would start and park every ready item in turn (design-questions.md Q1).
|
|
1275
|
+
const cap = ctx.config?.cost_cap_usd;
|
|
1276
|
+
if (typeof cap === 'number' && budgetCents(cap, 0) < 1) {
|
|
1277
|
+
ctx.say(`not starting builds: cost cap ${cap} USD leaves under a cent after the 10% margin`);
|
|
1278
|
+
return;
|
|
1279
|
+
}
|
|
1280
|
+
const blocked = launchBlocked(state, ctx.now());
|
|
1281
|
+
if (blocked !== null) {
|
|
1282
|
+
ctx.say(`not starting builds: waiting for the session limit to reset at ${blocked}`);
|
|
1283
|
+
return;
|
|
1284
|
+
}
|
|
664
1285
|
const max = buildSlots(ctx);
|
|
665
1286
|
const template = ctx.config?.template ?? 'quick-feature';
|
|
666
1287
|
let busy = building(state);
|
|
@@ -714,7 +1335,9 @@ function startBuilds(ctx, io, state, budget) {
|
|
|
714
1335
|
// Held to these until it leaves, whatever later runs say (autopilot-stop-controls/AC-6).
|
|
715
1336
|
stops: stopsInForce(ctx.config),
|
|
716
1337
|
};
|
|
717
|
-
|
|
1338
|
+
// Only the gate can refuse here, and it was checked above; nothing is tracked without an agent.
|
|
1339
|
+
if (!launchAgent(ctx, io, state, item, agentPrompt(item, itemPauses(item)), budget))
|
|
1340
|
+
return;
|
|
718
1341
|
if (ctx.dryRun)
|
|
719
1342
|
return;
|
|
720
1343
|
state.items.push(item);
|
|
@@ -797,7 +1420,7 @@ export function tick(ctx, state, budget) {
|
|
|
797
1420
|
if (item.step === 'building')
|
|
798
1421
|
advanceBuilding(ctx, io, state, item, budget);
|
|
799
1422
|
else if (item.step === 'approved')
|
|
800
|
-
advanceApproved(ctx, io, state, item);
|
|
1423
|
+
advanceApproved(ctx, io, state, item, budget);
|
|
801
1424
|
else if (item.step === 'pr-open')
|
|
802
1425
|
advancePr(ctx, io, state, item, budget);
|
|
803
1426
|
else if (item.step === 'awaiting-merge')
|