@rasensio/aidlc 1.66.0 → 1.72.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. package/README.md +2 -1
  2. package/dist/autopilot/driver.d.ts +35 -2
  3. package/dist/autopilot/driver.d.ts.map +1 -1
  4. package/dist/autopilot/driver.js +650 -27
  5. package/dist/autopilot/driver.js.map +1 -1
  6. package/dist/autopilot/presets.d.ts.map +1 -1
  7. package/dist/autopilot/presets.js +5 -0
  8. package/dist/autopilot/presets.js.map +1 -1
  9. package/dist/autopilot/result.d.ts +133 -0
  10. package/dist/autopilot/result.d.ts.map +1 -0
  11. package/dist/autopilot/result.js +431 -0
  12. package/dist/autopilot/result.js.map +1 -0
  13. package/dist/autopilot/run.d.ts +2 -0
  14. package/dist/autopilot/run.d.ts.map +1 -1
  15. package/dist/autopilot/run.js +33 -6
  16. package/dist/autopilot/run.js.map +1 -1
  17. package/dist/autopilot/state.d.ts +51 -0
  18. package/dist/autopilot/state.d.ts.map +1 -1
  19. package/dist/autopilot/state.js.map +1 -1
  20. package/dist/autopilot/typed.d.ts +98 -0
  21. package/dist/autopilot/typed.d.ts.map +1 -0
  22. package/dist/autopilot/typed.js +122 -0
  23. package/dist/autopilot/typed.js.map +1 -0
  24. package/dist/cli.d.ts.map +1 -1
  25. package/dist/cli.js +2 -0
  26. package/dist/cli.js.map +1 -1
  27. package/dist/commands/hook.d.ts +37 -0
  28. package/dist/commands/hook.d.ts.map +1 -0
  29. package/dist/commands/hook.js +208 -0
  30. package/dist/commands/hook.js.map +1 -0
  31. package/dist/commands/init.d.ts +18 -0
  32. package/dist/commands/init.d.ts.map +1 -1
  33. package/dist/commands/init.js +66 -3
  34. package/dist/commands/init.js.map +1 -1
  35. package/dist/commands/knowledge.d.ts.map +1 -1
  36. package/dist/commands/knowledge.js +8 -4
  37. package/dist/commands/knowledge.js.map +1 -1
  38. package/dist/commands/roadmap.d.ts +20 -3
  39. package/dist/commands/roadmap.d.ts.map +1 -1
  40. package/dist/commands/roadmap.js +60 -15
  41. package/dist/commands/roadmap.js.map +1 -1
  42. package/dist/commands/update.d.ts.map +1 -1
  43. package/dist/commands/update.js +8 -2
  44. package/dist/commands/update.js.map +1 -1
  45. package/dist/compile/adapters/claude-code.d.ts.map +1 -1
  46. package/dist/compile/adapters/claude-code.js +35 -4
  47. package/dist/compile/adapters/claude-code.js.map +1 -1
  48. package/dist/compile/adapters/codex.d.ts +12 -0
  49. package/dist/compile/adapters/codex.d.ts.map +1 -1
  50. package/dist/compile/adapters/codex.js +58 -8
  51. package/dist/compile/adapters/codex.js.map +1 -1
  52. package/dist/compile/loaders.d.ts.map +1 -1
  53. package/dist/compile/loaders.js +2 -0
  54. package/dist/compile/loaders.js.map +1 -1
  55. package/dist/compile/skill-metadata.d.ts +82 -0
  56. package/dist/compile/skill-metadata.d.ts.map +1 -0
  57. package/dist/compile/skill-metadata.js +156 -0
  58. package/dist/compile/skill-metadata.js.map +1 -0
  59. package/dist/concurrency/claim-store.d.ts +17 -0
  60. package/dist/concurrency/claim-store.d.ts.map +1 -1
  61. package/dist/concurrency/claim-store.js +42 -2
  62. package/dist/concurrency/claim-store.js.map +1 -1
  63. package/dist/concurrency/common-dir.d.ts +15 -0
  64. package/dist/concurrency/common-dir.d.ts.map +1 -1
  65. package/dist/concurrency/common-dir.js +39 -0
  66. package/dist/concurrency/common-dir.js.map +1 -1
  67. package/dist/core/types.d.ts +17 -0
  68. package/dist/core/types.d.ts.map +1 -1
  69. package/dist/cost/hooks/arm.d.ts +9 -1
  70. package/dist/cost/hooks/arm.d.ts.map +1 -1
  71. package/dist/cost/hooks/arm.js +16 -2
  72. package/dist/cost/hooks/arm.js.map +1 -1
  73. package/dist/cost/providers/codex.d.ts +13 -0
  74. package/dist/cost/providers/codex.d.ts.map +1 -1
  75. package/dist/cost/providers/codex.js +74 -2
  76. package/dist/cost/providers/codex.js.map +1 -1
  77. package/dist/doctor/migrations/index.d.ts.map +1 -1
  78. package/dist/doctor/migrations/index.js +5 -0
  79. package/dist/doctor/migrations/index.js.map +1 -1
  80. package/dist/doctor/migrations/roadmap-mermaid-graphs.d.ts +22 -0
  81. package/dist/doctor/migrations/roadmap-mermaid-graphs.d.ts.map +1 -0
  82. package/dist/doctor/migrations/roadmap-mermaid-graphs.js +74 -0
  83. package/dist/doctor/migrations/roadmap-mermaid-graphs.js.map +1 -0
  84. package/dist/hooks/guard.d.ts +41 -0
  85. package/dist/hooks/guard.d.ts.map +1 -0
  86. package/dist/hooks/guard.js +108 -0
  87. package/dist/hooks/guard.js.map +1 -0
  88. package/dist/hooks/install.d.ts +78 -0
  89. package/dist/hooks/install.d.ts.map +1 -0
  90. package/dist/hooks/install.js +207 -0
  91. package/dist/hooks/install.js.map +1 -0
  92. package/dist/hooks/lifecycle-context.d.ts +65 -0
  93. package/dist/hooks/lifecycle-context.d.ts.map +1 -0
  94. package/dist/hooks/lifecycle-context.js +212 -0
  95. package/dist/hooks/lifecycle-context.js.map +1 -0
  96. package/dist/knowledge/engine.d.ts +19 -2
  97. package/dist/knowledge/engine.d.ts.map +1 -1
  98. package/dist/knowledge/engine.js +35 -11
  99. package/dist/knowledge/engine.js.map +1 -1
  100. package/dist/knowledge/index/knowledge-index.d.ts +28 -2
  101. package/dist/knowledge/index/knowledge-index.d.ts.map +1 -1
  102. package/dist/knowledge/index/sqlite-index.d.ts +19 -1
  103. package/dist/knowledge/index/sqlite-index.d.ts.map +1 -1
  104. package/dist/knowledge/index/sqlite-index.js +66 -5
  105. package/dist/knowledge/index/sqlite-index.js.map +1 -1
  106. package/dist/knowledge/source-fingerprint.d.ts +71 -0
  107. package/dist/knowledge/source-fingerprint.d.ts.map +1 -0
  108. package/dist/knowledge/source-fingerprint.js +105 -0
  109. package/dist/knowledge/source-fingerprint.js.map +1 -0
  110. package/dist/knowledge/sync.d.ts +12 -1
  111. package/dist/knowledge/sync.d.ts.map +1 -1
  112. package/dist/knowledge/sync.js +23 -3
  113. package/dist/knowledge/sync.js.map +1 -1
  114. package/dist/knowledge/types.d.ts +10 -1
  115. package/dist/knowledge/types.d.ts.map +1 -1
  116. package/dist/knowledge/types.js +10 -1
  117. package/dist/knowledge/types.js.map +1 -1
  118. package/dist/menu/roster.d.ts.map +1 -1
  119. package/dist/menu/roster.js +2 -0
  120. package/dist/menu/roster.js.map +1 -1
  121. package/dist/roadmap/graph.d.ts +73 -28
  122. package/dist/roadmap/graph.d.ts.map +1 -1
  123. package/dist/roadmap/graph.js +184 -125
  124. package/dist/roadmap/graph.js.map +1 -1
  125. package/dist/roadmap/next.d.ts +5 -0
  126. package/dist/roadmap/next.d.ts.map +1 -1
  127. package/dist/roadmap/next.js +7 -1
  128. package/dist/roadmap/next.js.map +1 -1
  129. package/dist/roadmap/open-browser.d.ts +34 -0
  130. package/dist/roadmap/open-browser.d.ts.map +1 -0
  131. package/dist/roadmap/open-browser.js +40 -0
  132. package/dist/roadmap/open-browser.js.map +1 -0
  133. package/dist/roadmap/paths.d.ts +5 -1
  134. package/dist/roadmap/paths.d.ts.map +1 -1
  135. package/dist/roadmap/paths.js +5 -9
  136. package/dist/roadmap/paths.js.map +1 -1
  137. package/dist/roadmap/view-assets.d.ts +18 -0
  138. package/dist/roadmap/view-assets.d.ts.map +1 -0
  139. package/dist/roadmap/view-assets.js +301 -0
  140. package/dist/roadmap/view-assets.js.map +1 -0
  141. package/dist/roadmap/view-html.d.ts +37 -0
  142. package/dist/roadmap/view-html.d.ts.map +1 -0
  143. package/dist/roadmap/view-html.js +317 -0
  144. package/dist/roadmap/view-html.js.map +1 -0
  145. package/dist/roadmap/view-model.d.ts +93 -0
  146. package/dist/roadmap/view-model.d.ts.map +1 -0
  147. package/dist/roadmap/view-model.js +194 -0
  148. package/dist/roadmap/view-model.js.map +1 -0
  149. package/package.json +2 -2
@@ -15,7 +15,12 @@
15
15
  *
16
16
  * @module
17
17
  */
18
+ import { mapUnits } from '../cost/providers/codex.js';
19
+ import { microToUsd, priceUnits } from '../cost/rates.js';
20
+ import { MODEL_ID_RE } from '../cost/types.js';
21
+ import { REASON_MAX, RESET_WAIT_MAX_MS, codexEnd, emptyCodexProgress, firstLine, readClaudeResult, readCodexEvents, refusalSuffix, resetAt, stoppedBy, summaryLines, } from './result.js';
18
22
  import { LEGACY_STOPS, pauseEnv, pendingReview, stopsInForce } from './stops.js';
23
+ import { OUTCOME_SCHEMA, UUID_RE, budgetCents, resumePrompt, typedArgv, typedHarness, withOutcomeContract, } from './typed.js';
19
24
  const RATE_LIMIT_RE = /rate limit|\b429\b|secondary rate/i;
20
25
  const BACKOFF_FIRST_MS = 5 * 60 * 1000;
21
26
  const BACKOFF_MAX_MS = 2 * 60 * 60 * 1000;
@@ -39,16 +44,27 @@ function makeIO(ctx) {
39
44
  ctx.say(`[dry-run] ${opts.cwd ? `(in ${opts.cwd}) ` : ''}${[cmd, ...args].join(' ')}`);
40
45
  return { status: 0, stdout: '', stderr: '', dry: true };
41
46
  }
42
- const r = ctx.run(cmd, args, opts);
43
- if (RATE_LIMIT_RE.test(`${r.stdout}\n${r.stderr}`)) {
47
+ const r = ctx.run(cmd, args, opts.cwd === undefined ? {} : { cwd: opts.cwd });
48
+ if (opts.echoes !== true && RATE_LIMIT_RE.test(`${r.stdout}\n${r.stderr}`)) {
44
49
  throw new RateLimited(`${cmd} ${args.slice(0, 2).join(' ')}: ${(r.stderr || r.stdout).trim().split('\n')[0]}`);
45
50
  }
46
51
  return r;
47
52
  };
48
- const cli = (args) => call(ctx.cli[0], [...ctx.cli.slice(1), ...args], { cwd: ctx.repo });
53
+ const cli = (args, opts = {}) => call(ctx.cli[0], [...ctx.cli.slice(1), ...args], { cwd: ctx.repo, ...opts });
49
54
  return { call, cli };
50
55
  }
51
56
  const fail = (r) => (r.stderr || r.stdout || `exit ${r.status}`).trim().split('\n').slice(-1)[0];
57
+ /**
58
+ * Git's whole message on one line, at most 400 characters. `fail` keeps the last
59
+ * line, and a refused merge's last line is `Aborting` — the files it would
60
+ * overwrite are the lines before (autopilot-merge-refusal-is-not-a-conflict/AC-7).
61
+ */
62
+ const gitMessage = (r) => (r.stderr.trim() ? r.stderr : r.stdout || `exit ${r.status}`)
63
+ .split('\n')
64
+ .map((l) => l.trim())
65
+ .filter(Boolean)
66
+ .join(' ')
67
+ .slice(0, 400);
52
68
  /**
53
69
  * Why `aidlc transition --json` refused: each violation, where `fail` alone would keep
54
70
  * the last line — a "next step" naming no cause, which is all a halted queue logged
@@ -158,16 +174,13 @@ const itemPauses = (item) => pauseEnv(itemStops(item), item.approved ?? []);
158
174
  * fails later prints a banner first, which names nothing (codex-preset-removed-flag/AC-6, AC-10).
159
175
  */
160
176
  export const LAUNCH_FAILURE_MS = 30_000;
161
- const REASON_MAX = 200;
162
177
  /**
163
178
  * The first non-blank line of `text`, terminal escapes and control characters removed,
164
179
  * at most 200 characters; null when there is none (codex-preset-removed-flag/AC-12).
165
180
  * The first line, not `fail`'s last: for an argument error the last is the usage line.
166
181
  */
167
182
  export function firstOutputLine(text) {
168
- const plain = text.replace(/\x1b\[[0-?]*[ -/]*[@-~]/g, '').replace(/[\x00-\x08\x0b-\x1f\x7f]/g, '');
169
- const line = plain.split('\n').map((l) => l.trim()).find((l) => l !== '');
170
- return line === undefined ? null : line.slice(0, REASON_MAX);
183
+ return firstLine(text, REASON_MAX);
171
184
  }
172
185
  /**
173
186
  * The park reason for an agent that failed to launch, or null when it did not or the
@@ -183,9 +196,92 @@ function launchFailure(ctx, item) {
183
196
  const line = firstOutputLine(exit.output);
184
197
  return line === null ? `agent failed to launch (${how}) with no output` : `agent failed to launch (${how}): ${line}`;
185
198
  }
186
- function launchAgent(ctx, item, prompt) {
199
+ /**
200
+ * The reset time of the latest session-limit wait still in force, or null. While there is
201
+ * one, no agent is launched for any item: the limit is the account's, so every launch
202
+ * would hit it (autopilot-reads-agent-result/AC-26).
203
+ */
204
+ export function launchBlocked(state, now) {
205
+ let latest = null;
206
+ for (const i of state.items) {
207
+ const w = i.resume_wait;
208
+ const until = w?.kind === 'session-limit' ? Date.parse(w.until) : Number.NaN;
209
+ if (until > now && (latest === null || until > latest))
210
+ latest = until;
211
+ }
212
+ // Re-rendered, not repeated: the stored text came from a file that can be edited.
213
+ return latest === null ? null : new Date(latest).toISOString();
214
+ }
215
+ /** The park reason when the item's spend has used up its cap (autopilot-reads-agent-result/AC-7, AC-18, AC-20). */
216
+ const capReached = (cap, spentMicro) => `cost cap ${cap} USD reached — spent ${microToUsd(spentMicro)}`;
217
+ const logFile = (ctx, item) => `${ctx.paths.logs}/${item.name}.log`;
218
+ /**
219
+ * Launch the item's agent. False when nothing was launched: a session-limit wait is in
220
+ * force — the net under each caller's own check (design D7) — or the item was parked,
221
+ * for a budget under a cent (autopilot-reads-agent-result/AC-18) or a session id that is
222
+ * not a UUID (AC-33).
223
+ *
224
+ * A preset command is launched typed (AC-1, AC-2): the outcome contract in the prompt
225
+ * (AC-5), the flags, the budget (AC-17), and the launch recorded on the item with where
226
+ * its output starts, so a pass in any process reads its result (AC-32). A custom
227
+ * command is launched exactly as configured, and clears any earlier mark (AC-3, AC-31).
228
+ *
229
+ * @param budget - the run's item budget, which a park here must reach.
230
+ */
231
+ function launchAgent(ctx, io, state, item, prompt, budget, resume) {
232
+ if (launchBlocked(state, ctx.now()) !== null)
233
+ return false;
234
+ const cap = ctx.config?.cost_cap_usd;
235
+ const spent = item.spent_micro_usd ?? 0;
236
+ const cents = typeof cap === 'number' ? budgetCents(cap, spent) : null;
237
+ if (typeof cap === 'number' && (cents ?? 0) < 1) {
238
+ park(ctx, io, state, item, capReached(cap, spent), budget);
239
+ return false;
240
+ }
187
241
  const command = ctx.config?.agent_command ?? [];
188
- const argv = command.map((a) => (a === '{prompt}' ? prompt : a));
242
+ const harness = typedHarness(command);
243
+ const log = logFile(ctx, item);
244
+ let argv;
245
+ let launch;
246
+ if (harness === null) {
247
+ argv = command.map((a) => (a === '{prompt}' ? prompt : a));
248
+ }
249
+ else {
250
+ const outcomeFile = `${ctx.paths.dir}/outcomes/${item.name}.json`;
251
+ const schemaFile = `${ctx.paths.dir}/outcome-schema.json`;
252
+ const typed = typedArgv(command, {
253
+ harness,
254
+ prompt: withOutcomeContract(prompt),
255
+ schemaFile,
256
+ outcomeFile,
257
+ maxBudgetCents: harness === 'claude' ? cents : null,
258
+ resume: resume?.session ?? null,
259
+ });
260
+ if (typed === null) {
261
+ park(ctx, io, state, item, `${resume?.label ?? 'session'} — no usable session id`, budget);
262
+ return false;
263
+ }
264
+ argv = typed;
265
+ launch = {
266
+ harness,
267
+ at: new Date(ctx.now()).toISOString(),
268
+ ...(typeof cap === 'number' ? { cap_usd: cap } : {}),
269
+ log_offset: 0,
270
+ ...(harness === 'codex' ? { outcome_file: outcomeFile, progress: emptyCodexProgress() } : {}),
271
+ ...(resume === undefined ? {} : { resumes: resume.session }),
272
+ };
273
+ // A resume continues the thread's rollout, which sits in its first launch's day
274
+ // directory, and Codex reports the model there once (design D10).
275
+ if (resume?.from?.harness === 'codex') {
276
+ if (resume.from.model !== undefined)
277
+ launch.model = resume.from.model;
278
+ launch.rollout_at = resume.from.rollout_at ?? resume.from.at;
279
+ }
280
+ if (harness === 'codex' && !ctx.dryRun) {
281
+ ctx.fs.write(schemaFile, `${JSON.stringify(OUTCOME_SCHEMA, null, 2)}\n`);
282
+ ctx.fs.write(outcomeFile, '');
283
+ }
284
+ }
189
285
  const env = {
190
286
  AIDLC_UNATTENDED: '1',
191
287
  AIDLC_AGENT: ctx.config?.agent_platform ?? 'claude-code',
@@ -193,10 +289,21 @@ function launchAgent(ctx, item, prompt) {
193
289
  AIDLC_PAUSE_AFTER: itemPauses(item),
194
290
  };
195
291
  if (ctx.dryRun) {
196
- ctx.say(`[dry-run] (in ${item.worktree}) launch agent: ${argv[0]} … (${argv.length} args)`);
197
- return;
292
+ ctx.say(`[dry-run] (in ${item.worktree}) launch agent: ${argv[0]} … (${argv.length} args${harness === null ? '' : `, typed ${harness}`})`);
293
+ return true;
294
+ }
295
+ if (launch === undefined) {
296
+ delete item.launch;
198
297
  }
199
- item.pid = ctx.spawnDetached(argv, { cwd: item.worktree, env, logFile: `${ctx.paths.logs}/${item.name}.log` });
298
+ else {
299
+ // Read before the spawn, not after: only this item's agent writes its log, and it is
300
+ // not running yet, so nothing can write in between.
301
+ launch.log_offset = ctx.fs.size?.(log) ?? 0;
302
+ if (launch.harness === 'codex')
303
+ launch.read_offset = launch.log_offset;
304
+ item.launch = launch;
305
+ }
306
+ item.pid = ctx.spawnDetached(argv, { cwd: item.worktree, env, logFile: log });
200
307
  item.pid_started_at = new Date(ctx.now()).toISOString();
201
308
  const id = ctx.identify?.(item.pid);
202
309
  if (id) {
@@ -205,6 +312,7 @@ function launchAgent(ctx, item, prompt) {
205
312
  item.pid_ns = id.ns;
206
313
  }
207
314
  tell(ctx, item, `agent started (pid ${item.pid})`);
315
+ return true;
208
316
  }
209
317
  function drop(state, item, budget) {
210
318
  state.items = state.items.filter((i) => i !== item);
@@ -212,7 +320,7 @@ function drop(state, item, budget) {
212
320
  budget?.onFinish(item.name);
213
321
  }
214
322
  function park(ctx, io, state, item, reason, budget) {
215
- const r = io.cli(['roadmap', 'park', item.name, '--reason', reason]);
323
+ const r = io.cli(['roadmap', 'park', item.name, '--reason', reason], { echoes: true });
216
324
  if (r.status !== 0) {
217
325
  failItem(ctx, state, item, `park failed: ${fail(r)}`);
218
326
  tell(ctx, item, `could not park — ${item.error}`);
@@ -223,10 +331,16 @@ function park(ctx, io, state, item, reason, budget) {
223
331
  notice(ctx, 'item parked', `${item.name}: ${reason}`);
224
332
  drop(state, item, budget);
225
333
  }
226
- /** The instance's `current_phase` in its worktree, or null when unreadable. Shared with the view. */
334
+ /**
335
+ * The instance's `current_phase` in its worktree, or null when unreadable. Shared with the
336
+ * view. Only a phase-shaped name: the agent writes this file, and the phase reaches park
337
+ * reasons committed to the roadmap (autopilot-reads-agent-result/AC-29).
338
+ */
227
339
  export function instancePhase(read, item) {
228
340
  const yaml = read(`${item.worktree}/.aidlc/state/${item.name}/instance.yaml`) ?? '';
229
- return /^current_phase:\s*(\S+)/m.exec(yaml)?.[1] ?? null;
341
+ // A name of word characters and hyphens ending at a space or the line's end: what `\S+`
342
+ // read for every real phase — a trailing comment too — without its escapes.
343
+ return /^current_phase:[ \t]*([A-Za-z0-9_-]{1,32})(?=\s|$)/m.exec(yaml)?.[1] ?? null;
230
344
  }
231
345
  function currentPhase(ctx, item) {
232
346
  return instancePhase(ctx.fs.read, item);
@@ -252,10 +366,225 @@ function awaitIfPaused(ctx, item) {
252
366
  notice(ctx, 'review needed', `${item.name}: paused after ${phase} — aidlc autopilot approve ${item.name}`);
253
367
  return true;
254
368
  }
369
+ /** How much of a log one read takes, and one pass at most (design, "While the agent runs"). */
370
+ const CHUNK_BYTES = 1024 * 1024;
371
+ const PASS_READ_BYTES = 8 * CHUNK_BYTES;
372
+ /**
373
+ * Fold what a Codex launch appended to the item's log since the last pass into its
374
+ * progress, at most 8 MiB a pass (autopilot-reads-agent-result/AC-20). Offsets are bytes,
375
+ * and a chunk is decoded only up to its last newline byte, so a character split across
376
+ * two chunks is never miscounted. A line still being written waits for the next pass —
377
+ * unless `final`, when the agent has exited and the line will never be finished. A line
378
+ * longer than a chunk is skipped; its tail fails to parse and is counted unreadable.
379
+ *
380
+ * @returns whether the whole log has been folded.
381
+ */
382
+ function followCodex(ctx, item, launch, final) {
383
+ const readRange = ctx.fs.readRange;
384
+ if (readRange === undefined)
385
+ return true;
386
+ const progress = (launch.progress ??= emptyCodexProgress());
387
+ const log = logFile(ctx, item);
388
+ let offset = launch.read_offset ?? launch.log_offset;
389
+ for (let left = PASS_READ_BYTES; left > 0;) {
390
+ const want = Math.min(CHUNK_BYTES, left);
391
+ const chunk = readRange(log, offset, want);
392
+ if (chunk.length === 0)
393
+ break;
394
+ const newline = chunk.lastIndexOf(0x0a);
395
+ if (newline === -1) {
396
+ if (chunk.length === CHUNK_BYTES) {
397
+ offset += chunk.length;
398
+ left -= chunk.length;
399
+ continue;
400
+ }
401
+ // The end of the log, mid-line: finished only when the agent will write no more.
402
+ if (final && chunk.length < want) {
403
+ readCodexEvents(chunk.toString('utf8'), progress);
404
+ offset += chunk.length;
405
+ }
406
+ break;
407
+ }
408
+ readCodexEvents(chunk.toString('utf8', 0, newline + 1), progress);
409
+ offset += newline + 1;
410
+ left -= newline + 1;
411
+ }
412
+ launch.read_offset = offset;
413
+ return offset >= (ctx.fs.size?.(log) ?? offset);
414
+ }
415
+ /** Rollout lookups for a thread's model before its launch is treated as unpriced (design D10). */
416
+ const MODEL_LOOKUPS = 3;
417
+ /** The model a Codex launch used, looked up at most three times, kept once found. */
418
+ function codexModelOf(ctx, launch) {
419
+ if (launch.model !== undefined)
420
+ return launch.model;
421
+ const thread = launch.progress?.thread_id ?? null;
422
+ if (thread === null || ctx.codexModel === undefined || (launch.model_lookups ?? 0) >= MODEL_LOOKUPS)
423
+ return null;
424
+ launch.model_lookups = (launch.model_lookups ?? 0) + 1;
425
+ const model = ctx.codexModel(thread, launch.rollout_at ?? launch.at);
426
+ // The shape the cost ledger accepts, and nothing else reaches state or a message (AC-21).
427
+ if (model !== null && MODEL_ID_RE.test(model))
428
+ launch.model = model;
429
+ return launch.model ?? null;
430
+ }
431
+ /**
432
+ * Whether the model lookup has given up for good, which is when "cannot be read" is said
433
+ * (AC-21) — after three tries, or at once when a turn completed with no usable thread id,
434
+ * since `thread.started` comes first or never.
435
+ */
436
+ const modelLookupGaveUp = (launch) => launch.model === undefined &&
437
+ ((launch.model_lookups ?? 0) >= MODEL_LOOKUPS || (launch.progress?.thread_id == null && (launch.progress?.turns ?? 0) > 0));
438
+ /**
439
+ * What a Codex launch's summed `turn.completed` usage costs, in micro-USD, priced as the
440
+ * Codex cost provider prices a turn — cached input counted inside input, not on top of
441
+ * it — at the effective rate for its model; null when it cannot be priced
442
+ * (autopilot-reads-agent-result/AC-20, AC-21). The one place the reading of Codex's
443
+ * usage lives: if the release run finds it a running total, this changes and nothing
444
+ * else does (design, Risk 1).
445
+ */
446
+ export function launchUsageMicro(ctx, launch) {
447
+ const usage = launch.progress?.usage;
448
+ if (usage === undefined)
449
+ return null;
450
+ const model = codexModelOf(ctx, launch);
451
+ const rates = ctx.rates?.();
452
+ if (model === null || rates === undefined)
453
+ return null;
454
+ const units = mapUnits({
455
+ input_tokens: usage.input,
456
+ cached_input_tokens: usage.cached,
457
+ cache_write_input_tokens: 0,
458
+ output_tokens: usage.output,
459
+ reasoning_output_tokens: 0,
460
+ total_tokens: usage.input + usage.output,
461
+ }, model);
462
+ const micro = priceUnits(units, rates);
463
+ // A figure the state file cannot hold is no figure: JSON writes Infinity as null, and the
464
+ // next load would read the spend as 0.
465
+ return micro !== null && Number.isSafeInteger(micro) ? micro : null;
466
+ }
467
+ /** What a running launch's session amounts to so far: a Codex launch's folded events; nothing for Claude. */
468
+ function sessionSoFar(launch, costMicro) {
469
+ return launch.harness === 'codex' && launch.progress !== undefined ? codexEnd(launch.progress, null, costMicro) : null;
470
+ }
471
+ /**
472
+ * Stop a running Codex launch whose priced usage, on top of the item's recorded spend,
473
+ * has reached the cap less the 10% margin (autopilot-reads-agent-result/AC-20). Codex has
474
+ * no budget flag, so this is the cap inside its run. Usage that cannot be priced is said
475
+ * once and stops nothing (AC-21).
476
+ *
477
+ * @returns whether it stopped and parked the item.
478
+ */
479
+ function stoppedAtCap(ctx, io, state, item, launch, cap, budget) {
480
+ if ((launch.progress?.turns ?? 0) === 0)
481
+ return false;
482
+ const priced = launchUsageMicro(ctx, launch);
483
+ if (priced === null) {
484
+ noteUnpriced(ctx, item, launch);
485
+ return false;
486
+ }
487
+ const spent = (item.spent_micro_usd ?? 0) + priced;
488
+ if (spent < Math.round(cap * 1e6) - Math.round(cap * 1e5))
489
+ return false;
490
+ if (!ctx.dryRun && item.pid !== null)
491
+ ctx.kill(item.pid);
492
+ appendSummary(ctx, item, summaryLines(sessionSoFar(launch, priced), 'stopped by autopilot at the cost cap'));
493
+ park(ctx, io, state, item, capReached(cap, spent), budget);
494
+ return true;
495
+ }
496
+ /**
497
+ * Say once for the item that its Codex spend cannot be priced, and where a rate goes
498
+ * (autopilot-reads-agent-result/AC-21). Only once it is known why: the model is read and
499
+ * the rate table has nothing for it, or the lookup has given up — not while the rollout
500
+ * may still turn up.
501
+ */
502
+ function noteUnpriced(ctx, item, launch) {
503
+ if (item.unpriced_noted === true)
504
+ return;
505
+ const why = launch.model !== undefined ? `no rate for ${launch.model}` : modelLookupGaveUp(launch) ? "the session's model cannot be read" : null;
506
+ if (why === null)
507
+ return;
508
+ item.unpriced_noted = true;
509
+ tell(ctx, item, `Codex spend cannot be priced: ${why}; a rate for its model goes under cost.rates in .aidlc/config.yaml — until then this run is not stopped for cost`);
510
+ }
511
+ /** How much of a Claude launch's segment is read for its result, and of a Codex outcome file. */
512
+ const RESULT_TAIL_BYTES = 1024 * 1024;
513
+ const OUTCOME_FILE_BYTES = 64 * 1024;
514
+ /**
515
+ * Read how a typed launch's session ended, once: store it on the launch, add its cost to
516
+ * the item's spend, and append the summary to the item's log. Read from the offset the
517
+ * launch recorded, so it is this launch's result whichever process saw it exit
518
+ * (autopilot-reads-agent-result/AC-6, AC-19, AC-28, AC-32).
519
+ *
520
+ * @returns false while a Codex launch's log is still being folded: its end is decided
521
+ * only once everything it wrote has been read, over several passes if need be.
522
+ */
523
+ function endSession(ctx, item, launch) {
524
+ const { readRange, size } = ctx.fs;
525
+ if (readRange === undefined || size === undefined) {
526
+ launch.end = null;
527
+ return true;
528
+ }
529
+ const log = logFile(ctx, item);
530
+ let end;
531
+ if (launch.harness === 'claude') {
532
+ const total = size(log);
533
+ const start = Math.max(launch.log_offset, total - RESULT_TAIL_BYTES);
534
+ end = readClaudeResult(readRange(log, start, Math.max(0, total - start)).toString('utf8'));
535
+ }
536
+ else {
537
+ if (!followCodex(ctx, item, launch, true))
538
+ return false;
539
+ const outcome = launch.outcome_file === undefined ? null : readRange(launch.outcome_file, 0, OUTCOME_FILE_BYTES).toString('utf8');
540
+ end = codexEnd(launch.progress ?? emptyCodexProgress(), outcome, launchUsageMicro(ctx, launch));
541
+ }
542
+ launch.end = end;
543
+ if (end !== null)
544
+ recordSpend(item, end);
545
+ appendSummary(ctx, item, summaryLines(end));
546
+ return true;
547
+ }
548
+ /**
549
+ * Add a session's cost to the item's spend (autopilot-reads-agent-result/AC-19). A
550
+ * session already on record adds only its increase. Claude reports a session's cost;
551
+ * a resumed Codex launch's usage counts only its own turns, so a Codex session's cost
552
+ * is what was recorded plus this launch's (design, Risk 1). Unpriced adds nothing.
553
+ */
554
+ function recordSpend(item, end) {
555
+ if (end.cost_micro === null)
556
+ return;
557
+ const costs = item.session_costs ?? {};
558
+ const id = end.session_id;
559
+ const previous = id === null ? 0 : costs[id] ?? 0;
560
+ const sessionCost = end.harness === 'codex' ? previous + end.cost_micro : end.cost_micro;
561
+ item.spent_micro_usd = (item.spent_micro_usd ?? 0) + Math.max(0, sessionCost - previous);
562
+ if (id !== null) {
563
+ costs[id] = Math.max(previous, sessionCost);
564
+ item.session_costs = costs;
565
+ }
566
+ }
567
+ /** Append lines to the item's log, on a line of their own. Nothing in a dry run. */
568
+ function appendSummary(ctx, item, lines) {
569
+ const { append, readRange, size } = ctx.fs;
570
+ if (ctx.dryRun || append === undefined)
571
+ return;
572
+ const log = logFile(ctx, item);
573
+ const total = size?.(log) ?? 0;
574
+ const last = total > 0 ? readRange?.(log, total - 1, 1) : undefined;
575
+ const lead = last !== undefined && last.length === 1 && last[0] !== 0x0a ? '\n' : '';
576
+ append(log, `${lead}${lines.join('\n')}\n`);
577
+ }
255
578
  function advanceBuilding(ctx, io, state, item, budget) {
256
579
  if (item.pid && ctx.isAlive(item.pid)) {
580
+ const launch = item.launch;
581
+ if (launch?.harness === 'codex')
582
+ followCodex(ctx, item, launch, false);
257
583
  const cap = ctx.config?.cost_cap_usd;
258
584
  if (typeof cap === 'number') {
585
+ if (launch?.harness === 'codex' && stoppedAtCap(ctx, io, state, item, launch, cap, budget))
586
+ return;
587
+ // Every agent, typed or not: the ledger's view of the cap stands (autopilot-reads-agent-result/AC-22).
259
588
  const r = io.cli(['cost', item.name, '--json']);
260
589
  let micro = 0;
261
590
  try {
@@ -267,19 +596,44 @@ function advanceBuilding(ctx, io, state, item, budget) {
267
596
  if (micro > cap * 1e6) {
268
597
  if (!ctx.dryRun)
269
598
  ctx.kill(item.pid);
599
+ if (launch !== undefined) {
600
+ const priced = launch.harness === 'codex' ? launchUsageMicro(ctx, launch) : null;
601
+ appendSummary(ctx, item, summaryLines(sessionSoFar(launch, priced), 'stopped by autopilot: aidlc cost passed the cap'));
602
+ }
270
603
  park(ctx, io, state, item, `cost cap ${cap} USD exceeded`, budget);
271
604
  }
272
605
  }
273
606
  return;
274
607
  }
608
+ // First, once per launch: every session that ended gets its summary, one that parked
609
+ // itself included, and a paused item's spend is on record before an approval works
610
+ // out the relaunch's budget (autopilot-reads-agent-result/AC-19, AC-28).
611
+ if (item.launch !== undefined && item.launch.end === undefined && !endSession(ctx, item, item.launch))
612
+ return;
275
613
  if (ctx.fs.exists(`${ctx.repo}/.aidlc/roadmap/hold/${item.file}`)) {
276
614
  tell(ctx, item, 'the agent parked it');
277
615
  notice(ctx, 'item parked', `${item.name}: parked by its agent — the reason is in the item's history`);
278
616
  drop(state, item, budget);
279
617
  return;
280
618
  }
281
- if (awaitIfPaused(ctx, item))
619
+ // After hold/: a person who parked a waiting item by hand has the last word.
620
+ if (item.resume_wait !== undefined) {
621
+ resumeWhenDue(ctx, io, state, item, item.resume_wait, budget);
622
+ return;
623
+ }
624
+ if (awaitIfPaused(ctx, item)) {
625
+ // The result led to a pause; after the approval it must not decide again (design D8).
626
+ delete item.launch;
627
+ return;
628
+ }
629
+ // Before the launch-failure check: a typed launch that exited inside 30 seconds but
630
+ // wrote a result is decided by it (autopilot-reads-agent-result/AC-6, AC-14).
631
+ const end = item.launch?.end;
632
+ if (item.launch !== undefined && end !== undefined && end !== null) {
633
+ decideFromResult(ctx, io, state, item, item.launch, end, budget);
282
634
  return;
635
+ }
636
+ // Untyped, or no readable result: as before (autopilot-reads-agent-result/AC-15, AC-31).
283
637
  // Before the phase: a relaunch that never started would otherwise push and open
284
638
  // the pull request as if it had finished (codex-preset-removed-flag/AC-9).
285
639
  const launch = launchFailure(ctx, item);
@@ -287,19 +641,197 @@ function advanceBuilding(ctx, io, state, item, budget) {
287
641
  park(ctx, io, state, item, launch, budget);
288
642
  return;
289
643
  }
644
+ byPhase(ctx, io, state, item, budget);
645
+ }
646
+ const stopLabel = (kind) => kind === 'connection' ? 'connection dropped' : kind === 'ended-early' ? 'ended before the release boundary' : 'session limit';
647
+ const waitingLine = (until, session) => `waiting for the session limit to reset at ${until}, then resuming session ${session}`;
648
+ /**
649
+ * A session that stopped on a dropped connection or a session limit is resumed once with
650
+ * its own session id — at the next pass, or once the limit resets — and parked when it
651
+ * cannot be (autopilot-reads-agent-result/AC-23..AC-27, AC-33). The item stays `building`
652
+ * with no agent while it waits (AC-26).
653
+ */
654
+ function decideResume(ctx, io, state, item, launch, end, error, stop, budget) {
655
+ const label = stopLabel(stop);
656
+ const line = firstLine(error, REASON_MAX) ?? 'error';
657
+ // A resume that stops again is AC-27's whether or not it reported its id again: a
658
+ // resumed Codex thread writes no second `thread.started`.
659
+ const resumed = launch.resumes !== undefined && UUID_RE.test(launch.resumes) ? launch.resumes : null;
660
+ const id = end.session_id ?? resumed;
661
+ if (launch.resumes !== undefined && id !== null) {
662
+ park(ctx, io, state, item, `${label} — resumed once, stopped again: ${line}; session ${id}`, budget);
663
+ return;
664
+ }
665
+ if (id === null) {
666
+ park(ctx, io, state, item, `${label} — no usable session id: ${line}`, budget);
667
+ return;
668
+ }
669
+ const now = ctx.now();
670
+ item.pid = null;
671
+ if (stop === 'connection') {
672
+ item.resume_wait = { kind: 'connection', until: new Date(now).toISOString(), session: id };
673
+ tell(ctx, item, `connection dropped — resuming session ${id} at the next pass`);
674
+ return;
675
+ }
676
+ const until = resetAt(error, now);
677
+ const at = new Date(until).toISOString();
678
+ // A weekly limit: hours of a build slot held, and every other launch blocked, for one resume (AC-25).
679
+ if (until - now > RESET_WAIT_MAX_MS) {
680
+ park(ctx, io, state, item, `session limit — resets ${at}; session ${id}`, budget);
681
+ return;
682
+ }
683
+ item.resume_wait = { kind: 'session-limit', until: at, session: id };
684
+ tell(ctx, item, waitingLine(at, id));
685
+ }
686
+ /**
687
+ * A waiting item, each pass: wait on until its time, then resume its session with the
688
+ * same preset flags and a budget worked out again (autopilot-reads-agent-result/AC-23,
689
+ * AC-24, AC-26). The id has been in `state.json` since it was read, so it is checked
690
+ * again here (AC-33); a resume needs the harness that started the session.
691
+ */
692
+ function resumeWhenDue(ctx, io, state, item, wait, budget) {
693
+ const label = stopLabel(wait.kind);
694
+ // Before the id is printed anywhere: it came back from a file that can be edited.
695
+ if (!UUID_RE.test(wait.session)) {
696
+ delete item.resume_wait;
697
+ park(ctx, io, state, item, `${label} — no usable session id`, budget);
698
+ return;
699
+ }
700
+ const until = Date.parse(wait.until);
701
+ if (ctx.now() < until) {
702
+ // The same text keeps the time it was first said, so the view shows how long it has waited.
703
+ tell(ctx, item, waitingLine(new Date(until).toISOString(), wait.session));
704
+ return;
705
+ }
706
+ const blocked = launchBlocked(state, ctx.now());
707
+ if (blocked !== null) {
708
+ tell(ctx, item, `${label} — resuming session ${wait.session} waits for the session limit to reset at ${blocked}`);
709
+ return;
710
+ }
711
+ delete item.resume_wait;
712
+ const from = item.launch;
713
+ if (from === undefined || typedHarness(ctx.config?.agent_command ?? []) !== from.harness) {
714
+ park(ctx, io, state, item, `${label} — the agent command changed since the session stopped; session ${wait.session}`, budget);
715
+ return;
716
+ }
717
+ tell(ctx, item, `${label} — resuming session ${wait.session}`);
718
+ // The stopped session's claim outlives it by the 60-minute staleness, and a resume may
719
+ // run under another id, whose `aidlc claim` it would refuse. From the worktree: an
720
+ // in-flight instance's state is on its branch, and `release` refuses an instance it
721
+ // cannot find. `echoes`: it prints the instance name (unattended-agent-ends-turn-to-wait/AC-7).
722
+ const released = io.call(ctx.cli[0], [...ctx.cli.slice(1), 'release', item.name], { cwd: item.worktree, echoes: true });
723
+ if (released.status !== 0) {
724
+ const said = `${label} — could not release the claim: ${fail(released)}; resuming anyway`;
725
+ tell(ctx, item, said);
726
+ // The item's own log too: it is where the resumed session's summary lands next.
727
+ appendSummary(ctx, item, [`[autopilot] ${said}`]);
728
+ }
729
+ launchAgent(ctx, io, state, item, resumePrompt(item, label), budget, { session: wait.session, label, from });
730
+ }
731
+ /** At deployment, the pull request; short of it, parked. `suffix` follows the park reason. */
732
+ function byPhase(ctx, io, state, item, budget, suffix = '') {
290
733
  const phase = currentPhase(ctx, item);
291
734
  if (phase !== 'deployment') {
292
- park(ctx, io, state, item, `agent exited before the release boundary (phase ${phase ?? 'unknown'})`, budget);
735
+ park(ctx, io, state, item, `agent exited before the release boundary (phase ${phase ?? 'unknown'})${suffix}`, budget);
293
736
  return;
294
737
  }
295
738
  openPullRequest(ctx, io, state, item);
296
739
  }
740
+ /**
741
+ * What a typed session's own result says happens next — first match wins
742
+ * (autopilot-reads-agent-result/AC-7..AC-13). Every park decided here ends with the
743
+ * refused tool calls, after any other suffix (AC-16). Only the pull request keeps the
744
+ * launch's end: a failed push resumed by `aidlc autopilot run` decides the same way
745
+ * again; every other row parks the item or waits to resume it (design D8).
746
+ */
747
+ function decideFromResult(ctx, io, state, item, launch, end, budget) {
748
+ const refusals = refusalSuffix(end);
749
+ const parkWith = (reason) => park(ctx, io, state, item, `${reason}${refusals}`, budget);
750
+ const spent = item.spent_micro_usd ?? 0;
751
+ if (end.limit === 'budget') {
752
+ // The cap the budget came from, so a config edited mid-run cannot leave the reason without one.
753
+ parkWith(launch.cap_usd === undefined ? `budget reached — spent ${microToUsd(spent)}` : capReached(launch.cap_usd, spent));
754
+ return;
755
+ }
756
+ if (end.limit === 'turns') {
757
+ parkWith(`turn limit reached after ${end.turns ?? 'an unknown number of'} turns`);
758
+ return;
759
+ }
760
+ if (end.error !== null) {
761
+ const stop = stoppedBy(end.error);
762
+ if (stop !== null) {
763
+ decideResume(ctx, io, state, item, launch, end, end.error, stop, budget);
764
+ return;
765
+ }
766
+ parkWith(`agent session failed: ${firstLine(end.error, REASON_MAX) ?? 'error'}`);
767
+ return;
768
+ }
769
+ const early = (reason) => endedEarly(ctx, io, state, item, launch, end, reason, budget);
770
+ if (end.outcome === null) {
771
+ // `byPhase`, with the early ending where it would park.
772
+ const phase = currentPhase(ctx, item);
773
+ if (phase === 'deployment')
774
+ openPullRequest(ctx, io, state, item);
775
+ else
776
+ early(`agent exited before the release boundary (phase ${phase ?? 'unknown'}) — the agent reported no outcome`);
777
+ return;
778
+ }
779
+ const { outcome, reason } = end.outcome;
780
+ const phase = currentPhase(ctx, item) ?? 'unknown';
781
+ if (outcome === 'parked') {
782
+ parkWith(`agent parked at ${phase}: ${reason}`);
783
+ return;
784
+ }
785
+ if (outcome === 'blocked') {
786
+ early(`agent blocked at ${phase}: ${reason}`);
787
+ return;
788
+ }
789
+ if (outcome === 'shipped' && phase === 'deployment') {
790
+ openPullRequest(ctx, io, state, item);
791
+ return;
792
+ }
793
+ // A `paused` that awaits review was handled before this; one that does not is a stop
794
+ // the phase files do not bear out (AC-11).
795
+ early(`agent reported ${outcome}, but the instance is at ${phase}`);
796
+ }
797
+ /**
798
+ * A session that ended before the release boundary with no stop autopilot honours — no
799
+ * outcome short of deployment, `blocked`, or a `shipped` or `paused` its phase does not
800
+ * bear out. Most likely it ended its turn to wait for a command, which under `claude -p`
801
+ * ends the process and the command with it. Resumed once at the next pass, like a dropped
802
+ * connection; parked with `reason` — today's text — when it cannot be
803
+ * (unattended-agent-ends-turn-to-wait/AC-3, AC-5, AC-6).
804
+ */
805
+ function endedEarly(ctx, io, state, item, launch, end, reason, budget) {
806
+ const refusals = refusalSuffix(end);
807
+ // A CI-fix or conflict relaunch: its pull request is open, so it is past the release
808
+ // boundary, and the early-ending prompt would be wrong about it. Parked as before.
809
+ if (item.pr) {
810
+ park(ctx, io, state, item, `${reason}${refusals}`, budget);
811
+ return;
812
+ }
813
+ // As `decideResume` reads it: a resumed Codex thread writes no second `thread.started`.
814
+ const resumed = launch.resumes !== undefined && UUID_RE.test(launch.resumes) ? launch.resumes : null;
815
+ const id = end.session_id ?? resumed;
816
+ // A resume of any kind: a session is resumed once (AC-5).
817
+ if (launch.resumes !== undefined) {
818
+ park(ctx, io, state, item, `${reason}; resumed once${id === null ? '' : `, session ${id}`}${refusals}`, budget);
819
+ return;
820
+ }
821
+ if (id === null) {
822
+ park(ctx, io, state, item, `${reason}${refusals}`, budget);
823
+ return;
824
+ }
825
+ item.pid = null;
826
+ item.resume_wait = { kind: 'ended-early', until: new Date(ctx.now()).toISOString(), session: id };
827
+ tell(ctx, item, `ended before the release boundary — resuming session ${id} at the next pass`);
828
+ }
297
829
  /**
298
830
  * The pass after `aidlc autopilot approve`: wait again at the next completed pause,
299
831
  * open the pull request when the agent is already at deployment, or relaunch it to
300
832
  * carry on (autopilot-stop-controls/AC-18).
301
833
  */
302
- function advanceApproved(ctx, io, state, item) {
834
+ function advanceApproved(ctx, io, state, item, budget) {
303
835
  if (awaitIfPaused(ctx, item))
304
836
  return;
305
837
  if (currentPhase(ctx, item) === 'deployment') {
@@ -311,9 +843,13 @@ function advanceApproved(ctx, io, state, item) {
311
843
  // since this pass serves it before `startBuilds` runs.
312
844
  if (building(state) >= buildSlots(ctx))
313
845
  return;
846
+ // And for the session limit, still `approved` (autopilot-reads-agent-result/AC-26).
847
+ if (launchBlocked(state, ctx.now()) !== null)
848
+ return;
314
849
  item.step = 'building';
315
850
  tell(ctx, item, 'approved — relaunching the agent to carry on');
316
- launchAgent(ctx, item, agentPrompt(item, itemPauses(item)));
851
+ // The run's budget: a park for the cap here must reach `run.finished`, or `--items n` never ends.
852
+ launchAgent(ctx, io, state, item, agentPrompt(item, itemPauses(item)), budget);
317
853
  }
318
854
  function openPullRequest(ctx, io, state, item) {
319
855
  const push = io.call('git', ['push', '-u', 'origin', item.branch], { cwd: item.worktree });
@@ -445,11 +981,55 @@ function handleConflict(ctx, io, state, item, budget) {
445
981
  park(ctx, io, state, item, `conflicts with main after ${CONFLICT_FIX_LIMIT} fix attempts`, budget);
446
982
  return;
447
983
  }
984
+ const blocked = launchBlocked(state, ctx.now());
985
+ if (blocked !== null) {
986
+ // No attempt is made, so none is counted; out of the merge queue, so the items
987
+ // behind it can merge meanwhile (autopilot-reads-agent-result/AC-26, design D7).
988
+ dequeue(state, item);
989
+ item.step = 'pr-open';
990
+ tell(ctx, item, `conflicts with main — the fix waits for the session limit to reset at ${blocked}`);
991
+ return;
992
+ }
448
993
  item.conflict_fixes = tries + 1;
449
994
  dequeue(state, item);
450
995
  item.step = 'building';
451
996
  tell(ctx, item, `conflicts with main — relaunching the agent to resolve it (attempt ${tries + 1} of ${CONFLICT_FIX_LIMIT})`);
452
- launchAgent(ctx, item, conflictPrompt(item));
997
+ launchAgent(ctx, io, state, item, conflictPrompt(item), budget);
998
+ }
999
+ /**
1000
+ * The two `merge=union` cost ledgers a worktree collects between commits. `name` is
1001
+ * the one `aidlc start` created and the agent claimed, so it holds no glob or
1002
+ * option character (autopilot-merge-refusal-is-not-a-conflict/AC-1).
1003
+ */
1004
+ const ledgerPaths = (name) => ['.aidlc/cost/unattributed.ndjson', `.aidlc/state/${name}/costs.ndjson`];
1005
+ const LEDGER_COMMIT_MESSAGE = 'chore(aidlc): commit cost ledgers before updating from main (autopilot)';
1006
+ /**
1007
+ * Commit the worktree's uncommitted cost ledgers, and nothing else, so git does not
1008
+ * refuse the update from main over a ledger both sides appended to — PR #125 on
1009
+ * 2026-10-06 (autopilot-merge-refusal-is-not-a-conflict/AC-1..AC-4). Null when
1010
+ * nothing needed committing or the commit was made; otherwise git's failure.
1011
+ */
1012
+ function commitLedgers(ctx, io, item) {
1013
+ const wt = { cwd: item.worktree };
1014
+ const candidates = ledgerPaths(item.name);
1015
+ const status = io.call('git', ['status', '--porcelain=v1', '--untracked-files=all', '--', ...candidates], wt);
1016
+ if (status.status !== 0)
1017
+ return status;
1018
+ // Exact matches only, so a quoted or renamed path fails safe; a deleted ledger is
1019
+ // never committed as a deletion (AC-2).
1020
+ const lines = status.stdout.split('\n');
1021
+ const dirty = candidates.filter((p) => lines.some((l) => l.slice(3) === p && !l.slice(0, 2).includes('D')));
1022
+ if (dirty.length === 0)
1023
+ return null;
1024
+ const add = io.call('git', ['add', '--', ...dirty], wt);
1025
+ if (add.status !== 0)
1026
+ return add;
1027
+ // By pathspec: another staged change stays staged and outside this commit (AC-2).
1028
+ const commit = io.call('git', ['commit', '-q', '-m', LEDGER_COMMIT_MESSAGE, '--', ...dirty], wt);
1029
+ if (commit.status !== 0)
1030
+ return commit;
1031
+ tell(ctx, item, 'committed its cost ledgers before updating from main');
1032
+ return null;
453
1033
  }
454
1034
  /**
455
1035
  * Bring the item's branch up to date with main in its own worktree, where the
@@ -469,16 +1049,35 @@ function upToDateWithMain(ctx, io, state, item, resume, budget) {
469
1049
  return false;
470
1050
  }
471
1051
  const contains = io.call('git', ['merge-base', '--is-ancestor', 'origin/main', 'HEAD'], wt);
1052
+ // Nothing is committed on this path: the head CI passed on is the head merged (AC-3).
472
1053
  if (contains.status === 0)
473
1054
  return true;
474
1055
  if (contains.status !== 1) {
475
1056
  failItem(ctx, state, item, `update from main failed: ${fail(contains)}`, resume);
476
1057
  return false;
477
1058
  }
1059
+ // Here, not at the end of the build: the cost hook can write after the agent's
1060
+ // last commit, and the head is about to move anyway (design D1).
1061
+ const ledgers = commitLedgers(ctx, io, item);
1062
+ if (ledgers !== null) {
1063
+ failItem(ctx, state, item, `update from main failed: could not commit the cost ledgers: ${gitMessage(ledgers)}`, resume);
1064
+ return false;
1065
+ }
478
1066
  const merge = io.call('git', ['merge', '--no-edit', 'origin/main'], wt);
479
1067
  if (merge.status !== 0) {
1068
+ // Read before the abort clears them. A merge git refused — local changes it would
1069
+ // overwrite — leaves none, and an agent relaunched over it found "a clean merge"
1070
+ // seven minutes later (autopilot-merge-refusal-is-not-a-conflict/AC-5, AC-8).
1071
+ const unmerged = io.call('git', ['diff', '--name-only', '--diff-filter=U'], wt);
1072
+ // Fails harmlessly when the merge never started.
480
1073
  io.call('git', ['merge', '--abort'], wt);
481
- handleConflict(ctx, io, state, item, budget);
1074
+ if (unmerged.status === 0 && unmerged.stdout.trim() !== '') {
1075
+ handleConflict(ctx, io, state, item, budget);
1076
+ }
1077
+ else {
1078
+ // A check that cannot answer is no conflict either: an agent cannot mend git (AC-6).
1079
+ failItem(ctx, state, item, `update from main failed: ${gitMessage(merge)}`, resume);
1080
+ }
482
1081
  return false;
483
1082
  }
484
1083
  dequeue(state, item);
@@ -538,14 +1137,23 @@ function advancePr(ctx, io, state, item, budget) {
538
1137
  return;
539
1138
  }
540
1139
  item.ci = 'failed';
541
- item.ci_failures = (item.ci_failures ?? 0) + 1;
542
- if (item.ci_failures >= CI_FAILURE_LIMIT) {
1140
+ const failures = (item.ci_failures ?? 0) + 1;
1141
+ if (failures >= CI_FAILURE_LIMIT) {
1142
+ item.ci_failures = failures;
543
1143
  park(ctx, io, state, item, `CI failed twice: ${failing.join(', ')}`, budget);
544
1144
  return;
545
1145
  }
1146
+ const blocked = launchBlocked(state, ctx.now());
1147
+ if (blocked !== null) {
1148
+ // Counted when the fix is launched, not before: the next pass reads the checks again
1149
+ // (autopilot-reads-agent-result/AC-26, design D7).
1150
+ tell(ctx, item, `CI failed (${failing.join(', ')}) — the fix waits for the session limit to reset at ${blocked}`);
1151
+ return;
1152
+ }
1153
+ item.ci_failures = failures;
546
1154
  tell(ctx, item, `CI failed (${failing.join(', ')}), relaunching the agent once`);
547
1155
  item.step = 'building';
548
- launchAgent(ctx, item, fixPrompt(item, failing));
1156
+ launchAgent(ctx, io, state, item, fixPrompt(item, failing), budget);
549
1157
  }
550
1158
  /**
551
1159
  * Mark the deployment artifact complete so `aidlc transition` can close the
@@ -646,7 +1254,8 @@ function serveQueue(ctx, io, state, budget) {
646
1254
  */
647
1255
  function parkUnstartable(ctx, io, file, failure) {
648
1256
  const error = failure.trim();
649
- const r = io.cli(['roadmap', 'park', '--item', file, '--reason', `could not start: ${error}`]);
1257
+ // The start failure was read for a rate limit already, when `aidlc start` printed it.
1258
+ const r = io.cli(['roadmap', 'park', '--item', file, '--reason', `could not start: ${error}`], { echoes: true });
650
1259
  if (r.status !== 0) {
651
1260
  ctx.say(`could not start ${file}: ${error} — and could not park it: ${fail(r)}`);
652
1261
  return false;
@@ -661,6 +1270,18 @@ function startBuilds(ctx, io, state, budget) {
661
1270
  ctx.say('not starting builds: no agent command configured — run `aidlc autopilot setup`');
662
1271
  return;
663
1272
  }
1273
+ // Before `aidlc start`, not at launch: a new item has spent nothing, so a cap this small
1274
+ // would start and park every ready item in turn (design-questions.md Q1).
1275
+ const cap = ctx.config?.cost_cap_usd;
1276
+ if (typeof cap === 'number' && budgetCents(cap, 0) < 1) {
1277
+ ctx.say(`not starting builds: cost cap ${cap} USD leaves under a cent after the 10% margin`);
1278
+ return;
1279
+ }
1280
+ const blocked = launchBlocked(state, ctx.now());
1281
+ if (blocked !== null) {
1282
+ ctx.say(`not starting builds: waiting for the session limit to reset at ${blocked}`);
1283
+ return;
1284
+ }
664
1285
  const max = buildSlots(ctx);
665
1286
  const template = ctx.config?.template ?? 'quick-feature';
666
1287
  let busy = building(state);
@@ -714,7 +1335,9 @@ function startBuilds(ctx, io, state, budget) {
714
1335
  // Held to these until it leaves, whatever later runs say (autopilot-stop-controls/AC-6).
715
1336
  stops: stopsInForce(ctx.config),
716
1337
  };
717
- launchAgent(ctx, item, agentPrompt(item, itemPauses(item)));
1338
+ // Only the gate can refuse here, and it was checked above; nothing is tracked without an agent.
1339
+ if (!launchAgent(ctx, io, state, item, agentPrompt(item, itemPauses(item)), budget))
1340
+ return;
718
1341
  if (ctx.dryRun)
719
1342
  return;
720
1343
  state.items.push(item);
@@ -797,7 +1420,7 @@ export function tick(ctx, state, budget) {
797
1420
  if (item.step === 'building')
798
1421
  advanceBuilding(ctx, io, state, item, budget);
799
1422
  else if (item.step === 'approved')
800
- advanceApproved(ctx, io, state, item);
1423
+ advanceApproved(ctx, io, state, item, budget);
801
1424
  else if (item.step === 'pr-open')
802
1425
  advancePr(ctx, io, state, item, budget);
803
1426
  else if (item.step === 'awaiting-merge')