@mmerterden/multi-agent-pipeline 20.2.0 → 20.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (621) hide show
  1. package/CHANGELOG.md +923 -1
  2. package/README.md +104 -81
  3. package/README.tr.md +103 -62
  4. package/docs/FIGMA_PIPELINE.md +35 -35
  5. package/docs/adr/0006-skills-core-external-split.md +1 -1
  6. package/docs/adr/0007-multi-tool-adapter-framework.md +6 -0
  7. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +7 -0
  8. package/docs/architecture.md +50 -14
  9. package/docs/best-practices.md +1 -1
  10. package/docs/ecosystem.md +56 -32
  11. package/docs/facts.json +10 -10
  12. package/docs/features.md +97 -5
  13. package/docs/recovery-guide.md +7 -14
  14. package/docs/server-readiness.md +31 -24
  15. package/index.js +1 -1
  16. package/install/_common.mjs +3 -5
  17. package/install/_platform-filter.mjs +23 -1
  18. package/install/_unattended-profile.mjs +321 -75
  19. package/install/claude.mjs +51 -10
  20. package/install/codex.mjs +2 -0
  21. package/install/copilot.mjs +2 -0
  22. package/install/index.mjs +30 -17
  23. package/install/templates/claude-hooks.json +16 -5
  24. package/install/templates/copilot-instructions.md +1 -1
  25. package/install/templates/multi-agent-autopilot-awake.plist.template +48 -0
  26. package/install/templates/multi-agent-autopilot.plist.template +12 -5
  27. package/install/unattended-profile-legacy.json +80 -0
  28. package/manifest.json +616 -483
  29. package/package.json +8 -3
  30. package/pipeline/agents/code-reviewer.md +10 -0
  31. package/pipeline/agents/plan-critic.md +98 -0
  32. package/pipeline/agents/security-auditor.md +10 -0
  33. package/pipeline/agents/task-clarifier.md +10 -0
  34. package/pipeline/commands/multi-agent/SKILL.md +2 -2
  35. package/pipeline/commands/multi-agent/analysis/SKILL.md +6 -2
  36. package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +12 -1
  37. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +4 -0
  38. package/pipeline/commands/multi-agent/autopilot/SKILL.md +6 -2
  39. package/pipeline/commands/multi-agent/autopilot-off/SKILL.md +36 -6
  40. package/pipeline/commands/multi-agent/autopilot-on/SKILL.md +77 -12
  41. package/pipeline/commands/multi-agent/autopilot-status/SKILL.md +29 -8
  42. package/pipeline/commands/multi-agent/build-optimize/SKILL.md +4 -0
  43. package/pipeline/commands/multi-agent/channels/SKILL.md +9 -9
  44. package/pipeline/commands/multi-agent/complaint-analysis/SKILL.md +5 -1
  45. package/pipeline/commands/multi-agent/create-jira/SKILL.md +4 -0
  46. package/pipeline/commands/multi-agent/design-check/SKILL.md +16 -11
  47. package/pipeline/commands/multi-agent/diff-explain/SKILL.md +4 -0
  48. package/pipeline/commands/multi-agent/doctor/SKILL.md +4 -0
  49. package/pipeline/commands/multi-agent/feedback/SKILL.md +5 -1
  50. package/pipeline/commands/multi-agent/forget/SKILL.md +4 -0
  51. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +4 -0
  52. package/pipeline/commands/multi-agent/graph/SKILL.md +4 -0
  53. package/pipeline/commands/multi-agent/help/SKILL.md +37 -39
  54. package/pipeline/commands/multi-agent/ios-coding-standard/SKILL.md +4 -0
  55. package/pipeline/commands/multi-agent/issue/SKILL.md +7 -1
  56. package/pipeline/commands/multi-agent/jira/SKILL.md +7 -1
  57. package/pipeline/commands/multi-agent/kill/SKILL.md +9 -3
  58. package/pipeline/commands/multi-agent/language/SKILL.md +4 -0
  59. package/pipeline/commands/multi-agent/log/SKILL.md +4 -0
  60. package/pipeline/commands/multi-agent/manual-test/SKILL.md +5 -0
  61. package/pipeline/commands/multi-agent/model/SKILL.md +4 -0
  62. package/pipeline/commands/multi-agent/prune-logs/SKILL.md +4 -0
  63. package/pipeline/commands/multi-agent/prune-prompts/SKILL.md +4 -0
  64. package/pipeline/commands/multi-agent/purge/SKILL.md +4 -0
  65. package/pipeline/commands/multi-agent/refactor/SKILL.md +5 -0
  66. package/pipeline/commands/multi-agent/research/SKILL.md +49 -0
  67. package/pipeline/commands/multi-agent/resume/SKILL.md +29 -6
  68. package/pipeline/commands/multi-agent/review/SKILL.md +28 -9
  69. package/pipeline/commands/multi-agent/review-analysis/SKILL.md +5 -1
  70. package/pipeline/commands/multi-agent/review-issue/SKILL.md +4 -0
  71. package/pipeline/commands/multi-agent/review-jira/SKILL.md +4 -0
  72. package/pipeline/commands/multi-agent/route-off/SKILL.md +4 -0
  73. package/pipeline/commands/multi-agent/route-on/SKILL.md +4 -0
  74. package/pipeline/commands/multi-agent/route-status/SKILL.md +4 -0
  75. package/pipeline/commands/multi-agent/routines/SKILL.md +4 -0
  76. package/pipeline/commands/multi-agent/save/SKILL.md +6 -2
  77. package/pipeline/commands/multi-agent/scaffold/SKILL.md +47 -0
  78. package/pipeline/commands/multi-agent/scan/SKILL.md +4 -0
  79. package/pipeline/commands/multi-agent/search/SKILL.md +12 -8
  80. package/pipeline/commands/multi-agent/security-review/SKILL.md +4 -0
  81. package/pipeline/commands/multi-agent/serve/SKILL.md +62 -0
  82. package/pipeline/commands/multi-agent/setup/SKILL.md +27 -36
  83. package/pipeline/commands/multi-agent/stack/SKILL.md +4 -0
  84. package/pipeline/commands/multi-agent/status/SKILL.md +10 -9
  85. package/pipeline/commands/multi-agent/steer/SKILL.md +5 -1
  86. package/pipeline/commands/multi-agent/store-ready/SKILL.md +4 -0
  87. package/pipeline/commands/multi-agent/sync/SKILL.md +13 -13
  88. package/pipeline/commands/multi-agent/test/SKILL.md +4 -0
  89. package/pipeline/commands/multi-agent/test-accessibility/SKILL.md +4 -0
  90. package/pipeline/commands/multi-agent/test-dark-mode/SKILL.md +4 -0
  91. package/pipeline/commands/multi-agent/test-dynamic-type/SKILL.md +4 -0
  92. package/pipeline/commands/multi-agent/test-screenshots/SKILL.md +5 -1
  93. package/pipeline/commands/multi-agent/testflight-validation/SKILL.md +4 -0
  94. package/pipeline/commands/multi-agent/uninstall/SKILL.md +5 -1
  95. package/pipeline/commands/multi-agent/update/SKILL.md +4 -0
  96. package/pipeline/contract/CHANGELOG.md +74 -0
  97. package/pipeline/contract/README.md +126 -0
  98. package/pipeline/contract/build.mjs +427 -0
  99. package/pipeline/contract/fixtures/answer-result.json +11 -0
  100. package/pipeline/contract/fixtures/error-invalid-request.json +8 -0
  101. package/pipeline/contract/fixtures/error-unauthorized.json +5 -0
  102. package/pipeline/contract/fixtures/error-unsigned.json +5 -0
  103. package/pipeline/contract/fixtures/issues-empty.json +18 -0
  104. package/pipeline/contract/fixtures/launch-plan.json +31 -0
  105. package/pipeline/contract/fixtures/runs-awaiting-question.json +70 -0
  106. package/pipeline/contract/fixtures/runs-empty.json +6 -0
  107. package/pipeline/contract/fixtures/runs-failed.json +84 -0
  108. package/pipeline/contract/fixtures/runs-old-schema.json +70 -0
  109. package/pipeline/contract/fixtures/runs-pr-opened-redacted.json +101 -0
  110. package/pipeline/contract/fixtures/runs-pr-opened.json +106 -0
  111. package/pipeline/contract/fixtures/runs-running.json +84 -0
  112. package/pipeline/contract/fixtures/worktrees-empty.json +4 -0
  113. package/pipeline/contract/frozen/toolbox.json +107 -0
  114. package/pipeline/contract/manifest.json +263 -0
  115. package/pipeline/contract/types/index.d.ts +343 -0
  116. package/pipeline/lib/_jira-auth.sh +6 -2
  117. package/pipeline/lib/account-resolver.sh +1 -1
  118. package/pipeline/lib/autopilot-state.sh +19 -0
  119. package/pipeline/lib/context-link-extractor.sh +12 -5
  120. package/pipeline/lib/credential-inventory.sh +12 -5
  121. package/pipeline/lib/credential-store.sh +116 -185
  122. package/pipeline/lib/fetch-confluence.sh +44 -3
  123. package/pipeline/lib/fetch-document.sh +3 -4
  124. package/pipeline/lib/fetch-fortify.sh +1 -1
  125. package/pipeline/lib/figma-mcp-refresh.sh +2 -2
  126. package/pipeline/lib/figma-token.sh +5 -1
  127. package/pipeline/lib/issue-fetcher.sh +233 -16
  128. package/pipeline/lib/json-file-lock.mjs +172 -0
  129. package/pipeline/lib/model-dispatch.sh +21 -12
  130. package/pipeline/lib/model-rung.sh +6 -1
  131. package/pipeline/lib/multi-repo-pipeline.sh +1 -1
  132. package/pipeline/lib/outbound-gate.mjs +46 -16
  133. package/pipeline/lib/parse-complaints.sh +14 -7
  134. package/pipeline/lib/plan-todos.sh +3 -3
  135. package/pipeline/lib/post-pr-review.sh +9 -9
  136. package/pipeline/lib/pr-request-location.mjs +85 -0
  137. package/pipeline/lib/regular-file.mjs +153 -0
  138. package/pipeline/lib/repo-hygiene.sh +17 -0
  139. package/pipeline/lib/route-state.sh +5 -1
  140. package/pipeline/lib/run-paths.sh +3 -2
  141. package/pipeline/lib/stack-detect.sh +19 -1
  142. package/pipeline/lib/unattended-profile-check.mjs +178 -0
  143. package/pipeline/lib/unattended-settings-location.mjs +28 -0
  144. package/pipeline/lib/unattended.mjs +76 -0
  145. package/pipeline/lib/unattended.sh +32 -0
  146. package/pipeline/lib/untrusted.mjs +76 -0
  147. package/pipeline/lib/user-facing.mjs +82 -0
  148. package/pipeline/lib/user-facing.sh +58 -0
  149. package/pipeline/multi-agent-refs/_dev-context.md +1 -1
  150. package/pipeline/multi-agent-refs/analysis/evidence.md +4 -0
  151. package/pipeline/multi-agent-refs/analysis/locked.md +2 -2
  152. package/pipeline/multi-agent-refs/analysis/render.md +4 -3
  153. package/pipeline/multi-agent-refs/analysis/resolve.md +3 -1
  154. package/pipeline/multi-agent-refs/analysis/synthesis.md +4 -5
  155. package/pipeline/multi-agent-refs/analysis-template.md +10 -17
  156. package/pipeline/multi-agent-refs/channels/jira.md +1 -1
  157. package/pipeline/multi-agent-refs/component-dispatch.md +3 -3
  158. package/pipeline/multi-agent-refs/conventions-defaults.md +32 -32
  159. package/pipeline/multi-agent-refs/cross-cli-contract.md +10 -17
  160. package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +37 -9
  161. package/pipeline/multi-agent-refs/features/autopilot-operations.md +277 -0
  162. package/pipeline/multi-agent-refs/features/constitution.md +196 -0
  163. package/pipeline/multi-agent-refs/features/doctor.md +6 -3
  164. package/pipeline/multi-agent-refs/features/external-context-injection.md +6 -1
  165. package/pipeline/multi-agent-refs/features/jira-context.md +1 -1
  166. package/pipeline/multi-agent-refs/features/maturity-followup.md +43 -23
  167. package/pipeline/multi-agent-refs/features/model-fallback.md +9 -0
  168. package/pipeline/multi-agent-refs/features/phone-api.md +306 -0
  169. package/pipeline/multi-agent-refs/features/plan-critic.md +159 -0
  170. package/pipeline/multi-agent-refs/features/research.md +150 -0
  171. package/pipeline/multi-agent-refs/features/review-decision.md +185 -0
  172. package/pipeline/multi-agent-refs/features/scaffold.md +160 -0
  173. package/pipeline/multi-agent-refs/features/security-audit.md +6 -2
  174. package/pipeline/multi-agent-refs/features/skill-conformance.md +4 -1
  175. package/pipeline/multi-agent-refs/features/stack-adapters.md +116 -0
  176. package/pipeline/multi-agent-refs/features/unattended-gates.md +316 -0
  177. package/pipeline/multi-agent-refs/features/unattended-security.md +731 -0
  178. package/pipeline/multi-agent-refs/features/url-enrichment.md +2 -2
  179. package/pipeline/multi-agent-refs/features/usage-reporting.md +127 -31
  180. package/pipeline/multi-agent-refs/features/verify-by-test.md +1 -1
  181. package/pipeline/multi-agent-refs/features/visual-evidence.md +7 -7
  182. package/pipeline/multi-agent-refs/keychain.md +6 -11
  183. package/pipeline/multi-agent-refs/payload-contracts.md +2 -2
  184. package/pipeline/multi-agent-refs/phases/log-format.md +2 -3
  185. package/pipeline/multi-agent-refs/phases/modes.md +10 -12
  186. package/pipeline/multi-agent-refs/phases/operations.md +11 -5
  187. package/pipeline/multi-agent-refs/phases/phase-0-init.md +73 -81
  188. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +22 -22
  189. package/pipeline/multi-agent-refs/phases/phase-2-dev.md +42 -66
  190. package/pipeline/multi-agent-refs/phases/phase-3-review.md +45 -79
  191. package/pipeline/multi-agent-refs/phases/phase-4-commit.md +28 -38
  192. package/pipeline/multi-agent-refs/phases/phase-5-report.md +27 -25
  193. package/pipeline/multi-agent-refs/phases.md +1 -1
  194. package/pipeline/multi-agent-refs/picker-contract.md +1 -1
  195. package/pipeline/multi-agent-refs/progress-contract.md +13 -16
  196. package/pipeline/multi-agent-refs/readiness-review.md +1 -1
  197. package/pipeline/multi-agent-refs/research/engine.md +91 -0
  198. package/pipeline/multi-agent-refs/rules.md +6 -4
  199. package/pipeline/multi-agent-refs/setup/repo-discovery.md +2 -2
  200. package/pipeline/multi-agent-refs/tracker-contract.md +2 -2
  201. package/pipeline/multi-agent-refs/unattended-contract.md +68 -16
  202. package/pipeline/rules/figma-pipeline.md +12 -12
  203. package/pipeline/schemas/agent-state.schema.json +480 -18
  204. package/pipeline/schemas/analysis-spec.schema.json +4 -4
  205. package/pipeline/schemas/answer-request.schema.json +24 -0
  206. package/pipeline/schemas/answer-result.schema.json +28 -0
  207. package/pipeline/schemas/autopilot-config.schema.json +111 -13
  208. package/pipeline/schemas/command-parameters.schema.json +99 -0
  209. package/pipeline/schemas/constitution.schema.json +56 -0
  210. package/pipeline/schemas/contract-error.schema.json +52 -0
  211. package/pipeline/schemas/design-check-config.schema.json +5 -1
  212. package/pipeline/schemas/issues.schema.json +61 -0
  213. package/pipeline/schemas/launch-plan.schema.json +46 -0
  214. package/pipeline/schemas/launch-request.schema.json +45 -0
  215. package/pipeline/schemas/launch.json +61 -0
  216. package/pipeline/schemas/launch.schema.json +84 -0
  217. package/pipeline/schemas/phases.json +2 -2
  218. package/pipeline/schemas/phases.schema.json +68 -0
  219. package/pipeline/schemas/phone-devices.schema.json +61 -0
  220. package/pipeline/schemas/phone-signed-request.schema.json +67 -0
  221. package/pipeline/schemas/plan-critique.schema.json +99 -0
  222. package/pipeline/schemas/plan-todos.schema.json +7 -7
  223. package/pipeline/schemas/planning-output.schema.json +5 -0
  224. package/pipeline/schemas/pr-request.schema.json +46 -0
  225. package/pipeline/schemas/prefs.schema.json +82 -7
  226. package/pipeline/schemas/research-output.schema.json +118 -0
  227. package/pipeline/schemas/review-file-exclusions.schema.json +25 -0
  228. package/pipeline/schemas/reviewer-output.schema.json +40 -4
  229. package/pipeline/schemas/run-questions.json +392 -0
  230. package/pipeline/schemas/run-questions.schema.json +118 -0
  231. package/pipeline/schemas/runs-index.schema.json +189 -0
  232. package/pipeline/schemas/scaffold-manifest.schema.json +43 -0
  233. package/pipeline/schemas/secret-patterns.schema.json +28 -0
  234. package/pipeline/schemas/stack-adapters.json +527 -0
  235. package/pipeline/schemas/stack-adapters.schema.json +184 -0
  236. package/pipeline/schemas/token-budget.json +1 -1
  237. package/pipeline/schemas/token-budget.schema.json +26 -0
  238. package/pipeline/schemas/triage-output.schema.json +64 -3
  239. package/pipeline/schemas/unattended-policy.json +139 -0
  240. package/pipeline/schemas/unattended-policy.schema.json +73 -0
  241. package/pipeline/schemas/unattended-profile.json +248 -0
  242. package/pipeline/schemas/unattended-profile.schema.json +198 -0
  243. package/pipeline/schemas/worktrees.schema.json +51 -0
  244. package/pipeline/scripts/README.md +1 -0
  245. package/pipeline/scripts/_autopilot-config.mjs +130 -0
  246. package/pipeline/scripts/_autopilot-ops.mjs +567 -0
  247. package/pipeline/scripts/_autopilot-outcomes.mjs +174 -0
  248. package/pipeline/scripts/_command-contract.mjs +384 -0
  249. package/pipeline/scripts/_cost.mjs +40 -0
  250. package/pipeline/scripts/_notices.mjs +160 -0
  251. package/pipeline/scripts/_phone-auth.mjs +485 -0
  252. package/pipeline/scripts/_pre-existing.mjs +294 -0
  253. package/pipeline/scripts/_redact.mjs +77 -0
  254. package/pipeline/scripts/_run-paths.mjs +4 -2
  255. package/pipeline/scripts/_stack-adapter.mjs +678 -0
  256. package/pipeline/scripts/_stack-routing.mjs +1 -1
  257. package/pipeline/scripts/agent-guard.py +348 -37
  258. package/pipeline/scripts/agent-guard.sh +41 -13
  259. package/pipeline/scripts/analysis-story-tree.mjs +79 -3
  260. package/pipeline/scripts/answer-question.mjs +181 -0
  261. package/pipeline/scripts/audit-log-rotate.sh +1 -4
  262. package/pipeline/scripts/audit-log.sh +4 -4
  263. package/pipeline/scripts/autopilot-arming.mjs +389 -21
  264. package/pipeline/scripts/autopilot-awake.mjs +255 -0
  265. package/pipeline/scripts/autopilot-intake.mjs +137 -36
  266. package/pipeline/scripts/autopilot-menubar.swift +156 -44
  267. package/pipeline/scripts/autopilot-publish.mjs +1625 -0
  268. package/pipeline/scripts/autopilot-runner.mjs +1678 -222
  269. package/pipeline/scripts/autopilot-status.sh +198 -33
  270. package/pipeline/scripts/build-lock.sh +120 -0
  271. package/pipeline/scripts/build-references.mjs +4 -1
  272. package/pipeline/scripts/build-stack-plugins.mjs +59 -22
  273. package/pipeline/scripts/capture-flush.sh +1 -1
  274. package/pipeline/scripts/capture-resume.sh +13 -9
  275. package/pipeline/scripts/check-derived-drift.mjs +52 -11
  276. package/pipeline/scripts/commands.mjs +88 -0
  277. package/pipeline/scripts/constitution.mjs +362 -0
  278. package/pipeline/scripts/contract-server.mjs +776 -0
  279. package/pipeline/scripts/cost-analyze.mjs +89 -39
  280. package/pipeline/scripts/diff-explain.mjs +12 -1
  281. package/pipeline/scripts/doctor.mjs +77 -28
  282. package/pipeline/scripts/evidence-gate.mjs +192 -12
  283. package/pipeline/scripts/feedback-send.mjs +4 -2
  284. package/pipeline/scripts/gate-ledger.mjs +449 -0
  285. package/pipeline/scripts/gc-abandoned.sh +132 -13
  286. package/pipeline/scripts/gen-facts.mjs +31 -15
  287. package/pipeline/scripts/gen-mode-dispatch.mjs +3 -3
  288. package/pipeline/scripts/github-ssh-setup.sh +140 -29
  289. package/pipeline/scripts/graph-mermaid.mjs +4 -1
  290. package/pipeline/scripts/issues.mjs +236 -0
  291. package/pipeline/scripts/jira-attach.sh +6 -2
  292. package/pipeline/scripts/jira-search.sh +4 -3
  293. package/pipeline/scripts/keychain-save.sh +125 -24
  294. package/pipeline/scripts/keychain.py +63 -93
  295. package/pipeline/scripts/launch-request.mjs +747 -0
  296. package/pipeline/scripts/localize-commands.mjs +4 -10
  297. package/pipeline/scripts/log-metric.sh +6 -5
  298. package/pipeline/scripts/maturity-followup.mjs +13 -4
  299. package/pipeline/scripts/memory-save.sh +25 -0
  300. package/pipeline/scripts/migrate-prefs.mjs +4 -3
  301. package/pipeline/scripts/open-questions-gate.mjs +276 -0
  302. package/pipeline/scripts/phase-tracker.sh +41 -27
  303. package/pipeline/scripts/phase0-exit-gate.mjs +22 -4
  304. package/pipeline/scripts/phone-devices.mjs +224 -0
  305. package/pipeline/scripts/plan-coverage-gate.mjs +200 -66
  306. package/pipeline/scripts/plan-critique-gate.mjs +591 -0
  307. package/pipeline/scripts/pr-request.mjs +188 -0
  308. package/pipeline/scripts/pre-commit-check.sh +115 -4
  309. package/pipeline/scripts/probe-evidence-capability.sh +44 -5
  310. package/pipeline/scripts/record-phase.mjs +71 -0
  311. package/pipeline/scripts/render-agent-log-cost.sh +17 -2
  312. package/pipeline/scripts/render-cost-summary.sh +1 -1
  313. package/pipeline/scripts/render-work-summary.sh +1 -1
  314. package/pipeline/scripts/require-supported-version.sh +4 -1
  315. package/pipeline/scripts/research-gate.mjs +704 -0
  316. package/pipeline/scripts/review-decision-gate.mjs +403 -0
  317. package/pipeline/scripts/routine-registry.mjs +5 -2
  318. package/pipeline/scripts/runs-index.mjs +135 -27
  319. package/pipeline/scripts/scaffold-gate.mjs +393 -0
  320. package/pipeline/scripts/skill-conformance.mjs +25 -8
  321. package/pipeline/scripts/skill-siblings.mjs +2 -1
  322. package/pipeline/scripts/smoke-cross-cli-behavior.sh +32 -6
  323. package/pipeline/scripts/smoke-schema-validation.sh +6 -2
  324. package/pipeline/scripts/spec-consistency-gate.mjs +469 -0
  325. package/pipeline/scripts/symbol-existence-gate.mjs +450 -0
  326. package/pipeline/scripts/test-gap-scan.mjs +40 -2
  327. package/pipeline/scripts/test-integrity-gate.mjs +20 -4
  328. package/pipeline/scripts/test-strength.mjs +484 -0
  329. package/pipeline/scripts/test-summary.mjs +651 -0
  330. package/pipeline/scripts/triage-memory.mjs +49 -9
  331. package/pipeline/scripts/unattended_policy.py +2786 -0
  332. package/pipeline/scripts/uninstall.mjs +10 -10
  333. package/pipeline/scripts/update-issue-progress.sh +1 -1
  334. package/pipeline/scripts/usage-identity.mjs +288 -0
  335. package/pipeline/scripts/usage-register.mjs +185 -63
  336. package/pipeline/scripts/usage-report.mjs +230 -66
  337. package/pipeline/scripts/validate-complaint-doc.mjs +28 -10
  338. package/pipeline/scripts/validate-planning.mjs +6 -0
  339. package/pipeline/scripts/verify-citations.mjs +151 -38
  340. package/pipeline/scripts/verify.mjs +58 -18
  341. package/pipeline/scripts/worktree-prepare.sh +126 -0
  342. package/pipeline/scripts/worktrees.mjs +124 -0
  343. package/pipeline/scripts/write-state.mjs +48 -17
  344. package/pipeline/skills/.skill-manifest.json +222 -226
  345. package/pipeline/skills/.skills-index.json +77 -88
  346. package/pipeline/skills/shared/README.md +44 -45
  347. package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +3 -2
  348. package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +3 -2
  349. package/pipeline/skills/shared/core/multi-agent/SKILL.md +10 -9
  350. package/pipeline/skills/shared/core/multi-agent-analysis/SKILL.md +6 -5
  351. package/pipeline/skills/shared/core/multi-agent-analysis-jira/SKILL.md +11 -3
  352. package/pipeline/skills/shared/core/multi-agent-analysis-resolve/SKILL.md +2 -1
  353. package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +4 -3
  354. package/pipeline/skills/shared/core/multi-agent-autopilot-off/SKILL.md +32 -6
  355. package/pipeline/skills/shared/core/multi-agent-autopilot-on/SKILL.md +66 -8
  356. package/pipeline/skills/shared/core/multi-agent-autopilot-status/SKILL.md +27 -9
  357. package/pipeline/skills/shared/core/multi-agent-build-optimize/SKILL.md +2 -1
  358. package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +7 -4
  359. package/pipeline/skills/shared/core/multi-agent-complaint-analysis/SKILL.md +3 -2
  360. package/pipeline/skills/shared/core/multi-agent-create-jira/SKILL.md +2 -1
  361. package/pipeline/skills/shared/core/multi-agent-design-check/SKILL.md +7 -5
  362. package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +2 -1
  363. package/pipeline/skills/shared/core/multi-agent-doctor/SKILL.md +3 -2
  364. package/pipeline/skills/shared/core/multi-agent-feedback/SKILL.md +2 -1
  365. package/pipeline/skills/shared/core/multi-agent-forget/SKILL.md +2 -1
  366. package/pipeline/skills/shared/core/multi-agent-garbage-collect/SKILL.md +2 -1
  367. package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +2 -1
  368. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +2 -1
  369. package/pipeline/skills/shared/core/multi-agent-ios-coding-standard/SKILL.md +2 -1
  370. package/pipeline/skills/shared/core/multi-agent-issue/SKILL.md +3 -2
  371. package/pipeline/skills/shared/core/multi-agent-jira/SKILL.md +2 -1
  372. package/pipeline/skills/shared/core/multi-agent-kill/SKILL.md +7 -4
  373. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -1
  374. package/pipeline/skills/shared/core/multi-agent-log/SKILL.md +2 -1
  375. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +2 -1
  376. package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +2 -1
  377. package/pipeline/skills/shared/core/multi-agent-prune-logs/SKILL.md +2 -1
  378. package/pipeline/skills/shared/core/multi-agent-prune-prompts/SKILL.md +2 -1
  379. package/pipeline/skills/shared/core/multi-agent-purge/SKILL.md +2 -1
  380. package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +2 -1
  381. package/pipeline/skills/shared/core/multi-agent-research/SKILL.md +35 -0
  382. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +7 -3
  383. package/pipeline/skills/shared/core/multi-agent-review/SKILL.md +7 -5
  384. package/pipeline/skills/shared/core/multi-agent-review-analysis/SKILL.md +2 -1
  385. package/pipeline/skills/shared/core/multi-agent-review-issue/SKILL.md +3 -2
  386. package/pipeline/skills/shared/core/multi-agent-review-jira/SKILL.md +2 -1
  387. package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +2 -1
  388. package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +2 -1
  389. package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +2 -1
  390. package/pipeline/skills/shared/core/multi-agent-routines/SKILL.md +2 -1
  391. package/pipeline/skills/shared/core/multi-agent-save/SKILL.md +3 -2
  392. package/pipeline/skills/shared/core/multi-agent-scaffold/SKILL.md +30 -0
  393. package/pipeline/skills/shared/core/multi-agent-scan/SKILL.md +2 -1
  394. package/pipeline/skills/shared/core/multi-agent-search/SKILL.md +4 -3
  395. package/pipeline/skills/shared/core/multi-agent-security-review/SKILL.md +2 -1
  396. package/pipeline/skills/shared/core/multi-agent-serve/SKILL.md +60 -0
  397. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +8 -8
  398. package/pipeline/skills/shared/core/multi-agent-stack/SKILL.md +2 -1
  399. package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +7 -8
  400. package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -1
  401. package/pipeline/skills/shared/core/multi-agent-store-ready/SKILL.md +2 -1
  402. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +10 -10
  403. package/pipeline/skills/shared/core/multi-agent-test/SKILL.md +2 -1
  404. package/pipeline/skills/shared/core/multi-agent-test-accessibility/SKILL.md +2 -1
  405. package/pipeline/skills/shared/core/multi-agent-test-dark-mode/SKILL.md +2 -1
  406. package/pipeline/skills/shared/core/multi-agent-test-dynamic-type/SKILL.md +2 -1
  407. package/pipeline/skills/shared/core/multi-agent-test-screenshots/SKILL.md +2 -1
  408. package/pipeline/skills/shared/core/multi-agent-testflight-validation/SKILL.md +2 -1
  409. package/pipeline/skills/shared/core/multi-agent-uninstall/SKILL.md +3 -2
  410. package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +2 -1
  411. package/pipeline/skills/shared/external/NOTICE-avdlee-swiftui-agent-skill.md +46 -0
  412. package/pipeline/skills/shared/external/NOTICE-dimillian-skills.md +9 -3
  413. package/pipeline/skills/shared/external/NOTICE-paul-hudson-skills.md +54 -0
  414. package/pipeline/skills/shared/external/NOTICE-swift-ios-skills.md +9 -11
  415. package/pipeline/skills/shared/external/NOTICE-vibeship-spawner-skills.md +203 -0
  416. package/pipeline/skills/shared/external/NOTICE-xcode-build-skills.md +10 -3
  417. package/pipeline/skills/shared/external/accessibility-compliance-accessibility-audit/SKILL.md +131 -25
  418. package/pipeline/skills/shared/external/agent-introspection-debugging/SKILL.md +4 -3
  419. package/pipeline/skills/shared/external/alarmkit/SKILL.md +2 -1
  420. package/pipeline/skills/shared/external/android-architecture/SKILL.md +85 -76
  421. package/pipeline/skills/shared/external/android-build-quality-gates/SKILL.md +51 -3
  422. package/pipeline/skills/shared/external/android-build-quality-gates/references/patterns.md +9 -0
  423. package/pipeline/skills/shared/external/android-datastore/SKILL.md +4 -3
  424. package/pipeline/skills/shared/external/android-design-tokens-codegen/SKILL.md +4 -3
  425. package/pipeline/skills/shared/external/android-jetpack-compose-expert/SKILL.md +106 -181
  426. package/pipeline/skills/shared/external/android-jetpack-compose-expert/references/patterns.md +147 -0
  427. package/pipeline/skills/shared/external/android-mvi-viewmodel/SKILL.md +45 -4
  428. package/pipeline/skills/shared/external/android-performance/SKILL.md +27 -2
  429. package/pipeline/skills/shared/external/android-performance/references/patterns.md +37 -37
  430. package/pipeline/skills/shared/external/api-patterns/SKILL.md +111 -67
  431. package/pipeline/skills/shared/external/api-patterns/references/contract-details.md +128 -0
  432. package/pipeline/skills/shared/external/api-security-best-practices/SKILL.md +136 -186
  433. package/pipeline/skills/shared/external/api-security-best-practices/references/abuse-controls.md +140 -0
  434. package/pipeline/skills/shared/external/api-security-best-practices/references/identity-and-access.md +207 -0
  435. package/pipeline/skills/shared/external/api-security-best-practices/references/operations.md +66 -0
  436. package/pipeline/skills/shared/external/api-security-best-practices/references/request-handling.md +175 -0
  437. package/pipeline/skills/shared/external/app-clips/SKILL.md +2 -0
  438. package/pipeline/skills/shared/external/app-intents/SKILL.md +2 -0
  439. package/pipeline/skills/shared/external/app-store-changelog/SKILL.md +4 -3
  440. package/pipeline/skills/shared/external/app-store-optimization/SKILL.md +2 -0
  441. package/pipeline/skills/shared/external/app-store-review/SKILL.md +2 -0
  442. package/pipeline/skills/shared/external/apple-on-device-ai/SKILL.md +2 -0
  443. package/pipeline/skills/shared/external/architecture/SKILL.md +106 -39
  444. package/pipeline/skills/shared/external/architecture/references/adr-and-review.md +76 -0
  445. package/pipeline/skills/shared/external/authentication/SKILL.md +2 -0
  446. package/pipeline/skills/shared/external/avkit/SKILL.md +2 -0
  447. package/pipeline/skills/shared/external/background-processing/SKILL.md +2 -1
  448. package/pipeline/skills/shared/external/backlog/SKILL.md +3 -2
  449. package/pipeline/skills/shared/external/callkit-voip/SKILL.md +2 -0
  450. package/pipeline/skills/shared/external/callkit-voip/evals/evals.json +1 -1
  451. package/pipeline/skills/shared/external/ci-cd-pipelines/SKILL.md +3 -2
  452. package/pipeline/skills/shared/external/clean-code/SKILL.md +192 -90
  453. package/pipeline/skills/shared/external/cloudkit-sync/SKILL.md +2 -1
  454. package/pipeline/skills/shared/external/cloudkit-sync/evals/evals.json +1 -1
  455. package/pipeline/skills/shared/external/compose-components/SKILL.md +4 -4
  456. package/pipeline/skills/shared/external/compose-navigation/SKILL.md +22 -22
  457. package/pipeline/skills/shared/external/compose-navigation/references/patterns.md +10 -10
  458. package/pipeline/skills/shared/external/compose-testing/SKILL.md +13 -6
  459. package/pipeline/skills/shared/external/compose-testing/references/patterns.md +59 -59
  460. package/pipeline/skills/shared/external/contacts-framework/SKILL.md +2 -0
  461. package/pipeline/skills/shared/external/context-compression/SKILL.md +118 -250
  462. package/pipeline/skills/shared/external/core-bluetooth/SKILL.md +2 -0
  463. package/pipeline/skills/shared/external/core-data/SKILL.md +2 -0
  464. package/pipeline/skills/shared/external/core-motion/SKILL.md +2 -0
  465. package/pipeline/skills/shared/external/core-nfc/SKILL.md +2 -0
  466. package/pipeline/skills/shared/external/coreml/SKILL.md +2 -1
  467. package/pipeline/skills/shared/external/council/SKILL.md +2 -1
  468. package/pipeline/skills/shared/external/cryptokit/SKILL.md +2 -0
  469. package/pipeline/skills/shared/external/css-modern/SKILL.md +3 -2
  470. package/pipeline/skills/shared/external/database-patterns/SKILL.md +3 -2
  471. package/pipeline/skills/shared/external/debugging-instruments/SKILL.md +2 -0
  472. package/pipeline/skills/shared/external/debugging-strategies/SKILL.md +130 -20
  473. package/pipeline/skills/shared/external/debugging-strategies/references/hard-cases.md +54 -0
  474. package/pipeline/skills/shared/external/device-integrity/SKILL.md +2 -0
  475. package/pipeline/skills/shared/external/docker-expert/SKILL.md +137 -381
  476. package/pipeline/skills/shared/external/docker-expert/references/patterns.md +98 -0
  477. package/pipeline/skills/shared/external/energykit/SKILL.md +2 -1
  478. package/pipeline/skills/shared/external/eventkit-calendar/SKILL.md +3 -1
  479. package/pipeline/skills/shared/external/eventkit-calendar/evals/evals.json +2 -2
  480. package/pipeline/skills/shared/external/fastapi-pro/SKILL.md +116 -185
  481. package/pipeline/skills/shared/external/fastapi-pro/references/app-structure.md +235 -0
  482. package/pipeline/skills/shared/external/fastapi-pro/references/testing-and-deployment.md +69 -0
  483. package/pipeline/skills/shared/external/firebase/SKILL.md +4 -3
  484. package/pipeline/skills/shared/external/github-actions-templates/SKILL.md +151 -293
  485. package/pipeline/skills/shared/external/github-actions-templates/references/patterns.md +96 -0
  486. package/pipeline/skills/shared/external/healthkit/SKILL.md +2 -0
  487. package/pipeline/skills/shared/external/hig-components-content/SKILL.md +69 -72
  488. package/pipeline/skills/shared/external/hig-components-content/references/content-views.md +111 -0
  489. package/pipeline/skills/shared/external/hig-components-layout/SKILL.md +77 -86
  490. package/pipeline/skills/shared/external/hig-components-layout/references/containers.md +98 -0
  491. package/pipeline/skills/shared/external/hig-components-status/SKILL.md +150 -78
  492. package/pipeline/skills/shared/external/hig-components-system/SKILL.md +76 -97
  493. package/pipeline/skills/shared/external/hig-components-system/references/surfaces.md +98 -0
  494. package/pipeline/skills/shared/external/hig-foundations/SKILL.md +48 -76
  495. package/pipeline/skills/shared/external/hig-foundations/references/foundations-detail.md +132 -0
  496. package/pipeline/skills/shared/external/hig-inputs/SKILL.md +147 -106
  497. package/pipeline/skills/shared/external/hig-patterns/SKILL.md +61 -77
  498. package/pipeline/skills/shared/external/hig-patterns/references/patterns.md +92 -0
  499. package/pipeline/skills/shared/external/hig-platforms/SKILL.md +147 -77
  500. package/pipeline/skills/shared/external/hig-technologies/SKILL.md +61 -121
  501. package/pipeline/skills/shared/external/hig-technologies/references/technologies.md +108 -0
  502. package/pipeline/skills/shared/external/homekit-matter/SKILL.md +2 -0
  503. package/pipeline/skills/shared/external/homekit-matter/evals/evals.json +1 -1
  504. package/pipeline/skills/shared/external/html-semantic/SKILL.md +3 -2
  505. package/pipeline/skills/shared/external/humanizer/SKILL.md +2 -1
  506. package/pipeline/skills/shared/external/ios-accessibility/SKILL.md +2 -0
  507. package/pipeline/skills/shared/external/ios-coding-standard/SKILL.md +2 -1
  508. package/pipeline/skills/shared/external/ios-coding-standard/references/STANDARD.md +14 -2
  509. package/pipeline/skills/shared/external/ios-coding-standard/references/rules.yml +2 -2
  510. package/pipeline/skills/shared/external/ios-debugger-agent/SKILL.md +4 -3
  511. package/pipeline/skills/shared/external/ios-localization/SKILL.md +8 -6
  512. package/pipeline/skills/shared/external/ios-localization/evals/evals.json +2 -2
  513. package/pipeline/skills/shared/external/ios-localization/references/string-catalogs.md +11 -11
  514. package/pipeline/skills/shared/external/ios-module-structure/SKILL.md +2 -1
  515. package/pipeline/skills/shared/external/ios-networking/SKILL.md +2 -0
  516. package/pipeline/skills/shared/external/ios-simulator/SKILL.md +2 -0
  517. package/pipeline/skills/shared/external/kotlin-coroutines-expert/SKILL.md +102 -193
  518. package/pipeline/skills/shared/external/kotlin-coroutines-expert/references/patterns.md +103 -0
  519. package/pipeline/skills/shared/external/live-activities/SKILL.md +4 -2
  520. package/pipeline/skills/shared/external/live-activities/evals/evals.json +2 -2
  521. package/pipeline/skills/shared/external/localization-reuse-map/SKILL.md +7 -9
  522. package/pipeline/skills/shared/external/localization-reuse-map/reference/sources-and-recipes.md +3 -3
  523. package/pipeline/skills/shared/external/localization-reuse-map/scripts/resolve-new-values.py +2 -2
  524. package/pipeline/skills/shared/external/macos-menubar-tuist-app/SKILL.md +4 -3
  525. package/pipeline/skills/shared/external/macos-spm-app-packaging/SKILL.md +5 -3
  526. package/pipeline/skills/shared/external/mapkit-location/SKILL.md +2 -0
  527. package/pipeline/skills/shared/external/mapkit-location/evals/evals.json +1 -1
  528. package/pipeline/skills/shared/external/metrickit-diagnostics/SKILL.md +2 -0
  529. package/pipeline/skills/shared/external/metrickit-diagnostics/evals/evals.json +1 -1
  530. package/pipeline/skills/shared/external/monorepo-architect/SKILL.md +149 -46
  531. package/pipeline/skills/shared/external/monorepo-architect/references/patterns.md +70 -0
  532. package/pipeline/skills/shared/external/musickit-audio/SKILL.md +2 -0
  533. package/pipeline/skills/shared/external/musickit-audio/evals/evals.json +1 -1
  534. package/pipeline/skills/shared/external/natural-language/SKILL.md +2 -0
  535. package/pipeline/skills/shared/external/nextjs-app-router/SKILL.md +3 -2
  536. package/pipeline/skills/shared/external/nodejs-backend-patterns/SKILL.md +91 -21
  537. package/pipeline/skills/shared/external/nodejs-backend-patterns/references/runtime-patterns.md +247 -0
  538. package/pipeline/skills/shared/external/nodejs-backend-patterns/references/testing-and-frameworks.md +57 -0
  539. package/pipeline/skills/shared/external/observability-engineer/SKILL.md +155 -232
  540. package/pipeline/skills/shared/external/observability-engineer/references/patterns.md +75 -0
  541. package/pipeline/skills/shared/external/passkit-wallet/SKILL.md +2 -0
  542. package/pipeline/skills/shared/external/passkit-wallet/evals/evals.json +2 -2
  543. package/pipeline/skills/shared/external/passkit-wallet/references/wallet-passes.md +18 -18
  544. package/pipeline/skills/shared/external/pdfkit/SKILL.md +2 -1
  545. package/pipeline/skills/shared/external/pencilkit-drawing/SKILL.md +2 -0
  546. package/pipeline/skills/shared/external/pencilkit-drawing/evals/evals.json +1 -1
  547. package/pipeline/skills/shared/external/permissionkit/SKILL.md +2 -0
  548. package/pipeline/skills/shared/external/photos-camera-media/SKILL.md +2 -1
  549. package/pipeline/skills/shared/external/push-notifications/SKILL.md +2 -0
  550. package/pipeline/skills/shared/external/python-patterns/SKILL.md +3 -2
  551. package/pipeline/skills/shared/external/react-best-practices/SKILL.md +3 -2
  552. package/pipeline/skills/shared/external/realitykit-ar/SKILL.md +2 -2
  553. package/pipeline/skills/shared/external/realitykit-ar/evals/evals.json +1 -1
  554. package/pipeline/skills/shared/external/rest-api-design/SKILL.md +3 -2
  555. package/pipeline/skills/shared/external/retrofit-networking/SKILL.md +7 -7
  556. package/pipeline/skills/shared/external/retrofit-networking/references/patterns.md +69 -69
  557. package/pipeline/skills/shared/external/room-database/references/patterns.md +126 -126
  558. package/pipeline/skills/shared/external/search-first/SKILL.md +2 -1
  559. package/pipeline/skills/shared/external/security-review/SKILL.md +4 -3
  560. package/pipeline/skills/shared/external/shareplay-activities/SKILL.md +2 -2
  561. package/pipeline/skills/shared/external/signal-community/SKILL.md +1 -1
  562. package/pipeline/skills/shared/external/skill-creator/SKILL.md +2 -1
  563. package/pipeline/skills/shared/external/speech-recognition/SKILL.md +2 -1
  564. package/pipeline/skills/shared/external/spm-build-analysis/SKILL.md +2 -0
  565. package/pipeline/skills/shared/external/storekit/SKILL.md +2 -0
  566. package/pipeline/skills/shared/external/swift-api-design-guidelines/SKILL.md +2 -0
  567. package/pipeline/skills/shared/external/swift-architecture/SKILL.md +2 -1
  568. package/pipeline/skills/shared/external/swift-charts/SKILL.md +2 -1
  569. package/pipeline/skills/shared/external/swift-codable/SKILL.md +2 -0
  570. package/pipeline/skills/shared/external/swift-concurrency/SKILL.md +3 -1
  571. package/pipeline/skills/shared/external/swift-concurrency-expert/SKILL.md +5 -4
  572. package/pipeline/skills/shared/external/swift-concurrency-pro/SKILL.md +2 -1
  573. package/pipeline/skills/shared/external/swift-formatstyle/SKILL.md +2 -0
  574. package/pipeline/skills/shared/external/swift-language/SKILL.md +3 -1
  575. package/pipeline/skills/shared/external/swift-security/SKILL.md +2 -1
  576. package/pipeline/skills/shared/external/swift-testing/SKILL.md +2 -0
  577. package/pipeline/skills/shared/external/swift-testing-pro/SKILL.md +2 -1
  578. package/pipeline/skills/shared/external/swift-testing-pro/references/new-features.md +10 -0
  579. package/pipeline/skills/shared/external/swiftdata/SKILL.md +2 -0
  580. package/pipeline/skills/shared/external/swiftdata-pro/SKILL.md +2 -1
  581. package/pipeline/skills/shared/external/swiftlint/SKILL.md +2 -0
  582. package/pipeline/skills/shared/external/swiftui-animation/SKILL.md +2 -0
  583. package/pipeline/skills/shared/external/swiftui-expert-skill/SKILL.md +3 -1
  584. package/pipeline/skills/shared/external/swiftui-gestures/SKILL.md +2 -0
  585. package/pipeline/skills/shared/external/swiftui-layout-components/SKILL.md +2 -0
  586. package/pipeline/skills/shared/external/swiftui-liquid-glass/SKILL.md +2 -0
  587. package/pipeline/skills/shared/external/swiftui-navigation/SKILL.md +2 -0
  588. package/pipeline/skills/shared/external/swiftui-patterns/SKILL.md +3 -1
  589. package/pipeline/skills/shared/external/swiftui-performance/SKILL.md +3 -1
  590. package/pipeline/skills/shared/external/swiftui-performance-audit/SKILL.md +5 -4
  591. package/pipeline/skills/shared/external/swiftui-pro/SKILL.md +2 -1
  592. package/pipeline/skills/shared/external/swiftui-ui-patterns/SKILL.md +5 -4
  593. package/pipeline/skills/shared/external/swiftui-uikit-interop/SKILL.md +2 -0
  594. package/pipeline/skills/shared/external/swiftui-view-refactor/SKILL.md +4 -3
  595. package/pipeline/skills/shared/external/swiftui-webkit/SKILL.md +2 -0
  596. package/pipeline/skills/shared/external/tailwind-css/SKILL.md +3 -2
  597. package/pipeline/skills/shared/external/testing-backend/SKILL.md +3 -2
  598. package/pipeline/skills/shared/external/tipkit/SKILL.md +2 -0
  599. package/pipeline/skills/shared/external/typescript-patterns/SKILL.md +3 -2
  600. package/pipeline/skills/shared/external/vision-framework/SKILL.md +2 -0
  601. package/pipeline/skills/shared/external/vue-composition/SKILL.md +3 -2
  602. package/pipeline/skills/shared/external/weatherkit/SKILL.md +2 -0
  603. package/pipeline/skills/shared/external/web-accessibility/SKILL.md +3 -2
  604. package/pipeline/skills/shared/external/web-performance/SKILL.md +3 -2
  605. package/pipeline/skills/shared/external/web-testing/SKILL.md +3 -2
  606. package/pipeline/skills/shared/external/widgetkit/SKILL.md +2 -0
  607. package/pipeline/skills/shared/external/widgetkit/references/widgetkit-advanced.md +1 -1
  608. package/pipeline/skills/shared/external/xcode-build-benchmark/SKILL.md +2 -0
  609. package/pipeline/skills/shared/external/xcode-build-fixer/SKILL.md +2 -0
  610. package/pipeline/skills/shared/external/xcode-build-orchestrator/SKILL.md +2 -0
  611. package/pipeline/skills/shared/external/xcode-compilation-analyzer/SKILL.md +2 -0
  612. package/pipeline/skills/shared/external/xcode-project-analyzer/SKILL.md +2 -0
  613. package/pipeline/skills/skills-index.md +40 -41
  614. package/docs/token-budget-history.md +0 -24
  615. package/pipeline/skills/shared/external/agentflow/SKILL.md +0 -199
  616. package/pipeline/skills/shared/external/android-ui-verification/SKILL.md +0 -66
  617. package/pipeline/skills/shared/external/api-security-best-practices/references/auth.md +0 -299
  618. package/pipeline/skills/shared/external/api-security-best-practices/references/input-validation.md +0 -255
  619. package/pipeline/skills/shared/external/api-security-best-practices/references/rate-limiting.md +0 -167
  620. package/pipeline/skills/shared/external/closed-loop-delivery/SKILL.md +0 -116
  621. package/pipeline/skills/shared/external/ios-developer/SKILL.md +0 -216
@@ -14,17 +14,34 @@
14
14
  * 4. TAKE the head of the queue, and only if this repo is free.
15
15
  * 5. RUN one `claude --bg` child, then SUPERVISED to the end. `--bg`
16
16
  * returns immediately, so the spawn proves only that a run began.
17
+ * 5c. RESEARCH a run parked on a maturity blocker or on open questions gets a
18
+ * research pass before it is left for a person: one more session
19
+ * (`/multi-agent:research <id> --autonomous`), then
20
+ * research-gate.mjs decides; proceed resumes the run, anything
21
+ * else leaves it parked. Bounded by config `maxAskRounds`.
17
22
  * 6. RECORD attempted.jsonl, clear the pid file, refresh status.json.
18
23
  *
24
+ * Around the run, three operations (_autopilot-ops.mjs): a sleep inhibitor held
25
+ * from the launch to the end of the tick; `maxParallelAgents`, a ceiling on
26
+ * runner-launched sessions working at once; and two jobs at most once per
27
+ * local day - the cleanup report over the worktrees this runner created, and
28
+ * the opt-in digest sent through `reportChannels`.
29
+ *
19
30
  * LIVENESS WITHOUT A LEASE. Lease arithmetic exists to arbitrate between rival
20
31
  * writers, and here there are none: one machine, one queue, one writer. The
21
- * question is only "is the runner I recorded still alive", and three cheap facts
22
- * answer it exactly - the pid responds to signal 0, the machine has not rebooted
23
- * since the pid was recorded, and the session is in `claude agents --json`. The
24
- * middle one is what makes a reboot safe without probing anything: after a
25
- * restart pids start low, so a recorded 4711 may well belong to something real
26
- * and unrelated, and a runner that trusted `kill -0` alone would decide another
27
- * runner was live and do nothing, forever.
32
+ * question is only "is the run I recorded still alive", and three cheap facts
33
+ * answer it - the machine has not rebooted since the claim was recorded, the
34
+ * session is in `claude agents --json`, and either the recorded pid responds to
35
+ * signal 0 or the session reports itself working. The boot check is what makes
36
+ * a reboot safe without probing anything: after a restart pids start low, so a
37
+ * recorded 4711 may well belong to something real and unrelated, and a runner
38
+ * that trusted `kill -0` alone would decide another runner was live and do
39
+ * nothing, forever. The session is what decides the rest, because the pid is
40
+ * the supervisor's and the session outlives it.
41
+ *
42
+ * A run that stops to ask a person is PARKED: its claim moves into `running`,
43
+ * where it holds its repo but not a slot, and it is settled from its own state
44
+ * once its session ends. Nothing retires it.
28
45
  *
29
46
  * WHAT IT WILL NOT DO. It does not merge. It does not remove a worktree holding
30
47
  * uncommitted work - that is stashed under a named message and the worktree
@@ -42,25 +59,67 @@
42
59
 
43
60
  import { execFileSync, spawnSync } from "node:child_process";
44
61
  import {
62
+ accessSync,
63
+ constants as fsConstants,
64
+ copyFileSync,
45
65
  existsSync,
46
66
  readFileSync,
67
+ realpathSync,
47
68
  writeFileSync,
48
69
  appendFileSync,
49
70
  mkdirSync,
71
+ renameSync,
50
72
  rmSync,
51
73
  chmodSync,
52
- readdirSync,
53
74
  statSync,
54
75
  openSync,
55
76
  readSync,
56
77
  closeSync,
57
78
  truncateSync,
58
79
  } from "node:fs";
59
- import { join } from "node:path";
80
+ import { delimiter, dirname, isAbsolute, join } from "node:path";
60
81
  import { homedir } from "node:os";
61
82
  import { randomUUID } from "node:crypto";
62
83
  import { runMain } from "../lib/fatal.mjs";
63
84
  import { invokedDirectly } from "../lib/invoked-directly.mjs";
85
+ import { readRegularJson } from "../lib/regular-file.mjs";
86
+ import { messages, outputLanguage } from "../lib/user-facing.mjs";
87
+ import {
88
+ ARTIFACTS_SUBDIR,
89
+ listRuns,
90
+ logsRoot,
91
+ resolveRunFile,
92
+ runDirCandidates,
93
+ } from "./_run-paths.mjs";
94
+ import { costUsd } from "./_cost.mjs";
95
+ import { configRefusal } from "./_autopilot-config.mjs";
96
+ import {
97
+ OUTCOME,
98
+ classifyOutcome,
99
+ isBlocked,
100
+ isParked,
101
+ isTerminal,
102
+ rateLimited,
103
+ } from "./_autopilot-outcomes.mjs";
104
+ import { prRequestsDir, runRequestsDir, unattendedLogsRoot } from "../lib/pr-request-location.mjs";
105
+ import { prefsHosts, profileGaps, projectSettingsGaps } from "../lib/unattended-profile-check.mjs";
106
+ import { unattendedSettingsPath } from "../lib/unattended-settings-location.mjs";
107
+ import { SUMMARY_SUFFIX, publish, reportAfterPublish, reportDigest } from "./autopilot-publish.mjs";
108
+ import { updateState } from "./gate-ledger.mjs";
109
+ import {
110
+ agentsInFlight,
111
+ buildDigest,
112
+ claimExclusive,
113
+ dailyDue,
114
+ digestText,
115
+ gcReport,
116
+ holdSleep,
117
+ localDate,
118
+ markDaily,
119
+ parallelCap,
120
+ pruneDaily,
121
+ sweepInhibitor,
122
+ } from "./_autopilot-ops.mjs";
64
123
 
65
124
  const ROOT = process.env.MA_AUTOPILOT_ROOT || join(homedir(), ".claude", "autopilot");
66
125
  // Siblings resolve from THIS file's own directory, not from ~/.claude/scripts.
@@ -69,7 +128,40 @@ const ROOT = process.env.MA_AUTOPILOT_ROOT || join(homedir(), ".claude", "autopi
69
128
  // host's path means a Codex-only or Copilot-only tick finds nothing and does
70
129
  // nothing, with no error anywhere. smoke-autopilot-hosts.sh covers this.
71
130
  const SCRIPTS = process.env.MA_AP_SCRIPTS || import.meta.dirname;
72
- const CLAUDE_BIN = process.env.MA_AP_CLAUDE_BIN || "claude";
131
+
132
+ function isExecutable(p) {
133
+ try {
134
+ accessSync(p, fsConstants.X_OK);
135
+ return statSync(p).isFile();
136
+ } catch {
137
+ return false;
138
+ }
139
+ }
140
+
141
+ /**
142
+ * The `claude` this tick launches.
143
+ *
144
+ * launchd does not read a login shell, so the bare word resolves against the
145
+ * plist's short PATH - and the native installer puts `claude` in ~/.local/bin,
146
+ * which no default PATH carries. Order: an explicit override, then the absolute
147
+ * path autopilot-on captured into the plist, then PATH plus ~/.local/bin.
148
+ *
149
+ * @returns {string}
150
+ */
151
+ export function resolveClaudeBin(env = process.env, home = homedir()) {
152
+ if (env.MA_AP_CLAUDE_BIN) return env.MA_AP_CLAUDE_BIN;
153
+ const installed = env.MA_AP_CLAUDE_INSTALLED_BIN;
154
+ if (installed && isAbsolute(installed) && isExecutable(installed)) return installed;
155
+ const dirs = [...String(env.PATH || "").split(delimiter), join(home, ".local", "bin")];
156
+ for (const d of dirs) {
157
+ if (!d) continue;
158
+ const p = join(d, "claude");
159
+ if (isExecutable(p)) return p;
160
+ }
161
+ return "claude";
162
+ }
163
+
164
+ const CLAUDE_BIN = resolveClaudeBin();
73
165
  const DRY = process.argv.includes("--dry-run");
74
166
  const POLL_MS = Number(process.env.MA_AP_POLL_MS || 15000);
75
167
  // A run that has not finished in this long is not going to, and holding the
@@ -87,20 +179,124 @@ const LIVE_STATUSES = new Set(["busy", "running", "waiting"]);
87
179
  // months it is the thing that fills the disk, and a full disk stops the runs it
88
180
  // was logging.
89
181
  const LOG_MAX_BYTES = Number(process.env.MA_AP_LOG_MAX_BYTES || 5 * 1024 * 1024);
182
+ // ticks.jsonl is telemetry: past this size only the newest TICKS_KEEP lines stay.
183
+ const TICKS_MAX_BYTES = Number(process.env.MA_AP_TICKS_MAX_BYTES || 2 * 1024 * 1024);
184
+ const TICKS_KEEP = Number(process.env.MA_AP_TICKS_KEEP || 5000);
185
+ // attempted.jsonl is the queue's memory, so past this size it is compacted
186
+ // rather than cut: rows younger than ATTEMPTED_KEEP_SEC stay whole, and older
187
+ // rows stay when intake still needs them (compactAttempted).
188
+ const ATTEMPTED_MAX_BYTES = Number(process.env.MA_AP_ATTEMPTED_MAX_BYTES || 4 * 1024 * 1024);
189
+ const ATTEMPTED_KEEP_SEC = 30 * 86400;
190
+ // The breaker and the rate-limit backoff read the attempts of this trailing
191
+ // window. A row count would let the `blocked-*` row every refused tick writes
192
+ // push the failures it is counting out of view.
193
+ const ATTEMPT_WINDOW_SEC = Number(process.env.MA_AP_ATTEMPT_WINDOW_SEC || 7 * 86400);
90
194
  // Consecutive failed attempts before the runner stops taking NEW work. The
91
195
  // failure this guards against is a machine-level one - an expired token, a full
92
196
  // disk, a `claude` that no longer launches - where every item fails the same
93
197
  // way and the queue is consumed one worthless run at a time. Zero disables it.
94
198
  const BREAKER_LIMIT = Number(process.env.MA_AP_BREAKER_LIMIT || 3);
199
+ // Publish refusals that describe the machine rather than the run's work: the
200
+ // remote could not be resolved, the push or the PR was refused by the host.
201
+ export const ENVIRONMENTAL_PUBLISH_GATES = new Set(["remote", "push", "pr-create"]);
95
202
  // How long an open breaker waits before letting ONE attempt through. Long
96
203
  // enough that a broken machine is not burning the queue (default 30 min, so at
97
204
  // a 15-minute tick it is every other tick at most), short enough that a machine
98
205
  // fixed at 3am is working again by morning without anyone touching it.
99
206
  const BREAKER_COOLDOWN_SEC = Number(process.env.MA_AP_BREAKER_COOLDOWN_SEC || 1800);
100
- // Outcomes that mean the attempt produced nothing. `needs-input` is NOT here:
101
- // a run parked on a question did work and is waiting for a person, which is the
102
- // system behaving correctly.
103
- const FAILED_OUTCOMES = new Set(["launch-failed", "timed-out", "no-state", "failed", "unknown"]);
207
+ // A rate limit belongs to the account, not the item, so after one the runner
208
+ // takes NO new item until the wait is over, doubling per consecutive limit up to
209
+ // the ceiling. Five hours is the longest usage window a limit resets on.
210
+ const RATE_LIMIT_BACKOFF_SEC = Number(process.env.MA_AP_RATE_LIMIT_BACKOFF_SEC ?? 1800);
211
+ const RATE_LIMIT_MAX_SEC = Number(process.env.MA_AP_RATE_LIMIT_MAX_SEC || 5 * 60 * 60);
212
+ // `--bg` returns once the session exists, so a launch that takes this long is
213
+ // not starting.
214
+ const LAUNCH_TIMEOUT_MS = 2 * 60 * 1000;
215
+ // How long an unparseable runner.pid is taken for a claim still being written.
216
+ const UNPARSEABLE_CLAIM_GRACE_MS = 60 * 1000;
217
+ // The deterministic half of a research round. The agent gathers; this decides.
218
+ const RESEARCH_GATE =
219
+ process.env.MA_AP_RESEARCH_GATE || join(import.meta.dirname, "research-gate.mjs");
220
+ // Tools a research pass may not call. Tracker text is untrusted and often
221
+ // confidential: the research_* tools send their query to an external service,
222
+ // and so does a web search built from that text. The local context_* index
223
+ // stays available.
224
+ export const RESEARCH_DENIED_TOOLS = Object.freeze([
225
+ "mcp__multi-agent-toolkit__research_ask",
226
+ "mcp__multi-agent-toolkit__research_search",
227
+ "WebSearch",
228
+ ]);
229
+ // How a session ends that lets a research pass write into the run's state: it
230
+ // left the list, it reported it was done, or it is idle waiting for a person.
231
+ const ENDED = new Set(["gone", "finished", "needs-input"]);
232
+ // The toolkit settings install --unattended writes into settings.json, read
233
+ // from the same profile so the two cannot drift. The URL policy is forced, not
234
+ // defaulted: an unattended child is never less strict than the profile.
235
+ const PROFILE_SCHEMA = (() => {
236
+ try {
237
+ return JSON.parse(
238
+ readFileSync(new URL("../schemas/unattended-profile.json", import.meta.url), "utf-8"),
239
+ );
240
+ } catch {
241
+ return null;
242
+ }
243
+ })();
244
+ const PROFILE_ENV = Object.fromEntries((PROFILE_SCHEMA?.env || []).map(([k, v]) => [k, v]));
245
+ const TOOLKIT_URL_POLICY = PROFILE_ENV.MCP_TOOLKIT_URL_POLICY || "strict";
246
+ const TOOLKIT_INDEX_DENY =
247
+ PROFILE_ENV.MCP_TOOLKIT_INDEX_DENY || "~/.ssh,~/.aws,~/.gnupg,~/Library/Keychains";
248
+ // The three PreToolUse registrations install/claude.mjs writes for agent-guard.sh,
249
+ // and the tool names each one has to cover. The web entry is checked against
250
+ // WebFetch and the toolkit tools that reach a host: a web tool, agent_run_steps
251
+ // (which dispatches the web tools) and both open_url tools.
252
+ const GUARD_MATCHERS = [
253
+ ["Bash", ["Bash"]],
254
+ ["Edit|Write|NotebookEdit", ["Edit", "Write", "NotebookEdit"]],
255
+ [
256
+ "WebFetch|mcp__multi-agent-toolkit__.*",
257
+ [
258
+ "WebFetch",
259
+ "mcp__multi-agent-toolkit__web_goto",
260
+ "mcp__multi-agent-toolkit__agent_run_steps",
261
+ "mcp__multi-agent-toolkit__ios_open_url",
262
+ "mcp__multi-agent-toolkit__android_open_url",
263
+ ],
264
+ ],
265
+ ];
266
+ // The hook command install/claude.mjs writes. Only this command, with $HOME
267
+ // written literally or expanded, counts as the guard: a command that merely
268
+ // mentions the script path can run anything and still look registered.
269
+ const GUARD_COMMAND = "bash $HOME/.claude/scripts/agent-guard.sh";
270
+ const MANAGED_SETTINGS = "/Library/Application Support/ClaudeCode/managed-settings.json";
271
+ // A phase record without a model is priced at the most expensive tier: an
272
+ // over-estimate stops the spend ceiling early, an under-estimate stops it late.
273
+ const CONSERVATIVE_PRICE = "fable";
274
+ // What the queue says when the arming script answers nothing at all. Read by a
275
+ // person in the menu bar and in the terminal, so it renders in the same
276
+ // preference every other line does.
277
+ const ARM_UNREADABLE = {
278
+ en: "the arming check returned no verdict - no new item is taken until it does",
279
+ tr: "hazırlık denetimi bir karar döndürmedi - dönene kadar yeni madde alınmaz",
280
+ };
281
+
282
+ const RATE_LIMITED_MSG = {
283
+ en: (min) => `rate limited - no new item is taken for ${min} min`,
284
+ tr: (min) => `hız sınırına takıldı - ${min} dk boyunca yeni madde alınmaz`,
285
+ };
286
+
287
+ /**
288
+ * The phase contract, for the running entry's "phase N/M Name". Beside the
289
+ * scripts in every install, as runs-index.mjs reads it; a host tree without it
290
+ * shows the number alone rather than failing the tick.
291
+ */
292
+ const PHASES = (() => {
293
+ try {
294
+ return JSON.parse(readFileSync(new URL("../schemas/phases.json", import.meta.url), "utf-8"))
295
+ .phases;
296
+ } catch {
297
+ return [];
298
+ }
299
+ })();
104
300
 
105
301
  const log = (m) => process.stdout.write(`autopilot-runner: ${m}\n`);
106
302
 
@@ -116,24 +312,31 @@ function run(cmd, args, opts = {}) {
116
312
  }
117
313
  }
118
314
 
315
+ /**
316
+ * Every JSON file the runner reads goes through here. Run state and trackers
317
+ * are written by the run, so anything but a bounded regular file - a FIFO would
318
+ * block this thread, and with it every run it supervises - reads as absent.
319
+ */
119
320
  function readJson(p, fallback) {
120
- if (!existsSync(p)) return fallback;
121
- try {
122
- return JSON.parse(readFileSync(p, "utf-8"));
123
- } catch {
124
- return fallback;
125
- }
321
+ return readRegularJson(p, fallback);
126
322
  }
127
323
 
128
324
  function ensureRoot() {
129
325
  mkdirSync(ROOT, { recursive: true, mode: 0o700 });
130
326
  }
131
327
 
328
+ /**
329
+ * Replace one of the runner's files whole: a temp file beside it, then a
330
+ * rename, so a reader sees the old document or the new one and never half of
331
+ * either, and a link in its place is replaced rather than written through.
332
+ */
132
333
  function writeState(name, obj) {
133
334
  ensureRoot();
134
335
  const p = join(ROOT, name);
135
- writeFileSync(p, JSON.stringify(obj, null, 2), { mode: 0o600 });
136
- chmodSync(p, 0o600);
336
+ const tmp = `${p}.${process.pid}.tmp`;
337
+ writeFileSync(tmp, JSON.stringify(obj, null, 2), { mode: 0o600 });
338
+ chmodSync(tmp, 0o600);
339
+ renameSync(tmp, p);
137
340
  }
138
341
 
139
342
  function record(entry) {
@@ -213,13 +416,40 @@ function tick(entry) {
213
416
  export function consecutiveFailures(lines) {
214
417
  let count = 0;
215
418
  let last = null;
419
+ // One run can write two rows - `timed-out` when supervision gave up, then
420
+ // the outcome a later tick retired it with. It is one attempt, counted once.
421
+ const counted = new Set();
216
422
  for (let i = lines.length - 1; i >= 0; i--) {
217
423
  const row = lines[i];
218
424
  if (!row || typeof row.outcome !== "string") continue;
219
425
  // A blocked item never ran: arming refused it, which says nothing about
220
- // whether a run would have worked.
221
- if (row.outcome.startsWith("blocked-")) continue;
222
- if (FAILED_OUTCOMES.has(row.outcome)) {
426
+ // whether a run would have worked. A rate limit has its own backoff, and
427
+ // counting it here too would stop the machine twice for one cause. `died`
428
+ // is a run this runner lost track of - a reboot, a lid closed - which says
429
+ // nothing about the machine either way, so it neither counts nor passes
430
+ // for the success that would end the chain.
431
+ if (
432
+ isBlocked(row.outcome) ||
433
+ row.outcome === OUTCOME.RATE_LIMITED ||
434
+ row.outcome === OUTCOME.DIED
435
+ ) {
436
+ continue;
437
+ }
438
+ // Every other RETRYABLE outcome is an attempt that produced nothing, and
439
+ // so is a publish the machine refused: no remote, a push or a PR the host
440
+ // would not take. Those park the item, but what they say is that the
441
+ // runner's own credential or network is broken, which every next item
442
+ // would meet at the same step. Other PARKED and TERMINAL outcomes did
443
+ // work, and end the chain.
444
+ if (
445
+ classifyOutcome(row.outcome) === "retryable" ||
446
+ (row.outcome === OUTCOME.VERIFICATION_FAILED &&
447
+ ENVIRONMENTAL_PUBLISH_GATES.has(row.publishGate))
448
+ ) {
449
+ if (typeof row.taskId === "string" && row.taskId) {
450
+ if (counted.has(row.taskId)) continue;
451
+ counted.add(row.taskId);
452
+ }
223
453
  count++;
224
454
  if (!last) last = row.outcome;
225
455
  continue;
@@ -229,24 +459,121 @@ export function consecutiveFailures(lines) {
229
459
  return { count, last };
230
460
  }
231
461
 
232
- function readAttempted() {
233
- const p = join(ROOT, "attempted.jsonl");
234
- if (!existsSync(p)) return [];
462
+ /**
463
+ * When the newest real attempt was recorded. A `blocked-*` row is written on
464
+ * every tick that arming refuses, so letting it move this clock would keep a
465
+ * breaker's cooldown from ever running out.
466
+ */
467
+ export function lastAttemptAt(lines) {
468
+ for (let i = lines.length - 1; i >= 0; i--) {
469
+ const row = lines[i];
470
+ if (!row || typeof row.outcome !== "string" || isBlocked(row.outcome)) continue;
471
+ return Number(row.at || 0);
472
+ }
473
+ return 0;
474
+ }
475
+
476
+ /**
477
+ * How long a rate limit still holds the queue, or null when it does not.
478
+ *
479
+ * @returns {{count:number, remainingSec:number}|null}
480
+ */
481
+ export function rateLimitBackoff(
482
+ lines,
483
+ { baseSec = RATE_LIMIT_BACKOFF_SEC, maxSec = RATE_LIMIT_MAX_SEC, now = Date.now() / 1000 } = {},
484
+ ) {
485
+ let count = 0;
486
+ let lastAt = 0;
487
+ for (let i = lines.length - 1; i >= 0; i--) {
488
+ const row = lines[i];
489
+ if (!row || typeof row.outcome !== "string" || isBlocked(row.outcome)) continue;
490
+ if (row.outcome !== OUTCOME.RATE_LIMITED) break;
491
+ count++;
492
+ if (!lastAt) lastAt = Number(row.at || 0);
493
+ }
494
+ if (!count || !(baseSec > 0)) return null;
495
+ const wait = Math.min(maxSec, baseSec * 2 ** (count - 1));
496
+ const remainingSec = Math.ceil(lastAt + wait - now);
497
+ return remainingSec > 0 ? { count, remainingSec } : null;
498
+ }
499
+
500
+ /** The attempts of the trailing window the breaker and the backoff judge. */
501
+ function readAttempted(now = Date.now() / 1000) {
502
+ const since = now - ATTEMPT_WINDOW_SEC;
503
+ return readAllAttempted().filter((r) => Number(r.at || 0) >= since);
504
+ }
505
+
506
+ /**
507
+ * The rows of attempted.jsonl worth keeping once it outgrows its cap.
508
+ *
509
+ * Rows younger than `keepSec` are kept whole. An older row is kept when intake
510
+ * or the cleanup report still reads it: the newest row of its item (what the
511
+ * item last did), and, for an item that is not finished, every row that is not
512
+ * a `blocked-*` refusal (its attempt count, research rounds and the task ids
513
+ * that tie a parked run to its worktree). What goes is the refusal a blocked
514
+ * tick writes every time, and the superseded rows of finished items.
515
+ *
516
+ * @returns {object[]} the kept rows, in their original order
517
+ */
518
+ export function compactAttempted(
519
+ rows,
520
+ { now = Date.now() / 1000, keepSec = ATTEMPTED_KEEP_SEC } = {},
521
+ ) {
522
+ const since = now - keepSec;
523
+ const newest = new Map();
524
+ rows.forEach((r, i) => {
525
+ const key = `${r.source}:${r.id}`;
526
+ const cur = newest.get(key);
527
+ if (!cur || Number(r.at || 0) >= Number(rows[cur].at || 0)) newest.set(key, i);
528
+ });
529
+ return rows.filter((r, i) => {
530
+ if (Number(r.at || 0) >= since) return true;
531
+ const key = `${r.source}:${r.id}`;
532
+ const last = newest.get(key);
533
+ if (last === i) return true;
534
+ if (isTerminal(rows[last].outcome)) return false;
535
+ return !isBlocked(r.outcome);
536
+ });
537
+ }
538
+
539
+ function replaceWhole(p, text) {
540
+ const tmp = `${p}.${process.pid}.tmp`;
541
+ writeFileSync(tmp, text, { mode: 0o600 });
542
+ chmodSync(tmp, 0o600);
543
+ renameSync(tmp, p);
544
+ }
545
+
546
+ /**
547
+ * Keep ticks.jsonl and attempted.jsonl bounded. Both are opened per append and
548
+ * never held, so replacing them by rename loses no writer.
549
+ */
550
+ function rotateJsonl() {
235
551
  try {
236
- return readFileSync(p, "utf-8")
237
- .split("\n")
238
- .filter(Boolean)
239
- .slice(-50)
240
- .map((l) => {
241
- try {
242
- return JSON.parse(l);
243
- } catch {
244
- return null;
245
- }
246
- })
247
- .filter(Boolean);
552
+ const ticks = join(ROOT, "ticks.jsonl");
553
+ if (existsSync(ticks) && statSync(ticks).size > TICKS_MAX_BYTES) {
554
+ const lines = readFileSync(ticks, "utf-8").split("\n").filter(Boolean);
555
+ replaceWhole(ticks, lines.slice(-TICKS_KEEP).join("\n") + "\n");
556
+ }
248
557
  } catch {
249
- return [];
558
+ // Telemetry never fails a tick.
559
+ }
560
+ try {
561
+ const attempted = join(ROOT, "attempted.jsonl");
562
+ if (existsSync(attempted) && statSync(attempted).size > ATTEMPTED_MAX_BYTES) {
563
+ const rows = readAllAttempted();
564
+ const kept = compactAttempted(rows);
565
+ copyFileSync(attempted, `${attempted}.1`);
566
+ chmodSync(`${attempted}.1`, 0o600);
567
+ replaceWhole(
568
+ attempted,
569
+ kept.map((r) => JSON.stringify(r)).join("\n") + (kept.length ? "\n" : ""),
570
+ );
571
+ log(
572
+ `attempted.jsonl compacted: ${rows.length} rows -> ${kept.length}, full log in attempted.jsonl.1`,
573
+ );
574
+ }
575
+ } catch (e) {
576
+ log(`attempted.jsonl could not be compacted: ${e?.message || e}`);
250
577
  }
251
578
  }
252
579
 
@@ -295,20 +622,32 @@ export function agentRow(sessionId, agentsJson) {
295
622
  }
296
623
  }
297
624
 
625
+ /** A listed session that is still working or waiting. A row with no status is not "ended". */
626
+ function sessionIsLive(row) {
627
+ return Boolean(row) && (!row.status || LIVE_STATUSES.has(row.status));
628
+ }
629
+
298
630
  /**
299
- * Is the recorded runner still running?
631
+ * Is the recorded run still running?
632
+ *
633
+ * The SESSION decides, not the runner's pid. The pid is this runner's own, and
634
+ * it exits whenever supervision stops short of the run's end - a timeout, an
635
+ * unreadable list - while the session it launched carries on. Judging by the
636
+ * pid would retire a live run on the very next tick. The pid still counts for
637
+ * one thing: a live supervisor and a listed session is a peer mid-poll, whatever
638
+ * status the row shows, and it will record the outcome itself.
300
639
  *
301
- * All three conditions, and the boot check is the one that matters after a
302
- * restart: a pid recorded before the current boot is stale by definition and
303
- * needs no probe at all.
640
+ * The boot check comes first: a claim recorded before the current boot is stale
641
+ * by definition and needs no probe at all.
304
642
  */
305
643
  export function holderLiveness(holder, { now = bootTime(), agentsJson } = {}) {
306
644
  if (!holder || !holder.pid) return "dead";
307
645
  if (holder.bootTime && now && holder.bootTime !== now) return "dead";
308
- if (!pidAlive(holder.pid)) return "dead";
309
646
  const row = agentRow(holder.sessionId, agentsJson);
310
647
  if (row === undefined) return "unknown";
311
- return row ? "live" : "dead";
648
+ if (!row) return "dead";
649
+ if (pidAlive(holder.pid)) return "live";
650
+ return sessionIsLive(row) ? "live" : "dead";
312
651
  }
313
652
 
314
653
  export function holderIsLive(holder, opts = {}) {
@@ -319,51 +658,156 @@ export function holderIsLive(holder, opts = {}) {
319
658
  * The run's own state file, which is what actually knows the worktree.
320
659
  *
321
660
  * The runner does not create the worktree - Phase 0 of the run does - and the
322
- * runner's `taskId` (`ap-<epoch>`) is not the run's task id. So the link is made
323
- * from disk: a state file whose `worktreePath` sits under this repo and which
324
- * was written after this claim started. `gc-abandoned.sh` builds the same index
325
- * for the same reason.
326
- *
327
- * Returns null rather than guessing. Every field it fills is one `retire()` can
328
- * act on; without them retire still records the attempt, it just cannot clean
329
- * anything up.
330
- */
331
- export function findRunState(repoPath, startedAtSec, logsRoot) {
332
- const root = logsRoot || join(homedir(), ".claude", "logs", "multi-agent");
333
- if (!repoPath || !existsSync(root)) return null;
661
+ * runner's `taskId` (`ap-<epoch>`) is not the run's task id. The link is the
662
+ * session id: the runner chose it, passed it as `--session-id` and as
663
+ * MULTI_AGENT_SESSION_ID, and Phase 0 records it as `sessionId`. Both layouts
664
+ * are read through _run-paths.mjs, because a run's files may be nested under
665
+ * its project or flat.
666
+ *
667
+ * Only a state with NO `sessionId` falls back to the older heuristic: written
668
+ * after this claim started, with a `worktreePath` under this repo. A state that
669
+ * names a different session is somebody else's run, however well it matches.
670
+ * `matchedBy` says which link was used, and only a session match authorises
671
+ * cleaning anything up.
672
+ *
673
+ * Returns null rather than guessing.
674
+ *
675
+ * @returns {{statePath:string, worktree:string|null, runTaskId:string, matchedBy:"session"|"heuristic"}|null}
676
+ */
677
+ export function findRunState({ sessionId, repoPath, startedAtSec } = {}) {
678
+ if (!sessionId && !repoPath) return null;
334
679
  let best = null;
335
- for (const dir of readdirSync(root, { withFileTypes: true })) {
336
- if (!dir.isDirectory()) continue;
337
- const p = join(root, dir.name, "agent-state.json");
338
- if (!existsSync(p)) continue;
339
- let st;
680
+ for (const run of listRuns()) {
681
+ const p = resolveRunFile(run.taskId, "agent-state.json", run.project);
682
+ if (!p) continue;
683
+ const doc = readJson(p, null);
684
+ if (!doc || typeof doc !== "object") continue;
685
+ const wt = typeof doc.worktreePath === "string" ? doc.worktreePath : null;
686
+ const runTaskId = doc.taskId || run.taskId;
687
+ if (sessionId && doc.sessionId === sessionId) {
688
+ return { statePath: p, worktree: wt, runTaskId, matchedBy: "session" };
689
+ }
690
+ if (doc.sessionId || !repoPath) continue;
691
+ let mtimeMs;
340
692
  try {
341
- st = statSync(p);
693
+ mtimeMs = statSync(p).mtimeMs;
342
694
  } catch {
343
695
  continue;
344
696
  }
345
697
  // A second of slack: the claim is stamped before the child is spawned.
346
- if (st.mtimeMs < (startedAtSec - 1) * 1000) continue;
347
- const doc = readJson(p, null);
348
- const wt = doc && doc.worktreePath;
698
+ if (mtimeMs < (startedAtSec - 1) * 1000) continue;
349
699
  if (!wt || !(wt === repoPath || wt.startsWith(repoPath + "/"))) continue;
350
- if (!best || st.mtimeMs > best.mtimeMs)
351
- best = { statePath: p, worktree: wt, mtimeMs: st.mtimeMs };
700
+ if (!best || mtimeMs > best.mtimeMs) {
701
+ best = { statePath: p, worktree: wt, runTaskId, matchedBy: "heuristic", mtimeMs };
702
+ }
352
703
  }
704
+ if (best) delete best.mtimeMs;
353
705
  return best;
354
706
  }
355
707
 
708
+ function realOr(p) {
709
+ try {
710
+ return realpathSync(p);
711
+ } catch {
712
+ return p;
713
+ }
714
+ }
715
+
716
+ /**
717
+ * May this directory be removed as a run's worktree?
718
+ *
719
+ * Only a LINKED worktree that sits inside the repo it belongs to. Never the
720
+ * repo path itself - a local-workspace run works in the user's own checkout,
721
+ * which may itself be a linked worktree git would happily remove - and never a
722
+ * main worktree, whose git dir is the repository.
723
+ */
724
+ export function removableWorktree(wt, repoPath) {
725
+ if (!wt || !repoPath || !existsSync(wt)) return false;
726
+ const w = realOr(wt);
727
+ const r = realOr(repoPath);
728
+ if (w === r || !w.startsWith(r + "/")) return false;
729
+ const gitPath = (flag) =>
730
+ run("git", ["-C", wt, "rev-parse", "--path-format=absolute", flag]).trim();
731
+ const gitDir = gitPath("--git-dir");
732
+ const commonDir = gitPath("--git-common-dir");
733
+ if (!gitDir || !commonDir) return false;
734
+ return realOr(gitDir) !== realOr(commonDir);
735
+ }
736
+
737
+ /**
738
+ * USD for one run, priced from the token counts its tracker recorded.
739
+ *
740
+ * Null - never 0 - when no tracker or no tokens are recorded: the spend ceiling
741
+ * has to be able to tell "cost nothing" from "cost is not known".
742
+ *
743
+ * @returns {number|null}
744
+ */
745
+ export function runUsd(runTaskId, statePath) {
746
+ const dirs = [];
747
+ if (statePath) dirs.push(dirname(statePath));
748
+ if (runTaskId) dirs.push(...runDirCandidates(runTaskId));
749
+ let trackerPath = null;
750
+ for (const d of dirs) {
751
+ for (const base of [d, join(d, ARTIFACTS_SUBDIR)]) {
752
+ const p = join(base, "tracker-state.json");
753
+ if (existsSync(p)) {
754
+ trackerPath = p;
755
+ break;
756
+ }
757
+ }
758
+ if (trackerPath) break;
759
+ }
760
+ if (!trackerPath) return null;
761
+ const tracker = readJson(trackerPath, null);
762
+ const prices = readJson(join(import.meta.dirname, "cost-table.json"), null)?.prices;
763
+ if (!tracker || !prices) return null;
764
+ let usd = 0;
765
+ let tokens = 0;
766
+ for (const ph of Object.values(tracker.phases || {})) {
767
+ const tin = Number(ph?.tokens_in || 0);
768
+ const tout = Number(ph?.tokens_out || 0);
769
+ const tcached = Number(ph?.tokens_cached || 0);
770
+ if (!(tin + tout + tcached > 0)) continue;
771
+ tokens += tin + tout + tcached;
772
+ usd += costUsd(prices[ph?.model] || prices[CONSERVATIVE_PRICE], tin, tout, tcached) ?? 0;
773
+ }
774
+ return tokens > 0 ? Math.round(usd * 10000) / 10000 : null;
775
+ }
776
+
356
777
  /**
357
778
  * Retire a run whose session is gone. The worktree is removed BY NAME, never by
358
779
  * sweeping: gc-worktrees.sh deliberately skips registered worktrees and always
359
780
  * will, and the runner is the only thing that knows which one it created.
781
+ *
782
+ * The run's own state speaks first. A run that opened its PR or finished after
783
+ * supervision stopped watching is recorded as exactly that, and a run parked on
784
+ * a question is left entirely alone - its state, its worktree and its
785
+ * attempt count. Only a run that ended without either is `died`.
360
786
  */
361
- function retire(holder) {
787
+ async function retire(holder) {
362
788
  const item = holder.item || {};
363
789
  const wt = holder.worktree;
364
790
  let stashed = null;
791
+ const pub = await publishStep(holder.sessionId, holder.statePath, item);
792
+ const state = holder.statePath ? readJson(holder.statePath, null) : null;
793
+ const prUrl = prFromState(state);
794
+ const ended = pub.outcome || outcomeFor({ reason: "gone" }, state, prUrl);
795
+ const usd = runUsd(holder.runTaskId, holder.statePath);
796
+
797
+ if (holder.parked || isParked(ended)) {
798
+ record({
799
+ source: item.source,
800
+ id: item.id,
801
+ taskId: holder.taskId,
802
+ outcome: ended,
803
+ usd,
804
+ ...publishGateOf(pub),
805
+ });
806
+ log(`${item.id}: parked (${ended}) - its session, state and worktree are left alone`);
807
+ return;
808
+ }
365
809
 
366
- if (wt && existsSync(wt)) {
810
+ if (wt && existsSync(wt) && holder.createdWorktree) {
367
811
  const dirty = run("git", ["-C", wt, "status", "--porcelain"]).trim();
368
812
  if (dirty) {
369
813
  // A day of edits is worth more than 750 MB. The stash MESSAGE carries the
@@ -376,30 +820,44 @@ function retire(holder) {
376
820
  // the stash commit's third parent, not in its tree. One honest reference
377
821
  // beats two, and this is the contract gc-abandoned.sh already keeps.
378
822
  const label = `autopilot/abandoned/${holder.taskId || item.id || "unknown"}`;
379
- run("git", ["-C", wt, "stash", "push", "-u", "-m", label]);
380
- stashed = label;
381
- log(`kept ${wt}: uncommitted work stashed as "${label}" (git stash list)`);
382
- } else if (holder.createdWorktree) {
823
+ if (stashWorktree(wt, label)) {
824
+ stashed = label;
825
+ log(`kept ${wt}: uncommitted work stashed as "${label}" (git stash list)`);
826
+ } else {
827
+ log(`kept ${wt}: uncommitted work left in place - git stash push failed`);
828
+ }
829
+ } else if (!removableWorktree(wt, holder.repoPath)) {
830
+ log(`kept ${wt}: not a linked worktree inside ${holder.repoPath || "its repo"}`);
831
+ } else if (agentRow(holder.sessionId)) {
383
832
  // A clean worktree is only safe to remove once nothing is writing to it.
384
833
  // Supervision can end while the run does not - a timeout, or an agent
385
834
  // list that could not be read - and `worktree remove --force` under a
386
835
  // live session is the one mistake here with nothing to undo it.
387
- if (agentRow(holder.sessionId)) {
388
- log(`kept ${wt}: session ${holder.sessionId} is still listed`);
389
- } else {
390
- run("git", ["-C", wt, "worktree", "remove", "--force", wt]);
391
- run("git", ["-C", holder.repoPath || wt, "worktree", "prune"]);
392
- log(`removed ${wt}`);
393
- }
836
+ log(`kept ${wt}: session ${holder.sessionId} is still listed`);
837
+ } else {
838
+ run("git", ["-C", wt, "worktree", "remove", "--force", wt]);
839
+ run("git", ["-C", holder.repoPath, "worktree", "prune"]);
840
+ log(`removed ${wt}`);
394
841
  }
395
842
  }
396
843
 
397
- if (holder.statePath && existsSync(holder.statePath)) {
398
- const st = readJson(holder.statePath, null);
399
- if (st) {
400
- st.status = "abandoned";
401
- st.abandonedBy = "autopilot-runner";
402
- writeFileSync(holder.statePath, JSON.stringify(st, null, 2));
844
+ if (isTerminal(ended)) {
845
+ record({ source: item.source, id: item.id, taskId: holder.taskId, outcome: ended, prUrl, usd });
846
+ return;
847
+ }
848
+
849
+ // Only a state this runner's session wrote is marked. A heuristic match may be
850
+ // an attended run in the same repo, and its state is not ours to rewrite. The
851
+ // write takes the state lock and bumps `rev`, as every other writer does.
852
+ if (holder.matchedBy === "session" && state) {
853
+ try {
854
+ await updateState(holder.statePath, (s) => ({
855
+ ...s,
856
+ status: "abandoned",
857
+ abandonedBy: "autopilot-runner",
858
+ }));
859
+ } catch (e) {
860
+ log(`${item.id}: could not mark the state abandoned: ${e?.message || e}`);
403
861
  }
404
862
  }
405
863
 
@@ -407,9 +865,75 @@ function retire(holder) {
407
865
  source: item.source,
408
866
  id: item.id,
409
867
  taskId: holder.taskId,
410
- outcome: "died",
868
+ outcome: OUTCOME.DIED,
411
869
  stashedTo: stashed,
870
+ usd,
871
+ });
872
+ }
873
+
874
+ /**
875
+ * Stash a worktree's uncommitted work, untracked files included. True only when
876
+ * git says it stashed: a stash reference is recorded for a person to find, and
877
+ * one that does not exist sends them looking for work that is still in the
878
+ * worktree, or gone.
879
+ */
880
+ export function stashWorktree(wt, label) {
881
+ const r = spawnSync("git", ["-C", wt, "stash", "push", "-u", "-m", label], {
882
+ encoding: "utf-8",
883
+ stdio: ["ignore", "pipe", "pipe"],
884
+ timeout: 120000,
885
+ });
886
+ if (r.status !== 0) return false;
887
+ const top = spawnSync("git", ["-C", wt, "stash", "list", "-n", "1", "--format=%s"], {
888
+ encoding: "utf-8",
889
+ stdio: ["ignore", "pipe", "pipe"],
890
+ });
891
+ return top.status === 0 && String(top.stdout).includes(label);
892
+ }
893
+
894
+ /**
895
+ * Settle the runs parked in `running`.
896
+ *
897
+ * A parked run holds its repo - its session may resume and write to the
898
+ * worktree - but not the supervisor, which returned when the run stopped to
899
+ * ask. Nothing about it is retired: once its session ends, whatever its state
900
+ * then records is written as a new outcome if it is one (a PR opened after the
901
+ * answer, a failure after it), the slot is released, and the state file and
902
+ * worktree are never touched.
903
+ */
904
+ async function reconcileParked() {
905
+ const q = readJson(join(ROOT, "queue.json"), null);
906
+ const parked = (q?.running || []).filter((r) => r.state === "parked");
907
+ if (!parked.length) return;
908
+ const settled = new Set();
909
+ for (const e of parked) {
910
+ const row = agentRow(e.sessionId);
911
+ if (row === undefined || sessionIsLive(row)) continue;
912
+ const pub = DRY ? { outcome: null } : await publishStep(e.sessionId, e.statePath, e);
913
+ const st = e.statePath ? readJson(e.statePath, null) : null;
914
+ const prUrl = prFromState(st);
915
+ const outcome = pub.outcome || outcomeFor({ reason: "gone" }, st, prUrl);
916
+ if (!isParked(outcome) && !DRY) {
917
+ record({
918
+ source: e.source,
919
+ id: e.id,
920
+ taskId: e.taskId,
921
+ outcome,
922
+ prUrl,
923
+ usd: runUsd(e.runTaskId, e.statePath),
924
+ ...publishGateOf(pub),
925
+ });
926
+ }
927
+ log(`${e.id}: parked session ended (${outcome}) - slot released`);
928
+ settled.add(e.id);
929
+ }
930
+ if (!settled.size || DRY) return;
931
+ const after = readJson(join(ROOT, "queue.json"), q);
932
+ writeState("queue.json", {
933
+ ...after,
934
+ running: (after.running || []).filter((r) => !(r.state === "parked" && settled.has(r.id))),
412
935
  });
936
+ refreshStatus();
413
937
  }
414
938
 
415
939
  function refreshStatus() {
@@ -417,15 +941,48 @@ function refreshStatus() {
417
941
  if (existsSync(s)) run("bash", [s, "--write"]);
418
942
  }
419
943
 
944
+ /**
945
+ * Does `bin/menubar` predate the source it was compiled from?
946
+ *
947
+ * The indicator is built from source on demand rather than shipped as a binary,
948
+ * so the compiled artifact is the one thing in the install that an update does
949
+ * not replace. Comparing the two timestamps is what ties it to the release: the
950
+ * installer lays the `.swift` down fresh, and this answers yes exactly once
951
+ * after it does.
952
+ */
953
+ export function indicatorStale(src, out) {
954
+ if (!existsSync(src)) return false;
955
+ if (!existsSync(out)) return true;
956
+ return statSync(src).mtimeMs > statSync(out).mtimeMs;
957
+ }
958
+
959
+ /** Compile the indicator when it is stale. Best effort: no swiftc, no indicator. */
960
+ function ensureIndicator() {
961
+ const src = join(SCRIPTS, "autopilot-menubar.swift");
962
+ const out = join(ROOT, "bin", "menubar");
963
+ if (!indicatorStale(src, out)) return;
964
+ mkdirSync(dirname(out), { recursive: true, mode: 0o700 });
965
+ const built = spawnSync("swiftc", ["-O", src, "-o", out], { stdio: "ignore" });
966
+ // No swiftc means no indicator and nothing else changes - a supported shape,
967
+ // and one that would otherwise put a line in the log on every tick for the
968
+ // life of the machine. A compiler that IS here and refused is worth saying.
969
+ if (built.error?.code === "ENOENT") return;
970
+ if (built.error || built.status !== 0) {
971
+ log("swiftc refused the menu bar indicator - the terminal status is unaffected");
972
+ return;
973
+ }
974
+ chmodSync(out, 0o700);
975
+ log("rebuilt the menu bar indicator from its source");
976
+ }
977
+
420
978
  /**
421
979
  * Put the slot back and drop the claim. One place, so every exit agrees.
422
980
  *
423
981
  * It re-reads queue.json rather than trusting a copy, because intake runs
424
- * between the read and here. The recovery path calls this too: `retire()` used
425
- * to record the outcome and delete the claim while leaving the item in
426
- * `running`, and nothing else on this machine ever removes one - intake copies
427
- * the list forward verbatim. One crashed runner was enough to wedge the queue
428
- * at "1/1 slots in use" on every tick from then on.
982
+ * between the read and here. The recovery path calls this too: nothing else on
983
+ * this machine removes an entry from `running` - intake copies the list forward
984
+ * verbatim - so a retirement that dropped the claim without this would leave
985
+ * the queue at "1/1 slots in use" on every tick.
429
986
  */
430
987
  function releaseSlot(id, pidPath, queue = { queued: [], running: [] }) {
431
988
  const after = readJson(join(ROOT, "queue.json"), queue);
@@ -437,6 +994,157 @@ function releaseSlot(id, pidPath, queue = { queued: [], running: [] }) {
437
994
  refreshStatus();
438
995
  }
439
996
 
997
+ /**
998
+ * The publish step for one ended session: verify the PR request it left, push,
999
+ * open the draft PR, then post what the config opts into
1000
+ * (autopilot-publish.mjs). `outcome` is null when there was nothing to publish,
1001
+ * and the run's own state then decides the attempt as before.
1002
+ *
1003
+ * @returns {Promise<{outcome: string|null, prUrl?: string}>}
1004
+ */
1005
+ async function publishStep(sessionId, statePath, item) {
1006
+ if (!sessionId || !statePath || !item?.repo) return { outcome: null };
1007
+ const config = readJson(join(ROOT, "config.json"), null);
1008
+ let pub;
1009
+ try {
1010
+ pub = await publish({ sessionId, statePath, item, config });
1011
+ } catch (err) {
1012
+ log(`${item.id}: publish step failed: ${err?.message || err}`);
1013
+ return { outcome: null };
1014
+ }
1015
+ if (!pub.outcome) return pub;
1016
+ if (pub.outcome === OUTCOME.VERIFICATION_FAILED) {
1017
+ log(`${item.id}: publish refused at ${pub.gate}: ${pub.reason}`);
1018
+ return pub;
1019
+ }
1020
+ log(`${item.id}: published (${pub.outcome})${pub.prUrl ? ` ${pub.prUrl}` : ""}`);
1021
+ if (pub.outcome === OUTCOME.PR_OPENED) {
1022
+ const reports = reportAfterPublish({
1023
+ prUrl: pub.prUrl,
1024
+ request: pub.request,
1025
+ item,
1026
+ state: readJson(statePath, null),
1027
+ config,
1028
+ scripts: outwardScripts(),
1029
+ });
1030
+ for (const r of reports) {
1031
+ log(`${item.id}: report ${r.channel} ${r.status}${r.reason ? ` (${r.reason})` : ""}`);
1032
+ }
1033
+ }
1034
+ return pub;
1035
+ }
1036
+
1037
+ /** The refusing publish gate, for the attempt row the breaker reads. */
1038
+ function publishGateOf(pub) {
1039
+ return pub?.outcome === OUTCOME.VERIFICATION_FAILED && pub.gate ? { publishGate: pub.gate } : {};
1040
+ }
1041
+
1042
+ /** Overrides for the outward scripts, so a test never reaches the real tracker. */
1043
+ function outwardScripts() {
1044
+ return process.env.MA_AP_JIRA_PUBLISH ? { jiraPublish: process.env.MA_AP_JIRA_PUBLISH } : {};
1045
+ }
1046
+
1047
+ /** Every row of attempted.jsonl; the digest needs a full day, not the recent tail. */
1048
+ function readAllAttempted() {
1049
+ const p = join(ROOT, "attempted.jsonl");
1050
+ if (!existsSync(p)) return [];
1051
+ try {
1052
+ return readFileSync(p, "utf-8")
1053
+ .split("\n")
1054
+ .filter(Boolean)
1055
+ .map((l) => {
1056
+ try {
1057
+ return JSON.parse(l);
1058
+ } catch {
1059
+ return null;
1060
+ }
1061
+ })
1062
+ .filter(Boolean);
1063
+ } catch {
1064
+ return [];
1065
+ }
1066
+ }
1067
+
1068
+ /**
1069
+ * The two jobs that run at most once per local day, on the first tick of it.
1070
+ * Each is stamped in daily.json whether or not it succeeded, so a job that
1071
+ * fails is retried tomorrow rather than on every tick; the failure is logged
1072
+ * and, for the digest, recorded in its file. Neither ever fails the tick.
1073
+ */
1074
+ function dailyJobs(config) {
1075
+ const date = localDate();
1076
+ const queue = readJson(join(ROOT, "queue.json"), { queued: [], running: [] });
1077
+ if (dailyDue(ROOT, "gc", date)) {
1078
+ try {
1079
+ gcReport({
1080
+ root: ROOT,
1081
+ gcScript: join(SCRIPTS, "gc-abandoned.sh"),
1082
+ logsRoot: logsRoot(),
1083
+ config,
1084
+ queue,
1085
+ date,
1086
+ log,
1087
+ });
1088
+ } catch (e) {
1089
+ log(`cleanup report failed: ${e?.message || e}`);
1090
+ }
1091
+ markDaily(ROOT, "gc", date);
1092
+ }
1093
+ if (config?.digest?.enabled === true && dailyDue(ROOT, "digest", date)) {
1094
+ try {
1095
+ const digest = buildDigest({
1096
+ lines: readAllAttempted(),
1097
+ queue,
1098
+ ceilingUsd: Number(config.costCeilingUsd ?? 25),
1099
+ });
1100
+ const text = digestText(digest);
1101
+ const sent = reportDigest({ text, config, scripts: outwardScripts() });
1102
+ for (const r of sent) {
1103
+ log(`digest: ${r.channel} ${r.status}${r.reason ? ` (${r.reason})` : ""}`);
1104
+ }
1105
+ writeFileSync(
1106
+ join(ROOT, `digest-${date}.json`),
1107
+ JSON.stringify({ date, ...digest, sent }, null, 2) + "\n",
1108
+ { mode: 0o600 },
1109
+ );
1110
+ writeFileSync(join(ROOT, `digest-${date}.md`), text, { mode: 0o600 });
1111
+ pruneDaily(ROOT, "digest-");
1112
+ log(
1113
+ `digest: ${digest.items} item(s), ${digest.prsOpened.length} PR(s) - digest-${date}.json`,
1114
+ );
1115
+ } catch (e) {
1116
+ log(`digest failed: ${e?.message || e}`);
1117
+ }
1118
+ markDaily(ROOT, "digest", date);
1119
+ }
1120
+ }
1121
+
1122
+ /**
1123
+ * The cleanup ledger: one line per worktree this runner can prove it created,
1124
+ * read by the daily cleanup report and by nothing that removes on its own.
1125
+ */
1126
+ function recordWorktree(claim) {
1127
+ if (!claim.worktree) return;
1128
+ try {
1129
+ ensureRoot();
1130
+ const p = join(ROOT, "worktrees.jsonl");
1131
+ appendFileSync(
1132
+ p,
1133
+ JSON.stringify({
1134
+ at: Math.floor(Date.now() / 1000),
1135
+ sessionId: claim.sessionId,
1136
+ taskId: claim.taskId,
1137
+ runTaskId: claim.runTaskId || null,
1138
+ repoPath: claim.repoPath || null,
1139
+ worktree: realOr(claim.worktree),
1140
+ }) + "\n",
1141
+ { mode: 0o600 },
1142
+ );
1143
+ } catch {
1144
+ // The ledger only feeds a report; a failed append costs one report line.
1145
+ }
1146
+ }
1147
+
440
1148
  /** The PR a run opened, wherever its state file recorded it. */
441
1149
  export function prFromState(state) {
442
1150
  const pr = state && state.pr;
@@ -447,22 +1155,521 @@ export function prFromState(state) {
447
1155
  }
448
1156
 
449
1157
  /**
450
- * What to call this attempt.
1158
+ * What to call this attempt. Every word it can return is classified in
1159
+ * _autopilot-outcomes.mjs, and test/autopilot-runner.test.mjs fails on one
1160
+ * that is not.
451
1161
  *
452
1162
  * The run's own state is the authority; the supervisor only says why waiting
453
- * stopped. "finished with no PR" is deliberately not "failed": a run can stop
1163
+ * stopped. The order is what the words mean:
1164
+ *
1165
+ * - a PR first. Phase 5 holds for input AFTER the PR is open, so a session
1166
+ * parked there has delivered, and calling it anything else re-queues work
1167
+ * that is already in review. A branch the runner pushed to a host with no
1168
+ * draft PR (`state.publish`) is delivered the same way.
1169
+ * - a state waiting on a person (`waitingFor`, `awaiting_input`) next, before
1170
+ * the supervisor's reasons: a run parked on a maturity question that
1171
+ * outlived the ceiling is still parked, not timed out.
1172
+ * - a rate limit before the state's own status, because a limited run's state
1173
+ * says only where it stopped.
1174
+ * - `verificationFailed`, written by gate-ledger.mjs `park` when an
1175
+ * unattended run's own gate rejected its work, before a plain `failed`:
1176
+ * retrying repeats the same work into the same rejection.
1177
+ *
1178
+ * "finished with no PR" is deliberately not "failed": a run can stop
454
1179
  * legitimately without opening one, and calling that a failure would put a
455
1180
  * healthy item into the retry path.
456
1181
  */
457
1182
  export function outcomeFor(supervised, state, prUrl) {
458
- if (supervised.reason === "timeout") return "timed-out";
459
- if (supervised.reason === "unknown") return "unknown";
460
- if (supervised.reason === "needs-input") return "needs-input";
461
- if (prUrl) return "pr-opened";
1183
+ if (prUrl) return OUTCOME.PR_OPENED;
1184
+ if (state?.publish?.outcome === OUTCOME.PUSHED_AWAITING_PR) return OUTCOME.PUSHED_AWAITING_PR;
1185
+ if (state && (state.waitingFor || state.status === "awaiting_input")) {
1186
+ return OUTCOME.AWAITING_ANSWER;
1187
+ }
1188
+ if (supervised.reason === "timeout") return OUTCOME.TIMED_OUT;
1189
+ if (supervised.reason === "unknown") return OUTCOME.UNKNOWN;
1190
+ if (supervised.reason === "needs-input") return OUTCOME.NEEDS_INPUT;
1191
+ if (rateLimited({ row: supervised.lastRow, state })) return OUTCOME.RATE_LIMITED;
1192
+ if (state && state.verificationFailed) return OUTCOME.VERIFICATION_FAILED;
462
1193
  const status = state && state.status;
463
- if (status === "complete" || status === "completed") return "completed-no-pr";
464
- if (status === "failed") return "failed";
465
- return state ? `stopped-${status || "unknown"}` : "no-state";
1194
+ if (status === "complete" || status === "completed") return OUTCOME.COMPLETED_NO_PR;
1195
+ if (status === "failed") return OUTCOME.FAILED;
1196
+ return state ? `stopped-${status || "unknown"}` : OUTCOME.NO_STATE;
1197
+ }
1198
+
1199
+ /**
1200
+ * The `claude` arguments for one item. The shape - background, a session id
1201
+ * this runner chose, no permission prompts, the unattended permission profile
1202
+ * as an extra settings file, one unattended `/multi-agent` item - is pending a
1203
+ * live launch trial; everything that launches or describes a launch derives
1204
+ * from here so a correction is one edit.
1205
+ */
1206
+ export function launchArgv(item, sessionId, settingsFile = unattendedSettingsPath()) {
1207
+ return [
1208
+ "--bg",
1209
+ "--session-id",
1210
+ sessionId,
1211
+ "--permission-prompts",
1212
+ "none",
1213
+ "--settings",
1214
+ settingsFile,
1215
+ "-p",
1216
+ `/multi-agent ${item.ref || item.id} autopilot`,
1217
+ ];
1218
+ }
1219
+
1220
+ /**
1221
+ * The child's environment. MULTI_AGENT_UNATTENDED=1 is the operator contract
1222
+ * (unattended-contract.md) that nobody is watching; MULTI_AGENT_SESSION_ID is
1223
+ * what Phase 0 records as `sessionId`, which is how findRunState finds this
1224
+ * run and no other. LOGS_ROOT puts the run's state under the unattended run
1225
+ * directory, the one place outside the worktree the sandbox lets it write
1226
+ * (an operator's own LOGS_ROOT is kept). MA_RUN_BASE_SHA is the checkout's
1227
+ * HEAD when the item was taken: agent-guard trusts a repository file as a
1228
+ * program only when it is unchanged from that commit, so a file the run
1229
+ * committed itself never counts.
1230
+ *
1231
+ * @param {string} sessionId
1232
+ * @param {Record<string,string|undefined>} [base]
1233
+ * @param {{baseSha?: string|null}} [opts]
1234
+ */
1235
+ export function launchEnv(sessionId, base = process.env, { baseSha = null } = {}) {
1236
+ const deny = [
1237
+ ...TOOLKIT_INDEX_DENY.split(","),
1238
+ ...String(base.MCP_TOOLKIT_INDEX_DENY || "").split(","),
1239
+ ]
1240
+ .map((s) => s.trim())
1241
+ .filter(Boolean);
1242
+ const { MA_RUN_BASE_SHA: _stale, ...rest } = base;
1243
+ return {
1244
+ ...rest,
1245
+ MULTI_AGENT_UNATTENDED: "1",
1246
+ MULTI_AGENT_SESSION_ID: sessionId,
1247
+ MCP_TOOLKIT_URL_POLICY: TOOLKIT_URL_POLICY,
1248
+ MCP_TOOLKIT_INDEX_DENY: [...new Set(deny)].join(","),
1249
+ LOGS_ROOT: base.LOGS_ROOT || unattendedLogsRoot(base),
1250
+ ...(baseSha ? { MA_RUN_BASE_SHA: baseSha } : {}),
1251
+ };
1252
+ }
1253
+
1254
+ /**
1255
+ * Ready the runner's side of a launch: both requests directories exist at
1256
+ * 0700 (the run's, under the unattended run directory, and the runner's own
1257
+ * for verdicts), and nothing is left under this session's name from before.
1258
+ * A request the publish step finds must be one this session wrote.
1259
+ */
1260
+ export function prepareLaunch(sessionId, env = process.env) {
1261
+ for (const dir of [prRequestsDir(env), runRequestsDir(env)]) {
1262
+ mkdirSync(dir, { recursive: true, mode: 0o700 });
1263
+ chmodSync(dir, 0o700);
1264
+ for (const suffix of [".json", ".verdict.json", SUMMARY_SUFFIX]) {
1265
+ rmSync(join(dir, `${sessionId}${suffix}`), { force: true });
1266
+ }
1267
+ }
1268
+ }
1269
+
1270
+ /**
1271
+ * Does the profile file still hold the OS sandbox the unattended contract
1272
+ * rests on, and does the checkout leave it alone? Judged by content, before
1273
+ * every launch (the first run, each research pass, each resume), so a run
1274
+ * that rewrote the file, or a repository whose own settings widen the child,
1275
+ * never reaches the next launch.
1276
+ *
1277
+ * @param {{home?: string, cwd?: string|null}} [opts]
1278
+ * @returns {{ok: boolean, reason?: string}}
1279
+ */
1280
+ export function sandboxPrecondition({ home = homedir(), cwd = null } = {}) {
1281
+ const path = unattendedSettingsPath(home);
1282
+ if (!PROFILE_SCHEMA)
1283
+ return {
1284
+ ok: false,
1285
+ reason: "schemas/unattended-profile.json is not installed beside the runner",
1286
+ };
1287
+ if (!existsSync(path))
1288
+ return {
1289
+ ok: false,
1290
+ reason: `the unattended profile ${path} is not installed (install --unattended writes it)`,
1291
+ };
1292
+ let doc;
1293
+ try {
1294
+ doc = JSON.parse(readFileSync(path, "utf-8"));
1295
+ } catch (err) {
1296
+ return { ok: false, reason: `the unattended profile ${path} does not parse: ${err.message}` };
1297
+ }
1298
+ const gaps = profileGaps(doc, PROFILE_SCHEMA, { extraHosts: prefsHosts(home) });
1299
+ const project = cwd
1300
+ ? [join(cwd, ".claude", "settings.json"), join(cwd, ".claude", "settings.local.json")]
1301
+ .map((f) => readJson(f, null))
1302
+ .filter((d) => d && typeof d === "object")
1303
+ : [];
1304
+ const all = [...gaps.sandbox, ...gaps.permissions, ...projectSettingsGaps(project)];
1305
+ if (all.length)
1306
+ return {
1307
+ ok: false,
1308
+ reason: `the unattended sandbox is not in force: ${all.join("; ")} (install --unattended restores the profile)`,
1309
+ };
1310
+ return { ok: true };
1311
+ }
1312
+
1313
+ /** The checkout's HEAD commit, or null when it has none. */
1314
+ function headSha(cwd) {
1315
+ const r = spawnSync("git", ["-C", cwd, "rev-parse", "--verify", "--quiet", "HEAD^{commit}"], {
1316
+ encoding: "utf-8",
1317
+ stdio: ["ignore", "pipe", "ignore"],
1318
+ timeout: 10000,
1319
+ });
1320
+ return r.status === 0 ? r.stdout.trim() || null : null;
1321
+ }
1322
+
1323
+ /** Does a hook matcher (a regex over the tool name, empty or `*` for all) cover this tool? */
1324
+ function matcherCovers(matcher, tool) {
1325
+ const m = typeof matcher === "string" ? matcher : "";
1326
+ if (m === "" || m === "*") return true;
1327
+ try {
1328
+ return new RegExp(`^(?:${m})$`).test(tool);
1329
+ } catch {
1330
+ return false;
1331
+ }
1332
+ }
1333
+
1334
+ /** The guard script, when the hook command is exactly the installer's and the script exists. */
1335
+ function guardScriptOf(command, home) {
1336
+ const script = join(home, ".claude", "scripts", "agent-guard.sh");
1337
+ const forms = [GUARD_COMMAND, GUARD_COMMAND.replace("$HOME", "${HOME}"), `bash ${script}`];
1338
+ return forms.includes(String(command || "").trim()) && existsSync(script) ? script : null;
1339
+ }
1340
+
1341
+ /**
1342
+ * Is agent-guard.sh registered where the child will load it?
1343
+ *
1344
+ * Read the way install/claude.mjs writes it: a PreToolUse entry whose matcher
1345
+ * covers the tool and whose command runs agent-guard.sh. All three matchers are
1346
+ * required: the Bash entry is the one that stops a push, the
1347
+ * Edit|Write|NotebookEdit entry the one that stops a write into a protected
1348
+ * path, and the WebFetch|mcp__multi-agent-toolkit__web_.* entry the one that
1349
+ * holds WebFetch and every toolkit tool to the network host policy. A
1350
+ * matcher is judged as Claude Code judges it, a regex over the tool name, so
1351
+ * the legacy `Bash(git push:*)` does not count. The hook command has to be the
1352
+ * one the installer writes and its script has to exist: a hook whose script is
1353
+ * missing exits non-zero without blocking anything, and a command that only
1354
+ * mentions the script path proves nothing. `disableAllHooks` in any file turns
1355
+ * every hook off.
1356
+ *
1357
+ * Files read for the registration: the user settings (CLAUDE_CONFIG_DIR, else
1358
+ * ~/.claude) and macOS managed settings. The checkout's project settings are
1359
+ * read only for `disableAllHooks`: the repo under work is not where the
1360
+ * guard's registration may come from.
1361
+ *
1362
+ * @returns {{ok: boolean, missing: string[], reason?: string}}
1363
+ */
1364
+ export function guardRegistration({
1365
+ home = homedir(),
1366
+ env = process.env,
1367
+ cwd = null,
1368
+ managedPath = MANAGED_SETTINGS,
1369
+ } = {}) {
1370
+ const userDir = env.CLAUDE_CONFIG_DIR || join(home, ".claude");
1371
+ const files = [join(userDir, "settings.json"), join(userDir, "settings.local.json"), managedPath];
1372
+ const project = cwd
1373
+ ? [join(cwd, ".claude", "settings.json"), join(cwd, ".claude", "settings.local.json")]
1374
+ : [];
1375
+ const read = (list) =>
1376
+ list.map((f) => readJson(f, null)).filter((d) => d && typeof d === "object");
1377
+ const docs = read(files);
1378
+ const off = [...docs, ...read(project)].find((d) => d.disableAllHooks === true);
1379
+ const required = GUARD_MATCHERS.map(([label]) => label);
1380
+ if (off) return { ok: false, missing: required, reason: "disableAllHooks is set" };
1381
+ const managed = readJson(managedPath, null);
1382
+ const sources = managed?.allowManagedHooksOnly === true ? [managed] : docs;
1383
+ const entries = sources.flatMap((d) =>
1384
+ Array.isArray(d?.hooks?.PreToolUse) ? d.hooks.PreToolUse : [],
1385
+ );
1386
+ const missing = GUARD_MATCHERS.filter(([, tools]) =>
1387
+ tools.some(
1388
+ (tool) =>
1389
+ !entries.some(
1390
+ (e) =>
1391
+ e &&
1392
+ matcherCovers(e.matcher, tool) &&
1393
+ Array.isArray(e.hooks) &&
1394
+ e.hooks.some((h) => h && h.type === "command" && guardScriptOf(h.command, home)),
1395
+ ),
1396
+ ),
1397
+ ).map(([label]) => label);
1398
+ if (missing.length)
1399
+ return {
1400
+ ok: false,
1401
+ missing,
1402
+ reason: `agent-guard.sh is not registered on ${missing.join(", ")}`,
1403
+ };
1404
+ // The child is launched with `--settings <profile>`; without the file there is
1405
+ // no dontAsk default and no narrow allow list, and the run stops at its first
1406
+ // tool call.
1407
+ const profile = unattendedSettingsPath(home);
1408
+ if (!existsSync(profile))
1409
+ return {
1410
+ ok: false,
1411
+ missing: [],
1412
+ reason: `the unattended permission profile ${profile} is not installed (install --unattended writes it)`,
1413
+ };
1414
+ return { ok: true, missing: [] };
1415
+ }
1416
+
1417
+ /**
1418
+ * Should this run's parked state get a research pass before a person is asked?
1419
+ *
1420
+ * Only the two parks research can act on: the maturity step (`waitingFor:
1421
+ * maturity`, or a Phase 0 halt that still carries blockers) and the Phase 1
1422
+ * open-questions gate. Everything else a run waits on - the channels menu, a
1423
+ * failed verification, a user test - is a person's call and stays one. The
1424
+ * ceiling counts every round this item has had, across runs, because a fresh
1425
+ * run of the same item is a fresh state file and would otherwise start at zero.
1426
+ *
1427
+ * @returns {{route: "research"|"none", kind?: "maturity"|"open-questions", reason?: string}}
1428
+ */
1429
+ export function researchRoute(state, prUrl, { roundsUsed = 0, maxAskRounds = 2 } = {}) {
1430
+ if (prUrl || !state || typeof state !== "object") return { route: "none" };
1431
+ let kind = null;
1432
+ const pq = state.pendingQuestion;
1433
+ if (state.waitingFor === "maturity") kind = "maturity";
1434
+ else if (state.waitingFor === "question" && pq?.stepId === "phase-1/open-questions") {
1435
+ kind = "open-questions";
1436
+ } else if (
1437
+ !state.waitingFor &&
1438
+ !state.verificationFailed &&
1439
+ (state.currentPhase ?? 0) === 0 &&
1440
+ state.status !== "complete" &&
1441
+ Array.isArray(state.maturity?.blockers) &&
1442
+ state.maturity.blockers.length > 0
1443
+ ) {
1444
+ kind = "maturity";
1445
+ }
1446
+ if (!kind) return { route: "none" };
1447
+ if (roundsUsed >= maxAskRounds) {
1448
+ return {
1449
+ route: "none",
1450
+ kind,
1451
+ reason: `${roundsUsed} research round(s) used, maxAskRounds ${maxAskRounds} - left for a person`,
1452
+ };
1453
+ }
1454
+ return { route: "research", kind };
1455
+ }
1456
+
1457
+ /** Research rounds already spent on one item, summed from attempted.jsonl. */
1458
+ export function researchRoundsUsed(lines, item) {
1459
+ let n = 0;
1460
+ for (const row of lines || []) {
1461
+ if (!row || row.source !== item?.source || row.id !== item?.id) continue;
1462
+ const r = Number(row.researchRounds || 0);
1463
+ if (Number.isFinite(r) && r > 0) n += r;
1464
+ }
1465
+ return n;
1466
+ }
1467
+
1468
+ /** The research session's arguments: the same launch shape, one command, fewer tools. */
1469
+ export function researchArgv(item, sessionId, statePath, settingsFile = unattendedSettingsPath()) {
1470
+ const state = /\s/.test(statePath) ? JSON.stringify(statePath) : statePath;
1471
+ return [
1472
+ "--bg",
1473
+ "--session-id",
1474
+ sessionId,
1475
+ "--permission-prompts",
1476
+ "none",
1477
+ "--settings",
1478
+ settingsFile,
1479
+ "--disallowedTools",
1480
+ RESEARCH_DENIED_TOOLS.join(","),
1481
+ "-p",
1482
+ `/multi-agent:research ${item.ref || item.id} --autonomous --state ${state}`,
1483
+ ];
1484
+ }
1485
+
1486
+ /** Continue a parked run after research closed its gaps. */
1487
+ export function resumeArgv(runTaskId, sessionId, settingsFile = unattendedSettingsPath()) {
1488
+ return [
1489
+ "--bg",
1490
+ "--session-id",
1491
+ sessionId,
1492
+ "--permission-prompts",
1493
+ "none",
1494
+ "--settings",
1495
+ settingsFile,
1496
+ "-p",
1497
+ `/multi-agent:resume ${runTaskId} autopilot`,
1498
+ ];
1499
+ }
1500
+
1501
+ /** Point the claim and the running entry at the session now in flight. */
1502
+ function rebindSession(claim, item, sessionId) {
1503
+ claim.sessionId = sessionId;
1504
+ writeState("runner.pid", claim);
1505
+ const q = readJson(join(ROOT, "queue.json"), null);
1506
+ if (!q) return;
1507
+ writeState("queue.json", {
1508
+ ...q,
1509
+ running: (q.running || []).map((r) => (r.id === item.id ? { ...r, sessionId } : r)),
1510
+ });
1511
+ }
1512
+
1513
+ function launch(argv, cwd, sessionId, baseSha) {
1514
+ const sandbox = sandboxPrecondition({ cwd });
1515
+ if (!sandbox.ok)
1516
+ return { status: 1, stderr: `${OUTCOME.BLOCKED_SANDBOX_UNAVAILABLE}: ${sandbox.reason}` };
1517
+ prepareLaunch(sessionId);
1518
+ return spawnSync(CLAUDE_BIN, argv, {
1519
+ cwd,
1520
+ env: launchEnv(sessionId, process.env, { baseSha }),
1521
+ encoding: "utf-8",
1522
+ stdio: ["ignore", "pipe", "pipe"],
1523
+ timeout: LAUNCH_TIMEOUT_MS,
1524
+ });
1525
+ }
1526
+
1527
+ /** Run research-gate.mjs on the parked state; its verdict, or null when it gave none. */
1528
+ function researchVerdict(statePath, sessionId) {
1529
+ if (!existsSync(RESEARCH_GATE)) return null;
1530
+ const r = spawnSync(process.execPath, [RESEARCH_GATE, "--state", statePath, "--json"], {
1531
+ env: launchEnv(sessionId),
1532
+ encoding: "utf-8",
1533
+ stdio: ["ignore", "pipe", "pipe"],
1534
+ timeout: LAUNCH_TIMEOUT_MS,
1535
+ });
1536
+ try {
1537
+ const out = JSON.parse(r.stdout || "");
1538
+ return typeof out.decision === "string" ? out : null;
1539
+ } catch {
1540
+ return null;
1541
+ }
1542
+ }
1543
+
1544
+ /**
1545
+ * Research rounds for a run that parked, until one does not proceed.
1546
+ *
1547
+ * Each round is two sessions at most: the research pass, then - only when
1548
+ * research-gate.mjs says the re-run check passed - the parked run resumed. A
1549
+ * resumed run that parks again on the other kind of gap gets another round
1550
+ * while the ceiling allows. Anything short of a clean proceed stops the loop
1551
+ * with the run's state as it is, which the caller records: a parked state is
1552
+ * awaiting-answer, exactly as it would have been without research.
1553
+ */
1554
+ function researchRounds(claim, item, run0, { roundsUsed, maxAskRounds }) {
1555
+ let { supervised, state, prUrl } = run0;
1556
+ let rounds = 0;
1557
+ for (;;) {
1558
+ if (!claim.statePath || !ENDED.has(supervised.reason)) break;
1559
+ const route = researchRoute(state, prUrl, { roundsUsed: roundsUsed + rounds, maxAskRounds });
1560
+ if (route.route !== "research") {
1561
+ if (route.reason) log(`${item.id}: ${route.reason}`);
1562
+ break;
1563
+ }
1564
+ const guard = guardRegistration({ cwd: item.localPath });
1565
+ if (!guard.ok) {
1566
+ log(`${item.id}: ${guard.reason} - no research session is launched, left for a person`);
1567
+ break;
1568
+ }
1569
+ rounds++;
1570
+ const rsid = randomUUID();
1571
+ rebindSession(claim, item, rsid);
1572
+ log(`${item.id}: parked on ${route.kind} - research round ${roundsUsed + rounds}`);
1573
+ const rchild = launch(
1574
+ researchArgv(item, rsid, claim.statePath),
1575
+ item.localPath,
1576
+ rsid,
1577
+ claim.baseSha,
1578
+ );
1579
+ if (rchild.status !== 0) {
1580
+ log(`${item.id}: research launch failed - left for a person`);
1581
+ break;
1582
+ }
1583
+ const rs = supervise(rsid, claim, item);
1584
+ if (!ENDED.has(rs.reason)) {
1585
+ supervised = rs;
1586
+ break;
1587
+ }
1588
+ const verdict = researchVerdict(claim.statePath, rsid);
1589
+ state = readJson(claim.statePath, state);
1590
+ if (!verdict || verdict.decision !== "proceed") {
1591
+ log(
1592
+ `${item.id}: research ${verdict ? verdict.decision : "gave no verdict"} - left for a person`,
1593
+ );
1594
+ break;
1595
+ }
1596
+ const runTaskId = claim.runTaskId || state?.taskId;
1597
+ if (!runTaskId) break;
1598
+ const guardAgain = guardRegistration({ cwd: item.localPath });
1599
+ if (!guardAgain.ok) {
1600
+ log(`${item.id}: ${guardAgain.reason} - the run is not resumed, left for a person`);
1601
+ break;
1602
+ }
1603
+ const dsid = randomUUID();
1604
+ rebindSession(claim, item, dsid);
1605
+ log(
1606
+ `${item.id}: research closed ${verdict.closed?.join(", ") || "the gaps"} - resuming ${runTaskId}`,
1607
+ );
1608
+ const dchild = launch(resumeArgv(runTaskId, dsid), item.localPath, dsid, claim.baseSha);
1609
+ if (dchild.status !== 0) {
1610
+ log(`${item.id}: resume launch failed - the run stays parked`);
1611
+ break;
1612
+ }
1613
+ supervised = supervise(dsid, claim, item);
1614
+ state = readJson(claim.statePath, state);
1615
+ prUrl = prFromState(state);
1616
+ }
1617
+ return { supervised, state, prUrl, rounds };
1618
+ }
1619
+
1620
+ /** Link the claim to the run's state; upgrade a heuristic match once the session's own appears. */
1621
+ function trackRunState(claim) {
1622
+ if (claim.matchedBy === "session") return;
1623
+ const found = findRunState({
1624
+ sessionId: claim.sessionId,
1625
+ repoPath: claim.repoPath,
1626
+ startedAtSec: claim.startedAt,
1627
+ });
1628
+ if (!found || found.statePath === claim.statePath) return;
1629
+ if (claim.statePath && found.matchedBy !== "session") return;
1630
+ claim.statePath = found.statePath;
1631
+ claim.worktree = found.worktree;
1632
+ claim.runTaskId = found.runTaskId;
1633
+ claim.matchedBy = found.matchedBy;
1634
+ // Phase 0 opened the worktree, not this runner - but this runner is the only
1635
+ // thing that will be around to remove it, and only when the state is
1636
+ // provably this session's.
1637
+ claim.createdWorktree = found.matchedBy === "session";
1638
+ writeState("runner.pid", claim);
1639
+ if (claim.createdWorktree) recordWorktree(claim);
1640
+ log(`tracking ${found.worktree || found.statePath} (matched by ${found.matchedBy})`);
1641
+ }
1642
+
1643
+ /** Mirror the run's current phase onto its `running` entry, which the status line renders. */
1644
+ function trackProgress(claim, id) {
1645
+ if (!claim.statePath) return;
1646
+ const st = readJson(claim.statePath, null);
1647
+ const phase = st && Number.isInteger(st.currentPhase) ? st.currentPhase : null;
1648
+ if (phase === null || phase === claim.phase) return;
1649
+ claim.phase = phase;
1650
+ const q = readJson(join(ROOT, "queue.json"), null);
1651
+ if (!q) return;
1652
+ const last = PHASES.length ? PHASES[PHASES.length - 1].id : null;
1653
+ const phaseName = PHASES.find((p) => p.id === phase)?.name ?? null;
1654
+ writeState("queue.json", {
1655
+ ...q,
1656
+ running: (q.running || []).map((r) =>
1657
+ r.id === id ? { ...r, phase, phaseName, phaseTotal: last } : r,
1658
+ ),
1659
+ });
1660
+ refreshStatus();
1661
+ }
1662
+
1663
+ /**
1664
+ * Has this session's own state recorded that the run stopped - to ask a
1665
+ * person, or for good? Only a state matched by session id is believed.
1666
+ */
1667
+ export function stateSaysStopped(claim) {
1668
+ if (!claim || claim.matchedBy !== "session" || !claim.statePath) return false;
1669
+ const st = readJson(claim.statePath, null);
1670
+ if (!st || typeof st !== "object") return false;
1671
+ if (st.waitingFor || st.status === "awaiting_input" || st.verificationFailed) return true;
1672
+ return ["complete", "completed", "failed"].includes(st.status);
466
1673
  }
467
1674
 
468
1675
  /**
@@ -474,45 +1681,43 @@ export function outcomeFor(supervised, state, prUrl) {
474
1681
  * is reached. An unreadable list ends it too, but says "unknown" rather than
475
1682
  * inventing an outcome.
476
1683
  *
1684
+ * A session is listed a moment AFTER `--bg` returns, so until it has been seen
1685
+ * once an absent row and an empty or unreadable list are the same thing - not
1686
+ * yet - and both get the registration grace.
1687
+ *
477
1688
  * The claim is enriched on the way: the first poll that finds the run's state
478
1689
  * file writes `worktree` and `statePath` into runner.pid, which is what lets a
479
- * LATER tick's retire() actually clean up. Those fields were read by retire()
480
- * from the day it was written and never once set by anything.
1690
+ * LATER tick's retire() clean up.
481
1691
  */
482
1692
  function supervise(sessionId, claim, item) {
483
1693
  const startedMs = Date.now();
484
1694
  const deadline = startedMs + MAX_RUN_MS;
485
1695
  let sawLive = false;
486
1696
  let sawBusy = false;
1697
+ let lastRow = null;
487
1698
  let reason;
488
1699
 
489
1700
  for (;;) {
490
1701
  const row = agentRow(sessionId);
1702
+ const inGrace = !sawLive && Date.now() - startedMs <= REGISTER_GRACE_MS;
491
1703
 
492
1704
  if (row === undefined) {
493
- reason = "unknown";
494
- break;
495
- }
496
- if (row) {
497
- sawLive = true;
498
- if (!claim.statePath) {
499
- const found = findRunState(claim.repoPath, claim.startedAt);
500
- if (found) {
501
- claim.statePath = found.statePath;
502
- claim.worktree = found.worktree;
503
- // Phase 0 opened it, not this runner - but this runner is the only
504
- // thing that will be around to remove it.
505
- claim.createdWorktree = true;
506
- writeState("runner.pid", claim);
507
- log(`tracking ${found.worktree}`);
508
- }
1705
+ if (!inGrace) {
1706
+ reason = "unknown";
1707
+ break;
509
1708
  }
1709
+ } else if (row) {
1710
+ sawLive = true;
1711
+ lastRow = row;
1712
+ trackRunState(claim);
1713
+ trackProgress(claim, item.id);
510
1714
  if (row.status === "busy" || row.status === "running") sawBusy = true;
511
1715
  if (row.status === "waiting") {
512
1716
  // Only terminal once the run has actually been working. A session can
513
1717
  // read as "waiting" in the second after launch, and ending supervision
514
- // there would abandon every run at birth.
515
- if (sawBusy) {
1718
+ // there would abandon every run at birth. A run that worked between two
1719
+ // polls is not caught by that, so the run's own state is asked too.
1720
+ if (sawBusy || stateSaysStopped(claim)) {
516
1721
  reason = "needs-input";
517
1722
  break;
518
1723
  }
@@ -520,7 +1725,7 @@ function supervise(sessionId, claim, item) {
520
1725
  reason = "finished";
521
1726
  break;
522
1727
  }
523
- } else if (sawLive || Date.now() - startedMs > REGISTER_GRACE_MS) {
1728
+ } else if (!inGrace) {
524
1729
  reason = "gone";
525
1730
  break;
526
1731
  }
@@ -532,28 +1737,101 @@ function supervise(sessionId, claim, item) {
532
1737
  sleepSync(POLL_MS);
533
1738
  }
534
1739
 
535
- if (!claim.statePath) {
536
- const found = findRunState(claim.repoPath, claim.startedAt);
537
- if (found) {
538
- claim.statePath = found.statePath;
539
- claim.worktree = found.worktree;
540
- claim.createdWorktree = true;
541
- }
542
- }
1740
+ trackRunState(claim);
543
1741
  const waitedSec = Math.round((Date.now() - startedMs) / 1000);
544
1742
  log(`${item.id}: supervision ended (${reason}) after ${waitedSec}s`);
545
- return { reason, waitedSec };
1743
+ return { reason, waitedSec, lastRow };
1744
+ }
1745
+
1746
+ /**
1747
+ * Hand a run that stopped to ask a person over from the claim to `running`.
1748
+ *
1749
+ * The claim is dropped because this supervisor is done: keeping it in
1750
+ * runner.pid would hold every other repo for as long as nobody answers. The
1751
+ * entry keeps what reconcileParked needs to settle it later, and keeps the
1752
+ * repo occupied, since the session may resume and write to its worktree.
1753
+ */
1754
+ function park(claim, item, outcome, pidPath) {
1755
+ const q = readJson(join(ROOT, "queue.json"), { queued: [], running: [] });
1756
+ writeState("queue.json", {
1757
+ ...q,
1758
+ running: (q.running || []).map((r) =>
1759
+ r.id === item.id
1760
+ ? {
1761
+ ...r,
1762
+ state: "parked",
1763
+ outcome,
1764
+ parkedAt: Math.floor(Date.now() / 1000),
1765
+ source: item.source,
1766
+ sessionId: claim.sessionId,
1767
+ taskId: claim.taskId,
1768
+ statePath: claim.statePath || null,
1769
+ runTaskId: claim.runTaskId || null,
1770
+ worktree: claim.worktree || null,
1771
+ }
1772
+ : r,
1773
+ ),
1774
+ });
1775
+ rmSync(pidPath, { force: true });
1776
+ refreshStatus();
1777
+ }
1778
+
1779
+ /**
1780
+ * The first queued item this tick may take: its repo is free and its checkout
1781
+ * exists. An item whose checkout is missing is recorded as `blocked-no-path`
1782
+ * once - not on every tick - and stepped over, so one moved directory does not
1783
+ * hold the whole queue.
1784
+ */
1785
+ function nextItem(queue, busyRepos, attempts) {
1786
+ for (const i of queue.queued || []) {
1787
+ if (busyRepos.has(i.repo)) continue;
1788
+ if (i.localPath && existsSync(i.localPath)) return i;
1789
+ const prev = attempts.findLast((a) => a.source === i.source && a.id === i.id);
1790
+ log(`${i.id}: no checkout at ${i.localPath || "(none configured)"} - not launched`);
1791
+ if (!DRY && prev?.outcome !== OUTCOME.BLOCKED_NO_PATH) {
1792
+ record({ source: i.source, id: i.id, outcome: OUTCOME.BLOCKED_NO_PATH });
1793
+ }
1794
+ }
1795
+ return null;
546
1796
  }
547
1797
 
548
- function main() {
1798
+ async function main() {
549
1799
  const pidPath = join(ROOT, "runner.pid");
1800
+ // An unattended run writes its state under the unattended run directory;
1801
+ // every lookup below (findRunState, the cleanup report, publish) reads it
1802
+ // there. An operator's own LOGS_ROOT is kept.
1803
+ if (!process.env.LOGS_ROOT) process.env.LOGS_ROOT = unattendedLogsRoot();
550
1804
 
551
1805
  // ---- 1. RECOVER --------------------------------------------------------
552
1806
  const holder = readJson(pidPath, null);
1807
+ // A claim file that does not parse names no runner and no session. Left in
1808
+ // place it would refuse every exclusive claim below, forever - but a young
1809
+ // one may be a claim another tick is writing this moment, so it is removed
1810
+ // only once it is older than that could take.
1811
+ if (!holder && existsSync(pidPath)) {
1812
+ let ageMs = Infinity;
1813
+ try {
1814
+ ageMs = Date.now() - statSync(pidPath).mtimeMs;
1815
+ } catch {
1816
+ // Gone between the check and the stat: nothing to remove.
1817
+ }
1818
+ if (ageMs < UNPARSEABLE_CLAIM_GRACE_MS) {
1819
+ log("runner.pid does not parse yet - another tick may be writing it; nothing to do");
1820
+ return 0;
1821
+ }
1822
+ if (!DRY) {
1823
+ log("runner.pid does not parse - removing it");
1824
+ rmSync(pidPath, { force: true });
1825
+ }
1826
+ }
553
1827
  if (holder) {
554
1828
  const liveness = holderLiveness(holder);
555
1829
  if (liveness === "live") {
556
- log(`another runner is live (pid ${holder.pid}) - nothing to do`);
1830
+ if (pidAlive(holder.pid)) {
1831
+ log(`another runner is live (pid ${holder.pid}) - nothing to do`);
1832
+ } else {
1833
+ log(`session ${holder.sessionId} is still running - it is retired once it ends`);
1834
+ }
557
1835
  return 0;
558
1836
  }
559
1837
  if (liveness === "unknown") {
@@ -565,7 +1843,7 @@ function main() {
565
1843
  }
566
1844
  log(`previous runner is gone (pid ${holder.pid}) - retiring its item`);
567
1845
  if (!DRY) {
568
- retire(holder);
1846
+ await retire(holder);
569
1847
  releaseSlot(holder.item && holder.item.id, pidPath);
570
1848
  }
571
1849
  }
@@ -575,8 +1853,31 @@ function main() {
575
1853
  log("not configured on this machine - nothing to do");
576
1854
  return 0;
577
1855
  }
1856
+ // The numbers in config.json are ceilings, and a value that is not a number
1857
+ // compares false with everything. Nothing is taken or published until the
1858
+ // config is corrected, and the queue says which field is wrong.
1859
+ const refusal = configRefusal(config);
1860
+ if (refusal) {
1861
+ log(`${refusal} - nothing is taken until it is corrected`);
1862
+ if (!DRY) {
1863
+ writeState("queue.json", {
1864
+ ...readJson(join(ROOT, "queue.json"), { queued: [], running: [] }),
1865
+ blockedReason: refusal,
1866
+ });
1867
+ refreshStatus();
1868
+ }
1869
+ tick({ action: "config-invalid", reason: refusal });
1870
+ return 0;
1871
+ }
578
1872
 
579
1873
  rotateLog();
1874
+ if (!DRY) rotateJsonl();
1875
+ if (!DRY) ensureIndicator();
1876
+ await reconcileParked();
1877
+ if (!DRY) {
1878
+ sweepInhibitor(ROOT, log);
1879
+ dailyJobs(config);
1880
+ }
580
1881
 
581
1882
  // ---- 1b. BREAKER -------------------------------------------------------
582
1883
  // Three failures in a row is not three unlucky items, it is one broken
@@ -600,7 +1901,7 @@ function main() {
600
1901
  // not, that probe fails, becomes the new most recent attempt, and the
601
1902
  // breaker closes again for another cooldown. One wasted run per cooldown is
602
1903
  // the price of not needing a human to notice.
603
- const lastAt = attempts.length ? Number(attempts[attempts.length - 1].at || 0) : 0;
1904
+ const lastAt = lastAttemptAt(attempts);
604
1905
  const sinceSec = lastAt ? Math.floor(Date.now() / 1000) - lastAt : Infinity;
605
1906
  const probeDue = sinceSec >= BREAKER_COOLDOWN_SEC;
606
1907
  if (!probeDue) {
@@ -630,25 +1931,73 @@ function main() {
630
1931
  tick({ action: "breaker-probe", failures: breaker.count, lastOutcome: breaker.last });
631
1932
  }
632
1933
 
1934
+ // ---- 1c. RATE LIMIT ----------------------------------------------------
1935
+ // The limit is the account's, so the next item would hit it too. Waiting it
1936
+ // out costs nothing; a queue that keeps launching into it records one wasted
1937
+ // attempt per item until every item is past its ceiling.
1938
+ const limited = rateLimitBackoff(attempts);
1939
+ if (limited) {
1940
+ const waitMin = Math.ceil(limited.remainingSec / 60);
1941
+ const reason = messages(RATE_LIMITED_MSG, outputLanguage())(waitMin);
1942
+ log(reason);
1943
+ if (!DRY) {
1944
+ writeState("queue.json", {
1945
+ ...readJson(join(ROOT, "queue.json"), { queued: [], running: [] }),
1946
+ blockedReason: reason,
1947
+ });
1948
+ refreshStatus();
1949
+ }
1950
+ tick({ action: "rate-limit-backoff", limits: limited.count, retryInSec: limited.remainingSec });
1951
+ return 0;
1952
+ }
1953
+
633
1954
  // ---- 2. INTAKE ---------------------------------------------------------
634
1955
  const intake = join(SCRIPTS, "autopilot-intake.mjs");
635
1956
  if (existsSync(intake)) run(process.execPath, [intake]);
636
1957
  const queue = readJson(join(ROOT, "queue.json"), { queued: [], running: [] });
637
1958
 
638
1959
  const running = queue.running || [];
1960
+ // A parked run holds its repo but not a slot: it is waiting for a person,
1961
+ // not working, and counting it would let one unanswered question stop the
1962
+ // queue for as long as nobody answers.
1963
+ const active = running.filter((r) => r.state !== "parked");
639
1964
  const slots = Number(config.slots ?? 1);
640
- if (running.length >= slots) {
641
- log(`${running.length}/${slots} slots in use - nothing to take`);
642
- tick({ action: "slots-full", running: running.length, slots });
1965
+ if (active.length >= slots) {
1966
+ log(`${active.length}/${slots} slots in use - nothing to take`);
1967
+ tick({ action: "slots-full", running: active.length, slots });
643
1968
  return 0;
644
1969
  }
645
1970
 
1971
+ // ---- 2b. PARALLEL CAP ---------------------------------------------------
1972
+ // A ceiling, never a raise: unset keeps exactly the behaviour above. When set,
1973
+ // it also counts parked runs whose session is working again after a person
1974
+ // answered, which `slots` deliberately does not. An agent list that cannot be
1975
+ // read is not zero: under a cap, nothing is taken until it can be.
1976
+ const cap = parallelCap(config);
1977
+ if (cap !== null) {
1978
+ const hasParked = running.some((r) => r.state === "parked");
1979
+ const inFlight = agentsInFlight(
1980
+ running,
1981
+ hasParked ? run(CLAUDE_BIN, ["agents", "--json"]) : "[]",
1982
+ );
1983
+ if (inFlight === undefined) {
1984
+ log("cannot read the agent list - no item is taken under maxParallelAgents");
1985
+ tick({ action: "parallel-cap-unknown", cap });
1986
+ return 0;
1987
+ }
1988
+ if (inFlight >= cap) {
1989
+ log(`${inFlight}/${cap} agent(s) in flight (maxParallelAgents) - nothing to take`);
1990
+ tick({ action: "parallel-cap", inFlight, cap });
1991
+ return 0;
1992
+ }
1993
+ }
1994
+
646
1995
  // Per-repo concurrency is always 1, whatever `slots` says: two runs in one
647
1996
  // checkout contend on .git/index.lock, and that is a named failure rather
648
1997
  // than a slow path. It also means the queue steps around a repo YOU are
649
1998
  // working in instead of competing with you for it.
650
1999
  const busyRepos = new Set(running.map((r) => r.repo));
651
- const next = (queue.queued || []).find((i) => !busyRepos.has(i.repo));
2000
+ const next = nextItem(queue, busyRepos, attempts);
652
2001
  if (!next) {
653
2002
  log(queue.emptyReason || "every queued item belongs to a repo already in flight");
654
2003
  tick({
@@ -662,12 +2011,33 @@ function main() {
662
2011
  // ---- 3. ARM ------------------------------------------------------------
663
2012
  const armScript = join(SCRIPTS, "autopilot-arming.mjs");
664
2013
  if (existsSync(armScript)) {
665
- const out = run(process.execPath, [armScript, "check", "--source", next.source, "--json"]);
2014
+ // --probe: the item's credential must be readable from THIS session, the
2015
+ // background one, where a keychain that locked on sleep answers differently
2016
+ // from the terminal that armed the mode. --repo adds the credential its
2017
+ // repo publishes with (gh for a GitHub host), which the publish step needs
2018
+ // after the run has been paid for.
2019
+ const out = run(process.execPath, [
2020
+ armScript,
2021
+ "check",
2022
+ "--source",
2023
+ next.source,
2024
+ "--repo",
2025
+ next.repo,
2026
+ "--probe",
2027
+ "--json",
2028
+ ]);
666
2029
  const arm = (() => {
667
2030
  try {
668
2031
  return JSON.parse(out);
669
2032
  } catch {
670
- return { armed: true };
2033
+ // No verdict is not a yes. The check it stands in for is a spend
2034
+ // ceiling and a credential gate, and reading silence as "armed" hands
2035
+ // the queue the one answer that costs money.
2036
+ return {
2037
+ armed: false,
2038
+ blockedBy: "internal",
2039
+ reason: messages(ARM_UNREADABLE, outputLanguage()),
2040
+ };
671
2041
  }
672
2042
  })();
673
2043
  if (!arm.armed) {
@@ -681,6 +2051,41 @@ function main() {
681
2051
  }
682
2052
  }
683
2053
 
2054
+ // ---- 3b. GUARD ----------------------------------------------------------
2055
+ // The unattended contract rests on agent-guard.sh running as the child's
2056
+ // PreToolUse hook: it is what blocks a push, a PR, an install or a write into
2057
+ // a protected path. A child launched without it would hold every outward
2058
+ // write the runner is meant to be the only source of.
2059
+ const guard = guardRegistration({ cwd: next.localPath });
2060
+ if (!guard.ok) {
2061
+ log(
2062
+ `not launching: ${guard.reason} (install registers it; see features/unattended-security.md)`,
2063
+ );
2064
+ if (!DRY) {
2065
+ record({ source: next.source, id: next.id, outcome: OUTCOME.BLOCKED_GUARD_MISSING });
2066
+ writeState("queue.json", { ...queue, blockedReason: guard.reason });
2067
+ refreshStatus();
2068
+ }
2069
+ tick({ action: "guard-missing", missing: guard.missing });
2070
+ return 0;
2071
+ }
2072
+
2073
+ // ---- 3c. SANDBOX --------------------------------------------------------
2074
+ // The OS sandbox in the profile file is the boundary for everything the
2075
+ // child runs from Bash. Without it, or with a checkout whose own settings
2076
+ // widen it, there is no launch.
2077
+ const sandbox = sandboxPrecondition({ cwd: next.localPath });
2078
+ if (!sandbox.ok) {
2079
+ log(`not launching: ${sandbox.reason} (see features/unattended-security.md)`);
2080
+ if (!DRY) {
2081
+ record({ source: next.source, id: next.id, outcome: OUTCOME.BLOCKED_SANDBOX_UNAVAILABLE });
2082
+ writeState("queue.json", { ...queue, blockedReason: sandbox.reason });
2083
+ refreshStatus();
2084
+ }
2085
+ tick({ action: "sandbox-unavailable" });
2086
+ return 0;
2087
+ }
2088
+
684
2089
  // ---- 4. TAKE -----------------------------------------------------------
685
2090
  const sessionId = randomUUID();
686
2091
  const taskId = `ap-${Date.now()}`;
@@ -691,6 +2096,7 @@ function main() {
691
2096
  taskId,
692
2097
  item: next,
693
2098
  repoPath: next.localPath,
2099
+ baseSha: headSha(next.localPath),
694
2100
  startedAt: Math.floor(Date.now() / 1000),
695
2101
  };
696
2102
 
@@ -699,7 +2105,14 @@ function main() {
699
2105
  return 0;
700
2106
  }
701
2107
 
702
- writeState("runner.pid", claim);
2108
+ // Exclusive create: two ticks that both passed RECOVER race here, and the
2109
+ // loser leaves before touching the queue or launching anything.
2110
+ ensureRoot();
2111
+ if (!claimExclusive(pidPath, claim)) {
2112
+ log("another tick claimed first - nothing to do");
2113
+ return 0;
2114
+ }
2115
+ chmodSync(pidPath, 0o600);
703
2116
  writeState("queue.json", {
704
2117
  ...queue,
705
2118
  running: [
@@ -707,11 +2120,12 @@ function main() {
707
2120
  {
708
2121
  id: next.id,
709
2122
  repo: next.repo,
710
- depth: config.depthRouter === "auto" ? undefined : "full",
711
2123
  stack: next.stack,
712
2124
  startedAt: claim.startedAt,
713
2125
  state: "running",
714
2126
  url: next.url,
2127
+ sessionId,
2128
+ taskId,
715
2129
  },
716
2130
  ],
717
2131
  queued: (queue.queued || []).filter((i) => i.id !== next.id),
@@ -719,86 +2133,128 @@ function main() {
719
2133
  refreshStatus();
720
2134
 
721
2135
  // ---- 5. RUN ------------------------------------------------------------
722
- // The child is a normal pipeline run in the mode that already exists: one
723
- // item, unattended, stopping at an open PR. Nothing about the per-item
724
- // `autopilot` mode changes because a queue is calling it.
725
- log(`running ${next.source}:${next.id}`);
726
- const child = spawnSync(
727
- CLAUDE_BIN,
728
- [
729
- "--bg",
730
- "--session-id",
731
- sessionId,
732
- "--permission-prompts",
733
- "none",
734
- "-p",
735
- `/multi-agent ${next.ref || next.id} autopilot`,
736
- ],
737
- { cwd: next.localPath || process.cwd(), encoding: "utf-8", stdio: ["ignore", "pipe", "pipe"] },
738
- );
739
- if (child.status !== 0) {
740
- const why = (child.stderr || child.error?.message || "").trim().slice(0, 300);
741
- record({ source: next.source, id: next.id, taskId, outcome: "launch-failed", note: why });
742
- releaseSlot(next.id, pidPath, queue);
743
- log(`${next.id}: launch-failed ${why}`);
744
- return 0;
745
- }
2136
+ // The sleep inhibitor covers the session and everything after it in this
2137
+ // tick - research, resume, publish - and is released on every way out.
2138
+ const sleepLock = holdSleep({ root: ROOT, log });
2139
+ try {
2140
+ // The child is a normal pipeline run in the mode that already exists: one
2141
+ // item, unattended, stopping at an open PR. Nothing about the per-item
2142
+ // `autopilot` mode changes because a queue is calling it.
2143
+ log(`running ${next.source}:${next.id}`);
2144
+ const child = launch(launchArgv(next, sessionId), next.localPath, sessionId, claim.baseSha);
2145
+ if (child.status !== 0) {
2146
+ const why = (child.stderr || child.error?.message || "").trim().slice(0, 300);
2147
+ // Nothing ran, so nothing was spent: 0 here is a measurement, not a guess.
2148
+ record({
2149
+ source: next.source,
2150
+ id: next.id,
2151
+ taskId,
2152
+ outcome: OUTCOME.LAUNCH_FAILED,
2153
+ note: why,
2154
+ usd: 0,
2155
+ });
2156
+ releaseSlot(next.id, pidPath, queue);
2157
+ log(`${next.id}: launch-failed ${why}`);
2158
+ return 0;
2159
+ }
746
2160
 
747
- // ---- 5b. SUPERVISE -----------------------------------------------------
748
- // `claude --bg` returns as soon as the session is started - `claude --help`
749
- // says so in as many words. The spawn above therefore proves only that a run
750
- // BEGAN. Everything below used to run immediately after it: the outcome was
751
- // recorded as "pr-opened" before any work happened, the slot was freed within
752
- // a second, and the per-repo guard three steps up became decorative because
753
- // `running` was already empty when the next tick read it.
754
- const supervised = supervise(sessionId, claim, next);
755
-
756
- // ---- 6. RECORD ---------------------------------------------------------
757
- // From the run's own state file, not from the child's exit code and not from
758
- // its stdout - with `--bg` the stdout is a session id, so the PR regex that
759
- // used to read it could never match.
760
- const finalState = claim.statePath ? readJson(claim.statePath, null) : null;
761
- const prUrl = prFromState(finalState);
762
- const outcome = outcomeFor(supervised, finalState, prUrl);
763
- record({
764
- source: next.source,
765
- id: next.id,
766
- taskId,
767
- outcome,
768
- prUrl,
769
- waitedSec: supervised.waitedSec,
770
- usd: 0,
771
- });
2161
+ // ---- 5b. SUPERVISE -----------------------------------------------------
2162
+ // `claude --bg` returns as soon as the session is started - `claude --help`
2163
+ // says so in as many words. The spawn above therefore proves only that a run
2164
+ // BEGAN; recording or releasing anything before the run ends would free the
2165
+ // slot within a second and leave the per-repo guard with nothing to guard.
2166
+ const firstRun = supervise(sessionId, claim, next);
2167
+ const firstState = claim.statePath ? readJson(claim.statePath, null) : null;
772
2168
 
773
- // "gone" and "finished" are the two reasons that mean the run ENDED. The
774
- // other three - a ceiling, an unreadable list, a session parked on a question
775
- // - say only that waiting stopped, and the run may still hold its worktree.
776
- // Deleting the claim there would strand it: the worktree and state path this
777
- // tick just learned are the only record, and the next tick would take another
778
- // item in the same repo. Keeping it means the next tick's RECOVER decides,
779
- // with retire() free to clean up and releaseSlot to free the slot.
780
- if (supervised.reason === "gone" || supervised.reason === "finished") {
781
- releaseSlot(next.id, pidPath, queue);
782
- } else {
783
- log(`${next.id}: claim kept for the next tick (${supervised.reason})`);
784
- }
785
- log(`${next.id}: ${outcome}${prUrl ? ` ${prUrl}` : ""} (${supervised.waitedSec}s)`);
786
- tick({
787
- action: "ran",
788
- source: next.source,
789
- id: next.id,
790
- taskId,
791
- outcome,
792
- reason: supervised.reason,
793
- waitedSec: supervised.waitedSec,
794
- prOpened: Boolean(prUrl),
795
- consecutiveFailuresBefore: breaker.count,
796
- });
797
- return 0;
2169
+ // ---- 5c. RESEARCH ------------------------------------------------------
2170
+ // A run parked on a gap research can act on gets its rounds here, while the
2171
+ // item is still claimed, so no other tick can take it in between.
2172
+ const {
2173
+ supervised,
2174
+ state: finalState,
2175
+ prUrl,
2176
+ rounds,
2177
+ } = researchRounds(
2178
+ claim,
2179
+ next,
2180
+ { supervised: firstRun, state: firstState, prUrl: prFromState(firstState) },
2181
+ {
2182
+ // The ceiling spans every run of the item, so the whole log is read,
2183
+ // not the tail the breaker looks at.
2184
+ roundsUsed: researchRoundsUsed(readAllAttempted(), next),
2185
+ maxAskRounds: Number(config.maxAskRounds ?? 2),
2186
+ },
2187
+ );
2188
+
2189
+ // ---- 5d. PUBLISH -------------------------------------------------------
2190
+ // A session that ended at the Phase 4 hand-off left a PR request; the runner
2191
+ // verifies it and is the one that pushes and opens the PR. A session that
2192
+ // may still be running (a ceiling, an unreadable list) is not published
2193
+ // from: the next tick's RECOVER does that once the session is gone.
2194
+ let pub = { outcome: null };
2195
+ let stateNow = finalState;
2196
+ let prNow = prUrl;
2197
+ if (!prUrl && ENDED.has(supervised.reason)) {
2198
+ pub = await publishStep(claim.sessionId, claim.statePath, next);
2199
+ if (pub.outcome) {
2200
+ stateNow = claim.statePath ? readJson(claim.statePath, finalState) : finalState;
2201
+ prNow = pub.prUrl || prFromState(stateNow);
2202
+ }
2203
+ }
2204
+
2205
+ // ---- 6. RECORD ---------------------------------------------------------
2206
+ // From the run's own state file, not from the child's exit code and not from
2207
+ // its stdout - with `--bg` the stdout is a session id and nothing else.
2208
+ const outcome = pub.outcome || outcomeFor(supervised, stateNow, prNow);
2209
+ record({
2210
+ source: next.source,
2211
+ id: next.id,
2212
+ taskId,
2213
+ outcome,
2214
+ prUrl: prNow,
2215
+ waitedSec: supervised.waitedSec,
2216
+ usd: runUsd(claim.runTaskId, claim.statePath),
2217
+ ...(rounds ? { researchRounds: rounds } : {}),
2218
+ ...publishGateOf(pub),
2219
+ });
2220
+
2221
+ // A delivered run releases its slot whatever its session is doing - Phase 5
2222
+ // can hold for input after the PR is open, and the item is finished with us.
2223
+ // A run parked on a question moves to `running` as parked, see park().
2224
+ // Otherwise "gone" and "finished" are the two reasons that mean the run
2225
+ // ENDED. The rest - a ceiling, an unreadable list - say only that waiting
2226
+ // stopped, and the run may still hold its worktree. Deleting the claim there
2227
+ // would strand it: the worktree and state path this tick just learned are
2228
+ // the only record, and the next tick would take another item in the same
2229
+ // repo. Keeping it means the next tick's RECOVER decides, by the session.
2230
+ if (isTerminal(outcome) || supervised.reason === "gone" || supervised.reason === "finished") {
2231
+ releaseSlot(next.id, pidPath, queue);
2232
+ } else if (isParked(outcome)) {
2233
+ park(claim, next, outcome, pidPath);
2234
+ log(`${next.id}: parked (${outcome}) - the repo stays held until its session ends`);
2235
+ } else {
2236
+ log(`${next.id}: claim kept for the next tick (${supervised.reason})`);
2237
+ }
2238
+ log(`${next.id}: ${outcome}${prNow ? ` ${prNow}` : ""} (${supervised.waitedSec}s)`);
2239
+ tick({
2240
+ action: "ran",
2241
+ source: next.source,
2242
+ id: next.id,
2243
+ taskId,
2244
+ outcome,
2245
+ reason: supervised.reason,
2246
+ waitedSec: supervised.waitedSec,
2247
+ prOpened: Boolean(prNow),
2248
+ consecutiveFailuresBefore: breaker.count,
2249
+ });
2250
+ return 0;
2251
+ } finally {
2252
+ sleepLock.release();
2253
+ }
798
2254
  }
799
2255
 
800
2256
  if (invokedDirectly(import.meta.url)) {
801
- runMain("autopilot-runner", () => {
802
- process.exit(main());
2257
+ runMain("autopilot-runner", async () => {
2258
+ process.exit(await main());
803
2259
  });
804
2260
  }