@tokenfactory/acc-runner 0.25.1 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (162) hide show
  1. package/README.md +77 -4
  2. package/package.json +1 -1
  3. package/dist/anthropic-auth.d.ts +0 -53
  4. package/dist/anthropic-auth.d.ts.map +0 -1
  5. package/dist/anthropic-auth.js +0 -89
  6. package/dist/anthropic-auth.js.map +0 -1
  7. package/dist/bin-resolve.d.ts +0 -16
  8. package/dist/bin-resolve.d.ts.map +0 -1
  9. package/dist/bin-resolve.js +0 -35
  10. package/dist/bin-resolve.js.map +0 -1
  11. package/dist/cli.d.ts +0 -3
  12. package/dist/cli.d.ts.map +0 -1
  13. package/dist/cli.js +0 -128
  14. package/dist/cli.js.map +0 -1
  15. package/dist/config.d.ts +0 -121
  16. package/dist/config.d.ts.map +0 -1
  17. package/dist/config.js +0 -396
  18. package/dist/config.js.map +0 -1
  19. package/dist/cost-pricing.d.ts +0 -130
  20. package/dist/cost-pricing.d.ts.map +0 -1
  21. package/dist/cost-pricing.js +0 -191
  22. package/dist/cost-pricing.js.map +0 -1
  23. package/dist/doctor.d.ts +0 -37
  24. package/dist/doctor.d.ts.map +0 -1
  25. package/dist/doctor.js +0 -658
  26. package/dist/doctor.js.map +0 -1
  27. package/dist/failure-classifier.d.ts +0 -112
  28. package/dist/failure-classifier.d.ts.map +0 -1
  29. package/dist/failure-classifier.js +0 -353
  30. package/dist/failure-classifier.js.map +0 -1
  31. package/dist/gh.d.ts +0 -22
  32. package/dist/gh.d.ts.map +0 -1
  33. package/dist/gh.js +0 -48
  34. package/dist/gh.js.map +0 -1
  35. package/dist/git.d.ts +0 -50
  36. package/dist/git.d.ts.map +0 -1
  37. package/dist/git.js +0 -127
  38. package/dist/git.js.map +0 -1
  39. package/dist/github-client.d.ts +0 -133
  40. package/dist/github-client.d.ts.map +0 -1
  41. package/dist/github-client.js +0 -234
  42. package/dist/github-client.js.map +0 -1
  43. package/dist/keychain.d.ts +0 -21
  44. package/dist/keychain.d.ts.map +0 -1
  45. package/dist/keychain.js +0 -45
  46. package/dist/keychain.js.map +0 -1
  47. package/dist/login.d.ts +0 -12
  48. package/dist/login.d.ts.map +0 -1
  49. package/dist/login.js +0 -133
  50. package/dist/login.js.map +0 -1
  51. package/dist/logout.d.ts +0 -2
  52. package/dist/logout.d.ts.map +0 -1
  53. package/dist/logout.js +0 -31
  54. package/dist/logout.js.map +0 -1
  55. package/dist/mcp-spawn.d.ts +0 -30
  56. package/dist/mcp-spawn.d.ts.map +0 -1
  57. package/dist/mcp-spawn.js +0 -145
  58. package/dist/mcp-spawn.js.map +0 -1
  59. package/dist/messaging.d.ts +0 -49
  60. package/dist/messaging.d.ts.map +0 -1
  61. package/dist/messaging.js +0 -36
  62. package/dist/messaging.js.map +0 -1
  63. package/dist/pkg-version.d.ts +0 -3
  64. package/dist/pkg-version.d.ts.map +0 -1
  65. package/dist/pkg-version.js +0 -20
  66. package/dist/pkg-version.js.map +0 -1
  67. package/dist/profiles/designer-prompt.d.ts +0 -18
  68. package/dist/profiles/designer-prompt.d.ts.map +0 -1
  69. package/dist/profiles/designer-prompt.js +0 -172
  70. package/dist/profiles/designer-prompt.js.map +0 -1
  71. package/dist/profiles/developer-prompt.d.ts +0 -24
  72. package/dist/profiles/developer-prompt.d.ts.map +0 -1
  73. package/dist/profiles/developer-prompt.js +0 -24
  74. package/dist/profiles/developer-prompt.js.map +0 -1
  75. package/dist/profiles/manager-prompt.d.ts +0 -34
  76. package/dist/profiles/manager-prompt.d.ts.map +0 -1
  77. package/dist/profiles/manager-prompt.js +0 -93
  78. package/dist/profiles/manager-prompt.js.map +0 -1
  79. package/dist/profiles/tester-prompt.d.ts +0 -28
  80. package/dist/profiles/tester-prompt.d.ts.map +0 -1
  81. package/dist/profiles/tester-prompt.js +0 -165
  82. package/dist/profiles/tester-prompt.js.map +0 -1
  83. package/dist/prompt.d.ts +0 -38
  84. package/dist/prompt.d.ts.map +0 -1
  85. package/dist/prompt.js +0 -78
  86. package/dist/prompt.js.map +0 -1
  87. package/dist/runtime/cache-dir.d.ts +0 -2
  88. package/dist/runtime/cache-dir.d.ts.map +0 -1
  89. package/dist/runtime/cache-dir.js +0 -15
  90. package/dist/runtime/cache-dir.js.map +0 -1
  91. package/dist/runtime/conflict-resolver.d.ts +0 -65
  92. package/dist/runtime/conflict-resolver.d.ts.map +0 -1
  93. package/dist/runtime/conflict-resolver.js +0 -477
  94. package/dist/runtime/conflict-resolver.js.map +0 -1
  95. package/dist/runtime/expand-args.d.ts +0 -28
  96. package/dist/runtime/expand-args.d.ts.map +0 -1
  97. package/dist/runtime/expand-args.js +0 -50
  98. package/dist/runtime/expand-args.js.map +0 -1
  99. package/dist/runtime/locks.d.ts +0 -21
  100. package/dist/runtime/locks.d.ts.map +0 -1
  101. package/dist/runtime/locks.js +0 -97
  102. package/dist/runtime/locks.js.map +0 -1
  103. package/dist/runtime/provision-mutex.d.ts +0 -37
  104. package/dist/runtime/provision-mutex.d.ts.map +0 -1
  105. package/dist/runtime/provision-mutex.js +0 -67
  106. package/dist/runtime/provision-mutex.js.map +0 -1
  107. package/dist/runtime/quarantine.d.ts +0 -26
  108. package/dist/runtime/quarantine.d.ts.map +0 -1
  109. package/dist/runtime/quarantine.js +0 -50
  110. package/dist/runtime/quarantine.js.map +0 -1
  111. package/dist/runtime/resolution-integrity.d.ts +0 -86
  112. package/dist/runtime/resolution-integrity.d.ts.map +0 -1
  113. package/dist/runtime/resolution-integrity.js +0 -248
  114. package/dist/runtime/resolution-integrity.js.map +0 -1
  115. package/dist/runtime/reviewer.d.ts +0 -81
  116. package/dist/runtime/reviewer.d.ts.map +0 -1
  117. package/dist/runtime/reviewer.js +0 -374
  118. package/dist/runtime/reviewer.js.map +0 -1
  119. package/dist/runtime/rework.d.ts +0 -48
  120. package/dist/runtime/rework.d.ts.map +0 -1
  121. package/dist/runtime/rework.js +0 -136
  122. package/dist/runtime/rework.js.map +0 -1
  123. package/dist/runtime/singleton.d.ts +0 -47
  124. package/dist/runtime/singleton.d.ts.map +0 -1
  125. package/dist/runtime/singleton.js +0 -200
  126. package/dist/runtime/singleton.js.map +0 -1
  127. package/dist/runtime/version-drift.d.ts +0 -31
  128. package/dist/runtime/version-drift.d.ts.map +0 -1
  129. package/dist/runtime/version-drift.js +0 -114
  130. package/dist/runtime/version-drift.js.map +0 -1
  131. package/dist/runtime/worktree.d.ts +0 -74
  132. package/dist/runtime/worktree.d.ts.map +0 -1
  133. package/dist/runtime/worktree.js +0 -206
  134. package/dist/runtime/worktree.js.map +0 -1
  135. package/dist/secrets/inject.d.ts +0 -70
  136. package/dist/secrets/inject.d.ts.map +0 -1
  137. package/dist/secrets/inject.js +0 -102
  138. package/dist/secrets/inject.js.map +0 -1
  139. package/dist/supabase.d.ts +0 -4
  140. package/dist/supabase.d.ts.map +0 -1
  141. package/dist/supabase.js +0 -36
  142. package/dist/supabase.js.map +0 -1
  143. package/dist/task-runner.d.ts +0 -313
  144. package/dist/task-runner.d.ts.map +0 -1
  145. package/dist/task-runner.js +0 -1766
  146. package/dist/task-runner.js.map +0 -1
  147. package/dist/token-provider.d.ts +0 -50
  148. package/dist/token-provider.d.ts.map +0 -1
  149. package/dist/token-provider.js +0 -177
  150. package/dist/token-provider.js.map +0 -1
  151. package/dist/types.d.ts +0 -120
  152. package/dist/types.d.ts.map +0 -1
  153. package/dist/types.js +0 -16
  154. package/dist/types.js.map +0 -1
  155. package/dist/version-check.d.ts +0 -17
  156. package/dist/version-check.d.ts.map +0 -1
  157. package/dist/version-check.js +0 -56
  158. package/dist/version-check.js.map +0 -1
  159. package/dist/watch.d.ts +0 -295
  160. package/dist/watch.d.ts.map +0 -1
  161. package/dist/watch.js +0 -1532
  162. package/dist/watch.js.map +0 -1
@@ -1,1766 +0,0 @@
1
- /**
2
- * Task execution path. Called by watch.ts when a task_assigned broadcast
3
- * lands on the runner channel.
4
- *
5
- * 1. Transition task → running (RPC).
6
- * 2. fetch_task_for_runner to get the task + adjacent agent/model/runner.
7
- * 3. Render the prompt, ensure repo is fresh, branch is checked out.
8
- * 4. Spawn `claude --print` with the prompt on stdin; pipe stdout/stderr
9
- * to acc.append_task_event so the History tab updates live.
10
- * 5. On success, push the branch and open a PR via `gh`.
11
- * 6. On failure, transition → failed and emit an error task_event.
12
- *
13
- * Cancellation: watch.ts holds a reference to the running child via the
14
- * returned controller and SIGTERMs it on task_cancelled.
15
- */
16
- import fs from "node:fs";
17
- import { existsSync, unlinkSync } from "node:fs";
18
- import os from "node:os";
19
- import path from "node:path";
20
- import { join } from "node:path";
21
- import { fileURLToPath, pathToFileURL } from "node:url";
22
- import { execa } from "execa";
23
- import { withAnthropicAuth } from "./anthropic-auth.js";
24
- import { loadProfile as defaultLoadProfile } from "./config.js";
25
- import { normalizeUsage, parseClaudeJson, priceUsdCents, toCliAlias, } from "./cost-pricing.js";
26
- import { classifyClaudeExit, extractResetTime, isGitProvisionContention, } from "./failure-classifier.js";
27
- import { startTaskGithubBudget, endTaskGithubBudget, } from "./github-client.js";
28
- import { git as defaultGit } from "./git.js";
29
- import { gh as defaultGh } from "./gh.js";
30
- import { writeMcpConfig as defaultWriteMcpConfig, } from "./mcp-spawn.js";
31
- import { postRunnerStateMessage as defaultPostRunnerStateMessage, } from "./messaging.js";
32
- import { branchForTask, prTitleForTask, renderTaskPrompt, } from "./prompt.js";
33
- import { acquireTaskLock as defaultAcquireTaskLock, TaskLockHeldError, } from "./runtime/locks.js";
34
- import { prepareTaskWorktree as defaultPrepareTaskWorktree, } from "./runtime/worktree.js";
35
- // v0.21 T-66-1: serialize the per-task worktree-provisioning critical section
36
- // against the shared clone so concurrent tasks can't collide on git's
37
- // repo-global locks (the T-65-2 race).
38
- import { withProvisionLock } from "./runtime/provision-mutex.js";
39
- import { parseConflictMeta, resolveConflict, MAX_CONFLICT_ATTEMPTS, } from "./runtime/conflict-resolver.js";
40
- import { parseReworkMeta, renderReworkPrompt, fetchPrHeadRef as defaultFetchPrHeadRef, } from "./runtime/rework.js";
41
- import { assertResolutionIntegrity, isConflictResolutionTask, summariseFailures, } from "./runtime/resolution-integrity.js";
42
- const LOG_BATCH_BYTES = 4 * 1024;
43
- /**
44
- * v0.6.0 REG-296: build the argv for `claude --print`. Splitting this
45
- * out keeps the model-passthrough logic unit-testable without spawning
46
- * a real subprocess. modelId is forwarded as `--model <id>` when
47
- * non-empty (whitespace counts as empty); otherwise omitted so claude
48
- * uses the operator's default model.
49
- */
50
- export function buildClaudeArgs(modelId) {
51
- const args = ["--print", "--dangerously-skip-permissions", "--output-format=json"];
52
- const trimmed = (modelId ?? "").trim();
53
- if (trimmed)
54
- args.push("--model", trimmed);
55
- return args;
56
- }
57
- function defaultSpawnClaude(cwd, modelId) {
58
- // --dangerously-skip-permissions: bypass claude's per-edit approval
59
- // prompts. The runner is non-interactive (no TTY for human approval)
60
- // and the permission boundary is enforced one level up — see the
61
- // "Runner sandbox" section in packages/acc-runner/README.md.
62
- // --output-format=json: emit a structured object so we can parse the
63
- // usage block for cost tracking. parseClaudeJson falls back to plain
64
- // text when older claude versions emit markdown directly.
65
- // --model: forwarded when the task pins a model (REG-296).
66
- // v0.74-B: withAnthropicAuth forwards ANTHROPIC_API_KEY (trimmed) when set
67
- // so claude authenticates via the API account (no per-session cap) instead
68
- // of the operator's interactive OAuth session. claude has no --api-key flag;
69
- // the env var is the only auth channel. See src/anthropic-auth.ts.
70
- return execa("claude", buildClaudeArgs(modelId), {
71
- cwd,
72
- stdin: "pipe",
73
- stdout: "pipe",
74
- stderr: "pipe",
75
- reject: false,
76
- env: withAnthropicAuth(process.env),
77
- });
78
- }
79
- async function defaultCheckoutBase(repoPath, baseBranch) {
80
- // Plain `git checkout <branch>` (not `-B`) so the integration
81
- // branch's existing ref is honored. With -B we'd reset the
82
- // integration branch to current HEAD, which is the exact bug
83
- // REG-301 was filed for in reverse — destructive instead of stale.
84
- await execa("git", ["checkout", baseBranch], { cwd: repoPath, env: process.env });
85
- }
86
- /**
87
- * v0.6.0 REG-295: expand `~/...` paths against $HOME so a task can
88
- * carry `repo_path_hint = "~/work/foo"` from a row populated on a
89
- * different machine. Absolute and relative paths pass through.
90
- */
91
- export function expandHomePath(p) {
92
- if (p === "~")
93
- return os.homedir();
94
- if (p.startsWith("~/"))
95
- return path.join(os.homedir(), p.slice(2));
96
- return p;
97
- }
98
- /**
99
- * v0.63 T-63-1: default claude health probe. A silent instant-empty exit is
100
- * ambiguous — it can be a broken host OR a momentarily-unavailable claude.
101
- * `claude --version` is a cheap, deterministic answer to "is the binary/host
102
- * actually broken?": if it returns 0 the host is fine (the exit was
103
- * transient → claude_unavailable, retry), and any spawn failure / non-zero
104
- * exit / timeout means the binary or its dynamic deps are genuinely broken
105
- * (→ env_broken, quarantine). Short timeout so a probe never wedges the run.
106
- */
107
- async function defaultHealthProbe() {
108
- try {
109
- const res = await execa("claude", ["--version"], {
110
- reject: false,
111
- timeout: 5_000,
112
- env: process.env,
113
- });
114
- if (res.exitCode === 0) {
115
- return { ok: true, detail: (res.stdout || "").trim().slice(0, 200) || "claude --version ok" };
116
- }
117
- return {
118
- ok: false,
119
- detail: `claude --version exited ${res.exitCode ?? "signal"}: ${(res.stderr || "").trim().slice(0, 200)}`,
120
- };
121
- }
122
- catch (err) {
123
- return { ok: false, detail: `claude --version spawn failed: ${err.message}` };
124
- }
125
- }
126
- const DEFAULT_CLAUDE_UNAVAILABLE_KNOBS = {
127
- windowMs: 15 * 60_000,
128
- alertThreshold: 5,
129
- backoffBaseMs: 30_000,
130
- backoffMaxMs: 15 * 60_000,
131
- };
132
- // Process-level state. runTask() is a fresh closure per task, but the module
133
- // is loaded once, so consecutive task runs in the same `acc-runner watch`
134
- // process share this — the only place a cross-task streak can live without
135
- // touching watch.ts (out of this track's file scope). A clean (exit-0) claude
136
- // run resets it: claude has recovered.
137
- let claudeUnavailableKnobs = { ...DEFAULT_CLAUDE_UNAVAILABLE_KNOBS };
138
- let claudeUnavailableHits = [];
139
- let claudeUnavailableStreak = 0;
140
- /** v0.63 T-63-1: override the silent-unavailability knobs (used by tests). */
141
- export function configureClaudeUnavailable(partial) {
142
- claudeUnavailableKnobs = { ...claudeUnavailableKnobs, ...partial };
143
- }
144
- /** v0.63 T-63-1: reset both the streak and the knobs (used by tests). */
145
- export function resetClaudeUnavailableState() {
146
- claudeUnavailableHits = [];
147
- claudeUnavailableStreak = 0;
148
- claudeUnavailableKnobs = { ...DEFAULT_CLAUDE_UNAVAILABLE_KNOBS };
149
- }
150
- /**
151
- * v0.63 T-63-1: record one silent instant-empty exit and compute the response
152
- * (exponential backoff + whether the sustained-streak INFRA alert trips).
153
- * Exported so the watch process and the unit suite share one definition.
154
- */
155
- export function recordClaudeUnavailable(now) {
156
- claudeUnavailableStreak += 1;
157
- claudeUnavailableHits.push(now);
158
- const cutoff = now - claudeUnavailableKnobs.windowMs;
159
- claudeUnavailableHits = claudeUnavailableHits.filter((t) => t >= cutoff);
160
- const windowCount = claudeUnavailableHits.length;
161
- const alert = windowCount >= claudeUnavailableKnobs.alertThreshold;
162
- const expo = claudeUnavailableKnobs.backoffBaseMs * 2 ** (claudeUnavailableStreak - 1);
163
- const backoffMs = Math.min(expo, claudeUnavailableKnobs.backoffMaxMs);
164
- return {
165
- streak: claudeUnavailableStreak,
166
- windowCount,
167
- alert,
168
- backoffMs,
169
- resumeAt: new Date(now + backoffMs).toISOString(),
170
- };
171
- }
172
- /** v0.63 T-63-1: a clean claude run proves the binary recovered — reset. */
173
- function clearClaudeUnavailableStreak() {
174
- claudeUnavailableStreak = 0;
175
- claudeUnavailableHits = [];
176
- }
177
- // v0.37-A HEAD-LOCK-GUARD: git worktree add/remove/prune can leave stale
178
- // .git/HEAD.lock or .git/index.lock behind. The virtiofs boundary keeps
179
- // the sandbox from clearing them, but the host runner process can. Call
180
- // this against the parent repo's path (the dir containing .git/) before
181
- // any worktree op.
182
- function clearGitLocks(repoRoot) {
183
- for (const name of ["HEAD.lock", "index.lock"]) {
184
- const p = join(repoRoot, ".git", name);
185
- try {
186
- if (existsSync(p)) {
187
- unlinkSync(p);
188
- process.stderr.write("[runner] cleared stale lock: " + p + "\n");
189
- }
190
- }
191
- catch { /* noop */ }
192
- }
193
- }
194
- async function appendEvent(supabase, taskId, kind, payload) {
195
- const { error } = await supabase.rpc("append_task_event", {
196
- p_task_id: taskId,
197
- p_kind: kind,
198
- p_payload: payload,
199
- });
200
- if (error) {
201
- // Logging failures shouldn't crash the run — print to local stderr.
202
- process.stderr.write(`[acc-runner] append_task_event failed: ${error.message}\n`);
203
- }
204
- }
205
- async function streamToEvents(stream, supabase, taskId, streamName) {
206
- let captured = "";
207
- let pending = "";
208
- const flush = async () => {
209
- if (!pending)
210
- return;
211
- const text = pending;
212
- pending = "";
213
- captured += text;
214
- await appendEvent(supabase, taskId, "log", { stream: streamName, text });
215
- };
216
- for await (const raw of stream) {
217
- const text = typeof raw === "string" ? raw : raw.toString("utf8");
218
- pending += text;
219
- if (pending.length >= LOG_BATCH_BYTES)
220
- await flush();
221
- }
222
- await flush();
223
- return captured;
224
- }
225
- /**
226
- * Pull the structured report block from claude's stdout. Recognises
227
- * both `--output-format=json` (extracts .result) and plain text. Falls
228
- * back to the entire stdout (capped) if no `## Summary` heading is
229
- * present — that way the PR body still tells the reviewer what happened.
230
- */
231
- export function extractReportFromOutput(stdout) {
232
- const parsed = parseClaudeJson(stdout);
233
- const text = parsed?.result ?? stdout;
234
- const idx = text.indexOf("## Summary");
235
- if (idx === -1) {
236
- const trimmed = text.trim();
237
- if (!trimmed)
238
- return "_(no report emitted)_";
239
- return trimmed.slice(-8000);
240
- }
241
- return text.slice(idx).trim().slice(0, 16000);
242
- }
243
- /**
244
- * v0.35-B: extract the PR number from a `gh pr create` URL. Returns null
245
- * when the URL doesn't carry a `/pull/<int>` segment so the caller can
246
- * fall back to the webhook-driven path. GitHub URLs encode PR ids as the
247
- * last path segment; anchors and query strings are tolerated.
248
- */
249
- export function parsePrNumber(url) {
250
- if (!url)
251
- return null;
252
- const m = url.match(/\/pull\/(\d+)(?:[/?#]|$)/);
253
- if (!m)
254
- return null;
255
- const n = parseInt(m[1], 10);
256
- return Number.isFinite(n) && n > 0 ? n : null;
257
- }
258
- /**
259
- * v0.35-B: write `pr_number` onto acc.tasks AND transition running →
260
- * needs-review in one RPC. Best-effort: a missing function (Track A
261
- * migration not yet applied) OR any error logs to stderr and returns
262
- * false so the caller falls back to the webhook-driven done path
263
- * (running → done via the matrix widened in 0156). The RPC contract is
264
- * `acc.set_task_pr(p_task_id text, p_pr_number int, p_pr_html_url text)`.
265
- */
266
- async function setTaskPr(supabase, taskId, prUrl) {
267
- const prNumber = parsePrNumber(prUrl);
268
- if (prNumber === null)
269
- return false;
270
- const { error } = await supabase.rpc("set_task_pr", {
271
- p_task_id: taskId,
272
- p_pr_number: prNumber,
273
- p_pr_html_url: prUrl,
274
- });
275
- if (error) {
276
- process.stderr.write(`[acc-runner] set_task_pr(${taskId}, ${prNumber}) failed: ${error.message} ` +
277
- `— falling back to webhook-driven running→done path\n`);
278
- return false;
279
- }
280
- return true;
281
- }
282
- /**
283
- * v0.53 T-53-4: emit a `task.first_commit` activity event the first time a
284
- * task's worktree produces a commit on its branch (the SLO funnel signal
285
- * for "coding actually started"). Additive — the runner protocol message
286
- * shapes are unchanged; this rides the existing `log_activity` RPC path.
287
- *
288
- * Best-effort and exactly-once per task run: it sits on the single
289
- * post-success, pre-push code path. A repo with no firstCommit support
290
- * (a GitRunner stub without the method) or a branch with no commits ahead
291
- * of the base simply emits nothing. Failures log to stderr and never block
292
- * the push.
293
- */
294
- async function emitFirstCommit(supabase, git, taskId, workdir, baseRef, branch) {
295
- try {
296
- const first = await git.firstCommit?.(workdir, baseRef, branch);
297
- if (!first)
298
- return;
299
- const { error } = await supabase.rpc("log_activity", {
300
- p_verb: "task.first_commit",
301
- p_target_id: taskId,
302
- p_target_type: "task",
303
- p_payload: {
304
- task_id: taskId,
305
- branch,
306
- sha: first.sha,
307
- ts: first.ts || null,
308
- },
309
- });
310
- if (error) {
311
- process.stderr.write(`[acc-runner] task.first_commit log_activity(${taskId}) failed: ${error.message}\n`);
312
- }
313
- }
314
- catch (err) {
315
- process.stderr.write(`[acc-runner] task.first_commit emit failed: ${err.message}\n`);
316
- }
317
- }
318
- /**
319
- * Build a cost event from claude's stdout. Always returns a payload
320
- * (zero-cost when usage isn't parseable) so the cap-tracking surface
321
- * stays consistent: one cost_events row per attempt.
322
- */
323
- export function buildCostEvent(taskId, stdout, fallbackModel, runnerId) {
324
- const parsed = parseClaudeJson(stdout);
325
- const model = parsed?.model ?? fallbackModel ?? "unknown";
326
- const usage = normalizeUsage(parsed?.usage);
327
- return {
328
- task_id: taskId,
329
- model,
330
- usd_cents: priceUsdCents(model, usage),
331
- runner_id: runnerId ?? null,
332
- ...usage,
333
- };
334
- }
335
- /**
336
- * v0.14-MESSAGING-RUNTIME-WIRE — dynamic-import the prelude module
337
- * referenced by a runner profile and return its default-exported string.
338
- *
339
- * The prelude path in the profile JSON (e.g. `./src/profiles/developer-
340
- * prompt.ts`) is interpreted from the runner package root — the parent
341
- * of the runner-profiles dir where the JSON lives. In source/dev the
342
- * .ts file is importable via tsx; in the npm-published artifact only
343
- * `dist/` ships, so we map `src/*.ts` → `dist/*.js` when the .ts target
344
- * doesn't exist. Returns empty string when no prelude is configured or
345
- * the import resolves to a non-string default (v0.13 byte-identical).
346
- *
347
- * Exported so tests can stub the resolution + import seam via the
348
- * `loadPromptPrelude` dep on RunTaskDeps — vitest's Vite-backed
349
- * transform pipeline does not handle dynamic imports of arbitrary file
350
- * URLs in the jsdom environment, so the production path is exercised
351
- * by the runner's own unit suite (node env) rather than by the
352
- * top-level integration test.
353
- */
354
- export async function loadPromptPrelude(profile) {
355
- if (!profile.promptPrelude || profile.promptPrelude.length === 0)
356
- return "";
357
- // Runner-profiles dir lives at <pkg-root>/runner-profiles/. Resolve
358
- // the prelude path from the package root so a JSON like
359
- // `./src/profiles/developer-prompt.ts` lands on a real file.
360
- const profilesRoot = process.env.ACC_RUNNER_PROFILES_DIR?.trim() ||
361
- path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "runner-profiles");
362
- const pkgRoot = path.resolve(profilesRoot, "..");
363
- let resolved = path.resolve(pkgRoot, profile.promptPrelude);
364
- // Production map: the published package ships dist/ only.
365
- if (!fs.existsSync(resolved)) {
366
- const distMapped = resolved
367
- .replace(/([/\\])src\1/, "$1dist$1")
368
- .replace(/\.ts$/, ".js");
369
- if (fs.existsSync(distMapped))
370
- resolved = distMapped;
371
- }
372
- const mod = (await import(/* @vite-ignore */ pathToFileURL(resolved).href));
373
- const def = mod.default;
374
- return typeof def === "string" ? def : "";
375
- }
376
- async function defaultPostCostEvent(supabase, event) {
377
- const { error } = await supabase.rpc("record_cost_event", {
378
- p_task_id: event.task_id,
379
- p_model: event.model,
380
- p_input_tokens: event.input_tokens,
381
- p_output_tokens: event.output_tokens,
382
- p_cache_read_tokens: event.cache_read_tokens,
383
- p_cache_write_tokens: event.cache_write_tokens,
384
- p_usd_cents: event.usd_cents,
385
- p_runner_id: event.runner_id ?? null,
386
- });
387
- if (error)
388
- throw new Error(error.message);
389
- }
390
- /**
391
- * v0.48 T-48-5: conflict_resolution task path.
392
- *
393
- * Called when the fetched task has type='conflict_resolution'. Does not
394
- * spawn Claude — resolves conflicts via GitHub API and transitions directly.
395
- * Runs inside the existing claim/lock/signal infrastructure of runTask.
396
- */
397
- async function runConflictResolutionPath(taskId, task, deps, postState_, appendEventFn) {
398
- await postState_("planning", "info", {
399
- runner_id: deps.session.runnerId,
400
- task_type: "conflict_resolution",
401
- });
402
- const meta = parseConflictMeta(task.description, {
403
- pr_number: task.pr_number,
404
- repo: task.repo,
405
- integration_branch: task.integration_branch,
406
- });
407
- if (!meta || !meta.repo) {
408
- const msg = "conflict_resolution task missing PR metadata (pr_number + repo in description or task fields)";
409
- await appendEventFn("error", { phase: "conflict_resolution", error: msg });
410
- await deps.supabase.rpc("transition_task", {
411
- p_task_id: taskId,
412
- p_new_status: "failed",
413
- });
414
- return { taskId, status: "failed", phase: "fetch", error: msg };
415
- }
416
- await postState_("coding", "info", {
417
- runner_id: deps.session.runnerId,
418
- pr_number: meta.pr_number,
419
- repo: meta.repo,
420
- });
421
- const conflictDeps = {
422
- supabase: deps.supabase,
423
- ...deps.conflictResolutionDeps,
424
- };
425
- const outcome = await resolveConflict(meta, conflictDeps);
426
- await appendEventFn("conflict_resolution_attempt", {
427
- action: outcome.action,
428
- reason: outcome.reason ?? null,
429
- pr_number: meta.pr_number,
430
- repo: meta.repo,
431
- files_resolved: outcome.files_resolved ?? [],
432
- files_skipped: outcome.files_skipped ?? [],
433
- max_attempts: MAX_CONFLICT_ATTEMPTS,
434
- });
435
- if (outcome.action === "resolved") {
436
- await postState_("done", "info", {
437
- runner_id: deps.session.runnerId,
438
- pr_number: meta.pr_number,
439
- resolution: "resolved",
440
- });
441
- await deps.supabase.rpc("transition_task", {
442
- p_task_id: taskId,
443
- p_new_status: "done",
444
- });
445
- // Best-effort activity log
446
- try {
447
- await deps.supabase.rpc("log_activity", {
448
- p_verb: "conflict.resolved",
449
- p_target_id: String(meta.pr_number),
450
- p_payload: {
451
- task_id: taskId,
452
- repo: meta.repo,
453
- files: outcome.files_resolved ?? [],
454
- reason: outcome.reason,
455
- },
456
- p_target_type: "pr",
457
- });
458
- }
459
- catch { /* best-effort */ }
460
- return { taskId, status: "ok", exitCode: 0 };
461
- }
462
- if (outcome.action === "policy_disabled") {
463
- // detect-and-skip: transition to done, no escalation
464
- await postState_("done", "info", {
465
- runner_id: deps.session.runnerId,
466
- skipped_reason: "conflict_policy.enabled=false",
467
- });
468
- await deps.supabase.rpc("transition_task", {
469
- p_task_id: taskId,
470
- p_new_status: "done",
471
- });
472
- return { taskId, status: "ok", exitCode: 0 };
473
- }
474
- if (outcome.action === "escalated") {
475
- // max attempts reached — escalate to operator
476
- const escalationMsg = outcome.reason ?? "conflict_unresolvable after max attempts";
477
- await postState_("blocked", "error_context", {
478
- task_id: taskId,
479
- phase: "conflict_resolution",
480
- error_class: "conflict_unresolvable",
481
- pr_number: meta.pr_number,
482
- repo: meta.repo,
483
- stderr_tail: escalationMsg,
484
- });
485
- try {
486
- await deps.supabase.rpc("log_activity", {
487
- p_verb: "conflict.escalated",
488
- p_target_id: String(meta.pr_number),
489
- p_payload: {
490
- task_id: taskId,
491
- repo: meta.repo,
492
- reason: escalationMsg,
493
- prior_attempts: MAX_CONFLICT_ATTEMPTS,
494
- },
495
- p_target_type: "pr",
496
- });
497
- }
498
- catch { /* best-effort */ }
499
- // Transition to done (not failed) so Sweep C doesn't create another task.
500
- // The escalation signal is on the message bus + activity log.
501
- await deps.supabase.rpc("transition_task", {
502
- p_task_id: taskId,
503
- p_new_status: "done",
504
- });
505
- return { taskId, status: "ok", exitCode: 0 };
506
- }
507
- // outcome.action === "unresolvable" — transition to failed so Sweep C can retry.
508
- const failMsg = outcome.reason ?? "conflict not automatically resolvable";
509
- await postState_("blocked", "error_context", {
510
- task_id: taskId,
511
- phase: "conflict_resolution",
512
- error_class: "conflict_unresolvable_single",
513
- pr_number: meta.pr_number,
514
- stderr_tail: failMsg,
515
- });
516
- await deps.supabase.rpc("transition_task", {
517
- p_task_id: taskId,
518
- p_new_status: "failed",
519
- });
520
- return {
521
- taskId,
522
- status: "failed",
523
- phase: "claude_exit",
524
- error: failMsg,
525
- };
526
- }
527
- export function runTask(taskId, deps) {
528
- const git = deps.git ?? defaultGit;
529
- const gh = deps.gh ?? defaultGh;
530
- const spawnClaude = deps.spawnClaude ?? defaultSpawnClaude;
531
- const checkoutBase = deps.checkoutBase ?? defaultCheckoutBase;
532
- const acquireLock = deps.acquireLock ?? defaultAcquireTaskLock;
533
- const prepareWorktree = deps.prepareWorktree ?? defaultPrepareTaskWorktree;
534
- const postCostEvent = deps.postCostEvent ?? ((event) => defaultPostCostEvent(deps.supabase, event));
535
- const postState = deps.postRunnerStateMessage ??
536
- ((args) => defaultPostRunnerStateMessage(deps.supabase, args));
537
- const loadProfile = deps.loadProfile ?? defaultLoadProfile;
538
- const loadPromptPreludeFn = deps.loadPromptPrelude ?? loadPromptPrelude;
539
- const healthProbe = deps.healthProbe ?? defaultHealthProbe;
540
- // Silence the unused-binding lint for `checkoutBase` — v0.11-F supersedes
541
- // the v0.6.0 REG-301 pre-spawn `git checkout <baseBranch>` with the
542
- // worktree's `add -B <branch> <path> <baseBranch>` start-point semantics,
543
- // but the dep is still accepted for back-compat with test fixtures that
544
- // inject a mock. Drop in v0.7.
545
- void checkoutBase;
546
- let child = null;
547
- let cancelled = false;
548
- // v0.14-MESSAGING-RUNTIME-WIRE: convenience wrapper that scopes every
549
- // bus message to this task + this runner. Best-effort: an underlying
550
- // RPC failure is logged inside defaultPostRunnerStateMessage and never
551
- // bubbles out, so this wrapper does not need its own try/catch.
552
- const postState_ = async (state, protocol, payload) => {
553
- await postState({
554
- task_id: taskId,
555
- sender_id: deps.session.runnerId,
556
- state,
557
- protocol,
558
- payload,
559
- });
560
- };
561
- // v0.6.1 (v0.11-D): release lock rows on every terminal exit path
562
- // below the claim. Best-effort — a release failure logs to stderr
563
- // but does not change the outcome the runner reports to the caller.
564
- // Calling release without first calling claim is a no-op at the
565
- // RPC level, so wiring this into both pre-claim and post-claim
566
- // returns would be safe; we only call it from post-claim returns
567
- // to keep the code path obvious to a reader.
568
- const releaseLocks = async () => {
569
- const { error } = await deps.supabase.rpc("release_task_locks", {
570
- p_task_id: taskId,
571
- });
572
- if (error) {
573
- process.stderr.write(`[acc-runner] release_task_locks(${taskId}) failed: ${error.message}\n`);
574
- }
575
- };
576
- // v0.12-RESUME — periodic signal loop. After the claim succeeds we
577
- // bump acc.tasks.last_runner_signal_at every signalIntervalMs ms so
578
- // the v0.12 /5m sweep distinguishes "runner alive, work in flight"
579
- // from "runner crashed, task stuck at running". Self-rescheduling
580
- // setTimeout (not setInterval) so a slow RPC doesn't queue up
581
- // overlapping firings; the loop stops as soon as `signalStopped`
582
- // flips in the outer try/finally below.
583
- const DEFAULT_SIGNAL_MS = 30_000;
584
- const signalIntervalMs = deps.signalIntervalMs ?? DEFAULT_SIGNAL_MS;
585
- let signalTimer = null;
586
- let signalStopped = false;
587
- // v0.68 (T-68-1): bound the signal RPC. supabase-js fetch has no default
588
- // timeout, so a single hung connection would freeze the self-rescheduling
589
- // chain below — last_runner_signal_at stops advancing while the runner
590
- // heartbeat keeps going, and the stale-running sweep reclaims a live task
591
- // mid-run (the signal-stale zombie symptom). Racing a timeout guarantees
592
- // the chain always advances to the next tick even when one call stalls.
593
- const SIGNAL_RPC_TIMEOUT_MS = 10_000;
594
- const updateSignal = async () => {
595
- let timer;
596
- try {
597
- const timeout = new Promise((_, reject) => {
598
- timer = setTimeout(() => reject(new Error(`update_task_signal timed out after ${SIGNAL_RPC_TIMEOUT_MS}ms`)), SIGNAL_RPC_TIMEOUT_MS);
599
- if (timer.unref)
600
- timer.unref();
601
- });
602
- const result = (await Promise.race([
603
- deps.supabase.rpc("update_task_signal", { p_task_id: taskId }),
604
- timeout,
605
- ]));
606
- const error = result?.error ?? null;
607
- if (error) {
608
- // Best-effort: a missed signal just means the sweep might pull
609
- // the task back if enough of them stack up. Log and continue.
610
- process.stderr.write(`[acc-runner] update_task_signal(${taskId}) failed: ${error.message}\n`);
611
- }
612
- }
613
- catch (err) {
614
- process.stderr.write(`[acc-runner] update_task_signal(${taskId}) ${err.message}\n`);
615
- }
616
- finally {
617
- if (timer)
618
- clearTimeout(timer);
619
- }
620
- };
621
- const scheduleSignal = () => {
622
- if (signalStopped)
623
- return;
624
- signalTimer = setTimeout(async () => {
625
- if (signalStopped)
626
- return;
627
- try {
628
- await updateSignal();
629
- }
630
- catch { /* logged inside */ }
631
- scheduleSignal();
632
- }, signalIntervalMs);
633
- // Detach so an in-flight signal timer doesn't keep the process
634
- // alive past `watch.ts` shutdown. The outer finally clears it
635
- // anyway; this is belt-and-suspenders for stray timers.
636
- if (signalTimer.unref)
637
- signalTimer.unref();
638
- };
639
- const stopSignalLoop = () => {
640
- signalStopped = true;
641
- if (signalTimer) {
642
- clearTimeout(signalTimer);
643
- signalTimer = null;
644
- }
645
- };
646
- const promise = (async () => {
647
- // 1. Atomic claim + transition to running. v0.11-D: replaces the
648
- // pre-v0.6.1 raw transition_task('running') call. The RPC
649
- // row-locks acc.tasks FOR UPDATE, checks file-path overlap
650
- // against other running tasks' locks, INSERTs a lock row +
651
- // transitions to running in one transaction. On overlap or
652
- // same-task race the task stays queued for another runner.
653
- const claim = await deps.supabase.rpc("claim_task_with_locks", {
654
- p_task_id: taskId,
655
- p_runner_id: deps.session.runnerId,
656
- });
657
- if (claim.error) {
658
- const msg = claim.error.message;
659
- // Includes 22023 (invalid_task_transition surfaced through the
660
- // nested acc.transition_task call) — task may already be past
661
- // 'running' (e.g. needs-review). Log and bail rather than crash.
662
- await appendEvent(deps.supabase, taskId, "error", {
663
- phase: "claim_locks",
664
- error: msg,
665
- });
666
- return { taskId, status: "failed", phase: "claim_locks", error: msg };
667
- }
668
- const claimResult = (claim.data ?? {});
669
- if (claimResult.ok !== true) {
670
- const conflicts = Array.isArray(claimResult.conflicts) ? claimResult.conflicts : [];
671
- await appendEvent(deps.supabase, taskId, "log", {
672
- phase: "claim_locks",
673
- stream: "stderr",
674
- conflicts,
675
- message: `file-lock conflict — other tasks hold overlapping paths: ${conflicts.join(", ")}`,
676
- runner_id: deps.session.runnerId,
677
- });
678
- const reason = `file-lock conflict with: ${conflicts.join(", ") || "(unknown)"}`;
679
- return {
680
- taskId,
681
- status: "failed",
682
- phase: "claim_locks",
683
- error: reason,
684
- };
685
- }
686
- // v0.6.1 (v0.11-D): every exit path below the successful claim
687
- // must release the lock row so the same paths free up for the
688
- // next runner. try/finally captures returns AND uncaught throws
689
- // alike — strictly stronger than explicit pre-return release at
690
- // each of the seven completion sites, and the runner CLI never
691
- // recovers from a thrown error inside runTask, so the lock
692
- // would otherwise leak until a future sweep job clears it.
693
- try {
694
- // v0.53 T-53-4: arm the per-task-execution GitHub soft budget. Any
695
- // direct REST call routed through github-client.githubFetch during
696
- // this task is counted against DEFAULT_TASK_GITHUB_BUDGET and fails
697
- // clean (retryable infra, never quarantine) once spent. NOTE: the
698
- // runner's current GitHub access is entirely via the `gh` CLI
699
- // (gh.ts, conflict-resolver, rework, reviewer) — those separate
700
- // processes are out of scope and remain UNBUDGETED here. Disarmed in
701
- // the finally so out-of-task callers (doctor) run unbudgeted.
702
- startTaskGithubBudget();
703
- // v0.12-RESUME: prime the signal column immediately on claim so
704
- // the sweep window resets from "now" and the first periodic tick
705
- // (after signalIntervalMs) refreshes it. Without this prime, a
706
- // task whose run takes < signalIntervalMs from claim to first
707
- // tick could race the sweep on borderline updated_at values.
708
- // Placed inside the outer try so an unexpected throw from the
709
- // RPC still runs the finally (stop loop, release locks).
710
- await updateSignal();
711
- scheduleSignal();
712
- // v0.14-MESSAGING-RUNTIME-WIRE: first bus event — runner has the
713
- // claim, will now fetch + spawn. planner subscribes via /messages.
714
- await postState_("planning", "info", { runner_id: deps.session.runnerId });
715
- // 2. Fetch task + adjacent rows.
716
- const fetched = await deps.supabase.rpc("fetch_task_for_runner", {
717
- p_task_id: taskId,
718
- });
719
- if (fetched.error || !fetched.data) {
720
- const msg = fetched.error?.message ?? "fetch_task_for_runner returned no data";
721
- await appendEvent(deps.supabase, taskId, "error", { phase: "fetch", error: msg });
722
- await postState_("blocked", "error_context", {
723
- task_id: taskId,
724
- phase: "plan",
725
- error_class: "fetch_task_failed",
726
- stderr_tail: msg,
727
- });
728
- await deps.supabase.rpc("transition_task", {
729
- p_task_id: taskId,
730
- p_new_status: "failed",
731
- });
732
- return { taskId, status: "failed", phase: "fetch", error: msg };
733
- }
734
- const result = fetched.data;
735
- const { task } = result;
736
- // v0.48 T-48-5: conflict_resolution tasks bypass Claude Code entirely.
737
- // The runner resolves conflicts on the real base∩head file set via the
738
- // GitHub API, pushes the resolution, and transitions the task to done
739
- // (or escalates after MAX_CONFLICT_ATTEMPTS failed attempts).
740
- if (task.type === "conflict_resolution") {
741
- return await runConflictResolutionPath(taskId, task, deps, postState_, appendEvent.bind(null, deps.supabase, taskId));
742
- }
743
- // v0.51 T-51-1: rework tasks run the normal Claude session, but against
744
- // the PR's EXISTING head branch — the worktree forks from
745
- // origin/<branch> instead of the integration branch, the prompt carries
746
- // the reviewer's verdict + comments, and the push goes back to the same
747
- // branch (same PR; no new PR is opened).
748
- let reworkMeta = null;
749
- if (task.type === "rework") {
750
- reworkMeta = parseReworkMeta(task.description, {
751
- pr_number: task.pr_number,
752
- repo: task.repo,
753
- branch: task.branch,
754
- });
755
- if (!reworkMeta) {
756
- const msg = "rework task missing PR metadata (acc-rework sentinel or pr_number + repo on task row)";
757
- await appendEvent(deps.supabase, taskId, "error", { phase: "rework", error: msg });
758
- await postState_("blocked", "error_context", {
759
- task_id: taskId,
760
- phase: "plan",
761
- error_class: "rework_meta_missing",
762
- stderr_tail: msg,
763
- });
764
- await deps.supabase.rpc("transition_task", {
765
- p_task_id: taskId,
766
- p_new_status: "failed",
767
- });
768
- return { taskId, status: "failed", phase: "fetch", error: msg };
769
- }
770
- if (!reworkMeta.branch) {
771
- try {
772
- reworkMeta.branch = await (deps.fetchPrHeadRef ?? defaultFetchPrHeadRef)(reworkMeta.repo, reworkMeta.pr_number);
773
- }
774
- catch { /* handled by the empty-branch check below */ }
775
- if (!reworkMeta.branch) {
776
- const msg = `rework task could not resolve head branch for PR #${reworkMeta.pr_number}`;
777
- await appendEvent(deps.supabase, taskId, "error", { phase: "rework", error: msg });
778
- await deps.supabase.rpc("transition_task", {
779
- p_task_id: taskId,
780
- p_new_status: "failed",
781
- });
782
- return { taskId, status: "failed", phase: "fetch", error: msg };
783
- }
784
- }
785
- }
786
- const branch = reworkMeta
787
- ? reworkMeta.branch
788
- : branchForTask(task.id, task.title, task.branch);
789
- // REG-295: per-task repo overrides — task row trumps config so one
790
- // runner can service tasks across multiple repos. Env-backed config
791
- // remains the fallback for tasks that don't carry a hint yet.
792
- const repoPath = task.repo_path_hint?.trim()
793
- ? expandHomePath(task.repo_path_hint.trim())
794
- : deps.cfg.repoPath;
795
- const targetRepo = task.repo?.trim() || deps.cfg.targetRepo;
796
- // REG-301: branch base comes from task → cfg → main. cfg.integrationBranch
797
- // already defaults to "acc/integration" so the third fallback only
798
- // matters when an operator zeroed it out via env.
799
- const integrationBranch = task.integration_branch?.trim() || deps.cfg.integrationBranch || "main";
800
- // v0.51 T-51-1: rework prompts replace the "open a PR" protocol header
801
- // with "update the existing PR's branch" + the reviewer's verdict.
802
- const renderedPrompt = reworkMeta
803
- ? renderReworkPrompt({ task, meta: reworkMeta, branch })
804
- : renderTaskPrompt({
805
- task,
806
- agent: result.agent,
807
- model: result.model,
808
- integrationBranch,
809
- targetRepo,
810
- });
811
- // v0.14-MESSAGING-RUNTIME-WIRE: profile-driven prompt prelude.
812
- // When loadProfile() returns null OR the profile has no prelude
813
- // OR the prelude module's default export is the empty string,
814
- // `prompt` is byte-identical to the v0.13 `renderedPrompt`.
815
- const activeProfile = loadProfile();
816
- let prelude = "";
817
- if (activeProfile && activeProfile.promptPrelude) {
818
- try {
819
- prelude = await loadPromptPreludeFn(activeProfile);
820
- }
821
- catch (err) {
822
- // A broken prelude module should not block the task — log and
823
- // fall through to v0.13 behavior. The operator surfaces this
824
- // via the per-task History stream so the misconfiguration is
825
- // discoverable without crashing the run.
826
- await appendEvent(deps.supabase, taskId, "error", {
827
- phase: "prelude",
828
- error: err.message,
829
- profile: activeProfile.name,
830
- prelude_path: activeProfile.promptPrelude,
831
- });
832
- }
833
- }
834
- const prompt = prelude ? `${prelude}\n\n${renderedPrompt}` : renderedPrompt;
835
- // v0.11-F: acquire the per-task PID lock before any worktree
836
- // side-effect. Same-machine parallel runners that picked up the same
837
- // task_id (e.g. two `acc-runner watch` processes seeing the same
838
- // broadcast) race here; the second one bails on TaskLockHeldError.
839
- let lock;
840
- try {
841
- lock = await acquireLock(taskId);
842
- }
843
- catch (err) {
844
- if (err instanceof TaskLockHeldError) {
845
- await appendEvent(deps.supabase, taskId, "error", {
846
- phase: "worktree_lock",
847
- error: err.message,
848
- held_by_pid: err.heldByPid,
849
- });
850
- await deps.supabase.rpc("transition_task", {
851
- p_task_id: taskId,
852
- p_new_status: "failed",
853
- });
854
- return { taskId, status: "failed", phase: "worktree_lock", error: err.message };
855
- }
856
- throw err;
857
- }
858
- // Worktree gets assigned inside the git-prep block; cleanup in the
859
- // outer finally handles both the success path and every early
860
- // return that follows.
861
- let worktree = null;
862
- let workdir = repoPath;
863
- // v0.19 T-64-1: the ref the task branch is actually forked from, set
864
- // once the base is resolved below. Used as the `base..branch` range for
865
- // the first-commit SLO so the signal stays accurate now that FRESH
866
- // branches fork from origin/<integration> rather than the local ref.
867
- let resolvedBaseRef = integrationBranch;
868
- // v0.57 T-57-2: conflict-resolution integrity guardrail. For tasks whose
869
- // job is to resolve a merge conflict (the Claude/worktree path that the
870
- // mechanical resolver hands off to when it bails), re-verify the
871
- // resolution from the runner's own vantage point before pushing —
872
- // zero conflict markers, typecheck green, and no silently-dropped /
873
- // regressed version bump. On failure the resolution is REJECTED (no
874
- // push, not marked resolved) and the task fails with a diagnostic that
875
- // names the offending assertion, so a broken merge is reworked rather
876
- // than shipped. A no-op for every non-conflict task.
877
- const runResolutionIntegrityGate = async () => {
878
- if (!isConflictResolutionTask(task))
879
- return null;
880
- const assertIntegrity = deps.assertResolutionIntegrity ?? assertResolutionIntegrity;
881
- let result;
882
- try {
883
- result = await assertIntegrity(workdir, {
884
- baseRef: integrationBranch,
885
- typecheckCmd: task.typecheck_cmd ?? undefined,
886
- // The #496 regression lived in the acc-runner package; also guard
887
- // the repo root. Each pair is skipped silently when its version is
888
- // unchanged or unreadable, so this never false-fails.
889
- versionTargets: [
890
- { packageJsonPath: "package.json", changelogPath: "CHANGELOG.md" },
891
- {
892
- packageJsonPath: "packages/acc-runner/package.json",
893
- changelogPath: "packages/acc-runner/CHANGELOG.md",
894
- },
895
- ],
896
- });
897
- }
898
- catch (err) {
899
- // A guardrail that itself crashed must fail closed: reject, never
900
- // wave a resolution through because the check errored.
901
- result = {
902
- ok: false,
903
- checked: [],
904
- failures: [
905
- {
906
- assertion: "typecheck",
907
- detail: `integrity check crashed: ${err.message?.slice(0, 200)}`,
908
- },
909
- ],
910
- };
911
- }
912
- if (result.ok) {
913
- await appendEvent(deps.supabase, taskId, "resolution_integrity_passed", {
914
- pr_number: task.pr_number ?? null,
915
- checked: result.checked,
916
- });
917
- return null;
918
- }
919
- const summary = summariseFailures(result.failures);
920
- const detail = result.failures
921
- .map((f) => `${f.assertion}: ${f.detail}`)
922
- .join("\n");
923
- await appendEvent(deps.supabase, taskId, "resolution_integrity_failed", {
924
- pr_number: task.pr_number ?? null,
925
- failed_assertions: result.failures.map((f) => f.assertion),
926
- files: result.failures.flatMap((f) => f.files ?? []),
927
- detail: detail.slice(-4000),
928
- });
929
- // Best-effort operator-facing activity event naming the failed assertion.
930
- try {
931
- await deps.supabase.rpc("log_activity", {
932
- p_verb: "conflict.integrity_rejected",
933
- p_target_id: String(task.pr_number ?? taskId),
934
- p_payload: {
935
- task_id: taskId,
936
- repo: task.repo ?? null,
937
- failed_assertions: result.failures.map((f) => f.assertion),
938
- },
939
- p_target_type: task.pr_number ? "pr" : "task",
940
- });
941
- }
942
- catch {
943
- /* best-effort */
944
- }
945
- await postState_("blocked", "error_context", {
946
- task_id: taskId,
947
- phase: "review",
948
- error_class: "resolution_integrity_failed",
949
- stderr_tail: detail.slice(-4000),
950
- });
951
- // Reject: do NOT push, do NOT mark resolved. Fail the task so the
952
- // resolution is reworked instead of merged broken.
953
- await deps.supabase.rpc("transition_task", {
954
- p_task_id: taskId,
955
- p_new_status: "failed",
956
- });
957
- return {
958
- taskId,
959
- status: "failed",
960
- phase: "push",
961
- error: `conflict-resolution integrity guardrail rejected resolution (${summary})`,
962
- };
963
- };
964
- try {
965
- // 3. Repo prep. v0.11-F: fetch on the shared clone (worktree
966
- // shares the object store) then provision an isolated worktree
967
- // at ~/.cache/acc-runner/work/<task_id>/ forked from the
968
- // integration branch. The redundant `git.checkout -B <branch>`
969
- // is a no-op inside the new worktree — kept so the v0.6.0
970
- // checkout seam stays observable in unit tests.
971
- try {
972
- // v0.19 T-64-1: cut a FRESH task branch from the LATEST
973
- // integration tip, not a stale LOCAL ref. `git.fetch` runs
974
- // `fetch --prune --all`, which refreshes origin/<integrationBranch>;
975
- // the worktree below then forks from that remote-tracking ref
976
- // (mirroring the v0.51 rework path, which forks from
977
- // origin/<branch>). T-63-1's PR conflicted in package.json /
978
- // CHANGELOG / task-runner.ts precisely because a 13-commit-stale
979
- // LOCAL integration ref (base 5f627e7) was used as the branch base.
980
- //
981
- // ROBUST FETCH (acceptance): a fetch failure (offline / network)
982
- // must NOT abort the task. We fall back to the stale local ref,
983
- // log a warning + a `runner.stale_base_fallback` activity event so
984
- // the staleness is auditable, and continue. The fix never
985
- // force-pushes and is idempotent.
986
- // v0.21 T-66-1: serialize the git plumbing below (fetch + stale-lock
987
- // clear + worktree add/remove) against the shared clone. Two tasks
988
- // running concurrently would otherwise collide on git's repo-global
989
- // locks and the loser would die in this pre-Claude phase (T-65-2).
990
- // Only this critical section is held; the Claude run + push that
991
- // follow stay concurrent up to concurrencyLimit. Serializing here
992
- // also makes clearGitLocks safe — it can no longer delete a lock a
993
- // concurrent provision is actively holding.
994
- const provisionCloneKey = deps.cfg.worktreeRepoPath ?? repoPath;
995
- worktree = await withProvisionLock(provisionCloneKey, async () => {
996
- let staleBaseFallback = false;
997
- try {
998
- await git.fetch(repoPath);
999
- }
1000
- catch (fetchErr) {
1001
- staleBaseFallback = true;
1002
- const detail = fetchErr.message;
1003
- process.stderr.write(`[acc-runner] base fetch failed for ${taskId}; cutting ${branch} ` +
1004
- `from stale local ${integrationBranch}: ${detail}\n`);
1005
- await appendEvent(deps.supabase, taskId, "log", {
1006
- phase: "git",
1007
- stream: "stderr",
1008
- event: "runner.stale_base_fallback",
1009
- base: integrationBranch,
1010
- branch,
1011
- detail,
1012
- runner_id: deps.session.runnerId,
1013
- });
1014
- // Best-effort audit trail on the runner timeline.
1015
- try {
1016
- await deps.supabase.rpc("log_activity", {
1017
- p_verb: "runner.stale_base_fallback",
1018
- p_target_id: deps.session.runnerId,
1019
- p_target_type: "runner",
1020
- p_payload: {
1021
- task_id: taskId,
1022
- base: integrationBranch,
1023
- branch,
1024
- detail,
1025
- },
1026
- });
1027
- }
1028
- catch { /* best-effort */ }
1029
- }
1030
- // v0.37-A: clear any stale .git/HEAD.lock or index.lock left over
1031
- // from a prior crashed run before `git worktree add` touches refs.
1032
- // v0.21 T-66-1: safe under concurrency now — held inside the
1033
- // provision lock, so it cannot stomp a peer's active lock.
1034
- clearGitLocks(repoPath);
1035
- // v0.51 T-51-1: rework tasks fork from the PR's remote head so
1036
- // Claude sees the branch's current contents. v0.19 T-64-1: FRESH
1037
- // tasks fork from the freshly-fetched origin tip
1038
- // (origin/<integrationBranch>), or — only when the fetch failed —
1039
- // the stale local ref. The crash-resume path inside
1040
- // prepareTaskWorktree returns the existing partial branch untouched
1041
- // regardless of this base (it never resets a resumed branch:
1042
- // REG-301 / v0.12-RESUME), so a stale base here can never clobber
1043
- // partially-committed work.
1044
- const worktreeBaseBranch = reworkMeta
1045
- ? `origin/${branch}`
1046
- : staleBaseFallback
1047
- ? integrationBranch
1048
- : `origin/${integrationBranch}`;
1049
- // Rework keeps its prior first-commit base (integration) so the SLO
1050
- // measures the whole PR effort; FRESH tasks use their true fork point.
1051
- resolvedBaseRef = reworkMeta ? integrationBranch : worktreeBaseBranch;
1052
- return await prepareWorktree({
1053
- repoPath,
1054
- // v0.41-B: when an operator has configured a separate mirror
1055
- // clone for worktree ops, prepareTaskWorktree runs add/remove/
1056
- // prune against it; `git fetch` above still targets repoPath.
1057
- // Undefined preserves v0.40 behaviour byte-for-byte.
1058
- worktreeRepoPath: deps.cfg.worktreeRepoPath,
1059
- taskId,
1060
- branch,
1061
- baseBranch: worktreeBaseBranch,
1062
- });
1063
- });
1064
- workdir = worktree.path;
1065
- // v0.12-RESUME: log resume-vs-fresh so an operator can audit
1066
- // how often the resume path actually fires. `resumed: true`
1067
- // means a prior runner crashed mid-task, the v0.12 sweep
1068
- // returned the task to queued, and this runner picked it
1069
- // back up with the prior worktree intact. Claude reads the
1070
- // partially-committed state and continues; the spawn is the
1071
- // same prompt either way (Claude is idempotent enough that
1072
- // re-running on a partially-edited worktree converges on
1073
- // the right final state).
1074
- if (worktree.resumed) {
1075
- await appendEvent(deps.supabase, taskId, "log", {
1076
- phase: "git",
1077
- stream: "stdout",
1078
- event: "worktree.resumed",
1079
- worktree_path: workdir,
1080
- branch,
1081
- runner_id: deps.session.runnerId,
1082
- });
1083
- }
1084
- await git.checkout(workdir, branch);
1085
- }
1086
- catch (err) {
1087
- const msg = err.message;
1088
- // v0.21 T-66-1: a git REPO-GLOBAL LOCK collision during provisioning
1089
- // (an external git process touched the clone while we held the
1090
- // provision mutex) is TRANSIENT — requeue the task so it re-provisions
1091
- // cleanly once the lock frees, instead of burning a terminal `failed`.
1092
- // The in-process mutex already removes the runner-vs-runner case; this
1093
- // covers the residual external-contention case.
1094
- if (isGitProvisionContention(msg)) {
1095
- await appendEvent(deps.supabase, taskId, "log", {
1096
- phase: "git",
1097
- stream: "stderr",
1098
- event: "runner.git_provision_contention",
1099
- error_class: "git_provision_contention",
1100
- error: msg,
1101
- branch,
1102
- runner_id: deps.session.runnerId,
1103
- });
1104
- // running → queued: re-dispatch for a clean retry (matches the
1105
- // stale-running sweep's lossless requeue; not a retry-budget burn).
1106
- await deps.supabase.rpc("transition_task", {
1107
- p_task_id: taskId,
1108
- p_new_status: "queued",
1109
- });
1110
- return {
1111
- taskId,
1112
- status: "requeued",
1113
- phase: "git",
1114
- error: `git_provision_contention: ${msg}`,
1115
- };
1116
- }
1117
- await appendEvent(deps.supabase, taskId, "error", {
1118
- phase: "git",
1119
- error: msg,
1120
- base: integrationBranch,
1121
- branch,
1122
- repo_path: repoPath,
1123
- worktree_path: worktree?.path ?? null,
1124
- });
1125
- await postState_("blocked", "error_context", {
1126
- task_id: taskId,
1127
- phase: "plan",
1128
- error_class: "git_prep_failed",
1129
- stderr_tail: msg,
1130
- });
1131
- await deps.supabase.rpc("transition_task", {
1132
- p_task_id: taskId,
1133
- p_new_status: "failed",
1134
- });
1135
- return { taskId, status: "failed", phase: "git", error: msg };
1136
- }
1137
- // 4. Spawn Claude.
1138
- if (cancelled) {
1139
- return { taskId, status: "cancelled" };
1140
- }
1141
- // v0.32-A: idempotency guard. If the runner was killed mid-task and
1142
- // the sweep returned this task to 'queued', a prior run may have
1143
- // already pushed the branch and opened a PR. Re-running Claude Code
1144
- // would fail at `git push` (non-fast-forward). Check for an existing
1145
- // open PR before spawning and skip the coding step if one is found.
1146
- // v0.51 T-51-1: skipped for rework tasks — their PR exists by design;
1147
- // the whole point is to spawn Claude against it.
1148
- if (!reworkMeta) {
1149
- const existingPrUrl = await gh.findOpenPR(workdir, branch);
1150
- if (existingPrUrl) {
1151
- await appendEvent(deps.supabase, taskId, "log", {
1152
- phase: "coding",
1153
- stream: "stdout",
1154
- event: "retry.pr_exists",
1155
- branch,
1156
- pr_url: existingPrUrl,
1157
- message: "open PR already exists for branch — skipping Claude Code spawn",
1158
- });
1159
- // v0.35-B: stamp pr_number + transition running→needs-review on
1160
- // the recovered task. On any failure the webhook-driven flow
1161
- // still moves running→done at merge time (matrix 0156).
1162
- await setTaskPr(deps.supabase, taskId, existingPrUrl);
1163
- await postState_("done", "info", {
1164
- runner_id: deps.session.runnerId,
1165
- pr_url: existingPrUrl,
1166
- skipped_reason: "pr_already_open",
1167
- });
1168
- return { taskId, status: "ok", prUrl: existingPrUrl, exitCode: 0 };
1169
- }
1170
- }
1171
- // v0.14-MESSAGING-RUNTIME-WIRE: worktree is ready, Claude is
1172
- // about to start the actual work. coding marks the transition
1173
- // into the long-running subprocess phase.
1174
- await postState_("coding", "info", {
1175
- runner_id: deps.session.runnerId,
1176
- branch,
1177
- model: result.model?.id ?? null,
1178
- });
1179
- // 4a. Provision .mcp.json so Claude Code auto-discovers acc-mcp-server.
1180
- // Best-effort: a failed write must not block the task. The MCP server
1181
- // is a context source, not a critical dependency for v0.5-C1.
1182
- let mcpCleanup = null;
1183
- if (deps.session) {
1184
- const writer = deps.writeMcpConfig ?? defaultWriteMcpConfig;
1185
- try {
1186
- mcpCleanup = await writer({
1187
- cwd: workdir,
1188
- taskId,
1189
- runnerId: deps.session.runnerId,
1190
- accessToken: deps.session.accessToken,
1191
- publicUrl: deps.publicUrl ?? deps.cfg.publicUrl,
1192
- supabaseUrl: deps.cfg.supabaseUrl,
1193
- supabaseAnonKey: deps.cfg.supabaseAnonKey,
1194
- });
1195
- }
1196
- catch (err) {
1197
- // Don't echo the token even on failure.
1198
- process.stderr.write(`[acc-runner] mcp .mcp.json write failed: ${err.message}\n`);
1199
- }
1200
- }
1201
- // v0.12-MODEL-ALIAS (REG-303/304): translate the ACC model alias
1202
- // (`claude-sonnet-4`) into the wire form `claude --model` actually
1203
- // accepts (`sonnet` or `claude-sonnet-4-6`). Unknown ids pass
1204
- // through verbatim so a future model not yet in the embedded
1205
- // table still spawns.
1206
- //
1207
- // v0.53 T-53-4: one claude attempt — spawn, stream both pipes to the
1208
- // History tab, post a cost event (one row per attempt), and time the
1209
- // run so the classifier's fast-exit env_broken heuristic can fire.
1210
- const runClaudeOnce = async () => {
1211
- const spawnedAt = Date.now();
1212
- child = spawnClaude(workdir, toCliAlias(result.model?.id));
1213
- if (child.stdin) {
1214
- child.stdin.write(prompt);
1215
- child.stdin.end();
1216
- }
1217
- const stdoutPromise = child.stdout
1218
- ? streamToEvents(child.stdout, deps.supabase, taskId, "stdout")
1219
- : Promise.resolve("");
1220
- const stderrPromise = child.stderr
1221
- ? streamToEvents(child.stderr, deps.supabase, taskId, "stderr")
1222
- : Promise.resolve("");
1223
- const [outcome, capturedStdout, capturedStderr] = await Promise.all([
1224
- child,
1225
- stdoutPromise,
1226
- stderrPromise,
1227
- ]);
1228
- // One cost_events row per attempt so cap tracking captures success,
1229
- // cancellation, retry, and non-zero exit alike. Best-effort: a
1230
- // failed POST logs to stderr but never bubbles past the runner.
1231
- {
1232
- const event = buildCostEvent(taskId, capturedStdout, result.model?.id, result.runner?.id);
1233
- try {
1234
- await postCostEvent(event);
1235
- }
1236
- catch (err) {
1237
- process.stderr.write(`[acc-runner] cost-event post failed: ${err.message}\n`);
1238
- }
1239
- }
1240
- return {
1241
- exitCode: outcome.exitCode,
1242
- stdout: capturedStdout,
1243
- stderr: capturedStderr,
1244
- durationMs: Date.now() - spawnedAt,
1245
- };
1246
- };
1247
- // v0.53 T-53-4 / v0.63 T-63-1: instant-empty tune. A lone instant-empty
1248
- // exit (the pd36 fast-exit fingerprint, now classified
1249
- // `claude_unavailable`) is too weak a signal to act on — v0.52 incident
1250
- // (d) quarantined a working machine on ONE 249ms empty exit. So when the
1251
- // FIRST exit is a *heuristic* claude_unavailable we log it and retry the
1252
- // spawn once; a single blip that the retry recovers stays silent. Only a
1253
- // CONFIRMED streak (two consecutive instant-empties) is acted on — and
1254
- // even then v0.63 routes it to pause + backoff + retry (NOT quarantine),
1255
- // gated by the health probe below. Definitive failures (dyld/OOM/exit
1256
- // 127, usage/auth patterns — classified.heuristic falsey) skip the retry
1257
- // and quarantine immediately, unchanged from v0.48.
1258
- let attempt = await runClaudeOnce();
1259
- const classifyIfFailed = (a) => a.exitCode === 0
1260
- ? null
1261
- : classifyClaudeExit(a.exitCode, a.stderr, a.stdout, a.durationMs);
1262
- let classified = classifyIfFailed(attempt);
1263
- let instantEmptyStreak = classified?.class === "claude_unavailable" && classified.heuristic ? 1 : 0;
1264
- if (instantEmptyStreak === 1 && !cancelled) {
1265
- await appendEvent(deps.supabase, taskId, "log", {
1266
- phase: "claude_exit",
1267
- stream: "stderr",
1268
- event: "claude_unavailable_blip",
1269
- attempt: 1,
1270
- duration_ms: attempt.durationMs,
1271
- message: "instant empty claude exit — single blip, retrying task once before pausing",
1272
- });
1273
- attempt = await runClaudeOnce();
1274
- classified = classifyIfFailed(attempt);
1275
- if (classified?.class === "claude_unavailable" && classified.heuristic) {
1276
- instantEmptyStreak = 2;
1277
- }
1278
- }
1279
- // Restore .mcp.json as soon as Claude is done — its MCP subprocess
1280
- // tree comes down with it, so leaving our token-bearing config on
1281
- // disk a moment longer is pure exposure surface.
1282
- if (mcpCleanup) {
1283
- try {
1284
- await mcpCleanup.restore();
1285
- }
1286
- catch (err) {
1287
- process.stderr.write(`[acc-runner] mcp .mcp.json restore failed: ${err.message}\n`);
1288
- }
1289
- }
1290
- if (cancelled) {
1291
- await appendEvent(deps.supabase, taskId, "cancelled", {
1292
- exit_code: attempt.exitCode,
1293
- });
1294
- return { taskId, status: "cancelled", exitCode: attempt.exitCode };
1295
- }
1296
- if (attempt.exitCode !== 0) {
1297
- // v0.48: classify the failure so the quarantine cause is consistent
1298
- // across the event, the bus message, and the RunTaskOutcome. `let`
1299
- // because v0.63's health probe can upgrade a claude_unavailable exit
1300
- // to env_broken when the probe proves the host is genuinely broken.
1301
- let cls = classified ??
1302
- classifyClaudeExit(attempt.exitCode, attempt.stderr, attempt.stdout, attempt.durationMs);
1303
- // v0.56 (T-56-1): capacity exhaustion is NOT a task failure. Do not
1304
- // transition the task to 'failed' (that would dead-letter it and burn
1305
- // the reviewer/automerge retry cap). Leave it 'running' with its
1306
- // signal loop stopped (the outer finally does this) so the
1307
- // stale-running sweep returns it to 'queued' with runner_id cleared —
1308
- // a lossless requeue for any runner that still has capacity. Surface
1309
- // the parsed reset time so watch.ts can pause until the window
1310
- // reopens. The locks are released by the outer finally exactly as on
1311
- // any other exit path.
1312
- if (cls.class === "capacity_exhausted") {
1313
- const resumeMs = extractResetTime(`${attempt.stderr}\n${attempt.stdout}`);
1314
- const resumeAt = resumeMs !== null ? new Date(resumeMs).toISOString() : null;
1315
- await appendEvent(deps.supabase, taskId, "capacity_paused", {
1316
- phase: "claude_exit",
1317
- exit_code: attempt.exitCode,
1318
- exit_class: cls.class,
1319
- resume_at: resumeAt,
1320
- detail: cls.detail,
1321
- runner_id: deps.session.runnerId,
1322
- });
1323
- return {
1324
- taskId,
1325
- status: "capacity_paused",
1326
- phase: "claude_exit",
1327
- exitCode: attempt.exitCode,
1328
- error: cls.detail,
1329
- capacity_exhausted: true,
1330
- resume_at: resumeAt,
1331
- };
1332
- }
1333
- // v0.63 (T-63-1): a bare silent instant-empty exit. Silence alone
1334
- // cannot tell "broken host" from "claude momentarily unavailable", so
1335
- // run an authoritative `claude --version` health probe as the
1336
- // tiebreaker. Probe OK → the host is fine → treat exactly like
1337
- // capacity_exhausted (pause + EXPONENTIAL backoff + retry; leave the
1338
- // task 'running' for the stale-running sweep; NEVER quarantine, NEVER
1339
- // burn the task/review retry caps). Probe FAILS → the binary/host is
1340
- // genuinely broken → upgrade to env_broken and fall through to the
1341
- // quarantine path (the pd36 contract, now gated by a real signal
1342
- // rather than silence). A human INFRA alert fires only after a
1343
- // SUSTAINED streak inside a rolling window, never on 2.
1344
- if (cls.class === "claude_unavailable") {
1345
- const probe = await healthProbe();
1346
- await appendEvent(deps.supabase, taskId, "log", {
1347
- phase: "claude_exit",
1348
- stream: "stderr",
1349
- event: "claude_health_probe",
1350
- ok: probe.ok,
1351
- detail: probe.detail,
1352
- consecutive_instant_empty: instantEmptyStreak,
1353
- });
1354
- if (probe.ok) {
1355
- const decision = recordClaudeUnavailable(Date.now());
1356
- await appendEvent(deps.supabase, taskId, "claude_unavailable", {
1357
- phase: "claude_exit",
1358
- exit_code: attempt.exitCode,
1359
- exit_class: cls.class,
1360
- consecutive_instant_empty: instantEmptyStreak,
1361
- duration_ms: attempt.durationMs,
1362
- resume_at: decision.resumeAt,
1363
- streak: decision.streak,
1364
- window_count: decision.windowCount,
1365
- detail: cls.detail,
1366
- runner_id: deps.session.runnerId,
1367
- });
1368
- // Best-effort audit trail on the runner timeline.
1369
- try {
1370
- await deps.supabase.rpc("log_activity", {
1371
- p_verb: "runner.claude_unavailable",
1372
- p_target_id: deps.session.runnerId,
1373
- p_target_type: "runner",
1374
- p_payload: {
1375
- task_id: taskId,
1376
- streak: decision.streak,
1377
- window_count: decision.windowCount,
1378
- backoff_ms: decision.backoffMs,
1379
- resume_at: decision.resumeAt,
1380
- detail: cls.detail,
1381
- },
1382
- });
1383
- }
1384
- catch { /* best-effort */ }
1385
- // Escalate to a human only on a SUSTAINED streak — not a single
1386
- // pause, not 2. Still NOT a quarantine: the runner stays online,
1387
- // paused + retrying, while an operator investigates infra.
1388
- if (decision.alert) {
1389
- try {
1390
- await deps.supabase.rpc("log_activity", {
1391
- p_verb: "runner.claude_unavailable_infra",
1392
- p_target_id: deps.session.runnerId,
1393
- p_target_type: "runner",
1394
- p_payload: {
1395
- task_id: taskId,
1396
- window_count: decision.windowCount,
1397
- streak: decision.streak,
1398
- detail: cls.detail,
1399
- },
1400
- });
1401
- }
1402
- catch { /* best-effort */ }
1403
- await postState_("blocked", "error_context", {
1404
- task_id: taskId,
1405
- phase: "code",
1406
- error_class: "claude_unavailable_sustained",
1407
- infra_alert: true,
1408
- window_count: decision.windowCount,
1409
- stderr_tail: cls.detail,
1410
- });
1411
- }
1412
- return {
1413
- taskId,
1414
- status: "capacity_paused",
1415
- phase: "claude_exit",
1416
- exitCode: attempt.exitCode,
1417
- error: `claude silently unavailable (instant-empty exit) — pausing ` +
1418
- `${Math.round(decision.backoffMs / 1000)}s then retrying, not quarantining`,
1419
- claude_unavailable: true,
1420
- resume_at: decision.resumeAt,
1421
- };
1422
- }
1423
- // Probe failed — the host/binary really is broken. This IS an
1424
- // env_broken; fall through to the quarantine path below.
1425
- cls = {
1426
- exitCode: attempt.exitCode,
1427
- class: "env_broken",
1428
- detail: `${cls.detail}; claude health probe failed: ${probe.detail}`,
1429
- };
1430
- }
1431
- // v0.48 / v0.63: by this point claude_unavailable has either returned
1432
- // (probe OK) or been rewritten to env_broken (probe failed), so any
1433
- // non-task_error class is a genuine quarantine cause. consecutive
1434
- // mirrors the instant-empty streak for an env_broken upgraded from the
1435
- // heuristic (≥1), so quarantine.json still records 1 (definitive) vs 2
1436
- // (heuristic-confirmed).
1437
- const quarantineCause = cls.class !== "task_error" ? cls.class : undefined;
1438
- const quarantineConsecutive = quarantineCause === "env_broken" ? Math.max(instantEmptyStreak, 1) : undefined;
1439
- await appendEvent(deps.supabase, taskId, "error", {
1440
- phase: "claude_exit",
1441
- exit_code: attempt.exitCode,
1442
- exit_class: cls.class,
1443
- exit_class_heuristic: cls.heuristic === true,
1444
- consecutive_instant_empty: instantEmptyStreak,
1445
- quarantine: quarantineCause !== undefined,
1446
- stderr_tail: attempt.stderr.slice(-2000),
1447
- });
1448
- await postState_("blocked", "error_context", {
1449
- task_id: taskId,
1450
- phase: "code",
1451
- error_class: `claude_exit_${attempt.exitCode ?? "unknown"}`,
1452
- // v0.48: include the classified failure class so the planner can
1453
- // distinguish machine-level failures from task-level failures.
1454
- exit_class: cls.class,
1455
- // Recommended cap from MESSAGING_PROTOCOLS.md: last ~4 KB.
1456
- stderr_tail: attempt.stderr.slice(-4000),
1457
- });
1458
- await deps.supabase.rpc("transition_task", {
1459
- p_task_id: taskId,
1460
- p_new_status: "failed",
1461
- });
1462
- return {
1463
- taskId,
1464
- status: "failed",
1465
- phase: "claude_exit",
1466
- exitCode: attempt.exitCode,
1467
- error: attempt.stderr.slice(-200).trim() || cls.detail,
1468
- quarantine_cause: quarantineCause,
1469
- quarantine_consecutive: quarantineConsecutive,
1470
- };
1471
- }
1472
- // Success: expose the final attempt's stdout under the name the
1473
- // downstream report/PR path expects.
1474
- const capturedStdout = attempt.stdout;
1475
- // v0.63 (T-63-1): a clean claude run proves the binary recovered — reset
1476
- // the silent-unavailability streak so the next blip restarts the backoff
1477
- // ladder from the base and the rolling INFRA-alert window starts fresh.
1478
- clearClaudeUnavailableStreak();
1479
- // v0.14-MESSAGING-RUNTIME-WIRE: Claude finished successfully.
1480
- // testing covers the post-exit window where the runner inspects
1481
- // the captured stdout (parses the report, builds the cost event)
1482
- // before publishing. reviewing is emitted just before opening
1483
- // the PR so a subscriber can pre-stage the review surface.
1484
- await postState_("testing", "info", {
1485
- runner_id: deps.session.runnerId,
1486
- exit_code: attempt.exitCode ?? 0,
1487
- });
1488
- // v0.51 T-51-1: rework completion path. Push the fixes to the SAME
1489
- // branch (same PR) — never open a new PR — then re-enqueue a review
1490
- // and flip the origin task back to needs-review so the existing
1491
- // automerge flow re-evaluates. No v0.32-A "PR exists" fallback on
1492
- // push failure here: a rework push that didn't land means the
1493
- // reviewer feedback was NOT addressed, so the task must fail.
1494
- if (reworkMeta) {
1495
- // v0.57 T-57-2: gate conflict-resolution reworks before the push.
1496
- const reworkIntegrity = await runResolutionIntegrityGate();
1497
- if (reworkIntegrity)
1498
- return reworkIntegrity;
1499
- try {
1500
- await git.push(workdir, branch);
1501
- }
1502
- catch (err) {
1503
- const msg = err.message;
1504
- await appendEvent(deps.supabase, taskId, "error", {
1505
- phase: "push",
1506
- error: msg,
1507
- rework: true,
1508
- pr_number: reworkMeta.pr_number,
1509
- });
1510
- await postState_("blocked", "error_context", {
1511
- task_id: taskId,
1512
- phase: "review",
1513
- error_class: "rework_push_failed",
1514
- stderr_tail: msg,
1515
- });
1516
- await deps.supabase.rpc("transition_task", {
1517
- p_task_id: taskId,
1518
- p_new_status: "failed",
1519
- });
1520
- return { taskId, status: "failed", phase: "push", error: msg };
1521
- }
1522
- // Order matters: request the fresh review BEFORE flipping the
1523
- // origin task back to needs-review. The automerge cron acts on the
1524
- // LATEST review_queue row — if the task re-entered candidacy while
1525
- // the stale completed reject was still latest, the cron would
1526
- // re-finalize that reject and burn a rework cycle without any
1527
- // re-review. With the pending row in place first, the cron sees
1528
- // review_pending and waits for the real verdict.
1529
- if (reworkMeta.origin_task_id) {
1530
- const requested = await deps.supabase.rpc("request_review", {
1531
- p_task_id: reworkMeta.origin_task_id,
1532
- p_pr_number: reworkMeta.pr_number,
1533
- });
1534
- if (requested.error) {
1535
- process.stderr.write(`[acc-runner] rework request_review(${reworkMeta.origin_task_id}) failed: ` +
1536
- `${requested.error.message}\n`);
1537
- }
1538
- const flipped = await deps.supabase.rpc("transition_task", {
1539
- p_task_id: reworkMeta.origin_task_id,
1540
- p_new_status: "needs-review",
1541
- });
1542
- if (flipped.error) {
1543
- // Best-effort: requires migration 0172's widened matrix
1544
- // (changes-requested → needs-review). On failure the pending
1545
- // review row still completes via the runner reviewer; the
1546
- // operator sees the origin task parked at changes-requested.
1547
- process.stderr.write(`[acc-runner] rework transition(${reworkMeta.origin_task_id}→needs-review) failed: ` +
1548
- `${flipped.error.message}\n`);
1549
- }
1550
- }
1551
- else {
1552
- await appendEvent(deps.supabase, taskId, "log", {
1553
- phase: "rework",
1554
- stream: "stderr",
1555
- message: "rework meta has no origin_task_id — re-review not auto-triggered",
1556
- });
1557
- }
1558
- await appendEvent(deps.supabase, taskId, "rework_pushed", {
1559
- pr_number: reworkMeta.pr_number,
1560
- repo: reworkMeta.repo,
1561
- branch,
1562
- cycle: reworkMeta.cycle,
1563
- origin_task_id: reworkMeta.origin_task_id || null,
1564
- });
1565
- await postState_("done", "info", {
1566
- runner_id: deps.session.runnerId,
1567
- pr_number: reworkMeta.pr_number,
1568
- rework: true,
1569
- });
1570
- await deps.supabase.rpc("transition_task", {
1571
- p_task_id: taskId,
1572
- p_new_status: "done",
1573
- });
1574
- return { taskId, status: "ok", exitCode: 0 };
1575
- }
1576
- // v0.53 T-53-4: the task's worktree now holds the agent's commits.
1577
- // Emit the first-commit SLO signal once, before push, so the funnel
1578
- // sees "coding started" independent of whether the push/PR succeeds.
1579
- await emitFirstCommit(deps.supabase, git, taskId, workdir, resolvedBaseRef, branch);
1580
- // v0.57 T-57-2: gate conflict-resolution tasks before the push so a
1581
- // broken merge never reaches a PR. No-op for every non-conflict task.
1582
- const integrityGate = await runResolutionIntegrityGate();
1583
- if (integrityGate)
1584
- return integrityGate;
1585
- // 5. Push + open PR. Both run from the worktree so the operator's
1586
- // shared clone never has the task branch checked out.
1587
- let prUrl = "";
1588
- try {
1589
- await git.push(workdir, branch);
1590
- }
1591
- catch (err) {
1592
- const msg = err.message;
1593
- // v0.32-A: before failing, check whether the branch was already
1594
- // pushed by a prior run (non-fast-forward error on re-push). If
1595
- // an open PR exists, the branch is already live — treat as success.
1596
- const existingPrUrlOnPushFail = await gh.findOpenPR(workdir, branch).catch(() => null);
1597
- if (existingPrUrlOnPushFail) {
1598
- await appendEvent(deps.supabase, taskId, "log", {
1599
- phase: "push",
1600
- stream: "stdout",
1601
- event: "retry.push_conflict_resolved",
1602
- branch,
1603
- pr_url: existingPrUrlOnPushFail,
1604
- message: "push failed but open PR exists — treating as idempotent success",
1605
- });
1606
- prUrl = existingPrUrlOnPushFail;
1607
- // Fall through to postState_("done") below.
1608
- }
1609
- else {
1610
- await appendEvent(deps.supabase, taskId, "error", { phase: "push", error: msg });
1611
- await postState_("blocked", "error_context", {
1612
- task_id: taskId,
1613
- phase: "review",
1614
- error_class: "git_push_failed",
1615
- stderr_tail: msg,
1616
- });
1617
- await deps.supabase.rpc("transition_task", {
1618
- p_task_id: taskId,
1619
- p_new_status: "failed",
1620
- });
1621
- return { taskId, status: "failed", phase: "push", error: msg };
1622
- }
1623
- }
1624
- await postState_("reviewing", "info", {
1625
- runner_id: deps.session.runnerId,
1626
- branch,
1627
- });
1628
- try {
1629
- const body = extractReportFromOutput(capturedStdout);
1630
- const pr = await gh.openPR(workdir, {
1631
- title: prTitleForTask(task.id, task.title),
1632
- body,
1633
- base: integrationBranch,
1634
- });
1635
- prUrl = pr.url;
1636
- }
1637
- catch (err) {
1638
- const msg = err.message;
1639
- // v0.8.2: gh pr create returns a non-zero exit when a PR already
1640
- // exists for this branch (common on runner-crash + re-pick-up).
1641
- // GitHub's error message embeds the existing PR URL. Extract it
1642
- // and treat the outcome as a successful PR open so set_task_pr can
1643
- // transition the task to needs-review rather than leaving it stuck.
1644
- const prAlreadyExistsMatch = /already exists/i.test(msg) &&
1645
- msg.match(/https:\/\/github\.com\/[^\s]+\/pull\/\d+/);
1646
- if (prAlreadyExistsMatch) {
1647
- prUrl = prAlreadyExistsMatch[0];
1648
- try {
1649
- await appendEvent(deps.supabase, taskId, "log", {
1650
- phase: "pr_open",
1651
- stream: "stdout",
1652
- event: "retry.pr_create_already_exists",
1653
- pr_url: prUrl,
1654
- message: "gh pr create reported PR already exists — recovered URL from error message",
1655
- });
1656
- }
1657
- catch (logErr) {
1658
- process.stderr.write(`[acc-runner] pr_open log failed: ${logErr.message}\n`);
1659
- }
1660
- }
1661
- else {
1662
- // Best-effort log; don't transition to failed — the push succeeded
1663
- // and the user can open a PR manually (webhook back-write picks
1664
- // it up). Wrapped so a logging RPC failure can't block the
1665
- // terminal status write below.
1666
- try {
1667
- await appendEvent(deps.supabase, taskId, "error", { phase: "pr_open", error: msg });
1668
- }
1669
- catch (logErr) {
1670
- process.stderr.write(`[acc-runner] pr_open error log failed: ${logErr.message}\n`);
1671
- }
1672
- }
1673
- }
1674
- // v0.35-B: write pr_number + transition running→needs-review via
1675
- // acc.set_task_pr. Best-effort: a missing RPC (Track A migration
1676
- // not yet applied) OR any other error falls back to the
1677
- // webhook-driven running→done path (matrix 0156). Only attempted
1678
- // when a PR URL is in hand — the push-success / PR-open-failure
1679
- // branch above leaves prUrl empty and is handled exclusively by
1680
- // the webhook back-write.
1681
- if (prUrl) {
1682
- await setTaskPr(deps.supabase, taskId, prUrl);
1683
- }
1684
- // v0.33-C: terminal status written before cleanup so a runner crash
1685
- // during cleanup doesn't leave the task non-terminal. Fires the
1686
- // 'done' bus event the instant a PR URL is confirmed (or null when
1687
- // the PR-open RPC failed but the push succeeded — the webhook
1688
- // back-write still finalises the task). v0.14-MESSAGING-RUNTIME-WIRE
1689
- // semantics: protocol='info' today; handoff requires a
1690
- // target_capability lookup the runner doesn't have yet. The 'done'
1691
- // bus state describes the runner's spawn lifecycle (PR opened,
1692
- // runner finished) and is independent of the task status — v0.35-B
1693
- // moves the task itself to needs-review via setTaskPr above.
1694
- await postState_("done", "info", {
1695
- runner_id: deps.session.runnerId,
1696
- pr_url: prUrl || null,
1697
- });
1698
- // v0.33-C: final event log is non-fatal — any throw here would
1699
- // skip the return statement and leave the worktree-cleanup finally
1700
- // running with the wrong outcome, but the terminal bus event above
1701
- // is already on the wire.
1702
- if (prUrl) {
1703
- try {
1704
- await appendEvent(deps.supabase, taskId, "pr-opened", { url: prUrl });
1705
- }
1706
- catch (logErr) {
1707
- process.stderr.write(`[acc-runner] pr-opened event log failed: ${logErr.message}\n`);
1708
- }
1709
- }
1710
- return { taskId, status: "ok", prUrl, exitCode: 0 };
1711
- }
1712
- finally {
1713
- // v0.11-F: tear down the per-task worktree and release the PID
1714
- // lock on every completion path (success, failure, cancellation,
1715
- // unexpected throw). Both legs are best-effort — leaking either
1716
- // resource is preferable to masking the original return value.
1717
- if (worktree) {
1718
- // v0.37-A: clear any stale .git/HEAD.lock or index.lock so the
1719
- // `git worktree remove` + `prune` inside cleanup() don't trip on
1720
- // a leftover lock from the prior worktree add or claude run.
1721
- clearGitLocks(repoPath);
1722
- try {
1723
- await worktree.cleanup();
1724
- }
1725
- catch (err) {
1726
- process.stderr.write(`[acc-runner] worktree cleanup failed: ${err.message}\n`);
1727
- }
1728
- }
1729
- try {
1730
- await lock.release();
1731
- }
1732
- catch (err) {
1733
- process.stderr.write(`[acc-runner] lock release failed: ${err.message}\n`);
1734
- }
1735
- }
1736
- }
1737
- finally {
1738
- // v0.12-RESUME: stop the periodic signal loop before releasing
1739
- // locks so a late-firing signal can't bump the column after the
1740
- // task transitions to a terminal status (the RPC is no-op on
1741
- // non-running rows anyway, but stopping early avoids the extra
1742
- // RPC round-trip).
1743
- stopSignalLoop();
1744
- // v0.53 T-53-4: disarm the per-task GitHub budget so any REST call
1745
- // made outside a task run is unbudgeted.
1746
- endTaskGithubBudget();
1747
- await releaseLocks();
1748
- }
1749
- })();
1750
- return {
1751
- taskId,
1752
- promise,
1753
- cancel() {
1754
- cancelled = true;
1755
- if (child && child.pid) {
1756
- try {
1757
- child.kill("SIGTERM");
1758
- }
1759
- catch {
1760
- // Already dead — ignore.
1761
- }
1762
- }
1763
- },
1764
- };
1765
- }
1766
- //# sourceMappingURL=task-runner.js.map