@tokenfactory/acc-runner 0.27.0 → 0.27.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (161) hide show
  1. package/dist/anthropic-auth.d.ts +53 -0
  2. package/dist/anthropic-auth.d.ts.map +1 -0
  3. package/dist/anthropic-auth.js +89 -0
  4. package/dist/anthropic-auth.js.map +1 -0
  5. package/dist/bin-resolve.d.ts +16 -0
  6. package/dist/bin-resolve.d.ts.map +1 -0
  7. package/dist/bin-resolve.js +35 -0
  8. package/dist/bin-resolve.js.map +1 -0
  9. package/dist/cli.d.ts +3 -0
  10. package/dist/cli.d.ts.map +1 -0
  11. package/dist/cli.js +128 -0
  12. package/dist/cli.js.map +1 -0
  13. package/dist/config.d.ts +121 -0
  14. package/dist/config.d.ts.map +1 -0
  15. package/dist/config.js +396 -0
  16. package/dist/config.js.map +1 -0
  17. package/dist/cost-pricing.d.ts +130 -0
  18. package/dist/cost-pricing.d.ts.map +1 -0
  19. package/dist/cost-pricing.js +191 -0
  20. package/dist/cost-pricing.js.map +1 -0
  21. package/dist/doctor.d.ts +49 -0
  22. package/dist/doctor.d.ts.map +1 -0
  23. package/dist/doctor.js +776 -0
  24. package/dist/doctor.js.map +1 -0
  25. package/dist/failure-classifier.d.ts +112 -0
  26. package/dist/failure-classifier.d.ts.map +1 -0
  27. package/dist/failure-classifier.js +353 -0
  28. package/dist/failure-classifier.js.map +1 -0
  29. package/dist/gh.d.ts +22 -0
  30. package/dist/gh.d.ts.map +1 -0
  31. package/dist/gh.js +48 -0
  32. package/dist/gh.js.map +1 -0
  33. package/dist/git.d.ts +50 -0
  34. package/dist/git.d.ts.map +1 -0
  35. package/dist/git.js +127 -0
  36. package/dist/git.js.map +1 -0
  37. package/dist/github-client.d.ts +133 -0
  38. package/dist/github-client.d.ts.map +1 -0
  39. package/dist/github-client.js +234 -0
  40. package/dist/github-client.js.map +1 -0
  41. package/dist/keychain.d.ts +21 -0
  42. package/dist/keychain.d.ts.map +1 -0
  43. package/dist/keychain.js +45 -0
  44. package/dist/keychain.js.map +1 -0
  45. package/dist/login.d.ts +12 -0
  46. package/dist/login.d.ts.map +1 -0
  47. package/dist/login.js +133 -0
  48. package/dist/login.js.map +1 -0
  49. package/dist/logout.d.ts +2 -0
  50. package/dist/logout.d.ts.map +1 -0
  51. package/dist/logout.js +31 -0
  52. package/dist/logout.js.map +1 -0
  53. package/dist/mcp-spawn.d.ts +30 -0
  54. package/dist/mcp-spawn.d.ts.map +1 -0
  55. package/dist/mcp-spawn.js +145 -0
  56. package/dist/mcp-spawn.js.map +1 -0
  57. package/dist/messaging.d.ts +49 -0
  58. package/dist/messaging.d.ts.map +1 -0
  59. package/dist/messaging.js +36 -0
  60. package/dist/messaging.js.map +1 -0
  61. package/dist/pkg-version.d.ts +3 -0
  62. package/dist/pkg-version.d.ts.map +1 -0
  63. package/dist/pkg-version.js +20 -0
  64. package/dist/pkg-version.js.map +1 -0
  65. package/dist/profiles/designer-prompt.d.ts +18 -0
  66. package/dist/profiles/designer-prompt.d.ts.map +1 -0
  67. package/dist/profiles/designer-prompt.js +172 -0
  68. package/dist/profiles/designer-prompt.js.map +1 -0
  69. package/dist/profiles/developer-prompt.d.ts +24 -0
  70. package/dist/profiles/developer-prompt.d.ts.map +1 -0
  71. package/dist/profiles/developer-prompt.js +24 -0
  72. package/dist/profiles/developer-prompt.js.map +1 -0
  73. package/dist/profiles/manager-prompt.d.ts +34 -0
  74. package/dist/profiles/manager-prompt.d.ts.map +1 -0
  75. package/dist/profiles/manager-prompt.js +93 -0
  76. package/dist/profiles/manager-prompt.js.map +1 -0
  77. package/dist/profiles/tester-prompt.d.ts +28 -0
  78. package/dist/profiles/tester-prompt.d.ts.map +1 -0
  79. package/dist/profiles/tester-prompt.js +165 -0
  80. package/dist/profiles/tester-prompt.js.map +1 -0
  81. package/dist/prompt.d.ts +38 -0
  82. package/dist/prompt.d.ts.map +1 -0
  83. package/dist/prompt.js +78 -0
  84. package/dist/prompt.js.map +1 -0
  85. package/dist/runtime/cache-dir.d.ts +10 -0
  86. package/dist/runtime/cache-dir.d.ts.map +1 -0
  87. package/dist/runtime/cache-dir.js +42 -0
  88. package/dist/runtime/cache-dir.js.map +1 -0
  89. package/dist/runtime/conflict-resolver.d.ts +65 -0
  90. package/dist/runtime/conflict-resolver.d.ts.map +1 -0
  91. package/dist/runtime/conflict-resolver.js +477 -0
  92. package/dist/runtime/conflict-resolver.js.map +1 -0
  93. package/dist/runtime/expand-args.d.ts +28 -0
  94. package/dist/runtime/expand-args.d.ts.map +1 -0
  95. package/dist/runtime/expand-args.js +50 -0
  96. package/dist/runtime/expand-args.js.map +1 -0
  97. package/dist/runtime/locks.d.ts +21 -0
  98. package/dist/runtime/locks.d.ts.map +1 -0
  99. package/dist/runtime/locks.js +103 -0
  100. package/dist/runtime/locks.js.map +1 -0
  101. package/dist/runtime/provision-mutex.d.ts +37 -0
  102. package/dist/runtime/provision-mutex.d.ts.map +1 -0
  103. package/dist/runtime/provision-mutex.js +67 -0
  104. package/dist/runtime/provision-mutex.js.map +1 -0
  105. package/dist/runtime/quarantine.d.ts +30 -0
  106. package/dist/runtime/quarantine.d.ts.map +1 -0
  107. package/dist/runtime/quarantine.js +53 -0
  108. package/dist/runtime/quarantine.js.map +1 -0
  109. package/dist/runtime/resolution-integrity.d.ts +86 -0
  110. package/dist/runtime/resolution-integrity.d.ts.map +1 -0
  111. package/dist/runtime/resolution-integrity.js +248 -0
  112. package/dist/runtime/resolution-integrity.js.map +1 -0
  113. package/dist/runtime/reviewer.d.ts +81 -0
  114. package/dist/runtime/reviewer.d.ts.map +1 -0
  115. package/dist/runtime/reviewer.js +374 -0
  116. package/dist/runtime/reviewer.js.map +1 -0
  117. package/dist/runtime/rework.d.ts +48 -0
  118. package/dist/runtime/rework.d.ts.map +1 -0
  119. package/dist/runtime/rework.js +136 -0
  120. package/dist/runtime/rework.js.map +1 -0
  121. package/dist/runtime/singleton.d.ts +66 -0
  122. package/dist/runtime/singleton.d.ts.map +1 -0
  123. package/dist/runtime/singleton.js +227 -0
  124. package/dist/runtime/singleton.js.map +1 -0
  125. package/dist/runtime/version-drift.d.ts +31 -0
  126. package/dist/runtime/version-drift.d.ts.map +1 -0
  127. package/dist/runtime/version-drift.js +114 -0
  128. package/dist/runtime/version-drift.js.map +1 -0
  129. package/dist/runtime/worktree.d.ts +74 -0
  130. package/dist/runtime/worktree.d.ts.map +1 -0
  131. package/dist/runtime/worktree.js +211 -0
  132. package/dist/runtime/worktree.js.map +1 -0
  133. package/dist/secrets/inject.d.ts +70 -0
  134. package/dist/secrets/inject.d.ts.map +1 -0
  135. package/dist/secrets/inject.js +102 -0
  136. package/dist/secrets/inject.js.map +1 -0
  137. package/dist/supabase.d.ts +4 -0
  138. package/dist/supabase.d.ts.map +1 -0
  139. package/dist/supabase.js +36 -0
  140. package/dist/supabase.js.map +1 -0
  141. package/dist/task-runner.d.ts +313 -0
  142. package/dist/task-runner.d.ts.map +1 -0
  143. package/dist/task-runner.js +1766 -0
  144. package/dist/task-runner.js.map +1 -0
  145. package/dist/token-provider.d.ts +50 -0
  146. package/dist/token-provider.d.ts.map +1 -0
  147. package/dist/token-provider.js +177 -0
  148. package/dist/token-provider.js.map +1 -0
  149. package/dist/types.d.ts +120 -0
  150. package/dist/types.d.ts.map +1 -0
  151. package/dist/types.js +16 -0
  152. package/dist/types.js.map +1 -0
  153. package/dist/version-check.d.ts +17 -0
  154. package/dist/version-check.d.ts.map +1 -0
  155. package/dist/version-check.js +56 -0
  156. package/dist/version-check.js.map +1 -0
  157. package/dist/watch.d.ts +296 -0
  158. package/dist/watch.d.ts.map +1 -0
  159. package/dist/watch.js +1570 -0
  160. package/dist/watch.js.map +1 -0
  161. package/package.json +1 -1
package/dist/watch.js ADDED
@@ -0,0 +1,1570 @@
1
+ /**
2
+ * `acc-runner watch` — long-running command. Subscribes to broadcast
3
+ * tasks, dispatches them concurrently (v0.10+), heartbeats every 4s.
4
+ * v0.2.5 refreshes the access token on startup + every 30 min, and exits
5
+ * loudly after 5 consecutive heartbeat failures.
6
+ *
7
+ * v0.3 adds a 30s polling fallback: if a Realtime broadcast is dropped
8
+ * during a reconnect, the next poll pass picks the task up via the
9
+ * `acc.list_assigned_queued_tasks` RPC. A `seen` set keeps the poll
10
+ * and the broadcast from fighting over the same task_id.
11
+ *
12
+ * v0.6.2 (v0.12-REVIEW-LOCAL) listens for `review_assigned` on the same
13
+ * `runner:<id>` channel and dispatches to runtime/reviewer.ts. Reviews
14
+ * are independently tracked (separate `seenReviews` set).
15
+ *
16
+ * v0.10 (T-49-2): concurrent task dispatch. pump() now launches up to
17
+ * `concurrencyLimit` (default 2) tasks simultaneously. Each task runs in
18
+ * its own isolated git worktree so concurrent tasks never share files.
19
+ * `createTaskAsAdmin` in api/_lib/supabase-admin.ts is the sanctioned
20
+ * Manager seed path that feeds this dispatch loop.
21
+ *
22
+ * v0.12.0 (T-52-7): ephemeral-runner support. Sessions load through the
23
+ * token-provider abstraction (keychain by default — byte-identical desktop
24
+ * behaviour; ACC_RUNNER_TOKEN_MODE=env for containers). Env-token runners
25
+ * register an acc-eph-<random> identity on boot, and ACC_RUNNER_IDLE_TTL_MIN
26
+ * arms a clean self-termination after n minutes without work.
27
+ */
28
+ import chalk from "chalk";
29
+ import { loadConfig } from "./config.js";
30
+ import { getTokenProvider } from "./token-provider.js";
31
+ import { machineIdentity, detectInstalledModels, RUNNER_CAPS } from "./login.js";
32
+ import { createRunnerClient } from "./supabase.js";
33
+ import { runTask } from "./task-runner.js";
34
+ import { runReview, } from "./runtime/reviewer.js";
35
+ import { getQuarantine, setQuarantine, } from "./runtime/quarantine.js";
36
+ import { acquireSingletonLock, singletonRunnerId, SingletonLockHeldError, } from "./runtime/singleton.js";
37
+ import { getClaudeVersion, recordTaskClaudeVersion, } from "./runtime/version-drift.js";
38
+ import { postRunnerStateMessage } from "./messaging.js";
39
+ import { checkVersion, compareSemver } from "./version-check.js";
40
+ import { PROTOCOL_VERSION } from "./types.js";
41
+ import { PACKAGE_VERSION } from "./pkg-version.js";
42
+ import { resolveAuthMode, describeAuthMode, authPrecedenceHint } from "./anthropic-auth.js";
43
+ export class RefreshReuseDetectedError extends Error {
44
+ constructor(message) {
45
+ super(message);
46
+ this.name = "RefreshReuseDetectedError";
47
+ }
48
+ }
49
+ const ONE_HOUR_MS = 60 * 60 * 1000;
50
+ const REFRESH_CHECK_MS = 30 * 60 * 1000;
51
+ // v0.12.0 (T-52-7): idle-TTL check cadence ceiling. The actual cadence is
52
+ // min(IDLE_CHECK_MS, ttl) so a tiny test TTL still gets checked in time.
53
+ const IDLE_CHECK_MS = 30_000;
54
+ const HEARTBEAT_REFRESH_AFTER = 3;
55
+ const HEARTBEAT_EXIT_AFTER = 5;
56
+ const POLL_INTERVAL_MS = 30_000;
57
+ // v0.30-B: widen the startup `pollSince` watermark so a task dispatched in
58
+ // the minutes before `watch` started is still caught by the first poll.
59
+ // The 30s window the previous code used was tighter than the gap between
60
+ // dispatch and runner restart in real operator workflows.
61
+ const STARTUP_POLL_LOOKBACK_MS = 5 * 60 * 1000;
62
+ // v0.62 (T-62-1): cadence of the periodic full claim scan — the realtime-miss
63
+ // safety net. The delta-poll's forward-only watermark can permanently skip a
64
+ // task assigned with an `updated_at` at/behind the watermark whose realtime
65
+ // `task_assigned` push was dropped (stale subscription). This full scan
66
+ // (epoch watermark, every assigned+queued task for this runner) re-claims any
67
+ // such task within one interval, so an alive-but-wedged runner self-recovers
68
+ // instead of needing a manual restart. Knob: `claim_scan_interval_ms`.
69
+ const DEFAULT_CLAIM_SCAN_INTERVAL_MS = 3 * 60 * 1000;
70
+ // v0.73 (T-73-2, G-M): claim-loop watchdog. A task can sit assigned+queued for
71
+ // this runner (acc.tasks.runner_id = me, status='queued') while the runner
72
+ // heartbeats fine but never claims it — the channel looks healthy (so T-68-1's
73
+ // auto-recovery never fires) but the claim path is wedged. CLAIM_WATCHDOG_STALE_MS
74
+ // is how long a task may stay assigned-but-unclaimed (with a live heartbeat)
75
+ // before the watchdog treats the claim path as wedged and forces the same
76
+ // recovery a terminal channel status does. Default 90s — comfortably under the
77
+ // server's 300s rescue grace, so we self-heal before the rescue churns rather
78
+ // than waiting for a manual `pkill acc-runner watch`. CLAIM_WATCHDOG_CHECK_MS is
79
+ // how often the watchdog re-evaluates; well under the threshold so detection +
80
+ // recovery land inside the grace window. Knobs: tests override both for speed.
81
+ const DEFAULT_CLAIM_WATCHDOG_STALE_MS = 90_000;
82
+ const DEFAULT_CLAIM_WATCHDOG_CHECK_MS = 30_000;
83
+ // v0.56 (T-56-1): capacity-aware pause. When claude reports a session/usage
84
+ // limit (api_error_status 429) the runner pauses claiming tasks AND reviews
85
+ // until the stated reset time. When no reset time is parseable, fall back to
86
+ // this fixed backoff. Resume fires at the target + a small random jitter so a
87
+ // fleet that all hit the limit on the same broadcast doesn't thunder back in
88
+ // lockstep the instant the window reopens.
89
+ const DEFAULT_CAPACITY_BACKOFF_MS = 30 * 60 * 1000;
90
+ const DEFAULT_RESUME_JITTER_MS = 15_000;
91
+ /**
92
+ * v0.6.0: exponential backoff between heartbeat attempts when the
93
+ * previous one failed. The dogfood loop fired a heartbeat every 4s
94
+ * even when the network was wedged, producing a wall of "fetch failed"
95
+ * stderr noise and reaching the EXIT_AFTER cap in 20s. Spreading
96
+ * attempts out gives flaky links time to recover before we tear down
97
+ * the watch process.
98
+ */
99
+ const HEARTBEAT_BACKOFF_MS = [4_000, 8_000, 16_000, 32_000, 60_000];
100
+ /**
101
+ * v0.6.0: classify the heartbeat RPC error so we (a) emit a useful
102
+ * stderr line for the operator and (b) only attempt a JWT refresh
103
+ * when the failure shape actually implies an auth problem. Refreshing
104
+ * on a transient network blip just burns the refresh-token rotation
105
+ * chain and surfaces a confusing "refresh failed" line on top of the
106
+ * real underlying network error.
107
+ */
108
+ export function classifyHeartbeatError(err) {
109
+ if (!err)
110
+ return "unknown";
111
+ const msg = (err.message ?? "").toLowerCase();
112
+ const code = (err.code ?? "").toLowerCase();
113
+ if (code === "pgrst301" ||
114
+ code === "401" ||
115
+ code === "403" ||
116
+ msg.includes("jwt") ||
117
+ msg.includes("invalid token") ||
118
+ msg.includes("token has expired") ||
119
+ msg.includes("unauthorized") ||
120
+ msg.includes("forbidden")) {
121
+ return "auth";
122
+ }
123
+ if (msg.includes("enotfound") || msg.includes("getaddrinfo") || msg.includes("eai_again")) {
124
+ return "dns";
125
+ }
126
+ if (msg.includes("econnreset") || msg.includes("connection reset")) {
127
+ return "reset";
128
+ }
129
+ if (msg.includes("certificate") ||
130
+ msg.includes("self-signed") ||
131
+ msg.includes("self signed") ||
132
+ msg.includes("tls") ||
133
+ msg.includes("ssl")) {
134
+ return "tls";
135
+ }
136
+ if (msg.includes("etimedout") || msg.includes("timeout")) {
137
+ return "timeout";
138
+ }
139
+ if (msg.includes("fetch failed") ||
140
+ msg.includes("econnrefused") ||
141
+ msg.includes("network") ||
142
+ msg.includes("socket hang up")) {
143
+ return "network";
144
+ }
145
+ return "unknown";
146
+ }
147
+ /** Pick the next inter-heartbeat delay from the backoff ladder. Caller
148
+ * passes the number of consecutive failures (1 → first retry). The
149
+ * return is at least baseMs so a tighter operator-configured cadence
150
+ * is never slowed by the backoff scheme. */
151
+ export function nextHeartbeatDelayMs(consecutiveFailures, baseMs) {
152
+ if (consecutiveFailures <= 0)
153
+ return baseMs;
154
+ const idx = Math.min(consecutiveFailures - 1, HEARTBEAT_BACKOFF_MS.length - 1);
155
+ return Math.max(baseMs, HEARTBEAT_BACKOFF_MS[idx]);
156
+ }
157
+ /**
158
+ * v0.12.0 (T-52-7): resolve ACC_RUNNER_IDLE_TTL_MIN into ms. Returns null
159
+ * (TTL disabled) when unset; warns and returns null on junk so a typo'd
160
+ * env never strands a container in "never exits" mode silently.
161
+ */
162
+ export function resolveIdleTtlMs(env = process.env) {
163
+ const raw = env.ACC_RUNNER_IDLE_TTL_MIN?.trim();
164
+ if (!raw)
165
+ return null;
166
+ const minutes = Number(raw);
167
+ if (!Number.isFinite(minutes) || minutes <= 0) {
168
+ process.stderr.write(`[acc-runner] ACC_RUNNER_IDLE_TTL_MIN=${JSON.stringify(raw)} is not a positive number; idle TTL disabled\n`);
169
+ return null;
170
+ }
171
+ return minutes * 60_000;
172
+ }
173
+ /** v0.12.0 (T-52-7): record work-related activity for the idle TTL clock. */
174
+ function touchActivity(state) {
175
+ state.lastActivityMs = Date.now();
176
+ }
177
+ function expiresSoon(session, leadMs = ONE_HOUR_MS) {
178
+ const expires = Date.parse(session.access_expires_at);
179
+ if (!Number.isFinite(expires))
180
+ return true;
181
+ return expires < Date.now() + leadMs;
182
+ }
183
+ async function refreshAccessToken(cfg, session) {
184
+ const res = await fetch(`${cfg.publicUrl.replace(/\/+$/, "")}/api/runner/refresh`, {
185
+ method: "POST",
186
+ headers: {
187
+ "Content-Type": "application/json",
188
+ "User-Agent": `acc-runner/${PROTOCOL_VERSION}`,
189
+ },
190
+ body: JSON.stringify({ refresh_token: session.refresh_token }),
191
+ });
192
+ if (res.status === 401) {
193
+ const body = (await res.json().catch(() => ({})));
194
+ if (body.error === "refresh_reuse_detected") {
195
+ throw new RefreshReuseDetectedError(body.hint ??
196
+ "refresh token reuse detected; keychain may be compromised. Run `acc-runner login` from a trusted machine.");
197
+ }
198
+ throw new Error(body.hint ?? "refresh token expired; run `acc-runner login`");
199
+ }
200
+ if (!res.ok) {
201
+ throw new Error(`refresh failed: ${res.status} ${await res.text()}`);
202
+ }
203
+ const body = (await res.json());
204
+ return {
205
+ ...session,
206
+ access_token: body.access_token,
207
+ access_expires_at: new Date(body.expires_at * 1000).toISOString(),
208
+ // v0.4-E: server rotates on every refresh. Persist the new refresh token
209
+ // so the next call doesn't replay the old one (which would now look like
210
+ // reuse and trip the chain-revocation path).
211
+ refresh_token: body.refresh_token ?? session.refresh_token,
212
+ };
213
+ }
214
+ // Centralised handler for the reuse-detected case. Loud red banner so the
215
+ // operator notices immediately, then exit(1) — there's no safe recovery
216
+ // from the watch loop, the keychain has to be reset by `acc-runner login`.
217
+ function failOnReuse(err) {
218
+ process.stderr.write(chalk.red.bold("\n[acc-runner] CRITICAL: refresh-token reuse detected.\n"));
219
+ process.stderr.write(chalk.red(`${err.message}\n`));
220
+ process.stderr.write(chalk.red("If this machine is the legitimate owner, an attacker may have rotated your tokens.\n" +
221
+ "Rotate any other credentials that were on this box and run `acc-runner login` from a trusted machine.\n"));
222
+ process.exit(1);
223
+ }
224
+ // Single source of truth for "this task is new — push it into the runner
225
+ // queue". Both the Realtime listener and the polling loop go through here.
226
+ function enqueue(state, taskId, factory) {
227
+ if (state.seen.has(taskId))
228
+ return false;
229
+ // v0.10 T-49-2: check the running Map (replaces v0.9 state.current check).
230
+ if (state.running.has(taskId)) {
231
+ state.seen.add(taskId);
232
+ return false;
233
+ }
234
+ state.seen.add(taskId);
235
+ state.queue.push(taskId);
236
+ touchActivity(state);
237
+ void pump(state, factory);
238
+ return true;
239
+ }
240
+ // Single source of truth for "this review is new — push it into the review
241
+ // queue". The Realtime listener AND the periodic review scan (v0.65 T-65-2)
242
+ // both go through here; the shared `seenReviews` guard dedupes a review that
243
+ // arrives on both paths. Returns true only when the review was actually
244
+ // enqueued (new + well-formed), so the scan can count what it recovered.
245
+ function enqueueReview(state, payload) {
246
+ const p = payload;
247
+ if (!p?.review_id || !p?.task_id || !p?.pr_number)
248
+ return false;
249
+ if (state.seenReviews.has(p.review_id))
250
+ return false;
251
+ state.seenReviews.add(p.review_id);
252
+ state.reviewQueue.push({
253
+ review_id: p.review_id,
254
+ task_id: p.task_id,
255
+ pr_number: p.pr_number,
256
+ });
257
+ touchActivity(state);
258
+ void pumpReviews(state);
259
+ return true;
260
+ }
261
+ function subscribeChannel(state, taskRunnerFactory) {
262
+ state.channel = state.supabase
263
+ .channel(`runner:${state.session.runner_id}`, {
264
+ config: { broadcast: { ack: false, self: false } },
265
+ })
266
+ .on("broadcast", { event: "task_assigned" }, ({ payload }) => {
267
+ const taskId = payload?.task_id;
268
+ if (taskId)
269
+ enqueue(state, taskId, taskRunnerFactory);
270
+ })
271
+ .on("broadcast", { event: "task_cancelled" }, ({ payload }) => {
272
+ const taskId = payload?.task_id;
273
+ if (!taskId)
274
+ return;
275
+ // v0.10 T-49-2: cancel from the running Map; fall back to queue removal.
276
+ const ctrl = state.running.get(taskId);
277
+ if (ctrl) {
278
+ ctrl.cancel();
279
+ }
280
+ else {
281
+ state.queue = state.queue.filter((t) => t !== taskId);
282
+ }
283
+ // Drop from seen so a re-queue (cancel → re-start) is honored on
284
+ // the next broadcast or poll.
285
+ state.seen.delete(taskId);
286
+ })
287
+ .on("broadcast", { event: "review_assigned" }, ({ payload }) => {
288
+ enqueueReview(state, payload);
289
+ })
290
+ .subscribe((status) => {
291
+ if (status === "SUBSCRIBED") {
292
+ // v0.68 (T-68-1): healthy subscribe — reset the resubscribe backoff.
293
+ state.resubscribeAttempts = 0;
294
+ console.log(chalk.green(`✓ Listening on runner:${state.session.runner_id}`));
295
+ }
296
+ else if (status === "CHANNEL_ERROR" ||
297
+ status === "TIMED_OUT" ||
298
+ status === "CLOSED") {
299
+ // v0.68 (T-68-1): the realtime channel dropped. Previously unhandled —
300
+ // which left the runner silently deaf to task_assigned / review_assigned
301
+ // broadcasts until token expiry or a manual restart (the stale-
302
+ // subscription root cause behind the repeated "task stuck in queued →
303
+ // swept to blocked" incidents). Recover it.
304
+ if (!state.stopped)
305
+ scheduleResubscribe(state, taskRunnerFactory, status);
306
+ }
307
+ });
308
+ }
309
+ // v0.68 (T-68-1): single-flight, backed-off recovery of a dropped realtime
310
+ // channel. Like doRefresh's remove+resubscribe, but WITHOUT a token rotation
311
+ // (the token is still valid; only the channel died) and driven by channel
312
+ // status rather than token expiry. Backoff 1s→30s, reset to 0 on a healthy
313
+ // SUBSCRIBED. After a fresh subscribe we run the poll + claim/review scans so
314
+ // anything the dead channel missed is recovered at once rather than waiting
315
+ // for the periodic 3-min scans.
316
+ function scheduleResubscribe(state, factory, status) {
317
+ if (state.stopped || state.resubscribing || state.resubscribeTimer)
318
+ return;
319
+ const delay = Math.min(30_000, 1_000 * 2 ** Math.min(state.resubscribeAttempts, 5));
320
+ process.stderr.write(`[acc-runner] realtime channel ${status}; resubscribing in ${delay}ms ` +
321
+ `(attempt ${state.resubscribeAttempts + 1})\n`);
322
+ state.resubscribeTimer = setTimeout(() => {
323
+ state.resubscribeTimer = null;
324
+ void resubscribeChannel(state, factory);
325
+ }, delay);
326
+ if (state.resubscribeTimer.unref)
327
+ state.resubscribeTimer.unref();
328
+ }
329
+ export async function resubscribeChannel(state, factory) {
330
+ if (state.stopped || state.resubscribing)
331
+ return;
332
+ state.resubscribing = true;
333
+ state.resubscribeAttempts += 1;
334
+ try {
335
+ try {
336
+ await state.supabase.removeChannel(state.channel);
337
+ }
338
+ catch {
339
+ /* channel may already be torn down */
340
+ }
341
+ subscribeChannel(state, factory);
342
+ void reassertRunningClaims(state);
343
+ // Recover anything the dead channel missed during the gap. Best-effort;
344
+ // the `seen` / `seenReviews` guards dedupe against a late re-delivery.
345
+ void pollOnce(state, factory);
346
+ void claimScan(state, factory, "periodic");
347
+ void reviewScan(state, "periodic");
348
+ }
349
+ finally {
350
+ state.resubscribing = false;
351
+ }
352
+ }
353
+ // A reconnect-triggering network blip also stalls the per-task
354
+ // update_task_signal loop, so last_runner_signal_at drifts stale and the 5-min
355
+ // server sweep can reclaim a task this runner is actively executing. Bump the
356
+ // signal for every in-flight claim the moment the channel is back. Best-effort.
357
+ export async function reassertRunningClaims(state) {
358
+ const taskIds = [...state.running.keys()];
359
+ if (taskIds.length === 0)
360
+ return;
361
+ await Promise.allSettled(taskIds.map(async (taskId) => {
362
+ try {
363
+ const { error } = await state.supabase.rpc("update_task_signal", {
364
+ p_task_id: taskId,
365
+ });
366
+ if (error) {
367
+ process.stderr.write(`[acc-runner] reconnect re-assert update_task_signal(${taskId}) failed: ${error.message}\n`);
368
+ }
369
+ }
370
+ catch (err) {
371
+ process.stderr.write(`[acc-runner] reconnect re-assert update_task_signal(${taskId}) ${err.message}\n`);
372
+ }
373
+ }));
374
+ }
375
+ // Single-flight poll. The watermark advances only after a clean response so
376
+ // a transient error doesn't permanently skip rows.
377
+ async function pollOnce(state, factory) {
378
+ if (state.stopped || state.polling)
379
+ return;
380
+ state.polling = true;
381
+ try {
382
+ const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
383
+ p_runner_id: state.session.runner_id,
384
+ p_since: state.pollSince,
385
+ });
386
+ if (error) {
387
+ process.stderr.write(`[acc-runner] poll failed: ${error.message}\n`);
388
+ return;
389
+ }
390
+ const rows = (data ?? []);
391
+ let maxSeen = state.pollSince;
392
+ for (const row of rows) {
393
+ if (!row?.id)
394
+ continue;
395
+ // v0.62 (T-62-1): advance the forward-only watermark ONLY for rows we
396
+ // actually claimed this pass (enqueue returned true). Previously the
397
+ // watermark advanced to max(updated_at) over EVERY returned row,
398
+ // including no-ops (a task already seen/in-flight via realtime or a
399
+ // prior poll). A high-updated_at no-op could then drag the watermark
400
+ // past a still-queued task assigned with an older updated_at, so the
401
+ // next delta-poll (`> watermark`) silently dropped it — only a restart
402
+ // (one-shot startup scan) recovered it. Advancing solely on real claims
403
+ // keeps the watermark a conservative "newest task I claimed by polling"
404
+ // mark and stops the delta-poll losing still-queued work. The periodic
405
+ // full claim scan (claimScan) is the belt to this braces.
406
+ const claimed = enqueue(state, row.id, factory);
407
+ if (claimed && row.updated_at && row.updated_at > maxSeen) {
408
+ maxSeen = row.updated_at;
409
+ }
410
+ }
411
+ state.pollSince = maxSeen;
412
+ }
413
+ finally {
414
+ state.polling = false;
415
+ }
416
+ }
417
+ /**
418
+ * v0.62 (T-62-1): full claim scan — lists EVERY assigned+queued task for this
419
+ * runner with an epoch watermark (`1970-…Z`) and enqueues any the in-process
420
+ * `seen` guard hasn't already accounted for. Used two ways:
421
+ *
422
+ * - once at startup (replaces the v0.34-B inline block), and
423
+ * - on a periodic timer (`claim_scan_interval_ms`, default 3 min) as the
424
+ * realtime-miss safety net: a runner that dropped a `task_assigned`
425
+ * broadcast self-recovers within one interval instead of sitting next to
426
+ * a free runner until a manual restart.
427
+ *
428
+ * Deliberately does NOT touch `state.pollSince` — the delta-poll keeps its own
429
+ * forward-only watermark. Idempotent: `enqueue`'s shared `seen`/`running`
430
+ * guards mean a task already taken by realtime or the delta-poll is a no-op
431
+ * here, so realtime + delta-poll + scan all seeing the same task still yields
432
+ * exactly one run. Single-flight via `claimScanning` so a slow RPC can't let
433
+ * two scans overlap.
434
+ */
435
+ async function claimScan(state, factory, context) {
436
+ if (state.stopped || state.claimScanning)
437
+ return;
438
+ state.claimScanning = true;
439
+ try {
440
+ const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
441
+ p_runner_id: state.session.runner_id,
442
+ p_since: "1970-01-01T00:00:00.000Z",
443
+ });
444
+ if (error) {
445
+ process.stderr.write(`[acc-runner] ${context} claim scan failed: ${error.message}\n`);
446
+ return;
447
+ }
448
+ const rows = (data ?? []);
449
+ let enqueued = 0;
450
+ for (const row of rows) {
451
+ if (!row?.id)
452
+ continue;
453
+ if (enqueue(state, row.id, factory))
454
+ enqueued += 1;
455
+ }
456
+ if (context === "startup") {
457
+ console.log(`[acc-runner] startup claim scan: ${rows.length} task(s) found`);
458
+ }
459
+ else if (enqueued > 0) {
460
+ // Only chirp when the safety net actually caught something — a periodic
461
+ // scan that finds nothing is the steady state and must stay quiet.
462
+ console.log(`[acc-runner] periodic claim scan recovered ${enqueued} realtime-missed task(s)`);
463
+ }
464
+ }
465
+ catch (err) {
466
+ process.stderr.write(`[acc-runner] ${context} claim scan failed: ${err.message}\n`);
467
+ }
468
+ finally {
469
+ state.claimScanning = false;
470
+ }
471
+ }
472
+ /**
473
+ * v0.73 (T-73-2, G-M): claim-loop watchdog — the self-heal for an
474
+ * "assigned-but-unclaimed while heartbeating" strand.
475
+ *
476
+ * SYMPTOM (prod 2026-06-10): a runner heartbeated fine (status online, fresh
477
+ * heartbeat, idle) but never claimed a task the dispatcher had ASSIGNED to it
478
+ * (acc.tasks.runner_id = me, status='queued'). The server rescue sweep
479
+ * re-dispatched every ~6 min but the runner never claimed; only a manual
480
+ * `pkill acc-runner watch && acc-runner watch` cleared it. T-68-1's channel
481
+ * auto-recovery did NOT catch it — the channel looked healthy but the claim
482
+ * path was wedged.
483
+ *
484
+ * Mechanism: each tick lists the runner's own assigned+queued (= unclaimed)
485
+ * tasks via the same `list_assigned_queued_tasks` RPC the claim scan uses, and
486
+ * tracks per task id the first instant it was OBSERVED as
487
+ * assigned-to-me-but-not-claimed. A task that stays unclaimed past
488
+ * `claimWatchdogStaleMs` (default 90s — comfortably under the server's 300s
489
+ * rescue grace) WHILE heartbeats are succeeding means the claim path is wedged:
490
+ * trigger the SAME single-flight recovery a terminal channel status does
491
+ * (resubscribeChannel: removeChannel + subscribeChannel + pollOnce + claimScan
492
+ * + reviewScan), logged as `claim_watchdog_resubscribe`.
493
+ *
494
+ * Reset semantics: a task's observed-stale timer is dropped the moment it
495
+ * leaves the assigned+queued set (claimed → status moved off 'queued', or
496
+ * unassigned → runner_id cleared). A heartbeat failure clears ALL timers — a
497
+ * network/auth outage is a different failure with its own handling, and we must
498
+ * not carry a stale age across it into a false trigger when heartbeats recover.
499
+ *
500
+ * Single-flight: respects the existing resubscribe guard/backoff
501
+ * (`resubscribing` / `resubscribeTimer`) so concurrent staleness can't stack
502
+ * resubscribes, and resets the stale timers on trigger so a recovery that
503
+ * didn't help waits another full threshold before re-firing. Bounded and
504
+ * side-effect-safe: never throws out of the loop.
505
+ */
506
+ async function claimWatchdogScan(state, factory) {
507
+ if (state.stopped || state.claimWatchdogChecking)
508
+ return;
509
+ state.claimWatchdogChecking = true;
510
+ try {
511
+ // Only treat "unclaimed" as wedged while heartbeats are healthy. A
512
+ // heartbeat failure means the network/auth path is down — the claim RPC
513
+ // would fail too, and the fix there is the heartbeat refresh/exit path, not
514
+ // a resubscribe. Clear the observed-stale timers so an age never spans an
515
+ // outage and falsely trips the watchdog the instant heartbeats recover.
516
+ if (state.heartbeatFailures > 0) {
517
+ state.claimWatchdogSince.clear();
518
+ return;
519
+ }
520
+ const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
521
+ p_runner_id: state.session.runner_id,
522
+ p_since: "1970-01-01T00:00:00.000Z",
523
+ });
524
+ if (error) {
525
+ process.stderr.write(`[acc-runner] claim watchdog scan failed: ${error.message}\n`);
526
+ return;
527
+ }
528
+ const rows = (data ?? []);
529
+ const now = Date.now();
530
+ const currentIds = new Set();
531
+ for (const row of rows) {
532
+ if (row?.id)
533
+ currentIds.add(row.id);
534
+ }
535
+ // Reset: drop timers for tasks no longer assigned+queued (claimed/unassigned).
536
+ for (const id of [...state.claimWatchdogSince.keys()]) {
537
+ if (!currentIds.has(id))
538
+ state.claimWatchdogSince.delete(id);
539
+ }
540
+ // Observe: stamp the first-seen instant for newly assigned-unclaimed tasks;
541
+ // collect any that have now been stale past the threshold.
542
+ const stale = [];
543
+ for (const id of currentIds) {
544
+ const since = state.claimWatchdogSince.get(id);
545
+ if (since === undefined) {
546
+ state.claimWatchdogSince.set(id, now);
547
+ }
548
+ else if (now - since >= state.claimWatchdogStaleMs) {
549
+ stale.push(id);
550
+ }
551
+ }
552
+ if (stale.length === 0)
553
+ return;
554
+ // Single-flight: never stack on a resubscribe already in flight or
555
+ // scheduled (the existing T-68-1 guard/backoff owns the recovery).
556
+ if (state.resubscribing || state.resubscribeTimer)
557
+ return;
558
+ process.stderr.write(`[acc-runner] claim_watchdog_resubscribe: task(s) ${stale.join(", ")} ` +
559
+ `assigned but unclaimed > ${state.claimWatchdogStaleMs}ms with a live ` +
560
+ `heartbeat; claim path looks wedged — forcing resubscribe + claim scan.\n`);
561
+ // Reset the stale timers so a recovery that didn't help waits another full
562
+ // threshold before re-triggering (backoff alongside the resubscribe guard).
563
+ for (const id of stale)
564
+ state.claimWatchdogSince.set(id, now);
565
+ void (async () => {
566
+ try {
567
+ await state.supabase.rpc("log_activity", {
568
+ p_verb: "runner.claim_watchdog_resubscribe",
569
+ p_target_id: state.session.runner_id,
570
+ p_payload: {
571
+ task_ids: stale,
572
+ stale_ms: state.claimWatchdogStaleMs,
573
+ version: PACKAGE_VERSION,
574
+ },
575
+ p_target_type: "runner",
576
+ });
577
+ }
578
+ catch { /* best-effort breadcrumb */ }
579
+ })();
580
+ void resubscribeChannel(state, factory);
581
+ }
582
+ catch (err) {
583
+ process.stderr.write(`[acc-runner] claim watchdog scan failed: ${err.message}\n`);
584
+ }
585
+ finally {
586
+ state.claimWatchdogChecking = false;
587
+ }
588
+ }
589
+ /**
590
+ * v0.65 (T-65-2): full review scan — the review twin of claimScan. Lists EVERY
591
+ * pending review assigned to this runner with an epoch watermark (`1970-…Z`)
592
+ * via `list_assigned_pending_reviews` and feeds each row into the existing
593
+ * `enqueueReview`, whose shared `seenReviews` guard dedupes a review the
594
+ * realtime `review_assigned` broadcast already delivered. Used two ways:
595
+ *
596
+ * - once at startup (next to the startup claim scan), and
597
+ * - on a periodic timer (`review_scan_interval_ms`, default 3 min) as the
598
+ * realtime-miss safety net: a runner that dropped a `review_assigned`
599
+ * broadcast recovers the review within one interval instead of letting it
600
+ * strand in acc.review_queue (PR #544's review stranded 16h last cycle).
601
+ *
602
+ * Idempotent: `enqueueReview`'s `seenReviews` guard means a review already
603
+ * taken by realtime is a no-op here, so realtime + scan both seeing the same
604
+ * review still yields exactly one review run. Single-flight via
605
+ * `reviewScanning` so a slow RPC can't let two scans overlap. Skipped while
606
+ * paused for capacity — matching `pumpReviews`, the runner deliberately holds
607
+ * reviews until its session window reopens.
608
+ */
609
+ async function reviewScan(state, context) {
610
+ if (state.stopped || state.reviewScanning)
611
+ return;
612
+ // v0.56 (T-56-1) parity with pumpReviews: don't claim reviews while paused
613
+ // for capacity. The scheduled resume re-drives the review pump.
614
+ if (state.pausedCapacity)
615
+ return;
616
+ state.reviewScanning = true;
617
+ try {
618
+ const { data, error } = await state.supabase.rpc("list_assigned_pending_reviews", {
619
+ p_runner_id: state.session.runner_id,
620
+ p_since: "1970-01-01T00:00:00.000Z",
621
+ });
622
+ if (error) {
623
+ process.stderr.write(`[acc-runner] ${context} review scan failed: ${error.message}\n`);
624
+ return;
625
+ }
626
+ const rows = (data ?? []);
627
+ let enqueued = 0;
628
+ for (const row of rows) {
629
+ if (enqueueReview(state, row))
630
+ enqueued += 1;
631
+ }
632
+ if (context === "startup") {
633
+ console.log(`[acc-runner] startup review scan: ${rows.length} pending review(s) found`);
634
+ }
635
+ else if (enqueued > 0) {
636
+ // Only chirp when the safety net actually caught something — a periodic
637
+ // scan that finds nothing is the steady state and must stay quiet.
638
+ console.log(`[acc-runner] periodic review scan recovered ${enqueued} realtime-missed review(s)`);
639
+ }
640
+ }
641
+ catch (err) {
642
+ process.stderr.write(`[acc-runner] ${context} review scan failed: ${err.message}\n`);
643
+ }
644
+ finally {
645
+ state.reviewScanning = false;
646
+ }
647
+ }
648
+ // Single-flight refresh. Recreates the Supabase client (so future RPC +
649
+ // channel auth use the new JWT) and resubscribes the broadcast channel.
650
+ async function doRefresh(state, taskRunnerFactory) {
651
+ // v0.12.0 (T-52-7): an env-token runner without ACC_RUNNER_REFRESH_TOKEN
652
+ // cannot rotate. Refusing here (instead of POSTing an empty token) keeps
653
+ // the failure mode a clear stderr line rather than a server-side 400.
654
+ if (!state.tokenProvider.canRefresh()) {
655
+ process.stderr.write("[acc-runner] token refresh unavailable (no refresh token in env); runner will exit when the access token expires\n");
656
+ return false;
657
+ }
658
+ if (state.refreshing)
659
+ return state.refreshing;
660
+ state.refreshing = (async () => {
661
+ try {
662
+ const updated = await refreshAccessToken(state.cfg, state.session);
663
+ state.session = updated;
664
+ await state.tokenProvider.save(updated);
665
+ try {
666
+ await state.supabase.removeChannel(state.channel);
667
+ }
668
+ catch {
669
+ /* channel may already be torn down */
670
+ }
671
+ state.supabase = createRunnerClient(state.cfg, state.session.access_token);
672
+ subscribeChannel(state, taskRunnerFactory);
673
+ return true;
674
+ }
675
+ catch (err) {
676
+ if (err instanceof RefreshReuseDetectedError)
677
+ failOnReuse(err);
678
+ process.stderr.write(`[acc-runner] refresh failed: ${err.message}\n`);
679
+ return false;
680
+ }
681
+ finally {
682
+ state.refreshing = null;
683
+ }
684
+ })();
685
+ return state.refreshing;
686
+ }
687
+ export async function watchCommand(options = {}) {
688
+ const cfg = loadConfig();
689
+ const exitFn = options.exit ?? ((code) => process.exit(code));
690
+ // v0.14.0 (T-54-2, GA-3): single-instance guard. Acquired BEFORE the
691
+ // session load + realtime connection so a duplicate exits immediately,
692
+ // before it can race the holder on worktree locks or git refs. A second
693
+ // runner for the same repo path is a hard error; a stale lock from a dead
694
+ // pid self-reclaims inside acquireSingletonLock.
695
+ const enforceSingleton = options.enforceSingleton ?? !options.taskRunnerFactory;
696
+ let singletonLock = null;
697
+ if (enforceSingleton) {
698
+ // v1.02-A: the lock is keyed by (repo path + ACC_RUNNER_ID). A second
699
+ // runner with a DISTINCT ACC_RUNNER_ID (and, by default, its own
700
+ // namespaced cache dir) acquires its own lock and coexists; only a true
701
+ // duplicate — same repo, same identity — is rejected here.
702
+ const runnerId = singletonRunnerId();
703
+ try {
704
+ singletonLock = await acquireSingletonLock(cfg.repoPath, process.pid, runnerId);
705
+ }
706
+ catch (err) {
707
+ if (err instanceof SingletonLockHeldError) {
708
+ process.stderr.write(chalk.red.bold("\n[acc-runner] ANOTHER RUNNER IS ALREADY ACTIVE.\n"));
709
+ process.stderr.write(chalk.red(`Repo path ${err.holder.repoPath} is served by pid ${err.holder.pid}` +
710
+ (err.holder.runnerId ? ` as ${err.holder.runnerId}` : "") +
711
+ ` (acc-runner v${err.holder.version}, since ${err.holder.acquiredAt}).\n` +
712
+ "Refusing to start a second runner with the same identity for the " +
713
+ "same repo — concurrent same-identity runners corrupt worktrees and " +
714
+ "race git refs.\n"));
715
+ process.stderr.write(chalk.gray(`To stop the other runner: kill ${err.holder.pid}\n` +
716
+ `If it is already dead, remove ${err.lockFile} (or just retry — ` +
717
+ "stale locks self-reclaim).\n" +
718
+ "To run a SECOND runner on this host, give it a distinct " +
719
+ "ACC_RUNNER_ID (its cache dir auto-isolates).\n"));
720
+ exitFn(1);
721
+ // exitFn is process.exit in production (never returns); in tests it
722
+ // is a stub, so re-throw to halt watchCommand cleanly.
723
+ throw err;
724
+ }
725
+ throw err;
726
+ }
727
+ }
728
+ const tokenProvider = getTokenProvider();
729
+ let session = await tokenProvider.load();
730
+ if (!session) {
731
+ process.stderr.write(tokenProvider.mode === "env"
732
+ ? "No session in env. Set ACC_RUNNER_ACCESS_TOKEN (token mode: env).\n"
733
+ : "Not logged in. Run `acc-runner login`.\n");
734
+ process.exit(1);
735
+ }
736
+ const v = await checkVersion(cfg.publicUrl);
737
+ if (!v.ok && v.reason === "outdated") {
738
+ throw new Error(`acc-runner is below the server's minimum version (${v.serverMin}). Run \`pnpm add -g @tokenfactory/acc-runner@latest\`.`);
739
+ }
740
+ // v0.31-B: warn when running behind current_version (above min but stale).
741
+ // Launchd daemon often runs an old binary silently; this surfaces it at
742
+ // startup before the first heartbeat so the operator can update promptly.
743
+ if (v.ok && v.serverCurrent && compareSemver(PACKAGE_VERSION, v.serverCurrent) < 0) {
744
+ process.stderr.write(`[acc-runner] WARNING: running v${PACKAGE_VERSION}, server current is v${v.serverCurrent}. ` +
745
+ `Run \`npm install -g @tokenfactory/acc-runner@latest\` then reload launchd.\n`);
746
+ }
747
+ // v0.74-B: surface how spawned claude sessions will authenticate. api-key
748
+ // (ANTHROPIC_API_KEY set) bills against the API account with no per-session
749
+ // cap; interactive-session falls back to the operator's OAuth login, which
750
+ // hits 429 session caps during long autonomous runs.
751
+ const authMode = resolveAuthMode();
752
+ console.log(chalk.gray(`Anthropic auth: ${describeAuthMode(authMode)}`));
753
+ // v0.76-A (GA-11): surface the API-key-vs-claude.ai-login precedence hint so
754
+ // a 429-on-a-key-configured-runner is diagnosable from the banner.
755
+ const authHint = authPrecedenceHint();
756
+ if (authHint)
757
+ console.log(chalk.gray(`Anthropic auth: ${authHint}`));
758
+ // Startup refresh: if the access token expires in less than an hour,
759
+ // grab a fresh one before opening the realtime connection.
760
+ // v0.12.0 (T-52-7): skipped when the provider has no refresh token —
761
+ // an env-token runner just runs out its 8h access token (the idle TTL
762
+ // normally exits it long before then).
763
+ if (tokenProvider.canRefresh() && expiresSoon(session)) {
764
+ try {
765
+ session = await refreshAccessToken(cfg, session);
766
+ await tokenProvider.save(session);
767
+ }
768
+ catch (err) {
769
+ if (err instanceof RefreshReuseDetectedError)
770
+ failOnReuse(err);
771
+ process.stderr.write(`Refresh failed; run \`acc-runner login\`. (${err.message})\n`);
772
+ process.exit(1);
773
+ }
774
+ }
775
+ // v0.12.0 (T-52-7): env-token (ephemeral) runners had no `login` flow, so
776
+ // nothing registered an acc.runners row yet. Register on boot under the
777
+ // acc-eph-<random> identity. register_runner may resolve to an existing
778
+ // row (same bound user + machine, e.g. a restarted container with a
779
+ // pinned hostname) — adopt whatever id it returns so the realtime
780
+ // channel, heartbeats, and polling all agree.
781
+ if (tokenProvider.mode === "env") {
782
+ const boot = createRunnerClient(cfg, session.access_token);
783
+ const models = await detectInstalledModels();
784
+ const { data: resolvedId, error: regErr } = await boot.rpc("register_runner", {
785
+ p_id: session.runner_id,
786
+ p_name: session.runner_id,
787
+ p_owner: session.email ?? session.user_id,
788
+ p_machine: machineIdentity(),
789
+ p_models: models,
790
+ p_caps: RUNNER_CAPS,
791
+ p_version: PACKAGE_VERSION,
792
+ });
793
+ if (regErr) {
794
+ process.stderr.write(`[acc-runner] ephemeral register_runner failed: ${regErr.message}\n`);
795
+ process.exit(1);
796
+ }
797
+ if (typeof resolvedId === "string" && resolvedId && resolvedId !== session.runner_id) {
798
+ session = { ...session, runner_id: resolvedId };
799
+ }
800
+ await tokenProvider.save(session);
801
+ console.log(chalk.green(`✓ Registered ephemeral runner: ${session.runner_id}`));
802
+ }
803
+ const taskRunnerFactory = options.taskRunnerFactory ?? runTask;
804
+ const heartbeatMs = options.heartbeatMs ?? 4_000;
805
+ const pollMs = options.pollMs ?? POLL_INTERVAL_MS;
806
+ const reviewerFactory = options.reviewerFactory ?? ((assignment, deps) => runReview(assignment, deps));
807
+ // v0.48: check for an existing quarantine file before opening the
808
+ // realtime connection. If quarantined, emit a loud warning — the
809
+ // runner will not claim tasks until `acc-runner quarantine clear` runs.
810
+ const startupQuarantine = await getQuarantine().catch(() => null);
811
+ if (startupQuarantine) {
812
+ process.stderr.write(chalk.red.bold(`[acc-runner] QUARANTINED (${startupQuarantine.cause}): ${startupQuarantine.detail}\n`));
813
+ process.stderr.write(chalk.red(`[acc-runner] Runner will NOT claim tasks. Resolve the issue then run ` +
814
+ `\`acc-runner quarantine clear\`.\n`));
815
+ }
816
+ // v0.12.0 (T-52-7): idle TTL — option override wins (tests), then env.
817
+ const idleTtlMs = options.idleTtlMs !== undefined ? options.idleTtlMs : resolveIdleTtlMs();
818
+ const state = {
819
+ cfg,
820
+ session,
821
+ tokenProvider,
822
+ supabase: createRunnerClient(cfg, session.access_token),
823
+ channel: undefined,
824
+ heartbeatTimer: undefined,
825
+ refreshTimer: undefined,
826
+ pollTimer: undefined,
827
+ claimScanTimer: undefined, // v0.62 (T-62-1)
828
+ claimScanIntervalMs: options.claimScanIntervalMs ?? DEFAULT_CLAIM_SCAN_INTERVAL_MS,
829
+ claimScanning: false,
830
+ reviewScanTimer: undefined, // v0.65 (T-65-2)
831
+ reviewScanIntervalMs: options.reviewScanIntervalMs ?? DEFAULT_CLAIM_SCAN_INTERVAL_MS,
832
+ reviewScanning: false,
833
+ // v0.73 (T-73-2, G-M): claim-loop watchdog.
834
+ claimWatchdogTimer: undefined,
835
+ claimWatchdogStaleMs: options.claimWatchdogStaleMs ?? DEFAULT_CLAIM_WATCHDOG_STALE_MS,
836
+ claimWatchdogSince: new Map(),
837
+ claimWatchdogChecking: false,
838
+ // v0.68 (T-68-1): realtime channel-health recovery state.
839
+ resubscribing: false,
840
+ resubscribeAttempts: 0,
841
+ resubscribeTimer: null,
842
+ // v0.10 T-49-2: concurrent running Map replaces v0.9 single `current`.
843
+ running: new Map(),
844
+ concurrencyLimit: options.concurrencyLimit ??
845
+ (() => {
846
+ // v1.02 (T-1780800001050101): ACC_RUNNER_CONCURRENCY env override.
847
+ // default 2; clamp to 1..50; fall back to 2 on NaN.
848
+ const parsed = Number.parseInt(process.env.ACC_RUNNER_CONCURRENCY ?? '', 10);
849
+ return Number.isNaN(parsed) ? 2 : Math.min(50, Math.max(1, parsed));
850
+ })(),
851
+ queue: [],
852
+ seen: new Set(),
853
+ pumping: false,
854
+ stopped: false,
855
+ heartbeatFailures: 0,
856
+ refreshing: null,
857
+ pollSince: new Date(Date.now() - STARTUP_POLL_LOOKBACK_MS).toISOString(),
858
+ polling: false,
859
+ reviewQueue: [],
860
+ seenReviews: new Set(),
861
+ reviewPumping: false,
862
+ runReview: reviewerFactory,
863
+ quarantined: startupQuarantine !== null,
864
+ authRetriedTasks: new Set(),
865
+ idleTtlMs,
866
+ lastActivityMs: Date.now(),
867
+ idleTimer: null,
868
+ idleExiting: false,
869
+ taskRunnerFactory,
870
+ pausedCapacity: false,
871
+ pausedUntil: null,
872
+ pauseTimer: null,
873
+ capacityProbe: false,
874
+ capacityBackoffMs: options.capacityBackoffMs ?? DEFAULT_CAPACITY_BACKOFF_MS,
875
+ resumeJitterMs: options.resumeJitterMs ?? DEFAULT_RESUME_JITTER_MS,
876
+ singletonLock,
877
+ };
878
+ subscribeChannel(state, taskRunnerFactory);
879
+ // v0.34-B / v0.62 (T-62-1): one-shot startup claim scan. The 5-minute
880
+ // pollSince watermark (v0.30-B) can miss a task dispatched (runner_id set)
881
+ // more than 5 minutes before the runner started whose realtime task_assigned
882
+ // event was lost during the restart window. claimScan() scans with an epoch
883
+ // watermark so any task currently assigned to this runner_id is enqueued
884
+ // regardless of dispatch age. Does NOT update state.pollSince — the ongoing
885
+ // poll loop keeps its 5-min watermark. The same routine runs periodically
886
+ // below as the realtime-miss safety net.
887
+ await claimScan(state, taskRunnerFactory, "startup");
888
+ // v0.65 (T-65-2): one-shot startup review scan, the review twin of the
889
+ // startup claim scan above. A `review_assigned` broadcast dropped during the
890
+ // restart window would otherwise strand the review in acc.review_queue until
891
+ // a manual restart; the epoch-watermark scan re-enqueues any pending review
892
+ // assigned to this runner regardless of when it was requested. enqueueReview's
893
+ // seenReviews guard keeps it idempotent against the realtime path.
894
+ await reviewScan(state, "startup");
895
+ // v0.30-B: fire an immediate poll right after subscribing so any task
896
+ // dispatched before the runner came online (where the fire-and-forget
897
+ // Realtime broadcast went to a dead channel) is picked up at startup
898
+ // instead of waiting `pollMs` for the first scheduled poll. Awaited so
899
+ // `state.polling` is back to false before the timer-driven polls and
900
+ // any caller-driven `pollOnce()` can run.
901
+ await pollOnce(state, taskRunnerFactory);
902
+ // v0.6.0: self-rescheduling setTimeout (not setInterval) so the
903
+ // backoff ladder in nextHeartbeatDelayMs actually pauses between
904
+ // attempts. setInterval would keep firing every heartbeatMs even
905
+ // while the link is down.
906
+ scheduleHeartbeat(state, taskRunnerFactory, heartbeatMs);
907
+ // v0.12.0 (T-52-7): no refresh timer when the provider can't rotate —
908
+ // doRefresh would just emit the same "unavailable" line every 30 min.
909
+ state.refreshTimer = setInterval(() => {
910
+ if (state.tokenProvider.canRefresh() && expiresSoon(state.session)) {
911
+ void doRefresh(state, taskRunnerFactory);
912
+ }
913
+ }, REFRESH_CHECK_MS);
914
+ state.pollTimer = setInterval(() => {
915
+ void pollOnce(state, taskRunnerFactory);
916
+ }, pollMs);
917
+ // v0.62 (T-62-1): periodic full claim scan — the realtime-miss safety net.
918
+ // Independent of the delta-poll watermark, so a task assigned with an
919
+ // updated_at at/behind the watermark (and whose task_assigned broadcast was
920
+ // dropped) is re-claimed within one interval rather than waiting for a
921
+ // manual restart. Single-flight inside claimScan(); enqueue's seen-guard
922
+ // keeps it from double-claiming work the realtime path or delta-poll took.
923
+ state.claimScanTimer = setInterval(() => {
924
+ void claimScan(state, taskRunnerFactory, "periodic");
925
+ }, state.claimScanIntervalMs);
926
+ // v0.65 (T-65-2): periodic full review scan — the realtime-miss safety net
927
+ // for reviews. Mirrors the claim-scan timer above: a `review_assigned`
928
+ // broadcast missed by a stale subscription is recovered within one interval
929
+ // instead of stranding in acc.review_queue. Single-flight inside
930
+ // reviewScan(); enqueueReview's seenReviews guard prevents double-claiming a
931
+ // review the realtime path already took.
932
+ state.reviewScanTimer = setInterval(() => {
933
+ void reviewScan(state, "periodic");
934
+ }, state.reviewScanIntervalMs);
935
+ // v0.73 (T-73-2, G-M): claim-loop watchdog timer. Independent of the claim
936
+ // scan (whose 3-min cadence is too coarse for a 90s threshold): it ages the
937
+ // runner's own assigned-but-unclaimed tasks and forces a resubscribe + claim
938
+ // scan when one stays unclaimed past the threshold while heartbeats succeed —
939
+ // the self-heal for the "online + heartbeating but never claims" strand that
940
+ // T-68-1's channel recovery can't see (healthy channel, wedged claim path).
941
+ const claimWatchdogCheckMs = options.claimWatchdogCheckMs ?? DEFAULT_CLAIM_WATCHDOG_CHECK_MS;
942
+ state.claimWatchdogTimer = setInterval(() => {
943
+ void claimWatchdogScan(state, taskRunnerFactory);
944
+ }, claimWatchdogCheckMs);
945
+ // v0.13-MULTI-RUNNER: refresh the runner's capability tags so an
946
+ // operator change to ACC_RUNNER_CAPABILITIES (or to ~/.config/acc-
947
+ // runner/config.json) takes effect on the next restart without
948
+ // requiring a full `acc-runner login` re-flow. Default-empty is the
949
+ // backward-compat path for 0.6.3 runners and matches any task with
950
+ // an empty required_capabilities array.
951
+ {
952
+ const { error: capsErr } = await state.supabase.rpc("set_runner_capabilities", {
953
+ p_id: state.session.runner_id,
954
+ p_capabilities: cfg.capabilities ?? [],
955
+ });
956
+ if (capsErr) {
957
+ // Non-fatal: a pre-0114 server (or an Old DB the runner is
958
+ // pointed at during local dev) will 404 this RPC. The watch
959
+ // loop continues so an operator on an older deploy isn't
960
+ // wedged by a missing function.
961
+ process.stderr.write(`[acc-runner] set_runner_capabilities failed (continuing): ${capsErr.message}\n`);
962
+ }
963
+ }
964
+ // Initial heartbeat fires immediately so the runner page reflects "online".
965
+ await state.supabase.rpc("heartbeat_runner", {
966
+ p_id: state.session.runner_id,
967
+ p_version: PACKAGE_VERSION,
968
+ });
969
+ console.log(chalk.gray(`Runner ${state.session.runner_id} is online. Ctrl-C to stop.`));
970
+ const stop = async () => {
971
+ if (state.stopped)
972
+ return;
973
+ state.stopped = true;
974
+ clearTimeout(state.heartbeatTimer);
975
+ clearInterval(state.refreshTimer);
976
+ clearInterval(state.pollTimer);
977
+ clearInterval(state.claimScanTimer); // v0.62 (T-62-1)
978
+ clearInterval(state.reviewScanTimer); // v0.65 (T-65-2)
979
+ clearInterval(state.claimWatchdogTimer); // v0.73 (T-73-2)
980
+ if (state.resubscribeTimer)
981
+ clearTimeout(state.resubscribeTimer); // v0.68 (T-68-1)
982
+ if (state.idleTimer)
983
+ clearInterval(state.idleTimer);
984
+ if (state.pauseTimer)
985
+ clearTimeout(state.pauseTimer); // v0.56 (T-56-1)
986
+ // v0.10 T-49-2: cancel every in-flight task (replaces v0.9 single cancel).
987
+ for (const ctrl of state.running.values()) {
988
+ ctrl.cancel();
989
+ }
990
+ try {
991
+ await state.supabase.removeChannel(state.channel);
992
+ }
993
+ catch { /* ignore */ }
994
+ // v0.12-RUNNER-HEALTH: log shutdown_intent BEFORE flipping status so
995
+ // the audit trail records the intent even if the status RPC fails.
996
+ // The 1-min runner-watchdog cron uses status='online'+stale heartbeat
997
+ // as its crash detector; this event gives operators a timestamped
998
+ // breadcrumb when investigating "why did this runner go offline?"
999
+ try {
1000
+ await state.supabase.rpc("log_activity", {
1001
+ p_verb: "runner.shutdown_intent",
1002
+ p_target_id: state.session.runner_id,
1003
+ p_payload: { version: PACKAGE_VERSION },
1004
+ p_target_type: "runner",
1005
+ });
1006
+ }
1007
+ catch { /* best-effort; network may already be gone */ }
1008
+ try {
1009
+ await state.supabase.rpc("set_runner_status", {
1010
+ p_id: state.session.runner_id,
1011
+ p_status: "offline",
1012
+ });
1013
+ }
1014
+ catch { /* network may already be gone */ }
1015
+ // v0.14.0 (T-54-2): release the single-instance lock last, so the slot
1016
+ // is only freed once this runner has fully deregistered. SIGINT/SIGTERM
1017
+ // and the idle-TTL exit all route through stop(), so all clean shutdown
1018
+ // paths release it. A hard crash leaves a stale lock that the next
1019
+ // start reclaims.
1020
+ try {
1021
+ await state.singletonLock?.release();
1022
+ }
1023
+ catch { /* best-effort */ }
1024
+ };
1025
+ // v0.12.0 (T-52-7): arm the idle-TTL timer. Desktop runners (no TTL env,
1026
+ // no option) never reach this branch — zero behaviour change. Armed after
1027
+ // `stop` exists so the exit sequence can reuse the one shutdown path.
1028
+ if (state.idleTtlMs !== null && state.idleTtlMs > 0) {
1029
+ const checkMs = options.idleCheckMs ?? Math.min(IDLE_CHECK_MS, state.idleTtlMs);
1030
+ state.idleTimer = setInterval(() => {
1031
+ void maybeIdleExit(state, stop, exitFn);
1032
+ }, checkMs);
1033
+ // Don't hold the event loop open for the idle check alone.
1034
+ state.idleTimer.unref?.();
1035
+ console.log(chalk.gray(`Idle TTL armed: exiting after ${Math.round(state.idleTtlMs / 60_000)} min without work.`));
1036
+ }
1037
+ if (!options.taskRunnerFactory) {
1038
+ // v0.33-C: best-effort terminal status flush on runner shutdown. If
1039
+ // launchd (or another OS-driven kill) sends SIGTERM while a task is
1040
+ // mid-run, transition_task('failed') as a last resort so the task
1041
+ // doesn't sit in 'running' until the 5-15 min sweep fires. SIGINT
1042
+ // (Ctrl-C, operator-driven) skips the flush — the sweep is the right
1043
+ // path there because the task should be retried as-is, not failed.
1044
+ // SIGKILL / OOM can't be caught, so the sweep is still the real
1045
+ // safety net; this just shrinks the window when the kill is catchable.
1046
+ const onSignal = (signal) => {
1047
+ void (async () => {
1048
+ // v0.10 T-49-2: on SIGTERM, transition ALL in-flight tasks to
1049
+ // failed (replaces v0.9 single-task transition). SIGINT (Ctrl-C)
1050
+ // still skips the flush — the 5-min sweep handles that path.
1051
+ if (signal === "SIGTERM" && state.running.size > 0) {
1052
+ await Promise.allSettled([...state.running.keys()].map(async (taskId) => {
1053
+ try {
1054
+ await state.supabase.rpc("transition_task", {
1055
+ p_task_id: taskId,
1056
+ p_new_status: "failed",
1057
+ });
1058
+ }
1059
+ catch { /* best-effort — sweep covers us if RPC can't get out */ }
1060
+ }));
1061
+ }
1062
+ await stop();
1063
+ process.exit(0);
1064
+ })();
1065
+ };
1066
+ process.once("SIGINT", () => onSignal("SIGINT"));
1067
+ process.once("SIGTERM", () => onSignal("SIGTERM"));
1068
+ }
1069
+ return {
1070
+ stop,
1071
+ pollOnce: () => pollOnce(state, taskRunnerFactory),
1072
+ claimScanOnce: () => claimScan(state, taskRunnerFactory, "periodic"), // v0.62 (T-62-1)
1073
+ reviewScanOnce: () => reviewScan(state, "periodic"), // v0.65 (T-65-2)
1074
+ claimWatchdogOnce: () => claimWatchdogScan(state, taskRunnerFactory), // v0.73 (T-73-2)
1075
+ runningCount: () => state.running.size,
1076
+ };
1077
+ }
1078
+ function scheduleHeartbeat(state, taskRunnerFactory, baseMs) {
1079
+ if (state.stopped)
1080
+ return;
1081
+ const delay = nextHeartbeatDelayMs(state.heartbeatFailures, baseMs);
1082
+ state.heartbeatTimer = setTimeout(() => {
1083
+ void (async () => {
1084
+ const { error } = await state.supabase.rpc("heartbeat_runner", {
1085
+ p_id: state.session.runner_id,
1086
+ p_version: PACKAGE_VERSION,
1087
+ });
1088
+ if (!error) {
1089
+ state.heartbeatFailures = 0;
1090
+ }
1091
+ else {
1092
+ state.heartbeatFailures += 1;
1093
+ const kind = classifyHeartbeatError(error);
1094
+ process.stderr.write(`[acc-runner] heartbeat failed (${kind}): ${error.message}\n`);
1095
+ // v0.6.0 REG-294 sibling: only refresh on an auth-shaped error
1096
+ // AND once we've crossed the consecutive-failure threshold. A
1097
+ // single 401 on a flaky network shouldn't burn a refresh, and a
1098
+ // DNS outage shouldn't trigger refresh at all.
1099
+ if (state.heartbeatFailures >= HEARTBEAT_REFRESH_AFTER &&
1100
+ kind === "auth") {
1101
+ process.stderr.write("[acc-runner] auth-shaped heartbeat failure, attempting JWT refresh\n");
1102
+ void doRefresh(state, taskRunnerFactory);
1103
+ }
1104
+ if (state.heartbeatFailures >= HEARTBEAT_EXIT_AFTER) {
1105
+ process.stderr.write("[acc-runner] heartbeat dead, exiting\n");
1106
+ clearTimeout(state.heartbeatTimer);
1107
+ clearInterval(state.refreshTimer);
1108
+ clearInterval(state.pollTimer);
1109
+ clearInterval(state.claimScanTimer); // v0.62 (T-62-1)
1110
+ clearInterval(state.reviewScanTimer); // v0.65 (T-65-2)
1111
+ clearInterval(state.claimWatchdogTimer); // v0.73 (T-73-2)
1112
+ if (state.resubscribeTimer)
1113
+ clearTimeout(state.resubscribeTimer); // v0.68 (T-68-1)
1114
+ process.exit(1);
1115
+ }
1116
+ }
1117
+ scheduleHeartbeat(state, taskRunnerFactory, baseMs);
1118
+ })();
1119
+ }, delay);
1120
+ }
1121
+ /**
1122
+ * v0.12.0 (T-52-7): idle-TTL exit check. Fires from the idle timer; exits
1123
+ * the process cleanly when the runner has had no queued or in-flight work
1124
+ * (tasks OR reviews) for idleTtlMs. The exit sequence:
1125
+ *
1126
+ * 1. log_activity runner.idle_ttl_exit — the audit breadcrumb the
1127
+ * autoscale sweep and operators read ("why did acc-eph-x vanish?").
1128
+ * 2. final heartbeat_runner — so the runners panel shows a fresh
1129
+ * timestamp right up to the clean exit (vs. a stale-heartbeat crash).
1130
+ * 3. stop() — shutdown_intent + set_runner_status('offline'), the same
1131
+ * deregistration path `acc-runner logout` and SIGTERM use.
1132
+ * 4. exit(0).
1133
+ *
1134
+ * In-flight work always wins: any running task, queued task, or queued
1135
+ * review resets the decision to "not idle" — the TTL never kills a runner
1136
+ * mid-task.
1137
+ */
1138
+ async function maybeIdleExit(state, stop, exitFn) {
1139
+ if (state.stopped || state.idleExiting || state.idleTtlMs === null)
1140
+ return;
1141
+ // v0.56 (T-56-1): a capacity pause is not idleness — the runner is
1142
+ // deliberately waiting for its session window to reopen. Don't TTL-exit
1143
+ // it out from under the scheduled resume.
1144
+ if (state.pausedCapacity)
1145
+ return;
1146
+ if (state.running.size > 0 ||
1147
+ state.queue.length > 0 ||
1148
+ state.reviewQueue.length > 0 ||
1149
+ state.reviewPumping) {
1150
+ return;
1151
+ }
1152
+ const idleMs = Date.now() - state.lastActivityMs;
1153
+ if (idleMs < state.idleTtlMs)
1154
+ return;
1155
+ state.idleExiting = true;
1156
+ const idleMinutes = Math.round(idleMs / 60_000);
1157
+ console.log(chalk.gray(`[acc-runner] idle TTL reached (${idleMinutes} min without work); exiting cleanly.`));
1158
+ try {
1159
+ await state.supabase.rpc("log_activity", {
1160
+ p_verb: "runner.idle_ttl_exit",
1161
+ p_target_id: state.session.runner_id,
1162
+ p_payload: { idle_minutes: idleMinutes, version: PACKAGE_VERSION },
1163
+ p_target_type: "runner",
1164
+ });
1165
+ }
1166
+ catch { /* best-effort */ }
1167
+ try {
1168
+ await state.supabase.rpc("heartbeat_runner", {
1169
+ p_id: state.session.runner_id,
1170
+ p_version: PACKAGE_VERSION,
1171
+ });
1172
+ }
1173
+ catch { /* best-effort */ }
1174
+ await stop();
1175
+ exitFn(0);
1176
+ }
1177
+ async function pumpReviews(state) {
1178
+ if (state.reviewPumping)
1179
+ return;
1180
+ // v0.56 (T-56-1): don't claim reviews while paused for capacity, and hold
1181
+ // reviews back during the single-probe window so the probe TASK confirms
1182
+ // the session reopened before we spend more capacity on reviews.
1183
+ if (state.pausedCapacity || state.capacityProbe)
1184
+ return;
1185
+ state.reviewPumping = true;
1186
+ try {
1187
+ while (!state.stopped &&
1188
+ !state.pausedCapacity &&
1189
+ !state.capacityProbe &&
1190
+ state.reviewQueue.length > 0) {
1191
+ const next = state.reviewQueue.shift();
1192
+ if (!next)
1193
+ continue;
1194
+ try {
1195
+ const outcome = await state.runReview(next, { supabase: state.supabase });
1196
+ const tag = outcome.decision === "reviewer_error" ? chalk.red : chalk.gray;
1197
+ console.log(tag(`[acc-runner] review ${next.review_id} task=${next.task_id} pr=${next.pr_number} decision=${outcome.decision} confidence=${outcome.confidence.toFixed(2)}`));
1198
+ // v0.56 (T-56-1): a reviewer that hit the session/usage limit reports
1199
+ // a DISTINCT outcome (reviewer_capacity, NOT reviewer_error) so the
1200
+ // api side (T-56-2) can exclude it from automerge retry caps and
1201
+ // re-dispatch the review. runReview has already submitted the
1202
+ // reviewer_capacity row; here we pause the whole runner exactly as a
1203
+ // capacity_exhausted task does.
1204
+ if (outcome.decision === "reviewer_capacity") {
1205
+ enterCapacityPause(state, next.task_id, outcome.resume_at ?? null, `reviewer ${next.review_id} hit session/usage limit`);
1206
+ return;
1207
+ }
1208
+ }
1209
+ catch (err) {
1210
+ process.stderr.write(`[acc-runner] review ${next.review_id} crashed: ${err.message}\n`);
1211
+ }
1212
+ finally {
1213
+ touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at review end
1214
+ }
1215
+ }
1216
+ }
1217
+ finally {
1218
+ state.reviewPumping = false;
1219
+ }
1220
+ }
1221
+ /**
1222
+ * v0.48 / v0.49 T-49-6: enter quarantine after a machine-level failure.
1223
+ *
1224
+ * Extracted from pump()'s completion handler so both the immediate
1225
+ * quarantine path and the auth_expired "refresh failed → quarantine"
1226
+ * path share one implementation. Idempotent: a no-op once the runner is
1227
+ * already quarantined, so concurrent machine-level failures don't fire
1228
+ * duplicate audit events. Sets the in-memory flag synchronously (pump()
1229
+ * reads it without an async hop) and best-effort persists + announces.
1230
+ */
1231
+ function enterQuarantine(state, taskId, cause, detail,
1232
+ // v0.53 T-53-4: how many consecutive instant-empty exits produced an
1233
+ // env_broken cause (1 = definitive, 2 = heuristic confirmed by the
1234
+ // retry). Persisted into quarantine.json's additive `consecutive` field.
1235
+ consecutive) {
1236
+ if (state.quarantined)
1237
+ return;
1238
+ state.quarantined = true;
1239
+ const qState = {
1240
+ cause,
1241
+ classifiedAt: new Date().toISOString(),
1242
+ taskId,
1243
+ detail,
1244
+ ...(consecutive !== undefined ? { consecutive } : {}),
1245
+ };
1246
+ void setQuarantine(qState).catch((err) => {
1247
+ process.stderr.write(`[acc-runner] setQuarantine failed: ${err.message}\n`);
1248
+ });
1249
+ process.stderr.write(chalk.red.bold(`[acc-runner] QUARANTINE: runner ${state.session.runner_id} quarantined ` +
1250
+ `(${cause}) after task ${taskId} failed.\n`));
1251
+ void postRunnerStateMessage(state.supabase, {
1252
+ task_id: taskId,
1253
+ sender_id: state.session.runner_id,
1254
+ state: "blocked",
1255
+ protocol: "error_context",
1256
+ payload: {
1257
+ capacity_alert: true,
1258
+ quarantine_cause: cause,
1259
+ runner_id: state.session.runner_id,
1260
+ detail,
1261
+ },
1262
+ kind: "blocked",
1263
+ }).catch((err) => {
1264
+ process.stderr.write(`[acc-runner] capacity alert post failed: ${err.message}\n`);
1265
+ });
1266
+ void (async () => {
1267
+ try {
1268
+ await state.supabase.rpc("log_activity", {
1269
+ p_verb: "runner.quarantine_enter",
1270
+ p_target_id: state.session.runner_id,
1271
+ p_payload: { cause, task_id: taskId, detail },
1272
+ p_target_type: "runner",
1273
+ });
1274
+ }
1275
+ catch { /* best-effort */ }
1276
+ })();
1277
+ }
1278
+ /**
1279
+ * v0.56 (T-56-1): enter a capacity pause.
1280
+ *
1281
+ * Triggered when a task OR a review reports capacity_exhausted (Claude
1282
+ * account out of session/usage capacity). The runner:
1283
+ * - stops claiming tasks AND reviews (pump/pumpReviews bail on the flag);
1284
+ * - schedules an automatic resume at the stated reset time (or a default
1285
+ * backoff when none was parseable) plus a small jitter;
1286
+ * - keeps heartbeating so the board shows online-but-paused, and emits a
1287
+ * `runner.paused_capacity` activity event + a runner-level bus message
1288
+ * carrying `resume_at` so health/board surfaces can show WHY the fleet
1289
+ * is idle (heartbeat_runner's RPC shape is frozen, so the detail is
1290
+ * carried additively through these existing channels).
1291
+ *
1292
+ * Idempotent: a second capacity hit while already paused only ever pushes
1293
+ * the resume later, never earlier, so a straggler task that 429s a moment
1294
+ * after the first one can't shorten the wait.
1295
+ */
1296
+ function enterCapacityPause(state, triggerTaskId, resumeAtIso, detail) {
1297
+ const now = Date.now();
1298
+ const parsed = resumeAtIso ? Date.parse(resumeAtIso) : NaN;
1299
+ const resumeMs = Number.isFinite(parsed) && parsed > now ? parsed : now + state.capacityBackoffMs;
1300
+ const source = Number.isFinite(parsed) && parsed > now ? "reset_time" : "default_backoff";
1301
+ // Already paused: only extend the window, never shorten it.
1302
+ if (state.pausedCapacity && state.pausedUntil !== null && resumeMs <= state.pausedUntil) {
1303
+ return;
1304
+ }
1305
+ state.pausedCapacity = true;
1306
+ state.pausedUntil = resumeMs;
1307
+ if (state.pauseTimer) {
1308
+ clearTimeout(state.pauseTimer);
1309
+ state.pauseTimer = null;
1310
+ }
1311
+ const jitter = state.resumeJitterMs > 0 ? Math.floor(Math.random() * state.resumeJitterMs) : 0;
1312
+ const delay = Math.max(0, resumeMs - now) + jitter;
1313
+ const resumeAtFinal = new Date(now + delay).toISOString();
1314
+ process.stderr.write(chalk.yellow(`[acc-runner] PAUSED (capacity): ${detail}. Not claiming tasks or reviews; ` +
1315
+ `resuming at ${resumeAtFinal} (${source}).\n`));
1316
+ void (async () => {
1317
+ try {
1318
+ await state.supabase.rpc("log_activity", {
1319
+ p_verb: "runner.paused_capacity",
1320
+ p_target_id: state.session.runner_id,
1321
+ p_payload: {
1322
+ resume_at: resumeAtFinal,
1323
+ source,
1324
+ detail,
1325
+ version: PACKAGE_VERSION,
1326
+ },
1327
+ p_target_type: "runner",
1328
+ });
1329
+ }
1330
+ catch { /* best-effort */ }
1331
+ })();
1332
+ // Runner-level bus message so operators see the paused status + reason on
1333
+ // /messages even though heartbeat_runner itself can't carry a detail field.
1334
+ // Anchored to the triggering task so the RPC can resolve the org scope
1335
+ // (post_agent_message looks up org_id from acc.tasks for a non-null id).
1336
+ void postRunnerStateMessage(state.supabase, {
1337
+ task_id: triggerTaskId,
1338
+ sender_id: state.session.runner_id,
1339
+ state: "blocked",
1340
+ protocol: "info",
1341
+ payload: {
1342
+ capacity_paused: true,
1343
+ runner_id: state.session.runner_id,
1344
+ resume_at: resumeAtFinal,
1345
+ source,
1346
+ detail,
1347
+ },
1348
+ kind: "blocked",
1349
+ }).catch((err) => {
1350
+ process.stderr.write(`[acc-runner] capacity pause bus post failed: ${err.message}\n`);
1351
+ });
1352
+ state.pauseTimer = setTimeout(() => {
1353
+ void resumeFromCapacity(state);
1354
+ }, delay);
1355
+ state.pauseTimer.unref?.();
1356
+ }
1357
+ /**
1358
+ * v0.56 (T-56-1): resume from a capacity pause.
1359
+ *
1360
+ * Clears the pause and arms the single-probe clamp: the runner claims at
1361
+ * most ONE task to confirm the window actually reopened before resuming
1362
+ * full task concurrency and reviews. If the probe itself 429s, the
1363
+ * completion handler re-enters the pause; if it succeeds, the probe clamp
1364
+ * clears and normal claiming resumes.
1365
+ */
1366
+ async function resumeFromCapacity(state) {
1367
+ if (state.stopped)
1368
+ return;
1369
+ state.pausedCapacity = false;
1370
+ state.pausedUntil = null;
1371
+ if (state.pauseTimer) {
1372
+ clearTimeout(state.pauseTimer);
1373
+ state.pauseTimer = null;
1374
+ }
1375
+ state.capacityProbe = true;
1376
+ console.log(chalk.gray(`[acc-runner] capacity window reopened — resuming with a single probe task.`));
1377
+ try {
1378
+ await state.supabase.rpc("log_activity", {
1379
+ p_verb: "runner.resume_capacity",
1380
+ p_target_id: state.session.runner_id,
1381
+ p_payload: { probe: true, version: PACKAGE_VERSION },
1382
+ p_target_type: "runner",
1383
+ });
1384
+ }
1385
+ catch { /* best-effort */ }
1386
+ // Poll first (the stale-running sweep may have re-queued the released task
1387
+ // to another runner; this picks up whatever is still assigned here), then
1388
+ // pump — the probe clamp bounds it to one task.
1389
+ await pollOnce(state, state.taskRunnerFactory);
1390
+ void pump(state, state.taskRunnerFactory);
1391
+ }
1392
+ /**
1393
+ * v0.10 T-49-2: concurrent task pump.
1394
+ *
1395
+ * Launches up to state.concurrencyLimit tasks in parallel. Each task
1396
+ * gets its own isolated git worktree (prepareTaskWorktree in task-runner)
1397
+ * so there is zero shared-file risk between concurrent runs. Database-side
1398
+ * `claim_task_with_locks` provides a second guard: it rejects a claim when
1399
+ * the task's file scope overlaps another in-flight task.
1400
+ *
1401
+ * Design: rather than `await`ing each task inside a while loop (v0.9
1402
+ * serial behaviour), tasks are launched fire-and-forget. Each task's
1403
+ * `.then()` handler removes it from `state.running` and re-fires pump()
1404
+ * so newly freed capacity immediately picks up the next queued item.
1405
+ * The `pumping` mutex prevents two concurrent pump() invocations from
1406
+ * both launching into the same concurrencyLimit slot.
1407
+ */
1408
+ async function pump(state, factory) {
1409
+ if (state.pumping)
1410
+ return;
1411
+ state.pumping = true;
1412
+ try {
1413
+ // v0.48: check quarantine before claiming any task.
1414
+ if (state.quarantined) {
1415
+ process.stderr.write(`[acc-runner] QUARANTINED: runner ${state.session.runner_id} is not claiming tasks.\n` +
1416
+ `[acc-runner] Fix the underlying issue then run \`acc-runner quarantine clear\`.\n`);
1417
+ return;
1418
+ }
1419
+ // v0.56 (T-56-1): out of session/usage capacity — don't claim. Queued
1420
+ // tasks stay queued; the scheduled resume re-drives pump().
1421
+ if (state.pausedCapacity) {
1422
+ return;
1423
+ }
1424
+ // v0.56 (T-56-1): during the post-resume probe window, claim exactly one
1425
+ // task to confirm the session reopened before unleashing full concurrency.
1426
+ const effectiveLimit = state.capacityProbe ? 1 : state.concurrencyLimit;
1427
+ // Fill available concurrency slots from the queue.
1428
+ while (!state.stopped &&
1429
+ !state.quarantined &&
1430
+ !state.pausedCapacity &&
1431
+ state.queue.length > 0 &&
1432
+ state.running.size < effectiveLimit) {
1433
+ const next = state.queue.shift();
1434
+ if (!next)
1435
+ continue;
1436
+ const ctrl = factory(next, {
1437
+ supabase: state.supabase,
1438
+ cfg: state.cfg,
1439
+ session: {
1440
+ accessToken: state.session.access_token,
1441
+ runnerId: state.session.runner_id,
1442
+ },
1443
+ publicUrl: state.cfg.publicUrl,
1444
+ });
1445
+ state.running.set(next, ctrl);
1446
+ // Detached completion handler. Runs AFTER pump() returns so the
1447
+ // running Map slot is occupied by the time the while condition
1448
+ // re-checks running.size on the next iteration.
1449
+ void ctrl.promise
1450
+ .then((outcome) => {
1451
+ state.running.delete(next);
1452
+ touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at task end
1453
+ // v0.56 (T-56-1): capacity exhaustion pauses the WHOLE runner —
1454
+ // never a task_error, never quarantine. The task was left 'running'
1455
+ // for the stale-running sweep to requeue losslessly (runner_id
1456
+ // cleared). Do not re-fire pump for new work; the scheduled resume
1457
+ // does that with a single probe.
1458
+ if (outcome.status === "capacity_paused" || outcome.capacity_exhausted) {
1459
+ enterCapacityPause(state, next, outcome.resume_at ?? null, outcome.error || "task hit session/usage limit");
1460
+ return;
1461
+ }
1462
+ // v0.21 T-66-1: a transient git-provision contention requeue. runTask
1463
+ // already returned the task to 'queued' server-side for a clean
1464
+ // retry — NOT a failure (no quarantine, no pause, no retry-budget
1465
+ // burn). Fall through to the pump re-fire below so the runner stays
1466
+ // healthy and the requeued task re-dispatches normally.
1467
+ if (outcome.status === "requeued") {
1468
+ process.stderr.write(`[acc-runner] task ${next} requeued (transient): ${outcome.error ?? "git provision contention"}\n`);
1469
+ }
1470
+ // v0.14.0 (T-54-2, GA-6): record the claude CLI version at the last
1471
+ // successful task so `doctor` can WARN when claude auto-updates
1472
+ // under the runner. Best-effort and detached — never blocks pump.
1473
+ if (outcome.status === "ok") {
1474
+ void getClaudeVersion()
1475
+ .then((v) => recordTaskClaudeVersion(v))
1476
+ .catch(() => {
1477
+ /* best-effort cache write */
1478
+ });
1479
+ }
1480
+ if (outcome.status === "failed") {
1481
+ const reason = outcome.error?.trim() ||
1482
+ (outcome.exitCode != null ? `exit ${outcome.exitCode}` : "unknown");
1483
+ process.stderr.write(`[acc-runner] task ${next} phase=${outcome.phase ?? "unknown"} failed: ${reason}\n`);
1484
+ // v0.48: machine-level failure → quarantine. Guard with
1485
+ // !state.quarantined so concurrent tasks don't fire duplicate
1486
+ // quarantine events when two machine-level failures land
1487
+ // simultaneously (both already in-flight; only the first to
1488
+ // resolve sets the flag and writes the audit trail).
1489
+ if (outcome.quarantine_cause && !state.quarantined) {
1490
+ const cause = outcome.quarantine_cause;
1491
+ // v0.49 T-49-6: a transient auth_expired (the access token
1492
+ // lapsed while the task ran) gets ONE JWT refresh before we
1493
+ // quarantine. If the refresh recovers, the runner stays
1494
+ // online and resumes claiming queued work; if it fails — or
1495
+ // this task already burned its one retry — we quarantine.
1496
+ // usage_limit / env_broken skip this path: a token refresh
1497
+ // can't fix a spent quota or a broken host.
1498
+ if (cause === "auth_expired" && !state.authRetriedTasks.has(next)) {
1499
+ state.authRetriedTasks.add(next);
1500
+ process.stderr.write(chalk.yellow(`[acc-runner] auth_expired after task ${next}; attempting one ` +
1501
+ `token refresh before quarantine.\n`));
1502
+ void doRefresh(state, factory).then((ok) => {
1503
+ if (ok) {
1504
+ process.stderr.write(chalk.green(`[acc-runner] token refresh recovered auth after task ${next}; ` +
1505
+ `staying online.\n`));
1506
+ void (async () => {
1507
+ try {
1508
+ await state.supabase.rpc("log_activity", {
1509
+ p_verb: "runner.auth_retry_recovered",
1510
+ p_target_id: state.session.runner_id,
1511
+ p_payload: { task_id: next, cause },
1512
+ p_target_type: "runner",
1513
+ });
1514
+ }
1515
+ catch { /* best-effort */ }
1516
+ })();
1517
+ // Resume claiming: the freed slot can pick up queued work
1518
+ // now that the JWT is fresh.
1519
+ void pump(state, factory);
1520
+ }
1521
+ else {
1522
+ enterQuarantine(state, next, cause, reason, outcome.quarantine_consecutive);
1523
+ }
1524
+ });
1525
+ return;
1526
+ }
1527
+ enterQuarantine(state, next, cause, reason, outcome.quarantine_consecutive);
1528
+ // Quarantined — do not re-fire pump for new tasks.
1529
+ // Other in-flight tasks (still in state.running) complete
1530
+ // normally; we just stop accepting new work.
1531
+ return;
1532
+ }
1533
+ }
1534
+ // v0.56 (T-56-1): a non-capacity completion (ok, or a genuine
1535
+ // task_error) proves the session window is open — clear the probe
1536
+ // clamp and let reviews resume alongside full task concurrency.
1537
+ if (state.capacityProbe) {
1538
+ state.capacityProbe = false;
1539
+ void pumpReviews(state);
1540
+ }
1541
+ // Task done (ok, failed without quarantine, or cancelled).
1542
+ // Re-fire pump so the next queued task can claim the freed slot.
1543
+ void pump(state, factory);
1544
+ })
1545
+ .catch((err) => {
1546
+ state.running.delete(next);
1547
+ touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at task end
1548
+ process.stderr.write(`[acc-runner] task ${next} crashed: ${err.message}\n`);
1549
+ // v0.56 (T-56-1): a crash is not a capacity signal — clear the probe
1550
+ // clamp so the runner doesn't stay stuck at concurrency 1.
1551
+ if (state.capacityProbe) {
1552
+ state.capacityProbe = false;
1553
+ void pumpReviews(state);
1554
+ }
1555
+ void pump(state, factory);
1556
+ });
1557
+ }
1558
+ // v0.56 (T-56-1): the probe found nothing to test with (no queued task at
1559
+ // resume) — clear the clamp and release any held reviews so an idle
1560
+ // resume doesn't strand pending reviews behind a probe that never runs.
1561
+ if (state.capacityProbe && state.running.size === 0 && state.queue.length === 0) {
1562
+ state.capacityProbe = false;
1563
+ void pumpReviews(state);
1564
+ }
1565
+ }
1566
+ finally {
1567
+ state.pumping = false;
1568
+ }
1569
+ }
1570
+ //# sourceMappingURL=watch.js.map