@tokenfactory/acc-runner 0.25.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (162) hide show
  1. package/README.md +77 -4
  2. package/package.json +1 -1
  3. package/dist/anthropic-auth.d.ts +0 -53
  4. package/dist/anthropic-auth.d.ts.map +0 -1
  5. package/dist/anthropic-auth.js +0 -89
  6. package/dist/anthropic-auth.js.map +0 -1
  7. package/dist/bin-resolve.d.ts +0 -16
  8. package/dist/bin-resolve.d.ts.map +0 -1
  9. package/dist/bin-resolve.js +0 -35
  10. package/dist/bin-resolve.js.map +0 -1
  11. package/dist/cli.d.ts +0 -3
  12. package/dist/cli.d.ts.map +0 -1
  13. package/dist/cli.js +0 -128
  14. package/dist/cli.js.map +0 -1
  15. package/dist/config.d.ts +0 -121
  16. package/dist/config.d.ts.map +0 -1
  17. package/dist/config.js +0 -396
  18. package/dist/config.js.map +0 -1
  19. package/dist/cost-pricing.d.ts +0 -130
  20. package/dist/cost-pricing.d.ts.map +0 -1
  21. package/dist/cost-pricing.js +0 -191
  22. package/dist/cost-pricing.js.map +0 -1
  23. package/dist/doctor.d.ts +0 -37
  24. package/dist/doctor.d.ts.map +0 -1
  25. package/dist/doctor.js +0 -658
  26. package/dist/doctor.js.map +0 -1
  27. package/dist/failure-classifier.d.ts +0 -112
  28. package/dist/failure-classifier.d.ts.map +0 -1
  29. package/dist/failure-classifier.js +0 -353
  30. package/dist/failure-classifier.js.map +0 -1
  31. package/dist/gh.d.ts +0 -22
  32. package/dist/gh.d.ts.map +0 -1
  33. package/dist/gh.js +0 -48
  34. package/dist/gh.js.map +0 -1
  35. package/dist/git.d.ts +0 -50
  36. package/dist/git.d.ts.map +0 -1
  37. package/dist/git.js +0 -127
  38. package/dist/git.js.map +0 -1
  39. package/dist/github-client.d.ts +0 -133
  40. package/dist/github-client.d.ts.map +0 -1
  41. package/dist/github-client.js +0 -234
  42. package/dist/github-client.js.map +0 -1
  43. package/dist/keychain.d.ts +0 -21
  44. package/dist/keychain.d.ts.map +0 -1
  45. package/dist/keychain.js +0 -45
  46. package/dist/keychain.js.map +0 -1
  47. package/dist/login.d.ts +0 -12
  48. package/dist/login.d.ts.map +0 -1
  49. package/dist/login.js +0 -133
  50. package/dist/login.js.map +0 -1
  51. package/dist/logout.d.ts +0 -2
  52. package/dist/logout.d.ts.map +0 -1
  53. package/dist/logout.js +0 -31
  54. package/dist/logout.js.map +0 -1
  55. package/dist/mcp-spawn.d.ts +0 -30
  56. package/dist/mcp-spawn.d.ts.map +0 -1
  57. package/dist/mcp-spawn.js +0 -145
  58. package/dist/mcp-spawn.js.map +0 -1
  59. package/dist/messaging.d.ts +0 -49
  60. package/dist/messaging.d.ts.map +0 -1
  61. package/dist/messaging.js +0 -36
  62. package/dist/messaging.js.map +0 -1
  63. package/dist/pkg-version.d.ts +0 -3
  64. package/dist/pkg-version.d.ts.map +0 -1
  65. package/dist/pkg-version.js +0 -20
  66. package/dist/pkg-version.js.map +0 -1
  67. package/dist/profiles/designer-prompt.d.ts +0 -18
  68. package/dist/profiles/designer-prompt.d.ts.map +0 -1
  69. package/dist/profiles/designer-prompt.js +0 -172
  70. package/dist/profiles/designer-prompt.js.map +0 -1
  71. package/dist/profiles/developer-prompt.d.ts +0 -24
  72. package/dist/profiles/developer-prompt.d.ts.map +0 -1
  73. package/dist/profiles/developer-prompt.js +0 -24
  74. package/dist/profiles/developer-prompt.js.map +0 -1
  75. package/dist/profiles/manager-prompt.d.ts +0 -34
  76. package/dist/profiles/manager-prompt.d.ts.map +0 -1
  77. package/dist/profiles/manager-prompt.js +0 -93
  78. package/dist/profiles/manager-prompt.js.map +0 -1
  79. package/dist/profiles/tester-prompt.d.ts +0 -28
  80. package/dist/profiles/tester-prompt.d.ts.map +0 -1
  81. package/dist/profiles/tester-prompt.js +0 -165
  82. package/dist/profiles/tester-prompt.js.map +0 -1
  83. package/dist/prompt.d.ts +0 -38
  84. package/dist/prompt.d.ts.map +0 -1
  85. package/dist/prompt.js +0 -78
  86. package/dist/prompt.js.map +0 -1
  87. package/dist/runtime/cache-dir.d.ts +0 -2
  88. package/dist/runtime/cache-dir.d.ts.map +0 -1
  89. package/dist/runtime/cache-dir.js +0 -15
  90. package/dist/runtime/cache-dir.js.map +0 -1
  91. package/dist/runtime/conflict-resolver.d.ts +0 -65
  92. package/dist/runtime/conflict-resolver.d.ts.map +0 -1
  93. package/dist/runtime/conflict-resolver.js +0 -477
  94. package/dist/runtime/conflict-resolver.js.map +0 -1
  95. package/dist/runtime/expand-args.d.ts +0 -28
  96. package/dist/runtime/expand-args.d.ts.map +0 -1
  97. package/dist/runtime/expand-args.js +0 -50
  98. package/dist/runtime/expand-args.js.map +0 -1
  99. package/dist/runtime/locks.d.ts +0 -21
  100. package/dist/runtime/locks.d.ts.map +0 -1
  101. package/dist/runtime/locks.js +0 -97
  102. package/dist/runtime/locks.js.map +0 -1
  103. package/dist/runtime/provision-mutex.d.ts +0 -37
  104. package/dist/runtime/provision-mutex.d.ts.map +0 -1
  105. package/dist/runtime/provision-mutex.js +0 -67
  106. package/dist/runtime/provision-mutex.js.map +0 -1
  107. package/dist/runtime/quarantine.d.ts +0 -26
  108. package/dist/runtime/quarantine.d.ts.map +0 -1
  109. package/dist/runtime/quarantine.js +0 -50
  110. package/dist/runtime/quarantine.js.map +0 -1
  111. package/dist/runtime/resolution-integrity.d.ts +0 -86
  112. package/dist/runtime/resolution-integrity.d.ts.map +0 -1
  113. package/dist/runtime/resolution-integrity.js +0 -248
  114. package/dist/runtime/resolution-integrity.js.map +0 -1
  115. package/dist/runtime/reviewer.d.ts +0 -81
  116. package/dist/runtime/reviewer.d.ts.map +0 -1
  117. package/dist/runtime/reviewer.js +0 -374
  118. package/dist/runtime/reviewer.js.map +0 -1
  119. package/dist/runtime/rework.d.ts +0 -48
  120. package/dist/runtime/rework.d.ts.map +0 -1
  121. package/dist/runtime/rework.js +0 -136
  122. package/dist/runtime/rework.js.map +0 -1
  123. package/dist/runtime/singleton.d.ts +0 -47
  124. package/dist/runtime/singleton.d.ts.map +0 -1
  125. package/dist/runtime/singleton.js +0 -200
  126. package/dist/runtime/singleton.js.map +0 -1
  127. package/dist/runtime/version-drift.d.ts +0 -31
  128. package/dist/runtime/version-drift.d.ts.map +0 -1
  129. package/dist/runtime/version-drift.js +0 -114
  130. package/dist/runtime/version-drift.js.map +0 -1
  131. package/dist/runtime/worktree.d.ts +0 -74
  132. package/dist/runtime/worktree.d.ts.map +0 -1
  133. package/dist/runtime/worktree.js +0 -206
  134. package/dist/runtime/worktree.js.map +0 -1
  135. package/dist/secrets/inject.d.ts +0 -70
  136. package/dist/secrets/inject.d.ts.map +0 -1
  137. package/dist/secrets/inject.js +0 -102
  138. package/dist/secrets/inject.js.map +0 -1
  139. package/dist/supabase.d.ts +0 -4
  140. package/dist/supabase.d.ts.map +0 -1
  141. package/dist/supabase.js +0 -36
  142. package/dist/supabase.js.map +0 -1
  143. package/dist/task-runner.d.ts +0 -313
  144. package/dist/task-runner.d.ts.map +0 -1
  145. package/dist/task-runner.js +0 -1766
  146. package/dist/task-runner.js.map +0 -1
  147. package/dist/token-provider.d.ts +0 -50
  148. package/dist/token-provider.d.ts.map +0 -1
  149. package/dist/token-provider.js +0 -177
  150. package/dist/token-provider.js.map +0 -1
  151. package/dist/types.d.ts +0 -120
  152. package/dist/types.d.ts.map +0 -1
  153. package/dist/types.js +0 -16
  154. package/dist/types.js.map +0 -1
  155. package/dist/version-check.d.ts +0 -17
  156. package/dist/version-check.d.ts.map +0 -1
  157. package/dist/version-check.js +0 -56
  158. package/dist/version-check.js.map +0 -1
  159. package/dist/watch.d.ts +0 -295
  160. package/dist/watch.d.ts.map +0 -1
  161. package/dist/watch.js +0 -1532
  162. package/dist/watch.js.map +0 -1
package/dist/watch.js DELETED
@@ -1,1532 +0,0 @@
1
- /**
2
- * `acc-runner watch` — long-running command. Subscribes to broadcast
3
- * tasks, dispatches them concurrently (v0.10+), heartbeats every 4s.
4
- * v0.2.5 refreshes the access token on startup + every 30 min, and exits
5
- * loudly after 5 consecutive heartbeat failures.
6
- *
7
- * v0.3 adds a 30s polling fallback: if a Realtime broadcast is dropped
8
- * during a reconnect, the next poll pass picks the task up via the
9
- * `acc.list_assigned_queued_tasks` RPC. A `seen` set keeps the poll
10
- * and the broadcast from fighting over the same task_id.
11
- *
12
- * v0.6.2 (v0.12-REVIEW-LOCAL) listens for `review_assigned` on the same
13
- * `runner:<id>` channel and dispatches to runtime/reviewer.ts. Reviews
14
- * are independently tracked (separate `seenReviews` set).
15
- *
16
- * v0.10 (T-49-2): concurrent task dispatch. pump() now launches up to
17
- * `concurrencyLimit` (default 2) tasks simultaneously. Each task runs in
18
- * its own isolated git worktree so concurrent tasks never share files.
19
- * `createTaskAsAdmin` in api/_lib/supabase-admin.ts is the sanctioned
20
- * Manager seed path that feeds this dispatch loop.
21
- *
22
- * v0.12.0 (T-52-7): ephemeral-runner support. Sessions load through the
23
- * token-provider abstraction (keychain by default — byte-identical desktop
24
- * behaviour; ACC_RUNNER_TOKEN_MODE=env for containers). Env-token runners
25
- * register an acc-eph-<random> identity on boot, and ACC_RUNNER_IDLE_TTL_MIN
26
- * arms a clean self-termination after n minutes without work.
27
- */
28
- import chalk from "chalk";
29
- import { loadConfig } from "./config.js";
30
- import { getTokenProvider } from "./token-provider.js";
31
- import { machineIdentity, detectInstalledModels, RUNNER_CAPS } from "./login.js";
32
- import { createRunnerClient } from "./supabase.js";
33
- import { runTask } from "./task-runner.js";
34
- import { runReview, } from "./runtime/reviewer.js";
35
- import { getQuarantine, setQuarantine, } from "./runtime/quarantine.js";
36
- import { acquireSingletonLock, SingletonLockHeldError, } from "./runtime/singleton.js";
37
- import { getClaudeVersion, recordTaskClaudeVersion, } from "./runtime/version-drift.js";
38
- import { postRunnerStateMessage } from "./messaging.js";
39
- import { checkVersion, compareSemver } from "./version-check.js";
40
- import { PROTOCOL_VERSION } from "./types.js";
41
- import { PACKAGE_VERSION } from "./pkg-version.js";
42
- import { resolveAuthMode, describeAuthMode, authPrecedenceHint } from "./anthropic-auth.js";
43
- export class RefreshReuseDetectedError extends Error {
44
- constructor(message) {
45
- super(message);
46
- this.name = "RefreshReuseDetectedError";
47
- }
48
- }
49
- const ONE_HOUR_MS = 60 * 60 * 1000;
50
- const REFRESH_CHECK_MS = 30 * 60 * 1000;
51
- // v0.12.0 (T-52-7): idle-TTL check cadence ceiling. The actual cadence is
52
- // min(IDLE_CHECK_MS, ttl) so a tiny test TTL still gets checked in time.
53
- const IDLE_CHECK_MS = 30_000;
54
- const HEARTBEAT_REFRESH_AFTER = 3;
55
- const HEARTBEAT_EXIT_AFTER = 5;
56
- const POLL_INTERVAL_MS = 30_000;
57
- // v0.30-B: widen the startup `pollSince` watermark so a task dispatched in
58
- // the minutes before `watch` started is still caught by the first poll.
59
- // The 30s window the previous code used was tighter than the gap between
60
- // dispatch and runner restart in real operator workflows.
61
- const STARTUP_POLL_LOOKBACK_MS = 5 * 60 * 1000;
62
- // v0.62 (T-62-1): cadence of the periodic full claim scan — the realtime-miss
63
- // safety net. The delta-poll's forward-only watermark can permanently skip a
64
- // task assigned with an `updated_at` at/behind the watermark whose realtime
65
- // `task_assigned` push was dropped (stale subscription). This full scan
66
- // (epoch watermark, every assigned+queued task for this runner) re-claims any
67
- // such task within one interval, so an alive-but-wedged runner self-recovers
68
- // instead of needing a manual restart. Knob: `claim_scan_interval_ms`.
69
- const DEFAULT_CLAIM_SCAN_INTERVAL_MS = 3 * 60 * 1000;
70
- // v0.73 (T-73-2, G-M): claim-loop watchdog. A task can sit assigned+queued for
71
- // this runner (acc.tasks.runner_id = me, status='queued') while the runner
72
- // heartbeats fine but never claims it — the channel looks healthy (so T-68-1's
73
- // auto-recovery never fires) but the claim path is wedged. CLAIM_WATCHDOG_STALE_MS
74
- // is how long a task may stay assigned-but-unclaimed (with a live heartbeat)
75
- // before the watchdog treats the claim path as wedged and forces the same
76
- // recovery a terminal channel status does. Default 90s — comfortably under the
77
- // server's 300s rescue grace, so we self-heal before the rescue churns rather
78
- // than waiting for a manual `pkill acc-runner watch`. CLAIM_WATCHDOG_CHECK_MS is
79
- // how often the watchdog re-evaluates; well under the threshold so detection +
80
- // recovery land inside the grace window. Knobs: tests override both for speed.
81
- const DEFAULT_CLAIM_WATCHDOG_STALE_MS = 90_000;
82
- const DEFAULT_CLAIM_WATCHDOG_CHECK_MS = 30_000;
83
- // v0.56 (T-56-1): capacity-aware pause. When claude reports a session/usage
84
- // limit (api_error_status 429) the runner pauses claiming tasks AND reviews
85
- // until the stated reset time. When no reset time is parseable, fall back to
86
- // this fixed backoff. Resume fires at the target + a small random jitter so a
87
- // fleet that all hit the limit on the same broadcast doesn't thunder back in
88
- // lockstep the instant the window reopens.
89
- const DEFAULT_CAPACITY_BACKOFF_MS = 30 * 60 * 1000;
90
- const DEFAULT_RESUME_JITTER_MS = 15_000;
91
- /**
92
- * v0.6.0: exponential backoff between heartbeat attempts when the
93
- * previous one failed. The dogfood loop fired a heartbeat every 4s
94
- * even when the network was wedged, producing a wall of "fetch failed"
95
- * stderr noise and reaching the EXIT_AFTER cap in 20s. Spreading
96
- * attempts out gives flaky links time to recover before we tear down
97
- * the watch process.
98
- */
99
- const HEARTBEAT_BACKOFF_MS = [4_000, 8_000, 16_000, 32_000, 60_000];
100
- /**
101
- * v0.6.0: classify the heartbeat RPC error so we (a) emit a useful
102
- * stderr line for the operator and (b) only attempt a JWT refresh
103
- * when the failure shape actually implies an auth problem. Refreshing
104
- * on a transient network blip just burns the refresh-token rotation
105
- * chain and surfaces a confusing "refresh failed" line on top of the
106
- * real underlying network error.
107
- */
108
- export function classifyHeartbeatError(err) {
109
- if (!err)
110
- return "unknown";
111
- const msg = (err.message ?? "").toLowerCase();
112
- const code = (err.code ?? "").toLowerCase();
113
- if (code === "pgrst301" ||
114
- code === "401" ||
115
- code === "403" ||
116
- msg.includes("jwt") ||
117
- msg.includes("invalid token") ||
118
- msg.includes("token has expired") ||
119
- msg.includes("unauthorized") ||
120
- msg.includes("forbidden")) {
121
- return "auth";
122
- }
123
- if (msg.includes("enotfound") || msg.includes("getaddrinfo") || msg.includes("eai_again")) {
124
- return "dns";
125
- }
126
- if (msg.includes("econnreset") || msg.includes("connection reset")) {
127
- return "reset";
128
- }
129
- if (msg.includes("certificate") ||
130
- msg.includes("self-signed") ||
131
- msg.includes("self signed") ||
132
- msg.includes("tls") ||
133
- msg.includes("ssl")) {
134
- return "tls";
135
- }
136
- if (msg.includes("etimedout") || msg.includes("timeout")) {
137
- return "timeout";
138
- }
139
- if (msg.includes("fetch failed") ||
140
- msg.includes("econnrefused") ||
141
- msg.includes("network") ||
142
- msg.includes("socket hang up")) {
143
- return "network";
144
- }
145
- return "unknown";
146
- }
147
- /** Pick the next inter-heartbeat delay from the backoff ladder. Caller
148
- * passes the number of consecutive failures (1 → first retry). The
149
- * return is at least baseMs so a tighter operator-configured cadence
150
- * is never slowed by the backoff scheme. */
151
- export function nextHeartbeatDelayMs(consecutiveFailures, baseMs) {
152
- if (consecutiveFailures <= 0)
153
- return baseMs;
154
- const idx = Math.min(consecutiveFailures - 1, HEARTBEAT_BACKOFF_MS.length - 1);
155
- return Math.max(baseMs, HEARTBEAT_BACKOFF_MS[idx]);
156
- }
157
- /**
158
- * v0.12.0 (T-52-7): resolve ACC_RUNNER_IDLE_TTL_MIN into ms. Returns null
159
- * (TTL disabled) when unset; warns and returns null on junk so a typo'd
160
- * env never strands a container in "never exits" mode silently.
161
- */
162
- export function resolveIdleTtlMs(env = process.env) {
163
- const raw = env.ACC_RUNNER_IDLE_TTL_MIN?.trim();
164
- if (!raw)
165
- return null;
166
- const minutes = Number(raw);
167
- if (!Number.isFinite(minutes) || minutes <= 0) {
168
- process.stderr.write(`[acc-runner] ACC_RUNNER_IDLE_TTL_MIN=${JSON.stringify(raw)} is not a positive number; idle TTL disabled\n`);
169
- return null;
170
- }
171
- return minutes * 60_000;
172
- }
173
- /** v0.12.0 (T-52-7): record work-related activity for the idle TTL clock. */
174
- function touchActivity(state) {
175
- state.lastActivityMs = Date.now();
176
- }
177
- function expiresSoon(session, leadMs = ONE_HOUR_MS) {
178
- const expires = Date.parse(session.access_expires_at);
179
- if (!Number.isFinite(expires))
180
- return true;
181
- return expires < Date.now() + leadMs;
182
- }
183
- async function refreshAccessToken(cfg, session) {
184
- const res = await fetch(`${cfg.publicUrl.replace(/\/+$/, "")}/api/runner/refresh`, {
185
- method: "POST",
186
- headers: {
187
- "Content-Type": "application/json",
188
- "User-Agent": `acc-runner/${PROTOCOL_VERSION}`,
189
- },
190
- body: JSON.stringify({ refresh_token: session.refresh_token }),
191
- });
192
- if (res.status === 401) {
193
- const body = (await res.json().catch(() => ({})));
194
- if (body.error === "refresh_reuse_detected") {
195
- throw new RefreshReuseDetectedError(body.hint ??
196
- "refresh token reuse detected; keychain may be compromised. Run `acc-runner login` from a trusted machine.");
197
- }
198
- throw new Error(body.hint ?? "refresh token expired; run `acc-runner login`");
199
- }
200
- if (!res.ok) {
201
- throw new Error(`refresh failed: ${res.status} ${await res.text()}`);
202
- }
203
- const body = (await res.json());
204
- return {
205
- ...session,
206
- access_token: body.access_token,
207
- access_expires_at: new Date(body.expires_at * 1000).toISOString(),
208
- // v0.4-E: server rotates on every refresh. Persist the new refresh token
209
- // so the next call doesn't replay the old one (which would now look like
210
- // reuse and trip the chain-revocation path).
211
- refresh_token: body.refresh_token ?? session.refresh_token,
212
- };
213
- }
214
- // Centralised handler for the reuse-detected case. Loud red banner so the
215
- // operator notices immediately, then exit(1) — there's no safe recovery
216
- // from the watch loop, the keychain has to be reset by `acc-runner login`.
217
- function failOnReuse(err) {
218
- process.stderr.write(chalk.red.bold("\n[acc-runner] CRITICAL: refresh-token reuse detected.\n"));
219
- process.stderr.write(chalk.red(`${err.message}\n`));
220
- process.stderr.write(chalk.red("If this machine is the legitimate owner, an attacker may have rotated your tokens.\n" +
221
- "Rotate any other credentials that were on this box and run `acc-runner login` from a trusted machine.\n"));
222
- process.exit(1);
223
- }
224
- // Single source of truth for "this task is new — push it into the runner
225
- // queue". Both the Realtime listener and the polling loop go through here.
226
- function enqueue(state, taskId, factory) {
227
- if (state.seen.has(taskId))
228
- return false;
229
- // v0.10 T-49-2: check the running Map (replaces v0.9 state.current check).
230
- if (state.running.has(taskId)) {
231
- state.seen.add(taskId);
232
- return false;
233
- }
234
- state.seen.add(taskId);
235
- state.queue.push(taskId);
236
- touchActivity(state);
237
- void pump(state, factory);
238
- return true;
239
- }
240
- // Single source of truth for "this review is new — push it into the review
241
- // queue". The Realtime listener AND the periodic review scan (v0.65 T-65-2)
242
- // both go through here; the shared `seenReviews` guard dedupes a review that
243
- // arrives on both paths. Returns true only when the review was actually
244
- // enqueued (new + well-formed), so the scan can count what it recovered.
245
- function enqueueReview(state, payload) {
246
- const p = payload;
247
- if (!p?.review_id || !p?.task_id || !p?.pr_number)
248
- return false;
249
- if (state.seenReviews.has(p.review_id))
250
- return false;
251
- state.seenReviews.add(p.review_id);
252
- state.reviewQueue.push({
253
- review_id: p.review_id,
254
- task_id: p.task_id,
255
- pr_number: p.pr_number,
256
- });
257
- touchActivity(state);
258
- void pumpReviews(state);
259
- return true;
260
- }
261
- function subscribeChannel(state, taskRunnerFactory) {
262
- state.channel = state.supabase
263
- .channel(`runner:${state.session.runner_id}`, {
264
- config: { broadcast: { ack: false, self: false } },
265
- })
266
- .on("broadcast", { event: "task_assigned" }, ({ payload }) => {
267
- const taskId = payload?.task_id;
268
- if (taskId)
269
- enqueue(state, taskId, taskRunnerFactory);
270
- })
271
- .on("broadcast", { event: "task_cancelled" }, ({ payload }) => {
272
- const taskId = payload?.task_id;
273
- if (!taskId)
274
- return;
275
- // v0.10 T-49-2: cancel from the running Map; fall back to queue removal.
276
- const ctrl = state.running.get(taskId);
277
- if (ctrl) {
278
- ctrl.cancel();
279
- }
280
- else {
281
- state.queue = state.queue.filter((t) => t !== taskId);
282
- }
283
- // Drop from seen so a re-queue (cancel → re-start) is honored on
284
- // the next broadcast or poll.
285
- state.seen.delete(taskId);
286
- })
287
- .on("broadcast", { event: "review_assigned" }, ({ payload }) => {
288
- enqueueReview(state, payload);
289
- })
290
- .subscribe((status) => {
291
- if (status === "SUBSCRIBED") {
292
- // v0.68 (T-68-1): healthy subscribe — reset the resubscribe backoff.
293
- state.resubscribeAttempts = 0;
294
- console.log(chalk.green(`✓ Listening on runner:${state.session.runner_id}`));
295
- }
296
- else if (status === "CHANNEL_ERROR" ||
297
- status === "TIMED_OUT" ||
298
- status === "CLOSED") {
299
- // v0.68 (T-68-1): the realtime channel dropped. Previously unhandled —
300
- // which left the runner silently deaf to task_assigned / review_assigned
301
- // broadcasts until token expiry or a manual restart (the stale-
302
- // subscription root cause behind the repeated "task stuck in queued →
303
- // swept to blocked" incidents). Recover it.
304
- if (!state.stopped)
305
- scheduleResubscribe(state, taskRunnerFactory, status);
306
- }
307
- });
308
- }
309
- // v0.68 (T-68-1): single-flight, backed-off recovery of a dropped realtime
310
- // channel. Like doRefresh's remove+resubscribe, but WITHOUT a token rotation
311
- // (the token is still valid; only the channel died) and driven by channel
312
- // status rather than token expiry. Backoff 1s→30s, reset to 0 on a healthy
313
- // SUBSCRIBED. After a fresh subscribe we run the poll + claim/review scans so
314
- // anything the dead channel missed is recovered at once rather than waiting
315
- // for the periodic 3-min scans.
316
- function scheduleResubscribe(state, factory, status) {
317
- if (state.stopped || state.resubscribing || state.resubscribeTimer)
318
- return;
319
- const delay = Math.min(30_000, 1_000 * 2 ** Math.min(state.resubscribeAttempts, 5));
320
- process.stderr.write(`[acc-runner] realtime channel ${status}; resubscribing in ${delay}ms ` +
321
- `(attempt ${state.resubscribeAttempts + 1})\n`);
322
- state.resubscribeTimer = setTimeout(() => {
323
- state.resubscribeTimer = null;
324
- void resubscribeChannel(state, factory);
325
- }, delay);
326
- if (state.resubscribeTimer.unref)
327
- state.resubscribeTimer.unref();
328
- }
329
- export async function resubscribeChannel(state, factory) {
330
- if (state.stopped || state.resubscribing)
331
- return;
332
- state.resubscribing = true;
333
- state.resubscribeAttempts += 1;
334
- try {
335
- try {
336
- await state.supabase.removeChannel(state.channel);
337
- }
338
- catch {
339
- /* channel may already be torn down */
340
- }
341
- subscribeChannel(state, factory);
342
- // Recover anything the dead channel missed during the gap. Best-effort;
343
- // the `seen` / `seenReviews` guards dedupe against a late re-delivery.
344
- void pollOnce(state, factory);
345
- void claimScan(state, factory, "periodic");
346
- void reviewScan(state, "periodic");
347
- }
348
- finally {
349
- state.resubscribing = false;
350
- }
351
- }
352
- // Single-flight poll. The watermark advances only after a clean response so
353
- // a transient error doesn't permanently skip rows.
354
- async function pollOnce(state, factory) {
355
- if (state.stopped || state.polling)
356
- return;
357
- state.polling = true;
358
- try {
359
- const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
360
- p_runner_id: state.session.runner_id,
361
- p_since: state.pollSince,
362
- });
363
- if (error) {
364
- process.stderr.write(`[acc-runner] poll failed: ${error.message}\n`);
365
- return;
366
- }
367
- const rows = (data ?? []);
368
- let maxSeen = state.pollSince;
369
- for (const row of rows) {
370
- if (!row?.id)
371
- continue;
372
- // v0.62 (T-62-1): advance the forward-only watermark ONLY for rows we
373
- // actually claimed this pass (enqueue returned true). Previously the
374
- // watermark advanced to max(updated_at) over EVERY returned row,
375
- // including no-ops (a task already seen/in-flight via realtime or a
376
- // prior poll). A high-updated_at no-op could then drag the watermark
377
- // past a still-queued task assigned with an older updated_at, so the
378
- // next delta-poll (`> watermark`) silently dropped it — only a restart
379
- // (one-shot startup scan) recovered it. Advancing solely on real claims
380
- // keeps the watermark a conservative "newest task I claimed by polling"
381
- // mark and stops the delta-poll losing still-queued work. The periodic
382
- // full claim scan (claimScan) is the belt to this braces.
383
- const claimed = enqueue(state, row.id, factory);
384
- if (claimed && row.updated_at && row.updated_at > maxSeen) {
385
- maxSeen = row.updated_at;
386
- }
387
- }
388
- state.pollSince = maxSeen;
389
- }
390
- finally {
391
- state.polling = false;
392
- }
393
- }
394
- /**
395
- * v0.62 (T-62-1): full claim scan — lists EVERY assigned+queued task for this
396
- * runner with an epoch watermark (`1970-…Z`) and enqueues any the in-process
397
- * `seen` guard hasn't already accounted for. Used two ways:
398
- *
399
- * - once at startup (replaces the v0.34-B inline block), and
400
- * - on a periodic timer (`claim_scan_interval_ms`, default 3 min) as the
401
- * realtime-miss safety net: a runner that dropped a `task_assigned`
402
- * broadcast self-recovers within one interval instead of sitting next to
403
- * a free runner until a manual restart.
404
- *
405
- * Deliberately does NOT touch `state.pollSince` — the delta-poll keeps its own
406
- * forward-only watermark. Idempotent: `enqueue`'s shared `seen`/`running`
407
- * guards mean a task already taken by realtime or the delta-poll is a no-op
408
- * here, so realtime + delta-poll + scan all seeing the same task still yields
409
- * exactly one run. Single-flight via `claimScanning` so a slow RPC can't let
410
- * two scans overlap.
411
- */
412
- async function claimScan(state, factory, context) {
413
- if (state.stopped || state.claimScanning)
414
- return;
415
- state.claimScanning = true;
416
- try {
417
- const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
418
- p_runner_id: state.session.runner_id,
419
- p_since: "1970-01-01T00:00:00.000Z",
420
- });
421
- if (error) {
422
- process.stderr.write(`[acc-runner] ${context} claim scan failed: ${error.message}\n`);
423
- return;
424
- }
425
- const rows = (data ?? []);
426
- let enqueued = 0;
427
- for (const row of rows) {
428
- if (!row?.id)
429
- continue;
430
- if (enqueue(state, row.id, factory))
431
- enqueued += 1;
432
- }
433
- if (context === "startup") {
434
- console.log(`[acc-runner] startup claim scan: ${rows.length} task(s) found`);
435
- }
436
- else if (enqueued > 0) {
437
- // Only chirp when the safety net actually caught something — a periodic
438
- // scan that finds nothing is the steady state and must stay quiet.
439
- console.log(`[acc-runner] periodic claim scan recovered ${enqueued} realtime-missed task(s)`);
440
- }
441
- }
442
- catch (err) {
443
- process.stderr.write(`[acc-runner] ${context} claim scan failed: ${err.message}\n`);
444
- }
445
- finally {
446
- state.claimScanning = false;
447
- }
448
- }
449
- /**
450
- * v0.73 (T-73-2, G-M): claim-loop watchdog — the self-heal for an
451
- * "assigned-but-unclaimed while heartbeating" strand.
452
- *
453
- * SYMPTOM (prod 2026-06-10): a runner heartbeated fine (status online, fresh
454
- * heartbeat, idle) but never claimed a task the dispatcher had ASSIGNED to it
455
- * (acc.tasks.runner_id = me, status='queued'). The server rescue sweep
456
- * re-dispatched every ~6 min but the runner never claimed; only a manual
457
- * `pkill acc-runner watch && acc-runner watch` cleared it. T-68-1's channel
458
- * auto-recovery did NOT catch it — the channel looked healthy but the claim
459
- * path was wedged.
460
- *
461
- * Mechanism: each tick lists the runner's own assigned+queued (= unclaimed)
462
- * tasks via the same `list_assigned_queued_tasks` RPC the claim scan uses, and
463
- * tracks per task id the first instant it was OBSERVED as
464
- * assigned-to-me-but-not-claimed. A task that stays unclaimed past
465
- * `claimWatchdogStaleMs` (default 90s — comfortably under the server's 300s
466
- * rescue grace) WHILE heartbeats are succeeding means the claim path is wedged:
467
- * trigger the SAME single-flight recovery a terminal channel status does
468
- * (resubscribeChannel: removeChannel + subscribeChannel + pollOnce + claimScan
469
- * + reviewScan), logged as `claim_watchdog_resubscribe`.
470
- *
471
- * Reset semantics: a task's observed-stale timer is dropped the moment it
472
- * leaves the assigned+queued set (claimed → status moved off 'queued', or
473
- * unassigned → runner_id cleared). A heartbeat failure clears ALL timers — a
474
- * network/auth outage is a different failure with its own handling, and we must
475
- * not carry a stale age across it into a false trigger when heartbeats recover.
476
- *
477
- * Single-flight: respects the existing resubscribe guard/backoff
478
- * (`resubscribing` / `resubscribeTimer`) so concurrent staleness can't stack
479
- * resubscribes, and resets the stale timers on trigger so a recovery that
480
- * didn't help waits another full threshold before re-firing. Bounded and
481
- * side-effect-safe: never throws out of the loop.
482
- */
483
- async function claimWatchdogScan(state, factory) {
484
- if (state.stopped || state.claimWatchdogChecking)
485
- return;
486
- state.claimWatchdogChecking = true;
487
- try {
488
- // Only treat "unclaimed" as wedged while heartbeats are healthy. A
489
- // heartbeat failure means the network/auth path is down — the claim RPC
490
- // would fail too, and the fix there is the heartbeat refresh/exit path, not
491
- // a resubscribe. Clear the observed-stale timers so an age never spans an
492
- // outage and falsely trips the watchdog the instant heartbeats recover.
493
- if (state.heartbeatFailures > 0) {
494
- state.claimWatchdogSince.clear();
495
- return;
496
- }
497
- const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
498
- p_runner_id: state.session.runner_id,
499
- p_since: "1970-01-01T00:00:00.000Z",
500
- });
501
- if (error) {
502
- process.stderr.write(`[acc-runner] claim watchdog scan failed: ${error.message}\n`);
503
- return;
504
- }
505
- const rows = (data ?? []);
506
- const now = Date.now();
507
- const currentIds = new Set();
508
- for (const row of rows) {
509
- if (row?.id)
510
- currentIds.add(row.id);
511
- }
512
- // Reset: drop timers for tasks no longer assigned+queued (claimed/unassigned).
513
- for (const id of [...state.claimWatchdogSince.keys()]) {
514
- if (!currentIds.has(id))
515
- state.claimWatchdogSince.delete(id);
516
- }
517
- // Observe: stamp the first-seen instant for newly assigned-unclaimed tasks;
518
- // collect any that have now been stale past the threshold.
519
- const stale = [];
520
- for (const id of currentIds) {
521
- const since = state.claimWatchdogSince.get(id);
522
- if (since === undefined) {
523
- state.claimWatchdogSince.set(id, now);
524
- }
525
- else if (now - since >= state.claimWatchdogStaleMs) {
526
- stale.push(id);
527
- }
528
- }
529
- if (stale.length === 0)
530
- return;
531
- // Single-flight: never stack on a resubscribe already in flight or
532
- // scheduled (the existing T-68-1 guard/backoff owns the recovery).
533
- if (state.resubscribing || state.resubscribeTimer)
534
- return;
535
- process.stderr.write(`[acc-runner] claim_watchdog_resubscribe: task(s) ${stale.join(", ")} ` +
536
- `assigned but unclaimed > ${state.claimWatchdogStaleMs}ms with a live ` +
537
- `heartbeat; claim path looks wedged — forcing resubscribe + claim scan.\n`);
538
- // Reset the stale timers so a recovery that didn't help waits another full
539
- // threshold before re-triggering (backoff alongside the resubscribe guard).
540
- for (const id of stale)
541
- state.claimWatchdogSince.set(id, now);
542
- void (async () => {
543
- try {
544
- await state.supabase.rpc("log_activity", {
545
- p_verb: "runner.claim_watchdog_resubscribe",
546
- p_target_id: state.session.runner_id,
547
- p_payload: {
548
- task_ids: stale,
549
- stale_ms: state.claimWatchdogStaleMs,
550
- version: PACKAGE_VERSION,
551
- },
552
- p_target_type: "runner",
553
- });
554
- }
555
- catch { /* best-effort breadcrumb */ }
556
- })();
557
- void resubscribeChannel(state, factory);
558
- }
559
- catch (err) {
560
- process.stderr.write(`[acc-runner] claim watchdog scan failed: ${err.message}\n`);
561
- }
562
- finally {
563
- state.claimWatchdogChecking = false;
564
- }
565
- }
566
- /**
567
- * v0.65 (T-65-2): full review scan — the review twin of claimScan. Lists EVERY
568
- * pending review assigned to this runner with an epoch watermark (`1970-…Z`)
569
- * via `list_assigned_pending_reviews` and feeds each row into the existing
570
- * `enqueueReview`, whose shared `seenReviews` guard dedupes a review the
571
- * realtime `review_assigned` broadcast already delivered. Used two ways:
572
- *
573
- * - once at startup (next to the startup claim scan), and
574
- * - on a periodic timer (`review_scan_interval_ms`, default 3 min) as the
575
- * realtime-miss safety net: a runner that dropped a `review_assigned`
576
- * broadcast recovers the review within one interval instead of letting it
577
- * strand in acc.review_queue (PR #544's review stranded 16h last cycle).
578
- *
579
- * Idempotent: `enqueueReview`'s `seenReviews` guard means a review already
580
- * taken by realtime is a no-op here, so realtime + scan both seeing the same
581
- * review still yields exactly one review run. Single-flight via
582
- * `reviewScanning` so a slow RPC can't let two scans overlap. Skipped while
583
- * paused for capacity — matching `pumpReviews`, the runner deliberately holds
584
- * reviews until its session window reopens.
585
- */
586
- async function reviewScan(state, context) {
587
- if (state.stopped || state.reviewScanning)
588
- return;
589
- // v0.56 (T-56-1) parity with pumpReviews: don't claim reviews while paused
590
- // for capacity. The scheduled resume re-drives the review pump.
591
- if (state.pausedCapacity)
592
- return;
593
- state.reviewScanning = true;
594
- try {
595
- const { data, error } = await state.supabase.rpc("list_assigned_pending_reviews", {
596
- p_runner_id: state.session.runner_id,
597
- p_since: "1970-01-01T00:00:00.000Z",
598
- });
599
- if (error) {
600
- process.stderr.write(`[acc-runner] ${context} review scan failed: ${error.message}\n`);
601
- return;
602
- }
603
- const rows = (data ?? []);
604
- let enqueued = 0;
605
- for (const row of rows) {
606
- if (enqueueReview(state, row))
607
- enqueued += 1;
608
- }
609
- if (context === "startup") {
610
- console.log(`[acc-runner] startup review scan: ${rows.length} pending review(s) found`);
611
- }
612
- else if (enqueued > 0) {
613
- // Only chirp when the safety net actually caught something — a periodic
614
- // scan that finds nothing is the steady state and must stay quiet.
615
- console.log(`[acc-runner] periodic review scan recovered ${enqueued} realtime-missed review(s)`);
616
- }
617
- }
618
- catch (err) {
619
- process.stderr.write(`[acc-runner] ${context} review scan failed: ${err.message}\n`);
620
- }
621
- finally {
622
- state.reviewScanning = false;
623
- }
624
- }
625
- // Single-flight refresh. Recreates the Supabase client (so future RPC +
626
- // channel auth use the new JWT) and resubscribes the broadcast channel.
627
- async function doRefresh(state, taskRunnerFactory) {
628
- // v0.12.0 (T-52-7): an env-token runner without ACC_RUNNER_REFRESH_TOKEN
629
- // cannot rotate. Refusing here (instead of POSTing an empty token) keeps
630
- // the failure mode a clear stderr line rather than a server-side 400.
631
- if (!state.tokenProvider.canRefresh()) {
632
- process.stderr.write("[acc-runner] token refresh unavailable (no refresh token in env); runner will exit when the access token expires\n");
633
- return false;
634
- }
635
- if (state.refreshing)
636
- return state.refreshing;
637
- state.refreshing = (async () => {
638
- try {
639
- const updated = await refreshAccessToken(state.cfg, state.session);
640
- state.session = updated;
641
- await state.tokenProvider.save(updated);
642
- try {
643
- await state.supabase.removeChannel(state.channel);
644
- }
645
- catch {
646
- /* channel may already be torn down */
647
- }
648
- state.supabase = createRunnerClient(state.cfg, state.session.access_token);
649
- subscribeChannel(state, taskRunnerFactory);
650
- return true;
651
- }
652
- catch (err) {
653
- if (err instanceof RefreshReuseDetectedError)
654
- failOnReuse(err);
655
- process.stderr.write(`[acc-runner] refresh failed: ${err.message}\n`);
656
- return false;
657
- }
658
- finally {
659
- state.refreshing = null;
660
- }
661
- })();
662
- return state.refreshing;
663
- }
664
- export async function watchCommand(options = {}) {
665
- const cfg = loadConfig();
666
- const exitFn = options.exit ?? ((code) => process.exit(code));
667
- // v0.14.0 (T-54-2, GA-3): single-instance guard. Acquired BEFORE the
668
- // session load + realtime connection so a duplicate exits immediately,
669
- // before it can race the holder on worktree locks or git refs. A second
670
- // runner for the same repo path is a hard error; a stale lock from a dead
671
- // pid self-reclaims inside acquireSingletonLock.
672
- const enforceSingleton = options.enforceSingleton ?? !options.taskRunnerFactory;
673
- let singletonLock = null;
674
- if (enforceSingleton) {
675
- try {
676
- singletonLock = await acquireSingletonLock(cfg.repoPath);
677
- }
678
- catch (err) {
679
- if (err instanceof SingletonLockHeldError) {
680
- process.stderr.write(chalk.red.bold("\n[acc-runner] ANOTHER RUNNER IS ALREADY ACTIVE.\n"));
681
- process.stderr.write(chalk.red(`Repo path ${err.holder.repoPath} is served by pid ${err.holder.pid} ` +
682
- `(acc-runner v${err.holder.version}, since ${err.holder.acquiredAt}).\n` +
683
- "Refusing to start a second runner for the same repo — concurrent " +
684
- "runners corrupt worktrees and race git refs.\n"));
685
- process.stderr.write(chalk.gray(`To stop the other runner: kill ${err.holder.pid}\n` +
686
- `If it is already dead, remove ${err.lockFile} (or just retry — ` +
687
- "stale locks self-reclaim).\n"));
688
- exitFn(1);
689
- // exitFn is process.exit in production (never returns); in tests it
690
- // is a stub, so re-throw to halt watchCommand cleanly.
691
- throw err;
692
- }
693
- throw err;
694
- }
695
- }
696
- const tokenProvider = getTokenProvider();
697
- let session = await tokenProvider.load();
698
- if (!session) {
699
- process.stderr.write(tokenProvider.mode === "env"
700
- ? "No session in env. Set ACC_RUNNER_ACCESS_TOKEN (token mode: env).\n"
701
- : "Not logged in. Run `acc-runner login`.\n");
702
- process.exit(1);
703
- }
704
- const v = await checkVersion(cfg.publicUrl);
705
- if (!v.ok && v.reason === "outdated") {
706
- throw new Error(`acc-runner is below the server's minimum version (${v.serverMin}). Run \`pnpm add -g @tokenfactory/acc-runner@latest\`.`);
707
- }
708
- // v0.31-B: warn when running behind current_version (above min but stale).
709
- // Launchd daemon often runs an old binary silently; this surfaces it at
710
- // startup before the first heartbeat so the operator can update promptly.
711
- if (v.ok && v.serverCurrent && compareSemver(PACKAGE_VERSION, v.serverCurrent) < 0) {
712
- process.stderr.write(`[acc-runner] WARNING: running v${PACKAGE_VERSION}, server current is v${v.serverCurrent}. ` +
713
- `Run \`npm install -g @tokenfactory/acc-runner@latest\` then reload launchd.\n`);
714
- }
715
- // v0.74-B: surface how spawned claude sessions will authenticate. api-key
716
- // (ANTHROPIC_API_KEY set) bills against the API account with no per-session
717
- // cap; interactive-session falls back to the operator's OAuth login, which
718
- // hits 429 session caps during long autonomous runs.
719
- const authMode = resolveAuthMode();
720
- console.log(chalk.gray(`Anthropic auth: ${describeAuthMode(authMode)}`));
721
- // v0.76-A (GA-11): surface the API-key-vs-claude.ai-login precedence hint so
722
- // a 429-on-a-key-configured-runner is diagnosable from the banner.
723
- const authHint = authPrecedenceHint();
724
- if (authHint)
725
- console.log(chalk.gray(`Anthropic auth: ${authHint}`));
726
- // Startup refresh: if the access token expires in less than an hour,
727
- // grab a fresh one before opening the realtime connection.
728
- // v0.12.0 (T-52-7): skipped when the provider has no refresh token —
729
- // an env-token runner just runs out its 8h access token (the idle TTL
730
- // normally exits it long before then).
731
- if (tokenProvider.canRefresh() && expiresSoon(session)) {
732
- try {
733
- session = await refreshAccessToken(cfg, session);
734
- await tokenProvider.save(session);
735
- }
736
- catch (err) {
737
- if (err instanceof RefreshReuseDetectedError)
738
- failOnReuse(err);
739
- process.stderr.write(`Refresh failed; run \`acc-runner login\`. (${err.message})\n`);
740
- process.exit(1);
741
- }
742
- }
743
- // v0.12.0 (T-52-7): env-token (ephemeral) runners had no `login` flow, so
744
- // nothing registered an acc.runners row yet. Register on boot under the
745
- // acc-eph-<random> identity. register_runner may resolve to an existing
746
- // row (same bound user + machine, e.g. a restarted container with a
747
- // pinned hostname) — adopt whatever id it returns so the realtime
748
- // channel, heartbeats, and polling all agree.
749
- if (tokenProvider.mode === "env") {
750
- const boot = createRunnerClient(cfg, session.access_token);
751
- const models = await detectInstalledModels();
752
- const { data: resolvedId, error: regErr } = await boot.rpc("register_runner", {
753
- p_id: session.runner_id,
754
- p_name: session.runner_id,
755
- p_owner: session.email ?? session.user_id,
756
- p_machine: machineIdentity(),
757
- p_models: models,
758
- p_caps: RUNNER_CAPS,
759
- p_version: PACKAGE_VERSION,
760
- });
761
- if (regErr) {
762
- process.stderr.write(`[acc-runner] ephemeral register_runner failed: ${regErr.message}\n`);
763
- process.exit(1);
764
- }
765
- if (typeof resolvedId === "string" && resolvedId && resolvedId !== session.runner_id) {
766
- session = { ...session, runner_id: resolvedId };
767
- }
768
- await tokenProvider.save(session);
769
- console.log(chalk.green(`✓ Registered ephemeral runner: ${session.runner_id}`));
770
- }
771
- const taskRunnerFactory = options.taskRunnerFactory ?? runTask;
772
- const heartbeatMs = options.heartbeatMs ?? 4_000;
773
- const pollMs = options.pollMs ?? POLL_INTERVAL_MS;
774
- const reviewerFactory = options.reviewerFactory ?? ((assignment, deps) => runReview(assignment, deps));
775
- // v0.48: check for an existing quarantine file before opening the
776
- // realtime connection. If quarantined, emit a loud warning — the
777
- // runner will not claim tasks until `acc-runner quarantine clear` runs.
778
- const startupQuarantine = await getQuarantine().catch(() => null);
779
- if (startupQuarantine) {
780
- process.stderr.write(chalk.red.bold(`[acc-runner] QUARANTINED (${startupQuarantine.cause}): ${startupQuarantine.detail}\n`));
781
- process.stderr.write(chalk.red(`[acc-runner] Runner will NOT claim tasks. Resolve the issue then run ` +
782
- `\`acc-runner quarantine clear\`.\n`));
783
- }
784
- // v0.12.0 (T-52-7): idle TTL — option override wins (tests), then env.
785
- const idleTtlMs = options.idleTtlMs !== undefined ? options.idleTtlMs : resolveIdleTtlMs();
786
- const state = {
787
- cfg,
788
- session,
789
- tokenProvider,
790
- supabase: createRunnerClient(cfg, session.access_token),
791
- channel: undefined,
792
- heartbeatTimer: undefined,
793
- refreshTimer: undefined,
794
- pollTimer: undefined,
795
- claimScanTimer: undefined, // v0.62 (T-62-1)
796
- claimScanIntervalMs: options.claimScanIntervalMs ?? DEFAULT_CLAIM_SCAN_INTERVAL_MS,
797
- claimScanning: false,
798
- reviewScanTimer: undefined, // v0.65 (T-65-2)
799
- reviewScanIntervalMs: options.reviewScanIntervalMs ?? DEFAULT_CLAIM_SCAN_INTERVAL_MS,
800
- reviewScanning: false,
801
- // v0.73 (T-73-2, G-M): claim-loop watchdog.
802
- claimWatchdogTimer: undefined,
803
- claimWatchdogStaleMs: options.claimWatchdogStaleMs ?? DEFAULT_CLAIM_WATCHDOG_STALE_MS,
804
- claimWatchdogSince: new Map(),
805
- claimWatchdogChecking: false,
806
- // v0.68 (T-68-1): realtime channel-health recovery state.
807
- resubscribing: false,
808
- resubscribeAttempts: 0,
809
- resubscribeTimer: null,
810
- // v0.10 T-49-2: concurrent running Map replaces v0.9 single `current`.
811
- running: new Map(),
812
- concurrencyLimit: options.concurrencyLimit ?? 2,
813
- queue: [],
814
- seen: new Set(),
815
- pumping: false,
816
- stopped: false,
817
- heartbeatFailures: 0,
818
- refreshing: null,
819
- pollSince: new Date(Date.now() - STARTUP_POLL_LOOKBACK_MS).toISOString(),
820
- polling: false,
821
- reviewQueue: [],
822
- seenReviews: new Set(),
823
- reviewPumping: false,
824
- runReview: reviewerFactory,
825
- quarantined: startupQuarantine !== null,
826
- authRetriedTasks: new Set(),
827
- idleTtlMs,
828
- lastActivityMs: Date.now(),
829
- idleTimer: null,
830
- idleExiting: false,
831
- taskRunnerFactory,
832
- pausedCapacity: false,
833
- pausedUntil: null,
834
- pauseTimer: null,
835
- capacityProbe: false,
836
- capacityBackoffMs: options.capacityBackoffMs ?? DEFAULT_CAPACITY_BACKOFF_MS,
837
- resumeJitterMs: options.resumeJitterMs ?? DEFAULT_RESUME_JITTER_MS,
838
- singletonLock,
839
- };
840
- subscribeChannel(state, taskRunnerFactory);
841
- // v0.34-B / v0.62 (T-62-1): one-shot startup claim scan. The 5-minute
842
- // pollSince watermark (v0.30-B) can miss a task dispatched (runner_id set)
843
- // more than 5 minutes before the runner started whose realtime task_assigned
844
- // event was lost during the restart window. claimScan() scans with an epoch
845
- // watermark so any task currently assigned to this runner_id is enqueued
846
- // regardless of dispatch age. Does NOT update state.pollSince — the ongoing
847
- // poll loop keeps its 5-min watermark. The same routine runs periodically
848
- // below as the realtime-miss safety net.
849
- await claimScan(state, taskRunnerFactory, "startup");
850
- // v0.65 (T-65-2): one-shot startup review scan, the review twin of the
851
- // startup claim scan above. A `review_assigned` broadcast dropped during the
852
- // restart window would otherwise strand the review in acc.review_queue until
853
- // a manual restart; the epoch-watermark scan re-enqueues any pending review
854
- // assigned to this runner regardless of when it was requested. enqueueReview's
855
- // seenReviews guard keeps it idempotent against the realtime path.
856
- await reviewScan(state, "startup");
857
- // v0.30-B: fire an immediate poll right after subscribing so any task
858
- // dispatched before the runner came online (where the fire-and-forget
859
- // Realtime broadcast went to a dead channel) is picked up at startup
860
- // instead of waiting `pollMs` for the first scheduled poll. Awaited so
861
- // `state.polling` is back to false before the timer-driven polls and
862
- // any caller-driven `pollOnce()` can run.
863
- await pollOnce(state, taskRunnerFactory);
864
- // v0.6.0: self-rescheduling setTimeout (not setInterval) so the
865
- // backoff ladder in nextHeartbeatDelayMs actually pauses between
866
- // attempts. setInterval would keep firing every heartbeatMs even
867
- // while the link is down.
868
- scheduleHeartbeat(state, taskRunnerFactory, heartbeatMs);
869
- // v0.12.0 (T-52-7): no refresh timer when the provider can't rotate —
870
- // doRefresh would just emit the same "unavailable" line every 30 min.
871
- state.refreshTimer = setInterval(() => {
872
- if (state.tokenProvider.canRefresh() && expiresSoon(state.session)) {
873
- void doRefresh(state, taskRunnerFactory);
874
- }
875
- }, REFRESH_CHECK_MS);
876
- state.pollTimer = setInterval(() => {
877
- void pollOnce(state, taskRunnerFactory);
878
- }, pollMs);
879
- // v0.62 (T-62-1): periodic full claim scan — the realtime-miss safety net.
880
- // Independent of the delta-poll watermark, so a task assigned with an
881
- // updated_at at/behind the watermark (and whose task_assigned broadcast was
882
- // dropped) is re-claimed within one interval rather than waiting for a
883
- // manual restart. Single-flight inside claimScan(); enqueue's seen-guard
884
- // keeps it from double-claiming work the realtime path or delta-poll took.
885
- state.claimScanTimer = setInterval(() => {
886
- void claimScan(state, taskRunnerFactory, "periodic");
887
- }, state.claimScanIntervalMs);
888
- // v0.65 (T-65-2): periodic full review scan — the realtime-miss safety net
889
- // for reviews. Mirrors the claim-scan timer above: a `review_assigned`
890
- // broadcast missed by a stale subscription is recovered within one interval
891
- // instead of stranding in acc.review_queue. Single-flight inside
892
- // reviewScan(); enqueueReview's seenReviews guard prevents double-claiming a
893
- // review the realtime path already took.
894
- state.reviewScanTimer = setInterval(() => {
895
- void reviewScan(state, "periodic");
896
- }, state.reviewScanIntervalMs);
897
- // v0.73 (T-73-2, G-M): claim-loop watchdog timer. Independent of the claim
898
- // scan (whose 3-min cadence is too coarse for a 90s threshold): it ages the
899
- // runner's own assigned-but-unclaimed tasks and forces a resubscribe + claim
900
- // scan when one stays unclaimed past the threshold while heartbeats succeed —
901
- // the self-heal for the "online + heartbeating but never claims" strand that
902
- // T-68-1's channel recovery can't see (healthy channel, wedged claim path).
903
- const claimWatchdogCheckMs = options.claimWatchdogCheckMs ?? DEFAULT_CLAIM_WATCHDOG_CHECK_MS;
904
- state.claimWatchdogTimer = setInterval(() => {
905
- void claimWatchdogScan(state, taskRunnerFactory);
906
- }, claimWatchdogCheckMs);
907
- // v0.13-MULTI-RUNNER: refresh the runner's capability tags so an
908
- // operator change to ACC_RUNNER_CAPABILITIES (or to ~/.config/acc-
909
- // runner/config.json) takes effect on the next restart without
910
- // requiring a full `acc-runner login` re-flow. Default-empty is the
911
- // backward-compat path for 0.6.3 runners and matches any task with
912
- // an empty required_capabilities array.
913
- {
914
- const { error: capsErr } = await state.supabase.rpc("set_runner_capabilities", {
915
- p_id: state.session.runner_id,
916
- p_capabilities: cfg.capabilities ?? [],
917
- });
918
- if (capsErr) {
919
- // Non-fatal: a pre-0114 server (or an Old DB the runner is
920
- // pointed at during local dev) will 404 this RPC. The watch
921
- // loop continues so an operator on an older deploy isn't
922
- // wedged by a missing function.
923
- process.stderr.write(`[acc-runner] set_runner_capabilities failed (continuing): ${capsErr.message}\n`);
924
- }
925
- }
926
- // Initial heartbeat fires immediately so the runner page reflects "online".
927
- await state.supabase.rpc("heartbeat_runner", {
928
- p_id: state.session.runner_id,
929
- p_version: PACKAGE_VERSION,
930
- });
931
- console.log(chalk.gray(`Runner ${state.session.runner_id} is online. Ctrl-C to stop.`));
932
- const stop = async () => {
933
- if (state.stopped)
934
- return;
935
- state.stopped = true;
936
- clearTimeout(state.heartbeatTimer);
937
- clearInterval(state.refreshTimer);
938
- clearInterval(state.pollTimer);
939
- clearInterval(state.claimScanTimer); // v0.62 (T-62-1)
940
- clearInterval(state.reviewScanTimer); // v0.65 (T-65-2)
941
- clearInterval(state.claimWatchdogTimer); // v0.73 (T-73-2)
942
- if (state.resubscribeTimer)
943
- clearTimeout(state.resubscribeTimer); // v0.68 (T-68-1)
944
- if (state.idleTimer)
945
- clearInterval(state.idleTimer);
946
- if (state.pauseTimer)
947
- clearTimeout(state.pauseTimer); // v0.56 (T-56-1)
948
- // v0.10 T-49-2: cancel every in-flight task (replaces v0.9 single cancel).
949
- for (const ctrl of state.running.values()) {
950
- ctrl.cancel();
951
- }
952
- try {
953
- await state.supabase.removeChannel(state.channel);
954
- }
955
- catch { /* ignore */ }
956
- // v0.12-RUNNER-HEALTH: log shutdown_intent BEFORE flipping status so
957
- // the audit trail records the intent even if the status RPC fails.
958
- // The 1-min runner-watchdog cron uses status='online'+stale heartbeat
959
- // as its crash detector; this event gives operators a timestamped
960
- // breadcrumb when investigating "why did this runner go offline?"
961
- try {
962
- await state.supabase.rpc("log_activity", {
963
- p_verb: "runner.shutdown_intent",
964
- p_target_id: state.session.runner_id,
965
- p_payload: { version: PACKAGE_VERSION },
966
- p_target_type: "runner",
967
- });
968
- }
969
- catch { /* best-effort; network may already be gone */ }
970
- try {
971
- await state.supabase.rpc("set_runner_status", {
972
- p_id: state.session.runner_id,
973
- p_status: "offline",
974
- });
975
- }
976
- catch { /* network may already be gone */ }
977
- // v0.14.0 (T-54-2): release the single-instance lock last, so the slot
978
- // is only freed once this runner has fully deregistered. SIGINT/SIGTERM
979
- // and the idle-TTL exit all route through stop(), so all clean shutdown
980
- // paths release it. A hard crash leaves a stale lock that the next
981
- // start reclaims.
982
- try {
983
- await state.singletonLock?.release();
984
- }
985
- catch { /* best-effort */ }
986
- };
987
- // v0.12.0 (T-52-7): arm the idle-TTL timer. Desktop runners (no TTL env,
988
- // no option) never reach this branch — zero behaviour change. Armed after
989
- // `stop` exists so the exit sequence can reuse the one shutdown path.
990
- if (state.idleTtlMs !== null && state.idleTtlMs > 0) {
991
- const checkMs = options.idleCheckMs ?? Math.min(IDLE_CHECK_MS, state.idleTtlMs);
992
- state.idleTimer = setInterval(() => {
993
- void maybeIdleExit(state, stop, exitFn);
994
- }, checkMs);
995
- // Don't hold the event loop open for the idle check alone.
996
- state.idleTimer.unref?.();
997
- console.log(chalk.gray(`Idle TTL armed: exiting after ${Math.round(state.idleTtlMs / 60_000)} min without work.`));
998
- }
999
- if (!options.taskRunnerFactory) {
1000
- // v0.33-C: best-effort terminal status flush on runner shutdown. If
1001
- // launchd (or another OS-driven kill) sends SIGTERM while a task is
1002
- // mid-run, transition_task('failed') as a last resort so the task
1003
- // doesn't sit in 'running' until the 5-15 min sweep fires. SIGINT
1004
- // (Ctrl-C, operator-driven) skips the flush — the sweep is the right
1005
- // path there because the task should be retried as-is, not failed.
1006
- // SIGKILL / OOM can't be caught, so the sweep is still the real
1007
- // safety net; this just shrinks the window when the kill is catchable.
1008
- const onSignal = (signal) => {
1009
- void (async () => {
1010
- // v0.10 T-49-2: on SIGTERM, transition ALL in-flight tasks to
1011
- // failed (replaces v0.9 single-task transition). SIGINT (Ctrl-C)
1012
- // still skips the flush — the 5-min sweep handles that path.
1013
- if (signal === "SIGTERM" && state.running.size > 0) {
1014
- await Promise.allSettled([...state.running.keys()].map(async (taskId) => {
1015
- try {
1016
- await state.supabase.rpc("transition_task", {
1017
- p_task_id: taskId,
1018
- p_new_status: "failed",
1019
- });
1020
- }
1021
- catch { /* best-effort — sweep covers us if RPC can't get out */ }
1022
- }));
1023
- }
1024
- await stop();
1025
- process.exit(0);
1026
- })();
1027
- };
1028
- process.once("SIGINT", () => onSignal("SIGINT"));
1029
- process.once("SIGTERM", () => onSignal("SIGTERM"));
1030
- }
1031
- return {
1032
- stop,
1033
- pollOnce: () => pollOnce(state, taskRunnerFactory),
1034
- claimScanOnce: () => claimScan(state, taskRunnerFactory, "periodic"), // v0.62 (T-62-1)
1035
- reviewScanOnce: () => reviewScan(state, "periodic"), // v0.65 (T-65-2)
1036
- claimWatchdogOnce: () => claimWatchdogScan(state, taskRunnerFactory), // v0.73 (T-73-2)
1037
- runningCount: () => state.running.size,
1038
- };
1039
- }
1040
- function scheduleHeartbeat(state, taskRunnerFactory, baseMs) {
1041
- if (state.stopped)
1042
- return;
1043
- const delay = nextHeartbeatDelayMs(state.heartbeatFailures, baseMs);
1044
- state.heartbeatTimer = setTimeout(() => {
1045
- void (async () => {
1046
- const { error } = await state.supabase.rpc("heartbeat_runner", {
1047
- p_id: state.session.runner_id,
1048
- p_version: PACKAGE_VERSION,
1049
- });
1050
- if (!error) {
1051
- state.heartbeatFailures = 0;
1052
- }
1053
- else {
1054
- state.heartbeatFailures += 1;
1055
- const kind = classifyHeartbeatError(error);
1056
- process.stderr.write(`[acc-runner] heartbeat failed (${kind}): ${error.message}\n`);
1057
- // v0.6.0 REG-294 sibling: only refresh on an auth-shaped error
1058
- // AND once we've crossed the consecutive-failure threshold. A
1059
- // single 401 on a flaky network shouldn't burn a refresh, and a
1060
- // DNS outage shouldn't trigger refresh at all.
1061
- if (state.heartbeatFailures >= HEARTBEAT_REFRESH_AFTER &&
1062
- kind === "auth") {
1063
- process.stderr.write("[acc-runner] auth-shaped heartbeat failure, attempting JWT refresh\n");
1064
- void doRefresh(state, taskRunnerFactory);
1065
- }
1066
- if (state.heartbeatFailures >= HEARTBEAT_EXIT_AFTER) {
1067
- process.stderr.write("[acc-runner] heartbeat dead, exiting\n");
1068
- clearTimeout(state.heartbeatTimer);
1069
- clearInterval(state.refreshTimer);
1070
- clearInterval(state.pollTimer);
1071
- clearInterval(state.claimScanTimer); // v0.62 (T-62-1)
1072
- clearInterval(state.reviewScanTimer); // v0.65 (T-65-2)
1073
- clearInterval(state.claimWatchdogTimer); // v0.73 (T-73-2)
1074
- if (state.resubscribeTimer)
1075
- clearTimeout(state.resubscribeTimer); // v0.68 (T-68-1)
1076
- process.exit(1);
1077
- }
1078
- }
1079
- scheduleHeartbeat(state, taskRunnerFactory, baseMs);
1080
- })();
1081
- }, delay);
1082
- }
1083
- /**
1084
- * v0.12.0 (T-52-7): idle-TTL exit check. Fires from the idle timer; exits
1085
- * the process cleanly when the runner has had no queued or in-flight work
1086
- * (tasks OR reviews) for idleTtlMs. The exit sequence:
1087
- *
1088
- * 1. log_activity runner.idle_ttl_exit — the audit breadcrumb the
1089
- * autoscale sweep and operators read ("why did acc-eph-x vanish?").
1090
- * 2. final heartbeat_runner — so the runners panel shows a fresh
1091
- * timestamp right up to the clean exit (vs. a stale-heartbeat crash).
1092
- * 3. stop() — shutdown_intent + set_runner_status('offline'), the same
1093
- * deregistration path `acc-runner logout` and SIGTERM use.
1094
- * 4. exit(0).
1095
- *
1096
- * In-flight work always wins: any running task, queued task, or queued
1097
- * review resets the decision to "not idle" — the TTL never kills a runner
1098
- * mid-task.
1099
- */
1100
- async function maybeIdleExit(state, stop, exitFn) {
1101
- if (state.stopped || state.idleExiting || state.idleTtlMs === null)
1102
- return;
1103
- // v0.56 (T-56-1): a capacity pause is not idleness — the runner is
1104
- // deliberately waiting for its session window to reopen. Don't TTL-exit
1105
- // it out from under the scheduled resume.
1106
- if (state.pausedCapacity)
1107
- return;
1108
- if (state.running.size > 0 ||
1109
- state.queue.length > 0 ||
1110
- state.reviewQueue.length > 0 ||
1111
- state.reviewPumping) {
1112
- return;
1113
- }
1114
- const idleMs = Date.now() - state.lastActivityMs;
1115
- if (idleMs < state.idleTtlMs)
1116
- return;
1117
- state.idleExiting = true;
1118
- const idleMinutes = Math.round(idleMs / 60_000);
1119
- console.log(chalk.gray(`[acc-runner] idle TTL reached (${idleMinutes} min without work); exiting cleanly.`));
1120
- try {
1121
- await state.supabase.rpc("log_activity", {
1122
- p_verb: "runner.idle_ttl_exit",
1123
- p_target_id: state.session.runner_id,
1124
- p_payload: { idle_minutes: idleMinutes, version: PACKAGE_VERSION },
1125
- p_target_type: "runner",
1126
- });
1127
- }
1128
- catch { /* best-effort */ }
1129
- try {
1130
- await state.supabase.rpc("heartbeat_runner", {
1131
- p_id: state.session.runner_id,
1132
- p_version: PACKAGE_VERSION,
1133
- });
1134
- }
1135
- catch { /* best-effort */ }
1136
- await stop();
1137
- exitFn(0);
1138
- }
1139
- async function pumpReviews(state) {
1140
- if (state.reviewPumping)
1141
- return;
1142
- // v0.56 (T-56-1): don't claim reviews while paused for capacity, and hold
1143
- // reviews back during the single-probe window so the probe TASK confirms
1144
- // the session reopened before we spend more capacity on reviews.
1145
- if (state.pausedCapacity || state.capacityProbe)
1146
- return;
1147
- state.reviewPumping = true;
1148
- try {
1149
- while (!state.stopped &&
1150
- !state.pausedCapacity &&
1151
- !state.capacityProbe &&
1152
- state.reviewQueue.length > 0) {
1153
- const next = state.reviewQueue.shift();
1154
- if (!next)
1155
- continue;
1156
- try {
1157
- const outcome = await state.runReview(next, { supabase: state.supabase });
1158
- const tag = outcome.decision === "reviewer_error" ? chalk.red : chalk.gray;
1159
- console.log(tag(`[acc-runner] review ${next.review_id} task=${next.task_id} pr=${next.pr_number} decision=${outcome.decision} confidence=${outcome.confidence.toFixed(2)}`));
1160
- // v0.56 (T-56-1): a reviewer that hit the session/usage limit reports
1161
- // a DISTINCT outcome (reviewer_capacity, NOT reviewer_error) so the
1162
- // api side (T-56-2) can exclude it from automerge retry caps and
1163
- // re-dispatch the review. runReview has already submitted the
1164
- // reviewer_capacity row; here we pause the whole runner exactly as a
1165
- // capacity_exhausted task does.
1166
- if (outcome.decision === "reviewer_capacity") {
1167
- enterCapacityPause(state, next.task_id, outcome.resume_at ?? null, `reviewer ${next.review_id} hit session/usage limit`);
1168
- return;
1169
- }
1170
- }
1171
- catch (err) {
1172
- process.stderr.write(`[acc-runner] review ${next.review_id} crashed: ${err.message}\n`);
1173
- }
1174
- finally {
1175
- touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at review end
1176
- }
1177
- }
1178
- }
1179
- finally {
1180
- state.reviewPumping = false;
1181
- }
1182
- }
1183
- /**
1184
- * v0.48 / v0.49 T-49-6: enter quarantine after a machine-level failure.
1185
- *
1186
- * Extracted from pump()'s completion handler so both the immediate
1187
- * quarantine path and the auth_expired "refresh failed → quarantine"
1188
- * path share one implementation. Idempotent: a no-op once the runner is
1189
- * already quarantined, so concurrent machine-level failures don't fire
1190
- * duplicate audit events. Sets the in-memory flag synchronously (pump()
1191
- * reads it without an async hop) and best-effort persists + announces.
1192
- */
1193
- function enterQuarantine(state, taskId, cause, detail,
1194
- // v0.53 T-53-4: how many consecutive instant-empty exits produced an
1195
- // env_broken cause (1 = definitive, 2 = heuristic confirmed by the
1196
- // retry). Persisted into quarantine.json's additive `consecutive` field.
1197
- consecutive) {
1198
- if (state.quarantined)
1199
- return;
1200
- state.quarantined = true;
1201
- const qState = {
1202
- cause,
1203
- classifiedAt: new Date().toISOString(),
1204
- taskId,
1205
- detail,
1206
- ...(consecutive !== undefined ? { consecutive } : {}),
1207
- };
1208
- void setQuarantine(qState).catch((err) => {
1209
- process.stderr.write(`[acc-runner] setQuarantine failed: ${err.message}\n`);
1210
- });
1211
- process.stderr.write(chalk.red.bold(`[acc-runner] QUARANTINE: runner ${state.session.runner_id} quarantined ` +
1212
- `(${cause}) after task ${taskId} failed.\n`));
1213
- void postRunnerStateMessage(state.supabase, {
1214
- task_id: taskId,
1215
- sender_id: state.session.runner_id,
1216
- state: "blocked",
1217
- protocol: "error_context",
1218
- payload: {
1219
- capacity_alert: true,
1220
- quarantine_cause: cause,
1221
- runner_id: state.session.runner_id,
1222
- detail,
1223
- },
1224
- kind: "blocked",
1225
- }).catch((err) => {
1226
- process.stderr.write(`[acc-runner] capacity alert post failed: ${err.message}\n`);
1227
- });
1228
- void (async () => {
1229
- try {
1230
- await state.supabase.rpc("log_activity", {
1231
- p_verb: "runner.quarantine_enter",
1232
- p_target_id: state.session.runner_id,
1233
- p_payload: { cause, task_id: taskId, detail },
1234
- p_target_type: "runner",
1235
- });
1236
- }
1237
- catch { /* best-effort */ }
1238
- })();
1239
- }
1240
- /**
1241
- * v0.56 (T-56-1): enter a capacity pause.
1242
- *
1243
- * Triggered when a task OR a review reports capacity_exhausted (Claude
1244
- * account out of session/usage capacity). The runner:
1245
- * - stops claiming tasks AND reviews (pump/pumpReviews bail on the flag);
1246
- * - schedules an automatic resume at the stated reset time (or a default
1247
- * backoff when none was parseable) plus a small jitter;
1248
- * - keeps heartbeating so the board shows online-but-paused, and emits a
1249
- * `runner.paused_capacity` activity event + a runner-level bus message
1250
- * carrying `resume_at` so health/board surfaces can show WHY the fleet
1251
- * is idle (heartbeat_runner's RPC shape is frozen, so the detail is
1252
- * carried additively through these existing channels).
1253
- *
1254
- * Idempotent: a second capacity hit while already paused only ever pushes
1255
- * the resume later, never earlier, so a straggler task that 429s a moment
1256
- * after the first one can't shorten the wait.
1257
- */
1258
- function enterCapacityPause(state, triggerTaskId, resumeAtIso, detail) {
1259
- const now = Date.now();
1260
- const parsed = resumeAtIso ? Date.parse(resumeAtIso) : NaN;
1261
- const resumeMs = Number.isFinite(parsed) && parsed > now ? parsed : now + state.capacityBackoffMs;
1262
- const source = Number.isFinite(parsed) && parsed > now ? "reset_time" : "default_backoff";
1263
- // Already paused: only extend the window, never shorten it.
1264
- if (state.pausedCapacity && state.pausedUntil !== null && resumeMs <= state.pausedUntil) {
1265
- return;
1266
- }
1267
- state.pausedCapacity = true;
1268
- state.pausedUntil = resumeMs;
1269
- if (state.pauseTimer) {
1270
- clearTimeout(state.pauseTimer);
1271
- state.pauseTimer = null;
1272
- }
1273
- const jitter = state.resumeJitterMs > 0 ? Math.floor(Math.random() * state.resumeJitterMs) : 0;
1274
- const delay = Math.max(0, resumeMs - now) + jitter;
1275
- const resumeAtFinal = new Date(now + delay).toISOString();
1276
- process.stderr.write(chalk.yellow(`[acc-runner] PAUSED (capacity): ${detail}. Not claiming tasks or reviews; ` +
1277
- `resuming at ${resumeAtFinal} (${source}).\n`));
1278
- void (async () => {
1279
- try {
1280
- await state.supabase.rpc("log_activity", {
1281
- p_verb: "runner.paused_capacity",
1282
- p_target_id: state.session.runner_id,
1283
- p_payload: {
1284
- resume_at: resumeAtFinal,
1285
- source,
1286
- detail,
1287
- version: PACKAGE_VERSION,
1288
- },
1289
- p_target_type: "runner",
1290
- });
1291
- }
1292
- catch { /* best-effort */ }
1293
- })();
1294
- // Runner-level bus message so operators see the paused status + reason on
1295
- // /messages even though heartbeat_runner itself can't carry a detail field.
1296
- // Anchored to the triggering task so the RPC can resolve the org scope
1297
- // (post_agent_message looks up org_id from acc.tasks for a non-null id).
1298
- void postRunnerStateMessage(state.supabase, {
1299
- task_id: triggerTaskId,
1300
- sender_id: state.session.runner_id,
1301
- state: "blocked",
1302
- protocol: "info",
1303
- payload: {
1304
- capacity_paused: true,
1305
- runner_id: state.session.runner_id,
1306
- resume_at: resumeAtFinal,
1307
- source,
1308
- detail,
1309
- },
1310
- kind: "blocked",
1311
- }).catch((err) => {
1312
- process.stderr.write(`[acc-runner] capacity pause bus post failed: ${err.message}\n`);
1313
- });
1314
- state.pauseTimer = setTimeout(() => {
1315
- void resumeFromCapacity(state);
1316
- }, delay);
1317
- state.pauseTimer.unref?.();
1318
- }
1319
- /**
1320
- * v0.56 (T-56-1): resume from a capacity pause.
1321
- *
1322
- * Clears the pause and arms the single-probe clamp: the runner claims at
1323
- * most ONE task to confirm the window actually reopened before resuming
1324
- * full task concurrency and reviews. If the probe itself 429s, the
1325
- * completion handler re-enters the pause; if it succeeds, the probe clamp
1326
- * clears and normal claiming resumes.
1327
- */
1328
- async function resumeFromCapacity(state) {
1329
- if (state.stopped)
1330
- return;
1331
- state.pausedCapacity = false;
1332
- state.pausedUntil = null;
1333
- if (state.pauseTimer) {
1334
- clearTimeout(state.pauseTimer);
1335
- state.pauseTimer = null;
1336
- }
1337
- state.capacityProbe = true;
1338
- console.log(chalk.gray(`[acc-runner] capacity window reopened — resuming with a single probe task.`));
1339
- try {
1340
- await state.supabase.rpc("log_activity", {
1341
- p_verb: "runner.resume_capacity",
1342
- p_target_id: state.session.runner_id,
1343
- p_payload: { probe: true, version: PACKAGE_VERSION },
1344
- p_target_type: "runner",
1345
- });
1346
- }
1347
- catch { /* best-effort */ }
1348
- // Poll first (the stale-running sweep may have re-queued the released task
1349
- // to another runner; this picks up whatever is still assigned here), then
1350
- // pump — the probe clamp bounds it to one task.
1351
- await pollOnce(state, state.taskRunnerFactory);
1352
- void pump(state, state.taskRunnerFactory);
1353
- }
1354
- /**
1355
- * v0.10 T-49-2: concurrent task pump.
1356
- *
1357
- * Launches up to state.concurrencyLimit tasks in parallel. Each task
1358
- * gets its own isolated git worktree (prepareTaskWorktree in task-runner)
1359
- * so there is zero shared-file risk between concurrent runs. Database-side
1360
- * `claim_task_with_locks` provides a second guard: it rejects a claim when
1361
- * the task's file scope overlaps another in-flight task.
1362
- *
1363
- * Design: rather than `await`ing each task inside a while loop (v0.9
1364
- * serial behaviour), tasks are launched fire-and-forget. Each task's
1365
- * `.then()` handler removes it from `state.running` and re-fires pump()
1366
- * so newly freed capacity immediately picks up the next queued item.
1367
- * The `pumping` mutex prevents two concurrent pump() invocations from
1368
- * both launching into the same concurrencyLimit slot.
1369
- */
1370
- async function pump(state, factory) {
1371
- if (state.pumping)
1372
- return;
1373
- state.pumping = true;
1374
- try {
1375
- // v0.48: check quarantine before claiming any task.
1376
- if (state.quarantined) {
1377
- process.stderr.write(`[acc-runner] QUARANTINED: runner ${state.session.runner_id} is not claiming tasks.\n` +
1378
- `[acc-runner] Fix the underlying issue then run \`acc-runner quarantine clear\`.\n`);
1379
- return;
1380
- }
1381
- // v0.56 (T-56-1): out of session/usage capacity — don't claim. Queued
1382
- // tasks stay queued; the scheduled resume re-drives pump().
1383
- if (state.pausedCapacity) {
1384
- return;
1385
- }
1386
- // v0.56 (T-56-1): during the post-resume probe window, claim exactly one
1387
- // task to confirm the session reopened before unleashing full concurrency.
1388
- const effectiveLimit = state.capacityProbe ? 1 : state.concurrencyLimit;
1389
- // Fill available concurrency slots from the queue.
1390
- while (!state.stopped &&
1391
- !state.quarantined &&
1392
- !state.pausedCapacity &&
1393
- state.queue.length > 0 &&
1394
- state.running.size < effectiveLimit) {
1395
- const next = state.queue.shift();
1396
- if (!next)
1397
- continue;
1398
- const ctrl = factory(next, {
1399
- supabase: state.supabase,
1400
- cfg: state.cfg,
1401
- session: {
1402
- accessToken: state.session.access_token,
1403
- runnerId: state.session.runner_id,
1404
- },
1405
- publicUrl: state.cfg.publicUrl,
1406
- });
1407
- state.running.set(next, ctrl);
1408
- // Detached completion handler. Runs AFTER pump() returns so the
1409
- // running Map slot is occupied by the time the while condition
1410
- // re-checks running.size on the next iteration.
1411
- void ctrl.promise
1412
- .then((outcome) => {
1413
- state.running.delete(next);
1414
- touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at task end
1415
- // v0.56 (T-56-1): capacity exhaustion pauses the WHOLE runner —
1416
- // never a task_error, never quarantine. The task was left 'running'
1417
- // for the stale-running sweep to requeue losslessly (runner_id
1418
- // cleared). Do not re-fire pump for new work; the scheduled resume
1419
- // does that with a single probe.
1420
- if (outcome.status === "capacity_paused" || outcome.capacity_exhausted) {
1421
- enterCapacityPause(state, next, outcome.resume_at ?? null, outcome.error || "task hit session/usage limit");
1422
- return;
1423
- }
1424
- // v0.21 T-66-1: a transient git-provision contention requeue. runTask
1425
- // already returned the task to 'queued' server-side for a clean
1426
- // retry — NOT a failure (no quarantine, no pause, no retry-budget
1427
- // burn). Fall through to the pump re-fire below so the runner stays
1428
- // healthy and the requeued task re-dispatches normally.
1429
- if (outcome.status === "requeued") {
1430
- process.stderr.write(`[acc-runner] task ${next} requeued (transient): ${outcome.error ?? "git provision contention"}\n`);
1431
- }
1432
- // v0.14.0 (T-54-2, GA-6): record the claude CLI version at the last
1433
- // successful task so `doctor` can WARN when claude auto-updates
1434
- // under the runner. Best-effort and detached — never blocks pump.
1435
- if (outcome.status === "ok") {
1436
- void getClaudeVersion()
1437
- .then((v) => recordTaskClaudeVersion(v))
1438
- .catch(() => {
1439
- /* best-effort cache write */
1440
- });
1441
- }
1442
- if (outcome.status === "failed") {
1443
- const reason = outcome.error?.trim() ||
1444
- (outcome.exitCode != null ? `exit ${outcome.exitCode}` : "unknown");
1445
- process.stderr.write(`[acc-runner] task ${next} phase=${outcome.phase ?? "unknown"} failed: ${reason}\n`);
1446
- // v0.48: machine-level failure → quarantine. Guard with
1447
- // !state.quarantined so concurrent tasks don't fire duplicate
1448
- // quarantine events when two machine-level failures land
1449
- // simultaneously (both already in-flight; only the first to
1450
- // resolve sets the flag and writes the audit trail).
1451
- if (outcome.quarantine_cause && !state.quarantined) {
1452
- const cause = outcome.quarantine_cause;
1453
- // v0.49 T-49-6: a transient auth_expired (the access token
1454
- // lapsed while the task ran) gets ONE JWT refresh before we
1455
- // quarantine. If the refresh recovers, the runner stays
1456
- // online and resumes claiming queued work; if it fails — or
1457
- // this task already burned its one retry — we quarantine.
1458
- // usage_limit / env_broken skip this path: a token refresh
1459
- // can't fix a spent quota or a broken host.
1460
- if (cause === "auth_expired" && !state.authRetriedTasks.has(next)) {
1461
- state.authRetriedTasks.add(next);
1462
- process.stderr.write(chalk.yellow(`[acc-runner] auth_expired after task ${next}; attempting one ` +
1463
- `token refresh before quarantine.\n`));
1464
- void doRefresh(state, factory).then((ok) => {
1465
- if (ok) {
1466
- process.stderr.write(chalk.green(`[acc-runner] token refresh recovered auth after task ${next}; ` +
1467
- `staying online.\n`));
1468
- void (async () => {
1469
- try {
1470
- await state.supabase.rpc("log_activity", {
1471
- p_verb: "runner.auth_retry_recovered",
1472
- p_target_id: state.session.runner_id,
1473
- p_payload: { task_id: next, cause },
1474
- p_target_type: "runner",
1475
- });
1476
- }
1477
- catch { /* best-effort */ }
1478
- })();
1479
- // Resume claiming: the freed slot can pick up queued work
1480
- // now that the JWT is fresh.
1481
- void pump(state, factory);
1482
- }
1483
- else {
1484
- enterQuarantine(state, next, cause, reason, outcome.quarantine_consecutive);
1485
- }
1486
- });
1487
- return;
1488
- }
1489
- enterQuarantine(state, next, cause, reason, outcome.quarantine_consecutive);
1490
- // Quarantined — do not re-fire pump for new tasks.
1491
- // Other in-flight tasks (still in state.running) complete
1492
- // normally; we just stop accepting new work.
1493
- return;
1494
- }
1495
- }
1496
- // v0.56 (T-56-1): a non-capacity completion (ok, or a genuine
1497
- // task_error) proves the session window is open — clear the probe
1498
- // clamp and let reviews resume alongside full task concurrency.
1499
- if (state.capacityProbe) {
1500
- state.capacityProbe = false;
1501
- void pumpReviews(state);
1502
- }
1503
- // Task done (ok, failed without quarantine, or cancelled).
1504
- // Re-fire pump so the next queued task can claim the freed slot.
1505
- void pump(state, factory);
1506
- })
1507
- .catch((err) => {
1508
- state.running.delete(next);
1509
- touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at task end
1510
- process.stderr.write(`[acc-runner] task ${next} crashed: ${err.message}\n`);
1511
- // v0.56 (T-56-1): a crash is not a capacity signal — clear the probe
1512
- // clamp so the runner doesn't stay stuck at concurrency 1.
1513
- if (state.capacityProbe) {
1514
- state.capacityProbe = false;
1515
- void pumpReviews(state);
1516
- }
1517
- void pump(state, factory);
1518
- });
1519
- }
1520
- // v0.56 (T-56-1): the probe found nothing to test with (no queued task at
1521
- // resume) — clear the clamp and release any held reviews so an idle
1522
- // resume doesn't strand pending reviews behind a probe that never runs.
1523
- if (state.capacityProbe && state.running.size === 0 && state.queue.length === 0) {
1524
- state.capacityProbe = false;
1525
- void pumpReviews(state);
1526
- }
1527
- }
1528
- finally {
1529
- state.pumping = false;
1530
- }
1531
- }
1532
- //# sourceMappingURL=watch.js.map