@tokenfactory/acc-runner 0.25.1 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +77 -4
- package/package.json +1 -1
- package/dist/anthropic-auth.d.ts +0 -53
- package/dist/anthropic-auth.d.ts.map +0 -1
- package/dist/anthropic-auth.js +0 -89
- package/dist/anthropic-auth.js.map +0 -1
- package/dist/bin-resolve.d.ts +0 -16
- package/dist/bin-resolve.d.ts.map +0 -1
- package/dist/bin-resolve.js +0 -35
- package/dist/bin-resolve.js.map +0 -1
- package/dist/cli.d.ts +0 -3
- package/dist/cli.d.ts.map +0 -1
- package/dist/cli.js +0 -128
- package/dist/cli.js.map +0 -1
- package/dist/config.d.ts +0 -121
- package/dist/config.d.ts.map +0 -1
- package/dist/config.js +0 -396
- package/dist/config.js.map +0 -1
- package/dist/cost-pricing.d.ts +0 -130
- package/dist/cost-pricing.d.ts.map +0 -1
- package/dist/cost-pricing.js +0 -191
- package/dist/cost-pricing.js.map +0 -1
- package/dist/doctor.d.ts +0 -37
- package/dist/doctor.d.ts.map +0 -1
- package/dist/doctor.js +0 -658
- package/dist/doctor.js.map +0 -1
- package/dist/failure-classifier.d.ts +0 -112
- package/dist/failure-classifier.d.ts.map +0 -1
- package/dist/failure-classifier.js +0 -353
- package/dist/failure-classifier.js.map +0 -1
- package/dist/gh.d.ts +0 -22
- package/dist/gh.d.ts.map +0 -1
- package/dist/gh.js +0 -48
- package/dist/gh.js.map +0 -1
- package/dist/git.d.ts +0 -50
- package/dist/git.d.ts.map +0 -1
- package/dist/git.js +0 -127
- package/dist/git.js.map +0 -1
- package/dist/github-client.d.ts +0 -133
- package/dist/github-client.d.ts.map +0 -1
- package/dist/github-client.js +0 -234
- package/dist/github-client.js.map +0 -1
- package/dist/keychain.d.ts +0 -21
- package/dist/keychain.d.ts.map +0 -1
- package/dist/keychain.js +0 -45
- package/dist/keychain.js.map +0 -1
- package/dist/login.d.ts +0 -12
- package/dist/login.d.ts.map +0 -1
- package/dist/login.js +0 -133
- package/dist/login.js.map +0 -1
- package/dist/logout.d.ts +0 -2
- package/dist/logout.d.ts.map +0 -1
- package/dist/logout.js +0 -31
- package/dist/logout.js.map +0 -1
- package/dist/mcp-spawn.d.ts +0 -30
- package/dist/mcp-spawn.d.ts.map +0 -1
- package/dist/mcp-spawn.js +0 -145
- package/dist/mcp-spawn.js.map +0 -1
- package/dist/messaging.d.ts +0 -49
- package/dist/messaging.d.ts.map +0 -1
- package/dist/messaging.js +0 -36
- package/dist/messaging.js.map +0 -1
- package/dist/pkg-version.d.ts +0 -3
- package/dist/pkg-version.d.ts.map +0 -1
- package/dist/pkg-version.js +0 -20
- package/dist/pkg-version.js.map +0 -1
- package/dist/profiles/designer-prompt.d.ts +0 -18
- package/dist/profiles/designer-prompt.d.ts.map +0 -1
- package/dist/profiles/designer-prompt.js +0 -172
- package/dist/profiles/designer-prompt.js.map +0 -1
- package/dist/profiles/developer-prompt.d.ts +0 -24
- package/dist/profiles/developer-prompt.d.ts.map +0 -1
- package/dist/profiles/developer-prompt.js +0 -24
- package/dist/profiles/developer-prompt.js.map +0 -1
- package/dist/profiles/manager-prompt.d.ts +0 -34
- package/dist/profiles/manager-prompt.d.ts.map +0 -1
- package/dist/profiles/manager-prompt.js +0 -93
- package/dist/profiles/manager-prompt.js.map +0 -1
- package/dist/profiles/tester-prompt.d.ts +0 -28
- package/dist/profiles/tester-prompt.d.ts.map +0 -1
- package/dist/profiles/tester-prompt.js +0 -165
- package/dist/profiles/tester-prompt.js.map +0 -1
- package/dist/prompt.d.ts +0 -38
- package/dist/prompt.d.ts.map +0 -1
- package/dist/prompt.js +0 -78
- package/dist/prompt.js.map +0 -1
- package/dist/runtime/cache-dir.d.ts +0 -2
- package/dist/runtime/cache-dir.d.ts.map +0 -1
- package/dist/runtime/cache-dir.js +0 -15
- package/dist/runtime/cache-dir.js.map +0 -1
- package/dist/runtime/conflict-resolver.d.ts +0 -65
- package/dist/runtime/conflict-resolver.d.ts.map +0 -1
- package/dist/runtime/conflict-resolver.js +0 -477
- package/dist/runtime/conflict-resolver.js.map +0 -1
- package/dist/runtime/expand-args.d.ts +0 -28
- package/dist/runtime/expand-args.d.ts.map +0 -1
- package/dist/runtime/expand-args.js +0 -50
- package/dist/runtime/expand-args.js.map +0 -1
- package/dist/runtime/locks.d.ts +0 -21
- package/dist/runtime/locks.d.ts.map +0 -1
- package/dist/runtime/locks.js +0 -97
- package/dist/runtime/locks.js.map +0 -1
- package/dist/runtime/provision-mutex.d.ts +0 -37
- package/dist/runtime/provision-mutex.d.ts.map +0 -1
- package/dist/runtime/provision-mutex.js +0 -67
- package/dist/runtime/provision-mutex.js.map +0 -1
- package/dist/runtime/quarantine.d.ts +0 -26
- package/dist/runtime/quarantine.d.ts.map +0 -1
- package/dist/runtime/quarantine.js +0 -50
- package/dist/runtime/quarantine.js.map +0 -1
- package/dist/runtime/resolution-integrity.d.ts +0 -86
- package/dist/runtime/resolution-integrity.d.ts.map +0 -1
- package/dist/runtime/resolution-integrity.js +0 -248
- package/dist/runtime/resolution-integrity.js.map +0 -1
- package/dist/runtime/reviewer.d.ts +0 -81
- package/dist/runtime/reviewer.d.ts.map +0 -1
- package/dist/runtime/reviewer.js +0 -374
- package/dist/runtime/reviewer.js.map +0 -1
- package/dist/runtime/rework.d.ts +0 -48
- package/dist/runtime/rework.d.ts.map +0 -1
- package/dist/runtime/rework.js +0 -136
- package/dist/runtime/rework.js.map +0 -1
- package/dist/runtime/singleton.d.ts +0 -47
- package/dist/runtime/singleton.d.ts.map +0 -1
- package/dist/runtime/singleton.js +0 -200
- package/dist/runtime/singleton.js.map +0 -1
- package/dist/runtime/version-drift.d.ts +0 -31
- package/dist/runtime/version-drift.d.ts.map +0 -1
- package/dist/runtime/version-drift.js +0 -114
- package/dist/runtime/version-drift.js.map +0 -1
- package/dist/runtime/worktree.d.ts +0 -74
- package/dist/runtime/worktree.d.ts.map +0 -1
- package/dist/runtime/worktree.js +0 -206
- package/dist/runtime/worktree.js.map +0 -1
- package/dist/secrets/inject.d.ts +0 -70
- package/dist/secrets/inject.d.ts.map +0 -1
- package/dist/secrets/inject.js +0 -102
- package/dist/secrets/inject.js.map +0 -1
- package/dist/supabase.d.ts +0 -4
- package/dist/supabase.d.ts.map +0 -1
- package/dist/supabase.js +0 -36
- package/dist/supabase.js.map +0 -1
- package/dist/task-runner.d.ts +0 -313
- package/dist/task-runner.d.ts.map +0 -1
- package/dist/task-runner.js +0 -1766
- package/dist/task-runner.js.map +0 -1
- package/dist/token-provider.d.ts +0 -50
- package/dist/token-provider.d.ts.map +0 -1
- package/dist/token-provider.js +0 -177
- package/dist/token-provider.js.map +0 -1
- package/dist/types.d.ts +0 -120
- package/dist/types.d.ts.map +0 -1
- package/dist/types.js +0 -16
- package/dist/types.js.map +0 -1
- package/dist/version-check.d.ts +0 -17
- package/dist/version-check.d.ts.map +0 -1
- package/dist/version-check.js +0 -56
- package/dist/version-check.js.map +0 -1
- package/dist/watch.d.ts +0 -295
- package/dist/watch.d.ts.map +0 -1
- package/dist/watch.js +0 -1532
- package/dist/watch.js.map +0 -1
package/dist/watch.js
DELETED
|
@@ -1,1532 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* `acc-runner watch` — long-running command. Subscribes to broadcast
|
|
3
|
-
* tasks, dispatches them concurrently (v0.10+), heartbeats every 4s.
|
|
4
|
-
* v0.2.5 refreshes the access token on startup + every 30 min, and exits
|
|
5
|
-
* loudly after 5 consecutive heartbeat failures.
|
|
6
|
-
*
|
|
7
|
-
* v0.3 adds a 30s polling fallback: if a Realtime broadcast is dropped
|
|
8
|
-
* during a reconnect, the next poll pass picks the task up via the
|
|
9
|
-
* `acc.list_assigned_queued_tasks` RPC. A `seen` set keeps the poll
|
|
10
|
-
* and the broadcast from fighting over the same task_id.
|
|
11
|
-
*
|
|
12
|
-
* v0.6.2 (v0.12-REVIEW-LOCAL) listens for `review_assigned` on the same
|
|
13
|
-
* `runner:<id>` channel and dispatches to runtime/reviewer.ts. Reviews
|
|
14
|
-
* are independently tracked (separate `seenReviews` set).
|
|
15
|
-
*
|
|
16
|
-
* v0.10 (T-49-2): concurrent task dispatch. pump() now launches up to
|
|
17
|
-
* `concurrencyLimit` (default 2) tasks simultaneously. Each task runs in
|
|
18
|
-
* its own isolated git worktree so concurrent tasks never share files.
|
|
19
|
-
* `createTaskAsAdmin` in api/_lib/supabase-admin.ts is the sanctioned
|
|
20
|
-
* Manager seed path that feeds this dispatch loop.
|
|
21
|
-
*
|
|
22
|
-
* v0.12.0 (T-52-7): ephemeral-runner support. Sessions load through the
|
|
23
|
-
* token-provider abstraction (keychain by default — byte-identical desktop
|
|
24
|
-
* behaviour; ACC_RUNNER_TOKEN_MODE=env for containers). Env-token runners
|
|
25
|
-
* register an acc-eph-<random> identity on boot, and ACC_RUNNER_IDLE_TTL_MIN
|
|
26
|
-
* arms a clean self-termination after n minutes without work.
|
|
27
|
-
*/
|
|
28
|
-
import chalk from "chalk";
|
|
29
|
-
import { loadConfig } from "./config.js";
|
|
30
|
-
import { getTokenProvider } from "./token-provider.js";
|
|
31
|
-
import { machineIdentity, detectInstalledModels, RUNNER_CAPS } from "./login.js";
|
|
32
|
-
import { createRunnerClient } from "./supabase.js";
|
|
33
|
-
import { runTask } from "./task-runner.js";
|
|
34
|
-
import { runReview, } from "./runtime/reviewer.js";
|
|
35
|
-
import { getQuarantine, setQuarantine, } from "./runtime/quarantine.js";
|
|
36
|
-
import { acquireSingletonLock, SingletonLockHeldError, } from "./runtime/singleton.js";
|
|
37
|
-
import { getClaudeVersion, recordTaskClaudeVersion, } from "./runtime/version-drift.js";
|
|
38
|
-
import { postRunnerStateMessage } from "./messaging.js";
|
|
39
|
-
import { checkVersion, compareSemver } from "./version-check.js";
|
|
40
|
-
import { PROTOCOL_VERSION } from "./types.js";
|
|
41
|
-
import { PACKAGE_VERSION } from "./pkg-version.js";
|
|
42
|
-
import { resolveAuthMode, describeAuthMode, authPrecedenceHint } from "./anthropic-auth.js";
|
|
43
|
-
export class RefreshReuseDetectedError extends Error {
|
|
44
|
-
constructor(message) {
|
|
45
|
-
super(message);
|
|
46
|
-
this.name = "RefreshReuseDetectedError";
|
|
47
|
-
}
|
|
48
|
-
}
|
|
49
|
-
const ONE_HOUR_MS = 60 * 60 * 1000;
|
|
50
|
-
const REFRESH_CHECK_MS = 30 * 60 * 1000;
|
|
51
|
-
// v0.12.0 (T-52-7): idle-TTL check cadence ceiling. The actual cadence is
|
|
52
|
-
// min(IDLE_CHECK_MS, ttl) so a tiny test TTL still gets checked in time.
|
|
53
|
-
const IDLE_CHECK_MS = 30_000;
|
|
54
|
-
const HEARTBEAT_REFRESH_AFTER = 3;
|
|
55
|
-
const HEARTBEAT_EXIT_AFTER = 5;
|
|
56
|
-
const POLL_INTERVAL_MS = 30_000;
|
|
57
|
-
// v0.30-B: widen the startup `pollSince` watermark so a task dispatched in
|
|
58
|
-
// the minutes before `watch` started is still caught by the first poll.
|
|
59
|
-
// The 30s window the previous code used was tighter than the gap between
|
|
60
|
-
// dispatch and runner restart in real operator workflows.
|
|
61
|
-
const STARTUP_POLL_LOOKBACK_MS = 5 * 60 * 1000;
|
|
62
|
-
// v0.62 (T-62-1): cadence of the periodic full claim scan — the realtime-miss
|
|
63
|
-
// safety net. The delta-poll's forward-only watermark can permanently skip a
|
|
64
|
-
// task assigned with an `updated_at` at/behind the watermark whose realtime
|
|
65
|
-
// `task_assigned` push was dropped (stale subscription). This full scan
|
|
66
|
-
// (epoch watermark, every assigned+queued task for this runner) re-claims any
|
|
67
|
-
// such task within one interval, so an alive-but-wedged runner self-recovers
|
|
68
|
-
// instead of needing a manual restart. Knob: `claim_scan_interval_ms`.
|
|
69
|
-
const DEFAULT_CLAIM_SCAN_INTERVAL_MS = 3 * 60 * 1000;
|
|
70
|
-
// v0.73 (T-73-2, G-M): claim-loop watchdog. A task can sit assigned+queued for
|
|
71
|
-
// this runner (acc.tasks.runner_id = me, status='queued') while the runner
|
|
72
|
-
// heartbeats fine but never claims it — the channel looks healthy (so T-68-1's
|
|
73
|
-
// auto-recovery never fires) but the claim path is wedged. CLAIM_WATCHDOG_STALE_MS
|
|
74
|
-
// is how long a task may stay assigned-but-unclaimed (with a live heartbeat)
|
|
75
|
-
// before the watchdog treats the claim path as wedged and forces the same
|
|
76
|
-
// recovery a terminal channel status does. Default 90s — comfortably under the
|
|
77
|
-
// server's 300s rescue grace, so we self-heal before the rescue churns rather
|
|
78
|
-
// than waiting for a manual `pkill acc-runner watch`. CLAIM_WATCHDOG_CHECK_MS is
|
|
79
|
-
// how often the watchdog re-evaluates; well under the threshold so detection +
|
|
80
|
-
// recovery land inside the grace window. Knobs: tests override both for speed.
|
|
81
|
-
const DEFAULT_CLAIM_WATCHDOG_STALE_MS = 90_000;
|
|
82
|
-
const DEFAULT_CLAIM_WATCHDOG_CHECK_MS = 30_000;
|
|
83
|
-
// v0.56 (T-56-1): capacity-aware pause. When claude reports a session/usage
|
|
84
|
-
// limit (api_error_status 429) the runner pauses claiming tasks AND reviews
|
|
85
|
-
// until the stated reset time. When no reset time is parseable, fall back to
|
|
86
|
-
// this fixed backoff. Resume fires at the target + a small random jitter so a
|
|
87
|
-
// fleet that all hit the limit on the same broadcast doesn't thunder back in
|
|
88
|
-
// lockstep the instant the window reopens.
|
|
89
|
-
const DEFAULT_CAPACITY_BACKOFF_MS = 30 * 60 * 1000;
|
|
90
|
-
const DEFAULT_RESUME_JITTER_MS = 15_000;
|
|
91
|
-
/**
|
|
92
|
-
* v0.6.0: exponential backoff between heartbeat attempts when the
|
|
93
|
-
* previous one failed. The dogfood loop fired a heartbeat every 4s
|
|
94
|
-
* even when the network was wedged, producing a wall of "fetch failed"
|
|
95
|
-
* stderr noise and reaching the EXIT_AFTER cap in 20s. Spreading
|
|
96
|
-
* attempts out gives flaky links time to recover before we tear down
|
|
97
|
-
* the watch process.
|
|
98
|
-
*/
|
|
99
|
-
const HEARTBEAT_BACKOFF_MS = [4_000, 8_000, 16_000, 32_000, 60_000];
|
|
100
|
-
/**
|
|
101
|
-
* v0.6.0: classify the heartbeat RPC error so we (a) emit a useful
|
|
102
|
-
* stderr line for the operator and (b) only attempt a JWT refresh
|
|
103
|
-
* when the failure shape actually implies an auth problem. Refreshing
|
|
104
|
-
* on a transient network blip just burns the refresh-token rotation
|
|
105
|
-
* chain and surfaces a confusing "refresh failed" line on top of the
|
|
106
|
-
* real underlying network error.
|
|
107
|
-
*/
|
|
108
|
-
export function classifyHeartbeatError(err) {
|
|
109
|
-
if (!err)
|
|
110
|
-
return "unknown";
|
|
111
|
-
const msg = (err.message ?? "").toLowerCase();
|
|
112
|
-
const code = (err.code ?? "").toLowerCase();
|
|
113
|
-
if (code === "pgrst301" ||
|
|
114
|
-
code === "401" ||
|
|
115
|
-
code === "403" ||
|
|
116
|
-
msg.includes("jwt") ||
|
|
117
|
-
msg.includes("invalid token") ||
|
|
118
|
-
msg.includes("token has expired") ||
|
|
119
|
-
msg.includes("unauthorized") ||
|
|
120
|
-
msg.includes("forbidden")) {
|
|
121
|
-
return "auth";
|
|
122
|
-
}
|
|
123
|
-
if (msg.includes("enotfound") || msg.includes("getaddrinfo") || msg.includes("eai_again")) {
|
|
124
|
-
return "dns";
|
|
125
|
-
}
|
|
126
|
-
if (msg.includes("econnreset") || msg.includes("connection reset")) {
|
|
127
|
-
return "reset";
|
|
128
|
-
}
|
|
129
|
-
if (msg.includes("certificate") ||
|
|
130
|
-
msg.includes("self-signed") ||
|
|
131
|
-
msg.includes("self signed") ||
|
|
132
|
-
msg.includes("tls") ||
|
|
133
|
-
msg.includes("ssl")) {
|
|
134
|
-
return "tls";
|
|
135
|
-
}
|
|
136
|
-
if (msg.includes("etimedout") || msg.includes("timeout")) {
|
|
137
|
-
return "timeout";
|
|
138
|
-
}
|
|
139
|
-
if (msg.includes("fetch failed") ||
|
|
140
|
-
msg.includes("econnrefused") ||
|
|
141
|
-
msg.includes("network") ||
|
|
142
|
-
msg.includes("socket hang up")) {
|
|
143
|
-
return "network";
|
|
144
|
-
}
|
|
145
|
-
return "unknown";
|
|
146
|
-
}
|
|
147
|
-
/** Pick the next inter-heartbeat delay from the backoff ladder. Caller
|
|
148
|
-
* passes the number of consecutive failures (1 → first retry). The
|
|
149
|
-
* return is at least baseMs so a tighter operator-configured cadence
|
|
150
|
-
* is never slowed by the backoff scheme. */
|
|
151
|
-
export function nextHeartbeatDelayMs(consecutiveFailures, baseMs) {
|
|
152
|
-
if (consecutiveFailures <= 0)
|
|
153
|
-
return baseMs;
|
|
154
|
-
const idx = Math.min(consecutiveFailures - 1, HEARTBEAT_BACKOFF_MS.length - 1);
|
|
155
|
-
return Math.max(baseMs, HEARTBEAT_BACKOFF_MS[idx]);
|
|
156
|
-
}
|
|
157
|
-
/**
|
|
158
|
-
* v0.12.0 (T-52-7): resolve ACC_RUNNER_IDLE_TTL_MIN into ms. Returns null
|
|
159
|
-
* (TTL disabled) when unset; warns and returns null on junk so a typo'd
|
|
160
|
-
* env never strands a container in "never exits" mode silently.
|
|
161
|
-
*/
|
|
162
|
-
export function resolveIdleTtlMs(env = process.env) {
|
|
163
|
-
const raw = env.ACC_RUNNER_IDLE_TTL_MIN?.trim();
|
|
164
|
-
if (!raw)
|
|
165
|
-
return null;
|
|
166
|
-
const minutes = Number(raw);
|
|
167
|
-
if (!Number.isFinite(minutes) || minutes <= 0) {
|
|
168
|
-
process.stderr.write(`[acc-runner] ACC_RUNNER_IDLE_TTL_MIN=${JSON.stringify(raw)} is not a positive number; idle TTL disabled\n`);
|
|
169
|
-
return null;
|
|
170
|
-
}
|
|
171
|
-
return minutes * 60_000;
|
|
172
|
-
}
|
|
173
|
-
/** v0.12.0 (T-52-7): record work-related activity for the idle TTL clock. */
|
|
174
|
-
function touchActivity(state) {
|
|
175
|
-
state.lastActivityMs = Date.now();
|
|
176
|
-
}
|
|
177
|
-
function expiresSoon(session, leadMs = ONE_HOUR_MS) {
|
|
178
|
-
const expires = Date.parse(session.access_expires_at);
|
|
179
|
-
if (!Number.isFinite(expires))
|
|
180
|
-
return true;
|
|
181
|
-
return expires < Date.now() + leadMs;
|
|
182
|
-
}
|
|
183
|
-
async function refreshAccessToken(cfg, session) {
|
|
184
|
-
const res = await fetch(`${cfg.publicUrl.replace(/\/+$/, "")}/api/runner/refresh`, {
|
|
185
|
-
method: "POST",
|
|
186
|
-
headers: {
|
|
187
|
-
"Content-Type": "application/json",
|
|
188
|
-
"User-Agent": `acc-runner/${PROTOCOL_VERSION}`,
|
|
189
|
-
},
|
|
190
|
-
body: JSON.stringify({ refresh_token: session.refresh_token }),
|
|
191
|
-
});
|
|
192
|
-
if (res.status === 401) {
|
|
193
|
-
const body = (await res.json().catch(() => ({})));
|
|
194
|
-
if (body.error === "refresh_reuse_detected") {
|
|
195
|
-
throw new RefreshReuseDetectedError(body.hint ??
|
|
196
|
-
"refresh token reuse detected; keychain may be compromised. Run `acc-runner login` from a trusted machine.");
|
|
197
|
-
}
|
|
198
|
-
throw new Error(body.hint ?? "refresh token expired; run `acc-runner login`");
|
|
199
|
-
}
|
|
200
|
-
if (!res.ok) {
|
|
201
|
-
throw new Error(`refresh failed: ${res.status} ${await res.text()}`);
|
|
202
|
-
}
|
|
203
|
-
const body = (await res.json());
|
|
204
|
-
return {
|
|
205
|
-
...session,
|
|
206
|
-
access_token: body.access_token,
|
|
207
|
-
access_expires_at: new Date(body.expires_at * 1000).toISOString(),
|
|
208
|
-
// v0.4-E: server rotates on every refresh. Persist the new refresh token
|
|
209
|
-
// so the next call doesn't replay the old one (which would now look like
|
|
210
|
-
// reuse and trip the chain-revocation path).
|
|
211
|
-
refresh_token: body.refresh_token ?? session.refresh_token,
|
|
212
|
-
};
|
|
213
|
-
}
|
|
214
|
-
// Centralised handler for the reuse-detected case. Loud red banner so the
|
|
215
|
-
// operator notices immediately, then exit(1) — there's no safe recovery
|
|
216
|
-
// from the watch loop, the keychain has to be reset by `acc-runner login`.
|
|
217
|
-
function failOnReuse(err) {
|
|
218
|
-
process.stderr.write(chalk.red.bold("\n[acc-runner] CRITICAL: refresh-token reuse detected.\n"));
|
|
219
|
-
process.stderr.write(chalk.red(`${err.message}\n`));
|
|
220
|
-
process.stderr.write(chalk.red("If this machine is the legitimate owner, an attacker may have rotated your tokens.\n" +
|
|
221
|
-
"Rotate any other credentials that were on this box and run `acc-runner login` from a trusted machine.\n"));
|
|
222
|
-
process.exit(1);
|
|
223
|
-
}
|
|
224
|
-
// Single source of truth for "this task is new — push it into the runner
|
|
225
|
-
// queue". Both the Realtime listener and the polling loop go through here.
|
|
226
|
-
function enqueue(state, taskId, factory) {
|
|
227
|
-
if (state.seen.has(taskId))
|
|
228
|
-
return false;
|
|
229
|
-
// v0.10 T-49-2: check the running Map (replaces v0.9 state.current check).
|
|
230
|
-
if (state.running.has(taskId)) {
|
|
231
|
-
state.seen.add(taskId);
|
|
232
|
-
return false;
|
|
233
|
-
}
|
|
234
|
-
state.seen.add(taskId);
|
|
235
|
-
state.queue.push(taskId);
|
|
236
|
-
touchActivity(state);
|
|
237
|
-
void pump(state, factory);
|
|
238
|
-
return true;
|
|
239
|
-
}
|
|
240
|
-
// Single source of truth for "this review is new — push it into the review
|
|
241
|
-
// queue". The Realtime listener AND the periodic review scan (v0.65 T-65-2)
|
|
242
|
-
// both go through here; the shared `seenReviews` guard dedupes a review that
|
|
243
|
-
// arrives on both paths. Returns true only when the review was actually
|
|
244
|
-
// enqueued (new + well-formed), so the scan can count what it recovered.
|
|
245
|
-
function enqueueReview(state, payload) {
|
|
246
|
-
const p = payload;
|
|
247
|
-
if (!p?.review_id || !p?.task_id || !p?.pr_number)
|
|
248
|
-
return false;
|
|
249
|
-
if (state.seenReviews.has(p.review_id))
|
|
250
|
-
return false;
|
|
251
|
-
state.seenReviews.add(p.review_id);
|
|
252
|
-
state.reviewQueue.push({
|
|
253
|
-
review_id: p.review_id,
|
|
254
|
-
task_id: p.task_id,
|
|
255
|
-
pr_number: p.pr_number,
|
|
256
|
-
});
|
|
257
|
-
touchActivity(state);
|
|
258
|
-
void pumpReviews(state);
|
|
259
|
-
return true;
|
|
260
|
-
}
|
|
261
|
-
function subscribeChannel(state, taskRunnerFactory) {
|
|
262
|
-
state.channel = state.supabase
|
|
263
|
-
.channel(`runner:${state.session.runner_id}`, {
|
|
264
|
-
config: { broadcast: { ack: false, self: false } },
|
|
265
|
-
})
|
|
266
|
-
.on("broadcast", { event: "task_assigned" }, ({ payload }) => {
|
|
267
|
-
const taskId = payload?.task_id;
|
|
268
|
-
if (taskId)
|
|
269
|
-
enqueue(state, taskId, taskRunnerFactory);
|
|
270
|
-
})
|
|
271
|
-
.on("broadcast", { event: "task_cancelled" }, ({ payload }) => {
|
|
272
|
-
const taskId = payload?.task_id;
|
|
273
|
-
if (!taskId)
|
|
274
|
-
return;
|
|
275
|
-
// v0.10 T-49-2: cancel from the running Map; fall back to queue removal.
|
|
276
|
-
const ctrl = state.running.get(taskId);
|
|
277
|
-
if (ctrl) {
|
|
278
|
-
ctrl.cancel();
|
|
279
|
-
}
|
|
280
|
-
else {
|
|
281
|
-
state.queue = state.queue.filter((t) => t !== taskId);
|
|
282
|
-
}
|
|
283
|
-
// Drop from seen so a re-queue (cancel → re-start) is honored on
|
|
284
|
-
// the next broadcast or poll.
|
|
285
|
-
state.seen.delete(taskId);
|
|
286
|
-
})
|
|
287
|
-
.on("broadcast", { event: "review_assigned" }, ({ payload }) => {
|
|
288
|
-
enqueueReview(state, payload);
|
|
289
|
-
})
|
|
290
|
-
.subscribe((status) => {
|
|
291
|
-
if (status === "SUBSCRIBED") {
|
|
292
|
-
// v0.68 (T-68-1): healthy subscribe — reset the resubscribe backoff.
|
|
293
|
-
state.resubscribeAttempts = 0;
|
|
294
|
-
console.log(chalk.green(`✓ Listening on runner:${state.session.runner_id}`));
|
|
295
|
-
}
|
|
296
|
-
else if (status === "CHANNEL_ERROR" ||
|
|
297
|
-
status === "TIMED_OUT" ||
|
|
298
|
-
status === "CLOSED") {
|
|
299
|
-
// v0.68 (T-68-1): the realtime channel dropped. Previously unhandled —
|
|
300
|
-
// which left the runner silently deaf to task_assigned / review_assigned
|
|
301
|
-
// broadcasts until token expiry or a manual restart (the stale-
|
|
302
|
-
// subscription root cause behind the repeated "task stuck in queued →
|
|
303
|
-
// swept to blocked" incidents). Recover it.
|
|
304
|
-
if (!state.stopped)
|
|
305
|
-
scheduleResubscribe(state, taskRunnerFactory, status);
|
|
306
|
-
}
|
|
307
|
-
});
|
|
308
|
-
}
|
|
309
|
-
// v0.68 (T-68-1): single-flight, backed-off recovery of a dropped realtime
|
|
310
|
-
// channel. Like doRefresh's remove+resubscribe, but WITHOUT a token rotation
|
|
311
|
-
// (the token is still valid; only the channel died) and driven by channel
|
|
312
|
-
// status rather than token expiry. Backoff 1s→30s, reset to 0 on a healthy
|
|
313
|
-
// SUBSCRIBED. After a fresh subscribe we run the poll + claim/review scans so
|
|
314
|
-
// anything the dead channel missed is recovered at once rather than waiting
|
|
315
|
-
// for the periodic 3-min scans.
|
|
316
|
-
function scheduleResubscribe(state, factory, status) {
|
|
317
|
-
if (state.stopped || state.resubscribing || state.resubscribeTimer)
|
|
318
|
-
return;
|
|
319
|
-
const delay = Math.min(30_000, 1_000 * 2 ** Math.min(state.resubscribeAttempts, 5));
|
|
320
|
-
process.stderr.write(`[acc-runner] realtime channel ${status}; resubscribing in ${delay}ms ` +
|
|
321
|
-
`(attempt ${state.resubscribeAttempts + 1})\n`);
|
|
322
|
-
state.resubscribeTimer = setTimeout(() => {
|
|
323
|
-
state.resubscribeTimer = null;
|
|
324
|
-
void resubscribeChannel(state, factory);
|
|
325
|
-
}, delay);
|
|
326
|
-
if (state.resubscribeTimer.unref)
|
|
327
|
-
state.resubscribeTimer.unref();
|
|
328
|
-
}
|
|
329
|
-
export async function resubscribeChannel(state, factory) {
|
|
330
|
-
if (state.stopped || state.resubscribing)
|
|
331
|
-
return;
|
|
332
|
-
state.resubscribing = true;
|
|
333
|
-
state.resubscribeAttempts += 1;
|
|
334
|
-
try {
|
|
335
|
-
try {
|
|
336
|
-
await state.supabase.removeChannel(state.channel);
|
|
337
|
-
}
|
|
338
|
-
catch {
|
|
339
|
-
/* channel may already be torn down */
|
|
340
|
-
}
|
|
341
|
-
subscribeChannel(state, factory);
|
|
342
|
-
// Recover anything the dead channel missed during the gap. Best-effort;
|
|
343
|
-
// the `seen` / `seenReviews` guards dedupe against a late re-delivery.
|
|
344
|
-
void pollOnce(state, factory);
|
|
345
|
-
void claimScan(state, factory, "periodic");
|
|
346
|
-
void reviewScan(state, "periodic");
|
|
347
|
-
}
|
|
348
|
-
finally {
|
|
349
|
-
state.resubscribing = false;
|
|
350
|
-
}
|
|
351
|
-
}
|
|
352
|
-
// Single-flight poll. The watermark advances only after a clean response so
|
|
353
|
-
// a transient error doesn't permanently skip rows.
|
|
354
|
-
async function pollOnce(state, factory) {
|
|
355
|
-
if (state.stopped || state.polling)
|
|
356
|
-
return;
|
|
357
|
-
state.polling = true;
|
|
358
|
-
try {
|
|
359
|
-
const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
|
|
360
|
-
p_runner_id: state.session.runner_id,
|
|
361
|
-
p_since: state.pollSince,
|
|
362
|
-
});
|
|
363
|
-
if (error) {
|
|
364
|
-
process.stderr.write(`[acc-runner] poll failed: ${error.message}\n`);
|
|
365
|
-
return;
|
|
366
|
-
}
|
|
367
|
-
const rows = (data ?? []);
|
|
368
|
-
let maxSeen = state.pollSince;
|
|
369
|
-
for (const row of rows) {
|
|
370
|
-
if (!row?.id)
|
|
371
|
-
continue;
|
|
372
|
-
// v0.62 (T-62-1): advance the forward-only watermark ONLY for rows we
|
|
373
|
-
// actually claimed this pass (enqueue returned true). Previously the
|
|
374
|
-
// watermark advanced to max(updated_at) over EVERY returned row,
|
|
375
|
-
// including no-ops (a task already seen/in-flight via realtime or a
|
|
376
|
-
// prior poll). A high-updated_at no-op could then drag the watermark
|
|
377
|
-
// past a still-queued task assigned with an older updated_at, so the
|
|
378
|
-
// next delta-poll (`> watermark`) silently dropped it — only a restart
|
|
379
|
-
// (one-shot startup scan) recovered it. Advancing solely on real claims
|
|
380
|
-
// keeps the watermark a conservative "newest task I claimed by polling"
|
|
381
|
-
// mark and stops the delta-poll losing still-queued work. The periodic
|
|
382
|
-
// full claim scan (claimScan) is the belt to this braces.
|
|
383
|
-
const claimed = enqueue(state, row.id, factory);
|
|
384
|
-
if (claimed && row.updated_at && row.updated_at > maxSeen) {
|
|
385
|
-
maxSeen = row.updated_at;
|
|
386
|
-
}
|
|
387
|
-
}
|
|
388
|
-
state.pollSince = maxSeen;
|
|
389
|
-
}
|
|
390
|
-
finally {
|
|
391
|
-
state.polling = false;
|
|
392
|
-
}
|
|
393
|
-
}
|
|
394
|
-
/**
|
|
395
|
-
* v0.62 (T-62-1): full claim scan — lists EVERY assigned+queued task for this
|
|
396
|
-
* runner with an epoch watermark (`1970-…Z`) and enqueues any the in-process
|
|
397
|
-
* `seen` guard hasn't already accounted for. Used two ways:
|
|
398
|
-
*
|
|
399
|
-
* - once at startup (replaces the v0.34-B inline block), and
|
|
400
|
-
* - on a periodic timer (`claim_scan_interval_ms`, default 3 min) as the
|
|
401
|
-
* realtime-miss safety net: a runner that dropped a `task_assigned`
|
|
402
|
-
* broadcast self-recovers within one interval instead of sitting next to
|
|
403
|
-
* a free runner until a manual restart.
|
|
404
|
-
*
|
|
405
|
-
* Deliberately does NOT touch `state.pollSince` — the delta-poll keeps its own
|
|
406
|
-
* forward-only watermark. Idempotent: `enqueue`'s shared `seen`/`running`
|
|
407
|
-
* guards mean a task already taken by realtime or the delta-poll is a no-op
|
|
408
|
-
* here, so realtime + delta-poll + scan all seeing the same task still yields
|
|
409
|
-
* exactly one run. Single-flight via `claimScanning` so a slow RPC can't let
|
|
410
|
-
* two scans overlap.
|
|
411
|
-
*/
|
|
412
|
-
async function claimScan(state, factory, context) {
|
|
413
|
-
if (state.stopped || state.claimScanning)
|
|
414
|
-
return;
|
|
415
|
-
state.claimScanning = true;
|
|
416
|
-
try {
|
|
417
|
-
const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
|
|
418
|
-
p_runner_id: state.session.runner_id,
|
|
419
|
-
p_since: "1970-01-01T00:00:00.000Z",
|
|
420
|
-
});
|
|
421
|
-
if (error) {
|
|
422
|
-
process.stderr.write(`[acc-runner] ${context} claim scan failed: ${error.message}\n`);
|
|
423
|
-
return;
|
|
424
|
-
}
|
|
425
|
-
const rows = (data ?? []);
|
|
426
|
-
let enqueued = 0;
|
|
427
|
-
for (const row of rows) {
|
|
428
|
-
if (!row?.id)
|
|
429
|
-
continue;
|
|
430
|
-
if (enqueue(state, row.id, factory))
|
|
431
|
-
enqueued += 1;
|
|
432
|
-
}
|
|
433
|
-
if (context === "startup") {
|
|
434
|
-
console.log(`[acc-runner] startup claim scan: ${rows.length} task(s) found`);
|
|
435
|
-
}
|
|
436
|
-
else if (enqueued > 0) {
|
|
437
|
-
// Only chirp when the safety net actually caught something — a periodic
|
|
438
|
-
// scan that finds nothing is the steady state and must stay quiet.
|
|
439
|
-
console.log(`[acc-runner] periodic claim scan recovered ${enqueued} realtime-missed task(s)`);
|
|
440
|
-
}
|
|
441
|
-
}
|
|
442
|
-
catch (err) {
|
|
443
|
-
process.stderr.write(`[acc-runner] ${context} claim scan failed: ${err.message}\n`);
|
|
444
|
-
}
|
|
445
|
-
finally {
|
|
446
|
-
state.claimScanning = false;
|
|
447
|
-
}
|
|
448
|
-
}
|
|
449
|
-
/**
|
|
450
|
-
* v0.73 (T-73-2, G-M): claim-loop watchdog — the self-heal for an
|
|
451
|
-
* "assigned-but-unclaimed while heartbeating" strand.
|
|
452
|
-
*
|
|
453
|
-
* SYMPTOM (prod 2026-06-10): a runner heartbeated fine (status online, fresh
|
|
454
|
-
* heartbeat, idle) but never claimed a task the dispatcher had ASSIGNED to it
|
|
455
|
-
* (acc.tasks.runner_id = me, status='queued'). The server rescue sweep
|
|
456
|
-
* re-dispatched every ~6 min but the runner never claimed; only a manual
|
|
457
|
-
* `pkill acc-runner watch && acc-runner watch` cleared it. T-68-1's channel
|
|
458
|
-
* auto-recovery did NOT catch it — the channel looked healthy but the claim
|
|
459
|
-
* path was wedged.
|
|
460
|
-
*
|
|
461
|
-
* Mechanism: each tick lists the runner's own assigned+queued (= unclaimed)
|
|
462
|
-
* tasks via the same `list_assigned_queued_tasks` RPC the claim scan uses, and
|
|
463
|
-
* tracks per task id the first instant it was OBSERVED as
|
|
464
|
-
* assigned-to-me-but-not-claimed. A task that stays unclaimed past
|
|
465
|
-
* `claimWatchdogStaleMs` (default 90s — comfortably under the server's 300s
|
|
466
|
-
* rescue grace) WHILE heartbeats are succeeding means the claim path is wedged:
|
|
467
|
-
* trigger the SAME single-flight recovery a terminal channel status does
|
|
468
|
-
* (resubscribeChannel: removeChannel + subscribeChannel + pollOnce + claimScan
|
|
469
|
-
* + reviewScan), logged as `claim_watchdog_resubscribe`.
|
|
470
|
-
*
|
|
471
|
-
* Reset semantics: a task's observed-stale timer is dropped the moment it
|
|
472
|
-
* leaves the assigned+queued set (claimed → status moved off 'queued', or
|
|
473
|
-
* unassigned → runner_id cleared). A heartbeat failure clears ALL timers — a
|
|
474
|
-
* network/auth outage is a different failure with its own handling, and we must
|
|
475
|
-
* not carry a stale age across it into a false trigger when heartbeats recover.
|
|
476
|
-
*
|
|
477
|
-
* Single-flight: respects the existing resubscribe guard/backoff
|
|
478
|
-
* (`resubscribing` / `resubscribeTimer`) so concurrent staleness can't stack
|
|
479
|
-
* resubscribes, and resets the stale timers on trigger so a recovery that
|
|
480
|
-
* didn't help waits another full threshold before re-firing. Bounded and
|
|
481
|
-
* side-effect-safe: never throws out of the loop.
|
|
482
|
-
*/
|
|
483
|
-
async function claimWatchdogScan(state, factory) {
|
|
484
|
-
if (state.stopped || state.claimWatchdogChecking)
|
|
485
|
-
return;
|
|
486
|
-
state.claimWatchdogChecking = true;
|
|
487
|
-
try {
|
|
488
|
-
// Only treat "unclaimed" as wedged while heartbeats are healthy. A
|
|
489
|
-
// heartbeat failure means the network/auth path is down — the claim RPC
|
|
490
|
-
// would fail too, and the fix there is the heartbeat refresh/exit path, not
|
|
491
|
-
// a resubscribe. Clear the observed-stale timers so an age never spans an
|
|
492
|
-
// outage and falsely trips the watchdog the instant heartbeats recover.
|
|
493
|
-
if (state.heartbeatFailures > 0) {
|
|
494
|
-
state.claimWatchdogSince.clear();
|
|
495
|
-
return;
|
|
496
|
-
}
|
|
497
|
-
const { data, error } = await state.supabase.rpc("list_assigned_queued_tasks", {
|
|
498
|
-
p_runner_id: state.session.runner_id,
|
|
499
|
-
p_since: "1970-01-01T00:00:00.000Z",
|
|
500
|
-
});
|
|
501
|
-
if (error) {
|
|
502
|
-
process.stderr.write(`[acc-runner] claim watchdog scan failed: ${error.message}\n`);
|
|
503
|
-
return;
|
|
504
|
-
}
|
|
505
|
-
const rows = (data ?? []);
|
|
506
|
-
const now = Date.now();
|
|
507
|
-
const currentIds = new Set();
|
|
508
|
-
for (const row of rows) {
|
|
509
|
-
if (row?.id)
|
|
510
|
-
currentIds.add(row.id);
|
|
511
|
-
}
|
|
512
|
-
// Reset: drop timers for tasks no longer assigned+queued (claimed/unassigned).
|
|
513
|
-
for (const id of [...state.claimWatchdogSince.keys()]) {
|
|
514
|
-
if (!currentIds.has(id))
|
|
515
|
-
state.claimWatchdogSince.delete(id);
|
|
516
|
-
}
|
|
517
|
-
// Observe: stamp the first-seen instant for newly assigned-unclaimed tasks;
|
|
518
|
-
// collect any that have now been stale past the threshold.
|
|
519
|
-
const stale = [];
|
|
520
|
-
for (const id of currentIds) {
|
|
521
|
-
const since = state.claimWatchdogSince.get(id);
|
|
522
|
-
if (since === undefined) {
|
|
523
|
-
state.claimWatchdogSince.set(id, now);
|
|
524
|
-
}
|
|
525
|
-
else if (now - since >= state.claimWatchdogStaleMs) {
|
|
526
|
-
stale.push(id);
|
|
527
|
-
}
|
|
528
|
-
}
|
|
529
|
-
if (stale.length === 0)
|
|
530
|
-
return;
|
|
531
|
-
// Single-flight: never stack on a resubscribe already in flight or
|
|
532
|
-
// scheduled (the existing T-68-1 guard/backoff owns the recovery).
|
|
533
|
-
if (state.resubscribing || state.resubscribeTimer)
|
|
534
|
-
return;
|
|
535
|
-
process.stderr.write(`[acc-runner] claim_watchdog_resubscribe: task(s) ${stale.join(", ")} ` +
|
|
536
|
-
`assigned but unclaimed > ${state.claimWatchdogStaleMs}ms with a live ` +
|
|
537
|
-
`heartbeat; claim path looks wedged — forcing resubscribe + claim scan.\n`);
|
|
538
|
-
// Reset the stale timers so a recovery that didn't help waits another full
|
|
539
|
-
// threshold before re-triggering (backoff alongside the resubscribe guard).
|
|
540
|
-
for (const id of stale)
|
|
541
|
-
state.claimWatchdogSince.set(id, now);
|
|
542
|
-
void (async () => {
|
|
543
|
-
try {
|
|
544
|
-
await state.supabase.rpc("log_activity", {
|
|
545
|
-
p_verb: "runner.claim_watchdog_resubscribe",
|
|
546
|
-
p_target_id: state.session.runner_id,
|
|
547
|
-
p_payload: {
|
|
548
|
-
task_ids: stale,
|
|
549
|
-
stale_ms: state.claimWatchdogStaleMs,
|
|
550
|
-
version: PACKAGE_VERSION,
|
|
551
|
-
},
|
|
552
|
-
p_target_type: "runner",
|
|
553
|
-
});
|
|
554
|
-
}
|
|
555
|
-
catch { /* best-effort breadcrumb */ }
|
|
556
|
-
})();
|
|
557
|
-
void resubscribeChannel(state, factory);
|
|
558
|
-
}
|
|
559
|
-
catch (err) {
|
|
560
|
-
process.stderr.write(`[acc-runner] claim watchdog scan failed: ${err.message}\n`);
|
|
561
|
-
}
|
|
562
|
-
finally {
|
|
563
|
-
state.claimWatchdogChecking = false;
|
|
564
|
-
}
|
|
565
|
-
}
|
|
566
|
-
/**
|
|
567
|
-
* v0.65 (T-65-2): full review scan — the review twin of claimScan. Lists EVERY
|
|
568
|
-
* pending review assigned to this runner with an epoch watermark (`1970-…Z`)
|
|
569
|
-
* via `list_assigned_pending_reviews` and feeds each row into the existing
|
|
570
|
-
* `enqueueReview`, whose shared `seenReviews` guard dedupes a review the
|
|
571
|
-
* realtime `review_assigned` broadcast already delivered. Used two ways:
|
|
572
|
-
*
|
|
573
|
-
* - once at startup (next to the startup claim scan), and
|
|
574
|
-
* - on a periodic timer (`review_scan_interval_ms`, default 3 min) as the
|
|
575
|
-
* realtime-miss safety net: a runner that dropped a `review_assigned`
|
|
576
|
-
* broadcast recovers the review within one interval instead of letting it
|
|
577
|
-
* strand in acc.review_queue (PR #544's review stranded 16h last cycle).
|
|
578
|
-
*
|
|
579
|
-
* Idempotent: `enqueueReview`'s `seenReviews` guard means a review already
|
|
580
|
-
* taken by realtime is a no-op here, so realtime + scan both seeing the same
|
|
581
|
-
* review still yields exactly one review run. Single-flight via
|
|
582
|
-
* `reviewScanning` so a slow RPC can't let two scans overlap. Skipped while
|
|
583
|
-
* paused for capacity — matching `pumpReviews`, the runner deliberately holds
|
|
584
|
-
* reviews until its session window reopens.
|
|
585
|
-
*/
|
|
586
|
-
async function reviewScan(state, context) {
|
|
587
|
-
if (state.stopped || state.reviewScanning)
|
|
588
|
-
return;
|
|
589
|
-
// v0.56 (T-56-1) parity with pumpReviews: don't claim reviews while paused
|
|
590
|
-
// for capacity. The scheduled resume re-drives the review pump.
|
|
591
|
-
if (state.pausedCapacity)
|
|
592
|
-
return;
|
|
593
|
-
state.reviewScanning = true;
|
|
594
|
-
try {
|
|
595
|
-
const { data, error } = await state.supabase.rpc("list_assigned_pending_reviews", {
|
|
596
|
-
p_runner_id: state.session.runner_id,
|
|
597
|
-
p_since: "1970-01-01T00:00:00.000Z",
|
|
598
|
-
});
|
|
599
|
-
if (error) {
|
|
600
|
-
process.stderr.write(`[acc-runner] ${context} review scan failed: ${error.message}\n`);
|
|
601
|
-
return;
|
|
602
|
-
}
|
|
603
|
-
const rows = (data ?? []);
|
|
604
|
-
let enqueued = 0;
|
|
605
|
-
for (const row of rows) {
|
|
606
|
-
if (enqueueReview(state, row))
|
|
607
|
-
enqueued += 1;
|
|
608
|
-
}
|
|
609
|
-
if (context === "startup") {
|
|
610
|
-
console.log(`[acc-runner] startup review scan: ${rows.length} pending review(s) found`);
|
|
611
|
-
}
|
|
612
|
-
else if (enqueued > 0) {
|
|
613
|
-
// Only chirp when the safety net actually caught something — a periodic
|
|
614
|
-
// scan that finds nothing is the steady state and must stay quiet.
|
|
615
|
-
console.log(`[acc-runner] periodic review scan recovered ${enqueued} realtime-missed review(s)`);
|
|
616
|
-
}
|
|
617
|
-
}
|
|
618
|
-
catch (err) {
|
|
619
|
-
process.stderr.write(`[acc-runner] ${context} review scan failed: ${err.message}\n`);
|
|
620
|
-
}
|
|
621
|
-
finally {
|
|
622
|
-
state.reviewScanning = false;
|
|
623
|
-
}
|
|
624
|
-
}
|
|
625
|
-
// Single-flight refresh. Recreates the Supabase client (so future RPC +
|
|
626
|
-
// channel auth use the new JWT) and resubscribes the broadcast channel.
|
|
627
|
-
async function doRefresh(state, taskRunnerFactory) {
|
|
628
|
-
// v0.12.0 (T-52-7): an env-token runner without ACC_RUNNER_REFRESH_TOKEN
|
|
629
|
-
// cannot rotate. Refusing here (instead of POSTing an empty token) keeps
|
|
630
|
-
// the failure mode a clear stderr line rather than a server-side 400.
|
|
631
|
-
if (!state.tokenProvider.canRefresh()) {
|
|
632
|
-
process.stderr.write("[acc-runner] token refresh unavailable (no refresh token in env); runner will exit when the access token expires\n");
|
|
633
|
-
return false;
|
|
634
|
-
}
|
|
635
|
-
if (state.refreshing)
|
|
636
|
-
return state.refreshing;
|
|
637
|
-
state.refreshing = (async () => {
|
|
638
|
-
try {
|
|
639
|
-
const updated = await refreshAccessToken(state.cfg, state.session);
|
|
640
|
-
state.session = updated;
|
|
641
|
-
await state.tokenProvider.save(updated);
|
|
642
|
-
try {
|
|
643
|
-
await state.supabase.removeChannel(state.channel);
|
|
644
|
-
}
|
|
645
|
-
catch {
|
|
646
|
-
/* channel may already be torn down */
|
|
647
|
-
}
|
|
648
|
-
state.supabase = createRunnerClient(state.cfg, state.session.access_token);
|
|
649
|
-
subscribeChannel(state, taskRunnerFactory);
|
|
650
|
-
return true;
|
|
651
|
-
}
|
|
652
|
-
catch (err) {
|
|
653
|
-
if (err instanceof RefreshReuseDetectedError)
|
|
654
|
-
failOnReuse(err);
|
|
655
|
-
process.stderr.write(`[acc-runner] refresh failed: ${err.message}\n`);
|
|
656
|
-
return false;
|
|
657
|
-
}
|
|
658
|
-
finally {
|
|
659
|
-
state.refreshing = null;
|
|
660
|
-
}
|
|
661
|
-
})();
|
|
662
|
-
return state.refreshing;
|
|
663
|
-
}
|
|
664
|
-
export async function watchCommand(options = {}) {
|
|
665
|
-
const cfg = loadConfig();
|
|
666
|
-
const exitFn = options.exit ?? ((code) => process.exit(code));
|
|
667
|
-
// v0.14.0 (T-54-2, GA-3): single-instance guard. Acquired BEFORE the
|
|
668
|
-
// session load + realtime connection so a duplicate exits immediately,
|
|
669
|
-
// before it can race the holder on worktree locks or git refs. A second
|
|
670
|
-
// runner for the same repo path is a hard error; a stale lock from a dead
|
|
671
|
-
// pid self-reclaims inside acquireSingletonLock.
|
|
672
|
-
const enforceSingleton = options.enforceSingleton ?? !options.taskRunnerFactory;
|
|
673
|
-
let singletonLock = null;
|
|
674
|
-
if (enforceSingleton) {
|
|
675
|
-
try {
|
|
676
|
-
singletonLock = await acquireSingletonLock(cfg.repoPath);
|
|
677
|
-
}
|
|
678
|
-
catch (err) {
|
|
679
|
-
if (err instanceof SingletonLockHeldError) {
|
|
680
|
-
process.stderr.write(chalk.red.bold("\n[acc-runner] ANOTHER RUNNER IS ALREADY ACTIVE.\n"));
|
|
681
|
-
process.stderr.write(chalk.red(`Repo path ${err.holder.repoPath} is served by pid ${err.holder.pid} ` +
|
|
682
|
-
`(acc-runner v${err.holder.version}, since ${err.holder.acquiredAt}).\n` +
|
|
683
|
-
"Refusing to start a second runner for the same repo — concurrent " +
|
|
684
|
-
"runners corrupt worktrees and race git refs.\n"));
|
|
685
|
-
process.stderr.write(chalk.gray(`To stop the other runner: kill ${err.holder.pid}\n` +
|
|
686
|
-
`If it is already dead, remove ${err.lockFile} (or just retry — ` +
|
|
687
|
-
"stale locks self-reclaim).\n"));
|
|
688
|
-
exitFn(1);
|
|
689
|
-
// exitFn is process.exit in production (never returns); in tests it
|
|
690
|
-
// is a stub, so re-throw to halt watchCommand cleanly.
|
|
691
|
-
throw err;
|
|
692
|
-
}
|
|
693
|
-
throw err;
|
|
694
|
-
}
|
|
695
|
-
}
|
|
696
|
-
const tokenProvider = getTokenProvider();
|
|
697
|
-
let session = await tokenProvider.load();
|
|
698
|
-
if (!session) {
|
|
699
|
-
process.stderr.write(tokenProvider.mode === "env"
|
|
700
|
-
? "No session in env. Set ACC_RUNNER_ACCESS_TOKEN (token mode: env).\n"
|
|
701
|
-
: "Not logged in. Run `acc-runner login`.\n");
|
|
702
|
-
process.exit(1);
|
|
703
|
-
}
|
|
704
|
-
const v = await checkVersion(cfg.publicUrl);
|
|
705
|
-
if (!v.ok && v.reason === "outdated") {
|
|
706
|
-
throw new Error(`acc-runner is below the server's minimum version (${v.serverMin}). Run \`pnpm add -g @tokenfactory/acc-runner@latest\`.`);
|
|
707
|
-
}
|
|
708
|
-
// v0.31-B: warn when running behind current_version (above min but stale).
|
|
709
|
-
// Launchd daemon often runs an old binary silently; this surfaces it at
|
|
710
|
-
// startup before the first heartbeat so the operator can update promptly.
|
|
711
|
-
if (v.ok && v.serverCurrent && compareSemver(PACKAGE_VERSION, v.serverCurrent) < 0) {
|
|
712
|
-
process.stderr.write(`[acc-runner] WARNING: running v${PACKAGE_VERSION}, server current is v${v.serverCurrent}. ` +
|
|
713
|
-
`Run \`npm install -g @tokenfactory/acc-runner@latest\` then reload launchd.\n`);
|
|
714
|
-
}
|
|
715
|
-
// v0.74-B: surface how spawned claude sessions will authenticate. api-key
|
|
716
|
-
// (ANTHROPIC_API_KEY set) bills against the API account with no per-session
|
|
717
|
-
// cap; interactive-session falls back to the operator's OAuth login, which
|
|
718
|
-
// hits 429 session caps during long autonomous runs.
|
|
719
|
-
const authMode = resolveAuthMode();
|
|
720
|
-
console.log(chalk.gray(`Anthropic auth: ${describeAuthMode(authMode)}`));
|
|
721
|
-
// v0.76-A (GA-11): surface the API-key-vs-claude.ai-login precedence hint so
|
|
722
|
-
// a 429-on-a-key-configured-runner is diagnosable from the banner.
|
|
723
|
-
const authHint = authPrecedenceHint();
|
|
724
|
-
if (authHint)
|
|
725
|
-
console.log(chalk.gray(`Anthropic auth: ${authHint}`));
|
|
726
|
-
// Startup refresh: if the access token expires in less than an hour,
|
|
727
|
-
// grab a fresh one before opening the realtime connection.
|
|
728
|
-
// v0.12.0 (T-52-7): skipped when the provider has no refresh token —
|
|
729
|
-
// an env-token runner just runs out its 8h access token (the idle TTL
|
|
730
|
-
// normally exits it long before then).
|
|
731
|
-
if (tokenProvider.canRefresh() && expiresSoon(session)) {
|
|
732
|
-
try {
|
|
733
|
-
session = await refreshAccessToken(cfg, session);
|
|
734
|
-
await tokenProvider.save(session);
|
|
735
|
-
}
|
|
736
|
-
catch (err) {
|
|
737
|
-
if (err instanceof RefreshReuseDetectedError)
|
|
738
|
-
failOnReuse(err);
|
|
739
|
-
process.stderr.write(`Refresh failed; run \`acc-runner login\`. (${err.message})\n`);
|
|
740
|
-
process.exit(1);
|
|
741
|
-
}
|
|
742
|
-
}
|
|
743
|
-
// v0.12.0 (T-52-7): env-token (ephemeral) runners had no `login` flow, so
|
|
744
|
-
// nothing registered an acc.runners row yet. Register on boot under the
|
|
745
|
-
// acc-eph-<random> identity. register_runner may resolve to an existing
|
|
746
|
-
// row (same bound user + machine, e.g. a restarted container with a
|
|
747
|
-
// pinned hostname) — adopt whatever id it returns so the realtime
|
|
748
|
-
// channel, heartbeats, and polling all agree.
|
|
749
|
-
if (tokenProvider.mode === "env") {
|
|
750
|
-
const boot = createRunnerClient(cfg, session.access_token);
|
|
751
|
-
const models = await detectInstalledModels();
|
|
752
|
-
const { data: resolvedId, error: regErr } = await boot.rpc("register_runner", {
|
|
753
|
-
p_id: session.runner_id,
|
|
754
|
-
p_name: session.runner_id,
|
|
755
|
-
p_owner: session.email ?? session.user_id,
|
|
756
|
-
p_machine: machineIdentity(),
|
|
757
|
-
p_models: models,
|
|
758
|
-
p_caps: RUNNER_CAPS,
|
|
759
|
-
p_version: PACKAGE_VERSION,
|
|
760
|
-
});
|
|
761
|
-
if (regErr) {
|
|
762
|
-
process.stderr.write(`[acc-runner] ephemeral register_runner failed: ${regErr.message}\n`);
|
|
763
|
-
process.exit(1);
|
|
764
|
-
}
|
|
765
|
-
if (typeof resolvedId === "string" && resolvedId && resolvedId !== session.runner_id) {
|
|
766
|
-
session = { ...session, runner_id: resolvedId };
|
|
767
|
-
}
|
|
768
|
-
await tokenProvider.save(session);
|
|
769
|
-
console.log(chalk.green(`✓ Registered ephemeral runner: ${session.runner_id}`));
|
|
770
|
-
}
|
|
771
|
-
const taskRunnerFactory = options.taskRunnerFactory ?? runTask;
|
|
772
|
-
const heartbeatMs = options.heartbeatMs ?? 4_000;
|
|
773
|
-
const pollMs = options.pollMs ?? POLL_INTERVAL_MS;
|
|
774
|
-
const reviewerFactory = options.reviewerFactory ?? ((assignment, deps) => runReview(assignment, deps));
|
|
775
|
-
// v0.48: check for an existing quarantine file before opening the
|
|
776
|
-
// realtime connection. If quarantined, emit a loud warning — the
|
|
777
|
-
// runner will not claim tasks until `acc-runner quarantine clear` runs.
|
|
778
|
-
const startupQuarantine = await getQuarantine().catch(() => null);
|
|
779
|
-
if (startupQuarantine) {
|
|
780
|
-
process.stderr.write(chalk.red.bold(`[acc-runner] QUARANTINED (${startupQuarantine.cause}): ${startupQuarantine.detail}\n`));
|
|
781
|
-
process.stderr.write(chalk.red(`[acc-runner] Runner will NOT claim tasks. Resolve the issue then run ` +
|
|
782
|
-
`\`acc-runner quarantine clear\`.\n`));
|
|
783
|
-
}
|
|
784
|
-
// v0.12.0 (T-52-7): idle TTL — option override wins (tests), then env.
|
|
785
|
-
const idleTtlMs = options.idleTtlMs !== undefined ? options.idleTtlMs : resolveIdleTtlMs();
|
|
786
|
-
const state = {
|
|
787
|
-
cfg,
|
|
788
|
-
session,
|
|
789
|
-
tokenProvider,
|
|
790
|
-
supabase: createRunnerClient(cfg, session.access_token),
|
|
791
|
-
channel: undefined,
|
|
792
|
-
heartbeatTimer: undefined,
|
|
793
|
-
refreshTimer: undefined,
|
|
794
|
-
pollTimer: undefined,
|
|
795
|
-
claimScanTimer: undefined, // v0.62 (T-62-1)
|
|
796
|
-
claimScanIntervalMs: options.claimScanIntervalMs ?? DEFAULT_CLAIM_SCAN_INTERVAL_MS,
|
|
797
|
-
claimScanning: false,
|
|
798
|
-
reviewScanTimer: undefined, // v0.65 (T-65-2)
|
|
799
|
-
reviewScanIntervalMs: options.reviewScanIntervalMs ?? DEFAULT_CLAIM_SCAN_INTERVAL_MS,
|
|
800
|
-
reviewScanning: false,
|
|
801
|
-
// v0.73 (T-73-2, G-M): claim-loop watchdog.
|
|
802
|
-
claimWatchdogTimer: undefined,
|
|
803
|
-
claimWatchdogStaleMs: options.claimWatchdogStaleMs ?? DEFAULT_CLAIM_WATCHDOG_STALE_MS,
|
|
804
|
-
claimWatchdogSince: new Map(),
|
|
805
|
-
claimWatchdogChecking: false,
|
|
806
|
-
// v0.68 (T-68-1): realtime channel-health recovery state.
|
|
807
|
-
resubscribing: false,
|
|
808
|
-
resubscribeAttempts: 0,
|
|
809
|
-
resubscribeTimer: null,
|
|
810
|
-
// v0.10 T-49-2: concurrent running Map replaces v0.9 single `current`.
|
|
811
|
-
running: new Map(),
|
|
812
|
-
concurrencyLimit: options.concurrencyLimit ?? 2,
|
|
813
|
-
queue: [],
|
|
814
|
-
seen: new Set(),
|
|
815
|
-
pumping: false,
|
|
816
|
-
stopped: false,
|
|
817
|
-
heartbeatFailures: 0,
|
|
818
|
-
refreshing: null,
|
|
819
|
-
pollSince: new Date(Date.now() - STARTUP_POLL_LOOKBACK_MS).toISOString(),
|
|
820
|
-
polling: false,
|
|
821
|
-
reviewQueue: [],
|
|
822
|
-
seenReviews: new Set(),
|
|
823
|
-
reviewPumping: false,
|
|
824
|
-
runReview: reviewerFactory,
|
|
825
|
-
quarantined: startupQuarantine !== null,
|
|
826
|
-
authRetriedTasks: new Set(),
|
|
827
|
-
idleTtlMs,
|
|
828
|
-
lastActivityMs: Date.now(),
|
|
829
|
-
idleTimer: null,
|
|
830
|
-
idleExiting: false,
|
|
831
|
-
taskRunnerFactory,
|
|
832
|
-
pausedCapacity: false,
|
|
833
|
-
pausedUntil: null,
|
|
834
|
-
pauseTimer: null,
|
|
835
|
-
capacityProbe: false,
|
|
836
|
-
capacityBackoffMs: options.capacityBackoffMs ?? DEFAULT_CAPACITY_BACKOFF_MS,
|
|
837
|
-
resumeJitterMs: options.resumeJitterMs ?? DEFAULT_RESUME_JITTER_MS,
|
|
838
|
-
singletonLock,
|
|
839
|
-
};
|
|
840
|
-
subscribeChannel(state, taskRunnerFactory);
|
|
841
|
-
// v0.34-B / v0.62 (T-62-1): one-shot startup claim scan. The 5-minute
|
|
842
|
-
// pollSince watermark (v0.30-B) can miss a task dispatched (runner_id set)
|
|
843
|
-
// more than 5 minutes before the runner started whose realtime task_assigned
|
|
844
|
-
// event was lost during the restart window. claimScan() scans with an epoch
|
|
845
|
-
// watermark so any task currently assigned to this runner_id is enqueued
|
|
846
|
-
// regardless of dispatch age. Does NOT update state.pollSince — the ongoing
|
|
847
|
-
// poll loop keeps its 5-min watermark. The same routine runs periodically
|
|
848
|
-
// below as the realtime-miss safety net.
|
|
849
|
-
await claimScan(state, taskRunnerFactory, "startup");
|
|
850
|
-
// v0.65 (T-65-2): one-shot startup review scan, the review twin of the
|
|
851
|
-
// startup claim scan above. A `review_assigned` broadcast dropped during the
|
|
852
|
-
// restart window would otherwise strand the review in acc.review_queue until
|
|
853
|
-
// a manual restart; the epoch-watermark scan re-enqueues any pending review
|
|
854
|
-
// assigned to this runner regardless of when it was requested. enqueueReview's
|
|
855
|
-
// seenReviews guard keeps it idempotent against the realtime path.
|
|
856
|
-
await reviewScan(state, "startup");
|
|
857
|
-
// v0.30-B: fire an immediate poll right after subscribing so any task
|
|
858
|
-
// dispatched before the runner came online (where the fire-and-forget
|
|
859
|
-
// Realtime broadcast went to a dead channel) is picked up at startup
|
|
860
|
-
// instead of waiting `pollMs` for the first scheduled poll. Awaited so
|
|
861
|
-
// `state.polling` is back to false before the timer-driven polls and
|
|
862
|
-
// any caller-driven `pollOnce()` can run.
|
|
863
|
-
await pollOnce(state, taskRunnerFactory);
|
|
864
|
-
// v0.6.0: self-rescheduling setTimeout (not setInterval) so the
|
|
865
|
-
// backoff ladder in nextHeartbeatDelayMs actually pauses between
|
|
866
|
-
// attempts. setInterval would keep firing every heartbeatMs even
|
|
867
|
-
// while the link is down.
|
|
868
|
-
scheduleHeartbeat(state, taskRunnerFactory, heartbeatMs);
|
|
869
|
-
// v0.12.0 (T-52-7): no refresh timer when the provider can't rotate —
|
|
870
|
-
// doRefresh would just emit the same "unavailable" line every 30 min.
|
|
871
|
-
state.refreshTimer = setInterval(() => {
|
|
872
|
-
if (state.tokenProvider.canRefresh() && expiresSoon(state.session)) {
|
|
873
|
-
void doRefresh(state, taskRunnerFactory);
|
|
874
|
-
}
|
|
875
|
-
}, REFRESH_CHECK_MS);
|
|
876
|
-
state.pollTimer = setInterval(() => {
|
|
877
|
-
void pollOnce(state, taskRunnerFactory);
|
|
878
|
-
}, pollMs);
|
|
879
|
-
// v0.62 (T-62-1): periodic full claim scan — the realtime-miss safety net.
|
|
880
|
-
// Independent of the delta-poll watermark, so a task assigned with an
|
|
881
|
-
// updated_at at/behind the watermark (and whose task_assigned broadcast was
|
|
882
|
-
// dropped) is re-claimed within one interval rather than waiting for a
|
|
883
|
-
// manual restart. Single-flight inside claimScan(); enqueue's seen-guard
|
|
884
|
-
// keeps it from double-claiming work the realtime path or delta-poll took.
|
|
885
|
-
state.claimScanTimer = setInterval(() => {
|
|
886
|
-
void claimScan(state, taskRunnerFactory, "periodic");
|
|
887
|
-
}, state.claimScanIntervalMs);
|
|
888
|
-
// v0.65 (T-65-2): periodic full review scan — the realtime-miss safety net
|
|
889
|
-
// for reviews. Mirrors the claim-scan timer above: a `review_assigned`
|
|
890
|
-
// broadcast missed by a stale subscription is recovered within one interval
|
|
891
|
-
// instead of stranding in acc.review_queue. Single-flight inside
|
|
892
|
-
// reviewScan(); enqueueReview's seenReviews guard prevents double-claiming a
|
|
893
|
-
// review the realtime path already took.
|
|
894
|
-
state.reviewScanTimer = setInterval(() => {
|
|
895
|
-
void reviewScan(state, "periodic");
|
|
896
|
-
}, state.reviewScanIntervalMs);
|
|
897
|
-
// v0.73 (T-73-2, G-M): claim-loop watchdog timer. Independent of the claim
|
|
898
|
-
// scan (whose 3-min cadence is too coarse for a 90s threshold): it ages the
|
|
899
|
-
// runner's own assigned-but-unclaimed tasks and forces a resubscribe + claim
|
|
900
|
-
// scan when one stays unclaimed past the threshold while heartbeats succeed —
|
|
901
|
-
// the self-heal for the "online + heartbeating but never claims" strand that
|
|
902
|
-
// T-68-1's channel recovery can't see (healthy channel, wedged claim path).
|
|
903
|
-
const claimWatchdogCheckMs = options.claimWatchdogCheckMs ?? DEFAULT_CLAIM_WATCHDOG_CHECK_MS;
|
|
904
|
-
state.claimWatchdogTimer = setInterval(() => {
|
|
905
|
-
void claimWatchdogScan(state, taskRunnerFactory);
|
|
906
|
-
}, claimWatchdogCheckMs);
|
|
907
|
-
// v0.13-MULTI-RUNNER: refresh the runner's capability tags so an
|
|
908
|
-
// operator change to ACC_RUNNER_CAPABILITIES (or to ~/.config/acc-
|
|
909
|
-
// runner/config.json) takes effect on the next restart without
|
|
910
|
-
// requiring a full `acc-runner login` re-flow. Default-empty is the
|
|
911
|
-
// backward-compat path for 0.6.3 runners and matches any task with
|
|
912
|
-
// an empty required_capabilities array.
|
|
913
|
-
{
|
|
914
|
-
const { error: capsErr } = await state.supabase.rpc("set_runner_capabilities", {
|
|
915
|
-
p_id: state.session.runner_id,
|
|
916
|
-
p_capabilities: cfg.capabilities ?? [],
|
|
917
|
-
});
|
|
918
|
-
if (capsErr) {
|
|
919
|
-
// Non-fatal: a pre-0114 server (or an Old DB the runner is
|
|
920
|
-
// pointed at during local dev) will 404 this RPC. The watch
|
|
921
|
-
// loop continues so an operator on an older deploy isn't
|
|
922
|
-
// wedged by a missing function.
|
|
923
|
-
process.stderr.write(`[acc-runner] set_runner_capabilities failed (continuing): ${capsErr.message}\n`);
|
|
924
|
-
}
|
|
925
|
-
}
|
|
926
|
-
// Initial heartbeat fires immediately so the runner page reflects "online".
|
|
927
|
-
await state.supabase.rpc("heartbeat_runner", {
|
|
928
|
-
p_id: state.session.runner_id,
|
|
929
|
-
p_version: PACKAGE_VERSION,
|
|
930
|
-
});
|
|
931
|
-
console.log(chalk.gray(`Runner ${state.session.runner_id} is online. Ctrl-C to stop.`));
|
|
932
|
-
const stop = async () => {
|
|
933
|
-
if (state.stopped)
|
|
934
|
-
return;
|
|
935
|
-
state.stopped = true;
|
|
936
|
-
clearTimeout(state.heartbeatTimer);
|
|
937
|
-
clearInterval(state.refreshTimer);
|
|
938
|
-
clearInterval(state.pollTimer);
|
|
939
|
-
clearInterval(state.claimScanTimer); // v0.62 (T-62-1)
|
|
940
|
-
clearInterval(state.reviewScanTimer); // v0.65 (T-65-2)
|
|
941
|
-
clearInterval(state.claimWatchdogTimer); // v0.73 (T-73-2)
|
|
942
|
-
if (state.resubscribeTimer)
|
|
943
|
-
clearTimeout(state.resubscribeTimer); // v0.68 (T-68-1)
|
|
944
|
-
if (state.idleTimer)
|
|
945
|
-
clearInterval(state.idleTimer);
|
|
946
|
-
if (state.pauseTimer)
|
|
947
|
-
clearTimeout(state.pauseTimer); // v0.56 (T-56-1)
|
|
948
|
-
// v0.10 T-49-2: cancel every in-flight task (replaces v0.9 single cancel).
|
|
949
|
-
for (const ctrl of state.running.values()) {
|
|
950
|
-
ctrl.cancel();
|
|
951
|
-
}
|
|
952
|
-
try {
|
|
953
|
-
await state.supabase.removeChannel(state.channel);
|
|
954
|
-
}
|
|
955
|
-
catch { /* ignore */ }
|
|
956
|
-
// v0.12-RUNNER-HEALTH: log shutdown_intent BEFORE flipping status so
|
|
957
|
-
// the audit trail records the intent even if the status RPC fails.
|
|
958
|
-
// The 1-min runner-watchdog cron uses status='online'+stale heartbeat
|
|
959
|
-
// as its crash detector; this event gives operators a timestamped
|
|
960
|
-
// breadcrumb when investigating "why did this runner go offline?"
|
|
961
|
-
try {
|
|
962
|
-
await state.supabase.rpc("log_activity", {
|
|
963
|
-
p_verb: "runner.shutdown_intent",
|
|
964
|
-
p_target_id: state.session.runner_id,
|
|
965
|
-
p_payload: { version: PACKAGE_VERSION },
|
|
966
|
-
p_target_type: "runner",
|
|
967
|
-
});
|
|
968
|
-
}
|
|
969
|
-
catch { /* best-effort; network may already be gone */ }
|
|
970
|
-
try {
|
|
971
|
-
await state.supabase.rpc("set_runner_status", {
|
|
972
|
-
p_id: state.session.runner_id,
|
|
973
|
-
p_status: "offline",
|
|
974
|
-
});
|
|
975
|
-
}
|
|
976
|
-
catch { /* network may already be gone */ }
|
|
977
|
-
// v0.14.0 (T-54-2): release the single-instance lock last, so the slot
|
|
978
|
-
// is only freed once this runner has fully deregistered. SIGINT/SIGTERM
|
|
979
|
-
// and the idle-TTL exit all route through stop(), so all clean shutdown
|
|
980
|
-
// paths release it. A hard crash leaves a stale lock that the next
|
|
981
|
-
// start reclaims.
|
|
982
|
-
try {
|
|
983
|
-
await state.singletonLock?.release();
|
|
984
|
-
}
|
|
985
|
-
catch { /* best-effort */ }
|
|
986
|
-
};
|
|
987
|
-
// v0.12.0 (T-52-7): arm the idle-TTL timer. Desktop runners (no TTL env,
|
|
988
|
-
// no option) never reach this branch — zero behaviour change. Armed after
|
|
989
|
-
// `stop` exists so the exit sequence can reuse the one shutdown path.
|
|
990
|
-
if (state.idleTtlMs !== null && state.idleTtlMs > 0) {
|
|
991
|
-
const checkMs = options.idleCheckMs ?? Math.min(IDLE_CHECK_MS, state.idleTtlMs);
|
|
992
|
-
state.idleTimer = setInterval(() => {
|
|
993
|
-
void maybeIdleExit(state, stop, exitFn);
|
|
994
|
-
}, checkMs);
|
|
995
|
-
// Don't hold the event loop open for the idle check alone.
|
|
996
|
-
state.idleTimer.unref?.();
|
|
997
|
-
console.log(chalk.gray(`Idle TTL armed: exiting after ${Math.round(state.idleTtlMs / 60_000)} min without work.`));
|
|
998
|
-
}
|
|
999
|
-
if (!options.taskRunnerFactory) {
|
|
1000
|
-
// v0.33-C: best-effort terminal status flush on runner shutdown. If
|
|
1001
|
-
// launchd (or another OS-driven kill) sends SIGTERM while a task is
|
|
1002
|
-
// mid-run, transition_task('failed') as a last resort so the task
|
|
1003
|
-
// doesn't sit in 'running' until the 5-15 min sweep fires. SIGINT
|
|
1004
|
-
// (Ctrl-C, operator-driven) skips the flush — the sweep is the right
|
|
1005
|
-
// path there because the task should be retried as-is, not failed.
|
|
1006
|
-
// SIGKILL / OOM can't be caught, so the sweep is still the real
|
|
1007
|
-
// safety net; this just shrinks the window when the kill is catchable.
|
|
1008
|
-
const onSignal = (signal) => {
|
|
1009
|
-
void (async () => {
|
|
1010
|
-
// v0.10 T-49-2: on SIGTERM, transition ALL in-flight tasks to
|
|
1011
|
-
// failed (replaces v0.9 single-task transition). SIGINT (Ctrl-C)
|
|
1012
|
-
// still skips the flush — the 5-min sweep handles that path.
|
|
1013
|
-
if (signal === "SIGTERM" && state.running.size > 0) {
|
|
1014
|
-
await Promise.allSettled([...state.running.keys()].map(async (taskId) => {
|
|
1015
|
-
try {
|
|
1016
|
-
await state.supabase.rpc("transition_task", {
|
|
1017
|
-
p_task_id: taskId,
|
|
1018
|
-
p_new_status: "failed",
|
|
1019
|
-
});
|
|
1020
|
-
}
|
|
1021
|
-
catch { /* best-effort — sweep covers us if RPC can't get out */ }
|
|
1022
|
-
}));
|
|
1023
|
-
}
|
|
1024
|
-
await stop();
|
|
1025
|
-
process.exit(0);
|
|
1026
|
-
})();
|
|
1027
|
-
};
|
|
1028
|
-
process.once("SIGINT", () => onSignal("SIGINT"));
|
|
1029
|
-
process.once("SIGTERM", () => onSignal("SIGTERM"));
|
|
1030
|
-
}
|
|
1031
|
-
return {
|
|
1032
|
-
stop,
|
|
1033
|
-
pollOnce: () => pollOnce(state, taskRunnerFactory),
|
|
1034
|
-
claimScanOnce: () => claimScan(state, taskRunnerFactory, "periodic"), // v0.62 (T-62-1)
|
|
1035
|
-
reviewScanOnce: () => reviewScan(state, "periodic"), // v0.65 (T-65-2)
|
|
1036
|
-
claimWatchdogOnce: () => claimWatchdogScan(state, taskRunnerFactory), // v0.73 (T-73-2)
|
|
1037
|
-
runningCount: () => state.running.size,
|
|
1038
|
-
};
|
|
1039
|
-
}
|
|
1040
|
-
function scheduleHeartbeat(state, taskRunnerFactory, baseMs) {
|
|
1041
|
-
if (state.stopped)
|
|
1042
|
-
return;
|
|
1043
|
-
const delay = nextHeartbeatDelayMs(state.heartbeatFailures, baseMs);
|
|
1044
|
-
state.heartbeatTimer = setTimeout(() => {
|
|
1045
|
-
void (async () => {
|
|
1046
|
-
const { error } = await state.supabase.rpc("heartbeat_runner", {
|
|
1047
|
-
p_id: state.session.runner_id,
|
|
1048
|
-
p_version: PACKAGE_VERSION,
|
|
1049
|
-
});
|
|
1050
|
-
if (!error) {
|
|
1051
|
-
state.heartbeatFailures = 0;
|
|
1052
|
-
}
|
|
1053
|
-
else {
|
|
1054
|
-
state.heartbeatFailures += 1;
|
|
1055
|
-
const kind = classifyHeartbeatError(error);
|
|
1056
|
-
process.stderr.write(`[acc-runner] heartbeat failed (${kind}): ${error.message}\n`);
|
|
1057
|
-
// v0.6.0 REG-294 sibling: only refresh on an auth-shaped error
|
|
1058
|
-
// AND once we've crossed the consecutive-failure threshold. A
|
|
1059
|
-
// single 401 on a flaky network shouldn't burn a refresh, and a
|
|
1060
|
-
// DNS outage shouldn't trigger refresh at all.
|
|
1061
|
-
if (state.heartbeatFailures >= HEARTBEAT_REFRESH_AFTER &&
|
|
1062
|
-
kind === "auth") {
|
|
1063
|
-
process.stderr.write("[acc-runner] auth-shaped heartbeat failure, attempting JWT refresh\n");
|
|
1064
|
-
void doRefresh(state, taskRunnerFactory);
|
|
1065
|
-
}
|
|
1066
|
-
if (state.heartbeatFailures >= HEARTBEAT_EXIT_AFTER) {
|
|
1067
|
-
process.stderr.write("[acc-runner] heartbeat dead, exiting\n");
|
|
1068
|
-
clearTimeout(state.heartbeatTimer);
|
|
1069
|
-
clearInterval(state.refreshTimer);
|
|
1070
|
-
clearInterval(state.pollTimer);
|
|
1071
|
-
clearInterval(state.claimScanTimer); // v0.62 (T-62-1)
|
|
1072
|
-
clearInterval(state.reviewScanTimer); // v0.65 (T-65-2)
|
|
1073
|
-
clearInterval(state.claimWatchdogTimer); // v0.73 (T-73-2)
|
|
1074
|
-
if (state.resubscribeTimer)
|
|
1075
|
-
clearTimeout(state.resubscribeTimer); // v0.68 (T-68-1)
|
|
1076
|
-
process.exit(1);
|
|
1077
|
-
}
|
|
1078
|
-
}
|
|
1079
|
-
scheduleHeartbeat(state, taskRunnerFactory, baseMs);
|
|
1080
|
-
})();
|
|
1081
|
-
}, delay);
|
|
1082
|
-
}
|
|
1083
|
-
/**
|
|
1084
|
-
* v0.12.0 (T-52-7): idle-TTL exit check. Fires from the idle timer; exits
|
|
1085
|
-
* the process cleanly when the runner has had no queued or in-flight work
|
|
1086
|
-
* (tasks OR reviews) for idleTtlMs. The exit sequence:
|
|
1087
|
-
*
|
|
1088
|
-
* 1. log_activity runner.idle_ttl_exit — the audit breadcrumb the
|
|
1089
|
-
* autoscale sweep and operators read ("why did acc-eph-x vanish?").
|
|
1090
|
-
* 2. final heartbeat_runner — so the runners panel shows a fresh
|
|
1091
|
-
* timestamp right up to the clean exit (vs. a stale-heartbeat crash).
|
|
1092
|
-
* 3. stop() — shutdown_intent + set_runner_status('offline'), the same
|
|
1093
|
-
* deregistration path `acc-runner logout` and SIGTERM use.
|
|
1094
|
-
* 4. exit(0).
|
|
1095
|
-
*
|
|
1096
|
-
* In-flight work always wins: any running task, queued task, or queued
|
|
1097
|
-
* review resets the decision to "not idle" — the TTL never kills a runner
|
|
1098
|
-
* mid-task.
|
|
1099
|
-
*/
|
|
1100
|
-
async function maybeIdleExit(state, stop, exitFn) {
|
|
1101
|
-
if (state.stopped || state.idleExiting || state.idleTtlMs === null)
|
|
1102
|
-
return;
|
|
1103
|
-
// v0.56 (T-56-1): a capacity pause is not idleness — the runner is
|
|
1104
|
-
// deliberately waiting for its session window to reopen. Don't TTL-exit
|
|
1105
|
-
// it out from under the scheduled resume.
|
|
1106
|
-
if (state.pausedCapacity)
|
|
1107
|
-
return;
|
|
1108
|
-
if (state.running.size > 0 ||
|
|
1109
|
-
state.queue.length > 0 ||
|
|
1110
|
-
state.reviewQueue.length > 0 ||
|
|
1111
|
-
state.reviewPumping) {
|
|
1112
|
-
return;
|
|
1113
|
-
}
|
|
1114
|
-
const idleMs = Date.now() - state.lastActivityMs;
|
|
1115
|
-
if (idleMs < state.idleTtlMs)
|
|
1116
|
-
return;
|
|
1117
|
-
state.idleExiting = true;
|
|
1118
|
-
const idleMinutes = Math.round(idleMs / 60_000);
|
|
1119
|
-
console.log(chalk.gray(`[acc-runner] idle TTL reached (${idleMinutes} min without work); exiting cleanly.`));
|
|
1120
|
-
try {
|
|
1121
|
-
await state.supabase.rpc("log_activity", {
|
|
1122
|
-
p_verb: "runner.idle_ttl_exit",
|
|
1123
|
-
p_target_id: state.session.runner_id,
|
|
1124
|
-
p_payload: { idle_minutes: idleMinutes, version: PACKAGE_VERSION },
|
|
1125
|
-
p_target_type: "runner",
|
|
1126
|
-
});
|
|
1127
|
-
}
|
|
1128
|
-
catch { /* best-effort */ }
|
|
1129
|
-
try {
|
|
1130
|
-
await state.supabase.rpc("heartbeat_runner", {
|
|
1131
|
-
p_id: state.session.runner_id,
|
|
1132
|
-
p_version: PACKAGE_VERSION,
|
|
1133
|
-
});
|
|
1134
|
-
}
|
|
1135
|
-
catch { /* best-effort */ }
|
|
1136
|
-
await stop();
|
|
1137
|
-
exitFn(0);
|
|
1138
|
-
}
|
|
1139
|
-
async function pumpReviews(state) {
|
|
1140
|
-
if (state.reviewPumping)
|
|
1141
|
-
return;
|
|
1142
|
-
// v0.56 (T-56-1): don't claim reviews while paused for capacity, and hold
|
|
1143
|
-
// reviews back during the single-probe window so the probe TASK confirms
|
|
1144
|
-
// the session reopened before we spend more capacity on reviews.
|
|
1145
|
-
if (state.pausedCapacity || state.capacityProbe)
|
|
1146
|
-
return;
|
|
1147
|
-
state.reviewPumping = true;
|
|
1148
|
-
try {
|
|
1149
|
-
while (!state.stopped &&
|
|
1150
|
-
!state.pausedCapacity &&
|
|
1151
|
-
!state.capacityProbe &&
|
|
1152
|
-
state.reviewQueue.length > 0) {
|
|
1153
|
-
const next = state.reviewQueue.shift();
|
|
1154
|
-
if (!next)
|
|
1155
|
-
continue;
|
|
1156
|
-
try {
|
|
1157
|
-
const outcome = await state.runReview(next, { supabase: state.supabase });
|
|
1158
|
-
const tag = outcome.decision === "reviewer_error" ? chalk.red : chalk.gray;
|
|
1159
|
-
console.log(tag(`[acc-runner] review ${next.review_id} task=${next.task_id} pr=${next.pr_number} decision=${outcome.decision} confidence=${outcome.confidence.toFixed(2)}`));
|
|
1160
|
-
// v0.56 (T-56-1): a reviewer that hit the session/usage limit reports
|
|
1161
|
-
// a DISTINCT outcome (reviewer_capacity, NOT reviewer_error) so the
|
|
1162
|
-
// api side (T-56-2) can exclude it from automerge retry caps and
|
|
1163
|
-
// re-dispatch the review. runReview has already submitted the
|
|
1164
|
-
// reviewer_capacity row; here we pause the whole runner exactly as a
|
|
1165
|
-
// capacity_exhausted task does.
|
|
1166
|
-
if (outcome.decision === "reviewer_capacity") {
|
|
1167
|
-
enterCapacityPause(state, next.task_id, outcome.resume_at ?? null, `reviewer ${next.review_id} hit session/usage limit`);
|
|
1168
|
-
return;
|
|
1169
|
-
}
|
|
1170
|
-
}
|
|
1171
|
-
catch (err) {
|
|
1172
|
-
process.stderr.write(`[acc-runner] review ${next.review_id} crashed: ${err.message}\n`);
|
|
1173
|
-
}
|
|
1174
|
-
finally {
|
|
1175
|
-
touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at review end
|
|
1176
|
-
}
|
|
1177
|
-
}
|
|
1178
|
-
}
|
|
1179
|
-
finally {
|
|
1180
|
-
state.reviewPumping = false;
|
|
1181
|
-
}
|
|
1182
|
-
}
|
|
1183
|
-
/**
|
|
1184
|
-
* v0.48 / v0.49 T-49-6: enter quarantine after a machine-level failure.
|
|
1185
|
-
*
|
|
1186
|
-
* Extracted from pump()'s completion handler so both the immediate
|
|
1187
|
-
* quarantine path and the auth_expired "refresh failed → quarantine"
|
|
1188
|
-
* path share one implementation. Idempotent: a no-op once the runner is
|
|
1189
|
-
* already quarantined, so concurrent machine-level failures don't fire
|
|
1190
|
-
* duplicate audit events. Sets the in-memory flag synchronously (pump()
|
|
1191
|
-
* reads it without an async hop) and best-effort persists + announces.
|
|
1192
|
-
*/
|
|
1193
|
-
function enterQuarantine(state, taskId, cause, detail,
|
|
1194
|
-
// v0.53 T-53-4: how many consecutive instant-empty exits produced an
|
|
1195
|
-
// env_broken cause (1 = definitive, 2 = heuristic confirmed by the
|
|
1196
|
-
// retry). Persisted into quarantine.json's additive `consecutive` field.
|
|
1197
|
-
consecutive) {
|
|
1198
|
-
if (state.quarantined)
|
|
1199
|
-
return;
|
|
1200
|
-
state.quarantined = true;
|
|
1201
|
-
const qState = {
|
|
1202
|
-
cause,
|
|
1203
|
-
classifiedAt: new Date().toISOString(),
|
|
1204
|
-
taskId,
|
|
1205
|
-
detail,
|
|
1206
|
-
...(consecutive !== undefined ? { consecutive } : {}),
|
|
1207
|
-
};
|
|
1208
|
-
void setQuarantine(qState).catch((err) => {
|
|
1209
|
-
process.stderr.write(`[acc-runner] setQuarantine failed: ${err.message}\n`);
|
|
1210
|
-
});
|
|
1211
|
-
process.stderr.write(chalk.red.bold(`[acc-runner] QUARANTINE: runner ${state.session.runner_id} quarantined ` +
|
|
1212
|
-
`(${cause}) after task ${taskId} failed.\n`));
|
|
1213
|
-
void postRunnerStateMessage(state.supabase, {
|
|
1214
|
-
task_id: taskId,
|
|
1215
|
-
sender_id: state.session.runner_id,
|
|
1216
|
-
state: "blocked",
|
|
1217
|
-
protocol: "error_context",
|
|
1218
|
-
payload: {
|
|
1219
|
-
capacity_alert: true,
|
|
1220
|
-
quarantine_cause: cause,
|
|
1221
|
-
runner_id: state.session.runner_id,
|
|
1222
|
-
detail,
|
|
1223
|
-
},
|
|
1224
|
-
kind: "blocked",
|
|
1225
|
-
}).catch((err) => {
|
|
1226
|
-
process.stderr.write(`[acc-runner] capacity alert post failed: ${err.message}\n`);
|
|
1227
|
-
});
|
|
1228
|
-
void (async () => {
|
|
1229
|
-
try {
|
|
1230
|
-
await state.supabase.rpc("log_activity", {
|
|
1231
|
-
p_verb: "runner.quarantine_enter",
|
|
1232
|
-
p_target_id: state.session.runner_id,
|
|
1233
|
-
p_payload: { cause, task_id: taskId, detail },
|
|
1234
|
-
p_target_type: "runner",
|
|
1235
|
-
});
|
|
1236
|
-
}
|
|
1237
|
-
catch { /* best-effort */ }
|
|
1238
|
-
})();
|
|
1239
|
-
}
|
|
1240
|
-
/**
|
|
1241
|
-
* v0.56 (T-56-1): enter a capacity pause.
|
|
1242
|
-
*
|
|
1243
|
-
* Triggered when a task OR a review reports capacity_exhausted (Claude
|
|
1244
|
-
* account out of session/usage capacity). The runner:
|
|
1245
|
-
* - stops claiming tasks AND reviews (pump/pumpReviews bail on the flag);
|
|
1246
|
-
* - schedules an automatic resume at the stated reset time (or a default
|
|
1247
|
-
* backoff when none was parseable) plus a small jitter;
|
|
1248
|
-
* - keeps heartbeating so the board shows online-but-paused, and emits a
|
|
1249
|
-
* `runner.paused_capacity` activity event + a runner-level bus message
|
|
1250
|
-
* carrying `resume_at` so health/board surfaces can show WHY the fleet
|
|
1251
|
-
* is idle (heartbeat_runner's RPC shape is frozen, so the detail is
|
|
1252
|
-
* carried additively through these existing channels).
|
|
1253
|
-
*
|
|
1254
|
-
* Idempotent: a second capacity hit while already paused only ever pushes
|
|
1255
|
-
* the resume later, never earlier, so a straggler task that 429s a moment
|
|
1256
|
-
* after the first one can't shorten the wait.
|
|
1257
|
-
*/
|
|
1258
|
-
function enterCapacityPause(state, triggerTaskId, resumeAtIso, detail) {
|
|
1259
|
-
const now = Date.now();
|
|
1260
|
-
const parsed = resumeAtIso ? Date.parse(resumeAtIso) : NaN;
|
|
1261
|
-
const resumeMs = Number.isFinite(parsed) && parsed > now ? parsed : now + state.capacityBackoffMs;
|
|
1262
|
-
const source = Number.isFinite(parsed) && parsed > now ? "reset_time" : "default_backoff";
|
|
1263
|
-
// Already paused: only extend the window, never shorten it.
|
|
1264
|
-
if (state.pausedCapacity && state.pausedUntil !== null && resumeMs <= state.pausedUntil) {
|
|
1265
|
-
return;
|
|
1266
|
-
}
|
|
1267
|
-
state.pausedCapacity = true;
|
|
1268
|
-
state.pausedUntil = resumeMs;
|
|
1269
|
-
if (state.pauseTimer) {
|
|
1270
|
-
clearTimeout(state.pauseTimer);
|
|
1271
|
-
state.pauseTimer = null;
|
|
1272
|
-
}
|
|
1273
|
-
const jitter = state.resumeJitterMs > 0 ? Math.floor(Math.random() * state.resumeJitterMs) : 0;
|
|
1274
|
-
const delay = Math.max(0, resumeMs - now) + jitter;
|
|
1275
|
-
const resumeAtFinal = new Date(now + delay).toISOString();
|
|
1276
|
-
process.stderr.write(chalk.yellow(`[acc-runner] PAUSED (capacity): ${detail}. Not claiming tasks or reviews; ` +
|
|
1277
|
-
`resuming at ${resumeAtFinal} (${source}).\n`));
|
|
1278
|
-
void (async () => {
|
|
1279
|
-
try {
|
|
1280
|
-
await state.supabase.rpc("log_activity", {
|
|
1281
|
-
p_verb: "runner.paused_capacity",
|
|
1282
|
-
p_target_id: state.session.runner_id,
|
|
1283
|
-
p_payload: {
|
|
1284
|
-
resume_at: resumeAtFinal,
|
|
1285
|
-
source,
|
|
1286
|
-
detail,
|
|
1287
|
-
version: PACKAGE_VERSION,
|
|
1288
|
-
},
|
|
1289
|
-
p_target_type: "runner",
|
|
1290
|
-
});
|
|
1291
|
-
}
|
|
1292
|
-
catch { /* best-effort */ }
|
|
1293
|
-
})();
|
|
1294
|
-
// Runner-level bus message so operators see the paused status + reason on
|
|
1295
|
-
// /messages even though heartbeat_runner itself can't carry a detail field.
|
|
1296
|
-
// Anchored to the triggering task so the RPC can resolve the org scope
|
|
1297
|
-
// (post_agent_message looks up org_id from acc.tasks for a non-null id).
|
|
1298
|
-
void postRunnerStateMessage(state.supabase, {
|
|
1299
|
-
task_id: triggerTaskId,
|
|
1300
|
-
sender_id: state.session.runner_id,
|
|
1301
|
-
state: "blocked",
|
|
1302
|
-
protocol: "info",
|
|
1303
|
-
payload: {
|
|
1304
|
-
capacity_paused: true,
|
|
1305
|
-
runner_id: state.session.runner_id,
|
|
1306
|
-
resume_at: resumeAtFinal,
|
|
1307
|
-
source,
|
|
1308
|
-
detail,
|
|
1309
|
-
},
|
|
1310
|
-
kind: "blocked",
|
|
1311
|
-
}).catch((err) => {
|
|
1312
|
-
process.stderr.write(`[acc-runner] capacity pause bus post failed: ${err.message}\n`);
|
|
1313
|
-
});
|
|
1314
|
-
state.pauseTimer = setTimeout(() => {
|
|
1315
|
-
void resumeFromCapacity(state);
|
|
1316
|
-
}, delay);
|
|
1317
|
-
state.pauseTimer.unref?.();
|
|
1318
|
-
}
|
|
1319
|
-
/**
|
|
1320
|
-
* v0.56 (T-56-1): resume from a capacity pause.
|
|
1321
|
-
*
|
|
1322
|
-
* Clears the pause and arms the single-probe clamp: the runner claims at
|
|
1323
|
-
* most ONE task to confirm the window actually reopened before resuming
|
|
1324
|
-
* full task concurrency and reviews. If the probe itself 429s, the
|
|
1325
|
-
* completion handler re-enters the pause; if it succeeds, the probe clamp
|
|
1326
|
-
* clears and normal claiming resumes.
|
|
1327
|
-
*/
|
|
1328
|
-
async function resumeFromCapacity(state) {
|
|
1329
|
-
if (state.stopped)
|
|
1330
|
-
return;
|
|
1331
|
-
state.pausedCapacity = false;
|
|
1332
|
-
state.pausedUntil = null;
|
|
1333
|
-
if (state.pauseTimer) {
|
|
1334
|
-
clearTimeout(state.pauseTimer);
|
|
1335
|
-
state.pauseTimer = null;
|
|
1336
|
-
}
|
|
1337
|
-
state.capacityProbe = true;
|
|
1338
|
-
console.log(chalk.gray(`[acc-runner] capacity window reopened — resuming with a single probe task.`));
|
|
1339
|
-
try {
|
|
1340
|
-
await state.supabase.rpc("log_activity", {
|
|
1341
|
-
p_verb: "runner.resume_capacity",
|
|
1342
|
-
p_target_id: state.session.runner_id,
|
|
1343
|
-
p_payload: { probe: true, version: PACKAGE_VERSION },
|
|
1344
|
-
p_target_type: "runner",
|
|
1345
|
-
});
|
|
1346
|
-
}
|
|
1347
|
-
catch { /* best-effort */ }
|
|
1348
|
-
// Poll first (the stale-running sweep may have re-queued the released task
|
|
1349
|
-
// to another runner; this picks up whatever is still assigned here), then
|
|
1350
|
-
// pump — the probe clamp bounds it to one task.
|
|
1351
|
-
await pollOnce(state, state.taskRunnerFactory);
|
|
1352
|
-
void pump(state, state.taskRunnerFactory);
|
|
1353
|
-
}
|
|
1354
|
-
/**
|
|
1355
|
-
* v0.10 T-49-2: concurrent task pump.
|
|
1356
|
-
*
|
|
1357
|
-
* Launches up to state.concurrencyLimit tasks in parallel. Each task
|
|
1358
|
-
* gets its own isolated git worktree (prepareTaskWorktree in task-runner)
|
|
1359
|
-
* so there is zero shared-file risk between concurrent runs. Database-side
|
|
1360
|
-
* `claim_task_with_locks` provides a second guard: it rejects a claim when
|
|
1361
|
-
* the task's file scope overlaps another in-flight task.
|
|
1362
|
-
*
|
|
1363
|
-
* Design: rather than `await`ing each task inside a while loop (v0.9
|
|
1364
|
-
* serial behaviour), tasks are launched fire-and-forget. Each task's
|
|
1365
|
-
* `.then()` handler removes it from `state.running` and re-fires pump()
|
|
1366
|
-
* so newly freed capacity immediately picks up the next queued item.
|
|
1367
|
-
* The `pumping` mutex prevents two concurrent pump() invocations from
|
|
1368
|
-
* both launching into the same concurrencyLimit slot.
|
|
1369
|
-
*/
|
|
1370
|
-
async function pump(state, factory) {
|
|
1371
|
-
if (state.pumping)
|
|
1372
|
-
return;
|
|
1373
|
-
state.pumping = true;
|
|
1374
|
-
try {
|
|
1375
|
-
// v0.48: check quarantine before claiming any task.
|
|
1376
|
-
if (state.quarantined) {
|
|
1377
|
-
process.stderr.write(`[acc-runner] QUARANTINED: runner ${state.session.runner_id} is not claiming tasks.\n` +
|
|
1378
|
-
`[acc-runner] Fix the underlying issue then run \`acc-runner quarantine clear\`.\n`);
|
|
1379
|
-
return;
|
|
1380
|
-
}
|
|
1381
|
-
// v0.56 (T-56-1): out of session/usage capacity — don't claim. Queued
|
|
1382
|
-
// tasks stay queued; the scheduled resume re-drives pump().
|
|
1383
|
-
if (state.pausedCapacity) {
|
|
1384
|
-
return;
|
|
1385
|
-
}
|
|
1386
|
-
// v0.56 (T-56-1): during the post-resume probe window, claim exactly one
|
|
1387
|
-
// task to confirm the session reopened before unleashing full concurrency.
|
|
1388
|
-
const effectiveLimit = state.capacityProbe ? 1 : state.concurrencyLimit;
|
|
1389
|
-
// Fill available concurrency slots from the queue.
|
|
1390
|
-
while (!state.stopped &&
|
|
1391
|
-
!state.quarantined &&
|
|
1392
|
-
!state.pausedCapacity &&
|
|
1393
|
-
state.queue.length > 0 &&
|
|
1394
|
-
state.running.size < effectiveLimit) {
|
|
1395
|
-
const next = state.queue.shift();
|
|
1396
|
-
if (!next)
|
|
1397
|
-
continue;
|
|
1398
|
-
const ctrl = factory(next, {
|
|
1399
|
-
supabase: state.supabase,
|
|
1400
|
-
cfg: state.cfg,
|
|
1401
|
-
session: {
|
|
1402
|
-
accessToken: state.session.access_token,
|
|
1403
|
-
runnerId: state.session.runner_id,
|
|
1404
|
-
},
|
|
1405
|
-
publicUrl: state.cfg.publicUrl,
|
|
1406
|
-
});
|
|
1407
|
-
state.running.set(next, ctrl);
|
|
1408
|
-
// Detached completion handler. Runs AFTER pump() returns so the
|
|
1409
|
-
// running Map slot is occupied by the time the while condition
|
|
1410
|
-
// re-checks running.size on the next iteration.
|
|
1411
|
-
void ctrl.promise
|
|
1412
|
-
.then((outcome) => {
|
|
1413
|
-
state.running.delete(next);
|
|
1414
|
-
touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at task end
|
|
1415
|
-
// v0.56 (T-56-1): capacity exhaustion pauses the WHOLE runner —
|
|
1416
|
-
// never a task_error, never quarantine. The task was left 'running'
|
|
1417
|
-
// for the stale-running sweep to requeue losslessly (runner_id
|
|
1418
|
-
// cleared). Do not re-fire pump for new work; the scheduled resume
|
|
1419
|
-
// does that with a single probe.
|
|
1420
|
-
if (outcome.status === "capacity_paused" || outcome.capacity_exhausted) {
|
|
1421
|
-
enterCapacityPause(state, next, outcome.resume_at ?? null, outcome.error || "task hit session/usage limit");
|
|
1422
|
-
return;
|
|
1423
|
-
}
|
|
1424
|
-
// v0.21 T-66-1: a transient git-provision contention requeue. runTask
|
|
1425
|
-
// already returned the task to 'queued' server-side for a clean
|
|
1426
|
-
// retry — NOT a failure (no quarantine, no pause, no retry-budget
|
|
1427
|
-
// burn). Fall through to the pump re-fire below so the runner stays
|
|
1428
|
-
// healthy and the requeued task re-dispatches normally.
|
|
1429
|
-
if (outcome.status === "requeued") {
|
|
1430
|
-
process.stderr.write(`[acc-runner] task ${next} requeued (transient): ${outcome.error ?? "git provision contention"}\n`);
|
|
1431
|
-
}
|
|
1432
|
-
// v0.14.0 (T-54-2, GA-6): record the claude CLI version at the last
|
|
1433
|
-
// successful task so `doctor` can WARN when claude auto-updates
|
|
1434
|
-
// under the runner. Best-effort and detached — never blocks pump.
|
|
1435
|
-
if (outcome.status === "ok") {
|
|
1436
|
-
void getClaudeVersion()
|
|
1437
|
-
.then((v) => recordTaskClaudeVersion(v))
|
|
1438
|
-
.catch(() => {
|
|
1439
|
-
/* best-effort cache write */
|
|
1440
|
-
});
|
|
1441
|
-
}
|
|
1442
|
-
if (outcome.status === "failed") {
|
|
1443
|
-
const reason = outcome.error?.trim() ||
|
|
1444
|
-
(outcome.exitCode != null ? `exit ${outcome.exitCode}` : "unknown");
|
|
1445
|
-
process.stderr.write(`[acc-runner] task ${next} phase=${outcome.phase ?? "unknown"} failed: ${reason}\n`);
|
|
1446
|
-
// v0.48: machine-level failure → quarantine. Guard with
|
|
1447
|
-
// !state.quarantined so concurrent tasks don't fire duplicate
|
|
1448
|
-
// quarantine events when two machine-level failures land
|
|
1449
|
-
// simultaneously (both already in-flight; only the first to
|
|
1450
|
-
// resolve sets the flag and writes the audit trail).
|
|
1451
|
-
if (outcome.quarantine_cause && !state.quarantined) {
|
|
1452
|
-
const cause = outcome.quarantine_cause;
|
|
1453
|
-
// v0.49 T-49-6: a transient auth_expired (the access token
|
|
1454
|
-
// lapsed while the task ran) gets ONE JWT refresh before we
|
|
1455
|
-
// quarantine. If the refresh recovers, the runner stays
|
|
1456
|
-
// online and resumes claiming queued work; if it fails — or
|
|
1457
|
-
// this task already burned its one retry — we quarantine.
|
|
1458
|
-
// usage_limit / env_broken skip this path: a token refresh
|
|
1459
|
-
// can't fix a spent quota or a broken host.
|
|
1460
|
-
if (cause === "auth_expired" && !state.authRetriedTasks.has(next)) {
|
|
1461
|
-
state.authRetriedTasks.add(next);
|
|
1462
|
-
process.stderr.write(chalk.yellow(`[acc-runner] auth_expired after task ${next}; attempting one ` +
|
|
1463
|
-
`token refresh before quarantine.\n`));
|
|
1464
|
-
void doRefresh(state, factory).then((ok) => {
|
|
1465
|
-
if (ok) {
|
|
1466
|
-
process.stderr.write(chalk.green(`[acc-runner] token refresh recovered auth after task ${next}; ` +
|
|
1467
|
-
`staying online.\n`));
|
|
1468
|
-
void (async () => {
|
|
1469
|
-
try {
|
|
1470
|
-
await state.supabase.rpc("log_activity", {
|
|
1471
|
-
p_verb: "runner.auth_retry_recovered",
|
|
1472
|
-
p_target_id: state.session.runner_id,
|
|
1473
|
-
p_payload: { task_id: next, cause },
|
|
1474
|
-
p_target_type: "runner",
|
|
1475
|
-
});
|
|
1476
|
-
}
|
|
1477
|
-
catch { /* best-effort */ }
|
|
1478
|
-
})();
|
|
1479
|
-
// Resume claiming: the freed slot can pick up queued work
|
|
1480
|
-
// now that the JWT is fresh.
|
|
1481
|
-
void pump(state, factory);
|
|
1482
|
-
}
|
|
1483
|
-
else {
|
|
1484
|
-
enterQuarantine(state, next, cause, reason, outcome.quarantine_consecutive);
|
|
1485
|
-
}
|
|
1486
|
-
});
|
|
1487
|
-
return;
|
|
1488
|
-
}
|
|
1489
|
-
enterQuarantine(state, next, cause, reason, outcome.quarantine_consecutive);
|
|
1490
|
-
// Quarantined — do not re-fire pump for new tasks.
|
|
1491
|
-
// Other in-flight tasks (still in state.running) complete
|
|
1492
|
-
// normally; we just stop accepting new work.
|
|
1493
|
-
return;
|
|
1494
|
-
}
|
|
1495
|
-
}
|
|
1496
|
-
// v0.56 (T-56-1): a non-capacity completion (ok, or a genuine
|
|
1497
|
-
// task_error) proves the session window is open — clear the probe
|
|
1498
|
-
// clamp and let reviews resume alongside full task concurrency.
|
|
1499
|
-
if (state.capacityProbe) {
|
|
1500
|
-
state.capacityProbe = false;
|
|
1501
|
-
void pumpReviews(state);
|
|
1502
|
-
}
|
|
1503
|
-
// Task done (ok, failed without quarantine, or cancelled).
|
|
1504
|
-
// Re-fire pump so the next queued task can claim the freed slot.
|
|
1505
|
-
void pump(state, factory);
|
|
1506
|
-
})
|
|
1507
|
-
.catch((err) => {
|
|
1508
|
-
state.running.delete(next);
|
|
1509
|
-
touchActivity(state); // v0.12.0 (T-52-7): idle clock restarts at task end
|
|
1510
|
-
process.stderr.write(`[acc-runner] task ${next} crashed: ${err.message}\n`);
|
|
1511
|
-
// v0.56 (T-56-1): a crash is not a capacity signal — clear the probe
|
|
1512
|
-
// clamp so the runner doesn't stay stuck at concurrency 1.
|
|
1513
|
-
if (state.capacityProbe) {
|
|
1514
|
-
state.capacityProbe = false;
|
|
1515
|
-
void pumpReviews(state);
|
|
1516
|
-
}
|
|
1517
|
-
void pump(state, factory);
|
|
1518
|
-
});
|
|
1519
|
-
}
|
|
1520
|
-
// v0.56 (T-56-1): the probe found nothing to test with (no queued task at
|
|
1521
|
-
// resume) — clear the clamp and release any held reviews so an idle
|
|
1522
|
-
// resume doesn't strand pending reviews behind a probe that never runs.
|
|
1523
|
-
if (state.capacityProbe && state.running.size === 0 && state.queue.length === 0) {
|
|
1524
|
-
state.capacityProbe = false;
|
|
1525
|
-
void pumpReviews(state);
|
|
1526
|
-
}
|
|
1527
|
-
}
|
|
1528
|
-
finally {
|
|
1529
|
-
state.pumping = false;
|
|
1530
|
-
}
|
|
1531
|
-
}
|
|
1532
|
-
//# sourceMappingURL=watch.js.map
|