@tokenfactory/acc-runner 0.25.1 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +77 -4
- package/package.json +1 -1
- package/dist/anthropic-auth.d.ts +0 -53
- package/dist/anthropic-auth.d.ts.map +0 -1
- package/dist/anthropic-auth.js +0 -89
- package/dist/anthropic-auth.js.map +0 -1
- package/dist/bin-resolve.d.ts +0 -16
- package/dist/bin-resolve.d.ts.map +0 -1
- package/dist/bin-resolve.js +0 -35
- package/dist/bin-resolve.js.map +0 -1
- package/dist/cli.d.ts +0 -3
- package/dist/cli.d.ts.map +0 -1
- package/dist/cli.js +0 -128
- package/dist/cli.js.map +0 -1
- package/dist/config.d.ts +0 -121
- package/dist/config.d.ts.map +0 -1
- package/dist/config.js +0 -396
- package/dist/config.js.map +0 -1
- package/dist/cost-pricing.d.ts +0 -130
- package/dist/cost-pricing.d.ts.map +0 -1
- package/dist/cost-pricing.js +0 -191
- package/dist/cost-pricing.js.map +0 -1
- package/dist/doctor.d.ts +0 -37
- package/dist/doctor.d.ts.map +0 -1
- package/dist/doctor.js +0 -658
- package/dist/doctor.js.map +0 -1
- package/dist/failure-classifier.d.ts +0 -112
- package/dist/failure-classifier.d.ts.map +0 -1
- package/dist/failure-classifier.js +0 -353
- package/dist/failure-classifier.js.map +0 -1
- package/dist/gh.d.ts +0 -22
- package/dist/gh.d.ts.map +0 -1
- package/dist/gh.js +0 -48
- package/dist/gh.js.map +0 -1
- package/dist/git.d.ts +0 -50
- package/dist/git.d.ts.map +0 -1
- package/dist/git.js +0 -127
- package/dist/git.js.map +0 -1
- package/dist/github-client.d.ts +0 -133
- package/dist/github-client.d.ts.map +0 -1
- package/dist/github-client.js +0 -234
- package/dist/github-client.js.map +0 -1
- package/dist/keychain.d.ts +0 -21
- package/dist/keychain.d.ts.map +0 -1
- package/dist/keychain.js +0 -45
- package/dist/keychain.js.map +0 -1
- package/dist/login.d.ts +0 -12
- package/dist/login.d.ts.map +0 -1
- package/dist/login.js +0 -133
- package/dist/login.js.map +0 -1
- package/dist/logout.d.ts +0 -2
- package/dist/logout.d.ts.map +0 -1
- package/dist/logout.js +0 -31
- package/dist/logout.js.map +0 -1
- package/dist/mcp-spawn.d.ts +0 -30
- package/dist/mcp-spawn.d.ts.map +0 -1
- package/dist/mcp-spawn.js +0 -145
- package/dist/mcp-spawn.js.map +0 -1
- package/dist/messaging.d.ts +0 -49
- package/dist/messaging.d.ts.map +0 -1
- package/dist/messaging.js +0 -36
- package/dist/messaging.js.map +0 -1
- package/dist/pkg-version.d.ts +0 -3
- package/dist/pkg-version.d.ts.map +0 -1
- package/dist/pkg-version.js +0 -20
- package/dist/pkg-version.js.map +0 -1
- package/dist/profiles/designer-prompt.d.ts +0 -18
- package/dist/profiles/designer-prompt.d.ts.map +0 -1
- package/dist/profiles/designer-prompt.js +0 -172
- package/dist/profiles/designer-prompt.js.map +0 -1
- package/dist/profiles/developer-prompt.d.ts +0 -24
- package/dist/profiles/developer-prompt.d.ts.map +0 -1
- package/dist/profiles/developer-prompt.js +0 -24
- package/dist/profiles/developer-prompt.js.map +0 -1
- package/dist/profiles/manager-prompt.d.ts +0 -34
- package/dist/profiles/manager-prompt.d.ts.map +0 -1
- package/dist/profiles/manager-prompt.js +0 -93
- package/dist/profiles/manager-prompt.js.map +0 -1
- package/dist/profiles/tester-prompt.d.ts +0 -28
- package/dist/profiles/tester-prompt.d.ts.map +0 -1
- package/dist/profiles/tester-prompt.js +0 -165
- package/dist/profiles/tester-prompt.js.map +0 -1
- package/dist/prompt.d.ts +0 -38
- package/dist/prompt.d.ts.map +0 -1
- package/dist/prompt.js +0 -78
- package/dist/prompt.js.map +0 -1
- package/dist/runtime/cache-dir.d.ts +0 -2
- package/dist/runtime/cache-dir.d.ts.map +0 -1
- package/dist/runtime/cache-dir.js +0 -15
- package/dist/runtime/cache-dir.js.map +0 -1
- package/dist/runtime/conflict-resolver.d.ts +0 -65
- package/dist/runtime/conflict-resolver.d.ts.map +0 -1
- package/dist/runtime/conflict-resolver.js +0 -477
- package/dist/runtime/conflict-resolver.js.map +0 -1
- package/dist/runtime/expand-args.d.ts +0 -28
- package/dist/runtime/expand-args.d.ts.map +0 -1
- package/dist/runtime/expand-args.js +0 -50
- package/dist/runtime/expand-args.js.map +0 -1
- package/dist/runtime/locks.d.ts +0 -21
- package/dist/runtime/locks.d.ts.map +0 -1
- package/dist/runtime/locks.js +0 -97
- package/dist/runtime/locks.js.map +0 -1
- package/dist/runtime/provision-mutex.d.ts +0 -37
- package/dist/runtime/provision-mutex.d.ts.map +0 -1
- package/dist/runtime/provision-mutex.js +0 -67
- package/dist/runtime/provision-mutex.js.map +0 -1
- package/dist/runtime/quarantine.d.ts +0 -26
- package/dist/runtime/quarantine.d.ts.map +0 -1
- package/dist/runtime/quarantine.js +0 -50
- package/dist/runtime/quarantine.js.map +0 -1
- package/dist/runtime/resolution-integrity.d.ts +0 -86
- package/dist/runtime/resolution-integrity.d.ts.map +0 -1
- package/dist/runtime/resolution-integrity.js +0 -248
- package/dist/runtime/resolution-integrity.js.map +0 -1
- package/dist/runtime/reviewer.d.ts +0 -81
- package/dist/runtime/reviewer.d.ts.map +0 -1
- package/dist/runtime/reviewer.js +0 -374
- package/dist/runtime/reviewer.js.map +0 -1
- package/dist/runtime/rework.d.ts +0 -48
- package/dist/runtime/rework.d.ts.map +0 -1
- package/dist/runtime/rework.js +0 -136
- package/dist/runtime/rework.js.map +0 -1
- package/dist/runtime/singleton.d.ts +0 -47
- package/dist/runtime/singleton.d.ts.map +0 -1
- package/dist/runtime/singleton.js +0 -200
- package/dist/runtime/singleton.js.map +0 -1
- package/dist/runtime/version-drift.d.ts +0 -31
- package/dist/runtime/version-drift.d.ts.map +0 -1
- package/dist/runtime/version-drift.js +0 -114
- package/dist/runtime/version-drift.js.map +0 -1
- package/dist/runtime/worktree.d.ts +0 -74
- package/dist/runtime/worktree.d.ts.map +0 -1
- package/dist/runtime/worktree.js +0 -206
- package/dist/runtime/worktree.js.map +0 -1
- package/dist/secrets/inject.d.ts +0 -70
- package/dist/secrets/inject.d.ts.map +0 -1
- package/dist/secrets/inject.js +0 -102
- package/dist/secrets/inject.js.map +0 -1
- package/dist/supabase.d.ts +0 -4
- package/dist/supabase.d.ts.map +0 -1
- package/dist/supabase.js +0 -36
- package/dist/supabase.js.map +0 -1
- package/dist/task-runner.d.ts +0 -313
- package/dist/task-runner.d.ts.map +0 -1
- package/dist/task-runner.js +0 -1766
- package/dist/task-runner.js.map +0 -1
- package/dist/token-provider.d.ts +0 -50
- package/dist/token-provider.d.ts.map +0 -1
- package/dist/token-provider.js +0 -177
- package/dist/token-provider.js.map +0 -1
- package/dist/types.d.ts +0 -120
- package/dist/types.d.ts.map +0 -1
- package/dist/types.js +0 -16
- package/dist/types.js.map +0 -1
- package/dist/version-check.d.ts +0 -17
- package/dist/version-check.d.ts.map +0 -1
- package/dist/version-check.js +0 -56
- package/dist/version-check.js.map +0 -1
- package/dist/watch.d.ts +0 -295
- package/dist/watch.d.ts.map +0 -1
- package/dist/watch.js +0 -1532
- package/dist/watch.js.map +0 -1
package/dist/task-runner.js
DELETED
|
@@ -1,1766 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Task execution path. Called by watch.ts when a task_assigned broadcast
|
|
3
|
-
* lands on the runner channel.
|
|
4
|
-
*
|
|
5
|
-
* 1. Transition task → running (RPC).
|
|
6
|
-
* 2. fetch_task_for_runner to get the task + adjacent agent/model/runner.
|
|
7
|
-
* 3. Render the prompt, ensure repo is fresh, branch is checked out.
|
|
8
|
-
* 4. Spawn `claude --print` with the prompt on stdin; pipe stdout/stderr
|
|
9
|
-
* to acc.append_task_event so the History tab updates live.
|
|
10
|
-
* 5. On success, push the branch and open a PR via `gh`.
|
|
11
|
-
* 6. On failure, transition → failed and emit an error task_event.
|
|
12
|
-
*
|
|
13
|
-
* Cancellation: watch.ts holds a reference to the running child via the
|
|
14
|
-
* returned controller and SIGTERMs it on task_cancelled.
|
|
15
|
-
*/
|
|
16
|
-
import fs from "node:fs";
|
|
17
|
-
import { existsSync, unlinkSync } from "node:fs";
|
|
18
|
-
import os from "node:os";
|
|
19
|
-
import path from "node:path";
|
|
20
|
-
import { join } from "node:path";
|
|
21
|
-
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
22
|
-
import { execa } from "execa";
|
|
23
|
-
import { withAnthropicAuth } from "./anthropic-auth.js";
|
|
24
|
-
import { loadProfile as defaultLoadProfile } from "./config.js";
|
|
25
|
-
import { normalizeUsage, parseClaudeJson, priceUsdCents, toCliAlias, } from "./cost-pricing.js";
|
|
26
|
-
import { classifyClaudeExit, extractResetTime, isGitProvisionContention, } from "./failure-classifier.js";
|
|
27
|
-
import { startTaskGithubBudget, endTaskGithubBudget, } from "./github-client.js";
|
|
28
|
-
import { git as defaultGit } from "./git.js";
|
|
29
|
-
import { gh as defaultGh } from "./gh.js";
|
|
30
|
-
import { writeMcpConfig as defaultWriteMcpConfig, } from "./mcp-spawn.js";
|
|
31
|
-
import { postRunnerStateMessage as defaultPostRunnerStateMessage, } from "./messaging.js";
|
|
32
|
-
import { branchForTask, prTitleForTask, renderTaskPrompt, } from "./prompt.js";
|
|
33
|
-
import { acquireTaskLock as defaultAcquireTaskLock, TaskLockHeldError, } from "./runtime/locks.js";
|
|
34
|
-
import { prepareTaskWorktree as defaultPrepareTaskWorktree, } from "./runtime/worktree.js";
|
|
35
|
-
// v0.21 T-66-1: serialize the per-task worktree-provisioning critical section
|
|
36
|
-
// against the shared clone so concurrent tasks can't collide on git's
|
|
37
|
-
// repo-global locks (the T-65-2 race).
|
|
38
|
-
import { withProvisionLock } from "./runtime/provision-mutex.js";
|
|
39
|
-
import { parseConflictMeta, resolveConflict, MAX_CONFLICT_ATTEMPTS, } from "./runtime/conflict-resolver.js";
|
|
40
|
-
import { parseReworkMeta, renderReworkPrompt, fetchPrHeadRef as defaultFetchPrHeadRef, } from "./runtime/rework.js";
|
|
41
|
-
import { assertResolutionIntegrity, isConflictResolutionTask, summariseFailures, } from "./runtime/resolution-integrity.js";
|
|
42
|
-
const LOG_BATCH_BYTES = 4 * 1024;
|
|
43
|
-
/**
|
|
44
|
-
* v0.6.0 REG-296: build the argv for `claude --print`. Splitting this
|
|
45
|
-
* out keeps the model-passthrough logic unit-testable without spawning
|
|
46
|
-
* a real subprocess. modelId is forwarded as `--model <id>` when
|
|
47
|
-
* non-empty (whitespace counts as empty); otherwise omitted so claude
|
|
48
|
-
* uses the operator's default model.
|
|
49
|
-
*/
|
|
50
|
-
export function buildClaudeArgs(modelId) {
|
|
51
|
-
const args = ["--print", "--dangerously-skip-permissions", "--output-format=json"];
|
|
52
|
-
const trimmed = (modelId ?? "").trim();
|
|
53
|
-
if (trimmed)
|
|
54
|
-
args.push("--model", trimmed);
|
|
55
|
-
return args;
|
|
56
|
-
}
|
|
57
|
-
function defaultSpawnClaude(cwd, modelId) {
|
|
58
|
-
// --dangerously-skip-permissions: bypass claude's per-edit approval
|
|
59
|
-
// prompts. The runner is non-interactive (no TTY for human approval)
|
|
60
|
-
// and the permission boundary is enforced one level up — see the
|
|
61
|
-
// "Runner sandbox" section in packages/acc-runner/README.md.
|
|
62
|
-
// --output-format=json: emit a structured object so we can parse the
|
|
63
|
-
// usage block for cost tracking. parseClaudeJson falls back to plain
|
|
64
|
-
// text when older claude versions emit markdown directly.
|
|
65
|
-
// --model: forwarded when the task pins a model (REG-296).
|
|
66
|
-
// v0.74-B: withAnthropicAuth forwards ANTHROPIC_API_KEY (trimmed) when set
|
|
67
|
-
// so claude authenticates via the API account (no per-session cap) instead
|
|
68
|
-
// of the operator's interactive OAuth session. claude has no --api-key flag;
|
|
69
|
-
// the env var is the only auth channel. See src/anthropic-auth.ts.
|
|
70
|
-
return execa("claude", buildClaudeArgs(modelId), {
|
|
71
|
-
cwd,
|
|
72
|
-
stdin: "pipe",
|
|
73
|
-
stdout: "pipe",
|
|
74
|
-
stderr: "pipe",
|
|
75
|
-
reject: false,
|
|
76
|
-
env: withAnthropicAuth(process.env),
|
|
77
|
-
});
|
|
78
|
-
}
|
|
79
|
-
async function defaultCheckoutBase(repoPath, baseBranch) {
|
|
80
|
-
// Plain `git checkout <branch>` (not `-B`) so the integration
|
|
81
|
-
// branch's existing ref is honored. With -B we'd reset the
|
|
82
|
-
// integration branch to current HEAD, which is the exact bug
|
|
83
|
-
// REG-301 was filed for in reverse — destructive instead of stale.
|
|
84
|
-
await execa("git", ["checkout", baseBranch], { cwd: repoPath, env: process.env });
|
|
85
|
-
}
|
|
86
|
-
/**
|
|
87
|
-
* v0.6.0 REG-295: expand `~/...` paths against $HOME so a task can
|
|
88
|
-
* carry `repo_path_hint = "~/work/foo"` from a row populated on a
|
|
89
|
-
* different machine. Absolute and relative paths pass through.
|
|
90
|
-
*/
|
|
91
|
-
export function expandHomePath(p) {
|
|
92
|
-
if (p === "~")
|
|
93
|
-
return os.homedir();
|
|
94
|
-
if (p.startsWith("~/"))
|
|
95
|
-
return path.join(os.homedir(), p.slice(2));
|
|
96
|
-
return p;
|
|
97
|
-
}
|
|
98
|
-
/**
|
|
99
|
-
* v0.63 T-63-1: default claude health probe. A silent instant-empty exit is
|
|
100
|
-
* ambiguous — it can be a broken host OR a momentarily-unavailable claude.
|
|
101
|
-
* `claude --version` is a cheap, deterministic answer to "is the binary/host
|
|
102
|
-
* actually broken?": if it returns 0 the host is fine (the exit was
|
|
103
|
-
* transient → claude_unavailable, retry), and any spawn failure / non-zero
|
|
104
|
-
* exit / timeout means the binary or its dynamic deps are genuinely broken
|
|
105
|
-
* (→ env_broken, quarantine). Short timeout so a probe never wedges the run.
|
|
106
|
-
*/
|
|
107
|
-
async function defaultHealthProbe() {
|
|
108
|
-
try {
|
|
109
|
-
const res = await execa("claude", ["--version"], {
|
|
110
|
-
reject: false,
|
|
111
|
-
timeout: 5_000,
|
|
112
|
-
env: process.env,
|
|
113
|
-
});
|
|
114
|
-
if (res.exitCode === 0) {
|
|
115
|
-
return { ok: true, detail: (res.stdout || "").trim().slice(0, 200) || "claude --version ok" };
|
|
116
|
-
}
|
|
117
|
-
return {
|
|
118
|
-
ok: false,
|
|
119
|
-
detail: `claude --version exited ${res.exitCode ?? "signal"}: ${(res.stderr || "").trim().slice(0, 200)}`,
|
|
120
|
-
};
|
|
121
|
-
}
|
|
122
|
-
catch (err) {
|
|
123
|
-
return { ok: false, detail: `claude --version spawn failed: ${err.message}` };
|
|
124
|
-
}
|
|
125
|
-
}
|
|
126
|
-
const DEFAULT_CLAUDE_UNAVAILABLE_KNOBS = {
|
|
127
|
-
windowMs: 15 * 60_000,
|
|
128
|
-
alertThreshold: 5,
|
|
129
|
-
backoffBaseMs: 30_000,
|
|
130
|
-
backoffMaxMs: 15 * 60_000,
|
|
131
|
-
};
|
|
132
|
-
// Process-level state. runTask() is a fresh closure per task, but the module
|
|
133
|
-
// is loaded once, so consecutive task runs in the same `acc-runner watch`
|
|
134
|
-
// process share this — the only place a cross-task streak can live without
|
|
135
|
-
// touching watch.ts (out of this track's file scope). A clean (exit-0) claude
|
|
136
|
-
// run resets it: claude has recovered.
|
|
137
|
-
let claudeUnavailableKnobs = { ...DEFAULT_CLAUDE_UNAVAILABLE_KNOBS };
|
|
138
|
-
let claudeUnavailableHits = [];
|
|
139
|
-
let claudeUnavailableStreak = 0;
|
|
140
|
-
/** v0.63 T-63-1: override the silent-unavailability knobs (used by tests). */
|
|
141
|
-
export function configureClaudeUnavailable(partial) {
|
|
142
|
-
claudeUnavailableKnobs = { ...claudeUnavailableKnobs, ...partial };
|
|
143
|
-
}
|
|
144
|
-
/** v0.63 T-63-1: reset both the streak and the knobs (used by tests). */
|
|
145
|
-
export function resetClaudeUnavailableState() {
|
|
146
|
-
claudeUnavailableHits = [];
|
|
147
|
-
claudeUnavailableStreak = 0;
|
|
148
|
-
claudeUnavailableKnobs = { ...DEFAULT_CLAUDE_UNAVAILABLE_KNOBS };
|
|
149
|
-
}
|
|
150
|
-
/**
|
|
151
|
-
* v0.63 T-63-1: record one silent instant-empty exit and compute the response
|
|
152
|
-
* (exponential backoff + whether the sustained-streak INFRA alert trips).
|
|
153
|
-
* Exported so the watch process and the unit suite share one definition.
|
|
154
|
-
*/
|
|
155
|
-
export function recordClaudeUnavailable(now) {
|
|
156
|
-
claudeUnavailableStreak += 1;
|
|
157
|
-
claudeUnavailableHits.push(now);
|
|
158
|
-
const cutoff = now - claudeUnavailableKnobs.windowMs;
|
|
159
|
-
claudeUnavailableHits = claudeUnavailableHits.filter((t) => t >= cutoff);
|
|
160
|
-
const windowCount = claudeUnavailableHits.length;
|
|
161
|
-
const alert = windowCount >= claudeUnavailableKnobs.alertThreshold;
|
|
162
|
-
const expo = claudeUnavailableKnobs.backoffBaseMs * 2 ** (claudeUnavailableStreak - 1);
|
|
163
|
-
const backoffMs = Math.min(expo, claudeUnavailableKnobs.backoffMaxMs);
|
|
164
|
-
return {
|
|
165
|
-
streak: claudeUnavailableStreak,
|
|
166
|
-
windowCount,
|
|
167
|
-
alert,
|
|
168
|
-
backoffMs,
|
|
169
|
-
resumeAt: new Date(now + backoffMs).toISOString(),
|
|
170
|
-
};
|
|
171
|
-
}
|
|
172
|
-
/** v0.63 T-63-1: a clean claude run proves the binary recovered — reset. */
|
|
173
|
-
function clearClaudeUnavailableStreak() {
|
|
174
|
-
claudeUnavailableStreak = 0;
|
|
175
|
-
claudeUnavailableHits = [];
|
|
176
|
-
}
|
|
177
|
-
// v0.37-A HEAD-LOCK-GUARD: git worktree add/remove/prune can leave stale
|
|
178
|
-
// .git/HEAD.lock or .git/index.lock behind. The virtiofs boundary keeps
|
|
179
|
-
// the sandbox from clearing them, but the host runner process can. Call
|
|
180
|
-
// this against the parent repo's path (the dir containing .git/) before
|
|
181
|
-
// any worktree op.
|
|
182
|
-
function clearGitLocks(repoRoot) {
|
|
183
|
-
for (const name of ["HEAD.lock", "index.lock"]) {
|
|
184
|
-
const p = join(repoRoot, ".git", name);
|
|
185
|
-
try {
|
|
186
|
-
if (existsSync(p)) {
|
|
187
|
-
unlinkSync(p);
|
|
188
|
-
process.stderr.write("[runner] cleared stale lock: " + p + "\n");
|
|
189
|
-
}
|
|
190
|
-
}
|
|
191
|
-
catch { /* noop */ }
|
|
192
|
-
}
|
|
193
|
-
}
|
|
194
|
-
async function appendEvent(supabase, taskId, kind, payload) {
|
|
195
|
-
const { error } = await supabase.rpc("append_task_event", {
|
|
196
|
-
p_task_id: taskId,
|
|
197
|
-
p_kind: kind,
|
|
198
|
-
p_payload: payload,
|
|
199
|
-
});
|
|
200
|
-
if (error) {
|
|
201
|
-
// Logging failures shouldn't crash the run — print to local stderr.
|
|
202
|
-
process.stderr.write(`[acc-runner] append_task_event failed: ${error.message}\n`);
|
|
203
|
-
}
|
|
204
|
-
}
|
|
205
|
-
async function streamToEvents(stream, supabase, taskId, streamName) {
|
|
206
|
-
let captured = "";
|
|
207
|
-
let pending = "";
|
|
208
|
-
const flush = async () => {
|
|
209
|
-
if (!pending)
|
|
210
|
-
return;
|
|
211
|
-
const text = pending;
|
|
212
|
-
pending = "";
|
|
213
|
-
captured += text;
|
|
214
|
-
await appendEvent(supabase, taskId, "log", { stream: streamName, text });
|
|
215
|
-
};
|
|
216
|
-
for await (const raw of stream) {
|
|
217
|
-
const text = typeof raw === "string" ? raw : raw.toString("utf8");
|
|
218
|
-
pending += text;
|
|
219
|
-
if (pending.length >= LOG_BATCH_BYTES)
|
|
220
|
-
await flush();
|
|
221
|
-
}
|
|
222
|
-
await flush();
|
|
223
|
-
return captured;
|
|
224
|
-
}
|
|
225
|
-
/**
|
|
226
|
-
* Pull the structured report block from claude's stdout. Recognises
|
|
227
|
-
* both `--output-format=json` (extracts .result) and plain text. Falls
|
|
228
|
-
* back to the entire stdout (capped) if no `## Summary` heading is
|
|
229
|
-
* present — that way the PR body still tells the reviewer what happened.
|
|
230
|
-
*/
|
|
231
|
-
export function extractReportFromOutput(stdout) {
|
|
232
|
-
const parsed = parseClaudeJson(stdout);
|
|
233
|
-
const text = parsed?.result ?? stdout;
|
|
234
|
-
const idx = text.indexOf("## Summary");
|
|
235
|
-
if (idx === -1) {
|
|
236
|
-
const trimmed = text.trim();
|
|
237
|
-
if (!trimmed)
|
|
238
|
-
return "_(no report emitted)_";
|
|
239
|
-
return trimmed.slice(-8000);
|
|
240
|
-
}
|
|
241
|
-
return text.slice(idx).trim().slice(0, 16000);
|
|
242
|
-
}
|
|
243
|
-
/**
|
|
244
|
-
* v0.35-B: extract the PR number from a `gh pr create` URL. Returns null
|
|
245
|
-
* when the URL doesn't carry a `/pull/<int>` segment so the caller can
|
|
246
|
-
* fall back to the webhook-driven path. GitHub URLs encode PR ids as the
|
|
247
|
-
* last path segment; anchors and query strings are tolerated.
|
|
248
|
-
*/
|
|
249
|
-
export function parsePrNumber(url) {
|
|
250
|
-
if (!url)
|
|
251
|
-
return null;
|
|
252
|
-
const m = url.match(/\/pull\/(\d+)(?:[/?#]|$)/);
|
|
253
|
-
if (!m)
|
|
254
|
-
return null;
|
|
255
|
-
const n = parseInt(m[1], 10);
|
|
256
|
-
return Number.isFinite(n) && n > 0 ? n : null;
|
|
257
|
-
}
|
|
258
|
-
/**
|
|
259
|
-
* v0.35-B: write `pr_number` onto acc.tasks AND transition running →
|
|
260
|
-
* needs-review in one RPC. Best-effort: a missing function (Track A
|
|
261
|
-
* migration not yet applied) OR any error logs to stderr and returns
|
|
262
|
-
* false so the caller falls back to the webhook-driven done path
|
|
263
|
-
* (running → done via the matrix widened in 0156). The RPC contract is
|
|
264
|
-
* `acc.set_task_pr(p_task_id text, p_pr_number int, p_pr_html_url text)`.
|
|
265
|
-
*/
|
|
266
|
-
async function setTaskPr(supabase, taskId, prUrl) {
|
|
267
|
-
const prNumber = parsePrNumber(prUrl);
|
|
268
|
-
if (prNumber === null)
|
|
269
|
-
return false;
|
|
270
|
-
const { error } = await supabase.rpc("set_task_pr", {
|
|
271
|
-
p_task_id: taskId,
|
|
272
|
-
p_pr_number: prNumber,
|
|
273
|
-
p_pr_html_url: prUrl,
|
|
274
|
-
});
|
|
275
|
-
if (error) {
|
|
276
|
-
process.stderr.write(`[acc-runner] set_task_pr(${taskId}, ${prNumber}) failed: ${error.message} ` +
|
|
277
|
-
`— falling back to webhook-driven running→done path\n`);
|
|
278
|
-
return false;
|
|
279
|
-
}
|
|
280
|
-
return true;
|
|
281
|
-
}
|
|
282
|
-
/**
|
|
283
|
-
* v0.53 T-53-4: emit a `task.first_commit` activity event the first time a
|
|
284
|
-
* task's worktree produces a commit on its branch (the SLO funnel signal
|
|
285
|
-
* for "coding actually started"). Additive — the runner protocol message
|
|
286
|
-
* shapes are unchanged; this rides the existing `log_activity` RPC path.
|
|
287
|
-
*
|
|
288
|
-
* Best-effort and exactly-once per task run: it sits on the single
|
|
289
|
-
* post-success, pre-push code path. A repo with no firstCommit support
|
|
290
|
-
* (a GitRunner stub without the method) or a branch with no commits ahead
|
|
291
|
-
* of the base simply emits nothing. Failures log to stderr and never block
|
|
292
|
-
* the push.
|
|
293
|
-
*/
|
|
294
|
-
async function emitFirstCommit(supabase, git, taskId, workdir, baseRef, branch) {
|
|
295
|
-
try {
|
|
296
|
-
const first = await git.firstCommit?.(workdir, baseRef, branch);
|
|
297
|
-
if (!first)
|
|
298
|
-
return;
|
|
299
|
-
const { error } = await supabase.rpc("log_activity", {
|
|
300
|
-
p_verb: "task.first_commit",
|
|
301
|
-
p_target_id: taskId,
|
|
302
|
-
p_target_type: "task",
|
|
303
|
-
p_payload: {
|
|
304
|
-
task_id: taskId,
|
|
305
|
-
branch,
|
|
306
|
-
sha: first.sha,
|
|
307
|
-
ts: first.ts || null,
|
|
308
|
-
},
|
|
309
|
-
});
|
|
310
|
-
if (error) {
|
|
311
|
-
process.stderr.write(`[acc-runner] task.first_commit log_activity(${taskId}) failed: ${error.message}\n`);
|
|
312
|
-
}
|
|
313
|
-
}
|
|
314
|
-
catch (err) {
|
|
315
|
-
process.stderr.write(`[acc-runner] task.first_commit emit failed: ${err.message}\n`);
|
|
316
|
-
}
|
|
317
|
-
}
|
|
318
|
-
/**
|
|
319
|
-
* Build a cost event from claude's stdout. Always returns a payload
|
|
320
|
-
* (zero-cost when usage isn't parseable) so the cap-tracking surface
|
|
321
|
-
* stays consistent: one cost_events row per attempt.
|
|
322
|
-
*/
|
|
323
|
-
export function buildCostEvent(taskId, stdout, fallbackModel, runnerId) {
|
|
324
|
-
const parsed = parseClaudeJson(stdout);
|
|
325
|
-
const model = parsed?.model ?? fallbackModel ?? "unknown";
|
|
326
|
-
const usage = normalizeUsage(parsed?.usage);
|
|
327
|
-
return {
|
|
328
|
-
task_id: taskId,
|
|
329
|
-
model,
|
|
330
|
-
usd_cents: priceUsdCents(model, usage),
|
|
331
|
-
runner_id: runnerId ?? null,
|
|
332
|
-
...usage,
|
|
333
|
-
};
|
|
334
|
-
}
|
|
335
|
-
/**
|
|
336
|
-
* v0.14-MESSAGING-RUNTIME-WIRE — dynamic-import the prelude module
|
|
337
|
-
* referenced by a runner profile and return its default-exported string.
|
|
338
|
-
*
|
|
339
|
-
* The prelude path in the profile JSON (e.g. `./src/profiles/developer-
|
|
340
|
-
* prompt.ts`) is interpreted from the runner package root — the parent
|
|
341
|
-
* of the runner-profiles dir where the JSON lives. In source/dev the
|
|
342
|
-
* .ts file is importable via tsx; in the npm-published artifact only
|
|
343
|
-
* `dist/` ships, so we map `src/*.ts` → `dist/*.js` when the .ts target
|
|
344
|
-
* doesn't exist. Returns empty string when no prelude is configured or
|
|
345
|
-
* the import resolves to a non-string default (v0.13 byte-identical).
|
|
346
|
-
*
|
|
347
|
-
* Exported so tests can stub the resolution + import seam via the
|
|
348
|
-
* `loadPromptPrelude` dep on RunTaskDeps — vitest's Vite-backed
|
|
349
|
-
* transform pipeline does not handle dynamic imports of arbitrary file
|
|
350
|
-
* URLs in the jsdom environment, so the production path is exercised
|
|
351
|
-
* by the runner's own unit suite (node env) rather than by the
|
|
352
|
-
* top-level integration test.
|
|
353
|
-
*/
|
|
354
|
-
export async function loadPromptPrelude(profile) {
|
|
355
|
-
if (!profile.promptPrelude || profile.promptPrelude.length === 0)
|
|
356
|
-
return "";
|
|
357
|
-
// Runner-profiles dir lives at <pkg-root>/runner-profiles/. Resolve
|
|
358
|
-
// the prelude path from the package root so a JSON like
|
|
359
|
-
// `./src/profiles/developer-prompt.ts` lands on a real file.
|
|
360
|
-
const profilesRoot = process.env.ACC_RUNNER_PROFILES_DIR?.trim() ||
|
|
361
|
-
path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "runner-profiles");
|
|
362
|
-
const pkgRoot = path.resolve(profilesRoot, "..");
|
|
363
|
-
let resolved = path.resolve(pkgRoot, profile.promptPrelude);
|
|
364
|
-
// Production map: the published package ships dist/ only.
|
|
365
|
-
if (!fs.existsSync(resolved)) {
|
|
366
|
-
const distMapped = resolved
|
|
367
|
-
.replace(/([/\\])src\1/, "$1dist$1")
|
|
368
|
-
.replace(/\.ts$/, ".js");
|
|
369
|
-
if (fs.existsSync(distMapped))
|
|
370
|
-
resolved = distMapped;
|
|
371
|
-
}
|
|
372
|
-
const mod = (await import(/* @vite-ignore */ pathToFileURL(resolved).href));
|
|
373
|
-
const def = mod.default;
|
|
374
|
-
return typeof def === "string" ? def : "";
|
|
375
|
-
}
|
|
376
|
-
async function defaultPostCostEvent(supabase, event) {
|
|
377
|
-
const { error } = await supabase.rpc("record_cost_event", {
|
|
378
|
-
p_task_id: event.task_id,
|
|
379
|
-
p_model: event.model,
|
|
380
|
-
p_input_tokens: event.input_tokens,
|
|
381
|
-
p_output_tokens: event.output_tokens,
|
|
382
|
-
p_cache_read_tokens: event.cache_read_tokens,
|
|
383
|
-
p_cache_write_tokens: event.cache_write_tokens,
|
|
384
|
-
p_usd_cents: event.usd_cents,
|
|
385
|
-
p_runner_id: event.runner_id ?? null,
|
|
386
|
-
});
|
|
387
|
-
if (error)
|
|
388
|
-
throw new Error(error.message);
|
|
389
|
-
}
|
|
390
|
-
/**
|
|
391
|
-
* v0.48 T-48-5: conflict_resolution task path.
|
|
392
|
-
*
|
|
393
|
-
* Called when the fetched task has type='conflict_resolution'. Does not
|
|
394
|
-
* spawn Claude — resolves conflicts via GitHub API and transitions directly.
|
|
395
|
-
* Runs inside the existing claim/lock/signal infrastructure of runTask.
|
|
396
|
-
*/
|
|
397
|
-
async function runConflictResolutionPath(taskId, task, deps, postState_, appendEventFn) {
|
|
398
|
-
await postState_("planning", "info", {
|
|
399
|
-
runner_id: deps.session.runnerId,
|
|
400
|
-
task_type: "conflict_resolution",
|
|
401
|
-
});
|
|
402
|
-
const meta = parseConflictMeta(task.description, {
|
|
403
|
-
pr_number: task.pr_number,
|
|
404
|
-
repo: task.repo,
|
|
405
|
-
integration_branch: task.integration_branch,
|
|
406
|
-
});
|
|
407
|
-
if (!meta || !meta.repo) {
|
|
408
|
-
const msg = "conflict_resolution task missing PR metadata (pr_number + repo in description or task fields)";
|
|
409
|
-
await appendEventFn("error", { phase: "conflict_resolution", error: msg });
|
|
410
|
-
await deps.supabase.rpc("transition_task", {
|
|
411
|
-
p_task_id: taskId,
|
|
412
|
-
p_new_status: "failed",
|
|
413
|
-
});
|
|
414
|
-
return { taskId, status: "failed", phase: "fetch", error: msg };
|
|
415
|
-
}
|
|
416
|
-
await postState_("coding", "info", {
|
|
417
|
-
runner_id: deps.session.runnerId,
|
|
418
|
-
pr_number: meta.pr_number,
|
|
419
|
-
repo: meta.repo,
|
|
420
|
-
});
|
|
421
|
-
const conflictDeps = {
|
|
422
|
-
supabase: deps.supabase,
|
|
423
|
-
...deps.conflictResolutionDeps,
|
|
424
|
-
};
|
|
425
|
-
const outcome = await resolveConflict(meta, conflictDeps);
|
|
426
|
-
await appendEventFn("conflict_resolution_attempt", {
|
|
427
|
-
action: outcome.action,
|
|
428
|
-
reason: outcome.reason ?? null,
|
|
429
|
-
pr_number: meta.pr_number,
|
|
430
|
-
repo: meta.repo,
|
|
431
|
-
files_resolved: outcome.files_resolved ?? [],
|
|
432
|
-
files_skipped: outcome.files_skipped ?? [],
|
|
433
|
-
max_attempts: MAX_CONFLICT_ATTEMPTS,
|
|
434
|
-
});
|
|
435
|
-
if (outcome.action === "resolved") {
|
|
436
|
-
await postState_("done", "info", {
|
|
437
|
-
runner_id: deps.session.runnerId,
|
|
438
|
-
pr_number: meta.pr_number,
|
|
439
|
-
resolution: "resolved",
|
|
440
|
-
});
|
|
441
|
-
await deps.supabase.rpc("transition_task", {
|
|
442
|
-
p_task_id: taskId,
|
|
443
|
-
p_new_status: "done",
|
|
444
|
-
});
|
|
445
|
-
// Best-effort activity log
|
|
446
|
-
try {
|
|
447
|
-
await deps.supabase.rpc("log_activity", {
|
|
448
|
-
p_verb: "conflict.resolved",
|
|
449
|
-
p_target_id: String(meta.pr_number),
|
|
450
|
-
p_payload: {
|
|
451
|
-
task_id: taskId,
|
|
452
|
-
repo: meta.repo,
|
|
453
|
-
files: outcome.files_resolved ?? [],
|
|
454
|
-
reason: outcome.reason,
|
|
455
|
-
},
|
|
456
|
-
p_target_type: "pr",
|
|
457
|
-
});
|
|
458
|
-
}
|
|
459
|
-
catch { /* best-effort */ }
|
|
460
|
-
return { taskId, status: "ok", exitCode: 0 };
|
|
461
|
-
}
|
|
462
|
-
if (outcome.action === "policy_disabled") {
|
|
463
|
-
// detect-and-skip: transition to done, no escalation
|
|
464
|
-
await postState_("done", "info", {
|
|
465
|
-
runner_id: deps.session.runnerId,
|
|
466
|
-
skipped_reason: "conflict_policy.enabled=false",
|
|
467
|
-
});
|
|
468
|
-
await deps.supabase.rpc("transition_task", {
|
|
469
|
-
p_task_id: taskId,
|
|
470
|
-
p_new_status: "done",
|
|
471
|
-
});
|
|
472
|
-
return { taskId, status: "ok", exitCode: 0 };
|
|
473
|
-
}
|
|
474
|
-
if (outcome.action === "escalated") {
|
|
475
|
-
// max attempts reached — escalate to operator
|
|
476
|
-
const escalationMsg = outcome.reason ?? "conflict_unresolvable after max attempts";
|
|
477
|
-
await postState_("blocked", "error_context", {
|
|
478
|
-
task_id: taskId,
|
|
479
|
-
phase: "conflict_resolution",
|
|
480
|
-
error_class: "conflict_unresolvable",
|
|
481
|
-
pr_number: meta.pr_number,
|
|
482
|
-
repo: meta.repo,
|
|
483
|
-
stderr_tail: escalationMsg,
|
|
484
|
-
});
|
|
485
|
-
try {
|
|
486
|
-
await deps.supabase.rpc("log_activity", {
|
|
487
|
-
p_verb: "conflict.escalated",
|
|
488
|
-
p_target_id: String(meta.pr_number),
|
|
489
|
-
p_payload: {
|
|
490
|
-
task_id: taskId,
|
|
491
|
-
repo: meta.repo,
|
|
492
|
-
reason: escalationMsg,
|
|
493
|
-
prior_attempts: MAX_CONFLICT_ATTEMPTS,
|
|
494
|
-
},
|
|
495
|
-
p_target_type: "pr",
|
|
496
|
-
});
|
|
497
|
-
}
|
|
498
|
-
catch { /* best-effort */ }
|
|
499
|
-
// Transition to done (not failed) so Sweep C doesn't create another task.
|
|
500
|
-
// The escalation signal is on the message bus + activity log.
|
|
501
|
-
await deps.supabase.rpc("transition_task", {
|
|
502
|
-
p_task_id: taskId,
|
|
503
|
-
p_new_status: "done",
|
|
504
|
-
});
|
|
505
|
-
return { taskId, status: "ok", exitCode: 0 };
|
|
506
|
-
}
|
|
507
|
-
// outcome.action === "unresolvable" — transition to failed so Sweep C can retry.
|
|
508
|
-
const failMsg = outcome.reason ?? "conflict not automatically resolvable";
|
|
509
|
-
await postState_("blocked", "error_context", {
|
|
510
|
-
task_id: taskId,
|
|
511
|
-
phase: "conflict_resolution",
|
|
512
|
-
error_class: "conflict_unresolvable_single",
|
|
513
|
-
pr_number: meta.pr_number,
|
|
514
|
-
stderr_tail: failMsg,
|
|
515
|
-
});
|
|
516
|
-
await deps.supabase.rpc("transition_task", {
|
|
517
|
-
p_task_id: taskId,
|
|
518
|
-
p_new_status: "failed",
|
|
519
|
-
});
|
|
520
|
-
return {
|
|
521
|
-
taskId,
|
|
522
|
-
status: "failed",
|
|
523
|
-
phase: "claude_exit",
|
|
524
|
-
error: failMsg,
|
|
525
|
-
};
|
|
526
|
-
}
|
|
527
|
-
export function runTask(taskId, deps) {
|
|
528
|
-
const git = deps.git ?? defaultGit;
|
|
529
|
-
const gh = deps.gh ?? defaultGh;
|
|
530
|
-
const spawnClaude = deps.spawnClaude ?? defaultSpawnClaude;
|
|
531
|
-
const checkoutBase = deps.checkoutBase ?? defaultCheckoutBase;
|
|
532
|
-
const acquireLock = deps.acquireLock ?? defaultAcquireTaskLock;
|
|
533
|
-
const prepareWorktree = deps.prepareWorktree ?? defaultPrepareTaskWorktree;
|
|
534
|
-
const postCostEvent = deps.postCostEvent ?? ((event) => defaultPostCostEvent(deps.supabase, event));
|
|
535
|
-
const postState = deps.postRunnerStateMessage ??
|
|
536
|
-
((args) => defaultPostRunnerStateMessage(deps.supabase, args));
|
|
537
|
-
const loadProfile = deps.loadProfile ?? defaultLoadProfile;
|
|
538
|
-
const loadPromptPreludeFn = deps.loadPromptPrelude ?? loadPromptPrelude;
|
|
539
|
-
const healthProbe = deps.healthProbe ?? defaultHealthProbe;
|
|
540
|
-
// Silence the unused-binding lint for `checkoutBase` — v0.11-F supersedes
|
|
541
|
-
// the v0.6.0 REG-301 pre-spawn `git checkout <baseBranch>` with the
|
|
542
|
-
// worktree's `add -B <branch> <path> <baseBranch>` start-point semantics,
|
|
543
|
-
// but the dep is still accepted for back-compat with test fixtures that
|
|
544
|
-
// inject a mock. Drop in v0.7.
|
|
545
|
-
void checkoutBase;
|
|
546
|
-
let child = null;
|
|
547
|
-
let cancelled = false;
|
|
548
|
-
// v0.14-MESSAGING-RUNTIME-WIRE: convenience wrapper that scopes every
|
|
549
|
-
// bus message to this task + this runner. Best-effort: an underlying
|
|
550
|
-
// RPC failure is logged inside defaultPostRunnerStateMessage and never
|
|
551
|
-
// bubbles out, so this wrapper does not need its own try/catch.
|
|
552
|
-
const postState_ = async (state, protocol, payload) => {
|
|
553
|
-
await postState({
|
|
554
|
-
task_id: taskId,
|
|
555
|
-
sender_id: deps.session.runnerId,
|
|
556
|
-
state,
|
|
557
|
-
protocol,
|
|
558
|
-
payload,
|
|
559
|
-
});
|
|
560
|
-
};
|
|
561
|
-
// v0.6.1 (v0.11-D): release lock rows on every terminal exit path
|
|
562
|
-
// below the claim. Best-effort — a release failure logs to stderr
|
|
563
|
-
// but does not change the outcome the runner reports to the caller.
|
|
564
|
-
// Calling release without first calling claim is a no-op at the
|
|
565
|
-
// RPC level, so wiring this into both pre-claim and post-claim
|
|
566
|
-
// returns would be safe; we only call it from post-claim returns
|
|
567
|
-
// to keep the code path obvious to a reader.
|
|
568
|
-
const releaseLocks = async () => {
|
|
569
|
-
const { error } = await deps.supabase.rpc("release_task_locks", {
|
|
570
|
-
p_task_id: taskId,
|
|
571
|
-
});
|
|
572
|
-
if (error) {
|
|
573
|
-
process.stderr.write(`[acc-runner] release_task_locks(${taskId}) failed: ${error.message}\n`);
|
|
574
|
-
}
|
|
575
|
-
};
|
|
576
|
-
// v0.12-RESUME — periodic signal loop. After the claim succeeds we
|
|
577
|
-
// bump acc.tasks.last_runner_signal_at every signalIntervalMs ms so
|
|
578
|
-
// the v0.12 /5m sweep distinguishes "runner alive, work in flight"
|
|
579
|
-
// from "runner crashed, task stuck at running". Self-rescheduling
|
|
580
|
-
// setTimeout (not setInterval) so a slow RPC doesn't queue up
|
|
581
|
-
// overlapping firings; the loop stops as soon as `signalStopped`
|
|
582
|
-
// flips in the outer try/finally below.
|
|
583
|
-
const DEFAULT_SIGNAL_MS = 30_000;
|
|
584
|
-
const signalIntervalMs = deps.signalIntervalMs ?? DEFAULT_SIGNAL_MS;
|
|
585
|
-
let signalTimer = null;
|
|
586
|
-
let signalStopped = false;
|
|
587
|
-
// v0.68 (T-68-1): bound the signal RPC. supabase-js fetch has no default
|
|
588
|
-
// timeout, so a single hung connection would freeze the self-rescheduling
|
|
589
|
-
// chain below — last_runner_signal_at stops advancing while the runner
|
|
590
|
-
// heartbeat keeps going, and the stale-running sweep reclaims a live task
|
|
591
|
-
// mid-run (the signal-stale zombie symptom). Racing a timeout guarantees
|
|
592
|
-
// the chain always advances to the next tick even when one call stalls.
|
|
593
|
-
const SIGNAL_RPC_TIMEOUT_MS = 10_000;
|
|
594
|
-
const updateSignal = async () => {
|
|
595
|
-
let timer;
|
|
596
|
-
try {
|
|
597
|
-
const timeout = new Promise((_, reject) => {
|
|
598
|
-
timer = setTimeout(() => reject(new Error(`update_task_signal timed out after ${SIGNAL_RPC_TIMEOUT_MS}ms`)), SIGNAL_RPC_TIMEOUT_MS);
|
|
599
|
-
if (timer.unref)
|
|
600
|
-
timer.unref();
|
|
601
|
-
});
|
|
602
|
-
const result = (await Promise.race([
|
|
603
|
-
deps.supabase.rpc("update_task_signal", { p_task_id: taskId }),
|
|
604
|
-
timeout,
|
|
605
|
-
]));
|
|
606
|
-
const error = result?.error ?? null;
|
|
607
|
-
if (error) {
|
|
608
|
-
// Best-effort: a missed signal just means the sweep might pull
|
|
609
|
-
// the task back if enough of them stack up. Log and continue.
|
|
610
|
-
process.stderr.write(`[acc-runner] update_task_signal(${taskId}) failed: ${error.message}\n`);
|
|
611
|
-
}
|
|
612
|
-
}
|
|
613
|
-
catch (err) {
|
|
614
|
-
process.stderr.write(`[acc-runner] update_task_signal(${taskId}) ${err.message}\n`);
|
|
615
|
-
}
|
|
616
|
-
finally {
|
|
617
|
-
if (timer)
|
|
618
|
-
clearTimeout(timer);
|
|
619
|
-
}
|
|
620
|
-
};
|
|
621
|
-
const scheduleSignal = () => {
|
|
622
|
-
if (signalStopped)
|
|
623
|
-
return;
|
|
624
|
-
signalTimer = setTimeout(async () => {
|
|
625
|
-
if (signalStopped)
|
|
626
|
-
return;
|
|
627
|
-
try {
|
|
628
|
-
await updateSignal();
|
|
629
|
-
}
|
|
630
|
-
catch { /* logged inside */ }
|
|
631
|
-
scheduleSignal();
|
|
632
|
-
}, signalIntervalMs);
|
|
633
|
-
// Detach so an in-flight signal timer doesn't keep the process
|
|
634
|
-
// alive past `watch.ts` shutdown. The outer finally clears it
|
|
635
|
-
// anyway; this is belt-and-suspenders for stray timers.
|
|
636
|
-
if (signalTimer.unref)
|
|
637
|
-
signalTimer.unref();
|
|
638
|
-
};
|
|
639
|
-
const stopSignalLoop = () => {
|
|
640
|
-
signalStopped = true;
|
|
641
|
-
if (signalTimer) {
|
|
642
|
-
clearTimeout(signalTimer);
|
|
643
|
-
signalTimer = null;
|
|
644
|
-
}
|
|
645
|
-
};
|
|
646
|
-
const promise = (async () => {
|
|
647
|
-
// 1. Atomic claim + transition to running. v0.11-D: replaces the
|
|
648
|
-
// pre-v0.6.1 raw transition_task('running') call. The RPC
|
|
649
|
-
// row-locks acc.tasks FOR UPDATE, checks file-path overlap
|
|
650
|
-
// against other running tasks' locks, INSERTs a lock row +
|
|
651
|
-
// transitions to running in one transaction. On overlap or
|
|
652
|
-
// same-task race the task stays queued for another runner.
|
|
653
|
-
const claim = await deps.supabase.rpc("claim_task_with_locks", {
|
|
654
|
-
p_task_id: taskId,
|
|
655
|
-
p_runner_id: deps.session.runnerId,
|
|
656
|
-
});
|
|
657
|
-
if (claim.error) {
|
|
658
|
-
const msg = claim.error.message;
|
|
659
|
-
// Includes 22023 (invalid_task_transition surfaced through the
|
|
660
|
-
// nested acc.transition_task call) — task may already be past
|
|
661
|
-
// 'running' (e.g. needs-review). Log and bail rather than crash.
|
|
662
|
-
await appendEvent(deps.supabase, taskId, "error", {
|
|
663
|
-
phase: "claim_locks",
|
|
664
|
-
error: msg,
|
|
665
|
-
});
|
|
666
|
-
return { taskId, status: "failed", phase: "claim_locks", error: msg };
|
|
667
|
-
}
|
|
668
|
-
const claimResult = (claim.data ?? {});
|
|
669
|
-
if (claimResult.ok !== true) {
|
|
670
|
-
const conflicts = Array.isArray(claimResult.conflicts) ? claimResult.conflicts : [];
|
|
671
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
672
|
-
phase: "claim_locks",
|
|
673
|
-
stream: "stderr",
|
|
674
|
-
conflicts,
|
|
675
|
-
message: `file-lock conflict — other tasks hold overlapping paths: ${conflicts.join(", ")}`,
|
|
676
|
-
runner_id: deps.session.runnerId,
|
|
677
|
-
});
|
|
678
|
-
const reason = `file-lock conflict with: ${conflicts.join(", ") || "(unknown)"}`;
|
|
679
|
-
return {
|
|
680
|
-
taskId,
|
|
681
|
-
status: "failed",
|
|
682
|
-
phase: "claim_locks",
|
|
683
|
-
error: reason,
|
|
684
|
-
};
|
|
685
|
-
}
|
|
686
|
-
// v0.6.1 (v0.11-D): every exit path below the successful claim
|
|
687
|
-
// must release the lock row so the same paths free up for the
|
|
688
|
-
// next runner. try/finally captures returns AND uncaught throws
|
|
689
|
-
// alike — strictly stronger than explicit pre-return release at
|
|
690
|
-
// each of the seven completion sites, and the runner CLI never
|
|
691
|
-
// recovers from a thrown error inside runTask, so the lock
|
|
692
|
-
// would otherwise leak until a future sweep job clears it.
|
|
693
|
-
try {
|
|
694
|
-
// v0.53 T-53-4: arm the per-task-execution GitHub soft budget. Any
|
|
695
|
-
// direct REST call routed through github-client.githubFetch during
|
|
696
|
-
// this task is counted against DEFAULT_TASK_GITHUB_BUDGET and fails
|
|
697
|
-
// clean (retryable infra, never quarantine) once spent. NOTE: the
|
|
698
|
-
// runner's current GitHub access is entirely via the `gh` CLI
|
|
699
|
-
// (gh.ts, conflict-resolver, rework, reviewer) — those separate
|
|
700
|
-
// processes are out of scope and remain UNBUDGETED here. Disarmed in
|
|
701
|
-
// the finally so out-of-task callers (doctor) run unbudgeted.
|
|
702
|
-
startTaskGithubBudget();
|
|
703
|
-
// v0.12-RESUME: prime the signal column immediately on claim so
|
|
704
|
-
// the sweep window resets from "now" and the first periodic tick
|
|
705
|
-
// (after signalIntervalMs) refreshes it. Without this prime, a
|
|
706
|
-
// task whose run takes < signalIntervalMs from claim to first
|
|
707
|
-
// tick could race the sweep on borderline updated_at values.
|
|
708
|
-
// Placed inside the outer try so an unexpected throw from the
|
|
709
|
-
// RPC still runs the finally (stop loop, release locks).
|
|
710
|
-
await updateSignal();
|
|
711
|
-
scheduleSignal();
|
|
712
|
-
// v0.14-MESSAGING-RUNTIME-WIRE: first bus event — runner has the
|
|
713
|
-
// claim, will now fetch + spawn. planner subscribes via /messages.
|
|
714
|
-
await postState_("planning", "info", { runner_id: deps.session.runnerId });
|
|
715
|
-
// 2. Fetch task + adjacent rows.
|
|
716
|
-
const fetched = await deps.supabase.rpc("fetch_task_for_runner", {
|
|
717
|
-
p_task_id: taskId,
|
|
718
|
-
});
|
|
719
|
-
if (fetched.error || !fetched.data) {
|
|
720
|
-
const msg = fetched.error?.message ?? "fetch_task_for_runner returned no data";
|
|
721
|
-
await appendEvent(deps.supabase, taskId, "error", { phase: "fetch", error: msg });
|
|
722
|
-
await postState_("blocked", "error_context", {
|
|
723
|
-
task_id: taskId,
|
|
724
|
-
phase: "plan",
|
|
725
|
-
error_class: "fetch_task_failed",
|
|
726
|
-
stderr_tail: msg,
|
|
727
|
-
});
|
|
728
|
-
await deps.supabase.rpc("transition_task", {
|
|
729
|
-
p_task_id: taskId,
|
|
730
|
-
p_new_status: "failed",
|
|
731
|
-
});
|
|
732
|
-
return { taskId, status: "failed", phase: "fetch", error: msg };
|
|
733
|
-
}
|
|
734
|
-
const result = fetched.data;
|
|
735
|
-
const { task } = result;
|
|
736
|
-
// v0.48 T-48-5: conflict_resolution tasks bypass Claude Code entirely.
|
|
737
|
-
// The runner resolves conflicts on the real base∩head file set via the
|
|
738
|
-
// GitHub API, pushes the resolution, and transitions the task to done
|
|
739
|
-
// (or escalates after MAX_CONFLICT_ATTEMPTS failed attempts).
|
|
740
|
-
if (task.type === "conflict_resolution") {
|
|
741
|
-
return await runConflictResolutionPath(taskId, task, deps, postState_, appendEvent.bind(null, deps.supabase, taskId));
|
|
742
|
-
}
|
|
743
|
-
// v0.51 T-51-1: rework tasks run the normal Claude session, but against
|
|
744
|
-
// the PR's EXISTING head branch — the worktree forks from
|
|
745
|
-
// origin/<branch> instead of the integration branch, the prompt carries
|
|
746
|
-
// the reviewer's verdict + comments, and the push goes back to the same
|
|
747
|
-
// branch (same PR; no new PR is opened).
|
|
748
|
-
let reworkMeta = null;
|
|
749
|
-
if (task.type === "rework") {
|
|
750
|
-
reworkMeta = parseReworkMeta(task.description, {
|
|
751
|
-
pr_number: task.pr_number,
|
|
752
|
-
repo: task.repo,
|
|
753
|
-
branch: task.branch,
|
|
754
|
-
});
|
|
755
|
-
if (!reworkMeta) {
|
|
756
|
-
const msg = "rework task missing PR metadata (acc-rework sentinel or pr_number + repo on task row)";
|
|
757
|
-
await appendEvent(deps.supabase, taskId, "error", { phase: "rework", error: msg });
|
|
758
|
-
await postState_("blocked", "error_context", {
|
|
759
|
-
task_id: taskId,
|
|
760
|
-
phase: "plan",
|
|
761
|
-
error_class: "rework_meta_missing",
|
|
762
|
-
stderr_tail: msg,
|
|
763
|
-
});
|
|
764
|
-
await deps.supabase.rpc("transition_task", {
|
|
765
|
-
p_task_id: taskId,
|
|
766
|
-
p_new_status: "failed",
|
|
767
|
-
});
|
|
768
|
-
return { taskId, status: "failed", phase: "fetch", error: msg };
|
|
769
|
-
}
|
|
770
|
-
if (!reworkMeta.branch) {
|
|
771
|
-
try {
|
|
772
|
-
reworkMeta.branch = await (deps.fetchPrHeadRef ?? defaultFetchPrHeadRef)(reworkMeta.repo, reworkMeta.pr_number);
|
|
773
|
-
}
|
|
774
|
-
catch { /* handled by the empty-branch check below */ }
|
|
775
|
-
if (!reworkMeta.branch) {
|
|
776
|
-
const msg = `rework task could not resolve head branch for PR #${reworkMeta.pr_number}`;
|
|
777
|
-
await appendEvent(deps.supabase, taskId, "error", { phase: "rework", error: msg });
|
|
778
|
-
await deps.supabase.rpc("transition_task", {
|
|
779
|
-
p_task_id: taskId,
|
|
780
|
-
p_new_status: "failed",
|
|
781
|
-
});
|
|
782
|
-
return { taskId, status: "failed", phase: "fetch", error: msg };
|
|
783
|
-
}
|
|
784
|
-
}
|
|
785
|
-
}
|
|
786
|
-
const branch = reworkMeta
|
|
787
|
-
? reworkMeta.branch
|
|
788
|
-
: branchForTask(task.id, task.title, task.branch);
|
|
789
|
-
// REG-295: per-task repo overrides — task row trumps config so one
|
|
790
|
-
// runner can service tasks across multiple repos. Env-backed config
|
|
791
|
-
// remains the fallback for tasks that don't carry a hint yet.
|
|
792
|
-
const repoPath = task.repo_path_hint?.trim()
|
|
793
|
-
? expandHomePath(task.repo_path_hint.trim())
|
|
794
|
-
: deps.cfg.repoPath;
|
|
795
|
-
const targetRepo = task.repo?.trim() || deps.cfg.targetRepo;
|
|
796
|
-
// REG-301: branch base comes from task → cfg → main. cfg.integrationBranch
|
|
797
|
-
// already defaults to "acc/integration" so the third fallback only
|
|
798
|
-
// matters when an operator zeroed it out via env.
|
|
799
|
-
const integrationBranch = task.integration_branch?.trim() || deps.cfg.integrationBranch || "main";
|
|
800
|
-
// v0.51 T-51-1: rework prompts replace the "open a PR" protocol header
|
|
801
|
-
// with "update the existing PR's branch" + the reviewer's verdict.
|
|
802
|
-
const renderedPrompt = reworkMeta
|
|
803
|
-
? renderReworkPrompt({ task, meta: reworkMeta, branch })
|
|
804
|
-
: renderTaskPrompt({
|
|
805
|
-
task,
|
|
806
|
-
agent: result.agent,
|
|
807
|
-
model: result.model,
|
|
808
|
-
integrationBranch,
|
|
809
|
-
targetRepo,
|
|
810
|
-
});
|
|
811
|
-
// v0.14-MESSAGING-RUNTIME-WIRE: profile-driven prompt prelude.
|
|
812
|
-
// When loadProfile() returns null OR the profile has no prelude
|
|
813
|
-
// OR the prelude module's default export is the empty string,
|
|
814
|
-
// `prompt` is byte-identical to the v0.13 `renderedPrompt`.
|
|
815
|
-
const activeProfile = loadProfile();
|
|
816
|
-
let prelude = "";
|
|
817
|
-
if (activeProfile && activeProfile.promptPrelude) {
|
|
818
|
-
try {
|
|
819
|
-
prelude = await loadPromptPreludeFn(activeProfile);
|
|
820
|
-
}
|
|
821
|
-
catch (err) {
|
|
822
|
-
// A broken prelude module should not block the task — log and
|
|
823
|
-
// fall through to v0.13 behavior. The operator surfaces this
|
|
824
|
-
// via the per-task History stream so the misconfiguration is
|
|
825
|
-
// discoverable without crashing the run.
|
|
826
|
-
await appendEvent(deps.supabase, taskId, "error", {
|
|
827
|
-
phase: "prelude",
|
|
828
|
-
error: err.message,
|
|
829
|
-
profile: activeProfile.name,
|
|
830
|
-
prelude_path: activeProfile.promptPrelude,
|
|
831
|
-
});
|
|
832
|
-
}
|
|
833
|
-
}
|
|
834
|
-
const prompt = prelude ? `${prelude}\n\n${renderedPrompt}` : renderedPrompt;
|
|
835
|
-
// v0.11-F: acquire the per-task PID lock before any worktree
|
|
836
|
-
// side-effect. Same-machine parallel runners that picked up the same
|
|
837
|
-
// task_id (e.g. two `acc-runner watch` processes seeing the same
|
|
838
|
-
// broadcast) race here; the second one bails on TaskLockHeldError.
|
|
839
|
-
let lock;
|
|
840
|
-
try {
|
|
841
|
-
lock = await acquireLock(taskId);
|
|
842
|
-
}
|
|
843
|
-
catch (err) {
|
|
844
|
-
if (err instanceof TaskLockHeldError) {
|
|
845
|
-
await appendEvent(deps.supabase, taskId, "error", {
|
|
846
|
-
phase: "worktree_lock",
|
|
847
|
-
error: err.message,
|
|
848
|
-
held_by_pid: err.heldByPid,
|
|
849
|
-
});
|
|
850
|
-
await deps.supabase.rpc("transition_task", {
|
|
851
|
-
p_task_id: taskId,
|
|
852
|
-
p_new_status: "failed",
|
|
853
|
-
});
|
|
854
|
-
return { taskId, status: "failed", phase: "worktree_lock", error: err.message };
|
|
855
|
-
}
|
|
856
|
-
throw err;
|
|
857
|
-
}
|
|
858
|
-
// Worktree gets assigned inside the git-prep block; cleanup in the
|
|
859
|
-
// outer finally handles both the success path and every early
|
|
860
|
-
// return that follows.
|
|
861
|
-
let worktree = null;
|
|
862
|
-
let workdir = repoPath;
|
|
863
|
-
// v0.19 T-64-1: the ref the task branch is actually forked from, set
|
|
864
|
-
// once the base is resolved below. Used as the `base..branch` range for
|
|
865
|
-
// the first-commit SLO so the signal stays accurate now that FRESH
|
|
866
|
-
// branches fork from origin/<integration> rather than the local ref.
|
|
867
|
-
let resolvedBaseRef = integrationBranch;
|
|
868
|
-
// v0.57 T-57-2: conflict-resolution integrity guardrail. For tasks whose
|
|
869
|
-
// job is to resolve a merge conflict (the Claude/worktree path that the
|
|
870
|
-
// mechanical resolver hands off to when it bails), re-verify the
|
|
871
|
-
// resolution from the runner's own vantage point before pushing —
|
|
872
|
-
// zero conflict markers, typecheck green, and no silently-dropped /
|
|
873
|
-
// regressed version bump. On failure the resolution is REJECTED (no
|
|
874
|
-
// push, not marked resolved) and the task fails with a diagnostic that
|
|
875
|
-
// names the offending assertion, so a broken merge is reworked rather
|
|
876
|
-
// than shipped. A no-op for every non-conflict task.
|
|
877
|
-
const runResolutionIntegrityGate = async () => {
|
|
878
|
-
if (!isConflictResolutionTask(task))
|
|
879
|
-
return null;
|
|
880
|
-
const assertIntegrity = deps.assertResolutionIntegrity ?? assertResolutionIntegrity;
|
|
881
|
-
let result;
|
|
882
|
-
try {
|
|
883
|
-
result = await assertIntegrity(workdir, {
|
|
884
|
-
baseRef: integrationBranch,
|
|
885
|
-
typecheckCmd: task.typecheck_cmd ?? undefined,
|
|
886
|
-
// The #496 regression lived in the acc-runner package; also guard
|
|
887
|
-
// the repo root. Each pair is skipped silently when its version is
|
|
888
|
-
// unchanged or unreadable, so this never false-fails.
|
|
889
|
-
versionTargets: [
|
|
890
|
-
{ packageJsonPath: "package.json", changelogPath: "CHANGELOG.md" },
|
|
891
|
-
{
|
|
892
|
-
packageJsonPath: "packages/acc-runner/package.json",
|
|
893
|
-
changelogPath: "packages/acc-runner/CHANGELOG.md",
|
|
894
|
-
},
|
|
895
|
-
],
|
|
896
|
-
});
|
|
897
|
-
}
|
|
898
|
-
catch (err) {
|
|
899
|
-
// A guardrail that itself crashed must fail closed: reject, never
|
|
900
|
-
// wave a resolution through because the check errored.
|
|
901
|
-
result = {
|
|
902
|
-
ok: false,
|
|
903
|
-
checked: [],
|
|
904
|
-
failures: [
|
|
905
|
-
{
|
|
906
|
-
assertion: "typecheck",
|
|
907
|
-
detail: `integrity check crashed: ${err.message?.slice(0, 200)}`,
|
|
908
|
-
},
|
|
909
|
-
],
|
|
910
|
-
};
|
|
911
|
-
}
|
|
912
|
-
if (result.ok) {
|
|
913
|
-
await appendEvent(deps.supabase, taskId, "resolution_integrity_passed", {
|
|
914
|
-
pr_number: task.pr_number ?? null,
|
|
915
|
-
checked: result.checked,
|
|
916
|
-
});
|
|
917
|
-
return null;
|
|
918
|
-
}
|
|
919
|
-
const summary = summariseFailures(result.failures);
|
|
920
|
-
const detail = result.failures
|
|
921
|
-
.map((f) => `${f.assertion}: ${f.detail}`)
|
|
922
|
-
.join("\n");
|
|
923
|
-
await appendEvent(deps.supabase, taskId, "resolution_integrity_failed", {
|
|
924
|
-
pr_number: task.pr_number ?? null,
|
|
925
|
-
failed_assertions: result.failures.map((f) => f.assertion),
|
|
926
|
-
files: result.failures.flatMap((f) => f.files ?? []),
|
|
927
|
-
detail: detail.slice(-4000),
|
|
928
|
-
});
|
|
929
|
-
// Best-effort operator-facing activity event naming the failed assertion.
|
|
930
|
-
try {
|
|
931
|
-
await deps.supabase.rpc("log_activity", {
|
|
932
|
-
p_verb: "conflict.integrity_rejected",
|
|
933
|
-
p_target_id: String(task.pr_number ?? taskId),
|
|
934
|
-
p_payload: {
|
|
935
|
-
task_id: taskId,
|
|
936
|
-
repo: task.repo ?? null,
|
|
937
|
-
failed_assertions: result.failures.map((f) => f.assertion),
|
|
938
|
-
},
|
|
939
|
-
p_target_type: task.pr_number ? "pr" : "task",
|
|
940
|
-
});
|
|
941
|
-
}
|
|
942
|
-
catch {
|
|
943
|
-
/* best-effort */
|
|
944
|
-
}
|
|
945
|
-
await postState_("blocked", "error_context", {
|
|
946
|
-
task_id: taskId,
|
|
947
|
-
phase: "review",
|
|
948
|
-
error_class: "resolution_integrity_failed",
|
|
949
|
-
stderr_tail: detail.slice(-4000),
|
|
950
|
-
});
|
|
951
|
-
// Reject: do NOT push, do NOT mark resolved. Fail the task so the
|
|
952
|
-
// resolution is reworked instead of merged broken.
|
|
953
|
-
await deps.supabase.rpc("transition_task", {
|
|
954
|
-
p_task_id: taskId,
|
|
955
|
-
p_new_status: "failed",
|
|
956
|
-
});
|
|
957
|
-
return {
|
|
958
|
-
taskId,
|
|
959
|
-
status: "failed",
|
|
960
|
-
phase: "push",
|
|
961
|
-
error: `conflict-resolution integrity guardrail rejected resolution (${summary})`,
|
|
962
|
-
};
|
|
963
|
-
};
|
|
964
|
-
try {
|
|
965
|
-
// 3. Repo prep. v0.11-F: fetch on the shared clone (worktree
|
|
966
|
-
// shares the object store) then provision an isolated worktree
|
|
967
|
-
// at ~/.cache/acc-runner/work/<task_id>/ forked from the
|
|
968
|
-
// integration branch. The redundant `git.checkout -B <branch>`
|
|
969
|
-
// is a no-op inside the new worktree — kept so the v0.6.0
|
|
970
|
-
// checkout seam stays observable in unit tests.
|
|
971
|
-
try {
|
|
972
|
-
// v0.19 T-64-1: cut a FRESH task branch from the LATEST
|
|
973
|
-
// integration tip, not a stale LOCAL ref. `git.fetch` runs
|
|
974
|
-
// `fetch --prune --all`, which refreshes origin/<integrationBranch>;
|
|
975
|
-
// the worktree below then forks from that remote-tracking ref
|
|
976
|
-
// (mirroring the v0.51 rework path, which forks from
|
|
977
|
-
// origin/<branch>). T-63-1's PR conflicted in package.json /
|
|
978
|
-
// CHANGELOG / task-runner.ts precisely because a 13-commit-stale
|
|
979
|
-
// LOCAL integration ref (base 5f627e7) was used as the branch base.
|
|
980
|
-
//
|
|
981
|
-
// ROBUST FETCH (acceptance): a fetch failure (offline / network)
|
|
982
|
-
// must NOT abort the task. We fall back to the stale local ref,
|
|
983
|
-
// log a warning + a `runner.stale_base_fallback` activity event so
|
|
984
|
-
// the staleness is auditable, and continue. The fix never
|
|
985
|
-
// force-pushes and is idempotent.
|
|
986
|
-
// v0.21 T-66-1: serialize the git plumbing below (fetch + stale-lock
|
|
987
|
-
// clear + worktree add/remove) against the shared clone. Two tasks
|
|
988
|
-
// running concurrently would otherwise collide on git's repo-global
|
|
989
|
-
// locks and the loser would die in this pre-Claude phase (T-65-2).
|
|
990
|
-
// Only this critical section is held; the Claude run + push that
|
|
991
|
-
// follow stay concurrent up to concurrencyLimit. Serializing here
|
|
992
|
-
// also makes clearGitLocks safe — it can no longer delete a lock a
|
|
993
|
-
// concurrent provision is actively holding.
|
|
994
|
-
const provisionCloneKey = deps.cfg.worktreeRepoPath ?? repoPath;
|
|
995
|
-
worktree = await withProvisionLock(provisionCloneKey, async () => {
|
|
996
|
-
let staleBaseFallback = false;
|
|
997
|
-
try {
|
|
998
|
-
await git.fetch(repoPath);
|
|
999
|
-
}
|
|
1000
|
-
catch (fetchErr) {
|
|
1001
|
-
staleBaseFallback = true;
|
|
1002
|
-
const detail = fetchErr.message;
|
|
1003
|
-
process.stderr.write(`[acc-runner] base fetch failed for ${taskId}; cutting ${branch} ` +
|
|
1004
|
-
`from stale local ${integrationBranch}: ${detail}\n`);
|
|
1005
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1006
|
-
phase: "git",
|
|
1007
|
-
stream: "stderr",
|
|
1008
|
-
event: "runner.stale_base_fallback",
|
|
1009
|
-
base: integrationBranch,
|
|
1010
|
-
branch,
|
|
1011
|
-
detail,
|
|
1012
|
-
runner_id: deps.session.runnerId,
|
|
1013
|
-
});
|
|
1014
|
-
// Best-effort audit trail on the runner timeline.
|
|
1015
|
-
try {
|
|
1016
|
-
await deps.supabase.rpc("log_activity", {
|
|
1017
|
-
p_verb: "runner.stale_base_fallback",
|
|
1018
|
-
p_target_id: deps.session.runnerId,
|
|
1019
|
-
p_target_type: "runner",
|
|
1020
|
-
p_payload: {
|
|
1021
|
-
task_id: taskId,
|
|
1022
|
-
base: integrationBranch,
|
|
1023
|
-
branch,
|
|
1024
|
-
detail,
|
|
1025
|
-
},
|
|
1026
|
-
});
|
|
1027
|
-
}
|
|
1028
|
-
catch { /* best-effort */ }
|
|
1029
|
-
}
|
|
1030
|
-
// v0.37-A: clear any stale .git/HEAD.lock or index.lock left over
|
|
1031
|
-
// from a prior crashed run before `git worktree add` touches refs.
|
|
1032
|
-
// v0.21 T-66-1: safe under concurrency now — held inside the
|
|
1033
|
-
// provision lock, so it cannot stomp a peer's active lock.
|
|
1034
|
-
clearGitLocks(repoPath);
|
|
1035
|
-
// v0.51 T-51-1: rework tasks fork from the PR's remote head so
|
|
1036
|
-
// Claude sees the branch's current contents. v0.19 T-64-1: FRESH
|
|
1037
|
-
// tasks fork from the freshly-fetched origin tip
|
|
1038
|
-
// (origin/<integrationBranch>), or — only when the fetch failed —
|
|
1039
|
-
// the stale local ref. The crash-resume path inside
|
|
1040
|
-
// prepareTaskWorktree returns the existing partial branch untouched
|
|
1041
|
-
// regardless of this base (it never resets a resumed branch:
|
|
1042
|
-
// REG-301 / v0.12-RESUME), so a stale base here can never clobber
|
|
1043
|
-
// partially-committed work.
|
|
1044
|
-
const worktreeBaseBranch = reworkMeta
|
|
1045
|
-
? `origin/${branch}`
|
|
1046
|
-
: staleBaseFallback
|
|
1047
|
-
? integrationBranch
|
|
1048
|
-
: `origin/${integrationBranch}`;
|
|
1049
|
-
// Rework keeps its prior first-commit base (integration) so the SLO
|
|
1050
|
-
// measures the whole PR effort; FRESH tasks use their true fork point.
|
|
1051
|
-
resolvedBaseRef = reworkMeta ? integrationBranch : worktreeBaseBranch;
|
|
1052
|
-
return await prepareWorktree({
|
|
1053
|
-
repoPath,
|
|
1054
|
-
// v0.41-B: when an operator has configured a separate mirror
|
|
1055
|
-
// clone for worktree ops, prepareTaskWorktree runs add/remove/
|
|
1056
|
-
// prune against it; `git fetch` above still targets repoPath.
|
|
1057
|
-
// Undefined preserves v0.40 behaviour byte-for-byte.
|
|
1058
|
-
worktreeRepoPath: deps.cfg.worktreeRepoPath,
|
|
1059
|
-
taskId,
|
|
1060
|
-
branch,
|
|
1061
|
-
baseBranch: worktreeBaseBranch,
|
|
1062
|
-
});
|
|
1063
|
-
});
|
|
1064
|
-
workdir = worktree.path;
|
|
1065
|
-
// v0.12-RESUME: log resume-vs-fresh so an operator can audit
|
|
1066
|
-
// how often the resume path actually fires. `resumed: true`
|
|
1067
|
-
// means a prior runner crashed mid-task, the v0.12 sweep
|
|
1068
|
-
// returned the task to queued, and this runner picked it
|
|
1069
|
-
// back up with the prior worktree intact. Claude reads the
|
|
1070
|
-
// partially-committed state and continues; the spawn is the
|
|
1071
|
-
// same prompt either way (Claude is idempotent enough that
|
|
1072
|
-
// re-running on a partially-edited worktree converges on
|
|
1073
|
-
// the right final state).
|
|
1074
|
-
if (worktree.resumed) {
|
|
1075
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1076
|
-
phase: "git",
|
|
1077
|
-
stream: "stdout",
|
|
1078
|
-
event: "worktree.resumed",
|
|
1079
|
-
worktree_path: workdir,
|
|
1080
|
-
branch,
|
|
1081
|
-
runner_id: deps.session.runnerId,
|
|
1082
|
-
});
|
|
1083
|
-
}
|
|
1084
|
-
await git.checkout(workdir, branch);
|
|
1085
|
-
}
|
|
1086
|
-
catch (err) {
|
|
1087
|
-
const msg = err.message;
|
|
1088
|
-
// v0.21 T-66-1: a git REPO-GLOBAL LOCK collision during provisioning
|
|
1089
|
-
// (an external git process touched the clone while we held the
|
|
1090
|
-
// provision mutex) is TRANSIENT — requeue the task so it re-provisions
|
|
1091
|
-
// cleanly once the lock frees, instead of burning a terminal `failed`.
|
|
1092
|
-
// The in-process mutex already removes the runner-vs-runner case; this
|
|
1093
|
-
// covers the residual external-contention case.
|
|
1094
|
-
if (isGitProvisionContention(msg)) {
|
|
1095
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1096
|
-
phase: "git",
|
|
1097
|
-
stream: "stderr",
|
|
1098
|
-
event: "runner.git_provision_contention",
|
|
1099
|
-
error_class: "git_provision_contention",
|
|
1100
|
-
error: msg,
|
|
1101
|
-
branch,
|
|
1102
|
-
runner_id: deps.session.runnerId,
|
|
1103
|
-
});
|
|
1104
|
-
// running → queued: re-dispatch for a clean retry (matches the
|
|
1105
|
-
// stale-running sweep's lossless requeue; not a retry-budget burn).
|
|
1106
|
-
await deps.supabase.rpc("transition_task", {
|
|
1107
|
-
p_task_id: taskId,
|
|
1108
|
-
p_new_status: "queued",
|
|
1109
|
-
});
|
|
1110
|
-
return {
|
|
1111
|
-
taskId,
|
|
1112
|
-
status: "requeued",
|
|
1113
|
-
phase: "git",
|
|
1114
|
-
error: `git_provision_contention: ${msg}`,
|
|
1115
|
-
};
|
|
1116
|
-
}
|
|
1117
|
-
await appendEvent(deps.supabase, taskId, "error", {
|
|
1118
|
-
phase: "git",
|
|
1119
|
-
error: msg,
|
|
1120
|
-
base: integrationBranch,
|
|
1121
|
-
branch,
|
|
1122
|
-
repo_path: repoPath,
|
|
1123
|
-
worktree_path: worktree?.path ?? null,
|
|
1124
|
-
});
|
|
1125
|
-
await postState_("blocked", "error_context", {
|
|
1126
|
-
task_id: taskId,
|
|
1127
|
-
phase: "plan",
|
|
1128
|
-
error_class: "git_prep_failed",
|
|
1129
|
-
stderr_tail: msg,
|
|
1130
|
-
});
|
|
1131
|
-
await deps.supabase.rpc("transition_task", {
|
|
1132
|
-
p_task_id: taskId,
|
|
1133
|
-
p_new_status: "failed",
|
|
1134
|
-
});
|
|
1135
|
-
return { taskId, status: "failed", phase: "git", error: msg };
|
|
1136
|
-
}
|
|
1137
|
-
// 4. Spawn Claude.
|
|
1138
|
-
if (cancelled) {
|
|
1139
|
-
return { taskId, status: "cancelled" };
|
|
1140
|
-
}
|
|
1141
|
-
// v0.32-A: idempotency guard. If the runner was killed mid-task and
|
|
1142
|
-
// the sweep returned this task to 'queued', a prior run may have
|
|
1143
|
-
// already pushed the branch and opened a PR. Re-running Claude Code
|
|
1144
|
-
// would fail at `git push` (non-fast-forward). Check for an existing
|
|
1145
|
-
// open PR before spawning and skip the coding step if one is found.
|
|
1146
|
-
// v0.51 T-51-1: skipped for rework tasks — their PR exists by design;
|
|
1147
|
-
// the whole point is to spawn Claude against it.
|
|
1148
|
-
if (!reworkMeta) {
|
|
1149
|
-
const existingPrUrl = await gh.findOpenPR(workdir, branch);
|
|
1150
|
-
if (existingPrUrl) {
|
|
1151
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1152
|
-
phase: "coding",
|
|
1153
|
-
stream: "stdout",
|
|
1154
|
-
event: "retry.pr_exists",
|
|
1155
|
-
branch,
|
|
1156
|
-
pr_url: existingPrUrl,
|
|
1157
|
-
message: "open PR already exists for branch — skipping Claude Code spawn",
|
|
1158
|
-
});
|
|
1159
|
-
// v0.35-B: stamp pr_number + transition running→needs-review on
|
|
1160
|
-
// the recovered task. On any failure the webhook-driven flow
|
|
1161
|
-
// still moves running→done at merge time (matrix 0156).
|
|
1162
|
-
await setTaskPr(deps.supabase, taskId, existingPrUrl);
|
|
1163
|
-
await postState_("done", "info", {
|
|
1164
|
-
runner_id: deps.session.runnerId,
|
|
1165
|
-
pr_url: existingPrUrl,
|
|
1166
|
-
skipped_reason: "pr_already_open",
|
|
1167
|
-
});
|
|
1168
|
-
return { taskId, status: "ok", prUrl: existingPrUrl, exitCode: 0 };
|
|
1169
|
-
}
|
|
1170
|
-
}
|
|
1171
|
-
// v0.14-MESSAGING-RUNTIME-WIRE: worktree is ready, Claude is
|
|
1172
|
-
// about to start the actual work. coding marks the transition
|
|
1173
|
-
// into the long-running subprocess phase.
|
|
1174
|
-
await postState_("coding", "info", {
|
|
1175
|
-
runner_id: deps.session.runnerId,
|
|
1176
|
-
branch,
|
|
1177
|
-
model: result.model?.id ?? null,
|
|
1178
|
-
});
|
|
1179
|
-
// 4a. Provision .mcp.json so Claude Code auto-discovers acc-mcp-server.
|
|
1180
|
-
// Best-effort: a failed write must not block the task. The MCP server
|
|
1181
|
-
// is a context source, not a critical dependency for v0.5-C1.
|
|
1182
|
-
let mcpCleanup = null;
|
|
1183
|
-
if (deps.session) {
|
|
1184
|
-
const writer = deps.writeMcpConfig ?? defaultWriteMcpConfig;
|
|
1185
|
-
try {
|
|
1186
|
-
mcpCleanup = await writer({
|
|
1187
|
-
cwd: workdir,
|
|
1188
|
-
taskId,
|
|
1189
|
-
runnerId: deps.session.runnerId,
|
|
1190
|
-
accessToken: deps.session.accessToken,
|
|
1191
|
-
publicUrl: deps.publicUrl ?? deps.cfg.publicUrl,
|
|
1192
|
-
supabaseUrl: deps.cfg.supabaseUrl,
|
|
1193
|
-
supabaseAnonKey: deps.cfg.supabaseAnonKey,
|
|
1194
|
-
});
|
|
1195
|
-
}
|
|
1196
|
-
catch (err) {
|
|
1197
|
-
// Don't echo the token even on failure.
|
|
1198
|
-
process.stderr.write(`[acc-runner] mcp .mcp.json write failed: ${err.message}\n`);
|
|
1199
|
-
}
|
|
1200
|
-
}
|
|
1201
|
-
// v0.12-MODEL-ALIAS (REG-303/304): translate the ACC model alias
|
|
1202
|
-
// (`claude-sonnet-4`) into the wire form `claude --model` actually
|
|
1203
|
-
// accepts (`sonnet` or `claude-sonnet-4-6`). Unknown ids pass
|
|
1204
|
-
// through verbatim so a future model not yet in the embedded
|
|
1205
|
-
// table still spawns.
|
|
1206
|
-
//
|
|
1207
|
-
// v0.53 T-53-4: one claude attempt — spawn, stream both pipes to the
|
|
1208
|
-
// History tab, post a cost event (one row per attempt), and time the
|
|
1209
|
-
// run so the classifier's fast-exit env_broken heuristic can fire.
|
|
1210
|
-
const runClaudeOnce = async () => {
|
|
1211
|
-
const spawnedAt = Date.now();
|
|
1212
|
-
child = spawnClaude(workdir, toCliAlias(result.model?.id));
|
|
1213
|
-
if (child.stdin) {
|
|
1214
|
-
child.stdin.write(prompt);
|
|
1215
|
-
child.stdin.end();
|
|
1216
|
-
}
|
|
1217
|
-
const stdoutPromise = child.stdout
|
|
1218
|
-
? streamToEvents(child.stdout, deps.supabase, taskId, "stdout")
|
|
1219
|
-
: Promise.resolve("");
|
|
1220
|
-
const stderrPromise = child.stderr
|
|
1221
|
-
? streamToEvents(child.stderr, deps.supabase, taskId, "stderr")
|
|
1222
|
-
: Promise.resolve("");
|
|
1223
|
-
const [outcome, capturedStdout, capturedStderr] = await Promise.all([
|
|
1224
|
-
child,
|
|
1225
|
-
stdoutPromise,
|
|
1226
|
-
stderrPromise,
|
|
1227
|
-
]);
|
|
1228
|
-
// One cost_events row per attempt so cap tracking captures success,
|
|
1229
|
-
// cancellation, retry, and non-zero exit alike. Best-effort: a
|
|
1230
|
-
// failed POST logs to stderr but never bubbles past the runner.
|
|
1231
|
-
{
|
|
1232
|
-
const event = buildCostEvent(taskId, capturedStdout, result.model?.id, result.runner?.id);
|
|
1233
|
-
try {
|
|
1234
|
-
await postCostEvent(event);
|
|
1235
|
-
}
|
|
1236
|
-
catch (err) {
|
|
1237
|
-
process.stderr.write(`[acc-runner] cost-event post failed: ${err.message}\n`);
|
|
1238
|
-
}
|
|
1239
|
-
}
|
|
1240
|
-
return {
|
|
1241
|
-
exitCode: outcome.exitCode,
|
|
1242
|
-
stdout: capturedStdout,
|
|
1243
|
-
stderr: capturedStderr,
|
|
1244
|
-
durationMs: Date.now() - spawnedAt,
|
|
1245
|
-
};
|
|
1246
|
-
};
|
|
1247
|
-
// v0.53 T-53-4 / v0.63 T-63-1: instant-empty tune. A lone instant-empty
|
|
1248
|
-
// exit (the pd36 fast-exit fingerprint, now classified
|
|
1249
|
-
// `claude_unavailable`) is too weak a signal to act on — v0.52 incident
|
|
1250
|
-
// (d) quarantined a working machine on ONE 249ms empty exit. So when the
|
|
1251
|
-
// FIRST exit is a *heuristic* claude_unavailable we log it and retry the
|
|
1252
|
-
// spawn once; a single blip that the retry recovers stays silent. Only a
|
|
1253
|
-
// CONFIRMED streak (two consecutive instant-empties) is acted on — and
|
|
1254
|
-
// even then v0.63 routes it to pause + backoff + retry (NOT quarantine),
|
|
1255
|
-
// gated by the health probe below. Definitive failures (dyld/OOM/exit
|
|
1256
|
-
// 127, usage/auth patterns — classified.heuristic falsey) skip the retry
|
|
1257
|
-
// and quarantine immediately, unchanged from v0.48.
|
|
1258
|
-
let attempt = await runClaudeOnce();
|
|
1259
|
-
const classifyIfFailed = (a) => a.exitCode === 0
|
|
1260
|
-
? null
|
|
1261
|
-
: classifyClaudeExit(a.exitCode, a.stderr, a.stdout, a.durationMs);
|
|
1262
|
-
let classified = classifyIfFailed(attempt);
|
|
1263
|
-
let instantEmptyStreak = classified?.class === "claude_unavailable" && classified.heuristic ? 1 : 0;
|
|
1264
|
-
if (instantEmptyStreak === 1 && !cancelled) {
|
|
1265
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1266
|
-
phase: "claude_exit",
|
|
1267
|
-
stream: "stderr",
|
|
1268
|
-
event: "claude_unavailable_blip",
|
|
1269
|
-
attempt: 1,
|
|
1270
|
-
duration_ms: attempt.durationMs,
|
|
1271
|
-
message: "instant empty claude exit — single blip, retrying task once before pausing",
|
|
1272
|
-
});
|
|
1273
|
-
attempt = await runClaudeOnce();
|
|
1274
|
-
classified = classifyIfFailed(attempt);
|
|
1275
|
-
if (classified?.class === "claude_unavailable" && classified.heuristic) {
|
|
1276
|
-
instantEmptyStreak = 2;
|
|
1277
|
-
}
|
|
1278
|
-
}
|
|
1279
|
-
// Restore .mcp.json as soon as Claude is done — its MCP subprocess
|
|
1280
|
-
// tree comes down with it, so leaving our token-bearing config on
|
|
1281
|
-
// disk a moment longer is pure exposure surface.
|
|
1282
|
-
if (mcpCleanup) {
|
|
1283
|
-
try {
|
|
1284
|
-
await mcpCleanup.restore();
|
|
1285
|
-
}
|
|
1286
|
-
catch (err) {
|
|
1287
|
-
process.stderr.write(`[acc-runner] mcp .mcp.json restore failed: ${err.message}\n`);
|
|
1288
|
-
}
|
|
1289
|
-
}
|
|
1290
|
-
if (cancelled) {
|
|
1291
|
-
await appendEvent(deps.supabase, taskId, "cancelled", {
|
|
1292
|
-
exit_code: attempt.exitCode,
|
|
1293
|
-
});
|
|
1294
|
-
return { taskId, status: "cancelled", exitCode: attempt.exitCode };
|
|
1295
|
-
}
|
|
1296
|
-
if (attempt.exitCode !== 0) {
|
|
1297
|
-
// v0.48: classify the failure so the quarantine cause is consistent
|
|
1298
|
-
// across the event, the bus message, and the RunTaskOutcome. `let`
|
|
1299
|
-
// because v0.63's health probe can upgrade a claude_unavailable exit
|
|
1300
|
-
// to env_broken when the probe proves the host is genuinely broken.
|
|
1301
|
-
let cls = classified ??
|
|
1302
|
-
classifyClaudeExit(attempt.exitCode, attempt.stderr, attempt.stdout, attempt.durationMs);
|
|
1303
|
-
// v0.56 (T-56-1): capacity exhaustion is NOT a task failure. Do not
|
|
1304
|
-
// transition the task to 'failed' (that would dead-letter it and burn
|
|
1305
|
-
// the reviewer/automerge retry cap). Leave it 'running' with its
|
|
1306
|
-
// signal loop stopped (the outer finally does this) so the
|
|
1307
|
-
// stale-running sweep returns it to 'queued' with runner_id cleared —
|
|
1308
|
-
// a lossless requeue for any runner that still has capacity. Surface
|
|
1309
|
-
// the parsed reset time so watch.ts can pause until the window
|
|
1310
|
-
// reopens. The locks are released by the outer finally exactly as on
|
|
1311
|
-
// any other exit path.
|
|
1312
|
-
if (cls.class === "capacity_exhausted") {
|
|
1313
|
-
const resumeMs = extractResetTime(`${attempt.stderr}\n${attempt.stdout}`);
|
|
1314
|
-
const resumeAt = resumeMs !== null ? new Date(resumeMs).toISOString() : null;
|
|
1315
|
-
await appendEvent(deps.supabase, taskId, "capacity_paused", {
|
|
1316
|
-
phase: "claude_exit",
|
|
1317
|
-
exit_code: attempt.exitCode,
|
|
1318
|
-
exit_class: cls.class,
|
|
1319
|
-
resume_at: resumeAt,
|
|
1320
|
-
detail: cls.detail,
|
|
1321
|
-
runner_id: deps.session.runnerId,
|
|
1322
|
-
});
|
|
1323
|
-
return {
|
|
1324
|
-
taskId,
|
|
1325
|
-
status: "capacity_paused",
|
|
1326
|
-
phase: "claude_exit",
|
|
1327
|
-
exitCode: attempt.exitCode,
|
|
1328
|
-
error: cls.detail,
|
|
1329
|
-
capacity_exhausted: true,
|
|
1330
|
-
resume_at: resumeAt,
|
|
1331
|
-
};
|
|
1332
|
-
}
|
|
1333
|
-
// v0.63 (T-63-1): a bare silent instant-empty exit. Silence alone
|
|
1334
|
-
// cannot tell "broken host" from "claude momentarily unavailable", so
|
|
1335
|
-
// run an authoritative `claude --version` health probe as the
|
|
1336
|
-
// tiebreaker. Probe OK → the host is fine → treat exactly like
|
|
1337
|
-
// capacity_exhausted (pause + EXPONENTIAL backoff + retry; leave the
|
|
1338
|
-
// task 'running' for the stale-running sweep; NEVER quarantine, NEVER
|
|
1339
|
-
// burn the task/review retry caps). Probe FAILS → the binary/host is
|
|
1340
|
-
// genuinely broken → upgrade to env_broken and fall through to the
|
|
1341
|
-
// quarantine path (the pd36 contract, now gated by a real signal
|
|
1342
|
-
// rather than silence). A human INFRA alert fires only after a
|
|
1343
|
-
// SUSTAINED streak inside a rolling window, never on 2.
|
|
1344
|
-
if (cls.class === "claude_unavailable") {
|
|
1345
|
-
const probe = await healthProbe();
|
|
1346
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1347
|
-
phase: "claude_exit",
|
|
1348
|
-
stream: "stderr",
|
|
1349
|
-
event: "claude_health_probe",
|
|
1350
|
-
ok: probe.ok,
|
|
1351
|
-
detail: probe.detail,
|
|
1352
|
-
consecutive_instant_empty: instantEmptyStreak,
|
|
1353
|
-
});
|
|
1354
|
-
if (probe.ok) {
|
|
1355
|
-
const decision = recordClaudeUnavailable(Date.now());
|
|
1356
|
-
await appendEvent(deps.supabase, taskId, "claude_unavailable", {
|
|
1357
|
-
phase: "claude_exit",
|
|
1358
|
-
exit_code: attempt.exitCode,
|
|
1359
|
-
exit_class: cls.class,
|
|
1360
|
-
consecutive_instant_empty: instantEmptyStreak,
|
|
1361
|
-
duration_ms: attempt.durationMs,
|
|
1362
|
-
resume_at: decision.resumeAt,
|
|
1363
|
-
streak: decision.streak,
|
|
1364
|
-
window_count: decision.windowCount,
|
|
1365
|
-
detail: cls.detail,
|
|
1366
|
-
runner_id: deps.session.runnerId,
|
|
1367
|
-
});
|
|
1368
|
-
// Best-effort audit trail on the runner timeline.
|
|
1369
|
-
try {
|
|
1370
|
-
await deps.supabase.rpc("log_activity", {
|
|
1371
|
-
p_verb: "runner.claude_unavailable",
|
|
1372
|
-
p_target_id: deps.session.runnerId,
|
|
1373
|
-
p_target_type: "runner",
|
|
1374
|
-
p_payload: {
|
|
1375
|
-
task_id: taskId,
|
|
1376
|
-
streak: decision.streak,
|
|
1377
|
-
window_count: decision.windowCount,
|
|
1378
|
-
backoff_ms: decision.backoffMs,
|
|
1379
|
-
resume_at: decision.resumeAt,
|
|
1380
|
-
detail: cls.detail,
|
|
1381
|
-
},
|
|
1382
|
-
});
|
|
1383
|
-
}
|
|
1384
|
-
catch { /* best-effort */ }
|
|
1385
|
-
// Escalate to a human only on a SUSTAINED streak — not a single
|
|
1386
|
-
// pause, not 2. Still NOT a quarantine: the runner stays online,
|
|
1387
|
-
// paused + retrying, while an operator investigates infra.
|
|
1388
|
-
if (decision.alert) {
|
|
1389
|
-
try {
|
|
1390
|
-
await deps.supabase.rpc("log_activity", {
|
|
1391
|
-
p_verb: "runner.claude_unavailable_infra",
|
|
1392
|
-
p_target_id: deps.session.runnerId,
|
|
1393
|
-
p_target_type: "runner",
|
|
1394
|
-
p_payload: {
|
|
1395
|
-
task_id: taskId,
|
|
1396
|
-
window_count: decision.windowCount,
|
|
1397
|
-
streak: decision.streak,
|
|
1398
|
-
detail: cls.detail,
|
|
1399
|
-
},
|
|
1400
|
-
});
|
|
1401
|
-
}
|
|
1402
|
-
catch { /* best-effort */ }
|
|
1403
|
-
await postState_("blocked", "error_context", {
|
|
1404
|
-
task_id: taskId,
|
|
1405
|
-
phase: "code",
|
|
1406
|
-
error_class: "claude_unavailable_sustained",
|
|
1407
|
-
infra_alert: true,
|
|
1408
|
-
window_count: decision.windowCount,
|
|
1409
|
-
stderr_tail: cls.detail,
|
|
1410
|
-
});
|
|
1411
|
-
}
|
|
1412
|
-
return {
|
|
1413
|
-
taskId,
|
|
1414
|
-
status: "capacity_paused",
|
|
1415
|
-
phase: "claude_exit",
|
|
1416
|
-
exitCode: attempt.exitCode,
|
|
1417
|
-
error: `claude silently unavailable (instant-empty exit) — pausing ` +
|
|
1418
|
-
`${Math.round(decision.backoffMs / 1000)}s then retrying, not quarantining`,
|
|
1419
|
-
claude_unavailable: true,
|
|
1420
|
-
resume_at: decision.resumeAt,
|
|
1421
|
-
};
|
|
1422
|
-
}
|
|
1423
|
-
// Probe failed — the host/binary really is broken. This IS an
|
|
1424
|
-
// env_broken; fall through to the quarantine path below.
|
|
1425
|
-
cls = {
|
|
1426
|
-
exitCode: attempt.exitCode,
|
|
1427
|
-
class: "env_broken",
|
|
1428
|
-
detail: `${cls.detail}; claude health probe failed: ${probe.detail}`,
|
|
1429
|
-
};
|
|
1430
|
-
}
|
|
1431
|
-
// v0.48 / v0.63: by this point claude_unavailable has either returned
|
|
1432
|
-
// (probe OK) or been rewritten to env_broken (probe failed), so any
|
|
1433
|
-
// non-task_error class is a genuine quarantine cause. consecutive
|
|
1434
|
-
// mirrors the instant-empty streak for an env_broken upgraded from the
|
|
1435
|
-
// heuristic (≥1), so quarantine.json still records 1 (definitive) vs 2
|
|
1436
|
-
// (heuristic-confirmed).
|
|
1437
|
-
const quarantineCause = cls.class !== "task_error" ? cls.class : undefined;
|
|
1438
|
-
const quarantineConsecutive = quarantineCause === "env_broken" ? Math.max(instantEmptyStreak, 1) : undefined;
|
|
1439
|
-
await appendEvent(deps.supabase, taskId, "error", {
|
|
1440
|
-
phase: "claude_exit",
|
|
1441
|
-
exit_code: attempt.exitCode,
|
|
1442
|
-
exit_class: cls.class,
|
|
1443
|
-
exit_class_heuristic: cls.heuristic === true,
|
|
1444
|
-
consecutive_instant_empty: instantEmptyStreak,
|
|
1445
|
-
quarantine: quarantineCause !== undefined,
|
|
1446
|
-
stderr_tail: attempt.stderr.slice(-2000),
|
|
1447
|
-
});
|
|
1448
|
-
await postState_("blocked", "error_context", {
|
|
1449
|
-
task_id: taskId,
|
|
1450
|
-
phase: "code",
|
|
1451
|
-
error_class: `claude_exit_${attempt.exitCode ?? "unknown"}`,
|
|
1452
|
-
// v0.48: include the classified failure class so the planner can
|
|
1453
|
-
// distinguish machine-level failures from task-level failures.
|
|
1454
|
-
exit_class: cls.class,
|
|
1455
|
-
// Recommended cap from MESSAGING_PROTOCOLS.md: last ~4 KB.
|
|
1456
|
-
stderr_tail: attempt.stderr.slice(-4000),
|
|
1457
|
-
});
|
|
1458
|
-
await deps.supabase.rpc("transition_task", {
|
|
1459
|
-
p_task_id: taskId,
|
|
1460
|
-
p_new_status: "failed",
|
|
1461
|
-
});
|
|
1462
|
-
return {
|
|
1463
|
-
taskId,
|
|
1464
|
-
status: "failed",
|
|
1465
|
-
phase: "claude_exit",
|
|
1466
|
-
exitCode: attempt.exitCode,
|
|
1467
|
-
error: attempt.stderr.slice(-200).trim() || cls.detail,
|
|
1468
|
-
quarantine_cause: quarantineCause,
|
|
1469
|
-
quarantine_consecutive: quarantineConsecutive,
|
|
1470
|
-
};
|
|
1471
|
-
}
|
|
1472
|
-
// Success: expose the final attempt's stdout under the name the
|
|
1473
|
-
// downstream report/PR path expects.
|
|
1474
|
-
const capturedStdout = attempt.stdout;
|
|
1475
|
-
// v0.63 (T-63-1): a clean claude run proves the binary recovered — reset
|
|
1476
|
-
// the silent-unavailability streak so the next blip restarts the backoff
|
|
1477
|
-
// ladder from the base and the rolling INFRA-alert window starts fresh.
|
|
1478
|
-
clearClaudeUnavailableStreak();
|
|
1479
|
-
// v0.14-MESSAGING-RUNTIME-WIRE: Claude finished successfully.
|
|
1480
|
-
// testing covers the post-exit window where the runner inspects
|
|
1481
|
-
// the captured stdout (parses the report, builds the cost event)
|
|
1482
|
-
// before publishing. reviewing is emitted just before opening
|
|
1483
|
-
// the PR so a subscriber can pre-stage the review surface.
|
|
1484
|
-
await postState_("testing", "info", {
|
|
1485
|
-
runner_id: deps.session.runnerId,
|
|
1486
|
-
exit_code: attempt.exitCode ?? 0,
|
|
1487
|
-
});
|
|
1488
|
-
// v0.51 T-51-1: rework completion path. Push the fixes to the SAME
|
|
1489
|
-
// branch (same PR) — never open a new PR — then re-enqueue a review
|
|
1490
|
-
// and flip the origin task back to needs-review so the existing
|
|
1491
|
-
// automerge flow re-evaluates. No v0.32-A "PR exists" fallback on
|
|
1492
|
-
// push failure here: a rework push that didn't land means the
|
|
1493
|
-
// reviewer feedback was NOT addressed, so the task must fail.
|
|
1494
|
-
if (reworkMeta) {
|
|
1495
|
-
// v0.57 T-57-2: gate conflict-resolution reworks before the push.
|
|
1496
|
-
const reworkIntegrity = await runResolutionIntegrityGate();
|
|
1497
|
-
if (reworkIntegrity)
|
|
1498
|
-
return reworkIntegrity;
|
|
1499
|
-
try {
|
|
1500
|
-
await git.push(workdir, branch);
|
|
1501
|
-
}
|
|
1502
|
-
catch (err) {
|
|
1503
|
-
const msg = err.message;
|
|
1504
|
-
await appendEvent(deps.supabase, taskId, "error", {
|
|
1505
|
-
phase: "push",
|
|
1506
|
-
error: msg,
|
|
1507
|
-
rework: true,
|
|
1508
|
-
pr_number: reworkMeta.pr_number,
|
|
1509
|
-
});
|
|
1510
|
-
await postState_("blocked", "error_context", {
|
|
1511
|
-
task_id: taskId,
|
|
1512
|
-
phase: "review",
|
|
1513
|
-
error_class: "rework_push_failed",
|
|
1514
|
-
stderr_tail: msg,
|
|
1515
|
-
});
|
|
1516
|
-
await deps.supabase.rpc("transition_task", {
|
|
1517
|
-
p_task_id: taskId,
|
|
1518
|
-
p_new_status: "failed",
|
|
1519
|
-
});
|
|
1520
|
-
return { taskId, status: "failed", phase: "push", error: msg };
|
|
1521
|
-
}
|
|
1522
|
-
// Order matters: request the fresh review BEFORE flipping the
|
|
1523
|
-
// origin task back to needs-review. The automerge cron acts on the
|
|
1524
|
-
// LATEST review_queue row — if the task re-entered candidacy while
|
|
1525
|
-
// the stale completed reject was still latest, the cron would
|
|
1526
|
-
// re-finalize that reject and burn a rework cycle without any
|
|
1527
|
-
// re-review. With the pending row in place first, the cron sees
|
|
1528
|
-
// review_pending and waits for the real verdict.
|
|
1529
|
-
if (reworkMeta.origin_task_id) {
|
|
1530
|
-
const requested = await deps.supabase.rpc("request_review", {
|
|
1531
|
-
p_task_id: reworkMeta.origin_task_id,
|
|
1532
|
-
p_pr_number: reworkMeta.pr_number,
|
|
1533
|
-
});
|
|
1534
|
-
if (requested.error) {
|
|
1535
|
-
process.stderr.write(`[acc-runner] rework request_review(${reworkMeta.origin_task_id}) failed: ` +
|
|
1536
|
-
`${requested.error.message}\n`);
|
|
1537
|
-
}
|
|
1538
|
-
const flipped = await deps.supabase.rpc("transition_task", {
|
|
1539
|
-
p_task_id: reworkMeta.origin_task_id,
|
|
1540
|
-
p_new_status: "needs-review",
|
|
1541
|
-
});
|
|
1542
|
-
if (flipped.error) {
|
|
1543
|
-
// Best-effort: requires migration 0172's widened matrix
|
|
1544
|
-
// (changes-requested → needs-review). On failure the pending
|
|
1545
|
-
// review row still completes via the runner reviewer; the
|
|
1546
|
-
// operator sees the origin task parked at changes-requested.
|
|
1547
|
-
process.stderr.write(`[acc-runner] rework transition(${reworkMeta.origin_task_id}→needs-review) failed: ` +
|
|
1548
|
-
`${flipped.error.message}\n`);
|
|
1549
|
-
}
|
|
1550
|
-
}
|
|
1551
|
-
else {
|
|
1552
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1553
|
-
phase: "rework",
|
|
1554
|
-
stream: "stderr",
|
|
1555
|
-
message: "rework meta has no origin_task_id — re-review not auto-triggered",
|
|
1556
|
-
});
|
|
1557
|
-
}
|
|
1558
|
-
await appendEvent(deps.supabase, taskId, "rework_pushed", {
|
|
1559
|
-
pr_number: reworkMeta.pr_number,
|
|
1560
|
-
repo: reworkMeta.repo,
|
|
1561
|
-
branch,
|
|
1562
|
-
cycle: reworkMeta.cycle,
|
|
1563
|
-
origin_task_id: reworkMeta.origin_task_id || null,
|
|
1564
|
-
});
|
|
1565
|
-
await postState_("done", "info", {
|
|
1566
|
-
runner_id: deps.session.runnerId,
|
|
1567
|
-
pr_number: reworkMeta.pr_number,
|
|
1568
|
-
rework: true,
|
|
1569
|
-
});
|
|
1570
|
-
await deps.supabase.rpc("transition_task", {
|
|
1571
|
-
p_task_id: taskId,
|
|
1572
|
-
p_new_status: "done",
|
|
1573
|
-
});
|
|
1574
|
-
return { taskId, status: "ok", exitCode: 0 };
|
|
1575
|
-
}
|
|
1576
|
-
// v0.53 T-53-4: the task's worktree now holds the agent's commits.
|
|
1577
|
-
// Emit the first-commit SLO signal once, before push, so the funnel
|
|
1578
|
-
// sees "coding started" independent of whether the push/PR succeeds.
|
|
1579
|
-
await emitFirstCommit(deps.supabase, git, taskId, workdir, resolvedBaseRef, branch);
|
|
1580
|
-
// v0.57 T-57-2: gate conflict-resolution tasks before the push so a
|
|
1581
|
-
// broken merge never reaches a PR. No-op for every non-conflict task.
|
|
1582
|
-
const integrityGate = await runResolutionIntegrityGate();
|
|
1583
|
-
if (integrityGate)
|
|
1584
|
-
return integrityGate;
|
|
1585
|
-
// 5. Push + open PR. Both run from the worktree so the operator's
|
|
1586
|
-
// shared clone never has the task branch checked out.
|
|
1587
|
-
let prUrl = "";
|
|
1588
|
-
try {
|
|
1589
|
-
await git.push(workdir, branch);
|
|
1590
|
-
}
|
|
1591
|
-
catch (err) {
|
|
1592
|
-
const msg = err.message;
|
|
1593
|
-
// v0.32-A: before failing, check whether the branch was already
|
|
1594
|
-
// pushed by a prior run (non-fast-forward error on re-push). If
|
|
1595
|
-
// an open PR exists, the branch is already live — treat as success.
|
|
1596
|
-
const existingPrUrlOnPushFail = await gh.findOpenPR(workdir, branch).catch(() => null);
|
|
1597
|
-
if (existingPrUrlOnPushFail) {
|
|
1598
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1599
|
-
phase: "push",
|
|
1600
|
-
stream: "stdout",
|
|
1601
|
-
event: "retry.push_conflict_resolved",
|
|
1602
|
-
branch,
|
|
1603
|
-
pr_url: existingPrUrlOnPushFail,
|
|
1604
|
-
message: "push failed but open PR exists — treating as idempotent success",
|
|
1605
|
-
});
|
|
1606
|
-
prUrl = existingPrUrlOnPushFail;
|
|
1607
|
-
// Fall through to postState_("done") below.
|
|
1608
|
-
}
|
|
1609
|
-
else {
|
|
1610
|
-
await appendEvent(deps.supabase, taskId, "error", { phase: "push", error: msg });
|
|
1611
|
-
await postState_("blocked", "error_context", {
|
|
1612
|
-
task_id: taskId,
|
|
1613
|
-
phase: "review",
|
|
1614
|
-
error_class: "git_push_failed",
|
|
1615
|
-
stderr_tail: msg,
|
|
1616
|
-
});
|
|
1617
|
-
await deps.supabase.rpc("transition_task", {
|
|
1618
|
-
p_task_id: taskId,
|
|
1619
|
-
p_new_status: "failed",
|
|
1620
|
-
});
|
|
1621
|
-
return { taskId, status: "failed", phase: "push", error: msg };
|
|
1622
|
-
}
|
|
1623
|
-
}
|
|
1624
|
-
await postState_("reviewing", "info", {
|
|
1625
|
-
runner_id: deps.session.runnerId,
|
|
1626
|
-
branch,
|
|
1627
|
-
});
|
|
1628
|
-
try {
|
|
1629
|
-
const body = extractReportFromOutput(capturedStdout);
|
|
1630
|
-
const pr = await gh.openPR(workdir, {
|
|
1631
|
-
title: prTitleForTask(task.id, task.title),
|
|
1632
|
-
body,
|
|
1633
|
-
base: integrationBranch,
|
|
1634
|
-
});
|
|
1635
|
-
prUrl = pr.url;
|
|
1636
|
-
}
|
|
1637
|
-
catch (err) {
|
|
1638
|
-
const msg = err.message;
|
|
1639
|
-
// v0.8.2: gh pr create returns a non-zero exit when a PR already
|
|
1640
|
-
// exists for this branch (common on runner-crash + re-pick-up).
|
|
1641
|
-
// GitHub's error message embeds the existing PR URL. Extract it
|
|
1642
|
-
// and treat the outcome as a successful PR open so set_task_pr can
|
|
1643
|
-
// transition the task to needs-review rather than leaving it stuck.
|
|
1644
|
-
const prAlreadyExistsMatch = /already exists/i.test(msg) &&
|
|
1645
|
-
msg.match(/https:\/\/github\.com\/[^\s]+\/pull\/\d+/);
|
|
1646
|
-
if (prAlreadyExistsMatch) {
|
|
1647
|
-
prUrl = prAlreadyExistsMatch[0];
|
|
1648
|
-
try {
|
|
1649
|
-
await appendEvent(deps.supabase, taskId, "log", {
|
|
1650
|
-
phase: "pr_open",
|
|
1651
|
-
stream: "stdout",
|
|
1652
|
-
event: "retry.pr_create_already_exists",
|
|
1653
|
-
pr_url: prUrl,
|
|
1654
|
-
message: "gh pr create reported PR already exists — recovered URL from error message",
|
|
1655
|
-
});
|
|
1656
|
-
}
|
|
1657
|
-
catch (logErr) {
|
|
1658
|
-
process.stderr.write(`[acc-runner] pr_open log failed: ${logErr.message}\n`);
|
|
1659
|
-
}
|
|
1660
|
-
}
|
|
1661
|
-
else {
|
|
1662
|
-
// Best-effort log; don't transition to failed — the push succeeded
|
|
1663
|
-
// and the user can open a PR manually (webhook back-write picks
|
|
1664
|
-
// it up). Wrapped so a logging RPC failure can't block the
|
|
1665
|
-
// terminal status write below.
|
|
1666
|
-
try {
|
|
1667
|
-
await appendEvent(deps.supabase, taskId, "error", { phase: "pr_open", error: msg });
|
|
1668
|
-
}
|
|
1669
|
-
catch (logErr) {
|
|
1670
|
-
process.stderr.write(`[acc-runner] pr_open error log failed: ${logErr.message}\n`);
|
|
1671
|
-
}
|
|
1672
|
-
}
|
|
1673
|
-
}
|
|
1674
|
-
// v0.35-B: write pr_number + transition running→needs-review via
|
|
1675
|
-
// acc.set_task_pr. Best-effort: a missing RPC (Track A migration
|
|
1676
|
-
// not yet applied) OR any other error falls back to the
|
|
1677
|
-
// webhook-driven running→done path (matrix 0156). Only attempted
|
|
1678
|
-
// when a PR URL is in hand — the push-success / PR-open-failure
|
|
1679
|
-
// branch above leaves prUrl empty and is handled exclusively by
|
|
1680
|
-
// the webhook back-write.
|
|
1681
|
-
if (prUrl) {
|
|
1682
|
-
await setTaskPr(deps.supabase, taskId, prUrl);
|
|
1683
|
-
}
|
|
1684
|
-
// v0.33-C: terminal status written before cleanup so a runner crash
|
|
1685
|
-
// during cleanup doesn't leave the task non-terminal. Fires the
|
|
1686
|
-
// 'done' bus event the instant a PR URL is confirmed (or null when
|
|
1687
|
-
// the PR-open RPC failed but the push succeeded — the webhook
|
|
1688
|
-
// back-write still finalises the task). v0.14-MESSAGING-RUNTIME-WIRE
|
|
1689
|
-
// semantics: protocol='info' today; handoff requires a
|
|
1690
|
-
// target_capability lookup the runner doesn't have yet. The 'done'
|
|
1691
|
-
// bus state describes the runner's spawn lifecycle (PR opened,
|
|
1692
|
-
// runner finished) and is independent of the task status — v0.35-B
|
|
1693
|
-
// moves the task itself to needs-review via setTaskPr above.
|
|
1694
|
-
await postState_("done", "info", {
|
|
1695
|
-
runner_id: deps.session.runnerId,
|
|
1696
|
-
pr_url: prUrl || null,
|
|
1697
|
-
});
|
|
1698
|
-
// v0.33-C: final event log is non-fatal — any throw here would
|
|
1699
|
-
// skip the return statement and leave the worktree-cleanup finally
|
|
1700
|
-
// running with the wrong outcome, but the terminal bus event above
|
|
1701
|
-
// is already on the wire.
|
|
1702
|
-
if (prUrl) {
|
|
1703
|
-
try {
|
|
1704
|
-
await appendEvent(deps.supabase, taskId, "pr-opened", { url: prUrl });
|
|
1705
|
-
}
|
|
1706
|
-
catch (logErr) {
|
|
1707
|
-
process.stderr.write(`[acc-runner] pr-opened event log failed: ${logErr.message}\n`);
|
|
1708
|
-
}
|
|
1709
|
-
}
|
|
1710
|
-
return { taskId, status: "ok", prUrl, exitCode: 0 };
|
|
1711
|
-
}
|
|
1712
|
-
finally {
|
|
1713
|
-
// v0.11-F: tear down the per-task worktree and release the PID
|
|
1714
|
-
// lock on every completion path (success, failure, cancellation,
|
|
1715
|
-
// unexpected throw). Both legs are best-effort — leaking either
|
|
1716
|
-
// resource is preferable to masking the original return value.
|
|
1717
|
-
if (worktree) {
|
|
1718
|
-
// v0.37-A: clear any stale .git/HEAD.lock or index.lock so the
|
|
1719
|
-
// `git worktree remove` + `prune` inside cleanup() don't trip on
|
|
1720
|
-
// a leftover lock from the prior worktree add or claude run.
|
|
1721
|
-
clearGitLocks(repoPath);
|
|
1722
|
-
try {
|
|
1723
|
-
await worktree.cleanup();
|
|
1724
|
-
}
|
|
1725
|
-
catch (err) {
|
|
1726
|
-
process.stderr.write(`[acc-runner] worktree cleanup failed: ${err.message}\n`);
|
|
1727
|
-
}
|
|
1728
|
-
}
|
|
1729
|
-
try {
|
|
1730
|
-
await lock.release();
|
|
1731
|
-
}
|
|
1732
|
-
catch (err) {
|
|
1733
|
-
process.stderr.write(`[acc-runner] lock release failed: ${err.message}\n`);
|
|
1734
|
-
}
|
|
1735
|
-
}
|
|
1736
|
-
}
|
|
1737
|
-
finally {
|
|
1738
|
-
// v0.12-RESUME: stop the periodic signal loop before releasing
|
|
1739
|
-
// locks so a late-firing signal can't bump the column after the
|
|
1740
|
-
// task transitions to a terminal status (the RPC is no-op on
|
|
1741
|
-
// non-running rows anyway, but stopping early avoids the extra
|
|
1742
|
-
// RPC round-trip).
|
|
1743
|
-
stopSignalLoop();
|
|
1744
|
-
// v0.53 T-53-4: disarm the per-task GitHub budget so any REST call
|
|
1745
|
-
// made outside a task run is unbudgeted.
|
|
1746
|
-
endTaskGithubBudget();
|
|
1747
|
-
await releaseLocks();
|
|
1748
|
-
}
|
|
1749
|
-
})();
|
|
1750
|
-
return {
|
|
1751
|
-
taskId,
|
|
1752
|
-
promise,
|
|
1753
|
-
cancel() {
|
|
1754
|
-
cancelled = true;
|
|
1755
|
-
if (child && child.pid) {
|
|
1756
|
-
try {
|
|
1757
|
-
child.kill("SIGTERM");
|
|
1758
|
-
}
|
|
1759
|
-
catch {
|
|
1760
|
-
// Already dead — ignore.
|
|
1761
|
-
}
|
|
1762
|
-
}
|
|
1763
|
-
},
|
|
1764
|
-
};
|
|
1765
|
-
}
|
|
1766
|
-
//# sourceMappingURL=task-runner.js.map
|