hilos-agent 0.7.0 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -15
- package/bin/hilos-agent.mjs +14 -6
- package/package.json +1 -1
- package/src/acp-session.mjs +553 -0
- package/src/agent-events.mjs +764 -45
- package/src/attachments.mjs +310 -0
- package/src/claude-permissions.mjs +445 -0
- package/src/cli.mjs +56 -0
- package/src/codex-mcp-session.mjs +619 -0
- package/src/config.mjs +84 -0
- package/src/daemon.mjs +23 -0
- package/src/handler.mjs +1210 -100
- package/src/hook.mjs +408 -91
- package/src/mcp.mjs +1 -0
- package/src/model-resolve.mjs +180 -11
- package/src/permission-gate.mjs +269 -0
- package/src/progress-emitter.mjs +105 -3
- package/src/queue.mjs +21 -5
- package/src/redact.mjs +11 -1
- package/src/reply-bridge.mjs +484 -0
- package/src/resume.mjs +48 -11
- package/src/run.mjs +135 -4
- package/src/thread-pr.mjs +305 -0
- package/src/transcript.mjs +153 -0
package/src/handler.mjs
CHANGED
|
@@ -25,6 +25,7 @@ import {
|
|
|
25
25
|
prTitleBody,
|
|
26
26
|
mentionHandle,
|
|
27
27
|
detectPrContinuation,
|
|
28
|
+
selfDrivenShipPlan,
|
|
28
29
|
} from "./daemon.mjs";
|
|
29
30
|
import {
|
|
30
31
|
runCli,
|
|
@@ -36,17 +37,28 @@ import {
|
|
|
36
37
|
scrubHilosEnv,
|
|
37
38
|
envForCwd,
|
|
38
39
|
} from "./cli.mjs";
|
|
39
|
-
import { makeStreamParser } from "./agent-events.mjs";
|
|
40
|
+
import { makeStreamParser, createUsageFold } from "./agent-events.mjs";
|
|
40
41
|
import {
|
|
41
42
|
detectVendor,
|
|
42
43
|
codeStreamArgs,
|
|
43
44
|
codeDirArgs,
|
|
45
|
+
codeImageArgs,
|
|
44
46
|
codeProjectKey,
|
|
47
|
+
imagesNeedReading,
|
|
45
48
|
attachTarget,
|
|
46
49
|
createProgressEmitter,
|
|
47
50
|
fastChatCmd,
|
|
48
51
|
} from "./progress-emitter.mjs";
|
|
52
|
+
import { createTranscriptTap } from "./transcript.mjs";
|
|
53
|
+
import { imagePromptNote, renderAttachmentLine } from "./attachments.mjs";
|
|
49
54
|
import { resolveFollowupMode, classifyFollowupCue, normalizeSignal } from "./followup.mjs";
|
|
55
|
+
import {
|
|
56
|
+
anchorIterateDecision,
|
|
57
|
+
confirmAnchorOpen,
|
|
58
|
+
prActionRelayDecision,
|
|
59
|
+
prActionReply,
|
|
60
|
+
staleAnchorNote,
|
|
61
|
+
} from "./thread-pr.mjs";
|
|
50
62
|
import { buildResumeArgs, resumeDecision, readStateEntry, writeState, HILOS_DIR } from "./resume.mjs";
|
|
51
63
|
import { createModelArgsResolver } from "./model-resolve.mjs";
|
|
52
64
|
import {
|
|
@@ -58,6 +70,18 @@ import {
|
|
|
58
70
|
import { buildMemoryBlock } from "./memory.mjs";
|
|
59
71
|
import { deployFolder, resolveDeployTarget } from "./deploy.mjs";
|
|
60
72
|
import { runOpenCodeHttpSession } from "./opencode-session.mjs";
|
|
73
|
+
import { runAcpSession } from "./acp-session.mjs";
|
|
74
|
+
import {
|
|
75
|
+
shouldGateClaudePermissions,
|
|
76
|
+
startClaudePermissionServer,
|
|
77
|
+
} from "./claude-permissions.mjs";
|
|
78
|
+
import {
|
|
79
|
+
codexMcpTransportUnavailable,
|
|
80
|
+
codexSandboxFromArgs,
|
|
81
|
+
runCodexMcpSession,
|
|
82
|
+
shouldGateCodexPermissions,
|
|
83
|
+
} from "./codex-mcp-session.mjs";
|
|
84
|
+
import { createUngatedRunNotice } from "./permission-gate.mjs";
|
|
61
85
|
|
|
62
86
|
/**
|
|
63
87
|
* The environment for a coding/chat CLI run. runCli always strips HILOS_* on top
|
|
@@ -138,6 +162,49 @@ function compactRunMarker(status, branch) {
|
|
|
138
162
|
return `Failed:${b}`;
|
|
139
163
|
}
|
|
140
164
|
|
|
165
|
+
/**
|
|
166
|
+
* A `--model x` / `-m x` the operator pinned on the coding command (0787).
|
|
167
|
+
* `modelArgsFor` deliberately returns [] in that case — the hand-pin wins — so
|
|
168
|
+
* this is where a pinned id is recovered for the receipt.
|
|
169
|
+
*/
|
|
170
|
+
export function pinnedModelId(codingCmd) {
|
|
171
|
+
const m = /(?:^|\s)(?:--model|-m)(?:\s+|=)("[^"]+"|'[^']+'|\S+)/.exec(String(codingCmd || ""));
|
|
172
|
+
return m ? m[1].replace(/^["']|["']$/g, "") : "";
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* The `usage` argument for a report (0787), or `{}` when the CLI told us
|
|
177
|
+
* nothing — an omitted field is the honest answer, and the card degrades to
|
|
178
|
+
* "Tokens unavailable" rather than claiming a free run.
|
|
179
|
+
*
|
|
180
|
+
* TAKES from the fold. Every report writes its OWN ledger row server-side, so a
|
|
181
|
+
* run that reports more than once (a proposal card and then a shipped card, a
|
|
182
|
+
* revision round per loop) must carry only what is new since the last report;
|
|
183
|
+
* a running total would bill the same tokens twice.
|
|
184
|
+
*
|
|
185
|
+
* The model: whatever the stream named wins. Codex names none, so the id
|
|
186
|
+
* resolved at spawn — or pinned on the command — stands in, and failing both,
|
|
187
|
+
* the CLI's own name. The ledger has to say what ran.
|
|
188
|
+
*/
|
|
189
|
+
export function reportUsageArgs(fold, { vendor, modelId, runId } = {}) {
|
|
190
|
+
const totals = fold && typeof fold.take === "function" ? fold.take() : null;
|
|
191
|
+
if (!totals) return {};
|
|
192
|
+
const cli = vendor && vendor !== "unknown" ? vendor : null;
|
|
193
|
+
const model = totals.model || modelId || cli || "";
|
|
194
|
+
if (!model) return {};
|
|
195
|
+
const usage = {
|
|
196
|
+
model,
|
|
197
|
+
inputTokens: totals.inputTokens,
|
|
198
|
+
outputTokens: totals.outputTokens,
|
|
199
|
+
cacheReadTokens: totals.cacheReadTokens,
|
|
200
|
+
cacheCreationTokens: totals.cacheCreationTokens,
|
|
201
|
+
};
|
|
202
|
+
if (cli) usage.vendor = cli;
|
|
203
|
+
if (typeof totals.costUsd === "number") usage.costUsd = totals.costUsd;
|
|
204
|
+
if (runId) usage.runId = runId;
|
|
205
|
+
return { usage };
|
|
206
|
+
}
|
|
207
|
+
|
|
141
208
|
// Every helper below runs a tool in a directory we choose, so each one hands
|
|
142
209
|
// the child a PWD that matches that directory instead of the daemon's launch
|
|
143
210
|
// dir (0615). git and gh both use the real cwd, so this is hygiene rather than
|
|
@@ -152,6 +219,9 @@ function defaultDeps() {
|
|
|
152
219
|
// selected trust boundary without spawning a real model process.
|
|
153
220
|
runCli: (opts) => runCli(opts),
|
|
154
221
|
runOpenCodeHttpSession: (opts) => runOpenCodeHttpSession(opts),
|
|
222
|
+
runAcpSession: (opts) => runAcpSession(opts),
|
|
223
|
+
runCodexMcpSession: (opts) => runCodexMcpSession(opts),
|
|
224
|
+
startClaudePermissionServer: (opts) => startClaudePermissionServer(opts),
|
|
155
225
|
// Does a path exist on disk? Injectable so folder mode's "missing folder"
|
|
156
226
|
// guard is unit-testable without touching the real filesystem.
|
|
157
227
|
pathExists: (p) => existsSync(p),
|
|
@@ -164,7 +234,7 @@ function defaultDeps() {
|
|
|
164
234
|
{ cwd, env: cwdEnv(cwd), encoding: "utf8" },
|
|
165
235
|
);
|
|
166
236
|
const url = (r.stdout || "").trim().split("\n").filter(Boolean).pop() || null;
|
|
167
|
-
return { ok: r.status === 0, url, stderr: r.stderr || "" };
|
|
237
|
+
return { ok: r.status === 0, url, stdout: r.stdout || "", stderr: r.stderr || "" };
|
|
168
238
|
},
|
|
169
239
|
// The open PR for a branch, if one already exists — so when the CLI opened a
|
|
170
240
|
// PR itself we report THAT instead of opening a duplicate.
|
|
@@ -189,6 +259,37 @@ function defaultDeps() {
|
|
|
189
259
|
const out = (r.stdout || "").trim();
|
|
190
260
|
return r.status === 0 && out ? out : null;
|
|
191
261
|
},
|
|
262
|
+
// The live state of a PR (0704): is it still OPEN, what is its head branch,
|
|
263
|
+
// and does that head live in the base repo? The staleness guard for adopting
|
|
264
|
+
// the thread's pull request — a merged/closed anchor must never take a push,
|
|
265
|
+
// and a fork head is another repository's branch. Returns null when `gh`
|
|
266
|
+
// can't answer, which the caller reads as "no proof, no adoption".
|
|
267
|
+
prState: (cwd, ref) => {
|
|
268
|
+
const r = spawnSync(
|
|
269
|
+
"gh",
|
|
270
|
+
[
|
|
271
|
+
"pr",
|
|
272
|
+
"view",
|
|
273
|
+
String(ref),
|
|
274
|
+
"--json",
|
|
275
|
+
"state,headRefName,headRepository,headRepositoryOwner",
|
|
276
|
+
],
|
|
277
|
+
{ cwd, env: cwdEnv(cwd), encoding: "utf8" },
|
|
278
|
+
);
|
|
279
|
+
if (r.status !== 0) return null;
|
|
280
|
+
try {
|
|
281
|
+
const j = JSON.parse(r.stdout || "{}");
|
|
282
|
+
const owner = j.headRepositoryOwner?.login || j.headRepositoryOwner?.name || null;
|
|
283
|
+
const repo = j.headRepository?.name || null;
|
|
284
|
+
return {
|
|
285
|
+
state: typeof j.state === "string" ? j.state : "",
|
|
286
|
+
headRefName: typeof j.headRefName === "string" ? j.headRefName : null,
|
|
287
|
+
headRepoFullName: owner && repo ? `${owner}/${repo}` : null,
|
|
288
|
+
};
|
|
289
|
+
} catch {
|
|
290
|
+
return null;
|
|
291
|
+
}
|
|
292
|
+
},
|
|
192
293
|
sleep: (ms) => new Promise((r) => setTimeout(r, ms)),
|
|
193
294
|
now: () => Date.now(),
|
|
194
295
|
};
|
|
@@ -199,7 +300,13 @@ function defaultDeps() {
|
|
|
199
300
|
* Both repo and direct-folder runs use this exact callback contract so neither
|
|
200
301
|
* path can accidentally become the ungated exception.
|
|
201
302
|
*/
|
|
202
|
-
function openCodePermissionCallbacks({
|
|
303
|
+
function openCodePermissionCallbacks({
|
|
304
|
+
tool,
|
|
305
|
+
channelId,
|
|
306
|
+
threadRoot,
|
|
307
|
+
runId = null,
|
|
308
|
+
provider = "opencode",
|
|
309
|
+
}) {
|
|
203
310
|
return {
|
|
204
311
|
requestPermission: async (request) => {
|
|
205
312
|
const detail =
|
|
@@ -216,7 +323,7 @@ function openCodePermissionCallbacks({ tool, channelId, threadRoot, runId = null
|
|
|
216
323
|
channelId,
|
|
217
324
|
threadRootId: threadRoot,
|
|
218
325
|
...(runId ? { runId } : {}),
|
|
219
|
-
provider
|
|
326
|
+
provider,
|
|
220
327
|
vendorSessionId: request.sessionId,
|
|
221
328
|
vendorRequestId: request.vendorRequestId,
|
|
222
329
|
action: request.action,
|
|
@@ -241,7 +348,7 @@ function openCodePermissionCallbacks({ tool, channelId, threadRoot, runId = null
|
|
|
241
348
|
}
|
|
242
349
|
return tool("get_permission_decision", {
|
|
243
350
|
requestId,
|
|
244
|
-
provider
|
|
351
|
+
provider,
|
|
245
352
|
vendorSessionId: request.sessionId,
|
|
246
353
|
vendorRequestId: request.vendorRequestId,
|
|
247
354
|
...(failClosed ? { failClosed: true } : {}),
|
|
@@ -250,6 +357,145 @@ function openCodePermissionCallbacks({ tool, channelId, threadRoot, runId = null
|
|
|
250
357
|
};
|
|
251
358
|
}
|
|
252
359
|
|
|
360
|
+
/**
|
|
361
|
+
* The post_progress sender both run lanes use, with the 0782 stop check folded
|
|
362
|
+
* into the beat they already send.
|
|
363
|
+
*
|
|
364
|
+
* A person can press Stop on the live card from anywhere — the person who
|
|
365
|
+
* delegated the work usually is not the one at this laptop. The server records
|
|
366
|
+
* that on the run's own row and answers the NEXT heartbeat with
|
|
367
|
+
* `stopRequested: true` (plus `stoppedBy` when it knows the name). So the daemon
|
|
368
|
+
* learns it on its existing cadence: no poll tool, no second timer, nothing new
|
|
369
|
+
* to keep alive.
|
|
370
|
+
*
|
|
371
|
+
* `onStopRequested` is called at most once per sender and must never throw into
|
|
372
|
+
* the run — it is the daemon's cue to tear the process group down.
|
|
373
|
+
*
|
|
374
|
+
* @param {{ tool: Function, statusId: string, runId?: string|null,
|
|
375
|
+
* onStopRequested?: (info: {by: string|null}) => any }} o
|
|
376
|
+
*/
|
|
377
|
+
export function createProgressSender({ tool, statusId, runId = null, onStopRequested }) {
|
|
378
|
+
let inflight = Promise.resolve();
|
|
379
|
+
let stopSeen = false;
|
|
380
|
+
const takeStop = (res) => {
|
|
381
|
+
if (!res || res.stopRequested !== true) return false;
|
|
382
|
+
if (stopSeen || !onStopRequested) return true;
|
|
383
|
+
stopSeen = true;
|
|
384
|
+
try {
|
|
385
|
+
onStopRequested({ by: typeof res.stoppedBy === "string" ? res.stoppedBy : null });
|
|
386
|
+
} catch {
|
|
387
|
+
/* the stop hand-off must never break the run's teardown */
|
|
388
|
+
}
|
|
389
|
+
return true;
|
|
390
|
+
};
|
|
391
|
+
return {
|
|
392
|
+
/** @param {object} p — the emitter's snapshot. */
|
|
393
|
+
send(p) {
|
|
394
|
+
try {
|
|
395
|
+
const r = tool("post_progress", {
|
|
396
|
+
messageId: statusId,
|
|
397
|
+
...(runId ? { runId } : {}),
|
|
398
|
+
progress: p,
|
|
399
|
+
});
|
|
400
|
+
if (r && typeof r.then === "function") {
|
|
401
|
+
const done = r.then((res) => void takeStop(res), () => {});
|
|
402
|
+
inflight = Promise.all([inflight, done]).then(
|
|
403
|
+
() => {},
|
|
404
|
+
() => {},
|
|
405
|
+
);
|
|
406
|
+
}
|
|
407
|
+
} catch {
|
|
408
|
+
/* a progress send must never break the run */
|
|
409
|
+
}
|
|
410
|
+
},
|
|
411
|
+
/**
|
|
412
|
+
* Ask the server outright, and WAIT for the answer.
|
|
413
|
+
*
|
|
414
|
+
* The heartbeat above only fires when the CLI says something. A run sitting
|
|
415
|
+
* inside one silent ten-minute command emits nothing, so nothing would carry
|
|
416
|
+
* a stop home — the person would press Stop and watch their laptop keep
|
|
417
|
+
* going. This is the same call, on a timer, and it is also the authoritative
|
|
418
|
+
* pre-ship check: `state` echoes the card's CURRENT lifecycle so asking
|
|
419
|
+
* never repaints it (a stopped run's card is frozen server-side anyway).
|
|
420
|
+
*
|
|
421
|
+
* @param {"working"|"done"|"error"} [state]
|
|
422
|
+
* @returns {Promise<boolean>} true when a person has stopped this run.
|
|
423
|
+
*/
|
|
424
|
+
async poll(state = "working") {
|
|
425
|
+
try {
|
|
426
|
+
const res = await tool("post_progress", {
|
|
427
|
+
messageId: statusId,
|
|
428
|
+
...(runId ? { runId } : {}),
|
|
429
|
+
progress: { state },
|
|
430
|
+
});
|
|
431
|
+
return takeStop(res);
|
|
432
|
+
} catch {
|
|
433
|
+
// A stop check that can't reach the server is not a stop. The run
|
|
434
|
+
// continues; the next tick asks again.
|
|
435
|
+
return false;
|
|
436
|
+
}
|
|
437
|
+
},
|
|
438
|
+
/** Every dispatched write, so a caller can wait them out before settling. */
|
|
439
|
+
drain() {
|
|
440
|
+
return inflight;
|
|
441
|
+
},
|
|
442
|
+
};
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
/** How often a run asks whether it has been stopped while its CLI is silent.
|
|
446
|
+
* Well under the card's own staleness window, far above a chatty write rate. */
|
|
447
|
+
export const STOP_POLL_MS = 15_000;
|
|
448
|
+
|
|
449
|
+
/**
|
|
450
|
+
* Arm the silent-work stop poll for ONE active job. Timers are injectable so
|
|
451
|
+
* this is unit-tested on a fake clock, and the returned `stop()` must be called
|
|
452
|
+
* from the same finally that tears the run down — one timer per job, never a
|
|
453
|
+
* timer that outlives the work it was watching.
|
|
454
|
+
*
|
|
455
|
+
* @param {{ poll: () => Promise<boolean>, intervalMs?: number,
|
|
456
|
+
* setTimer?: Function, clearTimer?: Function }} o
|
|
457
|
+
*/
|
|
458
|
+
export function createStopPoller({
|
|
459
|
+
poll,
|
|
460
|
+
intervalMs = STOP_POLL_MS,
|
|
461
|
+
setTimer = (fn, ms) => {
|
|
462
|
+
const id = setInterval(fn, ms);
|
|
463
|
+
if (id && typeof id.unref === "function") id.unref();
|
|
464
|
+
return id;
|
|
465
|
+
},
|
|
466
|
+
clearTimer = (id) => clearInterval(id),
|
|
467
|
+
} = {}) {
|
|
468
|
+
let id = null;
|
|
469
|
+
let asking = false;
|
|
470
|
+
let done = false;
|
|
471
|
+
const tick = async () => {
|
|
472
|
+
// Never stack asks: a slow server must not queue a burst of stop checks.
|
|
473
|
+
if (asking || done) return;
|
|
474
|
+
asking = true;
|
|
475
|
+
try {
|
|
476
|
+
if (await poll()) {
|
|
477
|
+
done = true; // the stop is handed over once; teardown owns the rest
|
|
478
|
+
stop();
|
|
479
|
+
}
|
|
480
|
+
} catch {
|
|
481
|
+
/* a stop check must never break the run */
|
|
482
|
+
} finally {
|
|
483
|
+
asking = false;
|
|
484
|
+
}
|
|
485
|
+
};
|
|
486
|
+
function stop() {
|
|
487
|
+
if (id == null) return;
|
|
488
|
+
try {
|
|
489
|
+
clearTimer(id);
|
|
490
|
+
} catch {
|
|
491
|
+
/* ignore */
|
|
492
|
+
}
|
|
493
|
+
id = null;
|
|
494
|
+
}
|
|
495
|
+
id = setTimer(() => void tick(), intervalMs);
|
|
496
|
+
return { stop, tick };
|
|
497
|
+
}
|
|
498
|
+
|
|
253
499
|
/** The local HTTP bridge can own only a local OpenCode server. An explicit
|
|
254
500
|
* `--attach` remains on OpenCode's CLI responder, which rejects unanswered asks
|
|
255
501
|
* fail closed; taking over a remote server requires a separate authenticated
|
|
@@ -268,6 +514,210 @@ export function shouldUseRuntimePermissionBridge({
|
|
|
268
514
|
);
|
|
269
515
|
}
|
|
270
516
|
|
|
517
|
+
/** ACP transport (0759; slice 1 opencode, slice 2 cursor). Only runs the
|
|
518
|
+
* workspace already gates with runtime permissions qualify, and only when the
|
|
519
|
+
* vendor's own auto-allow escape hatches are absent — over ACP, hilos must be
|
|
520
|
+
* the one answering asks, so a run configured to never ask has nothing to gate
|
|
521
|
+
* here.
|
|
522
|
+
*
|
|
523
|
+
* 0778 removed two limits that were never about correctness. `acpTransport`
|
|
524
|
+
* now defaults ON (config.mjs) for the vendors whose adapter is proven, and a
|
|
525
|
+
* RESUMED run no longer disqualifies: both live vendors advertise ACP's
|
|
526
|
+
* `loadSession` capability, so acp-session.mjs resumes the session and keeps
|
|
527
|
+
* raising cards. Setting `acpTransport: false` still opts a workspace out. */
|
|
528
|
+
export function shouldUseAcpTransport({
|
|
529
|
+
vendor,
|
|
530
|
+
acpTransport,
|
|
531
|
+
runtimePermissions,
|
|
532
|
+
codeArgs,
|
|
533
|
+
codingCmd,
|
|
534
|
+
}) {
|
|
535
|
+
if (acpTransport !== true) return false;
|
|
536
|
+
if (runtimePermissions !== true) return false;
|
|
537
|
+
if (vendor === "opencode") {
|
|
538
|
+
return shouldUseRuntimePermissionBridge({
|
|
539
|
+
vendor,
|
|
540
|
+
runtimePermissions,
|
|
541
|
+
codeArgs,
|
|
542
|
+
codingCmd,
|
|
543
|
+
});
|
|
544
|
+
}
|
|
545
|
+
if (vendor === "cursor") {
|
|
546
|
+
// -f/--force is cursor's "allow everything" switch — the exact analog of
|
|
547
|
+
// opencode's --auto.
|
|
548
|
+
return !codeArgs.includes("-f") && !codeArgs.includes("--force");
|
|
549
|
+
}
|
|
550
|
+
return false;
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
/**
|
|
554
|
+
* The env a gated claude run gets (0777).
|
|
555
|
+
*
|
|
556
|
+
* A permission card can legitimately wait on a person for minutes — a 150s wait
|
|
557
|
+
* was verified live — so the CLI's own MCP tool timeout must not cut the ask
|
|
558
|
+
* short before hilos's run deadline does. Everything else about the env is
|
|
559
|
+
* unchanged (runCli still strips HILOS_* itself).
|
|
560
|
+
*/
|
|
561
|
+
function claudeGateEnv(cfg) {
|
|
562
|
+
const base = codingChildEnv(cfg) || process.env;
|
|
563
|
+
const budget = Math.max(60_000, Number(cfg?.runTimeoutMs) || 0) + 60_000;
|
|
564
|
+
return { ...base, MCP_TOOL_TIMEOUT: String(budget) };
|
|
565
|
+
}
|
|
566
|
+
|
|
567
|
+
/**
|
|
568
|
+
* Run claude_code through its permission-prompt seam (0777).
|
|
569
|
+
*
|
|
570
|
+
* Deliberately NOT a new transport: the ordinary argv run is untouched — same
|
|
571
|
+
* runCli, same resume args, same streaming — and the gate is two extra flags
|
|
572
|
+
* plus a loopback MCP server that lives exactly as long as the run. The prompt
|
|
573
|
+
* stays the final argument.
|
|
574
|
+
*/
|
|
575
|
+
async function runClaudeGatedCli({
|
|
576
|
+
deps,
|
|
577
|
+
cfg,
|
|
578
|
+
onGateDropped,
|
|
579
|
+
cmd,
|
|
580
|
+
codeArgs,
|
|
581
|
+
prompt,
|
|
582
|
+
cwd,
|
|
583
|
+
signal,
|
|
584
|
+
onData,
|
|
585
|
+
permissionCallbacks,
|
|
586
|
+
sessionId = "",
|
|
587
|
+
resolveSessionId,
|
|
588
|
+
log = console,
|
|
589
|
+
}) {
|
|
590
|
+
const server = await deps.startClaudePermissionServer({
|
|
591
|
+
...permissionCallbacks,
|
|
592
|
+
sessionId,
|
|
593
|
+
// A FRESH run has no session id to hand over: claude reveals its own in the
|
|
594
|
+
// init frame of its stream, which the progress emitter is already folding.
|
|
595
|
+
// Pulling from that snapshot means an ask carries the CLI's real session id
|
|
596
|
+
// without a second parser — and hilos rejects an empty one outright.
|
|
597
|
+
resolveSessionId,
|
|
598
|
+
timeoutMs: cfg.runTimeoutMs,
|
|
599
|
+
signal,
|
|
600
|
+
log,
|
|
601
|
+
});
|
|
602
|
+
try {
|
|
603
|
+
return await deps.runCli({
|
|
604
|
+
cmd,
|
|
605
|
+
args: [...codeArgs, ...server.args, prompt],
|
|
606
|
+
cwd,
|
|
607
|
+
timeoutMs: cfg.runTimeoutMs,
|
|
608
|
+
label: "coding",
|
|
609
|
+
signal,
|
|
610
|
+
env: claudeGateEnv(cfg),
|
|
611
|
+
onData,
|
|
612
|
+
// 0785 — a CLI that rejects the gate flags is retried without them. This
|
|
613
|
+
// fires BEFORE that ungated retry is spawned, so the room hears about the
|
|
614
|
+
// downgrade first rather than after the fact.
|
|
615
|
+
onPermissionGateDropped: onGateDropped,
|
|
616
|
+
});
|
|
617
|
+
} finally {
|
|
618
|
+
try {
|
|
619
|
+
await server.close();
|
|
620
|
+
} catch {
|
|
621
|
+
/* tearing the gate down must never fail a finished run */
|
|
622
|
+
}
|
|
623
|
+
}
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
/**
|
|
627
|
+
* Will this codex run take the gated transport (0785)?
|
|
628
|
+
*
|
|
629
|
+
* Answerable before the run's argv exists, because the only input that can
|
|
630
|
+
* change the answer is the operator's OWN command — every flag the daemon
|
|
631
|
+
* appends later (model, dir, image, resume, stream) is ours and none of them is
|
|
632
|
+
* the bypass tier. Which matters because the PROMPT is written before the
|
|
633
|
+
* transport is chosen, and what the prompt may claim about images depends on it.
|
|
634
|
+
*/
|
|
635
|
+
function codexRunIsGated(cfg, caps) {
|
|
636
|
+
return shouldGateCodexPermissions({
|
|
637
|
+
vendor: detectVendor(cfg?.codingCmd),
|
|
638
|
+
runtimePermissions: caps?.runtimePermissions,
|
|
639
|
+
codeArgs: String(cfg?.codingCmd || "").split(" ").filter(Boolean).slice(1),
|
|
640
|
+
});
|
|
641
|
+
}
|
|
642
|
+
|
|
643
|
+
/**
|
|
644
|
+
* Run codex through its gated transport, with an honest fallback (0785).
|
|
645
|
+
*
|
|
646
|
+
* A granted workspace ALWAYS gets `codex mcp-server` — that is what the grant
|
|
647
|
+
* means, and it is the only codex transport that can ask (`codex exec` has no
|
|
648
|
+
* approval channel at any flag combination, see codex-mcp-session.mjs). The one
|
|
649
|
+
* thing the grant can't conjure is the subcommand itself: a codex old enough
|
|
650
|
+
* not to have it never answers the MCP handshake, and before this the run just
|
|
651
|
+
* failed. Now it degrades to the plain exec argv — the ungated run every codex
|
|
652
|
+
* did before 0777, never worse.
|
|
653
|
+
*
|
|
654
|
+
* Two rules make that degrade safe. It happens ONLY on positive evidence that
|
|
655
|
+
* the subcommand is absent (codexMcpTransportUnavailable — every other
|
|
656
|
+
* pre-handshake failure is returned as the failed run it is, because a wrongly
|
|
657
|
+
* failed run is recoverable and a wrongly ungated one is not). And the room is
|
|
658
|
+
* told BEFORE the ungated child is spawned, so the warning arrives while the
|
|
659
|
+
* run can still be stopped.
|
|
660
|
+
*
|
|
661
|
+
* The exec argv is the run's real one (model, images, resume, stream flags),
|
|
662
|
+
* so the fallback is the same run the ungated lane would have made.
|
|
663
|
+
*/
|
|
664
|
+
async function runCodexGatedSession({
|
|
665
|
+
deps,
|
|
666
|
+
cfg,
|
|
667
|
+
onGateDropped,
|
|
668
|
+
cmd,
|
|
669
|
+
codeArgs,
|
|
670
|
+
prompt,
|
|
671
|
+
cwd,
|
|
672
|
+
model,
|
|
673
|
+
resumeThreadId = null,
|
|
674
|
+
signal,
|
|
675
|
+
onData,
|
|
676
|
+
onEvent,
|
|
677
|
+
permissionCallbacks,
|
|
678
|
+
}) {
|
|
679
|
+
const run = await deps.runCodexMcpSession({
|
|
680
|
+
cmd,
|
|
681
|
+
cwd,
|
|
682
|
+
prompt,
|
|
683
|
+
sandbox: codexSandboxFromArgs(codeArgs),
|
|
684
|
+
// 0785 — the resolved tier (0783) or a hand-pinned id. The gated transport
|
|
685
|
+
// took the account default before this, so a preset was silently ignored
|
|
686
|
+
// exactly where the operator was most likely to have set one.
|
|
687
|
+
model: model || null,
|
|
688
|
+
resumeThreadId,
|
|
689
|
+
timeoutMs: cfg.runTimeoutMs,
|
|
690
|
+
signal,
|
|
691
|
+
env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
|
|
692
|
+
onData,
|
|
693
|
+
onEvent,
|
|
694
|
+
...permissionCallbacks,
|
|
695
|
+
});
|
|
696
|
+
if (!codexMcpTransportUnavailable(run)) return run;
|
|
697
|
+
console.log(
|
|
698
|
+
" code → this codex has no `mcp-server`; running WITHOUT the hilos permission gate",
|
|
699
|
+
);
|
|
700
|
+
// Warn FIRST — before a single ungated command can run.
|
|
701
|
+
if (typeof onGateDropped === "function") {
|
|
702
|
+
try {
|
|
703
|
+
await onGateDropped();
|
|
704
|
+
} catch {
|
|
705
|
+
/* telling the room must never break the run */
|
|
706
|
+
}
|
|
707
|
+
}
|
|
708
|
+
const fallback = await deps.runCli({
|
|
709
|
+
cmd,
|
|
710
|
+
args: [...codeArgs, prompt],
|
|
711
|
+
cwd,
|
|
712
|
+
timeoutMs: cfg.runTimeoutMs,
|
|
713
|
+
label: "coding",
|
|
714
|
+
signal,
|
|
715
|
+
env: codingChildEnv(cfg),
|
|
716
|
+
onData,
|
|
717
|
+
});
|
|
718
|
+
return { ...fallback, permissionGateDropped: true };
|
|
719
|
+
}
|
|
720
|
+
|
|
271
721
|
async function awaitDecision({ tool, channelId, reportMessageId, cfg, deps, parentId, signal }) {
|
|
272
722
|
if (!reportMessageId) return { kind: "timeout" };
|
|
273
723
|
const deadline = deps.now() + cfg.decisionTimeoutMs;
|
|
@@ -318,7 +768,7 @@ async function linkPrFromUrl({ tool, channelId, url }) {
|
|
|
318
768
|
await tool("link_pr", { channelId, repoFullName: m[1], prNumber: Number(m[2]) }).catch(() => {});
|
|
319
769
|
}
|
|
320
770
|
|
|
321
|
-
async function applyDecision({ decision, repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, existingPrUrl, settleId, runId }) {
|
|
771
|
+
async function applyDecision({ decision, repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, existingPrUrl, settleId, runId, usageArgs = {} }) {
|
|
322
772
|
const tag = requesterTag(requester);
|
|
323
773
|
const lead = tag ? `${tag} — ` : "";
|
|
324
774
|
// `parentId` here is the run's thread root. Terminal outcomes ask the server
|
|
@@ -377,13 +827,21 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
|
|
|
377
827
|
: `${lead}pushed \`${branch}\`. Open a PR manually — \`gh\` failed.`,
|
|
378
828
|
prUrl: pr.ok && pr.url ? pr.url : undefined,
|
|
379
829
|
caveats: pr.ok ? [] : [`gh pr create failed: ${(pr.stderr || "").trim().slice(0, 200)}`],
|
|
830
|
+
// 0787 — what the run cost, when the caller had a receipt to hand over.
|
|
831
|
+
// Ungated runs land here with the whole run's usage; a gated one already
|
|
832
|
+
// spent it on the proposal card and passes {}.
|
|
833
|
+
...usageArgs,
|
|
380
834
|
};
|
|
381
|
-
|
|
835
|
+
// 0792 — keep the id of the card this run settled onto, so the caller can
|
|
836
|
+
// mark it with the run's transcript. `settleId` is that card when the run
|
|
837
|
+
// streamed onto a live status message; otherwise it is the fresh report.
|
|
838
|
+
const reportRes = await tool(
|
|
382
839
|
"post_report",
|
|
383
840
|
settleId
|
|
384
841
|
? { ...reportArgs, messageId: settleId, broadcast: true }
|
|
385
842
|
: { ...reportArgs, parentId, broadcast: true },
|
|
386
843
|
);
|
|
844
|
+
const reportMsgId = settleId || reportRes?.messageId || null;
|
|
387
845
|
// Work produced a PR → attach it to the channel so its live pill shows up.
|
|
388
846
|
await linkPrFromUrl({ tool, channelId, url: pr.ok ? pr.url : null });
|
|
389
847
|
// Bind the PR to the thread's run (0279/0280) so a follow-up continues it.
|
|
@@ -393,10 +851,13 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
|
|
|
393
851
|
runId,
|
|
394
852
|
...(pr.ok && pr.url ? { prUrl: pr.url } : {}),
|
|
395
853
|
status: "awaiting_review",
|
|
396
|
-
|
|
854
|
+
// 0792: record the card even when it was a fresh report rather than a
|
|
855
|
+
// settle — it is the run's card either way, and the transcript upload
|
|
856
|
+
// falls back to this exact field when it is not handed an id.
|
|
857
|
+
...(reportMsgId ? { reportMsgId } : {}),
|
|
397
858
|
}).catch(() => {});
|
|
398
859
|
}
|
|
399
|
-
return { status: "pushed", branch, prUrl: pr.ok ? pr.url : null };
|
|
860
|
+
return { status: "pushed", branch, prUrl: pr.ok ? pr.url : null, reportMsgId };
|
|
400
861
|
}
|
|
401
862
|
|
|
402
863
|
if (decision.kind === "rejected") {
|
|
@@ -431,55 +892,100 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
|
|
|
431
892
|
* The coding CLI committed the work itself (autonomous run with skip-permissions
|
|
432
893
|
* in a repo whose docs prescribe a commit+PR flow), so the working tree is clean
|
|
433
894
|
* and the daemon's diff is empty. Don't claim "no changes": adopt what it did —
|
|
434
|
-
* push
|
|
435
|
-
* open one, and report it.
|
|
895
|
+
* push its HEAD under the daemon-owned task branch, reuse the PR it opened or
|
|
896
|
+
* open one, and report it. Keeping the intended head matters: a child that
|
|
897
|
+
* switched to `main` would otherwise make `gh` try to open main → main. Used
|
|
898
|
+
* only when NOT gated (bias-to-action).
|
|
436
899
|
*/
|
|
437
|
-
async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, settleId, runId }) {
|
|
900
|
+
async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, settleId, runId, usageArgs = {} }) {
|
|
438
901
|
const tag = requesterTag(requester);
|
|
439
902
|
const lead = tag ? `${tag} — ` : "";
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
const
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
let
|
|
903
|
+
const currentBranch =
|
|
904
|
+
(deps.git(repoPath, ["symbolic-ref", "--quiet", "--short", "HEAD"]).stdout || "").trim();
|
|
905
|
+
const ship = selfDrivenShipPlan(currentBranch, branch);
|
|
906
|
+
const push = deps.git(repoPath, ship.pushArgs);
|
|
907
|
+
let prUrl = deps.findPR ? deps.findPR(repoPath, ship.headBranch) : null;
|
|
908
|
+
let prAttempt = null;
|
|
446
909
|
if (!prUrl && push.status === 0) {
|
|
447
|
-
const { title, body } = prTitleBody(task,
|
|
448
|
-
|
|
449
|
-
|
|
910
|
+
const { title, body } = prTitleBody(task, ship.headBranch);
|
|
911
|
+
prAttempt = deps.openPR(repoPath, {
|
|
912
|
+
title,
|
|
913
|
+
body,
|
|
914
|
+
branch: ship.headBranch,
|
|
915
|
+
base: cfg.defaultBranch,
|
|
916
|
+
});
|
|
917
|
+
prUrl = prAttempt.ok && prAttempt.url ? prAttempt.url : null;
|
|
450
918
|
}
|
|
451
|
-
const { title } = prTitleBody(task,
|
|
919
|
+
const { title } = prTitleBody(task, ship.headBranch);
|
|
920
|
+
const recoveredNote = ship.recoveredFrom
|
|
921
|
+
? ` The child had switched to \`${ship.recoveredFrom}\`; hilos recovered its HEAD onto \`${ship.headBranch}\`.`
|
|
922
|
+
: "";
|
|
923
|
+
const prFailure =
|
|
924
|
+
prAttempt && !prUrl
|
|
925
|
+
? oneLine(
|
|
926
|
+
(prAttempt.stderr || prAttempt.stdout || "gh did not return a pull request URL").trim(),
|
|
927
|
+
200,
|
|
928
|
+
)
|
|
929
|
+
: "";
|
|
452
930
|
// Seamless single card (0289): settle onto the live status message when one was
|
|
453
931
|
// streaming this run; the workspace setting decides channel visibility.
|
|
454
932
|
// Otherwise post a fresh final report.
|
|
455
933
|
const reportArgs = {
|
|
456
934
|
channelId,
|
|
457
|
-
title:
|
|
935
|
+
title: prUrl
|
|
936
|
+
? `Shipped: ${title}`
|
|
937
|
+
: push.status === 0
|
|
938
|
+
? `PR failed: ${title}`
|
|
939
|
+
: `Push failed: ${title}`,
|
|
458
940
|
summary: prUrl
|
|
459
|
-
? `${lead}the coding agent committed
|
|
941
|
+
? `${lead}the coding agent committed the work itself and opened a pull request from \`${ship.headBranch}\`.${recoveredNote}`
|
|
460
942
|
: push.status === 0
|
|
461
|
-
? `${lead}the coding agent committed and pushed \`${
|
|
462
|
-
: `${lead}the coding agent committed
|
|
943
|
+
? `${lead}the coding agent committed and pushed \`${ship.headBranch}\`, but GitHub rejected PR creation.${recoveredNote}`
|
|
944
|
+
: `${lead}the coding agent committed locally, but pushing \`${ship.headBranch}\` failed.${recoveredNote}`,
|
|
463
945
|
prUrl: prUrl || undefined,
|
|
464
|
-
caveats:
|
|
946
|
+
caveats:
|
|
947
|
+
push.status !== 0
|
|
948
|
+
? [`git push failed: ${(push.stderr || "").trim().slice(0, 200)}`]
|
|
949
|
+
: prFailure
|
|
950
|
+
? [`gh pr create failed: ${prFailure}`]
|
|
951
|
+
: [],
|
|
952
|
+
// 0787 — an autonomous run's receipt rides its one and only report.
|
|
953
|
+
...usageArgs,
|
|
465
954
|
};
|
|
466
|
-
await tool(
|
|
955
|
+
const reportRes = await tool(
|
|
467
956
|
"post_report",
|
|
468
957
|
settleId
|
|
469
958
|
? { ...reportArgs, messageId: settleId, broadcast: true }
|
|
470
959
|
: { ...reportArgs, parentId, broadcast: true },
|
|
471
960
|
);
|
|
961
|
+
// 0792 — the card this run settled onto, for the transcript stamp.
|
|
962
|
+
const reportMsgId = settleId || reportRes?.messageId || null;
|
|
472
963
|
if (prUrl) await linkPrFromUrl({ tool, channelId, url: prUrl });
|
|
473
964
|
// Bind the PR to the thread's run (0279/0280). Best-effort; only when recorded.
|
|
474
965
|
if (runId) {
|
|
475
966
|
await tool("update_run", {
|
|
476
967
|
runId,
|
|
477
968
|
...(prUrl ? { prUrl } : {}),
|
|
478
|
-
status: "awaiting_review",
|
|
479
|
-
...(
|
|
969
|
+
status: prUrl ? "awaiting_review" : "failed",
|
|
970
|
+
...(!prUrl
|
|
971
|
+
? {
|
|
972
|
+
reason:
|
|
973
|
+
push.status !== 0
|
|
974
|
+
? oneLine((push.stderr || "git push failed").trim(), 300)
|
|
975
|
+
: prFailure || "GitHub did not return a pull request URL",
|
|
976
|
+
}
|
|
977
|
+
: {}),
|
|
978
|
+
...(reportMsgId ? { reportMsgId } : {}),
|
|
480
979
|
}).catch(() => {});
|
|
481
980
|
}
|
|
482
|
-
|
|
981
|
+
const status =
|
|
982
|
+
push.status !== 0 ? "commit-local" : prUrl ? "pushed" : "pr-failed";
|
|
983
|
+
return {
|
|
984
|
+
status,
|
|
985
|
+
branch: ship.headBranch,
|
|
986
|
+
prUrl,
|
|
987
|
+
reportMsgId,
|
|
988
|
+
};
|
|
483
989
|
}
|
|
484
990
|
|
|
485
991
|
// Sentinel the router model emits when the latest message is a request to change
|
|
@@ -580,16 +1086,21 @@ async function routeIntent({ name, repoFullName, transcript, workspaceMemory, cf
|
|
|
580
1086
|
* agent, edit files now" — led by the router's distilled `brief` (what to build),
|
|
581
1087
|
* with the conversation included only as background. Falls back to the raw
|
|
582
1088
|
* mention when there's no brief.
|
|
583
|
-
* @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, repoFullName?: string }} [o]
|
|
1089
|
+
* @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, repoFullName?: string, images?: {path: string, name: string, type: string}[], imagesReadable?: boolean }} [o]
|
|
584
1090
|
*/
|
|
585
1091
|
export function codeTaskPrompt(o) {
|
|
586
|
-
const { message, context, brief, repoFullName } = o || {};
|
|
1092
|
+
const { message, context, brief, repoFullName, images, imagesReadable } = o || {};
|
|
587
1093
|
const task = (brief && brief.trim()) || String(message?.body || "").trim();
|
|
588
1094
|
const transcript = context?.transcript?.trim();
|
|
589
1095
|
const where = repoFullName ? ` in the git repository ${repoFullName}` : "";
|
|
590
1096
|
let p =
|
|
591
1097
|
`You are a coding agent working${where}. Implement the following by EDITING FILES now — ` +
|
|
592
1098
|
`make the changes directly, do not just describe them, do not ask questions:\n\n${task}`;
|
|
1099
|
+
// 0779: the screenshots the request was about, already on this machine. Right
|
|
1100
|
+
// under the task, because "fix this spacing" only means something next to the
|
|
1101
|
+
// picture of the spacing.
|
|
1102
|
+
const imageNote = imagePromptNote(images, { readable: imagesReadable });
|
|
1103
|
+
if (imageNote) p += `\n\n${imageNote}`;
|
|
593
1104
|
// The daemon owns git: it stages, commits, pushes, and opens the PR after the
|
|
594
1105
|
// CLI finishes. An autonomous CLI run with skip-permissions inside a repo whose
|
|
595
1106
|
// docs prescribe a commit+PR workflow will otherwise do all of that itself,
|
|
@@ -612,16 +1123,19 @@ export function codeTaskPrompt(o) {
|
|
|
612
1123
|
* imperative "edit files now" framing as codeTaskPrompt, but it tells the agent
|
|
613
1124
|
* its edits land straight in the folder and bans git/gh (there's nothing for the
|
|
614
1125
|
* daemon to commit; the changes ARE the deliverable).
|
|
615
|
-
* @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, folderPath?: string }} [o]
|
|
1126
|
+
* @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, folderPath?: string, images?: {path: string, name: string, type: string}[], imagesReadable?: boolean }} [o]
|
|
616
1127
|
*/
|
|
617
1128
|
export function folderTaskPrompt(o) {
|
|
618
|
-
const { message, context, brief, folderPath } = o || {};
|
|
1129
|
+
const { message, context, brief, folderPath, images, imagesReadable } = o || {};
|
|
619
1130
|
const task = (brief && brief.trim()) || String(message?.body || "").trim();
|
|
620
1131
|
const where = folderPath ? ` in the local folder ${folderPath}` : "";
|
|
621
1132
|
let p =
|
|
622
1133
|
`You are a coding agent working directly${where}. Implement the following by EDITING ` +
|
|
623
1134
|
`FILES now — make the changes directly in this folder, do not just describe them, do not ` +
|
|
624
1135
|
`ask questions:\n\n${task}`;
|
|
1136
|
+
// 0779 — same image handoff as the repo path.
|
|
1137
|
+
const imageNote = imagePromptNote(images, { readable: imagesReadable });
|
|
1138
|
+
if (imageNote) p += `\n\n${imageNote}`;
|
|
625
1139
|
p +=
|
|
626
1140
|
`\n\nIMPORTANT: your edits apply DIRECTLY to the user's folder — there is no branch, no ` +
|
|
627
1141
|
`commit, and no pull request. Do NOT run git; do NOT stage, commit, push, create branches, ` +
|
|
@@ -664,7 +1178,17 @@ async function fetchContext({ channelId, tool, parentId }) {
|
|
|
664
1178
|
}));
|
|
665
1179
|
rows = messages;
|
|
666
1180
|
}
|
|
667
|
-
|
|
1181
|
+
// 0779: the rows carry resolved `attachments`, and the body carries the
|
|
1182
|
+
// unresolvable `` pill token. Flatten the token and
|
|
1183
|
+
// name the file, so the daemon's chat replies and routing decisions stop
|
|
1184
|
+
// being handed a reference no model can follow.
|
|
1185
|
+
return {
|
|
1186
|
+
rows,
|
|
1187
|
+
transcript: rows
|
|
1188
|
+
.map((m) => `${m.author}: ${renderAttachmentLine(m.body, m.attachments)}`)
|
|
1189
|
+
.join("\n")
|
|
1190
|
+
.slice(-6000),
|
|
1191
|
+
};
|
|
668
1192
|
}
|
|
669
1193
|
|
|
670
1194
|
/**
|
|
@@ -1242,9 +1766,26 @@ async function handleFolderDeploy({
|
|
|
1242
1766
|
* live), request-changes = re-run in place (bounded), reject = revert the run's
|
|
1243
1767
|
* own changes (git repos only).
|
|
1244
1768
|
*/
|
|
1245
|
-
async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps, signal, parentId, folderPath, brief, workspaceMemory, context }) {
|
|
1769
|
+
async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps, signal, parentId, folderPath, brief, workspaceMemory, context, onStopRequested }) {
|
|
1246
1770
|
void me;
|
|
1247
1771
|
const git = deps.git;
|
|
1772
|
+
// 0779 — screenshots the poll loop already pulled to a temp dir at pickup.
|
|
1773
|
+
// The prompt names them; the argv carries them for a vendor that takes one.
|
|
1774
|
+
const localImages = Array.isArray(message?.images) ? message.images : [];
|
|
1775
|
+
const folderImagePromptArgs = localImages.length
|
|
1776
|
+
? {
|
|
1777
|
+
images: localImages,
|
|
1778
|
+
imagesReadable: imagesNeedReading(detectVendor(cfg.codingCmd), {
|
|
1779
|
+
gated: codexRunIsGated(cfg, caps),
|
|
1780
|
+
}),
|
|
1781
|
+
}
|
|
1782
|
+
: {};
|
|
1783
|
+
// 0785 — one line per run, whichever CLI turns out to be ungateable.
|
|
1784
|
+
const noticeUngatedRun = createUngatedRunNotice({
|
|
1785
|
+
// Read at post time, not now: the thread root is the ack this run posts.
|
|
1786
|
+
post: (body) => tool("post_message", { channelId, parentId: threadRoot, body }),
|
|
1787
|
+
log: console,
|
|
1788
|
+
});
|
|
1248
1789
|
const tag = requesterTag(message.author);
|
|
1249
1790
|
const lead = tag ? `${tag} — ` : "";
|
|
1250
1791
|
|
|
@@ -1291,6 +1832,33 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1291
1832
|
// supports it, else the edit-in-place heartbeat, else nothing. Kept across
|
|
1292
1833
|
// request-changes re-runs on the same status message.
|
|
1293
1834
|
let progressId = null;
|
|
1835
|
+
// 0787 — what the run cost, folded across every round of the revise loop and
|
|
1836
|
+
// taken at each report so each round's spend is reported exactly once. The
|
|
1837
|
+
// model id the stream names wins; `resolvedModelId` is the spawn-time
|
|
1838
|
+
// fallback for a CLI (codex) whose stream never says.
|
|
1839
|
+
const runUsage = createUsageFold();
|
|
1840
|
+
const cliVendor = detectVendor(cfg.codingCmd);
|
|
1841
|
+
let resolvedModelId = pinnedModelId(cfg.codingCmd);
|
|
1842
|
+
let lastTerminalState = "done";
|
|
1843
|
+
/**
|
|
1844
|
+
* Accounting-only settlement (0787). Plenty of terminal exits produce no
|
|
1845
|
+
* report at all — nothing changed, the CLI never started, a person pressed
|
|
1846
|
+
* Stop — and the tokens those runs spent were being dropped on the floor.
|
|
1847
|
+
* `post_progress` is the terminal call the daemon already makes on every one
|
|
1848
|
+
* of them, it is already run-scoped, and it already carries `runId`, so the
|
|
1849
|
+
* spend rides home on a call that was happening anyway rather than on a new
|
|
1850
|
+
* tool. Only ever the DELTA, so a later report cannot re-report it; a card is
|
|
1851
|
+
* required because there is no other message to hang a terminal update on.
|
|
1852
|
+
*/
|
|
1853
|
+
const settleRunUsage = async () => {
|
|
1854
|
+
const args = reportUsageArgs(runUsage, { vendor: cliVendor, modelId: resolvedModelId });
|
|
1855
|
+
if (!args.usage || !progressId) return;
|
|
1856
|
+
await tool("post_progress", {
|
|
1857
|
+
messageId: progressId,
|
|
1858
|
+
progress: { state: lastTerminalState },
|
|
1859
|
+
usage: args.usage,
|
|
1860
|
+
}).catch(() => {});
|
|
1861
|
+
};
|
|
1294
1862
|
const folderWorking = (elapsedMs, lastLine) => {
|
|
1295
1863
|
const base = `Working in \`${folderPath}\` — ${fmtElapsed(elapsedMs)} elapsed. I'll report what changed when it's done.`;
|
|
1296
1864
|
const tail = oneLine(lastLine);
|
|
@@ -1305,7 +1873,10 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1305
1873
|
const streamOn = Boolean(caps.postProgress);
|
|
1306
1874
|
const streamArgs = streamOn ? codeStreamArgs(vendor) : [];
|
|
1307
1875
|
let emitter = null;
|
|
1308
|
-
|
|
1876
|
+
// Every dispatched progress write, so the settle can wait them out (0289).
|
|
1877
|
+
let drainProgress = async () => {};
|
|
1878
|
+
// 0782 — the silent-work stop poll, torn down in the same finally as the run.
|
|
1879
|
+
let stopPoller = null;
|
|
1309
1880
|
let stopHeartbeat = () => {};
|
|
1310
1881
|
let lastLine = "";
|
|
1311
1882
|
// With stream args the CLI's stdout is NDJSON events, not prose — parse it and
|
|
@@ -1317,6 +1888,11 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1317
1888
|
const foldResultEvents = (events) => {
|
|
1318
1889
|
for (const ev of events || []) {
|
|
1319
1890
|
if (ev && ev.t === "result" && ev.summary) resultText = ev.summary;
|
|
1891
|
+
// 0787 — read the numbers off the SAME parse the summary comes from.
|
|
1892
|
+
// This parser exists whenever stream args do, which is a superset of the
|
|
1893
|
+
// cases where a progress emitter exists, so folder usage never depends
|
|
1894
|
+
// on the status card having been posted.
|
|
1895
|
+
runUsage.push(ev);
|
|
1320
1896
|
}
|
|
1321
1897
|
};
|
|
1322
1898
|
if (streamOn) {
|
|
@@ -1330,21 +1906,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1330
1906
|
}
|
|
1331
1907
|
const statusId = progressId;
|
|
1332
1908
|
if (statusId) {
|
|
1909
|
+
// 0782 — a folder run has no runs row, so the stop check is scoped to
|
|
1910
|
+
// the card's thread by the server; it degrades to "no stop" quietly.
|
|
1911
|
+
const sender = createProgressSender({ tool, statusId, onStopRequested });
|
|
1333
1912
|
emitter = createProgressEmitter({
|
|
1334
1913
|
parser: makeStreamParser(vendor),
|
|
1335
1914
|
now: deps.now,
|
|
1336
1915
|
throttleMs: cfg.progressMs,
|
|
1337
|
-
send:
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
|
|
1342
|
-
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
/* a progress send must never break the run */
|
|
1346
|
-
}
|
|
1347
|
-
},
|
|
1916
|
+
send: sender.send,
|
|
1917
|
+
});
|
|
1918
|
+
drainProgress = () => sender.drain();
|
|
1919
|
+
// The heartbeat only fires when the CLI speaks. This asks anyway, so a
|
|
1920
|
+
// run inside one long silent command is still stoppable.
|
|
1921
|
+
stopPoller = createStopPoller({
|
|
1922
|
+
poll: () => sender.poll("working"),
|
|
1923
|
+
intervalMs: cfg.stopPollMs || STOP_POLL_MS,
|
|
1348
1924
|
});
|
|
1349
1925
|
}
|
|
1350
1926
|
} else if (caps.editMessage && cfg.heartbeatMs > 0) {
|
|
@@ -1377,14 +1953,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1377
1953
|
let run;
|
|
1378
1954
|
// Model preset (0504): same run-time resolution as the repo path.
|
|
1379
1955
|
const modelArgs = await modelArgsFor(cfg, vendor);
|
|
1956
|
+
// 0787: the id we actually handed the CLI, kept for the usage receipt when
|
|
1957
|
+
// the vendor's own stream never names a model.
|
|
1958
|
+
if (modelArgs[0] === "--model" && modelArgs[1]) resolvedModelId = modelArgs[1];
|
|
1380
1959
|
// Project pin (0608): opencode reads its project from PWD, so without this
|
|
1381
1960
|
// a folder run could edit the daemon's launch directory instead of the
|
|
1382
1961
|
// folder the channel is linked to. [] for every other vendor.
|
|
1383
1962
|
const dirArgs = codeDirArgs(vendor, folderPath, cfg.codingCmd);
|
|
1963
|
+
// 0779: [] for every vendor without a verified image flag — their argv is
|
|
1964
|
+
// byte-identical to before, and the prompt note still names the files.
|
|
1965
|
+
const imageArgs = codeImageArgs(vendor, localImages);
|
|
1384
1966
|
const codeArgs = [
|
|
1385
1967
|
...parts.slice(1),
|
|
1386
1968
|
...modelArgs,
|
|
1387
1969
|
...dirArgs,
|
|
1970
|
+
...imageArgs,
|
|
1388
1971
|
...streamArgs,
|
|
1389
1972
|
];
|
|
1390
1973
|
const handleCliData = (c) => {
|
|
@@ -1405,14 +1988,78 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1405
1988
|
}
|
|
1406
1989
|
}
|
|
1407
1990
|
};
|
|
1991
|
+
// The structured transports (ACP, codex's MCP server) produce events
|
|
1992
|
+
// directly — there is no NDJSON on stdout for `resultParser` to read, so
|
|
1993
|
+
// this is where THEIR usage joins the fold (0787). A stdout run never
|
|
1994
|
+
// reaches here, and a structured run never reaches resultParser, so the two
|
|
1995
|
+
// sources can't double-count the same tokens.
|
|
1996
|
+
const handleCliEvent = (ev) => {
|
|
1997
|
+
try {
|
|
1998
|
+
emitter?.foldEvent(ev);
|
|
1999
|
+
} catch {
|
|
2000
|
+
/* a progress fold must never break the run */
|
|
2001
|
+
}
|
|
2002
|
+
try {
|
|
2003
|
+
runUsage.push(ev);
|
|
2004
|
+
} catch {
|
|
2005
|
+
/* accounting must never break the run */
|
|
2006
|
+
}
|
|
2007
|
+
};
|
|
1408
2008
|
try {
|
|
1409
|
-
const
|
|
2009
|
+
const useAcpTransport = shouldUseAcpTransport({
|
|
1410
2010
|
vendor,
|
|
2011
|
+
acpTransport: cfg.acpTransport,
|
|
1411
2012
|
runtimePermissions: caps.runtimePermissions,
|
|
1412
2013
|
codeArgs,
|
|
1413
2014
|
codingCmd: cfg.codingCmd,
|
|
1414
2015
|
});
|
|
1415
|
-
|
|
2016
|
+
const useRuntimePermissionBridge =
|
|
2017
|
+
!useAcpTransport &&
|
|
2018
|
+
shouldUseRuntimePermissionBridge({
|
|
2019
|
+
vendor,
|
|
2020
|
+
runtimePermissions: caps.runtimePermissions,
|
|
2021
|
+
codeArgs,
|
|
2022
|
+
codingCmd: cfg.codingCmd,
|
|
2023
|
+
});
|
|
2024
|
+
// 0777: the two vendors that had no gate at all. Neither is an ACP
|
|
2025
|
+
// adapter — each CLI turned out to have its own native seam (see the
|
|
2026
|
+
// module headers), so both compose with everything already here.
|
|
2027
|
+
const gateCodexPermissions =
|
|
2028
|
+
!useAcpTransport &&
|
|
2029
|
+
!useRuntimePermissionBridge &&
|
|
2030
|
+
shouldGateCodexPermissions({
|
|
2031
|
+
vendor,
|
|
2032
|
+
runtimePermissions: caps.runtimePermissions,
|
|
2033
|
+
codeArgs,
|
|
2034
|
+
});
|
|
2035
|
+
const gateClaudePermissions =
|
|
2036
|
+
!useAcpTransport &&
|
|
2037
|
+
!useRuntimePermissionBridge &&
|
|
2038
|
+
!gateCodexPermissions &&
|
|
2039
|
+
shouldGateClaudePermissions({
|
|
2040
|
+
vendor,
|
|
2041
|
+
runtimePermissions: caps.runtimePermissions,
|
|
2042
|
+
codeArgs,
|
|
2043
|
+
});
|
|
2044
|
+
if (useAcpTransport) {
|
|
2045
|
+
run = await deps.runAcpSession({
|
|
2046
|
+
cmd: parts[0],
|
|
2047
|
+
vendor,
|
|
2048
|
+
cwd: folderPath,
|
|
2049
|
+
prompt: memoryPreamble(workspaceMemory) + promptText,
|
|
2050
|
+
timeoutMs: cfg.runTimeoutMs,
|
|
2051
|
+
signal,
|
|
2052
|
+
env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
|
|
2053
|
+
onData: handleCliData,
|
|
2054
|
+
onEvent: handleCliEvent,
|
|
2055
|
+
...openCodePermissionCallbacks({
|
|
2056
|
+
tool,
|
|
2057
|
+
channelId,
|
|
2058
|
+
threadRoot,
|
|
2059
|
+
provider: vendor,
|
|
2060
|
+
}),
|
|
2061
|
+
});
|
|
2062
|
+
} else if (useRuntimePermissionBridge) {
|
|
1416
2063
|
run = await deps.runOpenCodeHttpSession({
|
|
1417
2064
|
cmd: parts[0],
|
|
1418
2065
|
args: codeArgs,
|
|
@@ -1428,6 +2075,45 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1428
2075
|
threadRoot,
|
|
1429
2076
|
}),
|
|
1430
2077
|
});
|
|
2078
|
+
} else if (gateCodexPermissions) {
|
|
2079
|
+
run = await runCodexGatedSession({
|
|
2080
|
+
deps,
|
|
2081
|
+
cfg,
|
|
2082
|
+
onGateDropped: () => noticeUngatedRun(parts[0]),
|
|
2083
|
+
cmd: parts[0],
|
|
2084
|
+
codeArgs,
|
|
2085
|
+
cwd: folderPath,
|
|
2086
|
+
prompt: memoryPreamble(workspaceMemory) + promptText,
|
|
2087
|
+
model: resolvedModelId,
|
|
2088
|
+
signal,
|
|
2089
|
+
onData: handleCliData,
|
|
2090
|
+
onEvent: handleCliEvent,
|
|
2091
|
+
permissionCallbacks: openCodePermissionCallbacks({
|
|
2092
|
+
tool,
|
|
2093
|
+
channelId,
|
|
2094
|
+
threadRoot,
|
|
2095
|
+
provider: vendor,
|
|
2096
|
+
}),
|
|
2097
|
+
});
|
|
2098
|
+
} else if (gateClaudePermissions) {
|
|
2099
|
+
run = await runClaudeGatedCli({
|
|
2100
|
+
deps,
|
|
2101
|
+
cfg,
|
|
2102
|
+
onGateDropped: () => noticeUngatedRun(parts[0]),
|
|
2103
|
+
cmd: parts[0],
|
|
2104
|
+
codeArgs,
|
|
2105
|
+
prompt: memoryPreamble(workspaceMemory) + promptText,
|
|
2106
|
+
cwd: folderPath,
|
|
2107
|
+
signal,
|
|
2108
|
+
onData: handleCliData,
|
|
2109
|
+
resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
|
|
2110
|
+
permissionCallbacks: openCodePermissionCallbacks({
|
|
2111
|
+
tool,
|
|
2112
|
+
channelId,
|
|
2113
|
+
threadRoot,
|
|
2114
|
+
provider: vendor,
|
|
2115
|
+
}),
|
|
2116
|
+
});
|
|
1431
2117
|
} else {
|
|
1432
2118
|
run = await deps.runCli({
|
|
1433
2119
|
cmd: parts[0],
|
|
@@ -1443,10 +2129,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1443
2129
|
onData: handleCliData,
|
|
1444
2130
|
});
|
|
1445
2131
|
}
|
|
2132
|
+
// 0785 backstop. The notice normally goes out from `onGateDropped`,
|
|
2133
|
+
// BEFORE the ungated child is spawned — the room has to be warned while
|
|
2134
|
+
// the run can still be stopped, not told afterwards what it already did.
|
|
2135
|
+
// This catches a degrade that reached us without firing that hook; the
|
|
2136
|
+
// latch makes it a no-op in the ordinary case.
|
|
2137
|
+
if (run?.permissionGateDropped) await noticeUngatedRun(parts[0]);
|
|
1446
2138
|
} finally {
|
|
1447
2139
|
stopHeartbeat();
|
|
2140
|
+
stopPoller?.stop();
|
|
1448
2141
|
if (emitter) {
|
|
1449
2142
|
const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
|
|
2143
|
+
// Remembered for the accounting-only settlement (0787): an exit with no
|
|
2144
|
+
// report re-sends this same terminal state, so the card is never
|
|
2145
|
+
// repainted into something it wasn't.
|
|
2146
|
+
lastTerminalState = errored ? "error" : "done";
|
|
1450
2147
|
let reason = "";
|
|
1451
2148
|
if (errored && run && !run.aborted) {
|
|
1452
2149
|
const stderrTail = oneLine((run.stderr || "").trim().split("\n").slice(-3).join(" "), 200);
|
|
@@ -1461,7 +2158,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1461
2158
|
/* ignore */
|
|
1462
2159
|
}
|
|
1463
2160
|
try {
|
|
1464
|
-
await
|
|
2161
|
+
await drainProgress();
|
|
1465
2162
|
} catch {
|
|
1466
2163
|
/* a drain failure must never break the run */
|
|
1467
2164
|
}
|
|
@@ -1529,7 +2226,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1529
2226
|
};
|
|
1530
2227
|
|
|
1531
2228
|
// --- Run ---
|
|
1532
|
-
let runResult = await runFolderCli(folderTaskPrompt({ message, context, brief, folderPath }));
|
|
2229
|
+
let runResult = await runFolderCli(folderTaskPrompt({ message, context, brief, folderPath, ...folderImagePromptArgs }));
|
|
1533
2230
|
if (runResult.aborted) {
|
|
1534
2231
|
await tool("post_message", {
|
|
1535
2232
|
channelId,
|
|
@@ -1537,6 +2234,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1537
2234
|
broadcast: Boolean(threadRoot),
|
|
1538
2235
|
body: `Stopped — I left \`${folderPath}\` as it was.`,
|
|
1539
2236
|
});
|
|
2237
|
+
await settleRunUsage(); // a stopped run still spent tokens (0787)
|
|
1540
2238
|
return { status: "cancelled" };
|
|
1541
2239
|
}
|
|
1542
2240
|
|
|
@@ -1559,6 +2257,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1559
2257
|
body = `The run didn't finish${why}${tail}. ${left} — mention me to retry.`;
|
|
1560
2258
|
}
|
|
1561
2259
|
await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body });
|
|
2260
|
+
await settleRunUsage(); // no report on this exit — settle the spend anyway (0787)
|
|
1562
2261
|
return { status: "run-failed" };
|
|
1563
2262
|
}
|
|
1564
2263
|
if (!runResult.failed && isGit && !hasChanges(changed)) {
|
|
@@ -1568,13 +2267,22 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1568
2267
|
broadcast: Boolean(threadRoot),
|
|
1569
2268
|
body: `The run finished but nothing changed in \`${folderPath}\`. Mention me to try a different approach.`,
|
|
1570
2269
|
});
|
|
2270
|
+
await settleRunUsage(); // "nothing changed" is not "nothing spent" (0787)
|
|
1571
2271
|
return { status: "no-changes" };
|
|
1572
2272
|
}
|
|
1573
2273
|
|
|
1574
2274
|
// --- Report card + decision loop ---
|
|
1575
2275
|
const postFolderReport = async (rr, ch) => {
|
|
1576
2276
|
const report = buildReport(rr, ch);
|
|
1577
|
-
const res = await tool("post_report", {
|
|
2277
|
+
const res = await tool("post_report", {
|
|
2278
|
+
channelId,
|
|
2279
|
+
parentId: threadRoot,
|
|
2280
|
+
broadcast: true,
|
|
2281
|
+
...report,
|
|
2282
|
+
// 0787 — a folder run has no runs row, so the receipt rides the card and
|
|
2283
|
+
// the ledger row lands unattached to a run. Still the honest number.
|
|
2284
|
+
...reportUsageArgs(runUsage, { vendor: cliVendor, modelId: resolvedModelId }),
|
|
2285
|
+
});
|
|
1578
2286
|
return res?.messageId ?? null;
|
|
1579
2287
|
};
|
|
1580
2288
|
let reportMessageId = await postFolderReport(runResult, changed);
|
|
@@ -1589,11 +2297,12 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1589
2297
|
body: `Revising in \`${folderPath}\` with your feedback${decision.note ? `: ${decision.note}` : ""} (round ${round + 1}/${maxRounds}).`,
|
|
1590
2298
|
});
|
|
1591
2299
|
runResult = await runFolderCli(
|
|
1592
|
-
folderTaskPrompt({ message, context, brief, folderPath }) +
|
|
2300
|
+
folderTaskPrompt({ message, context, brief, folderPath, ...folderImagePromptArgs }) +
|
|
1593
2301
|
`\n\nReviewer feedback to address: ${decision.note || "(see the channel)"}`,
|
|
1594
2302
|
);
|
|
1595
2303
|
if (runResult.aborted) {
|
|
1596
2304
|
await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body: `Stopped — I left \`${folderPath}\` as it was.` });
|
|
2305
|
+
await settleRunUsage();
|
|
1597
2306
|
return { status: "cancelled" };
|
|
1598
2307
|
}
|
|
1599
2308
|
changed = computeChanged();
|
|
@@ -1668,6 +2377,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
|
|
|
1668
2377
|
|
|
1669
2378
|
if (decision.kind === "cancelled") {
|
|
1670
2379
|
await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body: `Stopped — the changes so far are still in \`${folderPath}\`.` });
|
|
2380
|
+
await settleRunUsage();
|
|
1671
2381
|
return { status: "cancelled" };
|
|
1672
2382
|
}
|
|
1673
2383
|
|
|
@@ -1691,6 +2401,10 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
1691
2401
|
const deps = depsOverride ? { ...defaultDeps(), ...depsOverride } : defaultDeps();
|
|
1692
2402
|
const git = deps.git;
|
|
1693
2403
|
const signal = opts.signal;
|
|
2404
|
+
// 0782 — the poll loop's hand-off for "a person pressed Stop on the card": it
|
|
2405
|
+
// posts the notice naming them and aborts this job's signal, which is what
|
|
2406
|
+
// tears the coding CLI's process group down (runCli's abort path).
|
|
2407
|
+
const onStopRequested = opts.onStopRequested;
|
|
1694
2408
|
// When the mention was a thread reply, keep the whole exchange in that thread.
|
|
1695
2409
|
const parentId = message.parentId ?? null;
|
|
1696
2410
|
|
|
@@ -1738,6 +2452,43 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
1738
2452
|
}
|
|
1739
2453
|
}
|
|
1740
2454
|
|
|
2455
|
+
// MERGE / CLOSE RELAY (0704): "merge it" in a PR thread used to get an answer
|
|
2456
|
+
// describing the button a human should press. Now it is executed — through the
|
|
2457
|
+
// server, which is the authority: merge_pr / close_pr take the id of the human
|
|
2458
|
+
// message that asked, re-classify that text with a forced-tool model call, and
|
|
2459
|
+
// check the author's workspace role before spending the App's write token. The
|
|
2460
|
+
// daemon carries no new trust; it only stops swallowing the request.
|
|
2461
|
+
//
|
|
2462
|
+
// Runs BEFORE the chat/code split for the same reason the server runs it before
|
|
2463
|
+
// the coding gates: a merge instruction must never be read as a request to
|
|
2464
|
+
// start a new run. Every "no" here falls through to today's behavior.
|
|
2465
|
+
{
|
|
2466
|
+
const relay = prActionRelayDecision({
|
|
2467
|
+
body: message.body,
|
|
2468
|
+
threadPr: message.threadPr,
|
|
2469
|
+
mode,
|
|
2470
|
+
// A manager-routed dispatch is a brief to implement, not an instruction to
|
|
2471
|
+
// land something — its body is assembled text, never a person's sentence.
|
|
2472
|
+
canRelay: Boolean(caps.prActions) && !message.dispatch,
|
|
2473
|
+
});
|
|
2474
|
+
if (relay.relay) {
|
|
2475
|
+
const verb = relay.action;
|
|
2476
|
+
console.log(` ${verb} → relaying PR #${relay.anchor.prNumber} to the server (it decides)`);
|
|
2477
|
+
const result = await tool(verb === "merge" ? "merge_pr" : "close_pr", {
|
|
2478
|
+
channelId,
|
|
2479
|
+
// The mention itself is the instruction on record. The server verifies
|
|
2480
|
+
// it is a human's message, in this channel, from an owner or admin, and
|
|
2481
|
+
// that it really asks for THIS action — citing anything else fails.
|
|
2482
|
+
instructionMessageId: message.id,
|
|
2483
|
+
}).catch((e) => ({ ok: false, error: `I couldn't reach hilos to ${verb} that pull request: ${e?.message || e}` }));
|
|
2484
|
+
const reply = prActionReply(result);
|
|
2485
|
+
if (reply.post) {
|
|
2486
|
+
await tool("post_message", { channelId, parentId, body: reply.post }).catch(() => {});
|
|
2487
|
+
}
|
|
2488
|
+
return { status: reply.ok ? `pr-${verb}d` : "pr-action-refused", action: verb };
|
|
2489
|
+
}
|
|
2490
|
+
}
|
|
2491
|
+
|
|
1741
2492
|
// No repo linked → either FOLDER mode (0322: a channel mapped to a plain local
|
|
1742
2493
|
// folder can still get coding work done) or, failing that, just reply. A linked
|
|
1743
2494
|
// repo always wins — the folder map is only consulted when there's no repo link.
|
|
@@ -1868,6 +2619,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
1868
2619
|
brief: routed.task,
|
|
1869
2620
|
workspaceMemory,
|
|
1870
2621
|
context,
|
|
2622
|
+
onStopRequested,
|
|
1871
2623
|
});
|
|
1872
2624
|
}
|
|
1873
2625
|
// Not a coding task → post the router's reply if it produced one, else fall
|
|
@@ -2034,22 +2786,65 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2034
2786
|
git(repoPath, ["fetch", "origin", cfg.defaultBranch]);
|
|
2035
2787
|
|
|
2036
2788
|
// Same-branch iteration: continue an existing PR on its own branch and push
|
|
2037
|
-
// onto it, so the PR updates in place instead of a duplicate opening.
|
|
2789
|
+
// onto it, so the PR updates in place instead of a duplicate opening. Three ways
|
|
2038
2790
|
// in, and an explicit token WINS when both apply (they agree — same PR):
|
|
2039
2791
|
// 1. detectPrContinuation — an explicit "PR <url>" in the text (the server's
|
|
2040
2792
|
// "request changes" rework ping, or a person naming a PR). The fallback.
|
|
2041
2793
|
// 2. the run resolver's 'iterate' (0281) — a plain follow-up reply in a thread
|
|
2042
2794
|
// whose run already owns an OPEN PR. The PRIMARY path.
|
|
2795
|
+
// 3. the THREAD ANCHOR (0704) — no active run owns this thread, but the server
|
|
2796
|
+
// says (on list_mentions, as `threadPr`) that the thread is about a pull
|
|
2797
|
+
// request: a second agent asked here, or a reply long after the first run
|
|
2798
|
+
// settled, still amends THAT PR. Same shape as the server's own anchor
|
|
2799
|
+
// fallback, and it never overrides 1 or 2.
|
|
2043
2800
|
// Otherwise, a fresh branch off the default (today's behavior).
|
|
2044
2801
|
const cont = detectPrContinuation(message.body, repoFullName);
|
|
2045
|
-
const
|
|
2802
|
+
const runIterateUrl = (followupMode === "iterate" && activeRun?.prUrl) || null;
|
|
2803
|
+
// The anchor is a HYPOTHESIS (its recorded state may be stale or unknown);
|
|
2804
|
+
// `gh` is the proof. No proof, no adoption — the run cuts a fresh branch and
|
|
2805
|
+
// says so on the status card rather than pushing onto a merged PR.
|
|
2806
|
+
let anchorAdopt = null;
|
|
2807
|
+
let anchorNote = "";
|
|
2808
|
+
if (!cont?.url && !runIterateUrl) {
|
|
2809
|
+
const decision = anchorIterateDecision({
|
|
2810
|
+
threadPr: message.threadPr,
|
|
2811
|
+
repoFullName,
|
|
2812
|
+
signal: followupSignal,
|
|
2813
|
+
});
|
|
2814
|
+
if (decision.adopt && deps.prState) {
|
|
2815
|
+
const live = deps.prState(repoPath, decision.anchor.prUrl);
|
|
2816
|
+
const confirmed = confirmAnchorOpen({ live, anchor: decision.anchor });
|
|
2817
|
+
if (confirmed.ok) {
|
|
2818
|
+
anchorAdopt = {
|
|
2819
|
+
prUrl: decision.anchor.prUrl,
|
|
2820
|
+
prNumber: decision.anchor.prNumber,
|
|
2821
|
+
branch: confirmed.branch,
|
|
2822
|
+
};
|
|
2823
|
+
console.log(
|
|
2824
|
+
` code → thread anchor: PR #${decision.anchor.prNumber} is open on \`${confirmed.branch}\` — adopting it`,
|
|
2825
|
+
);
|
|
2826
|
+
} else {
|
|
2827
|
+
anchorNote = staleAnchorNote({ reason: confirmed.reason, anchor: decision.anchor });
|
|
2828
|
+
console.log(
|
|
2829
|
+
` ! thread anchor PR #${decision.anchor.prNumber} not adopted (${confirmed.reason}) — new branch`,
|
|
2830
|
+
);
|
|
2831
|
+
}
|
|
2832
|
+
} else if (decision.adopt) {
|
|
2833
|
+
anchorNote = staleAnchorNote({ reason: "unverified", anchor: decision.anchor });
|
|
2834
|
+
} else if (decision.reason === "anchor-merged" || decision.reason === "anchor-closed") {
|
|
2835
|
+
anchorNote = staleAnchorNote({ reason: decision.reason, anchor: decision.anchor });
|
|
2836
|
+
}
|
|
2837
|
+
}
|
|
2838
|
+
const iterateUrl = cont?.url || runIterateUrl || anchorAdopt?.prUrl || null;
|
|
2046
2839
|
let branch = null;
|
|
2047
2840
|
let continuingPrUrl = null;
|
|
2048
|
-
if (iterateUrl && deps.prHeadRef) {
|
|
2841
|
+
if (iterateUrl && (deps.prHeadRef || anchorAdopt)) {
|
|
2049
2842
|
// Resolve the PR's head branch (source of truth); for a run-based iterate the
|
|
2050
|
-
// run's recorded branch is a fallback if gh can't view the PR.
|
|
2843
|
+
// run's recorded branch is a fallback if gh can't view the PR. An adopted
|
|
2844
|
+
// anchor already has its head branch straight from the live PR.
|
|
2051
2845
|
const headRef =
|
|
2052
|
-
|
|
2846
|
+
(anchorAdopt?.prUrl === iterateUrl ? anchorAdopt.branch : null) ||
|
|
2847
|
+
deps.prHeadRef?.(repoPath, iterateUrl) ||
|
|
2053
2848
|
(followupMode === "iterate" ? activeRun?.branch || null : null);
|
|
2054
2849
|
if (headRef) {
|
|
2055
2850
|
// Fetch the PR branch, then point a local branch at its tip via FETCH_HEAD
|
|
@@ -2067,6 +2862,12 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2067
2862
|
console.log(
|
|
2068
2863
|
` ! couldn't check out PR branch \`${headRef}\` (falling back to a new branch): ${(sw.stderr || "").trim().slice(0, 200)}`,
|
|
2069
2864
|
);
|
|
2865
|
+
// An adopted anchor that can't be checked out is the same broken promise
|
|
2866
|
+
// as a stale one — the thread asked about a PR and gets a new branch, so
|
|
2867
|
+
// the status card has to say it.
|
|
2868
|
+
if (anchorAdopt?.prUrl === iterateUrl) {
|
|
2869
|
+
anchorNote = staleAnchorNote({ reason: "checkout-failed", anchor: anchorAdopt });
|
|
2870
|
+
}
|
|
2070
2871
|
}
|
|
2071
2872
|
}
|
|
2072
2873
|
}
|
|
@@ -2118,6 +2919,12 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2118
2919
|
} else if (effectiveMode === "redirect") {
|
|
2119
2920
|
const n = activeRun?.prNumber || prNumberFromUrl(activeRun?.prUrl);
|
|
2120
2921
|
ackBody = `Starting a new PR${n ? ` (separate from #${n})` : ""} on \`${branch}\`.`;
|
|
2922
|
+
} else if (anchorNote) {
|
|
2923
|
+
// The thread is about a pull request this run could NOT adopt (0704). Say it
|
|
2924
|
+
// on the status card itself — silence here would read as "pushing to my PR"
|
|
2925
|
+
// while a fresh branch is being cut. Still one card, per the run-output
|
|
2926
|
+
// contract: no extra chat message.
|
|
2927
|
+
ackBody = `${anchorNote} ${ackBody}`;
|
|
2121
2928
|
}
|
|
2122
2929
|
const ack = await tool("post_message", { channelId, parentId, body: ackBody });
|
|
2123
2930
|
const ackId = ack?.messageId ?? null;
|
|
@@ -2171,6 +2978,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2171
2978
|
threadRootId: threadRoot,
|
|
2172
2979
|
taskText: message.body,
|
|
2173
2980
|
branch,
|
|
2981
|
+
// 0782 — the mention we picked up. The server reads its human author and
|
|
2982
|
+
// records them as the run's requester, which is who (besides workspace
|
|
2983
|
+
// admins) may stop this run from its card. The thread root is NOT that
|
|
2984
|
+
// person: it can be an ack this agent wrote, or someone else's thread.
|
|
2985
|
+
requestedByMessageId: message.id,
|
|
2174
2986
|
// Map an unrecognized command to null rather than the off-vocabulary
|
|
2175
2987
|
// "unknown" — provider is documented as
|
|
2176
2988
|
// claude_code|codex|cursor|opencode|antigravity|hermes|hilos.
|
|
@@ -2186,7 +2998,14 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2186
2998
|
}
|
|
2187
2999
|
// Keep the inspectable continuation/redirect statement (don't overwrite it with
|
|
2188
3000
|
// an LLM plan) so a human can correct the routing before code lands.
|
|
2189
|
-
if (
|
|
3001
|
+
if (
|
|
3002
|
+
ackId &&
|
|
3003
|
+
caps.editMessage &&
|
|
3004
|
+
chatCmdFor(cfg) &&
|
|
3005
|
+
!continuingPrUrl &&
|
|
3006
|
+
!anchorNote &&
|
|
3007
|
+
effectiveMode !== "redirect"
|
|
3008
|
+
) {
|
|
2190
3009
|
// Feed the ack the router's distilled brief AND the conversation — not the raw
|
|
2191
3010
|
// mention — so it states a real plan instead of "what's the task?".
|
|
2192
3011
|
const plan = await proposePlanAck({
|
|
@@ -2210,7 +3029,35 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2210
3029
|
const streamOn = Boolean(caps.postProgress);
|
|
2211
3030
|
const vendor = detectVendor(cfg.codingCmd);
|
|
2212
3031
|
const streamArgs = codeStreamArgs(vendor);
|
|
3032
|
+
// 0792 — the run's own record, when the operator asked for one. Three gates,
|
|
3033
|
+
// all of which have to say yes: the operator's config, the server offering
|
|
3034
|
+
// the tool, and a durable run row to hang the object off. The tap lives at
|
|
3035
|
+
// TASK scope, not per-CLI-invocation, so a resume retry and the gate:true
|
|
3036
|
+
// revision rounds all land in one transcript — the run is the unit, not the
|
|
3037
|
+
// spawn.
|
|
3038
|
+
const transcriptTap =
|
|
3039
|
+
cfg.uploadTranscripts && caps.uploadTranscript && runId ? createTranscriptTap() : null;
|
|
3040
|
+
// 0779 — screenshots the poll loop pulled to a temp dir at pickup. The prompt
|
|
3041
|
+
// names them; the argv carries them for a vendor with a verified image flag.
|
|
3042
|
+
const localImages = Array.isArray(message?.images) ? message.images : [];
|
|
3043
|
+
const codeImagePromptArgs = localImages.length
|
|
3044
|
+
? {
|
|
3045
|
+
images: localImages,
|
|
3046
|
+
imagesReadable: imagesNeedReading(vendor, { gated: codexRunIsGated(cfg, caps) }),
|
|
3047
|
+
}
|
|
3048
|
+
: {};
|
|
3049
|
+
// 0785 — one line per run, whichever CLI turns out to be ungateable.
|
|
3050
|
+
const noticeUngatedRun = createUngatedRunNotice({
|
|
3051
|
+
// Read at post time: the thread root is the ack this run posts.
|
|
3052
|
+
post: (body) => tool("post_message", { channelId, parentId: threadRoot, body }),
|
|
3053
|
+
log: console,
|
|
3054
|
+
});
|
|
2213
3055
|
let runSessionId = null; // captured from the stream for 0282 (resume)
|
|
3056
|
+
// 0787 — what the run cost, folded across every runAndStage call (a gated
|
|
3057
|
+
// iterate runs the CLI more than once) and taken at each report.
|
|
3058
|
+
const runUsage = createUsageFold();
|
|
3059
|
+
let resolvedModelId = pinnedModelId(cfg.codingCmd);
|
|
3060
|
+
let lastTerminalState = "done";
|
|
2214
3061
|
const machine = hostname();
|
|
2215
3062
|
|
|
2216
3063
|
// Session resume (0282): on an ITERATE we can resume the coding agent's SESSION so
|
|
@@ -2221,8 +3068,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2221
3068
|
// (get_active_run) with no local match means the session likely lives on another
|
|
2222
3069
|
// machine/instance — a bad `--resume` id makes claude error → empty diff → a failed
|
|
2223
3070
|
// run, so we DON'T resume and degrade to today's branch+feedback (never worse).
|
|
2224
|
-
// claude_code + cursor + opencode have proven resume flags
|
|
2225
|
-
//
|
|
3071
|
+
// claude_code + cursor + opencode + codex have proven resume flags
|
|
3072
|
+
// (0282/0573/0608/0783); unknown → []. Note codex's is a SUBCOMMAND (`exec
|
|
3073
|
+
// resume <id>`), not a flag — it still splices in here, because flags placed
|
|
3074
|
+
// before it are parsed as `exec`'s own (verified live on codex-cli 0.144.1).
|
|
3075
|
+
// The gate itself is pure (resumeDecision in resume.mjs):
|
|
2226
3076
|
// vendor, machine, project and server agreement all have to line up, and any
|
|
2227
3077
|
// "no" degrades to the branch+feedback iterate rather than risking a bad id.
|
|
2228
3078
|
const projectKey = codeProjectKey(vendor, repoPath, cfg.codingCmd);
|
|
@@ -2283,6 +3133,25 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2283
3133
|
clearInterval(beat);
|
|
2284
3134
|
};
|
|
2285
3135
|
};
|
|
3136
|
+
/**
|
|
3137
|
+
* The authoritative check before anything leaves this machine (0782).
|
|
3138
|
+
*
|
|
3139
|
+
* The in-run poll stops when the CLI does, and the git work that follows —
|
|
3140
|
+
* commit, push, `gh pr create` — can take a minute with nothing streaming. A
|
|
3141
|
+
* stop that lands in THAT window would otherwise ship work nobody is waiting
|
|
3142
|
+
* on. So we ask the server one more time, in the shape that changes nothing:
|
|
3143
|
+
* `state` echoes the card's current lifecycle, and a stopped run's card is
|
|
3144
|
+
* frozen server-side regardless. A stop here fires the same hand-off the
|
|
3145
|
+
* heartbeat does (the room hears who stopped it; the job's signal aborts).
|
|
3146
|
+
*/
|
|
3147
|
+
const stoppedBeforeShip = async (state = "done") => {
|
|
3148
|
+
// Only the streaming card carries progress metadata. The legacy heartbeat's
|
|
3149
|
+
// message is plain text, and stamping a lifecycle onto it would turn a
|
|
3150
|
+
// status line into a run card — so that path simply has no pre-ship check.
|
|
3151
|
+
if (!streamOn || !progressId) return false;
|
|
3152
|
+
const sender = createProgressSender({ tool, statusId: progressId, runId, onStopRequested });
|
|
3153
|
+
return await sender.poll(state);
|
|
3154
|
+
};
|
|
2286
3155
|
// Once the run ends, retire the "still working…" progress reply so it doesn't
|
|
2287
3156
|
// sit there claiming the agent is alive. No-op when no beat ever fired (short
|
|
2288
3157
|
// run) or without edit_message. Best-effort.
|
|
@@ -2292,6 +3161,25 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2292
3161
|
}
|
|
2293
3162
|
};
|
|
2294
3163
|
|
|
3164
|
+
/**
|
|
3165
|
+
* Accounting-only settlement (0787) — the repo lane's copy of the folder
|
|
3166
|
+
* lane's. A run that ends with no report (no changes, a CLI that never
|
|
3167
|
+
* started, a person pressing Stop, a self-committing agent under the gate)
|
|
3168
|
+
* still spent tokens, and `post_progress` is the terminal, run-scoped call
|
|
3169
|
+
* the daemon already makes on all of them. Only ever the DELTA, so calling it
|
|
3170
|
+
* on a path that later reports anyway is harmless: the fold is already empty.
|
|
3171
|
+
*/
|
|
3172
|
+
const settleRunUsage = async () => {
|
|
3173
|
+
const args = reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId });
|
|
3174
|
+
if (!args.usage || !progressId) return;
|
|
3175
|
+
await tool("post_progress", {
|
|
3176
|
+
messageId: progressId,
|
|
3177
|
+
...(runId ? { runId } : {}),
|
|
3178
|
+
progress: { state: lastTerminalState },
|
|
3179
|
+
usage: args.usage,
|
|
3180
|
+
}).catch(() => {});
|
|
3181
|
+
};
|
|
3182
|
+
|
|
2295
3183
|
const parts = cfg.codingCmd.split(" ").filter(Boolean);
|
|
2296
3184
|
// Run the CLI and stage everything it changed; return the diff stats (no post).
|
|
2297
3185
|
// Workspace memory (the project's soul) is prepended so the coding agent has
|
|
@@ -2311,7 +3199,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2311
3199
|
// when post_report settles the report, it would re-write metadata WITHOUT the
|
|
2312
3200
|
// report (it read the pre-settle snapshot) and clobber it. Draining every send
|
|
2313
3201
|
// here guarantees no progress write is in flight once we settle.
|
|
2314
|
-
let
|
|
3202
|
+
let drainProgress = async () => {};
|
|
3203
|
+
// 0782 — the silent-work stop poll, torn down in the same finally as the run.
|
|
3204
|
+
let stopPoller = null;
|
|
2315
3205
|
if (streamOn) {
|
|
2316
3206
|
if (!progressId) {
|
|
2317
3207
|
try {
|
|
@@ -2326,27 +3216,35 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2326
3216
|
}
|
|
2327
3217
|
}
|
|
2328
3218
|
const statusId = progressId;
|
|
2329
|
-
|
|
2330
|
-
|
|
2331
|
-
|
|
2332
|
-
|
|
2333
|
-
|
|
2334
|
-
|
|
2335
|
-
|
|
2336
|
-
|
|
2337
|
-
|
|
2338
|
-
|
|
2339
|
-
|
|
2340
|
-
|
|
2341
|
-
|
|
2342
|
-
|
|
2343
|
-
|
|
2344
|
-
|
|
2345
|
-
|
|
2346
|
-
|
|
2347
|
-
|
|
2348
|
-
|
|
2349
|
-
|
|
3219
|
+
// The module never imports MCP — the send fn is injected here. It must
|
|
3220
|
+
// never throw into the run (createProgressEmitter also guards). 0782:
|
|
3221
|
+
// the sender also carries the stop signal home on this same beat, bound
|
|
3222
|
+
// to THIS run's id so a stop can't be read off a neighbouring thread.
|
|
3223
|
+
const sender = statusId
|
|
3224
|
+
? createProgressSender({ tool, statusId, runId, onStopRequested })
|
|
3225
|
+
: null;
|
|
3226
|
+
// The emitter is built whether or not the status card posted (0787). The
|
|
3227
|
+
// stream flags are already on the argv either way, so the CLI is speaking
|
|
3228
|
+
// NDJSON regardless; gating the parser on a card meant one transient
|
|
3229
|
+
// post_message failure silently cost the run its session id (the resume
|
|
3230
|
+
// record's only source on an argv run) and its usage. A card is where
|
|
3231
|
+
// progress is SHOWN, never how the stream is read — this is what the
|
|
3232
|
+
// folder path has always done. With no card the sends are dropped on the
|
|
3233
|
+
// floor rather than skipped, so nothing else in the run changes shape.
|
|
3234
|
+
emitter = createProgressEmitter({
|
|
3235
|
+
parser: makeStreamParser(vendor),
|
|
3236
|
+
now: deps.now,
|
|
3237
|
+
throttleMs: cfg.progressMs,
|
|
3238
|
+
send: sender ? sender.send : () => {},
|
|
3239
|
+
});
|
|
3240
|
+
if (sender) {
|
|
3241
|
+
drainProgress = () => sender.drain();
|
|
3242
|
+
// The heartbeat only fires when the CLI speaks. This asks anyway, so a
|
|
3243
|
+
// run inside one long silent command (a full test suite, a slow install)
|
|
3244
|
+
// is still stoppable from the room.
|
|
3245
|
+
stopPoller = createStopPoller({
|
|
3246
|
+
poll: () => sender.poll("working"),
|
|
3247
|
+
intervalMs: cfg.stopPollMs || STOP_POLL_MS,
|
|
2350
3248
|
});
|
|
2351
3249
|
}
|
|
2352
3250
|
} else {
|
|
@@ -2360,10 +3258,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2360
3258
|
// fallback) suppresses it. buildResumeArgs is [] unless vendor+session make
|
|
2361
3259
|
// resume safe, so a non-resume run is byte-identical to the pre-0282 ARGV.
|
|
2362
3260
|
const resumeArgs = resume ? buildResumeArgs(vendor, resumeSessionId) : [];
|
|
2363
|
-
// Model preset (0504): resolved at run time against the
|
|
2364
|
-
//
|
|
2365
|
-
//
|
|
3261
|
+
// Model preset (0504, codex in 0783): resolved at run time against the
|
|
3262
|
+
// CLI's own model list — [] when unset/unresolvable, so the tool's default
|
|
3263
|
+
// stands. Inserted before resume/stream flags, after the base — an order
|
|
3264
|
+
// codex depends on, since its resume is a subcommand and the model flag
|
|
3265
|
+
// has to reach `exec`, i.e. sit BEFORE `resume`.
|
|
2366
3266
|
const modelArgs = await modelArgsFor(cfg, vendor);
|
|
3267
|
+
// 0787: remember the id we actually handed the CLI — the usage receipt for
|
|
3268
|
+
// a vendor whose stream never names a model (codex) has nothing else to say.
|
|
3269
|
+
if (modelArgs[0] === "--model" && modelArgs[1]) resolvedModelId = modelArgs[1];
|
|
2367
3270
|
// Project pin (0608, opencode only): the CLI resolves its project from PWD,
|
|
2368
3271
|
// not the spawn cwd, and its sessions are per project — `--dir` makes both
|
|
2369
3272
|
// deterministic. [] for every other vendor (and for an `--attach`ed run,
|
|
@@ -2371,13 +3274,29 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2371
3274
|
// The spawned PWD now matches the cwd too (0615); `--dir` stays as the
|
|
2372
3275
|
// CLI's own explicit contract, and to keep attach runs off our local path.
|
|
2373
3276
|
const dirArgs = codeDirArgs(vendor, repoPath, cfg.codingCmd);
|
|
3277
|
+
// 0779: BEFORE resumeArgs, for the same reason modelArgs are — codex's
|
|
3278
|
+
// resume is a SUBCOMMAND (`exec resume <id>`) and `--image=` belongs to
|
|
3279
|
+
// `exec`, so it has to sit ahead of it. [] for every vendor without a
|
|
3280
|
+
// verified flag, leaving their argv byte-identical to before.
|
|
3281
|
+
const imageArgs = codeImageArgs(vendor, localImages);
|
|
2374
3282
|
const codeArgs = streamOn
|
|
2375
|
-
? [...parts.slice(1), ...modelArgs, ...dirArgs, ...resumeArgs, ...streamArgs]
|
|
2376
|
-
: [...parts.slice(1), ...modelArgs, ...dirArgs, ...resumeArgs];
|
|
3283
|
+
? [...parts.slice(1), ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs, ...streamArgs]
|
|
3284
|
+
: [...parts.slice(1), ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs];
|
|
2377
3285
|
const handleCliData = (c) => {
|
|
2378
3286
|
// Keep tracking lastLine as a fallback (legacy heartbeat / honesty).
|
|
2379
3287
|
const lines = String(c).split("\n").map((s) => s.trim()).filter(Boolean);
|
|
2380
3288
|
if (lines.length) lastLine = lines[lines.length - 1];
|
|
3289
|
+
// 0792 — the raw tail, before the parser reduces it to eight steps. Every
|
|
3290
|
+
// transport routes its child's stdout through this one callback, so the
|
|
3291
|
+
// tap sees an argv run, a permission-bridged run, and an ACP session
|
|
3292
|
+
// alike; a transport that writes nothing to stdout simply leaves it empty.
|
|
3293
|
+
if (transcriptTap) {
|
|
3294
|
+
try {
|
|
3295
|
+
transcriptTap.push(c);
|
|
3296
|
+
} catch {
|
|
3297
|
+
/* keeping a record must never break the run */
|
|
3298
|
+
}
|
|
3299
|
+
}
|
|
2381
3300
|
if (emitter) {
|
|
2382
3301
|
try {
|
|
2383
3302
|
emitter.feed(c);
|
|
@@ -2386,19 +3305,89 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2386
3305
|
}
|
|
2387
3306
|
}
|
|
2388
3307
|
};
|
|
3308
|
+
// The structured transports (ACP, `codex mcp-server`) put only the
|
|
3309
|
+
// assistant's PROSE on stdout — every tool call travels here instead. The
|
|
3310
|
+
// live card has always read this; the transcript has to as well, or a gated
|
|
3311
|
+
// run uploads a page of narration with no execution anywhere in it.
|
|
3312
|
+
const handleCliEvent = (ev) => {
|
|
3313
|
+
if (transcriptTap) {
|
|
3314
|
+
try {
|
|
3315
|
+
transcriptTap.pushEvent(ev);
|
|
3316
|
+
} catch {
|
|
3317
|
+
/* keeping a record must never break the run */
|
|
3318
|
+
}
|
|
3319
|
+
}
|
|
3320
|
+
try {
|
|
3321
|
+
emitter?.foldEvent(ev);
|
|
3322
|
+
} catch {
|
|
3323
|
+
/* a progress fold must never break the run */
|
|
3324
|
+
}
|
|
3325
|
+
};
|
|
2389
3326
|
let run;
|
|
2390
3327
|
try {
|
|
2391
3328
|
// OpenCode's own non-interactive CLI auto-rejects every permission ask
|
|
2392
3329
|
// unless --auto is present. For the gated tiers, bypass that responder
|
|
2393
3330
|
// and own the authenticated HTTP session + SSE stream directly so hilos
|
|
2394
3331
|
// is the sole authority answering the paused tool call (0593).
|
|
2395
|
-
const
|
|
3332
|
+
const useAcpTransport = shouldUseAcpTransport({
|
|
2396
3333
|
vendor,
|
|
3334
|
+
acpTransport: cfg.acpTransport,
|
|
2397
3335
|
runtimePermissions: caps.runtimePermissions,
|
|
2398
3336
|
codeArgs,
|
|
2399
3337
|
codingCmd: cfg.codingCmd,
|
|
2400
3338
|
});
|
|
2401
|
-
|
|
3339
|
+
const useRuntimePermissionBridge =
|
|
3340
|
+
!useAcpTransport &&
|
|
3341
|
+
shouldUseRuntimePermissionBridge({
|
|
3342
|
+
vendor,
|
|
3343
|
+
runtimePermissions: caps.runtimePermissions,
|
|
3344
|
+
codeArgs,
|
|
3345
|
+
codingCmd: cfg.codingCmd,
|
|
3346
|
+
});
|
|
3347
|
+
// 0777, and the reason it composes with resume (0778): claude keeps its
|
|
3348
|
+
// ordinary argv run (resumeArgs included), and codex's gated transport
|
|
3349
|
+
// has its OWN resume — `codex-reply {threadId}` — so a gated iterate
|
|
3350
|
+
// continues the same thread instead of trading continuity for a gate.
|
|
3351
|
+
const gateCodexPermissions =
|
|
3352
|
+
!useAcpTransport &&
|
|
3353
|
+
!useRuntimePermissionBridge &&
|
|
3354
|
+
shouldGateCodexPermissions({
|
|
3355
|
+
vendor,
|
|
3356
|
+
runtimePermissions: caps.runtimePermissions,
|
|
3357
|
+
codeArgs,
|
|
3358
|
+
});
|
|
3359
|
+
const gateClaudePermissions =
|
|
3360
|
+
!useAcpTransport &&
|
|
3361
|
+
!useRuntimePermissionBridge &&
|
|
3362
|
+
!gateCodexPermissions &&
|
|
3363
|
+
shouldGateClaudePermissions({
|
|
3364
|
+
vendor,
|
|
3365
|
+
runtimePermissions: caps.runtimePermissions,
|
|
3366
|
+
codeArgs,
|
|
3367
|
+
});
|
|
3368
|
+
if (useAcpTransport) {
|
|
3369
|
+
run = await deps.runAcpSession({
|
|
3370
|
+
cmd: parts[0],
|
|
3371
|
+
vendor,
|
|
3372
|
+
cwd: repoPath,
|
|
3373
|
+
prompt: memoryPreamble(workspaceMemory) + promptText,
|
|
3374
|
+
// 0778: approvals AND continuity. `resume:false` (the never-worse
|
|
3375
|
+
// retry) drops it exactly like buildResumeArgs does.
|
|
3376
|
+
resumeSessionId: resume ? resumeSessionId : null,
|
|
3377
|
+
timeoutMs: cfg.runTimeoutMs,
|
|
3378
|
+
signal,
|
|
3379
|
+
env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
|
|
3380
|
+
onData: handleCliData,
|
|
3381
|
+
onEvent: handleCliEvent,
|
|
3382
|
+
...openCodePermissionCallbacks({
|
|
3383
|
+
tool,
|
|
3384
|
+
channelId,
|
|
3385
|
+
threadRoot,
|
|
3386
|
+
runId,
|
|
3387
|
+
provider: vendor,
|
|
3388
|
+
}),
|
|
3389
|
+
});
|
|
3390
|
+
} else if (useRuntimePermissionBridge) {
|
|
2402
3391
|
run = await deps.runOpenCodeHttpSession({
|
|
2403
3392
|
cmd: parts[0],
|
|
2404
3393
|
args: codeArgs,
|
|
@@ -2418,8 +3407,59 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2418
3407
|
runId,
|
|
2419
3408
|
}),
|
|
2420
3409
|
});
|
|
3410
|
+
} else if (gateCodexPermissions) {
|
|
3411
|
+
run = await runCodexGatedSession({
|
|
3412
|
+
deps,
|
|
3413
|
+
cfg,
|
|
3414
|
+
onGateDropped: () => noticeUngatedRun(parts[0]),
|
|
3415
|
+
cmd: parts[0],
|
|
3416
|
+
codeArgs,
|
|
3417
|
+
cwd: repoPath,
|
|
3418
|
+
prompt: memoryPreamble(workspaceMemory) + promptText,
|
|
3419
|
+
model: resolvedModelId,
|
|
3420
|
+
// Codex's own resume over this transport. `resume:false` (the
|
|
3421
|
+
// never-worse retry) drops it exactly like buildResumeArgs does.
|
|
3422
|
+
resumeThreadId: resume ? resumeSessionId : null,
|
|
3423
|
+
signal,
|
|
3424
|
+
onData: handleCliData,
|
|
3425
|
+
onEvent: handleCliEvent,
|
|
3426
|
+
permissionCallbacks: openCodePermissionCallbacks({
|
|
3427
|
+
tool,
|
|
3428
|
+
channelId,
|
|
3429
|
+
threadRoot,
|
|
3430
|
+
runId,
|
|
3431
|
+
provider: vendor,
|
|
3432
|
+
}),
|
|
3433
|
+
});
|
|
3434
|
+
} else if (gateClaudePermissions) {
|
|
3435
|
+
run = await runClaudeGatedCli({
|
|
3436
|
+
deps,
|
|
3437
|
+
cfg,
|
|
3438
|
+
onGateDropped: () => noticeUngatedRun(parts[0]),
|
|
3439
|
+
cmd: parts[0],
|
|
3440
|
+
// codeArgs already carries this run's resume flags, so a gated
|
|
3441
|
+
// iterate resumes AND raises cards — the two never traded off.
|
|
3442
|
+
codeArgs,
|
|
3443
|
+
prompt: memoryPreamble(workspaceMemory) + promptText,
|
|
3444
|
+
cwd: repoPath,
|
|
3445
|
+
signal,
|
|
3446
|
+
onData: handleCliData,
|
|
3447
|
+
sessionId: resume ? resumeSessionId ?? "" : "",
|
|
3448
|
+
resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
|
|
3449
|
+
permissionCallbacks: openCodePermissionCallbacks({
|
|
3450
|
+
tool,
|
|
3451
|
+
channelId,
|
|
3452
|
+
threadRoot,
|
|
3453
|
+
runId,
|
|
3454
|
+
provider: vendor,
|
|
3455
|
+
}),
|
|
3456
|
+
});
|
|
2421
3457
|
} else {
|
|
2422
|
-
|
|
3458
|
+
// deps.runCli, not the bare import: `defaultDeps` wraps the very same
|
|
3459
|
+
// function, so production is byte-identical, but the ungated repo run
|
|
3460
|
+
// was the ONE coding path that escaped the injectable runner — which is
|
|
3461
|
+
// why nothing above the unit tests could ever drive it (0787).
|
|
3462
|
+
run = await deps.runCli({
|
|
2423
3463
|
cmd: parts[0],
|
|
2424
3464
|
args: [...codeArgs, memoryPreamble(workspaceMemory) + promptText],
|
|
2425
3465
|
cwd: repoPath,
|
|
@@ -2430,8 +3470,12 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2430
3470
|
onData: handleCliData,
|
|
2431
3471
|
});
|
|
2432
3472
|
}
|
|
3473
|
+
// 0785 — the gate was expected here and the CLI couldn't hold it. Say so
|
|
3474
|
+
// in the room, once per run, rather than only in the daemon's console.
|
|
3475
|
+
if (run?.permissionGateDropped) await noticeUngatedRun(parts[0]);
|
|
2433
3476
|
} finally {
|
|
2434
3477
|
stopHeartbeat();
|
|
3478
|
+
stopPoller?.stop();
|
|
2435
3479
|
if (run?.sessionId) runSessionId = run.sessionId;
|
|
2436
3480
|
if (emitter) {
|
|
2437
3481
|
// Terminal state: flip the status card off "working" (state 'done'/'error')
|
|
@@ -2440,6 +3484,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2440
3484
|
// (0289) — the card cross-fades run → report; for no-changes/failed/gate
|
|
2441
3485
|
// outcomes this terminal 'done'/'error' is the card's final state.
|
|
2442
3486
|
const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
|
|
3487
|
+
// Remembered so an exit with no report re-sends this same state (0787)
|
|
3488
|
+
// rather than repainting the card into something it wasn't.
|
|
3489
|
+
lastTerminalState = errored ? "error" : "done";
|
|
2443
3490
|
// On error, carry an honest reason onto the card (0294) — the same signal
|
|
2444
3491
|
// the chat note uses: spawn error message, else exit status, plus a short
|
|
2445
3492
|
// stderr tail. Sanitized again server-side; an old server just drops it.
|
|
@@ -2464,7 +3511,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2464
3511
|
// BEFORE runAndStage returns, so nothing is in flight when the caller
|
|
2465
3512
|
// settles the report onto this card — no lost-update clobber (0289).
|
|
2466
3513
|
try {
|
|
2467
|
-
await
|
|
3514
|
+
await drainProgress();
|
|
2468
3515
|
} catch {
|
|
2469
3516
|
/* a drain failure must never break the run */
|
|
2470
3517
|
}
|
|
@@ -2474,6 +3521,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2474
3521
|
} catch {
|
|
2475
3522
|
/* ignore */
|
|
2476
3523
|
}
|
|
3524
|
+
// 0787 — read the run's cost AFTER done() (the terminal usage frame
|
|
3525
|
+
// usually arrives in the parser flush). Folded, not sent yet: the
|
|
3526
|
+
// report call is what carries it home.
|
|
3527
|
+
try {
|
|
3528
|
+
const spend = emitter.usage?.();
|
|
3529
|
+
if (spend) runUsage.push({ t: "usage", ...spend });
|
|
3530
|
+
} catch {
|
|
3531
|
+
/* accounting must never break the run */
|
|
3532
|
+
}
|
|
2477
3533
|
}
|
|
2478
3534
|
}
|
|
2479
3535
|
if (run.aborted || signal?.aborted) {
|
|
@@ -2541,8 +3597,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2541
3597
|
sessionId: sid,
|
|
2542
3598
|
machine,
|
|
2543
3599
|
// The CLI that created the session (a `ses_…` is meaningless to
|
|
2544
|
-
// claude's `--resume`) and the project it belongs to — opencode
|
|
2545
|
-
//
|
|
3600
|
+
// claude's `--resume`) and the project it belongs to — opencode and
|
|
3601
|
+
// codex both scope sessions per project. The gate above requires
|
|
3602
|
+
// both (0608, 0783).
|
|
2546
3603
|
vendor,
|
|
2547
3604
|
cwd: projectKey,
|
|
2548
3605
|
updatedAt: new Date().toISOString(),
|
|
@@ -2553,6 +3610,32 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2553
3610
|
}
|
|
2554
3611
|
};
|
|
2555
3612
|
|
|
3613
|
+
// 0792 — ship the run's record, AFTER its report. Ordering is the whole
|
|
3614
|
+
// point: the transcript is a fact about finished work, and the card it marks
|
|
3615
|
+
// has to exist before it can be marked. Best-effort in every direction — no
|
|
3616
|
+
// tap, no stream, no server tool, or a refused upload all leave the run
|
|
3617
|
+
// exactly as it would have ended without this feature.
|
|
3618
|
+
let transcriptUploaded = false;
|
|
3619
|
+
const uploadTranscript = async (reportMsgId) => {
|
|
3620
|
+
if (!transcriptTap || !runId || transcriptUploaded) return;
|
|
3621
|
+
let text = "";
|
|
3622
|
+
try {
|
|
3623
|
+
text = transcriptTap.text();
|
|
3624
|
+
} catch {
|
|
3625
|
+
return;
|
|
3626
|
+
}
|
|
3627
|
+
// Nothing to say: a transport that never wrote to stdout (an ACP session, a
|
|
3628
|
+
// vendor with no stream) has no transcript, and an empty upload is worse
|
|
3629
|
+
// than none — it would put an empty object behind a Transcript control.
|
|
3630
|
+
if (!text) return;
|
|
3631
|
+
transcriptUploaded = true;
|
|
3632
|
+
await tool("upload_run_transcript", {
|
|
3633
|
+
runId,
|
|
3634
|
+
transcript: text,
|
|
3635
|
+
...(reportMsgId ? { messageId: reportMsgId } : {}),
|
|
3636
|
+
}).catch(() => {});
|
|
3637
|
+
};
|
|
3638
|
+
|
|
2556
3639
|
// Post a proposal card (approve-before-push mode) from a staged change.
|
|
2557
3640
|
const postProposal = async (staged) => {
|
|
2558
3641
|
const report = buildProposalReport({
|
|
@@ -2566,7 +3649,16 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2566
3649
|
stat: staged.stat,
|
|
2567
3650
|
runFailed: staged.runFailed,
|
|
2568
3651
|
});
|
|
2569
|
-
const res = await tool("post_report", {
|
|
3652
|
+
const res = await tool("post_report", {
|
|
3653
|
+
channelId,
|
|
3654
|
+
parentId: threadRoot,
|
|
3655
|
+
broadcast: true,
|
|
3656
|
+
...report,
|
|
3657
|
+
// 0787 — the proposal card is where a gated run's spend lands, because
|
|
3658
|
+
// it is the card the run produced. The later "Shipped" report carries
|
|
3659
|
+
// only what the CLI spent AFTER this one (usually nothing).
|
|
3660
|
+
...reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
|
|
3661
|
+
});
|
|
2570
3662
|
return { reportMessageId: res?.messageId ?? null, stat: staged.stat };
|
|
2571
3663
|
};
|
|
2572
3664
|
|
|
@@ -2605,6 +3697,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2605
3697
|
body,
|
|
2606
3698
|
});
|
|
2607
3699
|
await updateChannelMarker(compactRunMarker("cancelled", branch));
|
|
3700
|
+
// Every repo-lane cancel funnels through here, so one settle covers them
|
|
3701
|
+
// all: a stopped run still spent tokens (0787).
|
|
3702
|
+
await settleRunUsage();
|
|
2608
3703
|
return { status: "cancelled", branch };
|
|
2609
3704
|
};
|
|
2610
3705
|
|
|
@@ -2633,7 +3728,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2633
3728
|
|
|
2634
3729
|
const codePrompt =
|
|
2635
3730
|
(teamMemoryBlock ? teamMemoryBlock + "\n\n" : "") +
|
|
2636
|
-
codeTaskPrompt({ message, context, brief: routed.task, repoFullName });
|
|
3731
|
+
codeTaskPrompt({ message, context, brief: routed.task, repoFullName, ...codeImagePromptArgs });
|
|
2637
3732
|
let staged = await runAndStage(codePrompt);
|
|
2638
3733
|
if (staged.aborted) return await postStopped();
|
|
2639
3734
|
// Never-worse-than-today (0282): if we RESUMED a session and that run FAILED (a
|
|
@@ -2672,11 +3767,13 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2672
3767
|
});
|
|
2673
3768
|
await finalizeProgress(`\`${branch}\` has the agent's own commits — needs review.`);
|
|
2674
3769
|
await updateChannelMarker(compactRunMarker("changes", branch));
|
|
3770
|
+
await settleRunUsage(); // no report on this exit (0787)
|
|
2675
3771
|
return { status: "self-committed-gated", branch };
|
|
2676
3772
|
}
|
|
2677
3773
|
// When a live status card was streaming, settle the report onto it (0289) —
|
|
2678
3774
|
// the card cross-fades run → report. Skip finalizeProgress then (it edits the
|
|
2679
3775
|
// card's body, which the settle overwrites with the report anyway).
|
|
3776
|
+
if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
|
|
2680
3777
|
const selfSettleId = streamOn && progressId ? progressId : null;
|
|
2681
3778
|
if (!selfSettleId) await finalizeProgress(`Agent shipped \`${branch}\` itself — report below.`);
|
|
2682
3779
|
const selfResult = await shipSelfDriven({
|
|
@@ -2691,8 +3788,10 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2691
3788
|
parentId: threadRoot,
|
|
2692
3789
|
settleId: selfSettleId,
|
|
2693
3790
|
runId,
|
|
3791
|
+
usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
|
|
2694
3792
|
});
|
|
2695
3793
|
await recordSession(selfResult.prUrl);
|
|
3794
|
+
await uploadTranscript(selfResult.reportMsgId);
|
|
2696
3795
|
await updateChannelMarker(compactRunMarker(selfResult.status, selfResult.branch));
|
|
2697
3796
|
return selfResult;
|
|
2698
3797
|
}
|
|
@@ -2730,6 +3829,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2730
3829
|
await updateChannelMarker(
|
|
2731
3830
|
compactRunMarker(staged.failed ? "run-failed" : "no-changes", branch),
|
|
2732
3831
|
);
|
|
3832
|
+
await settleRunUsage(); // "no changes" is not "no spend" (0787)
|
|
2733
3833
|
return { status: staged.failed ? "run-failed" : "no-changes" };
|
|
2734
3834
|
}
|
|
2735
3835
|
|
|
@@ -2738,7 +3838,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2738
3838
|
// propose-a-diff-and-wait flow for users who want approve-before-push.
|
|
2739
3839
|
if (!cfg.gate) {
|
|
2740
3840
|
// A "stop" that lands between the run finishing and the ship must still win.
|
|
2741
|
-
|
|
3841
|
+
// The signal covers a stop we already heard; the server check covers one that
|
|
3842
|
+
// landed while the CLI was silent and nothing was carrying it home.
|
|
3843
|
+
if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
|
|
2742
3844
|
// Seamless single card (0289): when a live status card streamed this run,
|
|
2743
3845
|
// settle the report onto it in place instead of posting a separate report
|
|
2744
3846
|
// message. Only the DEFAULT gate:false report settles in place; gate:true's
|
|
@@ -2762,6 +3864,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2762
3864
|
existingPrUrl: continuingPrUrl,
|
|
2763
3865
|
settleId,
|
|
2764
3866
|
runId,
|
|
3867
|
+
usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
|
|
2765
3868
|
});
|
|
2766
3869
|
// The settle already flipped the card to the report (body + metadata); editing
|
|
2767
3870
|
// the body again would clobber the report summary, so only finalize when we
|
|
@@ -2770,6 +3873,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2770
3873
|
// Re-record with the now-known PR url so the local session record + the run row
|
|
2771
3874
|
// carry the PR the session belongs to (best-effort; session id unchanged).
|
|
2772
3875
|
await recordSession(result.prUrl);
|
|
3876
|
+
await uploadTranscript(result.reportMsgId);
|
|
2773
3877
|
await updateChannelMarker(compactRunMarker(result.status, result.branch));
|
|
2774
3878
|
return { ...result, stat: staged.stat, sessionId: runSessionId };
|
|
2775
3879
|
}
|
|
@@ -2806,7 +3910,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2806
3910
|
}
|
|
2807
3911
|
|
|
2808
3912
|
// A stop during the final wait, too — don't ship a cancelled run.
|
|
2809
|
-
if (signal?.aborted) return await postStopped();
|
|
3913
|
+
if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
|
|
2810
3914
|
const result = await applyDecision({
|
|
2811
3915
|
decision,
|
|
2812
3916
|
repoPath,
|
|
@@ -2822,9 +3926,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
|
|
|
2822
3926
|
parentId: threadRoot,
|
|
2823
3927
|
existingPrUrl: continuingPrUrl,
|
|
2824
3928
|
runId,
|
|
3929
|
+
// Normally {} — a gated run already reported its spend on the proposal card
|
|
3930
|
+
// it is now shipping. Anything the CLI spent after that still comes home.
|
|
3931
|
+
usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
|
|
2825
3932
|
});
|
|
2826
3933
|
await finalizeProgress(`Done on \`${branch}\` — see the report below.`);
|
|
2827
3934
|
await recordSession(result.prUrl);
|
|
3935
|
+
// The gated lane's shipped report is a fresh message; `result.reportMsgId`
|
|
3936
|
+
// names it, and the proposal card is the fallback when it never posted.
|
|
3937
|
+
await uploadTranscript(result.reportMsgId || proposal.reportMessageId);
|
|
2828
3938
|
await updateChannelMarker(compactRunMarker(result.status, result.branch));
|
|
2829
3939
|
return { ...result, reportMessageId: proposal.reportMessageId, stat: proposal.stat, rounds: round, sessionId: runSessionId };
|
|
2830
3940
|
} finally {
|