talon-agent 5.18.2 → 5.19.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/README.md +2 -1
  2. package/package.json +2 -2
  3. package/prompts/system/agent-brief.md +20 -3
  4. package/src/app.ts +13 -0
  5. package/src/backend/claude-sdk/handler.ts +4 -4
  6. package/src/backend/claude-sdk/mcp-ready.ts +16 -2
  7. package/src/backend/claude-sdk/one-shot.ts +32 -2
  8. package/src/backend/claude-sdk/stream.ts +2 -2
  9. package/src/backend/codex/auth.ts +1 -1
  10. package/src/backend/codex/handler/message.ts +11 -11
  11. package/src/backend/codex/init.ts +4 -9
  12. package/src/backend/codex/mcp-config.ts +1 -2
  13. package/src/backend/codex/oauth-incompat.ts +8 -4
  14. package/src/backend/codex/one-shot.ts +1 -1
  15. package/src/backend/openai-agents/builtins.ts +55 -27
  16. package/src/backend/openai-agents/factory.ts +3 -3
  17. package/src/backend/openai-agents/handler/message.ts +3 -5
  18. package/src/backend/openai-agents/mcp-pool.ts +10 -27
  19. package/src/backend/remote-server/chat-turn.ts +6 -6
  20. package/src/backend/remote-server/events.ts +1 -5
  21. package/src/backend/remote-server/index.ts +0 -1
  22. package/src/backend/remote-server/messages.ts +3 -7
  23. package/src/backend/remote-server/one-shot.ts +1 -3
  24. package/src/backend/remote-server/session-helpers.ts +1 -4
  25. package/src/backend/remote-server/sse-stream.ts +8 -9
  26. package/src/backend/runtime/metrics.ts +7 -13
  27. package/src/backend/runtime/sleep.ts +1 -2
  28. package/src/backend/runtime/turn/handle-retry.ts +48 -2
  29. package/src/backend/runtime/turn/handler-to-events.ts +3 -4
  30. package/src/bootstrap.ts +9 -1
  31. package/src/cli/doctor.ts +3 -0
  32. package/src/cli/index.ts +8 -10
  33. package/src/cli/logs.ts +148 -9
  34. package/src/cli/setup.ts +9 -11
  35. package/src/cli/status.ts +29 -0
  36. package/src/core/agent-runtime/README.md +5 -19
  37. package/src/core/agent-runtime/events.ts +3 -39
  38. package/src/core/agent-runtime/model-ref.ts +0 -8
  39. package/src/core/agents/registry.ts +65 -2
  40. package/src/core/auth/expiry-monitor.ts +9 -1
  41. package/src/core/auth/login-flow.ts +9 -1
  42. package/src/core/auth/status.ts +31 -3
  43. package/src/core/background/cron/scheduler.ts +25 -5
  44. package/src/core/background/dream/index.ts +29 -10
  45. package/src/core/background/failure-backoff.ts +30 -0
  46. package/src/core/background/heartbeat/agent.ts +2 -27
  47. package/src/core/background/heartbeat/index.ts +0 -2
  48. package/src/core/background/heartbeat/scheduler.ts +21 -13
  49. package/src/core/background/heartbeat/state.ts +10 -2
  50. package/src/core/background/isolated-agent.ts +6 -2
  51. package/src/core/background/pulse/pulse.ts +9 -0
  52. package/src/core/background/triggers/exit.ts +54 -0
  53. package/src/core/background/triggers/index.ts +1 -3
  54. package/src/core/background/triggers/resume.ts +2 -4
  55. package/src/core/backup/archive/tar.ts +14 -4
  56. package/src/core/backup/plan.ts +1 -0
  57. package/src/core/backup/restore.ts +21 -13
  58. package/src/core/backup/scheduler.ts +28 -9
  59. package/src/core/backup/snapshot.ts +43 -10
  60. package/src/core/backup/store.ts +7 -19
  61. package/src/core/backup/targets.ts +69 -17
  62. package/src/core/config/index.ts +14 -0
  63. package/src/core/daemon/crash-marker.ts +141 -0
  64. package/src/core/daemon/crash.ts +9 -2
  65. package/src/core/daemon/handoff.ts +15 -0
  66. package/src/core/daemon/health-alerts.ts +297 -0
  67. package/src/core/daemon/log-reader.ts +289 -0
  68. package/src/core/doctor/index.ts +18 -2
  69. package/src/core/doctor/logs.ts +124 -0
  70. package/src/core/doctor/types.ts +1 -1
  71. package/src/core/engine/backend-controller/index.ts +1 -13
  72. package/src/core/engine/backend-router/router.ts +1 -1
  73. package/src/core/engine/dispatcher.ts +55 -2
  74. package/src/core/engine/fault-text.ts +40 -0
  75. package/src/core/engine/gateway-actions/agents/index.ts +3 -2
  76. package/src/core/engine/gateway-actions/agents/report.ts +62 -0
  77. package/src/core/engine/gateway-actions/history.ts +2 -4
  78. package/src/core/engine/gateway-actions/mesh.ts +26 -14
  79. package/src/core/engine/gateway-actions/native/exec.ts +13 -16
  80. package/src/core/engine/gateway-actions/native/teleport.ts +1 -1
  81. package/src/core/engine/gateway.ts +60 -1
  82. package/src/core/engine/turn-health.ts +222 -0
  83. package/src/core/errors.ts +2 -2
  84. package/src/core/frontend-runtime/admin-notify.ts +1 -1
  85. package/src/core/frontend-runtime/alerts.ts +130 -0
  86. package/src/core/mcp-hub/children.ts +78 -29
  87. package/src/core/mcp-hub/index.ts +21 -18
  88. package/src/core/mcp-hub/proxy-server.ts +8 -4
  89. package/src/core/mcp-hub/talon-server.ts +5 -12
  90. package/src/core/mesh/credentials/store.ts +16 -1
  91. package/src/core/mesh/devices/registry.ts +1 -17
  92. package/src/core/mesh/devices/service.ts +49 -17
  93. package/src/core/mesh/devices/teleport.ts +14 -2
  94. package/src/core/mesh/links/node-binaries.ts +13 -6
  95. package/src/core/mesh/persist.ts +22 -10
  96. package/src/core/mesh/transfers/device-files.ts +5 -23
  97. package/src/core/mesh/transfers/transfers.ts +16 -3
  98. package/src/core/models/active-model.ts +2 -55
  99. package/src/core/plugin/actions.ts +19 -20
  100. package/src/core/plugin/builtins.ts +80 -90
  101. package/src/core/plugin/index.ts +1 -4
  102. package/src/core/plugin/loader.ts +25 -33
  103. package/src/core/plugin/mcp.ts +3 -5
  104. package/src/core/plugin/registry.ts +19 -35
  105. package/src/core/plugin/types.ts +2 -5
  106. package/src/core/prompt/assemble.ts +15 -3
  107. package/src/core/scripts/lua.ts +6 -2
  108. package/src/core/tasks/table.ts +8 -2
  109. package/src/core/tools/bridge.ts +2 -4
  110. package/src/core/tools/chat/cross-send.ts +1 -1
  111. package/src/core/tools/chat/messaging.ts +1 -1
  112. package/src/core/tools/index.ts +2 -2
  113. package/src/core/tools/mcp-env.ts +2 -59
  114. package/src/core/tools/ops/agents.ts +22 -1
  115. package/src/core/tools/schemas.ts +4 -9
  116. package/src/core/vfs/fusefs.ts +0 -5
  117. package/src/core/vfs/index.ts +9 -2
  118. package/src/core/vfs/mounts/diagnostics.ts +109 -0
  119. package/src/core/vfs/mounts/proc.ts +17 -1
  120. package/src/core/vfs/workspace.ts +7 -3
  121. package/src/core/weaver/shuttle.ts +8 -1
  122. package/src/core/weaver/turn-log.ts +320 -0
  123. package/src/core/weaver/weaver.ts +48 -5
  124. package/src/frontend/discord/actions/index.ts +8 -1
  125. package/src/frontend/discord/diagnostics.ts +82 -6
  126. package/src/frontend/discord/handlers/index.ts +0 -2
  127. package/src/frontend/discord/middleware.ts +7 -13
  128. package/src/frontend/discord/runtime.ts +1 -3
  129. package/src/frontend/health/delivery.ts +115 -0
  130. package/src/frontend/health/outage.ts +116 -0
  131. package/src/frontend/native/bridge/routes/chats.ts +3 -5
  132. package/src/frontend/native/bridge/server.ts +106 -21
  133. package/src/frontend/native/index.ts +1 -1
  134. package/src/frontend/native/media/media.ts +5 -1
  135. package/src/frontend/native/runtime.ts +12 -7
  136. package/src/frontend/native/surface/handlers.ts +1 -1
  137. package/src/frontend/native/surface/memory.ts +1 -1
  138. package/src/frontend/native/surface/models.ts +3 -3
  139. package/src/frontend/native/surface/settings.ts +20 -8
  140. package/src/frontend/native/turn/context.ts +6 -8
  141. package/src/frontend/native/turn/turn-meta.ts +2 -5
  142. package/src/frontend/native/turn/turn.ts +8 -10
  143. package/src/frontend/presentation/format.ts +2 -4
  144. package/src/frontend/presentation/session-status.ts +2 -6
  145. package/src/frontend/teams/actions.ts +8 -1
  146. package/src/frontend/teams/graph.ts +0 -1
  147. package/src/frontend/teams/index.ts +1 -4
  148. package/src/frontend/teams/poll.ts +40 -2
  149. package/src/frontend/teams/runtime.ts +14 -5
  150. package/src/frontend/telegram/actions/index.ts +4 -1
  151. package/src/frontend/telegram/actions/send.ts +8 -0
  152. package/src/frontend/telegram/handlers/context.ts +13 -2
  153. package/src/frontend/telegram/handlers/delivery.ts +12 -9
  154. package/src/frontend/telegram/handlers/index.ts +0 -2
  155. package/src/frontend/telegram/index.ts +35 -9
  156. package/src/frontend/telegram/polling/poll-health.ts +110 -0
  157. package/src/frontend/telegram/userbot.ts +100 -36
  158. package/src/frontend/terminal/builtins/session.ts +2 -2
  159. package/src/frontend/terminal/index.ts +1 -3
  160. package/src/frontend/terminal/renderer.ts +2 -18
  161. package/src/frontend/whatsapp/actions/index.ts +12 -1
  162. package/src/frontend/whatsapp/actions/messaging.ts +7 -2
  163. package/src/frontend/whatsapp/connection/connection.ts +29 -9
  164. package/src/frontend/whatsapp/connection/health.ts +89 -0
  165. package/src/frontend/whatsapp/connection/identity.ts +4 -4
  166. package/src/frontend/whatsapp/runtime.ts +11 -5
  167. package/src/native/blake3.ts +28 -2
  168. package/src/native/fusefs.ts +23 -5
  169. package/src/native/registry.ts +1 -1
  170. package/src/native/warden.ts +33 -5
  171. package/src/plugins/github/index.ts +0 -1
  172. package/src/plugins/mempalace/index.ts +9 -3
  173. package/src/plugins/playwright/index.ts +2 -4
  174. package/src/plugins/playwright/provision.ts +12 -4
  175. package/src/storage/chat-settings.ts +6 -1
  176. package/src/storage/cron.ts +29 -4
  177. package/src/storage/daily-log.ts +43 -47
  178. package/src/storage/db.ts +61 -33
  179. package/src/storage/history.ts +6 -1
  180. package/src/storage/journal.ts +9 -2
  181. package/src/storage/kv.ts +19 -6
  182. package/src/storage/media-index.ts +28 -6
  183. package/src/storage/repositories/chat-settings-repo.ts +8 -2
  184. package/src/storage/repositories/sessions-repo.ts +10 -3
  185. package/src/storage/scripts.ts +24 -13
  186. package/src/storage/sessions.ts +11 -2
  187. package/src/storage/skills.ts +21 -2
  188. package/src/storage/stickers.ts +17 -3
  189. package/src/storage/triggers.ts +8 -3
  190. package/src/storage/turn-meta.ts +25 -7
  191. package/src/util/log.ts +189 -6
  192. package/src/util/logging/turn-scope.ts +85 -0
  193. package/src/util/time.ts +3 -3
  194. package/src/util/watchdog.ts +30 -0
  195. package/src/core/engine/backend-controller/legacy.ts +0 -111
package/README.md CHANGED
@@ -203,7 +203,7 @@ The WhatsApp frontend drives a real WhatsApp account over Baileys multi-device
203
203
  // The bot account's own number, E.164 digits, no "+". Omit for QR pairing.
204
204
  "pairingNumber": "353871234567",
205
205
  // Who may DM it — bare numbers or full JIDs. Empty disables DMs.
206
- "allowedJids": ["353834733284"],
206
+ "allowedJids": ["447700900102"],
207
207
  // Which groups it serves: "listed" | "with-allowed-user" | "all"
208
208
  "groupPolicy": "with-allowed-user",
209
209
  // In groups: reply only when mentioned/quoted, or to everything
@@ -487,6 +487,7 @@ Config file: `~/.talon/config.json`
487
487
  | `heartbeatIntervalMinutes` | `60` | Heartbeat interval |
488
488
  | `heartbeatModel` | --- | Model for the heartbeat agent (falls back to `model`) |
489
489
  | `heartbeatEffort` | --- | Reasoning effort for the heartbeat agent: `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Unset = the model's own default |
490
+ | `alerts` | on | Operator alerts to the admin chat — full disk, crash, error spike, dead frontend — once per fault per cooldown, plus a recovery notice: `{ "enabled": true, "cooldownMinutes": 30 }`. Active alerts also show in `talon status` |
490
491
  | `router` | --- | Plan-aware routing for background work: `{ "enabled": true, "ceilingPercent": 85 }`. Unpinned sub-agents, cron `query` jobs and heartbeats run on whichever backend has the most plan headroom, skipping any whose tightest window is at or above the ceiling. `enabled: false` restores inherit-the-caller's-backend ([Backends](docs/backends.md)) |
491
492
  | `backendBudgets` | --- | Soft token budgets for backends with no usage API, e.g. `{ "openai-agents": { "tokensPer5h": 2000000, "tokensPerDay": 8000000 } }`. Talon's own rolling ledger is measured against these so such a backend still has a headroom signal — and it is what opts an idle backend into routing. `agy` reports its real quota windows (via `agy -p /usage`); a budget there is only a fallback for when that read fails |
492
493
  | `dreamModel` | --- | Model for dream / memory consolidation (falls back to `model`) |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "5.18.2",
3
+ "version": "5.19.1",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "The Falconry",
6
6
  "license": "Apache-2.0",
@@ -101,7 +101,7 @@
101
101
  "build:fusefs": "node native/talon-fusefs/build.mjs"
102
102
  },
103
103
  "dependencies": {
104
- "@anthropic-ai/claude-agent-sdk": "^0.3.277",
104
+ "@anthropic-ai/claude-agent-sdk": "^0.3.283",
105
105
  "@anthropic-ai/sdk": "^0.104.1",
106
106
  "@brave/brave-search-mcp-server": "^2.0.75",
107
107
  "@clack/prompts": "^1.2.0",
@@ -19,14 +19,31 @@ a useful result; silence is not.
19
19
 
20
20
  ## Talking to your parent
21
21
 
22
- - `check_inbox()` drains any instructions your parent has sent you. Check it
23
- at natural milestones — after a phase of work, before a long operation, and
24
- before you report. Messages are not delivered to you any other way.
22
+ - `check_inbox()` drains anything sent to you — instructions from your parent
23
+ and notes from peers. Each message names its sender. Check it at natural
24
+ milestones: after a phase of work, before a long operation, and before you
25
+ report. Messages are not delivered to you any other way.
25
26
  - `message_parent(text)` sends an interim note (a finding worth acting on now,
26
27
  a question, a heads-up that this will take a while). Use it sparingly: each
27
28
  one wakes your parent. It does **not** end your run and does **not** count
28
29
  as your result.
29
30
 
31
+ ## Talking to your peers
32
+
33
+ Your parent may have spawned others alongside you. `list_peers()` shows them —
34
+ id, label, and what each was asked to do — and `message_peer(agent_id, text)`
35
+ sends one of them a note directly, without going through your parent.
36
+
37
+ Use it when something you found changes _their_ work and waiting would waste
38
+ it: a fact you both need, a dead end worth not repeating, a correction to
39
+ something you told them earlier. Don't narrate your progress at them — a peer
40
+ pays for every message with context it could have spent on its own job.
41
+
42
+ You can only address peers, and they see your note at their next
43
+ `check_inbox`, so it is not an interrupt. Your report still goes to your
44
+ parent: peer messages are for coordination, never a substitute for
45
+ `report_result`.
46
+
30
47
  ## Delegating further
31
48
 
32
49
  {% if canSpawn %}You may spawn your own sub-agents with `spawn_agent` (current depth {{depth}}, cap {{maxDepth}}) when the work genuinely splits into independent pieces. You are then responsible for them: `wait_for_agent`, `send_to_agent`, `kill_agent`, and folding their reports into yours.{% else %}You are at the maximum sub-agent depth ({{maxDepth}}) — `spawn_agent` will be refused. Do this work yourself.{% endif %}
package/src/app.ts CHANGED
@@ -42,6 +42,12 @@ import {
42
42
  handleUncaughtException,
43
43
  handleUnhandledRejection,
44
44
  } from "./core/daemon/crash.js";
45
+ import { writeCrashMarker } from "./core/daemon/crash-marker.js";
46
+ import {
47
+ announceLastCrash,
48
+ startHealthAlerts,
49
+ stopHealthAlerts,
50
+ } from "./core/daemon/health-alerts.js";
45
51
  import { log, logError, logWarn } from "./util/log.js";
46
52
  import { bootPhase, bootReport } from "./core/daemon/boot-timer.js";
47
53
  import {
@@ -352,6 +358,7 @@ async function gracefulShutdown(signal: string): Promise<void> {
352
358
  await shutdownStep("sub-agents", shutdownAgents);
353
359
  await shutdownStep("watchdog", stopWatchdog);
354
360
  await shutdownStep("resource sampler", stopResourceSampler);
361
+ await shutdownStep("health alerts", stopHealthAlerts);
355
362
  await shutdownStep("upload cleanup", stopUploadCleanup);
356
363
  await shutdownStep("mcp hub", async () => {
357
364
  const { shutdownHub } = await import("./core/mcp-hub/index.js");
@@ -452,6 +459,11 @@ async function main(): Promise<void> {
452
459
  await import("./core/frontend-runtime/admin-notify.js");
453
460
  await notifyAdmin(restoreReport);
454
461
  }
462
+ // Same reasoning for a crash: the process that died couldn't say so,
463
+ // so the marker it left is announced now. The probes start here too —
464
+ // an alert raised before a frontend can carry it is wasted.
465
+ announceLastCrash();
466
+ startHealthAlerts();
455
467
 
456
468
  const bootMs = Math.round(process.uptime() * 1000);
457
469
  recordBootMetrics(bootMs);
@@ -465,6 +477,7 @@ async function main(): Promise<void> {
465
477
 
466
478
  main().catch((err) => {
467
479
  crashCleanup(crashHooks);
480
+ crashStep("crash marker", () => writeCrashMarker("startup", err));
468
481
  crashStep("startup report", () =>
469
482
  logError("bot", "Fatal startup error", err),
470
483
  );
@@ -83,10 +83,10 @@ import {
83
83
  // The SDK's PostToolBatch hook is the canonical loop-terminator — it returns
84
84
  // `{ continue: false }` after `end_turn`/`send`, and the SDK is supposed to
85
85
  // emit a `result` SDKMessage and close the async iterator immediately after.
86
- // In practice (observed 2026-05-19 14:52Z, chat 352042062, contextTokens=251464,
87
- // numApiCalls=50) the SDK can emit `result` and then ghost — the for-await loop
88
- // stays parked forever, holding the dispatcher context and the typing-indicator
89
- // pulse for hours until someone manually `/restart`s.
86
+ // In practice (observed 2026-05-19 on a ~250k-token, 50-API-call turn) the
87
+ // SDK can emit `result` and then ghost — the for-await loop stays parked
88
+ // forever, holding the dispatcher context and the typing-indicator pulse for
89
+ // hours until someone manually `/restart`s.
90
90
  //
91
91
  // Workaround: arm a short timer the moment `result` is processed. If the
92
92
  // iterator hasn't closed by the grace deadline, abort the controller and
@@ -35,9 +35,11 @@ export async function waitForMcpServersReady(
35
35
  while (Date.now() < deadline) {
36
36
  let statuses;
37
37
  try {
38
- statuses = await qi.mcpServerStatus();
38
+ // Bound the call itself: a CLI that never answers the control
39
+ // request would otherwise park here past the deadline.
40
+ statuses = await beforeDeadline(qi.mcpServerStatus(), deadline);
39
41
  } catch {
40
- return; // status query unsupported/failed — don't block the turn
42
+ return; // status query unsupported/failed/timed out — don't block the turn
41
43
  }
42
44
  // Best-effort: a backend (or stub) may report status in an unexpected
43
45
  // shape (undefined / non-array). Treat anything non-iterable as
@@ -52,3 +54,15 @@ export async function waitForMcpServersReady(
52
54
  await new Promise((resolve) => setTimeout(resolve, pollMs));
53
55
  }
54
56
  }
57
+
58
+ /** `promise`, or a rejection once `deadline` (epoch ms) passes first. */
59
+ function beforeDeadline<T>(promise: Promise<T>, deadline: number): Promise<T> {
60
+ let timer: ReturnType<typeof setTimeout> | undefined;
61
+ const expired = new Promise<never>((_, reject) => {
62
+ timer = setTimeout(
63
+ () => reject(new Error("mcpServerStatus timed out")),
64
+ Math.max(0, deadline - Date.now()),
65
+ );
66
+ });
67
+ return Promise.race([promise, expired]).finally(() => clearTimeout(timer));
68
+ }
@@ -279,6 +279,34 @@ async function formatAndAppendMessage(
279
279
  * spawner (us) and contains the env vars Talon set when launching the SDK
280
280
  * subprocess via MCP launcher. SIGTERM with a short grace, then SIGKILL.
281
281
  */
282
+ /**
283
+ * Decide whether a `/proc` entry is a run orphan this sweep may kill.
284
+ *
285
+ * Carrying the chat id is necessary but NOT sufficient. Every chat-scoped
286
+ * child Talon spawns inherits `TALON_CHAT_ID` — including trigger watchers
287
+ * (`core/background/triggers/spawn.ts` sets it alongside `TALON_TRIGGER_ID`)
288
+ * and their descendants, which are long-lived by design and belong to no run.
289
+ * Matching on the chat id alone made every orphan sweep kill that chat's
290
+ * triggers: the warden respawned them, the next sweep killed them again, and
291
+ * the only visible symptom was a trigger stuck in "errored" with no output.
292
+ *
293
+ * Two independent guards, because this function issues SIGKILL:
294
+ * - refuse anything tagged `TALON_TRIGGER_ID` (a trigger, or its child);
295
+ * - require the argv to actually be the `claude` SDK binary, which is the
296
+ * only thing this sweep was ever meant to reap.
297
+ */
298
+ export function isEvictableOrphan(
299
+ envEntries: string[],
300
+ argv: string[],
301
+ target: string,
302
+ ): boolean {
303
+ if (!envEntries.includes(target)) return false;
304
+ if (envEntries.some((entry) => entry.startsWith("TALON_TRIGGER_ID="))) {
305
+ return false;
306
+ }
307
+ return argv.some((arg) => arg === "claude" || arg.endsWith("/claude"));
308
+ }
309
+
282
310
  export async function evictOrphanSubprocesses(contextLabel: string): Promise<{
283
311
  found: number;
284
312
  termed: number;
@@ -306,12 +334,14 @@ export async function evictOrphanSubprocesses(contextLabel: string): Promise<{
306
334
  if (!Number.isInteger(pid) || pid === myPid) continue;
307
335
  try {
308
336
  const environRaw = await readFile(`/proc/${pid}/environ`, "utf-8");
337
+ const argvRaw = await readFile(`/proc/${pid}/cmdline`, "utf-8");
309
338
  // /proc/<pid>/environ is NUL-delimited. Split on \0 and match exact
310
339
  // entries — a raw .includes() can false-positive on other vars whose
311
340
  // value happens to contain the substring. Since this code can SIGKILL,
312
341
  // err on the side of strict matching. (Copilot review on #144.)
313
- const envEntries = environRaw.split("\0");
314
- if (envEntries.includes(target)) {
342
+ if (
343
+ isEvictableOrphan(environRaw.split("\0"), argvRaw.split("\0"), target)
344
+ ) {
315
345
  matched.push(pid);
316
346
  }
317
347
  } catch {
@@ -2,8 +2,8 @@
2
2
  * Typed stream processing helpers for SDK messages.
3
3
  *
4
4
  * Each function operates on a properly narrowed SDK message type —
5
- * no Record<string, unknown> casts. The StreamState accumulator
6
- * replaces the scattered local variables from the original handler.
5
+ * no Record<string, unknown> casts. StreamState is the turn's single
6
+ * accumulator.
7
7
  */
8
8
 
9
9
  import type {
@@ -302,7 +302,7 @@ export function isChatGptModelMismatchError(message: string): boolean {
302
302
 
303
303
  /**
304
304
  * Detect the *silent* OAuth-incompat exit shape — the one that hit
305
- * Pandario on 2026-05-20 at 23:13Z.
305
+ * a group chat on 2026-05-20 at 23:13Z.
306
306
  *
307
307
  * On a free ChatGPT-OAuth credential the Codex CLI silently rejects
308
308
  * most model strings (only `gpt-5.5` is verified working). Crucially,
@@ -205,17 +205,6 @@ async function maybeFallbackForChatGptMismatch(
205
205
  return await handleMessage({ ...params, model: fallbackModel }, true);
206
206
  }
207
207
 
208
- // ── Model resolution ────────────────────────────────────────────────────────
209
-
210
- /**
211
- * Codex accepts arbitrary model strings; we pass through whatever the
212
- * caller resolved (chat-settings → config) and fall back to the auth-aware
213
- * default: `gpt-5-codex` when an API key is present, `gpt-5.5` when only
214
- * ChatGPT OAuth is configured (because `gpt-5-codex` is rejected with a
215
- * 400 on ChatGPT-mode accounts). A model known to be OAuth-incompat on a
216
- * ChatGPT-OAuth account is swapped pre-emptively rather than letting the
217
- * first turn fail.
218
- */
219
208
  /**
220
209
  * The turn's user prompt: the shared framing every backend emits, plus
221
210
  * whatever this turn's memory retrieval produced. `formatUserPrompt` is
@@ -233,6 +222,17 @@ function buildTurnPrompt(params: QueryParams): string {
233
222
  });
234
223
  }
235
224
 
225
+ // ── Model resolution ────────────────────────────────────────────────────────
226
+
227
+ /**
228
+ * Codex accepts arbitrary model strings; we pass through whatever the
229
+ * caller resolved (chat-settings → config) and fall back to the auth-aware
230
+ * default: `gpt-5-codex` when an API key is present, `gpt-5.5` when only
231
+ * ChatGPT OAuth is configured (because `gpt-5-codex` is rejected with a
232
+ * 400 on ChatGPT-mode accounts). A model known to be OAuth-incompat on a
233
+ * ChatGPT-OAuth account is swapped pre-emptively rather than letting the
234
+ * first turn fail.
235
+ */
236
236
  function resolveCodexModel(chatId: string, requested: string | undefined) {
237
237
  const authInfo = getCodexAuthInfo();
238
238
  const authAwareDefault =
@@ -85,15 +85,10 @@ export function initCodexAgent(
85
85
  // credential fingerprint has changed since last load). The store is
86
86
  // a no-op for non-OAuth credentials but the loader is cheap, so we
87
87
  // call it for all modes uniformly.
88
- // Fire-and-forget: the loader keeps its async signature (historical;
89
- // kv reads are sync now) but `initCodexAgent` is sync and changing it
90
- // to async would ripple through every caller in bootstrap.ts.
91
- // The store is best-effort anyway — if a turn races with the
92
- // first load, `isKnownOAuthIncompat` defaults to false and the
93
- // turn proceeds without the runtime-learned filter (the curated
94
- // list still applies). Errors are already swallowed inside
95
- // `loadOAuthIncompatStore`, so the .catch() here is purely
96
- // defensive against a synchronous throw in the function body.
88
+ // Fire-and-forget: `initCodexAgent` is sync. The store is best-effort —
89
+ // a turn that races the first load runs without the runtime-learned
90
+ // filter (the curated list still applies). The loader swallows its own
91
+ // errors; the .catch() only guards a throw in its body.
97
92
  loadOAuthIncompatStore(computeAuthFingerprint(authInfo)).catch(() => {
98
93
  /* logged inside loadOAuthIncompatStore */
99
94
  });
@@ -12,8 +12,7 @@
12
12
  * hub (`core/mcp-hub`) — Codex connects with `mcp_servers.<name>.url`
13
13
  * (the same transport `codex mcp add --url` configures). The hub hosts
14
14
  * Talon's tools in-process and shares plugin/brave children across
15
- * chats, so Codex no longer spawns a subprocess pair per server per
16
- * turn:
15
+ * chats, so Codex spawns no MCP subprocesses of its own:
17
16
  *
18
17
  * {
19
18
  * mcp_servers: {
@@ -75,6 +75,7 @@
75
75
  * stay synchronous (they only touch the in-memory set).
76
76
  */
77
77
 
78
+ import { scryptSync } from "node:crypto";
78
79
  import { logDebug } from "../../util/log.js";
79
80
  import { files } from "../../util/paths.js";
80
81
  import { kvGet, kvSet } from "../../storage/kv.js";
@@ -118,7 +119,7 @@ let memoryStore: InMemoryStore | null = null;
118
119
  *
119
120
  * Fingerprint shape:
120
121
  * - `mode:source` for ChatGPT OAuth (no token to hash)
121
- * - `mode:source:<first 16 chars of key>` for api-key billing
122
+ * - `mode:source:<scrypt(key), 16 hex>` for api-key billing
122
123
  * - `mode:source` for missing/empty
123
124
  *
124
125
  * Constant-folded by `mode === "none"` — no learning happens without
@@ -127,9 +128,12 @@ let memoryStore: InMemoryStore | null = null;
127
128
  export function computeAuthFingerprint(info: CodexAuthInfo): string {
128
129
  const base = `${info.mode}:${info.source}`;
129
130
  if (info.mode === "api-key" && info.apiKey) {
130
- // First 16 chars is enough to distinguish keys for fingerprint
131
- // purposes without storing the full secret on disk.
132
- return `${base}:${info.apiKey.slice(0, 16)}`;
131
+ // A digest, never key material: the fingerprint is persisted in kv
132
+ // and appears in debug logs.
133
+ // scrypt, not a bare hash: the input is a credential. Computed once
134
+ // per codex init.
135
+ const digest = scryptSync(info.apiKey, "talon.codex.fingerprint", 8);
136
+ return `${base}:${digest.toString("hex")}`;
133
137
  }
134
138
  return base;
135
139
  }
@@ -40,7 +40,7 @@ import { toCodexReasoningEffort } from "./effort.js";
40
40
  * `config.heartbeatModel ?? config.model`. If that's an OAuth-incompat
41
41
  * id (curated `apiKeyOnly: true` or runtime-learned) AND the active
42
42
  * Codex credential is ChatGPT OAuth, swap to `gpt-5.5` to avoid the
43
- * silent exit-1 failure mode that hit Pandario on 2026-05-20 23:13Z.
43
+ * silent exit-1 failure mode that hit a group chat on 2026-05-20 23:13Z.
44
44
  *
45
45
  * Returns the resolved model id, whether a swap occurred, and an
46
46
  * optional reason string for the run log.
@@ -31,9 +31,10 @@
31
31
  * shared prompt vocabulary applies uniformly.
32
32
  */
33
33
  import { tool } from "@openai/agents";
34
- import { spawn } from "node:child_process";
34
+ import { spawn, type ChildProcess } from "node:child_process";
35
35
  import { readFile, writeFile, mkdir, glob } from "node:fs/promises";
36
36
  import { dirname, resolve as resolvePath } from "node:path";
37
+ import { createOutputCapture } from "../../util/exec-output.js";
37
38
  import { expandFsPath as expandPath } from "../../util/fs-path.js";
38
39
 
39
40
  // ── Read ────────────────────────────────────────────────────────────────────
@@ -204,6 +205,12 @@ const editTool = tool({
204
205
 
205
206
  const BASH_DEFAULT_TIMEOUT_MS = 30_000;
206
207
  const BASH_MAX_TIMEOUT_MS = 600_000;
208
+ /**
209
+ * After a timeout kill, how long to wait for `close` before settling with
210
+ * what was captured. `close` waits for the stdio pipes to drain, which a
211
+ * descendant that escaped the kill can hold open indefinitely.
212
+ */
213
+ const BASH_CLOSE_GRACE_MS = 2_000;
207
214
 
208
215
  interface BashInput {
209
216
  command: string;
@@ -235,43 +242,64 @@ function runShell(
235
242
  ],
236
243
  }
237
244
  : { cmd: "bash", args: ["-lc", command] };
245
+ // detached → own process group on POSIX, so the timeout kill reaches
246
+ // the shell's children too. A surviving child (the `sleep` in
247
+ // `sleep 30; echo`) holds stdout open, and `close` waits for it.
248
+ const detached = process.platform !== "win32";
238
249
  const child = spawn(shell.cmd, shell.args, {
239
250
  cwd: process.cwd(),
240
251
  env: process.env,
241
252
  stdio: ["ignore", "pipe", "pipe"],
253
+ detached,
242
254
  });
243
- let stdout = "";
244
- let stderr = "";
245
- child.stdout.on("data", (b: Buffer) => {
246
- stdout += b.toString("utf8");
247
- });
248
- child.stderr.on("data", (b: Buffer) => {
249
- stderr += b.toString("utf8");
250
- });
255
+ // Bounded: an unbounded stream (`yes`, a verbose build) otherwise grows
256
+ // until V8's string limit throws inside the data listener.
257
+ const out = createOutputCapture();
258
+ const err = createOutputCapture();
259
+ child.stdout.on("data", out.push);
260
+ child.stderr.on("data", err.push);
261
+ let settled = false;
262
+ let timedOut = false;
263
+ let closeGrace: NodeJS.Timeout | undefined;
264
+ const finish = (code: number, stderr = err.value()): void => {
265
+ if (settled) return;
266
+ settled = true;
267
+ clearTimeout(killer);
268
+ if (closeGrace) clearTimeout(closeGrace);
269
+ resolveResult({ stdout: out.value(), stderr, code, timedOut });
270
+ };
251
271
  const killer = setTimeout(() => {
252
272
  timedOut = true;
253
- child.kill("SIGKILL");
273
+ killGroup(child, detached);
274
+ // A descendant that left the group can still hold the pipes open;
275
+ // don't let it pin the tool call past the timeout.
276
+ closeGrace = setTimeout(() => finish(-1), BASH_CLOSE_GRACE_MS);
254
277
  }, timeoutMs);
255
- let timedOut = false;
256
- child.on("error", (err) => {
257
- // spawn failed (e.g. bash not in PATH on Windows). Resolve instead
258
- // of letting Node emit an uncaught error — the tool returns the
259
- // diagnostic so callers can surface it rather than crashing.
260
- clearTimeout(killer);
261
- resolveResult({
262
- stdout,
263
- stderr: err.message,
264
- code: -1,
265
- timedOut: false,
266
- });
267
- });
268
- child.on("close", (code) => {
269
- clearTimeout(killer);
270
- resolveResult({ stdout, stderr, code: code ?? -1, timedOut });
271
- });
278
+ // spawn failed (e.g. bash not in PATH on Windows). Resolve instead
279
+ // of letting Node emit an uncaught error — the tool returns the
280
+ // diagnostic so callers can surface it rather than crashing.
281
+ child.on("error", (spawnErr) => finish(-1, spawnErr.message));
282
+ child.on("close", (code) => finish(code ?? -1));
272
283
  });
273
284
  }
274
285
 
286
+ /** SIGKILL the child's whole process group, falling back to the child. */
287
+ function killGroup(child: ChildProcess, detached: boolean): void {
288
+ if (detached && child.pid) {
289
+ try {
290
+ process.kill(-child.pid, "SIGKILL");
291
+ return;
292
+ } catch {
293
+ // group already gone — fall through to the direct kill
294
+ }
295
+ }
296
+ try {
297
+ child.kill("SIGKILL");
298
+ } catch {
299
+ // already dead
300
+ }
301
+ }
302
+
275
303
  const bashTool = tool({
276
304
  name: "Bash",
277
305
  description:
@@ -93,9 +93,9 @@ const openAIAgentsFactory: BackendFactory = {
93
93
 
94
94
  return {
95
95
  backend,
96
- // Cleanup: close every per-chat MCP bundle in the pool so the
97
- // ~15 plugin subprocesses per active chat don't outlive the
98
- // backend itself, then drop the cached state.
96
+ // Cleanup: close every per-chat MCP bundle in the pool so its hub
97
+ // sessions don't outlive the backend itself, then drop the cached
98
+ // state.
99
99
  cleanup: async () => {
100
100
  await releaseAllBundles();
101
101
  resetState();
@@ -118,8 +118,7 @@ function buildTurnPrompt(
118
118
 
119
119
  /**
120
120
  * Acquire the per-chat MCP bundle. Persistent across turns — built on
121
- * first use, kept alive until `releaseBundle(chatId)`. Avoids the
122
- * ~15-subprocess re-spawn the original per-turn build caused.
121
+ * first use, kept until the backend's cleanup releases the pool.
123
122
  */
124
123
  async function acquireMcpBundle(
125
124
  chatId: string,
@@ -366,7 +365,7 @@ export async function handleMessage(
366
365
  } catch (err) {
367
366
  // Swallow the terminator abort — the turn completed via a delivery tool.
368
367
  if (!isTerminatorAbort(streamState, err)) {
369
- // MCP bundle is retained across a retry — subprocesses are
368
+ // MCP bundle is retained across a retry — its servers are
370
369
  // stateless wrt the model conversation. See `mcp-pool.ts`.
371
370
  const outcome = await applyRetryDecision({
372
371
  err,
@@ -400,8 +399,7 @@ export async function handleMessage(
400
399
  activeAborts.delete(chatId);
401
400
  }
402
401
  // MCP bundle is NOT closed here — it persists across turns via the
403
- // pool in `mcp-pool.ts`. Release happens on chat rebind, `/reset`, and
404
- // at backend cleanup.
402
+ // pool in `mcp-pool.ts` and is released at backend cleanup.
405
403
  }
406
404
 
407
405
  // ── Post-loop accounting ──────────────────────────────────────────────────
@@ -4,17 +4,11 @@
4
4
  * Every server is a lightweight `MCPServerStreamableHttp` client
5
5
  * pointing at the daemon's MCP hub (`core/mcp-hub`): Talon's own tools
6
6
  * run in-process there, and plugin/brave servers are hub-managed
7
- * children shared across chats and reaped when idle.
7
+ * children shared across chats and reaped when idle. A bundle is just
8
+ * HTTP client objects; the process count is owned and bounded by the hub.
8
9
  *
9
- * Historical note: this pool used to hold one **subprocess set** per
10
- * chat (every plugin × every chat, held until release) — the daemon's
11
- * memory grew linearly with the number of chats. With the hub, a
12
- * bundle is just HTTP client objects; the process count is owned and
13
- * bounded by the hub.
14
- *
15
- * The bundle is still cached per chat (and released on reset/rebind)
16
- * so `cacheToolsList` survives across turns and each turn skips the
17
- * connect handshake.
10
+ * The bundle is cached per chat so `cacheToolsList` survives across
11
+ * turns and each turn skips the connect handshake.
18
12
  *
19
13
  * Concurrency: `getOrCreateBundle` serialises the build-or-return
20
14
  * decision on a per-chat in-flight Promise so two concurrent turns from
@@ -36,18 +30,13 @@ import { frontendsForChat } from "../runtime/frontends.js";
36
30
  import { log, logWarn } from "../../util/log.js";
37
31
 
38
32
  /**
39
- * One per-chat bundle. Subprocesses stay alive until `close()` is
33
+ * One per-chat bundle. Its hub sessions stay open until `close()` is
40
34
  * called via `releaseBundle()` / `releaseAllBundles()`.
41
35
  */
42
36
  export interface OpenAIAgentsMcpBundle {
43
- /**
44
- * Connected MCP servers ready to pass to `new Agent({ mcpServers })`.
45
- * `connectMcpServers` returns the structural `MCPServer` type — the
46
- * underlying instances are `MCPServerStdio` but the Agent constructor
47
- * only needs the interface.
48
- */
37
+ /** Connected MCP servers ready to pass to `new Agent({ mcpServers })`. */
49
38
  servers: MCPServer[];
50
- /** Close every spawned subprocess. Safe to call multiple times. */
39
+ /** Close every server's hub session. Safe to call multiple times. */
51
40
  close: () => Promise<void>;
52
41
  /** Servers that failed to connect — exposed for diagnostics. */
53
42
  failed: ReadonlyArray<{ name: string; error: string }>;
@@ -115,14 +104,8 @@ export async function getOrCreateBundle(
115
104
  * Close the bundle for `chatId` and drop it from the pool. No-op when
116
105
  * the chat has no live bundle.
117
106
  *
118
- * Call when:
119
- * - The chat rebinds to a non-openai-agents backend.
120
- * - The user runs `/reset`.
121
- * - The chat is destroyed.
122
- *
123
107
  * If `getOrCreateBundle` is in flight when called, releases the bundle
124
- * once the in-flight build resolves to avoid leaving an unreleased
125
- * subprocess set.
108
+ * once the in-flight build resolves so it is not left open.
126
109
  */
127
110
  export async function releaseBundle(chatId: string): Promise<void> {
128
111
  // If a build is in flight, wait for it then close the result.
@@ -152,8 +135,8 @@ export async function releaseBundle(chatId: string): Promise<void> {
152
135
 
153
136
  /**
154
137
  * Close every live bundle. Used by the backend factory's `cleanup`
155
- * hook so unbinding the openai-agents backend leaves no orphan MCP
156
- * subprocesses.
138
+ * hook so unbinding the openai-agents backend leaves no hub sessions
139
+ * behind.
157
140
  */
158
141
  export async function releaseAllBundles(): Promise<void> {
159
142
  const ids = [...bundles.keys()];
@@ -253,12 +253,6 @@ export async function runRemoteChatTurn<TClient extends RemoteAgentClient>(
253
253
  });
254
254
  }
255
255
 
256
- /**
257
- * If the SSE loop missed the usage info, fall back to the session summary
258
- * endpoint (which always reflects the final server state). Best-effort:
259
- * session summaries can race on cancellation, so a failure leaves the
260
- * counts at zero.
261
- */
262
256
  /**
263
257
  * The turn's user prompt: the shared framing every backend emits, plus
264
258
  * whatever this turn's memory retrieval produced. `formatUserPrompt` is
@@ -276,6 +270,12 @@ function buildTurnPrompt(params: QueryParams): string {
276
270
  });
277
271
  }
278
272
 
273
+ /**
274
+ * If the SSE loop missed the usage info, fall back to the session summary
275
+ * endpoint (which always reflects the final server state). Best-effort:
276
+ * session summaries can race on cancellation, so a failure leaves the
277
+ * counts at zero.
278
+ */
279
279
  async function fillUsageFromSummary(
280
280
  oc: RemoteSessionClient,
281
281
  sessionId: string,
@@ -39,11 +39,7 @@ import {
39
39
  type RemoteAssistantInfo,
40
40
  } from "./session-helpers.js";
41
41
  import { log, logDebug } from "../../util/log.js";
42
-
43
- /** Format an error for a debug log line. */
44
- function errMsg(err: unknown): string {
45
- return err instanceof Error ? err.message : String(err);
46
- }
42
+ import { errMsg } from "./state.js";
47
43
 
48
44
  // ── Streaming timing ───────────────────────────────────────────────────────
49
45
 
@@ -19,7 +19,6 @@
19
19
  * - Bindings (server-bindings.ts, chat-turn.ts, turn.ts, factory.ts) —
20
20
  * the helpers closed over one backend's state, the SSE-driven turn,
21
21
  * the chat-turn orchestration, and the registry factory composition.
22
- * This is where the code that used to be copied per backend lives.
23
22
  *
24
23
  * - Profiles (`profiles/bind.ts` + `profiles/{kilo,opencode}.ts`) —
25
24
  * `bindRemoteProfile` closes all of the above over one driver's