talon-agent 5.10.0 → 5.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -508,6 +508,33 @@ Config file: `~/.talon/config.json`
508
508
  | `memory` | --- | Long-term memory backend selection: `mempalace` or `mem0` (see above) |
509
509
  | `mempalace` | --- | Legacy MemPalace plugin config (prefer `memory`) |
510
510
  | `playwright` | --- | Playwright plugin config (see above) |
511
+ | `triggers` | --- | Per-chat trigger caps, e.g. `{ "maxActivePerChat": 5, "maxPersistentPerChat": 3 }` ([Scaling limits](#scaling-limits)) |
512
+ | `agents` | --- | Sub-agent caps: `{ "maxConcurrent": 6, "maxDepth": 2, "defaultTimeoutMs": 900000 }` ([docs/agents.md](docs/agents.md)) |
513
+
514
+ ### Scaling limits
515
+
516
+ Two caps bound how much background work Talon keeps alive. Both keep their
517
+ historical defaults and are raised in `~/.talon/config.json`; the error a
518
+ capped tool returns names the key to raise.
519
+
520
+ | Key | Default | Bounds | What it caps |
521
+ | ------------------------------- | ------- | ------ | ------------------------------------------------------------ |
522
+ | `triggers.maxActivePerChat` | `5` | 1–50 | Active (running or pending) triggers per chat |
523
+ | `triggers.maxPersistentPerChat` | unset | 1–50 | Optional separate budget for persistent triggers (see below) |
524
+ | `agents.maxConcurrent` | `6` | 1–64 | Live sub-agents daemon-wide, children included |
525
+
526
+ With `maxPersistentPerChat` unset, persistent and ad-hoc triggers share
527
+ `maxActivePerChat`, exactly as before. Set it and the two draw from separate
528
+ budgets: persistent triggers count only against `maxPersistentPerChat`, and
529
+ `maxActivePerChat` then bounds ad-hoc (non-persistent) triggers only — so a
530
+ chat running long-lived watchers still has room for a short CI wait. Caps are
531
+ checked at `trigger_create`; lowering one never kills a trigger already
532
+ running, and persistent triggers resumed after a restart are not re-checked.
533
+
534
+ ```json
535
+ "triggers": { "maxActivePerChat": 5, "maxPersistentPerChat": 6 },
536
+ "agents": { "maxConcurrent": 12 }
537
+ ```
511
538
 
512
539
  ### Background reasoning effort
513
540
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "5.10.0",
3
+ "version": "5.11.0",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
package/src/app.ts CHANGED
@@ -6,7 +6,7 @@
6
6
  * are loaded dynamically — only the selected platform's dependencies are required.
7
7
  */
8
8
 
9
- import { getFrontends } from "./core/config/index.js";
9
+ import { ConfigFileError, getFrontends } from "./core/config/index.js";
10
10
  import { startUploadCleanup, stopUploadCleanup } from "./core/vfs/workspace.js";
11
11
  import { flushDatabase } from "./storage/db.js";
12
12
  import { getActiveCount, stopAllTurns } from "./core/engine/dispatcher.js";
@@ -40,6 +40,7 @@ import {
40
40
  crashCleanup,
41
41
  crashStep,
42
42
  handleUncaughtException,
43
+ handleUnhandledRejection,
43
44
  } from "./core/daemon/crash.js";
44
45
  import { log, logError, logWarn } from "./util/log.js";
45
46
  import { bootPhase, bootReport } from "./core/daemon/boot-timer.js";
@@ -119,9 +120,31 @@ async function applyStagedRestore(): Promise<string | null> {
119
120
  );
120
121
  }
121
122
 
122
- const restoreReport = await bootPhase("staged restore", applyStagedRestore);
123
+ /**
124
+ * A present-but-invalid config.json is fatal at startup: print the file
125
+ * path and every problem, then exit non-zero. Booting on defaults instead
126
+ * would put the daemon in a surprising state (wrong frontend, no plugins).
127
+ */
128
+ async function withConfigGuard<T>(fn: () => Promise<T>): Promise<T> {
129
+ try {
130
+ return await fn();
131
+ } catch (err) {
132
+ if (err instanceof ConfigFileError) {
133
+ logError("config", err.message);
134
+ console.error(`talon: ${err.message}`);
135
+ process.exit(1);
136
+ }
137
+ throw err;
138
+ }
139
+ }
140
+
141
+ const restoreReport = await withConfigGuard(() =>
142
+ bootPhase("staged restore", applyStagedRestore),
143
+ );
123
144
 
124
- const { config } = await bootPhase("bootstrap", () => bootstrap());
145
+ const { config } = await withConfigGuard(() =>
146
+ bootPhase("bootstrap", () => bootstrap()),
147
+ );
125
148
 
126
149
  // Record this process as the daemon. The gateway port is appended once
127
150
  // the gateway binds (it may fall back from the default on EADDRINUSE).
@@ -361,12 +384,7 @@ process.on("uncaughtException", (err) =>
361
384
  handleUncaughtException(err, crashHooks),
362
385
  );
363
386
 
364
- process.on("unhandledRejection", (reason) => {
365
- logWarn(
366
- "bot",
367
- `Unhandled rejection: ${reason instanceof Error ? reason.message : reason}`,
368
- );
369
- });
387
+ process.on("unhandledRejection", handleUnhandledRejection);
370
388
 
371
389
  // ── Start ────────────────────────────────────────────────────────────────────
372
390
 
@@ -21,7 +21,14 @@ import type {
21
21
  PlanWindow,
22
22
  } from "../../core/agent-runtime/capabilities.js";
23
23
 
24
- const USAGE_ENDPOINT = "https://api.anthropic.com/api/oauth/usage";
24
+ // `cedar_ember=1` asks the endpoint to include banked limit resets (the
25
+ // claude.ai "Reset for free" grants); `skip_spend=1` drops the spend block we
26
+ // don't render. Resets are only reported to the CLI surface — any other
27
+ // user agent gets `ineligible_reason: "surface"` — so the request identifies
28
+ // as the CLI, which is what the Agent SDK runs anyway.
29
+ const USAGE_ENDPOINT =
30
+ "https://api.anthropic.com/api/oauth/usage?cedar_ember=1&skip_spend=1";
31
+ const CLI_USER_AGENT = "claude-cli/2.1.280 (external, cli)";
25
32
  const REQUEST_TIMEOUT_MS = 5_000;
26
33
  const CACHE_TTL_MS = 60_000;
27
34
 
@@ -77,6 +84,48 @@ function windowLabel(limit: RawLimit): string | undefined {
77
84
  return undefined;
78
85
  }
79
86
 
87
+ interface RawResetGrant {
88
+ resets_left?: number;
89
+ ends_at?: string | null;
90
+ paused?: boolean;
91
+ }
92
+
93
+ /**
94
+ * Banked limit resets still usable: unpaused grants whose window hasn't
95
+ * closed. Returns the count and the soonest deadline among grants that still
96
+ * hold a reset, or undefined when there's nothing to offer.
97
+ */
98
+ export function parseBankedResets(
99
+ body: unknown,
100
+ now = Date.now(),
101
+ ): { count: number; expiresAt?: string } | undefined {
102
+ const program = (body as { cedar_ember?: unknown } | null)?.cedar_ember as
103
+ { eligible?: boolean; grants?: unknown } | null | undefined;
104
+ if (!program || program.eligible === false || !Array.isArray(program.grants))
105
+ return undefined;
106
+
107
+ let count = 0;
108
+ let expiresAt: string | undefined;
109
+ for (const grant of program.grants as RawResetGrant[]) {
110
+ const left = grant.resets_left;
111
+ if (typeof left !== "number" || !Number.isFinite(left) || left <= 0)
112
+ continue;
113
+ if (grant.paused === true) continue;
114
+ const ends =
115
+ typeof grant.ends_at === "string" ? Date.parse(grant.ends_at) : NaN;
116
+ if (Number.isFinite(ends) && ends <= now) continue;
117
+ count += Math.floor(left);
118
+ if (
119
+ typeof grant.ends_at === "string" &&
120
+ Number.isFinite(ends) &&
121
+ (!expiresAt || ends < Date.parse(expiresAt))
122
+ )
123
+ expiresAt = grant.ends_at;
124
+ }
125
+ if (count <= 0) return undefined;
126
+ return { count, ...(expiresAt ? { expiresAt } : {}) };
127
+ }
128
+
80
129
  export function parsePlanUsage(
81
130
  body: unknown,
82
131
  subscriptionType?: string,
@@ -103,9 +152,16 @@ export function parsePlanUsage(
103
152
  }
104
153
 
105
154
  if (windows.length === 0) return undefined;
155
+ const banked = parseBankedResets(body);
106
156
  return {
107
157
  ...(subscriptionType ? { plan: subscriptionType } : {}),
108
158
  windows,
159
+ ...(banked
160
+ ? {
161
+ resetsAvailable: banked.count,
162
+ ...(banked.expiresAt ? { resetsExpireAt: banked.expiresAt } : {}),
163
+ }
164
+ : {}),
109
165
  fetchedAt: Date.now(),
110
166
  };
111
167
  }
@@ -119,6 +175,7 @@ async function load(): Promise<PlanUsage | undefined> {
119
175
  headers: {
120
176
  Authorization: `Bearer ${creds.accessToken}`,
121
177
  "anthropic-beta": "oauth-2025-04-20",
178
+ "User-Agent": CLI_USER_AGENT,
122
179
  },
123
180
  signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
124
181
  });
package/src/bootstrap.ts CHANGED
@@ -607,7 +607,10 @@ function initRecurringAgents(
607
607
  * same delivery shape, so they are wired together.
608
608
  */
609
609
  function initWakeSubsystems(config: TalonConfig): void {
610
- initTriggers({ execute: dispatcherExecute });
610
+ initTriggers({
611
+ execute: dispatcherExecute,
612
+ ...(config.triggers ? { caps: config.triggers } : {}),
613
+ });
611
614
  initAgents({
612
615
  execute: dispatcherExecute,
613
616
  ...(config.agents ? { caps: config.agents } : {}),
@@ -261,6 +261,8 @@ export interface PlanUsage {
261
261
  windows: PlanWindow[];
262
262
  /** How many one-shot rate-limit resets are still banked, when the plan has them. */
263
263
  resetsAvailable?: number;
264
+ /** ISO time the soonest-expiring banked reset must be used by, when the plan says. */
265
+ resetsExpireAt?: string;
264
266
  /** Epoch ms of the read, so renderers can flag figures as aged. */
265
267
  fetchedAt: number;
266
268
  }
@@ -141,7 +141,8 @@ export class AgentRegistry {
141
141
  ok: false,
142
142
  error:
143
143
  `Sub-agent concurrency cap reached (${caps.maxConcurrent} live). ` +
144
- `Wait for one to finish (wait_for_agent) or kill one (kill_agent).`,
144
+ `Wait for one to finish (wait_for_agent) or kill one (kill_agent), ` +
145
+ `or raise agents.maxConcurrent in ~/.talon/config.json.`,
145
146
  };
146
147
  }
147
148
 
@@ -0,0 +1,78 @@
1
+ /**
2
+ * Per-chat trigger caps — how many watcher scripts one chat may keep active.
3
+ *
4
+ * Configured by `config.triggers` (see core/config) and wired in once via
5
+ * initTriggers. The check is a pure function of the chat's active triggers so
6
+ * the gateway handler and the tests share one rule.
7
+ */
8
+
9
+ import { MAX_ACTIVE_PER_CHAT } from "../../../storage/triggers.js";
10
+
11
+ export type TriggerCaps = {
12
+ /** Active triggers per chat — all of them, or only ad-hoc ones when
13
+ * `maxPersistentPerChat` is set. */
14
+ readonly maxActivePerChat: number;
15
+ /** Optional separate budget for persistent triggers. */
16
+ readonly maxPersistentPerChat?: number;
17
+ };
18
+
19
+ export const DEFAULT_TRIGGER_CAPS: TriggerCaps = {
20
+ maxActivePerChat: MAX_ACTIVE_PER_CHAT,
21
+ };
22
+
23
+ const capsHolder: { caps: TriggerCaps } = { caps: DEFAULT_TRIGGER_CAPS };
24
+
25
+ /** Replace the live caps. Missing fields fall back to the defaults. */
26
+ export function setTriggerCaps(caps?: Partial<TriggerCaps>): void {
27
+ const next: TriggerCaps = {
28
+ maxActivePerChat:
29
+ caps?.maxActivePerChat ?? DEFAULT_TRIGGER_CAPS.maxActivePerChat,
30
+ ...(caps?.maxPersistentPerChat !== undefined
31
+ ? { maxPersistentPerChat: caps.maxPersistentPerChat }
32
+ : {}),
33
+ };
34
+ capsHolder.caps = next;
35
+ }
36
+
37
+ export function getTriggerCaps(): TriggerCaps {
38
+ return capsHolder.caps;
39
+ }
40
+
41
+ /**
42
+ * Why a new trigger may not be created, or null when there is room.
43
+ *
44
+ * Without `maxPersistentPerChat`, every active trigger shares
45
+ * `maxActivePerChat` (the historical behaviour). With it, persistent and
46
+ * ad-hoc triggers draw from separate budgets.
47
+ */
48
+ export function triggerCapError(
49
+ active: ReadonlyArray<{ persistent?: boolean }>,
50
+ persistent: boolean,
51
+ caps: TriggerCaps = capsHolder.caps,
52
+ ): string | null {
53
+ const configPath = "~/.talon/config.json";
54
+ if (caps.maxPersistentPerChat === undefined) {
55
+ if (active.length < caps.maxActivePerChat) return null;
56
+ return (
57
+ `Per-chat trigger cap reached (${caps.maxActivePerChat} active). ` +
58
+ `Cancel one before creating another, or raise ` +
59
+ `triggers.maxActivePerChat in ${configPath}.`
60
+ );
61
+ }
62
+ if (persistent) {
63
+ const count = active.filter((t) => t.persistent === true).length;
64
+ if (count < caps.maxPersistentPerChat) return null;
65
+ return (
66
+ `Per-chat persistent trigger cap reached ` +
67
+ `(${caps.maxPersistentPerChat} persistent active). Cancel one before ` +
68
+ `creating another, or raise triggers.maxPersistentPerChat in ${configPath}.`
69
+ );
70
+ }
71
+ const count = active.filter((t) => t.persistent !== true).length;
72
+ if (count < caps.maxActivePerChat) return null;
73
+ return (
74
+ `Per-chat ad-hoc trigger cap reached (${caps.maxActivePerChat} ` +
75
+ `non-persistent active). Cancel one before creating another, or raise ` +
76
+ `triggers.maxActivePerChat in ${configPath}.`
77
+ );
78
+ }
@@ -5,6 +5,7 @@
5
5
  * Split by responsibility:
6
6
  * - `state` — injected deps + the child/timeout/log/buffer registries,
7
7
  * the warden set, timing constants, init + getRunningCount
8
+ * - `caps` — per-chat active-trigger caps (config.triggers)
8
9
  * - `command` — interpreter resolution per script language
9
10
  * - `output` — stdout/stderr capture, payload truncation, wake firing
10
11
  * - `exit` — timeout/cancel/shutdown, child kill, finalizeExit, failTrigger
@@ -20,6 +21,11 @@ import { handleStdoutLine } from "./output.js";
20
21
  import { handleTimeout, finalizeExit } from "./exit.js";
21
22
 
22
23
  export { initTriggers, getRunningCount } from "./state.js";
24
+ export {
25
+ getTriggerCaps,
26
+ triggerCapError,
27
+ DEFAULT_TRIGGER_CAPS,
28
+ } from "./caps.js";
23
29
  export { commandForLanguage } from "./command.js";
24
30
  export { spawnTrigger } from "./spawn.js";
25
31
  export { cancelTrigger, shutdownTriggers } from "./exit.js";
@@ -9,12 +9,15 @@ import type { ChildProcess } from "node:child_process";
9
9
  import type { WriteStream } from "node:fs";
10
10
  import { execute as dispatcherExecute } from "../../engine/dispatcher.js";
11
11
  import { log } from "../../../util/log.js";
12
+ import { getTriggerCaps, setTriggerCaps, type TriggerCaps } from "./caps.js";
12
13
 
13
14
  // ── Dependencies (injected at startup) ──────────────────────────────────────
14
15
 
15
16
  export type TriggerDeps = {
16
17
  /** Used for terminal "fired"/"errored" wake prompts that go through the model. */
17
18
  execute: typeof dispatcherExecute;
19
+ /** Per-chat caps from `config.triggers`; defaults apply when absent. */
20
+ caps?: Partial<TriggerCaps>;
18
21
  };
19
22
 
20
23
  /** Reassignable on a holder object so submodules see the injected deps. */
@@ -52,7 +55,15 @@ export const WARDEN_GRACE_SLACK_MS = 2_000;
52
55
  export function initTriggers(d: TriggerDeps): void {
53
56
  depsHolder.deps = d;
54
57
  lifecycle.shuttingDown = false;
55
- log("triggers", "Initialized");
58
+ setTriggerCaps(d.caps);
59
+ const caps = getTriggerCaps();
60
+ log(
61
+ "triggers",
62
+ `Initialized — maxActivePerChat=${caps.maxActivePerChat}` +
63
+ (caps.maxPersistentPerChat !== undefined
64
+ ? ` maxPersistentPerChat=${caps.maxPersistentPerChat}`
65
+ : ""),
66
+ );
56
67
  }
57
68
 
58
69
  /** Number of triggers currently running. */
@@ -511,6 +511,25 @@ const configSchema = z.object({
511
511
  .default(15 * 60 * 1000),
512
512
  })
513
513
  .optional(),
514
+ /**
515
+ * Triggers — per-chat caps on active watcher scripts (running or
516
+ * pending). Checked when `trigger_create` runs; triggers already running
517
+ * are never killed when a cap is lowered.
518
+ *
519
+ * - `maxActivePerChat` — active triggers per chat (default 5). When
520
+ * `maxPersistentPerChat` is unset this counts persistent and ad-hoc
521
+ * triggers together, exactly as before.
522
+ * - `maxPersistentPerChat` — optional separate budget for persistent
523
+ * triggers. When set, persistent triggers count only against it and
524
+ * `maxActivePerChat` bounds ad-hoc (non-persistent) triggers only, so
525
+ * long-lived watchers can't starve short ad-hoc ones.
526
+ */
527
+ triggers: z
528
+ .object({
529
+ maxActivePerChat: z.number().int().min(1).max(50).default(5),
530
+ maxPersistentPerChat: z.number().int().min(1).max(50).optional(),
531
+ })
532
+ .optional(),
514
533
  /**
515
534
  * Backups & checkpoints (docs/backups.md). Talon's only safety net, so
516
535
  * it is on by default: every `intervalHours` it writes a snapshot of
@@ -785,15 +804,84 @@ const DEFAULT_CONFIG = {
785
804
  pulseIntervalMs: 300000,
786
805
  };
787
806
 
807
+ /**
808
+ * A config.json that exists but cannot be used — unreadable, not JSON, or
809
+ * rejected by the schema. Thrown instead of falling back to defaults: a
810
+ * daemon that silently boots on defaults (e.g. the telegram frontend) is
811
+ * far more surprising than one that refuses to start. The file on disk is
812
+ * never touched. `issues` carries one line per problem for callers that
813
+ * want to render them individually.
814
+ */
815
+ export class ConfigFileError extends Error {
816
+ constructor(
817
+ message: string,
818
+ readonly path: string,
819
+ readonly issues: readonly string[] = [],
820
+ ) {
821
+ super(message);
822
+ this.name = "ConfigFileError";
823
+ }
824
+ }
825
+
826
+ /**
827
+ * Append a line/column hint to a JSON.parse error message. Recent V8
828
+ * already includes "(line L column C)"; older runtimes only report
829
+ * "at position N", so derive it from the raw text in that case.
830
+ */
831
+ function describeJsonError(err: unknown, raw: string): string {
832
+ const message = err instanceof Error ? err.message : String(err);
833
+ if (/\(line \d+ column \d+\)/.test(message)) return message;
834
+ const match = /at position (\d+)/.exec(message);
835
+ if (!match) return message;
836
+ const before = raw.slice(0, Number(match[1]));
837
+ const line = before.split("\n").length;
838
+ const column = before.length - before.lastIndexOf("\n");
839
+ return `${message} (line ${line} column ${column})`;
840
+ }
841
+
842
+ /** Render zod issues as `path: message` lines (`(root)` for top-level). */
843
+ function formatSchemaIssues(error: z.ZodError): string[] {
844
+ return error.issues.map((issue) => {
845
+ const path = issue.path.map(String).join(".") || "(root)";
846
+ return `${path}: ${issue.message}`;
847
+ });
848
+ }
849
+
850
+ /**
851
+ * Read config.json. A missing file is `{}` (first run — defaults apply);
852
+ * a present file that cannot be read or parsed throws ConfigFileError.
853
+ */
788
854
  function loadConfigFile(): Record<string, unknown> {
855
+ if (!existsSync(CONFIG_FILE)) return {};
856
+ let raw: string;
789
857
  try {
790
- if (existsSync(CONFIG_FILE)) {
791
- return JSON.parse(readFileSync(CONFIG_FILE, "utf-8"));
792
- }
793
- } catch {
794
- /* corrupt — will be recreated */
858
+ raw = readFileSync(CONFIG_FILE, "utf-8");
859
+ } catch (err) {
860
+ throw new ConfigFileError(
861
+ `Cannot read ${CONFIG_FILE}: ${err instanceof Error ? err.message : err}`,
862
+ CONFIG_FILE,
863
+ );
864
+ }
865
+ let data: unknown;
866
+ try {
867
+ data = JSON.parse(raw);
868
+ } catch (err) {
869
+ const detail = describeJsonError(err, raw);
870
+ throw new ConfigFileError(
871
+ `Invalid JSON in ${CONFIG_FILE}: ${detail}. ` +
872
+ `The file was left untouched — fix it and start Talon again.`,
873
+ CONFIG_FILE,
874
+ [detail],
875
+ );
876
+ }
877
+ if (data === null || typeof data !== "object" || Array.isArray(data)) {
878
+ throw new ConfigFileError(
879
+ `Invalid config in ${CONFIG_FILE}: the top level must be a JSON object.`,
880
+ CONFIG_FILE,
881
+ ["(root): expected a JSON object"],
882
+ );
795
883
  }
796
- return {};
884
+ return data as Record<string, unknown>;
797
885
  }
798
886
 
799
887
  function normalizeDeprecatedFrontendConfig(
@@ -879,7 +967,18 @@ export function loadConfig(): TalonConfig {
879
967
  }
880
968
  }
881
969
 
882
- const parsed = configSchema.parse(fileConfig);
970
+ const result = configSchema.safeParse(fileConfig);
971
+ if (!result.success) {
972
+ const issues = formatSchemaIssues(result.error);
973
+ throw new ConfigFileError(
974
+ `Invalid config in ${CONFIG_FILE}:\n` +
975
+ issues.map((line) => ` - ${line}`).join("\n") +
976
+ `\nThe file was left untouched — fix it and start Talon again.`,
977
+ CONFIG_FILE,
978
+ issues,
979
+ );
980
+ }
981
+ const parsed = result.data;
883
982
 
884
983
  // The soul kernel is gone (#953). Its config block still parses so an
885
984
  // existing config.json keeps loading, but it no longer does anything —
@@ -80,3 +80,26 @@ export function handleUncaughtException(err: Error, hooks: CrashHooks): void {
80
80
  crashStep("crash report", () => logError("bot", "Uncaught exception", err));
81
81
  process.exit(1);
82
82
  }
83
+
84
+ /**
85
+ * `process.on("unhandledRejection")` body. Report — never crash — but keep
86
+ * the stack: a bare "Unhandled rejection: ENOSPC: no space left on device,
87
+ * write" says nothing about which code path forgot its `.catch()`. Async fs
88
+ * errors carry `path`/`syscall` rather than useful frames, so those ride
89
+ * along in the message too.
90
+ */
91
+ export function handleUnhandledRejection(reason: unknown): void {
92
+ crashStep("rejection report", () => {
93
+ if (!(reason instanceof Error)) {
94
+ logError("bot", `Unhandled rejection: ${String(reason)}`);
95
+ return;
96
+ }
97
+ const { syscall, path } = reason as NodeJS.ErrnoException;
98
+ const where = [syscall, path].filter(Boolean).join(" ");
99
+ logError(
100
+ "bot",
101
+ `Unhandled rejection: ${reason.message}${where ? ` (${where})` : ""}`,
102
+ reason,
103
+ );
104
+ });
105
+ }
@@ -19,12 +19,12 @@ import {
19
19
  validateTimeout,
20
20
  writeScriptFile,
21
21
  DEFAULT_TIMEOUT_SECONDS,
22
- MAX_ACTIVE_PER_CHAT,
23
22
  type TriggerLanguage,
24
23
  } from "../../../storage/triggers.js";
25
24
  import {
26
25
  cancelTrigger,
27
26
  spawnTrigger,
27
+ triggerCapError,
28
28
  } from "../../background/triggers/index.js";
29
29
  import { log } from "../../../util/log.js";
30
30
  import { validateJobModelOverride } from "./validation.js";
@@ -61,13 +61,11 @@ export const triggerHandlers: SharedActionHandlers = {
61
61
  error: `A trigger named "${name}" already exists in this chat. Cancel it first or pick a different name.`,
62
62
  };
63
63
  }
64
- const active = getActiveTriggersForChat(chatKey);
65
- if (active.length >= MAX_ACTIVE_PER_CHAT) {
66
- return {
67
- ok: false,
68
- error: `Per-chat trigger cap reached (${MAX_ACTIVE_PER_CHAT} active). Cancel one before creating another.`,
69
- };
70
- }
64
+ const capErr = triggerCapError(
65
+ getActiveTriggersForChat(chatKey),
66
+ persistent,
67
+ );
68
+ if (capErr) return { ok: false, error: capErr };
71
69
 
72
70
  // Validate the model up front so a bad id is rejected here instead of
73
71
  // silently failing at fire time.
@@ -40,7 +40,9 @@
40
40
  * 2. `TALON_BRIDGE_URL/health` stops responding for several
41
41
  * consecutive pings — Talon's gateway is gone. Catches the
42
42
  * "kilo serve / opencode serve outlives Talon" case where those
43
- * daemons keep our stdin open across Talon restarts.
43
+ * daemons keep our stdin open across Talon restarts. A ping that only
44
+ * times out (port still bound, gateway busy) is tolerated for minutes,
45
+ * not seconds — see BridgeWatchdog.
44
46
  */
45
47
 
46
48
  import crossSpawn from "cross-spawn";
@@ -203,7 +205,75 @@ export function wrapMcpCommand(command: readonly string[]): string[] {
203
205
  // within ~1 minute.
204
206
  const BRIDGE_PING_INTERVAL_MS = 15_000;
205
207
  const BRIDGE_PING_TIMEOUT_MS = 2_000;
206
- const BRIDGE_FAILURES_BEFORE_EXIT = 4;
208
+ export const BRIDGE_FAILURES_BEFORE_EXIT = 4;
209
+ // A ping that TIMES OUT means the port is still bound — the kernel accepted
210
+ // the connection — but the gateway is too busy to answer within 2s (event
211
+ // loop saturated by a burst of agents, a big synchronous write, …). That is
212
+ // a live Talon, not a dead one, so it gets a much longer budget (~5 min)
213
+ // before the child is evicted. Only a truly wedged daemon reaches it.
214
+ export const BRIDGE_UNRESPONSIVE_BEFORE_EXIT = 20;
215
+
216
+ /**
217
+ * Outcome of one bridge health ping:
218
+ * - "ok": /health answered 2xx.
219
+ * - "unreachable": nothing healthy behind the port (connection refused or
220
+ * reset, non-2xx reply) — Talon is gone or restarting.
221
+ * - "unresponsive": the request timed out — something holds the port but
222
+ * is slow to answer; Talon is alive but busy.
223
+ */
224
+ export type BridgePingOutcome = "ok" | "unreachable" | "unresponsive";
225
+
226
+ /** Classify a rejected health fetch. Timeouts/aborts mean "busy", not "gone". */
227
+ export function classifyBridgePingError(
228
+ err: unknown,
229
+ ): Exclude<BridgePingOutcome, "ok"> {
230
+ const name = (err as { name?: unknown } | null)?.name;
231
+ return name === "TimeoutError" || name === "AbortError"
232
+ ? "unresponsive"
233
+ : "unreachable";
234
+ }
235
+
236
+ /** Ping `${bridgeUrl}/health` once. Never throws. */
237
+ export async function pingBridge(
238
+ bridgeUrl: string,
239
+ timeoutMs: number = BRIDGE_PING_TIMEOUT_MS,
240
+ ): Promise<BridgePingOutcome> {
241
+ try {
242
+ const resp = await fetch(`${bridgeUrl}/health`, {
243
+ signal: AbortSignal.timeout(timeoutMs),
244
+ });
245
+ // Drain the body so the socket is released promptly.
246
+ await resp.arrayBuffer().catch(() => undefined);
247
+ return resp.ok ? "ok" : "unreachable";
248
+ } catch (err) {
249
+ return classifyBridgePingError(err);
250
+ }
251
+ }
252
+
253
+ /**
254
+ * Consecutive-failure bookkeeping for the bridge watchdog. `record()`
255
+ * returns true once the child should be shut down: after
256
+ * BRIDGE_FAILURES_BEFORE_EXIT failures ending in an "unreachable" ping (the
257
+ * port is really closed), or after BRIDGE_UNRESPONSIVE_BEFORE_EXIT failures
258
+ * of any kind (the daemon is wedged, not just busy).
259
+ */
260
+ export class BridgeWatchdog {
261
+ consecutiveFailures = 0;
262
+
263
+ record(outcome: BridgePingOutcome): boolean {
264
+ if (outcome === "ok") {
265
+ this.consecutiveFailures = 0;
266
+ return false;
267
+ }
268
+ this.consecutiveFailures += 1;
269
+ if (this.consecutiveFailures >= BRIDGE_UNRESPONSIVE_BEFORE_EXIT)
270
+ return true;
271
+ return (
272
+ outcome === "unreachable" &&
273
+ this.consecutiveFailures >= BRIDGE_FAILURES_BEFORE_EXIT
274
+ );
275
+ }
276
+ }
207
277
 
208
278
  /**
209
279
  * Run the supervisor over `argvTail` = [cmd, ...args].
@@ -346,29 +416,19 @@ export function runSupervisor(argvTail: string[]): Promise<never> {
346
416
  // (every Talon-spawned MCP server has it; ad-hoc supervisor uses
347
417
  // without the env var keep the stdin-EOF-only behavior).
348
418
  if (BRIDGE_URL) {
349
- let consecutiveFailures = 0;
419
+ const watchdog = new BridgeWatchdog();
350
420
  const tick = async (): Promise<void> => {
351
421
  if (terminating) return;
352
- try {
353
- const resp = await fetch(`${BRIDGE_URL}/health`, {
354
- signal: AbortSignal.timeout(BRIDGE_PING_TIMEOUT_MS),
355
- });
356
- if (resp.ok) {
357
- consecutiveFailures = 0;
358
- return;
359
- }
360
- consecutiveFailures += 1;
361
- } catch {
362
- consecutiveFailures += 1;
363
- }
364
- if (consecutiveFailures >= BRIDGE_FAILURES_BEFORE_EXIT) {
365
- // Talon's gateway is gone. The MCP child has nothing useful to
366
- // serve — bridge calls would 404 against a dead port — so shut
367
- // down. Kilo/OpenCode notice the stdio close on the next
368
- // interaction and drop the registration on their side.
422
+ const outcome = await pingBridge(BRIDGE_URL);
423
+ if (terminating) return;
424
+ if (watchdog.record(outcome)) {
425
+ // Talon's gateway is gone (or wedged for minutes). The MCP child
426
+ // has nothing useful to serve — bridge calls would 404 against a
427
+ // dead port — so shut down. Kilo/OpenCode notice the stdio close
428
+ // on the next interaction and drop the registration on their side.
369
429
  process.stderr.write(
370
- `mcp-launcher: bridge ${BRIDGE_URL} unreachable for ${
371
- consecutiveFailures * (BRIDGE_PING_INTERVAL_MS / 1000)
430
+ `mcp-launcher: bridge ${BRIDGE_URL} ${outcome} for ${
431
+ watchdog.consecutiveFailures * (BRIDGE_PING_INTERVAL_MS / 1000)
372
432
  }s; shutting down child\n`,
373
433
  );
374
434
  terminate(0);
@@ -57,7 +57,8 @@ shutdown/crash are respawned (not ones that exited on their own), the
57
57
  script must be safe to re-run from scratch, and timeout_seconds is ignored
58
58
  (persistent triggers run until cancelled or until Talon shuts down).
59
59
 
60
- Per-chat cap of 5 active triggers.`;
60
+ Per-chat cap on active triggers (default 5; config triggers.maxActivePerChat,
61
+ with an optional separate triggers.maxPersistentPerChat budget).`;
61
62
 
62
63
  export const triggerTools: ToolDefinition[] = [
63
64
  {
@@ -462,7 +462,10 @@ export function renderUsageMessage(
462
462
  if (entry.plan.resetsAvailable) {
463
463
  const n = entry.plan.resetsAvailable;
464
464
  const resets = `usage limit reset${n === 1 ? "" : "s"} available`;
465
- lines.push(` • You have ${fmt.bold(String(n))} ${resets}`);
465
+ const by = entry.plan.resetsExpireLabel
466
+ ? ` ${fmt.escape(`(use by ${entry.plan.resetsExpireLabel})`)}`
467
+ : "";
468
+ lines.push(` • You have ${fmt.bold(String(n))} ${resets}${by}`);
466
469
  }
467
470
  for (const w of entry.plan.windows) {
468
471
  const reset = w.resetLabel ? ` reset ${w.resetLabel}` : "";
@@ -404,6 +404,8 @@ export interface PlanDisplay {
404
404
  windows: PlanWindowDisplay[];
405
405
  /** Set only when the account still has one-shot rate-limit resets left. */
406
406
  resetsAvailable: number | undefined;
407
+ /** Deadline for the soonest-expiring banked reset, when the plan reports one. */
408
+ resetsExpireLabel?: string;
407
409
  /** Set only when the figures have aged, e.g. "12m ago". */
408
410
  ageLabel: string | undefined;
409
411
  }
@@ -432,6 +434,9 @@ export function buildPlanDisplay(
432
434
  return {
433
435
  plan: usage.plan,
434
436
  resetsAvailable: usage.resetsAvailable,
437
+ resetsExpireLabel: usage.resetsAvailable
438
+ ? planResetLabel(usage.resetsExpireAt)
439
+ : undefined,
435
440
  ageLabel:
436
441
  age > PLAN_STALE_AFTER_MS
437
442
  ? formatRelativeAge(usage.fetchedAt)
@@ -38,7 +38,7 @@ import type { Trigger, TriggerLanguage } from "./repositories/triggers-repo.js";
38
38
 
39
39
  export const DEFAULT_TIMEOUT_SECONDS = 24 * 60 * 60; // 24h
40
40
  export const MAX_TIMEOUT_SECONDS = 7 * 24 * 60 * 60; // 7d
41
- /** Per-chat soft cap on simultaneously active triggers. */
41
+ /** Default per-chat cap on simultaneously active triggers (`config.triggers.maxActivePerChat`). */
42
42
  export const MAX_ACTIVE_PER_CHAT = 5;
43
43
  /** Truncate fire payloads at this many bytes to keep wake prompts sane. */
44
44
  export const FIRE_PAYLOAD_MAX_BYTES = 4_096;