talon-agent 3.6.1 → 3.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,3 +1,7 @@
1
+ <p align="center">
2
+ <img src="docs/assets/talon-hero.png" alt="Talon — multi-platform agentic AI harness" width="880">
3
+ </p>
4
+
1
5
  # Talon
2
6
 
3
7
  [![Node.js](https://img.shields.io/badge/node-%3E%3D22-339933?logo=nodedotjs&logoColor=white)](https://nodejs.org)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.6.1",
3
+ "version": "3.6.3",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -105,7 +105,7 @@
105
105
  "@openai/agents": "^0.13.0",
106
106
  "@openai/codex-sdk": "^0.145.0",
107
107
  "@opencode-ai/sdk": "^1.17.4",
108
- "@playwright/mcp": "^0.0.78",
108
+ "@playwright/mcp": "0.0.56",
109
109
  "@types/cross-spawn": "^6.0.6",
110
110
  "big-integer": "^1.6.52",
111
111
  "cheerio": "^1.2.0",
package/src/app.ts CHANGED
@@ -22,6 +22,7 @@ import {
22
22
  runStartupCatchup,
23
23
  } from "./core/background/cron.js";
24
24
  import { shutdownTriggers } from "./core/background/triggers/index.js";
25
+ import { pruneSettledTriggers } from "./storage/trigger-store.js";
25
26
  import { startWatchdog, stopWatchdog } from "./util/watchdog.js";
26
27
  import { log, logError, logWarn } from "./util/log.js";
27
28
  import {
@@ -94,6 +95,7 @@ onBackendChange((holder, newBackend, info) => {
94
95
  // ── Graceful shutdown ────────────────────────────────────────────────────────
95
96
 
96
97
  let shuttingDown = false;
98
+ let triggerPruneTimer: ReturnType<typeof setInterval> | null = null;
97
99
 
98
100
  const SHUTDOWN_TIMEOUT_MS = 15_000;
99
101
  const DRAIN_TIMEOUT_MS = 5_000;
@@ -167,6 +169,10 @@ async function gracefulShutdown(signal: string): Promise<void> {
167
169
  await awaitHeartbeat();
168
170
  });
169
171
  await shutdownStep("cron timer", stopCronTimer);
172
+ await shutdownStep("trigger prune timer", () => {
173
+ if (triggerPruneTimer) clearInterval(triggerPruneTimer);
174
+ triggerPruneTimer = null;
175
+ });
170
176
  await shutdownStep("triggers", shutdownTriggers);
171
177
  await shutdownStep("watchdog", stopWatchdog);
172
178
  await shutdownStep("upload cleanup", stopUploadCleanup);
@@ -225,6 +231,38 @@ async function main(): Promise<void> {
225
231
  startWatchdog(config.workspace);
226
232
  startUploadCleanup(config.workspace);
227
233
 
234
+ // Cron MUST start before the frontends are awaited: a long-polling
235
+ // frontend's start() blocks for the entire process lifetime, so anything
236
+ // sequenced after that await effectively runs at shutdown. (Regression
237
+ // #396→3.5.0: startCronTimer() sat after the frontend await and no
238
+ // scheduled job fired for 23 days.) Message delivery inside cron uses the
239
+ // frontend's send API, which works as soon as init() has completed —
240
+ // it does not depend on the polling loop being up.
241
+ //
242
+ // Catch-up replays runs that came due while Talon was down (per-job
243
+ // policy; default for new jobs is "once"). Kicking it off first gives it
244
+ // the ~60s head start to take each replayed job's in-flight lock before
245
+ // the first scheduled tick, so a replay can't race a scheduled run.
246
+ // Fire-and-forget so a slow replay never blocks startup.
247
+ runStartupCatchup().catch((err) =>
248
+ logError("cron", "startup catch-up failed", err),
249
+ );
250
+ startCronTimer();
251
+
252
+ // Sweep settled triggers (fired/errored/cancelled/timed_out/terminated)
253
+ // past their retention window so the trigger list doesn't accumulate
254
+ // corpses forever. Once at boot, then daily.
255
+ const pruned = pruneSettledTriggers();
256
+ if (pruned > 0) log("triggers", `Pruned ${pruned} settled trigger(s)`);
257
+ triggerPruneTimer = setInterval(
258
+ () => {
259
+ const n = pruneSettledTriggers();
260
+ if (n > 0) log("triggers", `Pruned ${n} settled trigger(s)`);
261
+ },
262
+ 24 * 60 * 60_000,
263
+ );
264
+ triggerPruneTimer.unref();
265
+
228
266
  // A stdin-reading frontend (terminal) blocks in start() for the
229
267
  // process lifetime — run it without awaiting alongside the others.
230
268
  const stdinFrontends = frontends.filter(
@@ -248,15 +286,8 @@ async function main(): Promise<void> {
248
286
  );
249
287
  }
250
288
 
251
- // Replay any runs that came due while Talon was down (per-job catch-up
252
- // policy; default skip = fast no-op), THEN start the live cron tick. Kicking
253
- // catch-up off first gives it the ~60s head start to take each replayed job's
254
- // in-flight lock before the first scheduled tick, so a replay can't race a
255
- // scheduled run. Fire-and-forget so a slow replay never blocks startup.
256
- runStartupCatchup().catch((err) =>
257
- logError("cron", "startup catch-up failed", err),
258
- );
259
- startCronTimer();
289
+ // NOTE: nothing may be sequenced after this point — the await above only
290
+ // resolves when the frontends stop (i.e. at shutdown).
260
291
  }
261
292
 
262
293
  main().catch((err) => {
@@ -116,30 +116,56 @@ const JOB_HEALTH: JobHealthOptions = {
116
116
  // fires a handful of times, not hundreds.
117
117
  const CATCHUP_MAX = 5;
118
118
 
119
+ // Wall-clock watermark of the last fully-evaluated tick. Cron dueness is
120
+ // window-based — "did a fire time land inside (watermark, now]?" — instead of
121
+ // "does the current minute match?". setInterval ticks drift under event-loop
122
+ // load, so a tick landing at :29:59 followed by one at :31:01 skips the :30
123
+ // minute entirely under minute-equality; the window formulation cannot miss
124
+ // it. Initialized to process start: anything earlier is startup-catch-up
125
+ // territory (per-job catchup policy), not live-tick recovery.
126
+ let lastTickMs = Date.now();
127
+
128
+ // Hard cap on how far back a live tick will look. Bounds the window when the
129
+ // watermark lags (load-shed ticks don't advance it) and keeps a pathological
130
+ // stall from replaying ancient fire times outside the catch-up policy.
131
+ const MAX_TICK_LOOKBACK_MS = 10 * 60_000;
132
+
119
133
  async function runCronTick(): Promise<void> {
120
134
  if (!deps) return;
121
- if (getActiveCount() > 10) return; // safety valve — don't pile on if heavily loaded
135
+ // Safety valve — don't pile on if heavily loaded. The watermark is NOT
136
+ // advanced, so the skipped window is re-covered by the next tick (bounded
137
+ // by MAX_TICK_LOOKBACK_MS).
138
+ if (getActiveCount() > 10) return;
122
139
 
123
140
  const now = new Date();
124
141
  const nowMs = now.getTime();
142
+ const windowStartMs = Math.max(lastTickMs, nowMs - MAX_TICK_LOOKBACK_MS);
125
143
  const jobs = getAllCronJobs();
126
144
  pruneJobHealth(new Set(jobs.map((j) => j.id)));
127
145
 
146
+ let loadShed = false;
128
147
  for (const job of jobs) {
129
148
  if (!job.enabled) continue;
130
149
  // Expiry takes priority over dueness: a job past its end time is disabled
131
150
  // and skipped even if this minute would otherwise match.
132
151
  if (expireIfPast(job, nowMs)) continue;
133
152
  if (runningJobs.has(job.id)) continue; // already in-flight this tick or a previous one
134
- if (!isDue(job, now)) continue;
153
+ if (!isDue(job, now, windowStartMs)) continue;
135
154
  if (!jobAllowsRun(job.id, nowMs, JOB_HEALTH)) {
136
155
  log("cron", `Skipping "${job.name}" [${job.id}] — breaker open`);
137
156
  continue;
138
157
  }
139
- if (getActiveCount() > 10) break;
158
+ if (getActiveCount() > 10) {
159
+ loadShed = true;
160
+ break;
161
+ }
140
162
 
141
163
  await runScheduled(job);
142
164
  }
165
+
166
+ // Only advance the watermark when every job was evaluated — a load-shed
167
+ // break leaves it in place so unevaluated jobs keep their window.
168
+ if (!loadShed) lastTickMs = nowMs;
143
169
  }
144
170
 
145
171
  /**
@@ -343,7 +369,7 @@ export async function runJobNow(
343
369
  const warnedBadSchedule = new Set<string>();
344
370
  const MAX_WARNED_SCHEDULES = 200;
345
371
 
346
- function isDue(job: CronJob, now: Date): boolean {
372
+ function isDue(job: CronJob, now: Date, windowStartMs: number): boolean {
347
373
  const nowMs = now.getTime();
348
374
 
349
375
  // Not-before gate (both modes): never fire before startAt.
@@ -354,7 +380,9 @@ function isDue(job: CronJob, now: Date): boolean {
354
380
  // a burst.
355
381
  if (job.lastRunAt !== undefined && job.lastRunAt > nowMs) return false;
356
382
 
357
- return isIntervalJob(job) ? isIntervalDue(job, nowMs) : isCronDue(job, now);
383
+ return isIntervalJob(job)
384
+ ? isIntervalDue(job, nowMs)
385
+ : isCronDue(job, now, windowStartMs);
358
386
  }
359
387
 
360
388
  /**
@@ -366,11 +394,17 @@ function isIntervalDue(job: CronJob, nowMs: number): boolean {
366
394
  return nowMs - intervalAnchor(job) >= (job.everyMs as number);
367
395
  }
368
396
 
369
- /** Cron mode: due when the current minute matches a fire time. */
370
- function isCronDue(job: CronJob, now: Date): boolean {
397
+ /**
398
+ * Cron mode: due when a scheduled fire time fell inside (floor, now], where
399
+ * the floor is the tick window start raised by lastRunAt (dedupe — never
400
+ * re-fire a slot that already ran) and startAt (never count fire times from
401
+ * before the job's not-before gate). Window semantics make dueness immune to
402
+ * tick drift: a fire time in a minute no tick landed on is still caught by
403
+ * the next tick, because the window spans the gap.
404
+ */
405
+ function isCronDue(job: CronJob, now: Date, windowStartMs: number): boolean {
371
406
  if (!job.schedule) return false;
372
407
  try {
373
- const oneMinuteAgo = new Date(now.getTime() - 60_000);
374
408
  const cron = new Cron(job.schedule, {
375
409
  timezone: job.timezone ?? undefined,
376
410
  });
@@ -380,12 +414,15 @@ function isCronDue(job: CronJob, now: Date): boolean {
380
414
  // the job is actually due right now)
381
415
  warnedBadSchedule.delete(job.id);
382
416
 
383
- const next = cron.nextRun(oneMinuteAgo);
384
- if (!next) return false;
385
-
386
- const nowMinute = Math.floor(now.getTime() / 60_000);
387
- const nextMinute = Math.floor(next.getTime() / 60_000);
388
- if (nowMinute !== nextMinute) return false;
417
+ const floorMs = Math.max(
418
+ windowStartMs,
419
+ job.lastRunAt ?? 0,
420
+ job.startAt ?? 0,
421
+ );
422
+ // croner's nextRun is strictly-after its argument, so the fire time at
423
+ // exactly floorMs is excluded — (floor, now].
424
+ const next = cron.nextRun(new Date(floorMs));
425
+ if (!next || next.getTime() > now.getTime()) return false;
389
426
 
390
427
  // Prevent duplicate runs — ensure at least 55 seconds since last execution
391
428
  if (job.lastRunAt && now.getTime() - job.lastRunAt < 55_000) return false;
@@ -407,6 +444,14 @@ function isCronDue(job: CronJob, now: Date): boolean {
407
444
  }
408
445
  }
409
446
 
447
+ // Internal exports for tests — window-based dueness is regression-critical
448
+ // (a drifted tick must not skip a scheduled minute).
449
+ export const _cronInternals = {
450
+ isDue,
451
+ isCronDue,
452
+ MAX_TICK_LOOKBACK_MS,
453
+ };
454
+
410
455
  const CRON_JOB_TIMEOUT_MS = 10 * 60_000; // 10-minute max per job
411
456
 
412
457
  export async function executeJob(job: CronJob): Promise<ExecuteJobResult> {
@@ -198,8 +198,13 @@ export const cronHandlers: SharedActionHandlers = {
198
198
  maxRuns = m;
199
199
  }
200
200
 
201
- // Missed-run catch-up policy.
202
- let catchup: CatchupPolicy | undefined;
201
+ // Missed-run catch-up policy. New jobs default to "once": a run that
202
+ // came due while Talon was down (or while the scheduler was wedged)
203
+ // replays a single time at startup instead of being lost silently — a
204
+ // live audit found one-shot reminders that missed their date under the
205
+ // old "skip" default and quietly rolled over a full year. Explicit
206
+ // "skip" remains available for jobs where a late run is worthless.
207
+ let catchup: CatchupPolicy = "once";
203
208
  if (provided(body.catchup)) {
204
209
  catchup = String(body.catchup) as CatchupPolicy;
205
210
  if (!CATCHUP_POLICIES.has(catchup))
@@ -245,7 +250,7 @@ export const cronHandlers: SharedActionHandlers = {
245
250
  ...(startAt !== undefined ? { startAt } : {}),
246
251
  ...(endAt !== undefined ? { endAt } : {}),
247
252
  ...(maxRuns !== undefined ? { maxRuns } : {}),
248
- ...(catchup ? { catchup } : {}),
253
+ catchup,
249
254
  ...(model ? { model } : {}),
250
255
  ...(provider ? { provider } : {}),
251
256
  ...(instructions ? { instructions } : {}),
@@ -44,7 +44,7 @@ Lifecycle (all optional):
44
44
  • once: true — run a single time, then auto-disable (a one-shot). For "run at 3pm tomorrow", pair a cron/interval that next fires then with once.
45
45
  • max_runs: N — auto-disable after N runs.
46
46
  • start_at / end_at — ISO-8601 timestamp (or epoch ms). Don't fire before start_at; auto-disable after end_at.
47
- • catchup — what to do with runs missed while Talon was down: "skip" (default), "once" (one catch-up run), or "all" (replay each missed run, capped).
47
+ • catchup — what to do with runs missed while Talon was down: "once" (default — one catch-up run), "skip" (drop missed runs), or "all" (replay each missed run, capped).
48
48
 
49
49
  Model: leave "model"/"provider" unset to use this chat's model. Set "model" for a valid model on this chat's backend, or set both "provider" and "model" for another backend that supports isolated jobs. "instructions" can provide a short system brief for query jobs.`,
50
50
  schema: {
@@ -95,7 +95,7 @@ Model: leave "model"/"provider" unset to use this chat's model. Set "model" for
95
95
  .enum(["skip", "once", "all"])
96
96
  .optional()
97
97
  .describe(
98
- "Missed-run policy for downtime: skip (default), once, or all (capped).",
98
+ "Missed-run policy for downtime: once (default — one catch-up run), skip, or all (capped).",
99
99
  ),
100
100
  model: z
101
101
  .string()
@@ -17,10 +17,22 @@
17
17
  * "browser": "firefox",
18
18
  * "endpointFile": "/home/dylan/camoufox-endpoint.txt"
19
19
  * }
20
+ *
21
+ * VERSION COUPLING (endpoint mode): the WebSocket handshake requires the
22
+ * client (playwright-core bundled inside @playwright/mcp) and the remote
23
+ * browser server (e.g. the python-playwright process hosting Camoufox) to be
24
+ * on the SAME playwright minor version — a mismatch fails every tool call
25
+ * with "428 Precondition Required". @playwright/mcp is therefore pinned
26
+ * exactly in package.json (0.0.56 → playwright 1.58.x, matching python
27
+ * playwright 1.58 which hosts Camoufox — camoufox itself caps playwright at
28
+ * <1.61, so the node client cannot chase latest). Bump BOTH sides together,
29
+ * deliberately — do not let a routine dependency bump move one without the
30
+ * other.
20
31
  */
21
32
 
22
- import { existsSync, readFileSync } from "node:fs";
23
- import { resolve } from "node:path";
33
+ import { existsSync, readFileSync, writeFileSync } from "node:fs";
34
+ import { join, resolve } from "node:path";
35
+ import { tmpdir } from "node:os";
24
36
  import type { TalonPlugin } from "../../core/plugin/types.js";
25
37
  import { log } from "../../util/log.js";
26
38
 
@@ -55,8 +67,24 @@ export function createPlaywrightPlugin(config: {
55
67
  const args: string[] = [];
56
68
 
57
69
  if (endpoint) {
58
- // Connect to existing browser (e.g. Camoufox websocket server)
59
- args.push("--endpoint", endpoint);
70
+ // Connect to the existing browser (e.g. the Camoufox websocket server)
71
+ // via a generated MCP config file: `browser.remoteEndpoint` is the
72
+ // stable, documented way to attach to a running Playwright server and —
73
+ // unlike the newer `--endpoint` flag — exists across the @playwright/mcp
74
+ // versions this repo can pin (the pin tracks the python playwright
75
+ // version hosting Camoufox; see the version-coupling note above).
76
+ const mcpConfig = {
77
+ browser: {
78
+ ...(browser !== "chromium" ? { browserName: browser } : {}),
79
+ remoteEndpoint: endpoint,
80
+ },
81
+ };
82
+ const configPath = join(
83
+ tmpdir(),
84
+ `talon-playwright-mcp-${process.pid}.json`,
85
+ );
86
+ writeFileSync(configPath, JSON.stringify(mcpConfig));
87
+ args.push("--config", configPath);
60
88
  } else {
61
89
  args.push("--no-sandbox");
62
90
 
@@ -243,6 +243,33 @@ export function updateTrigger(
243
243
  });
244
244
  }
245
245
 
246
+ /**
247
+ * How long a settled trigger (fired/errored/cancelled/timed_out/terminated)
248
+ * is kept for post-mortem inspection before the daily sweep removes it.
249
+ * Without a sweep the trigger list accumulates corpses forever — a live
250
+ * audit found 12 of 14 listed triggers dead, some over a week old.
251
+ */
252
+ export const SETTLED_TRIGGER_TTL_MS = 72 * 60 * 60_000; // 3 days
253
+
254
+ /**
255
+ * Delete settled triggers whose terminal state is older than `ttlMs`
256
+ * (script + log files included, via deleteTrigger). Running/pending
257
+ * triggers are never touched. Returns how many were pruned.
258
+ */
259
+ export function pruneSettledTriggers(
260
+ nowMs = Date.now(),
261
+ ttlMs = SETTLED_TRIGGER_TTL_MS,
262
+ ): number {
263
+ let pruned = 0;
264
+ for (const t of getAllTriggers()) {
265
+ if (t.status === "running" || t.status === "pending") continue;
266
+ const settledAt = t.endedAt ?? t.lastFireAt ?? t.startedAt ?? t.createdAt;
267
+ if (nowMs - settledAt < ttlMs) continue;
268
+ if (deleteTrigger(t.id)) pruned++;
269
+ }
270
+ return pruned;
271
+ }
272
+
246
273
  /** Delete a trigger and best-effort clean up its on-disk script + log. */
247
274
  export function deleteTrigger(id: string): boolean {
248
275
  const t = repo.get(id);