talon-agent 5.26.0 → 5.26.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -54,8 +54,8 @@ export function resolveMediaInput(
54
54
 
55
55
  /** Telegram's hard limit on media captions, counted after entity parsing. */
56
56
  export const TELEGRAM_MAX_CAPTION = 1024;
57
- /** Visible length a truncated caption is cut to, leaving room for the "…". */
58
- const TRUNCATED_CAPTION_LEN = 1000;
57
+ /** Below this head budget the splitter is fighting pathological markup; give up. */
58
+ const MIN_CAPTION_HEAD = 64;
59
59
 
60
60
  /**
61
61
  * Drop every `<...>` tag in one linear pass. Only used to *measure* the
@@ -104,35 +104,76 @@ export function visibleCaptionText(html: string): string {
104
104
  export type FittedCaption = {
105
105
  caption?: string;
106
106
  parse_mode?: "HTML";
107
- /** Full caption text to deliver as a follow-up message when it was cut. */
107
+ /** Caption text that did not fit, to deliver as follow-up message(s). */
108
108
  overflow?: string;
109
109
  };
110
110
 
111
+ /** Visible (Telegram-counted) length of a markdown caption once rendered. */
112
+ export function captionUnits(markdown: string): number {
113
+ // Telegram counts UTF-16 code units, which is what .length measures.
114
+ return visibleCaptionText(markdownToTelegramHtml(markdown)).length;
115
+ }
116
+
117
+ /**
118
+ * Split a markdown caption into the leading part that fits `max` visible
119
+ * units and the remainder. Splits the *markdown source* with the shared
120
+ * message splitter (paragraph → newline → space boundaries, surrogate-safe,
121
+ * ``` fences closed/reopened), so the head is rendered to HTML on its own
122
+ * and can never strand a tag or entity. Returns null if no clean head fits.
123
+ */
124
+ export function splitCaption(
125
+ text: string,
126
+ max: number = TELEGRAM_MAX_CAPTION,
127
+ ): { head: string; rest: string } | null {
128
+ // Rendering usually shrinks markdown (markers and link URLs vanish), so
129
+ // the first try nearly always fits; shrink the budget if it does not.
130
+ for (let budget = max; budget >= MIN_CAPTION_HEAD;) {
131
+ const chunks = splitMessage(text, budget);
132
+ const head = chunks[0] ?? "";
133
+ if (head.trim() && captionUnits(head) <= max) {
134
+ const at = text.indexOf(head);
135
+ // The splitter only trims at boundaries, so the head is normally a
136
+ // verbatim prefix; when it closed a ``` fence it is not, and the
137
+ // splitter's own reopened chunks carry the remainder instead.
138
+ const rest =
139
+ at >= 0
140
+ ? text.slice(at + head.length).replace(/^\s+/, "")
141
+ : chunks.slice(1).join("\n\n");
142
+ return { head, rest };
143
+ }
144
+ budget = Math.floor(budget * 0.8);
145
+ }
146
+ return null;
147
+ }
148
+
111
149
  /**
112
150
  * Convert a markdown caption to what Telegram accepts. Captions over 1024
113
151
  * visible characters are rejected outright ("message caption is too long"),
114
- * losing the media along with them — so an oversized caption is cut to a
115
- * plain-text preview (no parse_mode: a cut through HTML could strand a tag
116
- * or entity) and the full text is returned as `overflow` for the caller to
117
- * send as a normal, chunked text message.
152
+ * losing the media along with them — so an oversized caption is split: the
153
+ * leading part that fits rides on the media (still formatted), and the rest
154
+ * is returned as `overflow` for the caller to send as follow-up text.
118
155
  */
119
156
  export function fitCaption(raw: unknown): FittedCaption {
120
157
  if (!raw) return {};
121
158
  const text = String(raw);
122
159
  const html = markdownToTelegramHtml(text);
123
- const visible = visibleCaptionText(html);
124
- // Telegram counts UTF-16 code units, which is what .length measures.
125
- if (visible.length <= TELEGRAM_MAX_CAPTION)
160
+ if (visibleCaptionText(html).length <= TELEGRAM_MAX_CAPTION)
126
161
  return { caption: html, parse_mode: "HTML" };
127
- let cut = visible.slice(0, TRUNCATED_CAPTION_LEN);
128
- // Don't strand half a surrogate pair at the cut.
129
- const last = cut.charCodeAt(cut.length - 1);
130
- if (last >= 0xd800 && last <= 0xdbff) cut = cut.slice(0, -1);
131
- return { caption: `${cut.trimEnd()}…`, overflow: text };
162
+ const split = splitCaption(text);
163
+ if (split) {
164
+ return {
165
+ caption: markdownToTelegramHtml(split.head),
166
+ parse_mode: "HTML",
167
+ ...(split.rest ? { overflow: split.rest } : {}),
168
+ };
169
+ }
170
+ // No clean boundary fits (pathological markup): send the media bare and
171
+ // deliver the whole caption as text rather than cut through it.
172
+ return { overflow: text };
132
173
  }
133
174
 
134
175
  /**
135
- * Deliver the full text of a caption that did not fit, threaded as a reply
176
+ * Deliver the part of a caption that did not fit, threaded as a reply
136
177
  * to the media it belongs to. Best-effort: the media already landed, so a
137
178
  * failure here is reported as a warning rather than failing the send.
138
179
  */
@@ -167,7 +208,7 @@ async function sendCaptionOverflow(
167
208
  `Caption overflow follow-up failed (chat=${chatId}): ${msg}`,
168
209
  );
169
210
  return {
170
- warning: `Media sent with a truncated caption, but sending the full caption text failed: ${msg}`,
211
+ warning: `Media sent, but the rest of its caption (past Telegram's ${TELEGRAM_MAX_CAPTION}-char limit) failed to send: ${msg}`,
171
212
  };
172
213
  }
173
214
  }
@@ -176,8 +217,8 @@ function overflowResult(
176
217
  r: { message_ids: number[] } | { warning: string },
177
218
  ): Record<string, unknown> {
178
219
  return "warning" in r
179
- ? { caption_truncated: true, warning: r.warning }
180
- : { caption_truncated: true, caption_message_ids: r.message_ids };
220
+ ? { caption_split: true, warning: r.warning }
221
+ : { caption_split: true, caption_message_ids: r.message_ids };
181
222
  }
182
223
 
183
224
  const sendMediaFile: TelegramActionHandlers[string] = async (
@@ -21,6 +21,7 @@ import {
21
21
  richMessagesAvailable,
22
22
  } from "./rich-messages.js";
23
23
  import { toPositiveId } from "./coerce.js";
24
+ import { captionUnits, TELEGRAM_MAX_CAPTION } from "./media.js";
24
25
  import { resolveThreadId } from "../topics.js";
25
26
  import { TELEGRAM_MAX_TEXT, type TelegramActionHandlers } from "./types.js";
26
27
 
@@ -260,6 +261,14 @@ export const messagingHandlers: TelegramActionHandlers = {
260
261
  // Media messages have captions, not text — editMessageText on them fails
261
262
  // with "there is no text in the message to edit".
262
263
  if (body.is_caption === true) {
264
+ // An edit cannot spill into a follow-up message the way a send does,
265
+ // so refuse up front with guidance rather than a raw Bot API 400.
266
+ const units = captionUnits(text);
267
+ if (units > TELEGRAM_MAX_CAPTION)
268
+ return {
269
+ ok: false,
270
+ error: `Caption too long (${units} chars, max ${TELEGRAM_MAX_CAPTION}) — shorten it, or send the rest as a separate message`,
271
+ };
263
272
  await withRetry(async () => {
264
273
  try {
265
274
  await bot.api.editMessageCaption(chatId, Number(body.message_id), {
@@ -11,8 +11,10 @@ import type { TalonConfig } from "../../../core/config/index.js";
11
11
  import { respawnSelf } from "../../../core/daemon/respawn.js";
12
12
  import { isStaleCommand } from "../polling/stale-command.js";
13
13
  import {
14
+ describeCheckpoint,
14
15
  getRepoRoot,
15
16
  runSelfUpdate,
17
+ wantsForce,
16
18
  } from "../../../core/update/self-update.js";
17
19
  import { forceDream } from "../../../core/background/dream/index.js";
18
20
  import { escapeHtml } from "../formatting.js";
@@ -169,7 +171,8 @@ function registerRestartCommand(bot: Bot): void {
169
171
  });
170
172
  }
171
173
 
172
- // /update — pull latest, reinstall, run setup, restart. Only wired
174
+ // /update [force] — pull latest, reinstall, run setup, restart. Refused
175
+ // when the pre-update checkpoint fails unless "force" is given. Only wired
173
176
  // up for developer builds running from a git checkout; packaged
174
177
  // binaries have no source tree (getRepoRoot() === null) so the
175
178
  // command stays absent entirely.
@@ -187,8 +190,11 @@ function registerUpdateCommand(
187
190
  if (isStaleCommand(ctx.message?.date, "/update")) return;
188
191
  const remote = config.update?.remote ?? "origin";
189
192
  const branch = config.update?.branch ?? "main";
193
+ const force = wantsForce(typeof ctx.match === "string" ? ctx.match : "");
190
194
  const sent = await ctx.reply(
191
- `⏳ Updating from <code>${escapeHtml(remote)}/${escapeHtml(branch)}</code>…`,
195
+ `⏳ Updating from <code>${escapeHtml(remote)}/${escapeHtml(branch)}</code>` +
196
+ (force ? " (forced: a failed checkpoint will not stop it)" : "") +
197
+ "…",
192
198
  { parse_mode: "HTML" },
193
199
  );
194
200
  const edit = (text: string) =>
@@ -204,12 +210,24 @@ function registerUpdateCommand(
204
210
  branch,
205
211
  setup: config.update?.setup,
206
212
  repoRoot: updateRepoRoot,
213
+ force,
207
214
  })
208
215
  .then(async (res) => {
216
+ if (res.checkpointRefused) {
217
+ await edit(
218
+ `🛑 Update refused: ${escapeHtml(res.error ?? "the pre-update checkpoint failed")}\n\n` +
219
+ `Send <code>/update force</code> to update without a checkpoint.`,
220
+ );
221
+ return;
222
+ }
223
+ const note = res.checkpoint
224
+ ? `\n${escapeHtml(describeCheckpoint(res.checkpoint))}`
225
+ : "";
209
226
  if (!res.ok) {
210
227
  const tail = res.steps[res.steps.length - 1]?.output ?? "";
211
228
  await edit(
212
229
  `⚠️ Update failed: ${escapeHtml(res.error ?? "unknown error")}` +
230
+ note +
213
231
  (tail ? `\n\n<pre>${escapeHtml(tail.slice(-1500))}</pre>` : ""),
214
232
  );
215
233
  return;
@@ -221,7 +239,7 @@ function registerUpdateCommand(
221
239
  return;
222
240
  }
223
241
  await edit(
224
- `✅ Updated <code>${escapeHtml(res.before ?? "?")}</code> → <code>${escapeHtml(res.after ?? "?")}</code>. ♻️ Restarting…`,
242
+ `✅ Updated <code>${escapeHtml(res.before ?? "?")}</code> → <code>${escapeHtml(res.after ?? "?")}</code>.${note}\n♻️ Restarting…`,
225
243
  );
226
244
  // The successor documents any provisioning changes (plugin
227
245
  // runtime upgrades, migrations) back to this chat once it's up.
@@ -22,7 +22,7 @@ import * as repo from "./repo.js";
22
22
  * handle stays inside storage/ (`db-handle-stays-in-storage`), and a
23
23
  * snapshot of the database is a storage concern with a storage API.
24
24
  */
25
- export { snapshotDatabase, snapshotSqliteFile } from "../db.js";
25
+ export { databasePath, snapshotDatabase, snapshotSqliteFile } from "../db.js";
26
26
 
27
27
  export type { BackupRecord, BackupRemoteRecord } from "./repo.js";
28
28
  import type { BackupRecord, BackupRemoteRecord } from "./repo.js";
@@ -313,6 +313,19 @@ export function clearAllChatModels(chatId: string): void {
313
313
  persist(chatId);
314
314
  }
315
315
 
316
+ /**
317
+ * Drop only the legacy single-slot `model` field, keeping every
318
+ * per-backend pick. The boot reconcile uses it: a stale legacy value must
319
+ * go, but the picks the chat made on other backends are still good.
320
+ */
321
+ export function clearLegacyChatModel(chatId: string): void {
322
+ const entry = cache.get(chatId);
323
+ if (!entry || entry.model === undefined) return;
324
+ delete entry.model;
325
+ cleanupEmpty(chatId);
326
+ persist(chatId);
327
+ }
328
+
316
329
  /**
317
330
  * @deprecated Prefer `setChatModelForBackend(chatId, backendId, model)`
318
331
  * which is explicit about which backend's slot is being mutated.
package/src/storage/db.ts CHANGED
@@ -145,6 +145,11 @@ function defaultPath(): string {
145
145
  return process.env.TALON_DB_PATH || files.database;
146
146
  }
147
147
 
148
+ /** Where the process-wide database lives (or will, once opened). */
149
+ export function databasePath(): string {
150
+ return defaultPath();
151
+ }
152
+
148
153
  /**
149
154
  * Open (or return) the process-wide database. The first call wins the
150
155
  * path; tests pass an explicit tmp path and call closeDatabase() in
@@ -48,15 +48,16 @@ function isMediaEntry(value: unknown): value is MediaEntry {
48
48
 
49
49
  /**
50
50
  * Run the one-time import of the legacy JSON store, then sweep
51
- * expired entries. Idempotent; called once at boot.
51
+ * expired entries (unless `purgeExpired: false` — a boot with no safety
52
+ * checkpoint deletes nothing). Idempotent; called once at boot.
52
53
  */
53
- export function loadMediaIndex(): void {
54
+ export function loadMediaIndex(options: { purgeExpired?: boolean } = {}): void {
54
55
  try {
55
56
  importLegacyMediaIndex();
56
57
  } catch (err) {
57
58
  logError("media", "Media index load failed", err);
58
59
  }
59
- purgeExpired();
60
+ if (options.purgeExpired !== false) purgeExpired();
60
61
  }
61
62
 
62
63
  /** Legacy shape: bare MediaEntry[]. */
@@ -23,6 +23,7 @@ import { recordError } from "../util/watchdog.js";
23
23
  import { files } from "../util/paths.js";
24
24
  import { importLegacyJson } from "./legacy-import.js";
25
25
  import { dbErrorFields } from "./db.js";
26
+ import { kvGet, kvSet } from "./kv.js";
26
27
  import * as repo from "./repositories/sessions-repo.js";
27
28
 
28
29
  export type {
@@ -534,18 +535,76 @@ function removeSessionRow(chatId: string): void {
534
535
  }
535
536
  }
536
537
 
538
+ // ── Replaced-session archive ───────────────────────────────────────────────
539
+
540
+ /** A backend session id a reset replaced — kept so it can be re-linked. */
541
+ export type ArchivedSession = {
542
+ chatId: string;
543
+ sessionId: string;
544
+ /** Why the session was replaced ("reset", "backend-unavailable", …). */
545
+ reason: string;
546
+ /** When it was replaced (ms epoch). */
547
+ at: number;
548
+ turns: number;
549
+ lastModel?: string;
550
+ sessionName?: string;
551
+ };
552
+
553
+ const SESSION_ARCHIVE_KEY = "sessions.archive";
554
+ /** Newest entries kept; the transcripts themselves stay on disk regardless. */
555
+ const SESSION_ARCHIVE_MAX = 500;
556
+
557
+ /**
558
+ * Remember a session id that is about to be dropped. A reset used to
559
+ * forget it outright, leaving the backend transcript on disk with nothing
560
+ * pointing at it; with the id kept, a chat reset by mistake (or by a boot
561
+ * reconcile) can be re-linked to its old conversation. Never throws.
562
+ */
563
+ function archiveSessionId(entry: ArchivedSession): void {
564
+ try {
565
+ const prior = kvGet<ArchivedSession[]>(SESSION_ARCHIVE_KEY);
566
+ const list = Array.isArray(prior) ? prior : [];
567
+ list.push(entry);
568
+ kvSet(SESSION_ARCHIVE_KEY, list.slice(-SESSION_ARCHIVE_MAX));
569
+ } catch (err) {
570
+ logError("sessions", `Failed to archive session chat=${entry.chatId}`, err);
571
+ }
572
+ }
573
+
574
+ /** Archived (replaced) session ids, oldest first; one chat's when given. */
575
+ export function getArchivedSessions(chatId?: string): ArchivedSession[] {
576
+ const list = kvGet<ArchivedSession[]>(SESSION_ARCHIVE_KEY);
577
+ if (!Array.isArray(list)) return [];
578
+ return chatId ? list.filter((e) => e.chatId === chatId) : list;
579
+ }
580
+
537
581
  /**
538
582
  * Reset the chat's conversation state (backend session id, turns, usage)
539
583
  * while carrying its metrics forward. Resets fire on /new, model switches
540
584
  * and error recovery — none of which should erase the chat's accounting
541
585
  * history (that's what makes per-session metrics survive anything short
542
586
  * of deleting the chat). Use deleteSession() to drop the chat entirely.
587
+ *
588
+ * The replaced backend session id is archived (see getArchivedSessions),
589
+ * never just discarded. Talon's own chat history is untouched: clearing
590
+ * it is a separate, explicit call.
543
591
  */
544
- export function resetSession(chatId: string): void {
592
+ export function resetSession(chatId: string, reason = "reset"): void {
545
593
  const session = cache.get(chatId);
546
594
  const turns = session?.turns ?? 0;
547
595
  const name = session?.sessionName;
548
596
  const metrics = session?.metrics;
597
+ if (session?.sessionId) {
598
+ archiveSessionId({
599
+ chatId,
600
+ sessionId: session.sessionId,
601
+ reason,
602
+ at: Date.now(),
603
+ turns,
604
+ ...(session.lastModel ? { lastModel: session.lastModel } : {}),
605
+ ...(name ? { sessionName: name } : {}),
606
+ });
607
+ }
549
608
  removeSessionRow(chatId);
550
609
  const hasHistory =
551
610
  metrics &&