talon-agent 5.26.6 → 5.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/app.ts +44 -13
- package/src/backend/claude-sdk/factory.ts +2 -0
- package/src/backend/claude-sdk/one-shot.ts +22 -1
- package/src/backend/codex/factory.ts +2 -0
- package/src/backend/codex/one-shot.ts +68 -19
- package/src/backend/runtime/one-shot-hooks.ts +24 -0
- package/src/core/agent-runtime/capabilities.ts +7 -0
- package/src/core/agents/index.ts +2 -0
- package/src/core/agents/prompt.ts +75 -0
- package/src/core/agents/registry.ts +317 -7
- package/src/core/agents/runner.ts +331 -10
- package/src/core/backup/{restore-guard.ts → restore/guard.ts} +5 -5
- package/src/core/backup/restore/notice.ts +164 -0
- package/src/core/backup/restore.ts +17 -5
- package/src/core/mesh/credentials/store.ts +9 -1
- package/src/core/mesh/persist.ts +9 -0
- package/src/core/types.ts +15 -0
- package/src/frontend/discord/actions/channels.ts +6 -3
- package/src/frontend/discord/actions/index.ts +24 -0
- package/src/frontend/discord/commands/backup.ts +1 -0
- package/src/frontend/discord/handlers/registry.ts +35 -0
- package/src/frontend/native/commands/backup.ts +1 -0
- package/src/frontend/native/turn/actions.ts +29 -1
- package/src/frontend/telegram/commands/backup.ts +1 -0
- package/src/storage/agents/repo.ts +200 -0
- package/src/storage/sql/agents.sql +43 -0
- package/src/storage/sql/schema.sql +54 -0
- package/src/storage/sql/statements.generated.ts +87 -1
|
@@ -24,8 +24,19 @@
|
|
|
24
24
|
* `spawnAgent` returns as soon as the run is under way: the caller (a chat
|
|
25
25
|
* turn or another agent) keeps working and hears back through the wake turn
|
|
26
26
|
* or its mailbox.
|
|
27
|
+
*
|
|
28
|
+
* - **Restarts are not deaths.** Every agent is mirrored to the `agents`
|
|
29
|
+
* table (see the registry's `persist` hook). A daemon shutdown parks the
|
|
30
|
+
* live ones (`interruptAgentsForRestart`) before it tears the backends
|
|
31
|
+
* down, so the abort that follows is not recorded as a kill and the
|
|
32
|
+
* parent is not told the agent died. On the next boot
|
|
33
|
+
* `resumeAgentsAfterRestart` brings each one back under its original id:
|
|
34
|
+
* on a backend that can resume (Claude SDK session, Codex thread) it
|
|
35
|
+
* continues its own conversation with a short "you were interrupted"
|
|
36
|
+
* note; elsewhere it is re-briefed with the tail of its previous run log.
|
|
27
37
|
*/
|
|
28
38
|
|
|
39
|
+
import { readFile } from "node:fs/promises";
|
|
29
40
|
import { dirs } from "../../util/paths.js";
|
|
30
41
|
import { log, logError, logWarn } from "../../util/log.js";
|
|
31
42
|
import {
|
|
@@ -61,9 +72,14 @@ import {
|
|
|
61
72
|
import {
|
|
62
73
|
agentLogHeader,
|
|
63
74
|
agentLogPath,
|
|
75
|
+
agentResumeLogHeader,
|
|
64
76
|
buildAgentPrompt,
|
|
65
77
|
buildAgentSystemPrompt,
|
|
78
|
+
buildRebriefPrompt,
|
|
79
|
+
buildResumePrompt,
|
|
66
80
|
} from "./prompt.js";
|
|
81
|
+
import * as agentsRepo from "../../storage/agents/repo.js";
|
|
82
|
+
import type { PersistedAgent } from "../../storage/agents/repo.js";
|
|
67
83
|
import { agentRegistry } from "./registry.js";
|
|
68
84
|
import type {
|
|
69
85
|
AgentCaps,
|
|
@@ -259,6 +275,10 @@ export async function spawnAgent(
|
|
|
259
275
|
...(spec.reasoningEffort
|
|
260
276
|
? { reasoningEffort: spec.reasoningEffort }
|
|
261
277
|
: {}),
|
|
278
|
+
...(spec.model ? { requestedModel: spec.model } : {}),
|
|
279
|
+
timeoutMs: spec.timeoutMs ?? capsHolder.caps.defaultTimeoutMs,
|
|
280
|
+
cwd: dirs.workspace,
|
|
281
|
+
...(spec.preflight ? { preflight: true } : {}),
|
|
262
282
|
},
|
|
263
283
|
capsHolder.caps,
|
|
264
284
|
);
|
|
@@ -295,6 +315,16 @@ export async function spawnAgent(
|
|
|
295
315
|
};
|
|
296
316
|
}
|
|
297
317
|
|
|
318
|
+
/**
|
|
319
|
+
* How a restarted run picks up. `sessionId` set = continue that backend
|
|
320
|
+
* conversation; unset = a fresh conversation re-briefed with `prompt`.
|
|
321
|
+
*/
|
|
322
|
+
interface ResumePlan {
|
|
323
|
+
readonly prompt: string;
|
|
324
|
+
readonly sessionId?: string;
|
|
325
|
+
readonly interruptedAt: number;
|
|
326
|
+
}
|
|
327
|
+
|
|
298
328
|
/** Build the one-shot params for a run, wired to its log and text capture. */
|
|
299
329
|
async function buildRunParams(
|
|
300
330
|
record: AgentRecord,
|
|
@@ -302,15 +332,26 @@ async function buildRunParams(
|
|
|
302
332
|
model: string,
|
|
303
333
|
abortController: AbortController,
|
|
304
334
|
capture: { last: string },
|
|
335
|
+
resume?: ResumePlan,
|
|
305
336
|
): Promise<OneShotAgentParams> {
|
|
306
337
|
const appendLog = await openRunLog(
|
|
307
338
|
agentLogPath(record.id),
|
|
308
|
-
|
|
339
|
+
resume
|
|
340
|
+
? agentResumeLogHeader(
|
|
341
|
+
record,
|
|
342
|
+
model,
|
|
343
|
+
resume.interruptedAt,
|
|
344
|
+
resume.sessionId,
|
|
345
|
+
)
|
|
346
|
+
: agentLogHeader(record, model),
|
|
309
347
|
);
|
|
348
|
+
const id = record.id;
|
|
310
349
|
return {
|
|
311
|
-
prompt:
|
|
312
|
-
|
|
313
|
-
|
|
350
|
+
prompt: resume
|
|
351
|
+
? resume.prompt
|
|
352
|
+
: buildAgentPrompt(record.brief, {
|
|
353
|
+
preflight: spec.preflight === true,
|
|
354
|
+
}),
|
|
314
355
|
systemPrompt: buildAgentSystemPrompt({
|
|
315
356
|
agentId: record.id,
|
|
316
357
|
label: record.label,
|
|
@@ -327,6 +368,10 @@ async function buildRunParams(
|
|
|
327
368
|
const trimmed = text.trim();
|
|
328
369
|
if (trimmed) capture.last = trimmed;
|
|
329
370
|
},
|
|
371
|
+
// Persisted the moment the backend reports it, so a restart at any
|
|
372
|
+
// point after the first message can resume the conversation.
|
|
373
|
+
onSessionId: (sessionId) => agentRegistry.setSessionId(id, sessionId),
|
|
374
|
+
...(resume?.sessionId ? { resumeSessionId: resume.sessionId } : {}),
|
|
330
375
|
...(spec.reasoningEffort ? { reasoningEffort: spec.reasoningEffort } : {}),
|
|
331
376
|
};
|
|
332
377
|
}
|
|
@@ -389,6 +434,7 @@ async function runAgent(
|
|
|
389
434
|
spec: AgentSpawnSpec,
|
|
390
435
|
resolved: { model: string; background: BackgroundRunner },
|
|
391
436
|
acquired: Awaited<ReturnType<typeof acquireBackendInstance>>,
|
|
437
|
+
resume?: ResumePlan,
|
|
392
438
|
): Promise<void> {
|
|
393
439
|
const { model, background } = resolved;
|
|
394
440
|
const { release } = acquired;
|
|
@@ -419,8 +465,11 @@ async function runAgent(
|
|
|
419
465
|
model,
|
|
420
466
|
abortController,
|
|
421
467
|
capture,
|
|
468
|
+
resume,
|
|
422
469
|
);
|
|
423
|
-
if (
|
|
470
|
+
if (agentRegistry.isInterrupted(id)) {
|
|
471
|
+
settled = null;
|
|
472
|
+
} else if (abortController.signal.aborted) {
|
|
424
473
|
// A kill that lands during startup — while the backend is being
|
|
425
474
|
// acquired or the log opened — must not be lost. Handing an
|
|
426
475
|
// already-aborted signal to a backend relies on it checking, and not
|
|
@@ -441,18 +490,38 @@ async function runAgent(
|
|
|
441
490
|
evictLabel: agentContextLabel(id),
|
|
442
491
|
});
|
|
443
492
|
recordBackendRunUsage(record.backendId, usage ?? undefined);
|
|
444
|
-
|
|
445
|
-
|
|
493
|
+
// A backend may swallow the shutdown abort and return normally — the
|
|
494
|
+
// run still did not finish, so it must not settle as done/failed, and
|
|
495
|
+
// it says nothing about the backend's health either way.
|
|
496
|
+
if (agentRegistry.isInterrupted(id)) {
|
|
497
|
+
settled = null;
|
|
498
|
+
} else {
|
|
499
|
+
recordBackendRunSuccess(record.backendId);
|
|
500
|
+
settled = settleSuccess(id, task, capture.last, usage ?? undefined);
|
|
501
|
+
}
|
|
446
502
|
}
|
|
447
503
|
} catch (err) {
|
|
448
|
-
|
|
449
|
-
|
|
504
|
+
if (agentRegistry.isInterrupted(id)) {
|
|
505
|
+
settled = null;
|
|
506
|
+
} else {
|
|
507
|
+
recordBackendRunFailure(record.backendId, err);
|
|
508
|
+
settled = settleFailure(id, task, err);
|
|
509
|
+
}
|
|
450
510
|
} finally {
|
|
451
511
|
await release().catch((err: unknown) =>
|
|
452
512
|
logError("agents", `failed to release backend for ${id}`, err),
|
|
453
513
|
);
|
|
454
514
|
}
|
|
455
515
|
|
|
516
|
+
if (agentRegistry.isInterrupted(id)) {
|
|
517
|
+
// Parked by a daemon shutdown: the persisted row stays `running` for
|
|
518
|
+
// the next boot to resume. No settlement, no delivery, no reaping — the
|
|
519
|
+
// parent is not told its agent died, because it didn't.
|
|
520
|
+
task.fail(new Error("interrupted by daemon shutdown — will resume"));
|
|
521
|
+
agentRegistry.releaseInterrupted(id);
|
|
522
|
+
log("agents", `${id} "${record.label}" interrupted by shutdown — parked`);
|
|
523
|
+
return;
|
|
524
|
+
}
|
|
456
525
|
if (!settled) return;
|
|
457
526
|
log(
|
|
458
527
|
"agents",
|
|
@@ -491,12 +560,264 @@ export function killAgent(agentId: string): boolean {
|
|
|
491
560
|
return agentRegistry.requestKill(agentId);
|
|
492
561
|
}
|
|
493
562
|
|
|
563
|
+
/**
|
|
564
|
+
* Park every live agent for the next boot. Called FIRST in a graceful
|
|
565
|
+
* shutdown — before the frontends and the backend pool go down, since
|
|
566
|
+
* tearing a backend down aborts the runs on it and an unparked agent would
|
|
567
|
+
* record that abort as its death. Idempotent.
|
|
568
|
+
*/
|
|
569
|
+
export function interruptAgentsForRestart(): number {
|
|
570
|
+
const parked = agentRegistry.interruptAll();
|
|
571
|
+
if (parked > 0) {
|
|
572
|
+
log("agents", `Shutdown: parked ${parked} running agent(s) for resume`);
|
|
573
|
+
}
|
|
574
|
+
return parked;
|
|
575
|
+
}
|
|
576
|
+
|
|
494
577
|
/**
|
|
495
578
|
* Abort every live agent — the shutdown lever, alongside heartbeat's and
|
|
496
|
-
* cron's.
|
|
579
|
+
* cron's. Agents are parked first (a no-op if the shutdown already did), so
|
|
580
|
+
* this abort ends the process's hold on them without ending the agents:
|
|
581
|
+
* the next boot resumes them. Returns how many aborts were requested.
|
|
497
582
|
*/
|
|
498
583
|
export function shutdownAgents(): number {
|
|
584
|
+
interruptAgentsForRestart();
|
|
499
585
|
const killed = agentRegistry.killAll();
|
|
500
586
|
if (killed > 0) log("agents", `Shutdown: aborted ${killed} running agent(s)`);
|
|
501
587
|
return killed;
|
|
502
588
|
}
|
|
589
|
+
|
|
590
|
+
// ── Resume after restart ─────────────────────────────────────────────────────
|
|
591
|
+
|
|
592
|
+
/** A restart may resume one agent at most this many times. */
|
|
593
|
+
export const MAX_AGENT_RESUMES = 3;
|
|
594
|
+
/** An agent interrupted longer ago than this is not resumed. */
|
|
595
|
+
const AGENT_RESUME_STALE_MS = 24 * 60 * 60 * 1000;
|
|
596
|
+
/** A resumed run always gets at least this much wall-clock. */
|
|
597
|
+
const AGENT_RESUME_MIN_TIMEOUT_MS = 10 * 60 * 1000;
|
|
598
|
+
/** Settled rows are kept this long for inspection, then pruned at boot. */
|
|
599
|
+
const SETTLED_RETENTION_MS = 7 * 24 * 60 * 60 * 1000;
|
|
600
|
+
/** How much of the previous run log a re-briefed agent is shown. */
|
|
601
|
+
const REBRIEF_LOG_TAIL_CHARS = 12_000;
|
|
602
|
+
|
|
603
|
+
/** The last `max` characters of an agent's previous run log, if any. */
|
|
604
|
+
async function previousLogTail(agentId: string, max: number): Promise<string> {
|
|
605
|
+
try {
|
|
606
|
+
const text = await readFile(agentLogPath(agentId), "utf-8");
|
|
607
|
+
return text.length > max ? `…${text.slice(text.length - max)}` : text;
|
|
608
|
+
} catch {
|
|
609
|
+
return "";
|
|
610
|
+
}
|
|
611
|
+
}
|
|
612
|
+
|
|
613
|
+
/** Settle a restored agent without running it, and tell its parent. */
|
|
614
|
+
async function settleRestored(
|
|
615
|
+
record: AgentRecord,
|
|
616
|
+
patch: Parameters<typeof agentRegistry.settle>[1],
|
|
617
|
+
): Promise<void> {
|
|
618
|
+
const settled = agentRegistry.settle(record.id, patch);
|
|
619
|
+
if (!settled) return;
|
|
620
|
+
log(
|
|
621
|
+
"agents",
|
|
622
|
+
`${record.id} "${record.label}" → ${settled.state} (after restart)`,
|
|
623
|
+
);
|
|
624
|
+
reapChildren(settled);
|
|
625
|
+
await deliverSettlement(settled).catch((err: unknown) =>
|
|
626
|
+
logError("agents", `delivery failed for ${record.id}`, err),
|
|
627
|
+
);
|
|
628
|
+
}
|
|
629
|
+
|
|
630
|
+
/** Mark a row that cannot even be restored (its parent agent is gone). */
|
|
631
|
+
function abandonRow(saved: PersistedAgent, error: string): void {
|
|
632
|
+
try {
|
|
633
|
+
agentsRepo.upsert({
|
|
634
|
+
...saved,
|
|
635
|
+
state: "failed",
|
|
636
|
+
error,
|
|
637
|
+
endedAt: Date.now(),
|
|
638
|
+
updatedAt: Date.now(),
|
|
639
|
+
});
|
|
640
|
+
} catch (err) {
|
|
641
|
+
logError("agents", `could not mark ${saved.id} abandoned`, err);
|
|
642
|
+
}
|
|
643
|
+
}
|
|
644
|
+
|
|
645
|
+
/**
|
|
646
|
+
* Resolve the model a resumed run uses: the one it ran on, else the one it
|
|
647
|
+
* asked for, else the backend default — a model withdrawn across the
|
|
648
|
+
* restart must not strand the agent.
|
|
649
|
+
*/
|
|
650
|
+
async function resolveResumeRun(
|
|
651
|
+
backend: Backend,
|
|
652
|
+
backendId: string,
|
|
653
|
+
candidates: ReadonlyArray<string | undefined>,
|
|
654
|
+
): Promise<Awaited<ReturnType<typeof resolveRun>>> {
|
|
655
|
+
let last: Awaited<ReturnType<typeof resolveRun>> = {
|
|
656
|
+
ok: false,
|
|
657
|
+
error: "no model",
|
|
658
|
+
};
|
|
659
|
+
const tried = new Set<string | undefined>();
|
|
660
|
+
for (const candidate of [...candidates, undefined]) {
|
|
661
|
+
if (tried.has(candidate)) continue;
|
|
662
|
+
tried.add(candidate);
|
|
663
|
+
last = await resolveRun(backend, backendId, candidate);
|
|
664
|
+
if (last.ok) return last;
|
|
665
|
+
}
|
|
666
|
+
return last;
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
/** Bring one interrupted agent back. Never throws. */
|
|
670
|
+
async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
|
|
671
|
+
// A crash leaves no interruption stamp: charge the run up to its last
|
|
672
|
+
// persisted write, which is as close as anyone can know.
|
|
673
|
+
const interruptedAt = saved.interruptedAt ?? saved.updatedAt;
|
|
674
|
+
const elapsedMs =
|
|
675
|
+
saved.interruptedAt === undefined && saved.startedAt !== undefined
|
|
676
|
+
? saved.elapsedMs + Math.max(0, saved.updatedAt - saved.startedAt)
|
|
677
|
+
: saved.elapsedMs;
|
|
678
|
+
|
|
679
|
+
// Already back (a second resume pass in the same process) — leave it be.
|
|
680
|
+
if (agentRegistry.isLive(saved.id)) return;
|
|
681
|
+
const record = agentRegistry.restore({ ...saved, elapsedMs, interruptedAt });
|
|
682
|
+
if (!record) {
|
|
683
|
+
abandonRow(
|
|
684
|
+
saved,
|
|
685
|
+
"interrupted by a daemon restart; its parent agent did not survive it",
|
|
686
|
+
);
|
|
687
|
+
logWarn("agents", `${saved.id}: parent gone after restart — abandoned`);
|
|
688
|
+
return;
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
if (saved.reported) {
|
|
692
|
+
// It finished its job (report_result landed) and was only waiting to
|
|
693
|
+
// wind down — deliver what it said instead of running it again.
|
|
694
|
+
await settleRestored(record, { state: "done" });
|
|
695
|
+
return;
|
|
696
|
+
}
|
|
697
|
+
if (saved.resumeCount >= MAX_AGENT_RESUMES) {
|
|
698
|
+
await settleRestored(record, {
|
|
699
|
+
state: "failed",
|
|
700
|
+
error:
|
|
701
|
+
`interrupted by ${saved.resumeCount + 1} daemon restarts; not ` +
|
|
702
|
+
`resumed again. Its run log is ${agentLogPath(saved.id)}.`,
|
|
703
|
+
});
|
|
704
|
+
return;
|
|
705
|
+
}
|
|
706
|
+
if (now - interruptedAt > AGENT_RESUME_STALE_MS) {
|
|
707
|
+
await settleRestored(record, {
|
|
708
|
+
state: "failed",
|
|
709
|
+
error:
|
|
710
|
+
`interrupted by a daemon restart at ${new Date(interruptedAt).toISOString()} ` +
|
|
711
|
+
`and the daemon was down too long to resume it. Its run log is ` +
|
|
712
|
+
`${agentLogPath(saved.id)}.`,
|
|
713
|
+
});
|
|
714
|
+
return;
|
|
715
|
+
}
|
|
716
|
+
|
|
717
|
+
const backendId = saved.backendId;
|
|
718
|
+
let acquired: Awaited<ReturnType<typeof acquireBackendInstance>>;
|
|
719
|
+
try {
|
|
720
|
+
acquired = await acquireBackendInstance(backendId);
|
|
721
|
+
} catch (err) {
|
|
722
|
+
await settleRestored(record, {
|
|
723
|
+
state: "failed",
|
|
724
|
+
error:
|
|
725
|
+
`interrupted by a daemon restart, and its backend "${backendId}" ` +
|
|
726
|
+
`is unavailable after it: ${errText(err)}`,
|
|
727
|
+
});
|
|
728
|
+
return;
|
|
729
|
+
}
|
|
730
|
+
const resolved = await resolveResumeRun(acquired.backend, backendId, [
|
|
731
|
+
saved.model,
|
|
732
|
+
saved.requestedModel,
|
|
733
|
+
]);
|
|
734
|
+
if (!resolved.ok) {
|
|
735
|
+
await acquired.release().catch(() => {});
|
|
736
|
+
await settleRestored(record, {
|
|
737
|
+
state: "failed",
|
|
738
|
+
error: `interrupted by a daemon restart and could not resume: ${resolved.error}`,
|
|
739
|
+
});
|
|
740
|
+
return;
|
|
741
|
+
}
|
|
742
|
+
|
|
743
|
+
const canResume =
|
|
744
|
+
resolved.background.supportsResume === true &&
|
|
745
|
+
saved.sessionId !== undefined;
|
|
746
|
+
const minutes = Math.round(elapsedMs / 60_000);
|
|
747
|
+
const plan: ResumePlan = canResume
|
|
748
|
+
? {
|
|
749
|
+
prompt: buildResumePrompt({ interruptedAt, elapsedMinutes: minutes }),
|
|
750
|
+
sessionId: saved.sessionId!,
|
|
751
|
+
interruptedAt,
|
|
752
|
+
}
|
|
753
|
+
: {
|
|
754
|
+
prompt: buildRebriefPrompt({
|
|
755
|
+
brief: saved.brief,
|
|
756
|
+
preflight: saved.preflight === true,
|
|
757
|
+
interruptedAt,
|
|
758
|
+
elapsedMinutes: minutes,
|
|
759
|
+
logPath: agentLogPath(saved.id),
|
|
760
|
+
logTail: await previousLogTail(saved.id, REBRIEF_LOG_TAIL_CHARS),
|
|
761
|
+
}),
|
|
762
|
+
interruptedAt,
|
|
763
|
+
};
|
|
764
|
+
|
|
765
|
+
const budget =
|
|
766
|
+
(saved.timeoutMs ?? capsHolder.caps.defaultTimeoutMs) - elapsedMs;
|
|
767
|
+
const timeoutMs = Math.max(AGENT_RESUME_MIN_TIMEOUT_MS, budget);
|
|
768
|
+
const spec: AgentSpawnSpec = {
|
|
769
|
+
brief: saved.brief,
|
|
770
|
+
label: saved.label,
|
|
771
|
+
parent: record.parent,
|
|
772
|
+
backendId,
|
|
773
|
+
...(saved.requestedModel ? { model: saved.requestedModel } : {}),
|
|
774
|
+
...(record.reasoningEffort
|
|
775
|
+
? { reasoningEffort: record.reasoningEffort }
|
|
776
|
+
: {}),
|
|
777
|
+
timeoutMs,
|
|
778
|
+
...(saved.preflight ? { preflight: true } : {}),
|
|
779
|
+
};
|
|
780
|
+
agentRegistry.markResumed(record.id);
|
|
781
|
+
log(
|
|
782
|
+
"agents",
|
|
783
|
+
`${record.id} "${record.label}" resuming after restart ` +
|
|
784
|
+
`(${canResume ? `session ${saved.sessionId}` : "re-briefed"}, ` +
|
|
785
|
+
`${backendId}/${resolved.model}, ${Math.round(timeoutMs / 1000)}s left, ` +
|
|
786
|
+
`resume #${saved.resumeCount + 1})`,
|
|
787
|
+
);
|
|
788
|
+
void runAgent(record, spec, resolved, acquired, plan);
|
|
789
|
+
}
|
|
790
|
+
|
|
791
|
+
/**
|
|
792
|
+
* Boot-time: respawn every agent the previous process left running. Call
|
|
793
|
+
* once, after the dispatcher, backend pool and frontends are up (a resumed
|
|
794
|
+
* agent reaches its tools and its parent through them). Parents come back
|
|
795
|
+
* before their children, so a child re-attaches to its live parent.
|
|
796
|
+
* Returns how many rows were considered.
|
|
797
|
+
*/
|
|
798
|
+
export async function resumeAgentsAfterRestart(): Promise<number> {
|
|
799
|
+
const now = Date.now();
|
|
800
|
+
try {
|
|
801
|
+
const pruned = agentsRepo.pruneSettled(now - SETTLED_RETENTION_MS);
|
|
802
|
+
if (pruned > 0) log("agents", `Pruned ${pruned} settled agent row(s)`);
|
|
803
|
+
} catch (err) {
|
|
804
|
+
logError("agents", "prune of settled agent rows failed", err);
|
|
805
|
+
}
|
|
806
|
+
let rows: PersistedAgent[];
|
|
807
|
+
try {
|
|
808
|
+
rows = agentsRepo.listInterrupted();
|
|
809
|
+
} catch (err) {
|
|
810
|
+
logError("agents", "could not read interrupted agents", err);
|
|
811
|
+
return 0;
|
|
812
|
+
}
|
|
813
|
+
if (rows.length === 0) return 0;
|
|
814
|
+
log("agents", `Resuming ${rows.length} agent(s) interrupted by the restart`);
|
|
815
|
+
for (const saved of rows) {
|
|
816
|
+
try {
|
|
817
|
+
await resumeOne(saved, now);
|
|
818
|
+
} catch (err) {
|
|
819
|
+
logError("agents", `resume of ${saved.id} failed`, err);
|
|
820
|
+
}
|
|
821
|
+
}
|
|
822
|
+
return rows.length;
|
|
823
|
+
}
|
|
@@ -22,11 +22,11 @@
|
|
|
22
22
|
*/
|
|
23
23
|
|
|
24
24
|
import { chmod } from "node:fs/promises";
|
|
25
|
-
import { logWarn } from "
|
|
26
|
-
import { TalonError } from "
|
|
27
|
-
import { verifyManifest } from "
|
|
28
|
-
import { resolvePassphrase } from "
|
|
29
|
-
import type { BackupSettings, Manifest } from "
|
|
25
|
+
import { logWarn } from "../../../util/log.js";
|
|
26
|
+
import { TalonError } from "../../errors.js";
|
|
27
|
+
import { verifyManifest } from "../archive/manifest-auth.js";
|
|
28
|
+
import { resolvePassphrase } from "../passphrase.js";
|
|
29
|
+
import type { BackupSettings, Manifest } from "../types.js";
|
|
30
30
|
|
|
31
31
|
export type ManifestTrust = {
|
|
32
32
|
/** Operator override for unauthenticated manifests. */
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Reporting a staged restore back to the chat that asked for it.
|
|
3
|
+
*
|
|
4
|
+
* `/backup restore <id>` stages the request and restarts; the restore runs
|
|
5
|
+
* in the next boot before any frontend exists (see `applyStagedRestore` in
|
|
6
|
+
* app.ts). Once the frontends are up, the "♻️ Restored snapshot …" line is
|
|
7
|
+
* delivered here: to the requesting chat, on the frontend it came from,
|
|
8
|
+
* and — when that chat can't be reached (frontend disabled, delivery
|
|
9
|
+
* failing) or the request never said who asked — to the operator's
|
|
10
|
+
* primary chat through the admin notifier, which is where it always went
|
|
11
|
+
* before.
|
|
12
|
+
*
|
|
13
|
+
* Core never imports src/frontend, so delivery goes through the same
|
|
14
|
+
* cross-send broker `send_via` uses: each enabled frontend's action
|
|
15
|
+
* handler, keyed by frontend name.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import { log, logWarn } from "../../../util/log.js";
|
|
19
|
+
import {
|
|
20
|
+
isNativeChatId,
|
|
21
|
+
isTelegramChatId,
|
|
22
|
+
numericChatIdFor,
|
|
23
|
+
} from "../../frontend-runtime/chat-id.js";
|
|
24
|
+
import { crossSendTarget } from "../../engine/gateway-actions/cross-send.js";
|
|
25
|
+
import type { RestoreReport } from "../restore.js";
|
|
26
|
+
|
|
27
|
+
/** Who asked for a staged restore, as recorded in restore-pending.json. */
|
|
28
|
+
export type RestoreRequester = {
|
|
29
|
+
/** The requesting chat's key (Telegram id, `d_…`, `discord_…`). */
|
|
30
|
+
requestedBy?: string;
|
|
31
|
+
/** The frontend the request came from. Absent in files staged before it existed. */
|
|
32
|
+
frontend?: string;
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* The frontend a requester belongs to: the recorded one when the request
|
|
37
|
+
* carries it, else inferred from the shape of the chat key — Telegram's
|
|
38
|
+
* ids are numeric, native's start `d_`, Discord's `discord_`. Undefined
|
|
39
|
+
* when neither says.
|
|
40
|
+
*/
|
|
41
|
+
export function requesterFrontend(
|
|
42
|
+
requester: RestoreRequester,
|
|
43
|
+
): string | undefined {
|
|
44
|
+
const explicit =
|
|
45
|
+
typeof requester.frontend === "string"
|
|
46
|
+
? requester.frontend.trim().toLowerCase()
|
|
47
|
+
: "";
|
|
48
|
+
if (explicit) return explicit;
|
|
49
|
+
const key =
|
|
50
|
+
typeof requester.requestedBy === "string" ? requester.requestedBy : "";
|
|
51
|
+
if (!key) return undefined;
|
|
52
|
+
if (isTelegramChatId(key)) return "telegram";
|
|
53
|
+
if (isNativeChatId(key)) return "native";
|
|
54
|
+
if (key.startsWith("discord_")) return "discord";
|
|
55
|
+
return undefined;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** The confirmation line a successful staged restore reports. */
|
|
59
|
+
export function formatRestoreNotice(
|
|
60
|
+
report: Pick<RestoreReport, "id" | "checkpointId">,
|
|
61
|
+
): string {
|
|
62
|
+
return (
|
|
63
|
+
`♻️ Restored snapshot ${report.id}` +
|
|
64
|
+
(report.checkpointId
|
|
65
|
+
? ` (previous state saved as checkpoint ${report.checkpointId})`
|
|
66
|
+
: "")
|
|
67
|
+
);
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Send `text` to one chat through its frontend's registered action
|
|
72
|
+
* handler. The chat key rides in `target` so a frontend that has not
|
|
73
|
+
* seen the chat since the restart (a Discord channel nobody has spoken
|
|
74
|
+
* in yet, a native chat the restored database doesn't list) can adopt it.
|
|
75
|
+
* Resolves true only when the frontend reports the message delivered.
|
|
76
|
+
*/
|
|
77
|
+
async function sendToRequester(
|
|
78
|
+
frontend: string,
|
|
79
|
+
chatKey: string,
|
|
80
|
+
text: string,
|
|
81
|
+
): Promise<boolean> {
|
|
82
|
+
const handler = crossSendTarget(frontend);
|
|
83
|
+
if (!handler) return false;
|
|
84
|
+
const result = await handler(
|
|
85
|
+
{ action: "send_message", text, target: chatKey },
|
|
86
|
+
numericChatIdFor(chatKey),
|
|
87
|
+
);
|
|
88
|
+
return Boolean(result && result.ok === true);
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
const RESTORE_NOTICE_ATTEMPTS = 6;
|
|
92
|
+
const RESTORE_NOTICE_DELAY_MS = 5_000;
|
|
93
|
+
|
|
94
|
+
export type RestoreNoticeOptions = {
|
|
95
|
+
text: string;
|
|
96
|
+
requester: RestoreRequester;
|
|
97
|
+
/** The operator's primary chat — the fallback. */
|
|
98
|
+
notifyAdmin: (text: string) => Promise<unknown>;
|
|
99
|
+
send?: (frontend: string, chatKey: string, text: string) => Promise<boolean>;
|
|
100
|
+
/** Whether a frontend is enabled at all (no point retrying one that isn't). */
|
|
101
|
+
isEnabled?: (frontend: string) => boolean;
|
|
102
|
+
attempts?: number;
|
|
103
|
+
sleep?: (ms: number) => Promise<void>;
|
|
104
|
+
};
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Deliver the restore notice to the requesting chat, falling back to the
|
|
108
|
+
* admin's primary chat. Retries a while when the requester's frontend is
|
|
109
|
+
* enabled but can't deliver yet — the frontends have just started, and a
|
|
110
|
+
* Discord client may still be logging in. Never throws; resolves with
|
|
111
|
+
* where the notice went.
|
|
112
|
+
*/
|
|
113
|
+
export async function deliverRestoreNotice(
|
|
114
|
+
options: RestoreNoticeOptions,
|
|
115
|
+
): Promise<"requester" | "admin"> {
|
|
116
|
+
const {
|
|
117
|
+
text,
|
|
118
|
+
requester,
|
|
119
|
+
notifyAdmin,
|
|
120
|
+
send = sendToRequester,
|
|
121
|
+
isEnabled = (name) => crossSendTarget(name) !== undefined,
|
|
122
|
+
attempts = RESTORE_NOTICE_ATTEMPTS,
|
|
123
|
+
sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms).unref?.()),
|
|
124
|
+
} = options;
|
|
125
|
+
const frontend = requesterFrontend(requester);
|
|
126
|
+
const chatKey =
|
|
127
|
+
typeof requester.requestedBy === "string" ? requester.requestedBy : "";
|
|
128
|
+
|
|
129
|
+
if (frontend && chatKey && isEnabled(frontend)) {
|
|
130
|
+
for (let attempt = 1; attempt <= attempts; attempt++) {
|
|
131
|
+
try {
|
|
132
|
+
if (await send(frontend, chatKey, text)) {
|
|
133
|
+
log("backup", `Restore reported to ${frontend} chat ${chatKey}`);
|
|
134
|
+
return "requester";
|
|
135
|
+
}
|
|
136
|
+
} catch (err) {
|
|
137
|
+
logWarn(
|
|
138
|
+
"backup",
|
|
139
|
+
`Restore report to ${frontend} chat ${chatKey} failed: ${err instanceof Error ? err.message : String(err)}`,
|
|
140
|
+
);
|
|
141
|
+
}
|
|
142
|
+
if (attempt < attempts) await sleep(RESTORE_NOTICE_DELAY_MS);
|
|
143
|
+
}
|
|
144
|
+
logWarn(
|
|
145
|
+
"backup",
|
|
146
|
+
`Could not reach ${frontend} chat ${chatKey}; reporting the restore to the admin chat instead`,
|
|
147
|
+
);
|
|
148
|
+
} else if (chatKey || frontend) {
|
|
149
|
+
logWarn(
|
|
150
|
+
"backup",
|
|
151
|
+
`Restore requester ${frontend ?? "?"}:${chatKey || "?"} is not reachable here; reporting to the admin chat`,
|
|
152
|
+
);
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
try {
|
|
156
|
+
await notifyAdmin(text);
|
|
157
|
+
} catch (err) {
|
|
158
|
+
logWarn(
|
|
159
|
+
"backup",
|
|
160
|
+
`Admin restore report failed: ${err instanceof Error ? err.message : String(err)}`,
|
|
161
|
+
);
|
|
162
|
+
}
|
|
163
|
+
return "admin";
|
|
164
|
+
}
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* The rules that make this safe to run on a live home directory:
|
|
5
5
|
*
|
|
6
6
|
* 1. Verify before you touch anything. The manifest's signature is
|
|
7
|
-
* checked (restore
|
|
7
|
+
* checked (restore/guard.ts), then every part's sha256 and — for
|
|
8
8
|
* encrypted parts — every record's tag; a part that fails is a
|
|
9
9
|
* stopped restore, not a half-applied one.
|
|
10
10
|
* 2. Stage, then swap. The archive is extracted into a staging
|
|
@@ -55,7 +55,7 @@ import {
|
|
|
55
55
|
authenticateManifest,
|
|
56
56
|
makePrivate,
|
|
57
57
|
type ManifestTrust,
|
|
58
|
-
} from "./restore
|
|
58
|
+
} from "./restore/guard.js";
|
|
59
59
|
import { buildSnapshot } from "./snapshot.js";
|
|
60
60
|
import {
|
|
61
61
|
relocateRoot,
|
|
@@ -84,6 +84,12 @@ export type RestorePending = {
|
|
|
84
84
|
requestedAt: number;
|
|
85
85
|
/** Chat key that asked, so the boot can report back. */
|
|
86
86
|
requestedBy?: string;
|
|
87
|
+
/**
|
|
88
|
+
* Frontend the request came from ("telegram", "discord", "native"), so
|
|
89
|
+
* the report goes back the way it came. Absent in files staged before
|
|
90
|
+
* it was recorded — the boot then infers it from `requestedBy`.
|
|
91
|
+
*/
|
|
92
|
+
frontend?: string;
|
|
87
93
|
};
|
|
88
94
|
|
|
89
95
|
export type RestoreReport = {
|
|
@@ -507,7 +513,7 @@ export type RestoreOptions = {
|
|
|
507
513
|
beforeApply?: () => void | Promise<void>;
|
|
508
514
|
/** Skip the automatic pre-restore checkpoint (it has already been taken). */
|
|
509
515
|
skipCheckpoint?: boolean;
|
|
510
|
-
/** Restore a manifest that carries no signature (see restore
|
|
516
|
+
/** Restore a manifest that carries no signature (see restore/guard.ts). */
|
|
511
517
|
allowUnauthenticated?: ManifestTrust["allowUnauthenticated"];
|
|
512
518
|
/**
|
|
513
519
|
* Restoring onto a different machine: relocate the session stores and
|
|
@@ -641,7 +647,9 @@ export async function applyPendingRestore(options: {
|
|
|
641
647
|
settings: BackupSettings;
|
|
642
648
|
home?: string;
|
|
643
649
|
beforeApply?: () => void | Promise<void>;
|
|
644
|
-
}): Promise<
|
|
650
|
+
}): Promise<
|
|
651
|
+
(RestoreReport & { requestedBy?: string; frontend?: string }) | null
|
|
652
|
+
> {
|
|
645
653
|
const home = options.home ?? dirs.root;
|
|
646
654
|
const pending = await readRestorePending(home);
|
|
647
655
|
if (!pending) return null;
|
|
@@ -654,7 +662,11 @@ export async function applyPendingRestore(options: {
|
|
|
654
662
|
beforeApply: options.beforeApply,
|
|
655
663
|
});
|
|
656
664
|
await clearRestorePending(home);
|
|
657
|
-
return {
|
|
665
|
+
return {
|
|
666
|
+
...report,
|
|
667
|
+
requestedBy: pending.requestedBy,
|
|
668
|
+
frontend: pending.frontend,
|
|
669
|
+
};
|
|
658
670
|
} catch (err) {
|
|
659
671
|
logWarn("backup", `Staged restore of ${pending.id} failed: ${String(err)}`);
|
|
660
672
|
await clearRestorePending(home);
|