talon-agent 5.2.1 → 5.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/app.ts +94 -1
- package/src/backend/codex/mcp-config.ts +1 -1
- package/src/backend/openai-agents/mcp-pool.ts +1 -1
- package/src/backend/runtime/index.ts +1 -1
- package/src/cli/commands/backup.ts +396 -0
- package/src/cli/events.ts +14 -0
- package/src/cli/index.ts +64 -45
- package/src/core/backup/archive/digest.ts +77 -0
- package/src/core/backup/archive/tar.ts +567 -0
- package/src/core/backup/archive/zstd.ts +31 -0
- package/src/core/backup/index.ts +54 -0
- package/src/core/backup/plan.ts +273 -0
- package/src/core/backup/restore.ts +410 -0
- package/src/core/backup/scheduler.ts +357 -0
- package/src/core/backup/snapshot.ts +408 -0
- package/src/core/backup/status.ts +194 -0
- package/src/core/backup/store.ts +312 -0
- package/src/core/backup/targets.ts +281 -0
- package/src/core/backup/types.ts +96 -0
- package/src/core/backup/upload.ts +172 -0
- package/src/core/bus/events.ts +45 -1
- package/src/core/config/index.ts +52 -0
- package/src/core/daemon/handoff.ts +192 -0
- package/src/core/daemon/respawn.ts +127 -52
- package/src/core/engine/gateway-actions/backup/index.ts +129 -0
- package/src/core/engine/gateway-actions/index.ts +4 -0
- package/src/core/mcp-hub/talon-server.ts +1 -1
- package/src/core/plugin/actions.ts +34 -0
- package/src/core/plugin/index.ts +5 -1
- package/src/core/tools/{ops/bridge.ts → bridge.ts} +7 -2
- package/src/core/tools/index.ts +2 -0
- package/src/core/tools/ops/backup.ts +67 -0
- package/src/core/tools/types.ts +2 -1
- package/src/core/update/self-update.ts +47 -0
- package/src/frontend/discord/callbacks/components/index.ts +3 -0
- package/src/frontend/discord/commands/backup.ts +203 -0
- package/src/frontend/discord/commands/definitions.ts +35 -0
- package/src/frontend/discord/commands/router.ts +3 -0
- package/src/frontend/telegram/callbacks/backup.ts +55 -0
- package/src/frontend/telegram/callbacks/index.ts +8 -0
- package/src/frontend/telegram/commands/backup.ts +209 -0
- package/src/frontend/telegram/commands/definitions.ts +4 -0
- package/src/frontend/telegram/commands/index.ts +2 -0
- package/src/index.ts +13 -5
- package/src/plugins/playwright/index.ts +39 -3
- package/src/plugins/playwright/provision.ts +21 -0
- package/src/plugins/playwright/version-coupling.ts +195 -0
- package/src/storage/backup/index.ts +82 -0
- package/src/storage/backup/repo.ts +164 -0
- package/src/storage/db.ts +20 -0
- package/src/storage/sql/backups.sql +46 -0
- package/src/storage/sql/db.sql +8 -0
- package/src/storage/sql/schema.sql +30 -0
- package/src/storage/sql/statements.generated.ts +60 -1
- package/src/util/log.ts +129 -81
- package/src/util/paths.ts +5 -0
- /package/src/core/tools/{ops/mcp-env.ts → mcp-env.ts} +0 -0
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Getting a snapshot off this machine — and keeping the remote tidy.
|
|
3
|
+
*
|
|
4
|
+
* Targets run in parallel (they are independent networks) but the parts
|
|
5
|
+
* of one snapshot go up sequentially, manifest last. That order is the
|
|
6
|
+
* completeness marker: a remote snapshot with no manifest.json is an
|
|
7
|
+
* interrupted upload, and nothing will ever mistake it for a restorable
|
|
8
|
+
* backup.
|
|
9
|
+
*
|
|
10
|
+
* One target failing is not a failed backup. The snapshot is already on
|
|
11
|
+
* local disk by the time we get here, so a target that errors records
|
|
12
|
+
* `failed` with its reason and the run carries on — the next run retries
|
|
13
|
+
* it. The manifest keeps the per-target state alongside the parts, so a
|
|
14
|
+
* snapshot restored onto a new machine still knows where its copies are.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { bus } from "../bus/index.js";
|
|
18
|
+
import { log, logWarn } from "../../util/log.js";
|
|
19
|
+
import { dirs } from "../../util/paths.js";
|
|
20
|
+
import {
|
|
21
|
+
deleteBackupRemote,
|
|
22
|
+
recordBackupRemote,
|
|
23
|
+
} from "../../storage/backup/index.js";
|
|
24
|
+
import {
|
|
25
|
+
partPath,
|
|
26
|
+
reindexSnapshot,
|
|
27
|
+
selectPrunable,
|
|
28
|
+
writeManifest,
|
|
29
|
+
} from "./store.js";
|
|
30
|
+
import type { BackupTarget } from "./targets.js";
|
|
31
|
+
import type { Manifest, RemoteState } from "./types.js";
|
|
32
|
+
|
|
33
|
+
function recordState(id: string, targetId: string, state: RemoteState): void {
|
|
34
|
+
recordBackupRemote({
|
|
35
|
+
backupId: id,
|
|
36
|
+
targetId,
|
|
37
|
+
status: state.status,
|
|
38
|
+
remoteId: state.remoteId,
|
|
39
|
+
uploadedAt: state.uploadedAt,
|
|
40
|
+
error: state.error,
|
|
41
|
+
});
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** Push every part, then the manifest. Throws with the target's own words. */
|
|
45
|
+
async function sendSnapshot(
|
|
46
|
+
target: BackupTarget,
|
|
47
|
+
manifest: Manifest,
|
|
48
|
+
home: string,
|
|
49
|
+
): Promise<{ state: RemoteState; deduplicated: boolean }> {
|
|
50
|
+
let deduplicated = manifest.parts.length > 0;
|
|
51
|
+
for (const part of manifest.parts) {
|
|
52
|
+
const result = await target.upload(
|
|
53
|
+
manifest.id,
|
|
54
|
+
{ ...part, path: partPath(manifest.id, part.name, home) },
|
|
55
|
+
manifest,
|
|
56
|
+
);
|
|
57
|
+
if (!result.deduplicated) deduplicated = false;
|
|
58
|
+
}
|
|
59
|
+
const { remoteId } = await target.uploadManifest(manifest.id, manifest);
|
|
60
|
+
return {
|
|
61
|
+
state: { status: "uploaded", remoteId, uploadedAt: Date.now() },
|
|
62
|
+
deduplicated,
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Upload one snapshot to every target. Returns the manifest with its
|
|
68
|
+
* `remote` map filled in; it is rewritten on disk and reindexed so the
|
|
69
|
+
* status surfaces can answer without asking the network.
|
|
70
|
+
*/
|
|
71
|
+
export async function uploadSnapshot(
|
|
72
|
+
manifest: Manifest,
|
|
73
|
+
targets: readonly BackupTarget[],
|
|
74
|
+
home: string = dirs.root,
|
|
75
|
+
): Promise<Manifest> {
|
|
76
|
+
if (targets.length === 0) return manifest;
|
|
77
|
+
await Promise.all(
|
|
78
|
+
targets.map(async (target) => {
|
|
79
|
+
if (!target.ready) {
|
|
80
|
+
const state: RemoteState = {
|
|
81
|
+
status: "pending",
|
|
82
|
+
error: target.detail ?? "target not ready",
|
|
83
|
+
};
|
|
84
|
+
manifest.remote[target.id] = state;
|
|
85
|
+
recordState(manifest.id, target.id, state);
|
|
86
|
+
logWarn("backup", `Target ${target.id} not ready: ${state.error}`);
|
|
87
|
+
return;
|
|
88
|
+
}
|
|
89
|
+
try {
|
|
90
|
+
const { state, deduplicated } = await sendSnapshot(
|
|
91
|
+
target,
|
|
92
|
+
manifest,
|
|
93
|
+
home,
|
|
94
|
+
);
|
|
95
|
+
manifest.remote[target.id] = state;
|
|
96
|
+
recordState(manifest.id, target.id, state);
|
|
97
|
+
bus.publish({
|
|
98
|
+
type: "backup.uploaded",
|
|
99
|
+
snapshotId: manifest.id,
|
|
100
|
+
targetId: target.id,
|
|
101
|
+
bytes: manifest.sizeBytes,
|
|
102
|
+
deduplicated,
|
|
103
|
+
});
|
|
104
|
+
log(
|
|
105
|
+
"backup",
|
|
106
|
+
`Uploaded ${manifest.id} to ${target.id}${deduplicated ? " (deduplicated)" : ""}`,
|
|
107
|
+
);
|
|
108
|
+
} catch (err) {
|
|
109
|
+
const state: RemoteState = {
|
|
110
|
+
status: "failed",
|
|
111
|
+
error: err instanceof Error ? err.message : String(err),
|
|
112
|
+
};
|
|
113
|
+
manifest.remote[target.id] = state;
|
|
114
|
+
recordState(manifest.id, target.id, state);
|
|
115
|
+
logWarn("backup", `Upload to ${target.id} failed: ${state.error}`);
|
|
116
|
+
}
|
|
117
|
+
}),
|
|
118
|
+
);
|
|
119
|
+
await writeManifest(manifest, home);
|
|
120
|
+
reindexSnapshot(manifest);
|
|
121
|
+
return manifest;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Apply the remote retention policy on each target. Same rule as local:
|
|
126
|
+
* the newest `keep` unpinned snapshots survive, pinned ones always do.
|
|
127
|
+
* A target that cannot list is skipped — deleting on a partial listing is
|
|
128
|
+
* how a retention pass turns into data loss.
|
|
129
|
+
*/
|
|
130
|
+
export async function pruneRemote(
|
|
131
|
+
targets: readonly BackupTarget[],
|
|
132
|
+
keep: number,
|
|
133
|
+
): Promise<void> {
|
|
134
|
+
for (const target of targets) {
|
|
135
|
+
if (!target.ready) continue;
|
|
136
|
+
let snapshots;
|
|
137
|
+
try {
|
|
138
|
+
snapshots = await target.list();
|
|
139
|
+
} catch (err) {
|
|
140
|
+
logWarn(
|
|
141
|
+
"backup",
|
|
142
|
+
`Cannot list ${target.id}, skipping remote prune: ${String(err)}`,
|
|
143
|
+
);
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
const doomed = selectPrunable(
|
|
147
|
+
snapshots.map((entry) => ({
|
|
148
|
+
id: entry.snapshotId,
|
|
149
|
+
createdAt: entry.manifest?.createdAt ?? 0,
|
|
150
|
+
pinned: entry.manifest?.pinned === true,
|
|
151
|
+
})),
|
|
152
|
+
keep,
|
|
153
|
+
);
|
|
154
|
+
for (const victim of doomed) {
|
|
155
|
+
try {
|
|
156
|
+
await target.remove(victim.id);
|
|
157
|
+
deleteBackupRemote(victim.id, target.id);
|
|
158
|
+
} catch (err) {
|
|
159
|
+
logWarn(
|
|
160
|
+
"backup",
|
|
161
|
+
`Could not delete ${victim.id} from ${target.id}: ${String(err)}`,
|
|
162
|
+
);
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
if (doomed.length > 0) {
|
|
166
|
+
log(
|
|
167
|
+
"backup",
|
|
168
|
+
`Pruned ${doomed.length} snapshot(s) from ${target.id} (keepRemote=${keep})`,
|
|
169
|
+
);
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
package/src/core/bus/events.ts
CHANGED
|
@@ -15,6 +15,9 @@
|
|
|
15
15
|
* `agent.spawned` when an isolated sub-agent run starts, `agent.settled`
|
|
16
16
|
* when it reaches a terminal state, `agent.message` when a note or a
|
|
17
17
|
* report crosses between an agent and its parent.
|
|
18
|
+
* - `backup.*` — the snapshot lifecycle, published by `core/backup`:
|
|
19
|
+
* `backup.started` / `backup.completed` / `backup.failed` per run, and
|
|
20
|
+
* `backup.uploaded` once per target a snapshot reaches.
|
|
18
21
|
* - `turn.*` — the chat-domain moments inside a turn that other
|
|
19
22
|
* subsystems key off: `turn.started` fires once the warp is bound and
|
|
20
23
|
* the backend is about to run (never for a no-model refusal);
|
|
@@ -91,6 +94,43 @@ interface AgentMessageEvent {
|
|
|
91
94
|
readonly kind: "message" | "result";
|
|
92
95
|
}
|
|
93
96
|
|
|
97
|
+
/** A snapshot run began. `trigger` is what asked for it, never content. */
|
|
98
|
+
interface BackupStartedEvent {
|
|
99
|
+
readonly type: "backup.started";
|
|
100
|
+
readonly kind: "backup" | "checkpoint";
|
|
101
|
+
/** "schedule", "manual", "pre-update", "pre-restore", … */
|
|
102
|
+
readonly trigger: string;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** A snapshot was written and indexed. */
|
|
106
|
+
interface BackupCompletedEvent {
|
|
107
|
+
readonly type: "backup.completed";
|
|
108
|
+
/** Not `id`: the bus stamps every published event with its own numeric id. */
|
|
109
|
+
readonly snapshotId: string;
|
|
110
|
+
readonly kind: "backup" | "checkpoint";
|
|
111
|
+
readonly sizeBytes: number;
|
|
112
|
+
readonly parts: number;
|
|
113
|
+
readonly durationMs: number;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/** A snapshot run failed. `error` is the reason, never a path's contents. */
|
|
117
|
+
interface BackupFailedEvent {
|
|
118
|
+
readonly type: "backup.failed";
|
|
119
|
+
readonly trigger: string;
|
|
120
|
+
readonly error: string;
|
|
121
|
+
readonly consecutiveFailures: number;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** One snapshot reached one remote target. */
|
|
125
|
+
interface BackupUploadedEvent {
|
|
126
|
+
readonly type: "backup.uploaded";
|
|
127
|
+
readonly snapshotId: string;
|
|
128
|
+
readonly targetId: string;
|
|
129
|
+
readonly bytes: number;
|
|
130
|
+
/** True when the target already held every part (content-addressed reuse). */
|
|
131
|
+
readonly deduplicated: boolean;
|
|
132
|
+
}
|
|
133
|
+
|
|
94
134
|
export type TalonEvent =
|
|
95
135
|
| TaskStartedEvent
|
|
96
136
|
| TaskSettledEvent
|
|
@@ -98,7 +138,11 @@ export type TalonEvent =
|
|
|
98
138
|
| TurnCompletedEvent
|
|
99
139
|
| AgentSpawnedEvent
|
|
100
140
|
| AgentSettledEvent
|
|
101
|
-
| AgentMessageEvent
|
|
141
|
+
| AgentMessageEvent
|
|
142
|
+
| BackupStartedEvent
|
|
143
|
+
| BackupCompletedEvent
|
|
144
|
+
| BackupFailedEvent
|
|
145
|
+
| BackupUploadedEvent;
|
|
102
146
|
|
|
103
147
|
export type TalonEventType = TalonEvent["type"];
|
|
104
148
|
|
package/src/core/config/index.ts
CHANGED
|
@@ -6,6 +6,7 @@ import { hardenTalonPermissions } from "./harden.js";
|
|
|
6
6
|
import { setTimezone } from "../../util/time.js";
|
|
7
7
|
import { BACKEND_IDS } from "../agent-runtime/model-ref.js";
|
|
8
8
|
import { REASONING_LEVEL_ORDER } from "../models/reasoning-levels.js";
|
|
9
|
+
import { DEFAULT_BACKUP_SETTINGS } from "../backup/plan.js";
|
|
9
10
|
import {
|
|
10
11
|
assembleSystemPrompt,
|
|
11
12
|
joinSystemPromptParts,
|
|
@@ -458,6 +459,57 @@ const configSchema = z.object({
|
|
|
458
459
|
.default(15 * 60 * 1000),
|
|
459
460
|
})
|
|
460
461
|
.optional(),
|
|
462
|
+
/**
|
|
463
|
+
* Backups & checkpoints (docs/backups.md). Talon's only safety net, so
|
|
464
|
+
* it is on by default: every `intervalHours` it writes a snapshot of
|
|
465
|
+
* the identity, state, database and memory under ~/.talon/backups/,
|
|
466
|
+
* keeps `keepLocal` of them, and uploads to whatever remote targets
|
|
467
|
+
* are registered (`targets: []` keeps everything local).
|
|
468
|
+
*
|
|
469
|
+
* - `workspaceInclude` — the workspace is mostly bulk that can be
|
|
470
|
+
* refetched; this is the subset that IS the agent.
|
|
471
|
+
* - `extraPaths` — absolute or `~/…` paths outside ~/.talon worth
|
|
472
|
+
* carrying along (a Claude Code memory directory, say).
|
|
473
|
+
* - `checkpointBeforeUpdate` — pinned checkpoint before `/update`,
|
|
474
|
+
* so a bad update is one restore away from undone.
|
|
475
|
+
* - `notifyChatId` — where failures are reported; falls back to the
|
|
476
|
+
* admin chat.
|
|
477
|
+
*/
|
|
478
|
+
backup: z
|
|
479
|
+
.object({
|
|
480
|
+
enabled: z.boolean().default(DEFAULT_BACKUP_SETTINGS.enabled),
|
|
481
|
+
intervalHours: z
|
|
482
|
+
.number()
|
|
483
|
+
.int()
|
|
484
|
+
.min(1)
|
|
485
|
+
.max(168)
|
|
486
|
+
.default(DEFAULT_BACKUP_SETTINGS.intervalHours),
|
|
487
|
+
keepLocal: z
|
|
488
|
+
.number()
|
|
489
|
+
.int()
|
|
490
|
+
.min(1)
|
|
491
|
+
.max(1000)
|
|
492
|
+
.default(DEFAULT_BACKUP_SETTINGS.keepLocal),
|
|
493
|
+
keepRemote: z
|
|
494
|
+
.number()
|
|
495
|
+
.int()
|
|
496
|
+
.min(1)
|
|
497
|
+
.max(1000)
|
|
498
|
+
.default(DEFAULT_BACKUP_SETTINGS.keepRemote),
|
|
499
|
+
includePalace: z.boolean().default(DEFAULT_BACKUP_SETTINGS.includePalace),
|
|
500
|
+
workspaceInclude: z
|
|
501
|
+
.array(z.string().min(1))
|
|
502
|
+
.default([...DEFAULT_BACKUP_SETTINGS.workspaceInclude]),
|
|
503
|
+
extraPaths: z.array(z.string().min(1)).default([]),
|
|
504
|
+
/** Unset = every registered target; `[]` = local only. */
|
|
505
|
+
targets: z.array(z.string().min(1)).optional(),
|
|
506
|
+
checkpointBeforeUpdate: z
|
|
507
|
+
.boolean()
|
|
508
|
+
.default(DEFAULT_BACKUP_SETTINGS.checkpointBeforeUpdate),
|
|
509
|
+
notifyChatId: z.string().optional(),
|
|
510
|
+
})
|
|
511
|
+
.strict()
|
|
512
|
+
.optional(),
|
|
461
513
|
braveApiKey: z.string().optional(),
|
|
462
514
|
/**
|
|
463
515
|
* Codex-specific OpenAI API key. Prefer this, CODEX_API_KEY, or
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Handoff watcher — the process that proves a `/restart` or `/update`
|
|
3
|
+
* actually landed.
|
|
4
|
+
*
|
|
5
|
+
* The outgoing daemon spawns its successor and then calls
|
|
6
|
+
* `process.exit()`. Until 2026-09-18 that was the whole handoff, which
|
|
7
|
+
* means nothing in the system knew whether the successor had come up.
|
|
8
|
+
* On that day one didn't: it was spawned, lived about twenty seconds,
|
|
9
|
+
* never bound its gateway, and died — and because the dying parent was
|
|
10
|
+
* the only party to the handoff, Talon simply stayed down until a human
|
|
11
|
+
* noticed forty-five minutes later.
|
|
12
|
+
*
|
|
13
|
+
* So the handoff gets a witness. `spawnSuccessor()` (./respawn.ts)
|
|
14
|
+
* starts this watcher detached, sharing the successor's respawn.log fd.
|
|
15
|
+
* It polls identity-verified discovery (./discovery.ts — `app: "talon"`,
|
|
16
|
+
* `mode: "daemon"`, matching pid) until the successor answers /health or
|
|
17
|
+
* the window closes. Only a /health answer counts: discovery will also
|
|
18
|
+
* report a daemon whose pid is merely alive, and "the process exists" is
|
|
19
|
+
* exactly the claim that was false for 20 seconds on 2026-09-18. If the
|
|
20
|
+
* successor never serves, the watcher starts the daemon exactly
|
|
21
|
+
* the way `talon start` does (./control.ts — same spawn, same boot
|
|
22
|
+
* verification) and says why in the log. The watcher is tiny on purpose:
|
|
23
|
+
* `src/index.ts` dispatches its subcommand before the app graph loads,
|
|
24
|
+
* so it costs a bare runtime and these three modules.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import { dirname, resolve } from "node:path";
|
|
28
|
+
import { log, logError, logWarn } from "../../util/log.js";
|
|
29
|
+
import { startDaemon, type StartOutcome } from "./control.js";
|
|
30
|
+
import { findRunningInstance, type RunningInstance } from "./discovery.js";
|
|
31
|
+
import { isProcessAlive } from "./pidfile.js";
|
|
32
|
+
|
|
33
|
+
/** Hidden subcommand: `talon _handoff-watch <successor-pid>`. */
|
|
34
|
+
export const HANDOFF_WATCH_SUBCOMMAND = "_handoff-watch";
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* How long a successor gets to answer /health. Generous on purpose: a
|
|
38
|
+
* cold boot with every plugin and MCP server takes ~10s on the reference
|
|
39
|
+
* host, and `startDaemon()` itself waits 30s before calling a boot late.
|
|
40
|
+
*/
|
|
41
|
+
const HANDOFF_WINDOW_MS = 90_000;
|
|
42
|
+
const POLL_MS = 500;
|
|
43
|
+
|
|
44
|
+
export type HandoffOutcome =
|
|
45
|
+
| { ok: true; via: "successor" | "restart"; pid: number; port?: number }
|
|
46
|
+
| { ok: false; reason: string };
|
|
47
|
+
|
|
48
|
+
export type WatchHandoffOptions = {
|
|
49
|
+
/** The pid `spawnSuccessor()` created. */
|
|
50
|
+
childPid: number;
|
|
51
|
+
/** Repo/package root, for the `talon start` equivalent. */
|
|
52
|
+
pkgRoot: string;
|
|
53
|
+
windowMs?: number;
|
|
54
|
+
pollMs?: number;
|
|
55
|
+
pidfilePath?: string;
|
|
56
|
+
/** Injection seams for tests. */
|
|
57
|
+
find?: typeof findRunningInstance;
|
|
58
|
+
alive?: (pid: number) => boolean;
|
|
59
|
+
start?: typeof startDaemon;
|
|
60
|
+
sleep?: (ms: number) => Promise<void>;
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
function defaultSleep(ms: number): Promise<void> {
|
|
64
|
+
return new Promise((r) => setTimeout(r, ms));
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** Why the wait ended without a live daemon. */
|
|
68
|
+
type WaitFailure = "successor-exited" | "window-expired";
|
|
69
|
+
|
|
70
|
+
/** Discovery found a process; only a /health answer proves it serves. */
|
|
71
|
+
function isServing(instance: RunningInstance | null): boolean {
|
|
72
|
+
return instance?.health !== undefined;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Poll until a serving daemon answers, the successor process
|
|
77
|
+
* disappears, or the window closes. A daemon that isn't our child still
|
|
78
|
+
* counts: the goal is a live Talon, not a particular pid.
|
|
79
|
+
*/
|
|
80
|
+
async function awaitDaemon(
|
|
81
|
+
opts: WatchHandoffOptions,
|
|
82
|
+
): Promise<RunningInstance | WaitFailure> {
|
|
83
|
+
const find = opts.find ?? findRunningInstance;
|
|
84
|
+
const alive = opts.alive ?? isProcessAlive;
|
|
85
|
+
const sleep = opts.sleep ?? defaultSleep;
|
|
86
|
+
const pollMs = opts.pollMs ?? POLL_MS;
|
|
87
|
+
const deadline = Date.now() + (opts.windowMs ?? HANDOFF_WINDOW_MS);
|
|
88
|
+
|
|
89
|
+
while (Date.now() < deadline) {
|
|
90
|
+
const instance = await find(opts.pidfilePath);
|
|
91
|
+
if (instance && isServing(instance)) return instance;
|
|
92
|
+
if (!alive(opts.childPid)) return "successor-exited";
|
|
93
|
+
await sleep(pollMs);
|
|
94
|
+
}
|
|
95
|
+
return "window-expired";
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
const FAILURE_DETAIL: Record<WaitFailure, string> = {
|
|
99
|
+
"successor-exited": "the successor exited before serving /health",
|
|
100
|
+
"window-expired": "the successor never answered /health in time",
|
|
101
|
+
};
|
|
102
|
+
|
|
103
|
+
function describeStart(outcome: StartOutcome): string {
|
|
104
|
+
if (outcome.ok) return `started (pid ${outcome.pid})`;
|
|
105
|
+
if (outcome.reason === "already-running") {
|
|
106
|
+
return `already running (pid ${outcome.instance.pid})`;
|
|
107
|
+
}
|
|
108
|
+
if (outcome.reason === "boot-timeout") return "spawned but not yet healthy";
|
|
109
|
+
return `${outcome.reason}${outcome.detail ? `: ${outcome.detail}` : ""}`;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function toOutcome(started: StartOutcome, why: string): HandoffOutcome {
|
|
113
|
+
if (started.ok) {
|
|
114
|
+
return { ok: true, via: "restart", pid: started.pid, port: started.port };
|
|
115
|
+
}
|
|
116
|
+
if (started.reason === "already-running") {
|
|
117
|
+
const inst = started.instance;
|
|
118
|
+
// `talon start` refuses while a pid is alive — right, since a second
|
|
119
|
+
// daemon would fight the first for Telegram's getUpdates. But an
|
|
120
|
+
// alive pid that has never served /health is the failure, not the
|
|
121
|
+
// recovery, so it is reported as one.
|
|
122
|
+
if (!isServing(inst)) {
|
|
123
|
+
return {
|
|
124
|
+
ok: false,
|
|
125
|
+
reason: `${why}; pid ${inst.pid} is alive but not serving — kill it and run \`talon start\``,
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
return { ok: true, via: "restart", pid: inst.pid, port: inst.port };
|
|
129
|
+
}
|
|
130
|
+
return { ok: false, reason: `${why}; restart ${describeStart(started)}` };
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Verify the handoff, and repair it if it failed. Never throws: this
|
|
135
|
+
* process exists only to make the outcome known.
|
|
136
|
+
*/
|
|
137
|
+
export async function watchHandoff(
|
|
138
|
+
opts: WatchHandoffOptions,
|
|
139
|
+
): Promise<HandoffOutcome> {
|
|
140
|
+
const result = await awaitDaemon(opts);
|
|
141
|
+
if (typeof result !== "string") {
|
|
142
|
+
const via = result.pid === opts.childPid ? "successor" : "restart";
|
|
143
|
+
log(
|
|
144
|
+
"shutdown",
|
|
145
|
+
`Handoff verified — daemon pid ${result.pid} serving on ` +
|
|
146
|
+
`:${result.port ?? "?"} (${via})`,
|
|
147
|
+
);
|
|
148
|
+
return { ok: true, via, pid: result.pid, port: result.port };
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const why = FAILURE_DETAIL[result];
|
|
152
|
+
logWarn(
|
|
153
|
+
"shutdown",
|
|
154
|
+
`Handoff failed — ${why}; starting Talon the way \`talon start\` does`,
|
|
155
|
+
);
|
|
156
|
+
const start = opts.start ?? startDaemon;
|
|
157
|
+
const started = await start({
|
|
158
|
+
pkgRoot: opts.pkgRoot,
|
|
159
|
+
pidfilePath: opts.pidfilePath,
|
|
160
|
+
});
|
|
161
|
+
const outcome = toOutcome(started, why);
|
|
162
|
+
if (outcome.ok)
|
|
163
|
+
log("shutdown", `Handoff recovered — ${describeStart(started)}`);
|
|
164
|
+
else logError("shutdown", `Handoff unrecoverable — ${outcome.reason}`);
|
|
165
|
+
return outcome;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** src/core/daemon/ → the package root. */
|
|
169
|
+
function packageRoot(): string {
|
|
170
|
+
const here = import.meta.dirname ?? process.cwd();
|
|
171
|
+
return here.includes("$bunfs") || here.includes("~BUN")
|
|
172
|
+
? dirname(process.execPath)
|
|
173
|
+
: resolve(here, "..", "..", "..");
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/** Entry point for `talon _handoff-watch <pid>` (src/index.ts). */
|
|
177
|
+
export async function runHandoffWatch(argv: readonly string[]): Promise<void> {
|
|
178
|
+
const childPid = Number.parseInt(argv[0] ?? "", 10);
|
|
179
|
+
if (!Number.isInteger(childPid) || childPid <= 0) {
|
|
180
|
+
logError("shutdown", `Handoff watcher got no successor pid (${argv[0]})`);
|
|
181
|
+
process.exitCode = 2;
|
|
182
|
+
return;
|
|
183
|
+
}
|
|
184
|
+
log("shutdown", `Handoff watcher armed for pid ${childPid}`);
|
|
185
|
+
try {
|
|
186
|
+
const outcome = await watchHandoff({ childPid, pkgRoot: packageRoot() });
|
|
187
|
+
process.exitCode = outcome.ok ? 0 : 1;
|
|
188
|
+
} catch (err) {
|
|
189
|
+
logError("shutdown", "Handoff watcher crashed", err);
|
|
190
|
+
process.exitCode = 1;
|
|
191
|
+
}
|
|
192
|
+
}
|