@songsid/agend 2.1.6-beta.1 → 2.1.6-beta.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-endpoint.d.ts +18 -0
- package/dist/agent-endpoint.js +53 -2
- package/dist/agent-endpoint.js.map +1 -1
- package/dist/apply-job.d.ts +114 -0
- package/dist/apply-job.js +214 -0
- package/dist/apply-job.js.map +1 -0
- package/dist/backend/claude-code.d.ts +25 -0
- package/dist/backend/claude-code.js +127 -0
- package/dist/backend/claude-code.js.map +1 -1
- package/dist/backend/codex.d.ts +41 -0
- package/dist/backend/codex.js +133 -0
- package/dist/backend/codex.js.map +1 -1
- package/dist/backend/credential-profile.d.ts +159 -0
- package/dist/backend/credential-profile.js +340 -0
- package/dist/backend/credential-profile.js.map +1 -0
- package/dist/backend/factory.js +4 -1
- package/dist/backend/factory.js.map +1 -1
- package/dist/backend/kiro-auth-store.d.ts +25 -0
- package/dist/backend/kiro-auth-store.js +61 -0
- package/dist/backend/kiro-auth-store.js.map +1 -0
- package/dist/backend/kiro.d.ts +8 -0
- package/dist/backend/kiro.js +31 -2
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/muse.d.ts +90 -0
- package/dist/backend/muse.js +437 -0
- package/dist/backend/muse.js.map +1 -0
- package/dist/backend/types.d.ts +41 -0
- package/dist/backend/types.js.map +1 -1
- package/dist/channel/access-manager.d.ts +20 -0
- package/dist/channel/access-manager.js +25 -0
- package/dist/channel/access-manager.js.map +1 -1
- package/dist/channel/adapters/discord.d.ts +17 -0
- package/dist/channel/adapters/discord.js +72 -11
- package/dist/channel/adapters/discord.js.map +1 -1
- package/dist/channel/adapters/telegram.d.ts +6 -0
- package/dist/channel/adapters/telegram.js +36 -0
- package/dist/channel/adapters/telegram.js.map +1 -1
- package/dist/channel/markdown-chunk.d.ts +8 -3
- package/dist/channel/markdown-chunk.js +12 -3
- package/dist/channel/markdown-chunk.js.map +1 -1
- package/dist/channel/mcp-server.js +9 -12
- package/dist/channel/mcp-server.js.map +1 -1
- package/dist/channel/mcp-tools.d.ts +6 -2
- package/dist/channel/mcp-tools.js +6 -36
- package/dist/channel/mcp-tools.js.map +1 -1
- package/dist/channel/tool-router.js +32 -3
- package/dist/channel/tool-router.js.map +1 -1
- package/dist/channel/types.d.ts +18 -0
- package/dist/cli.js +113 -0
- package/dist/cli.js.map +1 -1
- package/dist/config-validator.js +45 -5
- package/dist/config-validator.js.map +1 -1
- package/dist/connection-secrets.d.ts +96 -0
- package/dist/connection-secrets.js +25 -0
- package/dist/connection-secrets.js.map +1 -0
- package/dist/daemon.d.ts +61 -0
- package/dist/daemon.js +359 -88
- package/dist/daemon.js.map +1 -1
- package/dist/event-log.d.ts +9 -0
- package/dist/event-log.js +21 -0
- package/dist/event-log.js.map +1 -1
- package/dist/fleet-level-config.d.ts +42 -0
- package/dist/fleet-level-config.js +78 -0
- package/dist/fleet-level-config.js.map +1 -0
- package/dist/fleet-lock.d.ts +21 -0
- package/dist/fleet-lock.js +42 -8
- package/dist/fleet-lock.js.map +1 -1
- package/dist/fleet-manager.d.ts +425 -2
- package/dist/fleet-manager.js +2014 -182
- package/dist/fleet-manager.js.map +1 -1
- package/dist/general-knowledge/skills/credential-profiles/SKILL.md +145 -0
- package/dist/general-knowledge/skills/worker-collaboration/SKILL.md +11 -1
- package/dist/instance-config-impact.d.ts +55 -0
- package/dist/instance-config-impact.js +152 -0
- package/dist/instance-config-impact.js.map +1 -0
- package/dist/instance-lifecycle.js +8 -4
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/locale.js +24 -0
- package/dist/locale.js.map +1 -1
- package/dist/logger.js +42 -25
- package/dist/logger.js.map +1 -1
- package/dist/muse-usage-relay.d.ts +72 -0
- package/dist/muse-usage-relay.js +419 -0
- package/dist/muse-usage-relay.js.map +1 -0
- package/dist/outbound-handlers.d.ts +5 -1
- package/dist/outbound-handlers.js +199 -1
- package/dist/outbound-handlers.js.map +1 -1
- package/dist/outbound-schemas.d.ts +7 -5
- package/dist/outbound-schemas.js +3 -2
- package/dist/outbound-schemas.js.map +1 -1
- package/dist/provider-probe.d.ts +48 -0
- package/dist/provider-probe.js +121 -0
- package/dist/provider-probe.js.map +1 -0
- package/dist/provider-secret-registry.d.ts +98 -0
- package/dist/provider-secret-registry.js +334 -0
- package/dist/provider-secret-registry.js.map +1 -0
- package/dist/quickstart-api.d.ts +164 -0
- package/dist/quickstart-api.js +351 -0
- package/dist/quickstart-api.js.map +1 -0
- package/dist/quickstart.js +23 -50
- package/dist/quickstart.js.map +1 -1
- package/dist/scheduler/db.js +17 -5
- package/dist/scheduler/db.js.map +1 -1
- package/dist/scheduler/db.test.js +34 -1
- package/dist/scheduler/db.test.js.map +1 -1
- package/dist/scheduler/types.d.ts +4 -0
- package/dist/scheduler/types.js.map +1 -1
- package/dist/secret-store.d.ts +25 -0
- package/dist/secret-store.js +165 -0
- package/dist/secret-store.js.map +1 -0
- package/dist/self-restart-limit.d.ts +21 -0
- package/dist/self-restart-limit.js +129 -0
- package/dist/self-restart-limit.js.map +1 -0
- package/dist/settings-api.d.ts +116 -0
- package/dist/settings-api.js +530 -5
- package/dist/settings-api.js.map +1 -1
- package/dist/setup-auth.d.ts +113 -0
- package/dist/setup-auth.js +181 -0
- package/dist/setup-auth.js.map +1 -0
- package/dist/setup-form.d.ts +18 -0
- package/dist/setup-form.js +315 -0
- package/dist/setup-form.js.map +1 -0
- package/dist/setup-host.d.ts +162 -0
- package/dist/setup-host.js +496 -0
- package/dist/setup-host.js.map +1 -0
- package/dist/setup-marker.d.ts +8 -0
- package/dist/setup-marker.js +39 -0
- package/dist/setup-marker.js.map +1 -0
- package/dist/setup-tunnel-consent.d.ts +53 -0
- package/dist/setup-tunnel-consent.js +82 -0
- package/dist/setup-tunnel-consent.js.map +1 -0
- package/dist/setup-wizard.js +4 -0
- package/dist/setup-wizard.js.map +1 -1
- package/dist/steer-capability.js +4 -1
- package/dist/steer-capability.js.map +1 -1
- package/dist/tips.js +1 -1
- package/dist/tips.js.map +1 -1
- package/dist/tmux-manager.js +29 -4
- package/dist/tmux-manager.js.map +1 -1
- package/dist/tool-permissions-notice.d.ts +38 -0
- package/dist/tool-permissions-notice.js +92 -0
- package/dist/tool-permissions-notice.js.map +1 -0
- package/dist/tool-permissions.d.ts +120 -0
- package/dist/tool-permissions.js +277 -0
- package/dist/tool-permissions.js.map +1 -0
- package/dist/topic-commands.js +1 -0
- package/dist/topic-commands.js.map +1 -1
- package/dist/transcript-sources.d.ts +12 -1
- package/dist/transcript-sources.js +19 -3
- package/dist/transcript-sources.js.map +1 -1
- package/dist/tunnel/cloudflared.d.ts +71 -0
- package/dist/tunnel/cloudflared.js +447 -0
- package/dist/tunnel/cloudflared.js.map +1 -0
- package/dist/tunnel/lease.d.ts +75 -0
- package/dist/tunnel/lease.js +209 -0
- package/dist/tunnel/lease.js.map +1 -0
- package/dist/tunnel/manager.d.ts +56 -0
- package/dist/tunnel/manager.js +167 -0
- package/dist/tunnel/manager.js.map +1 -0
- package/dist/tunnel/types.d.ts +119 -0
- package/dist/tunnel/types.js +33 -0
- package/dist/tunnel/types.js.map +1 -0
- package/dist/types.d.ts +2 -0
- package/dist/ui/dashboard.html +4 -3
- package/dist/ui/settings.html +808 -156
- package/dist/ui/view.html +2 -2
- package/dist/usage/i18n-keys.d.ts +1 -1
- package/dist/usage/i18n-keys.js +1 -1
- package/dist/usage/i18n-keys.js.map +1 -1
- package/dist/usage/providers.d.ts +23 -14
- package/dist/usage/providers.js +220 -58
- package/dist/usage/providers.js.map +1 -1
- package/dist/usage/usage-api.d.ts +13 -0
- package/dist/usage/usage-api.js +77 -1
- package/dist/usage/usage-api.js.map +1 -1
- package/dist/web-api.js +8 -5
- package/dist/web-api.js.map +1 -1
- package/dist/web-auth.d.ts +75 -0
- package/dist/web-auth.js +208 -2
- package/dist/web-auth.js.map +1 -1
- package/dist/web-terminal.d.ts +30 -1
- package/dist/web-terminal.js +86 -3
- package/dist/web-terminal.js.map +1 -1
- package/package.json +2 -1
package/dist/fleet-manager.js
CHANGED
|
@@ -5,7 +5,6 @@ import { freemem, totalmem, cpus } from "node:os";
|
|
|
5
5
|
import { createServer } from "node:http";
|
|
6
6
|
import { join, dirname, basename } from "node:path";
|
|
7
7
|
import { fileURLToPath } from "node:url";
|
|
8
|
-
import { isDeepStrictEqual } from "node:util";
|
|
9
8
|
import { getAgendHome, ensureWorkspaceGit } from "./paths.js";
|
|
10
9
|
import { beginFullRestartProgress as persistFullRestartProgress, beginUpdateProgress as persistUpdateProgress, clearUpdateMarker, isUpdateInProgress, readUpdateProgress, setUpdateProgressStage, updateProgressOperation, } from "./update-marker.js";
|
|
11
10
|
import { formatUpdateProgress } from "./update-progress.js";
|
|
@@ -45,7 +44,7 @@ import { StatuslineWatcher } from "./statusline-watcher.js";
|
|
|
45
44
|
import { outboundHandlers } from "./outbound-handlers.js";
|
|
46
45
|
import { handleWebRequest, broadcastSseEvent } from "./web-api.js";
|
|
47
46
|
import { handleViewRequest, isViewPath } from "./view-api.js";
|
|
48
|
-
import { handleUsageRequest, isUsagePath, usageProviderIdForBackend } from "./usage/usage-api.js";
|
|
47
|
+
import { filterUsageProviders, formatDiscordUsageActivity, getUsageSnapshot, handleUsageRequest, isUsagePath, usageProviderIdForBackend } from "./usage/usage-api.js";
|
|
49
48
|
import { LOGIN_FLOWS, LOGIN_BACKEND_ALIASES, checkAuthStatus } from "./login-flows.js";
|
|
50
49
|
import { LoginSession } from "./login-manager.js";
|
|
51
50
|
import { LoginController, LOGIN_TOKEN_RESEND_PREFIX, POST_LOGIN_RECOVERY_DEADLINE_MS, announcePostLoginRecovery } from "./login-controller.js";
|
|
@@ -53,15 +52,30 @@ import { runBeforeDeadline } from "./deadline.js";
|
|
|
53
52
|
import { LoginWindowLock } from "./login-window-lock.js";
|
|
54
53
|
import { handleSettingsRequest } from "./settings-api.js";
|
|
55
54
|
import { setLocale, detectLocale, getLocale, t } from "./locale.js";
|
|
56
|
-
import { handleAgentRequest } from "./agent-endpoint.js";
|
|
55
|
+
import { handleAgentRequest, ToolNotPermittedError } from "./agent-endpoint.js";
|
|
57
56
|
import { ClassicChannelManager, getClassicBackendChoices, isSelectableClassicBackend, readClassicLastActivityAt } from "./classic-channel-manager.js";
|
|
58
57
|
import { assertExplicitInstanceRemoval } from "./instance-removal.js";
|
|
59
58
|
import { validateFleetConfig } from "./config-validator.js";
|
|
60
59
|
import { readLastInboundAt } from "./daemon.js";
|
|
61
60
|
import { clearPausedMarker } from "./pause-marker.js";
|
|
62
|
-
import { releaseProcessFleetLock } from "./fleet-lock.js";
|
|
61
|
+
import { isFleetStartCommandLine, readProcessCommandLine, releaseProcessFleetLock } from "./fleet-lock.js";
|
|
62
|
+
import { isSetupComplete, markSetupComplete } from "./setup-marker.js";
|
|
63
|
+
import { manualCleanupMessage, reapStaleTunnel } from "./tunnel/lease.js";
|
|
64
|
+
import { buildToolPermissionsNotice } from "./tool-permissions-notice.js";
|
|
63
65
|
import { GENERAL_PAUSE_ERROR, isGeneralInstance } from "./general-instance.js";
|
|
64
|
-
import {
|
|
66
|
+
import { mayUseTool, resolveToolSet, scheduleOpRefusal, toolForIpcType, toolRefusedMessage, } from "./tool-permissions.js";
|
|
67
|
+
import { decideWebGate, loadOrCreateWebToken, readWebToken } from "./web-auth.js";
|
|
68
|
+
import { fleetLevelDifferences, fleetLevelSignature } from "./fleet-level-config.js";
|
|
69
|
+
import { checkSelfRestartAllowance, recordSelfRestartAttempt } from "./self-restart-limit.js";
|
|
70
|
+
import { SecretStore } from "./secret-store.js";
|
|
71
|
+
import { opaqueId, safeSecretError, SECRET_CHALLENGE_TTL_MS, } from "./connection-secrets.js";
|
|
72
|
+
import { verifyDiscordToken, verifyTelegramToken } from "./provider-probe.js";
|
|
73
|
+
import { PROVIDER_SECRET_SPECS, providerSecretSpec, providerRegistryEnvKeys, isReservedProviderEnvKey, verifyProviderSecret, } from "./provider-secret-registry.js";
|
|
74
|
+
/** A self-restart is a whole service restart; 120s is the apply budget, not this. */
|
|
75
|
+
const SELF_RESTART_DEADLINE_MS = 300_000;
|
|
76
|
+
import { APPLY_FLEET_TARGET, ApplyJobStore, viewOf, } from "./apply-job.js";
|
|
77
|
+
import { instanceCredentialProfile } from "./backend/credential-profile.js";
|
|
78
|
+
import { classifyInstanceChange, CLASSIC_HOT_CONFIG_KEYS, HOT_INSTANCE_CONFIG_KEYS, hotConfigUpdate, } from "./instance-config-impact.js";
|
|
65
79
|
import { formatRestartProgressCompletion, RESTART_PROGRESS_TERMINAL_TIMEOUT_MS, RestartProgress, } from "./restart-progress.js";
|
|
66
80
|
import { launchFullRestartHelper } from "./full-restart.js";
|
|
67
81
|
import { collectRedundantInstanceDefaultPaths } from "./fleet-yaml-slim.js";
|
|
@@ -179,31 +193,6 @@ const DELIVERY_STATUS_EMOJIS = new Set(["👀", "⏳", "✅", "❌"]);
|
|
|
179
193
|
* emoji never changes the documented delivery-state protocol.
|
|
180
194
|
*/
|
|
181
195
|
const IGNORED_REACTION_EMOJIS = new Set(["📷"]);
|
|
182
|
-
const HOT_INSTANCE_CONFIG_KEYS = new Set([
|
|
183
|
-
"tool_progress",
|
|
184
|
-
"reply_completion_guard",
|
|
185
|
-
"mcp_proxy_reply",
|
|
186
|
-
"auto_pause_after",
|
|
187
|
-
"warm_cap",
|
|
188
|
-
"display_name",
|
|
189
|
-
"description",
|
|
190
|
-
"tags",
|
|
191
|
-
"log_level",
|
|
192
|
-
]);
|
|
193
|
-
function splitHotColdConfig(config) {
|
|
194
|
-
const hot = {};
|
|
195
|
-
const cold = {};
|
|
196
|
-
for (const [key, value] of Object.entries(config)) {
|
|
197
|
-
(HOT_INSTANCE_CONFIG_KEYS.has(key) ? hot : cold)[key] = value;
|
|
198
|
-
}
|
|
199
|
-
return { hot, cold };
|
|
200
|
-
}
|
|
201
|
-
function hotConfigUpdate(config) {
|
|
202
|
-
const update = {};
|
|
203
|
-
for (const key of HOT_INSTANCE_CONFIG_KEYS)
|
|
204
|
-
update[key] = config[key] ?? null;
|
|
205
|
-
return update;
|
|
206
|
-
}
|
|
207
196
|
/**
|
|
208
197
|
* How long a delivery waits out a disconnected instance IPC before giving up.
|
|
209
198
|
*
|
|
@@ -232,6 +221,8 @@ const CLEAR_CONFIRM_TIMEOUT_MS = 15_000;
|
|
|
232
221
|
/** Default lifetime for long-lived nonce prompts (clear overrides this to 15s). */
|
|
233
222
|
const NONCE_BUTTON_TIMEOUT_MS = 15 * 60_000;
|
|
234
223
|
const TIP_BUTTON_TIMEOUT_MS = 24 * 60 * 60_000;
|
|
224
|
+
/** How long shutdown will spend retiring still-armed button prompts. */
|
|
225
|
+
const NONCE_RETIRE_BUDGET_MS = 5_000;
|
|
235
226
|
const CLI_ENV_TTL_MS = 24 * 60 * 60 * 1000; // hard validity bound for the cached CLI env
|
|
236
227
|
/**
|
|
237
228
|
* How old the cached CLI env may be before `/model` re-probes it live.
|
|
@@ -449,6 +440,40 @@ export class FleetManager {
|
|
|
449
440
|
adapterRestarting = new Set();
|
|
450
441
|
// Adapter isolation: track state per adapter for retry + visibility
|
|
451
442
|
adapterState = new Map();
|
|
443
|
+
/** Web Settings secret rotation is deliberately separate from reconnect
|
|
444
|
+
* recovery: a rotation must build a fresh provider client with the new token,
|
|
445
|
+
* and stale callbacks from the old client must not win. */
|
|
446
|
+
connectionSecretChallenges = new Map();
|
|
447
|
+
connectionSecretChallengesByKey = new Map();
|
|
448
|
+
connectionSecretJobs = new Map();
|
|
449
|
+
connectionSecretJobSession = new Map();
|
|
450
|
+
connectionSecretInFlight = new Map();
|
|
451
|
+
connectionSecretGenerations = new Map();
|
|
452
|
+
/** Local epoch that fences a challenge across adapter replacement and
|
|
453
|
+
* provider reconnect generations. The adapter's own generation can reset
|
|
454
|
+
* when a new adapter object is constructed, so keep an independent epoch. */
|
|
455
|
+
connectionSecretAdapterRefs = new Map();
|
|
456
|
+
connectionSecretHealthGenerations = new Map();
|
|
457
|
+
/** Generic API-key verifier/apply state. The challenge scope contains the
|
|
458
|
+
* resolved spec/env key, so a request can never retarget another provider. */
|
|
459
|
+
providerSecretChallenges = new Map();
|
|
460
|
+
providerSecretChallengesByKey = new Map();
|
|
461
|
+
providerSecretJobs = new Map();
|
|
462
|
+
providerSecretJobSession = new Map();
|
|
463
|
+
providerSecretInFlight = new Map();
|
|
464
|
+
providerSecretGenerations = new Map();
|
|
465
|
+
/** Test seam only; production always uses the fixed HTTPS client. */
|
|
466
|
+
providerSecretHttpClient;
|
|
467
|
+
/** Code-owned activation hooks; never populated from a request. */
|
|
468
|
+
providerSecretReloadHooks = new Map();
|
|
469
|
+
/** In-memory snapshots for narrow hot consumers (currently Groq voice). */
|
|
470
|
+
providerSecretHotSnapshots = new Map();
|
|
471
|
+
/** Web Settings connection-binding step-up challenges and apply jobs. */
|
|
472
|
+
connectionBindingChallenges = new Map();
|
|
473
|
+
connectionBindingChallengesByKey = new Map();
|
|
474
|
+
connectionBindingJobs = new Map();
|
|
475
|
+
connectionBindingJobSession = new Map();
|
|
476
|
+
connectionBindingInFlight = new Map();
|
|
452
477
|
collabInstances = new Set();
|
|
453
478
|
// Health endpoint
|
|
454
479
|
healthServer = null;
|
|
@@ -462,6 +487,11 @@ export class FleetManager {
|
|
|
462
487
|
fullRestartLauncher = launchFullRestartHelper;
|
|
463
488
|
eventLogPruneTimer = null;
|
|
464
489
|
logRotateTimer = null;
|
|
490
|
+
discordPresenceTimer = null;
|
|
491
|
+
discordPresenceEagerTimer = null;
|
|
492
|
+
discordPresenceEagerPending = false;
|
|
493
|
+
discordPresenceInFlight = null;
|
|
494
|
+
static DISCORD_PRESENCE_REFRESH_MS = 15 * 60_000;
|
|
465
495
|
/** Days of event/activity history to keep. */
|
|
466
496
|
static EVENT_LOG_RETENTION_DAYS = 30;
|
|
467
497
|
watchdogTimer = null;
|
|
@@ -472,7 +502,29 @@ export class FleetManager {
|
|
|
472
502
|
mirrorTimer = null;
|
|
473
503
|
// Web UI: SSE clients + auth token
|
|
474
504
|
sseClients = new Set();
|
|
475
|
-
|
|
505
|
+
/**
|
|
506
|
+
* Read from disk on every access rather than cached at startup: `agend
|
|
507
|
+
* web-token rotate` runs in a separate process, and a cached copy would keep
|
|
508
|
+
* authorizing revoked links and cookies until the fleet restarted.
|
|
509
|
+
*/
|
|
510
|
+
get webToken() { return readWebToken(this.dataDir); }
|
|
511
|
+
/**
|
|
512
|
+
* Set while a Settings apply job is driving the reconcile. The reconcile
|
|
513
|
+
* stays the single doer; it just says out loud what it is doing to whom, so
|
|
514
|
+
* the job's rows are the work rather than a prediction of it.
|
|
515
|
+
*/
|
|
516
|
+
applyJobStoreCache = null;
|
|
517
|
+
/** The apply that currently owns the reconcile slot, reserved synchronously
|
|
518
|
+
* so a second request cannot slip in before the first one starts working. */
|
|
519
|
+
activeApplyJobId = null;
|
|
520
|
+
/** The fleet-level signature this process actually came up on. */
|
|
521
|
+
appliedFleetLevel = null;
|
|
522
|
+
/** The config behind that signature, kept so a "needs restart" log can name
|
|
523
|
+
* which keys moved rather than just asserting that something did. */
|
|
524
|
+
startupFleetConfig = null;
|
|
525
|
+
/** Set when the file on disk and the in-memory config disagree on a
|
|
526
|
+
* startup-only key at startup. See checkStartupSignatureConsistency(). */
|
|
527
|
+
fleetSignatureMismatch = null;
|
|
476
528
|
viewToken = null;
|
|
477
529
|
healthServerListening = false;
|
|
478
530
|
constructor(dataDir) {
|
|
@@ -548,26 +600,48 @@ export class FleetManager {
|
|
|
548
600
|
this.scheduleReconcile();
|
|
549
601
|
}
|
|
550
602
|
scheduleReconcile() {
|
|
551
|
-
|
|
603
|
+
const started = this.startExclusiveReconcile();
|
|
604
|
+
if (!started) {
|
|
552
605
|
this.reloadPending = true;
|
|
553
606
|
this.logger.info("Config reconciliation already running — coalesced reload request");
|
|
554
607
|
return;
|
|
555
608
|
}
|
|
556
|
-
|
|
557
|
-
this.reconcileInFlight = this.reconcileInstances()
|
|
558
|
-
.catch(err => {
|
|
609
|
+
started.catch(err => {
|
|
559
610
|
// Almost always a YAML parse error. Log-only meant the user edited
|
|
560
611
|
// fleet.yaml, sent SIGHUP, and got no reaction and no explanation.
|
|
561
612
|
this.logger.error({ err }, "SIGHUP config reload failed");
|
|
562
613
|
const message = err instanceof Error ? err.message : String(err);
|
|
563
614
|
this.notifyFleetError(t("fleet.reload_failed", message));
|
|
564
|
-
})
|
|
615
|
+
});
|
|
616
|
+
}
|
|
617
|
+
/**
|
|
618
|
+
* Take the reconcile slot, or refuse.
|
|
619
|
+
*
|
|
620
|
+
* Only one reconcile may touch lifecycle and config at a time — two of them
|
|
621
|
+
* stop and start the same instance in parallel. SIGHUP and a Settings apply
|
|
622
|
+
* are the same operation from two entrances, so they share the one slot: the
|
|
623
|
+
* signal coalesces into a pending replay, the apply is told the fleet is busy.
|
|
624
|
+
*
|
|
625
|
+
* The returned promise is the caller's to handle; the stored one is already
|
|
626
|
+
* handled, so a rejection never escapes as an unhandled rejection.
|
|
627
|
+
*/
|
|
628
|
+
startExclusiveReconcile(observer) {
|
|
629
|
+
if (this.reconcileInFlight)
|
|
630
|
+
return null;
|
|
631
|
+
this.reloadPending = false;
|
|
632
|
+
let settle;
|
|
633
|
+
const caller = new Promise((resolve, reject) => {
|
|
634
|
+
settle = (err, outcome) => (err ? reject(err instanceof Error ? err : new Error(String(err))) : resolve(outcome ?? {}));
|
|
635
|
+
});
|
|
636
|
+
this.reconcileInFlight = this.reconcileInstances(observer)
|
|
637
|
+
.then(outcome => settle(null, outcome), err => settle(err))
|
|
565
638
|
.finally(() => {
|
|
566
639
|
this.reconcileInFlight = null;
|
|
567
640
|
if (this.reloadPending && this.startupComplete) {
|
|
568
641
|
this.scheduleReconcile();
|
|
569
642
|
}
|
|
570
643
|
});
|
|
644
|
+
return caller;
|
|
571
645
|
}
|
|
572
646
|
/**
|
|
573
647
|
* Is the fleet going down (or coming back up) on purpose?
|
|
@@ -583,6 +657,39 @@ export class FleetManager {
|
|
|
583
657
|
}
|
|
584
658
|
finishStartup() {
|
|
585
659
|
this.startupComplete = true;
|
|
660
|
+
// Resolve whatever a previous run — or a setup host that crashed — left
|
|
661
|
+
// behind. A tunnel nobody is tracking is a public entrance nobody is
|
|
662
|
+
// watching, and the fleet starting is the moment there is finally a process
|
|
663
|
+
// around to notice. Never throws: a lease that cannot be resolved blocks
|
|
664
|
+
// the next tunnel and says so, it does not block the fleet.
|
|
665
|
+
this.announceToolPermissionsChange();
|
|
666
|
+
void reapStaleTunnel(this.dataDir)
|
|
667
|
+
.then(outcome => {
|
|
668
|
+
if (outcome.kind === "manual")
|
|
669
|
+
this.logger.warn({ tunnel: outcome }, manualCleanupMessage(outcome));
|
|
670
|
+
else if (outcome.kind === "reaped")
|
|
671
|
+
this.logger.info({ how: outcome.how, pid: outcome.pid }, "Reaped a leftover tunnel");
|
|
672
|
+
})
|
|
673
|
+
.catch(err => this.logger.warn({ err }, "Tunnel reaper failed"));
|
|
674
|
+
// An existing installation has never written the setup marker — it predates
|
|
675
|
+
// it — so `agend setup` would open a pre-fleet form for a fleet that plainly
|
|
676
|
+
// exists. A fleet that just came up on a config with agents in it is proof
|
|
677
|
+
// enough that setup happened.
|
|
678
|
+
if (Object.keys(this.fleetConfig?.instances ?? {}).length > 0 && !isSetupComplete(this.dataDir)) {
|
|
679
|
+
markSetupComplete(this.dataDir);
|
|
680
|
+
}
|
|
681
|
+
// After slimFleetConfigAtStartup() and the general/topic fixups, all of
|
|
682
|
+
// which may rewrite fleet.yaml — the baseline has to be what this process
|
|
683
|
+
// is actually running, compared against what a reconcile would load.
|
|
684
|
+
this.appliedFleetLevel = this.fleetLevelSignature();
|
|
685
|
+
this.startupFleetConfig = this.fleetConfig ? structuredClone(this.fleetConfig) : null;
|
|
686
|
+
this.checkStartupSignatureConsistency();
|
|
687
|
+
// A job from the process that just died cannot still be running here. The
|
|
688
|
+
// restart applied the saved config to every instance, so its open rows are
|
|
689
|
+
// finished — by the restart, which is what the user needs told.
|
|
690
|
+
for (const settled of this.applyJobs.settleAfterRestart()) {
|
|
691
|
+
this.logger.info({ jobId: settled.id }, "Settings apply job settled by fleet restart");
|
|
692
|
+
}
|
|
586
693
|
// We are the post-update fleet: the update is over by definition. Clearing
|
|
587
694
|
// it here (rather than in the update command, which exits before the new
|
|
588
695
|
// fleet is up) is what keeps the quiet window from outliving the restart.
|
|
@@ -708,11 +815,30 @@ export class FleetManager {
|
|
|
708
815
|
if (this.rawFleetDocument.errors.length > 0) {
|
|
709
816
|
throw new Error(`Invalid fleet.yaml: ${this.rawFleetDocument.errors[0].message}`);
|
|
710
817
|
}
|
|
711
|
-
|
|
712
|
-
|
|
818
|
+
const raw = loadRawFleetConfig(configPath);
|
|
819
|
+
const loaded = loadFleetConfig(configPath);
|
|
820
|
+
this.assertProviderSecretEnvKeys(loaded);
|
|
821
|
+
this.rawFleetConfig = raw;
|
|
822
|
+
this.fleetConfig = loaded;
|
|
713
823
|
this.savedFleetConfigSnapshot = structuredClone(this.fleetConfig);
|
|
714
824
|
return this.fleetConfig;
|
|
715
825
|
}
|
|
826
|
+
/**
|
|
827
|
+
* A channel's configurable bot_token_env is an env-key writer too. Refuse
|
|
828
|
+
* an overlap with a registry API key (or a process-reserved key) before the
|
|
829
|
+
* config becomes live; otherwise a Discord token could be written into
|
|
830
|
+
* GROQ_API_KEY by a perfectly valid-looking rotation request.
|
|
831
|
+
*/
|
|
832
|
+
assertProviderSecretEnvKeys(config) {
|
|
833
|
+
const registryKeys = providerRegistryEnvKeys();
|
|
834
|
+
const channels = config.channels ?? (config.channel ? [config.channel] : []);
|
|
835
|
+
for (const channel of channels) {
|
|
836
|
+
const key = channel.bot_token_env;
|
|
837
|
+
if (registryKeys.has(key) || isReservedProviderEnvKey(key)) {
|
|
838
|
+
throw new Error(`bot_token_env ${key} conflicts with a protected provider secret key`);
|
|
839
|
+
}
|
|
840
|
+
}
|
|
841
|
+
}
|
|
716
842
|
/** User-authored fleet.yaml, before defaults are merged into instances. */
|
|
717
843
|
getRawFleetConfig() {
|
|
718
844
|
return structuredClone(this.rawFleetConfig);
|
|
@@ -1116,31 +1242,56 @@ export class FleetManager {
|
|
|
1116
1242
|
*/
|
|
1117
1243
|
getActiveUsageProviderIds() {
|
|
1118
1244
|
const providers = new Set();
|
|
1119
|
-
|
|
1245
|
+
// Per instance, not per backend: two agents on two kiro subscriptions are
|
|
1246
|
+
// two rows, and filtering by the bare backend id would hide both.
|
|
1247
|
+
for (const [name, backend, profile] of this.activeBackendBindings()) {
|
|
1248
|
+
void name;
|
|
1120
1249
|
const provider = usageProviderIdForBackend(backend);
|
|
1121
|
-
if (provider)
|
|
1122
|
-
|
|
1250
|
+
if (!provider)
|
|
1251
|
+
continue;
|
|
1252
|
+
providers.add(profile ? `${provider}:${profile}` : provider);
|
|
1123
1253
|
}
|
|
1124
1254
|
return providers;
|
|
1125
1255
|
}
|
|
1126
|
-
/**
|
|
1127
|
-
|
|
1128
|
-
const
|
|
1129
|
-
const
|
|
1256
|
+
/** Subscription providers used by the running/paused instances owned by one adapter. */
|
|
1257
|
+
getUsageProviderIdsForAdapter(adapterId) {
|
|
1258
|
+
const providers = new Set();
|
|
1259
|
+
for (const [name, backend, profile] of this.activeBackendBindings()) {
|
|
1260
|
+
if (this.getInstanceAdapterId(name) !== adapterId)
|
|
1261
|
+
continue;
|
|
1262
|
+
const provider = usageProviderIdForBackend(backend);
|
|
1263
|
+
if (!provider)
|
|
1264
|
+
continue;
|
|
1265
|
+
providers.add(profile ? `${provider}:${profile}` : provider);
|
|
1266
|
+
}
|
|
1267
|
+
return providers;
|
|
1268
|
+
}
|
|
1269
|
+
/** `[instance, effective backend, credential profile]` for everything that is
|
|
1270
|
+
* running or paused — the one place both usage views agree on who is live. */
|
|
1271
|
+
activeBackendBindings() {
|
|
1272
|
+
const bindings = [];
|
|
1273
|
+
const add = (name, backend, profile) => {
|
|
1130
1274
|
const status = this.getInstanceStatus(name);
|
|
1131
1275
|
if (status !== "running" && status !== "paused")
|
|
1132
1276
|
return;
|
|
1133
1277
|
if (backend)
|
|
1134
|
-
|
|
1278
|
+
bindings.push([name, backend, profile]);
|
|
1135
1279
|
};
|
|
1136
1280
|
for (const [name, config] of Object.entries(this.fleetConfig?.instances ?? {})) {
|
|
1137
1281
|
// loadFleetConfig() has already merged the fleet default into each row.
|
|
1138
|
-
|
|
1282
|
+
const backend = config.backend ?? this.fleetConfig?.defaults?.backend ?? "claude-code";
|
|
1283
|
+
add(name, backend, instanceCredentialProfile(config, this.fleetConfig?.defaults, backend));
|
|
1139
1284
|
}
|
|
1140
1285
|
for (const channel of this.classicChannels?.getAll() ?? []) {
|
|
1141
|
-
|
|
1286
|
+
const backend = this.classicChannels?.getBackendByInstance(channel.instanceName, this.fleetConfig?.defaults?.backend);
|
|
1287
|
+
// Classic channels carry no backend_options, so they run the shared login.
|
|
1288
|
+
add(channel.instanceName, backend, null);
|
|
1142
1289
|
}
|
|
1143
|
-
return
|
|
1290
|
+
return bindings;
|
|
1291
|
+
}
|
|
1292
|
+
/** Effective backends with a running or persisted-paused fleet/Classic instance. */
|
|
1293
|
+
getActiveBackendIds() {
|
|
1294
|
+
return new Set(this.activeBackendBindings().map(([, backend]) => backend));
|
|
1144
1295
|
}
|
|
1145
1296
|
isClassicInstance(name) {
|
|
1146
1297
|
return this.classicChannels?.getAll().some(channel => channel.instanceName === name) ?? false;
|
|
@@ -1548,7 +1699,7 @@ export class FleetManager {
|
|
|
1548
1699
|
if (!wasRunning)
|
|
1549
1700
|
return;
|
|
1550
1701
|
const hotOnly = changedFields.length > 0
|
|
1551
|
-
&& changedFields.every(field => field
|
|
1702
|
+
&& changedFields.every(field => CLASSIC_HOT_CONFIG_KEYS.has(field));
|
|
1552
1703
|
if (hotOnly) {
|
|
1553
1704
|
this.applyHotConfigUpdate(instanceName, this.classicBehaviorUpdate(instanceName));
|
|
1554
1705
|
this.logger.info({ instanceName, fields: changedFields }, "Classic instance hot config reloaded");
|
|
@@ -1690,6 +1841,7 @@ export class FleetManager {
|
|
|
1690
1841
|
catch { /* consumed or absent */ }
|
|
1691
1842
|
// Auto-connect IPC — daemon.start() ensures socket is ready before resolving
|
|
1692
1843
|
await this.connectIpcToInstance(name);
|
|
1844
|
+
this.requestDiscordUsagePresenceRefresh();
|
|
1693
1845
|
}
|
|
1694
1846
|
/** Recreate a daemon for a marker-only paused instance after an explicit wake/delivery. */
|
|
1695
1847
|
async startPersistedPausedInstance(name) {
|
|
@@ -1945,6 +2097,22 @@ export class FleetManager {
|
|
|
1945
2097
|
names.add(channel.instanceName);
|
|
1946
2098
|
return [...names];
|
|
1947
2099
|
}
|
|
2100
|
+
/**
|
|
2101
|
+
* Probe the same executable set exposed by the web backend catalog.
|
|
2102
|
+
*
|
|
2103
|
+
* This deliberately has no cache: an `/install-cli` completion can add a
|
|
2104
|
+
* binary to PATH while the fleet process remains alive, and the next bare
|
|
2105
|
+
* `/login` must see it without requiring a restart or an explicit cache
|
|
2106
|
+
* invalidation call.
|
|
2107
|
+
*/
|
|
2108
|
+
probeInstalledBackends() {
|
|
2109
|
+
const installed = new Set();
|
|
2110
|
+
for (const [backend, info] of Object.entries(BACKEND_INSTALLATION_INFO)) {
|
|
2111
|
+
if (checkBinaryInstalled(info.binary))
|
|
2112
|
+
installed.add(backend);
|
|
2113
|
+
}
|
|
2114
|
+
return installed;
|
|
2115
|
+
}
|
|
1948
2116
|
/**
|
|
1949
2117
|
* One fleet-level notice per burst, not one per instance: a post-update herd
|
|
1950
2118
|
* fails many instances within the same second. Two notices per incident at
|
|
@@ -2174,7 +2342,9 @@ export class FleetManager {
|
|
|
2174
2342
|
}
|
|
2175
2343
|
/** Initialize auth before any adapter can answer /dashboard. */
|
|
2176
2344
|
initializeWebAuthTokens() {
|
|
2177
|
-
|
|
2345
|
+
// Creates web.token if absent; the value is then read back per request by
|
|
2346
|
+
// the `webToken` getter, so nothing is cached here.
|
|
2347
|
+
loadOrCreateWebToken(this.dataDir);
|
|
2178
2348
|
this.viewToken = randomBytes(24).toString("hex");
|
|
2179
2349
|
const viewTokenPath = join(this.dataDir, "view.token");
|
|
2180
2350
|
writeFileSync(viewTokenPath, this.viewToken, { encoding: "utf8", mode: 0o600 });
|
|
@@ -2738,6 +2908,7 @@ export class FleetManager {
|
|
|
2738
2908
|
}
|
|
2739
2909
|
if (topicMode && (fleet.channel || fleet.channels?.length)) {
|
|
2740
2910
|
await adapterStartup;
|
|
2911
|
+
this.startDiscordUsagePresence();
|
|
2741
2912
|
// Bind every fleet instance deterministically. Explicit channel_id wins;
|
|
2742
2913
|
// otherwise channels[0] is authoritative. Do not infer identity from
|
|
2743
2914
|
// concurrent adapter startup or whichever bot receives a message first.
|
|
@@ -2886,6 +3057,78 @@ export class FleetManager {
|
|
|
2886
3057
|
this.logger.warn({ health }, "Fleet started with problems — see /health");
|
|
2887
3058
|
}
|
|
2888
3059
|
}
|
|
3060
|
+
/** Keep Discord profile activity aligned with the same cached usage source as /usage. */
|
|
3061
|
+
startDiscordUsagePresence() {
|
|
3062
|
+
if (this.discordPresenceTimer)
|
|
3063
|
+
clearInterval(this.discordPresenceTimer);
|
|
3064
|
+
if (this.discordPresenceEagerTimer) {
|
|
3065
|
+
clearTimeout(this.discordPresenceEagerTimer);
|
|
3066
|
+
this.discordPresenceEagerTimer = null;
|
|
3067
|
+
}
|
|
3068
|
+
this.discordPresenceEagerPending = false;
|
|
3069
|
+
void this.refreshDiscordUsagePresence();
|
|
3070
|
+
this.discordPresenceTimer = setInterval(() => {
|
|
3071
|
+
void this.refreshDiscordUsagePresence();
|
|
3072
|
+
}, FleetManager.DISCORD_PRESENCE_REFRESH_MS);
|
|
3073
|
+
this.discordPresenceTimer.unref?.();
|
|
3074
|
+
}
|
|
3075
|
+
/**
|
|
3076
|
+
* Request one prompt presence refresh after an instance/adapter comes online.
|
|
3077
|
+
* Startup can bring several instances up together; a short coalescing window
|
|
3078
|
+
* keeps that herd on the existing shared refresh path and one usage fetch.
|
|
3079
|
+
*/
|
|
3080
|
+
requestDiscordUsagePresenceRefresh() {
|
|
3081
|
+
this.discordPresenceEagerPending = true;
|
|
3082
|
+
// During early fleet startup the interval is not installed yet; its first
|
|
3083
|
+
// refresh in startDiscordUsagePresence already covers all pending starts.
|
|
3084
|
+
if (!this.discordPresenceTimer || this.discordPresenceEagerTimer)
|
|
3085
|
+
return;
|
|
3086
|
+
this.discordPresenceEagerTimer = setTimeout(() => {
|
|
3087
|
+
this.discordPresenceEagerTimer = null;
|
|
3088
|
+
if (!this.discordPresenceEagerPending || !this.discordPresenceTimer)
|
|
3089
|
+
return;
|
|
3090
|
+
this.discordPresenceEagerPending = false;
|
|
3091
|
+
void this.refreshDiscordUsagePresence();
|
|
3092
|
+
}, 50);
|
|
3093
|
+
this.discordPresenceEagerTimer.unref?.();
|
|
3094
|
+
}
|
|
3095
|
+
refreshDiscordUsagePresence() {
|
|
3096
|
+
if (this.discordPresenceInFlight)
|
|
3097
|
+
return this.discordPresenceInFlight;
|
|
3098
|
+
const run = (async () => {
|
|
3099
|
+
const targets = [...this.adapters.values()]
|
|
3100
|
+
.filter(adapter => adapter.type === "discord" && typeof adapter.setActivity === "function");
|
|
3101
|
+
if (targets.length === 0)
|
|
3102
|
+
return;
|
|
3103
|
+
try {
|
|
3104
|
+
// Fetch the shared snapshot once, then scope the projection to each
|
|
3105
|
+
// adapter's own fleet/Classic instances. Passing the fleet-wide active
|
|
3106
|
+
// set here would make every bot advertise providers owned by a sibling
|
|
3107
|
+
// bot (notably ClassicBot's Grok/Antigravity rows).
|
|
3108
|
+
const payload = await getUsageSnapshot(false);
|
|
3109
|
+
for (const adapter of targets) {
|
|
3110
|
+
try {
|
|
3111
|
+
const scoped = filterUsageProviders(payload, this.getUsageProviderIdsForAdapter(adapter.id));
|
|
3112
|
+
adapter.setActivity?.(formatDiscordUsageActivity(scoped));
|
|
3113
|
+
}
|
|
3114
|
+
catch {
|
|
3115
|
+
// Presence is cosmetic; a failed update must not affect delivery.
|
|
3116
|
+
}
|
|
3117
|
+
}
|
|
3118
|
+
}
|
|
3119
|
+
catch {
|
|
3120
|
+
// Usage providers are best-effort and may be offline. Keep the last
|
|
3121
|
+
// activity rather than replacing it with an untruthful blank state.
|
|
3122
|
+
this.logger.debug("Discord usage presence refresh skipped");
|
|
3123
|
+
}
|
|
3124
|
+
})();
|
|
3125
|
+
const done = run.finally(() => {
|
|
3126
|
+
if (this.discordPresenceInFlight === done)
|
|
3127
|
+
this.discordPresenceInFlight = null;
|
|
3128
|
+
});
|
|
3129
|
+
this.discordPresenceInFlight = done;
|
|
3130
|
+
return done;
|
|
3131
|
+
}
|
|
2889
3132
|
/**
|
|
2890
3133
|
* Delete inbox files older than retentionDays (by mtime). Cleans the shared
|
|
2891
3134
|
* inbox (`<dataDir>/inbox`) and every workspace inbox
|
|
@@ -3003,6 +3246,11 @@ export class FleetManager {
|
|
|
3003
3246
|
}
|
|
3004
3247
|
bindAdapterHealth(adapter, adapterId) {
|
|
3005
3248
|
adapter.on("gateway_health", (snapshot) => {
|
|
3249
|
+
// A token rotation tears down the old EventEmitter before constructing
|
|
3250
|
+
// the replacement. A late health frame from that old client must never
|
|
3251
|
+
// make a failed/new-generation adapter look connected.
|
|
3252
|
+
if (this.adapters.get(adapterId) !== adapter)
|
|
3253
|
+
return;
|
|
3006
3254
|
const previous = this.adapterState.get(adapterId);
|
|
3007
3255
|
const status = snapshot.status === "connected" ? "connected"
|
|
3008
3256
|
: snapshot.status === "stopped" ? "failed"
|
|
@@ -3083,7 +3331,7 @@ export class FleetManager {
|
|
|
3083
3331
|
};
|
|
3084
3332
|
}
|
|
3085
3333
|
/** Start the primary adapter (backward-compatible, sets this.adapter) */
|
|
3086
|
-
async startSingleAdapter(fleet, channelConfig) {
|
|
3334
|
+
async startSingleAdapter(fleet, channelConfig, onStarted) {
|
|
3087
3335
|
const botToken = process.env[channelConfig.bot_token_env];
|
|
3088
3336
|
if (!botToken) {
|
|
3089
3337
|
this.logger.warn({ env: channelConfig.bot_token_env }, "Bot token env not set, skipping shared adapter");
|
|
@@ -3091,7 +3339,9 @@ export class FleetManager {
|
|
|
3091
3339
|
}
|
|
3092
3340
|
const accessDir = join(this.dataDir, "access");
|
|
3093
3341
|
mkdirSync(accessDir, { recursive: true });
|
|
3094
|
-
const
|
|
3342
|
+
const accessStatePath = join(accessDir, "access.json");
|
|
3343
|
+
const accessManager = new AccessManager(channelConfig.access ?? DEFAULT_OPEN_ACCESS, accessStatePath);
|
|
3344
|
+
this.warnIfAccessModeOverridden(accessManager, accessStatePath);
|
|
3095
3345
|
this.accessManager = accessManager;
|
|
3096
3346
|
const inboxDir = join(this.dataDir, "inbox");
|
|
3097
3347
|
mkdirSync(inboxDir, { recursive: true });
|
|
@@ -3107,13 +3357,20 @@ export class FleetManager {
|
|
|
3107
3357
|
this.worlds.set(adapterId, world);
|
|
3108
3358
|
this.adapters.set(adapterId, adapter);
|
|
3109
3359
|
this.bindAdapterHealth(adapter, adapterId);
|
|
3360
|
+
const isCurrentAdapter = () => this.adapters.get(adapterId) === adapter;
|
|
3110
3361
|
this.adapter.on("message", safeHandler(async (msg) => {
|
|
3362
|
+
if (!isCurrentAdapter())
|
|
3363
|
+
return;
|
|
3111
3364
|
await this.handleInboundMessage(msg);
|
|
3112
3365
|
}, this.logger, "adapter.message"));
|
|
3113
3366
|
this.adapter.on("reaction", safeHandler(async (r) => {
|
|
3367
|
+
if (!isCurrentAdapter())
|
|
3368
|
+
return;
|
|
3114
3369
|
await this.handleInboundReaction(r);
|
|
3115
3370
|
}, this.logger, "adapter.reaction"));
|
|
3116
3371
|
this.adapter.on("callback_query", safeHandler(async (data) => {
|
|
3372
|
+
if (!isCurrentAdapter())
|
|
3373
|
+
return;
|
|
3117
3374
|
if (await this.handleTipDismiss(data, adapterId, this.adapter ?? undefined))
|
|
3118
3375
|
return;
|
|
3119
3376
|
if (await this.handleTipUnlock(data, adapterId, this.adapter ?? undefined))
|
|
@@ -3154,6 +3411,8 @@ export class FleetManager {
|
|
|
3154
3411
|
this.bindTopicClosedHandler(adapter, adapterId, "adapter.topic_closed");
|
|
3155
3412
|
// Handle classic bot slash commands (/start, /stop, /chat, /compact, /save, /load)
|
|
3156
3413
|
this.adapter.on("slash_command", safeHandler(async (data) => {
|
|
3414
|
+
if (!isCurrentAdapter())
|
|
3415
|
+
return;
|
|
3157
3416
|
if (data.command === "start") {
|
|
3158
3417
|
await this.handleClassicStartSlash(data, adapterId);
|
|
3159
3418
|
}
|
|
@@ -3433,6 +3692,8 @@ export class FleetManager {
|
|
|
3433
3692
|
// Non-blocking: /model & status views read the cache; never delays startup.
|
|
3434
3693
|
this.probeCliEnvs();
|
|
3435
3694
|
this.adapter.on("started", safeHandler((username, userId) => {
|
|
3695
|
+
if (!isCurrentAdapter())
|
|
3696
|
+
return;
|
|
3436
3697
|
this.logger.info(`Bot @${username} polling started. Ensure no other service is polling this bot token.`);
|
|
3437
3698
|
// Concurrent startup can insert a secondary world first. Update the
|
|
3438
3699
|
// configured primary world, not Map insertion order.
|
|
@@ -3444,6 +3705,7 @@ export class FleetManager {
|
|
|
3444
3705
|
}
|
|
3445
3706
|
if (userId)
|
|
3446
3707
|
this.botUserId = userId;
|
|
3708
|
+
onStarted?.();
|
|
3447
3709
|
}, this.logger, "adapter.started"));
|
|
3448
3710
|
this.adapter.on("polling_conflict", safeHandler(({ attempt, delay }) => {
|
|
3449
3711
|
this.logger.warn(`409 Conflict (attempt ${attempt}), retry in ${delay / 1000}s`);
|
|
@@ -3452,10 +3714,14 @@ export class FleetManager {
|
|
|
3452
3714
|
this.logger.warn({ err: err instanceof Error ? err.message : String(err) }, "Adapter handler error");
|
|
3453
3715
|
}, this.logger, "adapter.handler_error"));
|
|
3454
3716
|
this.adapter.on("error", (err) => {
|
|
3717
|
+
if (!isCurrentAdapter())
|
|
3718
|
+
return;
|
|
3455
3719
|
this.logger.error({ err }, "Primary adapter fatal error");
|
|
3456
3720
|
this.restartAdapter(this.adapter, adapterId).catch(() => { });
|
|
3457
3721
|
});
|
|
3458
3722
|
this.adapter.on("new_group_detected", safeHandler(async (data) => {
|
|
3723
|
+
if (!isCurrentAdapter())
|
|
3724
|
+
return;
|
|
3459
3725
|
const adminMsg = t("alert.bot_added", data.groupTitle, data.groupId, data.source);
|
|
3460
3726
|
const generalId = this.findGeneralInstance();
|
|
3461
3727
|
// No user to promote: the bot was just added, nobody has run /start yet.
|
|
@@ -3467,6 +3733,9 @@ export class FleetManager {
|
|
|
3467
3733
|
if (fleet.channel?.group_id) {
|
|
3468
3734
|
this.adapter.setChatId(String(fleet.channel.group_id));
|
|
3469
3735
|
}
|
|
3736
|
+
if (this.discordPresenceTimer && this.adapter.type === "discord") {
|
|
3737
|
+
this.requestDiscordUsagePresenceRefresh();
|
|
3738
|
+
}
|
|
3470
3739
|
this.startTopicCleanupPoller();
|
|
3471
3740
|
// Prune stale external sessions every 5 minutes
|
|
3472
3741
|
this.sessionPruneTimer = setInterval(() => {
|
|
@@ -3474,7 +3743,7 @@ export class FleetManager {
|
|
|
3474
3743
|
}, 5 * 60 * 1000);
|
|
3475
3744
|
}
|
|
3476
3745
|
/** Start an additional (non-primary) adapter */
|
|
3477
|
-
async startAdditionalAdapter(channelConfig, registerCommands = true) {
|
|
3746
|
+
async startAdditionalAdapter(channelConfig, registerCommands = true, onStarted) {
|
|
3478
3747
|
const adapterId = channelConfig.id ?? channelConfig.type;
|
|
3479
3748
|
const botToken = process.env[channelConfig.bot_token_env];
|
|
3480
3749
|
if (!botToken) {
|
|
@@ -3483,7 +3752,9 @@ export class FleetManager {
|
|
|
3483
3752
|
}
|
|
3484
3753
|
const accessDir = join(this.dataDir, "access");
|
|
3485
3754
|
mkdirSync(accessDir, { recursive: true });
|
|
3486
|
-
const
|
|
3755
|
+
const accessStatePath = join(accessDir, `access-${adapterId}.json`);
|
|
3756
|
+
const accessManager = new AccessManager(channelConfig.access ?? DEFAULT_OPEN_ACCESS, accessStatePath);
|
|
3757
|
+
this.warnIfAccessModeOverridden(accessManager, accessStatePath);
|
|
3487
3758
|
const inboxDir = join(this.dataDir, "inbox");
|
|
3488
3759
|
mkdirSync(inboxDir, { recursive: true });
|
|
3489
3760
|
const adapter = await createAdapter(channelConfig, {
|
|
@@ -3497,14 +3768,21 @@ export class FleetManager {
|
|
|
3497
3768
|
this.worlds.set(adapterId, world);
|
|
3498
3769
|
this.adapters.set(adapterId, adapter);
|
|
3499
3770
|
this.bindAdapterHealth(adapter, adapterId);
|
|
3771
|
+
const isCurrentAdapter = () => this.adapters.get(adapterId) === adapter;
|
|
3500
3772
|
// Wire up event handlers (same as primary, routes through shared handleInboundMessage)
|
|
3501
3773
|
adapter.on("message", safeHandler(async (msg) => {
|
|
3774
|
+
if (!isCurrentAdapter())
|
|
3775
|
+
return;
|
|
3502
3776
|
await this.handleInboundMessage(msg);
|
|
3503
3777
|
}, this.logger, `adapter[${adapterId}].message`));
|
|
3504
3778
|
adapter.on("reaction", safeHandler(async (r) => {
|
|
3779
|
+
if (!isCurrentAdapter())
|
|
3780
|
+
return;
|
|
3505
3781
|
await this.handleInboundReaction(r);
|
|
3506
3782
|
}, this.logger, `adapter[${adapterId}].reaction`));
|
|
3507
3783
|
adapter.on("callback_query", safeHandler(async (data) => {
|
|
3784
|
+
if (!isCurrentAdapter())
|
|
3785
|
+
return;
|
|
3508
3786
|
if (await this.handleTipDismiss(data, adapterId, adapter))
|
|
3509
3787
|
return;
|
|
3510
3788
|
if (await this.handleTipUnlock(data, adapterId, adapter))
|
|
@@ -3545,6 +3823,8 @@ export class FleetManager {
|
|
|
3545
3823
|
this.bindTopicClosedHandler(adapter, adapterId, `adapter[${adapterId}].topic_closed`);
|
|
3546
3824
|
// Slash commands: classic bot + admin commands
|
|
3547
3825
|
adapter.on("slash_command", safeHandler(async (data) => {
|
|
3826
|
+
if (!isCurrentAdapter())
|
|
3827
|
+
return;
|
|
3548
3828
|
if (data.command === "start") {
|
|
3549
3829
|
await this.handleClassicStartSlash(data, adapterId);
|
|
3550
3830
|
}
|
|
@@ -3768,6 +4048,8 @@ export class FleetManager {
|
|
|
3768
4048
|
}
|
|
3769
4049
|
}, this.logger, `adapter[${adapterId}].slash_command`));
|
|
3770
4050
|
adapter.on("started", safeHandler((username, userId) => {
|
|
4051
|
+
if (!isCurrentAdapter())
|
|
4052
|
+
return;
|
|
3771
4053
|
this.logger.info(`[${adapterId}] Bot @${username} polling started.`);
|
|
3772
4054
|
const world = this.worlds.get(adapterId);
|
|
3773
4055
|
if (world) {
|
|
@@ -3775,14 +4057,19 @@ export class FleetManager {
|
|
|
3775
4057
|
if (userId)
|
|
3776
4058
|
world.botUserId = userId;
|
|
3777
4059
|
}
|
|
4060
|
+
onStarted?.();
|
|
3778
4061
|
}, this.logger, `adapter[${adapterId}].started`));
|
|
3779
4062
|
adapter.on("new_group_detected", safeHandler(async (data) => {
|
|
4063
|
+
if (!isCurrentAdapter())
|
|
4064
|
+
return;
|
|
3780
4065
|
const adminMsg = t("alert.bot_added", data.groupTitle, data.groupId, data.source);
|
|
3781
4066
|
const generalId = this.findGeneralInstance(adapterId);
|
|
3782
4067
|
if (generalId)
|
|
3783
4068
|
await this.promptClassicApproval({ generalName: generalId, message: adminMsg, groupId: data.groupId, scope: data.source === "telegram" ? "group" : "guild" });
|
|
3784
4069
|
}, this.logger, `adapter[${adapterId}].new_group_detected`));
|
|
3785
4070
|
adapter.on("error", (err) => {
|
|
4071
|
+
if (!isCurrentAdapter())
|
|
4072
|
+
return;
|
|
3786
4073
|
this.logger.error({ err, adapterId }, "Additional adapter fatal error");
|
|
3787
4074
|
this.restartAdapter(adapter, adapterId).catch(() => { });
|
|
3788
4075
|
});
|
|
@@ -3791,6 +4078,9 @@ export class FleetManager {
|
|
|
3791
4078
|
if (channelConfig.group_id) {
|
|
3792
4079
|
adapter.setChatId(String(channelConfig.group_id));
|
|
3793
4080
|
}
|
|
4081
|
+
if (this.discordPresenceTimer && adapter.type === "discord") {
|
|
4082
|
+
this.requestDiscordUsagePresenceRefresh();
|
|
4083
|
+
}
|
|
3794
4084
|
this.logger.info({ adapterId, type: channelConfig.type }, "Additional adapter started");
|
|
3795
4085
|
}
|
|
3796
4086
|
/** Connect IPC to a single instance with all handlers */
|
|
@@ -3861,22 +4151,15 @@ export class FleetManager {
|
|
|
3861
4151
|
}
|
|
3862
4152
|
await this.handleOutboundFromInstance(name, msg);
|
|
3863
4153
|
}
|
|
3864
|
-
else if (msg.type
|
|
3865
|
-
|
|
3866
|
-
|
|
3867
|
-
|
|
3868
|
-
|
|
3869
|
-
|
|
3870
|
-
|
|
3871
|
-
|
|
3872
|
-
|
|
3873
|
-
this.handleTaskCrud(name, msg);
|
|
3874
|
-
}
|
|
3875
|
-
else if (msg.type === "fleet_set_display_name") {
|
|
3876
|
-
this.handleSetDisplayName(name, msg);
|
|
3877
|
-
}
|
|
3878
|
-
else if (msg.type === "fleet_set_description") {
|
|
3879
|
-
this.handleSetDescription(name, msg);
|
|
4154
|
+
else if (toolForIpcType(msg.type) !== null) {
|
|
4155
|
+
// Sink 2 of 3. Each of these types IS a tool and reaches its own
|
|
4156
|
+
// handler without passing through the outbound path — which is how
|
|
4157
|
+
// `update_decision`, a coordinator-only tool, kept a way through.
|
|
4158
|
+
const typed = this.checkToolPermission("ipc-typed", name, toolForIpcType(msg.type));
|
|
4159
|
+
if (typed.allowed)
|
|
4160
|
+
this.dispatchTypedIpc(name, msg);
|
|
4161
|
+
else
|
|
4162
|
+
this.refuseTypedIpc(name, msg, typed.message);
|
|
3880
4163
|
}
|
|
3881
4164
|
else if (msg.type === "instance_process_state") {
|
|
3882
4165
|
this.cacheInstanceProcessStatus(name, msg.status);
|
|
@@ -4040,6 +4323,9 @@ export class FleetManager {
|
|
|
4040
4323
|
// watchdog/manual/error triggers must converge here instead of stop/start.
|
|
4041
4324
|
await adapter.reconnectGateway(previous?.lastError ?? "fleet adapter restart");
|
|
4042
4325
|
this.adapterState.set(id, { status: "connected", retryCount: 0 });
|
|
4326
|
+
if (this.discordPresenceTimer && adapter.type === "discord") {
|
|
4327
|
+
this.requestDiscordUsagePresenceRefresh();
|
|
4328
|
+
}
|
|
4043
4329
|
this.logger.info({ id }, "Adapter gateway rebuilt successfully");
|
|
4044
4330
|
}
|
|
4045
4331
|
catch (err) {
|
|
@@ -4065,6 +4351,9 @@ export class FleetManager {
|
|
|
4065
4351
|
await adapter.start();
|
|
4066
4352
|
this.logger.info({ id, attempt }, "Adapter restarted successfully");
|
|
4067
4353
|
this.adapterState.set(id, { status: "connected", retryCount: 0 });
|
|
4354
|
+
if (this.discordPresenceTimer && adapter.type === "discord") {
|
|
4355
|
+
this.requestDiscordUsagePresenceRefresh();
|
|
4356
|
+
}
|
|
4068
4357
|
return;
|
|
4069
4358
|
}
|
|
4070
4359
|
catch (err) {
|
|
@@ -4245,13 +4534,36 @@ export class FleetManager {
|
|
|
4245
4534
|
const target = this.routing.resolve(threadId);
|
|
4246
4535
|
if (!target)
|
|
4247
4536
|
return "fleet topic: no instance routed for this thread";
|
|
4248
|
-
// Fleet topic:
|
|
4537
|
+
// Fleet topic: this gate lets the copy through when the adapter is open OR
|
|
4538
|
+
// collab is on for the instance. "Through" is not "delivered" — access
|
|
4539
|
+
// control runs next, and the two arms are not symmetric there:
|
|
4540
|
+
//
|
|
4541
|
+
// open adapter → isAllowed() returns true outright: really admitted.
|
|
4542
|
+
// locked + collab → the bot's user id still has to be on the allowlist,
|
|
4543
|
+
// so the message is normally refused a few lines later
|
|
4544
|
+
// under "Access DENIED for non-allowed user".
|
|
4545
|
+
//
|
|
4546
|
+
// Collab alone therefore does not admit a bot on a locked adapter; it only
|
|
4547
|
+
// declines to drop the copy here and leaves the decision to access control.
|
|
4249
4548
|
const isOpen = this.getChannelConfig(msg.adapterId)?.access?.mode === "open";
|
|
4250
4549
|
if (!isOpen && !this.collabInstances.has(target.name)) {
|
|
4251
4550
|
return `fleet topic: adapter not open and collab off for ${target.name}`;
|
|
4252
4551
|
}
|
|
4253
4552
|
return null;
|
|
4254
4553
|
}
|
|
4554
|
+
/**
|
|
4555
|
+
* Say so when an adapter's access mode is coming from its state file rather
|
|
4556
|
+
* than from fleet.yaml. The state file wins by design — a pairing done at
|
|
4557
|
+
* runtime has to survive a restart — but that also means an edit to
|
|
4558
|
+
* `access.mode` silently does nothing, which is indistinguishable from the
|
|
4559
|
+
* fleet not having reloaded. Naming the file is what makes it fixable.
|
|
4560
|
+
*/
|
|
4561
|
+
warnIfAccessModeOverridden(accessManager, statePath) {
|
|
4562
|
+
const configured = accessManager.overriddenConfigMode();
|
|
4563
|
+
if (!configured)
|
|
4564
|
+
return;
|
|
4565
|
+
this.logger.warn({ statePath, configured, inEffect: accessManager.getMode() }, `access mode "${accessManager.getMode()}" comes from the state file and overrides "${configured}" in fleet.yaml; delete ${statePath} to go back to the configured value`);
|
|
4566
|
+
}
|
|
4255
4567
|
async handleInboundMessage(msg) {
|
|
4256
4568
|
const threadId = msg.threadId || undefined;
|
|
4257
4569
|
this.logger.debug({ source: msg.source, chatId: msg.chatId, threadId, userId: msg.userId, isBotMessage: msg.isBotMessage, textLen: (msg.text ?? "").length, text: (msg.text ?? "").slice(0, 80) }, "handleInboundMessage entry");
|
|
@@ -4818,6 +5130,55 @@ export class FleetManager {
|
|
|
4818
5130
|
}
|
|
4819
5131
|
}
|
|
4820
5132
|
/** Handle outbound tool calls from a daemon instance */
|
|
5133
|
+
/**
|
|
5134
|
+
* Would this instance be allowed to use this tool?
|
|
5135
|
+
*
|
|
5136
|
+
* Stage 1 asks and records; nothing is refused yet. The recording is not only
|
|
5137
|
+
* an observation window: `logActivity("tool_call")` sits on the outbound path
|
|
5138
|
+
* only, so today the fleet has no idea what the agent endpoint or the typed
|
|
5139
|
+
* IPC handlers are being asked to do — the one face with no authorization is
|
|
5140
|
+
* also the one face with no telemetry.
|
|
5141
|
+
*
|
|
5142
|
+
* The profile comes from the instance the socket belongs to. Deliberately not
|
|
5143
|
+
* `senderSessionName`, which arrives inside the message: a caller that fills
|
|
5144
|
+
* in its own identity has not been identified.
|
|
5145
|
+
*/
|
|
5146
|
+
/**
|
|
5147
|
+
* Say once, at startup, what the new default means for this fleet.
|
|
5148
|
+
*
|
|
5149
|
+
* Nothing is rewritten: an explicit `tool_set: full` is a choice somebody
|
|
5150
|
+
* made, and marking an instance as a coordinator is a judgement about how
|
|
5151
|
+
* their fleet is organised. Both stay theirs — this only makes sure they are
|
|
5152
|
+
* not discovered by an agent failing at three in the morning.
|
|
5153
|
+
*/
|
|
5154
|
+
announceToolPermissionsChange() {
|
|
5155
|
+
try {
|
|
5156
|
+
const thirtyDaysAgo = new Date(Date.now() - 30 * 864e5).toISOString().slice(0, 19).replace("T", " ");
|
|
5157
|
+
const recent = [...(this.eventLog?.toolUseByInstance(thirtyDaysAgo) ?? new Map())]
|
|
5158
|
+
.map(([instance, tools]) => ({ instance, tools }));
|
|
5159
|
+
const notice = buildToolPermissionsNotice({
|
|
5160
|
+
defaultsToolSet: this.fleetConfig?.defaults?.tool_set,
|
|
5161
|
+
instances: (this.fleetConfig?.instances ?? {}),
|
|
5162
|
+
recent,
|
|
5163
|
+
});
|
|
5164
|
+
if (notice)
|
|
5165
|
+
this.logger.warn({ notice }, notice);
|
|
5166
|
+
}
|
|
5167
|
+
catch (err) {
|
|
5168
|
+
// Advice is not worth failing a startup over.
|
|
5169
|
+
this.logger.debug({ err }, "tool-permissions: could not build the migration notice");
|
|
5170
|
+
}
|
|
5171
|
+
}
|
|
5172
|
+
checkToolPermission(sink, instanceName, tool) {
|
|
5173
|
+
const profile = resolveToolSet(this.fleetConfig?.instances[instanceName], instanceName);
|
|
5174
|
+
const allowed = mayUseTool(profile, tool);
|
|
5175
|
+
if (!allowed) {
|
|
5176
|
+
this.logger.warn({ sink, instance: instanceName, profile, tool, enforced: true }, "tool-permissions: refused");
|
|
5177
|
+
return { allowed: false, message: toolRefusedMessage(profile, tool) };
|
|
5178
|
+
}
|
|
5179
|
+
this.logger.debug({ sink, instance: instanceName, profile, tool, enforced: true }, "tool-permissions: allowed");
|
|
5180
|
+
return { allowed: true, message: "" };
|
|
5181
|
+
}
|
|
4821
5182
|
async handleOutboundFromInstance(instanceName, msg) {
|
|
4822
5183
|
this.touchActivity(instanceName);
|
|
4823
5184
|
this.setTopicIcon(instanceName, "green");
|
|
@@ -4839,6 +5200,19 @@ export class FleetManager {
|
|
|
4839
5200
|
this.logger.warn({ instanceName, tool, requestId, fleetRequestId, error }, "Fleet outbound result could not be returned — instance IPC is disconnected");
|
|
4840
5201
|
}
|
|
4841
5202
|
};
|
|
5203
|
+
// Sink 1 of 3, and the first thing decided. Every MCP call and every direct
|
|
5204
|
+
// write to channel.sock lands here, whether or not mcp-server was ever
|
|
5205
|
+
// involved — which is why this is the boundary and the tool list the model
|
|
5206
|
+
// was shown is not.
|
|
5207
|
+
//
|
|
5208
|
+
// Above the adapter check on purpose: "retry shortly" is the wrong answer
|
|
5209
|
+
// to a call that will never be allowed, and an agent that believes it is a
|
|
5210
|
+
// timing problem will keep trying.
|
|
5211
|
+
const permitted = this.checkToolPermission("ipc-outbound", instanceName, tool);
|
|
5212
|
+
if (!permitted.allowed) {
|
|
5213
|
+
respond(null, permitted.message);
|
|
5214
|
+
return;
|
|
5215
|
+
}
|
|
4842
5216
|
if (this.worlds.size === 0) {
|
|
4843
5217
|
respond(null, "Channel adapters are not ready — retry shortly");
|
|
4844
5218
|
return;
|
|
@@ -5096,6 +5470,15 @@ export class FleetManager {
|
|
|
5096
5470
|
*/
|
|
5097
5471
|
scheduleSourceAdapter(schedule) {
|
|
5098
5472
|
const chatId = String(schedule.reply_chat_id);
|
|
5473
|
+
// New schedules carry the adapter that owned the source chat at creation.
|
|
5474
|
+
// Keep that persona across source-instance rebinding, but never trust a
|
|
5475
|
+
// stale adapter after the chat has moved to another world.
|
|
5476
|
+
const persistedWorld = schedule.reply_adapter_id
|
|
5477
|
+
? this.worlds.get(schedule.reply_adapter_id)
|
|
5478
|
+
: undefined;
|
|
5479
|
+
if (persistedWorld && String(persistedWorld.groupId) === chatId) {
|
|
5480
|
+
return persistedWorld.adapter;
|
|
5481
|
+
}
|
|
5099
5482
|
const sourceAdapterId = schedule.source ? this.getInstanceAdapterId(schedule.source) : undefined;
|
|
5100
5483
|
const sourceWorld = sourceAdapterId ? this.worlds.get(sourceAdapterId) : undefined;
|
|
5101
5484
|
if (sourceWorld && String(sourceWorld.groupId) === chatId)
|
|
@@ -5127,6 +5510,58 @@ export class FleetManager {
|
|
|
5127
5510
|
threadId: schedule.reply_thread_id ?? undefined,
|
|
5128
5511
|
}).catch((err) => this.logger.error({ err }, "Failed to send schedule failure notification"));
|
|
5129
5512
|
}
|
|
5513
|
+
/**
|
|
5514
|
+
* The typed IPC messages, all through one door.
|
|
5515
|
+
*
|
|
5516
|
+
* They used to be five sibling branches on the dispatch, which is why the
|
|
5517
|
+
* permission question had five places to be forgotten. Routing them together
|
|
5518
|
+
* means the check above happens once and cannot be skipped by adding a
|
|
5519
|
+
* sixth — a new type has to appear in `IPC_TYPE_TOOLS` to be dispatched at
|
|
5520
|
+
* all.
|
|
5521
|
+
*/
|
|
5522
|
+
/**
|
|
5523
|
+
* Answer a refused typed message the way its handler would have.
|
|
5524
|
+
*
|
|
5525
|
+
* These are request/response over IPC: dropping the message silently leaves
|
|
5526
|
+
* the caller waiting for a reply that never comes, and a hung agent is a
|
|
5527
|
+
* worse failure than a refused one.
|
|
5528
|
+
*/
|
|
5529
|
+
refuseTypedIpc(name, msg, message) {
|
|
5530
|
+
const ipc = this.instanceIpcClients.get(name);
|
|
5531
|
+
const fleetRequestId = msg.fleetRequestId;
|
|
5532
|
+
if (!ipc || !fleetRequestId)
|
|
5533
|
+
return;
|
|
5534
|
+
const type = String(msg.type);
|
|
5535
|
+
const responseType = type.startsWith("fleet_schedule_") ? "fleet_schedule_response"
|
|
5536
|
+
: type.startsWith("fleet_decision_") ? "fleet_decision_response"
|
|
5537
|
+
: type === "fleet_task" ? "fleet_task_response"
|
|
5538
|
+
: type === "fleet_set_display_name" ? "fleet_display_name_response"
|
|
5539
|
+
: "fleet_description_response";
|
|
5540
|
+
ipc.send({ type: responseType, fleetRequestId, error: message });
|
|
5541
|
+
}
|
|
5542
|
+
dispatchTypedIpc(name, msg) {
|
|
5543
|
+
const type = String(msg.type);
|
|
5544
|
+
if (type.startsWith("fleet_schedule_")) {
|
|
5545
|
+
this.handleScheduleCrud(name, msg);
|
|
5546
|
+
return;
|
|
5547
|
+
}
|
|
5548
|
+
if (type.startsWith("fleet_decision_")) {
|
|
5549
|
+
this.handleDecisionCrud(name, msg);
|
|
5550
|
+
return;
|
|
5551
|
+
}
|
|
5552
|
+
if (type === "fleet_task") {
|
|
5553
|
+
this.handleTaskCrud(name, msg);
|
|
5554
|
+
return;
|
|
5555
|
+
}
|
|
5556
|
+
if (type === "fleet_set_display_name") {
|
|
5557
|
+
this.handleSetDisplayName(name, msg);
|
|
5558
|
+
return;
|
|
5559
|
+
}
|
|
5560
|
+
if (type === "fleet_set_description") {
|
|
5561
|
+
this.handleSetDescription(name, msg);
|
|
5562
|
+
return;
|
|
5563
|
+
}
|
|
5564
|
+
}
|
|
5130
5565
|
handleScheduleCrud(instanceName, msg) {
|
|
5131
5566
|
const fleetRequestId = msg.fleetRequestId;
|
|
5132
5567
|
const payload = (msg.payload ?? {});
|
|
@@ -5134,36 +5569,27 @@ export class FleetManager {
|
|
|
5134
5569
|
const ipc = this.instanceIpcClients.get(instanceName);
|
|
5135
5570
|
if (!ipc)
|
|
5136
5571
|
return;
|
|
5572
|
+
if (!this.scheduler) {
|
|
5573
|
+
// It did answer before, with whatever TypeError fell out of the try
|
|
5574
|
+
// block — "Cannot read properties of null (reading 'list')" is a stack
|
|
5575
|
+
// trace wearing an error message, and the agent reading it cannot tell
|
|
5576
|
+
// that the fleet simply has no scheduler.
|
|
5577
|
+
ipc.send({ type: "fleet_schedule_response", fleetRequestId, error: "Schedules are unavailable — the fleet scheduler is not running" });
|
|
5578
|
+
return;
|
|
5579
|
+
}
|
|
5137
5580
|
try {
|
|
5138
|
-
|
|
5139
|
-
|
|
5140
|
-
|
|
5141
|
-
|
|
5142
|
-
|
|
5143
|
-
|
|
5144
|
-
|
|
5145
|
-
|
|
5146
|
-
|
|
5147
|
-
|
|
5148
|
-
|
|
5149
|
-
|
|
5150
|
-
timezone: payload.timezone,
|
|
5151
|
-
silent: !!(payload.silent),
|
|
5152
|
-
};
|
|
5153
|
-
result = this.scheduler.create(params);
|
|
5154
|
-
break;
|
|
5155
|
-
}
|
|
5156
|
-
case "fleet_schedule_list":
|
|
5157
|
-
result = this.scheduler.list(payload.target);
|
|
5158
|
-
break;
|
|
5159
|
-
case "fleet_schedule_update":
|
|
5160
|
-
result = this.scheduler.update(payload.id, payload);
|
|
5161
|
-
break;
|
|
5162
|
-
case "fleet_schedule_delete":
|
|
5163
|
-
this.scheduler.delete(payload.id);
|
|
5164
|
-
result = "ok";
|
|
5165
|
-
break;
|
|
5166
|
-
}
|
|
5581
|
+
const op = String(msg.type).replace("fleet_schedule_", "");
|
|
5582
|
+
const result = this.performScheduleOp(instanceName, op, payload, {
|
|
5583
|
+
// The daemon sends its last chat id, which is unset until the instance
|
|
5584
|
+
// has had a chat message — and cross-instance traffic never sets it. So
|
|
5585
|
+
// a worker that only takes delegated tasks, the very instance #895 lets
|
|
5586
|
+
// self-schedule, hit "NOT NULL constraint failed: schedules.reply_chat_id".
|
|
5587
|
+
// No chat means no reply chat, exactly as on the agent endpoint.
|
|
5588
|
+
chatId: meta.chat_id ?? "",
|
|
5589
|
+
threadId: meta.thread_id || null,
|
|
5590
|
+
adapterId: meta.adapter_id || null,
|
|
5591
|
+
silent: !!(payload.silent),
|
|
5592
|
+
});
|
|
5167
5593
|
ipc.send({ type: "fleet_schedule_response", fleetRequestId, result });
|
|
5168
5594
|
}
|
|
5169
5595
|
catch (err) {
|
|
@@ -5175,8 +5601,15 @@ export class FleetManager {
|
|
|
5175
5601
|
const payload = (msg.payload ?? {});
|
|
5176
5602
|
const meta = (msg.meta ?? {});
|
|
5177
5603
|
const ipc = this.instanceIpcClients.get(instanceName);
|
|
5178
|
-
if (!ipc
|
|
5604
|
+
if (!ipc)
|
|
5605
|
+
return;
|
|
5606
|
+
if (!this.scheduler) {
|
|
5607
|
+
// Returning silently left the caller waiting for a response that was
|
|
5608
|
+
// never coming. Over IPC that is a hang, and a hang is a worse answer
|
|
5609
|
+
// than an error.
|
|
5610
|
+
ipc.send({ type: "fleet_decision_response", fleetRequestId, error: "Decisions are unavailable — the fleet scheduler is not running" });
|
|
5179
5611
|
return;
|
|
5612
|
+
}
|
|
5180
5613
|
const db = this.scheduler.db;
|
|
5181
5614
|
const projectRoot = meta.working_directory || this.fleetConfig?.instances[instanceName]?.working_directory || "";
|
|
5182
5615
|
try {
|
|
@@ -5292,23 +5725,76 @@ export class FleetManager {
|
|
|
5292
5725
|
async handleScheduleCrudHttp(instance, op, args) {
|
|
5293
5726
|
if (!this.scheduler)
|
|
5294
5727
|
return { error: "Scheduler not available" };
|
|
5728
|
+
if (op !== "create" && op !== "list" && op !== "update" && op !== "delete") {
|
|
5729
|
+
return { error: `Unknown schedule op: ${op}` };
|
|
5730
|
+
}
|
|
5731
|
+
// No bound chat on this path, as before: a schedule made through the agent
|
|
5732
|
+
// endpoint has nowhere of its own to reply.
|
|
5733
|
+
return this.performScheduleOp(instance, op, args, { chatId: "", threadId: null });
|
|
5734
|
+
}
|
|
5735
|
+
/**
|
|
5736
|
+
* The only place an agent's request creates, changes or removes a schedule
|
|
5737
|
+
* (#895). The IPC handler (MCP calls, direct channel.sock writes) and the
|
|
5738
|
+
* agent endpoint (agent-cli, HTTP agent mode) are adapters over this, so the
|
|
5739
|
+
* target check cannot be present on one face and missing on the other — the
|
|
5740
|
+
* shape #804 had to close twice.
|
|
5741
|
+
*
|
|
5742
|
+
* `caller` is the instance the server resolved for the request; any
|
|
5743
|
+
* `source` in `args` is ignored, and a schedule's source is always its caller.
|
|
5744
|
+
* A refusal is a ToolNotPermittedError: 403 on the agent endpoint, the error
|
|
5745
|
+
* of the schedule response on IPC.
|
|
5746
|
+
*/
|
|
5747
|
+
performScheduleOp(caller, op, args, reply) {
|
|
5748
|
+
const scheduler = this.scheduler;
|
|
5749
|
+
// Shape first, so the value the permission decision reads is the value the
|
|
5750
|
+
// scheduler gets. `target` used to be filtered through typeof for the
|
|
5751
|
+
// decision and passed raw to the scheduler, whose `target.startsWith`
|
|
5752
|
+
// then threw a TypeError for `target: 123` (#897). A non-string is refused
|
|
5753
|
+
// — never coerced: turning 123 into "123" would decide an identity
|
|
5754
|
+
// question on a value the caller did not send. `null` means absent, as
|
|
5755
|
+
// omitting it always did, and is removed before the scheduler sees it.
|
|
5756
|
+
if (args.target === null) {
|
|
5757
|
+
args = { ...args };
|
|
5758
|
+
delete args.target;
|
|
5759
|
+
}
|
|
5760
|
+
if (args.target !== undefined && typeof args.target !== "string") {
|
|
5761
|
+
throw new Error(`${op}_schedule: "target" must be an instance name (a string), not ${typeof args.target}.`);
|
|
5762
|
+
}
|
|
5763
|
+
if ((op === "update" || op === "delete") && typeof args.id !== "string") {
|
|
5764
|
+
throw new Error(`${op}_schedule: "id" must be a schedule id (a string) — get one from list_schedules.`);
|
|
5765
|
+
}
|
|
5766
|
+
const requestedTarget = args.target;
|
|
5767
|
+
if (op !== "list") {
|
|
5768
|
+
const profile = resolveToolSet(this.fleetConfig?.instances[caller], caller);
|
|
5769
|
+
const existing = op === "create" ? null : scheduler.get(args.id);
|
|
5770
|
+
const refusal = scheduleOpRefusal(profile, caller, op, { requestedTarget, existing });
|
|
5771
|
+
if (refusal) {
|
|
5772
|
+
this.logger.warn({ instance: caller, profile, op, target: requestedTarget ?? existing?.target, scheduleId: args.id }, "tool-permissions: schedule op refused");
|
|
5773
|
+
throw new ToolNotPermittedError(refusal);
|
|
5774
|
+
}
|
|
5775
|
+
}
|
|
5295
5776
|
switch (op) {
|
|
5296
5777
|
case "create":
|
|
5297
|
-
return
|
|
5778
|
+
return scheduler.create({
|
|
5298
5779
|
cron: args.cron,
|
|
5299
5780
|
at: args.at,
|
|
5300
5781
|
message: args.message,
|
|
5301
|
-
source:
|
|
5302
|
-
|
|
5782
|
+
source: caller,
|
|
5783
|
+
target: requestedTarget || caller,
|
|
5784
|
+
reply_chat_id: reply.chatId,
|
|
5785
|
+
reply_thread_id: reply.threadId,
|
|
5786
|
+
...(reply.adapterId !== undefined ? { reply_adapter_id: reply.adapterId } : {}),
|
|
5303
5787
|
label: args.label,
|
|
5304
5788
|
timezone: args.timezone,
|
|
5789
|
+
...(reply.silent !== undefined ? { silent: reply.silent } : {}),
|
|
5305
5790
|
});
|
|
5306
|
-
case "list":
|
|
5307
|
-
|
|
5791
|
+
case "list":
|
|
5792
|
+
return scheduler.list(requestedTarget);
|
|
5793
|
+
case "update":
|
|
5794
|
+
return scheduler.update(args.id, args);
|
|
5308
5795
|
case "delete":
|
|
5309
|
-
|
|
5796
|
+
scheduler.delete(args.id);
|
|
5310
5797
|
return "ok";
|
|
5311
|
-
default: return { error: `Unknown schedule op: ${op}` };
|
|
5312
5798
|
}
|
|
5313
5799
|
}
|
|
5314
5800
|
async handleDecisionCrudHttp(instance, op, args) {
|
|
@@ -5450,8 +5936,15 @@ export class FleetManager {
|
|
|
5450
5936
|
const payload = (msg.payload ?? {});
|
|
5451
5937
|
const meta = (msg.meta ?? {});
|
|
5452
5938
|
const ipc = this.instanceIpcClients.get(instanceName);
|
|
5453
|
-
if (!ipc
|
|
5939
|
+
if (!ipc)
|
|
5454
5940
|
return;
|
|
5941
|
+
if (!this.scheduler) {
|
|
5942
|
+
// Returning silently left the caller waiting for a response that was
|
|
5943
|
+
// never coming. Over IPC that is a hang, and a hang is a worse answer
|
|
5944
|
+
// than an error.
|
|
5945
|
+
ipc.send({ type: "fleet_task_response", fleetRequestId, error: "The task board is unavailable — the fleet scheduler is not running" });
|
|
5946
|
+
return;
|
|
5947
|
+
}
|
|
5455
5948
|
const db = this.scheduler.db;
|
|
5456
5949
|
const action = payload.action;
|
|
5457
5950
|
try {
|
|
@@ -6799,6 +7292,39 @@ export class FleetManager {
|
|
|
6799
7292
|
* (or an assist for one they stopped) must not stay clickable for the rest
|
|
6800
7293
|
* of its 15 minutes.
|
|
6801
7294
|
*/
|
|
7295
|
+
/**
|
|
7296
|
+
* Collapse every still-armed button prompt, for a fleet that is going away.
|
|
7297
|
+
*
|
|
7298
|
+
* The map is in memory; the buttons are in the chat for up to 24 hours. A
|
|
7299
|
+
* restart therefore leaves them looking live — pressing one takes the stale
|
|
7300
|
+
* branch, which acknowledges the click and drops the dismissal without
|
|
7301
|
+
* saying so. Retiring them here means the user meets a spent prompt instead
|
|
7302
|
+
* of a live-looking dead one.
|
|
7303
|
+
*
|
|
7304
|
+
* Best effort and bounded. Each collapse is a platform call, a day of tips
|
|
7305
|
+
* can be many of them, and shutdown still has instances to stop. Whatever
|
|
7306
|
+
* has not finished by the deadline is abandoned — that is exactly today's
|
|
7307
|
+
* behaviour, so the budget can only leave things no worse than before.
|
|
7308
|
+
*/
|
|
7309
|
+
async retirePendingNoncePrompts(budgetMs = NONCE_RETIRE_BUDGET_MS) {
|
|
7310
|
+
const entries = [...this.pendingNonceButtons.values()];
|
|
7311
|
+
this.pendingNonceButtons.clear();
|
|
7312
|
+
for (const entry of entries)
|
|
7313
|
+
if (entry.timer)
|
|
7314
|
+
clearTimeout(entry.timer);
|
|
7315
|
+
const collapses = entries
|
|
7316
|
+
.filter(entry => entry.messageId && entry.adapter.editMessageRemoveButtons)
|
|
7317
|
+
.map(entry => entry.adapter.editMessageRemoveButtons(entry.chatId, entry.messageId, entry.expiredText, entry.threadId).catch(err => this.logger.debug({ err, instanceName: entry.instanceName, prefix: entry.prefix }, "Failed to retire button prompt during shutdown")));
|
|
7318
|
+
if (!collapses.length)
|
|
7319
|
+
return;
|
|
7320
|
+
await Promise.race([
|
|
7321
|
+
Promise.allSettled(collapses),
|
|
7322
|
+
new Promise(resolve => {
|
|
7323
|
+
const timer = setTimeout(resolve, budgetMs);
|
|
7324
|
+
timer.unref?.();
|
|
7325
|
+
}),
|
|
7326
|
+
]);
|
|
7327
|
+
}
|
|
6802
7328
|
clearNoncePromptsForInstance(instanceName) {
|
|
6803
7329
|
for (const [nonce, entry] of this.pendingNonceButtons) {
|
|
6804
7330
|
if (entry.instanceName !== instanceName)
|
|
@@ -7739,11 +8265,39 @@ export class FleetManager {
|
|
|
7739
8265
|
for (const name of this.configuredBackendInstanceNames()) {
|
|
7740
8266
|
configured.add(this.backendNameOf(name));
|
|
7741
8267
|
}
|
|
7742
|
-
const
|
|
7743
|
-
|
|
7744
|
-
|
|
8268
|
+
const installed = this.probeInstalledBackends();
|
|
8269
|
+
const candidates = new Set([...installed, ...configured]);
|
|
8270
|
+
const unsupported = [];
|
|
8271
|
+
const choices = [...candidates].sort().flatMap(backend => {
|
|
8272
|
+
const flow = LOGIN_FLOWS[backend];
|
|
8273
|
+
const remoteLogin = !!flow && flow.remoteLogin !== "unsupported";
|
|
8274
|
+
const status = [];
|
|
8275
|
+
if (installed.has(backend))
|
|
8276
|
+
status.push(t("login.status_installed"));
|
|
8277
|
+
if (configured.has(backend))
|
|
8278
|
+
status.push(t("login.status_configured"));
|
|
8279
|
+
if (remoteLogin)
|
|
8280
|
+
status.push(t("login.status_auth"));
|
|
8281
|
+
else
|
|
8282
|
+
status.push(t("login.status_unsupported"));
|
|
8283
|
+
if (!remoteLogin) {
|
|
8284
|
+
unsupported.push({ backend, flow, status });
|
|
8285
|
+
return [];
|
|
8286
|
+
}
|
|
8287
|
+
return [{ action: backend, label: `${backend} · ${status.join(" · ")}` }];
|
|
8288
|
+
});
|
|
8289
|
+
if (unsupported.length) {
|
|
8290
|
+
const guidance = unsupported.map(({ backend, flow, status }) => `${backend} · ${status.join(" · ")} — ${flow?.remoteLogin === "unsupported"
|
|
8291
|
+
? t("login.remote_unsupported_agent_cli", backend, flow.command)
|
|
8292
|
+
: backend === "opencode" ? t("login.unsupported", backend) : t("login.no_remote_flow", backend)}`).join("\n");
|
|
8293
|
+
await chat.adapter.sendText(chat.chatId, guidance, { threadId: chat.threadId })
|
|
8294
|
+
.catch(err => this.logger.warn({ err }, "Failed to post unsupported login guidance"));
|
|
8295
|
+
}
|
|
7745
8296
|
if (choices.length === 0) {
|
|
7746
|
-
|
|
8297
|
+
if (!unsupported.length) {
|
|
8298
|
+
await chat.adapter.sendText(chat.chatId, t("login.none_available"), { threadId: chat.threadId })
|
|
8299
|
+
.catch(err => this.logger.warn({ err }, "Failed to post empty login chooser guidance"));
|
|
8300
|
+
}
|
|
7747
8301
|
return;
|
|
7748
8302
|
}
|
|
7749
8303
|
await this.postNonceButtonPrompt({
|
|
@@ -8332,6 +8886,12 @@ export class FleetManager {
|
|
|
8332
8886
|
await chat.adapter.sendText(chat.chatId, t("install.success_no_login", backend), { threadId: chat.threadId }).catch(() => { });
|
|
8333
8887
|
return;
|
|
8334
8888
|
}
|
|
8889
|
+
// The button is deliberately best-effort: it is a short-lived
|
|
8890
|
+
// capability and can be lost during an adapter reconnect. Always
|
|
8891
|
+
// publish a durable completion line first so a successful install can
|
|
8892
|
+
// never look like it silently disappeared. Include the binary that
|
|
8893
|
+
// passed the fresh-login-shell verification and the exact next step.
|
|
8894
|
+
await chat.adapter.sendText(chat.chatId, t("install.success", backend, info.binary), { threadId: chat.threadId }).catch(err => this.logger.warn({ err, backend }, "Failed to send durable install success notification"));
|
|
8335
8895
|
await this.postNonceButtonPrompt({
|
|
8336
8896
|
prefix: INSTALL_LOGIN_CALLBACK_PREFIX,
|
|
8337
8897
|
alertType: "login",
|
|
@@ -8583,6 +9143,8 @@ export class FleetManager {
|
|
|
8583
9143
|
// Grok reads AGENTS.md project docs; agy reads .agents/agents.md — the
|
|
8584
9144
|
// same files their writeConfig() appends fleet instructions to.
|
|
8585
9145
|
"grok": "AGENTS.md",
|
|
9146
|
+
// muse init scaffolds AGENTS.md and the binary reads it as project rules.
|
|
9147
|
+
"muse": "AGENTS.md",
|
|
8586
9148
|
"antigravity": ".agents/agents.md",
|
|
8587
9149
|
"mock": "CLAUDE.md",
|
|
8588
9150
|
};
|
|
@@ -8658,6 +9220,9 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
8658
9220
|
// vendor-canonical location is .grok/skills.
|
|
8659
9221
|
"opencode": [".agents", "skills"],
|
|
8660
9222
|
"grok": [".grok", "skills"],
|
|
9223
|
+
// Live-verified on muse 1.3.0: `muse skills list --source project` sees
|
|
9224
|
+
// .agents/skills and ignores .muse/skills.
|
|
9225
|
+
"muse": [".agents", "skills"],
|
|
8661
9226
|
"antigravity": [".agents", "skills"],
|
|
8662
9227
|
};
|
|
8663
9228
|
/** Copy general-knowledge steering + all role-eligible skills to General. */
|
|
@@ -9242,8 +9807,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9242
9807
|
* A model list is an aid: a vendor that stops answering must degrade to the
|
|
9243
9808
|
* previous list, never stall the command that asked for it.
|
|
9244
9809
|
*/
|
|
9245
|
-
async probeBackendBounded(backend) {
|
|
9246
|
-
const work = this.probeBackend(backend);
|
|
9810
|
+
async probeBackendBounded(backend, opts = {}) {
|
|
9811
|
+
const work = this.probeBackend(backend, opts);
|
|
9247
9812
|
work.catch(() => { });
|
|
9248
9813
|
let timer;
|
|
9249
9814
|
const deadline = new Promise(resolve => {
|
|
@@ -9307,12 +9872,23 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9307
9872
|
const label = option.description ? `${option.label} — ${option.description}` : option.label;
|
|
9308
9873
|
return option.id === currentModel ? `✓ ${label}` : label;
|
|
9309
9874
|
}
|
|
9310
|
-
/**
|
|
9311
|
-
|
|
9875
|
+
/**
|
|
9876
|
+
* Probe one backend's CLI env and cache it. Best-effort; never throws.
|
|
9877
|
+
*
|
|
9878
|
+
* `refreshVendorCatalog` is the "🔄 Refresh models" path only (#886): first
|
|
9879
|
+
* ask the CLI to refetch its own catalog, for backends whose probe merely
|
|
9880
|
+
* reads a file the CLI maintains. Every other caller leaves it off, so the
|
|
9881
|
+
* startup and /model probes behave exactly as before. A failed vendor
|
|
9882
|
+
* refresh fails the probe (null), which the menu reports instead of passing
|
|
9883
|
+
* the old list off as fresh.
|
|
9884
|
+
*/
|
|
9885
|
+
async probeBackend(backend, opts = {}) {
|
|
9312
9886
|
try {
|
|
9313
9887
|
const be = createBackend(backend, join(getAgendHome(), "cli-env"));
|
|
9314
9888
|
if (!be.probeCLIEnv)
|
|
9315
9889
|
return null;
|
|
9890
|
+
if (opts.refreshVendorCatalog && be.refreshModelCatalog)
|
|
9891
|
+
await be.refreshModelCatalog();
|
|
9316
9892
|
const probed = await be.probeCLIEnv({ workingDirectory: "", instanceDir: join(getAgendHome(), "cli-env"), instanceName: `probe-${backend}`, mcpServers: {} });
|
|
9317
9893
|
const env = { backend, probedAt: Date.now(), ...probed };
|
|
9318
9894
|
// An empty result must never overwrite a catalog we already have. Some
|
|
@@ -9361,22 +9937,48 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9361
9937
|
}
|
|
9362
9938
|
/** Best-effort model list for `/model`: cached CLI env first, else live probe. Never throws. */
|
|
9363
9939
|
async getModelOptions(instanceName, refresh = false, onLiveProbe) {
|
|
9940
|
+
return (await this.getModelOptionsWithSource(instanceName, { refresh, onLiveProbe })).models;
|
|
9941
|
+
}
|
|
9942
|
+
/**
|
|
9943
|
+
* The model list plus where it came from. `source` is "live" only when a probe
|
|
9944
|
+
* actually ran and answered; a refresh that failed or ran out of time reports
|
|
9945
|
+
* "cache" with the previous list, so a menu can say "could not refresh"
|
|
9946
|
+
* instead of presenting an old list as a fresh one.
|
|
9947
|
+
*/
|
|
9948
|
+
async getModelOptionsWithSource(instanceName, opts = {}) {
|
|
9364
9949
|
const backendName = this.backendNameForInstance(instanceName);
|
|
9365
9950
|
const cached = this.readCliEnv(backendName);
|
|
9366
|
-
if (!refresh && cached?.models.length && !this.cliEnvNeedsRefresh(cached))
|
|
9367
|
-
return cached.models;
|
|
9951
|
+
if (!opts.refresh && cached?.models.length && !this.cliEnvNeedsRefresh(cached)) {
|
|
9952
|
+
return { models: cached.models, source: "cache" };
|
|
9953
|
+
}
|
|
9368
9954
|
// About to go to the vendor: let the caller say so. A silent 1–10s pause on
|
|
9369
9955
|
// an interactive command reads as another hang, which is the wrong lesson to
|
|
9370
9956
|
// teach a user who has just been bitten by one.
|
|
9371
|
-
onLiveProbe?.();
|
|
9957
|
+
opts.onLiveProbe?.();
|
|
9372
9958
|
// Stale, missing, or a forced refresh → probe live (also refreshes the cache).
|
|
9373
9959
|
// A newly released model is invisible until this runs, which is why staleness
|
|
9374
9960
|
// triggers it rather than waiting for the 24h hard expiry or a cold start.
|
|
9375
|
-
const env = await this.probeBackendBounded(backendName);
|
|
9961
|
+
const env = await this.probeBackendBounded(backendName, { refreshVendorCatalog: opts.refreshVendorCatalog });
|
|
9376
9962
|
if (env?.models.length)
|
|
9377
|
-
return env.models;
|
|
9963
|
+
return { models: env.models, source: "live" };
|
|
9378
9964
|
// Probe failed or timed out: the previous list is still the best answer.
|
|
9379
|
-
return cached?.models ?? [];
|
|
9965
|
+
return { models: cached?.models ?? [], source: "cache" };
|
|
9966
|
+
}
|
|
9967
|
+
/**
|
|
9968
|
+
* The /model menu's choices, in the one order all three pickers share:
|
|
9969
|
+
* 🔄 Refresh first, then models, then (claude) "More models…". Refresh takes
|
|
9970
|
+
* a slot inside Discord's 25-option select cap, so it is paid for from the
|
|
9971
|
+
* model rows, never from "More models…".
|
|
9972
|
+
*/
|
|
9973
|
+
modelMenuChoices(instanceName, nonce, options, currentModel) {
|
|
9974
|
+
const isClaude = this.backendNameForInstance(instanceName) === "claude-code";
|
|
9975
|
+
const choices = [{ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:__refresh__`, label: t("model.refresh") }];
|
|
9976
|
+
for (const o of options.slice(0, isClaude ? 23 : 24)) {
|
|
9977
|
+
choices.push({ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:${o.id}`, label: this.modelChoiceLabel(o, currentModel) });
|
|
9978
|
+
}
|
|
9979
|
+
if (isClaude)
|
|
9980
|
+
choices.push({ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:__more__`, label: t("model.more") });
|
|
9981
|
+
return choices;
|
|
9380
9982
|
}
|
|
9381
9983
|
/**
|
|
9382
9984
|
* Model catalog behind the `list_models` tool.
|
|
@@ -9677,14 +10279,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9677
10279
|
// Raw id for ✓-matching options; display resolves an inherited CLI default.
|
|
9678
10280
|
const { model: currentModel, display: currentDisplay } = this.resolveInstanceModel(name);
|
|
9679
10281
|
const nonce = randomBytes(6).toString("hex");
|
|
9680
|
-
const
|
|
9681
|
-
const choices = options.slice(0, isClaude ? 24 : 25).map(o => ({
|
|
9682
|
-
id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:${o.id}`,
|
|
9683
|
-
label: this.modelChoiceLabel(o, currentModel),
|
|
9684
|
-
}));
|
|
9685
|
-
if (isClaude) {
|
|
9686
|
-
choices.push({ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:__more__`, label: t("model.more") });
|
|
9687
|
-
}
|
|
10282
|
+
const choices = this.modelMenuChoices(name, nonce, options, currentModel);
|
|
9688
10283
|
const timer = setTimeout(() => this.pendingModelSelects.delete(nonce), CLASSIC_BACKEND_SELECTION_TIMEOUT_MS);
|
|
9689
10284
|
timer.unref?.();
|
|
9690
10285
|
this.pendingModelSelects.set(nonce, { instanceName: name, model: "", userId: data.userId, channelId: data.channelId, timer, respond: data.respond, respondChoices: data.respondChoices });
|
|
@@ -9710,15 +10305,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9710
10305
|
}
|
|
9711
10306
|
const { model: currentModel, display: currentDisplay } = this.resolveInstanceModel(instanceName);
|
|
9712
10307
|
const nonce = randomBytes(6).toString("hex");
|
|
9713
|
-
const
|
|
9714
|
-
// Keep the more-models entry inside Discord's 25-option select cap.
|
|
9715
|
-
const choices = options.slice(0, isClaude ? 24 : 25).map(o => ({
|
|
9716
|
-
id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:${o.id}`,
|
|
9717
|
-
label: this.modelChoiceLabel(o, currentModel),
|
|
9718
|
-
}));
|
|
9719
|
-
if (isClaude) {
|
|
9720
|
-
choices.push({ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:__more__`, label: t("model.more") });
|
|
9721
|
-
}
|
|
10308
|
+
const choices = this.modelMenuChoices(instanceName, nonce, options, currentModel);
|
|
9722
10309
|
const respond = async (text) => {
|
|
9723
10310
|
await adapter.sendText(chatId, text, { threadId });
|
|
9724
10311
|
return undefined;
|
|
@@ -9802,6 +10389,64 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9802
10389
|
await pending.respond(t("model.more_unavailable")).catch(() => { });
|
|
9803
10390
|
}
|
|
9804
10391
|
}
|
|
10392
|
+
/**
|
|
10393
|
+
* "🔄 Refresh models" (#886): probe live past both caches and redraw the menu.
|
|
10394
|
+
*
|
|
10395
|
+
* Past AgEnD's cli-env cache (refresh=true) and, where the backend supports
|
|
10396
|
+
* it, past the CLI's own catalog cache too (codex: `codex debug models`).
|
|
10397
|
+
* Without the second half, a codex refresh re-read the same models_cache.json
|
|
10398
|
+
* and showed the old list as new.
|
|
10399
|
+
*
|
|
10400
|
+
* A failed refresh keeps the previous list and says so. It never blanks the
|
|
10401
|
+
* menu, because a picker with no rows cannot even offer another refresh.
|
|
10402
|
+
*/
|
|
10403
|
+
async refreshModelMenu(pending) {
|
|
10404
|
+
const { models, source } = await this.getModelOptionsWithSource(pending.instanceName, {
|
|
10405
|
+
refresh: true, refreshVendorCatalog: true,
|
|
10406
|
+
});
|
|
10407
|
+
if (models.length === 0) {
|
|
10408
|
+
await pending.respond(t("model.list_unavailable", pending.instanceName)).catch(() => { });
|
|
10409
|
+
return;
|
|
10410
|
+
}
|
|
10411
|
+
const { model: currentModel, display: currentDisplay } = this.resolveInstanceModel(pending.instanceName);
|
|
10412
|
+
const nonce = randomBytes(6).toString("hex");
|
|
10413
|
+
const choices = this.modelMenuChoices(pending.instanceName, nonce, models, currentModel);
|
|
10414
|
+
const status = source === "live" ? t("model.refreshed") : t("model.refresh_failed");
|
|
10415
|
+
const timer = setTimeout(() => {
|
|
10416
|
+
const p = this.pendingModelSelects.get(nonce);
|
|
10417
|
+
if (p) {
|
|
10418
|
+
this.pendingModelSelects.delete(nonce);
|
|
10419
|
+
p.respond(t("model.selection_expired")).catch(() => { });
|
|
10420
|
+
}
|
|
10421
|
+
}, CLASSIC_BACKEND_SELECTION_TIMEOUT_MS);
|
|
10422
|
+
timer.unref?.();
|
|
10423
|
+
this.pendingModelSelects.set(nonce, { ...pending, model: "", timer });
|
|
10424
|
+
try {
|
|
10425
|
+
if (pending.respondChoices) {
|
|
10426
|
+
// Discord select menu: edit the same interaction reply in place.
|
|
10427
|
+
await pending.respondChoices(`${status}\n${t("model.menu", `**${currentDisplay}**`)}`, choices);
|
|
10428
|
+
return;
|
|
10429
|
+
}
|
|
10430
|
+
if (pending.adapter && pending.adapterChatId) {
|
|
10431
|
+
// Telegram: retire the consumed keyboard, then post the redrawn menu.
|
|
10432
|
+
if (pending.menuMessageId && pending.adapter.editMessageRemoveButtons) {
|
|
10433
|
+
await pending.adapter.editMessageRemoveButtons(pending.adapterChatId, pending.menuMessageId, t("model.refresh"), pending.adapterThreadId).catch(() => { });
|
|
10434
|
+
}
|
|
10435
|
+
const menuMessageId = await pending.adapter.promptUser(pending.adapterChatId, `${status}\n${t("model.menu", currentDisplay)}`, choices, { threadId: pending.adapterThreadId });
|
|
10436
|
+
const fresh = this.pendingModelSelects.get(nonce);
|
|
10437
|
+
if (fresh)
|
|
10438
|
+
fresh.menuMessageId = menuMessageId;
|
|
10439
|
+
return;
|
|
10440
|
+
}
|
|
10441
|
+
await pending.respond(t("model.usage")).catch(() => { });
|
|
10442
|
+
}
|
|
10443
|
+
catch (err) {
|
|
10444
|
+
this.pendingModelSelects.delete(nonce);
|
|
10445
|
+
clearTimeout(timer);
|
|
10446
|
+
this.logger.warn({ err, instanceName: pending.instanceName }, "Refreshed model menu failed");
|
|
10447
|
+
await pending.respond(t("model.usage")).catch(() => { });
|
|
10448
|
+
}
|
|
10449
|
+
}
|
|
9805
10450
|
/** Consume a `/model` selection callback. Returns true for all model-select ids (incl. stale). */
|
|
9806
10451
|
async handleModelSelection(data) {
|
|
9807
10452
|
if (!data.callbackData.startsWith(MODEL_SELECT_CALLBACK_PREFIX))
|
|
@@ -9827,6 +10472,12 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9827
10472
|
await this.expandClaudeModelMenu(pending);
|
|
9828
10473
|
return true;
|
|
9829
10474
|
}
|
|
10475
|
+
// "🔄 Refresh models" is navigation too: re-probe past every cache and
|
|
10476
|
+
// redraw the same menu with what came back.
|
|
10477
|
+
if (model === "__refresh__") {
|
|
10478
|
+
await this.refreshModelMenu(pending);
|
|
10479
|
+
return true;
|
|
10480
|
+
}
|
|
9830
10481
|
// Send immediate "⏳ Switching..." feedback, then apply in background.
|
|
9831
10482
|
const progressText = t("model.switching", pending.instanceName, model);
|
|
9832
10483
|
let progressMsgId;
|
|
@@ -10360,6 +11011,15 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10360
11011
|
clearInterval(this.logRotateTimer);
|
|
10361
11012
|
this.logRotateTimer = null;
|
|
10362
11013
|
}
|
|
11014
|
+
if (this.discordPresenceTimer) {
|
|
11015
|
+
clearInterval(this.discordPresenceTimer);
|
|
11016
|
+
this.discordPresenceTimer = null;
|
|
11017
|
+
}
|
|
11018
|
+
if (this.discordPresenceEagerTimer) {
|
|
11019
|
+
clearTimeout(this.discordPresenceEagerTimer);
|
|
11020
|
+
this.discordPresenceEagerTimer = null;
|
|
11021
|
+
}
|
|
11022
|
+
this.discordPresenceEagerPending = false;
|
|
10363
11023
|
// Cancel-button timers were never cleared here. The idle-check interval is not
|
|
10364
11024
|
// unref'd, so it held the event loop open past shutdown and kept retrying
|
|
10365
11025
|
// deletes against an adapter that was already gone.
|
|
@@ -10397,11 +11057,9 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10397
11057
|
for (const pending of this.pendingClassicStarts.values())
|
|
10398
11058
|
clearTimeout(pending.timer);
|
|
10399
11059
|
this.pendingClassicStarts.clear();
|
|
10400
|
-
|
|
10401
|
-
|
|
10402
|
-
|
|
10403
|
-
}
|
|
10404
|
-
this.pendingNonceButtons.clear();
|
|
11060
|
+
// Adapters are still connected here — they are stopped further down — so
|
|
11061
|
+
// this is the last moment the prompts can be collapsed.
|
|
11062
|
+
await this.retirePendingNoncePrompts();
|
|
10405
11063
|
this.topicArchiver.stop();
|
|
10406
11064
|
this.scheduler?.shutdown();
|
|
10407
11065
|
// Stop instances in parallel batches to avoid long sequential waits.
|
|
@@ -10586,9 +11244,9 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10586
11244
|
* Whitelisted runtime fields are pushed into live daemons; all other instance
|
|
10587
11245
|
* fields, plus cold fleet-level settings, retain restart semantics.
|
|
10588
11246
|
*/
|
|
10589
|
-
async reconcileInstances() {
|
|
11247
|
+
async reconcileInstances(observe) {
|
|
10590
11248
|
if (!this.configPath)
|
|
10591
|
-
return;
|
|
11249
|
+
return {};
|
|
10592
11250
|
const oldConfig = this.fleetConfig;
|
|
10593
11251
|
const previousRawConfig = this.rawFleetConfig;
|
|
10594
11252
|
const previousRawDocument = this.rawFleetDocument;
|
|
@@ -10631,7 +11289,9 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10631
11289
|
? t("fleet.reload_removed_all", oldCount)
|
|
10632
11290
|
: t("fleet.reload_removed_half", oldCount, newCount);
|
|
10633
11291
|
this.notifyFleetError(t("fleet.reload_rejected", why));
|
|
10634
|
-
|
|
11292
|
+
// Reported, not swallowed: an apply job whose config was refused used to
|
|
11293
|
+
// mark every row done and tell the user "changes applied".
|
|
11294
|
+
return { rejected: why };
|
|
10635
11295
|
}
|
|
10636
11296
|
// Classic behavior settings share the fleet defaults but are not entries
|
|
10637
11297
|
// in fleet.yaml. Snapshot the old effective chain before reloading the
|
|
@@ -10662,29 +11322,31 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10662
11322
|
this.scheduler?.reload();
|
|
10663
11323
|
const newInstances = this.fleetConfig.instances;
|
|
10664
11324
|
const topicMode = this.fleetConfig?.channel?.mode === "topic";
|
|
10665
|
-
//
|
|
10666
|
-
//
|
|
10667
|
-
|
|
10668
|
-
|
|
10669
|
-
|
|
10670
|
-
|
|
10671
|
-
|
|
10672
|
-
|
|
10673
|
-
|
|
10674
|
-
|
|
10675
|
-
|
|
10676
|
-
|
|
10677
|
-
const oldFleetLevel = JSON.stringify({ channel: oldConfig?.channel, defaults: oldDefaultCold });
|
|
10678
|
-
const newFleetLevel = JSON.stringify({ channel: this.fleetConfig?.channel, defaults: newDefaultCold });
|
|
10679
|
-
if (oldFleetLevel !== newFleetLevel) {
|
|
10680
|
-
this.logger.warn("Fleet-level config changed (channel/defaults) — use /restart for full effect");
|
|
11325
|
+
// Only what a fresh process can adopt, and only relative to the signature
|
|
11326
|
+
// this process came up on: a Settings edit mutates this.fleetConfig in place
|
|
11327
|
+
// before the reload, so comparing the pre-load copy sees nothing at all.
|
|
11328
|
+
const newFleetLevel = this.fleetLevelSignature();
|
|
11329
|
+
if (this.appliedFleetLevel !== null && this.appliedFleetLevel !== newFleetLevel) {
|
|
11330
|
+
this.logger.warn({
|
|
11331
|
+
keys: fleetLevelDifferences(this.startupFleetConfig, this.fleetConfig),
|
|
11332
|
+
}, "Fleet-level config changed — restart AgEnD for it to take effect");
|
|
11333
|
+
// Terminal, and deliberately not "done": this reconcile cannot adopt a
|
|
11334
|
+
// fleet-level change, and saying otherwise would claim AgEnD is running
|
|
11335
|
+
// on a configuration it is not running on.
|
|
11336
|
+
observe?.(APPLY_FLEET_TARGET, "restart", "restart-required");
|
|
10681
11337
|
}
|
|
10682
11338
|
// Stop removed instances (skip classic bot instances — they're managed by classicBot.yaml)
|
|
10683
11339
|
const classicNames = new Set(this.classicChannels?.getAll().map(ch => ch.instanceName) ?? []);
|
|
10684
11340
|
for (const name of this.daemons.keys()) {
|
|
10685
11341
|
if (!(name in newInstances) && !classicNames.has(name)) {
|
|
10686
11342
|
this.logger.info({ name }, "Instance removed from config — stopping");
|
|
10687
|
-
|
|
11343
|
+
observe?.(name, "restart", "running");
|
|
11344
|
+
await this.stopInstance(name)
|
|
11345
|
+
.then(() => observe?.(name, "restart", "done"))
|
|
11346
|
+
.catch(err => {
|
|
11347
|
+
observe?.(name, "restart", "failed", err.message);
|
|
11348
|
+
this.logger.error({ err, name }, "Failed to stop removed instance");
|
|
11349
|
+
});
|
|
10688
11350
|
}
|
|
10689
11351
|
}
|
|
10690
11352
|
// Start new + reconcile modified instances. Hot values are always sent as a
|
|
@@ -10694,20 +11356,23 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10694
11356
|
if (!this.daemons.has(name)) {
|
|
10695
11357
|
// New instance — startInstance already calls connectIpcToInstance
|
|
10696
11358
|
this.logger.info({ name }, "New instance in config — starting");
|
|
11359
|
+
observe?.(name, "restart", "running");
|
|
10697
11360
|
await this.startInstanceUnattended(name, config, topicMode, "new instance");
|
|
11361
|
+
observe?.(name, "restart", "done");
|
|
10698
11362
|
}
|
|
10699
11363
|
else if (oldConfig?.instances[name]) {
|
|
10700
11364
|
const daemon = this.daemons.get(name);
|
|
10701
11365
|
const runtimeConfig = daemon.getConfigSnapshot?.() ?? oldConfig.instances[name];
|
|
10702
|
-
const
|
|
10703
|
-
|
|
10704
|
-
// Every field not explicitly classified hot is cold by default.
|
|
10705
|
-
if (!isDeepStrictEqual(oldParts.cold, newParts.cold)) {
|
|
11366
|
+
const change = classifyInstanceChange(runtimeConfig, config);
|
|
11367
|
+
if (change === "restart") {
|
|
10706
11368
|
this.logger.info({ name }, "Instance config changed — restarting");
|
|
11369
|
+
observe?.(name, "restart", "running");
|
|
10707
11370
|
await this.stopInstance(name).catch(() => { });
|
|
10708
11371
|
await this.startInstanceUnattended(name, config, topicMode, "modified instance");
|
|
11372
|
+
observe?.(name, "restart", "done");
|
|
10709
11373
|
}
|
|
10710
|
-
else if (
|
|
11374
|
+
else if (change === "hot") {
|
|
11375
|
+
observe?.(name, "hot", "running");
|
|
10711
11376
|
const update = hotConfigUpdate(config);
|
|
10712
11377
|
const ipc = this.instanceIpcClients.get(name);
|
|
10713
11378
|
const sent = ipc?.connected === true && ipc.send({ type: "config_update", config: update });
|
|
@@ -10718,6 +11383,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10718
11383
|
this.logger.warn({ name }, "Config-update IPC unavailable — applied hot config in-process");
|
|
10719
11384
|
}
|
|
10720
11385
|
this.logger.info({ name, fields: [...HOT_INSTANCE_CONFIG_KEYS] }, "Instance hot config reloaded");
|
|
11386
|
+
observe?.(name, "hot", "done");
|
|
10721
11387
|
}
|
|
10722
11388
|
}
|
|
10723
11389
|
}
|
|
@@ -10735,25 +11401,1155 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10735
11401
|
const autoPauseAfter = this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after);
|
|
10736
11402
|
if (old.backend !== backend || old.model !== model || old.autoPauseAfter !== autoPauseAfter) {
|
|
10737
11403
|
this.logger.info({ instanceName: ch.instanceName }, "Classic cold config changed — restarting");
|
|
11404
|
+
observe?.(ch.instanceName, "restart", "running");
|
|
10738
11405
|
await this.stopInstance(ch.instanceName).catch(() => { });
|
|
10739
11406
|
await this.startClassicInstanceUnattended(ch, "classic instance after fleet reload");
|
|
11407
|
+
observe?.(ch.instanceName, "restart", "done");
|
|
10740
11408
|
continue;
|
|
10741
11409
|
}
|
|
10742
11410
|
const toolProgress = this.classicChannels.getToolProgress(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.tool_progress);
|
|
10743
11411
|
const replyCompletionGuard = this.classicChannels.getReplyCompletionGuard(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.reply_completion_guard);
|
|
10744
11412
|
if (old.toolProgress !== toolProgress || old.replyCompletionGuard !== replyCompletionGuard) {
|
|
11413
|
+
observe?.(ch.instanceName, "hot", "running");
|
|
10745
11414
|
this.applyHotConfigUpdate(ch.instanceName, {
|
|
10746
11415
|
tool_progress: toolProgress,
|
|
10747
11416
|
reply_completion_guard: replyCompletionGuard,
|
|
10748
11417
|
});
|
|
10749
11418
|
this.logger.info({ instanceName: ch.instanceName }, "Classic inherited hot config reloaded");
|
|
11419
|
+
observe?.(ch.instanceName, "hot", "done");
|
|
10750
11420
|
}
|
|
10751
11421
|
}
|
|
10752
11422
|
}
|
|
10753
11423
|
// warm_cap is fleet-owned; enforce the reloaded value immediately against
|
|
10754
11424
|
// currently idle instances instead of waiting for a future state edge.
|
|
10755
11425
|
this.enforceWarmCap();
|
|
11426
|
+
// appliedFleetLevel is deliberately NOT updated here. It means "the
|
|
11427
|
+
// fleet-level config this process started on"; a reconcile does not restart
|
|
11428
|
+
// the process, so moving it would erase the fact that a restart is still
|
|
11429
|
+
// owed and silence every later reminder.
|
|
10756
11430
|
this.logger.info({ running: this.daemons.size, configured: Object.keys(newInstances).length }, "Reconcile complete");
|
|
11431
|
+
return {};
|
|
11432
|
+
}
|
|
11433
|
+
/**
|
|
11434
|
+
* `{channel, cold defaults}` as one comparable string — the part of the config
|
|
11435
|
+
* a running fleet process cannot adopt without restarting.
|
|
11436
|
+
*/
|
|
11437
|
+
fleetLevelSignature(config = this.fleetConfig) {
|
|
11438
|
+
return fleetLevelSignature(config);
|
|
11439
|
+
}
|
|
11440
|
+
/**
|
|
11441
|
+
* The config a reconcile is about to load, not the one held in memory.
|
|
11442
|
+
*
|
|
11443
|
+
* Settings mutates the in-memory object and writes the file; only the file
|
|
11444
|
+
* goes back through defaults expansion. Forecasting from memory therefore
|
|
11445
|
+
* misses every instance that a changed fleet default will restart — the user
|
|
11446
|
+
* is told "restart AgEnD" and not told that five agents are about to go down.
|
|
11447
|
+
*/
|
|
11448
|
+
nextFleetConfig() {
|
|
11449
|
+
if (!this.configPath)
|
|
11450
|
+
return this.fleetConfig;
|
|
11451
|
+
try {
|
|
11452
|
+
return loadFleetConfig(this.configPath);
|
|
11453
|
+
}
|
|
11454
|
+
catch (err) {
|
|
11455
|
+
// An unparseable file is the reconcile's problem to report; the forecast
|
|
11456
|
+
// falls back to what is running rather than failing the request.
|
|
11457
|
+
this.logger.debug({ err }, "Apply plan fell back to the in-memory config");
|
|
11458
|
+
return this.fleetConfig;
|
|
11459
|
+
}
|
|
11460
|
+
}
|
|
11461
|
+
/**
|
|
11462
|
+
* Does the config this process is running match the one a reconcile would
|
|
11463
|
+
* load off disk?
|
|
11464
|
+
*
|
|
11465
|
+
* If not, every apply reports a fleet-level change that a restart cannot
|
|
11466
|
+
* clear — restart, recompute, disagree again — a self-sustaining loop that
|
|
11467
|
+
* the rate limit can only slow to three naggings an hour. Startup rewrites
|
|
11468
|
+
* the file in three places before this point (slimFleetConfigAtStartup, the
|
|
11469
|
+
* general auto-create, the general fixup), so the two can genuinely diverge.
|
|
11470
|
+
*
|
|
11471
|
+
* An offer to restart that cannot possibly succeed is worse than no offer, so
|
|
11472
|
+
* the panel shows the mismatch instead of a button.
|
|
11473
|
+
*/
|
|
11474
|
+
checkStartupSignatureConsistency() {
|
|
11475
|
+
if (!this.configPath) {
|
|
11476
|
+
this.fleetSignatureMismatch = null;
|
|
11477
|
+
return;
|
|
11478
|
+
}
|
|
11479
|
+
let onDisk;
|
|
11480
|
+
try {
|
|
11481
|
+
onDisk = loadFleetConfig(this.configPath);
|
|
11482
|
+
}
|
|
11483
|
+
catch (err) {
|
|
11484
|
+
this.logger.warn({ err }, "Could not re-read fleet.yaml to check the startup signature");
|
|
11485
|
+
this.fleetSignatureMismatch = null;
|
|
11486
|
+
return;
|
|
11487
|
+
}
|
|
11488
|
+
if (fleetLevelSignature(onDisk) === this.appliedFleetLevel) {
|
|
11489
|
+
this.fleetSignatureMismatch = null;
|
|
11490
|
+
return;
|
|
11491
|
+
}
|
|
11492
|
+
this.fleetSignatureMismatch = fleetLevelDifferences(this.fleetConfig, onDisk);
|
|
11493
|
+
this.logger.warn({
|
|
11494
|
+
keys: this.fleetSignatureMismatch,
|
|
11495
|
+
configPath: this.configPath,
|
|
11496
|
+
}, "fleet.yaml and the running configuration disagree on startup-only keys — every apply will ask for a restart that cannot clear it");
|
|
11497
|
+
}
|
|
11498
|
+
/**
|
|
11499
|
+
* Is a live adapter already long-polling this bot token?
|
|
11500
|
+
*
|
|
11501
|
+
* Telegram's `getUpdates` has exactly one consumer: a second poller takes
|
|
11502
|
+
* turns with the first and both miss messages. The setup wizard's "post in
|
|
11503
|
+
* the group and I'll detect it" step is a second poller, so it has to know
|
|
11504
|
+
* when the answer is "not against this token, not while I'm running".
|
|
11505
|
+
*/
|
|
11506
|
+
isBotTokenInUse(token) {
|
|
11507
|
+
if (!token)
|
|
11508
|
+
return false;
|
|
11509
|
+
const configured = this.fleetConfig?.channels
|
|
11510
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []);
|
|
11511
|
+
for (const channel of configured) {
|
|
11512
|
+
const envVar = channel?.bot_token_env;
|
|
11513
|
+
if (!envVar)
|
|
11514
|
+
continue;
|
|
11515
|
+
// Compare the value, not the variable name: the same token can be
|
|
11516
|
+
// reached through a differently named variable.
|
|
11517
|
+
if (process.env[envVar] === token)
|
|
11518
|
+
return true;
|
|
11519
|
+
}
|
|
11520
|
+
return false;
|
|
11521
|
+
}
|
|
11522
|
+
/** Non-null when startup found the running config and fleet.yaml disagreeing. */
|
|
11523
|
+
fleetSignatureMismatchKeys() {
|
|
11524
|
+
return this.fleetSignatureMismatch;
|
|
11525
|
+
}
|
|
11526
|
+
/**
|
|
11527
|
+
* Restart AgEnD itself on behalf of a Settings apply.
|
|
11528
|
+
*
|
|
11529
|
+
* Deliberately a separate action from Apply: the panel is reachable from
|
|
11530
|
+
* outside the LAN, and "restart the whole fleet" must never be something a
|
|
11531
|
+
* single Apply click can carry along with it.
|
|
11532
|
+
*
|
|
11533
|
+
* The order below is the safety envelope, and the order matters:
|
|
11534
|
+
* concurrency and consistency first (cheap, and a restart during a reconcile
|
|
11535
|
+
* is the dangerous one), then the state checks, then the rate limit, then the
|
|
11536
|
+
* audit notice — and only once all of that holds is the attempt written to
|
|
11537
|
+
* disk and fsynced, before anything spawns.
|
|
11538
|
+
*/
|
|
11539
|
+
async requestSettingsSelfRestart(jobId, key) {
|
|
11540
|
+
// A retry after a lost response must not restart a second time. The key is
|
|
11541
|
+
// recorded on the job, which survives the restart it triggers.
|
|
11542
|
+
const already = this.applyJobs.findByRestartKey(key);
|
|
11543
|
+
if (already)
|
|
11544
|
+
return { ok: true, jobId: already.id, reused: true };
|
|
11545
|
+
if (this.reconcileInFlight || this.activeApplyJobId) {
|
|
11546
|
+
return { ok: false, status: 409, error: "a configuration reload is running — try again once it finishes" };
|
|
11547
|
+
}
|
|
11548
|
+
if (this.fleetSignatureMismatch) {
|
|
11549
|
+
// Restarting cannot clear this, so offering it would be a loop.
|
|
11550
|
+
return {
|
|
11551
|
+
ok: false,
|
|
11552
|
+
status: 409,
|
|
11553
|
+
error: `fleet.yaml and the running configuration disagree on ${this.fleetSignatureMismatch.join(", ")} — check fleet.log before restarting`,
|
|
11554
|
+
};
|
|
11555
|
+
}
|
|
11556
|
+
const job = this.applyJobs.get(jobId);
|
|
11557
|
+
const row = job?.targets.find(item => item.target === APPLY_FLEET_TARGET);
|
|
11558
|
+
if (!job || !row || row.status !== "restart-required") {
|
|
11559
|
+
return { ok: false, status: 409, error: "that apply has no pending fleet-level change" };
|
|
11560
|
+
}
|
|
11561
|
+
// The row alone is not enough: an old job keeps its row for the whole
|
|
11562
|
+
// retention window, so a change made and then reverted would still leave a
|
|
11563
|
+
// job that looks restartable. The live signature is the authority — and it
|
|
11564
|
+
// must be the same view of the config the plan uses.
|
|
11565
|
+
if (this.appliedFleetLevel === null || this.appliedFleetLevel === this.fleetLevelSignature(this.nextFleetConfig())) {
|
|
11566
|
+
return { ok: false, status: 409, error: "no fleet-level change is pending any more" };
|
|
11567
|
+
}
|
|
11568
|
+
const allowance = checkSelfRestartAllowance(this.dataDir);
|
|
11569
|
+
if (!allowance.allowed) {
|
|
11570
|
+
if (allowance.reason === "unreadable") {
|
|
11571
|
+
// Fail closed: with the limit's own state in doubt, "no attempts yet"
|
|
11572
|
+
// is the one reading that must not be assumed.
|
|
11573
|
+
return {
|
|
11574
|
+
ok: false,
|
|
11575
|
+
status: 503,
|
|
11576
|
+
error: "the restart rate-limit file cannot be read — remove self-restart.json from the data dir on the host, or run `agend restart` there",
|
|
11577
|
+
};
|
|
11578
|
+
}
|
|
11579
|
+
return {
|
|
11580
|
+
ok: false,
|
|
11581
|
+
status: 429,
|
|
11582
|
+
error: allowance.reason === "too-soon"
|
|
11583
|
+
? "AgEnD was restarted from Settings very recently"
|
|
11584
|
+
: "too many Settings-triggered restarts in the last hour",
|
|
11585
|
+
retryAfterSeconds: allowance.retryAfterSeconds,
|
|
11586
|
+
};
|
|
11587
|
+
}
|
|
11588
|
+
// Recorded before the notice, not after. If recording keeps failing (a
|
|
11589
|
+
// read-only data dir), posting first would let whoever holds the token spam
|
|
11590
|
+
// the channel with "restarting…" notices for restarts that never happen.
|
|
11591
|
+
// The cost is that a failed announcement still spends an attempt, which is
|
|
11592
|
+
// the right way round for a rate limit.
|
|
11593
|
+
if (!recordSelfRestartAttempt(this.dataDir)) {
|
|
11594
|
+
this.logger.error("Self-restart attempt could not be recorded — refusing to restart unmetered");
|
|
11595
|
+
return { ok: false, status: 503, error: "could not record the restart attempt" };
|
|
11596
|
+
}
|
|
11597
|
+
// Out-of-band notice before the restart, so a panel restart is visible where
|
|
11598
|
+
// the admins are. Refusing when it cannot be posted is the same rule as
|
|
11599
|
+
// refusing when the progress marker cannot be written: no untraceable
|
|
11600
|
+
// restarts.
|
|
11601
|
+
const notice = await this.postSelfRestartNotice();
|
|
11602
|
+
if (!notice) {
|
|
11603
|
+
return {
|
|
11604
|
+
ok: false,
|
|
11605
|
+
status: 409,
|
|
11606
|
+
error: "no chat channel is available to announce the restart — run `agend restart` on the host instead",
|
|
11607
|
+
};
|
|
11608
|
+
}
|
|
11609
|
+
// Consume the row: it moves to running, which is also what lets the next
|
|
11610
|
+
// process settle it (settleAfterRestart only touches non-terminal rows).
|
|
11611
|
+
this.applyJobs.update(jobId, current => {
|
|
11612
|
+
const target = current.targets.find(item => item.target === APPLY_FLEET_TARGET);
|
|
11613
|
+
if (target)
|
|
11614
|
+
target.status = "running";
|
|
11615
|
+
current.restart_key = key;
|
|
11616
|
+
current.deadlineMs = SELF_RESTART_DEADLINE_MS;
|
|
11617
|
+
});
|
|
11618
|
+
this.emitSseEvent("apply_progress", viewOf(this.applyJobs.get(jobId)));
|
|
11619
|
+
const launched = await this.requestFullRestart(notice.adapter, notice.chatId, notice.threadId, notice.messageId)
|
|
11620
|
+
.catch(err => {
|
|
11621
|
+
this.logger.error({ err }, "Settings-triggered self restart failed to launch");
|
|
11622
|
+
return false;
|
|
11623
|
+
});
|
|
11624
|
+
if (!launched) {
|
|
11625
|
+
this.applyJobs.setTargetStatus(jobId, APPLY_FLEET_TARGET, "failed", "restart could not be launched");
|
|
11626
|
+
this.applyJobs.finish(jobId, "restart could not be launched");
|
|
11627
|
+
this.emitSseEvent("apply_progress", viewOf(this.applyJobs.get(jobId)));
|
|
11628
|
+
return { ok: false, status: 409, error: "the restart could not be launched — see fleet.log" };
|
|
11629
|
+
}
|
|
11630
|
+
return { ok: true, jobId };
|
|
11631
|
+
}
|
|
11632
|
+
/** The audit notice, and the message the restart progress will edit. */
|
|
11633
|
+
async postSelfRestartNotice() {
|
|
11634
|
+
const groupId = this.fleetConfig?.channel?.group_id;
|
|
11635
|
+
const adapter = this.adapter;
|
|
11636
|
+
if (!groupId || !adapter)
|
|
11637
|
+
return null;
|
|
11638
|
+
const generalName = this.findGeneralInstance();
|
|
11639
|
+
const rawThreadId = generalName ? this.fleetConfig?.instances[generalName]?.topic_id : undefined;
|
|
11640
|
+
const threadId = rawThreadId != null ? String(rawThreadId) : undefined;
|
|
11641
|
+
try {
|
|
11642
|
+
const sent = await adapter.sendText(String(groupId), t("restart.settings_triggered"), { threadId });
|
|
11643
|
+
if (!sent?.messageId)
|
|
11644
|
+
return null;
|
|
11645
|
+
return { adapter, chatId: sent.chatId, threadId: sent.threadId, messageId: sent.messageId };
|
|
11646
|
+
}
|
|
11647
|
+
catch (err) {
|
|
11648
|
+
this.logger.error({ err }, "Could not announce the Settings-triggered restart — refusing to restart silently");
|
|
11649
|
+
return null;
|
|
11650
|
+
}
|
|
11651
|
+
}
|
|
11652
|
+
/** Jobs outlive this process on purpose; see apply-job.ts. */
|
|
11653
|
+
get applyJobs() {
|
|
11654
|
+
return (this.applyJobStoreCache ??= new ApplyJobStore(this.dataDir, Date.now, this.logger));
|
|
11655
|
+
}
|
|
11656
|
+
/** The only channel metadata exposed to the Settings secret UI. */
|
|
11657
|
+
listSecureConnections() {
|
|
11658
|
+
const channels = this.fleetConfig?.channels
|
|
11659
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []);
|
|
11660
|
+
return channels.map((channel, index) => {
|
|
11661
|
+
const id = channel.id ?? channel.type ?? `channel-${index}`;
|
|
11662
|
+
const world = this.worlds.get(id);
|
|
11663
|
+
const state = this.adapterState.get(id);
|
|
11664
|
+
return {
|
|
11665
|
+
id,
|
|
11666
|
+
type: channel.type,
|
|
11667
|
+
token_env: channel.bot_token_env,
|
|
11668
|
+
token_present: !!process.env[channel.bot_token_env],
|
|
11669
|
+
group_id: channel.group_id != null ? String(channel.group_id) : null,
|
|
11670
|
+
general_channel_id: channel.options?.general_channel_id != null
|
|
11671
|
+
? String(channel.options.general_channel_id)
|
|
11672
|
+
: null,
|
|
11673
|
+
status: state?.status ?? (world ? "starting" : "stopped"),
|
|
11674
|
+
...(world ? { identity: { id: world.botUserId ?? null, username: world.botUsername ?? null } } : {}),
|
|
11675
|
+
};
|
|
11676
|
+
});
|
|
11677
|
+
}
|
|
11678
|
+
secureConnectionChannel(connectionId) {
|
|
11679
|
+
const channels = this.fleetConfig?.channels
|
|
11680
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []);
|
|
11681
|
+
const matches = channels.filter((channel, index) => (channel.id ?? channel.type ?? `channel-${index}`) === connectionId);
|
|
11682
|
+
// Ambiguous fallback IDs (for example two unlabelled Discord channels)
|
|
11683
|
+
// must fail closed rather than rotating the first matching token.
|
|
11684
|
+
return matches.length === 1 ? matches[0] : undefined;
|
|
11685
|
+
}
|
|
11686
|
+
secureConnectionGeneration(connectionId) {
|
|
11687
|
+
const adapter = this.adapters.get(connectionId);
|
|
11688
|
+
const healthGeneration = adapter?.getHealthSnapshot?.().generation ?? 0;
|
|
11689
|
+
let generation = this.connectionSecretGenerations.get(connectionId) ?? 0;
|
|
11690
|
+
const previousAdapter = this.connectionSecretAdapterRefs.get(connectionId);
|
|
11691
|
+
const previousHealthGeneration = this.connectionSecretHealthGenerations.get(connectionId);
|
|
11692
|
+
if (this.connectionSecretAdapterRefs.has(connectionId) && previousAdapter !== adapter)
|
|
11693
|
+
generation++;
|
|
11694
|
+
if (this.connectionSecretHealthGenerations.has(connectionId) && previousHealthGeneration !== healthGeneration)
|
|
11695
|
+
generation++;
|
|
11696
|
+
this.connectionSecretAdapterRefs.set(connectionId, adapter);
|
|
11697
|
+
this.connectionSecretHealthGenerations.set(connectionId, healthGeneration);
|
|
11698
|
+
this.connectionSecretGenerations.set(connectionId, generation);
|
|
11699
|
+
return generation;
|
|
11700
|
+
}
|
|
11701
|
+
/** Provider API-key rows exposed to Settings (never the env key or secret). */
|
|
11702
|
+
providerSecretsEnabled() {
|
|
11703
|
+
return this.fleetConfig?.web?.provider_secrets === true;
|
|
11704
|
+
}
|
|
11705
|
+
listProviderSecrets() {
|
|
11706
|
+
return PROVIDER_SECRET_SPECS.map(spec => ({
|
|
11707
|
+
id: spec.id,
|
|
11708
|
+
display_name: spec.displayName,
|
|
11709
|
+
kind: spec.kind,
|
|
11710
|
+
token_present: !!process.env[spec.envKey],
|
|
11711
|
+
verifier: spec.verifier ? "available" : "unsupported",
|
|
11712
|
+
activation: spec.activation,
|
|
11713
|
+
stale_consumers: this.providerSecretStaleConsumers(spec.envKey),
|
|
11714
|
+
}));
|
|
11715
|
+
}
|
|
11716
|
+
providerSecretStaleConsumers(envKey) {
|
|
11717
|
+
// A child inherits the manager's environment at spawn. We cannot inspect
|
|
11718
|
+
// a child process's private environment safely, so report the conservative
|
|
11719
|
+
// set of already-running children; the UI can then say "restart these"
|
|
11720
|
+
// rather than claiming an existing process reloaded.
|
|
11721
|
+
if (envKey === "GROQ_API_KEY")
|
|
11722
|
+
return [];
|
|
11723
|
+
return [...this.children.keys()].sort();
|
|
11724
|
+
}
|
|
11725
|
+
providerSecretGeneration(envKey) {
|
|
11726
|
+
return this.providerSecretGenerations.get(envKey) ?? 0;
|
|
11727
|
+
}
|
|
11728
|
+
providerSecretEnvAllowed(envKey) {
|
|
11729
|
+
const configured = new Set((this.fleetConfig?.channels
|
|
11730
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []))
|
|
11731
|
+
.map(channel => channel.bot_token_env));
|
|
11732
|
+
return !configured.has(envKey) && providerRegistryEnvKeys().has(envKey);
|
|
11733
|
+
}
|
|
11734
|
+
async verifyProviderSecret(input) {
|
|
11735
|
+
const spec = providerSecretSpec(input.specId);
|
|
11736
|
+
if (!spec || !this.providerSecretEnvAllowed(spec.envKey)) {
|
|
11737
|
+
return { ok: false, status: "invalid", error: "provider secret is not configured" };
|
|
11738
|
+
}
|
|
11739
|
+
if (!spec.verifier)
|
|
11740
|
+
return { ok: false, status: "unsupported_verifier", error: "this provider has no supported verifier" };
|
|
11741
|
+
if (!input.secret || input.secret.length > 4096 || /[\r\n\0]/.test(input.secret) || /[^\x20-\x7e]/.test(input.secret)) {
|
|
11742
|
+
return { ok: false, status: "invalid", error: "secret is invalid" };
|
|
11743
|
+
}
|
|
11744
|
+
const challengeKey = `${input.sessionBinding}:api_key:${spec.id}:${spec.envKey}:${input.idempotencyKey}`;
|
|
11745
|
+
const existingId = this.providerSecretChallengesByKey.get(challengeKey);
|
|
11746
|
+
const existing = existingId ? this.providerSecretChallenges.get(existingId) : undefined;
|
|
11747
|
+
if (existing && existing.expiresAt > Date.now()) {
|
|
11748
|
+
return { ok: true, verification_id: existing.id, expires_at: existing.expiresAt, spec_id: spec.id, activation: spec.activation };
|
|
11749
|
+
}
|
|
11750
|
+
if (existingId)
|
|
11751
|
+
this.providerSecretChallengesByKey.delete(challengeKey);
|
|
11752
|
+
const result = await verifyProviderSecret(spec, input.secret, this.providerSecretHttpClient);
|
|
11753
|
+
if (!result.ok) {
|
|
11754
|
+
// Do not log provider detail: the HTTP verifier already redacted it and
|
|
11755
|
+
// this endpoint has no need to disclose whether a key was close to valid.
|
|
11756
|
+
this.logger.warn({ specId: spec.id, status: result.status }, "Provider API-key verification failed");
|
|
11757
|
+
return { ok: false, status: result.status, error: result.status === "unsupported_verifier" ? "this provider has no supported verifier" : "provider rejected or unavailable" };
|
|
11758
|
+
}
|
|
11759
|
+
const expiresAt = Date.now() + SECRET_CHALLENGE_TTL_MS;
|
|
11760
|
+
const challenge = {
|
|
11761
|
+
id: opaqueId("provider_verify"),
|
|
11762
|
+
specId: spec.id,
|
|
11763
|
+
envKey: spec.envKey,
|
|
11764
|
+
kind: "api_key",
|
|
11765
|
+
sessionBinding: input.sessionBinding,
|
|
11766
|
+
generation: this.providerSecretGeneration(spec.envKey),
|
|
11767
|
+
operation: "provider-secret.apply",
|
|
11768
|
+
idempotencyKey: input.idempotencyKey,
|
|
11769
|
+
expiresAt,
|
|
11770
|
+
secret: input.secret,
|
|
11771
|
+
};
|
|
11772
|
+
this.providerSecretChallenges.set(challenge.id, challenge);
|
|
11773
|
+
this.providerSecretChallengesByKey.set(challengeKey, challenge.id);
|
|
11774
|
+
const expiryTimer = setTimeout(() => {
|
|
11775
|
+
if (this.providerSecretChallenges.get(challenge.id) !== challenge)
|
|
11776
|
+
return;
|
|
11777
|
+
this.providerSecretChallenges.delete(challenge.id);
|
|
11778
|
+
if (this.providerSecretChallengesByKey.get(challengeKey) === challenge.id)
|
|
11779
|
+
this.providerSecretChallengesByKey.delete(challengeKey);
|
|
11780
|
+
}, SECRET_CHALLENGE_TTL_MS);
|
|
11781
|
+
expiryTimer.unref?.();
|
|
11782
|
+
return { ok: true, verification_id: challenge.id, expires_at: expiresAt, spec_id: spec.id, activation: spec.activation };
|
|
11783
|
+
}
|
|
11784
|
+
/** Naming aliases used by integrations that call this an API-key operation. */
|
|
11785
|
+
verifyProviderApiKey(input) {
|
|
11786
|
+
return this.verifyProviderSecret(input);
|
|
11787
|
+
}
|
|
11788
|
+
startProviderSecretApply(input) {
|
|
11789
|
+
for (const [jobId, job] of this.providerSecretJobs) {
|
|
11790
|
+
if (job.specId === input.specId && job.idempotencyKey === input.idempotencyKey
|
|
11791
|
+
&& this.providerSecretJobSession.get(jobId) === input.sessionBinding)
|
|
11792
|
+
return { job, reused: true };
|
|
11793
|
+
}
|
|
11794
|
+
const challenge = this.providerSecretChallenges.get(input.verificationId);
|
|
11795
|
+
const spec = providerSecretSpec(input.specId);
|
|
11796
|
+
if (!challenge || challenge.expiresAt <= Date.now()) {
|
|
11797
|
+
if (challenge)
|
|
11798
|
+
this.providerSecretChallenges.delete(input.verificationId);
|
|
11799
|
+
return { error: "verification expired; verify the secret again" };
|
|
11800
|
+
}
|
|
11801
|
+
if (!spec || challenge.specId !== spec.id || challenge.envKey !== spec.envKey || challenge.kind !== "api_key"
|
|
11802
|
+
|| challenge.sessionBinding !== input.sessionBinding || challenge.operation !== "provider-secret.apply"
|
|
11803
|
+
|| challenge.idempotencyKey !== input.idempotencyKey || challenge.generation !== this.providerSecretGeneration(challenge.envKey)) {
|
|
11804
|
+
return { error: "verification does not match this provider or session" };
|
|
11805
|
+
}
|
|
11806
|
+
const existingId = this.providerSecretInFlight.get(challenge.envKey);
|
|
11807
|
+
if (existingId) {
|
|
11808
|
+
const existing = this.providerSecretJobs.get(existingId) ?? null;
|
|
11809
|
+
if (existing?.idempotencyKey === input.idempotencyKey)
|
|
11810
|
+
return { job: existing, reused: true };
|
|
11811
|
+
return { busy: existing };
|
|
11812
|
+
}
|
|
11813
|
+
this.providerSecretChallenges.delete(input.verificationId);
|
|
11814
|
+
this.providerSecretChallengesByKey.delete(`${input.sessionBinding}:api_key:${spec.id}:${spec.envKey}:${input.idempotencyKey}`);
|
|
11815
|
+
const job = {
|
|
11816
|
+
id: opaqueId("provider_apply"), specId: spec.id, envKey: spec.envKey,
|
|
11817
|
+
idempotencyKey: input.idempotencyKey, result: "applying", status: "running", startedAt: Date.now(),
|
|
11818
|
+
stale_consumers: this.providerSecretStaleConsumers(spec.envKey),
|
|
11819
|
+
};
|
|
11820
|
+
this.providerSecretJobs.set(job.id, job);
|
|
11821
|
+
this.providerSecretJobSession.set(job.id, input.sessionBinding);
|
|
11822
|
+
this.providerSecretInFlight.set(spec.envKey, job.id);
|
|
11823
|
+
queueMicrotask(() => void this.runProviderSecretApply(job, challenge.secret));
|
|
11824
|
+
return { job, reused: false };
|
|
11825
|
+
}
|
|
11826
|
+
startProviderApiKeyApply(input) {
|
|
11827
|
+
return this.startProviderSecretApply(input);
|
|
11828
|
+
}
|
|
11829
|
+
getProviderSecretApply(jobId, sessionBinding) {
|
|
11830
|
+
if (this.providerSecretJobSession.get(jobId) !== sessionBinding)
|
|
11831
|
+
return null;
|
|
11832
|
+
return this.providerSecretJobs.get(jobId) ?? null;
|
|
11833
|
+
}
|
|
11834
|
+
getProviderApiKeyApply(jobId, sessionBinding) {
|
|
11835
|
+
return this.getProviderSecretApply(jobId, sessionBinding);
|
|
11836
|
+
}
|
|
11837
|
+
async runProviderSecretApply(job, secret) {
|
|
11838
|
+
const spec = providerSecretSpec(job.specId);
|
|
11839
|
+
const allowed = new Set([
|
|
11840
|
+
...PROVIDER_SECRET_SPECS.map(item => item.envKey),
|
|
11841
|
+
...(this.fleetConfig?.channels ?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : [])).map(channel => channel.bot_token_env),
|
|
11842
|
+
]);
|
|
11843
|
+
let store = null;
|
|
11844
|
+
let before = null;
|
|
11845
|
+
const previousProcessValue = process.env[job.envKey];
|
|
11846
|
+
let wrote = false;
|
|
11847
|
+
try {
|
|
11848
|
+
if (!spec || !this.providerSecretEnvAllowed(job.envKey))
|
|
11849
|
+
throw new Error("provider secret is not configured");
|
|
11850
|
+
// Construct inside the transaction: symlink/permission refusal must
|
|
11851
|
+
// settle the job as a safe failure, not escape the queued microtask.
|
|
11852
|
+
store = new SecretStore(join(this.dataDir, ".env"), allowed);
|
|
11853
|
+
before = store.write(job.envKey, secret);
|
|
11854
|
+
wrote = true;
|
|
11855
|
+
process.env[job.envKey] = secret;
|
|
11856
|
+
if (spec.activation === "reload_hook" && spec.reloadHookId) {
|
|
11857
|
+
await this.runProviderSecretReloadHook(spec.reloadHookId, secret, previousProcessValue);
|
|
11858
|
+
job.result = "reloaded";
|
|
11859
|
+
}
|
|
11860
|
+
else {
|
|
11861
|
+
job.result = "applied_next_use";
|
|
11862
|
+
}
|
|
11863
|
+
this.providerSecretGenerations.set(job.envKey, this.providerSecretGeneration(job.envKey) + 1);
|
|
11864
|
+
job.status = "done";
|
|
11865
|
+
job.finishedAt = Date.now();
|
|
11866
|
+
}
|
|
11867
|
+
catch (err) {
|
|
11868
|
+
const safe = safeSecretError(err, secret);
|
|
11869
|
+
this.logger.warn({ specId: job.specId, reason: safe }, "Provider API-key apply failed");
|
|
11870
|
+
try {
|
|
11871
|
+
if (wrote && before && store)
|
|
11872
|
+
store.restore(before);
|
|
11873
|
+
if (previousProcessValue === undefined)
|
|
11874
|
+
delete process.env[job.envKey];
|
|
11875
|
+
else
|
|
11876
|
+
process.env[job.envKey] = previousProcessValue;
|
|
11877
|
+
// SecretStore.write is itself transactional; when it fails before a
|
|
11878
|
+
// snapshot is returned there is no new value to roll back. Report the
|
|
11879
|
+
// truthful no-op rather than claiming rollback_failed.
|
|
11880
|
+
job.result = "rolled_back";
|
|
11881
|
+
if (!wrote || !before)
|
|
11882
|
+
job.error = "provider secret was not applied";
|
|
11883
|
+
}
|
|
11884
|
+
catch (rollbackErr) {
|
|
11885
|
+
this.logger.error({ specId: job.specId, reason: safeSecretError(rollbackErr, secret, previousProcessValue ? [previousProcessValue] : []) }, "Provider API-key rollback failed");
|
|
11886
|
+
job.result = "rollback_failed";
|
|
11887
|
+
job.error = "provider secret rollback failed; operator attention required";
|
|
11888
|
+
}
|
|
11889
|
+
job.status = "done";
|
|
11890
|
+
job.finishedAt = Date.now();
|
|
11891
|
+
}
|
|
11892
|
+
finally {
|
|
11893
|
+
this.providerSecretInFlight.delete(job.envKey);
|
|
11894
|
+
secret = "";
|
|
11895
|
+
}
|
|
11896
|
+
}
|
|
11897
|
+
/** Groq is currently read from process.env per voice request, so the hook is
|
|
11898
|
+
* intentionally a no-op. Keeping it as a named code-owned hook makes the
|
|
11899
|
+
* hot activation contract explicit and gives tests a failure seam; no generic
|
|
11900
|
+
* SIGHUP or caller-provided hook is ever executed. */
|
|
11901
|
+
async runProviderSecretReloadHook(hookId, _next, _previous) {
|
|
11902
|
+
if (hookId !== "groq.voice")
|
|
11903
|
+
throw new Error("unknown provider secret reload hook");
|
|
11904
|
+
const before = this.providerSecretHotSnapshots.get(hookId);
|
|
11905
|
+
this.providerSecretHotSnapshots.set(hookId, _next);
|
|
11906
|
+
const hook = this.providerSecretReloadHooks.get(hookId);
|
|
11907
|
+
try {
|
|
11908
|
+
if (hook)
|
|
11909
|
+
await hook(_next, before);
|
|
11910
|
+
}
|
|
11911
|
+
catch (err) {
|
|
11912
|
+
if (before === undefined)
|
|
11913
|
+
this.providerSecretHotSnapshots.delete(hookId);
|
|
11914
|
+
else
|
|
11915
|
+
this.providerSecretHotSnapshots.set(hookId, before);
|
|
11916
|
+
throw err;
|
|
11917
|
+
}
|
|
11918
|
+
}
|
|
11919
|
+
normalizeConnectionBinding(input) {
|
|
11920
|
+
// IDs arrive from JSON and may be Discord snowflakes. Do not accept a
|
|
11921
|
+
// number here: JSON.parse may already have rounded it before verification.
|
|
11922
|
+
if (typeof input.group_id !== "string")
|
|
11923
|
+
return null;
|
|
11924
|
+
const groupId = input.group_id.trim();
|
|
11925
|
+
if (!groupId || groupId.length > 128 || /[\r\n\0]/.test(groupId))
|
|
11926
|
+
return null;
|
|
11927
|
+
let general;
|
|
11928
|
+
if (input.general_channel_id === null || input.general_channel_id === undefined || input.general_channel_id === "") {
|
|
11929
|
+
general = input.general_channel_id === null ? null : undefined;
|
|
11930
|
+
}
|
|
11931
|
+
else if (typeof input.general_channel_id === "string") {
|
|
11932
|
+
general = input.general_channel_id.trim();
|
|
11933
|
+
if (!general || general.length > 128 || /[\r\n\0]/.test(general))
|
|
11934
|
+
return null;
|
|
11935
|
+
}
|
|
11936
|
+
else {
|
|
11937
|
+
return null;
|
|
11938
|
+
}
|
|
11939
|
+
return general === undefined ? { group_id: groupId } : { group_id: groupId, general_channel_id: general };
|
|
11940
|
+
}
|
|
11941
|
+
connectionBindingChannelConfig(channel, binding) {
|
|
11942
|
+
const candidate = structuredClone(channel);
|
|
11943
|
+
// IDs are intentionally normalized to strings at this boundary. Discord
|
|
11944
|
+
// snowflakes must never become YAML numbers (precision loss is silent).
|
|
11945
|
+
candidate.group_id = String(binding.group_id);
|
|
11946
|
+
if (binding.general_channel_id !== undefined) {
|
|
11947
|
+
const options = { ...(candidate.options ?? {}) };
|
|
11948
|
+
if (binding.general_channel_id === null)
|
|
11949
|
+
delete options.general_channel_id;
|
|
11950
|
+
else
|
|
11951
|
+
options.general_channel_id = String(binding.general_channel_id);
|
|
11952
|
+
if (Object.keys(options).length === 0)
|
|
11953
|
+
delete candidate.options;
|
|
11954
|
+
else
|
|
11955
|
+
candidate.options = options;
|
|
11956
|
+
}
|
|
11957
|
+
return candidate;
|
|
11958
|
+
}
|
|
11959
|
+
/** Verify a prospective group/guild binding without mutating fleet state. */
|
|
11960
|
+
async verifyConnectionBinding(input) {
|
|
11961
|
+
const channel = this.secureConnectionChannel(input.connectionId);
|
|
11962
|
+
const binding = this.normalizeConnectionBinding(input.binding);
|
|
11963
|
+
const adapter = this.adapters.get(input.connectionId);
|
|
11964
|
+
if (!channel || !binding || !adapter?.verifyBinding) {
|
|
11965
|
+
return { ok: false, error: "connection binding is unsupported or invalid" };
|
|
11966
|
+
}
|
|
11967
|
+
const key = `${input.sessionBinding}:${input.connectionId}:${input.idempotencyKey}`;
|
|
11968
|
+
const existingId = this.connectionBindingChallengesByKey.get(key);
|
|
11969
|
+
const existing = existingId ? this.connectionBindingChallenges.get(existingId) : undefined;
|
|
11970
|
+
if (existing && existing.expiresAt > Date.now()) {
|
|
11971
|
+
return { ok: true, verification_id: existing.id, expires_at: existing.expiresAt, binding: existing.binding, probe: existing.probe };
|
|
11972
|
+
}
|
|
11973
|
+
if (existingId)
|
|
11974
|
+
this.connectionBindingChallengesByKey.delete(key);
|
|
11975
|
+
const beforeGeneration = this.secureConnectionGeneration(input.connectionId);
|
|
11976
|
+
let probe;
|
|
11977
|
+
try {
|
|
11978
|
+
probe = await adapter.verifyBinding(binding.group_id, binding.general_channel_id ?? undefined);
|
|
11979
|
+
}
|
|
11980
|
+
catch (err) {
|
|
11981
|
+
this.logger.warn({ connectionId: input.connectionId, reason: safeSecretError(err) }, "Settings connection binding verification failed");
|
|
11982
|
+
return { ok: false, error: "binding verification failed" };
|
|
11983
|
+
}
|
|
11984
|
+
const afterGeneration = this.secureConnectionGeneration(input.connectionId);
|
|
11985
|
+
if (beforeGeneration !== afterGeneration || this.adapters.get(input.connectionId) !== adapter) {
|
|
11986
|
+
return { ok: false, error: "connection changed while binding was verified" };
|
|
11987
|
+
}
|
|
11988
|
+
if (probe.group_id !== binding.group_id || !probe.can_view || !probe.can_send) {
|
|
11989
|
+
return { ok: false, error: "provider did not confirm the requested binding" };
|
|
11990
|
+
}
|
|
11991
|
+
const expiresAt = Date.now() + SECRET_CHALLENGE_TTL_MS;
|
|
11992
|
+
const challenge = {
|
|
11993
|
+
id: opaqueId("binding_verify"),
|
|
11994
|
+
connectionId: input.connectionId,
|
|
11995
|
+
sessionBinding: input.sessionBinding,
|
|
11996
|
+
generation: afterGeneration,
|
|
11997
|
+
operation: "binding.apply",
|
|
11998
|
+
idempotencyKey: input.idempotencyKey,
|
|
11999
|
+
expiresAt,
|
|
12000
|
+
binding,
|
|
12001
|
+
probe,
|
|
12002
|
+
};
|
|
12003
|
+
this.connectionBindingChallenges.set(challenge.id, challenge);
|
|
12004
|
+
this.connectionBindingChallengesByKey.set(key, challenge.id);
|
|
12005
|
+
const expiryTimer = setTimeout(() => {
|
|
12006
|
+
if (this.connectionBindingChallenges.get(challenge.id) !== challenge)
|
|
12007
|
+
return;
|
|
12008
|
+
this.connectionBindingChallenges.delete(challenge.id);
|
|
12009
|
+
if (this.connectionBindingChallengesByKey.get(key) === challenge.id)
|
|
12010
|
+
this.connectionBindingChallengesByKey.delete(key);
|
|
12011
|
+
}, SECRET_CHALLENGE_TTL_MS);
|
|
12012
|
+
expiryTimer.unref?.();
|
|
12013
|
+
return { ok: true, verification_id: challenge.id, expires_at: expiresAt, binding, probe };
|
|
12014
|
+
}
|
|
12015
|
+
startConnectionBindingApply(input) {
|
|
12016
|
+
for (const [jobId, job] of this.connectionBindingJobs) {
|
|
12017
|
+
if (job.connectionId === input.connectionId && job.idempotencyKey === input.idempotencyKey
|
|
12018
|
+
&& this.connectionBindingJobSession.get(jobId) === input.sessionBinding)
|
|
12019
|
+
return { job, reused: true };
|
|
12020
|
+
}
|
|
12021
|
+
const challenge = this.connectionBindingChallenges.get(input.verificationId);
|
|
12022
|
+
if (!challenge || challenge.expiresAt <= Date.now()) {
|
|
12023
|
+
this.connectionBindingChallenges.delete(input.verificationId);
|
|
12024
|
+
return { error: "binding verification expired; verify the binding again" };
|
|
12025
|
+
}
|
|
12026
|
+
if (challenge.connectionId !== input.connectionId || challenge.sessionBinding !== input.sessionBinding
|
|
12027
|
+
|| challenge.operation !== "binding.apply" || challenge.idempotencyKey !== input.idempotencyKey
|
|
12028
|
+
|| challenge.generation !== this.secureConnectionGeneration(input.connectionId)) {
|
|
12029
|
+
return { error: "binding verification does not match this connection or session" };
|
|
12030
|
+
}
|
|
12031
|
+
const existingId = this.connectionBindingInFlight.get(input.connectionId);
|
|
12032
|
+
if (existingId) {
|
|
12033
|
+
const existing = this.connectionBindingJobs.get(existingId) ?? null;
|
|
12034
|
+
if (existing?.idempotencyKey === input.idempotencyKey)
|
|
12035
|
+
return { job: existing, reused: true };
|
|
12036
|
+
return { busy: existing };
|
|
12037
|
+
}
|
|
12038
|
+
this.connectionBindingChallenges.delete(input.verificationId);
|
|
12039
|
+
this.connectionBindingChallengesByKey.delete(`${input.sessionBinding}:${input.connectionId}:${input.idempotencyKey}`);
|
|
12040
|
+
const job = {
|
|
12041
|
+
id: opaqueId("binding_apply"), connectionId: input.connectionId, idempotencyKey: input.idempotencyKey,
|
|
12042
|
+
result: "applying", status: "running", startedAt: Date.now(),
|
|
12043
|
+
};
|
|
12044
|
+
this.connectionBindingJobs.set(job.id, job);
|
|
12045
|
+
this.connectionBindingJobSession.set(job.id, input.sessionBinding);
|
|
12046
|
+
this.connectionBindingInFlight.set(input.connectionId, job.id);
|
|
12047
|
+
queueMicrotask(() => void this.runConnectionBindingApply(job, challenge.binding));
|
|
12048
|
+
return { job, reused: false };
|
|
12049
|
+
}
|
|
12050
|
+
getConnectionBindingApply(jobId, sessionBinding) {
|
|
12051
|
+
if (this.connectionBindingJobSession.get(jobId) !== sessionBinding)
|
|
12052
|
+
return null;
|
|
12053
|
+
return this.connectionBindingJobs.get(jobId) ?? null;
|
|
12054
|
+
}
|
|
12055
|
+
async runConnectionBindingApply(job, binding) {
|
|
12056
|
+
try {
|
|
12057
|
+
await this.rebuildAdapterForBinding(job.connectionId, binding);
|
|
12058
|
+
job.result = "applied";
|
|
12059
|
+
}
|
|
12060
|
+
catch (err) {
|
|
12061
|
+
const reason = safeSecretError(err);
|
|
12062
|
+
job.result = /rollback failed/i.test(reason) ? "rollback_failed" : "rolled_back";
|
|
12063
|
+
job.error = job.result === "rollback_failed"
|
|
12064
|
+
? "binding rollback failed; adapter requires operator attention"
|
|
12065
|
+
: "binding was not applied; previous binding was restored";
|
|
12066
|
+
this.logger.warn({ connectionId: job.connectionId, reason: safeSecretError(err) }, "Settings connection binding apply failed");
|
|
12067
|
+
}
|
|
12068
|
+
finally {
|
|
12069
|
+
job.status = "done";
|
|
12070
|
+
job.finishedAt = Date.now();
|
|
12071
|
+
this.connectionBindingInFlight.delete(job.connectionId);
|
|
12072
|
+
}
|
|
12073
|
+
}
|
|
12074
|
+
/** Stop, rebuild and wait for a new adapter before committing YAML binding. */
|
|
12075
|
+
async rebuildAdapterForBinding(connectionId, binding) {
|
|
12076
|
+
const channel = this.secureConnectionChannel(connectionId);
|
|
12077
|
+
if (!channel || !this.fleetConfig)
|
|
12078
|
+
throw new Error("connection not found");
|
|
12079
|
+
const candidate = this.connectionBindingChannelConfig(channel, binding);
|
|
12080
|
+
const oldAdapter = this.adapters.get(connectionId);
|
|
12081
|
+
const oldWorld = this.worlds.get(connectionId);
|
|
12082
|
+
const oldPrimary = this.adapter;
|
|
12083
|
+
const oldAccess = this.accessManager;
|
|
12084
|
+
const oldState = this.adapterState.get(connectionId);
|
|
12085
|
+
const oldChannel = structuredClone(channel);
|
|
12086
|
+
const primary = this.getPrimaryAdapterId() === connectionId;
|
|
12087
|
+
if (primary && this.sessionPruneTimer) {
|
|
12088
|
+
clearInterval(this.sessionPruneTimer);
|
|
12089
|
+
this.sessionPruneTimer = null;
|
|
12090
|
+
}
|
|
12091
|
+
let fresh;
|
|
12092
|
+
let persistedBinding = false;
|
|
12093
|
+
try {
|
|
12094
|
+
this.adapterState.set(connectionId, { status: "retrying", retryCount: oldState?.retryCount ?? 0 });
|
|
12095
|
+
if (oldAdapter) {
|
|
12096
|
+
oldAdapter.removeAllListeners();
|
|
12097
|
+
await oldAdapter.stop().catch(() => { });
|
|
12098
|
+
if (this.adapters.get(connectionId) === oldAdapter)
|
|
12099
|
+
this.adapters.delete(connectionId);
|
|
12100
|
+
if (this.worlds.get(connectionId)?.adapter === oldAdapter)
|
|
12101
|
+
this.worlds.delete(connectionId);
|
|
12102
|
+
if (primary && this.adapter === oldAdapter)
|
|
12103
|
+
this.adapter = null;
|
|
12104
|
+
}
|
|
12105
|
+
let startedResolve = null;
|
|
12106
|
+
const started = new Promise(resolve => { startedResolve = resolve; });
|
|
12107
|
+
const onStarted = () => { startedResolve?.(); };
|
|
12108
|
+
if (primary)
|
|
12109
|
+
await this.startSingleAdapter(this.fleetConfig, candidate, onStarted);
|
|
12110
|
+
else
|
|
12111
|
+
await this.startAdditionalAdapter(candidate, true, onStarted);
|
|
12112
|
+
fresh = this.adapters.get(connectionId);
|
|
12113
|
+
if (!fresh)
|
|
12114
|
+
throw new Error("new adapter did not start");
|
|
12115
|
+
const deadline = Date.now() + 15_000;
|
|
12116
|
+
if (!fresh.getHealthSnapshot) {
|
|
12117
|
+
let timer;
|
|
12118
|
+
const timeout = new Promise((_, reject) => {
|
|
12119
|
+
timer = setTimeout(() => reject(new Error("new adapter did not become ready")), Math.max(1, deadline - Date.now()));
|
|
12120
|
+
timer.unref?.();
|
|
12121
|
+
});
|
|
12122
|
+
try {
|
|
12123
|
+
await Promise.race([started, timeout]);
|
|
12124
|
+
}
|
|
12125
|
+
finally {
|
|
12126
|
+
if (timer)
|
|
12127
|
+
clearTimeout(timer);
|
|
12128
|
+
}
|
|
12129
|
+
}
|
|
12130
|
+
else {
|
|
12131
|
+
while (Date.now() < deadline) {
|
|
12132
|
+
const health = fresh.getHealthSnapshot?.();
|
|
12133
|
+
if (health?.status === "connected" || this.adapterState.get(connectionId)?.status === "connected")
|
|
12134
|
+
break;
|
|
12135
|
+
await new Promise(resolve => setTimeout(resolve, 100));
|
|
12136
|
+
}
|
|
12137
|
+
const health = fresh.getHealthSnapshot?.();
|
|
12138
|
+
if (health && health.status !== "connected" && this.adapterState.get(connectionId)?.status !== "connected") {
|
|
12139
|
+
throw new Error("new adapter did not become connected");
|
|
12140
|
+
}
|
|
12141
|
+
}
|
|
12142
|
+
if (fresh.setChatId)
|
|
12143
|
+
fresh.setChatId(String(candidate.group_id));
|
|
12144
|
+
// Commit only after the replacement adapter is ready. No allowlist,
|
|
12145
|
+
// topic, instance or schedule fields are touched here.
|
|
12146
|
+
channel.group_id = String(binding.group_id);
|
|
12147
|
+
if (binding.general_channel_id !== undefined) {
|
|
12148
|
+
const options = { ...(channel.options ?? {}) };
|
|
12149
|
+
if (binding.general_channel_id === null)
|
|
12150
|
+
delete options.general_channel_id;
|
|
12151
|
+
else
|
|
12152
|
+
options.general_channel_id = String(binding.general_channel_id);
|
|
12153
|
+
if (Object.keys(options).length === 0)
|
|
12154
|
+
delete channel.options;
|
|
12155
|
+
else
|
|
12156
|
+
channel.options = options;
|
|
12157
|
+
}
|
|
12158
|
+
this.saveFleetConfig();
|
|
12159
|
+
persistedBinding = true;
|
|
12160
|
+
this.routing.rebuild(this.fleetConfig);
|
|
12161
|
+
this.reregisterClassicChannels();
|
|
12162
|
+
this.adapterState.set(connectionId, { status: "connected", retryCount: 0 });
|
|
12163
|
+
}
|
|
12164
|
+
catch (err) {
|
|
12165
|
+
if (fresh && fresh !== oldAdapter)
|
|
12166
|
+
await fresh.stop().catch(() => { });
|
|
12167
|
+
// Restore only the binding object in memory; unrelated connection and
|
|
12168
|
+
// instance state remains exactly as it was before the attempt.
|
|
12169
|
+
for (const key of Object.keys(channel)) {
|
|
12170
|
+
if (!(key in oldChannel))
|
|
12171
|
+
delete channel[key];
|
|
12172
|
+
}
|
|
12173
|
+
Object.assign(channel, oldChannel);
|
|
12174
|
+
this.adapters.delete(connectionId);
|
|
12175
|
+
this.worlds.delete(connectionId);
|
|
12176
|
+
this.adapterState.delete(connectionId);
|
|
12177
|
+
if (oldAdapter) {
|
|
12178
|
+
try {
|
|
12179
|
+
const onStarted = () => { };
|
|
12180
|
+
if (primary)
|
|
12181
|
+
await this.startSingleAdapter(this.fleetConfig, oldChannel, onStarted);
|
|
12182
|
+
else
|
|
12183
|
+
await this.startAdditionalAdapter(oldChannel, true, onStarted);
|
|
12184
|
+
this.adapterState.set(connectionId, oldState ?? { status: "connected", retryCount: 0 });
|
|
12185
|
+
}
|
|
12186
|
+
catch (restoreErr) {
|
|
12187
|
+
throw new Error(`binding rollback failed: ${safeSecretError(restoreErr)}`);
|
|
12188
|
+
}
|
|
12189
|
+
}
|
|
12190
|
+
else {
|
|
12191
|
+
if (primary)
|
|
12192
|
+
this.adapter = oldPrimary;
|
|
12193
|
+
if (oldWorld)
|
|
12194
|
+
this.worlds.set(connectionId, oldWorld);
|
|
12195
|
+
if (oldAdapter)
|
|
12196
|
+
this.adapters.set(connectionId, oldAdapter);
|
|
12197
|
+
this.accessManager = oldAccess;
|
|
12198
|
+
}
|
|
12199
|
+
// The binding is committed to YAML before routing is rebuilt. If the
|
|
12200
|
+
// post-commit rebuild fails, restore the durable document as well as the
|
|
12201
|
+
// in-memory channel; otherwise a reload would resurrect the failed
|
|
12202
|
+
// binding that the running fleet just rolled back.
|
|
12203
|
+
if (persistedBinding)
|
|
12204
|
+
this.saveFleetConfig();
|
|
12205
|
+
this.routing.rebuild(this.fleetConfig);
|
|
12206
|
+
this.reregisterClassicChannels();
|
|
12207
|
+
throw err;
|
|
12208
|
+
}
|
|
12209
|
+
}
|
|
12210
|
+
async verifyConnectionSecret(input) {
|
|
12211
|
+
const channel = this.secureConnectionChannel(input.connectionId);
|
|
12212
|
+
if (!channel || (channel.type !== "discord" && channel.type !== "telegram")) {
|
|
12213
|
+
return { ok: false, error: "connection not found or unsupported" };
|
|
12214
|
+
}
|
|
12215
|
+
if (!input.secret || input.secret.length > 4096 || /[\r\n\0]/.test(input.secret)) {
|
|
12216
|
+
return { ok: false, error: "secret is invalid" };
|
|
12217
|
+
}
|
|
12218
|
+
const challengeKey = `${input.sessionBinding}:${input.connectionId}:${input.idempotencyKey}`;
|
|
12219
|
+
const existingId = this.connectionSecretChallengesByKey.get(challengeKey);
|
|
12220
|
+
const existing = existingId ? this.connectionSecretChallenges.get(existingId) : undefined;
|
|
12221
|
+
if (existing && existing.expiresAt > Date.now()) {
|
|
12222
|
+
return { ok: true, verification_id: existing.id, expires_at: existing.expiresAt };
|
|
12223
|
+
}
|
|
12224
|
+
if (existingId)
|
|
12225
|
+
this.connectionSecretChallengesByKey.delete(challengeKey);
|
|
12226
|
+
// Fixed provider endpoints only. Never use a user-supplied URL and never
|
|
12227
|
+
// call Telegram getUpdates (the running adapter owns that long poll).
|
|
12228
|
+
const identity = channel.type === "discord"
|
|
12229
|
+
? await verifyDiscordToken(input.secret)
|
|
12230
|
+
: await verifyTelegramToken(input.secret);
|
|
12231
|
+
if (!identity.valid) {
|
|
12232
|
+
this.logger.warn({ connectionId: input.connectionId, provider: channel.type }, "Settings connection secret verification failed");
|
|
12233
|
+
return { ok: false, error: "provider rejected the secret" };
|
|
12234
|
+
}
|
|
12235
|
+
const expiresAt = Date.now() + SECRET_CHALLENGE_TTL_MS;
|
|
12236
|
+
const challenge = {
|
|
12237
|
+
id: opaqueId("verify"),
|
|
12238
|
+
connectionId: input.connectionId,
|
|
12239
|
+
sessionBinding: input.sessionBinding,
|
|
12240
|
+
generation: this.secureConnectionGeneration(input.connectionId),
|
|
12241
|
+
operation: "secret.apply",
|
|
12242
|
+
idempotencyKey: input.idempotencyKey,
|
|
12243
|
+
expiresAt,
|
|
12244
|
+
secret: input.secret,
|
|
12245
|
+
};
|
|
12246
|
+
this.connectionSecretChallenges.set(challenge.id, challenge);
|
|
12247
|
+
this.connectionSecretChallengesByKey.set(challengeKey, challenge.id);
|
|
12248
|
+
const expiryTimer = setTimeout(() => {
|
|
12249
|
+
if (this.connectionSecretChallenges.get(challenge.id) !== challenge)
|
|
12250
|
+
return;
|
|
12251
|
+
this.connectionSecretChallenges.delete(challenge.id);
|
|
12252
|
+
if (this.connectionSecretChallengesByKey.get(challengeKey) === challenge.id) {
|
|
12253
|
+
this.connectionSecretChallengesByKey.delete(challengeKey);
|
|
12254
|
+
}
|
|
12255
|
+
}, SECRET_CHALLENGE_TTL_MS);
|
|
12256
|
+
expiryTimer.unref?.();
|
|
12257
|
+
return {
|
|
12258
|
+
ok: true,
|
|
12259
|
+
verification_id: challenge.id,
|
|
12260
|
+
expires_at: expiresAt,
|
|
12261
|
+
identity: { id: identity.id, username: identity.username },
|
|
12262
|
+
};
|
|
12263
|
+
}
|
|
12264
|
+
startConnectionSecretApply(input) {
|
|
12265
|
+
for (const [jobId, job] of this.connectionSecretJobs) {
|
|
12266
|
+
if (job.connectionId === input.connectionId
|
|
12267
|
+
&& job.idempotencyKey === input.idempotencyKey
|
|
12268
|
+
&& this.connectionSecretJobSession.get(jobId) === input.sessionBinding) {
|
|
12269
|
+
return { job, reused: true };
|
|
12270
|
+
}
|
|
12271
|
+
}
|
|
12272
|
+
const challenge = this.connectionSecretChallenges.get(input.verificationId);
|
|
12273
|
+
if (!challenge || challenge.expiresAt <= Date.now()) {
|
|
12274
|
+
this.connectionSecretChallenges.delete(input.verificationId);
|
|
12275
|
+
return { error: "verification expired; verify the secret again" };
|
|
12276
|
+
}
|
|
12277
|
+
if (challenge.connectionId !== input.connectionId
|
|
12278
|
+
|| challenge.sessionBinding !== input.sessionBinding
|
|
12279
|
+
|| challenge.operation !== "secret.apply"
|
|
12280
|
+
|| challenge.idempotencyKey !== input.idempotencyKey
|
|
12281
|
+
|| challenge.generation !== this.secureConnectionGeneration(input.connectionId)) {
|
|
12282
|
+
return { error: "verification does not match this connection or session" };
|
|
12283
|
+
}
|
|
12284
|
+
const existingId = this.connectionSecretInFlight.get(input.connectionId);
|
|
12285
|
+
if (existingId) {
|
|
12286
|
+
const existing = this.connectionSecretJobs.get(existingId) ?? null;
|
|
12287
|
+
if (existing?.idempotencyKey === input.idempotencyKey)
|
|
12288
|
+
return { job: existing, reused: true };
|
|
12289
|
+
return { busy: existing };
|
|
12290
|
+
}
|
|
12291
|
+
// Consume the challenge before scheduling work. A lost HTTP response can
|
|
12292
|
+
// retry with the same idempotency key and rejoin the job, but a second
|
|
12293
|
+
// request cannot replay the secret into a second adapter.
|
|
12294
|
+
this.connectionSecretChallenges.delete(input.verificationId);
|
|
12295
|
+
this.connectionSecretChallengesByKey.delete(`${input.sessionBinding}:${input.connectionId}:${input.idempotencyKey}`);
|
|
12296
|
+
const job = {
|
|
12297
|
+
id: opaqueId("secret_apply"),
|
|
12298
|
+
connectionId: input.connectionId,
|
|
12299
|
+
idempotencyKey: input.idempotencyKey,
|
|
12300
|
+
result: "applying",
|
|
12301
|
+
status: "running",
|
|
12302
|
+
startedAt: Date.now(),
|
|
12303
|
+
};
|
|
12304
|
+
this.connectionSecretJobs.set(job.id, job);
|
|
12305
|
+
this.connectionSecretJobSession.set(job.id, input.sessionBinding);
|
|
12306
|
+
this.connectionSecretInFlight.set(input.connectionId, job.id);
|
|
12307
|
+
queueMicrotask(() => void this.runConnectionSecretApply(job, challenge.secret));
|
|
12308
|
+
return { job, reused: false };
|
|
12309
|
+
}
|
|
12310
|
+
getConnectionSecretApply(jobId, sessionBinding) {
|
|
12311
|
+
if (this.connectionSecretJobSession.get(jobId) !== sessionBinding)
|
|
12312
|
+
return null;
|
|
12313
|
+
return this.connectionSecretJobs.get(jobId) ?? null;
|
|
12314
|
+
}
|
|
12315
|
+
async runConnectionSecretApply(job, secret) {
|
|
12316
|
+
const channel = this.secureConnectionChannel(job.connectionId);
|
|
12317
|
+
const envKey = channel?.bot_token_env;
|
|
12318
|
+
const allowed = new Set((this.fleetConfig?.channels
|
|
12319
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []))
|
|
12320
|
+
.map(item => item.bot_token_env));
|
|
12321
|
+
let before = null;
|
|
12322
|
+
let oldToken;
|
|
12323
|
+
let replaced = false;
|
|
12324
|
+
try {
|
|
12325
|
+
if (!channel || !envKey || !allowed.has(envKey))
|
|
12326
|
+
throw new Error("connection is not configured for secret rotation");
|
|
12327
|
+
const owners = (this.fleetConfig?.channels
|
|
12328
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []))
|
|
12329
|
+
.filter(item => item.bot_token_env === envKey);
|
|
12330
|
+
if (owners.length !== 1)
|
|
12331
|
+
throw new Error("secret key is shared by multiple connections");
|
|
12332
|
+
const store = new SecretStore(join(this.dataDir, ".env"), allowed);
|
|
12333
|
+
before = store.write(envKey, secret);
|
|
12334
|
+
replaced = true;
|
|
12335
|
+
oldToken = process.env[envKey];
|
|
12336
|
+
process.env[envKey] = secret;
|
|
12337
|
+
const generation = this.secureConnectionGeneration(job.connectionId) + 1;
|
|
12338
|
+
this.connectionSecretGenerations.set(job.connectionId, generation);
|
|
12339
|
+
const applied = await this.rebuildAdapterForSecret(job.connectionId, channel);
|
|
12340
|
+
if (!applied) {
|
|
12341
|
+
job.result = "restart_required";
|
|
12342
|
+
job.status = "done";
|
|
12343
|
+
job.finishedAt = Date.now();
|
|
12344
|
+
return;
|
|
12345
|
+
}
|
|
12346
|
+
job.result = "applied";
|
|
12347
|
+
job.status = "done";
|
|
12348
|
+
job.finishedAt = Date.now();
|
|
12349
|
+
}
|
|
12350
|
+
catch (err) {
|
|
12351
|
+
const message = safeSecretError(err, secret);
|
|
12352
|
+
this.logger.warn({ connectionId: job.connectionId, reason: message }, "Settings connection secret apply failed");
|
|
12353
|
+
if (!replaced || !before || !envKey) {
|
|
12354
|
+
job.result = "rollback_failed";
|
|
12355
|
+
job.error = "secret was not applied";
|
|
12356
|
+
}
|
|
12357
|
+
else {
|
|
12358
|
+
try {
|
|
12359
|
+
const store = new SecretStore(join(this.dataDir, ".env"), new Set([envKey]));
|
|
12360
|
+
store.restore(before);
|
|
12361
|
+
if (oldToken === undefined)
|
|
12362
|
+
delete process.env[envKey];
|
|
12363
|
+
else
|
|
12364
|
+
process.env[envKey] = oldToken;
|
|
12365
|
+
// Build a fresh adapter from the restored token. If the old adapter
|
|
12366
|
+
// was stopped already, this is the only safe way to return to the
|
|
12367
|
+
// previous runtime without claiming a disk-only rollback succeeded.
|
|
12368
|
+
const restored = await this.rebuildAdapterForSecret(job.connectionId, channel, true);
|
|
12369
|
+
if (!restored)
|
|
12370
|
+
throw new Error("adapter rollback did not become connected");
|
|
12371
|
+
job.result = "rolled_back";
|
|
12372
|
+
}
|
|
12373
|
+
catch (rollbackErr) {
|
|
12374
|
+
this.logger.error({ connectionId: job.connectionId, reason: safeSecretError(rollbackErr, oldToken, [secret]) }, "Settings connection secret rollback failed");
|
|
12375
|
+
job.result = "rollback_failed";
|
|
12376
|
+
job.error = "secret rollback failed; adapter is disabled";
|
|
12377
|
+
}
|
|
12378
|
+
}
|
|
12379
|
+
job.status = "done";
|
|
12380
|
+
job.finishedAt = Date.now();
|
|
12381
|
+
}
|
|
12382
|
+
finally {
|
|
12383
|
+
this.connectionSecretInFlight.delete(job.connectionId);
|
|
12384
|
+
// Do not retain the token after the apply (success or rollback).
|
|
12385
|
+
secret = "";
|
|
12386
|
+
}
|
|
12387
|
+
}
|
|
12388
|
+
/** Stop the old provider client and construct a new one from process.env. */
|
|
12389
|
+
async rebuildAdapterForSecret(connectionId, channel, force = false) {
|
|
12390
|
+
const old = this.adapters.get(connectionId);
|
|
12391
|
+
if (!old && !force)
|
|
12392
|
+
return false; // The secret is valid on disk; the next start adopts it.
|
|
12393
|
+
const primary = this.getPrimaryAdapterId() === connectionId;
|
|
12394
|
+
const previousState = this.adapterState.get(connectionId);
|
|
12395
|
+
this.adapterState.set(connectionId, { status: "retrying", retryCount: previousState?.retryCount ?? 0 });
|
|
12396
|
+
if (primary && this.sessionPruneTimer) {
|
|
12397
|
+
clearInterval(this.sessionPruneTimer);
|
|
12398
|
+
this.sessionPruneTimer = null;
|
|
12399
|
+
}
|
|
12400
|
+
if (old) {
|
|
12401
|
+
old.removeAllListeners();
|
|
12402
|
+
await old.stop().catch(() => { });
|
|
12403
|
+
if (this.adapters.get(connectionId) === old)
|
|
12404
|
+
this.adapters.delete(connectionId);
|
|
12405
|
+
if (this.worlds.get(connectionId)?.adapter === old)
|
|
12406
|
+
this.worlds.delete(connectionId);
|
|
12407
|
+
if (primary && this.adapter === old)
|
|
12408
|
+
this.adapter = null;
|
|
12409
|
+
}
|
|
12410
|
+
let startedResolve = null;
|
|
12411
|
+
const started = new Promise(resolve => { startedResolve = resolve; });
|
|
12412
|
+
const onStarted = () => { startedResolve?.(); };
|
|
12413
|
+
if (primary)
|
|
12414
|
+
await this.startSingleAdapter(this.fleetConfig, channel, onStarted);
|
|
12415
|
+
else
|
|
12416
|
+
await this.startAdditionalAdapter(channel, true, onStarted);
|
|
12417
|
+
const fresh = this.adapters.get(connectionId);
|
|
12418
|
+
if (!fresh)
|
|
12419
|
+
throw new Error("new adapter did not start");
|
|
12420
|
+
const deadline = Date.now() + 15_000;
|
|
12421
|
+
// Telegram has no gateway health snapshot. Its start() method launches the
|
|
12422
|
+
// grammY polling loop in the background, so completion of start() is not a
|
|
12423
|
+
// connected signal. The adapter's `started` event is emitted only after the
|
|
12424
|
+
// first provider getMe succeeds; require that event before claiming apply.
|
|
12425
|
+
if (!fresh.getHealthSnapshot) {
|
|
12426
|
+
let timer;
|
|
12427
|
+
const timeout = new Promise((_, reject) => {
|
|
12428
|
+
timer = setTimeout(() => reject(new Error("new adapter did not emit started before deadline")), Math.max(1, deadline - Date.now()));
|
|
12429
|
+
timer.unref?.();
|
|
12430
|
+
});
|
|
12431
|
+
try {
|
|
12432
|
+
await Promise.race([started, timeout]);
|
|
12433
|
+
}
|
|
12434
|
+
finally {
|
|
12435
|
+
if (timer)
|
|
12436
|
+
clearTimeout(timer);
|
|
12437
|
+
}
|
|
12438
|
+
this.adapterState.set(connectionId, { status: "connected", retryCount: 0 });
|
|
12439
|
+
return true;
|
|
12440
|
+
}
|
|
12441
|
+
while (Date.now() < deadline) {
|
|
12442
|
+
const health = fresh.getHealthSnapshot?.();
|
|
12443
|
+
if (health?.status === "connected" || this.adapterState.get(connectionId)?.status === "connected")
|
|
12444
|
+
return true;
|
|
12445
|
+
await new Promise(resolve => setTimeout(resolve, 100));
|
|
12446
|
+
}
|
|
12447
|
+
throw new Error("new adapter did not become connected");
|
|
12448
|
+
}
|
|
12449
|
+
/**
|
|
12450
|
+
* What a reconcile is about to do, per target.
|
|
12451
|
+
*
|
|
12452
|
+
* A forecast, not the record: it is built from the same `classifyInstanceChange`
|
|
12453
|
+
* the reconcile decides with, but the rows that end up in the job are the ones
|
|
12454
|
+
* the reconcile reports as it works. A target the forecast missed (a Classic
|
|
12455
|
+
* instance inheriting a changed default) is added when it is first touched.
|
|
12456
|
+
*/
|
|
12457
|
+
planConfigApply() {
|
|
12458
|
+
const rows = [];
|
|
12459
|
+
const nextConfig = this.nextFleetConfig();
|
|
12460
|
+
const next = nextConfig?.instances ?? {};
|
|
12461
|
+
const classicNames = new Set(this.classicChannels?.getAll().map(ch => ch.instanceName) ?? []);
|
|
12462
|
+
for (const [name, config] of Object.entries(next)) {
|
|
12463
|
+
const daemon = this.daemons.get(name);
|
|
12464
|
+
if (!daemon) {
|
|
12465
|
+
rows.push({ target: name, kind: "restart" });
|
|
12466
|
+
continue;
|
|
12467
|
+
}
|
|
12468
|
+
const runtime = daemon.getConfigSnapshot?.();
|
|
12469
|
+
if (!runtime)
|
|
12470
|
+
continue;
|
|
12471
|
+
const change = classifyInstanceChange(runtime, config);
|
|
12472
|
+
if (change !== "none")
|
|
12473
|
+
rows.push({ target: name, kind: change === "restart" ? "restart" : "hot" });
|
|
12474
|
+
}
|
|
12475
|
+
for (const name of this.daemons.keys()) {
|
|
12476
|
+
if (!(name in next) && !classicNames.has(name))
|
|
12477
|
+
rows.push({ target: name, kind: "restart" });
|
|
12478
|
+
}
|
|
12479
|
+
if (this.appliedFleetLevel !== null && this.appliedFleetLevel !== this.fleetLevelSignature(nextConfig)) {
|
|
12480
|
+
rows.push({ target: APPLY_FLEET_TARGET, kind: "restart" });
|
|
12481
|
+
}
|
|
12482
|
+
return rows;
|
|
12483
|
+
}
|
|
12484
|
+
/**
|
|
12485
|
+
* Start (or re-join) a Settings apply.
|
|
12486
|
+
*
|
|
12487
|
+
* The key is the client's. Handing back the existing job for a repeated key is
|
|
12488
|
+
* the whole point: the retry after a lost response must not apply everything a
|
|
12489
|
+
* second time.
|
|
12490
|
+
*/
|
|
12491
|
+
startSettingsApply(key) {
|
|
12492
|
+
// The key is checked first on purpose: a retry of the apply that is running
|
|
12493
|
+
// right now must get its own job back, not "busy".
|
|
12494
|
+
const existing = this.applyJobs.findByKey(key);
|
|
12495
|
+
if (existing)
|
|
12496
|
+
return { job: existing, reused: true };
|
|
12497
|
+
if (this.reconcileInFlight || this.activeApplyJobId) {
|
|
12498
|
+
return { busy: this.activeApplyJobId ? this.applyJobs.get(this.activeApplyJobId) : null };
|
|
12499
|
+
}
|
|
12500
|
+
const job = this.applyJobs.create(key, this.planConfigApply());
|
|
12501
|
+
// Reserved synchronously: the work starts a microtask later, and a second
|
|
12502
|
+
// request arriving in that gap must see the slot taken.
|
|
12503
|
+
this.activeApplyJobId = job.id;
|
|
12504
|
+
// Start after the caller has its answer, so the first thing the page renders
|
|
12505
|
+
// is the whole plan with every row still pending — not a job the reconcile
|
|
12506
|
+
// has already half-finished synchronously.
|
|
12507
|
+
queueMicrotask(() => void this.runSettingsApply(job.id));
|
|
12508
|
+
return { job, reused: false };
|
|
12509
|
+
}
|
|
12510
|
+
async runSettingsApply(jobId) {
|
|
12511
|
+
const emit = () => {
|
|
12512
|
+
const job = this.applyJobs.get(jobId);
|
|
12513
|
+
// An accelerator only: these frames carry no event id, so a client that
|
|
12514
|
+
// reconnects cannot ask for what it missed. GET is the authority.
|
|
12515
|
+
if (job)
|
|
12516
|
+
this.emitSseEvent("apply_progress", viewOf(job));
|
|
12517
|
+
};
|
|
12518
|
+
// Passed in rather than parked on `this`: a shared field would let a second
|
|
12519
|
+
// reconcile redirect this job's reporting into another job's rows.
|
|
12520
|
+
const observer = (target, kind, status, error) => {
|
|
12521
|
+
this.applyJobs.update(jobId, job => {
|
|
12522
|
+
let row = job.targets.find(item => item.target === target);
|
|
12523
|
+
if (!row) {
|
|
12524
|
+
row = { target, kind, status: "pending" };
|
|
12525
|
+
job.targets.push(row);
|
|
12526
|
+
}
|
|
12527
|
+
row.kind = kind;
|
|
12528
|
+
row.status = status;
|
|
12529
|
+
if (error)
|
|
12530
|
+
row.error = error;
|
|
12531
|
+
});
|
|
12532
|
+
emit();
|
|
12533
|
+
};
|
|
12534
|
+
emit();
|
|
12535
|
+
try {
|
|
12536
|
+
const started = this.startExclusiveReconcile(observer);
|
|
12537
|
+
if (!started) {
|
|
12538
|
+
// The slot was reserved before the microtask, so this means a SIGHUP
|
|
12539
|
+
// reconcile started in between. Report it instead of applying twice.
|
|
12540
|
+
this.applyJobs.finish(jobId, "a config reload was already running");
|
|
12541
|
+
return;
|
|
12542
|
+
}
|
|
12543
|
+
const outcome = await started;
|
|
12544
|
+
this.applyJobs.finish(jobId, outcome.rejected);
|
|
12545
|
+
}
|
|
12546
|
+
catch (err) {
|
|
12547
|
+
this.applyJobs.finish(jobId, err instanceof Error ? err.message : String(err));
|
|
12548
|
+
}
|
|
12549
|
+
finally {
|
|
12550
|
+
this.activeApplyJobId = null;
|
|
12551
|
+
emit();
|
|
12552
|
+
}
|
|
10757
12553
|
}
|
|
10758
12554
|
async restartInstances() {
|
|
10759
12555
|
if (!this.configPath) {
|
|
@@ -11046,6 +12842,12 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
11046
12842
|
this.initializeWebAuthTokens();
|
|
11047
12843
|
this.healthServer = createServer((req, res) => {
|
|
11048
12844
|
res.setHeader("Content-Type", "application/json");
|
|
12845
|
+
// No Referer to a tunnel host, an upstream proxy, or any page linked from
|
|
12846
|
+
// the panel — the dashboard URL is itself a credential-bearing address.
|
|
12847
|
+
res.setHeader("Referrer-Policy", "no-referrer");
|
|
12848
|
+
// Authorization now depends on a cookie, so a shared cache (a tunnel, a
|
|
12849
|
+
// corporate proxy) must not serve one visitor's response to another.
|
|
12850
|
+
res.setHeader("Vary", "Cookie");
|
|
11049
12851
|
const requestPath = new URL(req.url ?? "/", `http://localhost:${port}`).pathname;
|
|
11050
12852
|
// Browsers request this automatically and AgEnD does not ship an icon.
|
|
11051
12853
|
// It is neither user data nor an API route, so do not turn the harmless
|
|
@@ -11071,15 +12873,25 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
11071
12873
|
// like the other /view data routes (usage-api.ts rejects non-GET).
|
|
11072
12874
|
}
|
|
11073
12875
|
else {
|
|
11074
|
-
// All other endpoints require a
|
|
12876
|
+
// All other endpoints require a session cookie or an X-Agend-Token
|
|
12877
|
+
// header; a `?token=` in the URL is only redeemed for a cookie on a GET.
|
|
11075
12878
|
// /ui/* will also re-check in web-api.ts, which is harmless.
|
|
11076
12879
|
const parsedUrl = new URL(req.url ?? "/", `http://localhost:${port}`);
|
|
11077
|
-
const
|
|
11078
|
-
|
|
11079
|
-
|
|
11080
|
-
|
|
11081
|
-
|
|
11082
|
-
|
|
12880
|
+
const decision = decideWebGate(req, parsedUrl, this.webToken);
|
|
12881
|
+
if (decision.kind === "reject") {
|
|
12882
|
+
res.writeHead(decision.status);
|
|
12883
|
+
res.end(JSON.stringify({ error: decision.message }));
|
|
12884
|
+
return;
|
|
12885
|
+
}
|
|
12886
|
+
if (decision.kind === "exchange") {
|
|
12887
|
+
res.setHeader("Set-Cookie", decision.setCookie);
|
|
12888
|
+
res.setHeader("Location", decision.location);
|
|
12889
|
+
// A cached redirect would replay a Set-Cookie for a rotated token.
|
|
12890
|
+
res.setHeader("Cache-Control", "no-store");
|
|
12891
|
+
res.writeHead(302);
|
|
12892
|
+
// Browsers follow the Location; a script that does not gets told why
|
|
12893
|
+
// its URL token stopped being echoed back as data.
|
|
12894
|
+
res.end(JSON.stringify({ redirect: decision.location }));
|
|
11083
12895
|
return;
|
|
11084
12896
|
}
|
|
11085
12897
|
}
|
|
@@ -11317,8 +13129,11 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
11317
13129
|
this.logger.info({ port }, afterTakeover
|
|
11318
13130
|
? "Health endpoint listening (after takeover)"
|
|
11319
13131
|
: "Health endpoint listening");
|
|
11320
|
-
|
|
11321
|
-
|
|
13132
|
+
// Never the token: fleet.log is readable by anything that can read the
|
|
13133
|
+
// data dir, is copied into bug reports, and is tailed in shared terminals.
|
|
13134
|
+
// `/dashboard` and `agend web` are the ways to get an authorized link.
|
|
13135
|
+
this.logger.info({ url: `http://localhost:${port}/ui` }, "Web UI available (open it with /dashboard or `agend web`)");
|
|
13136
|
+
this.logger.info({ url: `http://localhost:${port}/view` }, "Web View available");
|
|
11322
13137
|
};
|
|
11323
13138
|
this.healthServer.on("error", (err) => {
|
|
11324
13139
|
this.healthServerListening = false;
|
|
@@ -11335,8 +13150,25 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
11335
13150
|
if (existsSync(pidPath)) {
|
|
11336
13151
|
const oldPid = parseInt(readFileSync(pidPath, "utf-8").trim(), 10);
|
|
11337
13152
|
if (oldPid && oldPid !== process.pid) {
|
|
11338
|
-
|
|
11339
|
-
|
|
13153
|
+
// fleet.pid is a claim, not proof. A stale or wrong entry points
|
|
13154
|
+
// at whatever now holds that pid, and this used to SIGTERM it —
|
|
13155
|
+
// an unrelated process killed because a port was busy. Confirm
|
|
13156
|
+
// the target really is an AgEnD fleet, and when that cannot be
|
|
13157
|
+
// confirmed, do not signal: not killing costs a dashboard, and
|
|
13158
|
+
// killing costs somebody else's process.
|
|
13159
|
+
const commandLine = readProcessCommandLine(oldPid);
|
|
13160
|
+
if (isFleetStartCommandLine(commandLine)) {
|
|
13161
|
+
process.kill(oldPid, "SIGTERM");
|
|
13162
|
+
this.logger.info({ oldPid }, "Killed old fleet process");
|
|
13163
|
+
}
|
|
13164
|
+
else {
|
|
13165
|
+
this.logger.warn({
|
|
13166
|
+
oldPid,
|
|
13167
|
+
// Truncated: this is an unrelated process's command line, and
|
|
13168
|
+
// fleet.log is copied into bug reports.
|
|
13169
|
+
commandLine: commandLine ? `${commandLine.slice(0, 60)}${commandLine.length > 60 ? "…" : ""}` : "(unreadable)",
|
|
13170
|
+
}, "fleet.pid does not name an AgEnD fleet process — not signalling it");
|
|
13171
|
+
}
|
|
11340
13172
|
}
|
|
11341
13173
|
}
|
|
11342
13174
|
}
|