@songsid/agend 2.1.6-beta.1 → 2.1.6-beta.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-endpoint.d.ts +18 -0
- package/dist/agent-endpoint.js +53 -2
- package/dist/agent-endpoint.js.map +1 -1
- package/dist/apply-job.d.ts +114 -0
- package/dist/apply-job.js +214 -0
- package/dist/apply-job.js.map +1 -0
- package/dist/backend/claude-code.d.ts +25 -0
- package/dist/backend/claude-code.js +127 -0
- package/dist/backend/claude-code.js.map +1 -1
- package/dist/backend/codex.d.ts +41 -0
- package/dist/backend/codex.js +133 -0
- package/dist/backend/codex.js.map +1 -1
- package/dist/backend/credential-profile.d.ts +159 -0
- package/dist/backend/credential-profile.js +340 -0
- package/dist/backend/credential-profile.js.map +1 -0
- package/dist/backend/factory.js +4 -1
- package/dist/backend/factory.js.map +1 -1
- package/dist/backend/kiro-auth-store.d.ts +25 -0
- package/dist/backend/kiro-auth-store.js +61 -0
- package/dist/backend/kiro-auth-store.js.map +1 -0
- package/dist/backend/kiro.d.ts +8 -0
- package/dist/backend/kiro.js +31 -2
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/muse.d.ts +90 -0
- package/dist/backend/muse.js +437 -0
- package/dist/backend/muse.js.map +1 -0
- package/dist/backend/types.d.ts +41 -0
- package/dist/backend/types.js.map +1 -1
- package/dist/channel/access-manager.d.ts +20 -0
- package/dist/channel/access-manager.js +25 -0
- package/dist/channel/access-manager.js.map +1 -1
- package/dist/channel/adapters/discord.d.ts +17 -0
- package/dist/channel/adapters/discord.js +72 -11
- package/dist/channel/adapters/discord.js.map +1 -1
- package/dist/channel/adapters/telegram.d.ts +6 -0
- package/dist/channel/adapters/telegram.js +36 -0
- package/dist/channel/adapters/telegram.js.map +1 -1
- package/dist/channel/markdown-chunk.d.ts +8 -3
- package/dist/channel/markdown-chunk.js +12 -3
- package/dist/channel/markdown-chunk.js.map +1 -1
- package/dist/channel/mcp-server.js +9 -12
- package/dist/channel/mcp-server.js.map +1 -1
- package/dist/channel/mcp-tools.d.ts +6 -2
- package/dist/channel/mcp-tools.js +6 -36
- package/dist/channel/mcp-tools.js.map +1 -1
- package/dist/channel/tool-router.js +32 -3
- package/dist/channel/tool-router.js.map +1 -1
- package/dist/channel/types.d.ts +18 -0
- package/dist/cli.js +113 -0
- package/dist/cli.js.map +1 -1
- package/dist/config-validator.js +45 -5
- package/dist/config-validator.js.map +1 -1
- package/dist/connection-secrets.d.ts +96 -0
- package/dist/connection-secrets.js +25 -0
- package/dist/connection-secrets.js.map +1 -0
- package/dist/daemon.d.ts +61 -0
- package/dist/daemon.js +359 -88
- package/dist/daemon.js.map +1 -1
- package/dist/event-log.d.ts +9 -0
- package/dist/event-log.js +21 -0
- package/dist/event-log.js.map +1 -1
- package/dist/fleet-level-config.d.ts +42 -0
- package/dist/fleet-level-config.js +78 -0
- package/dist/fleet-level-config.js.map +1 -0
- package/dist/fleet-lock.d.ts +21 -0
- package/dist/fleet-lock.js +42 -8
- package/dist/fleet-lock.js.map +1 -1
- package/dist/fleet-manager.d.ts +417 -2
- package/dist/fleet-manager.js +1981 -182
- package/dist/fleet-manager.js.map +1 -1
- package/dist/general-knowledge/skills/credential-profiles/SKILL.md +145 -0
- package/dist/general-knowledge/skills/worker-collaboration/SKILL.md +11 -1
- package/dist/instance-config-impact.d.ts +55 -0
- package/dist/instance-config-impact.js +152 -0
- package/dist/instance-config-impact.js.map +1 -0
- package/dist/instance-lifecycle.js +8 -4
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/locale.js +24 -0
- package/dist/locale.js.map +1 -1
- package/dist/logger.js +42 -25
- package/dist/logger.js.map +1 -1
- package/dist/muse-usage-relay.d.ts +72 -0
- package/dist/muse-usage-relay.js +419 -0
- package/dist/muse-usage-relay.js.map +1 -0
- package/dist/outbound-handlers.d.ts +5 -1
- package/dist/outbound-handlers.js +199 -1
- package/dist/outbound-handlers.js.map +1 -1
- package/dist/outbound-schemas.d.ts +7 -5
- package/dist/outbound-schemas.js +3 -2
- package/dist/outbound-schemas.js.map +1 -1
- package/dist/provider-probe.d.ts +48 -0
- package/dist/provider-probe.js +121 -0
- package/dist/provider-probe.js.map +1 -0
- package/dist/provider-secret-registry.d.ts +98 -0
- package/dist/provider-secret-registry.js +334 -0
- package/dist/provider-secret-registry.js.map +1 -0
- package/dist/quickstart-api.d.ts +164 -0
- package/dist/quickstart-api.js +351 -0
- package/dist/quickstart-api.js.map +1 -0
- package/dist/quickstart.js +23 -50
- package/dist/quickstart.js.map +1 -1
- package/dist/scheduler/db.js +17 -5
- package/dist/scheduler/db.js.map +1 -1
- package/dist/scheduler/db.test.js +34 -1
- package/dist/scheduler/db.test.js.map +1 -1
- package/dist/scheduler/types.d.ts +4 -0
- package/dist/scheduler/types.js.map +1 -1
- package/dist/secret-store.d.ts +25 -0
- package/dist/secret-store.js +165 -0
- package/dist/secret-store.js.map +1 -0
- package/dist/self-restart-limit.d.ts +21 -0
- package/dist/self-restart-limit.js +129 -0
- package/dist/self-restart-limit.js.map +1 -0
- package/dist/settings-api.d.ts +116 -0
- package/dist/settings-api.js +530 -5
- package/dist/settings-api.js.map +1 -1
- package/dist/setup-auth.d.ts +113 -0
- package/dist/setup-auth.js +181 -0
- package/dist/setup-auth.js.map +1 -0
- package/dist/setup-form.d.ts +18 -0
- package/dist/setup-form.js +315 -0
- package/dist/setup-form.js.map +1 -0
- package/dist/setup-host.d.ts +162 -0
- package/dist/setup-host.js +496 -0
- package/dist/setup-host.js.map +1 -0
- package/dist/setup-marker.d.ts +8 -0
- package/dist/setup-marker.js +39 -0
- package/dist/setup-marker.js.map +1 -0
- package/dist/setup-tunnel-consent.d.ts +53 -0
- package/dist/setup-tunnel-consent.js +82 -0
- package/dist/setup-tunnel-consent.js.map +1 -0
- package/dist/setup-wizard.js +4 -0
- package/dist/setup-wizard.js.map +1 -1
- package/dist/steer-capability.js +4 -1
- package/dist/steer-capability.js.map +1 -1
- package/dist/tips.js +1 -1
- package/dist/tips.js.map +1 -1
- package/dist/tmux-manager.js +29 -4
- package/dist/tmux-manager.js.map +1 -1
- package/dist/tool-permissions-notice.d.ts +38 -0
- package/dist/tool-permissions-notice.js +92 -0
- package/dist/tool-permissions-notice.js.map +1 -0
- package/dist/tool-permissions.d.ts +120 -0
- package/dist/tool-permissions.js +277 -0
- package/dist/tool-permissions.js.map +1 -0
- package/dist/topic-commands.js +1 -0
- package/dist/topic-commands.js.map +1 -1
- package/dist/transcript-sources.d.ts +12 -1
- package/dist/transcript-sources.js +19 -3
- package/dist/transcript-sources.js.map +1 -1
- package/dist/tunnel/cloudflared.d.ts +71 -0
- package/dist/tunnel/cloudflared.js +447 -0
- package/dist/tunnel/cloudflared.js.map +1 -0
- package/dist/tunnel/lease.d.ts +75 -0
- package/dist/tunnel/lease.js +209 -0
- package/dist/tunnel/lease.js.map +1 -0
- package/dist/tunnel/manager.d.ts +56 -0
- package/dist/tunnel/manager.js +167 -0
- package/dist/tunnel/manager.js.map +1 -0
- package/dist/tunnel/types.d.ts +119 -0
- package/dist/tunnel/types.js +33 -0
- package/dist/tunnel/types.js.map +1 -0
- package/dist/types.d.ts +2 -0
- package/dist/ui/dashboard.html +4 -3
- package/dist/ui/settings.html +808 -156
- package/dist/ui/view.html +2 -2
- package/dist/usage/i18n-keys.d.ts +1 -1
- package/dist/usage/i18n-keys.js +1 -1
- package/dist/usage/i18n-keys.js.map +1 -1
- package/dist/usage/providers.d.ts +23 -14
- package/dist/usage/providers.js +220 -58
- package/dist/usage/providers.js.map +1 -1
- package/dist/usage/usage-api.d.ts +13 -0
- package/dist/usage/usage-api.js +69 -1
- package/dist/usage/usage-api.js.map +1 -1
- package/dist/web-api.js +8 -5
- package/dist/web-api.js.map +1 -1
- package/dist/web-auth.d.ts +75 -0
- package/dist/web-auth.js +208 -2
- package/dist/web-auth.js.map +1 -1
- package/dist/web-terminal.d.ts +30 -1
- package/dist/web-terminal.js +86 -3
- package/dist/web-terminal.js.map +1 -1
- package/package.json +2 -1
package/dist/fleet-manager.js
CHANGED
|
@@ -5,7 +5,6 @@ import { freemem, totalmem, cpus } from "node:os";
|
|
|
5
5
|
import { createServer } from "node:http";
|
|
6
6
|
import { join, dirname, basename } from "node:path";
|
|
7
7
|
import { fileURLToPath } from "node:url";
|
|
8
|
-
import { isDeepStrictEqual } from "node:util";
|
|
9
8
|
import { getAgendHome, ensureWorkspaceGit } from "./paths.js";
|
|
10
9
|
import { beginFullRestartProgress as persistFullRestartProgress, beginUpdateProgress as persistUpdateProgress, clearUpdateMarker, isUpdateInProgress, readUpdateProgress, setUpdateProgressStage, updateProgressOperation, } from "./update-marker.js";
|
|
11
10
|
import { formatUpdateProgress } from "./update-progress.js";
|
|
@@ -45,7 +44,7 @@ import { StatuslineWatcher } from "./statusline-watcher.js";
|
|
|
45
44
|
import { outboundHandlers } from "./outbound-handlers.js";
|
|
46
45
|
import { handleWebRequest, broadcastSseEvent } from "./web-api.js";
|
|
47
46
|
import { handleViewRequest, isViewPath } from "./view-api.js";
|
|
48
|
-
import { handleUsageRequest, isUsagePath, usageProviderIdForBackend } from "./usage/usage-api.js";
|
|
47
|
+
import { filterUsageProviders, formatDiscordUsageActivity, getUsageSnapshot, handleUsageRequest, isUsagePath, usageProviderIdForBackend } from "./usage/usage-api.js";
|
|
49
48
|
import { LOGIN_FLOWS, LOGIN_BACKEND_ALIASES, checkAuthStatus } from "./login-flows.js";
|
|
50
49
|
import { LoginSession } from "./login-manager.js";
|
|
51
50
|
import { LoginController, LOGIN_TOKEN_RESEND_PREFIX, POST_LOGIN_RECOVERY_DEADLINE_MS, announcePostLoginRecovery } from "./login-controller.js";
|
|
@@ -53,15 +52,30 @@ import { runBeforeDeadline } from "./deadline.js";
|
|
|
53
52
|
import { LoginWindowLock } from "./login-window-lock.js";
|
|
54
53
|
import { handleSettingsRequest } from "./settings-api.js";
|
|
55
54
|
import { setLocale, detectLocale, getLocale, t } from "./locale.js";
|
|
56
|
-
import { handleAgentRequest } from "./agent-endpoint.js";
|
|
55
|
+
import { handleAgentRequest, ToolNotPermittedError } from "./agent-endpoint.js";
|
|
57
56
|
import { ClassicChannelManager, getClassicBackendChoices, isSelectableClassicBackend, readClassicLastActivityAt } from "./classic-channel-manager.js";
|
|
58
57
|
import { assertExplicitInstanceRemoval } from "./instance-removal.js";
|
|
59
58
|
import { validateFleetConfig } from "./config-validator.js";
|
|
60
59
|
import { readLastInboundAt } from "./daemon.js";
|
|
61
60
|
import { clearPausedMarker } from "./pause-marker.js";
|
|
62
|
-
import { releaseProcessFleetLock } from "./fleet-lock.js";
|
|
61
|
+
import { isFleetStartCommandLine, readProcessCommandLine, releaseProcessFleetLock } from "./fleet-lock.js";
|
|
62
|
+
import { isSetupComplete, markSetupComplete } from "./setup-marker.js";
|
|
63
|
+
import { manualCleanupMessage, reapStaleTunnel } from "./tunnel/lease.js";
|
|
64
|
+
import { buildToolPermissionsNotice } from "./tool-permissions-notice.js";
|
|
63
65
|
import { GENERAL_PAUSE_ERROR, isGeneralInstance } from "./general-instance.js";
|
|
64
|
-
import {
|
|
66
|
+
import { mayUseTool, resolveToolSet, scheduleOpRefusal, toolForIpcType, toolRefusedMessage, } from "./tool-permissions.js";
|
|
67
|
+
import { decideWebGate, loadOrCreateWebToken, readWebToken } from "./web-auth.js";
|
|
68
|
+
import { fleetLevelDifferences, fleetLevelSignature } from "./fleet-level-config.js";
|
|
69
|
+
import { checkSelfRestartAllowance, recordSelfRestartAttempt } from "./self-restart-limit.js";
|
|
70
|
+
import { SecretStore } from "./secret-store.js";
|
|
71
|
+
import { opaqueId, safeSecretError, SECRET_CHALLENGE_TTL_MS, } from "./connection-secrets.js";
|
|
72
|
+
import { verifyDiscordToken, verifyTelegramToken } from "./provider-probe.js";
|
|
73
|
+
import { PROVIDER_SECRET_SPECS, providerSecretSpec, providerRegistryEnvKeys, isReservedProviderEnvKey, verifyProviderSecret, } from "./provider-secret-registry.js";
|
|
74
|
+
/** A self-restart is a whole service restart; 120s is the apply budget, not this. */
|
|
75
|
+
const SELF_RESTART_DEADLINE_MS = 300_000;
|
|
76
|
+
import { APPLY_FLEET_TARGET, ApplyJobStore, viewOf, } from "./apply-job.js";
|
|
77
|
+
import { instanceCredentialProfile } from "./backend/credential-profile.js";
|
|
78
|
+
import { classifyInstanceChange, CLASSIC_HOT_CONFIG_KEYS, HOT_INSTANCE_CONFIG_KEYS, hotConfigUpdate, } from "./instance-config-impact.js";
|
|
65
79
|
import { formatRestartProgressCompletion, RESTART_PROGRESS_TERMINAL_TIMEOUT_MS, RestartProgress, } from "./restart-progress.js";
|
|
66
80
|
import { launchFullRestartHelper } from "./full-restart.js";
|
|
67
81
|
import { collectRedundantInstanceDefaultPaths } from "./fleet-yaml-slim.js";
|
|
@@ -179,31 +193,6 @@ const DELIVERY_STATUS_EMOJIS = new Set(["👀", "⏳", "✅", "❌"]);
|
|
|
179
193
|
* emoji never changes the documented delivery-state protocol.
|
|
180
194
|
*/
|
|
181
195
|
const IGNORED_REACTION_EMOJIS = new Set(["📷"]);
|
|
182
|
-
const HOT_INSTANCE_CONFIG_KEYS = new Set([
|
|
183
|
-
"tool_progress",
|
|
184
|
-
"reply_completion_guard",
|
|
185
|
-
"mcp_proxy_reply",
|
|
186
|
-
"auto_pause_after",
|
|
187
|
-
"warm_cap",
|
|
188
|
-
"display_name",
|
|
189
|
-
"description",
|
|
190
|
-
"tags",
|
|
191
|
-
"log_level",
|
|
192
|
-
]);
|
|
193
|
-
function splitHotColdConfig(config) {
|
|
194
|
-
const hot = {};
|
|
195
|
-
const cold = {};
|
|
196
|
-
for (const [key, value] of Object.entries(config)) {
|
|
197
|
-
(HOT_INSTANCE_CONFIG_KEYS.has(key) ? hot : cold)[key] = value;
|
|
198
|
-
}
|
|
199
|
-
return { hot, cold };
|
|
200
|
-
}
|
|
201
|
-
function hotConfigUpdate(config) {
|
|
202
|
-
const update = {};
|
|
203
|
-
for (const key of HOT_INSTANCE_CONFIG_KEYS)
|
|
204
|
-
update[key] = config[key] ?? null;
|
|
205
|
-
return update;
|
|
206
|
-
}
|
|
207
196
|
/**
|
|
208
197
|
* How long a delivery waits out a disconnected instance IPC before giving up.
|
|
209
198
|
*
|
|
@@ -232,6 +221,8 @@ const CLEAR_CONFIRM_TIMEOUT_MS = 15_000;
|
|
|
232
221
|
/** Default lifetime for long-lived nonce prompts (clear overrides this to 15s). */
|
|
233
222
|
const NONCE_BUTTON_TIMEOUT_MS = 15 * 60_000;
|
|
234
223
|
const TIP_BUTTON_TIMEOUT_MS = 24 * 60 * 60_000;
|
|
224
|
+
/** How long shutdown will spend retiring still-armed button prompts. */
|
|
225
|
+
const NONCE_RETIRE_BUDGET_MS = 5_000;
|
|
235
226
|
const CLI_ENV_TTL_MS = 24 * 60 * 60 * 1000; // hard validity bound for the cached CLI env
|
|
236
227
|
/**
|
|
237
228
|
* How old the cached CLI env may be before `/model` re-probes it live.
|
|
@@ -449,6 +440,40 @@ export class FleetManager {
|
|
|
449
440
|
adapterRestarting = new Set();
|
|
450
441
|
// Adapter isolation: track state per adapter for retry + visibility
|
|
451
442
|
adapterState = new Map();
|
|
443
|
+
/** Web Settings secret rotation is deliberately separate from reconnect
|
|
444
|
+
* recovery: a rotation must build a fresh provider client with the new token,
|
|
445
|
+
* and stale callbacks from the old client must not win. */
|
|
446
|
+
connectionSecretChallenges = new Map();
|
|
447
|
+
connectionSecretChallengesByKey = new Map();
|
|
448
|
+
connectionSecretJobs = new Map();
|
|
449
|
+
connectionSecretJobSession = new Map();
|
|
450
|
+
connectionSecretInFlight = new Map();
|
|
451
|
+
connectionSecretGenerations = new Map();
|
|
452
|
+
/** Local epoch that fences a challenge across adapter replacement and
|
|
453
|
+
* provider reconnect generations. The adapter's own generation can reset
|
|
454
|
+
* when a new adapter object is constructed, so keep an independent epoch. */
|
|
455
|
+
connectionSecretAdapterRefs = new Map();
|
|
456
|
+
connectionSecretHealthGenerations = new Map();
|
|
457
|
+
/** Generic API-key verifier/apply state. The challenge scope contains the
|
|
458
|
+
* resolved spec/env key, so a request can never retarget another provider. */
|
|
459
|
+
providerSecretChallenges = new Map();
|
|
460
|
+
providerSecretChallengesByKey = new Map();
|
|
461
|
+
providerSecretJobs = new Map();
|
|
462
|
+
providerSecretJobSession = new Map();
|
|
463
|
+
providerSecretInFlight = new Map();
|
|
464
|
+
providerSecretGenerations = new Map();
|
|
465
|
+
/** Test seam only; production always uses the fixed HTTPS client. */
|
|
466
|
+
providerSecretHttpClient;
|
|
467
|
+
/** Code-owned activation hooks; never populated from a request. */
|
|
468
|
+
providerSecretReloadHooks = new Map();
|
|
469
|
+
/** In-memory snapshots for narrow hot consumers (currently Groq voice). */
|
|
470
|
+
providerSecretHotSnapshots = new Map();
|
|
471
|
+
/** Web Settings connection-binding step-up challenges and apply jobs. */
|
|
472
|
+
connectionBindingChallenges = new Map();
|
|
473
|
+
connectionBindingChallengesByKey = new Map();
|
|
474
|
+
connectionBindingJobs = new Map();
|
|
475
|
+
connectionBindingJobSession = new Map();
|
|
476
|
+
connectionBindingInFlight = new Map();
|
|
452
477
|
collabInstances = new Set();
|
|
453
478
|
// Health endpoint
|
|
454
479
|
healthServer = null;
|
|
@@ -462,6 +487,9 @@ export class FleetManager {
|
|
|
462
487
|
fullRestartLauncher = launchFullRestartHelper;
|
|
463
488
|
eventLogPruneTimer = null;
|
|
464
489
|
logRotateTimer = null;
|
|
490
|
+
discordPresenceTimer = null;
|
|
491
|
+
discordPresenceInFlight = null;
|
|
492
|
+
static DISCORD_PRESENCE_REFRESH_MS = 15 * 60_000;
|
|
465
493
|
/** Days of event/activity history to keep. */
|
|
466
494
|
static EVENT_LOG_RETENTION_DAYS = 30;
|
|
467
495
|
watchdogTimer = null;
|
|
@@ -472,7 +500,29 @@ export class FleetManager {
|
|
|
472
500
|
mirrorTimer = null;
|
|
473
501
|
// Web UI: SSE clients + auth token
|
|
474
502
|
sseClients = new Set();
|
|
475
|
-
|
|
503
|
+
/**
|
|
504
|
+
* Read from disk on every access rather than cached at startup: `agend
|
|
505
|
+
* web-token rotate` runs in a separate process, and a cached copy would keep
|
|
506
|
+
* authorizing revoked links and cookies until the fleet restarted.
|
|
507
|
+
*/
|
|
508
|
+
get webToken() { return readWebToken(this.dataDir); }
|
|
509
|
+
/**
|
|
510
|
+
* Set while a Settings apply job is driving the reconcile. The reconcile
|
|
511
|
+
* stays the single doer; it just says out loud what it is doing to whom, so
|
|
512
|
+
* the job's rows are the work rather than a prediction of it.
|
|
513
|
+
*/
|
|
514
|
+
applyJobStoreCache = null;
|
|
515
|
+
/** The apply that currently owns the reconcile slot, reserved synchronously
|
|
516
|
+
* so a second request cannot slip in before the first one starts working. */
|
|
517
|
+
activeApplyJobId = null;
|
|
518
|
+
/** The fleet-level signature this process actually came up on. */
|
|
519
|
+
appliedFleetLevel = null;
|
|
520
|
+
/** The config behind that signature, kept so a "needs restart" log can name
|
|
521
|
+
* which keys moved rather than just asserting that something did. */
|
|
522
|
+
startupFleetConfig = null;
|
|
523
|
+
/** Set when the file on disk and the in-memory config disagree on a
|
|
524
|
+
* startup-only key at startup. See checkStartupSignatureConsistency(). */
|
|
525
|
+
fleetSignatureMismatch = null;
|
|
476
526
|
viewToken = null;
|
|
477
527
|
healthServerListening = false;
|
|
478
528
|
constructor(dataDir) {
|
|
@@ -548,26 +598,48 @@ export class FleetManager {
|
|
|
548
598
|
this.scheduleReconcile();
|
|
549
599
|
}
|
|
550
600
|
scheduleReconcile() {
|
|
551
|
-
|
|
601
|
+
const started = this.startExclusiveReconcile();
|
|
602
|
+
if (!started) {
|
|
552
603
|
this.reloadPending = true;
|
|
553
604
|
this.logger.info("Config reconciliation already running — coalesced reload request");
|
|
554
605
|
return;
|
|
555
606
|
}
|
|
556
|
-
|
|
557
|
-
this.reconcileInFlight = this.reconcileInstances()
|
|
558
|
-
.catch(err => {
|
|
607
|
+
started.catch(err => {
|
|
559
608
|
// Almost always a YAML parse error. Log-only meant the user edited
|
|
560
609
|
// fleet.yaml, sent SIGHUP, and got no reaction and no explanation.
|
|
561
610
|
this.logger.error({ err }, "SIGHUP config reload failed");
|
|
562
611
|
const message = err instanceof Error ? err.message : String(err);
|
|
563
612
|
this.notifyFleetError(t("fleet.reload_failed", message));
|
|
564
|
-
})
|
|
613
|
+
});
|
|
614
|
+
}
|
|
615
|
+
/**
|
|
616
|
+
* Take the reconcile slot, or refuse.
|
|
617
|
+
*
|
|
618
|
+
* Only one reconcile may touch lifecycle and config at a time — two of them
|
|
619
|
+
* stop and start the same instance in parallel. SIGHUP and a Settings apply
|
|
620
|
+
* are the same operation from two entrances, so they share the one slot: the
|
|
621
|
+
* signal coalesces into a pending replay, the apply is told the fleet is busy.
|
|
622
|
+
*
|
|
623
|
+
* The returned promise is the caller's to handle; the stored one is already
|
|
624
|
+
* handled, so a rejection never escapes as an unhandled rejection.
|
|
625
|
+
*/
|
|
626
|
+
startExclusiveReconcile(observer) {
|
|
627
|
+
if (this.reconcileInFlight)
|
|
628
|
+
return null;
|
|
629
|
+
this.reloadPending = false;
|
|
630
|
+
let settle;
|
|
631
|
+
const caller = new Promise((resolve, reject) => {
|
|
632
|
+
settle = (err, outcome) => (err ? reject(err instanceof Error ? err : new Error(String(err))) : resolve(outcome ?? {}));
|
|
633
|
+
});
|
|
634
|
+
this.reconcileInFlight = this.reconcileInstances(observer)
|
|
635
|
+
.then(outcome => settle(null, outcome), err => settle(err))
|
|
565
636
|
.finally(() => {
|
|
566
637
|
this.reconcileInFlight = null;
|
|
567
638
|
if (this.reloadPending && this.startupComplete) {
|
|
568
639
|
this.scheduleReconcile();
|
|
569
640
|
}
|
|
570
641
|
});
|
|
642
|
+
return caller;
|
|
571
643
|
}
|
|
572
644
|
/**
|
|
573
645
|
* Is the fleet going down (or coming back up) on purpose?
|
|
@@ -583,6 +655,39 @@ export class FleetManager {
|
|
|
583
655
|
}
|
|
584
656
|
finishStartup() {
|
|
585
657
|
this.startupComplete = true;
|
|
658
|
+
// Resolve whatever a previous run — or a setup host that crashed — left
|
|
659
|
+
// behind. A tunnel nobody is tracking is a public entrance nobody is
|
|
660
|
+
// watching, and the fleet starting is the moment there is finally a process
|
|
661
|
+
// around to notice. Never throws: a lease that cannot be resolved blocks
|
|
662
|
+
// the next tunnel and says so, it does not block the fleet.
|
|
663
|
+
this.announceToolPermissionsChange();
|
|
664
|
+
void reapStaleTunnel(this.dataDir)
|
|
665
|
+
.then(outcome => {
|
|
666
|
+
if (outcome.kind === "manual")
|
|
667
|
+
this.logger.warn({ tunnel: outcome }, manualCleanupMessage(outcome));
|
|
668
|
+
else if (outcome.kind === "reaped")
|
|
669
|
+
this.logger.info({ how: outcome.how, pid: outcome.pid }, "Reaped a leftover tunnel");
|
|
670
|
+
})
|
|
671
|
+
.catch(err => this.logger.warn({ err }, "Tunnel reaper failed"));
|
|
672
|
+
// An existing installation has never written the setup marker — it predates
|
|
673
|
+
// it — so `agend setup` would open a pre-fleet form for a fleet that plainly
|
|
674
|
+
// exists. A fleet that just came up on a config with agents in it is proof
|
|
675
|
+
// enough that setup happened.
|
|
676
|
+
if (Object.keys(this.fleetConfig?.instances ?? {}).length > 0 && !isSetupComplete(this.dataDir)) {
|
|
677
|
+
markSetupComplete(this.dataDir);
|
|
678
|
+
}
|
|
679
|
+
// After slimFleetConfigAtStartup() and the general/topic fixups, all of
|
|
680
|
+
// which may rewrite fleet.yaml — the baseline has to be what this process
|
|
681
|
+
// is actually running, compared against what a reconcile would load.
|
|
682
|
+
this.appliedFleetLevel = this.fleetLevelSignature();
|
|
683
|
+
this.startupFleetConfig = this.fleetConfig ? structuredClone(this.fleetConfig) : null;
|
|
684
|
+
this.checkStartupSignatureConsistency();
|
|
685
|
+
// A job from the process that just died cannot still be running here. The
|
|
686
|
+
// restart applied the saved config to every instance, so its open rows are
|
|
687
|
+
// finished — by the restart, which is what the user needs told.
|
|
688
|
+
for (const settled of this.applyJobs.settleAfterRestart()) {
|
|
689
|
+
this.logger.info({ jobId: settled.id }, "Settings apply job settled by fleet restart");
|
|
690
|
+
}
|
|
586
691
|
// We are the post-update fleet: the update is over by definition. Clearing
|
|
587
692
|
// it here (rather than in the update command, which exits before the new
|
|
588
693
|
// fleet is up) is what keeps the quiet window from outliving the restart.
|
|
@@ -708,11 +813,30 @@ export class FleetManager {
|
|
|
708
813
|
if (this.rawFleetDocument.errors.length > 0) {
|
|
709
814
|
throw new Error(`Invalid fleet.yaml: ${this.rawFleetDocument.errors[0].message}`);
|
|
710
815
|
}
|
|
711
|
-
|
|
712
|
-
|
|
816
|
+
const raw = loadRawFleetConfig(configPath);
|
|
817
|
+
const loaded = loadFleetConfig(configPath);
|
|
818
|
+
this.assertProviderSecretEnvKeys(loaded);
|
|
819
|
+
this.rawFleetConfig = raw;
|
|
820
|
+
this.fleetConfig = loaded;
|
|
713
821
|
this.savedFleetConfigSnapshot = structuredClone(this.fleetConfig);
|
|
714
822
|
return this.fleetConfig;
|
|
715
823
|
}
|
|
824
|
+
/**
|
|
825
|
+
* A channel's configurable bot_token_env is an env-key writer too. Refuse
|
|
826
|
+
* an overlap with a registry API key (or a process-reserved key) before the
|
|
827
|
+
* config becomes live; otherwise a Discord token could be written into
|
|
828
|
+
* GROQ_API_KEY by a perfectly valid-looking rotation request.
|
|
829
|
+
*/
|
|
830
|
+
assertProviderSecretEnvKeys(config) {
|
|
831
|
+
const registryKeys = providerRegistryEnvKeys();
|
|
832
|
+
const channels = config.channels ?? (config.channel ? [config.channel] : []);
|
|
833
|
+
for (const channel of channels) {
|
|
834
|
+
const key = channel.bot_token_env;
|
|
835
|
+
if (registryKeys.has(key) || isReservedProviderEnvKey(key)) {
|
|
836
|
+
throw new Error(`bot_token_env ${key} conflicts with a protected provider secret key`);
|
|
837
|
+
}
|
|
838
|
+
}
|
|
839
|
+
}
|
|
716
840
|
/** User-authored fleet.yaml, before defaults are merged into instances. */
|
|
717
841
|
getRawFleetConfig() {
|
|
718
842
|
return structuredClone(this.rawFleetConfig);
|
|
@@ -1116,31 +1240,56 @@ export class FleetManager {
|
|
|
1116
1240
|
*/
|
|
1117
1241
|
getActiveUsageProviderIds() {
|
|
1118
1242
|
const providers = new Set();
|
|
1119
|
-
|
|
1243
|
+
// Per instance, not per backend: two agents on two kiro subscriptions are
|
|
1244
|
+
// two rows, and filtering by the bare backend id would hide both.
|
|
1245
|
+
for (const [name, backend, profile] of this.activeBackendBindings()) {
|
|
1246
|
+
void name;
|
|
1120
1247
|
const provider = usageProviderIdForBackend(backend);
|
|
1121
|
-
if (provider)
|
|
1122
|
-
|
|
1248
|
+
if (!provider)
|
|
1249
|
+
continue;
|
|
1250
|
+
providers.add(profile ? `${provider}:${profile}` : provider);
|
|
1123
1251
|
}
|
|
1124
1252
|
return providers;
|
|
1125
1253
|
}
|
|
1126
|
-
/**
|
|
1127
|
-
|
|
1128
|
-
const
|
|
1129
|
-
const
|
|
1254
|
+
/** Subscription providers used by the running/paused instances owned by one adapter. */
|
|
1255
|
+
getUsageProviderIdsForAdapter(adapterId) {
|
|
1256
|
+
const providers = new Set();
|
|
1257
|
+
for (const [name, backend, profile] of this.activeBackendBindings()) {
|
|
1258
|
+
if (this.getInstanceAdapterId(name) !== adapterId)
|
|
1259
|
+
continue;
|
|
1260
|
+
const provider = usageProviderIdForBackend(backend);
|
|
1261
|
+
if (!provider)
|
|
1262
|
+
continue;
|
|
1263
|
+
providers.add(profile ? `${provider}:${profile}` : provider);
|
|
1264
|
+
}
|
|
1265
|
+
return providers;
|
|
1266
|
+
}
|
|
1267
|
+
/** `[instance, effective backend, credential profile]` for everything that is
|
|
1268
|
+
* running or paused — the one place both usage views agree on who is live. */
|
|
1269
|
+
activeBackendBindings() {
|
|
1270
|
+
const bindings = [];
|
|
1271
|
+
const add = (name, backend, profile) => {
|
|
1130
1272
|
const status = this.getInstanceStatus(name);
|
|
1131
1273
|
if (status !== "running" && status !== "paused")
|
|
1132
1274
|
return;
|
|
1133
1275
|
if (backend)
|
|
1134
|
-
|
|
1276
|
+
bindings.push([name, backend, profile]);
|
|
1135
1277
|
};
|
|
1136
1278
|
for (const [name, config] of Object.entries(this.fleetConfig?.instances ?? {})) {
|
|
1137
1279
|
// loadFleetConfig() has already merged the fleet default into each row.
|
|
1138
|
-
|
|
1280
|
+
const backend = config.backend ?? this.fleetConfig?.defaults?.backend ?? "claude-code";
|
|
1281
|
+
add(name, backend, instanceCredentialProfile(config, this.fleetConfig?.defaults, backend));
|
|
1139
1282
|
}
|
|
1140
1283
|
for (const channel of this.classicChannels?.getAll() ?? []) {
|
|
1141
|
-
|
|
1284
|
+
const backend = this.classicChannels?.getBackendByInstance(channel.instanceName, this.fleetConfig?.defaults?.backend);
|
|
1285
|
+
// Classic channels carry no backend_options, so they run the shared login.
|
|
1286
|
+
add(channel.instanceName, backend, null);
|
|
1142
1287
|
}
|
|
1143
|
-
return
|
|
1288
|
+
return bindings;
|
|
1289
|
+
}
|
|
1290
|
+
/** Effective backends with a running or persisted-paused fleet/Classic instance. */
|
|
1291
|
+
getActiveBackendIds() {
|
|
1292
|
+
return new Set(this.activeBackendBindings().map(([, backend]) => backend));
|
|
1144
1293
|
}
|
|
1145
1294
|
isClassicInstance(name) {
|
|
1146
1295
|
return this.classicChannels?.getAll().some(channel => channel.instanceName === name) ?? false;
|
|
@@ -1548,7 +1697,7 @@ export class FleetManager {
|
|
|
1548
1697
|
if (!wasRunning)
|
|
1549
1698
|
return;
|
|
1550
1699
|
const hotOnly = changedFields.length > 0
|
|
1551
|
-
&& changedFields.every(field => field
|
|
1700
|
+
&& changedFields.every(field => CLASSIC_HOT_CONFIG_KEYS.has(field));
|
|
1552
1701
|
if (hotOnly) {
|
|
1553
1702
|
this.applyHotConfigUpdate(instanceName, this.classicBehaviorUpdate(instanceName));
|
|
1554
1703
|
this.logger.info({ instanceName, fields: changedFields }, "Classic instance hot config reloaded");
|
|
@@ -1945,6 +2094,22 @@ export class FleetManager {
|
|
|
1945
2094
|
names.add(channel.instanceName);
|
|
1946
2095
|
return [...names];
|
|
1947
2096
|
}
|
|
2097
|
+
/**
|
|
2098
|
+
* Probe the same executable set exposed by the web backend catalog.
|
|
2099
|
+
*
|
|
2100
|
+
* This deliberately has no cache: an `/install-cli` completion can add a
|
|
2101
|
+
* binary to PATH while the fleet process remains alive, and the next bare
|
|
2102
|
+
* `/login` must see it without requiring a restart or an explicit cache
|
|
2103
|
+
* invalidation call.
|
|
2104
|
+
*/
|
|
2105
|
+
probeInstalledBackends() {
|
|
2106
|
+
const installed = new Set();
|
|
2107
|
+
for (const [backend, info] of Object.entries(BACKEND_INSTALLATION_INFO)) {
|
|
2108
|
+
if (checkBinaryInstalled(info.binary))
|
|
2109
|
+
installed.add(backend);
|
|
2110
|
+
}
|
|
2111
|
+
return installed;
|
|
2112
|
+
}
|
|
1948
2113
|
/**
|
|
1949
2114
|
* One fleet-level notice per burst, not one per instance: a post-update herd
|
|
1950
2115
|
* fails many instances within the same second. Two notices per incident at
|
|
@@ -2174,7 +2339,9 @@ export class FleetManager {
|
|
|
2174
2339
|
}
|
|
2175
2340
|
/** Initialize auth before any adapter can answer /dashboard. */
|
|
2176
2341
|
initializeWebAuthTokens() {
|
|
2177
|
-
|
|
2342
|
+
// Creates web.token if absent; the value is then read back per request by
|
|
2343
|
+
// the `webToken` getter, so nothing is cached here.
|
|
2344
|
+
loadOrCreateWebToken(this.dataDir);
|
|
2178
2345
|
this.viewToken = randomBytes(24).toString("hex");
|
|
2179
2346
|
const viewTokenPath = join(this.dataDir, "view.token");
|
|
2180
2347
|
writeFileSync(viewTokenPath, this.viewToken, { encoding: "utf8", mode: 0o600 });
|
|
@@ -2738,6 +2905,7 @@ export class FleetManager {
|
|
|
2738
2905
|
}
|
|
2739
2906
|
if (topicMode && (fleet.channel || fleet.channels?.length)) {
|
|
2740
2907
|
await adapterStartup;
|
|
2908
|
+
this.startDiscordUsagePresence();
|
|
2741
2909
|
// Bind every fleet instance deterministically. Explicit channel_id wins;
|
|
2742
2910
|
// otherwise channels[0] is authoritative. Do not infer identity from
|
|
2743
2911
|
// concurrent adapter startup or whichever bot receives a message first.
|
|
@@ -2886,6 +3054,53 @@ export class FleetManager {
|
|
|
2886
3054
|
this.logger.warn({ health }, "Fleet started with problems — see /health");
|
|
2887
3055
|
}
|
|
2888
3056
|
}
|
|
3057
|
+
/** Keep Discord profile activity aligned with the same cached usage source as /usage. */
|
|
3058
|
+
startDiscordUsagePresence() {
|
|
3059
|
+
if (this.discordPresenceTimer)
|
|
3060
|
+
clearInterval(this.discordPresenceTimer);
|
|
3061
|
+
void this.refreshDiscordUsagePresence();
|
|
3062
|
+
this.discordPresenceTimer = setInterval(() => {
|
|
3063
|
+
void this.refreshDiscordUsagePresence();
|
|
3064
|
+
}, FleetManager.DISCORD_PRESENCE_REFRESH_MS);
|
|
3065
|
+
this.discordPresenceTimer.unref?.();
|
|
3066
|
+
}
|
|
3067
|
+
refreshDiscordUsagePresence() {
|
|
3068
|
+
if (this.discordPresenceInFlight)
|
|
3069
|
+
return this.discordPresenceInFlight;
|
|
3070
|
+
const run = (async () => {
|
|
3071
|
+
const targets = [...this.adapters.values()]
|
|
3072
|
+
.filter(adapter => adapter.type === "discord" && typeof adapter.setActivity === "function");
|
|
3073
|
+
if (targets.length === 0)
|
|
3074
|
+
return;
|
|
3075
|
+
try {
|
|
3076
|
+
// Fetch the shared snapshot once, then scope the projection to each
|
|
3077
|
+
// adapter's own fleet/Classic instances. Passing the fleet-wide active
|
|
3078
|
+
// set here would make every bot advertise providers owned by a sibling
|
|
3079
|
+
// bot (notably ClassicBot's Grok/Antigravity rows).
|
|
3080
|
+
const payload = await getUsageSnapshot(false);
|
|
3081
|
+
for (const adapter of targets) {
|
|
3082
|
+
try {
|
|
3083
|
+
const scoped = filterUsageProviders(payload, this.getUsageProviderIdsForAdapter(adapter.id));
|
|
3084
|
+
adapter.setActivity?.(formatDiscordUsageActivity(scoped));
|
|
3085
|
+
}
|
|
3086
|
+
catch {
|
|
3087
|
+
// Presence is cosmetic; a failed update must not affect delivery.
|
|
3088
|
+
}
|
|
3089
|
+
}
|
|
3090
|
+
}
|
|
3091
|
+
catch {
|
|
3092
|
+
// Usage providers are best-effort and may be offline. Keep the last
|
|
3093
|
+
// activity rather than replacing it with an untruthful blank state.
|
|
3094
|
+
this.logger.debug("Discord usage presence refresh skipped");
|
|
3095
|
+
}
|
|
3096
|
+
})();
|
|
3097
|
+
const done = run.finally(() => {
|
|
3098
|
+
if (this.discordPresenceInFlight === done)
|
|
3099
|
+
this.discordPresenceInFlight = null;
|
|
3100
|
+
});
|
|
3101
|
+
this.discordPresenceInFlight = done;
|
|
3102
|
+
return done;
|
|
3103
|
+
}
|
|
2889
3104
|
/**
|
|
2890
3105
|
* Delete inbox files older than retentionDays (by mtime). Cleans the shared
|
|
2891
3106
|
* inbox (`<dataDir>/inbox`) and every workspace inbox
|
|
@@ -3003,6 +3218,11 @@ export class FleetManager {
|
|
|
3003
3218
|
}
|
|
3004
3219
|
bindAdapterHealth(adapter, adapterId) {
|
|
3005
3220
|
adapter.on("gateway_health", (snapshot) => {
|
|
3221
|
+
// A token rotation tears down the old EventEmitter before constructing
|
|
3222
|
+
// the replacement. A late health frame from that old client must never
|
|
3223
|
+
// make a failed/new-generation adapter look connected.
|
|
3224
|
+
if (this.adapters.get(adapterId) !== adapter)
|
|
3225
|
+
return;
|
|
3006
3226
|
const previous = this.adapterState.get(adapterId);
|
|
3007
3227
|
const status = snapshot.status === "connected" ? "connected"
|
|
3008
3228
|
: snapshot.status === "stopped" ? "failed"
|
|
@@ -3083,7 +3303,7 @@ export class FleetManager {
|
|
|
3083
3303
|
};
|
|
3084
3304
|
}
|
|
3085
3305
|
/** Start the primary adapter (backward-compatible, sets this.adapter) */
|
|
3086
|
-
async startSingleAdapter(fleet, channelConfig) {
|
|
3306
|
+
async startSingleAdapter(fleet, channelConfig, onStarted) {
|
|
3087
3307
|
const botToken = process.env[channelConfig.bot_token_env];
|
|
3088
3308
|
if (!botToken) {
|
|
3089
3309
|
this.logger.warn({ env: channelConfig.bot_token_env }, "Bot token env not set, skipping shared adapter");
|
|
@@ -3091,7 +3311,9 @@ export class FleetManager {
|
|
|
3091
3311
|
}
|
|
3092
3312
|
const accessDir = join(this.dataDir, "access");
|
|
3093
3313
|
mkdirSync(accessDir, { recursive: true });
|
|
3094
|
-
const
|
|
3314
|
+
const accessStatePath = join(accessDir, "access.json");
|
|
3315
|
+
const accessManager = new AccessManager(channelConfig.access ?? DEFAULT_OPEN_ACCESS, accessStatePath);
|
|
3316
|
+
this.warnIfAccessModeOverridden(accessManager, accessStatePath);
|
|
3095
3317
|
this.accessManager = accessManager;
|
|
3096
3318
|
const inboxDir = join(this.dataDir, "inbox");
|
|
3097
3319
|
mkdirSync(inboxDir, { recursive: true });
|
|
@@ -3107,13 +3329,20 @@ export class FleetManager {
|
|
|
3107
3329
|
this.worlds.set(adapterId, world);
|
|
3108
3330
|
this.adapters.set(adapterId, adapter);
|
|
3109
3331
|
this.bindAdapterHealth(adapter, adapterId);
|
|
3332
|
+
const isCurrentAdapter = () => this.adapters.get(adapterId) === adapter;
|
|
3110
3333
|
this.adapter.on("message", safeHandler(async (msg) => {
|
|
3334
|
+
if (!isCurrentAdapter())
|
|
3335
|
+
return;
|
|
3111
3336
|
await this.handleInboundMessage(msg);
|
|
3112
3337
|
}, this.logger, "adapter.message"));
|
|
3113
3338
|
this.adapter.on("reaction", safeHandler(async (r) => {
|
|
3339
|
+
if (!isCurrentAdapter())
|
|
3340
|
+
return;
|
|
3114
3341
|
await this.handleInboundReaction(r);
|
|
3115
3342
|
}, this.logger, "adapter.reaction"));
|
|
3116
3343
|
this.adapter.on("callback_query", safeHandler(async (data) => {
|
|
3344
|
+
if (!isCurrentAdapter())
|
|
3345
|
+
return;
|
|
3117
3346
|
if (await this.handleTipDismiss(data, adapterId, this.adapter ?? undefined))
|
|
3118
3347
|
return;
|
|
3119
3348
|
if (await this.handleTipUnlock(data, adapterId, this.adapter ?? undefined))
|
|
@@ -3154,6 +3383,8 @@ export class FleetManager {
|
|
|
3154
3383
|
this.bindTopicClosedHandler(adapter, adapterId, "adapter.topic_closed");
|
|
3155
3384
|
// Handle classic bot slash commands (/start, /stop, /chat, /compact, /save, /load)
|
|
3156
3385
|
this.adapter.on("slash_command", safeHandler(async (data) => {
|
|
3386
|
+
if (!isCurrentAdapter())
|
|
3387
|
+
return;
|
|
3157
3388
|
if (data.command === "start") {
|
|
3158
3389
|
await this.handleClassicStartSlash(data, adapterId);
|
|
3159
3390
|
}
|
|
@@ -3433,6 +3664,8 @@ export class FleetManager {
|
|
|
3433
3664
|
// Non-blocking: /model & status views read the cache; never delays startup.
|
|
3434
3665
|
this.probeCliEnvs();
|
|
3435
3666
|
this.adapter.on("started", safeHandler((username, userId) => {
|
|
3667
|
+
if (!isCurrentAdapter())
|
|
3668
|
+
return;
|
|
3436
3669
|
this.logger.info(`Bot @${username} polling started. Ensure no other service is polling this bot token.`);
|
|
3437
3670
|
// Concurrent startup can insert a secondary world first. Update the
|
|
3438
3671
|
// configured primary world, not Map insertion order.
|
|
@@ -3444,6 +3677,7 @@ export class FleetManager {
|
|
|
3444
3677
|
}
|
|
3445
3678
|
if (userId)
|
|
3446
3679
|
this.botUserId = userId;
|
|
3680
|
+
onStarted?.();
|
|
3447
3681
|
}, this.logger, "adapter.started"));
|
|
3448
3682
|
this.adapter.on("polling_conflict", safeHandler(({ attempt, delay }) => {
|
|
3449
3683
|
this.logger.warn(`409 Conflict (attempt ${attempt}), retry in ${delay / 1000}s`);
|
|
@@ -3452,10 +3686,14 @@ export class FleetManager {
|
|
|
3452
3686
|
this.logger.warn({ err: err instanceof Error ? err.message : String(err) }, "Adapter handler error");
|
|
3453
3687
|
}, this.logger, "adapter.handler_error"));
|
|
3454
3688
|
this.adapter.on("error", (err) => {
|
|
3689
|
+
if (!isCurrentAdapter())
|
|
3690
|
+
return;
|
|
3455
3691
|
this.logger.error({ err }, "Primary adapter fatal error");
|
|
3456
3692
|
this.restartAdapter(this.adapter, adapterId).catch(() => { });
|
|
3457
3693
|
});
|
|
3458
3694
|
this.adapter.on("new_group_detected", safeHandler(async (data) => {
|
|
3695
|
+
if (!isCurrentAdapter())
|
|
3696
|
+
return;
|
|
3459
3697
|
const adminMsg = t("alert.bot_added", data.groupTitle, data.groupId, data.source);
|
|
3460
3698
|
const generalId = this.findGeneralInstance();
|
|
3461
3699
|
// No user to promote: the bot was just added, nobody has run /start yet.
|
|
@@ -3467,6 +3705,9 @@ export class FleetManager {
|
|
|
3467
3705
|
if (fleet.channel?.group_id) {
|
|
3468
3706
|
this.adapter.setChatId(String(fleet.channel.group_id));
|
|
3469
3707
|
}
|
|
3708
|
+
if (this.discordPresenceTimer && this.adapter.type === "discord") {
|
|
3709
|
+
void this.refreshDiscordUsagePresence();
|
|
3710
|
+
}
|
|
3470
3711
|
this.startTopicCleanupPoller();
|
|
3471
3712
|
// Prune stale external sessions every 5 minutes
|
|
3472
3713
|
this.sessionPruneTimer = setInterval(() => {
|
|
@@ -3474,7 +3715,7 @@ export class FleetManager {
|
|
|
3474
3715
|
}, 5 * 60 * 1000);
|
|
3475
3716
|
}
|
|
3476
3717
|
/** Start an additional (non-primary) adapter */
|
|
3477
|
-
async startAdditionalAdapter(channelConfig, registerCommands = true) {
|
|
3718
|
+
async startAdditionalAdapter(channelConfig, registerCommands = true, onStarted) {
|
|
3478
3719
|
const adapterId = channelConfig.id ?? channelConfig.type;
|
|
3479
3720
|
const botToken = process.env[channelConfig.bot_token_env];
|
|
3480
3721
|
if (!botToken) {
|
|
@@ -3483,7 +3724,9 @@ export class FleetManager {
|
|
|
3483
3724
|
}
|
|
3484
3725
|
const accessDir = join(this.dataDir, "access");
|
|
3485
3726
|
mkdirSync(accessDir, { recursive: true });
|
|
3486
|
-
const
|
|
3727
|
+
const accessStatePath = join(accessDir, `access-${adapterId}.json`);
|
|
3728
|
+
const accessManager = new AccessManager(channelConfig.access ?? DEFAULT_OPEN_ACCESS, accessStatePath);
|
|
3729
|
+
this.warnIfAccessModeOverridden(accessManager, accessStatePath);
|
|
3487
3730
|
const inboxDir = join(this.dataDir, "inbox");
|
|
3488
3731
|
mkdirSync(inboxDir, { recursive: true });
|
|
3489
3732
|
const adapter = await createAdapter(channelConfig, {
|
|
@@ -3497,14 +3740,21 @@ export class FleetManager {
|
|
|
3497
3740
|
this.worlds.set(adapterId, world);
|
|
3498
3741
|
this.adapters.set(adapterId, adapter);
|
|
3499
3742
|
this.bindAdapterHealth(adapter, adapterId);
|
|
3743
|
+
const isCurrentAdapter = () => this.adapters.get(adapterId) === adapter;
|
|
3500
3744
|
// Wire up event handlers (same as primary, routes through shared handleInboundMessage)
|
|
3501
3745
|
adapter.on("message", safeHandler(async (msg) => {
|
|
3746
|
+
if (!isCurrentAdapter())
|
|
3747
|
+
return;
|
|
3502
3748
|
await this.handleInboundMessage(msg);
|
|
3503
3749
|
}, this.logger, `adapter[${adapterId}].message`));
|
|
3504
3750
|
adapter.on("reaction", safeHandler(async (r) => {
|
|
3751
|
+
if (!isCurrentAdapter())
|
|
3752
|
+
return;
|
|
3505
3753
|
await this.handleInboundReaction(r);
|
|
3506
3754
|
}, this.logger, `adapter[${adapterId}].reaction`));
|
|
3507
3755
|
adapter.on("callback_query", safeHandler(async (data) => {
|
|
3756
|
+
if (!isCurrentAdapter())
|
|
3757
|
+
return;
|
|
3508
3758
|
if (await this.handleTipDismiss(data, adapterId, adapter))
|
|
3509
3759
|
return;
|
|
3510
3760
|
if (await this.handleTipUnlock(data, adapterId, adapter))
|
|
@@ -3545,6 +3795,8 @@ export class FleetManager {
|
|
|
3545
3795
|
this.bindTopicClosedHandler(adapter, adapterId, `adapter[${adapterId}].topic_closed`);
|
|
3546
3796
|
// Slash commands: classic bot + admin commands
|
|
3547
3797
|
adapter.on("slash_command", safeHandler(async (data) => {
|
|
3798
|
+
if (!isCurrentAdapter())
|
|
3799
|
+
return;
|
|
3548
3800
|
if (data.command === "start") {
|
|
3549
3801
|
await this.handleClassicStartSlash(data, adapterId);
|
|
3550
3802
|
}
|
|
@@ -3768,6 +4020,8 @@ export class FleetManager {
|
|
|
3768
4020
|
}
|
|
3769
4021
|
}, this.logger, `adapter[${adapterId}].slash_command`));
|
|
3770
4022
|
adapter.on("started", safeHandler((username, userId) => {
|
|
4023
|
+
if (!isCurrentAdapter())
|
|
4024
|
+
return;
|
|
3771
4025
|
this.logger.info(`[${adapterId}] Bot @${username} polling started.`);
|
|
3772
4026
|
const world = this.worlds.get(adapterId);
|
|
3773
4027
|
if (world) {
|
|
@@ -3775,14 +4029,19 @@ export class FleetManager {
|
|
|
3775
4029
|
if (userId)
|
|
3776
4030
|
world.botUserId = userId;
|
|
3777
4031
|
}
|
|
4032
|
+
onStarted?.();
|
|
3778
4033
|
}, this.logger, `adapter[${adapterId}].started`));
|
|
3779
4034
|
adapter.on("new_group_detected", safeHandler(async (data) => {
|
|
4035
|
+
if (!isCurrentAdapter())
|
|
4036
|
+
return;
|
|
3780
4037
|
const adminMsg = t("alert.bot_added", data.groupTitle, data.groupId, data.source);
|
|
3781
4038
|
const generalId = this.findGeneralInstance(adapterId);
|
|
3782
4039
|
if (generalId)
|
|
3783
4040
|
await this.promptClassicApproval({ generalName: generalId, message: adminMsg, groupId: data.groupId, scope: data.source === "telegram" ? "group" : "guild" });
|
|
3784
4041
|
}, this.logger, `adapter[${adapterId}].new_group_detected`));
|
|
3785
4042
|
adapter.on("error", (err) => {
|
|
4043
|
+
if (!isCurrentAdapter())
|
|
4044
|
+
return;
|
|
3786
4045
|
this.logger.error({ err, adapterId }, "Additional adapter fatal error");
|
|
3787
4046
|
this.restartAdapter(adapter, adapterId).catch(() => { });
|
|
3788
4047
|
});
|
|
@@ -3791,6 +4050,9 @@ export class FleetManager {
|
|
|
3791
4050
|
if (channelConfig.group_id) {
|
|
3792
4051
|
adapter.setChatId(String(channelConfig.group_id));
|
|
3793
4052
|
}
|
|
4053
|
+
if (this.discordPresenceTimer && adapter.type === "discord") {
|
|
4054
|
+
void this.refreshDiscordUsagePresence();
|
|
4055
|
+
}
|
|
3794
4056
|
this.logger.info({ adapterId, type: channelConfig.type }, "Additional adapter started");
|
|
3795
4057
|
}
|
|
3796
4058
|
/** Connect IPC to a single instance with all handlers */
|
|
@@ -3861,22 +4123,15 @@ export class FleetManager {
|
|
|
3861
4123
|
}
|
|
3862
4124
|
await this.handleOutboundFromInstance(name, msg);
|
|
3863
4125
|
}
|
|
3864
|
-
else if (msg.type
|
|
3865
|
-
|
|
3866
|
-
|
|
3867
|
-
|
|
3868
|
-
|
|
3869
|
-
|
|
3870
|
-
|
|
3871
|
-
|
|
3872
|
-
|
|
3873
|
-
this.handleTaskCrud(name, msg);
|
|
3874
|
-
}
|
|
3875
|
-
else if (msg.type === "fleet_set_display_name") {
|
|
3876
|
-
this.handleSetDisplayName(name, msg);
|
|
3877
|
-
}
|
|
3878
|
-
else if (msg.type === "fleet_set_description") {
|
|
3879
|
-
this.handleSetDescription(name, msg);
|
|
4126
|
+
else if (toolForIpcType(msg.type) !== null) {
|
|
4127
|
+
// Sink 2 of 3. Each of these types IS a tool and reaches its own
|
|
4128
|
+
// handler without passing through the outbound path — which is how
|
|
4129
|
+
// `update_decision`, a coordinator-only tool, kept a way through.
|
|
4130
|
+
const typed = this.checkToolPermission("ipc-typed", name, toolForIpcType(msg.type));
|
|
4131
|
+
if (typed.allowed)
|
|
4132
|
+
this.dispatchTypedIpc(name, msg);
|
|
4133
|
+
else
|
|
4134
|
+
this.refuseTypedIpc(name, msg, typed.message);
|
|
3880
4135
|
}
|
|
3881
4136
|
else if (msg.type === "instance_process_state") {
|
|
3882
4137
|
this.cacheInstanceProcessStatus(name, msg.status);
|
|
@@ -4040,6 +4295,9 @@ export class FleetManager {
|
|
|
4040
4295
|
// watchdog/manual/error triggers must converge here instead of stop/start.
|
|
4041
4296
|
await adapter.reconnectGateway(previous?.lastError ?? "fleet adapter restart");
|
|
4042
4297
|
this.adapterState.set(id, { status: "connected", retryCount: 0 });
|
|
4298
|
+
if (this.discordPresenceTimer && adapter.type === "discord") {
|
|
4299
|
+
void this.refreshDiscordUsagePresence();
|
|
4300
|
+
}
|
|
4043
4301
|
this.logger.info({ id }, "Adapter gateway rebuilt successfully");
|
|
4044
4302
|
}
|
|
4045
4303
|
catch (err) {
|
|
@@ -4065,6 +4323,9 @@ export class FleetManager {
|
|
|
4065
4323
|
await adapter.start();
|
|
4066
4324
|
this.logger.info({ id, attempt }, "Adapter restarted successfully");
|
|
4067
4325
|
this.adapterState.set(id, { status: "connected", retryCount: 0 });
|
|
4326
|
+
if (this.discordPresenceTimer && adapter.type === "discord") {
|
|
4327
|
+
void this.refreshDiscordUsagePresence();
|
|
4328
|
+
}
|
|
4068
4329
|
return;
|
|
4069
4330
|
}
|
|
4070
4331
|
catch (err) {
|
|
@@ -4245,13 +4506,36 @@ export class FleetManager {
|
|
|
4245
4506
|
const target = this.routing.resolve(threadId);
|
|
4246
4507
|
if (!target)
|
|
4247
4508
|
return "fleet topic: no instance routed for this thread";
|
|
4248
|
-
// Fleet topic:
|
|
4509
|
+
// Fleet topic: this gate lets the copy through when the adapter is open OR
|
|
4510
|
+
// collab is on for the instance. "Through" is not "delivered" — access
|
|
4511
|
+
// control runs next, and the two arms are not symmetric there:
|
|
4512
|
+
//
|
|
4513
|
+
// open adapter → isAllowed() returns true outright: really admitted.
|
|
4514
|
+
// locked + collab → the bot's user id still has to be on the allowlist,
|
|
4515
|
+
// so the message is normally refused a few lines later
|
|
4516
|
+
// under "Access DENIED for non-allowed user".
|
|
4517
|
+
//
|
|
4518
|
+
// Collab alone therefore does not admit a bot on a locked adapter; it only
|
|
4519
|
+
// declines to drop the copy here and leaves the decision to access control.
|
|
4249
4520
|
const isOpen = this.getChannelConfig(msg.adapterId)?.access?.mode === "open";
|
|
4250
4521
|
if (!isOpen && !this.collabInstances.has(target.name)) {
|
|
4251
4522
|
return `fleet topic: adapter not open and collab off for ${target.name}`;
|
|
4252
4523
|
}
|
|
4253
4524
|
return null;
|
|
4254
4525
|
}
|
|
4526
|
+
/**
|
|
4527
|
+
* Say so when an adapter's access mode is coming from its state file rather
|
|
4528
|
+
* than from fleet.yaml. The state file wins by design — a pairing done at
|
|
4529
|
+
* runtime has to survive a restart — but that also means an edit to
|
|
4530
|
+
* `access.mode` silently does nothing, which is indistinguishable from the
|
|
4531
|
+
* fleet not having reloaded. Naming the file is what makes it fixable.
|
|
4532
|
+
*/
|
|
4533
|
+
warnIfAccessModeOverridden(accessManager, statePath) {
|
|
4534
|
+
const configured = accessManager.overriddenConfigMode();
|
|
4535
|
+
if (!configured)
|
|
4536
|
+
return;
|
|
4537
|
+
this.logger.warn({ statePath, configured, inEffect: accessManager.getMode() }, `access mode "${accessManager.getMode()}" comes from the state file and overrides "${configured}" in fleet.yaml; delete ${statePath} to go back to the configured value`);
|
|
4538
|
+
}
|
|
4255
4539
|
async handleInboundMessage(msg) {
|
|
4256
4540
|
const threadId = msg.threadId || undefined;
|
|
4257
4541
|
this.logger.debug({ source: msg.source, chatId: msg.chatId, threadId, userId: msg.userId, isBotMessage: msg.isBotMessage, textLen: (msg.text ?? "").length, text: (msg.text ?? "").slice(0, 80) }, "handleInboundMessage entry");
|
|
@@ -4818,6 +5102,55 @@ export class FleetManager {
|
|
|
4818
5102
|
}
|
|
4819
5103
|
}
|
|
4820
5104
|
/** Handle outbound tool calls from a daemon instance */
|
|
5105
|
+
/**
|
|
5106
|
+
* Would this instance be allowed to use this tool?
|
|
5107
|
+
*
|
|
5108
|
+
* Stage 1 asks and records; nothing is refused yet. The recording is not only
|
|
5109
|
+
* an observation window: `logActivity("tool_call")` sits on the outbound path
|
|
5110
|
+
* only, so today the fleet has no idea what the agent endpoint or the typed
|
|
5111
|
+
* IPC handlers are being asked to do — the one face with no authorization is
|
|
5112
|
+
* also the one face with no telemetry.
|
|
5113
|
+
*
|
|
5114
|
+
* The profile comes from the instance the socket belongs to. Deliberately not
|
|
5115
|
+
* `senderSessionName`, which arrives inside the message: a caller that fills
|
|
5116
|
+
* in its own identity has not been identified.
|
|
5117
|
+
*/
|
|
5118
|
+
/**
|
|
5119
|
+
* Say once, at startup, what the new default means for this fleet.
|
|
5120
|
+
*
|
|
5121
|
+
* Nothing is rewritten: an explicit `tool_set: full` is a choice somebody
|
|
5122
|
+
* made, and marking an instance as a coordinator is a judgement about how
|
|
5123
|
+
* their fleet is organised. Both stay theirs — this only makes sure they are
|
|
5124
|
+
* not discovered by an agent failing at three in the morning.
|
|
5125
|
+
*/
|
|
5126
|
+
announceToolPermissionsChange() {
|
|
5127
|
+
try {
|
|
5128
|
+
const thirtyDaysAgo = new Date(Date.now() - 30 * 864e5).toISOString().slice(0, 19).replace("T", " ");
|
|
5129
|
+
const recent = [...(this.eventLog?.toolUseByInstance(thirtyDaysAgo) ?? new Map())]
|
|
5130
|
+
.map(([instance, tools]) => ({ instance, tools }));
|
|
5131
|
+
const notice = buildToolPermissionsNotice({
|
|
5132
|
+
defaultsToolSet: this.fleetConfig?.defaults?.tool_set,
|
|
5133
|
+
instances: (this.fleetConfig?.instances ?? {}),
|
|
5134
|
+
recent,
|
|
5135
|
+
});
|
|
5136
|
+
if (notice)
|
|
5137
|
+
this.logger.warn({ notice }, notice);
|
|
5138
|
+
}
|
|
5139
|
+
catch (err) {
|
|
5140
|
+
// Advice is not worth failing a startup over.
|
|
5141
|
+
this.logger.debug({ err }, "tool-permissions: could not build the migration notice");
|
|
5142
|
+
}
|
|
5143
|
+
}
|
|
5144
|
+
checkToolPermission(sink, instanceName, tool) {
|
|
5145
|
+
const profile = resolveToolSet(this.fleetConfig?.instances[instanceName], instanceName);
|
|
5146
|
+
const allowed = mayUseTool(profile, tool);
|
|
5147
|
+
if (!allowed) {
|
|
5148
|
+
this.logger.warn({ sink, instance: instanceName, profile, tool, enforced: true }, "tool-permissions: refused");
|
|
5149
|
+
return { allowed: false, message: toolRefusedMessage(profile, tool) };
|
|
5150
|
+
}
|
|
5151
|
+
this.logger.debug({ sink, instance: instanceName, profile, tool, enforced: true }, "tool-permissions: allowed");
|
|
5152
|
+
return { allowed: true, message: "" };
|
|
5153
|
+
}
|
|
4821
5154
|
async handleOutboundFromInstance(instanceName, msg) {
|
|
4822
5155
|
this.touchActivity(instanceName);
|
|
4823
5156
|
this.setTopicIcon(instanceName, "green");
|
|
@@ -4839,6 +5172,19 @@ export class FleetManager {
|
|
|
4839
5172
|
this.logger.warn({ instanceName, tool, requestId, fleetRequestId, error }, "Fleet outbound result could not be returned — instance IPC is disconnected");
|
|
4840
5173
|
}
|
|
4841
5174
|
};
|
|
5175
|
+
// Sink 1 of 3, and the first thing decided. Every MCP call and every direct
|
|
5176
|
+
// write to channel.sock lands here, whether or not mcp-server was ever
|
|
5177
|
+
// involved — which is why this is the boundary and the tool list the model
|
|
5178
|
+
// was shown is not.
|
|
5179
|
+
//
|
|
5180
|
+
// Above the adapter check on purpose: "retry shortly" is the wrong answer
|
|
5181
|
+
// to a call that will never be allowed, and an agent that believes it is a
|
|
5182
|
+
// timing problem will keep trying.
|
|
5183
|
+
const permitted = this.checkToolPermission("ipc-outbound", instanceName, tool);
|
|
5184
|
+
if (!permitted.allowed) {
|
|
5185
|
+
respond(null, permitted.message);
|
|
5186
|
+
return;
|
|
5187
|
+
}
|
|
4842
5188
|
if (this.worlds.size === 0) {
|
|
4843
5189
|
respond(null, "Channel adapters are not ready — retry shortly");
|
|
4844
5190
|
return;
|
|
@@ -5096,6 +5442,15 @@ export class FleetManager {
|
|
|
5096
5442
|
*/
|
|
5097
5443
|
scheduleSourceAdapter(schedule) {
|
|
5098
5444
|
const chatId = String(schedule.reply_chat_id);
|
|
5445
|
+
// New schedules carry the adapter that owned the source chat at creation.
|
|
5446
|
+
// Keep that persona across source-instance rebinding, but never trust a
|
|
5447
|
+
// stale adapter after the chat has moved to another world.
|
|
5448
|
+
const persistedWorld = schedule.reply_adapter_id
|
|
5449
|
+
? this.worlds.get(schedule.reply_adapter_id)
|
|
5450
|
+
: undefined;
|
|
5451
|
+
if (persistedWorld && String(persistedWorld.groupId) === chatId) {
|
|
5452
|
+
return persistedWorld.adapter;
|
|
5453
|
+
}
|
|
5099
5454
|
const sourceAdapterId = schedule.source ? this.getInstanceAdapterId(schedule.source) : undefined;
|
|
5100
5455
|
const sourceWorld = sourceAdapterId ? this.worlds.get(sourceAdapterId) : undefined;
|
|
5101
5456
|
if (sourceWorld && String(sourceWorld.groupId) === chatId)
|
|
@@ -5127,6 +5482,58 @@ export class FleetManager {
|
|
|
5127
5482
|
threadId: schedule.reply_thread_id ?? undefined,
|
|
5128
5483
|
}).catch((err) => this.logger.error({ err }, "Failed to send schedule failure notification"));
|
|
5129
5484
|
}
|
|
5485
|
+
/**
|
|
5486
|
+
* The typed IPC messages, all through one door.
|
|
5487
|
+
*
|
|
5488
|
+
* They used to be five sibling branches on the dispatch, which is why the
|
|
5489
|
+
* permission question had five places to be forgotten. Routing them together
|
|
5490
|
+
* means the check above happens once and cannot be skipped by adding a
|
|
5491
|
+
* sixth — a new type has to appear in `IPC_TYPE_TOOLS` to be dispatched at
|
|
5492
|
+
* all.
|
|
5493
|
+
*/
|
|
5494
|
+
/**
|
|
5495
|
+
* Answer a refused typed message the way its handler would have.
|
|
5496
|
+
*
|
|
5497
|
+
* These are request/response over IPC: dropping the message silently leaves
|
|
5498
|
+
* the caller waiting for a reply that never comes, and a hung agent is a
|
|
5499
|
+
* worse failure than a refused one.
|
|
5500
|
+
*/
|
|
5501
|
+
refuseTypedIpc(name, msg, message) {
|
|
5502
|
+
const ipc = this.instanceIpcClients.get(name);
|
|
5503
|
+
const fleetRequestId = msg.fleetRequestId;
|
|
5504
|
+
if (!ipc || !fleetRequestId)
|
|
5505
|
+
return;
|
|
5506
|
+
const type = String(msg.type);
|
|
5507
|
+
const responseType = type.startsWith("fleet_schedule_") ? "fleet_schedule_response"
|
|
5508
|
+
: type.startsWith("fleet_decision_") ? "fleet_decision_response"
|
|
5509
|
+
: type === "fleet_task" ? "fleet_task_response"
|
|
5510
|
+
: type === "fleet_set_display_name" ? "fleet_display_name_response"
|
|
5511
|
+
: "fleet_description_response";
|
|
5512
|
+
ipc.send({ type: responseType, fleetRequestId, error: message });
|
|
5513
|
+
}
|
|
5514
|
+
dispatchTypedIpc(name, msg) {
|
|
5515
|
+
const type = String(msg.type);
|
|
5516
|
+
if (type.startsWith("fleet_schedule_")) {
|
|
5517
|
+
this.handleScheduleCrud(name, msg);
|
|
5518
|
+
return;
|
|
5519
|
+
}
|
|
5520
|
+
if (type.startsWith("fleet_decision_")) {
|
|
5521
|
+
this.handleDecisionCrud(name, msg);
|
|
5522
|
+
return;
|
|
5523
|
+
}
|
|
5524
|
+
if (type === "fleet_task") {
|
|
5525
|
+
this.handleTaskCrud(name, msg);
|
|
5526
|
+
return;
|
|
5527
|
+
}
|
|
5528
|
+
if (type === "fleet_set_display_name") {
|
|
5529
|
+
this.handleSetDisplayName(name, msg);
|
|
5530
|
+
return;
|
|
5531
|
+
}
|
|
5532
|
+
if (type === "fleet_set_description") {
|
|
5533
|
+
this.handleSetDescription(name, msg);
|
|
5534
|
+
return;
|
|
5535
|
+
}
|
|
5536
|
+
}
|
|
5130
5537
|
handleScheduleCrud(instanceName, msg) {
|
|
5131
5538
|
const fleetRequestId = msg.fleetRequestId;
|
|
5132
5539
|
const payload = (msg.payload ?? {});
|
|
@@ -5134,36 +5541,27 @@ export class FleetManager {
|
|
|
5134
5541
|
const ipc = this.instanceIpcClients.get(instanceName);
|
|
5135
5542
|
if (!ipc)
|
|
5136
5543
|
return;
|
|
5544
|
+
if (!this.scheduler) {
|
|
5545
|
+
// It did answer before, with whatever TypeError fell out of the try
|
|
5546
|
+
// block — "Cannot read properties of null (reading 'list')" is a stack
|
|
5547
|
+
// trace wearing an error message, and the agent reading it cannot tell
|
|
5548
|
+
// that the fleet simply has no scheduler.
|
|
5549
|
+
ipc.send({ type: "fleet_schedule_response", fleetRequestId, error: "Schedules are unavailable — the fleet scheduler is not running" });
|
|
5550
|
+
return;
|
|
5551
|
+
}
|
|
5137
5552
|
try {
|
|
5138
|
-
|
|
5139
|
-
|
|
5140
|
-
|
|
5141
|
-
|
|
5142
|
-
|
|
5143
|
-
|
|
5144
|
-
|
|
5145
|
-
|
|
5146
|
-
|
|
5147
|
-
|
|
5148
|
-
|
|
5149
|
-
|
|
5150
|
-
timezone: payload.timezone,
|
|
5151
|
-
silent: !!(payload.silent),
|
|
5152
|
-
};
|
|
5153
|
-
result = this.scheduler.create(params);
|
|
5154
|
-
break;
|
|
5155
|
-
}
|
|
5156
|
-
case "fleet_schedule_list":
|
|
5157
|
-
result = this.scheduler.list(payload.target);
|
|
5158
|
-
break;
|
|
5159
|
-
case "fleet_schedule_update":
|
|
5160
|
-
result = this.scheduler.update(payload.id, payload);
|
|
5161
|
-
break;
|
|
5162
|
-
case "fleet_schedule_delete":
|
|
5163
|
-
this.scheduler.delete(payload.id);
|
|
5164
|
-
result = "ok";
|
|
5165
|
-
break;
|
|
5166
|
-
}
|
|
5553
|
+
const op = String(msg.type).replace("fleet_schedule_", "");
|
|
5554
|
+
const result = this.performScheduleOp(instanceName, op, payload, {
|
|
5555
|
+
// The daemon sends its last chat id, which is unset until the instance
|
|
5556
|
+
// has had a chat message — and cross-instance traffic never sets it. So
|
|
5557
|
+
// a worker that only takes delegated tasks, the very instance #895 lets
|
|
5558
|
+
// self-schedule, hit "NOT NULL constraint failed: schedules.reply_chat_id".
|
|
5559
|
+
// No chat means no reply chat, exactly as on the agent endpoint.
|
|
5560
|
+
chatId: meta.chat_id ?? "",
|
|
5561
|
+
threadId: meta.thread_id || null,
|
|
5562
|
+
adapterId: meta.adapter_id || null,
|
|
5563
|
+
silent: !!(payload.silent),
|
|
5564
|
+
});
|
|
5167
5565
|
ipc.send({ type: "fleet_schedule_response", fleetRequestId, result });
|
|
5168
5566
|
}
|
|
5169
5567
|
catch (err) {
|
|
@@ -5175,8 +5573,15 @@ export class FleetManager {
|
|
|
5175
5573
|
const payload = (msg.payload ?? {});
|
|
5176
5574
|
const meta = (msg.meta ?? {});
|
|
5177
5575
|
const ipc = this.instanceIpcClients.get(instanceName);
|
|
5178
|
-
if (!ipc
|
|
5576
|
+
if (!ipc)
|
|
5179
5577
|
return;
|
|
5578
|
+
if (!this.scheduler) {
|
|
5579
|
+
// Returning silently left the caller waiting for a response that was
|
|
5580
|
+
// never coming. Over IPC that is a hang, and a hang is a worse answer
|
|
5581
|
+
// than an error.
|
|
5582
|
+
ipc.send({ type: "fleet_decision_response", fleetRequestId, error: "Decisions are unavailable — the fleet scheduler is not running" });
|
|
5583
|
+
return;
|
|
5584
|
+
}
|
|
5180
5585
|
const db = this.scheduler.db;
|
|
5181
5586
|
const projectRoot = meta.working_directory || this.fleetConfig?.instances[instanceName]?.working_directory || "";
|
|
5182
5587
|
try {
|
|
@@ -5292,23 +5697,76 @@ export class FleetManager {
|
|
|
5292
5697
|
async handleScheduleCrudHttp(instance, op, args) {
|
|
5293
5698
|
if (!this.scheduler)
|
|
5294
5699
|
return { error: "Scheduler not available" };
|
|
5700
|
+
if (op !== "create" && op !== "list" && op !== "update" && op !== "delete") {
|
|
5701
|
+
return { error: `Unknown schedule op: ${op}` };
|
|
5702
|
+
}
|
|
5703
|
+
// No bound chat on this path, as before: a schedule made through the agent
|
|
5704
|
+
// endpoint has nowhere of its own to reply.
|
|
5705
|
+
return this.performScheduleOp(instance, op, args, { chatId: "", threadId: null });
|
|
5706
|
+
}
|
|
5707
|
+
/**
|
|
5708
|
+
* The only place an agent's request creates, changes or removes a schedule
|
|
5709
|
+
* (#895). The IPC handler (MCP calls, direct channel.sock writes) and the
|
|
5710
|
+
* agent endpoint (agent-cli, HTTP agent mode) are adapters over this, so the
|
|
5711
|
+
* target check cannot be present on one face and missing on the other — the
|
|
5712
|
+
* shape #804 had to close twice.
|
|
5713
|
+
*
|
|
5714
|
+
* `caller` is the instance the server resolved for the request; any
|
|
5715
|
+
* `source` in `args` is ignored, and a schedule's source is always its caller.
|
|
5716
|
+
* A refusal is a ToolNotPermittedError: 403 on the agent endpoint, the error
|
|
5717
|
+
* of the schedule response on IPC.
|
|
5718
|
+
*/
|
|
5719
|
+
performScheduleOp(caller, op, args, reply) {
|
|
5720
|
+
const scheduler = this.scheduler;
|
|
5721
|
+
// Shape first, so the value the permission decision reads is the value the
|
|
5722
|
+
// scheduler gets. `target` used to be filtered through typeof for the
|
|
5723
|
+
// decision and passed raw to the scheduler, whose `target.startsWith`
|
|
5724
|
+
// then threw a TypeError for `target: 123` (#897). A non-string is refused
|
|
5725
|
+
// — never coerced: turning 123 into "123" would decide an identity
|
|
5726
|
+
// question on a value the caller did not send. `null` means absent, as
|
|
5727
|
+
// omitting it always did, and is removed before the scheduler sees it.
|
|
5728
|
+
if (args.target === null) {
|
|
5729
|
+
args = { ...args };
|
|
5730
|
+
delete args.target;
|
|
5731
|
+
}
|
|
5732
|
+
if (args.target !== undefined && typeof args.target !== "string") {
|
|
5733
|
+
throw new Error(`${op}_schedule: "target" must be an instance name (a string), not ${typeof args.target}.`);
|
|
5734
|
+
}
|
|
5735
|
+
if ((op === "update" || op === "delete") && typeof args.id !== "string") {
|
|
5736
|
+
throw new Error(`${op}_schedule: "id" must be a schedule id (a string) — get one from list_schedules.`);
|
|
5737
|
+
}
|
|
5738
|
+
const requestedTarget = args.target;
|
|
5739
|
+
if (op !== "list") {
|
|
5740
|
+
const profile = resolveToolSet(this.fleetConfig?.instances[caller], caller);
|
|
5741
|
+
const existing = op === "create" ? null : scheduler.get(args.id);
|
|
5742
|
+
const refusal = scheduleOpRefusal(profile, caller, op, { requestedTarget, existing });
|
|
5743
|
+
if (refusal) {
|
|
5744
|
+
this.logger.warn({ instance: caller, profile, op, target: requestedTarget ?? existing?.target, scheduleId: args.id }, "tool-permissions: schedule op refused");
|
|
5745
|
+
throw new ToolNotPermittedError(refusal);
|
|
5746
|
+
}
|
|
5747
|
+
}
|
|
5295
5748
|
switch (op) {
|
|
5296
5749
|
case "create":
|
|
5297
|
-
return
|
|
5750
|
+
return scheduler.create({
|
|
5298
5751
|
cron: args.cron,
|
|
5299
5752
|
at: args.at,
|
|
5300
5753
|
message: args.message,
|
|
5301
|
-
source:
|
|
5302
|
-
|
|
5754
|
+
source: caller,
|
|
5755
|
+
target: requestedTarget || caller,
|
|
5756
|
+
reply_chat_id: reply.chatId,
|
|
5757
|
+
reply_thread_id: reply.threadId,
|
|
5758
|
+
...(reply.adapterId !== undefined ? { reply_adapter_id: reply.adapterId } : {}),
|
|
5303
5759
|
label: args.label,
|
|
5304
5760
|
timezone: args.timezone,
|
|
5761
|
+
...(reply.silent !== undefined ? { silent: reply.silent } : {}),
|
|
5305
5762
|
});
|
|
5306
|
-
case "list":
|
|
5307
|
-
|
|
5763
|
+
case "list":
|
|
5764
|
+
return scheduler.list(requestedTarget);
|
|
5765
|
+
case "update":
|
|
5766
|
+
return scheduler.update(args.id, args);
|
|
5308
5767
|
case "delete":
|
|
5309
|
-
|
|
5768
|
+
scheduler.delete(args.id);
|
|
5310
5769
|
return "ok";
|
|
5311
|
-
default: return { error: `Unknown schedule op: ${op}` };
|
|
5312
5770
|
}
|
|
5313
5771
|
}
|
|
5314
5772
|
async handleDecisionCrudHttp(instance, op, args) {
|
|
@@ -5450,8 +5908,15 @@ export class FleetManager {
|
|
|
5450
5908
|
const payload = (msg.payload ?? {});
|
|
5451
5909
|
const meta = (msg.meta ?? {});
|
|
5452
5910
|
const ipc = this.instanceIpcClients.get(instanceName);
|
|
5453
|
-
if (!ipc
|
|
5911
|
+
if (!ipc)
|
|
5912
|
+
return;
|
|
5913
|
+
if (!this.scheduler) {
|
|
5914
|
+
// Returning silently left the caller waiting for a response that was
|
|
5915
|
+
// never coming. Over IPC that is a hang, and a hang is a worse answer
|
|
5916
|
+
// than an error.
|
|
5917
|
+
ipc.send({ type: "fleet_task_response", fleetRequestId, error: "The task board is unavailable — the fleet scheduler is not running" });
|
|
5454
5918
|
return;
|
|
5919
|
+
}
|
|
5455
5920
|
const db = this.scheduler.db;
|
|
5456
5921
|
const action = payload.action;
|
|
5457
5922
|
try {
|
|
@@ -6799,6 +7264,39 @@ export class FleetManager {
|
|
|
6799
7264
|
* (or an assist for one they stopped) must not stay clickable for the rest
|
|
6800
7265
|
* of its 15 minutes.
|
|
6801
7266
|
*/
|
|
7267
|
+
/**
|
|
7268
|
+
* Collapse every still-armed button prompt, for a fleet that is going away.
|
|
7269
|
+
*
|
|
7270
|
+
* The map is in memory; the buttons are in the chat for up to 24 hours. A
|
|
7271
|
+
* restart therefore leaves them looking live — pressing one takes the stale
|
|
7272
|
+
* branch, which acknowledges the click and drops the dismissal without
|
|
7273
|
+
* saying so. Retiring them here means the user meets a spent prompt instead
|
|
7274
|
+
* of a live-looking dead one.
|
|
7275
|
+
*
|
|
7276
|
+
* Best effort and bounded. Each collapse is a platform call, a day of tips
|
|
7277
|
+
* can be many of them, and shutdown still has instances to stop. Whatever
|
|
7278
|
+
* has not finished by the deadline is abandoned — that is exactly today's
|
|
7279
|
+
* behaviour, so the budget can only leave things no worse than before.
|
|
7280
|
+
*/
|
|
7281
|
+
async retirePendingNoncePrompts(budgetMs = NONCE_RETIRE_BUDGET_MS) {
|
|
7282
|
+
const entries = [...this.pendingNonceButtons.values()];
|
|
7283
|
+
this.pendingNonceButtons.clear();
|
|
7284
|
+
for (const entry of entries)
|
|
7285
|
+
if (entry.timer)
|
|
7286
|
+
clearTimeout(entry.timer);
|
|
7287
|
+
const collapses = entries
|
|
7288
|
+
.filter(entry => entry.messageId && entry.adapter.editMessageRemoveButtons)
|
|
7289
|
+
.map(entry => entry.adapter.editMessageRemoveButtons(entry.chatId, entry.messageId, entry.expiredText, entry.threadId).catch(err => this.logger.debug({ err, instanceName: entry.instanceName, prefix: entry.prefix }, "Failed to retire button prompt during shutdown")));
|
|
7290
|
+
if (!collapses.length)
|
|
7291
|
+
return;
|
|
7292
|
+
await Promise.race([
|
|
7293
|
+
Promise.allSettled(collapses),
|
|
7294
|
+
new Promise(resolve => {
|
|
7295
|
+
const timer = setTimeout(resolve, budgetMs);
|
|
7296
|
+
timer.unref?.();
|
|
7297
|
+
}),
|
|
7298
|
+
]);
|
|
7299
|
+
}
|
|
6802
7300
|
clearNoncePromptsForInstance(instanceName) {
|
|
6803
7301
|
for (const [nonce, entry] of this.pendingNonceButtons) {
|
|
6804
7302
|
if (entry.instanceName !== instanceName)
|
|
@@ -7739,11 +8237,39 @@ export class FleetManager {
|
|
|
7739
8237
|
for (const name of this.configuredBackendInstanceNames()) {
|
|
7740
8238
|
configured.add(this.backendNameOf(name));
|
|
7741
8239
|
}
|
|
7742
|
-
const
|
|
7743
|
-
|
|
7744
|
-
|
|
8240
|
+
const installed = this.probeInstalledBackends();
|
|
8241
|
+
const candidates = new Set([...installed, ...configured]);
|
|
8242
|
+
const unsupported = [];
|
|
8243
|
+
const choices = [...candidates].sort().flatMap(backend => {
|
|
8244
|
+
const flow = LOGIN_FLOWS[backend];
|
|
8245
|
+
const remoteLogin = !!flow && flow.remoteLogin !== "unsupported";
|
|
8246
|
+
const status = [];
|
|
8247
|
+
if (installed.has(backend))
|
|
8248
|
+
status.push(t("login.status_installed"));
|
|
8249
|
+
if (configured.has(backend))
|
|
8250
|
+
status.push(t("login.status_configured"));
|
|
8251
|
+
if (remoteLogin)
|
|
8252
|
+
status.push(t("login.status_auth"));
|
|
8253
|
+
else
|
|
8254
|
+
status.push(t("login.status_unsupported"));
|
|
8255
|
+
if (!remoteLogin) {
|
|
8256
|
+
unsupported.push({ backend, flow, status });
|
|
8257
|
+
return [];
|
|
8258
|
+
}
|
|
8259
|
+
return [{ action: backend, label: `${backend} · ${status.join(" · ")}` }];
|
|
8260
|
+
});
|
|
8261
|
+
if (unsupported.length) {
|
|
8262
|
+
const guidance = unsupported.map(({ backend, flow, status }) => `${backend} · ${status.join(" · ")} — ${flow?.remoteLogin === "unsupported"
|
|
8263
|
+
? t("login.remote_unsupported_agent_cli", backend, flow.command)
|
|
8264
|
+
: backend === "opencode" ? t("login.unsupported", backend) : t("login.no_remote_flow", backend)}`).join("\n");
|
|
8265
|
+
await chat.adapter.sendText(chat.chatId, guidance, { threadId: chat.threadId })
|
|
8266
|
+
.catch(err => this.logger.warn({ err }, "Failed to post unsupported login guidance"));
|
|
8267
|
+
}
|
|
7745
8268
|
if (choices.length === 0) {
|
|
7746
|
-
|
|
8269
|
+
if (!unsupported.length) {
|
|
8270
|
+
await chat.adapter.sendText(chat.chatId, t("login.none_available"), { threadId: chat.threadId })
|
|
8271
|
+
.catch(err => this.logger.warn({ err }, "Failed to post empty login chooser guidance"));
|
|
8272
|
+
}
|
|
7747
8273
|
return;
|
|
7748
8274
|
}
|
|
7749
8275
|
await this.postNonceButtonPrompt({
|
|
@@ -8332,6 +8858,12 @@ export class FleetManager {
|
|
|
8332
8858
|
await chat.adapter.sendText(chat.chatId, t("install.success_no_login", backend), { threadId: chat.threadId }).catch(() => { });
|
|
8333
8859
|
return;
|
|
8334
8860
|
}
|
|
8861
|
+
// The button is deliberately best-effort: it is a short-lived
|
|
8862
|
+
// capability and can be lost during an adapter reconnect. Always
|
|
8863
|
+
// publish a durable completion line first so a successful install can
|
|
8864
|
+
// never look like it silently disappeared. Include the binary that
|
|
8865
|
+
// passed the fresh-login-shell verification and the exact next step.
|
|
8866
|
+
await chat.adapter.sendText(chat.chatId, t("install.success", backend, info.binary), { threadId: chat.threadId }).catch(err => this.logger.warn({ err, backend }, "Failed to send durable install success notification"));
|
|
8335
8867
|
await this.postNonceButtonPrompt({
|
|
8336
8868
|
prefix: INSTALL_LOGIN_CALLBACK_PREFIX,
|
|
8337
8869
|
alertType: "login",
|
|
@@ -8583,6 +9115,8 @@ export class FleetManager {
|
|
|
8583
9115
|
// Grok reads AGENTS.md project docs; agy reads .agents/agents.md — the
|
|
8584
9116
|
// same files their writeConfig() appends fleet instructions to.
|
|
8585
9117
|
"grok": "AGENTS.md",
|
|
9118
|
+
// muse init scaffolds AGENTS.md and the binary reads it as project rules.
|
|
9119
|
+
"muse": "AGENTS.md",
|
|
8586
9120
|
"antigravity": ".agents/agents.md",
|
|
8587
9121
|
"mock": "CLAUDE.md",
|
|
8588
9122
|
};
|
|
@@ -8658,6 +9192,9 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
8658
9192
|
// vendor-canonical location is .grok/skills.
|
|
8659
9193
|
"opencode": [".agents", "skills"],
|
|
8660
9194
|
"grok": [".grok", "skills"],
|
|
9195
|
+
// Live-verified on muse 1.3.0: `muse skills list --source project` sees
|
|
9196
|
+
// .agents/skills and ignores .muse/skills.
|
|
9197
|
+
"muse": [".agents", "skills"],
|
|
8661
9198
|
"antigravity": [".agents", "skills"],
|
|
8662
9199
|
};
|
|
8663
9200
|
/** Copy general-knowledge steering + all role-eligible skills to General. */
|
|
@@ -9242,8 +9779,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9242
9779
|
* A model list is an aid: a vendor that stops answering must degrade to the
|
|
9243
9780
|
* previous list, never stall the command that asked for it.
|
|
9244
9781
|
*/
|
|
9245
|
-
async probeBackendBounded(backend) {
|
|
9246
|
-
const work = this.probeBackend(backend);
|
|
9782
|
+
async probeBackendBounded(backend, opts = {}) {
|
|
9783
|
+
const work = this.probeBackend(backend, opts);
|
|
9247
9784
|
work.catch(() => { });
|
|
9248
9785
|
let timer;
|
|
9249
9786
|
const deadline = new Promise(resolve => {
|
|
@@ -9307,12 +9844,23 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9307
9844
|
const label = option.description ? `${option.label} — ${option.description}` : option.label;
|
|
9308
9845
|
return option.id === currentModel ? `✓ ${label}` : label;
|
|
9309
9846
|
}
|
|
9310
|
-
/**
|
|
9311
|
-
|
|
9847
|
+
/**
|
|
9848
|
+
* Probe one backend's CLI env and cache it. Best-effort; never throws.
|
|
9849
|
+
*
|
|
9850
|
+
* `refreshVendorCatalog` is the "🔄 Refresh models" path only (#886): first
|
|
9851
|
+
* ask the CLI to refetch its own catalog, for backends whose probe merely
|
|
9852
|
+
* reads a file the CLI maintains. Every other caller leaves it off, so the
|
|
9853
|
+
* startup and /model probes behave exactly as before. A failed vendor
|
|
9854
|
+
* refresh fails the probe (null), which the menu reports instead of passing
|
|
9855
|
+
* the old list off as fresh.
|
|
9856
|
+
*/
|
|
9857
|
+
async probeBackend(backend, opts = {}) {
|
|
9312
9858
|
try {
|
|
9313
9859
|
const be = createBackend(backend, join(getAgendHome(), "cli-env"));
|
|
9314
9860
|
if (!be.probeCLIEnv)
|
|
9315
9861
|
return null;
|
|
9862
|
+
if (opts.refreshVendorCatalog && be.refreshModelCatalog)
|
|
9863
|
+
await be.refreshModelCatalog();
|
|
9316
9864
|
const probed = await be.probeCLIEnv({ workingDirectory: "", instanceDir: join(getAgendHome(), "cli-env"), instanceName: `probe-${backend}`, mcpServers: {} });
|
|
9317
9865
|
const env = { backend, probedAt: Date.now(), ...probed };
|
|
9318
9866
|
// An empty result must never overwrite a catalog we already have. Some
|
|
@@ -9361,22 +9909,48 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9361
9909
|
}
|
|
9362
9910
|
/** Best-effort model list for `/model`: cached CLI env first, else live probe. Never throws. */
|
|
9363
9911
|
async getModelOptions(instanceName, refresh = false, onLiveProbe) {
|
|
9912
|
+
return (await this.getModelOptionsWithSource(instanceName, { refresh, onLiveProbe })).models;
|
|
9913
|
+
}
|
|
9914
|
+
/**
|
|
9915
|
+
* The model list plus where it came from. `source` is "live" only when a probe
|
|
9916
|
+
* actually ran and answered; a refresh that failed or ran out of time reports
|
|
9917
|
+
* "cache" with the previous list, so a menu can say "could not refresh"
|
|
9918
|
+
* instead of presenting an old list as a fresh one.
|
|
9919
|
+
*/
|
|
9920
|
+
async getModelOptionsWithSource(instanceName, opts = {}) {
|
|
9364
9921
|
const backendName = this.backendNameForInstance(instanceName);
|
|
9365
9922
|
const cached = this.readCliEnv(backendName);
|
|
9366
|
-
if (!refresh && cached?.models.length && !this.cliEnvNeedsRefresh(cached))
|
|
9367
|
-
return cached.models;
|
|
9923
|
+
if (!opts.refresh && cached?.models.length && !this.cliEnvNeedsRefresh(cached)) {
|
|
9924
|
+
return { models: cached.models, source: "cache" };
|
|
9925
|
+
}
|
|
9368
9926
|
// About to go to the vendor: let the caller say so. A silent 1–10s pause on
|
|
9369
9927
|
// an interactive command reads as another hang, which is the wrong lesson to
|
|
9370
9928
|
// teach a user who has just been bitten by one.
|
|
9371
|
-
onLiveProbe?.();
|
|
9929
|
+
opts.onLiveProbe?.();
|
|
9372
9930
|
// Stale, missing, or a forced refresh → probe live (also refreshes the cache).
|
|
9373
9931
|
// A newly released model is invisible until this runs, which is why staleness
|
|
9374
9932
|
// triggers it rather than waiting for the 24h hard expiry or a cold start.
|
|
9375
|
-
const env = await this.probeBackendBounded(backendName);
|
|
9933
|
+
const env = await this.probeBackendBounded(backendName, { refreshVendorCatalog: opts.refreshVendorCatalog });
|
|
9376
9934
|
if (env?.models.length)
|
|
9377
|
-
return env.models;
|
|
9935
|
+
return { models: env.models, source: "live" };
|
|
9378
9936
|
// Probe failed or timed out: the previous list is still the best answer.
|
|
9379
|
-
return cached?.models ?? [];
|
|
9937
|
+
return { models: cached?.models ?? [], source: "cache" };
|
|
9938
|
+
}
|
|
9939
|
+
/**
|
|
9940
|
+
* The /model menu's choices, in the one order all three pickers share:
|
|
9941
|
+
* 🔄 Refresh first, then models, then (claude) "More models…". Refresh takes
|
|
9942
|
+
* a slot inside Discord's 25-option select cap, so it is paid for from the
|
|
9943
|
+
* model rows, never from "More models…".
|
|
9944
|
+
*/
|
|
9945
|
+
modelMenuChoices(instanceName, nonce, options, currentModel) {
|
|
9946
|
+
const isClaude = this.backendNameForInstance(instanceName) === "claude-code";
|
|
9947
|
+
const choices = [{ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:__refresh__`, label: t("model.refresh") }];
|
|
9948
|
+
for (const o of options.slice(0, isClaude ? 23 : 24)) {
|
|
9949
|
+
choices.push({ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:${o.id}`, label: this.modelChoiceLabel(o, currentModel) });
|
|
9950
|
+
}
|
|
9951
|
+
if (isClaude)
|
|
9952
|
+
choices.push({ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:__more__`, label: t("model.more") });
|
|
9953
|
+
return choices;
|
|
9380
9954
|
}
|
|
9381
9955
|
/**
|
|
9382
9956
|
* Model catalog behind the `list_models` tool.
|
|
@@ -9677,14 +10251,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9677
10251
|
// Raw id for ✓-matching options; display resolves an inherited CLI default.
|
|
9678
10252
|
const { model: currentModel, display: currentDisplay } = this.resolveInstanceModel(name);
|
|
9679
10253
|
const nonce = randomBytes(6).toString("hex");
|
|
9680
|
-
const
|
|
9681
|
-
const choices = options.slice(0, isClaude ? 24 : 25).map(o => ({
|
|
9682
|
-
id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:${o.id}`,
|
|
9683
|
-
label: this.modelChoiceLabel(o, currentModel),
|
|
9684
|
-
}));
|
|
9685
|
-
if (isClaude) {
|
|
9686
|
-
choices.push({ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:__more__`, label: t("model.more") });
|
|
9687
|
-
}
|
|
10254
|
+
const choices = this.modelMenuChoices(name, nonce, options, currentModel);
|
|
9688
10255
|
const timer = setTimeout(() => this.pendingModelSelects.delete(nonce), CLASSIC_BACKEND_SELECTION_TIMEOUT_MS);
|
|
9689
10256
|
timer.unref?.();
|
|
9690
10257
|
this.pendingModelSelects.set(nonce, { instanceName: name, model: "", userId: data.userId, channelId: data.channelId, timer, respond: data.respond, respondChoices: data.respondChoices });
|
|
@@ -9710,15 +10277,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9710
10277
|
}
|
|
9711
10278
|
const { model: currentModel, display: currentDisplay } = this.resolveInstanceModel(instanceName);
|
|
9712
10279
|
const nonce = randomBytes(6).toString("hex");
|
|
9713
|
-
const
|
|
9714
|
-
// Keep the more-models entry inside Discord's 25-option select cap.
|
|
9715
|
-
const choices = options.slice(0, isClaude ? 24 : 25).map(o => ({
|
|
9716
|
-
id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:${o.id}`,
|
|
9717
|
-
label: this.modelChoiceLabel(o, currentModel),
|
|
9718
|
-
}));
|
|
9719
|
-
if (isClaude) {
|
|
9720
|
-
choices.push({ id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:__more__`, label: t("model.more") });
|
|
9721
|
-
}
|
|
10280
|
+
const choices = this.modelMenuChoices(instanceName, nonce, options, currentModel);
|
|
9722
10281
|
const respond = async (text) => {
|
|
9723
10282
|
await adapter.sendText(chatId, text, { threadId });
|
|
9724
10283
|
return undefined;
|
|
@@ -9802,6 +10361,64 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9802
10361
|
await pending.respond(t("model.more_unavailable")).catch(() => { });
|
|
9803
10362
|
}
|
|
9804
10363
|
}
|
|
10364
|
+
/**
|
|
10365
|
+
* "🔄 Refresh models" (#886): probe live past both caches and redraw the menu.
|
|
10366
|
+
*
|
|
10367
|
+
* Past AgEnD's cli-env cache (refresh=true) and, where the backend supports
|
|
10368
|
+
* it, past the CLI's own catalog cache too (codex: `codex debug models`).
|
|
10369
|
+
* Without the second half, a codex refresh re-read the same models_cache.json
|
|
10370
|
+
* and showed the old list as new.
|
|
10371
|
+
*
|
|
10372
|
+
* A failed refresh keeps the previous list and says so. It never blanks the
|
|
10373
|
+
* menu, because a picker with no rows cannot even offer another refresh.
|
|
10374
|
+
*/
|
|
10375
|
+
async refreshModelMenu(pending) {
|
|
10376
|
+
const { models, source } = await this.getModelOptionsWithSource(pending.instanceName, {
|
|
10377
|
+
refresh: true, refreshVendorCatalog: true,
|
|
10378
|
+
});
|
|
10379
|
+
if (models.length === 0) {
|
|
10380
|
+
await pending.respond(t("model.list_unavailable", pending.instanceName)).catch(() => { });
|
|
10381
|
+
return;
|
|
10382
|
+
}
|
|
10383
|
+
const { model: currentModel, display: currentDisplay } = this.resolveInstanceModel(pending.instanceName);
|
|
10384
|
+
const nonce = randomBytes(6).toString("hex");
|
|
10385
|
+
const choices = this.modelMenuChoices(pending.instanceName, nonce, models, currentModel);
|
|
10386
|
+
const status = source === "live" ? t("model.refreshed") : t("model.refresh_failed");
|
|
10387
|
+
const timer = setTimeout(() => {
|
|
10388
|
+
const p = this.pendingModelSelects.get(nonce);
|
|
10389
|
+
if (p) {
|
|
10390
|
+
this.pendingModelSelects.delete(nonce);
|
|
10391
|
+
p.respond(t("model.selection_expired")).catch(() => { });
|
|
10392
|
+
}
|
|
10393
|
+
}, CLASSIC_BACKEND_SELECTION_TIMEOUT_MS);
|
|
10394
|
+
timer.unref?.();
|
|
10395
|
+
this.pendingModelSelects.set(nonce, { ...pending, model: "", timer });
|
|
10396
|
+
try {
|
|
10397
|
+
if (pending.respondChoices) {
|
|
10398
|
+
// Discord select menu: edit the same interaction reply in place.
|
|
10399
|
+
await pending.respondChoices(`${status}\n${t("model.menu", `**${currentDisplay}**`)}`, choices);
|
|
10400
|
+
return;
|
|
10401
|
+
}
|
|
10402
|
+
if (pending.adapter && pending.adapterChatId) {
|
|
10403
|
+
// Telegram: retire the consumed keyboard, then post the redrawn menu.
|
|
10404
|
+
if (pending.menuMessageId && pending.adapter.editMessageRemoveButtons) {
|
|
10405
|
+
await pending.adapter.editMessageRemoveButtons(pending.adapterChatId, pending.menuMessageId, t("model.refresh"), pending.adapterThreadId).catch(() => { });
|
|
10406
|
+
}
|
|
10407
|
+
const menuMessageId = await pending.adapter.promptUser(pending.adapterChatId, `${status}\n${t("model.menu", currentDisplay)}`, choices, { threadId: pending.adapterThreadId });
|
|
10408
|
+
const fresh = this.pendingModelSelects.get(nonce);
|
|
10409
|
+
if (fresh)
|
|
10410
|
+
fresh.menuMessageId = menuMessageId;
|
|
10411
|
+
return;
|
|
10412
|
+
}
|
|
10413
|
+
await pending.respond(t("model.usage")).catch(() => { });
|
|
10414
|
+
}
|
|
10415
|
+
catch (err) {
|
|
10416
|
+
this.pendingModelSelects.delete(nonce);
|
|
10417
|
+
clearTimeout(timer);
|
|
10418
|
+
this.logger.warn({ err, instanceName: pending.instanceName }, "Refreshed model menu failed");
|
|
10419
|
+
await pending.respond(t("model.usage")).catch(() => { });
|
|
10420
|
+
}
|
|
10421
|
+
}
|
|
9805
10422
|
/** Consume a `/model` selection callback. Returns true for all model-select ids (incl. stale). */
|
|
9806
10423
|
async handleModelSelection(data) {
|
|
9807
10424
|
if (!data.callbackData.startsWith(MODEL_SELECT_CALLBACK_PREFIX))
|
|
@@ -9827,6 +10444,12 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9827
10444
|
await this.expandClaudeModelMenu(pending);
|
|
9828
10445
|
return true;
|
|
9829
10446
|
}
|
|
10447
|
+
// "🔄 Refresh models" is navigation too: re-probe past every cache and
|
|
10448
|
+
// redraw the same menu with what came back.
|
|
10449
|
+
if (model === "__refresh__") {
|
|
10450
|
+
await this.refreshModelMenu(pending);
|
|
10451
|
+
return true;
|
|
10452
|
+
}
|
|
9830
10453
|
// Send immediate "⏳ Switching..." feedback, then apply in background.
|
|
9831
10454
|
const progressText = t("model.switching", pending.instanceName, model);
|
|
9832
10455
|
let progressMsgId;
|
|
@@ -10360,6 +10983,10 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10360
10983
|
clearInterval(this.logRotateTimer);
|
|
10361
10984
|
this.logRotateTimer = null;
|
|
10362
10985
|
}
|
|
10986
|
+
if (this.discordPresenceTimer) {
|
|
10987
|
+
clearInterval(this.discordPresenceTimer);
|
|
10988
|
+
this.discordPresenceTimer = null;
|
|
10989
|
+
}
|
|
10363
10990
|
// Cancel-button timers were never cleared here. The idle-check interval is not
|
|
10364
10991
|
// unref'd, so it held the event loop open past shutdown and kept retrying
|
|
10365
10992
|
// deletes against an adapter that was already gone.
|
|
@@ -10397,11 +11024,9 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10397
11024
|
for (const pending of this.pendingClassicStarts.values())
|
|
10398
11025
|
clearTimeout(pending.timer);
|
|
10399
11026
|
this.pendingClassicStarts.clear();
|
|
10400
|
-
|
|
10401
|
-
|
|
10402
|
-
|
|
10403
|
-
}
|
|
10404
|
-
this.pendingNonceButtons.clear();
|
|
11027
|
+
// Adapters are still connected here — they are stopped further down — so
|
|
11028
|
+
// this is the last moment the prompts can be collapsed.
|
|
11029
|
+
await this.retirePendingNoncePrompts();
|
|
10405
11030
|
this.topicArchiver.stop();
|
|
10406
11031
|
this.scheduler?.shutdown();
|
|
10407
11032
|
// Stop instances in parallel batches to avoid long sequential waits.
|
|
@@ -10586,9 +11211,9 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10586
11211
|
* Whitelisted runtime fields are pushed into live daemons; all other instance
|
|
10587
11212
|
* fields, plus cold fleet-level settings, retain restart semantics.
|
|
10588
11213
|
*/
|
|
10589
|
-
async reconcileInstances() {
|
|
11214
|
+
async reconcileInstances(observe) {
|
|
10590
11215
|
if (!this.configPath)
|
|
10591
|
-
return;
|
|
11216
|
+
return {};
|
|
10592
11217
|
const oldConfig = this.fleetConfig;
|
|
10593
11218
|
const previousRawConfig = this.rawFleetConfig;
|
|
10594
11219
|
const previousRawDocument = this.rawFleetDocument;
|
|
@@ -10631,7 +11256,9 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10631
11256
|
? t("fleet.reload_removed_all", oldCount)
|
|
10632
11257
|
: t("fleet.reload_removed_half", oldCount, newCount);
|
|
10633
11258
|
this.notifyFleetError(t("fleet.reload_rejected", why));
|
|
10634
|
-
|
|
11259
|
+
// Reported, not swallowed: an apply job whose config was refused used to
|
|
11260
|
+
// mark every row done and tell the user "changes applied".
|
|
11261
|
+
return { rejected: why };
|
|
10635
11262
|
}
|
|
10636
11263
|
// Classic behavior settings share the fleet defaults but are not entries
|
|
10637
11264
|
// in fleet.yaml. Snapshot the old effective chain before reloading the
|
|
@@ -10662,29 +11289,31 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10662
11289
|
this.scheduler?.reload();
|
|
10663
11290
|
const newInstances = this.fleetConfig.instances;
|
|
10664
11291
|
const topicMode = this.fleetConfig?.channel?.mode === "topic";
|
|
10665
|
-
//
|
|
10666
|
-
//
|
|
10667
|
-
|
|
10668
|
-
|
|
10669
|
-
|
|
10670
|
-
|
|
10671
|
-
|
|
10672
|
-
|
|
10673
|
-
|
|
10674
|
-
|
|
10675
|
-
|
|
10676
|
-
|
|
10677
|
-
const oldFleetLevel = JSON.stringify({ channel: oldConfig?.channel, defaults: oldDefaultCold });
|
|
10678
|
-
const newFleetLevel = JSON.stringify({ channel: this.fleetConfig?.channel, defaults: newDefaultCold });
|
|
10679
|
-
if (oldFleetLevel !== newFleetLevel) {
|
|
10680
|
-
this.logger.warn("Fleet-level config changed (channel/defaults) — use /restart for full effect");
|
|
11292
|
+
// Only what a fresh process can adopt, and only relative to the signature
|
|
11293
|
+
// this process came up on: a Settings edit mutates this.fleetConfig in place
|
|
11294
|
+
// before the reload, so comparing the pre-load copy sees nothing at all.
|
|
11295
|
+
const newFleetLevel = this.fleetLevelSignature();
|
|
11296
|
+
if (this.appliedFleetLevel !== null && this.appliedFleetLevel !== newFleetLevel) {
|
|
11297
|
+
this.logger.warn({
|
|
11298
|
+
keys: fleetLevelDifferences(this.startupFleetConfig, this.fleetConfig),
|
|
11299
|
+
}, "Fleet-level config changed — restart AgEnD for it to take effect");
|
|
11300
|
+
// Terminal, and deliberately not "done": this reconcile cannot adopt a
|
|
11301
|
+
// fleet-level change, and saying otherwise would claim AgEnD is running
|
|
11302
|
+
// on a configuration it is not running on.
|
|
11303
|
+
observe?.(APPLY_FLEET_TARGET, "restart", "restart-required");
|
|
10681
11304
|
}
|
|
10682
11305
|
// Stop removed instances (skip classic bot instances — they're managed by classicBot.yaml)
|
|
10683
11306
|
const classicNames = new Set(this.classicChannels?.getAll().map(ch => ch.instanceName) ?? []);
|
|
10684
11307
|
for (const name of this.daemons.keys()) {
|
|
10685
11308
|
if (!(name in newInstances) && !classicNames.has(name)) {
|
|
10686
11309
|
this.logger.info({ name }, "Instance removed from config — stopping");
|
|
10687
|
-
|
|
11310
|
+
observe?.(name, "restart", "running");
|
|
11311
|
+
await this.stopInstance(name)
|
|
11312
|
+
.then(() => observe?.(name, "restart", "done"))
|
|
11313
|
+
.catch(err => {
|
|
11314
|
+
observe?.(name, "restart", "failed", err.message);
|
|
11315
|
+
this.logger.error({ err, name }, "Failed to stop removed instance");
|
|
11316
|
+
});
|
|
10688
11317
|
}
|
|
10689
11318
|
}
|
|
10690
11319
|
// Start new + reconcile modified instances. Hot values are always sent as a
|
|
@@ -10694,20 +11323,23 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10694
11323
|
if (!this.daemons.has(name)) {
|
|
10695
11324
|
// New instance — startInstance already calls connectIpcToInstance
|
|
10696
11325
|
this.logger.info({ name }, "New instance in config — starting");
|
|
11326
|
+
observe?.(name, "restart", "running");
|
|
10697
11327
|
await this.startInstanceUnattended(name, config, topicMode, "new instance");
|
|
11328
|
+
observe?.(name, "restart", "done");
|
|
10698
11329
|
}
|
|
10699
11330
|
else if (oldConfig?.instances[name]) {
|
|
10700
11331
|
const daemon = this.daemons.get(name);
|
|
10701
11332
|
const runtimeConfig = daemon.getConfigSnapshot?.() ?? oldConfig.instances[name];
|
|
10702
|
-
const
|
|
10703
|
-
|
|
10704
|
-
// Every field not explicitly classified hot is cold by default.
|
|
10705
|
-
if (!isDeepStrictEqual(oldParts.cold, newParts.cold)) {
|
|
11333
|
+
const change = classifyInstanceChange(runtimeConfig, config);
|
|
11334
|
+
if (change === "restart") {
|
|
10706
11335
|
this.logger.info({ name }, "Instance config changed — restarting");
|
|
11336
|
+
observe?.(name, "restart", "running");
|
|
10707
11337
|
await this.stopInstance(name).catch(() => { });
|
|
10708
11338
|
await this.startInstanceUnattended(name, config, topicMode, "modified instance");
|
|
11339
|
+
observe?.(name, "restart", "done");
|
|
10709
11340
|
}
|
|
10710
|
-
else if (
|
|
11341
|
+
else if (change === "hot") {
|
|
11342
|
+
observe?.(name, "hot", "running");
|
|
10711
11343
|
const update = hotConfigUpdate(config);
|
|
10712
11344
|
const ipc = this.instanceIpcClients.get(name);
|
|
10713
11345
|
const sent = ipc?.connected === true && ipc.send({ type: "config_update", config: update });
|
|
@@ -10718,6 +11350,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10718
11350
|
this.logger.warn({ name }, "Config-update IPC unavailable — applied hot config in-process");
|
|
10719
11351
|
}
|
|
10720
11352
|
this.logger.info({ name, fields: [...HOT_INSTANCE_CONFIG_KEYS] }, "Instance hot config reloaded");
|
|
11353
|
+
observe?.(name, "hot", "done");
|
|
10721
11354
|
}
|
|
10722
11355
|
}
|
|
10723
11356
|
}
|
|
@@ -10735,25 +11368,1155 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
10735
11368
|
const autoPauseAfter = this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after);
|
|
10736
11369
|
if (old.backend !== backend || old.model !== model || old.autoPauseAfter !== autoPauseAfter) {
|
|
10737
11370
|
this.logger.info({ instanceName: ch.instanceName }, "Classic cold config changed — restarting");
|
|
11371
|
+
observe?.(ch.instanceName, "restart", "running");
|
|
10738
11372
|
await this.stopInstance(ch.instanceName).catch(() => { });
|
|
10739
11373
|
await this.startClassicInstanceUnattended(ch, "classic instance after fleet reload");
|
|
11374
|
+
observe?.(ch.instanceName, "restart", "done");
|
|
10740
11375
|
continue;
|
|
10741
11376
|
}
|
|
10742
11377
|
const toolProgress = this.classicChannels.getToolProgress(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.tool_progress);
|
|
10743
11378
|
const replyCompletionGuard = this.classicChannels.getReplyCompletionGuard(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.reply_completion_guard);
|
|
10744
11379
|
if (old.toolProgress !== toolProgress || old.replyCompletionGuard !== replyCompletionGuard) {
|
|
11380
|
+
observe?.(ch.instanceName, "hot", "running");
|
|
10745
11381
|
this.applyHotConfigUpdate(ch.instanceName, {
|
|
10746
11382
|
tool_progress: toolProgress,
|
|
10747
11383
|
reply_completion_guard: replyCompletionGuard,
|
|
10748
11384
|
});
|
|
10749
11385
|
this.logger.info({ instanceName: ch.instanceName }, "Classic inherited hot config reloaded");
|
|
11386
|
+
observe?.(ch.instanceName, "hot", "done");
|
|
10750
11387
|
}
|
|
10751
11388
|
}
|
|
10752
11389
|
}
|
|
10753
11390
|
// warm_cap is fleet-owned; enforce the reloaded value immediately against
|
|
10754
11391
|
// currently idle instances instead of waiting for a future state edge.
|
|
10755
11392
|
this.enforceWarmCap();
|
|
11393
|
+
// appliedFleetLevel is deliberately NOT updated here. It means "the
|
|
11394
|
+
// fleet-level config this process started on"; a reconcile does not restart
|
|
11395
|
+
// the process, so moving it would erase the fact that a restart is still
|
|
11396
|
+
// owed and silence every later reminder.
|
|
10756
11397
|
this.logger.info({ running: this.daemons.size, configured: Object.keys(newInstances).length }, "Reconcile complete");
|
|
11398
|
+
return {};
|
|
11399
|
+
}
|
|
11400
|
+
/**
|
|
11401
|
+
* `{channel, cold defaults}` as one comparable string — the part of the config
|
|
11402
|
+
* a running fleet process cannot adopt without restarting.
|
|
11403
|
+
*/
|
|
11404
|
+
fleetLevelSignature(config = this.fleetConfig) {
|
|
11405
|
+
return fleetLevelSignature(config);
|
|
11406
|
+
}
|
|
11407
|
+
/**
|
|
11408
|
+
* The config a reconcile is about to load, not the one held in memory.
|
|
11409
|
+
*
|
|
11410
|
+
* Settings mutates the in-memory object and writes the file; only the file
|
|
11411
|
+
* goes back through defaults expansion. Forecasting from memory therefore
|
|
11412
|
+
* misses every instance that a changed fleet default will restart — the user
|
|
11413
|
+
* is told "restart AgEnD" and not told that five agents are about to go down.
|
|
11414
|
+
*/
|
|
11415
|
+
nextFleetConfig() {
|
|
11416
|
+
if (!this.configPath)
|
|
11417
|
+
return this.fleetConfig;
|
|
11418
|
+
try {
|
|
11419
|
+
return loadFleetConfig(this.configPath);
|
|
11420
|
+
}
|
|
11421
|
+
catch (err) {
|
|
11422
|
+
// An unparseable file is the reconcile's problem to report; the forecast
|
|
11423
|
+
// falls back to what is running rather than failing the request.
|
|
11424
|
+
this.logger.debug({ err }, "Apply plan fell back to the in-memory config");
|
|
11425
|
+
return this.fleetConfig;
|
|
11426
|
+
}
|
|
11427
|
+
}
|
|
11428
|
+
/**
|
|
11429
|
+
* Does the config this process is running match the one a reconcile would
|
|
11430
|
+
* load off disk?
|
|
11431
|
+
*
|
|
11432
|
+
* If not, every apply reports a fleet-level change that a restart cannot
|
|
11433
|
+
* clear — restart, recompute, disagree again — a self-sustaining loop that
|
|
11434
|
+
* the rate limit can only slow to three naggings an hour. Startup rewrites
|
|
11435
|
+
* the file in three places before this point (slimFleetConfigAtStartup, the
|
|
11436
|
+
* general auto-create, the general fixup), so the two can genuinely diverge.
|
|
11437
|
+
*
|
|
11438
|
+
* An offer to restart that cannot possibly succeed is worse than no offer, so
|
|
11439
|
+
* the panel shows the mismatch instead of a button.
|
|
11440
|
+
*/
|
|
11441
|
+
checkStartupSignatureConsistency() {
|
|
11442
|
+
if (!this.configPath) {
|
|
11443
|
+
this.fleetSignatureMismatch = null;
|
|
11444
|
+
return;
|
|
11445
|
+
}
|
|
11446
|
+
let onDisk;
|
|
11447
|
+
try {
|
|
11448
|
+
onDisk = loadFleetConfig(this.configPath);
|
|
11449
|
+
}
|
|
11450
|
+
catch (err) {
|
|
11451
|
+
this.logger.warn({ err }, "Could not re-read fleet.yaml to check the startup signature");
|
|
11452
|
+
this.fleetSignatureMismatch = null;
|
|
11453
|
+
return;
|
|
11454
|
+
}
|
|
11455
|
+
if (fleetLevelSignature(onDisk) === this.appliedFleetLevel) {
|
|
11456
|
+
this.fleetSignatureMismatch = null;
|
|
11457
|
+
return;
|
|
11458
|
+
}
|
|
11459
|
+
this.fleetSignatureMismatch = fleetLevelDifferences(this.fleetConfig, onDisk);
|
|
11460
|
+
this.logger.warn({
|
|
11461
|
+
keys: this.fleetSignatureMismatch,
|
|
11462
|
+
configPath: this.configPath,
|
|
11463
|
+
}, "fleet.yaml and the running configuration disagree on startup-only keys — every apply will ask for a restart that cannot clear it");
|
|
11464
|
+
}
|
|
11465
|
+
/**
|
|
11466
|
+
* Is a live adapter already long-polling this bot token?
|
|
11467
|
+
*
|
|
11468
|
+
* Telegram's `getUpdates` has exactly one consumer: a second poller takes
|
|
11469
|
+
* turns with the first and both miss messages. The setup wizard's "post in
|
|
11470
|
+
* the group and I'll detect it" step is a second poller, so it has to know
|
|
11471
|
+
* when the answer is "not against this token, not while I'm running".
|
|
11472
|
+
*/
|
|
11473
|
+
isBotTokenInUse(token) {
|
|
11474
|
+
if (!token)
|
|
11475
|
+
return false;
|
|
11476
|
+
const configured = this.fleetConfig?.channels
|
|
11477
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []);
|
|
11478
|
+
for (const channel of configured) {
|
|
11479
|
+
const envVar = channel?.bot_token_env;
|
|
11480
|
+
if (!envVar)
|
|
11481
|
+
continue;
|
|
11482
|
+
// Compare the value, not the variable name: the same token can be
|
|
11483
|
+
// reached through a differently named variable.
|
|
11484
|
+
if (process.env[envVar] === token)
|
|
11485
|
+
return true;
|
|
11486
|
+
}
|
|
11487
|
+
return false;
|
|
11488
|
+
}
|
|
11489
|
+
/** Non-null when startup found the running config and fleet.yaml disagreeing. */
|
|
11490
|
+
fleetSignatureMismatchKeys() {
|
|
11491
|
+
return this.fleetSignatureMismatch;
|
|
11492
|
+
}
|
|
11493
|
+
/**
|
|
11494
|
+
* Restart AgEnD itself on behalf of a Settings apply.
|
|
11495
|
+
*
|
|
11496
|
+
* Deliberately a separate action from Apply: the panel is reachable from
|
|
11497
|
+
* outside the LAN, and "restart the whole fleet" must never be something a
|
|
11498
|
+
* single Apply click can carry along with it.
|
|
11499
|
+
*
|
|
11500
|
+
* The order below is the safety envelope, and the order matters:
|
|
11501
|
+
* concurrency and consistency first (cheap, and a restart during a reconcile
|
|
11502
|
+
* is the dangerous one), then the state checks, then the rate limit, then the
|
|
11503
|
+
* audit notice — and only once all of that holds is the attempt written to
|
|
11504
|
+
* disk and fsynced, before anything spawns.
|
|
11505
|
+
*/
|
|
11506
|
+
async requestSettingsSelfRestart(jobId, key) {
|
|
11507
|
+
// A retry after a lost response must not restart a second time. The key is
|
|
11508
|
+
// recorded on the job, which survives the restart it triggers.
|
|
11509
|
+
const already = this.applyJobs.findByRestartKey(key);
|
|
11510
|
+
if (already)
|
|
11511
|
+
return { ok: true, jobId: already.id, reused: true };
|
|
11512
|
+
if (this.reconcileInFlight || this.activeApplyJobId) {
|
|
11513
|
+
return { ok: false, status: 409, error: "a configuration reload is running — try again once it finishes" };
|
|
11514
|
+
}
|
|
11515
|
+
if (this.fleetSignatureMismatch) {
|
|
11516
|
+
// Restarting cannot clear this, so offering it would be a loop.
|
|
11517
|
+
return {
|
|
11518
|
+
ok: false,
|
|
11519
|
+
status: 409,
|
|
11520
|
+
error: `fleet.yaml and the running configuration disagree on ${this.fleetSignatureMismatch.join(", ")} — check fleet.log before restarting`,
|
|
11521
|
+
};
|
|
11522
|
+
}
|
|
11523
|
+
const job = this.applyJobs.get(jobId);
|
|
11524
|
+
const row = job?.targets.find(item => item.target === APPLY_FLEET_TARGET);
|
|
11525
|
+
if (!job || !row || row.status !== "restart-required") {
|
|
11526
|
+
return { ok: false, status: 409, error: "that apply has no pending fleet-level change" };
|
|
11527
|
+
}
|
|
11528
|
+
// The row alone is not enough: an old job keeps its row for the whole
|
|
11529
|
+
// retention window, so a change made and then reverted would still leave a
|
|
11530
|
+
// job that looks restartable. The live signature is the authority — and it
|
|
11531
|
+
// must be the same view of the config the plan uses.
|
|
11532
|
+
if (this.appliedFleetLevel === null || this.appliedFleetLevel === this.fleetLevelSignature(this.nextFleetConfig())) {
|
|
11533
|
+
return { ok: false, status: 409, error: "no fleet-level change is pending any more" };
|
|
11534
|
+
}
|
|
11535
|
+
const allowance = checkSelfRestartAllowance(this.dataDir);
|
|
11536
|
+
if (!allowance.allowed) {
|
|
11537
|
+
if (allowance.reason === "unreadable") {
|
|
11538
|
+
// Fail closed: with the limit's own state in doubt, "no attempts yet"
|
|
11539
|
+
// is the one reading that must not be assumed.
|
|
11540
|
+
return {
|
|
11541
|
+
ok: false,
|
|
11542
|
+
status: 503,
|
|
11543
|
+
error: "the restart rate-limit file cannot be read — remove self-restart.json from the data dir on the host, or run `agend restart` there",
|
|
11544
|
+
};
|
|
11545
|
+
}
|
|
11546
|
+
return {
|
|
11547
|
+
ok: false,
|
|
11548
|
+
status: 429,
|
|
11549
|
+
error: allowance.reason === "too-soon"
|
|
11550
|
+
? "AgEnD was restarted from Settings very recently"
|
|
11551
|
+
: "too many Settings-triggered restarts in the last hour",
|
|
11552
|
+
retryAfterSeconds: allowance.retryAfterSeconds,
|
|
11553
|
+
};
|
|
11554
|
+
}
|
|
11555
|
+
// Recorded before the notice, not after. If recording keeps failing (a
|
|
11556
|
+
// read-only data dir), posting first would let whoever holds the token spam
|
|
11557
|
+
// the channel with "restarting…" notices for restarts that never happen.
|
|
11558
|
+
// The cost is that a failed announcement still spends an attempt, which is
|
|
11559
|
+
// the right way round for a rate limit.
|
|
11560
|
+
if (!recordSelfRestartAttempt(this.dataDir)) {
|
|
11561
|
+
this.logger.error("Self-restart attempt could not be recorded — refusing to restart unmetered");
|
|
11562
|
+
return { ok: false, status: 503, error: "could not record the restart attempt" };
|
|
11563
|
+
}
|
|
11564
|
+
// Out-of-band notice before the restart, so a panel restart is visible where
|
|
11565
|
+
// the admins are. Refusing when it cannot be posted is the same rule as
|
|
11566
|
+
// refusing when the progress marker cannot be written: no untraceable
|
|
11567
|
+
// restarts.
|
|
11568
|
+
const notice = await this.postSelfRestartNotice();
|
|
11569
|
+
if (!notice) {
|
|
11570
|
+
return {
|
|
11571
|
+
ok: false,
|
|
11572
|
+
status: 409,
|
|
11573
|
+
error: "no chat channel is available to announce the restart — run `agend restart` on the host instead",
|
|
11574
|
+
};
|
|
11575
|
+
}
|
|
11576
|
+
// Consume the row: it moves to running, which is also what lets the next
|
|
11577
|
+
// process settle it (settleAfterRestart only touches non-terminal rows).
|
|
11578
|
+
this.applyJobs.update(jobId, current => {
|
|
11579
|
+
const target = current.targets.find(item => item.target === APPLY_FLEET_TARGET);
|
|
11580
|
+
if (target)
|
|
11581
|
+
target.status = "running";
|
|
11582
|
+
current.restart_key = key;
|
|
11583
|
+
current.deadlineMs = SELF_RESTART_DEADLINE_MS;
|
|
11584
|
+
});
|
|
11585
|
+
this.emitSseEvent("apply_progress", viewOf(this.applyJobs.get(jobId)));
|
|
11586
|
+
const launched = await this.requestFullRestart(notice.adapter, notice.chatId, notice.threadId, notice.messageId)
|
|
11587
|
+
.catch(err => {
|
|
11588
|
+
this.logger.error({ err }, "Settings-triggered self restart failed to launch");
|
|
11589
|
+
return false;
|
|
11590
|
+
});
|
|
11591
|
+
if (!launched) {
|
|
11592
|
+
this.applyJobs.setTargetStatus(jobId, APPLY_FLEET_TARGET, "failed", "restart could not be launched");
|
|
11593
|
+
this.applyJobs.finish(jobId, "restart could not be launched");
|
|
11594
|
+
this.emitSseEvent("apply_progress", viewOf(this.applyJobs.get(jobId)));
|
|
11595
|
+
return { ok: false, status: 409, error: "the restart could not be launched — see fleet.log" };
|
|
11596
|
+
}
|
|
11597
|
+
return { ok: true, jobId };
|
|
11598
|
+
}
|
|
11599
|
+
/** The audit notice, and the message the restart progress will edit. */
|
|
11600
|
+
async postSelfRestartNotice() {
|
|
11601
|
+
const groupId = this.fleetConfig?.channel?.group_id;
|
|
11602
|
+
const adapter = this.adapter;
|
|
11603
|
+
if (!groupId || !adapter)
|
|
11604
|
+
return null;
|
|
11605
|
+
const generalName = this.findGeneralInstance();
|
|
11606
|
+
const rawThreadId = generalName ? this.fleetConfig?.instances[generalName]?.topic_id : undefined;
|
|
11607
|
+
const threadId = rawThreadId != null ? String(rawThreadId) : undefined;
|
|
11608
|
+
try {
|
|
11609
|
+
const sent = await adapter.sendText(String(groupId), t("restart.settings_triggered"), { threadId });
|
|
11610
|
+
if (!sent?.messageId)
|
|
11611
|
+
return null;
|
|
11612
|
+
return { adapter, chatId: sent.chatId, threadId: sent.threadId, messageId: sent.messageId };
|
|
11613
|
+
}
|
|
11614
|
+
catch (err) {
|
|
11615
|
+
this.logger.error({ err }, "Could not announce the Settings-triggered restart — refusing to restart silently");
|
|
11616
|
+
return null;
|
|
11617
|
+
}
|
|
11618
|
+
}
|
|
11619
|
+
/** Jobs outlive this process on purpose; see apply-job.ts. */
|
|
11620
|
+
get applyJobs() {
|
|
11621
|
+
return (this.applyJobStoreCache ??= new ApplyJobStore(this.dataDir, Date.now, this.logger));
|
|
11622
|
+
}
|
|
11623
|
+
/** The only channel metadata exposed to the Settings secret UI. */
|
|
11624
|
+
listSecureConnections() {
|
|
11625
|
+
const channels = this.fleetConfig?.channels
|
|
11626
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []);
|
|
11627
|
+
return channels.map((channel, index) => {
|
|
11628
|
+
const id = channel.id ?? channel.type ?? `channel-${index}`;
|
|
11629
|
+
const world = this.worlds.get(id);
|
|
11630
|
+
const state = this.adapterState.get(id);
|
|
11631
|
+
return {
|
|
11632
|
+
id,
|
|
11633
|
+
type: channel.type,
|
|
11634
|
+
token_env: channel.bot_token_env,
|
|
11635
|
+
token_present: !!process.env[channel.bot_token_env],
|
|
11636
|
+
group_id: channel.group_id != null ? String(channel.group_id) : null,
|
|
11637
|
+
general_channel_id: channel.options?.general_channel_id != null
|
|
11638
|
+
? String(channel.options.general_channel_id)
|
|
11639
|
+
: null,
|
|
11640
|
+
status: state?.status ?? (world ? "starting" : "stopped"),
|
|
11641
|
+
...(world ? { identity: { id: world.botUserId ?? null, username: world.botUsername ?? null } } : {}),
|
|
11642
|
+
};
|
|
11643
|
+
});
|
|
11644
|
+
}
|
|
11645
|
+
secureConnectionChannel(connectionId) {
|
|
11646
|
+
const channels = this.fleetConfig?.channels
|
|
11647
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []);
|
|
11648
|
+
const matches = channels.filter((channel, index) => (channel.id ?? channel.type ?? `channel-${index}`) === connectionId);
|
|
11649
|
+
// Ambiguous fallback IDs (for example two unlabelled Discord channels)
|
|
11650
|
+
// must fail closed rather than rotating the first matching token.
|
|
11651
|
+
return matches.length === 1 ? matches[0] : undefined;
|
|
11652
|
+
}
|
|
11653
|
+
secureConnectionGeneration(connectionId) {
|
|
11654
|
+
const adapter = this.adapters.get(connectionId);
|
|
11655
|
+
const healthGeneration = adapter?.getHealthSnapshot?.().generation ?? 0;
|
|
11656
|
+
let generation = this.connectionSecretGenerations.get(connectionId) ?? 0;
|
|
11657
|
+
const previousAdapter = this.connectionSecretAdapterRefs.get(connectionId);
|
|
11658
|
+
const previousHealthGeneration = this.connectionSecretHealthGenerations.get(connectionId);
|
|
11659
|
+
if (this.connectionSecretAdapterRefs.has(connectionId) && previousAdapter !== adapter)
|
|
11660
|
+
generation++;
|
|
11661
|
+
if (this.connectionSecretHealthGenerations.has(connectionId) && previousHealthGeneration !== healthGeneration)
|
|
11662
|
+
generation++;
|
|
11663
|
+
this.connectionSecretAdapterRefs.set(connectionId, adapter);
|
|
11664
|
+
this.connectionSecretHealthGenerations.set(connectionId, healthGeneration);
|
|
11665
|
+
this.connectionSecretGenerations.set(connectionId, generation);
|
|
11666
|
+
return generation;
|
|
11667
|
+
}
|
|
11668
|
+
/** Provider API-key rows exposed to Settings (never the env key or secret). */
|
|
11669
|
+
providerSecretsEnabled() {
|
|
11670
|
+
return this.fleetConfig?.web?.provider_secrets === true;
|
|
11671
|
+
}
|
|
11672
|
+
listProviderSecrets() {
|
|
11673
|
+
return PROVIDER_SECRET_SPECS.map(spec => ({
|
|
11674
|
+
id: spec.id,
|
|
11675
|
+
display_name: spec.displayName,
|
|
11676
|
+
kind: spec.kind,
|
|
11677
|
+
token_present: !!process.env[spec.envKey],
|
|
11678
|
+
verifier: spec.verifier ? "available" : "unsupported",
|
|
11679
|
+
activation: spec.activation,
|
|
11680
|
+
stale_consumers: this.providerSecretStaleConsumers(spec.envKey),
|
|
11681
|
+
}));
|
|
11682
|
+
}
|
|
11683
|
+
providerSecretStaleConsumers(envKey) {
|
|
11684
|
+
// A child inherits the manager's environment at spawn. We cannot inspect
|
|
11685
|
+
// a child process's private environment safely, so report the conservative
|
|
11686
|
+
// set of already-running children; the UI can then say "restart these"
|
|
11687
|
+
// rather than claiming an existing process reloaded.
|
|
11688
|
+
if (envKey === "GROQ_API_KEY")
|
|
11689
|
+
return [];
|
|
11690
|
+
return [...this.children.keys()].sort();
|
|
11691
|
+
}
|
|
11692
|
+
providerSecretGeneration(envKey) {
|
|
11693
|
+
return this.providerSecretGenerations.get(envKey) ?? 0;
|
|
11694
|
+
}
|
|
11695
|
+
providerSecretEnvAllowed(envKey) {
|
|
11696
|
+
const configured = new Set((this.fleetConfig?.channels
|
|
11697
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []))
|
|
11698
|
+
.map(channel => channel.bot_token_env));
|
|
11699
|
+
return !configured.has(envKey) && providerRegistryEnvKeys().has(envKey);
|
|
11700
|
+
}
|
|
11701
|
+
async verifyProviderSecret(input) {
|
|
11702
|
+
const spec = providerSecretSpec(input.specId);
|
|
11703
|
+
if (!spec || !this.providerSecretEnvAllowed(spec.envKey)) {
|
|
11704
|
+
return { ok: false, status: "invalid", error: "provider secret is not configured" };
|
|
11705
|
+
}
|
|
11706
|
+
if (!spec.verifier)
|
|
11707
|
+
return { ok: false, status: "unsupported_verifier", error: "this provider has no supported verifier" };
|
|
11708
|
+
if (!input.secret || input.secret.length > 4096 || /[\r\n\0]/.test(input.secret) || /[^\x20-\x7e]/.test(input.secret)) {
|
|
11709
|
+
return { ok: false, status: "invalid", error: "secret is invalid" };
|
|
11710
|
+
}
|
|
11711
|
+
const challengeKey = `${input.sessionBinding}:api_key:${spec.id}:${spec.envKey}:${input.idempotencyKey}`;
|
|
11712
|
+
const existingId = this.providerSecretChallengesByKey.get(challengeKey);
|
|
11713
|
+
const existing = existingId ? this.providerSecretChallenges.get(existingId) : undefined;
|
|
11714
|
+
if (existing && existing.expiresAt > Date.now()) {
|
|
11715
|
+
return { ok: true, verification_id: existing.id, expires_at: existing.expiresAt, spec_id: spec.id, activation: spec.activation };
|
|
11716
|
+
}
|
|
11717
|
+
if (existingId)
|
|
11718
|
+
this.providerSecretChallengesByKey.delete(challengeKey);
|
|
11719
|
+
const result = await verifyProviderSecret(spec, input.secret, this.providerSecretHttpClient);
|
|
11720
|
+
if (!result.ok) {
|
|
11721
|
+
// Do not log provider detail: the HTTP verifier already redacted it and
|
|
11722
|
+
// this endpoint has no need to disclose whether a key was close to valid.
|
|
11723
|
+
this.logger.warn({ specId: spec.id, status: result.status }, "Provider API-key verification failed");
|
|
11724
|
+
return { ok: false, status: result.status, error: result.status === "unsupported_verifier" ? "this provider has no supported verifier" : "provider rejected or unavailable" };
|
|
11725
|
+
}
|
|
11726
|
+
const expiresAt = Date.now() + SECRET_CHALLENGE_TTL_MS;
|
|
11727
|
+
const challenge = {
|
|
11728
|
+
id: opaqueId("provider_verify"),
|
|
11729
|
+
specId: spec.id,
|
|
11730
|
+
envKey: spec.envKey,
|
|
11731
|
+
kind: "api_key",
|
|
11732
|
+
sessionBinding: input.sessionBinding,
|
|
11733
|
+
generation: this.providerSecretGeneration(spec.envKey),
|
|
11734
|
+
operation: "provider-secret.apply",
|
|
11735
|
+
idempotencyKey: input.idempotencyKey,
|
|
11736
|
+
expiresAt,
|
|
11737
|
+
secret: input.secret,
|
|
11738
|
+
};
|
|
11739
|
+
this.providerSecretChallenges.set(challenge.id, challenge);
|
|
11740
|
+
this.providerSecretChallengesByKey.set(challengeKey, challenge.id);
|
|
11741
|
+
const expiryTimer = setTimeout(() => {
|
|
11742
|
+
if (this.providerSecretChallenges.get(challenge.id) !== challenge)
|
|
11743
|
+
return;
|
|
11744
|
+
this.providerSecretChallenges.delete(challenge.id);
|
|
11745
|
+
if (this.providerSecretChallengesByKey.get(challengeKey) === challenge.id)
|
|
11746
|
+
this.providerSecretChallengesByKey.delete(challengeKey);
|
|
11747
|
+
}, SECRET_CHALLENGE_TTL_MS);
|
|
11748
|
+
expiryTimer.unref?.();
|
|
11749
|
+
return { ok: true, verification_id: challenge.id, expires_at: expiresAt, spec_id: spec.id, activation: spec.activation };
|
|
11750
|
+
}
|
|
11751
|
+
/** Naming aliases used by integrations that call this an API-key operation. */
|
|
11752
|
+
verifyProviderApiKey(input) {
|
|
11753
|
+
return this.verifyProviderSecret(input);
|
|
11754
|
+
}
|
|
11755
|
+
startProviderSecretApply(input) {
|
|
11756
|
+
for (const [jobId, job] of this.providerSecretJobs) {
|
|
11757
|
+
if (job.specId === input.specId && job.idempotencyKey === input.idempotencyKey
|
|
11758
|
+
&& this.providerSecretJobSession.get(jobId) === input.sessionBinding)
|
|
11759
|
+
return { job, reused: true };
|
|
11760
|
+
}
|
|
11761
|
+
const challenge = this.providerSecretChallenges.get(input.verificationId);
|
|
11762
|
+
const spec = providerSecretSpec(input.specId);
|
|
11763
|
+
if (!challenge || challenge.expiresAt <= Date.now()) {
|
|
11764
|
+
if (challenge)
|
|
11765
|
+
this.providerSecretChallenges.delete(input.verificationId);
|
|
11766
|
+
return { error: "verification expired; verify the secret again" };
|
|
11767
|
+
}
|
|
11768
|
+
if (!spec || challenge.specId !== spec.id || challenge.envKey !== spec.envKey || challenge.kind !== "api_key"
|
|
11769
|
+
|| challenge.sessionBinding !== input.sessionBinding || challenge.operation !== "provider-secret.apply"
|
|
11770
|
+
|| challenge.idempotencyKey !== input.idempotencyKey || challenge.generation !== this.providerSecretGeneration(challenge.envKey)) {
|
|
11771
|
+
return { error: "verification does not match this provider or session" };
|
|
11772
|
+
}
|
|
11773
|
+
const existingId = this.providerSecretInFlight.get(challenge.envKey);
|
|
11774
|
+
if (existingId) {
|
|
11775
|
+
const existing = this.providerSecretJobs.get(existingId) ?? null;
|
|
11776
|
+
if (existing?.idempotencyKey === input.idempotencyKey)
|
|
11777
|
+
return { job: existing, reused: true };
|
|
11778
|
+
return { busy: existing };
|
|
11779
|
+
}
|
|
11780
|
+
this.providerSecretChallenges.delete(input.verificationId);
|
|
11781
|
+
this.providerSecretChallengesByKey.delete(`${input.sessionBinding}:api_key:${spec.id}:${spec.envKey}:${input.idempotencyKey}`);
|
|
11782
|
+
const job = {
|
|
11783
|
+
id: opaqueId("provider_apply"), specId: spec.id, envKey: spec.envKey,
|
|
11784
|
+
idempotencyKey: input.idempotencyKey, result: "applying", status: "running", startedAt: Date.now(),
|
|
11785
|
+
stale_consumers: this.providerSecretStaleConsumers(spec.envKey),
|
|
11786
|
+
};
|
|
11787
|
+
this.providerSecretJobs.set(job.id, job);
|
|
11788
|
+
this.providerSecretJobSession.set(job.id, input.sessionBinding);
|
|
11789
|
+
this.providerSecretInFlight.set(spec.envKey, job.id);
|
|
11790
|
+
queueMicrotask(() => void this.runProviderSecretApply(job, challenge.secret));
|
|
11791
|
+
return { job, reused: false };
|
|
11792
|
+
}
|
|
11793
|
+
startProviderApiKeyApply(input) {
|
|
11794
|
+
return this.startProviderSecretApply(input);
|
|
11795
|
+
}
|
|
11796
|
+
getProviderSecretApply(jobId, sessionBinding) {
|
|
11797
|
+
if (this.providerSecretJobSession.get(jobId) !== sessionBinding)
|
|
11798
|
+
return null;
|
|
11799
|
+
return this.providerSecretJobs.get(jobId) ?? null;
|
|
11800
|
+
}
|
|
11801
|
+
getProviderApiKeyApply(jobId, sessionBinding) {
|
|
11802
|
+
return this.getProviderSecretApply(jobId, sessionBinding);
|
|
11803
|
+
}
|
|
11804
|
+
async runProviderSecretApply(job, secret) {
|
|
11805
|
+
const spec = providerSecretSpec(job.specId);
|
|
11806
|
+
const allowed = new Set([
|
|
11807
|
+
...PROVIDER_SECRET_SPECS.map(item => item.envKey),
|
|
11808
|
+
...(this.fleetConfig?.channels ?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : [])).map(channel => channel.bot_token_env),
|
|
11809
|
+
]);
|
|
11810
|
+
let store = null;
|
|
11811
|
+
let before = null;
|
|
11812
|
+
const previousProcessValue = process.env[job.envKey];
|
|
11813
|
+
let wrote = false;
|
|
11814
|
+
try {
|
|
11815
|
+
if (!spec || !this.providerSecretEnvAllowed(job.envKey))
|
|
11816
|
+
throw new Error("provider secret is not configured");
|
|
11817
|
+
// Construct inside the transaction: symlink/permission refusal must
|
|
11818
|
+
// settle the job as a safe failure, not escape the queued microtask.
|
|
11819
|
+
store = new SecretStore(join(this.dataDir, ".env"), allowed);
|
|
11820
|
+
before = store.write(job.envKey, secret);
|
|
11821
|
+
wrote = true;
|
|
11822
|
+
process.env[job.envKey] = secret;
|
|
11823
|
+
if (spec.activation === "reload_hook" && spec.reloadHookId) {
|
|
11824
|
+
await this.runProviderSecretReloadHook(spec.reloadHookId, secret, previousProcessValue);
|
|
11825
|
+
job.result = "reloaded";
|
|
11826
|
+
}
|
|
11827
|
+
else {
|
|
11828
|
+
job.result = "applied_next_use";
|
|
11829
|
+
}
|
|
11830
|
+
this.providerSecretGenerations.set(job.envKey, this.providerSecretGeneration(job.envKey) + 1);
|
|
11831
|
+
job.status = "done";
|
|
11832
|
+
job.finishedAt = Date.now();
|
|
11833
|
+
}
|
|
11834
|
+
catch (err) {
|
|
11835
|
+
const safe = safeSecretError(err, secret);
|
|
11836
|
+
this.logger.warn({ specId: job.specId, reason: safe }, "Provider API-key apply failed");
|
|
11837
|
+
try {
|
|
11838
|
+
if (wrote && before && store)
|
|
11839
|
+
store.restore(before);
|
|
11840
|
+
if (previousProcessValue === undefined)
|
|
11841
|
+
delete process.env[job.envKey];
|
|
11842
|
+
else
|
|
11843
|
+
process.env[job.envKey] = previousProcessValue;
|
|
11844
|
+
// SecretStore.write is itself transactional; when it fails before a
|
|
11845
|
+
// snapshot is returned there is no new value to roll back. Report the
|
|
11846
|
+
// truthful no-op rather than claiming rollback_failed.
|
|
11847
|
+
job.result = "rolled_back";
|
|
11848
|
+
if (!wrote || !before)
|
|
11849
|
+
job.error = "provider secret was not applied";
|
|
11850
|
+
}
|
|
11851
|
+
catch (rollbackErr) {
|
|
11852
|
+
this.logger.error({ specId: job.specId, reason: safeSecretError(rollbackErr, secret, previousProcessValue ? [previousProcessValue] : []) }, "Provider API-key rollback failed");
|
|
11853
|
+
job.result = "rollback_failed";
|
|
11854
|
+
job.error = "provider secret rollback failed; operator attention required";
|
|
11855
|
+
}
|
|
11856
|
+
job.status = "done";
|
|
11857
|
+
job.finishedAt = Date.now();
|
|
11858
|
+
}
|
|
11859
|
+
finally {
|
|
11860
|
+
this.providerSecretInFlight.delete(job.envKey);
|
|
11861
|
+
secret = "";
|
|
11862
|
+
}
|
|
11863
|
+
}
|
|
11864
|
+
/** Groq is currently read from process.env per voice request, so the hook is
|
|
11865
|
+
* intentionally a no-op. Keeping it as a named code-owned hook makes the
|
|
11866
|
+
* hot activation contract explicit and gives tests a failure seam; no generic
|
|
11867
|
+
* SIGHUP or caller-provided hook is ever executed. */
|
|
11868
|
+
async runProviderSecretReloadHook(hookId, _next, _previous) {
|
|
11869
|
+
if (hookId !== "groq.voice")
|
|
11870
|
+
throw new Error("unknown provider secret reload hook");
|
|
11871
|
+
const before = this.providerSecretHotSnapshots.get(hookId);
|
|
11872
|
+
this.providerSecretHotSnapshots.set(hookId, _next);
|
|
11873
|
+
const hook = this.providerSecretReloadHooks.get(hookId);
|
|
11874
|
+
try {
|
|
11875
|
+
if (hook)
|
|
11876
|
+
await hook(_next, before);
|
|
11877
|
+
}
|
|
11878
|
+
catch (err) {
|
|
11879
|
+
if (before === undefined)
|
|
11880
|
+
this.providerSecretHotSnapshots.delete(hookId);
|
|
11881
|
+
else
|
|
11882
|
+
this.providerSecretHotSnapshots.set(hookId, before);
|
|
11883
|
+
throw err;
|
|
11884
|
+
}
|
|
11885
|
+
}
|
|
11886
|
+
normalizeConnectionBinding(input) {
|
|
11887
|
+
// IDs arrive from JSON and may be Discord snowflakes. Do not accept a
|
|
11888
|
+
// number here: JSON.parse may already have rounded it before verification.
|
|
11889
|
+
if (typeof input.group_id !== "string")
|
|
11890
|
+
return null;
|
|
11891
|
+
const groupId = input.group_id.trim();
|
|
11892
|
+
if (!groupId || groupId.length > 128 || /[\r\n\0]/.test(groupId))
|
|
11893
|
+
return null;
|
|
11894
|
+
let general;
|
|
11895
|
+
if (input.general_channel_id === null || input.general_channel_id === undefined || input.general_channel_id === "") {
|
|
11896
|
+
general = input.general_channel_id === null ? null : undefined;
|
|
11897
|
+
}
|
|
11898
|
+
else if (typeof input.general_channel_id === "string") {
|
|
11899
|
+
general = input.general_channel_id.trim();
|
|
11900
|
+
if (!general || general.length > 128 || /[\r\n\0]/.test(general))
|
|
11901
|
+
return null;
|
|
11902
|
+
}
|
|
11903
|
+
else {
|
|
11904
|
+
return null;
|
|
11905
|
+
}
|
|
11906
|
+
return general === undefined ? { group_id: groupId } : { group_id: groupId, general_channel_id: general };
|
|
11907
|
+
}
|
|
11908
|
+
connectionBindingChannelConfig(channel, binding) {
|
|
11909
|
+
const candidate = structuredClone(channel);
|
|
11910
|
+
// IDs are intentionally normalized to strings at this boundary. Discord
|
|
11911
|
+
// snowflakes must never become YAML numbers (precision loss is silent).
|
|
11912
|
+
candidate.group_id = String(binding.group_id);
|
|
11913
|
+
if (binding.general_channel_id !== undefined) {
|
|
11914
|
+
const options = { ...(candidate.options ?? {}) };
|
|
11915
|
+
if (binding.general_channel_id === null)
|
|
11916
|
+
delete options.general_channel_id;
|
|
11917
|
+
else
|
|
11918
|
+
options.general_channel_id = String(binding.general_channel_id);
|
|
11919
|
+
if (Object.keys(options).length === 0)
|
|
11920
|
+
delete candidate.options;
|
|
11921
|
+
else
|
|
11922
|
+
candidate.options = options;
|
|
11923
|
+
}
|
|
11924
|
+
return candidate;
|
|
11925
|
+
}
|
|
11926
|
+
/** Verify a prospective group/guild binding without mutating fleet state. */
|
|
11927
|
+
async verifyConnectionBinding(input) {
|
|
11928
|
+
const channel = this.secureConnectionChannel(input.connectionId);
|
|
11929
|
+
const binding = this.normalizeConnectionBinding(input.binding);
|
|
11930
|
+
const adapter = this.adapters.get(input.connectionId);
|
|
11931
|
+
if (!channel || !binding || !adapter?.verifyBinding) {
|
|
11932
|
+
return { ok: false, error: "connection binding is unsupported or invalid" };
|
|
11933
|
+
}
|
|
11934
|
+
const key = `${input.sessionBinding}:${input.connectionId}:${input.idempotencyKey}`;
|
|
11935
|
+
const existingId = this.connectionBindingChallengesByKey.get(key);
|
|
11936
|
+
const existing = existingId ? this.connectionBindingChallenges.get(existingId) : undefined;
|
|
11937
|
+
if (existing && existing.expiresAt > Date.now()) {
|
|
11938
|
+
return { ok: true, verification_id: existing.id, expires_at: existing.expiresAt, binding: existing.binding, probe: existing.probe };
|
|
11939
|
+
}
|
|
11940
|
+
if (existingId)
|
|
11941
|
+
this.connectionBindingChallengesByKey.delete(key);
|
|
11942
|
+
const beforeGeneration = this.secureConnectionGeneration(input.connectionId);
|
|
11943
|
+
let probe;
|
|
11944
|
+
try {
|
|
11945
|
+
probe = await adapter.verifyBinding(binding.group_id, binding.general_channel_id ?? undefined);
|
|
11946
|
+
}
|
|
11947
|
+
catch (err) {
|
|
11948
|
+
this.logger.warn({ connectionId: input.connectionId, reason: safeSecretError(err) }, "Settings connection binding verification failed");
|
|
11949
|
+
return { ok: false, error: "binding verification failed" };
|
|
11950
|
+
}
|
|
11951
|
+
const afterGeneration = this.secureConnectionGeneration(input.connectionId);
|
|
11952
|
+
if (beforeGeneration !== afterGeneration || this.adapters.get(input.connectionId) !== adapter) {
|
|
11953
|
+
return { ok: false, error: "connection changed while binding was verified" };
|
|
11954
|
+
}
|
|
11955
|
+
if (probe.group_id !== binding.group_id || !probe.can_view || !probe.can_send) {
|
|
11956
|
+
return { ok: false, error: "provider did not confirm the requested binding" };
|
|
11957
|
+
}
|
|
11958
|
+
const expiresAt = Date.now() + SECRET_CHALLENGE_TTL_MS;
|
|
11959
|
+
const challenge = {
|
|
11960
|
+
id: opaqueId("binding_verify"),
|
|
11961
|
+
connectionId: input.connectionId,
|
|
11962
|
+
sessionBinding: input.sessionBinding,
|
|
11963
|
+
generation: afterGeneration,
|
|
11964
|
+
operation: "binding.apply",
|
|
11965
|
+
idempotencyKey: input.idempotencyKey,
|
|
11966
|
+
expiresAt,
|
|
11967
|
+
binding,
|
|
11968
|
+
probe,
|
|
11969
|
+
};
|
|
11970
|
+
this.connectionBindingChallenges.set(challenge.id, challenge);
|
|
11971
|
+
this.connectionBindingChallengesByKey.set(key, challenge.id);
|
|
11972
|
+
const expiryTimer = setTimeout(() => {
|
|
11973
|
+
if (this.connectionBindingChallenges.get(challenge.id) !== challenge)
|
|
11974
|
+
return;
|
|
11975
|
+
this.connectionBindingChallenges.delete(challenge.id);
|
|
11976
|
+
if (this.connectionBindingChallengesByKey.get(key) === challenge.id)
|
|
11977
|
+
this.connectionBindingChallengesByKey.delete(key);
|
|
11978
|
+
}, SECRET_CHALLENGE_TTL_MS);
|
|
11979
|
+
expiryTimer.unref?.();
|
|
11980
|
+
return { ok: true, verification_id: challenge.id, expires_at: expiresAt, binding, probe };
|
|
11981
|
+
}
|
|
11982
|
+
startConnectionBindingApply(input) {
|
|
11983
|
+
for (const [jobId, job] of this.connectionBindingJobs) {
|
|
11984
|
+
if (job.connectionId === input.connectionId && job.idempotencyKey === input.idempotencyKey
|
|
11985
|
+
&& this.connectionBindingJobSession.get(jobId) === input.sessionBinding)
|
|
11986
|
+
return { job, reused: true };
|
|
11987
|
+
}
|
|
11988
|
+
const challenge = this.connectionBindingChallenges.get(input.verificationId);
|
|
11989
|
+
if (!challenge || challenge.expiresAt <= Date.now()) {
|
|
11990
|
+
this.connectionBindingChallenges.delete(input.verificationId);
|
|
11991
|
+
return { error: "binding verification expired; verify the binding again" };
|
|
11992
|
+
}
|
|
11993
|
+
if (challenge.connectionId !== input.connectionId || challenge.sessionBinding !== input.sessionBinding
|
|
11994
|
+
|| challenge.operation !== "binding.apply" || challenge.idempotencyKey !== input.idempotencyKey
|
|
11995
|
+
|| challenge.generation !== this.secureConnectionGeneration(input.connectionId)) {
|
|
11996
|
+
return { error: "binding verification does not match this connection or session" };
|
|
11997
|
+
}
|
|
11998
|
+
const existingId = this.connectionBindingInFlight.get(input.connectionId);
|
|
11999
|
+
if (existingId) {
|
|
12000
|
+
const existing = this.connectionBindingJobs.get(existingId) ?? null;
|
|
12001
|
+
if (existing?.idempotencyKey === input.idempotencyKey)
|
|
12002
|
+
return { job: existing, reused: true };
|
|
12003
|
+
return { busy: existing };
|
|
12004
|
+
}
|
|
12005
|
+
this.connectionBindingChallenges.delete(input.verificationId);
|
|
12006
|
+
this.connectionBindingChallengesByKey.delete(`${input.sessionBinding}:${input.connectionId}:${input.idempotencyKey}`);
|
|
12007
|
+
const job = {
|
|
12008
|
+
id: opaqueId("binding_apply"), connectionId: input.connectionId, idempotencyKey: input.idempotencyKey,
|
|
12009
|
+
result: "applying", status: "running", startedAt: Date.now(),
|
|
12010
|
+
};
|
|
12011
|
+
this.connectionBindingJobs.set(job.id, job);
|
|
12012
|
+
this.connectionBindingJobSession.set(job.id, input.sessionBinding);
|
|
12013
|
+
this.connectionBindingInFlight.set(input.connectionId, job.id);
|
|
12014
|
+
queueMicrotask(() => void this.runConnectionBindingApply(job, challenge.binding));
|
|
12015
|
+
return { job, reused: false };
|
|
12016
|
+
}
|
|
12017
|
+
getConnectionBindingApply(jobId, sessionBinding) {
|
|
12018
|
+
if (this.connectionBindingJobSession.get(jobId) !== sessionBinding)
|
|
12019
|
+
return null;
|
|
12020
|
+
return this.connectionBindingJobs.get(jobId) ?? null;
|
|
12021
|
+
}
|
|
12022
|
+
async runConnectionBindingApply(job, binding) {
|
|
12023
|
+
try {
|
|
12024
|
+
await this.rebuildAdapterForBinding(job.connectionId, binding);
|
|
12025
|
+
job.result = "applied";
|
|
12026
|
+
}
|
|
12027
|
+
catch (err) {
|
|
12028
|
+
const reason = safeSecretError(err);
|
|
12029
|
+
job.result = /rollback failed/i.test(reason) ? "rollback_failed" : "rolled_back";
|
|
12030
|
+
job.error = job.result === "rollback_failed"
|
|
12031
|
+
? "binding rollback failed; adapter requires operator attention"
|
|
12032
|
+
: "binding was not applied; previous binding was restored";
|
|
12033
|
+
this.logger.warn({ connectionId: job.connectionId, reason: safeSecretError(err) }, "Settings connection binding apply failed");
|
|
12034
|
+
}
|
|
12035
|
+
finally {
|
|
12036
|
+
job.status = "done";
|
|
12037
|
+
job.finishedAt = Date.now();
|
|
12038
|
+
this.connectionBindingInFlight.delete(job.connectionId);
|
|
12039
|
+
}
|
|
12040
|
+
}
|
|
12041
|
+
/** Stop, rebuild and wait for a new adapter before committing YAML binding. */
|
|
12042
|
+
async rebuildAdapterForBinding(connectionId, binding) {
|
|
12043
|
+
const channel = this.secureConnectionChannel(connectionId);
|
|
12044
|
+
if (!channel || !this.fleetConfig)
|
|
12045
|
+
throw new Error("connection not found");
|
|
12046
|
+
const candidate = this.connectionBindingChannelConfig(channel, binding);
|
|
12047
|
+
const oldAdapter = this.adapters.get(connectionId);
|
|
12048
|
+
const oldWorld = this.worlds.get(connectionId);
|
|
12049
|
+
const oldPrimary = this.adapter;
|
|
12050
|
+
const oldAccess = this.accessManager;
|
|
12051
|
+
const oldState = this.adapterState.get(connectionId);
|
|
12052
|
+
const oldChannel = structuredClone(channel);
|
|
12053
|
+
const primary = this.getPrimaryAdapterId() === connectionId;
|
|
12054
|
+
if (primary && this.sessionPruneTimer) {
|
|
12055
|
+
clearInterval(this.sessionPruneTimer);
|
|
12056
|
+
this.sessionPruneTimer = null;
|
|
12057
|
+
}
|
|
12058
|
+
let fresh;
|
|
12059
|
+
let persistedBinding = false;
|
|
12060
|
+
try {
|
|
12061
|
+
this.adapterState.set(connectionId, { status: "retrying", retryCount: oldState?.retryCount ?? 0 });
|
|
12062
|
+
if (oldAdapter) {
|
|
12063
|
+
oldAdapter.removeAllListeners();
|
|
12064
|
+
await oldAdapter.stop().catch(() => { });
|
|
12065
|
+
if (this.adapters.get(connectionId) === oldAdapter)
|
|
12066
|
+
this.adapters.delete(connectionId);
|
|
12067
|
+
if (this.worlds.get(connectionId)?.adapter === oldAdapter)
|
|
12068
|
+
this.worlds.delete(connectionId);
|
|
12069
|
+
if (primary && this.adapter === oldAdapter)
|
|
12070
|
+
this.adapter = null;
|
|
12071
|
+
}
|
|
12072
|
+
let startedResolve = null;
|
|
12073
|
+
const started = new Promise(resolve => { startedResolve = resolve; });
|
|
12074
|
+
const onStarted = () => { startedResolve?.(); };
|
|
12075
|
+
if (primary)
|
|
12076
|
+
await this.startSingleAdapter(this.fleetConfig, candidate, onStarted);
|
|
12077
|
+
else
|
|
12078
|
+
await this.startAdditionalAdapter(candidate, true, onStarted);
|
|
12079
|
+
fresh = this.adapters.get(connectionId);
|
|
12080
|
+
if (!fresh)
|
|
12081
|
+
throw new Error("new adapter did not start");
|
|
12082
|
+
const deadline = Date.now() + 15_000;
|
|
12083
|
+
if (!fresh.getHealthSnapshot) {
|
|
12084
|
+
let timer;
|
|
12085
|
+
const timeout = new Promise((_, reject) => {
|
|
12086
|
+
timer = setTimeout(() => reject(new Error("new adapter did not become ready")), Math.max(1, deadline - Date.now()));
|
|
12087
|
+
timer.unref?.();
|
|
12088
|
+
});
|
|
12089
|
+
try {
|
|
12090
|
+
await Promise.race([started, timeout]);
|
|
12091
|
+
}
|
|
12092
|
+
finally {
|
|
12093
|
+
if (timer)
|
|
12094
|
+
clearTimeout(timer);
|
|
12095
|
+
}
|
|
12096
|
+
}
|
|
12097
|
+
else {
|
|
12098
|
+
while (Date.now() < deadline) {
|
|
12099
|
+
const health = fresh.getHealthSnapshot?.();
|
|
12100
|
+
if (health?.status === "connected" || this.adapterState.get(connectionId)?.status === "connected")
|
|
12101
|
+
break;
|
|
12102
|
+
await new Promise(resolve => setTimeout(resolve, 100));
|
|
12103
|
+
}
|
|
12104
|
+
const health = fresh.getHealthSnapshot?.();
|
|
12105
|
+
if (health && health.status !== "connected" && this.adapterState.get(connectionId)?.status !== "connected") {
|
|
12106
|
+
throw new Error("new adapter did not become connected");
|
|
12107
|
+
}
|
|
12108
|
+
}
|
|
12109
|
+
if (fresh.setChatId)
|
|
12110
|
+
fresh.setChatId(String(candidate.group_id));
|
|
12111
|
+
// Commit only after the replacement adapter is ready. No allowlist,
|
|
12112
|
+
// topic, instance or schedule fields are touched here.
|
|
12113
|
+
channel.group_id = String(binding.group_id);
|
|
12114
|
+
if (binding.general_channel_id !== undefined) {
|
|
12115
|
+
const options = { ...(channel.options ?? {}) };
|
|
12116
|
+
if (binding.general_channel_id === null)
|
|
12117
|
+
delete options.general_channel_id;
|
|
12118
|
+
else
|
|
12119
|
+
options.general_channel_id = String(binding.general_channel_id);
|
|
12120
|
+
if (Object.keys(options).length === 0)
|
|
12121
|
+
delete channel.options;
|
|
12122
|
+
else
|
|
12123
|
+
channel.options = options;
|
|
12124
|
+
}
|
|
12125
|
+
this.saveFleetConfig();
|
|
12126
|
+
persistedBinding = true;
|
|
12127
|
+
this.routing.rebuild(this.fleetConfig);
|
|
12128
|
+
this.reregisterClassicChannels();
|
|
12129
|
+
this.adapterState.set(connectionId, { status: "connected", retryCount: 0 });
|
|
12130
|
+
}
|
|
12131
|
+
catch (err) {
|
|
12132
|
+
if (fresh && fresh !== oldAdapter)
|
|
12133
|
+
await fresh.stop().catch(() => { });
|
|
12134
|
+
// Restore only the binding object in memory; unrelated connection and
|
|
12135
|
+
// instance state remains exactly as it was before the attempt.
|
|
12136
|
+
for (const key of Object.keys(channel)) {
|
|
12137
|
+
if (!(key in oldChannel))
|
|
12138
|
+
delete channel[key];
|
|
12139
|
+
}
|
|
12140
|
+
Object.assign(channel, oldChannel);
|
|
12141
|
+
this.adapters.delete(connectionId);
|
|
12142
|
+
this.worlds.delete(connectionId);
|
|
12143
|
+
this.adapterState.delete(connectionId);
|
|
12144
|
+
if (oldAdapter) {
|
|
12145
|
+
try {
|
|
12146
|
+
const onStarted = () => { };
|
|
12147
|
+
if (primary)
|
|
12148
|
+
await this.startSingleAdapter(this.fleetConfig, oldChannel, onStarted);
|
|
12149
|
+
else
|
|
12150
|
+
await this.startAdditionalAdapter(oldChannel, true, onStarted);
|
|
12151
|
+
this.adapterState.set(connectionId, oldState ?? { status: "connected", retryCount: 0 });
|
|
12152
|
+
}
|
|
12153
|
+
catch (restoreErr) {
|
|
12154
|
+
throw new Error(`binding rollback failed: ${safeSecretError(restoreErr)}`);
|
|
12155
|
+
}
|
|
12156
|
+
}
|
|
12157
|
+
else {
|
|
12158
|
+
if (primary)
|
|
12159
|
+
this.adapter = oldPrimary;
|
|
12160
|
+
if (oldWorld)
|
|
12161
|
+
this.worlds.set(connectionId, oldWorld);
|
|
12162
|
+
if (oldAdapter)
|
|
12163
|
+
this.adapters.set(connectionId, oldAdapter);
|
|
12164
|
+
this.accessManager = oldAccess;
|
|
12165
|
+
}
|
|
12166
|
+
// The binding is committed to YAML before routing is rebuilt. If the
|
|
12167
|
+
// post-commit rebuild fails, restore the durable document as well as the
|
|
12168
|
+
// in-memory channel; otherwise a reload would resurrect the failed
|
|
12169
|
+
// binding that the running fleet just rolled back.
|
|
12170
|
+
if (persistedBinding)
|
|
12171
|
+
this.saveFleetConfig();
|
|
12172
|
+
this.routing.rebuild(this.fleetConfig);
|
|
12173
|
+
this.reregisterClassicChannels();
|
|
12174
|
+
throw err;
|
|
12175
|
+
}
|
|
12176
|
+
}
|
|
12177
|
+
async verifyConnectionSecret(input) {
|
|
12178
|
+
const channel = this.secureConnectionChannel(input.connectionId);
|
|
12179
|
+
if (!channel || (channel.type !== "discord" && channel.type !== "telegram")) {
|
|
12180
|
+
return { ok: false, error: "connection not found or unsupported" };
|
|
12181
|
+
}
|
|
12182
|
+
if (!input.secret || input.secret.length > 4096 || /[\r\n\0]/.test(input.secret)) {
|
|
12183
|
+
return { ok: false, error: "secret is invalid" };
|
|
12184
|
+
}
|
|
12185
|
+
const challengeKey = `${input.sessionBinding}:${input.connectionId}:${input.idempotencyKey}`;
|
|
12186
|
+
const existingId = this.connectionSecretChallengesByKey.get(challengeKey);
|
|
12187
|
+
const existing = existingId ? this.connectionSecretChallenges.get(existingId) : undefined;
|
|
12188
|
+
if (existing && existing.expiresAt > Date.now()) {
|
|
12189
|
+
return { ok: true, verification_id: existing.id, expires_at: existing.expiresAt };
|
|
12190
|
+
}
|
|
12191
|
+
if (existingId)
|
|
12192
|
+
this.connectionSecretChallengesByKey.delete(challengeKey);
|
|
12193
|
+
// Fixed provider endpoints only. Never use a user-supplied URL and never
|
|
12194
|
+
// call Telegram getUpdates (the running adapter owns that long poll).
|
|
12195
|
+
const identity = channel.type === "discord"
|
|
12196
|
+
? await verifyDiscordToken(input.secret)
|
|
12197
|
+
: await verifyTelegramToken(input.secret);
|
|
12198
|
+
if (!identity.valid) {
|
|
12199
|
+
this.logger.warn({ connectionId: input.connectionId, provider: channel.type }, "Settings connection secret verification failed");
|
|
12200
|
+
return { ok: false, error: "provider rejected the secret" };
|
|
12201
|
+
}
|
|
12202
|
+
const expiresAt = Date.now() + SECRET_CHALLENGE_TTL_MS;
|
|
12203
|
+
const challenge = {
|
|
12204
|
+
id: opaqueId("verify"),
|
|
12205
|
+
connectionId: input.connectionId,
|
|
12206
|
+
sessionBinding: input.sessionBinding,
|
|
12207
|
+
generation: this.secureConnectionGeneration(input.connectionId),
|
|
12208
|
+
operation: "secret.apply",
|
|
12209
|
+
idempotencyKey: input.idempotencyKey,
|
|
12210
|
+
expiresAt,
|
|
12211
|
+
secret: input.secret,
|
|
12212
|
+
};
|
|
12213
|
+
this.connectionSecretChallenges.set(challenge.id, challenge);
|
|
12214
|
+
this.connectionSecretChallengesByKey.set(challengeKey, challenge.id);
|
|
12215
|
+
const expiryTimer = setTimeout(() => {
|
|
12216
|
+
if (this.connectionSecretChallenges.get(challenge.id) !== challenge)
|
|
12217
|
+
return;
|
|
12218
|
+
this.connectionSecretChallenges.delete(challenge.id);
|
|
12219
|
+
if (this.connectionSecretChallengesByKey.get(challengeKey) === challenge.id) {
|
|
12220
|
+
this.connectionSecretChallengesByKey.delete(challengeKey);
|
|
12221
|
+
}
|
|
12222
|
+
}, SECRET_CHALLENGE_TTL_MS);
|
|
12223
|
+
expiryTimer.unref?.();
|
|
12224
|
+
return {
|
|
12225
|
+
ok: true,
|
|
12226
|
+
verification_id: challenge.id,
|
|
12227
|
+
expires_at: expiresAt,
|
|
12228
|
+
identity: { id: identity.id, username: identity.username },
|
|
12229
|
+
};
|
|
12230
|
+
}
|
|
12231
|
+
startConnectionSecretApply(input) {
|
|
12232
|
+
for (const [jobId, job] of this.connectionSecretJobs) {
|
|
12233
|
+
if (job.connectionId === input.connectionId
|
|
12234
|
+
&& job.idempotencyKey === input.idempotencyKey
|
|
12235
|
+
&& this.connectionSecretJobSession.get(jobId) === input.sessionBinding) {
|
|
12236
|
+
return { job, reused: true };
|
|
12237
|
+
}
|
|
12238
|
+
}
|
|
12239
|
+
const challenge = this.connectionSecretChallenges.get(input.verificationId);
|
|
12240
|
+
if (!challenge || challenge.expiresAt <= Date.now()) {
|
|
12241
|
+
this.connectionSecretChallenges.delete(input.verificationId);
|
|
12242
|
+
return { error: "verification expired; verify the secret again" };
|
|
12243
|
+
}
|
|
12244
|
+
if (challenge.connectionId !== input.connectionId
|
|
12245
|
+
|| challenge.sessionBinding !== input.sessionBinding
|
|
12246
|
+
|| challenge.operation !== "secret.apply"
|
|
12247
|
+
|| challenge.idempotencyKey !== input.idempotencyKey
|
|
12248
|
+
|| challenge.generation !== this.secureConnectionGeneration(input.connectionId)) {
|
|
12249
|
+
return { error: "verification does not match this connection or session" };
|
|
12250
|
+
}
|
|
12251
|
+
const existingId = this.connectionSecretInFlight.get(input.connectionId);
|
|
12252
|
+
if (existingId) {
|
|
12253
|
+
const existing = this.connectionSecretJobs.get(existingId) ?? null;
|
|
12254
|
+
if (existing?.idempotencyKey === input.idempotencyKey)
|
|
12255
|
+
return { job: existing, reused: true };
|
|
12256
|
+
return { busy: existing };
|
|
12257
|
+
}
|
|
12258
|
+
// Consume the challenge before scheduling work. A lost HTTP response can
|
|
12259
|
+
// retry with the same idempotency key and rejoin the job, but a second
|
|
12260
|
+
// request cannot replay the secret into a second adapter.
|
|
12261
|
+
this.connectionSecretChallenges.delete(input.verificationId);
|
|
12262
|
+
this.connectionSecretChallengesByKey.delete(`${input.sessionBinding}:${input.connectionId}:${input.idempotencyKey}`);
|
|
12263
|
+
const job = {
|
|
12264
|
+
id: opaqueId("secret_apply"),
|
|
12265
|
+
connectionId: input.connectionId,
|
|
12266
|
+
idempotencyKey: input.idempotencyKey,
|
|
12267
|
+
result: "applying",
|
|
12268
|
+
status: "running",
|
|
12269
|
+
startedAt: Date.now(),
|
|
12270
|
+
};
|
|
12271
|
+
this.connectionSecretJobs.set(job.id, job);
|
|
12272
|
+
this.connectionSecretJobSession.set(job.id, input.sessionBinding);
|
|
12273
|
+
this.connectionSecretInFlight.set(input.connectionId, job.id);
|
|
12274
|
+
queueMicrotask(() => void this.runConnectionSecretApply(job, challenge.secret));
|
|
12275
|
+
return { job, reused: false };
|
|
12276
|
+
}
|
|
12277
|
+
getConnectionSecretApply(jobId, sessionBinding) {
|
|
12278
|
+
if (this.connectionSecretJobSession.get(jobId) !== sessionBinding)
|
|
12279
|
+
return null;
|
|
12280
|
+
return this.connectionSecretJobs.get(jobId) ?? null;
|
|
12281
|
+
}
|
|
12282
|
+
async runConnectionSecretApply(job, secret) {
|
|
12283
|
+
const channel = this.secureConnectionChannel(job.connectionId);
|
|
12284
|
+
const envKey = channel?.bot_token_env;
|
|
12285
|
+
const allowed = new Set((this.fleetConfig?.channels
|
|
12286
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []))
|
|
12287
|
+
.map(item => item.bot_token_env));
|
|
12288
|
+
let before = null;
|
|
12289
|
+
let oldToken;
|
|
12290
|
+
let replaced = false;
|
|
12291
|
+
try {
|
|
12292
|
+
if (!channel || !envKey || !allowed.has(envKey))
|
|
12293
|
+
throw new Error("connection is not configured for secret rotation");
|
|
12294
|
+
const owners = (this.fleetConfig?.channels
|
|
12295
|
+
?? (this.fleetConfig?.channel ? [this.fleetConfig.channel] : []))
|
|
12296
|
+
.filter(item => item.bot_token_env === envKey);
|
|
12297
|
+
if (owners.length !== 1)
|
|
12298
|
+
throw new Error("secret key is shared by multiple connections");
|
|
12299
|
+
const store = new SecretStore(join(this.dataDir, ".env"), allowed);
|
|
12300
|
+
before = store.write(envKey, secret);
|
|
12301
|
+
replaced = true;
|
|
12302
|
+
oldToken = process.env[envKey];
|
|
12303
|
+
process.env[envKey] = secret;
|
|
12304
|
+
const generation = this.secureConnectionGeneration(job.connectionId) + 1;
|
|
12305
|
+
this.connectionSecretGenerations.set(job.connectionId, generation);
|
|
12306
|
+
const applied = await this.rebuildAdapterForSecret(job.connectionId, channel);
|
|
12307
|
+
if (!applied) {
|
|
12308
|
+
job.result = "restart_required";
|
|
12309
|
+
job.status = "done";
|
|
12310
|
+
job.finishedAt = Date.now();
|
|
12311
|
+
return;
|
|
12312
|
+
}
|
|
12313
|
+
job.result = "applied";
|
|
12314
|
+
job.status = "done";
|
|
12315
|
+
job.finishedAt = Date.now();
|
|
12316
|
+
}
|
|
12317
|
+
catch (err) {
|
|
12318
|
+
const message = safeSecretError(err, secret);
|
|
12319
|
+
this.logger.warn({ connectionId: job.connectionId, reason: message }, "Settings connection secret apply failed");
|
|
12320
|
+
if (!replaced || !before || !envKey) {
|
|
12321
|
+
job.result = "rollback_failed";
|
|
12322
|
+
job.error = "secret was not applied";
|
|
12323
|
+
}
|
|
12324
|
+
else {
|
|
12325
|
+
try {
|
|
12326
|
+
const store = new SecretStore(join(this.dataDir, ".env"), new Set([envKey]));
|
|
12327
|
+
store.restore(before);
|
|
12328
|
+
if (oldToken === undefined)
|
|
12329
|
+
delete process.env[envKey];
|
|
12330
|
+
else
|
|
12331
|
+
process.env[envKey] = oldToken;
|
|
12332
|
+
// Build a fresh adapter from the restored token. If the old adapter
|
|
12333
|
+
// was stopped already, this is the only safe way to return to the
|
|
12334
|
+
// previous runtime without claiming a disk-only rollback succeeded.
|
|
12335
|
+
const restored = await this.rebuildAdapterForSecret(job.connectionId, channel, true);
|
|
12336
|
+
if (!restored)
|
|
12337
|
+
throw new Error("adapter rollback did not become connected");
|
|
12338
|
+
job.result = "rolled_back";
|
|
12339
|
+
}
|
|
12340
|
+
catch (rollbackErr) {
|
|
12341
|
+
this.logger.error({ connectionId: job.connectionId, reason: safeSecretError(rollbackErr, oldToken, [secret]) }, "Settings connection secret rollback failed");
|
|
12342
|
+
job.result = "rollback_failed";
|
|
12343
|
+
job.error = "secret rollback failed; adapter is disabled";
|
|
12344
|
+
}
|
|
12345
|
+
}
|
|
12346
|
+
job.status = "done";
|
|
12347
|
+
job.finishedAt = Date.now();
|
|
12348
|
+
}
|
|
12349
|
+
finally {
|
|
12350
|
+
this.connectionSecretInFlight.delete(job.connectionId);
|
|
12351
|
+
// Do not retain the token after the apply (success or rollback).
|
|
12352
|
+
secret = "";
|
|
12353
|
+
}
|
|
12354
|
+
}
|
|
12355
|
+
/** Stop the old provider client and construct a new one from process.env. */
|
|
12356
|
+
async rebuildAdapterForSecret(connectionId, channel, force = false) {
|
|
12357
|
+
const old = this.adapters.get(connectionId);
|
|
12358
|
+
if (!old && !force)
|
|
12359
|
+
return false; // The secret is valid on disk; the next start adopts it.
|
|
12360
|
+
const primary = this.getPrimaryAdapterId() === connectionId;
|
|
12361
|
+
const previousState = this.adapterState.get(connectionId);
|
|
12362
|
+
this.adapterState.set(connectionId, { status: "retrying", retryCount: previousState?.retryCount ?? 0 });
|
|
12363
|
+
if (primary && this.sessionPruneTimer) {
|
|
12364
|
+
clearInterval(this.sessionPruneTimer);
|
|
12365
|
+
this.sessionPruneTimer = null;
|
|
12366
|
+
}
|
|
12367
|
+
if (old) {
|
|
12368
|
+
old.removeAllListeners();
|
|
12369
|
+
await old.stop().catch(() => { });
|
|
12370
|
+
if (this.adapters.get(connectionId) === old)
|
|
12371
|
+
this.adapters.delete(connectionId);
|
|
12372
|
+
if (this.worlds.get(connectionId)?.adapter === old)
|
|
12373
|
+
this.worlds.delete(connectionId);
|
|
12374
|
+
if (primary && this.adapter === old)
|
|
12375
|
+
this.adapter = null;
|
|
12376
|
+
}
|
|
12377
|
+
let startedResolve = null;
|
|
12378
|
+
const started = new Promise(resolve => { startedResolve = resolve; });
|
|
12379
|
+
const onStarted = () => { startedResolve?.(); };
|
|
12380
|
+
if (primary)
|
|
12381
|
+
await this.startSingleAdapter(this.fleetConfig, channel, onStarted);
|
|
12382
|
+
else
|
|
12383
|
+
await this.startAdditionalAdapter(channel, true, onStarted);
|
|
12384
|
+
const fresh = this.adapters.get(connectionId);
|
|
12385
|
+
if (!fresh)
|
|
12386
|
+
throw new Error("new adapter did not start");
|
|
12387
|
+
const deadline = Date.now() + 15_000;
|
|
12388
|
+
// Telegram has no gateway health snapshot. Its start() method launches the
|
|
12389
|
+
// grammY polling loop in the background, so completion of start() is not a
|
|
12390
|
+
// connected signal. The adapter's `started` event is emitted only after the
|
|
12391
|
+
// first provider getMe succeeds; require that event before claiming apply.
|
|
12392
|
+
if (!fresh.getHealthSnapshot) {
|
|
12393
|
+
let timer;
|
|
12394
|
+
const timeout = new Promise((_, reject) => {
|
|
12395
|
+
timer = setTimeout(() => reject(new Error("new adapter did not emit started before deadline")), Math.max(1, deadline - Date.now()));
|
|
12396
|
+
timer.unref?.();
|
|
12397
|
+
});
|
|
12398
|
+
try {
|
|
12399
|
+
await Promise.race([started, timeout]);
|
|
12400
|
+
}
|
|
12401
|
+
finally {
|
|
12402
|
+
if (timer)
|
|
12403
|
+
clearTimeout(timer);
|
|
12404
|
+
}
|
|
12405
|
+
this.adapterState.set(connectionId, { status: "connected", retryCount: 0 });
|
|
12406
|
+
return true;
|
|
12407
|
+
}
|
|
12408
|
+
while (Date.now() < deadline) {
|
|
12409
|
+
const health = fresh.getHealthSnapshot?.();
|
|
12410
|
+
if (health?.status === "connected" || this.adapterState.get(connectionId)?.status === "connected")
|
|
12411
|
+
return true;
|
|
12412
|
+
await new Promise(resolve => setTimeout(resolve, 100));
|
|
12413
|
+
}
|
|
12414
|
+
throw new Error("new adapter did not become connected");
|
|
12415
|
+
}
|
|
12416
|
+
/**
|
|
12417
|
+
* What a reconcile is about to do, per target.
|
|
12418
|
+
*
|
|
12419
|
+
* A forecast, not the record: it is built from the same `classifyInstanceChange`
|
|
12420
|
+
* the reconcile decides with, but the rows that end up in the job are the ones
|
|
12421
|
+
* the reconcile reports as it works. A target the forecast missed (a Classic
|
|
12422
|
+
* instance inheriting a changed default) is added when it is first touched.
|
|
12423
|
+
*/
|
|
12424
|
+
planConfigApply() {
|
|
12425
|
+
const rows = [];
|
|
12426
|
+
const nextConfig = this.nextFleetConfig();
|
|
12427
|
+
const next = nextConfig?.instances ?? {};
|
|
12428
|
+
const classicNames = new Set(this.classicChannels?.getAll().map(ch => ch.instanceName) ?? []);
|
|
12429
|
+
for (const [name, config] of Object.entries(next)) {
|
|
12430
|
+
const daemon = this.daemons.get(name);
|
|
12431
|
+
if (!daemon) {
|
|
12432
|
+
rows.push({ target: name, kind: "restart" });
|
|
12433
|
+
continue;
|
|
12434
|
+
}
|
|
12435
|
+
const runtime = daemon.getConfigSnapshot?.();
|
|
12436
|
+
if (!runtime)
|
|
12437
|
+
continue;
|
|
12438
|
+
const change = classifyInstanceChange(runtime, config);
|
|
12439
|
+
if (change !== "none")
|
|
12440
|
+
rows.push({ target: name, kind: change === "restart" ? "restart" : "hot" });
|
|
12441
|
+
}
|
|
12442
|
+
for (const name of this.daemons.keys()) {
|
|
12443
|
+
if (!(name in next) && !classicNames.has(name))
|
|
12444
|
+
rows.push({ target: name, kind: "restart" });
|
|
12445
|
+
}
|
|
12446
|
+
if (this.appliedFleetLevel !== null && this.appliedFleetLevel !== this.fleetLevelSignature(nextConfig)) {
|
|
12447
|
+
rows.push({ target: APPLY_FLEET_TARGET, kind: "restart" });
|
|
12448
|
+
}
|
|
12449
|
+
return rows;
|
|
12450
|
+
}
|
|
12451
|
+
/**
|
|
12452
|
+
* Start (or re-join) a Settings apply.
|
|
12453
|
+
*
|
|
12454
|
+
* The key is the client's. Handing back the existing job for a repeated key is
|
|
12455
|
+
* the whole point: the retry after a lost response must not apply everything a
|
|
12456
|
+
* second time.
|
|
12457
|
+
*/
|
|
12458
|
+
startSettingsApply(key) {
|
|
12459
|
+
// The key is checked first on purpose: a retry of the apply that is running
|
|
12460
|
+
// right now must get its own job back, not "busy".
|
|
12461
|
+
const existing = this.applyJobs.findByKey(key);
|
|
12462
|
+
if (existing)
|
|
12463
|
+
return { job: existing, reused: true };
|
|
12464
|
+
if (this.reconcileInFlight || this.activeApplyJobId) {
|
|
12465
|
+
return { busy: this.activeApplyJobId ? this.applyJobs.get(this.activeApplyJobId) : null };
|
|
12466
|
+
}
|
|
12467
|
+
const job = this.applyJobs.create(key, this.planConfigApply());
|
|
12468
|
+
// Reserved synchronously: the work starts a microtask later, and a second
|
|
12469
|
+
// request arriving in that gap must see the slot taken.
|
|
12470
|
+
this.activeApplyJobId = job.id;
|
|
12471
|
+
// Start after the caller has its answer, so the first thing the page renders
|
|
12472
|
+
// is the whole plan with every row still pending — not a job the reconcile
|
|
12473
|
+
// has already half-finished synchronously.
|
|
12474
|
+
queueMicrotask(() => void this.runSettingsApply(job.id));
|
|
12475
|
+
return { job, reused: false };
|
|
12476
|
+
}
|
|
12477
|
+
async runSettingsApply(jobId) {
|
|
12478
|
+
const emit = () => {
|
|
12479
|
+
const job = this.applyJobs.get(jobId);
|
|
12480
|
+
// An accelerator only: these frames carry no event id, so a client that
|
|
12481
|
+
// reconnects cannot ask for what it missed. GET is the authority.
|
|
12482
|
+
if (job)
|
|
12483
|
+
this.emitSseEvent("apply_progress", viewOf(job));
|
|
12484
|
+
};
|
|
12485
|
+
// Passed in rather than parked on `this`: a shared field would let a second
|
|
12486
|
+
// reconcile redirect this job's reporting into another job's rows.
|
|
12487
|
+
const observer = (target, kind, status, error) => {
|
|
12488
|
+
this.applyJobs.update(jobId, job => {
|
|
12489
|
+
let row = job.targets.find(item => item.target === target);
|
|
12490
|
+
if (!row) {
|
|
12491
|
+
row = { target, kind, status: "pending" };
|
|
12492
|
+
job.targets.push(row);
|
|
12493
|
+
}
|
|
12494
|
+
row.kind = kind;
|
|
12495
|
+
row.status = status;
|
|
12496
|
+
if (error)
|
|
12497
|
+
row.error = error;
|
|
12498
|
+
});
|
|
12499
|
+
emit();
|
|
12500
|
+
};
|
|
12501
|
+
emit();
|
|
12502
|
+
try {
|
|
12503
|
+
const started = this.startExclusiveReconcile(observer);
|
|
12504
|
+
if (!started) {
|
|
12505
|
+
// The slot was reserved before the microtask, so this means a SIGHUP
|
|
12506
|
+
// reconcile started in between. Report it instead of applying twice.
|
|
12507
|
+
this.applyJobs.finish(jobId, "a config reload was already running");
|
|
12508
|
+
return;
|
|
12509
|
+
}
|
|
12510
|
+
const outcome = await started;
|
|
12511
|
+
this.applyJobs.finish(jobId, outcome.rejected);
|
|
12512
|
+
}
|
|
12513
|
+
catch (err) {
|
|
12514
|
+
this.applyJobs.finish(jobId, err instanceof Error ? err.message : String(err));
|
|
12515
|
+
}
|
|
12516
|
+
finally {
|
|
12517
|
+
this.activeApplyJobId = null;
|
|
12518
|
+
emit();
|
|
12519
|
+
}
|
|
10757
12520
|
}
|
|
10758
12521
|
async restartInstances() {
|
|
10759
12522
|
if (!this.configPath) {
|
|
@@ -11046,6 +12809,12 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
11046
12809
|
this.initializeWebAuthTokens();
|
|
11047
12810
|
this.healthServer = createServer((req, res) => {
|
|
11048
12811
|
res.setHeader("Content-Type", "application/json");
|
|
12812
|
+
// No Referer to a tunnel host, an upstream proxy, or any page linked from
|
|
12813
|
+
// the panel — the dashboard URL is itself a credential-bearing address.
|
|
12814
|
+
res.setHeader("Referrer-Policy", "no-referrer");
|
|
12815
|
+
// Authorization now depends on a cookie, so a shared cache (a tunnel, a
|
|
12816
|
+
// corporate proxy) must not serve one visitor's response to another.
|
|
12817
|
+
res.setHeader("Vary", "Cookie");
|
|
11049
12818
|
const requestPath = new URL(req.url ?? "/", `http://localhost:${port}`).pathname;
|
|
11050
12819
|
// Browsers request this automatically and AgEnD does not ship an icon.
|
|
11051
12820
|
// It is neither user data nor an API route, so do not turn the harmless
|
|
@@ -11071,15 +12840,25 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
11071
12840
|
// like the other /view data routes (usage-api.ts rejects non-GET).
|
|
11072
12841
|
}
|
|
11073
12842
|
else {
|
|
11074
|
-
// All other endpoints require a
|
|
12843
|
+
// All other endpoints require a session cookie or an X-Agend-Token
|
|
12844
|
+
// header; a `?token=` in the URL is only redeemed for a cookie on a GET.
|
|
11075
12845
|
// /ui/* will also re-check in web-api.ts, which is harmless.
|
|
11076
12846
|
const parsedUrl = new URL(req.url ?? "/", `http://localhost:${port}`);
|
|
11077
|
-
const
|
|
11078
|
-
|
|
11079
|
-
|
|
11080
|
-
|
|
11081
|
-
|
|
11082
|
-
|
|
12847
|
+
const decision = decideWebGate(req, parsedUrl, this.webToken);
|
|
12848
|
+
if (decision.kind === "reject") {
|
|
12849
|
+
res.writeHead(decision.status);
|
|
12850
|
+
res.end(JSON.stringify({ error: decision.message }));
|
|
12851
|
+
return;
|
|
12852
|
+
}
|
|
12853
|
+
if (decision.kind === "exchange") {
|
|
12854
|
+
res.setHeader("Set-Cookie", decision.setCookie);
|
|
12855
|
+
res.setHeader("Location", decision.location);
|
|
12856
|
+
// A cached redirect would replay a Set-Cookie for a rotated token.
|
|
12857
|
+
res.setHeader("Cache-Control", "no-store");
|
|
12858
|
+
res.writeHead(302);
|
|
12859
|
+
// Browsers follow the Location; a script that does not gets told why
|
|
12860
|
+
// its URL token stopped being echoed back as data.
|
|
12861
|
+
res.end(JSON.stringify({ redirect: decision.location }));
|
|
11083
12862
|
return;
|
|
11084
12863
|
}
|
|
11085
12864
|
}
|
|
@@ -11317,8 +13096,11 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
11317
13096
|
this.logger.info({ port }, afterTakeover
|
|
11318
13097
|
? "Health endpoint listening (after takeover)"
|
|
11319
13098
|
: "Health endpoint listening");
|
|
11320
|
-
|
|
11321
|
-
|
|
13099
|
+
// Never the token: fleet.log is readable by anything that can read the
|
|
13100
|
+
// data dir, is copied into bug reports, and is tailed in shared terminals.
|
|
13101
|
+
// `/dashboard` and `agend web` are the ways to get an authorized link.
|
|
13102
|
+
this.logger.info({ url: `http://localhost:${port}/ui` }, "Web UI available (open it with /dashboard or `agend web`)");
|
|
13103
|
+
this.logger.info({ url: `http://localhost:${port}/view` }, "Web View available");
|
|
11322
13104
|
};
|
|
11323
13105
|
this.healthServer.on("error", (err) => {
|
|
11324
13106
|
this.healthServerListening = false;
|
|
@@ -11335,8 +13117,25 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
11335
13117
|
if (existsSync(pidPath)) {
|
|
11336
13118
|
const oldPid = parseInt(readFileSync(pidPath, "utf-8").trim(), 10);
|
|
11337
13119
|
if (oldPid && oldPid !== process.pid) {
|
|
11338
|
-
|
|
11339
|
-
|
|
13120
|
+
// fleet.pid is a claim, not proof. A stale or wrong entry points
|
|
13121
|
+
// at whatever now holds that pid, and this used to SIGTERM it —
|
|
13122
|
+
// an unrelated process killed because a port was busy. Confirm
|
|
13123
|
+
// the target really is an AgEnD fleet, and when that cannot be
|
|
13124
|
+
// confirmed, do not signal: not killing costs a dashboard, and
|
|
13125
|
+
// killing costs somebody else's process.
|
|
13126
|
+
const commandLine = readProcessCommandLine(oldPid);
|
|
13127
|
+
if (isFleetStartCommandLine(commandLine)) {
|
|
13128
|
+
process.kill(oldPid, "SIGTERM");
|
|
13129
|
+
this.logger.info({ oldPid }, "Killed old fleet process");
|
|
13130
|
+
}
|
|
13131
|
+
else {
|
|
13132
|
+
this.logger.warn({
|
|
13133
|
+
oldPid,
|
|
13134
|
+
// Truncated: this is an unrelated process's command line, and
|
|
13135
|
+
// fleet.log is copied into bug reports.
|
|
13136
|
+
commandLine: commandLine ? `${commandLine.slice(0, 60)}${commandLine.length > 60 ? "…" : ""}` : "(unreadable)",
|
|
13137
|
+
}, "fleet.pid does not name an AgEnD fleet process — not signalling it");
|
|
13138
|
+
}
|
|
11340
13139
|
}
|
|
11341
13140
|
}
|
|
11342
13141
|
}
|