@songsid/agend 2.1.4-beta.5 → 2.1.4-beta.50
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/README.zh-TW.md +1 -1
- package/dist/agent-endpoint.js +1 -1
- package/dist/agent-endpoint.js.map +1 -1
- package/dist/backend/antigravity.d.ts +12 -8
- package/dist/backend/antigravity.js +189 -42
- package/dist/backend/antigravity.js.map +1 -1
- package/dist/backend/claude-code.d.ts +62 -23
- package/dist/backend/claude-code.js +296 -43
- package/dist/backend/claude-code.js.map +1 -1
- package/dist/backend/codex.d.ts +18 -0
- package/dist/backend/codex.js +148 -8
- package/dist/backend/codex.js.map +1 -1
- package/dist/backend/gemini-cli.js +2 -2
- package/dist/backend/gemini-cli.js.map +1 -1
- package/dist/backend/grok.js +2 -2
- package/dist/backend/grok.js.map +1 -1
- package/dist/backend/kiro.d.ts +52 -2
- package/dist/backend/kiro.js +267 -12
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/opencode.d.ts +27 -11
- package/dist/backend/opencode.js +65 -41
- package/dist/backend/opencode.js.map +1 -1
- package/dist/backend/types.d.ts +97 -3
- package/dist/backend/types.js +12 -1
- package/dist/backend/types.js.map +1 -1
- package/dist/backend-outage.d.ts +61 -0
- package/dist/backend-outage.js +71 -0
- package/dist/backend-outage.js.map +1 -0
- package/dist/channel/adapters/discord.d.ts +44 -2
- package/dist/channel/adapters/discord.js +526 -78
- package/dist/channel/adapters/discord.js.map +1 -1
- package/dist/channel/adapters/telegram.d.ts +30 -0
- package/dist/channel/adapters/telegram.js +221 -19
- package/dist/channel/adapters/telegram.js.map +1 -1
- package/dist/channel/agy-mcp-launcher.d.ts +2 -0
- package/dist/channel/agy-mcp-launcher.js +28 -0
- package/dist/channel/agy-mcp-launcher.js.map +1 -0
- package/dist/channel/ipc-bridge.d.ts +2 -1
- package/dist/channel/ipc-bridge.js +33 -4
- package/dist/channel/ipc-bridge.js.map +1 -1
- package/dist/channel/mcp-server.js +28 -25
- package/dist/channel/mcp-server.js.map +1 -1
- package/dist/channel/mcp-tools.js +4 -2
- package/dist/channel/mcp-tools.js.map +1 -1
- package/dist/channel/message-queue.js +62 -1
- package/dist/channel/message-queue.js.map +1 -1
- package/dist/channel/types.d.ts +31 -1
- package/dist/classic-channel-manager.d.ts +61 -3
- package/dist/classic-channel-manager.js +256 -31
- package/dist/classic-channel-manager.js.map +1 -1
- package/dist/cli.js +109 -37
- package/dist/cli.js.map +1 -1
- package/dist/config-validator.js +48 -7
- package/dist/config-validator.js.map +1 -1
- package/dist/config.d.ts +4 -0
- package/dist/config.js +12 -1
- package/dist/config.js.map +1 -1
- package/dist/cross-instance-envelope.d.ts +6 -0
- package/dist/cross-instance-envelope.js +30 -0
- package/dist/cross-instance-envelope.js.map +1 -0
- package/dist/daemon.d.ts +429 -9
- package/dist/daemon.js +1843 -289
- package/dist/daemon.js.map +1 -1
- package/dist/doctor.d.ts +41 -0
- package/dist/doctor.js +267 -0
- package/dist/doctor.js.map +1 -0
- package/dist/fleet-context.d.ts +54 -0
- package/dist/fleet-manager.d.ts +398 -11
- package/dist/fleet-manager.js +2756 -499
- package/dist/fleet-manager.js.map +1 -1
- package/dist/fleet-yaml-slim.d.ts +10 -0
- package/dist/fleet-yaml-slim.js +59 -0
- package/dist/fleet-yaml-slim.js.map +1 -0
- package/dist/general-knowledge/skills/backend-providers/SKILL.md +114 -0
- package/dist/general-knowledge/skills/cross-instance-messaging/SKILL.md +3 -1
- package/dist/general-knowledge/skills/delegation-playbook/SKILL.md +52 -0
- package/dist/general-knowledge/skills/development-workflow/SKILL.md +28 -0
- package/dist/general-knowledge/skills/fleet-config/SKILL.md +1 -0
- package/dist/general-knowledge/skills/fleet-health/SKILL.md +1 -0
- package/dist/general-knowledge/skills/fleet-restart/SKILL.md +1 -0
- package/dist/general-knowledge/skills/instance-lifecycle/SKILL.md +1 -0
- package/dist/general-knowledge/skills/model-discovery/SKILL.md +64 -16
- package/dist/general-knowledge/skills/multi-channel/SKILL.md +1 -0
- package/dist/general-knowledge/skills/scheduling/SKILL.md +1 -0
- package/dist/general-knowledge/skills/session-management/SKILL.md +172 -8
- package/dist/general-knowledge/skills/tui-effort/SKILL.md +1 -0
- package/dist/general-knowledge/skills/worker-collaboration/SKILL.md +30 -0
- package/dist/instance-lifecycle.d.ts +99 -0
- package/dist/instance-lifecycle.js +339 -11
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/instructions.d.ts +12 -0
- package/dist/instructions.js +35 -1
- package/dist/instructions.js.map +1 -1
- package/dist/locale.js +1004 -288
- package/dist/locale.js.map +1 -1
- package/dist/login-flows.d.ts +91 -0
- package/dist/login-flows.js +170 -0
- package/dist/login-flows.js.map +1 -0
- package/dist/login-manager.d.ts +63 -0
- package/dist/login-manager.js +134 -0
- package/dist/login-manager.js.map +1 -0
- package/dist/network-family.d.ts +18 -0
- package/dist/network-family.js +20 -0
- package/dist/network-family.js.map +1 -0
- package/dist/outbound-handlers.d.ts +8 -0
- package/dist/outbound-handlers.js +101 -18
- package/dist/outbound-handlers.js.map +1 -1
- package/dist/outbound-schemas.d.ts +7 -1
- package/dist/outbound-schemas.js +9 -0
- package/dist/outbound-schemas.js.map +1 -1
- package/dist/pane-input-residue.d.ts +52 -0
- package/dist/pane-input-residue.js +107 -0
- package/dist/pane-input-residue.js.map +1 -0
- package/dist/reply-dedup.d.ts +7 -8
- package/dist/reply-dedup.js +0 -0
- package/dist/reply-dedup.js.map +1 -1
- package/dist/restart-progress.d.ts +2 -0
- package/dist/restart-progress.js +3 -0
- package/dist/restart-progress.js.map +1 -1
- package/dist/scheduler/db.d.ts +12 -0
- package/dist/scheduler/db.js +59 -0
- package/dist/scheduler/db.js.map +1 -1
- package/dist/service-installer.d.ts +11 -0
- package/dist/service-installer.js +84 -18
- package/dist/service-installer.js.map +1 -1
- package/dist/settings-api.js +1 -1
- package/dist/settings-api.js.map +1 -1
- package/dist/setup-wizard.js +2 -2
- package/dist/setup-wizard.js.map +1 -1
- package/dist/spawn-gate.d.ts +30 -0
- package/dist/spawn-gate.js +79 -0
- package/dist/spawn-gate.js.map +1 -0
- package/dist/storm-window.d.ts +83 -0
- package/dist/storm-window.js +251 -0
- package/dist/storm-window.js.map +1 -0
- package/dist/tips.d.ts +44 -0
- package/dist/tips.js +355 -0
- package/dist/tips.js.map +1 -0
- package/dist/tmux-manager.d.ts +16 -0
- package/dist/tmux-manager.js +110 -25
- package/dist/tmux-manager.js.map +1 -1
- package/dist/tool-progress.d.ts +40 -0
- package/dist/tool-progress.js +289 -0
- package/dist/tool-progress.js.map +1 -0
- package/dist/topic-commands.d.ts +42 -4
- package/dist/topic-commands.js +372 -67
- package/dist/topic-commands.js.map +1 -1
- package/dist/transcript-monitor.d.ts +15 -2
- package/dist/transcript-monitor.js +63 -17
- package/dist/transcript-monitor.js.map +1 -1
- package/dist/transcript-sources.d.ts +131 -0
- package/dist/transcript-sources.js +580 -0
- package/dist/transcript-sources.js.map +1 -0
- package/dist/types.d.ts +13 -0
- package/dist/ui/dashboard.html +55 -32
- package/dist/ui/settings.html +200 -60
- package/dist/ui/view.html +147 -30
- package/dist/usage/format-rich.d.ts +1 -1
- package/dist/usage/format-rich.js +28 -24
- package/dist/usage/format-rich.js.map +1 -1
- package/dist/usage/i18n-keys.d.ts +7 -0
- package/dist/usage/i18n-keys.js +34 -0
- package/dist/usage/i18n-keys.js.map +1 -0
- package/dist/usage/i18n.d.ts +5 -0
- package/dist/usage/i18n.js +27 -0
- package/dist/usage/i18n.js.map +1 -0
- package/dist/usage/providers.d.ts +21 -0
- package/dist/usage/providers.js +153 -75
- package/dist/usage/providers.js.map +1 -1
- package/dist/usage/usage-api.d.ts +11 -3
- package/dist/usage/usage-api.js +61 -24
- package/dist/usage/usage-api.js.map +1 -1
- package/dist/workflow-templates/default.md +2 -1
- package/package.json +2 -2
|
@@ -4,7 +4,7 @@ import { join, basename, dirname, resolve, sep as pathSep } from "node:path";
|
|
|
4
4
|
import { access, unlink } from "node:fs/promises";
|
|
5
5
|
import { getAgendHome, ensureWorkspaceGit } from "./paths.js";
|
|
6
6
|
import { DEFAULT_INSTANCE_CONFIG } from "./config.js";
|
|
7
|
-
import { sanitizeInstanceName } from "./topic-commands.js";
|
|
7
|
+
import { readStatuslineModel, sanitizeInstanceName } from "./topic-commands.js";
|
|
8
8
|
import { isModelCompatible } from "./backend/types.js";
|
|
9
9
|
import { safeHandler } from "./safe-async.js";
|
|
10
10
|
import { t } from "./locale.js";
|
|
@@ -12,13 +12,17 @@ import { clearPausedMarker, hasPausedMarker, readPausedAt, writePausedMarker } f
|
|
|
12
12
|
import { reportProviderRateLimit } from "./usage/provider-alerts.js";
|
|
13
13
|
import { isFleetStartCommandLine } from "./fleet-lock.js";
|
|
14
14
|
import { GENERAL_PAUSE_ERROR, isGeneralInstance } from "./general-instance.js";
|
|
15
|
+
import { BACKEND_AUTH_CHECKS, checkAuthStatus, LOGIN_FLOWS } from "./login-flows.js";
|
|
16
|
+
import { fetchClaudeUsage, fetchCodexUsage } from "./usage/providers.js";
|
|
15
17
|
export { isFleetStartCommandLine } from "./fleet-lock.js";
|
|
16
18
|
/** Shared CLI metadata used by startup validation and ClassicBot onboarding. */
|
|
17
19
|
export const BACKEND_INSTALLATION_INFO = {
|
|
18
20
|
"claude-code": { binary: "claude", install: "curl -fsSL https://claude.ai/install.sh | bash" },
|
|
19
21
|
"gemini-cli": { binary: "gemini", install: "npm i -g @google/gemini-cli" },
|
|
20
|
-
|
|
21
|
-
|
|
22
|
+
// The curl installer serves Linux and macOS (verified: 200, text/x-shellscript);
|
|
23
|
+
// the previous `brew install --cask` form only worked on macOS.
|
|
24
|
+
"kiro-cli": { binary: "kiro-cli", install: "curl -fsSL https://cli.kiro.dev/install | bash" },
|
|
25
|
+
codex: { binary: "codex", install: "curl -fsSL https://chatgpt.com/codex/install.sh | sh" },
|
|
22
26
|
opencode: { binary: "opencode", install: "curl -fsSL https://opencode.ai/install | bash" },
|
|
23
27
|
antigravity: { binary: "agy", install: "curl -fsSL https://antigravity.google/cli/install.sh | bash" },
|
|
24
28
|
grok: { binary: "grok", install: "curl -fsSL https://x.ai/cli/install.sh | bash" },
|
|
@@ -77,12 +81,116 @@ export function getUnsafeInstanceDaemonPidReason(pid, dataDir) {
|
|
|
77
81
|
}
|
|
78
82
|
/** Suppress duplicate auth alerts for the same backend within this window. */
|
|
79
83
|
const AUTH_ALERT_COOLDOWN_MS = 5 * 60_000;
|
|
84
|
+
/**
|
|
85
|
+
* Reuse one auth verification per backend for this long. Shared credentials
|
|
86
|
+
* expiring makes EVERY instance of the backend emit auth_error at once — one
|
|
87
|
+
* token-free check answers for all of them.
|
|
88
|
+
*/
|
|
89
|
+
const AUTH_VERIFY_CACHE_MS = 60_000;
|
|
90
|
+
const CODEX_QUOTA_VERIFY_TIMEOUT_MS = 5_000;
|
|
91
|
+
const CLAUDE_QUOTA_VERIFY_TIMEOUT_MS = 5_000;
|
|
92
|
+
/**
|
|
93
|
+
* Convert the live Codex usage row into a conservative quota verdict. A
|
|
94
|
+
* successful response with at least one window below 100% proves that stale
|
|
95
|
+
* terminal text is no longer current only when no other window is exhausted.
|
|
96
|
+
* Missing credentials, API errors, and metric-less responses prove nothing.
|
|
97
|
+
*/
|
|
98
|
+
export function codexQuotaVerdictFromUsage(usage) {
|
|
99
|
+
if (usage.status !== "ok")
|
|
100
|
+
return "unknown";
|
|
101
|
+
const windows = usage.metrics.filter(metric => metric.type === "percent"
|
|
102
|
+
&& typeof metric.used === "number"
|
|
103
|
+
&& Number.isFinite(metric.used)
|
|
104
|
+
&& metric.windowMs != null);
|
|
105
|
+
if (windows.length === 0)
|
|
106
|
+
return "unknown";
|
|
107
|
+
return windows.some(metric => (metric.used ?? 0) >= 100) ? "exhausted" : "available";
|
|
108
|
+
}
|
|
109
|
+
/** Run only the Codex usage provider, bounded independently of its network timeout. */
|
|
110
|
+
export async function verifyCodexQuotaStatus(fetchUsage = fetchCodexUsage, timeoutMs = CODEX_QUOTA_VERIFY_TIMEOUT_MS) {
|
|
111
|
+
let timer;
|
|
112
|
+
try {
|
|
113
|
+
const usage = await Promise.race([
|
|
114
|
+
fetchUsage(),
|
|
115
|
+
new Promise((_, reject) => {
|
|
116
|
+
timer = setTimeout(() => reject(new Error("Codex quota verification timed out")), timeoutMs);
|
|
117
|
+
timer.unref?.();
|
|
118
|
+
}),
|
|
119
|
+
]);
|
|
120
|
+
return codexQuotaVerdictFromUsage(usage);
|
|
121
|
+
}
|
|
122
|
+
catch {
|
|
123
|
+
return "unknown";
|
|
124
|
+
}
|
|
125
|
+
finally {
|
|
126
|
+
if (timer)
|
|
127
|
+
clearTimeout(timer);
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
/**
|
|
131
|
+
* Convert a LIVE Claude usage API row into a conservative verdict. The
|
|
132
|
+
* provider can degrade to statusline data when the network/API is unavailable;
|
|
133
|
+
* that fallback is intentionally unknown because stale local limits cannot
|
|
134
|
+
* disprove a current API credit-balance error.
|
|
135
|
+
*/
|
|
136
|
+
export function claudeQuotaVerdictFromUsage(usage) {
|
|
137
|
+
if (usage.status !== "ok" || /statusline/i.test(usage.hint ?? ""))
|
|
138
|
+
return "unknown";
|
|
139
|
+
const bounded = usage.metrics.filter(metric => {
|
|
140
|
+
if (typeof metric.used !== "number" || !Number.isFinite(metric.used))
|
|
141
|
+
return false;
|
|
142
|
+
if (metric.type === "percent")
|
|
143
|
+
return true;
|
|
144
|
+
return metric.type === "dollars"
|
|
145
|
+
&& typeof metric.limit === "number"
|
|
146
|
+
&& Number.isFinite(metric.limit)
|
|
147
|
+
&& metric.limit > 0;
|
|
148
|
+
});
|
|
149
|
+
if (bounded.length === 0)
|
|
150
|
+
return "unknown";
|
|
151
|
+
const exhausted = bounded.some(metric => metric.type === "percent"
|
|
152
|
+
? (metric.used ?? 0) >= 100
|
|
153
|
+
: (metric.used ?? 0) >= (metric.limit ?? Number.POSITIVE_INFINITY));
|
|
154
|
+
return exhausted ? "exhausted" : "available";
|
|
155
|
+
}
|
|
156
|
+
/** Run only the Claude usage provider, bounded independently of its network timeout. */
|
|
157
|
+
export async function verifyClaudeQuotaStatus(fetchUsage = fetchClaudeUsage, timeoutMs = CLAUDE_QUOTA_VERIFY_TIMEOUT_MS) {
|
|
158
|
+
let timer;
|
|
159
|
+
try {
|
|
160
|
+
const usage = await Promise.race([
|
|
161
|
+
fetchUsage(),
|
|
162
|
+
new Promise((_, reject) => {
|
|
163
|
+
timer = setTimeout(() => reject(new Error("Claude quota verification timed out")), timeoutMs);
|
|
164
|
+
timer.unref?.();
|
|
165
|
+
}),
|
|
166
|
+
]);
|
|
167
|
+
return claudeQuotaVerdictFromUsage(usage);
|
|
168
|
+
}
|
|
169
|
+
catch {
|
|
170
|
+
return "unknown";
|
|
171
|
+
}
|
|
172
|
+
finally {
|
|
173
|
+
if (timer)
|
|
174
|
+
clearTimeout(timer);
|
|
175
|
+
}
|
|
176
|
+
}
|
|
80
177
|
export class InstanceLifecycle {
|
|
81
178
|
ctx;
|
|
82
179
|
/** Active daemon processes: instanceName → Daemon */
|
|
83
180
|
daemons = new Map();
|
|
84
181
|
/** backend → last auth-error alert time, so one expiry sends one alert. */
|
|
85
182
|
lastAuthAlertAt = new Map();
|
|
183
|
+
/**
|
|
184
|
+
* backend → cached token-free auth verification (see AUTH_VERIFY_CACHE_MS).
|
|
185
|
+
* The PROMISE is cached, not the result: a shared credential expiring makes
|
|
186
|
+
* every instance of the backend fire in the same tick, and they must join
|
|
187
|
+
* one in-flight check instead of each spawning their own.
|
|
188
|
+
*/
|
|
189
|
+
authVerifyCache = new Map();
|
|
190
|
+
/** Same-tick Codex quota alerts join one live usage probe. */
|
|
191
|
+
codexQuotaVerifyInFlight = null;
|
|
192
|
+
/** Same-tick Claude quota alerts join one live usage probe. */
|
|
193
|
+
claudeQuotaVerifyInFlight = null;
|
|
86
194
|
/**
|
|
87
195
|
* Minimum gap between MCP-revival auto-restarts of one instance. Kept here —
|
|
88
196
|
* not in the daemon — because each restart replaces the daemon object, which
|
|
@@ -113,6 +221,10 @@ export class InstanceLifecycle {
|
|
|
113
221
|
this.ctx.logger.info({ name, kind }, "Incident notification suppressed — planned restart in progress");
|
|
114
222
|
return;
|
|
115
223
|
}
|
|
224
|
+
if (this.ctx.stormSuppressed?.(kind)) {
|
|
225
|
+
this.ctx.logger.info({ name, kind }, "Incident notification suppressed — included in tmux storm summary");
|
|
226
|
+
return;
|
|
227
|
+
}
|
|
116
228
|
this.ctx.notifyInstanceTopic(name, text);
|
|
117
229
|
}
|
|
118
230
|
/** Backend a running instance uses (config → fleet default). */
|
|
@@ -121,6 +233,67 @@ export class InstanceLifecycle {
|
|
|
121
233
|
?? this.ctx.fleetConfig?.defaults?.backend
|
|
122
234
|
?? "claude-code";
|
|
123
235
|
}
|
|
236
|
+
/**
|
|
237
|
+
* Confirm a pane-detected auth error with the backend's token-free status
|
|
238
|
+
* probe before pausing anything: pattern matching over terminal text also
|
|
239
|
+
* fires on an agent DISCUSSING a 401. "valid" means false positive — ignore.
|
|
240
|
+
* "invalid" and "unknown" (timeout, missing binary) both pause: for a
|
|
241
|
+
* suspected expiry, pausing too much is recoverable, delivering into a dead
|
|
242
|
+
* CLI is not. Cached per backend so simultaneous alerts run one check.
|
|
243
|
+
*/
|
|
244
|
+
verifyAuthError(name) {
|
|
245
|
+
return this.verifyBackendAuth(this.backendOf(name));
|
|
246
|
+
}
|
|
247
|
+
verifyCodexQuota() {
|
|
248
|
+
if (this.codexQuotaVerifyInFlight)
|
|
249
|
+
return this.codexQuotaVerifyInFlight;
|
|
250
|
+
const verify = this.ctx.verifyCodexQuota ?? (() => verifyCodexQuotaStatus());
|
|
251
|
+
const promise = verify().catch(() => "unknown");
|
|
252
|
+
this.codexQuotaVerifyInFlight = promise;
|
|
253
|
+
void promise.finally(() => {
|
|
254
|
+
if (this.codexQuotaVerifyInFlight === promise)
|
|
255
|
+
this.codexQuotaVerifyInFlight = null;
|
|
256
|
+
});
|
|
257
|
+
return promise;
|
|
258
|
+
}
|
|
259
|
+
verifyClaudeQuota() {
|
|
260
|
+
if (this.claudeQuotaVerifyInFlight)
|
|
261
|
+
return this.claudeQuotaVerifyInFlight;
|
|
262
|
+
const verify = this.ctx.verifyClaudeQuota ?? (() => verifyClaudeQuotaStatus());
|
|
263
|
+
const promise = verify().catch(() => "unknown");
|
|
264
|
+
this.claudeQuotaVerifyInFlight = promise;
|
|
265
|
+
void promise.finally(() => {
|
|
266
|
+
if (this.claudeQuotaVerifyInFlight === promise)
|
|
267
|
+
this.claudeQuotaVerifyInFlight = null;
|
|
268
|
+
});
|
|
269
|
+
return promise;
|
|
270
|
+
}
|
|
271
|
+
/** Same verification keyed by backend (used by startup pre-flight priming). */
|
|
272
|
+
verifyBackendAuth(backend) {
|
|
273
|
+
const cached = this.authVerifyCache.get(backend);
|
|
274
|
+
if (cached && Date.now() - cached.at < AUTH_VERIFY_CACHE_MS)
|
|
275
|
+
return cached.promise;
|
|
276
|
+
const check = LOGIN_FLOWS[backend]?.authCheck ?? BACKEND_AUTH_CHECKS[backend];
|
|
277
|
+
const promise = check ? checkAuthStatus(check) : Promise.resolve("unknown");
|
|
278
|
+
this.authVerifyCache.set(backend, { at: Date.now(), promise });
|
|
279
|
+
return promise;
|
|
280
|
+
}
|
|
281
|
+
/**
|
|
282
|
+
* Startup pre-flight: warm the per-backend auth verification cache so the
|
|
283
|
+
* detectors that fire seconds later (login-screen scan, MCP-died gate) get an
|
|
284
|
+
* instant answer. Advisory only — a pre-flight result alone never pauses
|
|
285
|
+
* anything, because e.g. codex on a custom provider runs fine while
|
|
286
|
+
* `codex login status` reports logged out (live-verified on this fleet).
|
|
287
|
+
*/
|
|
288
|
+
primeAuthVerification(backends) {
|
|
289
|
+
for (const backend of new Set(backends)) {
|
|
290
|
+
void this.verifyBackendAuth(backend).then(result => {
|
|
291
|
+
if (result === "invalid") {
|
|
292
|
+
this.ctx.logger.warn({ backend }, "Pre-flight auth check failed — marking backend as auth-suspect");
|
|
293
|
+
}
|
|
294
|
+
}).catch(() => { });
|
|
295
|
+
}
|
|
296
|
+
}
|
|
124
297
|
/**
|
|
125
298
|
* One alert per backend per cooldown, naming every affected instance — a CLI's
|
|
126
299
|
* credentials are shared, so N instances failing is ONE problem with ONE fix
|
|
@@ -141,7 +314,30 @@ export class InstanceLifecycle {
|
|
|
141
314
|
const scope = others.length
|
|
142
315
|
? `${affected.length} instances on \`${backend}\`: ${affected.join(", ")}`
|
|
143
316
|
: `\`${name}\` (${backend})`;
|
|
144
|
-
|
|
317
|
+
const remedy = backend === "opencode"
|
|
318
|
+
? "Run `opencode auth login` in a terminal, then wake the affected instance(s)."
|
|
319
|
+
: `Use \`/login ${backend}\` to re-login remotely.`;
|
|
320
|
+
this.notifyIncident(notificationTarget, "auth_error", `🔑 ${message}\n\nAffects ${scope}. Credentials are shared per backend — one re-login restores all of them; affected instances pause until then. ${remedy}`);
|
|
321
|
+
}
|
|
322
|
+
/**
|
|
323
|
+
* A fleet-wide backend outage (see backend-outage.ts): record the sighting so
|
|
324
|
+
* startup stops spending `--resume` attempts on it, and notify ONCE per
|
|
325
|
+
* outage at fleet level — kiro prints the line every 10s on every instance,
|
|
326
|
+
* so per-instance incidents would be N × (one message per cooldown).
|
|
327
|
+
*/
|
|
328
|
+
noteBackendOutage(name, message, notificationTarget) {
|
|
329
|
+
const backend = this.backendOf(name);
|
|
330
|
+
const result = this.ctx.backendOutage?.record(backend, name, message);
|
|
331
|
+
this.ctx.eventLog?.insert(name, "backend_unreachable", { backend, message });
|
|
332
|
+
if (result && !result.isNew) {
|
|
333
|
+
this.ctx.logger.info({ name, backend }, "backend outage sighting (already alerted)");
|
|
334
|
+
return;
|
|
335
|
+
}
|
|
336
|
+
const text = t("inst.backend_unreachable", backend, message);
|
|
337
|
+
if (this.ctx.notifyFleetError)
|
|
338
|
+
this.ctx.notifyFleetError(text);
|
|
339
|
+
else if (notificationTarget)
|
|
340
|
+
this.notifyIncident(notificationTarget, "pty_error", text);
|
|
145
341
|
}
|
|
146
342
|
/**
|
|
147
343
|
* System errors from a ClassicBot belong in the operator's General topic,
|
|
@@ -179,6 +375,10 @@ export class InstanceLifecycle {
|
|
|
179
375
|
this.ctx.logger.info({ name }, "Hang notification suppressed — planned restart in progress");
|
|
180
376
|
return;
|
|
181
377
|
}
|
|
378
|
+
if (this.ctx.stormSuppressed?.("hang")) {
|
|
379
|
+
this.ctx.logger.info({ name }, "Hang notification suppressed — included in tmux storm summary");
|
|
380
|
+
return;
|
|
381
|
+
}
|
|
182
382
|
// Check if instance has claimed tasks — nudge it to continue
|
|
183
383
|
const claimedTasks = this.ctx.listClaimedTasks(name);
|
|
184
384
|
if (claimedTasks.length > 0) {
|
|
@@ -207,6 +407,24 @@ export class InstanceLifecycle {
|
|
|
207
407
|
this.notifyIncident(generalName, "crash_respawn", t("inst.crashed_respawned_log", name));
|
|
208
408
|
}
|
|
209
409
|
}, this.ctx.logger, `daemon.crash_respawn[${name}]`));
|
|
410
|
+
daemon.on("startup_backend_unreachable", safeHandler(async (data) => {
|
|
411
|
+
// The daemon's crash-respawn hit the backend outage and paused itself.
|
|
412
|
+
// Stop it cleanly (session kept) and let the fleet's delayed retry bring
|
|
413
|
+
// it back once the backend answers — instead of `crashed` forever.
|
|
414
|
+
this.ctx.eventLog?.insert(name, "startup_backend_unreachable", { backend: data.backend });
|
|
415
|
+
this.ctx.logger.warn({ name, backend: data.backend }, "Respawn blocked by backend outage — stopping the daemon for a delayed startup retry");
|
|
416
|
+
if (this.ctx.handOffToStartupRetry) {
|
|
417
|
+
await this.ctx.handOffToStartupRetry(name, daemon);
|
|
418
|
+
return;
|
|
419
|
+
}
|
|
420
|
+
// No fleet coordinator: stop only if this daemon is still the registered one.
|
|
421
|
+
if (await this.stopIfCurrent(name, daemon))
|
|
422
|
+
this.ctx.scheduleStartupRetry?.(name, 0);
|
|
423
|
+
}, this.ctx.logger, `daemon.startup_backend_unreachable[${name}]`));
|
|
424
|
+
daemon.on("tmux_server_crash", safeHandler(() => {
|
|
425
|
+
this.ctx.eventLog?.insert(name, "tmux_server_crash", {});
|
|
426
|
+
this.ctx.logger.error({ name }, "tmux server crash joined fleet storm window");
|
|
427
|
+
}, this.ctx.logger, `daemon.tmux_server_crash[${name}]`));
|
|
210
428
|
daemon.on("snapshot_failed", safeHandler(() => {
|
|
211
429
|
this.ctx.eventLog?.insert(name, "snapshot_failed", {});
|
|
212
430
|
this.notifyIncident(name, "snapshot_failed", t("inst.restarted_no_context", name));
|
|
@@ -242,10 +460,26 @@ export class InstanceLifecycle {
|
|
|
242
460
|
this.notifyIncident(name, "crash_loop", t("inst.respawn_paused", name));
|
|
243
461
|
this.ctx.setTopicIcon(name, "red");
|
|
244
462
|
}, this.ctx.logger, `daemon.crash_loop[${name}]`));
|
|
245
|
-
daemon.on("mcp_died", safeHandler((data) => {
|
|
463
|
+
daemon.on("mcp_died", safeHandler(async (data) => {
|
|
464
|
+
const stormAtDetection = this.ctx.stormWindow?.isActive() === true;
|
|
246
465
|
this.ctx.eventLog?.insert(name, "mcp_died", { pid: data.pid });
|
|
247
466
|
this.ctx.logger.error({ name, pid: data.pid }, "MCP server died — instance cannot use agend tools");
|
|
248
467
|
this.ctx.webhookEmit("mcp_died", name, { pid: data.pid });
|
|
468
|
+
// Auth outranks MCP: when the CLI lost authentication, "MCP server died"
|
|
469
|
+
// is a symptom and the fix is /login, not a restart. The daemon flags a
|
|
470
|
+
// suspicion it already confirmed (login screen / 401 pattern); otherwise
|
|
471
|
+
// ask the cached token-free probe. Only a CONFIRMED invalid swaps the
|
|
472
|
+
// message — valid or uncertain keeps the accurate MCP report below.
|
|
473
|
+
const verdict = data.authSuspected ? "invalid" : await this.verifyAuthError(name);
|
|
474
|
+
if (verdict === "invalid") {
|
|
475
|
+
this.notifyAuthErrorOnce(name, "Sign-in expired — the CLI cannot run its MCP server (agend tools are down) until it is re-authenticated.", this.ptyErrorNotificationTarget(name) ?? name);
|
|
476
|
+
return;
|
|
477
|
+
}
|
|
478
|
+
if (stormAtDetection) {
|
|
479
|
+
this.ctx.stormSuppressed?.("mcp_died");
|
|
480
|
+
this.ctx.logger.info({ name, pid: data.pid }, "MCP death notification suppressed — included in tmux storm summary");
|
|
481
|
+
return;
|
|
482
|
+
}
|
|
249
483
|
// The CLI owns the MCP server's stdio pipes, so only restarting the CLI can
|
|
250
484
|
// restore its tools. With mcp_auto_restart (default) the daemon requests an
|
|
251
485
|
// idle-gated restart itself — an immediate one would interrupt whatever the
|
|
@@ -261,7 +495,23 @@ export class InstanceLifecycle {
|
|
|
261
495
|
this.ctx.eventLog?.insert(name, "mcp_proxy_reply", { correlationId: data.correlationId });
|
|
262
496
|
this.ctx.logger.warn({ name, correlationId: data.correlationId }, "MCP dead at turn end with no reply — daemon relayed the pane text to the channel");
|
|
263
497
|
}, this.ctx.logger, `daemon.mcp_proxy_reply[${name}]`));
|
|
498
|
+
daemon.on("malformed_tool_call", safeHandler((data) => {
|
|
499
|
+
this.ctx.eventLog?.insert(name, "malformed_tool_call", {
|
|
500
|
+
correlationId: data.correlationId,
|
|
501
|
+
recovered: data.recovered,
|
|
502
|
+
});
|
|
503
|
+
const notificationTarget = this.ptyErrorNotificationTarget(name);
|
|
504
|
+
if (notificationTarget) {
|
|
505
|
+
this.notifyIncident(notificationTarget, "malformed_tool_call", t(data.recovered ? "inst.malformed_tool_call_recovered" : "inst.malformed_tool_call_unrecoverable", name));
|
|
506
|
+
}
|
|
507
|
+
}, this.ctx.logger, `daemon.malformed_tool_call[${name}]`));
|
|
264
508
|
daemon.on("mcp_restart_requested", safeHandler((data) => {
|
|
509
|
+
if (this.ctx.stormWindow?.isActive()) {
|
|
510
|
+
this.ctx.eventLog?.insert(name, "mcp_auto_restart_suppressed", { trigger: data.trigger, reason: "tmux_storm" });
|
|
511
|
+
this.ctx.stormSuppressed?.("mcp_auto_restart");
|
|
512
|
+
this.ctx.logger.info({ name, trigger: data.trigger }, "MCP revival restart suppressed — tmux storm recovery will replace MCP");
|
|
513
|
+
return;
|
|
514
|
+
}
|
|
265
515
|
// The daemon object dies with the restart it asks for, so the loop guard
|
|
266
516
|
// lives here: if the previous auto-restart was under the cooldown, the new
|
|
267
517
|
// MCP server evidently died right back (broken install, OOM pressure) and
|
|
@@ -297,7 +547,7 @@ export class InstanceLifecycle {
|
|
|
297
547
|
}
|
|
298
548
|
await this.ctx.notifyInteractivePrompt(name, data.kind);
|
|
299
549
|
}, this.ctx.logger, `daemon.interactive_prompt[${name}]`));
|
|
300
|
-
daemon.on("pty_error", safeHandler((data) => {
|
|
550
|
+
daemon.on("pty_error", safeHandler(async (data) => {
|
|
301
551
|
this.ctx.eventLog?.insert(name, "pty_error", { type: data.type, action: data.action });
|
|
302
552
|
this.ctx.logger.warn({ name, errorType: data.type, action: data.action }, `PTY error: ${data.message}`);
|
|
303
553
|
// Antigravity's account-level cap is visible ONLY here: the quota summary
|
|
@@ -307,18 +557,69 @@ export class InstanceLifecycle {
|
|
|
307
557
|
if (data.type === "quota" && this.backendOf(name) === "antigravity") {
|
|
308
558
|
reportProviderRateLimit("antigravity", data.message);
|
|
309
559
|
}
|
|
560
|
+
// Codex keeps old errors in pane scrollback. After a restart, the new
|
|
561
|
+
// daemon has no occurrence baseline and can mistake yesterday's
|
|
562
|
+
// `You've hit your usage limit` for a current failure, pause, wake, then
|
|
563
|
+
// repeat forever. A live, non-LLM usage query distinguishes an
|
|
564
|
+
// available account from that stale text before any notification/pause.
|
|
565
|
+
// Only pause-class quota errors need this gate; low-quota notifications
|
|
566
|
+
// remain immediate. Unknown (timeout/auth/API failure) stays fail-closed.
|
|
567
|
+
if (data.type === "quota" && data.action === "pause" && this.backendOf(name) === "codex") {
|
|
568
|
+
const verdict = await this.verifyCodexQuota();
|
|
569
|
+
if (verdict === "available") {
|
|
570
|
+
this.ctx.logger.debug({ name, backend: "codex" }, "quota pattern ignored — live usage has capacity (stale pane history)");
|
|
571
|
+
return;
|
|
572
|
+
}
|
|
573
|
+
if (verdict === "unknown") {
|
|
574
|
+
this.ctx.logger.warn({ name, backend: "codex" }, "Codex quota verification failed or timed out — pausing conservatively");
|
|
575
|
+
}
|
|
576
|
+
}
|
|
577
|
+
// Claude's credit-balance message also remains in pane scrollback. Only
|
|
578
|
+
// a fresh usage API row can prove it stale; provider fallback to a local
|
|
579
|
+
// statusline, timeout, or missing credentials remains fail-closed.
|
|
580
|
+
if (data.type === "quota" && data.action === "pause" && this.backendOf(name) === "claude-code") {
|
|
581
|
+
const verdict = await this.verifyClaudeQuota();
|
|
582
|
+
if (verdict === "available") {
|
|
583
|
+
this.ctx.logger.debug({ name, backend: "claude-code" }, "quota pattern ignored — live usage has capacity (stale pane history)");
|
|
584
|
+
return;
|
|
585
|
+
}
|
|
586
|
+
if (verdict === "unknown") {
|
|
587
|
+
this.ctx.logger.warn({ name, backend: "claude-code" }, "Claude quota verification failed, degraded, or timed out — pausing conservatively");
|
|
588
|
+
}
|
|
589
|
+
}
|
|
590
|
+
// Pattern-matched auth errors get a second opinion from the real CLI
|
|
591
|
+
// before any pause/alert: an agent quoting "401 Unauthorized" in prose
|
|
592
|
+
// must not pause the fleet. A working credential ends the incident here.
|
|
593
|
+
if (data.type === "auth_error") {
|
|
594
|
+
const verdict = await this.verifyAuthError(name);
|
|
595
|
+
if (verdict === "valid") {
|
|
596
|
+
this.ctx.logger.info({ name, backend: this.backendOf(name) }, "auth-error pattern ignored — token-free auth check passed (likely conversation text)");
|
|
597
|
+
return;
|
|
598
|
+
}
|
|
599
|
+
}
|
|
310
600
|
const emoji = data.type === "rate_limit" || data.type === "timeout" ? "⏳" : data.type === "auth_error" ? "🔑" : "⚠️";
|
|
601
|
+
const incidentMessage = data.type === "model_error" && this.backendOf(name) === "claude-code"
|
|
602
|
+
? (() => {
|
|
603
|
+
const liveModel = readStatuslineModel(this.ctx.dataDir, name);
|
|
604
|
+
return liveModel
|
|
605
|
+
? t("inst.claude_model_fallback", liveModel)
|
|
606
|
+
: t("inst.claude_model_unavailable");
|
|
607
|
+
})()
|
|
608
|
+
: data.message;
|
|
311
609
|
const notificationTarget = this.ptyErrorNotificationTarget(name);
|
|
312
610
|
// Auth failures are a property of the BACKEND's shared credentials, not of
|
|
313
611
|
// one instance: every instance on that CLI fails at once, and one re-login
|
|
314
612
|
// fixes them all. Notify once per backend (listing who's affected) instead
|
|
315
613
|
// of N near-identical alerts, and suppress repeats fleet-wide.
|
|
316
|
-
if (data.type === "
|
|
614
|
+
if (data.type === "network" && data.fleetWide) {
|
|
615
|
+
this.noteBackendOutage(name, data.message, notificationTarget);
|
|
616
|
+
}
|
|
617
|
+
else if (data.type === "auth_error") {
|
|
317
618
|
if (notificationTarget)
|
|
318
619
|
this.notifyAuthErrorOnce(name, data.message, notificationTarget);
|
|
319
620
|
}
|
|
320
621
|
else if (notificationTarget) {
|
|
321
|
-
this.notifyIncident(notificationTarget, "pty_error", t("inst.notification", emoji, name,
|
|
622
|
+
this.notifyIncident(notificationTarget, "pty_error", t("inst.notification", emoji, name, incidentMessage, data.action));
|
|
322
623
|
}
|
|
323
624
|
this.ctx.webhookEmit("pty_error", name, { type: data.type, action: data.action, message: data.message });
|
|
324
625
|
// The CLI interrupted itself on this error, so any pending Cancel button is
|
|
@@ -386,12 +687,12 @@ export class InstanceLifecycle {
|
|
|
386
687
|
kind: "fleet-topic",
|
|
387
688
|
backend: backendName,
|
|
388
689
|
model: config.model ?? "default",
|
|
389
|
-
});
|
|
690
|
+
}, this.ctx.spawnGate, this.ctx.stormWindow, this.ctx.backendOutage);
|
|
390
691
|
// Catch errors from daemon internals (e.g. IPC server) to prevent crashing the fleet process
|
|
391
692
|
daemon.on("error", (err) => {
|
|
392
693
|
this.ctx.logger.error({ err, name }, "Daemon emitted error — instance isolated");
|
|
393
694
|
});
|
|
394
|
-
await daemon.
|
|
695
|
+
await InstanceLifecycle.startOrDispose(daemon, name, this.ctx.logger);
|
|
395
696
|
this.daemons.set(name, daemon);
|
|
396
697
|
daemon.on("auto_pause_requested", safeHandler(async () => {
|
|
397
698
|
await this.pause(name);
|
|
@@ -485,12 +786,38 @@ export class InstanceLifecycle {
|
|
|
485
786
|
}
|
|
486
787
|
this.ctx.startStatuslineWatcher(name);
|
|
487
788
|
}
|
|
789
|
+
/**
|
|
790
|
+
* Ownership boundary for a daemon that is not yet registered: if start()
|
|
791
|
+
* rejects, nothing else will ever dispose it, so do it here before
|
|
792
|
+
* rethrowing. The abort is best-effort — the start error is the one to
|
|
793
|
+
* surface.
|
|
794
|
+
*/
|
|
795
|
+
static async startOrDispose(daemon, name, logger) {
|
|
796
|
+
try {
|
|
797
|
+
await daemon.start();
|
|
798
|
+
}
|
|
799
|
+
catch (err) {
|
|
800
|
+
await daemon.abortStartup().catch(abortErr => logger.warn({ err: abortErr, name }, "Failed to dispose a daemon whose start() rejected"));
|
|
801
|
+
throw err;
|
|
802
|
+
}
|
|
803
|
+
}
|
|
804
|
+
/** Stop the registered daemon only if it is still `daemon` (a concurrent restart may have replaced it). */
|
|
805
|
+
async stopIfCurrent(name, daemon) {
|
|
806
|
+
if (this.daemons.get(name) !== daemon)
|
|
807
|
+
return false;
|
|
808
|
+
await this.stop(name);
|
|
809
|
+
return true;
|
|
810
|
+
}
|
|
488
811
|
async stop(name) {
|
|
489
812
|
this.ctx.setTopicIcon(name, "remove");
|
|
490
813
|
const daemon = this.daemons.get(name);
|
|
491
814
|
if (daemon) {
|
|
492
815
|
await daemon.stop();
|
|
493
|
-
|
|
816
|
+
// Identity-safe: while we awaited, a concurrent restart may have
|
|
817
|
+
// registered a FRESH daemon under this name — deleting by name alone
|
|
818
|
+
// would drop it from supervision while its CLI keeps running.
|
|
819
|
+
if (this.daemons.get(name) === daemon)
|
|
820
|
+
this.daemons.delete(name);
|
|
494
821
|
}
|
|
495
822
|
else {
|
|
496
823
|
const instanceDir = this.ctx.getInstanceDir(name);
|
|
@@ -802,6 +1129,7 @@ export class InstanceLifecycle {
|
|
|
802
1129
|
...(systemPrompt ? { systemPrompt } : {}),
|
|
803
1130
|
...(args.model ? { model: args.model } : {}),
|
|
804
1131
|
...(args.backend ? { backend: args.backend } : {}),
|
|
1132
|
+
...(args.backend_options ? { backend_options: args.backend_options } : {}),
|
|
805
1133
|
...(args.model_failover ? { model_failover: args.model_failover } : {}),
|
|
806
1134
|
...(args.tool_set ? { tool_set: args.tool_set } : {}),
|
|
807
1135
|
...(args.skipPermissions != null ? { skipPermissions: args.skipPermissions } : {}),
|