switchroom 0.18.14 → 0.18.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +3 -0
- package/dist/auth-broker/index.js +473 -49
- package/dist/cli/notion-write-pretool.mjs +3 -0
- package/dist/cli/switchroom.js +1200 -1067
- package/dist/host-control/main.js +56 -51
- package/dist/vault/approvals/kernel-server.js +19 -12
- package/dist/vault/broker/server.js +675 -668
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +81 -139
- package/telegram-plugin/dist/bridge/bridge.js +21 -0
- package/telegram-plugin/dist/gateway/gateway.js +531 -259
- package/telegram-plugin/dist/server.js +22 -1
- package/telegram-plugin/draft-stream.ts +78 -3
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +3 -4
- package/telegram-plugin/gateway/effort-command.ts +9 -7
- package/telegram-plugin/gateway/gateway.ts +310 -219
- package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
- package/telegram-plugin/gateway/model-command.ts +96 -18
- package/telegram-plugin/gateway/pending-session-command.ts +10 -8
- package/telegram-plugin/gateway/session-model-file.ts +38 -172
- package/telegram-plugin/litellm-local-notice.ts +189 -0
- package/telegram-plugin/model-unavailable.ts +214 -0
- package/telegram-plugin/quota-watch.ts +16 -4
- package/telegram-plugin/runtime-metrics.ts +47 -0
- package/telegram-plugin/send-gate-degraded.test.ts +9 -7
- package/telegram-plugin/send-gate.ts +34 -4
- package/telegram-plugin/session-tail.ts +14 -2
- package/telegram-plugin/stream-controller.ts +143 -20
- package/telegram-plugin/stream-reply-handler.ts +12 -2
- package/telegram-plugin/tests/bot-api.harness.ts +7 -2
- package/telegram-plugin/tests/draft-stream.test.ts +110 -1
- package/telegram-plugin/tests/effort-command.test.ts +4 -4
- package/telegram-plugin/tests/flood-windows-persistence.test.ts +2 -2
- package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
- package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
- package/telegram-plugin/tests/model-command.test.ts +84 -1
- package/telegram-plugin/tests/model-unavailable.test.ts +187 -0
- package/telegram-plugin/tests/operator-events-session-tail.test.ts +55 -0
- package/telegram-plugin/tests/quota-watch.test.ts +21 -0
- package/telegram-plugin/tests/reaction-gate-routing.test.ts +2 -2
- package/telegram-plugin/tests/runtime-metrics.test.ts +24 -0
- package/telegram-plugin/tests/session-model-file.test.ts +7 -155
- package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
- package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
- package/telegram-plugin/tests/throttle-tier.test.ts +176 -0
- package/telegram-plugin/tests/worker-activity-feed.test.ts +207 -0
- package/telegram-plugin/throttle-tier.ts +98 -1
- package/telegram-plugin/worker-activity-feed.ts +83 -8
|
@@ -320,11 +320,14 @@ import {
|
|
|
320
320
|
type ModelUnavailableDetection,
|
|
321
321
|
} from '../model-unavailable.js'
|
|
322
322
|
import {
|
|
323
|
+
build429ClassifiedMetric,
|
|
324
|
+
classify429Detail,
|
|
323
325
|
decideThrottleTier,
|
|
324
|
-
isAccountScopedThrottle,
|
|
325
326
|
throttleRetryInPlaceMaxMs,
|
|
326
327
|
} from '../throttle-tier.js'
|
|
327
328
|
import { createThrottleTierRunner } from './throttle-tier-wiring.js'
|
|
329
|
+
import { parseLitellmNoticeWindowMs } from '../litellm-local-notice.js'
|
|
330
|
+
import { createLitellmLocalNoticeRunner, decideRateLimitedSurface } from './litellm-local-notice-wiring.js'
|
|
328
331
|
import { runFleetAutoFallback, renderFallbackFailureNotice, evaluateFallbackFailureNotice, evaluateAllBlockedNotice, type FallbackFailureNoticeState, type FallbackAllBlockedNoticeState } from '../auto-fallback-fleet.js'
|
|
329
332
|
import { startRestartWatchdog } from './restart-watchdog.js'
|
|
330
333
|
import { validateStringArray } from './access-validator.js'
|
|
@@ -435,6 +438,8 @@ import { injectSlashCommand as injectSlashCommandImpl } from '../../src/agents/i
|
|
|
435
438
|
import { handleInjectCommand, type InjectDeps } from './inject-handler.js'
|
|
436
439
|
import {
|
|
437
440
|
parseModelCommand,
|
|
441
|
+
planModelCommand,
|
|
442
|
+
modelCommandReceiptLine,
|
|
438
443
|
handleModelCommand,
|
|
439
444
|
buildModelMenu,
|
|
440
445
|
handleModelMenuCallback,
|
|
@@ -460,18 +465,9 @@ import {
|
|
|
460
465
|
readSessionModelFileRaw,
|
|
461
466
|
restoreSessionModelFileRaw,
|
|
462
467
|
clearSessionModelFile,
|
|
463
|
-
clearSessionModelBootAttempts,
|
|
464
468
|
readConfiguredDefaultModel,
|
|
465
|
-
writeRelaunchModelIntent,
|
|
466
|
-
clearRelaunchModelIntent,
|
|
467
|
-
intentForRestartReason,
|
|
468
|
-
readSessionModelFile,
|
|
469
|
-
RELAUNCH_MODEL_INTENT_FILE,
|
|
470
|
-
GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX,
|
|
471
|
-
clearStaleGatewayShutdownIntent,
|
|
472
469
|
writeSessionEffortFile,
|
|
473
470
|
clearSessionEffortFile,
|
|
474
|
-
readSessionEffortFile,
|
|
475
471
|
} from './session-model-file.js'
|
|
476
472
|
import { discoverModels, selectModel } from '../../src/agents/model-picker.js'
|
|
477
473
|
import { resolveMainModel, SWITCHROOM_DEFAULT_THINKING_EFFORT } from '../../src/agents/scaffold.js'
|
|
@@ -1096,16 +1092,10 @@ function triggerSelfRestart(
|
|
|
1096
1092
|
)
|
|
1097
1093
|
return false
|
|
1098
1094
|
}
|
|
1099
|
-
// Session-model
|
|
1100
|
-
//
|
|
1101
|
-
//
|
|
1102
|
-
//
|
|
1103
|
-
// per-reason table classifies recovery/model-switch bounces as "keep"
|
|
1104
|
-
// and the deliberate inline restart button as "revert".
|
|
1105
|
-
{
|
|
1106
|
-
const smDir = resolveAgentDirFromEnv()
|
|
1107
|
-
if (smDir) writeRelaunchModelIntent(smDir, intentForRestartReason(reason), reason)
|
|
1108
|
-
}
|
|
1095
|
+
// Session-scoped /model (reference/rfcs/session-model-stickiness.md §0.1):
|
|
1096
|
+
// a `.session-model` carrier is consume-once — start.sh applies it on the
|
|
1097
|
+
// apply-relaunch and deletes it, so no boot needs a keep/revert intent. A
|
|
1098
|
+
// switchroom-managed bounce simply reverts to the configured default.
|
|
1109
1099
|
process.stderr.write(
|
|
1110
1100
|
`telegram gateway: restart-via-SIGTERM-PID1 agent=${targetAgent} reason=${reason} (docker)\n`,
|
|
1111
1101
|
)
|
|
@@ -1117,10 +1107,6 @@ function triggerSelfRestart(
|
|
|
1117
1107
|
return true
|
|
1118
1108
|
}
|
|
1119
1109
|
// Legacy systemd path.
|
|
1120
|
-
if (targetAgent === selfAgent) {
|
|
1121
|
-
const smDir = resolveAgentDirFromEnv()
|
|
1122
|
-
if (smDir) writeRelaunchModelIntent(smDir, intentForRestartReason(reason), reason)
|
|
1123
|
-
}
|
|
1124
1110
|
process.stderr.write(
|
|
1125
1111
|
`telegram gateway: restart-via-systemctl agent=${targetAgent} reason=${reason}\n`,
|
|
1126
1112
|
)
|
|
@@ -1140,24 +1126,6 @@ function triggerSelfRestart(
|
|
|
1140
1126
|
}
|
|
1141
1127
|
}
|
|
1142
1128
|
|
|
1143
|
-
// #3018 finding 4: a gateway-only bounce (supervisor relaunch, bare gateway
|
|
1144
|
-
// unit restart) leaves the shutdown handler's deploy-survival keep-intent
|
|
1145
|
-
// stamp on disk UNCONSUMED — start.sh only runs on a container-level boot.
|
|
1146
|
-
// If this gateway boot still sees a gateway-shutdown-stamped intent, the
|
|
1147
|
-
// preceding bounce was gateway-only: clear it so a genuine crash inside the
|
|
1148
|
-
// 10-min freshness window can't be converted into a "keep" (crash-reverts
|
|
1149
|
-
// policy intact). A real container stop/deploy consumes the file in start.sh
|
|
1150
|
-
// before any gateway boots, so a legitimate deploy stamp is never touched;
|
|
1151
|
-
// triggerSelfRestart / user-slash stamps use un-prefixed reasons.
|
|
1152
|
-
{
|
|
1153
|
-
const bootSmDir = resolveAgentDirFromEnv()
|
|
1154
|
-
if (bootSmDir != null && clearStaleGatewayShutdownIntent(bootSmDir)) {
|
|
1155
|
-
process.stderr.write(
|
|
1156
|
-
'telegram gateway: cleared stale gateway-shutdown relaunch-model intent (previous bounce was gateway-only — container never restarted)\n',
|
|
1157
|
-
)
|
|
1158
|
-
}
|
|
1159
|
-
}
|
|
1160
|
-
|
|
1161
1129
|
// Cached lazily — the claude CLI binary doesn't change inside a running
|
|
1162
1130
|
// gateway process; on `switchroom update` the gateway restarts, refreshing this.
|
|
1163
1131
|
let cachedClaudeCliVersion: string | null | undefined = undefined
|
|
@@ -1409,6 +1377,11 @@ type Access = {
|
|
|
1409
1377
|
parseMode?: 'html' | 'markdownv2' | 'text'
|
|
1410
1378
|
disableLinkPreview?: boolean
|
|
1411
1379
|
coalescingGapMs?: number
|
|
1380
|
+
/** Cooldown window (ms) for the litellm-local 429 notice — the debounced
|
|
1381
|
+
* "fleet token limiter engaged" message (litellm-local-notice.ts). Default
|
|
1382
|
+
* 15 min when unset/invalid (parseLitellmNoticeWindowMs). Projected from
|
|
1383
|
+
* channels.telegram.litellm_notice.window_ms by scaffold. */
|
|
1384
|
+
litellmNoticeWindowMs?: number
|
|
1412
1385
|
/** A2: max media attachments folded into one coalesced turn. Default 10
|
|
1413
1386
|
* (a full Telegram album / forwarded burst arrives as one turn). Set 1 to
|
|
1414
1387
|
* restore single-attachment behaviour. Projected from
|
|
@@ -1563,6 +1536,7 @@ function readAccessFile(): Access {
|
|
|
1563
1536
|
parseMode: parsed.parseMode,
|
|
1564
1537
|
disableLinkPreview: parsed.disableLinkPreview,
|
|
1565
1538
|
coalescingGapMs: parsed.coalescingGapMs,
|
|
1539
|
+
litellmNoticeWindowMs: parsed.litellmNoticeWindowMs,
|
|
1566
1540
|
coalesceMaxAttachments: parsed.coalesceMaxAttachments,
|
|
1567
1541
|
interruptSafeBoundary: parsed.interruptSafeBoundary,
|
|
1568
1542
|
interruptMaxWaitMs: parsed.interruptMaxWaitMs,
|
|
@@ -7700,14 +7674,99 @@ function emitGatewayOperatorEvent(event: OperatorEvent): void {
|
|
|
7700
7674
|
// through to the existing calm rate-limited card unchanged — an
|
|
7701
7675
|
// account-scoped throttle would be the wrong action for a server-wide
|
|
7702
7676
|
// condition.
|
|
7677
|
+
//
|
|
7678
|
+
// Classification is three-way (classify429Detail, throttle-tier.ts):
|
|
7679
|
+
// only `account-scoped` may enter the throttle tier. `litellm-local` —
|
|
7680
|
+
// the LiteLLM proxy's OWN tpm/rpm/router limiter tripping before the
|
|
7681
|
+
// request reached Anthropic — takes the calm path with NO broker mark and
|
|
7682
|
+
// NO failover: the condition is proxy-local and says nothing about the
|
|
7683
|
+
// account. Every rate-limited event ALSO emits one
|
|
7684
|
+
// `rate_limit_429_classified` runtime metric (PostHog + JSONL), fired
|
|
7685
|
+
// here — before the cooldown gate — so operators can correlate
|
|
7686
|
+
// account-scoped 429s with fleet TPM even when the card is suppressed.
|
|
7703
7687
|
let throttleEscalation: ModelUnavailableDetection | null = null
|
|
7704
7688
|
let escalationFired = false
|
|
7705
|
-
|
|
7689
|
+
// True when decideRateLimitedSurface already consulted (and armed) the
|
|
7690
|
+
// shared per-kind card cooldown for this event — the gate below must not
|
|
7691
|
+
// re-consult it, or the arm from the first consult would self-suppress.
|
|
7692
|
+
let rateLimitedCooldownConsulted = false
|
|
7693
|
+
const rateLimit429Classification =
|
|
7694
|
+
kind === 'rate-limited' ? classify429Detail(event.detail) : null
|
|
7695
|
+
if (rateLimit429Classification != null && rateLimit429Classification !== 'account-scoped') {
|
|
7696
|
+
// litellm-local / generic-transient: the calm path. NO broker
|
|
7697
|
+
// mark-throttled, NO throttle-tier runner, NO failover — for a
|
|
7698
|
+
// proxy-local cap trip those would bench an account that was never
|
|
7699
|
+
// touched.
|
|
7700
|
+
emitRuntimeMetric(
|
|
7701
|
+
build429ClassifiedMetric({
|
|
7702
|
+
agent,
|
|
7703
|
+
detail: event.detail,
|
|
7704
|
+
classification: rateLimit429Classification,
|
|
7705
|
+
action: 'calm',
|
|
7706
|
+
now: Date.now(),
|
|
7707
|
+
}),
|
|
7708
|
+
)
|
|
7709
|
+
// Surface decision — extracted (decideRateLimitedSurface,
|
|
7710
|
+
// litellm-local-notice-wiring.ts) so the ordering contract with the
|
|
7711
|
+
// shared per-kind card cooldown is pinnable by tests: litellm-local
|
|
7712
|
+
// resolves BEFORE the gate (never arms `${agent}:rate-limited`, never
|
|
7713
|
+
// suppressed by a cooldown a recent 529/generic card armed);
|
|
7714
|
+
// generic-transient consults the gate exactly once HERE.
|
|
7715
|
+
const surface = decideRateLimitedSurface({
|
|
7716
|
+
classification: rateLimit429Classification,
|
|
7717
|
+
agent,
|
|
7718
|
+
shouldEmitCard: (a) => shouldEmitOperatorEvent(a, 'rate-limited'),
|
|
7719
|
+
})
|
|
7720
|
+
if (surface === 'litellm-local-notice') {
|
|
7721
|
+
process.stderr.write(
|
|
7722
|
+
`telegram gateway: 429 classified litellm-proxy-local agent=${agent} — ` +
|
|
7723
|
+
`calm path, no account attribution, no failover\n`,
|
|
7724
|
+
)
|
|
7725
|
+
// The dedicated debounced notice REPLACES the generic "🚦 Rate limited"
|
|
7726
|
+
// card for this classification only: one calm message naming the fleet
|
|
7727
|
+
// token limiter (LiteLLM tpm_limit/rpm_limit) instead of a card that
|
|
7728
|
+
// reads like an Anthropic account problem. Classification, quota-ledger,
|
|
7729
|
+
// and failover behavior are untouched (nothing fired above on this
|
|
7730
|
+
// branch). Record into history ONLY when a notice actually posted — a
|
|
7731
|
+
// suppressed low-stakes proxy throttle must not overwrite a more
|
|
7732
|
+
// important most-recent event (e.g. credentials-expired) in the
|
|
7733
|
+
// /status enrichment (operator-events-history keeps the most recent
|
|
7734
|
+
// event per agent).
|
|
7735
|
+
const outcome = litellmLocalNoticeRunner.onRateLimited('litellm-local', agent)
|
|
7736
|
+
if (outcome === 'sent') {
|
|
7737
|
+
try {
|
|
7738
|
+
recordOperatorEvent(event)
|
|
7739
|
+
} catch { /* history is best-effort */ }
|
|
7740
|
+
}
|
|
7741
|
+
return
|
|
7742
|
+
}
|
|
7743
|
+
if (surface === 'cooldown-suppressed') {
|
|
7744
|
+
process.stderr.write(
|
|
7745
|
+
`telegram gateway: operator-event suppressed (cooldown) agent=${agent} kind=${kind}\n`,
|
|
7746
|
+
)
|
|
7747
|
+
return
|
|
7748
|
+
}
|
|
7749
|
+
// 'generic-card' — the gate passed (and armed) above; fall through to
|
|
7750
|
+
// the existing calm rate-limited card without re-consulting it.
|
|
7751
|
+
rateLimitedCooldownConsulted = true
|
|
7752
|
+
}
|
|
7753
|
+
if (rateLimit429Classification === 'account-scoped') {
|
|
7706
7754
|
const throttleDecision = decideThrottleTier({
|
|
7707
7755
|
detail: event.detail,
|
|
7708
7756
|
now: Date.now(),
|
|
7709
7757
|
thresholdMs: throttleRetryInPlaceMaxMs(),
|
|
7710
7758
|
})
|
|
7759
|
+
emitRuntimeMetric(
|
|
7760
|
+
build429ClassifiedMetric({
|
|
7761
|
+
agent,
|
|
7762
|
+
detail: event.detail,
|
|
7763
|
+
classification: 'account-scoped',
|
|
7764
|
+
// decideThrottleTier can't return 'none' for account-scoped wording;
|
|
7765
|
+
// map the two live actions onto the metric's vocabulary.
|
|
7766
|
+
action: throttleDecision.action === 'failover' ? 'failover' : 'throttle',
|
|
7767
|
+
now: Date.now(),
|
|
7768
|
+
}),
|
|
7769
|
+
)
|
|
7711
7770
|
if (throttleDecision.action === 'throttle') {
|
|
7712
7771
|
// Reset is near (≤ threshold) or unparseable (60s default): DO NOT
|
|
7713
7772
|
// fail over. Record throttled_until broker-side, post ONE lightweight
|
|
@@ -7754,7 +7813,7 @@ function emitGatewayOperatorEvent(event: OperatorEvent): void {
|
|
|
7754
7813
|
}
|
|
7755
7814
|
}
|
|
7756
7815
|
|
|
7757
|
-
if (!shouldEmitOperatorEvent(agent, kind)) {
|
|
7816
|
+
if (!rateLimitedCooldownConsulted && !shouldEmitOperatorEvent(agent, kind)) {
|
|
7758
7817
|
process.stderr.write(
|
|
7759
7818
|
`telegram gateway: operator-event suppressed (cooldown) agent=${agent} kind=${kind}\n`,
|
|
7760
7819
|
)
|
|
@@ -10332,16 +10391,6 @@ const ipcServer: IpcServer = createIpcServer({
|
|
|
10332
10391
|
// this call is safe wherever it sits relative to the cron early-return
|
|
10333
10392
|
// above (#3038 review finding 5).
|
|
10334
10393
|
bridgeDeadWatchdog.noteBridgeRegistered(client.agentName)
|
|
10335
|
-
// #3043 item 2: a REAL bridge registering is proof the boot came all the
|
|
10336
|
-
// way up healthy — clear start.sh's crashloop boot-attempts counter so only
|
|
10337
|
-
// boots that genuinely fail BEFORE the bridge registers accumulate toward
|
|
10338
|
-
// the 3-strike override clear. Without this, three quick operator
|
|
10339
|
-
// hand-bounces of a healthy agent (each <150s apart) spuriously wipe a
|
|
10340
|
-
// working model override. Best-effort; no-op when the file is absent.
|
|
10341
|
-
if (client.agentName != null) {
|
|
10342
|
-
const smBootDir = resolveAgentDirFromEnv()
|
|
10343
|
-
if (smBootDir != null) clearSessionModelBootAttempts(smBootDir)
|
|
10344
|
-
}
|
|
10345
10394
|
client.send({ type: 'status', status: 'agent_connected' })
|
|
10346
10395
|
|
|
10347
10396
|
// Phase 2b PR 3a — bridgeUp cutover. The state machine's `bridgeUp`
|
|
@@ -21775,15 +21824,10 @@ function buildModelDeps(restartCtx?: ModelDepsRestartContext): ModelMenuDeps & M
|
|
|
21775
21824
|
})
|
|
21776
21825
|
}
|
|
21777
21826
|
stampUserRestartReason(reason)
|
|
21778
|
-
// Model-switch restarts are
|
|
21779
|
-
//
|
|
21780
|
-
// the
|
|
21781
|
-
// revert
|
|
21782
|
-
// deliberately writes no intent of its own.
|
|
21783
|
-
{
|
|
21784
|
-
const smDir = resolveAgentDirFromEnv()
|
|
21785
|
-
if (smDir) writeRelaunchModelIntent(smDir, 'keep', reason)
|
|
21786
|
-
}
|
|
21827
|
+
// Model-switch restarts are the APPLY-relaunch for a `.session-model`
|
|
21828
|
+
// carrier the caller wrote immediately above: start.sh consumes it on
|
|
21829
|
+
// the very next boot (this dispatch's boot), then reverts thereafter.
|
|
21830
|
+
// No keep/revert intent is needed — the carrier is consume-once.
|
|
21787
21831
|
await sweepBeforeSelfRestart()
|
|
21788
21832
|
const hostdResp = await tryHostdDispatch(name, {
|
|
21789
21833
|
v: 1,
|
|
@@ -21804,14 +21848,11 @@ function buildModelDeps(restartCtx?: ModelDepsRestartContext): ModelMenuDeps & M
|
|
|
21804
21848
|
return
|
|
21805
21849
|
}
|
|
21806
21850
|
// hostd is configured but returned an error/denied result. No restart
|
|
21807
|
-
// is coming, so the
|
|
21808
|
-
//
|
|
21851
|
+
// is coming, so the carrier the caller wrote must not linger to be
|
|
21852
|
+
// consumed by an unrelated later boot — scheduleModelRelaunch's catch
|
|
21853
|
+
// rolls it back on this throw.
|
|
21809
21854
|
if (hostdResp.result !== 'started' && hostdResp.result !== 'completed') {
|
|
21810
21855
|
clearRestartMarker()
|
|
21811
|
-
{
|
|
21812
|
-
const smDir = resolveAgentDirFromEnv()
|
|
21813
|
-
if (smDir) clearRelaunchModelIntent(smDir)
|
|
21814
|
-
}
|
|
21815
21856
|
throw new Error(
|
|
21816
21857
|
`hostd restart failed (result=${hostdResp.result}): ${hostdResp.error ?? '(no details)'}`,
|
|
21817
21858
|
)
|
|
@@ -21820,11 +21861,11 @@ function buildModelDeps(restartCtx?: ModelDepsRestartContext): ModelMenuDeps & M
|
|
|
21820
21861
|
/**
|
|
21821
21862
|
* Switch TO a model that needs a relaunch (sr-* LiteLLM/OpenRouter ids,
|
|
21822
21863
|
* which claude's native `/model` picker rejects, and the sr-to-claude
|
|
21823
|
-
* direction). Write the
|
|
21824
|
-
* applies it on
|
|
21825
|
-
*
|
|
21826
|
-
*
|
|
21827
|
-
*
|
|
21864
|
+
* direction). Write the CONSUME-ONCE `.session-model` carrier (start.sh
|
|
21865
|
+
* applies it on the very next boot — this dispatch's apply-relaunch — and
|
|
21866
|
+
* deletes it, so it reverts on any subsequent restart), set the in-memory
|
|
21867
|
+
* session-model so /status stays honest across the restart window, then
|
|
21868
|
+
* run the SAME restart dispatch as scheduleRestart above.
|
|
21828
21869
|
*/
|
|
21829
21870
|
scheduleModelRelaunch: async (model: string, reason: string) => {
|
|
21830
21871
|
const agentDir = resolveAgentDirFromEnv()
|
|
@@ -21841,20 +21882,15 @@ function buildModelDeps(restartCtx?: ModelDepsRestartContext): ModelMenuDeps & M
|
|
|
21841
21882
|
try {
|
|
21842
21883
|
await deps.scheduleRestart(reason)
|
|
21843
21884
|
} catch (err) {
|
|
21844
|
-
// A restart already in flight OWNS the
|
|
21845
|
-
// boot
|
|
21846
|
-
//
|
|
21847
|
-
//
|
|
21848
|
-
//
|
|
21849
|
-
//
|
|
21850
|
-
// mis-launch the NEXT relaunch. The keep-intent goes with them
|
|
21851
|
-
// (belt-and-braces: scheduleRestart's failure branch clears it too):
|
|
21852
|
-
// a fresh keep on disk with no restart coming would wrongly KEEP
|
|
21853
|
-
// across a crash inside its 10-min window.
|
|
21885
|
+
// A restart already in flight OWNS the carrier we just wrote — its
|
|
21886
|
+
// boot will consume+apply our token, so the switch is queued, not
|
|
21887
|
+
// lost: keep the file + override and let the caller tell the operator
|
|
21888
|
+
// "~15s". Any OTHER dispatch failure means no restart is coming, so
|
|
21889
|
+
// roll BOTH back — a lingering carrier would lie to /status and be
|
|
21890
|
+
// consumed (mis-applied) by an unrelated later boot.
|
|
21854
21891
|
if ((err as { code?: string })?.code !== 'restart_in_flight') {
|
|
21855
21892
|
restoreSessionModelFileRaw(agentDir, prevFileRaw)
|
|
21856
21893
|
sessionModelSource.setOverride(prevOverride)
|
|
21857
|
-
clearRelaunchModelIntent(agentDir)
|
|
21858
21894
|
}
|
|
21859
21895
|
throw err
|
|
21860
21896
|
}
|
|
@@ -21875,19 +21911,24 @@ function modelMenuReplyMarkup(reply: ModelMenuReply): InlineKeyboard | undefined
|
|
|
21875
21911
|
|
|
21876
21912
|
/**
|
|
21877
21913
|
* Record a POSITIVELY-CONFIRMED typed `/model` switch: set the in-memory
|
|
21878
|
-
* override so `/status` reflects the live model
|
|
21879
|
-
*
|
|
21880
|
-
*
|
|
21881
|
-
*
|
|
21914
|
+
* override so `/status` reflects the live model. Shared by the live
|
|
21915
|
+
* `bot.command('model')` handler and the deferred (queued mid-turn) apply so
|
|
21916
|
+
* both record identically. Returns a warning suffix to append to the reply
|
|
21917
|
+
* body (currently always empty — kept for a stable signature).
|
|
21882
21918
|
*
|
|
21883
|
-
*
|
|
21884
|
-
*
|
|
21885
|
-
*
|
|
21919
|
+
* Session-scoped (rev 4): a live Claude `/model` switch writes NO
|
|
21920
|
+
* `.session-model` carrier — it applies in-session and the explicit
|
|
21921
|
+
* `claude --model <configured>` flag reverts it on the next boot, so it lasts
|
|
21922
|
+
* exactly until the next restart with no durable state. (sr-* switches never
|
|
21923
|
+
* reach here — they go through scheduleModelRelaunch, which owns the
|
|
21924
|
+
* consume-once carrier.) The `/status` honesty invariant lives here: only
|
|
21925
|
+
* `reply.selectedModel` records; an unverified inject records nothing.
|
|
21926
|
+
* `/model default` clears any in-memory override and any leftover carrier.
|
|
21886
21927
|
*/
|
|
21887
21928
|
function recordTypedModelSwitch(
|
|
21888
21929
|
reply: { text: string; selectedModel?: string },
|
|
21889
21930
|
requestedModelArg: string | null,
|
|
21890
|
-
|
|
21931
|
+
_deps: ModelCommandDeps,
|
|
21891
21932
|
): string {
|
|
21892
21933
|
const requested = requestedModelArg != null ? expandSrAlias(requestedModelArg) : null
|
|
21893
21934
|
if (requested?.toLowerCase() === 'default') {
|
|
@@ -21898,28 +21939,14 @@ function recordTypedModelSwitch(
|
|
|
21898
21939
|
}
|
|
21899
21940
|
if (!reply.selectedModel) return ''
|
|
21900
21941
|
sessionModelSource.setOverride(reply.selectedModel)
|
|
21901
|
-
const smDir = resolveAgentDirFromEnv()
|
|
21902
|
-
if (smDir && requested && isValidModelArg(requested) && !isSrModel(requested)) {
|
|
21903
|
-
try {
|
|
21904
|
-
writeSessionModelFile(
|
|
21905
|
-
smDir,
|
|
21906
|
-
requested,
|
|
21907
|
-
readConfiguredDefaultModel(smDir) ??
|
|
21908
|
-
resolveMainModel(deps.getConfiguredModel() ?? undefined),
|
|
21909
|
-
)
|
|
21910
|
-
} catch (err) {
|
|
21911
|
-
process.stderr.write(
|
|
21912
|
-
`telegram gateway: session-model persist failed (typed /model): ${(err as Error)?.message ?? String(err)}\n`,
|
|
21913
|
-
)
|
|
21914
|
-
return '\n⚠️ Couldn’t persist the sticky override — the switch is live now but won’t survive a relaunch.'
|
|
21915
|
-
}
|
|
21916
|
-
}
|
|
21917
21942
|
return ''
|
|
21918
21943
|
}
|
|
21919
21944
|
|
|
21920
21945
|
/**
|
|
21921
|
-
* Record a model-MENU callback outcome (
|
|
21922
|
-
*
|
|
21946
|
+
* Record a model-MENU callback outcome (set the live in-memory override, clear
|
|
21947
|
+
* a leftover carrier on a Default tap) and drive an sr-*→Claude graceful
|
|
21948
|
+
* restart when the tap crosses that boundary — the only menu path that writes a
|
|
21949
|
+
* consume-once `.session-model` carrier (a live Claude tap writes none, rev 4).
|
|
21923
21950
|
* Extracted from the live `mdl:*` dispatcher so the deferred (queued mid-turn)
|
|
21924
21951
|
* apply records + restarts identically. Does NOT edit any Telegram message —
|
|
21925
21952
|
* callers own the card edit. Returns a restart notice when a session restart
|
|
@@ -21933,29 +21960,13 @@ function recordModelMenuSideEffects(
|
|
|
21933
21960
|
prevSessionModel: string | null,
|
|
21934
21961
|
): { restartNotice?: string } {
|
|
21935
21962
|
// Record a successful session switch so /status reflects what's actually
|
|
21936
|
-
// running
|
|
21937
|
-
//
|
|
21938
|
-
// the
|
|
21939
|
-
//
|
|
21963
|
+
// running. Session-scoped (rev 4): a live Claude menu tap writes NO
|
|
21964
|
+
// `.session-model` carrier — it applies in-session (native picker) and
|
|
21965
|
+
// reverts on the next boot. Only the sr→Claude transition below (which
|
|
21966
|
+
// relaunches) writes the consume-once carrier. A confirmed "Default
|
|
21967
|
+
// (recommended)" selection clears any leftover carrier.
|
|
21940
21968
|
if (outcome.selectedModel) {
|
|
21941
21969
|
sessionModelSource.setOverride(outcome.selectedModel)
|
|
21942
|
-
const smDir = resolveAgentDirFromEnv()
|
|
21943
|
-
if (smDir && outcome.selectedModelToken) {
|
|
21944
|
-
try {
|
|
21945
|
-
writeSessionModelFile(
|
|
21946
|
-
smDir,
|
|
21947
|
-
outcome.selectedModelToken,
|
|
21948
|
-
readConfiguredDefaultModel(smDir) ??
|
|
21949
|
-
resolveMainModel(modelDeps.getConfiguredModel() ?? undefined),
|
|
21950
|
-
)
|
|
21951
|
-
} catch (err) {
|
|
21952
|
-
outcome.reply.text +=
|
|
21953
|
-
'\n⚠️ Couldn’t persist the sticky override — the switch is live now but won’t survive a relaunch.'
|
|
21954
|
-
process.stderr.write(
|
|
21955
|
-
`telegram gateway: session-model persist failed (menu): ${(err as Error)?.message ?? String(err)}\n`,
|
|
21956
|
-
)
|
|
21957
|
-
}
|
|
21958
|
-
}
|
|
21959
21970
|
}
|
|
21960
21971
|
if (outcome.clearedDefault) {
|
|
21961
21972
|
const smDir = resolveAgentDirFromEnv()
|
|
@@ -21967,9 +21978,11 @@ function recordModelMenuSideEffects(
|
|
|
21967
21978
|
// torn down — a graceful restart (same mechanism as /restart) is required.
|
|
21968
21979
|
if (outcome.selectedModel && isSrToClaudeTransition(prevSessionModel, outcome.selectedModel)) {
|
|
21969
21980
|
const agentName = getMyAgentName()
|
|
21970
|
-
// Carry the requested Claude model across the restart via the SAME
|
|
21971
|
-
// `.session-model`
|
|
21972
|
-
//
|
|
21981
|
+
// Carry the requested Claude model across the restart via the SAME
|
|
21982
|
+
// consume-once `.session-model` carrier a Claude → sr-* switch uses —
|
|
21983
|
+
// otherwise this transition's apply-relaunch boots the CONFIGURED default
|
|
21984
|
+
// and the tapped model is silently dropped. Applied on that one boot,
|
|
21985
|
+
// then reverts on the next restart (rev 4).
|
|
21973
21986
|
const agentDir = resolveAgentDirFromEnv()
|
|
21974
21987
|
const token = outcome.selectedModelToken
|
|
21975
21988
|
if (agentDir && token) {
|
|
@@ -22132,20 +22145,19 @@ function persistQueuedCommandForRestart(action: ShutdownResolutionAction): strin
|
|
|
22132
22145
|
switch (action.persist) {
|
|
22133
22146
|
case 'model': {
|
|
22134
22147
|
// #3042 blocker 2a: this token was QUEUED, never confirmed by claude.
|
|
22135
|
-
//
|
|
22136
|
-
//
|
|
22137
|
-
//
|
|
22138
|
-
//
|
|
22139
|
-
//
|
|
22148
|
+
// The carrier is written immediately before the bounce, so the next
|
|
22149
|
+
// boot IS its apply-relaunch: consume-once means a garbage token can
|
|
22150
|
+
// crash at most one boot before it reverts, but we still gate on
|
|
22151
|
+
// offline-trustable tokens (static Claude aliases, curated sr-* alias
|
|
22152
|
+
// targets) to avoid even that one crash-boot; anything else gets the
|
|
22153
|
+
// honest "couldn't verify — re-issue" card instead.
|
|
22140
22154
|
if (!isOfflineTrustedModelToken(action.arg)) {
|
|
22141
22155
|
return `↩️ Couldn’t verify \`${escapeHtmlForTg(action.cmd.targetLabel || action.arg)}\` as a known model without the live session — it was NOT saved. Re-issue \`/model ${escapeHtmlForTg(action.arg)}\` once the agent is back.`
|
|
22142
22156
|
}
|
|
22143
22157
|
const configured =
|
|
22144
22158
|
readConfiguredDefaultModel(agentDir) ?? resolveMainModel(undefined)
|
|
22159
|
+
// Consume-once carrier: applied by the next boot, then reverts.
|
|
22145
22160
|
writeSessionModelFile(agentDir, expandSrAlias(action.arg), configured)
|
|
22146
|
-
// Boot default is keep (#3039), but stamp explicit keep-intent for
|
|
22147
|
-
// reason-honesty in the boot notice.
|
|
22148
|
-
writeRelaunchModelIntent(agentDir, 'keep', 'queued /model carried across restart')
|
|
22149
22161
|
break
|
|
22150
22162
|
}
|
|
22151
22163
|
case 'clear-model':
|
|
@@ -22235,17 +22247,50 @@ bot.command('model', async ctx => {
|
|
|
22235
22247
|
const parsed = parseModelCommand(text) ?? { kind: 'show' as const }
|
|
22236
22248
|
const chatId = String(ctx.chat!.id)
|
|
22237
22249
|
const threadId = resolveThreadId(chatId, ctx.message?.message_thread_id)
|
|
22250
|
+
// #3177 — durable receipt FIRST. A typed /model must NEVER be invisible: even
|
|
22251
|
+
// if every downstream reply is shed/dropped, or the session is in a
|
|
22252
|
+
// phantom-idle window (turn atom cleared while claude is still busy), the
|
|
22253
|
+
// command leaves a greppable log line + a history row before any branch. This
|
|
22254
|
+
// is the fix for the finn 2026-07-12 zero-trace swallow (no log, no reply, no
|
|
22255
|
+
// ack, no deferred apply). `busyNow` folds BOTH busy signals (turn atom AND
|
|
22256
|
+
// the authoritative delivery-machine/approval gate).
|
|
22257
|
+
const busyNow = currentTurn !== null || turnInFlightForGate()
|
|
22258
|
+
process.stderr.write(modelCommandReceiptLine(getMyAgentName(), parsed, busyNow) + '\n')
|
|
22259
|
+
if (HISTORY_ENABLED && ctx.message?.message_id != null) {
|
|
22260
|
+
try {
|
|
22261
|
+
recordInbound({
|
|
22262
|
+
chat_id: chatId,
|
|
22263
|
+
thread_id: threadId ?? null,
|
|
22264
|
+
message_id: ctx.message.message_id,
|
|
22265
|
+
user: ctx.from?.username ?? (ctx.from?.id != null ? String(ctx.from.id) : null),
|
|
22266
|
+
user_id: ctx.from?.id != null ? String(ctx.from.id) : null,
|
|
22267
|
+
ts: ctx.message.date ?? Math.floor(Date.now() / 1000),
|
|
22268
|
+
text,
|
|
22269
|
+
})
|
|
22270
|
+
} catch (err) {
|
|
22271
|
+
process.stderr.write(`telegram gateway: /model recordInbound failed: ${(err as Error)?.message ?? String(err)}\n`)
|
|
22272
|
+
}
|
|
22273
|
+
}
|
|
22238
22274
|
const deps = buildModelDeps({ chatId, threadId })
|
|
22239
|
-
|
|
22275
|
+
// Route on a pure disposition (#3177) that folds BOTH busy signals so a
|
|
22276
|
+
// session busy by EITHER measure ack+queues instead of silently injecting
|
|
22277
|
+
// into a busy pane. Every branch below produces a visible action.
|
|
22278
|
+
const disposition = planModelCommand(parsed, {
|
|
22279
|
+
currentTurnActive: currentTurn !== null,
|
|
22280
|
+
turnInFlight: turnInFlightForGate(),
|
|
22281
|
+
menuEnabled: process.env.SWITCHROOM_MODEL_MENU !== '0',
|
|
22282
|
+
})
|
|
22283
|
+
if (disposition.kind === 'menu') {
|
|
22240
22284
|
const menu = await buildModelMenu(deps)
|
|
22241
22285
|
await switchroomReply(ctx, menu.text, { html: true, reply_markup: modelMenuReplyMarkup(menu) })
|
|
22242
22286
|
return
|
|
22243
22287
|
}
|
|
22244
|
-
// Mid-turn: instead of dead-ending ("Try again in a
|
|
22245
|
-
//
|
|
22246
|
-
//
|
|
22247
|
-
|
|
22248
|
-
|
|
22288
|
+
// Mid-turn (by either busy signal): instead of dead-ending ("Try again in a
|
|
22289
|
+
// moment") or silently injecting into a busy pane, ACK + QUEUE + apply-on-idle
|
|
22290
|
+
// + confirm (#3017/#3177). The typed set path either injects into claude's
|
|
22291
|
+
// input box or triggers a carrier restart — both unsafe while busy.
|
|
22292
|
+
if (disposition.kind === 'queue') {
|
|
22293
|
+
const target = disposition.target
|
|
22249
22294
|
const sent = await ctx.replyWithRichMessage(
|
|
22250
22295
|
richMessage(hardenCardBreaks(pendingCmdAckText('model', target, escapeHtmlForTg))),
|
|
22251
22296
|
threadId != null ? { message_thread_id: threadId } : {},
|
|
@@ -22285,39 +22330,38 @@ bot.command('model', async ctx => {
|
|
|
22285
22330
|
// is blocklisted for `/effort` since #2471), session-scoped — boot re-pins
|
|
22286
22331
|
// the configured default via start.sh's `--effort`. Implementation in
|
|
22287
22332
|
// effort-command.ts so it's unit-testable without booting the bot.
|
|
22333
|
+
|
|
22334
|
+
// The live session-effort override, in memory only (#3186, session-scoped
|
|
22335
|
+
// like /model rev 4). A confirmed live apply records here — NOT to the
|
|
22336
|
+
// `.session-effort` carrier — so it lasts exactly until the next restart
|
|
22337
|
+
// (start.sh's explicit `--effort <configured>` reverts it for free). Seeded
|
|
22338
|
+
// at boot from `.active-session-effort` (the effort sibling of
|
|
22339
|
+
// `.active-session-model`) so a queued-carrier apply-boot still shows the
|
|
22340
|
+
// honest live level on the /effort menu.
|
|
22341
|
+
let sessionEffortOverride: string | null = null
|
|
22342
|
+
|
|
22288
22343
|
function buildEffortDeps(): EffortCommandDeps {
|
|
22289
22344
|
return {
|
|
22290
|
-
// #
|
|
22291
|
-
//
|
|
22292
|
-
//
|
|
22293
|
-
//
|
|
22294
|
-
//
|
|
22345
|
+
// Session-scoped (#3186): a positively-confirmed live apply records the
|
|
22346
|
+
// level IN MEMORY only — no durable carrier. The `.session-effort`
|
|
22347
|
+
// carrier is written solely by persistQueuedCommandForRestart (a queued
|
|
22348
|
+
// mid-turn /effort carried across the bounce) and is consume-once at
|
|
22349
|
+
// boot. `/effort default` clears via clearSessionEffort below.
|
|
22295
22350
|
applyEffort: async (agent, level) => {
|
|
22296
22351
|
const result = await applyEffort(agent, level)
|
|
22297
|
-
if (result.ok)
|
|
22298
|
-
const agentDir = resolveAgentDirFromEnv()
|
|
22299
|
-
if (agentDir) {
|
|
22300
|
-
try {
|
|
22301
|
-
writeSessionEffortFile(agentDir, level, getConfiguredEffortForPersist())
|
|
22302
|
-
} catch (err) {
|
|
22303
|
-
process.stderr.write(
|
|
22304
|
-
`telegram gateway: session-effort persist failed level=${level}: ${(err as Error)?.message ?? String(err)}\n`,
|
|
22305
|
-
)
|
|
22306
|
-
}
|
|
22307
|
-
}
|
|
22308
|
-
}
|
|
22352
|
+
if (result.ok) sessionEffortOverride = level
|
|
22309
22353
|
return result
|
|
22310
22354
|
},
|
|
22311
22355
|
getAgentName: getMyAgentName,
|
|
22312
22356
|
getConfiguredEffort: () => getConfiguredEffortForPersist(),
|
|
22313
22357
|
clearSessionEffort: () => {
|
|
22358
|
+
sessionEffortOverride = null
|
|
22359
|
+
// Also drop any leftover queued-command carrier so the next boot can't
|
|
22360
|
+
// consume a stale level the user just cleared.
|
|
22314
22361
|
const agentDir = resolveAgentDirFromEnv()
|
|
22315
22362
|
if (agentDir) clearSessionEffortFile(agentDir)
|
|
22316
22363
|
},
|
|
22317
|
-
getSessionEffort: () =>
|
|
22318
|
-
const agentDir = resolveAgentDirFromEnv()
|
|
22319
|
-
return agentDir ? (readSessionEffortFile(agentDir)?.level ?? null) : null
|
|
22320
|
-
},
|
|
22364
|
+
getSessionEffort: () => sessionEffortOverride,
|
|
22321
22365
|
escapeHtml: escapeHtmlForTg,
|
|
22322
22366
|
}
|
|
22323
22367
|
}
|
|
@@ -22462,14 +22506,10 @@ bot.command('restart', async ctx => {
|
|
|
22462
22506
|
// greeting card shows "Restarted user: /restart from chat" instead
|
|
22463
22507
|
// of whatever reason the downstream CLI would default to.
|
|
22464
22508
|
stampUserRestartReason('user: /restart from chat')
|
|
22465
|
-
//
|
|
22466
|
-
//
|
|
22467
|
-
//
|
|
22468
|
-
//
|
|
22469
|
-
{
|
|
22470
|
-
const smDir = resolveAgentDirFromEnv()
|
|
22471
|
-
if (smDir) writeRelaunchModelIntent(smDir, 'keep', 'user: /restart from chat')
|
|
22472
|
-
}
|
|
22509
|
+
// Session-scoped (rev 4): /restart reverts any live /model override to the
|
|
22510
|
+
// configured default. A consume-once `.session-model` carrier (if one was
|
|
22511
|
+
// in flight) was already consumed by its own apply-relaunch, so nothing to
|
|
22512
|
+
// do here — start.sh boots the configured default.
|
|
22473
22513
|
await sweepBeforeSelfRestart()
|
|
22474
22514
|
const hostdResp = await tryHostdDispatch(getMyAgentName(), {
|
|
22475
22515
|
v: 1,
|
|
@@ -22628,12 +22668,8 @@ async function handleNewCommand(ctx: Context): Promise<void> {
|
|
|
22628
22668
|
// Stamp user attribution so the next greeting shows "Restarted user:
|
|
22629
22669
|
// /new" / "user: /reset" rather than the downstream CLI default.
|
|
22630
22670
|
stampUserRestartReason(`user: /${kind} from chat`)
|
|
22631
|
-
// /new and /reset
|
|
22632
|
-
//
|
|
22633
|
-
// Boot default is revert, so the keep-intent must land before dispatch.
|
|
22634
|
-
if (agentDir != null) {
|
|
22635
|
-
writeRelaunchModelIntent(agentDir, 'keep', `user: /${kind} from chat`)
|
|
22636
|
-
}
|
|
22671
|
+
// Session-scoped (rev 4): /new and /reset are restarts, so they revert any
|
|
22672
|
+
// live /model override to the configured default — no carrier to preserve.
|
|
22637
22673
|
await sweepBeforeSelfRestart()
|
|
22638
22674
|
const hostdResp = await tryHostdDispatch(getMyAgentName(), {
|
|
22639
22675
|
v: 1,
|
|
@@ -23413,6 +23449,46 @@ const throttleTierRunner = createThrottleTierRunner({
|
|
|
23413
23449
|
log: (m) => process.stderr.write(`telegram gateway: ${m}\n`),
|
|
23414
23450
|
})
|
|
23415
23451
|
|
|
23452
|
+
// ─── litellm-local 429 notice — side-effect wiring ──────────────────────────
|
|
23453
|
+
// State machine + text + config parsing live in litellm-local-notice.ts
|
|
23454
|
+
// (pure); the sequencing (classification guard → per-agent cooldown →
|
|
23455
|
+
// broadcast + metric) lives in litellm-local-notice-wiring.ts so it is
|
|
23456
|
+
// unit-testable with injected deps. This block only binds the real gateway
|
|
23457
|
+
// dependencies. Deliberately NO broker surface: the litellm-local calm
|
|
23458
|
+
// path's invariant is that account state is never touched.
|
|
23459
|
+
const litellmLocalNoticeRunner = createLitellmLocalNoticeRunner({
|
|
23460
|
+
listNoticeChats: () => loadAccess().allowFrom,
|
|
23461
|
+
sendNotice: (chat_id, markdown) => {
|
|
23462
|
+
// Topic routing — this notice REPLACES the generic operator-event card
|
|
23463
|
+
// for the litellm-local classification, so it must land where that card
|
|
23464
|
+
// would have: supergroup-mode agents route system notifications into the
|
|
23465
|
+
// alerts/admin alias topic ('compact-watchdog' kind, same resolution as
|
|
23466
|
+
// the emitGatewayOperatorEvent broadcast loop), while DM recipients get
|
|
23467
|
+
// a thread-less send (topicForRecipient guards the #2096 "message
|
|
23468
|
+
// thread not found" misrouting class).
|
|
23469
|
+
const noticeTopic = resolveAgentOutboundTopic({ kind: 'compact-watchdog' })
|
|
23470
|
+
const noticeSupergroup = resolveAgentSupergroupChatId()
|
|
23471
|
+
const noticeThread = topicForRecipient({
|
|
23472
|
+
recipientChatId: chat_id,
|
|
23473
|
+
resolvedTopic: noticeTopic,
|
|
23474
|
+
supergroupChatId: noticeSupergroup,
|
|
23475
|
+
})
|
|
23476
|
+
// Status notice, not the user's answer — silence the ping (same posture
|
|
23477
|
+
// as the throttle-tier / fleet-fallback announcements).
|
|
23478
|
+
void swallowingApiCall(
|
|
23479
|
+
// allow-raw-bot-api: wrapped in swallowingApiCall (retry policy)
|
|
23480
|
+
() => bot.api.sendRichMessage(chat_id, richMessage(markdown), {
|
|
23481
|
+
disable_notification: true,
|
|
23482
|
+
...(noticeThread != null ? { message_thread_id: noticeThread } : {}),
|
|
23483
|
+
}),
|
|
23484
|
+
{ chat_id: String(chat_id), verb: 'litellm-local-notice:notify' },
|
|
23485
|
+
)
|
|
23486
|
+
},
|
|
23487
|
+
windowMs: () => parseLitellmNoticeWindowMs(loadAccess().litellmNoticeWindowMs),
|
|
23488
|
+
emitMetric: (event) => emitRuntimeMetric(event),
|
|
23489
|
+
log: (m) => process.stderr.write(`telegram gateway: ${m}\n`),
|
|
23490
|
+
})
|
|
23491
|
+
|
|
23416
23492
|
/**
|
|
23417
23493
|
* Broadcast a fleet-fallback FAILURE notice to every authorized chat.
|
|
23418
23494
|
*
|
|
@@ -25551,7 +25627,7 @@ bot.on('callback_query:data', async ctx => {
|
|
|
25551
25627
|
// sr-* TARGET tap: switch TO a non-Claude (LiteLLM/OpenRouter) model.
|
|
25552
25628
|
// Parity with the text `/model sr-*` path — claude's native picker rejects
|
|
25553
25629
|
// unknown sr-* ids, so an in-place inject can't set them. Carry the token
|
|
25554
|
-
// across a graceful restart (the
|
|
25630
|
+
// across a graceful restart (the consume-once `.session-model` carrier) and
|
|
25555
25631
|
// relaunch `claude --model sr-*`. Session-only; reverts to the configured
|
|
25556
25632
|
// default on the next restart. The sr-* → Claude direction is handled below
|
|
25557
25633
|
// via the SELECT/alias outcome + isSrToClaudeTransition.
|
|
@@ -27960,46 +28036,36 @@ async function shutdown(signal: string): Promise<void> {
|
|
|
27960
28036
|
} catch (err) {
|
|
27961
28037
|
process.stderr.write(`telegram gateway: shutdown.clean_marker_write_failed err=${(err as Error).message}\n`)
|
|
27962
28038
|
}
|
|
27963
|
-
//
|
|
27964
|
-
//
|
|
27965
|
-
//
|
|
27966
|
-
//
|
|
27967
|
-
//
|
|
27968
|
-
// the
|
|
27969
|
-
// bounce — stamp keep-intent for an active `.session-model` override so the
|
|
27970
|
-
// chosen model survives and start.sh re-confirms it via `.session-model-alert`
|
|
27971
|
-
// on boot. Respect an intent an initiator already stamped (a /restart stamps
|
|
27972
|
-
// 'revert' before SIGTERM): only stamp when none exists. A crash routes
|
|
27973
|
-
// through the non-OS-signal branch and still reverts (safe side preserved).
|
|
27974
|
-
try {
|
|
27975
|
-
const smDir = resolveAgentDirFromEnv()
|
|
27976
|
-
if (smDir != null) {
|
|
27977
|
-
const hasOverride = readSessionModelFile(smDir) != null
|
|
27978
|
-
const intentAlreadyStamped = existsSync(join(smDir, RELAUNCH_MODEL_INTENT_FILE))
|
|
27979
|
-
if (hasOverride && !intentAlreadyStamped) {
|
|
27980
|
-
// The GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX makes this stamp
|
|
27981
|
-
// recognisable at the next GATEWAY boot: a gateway-only bounce never
|
|
27982
|
-
// runs start.sh, so a leftover stamp with this prefix is cleared at
|
|
27983
|
-
// boot (clearStaleGatewayShutdownIntent) instead of lingering to
|
|
27984
|
-
// convert a later genuine crash into a "keep" (#3018 finding 4).
|
|
27985
|
-
writeRelaunchModelIntent(smDir, 'keep', `${GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX} graceful ${signal} shutdown (deploy/rolling restart) — preserving user-chosen session model`)
|
|
27986
|
-
process.stderr.write(`telegram gateway: shutdown.session_model_keep_stamped signal=${signal}\n`)
|
|
27987
|
-
}
|
|
27988
|
-
}
|
|
27989
|
-
} catch (err) {
|
|
27990
|
-
process.stderr.write(`telegram gateway: shutdown.session_model_keep_stamp_failed err=${(err as Error).message}\n`)
|
|
27991
|
-
}
|
|
28039
|
+
// Session-scoped (rev 4): a graceful deploy/restart REVERTS a live /model
|
|
28040
|
+
// override to the configured default — there is no keep-intent to stamp.
|
|
28041
|
+
// (A live Claude override never had a `.session-model` carrier; an in-flight
|
|
28042
|
+
// sr-* carrier was already consumed by its own apply-relaunch.) A queued —
|
|
28043
|
+
// not-yet-applied — /model IS still persisted just below so it applies as
|
|
28044
|
+
// the agent boots (its apply-relaunch), then reverts on the next restart.
|
|
27992
28045
|
} else {
|
|
27993
28046
|
process.stderr.write(`telegram gateway: shutdown.clean_marker_skipped signal=${signal} (crash path — banner will fire on next boot)\n`)
|
|
27994
28047
|
}
|
|
27995
28048
|
|
|
27996
28049
|
// #3018 finding 3 + #3039: resolve any queued /model|/effort ack cards. The
|
|
27997
28050
|
// gateway (and with it the in-memory queue) is going away — persist each
|
|
27998
|
-
// typed choice to the
|
|
27999
|
-
// `.session-effort
|
|
28000
|
-
//
|
|
28001
|
-
//
|
|
28002
|
-
// a
|
|
28051
|
+
// typed choice to the consume-once boot carriers (`.session-model` /
|
|
28052
|
+
// `.session-effort`, #3184/#3186) so it still deterministically applies as
|
|
28053
|
+
// the agent boots (that boot consumes the carrier; later restarts revert),
|
|
28054
|
+
// and edit the ack card to say so. Only an unresolvable menu-tag selection
|
|
28055
|
+
// falls back to a re-issue note. Best-effort and time-bounded so a wedged
|
|
28056
|
+
// Telegram API can't block shutdown.
|
|
28057
|
+
//
|
|
28058
|
+
// DELIBERATE (rev 4, #3184 review LOW-3 — applies to BOTH carriers): this
|
|
28059
|
+
// runs on EVERY shutdown path, including crashes (uncaughtException/
|
|
28060
|
+
// unhandledRejection route here), not just the isOsSignal branch above. So
|
|
28061
|
+
// a mid-turn queued /model or /effort + crash can apply on the
|
|
28062
|
+
// crash-recovery boot — technically at odds with a literal "crash reverts"
|
|
28063
|
+
// reading of the session-scoped contract. Intended: it preserves #3178's
|
|
28064
|
+
// "a queued command never silently vanishes" guarantee (the ack card
|
|
28065
|
+
// promised the switch), the model side is gated to offline-trusted tokens
|
|
28066
|
+
// (the effort side is allowlist-gated at write, so a garbage level can't
|
|
28067
|
+
// even cost one crash-boot), and the consume-once carriers bound it to
|
|
28068
|
+
// exactly that one recovery boot — the following restart reverts to config.
|
|
28003
28069
|
const orphanedCmdActions = pendingCmdShutdownResolutionActions(pendingSessionCommand, escapeHtmlForTg)
|
|
28004
28070
|
const orphanedCmdEdits = orphanedCmdActions.map(a => ({
|
|
28005
28071
|
chatId: a.cmd.ackChatId,
|
|
@@ -28844,6 +28910,23 @@ void (async () => {
|
|
|
28844
28910
|
} catch { /* leave override as-is on a bad read */ }
|
|
28845
28911
|
}
|
|
28846
28912
|
|
|
28913
|
+
// Effort sibling (#3186): start.sh records the EFFECTIVE launched
|
|
28914
|
+
// effort to `.active-session-effort` every boot. Re-hydrate the
|
|
28915
|
+
// in-memory session-effort override so the /effort menu highlight
|
|
28916
|
+
// stays honest after a queued-carrier apply-boot. Only an effort
|
|
28917
|
+
// differing from the configured default counts as an override.
|
|
28918
|
+
const activeEffortPath = join(smAgentDir, '.active-session-effort')
|
|
28919
|
+
if (existsSync(activeEffortPath)) {
|
|
28920
|
+
try {
|
|
28921
|
+
const launchedEffort = readFileSync(activeEffortPath, 'utf8').trim()
|
|
28922
|
+
const configuredEffort = getConfiguredEffortForPersist()
|
|
28923
|
+
sessionEffortOverride =
|
|
28924
|
+
launchedEffort.length > 0 && launchedEffort !== configuredEffort
|
|
28925
|
+
? launchedEffort
|
|
28926
|
+
: null
|
|
28927
|
+
} catch { /* leave override as-is on a bad read */ }
|
|
28928
|
+
}
|
|
28929
|
+
|
|
28847
28930
|
const alertPath = join(smAgentDir, '.session-model-alert')
|
|
28848
28931
|
if (existsSync(alertPath)) {
|
|
28849
28932
|
let alertText: string | null = null
|
|
@@ -29066,6 +29149,14 @@ void (async () => {
|
|
|
29066
29149
|
},
|
|
29067
29150
|
),
|
|
29068
29151
|
},
|
|
29152
|
+
// #3084 follow-up: the feed's send/edit adapters transit the send
|
|
29153
|
+
// gate, which SHEDS (resolves undefined) any call made during an
|
|
29154
|
+
// open flood window. Give the feed the SAME on-disk window probe
|
|
29155
|
+
// robustApiCall + the held-card sweep read, so a running/first-
|
|
29156
|
+
// paint tick parks in cooldown instead of re-firing a shed send
|
|
29157
|
+
// every ~6s for the whole ban (the worker-feed shed-contract bug:
|
|
29158
|
+
// 565 `sent.message_id` crashes in one 6h ban).
|
|
29159
|
+
floodWaitRemainingMs: probeFloodWaitRemainingMs,
|
|
29069
29160
|
log: (msg) => process.stderr.write(`telegram gateway: ${msg}\n`),
|
|
29070
29161
|
})
|
|
29071
29162
|
subagentWatcher = startSubagentWatcher({
|