switchroom 0.18.14 → 0.18.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/agent-scheduler/index.js +3 -0
  2. package/dist/auth-broker/index.js +473 -49
  3. package/dist/cli/notion-write-pretool.mjs +3 -0
  4. package/dist/cli/switchroom.js +1200 -1067
  5. package/dist/host-control/main.js +56 -51
  6. package/dist/vault/approvals/kernel-server.js +19 -12
  7. package/dist/vault/broker/server.js +675 -668
  8. package/package.json +1 -1
  9. package/profiles/_base/start.sh.hbs +81 -139
  10. package/telegram-plugin/dist/bridge/bridge.js +21 -0
  11. package/telegram-plugin/dist/gateway/gateway.js +531 -259
  12. package/telegram-plugin/dist/server.js +22 -1
  13. package/telegram-plugin/draft-stream.ts +78 -3
  14. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +3 -4
  15. package/telegram-plugin/gateway/effort-command.ts +9 -7
  16. package/telegram-plugin/gateway/gateway.ts +310 -219
  17. package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
  18. package/telegram-plugin/gateway/model-command.ts +96 -18
  19. package/telegram-plugin/gateway/pending-session-command.ts +10 -8
  20. package/telegram-plugin/gateway/session-model-file.ts +38 -172
  21. package/telegram-plugin/litellm-local-notice.ts +189 -0
  22. package/telegram-plugin/model-unavailable.ts +214 -0
  23. package/telegram-plugin/quota-watch.ts +16 -4
  24. package/telegram-plugin/runtime-metrics.ts +47 -0
  25. package/telegram-plugin/send-gate-degraded.test.ts +9 -7
  26. package/telegram-plugin/send-gate.ts +34 -4
  27. package/telegram-plugin/session-tail.ts +14 -2
  28. package/telegram-plugin/stream-controller.ts +143 -20
  29. package/telegram-plugin/stream-reply-handler.ts +12 -2
  30. package/telegram-plugin/tests/bot-api.harness.ts +7 -2
  31. package/telegram-plugin/tests/draft-stream.test.ts +110 -1
  32. package/telegram-plugin/tests/effort-command.test.ts +4 -4
  33. package/telegram-plugin/tests/flood-windows-persistence.test.ts +2 -2
  34. package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
  35. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
  36. package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
  37. package/telegram-plugin/tests/model-command.test.ts +84 -1
  38. package/telegram-plugin/tests/model-unavailable.test.ts +187 -0
  39. package/telegram-plugin/tests/operator-events-session-tail.test.ts +55 -0
  40. package/telegram-plugin/tests/quota-watch.test.ts +21 -0
  41. package/telegram-plugin/tests/reaction-gate-routing.test.ts +2 -2
  42. package/telegram-plugin/tests/runtime-metrics.test.ts +24 -0
  43. package/telegram-plugin/tests/session-model-file.test.ts +7 -155
  44. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
  45. package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
  46. package/telegram-plugin/tests/throttle-tier.test.ts +176 -0
  47. package/telegram-plugin/tests/worker-activity-feed.test.ts +207 -0
  48. package/telegram-plugin/throttle-tier.ts +98 -1
  49. package/telegram-plugin/worker-activity-feed.ts +83 -8
@@ -320,11 +320,14 @@ import {
320
320
  type ModelUnavailableDetection,
321
321
  } from '../model-unavailable.js'
322
322
  import {
323
+ build429ClassifiedMetric,
324
+ classify429Detail,
323
325
  decideThrottleTier,
324
- isAccountScopedThrottle,
325
326
  throttleRetryInPlaceMaxMs,
326
327
  } from '../throttle-tier.js'
327
328
  import { createThrottleTierRunner } from './throttle-tier-wiring.js'
329
+ import { parseLitellmNoticeWindowMs } from '../litellm-local-notice.js'
330
+ import { createLitellmLocalNoticeRunner, decideRateLimitedSurface } from './litellm-local-notice-wiring.js'
328
331
  import { runFleetAutoFallback, renderFallbackFailureNotice, evaluateFallbackFailureNotice, evaluateAllBlockedNotice, type FallbackFailureNoticeState, type FallbackAllBlockedNoticeState } from '../auto-fallback-fleet.js'
329
332
  import { startRestartWatchdog } from './restart-watchdog.js'
330
333
  import { validateStringArray } from './access-validator.js'
@@ -435,6 +438,8 @@ import { injectSlashCommand as injectSlashCommandImpl } from '../../src/agents/i
435
438
  import { handleInjectCommand, type InjectDeps } from './inject-handler.js'
436
439
  import {
437
440
  parseModelCommand,
441
+ planModelCommand,
442
+ modelCommandReceiptLine,
438
443
  handleModelCommand,
439
444
  buildModelMenu,
440
445
  handleModelMenuCallback,
@@ -460,18 +465,9 @@ import {
460
465
  readSessionModelFileRaw,
461
466
  restoreSessionModelFileRaw,
462
467
  clearSessionModelFile,
463
- clearSessionModelBootAttempts,
464
468
  readConfiguredDefaultModel,
465
- writeRelaunchModelIntent,
466
- clearRelaunchModelIntent,
467
- intentForRestartReason,
468
- readSessionModelFile,
469
- RELAUNCH_MODEL_INTENT_FILE,
470
- GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX,
471
- clearStaleGatewayShutdownIntent,
472
469
  writeSessionEffortFile,
473
470
  clearSessionEffortFile,
474
- readSessionEffortFile,
475
471
  } from './session-model-file.js'
476
472
  import { discoverModels, selectModel } from '../../src/agents/model-picker.js'
477
473
  import { resolveMainModel, SWITCHROOM_DEFAULT_THINKING_EFFORT } from '../../src/agents/scaffold.js'
@@ -1096,16 +1092,10 @@ function triggerSelfRestart(
1096
1092
  )
1097
1093
  return false
1098
1094
  }
1099
- // Session-model stickiness (reference/rfcs/session-model-stickiness.md):
1100
- // boot default is REVERT, so every switchroom-managed bounce must stamp
1101
- // its intent BEFORE the SIGTERM is even scheduled (write-before-kill
1102
- // invariant — pinned by gateway-session-model-relaunch.test.ts). The
1103
- // per-reason table classifies recovery/model-switch bounces as "keep"
1104
- // and the deliberate inline restart button as "revert".
1105
- {
1106
- const smDir = resolveAgentDirFromEnv()
1107
- if (smDir) writeRelaunchModelIntent(smDir, intentForRestartReason(reason), reason)
1108
- }
1095
+ // Session-scoped /model (reference/rfcs/session-model-stickiness.md §0.1):
1096
+ // a `.session-model` carrier is consume-once — start.sh applies it on the
1097
+ // apply-relaunch and deletes it, so no boot needs a keep/revert intent. A
1098
+ // switchroom-managed bounce simply reverts to the configured default.
1109
1099
  process.stderr.write(
1110
1100
  `telegram gateway: restart-via-SIGTERM-PID1 agent=${targetAgent} reason=${reason} (docker)\n`,
1111
1101
  )
@@ -1117,10 +1107,6 @@ function triggerSelfRestart(
1117
1107
  return true
1118
1108
  }
1119
1109
  // Legacy systemd path.
1120
- if (targetAgent === selfAgent) {
1121
- const smDir = resolveAgentDirFromEnv()
1122
- if (smDir) writeRelaunchModelIntent(smDir, intentForRestartReason(reason), reason)
1123
- }
1124
1110
  process.stderr.write(
1125
1111
  `telegram gateway: restart-via-systemctl agent=${targetAgent} reason=${reason}\n`,
1126
1112
  )
@@ -1140,24 +1126,6 @@ function triggerSelfRestart(
1140
1126
  }
1141
1127
  }
1142
1128
 
1143
- // #3018 finding 4: a gateway-only bounce (supervisor relaunch, bare gateway
1144
- // unit restart) leaves the shutdown handler's deploy-survival keep-intent
1145
- // stamp on disk UNCONSUMED — start.sh only runs on a container-level boot.
1146
- // If this gateway boot still sees a gateway-shutdown-stamped intent, the
1147
- // preceding bounce was gateway-only: clear it so a genuine crash inside the
1148
- // 10-min freshness window can't be converted into a "keep" (crash-reverts
1149
- // policy intact). A real container stop/deploy consumes the file in start.sh
1150
- // before any gateway boots, so a legitimate deploy stamp is never touched;
1151
- // triggerSelfRestart / user-slash stamps use un-prefixed reasons.
1152
- {
1153
- const bootSmDir = resolveAgentDirFromEnv()
1154
- if (bootSmDir != null && clearStaleGatewayShutdownIntent(bootSmDir)) {
1155
- process.stderr.write(
1156
- 'telegram gateway: cleared stale gateway-shutdown relaunch-model intent (previous bounce was gateway-only — container never restarted)\n',
1157
- )
1158
- }
1159
- }
1160
-
1161
1129
  // Cached lazily — the claude CLI binary doesn't change inside a running
1162
1130
  // gateway process; on `switchroom update` the gateway restarts, refreshing this.
1163
1131
  let cachedClaudeCliVersion: string | null | undefined = undefined
@@ -1409,6 +1377,11 @@ type Access = {
1409
1377
  parseMode?: 'html' | 'markdownv2' | 'text'
1410
1378
  disableLinkPreview?: boolean
1411
1379
  coalescingGapMs?: number
1380
+ /** Cooldown window (ms) for the litellm-local 429 notice — the debounced
1381
+ * "fleet token limiter engaged" message (litellm-local-notice.ts). Default
1382
+ * 15 min when unset/invalid (parseLitellmNoticeWindowMs). Projected from
1383
+ * channels.telegram.litellm_notice.window_ms by scaffold. */
1384
+ litellmNoticeWindowMs?: number
1412
1385
  /** A2: max media attachments folded into one coalesced turn. Default 10
1413
1386
  * (a full Telegram album / forwarded burst arrives as one turn). Set 1 to
1414
1387
  * restore single-attachment behaviour. Projected from
@@ -1563,6 +1536,7 @@ function readAccessFile(): Access {
1563
1536
  parseMode: parsed.parseMode,
1564
1537
  disableLinkPreview: parsed.disableLinkPreview,
1565
1538
  coalescingGapMs: parsed.coalescingGapMs,
1539
+ litellmNoticeWindowMs: parsed.litellmNoticeWindowMs,
1566
1540
  coalesceMaxAttachments: parsed.coalesceMaxAttachments,
1567
1541
  interruptSafeBoundary: parsed.interruptSafeBoundary,
1568
1542
  interruptMaxWaitMs: parsed.interruptMaxWaitMs,
@@ -7700,14 +7674,99 @@ function emitGatewayOperatorEvent(event: OperatorEvent): void {
7700
7674
  // through to the existing calm rate-limited card unchanged — an
7701
7675
  // account-scoped throttle would be the wrong action for a server-wide
7702
7676
  // condition.
7677
+ //
7678
+ // Classification is three-way (classify429Detail, throttle-tier.ts):
7679
+ // only `account-scoped` may enter the throttle tier. `litellm-local` —
7680
+ // the LiteLLM proxy's OWN tpm/rpm/router limiter tripping before the
7681
+ // request reached Anthropic — takes the calm path with NO broker mark and
7682
+ // NO failover: the condition is proxy-local and says nothing about the
7683
+ // account. Every rate-limited event ALSO emits one
7684
+ // `rate_limit_429_classified` runtime metric (PostHog + JSONL), fired
7685
+ // here — before the cooldown gate — so operators can correlate
7686
+ // account-scoped 429s with fleet TPM even when the card is suppressed.
7703
7687
  let throttleEscalation: ModelUnavailableDetection | null = null
7704
7688
  let escalationFired = false
7705
- if (kind === 'rate-limited' && isAccountScopedThrottle(event.detail)) {
7689
+ // True when decideRateLimitedSurface already consulted (and armed) the
7690
+ // shared per-kind card cooldown for this event — the gate below must not
7691
+ // re-consult it, or the arm from the first consult would self-suppress.
7692
+ let rateLimitedCooldownConsulted = false
7693
+ const rateLimit429Classification =
7694
+ kind === 'rate-limited' ? classify429Detail(event.detail) : null
7695
+ if (rateLimit429Classification != null && rateLimit429Classification !== 'account-scoped') {
7696
+ // litellm-local / generic-transient: the calm path. NO broker
7697
+ // mark-throttled, NO throttle-tier runner, NO failover — for a
7698
+ // proxy-local cap trip those would bench an account that was never
7699
+ // touched.
7700
+ emitRuntimeMetric(
7701
+ build429ClassifiedMetric({
7702
+ agent,
7703
+ detail: event.detail,
7704
+ classification: rateLimit429Classification,
7705
+ action: 'calm',
7706
+ now: Date.now(),
7707
+ }),
7708
+ )
7709
+ // Surface decision — extracted (decideRateLimitedSurface,
7710
+ // litellm-local-notice-wiring.ts) so the ordering contract with the
7711
+ // shared per-kind card cooldown is pinnable by tests: litellm-local
7712
+ // resolves BEFORE the gate (never arms `${agent}:rate-limited`, never
7713
+ // suppressed by a cooldown a recent 529/generic card armed);
7714
+ // generic-transient consults the gate exactly once HERE.
7715
+ const surface = decideRateLimitedSurface({
7716
+ classification: rateLimit429Classification,
7717
+ agent,
7718
+ shouldEmitCard: (a) => shouldEmitOperatorEvent(a, 'rate-limited'),
7719
+ })
7720
+ if (surface === 'litellm-local-notice') {
7721
+ process.stderr.write(
7722
+ `telegram gateway: 429 classified litellm-proxy-local agent=${agent} — ` +
7723
+ `calm path, no account attribution, no failover\n`,
7724
+ )
7725
+ // The dedicated debounced notice REPLACES the generic "🚦 Rate limited"
7726
+ // card for this classification only: one calm message naming the fleet
7727
+ // token limiter (LiteLLM tpm_limit/rpm_limit) instead of a card that
7728
+ // reads like an Anthropic account problem. Classification, quota-ledger,
7729
+ // and failover behavior are untouched (nothing fired above on this
7730
+ // branch). Record into history ONLY when a notice actually posted — a
7731
+ // suppressed low-stakes proxy throttle must not overwrite a more
7732
+ // important most-recent event (e.g. credentials-expired) in the
7733
+ // /status enrichment (operator-events-history keeps the most recent
7734
+ // event per agent).
7735
+ const outcome = litellmLocalNoticeRunner.onRateLimited('litellm-local', agent)
7736
+ if (outcome === 'sent') {
7737
+ try {
7738
+ recordOperatorEvent(event)
7739
+ } catch { /* history is best-effort */ }
7740
+ }
7741
+ return
7742
+ }
7743
+ if (surface === 'cooldown-suppressed') {
7744
+ process.stderr.write(
7745
+ `telegram gateway: operator-event suppressed (cooldown) agent=${agent} kind=${kind}\n`,
7746
+ )
7747
+ return
7748
+ }
7749
+ // 'generic-card' — the gate passed (and armed) above; fall through to
7750
+ // the existing calm rate-limited card without re-consulting it.
7751
+ rateLimitedCooldownConsulted = true
7752
+ }
7753
+ if (rateLimit429Classification === 'account-scoped') {
7706
7754
  const throttleDecision = decideThrottleTier({
7707
7755
  detail: event.detail,
7708
7756
  now: Date.now(),
7709
7757
  thresholdMs: throttleRetryInPlaceMaxMs(),
7710
7758
  })
7759
+ emitRuntimeMetric(
7760
+ build429ClassifiedMetric({
7761
+ agent,
7762
+ detail: event.detail,
7763
+ classification: 'account-scoped',
7764
+ // decideThrottleTier can't return 'none' for account-scoped wording;
7765
+ // map the two live actions onto the metric's vocabulary.
7766
+ action: throttleDecision.action === 'failover' ? 'failover' : 'throttle',
7767
+ now: Date.now(),
7768
+ }),
7769
+ )
7711
7770
  if (throttleDecision.action === 'throttle') {
7712
7771
  // Reset is near (≤ threshold) or unparseable (60s default): DO NOT
7713
7772
  // fail over. Record throttled_until broker-side, post ONE lightweight
@@ -7754,7 +7813,7 @@ function emitGatewayOperatorEvent(event: OperatorEvent): void {
7754
7813
  }
7755
7814
  }
7756
7815
 
7757
- if (!shouldEmitOperatorEvent(agent, kind)) {
7816
+ if (!rateLimitedCooldownConsulted && !shouldEmitOperatorEvent(agent, kind)) {
7758
7817
  process.stderr.write(
7759
7818
  `telegram gateway: operator-event suppressed (cooldown) agent=${agent} kind=${kind}\n`,
7760
7819
  )
@@ -10332,16 +10391,6 @@ const ipcServer: IpcServer = createIpcServer({
10332
10391
  // this call is safe wherever it sits relative to the cron early-return
10333
10392
  // above (#3038 review finding 5).
10334
10393
  bridgeDeadWatchdog.noteBridgeRegistered(client.agentName)
10335
- // #3043 item 2: a REAL bridge registering is proof the boot came all the
10336
- // way up healthy — clear start.sh's crashloop boot-attempts counter so only
10337
- // boots that genuinely fail BEFORE the bridge registers accumulate toward
10338
- // the 3-strike override clear. Without this, three quick operator
10339
- // hand-bounces of a healthy agent (each <150s apart) spuriously wipe a
10340
- // working model override. Best-effort; no-op when the file is absent.
10341
- if (client.agentName != null) {
10342
- const smBootDir = resolveAgentDirFromEnv()
10343
- if (smBootDir != null) clearSessionModelBootAttempts(smBootDir)
10344
- }
10345
10394
  client.send({ type: 'status', status: 'agent_connected' })
10346
10395
 
10347
10396
  // Phase 2b PR 3a — bridgeUp cutover. The state machine's `bridgeUp`
@@ -21775,15 +21824,10 @@ function buildModelDeps(restartCtx?: ModelDepsRestartContext): ModelMenuDeps & M
21775
21824
  })
21776
21825
  }
21777
21826
  stampUserRestartReason(reason)
21778
- // Model-switch restarts are switchroom-managed relaunches: the session
21779
- // override (written by the caller before this dispatch) must survive
21780
- // the bounce, so stamp keep-intent BEFORE dispatch (boot default is
21781
- // revert). hostd shells through `switchroom agent restart`, which
21782
- // deliberately writes no intent of its own.
21783
- {
21784
- const smDir = resolveAgentDirFromEnv()
21785
- if (smDir) writeRelaunchModelIntent(smDir, 'keep', reason)
21786
- }
21827
+ // Model-switch restarts are the APPLY-relaunch for a `.session-model`
21828
+ // carrier the caller wrote immediately above: start.sh consumes it on
21829
+ // the very next boot (this dispatch's boot), then reverts thereafter.
21830
+ // No keep/revert intent is needed — the carrier is consume-once.
21787
21831
  await sweepBeforeSelfRestart()
21788
21832
  const hostdResp = await tryHostdDispatch(name, {
21789
21833
  v: 1,
@@ -21804,14 +21848,11 @@ function buildModelDeps(restartCtx?: ModelDepsRestartContext): ModelMenuDeps & M
21804
21848
  return
21805
21849
  }
21806
21850
  // hostd is configured but returned an error/denied result. No restart
21807
- // is coming, so the keep-intent stamped above must not linger — a
21808
- // crash within its 10-min freshness window would wrongly KEEP.
21851
+ // is coming, so the carrier the caller wrote must not linger to be
21852
+ // consumed by an unrelated later boot — scheduleModelRelaunch's catch
21853
+ // rolls it back on this throw.
21809
21854
  if (hostdResp.result !== 'started' && hostdResp.result !== 'completed') {
21810
21855
  clearRestartMarker()
21811
- {
21812
- const smDir = resolveAgentDirFromEnv()
21813
- if (smDir) clearRelaunchModelIntent(smDir)
21814
- }
21815
21856
  throw new Error(
21816
21857
  `hostd restart failed (result=${hostdResp.result}): ${hostdResp.error ?? '(no details)'}`,
21817
21858
  )
@@ -21820,11 +21861,11 @@ function buildModelDeps(restartCtx?: ModelDepsRestartContext): ModelMenuDeps & M
21820
21861
  /**
21821
21862
  * Switch TO a model that needs a relaunch (sr-* LiteLLM/OpenRouter ids,
21822
21863
  * which claude's native `/model` picker rejects, and the sr-to-claude
21823
- * direction). Write the DURABLE `.session-model` override (start.sh
21824
- * applies it on every keep-relaunch boot and launches `claude --model
21825
- * <token>`), set the in-memory session-model so /status stays honest
21826
- * across the restart window, then run the SAME restart dispatch as
21827
- * scheduleRestart above — which stamps the keep-intent this boot needs.
21864
+ * direction). Write the CONSUME-ONCE `.session-model` carrier (start.sh
21865
+ * applies it on the very next boot — this dispatch's apply-relaunch — and
21866
+ * deletes it, so it reverts on any subsequent restart), set the in-memory
21867
+ * session-model so /status stays honest across the restart window, then
21868
+ * run the SAME restart dispatch as scheduleRestart above.
21828
21869
  */
21829
21870
  scheduleModelRelaunch: async (model: string, reason: string) => {
21830
21871
  const agentDir = resolveAgentDirFromEnv()
@@ -21841,20 +21882,15 @@ function buildModelDeps(restartCtx?: ModelDepsRestartContext): ModelMenuDeps & M
21841
21882
  try {
21842
21883
  await deps.scheduleRestart(reason)
21843
21884
  } catch (err) {
21844
- // A restart already in flight OWNS the override we just wrote — its
21845
- // boot (stamped keep by the in-flight path's own intent, last-writer-
21846
- // wins) will apply our token, so the switch is queued, not lost: keep
21847
- // the file + override and let the caller tell the operator "~15s".
21848
- // Any OTHER dispatch failure means no restart is coming, so roll BOTH
21849
- // back — a lingering file/override would lie to /status and
21850
- // mis-launch the NEXT relaunch. The keep-intent goes with them
21851
- // (belt-and-braces: scheduleRestart's failure branch clears it too):
21852
- // a fresh keep on disk with no restart coming would wrongly KEEP
21853
- // across a crash inside its 10-min window.
21885
+ // A restart already in flight OWNS the carrier we just wrote — its
21886
+ // boot will consume+apply our token, so the switch is queued, not
21887
+ // lost: keep the file + override and let the caller tell the operator
21888
+ // "~15s". Any OTHER dispatch failure means no restart is coming, so
21889
+ // roll BOTH back — a lingering carrier would lie to /status and be
21890
+ // consumed (mis-applied) by an unrelated later boot.
21854
21891
  if ((err as { code?: string })?.code !== 'restart_in_flight') {
21855
21892
  restoreSessionModelFileRaw(agentDir, prevFileRaw)
21856
21893
  sessionModelSource.setOverride(prevOverride)
21857
- clearRelaunchModelIntent(agentDir)
21858
21894
  }
21859
21895
  throw err
21860
21896
  }
@@ -21875,19 +21911,24 @@ function modelMenuReplyMarkup(reply: ModelMenuReply): InlineKeyboard | undefined
21875
21911
 
21876
21912
  /**
21877
21913
  * Record a POSITIVELY-CONFIRMED typed `/model` switch: set the in-memory
21878
- * override so `/status` reflects the live model and persist the sticky
21879
- * `.session-model` carrier. Shared by the live `bot.command('model')` handler
21880
- * and the deferred (queued mid-turn) apply so both record identically. Returns
21881
- * a persist-warning suffix to append to the reply body (empty when clean).
21914
+ * override so `/status` reflects the live model. Shared by the live
21915
+ * `bot.command('model')` handler and the deferred (queued mid-turn) apply so
21916
+ * both record identically. Returns a warning suffix to append to the reply
21917
+ * body (currently always empty — kept for a stable signature).
21882
21918
  *
21883
- * The `/status` honesty invariant lives here: only `reply.selectedModel`
21884
- * (present only on a confirmed switch) records; an unverified inject records
21885
- * nothing. `/model default` clears the carrier idempotently.
21919
+ * Session-scoped (rev 4): a live Claude `/model` switch writes NO
21920
+ * `.session-model` carrier — it applies in-session and the explicit
21921
+ * `claude --model <configured>` flag reverts it on the next boot, so it lasts
21922
+ * exactly until the next restart with no durable state. (sr-* switches never
21923
+ * reach here — they go through scheduleModelRelaunch, which owns the
21924
+ * consume-once carrier.) The `/status` honesty invariant lives here: only
21925
+ * `reply.selectedModel` records; an unverified inject records nothing.
21926
+ * `/model default` clears any in-memory override and any leftover carrier.
21886
21927
  */
21887
21928
  function recordTypedModelSwitch(
21888
21929
  reply: { text: string; selectedModel?: string },
21889
21930
  requestedModelArg: string | null,
21890
- deps: ModelCommandDeps,
21931
+ _deps: ModelCommandDeps,
21891
21932
  ): string {
21892
21933
  const requested = requestedModelArg != null ? expandSrAlias(requestedModelArg) : null
21893
21934
  if (requested?.toLowerCase() === 'default') {
@@ -21898,28 +21939,14 @@ function recordTypedModelSwitch(
21898
21939
  }
21899
21940
  if (!reply.selectedModel) return ''
21900
21941
  sessionModelSource.setOverride(reply.selectedModel)
21901
- const smDir = resolveAgentDirFromEnv()
21902
- if (smDir && requested && isValidModelArg(requested) && !isSrModel(requested)) {
21903
- try {
21904
- writeSessionModelFile(
21905
- smDir,
21906
- requested,
21907
- readConfiguredDefaultModel(smDir) ??
21908
- resolveMainModel(deps.getConfiguredModel() ?? undefined),
21909
- )
21910
- } catch (err) {
21911
- process.stderr.write(
21912
- `telegram gateway: session-model persist failed (typed /model): ${(err as Error)?.message ?? String(err)}\n`,
21913
- )
21914
- return '\n⚠️ Couldn’t persist the sticky override — the switch is live now but won’t survive a relaunch.'
21915
- }
21916
- }
21917
21942
  return ''
21918
21943
  }
21919
21944
 
21920
21945
  /**
21921
- * Record a model-MENU callback outcome (persist/clear sticky override) and
21922
- * drive an sr-*→Claude graceful restart when the tap crosses that boundary.
21946
+ * Record a model-MENU callback outcome (set the live in-memory override, clear
21947
+ * a leftover carrier on a Default tap) and drive an sr-*→Claude graceful
21948
+ * restart when the tap crosses that boundary — the only menu path that writes a
21949
+ * consume-once `.session-model` carrier (a live Claude tap writes none, rev 4).
21923
21950
  * Extracted from the live `mdl:*` dispatcher so the deferred (queued mid-turn)
21924
21951
  * apply records + restarts identically. Does NOT edit any Telegram message —
21925
21952
  * callers own the card edit. Returns a restart notice when a session restart
@@ -21933,29 +21960,13 @@ function recordModelMenuSideEffects(
21933
21960
  prevSessionModel: string | null,
21934
21961
  ): { restartNotice?: string } {
21935
21962
  // Record a successful session switch so /status reflects what's actually
21936
- // running, and persist the STICKY override
21937
- // (reference/rfcs/session-model-stickiness.md): the canonical token (never
21938
- // the display label) goes to the durable `.session-model`; a confirmed
21939
- // "Default (recommended)" selection clears it instead.
21963
+ // running. Session-scoped (rev 4): a live Claude menu tap writes NO
21964
+ // `.session-model` carrier — it applies in-session (native picker) and
21965
+ // reverts on the next boot. Only the sr→Claude transition below (which
21966
+ // relaunches) writes the consume-once carrier. A confirmed "Default
21967
+ // (recommended)" selection clears any leftover carrier.
21940
21968
  if (outcome.selectedModel) {
21941
21969
  sessionModelSource.setOverride(outcome.selectedModel)
21942
- const smDir = resolveAgentDirFromEnv()
21943
- if (smDir && outcome.selectedModelToken) {
21944
- try {
21945
- writeSessionModelFile(
21946
- smDir,
21947
- outcome.selectedModelToken,
21948
- readConfiguredDefaultModel(smDir) ??
21949
- resolveMainModel(modelDeps.getConfiguredModel() ?? undefined),
21950
- )
21951
- } catch (err) {
21952
- outcome.reply.text +=
21953
- '\n⚠️ Couldn’t persist the sticky override — the switch is live now but won’t survive a relaunch.'
21954
- process.stderr.write(
21955
- `telegram gateway: session-model persist failed (menu): ${(err as Error)?.message ?? String(err)}\n`,
21956
- )
21957
- }
21958
- }
21959
21970
  }
21960
21971
  if (outcome.clearedDefault) {
21961
21972
  const smDir = resolveAgentDirFromEnv()
@@ -21967,9 +21978,11 @@ function recordModelMenuSideEffects(
21967
21978
  // torn down — a graceful restart (same mechanism as /restart) is required.
21968
21979
  if (outcome.selectedModel && isSrToClaudeTransition(prevSessionModel, outcome.selectedModel)) {
21969
21980
  const agentName = getMyAgentName()
21970
- // Carry the requested Claude model across the restart via the SAME durable
21971
- // `.session-model` override a Claude → sr-* switch uses — otherwise boot
21972
- // launches the CONFIGURED default and the tapped model is silently dropped.
21981
+ // Carry the requested Claude model across the restart via the SAME
21982
+ // consume-once `.session-model` carrier a Claude → sr-* switch uses —
21983
+ // otherwise this transition's apply-relaunch boots the CONFIGURED default
21984
+ // and the tapped model is silently dropped. Applied on that one boot,
21985
+ // then reverts on the next restart (rev 4).
21973
21986
  const agentDir = resolveAgentDirFromEnv()
21974
21987
  const token = outcome.selectedModelToken
21975
21988
  if (agentDir && token) {
@@ -22132,20 +22145,19 @@ function persistQueuedCommandForRestart(action: ShutdownResolutionAction): strin
22132
22145
  switch (action.persist) {
22133
22146
  case 'model': {
22134
22147
  // #3042 blocker 2a: this token was QUEUED, never confirmed by claude.
22135
- // Under the keep-by-default boot a garbage-but-shape-valid token
22136
- // persisted here would crashloop `claude --model <garbage>` with the
22137
- // gateway dead. Only offline-trustable tokens (static Claude aliases,
22138
- // curated sr-* alias targets) may be persisted unconfirmed; anything
22139
- // else gets the honest "couldn't verify — re-issue" card instead.
22148
+ // The carrier is written immediately before the bounce, so the next
22149
+ // boot IS its apply-relaunch: consume-once means a garbage token can
22150
+ // crash at most one boot before it reverts, but we still gate on
22151
+ // offline-trustable tokens (static Claude aliases, curated sr-* alias
22152
+ // targets) to avoid even that one crash-boot; anything else gets the
22153
+ // honest "couldn't verify — re-issue" card instead.
22140
22154
  if (!isOfflineTrustedModelToken(action.arg)) {
22141
22155
  return `↩️ Couldn’t verify \`${escapeHtmlForTg(action.cmd.targetLabel || action.arg)}\` as a known model without the live session — it was NOT saved. Re-issue \`/model ${escapeHtmlForTg(action.arg)}\` once the agent is back.`
22142
22156
  }
22143
22157
  const configured =
22144
22158
  readConfiguredDefaultModel(agentDir) ?? resolveMainModel(undefined)
22159
+ // Consume-once carrier: applied by the next boot, then reverts.
22145
22160
  writeSessionModelFile(agentDir, expandSrAlias(action.arg), configured)
22146
- // Boot default is keep (#3039), but stamp explicit keep-intent for
22147
- // reason-honesty in the boot notice.
22148
- writeRelaunchModelIntent(agentDir, 'keep', 'queued /model carried across restart')
22149
22161
  break
22150
22162
  }
22151
22163
  case 'clear-model':
@@ -22235,17 +22247,50 @@ bot.command('model', async ctx => {
22235
22247
  const parsed = parseModelCommand(text) ?? { kind: 'show' as const }
22236
22248
  const chatId = String(ctx.chat!.id)
22237
22249
  const threadId = resolveThreadId(chatId, ctx.message?.message_thread_id)
22250
+ // #3177 — durable receipt FIRST. A typed /model must NEVER be invisible: even
22251
+ // if every downstream reply is shed/dropped, or the session is in a
22252
+ // phantom-idle window (turn atom cleared while claude is still busy), the
22253
+ // command leaves a greppable log line + a history row before any branch. This
22254
+ // is the fix for the finn 2026-07-12 zero-trace swallow (no log, no reply, no
22255
+ // ack, no deferred apply). `busyNow` folds BOTH busy signals (turn atom AND
22256
+ // the authoritative delivery-machine/approval gate).
22257
+ const busyNow = currentTurn !== null || turnInFlightForGate()
22258
+ process.stderr.write(modelCommandReceiptLine(getMyAgentName(), parsed, busyNow) + '\n')
22259
+ if (HISTORY_ENABLED && ctx.message?.message_id != null) {
22260
+ try {
22261
+ recordInbound({
22262
+ chat_id: chatId,
22263
+ thread_id: threadId ?? null,
22264
+ message_id: ctx.message.message_id,
22265
+ user: ctx.from?.username ?? (ctx.from?.id != null ? String(ctx.from.id) : null),
22266
+ user_id: ctx.from?.id != null ? String(ctx.from.id) : null,
22267
+ ts: ctx.message.date ?? Math.floor(Date.now() / 1000),
22268
+ text,
22269
+ })
22270
+ } catch (err) {
22271
+ process.stderr.write(`telegram gateway: /model recordInbound failed: ${(err as Error)?.message ?? String(err)}\n`)
22272
+ }
22273
+ }
22238
22274
  const deps = buildModelDeps({ chatId, threadId })
22239
- if (parsed.kind === 'show' && process.env.SWITCHROOM_MODEL_MENU !== '0') {
22275
+ // Route on a pure disposition (#3177) that folds BOTH busy signals so a
22276
+ // session busy by EITHER measure ack+queues instead of silently injecting
22277
+ // into a busy pane. Every branch below produces a visible action.
22278
+ const disposition = planModelCommand(parsed, {
22279
+ currentTurnActive: currentTurn !== null,
22280
+ turnInFlight: turnInFlightForGate(),
22281
+ menuEnabled: process.env.SWITCHROOM_MODEL_MENU !== '0',
22282
+ })
22283
+ if (disposition.kind === 'menu') {
22240
22284
  const menu = await buildModelMenu(deps)
22241
22285
  await switchroomReply(ctx, menu.text, { html: true, reply_markup: modelMenuReplyMarkup(menu) })
22242
22286
  return
22243
22287
  }
22244
- // Mid-turn: instead of dead-ending ("Try again in a moment"), ACK + QUEUE +
22245
- // apply-on-idle + confirm (#3017). The typed set path either injects into
22246
- // claude's input box or triggers a carrier restart — both unsafe mid-turn.
22247
- if (parsed.kind === 'set' && deps.isBusy()) {
22248
- const target = expandSrAlias(parsed.model)
22288
+ // Mid-turn (by either busy signal): instead of dead-ending ("Try again in a
22289
+ // moment") or silently injecting into a busy pane, ACK + QUEUE + apply-on-idle
22290
+ // + confirm (#3017/#3177). The typed set path either injects into claude's
22291
+ // input box or triggers a carrier restart — both unsafe while busy.
22292
+ if (disposition.kind === 'queue') {
22293
+ const target = disposition.target
22249
22294
  const sent = await ctx.replyWithRichMessage(
22250
22295
  richMessage(hardenCardBreaks(pendingCmdAckText('model', target, escapeHtmlForTg))),
22251
22296
  threadId != null ? { message_thread_id: threadId } : {},
@@ -22285,39 +22330,38 @@ bot.command('model', async ctx => {
22285
22330
  // is blocklisted for `/effort` since #2471), session-scoped — boot re-pins
22286
22331
  // the configured default via start.sh's `--effort`. Implementation in
22287
22332
  // effort-command.ts so it's unit-testable without booting the bot.
22333
+
22334
+ // The live session-effort override, in memory only (#3186, session-scoped
22335
+ // like /model rev 4). A confirmed live apply records here — NOT to the
22336
+ // `.session-effort` carrier — so it lasts exactly until the next restart
22337
+ // (start.sh's explicit `--effort <configured>` reverts it for free). Seeded
22338
+ // at boot from `.active-session-effort` (the effort sibling of
22339
+ // `.active-session-model`) so a queued-carrier apply-boot still shows the
22340
+ // honest live level on the /effort menu.
22341
+ let sessionEffortOverride: string | null = null
22342
+
22288
22343
  function buildEffortDeps(): EffortCommandDeps {
22289
22344
  return {
22290
- // #3039: single persistence choke point — EVERY positively-confirmed
22291
- // effort apply (typed, menu tap, queued drain) durably records the level
22292
- // to `.session-effort`, which start.sh resolves into `--effort` on every
22293
- // boot. `/effort default` clears it via clearSessionEffort (the handler
22294
- // clears AFTER its restore-apply, so the wrapper's write is undone).
22345
+ // Session-scoped (#3186): a positively-confirmed live apply records the
22346
+ // level IN MEMORY only — no durable carrier. The `.session-effort`
22347
+ // carrier is written solely by persistQueuedCommandForRestart (a queued
22348
+ // mid-turn /effort carried across the bounce) and is consume-once at
22349
+ // boot. `/effort default` clears via clearSessionEffort below.
22295
22350
  applyEffort: async (agent, level) => {
22296
22351
  const result = await applyEffort(agent, level)
22297
- if (result.ok) {
22298
- const agentDir = resolveAgentDirFromEnv()
22299
- if (agentDir) {
22300
- try {
22301
- writeSessionEffortFile(agentDir, level, getConfiguredEffortForPersist())
22302
- } catch (err) {
22303
- process.stderr.write(
22304
- `telegram gateway: session-effort persist failed level=${level}: ${(err as Error)?.message ?? String(err)}\n`,
22305
- )
22306
- }
22307
- }
22308
- }
22352
+ if (result.ok) sessionEffortOverride = level
22309
22353
  return result
22310
22354
  },
22311
22355
  getAgentName: getMyAgentName,
22312
22356
  getConfiguredEffort: () => getConfiguredEffortForPersist(),
22313
22357
  clearSessionEffort: () => {
22358
+ sessionEffortOverride = null
22359
+ // Also drop any leftover queued-command carrier so the next boot can't
22360
+ // consume a stale level the user just cleared.
22314
22361
  const agentDir = resolveAgentDirFromEnv()
22315
22362
  if (agentDir) clearSessionEffortFile(agentDir)
22316
22363
  },
22317
- getSessionEffort: () => {
22318
- const agentDir = resolveAgentDirFromEnv()
22319
- return agentDir ? (readSessionEffortFile(agentDir)?.level ?? null) : null
22320
- },
22364
+ getSessionEffort: () => sessionEffortOverride,
22321
22365
  escapeHtml: escapeHtmlForTg,
22322
22366
  }
22323
22367
  }
@@ -22462,14 +22506,10 @@ bot.command('restart', async ctx => {
22462
22506
  // greeting card shows "Restarted user: /restart from chat" instead
22463
22507
  // of whatever reason the downstream CLI would default to.
22464
22508
  stampUserRestartReason('user: /restart from chat')
22465
- // #3039: /restart is "bounce the session", NOT "clear my model" — the
22466
- // durable override survives every restart and is cleared only by
22467
- // `/model default`. Stamp keep for reason-honesty in the boot notice
22468
- // (absence of intent keeps anyway under the keep-by-default boot).
22469
- {
22470
- const smDir = resolveAgentDirFromEnv()
22471
- if (smDir) writeRelaunchModelIntent(smDir, 'keep', 'user: /restart from chat')
22472
- }
22509
+ // Session-scoped (rev 4): /restart reverts any live /model override to the
22510
+ // configured default. A consume-once `.session-model` carrier (if one was
22511
+ // in flight) was already consumed by its own apply-relaunch, so nothing to
22512
+ // do here — start.sh boots the configured default.
22473
22513
  await sweepBeforeSelfRestart()
22474
22514
  const hostdResp = await tryHostdDispatch(getMyAgentName(), {
22475
22515
  v: 1,
@@ -22628,12 +22668,8 @@ async function handleNewCommand(ctx: Context): Promise<void> {
22628
22668
  // Stamp user attribution so the next greeting shows "Restarted user:
22629
22669
  // /new" / "user: /reset" rather than the downstream CLI default.
22630
22670
  stampUserRestartReason(`user: /${kind} from chat`)
22631
- // /new and /reset start a fresh CONVERSATION, not a fresh model choice:
22632
- // the sticky session-model override KEEPS across them (contract row 7).
22633
- // Boot default is revert, so the keep-intent must land before dispatch.
22634
- if (agentDir != null) {
22635
- writeRelaunchModelIntent(agentDir, 'keep', `user: /${kind} from chat`)
22636
- }
22671
+ // Session-scoped (rev 4): /new and /reset are restarts, so they revert any
22672
+ // live /model override to the configured default — no carrier to preserve.
22637
22673
  await sweepBeforeSelfRestart()
22638
22674
  const hostdResp = await tryHostdDispatch(getMyAgentName(), {
22639
22675
  v: 1,
@@ -23413,6 +23449,46 @@ const throttleTierRunner = createThrottleTierRunner({
23413
23449
  log: (m) => process.stderr.write(`telegram gateway: ${m}\n`),
23414
23450
  })
23415
23451
 
23452
+ // ─── litellm-local 429 notice — side-effect wiring ──────────────────────────
23453
+ // State machine + text + config parsing live in litellm-local-notice.ts
23454
+ // (pure); the sequencing (classification guard → per-agent cooldown →
23455
+ // broadcast + metric) lives in litellm-local-notice-wiring.ts so it is
23456
+ // unit-testable with injected deps. This block only binds the real gateway
23457
+ // dependencies. Deliberately NO broker surface: the litellm-local calm
23458
+ // path's invariant is that account state is never touched.
23459
+ const litellmLocalNoticeRunner = createLitellmLocalNoticeRunner({
23460
+ listNoticeChats: () => loadAccess().allowFrom,
23461
+ sendNotice: (chat_id, markdown) => {
23462
+ // Topic routing — this notice REPLACES the generic operator-event card
23463
+ // for the litellm-local classification, so it must land where that card
23464
+ // would have: supergroup-mode agents route system notifications into the
23465
+ // alerts/admin alias topic ('compact-watchdog' kind, same resolution as
23466
+ // the emitGatewayOperatorEvent broadcast loop), while DM recipients get
23467
+ // a thread-less send (topicForRecipient guards the #2096 "message
23468
+ // thread not found" misrouting class).
23469
+ const noticeTopic = resolveAgentOutboundTopic({ kind: 'compact-watchdog' })
23470
+ const noticeSupergroup = resolveAgentSupergroupChatId()
23471
+ const noticeThread = topicForRecipient({
23472
+ recipientChatId: chat_id,
23473
+ resolvedTopic: noticeTopic,
23474
+ supergroupChatId: noticeSupergroup,
23475
+ })
23476
+ // Status notice, not the user's answer — silence the ping (same posture
23477
+ // as the throttle-tier / fleet-fallback announcements).
23478
+ void swallowingApiCall(
23479
+ // allow-raw-bot-api: wrapped in swallowingApiCall (retry policy)
23480
+ () => bot.api.sendRichMessage(chat_id, richMessage(markdown), {
23481
+ disable_notification: true,
23482
+ ...(noticeThread != null ? { message_thread_id: noticeThread } : {}),
23483
+ }),
23484
+ { chat_id: String(chat_id), verb: 'litellm-local-notice:notify' },
23485
+ )
23486
+ },
23487
+ windowMs: () => parseLitellmNoticeWindowMs(loadAccess().litellmNoticeWindowMs),
23488
+ emitMetric: (event) => emitRuntimeMetric(event),
23489
+ log: (m) => process.stderr.write(`telegram gateway: ${m}\n`),
23490
+ })
23491
+
23416
23492
  /**
23417
23493
  * Broadcast a fleet-fallback FAILURE notice to every authorized chat.
23418
23494
  *
@@ -25551,7 +25627,7 @@ bot.on('callback_query:data', async ctx => {
25551
25627
  // sr-* TARGET tap: switch TO a non-Claude (LiteLLM/OpenRouter) model.
25552
25628
  // Parity with the text `/model sr-*` path — claude's native picker rejects
25553
25629
  // unknown sr-* ids, so an in-place inject can't set them. Carry the token
25554
- // across a graceful restart (the durable `.session-model` override) and
25630
+ // across a graceful restart (the consume-once `.session-model` carrier) and
25555
25631
  // relaunch `claude --model sr-*`. Session-only; reverts to the configured
25556
25632
  // default on the next restart. The sr-* → Claude direction is handled below
25557
25633
  // via the SELECT/alias outcome + isSrToClaudeTransition.
@@ -27960,46 +28036,36 @@ async function shutdown(signal: string): Promise<void> {
27960
28036
  } catch (err) {
27961
28037
  process.stderr.write(`telegram gateway: shutdown.clean_marker_write_failed err=${(err as Error).message}\n`)
27962
28038
  }
27963
- // #3017 — persist a Telegram-set model across a GRACEFUL deploy/restart.
27964
- // Boot default is REVERT, and an EXTERNAL deploy (SIGTERM to PID 1 from
27965
- // `switchroom apply` / `docker compose up`) never routes through
27966
- // triggerSelfRestart, so it stamps no `.relaunch-model-intent` and start.sh
27967
- // drops the user's `/model` choice (the overlord `fable`→`opus` revert on
27968
- // the v0.18.9 roll). A graceful OS-signal shutdown IS a clean, planned
27969
- // bounce — stamp keep-intent for an active `.session-model` override so the
27970
- // chosen model survives and start.sh re-confirms it via `.session-model-alert`
27971
- // on boot. Respect an intent an initiator already stamped (a /restart stamps
27972
- // 'revert' before SIGTERM): only stamp when none exists. A crash routes
27973
- // through the non-OS-signal branch and still reverts (safe side preserved).
27974
- try {
27975
- const smDir = resolveAgentDirFromEnv()
27976
- if (smDir != null) {
27977
- const hasOverride = readSessionModelFile(smDir) != null
27978
- const intentAlreadyStamped = existsSync(join(smDir, RELAUNCH_MODEL_INTENT_FILE))
27979
- if (hasOverride && !intentAlreadyStamped) {
27980
- // The GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX makes this stamp
27981
- // recognisable at the next GATEWAY boot: a gateway-only bounce never
27982
- // runs start.sh, so a leftover stamp with this prefix is cleared at
27983
- // boot (clearStaleGatewayShutdownIntent) instead of lingering to
27984
- // convert a later genuine crash into a "keep" (#3018 finding 4).
27985
- writeRelaunchModelIntent(smDir, 'keep', `${GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX} graceful ${signal} shutdown (deploy/rolling restart) — preserving user-chosen session model`)
27986
- process.stderr.write(`telegram gateway: shutdown.session_model_keep_stamped signal=${signal}\n`)
27987
- }
27988
- }
27989
- } catch (err) {
27990
- process.stderr.write(`telegram gateway: shutdown.session_model_keep_stamp_failed err=${(err as Error).message}\n`)
27991
- }
28039
+ // Session-scoped (rev 4): a graceful deploy/restart REVERTS a live /model
28040
+ // override to the configured default — there is no keep-intent to stamp.
28041
+ // (A live Claude override never had a `.session-model` carrier; an in-flight
28042
+ // sr-* carrier was already consumed by its own apply-relaunch.) A queued —
28043
+ // not-yet-applied — /model IS still persisted just below so it applies as
28044
+ // the agent boots (its apply-relaunch), then reverts on the next restart.
27992
28045
  } else {
27993
28046
  process.stderr.write(`telegram gateway: shutdown.clean_marker_skipped signal=${signal} (crash path — banner will fire on next boot)\n`)
27994
28047
  }
27995
28048
 
27996
28049
  // #3018 finding 3 + #3039: resolve any queued /model|/effort ack cards. The
27997
28050
  // gateway (and with it the in-memory queue) is going away — persist each
27998
- // typed choice to the durable boot carriers (`.session-model` /
27999
- // `.session-effort`) so it still deterministically applies as the agent
28000
- // boots, and edit the ack card to say so. Only an unresolvable menu-tag
28001
- // selection falls back to a re-issue note. Best-effort and time-bounded so
28002
- // a wedged Telegram API can't block shutdown.
28051
+ // typed choice to the consume-once boot carriers (`.session-model` /
28052
+ // `.session-effort`, #3184/#3186) so it still deterministically applies as
28053
+ // the agent boots (that boot consumes the carrier; later restarts revert),
28054
+ // and edit the ack card to say so. Only an unresolvable menu-tag selection
28055
+ // falls back to a re-issue note. Best-effort and time-bounded so a wedged
28056
+ // Telegram API can't block shutdown.
28057
+ //
28058
+ // DELIBERATE (rev 4, #3184 review LOW-3 — applies to BOTH carriers): this
28059
+ // runs on EVERY shutdown path, including crashes (uncaughtException/
28060
+ // unhandledRejection route here), not just the isOsSignal branch above. So
28061
+ // a mid-turn queued /model or /effort + crash can apply on the
28062
+ // crash-recovery boot — technically at odds with a literal "crash reverts"
28063
+ // reading of the session-scoped contract. Intended: it preserves #3178's
28064
+ // "a queued command never silently vanishes" guarantee (the ack card
28065
+ // promised the switch), the model side is gated to offline-trusted tokens
28066
+ // (the effort side is allowlist-gated at write, so a garbage level can't
28067
+ // even cost one crash-boot), and the consume-once carriers bound it to
28068
+ // exactly that one recovery boot — the following restart reverts to config.
28003
28069
  const orphanedCmdActions = pendingCmdShutdownResolutionActions(pendingSessionCommand, escapeHtmlForTg)
28004
28070
  const orphanedCmdEdits = orphanedCmdActions.map(a => ({
28005
28071
  chatId: a.cmd.ackChatId,
@@ -28844,6 +28910,23 @@ void (async () => {
28844
28910
  } catch { /* leave override as-is on a bad read */ }
28845
28911
  }
28846
28912
 
28913
+ // Effort sibling (#3186): start.sh records the EFFECTIVE launched
28914
+ // effort to `.active-session-effort` every boot. Re-hydrate the
28915
+ // in-memory session-effort override so the /effort menu highlight
28916
+ // stays honest after a queued-carrier apply-boot. Only an effort
28917
+ // differing from the configured default counts as an override.
28918
+ const activeEffortPath = join(smAgentDir, '.active-session-effort')
28919
+ if (existsSync(activeEffortPath)) {
28920
+ try {
28921
+ const launchedEffort = readFileSync(activeEffortPath, 'utf8').trim()
28922
+ const configuredEffort = getConfiguredEffortForPersist()
28923
+ sessionEffortOverride =
28924
+ launchedEffort.length > 0 && launchedEffort !== configuredEffort
28925
+ ? launchedEffort
28926
+ : null
28927
+ } catch { /* leave override as-is on a bad read */ }
28928
+ }
28929
+
28847
28930
  const alertPath = join(smAgentDir, '.session-model-alert')
28848
28931
  if (existsSync(alertPath)) {
28849
28932
  let alertText: string | null = null
@@ -29066,6 +29149,14 @@ void (async () => {
29066
29149
  },
29067
29150
  ),
29068
29151
  },
29152
+ // #3084 follow-up: the feed's send/edit adapters transit the send
29153
+ // gate, which SHEDS (resolves undefined) any call made during an
29154
+ // open flood window. Give the feed the SAME on-disk window probe
29155
+ // robustApiCall + the held-card sweep read, so a running/first-
29156
+ // paint tick parks in cooldown instead of re-firing a shed send
29157
+ // every ~6s for the whole ban (the worker-feed shed-contract bug:
29158
+ // 565 `sent.message_id` crashes in one 6h ban).
29159
+ floodWaitRemainingMs: probeFloodWaitRemainingMs,
29069
29160
  log: (msg) => process.stderr.write(`telegram gateway: ${msg}\n`),
29070
29161
  })
29071
29162
  subagentWatcher = startSubagentWatcher({