switchroom 0.19.23 → 0.19.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/dist/agent-scheduler/index.js +18 -7
  2. package/dist/auth-broker/index.js +117 -33
  3. package/dist/cli/autoaccept-poll.js +0 -1
  4. package/dist/cli/drive-write-pretool.mjs +5 -0
  5. package/dist/cli/ms-365-write-pretool.mjs +5 -0
  6. package/dist/cli/notion-write-pretool.mjs +18 -6
  7. package/dist/cli/switchroom.js +2916 -1481
  8. package/dist/host-control/main.js +116 -34
  9. package/dist/vault/approvals/kernel-server.js +115 -33
  10. package/dist/vault/broker/server.js +281 -76
  11. package/examples/switchroom.yaml +1 -1
  12. package/package.json +1 -1
  13. package/profiles/_base/start.sh.hbs +52 -13
  14. package/profiles/_shared/dev-protocol.md.hbs +3 -4
  15. package/skills/dev-protocol/SKILL.md +22 -15
  16. package/skills/switchroom-health/SKILL.md +19 -0
  17. package/skills/switchroom-release/SKILL.md +2 -1
  18. package/skills/switchroom-status/SKILL.md +1 -1
  19. package/telegram-plugin/auth-snapshot-format.ts +9 -2
  20. package/telegram-plugin/dist/gateway/gateway.js +6925 -6734
  21. package/telegram-plugin/gateway/gateway.ts +34 -35
  22. package/telegram-plugin/gateway/latest-turn-lookup.ts +60 -0
  23. package/telegram-plugin/gateway/outbound-send-path.ts +53 -21
  24. package/telegram-plugin/gateway/subagent-handback-marker.ts +1 -1
  25. package/telegram-plugin/gateway/turn-end.ts +1 -1
  26. package/telegram-plugin/quota-bar-format.ts +4 -1
  27. package/telegram-plugin/reply-owner-resolve.ts +110 -9
  28. package/telegram-plugin/send-gate-degraded.test.ts +45 -16
  29. package/telegram-plugin/send-gate.ts +185 -24
  30. package/telegram-plugin/tests/activity-card-send-gate.test.ts +9 -9
  31. package/telegram-plugin/tests/auth-snapshot-format.test.ts +42 -0
  32. package/telegram-plugin/tests/latest-turn-lookup.test.ts +77 -0
  33. package/telegram-plugin/tests/narrative-lane-golden.test.ts +23 -1
  34. package/telegram-plugin/tests/quota-bar-format.test.ts +50 -0
  35. package/telegram-plugin/tests/reply-owner-resolve.test.ts +531 -0
  36. package/telegram-plugin/tests/secret-detect-false-positives.test.ts +1 -1
  37. package/telegram-plugin/tests/send-reply-golden.test.ts +296 -28
  38. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +134 -28
  39. package/telegram-plugin/tests/stream-render-golden.test.ts +25 -3
  40. package/vendor/hindsight-memory/scripts/lib/config.py +61 -19
  41. package/vendor/hindsight-memory/scripts/lib/content.py +376 -1
  42. package/vendor/hindsight-memory/scripts/lib/english_words.txt +10799 -0
  43. package/vendor/hindsight-memory/scripts/recall.py +503 -252
  44. package/vendor/hindsight-memory/scripts/tests/test_recall_bank_slots.py +509 -0
  45. package/vendor/hindsight-memory/scripts/tests/test_recall_envelope_strip_telemetry.py +22 -5
  46. package/vendor/hindsight-memory/scripts/tests/test_recall_error_text.py +147 -0
  47. package/vendor/hindsight-memory/scripts/tests/test_recall_hook_budget.py +266 -0
  48. package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +0 -401
  49. package/vendor/hindsight-memory/scripts/tests/test_recall_no_lexical_gate.py +261 -0
  50. package/vendor/hindsight-memory/scripts/tests/test_recall_query_shaping.py +473 -0
  51. package/vendor/hindsight-memory/scripts/tests/test_recall_transcript_fallback.py +25 -8
  52. package/vendor/hindsight-memory/tests/test_content.py +218 -0
@@ -16,12 +16,16 @@
16
16
  * collapse; the latest snapshot lands floor-paced
17
17
  * - the floor is per-message (stream A does not delay stream B)
18
18
  * - the gate's no-op skip drops a repeat payload for the same message
19
- * - an open flood window sheds draft edits with ZERO API calls, and the
20
- * stream recovers with full state after the window closes
21
- * - a shed draft is NOT recorded as delivered — a later flush of the
22
- * SAME text (the completed answer) still lands
19
+ * - an open flood window COALESCES draft edits with ZERO API calls, and the
20
+ * newest state lands once the window closes (#3716 — cosmetic edits are
21
+ * never shed; the last edit of a burst is the one still on screen, so
22
+ * dropping it stranded the message on a stale body)
23
+ * - a draft held through a window still renders the completed answer — a
24
+ * later flush of the SAME text is then a benign no-op, not a loss
23
25
  * - the finalize flush is `critical`: never shed; waits out a short
24
26
  * window; fails fast (structured, logged) on a long one
27
+ * - shed-honesty (F2+F3) remains wired for any `SEND_GATE_SHED` the retry
28
+ * policy does return, pinned directly rather than through the gate
25
29
  * - regression pin: the controller passes messageId / editPayload /
26
30
  * priorityClass through the retry policy on every edit
27
31
  *
@@ -32,7 +36,7 @@
32
36
  */
33
37
  import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'
34
38
  import { createStreamController, type RetryPolicy } from '../stream-controller.js'
35
- import { createSendGate, type Clock, type SendGateConfig } from '../send-gate.js'
39
+ import { createSendGate, SEND_GATE_SHED, type Clock, type SendGateConfig } from '../send-gate.js'
36
40
  import { isFloodWaitActiveError } from '../retry-api-call.js'
37
41
  import { renderOutboundChunks } from '../render/rich-render.js'
38
42
  import { createMockBot, installBotResetHook } from './bot-api.harness.js'
@@ -221,7 +225,7 @@ describe('stream-controller × send gate (#3110)', () => {
221
225
  expect(bot.api.editMessageText).toHaveBeenCalledTimes(1)
222
226
  })
223
227
 
224
- it('open flood window: draft edits shed with ZERO API calls; full state lands after it closes', async () => {
228
+ it('open flood window: draft edits COALESCE with ZERO API calls; newest state lands after it closes (#3716)', async () => {
225
229
  const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
226
230
  const stream = createStreamController({
227
231
  bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
@@ -233,17 +237,24 @@ describe('stream-controller × send gate (#3110)', () => {
233
237
  await flush()
234
238
  void stream.update('draft b')
235
239
  await tick()
236
- // Both drafts shed as cosmetic — nothing reached the API.
240
+ // Flood safety is unchanged — still ZERO API calls while the window is
241
+ // open. What changed is the mechanism: the drafts are HELD (last-write-
242
+ // wins), not discarded, so no state is lost.
237
243
  expect(bot.api.editMessageText).not.toHaveBeenCalled()
238
- expect(gate.stats().global.shed).toBe(2)
244
+ expect(gate.stats().global.shed).toBe(0)
239
245
 
240
- await clock.advance(30_000) // window closes
246
+ await clock.advance(30_000) // window closes → the held draft lands
241
247
  void stream.update('draft c — full state')
242
248
  await tick()
243
- expect(editBodies()).toEqual(['draft c — full state'])
249
+ await clock.advance(1_500) // clear the per-message edit floor
250
+
251
+ // The guarantee is the newest state reaches the screen, never that some
252
+ // intermediate was dropped to get there.
253
+ expect(editBodies().at(-1)).toBe('draft c — full state')
254
+ expect(gate.stats().global.shed).toBe(0)
244
255
  })
245
256
 
246
- it('a shed draft is NOT recorded as delivered: a later finalize of the SAME text still lands', async () => {
257
+ it('a draft held through a window still renders: the completed answer reaches the screen exactly once', async () => {
247
258
  const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
248
259
  const stream = createStreamController({
249
260
  bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
@@ -253,12 +264,16 @@ describe('stream-controller × send gate (#3110)', () => {
253
264
  void stream.update('the completed answer')
254
265
  await flush()
255
266
  expect(bot.api.editMessageText).not.toHaveBeenCalled()
256
- expect(gate.stats().global.shed).toBe(1)
267
+ expect(gate.stats().global.shed).toBe(0)
257
268
 
258
- await clock.advance(30_000)
259
- // stream_reply done=true with the same text → finalize(text). If the shed
260
- // draft had been recorded as on-screen, draft-stream's dedupe would skip
261
- // this flush and the completed answer would never render.
269
+ await clock.advance(30_000) // window closes → the held draft lands
270
+
271
+ // stream_reply done=true with the same text → finalize(text). Pre-#3716
272
+ // the draft was SHED here, and the guarantee was that the stream must not
273
+ // record it as on-screen so this flush could re-deliver it. Now the draft
274
+ // is never dropped, so it renders on its own and the identical finalize is
275
+ // a benign no-op. Either way the user sees the completed answer — and now
276
+ // it costs one API call instead of two.
262
277
  await stream.finalize('the completed answer')
263
278
  expect(editBodies()).toEqual(['the completed answer'])
264
279
  })
@@ -277,7 +292,7 @@ describe('stream-controller × send gate (#3110)', () => {
277
292
  gate.openFloodWindow('global', clock.now() + 30_000) // short: <= 60s fail-fast ceiling
278
293
  void stream.update('draft while banned')
279
294
  await flush()
280
- expect(bot.api.editMessageText).not.toHaveBeenCalled() // draft shed
295
+ expect(bot.api.editMessageText).not.toHaveBeenCalled() // draft held, not sent
281
296
 
282
297
  const fin = stream.finalize('the answer')
283
298
  await flush()
@@ -286,9 +301,12 @@ describe('stream-controller × send gate (#3110)', () => {
286
301
 
287
302
  await clock.advance(30_000)
288
303
  await fin
304
+ // The critical finalize coalesced ONTO the held draft and upgraded its
305
+ // class, so the whole burst resolves as a single send carrying the final
306
+ // body — the draft is superseded rather than dropped.
289
307
  expect(editBodies()).toEqual(['the answer'])
290
308
  expect(editTimes).toEqual([30_000])
291
- expect(gate.stats().global.shed).toBe(1) // only the draft
309
+ expect(gate.stats().global.shed).toBe(0) // nothing is shed any more
292
310
  })
293
311
 
294
312
  it('finalize under a LONG window fails fast (structured FLOOD_WAIT_ACTIVE, logged) — no API call, no hang', async () => {
@@ -316,6 +334,40 @@ describe('stream-controller × send gate (#3110)', () => {
316
334
  }
317
335
  })
318
336
 
337
+ /**
338
+ * REGRESSION PIN for the trap #3716 opened. Once cosmetic edits stopped
339
+ * shedding they began OCCUPYING the driver, and draft-stream serializes its
340
+ * own flushes — so a draft parked behind a 6h ban held the finalize upstream
341
+ * of the gate, where the fail-fast path could never see it. `failedFast` went
342
+ * to 0 and the reply path wedged for the length of the ban: the exact failure
343
+ * the gate was built to eliminate, reintroduced by the fix for a different
344
+ * one. The preceding test does NOT catch this — it finalizes with no draft in
345
+ * flight.
346
+ */
347
+ it('a draft parked behind a LONG window never wedges the finalize behind it (#3716)', async () => {
348
+ const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
349
+ const logs: string[] = []
350
+ const stream = createStreamController({
351
+ bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
352
+ log: (m) => logs.push(m),
353
+ })
354
+
355
+ gate.openFloodWindow('global', clock.now() + 21_397_000) // the 2026-07-12 ban: ~5.9h
356
+ void stream.update('draft while banned')
357
+ await flush()
358
+ expect(gate.stats().global.shed).toBe(0) // held, not dropped
359
+
360
+ // The cosmetic draft settles its caller as soon as it is queued, so the
361
+ // finalize reaches the gate. If it did not, this await never returns.
362
+ const fin = stream.finalize('the answer')
363
+ await tick()
364
+ await fin
365
+
366
+ expect(gate.stats().global.failedFast).toBe(1)
367
+ expect(bot.api.editMessageText).not.toHaveBeenCalled()
368
+ expect(logs.some((m) => m.includes('FLOOD_WAIT_ACTIVE'))).toBe(true)
369
+ })
370
+
319
371
  it('REGRESSION PIN: every edit passes messageId / editPayload / priorityClass to the retry policy', async () => {
320
372
  // Spy retry with NO gate — pins exactly what the controller hands to
321
373
  // robustApiCall (the #3110 bypass was these fields being absent).
@@ -474,7 +526,7 @@ describe('stream-controller × send gate (#3110)', () => {
474
526
  expect(bot.api.editMessageText.mock.calls.length).toBe(editCalls)
475
527
  })
476
528
 
477
- it('a shed TAIL is not recorded as delivered: argument-less finalize() re-flushes and lands it (F2+F3)', async () => {
529
+ it('a TAIL suppressed by a msg-scoped window is HELD, not lost: the completed answer lands when it closes', async () => {
478
530
  const { clock, gate, retry } = makeGatedRetry({
479
531
  editFloorMs: 1500,
480
532
  globalPerSec: 1000, globalBurst: 100, perChatPerSec: 1000, perChatBurst: 100,
@@ -495,27 +547,81 @@ describe('stream-controller × send gate (#3110)', () => {
495
547
  const lastTailId = anchorId + pieceCount - 1
496
548
 
497
549
  // Flood window scoped to the LAST tail message only (H1 msg-edit scope):
498
- // its edit sheds; the anchor and other pieces are unaffected.
550
+ // its edit is suppressed; the anchor and other pieces are unaffected.
499
551
  gate.openFloodWindow(`msg-edit:1:${lastTailId}`, clock.now() + 30_000)
500
552
 
501
553
  void stream.update(b2)
502
554
  await tick()
503
- // The changed piece is the suppressed tail → shed, zero edits landed on
504
- // it; the flush is reported shed and the snapshot preserved (F2), NOT
505
- // recorded as delivered (F3).
555
+ // The changed piece is the suppressed tail → zero edits land on it while
556
+ // the window is open. Pre-#3716 it was SHED and the completed answer only
557
+ // survived because the stream refused to record it as delivered; now the
558
+ // edit is held by the gate, so the content is safe by construction.
506
559
  const tailEdits = () =>
507
560
  bot.api.editMessageText.mock.calls.filter(([, id]) => id === lastTailId)
508
561
  expect(tailEdits()).toHaveLength(0)
509
- expect(gate.stats().global.shed).toBe(1)
510
- expect(logs.some((m) => m.includes('shed by send gate'))).toBe(true)
562
+ expect(gate.stats().global.shed).toBe(0)
563
+ expect(logs.some((m) => m.includes('shed by send gate'))).toBe(false)
511
564
 
512
- // Window closes; the gateway-style ARGUMENT-LESS finalize (the
513
- // disconnect-flush / turn-end cleanup path) must re-deliver the shed
514
- // snapshot — pre-F2 the content was silently lost here.
565
+ // Window closes → the held tail edit lands on its own. The gateway-style
566
+ // ARGUMENT-LESS finalize (disconnect-flush / turn-end cleanup) is then a
567
+ // no-op rather than a rescue.
515
568
  await clock.advance(31_000)
516
569
  await stream.finalize()
517
570
  expect(tailEdits()).toHaveLength(1)
518
571
  const [, , tailBody] = tailEdits()[0]
519
572
  expect(String(tailBody)).toContain('tail v2')
520
573
  })
574
+
575
+ /**
576
+ * #3716 removed the gate's cosmetic-EDIT shed, so no edit the stream makes
577
+ * can return `SEND_GATE_SHED` any more. The sentinel is still the contract
578
+ * for non-edit cosmetic sends, and the controller's F2/F3 shed-honesty
579
+ * handling is the guard if any edit path is ever re-tagged — so pin it
580
+ * directly against the retry seam instead of through the gate, where it
581
+ * would silently rot into an assertion about behaviour that cannot occur.
582
+ */
583
+ it('F2+F3 shed-honesty is still wired: a SHED tail is not recorded as delivered and re-flushes', async () => {
584
+ const base = ('a_b_c_d_e ').repeat(3000)
585
+ const b1 = `${base}tail v1`
586
+ const b2 = `${base}tail v2 — the completed answer`
587
+ const pieceCount = renderOutboundChunks(b1).length
588
+ expect(pieceCount).toBeGreaterThan(1)
589
+
590
+ // Shed exactly one message id, chosen after the first flush assigns ids.
591
+ let shedId: number | null = null
592
+ const retry: RetryPolicy = async (fn, opts) => {
593
+ if (shedId != null && opts?.messageId === shedId) {
594
+ return SEND_GATE_SHED as never
595
+ }
596
+ return await fn()
597
+ }
598
+
599
+ const logs: string[] = []
600
+ const stream = createStreamController({
601
+ bot, chatId: '1', throttleMs: 250, retry, log: (m) => logs.push(m),
602
+ })
603
+ void stream.update(b1)
604
+ await flush()
605
+ const anchorId = stream.getMessageId() as number
606
+ shedId = anchorId + pieceCount - 1
607
+
608
+ const tailEdits = () =>
609
+ bot.api.editMessageText.mock.calls.filter(([, id]) => id === shedId)
610
+
611
+ void stream.update(b2)
612
+ await tick()
613
+ expect(tailEdits()).toHaveLength(0)
614
+ expect(logs.some((m) => m.includes('shed by send gate'))).toBe(true)
615
+
616
+ // The shed piece must NOT have been recorded as on screen: an
617
+ // argument-less finalize re-delivers the preserved snapshot.
618
+ shedId = null
619
+ await stream.finalize()
620
+ const landed = bot.api.editMessageText.mock.calls.filter(
621
+ ([, id]) => id === anchorId + pieceCount - 1,
622
+ )
623
+ expect(landed).toHaveLength(1)
624
+ const [, , tailBody] = landed[0]
625
+ expect(String(tailBody)).toContain('tail v2')
626
+ })
521
627
  })
@@ -39,6 +39,28 @@ import {
39
39
  TYPING_REFRESH_MS,
40
40
  } from '../typing-emitter.js'
41
41
  import type { CurrentTurn } from '../gateway/gateway.js'
42
+ import type { ReplyOwnerTier } from '../reply-owner-resolve.js'
43
+
44
+ /** The owner-resolution shape `resolveReplyOwnerTurn` returns, including the
45
+ * candidate set the content-gate bypass corroborates against. These fixtures
46
+ * never exercise the supersede path, so the candidates mirror the resolved turn
47
+ * (the corroborated shape) with no override needed. */
48
+ function ownerRes(turn: CurrentTurn | null, tier: ReplyOwnerTier) {
49
+ const id = turn?.turnId ?? null
50
+ return {
51
+ turn,
52
+ tier,
53
+ candidates: {
54
+ liveTurnId: tier === 'live' ? id : null,
55
+ originTurnId: null,
56
+ quotedTurnId: null,
57
+ latestEndedTurnId: id,
58
+ latestEndedAgeMs: 1_000,
59
+ latestEndedTtlMs: 60_000,
60
+ },
61
+ }
62
+ }
63
+
42
64
 
43
65
  const CHAT = '1001'
44
66
 
@@ -251,7 +273,7 @@ function makeSendReplyDeps(dedup: OutboundDedupCache, sharedSupersede?: FlushedT
251
273
  assertSendable: () => {},
252
274
  statusKey: key,
253
275
  streamKey: key,
254
- resolveReplyOwnerTurn: () => ({ turn: null, tier: 'none' as const }),
276
+ resolveReplyOwnerTurn: () => ownerRes(null, 'none'),
255
277
  getLastSubagentHandbackAt: () => null,
256
278
  findTurnByOriginId: () => null,
257
279
  findTurnByQuotedMessageId: () => null,
@@ -443,7 +465,7 @@ describe('F3 — flush record() → same-turn reworded reply collapse (end-to-en
443
465
  // The model's REAL reply lands late with a REWORDED version of the same
444
466
  // answer: no live turn, latest-ended tier, NO handback in flight (CASE A).
445
467
  const s = makeSendReplyDeps(new OutboundDedupCache(), supersede)
446
- s.deps.resolveReplyOwnerTurn = () => ({ turn, tier: 'latest-ended' as const })
468
+ s.deps.resolveReplyOwnerTurn = () => ownerRes(turn, 'latest-ended')
447
469
  // (getLastSubagentHandbackAt returns null in the base deps → own answer.)
448
470
 
449
471
  const res = await sendReply(s.deps, req(REWORDED))
@@ -496,7 +518,7 @@ describe('F5 — take()-before-record() interleaving delivers exactly one messag
496
518
  ).toBe('no-record') // record genuinely not written yet
497
519
 
498
520
  const s = makeSendReplyDeps(new OutboundDedupCache(), supersede)
499
- s.deps.resolveReplyOwnerTurn = () => ({ turn, tier: 'latest-ended' as const })
521
+ s.deps.resolveReplyOwnerTurn = () => ownerRes(turn, 'latest-ended')
500
522
  // The same answer landing again in the race window → latch backstop suppresses.
501
523
  const res = await sendReply(s.deps, req(ANSWER))
502
524
 
@@ -29,21 +29,23 @@ DEFAULTS = {
29
29
  # formatting. Set to 0 (or any non-positive value) to disable the cap
30
30
  # and inject everything Hindsight returns.
31
31
  "recallMaxMemories": 12,
32
- # Switchroom-local: minimum lexical (containment) overlap between the
33
- # user's query terms and a memory's text terms. Memories below this
34
- # threshold are dropped before formatting. 0.0 disables the gate
35
- # (current behaviour: inject everything Hindsight returns up to the
36
- # count cap). NOTE: Hindsight's HTTP recall API DOES return per-result
37
- # relevance scores (`scores.final`, plus `.semantic`/`.keyword`/
38
- # `.reranker`) — verified at runtime — and recall.py now reads and
39
- # sorts the merged set by `scores.final`. This lexical gate is a
40
- # separate quality filter layered on top — see #475. The metric is
41
- # containment, `|Q n M| / |M|`, not Jaccard: dividing by the union made
42
- # the score a function of prompt length rather than relevance — see
43
- # #3541 and recall.py's design note. At the 0.10 fleet default this is
44
- # close to a passthrough (a <=10-token memory clears it on one shared
45
- # word); precision is the engine rerank's job, not this gate's.
46
- "recallMinOverlap": 0.0,
32
+ # Switchroom-local: per-bank slot FLOORS inside `recallMaxMemories`. The
33
+ # merged multi-bank set is sorted globally by `scores.final` and then
34
+ # head-sliced, which is winner-take-all across banks: when both banks return
35
+ # more candidates than the cap, one bank's score distribution can fill every
36
+ # slot and the agent gets a dossier about its operator with none of its own
37
+ # working memory. These are FLOORS, not quotas: each side gets at most this
38
+ # many slots, only if it has that many results, and only up to HALF the cap
39
+ # between them — the other half is always awarded on pure global relevance,
40
+ # so composition still moves with the scores. 0 disables reservation for
41
+ # that side (the pure pre-fix head-slice). Env:
42
+ # HINDSIGHT_RECALL_OWN_BANK_MIN_SLOTS /
43
+ # HINDSIGHT_RECALL_ADDITIONAL_BANK_MIN_SLOTS. See recall.py's
44
+ # `_reserve_bank_slots` and `_reservable_slots`. Vendor default is 0/0
45
+ # (off); switchroom's scaffold opts in with 2 own / 1 additional against the
46
+ # cap of 6 its fleet actually deploys.
47
+ "recallOwnBankMinSlots": 0,
48
+ "recallAdditionalBankMinSlots": 0,
47
49
  "recallTypes": ["world", "experience"],
48
50
  # Switchroom-local: when True (default; Ken-approved ON) recall biases
49
51
  # toward synthesized `observation`-tier facts. Escape hatch: pin off via
@@ -91,6 +93,29 @@ DEFAULTS = {
91
93
  # by recallTranscriptTailBytes so the added per-turn read stays O(1).
92
94
  "recallContextTurns": 2,
93
95
  "recallMaxQueryChars": 800,
96
+ # Switchroom #3757 — BM25 term budget for the query put on the wire.
97
+ # `recallMaxQueryChars` bounds CHARACTERS; the server's keyword arm costs
98
+ # per DISTINCT TERM, because it OR-joins every token into one tsquery and
99
+ # Postgres native FTS ranks the entire matched set before the top-60
100
+ # heapsort. An 800-char composed query is ~96 distinct terms and matched
101
+ # 119,510 rows on the live `overlord` bank — 14.0s for the 3-arm UNION,
102
+ # and up to 94s under load, past the
103
+ # per-bank client timeout, so the agent got NOTHING on 96.8% of its
104
+ # own-bank recalls in the 7 days to 2026-07-27. 24 terms measures at
105
+ # 48,433 rows / 2.7s on the same bank while keeping the high-signal terms
106
+ # of the latest turn. Selection is recency-first (latest turn beats prior
107
+ # context), then by a selectivity proxy — see lib/content.shape_recall_query.
108
+ # 0 disables shaping entirely (rollback lever).
109
+ # Operator knob: `memory.recall.query_max_tokens` in switchroom.yaml.
110
+ "recallQueryMaxTokens": 24,
111
+ # Switchroom #3757 — extra terms to drop from the BM25 query on top of the
112
+ # built-in English stopword list. For BANK-SPECIFIC high-document-frequency
113
+ # words a generic stoplist cannot know about: on `overlord`, `switchroom`
114
+ # matches 27,090 of 135,443 rows (20%) and `agent` 26,496 (20%), purely
115
+ # because that is what the corpus is about. Empty by default — an operator
116
+ # sets it per-agent after reading `switchroom memory recall-log <agent>`.
117
+ # Operator knob: `memory.recall.query_stop_terms` in switchroom.yaml.
118
+ "recallQueryStopTerms": [],
94
119
  # Switchroom hindsight-leverage A2 (PR2) — latency bound for the multi-turn
95
120
  # composition. With recallContextTurns>1 now the default, EVERY recall reads
96
121
  # the transcript to slice the last N human turns. A long session's .jsonl can
@@ -204,6 +229,18 @@ DEFAULTS = {
204
229
  # can never push the hook past its ceiling. Slots still unfinished when the
205
230
  # deadline elapses are abandoned (daemon threads) and marked timed_out.
206
231
  "recallParallelDeadlineSeconds": 10,
232
+ # Switchroom #3757 — per-bank HTTP read timeout (seconds) for one recall
233
+ # request. Was a hardcoded `timeout=8` in recall.py, which made it BOTH the
234
+ # binding constraint on a slow bank AND un-tunable without hand-editing the
235
+ # installed plugin — and a hand-edit does not survive `switchroom apply`,
236
+ # which re-copies the plugin from `vendor/hindsight-memory` (that revert is
237
+ # exactly what put the 8s literal back on 2026-07-27). 12s matches the
238
+ # UserPromptSubmit hook ceiling in hooks.json; the shared
239
+ # `recallParallelDeadlineSeconds` (10s) is the tighter outer guard in the
240
+ # default configuration, so this is a per-request safety net rather than
241
+ # the primary bound. Non-positive values fall back to the default.
242
+ # Operator knob: `memory.recall.request_timeout_seconds` in switchroom.yaml.
243
+ "recallRequestTimeoutSeconds": 12,
207
244
  # Switchroom hindsight-leverage E1 / PR8 (#3369) — bounded transcript-grep
208
245
  # fallback. Boot reconciliation (reconcile_tail.py) closes the crash-loss
209
246
  # window at the NEXT SessionStart, but between an abrupt kill and that boot,
@@ -282,10 +319,11 @@ ENV_OVERRIDES = {
282
319
  # agents.<name>.memory.recall.max_memories (cascading through
283
320
  # defaults.memory.recall.max_memories) when present in switchroom.yaml.
284
321
  "HINDSIGHT_RECALL_MAX_MEMORIES": ("recallMaxMemories", int),
285
- # Switchroom-local: lexical-overlap threshold (#475). Float in
286
- # [0.0, 1.0]. Set by start.sh from agents.<name>.memory.recall.min_overlap
287
- # (cascading through defaults). 0.0 = off (current behaviour).
288
- "HINDSIGHT_RECALL_MIN_OVERLAP": ("recallMinOverlap", float),
322
+ # Switchroom-local: per-bank slot floors inside the count cap. Set by
323
+ # start.sh from agents.<name>.memory.recall.own_bank_min_slots /
324
+ # .additional_bank_min_slots (cascading through defaults). 0 = off.
325
+ "HINDSIGHT_RECALL_OWN_BANK_MIN_SLOTS": ("recallOwnBankMinSlots", int),
326
+ "HINDSIGHT_RECALL_ADDITIONAL_BANK_MIN_SLOTS": ("recallAdditionalBankMinSlots", int),
289
327
  # Switchroom-local: recall fact types (comma-separated). Set by start.sh
290
328
  # from agents.<name>.memory.recall.types only when the operator overrode
291
329
  # the switchroom default (world,experience,observation) — i.e. the
@@ -322,6 +360,10 @@ ENV_OVERRIDES = {
322
360
  "HINDSIGHT_RECALL_TRANSCRIPT_FALLBACK_MAX_CHARS": ("recallTranscriptFallbackMaxChars", int),
323
361
  "HINDSIGHT_RECALL_TRANSCRIPT_FALLBACK_DEADLINE_MS": ("recallTranscriptFallbackDeadlineMs", int),
324
362
  "HINDSIGHT_RECALL_MAX_QUERY_CHARS": ("recallMaxQueryChars", int),
363
+ # Switchroom #3757 — BM25 query shaping + per-request timeout.
364
+ "HINDSIGHT_RECALL_QUERY_MAX_TOKENS": ("recallQueryMaxTokens", int),
365
+ "HINDSIGHT_RECALL_QUERY_STOP_TERMS": ("recallQueryStopTerms", list),
366
+ "HINDSIGHT_RECALL_REQUEST_TIMEOUT_SECONDS": ("recallRequestTimeoutSeconds", int),
325
367
  "HINDSIGHT_RECALL_CONTEXT_TURNS": ("recallContextTurns", int),
326
368
  # Switchroom hindsight-leverage A2 — byte-tail bound for the multi-turn
327
369
  # transcript read (0 = read whole file / rollback lever).