switchroom 0.19.24 → 0.19.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/dist/agent-scheduler/index.js +15 -6
  2. package/dist/auth-broker/index.js +85 -27
  3. package/dist/cli/autoaccept-poll.js +0 -1
  4. package/dist/cli/drive-write-pretool.mjs +5 -0
  5. package/dist/cli/ms-365-write-pretool.mjs +5 -0
  6. package/dist/cli/notion-write-pretool.mjs +15 -5
  7. package/dist/cli/switchroom.js +2690 -1415
  8. package/dist/host-control/main.js +84 -28
  9. package/dist/vault/approvals/kernel-server.js +83 -27
  10. package/dist/vault/broker/server.js +249 -70
  11. package/examples/switchroom.yaml +1 -1
  12. package/package.json +1 -1
  13. package/profiles/_base/start.sh.hbs +52 -13
  14. package/skills/switchroom-health/SKILL.md +19 -0
  15. package/skills/switchroom-status/SKILL.md +1 -1
  16. package/telegram-plugin/auth-snapshot-format.ts +9 -2
  17. package/telegram-plugin/dist/gateway/gateway.js +5757 -5689
  18. package/telegram-plugin/quota-bar-format.ts +4 -1
  19. package/telegram-plugin/tests/auth-snapshot-format.test.ts +42 -0
  20. package/telegram-plugin/tests/quota-bar-format.test.ts +50 -0
  21. package/telegram-plugin/tests/secret-detect-false-positives.test.ts +1 -1
  22. package/vendor/hindsight-memory/scripts/lib/config.py +61 -19
  23. package/vendor/hindsight-memory/scripts/lib/content.py +376 -1
  24. package/vendor/hindsight-memory/scripts/lib/english_words.txt +10799 -0
  25. package/vendor/hindsight-memory/scripts/recall.py +503 -252
  26. package/vendor/hindsight-memory/scripts/tests/test_recall_bank_slots.py +509 -0
  27. package/vendor/hindsight-memory/scripts/tests/test_recall_envelope_strip_telemetry.py +22 -5
  28. package/vendor/hindsight-memory/scripts/tests/test_recall_error_text.py +147 -0
  29. package/vendor/hindsight-memory/scripts/tests/test_recall_hook_budget.py +266 -0
  30. package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +0 -401
  31. package/vendor/hindsight-memory/scripts/tests/test_recall_no_lexical_gate.py +261 -0
  32. package/vendor/hindsight-memory/scripts/tests/test_recall_query_shaping.py +473 -0
  33. package/vendor/hindsight-memory/scripts/tests/test_recall_transcript_fallback.py +25 -8
  34. package/vendor/hindsight-memory/tests/test_content.py +218 -0
@@ -737,19 +737,58 @@ export HINDSIGHT_RECALL_MAX_MEMORIES={{hindsightRecallMaxMemories}}
737
737
  {{#if (isNumber hindsightRecallCacheTtlSecs)}}
738
738
  export HINDSIGHT_RECALL_CACHE_TTL_SECS={{hindsightRecallCacheTtlSecs}}
739
739
  {{/if}}
740
- # Lexical-overlap relevance gate (#475). Drops memories whose containment
741
- # overlap (|Q n M| / |M|, see #3541) with the user's query is
742
- # below this threshold (range 0.0–1.0).
743
- # Plugin default is 0.0 (gate disabled); export only when the operator
744
- # overrode it via memory.recall.min_overlap in switchroom.yaml. Use 0.10:
745
- # it is a near-passthrough floor, not a precision control. Values at or
746
- # above 0.20 measurably starve recall — on production replay 0.20 leaves
747
- # ~41.9% of turns with NO memories at all, re-creating the bug #3541
748
- # fixed. Observe `overlap_dropped` via
749
- # `switchroom memory recall-log <agent>`.
750
- {{#if (isNumber hindsightRecallMinOverlap)}}
751
- export HINDSIGHT_RECALL_MIN_OVERLAP={{hindsightRecallMinOverlap}}
752
- {{/if}}
740
+ # BM25 term budget for the recall query (#3757). The hook composes a query
741
+ # from the last 2 turns and Hindsight OR-joins EVERY token into one
742
+ # to_tsquery; Postgres native FTS cannot top-k from the GIN index, so it
743
+ # ranks the entire matched set before the top-60 heapsort. Measured on the
744
+ # live `overlord` bank (135,565 rows, 3 fact-type arms): 96 distinct terms
745
+ # = 119,510 rows ranked = 14.0s, versus 24 terms = 48,433 rows = 2.7s.
746
+ # memory.recall.query_max_tokens; 0 disables shaping (rollback lever).
747
+ # Exported UNCONDITIONALLY with a concrete value — see the note on
748
+ # HINDSIGHT_RECALL_REQUEST_TIMEOUT_SECONDS below for why an unexported or
749
+ # empty knob is a silent-shadowing bug rather than a no-op.
750
+ export HINDSIGHT_RECALL_QUERY_MAX_TOKENS={{hindsightRecallQueryMaxTokens}}
751
+ # Bank-specific high-document-frequency terms to drop from the BM25 query
752
+ # (#3757, memory.recall.query_stop_terms). Postgres' `english` config
753
+ # already strips generic English stopwords; this is for words a generic
754
+ # stoplist cannot know are useless in THIS bank — on `overlord`,
755
+ # `switchroom` matched 27,075 rows (20%) and `agent` a similar share.
756
+ # JSON array (the plugin also accepts a comma-separated string), single-
757
+ # quoted so the JSON's double quotes reach the plugin intact. The schema
758
+ # constrains each term to `[\w./-]+`, so it cannot contain a quote. Exported
759
+ # unconditionally: with no operator override this is the literal `[]`, which
760
+ # `_cast_env` parses to an empty list and assigns, so the env still wins over
761
+ # a stale claude-code.json.
762
+ export HINDSIGHT_RECALL_QUERY_STOP_TERMS='{{hindsightRecallQueryStopTermsJson}}'
763
+ # Per-bank HTTP timeout for a single recall request (#3757). Was hardcoded
764
+ # at 8s in the plugin, which fired on 96.8% of overlord's own-bank recalls
765
+ # and returned the model nothing at all. Plugin default is now 12s; the
766
+ # shared multi-bank deadline (recallParallelDeadlineSeconds, default 10s)
767
+ # is normally the tighter bound, so this is the safety net, not the budget.
768
+ # Exported UNCONDITIONALLY, always carrying a concrete value (#3774 defect
769
+ # class). The plugin resolves DEFAULTS -> settings.json -> claude-code.json ->
770
+ # env, so a knob left unexported — or exported empty, which `_cast_env` turns
771
+ # into a skipped assignment — lets a stale hand-written claude-code.json shadow
772
+ # the shipped default with every test still green.
773
+ export HINDSIGHT_RECALL_REQUEST_TIMEOUT_SECONDS={{hindsightRecallRequestTimeoutSeconds}}
774
+ # Per-bank slot FLOORS inside the recall count cap. The merged multi-bank set
775
+ # is sorted globally by relevance then head-sliced, which is winner-take-all
776
+ # across banks: when both banks return more candidates than the cap, one bank's
777
+ # score distribution can fill every slot and the agent gets a dossier about its
778
+ # operator and none of its own session memory. Floors, not quotas — recall.py
779
+ # bounds the two floors to HALF the cap between them, so the rest is always won
780
+ # on pure relevance. Switchroom ships 2 own / 1 additional against the deployed
781
+ # cap of 6; 0/0 restores the pure head-slice.
782
+ #
783
+ # EXPORTED UNCONDITIONALLY (#3774), like the two #3757 knobs above. The resolved
784
+ # effective value (memory.recall.own_bank_min_slots / .additional_bank_min_slots
785
+ # from the cascade, else the shipped default) is always written, because
786
+ # ~/.hindsight/claude-code.json survives an apply and loads AFTER the plugin's
787
+ # settings.json — a conditional export would let a stale hand-edited value there
788
+ # shadow what switchroom stamps, silently and permanently.
789
+ # Observe `injected_own_bank_count` via `switchroom memory recall-log <agent>`.
790
+ export HINDSIGHT_RECALL_OWN_BANK_MIN_SLOTS={{hindsightRecallOwnBankMinSlots}}
791
+ export HINDSIGHT_RECALL_ADDITIONAL_BANK_MIN_SLOTS={{hindsightRecallAdditionalBankMinSlots}}
753
792
  # Recall fact types (memory.recall.types cascade). Switchroom default is
754
793
  # world,experience,observation (the synthesized `observation` tier is ON
755
794
  # by default, set in the plugin settings.json). Export only when the
@@ -80,6 +80,23 @@ done
80
80
 
81
81
  # Check Hindsight MCP reachable
82
82
  switchroom memory search "test" --agent assistant 2>/dev/null && echo "OK: memory search works" || echo "WARN: memory search failed"
83
+
84
+ # Reachable is NOT healthy. A bank that arrived already populated (restore,
85
+ # cross-version upgrade, vector-extension switch) never got its per-bank vector
86
+ # indexes, so recall falls back to a global index + post-filter and silently
87
+ # UNDER-RETURNS — a search still "works", it just returns almost nothing.
88
+ # This fleet ran that way for ~3 months. Check coverage, not just liveness.
89
+ # Exit codes are a contract (src/memory/hindsight-repair.ts): 0 = coverage
90
+ # confirmed complete, 3 = indexes MISSING, 4 = coverage could NOT be confirmed
91
+ # (no summary line, or zero banks scanned — a typo'd --bank looks like this),
92
+ # 1 = the check itself failed. 4 is not "fine": it means nothing was verified.
93
+ switchroom memory repair --all --dry-run >/tmp/sr-cov.txt 2>&1
94
+ case $? in
95
+ 0) echo "OK: vector index coverage complete" ;;
96
+ 3) echo "FAIL: vector index coverage MISSING — recall is under-returning; run 'switchroom memory repair --all'"; tail -2 /tmp/sr-cov.txt ;;
97
+ 4) echo "WARN: coverage NOT verified — the scan confirmed nothing (check the --bank/--schema names, or the hindsight backend)"; tail -2 /tmp/sr-cov.txt ;;
98
+ *) echo "WARN: coverage check could not run (hindsight < 0.8.5, container down, or docker unavailable)"; tail -2 /tmp/sr-cov.txt ;;
99
+ esac
83
100
  ```
84
101
 
85
102
  ## Step 3 — Interpret and report
@@ -108,6 +125,8 @@ For common failures, give the exact fix:
108
125
  | Fleet on the wrong account | `switchroom auth use <label>` (fleet-wide) or `switchroom auth agent override <agent> <label>` (one agent) |
109
126
  | Container unhealthy | `docker compose -p switchroom -f ~/.switchroom/compose/docker-compose.yml restart switchroom-<name>` |
110
127
  | Missing .mcp.json | `switchroom apply` (full reconcile + rewrite compose; bring up via `docker compose ... up -d`) or `switchroom agent reconcile <name>` (targeted) |
128
+ | Recall returns few/no results though hindsight is up (esp. after a restore or upgrade) | Missing per-bank vector index coverage. `switchroom memory repair --all --dry-run` to confirm, then `switchroom memory repair --all`. Idempotent, uses `CREATE INDEX CONCURRENTLY`, safe on a live fleet. Requires hindsight ≥ 0.8.5. |
129
+ | `switchroom doctor` shows a red/amber `hindsight version` line | The running backend drifted from the MCP contract switchroom captured. Older → bump the pinned image (`switchroom memory --update`). Newer → re-capture `tests/fixtures/hindsight-tools-list.snapshot.json`; until then any tool added upstream is invisible to agents. |
111
130
  | Bot token unresolved | Check vault: `switchroom vault list` |
112
131
  | Memory unreachable | Check Hindsight MCP server is running |
113
132
 
@@ -85,7 +85,7 @@ assistant — running (2h 14m)
85
85
  model: claude-sonnet-5 collection: general
86
86
 
87
87
  dev — running (45m)
88
- model: claude-opus-4-8 collection: coding
88
+ model: claude-opus-5 collection: coding
89
89
 
90
90
  coach — stopped
91
91
  last run: 3 days ago
@@ -21,6 +21,7 @@
21
21
  import type { QuotaResult, QuotaUtilization } from './quota-check.js';
22
22
  import { isProbeThin, refillNormalizedUtils } from '../src/auth/quota.js';
23
23
  import type { AccountState, LastQuotaSnapshot, ListStateData } from '../src/auth/broker/client.js';
24
+ import { effectiveServingLabel } from '../src/auth/broker/client.js';
24
25
  import { maskEmail } from './demo-mask.js';
25
26
  import { escapeMarkdown, codeSpanSafe } from './card-format.js';
26
27
 
@@ -1152,12 +1153,16 @@ export function buildSnapshotsFromState(
1152
1153
  quotas: QuotaResult[],
1153
1154
  ): AccountSnapshot[] {
1154
1155
  const out: AccountSnapshot[] = [];
1156
+ // `isActive` means SERVING, not pinned — see effectiveServingLabel. Reading
1157
+ // `state.active` here marked the rolled-off pin as "(active)" on the /usage
1158
+ // card after every soft-avoid roll.
1159
+ const serving = effectiveServingLabel(state);
1155
1160
  for (let i = 0; i < state.accounts.length; i++) {
1156
1161
  const acc: AccountState = state.accounts[i]!;
1157
1162
  const q = quotas[i];
1158
1163
  out.push({
1159
1164
  label: acc.label,
1160
- isActive: acc.label === state.active,
1165
+ isActive: acc.label === serving,
1161
1166
  quota: q && q.ok ? q.data : null,
1162
1167
  quotaError: q && !q.ok ? q.reason : undefined,
1163
1168
  expiresAtMs: acc.expiresAt,
@@ -1206,11 +1211,13 @@ export function reviveLastQuota(snap: LastQuotaSnapshot | null | undefined): Quo
1206
1211
  export function buildSnapshotsFromCachedState(
1207
1212
  state: ListStateData,
1208
1213
  ): AccountSnapshot[] {
1214
+ // SERVING, not pinned — same contract as buildSnapshotsFromState.
1215
+ const serving = effectiveServingLabel(state);
1209
1216
  return state.accounts.map((acc) => {
1210
1217
  const lq = acc.last_quota ?? null;
1211
1218
  return {
1212
1219
  label: acc.label,
1213
- isActive: acc.label === state.active,
1220
+ isActive: acc.label === serving,
1214
1221
  quota: reviveLastQuota(lq),
1215
1222
  quotaError: lq ? undefined : 'no cached quota (no probe since broker start)',
1216
1223
  expiresAtMs: acc.expiresAt,