switchroom 0.19.17 → 0.19.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/bin/run-hook.sh +148 -0
  2. package/bin/workspace-dynamic-hook.sh +147 -38
  3. package/dist/agent-scheduler/index.js +13 -4
  4. package/dist/auth-broker/index.js +32 -5
  5. package/dist/cli/drive-write-pretool.mjs +48 -5
  6. package/dist/cli/ms-365-write-pretool.mjs +40 -2
  7. package/dist/cli/notion-write-pretool.mjs +13 -4
  8. package/dist/cli/switchroom.js +10614 -8104
  9. package/dist/host-control/main.js +12849 -11446
  10. package/dist/vault/approvals/kernel-server.js +90 -12
  11. package/dist/vault/broker/server.js +277 -94
  12. package/package.json +5 -3
  13. package/profiles/_base/start.sh.hbs +69 -5
  14. package/profiles/coding/CLAUDE.md.hbs +1 -1
  15. package/profiles/default/CLAUDE.md.hbs +3 -3
  16. package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
  17. package/profiles/health-coach/CLAUDE.md.hbs +1 -1
  18. package/skills/mental-model-curator/SKILL.md +8 -6
  19. package/telegram-plugin/bridge/bridge.ts +25 -19
  20. package/telegram-plugin/bridge/mcp-instructions.ts +87 -0
  21. package/telegram-plugin/dist/bridge/bridge.js +28 -20
  22. package/telegram-plugin/dist/gateway/gateway.js +2077 -1087
  23. package/telegram-plugin/dist/server.js +32 -20
  24. package/telegram-plugin/gateway/always-allow-persist-queue.ts +97 -11
  25. package/telegram-plugin/gateway/boot-card.ts +5 -1
  26. package/telegram-plugin/gateway/boot-probes.ts +113 -0
  27. package/telegram-plugin/gateway/config-approval-handler.test.ts +54 -0
  28. package/telegram-plugin/gateway/config-approval-handler.ts +16 -1
  29. package/telegram-plugin/gateway/disconnect-flush.ts +17 -0
  30. package/telegram-plugin/gateway/gateway.ts +43 -1
  31. package/telegram-plugin/gateway/handback-preturn-signal.ts +61 -7
  32. package/telegram-plugin/gateway/ipc-protocol.ts +5 -0
  33. package/telegram-plugin/gateway/ipc-server.ts +13 -0
  34. package/telegram-plugin/gateway/liveness-wiring.ts +125 -5
  35. package/telegram-plugin/gateway/missed-approvals-store.ts +66 -17
  36. package/telegram-plugin/gateway/obligation-ledger.ts +84 -4
  37. package/telegram-plugin/gateway/pending-card-store.ts +46 -16
  38. package/telegram-plugin/gateway/resume-inbound-builder.ts +13 -4
  39. package/telegram-plugin/gateway/scoped-grant-store.ts +39 -14
  40. package/telegram-plugin/gateway/store-file.ts +244 -0
  41. package/telegram-plugin/gateway/stream-render.ts +24 -5
  42. package/telegram-plugin/hooks/secret-guard-pretool.mjs +249 -76
  43. package/telegram-plugin/hooks/tool-label-pretool.mjs +88 -2
  44. package/telegram-plugin/registry/turns-schema.test.ts +8 -3
  45. package/telegram-plugin/registry/turns-schema.ts +40 -12
  46. package/telegram-plugin/runtime-metrics.ts +14 -0
  47. package/telegram-plugin/silence-poke.ts +138 -0
  48. package/telegram-plugin/tests/boot-probe-drift.test.ts +152 -0
  49. package/telegram-plugin/tests/bridge-tool-parity.test.ts +95 -0
  50. package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +32 -0
  51. package/telegram-plugin/tests/handback-preturn-signal.test.ts +62 -0
  52. package/telegram-plugin/tests/helpers/liveness-wiring-fixture.ts +178 -0
  53. package/telegram-plugin/tests/ipc-server-validate-config-approval.test.ts +95 -0
  54. package/telegram-plugin/tests/mcp-instructions-budget.test.ts +184 -0
  55. package/telegram-plugin/tests/multitopic-routing-wiring.test.ts +22 -2
  56. package/telegram-plugin/tests/obligation-determinism.test.ts +114 -3
  57. package/telegram-plugin/tests/obligation-ledger.test.ts +310 -0
  58. package/telegram-plugin/tests/registry-turns.test.ts +13 -0
  59. package/telegram-plugin/tests/resume-inbound-builder.test.ts +15 -0
  60. package/telegram-plugin/tests/secret-guard-pretool.test.ts +347 -16
  61. package/telegram-plugin/tests/silence-poke-orphan-reap.test.ts +392 -0
  62. package/telegram-plugin/tests/silence-poke-teardown-notice.test.ts +301 -0
  63. package/telegram-plugin/tests/store-atomic-write.test.ts +411 -0
  64. package/telegram-plugin/tests/stream-render-golden.test.ts +103 -1
  65. package/telegram-plugin/tests/tool-activity-summary.test.ts +9 -2
  66. package/telegram-plugin/tests/tool-label-pretool.test.ts +94 -0
  67. package/telegram-plugin/tests/tts-normalize.test.ts +43 -0
  68. package/telegram-plugin/tests/voice-normalize-text.test.ts +212 -3
  69. package/telegram-plugin/tests/worker-feed-repeat-steps.test.ts +147 -0
  70. package/telegram-plugin/tts-normalize.ts +6 -4
  71. package/telegram-plugin/voice-normalize-text.ts +168 -11
  72. package/telegram-plugin/worker-activity-feed.ts +51 -1
  73. package/vendor/hindsight-memory/CHANGELOG.md +73 -0
  74. package/vendor/hindsight-memory/scripts/drain_pending.py +668 -56
  75. package/vendor/hindsight-memory/scripts/lib/client.py +124 -0
  76. package/vendor/hindsight-memory/scripts/lib/config.py +8 -3
  77. package/vendor/hindsight-memory/scripts/lib/directives.py +62 -4
  78. package/vendor/hindsight-memory/scripts/lib/pending.py +865 -33
  79. package/vendor/hindsight-memory/scripts/lib/retain_split.py +449 -0
  80. package/vendor/hindsight-memory/scripts/recall.py +257 -12
  81. package/vendor/hindsight-memory/scripts/retain.py +12 -6
  82. package/vendor/hindsight-memory/scripts/session_start.py +48 -0
  83. package/vendor/hindsight-memory/scripts/tests/test_client_document_exists.py +470 -0
  84. package/vendor/hindsight-memory/scripts/tests/test_directives.py +80 -9
  85. package/vendor/hindsight-memory/scripts/tests/test_pending_drops.py +2121 -0
  86. package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +362 -18
  87. package/vendor/hindsight-memory/scripts/tests/test_retain_split.py +430 -0
  88. package/vendor/hindsight-memory/scripts/tests/test_session_start_version_skew.py +204 -0
  89. package/vendor/hindsight-memory/settings.json +1 -1
  90. package/vendor/hindsight-memory/tests/test_drain_pending.py +102 -6
  91. package/vendor/hindsight-memory/tests/test_pending.py +32 -7
@@ -5,12 +5,15 @@ Openclaw HindsightClient (client.js), adapted for Python stdlib.
5
5
  """
6
6
 
7
7
  import json
8
+ import time
8
9
  import urllib.error
9
10
  import urllib.parse
10
11
  import urllib.request
11
12
  from pathlib import Path
12
13
  from typing import Optional
13
14
 
15
+ from .retain_split import part_document_id, part_metadata, split_retain_content
16
+
14
17
  DEFAULT_TIMEOUT = 15 # seconds
15
18
  HEALTH_CHECK_RETRIES = 3
16
19
  HEALTH_CHECK_DELAY = 2 # seconds
@@ -189,7 +192,85 @@ class HindsightClient:
189
192
  reconciliation) MUST use this so a bare async 200 can never falsely mark
190
193
  unpersisted work as committed. (Merge precondition: the daemon honours
191
194
  ``async=false`` as commit-before-ack — verified by the §1.1 probe.)
195
+
196
+ **Oversized content is split before it is posted** (``lib/retain_split``).
197
+ The daemon runs one sequential extraction LLM call per
198
+ ``retain_chunk_size`` (3000) chars, so server wall time is linear in
199
+ ``len(content)`` while the client has a single deadline for the whole
200
+ POST — past ~45,000 chars a retain cannot complete at ANY client
201
+ timeout and the memory is permanently unsaveable (measured: 154 of the
202
+ 629 entries in the 2026-07-25 fleet backlog). This is the enforcement
203
+ point precisely because every retain POST in the plugin goes through
204
+ it, so the bound is code-enforced rather than left to each producer.
205
+
206
+ Parts are posted SEQUENTIALLY, each with the caller's ``timeout`` (the
207
+ timeout is a per-HTTP-request read deadline, and each part is its own
208
+ request) and each with the caller's ``async_processing`` — so a
209
+ durability caller still gets commit-before-ack per part. Any part
210
+ failing raises, exactly as an unsplit failure does, so the caller's
211
+ existing enqueue/retry handling is unchanged; already-committed parts
212
+ carry deterministic ids and are upserted, not duplicated, on retry.
213
+
214
+ ``timeout`` bounds the CALLER'S TOTAL WALL TIME, not just each HTTP
215
+ request. Splitting must never turn one bounded POST into N of them:
216
+ the Stop hook passes ``timeout=15`` because it has a hook budget to
217
+ respect, and ``15 × N`` would stall the session. Once the budget is
218
+ spent no further part is started and the call raises, exactly as an
219
+ unsplit failure does, so each caller's EXISTING failure path handles
220
+ the remainder: the hook paths (``retain.py``, ``subagent_retain.py``,
221
+ ``reconcile_tail.py``) enqueue, and ``pending.enqueue`` queues the
222
+ remainder as bounded per-part entries the drainer can finish; the
223
+ drain paths count an attempt and keep the entry;
224
+ ``backfill_transcripts.py`` logs and backs off. The parts already
225
+ committed are upserted on the next attempt, not duplicated.
192
226
  """
227
+ parts = split_retain_content(content)
228
+ total = len(parts)
229
+ response = None
230
+ deadline = time.monotonic() + timeout
231
+ part_timeout = timeout
232
+ for index, part in enumerate(parts):
233
+ if index > 0:
234
+ remaining = deadline - time.monotonic()
235
+ if remaining <= 0:
236
+ raise TimeoutError(
237
+ f"retain wall budget of {timeout}s exhausted after "
238
+ f"{index}/{total} parts; no further part was started. "
239
+ f"The caller's own failure path decides the remainder: "
240
+ f"the hook paths enqueue it (as bounded per-part "
241
+ f"entries), the drain paths count an attempt and keep "
242
+ f"the entry queued."
243
+ )
244
+ # Clamp the request deadline to what is left of the budget so
245
+ # the final part cannot overrun it either.
246
+ part_timeout = max(1, int(remaining))
247
+ response = self._retain_one(
248
+ bank_id=bank_id,
249
+ content=part,
250
+ document_id=part_document_id(document_id, index, total),
251
+ context=context,
252
+ metadata=part_metadata(metadata, index, total),
253
+ tags=tags,
254
+ timeout=part_timeout,
255
+ async_processing=async_processing,
256
+ )
257
+ if total > 1 and isinstance(response, dict):
258
+ response = dict(response)
259
+ response["split_parts"] = total
260
+ return response
261
+
262
+ def _retain_one(
263
+ self,
264
+ bank_id: str,
265
+ content: str,
266
+ document_id: str,
267
+ context: Optional[str],
268
+ metadata: Optional[dict],
269
+ tags: Optional[list],
270
+ timeout: int,
271
+ async_processing: bool,
272
+ ) -> dict:
273
+ """POST exactly one retain item. Raises on any HTTP/transport error."""
193
274
  path = f"/v1/default/banks/{urllib.parse.quote(bank_id, safe='')}/memories"
194
275
  item = {
195
276
  "content": content,
@@ -206,6 +287,49 @@ class HindsightClient:
206
287
  }
207
288
  return self._request("POST", path, body, timeout=timeout)
208
289
 
290
+ def document_exists(
291
+ self,
292
+ bank_id: str,
293
+ document_id: str,
294
+ timeout: int = 30,
295
+ ) -> Optional[bool]:
296
+ """TRI-STATE presence check for one document (switchroom #3596).
297
+
298
+ ``True`` — the document is there. ``False`` — the server said 404.
299
+ ``None`` — **unknown** (transport error, 5xx, timeout).
300
+
301
+ The tri-state is load-bearing and must not be collapsed to a bool.
302
+ Two callers depend on it:
303
+
304
+ * the backlog drain's reconcile phase, which SKIPS a retain when the
305
+ memory already exists. Treating "unknown" as ``False`` there would
306
+ re-POST a document that is already durable — the duplicated-LLM-cost
307
+ bug this check exists to avoid (70.4% of one measured 5,751-entry
308
+ fleet backlog already existed as documents).
309
+ * commit-before-delete, which only deletes a queue entry once presence
310
+ is CONFIRMED. Treating "unknown" as ``True`` there would delete the
311
+ last on-disk copy of a turn on a flaky GET — the #3244 silent-loss
312
+ shape, reintroduced from the other direction.
313
+
314
+ Deliberately does NOT reuse ``_request()``: that wraps every
315
+ ``HTTPError`` into a ``RuntimeError``, which would make a 404
316
+ indistinguishable from a 503 without string-matching the message.
317
+ """
318
+ bank = urllib.parse.quote(bank_id, safe="")
319
+ did = urllib.parse.quote(document_id, safe="")
320
+ url = f"{self.api_url}/v1/default/banks/{bank}/documents/{did}"
321
+ req = urllib.request.Request(url, headers=self._headers(), method="GET")
322
+ try:
323
+ with urllib.request.urlopen(
324
+ req, timeout=self._resolve_timeout(timeout)
325
+ ) as resp:
326
+ resp.read()
327
+ return True
328
+ except urllib.error.HTTPError as e:
329
+ return False if e.code == 404 else None
330
+ except Exception:
331
+ return None
332
+
209
333
  def list_session_document_ids(
210
334
  self,
211
335
  bank_id: str,
@@ -29,15 +29,20 @@ DEFAULTS = {
29
29
  # formatting. Set to 0 (or any non-positive value) to disable the cap
30
30
  # and inject everything Hindsight returns.
31
31
  "recallMaxMemories": 12,
32
- # Switchroom-local: minimum lexical (Jaccard) overlap between the
32
+ # Switchroom-local: minimum lexical (containment) overlap between the
33
33
  # user's query terms and a memory's text terms. Memories below this
34
34
  # threshold are dropped before formatting. 0.0 disables the gate
35
35
  # (current behaviour: inject everything Hindsight returns up to the
36
36
  # count cap). NOTE: Hindsight's HTTP recall API DOES return per-result
37
37
  # relevance scores (`scores.final`, plus `.semantic`/`.keyword`/
38
38
  # `.reranker`) — verified at runtime — and recall.py now reads and
39
- # sorts the merged set by `scores.final`. This Jaccard gate is a
40
- # separate lexical-overlap quality filter layered on top — see #475.
39
+ # sorts the merged set by `scores.final`. This lexical gate is a
40
+ # separate quality filter layered on top — see #475. The metric is
41
+ # containment, `|Q n M| / |M|`, not Jaccard: dividing by the union made
42
+ # the score a function of prompt length rather than relevance — see
43
+ # #3541 and recall.py's design note. At the 0.10 fleet default this is
44
+ # close to a passthrough (a <=10-token memory clears it on one shared
45
+ # word); precision is the engine rerank's job, not this gate's.
41
46
  "recallMinOverlap": 0.0,
42
47
  "recallTypes": ["world", "experience"],
43
48
  # Switchroom-local: when True (default; Ken-approved ON) recall biases
@@ -23,10 +23,33 @@ from typing import Optional
23
23
 
24
24
  from .state import list_state_names, read_state, remove_state, write_state
25
25
 
26
- # Sanity cap on how many directives we ever inject into the prompt. Banks
27
- # with more active directives than this are pathological; truncate with a
28
- # footer so the agent knows there are more.
29
- MAX_DIRECTIVES = 15
26
+ # Sanity cap on how many directives we ever inject into the prompt.
27
+ #
28
+ # This number is a COST TRADEOFF, not an arbitrary limit, and the real cost is
29
+ # larger than "one block": `recall.py` rebuilds the <active_directives> block
30
+ # on EVERY UserPromptSubmit (the `format_active_directives_block` call there)
31
+ # with no per-session dedupe and no "unchanged since last turn" suppression,
32
+ # and `additionalContext` is APPENDED into the conversation. So the block is
33
+ # re-paid every turn and accumulates — roughly (block size) x (turn count) over
34
+ # a session, not once.
35
+ #
36
+ # Measured live 2026-07-25 (fleet REST `/directives`):
37
+ # overlord 12 active ~9.8 KB ~816 chars avg (one 2,981-char outlier)
38
+ # klanker 17 active ~9.9 KB ~585 chars avg
39
+ # gymbro 9 active ~9.3 KB ~1038 chars avg
40
+ # At the ~700-char fleet average, a bank sitting at this cap injects ~21 KB —
41
+ # on the order of 5,000-6,000 tokens per injection, per turn, cumulative across
42
+ # the session. 30 is chosen to clear the observed fleet maximum with headroom;
43
+ # it is NOT a size at which a runaway bank becomes harmless. The actual defence
44
+ # against pile-up is the doctor's WARN/FAIL on the active directive count
45
+ # (src/cli/doctor-memory.ts), not this cap. Raising it further is a legitimate
46
+ # call — make it deliberately, with the per-turn-times-turns cost in mind, and
47
+ # move the doctor thresholds with it.
48
+ #
49
+ # Banks with more active directives than this are pathological; we truncate
50
+ # with an in-prompt footer, a `directives_omitted` field on the recall_log row,
51
+ # and a stderr warning (see `format_active_directives_block`).
52
+ MAX_DIRECTIVES = 30
30
53
 
31
54
  # Hard timeout for the list_directives call. The recall hook is on the
32
55
  # UserPromptSubmit critical path — we cannot block it for long.
@@ -208,6 +231,16 @@ def invalidate_directives_cache(bank_id: Optional[str] = None) -> None:
208
231
  remove_state(name)
209
232
 
210
233
 
234
+ def count_omitted_directives(directives: list, max_directives: int = MAX_DIRECTIVES) -> int:
235
+ """How many directives `format_active_directives_block` would DROP.
236
+
237
+ Pure counterpart of the truncation branch below, so `recall.py` can put the
238
+ number on the recall_log row without re-deriving the cap. 0 when nothing is
239
+ dropped.
240
+ """
241
+ return max(0, len(directives) - max_directives)
242
+
243
+
211
244
  def format_active_directives_block(directives: list, max_directives: int = MAX_DIRECTIVES) -> Optional[str]:
212
245
  """Format directives into the <active_directives> block string.
213
246
 
@@ -252,6 +285,31 @@ def format_active_directives_block(directives: list, max_directives: int = MAX_D
252
285
  if omitted > 0:
253
286
  lines.append("")
254
287
  lines.append(f"(+{omitted} more, omitted)")
288
+ # The in-prompt footer above only tells the AGENT. This stderr warn is
289
+ # the same channel every other operational failure in this module uses
290
+ # (see `_fetch_directives_with_status`), and it is a LAST-RESORT
291
+ # breadcrumb only — do NOT rely on it reaching an operator.
292
+ #
293
+ # Measured 2026-07-25: `docker logs --tail 20000` across all 12 running
294
+ # agent containers returns ZERO `[Hindsight]` lines, and nothing under
295
+ # ~/.switchroom/logs/ contains them either, despite months of runtime
296
+ # and several long-standing stderr paths in recall.py. Claude Code
297
+ # appears to swallow hook stderr on a zero exit, so hook stderr is not
298
+ # an operator-visible channel.
299
+ #
300
+ # The channels that DO reach an operator:
301
+ # * the `directives_omitted` field on the recall_log row
302
+ # (state/recall_log.jsonl — see `count_omitted_directives`), and
303
+ # * `switchroom doctor`'s WARN/FAIL on the bank's active directive
304
+ # count (src/cli/doctor-memory.ts `classifyDirectiveCount`), which
305
+ # reads the count from the same REST surface this module fetches.
306
+ print(
307
+ f"[Hindsight] directive truncation: {total} active directives exceeds "
308
+ f"MAX_DIRECTIVES={max_directives} — {omitted} lowest-priority "
309
+ f"directive(s) were DROPPED from this turn's prompt. Merge or retire "
310
+ f"directives (mental-model-curator) or raise MAX_DIRECTIVES.",
311
+ file=sys.stderr,
312
+ )
255
313
 
256
314
  lines.append("</active_directives>")
257
315
  return "\n".join(lines)