switchroom 0.19.17 → 0.19.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/run-hook.sh +148 -0
- package/bin/workspace-dynamic-hook.sh +147 -38
- package/dist/agent-scheduler/index.js +13 -4
- package/dist/auth-broker/index.js +32 -5
- package/dist/cli/drive-write-pretool.mjs +48 -5
- package/dist/cli/ms-365-write-pretool.mjs +40 -2
- package/dist/cli/notion-write-pretool.mjs +13 -4
- package/dist/cli/switchroom.js +10614 -8104
- package/dist/host-control/main.js +12849 -11446
- package/dist/vault/approvals/kernel-server.js +90 -12
- package/dist/vault/broker/server.js +277 -94
- package/package.json +5 -3
- package/profiles/_base/start.sh.hbs +69 -5
- package/profiles/coding/CLAUDE.md.hbs +1 -1
- package/profiles/default/CLAUDE.md.hbs +3 -3
- package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
- package/profiles/health-coach/CLAUDE.md.hbs +1 -1
- package/skills/mental-model-curator/SKILL.md +8 -6
- package/telegram-plugin/bridge/bridge.ts +25 -19
- package/telegram-plugin/bridge/mcp-instructions.ts +87 -0
- package/telegram-plugin/dist/bridge/bridge.js +28 -20
- package/telegram-plugin/dist/gateway/gateway.js +2077 -1087
- package/telegram-plugin/dist/server.js +32 -20
- package/telegram-plugin/gateway/always-allow-persist-queue.ts +97 -11
- package/telegram-plugin/gateway/boot-card.ts +5 -1
- package/telegram-plugin/gateway/boot-probes.ts +113 -0
- package/telegram-plugin/gateway/config-approval-handler.test.ts +54 -0
- package/telegram-plugin/gateway/config-approval-handler.ts +16 -1
- package/telegram-plugin/gateway/disconnect-flush.ts +17 -0
- package/telegram-plugin/gateway/gateway.ts +43 -1
- package/telegram-plugin/gateway/handback-preturn-signal.ts +61 -7
- package/telegram-plugin/gateway/ipc-protocol.ts +5 -0
- package/telegram-plugin/gateway/ipc-server.ts +13 -0
- package/telegram-plugin/gateway/liveness-wiring.ts +125 -5
- package/telegram-plugin/gateway/missed-approvals-store.ts +66 -17
- package/telegram-plugin/gateway/obligation-ledger.ts +84 -4
- package/telegram-plugin/gateway/pending-card-store.ts +46 -16
- package/telegram-plugin/gateway/resume-inbound-builder.ts +13 -4
- package/telegram-plugin/gateway/scoped-grant-store.ts +39 -14
- package/telegram-plugin/gateway/store-file.ts +244 -0
- package/telegram-plugin/gateway/stream-render.ts +24 -5
- package/telegram-plugin/hooks/secret-guard-pretool.mjs +249 -76
- package/telegram-plugin/hooks/tool-label-pretool.mjs +88 -2
- package/telegram-plugin/registry/turns-schema.test.ts +8 -3
- package/telegram-plugin/registry/turns-schema.ts +40 -12
- package/telegram-plugin/runtime-metrics.ts +14 -0
- package/telegram-plugin/silence-poke.ts +138 -0
- package/telegram-plugin/tests/boot-probe-drift.test.ts +152 -0
- package/telegram-plugin/tests/bridge-tool-parity.test.ts +95 -0
- package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +32 -0
- package/telegram-plugin/tests/handback-preturn-signal.test.ts +62 -0
- package/telegram-plugin/tests/helpers/liveness-wiring-fixture.ts +178 -0
- package/telegram-plugin/tests/ipc-server-validate-config-approval.test.ts +95 -0
- package/telegram-plugin/tests/mcp-instructions-budget.test.ts +184 -0
- package/telegram-plugin/tests/multitopic-routing-wiring.test.ts +22 -2
- package/telegram-plugin/tests/obligation-determinism.test.ts +114 -3
- package/telegram-plugin/tests/obligation-ledger.test.ts +310 -0
- package/telegram-plugin/tests/registry-turns.test.ts +13 -0
- package/telegram-plugin/tests/resume-inbound-builder.test.ts +15 -0
- package/telegram-plugin/tests/secret-guard-pretool.test.ts +347 -16
- package/telegram-plugin/tests/silence-poke-orphan-reap.test.ts +392 -0
- package/telegram-plugin/tests/silence-poke-teardown-notice.test.ts +301 -0
- package/telegram-plugin/tests/store-atomic-write.test.ts +411 -0
- package/telegram-plugin/tests/stream-render-golden.test.ts +103 -1
- package/telegram-plugin/tests/tool-activity-summary.test.ts +9 -2
- package/telegram-plugin/tests/tool-label-pretool.test.ts +94 -0
- package/telegram-plugin/tests/tts-normalize.test.ts +43 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +212 -3
- package/telegram-plugin/tests/worker-feed-repeat-steps.test.ts +147 -0
- package/telegram-plugin/tts-normalize.ts +6 -4
- package/telegram-plugin/voice-normalize-text.ts +168 -11
- package/telegram-plugin/worker-activity-feed.ts +51 -1
- package/vendor/hindsight-memory/CHANGELOG.md +73 -0
- package/vendor/hindsight-memory/scripts/drain_pending.py +668 -56
- package/vendor/hindsight-memory/scripts/lib/client.py +124 -0
- package/vendor/hindsight-memory/scripts/lib/config.py +8 -3
- package/vendor/hindsight-memory/scripts/lib/directives.py +62 -4
- package/vendor/hindsight-memory/scripts/lib/pending.py +865 -33
- package/vendor/hindsight-memory/scripts/lib/retain_split.py +449 -0
- package/vendor/hindsight-memory/scripts/recall.py +257 -12
- package/vendor/hindsight-memory/scripts/retain.py +12 -6
- package/vendor/hindsight-memory/scripts/session_start.py +48 -0
- package/vendor/hindsight-memory/scripts/tests/test_client_document_exists.py +470 -0
- package/vendor/hindsight-memory/scripts/tests/test_directives.py +80 -9
- package/vendor/hindsight-memory/scripts/tests/test_pending_drops.py +2121 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +362 -18
- package/vendor/hindsight-memory/scripts/tests/test_retain_split.py +430 -0
- package/vendor/hindsight-memory/scripts/tests/test_session_start_version_skew.py +204 -0
- package/vendor/hindsight-memory/settings.json +1 -1
- package/vendor/hindsight-memory/tests/test_drain_pending.py +102 -6
- package/vendor/hindsight-memory/tests/test_pending.py +32 -7
|
@@ -5,12 +5,15 @@ Openclaw HindsightClient (client.js), adapted for Python stdlib.
|
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
7
|
import json
|
|
8
|
+
import time
|
|
8
9
|
import urllib.error
|
|
9
10
|
import urllib.parse
|
|
10
11
|
import urllib.request
|
|
11
12
|
from pathlib import Path
|
|
12
13
|
from typing import Optional
|
|
13
14
|
|
|
15
|
+
from .retain_split import part_document_id, part_metadata, split_retain_content
|
|
16
|
+
|
|
14
17
|
DEFAULT_TIMEOUT = 15 # seconds
|
|
15
18
|
HEALTH_CHECK_RETRIES = 3
|
|
16
19
|
HEALTH_CHECK_DELAY = 2 # seconds
|
|
@@ -189,7 +192,85 @@ class HindsightClient:
|
|
|
189
192
|
reconciliation) MUST use this so a bare async 200 can never falsely mark
|
|
190
193
|
unpersisted work as committed. (Merge precondition: the daemon honours
|
|
191
194
|
``async=false`` as commit-before-ack — verified by the §1.1 probe.)
|
|
195
|
+
|
|
196
|
+
**Oversized content is split before it is posted** (``lib/retain_split``).
|
|
197
|
+
The daemon runs one sequential extraction LLM call per
|
|
198
|
+
``retain_chunk_size`` (3000) chars, so server wall time is linear in
|
|
199
|
+
``len(content)`` while the client has a single deadline for the whole
|
|
200
|
+
POST — past ~45,000 chars a retain cannot complete at ANY client
|
|
201
|
+
timeout and the memory is permanently unsaveable (measured: 154 of the
|
|
202
|
+
629 entries in the 2026-07-25 fleet backlog). This is the enforcement
|
|
203
|
+
point precisely because every retain POST in the plugin goes through
|
|
204
|
+
it, so the bound is code-enforced rather than left to each producer.
|
|
205
|
+
|
|
206
|
+
Parts are posted SEQUENTIALLY, each with the caller's ``timeout`` (the
|
|
207
|
+
timeout is a per-HTTP-request read deadline, and each part is its own
|
|
208
|
+
request) and each with the caller's ``async_processing`` — so a
|
|
209
|
+
durability caller still gets commit-before-ack per part. Any part
|
|
210
|
+
failing raises, exactly as an unsplit failure does, so the caller's
|
|
211
|
+
existing enqueue/retry handling is unchanged; already-committed parts
|
|
212
|
+
carry deterministic ids and are upserted, not duplicated, on retry.
|
|
213
|
+
|
|
214
|
+
``timeout`` bounds the CALLER'S TOTAL WALL TIME, not just each HTTP
|
|
215
|
+
request. Splitting must never turn one bounded POST into N of them:
|
|
216
|
+
the Stop hook passes ``timeout=15`` because it has a hook budget to
|
|
217
|
+
respect, and ``15 × N`` would stall the session. Once the budget is
|
|
218
|
+
spent no further part is started and the call raises, exactly as an
|
|
219
|
+
unsplit failure does, so each caller's EXISTING failure path handles
|
|
220
|
+
the remainder: the hook paths (``retain.py``, ``subagent_retain.py``,
|
|
221
|
+
``reconcile_tail.py``) enqueue, and ``pending.enqueue`` queues the
|
|
222
|
+
remainder as bounded per-part entries the drainer can finish; the
|
|
223
|
+
drain paths count an attempt and keep the entry;
|
|
224
|
+
``backfill_transcripts.py`` logs and backs off. The parts already
|
|
225
|
+
committed are upserted on the next attempt, not duplicated.
|
|
192
226
|
"""
|
|
227
|
+
parts = split_retain_content(content)
|
|
228
|
+
total = len(parts)
|
|
229
|
+
response = None
|
|
230
|
+
deadline = time.monotonic() + timeout
|
|
231
|
+
part_timeout = timeout
|
|
232
|
+
for index, part in enumerate(parts):
|
|
233
|
+
if index > 0:
|
|
234
|
+
remaining = deadline - time.monotonic()
|
|
235
|
+
if remaining <= 0:
|
|
236
|
+
raise TimeoutError(
|
|
237
|
+
f"retain wall budget of {timeout}s exhausted after "
|
|
238
|
+
f"{index}/{total} parts; no further part was started. "
|
|
239
|
+
f"The caller's own failure path decides the remainder: "
|
|
240
|
+
f"the hook paths enqueue it (as bounded per-part "
|
|
241
|
+
f"entries), the drain paths count an attempt and keep "
|
|
242
|
+
f"the entry queued."
|
|
243
|
+
)
|
|
244
|
+
# Clamp the request deadline to what is left of the budget so
|
|
245
|
+
# the final part cannot overrun it either.
|
|
246
|
+
part_timeout = max(1, int(remaining))
|
|
247
|
+
response = self._retain_one(
|
|
248
|
+
bank_id=bank_id,
|
|
249
|
+
content=part,
|
|
250
|
+
document_id=part_document_id(document_id, index, total),
|
|
251
|
+
context=context,
|
|
252
|
+
metadata=part_metadata(metadata, index, total),
|
|
253
|
+
tags=tags,
|
|
254
|
+
timeout=part_timeout,
|
|
255
|
+
async_processing=async_processing,
|
|
256
|
+
)
|
|
257
|
+
if total > 1 and isinstance(response, dict):
|
|
258
|
+
response = dict(response)
|
|
259
|
+
response["split_parts"] = total
|
|
260
|
+
return response
|
|
261
|
+
|
|
262
|
+
def _retain_one(
|
|
263
|
+
self,
|
|
264
|
+
bank_id: str,
|
|
265
|
+
content: str,
|
|
266
|
+
document_id: str,
|
|
267
|
+
context: Optional[str],
|
|
268
|
+
metadata: Optional[dict],
|
|
269
|
+
tags: Optional[list],
|
|
270
|
+
timeout: int,
|
|
271
|
+
async_processing: bool,
|
|
272
|
+
) -> dict:
|
|
273
|
+
"""POST exactly one retain item. Raises on any HTTP/transport error."""
|
|
193
274
|
path = f"/v1/default/banks/{urllib.parse.quote(bank_id, safe='')}/memories"
|
|
194
275
|
item = {
|
|
195
276
|
"content": content,
|
|
@@ -206,6 +287,49 @@ class HindsightClient:
|
|
|
206
287
|
}
|
|
207
288
|
return self._request("POST", path, body, timeout=timeout)
|
|
208
289
|
|
|
290
|
+
def document_exists(
|
|
291
|
+
self,
|
|
292
|
+
bank_id: str,
|
|
293
|
+
document_id: str,
|
|
294
|
+
timeout: int = 30,
|
|
295
|
+
) -> Optional[bool]:
|
|
296
|
+
"""TRI-STATE presence check for one document (switchroom #3596).
|
|
297
|
+
|
|
298
|
+
``True`` — the document is there. ``False`` — the server said 404.
|
|
299
|
+
``None`` — **unknown** (transport error, 5xx, timeout).
|
|
300
|
+
|
|
301
|
+
The tri-state is load-bearing and must not be collapsed to a bool.
|
|
302
|
+
Two callers depend on it:
|
|
303
|
+
|
|
304
|
+
* the backlog drain's reconcile phase, which SKIPS a retain when the
|
|
305
|
+
memory already exists. Treating "unknown" as ``False`` there would
|
|
306
|
+
re-POST a document that is already durable — the duplicated-LLM-cost
|
|
307
|
+
bug this check exists to avoid (70.4% of one measured 5,751-entry
|
|
308
|
+
fleet backlog already existed as documents).
|
|
309
|
+
* commit-before-delete, which only deletes a queue entry once presence
|
|
310
|
+
is CONFIRMED. Treating "unknown" as ``True`` there would delete the
|
|
311
|
+
last on-disk copy of a turn on a flaky GET — the #3244 silent-loss
|
|
312
|
+
shape, reintroduced from the other direction.
|
|
313
|
+
|
|
314
|
+
Deliberately does NOT reuse ``_request()``: that wraps every
|
|
315
|
+
``HTTPError`` into a ``RuntimeError``, which would make a 404
|
|
316
|
+
indistinguishable from a 503 without string-matching the message.
|
|
317
|
+
"""
|
|
318
|
+
bank = urllib.parse.quote(bank_id, safe="")
|
|
319
|
+
did = urllib.parse.quote(document_id, safe="")
|
|
320
|
+
url = f"{self.api_url}/v1/default/banks/{bank}/documents/{did}"
|
|
321
|
+
req = urllib.request.Request(url, headers=self._headers(), method="GET")
|
|
322
|
+
try:
|
|
323
|
+
with urllib.request.urlopen(
|
|
324
|
+
req, timeout=self._resolve_timeout(timeout)
|
|
325
|
+
) as resp:
|
|
326
|
+
resp.read()
|
|
327
|
+
return True
|
|
328
|
+
except urllib.error.HTTPError as e:
|
|
329
|
+
return False if e.code == 404 else None
|
|
330
|
+
except Exception:
|
|
331
|
+
return None
|
|
332
|
+
|
|
209
333
|
def list_session_document_ids(
|
|
210
334
|
self,
|
|
211
335
|
bank_id: str,
|
|
@@ -29,15 +29,20 @@ DEFAULTS = {
|
|
|
29
29
|
# formatting. Set to 0 (or any non-positive value) to disable the cap
|
|
30
30
|
# and inject everything Hindsight returns.
|
|
31
31
|
"recallMaxMemories": 12,
|
|
32
|
-
# Switchroom-local: minimum lexical (
|
|
32
|
+
# Switchroom-local: minimum lexical (containment) overlap between the
|
|
33
33
|
# user's query terms and a memory's text terms. Memories below this
|
|
34
34
|
# threshold are dropped before formatting. 0.0 disables the gate
|
|
35
35
|
# (current behaviour: inject everything Hindsight returns up to the
|
|
36
36
|
# count cap). NOTE: Hindsight's HTTP recall API DOES return per-result
|
|
37
37
|
# relevance scores (`scores.final`, plus `.semantic`/`.keyword`/
|
|
38
38
|
# `.reranker`) — verified at runtime — and recall.py now reads and
|
|
39
|
-
# sorts the merged set by `scores.final`. This
|
|
40
|
-
# separate
|
|
39
|
+
# sorts the merged set by `scores.final`. This lexical gate is a
|
|
40
|
+
# separate quality filter layered on top — see #475. The metric is
|
|
41
|
+
# containment, `|Q n M| / |M|`, not Jaccard: dividing by the union made
|
|
42
|
+
# the score a function of prompt length rather than relevance — see
|
|
43
|
+
# #3541 and recall.py's design note. At the 0.10 fleet default this is
|
|
44
|
+
# close to a passthrough (a <=10-token memory clears it on one shared
|
|
45
|
+
# word); precision is the engine rerank's job, not this gate's.
|
|
41
46
|
"recallMinOverlap": 0.0,
|
|
42
47
|
"recallTypes": ["world", "experience"],
|
|
43
48
|
# Switchroom-local: when True (default; Ken-approved ON) recall biases
|
|
@@ -23,10 +23,33 @@ from typing import Optional
|
|
|
23
23
|
|
|
24
24
|
from .state import list_state_names, read_state, remove_state, write_state
|
|
25
25
|
|
|
26
|
-
# Sanity cap on how many directives we ever inject into the prompt.
|
|
27
|
-
#
|
|
28
|
-
#
|
|
29
|
-
|
|
26
|
+
# Sanity cap on how many directives we ever inject into the prompt.
|
|
27
|
+
#
|
|
28
|
+
# This number is a COST TRADEOFF, not an arbitrary limit, and the real cost is
|
|
29
|
+
# larger than "one block": `recall.py` rebuilds the <active_directives> block
|
|
30
|
+
# on EVERY UserPromptSubmit (the `format_active_directives_block` call there)
|
|
31
|
+
# with no per-session dedupe and no "unchanged since last turn" suppression,
|
|
32
|
+
# and `additionalContext` is APPENDED into the conversation. So the block is
|
|
33
|
+
# re-paid every turn and accumulates — roughly (block size) x (turn count) over
|
|
34
|
+
# a session, not once.
|
|
35
|
+
#
|
|
36
|
+
# Measured live 2026-07-25 (fleet REST `/directives`):
|
|
37
|
+
# overlord 12 active ~9.8 KB ~816 chars avg (one 2,981-char outlier)
|
|
38
|
+
# klanker 17 active ~9.9 KB ~585 chars avg
|
|
39
|
+
# gymbro 9 active ~9.3 KB ~1038 chars avg
|
|
40
|
+
# At the ~700-char fleet average, a bank sitting at this cap injects ~21 KB —
|
|
41
|
+
# on the order of 5,000-6,000 tokens per injection, per turn, cumulative across
|
|
42
|
+
# the session. 30 is chosen to clear the observed fleet maximum with headroom;
|
|
43
|
+
# it is NOT a size at which a runaway bank becomes harmless. The actual defence
|
|
44
|
+
# against pile-up is the doctor's WARN/FAIL on the active directive count
|
|
45
|
+
# (src/cli/doctor-memory.ts), not this cap. Raising it further is a legitimate
|
|
46
|
+
# call — make it deliberately, with the per-turn-times-turns cost in mind, and
|
|
47
|
+
# move the doctor thresholds with it.
|
|
48
|
+
#
|
|
49
|
+
# Banks with more active directives than this are pathological; we truncate
|
|
50
|
+
# with an in-prompt footer, a `directives_omitted` field on the recall_log row,
|
|
51
|
+
# and a stderr warning (see `format_active_directives_block`).
|
|
52
|
+
MAX_DIRECTIVES = 30
|
|
30
53
|
|
|
31
54
|
# Hard timeout for the list_directives call. The recall hook is on the
|
|
32
55
|
# UserPromptSubmit critical path — we cannot block it for long.
|
|
@@ -208,6 +231,16 @@ def invalidate_directives_cache(bank_id: Optional[str] = None) -> None:
|
|
|
208
231
|
remove_state(name)
|
|
209
232
|
|
|
210
233
|
|
|
234
|
+
def count_omitted_directives(directives: list, max_directives: int = MAX_DIRECTIVES) -> int:
|
|
235
|
+
"""How many directives `format_active_directives_block` would DROP.
|
|
236
|
+
|
|
237
|
+
Pure counterpart of the truncation branch below, so `recall.py` can put the
|
|
238
|
+
number on the recall_log row without re-deriving the cap. 0 when nothing is
|
|
239
|
+
dropped.
|
|
240
|
+
"""
|
|
241
|
+
return max(0, len(directives) - max_directives)
|
|
242
|
+
|
|
243
|
+
|
|
211
244
|
def format_active_directives_block(directives: list, max_directives: int = MAX_DIRECTIVES) -> Optional[str]:
|
|
212
245
|
"""Format directives into the <active_directives> block string.
|
|
213
246
|
|
|
@@ -252,6 +285,31 @@ def format_active_directives_block(directives: list, max_directives: int = MAX_D
|
|
|
252
285
|
if omitted > 0:
|
|
253
286
|
lines.append("")
|
|
254
287
|
lines.append(f"(+{omitted} more, omitted)")
|
|
288
|
+
# The in-prompt footer above only tells the AGENT. This stderr warn is
|
|
289
|
+
# the same channel every other operational failure in this module uses
|
|
290
|
+
# (see `_fetch_directives_with_status`), and it is a LAST-RESORT
|
|
291
|
+
# breadcrumb only — do NOT rely on it reaching an operator.
|
|
292
|
+
#
|
|
293
|
+
# Measured 2026-07-25: `docker logs --tail 20000` across all 12 running
|
|
294
|
+
# agent containers returns ZERO `[Hindsight]` lines, and nothing under
|
|
295
|
+
# ~/.switchroom/logs/ contains them either, despite months of runtime
|
|
296
|
+
# and several long-standing stderr paths in recall.py. Claude Code
|
|
297
|
+
# appears to swallow hook stderr on a zero exit, so hook stderr is not
|
|
298
|
+
# an operator-visible channel.
|
|
299
|
+
#
|
|
300
|
+
# The channels that DO reach an operator:
|
|
301
|
+
# * the `directives_omitted` field on the recall_log row
|
|
302
|
+
# (state/recall_log.jsonl — see `count_omitted_directives`), and
|
|
303
|
+
# * `switchroom doctor`'s WARN/FAIL on the bank's active directive
|
|
304
|
+
# count (src/cli/doctor-memory.ts `classifyDirectiveCount`), which
|
|
305
|
+
# reads the count from the same REST surface this module fetches.
|
|
306
|
+
print(
|
|
307
|
+
f"[Hindsight] directive truncation: {total} active directives exceeds "
|
|
308
|
+
f"MAX_DIRECTIVES={max_directives} — {omitted} lowest-priority "
|
|
309
|
+
f"directive(s) were DROPPED from this turn's prompt. Merge or retire "
|
|
310
|
+
f"directives (mental-model-curator) or raise MAX_DIRECTIVES.",
|
|
311
|
+
file=sys.stderr,
|
|
312
|
+
)
|
|
255
313
|
|
|
256
314
|
lines.append("</active_directives>")
|
|
257
315
|
return "\n".join(lines)
|