switchroom 0.19.17 → 0.19.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/bin/run-hook.sh +148 -0
  2. package/bin/workspace-dynamic-hook.sh +147 -38
  3. package/dist/agent-scheduler/index.js +13 -4
  4. package/dist/auth-broker/index.js +32 -5
  5. package/dist/cli/drive-write-pretool.mjs +48 -5
  6. package/dist/cli/ms-365-write-pretool.mjs +40 -2
  7. package/dist/cli/notion-write-pretool.mjs +13 -4
  8. package/dist/cli/switchroom.js +10614 -8104
  9. package/dist/host-control/main.js +12849 -11446
  10. package/dist/vault/approvals/kernel-server.js +90 -12
  11. package/dist/vault/broker/server.js +277 -94
  12. package/package.json +5 -3
  13. package/profiles/_base/start.sh.hbs +69 -5
  14. package/profiles/coding/CLAUDE.md.hbs +1 -1
  15. package/profiles/default/CLAUDE.md.hbs +3 -3
  16. package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
  17. package/profiles/health-coach/CLAUDE.md.hbs +1 -1
  18. package/skills/mental-model-curator/SKILL.md +8 -6
  19. package/telegram-plugin/bridge/bridge.ts +25 -19
  20. package/telegram-plugin/bridge/mcp-instructions.ts +87 -0
  21. package/telegram-plugin/dist/bridge/bridge.js +28 -20
  22. package/telegram-plugin/dist/gateway/gateway.js +2077 -1087
  23. package/telegram-plugin/dist/server.js +32 -20
  24. package/telegram-plugin/gateway/always-allow-persist-queue.ts +97 -11
  25. package/telegram-plugin/gateway/boot-card.ts +5 -1
  26. package/telegram-plugin/gateway/boot-probes.ts +113 -0
  27. package/telegram-plugin/gateway/config-approval-handler.test.ts +54 -0
  28. package/telegram-plugin/gateway/config-approval-handler.ts +16 -1
  29. package/telegram-plugin/gateway/disconnect-flush.ts +17 -0
  30. package/telegram-plugin/gateway/gateway.ts +43 -1
  31. package/telegram-plugin/gateway/handback-preturn-signal.ts +61 -7
  32. package/telegram-plugin/gateway/ipc-protocol.ts +5 -0
  33. package/telegram-plugin/gateway/ipc-server.ts +13 -0
  34. package/telegram-plugin/gateway/liveness-wiring.ts +125 -5
  35. package/telegram-plugin/gateway/missed-approvals-store.ts +66 -17
  36. package/telegram-plugin/gateway/obligation-ledger.ts +84 -4
  37. package/telegram-plugin/gateway/pending-card-store.ts +46 -16
  38. package/telegram-plugin/gateway/resume-inbound-builder.ts +13 -4
  39. package/telegram-plugin/gateway/scoped-grant-store.ts +39 -14
  40. package/telegram-plugin/gateway/store-file.ts +244 -0
  41. package/telegram-plugin/gateway/stream-render.ts +24 -5
  42. package/telegram-plugin/hooks/secret-guard-pretool.mjs +249 -76
  43. package/telegram-plugin/hooks/tool-label-pretool.mjs +88 -2
  44. package/telegram-plugin/registry/turns-schema.test.ts +8 -3
  45. package/telegram-plugin/registry/turns-schema.ts +40 -12
  46. package/telegram-plugin/runtime-metrics.ts +14 -0
  47. package/telegram-plugin/silence-poke.ts +138 -0
  48. package/telegram-plugin/tests/boot-probe-drift.test.ts +152 -0
  49. package/telegram-plugin/tests/bridge-tool-parity.test.ts +95 -0
  50. package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +32 -0
  51. package/telegram-plugin/tests/handback-preturn-signal.test.ts +62 -0
  52. package/telegram-plugin/tests/helpers/liveness-wiring-fixture.ts +178 -0
  53. package/telegram-plugin/tests/ipc-server-validate-config-approval.test.ts +95 -0
  54. package/telegram-plugin/tests/mcp-instructions-budget.test.ts +184 -0
  55. package/telegram-plugin/tests/multitopic-routing-wiring.test.ts +22 -2
  56. package/telegram-plugin/tests/obligation-determinism.test.ts +114 -3
  57. package/telegram-plugin/tests/obligation-ledger.test.ts +310 -0
  58. package/telegram-plugin/tests/registry-turns.test.ts +13 -0
  59. package/telegram-plugin/tests/resume-inbound-builder.test.ts +15 -0
  60. package/telegram-plugin/tests/secret-guard-pretool.test.ts +347 -16
  61. package/telegram-plugin/tests/silence-poke-orphan-reap.test.ts +392 -0
  62. package/telegram-plugin/tests/silence-poke-teardown-notice.test.ts +301 -0
  63. package/telegram-plugin/tests/store-atomic-write.test.ts +411 -0
  64. package/telegram-plugin/tests/stream-render-golden.test.ts +103 -1
  65. package/telegram-plugin/tests/tool-activity-summary.test.ts +9 -2
  66. package/telegram-plugin/tests/tool-label-pretool.test.ts +94 -0
  67. package/telegram-plugin/tests/tts-normalize.test.ts +43 -0
  68. package/telegram-plugin/tests/voice-normalize-text.test.ts +212 -3
  69. package/telegram-plugin/tests/worker-feed-repeat-steps.test.ts +147 -0
  70. package/telegram-plugin/tts-normalize.ts +6 -4
  71. package/telegram-plugin/voice-normalize-text.ts +168 -11
  72. package/telegram-plugin/worker-activity-feed.ts +51 -1
  73. package/vendor/hindsight-memory/CHANGELOG.md +73 -0
  74. package/vendor/hindsight-memory/scripts/drain_pending.py +668 -56
  75. package/vendor/hindsight-memory/scripts/lib/client.py +124 -0
  76. package/vendor/hindsight-memory/scripts/lib/config.py +8 -3
  77. package/vendor/hindsight-memory/scripts/lib/directives.py +62 -4
  78. package/vendor/hindsight-memory/scripts/lib/pending.py +865 -33
  79. package/vendor/hindsight-memory/scripts/lib/retain_split.py +449 -0
  80. package/vendor/hindsight-memory/scripts/recall.py +257 -12
  81. package/vendor/hindsight-memory/scripts/retain.py +12 -6
  82. package/vendor/hindsight-memory/scripts/session_start.py +48 -0
  83. package/vendor/hindsight-memory/scripts/tests/test_client_document_exists.py +470 -0
  84. package/vendor/hindsight-memory/scripts/tests/test_directives.py +80 -9
  85. package/vendor/hindsight-memory/scripts/tests/test_pending_drops.py +2121 -0
  86. package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +362 -18
  87. package/vendor/hindsight-memory/scripts/tests/test_retain_split.py +430 -0
  88. package/vendor/hindsight-memory/scripts/tests/test_session_start_version_skew.py +204 -0
  89. package/vendor/hindsight-memory/settings.json +1 -1
  90. package/vendor/hindsight-memory/tests/test_drain_pending.py +102 -6
  91. package/vendor/hindsight-memory/tests/test_pending.py +32 -7
@@ -10,32 +10,82 @@ the queue no longer drains it but the operator can still inspect via
10
10
  Boundaries
11
11
  ----------
12
12
  * Per-entry HTTP timeout: ``HINDSIGHT_DRAIN_TIMEOUT`` (default 5s), but
13
- clamped per entry to the budget still remaining (see below) so a
14
- single slow entry can never overshoot the wall-clock cap. The default
15
- timeout (5s) intentionally exceeds the default budget (4s): the clamp,
16
- not the raw timeout, is what bounds a slow entry.
13
+ clamped to the budget still remaining (see below) so a single slow
14
+ entry can never overshoot the wall-clock cap. The default timeout (5s)
15
+ intentionally exceeds the default budget (4s): the clamp, not the raw
16
+ timeout, is what bounds a slow entry.
17
17
  * Total wall-clock cap: ``HINDSIGHT_DRAIN_BUDGET_S`` (default 4s) so
18
18
  drain never blocks SessionStart longer than the upstream hook timeout
19
- permits. This is the authoritative bound; the per-entry timeout is
20
- clamped down to ``max(1, remaining budget)`` before each request, so
21
- even one slow upstream entry overshoots the budget by at most the
22
- clamp floor (~1s), not by ``HINDSIGHT_DRAIN_TIMEOUT - budget``.
19
+ permits. This is the authoritative bound. The clamp is recomputed
20
+ **before every request** — the presence GET and, if it falls through,
21
+ the POST — against the budget remaining *at that moment*, not once per
22
+ entry. Clamping once per entry spends the same allowance twice (a 9s
23
+ budget with an 8s timeout measured 16.0s), so the overshoot is bounded
24
+ by the clamp floor (~1s) only if each request re-reads the remaining
25
+ budget.
23
26
  * Stall guard: if ``STALL_THRESHOLD`` (3) consecutive entries fail with
24
27
  the same error class, we stop draining for this session — that's a
25
28
  systemic outage, not a transient flake, and continuing would only
26
29
  burn the SessionStart timeout budget. The remaining entries stay
27
30
  queued for the next session.
28
31
 
32
+ Backlog mode (switchroom #3596)
33
+ -------------------------------
34
+ The bounds above are sized for the SessionStart hook and CANNOT clear an
35
+ accumulated backlog. Worse, they CREATE one: the per-entry timeout is
36
+ clamped to the remaining hook budget (1-8s) while ``_retry_one`` posts
37
+ synchronously and a real retain takes 30-90s, so **the server commits the
38
+ document and the client always gives up before the ack**. The entry is
39
+ never deleted and is re-posted on every session start, forever, until it
40
+ hits ``MAX_ATTEMPTS`` and goes ``.dead``. The queue depth was a symptom
41
+ of that loop, not of lost memory: a full sweep of 5,751 queued entries on
42
+ this fleet (2026-07-25) found **4,048 (70.4%) already existed as
43
+ documents**, 3,815 of them with facts extracted.
44
+
45
+ ``--backlog`` is therefore a two-phase, out-of-hook replay:
46
+
47
+ * **Phase 1 — reconcile (free).** GET the document. If it exists, the
48
+ memory is already durable; retire the queue entry without a POST. No
49
+ LLM work, no cost, idempotent, resumable at any point. Only for
50
+ post-#3244 CONTENT-derived ``document_id``s — a pre-#3244 bare session
51
+ id is answered 200 by any retain in that session, so its 200 says
52
+ nothing about this entry (see ``_reconcilable_on_presence``).
53
+ * **Phase 2 — drain (real work).** Only for genuinely absent documents:
54
+ POST with a realistic timeout, then **re-GET to confirm the document
55
+ exists before retiring the entry**. A 200 is an ack, not proof
56
+ (switchroom #3244).
57
+
58
+ "Retire" never means ``os.remove``: an entry leaves the queue by MOVING
59
+ into the bounded ``pending-reconciled/`` archive (``pending.
60
+ archive_reconciled``), because every retire decision here rests on a 200.
61
+ That holds unconditionally for THIS module. It does not make the queue
62
+ immortal: if the archive cannot be written the entry stays queued, and a
63
+ disk that stays full eventually drives ``pending._evict_to_fit`` to shed the
64
+ OLDEST live entries on the ENQUEUE path (ledgered, and a ``switchroom
65
+ doctor`` failure). Nothing the drain does deletes a turn; a full disk does.
66
+
67
+ Pacing is not optional. The local model group backing retain has a small,
68
+ fixed number of lanes shared with live retains, reflect and consolidation,
69
+ so backlog replay defaults to **concurrency 1** with a sleep between
70
+ entries, and will pause entirely while an operator-supplied p95 probe
71
+ reports the upstream is already slow.
72
+
29
73
  Standalone usage::
30
74
 
31
- python3 drain_pending.py # one-shot drain, prints summary
75
+ python3 drain_pending.py # bounded in-hook drain
76
+ python3 drain_pending.py --backlog # two-phase backlog replay
77
+ python3 drain_pending.py --backlog --phase reconcile # free pass only
78
+ python3 drain_pending.py --backlog --dry-run
32
79
  """
33
80
 
34
81
  from __future__ import annotations
35
82
 
83
+ import argparse
36
84
  import os
85
+ import subprocess
37
86
  import sys
38
87
  import time
88
+ from concurrent.futures import ThreadPoolExecutor
39
89
 
40
90
  sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
41
91
 
@@ -43,16 +93,77 @@ from lib.client import HindsightClient
43
93
  from lib.config import debug_log, load_config
44
94
  from lib.pending import (
45
95
  MAX_ATTEMPTS,
46
- delete_entry,
96
+ archive_reconciled,
97
+ is_content_derived_document_id,
47
98
  iter_entries,
48
99
  mark_dead,
49
100
  update_attempt,
50
101
  )
102
+ from lib.retain_split import retain_client_deadline
51
103
 
52
104
 
53
105
  STALL_THRESHOLD = 3
54
106
 
55
107
 
108
+ def _clamp(timeout: int, budget: float, started: float) -> int:
109
+ """Per-REQUEST HTTP timeout: ``timeout`` capped by the budget LEFT NOW.
110
+
111
+ Called immediately before each request — the presence GET and the POST
112
+ — not once per entry. Computing it once and spending it twice makes the
113
+ per-entry cost additive: with the fleet's own settings (budget 9s,
114
+ timeout 8s) a single slow entry measured 16.0s against a 9s budget,
115
+ while the module docstring promised an overshoot of at most the clamp
116
+ floor.
117
+
118
+ Floor at 1s so a near-exhausted budget still gets one bounded shot
119
+ rather than a 0s (instant-fail) request. That floor is the entire
120
+ overshoot: at most ~1s per request, ~2s for a GET+POST entry.
121
+
122
+ The 1s floor is expressed TWICE and the two are mutually redundant:
123
+ the early ``remaining < 1`` return and the ``max(1, ...)`` below each
124
+ enforce it alone (``int(0.5)`` → 0 → 1; ``int(-5)`` → -5 → 1). That
125
+ redundancy — not a coverage hole — is why single mutations here
126
+ survive: ``<`` → ``<=``, ``max(1`` → ``max(0``, and deleting either
127
+ guard are all EQUIVALENT mutants, verified by exhaustive comparison
128
+ over the boundary plus 200k random remainders (#3599 review R3-L4,
129
+ which reported it as unpinned). What matters is the floor's OUTCOME,
130
+ and that IS pinned: removing both guards fails
131
+ ``ClampAndEnvKnobBoundaryTest.test_clamp_floors_at_one_second_never_zero``
132
+ and ``test_session_start_reconcile_respects_the_hook_budget``. Left as
133
+ two lines deliberately — the shortcut states the intent where a reader
134
+ looks for it, and no mutation of it can change behaviour.
135
+
136
+ THE EQUIVALENCE HAS A PRECONDITION, and it is not this function's
137
+ (#3599 review R4-Lb — the paragraph above used to claim it
138
+ unconditionally). ``max(1, min(timeout, int(remaining)))`` collapses to
139
+ ``max(0, ...)`` only while ``timeout >= 1``. With ``timeout == 0`` a
140
+ healthy budget takes the ``max`` branch and the ``max(0`` mutant returns
141
+ **0** — the instant-fail request the floor exists to prevent, on every
142
+ entry. The only callsite invariant that rules this out is
143
+ ``_per_entry_timeout()``'s own ``max(1, v)``, which is what turns
144
+ ``HINDSIGHT_DRAIN_TIMEOUT=0`` into 1. Both callers below take their
145
+ ``timeout`` from it (``drain``'s local, line ~396). So the honest
146
+ statement is: equivalent FOR EVERY REACHABLE INPUT, because
147
+ ``_per_entry_timeout`` floors first — a fact pinned by
148
+ ``ClampAndEnvKnobBoundaryTest.test_the_per_entry_timeout_is_floored_at_one_second``,
149
+ not by anything here. Change that floor and this proof dies with it.
150
+ """
151
+ remaining = budget - (time.monotonic() - started)
152
+ if remaining < 1:
153
+ return 1
154
+ return max(1, min(timeout, int(remaining)))
155
+
156
+
157
+ def _reconcilable_on_presence(entry: dict) -> bool:
158
+ """May a bare presence GET retire this entry? See ``pending.py``.
159
+
160
+ Only for post-#3244 content-derived ``document_id``s. A pre-#3244
161
+ entry's id is a bare session id, so a 200 reflects *some* retain in
162
+ that session, not this entry's content.
163
+ """
164
+ return is_content_derived_document_id(entry.get("document_id"))
165
+
166
+
56
167
  def _per_entry_timeout() -> int:
57
168
  raw = os.environ.get("HINDSIGHT_DRAIN_TIMEOUT", "5")
58
169
  try:
@@ -71,13 +182,134 @@ def _budget_seconds() -> float:
71
182
  return 4.0
72
183
 
73
184
 
185
+ def _env_num(name: str, default, cast=float, lo=None, hi=None):
186
+ """Read a numeric env knob, falling back to ``default`` on garbage."""
187
+ try:
188
+ v = cast(os.environ.get(name) or default)
189
+ except (TypeError, ValueError):
190
+ v = cast(default)
191
+ if lo is not None:
192
+ v = max(lo, v)
193
+ if hi is not None:
194
+ v = min(hi, v)
195
+ return v
196
+
197
+
198
+ def _backlog_timeout() -> int:
199
+ """Per-entry HTTP timeout in backlog mode.
200
+
201
+ A synchronous retain takes 30-90s on this fleet, so the SessionStart
202
+ default (5-8s, further clamped by the hook budget) guarantees a
203
+ client-side timeout on a request the server then commits anyway.
204
+
205
+ The default is DERIVED, not a literal: it is
206
+ ``retain_split.retain_client_deadline()`` (280s), the same deadline the
207
+ retain content bound is sized against. Those two must be ONE number.
208
+ This function shipped as a bare ``180`` (#3599), and against a 180s
209
+ deadline both halves of the retain budget break: a maximally-sized part
210
+ is ~276s of sequential extraction, and the SERVER per-call timeout
211
+ derived in ``src/setup/hindsight.ts`` (#3611) is 204s — so the drain
212
+ client would abandon a request the server is still legitimately working
213
+ on, leave the entry queued, and rebuild the re-post loop #3599 exists to
214
+ kill, one size class up.
215
+
216
+ ``HINDSIGHT_DRAIN_BACKLOG_TIMEOUT`` still overrides it outright;
217
+ ``HINDSIGHT_RETAIN_CLIENT_DEADLINE_S`` moves the derivation and the
218
+ content bound together.
219
+ """
220
+ return _env_num(
221
+ "HINDSIGHT_DRAIN_BACKLOG_TIMEOUT",
222
+ int(retain_client_deadline()),
223
+ int,
224
+ lo=1,
225
+ )
226
+
227
+
228
+ def _backlog_budget_seconds() -> float:
229
+ """Total wall-clock cap in backlog mode (default 1h)."""
230
+ return _env_num("HINDSIGHT_DRAIN_BACKLOG_BUDGET_S", 3600, float, lo=1.0)
231
+
232
+
233
+ def _backlog_concurrency() -> int:
234
+ """Entries retried in parallel in backlog mode.
235
+
236
+ **Defaults to 1, deliberately.** The local model group serving retain
237
+ has a small fixed number of lanes (4 on this fleet: 2 boxes x 2 slots)
238
+ SHARED with live retains, reflect and consolidation. A drain at width
239
+ 4 consumes the whole pool; run per-agent across 11 agents and it is
240
+ 44 lanes of demand against 4, which trips the latency watchdog. One
241
+ lane leaves the rest for live work. Raise it only if you know the
242
+ pool is idle.
243
+ """
244
+ return _env_num("HINDSIGHT_DRAIN_CONCURRENCY", 1, int, lo=1, hi=16)
245
+
246
+
247
+ def _backlog_sleep_seconds() -> float:
248
+ """Pause between phase-2 retains, so replay never runs flat out."""
249
+ return _env_num("HINDSIGHT_DRAIN_SLEEP_S", 2.0, float, lo=0.0)
250
+
251
+
252
+ def _p95_backoff_ms() -> int:
253
+ """Pause phase 2 while the upstream p95 exceeds this."""
254
+ return _env_num("HINDSIGHT_DRAIN_P95_BACKOFF_MS", 38000, int, lo=0)
255
+
256
+
257
+ def _p95_probe_ms() -> int:
258
+ """Current upstream p95 in ms, or ``-1`` when unknown.
259
+
260
+ The probe is an operator-supplied command (``HINDSIGHT_DRAIN_P95_CMD``)
261
+ that prints a millisecond figure on stdout. It is NOT built in: the
262
+ authoritative latency figure on this fleet lives in LiteLLM's spend
263
+ log in postgres, which an agent container cannot reach — inventing a
264
+ weaker in-container proxy for it would be a worse signal that looks
265
+ like a better one. Unset ⇒ no backoff, and ``--backlog`` says so.
266
+
267
+ A CONFIGURED-BUT-BROKEN probe is not the same as an unset one, and
268
+ used to be indistinguishable: ``out.returncode`` was ignored, so a
269
+ typo'd or unauthorized command produced empty stdout, ``int()`` raised,
270
+ and the bare ``except`` returned ``-1`` — silently disabling the very
271
+ backoff the operator had asked for. It still returns ``-1`` (a replay
272
+ that halts because its probe is broken is worse than one that runs
273
+ unpaced), but it now says so loudly on stderr, naming the exit status
274
+ and stderr tail, so the gap is visible in the drain log.
275
+ """
276
+ cmd = os.environ.get("HINDSIGHT_DRAIN_P95_CMD")
277
+ if not cmd:
278
+ return -1
279
+ try:
280
+ out = subprocess.run(
281
+ cmd, shell=True, capture_output=True, text=True, timeout=45
282
+ )
283
+ except Exception as e:
284
+ _blog(
285
+ f"p95 probe FAILED to run ({type(e).__name__}: {e}) — backoff is "
286
+ f"DISABLED for this run. Fix HINDSIGHT_DRAIN_P95_CMD."
287
+ )
288
+ return -1
289
+ if out.returncode != 0:
290
+ _blog(
291
+ f"p95 probe exited {out.returncode} — backoff is DISABLED for this "
292
+ f"run. Fix HINDSIGHT_DRAIN_P95_CMD. stderr: "
293
+ f"{(out.stderr or '').strip()[:200]}"
294
+ )
295
+ return -1
296
+ try:
297
+ return int(out.stdout.strip().splitlines()[-1])
298
+ except (ValueError, IndexError):
299
+ _blog(
300
+ f"p95 probe exited 0 but printed no millisecond figure — backoff is "
301
+ f"DISABLED for this run. stdout: {(out.stdout or '').strip()[:200]!r}"
302
+ )
303
+ return -1
304
+
305
+
74
306
  def _retry_one(entry: dict, timeout: int) -> None:
75
307
  """POST a single queued retain. Raises on failure.
76
308
 
77
309
  Posts ``async_processing=False`` (commit-before-ack, switchroom #3244 §1.1):
78
- the drain is a DURABILITY path — it deletes the pending entry on a 200, so
310
+ the drain is a DURABILITY path — it retires the pending entry on a 200, so
79
311
  the 200 must prove durable persistence, not merely ack-of-receipt. A bare
80
- async 200 followed by a dropped extraction would delete the queue entry
312
+ async 200 followed by a dropped extraction would retire the queue entry
81
313
  while the content never lands, and (for boot-reconcile remainders whose
82
314
  watermark already advanced) there is no reconcile backstop — silent loss
83
315
  (the #3244 bug). All drained entries — Stop-hook A2 failures, SessionEnd
@@ -97,29 +329,116 @@ def _retry_one(entry: dict, timeout: int) -> None:
97
329
  )
98
330
 
99
331
 
100
- def drain(config: dict | None = None) -> dict:
332
+ def _document_state(entry: dict, timeout: int = 30):
333
+ """Tri-state presence of this entry's document. See ``document_exists``.
334
+
335
+ ``True`` present / ``False`` absent / ``None`` unknown. Never raises —
336
+ an unknown must never be mistaken for an absence (which would re-POST
337
+ a durable document) nor for a presence (which would delete the last
338
+ on-disk copy of a turn).
339
+ """
340
+ did = entry.get("document_id")
341
+ bank = entry.get("bank_id")
342
+ if not did or not bank:
343
+ return None
344
+ try:
345
+ client = HindsightClient(entry["api_url"], entry.get("api_token"))
346
+ return client.document_exists(bank, did, timeout=timeout)
347
+ except Exception:
348
+ return None
349
+
350
+
351
+ def _record_failure(
352
+ config: dict,
353
+ path: str,
354
+ entry: dict,
355
+ e: Exception,
356
+ summary: dict,
357
+ ) -> str:
358
+ """Apply the per-entry failure policy. Returns the error class name.
359
+
360
+ Shared by the sequential (SessionStart) and backlog drains so both age
361
+ entries toward ``.dead`` on exactly the same schedule.
362
+ """
363
+ err_class = type(e).__name__
364
+ attempts = int(entry.get("attempt_count", 1))
365
+ if attempts >= MAX_ATTEMPTS:
366
+ marker = mark_dead(path, entry)
367
+ summary["dead"] += 1
368
+ print(
369
+ f"[Hindsight] drain_pending: entry exceeded {MAX_ATTEMPTS} "
370
+ f"attempts, marking dead at {marker} (last error: {err_class}: {e})",
371
+ file=sys.stderr,
372
+ )
373
+ else:
374
+ update_attempt(path, entry, e)
375
+ summary["retried"] += 1
376
+ debug_log(
377
+ config,
378
+ f"drain_pending: retry {attempts}/{MAX_ATTEMPTS} failed for {path} ({err_class}: {e})",
379
+ )
380
+ return err_class
381
+
382
+
383
+ def _new_summary() -> dict:
384
+ return {
385
+ "drained": 0,
386
+ "retried": 0,
387
+ "dead": 0,
388
+ "reconciled": 0,
389
+ "unknown": 0,
390
+ # Presence WAS established, but the entry could not be moved into
391
+ # the archive (ENOSPC, EACCES, read-only mount). It is still queued
392
+ # — archiving never falls back to a delete (#3599 review R3-M1) —
393
+ # so it must not be counted as drained/reconciled, which would
394
+ # report a retire that did not happen.
395
+ "archive_failed": 0,
396
+ "stalled": False,
397
+ "budget_exceeded": False,
398
+ }
399
+
400
+
401
+ def drain_backlog(config: dict | None = None, **kw) -> dict:
402
+ """Two-phase backlog replay, off the SessionStart budget entirely.
403
+
404
+ See the module docstring. Summary shape is ``drain()``'s plus
405
+ ``reconciled`` (already durable — no POST issued) and ``unknown``
406
+ (presence could not be established; left queued).
407
+ """
408
+ return drain(config, backlog=True, **kw)
409
+
410
+
411
+ def drain(
412
+ config: dict | None = None,
413
+ backlog: bool = False,
414
+ phase: str = "both",
415
+ dry_run: bool = False,
416
+ ) -> dict:
101
417
  """Walk the pending-retains directory and retry each entry.
102
418
 
419
+ ``backlog=False`` (default) is the bounded in-hook drain.
420
+ ``backlog=True`` is the operator backlog replay — see ``drain_backlog()``.
421
+
103
422
  Returns a summary dict::
104
423
 
105
- {"drained": int, # successful retries (entries deleted)
424
+ {"drained": int, # successful retries (entries archived)
106
425
  "retried": int, # failures kept for next session
107
426
  "dead": int, # entries promoted to .dead this run
427
+ "reconciled": int,# already durable, archived without a POST
428
+ "unknown": int, # presence unknown, left queued
429
+ "archive_failed": int, # durable, but the archive was unwritable
430
+ # so the entry is STILL QUEUED
108
431
  "stalled": bool, # stall guard tripped
109
432
  "budget_exceeded": bool}
110
433
  """
111
434
  config = config or load_config()
435
+ if backlog:
436
+ return _drain_backlog_impl(config, phase=phase, dry_run=dry_run)
112
437
  timeout = _per_entry_timeout()
113
438
  budget = _budget_seconds()
114
439
  started = time.monotonic()
115
440
 
116
- summary = {
117
- "drained": 0,
118
- "retried": 0,
119
- "dead": 0,
120
- "stalled": False,
121
- "budget_exceeded": False,
122
- }
441
+ summary = _new_summary()
123
442
 
124
443
  entries = iter_entries()
125
444
  if not entries:
@@ -138,41 +457,52 @@ def drain(config: dict | None = None) -> dict:
138
457
  debug_log(config, "drain_pending: total budget exceeded, stopping")
139
458
  break
140
459
 
141
- # Clamp the per-entry HTTP timeout to the budget still remaining
142
- # (#1094 item 2). Without this, a single slow entry using the full
143
- # HINDSIGHT_DRAIN_TIMEOUT (default 5s) overshoots the total budget
144
- # (default 4s). Floor at 1s so we still give a near-exhausted
145
- # budget one bounded shot rather than a 0s (instant-fail) request.
146
- remaining = budget - elapsed
147
- effective_timeout = max(1, min(timeout, int(remaining) if remaining >= 1 else 1))
460
+ # RECONCILE BEFORE RETRY — this is the fix for the re-post loop in
461
+ # the path that actually runs on every boot, not just in --backlog.
462
+ # 70.4% of the measured fleet backlog already existed as documents;
463
+ # re-POSTing those inside the hook is a guaranteed client timeout
464
+ # (the clamp is far below a 30-90s synchronous retain), 30-90s of
465
+ # wasted server-side extraction, and one more step toward .dead for
466
+ # a memory that was never actually lost.
467
+ #
468
+ # A GET is sub-second and it REPLACES that doomed POST, so this
469
+ # strictly reduces both hook latency and upstream load. Only a
470
+ # definite True retires the entry: False and None (unknown) fall
471
+ # through to the retry, because guessing "present" would retire the
472
+ # last on-disk copy of a turn.
473
+ #
474
+ # GATED ON THE ID SHAPE. A presence GET only proves *this* entry's
475
+ # content was committed when the document_id is content-derived
476
+ # (post-#3244 `retain.slice_document_id`). A pre-#3244 entry carries
477
+ # a bare session id, for which the bank answers 200 after ANY
478
+ # successful retain in that session — reconciling on that deletes a
479
+ # turn that was never committed. Such entries skip the free pass and
480
+ # go straight to the POST, which is the only thing that can make
481
+ # them durable.
482
+ if _reconcilable_on_presence(entry):
483
+ if _document_state(entry, timeout=_clamp(timeout, budget, started)) is True:
484
+ if archive_reconciled(path):
485
+ summary["reconciled"] += 1
486
+ else:
487
+ # Archive unwritable: the entry is STILL QUEUED (it is
488
+ # never deleted), so calling it reconciled would be a lie.
489
+ summary["archive_failed"] += 1
490
+ consecutive_failures = 0
491
+ last_error_class = None
492
+ continue
148
493
 
149
494
  try:
150
- _retry_one(entry, timeout=effective_timeout)
495
+ # Re-clamp: the GET above spent part of the budget, and the same
496
+ # allowance must not be handed out twice (#3599 review F2).
497
+ _retry_one(entry, timeout=_clamp(timeout, budget, started))
151
498
  except Exception as e:
152
- err_class = type(e).__name__
499
+ err_class = _record_failure(config, path, entry, e, summary)
153
500
  if err_class == last_error_class:
154
501
  consecutive_failures += 1
155
502
  else:
156
503
  consecutive_failures = 1
157
504
  last_error_class = err_class
158
505
 
159
- attempts = int(entry.get("attempt_count", 1))
160
- if attempts >= MAX_ATTEMPTS:
161
- marker = mark_dead(path, entry)
162
- summary["dead"] += 1
163
- print(
164
- f"[Hindsight] drain_pending: entry exceeded {MAX_ATTEMPTS} "
165
- f"attempts, marking dead at {marker} (last error: {err_class}: {e})",
166
- file=sys.stderr,
167
- )
168
- else:
169
- update_attempt(path, entry, e)
170
- summary["retried"] += 1
171
- debug_log(
172
- config,
173
- f"drain_pending: retry {attempts}/{MAX_ATTEMPTS} failed for {path} ({err_class}: {e})",
174
- )
175
-
176
506
  if consecutive_failures >= STALL_THRESHOLD:
177
507
  summary["stalled"] = True
178
508
  print(
@@ -184,23 +514,305 @@ def drain(config: dict | None = None) -> dict:
184
514
  break
185
515
  continue
186
516
 
187
- # Success — delete the entry.
188
- delete_entry(path)
189
- summary["drained"] += 1
517
+ # Success — retire the entry. ARCHIVED, not deleted.
518
+ #
519
+ # The evidence here is the POST's own 200 and nothing else: a 5s
520
+ # hook budget has no room for the confirming re-GET the backlog
521
+ # drain issues. Not a BARE 200 though (#3599 review R4-B3 corrects
522
+ # this comment, which used to say so): ``_retry_one`` posts
523
+ # ``async_processing=False``, so the 200 is a commit-before-ack
524
+ # (#3244 §1.1) — the daemon's statement that it durably committed,
525
+ # not merely that it received. Weaker than an independent read,
526
+ # much stronger than an async ack.
527
+ #
528
+ # The residual risk is a daemon that does not honour ``async=false``.
529
+ # Archiving rather than deleting keeps a recoverable copy for that
530
+ # case — bounded by ``pending-reconciled/``'s caps, so recoverable
531
+ # has a horizon; ``pending._trim_dir`` documents exactly what that
532
+ # horizon costs.
533
+ if archive_reconciled(path):
534
+ summary["drained"] += 1
535
+ else:
536
+ summary["archive_failed"] += 1
190
537
  consecutive_failures = 0
191
538
  last_error_class = None
192
539
 
193
540
  return summary
194
541
 
195
542
 
196
- def main() -> int:
543
+ def _blog(msg: str) -> None:
544
+ print(f"[Hindsight] drain_pending(backlog): {msg}", file=sys.stderr)
545
+
546
+
547
+ def _reconcile_phase(config: dict, summary: dict, dry_run: bool) -> None:
548
+ """PHASE 1 — free pass: drop entries whose document already exists.
549
+
550
+ This is the phase that makes backlog replay affordable. 70.4% of a
551
+ measured 5,751-entry fleet backlog was already durable; POSTing those
552
+ is duplicated LLM extraction for zero new memory. A GET costs nothing
553
+ on the model pool.
554
+
555
+ Only a definite ``True`` retires an entry, and only for a post-#3244
556
+ content-derived ``document_id`` (see ``_reconcilable_on_presence``) —
557
+ a bare session id's 200 says nothing about *this* entry's content.
558
+ ``False`` leaves it for phase 2; ``None`` (unknown) leaves it queued
559
+ and is counted — never guessed. Retiring MOVES the entry into
560
+ ``pending-reconciled/``; it is never ``os.remove``d.
561
+ """
562
+ entries = iter_entries()
563
+ if not entries:
564
+ return
565
+ _blog(f"phase 1 reconcile: checking {len(entries)} entries (no LLM cost)")
566
+ skipped = 0
567
+ for path, entry in entries:
568
+ if not _reconcilable_on_presence(entry):
569
+ skipped += 1
570
+ continue
571
+ state = _document_state(entry)
572
+ if state is True:
573
+ if dry_run or archive_reconciled(path):
574
+ summary["reconciled"] += 1
575
+ else:
576
+ summary["archive_failed"] += 1
577
+ elif state is None:
578
+ summary["unknown"] += 1
579
+ if skipped:
580
+ _blog(
581
+ f"phase 1: {skipped} entries have a pre-#3244 (bare session) "
582
+ f"document_id — a presence GET cannot prove THEIR content was "
583
+ f"committed, so they go to phase 2 rather than the free pass"
584
+ )
585
+ _blog(
586
+ f"phase 1 done: {summary['reconciled']} already durable "
587
+ f"(no POST issued), {summary['unknown']} unknown (left queued)"
588
+ + (
589
+ f", {summary['archive_failed']} confirmed durable but NOT retired "
590
+ f"(archive unwritable — still queued)"
591
+ if summary["archive_failed"]
592
+ else ""
593
+ )
594
+ )
595
+
596
+
597
+ def _wait_for_upstream(backoff_ms: int, started: float, budget: float) -> bool:
598
+ """Block while the p95 probe says upstream is slow. False ⇒ give up.
599
+
600
+ The wait is bounded by the SAME wall-clock budget as the drain itself.
601
+ An unbounded ``while True: sleep(120)`` would let a persistently
602
+ degraded upstream block ``drain_backlog()`` indefinitely, long past the
603
+ budget the operator set — the one guarantee this mode makes about how
604
+ long it will run.
605
+ """
606
+ if backoff_ms <= 0:
607
+ return True
608
+ while True:
609
+ p95 = _p95_probe_ms()
610
+ if p95 < 0 or p95 <= backoff_ms:
611
+ return True
612
+ if time.monotonic() - started + 120 > budget:
613
+ _blog(
614
+ f"upstream still slow (p95={p95}ms > {backoff_ms}ms) and the "
615
+ f"budget is exhausted — stopping rather than waiting past it. "
616
+ f"Remaining entries stay queued; re-run when upstream recovers."
617
+ )
618
+ return False
619
+ _blog(
620
+ f"BACKOFF: upstream p95={p95}ms > {backoff_ms}ms — pausing 120s so "
621
+ f"the replay never becomes the cause of a latency alarm"
622
+ )
623
+ time.sleep(120)
624
+
625
+
626
+ def _drain_backlog_impl(
627
+ config: dict, phase: str = "both", dry_run: bool = False
628
+ ) -> dict:
629
+ """Concurrent-capable, long-budget, two-phase backlog replay."""
630
+ summary = _new_summary()
631
+
632
+ if phase in ("reconcile", "both"):
633
+ _reconcile_phase(config, summary, dry_run)
634
+ if phase == "reconcile":
635
+ return summary
636
+
637
+ timeout = _backlog_timeout()
638
+ budget = _backlog_budget_seconds()
639
+ width = _backlog_concurrency()
640
+ sleep_s = _backlog_sleep_seconds()
641
+ backoff_ms = _p95_backoff_ms()
642
+ started = time.monotonic()
643
+
644
+ entries = iter_entries()
645
+ if not entries:
646
+ _blog("phase 2: nothing left to retain")
647
+ return summary
648
+
649
+ _blog(
650
+ f"phase 2 drain: {len(entries)} entries genuinely need retaining "
651
+ f"(concurrency={width} timeout={timeout}s budget={budget:.0f}s "
652
+ f"sleep={sleep_s}s p95_probe="
653
+ f"{'on' if os.environ.get('HINDSIGHT_DRAIN_P95_CMD') else 'unset'})"
654
+ )
655
+ if dry_run:
656
+ _blog("dry run — no retains issued")
657
+ return summary
658
+
659
+ consecutive_failures = 0
660
+ last_error_class: str | None = None
661
+
662
+ # BUDGET GRANULARITY: checked between waves, not mid-wave, so a run can
663
+ # overshoot `budget` by at most one wave — up to HINDSIGHT_DRAIN_BACKLOG_
664
+ # TIMEOUT (280s default, `_backlog_timeout`) plus the confirming GETs.
665
+ # That is deliberate:
666
+ # abandoning an in-flight wave would leave entries whose POST the server
667
+ # is still committing, and the whole point of commit-before-delete is not
668
+ # to guess about those. Unlike the in-hook drain (whose overshoot is
669
+ # clamped to ~1s because a SessionStart hook has a hard deadline), this
670
+ # mode has no deadline to miss — so a bounded overshoot is the cheaper
671
+ # trade.
672
+ for start in range(0, len(entries), width):
673
+ if time.monotonic() - started > budget:
674
+ summary["budget_exceeded"] = True
675
+ _blog("budget exhausted, stopping. Remaining entries stay queued — re-run to continue.")
676
+ break
677
+
678
+ if not _wait_for_upstream(backoff_ms, started, budget):
679
+ summary["budget_exceeded"] = True
680
+ break
681
+
682
+ wave = entries[start : start + width]
683
+ # Results are collected in SUBMISSION order (not completion order)
684
+ # so the stall guard sees a deterministic sequence and behaves
685
+ # identically to the sequential drain.
686
+ with ThreadPoolExecutor(max_workers=width) as pool:
687
+ futures = [
688
+ pool.submit(_retry_one, entry, timeout) for _path, entry in wave
689
+ ]
690
+ outcomes = []
691
+ for fut in futures:
692
+ try:
693
+ fut.result()
694
+ outcomes.append(None)
695
+ except Exception as e: # noqa: BLE001 — per-entry policy below
696
+ outcomes.append(e)
697
+
698
+ for (path, entry), err in zip(wave, outcomes):
699
+ if err is None:
700
+ # COMMIT-BEFORE-RETIRE. A 200 is an ack, not proof the
701
+ # document is durable (switchroom #3244), so re-GET before
702
+ # retiring the last on-disk copy. Anything other than a
703
+ # definite True keeps the entry. This confirming GET is
704
+ # meaningful even for a pre-#3244 bare-session id: unlike
705
+ # the free reconcile pass, it is corroborated by the
706
+ # synchronous POST of THIS entry's content that just
707
+ # returned 200. And the entry is archived, not deleted.
708
+ if _document_state(entry) is True:
709
+ if archive_reconciled(path):
710
+ summary["drained"] += 1
711
+ else:
712
+ summary["archive_failed"] += 1
713
+ else:
714
+ summary["unknown"] += 1
715
+ _blog(
716
+ f"posted but document not confirmed, keeping entry: "
717
+ f"{os.path.basename(path)}"
718
+ )
719
+ consecutive_failures = 0
720
+ last_error_class = None
721
+ continue
722
+
723
+ # STALL GUARD, evaluated BEFORE the failure is recorded for the
724
+ # rest of the wave. Recording first would let a wave of `width`
725
+ # identical timeouts bump attempt_count on all of them before
726
+ # the loop breaks — aging entries toward .dead FASTER than the
727
+ # sequential drain, the opposite of the point. Counting first
728
+ # and breaking immediately after the tripping entry makes the
729
+ # two paths bump exactly the same number of entries.
730
+ err_class = type(err).__name__
731
+ if err_class == last_error_class:
732
+ consecutive_failures += 1
733
+ else:
734
+ consecutive_failures = 1
735
+ last_error_class = err_class
736
+
737
+ _record_failure(config, path, entry, err, summary)
738
+
739
+ if consecutive_failures >= STALL_THRESHOLD:
740
+ summary["stalled"] = True
741
+ _blog(
742
+ f"{consecutive_failures} consecutive failures with "
743
+ f"{err_class}, stalling. Fix the upstream, then re-run. "
744
+ f"Remaining entries stay queued."
745
+ )
746
+ break
747
+
748
+ if summary["stalled"]:
749
+ break
750
+ if sleep_s:
751
+ time.sleep(sleep_s)
752
+
753
+ return summary
754
+
755
+
756
+ def _parse_args(argv: list[str] | None):
757
+ """Real argument parsing.
758
+
759
+ Was a bare ``"--backlog" in argv`` membership test, which silently ran
760
+ the 4-second in-hook drain on a typo like ``--backlogg`` while the
761
+ operator believed they had started a backlog replay.
762
+ """
763
+ ap = argparse.ArgumentParser(
764
+ prog="drain_pending.py",
765
+ description="Drain the hindsight pending-retains queue.",
766
+ )
767
+ ap.add_argument(
768
+ "--backlog",
769
+ action="store_true",
770
+ help="two-phase backlog replay, off the SessionStart budget",
771
+ )
772
+ ap.add_argument(
773
+ "--phase",
774
+ choices=["reconcile", "drain", "both"],
775
+ default="both",
776
+ help="with --backlog: run only the free reconcile pass, only the "
777
+ "retain pass, or both (default)",
778
+ )
779
+ ap.add_argument(
780
+ "--dry-run",
781
+ action="store_true",
782
+ help="with --backlog: report what would happen, issue no writes",
783
+ )
784
+ return ap.parse_args(argv)
785
+
786
+
787
+ def main(argv: list[str] | None = None) -> int:
788
+ args = _parse_args(sys.argv[1:] if argv is None else argv)
789
+ if (args.phase != "both" or args.dry_run) and not args.backlog:
790
+ print(
791
+ "[Hindsight] drain_pending: --phase/--dry-run require --backlog",
792
+ file=sys.stderr,
793
+ )
794
+ return 2
197
795
  config = load_config()
198
- summary = drain(config)
199
- if summary["drained"] or summary["retried"] or summary["dead"]:
796
+ summary = drain(
797
+ config, backlog=args.backlog, phase=args.phase, dry_run=args.dry_run
798
+ )
799
+ if any(
800
+ summary[k]
801
+ for k in (
802
+ "drained",
803
+ "retried",
804
+ "dead",
805
+ "reconciled",
806
+ "unknown",
807
+ "archive_failed",
808
+ )
809
+ ):
200
810
  print(
201
- f"[Hindsight] drain_pending: "
202
- f"drained={summary['drained']} retried={summary['retried']} "
203
- f"dead={summary['dead']} "
811
+ f"[Hindsight] drain_pending{'(backlog)' if args.backlog else ''}: "
812
+ f"drained={summary['drained']} reconciled={summary['reconciled']} "
813
+ f"retried={summary['retried']} dead={summary['dead']} "
814
+ f"unknown={summary['unknown']} "
815
+ f"archive_failed={summary['archive_failed']} "
204
816
  f"stalled={summary['stalled']} budget_exceeded={summary['budget_exceeded']}",
205
817
  file=sys.stderr,
206
818
  )