switchroom 0.18.24 → 0.18.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/dist/cli/switchroom.js +59 -11
  2. package/dist/host-control/main.js +1 -1
  3. package/package.json +1 -1
  4. package/telegram-plugin/dist/bridge/bridge.js +26 -0
  5. package/telegram-plugin/dist/gateway/gateway.js +1489 -829
  6. package/telegram-plugin/dist/server.js +26 -0
  7. package/telegram-plugin/gateway/callback-query-handlers.ts +7 -0
  8. package/telegram-plugin/gateway/gateway.ts +314 -3
  9. package/telegram-plugin/gateway/model-command.ts +188 -56
  10. package/telegram-plugin/gateway/redelivery-decision.ts +139 -0
  11. package/telegram-plugin/gateway/vault-grant-inbound-builders.ts +42 -1
  12. package/telegram-plugin/history.ts +118 -0
  13. package/telegram-plugin/registry/turns-schema.ts +89 -1
  14. package/telegram-plugin/session-tail.ts +185 -0
  15. package/telegram-plugin/subagent-watcher.ts +45 -0
  16. package/telegram-plugin/tests/crash-redelivery-resume-exclusion.test.ts +133 -0
  17. package/telegram-plugin/tests/crash-redelivery-wiring.test.ts +72 -0
  18. package/telegram-plugin/tests/history.test.ts +91 -0
  19. package/telegram-plugin/tests/model-command.test.ts +189 -12
  20. package/telegram-plugin/tests/redelivery-decision.test.ts +84 -0
  21. package/telegram-plugin/tests/registry-turns.test.ts +51 -0
  22. package/telegram-plugin/tests/session-model-source.test.ts +11 -0
  23. package/telegram-plugin/tests/session-tail.test.ts +145 -0
  24. package/telegram-plugin/tests/subagent-watcher.test.ts +50 -0
  25. package/telegram-plugin/tests/tool-activity-summary.test.ts +109 -0
  26. package/telegram-plugin/tests/trailing-answer-projector.test.ts +124 -0
  27. package/telegram-plugin/tests/vault-grant-inbound-builders.test.ts +125 -0
  28. package/telegram-plugin/tests/worker-feed-pin-persistence.test.ts +306 -0
  29. package/telegram-plugin/tool-activity-summary.ts +54 -3
  30. package/telegram-plugin/worker-activity-feed.ts +104 -0
  31. package/vendor/hindsight-memory/scripts/backfill_transcripts.py +762 -0
  32. package/vendor/hindsight-memory/scripts/drain_pending.py +13 -1
  33. package/vendor/hindsight-memory/scripts/lib/client.py +14 -4
  34. package/vendor/hindsight-memory/scripts/lib/config.py +8 -0
  35. package/vendor/hindsight-memory/scripts/lib/pacing.py +102 -0
  36. package/vendor/hindsight-memory/scripts/lib/watermark.py +213 -0
  37. package/vendor/hindsight-memory/scripts/reconcile_tail.py +344 -0
  38. package/vendor/hindsight-memory/scripts/retain.py +299 -143
  39. package/vendor/hindsight-memory/scripts/session_start.py +14 -0
  40. package/vendor/hindsight-memory/scripts/tests/test_backfill.py +362 -0
  41. package/vendor/hindsight-memory/scripts/tests/test_reconcile_durability.py +350 -0
  42. package/vendor/hindsight-memory/tests/test_hooks.py +8 -2
@@ -17,6 +17,7 @@ Exit codes:
17
17
  0 — always (graceful degradation on any error)
18
18
  """
19
19
 
20
+ import hashlib
20
21
  import json
21
22
  import os
22
23
  import sys
@@ -24,6 +25,7 @@ import time
24
25
 
25
26
  sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
26
27
 
28
+ from lib import watermark
27
29
  from lib.bank import derive_bank_id, ensure_bank_mission
28
30
  from lib.client import HindsightClient
29
31
  from lib.config import debug_log, load_config
@@ -32,6 +34,7 @@ from lib.content import (
32
34
  slice_last_turns_by_user_boundary,
33
35
  )
34
36
  from lib.daemon import get_api_url
37
+ from lib.pacing import inflight_lock
35
38
  from lib.state import increment_turn_count, track_retention
36
39
 
37
40
 
@@ -41,7 +44,14 @@ def read_transcript(transcript_path: str) -> list:
41
44
  Claude Code transcript format nests messages:
42
45
  {type: "user", message: {role: "user", content: "..."}, uuid: "...", ...}
43
46
  Also supports flat format for testing:
44
- {role: "user", content: "..."}
47
+ {role: "user", content: "...", uuid: "..."}
48
+
49
+ The per-entry ``uuid`` Claude Code stamps on every ``.jsonl`` line is
50
+ surfaced onto the returned message dict (switchroom #3244) so the
51
+ deterministic content-derived ``document_id`` and the durable retain
52
+ watermark can key on it. The uuid is stable across compaction and unique
53
+ per entry. Adding the key is inert for downstream formatting
54
+ (``lib/content`` reads only ``role``/``content``).
45
55
  """
46
56
  if not transcript_path or not os.path.isfile(transcript_path):
47
57
  return []
@@ -58,6 +68,12 @@ def read_transcript(transcript_path: str) -> list:
58
68
  if entry.get("type") in ("user", "assistant"):
59
69
  msg = entry.get("message", {})
60
70
  if isinstance(msg, dict) and msg.get("role"):
71
+ # Surface the transcript-entry uuid (nested format
72
+ # carries it on the OUTER entry, not the message).
73
+ uid = entry.get("uuid")
74
+ if uid is not None and "uuid" not in msg:
75
+ msg = dict(msg)
76
+ msg["uuid"] = uid
61
77
  messages.append(msg)
62
78
  # Flat format (testing / future compatibility)
63
79
  elif "role" in entry and "content" in entry:
@@ -123,6 +139,174 @@ def select_retain_window(
123
139
  return list(all_messages), True
124
140
 
125
141
 
142
+ def _ordered_uuids(all_messages: list) -> list:
143
+ """Transcript-order list of entry uuids (skips entries without one)."""
144
+ out = []
145
+ for m in all_messages:
146
+ if isinstance(m, dict):
147
+ uid = m.get("uuid")
148
+ if uid:
149
+ out.append(uid)
150
+ return out
151
+
152
+
153
+ def slice_document_id(session_id: str, messages_slice: list, transcript_text: str) -> str:
154
+ """Deterministic, content-derived retain ``document_id`` (switchroom #3244 §1).
155
+
156
+ ``{session_id}-r{start_uuid}-{end_uuid}`` from the FULL first/last transcript
157
+ entry uuids of the retained slice — NOT truncated (32+32 bits birthday-
158
+ collide across months of history and a collision is *silent loss*, not a
159
+ dup). The id is a pure function of *which turns* (which start/end uuids) are
160
+ retained, so any two writes of the **identical** slice — the live Stop-hook
161
+ path re-firing the same window, or boot reconciliation re-posting it —
162
+ compute the same id and upsert on the daemon's document_id key instead of
163
+ double-storing. No wall clock, no randomness.
164
+
165
+ CAUTION — this convergence is only for the *same slice boundaries*. The live
166
+ path slices a *sliding* ``retainEveryNTurns``+overlap window (see
167
+ ``select_retain_window``) while the PR2 backfill slices *non-overlapping*
168
+ fixed-size chunks; they do NOT produce the same boundaries for the same
169
+ turns, so their ids do NOT converge and the daemon cannot upsert one against
170
+ the other. The backfill stays duplicate-free via its total-loss gate (it
171
+ only re-posts sessions the live path wrote nothing for), NOT via id-identity
172
+ with the live path — do not conflate the two.
173
+
174
+ Fallback (legacy/flat transcripts with no per-entry uuid): a sha256 of the
175
+ formatted transcript content — still purely deterministic and convergent
176
+ across paths for identical content.
177
+ """
178
+ start = messages_slice[0].get("uuid") if messages_slice else None
179
+ end = messages_slice[-1].get("uuid") if messages_slice else None
180
+ if start and end:
181
+ return f"{session_id}-r{start}-{end}"
182
+ digest = hashlib.sha256((transcript_text or "").encode("utf-8")).hexdigest()[:32]
183
+ return f"{session_id}-r{digest}"
184
+
185
+
186
+ def build_retain_payload(
187
+ config: dict,
188
+ session_id: str,
189
+ messages_to_retain: list,
190
+ all_messages: list,
191
+ *,
192
+ bank_id: str,
193
+ api_url,
194
+ api_token,
195
+ retain_full_window: bool = True,
196
+ document_id=None,
197
+ ) -> dict | None:
198
+ """Pure transcript → retain payload + deterministic id (switchroom #3244 §1.4).
199
+
200
+ NETWORK-FREE and MISSION-WRITE-FREE by contract: it formats the slice,
201
+ computes the deterministic content-derived ``document_id`` (unless one is
202
+ supplied, e.g. the full-session compaction-tracked id), and assembles the
203
+ exact kwargs ``client.retain()`` / the pending-retains drainer expect. It
204
+ issues NO HTTP and never touches ``ensure_bank_mission`` — so the boot
205
+ reconciler and the PR2 dry-run can import it and stay genuinely write-free.
206
+
207
+ Returns ``{payload, document_id, message_count, last_uuid, ordered_uuids,
208
+ transcript}`` or ``None`` when the slice formats to nothing.
209
+ """
210
+ retain_roles = config.get("retainRoles", ["user", "assistant"])
211
+ include_tool_calls = config.get("retainToolCalls", True)
212
+ transcript, message_count = prepare_retention_transcript(
213
+ messages_to_retain, retain_roles, retain_full_window, include_tool_calls=include_tool_calls
214
+ )
215
+ if not transcript:
216
+ return None
217
+
218
+ if document_id is None:
219
+ document_id = slice_document_id(session_id, messages_to_retain, transcript)
220
+
221
+ template_vars = {
222
+ "session_id": session_id,
223
+ "bank_id": bank_id,
224
+ "timestamp": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
225
+ "user_id": os.environ.get("HINDSIGHT_USER_ID", ""),
226
+ }
227
+
228
+ def _resolve_template(value: str) -> str:
229
+ for k, v in template_vars.items():
230
+ value = value.replace(f"{{{k}}}", v)
231
+ return value
232
+
233
+ raw_tags = config.get("retainTags", [])
234
+ if raw_tags:
235
+ tags = []
236
+ for original in raw_tags:
237
+ resolved = _resolve_template(original)
238
+ if ":" in resolved and resolved.split(":", 1)[1] == "":
239
+ continue
240
+ tags.append(resolved)
241
+ if not tags:
242
+ tags = None
243
+ else:
244
+ tags = None
245
+
246
+ metadata = {
247
+ "retained_at": template_vars["timestamp"],
248
+ "message_count": str(message_count),
249
+ "session_id": session_id,
250
+ }
251
+ for k, v in config.get("retainMetadata", {}).items():
252
+ metadata[k] = _resolve_template(str(v))
253
+
254
+ # Topic tagging (switchroom PR6a) — best-effort, never fails a build.
255
+ try:
256
+ from lib.gateway_ipc import extract_topic_from_prompt
257
+
258
+ topic_chat_id = None
259
+ topic_thread_id = None
260
+ for m in reversed(messages_to_retain):
261
+ if not isinstance(m, dict) or m.get("role") != "user":
262
+ continue
263
+ content = m.get("content")
264
+ text = content if isinstance(content, str) else (
265
+ next((p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text"), "")
266
+ if isinstance(content, list) else ""
267
+ )
268
+ c_id, t_id = extract_topic_from_prompt(text)
269
+ if c_id is not None:
270
+ topic_chat_id, topic_thread_id = c_id, t_id
271
+ break
272
+ if topic_chat_id is not None:
273
+ metadata["chat_id"] = topic_chat_id
274
+ if topic_thread_id is not None:
275
+ metadata["thread_id"] = topic_thread_id
276
+ aliases_json = os.environ.get("HINDSIGHT_TOPIC_ALIASES_JSON", "")
277
+ if aliases_json:
278
+ try:
279
+ aliases = json.loads(aliases_json)
280
+ if isinstance(aliases, dict):
281
+ inverse = {str(v): k for k, v in aliases.items()}
282
+ alias = inverse.get(str(topic_thread_id))
283
+ if alias:
284
+ metadata["topic_alias"] = alias
285
+ except (json.JSONDecodeError, ValueError, TypeError):
286
+ pass
287
+ except Exception:
288
+ pass
289
+
290
+ payload = {
291
+ "api_url": api_url,
292
+ "api_token": api_token,
293
+ "bank_id": bank_id,
294
+ "content": transcript,
295
+ "document_id": document_id,
296
+ "context": config.get("retainContext", "claude-code"),
297
+ "metadata": metadata,
298
+ "tags": tags,
299
+ }
300
+ return {
301
+ "payload": payload,
302
+ "document_id": document_id,
303
+ "message_count": message_count,
304
+ "last_uuid": messages_to_retain[-1].get("uuid") if messages_to_retain else None,
305
+ "ordered_uuids": _ordered_uuids(all_messages),
306
+ "transcript": transcript,
307
+ }
308
+
309
+
126
310
  def run_retain(hook_input: dict, force: bool = False) -> dict:
127
311
  """Run the auto-retain flow.
128
312
 
@@ -211,17 +395,6 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
211
395
  else:
212
396
  debug_log(config, f"Full session retain: {len(all_messages)} messages")
213
397
 
214
- # Format transcript
215
- retain_roles = config.get("retainRoles", ["user", "assistant"])
216
- include_tool_calls = config.get("retainToolCalls", True)
217
- transcript, message_count = prepare_retention_transcript(
218
- messages_to_retain, retain_roles, retain_full_window, include_tool_calls=include_tool_calls
219
- )
220
-
221
- if not transcript:
222
- debug_log(config, "Empty transcript after formatting, skipping retain")
223
- return {"status": "skipped", "reason": "empty transcript after formatting"}
224
-
225
398
  # Resolve API URL
226
399
  def _dbg(*a):
227
400
  debug_log(config, *a)
@@ -250,15 +423,18 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
250
423
  bank_id = derive_bank_id(hook_input, config)
251
424
  ensure_bank_mission(client, bank_id, config, debug_fn=_dbg)
252
425
 
253
- # Document ID strategy:
254
- # - Chunked mode: each chunk gets a timestamped document_id.
255
- # - Full-session mode: uses session_id as base, but tracks message count
256
- # to detect compaction. When Claude Code compacts the conversation the
257
- # transcript shrinks — if we kept the same document_id we'd overwrite the
258
- # pre-compaction document with a shorter one, losing context. Instead we
259
- # increment a chunk counter so the old document is preserved.
426
+ # Document ID strategy (switchroom #3244 §1 — single deterministic namespace):
427
+ # - Chunked mode at retainEveryNTurns>1 (the deployed default): a
428
+ # *content-derived* id computed from the slice's first/last transcript
429
+ # uuids (``document_id=None`` lets build_retain_payload compute it). This
430
+ # REPLACES the former wall-clock ``{session_id}-{time.time()*1000}`` id so
431
+ # the live path, boot reconciliation, and the PR2 backfill all mint the
432
+ # SAME id for the same turns ⇒ upsert, not duplicate.
433
+ # - Full-session mode / n==1: unchanged — a stable ``session_id`` base with
434
+ # a compaction chunk counter so a shrinking transcript doesn't overwrite
435
+ # the pre-compaction document with a shorter one.
260
436
  if retain_mode == "chunked" and retain_every_n > 1:
261
- document_id = f"{session_id}-{int(time.time() * 1000)}"
437
+ document_id = None # build_retain_payload computes the content-derived id
262
438
  else:
263
439
  chunk_index, compacted = track_retention(session_id, len(all_messages))
264
440
  if compacted:
@@ -270,135 +446,80 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
270
446
  # chunk 0 → plain session_id (backwards compatible with existing docs)
271
447
  document_id = session_id if chunk_index == 0 else f"{session_id}-c{chunk_index}"
272
448
 
273
- # Resolve template variables in tags and metadata.
274
- # Supported variables: {session_id}, {bank_id}, {timestamp}, {user_id}
275
- template_vars = {
276
- "session_id": session_id,
277
- "bank_id": bank_id,
278
- "timestamp": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
279
- "user_id": os.environ.get("HINDSIGHT_USER_ID", ""),
280
- }
281
-
282
- def _resolve_template(value: str) -> str:
283
- for k, v in template_vars.items():
284
- value = value.replace(f"{{{k}}}", v)
285
- return value
286
-
287
- # Tags from config with template resolution.
288
- # Drop tags whose resolved form ends in an empty namespace part (e.g. "user:"
289
- # when HINDSIGHT_USER_ID is unset). Tags without ':' are preserved as-is.
290
- raw_tags = config.get("retainTags", [])
291
- if raw_tags:
292
- tags = []
293
- for original in raw_tags:
294
- resolved = _resolve_template(original)
295
- if ":" in resolved and resolved.split(":", 1)[1] == "":
296
- debug_log(config, f"Dropping tag '{original}' -> '{resolved}' (empty content after ':')")
297
- continue
298
- tags.append(resolved)
299
- if not tags:
300
- tags = None
301
- else:
302
- tags = None
303
-
304
- # Metadata: merge built-in defaults with user-configured extras
305
- metadata = {
306
- "retained_at": template_vars["timestamp"],
307
- "message_count": str(message_count),
308
- "session_id": session_id,
309
- }
310
- for k, v in config.get("retainMetadata", {}).items():
311
- metadata[k] = _resolve_template(str(v))
312
-
313
- # Switchroom PR6a — topic tagging for supergroup-mode agents.
314
- # Scan the messages we're retaining for the latest `<channel
315
- # chat_id=... message_thread_id=...>` envelope and stamp the
316
- # tuple into metadata. Downstream (recall.py) uses this to log
317
- # active-vs-source topic for binding-failure analysis and to
318
- # support hard-filter mode when an operator opts in.
319
- #
320
- # No-op for fleet-shared / DM topology where every inbound from
321
- # this agent carries the same chat_id (or no chat envelope at all
322
- # for interactive / cron-only sessions) — the metadata is added
323
- # but doesn't change behaviour.
324
- try:
325
- from lib.gateway_ipc import extract_topic_from_prompt
326
- topic_chat_id = None
327
- topic_thread_id = None
328
- # Walk in reverse — most recent user message is the authoritative
329
- # "active topic" at retain time.
330
- for m in reversed(messages_to_retain):
331
- if not isinstance(m, dict) or m.get("role") != "user":
332
- continue
333
- content = m.get("content")
334
- text = content if isinstance(content, str) else (
335
- # Claude Code list-content shape: [{type:"text", text:"..."}, ...]
336
- next((p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text"), "")
337
- if isinstance(content, list) else ""
338
- )
339
- c_id, t_id = extract_topic_from_prompt(text)
340
- if c_id is not None:
341
- topic_chat_id, topic_thread_id = c_id, t_id
342
- break
343
- if topic_chat_id is not None:
344
- metadata["chat_id"] = topic_chat_id
345
- if topic_thread_id is not None:
346
- metadata["thread_id"] = topic_thread_id
347
- # Resolve alias from operator-injected env map.
348
- aliases_json = os.environ.get("HINDSIGHT_TOPIC_ALIASES_JSON", "")
349
- if aliases_json:
350
- try:
351
- aliases = json.loads(aliases_json)
352
- # aliases is {alias_name: thread_id_int_or_str}; build
353
- # the inverse lookup once.
354
- if isinstance(aliases, dict):
355
- inverse = {str(v): k for k, v in aliases.items()}
356
- alias = inverse.get(str(topic_thread_id))
357
- if alias:
358
- metadata["topic_alias"] = alias
359
- except (json.JSONDecodeError, ValueError, TypeError):
360
- pass # malformed env is non-fatal
361
- except Exception as e:
362
- # Topic tagging is best-effort — never fail a retain over it.
363
- debug_log(config, f"Topic tagging skipped: {e}")
449
+ # Build the payload via the shared, network-free seam (§1.4). This carries
450
+ # the deterministic id, connection info, formatted transcript, tags and
451
+ # metadata — sufficient to retry from a different process (drain_pending).
452
+ built = build_retain_payload(
453
+ config,
454
+ session_id,
455
+ messages_to_retain,
456
+ all_messages,
457
+ bank_id=bank_id,
458
+ api_url=api_url,
459
+ api_token=api_token,
460
+ retain_full_window=retain_full_window,
461
+ document_id=document_id,
462
+ )
463
+ if built is None:
464
+ debug_log(config, "Empty transcript after formatting, skipping retain")
465
+ return {"status": "skipped", "reason": "empty transcript after formatting"}
364
466
 
467
+ payload = built["payload"]
468
+ document_id = built["document_id"]
365
469
  debug_log(
366
- config, f"Retaining to bank '{bank_id}', doc '{document_id}', {message_count} messages, {len(transcript)} chars"
470
+ config,
471
+ f"Retaining to bank '{bank_id}', doc '{document_id}', {built['message_count']} messages",
367
472
  )
368
- if tags:
369
- debug_log(config, f"Tags: {tags}")
370
473
 
371
- # Build the full payload up-front so we can hand it to the pending-
372
- # retains queue verbatim if the POST fails. The payload mirrors the
373
- # client.retain() kwargs plus connection info (api_url, api_token)
374
- # so the drainer can reconstruct the call from a different process.
375
- payload = {
376
- "api_url": api_url,
377
- "api_token": api_token,
378
- "bank_id": bank_id,
379
- "content": transcript,
380
- "document_id": document_id,
381
- "context": config.get("retainContext", "claude-code"),
382
- "metadata": metadata,
383
- "tags": tags,
384
- }
474
+ # POST to Hindsight retain API under the shared fleet pacing lock (§1.5) so
475
+ # at most one durability POST is in flight for this agent. The live Stop
476
+ # path acquires it NON-BLOCKING (turn latency must not regress): if a boot
477
+ # reconcile / backfill holds it, we do NOT wait — we return the failure
478
+ # payload so main() enqueues it and the next boot drain delivers it. A
479
+ # forced SessionEnd sweep waits (blocking) since it is the final flush.
480
+ with inflight_lock(blocking=force) as acquired:
481
+ if not acquired:
482
+ debug_log(config, "retain-inflight lock busy; deferring retain to pending-retains")
483
+ return {
484
+ "status": "failed",
485
+ "error": RuntimeError("retain-inflight lock busy; deferring to pending-retains"),
486
+ "payload": payload,
487
+ }
488
+ try:
489
+ # async_processing=False (§1.1): commit-before-ack so the 200 proves
490
+ # durable persistence before we advance the watermark. A bare async
491
+ # 200 (ack-of-receipt) must never mark unpersisted work committed.
492
+ response = client.retain(
493
+ bank_id=bank_id,
494
+ content=payload["content"],
495
+ document_id=document_id,
496
+ context=payload["context"],
497
+ metadata=payload["metadata"],
498
+ tags=payload["tags"],
499
+ timeout=15,
500
+ async_processing=False,
501
+ )
502
+ except Exception as e:
503
+ print(f"[Hindsight] Retain failed: {e}", file=sys.stderr)
504
+ return {"status": "failed", "error": e, "payload": payload}
505
+
506
+ debug_log(config, f"Retain response: {json.dumps(response)[:200]}")
385
507
 
386
- # POST to Hindsight retain API
508
+ # Advance the durable watermark ONLY after confirmed persistence (§1.1).
509
+ # Best-effort: a watermark write failure must never fail a landed retain —
510
+ # worst case the boot reconciler re-upserts the same slice idempotently.
387
511
  try:
388
- response = client.retain(
389
- bank_id=bank_id,
390
- content=transcript,
391
- document_id=document_id,
392
- context=payload["context"],
393
- metadata=metadata,
394
- tags=tags,
395
- timeout=15,
512
+ watermark.commit(
513
+ session_id,
514
+ built["last_uuid"],
515
+ document_id,
516
+ transcript_path=transcript_path,
517
+ ordered_uuids=built["ordered_uuids"],
396
518
  )
397
- debug_log(config, f"Retain response: {json.dumps(response)[:200]}")
398
- return {"status": "ok", "response": response}
399
- except Exception as e:
400
- print(f"[Hindsight] Retain failed: {e}", file=sys.stderr)
401
- return {"status": "failed", "error": e, "payload": payload}
519
+ except Exception as e: # pragma: no cover - defensive
520
+ debug_log(config, f"watermark commit skipped: {e}")
521
+
522
+ return {"status": "ok", "response": response}
402
523
 
403
524
 
404
525
  def main():
@@ -407,7 +528,42 @@ def main():
407
528
  except (json.JSONDecodeError, EOFError):
408
529
  print("[Hindsight] Failed to read hook input", file=sys.stderr)
409
530
  return
410
- run_retain(hook_input, force=False)
531
+
532
+ # Close hole A2 (switchroom #3244): the Stop entrypoint used to discard
533
+ # run_retain's return value, so a failed per-turn retain (daemon busy /
534
+ # unreachable / lock-deferred) was silently dropped — never queued, never
535
+ # surfaced. Mirror session_end.py: on a failed retain WITH a payload,
536
+ # durably enqueue it to pending-retains so the next SessionStart drain
537
+ # replays it. The deterministic content-derived document_id (§1) means a
538
+ # queued entry and a later boot-reconcile entry for the same window collide
539
+ # on id ⇒ upsert, not duplicate — so no id rewrite is needed here.
540
+ #
541
+ # We keep exit 0: retain.py is registered as an async Stop hook (hooks.json)
542
+ # whose exit code Claude Code discards. Durability comes from the enqueue,
543
+ # not the exit code.
544
+ result = run_retain(hook_input, force=False) or {}
545
+ if result.get("status") == "failed" and result.get("payload"):
546
+ try:
547
+ from lib.pending import MAX_ENTRIES, count as pending_count, enqueue as pending_enqueue
548
+
549
+ err = result.get("error") or RuntimeError("stop retain failed")
550
+ queued = pending_enqueue(result["payload"], err)
551
+ if queued is None:
552
+ print(
553
+ f"[Hindsight] pending-retains queue full ({MAX_ENTRIES} entries); "
554
+ f"dropping this Stop retain. Operator: drain manually, then run "
555
+ f"`switchroom doctor`.",
556
+ file=sys.stderr,
557
+ )
558
+ else:
559
+ print(
560
+ f"[Hindsight] Stop retain failed: queued to pending-retains "
561
+ f"(error: {type(err).__name__}: {err}, pending={pending_count()}). "
562
+ f"Will retry on next SessionStart.",
563
+ file=sys.stderr,
564
+ )
565
+ except Exception as e: # pragma: no cover - defensive
566
+ print(f"[Hindsight] Stop retain enqueue failed: {e}", file=sys.stderr)
411
567
 
412
568
 
413
569
  if __name__ == "__main__":
@@ -88,6 +88,20 @@ def main():
88
88
  # this up via run-hook.sh — see exit-code path in __main__.
89
89
  debug_log(config, f"drain_pending unexpected error (ignored): {e}")
90
90
 
91
+ # Boot reconciliation (switchroom #3244): AFTER the drain, diff the durable
92
+ # per-session watermark against the on-disk transcript tail and recover any
93
+ # un-committed human turns an abrupt session death (SIGKILL / OOM /
94
+ # watchdog) skipped — the drain only replays what SessionEnd managed to
95
+ # enqueue, which an abrupt kill never runs. This is the load-bearing
96
+ # guarantee; it is naturally cheap (skips sessions with no gap) and bounded
97
+ # (lookback / turn-cap / wall-clock budget, each enqueuing its remainder).
98
+ try:
99
+ from reconcile_tail import reconcile as reconcile_tail
100
+
101
+ reconcile_tail(config, hook_input=hook_input)
102
+ except Exception as e:
103
+ debug_log(config, f"reconcile_tail unexpected error (ignored): {e}")
104
+
91
105
 
92
106
  if __name__ == "__main__":
93
107
  try: