switchroom 0.18.24 → 0.18.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +59 -11
- package/dist/host-control/main.js +1 -1
- package/package.json +1 -1
- package/telegram-plugin/dist/bridge/bridge.js +26 -0
- package/telegram-plugin/dist/gateway/gateway.js +1489 -829
- package/telegram-plugin/dist/server.js +26 -0
- package/telegram-plugin/gateway/callback-query-handlers.ts +7 -0
- package/telegram-plugin/gateway/gateway.ts +314 -3
- package/telegram-plugin/gateway/model-command.ts +188 -56
- package/telegram-plugin/gateway/redelivery-decision.ts +139 -0
- package/telegram-plugin/gateway/vault-grant-inbound-builders.ts +42 -1
- package/telegram-plugin/history.ts +118 -0
- package/telegram-plugin/registry/turns-schema.ts +89 -1
- package/telegram-plugin/session-tail.ts +185 -0
- package/telegram-plugin/subagent-watcher.ts +45 -0
- package/telegram-plugin/tests/crash-redelivery-resume-exclusion.test.ts +133 -0
- package/telegram-plugin/tests/crash-redelivery-wiring.test.ts +72 -0
- package/telegram-plugin/tests/history.test.ts +91 -0
- package/telegram-plugin/tests/model-command.test.ts +189 -12
- package/telegram-plugin/tests/redelivery-decision.test.ts +84 -0
- package/telegram-plugin/tests/registry-turns.test.ts +51 -0
- package/telegram-plugin/tests/session-model-source.test.ts +11 -0
- package/telegram-plugin/tests/session-tail.test.ts +145 -0
- package/telegram-plugin/tests/subagent-watcher.test.ts +50 -0
- package/telegram-plugin/tests/tool-activity-summary.test.ts +109 -0
- package/telegram-plugin/tests/trailing-answer-projector.test.ts +124 -0
- package/telegram-plugin/tests/vault-grant-inbound-builders.test.ts +125 -0
- package/telegram-plugin/tests/worker-feed-pin-persistence.test.ts +306 -0
- package/telegram-plugin/tool-activity-summary.ts +54 -3
- package/telegram-plugin/worker-activity-feed.ts +104 -0
- package/vendor/hindsight-memory/scripts/backfill_transcripts.py +762 -0
- package/vendor/hindsight-memory/scripts/drain_pending.py +13 -1
- package/vendor/hindsight-memory/scripts/lib/client.py +14 -4
- package/vendor/hindsight-memory/scripts/lib/config.py +8 -0
- package/vendor/hindsight-memory/scripts/lib/pacing.py +102 -0
- package/vendor/hindsight-memory/scripts/lib/watermark.py +213 -0
- package/vendor/hindsight-memory/scripts/reconcile_tail.py +344 -0
- package/vendor/hindsight-memory/scripts/retain.py +299 -143
- package/vendor/hindsight-memory/scripts/session_start.py +14 -0
- package/vendor/hindsight-memory/scripts/tests/test_backfill.py +362 -0
- package/vendor/hindsight-memory/scripts/tests/test_reconcile_durability.py +350 -0
- package/vendor/hindsight-memory/tests/test_hooks.py +8 -2
|
@@ -17,6 +17,7 @@ Exit codes:
|
|
|
17
17
|
0 — always (graceful degradation on any error)
|
|
18
18
|
"""
|
|
19
19
|
|
|
20
|
+
import hashlib
|
|
20
21
|
import json
|
|
21
22
|
import os
|
|
22
23
|
import sys
|
|
@@ -24,6 +25,7 @@ import time
|
|
|
24
25
|
|
|
25
26
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
26
27
|
|
|
28
|
+
from lib import watermark
|
|
27
29
|
from lib.bank import derive_bank_id, ensure_bank_mission
|
|
28
30
|
from lib.client import HindsightClient
|
|
29
31
|
from lib.config import debug_log, load_config
|
|
@@ -32,6 +34,7 @@ from lib.content import (
|
|
|
32
34
|
slice_last_turns_by_user_boundary,
|
|
33
35
|
)
|
|
34
36
|
from lib.daemon import get_api_url
|
|
37
|
+
from lib.pacing import inflight_lock
|
|
35
38
|
from lib.state import increment_turn_count, track_retention
|
|
36
39
|
|
|
37
40
|
|
|
@@ -41,7 +44,14 @@ def read_transcript(transcript_path: str) -> list:
|
|
|
41
44
|
Claude Code transcript format nests messages:
|
|
42
45
|
{type: "user", message: {role: "user", content: "..."}, uuid: "...", ...}
|
|
43
46
|
Also supports flat format for testing:
|
|
44
|
-
{role: "user", content: "..."}
|
|
47
|
+
{role: "user", content: "...", uuid: "..."}
|
|
48
|
+
|
|
49
|
+
The per-entry ``uuid`` Claude Code stamps on every ``.jsonl`` line is
|
|
50
|
+
surfaced onto the returned message dict (switchroom #3244) so the
|
|
51
|
+
deterministic content-derived ``document_id`` and the durable retain
|
|
52
|
+
watermark can key on it. The uuid is stable across compaction and unique
|
|
53
|
+
per entry. Adding the key is inert for downstream formatting
|
|
54
|
+
(``lib/content`` reads only ``role``/``content``).
|
|
45
55
|
"""
|
|
46
56
|
if not transcript_path or not os.path.isfile(transcript_path):
|
|
47
57
|
return []
|
|
@@ -58,6 +68,12 @@ def read_transcript(transcript_path: str) -> list:
|
|
|
58
68
|
if entry.get("type") in ("user", "assistant"):
|
|
59
69
|
msg = entry.get("message", {})
|
|
60
70
|
if isinstance(msg, dict) and msg.get("role"):
|
|
71
|
+
# Surface the transcript-entry uuid (nested format
|
|
72
|
+
# carries it on the OUTER entry, not the message).
|
|
73
|
+
uid = entry.get("uuid")
|
|
74
|
+
if uid is not None and "uuid" not in msg:
|
|
75
|
+
msg = dict(msg)
|
|
76
|
+
msg["uuid"] = uid
|
|
61
77
|
messages.append(msg)
|
|
62
78
|
# Flat format (testing / future compatibility)
|
|
63
79
|
elif "role" in entry and "content" in entry:
|
|
@@ -123,6 +139,174 @@ def select_retain_window(
|
|
|
123
139
|
return list(all_messages), True
|
|
124
140
|
|
|
125
141
|
|
|
142
|
+
def _ordered_uuids(all_messages: list) -> list:
|
|
143
|
+
"""Transcript-order list of entry uuids (skips entries without one)."""
|
|
144
|
+
out = []
|
|
145
|
+
for m in all_messages:
|
|
146
|
+
if isinstance(m, dict):
|
|
147
|
+
uid = m.get("uuid")
|
|
148
|
+
if uid:
|
|
149
|
+
out.append(uid)
|
|
150
|
+
return out
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def slice_document_id(session_id: str, messages_slice: list, transcript_text: str) -> str:
|
|
154
|
+
"""Deterministic, content-derived retain ``document_id`` (switchroom #3244 §1).
|
|
155
|
+
|
|
156
|
+
``{session_id}-r{start_uuid}-{end_uuid}`` from the FULL first/last transcript
|
|
157
|
+
entry uuids of the retained slice — NOT truncated (32+32 bits birthday-
|
|
158
|
+
collide across months of history and a collision is *silent loss*, not a
|
|
159
|
+
dup). The id is a pure function of *which turns* (which start/end uuids) are
|
|
160
|
+
retained, so any two writes of the **identical** slice — the live Stop-hook
|
|
161
|
+
path re-firing the same window, or boot reconciliation re-posting it —
|
|
162
|
+
compute the same id and upsert on the daemon's document_id key instead of
|
|
163
|
+
double-storing. No wall clock, no randomness.
|
|
164
|
+
|
|
165
|
+
CAUTION — this convergence is only for the *same slice boundaries*. The live
|
|
166
|
+
path slices a *sliding* ``retainEveryNTurns``+overlap window (see
|
|
167
|
+
``select_retain_window``) while the PR2 backfill slices *non-overlapping*
|
|
168
|
+
fixed-size chunks; they do NOT produce the same boundaries for the same
|
|
169
|
+
turns, so their ids do NOT converge and the daemon cannot upsert one against
|
|
170
|
+
the other. The backfill stays duplicate-free via its total-loss gate (it
|
|
171
|
+
only re-posts sessions the live path wrote nothing for), NOT via id-identity
|
|
172
|
+
with the live path — do not conflate the two.
|
|
173
|
+
|
|
174
|
+
Fallback (legacy/flat transcripts with no per-entry uuid): a sha256 of the
|
|
175
|
+
formatted transcript content — still purely deterministic and convergent
|
|
176
|
+
across paths for identical content.
|
|
177
|
+
"""
|
|
178
|
+
start = messages_slice[0].get("uuid") if messages_slice else None
|
|
179
|
+
end = messages_slice[-1].get("uuid") if messages_slice else None
|
|
180
|
+
if start and end:
|
|
181
|
+
return f"{session_id}-r{start}-{end}"
|
|
182
|
+
digest = hashlib.sha256((transcript_text or "").encode("utf-8")).hexdigest()[:32]
|
|
183
|
+
return f"{session_id}-r{digest}"
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def build_retain_payload(
|
|
187
|
+
config: dict,
|
|
188
|
+
session_id: str,
|
|
189
|
+
messages_to_retain: list,
|
|
190
|
+
all_messages: list,
|
|
191
|
+
*,
|
|
192
|
+
bank_id: str,
|
|
193
|
+
api_url,
|
|
194
|
+
api_token,
|
|
195
|
+
retain_full_window: bool = True,
|
|
196
|
+
document_id=None,
|
|
197
|
+
) -> dict | None:
|
|
198
|
+
"""Pure transcript → retain payload + deterministic id (switchroom #3244 §1.4).
|
|
199
|
+
|
|
200
|
+
NETWORK-FREE and MISSION-WRITE-FREE by contract: it formats the slice,
|
|
201
|
+
computes the deterministic content-derived ``document_id`` (unless one is
|
|
202
|
+
supplied, e.g. the full-session compaction-tracked id), and assembles the
|
|
203
|
+
exact kwargs ``client.retain()`` / the pending-retains drainer expect. It
|
|
204
|
+
issues NO HTTP and never touches ``ensure_bank_mission`` — so the boot
|
|
205
|
+
reconciler and the PR2 dry-run can import it and stay genuinely write-free.
|
|
206
|
+
|
|
207
|
+
Returns ``{payload, document_id, message_count, last_uuid, ordered_uuids,
|
|
208
|
+
transcript}`` or ``None`` when the slice formats to nothing.
|
|
209
|
+
"""
|
|
210
|
+
retain_roles = config.get("retainRoles", ["user", "assistant"])
|
|
211
|
+
include_tool_calls = config.get("retainToolCalls", True)
|
|
212
|
+
transcript, message_count = prepare_retention_transcript(
|
|
213
|
+
messages_to_retain, retain_roles, retain_full_window, include_tool_calls=include_tool_calls
|
|
214
|
+
)
|
|
215
|
+
if not transcript:
|
|
216
|
+
return None
|
|
217
|
+
|
|
218
|
+
if document_id is None:
|
|
219
|
+
document_id = slice_document_id(session_id, messages_to_retain, transcript)
|
|
220
|
+
|
|
221
|
+
template_vars = {
|
|
222
|
+
"session_id": session_id,
|
|
223
|
+
"bank_id": bank_id,
|
|
224
|
+
"timestamp": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
225
|
+
"user_id": os.environ.get("HINDSIGHT_USER_ID", ""),
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
def _resolve_template(value: str) -> str:
|
|
229
|
+
for k, v in template_vars.items():
|
|
230
|
+
value = value.replace(f"{{{k}}}", v)
|
|
231
|
+
return value
|
|
232
|
+
|
|
233
|
+
raw_tags = config.get("retainTags", [])
|
|
234
|
+
if raw_tags:
|
|
235
|
+
tags = []
|
|
236
|
+
for original in raw_tags:
|
|
237
|
+
resolved = _resolve_template(original)
|
|
238
|
+
if ":" in resolved and resolved.split(":", 1)[1] == "":
|
|
239
|
+
continue
|
|
240
|
+
tags.append(resolved)
|
|
241
|
+
if not tags:
|
|
242
|
+
tags = None
|
|
243
|
+
else:
|
|
244
|
+
tags = None
|
|
245
|
+
|
|
246
|
+
metadata = {
|
|
247
|
+
"retained_at": template_vars["timestamp"],
|
|
248
|
+
"message_count": str(message_count),
|
|
249
|
+
"session_id": session_id,
|
|
250
|
+
}
|
|
251
|
+
for k, v in config.get("retainMetadata", {}).items():
|
|
252
|
+
metadata[k] = _resolve_template(str(v))
|
|
253
|
+
|
|
254
|
+
# Topic tagging (switchroom PR6a) — best-effort, never fails a build.
|
|
255
|
+
try:
|
|
256
|
+
from lib.gateway_ipc import extract_topic_from_prompt
|
|
257
|
+
|
|
258
|
+
topic_chat_id = None
|
|
259
|
+
topic_thread_id = None
|
|
260
|
+
for m in reversed(messages_to_retain):
|
|
261
|
+
if not isinstance(m, dict) or m.get("role") != "user":
|
|
262
|
+
continue
|
|
263
|
+
content = m.get("content")
|
|
264
|
+
text = content if isinstance(content, str) else (
|
|
265
|
+
next((p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text"), "")
|
|
266
|
+
if isinstance(content, list) else ""
|
|
267
|
+
)
|
|
268
|
+
c_id, t_id = extract_topic_from_prompt(text)
|
|
269
|
+
if c_id is not None:
|
|
270
|
+
topic_chat_id, topic_thread_id = c_id, t_id
|
|
271
|
+
break
|
|
272
|
+
if topic_chat_id is not None:
|
|
273
|
+
metadata["chat_id"] = topic_chat_id
|
|
274
|
+
if topic_thread_id is not None:
|
|
275
|
+
metadata["thread_id"] = topic_thread_id
|
|
276
|
+
aliases_json = os.environ.get("HINDSIGHT_TOPIC_ALIASES_JSON", "")
|
|
277
|
+
if aliases_json:
|
|
278
|
+
try:
|
|
279
|
+
aliases = json.loads(aliases_json)
|
|
280
|
+
if isinstance(aliases, dict):
|
|
281
|
+
inverse = {str(v): k for k, v in aliases.items()}
|
|
282
|
+
alias = inverse.get(str(topic_thread_id))
|
|
283
|
+
if alias:
|
|
284
|
+
metadata["topic_alias"] = alias
|
|
285
|
+
except (json.JSONDecodeError, ValueError, TypeError):
|
|
286
|
+
pass
|
|
287
|
+
except Exception:
|
|
288
|
+
pass
|
|
289
|
+
|
|
290
|
+
payload = {
|
|
291
|
+
"api_url": api_url,
|
|
292
|
+
"api_token": api_token,
|
|
293
|
+
"bank_id": bank_id,
|
|
294
|
+
"content": transcript,
|
|
295
|
+
"document_id": document_id,
|
|
296
|
+
"context": config.get("retainContext", "claude-code"),
|
|
297
|
+
"metadata": metadata,
|
|
298
|
+
"tags": tags,
|
|
299
|
+
}
|
|
300
|
+
return {
|
|
301
|
+
"payload": payload,
|
|
302
|
+
"document_id": document_id,
|
|
303
|
+
"message_count": message_count,
|
|
304
|
+
"last_uuid": messages_to_retain[-1].get("uuid") if messages_to_retain else None,
|
|
305
|
+
"ordered_uuids": _ordered_uuids(all_messages),
|
|
306
|
+
"transcript": transcript,
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
|
|
126
310
|
def run_retain(hook_input: dict, force: bool = False) -> dict:
|
|
127
311
|
"""Run the auto-retain flow.
|
|
128
312
|
|
|
@@ -211,17 +395,6 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
|
|
|
211
395
|
else:
|
|
212
396
|
debug_log(config, f"Full session retain: {len(all_messages)} messages")
|
|
213
397
|
|
|
214
|
-
# Format transcript
|
|
215
|
-
retain_roles = config.get("retainRoles", ["user", "assistant"])
|
|
216
|
-
include_tool_calls = config.get("retainToolCalls", True)
|
|
217
|
-
transcript, message_count = prepare_retention_transcript(
|
|
218
|
-
messages_to_retain, retain_roles, retain_full_window, include_tool_calls=include_tool_calls
|
|
219
|
-
)
|
|
220
|
-
|
|
221
|
-
if not transcript:
|
|
222
|
-
debug_log(config, "Empty transcript after formatting, skipping retain")
|
|
223
|
-
return {"status": "skipped", "reason": "empty transcript after formatting"}
|
|
224
|
-
|
|
225
398
|
# Resolve API URL
|
|
226
399
|
def _dbg(*a):
|
|
227
400
|
debug_log(config, *a)
|
|
@@ -250,15 +423,18 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
|
|
|
250
423
|
bank_id = derive_bank_id(hook_input, config)
|
|
251
424
|
ensure_bank_mission(client, bank_id, config, debug_fn=_dbg)
|
|
252
425
|
|
|
253
|
-
# Document ID strategy:
|
|
254
|
-
# - Chunked mode
|
|
255
|
-
#
|
|
256
|
-
#
|
|
257
|
-
#
|
|
258
|
-
#
|
|
259
|
-
#
|
|
426
|
+
# Document ID strategy (switchroom #3244 §1 — single deterministic namespace):
|
|
427
|
+
# - Chunked mode at retainEveryNTurns>1 (the deployed default): a
|
|
428
|
+
# *content-derived* id computed from the slice's first/last transcript
|
|
429
|
+
# uuids (``document_id=None`` lets build_retain_payload compute it). This
|
|
430
|
+
# REPLACES the former wall-clock ``{session_id}-{time.time()*1000}`` id so
|
|
431
|
+
# the live path, boot reconciliation, and the PR2 backfill all mint the
|
|
432
|
+
# SAME id for the same turns ⇒ upsert, not duplicate.
|
|
433
|
+
# - Full-session mode / n==1: unchanged — a stable ``session_id`` base with
|
|
434
|
+
# a compaction chunk counter so a shrinking transcript doesn't overwrite
|
|
435
|
+
# the pre-compaction document with a shorter one.
|
|
260
436
|
if retain_mode == "chunked" and retain_every_n > 1:
|
|
261
|
-
document_id =
|
|
437
|
+
document_id = None # build_retain_payload computes the content-derived id
|
|
262
438
|
else:
|
|
263
439
|
chunk_index, compacted = track_retention(session_id, len(all_messages))
|
|
264
440
|
if compacted:
|
|
@@ -270,135 +446,80 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
|
|
|
270
446
|
# chunk 0 → plain session_id (backwards compatible with existing docs)
|
|
271
447
|
document_id = session_id if chunk_index == 0 else f"{session_id}-c{chunk_index}"
|
|
272
448
|
|
|
273
|
-
#
|
|
274
|
-
#
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
raw_tags = config.get("retainTags", [])
|
|
291
|
-
if raw_tags:
|
|
292
|
-
tags = []
|
|
293
|
-
for original in raw_tags:
|
|
294
|
-
resolved = _resolve_template(original)
|
|
295
|
-
if ":" in resolved and resolved.split(":", 1)[1] == "":
|
|
296
|
-
debug_log(config, f"Dropping tag '{original}' -> '{resolved}' (empty content after ':')")
|
|
297
|
-
continue
|
|
298
|
-
tags.append(resolved)
|
|
299
|
-
if not tags:
|
|
300
|
-
tags = None
|
|
301
|
-
else:
|
|
302
|
-
tags = None
|
|
303
|
-
|
|
304
|
-
# Metadata: merge built-in defaults with user-configured extras
|
|
305
|
-
metadata = {
|
|
306
|
-
"retained_at": template_vars["timestamp"],
|
|
307
|
-
"message_count": str(message_count),
|
|
308
|
-
"session_id": session_id,
|
|
309
|
-
}
|
|
310
|
-
for k, v in config.get("retainMetadata", {}).items():
|
|
311
|
-
metadata[k] = _resolve_template(str(v))
|
|
312
|
-
|
|
313
|
-
# Switchroom PR6a — topic tagging for supergroup-mode agents.
|
|
314
|
-
# Scan the messages we're retaining for the latest `<channel
|
|
315
|
-
# chat_id=... message_thread_id=...>` envelope and stamp the
|
|
316
|
-
# tuple into metadata. Downstream (recall.py) uses this to log
|
|
317
|
-
# active-vs-source topic for binding-failure analysis and to
|
|
318
|
-
# support hard-filter mode when an operator opts in.
|
|
319
|
-
#
|
|
320
|
-
# No-op for fleet-shared / DM topology where every inbound from
|
|
321
|
-
# this agent carries the same chat_id (or no chat envelope at all
|
|
322
|
-
# for interactive / cron-only sessions) — the metadata is added
|
|
323
|
-
# but doesn't change behaviour.
|
|
324
|
-
try:
|
|
325
|
-
from lib.gateway_ipc import extract_topic_from_prompt
|
|
326
|
-
topic_chat_id = None
|
|
327
|
-
topic_thread_id = None
|
|
328
|
-
# Walk in reverse — most recent user message is the authoritative
|
|
329
|
-
# "active topic" at retain time.
|
|
330
|
-
for m in reversed(messages_to_retain):
|
|
331
|
-
if not isinstance(m, dict) or m.get("role") != "user":
|
|
332
|
-
continue
|
|
333
|
-
content = m.get("content")
|
|
334
|
-
text = content if isinstance(content, str) else (
|
|
335
|
-
# Claude Code list-content shape: [{type:"text", text:"..."}, ...]
|
|
336
|
-
next((p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text"), "")
|
|
337
|
-
if isinstance(content, list) else ""
|
|
338
|
-
)
|
|
339
|
-
c_id, t_id = extract_topic_from_prompt(text)
|
|
340
|
-
if c_id is not None:
|
|
341
|
-
topic_chat_id, topic_thread_id = c_id, t_id
|
|
342
|
-
break
|
|
343
|
-
if topic_chat_id is not None:
|
|
344
|
-
metadata["chat_id"] = topic_chat_id
|
|
345
|
-
if topic_thread_id is not None:
|
|
346
|
-
metadata["thread_id"] = topic_thread_id
|
|
347
|
-
# Resolve alias from operator-injected env map.
|
|
348
|
-
aliases_json = os.environ.get("HINDSIGHT_TOPIC_ALIASES_JSON", "")
|
|
349
|
-
if aliases_json:
|
|
350
|
-
try:
|
|
351
|
-
aliases = json.loads(aliases_json)
|
|
352
|
-
# aliases is {alias_name: thread_id_int_or_str}; build
|
|
353
|
-
# the inverse lookup once.
|
|
354
|
-
if isinstance(aliases, dict):
|
|
355
|
-
inverse = {str(v): k for k, v in aliases.items()}
|
|
356
|
-
alias = inverse.get(str(topic_thread_id))
|
|
357
|
-
if alias:
|
|
358
|
-
metadata["topic_alias"] = alias
|
|
359
|
-
except (json.JSONDecodeError, ValueError, TypeError):
|
|
360
|
-
pass # malformed env is non-fatal
|
|
361
|
-
except Exception as e:
|
|
362
|
-
# Topic tagging is best-effort — never fail a retain over it.
|
|
363
|
-
debug_log(config, f"Topic tagging skipped: {e}")
|
|
449
|
+
# Build the payload via the shared, network-free seam (§1.4). This carries
|
|
450
|
+
# the deterministic id, connection info, formatted transcript, tags and
|
|
451
|
+
# metadata — sufficient to retry from a different process (drain_pending).
|
|
452
|
+
built = build_retain_payload(
|
|
453
|
+
config,
|
|
454
|
+
session_id,
|
|
455
|
+
messages_to_retain,
|
|
456
|
+
all_messages,
|
|
457
|
+
bank_id=bank_id,
|
|
458
|
+
api_url=api_url,
|
|
459
|
+
api_token=api_token,
|
|
460
|
+
retain_full_window=retain_full_window,
|
|
461
|
+
document_id=document_id,
|
|
462
|
+
)
|
|
463
|
+
if built is None:
|
|
464
|
+
debug_log(config, "Empty transcript after formatting, skipping retain")
|
|
465
|
+
return {"status": "skipped", "reason": "empty transcript after formatting"}
|
|
364
466
|
|
|
467
|
+
payload = built["payload"]
|
|
468
|
+
document_id = built["document_id"]
|
|
365
469
|
debug_log(
|
|
366
|
-
config,
|
|
470
|
+
config,
|
|
471
|
+
f"Retaining to bank '{bank_id}', doc '{document_id}', {built['message_count']} messages",
|
|
367
472
|
)
|
|
368
|
-
if tags:
|
|
369
|
-
debug_log(config, f"Tags: {tags}")
|
|
370
473
|
|
|
371
|
-
#
|
|
372
|
-
#
|
|
373
|
-
#
|
|
374
|
-
#
|
|
375
|
-
payload
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
474
|
+
# POST to Hindsight retain API under the shared fleet pacing lock (§1.5) so
|
|
475
|
+
# at most one durability POST is in flight for this agent. The live Stop
|
|
476
|
+
# path acquires it NON-BLOCKING (turn latency must not regress): if a boot
|
|
477
|
+
# reconcile / backfill holds it, we do NOT wait — we return the failure
|
|
478
|
+
# payload so main() enqueues it and the next boot drain delivers it. A
|
|
479
|
+
# forced SessionEnd sweep waits (blocking) since it is the final flush.
|
|
480
|
+
with inflight_lock(blocking=force) as acquired:
|
|
481
|
+
if not acquired:
|
|
482
|
+
debug_log(config, "retain-inflight lock busy; deferring retain to pending-retains")
|
|
483
|
+
return {
|
|
484
|
+
"status": "failed",
|
|
485
|
+
"error": RuntimeError("retain-inflight lock busy; deferring to pending-retains"),
|
|
486
|
+
"payload": payload,
|
|
487
|
+
}
|
|
488
|
+
try:
|
|
489
|
+
# async_processing=False (§1.1): commit-before-ack so the 200 proves
|
|
490
|
+
# durable persistence before we advance the watermark. A bare async
|
|
491
|
+
# 200 (ack-of-receipt) must never mark unpersisted work committed.
|
|
492
|
+
response = client.retain(
|
|
493
|
+
bank_id=bank_id,
|
|
494
|
+
content=payload["content"],
|
|
495
|
+
document_id=document_id,
|
|
496
|
+
context=payload["context"],
|
|
497
|
+
metadata=payload["metadata"],
|
|
498
|
+
tags=payload["tags"],
|
|
499
|
+
timeout=15,
|
|
500
|
+
async_processing=False,
|
|
501
|
+
)
|
|
502
|
+
except Exception as e:
|
|
503
|
+
print(f"[Hindsight] Retain failed: {e}", file=sys.stderr)
|
|
504
|
+
return {"status": "failed", "error": e, "payload": payload}
|
|
505
|
+
|
|
506
|
+
debug_log(config, f"Retain response: {json.dumps(response)[:200]}")
|
|
385
507
|
|
|
386
|
-
#
|
|
508
|
+
# Advance the durable watermark ONLY after confirmed persistence (§1.1).
|
|
509
|
+
# Best-effort: a watermark write failure must never fail a landed retain —
|
|
510
|
+
# worst case the boot reconciler re-upserts the same slice idempotently.
|
|
387
511
|
try:
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
document_id
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
tags=tags,
|
|
395
|
-
timeout=15,
|
|
512
|
+
watermark.commit(
|
|
513
|
+
session_id,
|
|
514
|
+
built["last_uuid"],
|
|
515
|
+
document_id,
|
|
516
|
+
transcript_path=transcript_path,
|
|
517
|
+
ordered_uuids=built["ordered_uuids"],
|
|
396
518
|
)
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
return {"status": "failed", "error": e, "payload": payload}
|
|
519
|
+
except Exception as e: # pragma: no cover - defensive
|
|
520
|
+
debug_log(config, f"watermark commit skipped: {e}")
|
|
521
|
+
|
|
522
|
+
return {"status": "ok", "response": response}
|
|
402
523
|
|
|
403
524
|
|
|
404
525
|
def main():
|
|
@@ -407,7 +528,42 @@ def main():
|
|
|
407
528
|
except (json.JSONDecodeError, EOFError):
|
|
408
529
|
print("[Hindsight] Failed to read hook input", file=sys.stderr)
|
|
409
530
|
return
|
|
410
|
-
|
|
531
|
+
|
|
532
|
+
# Close hole A2 (switchroom #3244): the Stop entrypoint used to discard
|
|
533
|
+
# run_retain's return value, so a failed per-turn retain (daemon busy /
|
|
534
|
+
# unreachable / lock-deferred) was silently dropped — never queued, never
|
|
535
|
+
# surfaced. Mirror session_end.py: on a failed retain WITH a payload,
|
|
536
|
+
# durably enqueue it to pending-retains so the next SessionStart drain
|
|
537
|
+
# replays it. The deterministic content-derived document_id (§1) means a
|
|
538
|
+
# queued entry and a later boot-reconcile entry for the same window collide
|
|
539
|
+
# on id ⇒ upsert, not duplicate — so no id rewrite is needed here.
|
|
540
|
+
#
|
|
541
|
+
# We keep exit 0: retain.py is registered as an async Stop hook (hooks.json)
|
|
542
|
+
# whose exit code Claude Code discards. Durability comes from the enqueue,
|
|
543
|
+
# not the exit code.
|
|
544
|
+
result = run_retain(hook_input, force=False) or {}
|
|
545
|
+
if result.get("status") == "failed" and result.get("payload"):
|
|
546
|
+
try:
|
|
547
|
+
from lib.pending import MAX_ENTRIES, count as pending_count, enqueue as pending_enqueue
|
|
548
|
+
|
|
549
|
+
err = result.get("error") or RuntimeError("stop retain failed")
|
|
550
|
+
queued = pending_enqueue(result["payload"], err)
|
|
551
|
+
if queued is None:
|
|
552
|
+
print(
|
|
553
|
+
f"[Hindsight] pending-retains queue full ({MAX_ENTRIES} entries); "
|
|
554
|
+
f"dropping this Stop retain. Operator: drain manually, then run "
|
|
555
|
+
f"`switchroom doctor`.",
|
|
556
|
+
file=sys.stderr,
|
|
557
|
+
)
|
|
558
|
+
else:
|
|
559
|
+
print(
|
|
560
|
+
f"[Hindsight] Stop retain failed: queued to pending-retains "
|
|
561
|
+
f"(error: {type(err).__name__}: {err}, pending={pending_count()}). "
|
|
562
|
+
f"Will retry on next SessionStart.",
|
|
563
|
+
file=sys.stderr,
|
|
564
|
+
)
|
|
565
|
+
except Exception as e: # pragma: no cover - defensive
|
|
566
|
+
print(f"[Hindsight] Stop retain enqueue failed: {e}", file=sys.stderr)
|
|
411
567
|
|
|
412
568
|
|
|
413
569
|
if __name__ == "__main__":
|
|
@@ -88,6 +88,20 @@ def main():
|
|
|
88
88
|
# this up via run-hook.sh — see exit-code path in __main__.
|
|
89
89
|
debug_log(config, f"drain_pending unexpected error (ignored): {e}")
|
|
90
90
|
|
|
91
|
+
# Boot reconciliation (switchroom #3244): AFTER the drain, diff the durable
|
|
92
|
+
# per-session watermark against the on-disk transcript tail and recover any
|
|
93
|
+
# un-committed human turns an abrupt session death (SIGKILL / OOM /
|
|
94
|
+
# watchdog) skipped — the drain only replays what SessionEnd managed to
|
|
95
|
+
# enqueue, which an abrupt kill never runs. This is the load-bearing
|
|
96
|
+
# guarantee; it is naturally cheap (skips sessions with no gap) and bounded
|
|
97
|
+
# (lookback / turn-cap / wall-clock budget, each enqueuing its remainder).
|
|
98
|
+
try:
|
|
99
|
+
from reconcile_tail import reconcile as reconcile_tail
|
|
100
|
+
|
|
101
|
+
reconcile_tail(config, hook_input=hook_input)
|
|
102
|
+
except Exception as e:
|
|
103
|
+
debug_log(config, f"reconcile_tail unexpected error (ignored): {e}")
|
|
104
|
+
|
|
91
105
|
|
|
92
106
|
if __name__ == "__main__":
|
|
93
107
|
try:
|