switchroom 0.21.14 → 0.21.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/bin/rules-sentinel-hook.sh +101 -0
  2. package/dist/agent-scheduler/index.js +7 -2
  3. package/dist/auth-broker/index.js +7 -2
  4. package/dist/cli/notion-write-pretool.mjs +7 -2
  5. package/dist/cli/switchroom.js +2601 -1009
  6. package/dist/host-control/main.js +8 -3
  7. package/dist/vault/approvals/kernel-server.js +7 -2
  8. package/dist/vault/broker/server.js +7 -2
  9. package/package.json +1 -1
  10. package/profiles/_base/start.sh.hbs +9 -0
  11. package/profiles/_shared/delegation-golden-rule.md.hbs +2 -0
  12. package/skills/mental-model-curator/SKILL.md +187 -56
  13. package/telegram-plugin/dist/gateway/gateway.js +11 -6
  14. package/vendor/hindsight-memory/hooks/hooks.json +10 -0
  15. package/vendor/hindsight-memory/scripts/lib/client.py +14 -0
  16. package/vendor/hindsight-memory/scripts/lib/config.py +22 -0
  17. package/vendor/hindsight-memory/scripts/lib/directives.py +45 -7
  18. package/vendor/hindsight-memory/scripts/lib/recall_buffer.py +236 -0
  19. package/vendor/hindsight-memory/scripts/lib/watermark.py +27 -0
  20. package/vendor/hindsight-memory/scripts/prefetch.py +156 -0
  21. package/vendor/hindsight-memory/scripts/recall.py +520 -5
  22. package/vendor/hindsight-memory/scripts/reconcile_tail.py +4 -12
  23. package/vendor/hindsight-memory/scripts/retain.py +167 -28
  24. package/vendor/hindsight-memory/scripts/tests/test_config_retain_tool_calls_env.py +98 -0
  25. package/vendor/hindsight-memory/scripts/tests/test_directives.py +52 -0
  26. package/vendor/hindsight-memory/scripts/tests/test_incremental_sweep.py +293 -0
  27. package/vendor/hindsight-memory/scripts/tests/test_prefetch_pipeline.py +247 -0
  28. package/vendor/hindsight-memory/scripts/tests/test_profile_capture_nudge.py +335 -0
  29. package/vendor/hindsight-memory/scripts/tests/test_recall_buffer.py +143 -0
  30. package/vendor/hindsight-memory/scripts/tests/test_recall_buffer_join.py +193 -0
  31. package/vendor/hindsight-memory/scripts/tests/test_recall_cap_truncation.py +133 -0
  32. package/vendor/hindsight-memory/scripts/tests/test_recall_junk_gate.py +168 -0
  33. package/vendor/hindsight-memory/scripts/tests/test_recall_no_score_floor.py +124 -0
  34. package/vendor/hindsight-memory/scripts/tests/test_recall_query_timestamp.py +376 -0
  35. package/vendor/hindsight-memory/scripts/tests/test_retain_delta.py +304 -0
  36. package/vendor/hindsight-memory/scripts/tests/test_retain_stop_hook_prefetch_gate.py +109 -0
@@ -178,6 +178,7 @@ class HindsightClient:
178
178
  tags_match: Optional[str] = None,
179
179
  tag_groups: Optional[object] = None,
180
180
  prefer_observations: Optional[bool] = None,
181
+ query_timestamp: Optional[str] = None,
181
182
  timeout: int = 10,
182
183
  ) -> dict:
183
184
  """Recall memories from a bank.
@@ -190,6 +191,17 @@ class HindsightClient:
190
191
  `prefer_observations=True` asks the engine to prefer deduped
191
192
  observation statements over the raw facts they supersede, backfilling
192
193
  the freed slots — denser coverage inside the same token/count budget.
194
+
195
+ ``query_timestamp`` is an ISO 8601 datetime naming when the query is
196
+ being asked, from the user's perspective. The engine uses it as the
197
+ anchor for resolving relative temporal expressions in the query ("last
198
+ week", "yesterday") and for recency scoring; absent, the server's own
199
+ current time is the anchor. Sent only when non-empty, so a ``None``
200
+ (the default) leaves the wire body byte-identical to a pre-field
201
+ client — the additive-field invariant switchroom P2 depends on. The
202
+ REST recall body validates this field's format (a malformed value
203
+ 400s: "Invalid query_timestamp format. Expected ISO format"), so the
204
+ caller is responsible for passing a well-formed ISO string.
193
205
  """
194
206
  path = f"/v1/default/banks/{urllib.parse.quote(bank_id, safe='')}/memories/recall"
195
207
  body = {
@@ -208,6 +220,8 @@ class HindsightClient:
208
220
  body["tag_groups"] = tag_groups
209
221
  if prefer_observations is not None:
210
222
  body["prefer_observations"] = prefer_observations
223
+ if query_timestamp:
224
+ body["query_timestamp"] = query_timestamp
211
225
  return self._request("POST", path, body, timeout=timeout)
212
226
 
213
227
  def retain(
@@ -122,6 +122,14 @@ DEFAULTS = {
122
122
  # out per-agent via memory.directive_capture_nudge=false →
123
123
  # HINDSIGHT_DIRECTIVE_CAPTURE_NUDGE (disables BOTH hooks).
124
124
  "directiveCaptureNudge": True,
125
+ # RFC phase4 P3 — operator-profile capture nudge (recall.py,
126
+ # UserPromptSubmit). Regex-detects a first-person durable self-statement by
127
+ # the operator and appends a terse advisory telling the model to persist it
128
+ # with an explicit retain tagged `profile:ken` into the agent's OWN bank.
129
+ # Pure detection — no model callsite, no silent hook-side write. Operators
130
+ # opt out per-agent via memory.profile_capture_nudge=false →
131
+ # HINDSIGHT_PROFILE_CAPTURE_NUDGE.
132
+ "profileCaptureNudge": True,
125
133
  # Switchroom #2873/#2903 Fix 6.2 — the BLOCKING half (Stage C
126
134
  # directive_verify.py Stop hook) split out from the advisory nudge. When
127
135
  # True (default) the verifier may block the stop once to re-prompt capture;
@@ -426,6 +434,15 @@ ENV_OVERRIDES = {
426
434
  "HINDSIGHT_RETAIN_OVERLAP_TURNS": ("retainOverlapTurns", int),
427
435
  "HINDSIGHT_RETAIN_CONTEXT": ("retainContext", str),
428
436
  "HINDSIGHT_RETAIN_TAGS": ("retainTags", list),
437
+ # `retainToolCalls` (RFC memory-redesign P4): whether retain stores tool_use
438
+ # inputs + tool_result content. Had a DEFAULTS entry (True) but — like the
439
+ # cadence knobs before them — no env channel and no scaffold stamp, so an
440
+ # operator could not opt an agent out and a docker-exec'd retain/backfill
441
+ # could not be steered. Adding the env key mirrors the yaml surface
442
+ # (`memory.retain.tool_calls`) and closes the same drift class. `false`
443
+ # (via `false`/`0`/`no`) resolves to Python False and lands over the True
444
+ # default; unset keeps True, byte-identical.
445
+ "HINDSIGHT_RETAIN_TOOL_CALLS": ("retainToolCalls", bool),
429
446
  # Switchroom-local: per-row observation scope on retains. Set by start.sh
430
447
  # from agents.<name>.memory.observation_scopes (cascading through
431
448
  # defaults.memory.observation_scopes) ONLY when the operator set it; unset
@@ -488,6 +505,11 @@ ENV_OVERRIDES = {
488
505
  # the operator overrode it; the switchroom default is on (settings.json
489
506
  # pins true; recall.py falls back to True).
490
507
  "HINDSIGHT_DIRECTIVE_CAPTURE_NUDGE": ("directiveCaptureNudge", bool),
508
+ # RFC phase4 P3: operator-profile capture nudge on/off. Set by start.sh from
509
+ # agents.<name>.memory.profile_capture_nudge only when the operator overrode
510
+ # it; the switchroom default is on (settings.json pins true; recall.py falls
511
+ # back to True).
512
+ "HINDSIGHT_PROFILE_CAPTURE_NUDGE": ("profileCaptureNudge", bool),
491
513
  # Switchroom #2873/#2903 Fix 6.2: the Stage C block on/off, independent of
492
514
  # the Stage B nudge. Set by start.sh from
493
515
  # agents.<name>.memory.directive_capture_verify only when the operator
@@ -47,8 +47,9 @@ from .state import list_state_names, read_state, remove_state, write_state
47
47
  # move the doctor thresholds with it.
48
48
  #
49
49
  # Banks with more active directives than this are pathological; we truncate
50
- # with an in-prompt footer, a `directives_omitted` field on the recall_log row,
51
- # and a stderr warning (see `format_active_directives_block`).
50
+ # with a LOUD, operator-directed in-prompt overflow notice (memory-RFC P7 — the
51
+ # in-turn channel), a `directives_omitted` field on the recall_log row, and a
52
+ # stderr warning (see `format_active_directives_block`).
52
53
  MAX_DIRECTIVES = 30
53
54
 
54
55
  # Hard timeout for the list_directives call. The recall hook is on the
@@ -272,6 +273,10 @@ def format_active_directives_block(directives: list, max_directives: int = MAX_D
272
273
  1. [P10] <name>: <content>
273
274
  2. [P9] <name>: <content>
274
275
  ...
276
+
277
+ === DIRECTIVE OVERFLOW — ACTION REQUIRED ===
278
+ <N of TOTAL directives dropped; agent is told to surface it to the
279
+ operator this turn and merge/retire directives> (only when omitted > 0)
275
280
  (+N more, omitted)
276
281
  </active_directives>
277
282
  """
@@ -301,12 +306,45 @@ def format_active_directives_block(directives: list, max_directives: int = MAX_D
301
306
  lines.append(f"{i}. [P{priority}] {name}: {content}")
302
307
 
303
308
  if omitted > 0:
309
+ # LOUD in-prompt overflow notice (memory-RFC P7). The prior footer was a
310
+ # single quiet parenthetical — `(+N more, omitted)` — which the agent
311
+ # could read past without registering that explicit, operator-authored
312
+ # instructions had been silently DROPPED from this turn. The three other
313
+ # signals this module records are either not operator-visible (stderr —
314
+ # swallowed, see below) or not in-turn (the `directives_omitted`
315
+ # recall_log row and `switchroom doctor`, both read after the fact). The
316
+ # in-prompt block is the one channel that is BOTH in-turn and reaches an
317
+ # operator — indirectly, because the agent relays it in its reply. So we
318
+ # make the footer loud and explicitly action-directed: state the loss,
319
+ # name the count and cap, and instruct the agent to surface it to the
320
+ # operator THIS turn. This is the P7 "loud channel" and is deliberately
321
+ # NOT a MAX_DIRECTIVES change (raising the cap only moves the silent-drop
322
+ # point; visibility has to ship first).
323
+ #
324
+ # The literal `(+N more, omitted)` marker is retained verbatim so the
325
+ # existing recall_log/doctor cross-checks and the directive-dedup parser
326
+ # (`parse_active_directives_block`) keep working unchanged.
304
327
  lines.append("")
328
+ lines.append("=== DIRECTIVE OVERFLOW — ACTION REQUIRED ===")
329
+ lines.append(
330
+ f"{omitted} of {total} active directives were DROPPED from this turn's "
331
+ f"prompt: the bank exceeds MAX_DIRECTIVES={max_directives}, so the "
332
+ f"{omitted} LOWEST-priority directive(s) are NOT in effect this turn. "
333
+ "This is silent loss of explicit, operator-authored instructions."
334
+ )
335
+ lines.append(
336
+ "ACTION: tell the operator in your reply this turn that directive "
337
+ "overflow is dropping rules, then merge or retire duplicate/stale "
338
+ "directives (mental-model-curator skill) to get the bank back under "
339
+ "the cap. Do not let this pass unmentioned."
340
+ )
305
341
  lines.append(f"(+{omitted} more, omitted)")
306
- # The in-prompt footer above only tells the AGENT. This stderr warn is
307
- # the same channel every other operational failure in this module uses
308
- # (see `_fetch_directives_with_status`), and it is a LAST-RESORT
309
- # breadcrumb only — do NOT rely on it reaching an operator.
342
+ # The in-prompt notice above is the operator-visible in-turn channel
343
+ # (P7): the agent reads it every turn the overflow persists and is told
344
+ # to relay it. This stderr warn is the same channel every other
345
+ # operational failure in this module uses (see
346
+ # `_fetch_directives_with_status`), and it is a LAST-RESORT breadcrumb
347
+ # only — do NOT rely on it reaching an operator.
310
348
  #
311
349
  # Measured 2026-07-25: `docker logs --tail 20000` across all 12 running
312
350
  # agent containers returns ZERO `[Hindsight]` lines, and nothing under
@@ -315,7 +353,7 @@ def format_active_directives_block(directives: list, max_directives: int = MAX_D
315
353
  # appears to swallow hook stderr on a zero exit, so hook stderr is not
316
354
  # an operator-visible channel.
317
355
  #
318
- # The channels that DO reach an operator:
356
+ # The other channels that reach an operator, but only AFTER the turn:
319
357
  # * the `directives_omitted` field on the recall_log row
320
358
  # (state/recall_log.jsonl — see `count_omitted_directives`), and
321
359
  # * `switchroom doctor`'s WARN/FAIL on the bank's active directive
@@ -0,0 +1,236 @@
1
+ """M4 P0 — prefetch buffer + sentinel primitive.
2
+
3
+ The end-of-turn-N Stop hook (``prefetch.py``) writes a prefetched recall
4
+ block into a per-session buffer file, then writes a ``buffer.done`` sentinel
5
+ LAST. The start-of-turn-N+1 UserPromptSubmit hook (``recall.py``) polls for a
6
+ FRESH sentinel (bounded by ``poll_for_sentinel``'s ``cap_ms``) and, on a hit,
7
+ consumes the buffered block instead of running recall synchronously.
8
+
9
+ Read-after-write safety (the whole ballgame — carve §4 note 1):
10
+
11
+ * The payload is written via temp-file + atomic ``os.replace`` (POSIX
12
+ atomic within the same directory).
13
+ * The sentinel is written LAST, also via temp-file + ``os.fsync`` +
14
+ atomic ``os.replace``, and carries a MONOTONIC token (a `time.time_ns()`
15
+ counter, not wall-clock-comparable across machines but perfectly
16
+ ordered within one) — never content or mtime comparison, since two
17
+ writes CAN land in the same filesystem mtime tick.
18
+ * ``read_if_fresh`` treats "sentinel absent" OR "sentinel token not newer
19
+ than ``last_consumed_token``" as a miss (``None``). A torn write
20
+ (payload present, sentinel absent — e.g. the producer crashed between
21
+ the two writes) is INDISTINGUISHABLE from "no fresh buffer yet" by
22
+ construction, so it fails closed to ``None`` rather than serving a
23
+ possibly-incomplete/stale payload.
24
+
25
+ State dir: ``$HOME/.hindsight/prefetch-buffer/`` (override via
26
+ ``HINDSIGHT_PREFETCH_BUFFER_DIR`` for tests), mirroring the
27
+ ``lib/watermark.py`` convention (a distinct dir, not a shared one, so a
28
+ polling reader and a writing producer never contend for the same files as
29
+ unrelated retain-watermark state).
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import json
35
+ import os
36
+ import re
37
+ import sys
38
+ import time
39
+ from typing import Optional
40
+
41
+ if sys.platform != "win32":
42
+ import fcntl
43
+ else: # pragma: no cover - switchroom agents are Linux only
44
+ fcntl = None
45
+
46
+ SCHEMA = 1
47
+
48
+ # Default poll behaviour (Hermes's 3s join, hand-rolled — P-REC/P-PRE tune
49
+ # the actual cap via config; this is the primitive's own conservative floor
50
+ # if a caller passes a cap smaller than one sleep tick).
51
+ _MIN_SLEEP_S = 0.02
52
+ _MAX_SLEEP_S = 0.1
53
+
54
+
55
+ def buffer_dir() -> str:
56
+ """Return the prefetch-buffer directory.
57
+
58
+ Override with ``HINDSIGHT_PREFETCH_BUFFER_DIR`` (tests). Default:
59
+ ``$HOME/.hindsight/prefetch-buffer/``.
60
+ """
61
+ override = os.environ.get("HINDSIGHT_PREFETCH_BUFFER_DIR")
62
+ if override:
63
+ return override
64
+ return os.path.join(os.path.expanduser("~"), ".hindsight", "prefetch-buffer")
65
+
66
+
67
+ def _ensure_dir() -> str:
68
+ d = buffer_dir()
69
+ if not os.path.isdir(d):
70
+ os.makedirs(d, mode=0o700, exist_ok=True)
71
+ else:
72
+ try:
73
+ mode = os.stat(d).st_mode & 0o777
74
+ if mode != 0o700:
75
+ os.chmod(d, 0o700)
76
+ except OSError:
77
+ pass
78
+ return d
79
+
80
+
81
+ def _safe_session(session_id: str) -> str:
82
+ """Sanitise a session id for use as a filename (no path traversal)."""
83
+ keep = [c if (c.isalnum() or c in "-_.") else "_" for c in (session_id or "unknown")]
84
+ name = "".join(keep)[:200]
85
+ return name.replace("..", "_") or "unknown"
86
+
87
+
88
+ def _buffer_path(session_id: str) -> str:
89
+ return os.path.join(buffer_dir(), f"{_safe_session(session_id)}.buffer.json")
90
+
91
+
92
+ def _sentinel_path(session_id: str) -> str:
93
+ return os.path.join(buffer_dir(), f"{_safe_session(session_id)}.buffer.done")
94
+
95
+
96
+ def write_buffer(session_id: str, context: str, telemetry: Optional[dict] = None) -> None:
97
+ """Write the prefetched recall payload for ``session_id``.
98
+
99
+ Atomic (temp file + ``os.replace`` within the same directory). Does NOT
100
+ write the sentinel — the caller MUST call ``write_sentinel`` after this,
101
+ and only once the payload write has returned, to preserve the
102
+ read-after-write ordering guarantee.
103
+ """
104
+ d = _ensure_dir()
105
+ final = _buffer_path(session_id)
106
+ tmp = final + f".tmp.{os.getpid()}"
107
+ entry = {
108
+ "schema": SCHEMA,
109
+ "session_id": session_id,
110
+ "context": context,
111
+ "telemetry": telemetry or {},
112
+ "written_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
113
+ }
114
+ with open(tmp, "w", encoding="utf-8") as f:
115
+ json.dump(entry, f, ensure_ascii=False)
116
+ f.flush()
117
+ os.fsync(f.fileno())
118
+ os.chmod(tmp, 0o600)
119
+ os.replace(tmp, final)
120
+
121
+
122
+ def write_sentinel(session_id: str) -> int:
123
+ """Write the ``buffer.done`` sentinel for ``session_id``, LAST.
124
+
125
+ Returns the monotonic token written (an ever-increasing integer derived
126
+ from ``time.time_ns()``, disambiguated against the previous token on
127
+ this session so two writes within the same nanosecond still order).
128
+ fsync'd before the atomic rename so a reader that observes the renamed
129
+ file is guaranteed to observe durable content.
130
+ """
131
+ d = _ensure_dir()
132
+ final = _sentinel_path(session_id)
133
+ token = time.time_ns()
134
+ # Guarantee strict monotonicity even under nanosecond-resolution ties or
135
+ # clock weirdness: never emit a token <= the previously written one.
136
+ prev = _read_sentinel_token(session_id)
137
+ if prev is not None and token <= prev:
138
+ token = prev + 1
139
+ tmp = final + f".tmp.{os.getpid()}"
140
+ with open(tmp, "w", encoding="utf-8") as f:
141
+ json.dump({"schema": SCHEMA, "token": token}, f)
142
+ f.flush()
143
+ os.fsync(f.fileno())
144
+ os.chmod(tmp, 0o600)
145
+ os.replace(tmp, final)
146
+ return token
147
+
148
+
149
+ def _read_sentinel_token(session_id: str) -> Optional[int]:
150
+ path = _sentinel_path(session_id)
151
+ if not os.path.isfile(path):
152
+ return None
153
+ try:
154
+ with open(path, encoding="utf-8") as f:
155
+ data = json.load(f)
156
+ token = data.get("token")
157
+ return int(token) if token is not None else None
158
+ except (OSError, json.JSONDecodeError, ValueError, TypeError):
159
+ return None
160
+
161
+
162
+ def _read_buffer_payload(session_id: str) -> Optional[dict]:
163
+ path = _buffer_path(session_id)
164
+ if not os.path.isfile(path):
165
+ return None
166
+ try:
167
+ with open(path, encoding="utf-8") as f:
168
+ return json.load(f)
169
+ except (OSError, json.JSONDecodeError):
170
+ return None
171
+
172
+
173
+ def sentinel_exists(session_id: str) -> bool:
174
+ """Return True iff a ``buffer.done`` sentinel has EVER been written for
175
+ ``session_id`` (regardless of freshness).
176
+
177
+ Cold-start guard (red-team MAJOR finding): a reader must be able to tell
178
+ "no producer has ever run for this session" from "a producer ran but
179
+ hasn't finished this turn's slice yet" WITHOUT paying the full poll cap
180
+ — the former should skip polling entirely (every session-open turn
181
+ would otherwise eat the full poll cap, the opposite of M4's latency
182
+ goal), the latter is exactly what polling is for.
183
+ """
184
+ return os.path.isfile(_sentinel_path(session_id))
185
+
186
+
187
+ def read_if_fresh(session_id: str, last_consumed_token: Optional[int]) -> tuple:
188
+ """Return ``(payload_dict | None, current_token)``.
189
+
190
+ ``payload_dict`` (when present) is ``{"context": str, "telemetry": dict}``.
191
+
192
+ Returns ``(None, token_or_last_consumed)`` when:
193
+ * no sentinel exists yet (nothing has been produced this session), or
194
+ * the sentinel's token is not strictly newer than ``last_consumed_token``
195
+ (already consumed — stale from the reader's perspective), or
196
+ * the buffer payload itself is missing/corrupt (a torn write, or a
197
+ sentinel written with no payload at all) — fail-closed, never serve
198
+ a partial/absent payload as fresh.
199
+ """
200
+ token = _read_sentinel_token(session_id)
201
+ if token is None:
202
+ return None, last_consumed_token
203
+ if last_consumed_token is not None and token <= last_consumed_token:
204
+ return None, token
205
+ payload = _read_buffer_payload(session_id)
206
+ if payload is None:
207
+ # Torn write: sentinel landed but payload didn't (or is corrupt).
208
+ # Fail-closed — never serve this as fresh.
209
+ return None, token
210
+ return {"context": payload.get("context", ""), "telemetry": payload.get("telemetry", {})}, token
211
+
212
+
213
+ def poll_for_sentinel(session_id: str, last_consumed_token: Optional[int], cap_ms: int) -> bool:
214
+ """Clock-bounded poll for a fresh sentinel. Never busy-spins.
215
+
216
+ Returns True as soon as ``read_if_fresh`` would return a payload; False
217
+ once the cumulative sleep time reaches ``cap_ms``. Short sleeps (20-100ms,
218
+ backing off) sum to at most ``cap_ms`` — never a hard-loop re-stat.
219
+ """
220
+ if cap_ms <= 0:
221
+ ctx, _ = read_if_fresh(session_id, last_consumed_token)
222
+ return ctx is not None
223
+
224
+ deadline = time.monotonic() + (cap_ms / 1000.0)
225
+ sleep_s = _MIN_SLEEP_S
226
+ while True:
227
+ ctx, _ = read_if_fresh(session_id, last_consumed_token)
228
+ if ctx is not None:
229
+ return True
230
+ remaining = deadline - time.monotonic()
231
+ if remaining <= 0:
232
+ return False
233
+ this_sleep = min(sleep_s, remaining, _MAX_SLEEP_S)
234
+ if this_sleep > 0:
235
+ time.sleep(this_sleep)
236
+ sleep_s = min(sleep_s * 1.5, _MAX_SLEEP_S)
@@ -92,6 +92,33 @@ def _lock_path(session_id: str) -> str:
92
92
  return os.path.join(watermark_dir(), f"{_safe_session(session_id)}.lock")
93
93
 
94
94
 
95
+ def tail_after(messages: list, last_uuid: Optional[str]) -> list:
96
+ """Return the transcript entries AFTER the committed watermark anchor.
97
+
98
+ Pure and IO-free — the single shared slice used by BOTH the boot reconciler
99
+ (``reconcile_tail``) and the incremental SessionEnd sweep (``retain``, the
100
+ switchroom memory-RFC P1 change). One implementation, so the reconcile /
101
+ watermark failure category cannot grow a second, divergent copy.
102
+
103
+ Two safety fallbacks, both returning the WHOLE transcript — a safe
104
+ re-upsert, never a skip, never an empty slice:
105
+
106
+ * ``last_uuid`` falsy (no committed watermark) — e.g. a short session that
107
+ never fired a per-window retain, so the force sweep must flush it whole.
108
+ * ``last_uuid`` absent from ``messages`` (compaction removed the anchor).
109
+
110
+ Callers in the retain seam depend on the whole-transcript fallback: an empty
111
+ or skipped slice there DELETES a turn rather than degrading it
112
+ (``retain.py`` §4.3 hazard). Never change a fallback here to return ``[]``.
113
+ """
114
+ if not last_uuid:
115
+ return list(messages)
116
+ for i, m in enumerate(messages):
117
+ if isinstance(m, dict) and m.get("uuid") == last_uuid:
118
+ return messages[i + 1:]
119
+ return list(messages)
120
+
121
+
95
122
  def load(session_id: str) -> Optional[dict]:
96
123
  """Return the stored watermark dict for ``session_id``, or ``None``."""
97
124
  p = _path(session_id)
@@ -0,0 +1,156 @@
1
+ #!/usr/bin/env python3
2
+ """M4 P-PRE — async Stop-hook prefetch producer.
3
+
4
+ carve-M4.md: moves recall injection off the synchronous UserPromptSubmit
5
+ path by speculatively retaining + recalling at the END of turn N (this
6
+ script, registered as an async Stop hook), so turn N+1's UserPromptSubmit
7
+ (`recall.py`) can join an already-warm buffer instead of paying the full
8
+ recall latency synchronously.
9
+
10
+ Entirely gated by `memoryPrefetchEnabled` (default OFF/falsy — Fix C, the
11
+ red-team-mandated per-agent kill switch). When off this script is a no-op:
12
+ it reads config, sees the flag off, and exits silently before touching
13
+ `lib.retain`, `lib.recall_buffer`, or the network.
14
+
15
+ Call order (test a-integration in `tests/test_prefetch_pipeline.py`):
16
+ 1. delta retain (`retain.run_retain(hook_input, force=False, delta=True)`)
17
+ — persists this turn's NEW slice using the Fix-A content-derived
18
+ document_id, never truncating the session document.
19
+ 2. speculative recall (best-effort query derived from the transcript's
20
+ last human turn — the strongest available proxy for what turn N+1
21
+ will ask about).
22
+ 3. `recall_buffer.write_buffer` then `recall_buffer.write_sentinel`
23
+ (STRICTLY in that order — the sentinel-ordering contract `lib/
24
+ recall_buffer.py` and `tests/test_recall_buffer.py` prove).
25
+
26
+ Silent on stdout always (an async Stop hook's stdout is not surfaced to the
27
+ transcript); errors go to stderr and NEVER raise past `main()` — a broken
28
+ prefetch must degrade to "recall.py's synchronous path runs as before",
29
+ never break the turn or leave a torn buffer (a crash between steps 2 and 3a
30
+ just means no sentinel is written this turn, which `read_if_fresh` already
31
+ treats as "nothing fresh" — fail-closed by construction).
32
+
33
+ Skips the same junk turns `recall.py` skips (`<task-notification>` prefix)
34
+ — a synthetic follow-up turn is not a real conversation shift worth
35
+ prefetching for.
36
+ """
37
+
38
+ from __future__ import annotations
39
+
40
+ import json
41
+ import sys
42
+
43
+ from lib.bank import derive_bank_id
44
+ from lib.client import HindsightClient
45
+ from lib.config import debug_log, load_config
46
+ from lib.content import _extract_text_content, strip_channel_envelope
47
+ from lib.daemon import get_api_url
48
+ from lib import recall_buffer
49
+ import retain as retain_module
50
+
51
+
52
+ def _last_human_prompt(transcript_path: str) -> str:
53
+ """Best-effort: the most recent human-authored user turn's text, used
54
+ as the speculative query for next turn's prefetch. Never raises —
55
+ returns "" on any read/parse failure (degrades to no query, which the
56
+ caller treats as "nothing to prefetch")."""
57
+ try:
58
+ messages = retain_module.read_transcript(transcript_path)
59
+ except Exception:
60
+ return ""
61
+ for msg in reversed(messages):
62
+ if msg.get("role") != "user":
63
+ continue
64
+ text = _extract_text_content(msg.get("content"), role="user")
65
+ text = strip_channel_envelope(text) if text else text
66
+ if text and text.strip():
67
+ return text.strip()
68
+ return ""
69
+
70
+
71
+ def run_prefetch(hook_input: dict, config: dict) -> bool:
72
+ """Execute one prefetch cycle. Returns True iff a buffer+sentinel pair
73
+ was written. Never raises — every internal step is wrapped so a bug in
74
+ ANY sub-step degrades to "no buffer written this turn", never a crash
75
+ that could take the Stop hook (and thus the turn) down with it."""
76
+ session_id = hook_input.get("session_id") or "unknown"
77
+
78
+ prompt = (hook_input.get("prompt") or hook_input.get("user_prompt") or "").strip()
79
+ if prompt.startswith("<task-notification"):
80
+ debug_log(config, "Prefetch: task-notification turn, skipping")
81
+ return False
82
+
83
+ # Step 1 — delta retain (Fix A: content-derived document_id, never the
84
+ # bare {session_id} document; never truncates).
85
+ try:
86
+ retain_module.run_retain(hook_input, force=False, delta=True)
87
+ except Exception as exc: # pragma: no cover - defensive
88
+ debug_log(config, f"Prefetch: delta retain failed, continuing without it: {exc}")
89
+
90
+ # Step 2 — speculative recall.
91
+ transcript_path = hook_input.get("transcript_path") or ""
92
+ query = _last_human_prompt(transcript_path)
93
+ if not query:
94
+ debug_log(config, "Prefetch: no usable query, nothing to prefetch")
95
+ return False
96
+
97
+ try:
98
+ bank_id = derive_bank_id(hook_input, config)
99
+ api_url = get_api_url(config)
100
+ client = HindsightClient(api_url)
101
+ response = client.recall(
102
+ bank_id,
103
+ query,
104
+ types=None,
105
+ timeout=config.get("memoryPrefetchTimeoutSeconds", 5),
106
+ )
107
+ results = (response or {}).get("results") or []
108
+ except Exception as exc: # pragma: no cover - defensive
109
+ debug_log(config, f"Prefetch: recall fetch failed: {exc}")
110
+ return False
111
+
112
+ if not results:
113
+ debug_log(config, "Prefetch: no candidates, nothing to buffer")
114
+ return False
115
+
116
+ from lib.content import format_memories
117
+
118
+ memories_block = format_memories(results)
119
+ if not memories_block:
120
+ return False
121
+
122
+ # Step 3 — write payload THEN sentinel, strictly in that order.
123
+ try:
124
+ recall_buffer.write_buffer(session_id, memories_block, {"result_count": len(results)})
125
+ recall_buffer.write_sentinel(session_id)
126
+ except Exception as exc: # pragma: no cover - defensive
127
+ debug_log(config, f"Prefetch: buffer write failed: {exc}")
128
+ return False
129
+
130
+ return True
131
+
132
+
133
+ def main():
134
+ config = load_config()
135
+ if not config.get("memoryPrefetchEnabled", False):
136
+ # Fix C, producer side: the whole mechanism is dark by default.
137
+ return
138
+
139
+ try:
140
+ hook_input = json.load(sys.stdin)
141
+ except (json.JSONDecodeError, EOFError):
142
+ print("[Hindsight] Prefetch: failed to read hook input", file=sys.stderr)
143
+ return
144
+
145
+ try:
146
+ run_prefetch(hook_input, config)
147
+ except Exception as exc: # pragma: no cover - defensive, must never break the turn
148
+ print(f"[Hindsight] Prefetch: unexpected error: {exc}", file=sys.stderr)
149
+
150
+
151
+ if __name__ == "__main__":
152
+ try:
153
+ main()
154
+ except Exception as e: # pragma: no cover - defensive
155
+ print(f"[Hindsight] Unexpected error in prefetch: {e}", file=sys.stderr)
156
+ sys.exit(0)