switchroom 0.21.14 → 0.21.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/bin/rules-sentinel-hook.sh +101 -0
  2. package/dist/agent-scheduler/index.js +7 -2
  3. package/dist/auth-broker/index.js +7 -2
  4. package/dist/cli/notion-write-pretool.mjs +7 -2
  5. package/dist/cli/switchroom.js +2601 -1009
  6. package/dist/host-control/main.js +8 -3
  7. package/dist/vault/approvals/kernel-server.js +7 -2
  8. package/dist/vault/broker/server.js +7 -2
  9. package/package.json +1 -1
  10. package/profiles/_base/start.sh.hbs +9 -0
  11. package/profiles/_shared/delegation-golden-rule.md.hbs +2 -0
  12. package/skills/mental-model-curator/SKILL.md +187 -56
  13. package/telegram-plugin/dist/gateway/gateway.js +11 -6
  14. package/vendor/hindsight-memory/hooks/hooks.json +10 -0
  15. package/vendor/hindsight-memory/scripts/lib/client.py +14 -0
  16. package/vendor/hindsight-memory/scripts/lib/config.py +22 -0
  17. package/vendor/hindsight-memory/scripts/lib/directives.py +45 -7
  18. package/vendor/hindsight-memory/scripts/lib/recall_buffer.py +236 -0
  19. package/vendor/hindsight-memory/scripts/lib/watermark.py +27 -0
  20. package/vendor/hindsight-memory/scripts/prefetch.py +156 -0
  21. package/vendor/hindsight-memory/scripts/recall.py +520 -5
  22. package/vendor/hindsight-memory/scripts/reconcile_tail.py +4 -12
  23. package/vendor/hindsight-memory/scripts/retain.py +167 -28
  24. package/vendor/hindsight-memory/scripts/tests/test_config_retain_tool_calls_env.py +98 -0
  25. package/vendor/hindsight-memory/scripts/tests/test_directives.py +52 -0
  26. package/vendor/hindsight-memory/scripts/tests/test_incremental_sweep.py +293 -0
  27. package/vendor/hindsight-memory/scripts/tests/test_prefetch_pipeline.py +247 -0
  28. package/vendor/hindsight-memory/scripts/tests/test_profile_capture_nudge.py +335 -0
  29. package/vendor/hindsight-memory/scripts/tests/test_recall_buffer.py +143 -0
  30. package/vendor/hindsight-memory/scripts/tests/test_recall_buffer_join.py +193 -0
  31. package/vendor/hindsight-memory/scripts/tests/test_recall_cap_truncation.py +133 -0
  32. package/vendor/hindsight-memory/scripts/tests/test_recall_junk_gate.py +168 -0
  33. package/vendor/hindsight-memory/scripts/tests/test_recall_no_score_floor.py +124 -0
  34. package/vendor/hindsight-memory/scripts/tests/test_recall_query_timestamp.py +376 -0
  35. package/vendor/hindsight-memory/scripts/tests/test_retain_delta.py +304 -0
  36. package/vendor/hindsight-memory/scripts/tests/test_retain_stop_hook_prefetch_gate.py +109 -0
@@ -0,0 +1,335 @@
1
+ """RFC phase4 P3 — unit + end-to-end tests for the operator-profile capture nudge.
2
+
3
+ Ken's first stated want is "save memories about him". Auto-retain stores
4
+ transcript facts, but nothing gives a DETERMINISTIC signal that a durable
5
+ *profile fact* about the operator himself just went by — so profile capture is
6
+ left to model discretion (the same per-agent lottery Stage A measured for
7
+ directives). P3 mirrors the shipped directive-capture nudge (recall.py #2848):
8
+ a POSITIVE regex detects a first-person durable self-statement, a NEGATIVE
9
+ regex scrubs the two false-positive shapes (questions, attributions to others)
10
+ BEFORE the positive match, and on a hit the UserPromptSubmit hook appends a
11
+ terse advisory telling the model to persist it with an explicit retain tagged
12
+ `profile:ken` into the agent's OWN bank. Pure regex — no model callsite.
13
+
14
+ These tests pin (all OUTCOME assertions):
15
+ * True positives — the RFC's listed profile shapes fire.
16
+ * True negatives — questions and third-/second-party attributions do NOT
17
+ fire (the negative-lookaround guard scrubs them first).
18
+ * The advisory carries the `profile:ken` tag instruction and targets the
19
+ agent's OWN bank (not a shared / cross-agent person bank).
20
+ * End-to-end through recall.main(): the advisory reaches the emitted
21
+ additionalContext, the recall_log row carries `profile_nudge: true`, and
22
+ the config knob OFF (`profileCaptureNudge: false`) suppresses both.
23
+
24
+ Stdlib-only; runs under `python3 -m unittest discover tests/`. The end-to-end
25
+ harness mirrors test_recall_envelope_strip_telemetry.py.
26
+ """
27
+
28
+ import io
29
+ import json
30
+ import os
31
+ import shutil
32
+ import sys
33
+ import tempfile
34
+ import unittest
35
+ from unittest.mock import patch
36
+
37
+ SCRIPTS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
38
+ if SCRIPTS_DIR not in sys.path:
39
+ sys.path.insert(0, SCRIPTS_DIR)
40
+
41
+ import recall # noqa: E402
42
+ from recall import ( # noqa: E402
43
+ _PROFILE_CAPTURE_NUDGE,
44
+ _combine_context,
45
+ _is_trivial_stateless,
46
+ looks_like_profile_statement,
47
+ )
48
+
49
+
50
+ # First-person durable self-statements that MUST fire the nudge. Drawn from the
51
+ # RFC P3 shapes ("I prefer …", "my … is …", "I always …", "remind me that I …")
52
+ # plus the conservative identity/situation set.
53
+ PROFILE_STATEMENTS = [
54
+ # --- stated preferences ---
55
+ "I prefer dark roast coffee",
56
+ "I'd prefer British spelling in my docs",
57
+ "my preference is tabs over spaces",
58
+ # --- durable self-facts: "my <ATTRIBUTE> is/are …" (tight allow-list) ---
59
+ "my timezone is Australia/Melbourne",
60
+ "my email is ken@example.com",
61
+ "my sister is Lisa",
62
+ "my kids are at school during the day",
63
+ "my name's Ken", # contraction form
64
+ # --- identity / situation ---
65
+ "I live in Melbourne",
66
+ "I work at Anthropic",
67
+ "I'm allergic to peanuts",
68
+ "I'm based in Australia",
69
+ # --- durable identity: diet / abstention / "I'm a <noun>" ---
70
+ "I'm a vegetarian",
71
+ "I don't eat meat",
72
+ "call me Ken",
73
+ # --- tastes (durable like/dislike framing) ---
74
+ "I hate em-dashes",
75
+ # --- durable habits ---
76
+ "I always take my coffee black",
77
+ "I usually work late on Thursdays",
78
+ "I never eat red meat",
79
+ # --- explicit memory framing about the operator himself ---
80
+ "remind me that I have a standing 9am standup",
81
+ "remember that I hate em-dashes",
82
+ ]
83
+
84
+ # Questions and attributions that MUST NOT fire — the negative-lookaround guard
85
+ # scrubs these BEFORE the positive match. A false positive here is nudge noise.
86
+ NON_PROFILE = [
87
+ # --- questions about the operator (not statements of a durable fact) ---
88
+ "do I prefer tea or coffee?",
89
+ "what's my timezone?",
90
+ "where is my email address stored?",
91
+ "how do I fix this bug",
92
+ "should I always run the tests first?",
93
+ "what do I usually do here",
94
+ "remind me what my calendar looks like",
95
+ # --- attributions to a third / second party ---
96
+ "you said I prefer tea",
97
+ "she claims my code is broken",
98
+ "he thinks I always overcomplicate things",
99
+ "they told me my access was revoked",
100
+ # --- neither: no first-person durable self-fact ---
101
+ "the project timezone is UTC",
102
+ "please run the tests",
103
+ "what time is it",
104
+ # --- discourse-marker "my <X> is" — not a durable profile fact. The
105
+ # positive arm uses a TIGHT identity allow-list, so a free noun never
106
+ # reaches the matcher; these confirm that. ---
107
+ "my guess is the cache is stale",
108
+ "my point is that we should ship it",
109
+ "my concern is the timeout",
110
+ # --- transient dev state "my <transient> is/are …". These are the exact
111
+ # over-fires the free-`\w+` arm produced; the allow-list must NOT fire
112
+ # on them (RFC favour-false-negatives constraint on this agent). ---
113
+ "my container is down",
114
+ "my build is failing",
115
+ "my code is broken",
116
+ "my PR is ready",
117
+ "my worktree is dirty",
118
+ "my server is down",
119
+ "my deploy is stuck",
120
+ "my tests are green",
121
+ "my branch is merged",
122
+ # --- pleasantry embedding a bare always/never after "I" ---
123
+ "I always appreciate your help",
124
+ "I never enjoy waiting, but thanks",
125
+ # --- "I'm a <hedge>" is a transient mood, not an "I'm a <noun>" identity ---
126
+ "I'm a bit tired",
127
+ "I'm a little confused about the config",
128
+ "I'm a big fan of shipping fast",
129
+ # --- "call me <phrasing>" as a request, not a name form ---
130
+ "call me back later",
131
+ "call me when the build finishes",
132
+ # --- the <channel …> envelope wrapper on its own must never trigger ---
133
+ '<channel user="ken" chat_id="123">',
134
+ ]
135
+
136
+
137
+ class TestProfileDetection(unittest.TestCase):
138
+ def test_profile_statements_fire_the_nudge(self):
139
+ for p in PROFILE_STATEMENTS:
140
+ with self.subTest(prompt=p):
141
+ self.assertTrue(
142
+ looks_like_profile_statement(p),
143
+ f"expected profile detection for {p!r}",
144
+ )
145
+
146
+ def test_questions_and_attributions_do_not_fire(self):
147
+ for p in NON_PROFILE:
148
+ with self.subTest(prompt=p):
149
+ self.assertFalse(
150
+ looks_like_profile_statement(p),
151
+ f"FALSE POSITIVE: profile nudge would fire for {p!r}",
152
+ )
153
+
154
+ def test_empty_and_non_string_are_false(self):
155
+ for bad in ("", " ", None, 123, [], {}):
156
+ with self.subTest(value=bad):
157
+ self.assertFalse(looks_like_profile_statement(bad))
158
+
159
+ def test_attribution_scrub_does_not_mask_a_real_self_fact(self):
160
+ # A message that opens with an attributed clause AND then states the
161
+ # operator's own durable fact must still fire — the negative guard
162
+ # scrubs only the attributed span (through end-of-sentence).
163
+ self.assertTrue(
164
+ looks_like_profile_statement(
165
+ "she thinks I'm wrong. my timezone is Melbourne"
166
+ )
167
+ )
168
+
169
+ def test_attribution_scrub_stops_at_a_comma(self):
170
+ # The attributed span must stop at a comma, not run to end-of-sentence:
171
+ # a real self-fact trailing the attributed clause in the SAME sentence
172
+ # must still reach the positive matcher (MINOR: greedy [^.?!]* → [^.?!,]*).
173
+ self.assertTrue(
174
+ looks_like_profile_statement(
175
+ "she said the deploy failed, my timezone is Melbourne"
176
+ )
177
+ )
178
+
179
+
180
+ class TestProfileNudgeString(unittest.TestCase):
181
+ def test_nudge_carries_the_profile_ken_tag_instruction(self):
182
+ self.assertIn("profile:ken", _PROFILE_CAPTURE_NUDGE)
183
+ self.assertIn("retain", _PROFILE_CAPTURE_NUDGE)
184
+ self.assertIn("profile_capture_check", _PROFILE_CAPTURE_NUDGE)
185
+
186
+ def test_nudge_targets_the_agents_own_bank_not_a_shared_one(self):
187
+ # Constraint 2 forbids a cross-agent person bank — the advisory must
188
+ # route to the agent's OWN bank and say so explicitly.
189
+ self.assertIn("OWN bank", _PROFILE_CAPTURE_NUDGE)
190
+ self.assertIn("shared", _PROFILE_CAPTURE_NUDGE)
191
+
192
+ def test_combine_appends_nudge_after_recall_block(self):
193
+ base = "<hindsight_memories>\n…\n</hindsight_memories>"
194
+ out = _combine_context(base, _PROFILE_CAPTURE_NUDGE)
195
+ self.assertTrue(out.startswith(base))
196
+ self.assertTrue(out.endswith(_PROFILE_CAPTURE_NUDGE))
197
+
198
+
199
+ class TestTrivialSkipUnaffected(unittest.TestCase):
200
+ """A profile statement carries personal/stateful signal and must never be
201
+ trivial-stateless-skipped; greetings still are."""
202
+
203
+ def test_profile_statements_are_never_trivial_skipped(self):
204
+ for p in PROFILE_STATEMENTS:
205
+ with self.subTest(prompt=p):
206
+ self.assertFalse(
207
+ _is_trivial_stateless("", p),
208
+ f"profile statement {p!r} was trivial-skipped",
209
+ )
210
+
211
+
212
+ # --- End-to-end harness (mirrors test_recall_envelope_strip_telemetry.py) ---
213
+
214
+
215
+ class _RecordingClient:
216
+ def __init__(self, memories=None, directives=None):
217
+ self._memories = memories if memories is not None else []
218
+ self._directives = directives if directives is not None else []
219
+ self.queries = []
220
+
221
+ def list_directives(self, bank_id, active_only=True, timeout=2):
222
+ return {"items": list(self._directives)}
223
+
224
+ def recall(self, bank_id, query, **kwargs):
225
+ self.queries.append(query)
226
+ return {"results": list(self._memories)}
227
+
228
+
229
+ def _run_main_with(client, prompt, config_extra=None):
230
+ hook_input = {
231
+ "prompt": prompt,
232
+ "session_id": "test-session",
233
+ "transcript_path": "",
234
+ "cwd": "/tmp",
235
+ }
236
+ config = {
237
+ "autoRecall": True,
238
+ "bankId": "test-bank",
239
+ "recallMaxTokens": 1024,
240
+ "recallBudget": "mid",
241
+ "recallContextTurns": 1,
242
+ "recallMaxQueryChars": 800,
243
+ "recallPromptPreamble": "",
244
+ }
245
+ if config_extra:
246
+ config.update(config_extra)
247
+
248
+ stdout = io.StringIO()
249
+ stderr = io.StringIO()
250
+ with patch.object(recall, "load_config", return_value=config), patch.object(
251
+ recall, "get_api_url", return_value="http://localhost:18888"
252
+ ), patch.object(recall, "HindsightClient", return_value=client), patch.object(
253
+ recall, "ensure_bank_mission", return_value=None
254
+ ), patch.object(recall, "write_state", return_value=None), patch(
255
+ "sys.stdin", new=io.StringIO(json.dumps(hook_input))
256
+ ), patch("sys.stdout", new=stdout), patch("sys.stderr", new=stderr):
257
+ recall.main()
258
+
259
+ raw = stdout.getvalue()
260
+ if not raw.strip():
261
+ return None, raw
262
+ parsed = json.loads(raw)
263
+ return parsed["hookSpecificOutput"]["additionalContext"], raw
264
+
265
+
266
+ # A profile-only prompt: matches the profile regex but NOT the directive regex,
267
+ # so these end-to-end assertions isolate the profile nudge cleanly.
268
+ PROFILE_PROMPT = "my timezone is Australia/Melbourne"
269
+
270
+
271
+ class _LogTestBase(unittest.TestCase):
272
+ def setUp(self):
273
+ self._tmpdir = tempfile.mkdtemp(prefix="profile-nudge-test-")
274
+ self._prev = os.environ.get("CLAUDE_PLUGIN_DATA")
275
+ os.environ["CLAUDE_PLUGIN_DATA"] = self._tmpdir
276
+
277
+ def tearDown(self):
278
+ shutil.rmtree(self._tmpdir, ignore_errors=True)
279
+ if self._prev is None:
280
+ os.environ.pop("CLAUDE_PLUGIN_DATA", None)
281
+ else:
282
+ os.environ["CLAUDE_PLUGIN_DATA"] = self._prev
283
+
284
+ def _read_log(self):
285
+ path = os.path.join(self._tmpdir, "state", "recall_log.jsonl")
286
+ if not os.path.isfile(path):
287
+ return []
288
+ with open(path, encoding="utf-8") as f:
289
+ return [json.loads(line) for line in f if line.strip()]
290
+
291
+
292
+ class ProfileNudgeEndToEnd(_LogTestBase):
293
+ def test_advisory_reaches_context_and_row_records_it(self):
294
+ client = _RecordingClient(memories=[{"text": "m", "type": "fact",
295
+ "mentioned_at": "2026-01-01", "id": "m1"}])
296
+ ctx, _raw = _run_main_with(client, prompt=PROFILE_PROMPT)
297
+ # The profile advisory is injected into the turn context, tagged and
298
+ # own-bank-scoped.
299
+ self.assertIsNotNone(ctx)
300
+ self.assertIn("profile:ken", ctx)
301
+ self.assertIn("OWN bank", ctx)
302
+ # The recall_log row carries the firing-rate boolean.
303
+ entries = self._read_log()
304
+ self.assertEqual(len(entries), 1)
305
+ self.assertTrue(entries[0]["profile_nudge"])
306
+
307
+ def test_knob_off_suppresses_nudge_and_row_is_false(self):
308
+ client = _RecordingClient(memories=[{"text": "m", "type": "fact",
309
+ "mentioned_at": "2026-01-01", "id": "m1"}])
310
+ ctx, _raw = _run_main_with(
311
+ client, prompt=PROFILE_PROMPT,
312
+ config_extra={"profileCaptureNudge": False},
313
+ )
314
+ # Nudge suppressed: the advisory is absent from context (memories may
315
+ # still be injected, but never the profile block).
316
+ if ctx is not None:
317
+ self.assertNotIn("profile:ken", ctx)
318
+ self.assertNotIn("profile_capture_check", ctx)
319
+ entries = self._read_log()
320
+ self.assertEqual(len(entries), 1)
321
+ self.assertFalse(entries[0]["profile_nudge"])
322
+
323
+ def test_non_profile_prompt_does_not_fire(self):
324
+ client = _RecordingClient(memories=[{"text": "m", "type": "fact",
325
+ "mentioned_at": "2026-01-01", "id": "m1"}])
326
+ ctx, _raw = _run_main_with(client, prompt="please run the tests")
327
+ if ctx is not None:
328
+ self.assertNotIn("profile:ken", ctx)
329
+ entries = self._read_log()
330
+ self.assertEqual(len(entries), 1)
331
+ self.assertFalse(entries[0]["profile_nudge"])
332
+
333
+
334
+ if __name__ == "__main__":
335
+ unittest.main()
@@ -0,0 +1,143 @@
1
+ """M4 P0 — buffer + sentinel primitive, OUTCOME tests (RED test a).
2
+
3
+ The sentinel/poll race is the whole safety story for M4's prefetch producer
4
+ (carve §4 note 1, red-team-verified sound). This module is the FOUNDATION
5
+ every other M4 packet depends on: P-REC's join logic and P-PRE's producer
6
+ both build on the read-after-write guarantee proven here.
7
+
8
+ Contract under test:
9
+ (i) ``buffer.done`` is written strictly AFTER the buffer payload
10
+ (mtime/content ordering — proven via a monotonic token, not wall
11
+ clock, since two writes can land in the same clock tick).
12
+ (ii) ``read_if_fresh(last_consumed_token)`` returns ``(None, token)``
13
+ when the sentinel is absent, or not newer than ``last_consumed``.
14
+ (iii) A torn write (payload present, sentinel absent) reads as ``None``
15
+ — fail-closed, never a stale/malformed payload leaking through.
16
+ (iv) ``poll_for_sentinel`` is clock-bounded: never exceeds its cap, no
17
+ busy-spin (sleeps sum to <= cap, with generous tolerance).
18
+
19
+ Stdlib-only (``python3 -m unittest discover tests/``).
20
+ """
21
+
22
+ import os
23
+ import shutil
24
+ import sys
25
+ import tempfile
26
+ import time
27
+ import unittest
28
+ from unittest import mock
29
+
30
+ SCRIPTS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
31
+ if SCRIPTS_DIR not in sys.path:
32
+ sys.path.insert(0, SCRIPTS_DIR)
33
+
34
+ from lib import recall_buffer # noqa: E402
35
+
36
+
37
+ class RecallBufferBase(unittest.TestCase):
38
+ def setUp(self):
39
+ self.tmp = tempfile.mkdtemp(prefix="hs-m4-buffer-")
40
+ self.env = mock.patch.dict(
41
+ os.environ, {"HINDSIGHT_PREFETCH_BUFFER_DIR": self.tmp}, clear=False
42
+ )
43
+ self.env.start()
44
+
45
+ def tearDown(self):
46
+ self.env.stop()
47
+ shutil.rmtree(self.tmp, ignore_errors=True)
48
+
49
+
50
+ class TestSentinelOrdering(RecallBufferBase):
51
+ def test_sentinel_written_strictly_after_payload(self):
52
+ session = "sess-order"
53
+ write_order = []
54
+ real_replace = os.replace
55
+
56
+ def _spy_replace(src, dst):
57
+ write_order.append(os.path.basename(dst))
58
+ return real_replace(src, dst)
59
+
60
+ with mock.patch("os.replace", side_effect=_spy_replace):
61
+ recall_buffer.write_buffer(session, "<hindsight_memories>x</hindsight_memories>", {"k": "v"})
62
+ recall_buffer.write_sentinel(session)
63
+
64
+ buffer_writes = [w for w in write_order if "buffer.json" in w]
65
+ sentinel_writes = [w for w in write_order if "buffer.done" in w]
66
+ self.assertEqual(len(buffer_writes), 1)
67
+ self.assertEqual(len(sentinel_writes), 1)
68
+ self.assertLess(
69
+ write_order.index(buffer_writes[0]),
70
+ write_order.index(sentinel_writes[0]),
71
+ "sentinel must be written strictly AFTER the buffer payload",
72
+ )
73
+
74
+ def test_read_if_fresh_returns_none_when_sentinel_absent(self):
75
+ session = "sess-nosentinel"
76
+ recall_buffer.write_buffer(session, "content", {})
77
+ ctx, token = recall_buffer.read_if_fresh(session, last_consumed_token=None)
78
+ self.assertIsNone(ctx)
79
+
80
+ def test_read_if_fresh_returns_none_when_sentinel_not_newer(self):
81
+ session = "sess-stale"
82
+ recall_buffer.write_buffer(session, "content", {})
83
+ token = recall_buffer.write_sentinel(session)
84
+ # Already consumed this exact token -> not fresh.
85
+ ctx, new_token = recall_buffer.read_if_fresh(session, last_consumed_token=token)
86
+ self.assertIsNone(ctx)
87
+ self.assertEqual(new_token, token)
88
+
89
+ def test_read_if_fresh_returns_payload_when_sentinel_is_newer(self):
90
+ session = "sess-fresh"
91
+ recall_buffer.write_buffer(session, "<hindsight_memories>fact</hindsight_memories>", {"telemetry": 1})
92
+ token = recall_buffer.write_sentinel(session)
93
+ ctx, new_token = recall_buffer.read_if_fresh(session, last_consumed_token=None)
94
+ self.assertIsNotNone(ctx)
95
+ self.assertIn("fact", ctx["context"])
96
+ self.assertEqual(ctx["telemetry"], {"telemetry": 1})
97
+ self.assertEqual(new_token, token)
98
+
99
+ def test_torn_write_payload_present_sentinel_absent_reads_as_none(self):
100
+ # Simulate a crash between write_buffer() and write_sentinel(): the
101
+ # payload landed, the sentinel never did. Must fail-closed to None,
102
+ # NEVER read the torn/possibly-incomplete payload as fresh.
103
+ session = "sess-torn"
104
+ recall_buffer.write_buffer(session, "<hindsight_memories>partial</hindsight_memories>", {})
105
+ # No write_sentinel() call — this IS the torn state.
106
+ ctx, token = recall_buffer.read_if_fresh(session, last_consumed_token=None)
107
+ self.assertIsNone(ctx, "torn write (payload without sentinel) must read as None, fail-closed")
108
+
109
+
110
+ class TestPollCap(RecallBufferBase):
111
+ def test_poll_returns_true_when_sentinel_appears(self):
112
+ session = "sess-poll-hit"
113
+ recall_buffer.write_buffer(session, "content", {})
114
+ recall_buffer.write_sentinel(session)
115
+ found = recall_buffer.poll_for_sentinel(session, last_consumed_token=None, cap_ms=500)
116
+ self.assertTrue(found)
117
+
118
+ def test_poll_never_exceeds_cap(self):
119
+ # No sentinel ever appears — poll must give up within cap_ms
120
+ # (generous tolerance for test-host scheduling jitter).
121
+ session = "sess-poll-miss"
122
+ cap_ms = 200
123
+ start = time.monotonic()
124
+ found = recall_buffer.poll_for_sentinel(session, last_consumed_token=None, cap_ms=cap_ms)
125
+ elapsed_ms = (time.monotonic() - start) * 1000
126
+ self.assertFalse(found)
127
+ self.assertLess(elapsed_ms, cap_ms + 150, "poll exceeded its cap by more than tolerance")
128
+
129
+ def test_poll_does_not_busy_spin(self):
130
+ # A busy-spin would burn far more than a handful of read attempts in
131
+ # the cap window; assert sleep is actually happening by checking the
132
+ # elapsed time is close to (not far below) the cap when nothing
133
+ # ever shows up.
134
+ session = "sess-poll-nospin"
135
+ cap_ms = 150
136
+ start = time.monotonic()
137
+ recall_buffer.poll_for_sentinel(session, last_consumed_token=None, cap_ms=cap_ms)
138
+ elapsed_ms = (time.monotonic() - start) * 1000
139
+ self.assertGreaterEqual(elapsed_ms, cap_ms * 0.5, "poll returned suspiciously fast — looks like it skipped sleeping")
140
+
141
+
142
+ if __name__ == "__main__":
143
+ unittest.main()
@@ -0,0 +1,193 @@
1
+ """M4 P-REC test (b) + red-team Fix B / Fix C — prefetch buffer join.
2
+
3
+ Asserts on RENDERED transport (`additionalContext`), never on an internal
4
+ flag alone, per the red-team's explicit strengthening. Covers:
5
+
6
+ * Fix C (kill switch): `memoryPrefetchEnabled` OFF (the default) leaves
7
+ the mechanism completely inert — synchronous recall runs exactly as
8
+ before, `lib.recall_buffer` is never touched.
9
+ * Fresh-hit: a producer-written buffer is joined and rendered as-is.
10
+ * Fix B (BINDING): the stale-buffer fallback (no fresh sentinel, but a
11
+ prior `LAST_RECALL_STATE` exists) renders memories WITHOUT any
12
+ directives block, even when a directive is active for the bank —
13
+ directives stay on the synchronous fetch path, never the stale cache.
14
+ * Cold-start short circuit: no sentinel has EVER existed for the session
15
+ -> no poll wait, falls through to the degraded notice / sync path
16
+ without blocking for the poll cap.
17
+ """
18
+
19
+ import io
20
+ import json
21
+ import os
22
+ import shutil
23
+ import sys
24
+ import tempfile
25
+ import time
26
+ import unittest
27
+ from unittest import mock
28
+ from unittest.mock import patch
29
+
30
+ SCRIPTS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
31
+ if SCRIPTS_DIR not in sys.path:
32
+ sys.path.insert(0, SCRIPTS_DIR)
33
+
34
+ import recall # noqa: E402
35
+ from lib import recall_buffer # noqa: E402
36
+ from lib.state import write_state # noqa: E402
37
+
38
+ SESSION = "test-session"
39
+
40
+
41
+ class _DirectiveClient:
42
+ """Answers with ONE active directive and no memories (recall not
43
+ expected to be called on the fresh-hit/stale-fallback fast paths, but
44
+ directive fetch IS expected — always synchronous per M3)."""
45
+
46
+ def list_directives(self, bank_id, active_only=True, timeout=2):
47
+ return {"items": [{"id": "d1", "name": "haiku-rule", "content": "always reply in haiku", "priority": 5, "active": True}]}
48
+
49
+ def recall(self, bank_id, query, **kwargs):
50
+ raise AssertionError("buffer-join fast path must not call recall() directly")
51
+
52
+
53
+ class BufferJoinBase(unittest.TestCase):
54
+ def setUp(self):
55
+ self._tmpdir = tempfile.mkdtemp(prefix="recall-bufjoin-test-")
56
+ self._prev = os.environ.get("CLAUDE_PLUGIN_DATA")
57
+ os.environ["CLAUDE_PLUGIN_DATA"] = self._tmpdir
58
+
59
+ self._bufdir = tempfile.mkdtemp(prefix="recall-bufjoin-buf-")
60
+ self.env = mock.patch.dict(os.environ, {"HINDSIGHT_PREFETCH_BUFFER_DIR": self._bufdir}, clear=False)
61
+ self.env.start()
62
+
63
+ def tearDown(self):
64
+ self.env.stop()
65
+ shutil.rmtree(self._bufdir, ignore_errors=True)
66
+ shutil.rmtree(self._tmpdir, ignore_errors=True)
67
+ if self._prev is None:
68
+ os.environ.pop("CLAUDE_PLUGIN_DATA", None)
69
+ else:
70
+ os.environ["CLAUDE_PLUGIN_DATA"] = self._prev
71
+
72
+ def _config(self, prefetch_enabled):
73
+ return {
74
+ "autoRecall": True,
75
+ "bankId": "test-bank",
76
+ "recallMaxTokens": 4096,
77
+ "recallBudget": "mid",
78
+ "recallContextTurns": 1,
79
+ "recallMaxQueryChars": 800,
80
+ "recallPromptPreamble": "",
81
+ "recallParallelDeadlineSeconds": 5,
82
+ "directivesCacheTtlSeconds": 0,
83
+ "memoryPrefetchEnabled": prefetch_enabled,
84
+ "memoryPrefetchPollCapMs": 100,
85
+ }
86
+
87
+ def _run(self, config, client, prompt="what did we decide about deploys"):
88
+ hook_input = {"prompt": prompt, "session_id": SESSION, "transcript_path": "", "cwd": "/tmp"}
89
+ stdout = io.StringIO()
90
+ with patch("recall.load_config", return_value=config), \
91
+ patch("recall.get_api_url", return_value="http://fake"), \
92
+ patch("recall.HindsightClient", return_value=client), \
93
+ patch("recall.ensure_bank_mission"), \
94
+ patch("sys.stdin", io.StringIO(json.dumps(hook_input))), \
95
+ patch("sys.stdout", stdout):
96
+ recall.main()
97
+ return stdout.getvalue()
98
+
99
+
100
+ class FreshHitTests(BufferJoinBase):
101
+ def test_fresh_buffer_hit_is_rendered_with_directives_layered_on(self):
102
+ recall_buffer.write_buffer(SESSION, "- a prefetched memory", {})
103
+ recall_buffer.write_sentinel(SESSION)
104
+
105
+ out = self._run(self._config(prefetch_enabled=True), _DirectiveClient())
106
+ self.assertTrue(out)
107
+ ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
108
+ self.assertIn("a prefetched memory", ctx)
109
+ self.assertIn("always reply in haiku", ctx)
110
+
111
+
112
+ class StaleFallbackFixBTests(BufferJoinBase):
113
+ def test_stale_fallback_renders_memories_without_any_directives_block(self):
114
+ # No fresh sentinel this turn (buffer is cold), but a prior turn's
115
+ # cache DOES exist and DOES include memories_context (directive-free
116
+ # per Fix B) plus a bundled directive in `context` that must NOT
117
+ # leak through the stale path.
118
+ write_state(
119
+ recall.LAST_RECALL_STATE,
120
+ {
121
+ "context": "## Active Directives\n- always reply in haiku\n\n- a stale cached memory",
122
+ "memories_context": "- a stale cached memory",
123
+ "saved_at": "2026-01-01T00:00:00Z",
124
+ "bank_id": "test-bank",
125
+ "result_count": 1,
126
+ "directive_count": 1,
127
+ },
128
+ )
129
+ out = self._run(self._config(prefetch_enabled=True), _DirectiveClient())
130
+ self.assertTrue(out)
131
+ ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
132
+ self.assertIn("a stale cached memory", ctx)
133
+ self.assertIn("stale", ctx.lower())
134
+ # Fix B, BINDING: the fresh directive fetch (synchronous, M3 rule)
135
+ # IS allowed to appear — what must NEVER appear is the STALE
136
+ # directives text sourced from the cached `context` field. Prove
137
+ # this by using a client whose directive text differs from what a
138
+ # stale-context leak would have produced, and confirming there is
139
+ # exactly one occurrence (the fresh fetch), not a duplicate/leaked
140
+ # second copy from the cache.
141
+ self.assertEqual(ctx.count("always reply in haiku"), 1)
142
+
143
+ def test_stale_fallback_directive_free_field_never_leaks_bare_context_directives(self):
144
+ # A cache row is deliberately malformed/legacy-shaped (missing the
145
+ # M4 `memories_context` field entirely, i.e. pre-M4 cache on disk).
146
+ # The fallback must degrade to "nothing stale to show", NEVER fall
147
+ # back to the directive-contaminated `context` field.
148
+ write_state(
149
+ recall.LAST_RECALL_STATE,
150
+ {
151
+ "context": "## Active Directives\n- always reply in haiku\n\n- a stale cached memory",
152
+ "saved_at": "2026-01-01T00:00:00Z",
153
+ "bank_id": "test-bank",
154
+ "result_count": 1,
155
+ "directive_count": 1,
156
+ },
157
+ )
158
+ client = _DirectiveClient()
159
+ out = self._run(self._config(prefetch_enabled=True), client)
160
+ self.assertTrue(out)
161
+ ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
162
+ self.assertNotIn("a stale cached memory", ctx, "must not fall back to the directive-contaminated context field")
163
+
164
+
165
+ class ColdStartShortCircuitTests(BufferJoinBase):
166
+ def test_no_sentinel_ever_written_skips_the_poll_wait(self):
167
+ config = self._config(prefetch_enabled=True)
168
+ config["memoryPrefetchPollCapMs"] = 5000 # would dominate the test if NOT short-circuited
169
+ start = time.monotonic()
170
+ out = self._run(config, _DirectiveClient())
171
+ elapsed_ms = (time.monotonic() - start) * 1000
172
+ self.assertTrue(out)
173
+ self.assertLess(elapsed_ms, 1000, "cold-start (no sentinel ever) must not wait out the poll cap")
174
+
175
+
176
+ class KillSwitchOffTests(BufferJoinBase):
177
+ def test_flag_off_never_touches_the_buffer_module(self):
178
+ recall_buffer.write_buffer(SESSION, "- a prefetched memory", {})
179
+ recall_buffer.write_sentinel(SESSION)
180
+
181
+ with patch("recall.recall_buffer.read_if_fresh", side_effect=AssertionError("must not be called when flag is off")):
182
+ client = _DirectiveClient()
183
+ client.recall = lambda *a, **kw: {"results": []} # sync path IS allowed to call recall
184
+ out = self._run(self._config(prefetch_enabled=False), client)
185
+ self.assertTrue(out)
186
+ ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
187
+ # Byte-identical-behaviour proof: the buffer's content must NOT
188
+ # appear (it was never consulted) even though it exists on disk.
189
+ self.assertNotIn("a prefetched memory", ctx)
190
+
191
+
192
+ if __name__ == "__main__":
193
+ unittest.main()