switchroom 0.21.14 → 0.21.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/rules-sentinel-hook.sh +101 -0
- package/dist/agent-scheduler/index.js +7 -2
- package/dist/auth-broker/index.js +7 -2
- package/dist/cli/notion-write-pretool.mjs +7 -2
- package/dist/cli/switchroom.js +2601 -1009
- package/dist/host-control/main.js +8 -3
- package/dist/vault/approvals/kernel-server.js +7 -2
- package/dist/vault/broker/server.js +7 -2
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +9 -0
- package/profiles/_shared/delegation-golden-rule.md.hbs +2 -0
- package/skills/mental-model-curator/SKILL.md +187 -56
- package/telegram-plugin/dist/gateway/gateway.js +11 -6
- package/vendor/hindsight-memory/hooks/hooks.json +10 -0
- package/vendor/hindsight-memory/scripts/lib/client.py +14 -0
- package/vendor/hindsight-memory/scripts/lib/config.py +22 -0
- package/vendor/hindsight-memory/scripts/lib/directives.py +45 -7
- package/vendor/hindsight-memory/scripts/lib/recall_buffer.py +236 -0
- package/vendor/hindsight-memory/scripts/lib/watermark.py +27 -0
- package/vendor/hindsight-memory/scripts/prefetch.py +156 -0
- package/vendor/hindsight-memory/scripts/recall.py +520 -5
- package/vendor/hindsight-memory/scripts/reconcile_tail.py +4 -12
- package/vendor/hindsight-memory/scripts/retain.py +167 -28
- package/vendor/hindsight-memory/scripts/tests/test_config_retain_tool_calls_env.py +98 -0
- package/vendor/hindsight-memory/scripts/tests/test_directives.py +52 -0
- package/vendor/hindsight-memory/scripts/tests/test_incremental_sweep.py +293 -0
- package/vendor/hindsight-memory/scripts/tests/test_prefetch_pipeline.py +247 -0
- package/vendor/hindsight-memory/scripts/tests/test_profile_capture_nudge.py +335 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_buffer.py +143 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_buffer_join.py +193 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_cap_truncation.py +133 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_junk_gate.py +168 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_no_score_floor.py +124 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_query_timestamp.py +376 -0
- package/vendor/hindsight-memory/scripts/tests/test_retain_delta.py +304 -0
- package/vendor/hindsight-memory/scripts/tests/test_retain_stop_hook_prefetch_gate.py +109 -0
|
@@ -0,0 +1,335 @@
|
|
|
1
|
+
"""RFC phase4 P3 — unit + end-to-end tests for the operator-profile capture nudge.
|
|
2
|
+
|
|
3
|
+
Ken's first stated want is "save memories about him". Auto-retain stores
|
|
4
|
+
transcript facts, but nothing gives a DETERMINISTIC signal that a durable
|
|
5
|
+
*profile fact* about the operator himself just went by — so profile capture is
|
|
6
|
+
left to model discretion (the same per-agent lottery Stage A measured for
|
|
7
|
+
directives). P3 mirrors the shipped directive-capture nudge (recall.py #2848):
|
|
8
|
+
a POSITIVE regex detects a first-person durable self-statement, a NEGATIVE
|
|
9
|
+
regex scrubs the two false-positive shapes (questions, attributions to others)
|
|
10
|
+
BEFORE the positive match, and on a hit the UserPromptSubmit hook appends a
|
|
11
|
+
terse advisory telling the model to persist it with an explicit retain tagged
|
|
12
|
+
`profile:ken` into the agent's OWN bank. Pure regex — no model callsite.
|
|
13
|
+
|
|
14
|
+
These tests pin (all OUTCOME assertions):
|
|
15
|
+
* True positives — the RFC's listed profile shapes fire.
|
|
16
|
+
* True negatives — questions and third-/second-party attributions do NOT
|
|
17
|
+
fire (the negative-lookaround guard scrubs them first).
|
|
18
|
+
* The advisory carries the `profile:ken` tag instruction and targets the
|
|
19
|
+
agent's OWN bank (not a shared / cross-agent person bank).
|
|
20
|
+
* End-to-end through recall.main(): the advisory reaches the emitted
|
|
21
|
+
additionalContext, the recall_log row carries `profile_nudge: true`, and
|
|
22
|
+
the config knob OFF (`profileCaptureNudge: false`) suppresses both.
|
|
23
|
+
|
|
24
|
+
Stdlib-only; runs under `python3 -m unittest discover tests/`. The end-to-end
|
|
25
|
+
harness mirrors test_recall_envelope_strip_telemetry.py.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
import io
|
|
29
|
+
import json
|
|
30
|
+
import os
|
|
31
|
+
import shutil
|
|
32
|
+
import sys
|
|
33
|
+
import tempfile
|
|
34
|
+
import unittest
|
|
35
|
+
from unittest.mock import patch
|
|
36
|
+
|
|
37
|
+
SCRIPTS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
|
|
38
|
+
if SCRIPTS_DIR not in sys.path:
|
|
39
|
+
sys.path.insert(0, SCRIPTS_DIR)
|
|
40
|
+
|
|
41
|
+
import recall # noqa: E402
|
|
42
|
+
from recall import ( # noqa: E402
|
|
43
|
+
_PROFILE_CAPTURE_NUDGE,
|
|
44
|
+
_combine_context,
|
|
45
|
+
_is_trivial_stateless,
|
|
46
|
+
looks_like_profile_statement,
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
# First-person durable self-statements that MUST fire the nudge. Drawn from the
|
|
51
|
+
# RFC P3 shapes ("I prefer …", "my … is …", "I always …", "remind me that I …")
|
|
52
|
+
# plus the conservative identity/situation set.
|
|
53
|
+
PROFILE_STATEMENTS = [
|
|
54
|
+
# --- stated preferences ---
|
|
55
|
+
"I prefer dark roast coffee",
|
|
56
|
+
"I'd prefer British spelling in my docs",
|
|
57
|
+
"my preference is tabs over spaces",
|
|
58
|
+
# --- durable self-facts: "my <ATTRIBUTE> is/are …" (tight allow-list) ---
|
|
59
|
+
"my timezone is Australia/Melbourne",
|
|
60
|
+
"my email is ken@example.com",
|
|
61
|
+
"my sister is Lisa",
|
|
62
|
+
"my kids are at school during the day",
|
|
63
|
+
"my name's Ken", # contraction form
|
|
64
|
+
# --- identity / situation ---
|
|
65
|
+
"I live in Melbourne",
|
|
66
|
+
"I work at Anthropic",
|
|
67
|
+
"I'm allergic to peanuts",
|
|
68
|
+
"I'm based in Australia",
|
|
69
|
+
# --- durable identity: diet / abstention / "I'm a <noun>" ---
|
|
70
|
+
"I'm a vegetarian",
|
|
71
|
+
"I don't eat meat",
|
|
72
|
+
"call me Ken",
|
|
73
|
+
# --- tastes (durable like/dislike framing) ---
|
|
74
|
+
"I hate em-dashes",
|
|
75
|
+
# --- durable habits ---
|
|
76
|
+
"I always take my coffee black",
|
|
77
|
+
"I usually work late on Thursdays",
|
|
78
|
+
"I never eat red meat",
|
|
79
|
+
# --- explicit memory framing about the operator himself ---
|
|
80
|
+
"remind me that I have a standing 9am standup",
|
|
81
|
+
"remember that I hate em-dashes",
|
|
82
|
+
]
|
|
83
|
+
|
|
84
|
+
# Questions and attributions that MUST NOT fire — the negative-lookaround guard
|
|
85
|
+
# scrubs these BEFORE the positive match. A false positive here is nudge noise.
|
|
86
|
+
NON_PROFILE = [
|
|
87
|
+
# --- questions about the operator (not statements of a durable fact) ---
|
|
88
|
+
"do I prefer tea or coffee?",
|
|
89
|
+
"what's my timezone?",
|
|
90
|
+
"where is my email address stored?",
|
|
91
|
+
"how do I fix this bug",
|
|
92
|
+
"should I always run the tests first?",
|
|
93
|
+
"what do I usually do here",
|
|
94
|
+
"remind me what my calendar looks like",
|
|
95
|
+
# --- attributions to a third / second party ---
|
|
96
|
+
"you said I prefer tea",
|
|
97
|
+
"she claims my code is broken",
|
|
98
|
+
"he thinks I always overcomplicate things",
|
|
99
|
+
"they told me my access was revoked",
|
|
100
|
+
# --- neither: no first-person durable self-fact ---
|
|
101
|
+
"the project timezone is UTC",
|
|
102
|
+
"please run the tests",
|
|
103
|
+
"what time is it",
|
|
104
|
+
# --- discourse-marker "my <X> is" — not a durable profile fact. The
|
|
105
|
+
# positive arm uses a TIGHT identity allow-list, so a free noun never
|
|
106
|
+
# reaches the matcher; these confirm that. ---
|
|
107
|
+
"my guess is the cache is stale",
|
|
108
|
+
"my point is that we should ship it",
|
|
109
|
+
"my concern is the timeout",
|
|
110
|
+
# --- transient dev state "my <transient> is/are …". These are the exact
|
|
111
|
+
# over-fires the free-`\w+` arm produced; the allow-list must NOT fire
|
|
112
|
+
# on them (RFC favour-false-negatives constraint on this agent). ---
|
|
113
|
+
"my container is down",
|
|
114
|
+
"my build is failing",
|
|
115
|
+
"my code is broken",
|
|
116
|
+
"my PR is ready",
|
|
117
|
+
"my worktree is dirty",
|
|
118
|
+
"my server is down",
|
|
119
|
+
"my deploy is stuck",
|
|
120
|
+
"my tests are green",
|
|
121
|
+
"my branch is merged",
|
|
122
|
+
# --- pleasantry embedding a bare always/never after "I" ---
|
|
123
|
+
"I always appreciate your help",
|
|
124
|
+
"I never enjoy waiting, but thanks",
|
|
125
|
+
# --- "I'm a <hedge>" is a transient mood, not an "I'm a <noun>" identity ---
|
|
126
|
+
"I'm a bit tired",
|
|
127
|
+
"I'm a little confused about the config",
|
|
128
|
+
"I'm a big fan of shipping fast",
|
|
129
|
+
# --- "call me <phrasing>" as a request, not a name form ---
|
|
130
|
+
"call me back later",
|
|
131
|
+
"call me when the build finishes",
|
|
132
|
+
# --- the <channel …> envelope wrapper on its own must never trigger ---
|
|
133
|
+
'<channel user="ken" chat_id="123">',
|
|
134
|
+
]
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class TestProfileDetection(unittest.TestCase):
|
|
138
|
+
def test_profile_statements_fire_the_nudge(self):
|
|
139
|
+
for p in PROFILE_STATEMENTS:
|
|
140
|
+
with self.subTest(prompt=p):
|
|
141
|
+
self.assertTrue(
|
|
142
|
+
looks_like_profile_statement(p),
|
|
143
|
+
f"expected profile detection for {p!r}",
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
def test_questions_and_attributions_do_not_fire(self):
|
|
147
|
+
for p in NON_PROFILE:
|
|
148
|
+
with self.subTest(prompt=p):
|
|
149
|
+
self.assertFalse(
|
|
150
|
+
looks_like_profile_statement(p),
|
|
151
|
+
f"FALSE POSITIVE: profile nudge would fire for {p!r}",
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
def test_empty_and_non_string_are_false(self):
|
|
155
|
+
for bad in ("", " ", None, 123, [], {}):
|
|
156
|
+
with self.subTest(value=bad):
|
|
157
|
+
self.assertFalse(looks_like_profile_statement(bad))
|
|
158
|
+
|
|
159
|
+
def test_attribution_scrub_does_not_mask_a_real_self_fact(self):
|
|
160
|
+
# A message that opens with an attributed clause AND then states the
|
|
161
|
+
# operator's own durable fact must still fire — the negative guard
|
|
162
|
+
# scrubs only the attributed span (through end-of-sentence).
|
|
163
|
+
self.assertTrue(
|
|
164
|
+
looks_like_profile_statement(
|
|
165
|
+
"she thinks I'm wrong. my timezone is Melbourne"
|
|
166
|
+
)
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
def test_attribution_scrub_stops_at_a_comma(self):
|
|
170
|
+
# The attributed span must stop at a comma, not run to end-of-sentence:
|
|
171
|
+
# a real self-fact trailing the attributed clause in the SAME sentence
|
|
172
|
+
# must still reach the positive matcher (MINOR: greedy [^.?!]* → [^.?!,]*).
|
|
173
|
+
self.assertTrue(
|
|
174
|
+
looks_like_profile_statement(
|
|
175
|
+
"she said the deploy failed, my timezone is Melbourne"
|
|
176
|
+
)
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
class TestProfileNudgeString(unittest.TestCase):
|
|
181
|
+
def test_nudge_carries_the_profile_ken_tag_instruction(self):
|
|
182
|
+
self.assertIn("profile:ken", _PROFILE_CAPTURE_NUDGE)
|
|
183
|
+
self.assertIn("retain", _PROFILE_CAPTURE_NUDGE)
|
|
184
|
+
self.assertIn("profile_capture_check", _PROFILE_CAPTURE_NUDGE)
|
|
185
|
+
|
|
186
|
+
def test_nudge_targets_the_agents_own_bank_not_a_shared_one(self):
|
|
187
|
+
# Constraint 2 forbids a cross-agent person bank — the advisory must
|
|
188
|
+
# route to the agent's OWN bank and say so explicitly.
|
|
189
|
+
self.assertIn("OWN bank", _PROFILE_CAPTURE_NUDGE)
|
|
190
|
+
self.assertIn("shared", _PROFILE_CAPTURE_NUDGE)
|
|
191
|
+
|
|
192
|
+
def test_combine_appends_nudge_after_recall_block(self):
|
|
193
|
+
base = "<hindsight_memories>\n…\n</hindsight_memories>"
|
|
194
|
+
out = _combine_context(base, _PROFILE_CAPTURE_NUDGE)
|
|
195
|
+
self.assertTrue(out.startswith(base))
|
|
196
|
+
self.assertTrue(out.endswith(_PROFILE_CAPTURE_NUDGE))
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
class TestTrivialSkipUnaffected(unittest.TestCase):
|
|
200
|
+
"""A profile statement carries personal/stateful signal and must never be
|
|
201
|
+
trivial-stateless-skipped; greetings still are."""
|
|
202
|
+
|
|
203
|
+
def test_profile_statements_are_never_trivial_skipped(self):
|
|
204
|
+
for p in PROFILE_STATEMENTS:
|
|
205
|
+
with self.subTest(prompt=p):
|
|
206
|
+
self.assertFalse(
|
|
207
|
+
_is_trivial_stateless("", p),
|
|
208
|
+
f"profile statement {p!r} was trivial-skipped",
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
# --- End-to-end harness (mirrors test_recall_envelope_strip_telemetry.py) ---
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
class _RecordingClient:
|
|
216
|
+
def __init__(self, memories=None, directives=None):
|
|
217
|
+
self._memories = memories if memories is not None else []
|
|
218
|
+
self._directives = directives if directives is not None else []
|
|
219
|
+
self.queries = []
|
|
220
|
+
|
|
221
|
+
def list_directives(self, bank_id, active_only=True, timeout=2):
|
|
222
|
+
return {"items": list(self._directives)}
|
|
223
|
+
|
|
224
|
+
def recall(self, bank_id, query, **kwargs):
|
|
225
|
+
self.queries.append(query)
|
|
226
|
+
return {"results": list(self._memories)}
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _run_main_with(client, prompt, config_extra=None):
|
|
230
|
+
hook_input = {
|
|
231
|
+
"prompt": prompt,
|
|
232
|
+
"session_id": "test-session",
|
|
233
|
+
"transcript_path": "",
|
|
234
|
+
"cwd": "/tmp",
|
|
235
|
+
}
|
|
236
|
+
config = {
|
|
237
|
+
"autoRecall": True,
|
|
238
|
+
"bankId": "test-bank",
|
|
239
|
+
"recallMaxTokens": 1024,
|
|
240
|
+
"recallBudget": "mid",
|
|
241
|
+
"recallContextTurns": 1,
|
|
242
|
+
"recallMaxQueryChars": 800,
|
|
243
|
+
"recallPromptPreamble": "",
|
|
244
|
+
}
|
|
245
|
+
if config_extra:
|
|
246
|
+
config.update(config_extra)
|
|
247
|
+
|
|
248
|
+
stdout = io.StringIO()
|
|
249
|
+
stderr = io.StringIO()
|
|
250
|
+
with patch.object(recall, "load_config", return_value=config), patch.object(
|
|
251
|
+
recall, "get_api_url", return_value="http://localhost:18888"
|
|
252
|
+
), patch.object(recall, "HindsightClient", return_value=client), patch.object(
|
|
253
|
+
recall, "ensure_bank_mission", return_value=None
|
|
254
|
+
), patch.object(recall, "write_state", return_value=None), patch(
|
|
255
|
+
"sys.stdin", new=io.StringIO(json.dumps(hook_input))
|
|
256
|
+
), patch("sys.stdout", new=stdout), patch("sys.stderr", new=stderr):
|
|
257
|
+
recall.main()
|
|
258
|
+
|
|
259
|
+
raw = stdout.getvalue()
|
|
260
|
+
if not raw.strip():
|
|
261
|
+
return None, raw
|
|
262
|
+
parsed = json.loads(raw)
|
|
263
|
+
return parsed["hookSpecificOutput"]["additionalContext"], raw
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
# A profile-only prompt: matches the profile regex but NOT the directive regex,
|
|
267
|
+
# so these end-to-end assertions isolate the profile nudge cleanly.
|
|
268
|
+
PROFILE_PROMPT = "my timezone is Australia/Melbourne"
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
class _LogTestBase(unittest.TestCase):
|
|
272
|
+
def setUp(self):
|
|
273
|
+
self._tmpdir = tempfile.mkdtemp(prefix="profile-nudge-test-")
|
|
274
|
+
self._prev = os.environ.get("CLAUDE_PLUGIN_DATA")
|
|
275
|
+
os.environ["CLAUDE_PLUGIN_DATA"] = self._tmpdir
|
|
276
|
+
|
|
277
|
+
def tearDown(self):
|
|
278
|
+
shutil.rmtree(self._tmpdir, ignore_errors=True)
|
|
279
|
+
if self._prev is None:
|
|
280
|
+
os.environ.pop("CLAUDE_PLUGIN_DATA", None)
|
|
281
|
+
else:
|
|
282
|
+
os.environ["CLAUDE_PLUGIN_DATA"] = self._prev
|
|
283
|
+
|
|
284
|
+
def _read_log(self):
|
|
285
|
+
path = os.path.join(self._tmpdir, "state", "recall_log.jsonl")
|
|
286
|
+
if not os.path.isfile(path):
|
|
287
|
+
return []
|
|
288
|
+
with open(path, encoding="utf-8") as f:
|
|
289
|
+
return [json.loads(line) for line in f if line.strip()]
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
class ProfileNudgeEndToEnd(_LogTestBase):
|
|
293
|
+
def test_advisory_reaches_context_and_row_records_it(self):
|
|
294
|
+
client = _RecordingClient(memories=[{"text": "m", "type": "fact",
|
|
295
|
+
"mentioned_at": "2026-01-01", "id": "m1"}])
|
|
296
|
+
ctx, _raw = _run_main_with(client, prompt=PROFILE_PROMPT)
|
|
297
|
+
# The profile advisory is injected into the turn context, tagged and
|
|
298
|
+
# own-bank-scoped.
|
|
299
|
+
self.assertIsNotNone(ctx)
|
|
300
|
+
self.assertIn("profile:ken", ctx)
|
|
301
|
+
self.assertIn("OWN bank", ctx)
|
|
302
|
+
# The recall_log row carries the firing-rate boolean.
|
|
303
|
+
entries = self._read_log()
|
|
304
|
+
self.assertEqual(len(entries), 1)
|
|
305
|
+
self.assertTrue(entries[0]["profile_nudge"])
|
|
306
|
+
|
|
307
|
+
def test_knob_off_suppresses_nudge_and_row_is_false(self):
|
|
308
|
+
client = _RecordingClient(memories=[{"text": "m", "type": "fact",
|
|
309
|
+
"mentioned_at": "2026-01-01", "id": "m1"}])
|
|
310
|
+
ctx, _raw = _run_main_with(
|
|
311
|
+
client, prompt=PROFILE_PROMPT,
|
|
312
|
+
config_extra={"profileCaptureNudge": False},
|
|
313
|
+
)
|
|
314
|
+
# Nudge suppressed: the advisory is absent from context (memories may
|
|
315
|
+
# still be injected, but never the profile block).
|
|
316
|
+
if ctx is not None:
|
|
317
|
+
self.assertNotIn("profile:ken", ctx)
|
|
318
|
+
self.assertNotIn("profile_capture_check", ctx)
|
|
319
|
+
entries = self._read_log()
|
|
320
|
+
self.assertEqual(len(entries), 1)
|
|
321
|
+
self.assertFalse(entries[0]["profile_nudge"])
|
|
322
|
+
|
|
323
|
+
def test_non_profile_prompt_does_not_fire(self):
|
|
324
|
+
client = _RecordingClient(memories=[{"text": "m", "type": "fact",
|
|
325
|
+
"mentioned_at": "2026-01-01", "id": "m1"}])
|
|
326
|
+
ctx, _raw = _run_main_with(client, prompt="please run the tests")
|
|
327
|
+
if ctx is not None:
|
|
328
|
+
self.assertNotIn("profile:ken", ctx)
|
|
329
|
+
entries = self._read_log()
|
|
330
|
+
self.assertEqual(len(entries), 1)
|
|
331
|
+
self.assertFalse(entries[0]["profile_nudge"])
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
if __name__ == "__main__":
|
|
335
|
+
unittest.main()
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""M4 P0 — buffer + sentinel primitive, OUTCOME tests (RED test a).
|
|
2
|
+
|
|
3
|
+
The sentinel/poll race is the whole safety story for M4's prefetch producer
|
|
4
|
+
(carve §4 note 1, red-team-verified sound). This module is the FOUNDATION
|
|
5
|
+
every other M4 packet depends on: P-REC's join logic and P-PRE's producer
|
|
6
|
+
both build on the read-after-write guarantee proven here.
|
|
7
|
+
|
|
8
|
+
Contract under test:
|
|
9
|
+
(i) ``buffer.done`` is written strictly AFTER the buffer payload
|
|
10
|
+
(mtime/content ordering — proven via a monotonic token, not wall
|
|
11
|
+
clock, since two writes can land in the same clock tick).
|
|
12
|
+
(ii) ``read_if_fresh(last_consumed_token)`` returns ``(None, token)``
|
|
13
|
+
when the sentinel is absent, or not newer than ``last_consumed``.
|
|
14
|
+
(iii) A torn write (payload present, sentinel absent) reads as ``None``
|
|
15
|
+
— fail-closed, never a stale/malformed payload leaking through.
|
|
16
|
+
(iv) ``poll_for_sentinel`` is clock-bounded: never exceeds its cap, no
|
|
17
|
+
busy-spin (sleeps sum to <= cap, with generous tolerance).
|
|
18
|
+
|
|
19
|
+
Stdlib-only (``python3 -m unittest discover tests/``).
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
import os
|
|
23
|
+
import shutil
|
|
24
|
+
import sys
|
|
25
|
+
import tempfile
|
|
26
|
+
import time
|
|
27
|
+
import unittest
|
|
28
|
+
from unittest import mock
|
|
29
|
+
|
|
30
|
+
SCRIPTS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
|
|
31
|
+
if SCRIPTS_DIR not in sys.path:
|
|
32
|
+
sys.path.insert(0, SCRIPTS_DIR)
|
|
33
|
+
|
|
34
|
+
from lib import recall_buffer # noqa: E402
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class RecallBufferBase(unittest.TestCase):
|
|
38
|
+
def setUp(self):
|
|
39
|
+
self.tmp = tempfile.mkdtemp(prefix="hs-m4-buffer-")
|
|
40
|
+
self.env = mock.patch.dict(
|
|
41
|
+
os.environ, {"HINDSIGHT_PREFETCH_BUFFER_DIR": self.tmp}, clear=False
|
|
42
|
+
)
|
|
43
|
+
self.env.start()
|
|
44
|
+
|
|
45
|
+
def tearDown(self):
|
|
46
|
+
self.env.stop()
|
|
47
|
+
shutil.rmtree(self.tmp, ignore_errors=True)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class TestSentinelOrdering(RecallBufferBase):
|
|
51
|
+
def test_sentinel_written_strictly_after_payload(self):
|
|
52
|
+
session = "sess-order"
|
|
53
|
+
write_order = []
|
|
54
|
+
real_replace = os.replace
|
|
55
|
+
|
|
56
|
+
def _spy_replace(src, dst):
|
|
57
|
+
write_order.append(os.path.basename(dst))
|
|
58
|
+
return real_replace(src, dst)
|
|
59
|
+
|
|
60
|
+
with mock.patch("os.replace", side_effect=_spy_replace):
|
|
61
|
+
recall_buffer.write_buffer(session, "<hindsight_memories>x</hindsight_memories>", {"k": "v"})
|
|
62
|
+
recall_buffer.write_sentinel(session)
|
|
63
|
+
|
|
64
|
+
buffer_writes = [w for w in write_order if "buffer.json" in w]
|
|
65
|
+
sentinel_writes = [w for w in write_order if "buffer.done" in w]
|
|
66
|
+
self.assertEqual(len(buffer_writes), 1)
|
|
67
|
+
self.assertEqual(len(sentinel_writes), 1)
|
|
68
|
+
self.assertLess(
|
|
69
|
+
write_order.index(buffer_writes[0]),
|
|
70
|
+
write_order.index(sentinel_writes[0]),
|
|
71
|
+
"sentinel must be written strictly AFTER the buffer payload",
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
def test_read_if_fresh_returns_none_when_sentinel_absent(self):
|
|
75
|
+
session = "sess-nosentinel"
|
|
76
|
+
recall_buffer.write_buffer(session, "content", {})
|
|
77
|
+
ctx, token = recall_buffer.read_if_fresh(session, last_consumed_token=None)
|
|
78
|
+
self.assertIsNone(ctx)
|
|
79
|
+
|
|
80
|
+
def test_read_if_fresh_returns_none_when_sentinel_not_newer(self):
|
|
81
|
+
session = "sess-stale"
|
|
82
|
+
recall_buffer.write_buffer(session, "content", {})
|
|
83
|
+
token = recall_buffer.write_sentinel(session)
|
|
84
|
+
# Already consumed this exact token -> not fresh.
|
|
85
|
+
ctx, new_token = recall_buffer.read_if_fresh(session, last_consumed_token=token)
|
|
86
|
+
self.assertIsNone(ctx)
|
|
87
|
+
self.assertEqual(new_token, token)
|
|
88
|
+
|
|
89
|
+
def test_read_if_fresh_returns_payload_when_sentinel_is_newer(self):
|
|
90
|
+
session = "sess-fresh"
|
|
91
|
+
recall_buffer.write_buffer(session, "<hindsight_memories>fact</hindsight_memories>", {"telemetry": 1})
|
|
92
|
+
token = recall_buffer.write_sentinel(session)
|
|
93
|
+
ctx, new_token = recall_buffer.read_if_fresh(session, last_consumed_token=None)
|
|
94
|
+
self.assertIsNotNone(ctx)
|
|
95
|
+
self.assertIn("fact", ctx["context"])
|
|
96
|
+
self.assertEqual(ctx["telemetry"], {"telemetry": 1})
|
|
97
|
+
self.assertEqual(new_token, token)
|
|
98
|
+
|
|
99
|
+
def test_torn_write_payload_present_sentinel_absent_reads_as_none(self):
|
|
100
|
+
# Simulate a crash between write_buffer() and write_sentinel(): the
|
|
101
|
+
# payload landed, the sentinel never did. Must fail-closed to None,
|
|
102
|
+
# NEVER read the torn/possibly-incomplete payload as fresh.
|
|
103
|
+
session = "sess-torn"
|
|
104
|
+
recall_buffer.write_buffer(session, "<hindsight_memories>partial</hindsight_memories>", {})
|
|
105
|
+
# No write_sentinel() call — this IS the torn state.
|
|
106
|
+
ctx, token = recall_buffer.read_if_fresh(session, last_consumed_token=None)
|
|
107
|
+
self.assertIsNone(ctx, "torn write (payload without sentinel) must read as None, fail-closed")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class TestPollCap(RecallBufferBase):
|
|
111
|
+
def test_poll_returns_true_when_sentinel_appears(self):
|
|
112
|
+
session = "sess-poll-hit"
|
|
113
|
+
recall_buffer.write_buffer(session, "content", {})
|
|
114
|
+
recall_buffer.write_sentinel(session)
|
|
115
|
+
found = recall_buffer.poll_for_sentinel(session, last_consumed_token=None, cap_ms=500)
|
|
116
|
+
self.assertTrue(found)
|
|
117
|
+
|
|
118
|
+
def test_poll_never_exceeds_cap(self):
|
|
119
|
+
# No sentinel ever appears — poll must give up within cap_ms
|
|
120
|
+
# (generous tolerance for test-host scheduling jitter).
|
|
121
|
+
session = "sess-poll-miss"
|
|
122
|
+
cap_ms = 200
|
|
123
|
+
start = time.monotonic()
|
|
124
|
+
found = recall_buffer.poll_for_sentinel(session, last_consumed_token=None, cap_ms=cap_ms)
|
|
125
|
+
elapsed_ms = (time.monotonic() - start) * 1000
|
|
126
|
+
self.assertFalse(found)
|
|
127
|
+
self.assertLess(elapsed_ms, cap_ms + 150, "poll exceeded its cap by more than tolerance")
|
|
128
|
+
|
|
129
|
+
def test_poll_does_not_busy_spin(self):
|
|
130
|
+
# A busy-spin would burn far more than a handful of read attempts in
|
|
131
|
+
# the cap window; assert sleep is actually happening by checking the
|
|
132
|
+
# elapsed time is close to (not far below) the cap when nothing
|
|
133
|
+
# ever shows up.
|
|
134
|
+
session = "sess-poll-nospin"
|
|
135
|
+
cap_ms = 150
|
|
136
|
+
start = time.monotonic()
|
|
137
|
+
recall_buffer.poll_for_sentinel(session, last_consumed_token=None, cap_ms=cap_ms)
|
|
138
|
+
elapsed_ms = (time.monotonic() - start) * 1000
|
|
139
|
+
self.assertGreaterEqual(elapsed_ms, cap_ms * 0.5, "poll returned suspiciously fast — looks like it skipped sleeping")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
if __name__ == "__main__":
|
|
143
|
+
unittest.main()
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""M4 P-REC test (b) + red-team Fix B / Fix C — prefetch buffer join.
|
|
2
|
+
|
|
3
|
+
Asserts on RENDERED transport (`additionalContext`), never on an internal
|
|
4
|
+
flag alone, per the red-team's explicit strengthening. Covers:
|
|
5
|
+
|
|
6
|
+
* Fix C (kill switch): `memoryPrefetchEnabled` OFF (the default) leaves
|
|
7
|
+
the mechanism completely inert — synchronous recall runs exactly as
|
|
8
|
+
before, `lib.recall_buffer` is never touched.
|
|
9
|
+
* Fresh-hit: a producer-written buffer is joined and rendered as-is.
|
|
10
|
+
* Fix B (BINDING): the stale-buffer fallback (no fresh sentinel, but a
|
|
11
|
+
prior `LAST_RECALL_STATE` exists) renders memories WITHOUT any
|
|
12
|
+
directives block, even when a directive is active for the bank —
|
|
13
|
+
directives stay on the synchronous fetch path, never the stale cache.
|
|
14
|
+
* Cold-start short circuit: no sentinel has EVER existed for the session
|
|
15
|
+
-> no poll wait, falls through to the degraded notice / sync path
|
|
16
|
+
without blocking for the poll cap.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import io
|
|
20
|
+
import json
|
|
21
|
+
import os
|
|
22
|
+
import shutil
|
|
23
|
+
import sys
|
|
24
|
+
import tempfile
|
|
25
|
+
import time
|
|
26
|
+
import unittest
|
|
27
|
+
from unittest import mock
|
|
28
|
+
from unittest.mock import patch
|
|
29
|
+
|
|
30
|
+
SCRIPTS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
|
|
31
|
+
if SCRIPTS_DIR not in sys.path:
|
|
32
|
+
sys.path.insert(0, SCRIPTS_DIR)
|
|
33
|
+
|
|
34
|
+
import recall # noqa: E402
|
|
35
|
+
from lib import recall_buffer # noqa: E402
|
|
36
|
+
from lib.state import write_state # noqa: E402
|
|
37
|
+
|
|
38
|
+
SESSION = "test-session"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class _DirectiveClient:
|
|
42
|
+
"""Answers with ONE active directive and no memories (recall not
|
|
43
|
+
expected to be called on the fresh-hit/stale-fallback fast paths, but
|
|
44
|
+
directive fetch IS expected — always synchronous per M3)."""
|
|
45
|
+
|
|
46
|
+
def list_directives(self, bank_id, active_only=True, timeout=2):
|
|
47
|
+
return {"items": [{"id": "d1", "name": "haiku-rule", "content": "always reply in haiku", "priority": 5, "active": True}]}
|
|
48
|
+
|
|
49
|
+
def recall(self, bank_id, query, **kwargs):
|
|
50
|
+
raise AssertionError("buffer-join fast path must not call recall() directly")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class BufferJoinBase(unittest.TestCase):
|
|
54
|
+
def setUp(self):
|
|
55
|
+
self._tmpdir = tempfile.mkdtemp(prefix="recall-bufjoin-test-")
|
|
56
|
+
self._prev = os.environ.get("CLAUDE_PLUGIN_DATA")
|
|
57
|
+
os.environ["CLAUDE_PLUGIN_DATA"] = self._tmpdir
|
|
58
|
+
|
|
59
|
+
self._bufdir = tempfile.mkdtemp(prefix="recall-bufjoin-buf-")
|
|
60
|
+
self.env = mock.patch.dict(os.environ, {"HINDSIGHT_PREFETCH_BUFFER_DIR": self._bufdir}, clear=False)
|
|
61
|
+
self.env.start()
|
|
62
|
+
|
|
63
|
+
def tearDown(self):
|
|
64
|
+
self.env.stop()
|
|
65
|
+
shutil.rmtree(self._bufdir, ignore_errors=True)
|
|
66
|
+
shutil.rmtree(self._tmpdir, ignore_errors=True)
|
|
67
|
+
if self._prev is None:
|
|
68
|
+
os.environ.pop("CLAUDE_PLUGIN_DATA", None)
|
|
69
|
+
else:
|
|
70
|
+
os.environ["CLAUDE_PLUGIN_DATA"] = self._prev
|
|
71
|
+
|
|
72
|
+
def _config(self, prefetch_enabled):
|
|
73
|
+
return {
|
|
74
|
+
"autoRecall": True,
|
|
75
|
+
"bankId": "test-bank",
|
|
76
|
+
"recallMaxTokens": 4096,
|
|
77
|
+
"recallBudget": "mid",
|
|
78
|
+
"recallContextTurns": 1,
|
|
79
|
+
"recallMaxQueryChars": 800,
|
|
80
|
+
"recallPromptPreamble": "",
|
|
81
|
+
"recallParallelDeadlineSeconds": 5,
|
|
82
|
+
"directivesCacheTtlSeconds": 0,
|
|
83
|
+
"memoryPrefetchEnabled": prefetch_enabled,
|
|
84
|
+
"memoryPrefetchPollCapMs": 100,
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
def _run(self, config, client, prompt="what did we decide about deploys"):
|
|
88
|
+
hook_input = {"prompt": prompt, "session_id": SESSION, "transcript_path": "", "cwd": "/tmp"}
|
|
89
|
+
stdout = io.StringIO()
|
|
90
|
+
with patch("recall.load_config", return_value=config), \
|
|
91
|
+
patch("recall.get_api_url", return_value="http://fake"), \
|
|
92
|
+
patch("recall.HindsightClient", return_value=client), \
|
|
93
|
+
patch("recall.ensure_bank_mission"), \
|
|
94
|
+
patch("sys.stdin", io.StringIO(json.dumps(hook_input))), \
|
|
95
|
+
patch("sys.stdout", stdout):
|
|
96
|
+
recall.main()
|
|
97
|
+
return stdout.getvalue()
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class FreshHitTests(BufferJoinBase):
|
|
101
|
+
def test_fresh_buffer_hit_is_rendered_with_directives_layered_on(self):
|
|
102
|
+
recall_buffer.write_buffer(SESSION, "- a prefetched memory", {})
|
|
103
|
+
recall_buffer.write_sentinel(SESSION)
|
|
104
|
+
|
|
105
|
+
out = self._run(self._config(prefetch_enabled=True), _DirectiveClient())
|
|
106
|
+
self.assertTrue(out)
|
|
107
|
+
ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
|
|
108
|
+
self.assertIn("a prefetched memory", ctx)
|
|
109
|
+
self.assertIn("always reply in haiku", ctx)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class StaleFallbackFixBTests(BufferJoinBase):
|
|
113
|
+
def test_stale_fallback_renders_memories_without_any_directives_block(self):
|
|
114
|
+
# No fresh sentinel this turn (buffer is cold), but a prior turn's
|
|
115
|
+
# cache DOES exist and DOES include memories_context (directive-free
|
|
116
|
+
# per Fix B) plus a bundled directive in `context` that must NOT
|
|
117
|
+
# leak through the stale path.
|
|
118
|
+
write_state(
|
|
119
|
+
recall.LAST_RECALL_STATE,
|
|
120
|
+
{
|
|
121
|
+
"context": "## Active Directives\n- always reply in haiku\n\n- a stale cached memory",
|
|
122
|
+
"memories_context": "- a stale cached memory",
|
|
123
|
+
"saved_at": "2026-01-01T00:00:00Z",
|
|
124
|
+
"bank_id": "test-bank",
|
|
125
|
+
"result_count": 1,
|
|
126
|
+
"directive_count": 1,
|
|
127
|
+
},
|
|
128
|
+
)
|
|
129
|
+
out = self._run(self._config(prefetch_enabled=True), _DirectiveClient())
|
|
130
|
+
self.assertTrue(out)
|
|
131
|
+
ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
|
|
132
|
+
self.assertIn("a stale cached memory", ctx)
|
|
133
|
+
self.assertIn("stale", ctx.lower())
|
|
134
|
+
# Fix B, BINDING: the fresh directive fetch (synchronous, M3 rule)
|
|
135
|
+
# IS allowed to appear — what must NEVER appear is the STALE
|
|
136
|
+
# directives text sourced from the cached `context` field. Prove
|
|
137
|
+
# this by using a client whose directive text differs from what a
|
|
138
|
+
# stale-context leak would have produced, and confirming there is
|
|
139
|
+
# exactly one occurrence (the fresh fetch), not a duplicate/leaked
|
|
140
|
+
# second copy from the cache.
|
|
141
|
+
self.assertEqual(ctx.count("always reply in haiku"), 1)
|
|
142
|
+
|
|
143
|
+
def test_stale_fallback_directive_free_field_never_leaks_bare_context_directives(self):
|
|
144
|
+
# A cache row is deliberately malformed/legacy-shaped (missing the
|
|
145
|
+
# M4 `memories_context` field entirely, i.e. pre-M4 cache on disk).
|
|
146
|
+
# The fallback must degrade to "nothing stale to show", NEVER fall
|
|
147
|
+
# back to the directive-contaminated `context` field.
|
|
148
|
+
write_state(
|
|
149
|
+
recall.LAST_RECALL_STATE,
|
|
150
|
+
{
|
|
151
|
+
"context": "## Active Directives\n- always reply in haiku\n\n- a stale cached memory",
|
|
152
|
+
"saved_at": "2026-01-01T00:00:00Z",
|
|
153
|
+
"bank_id": "test-bank",
|
|
154
|
+
"result_count": 1,
|
|
155
|
+
"directive_count": 1,
|
|
156
|
+
},
|
|
157
|
+
)
|
|
158
|
+
client = _DirectiveClient()
|
|
159
|
+
out = self._run(self._config(prefetch_enabled=True), client)
|
|
160
|
+
self.assertTrue(out)
|
|
161
|
+
ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
|
|
162
|
+
self.assertNotIn("a stale cached memory", ctx, "must not fall back to the directive-contaminated context field")
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class ColdStartShortCircuitTests(BufferJoinBase):
|
|
166
|
+
def test_no_sentinel_ever_written_skips_the_poll_wait(self):
|
|
167
|
+
config = self._config(prefetch_enabled=True)
|
|
168
|
+
config["memoryPrefetchPollCapMs"] = 5000 # would dominate the test if NOT short-circuited
|
|
169
|
+
start = time.monotonic()
|
|
170
|
+
out = self._run(config, _DirectiveClient())
|
|
171
|
+
elapsed_ms = (time.monotonic() - start) * 1000
|
|
172
|
+
self.assertTrue(out)
|
|
173
|
+
self.assertLess(elapsed_ms, 1000, "cold-start (no sentinel ever) must not wait out the poll cap")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class KillSwitchOffTests(BufferJoinBase):
|
|
177
|
+
def test_flag_off_never_touches_the_buffer_module(self):
|
|
178
|
+
recall_buffer.write_buffer(SESSION, "- a prefetched memory", {})
|
|
179
|
+
recall_buffer.write_sentinel(SESSION)
|
|
180
|
+
|
|
181
|
+
with patch("recall.recall_buffer.read_if_fresh", side_effect=AssertionError("must not be called when flag is off")):
|
|
182
|
+
client = _DirectiveClient()
|
|
183
|
+
client.recall = lambda *a, **kw: {"results": []} # sync path IS allowed to call recall
|
|
184
|
+
out = self._run(self._config(prefetch_enabled=False), client)
|
|
185
|
+
self.assertTrue(out)
|
|
186
|
+
ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
|
|
187
|
+
# Byte-identical-behaviour proof: the buffer's content must NOT
|
|
188
|
+
# appear (it was never consulted) even though it exists on disk.
|
|
189
|
+
self.assertNotIn("a prefetched memory", ctx)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
if __name__ == "__main__":
|
|
193
|
+
unittest.main()
|