switchroom 0.21.18 → 0.21.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -92,6 +92,13 @@ _IMPORT_ELAPSED_SECONDS = time.monotonic() - _IMPORT_START_MONOTONIC
92
92
 
93
93
  LAST_RECALL_STATE = "last_recall.json"
94
94
  RECALL_CACHE_STATE = "recall_cache.json"
95
+ # M4 P-REC F3 — the last prefetch-buffer sentinel token this session has
96
+ # already CONSUMED, keyed by session_id. Persisted across turns so a buffer
97
+ # produced at turn N and consumed at N+1 is not re-served as "fresh" at
98
+ # N+2..N+k when no newer sentinel has landed (the stale-buffer-served-as-fresh
99
+ # class). A dict {session_id: token}; capped like the other per-session state.
100
+ PREFETCH_CONSUMED_STATE = "prefetch_consumed.json"
101
+ _PREFETCH_CONSUMED_MAX_SESSIONS = 10000
95
102
 
96
103
  # Switchroom hindsight-leverage A3 — label for the directives fetch slot in the
97
104
  # parallel fan-out. Distinct from any bank_id (banks can't start with "__") so a
@@ -500,12 +507,6 @@ def _emit_cached_context(context: str) -> None:
500
507
  )
501
508
 
502
509
 
503
- _PREFETCH_DEGRADED_NOTICE = (
504
- "⏳ prefetch not ready and no prior recall is cached for this session — "
505
- "proceeding without injected memory this turn."
506
- )
507
-
508
-
509
510
  def stale_recall_notice(memories_context: str) -> str:
510
511
  """Wrap a PRIOR turn's cached memories-only block in an explicit
511
512
  staleness marker for the M4 prefetch-buffer fallback path.
@@ -528,6 +529,100 @@ def stale_recall_notice(memories_context: str) -> str:
528
529
  )
529
530
 
530
531
 
532
+ def _read_consumed_token(session_id: str) -> "int | None":
533
+ """Return the last prefetch sentinel token this session already consumed,
534
+ or None if none is recorded. Failure-tolerant — any read/parse error
535
+ returns None (treat as "nothing consumed yet", the fail-open direction that
536
+ at worst re-serves once, never suppresses forever)."""
537
+ state = read_state(PREFETCH_CONSUMED_STATE, {}) or {}
538
+ if not isinstance(state, dict):
539
+ return None
540
+ token = state.get(session_id)
541
+ return int(token) if isinstance(token, (int, float)) else None
542
+
543
+
544
+ def _write_consumed_token(session_id: str, token: int) -> None:
545
+ """Persist `token` as the last sentinel this session has consumed. Bounded
546
+ to `_PREFETCH_CONSUMED_MAX_SESSIONS` entries. Best-effort — write_state
547
+ never raises."""
548
+ state = read_state(PREFETCH_CONSUMED_STATE, {}) or {}
549
+ if not isinstance(state, dict):
550
+ state = {}
551
+ state[session_id] = int(token)
552
+ if len(state) > _PREFETCH_CONSUMED_MAX_SESSIONS:
553
+ # Drop the numerically-smallest tokens (oldest producers) first.
554
+ for k in sorted(
555
+ state.keys(),
556
+ key=lambda k: state[k] if isinstance(state[k], (int, float)) else 0,
557
+ )[: len(state) - _PREFETCH_CONSUMED_MAX_SESSIONS]:
558
+ state.pop(k, None)
559
+ write_state(PREFETCH_CONSUMED_STATE, state)
560
+
561
+
562
+ def curate_recall_results(results, config, bank_id):
563
+ """M4 F5 — the SHARED recall curation pipeline, so the async prefetch
564
+ producer (`prefetch.py`) can never drift from the guarantees the
565
+ synchronous recall path enforces.
566
+
567
+ Applies, in order, the exact same module-level primitives the synchronous
568
+ path composes inline: drop demote-tagged memories (`_is_demoted_memory`),
569
+ sort by engine relevance (`_sort_by_final_score`), apply the optional
570
+ absolute score floor when it is configured for ALL turns
571
+ (`_filter_by_min_score`; the `degraded`-scope floor is a synchronous-path
572
+ concept the speculative single-bank prefetch cannot evaluate, so it is
573
+ honoured here only when `recallMinScoreScope` is `all`, matching what the
574
+ sync path does on a healthy own-bank turn), and finally the
575
+ `recallMaxMemories` head cap with per-bank slot reservation
576
+ (`_reserve_bank_slots`).
577
+
578
+ Returns the curated result list. Never mutates the caller's list in a way
579
+ that changes its length before returning (it rebinds internally). Pure —
580
+ no I/O, no network — so it is safe to call from the producer hook.
581
+ """
582
+ if not results:
583
+ return results
584
+
585
+ # 1) demote-from-recall drop (Switchroom #432 4.4) — before any cap so the
586
+ # cap fills with non-demoted hits.
587
+ results = [m for m in results if not _is_demoted_memory(m)]
588
+
589
+ # 2) relevance sort (Phase-1 bank-starvation fix) — in place on our copy.
590
+ _sort_by_final_score(results)
591
+
592
+ # 3) optional absolute score floor (#3837). The sync path binds this on
593
+ # `all` scope always, and on `degraded` scope only when the own-bank
594
+ # read degraded. A prefetch is a healthy speculative read (no degraded
595
+ # signal), so we bind only the `all` scope here — identical to the sync
596
+ # path's decision on a non-degraded turn.
597
+ min_score_floor = config.get("recallMinScore", 0.0)
598
+ if isinstance(min_score_floor, bool) or not isinstance(min_score_floor, (int, float)):
599
+ min_score_floor = 0.0
600
+ min_score_floor = float(min_score_floor)
601
+ min_score_scope = config.get("recallMinScoreScope", "degraded")
602
+ if min_score_floor > 0 and min_score_scope == "all":
603
+ results, _dropped = _filter_by_min_score(results, min_score_floor)
604
+
605
+ # 4) recallMaxMemories head cap + per-bank slot reservation. `_reserve_
606
+ # bank_slots` is a passthrough when cap<=0 or the set already fits, and
607
+ # with the fleet-default 0/0 floors on a single-bank prefetch it reduces
608
+ # to a plain head-slice — the same slice the sync path takes.
609
+ recall_max_memories = config.get("recallMaxMemories", 0)
610
+ if (
611
+ isinstance(recall_max_memories, int)
612
+ and recall_max_memories > 0
613
+ and len(results) > recall_max_memories
614
+ ):
615
+ results, _own, _add = _reserve_bank_slots(
616
+ results,
617
+ recall_max_memories,
618
+ bank_id,
619
+ config.get("recallOwnBankMinSlots", 0),
620
+ config.get("recallAdditionalBankMinSlots", 0),
621
+ )
622
+
623
+ return results
624
+
625
+
531
626
  def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool:
532
627
  """M4 P-REC Fix C (consumer side) — join the Stop-hook producer's
533
628
  prefetch buffer for this session instead of running recall
@@ -535,10 +630,12 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
535
630
 
536
631
  Gated entirely by `config.get("memoryPrefetchEnabled", False)` at the
537
632
  caller; this function assumes the flag is already on. Returns True iff
538
- it emitted an `additionalContext` payload (fresh hit, stale fallback, or
539
- the explicit degraded notice) and the caller should return without
540
- running the synchronous path. Returns False on a clean no-op miss (flag
541
- effectively off / nothing to say) so the caller falls through.
633
+ it emitted an `additionalContext` payload (fresh buffer hit, stale
634
+ fallback, or a directives-only block) and the caller should return
635
+ without running the synchronous path. Returns False on a miss with
636
+ nothing to serve — cold start, no fresh buffer, no stale prior recall,
637
+ no directives — so the caller falls through to SYNCHRONOUS recall (M4
638
+ F4: the first turn of a fresh/post-restart session must still recall).
542
639
 
543
640
  Never raises past this function's own boundary in normal operation —
544
641
  every internal step is wrapped so a bug here degrades to "fall through
@@ -547,6 +644,13 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
547
644
  """
548
645
  session_id = hook_input.get("session_id") or "unknown"
549
646
 
647
+ # M4 F3 — the sentinel token this session has ALREADY consumed on a prior
648
+ # turn. Threaded into both `poll_for_sentinel` and `read_if_fresh` so a
649
+ # buffer produced at turn N and consumed at N+1 is treated as STALE (not
650
+ # re-served as fresh) at N+2..N+k until a strictly-newer sentinel lands.
651
+ # None on a cold session (nothing consumed yet).
652
+ last_consumed = _read_consumed_token(session_id)
653
+
550
654
  # Cold-start short-circuit (red-team MAJOR finding): if this session has
551
655
  # NEVER produced a sentinel, polling the full cap on every single
552
656
  # session-open turn would cost ~the poll cap on every fresh session —
@@ -556,9 +660,24 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
556
660
  debug_log(config, "Prefetch buffer: no sentinel ever written for this session, cold-start skip")
557
661
  else:
558
662
  cap_ms = int(config.get("memoryPrefetchPollCapMs", 400))
559
- recall_buffer.poll_for_sentinel(session_id, last_consumed_token=None, cap_ms=cap_ms)
560
-
561
- payload, _token = recall_buffer.read_if_fresh(session_id, last_consumed_token=None)
663
+ recall_buffer.poll_for_sentinel(session_id, last_consumed_token=last_consumed, cap_ms=cap_ms)
664
+
665
+ payload, _token = recall_buffer.read_if_fresh(session_id, last_consumed_token=last_consumed)
666
+
667
+ # #4778 — TOPIC-RELEVANCE GUARD. `read_if_fresh` only proves the buffer is
668
+ # FRESH (a strictly-newer sentinel this session), never that it is ON TOPIC:
669
+ # the producer built it from turn N's last human prompt, so on a topic PIVOT
670
+ # a perfectly-fresh buffer holds the WRONG-topic memories. Gate the join on
671
+ # topical similarity between the buffered query and turn N+1's ACTUAL prompt
672
+ # BEFORE the directive fetch below; a mismatch marks the token consumed (so
673
+ # it is never reconsidered) and falls through to SYNCHRONOUS recall — which
674
+ # fetches correct, current-topic memories AND re-injects directives itself.
675
+ # This is layered ON TOP of the F3 freshness machinery, not a replacement.
676
+ if payload is not None and not _prefetch_topic_matches(prompt, payload.get("query", ""), config):
677
+ if _token is not None:
678
+ _write_consumed_token(session_id, _token)
679
+ debug_log(config, "Prefetch buffer: topic mismatch vs current prompt, falling through to synchronous recall")
680
+ return False
562
681
 
563
682
  # Directives stay on the synchronous, always-fresh path (M3 rule) even
564
683
  # in the fast path — fetched here directly, never from the buffer.
@@ -584,6 +703,12 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
584
703
  directives_block = None
585
704
 
586
705
  if payload is not None:
706
+ # M4 F3 — record this token as consumed BEFORE emitting, so the next
707
+ # turn's `read_if_fresh` rejects the same buffer as stale unless the
708
+ # producer has since written a strictly-newer sentinel. `_token` is the
709
+ # sentinel token `read_if_fresh` just validated as fresh.
710
+ if _token is not None:
711
+ _write_consumed_token(session_id, _token)
587
712
  memories_block = payload.get("context") or ""
588
713
  parts = [b for b in (_dir_notice, directives_block, memories_block) if b]
589
714
  if not parts:
@@ -607,11 +732,15 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
607
732
  _emit_cached_context("\n\n".join([b for b in (_dir_notice, directives_block) if b]))
608
733
  return True
609
734
 
610
- # Nothing fresh, nothing stale, nothing cached — say so explicitly
611
- # rather than silently emitting no context (so a degraded turn is
612
- # legible, matching the #3619 degraded-disclosure precedent).
613
- _emit_cached_context(_PREFETCH_DEGRADED_NOTICE)
614
- return True
735
+ # M4 F4 — nothing fresh, nothing stale, nothing cached (the cold-start /
736
+ # first-turn / no-prior-recall case). Return False so the caller runs the
737
+ # SYNCHRONOUS recall path, which is exactly the path that should run on a
738
+ # turn where no prefetch could possibly exist yet. Emitting a degraded
739
+ # notice and returning True here (the previous behaviour) short-circuited
740
+ # sync recall and handed a fresh/post-restart session ZERO memories plus a
741
+ # degraded banner on its very first turn — the opposite of the docstring's
742
+ # "a miss falls through to sync recall" contract.
743
+ return False
615
744
 
616
745
 
617
746
  def _emit_directives_only(config: dict, hook_input: dict) -> None:
@@ -774,6 +903,78 @@ def _overlap_tokens(text) -> set:
774
903
  return out
775
904
 
776
905
 
906
+ _PREFETCH_TOPIC_OVERLAP_DEFAULT = 0.3
907
+
908
+ # #4778 review MAJOR — short-prompt Jaccard floor. Jaccard is probabilistic, not
909
+ # absolute, on SHORT prompts: two ~2-content-token turns that share ONE incidental
910
+ # token while EACH carries a divergent token give 1/3 = 0.333 >= 0.30 and wrongly
911
+ # clear the ratio bar ("call mom" buffer served on a "call ended" turn — the pivot
912
+ # signature). When either token set is small (fewer than ``MIN_SMALL_SET_TOKENS``)
913
+ # such a partial overlap must supply at least ``_PREFETCH_MIN_SMALL_SET_INTERSECTION``
914
+ # shared tokens, not just clear the ratio.
915
+ #
916
+ # The floor fires ONLY on that divergent-on-both-sides shape. It deliberately
917
+ # EXEMPTS full containment (``intersection == min(|A|, |B|)`` — the smaller set is
918
+ # wholly shared: identical or subset/narrowing queries like "decide" vs "decide",
919
+ # or "restart klanker" vs "restart the klanker agent"), which is a legitimate
920
+ # topical match regardless of brevity and must still be served. Long prompts (both
921
+ # sides at or above ``MIN_SMALL_SET_TOKENS``) skip the floor entirely and use the
922
+ # Jaccard ratio exactly as before. The floor only ever ANDs a stricter condition
923
+ # onto the ratio — it can reject, never serve.
924
+ MIN_SMALL_SET_TOKENS = 3
925
+ _PREFETCH_MIN_SMALL_SET_INTERSECTION = 2
926
+
927
+
928
+ def _prefetch_topic_matches(prompt, buffered_query, config) -> bool:
929
+ """#4778 — True iff turn N+1's ``prompt`` is topically close enough to the
930
+ ``buffered_query`` the M4 producer used to build the warm buffer.
931
+
932
+ Jaccard overlap (``|A∩B| / |A∪B|``) on ``_overlap_tokens`` — the same
933
+ stop-word-stripped, digit-preserving, dependency-free tokenizer the
934
+ transcript-fallback keyword match uses (unicode-safe via ``str.isalnum``,
935
+ characterised in ``tests/test_overlap_tokens.py``) — against
936
+ ``memoryPrefetchMinTopicOverlap`` (default 0.3). Jaccard, not the overlap
937
+ coefficient: both score 0 on a true pivot (disjoint content tokens), so the
938
+ correctness guarantee is identical, but Jaccard cannot be driven to 1.0 by a
939
+ single incidental shared token in a short prompt — it biases the residual
940
+ error toward a harmless false-MISS (sync recall, slower) rather than a
941
+ false-HIT (wrong-topic memories, the bug).
942
+
943
+ Fail-safe: an empty token set on EITHER side — a contentless prompt, or a
944
+ legacy/torn buffer whose ``query`` is ``""`` — returns False, so the caller
945
+ falls through to synchronous recall. A miss only ever costs latency; it can
946
+ never serve wrong-topic memories. Both sides are stripped of the ``<channel>``
947
+ envelope first, matching the producer's own query derivation.
948
+ """
949
+ now_tokens = _overlap_tokens(strip_channel_envelope(prompt or ""))
950
+ buf_tokens = _overlap_tokens(strip_channel_envelope(buffered_query or ""))
951
+ if not now_tokens or not buf_tokens:
952
+ return False
953
+ intersection = len(now_tokens & buf_tokens)
954
+ if intersection == 0:
955
+ return False
956
+ # Short-prompt floor (#4778 review MAJOR): when either side is small, a single
957
+ # incidental shared token must not clear the guard on the Jaccard ratio alone.
958
+ # Exempt full containment (intersection == smaller set) — identical/subset
959
+ # queries are a legitimate topical match, never the divergent-on-both-sides
960
+ # pivot this floor targets. ANDed with the ratio check below: stricter, never
961
+ # looser, and long prompts (both sides >= MIN_SMALL_SET_TOKENS) are untouched.
962
+ min_set = min(len(now_tokens), len(buf_tokens))
963
+ if min_set < MIN_SMALL_SET_TOKENS \
964
+ and intersection < min_set \
965
+ and intersection < _PREFETCH_MIN_SMALL_SET_INTERSECTION:
966
+ return False
967
+ union = len(now_tokens | buf_tokens)
968
+ threshold = config.get("memoryPrefetchMinTopicOverlap", _PREFETCH_TOPIC_OVERLAP_DEFAULT)
969
+ try:
970
+ threshold = float(threshold)
971
+ except (TypeError, ValueError):
972
+ threshold = _PREFETCH_TOPIC_OVERLAP_DEFAULT
973
+ if not (0.0 <= threshold <= 1.0):
974
+ threshold = _PREFETCH_TOPIC_OVERLAP_DEFAULT
975
+ return (intersection / union) >= threshold
976
+
977
+
777
978
  def _result_final_score(m) -> float:
778
979
  """Return a result's engine relevance score (`scores.final`).
779
980
 
@@ -278,7 +278,9 @@ class NoDuplicateInjectionTests(InvalidationBase):
278
278
  config = self._config(prefetch_enabled=True)
279
279
  marker = "- unique-fact-abc123 decided at standup"
280
280
 
281
- recall_buffer.write_buffer(SESSION, marker, {})
281
+ # #4778 — on-topic with the `_consume` prompt ("what did we decide") so
282
+ # the topic guard passes and this stays a genuine warm-buffer hit.
283
+ recall_buffer.write_buffer(SESSION, marker, {}, query="what did we decide")
282
284
  recall_buffer.write_sentinel(SESSION)
283
285
 
284
286
  # Turn N+1: consumer injects the buffered block once.
@@ -311,7 +313,10 @@ class ReplyPathDoesNotBlockOnRecallTests(InvalidationBase):
311
313
 
312
314
  def test_fresh_buffer_served_without_calling_recall(self):
313
315
  config = self._config(prefetch_enabled=True)
314
- recall_buffer.write_buffer(SESSION, "- a prefetched memory xyz", {})
316
+ # #4778 — on-topic with the `_consume` prompt so the topic guard passes;
317
+ # the point of THIS test is that a matched fresh buffer never blocks on a
318
+ # synchronous recall, so the query must clear the guard.
319
+ recall_buffer.write_buffer(SESSION, "- a prefetched memory xyz", {}, query="what did we decide")
315
320
  recall_buffer.write_sentinel(SESSION)
316
321
 
317
322
  class _ExplodingRecallClient:
@@ -147,7 +147,7 @@ class CallOrderTests(PrefetchPipelineBase):
147
147
  order.append("retain")
148
148
  return {"status": "ok"}
149
149
 
150
- def _spy_write_buffer(session_id, context, telemetry=None):
150
+ def _spy_write_buffer(session_id, context, telemetry=None, query=None):
151
151
  order.append("buffer")
152
152
 
153
153
  def _spy_write_sentinel(session_id):
@@ -233,14 +233,105 @@ class KillSwitchOffTests(PrefetchPipelineBase):
233
233
 
234
234
 
235
235
  class JunkGateTests(PrefetchPipelineBase):
236
- def test_task_notification_turn_is_skipped(self):
237
- _write_transcript(self.transcript_path, [("user", "irrelevant")])
238
- with patch("prefetch.retain_module.run_retain", side_effect=AssertionError("must not retain a task-notification turn")):
239
- wrote = prefetch.run_prefetch(
240
- self._hook_input("<task-notification>done</task-notification>"), self._config()
241
- )
236
+ def test_task_notification_in_transcript_skips_before_retain(self):
237
+ # F6 (regression guard): a real Stop-hook input carries NO `prompt`
238
+ # field — only session_id / transcript_path / stop_hook_active — so the
239
+ # junk gate MUST derive from the transcript's last human turn. When that
240
+ # turn is a `<task-notification>` envelope, prefetch skips ENTIRELY:
241
+ # no delta retain, no recall, no buffer. This fails on the pre-fix code,
242
+ # whose `hook_input.get("prompt")` gate is permanently empty on real
243
+ # input and therefore lets retain run on a task-notification turn.
244
+ _write_transcript(
245
+ self.transcript_path,
246
+ [("user", "<task-notification>sub-agent done</task-notification>")],
247
+ )
248
+ real_stop_input = {
249
+ "session_id": SESSION,
250
+ "transcript_path": self.transcript_path,
251
+ "stop_hook_active": True,
252
+ "cwd": "/tmp",
253
+ } # deliberately NO `prompt` / `user_prompt` key — mirrors real input
254
+ retain_spy = mock.Mock(return_value={"status": "ok"})
255
+ with patch("prefetch.retain_module.run_retain", retain_spy):
256
+ wrote = prefetch.run_prefetch(real_stop_input, self._config())
242
257
  self.assertFalse(wrote)
258
+ retain_spy.assert_not_called()
243
259
  self.assertFalse(os.path.isfile(recall_buffer._buffer_path(SESSION)))
260
+ self.assertFalse(os.path.isfile(recall_buffer._sentinel_path(SESSION)))
261
+
262
+ def test_gate_keys_off_transcript_not_a_phantom_prompt_field(self):
263
+ # The mirror of the above: a genuine human turn in the transcript with
264
+ # a STRAY `prompt="<task-notification>"` on the hook input (the shape the
265
+ # old harness faked). The gate must key off the transcript (a real turn)
266
+ # and PROCEED, proving it no longer reads `hook_input["prompt"]`. Fails
267
+ # on the pre-fix code, which reads the phantom field and wrongly skips.
268
+ _write_transcript(self.transcript_path, [("user", "we shipped the release on friday")])
269
+ client = mock.Mock()
270
+ client.recall.side_effect = lambda *a, **kw: {"results": [
271
+ {"text": "we shipped the release on friday", "type": "fact",
272
+ "mentioned_at": "2026-01-01", "id": "r1", "scores": {"final": 0.9}},
273
+ ]}
274
+ hook_input = {
275
+ "session_id": SESSION,
276
+ "transcript_path": self.transcript_path,
277
+ "prompt": "<task-notification>phantom</task-notification>",
278
+ "cwd": "/tmp",
279
+ }
280
+ with patch("prefetch.retain_module.run_retain", return_value={"status": "ok"}), \
281
+ patch("prefetch.HindsightClient", return_value=client), \
282
+ patch("prefetch.get_api_url", return_value="http://fake"):
283
+ wrote = prefetch.run_prefetch(hook_input, self._config())
284
+ self.assertTrue(wrote)
285
+ self.assertTrue(os.path.isfile(recall_buffer._buffer_path(SESSION)))
286
+
287
+
288
+ class CurationSharingTests(PrefetchPipelineBase):
289
+ """F5 — the producer must run recalled candidates through recall's shared
290
+ curation pipeline (demote-drop + `recallMaxMemories` cap) BEFORE buffering,
291
+ so the buffer path cannot inject what the synchronous path would filter."""
292
+
293
+ def _run_with_results(self, results, config):
294
+ _write_transcript(self.transcript_path, [("user", "what do you remember about deploys")])
295
+ client = mock.Mock()
296
+ client.recall.side_effect = lambda *a, **kw: {"results": [dict(m) for m in results]}
297
+ with patch("prefetch.retain_module.run_retain", return_value={"status": "ok"}), \
298
+ patch("prefetch.HindsightClient", return_value=client), \
299
+ patch("prefetch.get_api_url", return_value="http://fake"):
300
+ wrote = prefetch.run_prefetch(self._hook_input("what do you remember about deploys"), config)
301
+ buffered = ""
302
+ if wrote:
303
+ payload = recall_buffer._read_buffer_payload(SESSION) or {}
304
+ buffered = payload.get("context") or ""
305
+ return wrote, buffered
306
+
307
+ def test_demote_tagged_memory_is_dropped_before_buffering(self):
308
+ results = [
309
+ {"text": "keep me visible", "type": "fact", "mentioned_at": "2026-01-01",
310
+ "id": "k1", "scores": {"final": 0.9}},
311
+ {"text": "secret demoted note", "type": "fact", "mentioned_at": "2026-01-01",
312
+ "id": "d1", "scores": {"final": 0.8}, "tags": ["demote-from-recall"]},
313
+ ]
314
+ _wrote, buffered = self._run_with_results(results, self._config())
315
+ self.assertIn("keep me visible", buffered)
316
+ self.assertNotIn(
317
+ "secret demoted note", buffered,
318
+ "a demote-from-recall memory must never reach the prefetch buffer",
319
+ )
320
+
321
+ def test_candidate_set_above_cap_is_truncated_before_buffering(self):
322
+ results = [
323
+ {"text": f"candidate {i}", "type": "fact", "mentioned_at": "2026-01-01",
324
+ "id": f"m{i}", "scores": {"final": 1.0 - i * 0.01}}
325
+ for i in range(12)
326
+ ]
327
+ config = self._config()
328
+ config["recallMaxMemories"] = 8
329
+ _wrote, buffered = self._run_with_results(results, config)
330
+ injected = sum(1 for i in range(12) if f"candidate {i}" in buffered)
331
+ self.assertEqual(
332
+ injected, 8,
333
+ "the prefetch buffer must honour recallMaxMemories, same as the sync path",
334
+ )
244
335
 
245
336
 
246
337
  if __name__ == "__main__":