switchroom 0.21.18 → 0.21.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +247 -3
- package/dist/host-control/main.js +1 -1
- package/package.json +1 -1
- package/skills/switchroom-release/SKILL.md +6 -1
- package/telegram-plugin/connection-drop.ts +83 -0
- package/telegram-plugin/dist/bridge/bridge.js +41 -2
- package/telegram-plugin/dist/gateway/gateway.js +48 -6
- package/telegram-plugin/dist/server.js +46 -3
- package/telegram-plugin/llm-error-present.ts +25 -0
- package/telegram-plugin/session-tail.ts +38 -1
- package/telegram-plugin/tests/llm-error-present.test.ts +183 -0
- package/vendor/hindsight-memory/.claude-plugin/plugin.json +1 -1
- package/vendor/hindsight-memory/CHANGELOG.md +20 -0
- package/vendor/hindsight-memory/scripts/lib/recall_buffer.py +26 -3
- package/vendor/hindsight-memory/scripts/prefetch.py +40 -6
- package/vendor/hindsight-memory/scripts/recall.py +219 -18
- package/vendor/hindsight-memory/scripts/tests/test_prefetch_invalidation.py +7 -2
- package/vendor/hindsight-memory/scripts/tests/test_prefetch_pipeline.py +98 -7
- package/vendor/hindsight-memory/scripts/tests/test_prefetch_topic_guard.py +279 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_buffer_join.py +74 -1
|
@@ -92,6 +92,13 @@ _IMPORT_ELAPSED_SECONDS = time.monotonic() - _IMPORT_START_MONOTONIC
|
|
|
92
92
|
|
|
93
93
|
LAST_RECALL_STATE = "last_recall.json"
|
|
94
94
|
RECALL_CACHE_STATE = "recall_cache.json"
|
|
95
|
+
# M4 P-REC F3 — the last prefetch-buffer sentinel token this session has
|
|
96
|
+
# already CONSUMED, keyed by session_id. Persisted across turns so a buffer
|
|
97
|
+
# produced at turn N and consumed at N+1 is not re-served as "fresh" at
|
|
98
|
+
# N+2..N+k when no newer sentinel has landed (the stale-buffer-served-as-fresh
|
|
99
|
+
# class). A dict {session_id: token}; capped like the other per-session state.
|
|
100
|
+
PREFETCH_CONSUMED_STATE = "prefetch_consumed.json"
|
|
101
|
+
_PREFETCH_CONSUMED_MAX_SESSIONS = 10000
|
|
95
102
|
|
|
96
103
|
# Switchroom hindsight-leverage A3 — label for the directives fetch slot in the
|
|
97
104
|
# parallel fan-out. Distinct from any bank_id (banks can't start with "__") so a
|
|
@@ -500,12 +507,6 @@ def _emit_cached_context(context: str) -> None:
|
|
|
500
507
|
)
|
|
501
508
|
|
|
502
509
|
|
|
503
|
-
_PREFETCH_DEGRADED_NOTICE = (
|
|
504
|
-
"⏳ prefetch not ready and no prior recall is cached for this session — "
|
|
505
|
-
"proceeding without injected memory this turn."
|
|
506
|
-
)
|
|
507
|
-
|
|
508
|
-
|
|
509
510
|
def stale_recall_notice(memories_context: str) -> str:
|
|
510
511
|
"""Wrap a PRIOR turn's cached memories-only block in an explicit
|
|
511
512
|
staleness marker for the M4 prefetch-buffer fallback path.
|
|
@@ -528,6 +529,100 @@ def stale_recall_notice(memories_context: str) -> str:
|
|
|
528
529
|
)
|
|
529
530
|
|
|
530
531
|
|
|
532
|
+
def _read_consumed_token(session_id: str) -> "int | None":
|
|
533
|
+
"""Return the last prefetch sentinel token this session already consumed,
|
|
534
|
+
or None if none is recorded. Failure-tolerant — any read/parse error
|
|
535
|
+
returns None (treat as "nothing consumed yet", the fail-open direction that
|
|
536
|
+
at worst re-serves once, never suppresses forever)."""
|
|
537
|
+
state = read_state(PREFETCH_CONSUMED_STATE, {}) or {}
|
|
538
|
+
if not isinstance(state, dict):
|
|
539
|
+
return None
|
|
540
|
+
token = state.get(session_id)
|
|
541
|
+
return int(token) if isinstance(token, (int, float)) else None
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _write_consumed_token(session_id: str, token: int) -> None:
|
|
545
|
+
"""Persist `token` as the last sentinel this session has consumed. Bounded
|
|
546
|
+
to `_PREFETCH_CONSUMED_MAX_SESSIONS` entries. Best-effort — write_state
|
|
547
|
+
never raises."""
|
|
548
|
+
state = read_state(PREFETCH_CONSUMED_STATE, {}) or {}
|
|
549
|
+
if not isinstance(state, dict):
|
|
550
|
+
state = {}
|
|
551
|
+
state[session_id] = int(token)
|
|
552
|
+
if len(state) > _PREFETCH_CONSUMED_MAX_SESSIONS:
|
|
553
|
+
# Drop the numerically-smallest tokens (oldest producers) first.
|
|
554
|
+
for k in sorted(
|
|
555
|
+
state.keys(),
|
|
556
|
+
key=lambda k: state[k] if isinstance(state[k], (int, float)) else 0,
|
|
557
|
+
)[: len(state) - _PREFETCH_CONSUMED_MAX_SESSIONS]:
|
|
558
|
+
state.pop(k, None)
|
|
559
|
+
write_state(PREFETCH_CONSUMED_STATE, state)
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def curate_recall_results(results, config, bank_id):
|
|
563
|
+
"""M4 F5 — the SHARED recall curation pipeline, so the async prefetch
|
|
564
|
+
producer (`prefetch.py`) can never drift from the guarantees the
|
|
565
|
+
synchronous recall path enforces.
|
|
566
|
+
|
|
567
|
+
Applies, in order, the exact same module-level primitives the synchronous
|
|
568
|
+
path composes inline: drop demote-tagged memories (`_is_demoted_memory`),
|
|
569
|
+
sort by engine relevance (`_sort_by_final_score`), apply the optional
|
|
570
|
+
absolute score floor when it is configured for ALL turns
|
|
571
|
+
(`_filter_by_min_score`; the `degraded`-scope floor is a synchronous-path
|
|
572
|
+
concept the speculative single-bank prefetch cannot evaluate, so it is
|
|
573
|
+
honoured here only when `recallMinScoreScope` is `all`, matching what the
|
|
574
|
+
sync path does on a healthy own-bank turn), and finally the
|
|
575
|
+
`recallMaxMemories` head cap with per-bank slot reservation
|
|
576
|
+
(`_reserve_bank_slots`).
|
|
577
|
+
|
|
578
|
+
Returns the curated result list. Never mutates the caller's list in a way
|
|
579
|
+
that changes its length before returning (it rebinds internally). Pure —
|
|
580
|
+
no I/O, no network — so it is safe to call from the producer hook.
|
|
581
|
+
"""
|
|
582
|
+
if not results:
|
|
583
|
+
return results
|
|
584
|
+
|
|
585
|
+
# 1) demote-from-recall drop (Switchroom #432 4.4) — before any cap so the
|
|
586
|
+
# cap fills with non-demoted hits.
|
|
587
|
+
results = [m for m in results if not _is_demoted_memory(m)]
|
|
588
|
+
|
|
589
|
+
# 2) relevance sort (Phase-1 bank-starvation fix) — in place on our copy.
|
|
590
|
+
_sort_by_final_score(results)
|
|
591
|
+
|
|
592
|
+
# 3) optional absolute score floor (#3837). The sync path binds this on
|
|
593
|
+
# `all` scope always, and on `degraded` scope only when the own-bank
|
|
594
|
+
# read degraded. A prefetch is a healthy speculative read (no degraded
|
|
595
|
+
# signal), so we bind only the `all` scope here — identical to the sync
|
|
596
|
+
# path's decision on a non-degraded turn.
|
|
597
|
+
min_score_floor = config.get("recallMinScore", 0.0)
|
|
598
|
+
if isinstance(min_score_floor, bool) or not isinstance(min_score_floor, (int, float)):
|
|
599
|
+
min_score_floor = 0.0
|
|
600
|
+
min_score_floor = float(min_score_floor)
|
|
601
|
+
min_score_scope = config.get("recallMinScoreScope", "degraded")
|
|
602
|
+
if min_score_floor > 0 and min_score_scope == "all":
|
|
603
|
+
results, _dropped = _filter_by_min_score(results, min_score_floor)
|
|
604
|
+
|
|
605
|
+
# 4) recallMaxMemories head cap + per-bank slot reservation. `_reserve_
|
|
606
|
+
# bank_slots` is a passthrough when cap<=0 or the set already fits, and
|
|
607
|
+
# with the fleet-default 0/0 floors on a single-bank prefetch it reduces
|
|
608
|
+
# to a plain head-slice — the same slice the sync path takes.
|
|
609
|
+
recall_max_memories = config.get("recallMaxMemories", 0)
|
|
610
|
+
if (
|
|
611
|
+
isinstance(recall_max_memories, int)
|
|
612
|
+
and recall_max_memories > 0
|
|
613
|
+
and len(results) > recall_max_memories
|
|
614
|
+
):
|
|
615
|
+
results, _own, _add = _reserve_bank_slots(
|
|
616
|
+
results,
|
|
617
|
+
recall_max_memories,
|
|
618
|
+
bank_id,
|
|
619
|
+
config.get("recallOwnBankMinSlots", 0),
|
|
620
|
+
config.get("recallAdditionalBankMinSlots", 0),
|
|
621
|
+
)
|
|
622
|
+
|
|
623
|
+
return results
|
|
624
|
+
|
|
625
|
+
|
|
531
626
|
def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool:
|
|
532
627
|
"""M4 P-REC Fix C (consumer side) — join the Stop-hook producer's
|
|
533
628
|
prefetch buffer for this session instead of running recall
|
|
@@ -535,10 +630,12 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
535
630
|
|
|
536
631
|
Gated entirely by `config.get("memoryPrefetchEnabled", False)` at the
|
|
537
632
|
caller; this function assumes the flag is already on. Returns True iff
|
|
538
|
-
it emitted an `additionalContext` payload (fresh hit, stale
|
|
539
|
-
|
|
540
|
-
running the synchronous path. Returns False on a
|
|
541
|
-
|
|
633
|
+
it emitted an `additionalContext` payload (fresh buffer hit, stale
|
|
634
|
+
fallback, or a directives-only block) and the caller should return
|
|
635
|
+
without running the synchronous path. Returns False on a miss with
|
|
636
|
+
nothing to serve — cold start, no fresh buffer, no stale prior recall,
|
|
637
|
+
no directives — so the caller falls through to SYNCHRONOUS recall (M4
|
|
638
|
+
F4: the first turn of a fresh/post-restart session must still recall).
|
|
542
639
|
|
|
543
640
|
Never raises past this function's own boundary in normal operation —
|
|
544
641
|
every internal step is wrapped so a bug here degrades to "fall through
|
|
@@ -547,6 +644,13 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
547
644
|
"""
|
|
548
645
|
session_id = hook_input.get("session_id") or "unknown"
|
|
549
646
|
|
|
647
|
+
# M4 F3 — the sentinel token this session has ALREADY consumed on a prior
|
|
648
|
+
# turn. Threaded into both `poll_for_sentinel` and `read_if_fresh` so a
|
|
649
|
+
# buffer produced at turn N and consumed at N+1 is treated as STALE (not
|
|
650
|
+
# re-served as fresh) at N+2..N+k until a strictly-newer sentinel lands.
|
|
651
|
+
# None on a cold session (nothing consumed yet).
|
|
652
|
+
last_consumed = _read_consumed_token(session_id)
|
|
653
|
+
|
|
550
654
|
# Cold-start short-circuit (red-team MAJOR finding): if this session has
|
|
551
655
|
# NEVER produced a sentinel, polling the full cap on every single
|
|
552
656
|
# session-open turn would cost ~the poll cap on every fresh session —
|
|
@@ -556,9 +660,24 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
556
660
|
debug_log(config, "Prefetch buffer: no sentinel ever written for this session, cold-start skip")
|
|
557
661
|
else:
|
|
558
662
|
cap_ms = int(config.get("memoryPrefetchPollCapMs", 400))
|
|
559
|
-
recall_buffer.poll_for_sentinel(session_id, last_consumed_token=
|
|
560
|
-
|
|
561
|
-
payload, _token = recall_buffer.read_if_fresh(session_id, last_consumed_token=
|
|
663
|
+
recall_buffer.poll_for_sentinel(session_id, last_consumed_token=last_consumed, cap_ms=cap_ms)
|
|
664
|
+
|
|
665
|
+
payload, _token = recall_buffer.read_if_fresh(session_id, last_consumed_token=last_consumed)
|
|
666
|
+
|
|
667
|
+
# #4778 — TOPIC-RELEVANCE GUARD. `read_if_fresh` only proves the buffer is
|
|
668
|
+
# FRESH (a strictly-newer sentinel this session), never that it is ON TOPIC:
|
|
669
|
+
# the producer built it from turn N's last human prompt, so on a topic PIVOT
|
|
670
|
+
# a perfectly-fresh buffer holds the WRONG-topic memories. Gate the join on
|
|
671
|
+
# topical similarity between the buffered query and turn N+1's ACTUAL prompt
|
|
672
|
+
# BEFORE the directive fetch below; a mismatch marks the token consumed (so
|
|
673
|
+
# it is never reconsidered) and falls through to SYNCHRONOUS recall — which
|
|
674
|
+
# fetches correct, current-topic memories AND re-injects directives itself.
|
|
675
|
+
# This is layered ON TOP of the F3 freshness machinery, not a replacement.
|
|
676
|
+
if payload is not None and not _prefetch_topic_matches(prompt, payload.get("query", ""), config):
|
|
677
|
+
if _token is not None:
|
|
678
|
+
_write_consumed_token(session_id, _token)
|
|
679
|
+
debug_log(config, "Prefetch buffer: topic mismatch vs current prompt, falling through to synchronous recall")
|
|
680
|
+
return False
|
|
562
681
|
|
|
563
682
|
# Directives stay on the synchronous, always-fresh path (M3 rule) even
|
|
564
683
|
# in the fast path — fetched here directly, never from the buffer.
|
|
@@ -584,6 +703,12 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
584
703
|
directives_block = None
|
|
585
704
|
|
|
586
705
|
if payload is not None:
|
|
706
|
+
# M4 F3 — record this token as consumed BEFORE emitting, so the next
|
|
707
|
+
# turn's `read_if_fresh` rejects the same buffer as stale unless the
|
|
708
|
+
# producer has since written a strictly-newer sentinel. `_token` is the
|
|
709
|
+
# sentinel token `read_if_fresh` just validated as fresh.
|
|
710
|
+
if _token is not None:
|
|
711
|
+
_write_consumed_token(session_id, _token)
|
|
587
712
|
memories_block = payload.get("context") or ""
|
|
588
713
|
parts = [b for b in (_dir_notice, directives_block, memories_block) if b]
|
|
589
714
|
if not parts:
|
|
@@ -607,11 +732,15 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
607
732
|
_emit_cached_context("\n\n".join([b for b in (_dir_notice, directives_block) if b]))
|
|
608
733
|
return True
|
|
609
734
|
|
|
610
|
-
#
|
|
611
|
-
#
|
|
612
|
-
#
|
|
613
|
-
|
|
614
|
-
|
|
735
|
+
# M4 F4 — nothing fresh, nothing stale, nothing cached (the cold-start /
|
|
736
|
+
# first-turn / no-prior-recall case). Return False so the caller runs the
|
|
737
|
+
# SYNCHRONOUS recall path, which is exactly the path that should run on a
|
|
738
|
+
# turn where no prefetch could possibly exist yet. Emitting a degraded
|
|
739
|
+
# notice and returning True here (the previous behaviour) short-circuited
|
|
740
|
+
# sync recall and handed a fresh/post-restart session ZERO memories plus a
|
|
741
|
+
# degraded banner on its very first turn — the opposite of the docstring's
|
|
742
|
+
# "a miss falls through to sync recall" contract.
|
|
743
|
+
return False
|
|
615
744
|
|
|
616
745
|
|
|
617
746
|
def _emit_directives_only(config: dict, hook_input: dict) -> None:
|
|
@@ -774,6 +903,78 @@ def _overlap_tokens(text) -> set:
|
|
|
774
903
|
return out
|
|
775
904
|
|
|
776
905
|
|
|
906
|
+
_PREFETCH_TOPIC_OVERLAP_DEFAULT = 0.3
|
|
907
|
+
|
|
908
|
+
# #4778 review MAJOR — short-prompt Jaccard floor. Jaccard is probabilistic, not
|
|
909
|
+
# absolute, on SHORT prompts: two ~2-content-token turns that share ONE incidental
|
|
910
|
+
# token while EACH carries a divergent token give 1/3 = 0.333 >= 0.30 and wrongly
|
|
911
|
+
# clear the ratio bar ("call mom" buffer served on a "call ended" turn — the pivot
|
|
912
|
+
# signature). When either token set is small (fewer than ``MIN_SMALL_SET_TOKENS``)
|
|
913
|
+
# such a partial overlap must supply at least ``_PREFETCH_MIN_SMALL_SET_INTERSECTION``
|
|
914
|
+
# shared tokens, not just clear the ratio.
|
|
915
|
+
#
|
|
916
|
+
# The floor fires ONLY on that divergent-on-both-sides shape. It deliberately
|
|
917
|
+
# EXEMPTS full containment (``intersection == min(|A|, |B|)`` — the smaller set is
|
|
918
|
+
# wholly shared: identical or subset/narrowing queries like "decide" vs "decide",
|
|
919
|
+
# or "restart klanker" vs "restart the klanker agent"), which is a legitimate
|
|
920
|
+
# topical match regardless of brevity and must still be served. Long prompts (both
|
|
921
|
+
# sides at or above ``MIN_SMALL_SET_TOKENS``) skip the floor entirely and use the
|
|
922
|
+
# Jaccard ratio exactly as before. The floor only ever ANDs a stricter condition
|
|
923
|
+
# onto the ratio — it can reject, never serve.
|
|
924
|
+
MIN_SMALL_SET_TOKENS = 3
|
|
925
|
+
_PREFETCH_MIN_SMALL_SET_INTERSECTION = 2
|
|
926
|
+
|
|
927
|
+
|
|
928
|
+
def _prefetch_topic_matches(prompt, buffered_query, config) -> bool:
|
|
929
|
+
"""#4778 — True iff turn N+1's ``prompt`` is topically close enough to the
|
|
930
|
+
``buffered_query`` the M4 producer used to build the warm buffer.
|
|
931
|
+
|
|
932
|
+
Jaccard overlap (``|A∩B| / |A∪B|``) on ``_overlap_tokens`` — the same
|
|
933
|
+
stop-word-stripped, digit-preserving, dependency-free tokenizer the
|
|
934
|
+
transcript-fallback keyword match uses (unicode-safe via ``str.isalnum``,
|
|
935
|
+
characterised in ``tests/test_overlap_tokens.py``) — against
|
|
936
|
+
``memoryPrefetchMinTopicOverlap`` (default 0.3). Jaccard, not the overlap
|
|
937
|
+
coefficient: both score 0 on a true pivot (disjoint content tokens), so the
|
|
938
|
+
correctness guarantee is identical, but Jaccard cannot be driven to 1.0 by a
|
|
939
|
+
single incidental shared token in a short prompt — it biases the residual
|
|
940
|
+
error toward a harmless false-MISS (sync recall, slower) rather than a
|
|
941
|
+
false-HIT (wrong-topic memories, the bug).
|
|
942
|
+
|
|
943
|
+
Fail-safe: an empty token set on EITHER side — a contentless prompt, or a
|
|
944
|
+
legacy/torn buffer whose ``query`` is ``""`` — returns False, so the caller
|
|
945
|
+
falls through to synchronous recall. A miss only ever costs latency; it can
|
|
946
|
+
never serve wrong-topic memories. Both sides are stripped of the ``<channel>``
|
|
947
|
+
envelope first, matching the producer's own query derivation.
|
|
948
|
+
"""
|
|
949
|
+
now_tokens = _overlap_tokens(strip_channel_envelope(prompt or ""))
|
|
950
|
+
buf_tokens = _overlap_tokens(strip_channel_envelope(buffered_query or ""))
|
|
951
|
+
if not now_tokens or not buf_tokens:
|
|
952
|
+
return False
|
|
953
|
+
intersection = len(now_tokens & buf_tokens)
|
|
954
|
+
if intersection == 0:
|
|
955
|
+
return False
|
|
956
|
+
# Short-prompt floor (#4778 review MAJOR): when either side is small, a single
|
|
957
|
+
# incidental shared token must not clear the guard on the Jaccard ratio alone.
|
|
958
|
+
# Exempt full containment (intersection == smaller set) — identical/subset
|
|
959
|
+
# queries are a legitimate topical match, never the divergent-on-both-sides
|
|
960
|
+
# pivot this floor targets. ANDed with the ratio check below: stricter, never
|
|
961
|
+
# looser, and long prompts (both sides >= MIN_SMALL_SET_TOKENS) are untouched.
|
|
962
|
+
min_set = min(len(now_tokens), len(buf_tokens))
|
|
963
|
+
if min_set < MIN_SMALL_SET_TOKENS \
|
|
964
|
+
and intersection < min_set \
|
|
965
|
+
and intersection < _PREFETCH_MIN_SMALL_SET_INTERSECTION:
|
|
966
|
+
return False
|
|
967
|
+
union = len(now_tokens | buf_tokens)
|
|
968
|
+
threshold = config.get("memoryPrefetchMinTopicOverlap", _PREFETCH_TOPIC_OVERLAP_DEFAULT)
|
|
969
|
+
try:
|
|
970
|
+
threshold = float(threshold)
|
|
971
|
+
except (TypeError, ValueError):
|
|
972
|
+
threshold = _PREFETCH_TOPIC_OVERLAP_DEFAULT
|
|
973
|
+
if not (0.0 <= threshold <= 1.0):
|
|
974
|
+
threshold = _PREFETCH_TOPIC_OVERLAP_DEFAULT
|
|
975
|
+
return (intersection / union) >= threshold
|
|
976
|
+
|
|
977
|
+
|
|
777
978
|
def _result_final_score(m) -> float:
|
|
778
979
|
"""Return a result's engine relevance score (`scores.final`).
|
|
779
980
|
|
|
@@ -278,7 +278,9 @@ class NoDuplicateInjectionTests(InvalidationBase):
|
|
|
278
278
|
config = self._config(prefetch_enabled=True)
|
|
279
279
|
marker = "- unique-fact-abc123 decided at standup"
|
|
280
280
|
|
|
281
|
-
|
|
281
|
+
# #4778 — on-topic with the `_consume` prompt ("what did we decide") so
|
|
282
|
+
# the topic guard passes and this stays a genuine warm-buffer hit.
|
|
283
|
+
recall_buffer.write_buffer(SESSION, marker, {}, query="what did we decide")
|
|
282
284
|
recall_buffer.write_sentinel(SESSION)
|
|
283
285
|
|
|
284
286
|
# Turn N+1: consumer injects the buffered block once.
|
|
@@ -311,7 +313,10 @@ class ReplyPathDoesNotBlockOnRecallTests(InvalidationBase):
|
|
|
311
313
|
|
|
312
314
|
def test_fresh_buffer_served_without_calling_recall(self):
|
|
313
315
|
config = self._config(prefetch_enabled=True)
|
|
314
|
-
|
|
316
|
+
# #4778 — on-topic with the `_consume` prompt so the topic guard passes;
|
|
317
|
+
# the point of THIS test is that a matched fresh buffer never blocks on a
|
|
318
|
+
# synchronous recall, so the query must clear the guard.
|
|
319
|
+
recall_buffer.write_buffer(SESSION, "- a prefetched memory xyz", {}, query="what did we decide")
|
|
315
320
|
recall_buffer.write_sentinel(SESSION)
|
|
316
321
|
|
|
317
322
|
class _ExplodingRecallClient:
|
|
@@ -147,7 +147,7 @@ class CallOrderTests(PrefetchPipelineBase):
|
|
|
147
147
|
order.append("retain")
|
|
148
148
|
return {"status": "ok"}
|
|
149
149
|
|
|
150
|
-
def _spy_write_buffer(session_id, context, telemetry=None):
|
|
150
|
+
def _spy_write_buffer(session_id, context, telemetry=None, query=None):
|
|
151
151
|
order.append("buffer")
|
|
152
152
|
|
|
153
153
|
def _spy_write_sentinel(session_id):
|
|
@@ -233,14 +233,105 @@ class KillSwitchOffTests(PrefetchPipelineBase):
|
|
|
233
233
|
|
|
234
234
|
|
|
235
235
|
class JunkGateTests(PrefetchPipelineBase):
|
|
236
|
-
def
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
236
|
+
def test_task_notification_in_transcript_skips_before_retain(self):
|
|
237
|
+
# F6 (regression guard): a real Stop-hook input carries NO `prompt`
|
|
238
|
+
# field — only session_id / transcript_path / stop_hook_active — so the
|
|
239
|
+
# junk gate MUST derive from the transcript's last human turn. When that
|
|
240
|
+
# turn is a `<task-notification>` envelope, prefetch skips ENTIRELY:
|
|
241
|
+
# no delta retain, no recall, no buffer. This fails on the pre-fix code,
|
|
242
|
+
# whose `hook_input.get("prompt")` gate is permanently empty on real
|
|
243
|
+
# input and therefore lets retain run on a task-notification turn.
|
|
244
|
+
_write_transcript(
|
|
245
|
+
self.transcript_path,
|
|
246
|
+
[("user", "<task-notification>sub-agent done</task-notification>")],
|
|
247
|
+
)
|
|
248
|
+
real_stop_input = {
|
|
249
|
+
"session_id": SESSION,
|
|
250
|
+
"transcript_path": self.transcript_path,
|
|
251
|
+
"stop_hook_active": True,
|
|
252
|
+
"cwd": "/tmp",
|
|
253
|
+
} # deliberately NO `prompt` / `user_prompt` key — mirrors real input
|
|
254
|
+
retain_spy = mock.Mock(return_value={"status": "ok"})
|
|
255
|
+
with patch("prefetch.retain_module.run_retain", retain_spy):
|
|
256
|
+
wrote = prefetch.run_prefetch(real_stop_input, self._config())
|
|
242
257
|
self.assertFalse(wrote)
|
|
258
|
+
retain_spy.assert_not_called()
|
|
243
259
|
self.assertFalse(os.path.isfile(recall_buffer._buffer_path(SESSION)))
|
|
260
|
+
self.assertFalse(os.path.isfile(recall_buffer._sentinel_path(SESSION)))
|
|
261
|
+
|
|
262
|
+
def test_gate_keys_off_transcript_not_a_phantom_prompt_field(self):
|
|
263
|
+
# The mirror of the above: a genuine human turn in the transcript with
|
|
264
|
+
# a STRAY `prompt="<task-notification>"` on the hook input (the shape the
|
|
265
|
+
# old harness faked). The gate must key off the transcript (a real turn)
|
|
266
|
+
# and PROCEED, proving it no longer reads `hook_input["prompt"]`. Fails
|
|
267
|
+
# on the pre-fix code, which reads the phantom field and wrongly skips.
|
|
268
|
+
_write_transcript(self.transcript_path, [("user", "we shipped the release on friday")])
|
|
269
|
+
client = mock.Mock()
|
|
270
|
+
client.recall.side_effect = lambda *a, **kw: {"results": [
|
|
271
|
+
{"text": "we shipped the release on friday", "type": "fact",
|
|
272
|
+
"mentioned_at": "2026-01-01", "id": "r1", "scores": {"final": 0.9}},
|
|
273
|
+
]}
|
|
274
|
+
hook_input = {
|
|
275
|
+
"session_id": SESSION,
|
|
276
|
+
"transcript_path": self.transcript_path,
|
|
277
|
+
"prompt": "<task-notification>phantom</task-notification>",
|
|
278
|
+
"cwd": "/tmp",
|
|
279
|
+
}
|
|
280
|
+
with patch("prefetch.retain_module.run_retain", return_value={"status": "ok"}), \
|
|
281
|
+
patch("prefetch.HindsightClient", return_value=client), \
|
|
282
|
+
patch("prefetch.get_api_url", return_value="http://fake"):
|
|
283
|
+
wrote = prefetch.run_prefetch(hook_input, self._config())
|
|
284
|
+
self.assertTrue(wrote)
|
|
285
|
+
self.assertTrue(os.path.isfile(recall_buffer._buffer_path(SESSION)))
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
class CurationSharingTests(PrefetchPipelineBase):
|
|
289
|
+
"""F5 — the producer must run recalled candidates through recall's shared
|
|
290
|
+
curation pipeline (demote-drop + `recallMaxMemories` cap) BEFORE buffering,
|
|
291
|
+
so the buffer path cannot inject what the synchronous path would filter."""
|
|
292
|
+
|
|
293
|
+
def _run_with_results(self, results, config):
|
|
294
|
+
_write_transcript(self.transcript_path, [("user", "what do you remember about deploys")])
|
|
295
|
+
client = mock.Mock()
|
|
296
|
+
client.recall.side_effect = lambda *a, **kw: {"results": [dict(m) for m in results]}
|
|
297
|
+
with patch("prefetch.retain_module.run_retain", return_value={"status": "ok"}), \
|
|
298
|
+
patch("prefetch.HindsightClient", return_value=client), \
|
|
299
|
+
patch("prefetch.get_api_url", return_value="http://fake"):
|
|
300
|
+
wrote = prefetch.run_prefetch(self._hook_input("what do you remember about deploys"), config)
|
|
301
|
+
buffered = ""
|
|
302
|
+
if wrote:
|
|
303
|
+
payload = recall_buffer._read_buffer_payload(SESSION) or {}
|
|
304
|
+
buffered = payload.get("context") or ""
|
|
305
|
+
return wrote, buffered
|
|
306
|
+
|
|
307
|
+
def test_demote_tagged_memory_is_dropped_before_buffering(self):
|
|
308
|
+
results = [
|
|
309
|
+
{"text": "keep me visible", "type": "fact", "mentioned_at": "2026-01-01",
|
|
310
|
+
"id": "k1", "scores": {"final": 0.9}},
|
|
311
|
+
{"text": "secret demoted note", "type": "fact", "mentioned_at": "2026-01-01",
|
|
312
|
+
"id": "d1", "scores": {"final": 0.8}, "tags": ["demote-from-recall"]},
|
|
313
|
+
]
|
|
314
|
+
_wrote, buffered = self._run_with_results(results, self._config())
|
|
315
|
+
self.assertIn("keep me visible", buffered)
|
|
316
|
+
self.assertNotIn(
|
|
317
|
+
"secret demoted note", buffered,
|
|
318
|
+
"a demote-from-recall memory must never reach the prefetch buffer",
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
def test_candidate_set_above_cap_is_truncated_before_buffering(self):
|
|
322
|
+
results = [
|
|
323
|
+
{"text": f"candidate {i}", "type": "fact", "mentioned_at": "2026-01-01",
|
|
324
|
+
"id": f"m{i}", "scores": {"final": 1.0 - i * 0.01}}
|
|
325
|
+
for i in range(12)
|
|
326
|
+
]
|
|
327
|
+
config = self._config()
|
|
328
|
+
config["recallMaxMemories"] = 8
|
|
329
|
+
_wrote, buffered = self._run_with_results(results, config)
|
|
330
|
+
injected = sum(1 for i in range(12) if f"candidate {i}" in buffered)
|
|
331
|
+
self.assertEqual(
|
|
332
|
+
injected, 8,
|
|
333
|
+
"the prefetch buffer must honour recallMaxMemories, same as the sync path",
|
|
334
|
+
)
|
|
244
335
|
|
|
245
336
|
|
|
246
337
|
if __name__ == "__main__":
|