switchroom 0.21.18 → 0.21.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -42,6 +42,7 @@ function isMultiAgentEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
42
42
  }
43
43
  import { classifyClaudeError, type OperatorEventKind } from './operator-events.js'
44
44
  import { isLitellmProxyLocal429, isTransientUpstreamSignal } from './model-unavailable.js'
45
+ import { isConnectionDropText } from './connection-drop.js'
45
46
  import { createToolLabelSidecar, type ToolLabelSidecar, type SidecarOptions } from './tool-label-sidecar.js'
46
47
  import { isModelSentinel } from './model-label.js'
47
48
 
@@ -960,6 +961,25 @@ function extractRetryState(obj: Record<string, unknown>): {
960
961
  }
961
962
  }
962
963
 
964
+ /**
965
+ * The OperatorEventKinds on which a connection-drop wording may be flagged as a
966
+ * genuine transport drop. INCLUSION (not exclusion) by design: a positively
967
+ * classified auth/quota/credit/rate-limit/overload wall is NEVER in this set,
968
+ * so a wrapped terminal error whose outer text coincidentally carries a drop
969
+ * wording is never mislabelled a drop. `transport-transient` is the natural
970
+ * mid-stream-abort kind; `unknown-4xx`/`unknown-5xx` is where a drop-worded
971
+ * line lands when `classifyClaudeError` does not recognise its wording (the
972
+ * exact Path-B bug this closes). Mirrors Path A's `transient`/`unknown` gate in
973
+ * `parseLlmError`, keeping the two classifiers in agreement.
974
+ */
975
+ const CONNECTION_DROP_ELIGIBLE_KINDS: ReadonlySet<OperatorEventKind> =
976
+ new Set<OperatorEventKind>(['transport-transient', 'unknown-5xx', 'unknown-4xx'])
977
+
978
+ /** True when `kind` may carry the connection-drop flag AND the text is a drop wording. */
979
+ function isConnectionDrop(kind: OperatorEventKind, scanText: string): boolean {
980
+ return CONNECTION_DROP_ELIGIBLE_KINDS.has(kind) && isConnectionDropText(scanText)
981
+ }
982
+
963
983
  export function detectErrorInTranscriptLine(
964
984
  line: string,
965
985
  ): {
@@ -972,6 +992,15 @@ export function detectErrorInTranscriptLine(
972
992
  * error mid-retry is `transient:true, terminal:false`; the caller
973
993
  * suppresses it (no operator card until the failure is terminal). */
974
994
  terminal: boolean
995
+ /**
996
+ * True when this line is a mid-stream connection / SSE drop (a transport
997
+ * connection loss), per the canonical `isConnectionDropText` matcher, gated
998
+ * to the transport/unknown kinds. Set consistently with `parseLlmError`'s
999
+ * `ParsedLlmError.connectionDrop` so a later PR can gate auto-resume on ONE
1000
+ * reliable discriminator across both classification paths. Classification
1001
+ * only in this PR; no caller acts on it yet.
1002
+ */
1003
+ connectionDrop: boolean
975
1004
  } | null {
976
1005
  if (!line || line.length > 2 * 1024 * 1024) return null
977
1006
  let obj: Record<string, unknown>
@@ -1042,6 +1071,7 @@ export function detectErrorInTranscriptLine(
1042
1071
  detail: text || errStr || 'api error',
1043
1072
  transient: kind === 'rate-limited',
1044
1073
  terminal: true,
1074
+ connectionDrop: isConnectionDrop(kind, `${text}\n${errStr}`),
1045
1075
  }
1046
1076
  }
1047
1077
 
@@ -1089,7 +1119,14 @@ export function detectErrorInTranscriptLine(
1089
1119
  ? retry.retryAttempt >= retry.maxRetries
1090
1120
  : isErrorLine
1091
1121
 
1092
- return { kind, raw, detail, transient, terminal }
1122
+ return {
1123
+ kind,
1124
+ raw,
1125
+ detail,
1126
+ transient,
1127
+ terminal,
1128
+ connectionDrop: isConnectionDrop(kind, `${detail}\n${String(type ?? '')}`),
1129
+ }
1093
1130
  }
1094
1131
 
1095
1132
  function extractDetailMessage(obj: Record<string, unknown> | null): string | null {
@@ -27,6 +27,7 @@ import {
27
27
  } from '../llm-error-present.js'
28
28
  import { truncateDetailPreservingRequestId } from '../raw-error-scrub.js'
29
29
  import { projectTranscriptLine, detectErrorInTranscriptLine } from '../session-tail.js'
30
+ import { isConnectionDropText } from '../connection-drop.js'
30
31
  import { renderOperatorEvent, type OperatorEvent } from '../operator-events.js'
31
32
  import { redact } from '../secret-detect/redact.js'
32
33
 
@@ -137,6 +138,188 @@ describe('parseLlmError — classification table', () => {
137
138
  })
138
139
  })
139
140
 
141
+ // ─── Connection-drop classification (PR 0: unify drop classification) ─────────
142
+ //
143
+ // The two classifiers historically disagreed about a mid-stream connection /
144
+ // SSE drop: Path A (`parseLlmError`) mapped it to a transient/network error,
145
+ // while Path B (`detectErrorInTranscriptLine`) let a drop-worded line whose
146
+ // wording `classifyClaudeError` did not recognise fall through to a generic
147
+ // `unknown-*` terminal. These blocks assert the SAME line is now identifiable
148
+ // as a connection drop on BOTH paths — and, critically, that auth / quota /
149
+ // overload / provider-credit walls are NEVER flagged as drops on either path
150
+ // (the greedy-matcher failure mode that would later cause a wrong auto-resume).
151
+
152
+ const OPENROUTER_402_CREDIT =
153
+ 'litellm.APIError: OpenrouterException - {"error":{"code":402,"message":"Your account or API key has insufficient credits. Add more credits and retry the request.","metadata":{"provider_name":"openrouter"}}}'
154
+
155
+ /** Build the v2.1.x `isApiErrorMessage` synthetic-assistant transcript line. */
156
+ function apiErrorLine(text: string, opts: { error?: string; status?: number } = {}): string {
157
+ return JSON.stringify({
158
+ type: 'assistant',
159
+ message: { role: 'assistant', model: '<synthetic>', content: [{ type: 'text', text }] },
160
+ isApiErrorMessage: true,
161
+ ...(opts.error != null ? { error: opts.error } : { error: '' }),
162
+ ...(opts.status != null ? { apiErrorStatus: opts.status } : {}),
163
+ })
164
+ }
165
+
166
+ describe('connection-drop — identifiable on BOTH classification paths', () => {
167
+ // The canonical mid-stream abort shape: isApiErrorMessage + error:"server_error",
168
+ // NO status. Path A sees the worded string; Path B sees the transcript line.
169
+ it('a "Connection closed mid-response" line is a connection-drop on Path A AND Path B', () => {
170
+ const wording = 'API Error: Connection closed mid-response'
171
+
172
+ // Path A
173
+ const a = parseLlmError(wording)
174
+ expect(a.connectionDrop).toBe(true)
175
+ expect(a.kind).toBe('transient')
176
+ expect(a.source).toBe('network')
177
+
178
+ // Path B
179
+ const b = detectErrorInTranscriptLine(apiErrorLine(wording, { error: 'server_error' }))
180
+ expect(b).not.toBeNull()
181
+ expect(b!.connectionDrop).toBe(true)
182
+ expect(b!.kind).toBe('transport-transient')
183
+ })
184
+
185
+ // The exact Path-B bug: a drop-worded line whose wording `classifyClaudeError`
186
+ // does NOT recognise (no server_error type, no status) fell through to
187
+ // unknown-5xx. It must STILL be flagged a connection-drop.
188
+ it('a "socket hang up" line unknown to classifyClaudeError is still a connection-drop on both paths', () => {
189
+ const wording = 'API Error: socket hang up'
190
+
191
+ const a = parseLlmError(wording)
192
+ expect(a.connectionDrop).toBe(true)
193
+
194
+ const b = detectErrorInTranscriptLine(apiErrorLine(wording))
195
+ expect(b).not.toBeNull()
196
+ expect(b!.kind).toBe('unknown-5xx') // classifyClaudeError does not recognise the wording…
197
+ expect(b!.connectionDrop).toBe(true) // …yet the drop is identified anyway.
198
+ })
199
+
200
+ // The wrapper-error line shape (`type:"error"`, nested error object).
201
+ it('a server_error wrapper line carrying a drop wording is a connection-drop on Path B', () => {
202
+ const line = JSON.stringify({
203
+ type: 'error',
204
+ error: { type: 'server_error', message: 'Connection closed mid-response' },
205
+ })
206
+ const b = detectErrorInTranscriptLine(line)
207
+ expect(b).not.toBeNull()
208
+ expect(b!.kind).toBe('transport-transient')
209
+ expect(b!.connectionDrop).toBe(true)
210
+ })
211
+
212
+ // EPIPE is a genuine drop wording that `detectModelUnavailable` does not
213
+ // classify as network — Path A lands it on `unknown`, but it must still be a
214
+ // connection-drop (proving the discriminator is NOT tied to source:'network').
215
+ it('EPIPE is a connection-drop on Path A even though its kind stays unknown', () => {
216
+ const a = parseLlmError('write EPIPE: half-dead socket')
217
+ expect(a.kind).toBe('unknown')
218
+ expect(a.connectionDrop).toBe(true)
219
+ })
220
+ })
221
+
222
+ describe('connection-drop — negatives (auth / quota / overload / provider-credit are NEVER drops)', () => {
223
+ it('auth is not a connection-drop on either path', () => {
224
+ const a = parseLlmError('authentication_error: OAuth token expired, please refresh')
225
+ expect(a.kind).toBe('auth')
226
+ expect(a.connectionDrop).toBe(false)
227
+
228
+ const b = detectErrorInTranscriptLine(
229
+ apiErrorLine('authentication_error: OAuth token expired', { error: 'authentication_error', status: 401 }),
230
+ )
231
+ expect(b).not.toBeNull()
232
+ expect(b!.connectionDrop).toBe(false)
233
+ })
234
+
235
+ it('quota_wall is not a connection-drop on either path', () => {
236
+ const a = parseLlmError("You've hit your limit · resets 5pm")
237
+ expect(a.kind).toBe('quota_wall')
238
+ expect(a.connectionDrop).toBe(false)
239
+
240
+ const b = detectErrorInTranscriptLine(
241
+ apiErrorLine("You've hit your limit · resets 5pm", { error: 'rate_limit_error', status: 429 }),
242
+ )
243
+ expect(b).not.toBeNull()
244
+ expect(b!.kind).toBe('quota-exhausted')
245
+ expect(b!.connectionDrop).toBe(false)
246
+ })
247
+
248
+ it('overload_529 is not a connection-drop on either path', () => {
249
+ const a = parseLlmError('Overloaded (overloaded_error) — HTTP 529')
250
+ expect(a.kind).toBe('overload_529')
251
+ expect(a.connectionDrop).toBe(false)
252
+
253
+ const b = detectErrorInTranscriptLine(apiErrorLine('Overloaded', { error: 'overloaded_error' }))
254
+ expect(b).not.toBeNull()
255
+ expect(b!.connectionDrop).toBe(false)
256
+ })
257
+
258
+ it('provider_credit is not a connection-drop on either path', () => {
259
+ const a = parseLlmError(OPENROUTER_402_CREDIT)
260
+ expect(a.kind).toBe('provider_credit')
261
+ expect(a.connectionDrop).toBe(false)
262
+
263
+ const b = detectErrorInTranscriptLine(apiErrorLine(OPENROUTER_402_CREDIT, { status: 402 }))
264
+ expect(b).not.toBeNull()
265
+ expect(b!.kind).toBe('provider-credit-exhausted')
266
+ expect(b!.connectionDrop).toBe(false)
267
+ })
268
+
269
+ // The single highest-risk case: a wrapped auth/quota wall whose OUTER text
270
+ // coincidentally carries a drop wording ("fetch failed") must NOT be flagged a
271
+ // drop — the classification wins over the wording, on both paths.
272
+ it('a wrapped auth error whose outer text says "fetch failed" is NOT a connection-drop', () => {
273
+ const a = parseLlmError('authentication_error: invalid api key (underlying: fetch failed)')
274
+ expect(a.kind).toBe('auth')
275
+ expect(a.connectionDrop).toBe(false)
276
+
277
+ const b = detectErrorInTranscriptLine(
278
+ JSON.stringify({
279
+ type: 'error',
280
+ error: { type: 'authentication_error', message: 'invalid api key; underlying: fetch failed' },
281
+ }),
282
+ )
283
+ expect(b).not.toBeNull()
284
+ expect(b!.kind).toBe('credentials-invalid')
285
+ expect(b!.connectionDrop).toBe(false)
286
+ })
287
+ })
288
+
289
+ describe('isConnectionDropText — the canonical matcher is conservative', () => {
290
+ it('matches the canonical drop wordings', () => {
291
+ for (const s of [
292
+ 'socket hang up',
293
+ 'read ECONNRESET',
294
+ 'connect ECONNREFUSED 1.2.3.4:443',
295
+ 'write EPIPE',
296
+ 'fetch failed',
297
+ 'network error',
298
+ 'Connection closed mid-response',
299
+ 'connection lost',
300
+ 'Premature close',
301
+ 'the stream disconnected unexpectedly',
302
+ ]) {
303
+ expect(isConnectionDropText(s)).toBe(true)
304
+ }
305
+ })
306
+
307
+ it('does NOT over-match "upstream"/"downstream"/"terminated" or auth/quota wording', () => {
308
+ for (const s of [
309
+ 'the upstream provider returned a result',
310
+ 'downstream consumer finished',
311
+ 'the worker terminated cleanly',
312
+ 'authentication_error: invalid api key',
313
+ "You've hit your usage limit",
314
+ 'overloaded_error',
315
+ 'insufficient credits',
316
+ '',
317
+ ]) {
318
+ expect(isConnectionDropText(s)).toBe(false)
319
+ }
320
+ })
321
+ })
322
+
140
323
  // ─── local-time render ───────────────────────────────────────────────────────
141
324
 
142
325
  describe('renderLlmError / formatResetLocal — local-time rendering', () => {
@@ -75,8 +75,18 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
75
75
  that could take the Stop hook (and thus the turn) down with it."""
76
76
  session_id = hook_input.get("session_id") or "unknown"
77
77
 
78
- prompt = (hook_input.get("prompt") or hook_input.get("user_prompt") or "").strip()
79
- if prompt.startswith("<task-notification"):
78
+ # F6 — junk gate derived from the TRANSCRIPT's last human turn, not from a
79
+ # `hook_input["prompt"]` field. A Stop hook's input carries only
80
+ # session_id/transcript_path/stop_hook_active — never `prompt`/`user_prompt`
81
+ # — so the old `hook_input.get("prompt")` gate was permanently empty and
82
+ # NEVER fired, meaning prefetch would run on `<task-notification>` turns once
83
+ # lit. `_last_human_prompt` reads the same last-human turn the speculative
84
+ # query is derived from, so the gate now fires on exactly the synthetic
85
+ # follow-up turns `recall.py`'s synchronous gate skips (honouring the same
86
+ # `recallSkipTaskNotification` switch).
87
+ transcript_path = hook_input.get("transcript_path") or ""
88
+ query = _last_human_prompt(transcript_path)
89
+ if config.get("recallSkipTaskNotification", True) and query.startswith("<task-notification"):
80
90
  debug_log(config, "Prefetch: task-notification turn, skipping")
81
91
  return False
82
92
 
@@ -97,9 +107,8 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
97
107
  except Exception as exc: # pragma: no cover - defensive
98
108
  debug_log(config, f"Prefetch: delta retain failed, continuing without it: {exc}")
99
109
 
100
- # Step 2 — speculative recall.
101
- transcript_path = hook_input.get("transcript_path") or ""
102
- query = _last_human_prompt(transcript_path)
110
+ # Step 2 — speculative recall (query derived above from the transcript's
111
+ # last human turn).
103
112
  if not query:
104
113
  debug_log(config, "Prefetch: no usable query, nothing to prefetch")
105
114
  return False
@@ -123,6 +132,25 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
123
132
  debug_log(config, "Prefetch: no candidates, nothing to buffer")
124
133
  return False
125
134
 
135
+ # F5 — CURATE before buffering, through the SAME pipeline the synchronous
136
+ # recall path enforces (demote-drop, relevance sort, score floor, and the
137
+ # `recallMaxMemories` cap). Calling recall's shared helper — rather than
138
+ # `format_memories(results)` on the raw set — is what stops a demoted or
139
+ # uncapped memory from reaching the buffer and bypassing curation the sync
140
+ # path applies (carve §6.1 sharing requirement). Imported lazily so the
141
+ # flag-off no-op in `main()` never pays recall.py's import cost.
142
+ try:
143
+ import recall # noqa: PLC0415 - lazy, kept off the flag-off no-op path
144
+
145
+ results = recall.curate_recall_results(results, config, bank_id)
146
+ except Exception as exc: # pragma: no cover - defensive
147
+ debug_log(config, f"Prefetch: curation failed, skipping buffer: {exc}")
148
+ return False
149
+
150
+ if not results:
151
+ debug_log(config, "Prefetch: all candidates filtered by curation, nothing to buffer")
152
+ return False
153
+
126
154
  from lib.content import format_memories
127
155
 
128
156
  memories_block = format_memories(results)
@@ -92,6 +92,13 @@ _IMPORT_ELAPSED_SECONDS = time.monotonic() - _IMPORT_START_MONOTONIC
92
92
 
93
93
  LAST_RECALL_STATE = "last_recall.json"
94
94
  RECALL_CACHE_STATE = "recall_cache.json"
95
+ # M4 P-REC F3 — the last prefetch-buffer sentinel token this session has
96
+ # already CONSUMED, keyed by session_id. Persisted across turns so a buffer
97
+ # produced at turn N and consumed at N+1 is not re-served as "fresh" at
98
+ # N+2..N+k when no newer sentinel has landed (the stale-buffer-served-as-fresh
99
+ # class). A dict {session_id: token}; capped like the other per-session state.
100
+ PREFETCH_CONSUMED_STATE = "prefetch_consumed.json"
101
+ _PREFETCH_CONSUMED_MAX_SESSIONS = 10000
95
102
 
96
103
  # Switchroom hindsight-leverage A3 — label for the directives fetch slot in the
97
104
  # parallel fan-out. Distinct from any bank_id (banks can't start with "__") so a
@@ -500,12 +507,6 @@ def _emit_cached_context(context: str) -> None:
500
507
  )
501
508
 
502
509
 
503
- _PREFETCH_DEGRADED_NOTICE = (
504
- "⏳ prefetch not ready and no prior recall is cached for this session — "
505
- "proceeding without injected memory this turn."
506
- )
507
-
508
-
509
510
  def stale_recall_notice(memories_context: str) -> str:
510
511
  """Wrap a PRIOR turn's cached memories-only block in an explicit
511
512
  staleness marker for the M4 prefetch-buffer fallback path.
@@ -528,6 +529,100 @@ def stale_recall_notice(memories_context: str) -> str:
528
529
  )
529
530
 
530
531
 
532
+ def _read_consumed_token(session_id: str) -> "int | None":
533
+ """Return the last prefetch sentinel token this session already consumed,
534
+ or None if none is recorded. Failure-tolerant — any read/parse error
535
+ returns None (treat as "nothing consumed yet", the fail-open direction that
536
+ at worst re-serves once, never suppresses forever)."""
537
+ state = read_state(PREFETCH_CONSUMED_STATE, {}) or {}
538
+ if not isinstance(state, dict):
539
+ return None
540
+ token = state.get(session_id)
541
+ return int(token) if isinstance(token, (int, float)) else None
542
+
543
+
544
+ def _write_consumed_token(session_id: str, token: int) -> None:
545
+ """Persist `token` as the last sentinel this session has consumed. Bounded
546
+ to `_PREFETCH_CONSUMED_MAX_SESSIONS` entries. Best-effort — write_state
547
+ never raises."""
548
+ state = read_state(PREFETCH_CONSUMED_STATE, {}) or {}
549
+ if not isinstance(state, dict):
550
+ state = {}
551
+ state[session_id] = int(token)
552
+ if len(state) > _PREFETCH_CONSUMED_MAX_SESSIONS:
553
+ # Drop the numerically-smallest tokens (oldest producers) first.
554
+ for k in sorted(
555
+ state.keys(),
556
+ key=lambda k: state[k] if isinstance(state[k], (int, float)) else 0,
557
+ )[: len(state) - _PREFETCH_CONSUMED_MAX_SESSIONS]:
558
+ state.pop(k, None)
559
+ write_state(PREFETCH_CONSUMED_STATE, state)
560
+
561
+
562
+ def curate_recall_results(results, config, bank_id):
563
+ """M4 F5 — the SHARED recall curation pipeline, so the async prefetch
564
+ producer (`prefetch.py`) can never drift from the guarantees the
565
+ synchronous recall path enforces.
566
+
567
+ Applies, in order, the exact same module-level primitives the synchronous
568
+ path composes inline: drop demote-tagged memories (`_is_demoted_memory`),
569
+ sort by engine relevance (`_sort_by_final_score`), apply the optional
570
+ absolute score floor when it is configured for ALL turns
571
+ (`_filter_by_min_score`; the `degraded`-scope floor is a synchronous-path
572
+ concept the speculative single-bank prefetch cannot evaluate, so it is
573
+ honoured here only when `recallMinScoreScope` is `all`, matching what the
574
+ sync path does on a healthy own-bank turn), and finally the
575
+ `recallMaxMemories` head cap with per-bank slot reservation
576
+ (`_reserve_bank_slots`).
577
+
578
+ Returns the curated result list. Never mutates the caller's list in a way
579
+ that changes its length before returning (it rebinds internally). Pure —
580
+ no I/O, no network — so it is safe to call from the producer hook.
581
+ """
582
+ if not results:
583
+ return results
584
+
585
+ # 1) demote-from-recall drop (Switchroom #432 4.4) — before any cap so the
586
+ # cap fills with non-demoted hits.
587
+ results = [m for m in results if not _is_demoted_memory(m)]
588
+
589
+ # 2) relevance sort (Phase-1 bank-starvation fix) — in place on our copy.
590
+ _sort_by_final_score(results)
591
+
592
+ # 3) optional absolute score floor (#3837). The sync path binds this on
593
+ # `all` scope always, and on `degraded` scope only when the own-bank
594
+ # read degraded. A prefetch is a healthy speculative read (no degraded
595
+ # signal), so we bind only the `all` scope here — identical to the sync
596
+ # path's decision on a non-degraded turn.
597
+ min_score_floor = config.get("recallMinScore", 0.0)
598
+ if isinstance(min_score_floor, bool) or not isinstance(min_score_floor, (int, float)):
599
+ min_score_floor = 0.0
600
+ min_score_floor = float(min_score_floor)
601
+ min_score_scope = config.get("recallMinScoreScope", "degraded")
602
+ if min_score_floor > 0 and min_score_scope == "all":
603
+ results, _dropped = _filter_by_min_score(results, min_score_floor)
604
+
605
+ # 4) recallMaxMemories head cap + per-bank slot reservation. `_reserve_
606
+ # bank_slots` is a passthrough when cap<=0 or the set already fits, and
607
+ # with the fleet-default 0/0 floors on a single-bank prefetch it reduces
608
+ # to a plain head-slice — the same slice the sync path takes.
609
+ recall_max_memories = config.get("recallMaxMemories", 0)
610
+ if (
611
+ isinstance(recall_max_memories, int)
612
+ and recall_max_memories > 0
613
+ and len(results) > recall_max_memories
614
+ ):
615
+ results, _own, _add = _reserve_bank_slots(
616
+ results,
617
+ recall_max_memories,
618
+ bank_id,
619
+ config.get("recallOwnBankMinSlots", 0),
620
+ config.get("recallAdditionalBankMinSlots", 0),
621
+ )
622
+
623
+ return results
624
+
625
+
531
626
  def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool:
532
627
  """M4 P-REC Fix C (consumer side) — join the Stop-hook producer's
533
628
  prefetch buffer for this session instead of running recall
@@ -535,10 +630,12 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
535
630
 
536
631
  Gated entirely by `config.get("memoryPrefetchEnabled", False)` at the
537
632
  caller; this function assumes the flag is already on. Returns True iff
538
- it emitted an `additionalContext` payload (fresh hit, stale fallback, or
539
- the explicit degraded notice) and the caller should return without
540
- running the synchronous path. Returns False on a clean no-op miss (flag
541
- effectively off / nothing to say) so the caller falls through.
633
+ it emitted an `additionalContext` payload (fresh buffer hit, stale
634
+ fallback, or a directives-only block) and the caller should return
635
+ without running the synchronous path. Returns False on a miss with
636
+ nothing to serve — cold start, no fresh buffer, no stale prior recall,
637
+ no directives — so the caller falls through to SYNCHRONOUS recall (M4
638
+ F4: the first turn of a fresh/post-restart session must still recall).
542
639
 
543
640
  Never raises past this function's own boundary in normal operation —
544
641
  every internal step is wrapped so a bug here degrades to "fall through
@@ -547,6 +644,13 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
547
644
  """
548
645
  session_id = hook_input.get("session_id") or "unknown"
549
646
 
647
+ # M4 F3 — the sentinel token this session has ALREADY consumed on a prior
648
+ # turn. Threaded into both `poll_for_sentinel` and `read_if_fresh` so a
649
+ # buffer produced at turn N and consumed at N+1 is treated as STALE (not
650
+ # re-served as fresh) at N+2..N+k until a strictly-newer sentinel lands.
651
+ # None on a cold session (nothing consumed yet).
652
+ last_consumed = _read_consumed_token(session_id)
653
+
550
654
  # Cold-start short-circuit (red-team MAJOR finding): if this session has
551
655
  # NEVER produced a sentinel, polling the full cap on every single
552
656
  # session-open turn would cost ~the poll cap on every fresh session —
@@ -556,9 +660,9 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
556
660
  debug_log(config, "Prefetch buffer: no sentinel ever written for this session, cold-start skip")
557
661
  else:
558
662
  cap_ms = int(config.get("memoryPrefetchPollCapMs", 400))
559
- recall_buffer.poll_for_sentinel(session_id, last_consumed_token=None, cap_ms=cap_ms)
663
+ recall_buffer.poll_for_sentinel(session_id, last_consumed_token=last_consumed, cap_ms=cap_ms)
560
664
 
561
- payload, _token = recall_buffer.read_if_fresh(session_id, last_consumed_token=None)
665
+ payload, _token = recall_buffer.read_if_fresh(session_id, last_consumed_token=last_consumed)
562
666
 
563
667
  # Directives stay on the synchronous, always-fresh path (M3 rule) even
564
668
  # in the fast path — fetched here directly, never from the buffer.
@@ -584,6 +688,12 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
584
688
  directives_block = None
585
689
 
586
690
  if payload is not None:
691
+ # M4 F3 — record this token as consumed BEFORE emitting, so the next
692
+ # turn's `read_if_fresh` rejects the same buffer as stale unless the
693
+ # producer has since written a strictly-newer sentinel. `_token` is the
694
+ # sentinel token `read_if_fresh` just validated as fresh.
695
+ if _token is not None:
696
+ _write_consumed_token(session_id, _token)
587
697
  memories_block = payload.get("context") or ""
588
698
  parts = [b for b in (_dir_notice, directives_block, memories_block) if b]
589
699
  if not parts:
@@ -607,11 +717,15 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
607
717
  _emit_cached_context("\n\n".join([b for b in (_dir_notice, directives_block) if b]))
608
718
  return True
609
719
 
610
- # Nothing fresh, nothing stale, nothing cached — say so explicitly
611
- # rather than silently emitting no context (so a degraded turn is
612
- # legible, matching the #3619 degraded-disclosure precedent).
613
- _emit_cached_context(_PREFETCH_DEGRADED_NOTICE)
614
- return True
720
+ # M4 F4 — nothing fresh, nothing stale, nothing cached (the cold-start /
721
+ # first-turn / no-prior-recall case). Return False so the caller runs the
722
+ # SYNCHRONOUS recall path, which is exactly the path that should run on a
723
+ # turn where no prefetch could possibly exist yet. Emitting a degraded
724
+ # notice and returning True here (the previous behaviour) short-circuited
725
+ # sync recall and handed a fresh/post-restart session ZERO memories plus a
726
+ # degraded banner on its very first turn — the opposite of the docstring's
727
+ # "a miss falls through to sync recall" contract.
728
+ return False
615
729
 
616
730
 
617
731
  def _emit_directives_only(config: dict, hook_input: dict) -> None:
@@ -233,14 +233,105 @@ class KillSwitchOffTests(PrefetchPipelineBase):
233
233
 
234
234
 
235
235
  class JunkGateTests(PrefetchPipelineBase):
236
- def test_task_notification_turn_is_skipped(self):
237
- _write_transcript(self.transcript_path, [("user", "irrelevant")])
238
- with patch("prefetch.retain_module.run_retain", side_effect=AssertionError("must not retain a task-notification turn")):
239
- wrote = prefetch.run_prefetch(
240
- self._hook_input("<task-notification>done</task-notification>"), self._config()
241
- )
236
+ def test_task_notification_in_transcript_skips_before_retain(self):
237
+ # F6 (regression guard): a real Stop-hook input carries NO `prompt`
238
+ # field — only session_id / transcript_path / stop_hook_active — so the
239
+ # junk gate MUST derive from the transcript's last human turn. When that
240
+ # turn is a `<task-notification>` envelope, prefetch skips ENTIRELY:
241
+ # no delta retain, no recall, no buffer. This fails on the pre-fix code,
242
+ # whose `hook_input.get("prompt")` gate is permanently empty on real
243
+ # input and therefore lets retain run on a task-notification turn.
244
+ _write_transcript(
245
+ self.transcript_path,
246
+ [("user", "<task-notification>sub-agent done</task-notification>")],
247
+ )
248
+ real_stop_input = {
249
+ "session_id": SESSION,
250
+ "transcript_path": self.transcript_path,
251
+ "stop_hook_active": True,
252
+ "cwd": "/tmp",
253
+ } # deliberately NO `prompt` / `user_prompt` key — mirrors real input
254
+ retain_spy = mock.Mock(return_value={"status": "ok"})
255
+ with patch("prefetch.retain_module.run_retain", retain_spy):
256
+ wrote = prefetch.run_prefetch(real_stop_input, self._config())
242
257
  self.assertFalse(wrote)
258
+ retain_spy.assert_not_called()
243
259
  self.assertFalse(os.path.isfile(recall_buffer._buffer_path(SESSION)))
260
+ self.assertFalse(os.path.isfile(recall_buffer._sentinel_path(SESSION)))
261
+
262
+ def test_gate_keys_off_transcript_not_a_phantom_prompt_field(self):
263
+ # The mirror of the above: a genuine human turn in the transcript with
264
+ # a STRAY `prompt="<task-notification>"` on the hook input (the shape the
265
+ # old harness faked). The gate must key off the transcript (a real turn)
266
+ # and PROCEED, proving it no longer reads `hook_input["prompt"]`. Fails
267
+ # on the pre-fix code, which reads the phantom field and wrongly skips.
268
+ _write_transcript(self.transcript_path, [("user", "we shipped the release on friday")])
269
+ client = mock.Mock()
270
+ client.recall.side_effect = lambda *a, **kw: {"results": [
271
+ {"text": "we shipped the release on friday", "type": "fact",
272
+ "mentioned_at": "2026-01-01", "id": "r1", "scores": {"final": 0.9}},
273
+ ]}
274
+ hook_input = {
275
+ "session_id": SESSION,
276
+ "transcript_path": self.transcript_path,
277
+ "prompt": "<task-notification>phantom</task-notification>",
278
+ "cwd": "/tmp",
279
+ }
280
+ with patch("prefetch.retain_module.run_retain", return_value={"status": "ok"}), \
281
+ patch("prefetch.HindsightClient", return_value=client), \
282
+ patch("prefetch.get_api_url", return_value="http://fake"):
283
+ wrote = prefetch.run_prefetch(hook_input, self._config())
284
+ self.assertTrue(wrote)
285
+ self.assertTrue(os.path.isfile(recall_buffer._buffer_path(SESSION)))
286
+
287
+
288
+ class CurationSharingTests(PrefetchPipelineBase):
289
+ """F5 — the producer must run recalled candidates through recall's shared
290
+ curation pipeline (demote-drop + `recallMaxMemories` cap) BEFORE buffering,
291
+ so the buffer path cannot inject what the synchronous path would filter."""
292
+
293
+ def _run_with_results(self, results, config):
294
+ _write_transcript(self.transcript_path, [("user", "what do you remember about deploys")])
295
+ client = mock.Mock()
296
+ client.recall.side_effect = lambda *a, **kw: {"results": [dict(m) for m in results]}
297
+ with patch("prefetch.retain_module.run_retain", return_value={"status": "ok"}), \
298
+ patch("prefetch.HindsightClient", return_value=client), \
299
+ patch("prefetch.get_api_url", return_value="http://fake"):
300
+ wrote = prefetch.run_prefetch(self._hook_input("what do you remember about deploys"), config)
301
+ buffered = ""
302
+ if wrote:
303
+ payload = recall_buffer._read_buffer_payload(SESSION) or {}
304
+ buffered = payload.get("context") or ""
305
+ return wrote, buffered
306
+
307
+ def test_demote_tagged_memory_is_dropped_before_buffering(self):
308
+ results = [
309
+ {"text": "keep me visible", "type": "fact", "mentioned_at": "2026-01-01",
310
+ "id": "k1", "scores": {"final": 0.9}},
311
+ {"text": "secret demoted note", "type": "fact", "mentioned_at": "2026-01-01",
312
+ "id": "d1", "scores": {"final": 0.8}, "tags": ["demote-from-recall"]},
313
+ ]
314
+ _wrote, buffered = self._run_with_results(results, self._config())
315
+ self.assertIn("keep me visible", buffered)
316
+ self.assertNotIn(
317
+ "secret demoted note", buffered,
318
+ "a demote-from-recall memory must never reach the prefetch buffer",
319
+ )
320
+
321
+ def test_candidate_set_above_cap_is_truncated_before_buffering(self):
322
+ results = [
323
+ {"text": f"candidate {i}", "type": "fact", "mentioned_at": "2026-01-01",
324
+ "id": f"m{i}", "scores": {"final": 1.0 - i * 0.01}}
325
+ for i in range(12)
326
+ ]
327
+ config = self._config()
328
+ config["recallMaxMemories"] = 8
329
+ _wrote, buffered = self._run_with_results(results, config)
330
+ injected = sum(1 for i in range(12) if f"candidate {i}" in buffered)
331
+ self.assertEqual(
332
+ injected, 8,
333
+ "the prefetch buffer must honour recallMaxMemories, same as the sync path",
334
+ )
244
335
 
245
336
 
246
337
  if __name__ == "__main__":