switchroom 0.21.18 → 0.21.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17278,6 +17278,34 @@ var init_operator_events = __esm(() => {
17278
17278
  ]);
17279
17279
  });
17280
17280
 
17281
+ // connection-drop.ts
17282
+ function isConnectionDropText(text) {
17283
+ if (typeof text !== "string" || text.length === 0)
17284
+ return false;
17285
+ const lower = text.toLowerCase();
17286
+ return CONNECTION_DROP_SIGNALS.some((s) => lower.includes(s));
17287
+ }
17288
+ var CONNECTION_DROP_SIGNALS;
17289
+ var init_connection_drop = __esm(() => {
17290
+ CONNECTION_DROP_SIGNALS = [
17291
+ "socket hang up",
17292
+ "econnreset",
17293
+ "econnrefused",
17294
+ "etimedout",
17295
+ "epipe",
17296
+ "fetch failed",
17297
+ "network error",
17298
+ "connection refused",
17299
+ "connection reset",
17300
+ "connection closed",
17301
+ "connection lost",
17302
+ "mid-response",
17303
+ "premature close",
17304
+ "stream disconnected",
17305
+ "stream closed"
17306
+ ];
17307
+ });
17308
+
17281
17309
  // tool-label-sidecar.ts
17282
17310
  import { existsSync as existsSync3, readFileSync as readFileSync2, statSync as statSync3 } from "node:fs";
17283
17311
  import { join as join3 } from "node:path";
@@ -17761,6 +17789,9 @@ function extractRetryState(obj) {
17761
17789
  maxRetries: typeof obj.maxRetries === "number" ? obj.maxRetries : null
17762
17790
  };
17763
17791
  }
17792
+ function isConnectionDrop(kind, scanText) {
17793
+ return CONNECTION_DROP_ELIGIBLE_KINDS.has(kind) && isConnectionDropText(scanText);
17794
+ }
17764
17795
  function detectErrorInTranscriptLine(line) {
17765
17796
  if (!line || line.length > 2 * 1024 * 1024)
17766
17797
  return null;
@@ -17785,7 +17816,9 @@ ${errStr}`) ? "rate-limited" : "quota-exhausted" : classifyClaudeError({ type: e
17785
17816
  raw: obj,
17786
17817
  detail: text || errStr || "api error",
17787
17818
  transient: kind2 === "rate-limited",
17788
- terminal: true
17819
+ terminal: true,
17820
+ connectionDrop: isConnectionDrop(kind2, `${text}
17821
+ ${errStr}`)
17789
17822
  };
17790
17823
  }
17791
17824
  const isErrorLine = type === "api_error" || type === "error";
@@ -17798,7 +17831,15 @@ ${errStr}`) ? "rate-limited" : "quota-exhausted" : classifyClaudeError({ type: e
17798
17831
  const transient = kind === "rate-limited" || kind === "transport-transient";
17799
17832
  const retry = extractRetryState(obj);
17800
17833
  const terminal = !transient ? true : retry.retryAttempt != null && retry.maxRetries != null ? retry.retryAttempt >= retry.maxRetries : isErrorLine;
17801
- return { kind, raw, detail, transient, terminal };
17834
+ return {
17835
+ kind,
17836
+ raw,
17837
+ detail,
17838
+ transient,
17839
+ terminal,
17840
+ connectionDrop: isConnectionDrop(kind, `${detail}
17841
+ ${String(type ?? "")}`)
17842
+ };
17802
17843
  }
17803
17844
  function extractDetailMessage(obj) {
17804
17845
  if (!obj)
@@ -18234,13 +18275,15 @@ function startSessionTail(config2) {
18234
18275
  }
18235
18276
  };
18236
18277
  }
18237
- var MAX_JSONL_LINE_BYTES, MAX_ERROR_TEXT_CHARS = 500;
18278
+ var MAX_JSONL_LINE_BYTES, MAX_ERROR_TEXT_CHARS = 500, CONNECTION_DROP_ELIGIBLE_KINDS;
18238
18279
  var init_session_tail = __esm(() => {
18239
18280
  init_operator_events();
18240
18281
  init_model_unavailable();
18282
+ init_connection_drop();
18241
18283
  init_tool_label_sidecar();
18242
18284
  init_model_label();
18243
18285
  MAX_JSONL_LINE_BYTES = 2 * 1024 * 1024;
18286
+ CONNECTION_DROP_ELIGIBLE_KINDS = new Set(["transport-transient", "unknown-5xx", "unknown-4xx"]);
18244
18287
  });
18245
18288
 
18246
18289
  // ../node_modules/.bun/@xterm+headless@6.0.0/node_modules/@xterm/headless/lib-headless/xterm-headless.js
@@ -41,6 +41,7 @@ import {
41
41
  type ProviderCreditEntry,
42
42
  } from './provider-credit.js'
43
43
  import { classifyClaudeError } from './operator-events.js'
44
+ import { isConnectionDropText } from './connection-drop.js'
44
45
  import { stripRawErrorBytes, extractRequestId } from './raw-error-scrub.js'
45
46
  import { fmtLocalClock, tzAbbrev } from './shared/local-time.js'
46
47
 
@@ -94,6 +95,18 @@ export interface ParsedLlmError {
94
95
  */
95
96
  providerId?: string
96
97
  source: LlmErrorSource
98
+ /**
99
+ * True when this error is a mid-stream connection / SSE drop (a transport
100
+ * connection loss), per the canonical {@link isConnectionDropText} matcher.
101
+ * Set consistently on BOTH classification paths (here and
102
+ * `detectErrorInTranscriptLine`) so a later PR can gate auto-resume on ONE
103
+ * reliable discriminator. Only ever true for the `transient`/`unknown`
104
+ * families — never for a positively-identified auth/quota/overload/credit
105
+ * wall (guarded below), so it can never trigger a wrong auto-resume of a
106
+ * genuinely terminal error. Classification only in this PR; no consumer acts
107
+ * on it yet.
108
+ */
109
+ connectionDrop: boolean
97
110
  /** True when the harness is still retrying this error internally (mid-retry). */
98
111
  autoRetrying: boolean
99
112
  /** True when the failure is final (NOT an in-flight retry). */
@@ -169,6 +182,17 @@ export function parseLlmError(
169
182
  }
170
183
  }
171
184
 
185
+ // Connection-drop discriminator. Gated to the transient/unknown families so a
186
+ // positively-classified auth/quota/overload/credit wall is NEVER flagged as a
187
+ // drop, even if its raw text coincidentally carries a drop wording (e.g. a
188
+ // LiteLLM-wrapped 401 whose outer text says "fetch failed"). The gate mirrors
189
+ // Path B's inclusion set (`transport-transient`/`unknown-*`), keeping the two
190
+ // classifiers in agreement. `source==='network'` is not required: a drop
191
+ // wording that `detectModelUnavailable` does not recognise (e.g. `EPIPE`)
192
+ // lands on `unknown` here, and must still be identifiable as a drop.
193
+ const connectionDrop =
194
+ (kind === 'transient' || kind === 'unknown') && isConnectionDropText(text)
195
+
172
196
  return {
173
197
  kind,
174
198
  coreText: buildCoreText(kind, source),
@@ -178,6 +202,7 @@ export function parseLlmError(
178
202
  ...(requestId != null ? { requestId } : {}),
179
203
  ...(providerId != null ? { providerId } : {}),
180
204
  source,
205
+ connectionDrop,
181
206
  autoRetrying,
182
207
  terminal,
183
208
  }
@@ -42,6 +42,7 @@ function isMultiAgentEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
42
42
  }
43
43
  import { classifyClaudeError, type OperatorEventKind } from './operator-events.js'
44
44
  import { isLitellmProxyLocal429, isTransientUpstreamSignal } from './model-unavailable.js'
45
+ import { isConnectionDropText } from './connection-drop.js'
45
46
  import { createToolLabelSidecar, type ToolLabelSidecar, type SidecarOptions } from './tool-label-sidecar.js'
46
47
  import { isModelSentinel } from './model-label.js'
47
48
 
@@ -960,6 +961,25 @@ function extractRetryState(obj: Record<string, unknown>): {
960
961
  }
961
962
  }
962
963
 
964
+ /**
965
+ * The OperatorEventKinds on which a connection-drop wording may be flagged as a
966
+ * genuine transport drop. INCLUSION (not exclusion) by design: a positively
967
+ * classified auth/quota/credit/rate-limit/overload wall is NEVER in this set,
968
+ * so a wrapped terminal error whose outer text coincidentally carries a drop
969
+ * wording is never mislabelled a drop. `transport-transient` is the natural
970
+ * mid-stream-abort kind; `unknown-4xx`/`unknown-5xx` is where a drop-worded
971
+ * line lands when `classifyClaudeError` does not recognise its wording (the
972
+ * exact Path-B bug this closes). Mirrors Path A's `transient`/`unknown` gate in
973
+ * `parseLlmError`, keeping the two classifiers in agreement.
974
+ */
975
+ const CONNECTION_DROP_ELIGIBLE_KINDS: ReadonlySet<OperatorEventKind> =
976
+ new Set<OperatorEventKind>(['transport-transient', 'unknown-5xx', 'unknown-4xx'])
977
+
978
+ /** True when `kind` may carry the connection-drop flag AND the text is a drop wording. */
979
+ function isConnectionDrop(kind: OperatorEventKind, scanText: string): boolean {
980
+ return CONNECTION_DROP_ELIGIBLE_KINDS.has(kind) && isConnectionDropText(scanText)
981
+ }
982
+
963
983
  export function detectErrorInTranscriptLine(
964
984
  line: string,
965
985
  ): {
@@ -972,6 +992,15 @@ export function detectErrorInTranscriptLine(
972
992
  * error mid-retry is `transient:true, terminal:false`; the caller
973
993
  * suppresses it (no operator card until the failure is terminal). */
974
994
  terminal: boolean
995
+ /**
996
+ * True when this line is a mid-stream connection / SSE drop (a transport
997
+ * connection loss), per the canonical `isConnectionDropText` matcher, gated
998
+ * to the transport/unknown kinds. Set consistently with `parseLlmError`'s
999
+ * `ParsedLlmError.connectionDrop` so a later PR can gate auto-resume on ONE
1000
+ * reliable discriminator across both classification paths. Classification
1001
+ * only in this PR; no caller acts on it yet.
1002
+ */
1003
+ connectionDrop: boolean
975
1004
  } | null {
976
1005
  if (!line || line.length > 2 * 1024 * 1024) return null
977
1006
  let obj: Record<string, unknown>
@@ -1042,6 +1071,7 @@ export function detectErrorInTranscriptLine(
1042
1071
  detail: text || errStr || 'api error',
1043
1072
  transient: kind === 'rate-limited',
1044
1073
  terminal: true,
1074
+ connectionDrop: isConnectionDrop(kind, `${text}\n${errStr}`),
1045
1075
  }
1046
1076
  }
1047
1077
 
@@ -1089,7 +1119,14 @@ export function detectErrorInTranscriptLine(
1089
1119
  ? retry.retryAttempt >= retry.maxRetries
1090
1120
  : isErrorLine
1091
1121
 
1092
- return { kind, raw, detail, transient, terminal }
1122
+ return {
1123
+ kind,
1124
+ raw,
1125
+ detail,
1126
+ transient,
1127
+ terminal,
1128
+ connectionDrop: isConnectionDrop(kind, `${detail}\n${String(type ?? '')}`),
1129
+ }
1093
1130
  }
1094
1131
 
1095
1132
  function extractDetailMessage(obj: Record<string, unknown> | null): string | null {
@@ -27,6 +27,7 @@ import {
27
27
  } from '../llm-error-present.js'
28
28
  import { truncateDetailPreservingRequestId } from '../raw-error-scrub.js'
29
29
  import { projectTranscriptLine, detectErrorInTranscriptLine } from '../session-tail.js'
30
+ import { isConnectionDropText } from '../connection-drop.js'
30
31
  import { renderOperatorEvent, type OperatorEvent } from '../operator-events.js'
31
32
  import { redact } from '../secret-detect/redact.js'
32
33
 
@@ -137,6 +138,188 @@ describe('parseLlmError — classification table', () => {
137
138
  })
138
139
  })
139
140
 
141
+ // ─── Connection-drop classification (PR 0: unify drop classification) ─────────
142
+ //
143
+ // The two classifiers historically disagreed about a mid-stream connection /
144
+ // SSE drop: Path A (`parseLlmError`) mapped it to a transient/network error,
145
+ // while Path B (`detectErrorInTranscriptLine`) let a drop-worded line whose
146
+ // wording `classifyClaudeError` did not recognise fall through to a generic
147
+ // `unknown-*` terminal. These blocks assert the SAME line is now identifiable
148
+ // as a connection drop on BOTH paths — and, critically, that auth / quota /
149
+ // overload / provider-credit walls are NEVER flagged as drops on either path
150
+ // (the greedy-matcher failure mode that would later cause a wrong auto-resume).
151
+
152
+ const OPENROUTER_402_CREDIT =
153
+ 'litellm.APIError: OpenrouterException - {"error":{"code":402,"message":"Your account or API key has insufficient credits. Add more credits and retry the request.","metadata":{"provider_name":"openrouter"}}}'
154
+
155
+ /** Build the v2.1.x `isApiErrorMessage` synthetic-assistant transcript line. */
156
+ function apiErrorLine(text: string, opts: { error?: string; status?: number } = {}): string {
157
+ return JSON.stringify({
158
+ type: 'assistant',
159
+ message: { role: 'assistant', model: '<synthetic>', content: [{ type: 'text', text }] },
160
+ isApiErrorMessage: true,
161
+ ...(opts.error != null ? { error: opts.error } : { error: '' }),
162
+ ...(opts.status != null ? { apiErrorStatus: opts.status } : {}),
163
+ })
164
+ }
165
+
166
+ describe('connection-drop — identifiable on BOTH classification paths', () => {
167
+ // The canonical mid-stream abort shape: isApiErrorMessage + error:"server_error",
168
+ // NO status. Path A sees the worded string; Path B sees the transcript line.
169
+ it('a "Connection closed mid-response" line is a connection-drop on Path A AND Path B', () => {
170
+ const wording = 'API Error: Connection closed mid-response'
171
+
172
+ // Path A
173
+ const a = parseLlmError(wording)
174
+ expect(a.connectionDrop).toBe(true)
175
+ expect(a.kind).toBe('transient')
176
+ expect(a.source).toBe('network')
177
+
178
+ // Path B
179
+ const b = detectErrorInTranscriptLine(apiErrorLine(wording, { error: 'server_error' }))
180
+ expect(b).not.toBeNull()
181
+ expect(b!.connectionDrop).toBe(true)
182
+ expect(b!.kind).toBe('transport-transient')
183
+ })
184
+
185
+ // The exact Path-B bug: a drop-worded line whose wording `classifyClaudeError`
186
+ // does NOT recognise (no server_error type, no status) fell through to
187
+ // unknown-5xx. It must STILL be flagged a connection-drop.
188
+ it('a "socket hang up" line unknown to classifyClaudeError is still a connection-drop on both paths', () => {
189
+ const wording = 'API Error: socket hang up'
190
+
191
+ const a = parseLlmError(wording)
192
+ expect(a.connectionDrop).toBe(true)
193
+
194
+ const b = detectErrorInTranscriptLine(apiErrorLine(wording))
195
+ expect(b).not.toBeNull()
196
+ expect(b!.kind).toBe('unknown-5xx') // classifyClaudeError does not recognise the wording…
197
+ expect(b!.connectionDrop).toBe(true) // …yet the drop is identified anyway.
198
+ })
199
+
200
+ // The wrapper-error line shape (`type:"error"`, nested error object).
201
+ it('a server_error wrapper line carrying a drop wording is a connection-drop on Path B', () => {
202
+ const line = JSON.stringify({
203
+ type: 'error',
204
+ error: { type: 'server_error', message: 'Connection closed mid-response' },
205
+ })
206
+ const b = detectErrorInTranscriptLine(line)
207
+ expect(b).not.toBeNull()
208
+ expect(b!.kind).toBe('transport-transient')
209
+ expect(b!.connectionDrop).toBe(true)
210
+ })
211
+
212
+ // EPIPE is a genuine drop wording that `detectModelUnavailable` does not
213
+ // classify as network — Path A lands it on `unknown`, but it must still be a
214
+ // connection-drop (proving the discriminator is NOT tied to source:'network').
215
+ it('EPIPE is a connection-drop on Path A even though its kind stays unknown', () => {
216
+ const a = parseLlmError('write EPIPE: half-dead socket')
217
+ expect(a.kind).toBe('unknown')
218
+ expect(a.connectionDrop).toBe(true)
219
+ })
220
+ })
221
+
222
+ describe('connection-drop — negatives (auth / quota / overload / provider-credit are NEVER drops)', () => {
223
+ it('auth is not a connection-drop on either path', () => {
224
+ const a = parseLlmError('authentication_error: OAuth token expired, please refresh')
225
+ expect(a.kind).toBe('auth')
226
+ expect(a.connectionDrop).toBe(false)
227
+
228
+ const b = detectErrorInTranscriptLine(
229
+ apiErrorLine('authentication_error: OAuth token expired', { error: 'authentication_error', status: 401 }),
230
+ )
231
+ expect(b).not.toBeNull()
232
+ expect(b!.connectionDrop).toBe(false)
233
+ })
234
+
235
+ it('quota_wall is not a connection-drop on either path', () => {
236
+ const a = parseLlmError("You've hit your limit · resets 5pm")
237
+ expect(a.kind).toBe('quota_wall')
238
+ expect(a.connectionDrop).toBe(false)
239
+
240
+ const b = detectErrorInTranscriptLine(
241
+ apiErrorLine("You've hit your limit · resets 5pm", { error: 'rate_limit_error', status: 429 }),
242
+ )
243
+ expect(b).not.toBeNull()
244
+ expect(b!.kind).toBe('quota-exhausted')
245
+ expect(b!.connectionDrop).toBe(false)
246
+ })
247
+
248
+ it('overload_529 is not a connection-drop on either path', () => {
249
+ const a = parseLlmError('Overloaded (overloaded_error) — HTTP 529')
250
+ expect(a.kind).toBe('overload_529')
251
+ expect(a.connectionDrop).toBe(false)
252
+
253
+ const b = detectErrorInTranscriptLine(apiErrorLine('Overloaded', { error: 'overloaded_error' }))
254
+ expect(b).not.toBeNull()
255
+ expect(b!.connectionDrop).toBe(false)
256
+ })
257
+
258
+ it('provider_credit is not a connection-drop on either path', () => {
259
+ const a = parseLlmError(OPENROUTER_402_CREDIT)
260
+ expect(a.kind).toBe('provider_credit')
261
+ expect(a.connectionDrop).toBe(false)
262
+
263
+ const b = detectErrorInTranscriptLine(apiErrorLine(OPENROUTER_402_CREDIT, { status: 402 }))
264
+ expect(b).not.toBeNull()
265
+ expect(b!.kind).toBe('provider-credit-exhausted')
266
+ expect(b!.connectionDrop).toBe(false)
267
+ })
268
+
269
+ // The single highest-risk case: a wrapped auth/quota wall whose OUTER text
270
+ // coincidentally carries a drop wording ("fetch failed") must NOT be flagged a
271
+ // drop — the classification wins over the wording, on both paths.
272
+ it('a wrapped auth error whose outer text says "fetch failed" is NOT a connection-drop', () => {
273
+ const a = parseLlmError('authentication_error: invalid api key (underlying: fetch failed)')
274
+ expect(a.kind).toBe('auth')
275
+ expect(a.connectionDrop).toBe(false)
276
+
277
+ const b = detectErrorInTranscriptLine(
278
+ JSON.stringify({
279
+ type: 'error',
280
+ error: { type: 'authentication_error', message: 'invalid api key; underlying: fetch failed' },
281
+ }),
282
+ )
283
+ expect(b).not.toBeNull()
284
+ expect(b!.kind).toBe('credentials-invalid')
285
+ expect(b!.connectionDrop).toBe(false)
286
+ })
287
+ })
288
+
289
+ describe('isConnectionDropText — the canonical matcher is conservative', () => {
290
+ it('matches the canonical drop wordings', () => {
291
+ for (const s of [
292
+ 'socket hang up',
293
+ 'read ECONNRESET',
294
+ 'connect ECONNREFUSED 1.2.3.4:443',
295
+ 'write EPIPE',
296
+ 'fetch failed',
297
+ 'network error',
298
+ 'Connection closed mid-response',
299
+ 'connection lost',
300
+ 'Premature close',
301
+ 'the stream disconnected unexpectedly',
302
+ ]) {
303
+ expect(isConnectionDropText(s)).toBe(true)
304
+ }
305
+ })
306
+
307
+ it('does NOT over-match "upstream"/"downstream"/"terminated" or auth/quota wording', () => {
308
+ for (const s of [
309
+ 'the upstream provider returned a result',
310
+ 'downstream consumer finished',
311
+ 'the worker terminated cleanly',
312
+ 'authentication_error: invalid api key',
313
+ "You've hit your usage limit",
314
+ 'overloaded_error',
315
+ 'insufficient credits',
316
+ '',
317
+ ]) {
318
+ expect(isConnectionDropText(s)).toBe(false)
319
+ }
320
+ })
321
+ })
322
+
140
323
  // ─── local-time render ───────────────────────────────────────────────────────
141
324
 
142
325
  describe('renderLlmError / formatResetLocal — local-time rendering', () => {
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "hindsight-memory",
3
3
  "description": "Automatic long-term memory for Claude Code via Hindsight. Recalls relevant memories before each prompt and retains conversation transcripts after each response.",
4
- "version": "0.4.0",
4
+ "version": "0.5.0",
5
5
  "author": {"name": "Hindsight Team", "url": "https://vectorize.io/hindsight"},
6
6
  "license": "MIT",
7
7
  "keywords": ["memory", "hindsight", "recall", "retain"]
@@ -1,7 +1,27 @@
1
1
  # Changelog
2
2
 
3
+ > **Versioning scheme (switchroom, #4779).** The manifest `version` in
4
+ > `.claude-plugin/plugin.json` MUST be bumped on every substantive change to
5
+ > this vendored tree — new/removed `scripts/`, changed hook wiring, or any
6
+ > behaviour an agent would observe. It sat frozen at `0.4.0` across the entire
7
+ > M4/M5 async-recall-prefetch rewrite, so a version-string check could not tell
8
+ > a June pre-M4 tree from the shipped build and `test-harness` silently ran the
9
+ > stale one. The version is only a COARSE signal, though — the authoritative
10
+ > drift guard is the `scripts/`-tree hash in
11
+ > `detectHindsightPluginTreeDrift` (`src/agents/drift.ts`), surfaced by
12
+ > `switchroom doctor` and the gateway boot-card. Bump the version AND rely on
13
+ > the hash; never the version alone.
14
+
3
15
  ## [Unreleased]
4
16
 
17
+ ### Changed (switchroom divergence)
18
+
19
+ - **Manifest version bumped `0.4.0` → `0.5.0`** to reflect the M4/M5
20
+ async-recall-prefetch tree (adds `scripts/prefetch.py`,
21
+ `scripts/lib/recall_buffer.py`, `scripts/orientation.py`, and the
22
+ `memoryPrefetch*` settings). Per the scheme note above this is the first
23
+ bump of the discipline that stops build drift from going invisible (#4779).
24
+
5
25
  ### Added (switchroom divergence)
6
26
 
7
27
  - **Per-row `observation_scopes` on every retain.** Hindsight accepts and
@@ -93,13 +93,27 @@ def _sentinel_path(session_id: str) -> str:
93
93
  return os.path.join(buffer_dir(), f"{_safe_session(session_id)}.buffer.done")
94
94
 
95
95
 
96
- def write_buffer(session_id: str, context: str, telemetry: Optional[dict] = None) -> None:
96
+ def write_buffer(
97
+ session_id: str,
98
+ context: str,
99
+ telemetry: Optional[dict] = None,
100
+ query: Optional[str] = None,
101
+ ) -> None:
97
102
  """Write the prefetched recall payload for ``session_id``.
98
103
 
99
104
  Atomic (temp file + ``os.replace`` within the same directory). Does NOT
100
105
  write the sentinel — the caller MUST call ``write_sentinel`` after this,
101
106
  and only once the payload write has returned, to preserve the
102
107
  read-after-write ordering guarantee.
108
+
109
+ ``query`` (#4778) is the speculative query the producer used to build this
110
+ buffer. It is stored verbatim so the consumer can gate the join on topical
111
+ similarity between it and turn N+1's ACTUAL prompt — the buffer is keyed by
112
+ ``session_id`` alone and, without this, a fresh buffer built for the prior
113
+ turn is served on a topic pivot regardless of relevance. A ``None``/absent
114
+ query is stored as ``""``, which the consumer treats as "cannot establish a
115
+ topic match" and falls through to synchronous recall (fail-safe; also the
116
+ backward-compat behaviour for buffers written before this field existed).
103
117
  """
104
118
  d = _ensure_dir()
105
119
  final = _buffer_path(session_id)
@@ -108,6 +122,7 @@ def write_buffer(session_id: str, context: str, telemetry: Optional[dict] = None
108
122
  "schema": SCHEMA,
109
123
  "session_id": session_id,
110
124
  "context": context,
125
+ "query": query or "",
111
126
  "telemetry": telemetry or {},
112
127
  "written_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
113
128
  }
@@ -187,7 +202,11 @@ def sentinel_exists(session_id: str) -> bool:
187
202
  def read_if_fresh(session_id: str, last_consumed_token: Optional[int]) -> tuple:
188
203
  """Return ``(payload_dict | None, current_token)``.
189
204
 
190
- ``payload_dict`` (when present) is ``{"context": str, "telemetry": dict}``.
205
+ ``payload_dict`` (when present) is
206
+ ``{"context": str, "query": str, "telemetry": dict}``. ``query`` (#4778) is
207
+ the speculative query the buffer was built for — the consumer uses it to
208
+ gate the join on topical similarity to turn N+1's prompt. It is ``""`` for a
209
+ legacy buffer written before the field existed.
191
210
 
192
211
  Returns ``(None, token_or_last_consumed)`` when:
193
212
  * no sentinel exists yet (nothing has been produced this session), or
@@ -207,7 +226,11 @@ def read_if_fresh(session_id: str, last_consumed_token: Optional[int]) -> tuple:
207
226
  # Torn write: sentinel landed but payload didn't (or is corrupt).
208
227
  # Fail-closed — never serve this as fresh.
209
228
  return None, token
210
- return {"context": payload.get("context", ""), "telemetry": payload.get("telemetry", {})}, token
229
+ return {
230
+ "context": payload.get("context", ""),
231
+ "query": payload.get("query", ""),
232
+ "telemetry": payload.get("telemetry", {}),
233
+ }, token
211
234
 
212
235
 
213
236
  def invalidate(session_id: str) -> None:
@@ -75,8 +75,18 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
75
75
  that could take the Stop hook (and thus the turn) down with it."""
76
76
  session_id = hook_input.get("session_id") or "unknown"
77
77
 
78
- prompt = (hook_input.get("prompt") or hook_input.get("user_prompt") or "").strip()
79
- if prompt.startswith("<task-notification"):
78
+ # F6 — junk gate derived from the TRANSCRIPT's last human turn, not from a
79
+ # `hook_input["prompt"]` field. A Stop hook's input carries only
80
+ # session_id/transcript_path/stop_hook_active — never `prompt`/`user_prompt`
81
+ # — so the old `hook_input.get("prompt")` gate was permanently empty and
82
+ # NEVER fired, meaning prefetch would run on `<task-notification>` turns once
83
+ # lit. `_last_human_prompt` reads the same last-human turn the speculative
84
+ # query is derived from, so the gate now fires on exactly the synthetic
85
+ # follow-up turns `recall.py`'s synchronous gate skips (honouring the same
86
+ # `recallSkipTaskNotification` switch).
87
+ transcript_path = hook_input.get("transcript_path") or ""
88
+ query = _last_human_prompt(transcript_path)
89
+ if config.get("recallSkipTaskNotification", True) and query.startswith("<task-notification"):
80
90
  debug_log(config, "Prefetch: task-notification turn, skipping")
81
91
  return False
82
92
 
@@ -97,9 +107,8 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
97
107
  except Exception as exc: # pragma: no cover - defensive
98
108
  debug_log(config, f"Prefetch: delta retain failed, continuing without it: {exc}")
99
109
 
100
- # Step 2 — speculative recall.
101
- transcript_path = hook_input.get("transcript_path") or ""
102
- query = _last_human_prompt(transcript_path)
110
+ # Step 2 — speculative recall (query derived above from the transcript's
111
+ # last human turn).
103
112
  if not query:
104
113
  debug_log(config, "Prefetch: no usable query, nothing to prefetch")
105
114
  return False
@@ -123,6 +132,25 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
123
132
  debug_log(config, "Prefetch: no candidates, nothing to buffer")
124
133
  return False
125
134
 
135
+ # F5 — CURATE before buffering, through the SAME pipeline the synchronous
136
+ # recall path enforces (demote-drop, relevance sort, score floor, and the
137
+ # `recallMaxMemories` cap). Calling recall's shared helper — rather than
138
+ # `format_memories(results)` on the raw set — is what stops a demoted or
139
+ # uncapped memory from reaching the buffer and bypassing curation the sync
140
+ # path applies (carve §6.1 sharing requirement). Imported lazily so the
141
+ # flag-off no-op in `main()` never pays recall.py's import cost.
142
+ try:
143
+ import recall # noqa: PLC0415 - lazy, kept off the flag-off no-op path
144
+
145
+ results = recall.curate_recall_results(results, config, bank_id)
146
+ except Exception as exc: # pragma: no cover - defensive
147
+ debug_log(config, f"Prefetch: curation failed, skipping buffer: {exc}")
148
+ return False
149
+
150
+ if not results:
151
+ debug_log(config, "Prefetch: all candidates filtered by curation, nothing to buffer")
152
+ return False
153
+
126
154
  from lib.content import format_memories
127
155
 
128
156
  memories_block = format_memories(results)
@@ -131,7 +159,13 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
131
159
 
132
160
  # Step 3 — write payload THEN sentinel, strictly in that order.
133
161
  try:
134
- recall_buffer.write_buffer(session_id, memories_block, {"result_count": len(results)})
162
+ # #4778 — persist the speculative `query` alongside the buffer so the
163
+ # consumer can gate the turn-N+1 join on topical similarity, not just
164
+ # session freshness. Without it a fresh buffer built for THIS turn's
165
+ # prompt is served on next turn's prompt even after a topic pivot.
166
+ recall_buffer.write_buffer(
167
+ session_id, memories_block, {"result_count": len(results)}, query=query
168
+ )
135
169
  recall_buffer.write_sentinel(session_id)
136
170
  except Exception as exc: # pragma: no cover - defensive
137
171
  debug_log(config, f"Prefetch: buffer write failed: {exc}")