switchroom 0.21.18 → 0.21.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +150 -2
- package/dist/host-control/main.js +1 -1
- package/package.json +1 -1
- package/skills/switchroom-release/SKILL.md +6 -1
- package/telegram-plugin/connection-drop.ts +83 -0
- package/telegram-plugin/dist/bridge/bridge.js +41 -2
- package/telegram-plugin/dist/gateway/gateway.js +48 -6
- package/telegram-plugin/dist/server.js +46 -3
- package/telegram-plugin/llm-error-present.ts +25 -0
- package/telegram-plugin/session-tail.ts +38 -1
- package/telegram-plugin/tests/llm-error-present.test.ts +183 -0
- package/vendor/hindsight-memory/scripts/prefetch.py +33 -5
- package/vendor/hindsight-memory/scripts/recall.py +131 -17
- package/vendor/hindsight-memory/scripts/tests/test_prefetch_pipeline.py +97 -6
- package/vendor/hindsight-memory/scripts/tests/test_recall_buffer_join.py +70 -0
|
@@ -42,6 +42,7 @@ function isMultiAgentEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
|
|
|
42
42
|
}
|
|
43
43
|
import { classifyClaudeError, type OperatorEventKind } from './operator-events.js'
|
|
44
44
|
import { isLitellmProxyLocal429, isTransientUpstreamSignal } from './model-unavailable.js'
|
|
45
|
+
import { isConnectionDropText } from './connection-drop.js'
|
|
45
46
|
import { createToolLabelSidecar, type ToolLabelSidecar, type SidecarOptions } from './tool-label-sidecar.js'
|
|
46
47
|
import { isModelSentinel } from './model-label.js'
|
|
47
48
|
|
|
@@ -960,6 +961,25 @@ function extractRetryState(obj: Record<string, unknown>): {
|
|
|
960
961
|
}
|
|
961
962
|
}
|
|
962
963
|
|
|
964
|
+
/**
|
|
965
|
+
* The OperatorEventKinds on which a connection-drop wording may be flagged as a
|
|
966
|
+
* genuine transport drop. INCLUSION (not exclusion) by design: a positively
|
|
967
|
+
* classified auth/quota/credit/rate-limit/overload wall is NEVER in this set,
|
|
968
|
+
* so a wrapped terminal error whose outer text coincidentally carries a drop
|
|
969
|
+
* wording is never mislabelled a drop. `transport-transient` is the natural
|
|
970
|
+
* mid-stream-abort kind; `unknown-4xx`/`unknown-5xx` is where a drop-worded
|
|
971
|
+
* line lands when `classifyClaudeError` does not recognise its wording (the
|
|
972
|
+
* exact Path-B bug this closes). Mirrors Path A's `transient`/`unknown` gate in
|
|
973
|
+
* `parseLlmError`, keeping the two classifiers in agreement.
|
|
974
|
+
*/
|
|
975
|
+
const CONNECTION_DROP_ELIGIBLE_KINDS: ReadonlySet<OperatorEventKind> =
|
|
976
|
+
new Set<OperatorEventKind>(['transport-transient', 'unknown-5xx', 'unknown-4xx'])
|
|
977
|
+
|
|
978
|
+
/** True when `kind` may carry the connection-drop flag AND the text is a drop wording. */
|
|
979
|
+
function isConnectionDrop(kind: OperatorEventKind, scanText: string): boolean {
|
|
980
|
+
return CONNECTION_DROP_ELIGIBLE_KINDS.has(kind) && isConnectionDropText(scanText)
|
|
981
|
+
}
|
|
982
|
+
|
|
963
983
|
export function detectErrorInTranscriptLine(
|
|
964
984
|
line: string,
|
|
965
985
|
): {
|
|
@@ -972,6 +992,15 @@ export function detectErrorInTranscriptLine(
|
|
|
972
992
|
* error mid-retry is `transient:true, terminal:false`; the caller
|
|
973
993
|
* suppresses it (no operator card until the failure is terminal). */
|
|
974
994
|
terminal: boolean
|
|
995
|
+
/**
|
|
996
|
+
* True when this line is a mid-stream connection / SSE drop (a transport
|
|
997
|
+
* connection loss), per the canonical `isConnectionDropText` matcher, gated
|
|
998
|
+
* to the transport/unknown kinds. Set consistently with `parseLlmError`'s
|
|
999
|
+
* `ParsedLlmError.connectionDrop` so a later PR can gate auto-resume on ONE
|
|
1000
|
+
* reliable discriminator across both classification paths. Classification
|
|
1001
|
+
* only in this PR; no caller acts on it yet.
|
|
1002
|
+
*/
|
|
1003
|
+
connectionDrop: boolean
|
|
975
1004
|
} | null {
|
|
976
1005
|
if (!line || line.length > 2 * 1024 * 1024) return null
|
|
977
1006
|
let obj: Record<string, unknown>
|
|
@@ -1042,6 +1071,7 @@ export function detectErrorInTranscriptLine(
|
|
|
1042
1071
|
detail: text || errStr || 'api error',
|
|
1043
1072
|
transient: kind === 'rate-limited',
|
|
1044
1073
|
terminal: true,
|
|
1074
|
+
connectionDrop: isConnectionDrop(kind, `${text}\n${errStr}`),
|
|
1045
1075
|
}
|
|
1046
1076
|
}
|
|
1047
1077
|
|
|
@@ -1089,7 +1119,14 @@ export function detectErrorInTranscriptLine(
|
|
|
1089
1119
|
? retry.retryAttempt >= retry.maxRetries
|
|
1090
1120
|
: isErrorLine
|
|
1091
1121
|
|
|
1092
|
-
return {
|
|
1122
|
+
return {
|
|
1123
|
+
kind,
|
|
1124
|
+
raw,
|
|
1125
|
+
detail,
|
|
1126
|
+
transient,
|
|
1127
|
+
terminal,
|
|
1128
|
+
connectionDrop: isConnectionDrop(kind, `${detail}\n${String(type ?? '')}`),
|
|
1129
|
+
}
|
|
1093
1130
|
}
|
|
1094
1131
|
|
|
1095
1132
|
function extractDetailMessage(obj: Record<string, unknown> | null): string | null {
|
|
@@ -27,6 +27,7 @@ import {
|
|
|
27
27
|
} from '../llm-error-present.js'
|
|
28
28
|
import { truncateDetailPreservingRequestId } from '../raw-error-scrub.js'
|
|
29
29
|
import { projectTranscriptLine, detectErrorInTranscriptLine } from '../session-tail.js'
|
|
30
|
+
import { isConnectionDropText } from '../connection-drop.js'
|
|
30
31
|
import { renderOperatorEvent, type OperatorEvent } from '../operator-events.js'
|
|
31
32
|
import { redact } from '../secret-detect/redact.js'
|
|
32
33
|
|
|
@@ -137,6 +138,188 @@ describe('parseLlmError — classification table', () => {
|
|
|
137
138
|
})
|
|
138
139
|
})
|
|
139
140
|
|
|
141
|
+
// ─── Connection-drop classification (PR 0: unify drop classification) ─────────
|
|
142
|
+
//
|
|
143
|
+
// The two classifiers historically disagreed about a mid-stream connection /
|
|
144
|
+
// SSE drop: Path A (`parseLlmError`) mapped it to a transient/network error,
|
|
145
|
+
// while Path B (`detectErrorInTranscriptLine`) let a drop-worded line whose
|
|
146
|
+
// wording `classifyClaudeError` did not recognise fall through to a generic
|
|
147
|
+
// `unknown-*` terminal. These blocks assert the SAME line is now identifiable
|
|
148
|
+
// as a connection drop on BOTH paths — and, critically, that auth / quota /
|
|
149
|
+
// overload / provider-credit walls are NEVER flagged as drops on either path
|
|
150
|
+
// (the greedy-matcher failure mode that would later cause a wrong auto-resume).
|
|
151
|
+
|
|
152
|
+
const OPENROUTER_402_CREDIT =
|
|
153
|
+
'litellm.APIError: OpenrouterException - {"error":{"code":402,"message":"Your account or API key has insufficient credits. Add more credits and retry the request.","metadata":{"provider_name":"openrouter"}}}'
|
|
154
|
+
|
|
155
|
+
/** Build the v2.1.x `isApiErrorMessage` synthetic-assistant transcript line. */
|
|
156
|
+
function apiErrorLine(text: string, opts: { error?: string; status?: number } = {}): string {
|
|
157
|
+
return JSON.stringify({
|
|
158
|
+
type: 'assistant',
|
|
159
|
+
message: { role: 'assistant', model: '<synthetic>', content: [{ type: 'text', text }] },
|
|
160
|
+
isApiErrorMessage: true,
|
|
161
|
+
...(opts.error != null ? { error: opts.error } : { error: '' }),
|
|
162
|
+
...(opts.status != null ? { apiErrorStatus: opts.status } : {}),
|
|
163
|
+
})
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
describe('connection-drop — identifiable on BOTH classification paths', () => {
|
|
167
|
+
// The canonical mid-stream abort shape: isApiErrorMessage + error:"server_error",
|
|
168
|
+
// NO status. Path A sees the worded string; Path B sees the transcript line.
|
|
169
|
+
it('a "Connection closed mid-response" line is a connection-drop on Path A AND Path B', () => {
|
|
170
|
+
const wording = 'API Error: Connection closed mid-response'
|
|
171
|
+
|
|
172
|
+
// Path A
|
|
173
|
+
const a = parseLlmError(wording)
|
|
174
|
+
expect(a.connectionDrop).toBe(true)
|
|
175
|
+
expect(a.kind).toBe('transient')
|
|
176
|
+
expect(a.source).toBe('network')
|
|
177
|
+
|
|
178
|
+
// Path B
|
|
179
|
+
const b = detectErrorInTranscriptLine(apiErrorLine(wording, { error: 'server_error' }))
|
|
180
|
+
expect(b).not.toBeNull()
|
|
181
|
+
expect(b!.connectionDrop).toBe(true)
|
|
182
|
+
expect(b!.kind).toBe('transport-transient')
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
// The exact Path-B bug: a drop-worded line whose wording `classifyClaudeError`
|
|
186
|
+
// does NOT recognise (no server_error type, no status) fell through to
|
|
187
|
+
// unknown-5xx. It must STILL be flagged a connection-drop.
|
|
188
|
+
it('a "socket hang up" line unknown to classifyClaudeError is still a connection-drop on both paths', () => {
|
|
189
|
+
const wording = 'API Error: socket hang up'
|
|
190
|
+
|
|
191
|
+
const a = parseLlmError(wording)
|
|
192
|
+
expect(a.connectionDrop).toBe(true)
|
|
193
|
+
|
|
194
|
+
const b = detectErrorInTranscriptLine(apiErrorLine(wording))
|
|
195
|
+
expect(b).not.toBeNull()
|
|
196
|
+
expect(b!.kind).toBe('unknown-5xx') // classifyClaudeError does not recognise the wording…
|
|
197
|
+
expect(b!.connectionDrop).toBe(true) // …yet the drop is identified anyway.
|
|
198
|
+
})
|
|
199
|
+
|
|
200
|
+
// The wrapper-error line shape (`type:"error"`, nested error object).
|
|
201
|
+
it('a server_error wrapper line carrying a drop wording is a connection-drop on Path B', () => {
|
|
202
|
+
const line = JSON.stringify({
|
|
203
|
+
type: 'error',
|
|
204
|
+
error: { type: 'server_error', message: 'Connection closed mid-response' },
|
|
205
|
+
})
|
|
206
|
+
const b = detectErrorInTranscriptLine(line)
|
|
207
|
+
expect(b).not.toBeNull()
|
|
208
|
+
expect(b!.kind).toBe('transport-transient')
|
|
209
|
+
expect(b!.connectionDrop).toBe(true)
|
|
210
|
+
})
|
|
211
|
+
|
|
212
|
+
// EPIPE is a genuine drop wording that `detectModelUnavailable` does not
|
|
213
|
+
// classify as network — Path A lands it on `unknown`, but it must still be a
|
|
214
|
+
// connection-drop (proving the discriminator is NOT tied to source:'network').
|
|
215
|
+
it('EPIPE is a connection-drop on Path A even though its kind stays unknown', () => {
|
|
216
|
+
const a = parseLlmError('write EPIPE: half-dead socket')
|
|
217
|
+
expect(a.kind).toBe('unknown')
|
|
218
|
+
expect(a.connectionDrop).toBe(true)
|
|
219
|
+
})
|
|
220
|
+
})
|
|
221
|
+
|
|
222
|
+
describe('connection-drop — negatives (auth / quota / overload / provider-credit are NEVER drops)', () => {
|
|
223
|
+
it('auth is not a connection-drop on either path', () => {
|
|
224
|
+
const a = parseLlmError('authentication_error: OAuth token expired, please refresh')
|
|
225
|
+
expect(a.kind).toBe('auth')
|
|
226
|
+
expect(a.connectionDrop).toBe(false)
|
|
227
|
+
|
|
228
|
+
const b = detectErrorInTranscriptLine(
|
|
229
|
+
apiErrorLine('authentication_error: OAuth token expired', { error: 'authentication_error', status: 401 }),
|
|
230
|
+
)
|
|
231
|
+
expect(b).not.toBeNull()
|
|
232
|
+
expect(b!.connectionDrop).toBe(false)
|
|
233
|
+
})
|
|
234
|
+
|
|
235
|
+
it('quota_wall is not a connection-drop on either path', () => {
|
|
236
|
+
const a = parseLlmError("You've hit your limit · resets 5pm")
|
|
237
|
+
expect(a.kind).toBe('quota_wall')
|
|
238
|
+
expect(a.connectionDrop).toBe(false)
|
|
239
|
+
|
|
240
|
+
const b = detectErrorInTranscriptLine(
|
|
241
|
+
apiErrorLine("You've hit your limit · resets 5pm", { error: 'rate_limit_error', status: 429 }),
|
|
242
|
+
)
|
|
243
|
+
expect(b).not.toBeNull()
|
|
244
|
+
expect(b!.kind).toBe('quota-exhausted')
|
|
245
|
+
expect(b!.connectionDrop).toBe(false)
|
|
246
|
+
})
|
|
247
|
+
|
|
248
|
+
it('overload_529 is not a connection-drop on either path', () => {
|
|
249
|
+
const a = parseLlmError('Overloaded (overloaded_error) — HTTP 529')
|
|
250
|
+
expect(a.kind).toBe('overload_529')
|
|
251
|
+
expect(a.connectionDrop).toBe(false)
|
|
252
|
+
|
|
253
|
+
const b = detectErrorInTranscriptLine(apiErrorLine('Overloaded', { error: 'overloaded_error' }))
|
|
254
|
+
expect(b).not.toBeNull()
|
|
255
|
+
expect(b!.connectionDrop).toBe(false)
|
|
256
|
+
})
|
|
257
|
+
|
|
258
|
+
it('provider_credit is not a connection-drop on either path', () => {
|
|
259
|
+
const a = parseLlmError(OPENROUTER_402_CREDIT)
|
|
260
|
+
expect(a.kind).toBe('provider_credit')
|
|
261
|
+
expect(a.connectionDrop).toBe(false)
|
|
262
|
+
|
|
263
|
+
const b = detectErrorInTranscriptLine(apiErrorLine(OPENROUTER_402_CREDIT, { status: 402 }))
|
|
264
|
+
expect(b).not.toBeNull()
|
|
265
|
+
expect(b!.kind).toBe('provider-credit-exhausted')
|
|
266
|
+
expect(b!.connectionDrop).toBe(false)
|
|
267
|
+
})
|
|
268
|
+
|
|
269
|
+
// The single highest-risk case: a wrapped auth/quota wall whose OUTER text
|
|
270
|
+
// coincidentally carries a drop wording ("fetch failed") must NOT be flagged a
|
|
271
|
+
// drop — the classification wins over the wording, on both paths.
|
|
272
|
+
it('a wrapped auth error whose outer text says "fetch failed" is NOT a connection-drop', () => {
|
|
273
|
+
const a = parseLlmError('authentication_error: invalid api key (underlying: fetch failed)')
|
|
274
|
+
expect(a.kind).toBe('auth')
|
|
275
|
+
expect(a.connectionDrop).toBe(false)
|
|
276
|
+
|
|
277
|
+
const b = detectErrorInTranscriptLine(
|
|
278
|
+
JSON.stringify({
|
|
279
|
+
type: 'error',
|
|
280
|
+
error: { type: 'authentication_error', message: 'invalid api key; underlying: fetch failed' },
|
|
281
|
+
}),
|
|
282
|
+
)
|
|
283
|
+
expect(b).not.toBeNull()
|
|
284
|
+
expect(b!.kind).toBe('credentials-invalid')
|
|
285
|
+
expect(b!.connectionDrop).toBe(false)
|
|
286
|
+
})
|
|
287
|
+
})
|
|
288
|
+
|
|
289
|
+
describe('isConnectionDropText — the canonical matcher is conservative', () => {
|
|
290
|
+
it('matches the canonical drop wordings', () => {
|
|
291
|
+
for (const s of [
|
|
292
|
+
'socket hang up',
|
|
293
|
+
'read ECONNRESET',
|
|
294
|
+
'connect ECONNREFUSED 1.2.3.4:443',
|
|
295
|
+
'write EPIPE',
|
|
296
|
+
'fetch failed',
|
|
297
|
+
'network error',
|
|
298
|
+
'Connection closed mid-response',
|
|
299
|
+
'connection lost',
|
|
300
|
+
'Premature close',
|
|
301
|
+
'the stream disconnected unexpectedly',
|
|
302
|
+
]) {
|
|
303
|
+
expect(isConnectionDropText(s)).toBe(true)
|
|
304
|
+
}
|
|
305
|
+
})
|
|
306
|
+
|
|
307
|
+
it('does NOT over-match "upstream"/"downstream"/"terminated" or auth/quota wording', () => {
|
|
308
|
+
for (const s of [
|
|
309
|
+
'the upstream provider returned a result',
|
|
310
|
+
'downstream consumer finished',
|
|
311
|
+
'the worker terminated cleanly',
|
|
312
|
+
'authentication_error: invalid api key',
|
|
313
|
+
"You've hit your usage limit",
|
|
314
|
+
'overloaded_error',
|
|
315
|
+
'insufficient credits',
|
|
316
|
+
'',
|
|
317
|
+
]) {
|
|
318
|
+
expect(isConnectionDropText(s)).toBe(false)
|
|
319
|
+
}
|
|
320
|
+
})
|
|
321
|
+
})
|
|
322
|
+
|
|
140
323
|
// ─── local-time render ───────────────────────────────────────────────────────
|
|
141
324
|
|
|
142
325
|
describe('renderLlmError / formatResetLocal — local-time rendering', () => {
|
|
@@ -75,8 +75,18 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
|
|
|
75
75
|
that could take the Stop hook (and thus the turn) down with it."""
|
|
76
76
|
session_id = hook_input.get("session_id") or "unknown"
|
|
77
77
|
|
|
78
|
-
|
|
79
|
-
|
|
78
|
+
# F6 — junk gate derived from the TRANSCRIPT's last human turn, not from a
|
|
79
|
+
# `hook_input["prompt"]` field. A Stop hook's input carries only
|
|
80
|
+
# session_id/transcript_path/stop_hook_active — never `prompt`/`user_prompt`
|
|
81
|
+
# — so the old `hook_input.get("prompt")` gate was permanently empty and
|
|
82
|
+
# NEVER fired, meaning prefetch would run on `<task-notification>` turns once
|
|
83
|
+
# lit. `_last_human_prompt` reads the same last-human turn the speculative
|
|
84
|
+
# query is derived from, so the gate now fires on exactly the synthetic
|
|
85
|
+
# follow-up turns `recall.py`'s synchronous gate skips (honouring the same
|
|
86
|
+
# `recallSkipTaskNotification` switch).
|
|
87
|
+
transcript_path = hook_input.get("transcript_path") or ""
|
|
88
|
+
query = _last_human_prompt(transcript_path)
|
|
89
|
+
if config.get("recallSkipTaskNotification", True) and query.startswith("<task-notification"):
|
|
80
90
|
debug_log(config, "Prefetch: task-notification turn, skipping")
|
|
81
91
|
return False
|
|
82
92
|
|
|
@@ -97,9 +107,8 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
|
|
|
97
107
|
except Exception as exc: # pragma: no cover - defensive
|
|
98
108
|
debug_log(config, f"Prefetch: delta retain failed, continuing without it: {exc}")
|
|
99
109
|
|
|
100
|
-
# Step 2 — speculative recall
|
|
101
|
-
|
|
102
|
-
query = _last_human_prompt(transcript_path)
|
|
110
|
+
# Step 2 — speculative recall (query derived above from the transcript's
|
|
111
|
+
# last human turn).
|
|
103
112
|
if not query:
|
|
104
113
|
debug_log(config, "Prefetch: no usable query, nothing to prefetch")
|
|
105
114
|
return False
|
|
@@ -123,6 +132,25 @@ def run_prefetch(hook_input: dict, config: dict) -> bool:
|
|
|
123
132
|
debug_log(config, "Prefetch: no candidates, nothing to buffer")
|
|
124
133
|
return False
|
|
125
134
|
|
|
135
|
+
# F5 — CURATE before buffering, through the SAME pipeline the synchronous
|
|
136
|
+
# recall path enforces (demote-drop, relevance sort, score floor, and the
|
|
137
|
+
# `recallMaxMemories` cap). Calling recall's shared helper — rather than
|
|
138
|
+
# `format_memories(results)` on the raw set — is what stops a demoted or
|
|
139
|
+
# uncapped memory from reaching the buffer and bypassing curation the sync
|
|
140
|
+
# path applies (carve §6.1 sharing requirement). Imported lazily so the
|
|
141
|
+
# flag-off no-op in `main()` never pays recall.py's import cost.
|
|
142
|
+
try:
|
|
143
|
+
import recall # noqa: PLC0415 - lazy, kept off the flag-off no-op path
|
|
144
|
+
|
|
145
|
+
results = recall.curate_recall_results(results, config, bank_id)
|
|
146
|
+
except Exception as exc: # pragma: no cover - defensive
|
|
147
|
+
debug_log(config, f"Prefetch: curation failed, skipping buffer: {exc}")
|
|
148
|
+
return False
|
|
149
|
+
|
|
150
|
+
if not results:
|
|
151
|
+
debug_log(config, "Prefetch: all candidates filtered by curation, nothing to buffer")
|
|
152
|
+
return False
|
|
153
|
+
|
|
126
154
|
from lib.content import format_memories
|
|
127
155
|
|
|
128
156
|
memories_block = format_memories(results)
|
|
@@ -92,6 +92,13 @@ _IMPORT_ELAPSED_SECONDS = time.monotonic() - _IMPORT_START_MONOTONIC
|
|
|
92
92
|
|
|
93
93
|
LAST_RECALL_STATE = "last_recall.json"
|
|
94
94
|
RECALL_CACHE_STATE = "recall_cache.json"
|
|
95
|
+
# M4 P-REC F3 — the last prefetch-buffer sentinel token this session has
|
|
96
|
+
# already CONSUMED, keyed by session_id. Persisted across turns so a buffer
|
|
97
|
+
# produced at turn N and consumed at N+1 is not re-served as "fresh" at
|
|
98
|
+
# N+2..N+k when no newer sentinel has landed (the stale-buffer-served-as-fresh
|
|
99
|
+
# class). A dict {session_id: token}; capped like the other per-session state.
|
|
100
|
+
PREFETCH_CONSUMED_STATE = "prefetch_consumed.json"
|
|
101
|
+
_PREFETCH_CONSUMED_MAX_SESSIONS = 10000
|
|
95
102
|
|
|
96
103
|
# Switchroom hindsight-leverage A3 — label for the directives fetch slot in the
|
|
97
104
|
# parallel fan-out. Distinct from any bank_id (banks can't start with "__") so a
|
|
@@ -500,12 +507,6 @@ def _emit_cached_context(context: str) -> None:
|
|
|
500
507
|
)
|
|
501
508
|
|
|
502
509
|
|
|
503
|
-
_PREFETCH_DEGRADED_NOTICE = (
|
|
504
|
-
"⏳ prefetch not ready and no prior recall is cached for this session — "
|
|
505
|
-
"proceeding without injected memory this turn."
|
|
506
|
-
)
|
|
507
|
-
|
|
508
|
-
|
|
509
510
|
def stale_recall_notice(memories_context: str) -> str:
|
|
510
511
|
"""Wrap a PRIOR turn's cached memories-only block in an explicit
|
|
511
512
|
staleness marker for the M4 prefetch-buffer fallback path.
|
|
@@ -528,6 +529,100 @@ def stale_recall_notice(memories_context: str) -> str:
|
|
|
528
529
|
)
|
|
529
530
|
|
|
530
531
|
|
|
532
|
+
def _read_consumed_token(session_id: str) -> "int | None":
|
|
533
|
+
"""Return the last prefetch sentinel token this session already consumed,
|
|
534
|
+
or None if none is recorded. Failure-tolerant — any read/parse error
|
|
535
|
+
returns None (treat as "nothing consumed yet", the fail-open direction that
|
|
536
|
+
at worst re-serves once, never suppresses forever)."""
|
|
537
|
+
state = read_state(PREFETCH_CONSUMED_STATE, {}) or {}
|
|
538
|
+
if not isinstance(state, dict):
|
|
539
|
+
return None
|
|
540
|
+
token = state.get(session_id)
|
|
541
|
+
return int(token) if isinstance(token, (int, float)) else None
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _write_consumed_token(session_id: str, token: int) -> None:
|
|
545
|
+
"""Persist `token` as the last sentinel this session has consumed. Bounded
|
|
546
|
+
to `_PREFETCH_CONSUMED_MAX_SESSIONS` entries. Best-effort — write_state
|
|
547
|
+
never raises."""
|
|
548
|
+
state = read_state(PREFETCH_CONSUMED_STATE, {}) or {}
|
|
549
|
+
if not isinstance(state, dict):
|
|
550
|
+
state = {}
|
|
551
|
+
state[session_id] = int(token)
|
|
552
|
+
if len(state) > _PREFETCH_CONSUMED_MAX_SESSIONS:
|
|
553
|
+
# Drop the numerically-smallest tokens (oldest producers) first.
|
|
554
|
+
for k in sorted(
|
|
555
|
+
state.keys(),
|
|
556
|
+
key=lambda k: state[k] if isinstance(state[k], (int, float)) else 0,
|
|
557
|
+
)[: len(state) - _PREFETCH_CONSUMED_MAX_SESSIONS]:
|
|
558
|
+
state.pop(k, None)
|
|
559
|
+
write_state(PREFETCH_CONSUMED_STATE, state)
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def curate_recall_results(results, config, bank_id):
|
|
563
|
+
"""M4 F5 — the SHARED recall curation pipeline, so the async prefetch
|
|
564
|
+
producer (`prefetch.py`) can never drift from the guarantees the
|
|
565
|
+
synchronous recall path enforces.
|
|
566
|
+
|
|
567
|
+
Applies, in order, the exact same module-level primitives the synchronous
|
|
568
|
+
path composes inline: drop demote-tagged memories (`_is_demoted_memory`),
|
|
569
|
+
sort by engine relevance (`_sort_by_final_score`), apply the optional
|
|
570
|
+
absolute score floor when it is configured for ALL turns
|
|
571
|
+
(`_filter_by_min_score`; the `degraded`-scope floor is a synchronous-path
|
|
572
|
+
concept the speculative single-bank prefetch cannot evaluate, so it is
|
|
573
|
+
honoured here only when `recallMinScoreScope` is `all`, matching what the
|
|
574
|
+
sync path does on a healthy own-bank turn), and finally the
|
|
575
|
+
`recallMaxMemories` head cap with per-bank slot reservation
|
|
576
|
+
(`_reserve_bank_slots`).
|
|
577
|
+
|
|
578
|
+
Returns the curated result list. Never mutates the caller's list in a way
|
|
579
|
+
that changes its length before returning (it rebinds internally). Pure —
|
|
580
|
+
no I/O, no network — so it is safe to call from the producer hook.
|
|
581
|
+
"""
|
|
582
|
+
if not results:
|
|
583
|
+
return results
|
|
584
|
+
|
|
585
|
+
# 1) demote-from-recall drop (Switchroom #432 4.4) — before any cap so the
|
|
586
|
+
# cap fills with non-demoted hits.
|
|
587
|
+
results = [m for m in results if not _is_demoted_memory(m)]
|
|
588
|
+
|
|
589
|
+
# 2) relevance sort (Phase-1 bank-starvation fix) — in place on our copy.
|
|
590
|
+
_sort_by_final_score(results)
|
|
591
|
+
|
|
592
|
+
# 3) optional absolute score floor (#3837). The sync path binds this on
|
|
593
|
+
# `all` scope always, and on `degraded` scope only when the own-bank
|
|
594
|
+
# read degraded. A prefetch is a healthy speculative read (no degraded
|
|
595
|
+
# signal), so we bind only the `all` scope here — identical to the sync
|
|
596
|
+
# path's decision on a non-degraded turn.
|
|
597
|
+
min_score_floor = config.get("recallMinScore", 0.0)
|
|
598
|
+
if isinstance(min_score_floor, bool) or not isinstance(min_score_floor, (int, float)):
|
|
599
|
+
min_score_floor = 0.0
|
|
600
|
+
min_score_floor = float(min_score_floor)
|
|
601
|
+
min_score_scope = config.get("recallMinScoreScope", "degraded")
|
|
602
|
+
if min_score_floor > 0 and min_score_scope == "all":
|
|
603
|
+
results, _dropped = _filter_by_min_score(results, min_score_floor)
|
|
604
|
+
|
|
605
|
+
# 4) recallMaxMemories head cap + per-bank slot reservation. `_reserve_
|
|
606
|
+
# bank_slots` is a passthrough when cap<=0 or the set already fits, and
|
|
607
|
+
# with the fleet-default 0/0 floors on a single-bank prefetch it reduces
|
|
608
|
+
# to a plain head-slice — the same slice the sync path takes.
|
|
609
|
+
recall_max_memories = config.get("recallMaxMemories", 0)
|
|
610
|
+
if (
|
|
611
|
+
isinstance(recall_max_memories, int)
|
|
612
|
+
and recall_max_memories > 0
|
|
613
|
+
and len(results) > recall_max_memories
|
|
614
|
+
):
|
|
615
|
+
results, _own, _add = _reserve_bank_slots(
|
|
616
|
+
results,
|
|
617
|
+
recall_max_memories,
|
|
618
|
+
bank_id,
|
|
619
|
+
config.get("recallOwnBankMinSlots", 0),
|
|
620
|
+
config.get("recallAdditionalBankMinSlots", 0),
|
|
621
|
+
)
|
|
622
|
+
|
|
623
|
+
return results
|
|
624
|
+
|
|
625
|
+
|
|
531
626
|
def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool:
|
|
532
627
|
"""M4 P-REC Fix C (consumer side) — join the Stop-hook producer's
|
|
533
628
|
prefetch buffer for this session instead of running recall
|
|
@@ -535,10 +630,12 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
535
630
|
|
|
536
631
|
Gated entirely by `config.get("memoryPrefetchEnabled", False)` at the
|
|
537
632
|
caller; this function assumes the flag is already on. Returns True iff
|
|
538
|
-
it emitted an `additionalContext` payload (fresh hit, stale
|
|
539
|
-
|
|
540
|
-
running the synchronous path. Returns False on a
|
|
541
|
-
|
|
633
|
+
it emitted an `additionalContext` payload (fresh buffer hit, stale
|
|
634
|
+
fallback, or a directives-only block) and the caller should return
|
|
635
|
+
without running the synchronous path. Returns False on a miss with
|
|
636
|
+
nothing to serve — cold start, no fresh buffer, no stale prior recall,
|
|
637
|
+
no directives — so the caller falls through to SYNCHRONOUS recall (M4
|
|
638
|
+
F4: the first turn of a fresh/post-restart session must still recall).
|
|
542
639
|
|
|
543
640
|
Never raises past this function's own boundary in normal operation —
|
|
544
641
|
every internal step is wrapped so a bug here degrades to "fall through
|
|
@@ -547,6 +644,13 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
547
644
|
"""
|
|
548
645
|
session_id = hook_input.get("session_id") or "unknown"
|
|
549
646
|
|
|
647
|
+
# M4 F3 — the sentinel token this session has ALREADY consumed on a prior
|
|
648
|
+
# turn. Threaded into both `poll_for_sentinel` and `read_if_fresh` so a
|
|
649
|
+
# buffer produced at turn N and consumed at N+1 is treated as STALE (not
|
|
650
|
+
# re-served as fresh) at N+2..N+k until a strictly-newer sentinel lands.
|
|
651
|
+
# None on a cold session (nothing consumed yet).
|
|
652
|
+
last_consumed = _read_consumed_token(session_id)
|
|
653
|
+
|
|
550
654
|
# Cold-start short-circuit (red-team MAJOR finding): if this session has
|
|
551
655
|
# NEVER produced a sentinel, polling the full cap on every single
|
|
552
656
|
# session-open turn would cost ~the poll cap on every fresh session —
|
|
@@ -556,9 +660,9 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
556
660
|
debug_log(config, "Prefetch buffer: no sentinel ever written for this session, cold-start skip")
|
|
557
661
|
else:
|
|
558
662
|
cap_ms = int(config.get("memoryPrefetchPollCapMs", 400))
|
|
559
|
-
recall_buffer.poll_for_sentinel(session_id, last_consumed_token=
|
|
663
|
+
recall_buffer.poll_for_sentinel(session_id, last_consumed_token=last_consumed, cap_ms=cap_ms)
|
|
560
664
|
|
|
561
|
-
payload, _token = recall_buffer.read_if_fresh(session_id, last_consumed_token=
|
|
665
|
+
payload, _token = recall_buffer.read_if_fresh(session_id, last_consumed_token=last_consumed)
|
|
562
666
|
|
|
563
667
|
# Directives stay on the synchronous, always-fresh path (M3 rule) even
|
|
564
668
|
# in the fast path — fetched here directly, never from the buffer.
|
|
@@ -584,6 +688,12 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
584
688
|
directives_block = None
|
|
585
689
|
|
|
586
690
|
if payload is not None:
|
|
691
|
+
# M4 F3 — record this token as consumed BEFORE emitting, so the next
|
|
692
|
+
# turn's `read_if_fresh` rejects the same buffer as stale unless the
|
|
693
|
+
# producer has since written a strictly-newer sentinel. `_token` is the
|
|
694
|
+
# sentinel token `read_if_fresh` just validated as fresh.
|
|
695
|
+
if _token is not None:
|
|
696
|
+
_write_consumed_token(session_id, _token)
|
|
587
697
|
memories_block = payload.get("context") or ""
|
|
588
698
|
parts = [b for b in (_dir_notice, directives_block, memories_block) if b]
|
|
589
699
|
if not parts:
|
|
@@ -607,11 +717,15 @@ def _handle_prefetch_buffer(config: dict, hook_input: dict, prompt: str) -> bool
|
|
|
607
717
|
_emit_cached_context("\n\n".join([b for b in (_dir_notice, directives_block) if b]))
|
|
608
718
|
return True
|
|
609
719
|
|
|
610
|
-
#
|
|
611
|
-
#
|
|
612
|
-
#
|
|
613
|
-
|
|
614
|
-
|
|
720
|
+
# M4 F4 — nothing fresh, nothing stale, nothing cached (the cold-start /
|
|
721
|
+
# first-turn / no-prior-recall case). Return False so the caller runs the
|
|
722
|
+
# SYNCHRONOUS recall path, which is exactly the path that should run on a
|
|
723
|
+
# turn where no prefetch could possibly exist yet. Emitting a degraded
|
|
724
|
+
# notice and returning True here (the previous behaviour) short-circuited
|
|
725
|
+
# sync recall and handed a fresh/post-restart session ZERO memories plus a
|
|
726
|
+
# degraded banner on its very first turn — the opposite of the docstring's
|
|
727
|
+
# "a miss falls through to sync recall" contract.
|
|
728
|
+
return False
|
|
615
729
|
|
|
616
730
|
|
|
617
731
|
def _emit_directives_only(config: dict, hook_input: dict) -> None:
|
|
@@ -233,14 +233,105 @@ class KillSwitchOffTests(PrefetchPipelineBase):
|
|
|
233
233
|
|
|
234
234
|
|
|
235
235
|
class JunkGateTests(PrefetchPipelineBase):
|
|
236
|
-
def
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
236
|
+
def test_task_notification_in_transcript_skips_before_retain(self):
|
|
237
|
+
# F6 (regression guard): a real Stop-hook input carries NO `prompt`
|
|
238
|
+
# field — only session_id / transcript_path / stop_hook_active — so the
|
|
239
|
+
# junk gate MUST derive from the transcript's last human turn. When that
|
|
240
|
+
# turn is a `<task-notification>` envelope, prefetch skips ENTIRELY:
|
|
241
|
+
# no delta retain, no recall, no buffer. This fails on the pre-fix code,
|
|
242
|
+
# whose `hook_input.get("prompt")` gate is permanently empty on real
|
|
243
|
+
# input and therefore lets retain run on a task-notification turn.
|
|
244
|
+
_write_transcript(
|
|
245
|
+
self.transcript_path,
|
|
246
|
+
[("user", "<task-notification>sub-agent done</task-notification>")],
|
|
247
|
+
)
|
|
248
|
+
real_stop_input = {
|
|
249
|
+
"session_id": SESSION,
|
|
250
|
+
"transcript_path": self.transcript_path,
|
|
251
|
+
"stop_hook_active": True,
|
|
252
|
+
"cwd": "/tmp",
|
|
253
|
+
} # deliberately NO `prompt` / `user_prompt` key — mirrors real input
|
|
254
|
+
retain_spy = mock.Mock(return_value={"status": "ok"})
|
|
255
|
+
with patch("prefetch.retain_module.run_retain", retain_spy):
|
|
256
|
+
wrote = prefetch.run_prefetch(real_stop_input, self._config())
|
|
242
257
|
self.assertFalse(wrote)
|
|
258
|
+
retain_spy.assert_not_called()
|
|
243
259
|
self.assertFalse(os.path.isfile(recall_buffer._buffer_path(SESSION)))
|
|
260
|
+
self.assertFalse(os.path.isfile(recall_buffer._sentinel_path(SESSION)))
|
|
261
|
+
|
|
262
|
+
def test_gate_keys_off_transcript_not_a_phantom_prompt_field(self):
|
|
263
|
+
# The mirror of the above: a genuine human turn in the transcript with
|
|
264
|
+
# a STRAY `prompt="<task-notification>"` on the hook input (the shape the
|
|
265
|
+
# old harness faked). The gate must key off the transcript (a real turn)
|
|
266
|
+
# and PROCEED, proving it no longer reads `hook_input["prompt"]`. Fails
|
|
267
|
+
# on the pre-fix code, which reads the phantom field and wrongly skips.
|
|
268
|
+
_write_transcript(self.transcript_path, [("user", "we shipped the release on friday")])
|
|
269
|
+
client = mock.Mock()
|
|
270
|
+
client.recall.side_effect = lambda *a, **kw: {"results": [
|
|
271
|
+
{"text": "we shipped the release on friday", "type": "fact",
|
|
272
|
+
"mentioned_at": "2026-01-01", "id": "r1", "scores": {"final": 0.9}},
|
|
273
|
+
]}
|
|
274
|
+
hook_input = {
|
|
275
|
+
"session_id": SESSION,
|
|
276
|
+
"transcript_path": self.transcript_path,
|
|
277
|
+
"prompt": "<task-notification>phantom</task-notification>",
|
|
278
|
+
"cwd": "/tmp",
|
|
279
|
+
}
|
|
280
|
+
with patch("prefetch.retain_module.run_retain", return_value={"status": "ok"}), \
|
|
281
|
+
patch("prefetch.HindsightClient", return_value=client), \
|
|
282
|
+
patch("prefetch.get_api_url", return_value="http://fake"):
|
|
283
|
+
wrote = prefetch.run_prefetch(hook_input, self._config())
|
|
284
|
+
self.assertTrue(wrote)
|
|
285
|
+
self.assertTrue(os.path.isfile(recall_buffer._buffer_path(SESSION)))
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
class CurationSharingTests(PrefetchPipelineBase):
|
|
289
|
+
"""F5 — the producer must run recalled candidates through recall's shared
|
|
290
|
+
curation pipeline (demote-drop + `recallMaxMemories` cap) BEFORE buffering,
|
|
291
|
+
so the buffer path cannot inject what the synchronous path would filter."""
|
|
292
|
+
|
|
293
|
+
def _run_with_results(self, results, config):
|
|
294
|
+
_write_transcript(self.transcript_path, [("user", "what do you remember about deploys")])
|
|
295
|
+
client = mock.Mock()
|
|
296
|
+
client.recall.side_effect = lambda *a, **kw: {"results": [dict(m) for m in results]}
|
|
297
|
+
with patch("prefetch.retain_module.run_retain", return_value={"status": "ok"}), \
|
|
298
|
+
patch("prefetch.HindsightClient", return_value=client), \
|
|
299
|
+
patch("prefetch.get_api_url", return_value="http://fake"):
|
|
300
|
+
wrote = prefetch.run_prefetch(self._hook_input("what do you remember about deploys"), config)
|
|
301
|
+
buffered = ""
|
|
302
|
+
if wrote:
|
|
303
|
+
payload = recall_buffer._read_buffer_payload(SESSION) or {}
|
|
304
|
+
buffered = payload.get("context") or ""
|
|
305
|
+
return wrote, buffered
|
|
306
|
+
|
|
307
|
+
def test_demote_tagged_memory_is_dropped_before_buffering(self):
|
|
308
|
+
results = [
|
|
309
|
+
{"text": "keep me visible", "type": "fact", "mentioned_at": "2026-01-01",
|
|
310
|
+
"id": "k1", "scores": {"final": 0.9}},
|
|
311
|
+
{"text": "secret demoted note", "type": "fact", "mentioned_at": "2026-01-01",
|
|
312
|
+
"id": "d1", "scores": {"final": 0.8}, "tags": ["demote-from-recall"]},
|
|
313
|
+
]
|
|
314
|
+
_wrote, buffered = self._run_with_results(results, self._config())
|
|
315
|
+
self.assertIn("keep me visible", buffered)
|
|
316
|
+
self.assertNotIn(
|
|
317
|
+
"secret demoted note", buffered,
|
|
318
|
+
"a demote-from-recall memory must never reach the prefetch buffer",
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
def test_candidate_set_above_cap_is_truncated_before_buffering(self):
|
|
322
|
+
results = [
|
|
323
|
+
{"text": f"candidate {i}", "type": "fact", "mentioned_at": "2026-01-01",
|
|
324
|
+
"id": f"m{i}", "scores": {"final": 1.0 - i * 0.01}}
|
|
325
|
+
for i in range(12)
|
|
326
|
+
]
|
|
327
|
+
config = self._config()
|
|
328
|
+
config["recallMaxMemories"] = 8
|
|
329
|
+
_wrote, buffered = self._run_with_results(results, config)
|
|
330
|
+
injected = sum(1 for i in range(12) if f"candidate {i}" in buffered)
|
|
331
|
+
self.assertEqual(
|
|
332
|
+
injected, 8,
|
|
333
|
+
"the prefetch buffer must honour recallMaxMemories, same as the sync path",
|
|
334
|
+
)
|
|
244
335
|
|
|
245
336
|
|
|
246
337
|
if __name__ == "__main__":
|