dsh-session-recall 0.7.4 → 0.7.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.d.ts CHANGED
@@ -56,6 +56,19 @@ interface RecallConfig {
56
56
  * same project are invisible unless this is set to `false`.
57
57
  */
58
58
  callerTreeOnly?: boolean;
59
+ /**
60
+ * When the session index itself fails (`SESSION_QUERY_PERSISTENCE_FAILED`,
61
+ * e.g. one un-migratable artifact failing every indexed search —
62
+ * deepseek-harness discussion #7995), retry with a direct scan of the
63
+ * persisted session logs instead of failing the call. Default `true`.
64
+ */
65
+ rawScanFallback?: boolean;
66
+ /** Max persisted sessions visited by one degraded raw scan. Default `200`. */
67
+ rawScanMaxSessions?: number;
68
+ /** Wall-clock ceiling (ms) for one degraded raw scan. Default `20000`. */
69
+ rawScanMaxDurationMs?: number;
70
+ /** Compressed-size ceiling (bytes) per session log in a degraded scan. Default 8 MiB. */
71
+ rawScanMaxSessionBytes?: number;
59
72
  }
60
73
  /** Validated, fully defaulted configuration. */
61
74
  interface NormalizedRecallConfig {
@@ -72,6 +85,10 @@ interface NormalizedRecallConfig {
72
85
  readonly recencyHalfLifeDays: number | undefined;
73
86
  readonly pinnedCwds: readonly string[];
74
87
  readonly callerTreeOnly: boolean;
88
+ readonly rawScanFallback: boolean;
89
+ readonly rawScanMaxSessions: number;
90
+ readonly rawScanMaxDurationMs: number;
91
+ readonly rawScanMaxSessionBytes: number;
75
92
  }
76
93
  /** Default, clamp, and cross-check every optional field. */
77
94
  declare function normalizeRecallConfig(config?: RecallConfig): NormalizedRecallConfig;
@@ -107,7 +124,7 @@ interface RecallScope {
107
124
  /** How the matches in a result were produced (v0.5 diagnostics). */
108
125
  interface RecallDiagnostics {
109
126
  /** Which engine produced the matches. */
110
- readonly source: 'fts' | 'cjk-fallback' | 'session-scan';
127
+ readonly source: 'fts' | 'cjk-fallback' | 'session-scan' | 'raw-scan';
111
128
  /** Sessions visited by the fallback scan, when it ran. */
112
129
  readonly scanned?: number;
113
130
  /** The fallback scan budget (`cjkFallbackScanMax`), when a scan ran. */
@@ -244,6 +261,70 @@ declare function cjkFallbackHint(matched: number, enabled: boolean): string | nu
244
261
  /** The zero-hit hint, shown only when both the full-text and substring paths miss. */
245
262
  declare function cjkZeroHitHint(query: string, zeroHits: boolean, enabled: boolean): string | null;
246
263
  //#endregion
264
+ //#region src/resilient.d.ts
265
+ /** Default scan budget (sessions visited per degraded call). */
266
+ declare const RAW_SCAN_DEFAULT_MAX_SESSIONS = 300;
267
+ /** Hard ceiling for the scan budget. */
268
+ declare const RAW_SCAN_MAX_SESSIONS_MAX = 2000;
269
+ /** Where the harness persists sessions: `$DSH_HOME/sessions` (default `~/.dsh/sessions`). */
270
+ declare function discoverSessionsRoot(): string;
271
+ /** True when an error means "the index is unavailable", not "the query was bad". */
272
+ declare function isIndexOutage(error: unknown): boolean;
273
+ interface RawScanOptions {
274
+ /** Sessions root (workspace directories below this). */
275
+ root: string;
276
+ /** Normalized search query; whitespace-separated terms, all must match. */
277
+ query: string;
278
+ /** Restrict to sessions started in this cwd (unless `allProjects`). */
279
+ cwd: string | null;
280
+ /** Ignore the cwd scope. */
281
+ allProjects: boolean;
282
+ /** Restrict to one session id (the `session_id` argument), or null. */
283
+ sessionId: string | null;
284
+ /** Only sessions newer than this many days (0 = no time filter). */
285
+ sinceDays: number;
286
+ /** Only tool events whose tool name contains one of these substrings. */
287
+ tools: readonly string[] | null;
288
+ /** Only failed tool calls (tool results carrying an error flag). */
289
+ errorsOnly: boolean;
290
+ /** Max sessions to return. */
291
+ limit: number;
292
+ /** Scan budget: max sessions visited. */
293
+ maxSessions: number;
294
+ /** Wall-clock ceiling in ms for the whole pass (0 = no ceiling). */
295
+ maxDurationMs: number;
296
+ /** Compressed-size ceiling per session log; larger logs are skipped (pure-JS zstd is slow). */
297
+ maxSessionBytes: number;
298
+ }
299
+ interface RawScanResult {
300
+ items: RecallItem[];
301
+ /** Sessions whose events were actually scanned. */
302
+ scanned: number;
303
+ /** The scan budget that applied. */
304
+ budget: number;
305
+ /** Sessions skipped because their log could not be decompressed or parsed. */
306
+ unreadable: number;
307
+ /** False when the scan stopped early (budget/ceiling): results cover the newest sessions only. */
308
+ coveredAll: boolean;
309
+ /** In-scope sessions skipped because their log exceeded `maxSessionBytes`. */
310
+ skippedOversized: number;
311
+ /** Session directories that existed under the root when the scan started. */
312
+ totalCandidates: number;
313
+ }
314
+ /**
315
+ * Scan persisted session logs directly, honoring the same scope filters as
316
+ * the indexed path. Never throws for per-session problems; returns what it
317
+ * found plus scan counters.
318
+ *
319
+ * Cost model (newest-first): the header line of every session is read
320
+ * through the streaming decompressor (milliseconds each); only sessions
321
+ * passing the scope filters are decompressed fully, and only sessions whose
322
+ * bytes mention every term are parsed line by line. `maxDurationMs` puts a
323
+ * wall-clock ceiling on the pass — when it trips, `coveredAll` is false and
324
+ * the caller should say so (results cover the newest sessions only).
325
+ */
326
+ declare function rawScanSessions(options: RawScanOptions): Promise<RawScanResult>;
327
+ //#endregion
247
328
  //#region src/util.d.ts
248
329
  /** Small pure helpers shared by the recall tool and its renderers. */
249
330
  /** Clamp `n` into the inclusive `[lo, hi]` range. */
@@ -274,4 +355,4 @@ declare const inject: string[];
274
355
  /** Plugin entry: mount the `recall` tool on the global tool registry. */
275
356
  declare function apply(ctx: Context, config?: RecallConfig): void;
276
357
  //#endregion
277
- export { ALL_PROJECTS_POLICIES, type AllProjectsPolicy, type NormalizedRecallConfig, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, type RankableItem, type RankingOptions, type RecallApprovalVerdict, type RecallApprover, type RecallArgs, type RecallBestMatch, type RecallConfig, type RecallDiagnostics, type RecallItem, type RecallQueryEngine, type RecallResult, type RecallScope, type RedactionMode, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, firstLineClipped, formatDate, hasCJK, id8, inject, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
358
+ export { ALL_PROJECTS_POLICIES, type AllProjectsPolicy, type NormalizedRecallConfig, RAW_SCAN_DEFAULT_MAX_SESSIONS, RAW_SCAN_MAX_SESSIONS_MAX, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, type RankableItem, type RankingOptions, type RawScanOptions, type RawScanResult, type RecallApprovalVerdict, type RecallApprover, type RecallArgs, type RecallBestMatch, type RecallConfig, type RecallDiagnostics, type RecallItem, type RecallQueryEngine, type RecallResult, type RecallScope, type RedactionMode, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, discoverSessionsRoot, firstLineClipped, formatDate, hasCJK, id8, inject, isIndexOutage, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, rawScanSessions, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
package/lib/index.js CHANGED
@@ -1,7 +1,420 @@
1
1
  import { defineTool } from "@deepseek-ai/dsh-tools";
2
2
  import { SessionSearchCursor } from "@deepseek-ai/dsh-session-query";
3
3
  import { SessionId } from "@deepseek-ai/dsh-session";
4
+ import { closeSync, openSync, readFileSync, readSync, readdirSync, statSync } from "node:fs";
5
+ import { homedir } from "node:os";
6
+ import { join } from "node:path";
4
7
  import { createHash } from "node:crypto";
8
+ //#region src/util.ts
9
+ /** Small pure helpers shared by the recall tool and its renderers. */
10
+ /** Clamp `n` into the inclusive `[lo, hi]` range. */
11
+ function clamp(n, lo, hi) {
12
+ return Math.min(hi, Math.max(lo, n));
13
+ }
14
+ /**
15
+ * Short session slug for humans: strips the web-profile `session-` prefix and
16
+ * keeps the first 8 characters of what remains.
17
+ */
18
+ function id8(sessionId) {
19
+ return (sessionId.startsWith("session-") ? sessionId.slice(8) : sessionId).slice(0, 8);
20
+ }
21
+ /** Format Unix epoch milliseconds as a local `YYYY-MM-DD` date. */
22
+ function formatDate(epochMs) {
23
+ const d = new Date(epochMs);
24
+ const m = String(d.getMonth() + 1).padStart(2, "0");
25
+ const day = String(d.getDate()).padStart(2, "0");
26
+ return `${d.getFullYear()}-${m}-${day}`;
27
+ }
28
+ const CJK_RE = /[\u{3400}-\u{4DBF}\u{4E00}-\u{9FFF}\u{F900}-\u{FAFF}\u{3040}-\u{30FF}\u{AC00}-\u{D7AF}\u{3000}-\u{303F}]/u;
29
+ /** Whether `text` contains at least one CJK ideograph, kana, or hangul character. */
30
+ function hasCJK(text) {
31
+ return CJK_RE.test(text);
32
+ }
33
+ /** Collapse every whitespace run and trim; used to normalize model-supplied queries. */
34
+ function normalizeQuery(text) {
35
+ return text.trim().replaceAll(/\s+/g, " ");
36
+ }
37
+ /** Split into whitespace-separated terms, dropping empty pieces. */
38
+ function splitTerms(text) {
39
+ return text.split(/\s+/u).filter((term) => term.length > 0);
40
+ }
41
+ /** First line of `text` with control characters stripped, clipped to `limit` code points. */
42
+ function firstLineClipped(text, limit) {
43
+ const line = text.split("\n", 1)[0] ?? "";
44
+ let out = "";
45
+ for (const ch of line) {
46
+ if ((ch.codePointAt(0) ?? 0) < 32) continue;
47
+ out += ch;
48
+ if (out.length >= limit) break;
49
+ }
50
+ return out;
51
+ }
52
+ /**
53
+ * Single-line snippet clipped around the first case-insensitive occurrence of
54
+ * `query`, with ellipses at either end when text was cut. Falls back to a
55
+ * head clip when the query is empty or absent.
56
+ */
57
+ function snippetAround(text, query, limit) {
58
+ const flat = text.replaceAll(/\s+/g, " ");
59
+ const q = normalizeQuery(query).toLowerCase();
60
+ const idx = q === "" ? -1 : flat.toLowerCase().indexOf(q);
61
+ if (idx < 0) return flat.slice(0, limit);
62
+ const pad = Math.floor(limit / 3);
63
+ const start = Math.max(0, idx - pad);
64
+ const end = Math.min(flat.length, start + limit);
65
+ return `${start > 0 ? "…" : ""}${flat.slice(start, end)}${end < flat.length ? "…" : ""}`;
66
+ }
67
+ //#endregion
68
+ //#region src/resilient.ts
69
+ /**
70
+ * Resilient raw-log scan: the recall tool's degraded mode.
71
+ *
72
+ * When the session index itself is unavailable — most notably
73
+ * `SESSION_QUERY_PERSISTENCE_FAILED`, where one un-migratable session
74
+ * artifact fails every indexed search (see deepseek-harness discussion
75
+ * #7995: the v0→v1 migration gate rejects `subagent/descriptor` version 2,
76
+ * the only version the writer emits) — recall degrades to scanning the
77
+ * persisted session logs directly instead of failing the whole call.
78
+ *
79
+ * Design rules, in order:
80
+ * 1. Never throw on a bad session: a session that cannot be decompressed or
81
+ * parsed is counted as unreadable and skipped. The whole point of this
82
+ * module is that no single artifact can take the search down.
83
+ * 2. Scope first, scan second: the header line (cwd, createdAt) is read
84
+ * before the event scan, so cwd/time/session_id filters cost one line.
85
+ * 3. Same match semantics as the indexed path: whitespace-separated terms,
86
+ * all must match; ASCII terms match on word boundaries, CJK terms match
87
+ * as substrings.
88
+ *
89
+ * @module dsh-session-recall/resilient
90
+ */
91
+ let fzstdModule = null;
92
+ async function getFzstd() {
93
+ if (fzstdModule == null) fzstdModule = await import("./esm-DyxhIKe9.js");
94
+ return fzstdModule;
95
+ }
96
+ /** Default scan budget (sessions visited per degraded call). */
97
+ const RAW_SCAN_DEFAULT_MAX_SESSIONS = 300;
98
+ /** Hard ceiling for the scan budget. */
99
+ const RAW_SCAN_MAX_SESSIONS_MAX = 2e3;
100
+ /** Characters of context around the first matched term. */
101
+ const SNIPPET_CHARS$1 = 200;
102
+ /** Where the harness persists sessions: `$DSH_HOME/sessions` (default `~/.dsh/sessions`). */
103
+ function discoverSessionsRoot() {
104
+ const dshHome = process.env.DSH_HOME ?? join(homedir(), ".dsh");
105
+ return join(dshHome, "sessions");
106
+ }
107
+ /** True when an error means "the index is unavailable", not "the query was bad". */
108
+ function isIndexOutage(error) {
109
+ return typeof error === "object" && error !== null && "code" in error && String(error.code) === "SESSION_QUERY_PERSISTENCE_FAILED";
110
+ }
111
+ /** Tolerant single-line JSON parse: garbage lines yield null, never throw. */
112
+ function parseLine(line) {
113
+ try {
114
+ const value = JSON.parse(line);
115
+ return typeof value === "object" && value !== null ? value : null;
116
+ } catch {
117
+ return null;
118
+ }
119
+ }
120
+ /** Pull the searchable text, tool name, and error flag out of one event. */
121
+ function eventFacts(event) {
122
+ const data = event["data"];
123
+ if (typeof data !== "object" || data === null) return null;
124
+ const record = data;
125
+ const type = typeof event["type"] === "string" ? event["type"] : "";
126
+ const texts = [];
127
+ const pushBlocks = (blocks) => {
128
+ if (!Array.isArray(blocks)) return;
129
+ for (const block of blocks) if (typeof block === "string") texts.push(block);
130
+ else if (typeof block === "object" && block !== null) {
131
+ const b = block;
132
+ if (typeof b["text"] === "string") texts.push(b["text"]);
133
+ if (Array.isArray(b["content"])) pushBlocks(b["content"]);
134
+ }
135
+ };
136
+ let tool = null;
137
+ if (type === "tool/call" && typeof record["name"] === "string") {
138
+ tool = record["name"];
139
+ if (typeof record["arguments"] === "string") texts.push(record["arguments"]);
140
+ }
141
+ const message = record["message"];
142
+ if (typeof message === "object" && message !== null) {
143
+ const m = message;
144
+ if (Array.isArray(m["content"])) pushBlocks(m["content"]);
145
+ }
146
+ if (Array.isArray(record["content"])) pushBlocks(record["content"]);
147
+ if (texts.length === 0 && typeof record["label"] === "string") texts.push(record["label"]);
148
+ const isError = typeof message === "object" && message !== null && (message["isError"] === true || message["is_error"] === true) || record["isError"] === true || record["is_error"] === true;
149
+ const text = texts.join("\n");
150
+ if (text === "" && tool == null) return null;
151
+ return {
152
+ text,
153
+ tool,
154
+ isError
155
+ };
156
+ }
157
+ /** All terms must match: ASCII on word boundaries, CJK as substrings. */
158
+ function textMatches(text, terms) {
159
+ if (terms.length === 0) return false;
160
+ const haystack = text.toLowerCase();
161
+ for (const term of terms) {
162
+ const needle = term.toLowerCase();
163
+ if (/[\u3400-\u9fff\uf900-\ufaff\u3040-\u30ff]/.test(term)) {
164
+ if (!haystack.includes(needle)) return false;
165
+ } else {
166
+ const escaped = needle.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
167
+ if (!new RegExp(`(^|[^a-z0-9_])${escaped}([^a-z0-9_]|$)`, "u").test(haystack)) return false;
168
+ }
169
+ }
170
+ return true;
171
+ }
172
+ /** List session directories under the root, newest first. */
173
+ function listSessionDirs(root) {
174
+ const out = [];
175
+ let workspaces = [];
176
+ try {
177
+ workspaces = readdirSync(root, { withFileTypes: true }).filter((e) => e.isDirectory()).map((e) => e.name);
178
+ } catch {
179
+ return out;
180
+ }
181
+ for (const ws of workspaces) {
182
+ const wsDir = join(root, ws);
183
+ let sessions = [];
184
+ try {
185
+ sessions = readdirSync(wsDir, { withFileTypes: true }).filter((e) => e.isDirectory()).map((e) => e.name);
186
+ } catch {
187
+ continue;
188
+ }
189
+ for (const id of sessions) {
190
+ const dir = join(wsDir, id);
191
+ try {
192
+ out.push({
193
+ dir,
194
+ mtimeMs: statSync(dir).mtimeMs
195
+ });
196
+ } catch {}
197
+ }
198
+ }
199
+ out.sort((a, b) => b.mtimeMs - a.mtimeMs);
200
+ return out;
201
+ }
202
+ /** Decompress one session log fully. Plain `.jsonl` is read as-is; `.zstd` via fzstd. */
203
+ async function readSessionLog(dir) {
204
+ for (const name of ["session.jsonl.zstd", "session.jsonl"]) {
205
+ let bytes;
206
+ try {
207
+ bytes = readFileSync(join(dir, name));
208
+ } catch {
209
+ continue;
210
+ }
211
+ if (name.endsWith(".zstd")) try {
212
+ const fzstd = await getFzstd();
213
+ return Buffer.from(fzstd.decompress(new Uint8Array(bytes)));
214
+ } catch {
215
+ return null;
216
+ }
217
+ return bytes;
218
+ }
219
+ return null;
220
+ }
221
+ /**
222
+ * Read just the session header line without decompressing the whole log:
223
+ * zstd input is pushed through the streaming decompressor in small chunks
224
+ * and stops at the first newline. Costs milliseconds per session, which is
225
+ * what makes scanning a whole store affordable.
226
+ */
227
+ async function readSessionHeaderStreaming(dir) {
228
+ const fromLine = (line) => {
229
+ const header = parseLine(line);
230
+ if (header == null) return null;
231
+ const id = typeof header["id"] === "string" ? header["id"] : null;
232
+ if (id == null) return null;
233
+ return {
234
+ id,
235
+ createdAt: typeof header["createdAt"] === "number" ? header["createdAt"] : NaN,
236
+ cwd: typeof header["cwd"] === "string" ? header["cwd"] : null
237
+ };
238
+ };
239
+ for (const name of ["session.jsonl.zstd", "session.jsonl"]) {
240
+ const path = join(dir, name);
241
+ let fd;
242
+ try {
243
+ fd = openSync(path, "r");
244
+ } catch {
245
+ continue;
246
+ }
247
+ try {
248
+ const chunks = [];
249
+ const input = Buffer.alloc(16384);
250
+ let fzstd = null;
251
+ for (;;) {
252
+ let bytes;
253
+ try {
254
+ bytes = readSync(fd, input, 0, input.length, null);
255
+ } catch {
256
+ return null;
257
+ }
258
+ if (bytes <= 0) break;
259
+ if (name.endsWith(".zstd")) {
260
+ if (fzstd == null) try {
261
+ fzstd = await getFzstd();
262
+ } catch {
263
+ return null;
264
+ }
265
+ try {
266
+ new fzstd.Decompress((chunk) => {
267
+ chunks.push(Buffer.from(chunk));
268
+ }).push(input.subarray(0, bytes));
269
+ } catch {
270
+ return null;
271
+ }
272
+ } else chunks.push(Buffer.from(input.subarray(0, bytes)));
273
+ const soFar = Buffer.concat(chunks);
274
+ const nl = soFar.indexOf(10);
275
+ if (nl >= 0) return fromLine(soFar.subarray(0, nl).toString("utf-8"));
276
+ if (soFar.length > 1048576) return null;
277
+ }
278
+ return fromLine(Buffer.concat(chunks).toString("utf-8"));
279
+ } finally {
280
+ closeSync(fd);
281
+ }
282
+ }
283
+ return null;
284
+ }
285
+ /** The persisted log file of a session dir: path + compressed size, or null when none exists. */
286
+ function sessionLogPath(dir) {
287
+ for (const name of ["session.jsonl.zstd", "session.jsonl"]) try {
288
+ const path = join(dir, name);
289
+ return {
290
+ path,
291
+ size: statSync(path).size
292
+ };
293
+ } catch {
294
+ continue;
295
+ }
296
+ return null;
297
+ }
298
+ /**
299
+ * Cheap superset gate: every term must appear in the decompressed bytes
300
+ * (exact or lowercase form) before the expensive line-by-line parse runs.
301
+ * Word-boundary semantics are still enforced by `textMatches` later.
302
+ */
303
+ function bytesMentionAllTerms(buf, terms) {
304
+ for (const term of terms) if (buf.indexOf(term, 0, "utf-8") < 0 && buf.indexOf(term.toLowerCase(), 0, "utf-8") < 0) return false;
305
+ return true;
306
+ }
307
+ /**
308
+ * Scan persisted session logs directly, honoring the same scope filters as
309
+ * the indexed path. Never throws for per-session problems; returns what it
310
+ * found plus scan counters.
311
+ *
312
+ * Cost model (newest-first): the header line of every session is read
313
+ * through the streaming decompressor (milliseconds each); only sessions
314
+ * passing the scope filters are decompressed fully, and only sessions whose
315
+ * bytes mention every term are parsed line by line. `maxDurationMs` puts a
316
+ * wall-clock ceiling on the pass — when it trips, `coveredAll` is false and
317
+ * the caller should say so (results cover the newest sessions only).
318
+ */
319
+ async function rawScanSessions(options) {
320
+ const terms = splitTerms(options.query);
321
+ const sinceMs = options.sinceDays > 0 ? Date.now() - options.sinceDays * 24 * 60 * 60 * 1e3 : 0;
322
+ const toolsFilter = options.tools != null && options.tools.length > 0 ? options.tools.map((t) => t.toLowerCase()) : null;
323
+ const deadline = options.maxDurationMs > 0 ? Date.now() + options.maxDurationMs : 0;
324
+ const candidates = listSessionDirs(options.root);
325
+ const items = [];
326
+ let scanned = 0;
327
+ let unreadable = 0;
328
+ let skippedOversized = 0;
329
+ let coveredAll = true;
330
+ for (const candidate of candidates) {
331
+ if (items.length >= options.limit) break;
332
+ if (scanned >= options.maxSessions || deadline > 0 && Date.now() > deadline) {
333
+ coveredAll = false;
334
+ break;
335
+ }
336
+ const logInfo = sessionLogPath(candidate.dir);
337
+ if (logInfo == null) continue;
338
+ const header = await readSessionHeaderStreaming(candidate.dir);
339
+ if (header == null) {
340
+ unreadable += 1;
341
+ continue;
342
+ }
343
+ if (options.sessionId != null && header.id !== options.sessionId) continue;
344
+ if (!options.allProjects && options.cwd != null && header.cwd !== options.cwd) continue;
345
+ if (sinceMs > 0 && !(header.createdAt >= sinceMs)) continue;
346
+ if (logInfo.size > options.maxSessionBytes) {
347
+ skippedOversized += 1;
348
+ continue;
349
+ }
350
+ const log = await readSessionLog(candidate.dir);
351
+ if (log == null) {
352
+ unreadable += 1;
353
+ continue;
354
+ }
355
+ scanned += 1;
356
+ if (!bytesMentionAllTerms(log, terms)) continue;
357
+ const callTools = /* @__PURE__ */ new Map();
358
+ let best = null;
359
+ for (const line of log.toString("utf-8").split("\n")) {
360
+ if (line === "") continue;
361
+ const event = parseLine(line);
362
+ if (event == null) continue;
363
+ const type = typeof event["type"] === "string" ? event["type"] : "";
364
+ const facts = eventFacts(event);
365
+ if (facts == null) continue;
366
+ let tool = facts.tool;
367
+ if (tool == null && type === "tool/result") {
368
+ const source = event["data"]?.["message"];
369
+ const callId = typeof source === "object" && source !== null ? source["source"]?.["callId"] : void 0;
370
+ if (typeof callId === "string") tool = callTools.get(callId) ?? null;
371
+ } else if (tool != null) {
372
+ const data = event["data"];
373
+ const callId = typeof data?.["callId"] === "string" ? data["callId"] : null;
374
+ if (callId != null) callTools.set(callId, tool);
375
+ }
376
+ if (toolsFilter != null || options.errorsOnly) {
377
+ if (!(type === "tool/call" || type === "tool/result")) continue;
378
+ if (toolsFilter != null && (tool == null || !toolsFilter.some((t) => tool.toLowerCase().includes(t)))) continue;
379
+ if (options.errorsOnly && !(type === "tool/result" && facts.isError)) continue;
380
+ }
381
+ if (!textMatches(facts.text, terms)) continue;
382
+ const seq = typeof event["seq"] === "number" ? event["seq"] : 0;
383
+ const time = typeof event["time"] === "number" ? event["time"] : 0;
384
+ if (best == null || seq < best.seq) best = {
385
+ seq,
386
+ type,
387
+ time,
388
+ text: facts.text
389
+ };
390
+ }
391
+ if (best != null) items.push({
392
+ sessionId: header.id,
393
+ id8: id8(header.id),
394
+ title: null,
395
+ createdAt: Number.isFinite(header.createdAt) ? header.createdAt : 0,
396
+ cwd: header.cwd,
397
+ live: false,
398
+ persisted: true,
399
+ bestMatch: {
400
+ seq: best.seq,
401
+ type: best.type,
402
+ time: best.time,
403
+ snippet: snippetAround(best.text, terms[0] ?? "", SNIPPET_CHARS$1)
404
+ }
405
+ });
406
+ }
407
+ return {
408
+ items,
409
+ scanned,
410
+ budget: options.maxSessions,
411
+ unreadable,
412
+ skippedOversized,
413
+ coveredAll,
414
+ totalCandidates: candidates.length
415
+ };
416
+ }
417
+ //#endregion
5
418
  //#region src/redact.ts
6
419
  /**
7
420
  * Redaction for recall output text (snippets and titles).
@@ -98,7 +511,11 @@ function normalizeRecallConfig(config) {
98
511
  allProjectsPolicy: normalizePolicy(config?.allProjectsPolicy),
99
512
  recencyHalfLifeDays: typeof config?.recencyHalfLifeDays === "number" && Number.isFinite(config.recencyHalfLifeDays) && config.recencyHalfLifeDays > 0 ? Math.min(3650, Math.trunc(config.recencyHalfLifeDays)) : void 0,
100
513
  pinnedCwds: stringList(config?.pinnedCwds),
101
- callerTreeOnly: config?.callerTreeOnly !== false
514
+ callerTreeOnly: config?.callerTreeOnly !== false,
515
+ rawScanFallback: config?.rawScanFallback !== false,
516
+ rawScanMaxSessions: intIn(config?.rawScanMaxSessions, 200, 1, 2e3),
517
+ rawScanMaxDurationMs: intIn(config?.rawScanMaxDurationMs, 2e4, 1e3, 12e4),
518
+ rawScanMaxSessionBytes: intIn(config?.rawScanMaxSessionBytes, 8388608, 262144, 67108864)
102
519
  };
103
520
  }
104
521
  /** Whether a session cwd is searchable under the allowlist/denylist policy. */
@@ -146,66 +563,6 @@ function rankItems(items, options) {
146
563
  return decorated.map((entry) => entry.item);
147
564
  }
148
565
  //#endregion
149
- //#region src/util.ts
150
- /** Small pure helpers shared by the recall tool and its renderers. */
151
- /** Clamp `n` into the inclusive `[lo, hi]` range. */
152
- function clamp(n, lo, hi) {
153
- return Math.min(hi, Math.max(lo, n));
154
- }
155
- /**
156
- * Short session slug for humans: strips the web-profile `session-` prefix and
157
- * keeps the first 8 characters of what remains.
158
- */
159
- function id8(sessionId) {
160
- return (sessionId.startsWith("session-") ? sessionId.slice(8) : sessionId).slice(0, 8);
161
- }
162
- /** Format Unix epoch milliseconds as a local `YYYY-MM-DD` date. */
163
- function formatDate(epochMs) {
164
- const d = new Date(epochMs);
165
- const m = String(d.getMonth() + 1).padStart(2, "0");
166
- const day = String(d.getDate()).padStart(2, "0");
167
- return `${d.getFullYear()}-${m}-${day}`;
168
- }
169
- const CJK_RE = /[\u{3400}-\u{4DBF}\u{4E00}-\u{9FFF}\u{F900}-\u{FAFF}\u{3040}-\u{30FF}\u{AC00}-\u{D7AF}\u{3000}-\u{303F}]/u;
170
- /** Whether `text` contains at least one CJK ideograph, kana, or hangul character. */
171
- function hasCJK(text) {
172
- return CJK_RE.test(text);
173
- }
174
- /** Collapse every whitespace run and trim; used to normalize model-supplied queries. */
175
- function normalizeQuery(text) {
176
- return text.trim().replaceAll(/\s+/g, " ");
177
- }
178
- /** Split into whitespace-separated terms, dropping empty pieces. */
179
- function splitTerms(text) {
180
- return text.split(/\s+/u).filter((term) => term.length > 0);
181
- }
182
- /** First line of `text` with control characters stripped, clipped to `limit` code points. */
183
- function firstLineClipped(text, limit) {
184
- const line = text.split("\n", 1)[0] ?? "";
185
- let out = "";
186
- for (const ch of line) {
187
- if ((ch.codePointAt(0) ?? 0) < 32) continue;
188
- out += ch;
189
- if (out.length >= limit) break;
190
- }
191
- return out;
192
- }
193
- /**
194
- * Single-line snippet clipped around the first case-insensitive occurrence of
195
- * `query`, with ellipses at either end when text was cut. Falls back to a
196
- * head clip when the query is empty or absent.
197
- */
198
- function snippetAround(text, query, limit) {
199
- const flat = text.replaceAll(/\s+/g, " ");
200
- const q = normalizeQuery(query).toLowerCase();
201
- const idx = q === "" ? -1 : flat.toLowerCase().indexOf(q);
202
- if (idx < 0) return flat.slice(0, limit);
203
- const pad = Math.floor(limit / 3);
204
- const start = Math.max(0, idx - pad);
205
- const end = Math.min(flat.length, start + limit);
206
- return `${start > 0 ? "…" : ""}${flat.slice(start, end)}${end < flat.length ? "…" : ""}`;
207
- }
208
- //#endregion
209
566
  //#region src/render.ts
210
567
  const SNIPPET_CHARS = 120;
211
568
  function sessionLabel(result, index) {
@@ -545,7 +902,8 @@ const recallOutputSchema = {
545
902
  enum: [
546
903
  "fts",
547
904
  "cjk-fallback",
548
- "session-scan"
905
+ "session-scan",
906
+ "raw-scan"
549
907
  ]
550
908
  },
551
909
  scanned: { type: "integer" },
@@ -605,7 +963,7 @@ function createRecallTool(config, engine, approver) {
605
963
  return defineTool({
606
964
  name: "recall",
607
965
  description: RECALL_TOOL_DESCRIPTION,
608
- timeoutMs: 1e4,
966
+ timeoutMs: 3e4,
609
967
  isConcurrencySafe: () => true,
610
968
  parameters: {
611
969
  query: {
@@ -815,6 +1173,42 @@ function createRecallTool(config, engine, approver) {
815
1173
  diagnostics
816
1174
  };
817
1175
  } catch (error) {
1176
+ if (cfg.rawScanFallback && isIndexOutage(error)) try {
1177
+ const scan = await rawScanSessions({
1178
+ root: discoverSessionsRoot(),
1179
+ query,
1180
+ cwd: agentCwd,
1181
+ allProjects: wantAll,
1182
+ sessionId: args.session_id ?? null,
1183
+ sinceDays: args.since_days ?? 0,
1184
+ tools: args.tools ?? null,
1185
+ errorsOnly: args.errors_only === true,
1186
+ limit,
1187
+ maxSessions: cfg.rawScanMaxSessions,
1188
+ maxDurationMs: cfg.rawScanMaxDurationMs,
1189
+ maxSessionBytes: cfg.rawScanMaxSessionBytes
1190
+ });
1191
+ let scanItems = scan.items.filter((item) => cwdAllowed(item.cwd, cfg));
1192
+ if (allowedIds != null) scanItems = scanItems.filter((item) => allowedIds.has(item.sessionId));
1193
+ scanItems = scanItems.slice(0, limit);
1194
+ const red = applyRedaction(scanItems);
1195
+ return {
1196
+ query,
1197
+ scope,
1198
+ count: red.items.length,
1199
+ hasMore: false,
1200
+ items: red.items,
1201
+ nextCursor: null,
1202
+ hint: joinHints(`degraded mode: the session index is unavailable (${errCode(error) ?? "persistence failure"}), so recall scanned ${scan.scanned} persisted session log(s) directly — order is by session recency, live sessions are not included${scan.unreadable > 0 ? `, ${scan.unreadable} unreadable log(s) were skipped` : ""}${scan.skippedOversized > 0 ? `, ${scan.skippedOversized} oversized log(s) beyond rawScanMaxSessionBytes were skipped` : ""}${scan.coveredAll ? "" : `, and the wall-clock budget stopped the pass after the newest ${scan.scanned} of ${scan.totalCandidates} sessions — narrow the scope (since_days, session_id) or raise rawScanMaxDurationMs to reach older history`}.`, red.hint),
1203
+ redacted: red.redacted,
1204
+ diagnostics: {
1205
+ source: "raw-scan",
1206
+ scanned: scan.scanned,
1207
+ scanBudget: scan.budget,
1208
+ ranked: false
1209
+ }
1210
+ };
1211
+ } catch {}
818
1212
  return recallError(query, error);
819
1213
  }
820
1214
  },
@@ -872,4 +1266,4 @@ function apply(ctx, config) {
872
1266
  }, "session-recall lifecycle");
873
1267
  }
874
1268
  //#endregion
875
- export { ALL_PROJECTS_POLICIES, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, firstLineClipped, formatDate, hasCJK, id8, inject, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
1269
+ export { ALL_PROJECTS_POLICIES, RAW_SCAN_DEFAULT_MAX_SESSIONS, RAW_SCAN_MAX_SESSIONS_MAX, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, discoverSessionsRoot, firstLineClipped, formatDate, hasCJK, id8, inject, isIndexOutage, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, rawScanSessions, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };