dsh-session-recall 0.7.4 → 0.7.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/README.zh.md +14 -0
- package/lib/esm-DyxhIKe9.js +662 -0
- package/lib/index.d.ts +83 -2
- package/lib/index.js +458 -64
- package/package.json +2 -1
package/lib/index.d.ts
CHANGED
|
@@ -56,6 +56,19 @@ interface RecallConfig {
|
|
|
56
56
|
* same project are invisible unless this is set to `false`.
|
|
57
57
|
*/
|
|
58
58
|
callerTreeOnly?: boolean;
|
|
59
|
+
/**
|
|
60
|
+
* When the session index itself fails (`SESSION_QUERY_PERSISTENCE_FAILED`,
|
|
61
|
+
* e.g. one un-migratable artifact failing every indexed search —
|
|
62
|
+
* deepseek-harness discussion #7995), retry with a direct scan of the
|
|
63
|
+
* persisted session logs instead of failing the call. Default `true`.
|
|
64
|
+
*/
|
|
65
|
+
rawScanFallback?: boolean;
|
|
66
|
+
/** Max persisted sessions visited by one degraded raw scan. Default `200`. */
|
|
67
|
+
rawScanMaxSessions?: number;
|
|
68
|
+
/** Wall-clock ceiling (ms) for one degraded raw scan. Default `20000`. */
|
|
69
|
+
rawScanMaxDurationMs?: number;
|
|
70
|
+
/** Compressed-size ceiling (bytes) per session log in a degraded scan. Default 8 MiB. */
|
|
71
|
+
rawScanMaxSessionBytes?: number;
|
|
59
72
|
}
|
|
60
73
|
/** Validated, fully defaulted configuration. */
|
|
61
74
|
interface NormalizedRecallConfig {
|
|
@@ -72,6 +85,10 @@ interface NormalizedRecallConfig {
|
|
|
72
85
|
readonly recencyHalfLifeDays: number | undefined;
|
|
73
86
|
readonly pinnedCwds: readonly string[];
|
|
74
87
|
readonly callerTreeOnly: boolean;
|
|
88
|
+
readonly rawScanFallback: boolean;
|
|
89
|
+
readonly rawScanMaxSessions: number;
|
|
90
|
+
readonly rawScanMaxDurationMs: number;
|
|
91
|
+
readonly rawScanMaxSessionBytes: number;
|
|
75
92
|
}
|
|
76
93
|
/** Default, clamp, and cross-check every optional field. */
|
|
77
94
|
declare function normalizeRecallConfig(config?: RecallConfig): NormalizedRecallConfig;
|
|
@@ -107,7 +124,7 @@ interface RecallScope {
|
|
|
107
124
|
/** How the matches in a result were produced (v0.5 diagnostics). */
|
|
108
125
|
interface RecallDiagnostics {
|
|
109
126
|
/** Which engine produced the matches. */
|
|
110
|
-
readonly source: 'fts' | 'cjk-fallback' | 'session-scan';
|
|
127
|
+
readonly source: 'fts' | 'cjk-fallback' | 'session-scan' | 'raw-scan';
|
|
111
128
|
/** Sessions visited by the fallback scan, when it ran. */
|
|
112
129
|
readonly scanned?: number;
|
|
113
130
|
/** The fallback scan budget (`cjkFallbackScanMax`), when a scan ran. */
|
|
@@ -244,6 +261,70 @@ declare function cjkFallbackHint(matched: number, enabled: boolean): string | nu
|
|
|
244
261
|
/** The zero-hit hint, shown only when both the full-text and substring paths miss. */
|
|
245
262
|
declare function cjkZeroHitHint(query: string, zeroHits: boolean, enabled: boolean): string | null;
|
|
246
263
|
//#endregion
|
|
264
|
+
//#region src/resilient.d.ts
|
|
265
|
+
/** Default scan budget (sessions visited per degraded call). */
|
|
266
|
+
declare const RAW_SCAN_DEFAULT_MAX_SESSIONS = 300;
|
|
267
|
+
/** Hard ceiling for the scan budget. */
|
|
268
|
+
declare const RAW_SCAN_MAX_SESSIONS_MAX = 2000;
|
|
269
|
+
/** Where the harness persists sessions: `$DSH_HOME/sessions` (default `~/.dsh/sessions`). */
|
|
270
|
+
declare function discoverSessionsRoot(): string;
|
|
271
|
+
/** True when an error means "the index is unavailable", not "the query was bad". */
|
|
272
|
+
declare function isIndexOutage(error: unknown): boolean;
|
|
273
|
+
interface RawScanOptions {
|
|
274
|
+
/** Sessions root (workspace directories below this). */
|
|
275
|
+
root: string;
|
|
276
|
+
/** Normalized search query; whitespace-separated terms, all must match. */
|
|
277
|
+
query: string;
|
|
278
|
+
/** Restrict to sessions started in this cwd (unless `allProjects`). */
|
|
279
|
+
cwd: string | null;
|
|
280
|
+
/** Ignore the cwd scope. */
|
|
281
|
+
allProjects: boolean;
|
|
282
|
+
/** Restrict to one session id (the `session_id` argument), or null. */
|
|
283
|
+
sessionId: string | null;
|
|
284
|
+
/** Only sessions newer than this many days (0 = no time filter). */
|
|
285
|
+
sinceDays: number;
|
|
286
|
+
/** Only tool events whose tool name contains one of these substrings. */
|
|
287
|
+
tools: readonly string[] | null;
|
|
288
|
+
/** Only failed tool calls (tool results carrying an error flag). */
|
|
289
|
+
errorsOnly: boolean;
|
|
290
|
+
/** Max sessions to return. */
|
|
291
|
+
limit: number;
|
|
292
|
+
/** Scan budget: max sessions visited. */
|
|
293
|
+
maxSessions: number;
|
|
294
|
+
/** Wall-clock ceiling in ms for the whole pass (0 = no ceiling). */
|
|
295
|
+
maxDurationMs: number;
|
|
296
|
+
/** Compressed-size ceiling per session log; larger logs are skipped (pure-JS zstd is slow). */
|
|
297
|
+
maxSessionBytes: number;
|
|
298
|
+
}
|
|
299
|
+
interface RawScanResult {
|
|
300
|
+
items: RecallItem[];
|
|
301
|
+
/** Sessions whose events were actually scanned. */
|
|
302
|
+
scanned: number;
|
|
303
|
+
/** The scan budget that applied. */
|
|
304
|
+
budget: number;
|
|
305
|
+
/** Sessions skipped because their log could not be decompressed or parsed. */
|
|
306
|
+
unreadable: number;
|
|
307
|
+
/** False when the scan stopped early (budget/ceiling): results cover the newest sessions only. */
|
|
308
|
+
coveredAll: boolean;
|
|
309
|
+
/** In-scope sessions skipped because their log exceeded `maxSessionBytes`. */
|
|
310
|
+
skippedOversized: number;
|
|
311
|
+
/** Session directories that existed under the root when the scan started. */
|
|
312
|
+
totalCandidates: number;
|
|
313
|
+
}
|
|
314
|
+
/**
|
|
315
|
+
* Scan persisted session logs directly, honoring the same scope filters as
|
|
316
|
+
* the indexed path. Never throws for per-session problems; returns what it
|
|
317
|
+
* found plus scan counters.
|
|
318
|
+
*
|
|
319
|
+
* Cost model (newest-first): the header line of every session is read
|
|
320
|
+
* through the streaming decompressor (milliseconds each); only sessions
|
|
321
|
+
* passing the scope filters are decompressed fully, and only sessions whose
|
|
322
|
+
* bytes mention every term are parsed line by line. `maxDurationMs` puts a
|
|
323
|
+
* wall-clock ceiling on the pass — when it trips, `coveredAll` is false and
|
|
324
|
+
* the caller should say so (results cover the newest sessions only).
|
|
325
|
+
*/
|
|
326
|
+
declare function rawScanSessions(options: RawScanOptions): Promise<RawScanResult>;
|
|
327
|
+
//#endregion
|
|
247
328
|
//#region src/util.d.ts
|
|
248
329
|
/** Small pure helpers shared by the recall tool and its renderers. */
|
|
249
330
|
/** Clamp `n` into the inclusive `[lo, hi]` range. */
|
|
@@ -274,4 +355,4 @@ declare const inject: string[];
|
|
|
274
355
|
/** Plugin entry: mount the `recall` tool on the global tool registry. */
|
|
275
356
|
declare function apply(ctx: Context, config?: RecallConfig): void;
|
|
276
357
|
//#endregion
|
|
277
|
-
export { ALL_PROJECTS_POLICIES, type AllProjectsPolicy, type NormalizedRecallConfig, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, type RankableItem, type RankingOptions, type RecallApprovalVerdict, type RecallApprover, type RecallArgs, type RecallBestMatch, type RecallConfig, type RecallDiagnostics, type RecallItem, type RecallQueryEngine, type RecallResult, type RecallScope, type RedactionMode, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, firstLineClipped, formatDate, hasCJK, id8, inject, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
|
|
358
|
+
export { ALL_PROJECTS_POLICIES, type AllProjectsPolicy, type NormalizedRecallConfig, RAW_SCAN_DEFAULT_MAX_SESSIONS, RAW_SCAN_MAX_SESSIONS_MAX, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, type RankableItem, type RankingOptions, type RawScanOptions, type RawScanResult, type RecallApprovalVerdict, type RecallApprover, type RecallArgs, type RecallBestMatch, type RecallConfig, type RecallDiagnostics, type RecallItem, type RecallQueryEngine, type RecallResult, type RecallScope, type RedactionMode, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, discoverSessionsRoot, firstLineClipped, formatDate, hasCJK, id8, inject, isIndexOutage, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, rawScanSessions, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
|
package/lib/index.js
CHANGED
|
@@ -1,7 +1,420 @@
|
|
|
1
1
|
import { defineTool } from "@deepseek-ai/dsh-tools";
|
|
2
2
|
import { SessionSearchCursor } from "@deepseek-ai/dsh-session-query";
|
|
3
3
|
import { SessionId } from "@deepseek-ai/dsh-session";
|
|
4
|
+
import { closeSync, openSync, readFileSync, readSync, readdirSync, statSync } from "node:fs";
|
|
5
|
+
import { homedir } from "node:os";
|
|
6
|
+
import { join } from "node:path";
|
|
4
7
|
import { createHash } from "node:crypto";
|
|
8
|
+
//#region src/util.ts
|
|
9
|
+
/** Small pure helpers shared by the recall tool and its renderers. */
|
|
10
|
+
/** Clamp `n` into the inclusive `[lo, hi]` range. */
|
|
11
|
+
function clamp(n, lo, hi) {
|
|
12
|
+
return Math.min(hi, Math.max(lo, n));
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Short session slug for humans: strips the web-profile `session-` prefix and
|
|
16
|
+
* keeps the first 8 characters of what remains.
|
|
17
|
+
*/
|
|
18
|
+
function id8(sessionId) {
|
|
19
|
+
return (sessionId.startsWith("session-") ? sessionId.slice(8) : sessionId).slice(0, 8);
|
|
20
|
+
}
|
|
21
|
+
/** Format Unix epoch milliseconds as a local `YYYY-MM-DD` date. */
|
|
22
|
+
function formatDate(epochMs) {
|
|
23
|
+
const d = new Date(epochMs);
|
|
24
|
+
const m = String(d.getMonth() + 1).padStart(2, "0");
|
|
25
|
+
const day = String(d.getDate()).padStart(2, "0");
|
|
26
|
+
return `${d.getFullYear()}-${m}-${day}`;
|
|
27
|
+
}
|
|
28
|
+
const CJK_RE = /[\u{3400}-\u{4DBF}\u{4E00}-\u{9FFF}\u{F900}-\u{FAFF}\u{3040}-\u{30FF}\u{AC00}-\u{D7AF}\u{3000}-\u{303F}]/u;
|
|
29
|
+
/** Whether `text` contains at least one CJK ideograph, kana, or hangul character. */
|
|
30
|
+
function hasCJK(text) {
|
|
31
|
+
return CJK_RE.test(text);
|
|
32
|
+
}
|
|
33
|
+
/** Collapse every whitespace run and trim; used to normalize model-supplied queries. */
|
|
34
|
+
function normalizeQuery(text) {
|
|
35
|
+
return text.trim().replaceAll(/\s+/g, " ");
|
|
36
|
+
}
|
|
37
|
+
/** Split into whitespace-separated terms, dropping empty pieces. */
|
|
38
|
+
function splitTerms(text) {
|
|
39
|
+
return text.split(/\s+/u).filter((term) => term.length > 0);
|
|
40
|
+
}
|
|
41
|
+
/** First line of `text` with control characters stripped, clipped to `limit` code points. */
|
|
42
|
+
function firstLineClipped(text, limit) {
|
|
43
|
+
const line = text.split("\n", 1)[0] ?? "";
|
|
44
|
+
let out = "";
|
|
45
|
+
for (const ch of line) {
|
|
46
|
+
if ((ch.codePointAt(0) ?? 0) < 32) continue;
|
|
47
|
+
out += ch;
|
|
48
|
+
if (out.length >= limit) break;
|
|
49
|
+
}
|
|
50
|
+
return out;
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* Single-line snippet clipped around the first case-insensitive occurrence of
|
|
54
|
+
* `query`, with ellipses at either end when text was cut. Falls back to a
|
|
55
|
+
* head clip when the query is empty or absent.
|
|
56
|
+
*/
|
|
57
|
+
function snippetAround(text, query, limit) {
|
|
58
|
+
const flat = text.replaceAll(/\s+/g, " ");
|
|
59
|
+
const q = normalizeQuery(query).toLowerCase();
|
|
60
|
+
const idx = q === "" ? -1 : flat.toLowerCase().indexOf(q);
|
|
61
|
+
if (idx < 0) return flat.slice(0, limit);
|
|
62
|
+
const pad = Math.floor(limit / 3);
|
|
63
|
+
const start = Math.max(0, idx - pad);
|
|
64
|
+
const end = Math.min(flat.length, start + limit);
|
|
65
|
+
return `${start > 0 ? "…" : ""}${flat.slice(start, end)}${end < flat.length ? "…" : ""}`;
|
|
66
|
+
}
|
|
67
|
+
//#endregion
|
|
68
|
+
//#region src/resilient.ts
|
|
69
|
+
/**
|
|
70
|
+
* Resilient raw-log scan: the recall tool's degraded mode.
|
|
71
|
+
*
|
|
72
|
+
* When the session index itself is unavailable — most notably
|
|
73
|
+
* `SESSION_QUERY_PERSISTENCE_FAILED`, where one un-migratable session
|
|
74
|
+
* artifact fails every indexed search (see deepseek-harness discussion
|
|
75
|
+
* #7995: the v0→v1 migration gate rejects `subagent/descriptor` version 2,
|
|
76
|
+
* the only version the writer emits) — recall degrades to scanning the
|
|
77
|
+
* persisted session logs directly instead of failing the whole call.
|
|
78
|
+
*
|
|
79
|
+
* Design rules, in order:
|
|
80
|
+
* 1. Never throw on a bad session: a session that cannot be decompressed or
|
|
81
|
+
* parsed is counted as unreadable and skipped. The whole point of this
|
|
82
|
+
* module is that no single artifact can take the search down.
|
|
83
|
+
* 2. Scope first, scan second: the header line (cwd, createdAt) is read
|
|
84
|
+
* before the event scan, so cwd/time/session_id filters cost one line.
|
|
85
|
+
* 3. Same match semantics as the indexed path: whitespace-separated terms,
|
|
86
|
+
* all must match; ASCII terms match on word boundaries, CJK terms match
|
|
87
|
+
* as substrings.
|
|
88
|
+
*
|
|
89
|
+
* @module dsh-session-recall/resilient
|
|
90
|
+
*/
|
|
91
|
+
let fzstdModule = null;
|
|
92
|
+
async function getFzstd() {
|
|
93
|
+
if (fzstdModule == null) fzstdModule = await import("./esm-DyxhIKe9.js");
|
|
94
|
+
return fzstdModule;
|
|
95
|
+
}
|
|
96
|
+
/** Default scan budget (sessions visited per degraded call). */
|
|
97
|
+
const RAW_SCAN_DEFAULT_MAX_SESSIONS = 300;
|
|
98
|
+
/** Hard ceiling for the scan budget. */
|
|
99
|
+
const RAW_SCAN_MAX_SESSIONS_MAX = 2e3;
|
|
100
|
+
/** Characters of context around the first matched term. */
|
|
101
|
+
const SNIPPET_CHARS$1 = 200;
|
|
102
|
+
/** Where the harness persists sessions: `$DSH_HOME/sessions` (default `~/.dsh/sessions`). */
|
|
103
|
+
function discoverSessionsRoot() {
|
|
104
|
+
const dshHome = process.env.DSH_HOME ?? join(homedir(), ".dsh");
|
|
105
|
+
return join(dshHome, "sessions");
|
|
106
|
+
}
|
|
107
|
+
/** True when an error means "the index is unavailable", not "the query was bad". */
|
|
108
|
+
function isIndexOutage(error) {
|
|
109
|
+
return typeof error === "object" && error !== null && "code" in error && String(error.code) === "SESSION_QUERY_PERSISTENCE_FAILED";
|
|
110
|
+
}
|
|
111
|
+
/** Tolerant single-line JSON parse: garbage lines yield null, never throw. */
|
|
112
|
+
function parseLine(line) {
|
|
113
|
+
try {
|
|
114
|
+
const value = JSON.parse(line);
|
|
115
|
+
return typeof value === "object" && value !== null ? value : null;
|
|
116
|
+
} catch {
|
|
117
|
+
return null;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
/** Pull the searchable text, tool name, and error flag out of one event. */
|
|
121
|
+
function eventFacts(event) {
|
|
122
|
+
const data = event["data"];
|
|
123
|
+
if (typeof data !== "object" || data === null) return null;
|
|
124
|
+
const record = data;
|
|
125
|
+
const type = typeof event["type"] === "string" ? event["type"] : "";
|
|
126
|
+
const texts = [];
|
|
127
|
+
const pushBlocks = (blocks) => {
|
|
128
|
+
if (!Array.isArray(blocks)) return;
|
|
129
|
+
for (const block of blocks) if (typeof block === "string") texts.push(block);
|
|
130
|
+
else if (typeof block === "object" && block !== null) {
|
|
131
|
+
const b = block;
|
|
132
|
+
if (typeof b["text"] === "string") texts.push(b["text"]);
|
|
133
|
+
if (Array.isArray(b["content"])) pushBlocks(b["content"]);
|
|
134
|
+
}
|
|
135
|
+
};
|
|
136
|
+
let tool = null;
|
|
137
|
+
if (type === "tool/call" && typeof record["name"] === "string") {
|
|
138
|
+
tool = record["name"];
|
|
139
|
+
if (typeof record["arguments"] === "string") texts.push(record["arguments"]);
|
|
140
|
+
}
|
|
141
|
+
const message = record["message"];
|
|
142
|
+
if (typeof message === "object" && message !== null) {
|
|
143
|
+
const m = message;
|
|
144
|
+
if (Array.isArray(m["content"])) pushBlocks(m["content"]);
|
|
145
|
+
}
|
|
146
|
+
if (Array.isArray(record["content"])) pushBlocks(record["content"]);
|
|
147
|
+
if (texts.length === 0 && typeof record["label"] === "string") texts.push(record["label"]);
|
|
148
|
+
const isError = typeof message === "object" && message !== null && (message["isError"] === true || message["is_error"] === true) || record["isError"] === true || record["is_error"] === true;
|
|
149
|
+
const text = texts.join("\n");
|
|
150
|
+
if (text === "" && tool == null) return null;
|
|
151
|
+
return {
|
|
152
|
+
text,
|
|
153
|
+
tool,
|
|
154
|
+
isError
|
|
155
|
+
};
|
|
156
|
+
}
|
|
157
|
+
/** All terms must match: ASCII on word boundaries, CJK as substrings. */
|
|
158
|
+
function textMatches(text, terms) {
|
|
159
|
+
if (terms.length === 0) return false;
|
|
160
|
+
const haystack = text.toLowerCase();
|
|
161
|
+
for (const term of terms) {
|
|
162
|
+
const needle = term.toLowerCase();
|
|
163
|
+
if (/[\u3400-\u9fff\uf900-\ufaff\u3040-\u30ff]/.test(term)) {
|
|
164
|
+
if (!haystack.includes(needle)) return false;
|
|
165
|
+
} else {
|
|
166
|
+
const escaped = needle.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
167
|
+
if (!new RegExp(`(^|[^a-z0-9_])${escaped}([^a-z0-9_]|$)`, "u").test(haystack)) return false;
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
return true;
|
|
171
|
+
}
|
|
172
|
+
/** List session directories under the root, newest first. */
|
|
173
|
+
function listSessionDirs(root) {
|
|
174
|
+
const out = [];
|
|
175
|
+
let workspaces = [];
|
|
176
|
+
try {
|
|
177
|
+
workspaces = readdirSync(root, { withFileTypes: true }).filter((e) => e.isDirectory()).map((e) => e.name);
|
|
178
|
+
} catch {
|
|
179
|
+
return out;
|
|
180
|
+
}
|
|
181
|
+
for (const ws of workspaces) {
|
|
182
|
+
const wsDir = join(root, ws);
|
|
183
|
+
let sessions = [];
|
|
184
|
+
try {
|
|
185
|
+
sessions = readdirSync(wsDir, { withFileTypes: true }).filter((e) => e.isDirectory()).map((e) => e.name);
|
|
186
|
+
} catch {
|
|
187
|
+
continue;
|
|
188
|
+
}
|
|
189
|
+
for (const id of sessions) {
|
|
190
|
+
const dir = join(wsDir, id);
|
|
191
|
+
try {
|
|
192
|
+
out.push({
|
|
193
|
+
dir,
|
|
194
|
+
mtimeMs: statSync(dir).mtimeMs
|
|
195
|
+
});
|
|
196
|
+
} catch {}
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
out.sort((a, b) => b.mtimeMs - a.mtimeMs);
|
|
200
|
+
return out;
|
|
201
|
+
}
|
|
202
|
+
/** Decompress one session log fully. Plain `.jsonl` is read as-is; `.zstd` via fzstd. */
|
|
203
|
+
async function readSessionLog(dir) {
|
|
204
|
+
for (const name of ["session.jsonl.zstd", "session.jsonl"]) {
|
|
205
|
+
let bytes;
|
|
206
|
+
try {
|
|
207
|
+
bytes = readFileSync(join(dir, name));
|
|
208
|
+
} catch {
|
|
209
|
+
continue;
|
|
210
|
+
}
|
|
211
|
+
if (name.endsWith(".zstd")) try {
|
|
212
|
+
const fzstd = await getFzstd();
|
|
213
|
+
return Buffer.from(fzstd.decompress(new Uint8Array(bytes)));
|
|
214
|
+
} catch {
|
|
215
|
+
return null;
|
|
216
|
+
}
|
|
217
|
+
return bytes;
|
|
218
|
+
}
|
|
219
|
+
return null;
|
|
220
|
+
}
|
|
221
|
+
/**
|
|
222
|
+
* Read just the session header line without decompressing the whole log:
|
|
223
|
+
* zstd input is pushed through the streaming decompressor in small chunks
|
|
224
|
+
* and stops at the first newline. Costs milliseconds per session, which is
|
|
225
|
+
* what makes scanning a whole store affordable.
|
|
226
|
+
*/
|
|
227
|
+
async function readSessionHeaderStreaming(dir) {
|
|
228
|
+
const fromLine = (line) => {
|
|
229
|
+
const header = parseLine(line);
|
|
230
|
+
if (header == null) return null;
|
|
231
|
+
const id = typeof header["id"] === "string" ? header["id"] : null;
|
|
232
|
+
if (id == null) return null;
|
|
233
|
+
return {
|
|
234
|
+
id,
|
|
235
|
+
createdAt: typeof header["createdAt"] === "number" ? header["createdAt"] : NaN,
|
|
236
|
+
cwd: typeof header["cwd"] === "string" ? header["cwd"] : null
|
|
237
|
+
};
|
|
238
|
+
};
|
|
239
|
+
for (const name of ["session.jsonl.zstd", "session.jsonl"]) {
|
|
240
|
+
const path = join(dir, name);
|
|
241
|
+
let fd;
|
|
242
|
+
try {
|
|
243
|
+
fd = openSync(path, "r");
|
|
244
|
+
} catch {
|
|
245
|
+
continue;
|
|
246
|
+
}
|
|
247
|
+
try {
|
|
248
|
+
const chunks = [];
|
|
249
|
+
const input = Buffer.alloc(16384);
|
|
250
|
+
let fzstd = null;
|
|
251
|
+
for (;;) {
|
|
252
|
+
let bytes;
|
|
253
|
+
try {
|
|
254
|
+
bytes = readSync(fd, input, 0, input.length, null);
|
|
255
|
+
} catch {
|
|
256
|
+
return null;
|
|
257
|
+
}
|
|
258
|
+
if (bytes <= 0) break;
|
|
259
|
+
if (name.endsWith(".zstd")) {
|
|
260
|
+
if (fzstd == null) try {
|
|
261
|
+
fzstd = await getFzstd();
|
|
262
|
+
} catch {
|
|
263
|
+
return null;
|
|
264
|
+
}
|
|
265
|
+
try {
|
|
266
|
+
new fzstd.Decompress((chunk) => {
|
|
267
|
+
chunks.push(Buffer.from(chunk));
|
|
268
|
+
}).push(input.subarray(0, bytes));
|
|
269
|
+
} catch {
|
|
270
|
+
return null;
|
|
271
|
+
}
|
|
272
|
+
} else chunks.push(Buffer.from(input.subarray(0, bytes)));
|
|
273
|
+
const soFar = Buffer.concat(chunks);
|
|
274
|
+
const nl = soFar.indexOf(10);
|
|
275
|
+
if (nl >= 0) return fromLine(soFar.subarray(0, nl).toString("utf-8"));
|
|
276
|
+
if (soFar.length > 1048576) return null;
|
|
277
|
+
}
|
|
278
|
+
return fromLine(Buffer.concat(chunks).toString("utf-8"));
|
|
279
|
+
} finally {
|
|
280
|
+
closeSync(fd);
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
return null;
|
|
284
|
+
}
|
|
285
|
+
/** The persisted log file of a session dir: path + compressed size, or null when none exists. */
|
|
286
|
+
function sessionLogPath(dir) {
|
|
287
|
+
for (const name of ["session.jsonl.zstd", "session.jsonl"]) try {
|
|
288
|
+
const path = join(dir, name);
|
|
289
|
+
return {
|
|
290
|
+
path,
|
|
291
|
+
size: statSync(path).size
|
|
292
|
+
};
|
|
293
|
+
} catch {
|
|
294
|
+
continue;
|
|
295
|
+
}
|
|
296
|
+
return null;
|
|
297
|
+
}
|
|
298
|
+
/**
|
|
299
|
+
* Cheap superset gate: every term must appear in the decompressed bytes
|
|
300
|
+
* (exact or lowercase form) before the expensive line-by-line parse runs.
|
|
301
|
+
* Word-boundary semantics are still enforced by `textMatches` later.
|
|
302
|
+
*/
|
|
303
|
+
function bytesMentionAllTerms(buf, terms) {
|
|
304
|
+
for (const term of terms) if (buf.indexOf(term, 0, "utf-8") < 0 && buf.indexOf(term.toLowerCase(), 0, "utf-8") < 0) return false;
|
|
305
|
+
return true;
|
|
306
|
+
}
|
|
307
|
+
/**
|
|
308
|
+
* Scan persisted session logs directly, honoring the same scope filters as
|
|
309
|
+
* the indexed path. Never throws for per-session problems; returns what it
|
|
310
|
+
* found plus scan counters.
|
|
311
|
+
*
|
|
312
|
+
* Cost model (newest-first): the header line of every session is read
|
|
313
|
+
* through the streaming decompressor (milliseconds each); only sessions
|
|
314
|
+
* passing the scope filters are decompressed fully, and only sessions whose
|
|
315
|
+
* bytes mention every term are parsed line by line. `maxDurationMs` puts a
|
|
316
|
+
* wall-clock ceiling on the pass — when it trips, `coveredAll` is false and
|
|
317
|
+
* the caller should say so (results cover the newest sessions only).
|
|
318
|
+
*/
|
|
319
|
+
async function rawScanSessions(options) {
|
|
320
|
+
const terms = splitTerms(options.query);
|
|
321
|
+
const sinceMs = options.sinceDays > 0 ? Date.now() - options.sinceDays * 24 * 60 * 60 * 1e3 : 0;
|
|
322
|
+
const toolsFilter = options.tools != null && options.tools.length > 0 ? options.tools.map((t) => t.toLowerCase()) : null;
|
|
323
|
+
const deadline = options.maxDurationMs > 0 ? Date.now() + options.maxDurationMs : 0;
|
|
324
|
+
const candidates = listSessionDirs(options.root);
|
|
325
|
+
const items = [];
|
|
326
|
+
let scanned = 0;
|
|
327
|
+
let unreadable = 0;
|
|
328
|
+
let skippedOversized = 0;
|
|
329
|
+
let coveredAll = true;
|
|
330
|
+
for (const candidate of candidates) {
|
|
331
|
+
if (items.length >= options.limit) break;
|
|
332
|
+
if (scanned >= options.maxSessions || deadline > 0 && Date.now() > deadline) {
|
|
333
|
+
coveredAll = false;
|
|
334
|
+
break;
|
|
335
|
+
}
|
|
336
|
+
const logInfo = sessionLogPath(candidate.dir);
|
|
337
|
+
if (logInfo == null) continue;
|
|
338
|
+
const header = await readSessionHeaderStreaming(candidate.dir);
|
|
339
|
+
if (header == null) {
|
|
340
|
+
unreadable += 1;
|
|
341
|
+
continue;
|
|
342
|
+
}
|
|
343
|
+
if (options.sessionId != null && header.id !== options.sessionId) continue;
|
|
344
|
+
if (!options.allProjects && options.cwd != null && header.cwd !== options.cwd) continue;
|
|
345
|
+
if (sinceMs > 0 && !(header.createdAt >= sinceMs)) continue;
|
|
346
|
+
if (logInfo.size > options.maxSessionBytes) {
|
|
347
|
+
skippedOversized += 1;
|
|
348
|
+
continue;
|
|
349
|
+
}
|
|
350
|
+
const log = await readSessionLog(candidate.dir);
|
|
351
|
+
if (log == null) {
|
|
352
|
+
unreadable += 1;
|
|
353
|
+
continue;
|
|
354
|
+
}
|
|
355
|
+
scanned += 1;
|
|
356
|
+
if (!bytesMentionAllTerms(log, terms)) continue;
|
|
357
|
+
const callTools = /* @__PURE__ */ new Map();
|
|
358
|
+
let best = null;
|
|
359
|
+
for (const line of log.toString("utf-8").split("\n")) {
|
|
360
|
+
if (line === "") continue;
|
|
361
|
+
const event = parseLine(line);
|
|
362
|
+
if (event == null) continue;
|
|
363
|
+
const type = typeof event["type"] === "string" ? event["type"] : "";
|
|
364
|
+
const facts = eventFacts(event);
|
|
365
|
+
if (facts == null) continue;
|
|
366
|
+
let tool = facts.tool;
|
|
367
|
+
if (tool == null && type === "tool/result") {
|
|
368
|
+
const source = event["data"]?.["message"];
|
|
369
|
+
const callId = typeof source === "object" && source !== null ? source["source"]?.["callId"] : void 0;
|
|
370
|
+
if (typeof callId === "string") tool = callTools.get(callId) ?? null;
|
|
371
|
+
} else if (tool != null) {
|
|
372
|
+
const data = event["data"];
|
|
373
|
+
const callId = typeof data?.["callId"] === "string" ? data["callId"] : null;
|
|
374
|
+
if (callId != null) callTools.set(callId, tool);
|
|
375
|
+
}
|
|
376
|
+
if (toolsFilter != null || options.errorsOnly) {
|
|
377
|
+
if (!(type === "tool/call" || type === "tool/result")) continue;
|
|
378
|
+
if (toolsFilter != null && (tool == null || !toolsFilter.some((t) => tool.toLowerCase().includes(t)))) continue;
|
|
379
|
+
if (options.errorsOnly && !(type === "tool/result" && facts.isError)) continue;
|
|
380
|
+
}
|
|
381
|
+
if (!textMatches(facts.text, terms)) continue;
|
|
382
|
+
const seq = typeof event["seq"] === "number" ? event["seq"] : 0;
|
|
383
|
+
const time = typeof event["time"] === "number" ? event["time"] : 0;
|
|
384
|
+
if (best == null || seq < best.seq) best = {
|
|
385
|
+
seq,
|
|
386
|
+
type,
|
|
387
|
+
time,
|
|
388
|
+
text: facts.text
|
|
389
|
+
};
|
|
390
|
+
}
|
|
391
|
+
if (best != null) items.push({
|
|
392
|
+
sessionId: header.id,
|
|
393
|
+
id8: id8(header.id),
|
|
394
|
+
title: null,
|
|
395
|
+
createdAt: Number.isFinite(header.createdAt) ? header.createdAt : 0,
|
|
396
|
+
cwd: header.cwd,
|
|
397
|
+
live: false,
|
|
398
|
+
persisted: true,
|
|
399
|
+
bestMatch: {
|
|
400
|
+
seq: best.seq,
|
|
401
|
+
type: best.type,
|
|
402
|
+
time: best.time,
|
|
403
|
+
snippet: snippetAround(best.text, terms[0] ?? "", SNIPPET_CHARS$1)
|
|
404
|
+
}
|
|
405
|
+
});
|
|
406
|
+
}
|
|
407
|
+
return {
|
|
408
|
+
items,
|
|
409
|
+
scanned,
|
|
410
|
+
budget: options.maxSessions,
|
|
411
|
+
unreadable,
|
|
412
|
+
skippedOversized,
|
|
413
|
+
coveredAll,
|
|
414
|
+
totalCandidates: candidates.length
|
|
415
|
+
};
|
|
416
|
+
}
|
|
417
|
+
//#endregion
|
|
5
418
|
//#region src/redact.ts
|
|
6
419
|
/**
|
|
7
420
|
* Redaction for recall output text (snippets and titles).
|
|
@@ -98,7 +511,11 @@ function normalizeRecallConfig(config) {
|
|
|
98
511
|
allProjectsPolicy: normalizePolicy(config?.allProjectsPolicy),
|
|
99
512
|
recencyHalfLifeDays: typeof config?.recencyHalfLifeDays === "number" && Number.isFinite(config.recencyHalfLifeDays) && config.recencyHalfLifeDays > 0 ? Math.min(3650, Math.trunc(config.recencyHalfLifeDays)) : void 0,
|
|
100
513
|
pinnedCwds: stringList(config?.pinnedCwds),
|
|
101
|
-
callerTreeOnly: config?.callerTreeOnly !== false
|
|
514
|
+
callerTreeOnly: config?.callerTreeOnly !== false,
|
|
515
|
+
rawScanFallback: config?.rawScanFallback !== false,
|
|
516
|
+
rawScanMaxSessions: intIn(config?.rawScanMaxSessions, 200, 1, 2e3),
|
|
517
|
+
rawScanMaxDurationMs: intIn(config?.rawScanMaxDurationMs, 2e4, 1e3, 12e4),
|
|
518
|
+
rawScanMaxSessionBytes: intIn(config?.rawScanMaxSessionBytes, 8388608, 262144, 67108864)
|
|
102
519
|
};
|
|
103
520
|
}
|
|
104
521
|
/** Whether a session cwd is searchable under the allowlist/denylist policy. */
|
|
@@ -146,66 +563,6 @@ function rankItems(items, options) {
|
|
|
146
563
|
return decorated.map((entry) => entry.item);
|
|
147
564
|
}
|
|
148
565
|
//#endregion
|
|
149
|
-
//#region src/util.ts
|
|
150
|
-
/** Small pure helpers shared by the recall tool and its renderers. */
|
|
151
|
-
/** Clamp `n` into the inclusive `[lo, hi]` range. */
|
|
152
|
-
function clamp(n, lo, hi) {
|
|
153
|
-
return Math.min(hi, Math.max(lo, n));
|
|
154
|
-
}
|
|
155
|
-
/**
|
|
156
|
-
* Short session slug for humans: strips the web-profile `session-` prefix and
|
|
157
|
-
* keeps the first 8 characters of what remains.
|
|
158
|
-
*/
|
|
159
|
-
function id8(sessionId) {
|
|
160
|
-
return (sessionId.startsWith("session-") ? sessionId.slice(8) : sessionId).slice(0, 8);
|
|
161
|
-
}
|
|
162
|
-
/** Format Unix epoch milliseconds as a local `YYYY-MM-DD` date. */
|
|
163
|
-
function formatDate(epochMs) {
|
|
164
|
-
const d = new Date(epochMs);
|
|
165
|
-
const m = String(d.getMonth() + 1).padStart(2, "0");
|
|
166
|
-
const day = String(d.getDate()).padStart(2, "0");
|
|
167
|
-
return `${d.getFullYear()}-${m}-${day}`;
|
|
168
|
-
}
|
|
169
|
-
const CJK_RE = /[\u{3400}-\u{4DBF}\u{4E00}-\u{9FFF}\u{F900}-\u{FAFF}\u{3040}-\u{30FF}\u{AC00}-\u{D7AF}\u{3000}-\u{303F}]/u;
|
|
170
|
-
/** Whether `text` contains at least one CJK ideograph, kana, or hangul character. */
|
|
171
|
-
function hasCJK(text) {
|
|
172
|
-
return CJK_RE.test(text);
|
|
173
|
-
}
|
|
174
|
-
/** Collapse every whitespace run and trim; used to normalize model-supplied queries. */
|
|
175
|
-
function normalizeQuery(text) {
|
|
176
|
-
return text.trim().replaceAll(/\s+/g, " ");
|
|
177
|
-
}
|
|
178
|
-
/** Split into whitespace-separated terms, dropping empty pieces. */
|
|
179
|
-
function splitTerms(text) {
|
|
180
|
-
return text.split(/\s+/u).filter((term) => term.length > 0);
|
|
181
|
-
}
|
|
182
|
-
/** First line of `text` with control characters stripped, clipped to `limit` code points. */
|
|
183
|
-
function firstLineClipped(text, limit) {
|
|
184
|
-
const line = text.split("\n", 1)[0] ?? "";
|
|
185
|
-
let out = "";
|
|
186
|
-
for (const ch of line) {
|
|
187
|
-
if ((ch.codePointAt(0) ?? 0) < 32) continue;
|
|
188
|
-
out += ch;
|
|
189
|
-
if (out.length >= limit) break;
|
|
190
|
-
}
|
|
191
|
-
return out;
|
|
192
|
-
}
|
|
193
|
-
/**
|
|
194
|
-
* Single-line snippet clipped around the first case-insensitive occurrence of
|
|
195
|
-
* `query`, with ellipses at either end when text was cut. Falls back to a
|
|
196
|
-
* head clip when the query is empty or absent.
|
|
197
|
-
*/
|
|
198
|
-
function snippetAround(text, query, limit) {
|
|
199
|
-
const flat = text.replaceAll(/\s+/g, " ");
|
|
200
|
-
const q = normalizeQuery(query).toLowerCase();
|
|
201
|
-
const idx = q === "" ? -1 : flat.toLowerCase().indexOf(q);
|
|
202
|
-
if (idx < 0) return flat.slice(0, limit);
|
|
203
|
-
const pad = Math.floor(limit / 3);
|
|
204
|
-
const start = Math.max(0, idx - pad);
|
|
205
|
-
const end = Math.min(flat.length, start + limit);
|
|
206
|
-
return `${start > 0 ? "…" : ""}${flat.slice(start, end)}${end < flat.length ? "…" : ""}`;
|
|
207
|
-
}
|
|
208
|
-
//#endregion
|
|
209
566
|
//#region src/render.ts
|
|
210
567
|
const SNIPPET_CHARS = 120;
|
|
211
568
|
function sessionLabel(result, index) {
|
|
@@ -545,7 +902,8 @@ const recallOutputSchema = {
|
|
|
545
902
|
enum: [
|
|
546
903
|
"fts",
|
|
547
904
|
"cjk-fallback",
|
|
548
|
-
"session-scan"
|
|
905
|
+
"session-scan",
|
|
906
|
+
"raw-scan"
|
|
549
907
|
]
|
|
550
908
|
},
|
|
551
909
|
scanned: { type: "integer" },
|
|
@@ -605,7 +963,7 @@ function createRecallTool(config, engine, approver) {
|
|
|
605
963
|
return defineTool({
|
|
606
964
|
name: "recall",
|
|
607
965
|
description: RECALL_TOOL_DESCRIPTION,
|
|
608
|
-
timeoutMs:
|
|
966
|
+
timeoutMs: 3e4,
|
|
609
967
|
isConcurrencySafe: () => true,
|
|
610
968
|
parameters: {
|
|
611
969
|
query: {
|
|
@@ -815,6 +1173,42 @@ function createRecallTool(config, engine, approver) {
|
|
|
815
1173
|
diagnostics
|
|
816
1174
|
};
|
|
817
1175
|
} catch (error) {
|
|
1176
|
+
if (cfg.rawScanFallback && isIndexOutage(error)) try {
|
|
1177
|
+
const scan = await rawScanSessions({
|
|
1178
|
+
root: discoverSessionsRoot(),
|
|
1179
|
+
query,
|
|
1180
|
+
cwd: agentCwd,
|
|
1181
|
+
allProjects: wantAll,
|
|
1182
|
+
sessionId: args.session_id ?? null,
|
|
1183
|
+
sinceDays: args.since_days ?? 0,
|
|
1184
|
+
tools: args.tools ?? null,
|
|
1185
|
+
errorsOnly: args.errors_only === true,
|
|
1186
|
+
limit,
|
|
1187
|
+
maxSessions: cfg.rawScanMaxSessions,
|
|
1188
|
+
maxDurationMs: cfg.rawScanMaxDurationMs,
|
|
1189
|
+
maxSessionBytes: cfg.rawScanMaxSessionBytes
|
|
1190
|
+
});
|
|
1191
|
+
let scanItems = scan.items.filter((item) => cwdAllowed(item.cwd, cfg));
|
|
1192
|
+
if (allowedIds != null) scanItems = scanItems.filter((item) => allowedIds.has(item.sessionId));
|
|
1193
|
+
scanItems = scanItems.slice(0, limit);
|
|
1194
|
+
const red = applyRedaction(scanItems);
|
|
1195
|
+
return {
|
|
1196
|
+
query,
|
|
1197
|
+
scope,
|
|
1198
|
+
count: red.items.length,
|
|
1199
|
+
hasMore: false,
|
|
1200
|
+
items: red.items,
|
|
1201
|
+
nextCursor: null,
|
|
1202
|
+
hint: joinHints(`degraded mode: the session index is unavailable (${errCode(error) ?? "persistence failure"}), so recall scanned ${scan.scanned} persisted session log(s) directly — order is by session recency, live sessions are not included${scan.unreadable > 0 ? `, ${scan.unreadable} unreadable log(s) were skipped` : ""}${scan.skippedOversized > 0 ? `, ${scan.skippedOversized} oversized log(s) beyond rawScanMaxSessionBytes were skipped` : ""}${scan.coveredAll ? "" : `, and the wall-clock budget stopped the pass after the newest ${scan.scanned} of ${scan.totalCandidates} sessions — narrow the scope (since_days, session_id) or raise rawScanMaxDurationMs to reach older history`}.`, red.hint),
|
|
1203
|
+
redacted: red.redacted,
|
|
1204
|
+
diagnostics: {
|
|
1205
|
+
source: "raw-scan",
|
|
1206
|
+
scanned: scan.scanned,
|
|
1207
|
+
scanBudget: scan.budget,
|
|
1208
|
+
ranked: false
|
|
1209
|
+
}
|
|
1210
|
+
};
|
|
1211
|
+
} catch {}
|
|
818
1212
|
return recallError(query, error);
|
|
819
1213
|
}
|
|
820
1214
|
},
|
|
@@ -872,4 +1266,4 @@ function apply(ctx, config) {
|
|
|
872
1266
|
}, "session-recall lifecycle");
|
|
873
1267
|
}
|
|
874
1268
|
//#endregion
|
|
875
|
-
export { ALL_PROJECTS_POLICIES, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, firstLineClipped, formatDate, hasCJK, id8, inject, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
|
|
1269
|
+
export { ALL_PROJECTS_POLICIES, RAW_SCAN_DEFAULT_MAX_SESSIONS, RAW_SCAN_MAX_SESSIONS_MAX, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, discoverSessionsRoot, firstLineClipped, formatDate, hasCJK, id8, inject, isIndexOutage, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, rawScanSessions, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
|