dsh-session-recall 0.7.3 → 0.7.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +24 -0
- package/README.zh.md +14 -0
- package/lib/esm-DyxhIKe9.js +662 -0
- package/lib/index.d.ts +93 -4
- package/lib/index.js +462 -67
- package/package.json +15 -13
package/lib/index.js
CHANGED
|
@@ -1,7 +1,420 @@
|
|
|
1
1
|
import { defineTool } from "@deepseek-ai/dsh-tools";
|
|
2
2
|
import { SessionSearchCursor } from "@deepseek-ai/dsh-session-query";
|
|
3
3
|
import { SessionId } from "@deepseek-ai/dsh-session";
|
|
4
|
+
import { closeSync, openSync, readFileSync, readSync, readdirSync, statSync } from "node:fs";
|
|
5
|
+
import { homedir } from "node:os";
|
|
6
|
+
import { join } from "node:path";
|
|
4
7
|
import { createHash } from "node:crypto";
|
|
8
|
+
//#region src/util.ts
|
|
9
|
+
/** Small pure helpers shared by the recall tool and its renderers. */
|
|
10
|
+
/** Clamp `n` into the inclusive `[lo, hi]` range. */
|
|
11
|
+
function clamp(n, lo, hi) {
|
|
12
|
+
return Math.min(hi, Math.max(lo, n));
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Short session slug for humans: strips the web-profile `session-` prefix and
|
|
16
|
+
* keeps the first 8 characters of what remains.
|
|
17
|
+
*/
|
|
18
|
+
function id8(sessionId) {
|
|
19
|
+
return (sessionId.startsWith("session-") ? sessionId.slice(8) : sessionId).slice(0, 8);
|
|
20
|
+
}
|
|
21
|
+
/** Format Unix epoch milliseconds as a local `YYYY-MM-DD` date. */
|
|
22
|
+
function formatDate(epochMs) {
|
|
23
|
+
const d = new Date(epochMs);
|
|
24
|
+
const m = String(d.getMonth() + 1).padStart(2, "0");
|
|
25
|
+
const day = String(d.getDate()).padStart(2, "0");
|
|
26
|
+
return `${d.getFullYear()}-${m}-${day}`;
|
|
27
|
+
}
|
|
28
|
+
const CJK_RE = /[\u{3400}-\u{4DBF}\u{4E00}-\u{9FFF}\u{F900}-\u{FAFF}\u{3040}-\u{30FF}\u{AC00}-\u{D7AF}\u{3000}-\u{303F}]/u;
|
|
29
|
+
/** Whether `text` contains at least one CJK ideograph, kana, or hangul character. */
|
|
30
|
+
function hasCJK(text) {
|
|
31
|
+
return CJK_RE.test(text);
|
|
32
|
+
}
|
|
33
|
+
/** Collapse every whitespace run and trim; used to normalize model-supplied queries. */
|
|
34
|
+
function normalizeQuery(text) {
|
|
35
|
+
return text.trim().replaceAll(/\s+/g, " ");
|
|
36
|
+
}
|
|
37
|
+
/** Split into whitespace-separated terms, dropping empty pieces. */
|
|
38
|
+
function splitTerms(text) {
|
|
39
|
+
return text.split(/\s+/u).filter((term) => term.length > 0);
|
|
40
|
+
}
|
|
41
|
+
/** First line of `text` with control characters stripped, clipped to `limit` code points. */
|
|
42
|
+
function firstLineClipped(text, limit) {
|
|
43
|
+
const line = text.split("\n", 1)[0] ?? "";
|
|
44
|
+
let out = "";
|
|
45
|
+
for (const ch of line) {
|
|
46
|
+
if ((ch.codePointAt(0) ?? 0) < 32) continue;
|
|
47
|
+
out += ch;
|
|
48
|
+
if (out.length >= limit) break;
|
|
49
|
+
}
|
|
50
|
+
return out;
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* Single-line snippet clipped around the first case-insensitive occurrence of
|
|
54
|
+
* `query`, with ellipses at either end when text was cut. Falls back to a
|
|
55
|
+
* head clip when the query is empty or absent.
|
|
56
|
+
*/
|
|
57
|
+
function snippetAround(text, query, limit) {
|
|
58
|
+
const flat = text.replaceAll(/\s+/g, " ");
|
|
59
|
+
const q = normalizeQuery(query).toLowerCase();
|
|
60
|
+
const idx = q === "" ? -1 : flat.toLowerCase().indexOf(q);
|
|
61
|
+
if (idx < 0) return flat.slice(0, limit);
|
|
62
|
+
const pad = Math.floor(limit / 3);
|
|
63
|
+
const start = Math.max(0, idx - pad);
|
|
64
|
+
const end = Math.min(flat.length, start + limit);
|
|
65
|
+
return `${start > 0 ? "…" : ""}${flat.slice(start, end)}${end < flat.length ? "…" : ""}`;
|
|
66
|
+
}
|
|
67
|
+
//#endregion
|
|
68
|
+
//#region src/resilient.ts
|
|
69
|
+
/**
|
|
70
|
+
* Resilient raw-log scan: the recall tool's degraded mode.
|
|
71
|
+
*
|
|
72
|
+
* When the session index itself is unavailable — most notably
|
|
73
|
+
* `SESSION_QUERY_PERSISTENCE_FAILED`, where one un-migratable session
|
|
74
|
+
* artifact fails every indexed search (see deepseek-harness discussion
|
|
75
|
+
* #7995: the v0→v1 migration gate rejects `subagent/descriptor` version 2,
|
|
76
|
+
* the only version the writer emits) — recall degrades to scanning the
|
|
77
|
+
* persisted session logs directly instead of failing the whole call.
|
|
78
|
+
*
|
|
79
|
+
* Design rules, in order:
|
|
80
|
+
* 1. Never throw on a bad session: a session that cannot be decompressed or
|
|
81
|
+
* parsed is counted as unreadable and skipped. The whole point of this
|
|
82
|
+
* module is that no single artifact can take the search down.
|
|
83
|
+
* 2. Scope first, scan second: the header line (cwd, createdAt) is read
|
|
84
|
+
* before the event scan, so cwd/time/session_id filters cost one line.
|
|
85
|
+
* 3. Same match semantics as the indexed path: whitespace-separated terms,
|
|
86
|
+
* all must match; ASCII terms match on word boundaries, CJK terms match
|
|
87
|
+
* as substrings.
|
|
88
|
+
*
|
|
89
|
+
* @module dsh-session-recall/resilient
|
|
90
|
+
*/
|
|
91
|
+
let fzstdModule = null;
|
|
92
|
+
async function getFzstd() {
|
|
93
|
+
if (fzstdModule == null) fzstdModule = await import("./esm-DyxhIKe9.js");
|
|
94
|
+
return fzstdModule;
|
|
95
|
+
}
|
|
96
|
+
/** Default scan budget (sessions visited per degraded call). */
|
|
97
|
+
const RAW_SCAN_DEFAULT_MAX_SESSIONS = 300;
|
|
98
|
+
/** Hard ceiling for the scan budget. */
|
|
99
|
+
const RAW_SCAN_MAX_SESSIONS_MAX = 2e3;
|
|
100
|
+
/** Characters of context around the first matched term. */
|
|
101
|
+
const SNIPPET_CHARS$1 = 200;
|
|
102
|
+
/** Where the harness persists sessions: `$DSH_HOME/sessions` (default `~/.dsh/sessions`). */
|
|
103
|
+
function discoverSessionsRoot() {
|
|
104
|
+
const dshHome = process.env.DSH_HOME ?? join(homedir(), ".dsh");
|
|
105
|
+
return join(dshHome, "sessions");
|
|
106
|
+
}
|
|
107
|
+
/** True when an error means "the index is unavailable", not "the query was bad". */
|
|
108
|
+
function isIndexOutage(error) {
|
|
109
|
+
return typeof error === "object" && error !== null && "code" in error && String(error.code) === "SESSION_QUERY_PERSISTENCE_FAILED";
|
|
110
|
+
}
|
|
111
|
+
/** Tolerant single-line JSON parse: garbage lines yield null, never throw. */
|
|
112
|
+
function parseLine(line) {
|
|
113
|
+
try {
|
|
114
|
+
const value = JSON.parse(line);
|
|
115
|
+
return typeof value === "object" && value !== null ? value : null;
|
|
116
|
+
} catch {
|
|
117
|
+
return null;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
/** Pull the searchable text, tool name, and error flag out of one event. */
|
|
121
|
+
function eventFacts(event) {
|
|
122
|
+
const data = event["data"];
|
|
123
|
+
if (typeof data !== "object" || data === null) return null;
|
|
124
|
+
const record = data;
|
|
125
|
+
const type = typeof event["type"] === "string" ? event["type"] : "";
|
|
126
|
+
const texts = [];
|
|
127
|
+
const pushBlocks = (blocks) => {
|
|
128
|
+
if (!Array.isArray(blocks)) return;
|
|
129
|
+
for (const block of blocks) if (typeof block === "string") texts.push(block);
|
|
130
|
+
else if (typeof block === "object" && block !== null) {
|
|
131
|
+
const b = block;
|
|
132
|
+
if (typeof b["text"] === "string") texts.push(b["text"]);
|
|
133
|
+
if (Array.isArray(b["content"])) pushBlocks(b["content"]);
|
|
134
|
+
}
|
|
135
|
+
};
|
|
136
|
+
let tool = null;
|
|
137
|
+
if (type === "tool/call" && typeof record["name"] === "string") {
|
|
138
|
+
tool = record["name"];
|
|
139
|
+
if (typeof record["arguments"] === "string") texts.push(record["arguments"]);
|
|
140
|
+
}
|
|
141
|
+
const message = record["message"];
|
|
142
|
+
if (typeof message === "object" && message !== null) {
|
|
143
|
+
const m = message;
|
|
144
|
+
if (Array.isArray(m["content"])) pushBlocks(m["content"]);
|
|
145
|
+
}
|
|
146
|
+
if (Array.isArray(record["content"])) pushBlocks(record["content"]);
|
|
147
|
+
if (texts.length === 0 && typeof record["label"] === "string") texts.push(record["label"]);
|
|
148
|
+
const isError = typeof message === "object" && message !== null && (message["isError"] === true || message["is_error"] === true) || record["isError"] === true || record["is_error"] === true;
|
|
149
|
+
const text = texts.join("\n");
|
|
150
|
+
if (text === "" && tool == null) return null;
|
|
151
|
+
return {
|
|
152
|
+
text,
|
|
153
|
+
tool,
|
|
154
|
+
isError
|
|
155
|
+
};
|
|
156
|
+
}
|
|
157
|
+
/** All terms must match: ASCII on word boundaries, CJK as substrings. */
|
|
158
|
+
function textMatches(text, terms) {
|
|
159
|
+
if (terms.length === 0) return false;
|
|
160
|
+
const haystack = text.toLowerCase();
|
|
161
|
+
for (const term of terms) {
|
|
162
|
+
const needle = term.toLowerCase();
|
|
163
|
+
if (/[\u3400-\u9fff\uf900-\ufaff\u3040-\u30ff]/.test(term)) {
|
|
164
|
+
if (!haystack.includes(needle)) return false;
|
|
165
|
+
} else {
|
|
166
|
+
const escaped = needle.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
167
|
+
if (!new RegExp(`(^|[^a-z0-9_])${escaped}([^a-z0-9_]|$)`, "u").test(haystack)) return false;
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
return true;
|
|
171
|
+
}
|
|
172
|
+
/** List session directories under the root, newest first. */
|
|
173
|
+
function listSessionDirs(root) {
|
|
174
|
+
const out = [];
|
|
175
|
+
let workspaces = [];
|
|
176
|
+
try {
|
|
177
|
+
workspaces = readdirSync(root, { withFileTypes: true }).filter((e) => e.isDirectory()).map((e) => e.name);
|
|
178
|
+
} catch {
|
|
179
|
+
return out;
|
|
180
|
+
}
|
|
181
|
+
for (const ws of workspaces) {
|
|
182
|
+
const wsDir = join(root, ws);
|
|
183
|
+
let sessions = [];
|
|
184
|
+
try {
|
|
185
|
+
sessions = readdirSync(wsDir, { withFileTypes: true }).filter((e) => e.isDirectory()).map((e) => e.name);
|
|
186
|
+
} catch {
|
|
187
|
+
continue;
|
|
188
|
+
}
|
|
189
|
+
for (const id of sessions) {
|
|
190
|
+
const dir = join(wsDir, id);
|
|
191
|
+
try {
|
|
192
|
+
out.push({
|
|
193
|
+
dir,
|
|
194
|
+
mtimeMs: statSync(dir).mtimeMs
|
|
195
|
+
});
|
|
196
|
+
} catch {}
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
out.sort((a, b) => b.mtimeMs - a.mtimeMs);
|
|
200
|
+
return out;
|
|
201
|
+
}
|
|
202
|
+
/** Decompress one session log fully. Plain `.jsonl` is read as-is; `.zstd` via fzstd. */
|
|
203
|
+
async function readSessionLog(dir) {
|
|
204
|
+
for (const name of ["session.jsonl.zstd", "session.jsonl"]) {
|
|
205
|
+
let bytes;
|
|
206
|
+
try {
|
|
207
|
+
bytes = readFileSync(join(dir, name));
|
|
208
|
+
} catch {
|
|
209
|
+
continue;
|
|
210
|
+
}
|
|
211
|
+
if (name.endsWith(".zstd")) try {
|
|
212
|
+
const fzstd = await getFzstd();
|
|
213
|
+
return Buffer.from(fzstd.decompress(new Uint8Array(bytes)));
|
|
214
|
+
} catch {
|
|
215
|
+
return null;
|
|
216
|
+
}
|
|
217
|
+
return bytes;
|
|
218
|
+
}
|
|
219
|
+
return null;
|
|
220
|
+
}
|
|
221
|
+
/**
|
|
222
|
+
* Read just the session header line without decompressing the whole log:
|
|
223
|
+
* zstd input is pushed through the streaming decompressor in small chunks
|
|
224
|
+
* and stops at the first newline. Costs milliseconds per session, which is
|
|
225
|
+
* what makes scanning a whole store affordable.
|
|
226
|
+
*/
|
|
227
|
+
async function readSessionHeaderStreaming(dir) {
|
|
228
|
+
const fromLine = (line) => {
|
|
229
|
+
const header = parseLine(line);
|
|
230
|
+
if (header == null) return null;
|
|
231
|
+
const id = typeof header["id"] === "string" ? header["id"] : null;
|
|
232
|
+
if (id == null) return null;
|
|
233
|
+
return {
|
|
234
|
+
id,
|
|
235
|
+
createdAt: typeof header["createdAt"] === "number" ? header["createdAt"] : NaN,
|
|
236
|
+
cwd: typeof header["cwd"] === "string" ? header["cwd"] : null
|
|
237
|
+
};
|
|
238
|
+
};
|
|
239
|
+
for (const name of ["session.jsonl.zstd", "session.jsonl"]) {
|
|
240
|
+
const path = join(dir, name);
|
|
241
|
+
let fd;
|
|
242
|
+
try {
|
|
243
|
+
fd = openSync(path, "r");
|
|
244
|
+
} catch {
|
|
245
|
+
continue;
|
|
246
|
+
}
|
|
247
|
+
try {
|
|
248
|
+
const chunks = [];
|
|
249
|
+
const input = Buffer.alloc(16384);
|
|
250
|
+
let fzstd = null;
|
|
251
|
+
for (;;) {
|
|
252
|
+
let bytes;
|
|
253
|
+
try {
|
|
254
|
+
bytes = readSync(fd, input, 0, input.length, null);
|
|
255
|
+
} catch {
|
|
256
|
+
return null;
|
|
257
|
+
}
|
|
258
|
+
if (bytes <= 0) break;
|
|
259
|
+
if (name.endsWith(".zstd")) {
|
|
260
|
+
if (fzstd == null) try {
|
|
261
|
+
fzstd = await getFzstd();
|
|
262
|
+
} catch {
|
|
263
|
+
return null;
|
|
264
|
+
}
|
|
265
|
+
try {
|
|
266
|
+
new fzstd.Decompress((chunk) => {
|
|
267
|
+
chunks.push(Buffer.from(chunk));
|
|
268
|
+
}).push(input.subarray(0, bytes));
|
|
269
|
+
} catch {
|
|
270
|
+
return null;
|
|
271
|
+
}
|
|
272
|
+
} else chunks.push(Buffer.from(input.subarray(0, bytes)));
|
|
273
|
+
const soFar = Buffer.concat(chunks);
|
|
274
|
+
const nl = soFar.indexOf(10);
|
|
275
|
+
if (nl >= 0) return fromLine(soFar.subarray(0, nl).toString("utf-8"));
|
|
276
|
+
if (soFar.length > 1048576) return null;
|
|
277
|
+
}
|
|
278
|
+
return fromLine(Buffer.concat(chunks).toString("utf-8"));
|
|
279
|
+
} finally {
|
|
280
|
+
closeSync(fd);
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
return null;
|
|
284
|
+
}
|
|
285
|
+
/** The persisted log file of a session dir: path + compressed size, or null when none exists. */
|
|
286
|
+
function sessionLogPath(dir) {
|
|
287
|
+
for (const name of ["session.jsonl.zstd", "session.jsonl"]) try {
|
|
288
|
+
const path = join(dir, name);
|
|
289
|
+
return {
|
|
290
|
+
path,
|
|
291
|
+
size: statSync(path).size
|
|
292
|
+
};
|
|
293
|
+
} catch {
|
|
294
|
+
continue;
|
|
295
|
+
}
|
|
296
|
+
return null;
|
|
297
|
+
}
|
|
298
|
+
/**
|
|
299
|
+
* Cheap superset gate: every term must appear in the decompressed bytes
|
|
300
|
+
* (exact or lowercase form) before the expensive line-by-line parse runs.
|
|
301
|
+
* Word-boundary semantics are still enforced by `textMatches` later.
|
|
302
|
+
*/
|
|
303
|
+
function bytesMentionAllTerms(buf, terms) {
|
|
304
|
+
for (const term of terms) if (buf.indexOf(term, 0, "utf-8") < 0 && buf.indexOf(term.toLowerCase(), 0, "utf-8") < 0) return false;
|
|
305
|
+
return true;
|
|
306
|
+
}
|
|
307
|
+
/**
|
|
308
|
+
* Scan persisted session logs directly, honoring the same scope filters as
|
|
309
|
+
* the indexed path. Never throws for per-session problems; returns what it
|
|
310
|
+
* found plus scan counters.
|
|
311
|
+
*
|
|
312
|
+
* Cost model (newest-first): the header line of every session is read
|
|
313
|
+
* through the streaming decompressor (milliseconds each); only sessions
|
|
314
|
+
* passing the scope filters are decompressed fully, and only sessions whose
|
|
315
|
+
* bytes mention every term are parsed line by line. `maxDurationMs` puts a
|
|
316
|
+
* wall-clock ceiling on the pass — when it trips, `coveredAll` is false and
|
|
317
|
+
* the caller should say so (results cover the newest sessions only).
|
|
318
|
+
*/
|
|
319
|
+
async function rawScanSessions(options) {
|
|
320
|
+
const terms = splitTerms(options.query);
|
|
321
|
+
const sinceMs = options.sinceDays > 0 ? Date.now() - options.sinceDays * 24 * 60 * 60 * 1e3 : 0;
|
|
322
|
+
const toolsFilter = options.tools != null && options.tools.length > 0 ? options.tools.map((t) => t.toLowerCase()) : null;
|
|
323
|
+
const deadline = options.maxDurationMs > 0 ? Date.now() + options.maxDurationMs : 0;
|
|
324
|
+
const candidates = listSessionDirs(options.root);
|
|
325
|
+
const items = [];
|
|
326
|
+
let scanned = 0;
|
|
327
|
+
let unreadable = 0;
|
|
328
|
+
let skippedOversized = 0;
|
|
329
|
+
let coveredAll = true;
|
|
330
|
+
for (const candidate of candidates) {
|
|
331
|
+
if (items.length >= options.limit) break;
|
|
332
|
+
if (scanned >= options.maxSessions || deadline > 0 && Date.now() > deadline) {
|
|
333
|
+
coveredAll = false;
|
|
334
|
+
break;
|
|
335
|
+
}
|
|
336
|
+
const logInfo = sessionLogPath(candidate.dir);
|
|
337
|
+
if (logInfo == null) continue;
|
|
338
|
+
const header = await readSessionHeaderStreaming(candidate.dir);
|
|
339
|
+
if (header == null) {
|
|
340
|
+
unreadable += 1;
|
|
341
|
+
continue;
|
|
342
|
+
}
|
|
343
|
+
if (options.sessionId != null && header.id !== options.sessionId) continue;
|
|
344
|
+
if (!options.allProjects && options.cwd != null && header.cwd !== options.cwd) continue;
|
|
345
|
+
if (sinceMs > 0 && !(header.createdAt >= sinceMs)) continue;
|
|
346
|
+
if (logInfo.size > options.maxSessionBytes) {
|
|
347
|
+
skippedOversized += 1;
|
|
348
|
+
continue;
|
|
349
|
+
}
|
|
350
|
+
const log = await readSessionLog(candidate.dir);
|
|
351
|
+
if (log == null) {
|
|
352
|
+
unreadable += 1;
|
|
353
|
+
continue;
|
|
354
|
+
}
|
|
355
|
+
scanned += 1;
|
|
356
|
+
if (!bytesMentionAllTerms(log, terms)) continue;
|
|
357
|
+
const callTools = /* @__PURE__ */ new Map();
|
|
358
|
+
let best = null;
|
|
359
|
+
for (const line of log.toString("utf-8").split("\n")) {
|
|
360
|
+
if (line === "") continue;
|
|
361
|
+
const event = parseLine(line);
|
|
362
|
+
if (event == null) continue;
|
|
363
|
+
const type = typeof event["type"] === "string" ? event["type"] : "";
|
|
364
|
+
const facts = eventFacts(event);
|
|
365
|
+
if (facts == null) continue;
|
|
366
|
+
let tool = facts.tool;
|
|
367
|
+
if (tool == null && type === "tool/result") {
|
|
368
|
+
const source = event["data"]?.["message"];
|
|
369
|
+
const callId = typeof source === "object" && source !== null ? source["source"]?.["callId"] : void 0;
|
|
370
|
+
if (typeof callId === "string") tool = callTools.get(callId) ?? null;
|
|
371
|
+
} else if (tool != null) {
|
|
372
|
+
const data = event["data"];
|
|
373
|
+
const callId = typeof data?.["callId"] === "string" ? data["callId"] : null;
|
|
374
|
+
if (callId != null) callTools.set(callId, tool);
|
|
375
|
+
}
|
|
376
|
+
if (toolsFilter != null || options.errorsOnly) {
|
|
377
|
+
if (!(type === "tool/call" || type === "tool/result")) continue;
|
|
378
|
+
if (toolsFilter != null && (tool == null || !toolsFilter.some((t) => tool.toLowerCase().includes(t)))) continue;
|
|
379
|
+
if (options.errorsOnly && !(type === "tool/result" && facts.isError)) continue;
|
|
380
|
+
}
|
|
381
|
+
if (!textMatches(facts.text, terms)) continue;
|
|
382
|
+
const seq = typeof event["seq"] === "number" ? event["seq"] : 0;
|
|
383
|
+
const time = typeof event["time"] === "number" ? event["time"] : 0;
|
|
384
|
+
if (best == null || seq < best.seq) best = {
|
|
385
|
+
seq,
|
|
386
|
+
type,
|
|
387
|
+
time,
|
|
388
|
+
text: facts.text
|
|
389
|
+
};
|
|
390
|
+
}
|
|
391
|
+
if (best != null) items.push({
|
|
392
|
+
sessionId: header.id,
|
|
393
|
+
id8: id8(header.id),
|
|
394
|
+
title: null,
|
|
395
|
+
createdAt: Number.isFinite(header.createdAt) ? header.createdAt : 0,
|
|
396
|
+
cwd: header.cwd,
|
|
397
|
+
live: false,
|
|
398
|
+
persisted: true,
|
|
399
|
+
bestMatch: {
|
|
400
|
+
seq: best.seq,
|
|
401
|
+
type: best.type,
|
|
402
|
+
time: best.time,
|
|
403
|
+
snippet: snippetAround(best.text, terms[0] ?? "", SNIPPET_CHARS$1)
|
|
404
|
+
}
|
|
405
|
+
});
|
|
406
|
+
}
|
|
407
|
+
return {
|
|
408
|
+
items,
|
|
409
|
+
scanned,
|
|
410
|
+
budget: options.maxSessions,
|
|
411
|
+
unreadable,
|
|
412
|
+
skippedOversized,
|
|
413
|
+
coveredAll,
|
|
414
|
+
totalCandidates: candidates.length
|
|
415
|
+
};
|
|
416
|
+
}
|
|
417
|
+
//#endregion
|
|
5
418
|
//#region src/redact.ts
|
|
6
419
|
/**
|
|
7
420
|
* Redaction for recall output text (snippets and titles).
|
|
@@ -98,7 +511,11 @@ function normalizeRecallConfig(config) {
|
|
|
98
511
|
allProjectsPolicy: normalizePolicy(config?.allProjectsPolicy),
|
|
99
512
|
recencyHalfLifeDays: typeof config?.recencyHalfLifeDays === "number" && Number.isFinite(config.recencyHalfLifeDays) && config.recencyHalfLifeDays > 0 ? Math.min(3650, Math.trunc(config.recencyHalfLifeDays)) : void 0,
|
|
100
513
|
pinnedCwds: stringList(config?.pinnedCwds),
|
|
101
|
-
callerTreeOnly: config?.callerTreeOnly !== false
|
|
514
|
+
callerTreeOnly: config?.callerTreeOnly !== false,
|
|
515
|
+
rawScanFallback: config?.rawScanFallback !== false,
|
|
516
|
+
rawScanMaxSessions: intIn(config?.rawScanMaxSessions, 200, 1, 2e3),
|
|
517
|
+
rawScanMaxDurationMs: intIn(config?.rawScanMaxDurationMs, 2e4, 1e3, 12e4),
|
|
518
|
+
rawScanMaxSessionBytes: intIn(config?.rawScanMaxSessionBytes, 8388608, 262144, 67108864)
|
|
102
519
|
};
|
|
103
520
|
}
|
|
104
521
|
/** Whether a session cwd is searchable under the allowlist/denylist policy. */
|
|
@@ -146,66 +563,6 @@ function rankItems(items, options) {
|
|
|
146
563
|
return decorated.map((entry) => entry.item);
|
|
147
564
|
}
|
|
148
565
|
//#endregion
|
|
149
|
-
//#region src/util.ts
|
|
150
|
-
/** Small pure helpers shared by the recall tool and its renderers. */
|
|
151
|
-
/** Clamp `n` into the inclusive `[lo, hi]` range. */
|
|
152
|
-
function clamp(n, lo, hi) {
|
|
153
|
-
return Math.min(hi, Math.max(lo, n));
|
|
154
|
-
}
|
|
155
|
-
/**
|
|
156
|
-
* Short session slug for humans: strips the web-profile `session-` prefix and
|
|
157
|
-
* keeps the first 8 characters of what remains.
|
|
158
|
-
*/
|
|
159
|
-
function id8(sessionId) {
|
|
160
|
-
return (sessionId.startsWith("session-") ? sessionId.slice(8) : sessionId).slice(0, 8);
|
|
161
|
-
}
|
|
162
|
-
/** Format Unix epoch milliseconds as a local `YYYY-MM-DD` date. */
|
|
163
|
-
function formatDate(epochMs) {
|
|
164
|
-
const d = new Date(epochMs);
|
|
165
|
-
const m = String(d.getMonth() + 1).padStart(2, "0");
|
|
166
|
-
const day = String(d.getDate()).padStart(2, "0");
|
|
167
|
-
return `${d.getFullYear()}-${m}-${day}`;
|
|
168
|
-
}
|
|
169
|
-
const CJK_RE = /[\u{3400}-\u{4DBF}\u{4E00}-\u{9FFF}\u{F900}-\u{FAFF}\u{3040}-\u{30FF}\u{AC00}-\u{D7AF}\u{3000}-\u{303F}]/u;
|
|
170
|
-
/** Whether `text` contains at least one CJK ideograph, kana, or hangul character. */
|
|
171
|
-
function hasCJK(text) {
|
|
172
|
-
return CJK_RE.test(text);
|
|
173
|
-
}
|
|
174
|
-
/** Collapse every whitespace run and trim; used to normalize model-supplied queries. */
|
|
175
|
-
function normalizeQuery(text) {
|
|
176
|
-
return text.trim().replaceAll(/\s+/g, " ");
|
|
177
|
-
}
|
|
178
|
-
/** Split into whitespace-separated terms, dropping empty pieces. */
|
|
179
|
-
function splitTerms(text) {
|
|
180
|
-
return text.split(/\s+/u).filter((term) => term.length > 0);
|
|
181
|
-
}
|
|
182
|
-
/** First line of `text` with control characters stripped, clipped to `limit` code points. */
|
|
183
|
-
function firstLineClipped(text, limit) {
|
|
184
|
-
const line = text.split("\n", 1)[0] ?? "";
|
|
185
|
-
let out = "";
|
|
186
|
-
for (const ch of line) {
|
|
187
|
-
if ((ch.codePointAt(0) ?? 0) < 32) continue;
|
|
188
|
-
out += ch;
|
|
189
|
-
if (out.length >= limit) break;
|
|
190
|
-
}
|
|
191
|
-
return out;
|
|
192
|
-
}
|
|
193
|
-
/**
|
|
194
|
-
* Single-line snippet clipped around the first case-insensitive occurrence of
|
|
195
|
-
* `query`, with ellipses at either end when text was cut. Falls back to a
|
|
196
|
-
* head clip when the query is empty or absent.
|
|
197
|
-
*/
|
|
198
|
-
function snippetAround(text, query, limit) {
|
|
199
|
-
const flat = text.replaceAll(/\s+/g, " ");
|
|
200
|
-
const q = normalizeQuery(query).toLowerCase();
|
|
201
|
-
const idx = q === "" ? -1 : flat.toLowerCase().indexOf(q);
|
|
202
|
-
if (idx < 0) return flat.slice(0, limit);
|
|
203
|
-
const pad = Math.floor(limit / 3);
|
|
204
|
-
const start = Math.max(0, idx - pad);
|
|
205
|
-
const end = Math.min(flat.length, start + limit);
|
|
206
|
-
return `${start > 0 ? "…" : ""}${flat.slice(start, end)}${end < flat.length ? "…" : ""}`;
|
|
207
|
-
}
|
|
208
|
-
//#endregion
|
|
209
566
|
//#region src/render.ts
|
|
210
567
|
const SNIPPET_CHARS = 120;
|
|
211
568
|
function sessionLabel(result, index) {
|
|
@@ -545,7 +902,8 @@ const recallOutputSchema = {
|
|
|
545
902
|
enum: [
|
|
546
903
|
"fts",
|
|
547
904
|
"cjk-fallback",
|
|
548
|
-
"session-scan"
|
|
905
|
+
"session-scan",
|
|
906
|
+
"raw-scan"
|
|
549
907
|
]
|
|
550
908
|
},
|
|
551
909
|
scanned: { type: "integer" },
|
|
@@ -605,7 +963,7 @@ function createRecallTool(config, engine, approver) {
|
|
|
605
963
|
return defineTool({
|
|
606
964
|
name: "recall",
|
|
607
965
|
description: RECALL_TOOL_DESCRIPTION,
|
|
608
|
-
timeoutMs:
|
|
966
|
+
timeoutMs: 3e4,
|
|
609
967
|
isConcurrencySafe: () => true,
|
|
610
968
|
parameters: {
|
|
611
969
|
query: {
|
|
@@ -815,6 +1173,42 @@ function createRecallTool(config, engine, approver) {
|
|
|
815
1173
|
diagnostics
|
|
816
1174
|
};
|
|
817
1175
|
} catch (error) {
|
|
1176
|
+
if (cfg.rawScanFallback && isIndexOutage(error)) try {
|
|
1177
|
+
const scan = await rawScanSessions({
|
|
1178
|
+
root: discoverSessionsRoot(),
|
|
1179
|
+
query,
|
|
1180
|
+
cwd: agentCwd,
|
|
1181
|
+
allProjects: wantAll,
|
|
1182
|
+
sessionId: args.session_id ?? null,
|
|
1183
|
+
sinceDays: args.since_days ?? 0,
|
|
1184
|
+
tools: args.tools ?? null,
|
|
1185
|
+
errorsOnly: args.errors_only === true,
|
|
1186
|
+
limit,
|
|
1187
|
+
maxSessions: cfg.rawScanMaxSessions,
|
|
1188
|
+
maxDurationMs: cfg.rawScanMaxDurationMs,
|
|
1189
|
+
maxSessionBytes: cfg.rawScanMaxSessionBytes
|
|
1190
|
+
});
|
|
1191
|
+
let scanItems = scan.items.filter((item) => cwdAllowed(item.cwd, cfg));
|
|
1192
|
+
if (allowedIds != null) scanItems = scanItems.filter((item) => allowedIds.has(item.sessionId));
|
|
1193
|
+
scanItems = scanItems.slice(0, limit);
|
|
1194
|
+
const red = applyRedaction(scanItems);
|
|
1195
|
+
return {
|
|
1196
|
+
query,
|
|
1197
|
+
scope,
|
|
1198
|
+
count: red.items.length,
|
|
1199
|
+
hasMore: false,
|
|
1200
|
+
items: red.items,
|
|
1201
|
+
nextCursor: null,
|
|
1202
|
+
hint: joinHints(`degraded mode: the session index is unavailable (${errCode(error) ?? "persistence failure"}), so recall scanned ${scan.scanned} persisted session log(s) directly — order is by session recency, live sessions are not included${scan.unreadable > 0 ? `, ${scan.unreadable} unreadable log(s) were skipped` : ""}${scan.skippedOversized > 0 ? `, ${scan.skippedOversized} oversized log(s) beyond rawScanMaxSessionBytes were skipped` : ""}${scan.coveredAll ? "" : `, and the wall-clock budget stopped the pass after the newest ${scan.scanned} of ${scan.totalCandidates} sessions — narrow the scope (since_days, session_id) or raise rawScanMaxDurationMs to reach older history`}.`, red.hint),
|
|
1203
|
+
redacted: red.redacted,
|
|
1204
|
+
diagnostics: {
|
|
1205
|
+
source: "raw-scan",
|
|
1206
|
+
scanned: scan.scanned,
|
|
1207
|
+
scanBudget: scan.budget,
|
|
1208
|
+
ranked: false
|
|
1209
|
+
}
|
|
1210
|
+
};
|
|
1211
|
+
} catch {}
|
|
818
1212
|
return recallError(query, error);
|
|
819
1213
|
}
|
|
820
1214
|
},
|
|
@@ -844,11 +1238,12 @@ const name = "session-recall";
|
|
|
844
1238
|
const inject = ["tools", "sessionQuery"];
|
|
845
1239
|
/**
|
|
846
1240
|
* Build the optional approval seam for `allProjectsPolicy: 'confirm'`: ask
|
|
847
|
-
* `ctx.approval` when the service is composed and the call carries an agent;
|
|
848
|
-
* fail closed (`'unavailable'`) otherwise.
|
|
1241
|
+
* `ctx.get('approval')` when the service is composed and the call carries an agent;
|
|
1242
|
+
* fail closed (`'unavailable'`) otherwise. Direct `ctx.approval` throws unless
|
|
1243
|
+
* the service is declared in `inject`, and this seam is optional. Never throws.
|
|
849
1244
|
*/
|
|
850
1245
|
function makeApprover(ctx) {
|
|
851
|
-
const approval = ctx.approval;
|
|
1246
|
+
const approval = ctx.get("approval");
|
|
852
1247
|
if (approval == null || typeof approval.request !== "function") return void 0;
|
|
853
1248
|
return async (exec, reason) => {
|
|
854
1249
|
const agent = exec.agent;
|
|
@@ -871,4 +1266,4 @@ function apply(ctx, config) {
|
|
|
871
1266
|
}, "session-recall lifecycle");
|
|
872
1267
|
}
|
|
873
1268
|
//#endregion
|
|
874
|
-
export { ALL_PROJECTS_POLICIES, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, firstLineClipped, formatDate, hasCJK, id8, inject, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
|
|
1269
|
+
export { ALL_PROJECTS_POLICIES, RAW_SCAN_DEFAULT_MAX_SESSIONS, RAW_SCAN_MAX_SESSIONS_MAX, RECALL_TOOL_DESCRIPTION, REDACTION_MODES, apply, cjkFallbackHint, cjkZeroHitHint, clamp, createRecallTool, cwdAllowed, discoverSessionsRoot, firstLineClipped, formatDate, hasCJK, id8, inject, isIndexOutage, name, normalizeQuery, normalizeRecallConfig, normalizeRedactionMode, rankItems, rankingActive, rawScanSessions, recallContentBlocks, recallPresentationMeta, redactText, renderRecallText, snippetAround };
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-session-recall",
|
|
3
3
|
"description": "Deterministic cross-session transcript retrieval for DeepSeek Harness: the model-facing `recall` tool searches past session logs with explicit scope control",
|
|
4
|
-
"version": "0.7.
|
|
4
|
+
"version": "0.7.5",
|
|
5
5
|
"publishConfig": {
|
|
6
6
|
"access": "public"
|
|
7
7
|
},
|
|
@@ -39,22 +39,24 @@
|
|
|
39
39
|
"full-text-search"
|
|
40
40
|
],
|
|
41
41
|
"peerDependencies": {
|
|
42
|
-
"@deepseek-ai/cordis": "
|
|
43
|
-
"@deepseek-ai/dsh-agent": "^0.1.
|
|
44
|
-
"@deepseek-ai/dsh-llm": "^0.1.
|
|
45
|
-
"@deepseek-ai/dsh-session": "^0.1.
|
|
46
|
-
"@deepseek-ai/dsh-session-query": "^0.1.
|
|
47
|
-
"@deepseek-ai/dsh-tools": "^0.1.
|
|
42
|
+
"@deepseek-ai/cordis": "4.0.2",
|
|
43
|
+
"@deepseek-ai/dsh-agent": "^0.1.5-rc.3",
|
|
44
|
+
"@deepseek-ai/dsh-llm": "^0.1.5-rc.3",
|
|
45
|
+
"@deepseek-ai/dsh-session": "^0.1.5-rc.3",
|
|
46
|
+
"@deepseek-ai/dsh-session-query": "^0.1.5-rc.3",
|
|
47
|
+
"@deepseek-ai/dsh-tools": "^0.1.5-rc.3"
|
|
48
48
|
},
|
|
49
49
|
"devDependencies": {
|
|
50
|
-
"@deepseek-ai/cordis": "
|
|
51
|
-
"@deepseek-ai/dsh-agent": "^0.1.
|
|
52
|
-
"@deepseek-ai/dsh-llm": "^0.1.
|
|
53
|
-
"@deepseek-ai/dsh-session": "^0.1.
|
|
54
|
-
"@deepseek-ai/dsh-session-query": "^0.1.
|
|
55
|
-
"@deepseek-ai/dsh-tools": "^0.1.
|
|
50
|
+
"@deepseek-ai/cordis": "4.0.2",
|
|
51
|
+
"@deepseek-ai/dsh-agent": "^0.1.5-rc.3",
|
|
52
|
+
"@deepseek-ai/dsh-llm": "^0.1.5-rc.3",
|
|
53
|
+
"@deepseek-ai/dsh-session": "^0.1.5-rc.3",
|
|
54
|
+
"@deepseek-ai/dsh-session-query": "^0.1.5-rc.3",
|
|
55
|
+
"@deepseek-ai/dsh-tools": "^0.1.5-rc.3",
|
|
56
|
+
"@deepseek-ai/dsh-util-values": "^0.1.5-rc.3",
|
|
56
57
|
"@types/js-yaml": "^4.0.9",
|
|
57
58
|
"@types/node": "^26.2.0",
|
|
59
|
+
"fzstd": "^0.1.1",
|
|
58
60
|
"js-yaml": "^4.2.0",
|
|
59
61
|
"tsdown": "^0.22.14",
|
|
60
62
|
"typescript": "^7.0.2",
|