@deathbeam/pi-archive 0.0.2 → 0.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/index.ts +91 -27
- package/package.json +1 -1
package/index.ts
CHANGED
|
@@ -5,7 +5,9 @@ import path from "node:path";
|
|
|
5
5
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
6
6
|
|
|
7
7
|
const MAX_TOTAL_BYTES = 400 * 1024 * 1024;
|
|
8
|
-
const
|
|
8
|
+
const DEFAULT_PER_SESSION = 3;
|
|
9
|
+
const MAX_PER_SESSION = 1000;
|
|
10
|
+
const MAX_OFFSET = 10000;
|
|
9
11
|
const MAX_SEARCH_LIMIT = 100;
|
|
10
12
|
const READ_CHUNK_BYTES = 64 * 1024;
|
|
11
13
|
const MAX_LINE_BYTES = 16 * 1024 * 1024;
|
|
@@ -25,6 +27,8 @@ export interface SearchOptions {
|
|
|
25
27
|
currentDir?: string;
|
|
26
28
|
sessionFilter?: string; // substring matched against the encoded cwd dir name
|
|
27
29
|
limit?: number;
|
|
30
|
+
perSession?: number;
|
|
31
|
+
offset?: number;
|
|
28
32
|
}
|
|
29
33
|
|
|
30
34
|
function parseJsonLine(line: string): any | null {
|
|
@@ -49,6 +53,7 @@ export function entryText(entry: unknown): { role: string; text: string } | null
|
|
|
49
53
|
const content = e?.message?.content;
|
|
50
54
|
if (
|
|
51
55
|
e?.type === "message" &&
|
|
56
|
+
!(e.message.role === "toolResult" && e.message.toolName === "search_archive") &&
|
|
52
57
|
(Array.isArray(content) || (e.message.role === "user" && typeof content === "string"))
|
|
53
58
|
) {
|
|
54
59
|
const text =
|
|
@@ -182,16 +187,30 @@ export async function searchSessions(
|
|
|
182
187
|
opts: SearchOptions = {},
|
|
183
188
|
signal?: AbortSignal,
|
|
184
189
|
onProgress?: (filesScanned: number, matches: number) => void,
|
|
185
|
-
): Promise<{
|
|
190
|
+
): Promise<{
|
|
191
|
+
matches: ArchiveMatch[];
|
|
192
|
+
bytesScanned: number;
|
|
193
|
+
filesScanned: number;
|
|
194
|
+
truncated: boolean;
|
|
195
|
+
limitReached?: boolean;
|
|
196
|
+
scanIncomplete?: boolean;
|
|
197
|
+
perSession?: number;
|
|
198
|
+
offset?: number;
|
|
199
|
+
}> {
|
|
186
200
|
const terms = parseTerms(query);
|
|
187
|
-
if (
|
|
188
|
-
opts.limit !== undefined &&
|
|
189
|
-
(!Number.isSafeInteger(opts.limit) || opts.limit < 1 || opts.limit > MAX_SEARCH_LIMIT)
|
|
190
|
-
) {
|
|
201
|
+
if (opts.limit !== undefined && (!Number.isSafeInteger(opts.limit) || opts.limit < 1 || opts.limit > MAX_SEARCH_LIMIT)) {
|
|
191
202
|
throw new RangeError(`limit must be an integer from 1 to ${MAX_SEARCH_LIMIT}`);
|
|
192
203
|
}
|
|
204
|
+
if (opts.perSession !== undefined && (!Number.isSafeInteger(opts.perSession) || opts.perSession < 1 || opts.perSession > MAX_PER_SESSION)) {
|
|
205
|
+
throw new RangeError(`perSession must be an integer from 1 to ${MAX_PER_SESSION}`);
|
|
206
|
+
}
|
|
207
|
+
if (opts.offset !== undefined && (!Number.isSafeInteger(opts.offset) || opts.offset < 0 || opts.offset > MAX_OFFSET)) {
|
|
208
|
+
throw new RangeError(`offset must be an integer from 0 to ${MAX_OFFSET}`);
|
|
209
|
+
}
|
|
193
210
|
if (terms.length === 0) return { matches: [], bytesScanned: 0, filesScanned: 0, truncated: false };
|
|
194
211
|
const limit = opts.limit ?? 20;
|
|
212
|
+
const perSession = opts.perSession ?? DEFAULT_PER_SESSION;
|
|
213
|
+
const offset = opts.offset ?? 0;
|
|
195
214
|
const filter = opts.sessionFilter?.toLowerCase();
|
|
196
215
|
const files: { file: string; dir: string; name: string; mtime: number }[] = [];
|
|
197
216
|
let rootEntries: fs.Dirent[];
|
|
@@ -235,24 +254,29 @@ export async function searchSessions(
|
|
|
235
254
|
files.sort((a, b) => rank(a) - rank(b) || b.mtime - a.mtime);
|
|
236
255
|
|
|
237
256
|
const matches: ArchiveMatch[] = [];
|
|
257
|
+
// ponytail: live offsets can shift as sessions grow; use entry-ID cursors if paging must be exact.
|
|
258
|
+
let skipped = 0;
|
|
238
259
|
let bytes = 0;
|
|
239
260
|
let filesScanned = 0;
|
|
240
|
-
let
|
|
261
|
+
let scanIncomplete = incompleteDiscovery;
|
|
262
|
+
let limitReached = false;
|
|
241
263
|
for (const { file, dir, name } of files) {
|
|
242
264
|
if (signal?.aborted || bytes >= MAX_TOTAL_BYTES) {
|
|
243
|
-
|
|
265
|
+
scanIncomplete = true;
|
|
244
266
|
break;
|
|
245
267
|
}
|
|
246
268
|
onProgress?.(filesScanned, matches.length);
|
|
247
|
-
|
|
269
|
+
// ponytail: newest hits require scanning each file; index offsets if archive searches get slow.
|
|
270
|
+
const fileMessages: ArchiveMatch[] = [];
|
|
271
|
+
const fileTools: ArchiveMatch[] = [];
|
|
248
272
|
const scan = await scanJsonLines(file, MAX_TOTAL_BYTES - bytes, signal, (line) => {
|
|
249
|
-
if (fileMatches >= MAX_PER_FILE || matches.length >= limit) return false;
|
|
250
273
|
if (!line || !terms.some((t) => line.toLowerCase().includes(t))) return true;
|
|
251
274
|
const parsed = parseJsonLine(line);
|
|
252
275
|
if (!parsed) return true;
|
|
253
276
|
const et = entryText(parsed);
|
|
254
277
|
if (!et || !matchesAll(et.text, terms)) return true;
|
|
255
|
-
|
|
278
|
+
const bucket = et.role === "toolResult" ? fileTools : fileMessages;
|
|
279
|
+
bucket.push({
|
|
256
280
|
file,
|
|
257
281
|
project: projectLabel(dir),
|
|
258
282
|
date: fileDate(name),
|
|
@@ -261,18 +285,31 @@ export async function searchSessions(
|
|
|
261
285
|
currentSession: file === opts.currentFile,
|
|
262
286
|
sameProject: currentDirName !== undefined && dir === currentDirName,
|
|
263
287
|
});
|
|
264
|
-
|
|
265
|
-
return
|
|
288
|
+
if (bucket.length > perSession) bucket.shift();
|
|
289
|
+
return true;
|
|
266
290
|
});
|
|
267
291
|
bytes += scan.bytes;
|
|
268
292
|
filesScanned++;
|
|
269
|
-
|
|
270
|
-
|
|
293
|
+
// ponytail: prefer dialogue with one tool hit; add relevance ranking if noisy matches persist.
|
|
294
|
+
const keepMessages = fileTools.length && perSession > 1 ? perSession - 1 : perSession;
|
|
295
|
+
const selected = fileMessages.slice(-keepMessages).reverse();
|
|
296
|
+
const toolSlots = perSession - selected.length;
|
|
297
|
+
if (toolSlots) selected.push(...fileTools.slice(-toolSlots).reverse());
|
|
298
|
+
for (const hit of selected) {
|
|
299
|
+
if (skipped < offset) skipped++;
|
|
300
|
+
else if (matches.length < limit) matches.push(hit);
|
|
301
|
+
}
|
|
302
|
+
if (signal?.aborted || scan.incomplete) {
|
|
303
|
+
scanIncomplete = true;
|
|
304
|
+
break;
|
|
305
|
+
}
|
|
306
|
+
if (scan.oversizedLines > 0) scanIncomplete = true;
|
|
307
|
+
if (matches.length >= limit) {
|
|
308
|
+
limitReached = true;
|
|
271
309
|
break;
|
|
272
310
|
}
|
|
273
|
-
if (scan.oversizedLines > 0) truncated = true;
|
|
274
311
|
}
|
|
275
|
-
return { matches, bytesScanned: bytes, filesScanned, truncated };
|
|
312
|
+
return { matches, bytesScanned: bytes, filesScanned, truncated: scanIncomplete || limitReached, limitReached, scanIncomplete, perSession, offset };
|
|
276
313
|
}
|
|
277
314
|
|
|
278
315
|
export async function firstUserTitle(file: string, maxLen = 120): Promise<string | null> {
|
|
@@ -313,15 +350,26 @@ export function recentSessions(
|
|
|
313
350
|
}
|
|
314
351
|
|
|
315
352
|
export function formatResults(query: string, result: Awaited<ReturnType<typeof searchSessions>>): string {
|
|
316
|
-
const
|
|
353
|
+
const notes = [
|
|
354
|
+
result.limitReached ? "Result limit reached; increase offset to check for more." : "",
|
|
355
|
+
result.truncated && (!result.limitReached || result.scanIncomplete)
|
|
356
|
+
? "Search incomplete; narrow the search or check archive access."
|
|
357
|
+
: "",
|
|
358
|
+
].filter(Boolean);
|
|
359
|
+
const note = notes.length ? `\n(${notes.join(" ")})` : "";
|
|
317
360
|
if (result.matches.length === 0) {
|
|
318
|
-
return `No matches for "${query}" in ${result.filesScanned} scanned session files.${note}`;
|
|
361
|
+
return `No matches for "${query}"${result.offset ? ` after offset ${result.offset}` : ""} in ${result.filesScanned} scanned session files.${note}`;
|
|
362
|
+
}
|
|
363
|
+
const blocks: string[] = [];
|
|
364
|
+
let lastFile = "";
|
|
365
|
+
for (const m of result.matches) {
|
|
366
|
+
if (m.file !== lastFile) {
|
|
367
|
+
blocks.push(`${m.file}\n[${m.currentSession ? "this session" : m.sameProject ? "other session in this project" : m.project} | ${m.date}]`);
|
|
368
|
+
lastFile = m.file;
|
|
369
|
+
}
|
|
370
|
+
blocks[blocks.length - 1] += `\n- ${m.role}: ${m.excerpt.replace(/\s+/g, " ")}`;
|
|
319
371
|
}
|
|
320
|
-
|
|
321
|
-
(m) =>
|
|
322
|
-
`${m.file}\n[${m.currentSession ? "this session" : m.sameProject ? "this project" : m.project} | ${m.date} | ${m.role}] ${m.excerpt.replace(/\s+/g, " ")}`,
|
|
323
|
-
);
|
|
324
|
-
return `Found ${result.matches.length} match(es) for "${query}" — at most ${MAX_PER_FILE} per session file (read or grep the file for more):\n\n${lines.join("\n\n")}${note}`;
|
|
372
|
+
return `Found ${result.matches.length} match(es) for "${query}"${result.offset ? ` at offset ${result.offset}` : ""} — at most ${result.perSession ?? DEFAULT_PER_SESSION} per session file (read or grep the file for more):\n\n${blocks.join("\n\n")}${note}`;
|
|
325
373
|
}
|
|
326
374
|
|
|
327
375
|
export default async function piArchive(pi: ExtensionAPI) {
|
|
@@ -332,9 +380,9 @@ export default async function piArchive(pi: ExtensionAPI) {
|
|
|
332
380
|
name: "search_archive",
|
|
333
381
|
label: "Search archive",
|
|
334
382
|
description:
|
|
335
|
-
"Search
|
|
336
|
-
"
|
|
337
|
-
"
|
|
383
|
+
"Search raw Pi transcripts from this session and older sessions, including messages compacted away. " +
|
|
384
|
+
"If a compaction summary lacks a detail needed for the current task (exact code, output, error, or decision), search here instead of guessing. " +
|
|
385
|
+
"Results identify this session, another session in this project, or another project; old hits may be stale, so verify against current context or files.",
|
|
338
386
|
parameters: Type.Object({
|
|
339
387
|
query: Type.String({ description: "Search terms; all must appear (case-insensitive)" }),
|
|
340
388
|
session: Type.Optional(
|
|
@@ -347,6 +395,20 @@ export default async function piArchive(pi: ExtensionAPI) {
|
|
|
347
395
|
description: `Max matches (default 20, max ${MAX_SEARCH_LIMIT})`,
|
|
348
396
|
}),
|
|
349
397
|
),
|
|
398
|
+
perSession: Type.Optional(
|
|
399
|
+
Type.Integer({
|
|
400
|
+
minimum: 1,
|
|
401
|
+
maximum: MAX_PER_SESSION,
|
|
402
|
+
description: `Max hits per session file (default ${DEFAULT_PER_SESSION}, max ${MAX_PER_SESSION}); raise for more hits in one session`,
|
|
403
|
+
}),
|
|
404
|
+
),
|
|
405
|
+
offset: Type.Optional(
|
|
406
|
+
Type.Integer({
|
|
407
|
+
minimum: 0,
|
|
408
|
+
maximum: MAX_OFFSET,
|
|
409
|
+
description: `Skip ranked hits across sessions (default 0, max ${MAX_OFFSET}); keep perSession unchanged when paging. New messages can shift offsets.`,
|
|
410
|
+
}),
|
|
411
|
+
),
|
|
350
412
|
}),
|
|
351
413
|
async execute(_toolCallId, params, signal, onUpdate, ctx) {
|
|
352
414
|
const sessionDir = ctx.sessionManager.getSessionDir();
|
|
@@ -359,6 +421,8 @@ export default async function piArchive(pi: ExtensionAPI) {
|
|
|
359
421
|
currentDir: sessionDir,
|
|
360
422
|
sessionFilter: params.session,
|
|
361
423
|
limit: params.limit,
|
|
424
|
+
perSession: params.perSession,
|
|
425
|
+
offset: params.offset,
|
|
362
426
|
},
|
|
363
427
|
signal,
|
|
364
428
|
(files, hits) =>
|