pim-agent 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/bin/pim.ts +1 -1
  2. package/package.json +2 -2
  3. package/packages/core/src/extensions/CoreExtensions.ts +19 -16
  4. package/packages/core/src/extensions/apply-patch/render.ts +16 -55
  5. package/packages/core/src/extensions/system-prompt/index.ts +29 -21
  6. package/packages/core/src/extensions/system-prompt/prompt.ts +5 -1
  7. package/packages/core/src/picker/PickerService.ts +23 -1
  8. package/packages/core/src/picker/token.ts +12 -5
  9. package/packages/core/src/session/EventLog.ts +22 -79
  10. package/packages/core/src/session/SearchIndex.ts +752 -0
  11. package/packages/core/src/session/SearchTokens.ts +97 -0
  12. package/packages/core/src/session/SessionDigest.ts +161 -0
  13. package/packages/core/src/session/SessionHost.ts +17 -1
  14. package/packages/core/src/session/SessionMeta.ts +267 -0
  15. package/packages/core/src/session/SessionName.ts +18 -0
  16. package/packages/core/src/session/SessionRegistry.ts +89 -14
  17. package/packages/core/src/session/WriteMark.ts +15 -1
  18. package/packages/core/src/shared/DiffLines.ts +1 -1
  19. package/packages/core/src/shared/DiffPatch.ts +70 -0
  20. package/packages/core/src/shared/DiffRenderer.ts +1 -1
  21. package/packages/core/src/shared/FileWatch.ts +15 -1
  22. package/packages/core/src/shared/Git.ts +483 -42
  23. package/packages/core/src/shared/GitMonitor.ts +221 -0
  24. package/packages/core/src/shared/Levenshtein.ts +54 -1
  25. package/packages/core/src/shared/Proc.ts +21 -1
  26. package/packages/core/src/shared/RepoDiff.ts +741 -0
  27. package/packages/core/src/shared/Surface.ts +2 -0
  28. package/packages/core/src/shared/Updater.ts +21 -5
  29. package/packages/core/src/view/DiffExpand.ts +214 -0
  30. package/packages/core/src/view/DiffLayout.ts +7 -7
  31. package/packages/core/src/view/DiffPairs.ts +47 -0
  32. package/packages/core/src/view/MovePath.ts +70 -0
  33. package/packages/daemon/src/WebSurface.ts +1 -2
  34. package/packages/protocol/src/Command.ts +129 -2
  35. package/packages/protocol/src/Diff.ts +15 -0
  36. package/packages/protocol/src/ServerEvent.ts +157 -2
  37. package/packages/server/src/ClientConnection.ts +15 -3
  38. package/packages/server/src/ProbeClient.ts +90 -28
  39. package/packages/server/src/SessionCatalogue.ts +292 -38
  40. package/packages/server/src/SessionStream.ts +69 -43
  41. package/packages/server/src/StaticClient.ts +95 -12
  42. package/packages/server/src/WsGateway.ts +264 -29
  43. package/packages/server/src/probe.ts +61 -7
  44. package/packages/telegram/src/Session.ts +1 -0
  45. package/packages/tui/src/extensions/command-picker/index.ts +64 -32
  46. package/packages/tui/src/extensions/footer/index.ts +19 -43
  47. package/packages/tui/src/extensions/session-lease/index.ts +8 -3
  48. package/packages/web/dist/client/assets/commit-mono-latin-300-normal-B-iV2FbL.woff2 +0 -0
  49. package/packages/web/dist/client/assets/commit-mono-latin-300-normal-kM0OTZCv.woff +0 -0
  50. package/packages/web/dist/client/assets/{core-CGdQx9la.js → core-CA0aSzPu.js} +1 -1
  51. package/packages/web/dist/client/assets/index-B9be-i9d.css +1 -0
  52. package/packages/web/dist/client/assets/index-CpinM5NK.js +45 -0
  53. package/packages/web/dist/client/index.html +2 -2
  54. package/packages/protocol/src/Protocol.ts +0 -6
  55. package/packages/web/dist/client/assets/index-BMNwleWO.js +0 -40
  56. package/packages/web/dist/client/assets/index-CqjB2RPr.css +0 -1
@@ -0,0 +1,752 @@
1
+ import { parseSessionEntries } from "@earendil-works/pi-coding-agent";
2
+
3
+ import { Attachments } from "../attachments/Attachments";
4
+ import { Levenshtein } from "../shared/Levenshtein";
5
+ import { Pool } from "../shared/Pool";
6
+ import { MessageText } from "./MessageText";
7
+ import { SearchTokens, type SearchToken } from "./SearchTokens";
8
+ import {
9
+ NEWLINE,
10
+ SessionDigest,
11
+ type DigestParts,
12
+ type Durable,
13
+ } from "./SessionDigest";
14
+ import type { SessionSummary } from "./SessionRegistry";
15
+
16
+ /** Half-open character offsets into the string they mark, non-overlapping, ascending. */
17
+ export type SearchRange = readonly [start: number, end: number];
18
+
19
+ export type SearchSnippet = {
20
+ readonly seq: number;
21
+ readonly role: "user" | "assistant";
22
+ /** The windowed snippet, verbatim from the message. */
23
+ readonly text: string;
24
+ /** Offsets into `text`, not into the message. */
25
+ readonly ranges: readonly SearchRange[];
26
+ /** The window opened past the message's first word. */
27
+ readonly cutHead?: true;
28
+ };
29
+
30
+ export type SearchHit = {
31
+ readonly sessionId: string;
32
+ readonly cwd: string;
33
+ /** This machine's path to the session file: local to whoever reads it, and never fit for a wire. */
34
+ readonly path: string;
35
+ readonly title?: string;
36
+ readonly named?: true;
37
+ /** Offsets into the clamped `title`. */
38
+ readonly titleRanges: readonly SearchRange[];
39
+ /** The session's opening ask, for a row whose name is all that matched and so has no snippet to show. */
40
+ readonly opening?: string;
41
+ /** End of the last completed turn, falling back to when the session started: a row always has a clock to print. */
42
+ readonly settledAt: number;
43
+ readonly snippets: readonly SearchSnippet[];
44
+ /** Matching messages in this session, before the snippet cut. */
45
+ readonly total: number;
46
+ /** This hit was reached through typo expansion, so it ranks below every exact hit. */
47
+ readonly typos: boolean;
48
+ };
49
+
50
+ export type SearchAnswer = {
51
+ readonly hits: readonly SearchHit[];
52
+ /** Query words dropped because they were too rare; the UI says these out loud. */
53
+ readonly dropped: readonly string[];
54
+ /** Sessions searched. */
55
+ readonly scanned: number;
56
+ };
57
+
58
+ export type SearchOptions = {
59
+ readonly limit?: number;
60
+ readonly cwd?: string;
61
+ /** Applied before the limit, so the server can exclude sessions without core knowing why. */
62
+ readonly accept?: (sessionId: string) => boolean;
63
+ };
64
+
65
+ export type SearchIndexDeps = {
66
+ /** `SessionRegistry.list` in production; a fake in tests. */
67
+ readonly list: () => Promise<readonly SessionSummary[]>;
68
+ /** Only the refresh throttle reads it, and only a test replaces it. */
69
+ readonly now?: () => number;
70
+ };
71
+
72
+ type Role = "user" | "assistant";
73
+
74
+ type Turn = {
75
+ readonly seq: number;
76
+ readonly role: Role;
77
+ readonly text: string;
78
+ };
79
+
80
+ type Entry = {
81
+ readonly sessionId: string;
82
+ readonly cwd: string;
83
+ readonly path: string;
84
+ readonly createdAt: number;
85
+ modifiedAt: number;
86
+ /** Durable bytes taken from the file, and the resume point of the next tail read. */
87
+ offset: number;
88
+ /** Durable lines taken, so an appended one keeps counting from the right ordinal. */
89
+ seq: number;
90
+ parts: DigestParts;
91
+ digest: SessionDigest;
92
+ readonly turns: Turn[];
93
+ };
94
+
95
+ type Posting = {
96
+ readonly entry: Entry;
97
+ readonly turn: Turn;
98
+ };
99
+
100
+ /** What one file's read yielded: `full` replaces the entry, otherwise it extends it. */
101
+ type Read = {
102
+ readonly full: boolean;
103
+ readonly body?: Durable;
104
+ };
105
+
106
+ /** One file's read, folded into its entry and with the bytes already let go. */
107
+ type Taken = {
108
+ readonly entry: Entry;
109
+ /** Turns the entry held before this read, so only the appended ones are posted. */
110
+ readonly from: number;
111
+ /** A whole file read over an entry we already had, so every ordinal it owned is stale. */
112
+ readonly replaced: boolean;
113
+ };
114
+
115
+ type Plan = {
116
+ /** Terms reached with no typos: the word itself, and prefixes of it when it is last. */
117
+ readonly exact: readonly string[];
118
+ readonly typo: readonly string[];
119
+ };
120
+
121
+ type Group = {
122
+ readonly turns: Turn[];
123
+ title: boolean;
124
+ exact: boolean;
125
+ };
126
+
127
+ /**
128
+ * Everything `byRank` weighs, taken from the group alone: no message is
129
+ * tokenised until the sort has cut the losers, and four in five matched
130
+ * sessions never reach a snippet.
131
+ */
132
+ type Ranked = {
133
+ readonly entry: Entry;
134
+ readonly group: Group;
135
+ readonly terms: ReadonlySet<string>;
136
+ readonly typos: boolean;
137
+ readonly field: number;
138
+ readonly settledAt: number;
139
+ };
140
+
141
+ const FILE_READS = 16;
142
+
143
+ /** The granularity at which the catalogue already decided file changes become visible. */
144
+ const REFRESH_MS = 500;
145
+
146
+ /** Typesense's `typo_tokens_threshold`: below this many results, and only then, typos are allowed in. */
147
+ const ENOUGH = 1;
148
+
149
+ /** Typesense's `max_candidates`; unbounded expansion is where precision dies. */
150
+ const MAX_CANDIDATES = 4;
151
+
152
+ const MAX_TYPOS = 2;
153
+
154
+ /** Typesense's `highlight_affix_num_tokens`, for the run-up only: how far back of the match a window opens. */
155
+ const AFFIX = 4;
156
+
157
+ /**
158
+ * How far past the window's start a snippet runs. A row is cut to pixels, and
159
+ * only the row knows how many it has, so the tail is not a window at all: it is
160
+ * more text than the widest row can draw — the modal is capped at 30rem, which
161
+ * is about 59 of these characters — and the client's own ellipsis does the
162
+ * cutting. A word count here cut short rows shorter still, on the wide screens
163
+ * that had the most room to spare.
164
+ */
165
+ const TAIL = 100;
166
+
167
+ const SNIPPETS = 2;
168
+
169
+ /** What a row draws of an opening ask before its own box cuts it anyway. */
170
+ const PREVIEW = 200;
171
+
172
+ const LIMIT = 20;
173
+
174
+ /** Cheap reject before decoding a line: no text part, nothing to index. */
175
+ const TEXT = Buffer.from('"text"');
176
+
177
+ /**
178
+ * An inverted index over what was said in every session on disk: `user:text`
179
+ * and `assistant:text` only, which is a sixtieth of the bytes a session tree
180
+ * weighs. Built lazily on the first query and refreshed by later ones, never
181
+ * by a watcher, so the cold cost belongs to whoever searches.
182
+ */
183
+ export class SearchIndex {
184
+ private readonly list: () => Promise<readonly SessionSummary[]>;
185
+ private readonly now: () => number;
186
+ private readonly entries = new Map<string, Entry>();
187
+ private readonly postings = new Map<string, number[]>();
188
+ private readonly titles = new Map<string, Entry[]>();
189
+ private turns: Posting[] = [];
190
+ private vocabulary: readonly string[] = [];
191
+ private vocabularyStale = true;
192
+ private built?: Promise<void>;
193
+ private syncing?: Promise<void>;
194
+ private syncedAt = 0;
195
+
196
+ public constructor(deps: SearchIndexDeps) {
197
+ this.list = deps.list;
198
+ this.now = deps.now ?? Date.now;
199
+ }
200
+
201
+ /** Memoised: concurrent callers share one build. */
202
+ public ready(): Promise<void> {
203
+ this.built ??= this.sync().catch((error: unknown) => {
204
+ this.built = undefined;
205
+ throw error;
206
+ });
207
+ return this.built;
208
+ }
209
+
210
+ public async search(
211
+ query: string,
212
+ options: SearchOptions = {}
213
+ ): Promise<SearchAnswer> {
214
+ await this.ready();
215
+ await this.refresh();
216
+
217
+ const scope = this.scope(options);
218
+ const dropped: string[] = [];
219
+ let kept = SearchTokens.words(query);
220
+ let ranked = kept.length === 0 ? [] : this.run(kept, scope);
221
+ while (ranked.length < ENOUGH && kept.length > 1) {
222
+ const rarest = this.rarest(kept);
223
+ dropped.push(kept[rarest]!);
224
+ kept = kept.filter((_, index) => index !== rarest);
225
+ ranked = this.run(kept, scope);
226
+ }
227
+
228
+ return {
229
+ hits: ranked
230
+ .sort(byRank)
231
+ .slice(0, options.limit ?? LIMIT)
232
+ .map((one) => hitOf(one)),
233
+ dropped,
234
+ scanned: scope.size,
235
+ };
236
+ }
237
+
238
+ /** Exact and prefix first; typos only widen a result set that came back thin. */
239
+ private run(words: readonly string[], scope: ReadonlySet<Entry>): Ranked[] {
240
+ const reach = Math.min(MAX_TYPOS, Math.max(...words.map(gateOf)));
241
+ const last = words.length - 1;
242
+ let ranked: Ranked[] = [];
243
+ for (let typos = 0; typos <= reach; typos++) {
244
+ const plans = words.map((word, at) =>
245
+ this.plan(word, at === last, typos)
246
+ );
247
+ ranked = this.collect(plans, scope);
248
+ if (ranked.length >= ENOUGH) {
249
+ break;
250
+ }
251
+ }
252
+ return ranked;
253
+ }
254
+
255
+ private plan(word: string, last: boolean, typos: number): Plan {
256
+ const exact = this.known(word) ? [word] : [];
257
+ if (last) {
258
+ const prefixes = this.vocabularyOf().filter(
259
+ (term) => term !== word && term.startsWith(word)
260
+ );
261
+ exact.push(...this.best(prefixes));
262
+ }
263
+
264
+ const allowed = Math.min(typos, gateOf(word));
265
+ if (allowed === 0) {
266
+ return { exact, typo: [] };
267
+ }
268
+ const reached = new Set(exact);
269
+ const near = this.vocabularyOf()
270
+ .filter((term) => !reached.has(term))
271
+ .map((term) => ({
272
+ term,
273
+ distance: Levenshtein.damerau(word, term, allowed),
274
+ }))
275
+ .filter((candidate) => candidate.distance <= allowed)
276
+ .sort(
277
+ (a, b) =>
278
+ a.distance - b.distance ||
279
+ this.frequencyOf(b.term) - this.frequencyOf(a.term) ||
280
+ a.term.localeCompare(b.term)
281
+ );
282
+ return {
283
+ exact,
284
+ typo: near.slice(0, MAX_CANDIDATES).map((candidate) => candidate.term),
285
+ };
286
+ }
287
+
288
+ private collect(plans: readonly Plan[], scope: ReadonlySet<Entry>): Ranked[] {
289
+ const turnsExact = plans.map((plan) =>
290
+ this.union(this.postings, plan.exact)
291
+ );
292
+ const titlesExact = plans.map((plan) =>
293
+ this.union(this.titles, plan.exact)
294
+ );
295
+ const everything = plans.map((plan) => [...plan.exact, ...plan.typo]);
296
+ const turns = everything.map((terms) => this.union(this.postings, terms));
297
+ const titles = everything.map((terms) => this.union(this.titles, terms));
298
+ const terms = new Set(everything.flat());
299
+
300
+ const groups = new Map<Entry, Group>();
301
+ for (const ordinal of intersect(turns)) {
302
+ const posting = this.turns[ordinal]!;
303
+ if (!scope.has(posting.entry)) {
304
+ continue;
305
+ }
306
+ const group = groupIn(groups, posting.entry);
307
+ group.turns.push(posting.turn);
308
+ group.exact ||= turnsExact.every((found) => found.has(ordinal));
309
+ }
310
+ for (const entry of intersect(titles)) {
311
+ if (!scope.has(entry)) {
312
+ continue;
313
+ }
314
+ const group = groupIn(groups, entry);
315
+ group.title = true;
316
+ group.exact ||= titlesExact.every((found) => found.has(entry));
317
+ }
318
+
319
+ return [...groups].map(([entry, group]) => rankedOf(entry, group, terms));
320
+ }
321
+
322
+ private scope(options: SearchOptions): ReadonlySet<Entry> {
323
+ const scoped = new Set<Entry>();
324
+ for (const entry of this.entries.values()) {
325
+ const wanted =
326
+ (options.cwd === undefined || entry.cwd === options.cwd) &&
327
+ (options.accept === undefined || options.accept(entry.sessionId));
328
+ if (wanted) {
329
+ scoped.add(entry);
330
+ }
331
+ }
332
+ return scoped;
333
+ }
334
+
335
+ /** "Words that have the least individual results are dropped first." */
336
+ private rarest(words: readonly string[]): number {
337
+ let rarest = 0;
338
+ for (const [at, word] of words.entries()) {
339
+ if (this.frequencyOf(word) < this.frequencyOf(words[rarest]!)) {
340
+ rarest = at;
341
+ }
342
+ }
343
+ return rarest;
344
+ }
345
+
346
+ private frequencyOf(term: string): number {
347
+ return (
348
+ (this.postings.get(term)?.length ?? 0) +
349
+ (this.titles.get(term)?.length ?? 0)
350
+ );
351
+ }
352
+
353
+ private known(term: string): boolean {
354
+ return this.postings.has(term) || this.titles.has(term);
355
+ }
356
+
357
+ private best(terms: readonly string[]): readonly string[] {
358
+ return [...terms]
359
+ .sort(
360
+ (a, b) =>
361
+ this.frequencyOf(b) - this.frequencyOf(a) || a.localeCompare(b)
362
+ )
363
+ .slice(0, MAX_CANDIDATES);
364
+ }
365
+
366
+ private union<T>(
367
+ index: ReadonlyMap<string, T[]>,
368
+ terms: readonly string[]
369
+ ): ReadonlySet<T> {
370
+ const found = new Set<T>();
371
+ for (const term of terms) {
372
+ for (const value of index.get(term) ?? []) {
373
+ found.add(value);
374
+ }
375
+ }
376
+ return found;
377
+ }
378
+
379
+ private vocabularyOf(): readonly string[] {
380
+ if (this.vocabularyStale) {
381
+ this.vocabulary = [
382
+ ...new Set([...this.postings.keys(), ...this.titles.keys()]),
383
+ ];
384
+ this.vocabularyStale = false;
385
+ }
386
+ return this.vocabulary;
387
+ }
388
+
389
+ private async refresh(): Promise<void> {
390
+ if (this.now() - this.syncedAt < REFRESH_MS) {
391
+ return;
392
+ }
393
+ this.syncing ??= this.sync().finally(() => {
394
+ this.syncing = undefined;
395
+ });
396
+ await this.syncing;
397
+ }
398
+
399
+ private async sync(): Promise<void> {
400
+ const summaries = await this.list();
401
+ const taken = await Pool.mapPooled(summaries, FILE_READS, (summary) =>
402
+ this.take(summary)
403
+ );
404
+
405
+ const alive = new Set(summaries.map((summary) => summary.sessionId));
406
+ const appended: { readonly entry: Entry; readonly from: number }[] = [];
407
+ let replaced = false;
408
+ for (const one of taken) {
409
+ if (one === undefined) {
410
+ continue;
411
+ }
412
+ replaced ||= one.replaced;
413
+ this.entries.set(one.entry.sessionId, one.entry);
414
+ appended.push({ entry: one.entry, from: one.from });
415
+ }
416
+
417
+ for (const sessionId of this.entries.keys()) {
418
+ if (!alive.has(sessionId)) {
419
+ this.entries.delete(sessionId);
420
+ replaced = true;
421
+ }
422
+ }
423
+
424
+ // A posting is an ordinal into one flat array of turns, which only holds
425
+ // while every entry keeps the turns it had; a replaced or departed session
426
+ // invalidates the lot, and an appended one does not.
427
+ if (replaced) {
428
+ this.reindex();
429
+ } else {
430
+ for (const { entry, from } of appended) {
431
+ this.post(entry, entry.turns.slice(from));
432
+ }
433
+ }
434
+ this.retitle();
435
+ this.vocabularyStale = true;
436
+ this.syncedAt = this.now();
437
+ }
438
+
439
+ /**
440
+ * One file's bytes end here, with the turns and parts taken out of them:
441
+ * held until the pool drained instead, a whole session tree would be
442
+ * resident at once. The entry lands in `entries` in the caller's order.
443
+ */
444
+ private async take(summary: SessionSummary): Promise<Taken | undefined> {
445
+ const read = await this.read(summary);
446
+ if (read === undefined) {
447
+ return undefined;
448
+ }
449
+ const known = this.entries.get(summary.sessionId);
450
+ const entry = read.full ? blank(summary) : known!;
451
+ const from = entry.turns.length;
452
+ entry.modifiedAt = summary.modifiedAt;
453
+ if (read.body !== undefined) {
454
+ absorb(entry, read.body);
455
+ }
456
+ return { entry, from, replaced: read.full && known !== undefined };
457
+ }
458
+
459
+ /**
460
+ * A session file is append-only, so the usual read is the bytes past the
461
+ * stored offset. Only a file that shrank or whose mtime went backwards has
462
+ * been rewritten under us, and only that forces the whole file again.
463
+ */
464
+ private async read(summary: SessionSummary): Promise<Read | undefined> {
465
+ const known = this.entries.get(summary.sessionId);
466
+ const file = Bun.file(summary.path);
467
+ try {
468
+ if (known === undefined) {
469
+ return { full: true, body: SessionDigest.durable(await file.bytes()) };
470
+ }
471
+ if (summary.modifiedAt === known.modifiedAt) {
472
+ return undefined;
473
+ }
474
+ const shrank = (await file.stat()).size < known.offset;
475
+ if (shrank || summary.modifiedAt < known.modifiedAt) {
476
+ return { full: true, body: SessionDigest.durable(await file.bytes()) };
477
+ }
478
+ const tail = await file.slice(known.offset).bytes();
479
+ return { full: false, body: SessionDigest.durable(tail) };
480
+ } catch {
481
+ return undefined;
482
+ }
483
+ }
484
+
485
+ private post(entry: Entry, turns: readonly Turn[]): void {
486
+ for (const turn of turns) {
487
+ const ordinal = this.turns.length;
488
+ this.turns.push({ entry, turn });
489
+ for (const token of tokensOf(turn.text)) {
490
+ add(this.postings, token, ordinal);
491
+ }
492
+ }
493
+ }
494
+
495
+ private reindex(): void {
496
+ this.turns = [];
497
+ this.postings.clear();
498
+ for (const entry of this.entries.values()) {
499
+ this.post(entry, entry.turns);
500
+ }
501
+ }
502
+
503
+ private retitle(): void {
504
+ this.titles.clear();
505
+ for (const entry of this.entries.values()) {
506
+ const title = entry.digest.title;
507
+ if (title === undefined) {
508
+ continue;
509
+ }
510
+ for (const token of tokensOf(title)) {
511
+ add(this.titles, token, entry);
512
+ }
513
+ }
514
+ }
515
+ }
516
+
517
+ function blank(summary: SessionSummary): Entry {
518
+ return {
519
+ sessionId: summary.sessionId,
520
+ cwd: summary.cwd,
521
+ path: summary.path,
522
+ createdAt: summary.createdAt,
523
+ modifiedAt: summary.modifiedAt,
524
+ offset: 0,
525
+ seq: 0,
526
+ parts: {},
527
+ digest: {},
528
+ turns: [],
529
+ };
530
+ }
531
+
532
+ function absorb(entry: Entry, body: Durable): void {
533
+ const read = spokenIn(body, entry.seq);
534
+ entry.turns.push(...read.turns);
535
+ entry.seq += read.lines;
536
+ entry.offset += body.end;
537
+ entry.parts = SessionDigest.merge(entry.parts, SessionDigest.partsOf(body));
538
+ entry.digest = SessionDigest.of(entry.parts);
539
+ }
540
+
541
+ function spokenIn(
542
+ body: Durable,
543
+ from: number
544
+ ): { readonly turns: readonly Turn[]; readonly lines: number } {
545
+ const { bytes, end } = body;
546
+ const turns: Turn[] = [];
547
+ let marker = bytes.indexOf(TEXT);
548
+ let at = 0;
549
+ let lines = 0;
550
+ while (at < end) {
551
+ const to = bytes.indexOf(NEWLINE, at);
552
+ lines += 1;
553
+ if (marker !== -1 && marker < at) {
554
+ marker = bytes.indexOf(TEXT, at);
555
+ }
556
+ if (marker !== -1 && marker < to) {
557
+ const turn = turnOf(bytes.toString("utf8", at, to), from + lines);
558
+ if (turn !== undefined) {
559
+ turns.push(turn);
560
+ }
561
+ }
562
+ at = to + 1;
563
+ }
564
+ return { turns, lines };
565
+ }
566
+
567
+ function turnOf(line: string, seq: number): Turn | undefined {
568
+ const entry = parseSessionEntries(line)[0];
569
+ if (entry?.type !== "message") {
570
+ return undefined;
571
+ }
572
+ const message = entry.message;
573
+ if (message.role !== "user" && message.role !== "assistant") {
574
+ return undefined;
575
+ }
576
+ const said = MessageText.textOf(message.content);
577
+ const text = (
578
+ message.role === "user" ? Attachments.parse(said).text : said
579
+ ).trim();
580
+ return text === "" ? undefined : { seq, role: message.role, text };
581
+ }
582
+
583
+ function rankedOf(
584
+ entry: Entry,
585
+ group: Group,
586
+ terms: ReadonlySet<string>
587
+ ): Ranked {
588
+ const { title, settledAt } = entry.digest;
589
+ return {
590
+ entry,
591
+ group,
592
+ terms,
593
+ typos: !group.exact,
594
+ field: fieldOf(group, title),
595
+ settledAt: settledAt ?? entry.createdAt,
596
+ };
597
+ }
598
+
599
+ function hitOf(ranked: Ranked): SearchHit {
600
+ const { entry, group, terms } = ranked;
601
+ const { title, named } = entry.digest;
602
+ const spoken = [...group.turns].sort(byRecognition);
603
+ const opening = entry.parts.opening;
604
+ return {
605
+ sessionId: entry.sessionId,
606
+ cwd: entry.cwd,
607
+ path: entry.path,
608
+ ...(title === undefined ? {} : { title }),
609
+ ...(named === undefined ? {} : { named }),
610
+ ...(named === true && opening !== undefined
611
+ ? { opening: opening.slice(0, PREVIEW) }
612
+ : {}),
613
+ settledAt: ranked.settledAt,
614
+ titleRanges:
615
+ group.title && title !== undefined ? rangesOf(title, terms) : [],
616
+ snippets: spoken.slice(0, SNIPPETS).map((turn) => snippetOf(turn, terms)),
617
+ total: group.turns.length,
618
+ typos: ranked.typos,
619
+ };
620
+ }
621
+
622
+ function snippetOf(turn: Turn, terms: ReadonlySet<string>): SearchSnippet {
623
+ const tokens = [...SearchTokens.scan(turn.text)].sort(
624
+ (a, b) => a.start - b.start || b.end - a.end
625
+ );
626
+ const at = tokens.findIndex((token) => terms.has(token.text));
627
+ const head = at - AFFIX;
628
+ const from = head <= 0 ? 0 : tokens[head]!.start;
629
+ const to = Math.min(turn.text.length, from + TAIL);
630
+ return {
631
+ seq: turn.seq,
632
+ role: turn.role,
633
+ text: turn.text.slice(from, to),
634
+ ranges: rangesIn(tokens, terms, from, to),
635
+ ...(from > 0 ? { cutHead: true as const } : {}),
636
+ };
637
+ }
638
+
639
+ function rangesOf(
640
+ text: string,
641
+ terms: ReadonlySet<string>
642
+ ): readonly SearchRange[] {
643
+ return rangesIn(SearchTokens.scan(text), terms, 0, text.length);
644
+ }
645
+
646
+ /** Offsets into the cut the caller is about to make, not into the string the tokens came from. */
647
+ function rangesIn(
648
+ tokens: readonly SearchToken[],
649
+ terms: ReadonlySet<string>,
650
+ from: number,
651
+ to: number
652
+ ): readonly SearchRange[] {
653
+ return merge(
654
+ tokens
655
+ .filter(
656
+ (token) =>
657
+ terms.has(token.text) && token.start >= from && token.end <= to
658
+ )
659
+ .map((token) => [token.start - from, token.end - from] as SearchRange)
660
+ );
661
+ }
662
+
663
+ function merge(ranges: readonly SearchRange[]): readonly SearchRange[] {
664
+ const marked: SearchRange[] = [];
665
+ for (const [start, end] of [...ranges].sort(
666
+ (a, b) => a[0] - b[0] || a[1] - b[1]
667
+ )) {
668
+ const last = marked.at(-1);
669
+ if (last !== undefined && start <= last[1]) {
670
+ marked[marked.length - 1] = [last[0], Math.max(last[1], end)];
671
+ } else {
672
+ marked.push([start, end]);
673
+ }
674
+ }
675
+ return marked;
676
+ }
677
+
678
+ /** What you asked is more memorable than what the agent answered. */
679
+ function byRecognition(a: Turn, b: Turn): number {
680
+ return a.role === b.role ? a.seq - b.seq : a.role === "user" ? -1 : 1;
681
+ }
682
+
683
+ /** The turn a hit will lead with, without sorting the ones it will never show. */
684
+ function leadOf(turns: readonly Turn[]): Turn | undefined {
685
+ let lead: Turn | undefined;
686
+ for (const turn of turns) {
687
+ if (lead === undefined || byRecognition(turn, lead) < 0) {
688
+ lead = turn;
689
+ }
690
+ }
691
+ return lead;
692
+ }
693
+
694
+ function byRank(a: Ranked, b: Ranked): number {
695
+ if (a.typos !== b.typos) {
696
+ return a.typos ? 1 : -1;
697
+ }
698
+ return a.field - b.field || b.settledAt - a.settledAt;
699
+ }
700
+
701
+ /**
702
+ * A title match outranks a spoken one. The title index is built from the very
703
+ * tokens `rangesOf` marks, so a titled group always has a range to show, and
704
+ * the field is settled without cutting a single snippet.
705
+ */
706
+ function fieldOf(group: Group, title: string | undefined): number {
707
+ if (group.title && title !== undefined) {
708
+ return 0;
709
+ }
710
+ return leadOf(group.turns)?.role === "user" ? 1 : 2;
711
+ }
712
+
713
+ /** Typesense's `min_len_1typo` and `min_len_2typo`. */
714
+ function gateOf(word: string): number {
715
+ if (word.length < 4) {
716
+ return 0;
717
+ }
718
+ return word.length < 7 ? 1 : 2;
719
+ }
720
+
721
+ function tokensOf(text: string): ReadonlySet<string> {
722
+ return new Set(SearchTokens.scan(text).map((token) => token.text));
723
+ }
724
+
725
+ function add<T>(into: Map<string, T[]>, key: string, value: T): void {
726
+ const found = into.get(key);
727
+ if (found === undefined) {
728
+ into.set(key, [value]);
729
+ } else {
730
+ found.push(value);
731
+ }
732
+ }
733
+
734
+ function groupIn(groups: Map<Entry, Group>, entry: Entry): Group {
735
+ const found = groups.get(entry) ?? { turns: [], title: false, exact: false };
736
+ groups.set(entry, found);
737
+ return found;
738
+ }
739
+
740
+ function intersect<T>(sets: readonly ReadonlySet<T>[]): ReadonlySet<T> {
741
+ const [first, ...rest] = [...sets].sort((a, b) => a.size - b.size);
742
+ if (first === undefined) {
743
+ return new Set<T>();
744
+ }
745
+ const found = new Set<T>();
746
+ for (const value of first) {
747
+ if (rest.every((set) => set.has(value))) {
748
+ found.add(value);
749
+ }
750
+ }
751
+ return found;
752
+ }