@zosmaai/pi-llm-wiki 0.10.7 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/CHANGELOG.md +4 -0
  2. package/README.de.md +35 -4
  3. package/README.es.md +260 -170
  4. package/README.fr.md +35 -4
  5. package/README.hi.md +35 -4
  6. package/README.ja.md +35 -4
  7. package/README.ko.md +35 -4
  8. package/README.md +38 -3
  9. package/README.pt.md +35 -4
  10. package/README.ru.md +35 -4
  11. package/README.zh.md +260 -170
  12. package/assets/demo.gif +0 -0
  13. package/dist/extensions/llm-wiki/lib/bootstrap.js +71 -0
  14. package/dist/extensions/llm-wiki/lib/embeddings.js +401 -0
  15. package/dist/extensions/llm-wiki/lib/guardrails.js +232 -0
  16. package/dist/extensions/llm-wiki/lib/indexing.js +78 -0
  17. package/dist/extensions/llm-wiki/lib/ingest-worker.js +310 -0
  18. package/dist/extensions/llm-wiki/lib/inject.js +65 -0
  19. package/dist/extensions/llm-wiki/lib/knowledge-document.js +442 -0
  20. package/dist/extensions/llm-wiki/lib/knowledge-links.js +206 -0
  21. package/dist/extensions/llm-wiki/lib/legacy-repair.js +443 -0
  22. package/dist/extensions/llm-wiki/lib/metadata.js +499 -0
  23. package/dist/extensions/llm-wiki/lib/model-command.js +86 -0
  24. package/dist/extensions/llm-wiki/lib/observation.js +283 -0
  25. package/dist/extensions/llm-wiki/lib/recall.js +875 -0
  26. package/dist/extensions/llm-wiki/lib/retro.js +158 -0
  27. package/dist/extensions/llm-wiki/lib/runtime.js +191 -0
  28. package/dist/extensions/llm-wiki/lib/source-extractors.js +426 -0
  29. package/dist/extensions/llm-wiki/lib/source-packet.js +229 -0
  30. package/dist/extensions/llm-wiki/lib/subagent.js +41 -0
  31. package/dist/extensions/llm-wiki/lib/task-config.js +172 -0
  32. package/dist/extensions/llm-wiki/lib/tools.js +1192 -0
  33. package/dist/extensions/llm-wiki/lib/trajectories-command.js +51 -0
  34. package/dist/extensions/llm-wiki/lib/trajectory.js +467 -0
  35. package/dist/extensions/llm-wiki/lib/utils.js +347 -0
  36. package/dist/extensions/llm-wiki/lib/vault-format.js +247 -0
  37. package/dist/extensions/llm-wiki/lib/visible-status.js +31 -0
  38. package/dist/extensions/llm-wiki/lib/wiki-service.js +128 -0
  39. package/dist/mcp/exec.js +121 -0
  40. package/dist/mcp/index.js +229 -0
  41. package/dist/mcp/operations.js +130 -0
  42. package/dist/package.json +1 -0
  43. package/docs/superpowers/plans/2026-08-02-okf-foundation.md +1579 -0
  44. package/docs/superpowers/plans/2026-08-03-okf-foundation-remediation.md +3005 -0
  45. package/docs/superpowers/plans/2026-08-06-okf-foundation-release-remediation.md +1174 -0
  46. package/docs/superpowers/specs/2026-08-02-okf-foundation-design.md +578 -0
  47. package/docs/superpowers/specs/2026-08-02-okf-v0.2-interoperability-design.md +538 -0
  48. package/extensions/llm-wiki/index.ts +22 -36
  49. package/extensions/llm-wiki/lib/bootstrap.ts +84 -0
  50. package/extensions/llm-wiki/lib/embeddings.ts +9 -3
  51. package/extensions/llm-wiki/lib/guardrails.ts +174 -29
  52. package/extensions/llm-wiki/lib/indexing.ts +2 -1
  53. package/extensions/llm-wiki/lib/ingest-worker.ts +170 -29
  54. package/extensions/llm-wiki/lib/knowledge-document.ts +661 -0
  55. package/extensions/llm-wiki/lib/knowledge-links.ts +282 -0
  56. package/extensions/llm-wiki/lib/legacy-repair.ts +572 -0
  57. package/extensions/llm-wiki/lib/metadata.ts +531 -116
  58. package/extensions/llm-wiki/lib/observation.ts +37 -43
  59. package/extensions/llm-wiki/lib/recall.ts +61 -33
  60. package/extensions/llm-wiki/lib/retro.ts +65 -41
  61. package/extensions/llm-wiki/lib/source-extractors.ts +12 -17
  62. package/extensions/llm-wiki/lib/source-packet.ts +44 -31
  63. package/extensions/llm-wiki/lib/tools.ts +406 -348
  64. package/extensions/llm-wiki/lib/trajectory.ts +15 -1
  65. package/extensions/llm-wiki/lib/utils.ts +121 -130
  66. package/extensions/llm-wiki/lib/vault-format.ts +363 -0
  67. package/extensions/llm-wiki/lib/wiki-service.ts +183 -0
  68. package/mcp/exec.ts +122 -0
  69. package/mcp/index.ts +60 -250
  70. package/mcp/operations.ts +176 -0
  71. package/package.json +8 -2
  72. package/scripts/migrate-llm-wiki.js +801 -0
  73. package/skills/llm-wiki/SKILL.md +8 -6
@@ -0,0 +1,875 @@
1
+ import { existsSync, readFileSync } from "node:fs";
2
+ import { join } from "node:path";
3
+ import { Type } from "typebox";
4
+ import { cosineSimilarity, normalizeVector, readEmbeddingStore, resolveEmbedder, } from "./embeddings.js";
5
+ import { parseKnowledgeDocument } from "./knowledge-document.js";
6
+ import { getPersonalWikiPaths, isPersonalVault, readJson, resolveVaultPaths, } from "./utils.js";
7
+ import { inspectVaultFormat } from "./vault-format.js";
8
+ /** Default blend weight when none is configured. */
9
+ export const DEFAULT_SEMANTIC_WEIGHT = 0.5;
10
+ /**
11
+ * Lexical points a perfect (cosine = 1) semantic match is worth at full
12
+ * weight. Chosen so a strong paraphrase match (cosine ≳ 0.84) at the default
13
+ * weight (0.5) clears the auto-injection threshold (minScore = 5) on its own,
14
+ * while weak/incidental similarity stays below it.
15
+ */
16
+ export const SEMANTIC_SCALE = 12;
17
+ /**
18
+ * Minimum cosine for a page with NO lexical match to even be considered a
19
+ * semantic candidate. Keeps the candidate set bounded (near-orthogonal pages
20
+ * are ignored) instead of pulling in the entire embedded vault.
21
+ */
22
+ export const SEMANTIC_MIN_COSINE = 0.2;
23
+ /**
24
+ * Blend a lexical score with a cosine similarity. The lexical score keeps its
25
+ * original absolute scale (so `minScore` semantics survive); the semantic
26
+ * signal is added as a bounded, weighted boost on a comparable scale. With no
27
+ * semantic signal (cosine ≤ 0) this is the identity on the lexical score, so
28
+ * the pure-lexical path is preserved exactly.
29
+ */
30
+ export function fuseScores(lexical, cosine, weight) {
31
+ return lexical + weight * SEMANTIC_SCALE * Math.max(cosine, 0);
32
+ }
33
+ /**
34
+ * Normalize text for recall matching.
35
+ *
36
+ * Wiki queries are often short and multilingual (for example: "继续学习pi").
37
+ * Normalization keeps CJK characters intact, lowercases Latin text, removes
38
+ * punctuation boundaries, and makes hyphenated page IDs match space-separated
39
+ * queries.
40
+ */
41
+ function normalizeText(value) {
42
+ return flattenSearchValue(value)
43
+ .toLowerCase()
44
+ .normalize("NFKC")
45
+ .replace(/[\-_./\\]+/g, " ")
46
+ .replace(/[\p{P}\p{S}]+/gu, " ")
47
+ .replace(/\s+/g, " ")
48
+ .trim();
49
+ }
50
+ function compactText(value) {
51
+ return value.replace(/\s+/g, "");
52
+ }
53
+ function flattenSearchValue(value) {
54
+ if (value == null)
55
+ return "";
56
+ if (Array.isArray(value))
57
+ return value.map(flattenSearchValue).join(" ");
58
+ if (typeof value === "object")
59
+ return Object.values(value).map(flattenSearchValue).join(" ");
60
+ return String(value);
61
+ }
62
+ function unique(values) {
63
+ return [...new Set(values.filter(Boolean))];
64
+ }
65
+ /**
66
+ * Tokenize with support for CJK short queries and English/kebab-case terms.
67
+ *
68
+ * Besides whitespace tokens, this returns Latin/digit runs ("pi", "recall")
69
+ * and overlapping CJK bigrams/trigrams. The full normalized query is also kept
70
+ * so exact short phrases still rank highest.
71
+ */
72
+ function queryTerms(query) {
73
+ const normalized = normalizeText(query);
74
+ const compact = compactText(normalized);
75
+ const terms = [];
76
+ if (normalized)
77
+ terms.push(normalized);
78
+ if (compact && compact !== normalized)
79
+ terms.push(compact);
80
+ for (const part of normalized.split(/\s+/)) {
81
+ if (part.length >= 2)
82
+ terms.push(part);
83
+ }
84
+ const latinRuns = normalized.match(/[a-z0-9]{2,}/g) ?? [];
85
+ terms.push(...latinRuns);
86
+ const cjkRuns = normalized.match(/[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}]+/gu) ?? [];
87
+ for (const run of cjkRuns) {
88
+ for (let size = 2; size <= 3; size++) {
89
+ if (run.length < size)
90
+ continue;
91
+ for (let i = 0; i <= run.length - size; i++) {
92
+ terms.push(run.slice(i, i + size));
93
+ }
94
+ }
95
+ }
96
+ return unique(terms).slice(0, 30);
97
+ }
98
+ function includesTerm(haystack, term) {
99
+ if (!haystack || !term)
100
+ return false;
101
+ return haystack.includes(term) || compactText(haystack).includes(compactText(term));
102
+ }
103
+ function scoreField(value, terms, weight) {
104
+ const text = normalizeText(value);
105
+ if (!text)
106
+ return 0;
107
+ let score = 0;
108
+ for (const term of terms) {
109
+ if (includesTerm(text, term))
110
+ score += weight;
111
+ }
112
+ return score;
113
+ }
114
+ // ─── Common English stopwords ─────────────────────────
115
+ const STOPWORDS = new Set([
116
+ "the",
117
+ "this",
118
+ "that",
119
+ "with",
120
+ "from",
121
+ "have",
122
+ "been",
123
+ "were",
124
+ "they",
125
+ "their",
126
+ "them",
127
+ "will",
128
+ "would",
129
+ "could",
130
+ "should",
131
+ "about",
132
+ "there",
133
+ "which",
134
+ "what",
135
+ "when",
136
+ "where",
137
+ "than",
138
+ "then",
139
+ "also",
140
+ "just",
141
+ "more",
142
+ "some",
143
+ "such",
144
+ "only",
145
+ "other",
146
+ "into",
147
+ "over",
148
+ "very",
149
+ "after",
150
+ "before",
151
+ "because",
152
+ "between",
153
+ "through",
154
+ "during",
155
+ "without",
156
+ "within",
157
+ "along",
158
+ "these",
159
+ "those",
160
+ "page",
161
+ "section",
162
+ "note",
163
+ "info",
164
+ "type",
165
+ "used",
166
+ "using",
167
+ ]);
168
+ /**
169
+ * Split a page's body into chunks by headings.
170
+ * Each heading and its following content become one chunk.
171
+ * Content before the first heading becomes the intro chunk.
172
+ */
173
+ function chunkPage(body) {
174
+ if (!body.trim())
175
+ return [];
176
+ const chunks = [];
177
+ const lines = body.split("\n");
178
+ let currentHeading = "";
179
+ let currentLevel = 0;
180
+ let currentContent = [];
181
+ for (const line of lines) {
182
+ const headingMatch = line.trim().match(/^(#{1,6})\s+(.+)$/);
183
+ if (headingMatch) {
184
+ // Save previous chunk
185
+ if (currentContent.length > 0 || currentHeading) {
186
+ chunks.push({
187
+ heading: currentHeading,
188
+ content: currentContent.join("\n").trim(),
189
+ level: currentLevel,
190
+ });
191
+ }
192
+ currentHeading = headingMatch[2].trim();
193
+ currentLevel = headingMatch[1].length;
194
+ currentContent = [];
195
+ }
196
+ else {
197
+ currentContent.push(line);
198
+ }
199
+ }
200
+ // Save last chunk
201
+ if (currentContent.length > 0 || currentHeading) {
202
+ chunks.push({
203
+ heading: currentHeading,
204
+ content: currentContent.join("\n").trim(),
205
+ level: currentLevel,
206
+ });
207
+ }
208
+ return chunks;
209
+ }
210
+ function parsePage(path, id) {
211
+ if (!existsSync(path))
212
+ return undefined;
213
+ const content = readFileSync(path, "utf-8");
214
+ const result = parseKnowledgeDocument(content, `${id}.md`);
215
+ if (!result.ok)
216
+ return undefined;
217
+ return { frontmatter: result.document.frontmatter, body: result.document.body };
218
+ }
219
+ function pagePreview(content) {
220
+ const body = content.replace(/^---\n[\s\S]*?\n---\n?/, "");
221
+ return body.trim().slice(0, 200).replace(/\n/g, " ");
222
+ }
223
+ /**
224
+ * Get a preview of the best-matching chunk, or fall back to the page intro.
225
+ * Shows the heading (if any) and the first ~200 chars of content.
226
+ */
227
+ function chunkPreview(heading, content) {
228
+ const trimmed = content.slice(0, 180).replace(/\n/g, " ");
229
+ if (heading) {
230
+ return `#${heading} — ${trimmed}`;
231
+ }
232
+ return trimmed;
233
+ }
234
+ /**
235
+ * Extract distinctive terms from the top search results for query expansion.
236
+ * Pseudo-relevance feedback: terms from top-matching pages that aren't in
237
+ * the original query become expansion candidates.
238
+ */
239
+ function extractExpansionTerms(scored, originalQuery, paths, maxTerms = 6) {
240
+ const topResults = scored.slice(0, Math.min(3, scored.length));
241
+ if (topResults.length === 0)
242
+ return [];
243
+ const originalNorm = normalizeText(originalQuery);
244
+ const termFreq = new Map();
245
+ for (const { id, pagePath, entry } of topResults) {
246
+ // Collect text from registry metadata + file content
247
+ const metaText = normalizeText([entry.title, entry.aliases, entry.tags, entry.summary, entry.description]
248
+ .filter(Boolean)
249
+ .join(" "));
250
+ for (const w of metaText.split(/\s+/)) {
251
+ if (w.length >= 4 && !originalNorm.includes(w) && !STOPWORDS.has(w)) {
252
+ termFreq.set(w, (termFreq.get(w) || 0) + 1);
253
+ }
254
+ }
255
+ // Also extract from file body
256
+ if (existsSync(pagePath)) {
257
+ const parsed = parsePage(pagePath, id);
258
+ if (parsed) {
259
+ const bodyNorm = normalizeText(parsed.body);
260
+ for (const w of bodyNorm.split(/\s+/)) {
261
+ if (w.length >= 4 && !originalNorm.includes(w) && !STOPWORDS.has(w)) {
262
+ termFreq.set(w, (termFreq.get(w) || 0) + 1);
263
+ }
264
+ }
265
+ }
266
+ }
267
+ }
268
+ // Sort by frequency descending, take top N
269
+ return Array.from(termFreq.entries())
270
+ .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))
271
+ .slice(0, maxTerms)
272
+ .map(([term]) => term);
273
+ }
274
+ /**
275
+ * Search a single vault's registry for pages matching a query.
276
+ * Returns up to `maxResults` matches, each with a content preview.
277
+ * Results below `minScore` are excluded (default 0 = no filtering).
278
+ */
279
+ export function searchWiki(paths, query, maxResults = 5, minScore = 0, semantic) {
280
+ const registry = readJson(join(paths.meta, "registry.json"), {
281
+ version: "1.0",
282
+ last_updated: "",
283
+ pages: {},
284
+ });
285
+ const terms = queryTerms(query);
286
+ if (terms.length === 0)
287
+ return [];
288
+ // Read this vault's precomputed embedding sidecar (synchronous, offline).
289
+ // Missing/empty sidecar => no semantic signal => pure lexical, by construction.
290
+ const embeddingStore = semantic
291
+ ? readEmbeddingStore(paths)
292
+ : undefined;
293
+ const scored = [];
294
+ for (const [id, entry] of Object.entries(registry.pages)) {
295
+ const pagePath = join(paths.wiki, `${id}.md`);
296
+ const pageExists = existsSync(pagePath);
297
+ const parsed = parsePage(pagePath, id);
298
+ if (pageExists && !parsed)
299
+ continue;
300
+ const frontmatter = parsed?.frontmatter ?? {};
301
+ const body = parsed?.body ?? "";
302
+ let score = 0;
303
+ // Strong identifiers: exact command/short-query aliases should win.
304
+ score += scoreField(id, terms, 3);
305
+ score += scoreField(entry.title, terms, 5);
306
+ score += scoreField(frontmatter.title, terms, 5);
307
+ score += scoreField(entry.type, terms, 1);
308
+ // Recall-oriented metadata.
309
+ score += scoreField(entry.aliases, terms, 6);
310
+ score += scoreField(frontmatter.aliases, terms, 6);
311
+ score += scoreField(entry.recall_triggers, terms, 7);
312
+ score += scoreField(frontmatter.recall_triggers, terms, 7);
313
+ score += scoreField(entry.summary, terms, 3);
314
+ score += scoreField(frontmatter.summary, terms, 3);
315
+ score += scoreField(entry.description, terms, 3);
316
+ score += scoreField(frontmatter.description, terms, 3);
317
+ // General metadata from the registry/frontmatter.
318
+ score += scoreField(entry.tags, terms, 2);
319
+ score += scoreField(entry.category, terms, 2);
320
+ score += scoreField(entry.domain, terms, 2);
321
+ score += scoreField(frontmatter.tags, terms, 2);
322
+ score += scoreField(frontmatter.category, terms, 2);
323
+ score += scoreField(frontmatter.domain, terms, 2);
324
+ // Body search: use chunk-level indexing for more precise matching.
325
+ // Each section of the page is scored independently, so a query about
326
+ // "Postgres" matches only the Postgres section, not the whole page.
327
+ let bestChunkScore = 0;
328
+ let bestChunkHeading = "";
329
+ let bestChunkContent = "";
330
+ if (body.trim()) {
331
+ const chunks = chunkPage(body);
332
+ for (const chunk of chunks) {
333
+ let chunkScore = 0;
334
+ // Heading gets a strong boost
335
+ chunkScore += scoreField(chunk.heading, terms, 4);
336
+ // Chunk body content
337
+ chunkScore += scoreField(chunk.content, terms, 1);
338
+ if (chunkScore > bestChunkScore) {
339
+ bestChunkScore = chunkScore;
340
+ bestChunkHeading = chunk.heading;
341
+ bestChunkContent = chunk.content;
342
+ }
343
+ }
344
+ }
345
+ // Add best chunk score to total page score
346
+ score += bestChunkScore;
347
+ // Semantic candidacy: a page with no lexical match can still qualify if its
348
+ // precomputed vector is sufficiently close to the query vector. The boost
349
+ // itself is applied AFTER pseudo-relevance feedback so PRF stays lexical.
350
+ let semCos = 0;
351
+ if (semantic && embeddingStore) {
352
+ const vec = embeddingStore.entries[id]?.vector;
353
+ if (vec && vec.length === semantic.queryVector.length) {
354
+ semCos = cosineSimilarity(semantic.queryVector, vec);
355
+ }
356
+ }
357
+ const semEligible = semCos >= SEMANTIC_MIN_COSINE;
358
+ if (score > 0 || semEligible) {
359
+ scored.push({
360
+ id,
361
+ entry,
362
+ score,
363
+ pagePath,
364
+ bestChunkPreview: bestChunkContent ? chunkPreview(bestChunkHeading, bestChunkContent) : "",
365
+ semCos,
366
+ });
367
+ }
368
+ }
369
+ scored.sort((a, b) => b.score - a.score || a.id.localeCompare(b.id));
370
+ // ── Pseudo-Relevance Feedback (PRF) ─────────────────
371
+ // Extract distinctive terms from the top 3 results and use them to
372
+ // boost semantically related pages. This gives "semantic" expansion
373
+ // without external dependencies: if an "Authentication" page mentions
374
+ // JWT, OAuth, and sessions, those terms boost other pages that discuss
375
+ // related concepts.
376
+ const expansionTerms = extractExpansionTerms(scored, query, paths, 6);
377
+ if (expansionTerms.length > 0) {
378
+ const expTermList = queryTerms(expansionTerms.join(" "));
379
+ // Apply expansion scoring to the top 25 results (cheap re-read)
380
+ const expansionCandidates = scored.slice(0, Math.min(25, scored.length));
381
+ for (const item of expansionCandidates) {
382
+ const parsed = parsePage(item.pagePath, item.id);
383
+ const body = parsed?.body ?? "";
384
+ let expChunkScore = 0;
385
+ if (body.trim()) {
386
+ const chunks = chunkPage(body);
387
+ for (const chunk of chunks) {
388
+ let cs = 0;
389
+ cs += scoreField(chunk.heading, expTermList, 2); // half weight
390
+ cs += scoreField(chunk.content, expTermList, 0.5);
391
+ if (cs > expChunkScore)
392
+ expChunkScore = cs;
393
+ }
394
+ }
395
+ // Dampened addition — expansion contributes at most 40%
396
+ item.score += expChunkScore * 0.4;
397
+ }
398
+ }
399
+ // ── Semantic fusion ─────────────────────────────────
400
+ // Blend the precomputed cosine similarity into the (lexical + PRF) score.
401
+ // Applied last so PRF expansion remains purely lexical and so a strongly
402
+ // paraphrase-relevant page that lexical missed can clear `minScore`. With no
403
+ // semantic context every boost is 0, leaving the lexical ranking untouched.
404
+ if (semantic) {
405
+ for (const item of scored) {
406
+ item.score = fuseScores(item.score, item.semCos, semantic.weight);
407
+ }
408
+ }
409
+ // Re-sort after expansion + semantic scoring
410
+ scored.sort((a, b) => b.score - a.score || a.id.localeCompare(b.id));
411
+ const top = scored.filter((s) => s.score >= minScore).slice(0, maxResults);
412
+ return top.map(({ id, entry, pagePath, score, bestChunkPreview }) => {
413
+ let preview = bestChunkPreview;
414
+ if (!preview && existsSync(pagePath)) {
415
+ // Fallback: no chunk matched, show page intro
416
+ preview = pagePreview(readFileSync(pagePath, "utf-8"));
417
+ }
418
+ return {
419
+ id,
420
+ title: String(entry.title || id),
421
+ type: String(entry.type || "page"),
422
+ preview,
423
+ path: pagePath,
424
+ score,
425
+ };
426
+ });
427
+ }
428
+ /**
429
+ * Search both project/primary vault and personal vault, merging results.
430
+ * Personal results are appended after primary results, deduplicated by page ID.
431
+ *
432
+ * @param minScore - Minimum relevance score (default 0 = no filter).
433
+ * @param includePersonal - Whether to search the personal vault (default true).
434
+ * Auto-injection should pass false to avoid personal-vault contamination.
435
+ */
436
+ export function searchWikiLayered(primaryPaths, query, maxResults = 5, minScore = 0, includePersonal = true, semantic) {
437
+ // Search primary vault
438
+ const primaryResults = searchWiki(primaryPaths, query, maxResults, minScore, semantic);
439
+ // If primary is already the personal vault, no layered search needed
440
+ if (isPersonalVault(primaryPaths))
441
+ return primaryResults;
442
+ // Search personal vault as secondary layer (only when explicitly requested)
443
+ let personalResults = [];
444
+ if (includePersonal) {
445
+ const personalPaths = getPersonalWikiPaths();
446
+ if (existsSync(join(personalPaths.dotWiki, "config.json"))) {
447
+ personalResults = searchWiki(personalPaths, query, maxResults, minScore, semantic);
448
+ }
449
+ }
450
+ // Merge: personal results first (they're the user's accumulated knowledge),
451
+ // then primary results (project-specific). Deduplicate by page ID.
452
+ const seen = new Set();
453
+ const merged = [];
454
+ for (const r of [...personalResults, ...primaryResults]) {
455
+ if (seen.has(r.id))
456
+ continue;
457
+ seen.add(r.id);
458
+ // If it's from personal vault, tag it
459
+ if (personalResults.includes(r)) {
460
+ merged.push({ ...r, vaultLabel: "📓 personal" });
461
+ }
462
+ else {
463
+ merged.push(r);
464
+ }
465
+ }
466
+ return merged.slice(0, maxResults);
467
+ }
468
+ // ─── Async hybrid entry point (the single, cached query embedding) ───
469
+ /**
470
+ * Cache of query string → normalized embedding vector. The query embedding is
471
+ * the ONLY embedding call in the recall hot path; caching collapses repeated
472
+ * recalls of the same query within a session (e.g. auto-injection + an explicit
473
+ * wiki_recall) into a single network call, satisfying the #67 "single cached
474
+ * query-embedding lookup" bound.
475
+ */
476
+ const queryEmbeddingCache = new Map();
477
+ const QUERY_CACHE_MAX = 256;
478
+ function queryCacheKey(model, query) {
479
+ return `${model}\u0000${normalizeText(query)}`;
480
+ }
481
+ /** Test-only: reset the module-level query-embedding cache. */
482
+ export function __clearQueryEmbeddingCache() {
483
+ queryEmbeddingCache.clear();
484
+ }
485
+ /** True if a vault has at least one stored embedding vector. */
486
+ function storeHasEntries(paths) {
487
+ return Object.keys(readEmbeddingStore(paths).entries).length > 0;
488
+ }
489
+ /**
490
+ * Embed the query string once (cached), returning a normalized vector, or
491
+ * `undefined` when no embedder is configured or the call yields nothing.
492
+ */
493
+ async function embedQuery(embedder, query) {
494
+ const key = queryCacheKey(embedder.model, query);
495
+ const cached = queryEmbeddingCache.get(key);
496
+ if (cached)
497
+ return cached;
498
+ const [raw] = await embedder.embed([query]);
499
+ if (!raw || raw.length === 0)
500
+ return undefined;
501
+ const vec = normalizeVector(raw);
502
+ if (queryEmbeddingCache.size >= QUERY_CACHE_MAX) {
503
+ const oldest = queryEmbeddingCache.keys().next().value;
504
+ if (oldest !== undefined)
505
+ queryEmbeddingCache.delete(oldest);
506
+ }
507
+ queryEmbeddingCache.set(key, vec);
508
+ return vec;
509
+ }
510
+ /**
511
+ * Hybrid layered recall: lexical scoring blended with semantic cosine ranking.
512
+ *
513
+ * Design (issue #67): page vectors are precomputed at write time (#66); the
514
+ * ONLY per-query embedding work is a single, cached lookup of the (short) query
515
+ * string. If no vault has embeddings, the query embedding is skipped entirely
516
+ * and this degrades to exactly `searchWikiLayered` (pure lexical, zero network).
517
+ * Likewise when no embedder is configured. `opts.embedder` is an injection seam
518
+ * for tests (mirrors `embedPages`) so unit tests never touch the network.
519
+ */
520
+ export async function searchWikiHybrid(primaryPaths, query, maxResults = 5, minScore = 0, includePersonal = true, opts = {}) {
521
+ // Pure-lexical fast path: no semantic signal anywhere => no embedding call.
522
+ let anyEmbeddings = storeHasEntries(primaryPaths);
523
+ if (!anyEmbeddings && includePersonal && !isPersonalVault(primaryPaths)) {
524
+ const personalPaths = getPersonalWikiPaths();
525
+ if (existsSync(join(personalPaths.dotWiki, "config.json"))) {
526
+ anyEmbeddings = storeHasEntries(personalPaths);
527
+ }
528
+ }
529
+ if (!anyEmbeddings) {
530
+ return searchWikiLayered(primaryPaths, query, maxResults, minScore, includePersonal);
531
+ }
532
+ const embedder = opts.embedder ?? (opts.config ? resolveEmbedder(opts.config) : undefined);
533
+ if (!embedder) {
534
+ // Embeddings exist but no embedder configured to embed the query: fall back
535
+ // to pure lexical rather than guess. (Degrades gracefully.)
536
+ return searchWikiLayered(primaryPaths, query, maxResults, minScore, includePersonal);
537
+ }
538
+ let semantic;
539
+ try {
540
+ const queryVector = await embedQuery(embedder, query);
541
+ if (queryVector) {
542
+ const weight = opts.config?.semanticWeight ?? DEFAULT_SEMANTIC_WEIGHT;
543
+ semantic = { queryVector, weight };
544
+ }
545
+ }
546
+ catch {
547
+ // Network/embedding failure must never break recall — fall back to lexical.
548
+ semantic = undefined;
549
+ }
550
+ return searchWikiLayered(primaryPaths, query, maxResults, minScore, includePersonal, semantic);
551
+ }
552
+ /**
553
+ * Default page-count gate for two-stage (links-first) recall (issue #68).
554
+ * When a vault's registered page count exceeds this, recall returns ranked
555
+ * links (expand on demand via `read`) instead of inline content previews.
556
+ */
557
+ export const DEFAULT_RECALL_LINKS_THRESHOLD = 50;
558
+ /** Max characters of the 1-line snippet shown beside a link in links-first mode. */
559
+ const LINKS_SNIPPET_MAX = 80;
560
+ /** Count the registered pages of a single vault (O(1), no page-body I/O). */
561
+ function registryPageCount(paths) {
562
+ const registry = readJson(join(paths.meta, "registry.json"), {
563
+ version: "1.0",
564
+ last_updated: "",
565
+ pages: {},
566
+ });
567
+ return Object.keys(registry.pages).length;
568
+ }
569
+ /**
570
+ * Total registered page count across the vault(s) recall will actually search.
571
+ * Mirrors `searchWikiLayered`'s vault selection so the two-stage gate is keyed
572
+ * to the same corpus the agent sees. Reads only `registry.json` — never a page
573
+ * body — so the gate stays cheap as the vault grows.
574
+ */
575
+ export function vaultPageCount(primaryPaths, includePersonal = true) {
576
+ let count = registryPageCount(primaryPaths);
577
+ if (includePersonal && !isPersonalVault(primaryPaths)) {
578
+ const personalPaths = getPersonalWikiPaths();
579
+ if (existsSync(join(personalPaths.dotWiki, "config.json"))) {
580
+ count += registryPageCount(personalPaths);
581
+ }
582
+ }
583
+ return count;
584
+ }
585
+ /**
586
+ * Decide whether recall should use links-first (stage 1) rendering: true when
587
+ * the vault page count is STRICTLY GREATER THAN the configured threshold.
588
+ * Threshold 0 forces links-first for any non-empty vault; a very large value
589
+ * keeps previews inline always. Default `DEFAULT_RECALL_LINKS_THRESHOLD`.
590
+ */
591
+ export function shouldUseLinksFirst(pageCount, config) {
592
+ const threshold = config?.recallLinksThreshold ?? DEFAULT_RECALL_LINKS_THRESHOLD;
593
+ return pageCount > threshold;
594
+ }
595
+ /** One-line snippet for links-first rendering, derived from the chunk preview. */
596
+ function linkSnippet(preview) {
597
+ const oneLine = preview.replace(/\s+/g, " ").trim();
598
+ if (!oneLine)
599
+ return "";
600
+ return oneLine.length > LINKS_SNIPPET_MAX ? `${oneLine.slice(0, LINKS_SNIPPET_MAX)}…` : oneLine;
601
+ }
602
+ /**
603
+ * Default cap on chars of a skill/case body inlined directly into a recall
604
+ * block. Overridable per-vault via `recallSkillInlineMax` (0 disables inlining).
605
+ * Mirrors `DEFAULT_RECALL_LINKS_THRESHOLD` — the sibling context-window lever.
606
+ */
607
+ export const DEFAULT_RECALL_SKILL_INLINE_MAX = 1600;
608
+ /**
609
+ * Skills/working-memory carve-out from links-first: short, high-value
610
+ * procedural pages (`skill`/`case`) are meant to be APPLIED immediately, so we
611
+ * inline their body directly rather than make the agent expand a link it often
612
+ * skips (adherence > context-economy for these page types). Returns null for
613
+ * non-skill pages or when the body can't be read.
614
+ */
615
+ function isSkillOrCase(r) {
616
+ return (r.type === "skill" ||
617
+ r.type === "case" ||
618
+ r.id.startsWith("skills/") ||
619
+ r.id.startsWith("cases/"));
620
+ }
621
+ /**
622
+ * Inlined body for a skill/case page, or null. `max <= 0` disables inlining and
623
+ * short-circuits BEFORE any filesystem access, so a vault that opts out keeps
624
+ * recall page-body-I/O-free (issue #68's cheap-recall invariant). Otherwise the
625
+ * read is bounded: it fires only for skill/case results (which exist only when
626
+ * the trajectories feature is on) and only the top-N ranked hits.
627
+ */
628
+ function inlineSkillBody(r, max = DEFAULT_RECALL_SKILL_INLINE_MAX) {
629
+ if (max <= 0)
630
+ return null;
631
+ if (!isSkillOrCase(r))
632
+ return null;
633
+ if (!r.path || !existsSync(r.path))
634
+ return null;
635
+ // Normalize CRLF first so the LF-anchored frontmatter strip below works on
636
+ // Windows-authored / git-autocrlf'd vaults (otherwise the raw YAML leaks in).
637
+ let body = readFileSync(r.path, "utf-8").replace(/\r\n/g, "\n");
638
+ body = body.replace(/^---\n[\s\S]*?\n---\n/, "").trim(); // strip YAML frontmatter
639
+ if (!body)
640
+ return null;
641
+ if (body.length > max) {
642
+ body = `${body.slice(0, max)}\n…(truncated — \`read\` the path above for the full page)`;
643
+ }
644
+ return body;
645
+ }
646
+ /**
647
+ * A backtick fence guaranteed longer than any backtick run inside `body`.
648
+ * Skill/case pages routinely embed their own fenced code blocks; CommonMark
649
+ * closes a fenced block only on a fence of length >= the opener, so opening
650
+ * with (longest inner run + 1, min 3) keeps an inlined body from terminating
651
+ * the wrapper early — and stays safe even when truncation cuts mid-fence.
652
+ */
653
+ function codeFenceFor(body) {
654
+ let longest = 0;
655
+ for (const run of body.match(/`+/g) ?? [])
656
+ longest = Math.max(longest, run.length);
657
+ return "`".repeat(Math.max(3, longest + 1));
658
+ }
659
+ /** Indented, fence-safe lines wrapping an inlined skill/case body. */
660
+ function inlineBlockLines(body, indent) {
661
+ const fence = codeFenceFor(body);
662
+ return [
663
+ "",
664
+ `${indent}${fence}`,
665
+ ...body.split("\n").map((line) => `${indent}${line}`),
666
+ `${indent}${fence}`,
667
+ ];
668
+ }
669
+ /**
670
+ * Format recall results as a compact system-prompt section.
671
+ *
672
+ * Two render modes (issue #68):
673
+ * - Default / `linksOnly: false` — preview-inline. For ordinary pages this is
674
+ * byte-for-byte the pre-fix small-vault rendering (no regression); the
675
+ * resolvable read-path + new footer copy are confined to links-first, where
676
+ * there is no inline content and the agent MUST resolve a link.
677
+ * - `linksOnly: true` — stage-1 "links-first": a ranked list of links carrying
678
+ * id, title, type, score, and a single short snippet, each with a resolvable
679
+ * `read <path>`. The agent expands the links it wants on demand (stage 2).
680
+ * Used above the vault-size threshold to keep large vaults from flooding context.
681
+ *
682
+ * `skillInlineMax` (default `DEFAULT_RECALL_SKILL_INLINE_MAX`) caps how much of a
683
+ * skill/case body is inlined; 0 disables inlining (pure links-first for those too).
684
+ */
685
+ export function formatRecallContext(results, opts = {}) {
686
+ if (results.length === 0)
687
+ return "";
688
+ const skillInlineMax = opts.skillInlineMax ?? DEFAULT_RECALL_SKILL_INLINE_MAX;
689
+ const hasLayered = results.some((r) => r.vaultLabel);
690
+ const label = hasLayered ? " (personal + project)" : "";
691
+ // Salience nudge: when a distilled skill/case matches, tell the agent to
692
+ // apply it BEFORE experimenting (the dominant cost is recall non-adherence).
693
+ const hasSkill = results.some(isSkillOrCase);
694
+ const skillNudge = "⚠\ufe0f A distilled skill/case below matches this task — read and APPLY it BEFORE experimenting on your own.";
695
+ if (opts.linksOnly) {
696
+ const lines = [
697
+ "## Relevant Wiki Knowledge (links-first)",
698
+ "",
699
+ `_${results.length} page(s) matched your query${label}, ranked. Two-stage recall: links only — open the ones you need to read their full content._`,
700
+ "",
701
+ ];
702
+ if (hasSkill)
703
+ lines.splice(1, 0, "", skillNudge);
704
+ results.forEach((r, i) => {
705
+ const vaultTag = r.vaultLabel ? ` ${r.vaultLabel}` : "";
706
+ const snippet = linkSnippet(r.preview);
707
+ const tail = snippet ? ` — ${snippet}` : "";
708
+ lines.push(`${i + 1}. **[[${r.id}]]** — *${r.type}* — score ${r.score.toFixed(1)}${vaultTag} — ${r.title}${tail}`);
709
+ // Surface a read-resolvable path so expansion is a single, first-try
710
+ // `read` (issue: wikilink ids aren't resolvable by the file read tool).
711
+ if (r.path)
712
+ lines.push(` ↳ \`read ${r.path}\``);
713
+ // Skills/case carve-out: inline the body so the agent doesn't have to
714
+ // (and often won't) expand the link before acting.
715
+ const inl = inlineSkillBody(r, skillInlineMax);
716
+ if (inl)
717
+ lines.push(...inlineBlockLines(inl, " "));
718
+ });
719
+ lines.push("", "Call `read` on the exact path shown under each link to pull its full content." +
720
+ " Add new findings via wiki_ensure_page or wiki_retro.", "");
721
+ return lines.join("\n");
722
+ }
723
+ const lines = [
724
+ "## Relevant Wiki Knowledge",
725
+ "",
726
+ `_${results.length} page(s) matched your query${label}._`,
727
+ "",
728
+ ];
729
+ if (hasSkill)
730
+ lines.splice(1, 0, "", skillNudge);
731
+ for (const r of results) {
732
+ const vaultTag = r.vaultLabel ? ` ${r.vaultLabel}` : "";
733
+ lines.push(`- **[[${r.id}]]** — *${r.type}* — ${r.title}${vaultTag}`);
734
+ // Skills/case carve-out: inline the body (adherence > context-economy). Only
735
+ // here does the default path deviate from the pre-fix small-vault render —
736
+ // and only when a skill/case matched (i.e. the trajectories feature is on),
737
+ // so ordinary pages stay byte-for-byte unchanged (#68 no-regression promise).
738
+ const inl = inlineSkillBody(r, skillInlineMax);
739
+ if (inl) {
740
+ // Resolvable path so a truncated inline body is one `read` away.
741
+ if (r.path)
742
+ lines.push(` ↳ \`read ${r.path}\``);
743
+ lines.push(...inlineBlockLines(inl, " "));
744
+ }
745
+ else if (r.preview) {
746
+ // Truncate preview to one line
747
+ const preview = r.preview.length > 120 ? `${r.preview.slice(0, 120)}…` : r.preview;
748
+ lines.push(` ${preview}`);
749
+ }
750
+ lines.push("");
751
+ }
752
+ lines.push("Use `read` to view full pages. Add new findings via wiki_ensure_page or wiki_retro.", "");
753
+ return lines.join("\n");
754
+ }
755
+ // ─── Tool Registration ──────────────────────────────────
756
+ /**
757
+ * Register the `wiki_recall` tool.
758
+ * The model can call this explicitly to search the wiki.
759
+ * It is also called automatically via before_agent_start hook.
760
+ */
761
+ export function registerWikiRecall(pi, runtime) {
762
+ pi.registerTool({
763
+ name: "wiki_recall",
764
+ label: "Wiki Recall",
765
+ description: "Search the wiki for pages relevant to a query. " +
766
+ "Returns matching page IDs, titles, types, and content previews (small vaults) " +
767
+ "or a ranked list of links to expand with `read` (large vaults, two-stage recall). " +
768
+ "Called automatically at session start — use explicitly to dig deeper.",
769
+ promptSnippet: "Recall wiki knowledge relevant to the current task",
770
+ promptGuidelines: [
771
+ "Use wiki_recall at the START of every task to find relevant wiki knowledge.",
772
+ "The extension auto-calls wiki_recall — but calling it explicitly with specific terms gets better results.",
773
+ ],
774
+ parameters: Type.Object({
775
+ query: Type.String({
776
+ description: "Search query — use the user's full request or key terms",
777
+ }),
778
+ max_results: Type.Optional(Type.Number({ description: "Max results (default: 5, max: 10)", default: 5 })),
779
+ }),
780
+ async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
781
+ const paths = resolveVaultPaths(ctx.cwd ?? process.cwd());
782
+ if (!existsSync(join(paths.dotWiki, "config.json"))) {
783
+ return {
784
+ content: [
785
+ {
786
+ type: "text",
787
+ text: "No wiki vault found at this location. Initialize one with wiki_bootstrap first.",
788
+ },
789
+ ],
790
+ details: { error: "no_vault" },
791
+ isError: true,
792
+ };
793
+ }
794
+ const maxResults = Math.min(params.max_results ?? 5, 10);
795
+ const vaultDiagnostics = inspectVaultFormat(paths).diagnostics;
796
+ const diagnosticText = vaultDiagnostics.length
797
+ ? `\n\nDiagnostics: ${vaultDiagnostics.map((diagnostic) => diagnostic.code).join(", ")}`
798
+ : "";
799
+ // Use layered hybrid search: personal vault + project vault, blending
800
+ // lexical scoring with precomputed semantic embeddings when available.
801
+ // No embeddings / no embedder => pure lexical, no network call.
802
+ if (runtime)
803
+ runtime.ensureConfig(ctx.cwd ?? paths.root);
804
+ const results = await searchWikiHybrid(paths, params.query, maxResults, 0, true, {
805
+ config: runtime?.config,
806
+ });
807
+ if (results.length === 0) {
808
+ return {
809
+ content: [
810
+ {
811
+ type: "text",
812
+ text: `No wiki pages found matching "${params.query}". The wiki is empty — use wiki_retro to start building knowledge.${diagnosticText}`,
813
+ },
814
+ ],
815
+ details: {
816
+ query: params.query,
817
+ matches: [],
818
+ diagnostics: vaultDiagnostics,
819
+ },
820
+ };
821
+ }
822
+ const hasPersonal = results.some((r) => r.vaultLabel);
823
+ const layerTag = hasPersonal ? " (personal + project)" : "";
824
+ // Two-stage gate (issue #68): large vaults return ranked LINKS only;
825
+ // the agent expands chosen links on demand via `read`. Small vaults keep
826
+ // the inline-preview behavior. Page count is read from the registry only.
827
+ const linksFirst = shouldUseLinksFirst(vaultPageCount(paths, true), runtime?.config);
828
+ if (linksFirst) {
829
+ const linkLines = results
830
+ .map((r, i) => {
831
+ const vault = r.vaultLabel ? ` ${r.vaultLabel}` : "";
832
+ const snippet = linkSnippet(r.preview);
833
+ const tail = snippet ? ` — ${snippet}` : "";
834
+ return `${i + 1}. [[${r.id}]] — ${r.title} (${r.type}, score ${r.score.toFixed(1)})${vault}\n Path: ${r.path}${tail}`;
835
+ })
836
+ .join("\n");
837
+ const text = [
838
+ `Found ${results.length} wiki page(s) matching "${params.query}"${layerTag} (two-stage recall — ranked links, expand on demand):`,
839
+ "",
840
+ linkLines,
841
+ "",
842
+ "Call `read` on the path(s) you need to pull full content.",
843
+ ].join("\n") + diagnosticText;
844
+ return {
845
+ content: [{ type: "text", text }],
846
+ details: {
847
+ query: params.query,
848
+ mode: "links",
849
+ matches: results,
850
+ diagnostics: vaultDiagnostics,
851
+ },
852
+ };
853
+ }
854
+ return {
855
+ content: [
856
+ {
857
+ type: "text",
858
+ text: `Found ${results.length} wiki page(s) matching "${params.query}"${layerTag}:\n\n${results
859
+ .map((r) => {
860
+ const vault = r.vaultLabel ? ` ${r.vaultLabel}` : "";
861
+ return `## [[${r.id}]] — ${r.title}${vault}\nType: ${r.type}\nPath: ${r.path}\n\n${r.preview}`;
862
+ })
863
+ .join("\n\n---\n\n")}${diagnosticText}`,
864
+ },
865
+ ],
866
+ details: {
867
+ query: params.query,
868
+ mode: "preview",
869
+ matches: results,
870
+ diagnostics: vaultDiagnostics,
871
+ },
872
+ };
873
+ },
874
+ });
875
+ }