@zosmaai/pi-llm-wiki 0.10.7 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/CHANGELOG.md +4 -0
  2. package/README.de.md +35 -4
  3. package/README.es.md +260 -170
  4. package/README.fr.md +35 -4
  5. package/README.hi.md +35 -4
  6. package/README.ja.md +35 -4
  7. package/README.ko.md +35 -4
  8. package/README.md +38 -3
  9. package/README.pt.md +35 -4
  10. package/README.ru.md +35 -4
  11. package/README.zh.md +260 -170
  12. package/assets/demo.gif +0 -0
  13. package/dist/extensions/llm-wiki/lib/bootstrap.js +71 -0
  14. package/dist/extensions/llm-wiki/lib/embeddings.js +401 -0
  15. package/dist/extensions/llm-wiki/lib/guardrails.js +232 -0
  16. package/dist/extensions/llm-wiki/lib/indexing.js +78 -0
  17. package/dist/extensions/llm-wiki/lib/ingest-worker.js +310 -0
  18. package/dist/extensions/llm-wiki/lib/inject.js +65 -0
  19. package/dist/extensions/llm-wiki/lib/knowledge-document.js +442 -0
  20. package/dist/extensions/llm-wiki/lib/knowledge-links.js +206 -0
  21. package/dist/extensions/llm-wiki/lib/legacy-repair.js +443 -0
  22. package/dist/extensions/llm-wiki/lib/metadata.js +499 -0
  23. package/dist/extensions/llm-wiki/lib/model-command.js +86 -0
  24. package/dist/extensions/llm-wiki/lib/observation.js +283 -0
  25. package/dist/extensions/llm-wiki/lib/recall.js +875 -0
  26. package/dist/extensions/llm-wiki/lib/retro.js +158 -0
  27. package/dist/extensions/llm-wiki/lib/runtime.js +191 -0
  28. package/dist/extensions/llm-wiki/lib/source-extractors.js +426 -0
  29. package/dist/extensions/llm-wiki/lib/source-packet.js +229 -0
  30. package/dist/extensions/llm-wiki/lib/subagent.js +41 -0
  31. package/dist/extensions/llm-wiki/lib/task-config.js +172 -0
  32. package/dist/extensions/llm-wiki/lib/tools.js +1192 -0
  33. package/dist/extensions/llm-wiki/lib/trajectories-command.js +51 -0
  34. package/dist/extensions/llm-wiki/lib/trajectory.js +467 -0
  35. package/dist/extensions/llm-wiki/lib/utils.js +347 -0
  36. package/dist/extensions/llm-wiki/lib/vault-format.js +247 -0
  37. package/dist/extensions/llm-wiki/lib/visible-status.js +31 -0
  38. package/dist/extensions/llm-wiki/lib/wiki-service.js +128 -0
  39. package/dist/mcp/exec.js +121 -0
  40. package/dist/mcp/index.js +229 -0
  41. package/dist/mcp/operations.js +130 -0
  42. package/dist/package.json +1 -0
  43. package/docs/superpowers/plans/2026-08-02-okf-foundation.md +1579 -0
  44. package/docs/superpowers/plans/2026-08-03-okf-foundation-remediation.md +3005 -0
  45. package/docs/superpowers/plans/2026-08-06-okf-foundation-release-remediation.md +1174 -0
  46. package/docs/superpowers/specs/2026-08-02-okf-foundation-design.md +578 -0
  47. package/docs/superpowers/specs/2026-08-02-okf-v0.2-interoperability-design.md +538 -0
  48. package/extensions/llm-wiki/index.ts +22 -36
  49. package/extensions/llm-wiki/lib/bootstrap.ts +84 -0
  50. package/extensions/llm-wiki/lib/embeddings.ts +9 -3
  51. package/extensions/llm-wiki/lib/guardrails.ts +174 -29
  52. package/extensions/llm-wiki/lib/indexing.ts +2 -1
  53. package/extensions/llm-wiki/lib/ingest-worker.ts +170 -29
  54. package/extensions/llm-wiki/lib/knowledge-document.ts +661 -0
  55. package/extensions/llm-wiki/lib/knowledge-links.ts +282 -0
  56. package/extensions/llm-wiki/lib/legacy-repair.ts +572 -0
  57. package/extensions/llm-wiki/lib/metadata.ts +531 -116
  58. package/extensions/llm-wiki/lib/observation.ts +37 -43
  59. package/extensions/llm-wiki/lib/recall.ts +61 -33
  60. package/extensions/llm-wiki/lib/retro.ts +65 -41
  61. package/extensions/llm-wiki/lib/source-extractors.ts +12 -17
  62. package/extensions/llm-wiki/lib/source-packet.ts +44 -31
  63. package/extensions/llm-wiki/lib/tools.ts +406 -348
  64. package/extensions/llm-wiki/lib/trajectory.ts +15 -1
  65. package/extensions/llm-wiki/lib/utils.ts +121 -130
  66. package/extensions/llm-wiki/lib/vault-format.ts +363 -0
  67. package/extensions/llm-wiki/lib/wiki-service.ts +183 -0
  68. package/mcp/exec.ts +122 -0
  69. package/mcp/index.ts +60 -250
  70. package/mcp/operations.ts +176 -0
  71. package/package.json +8 -2
  72. package/scripts/migrate-llm-wiki.js +801 -0
  73. package/skills/llm-wiki/SKILL.md +8 -6
@@ -0,0 +1,426 @@
1
+ import { open } from "node:fs/promises";
2
+ import { NodeHtmlMarkdown } from "node-html-markdown";
3
+ import { exec } from "./utils.js";
4
+ // ---------------------------------------------------------------------------
5
+ // Binary magic byte detection
6
+ // ---------------------------------------------------------------------------
7
+ const BINARY_SIGNATURES = [
8
+ // Archives & documents
9
+ { bytes: [0x50, 0x4b, 0x03, 0x04], format: "zip" }, // ZIP / DOCX / XLSX / PPTX / JAR
10
+ { bytes: [0x25, 0x50, 0x44, 0x46], format: "pdf" }, // %PDF
11
+ { bytes: [0x37, 0x7a, 0xbc, 0xaf], format: "7z" }, // 7-Zip
12
+ { bytes: [0x1f, 0x8b], format: "gzip" }, // gzip / .tar.gz
13
+ // Images
14
+ { bytes: [0x89, 0x50, 0x4e, 0x47], format: "png" }, // PNG
15
+ { bytes: [0xff, 0xd8, 0xff], format: "jpeg" }, // JPEG
16
+ { bytes: [0x47, 0x49, 0x46, 0x38], format: "gif" }, // GIF8
17
+ { bytes: [0x42, 0x4d], format: "bmp" }, // BMP
18
+ { bytes: [0x49, 0x49, 0x2a, 0x00], format: "tiff" }, // TIFF (little-endian)
19
+ { bytes: [0x4d, 0x4d, 0x00, 0x2a], format: "tiff" }, // TIFF (big-endian)
20
+ { bytes: [0x52, 0x49, 0x46, 0x46], format: "riff" }, // RIFF (WAV / AVI / WebP)
21
+ // Executables & binaries
22
+ { bytes: [0x4d, 0x5a], format: "exe" }, // Windows PE (EXE / DLL)
23
+ { bytes: [0xcf, 0xfa, 0xed, 0xfe], format: "macho" }, // Mach-O 64-bit LE
24
+ { bytes: [0xce, 0xfa, 0xed, 0xfe], format: "macho" }, // Mach-O 32-bit LE
25
+ { bytes: [0xfe, 0xed, 0xfa, 0xcf], format: "macho" }, // Mach-O 64-bit BE
26
+ { bytes: [0xfe, 0xed, 0xfa, 0xce], format: "macho" }, // Mach-O 32-bit BE
27
+ { bytes: [0xca, 0xfe, 0xba, 0xbe], format: "class" }, // Java .class / Mach-O FAT
28
+ { bytes: [0x7f, 0x45, 0x4c, 0x46], format: "elf" }, // ELF binary
29
+ { bytes: [0x00, 0x61, 0x73, 0x6d], format: "wasm" }, // WebAssembly
30
+ // Data & media
31
+ { bytes: [0x53, 0x51, 0x4c, 0x69], format: "sqlite" }, // SQLite
32
+ { bytes: [0x49, 0x44, 0x33], format: "mp3" }, // MP3 (ID3 tag)
33
+ ];
34
+ /**
35
+ * Reads the first 8 bytes of `filePath` and checks them against known binary
36
+ * magic byte signatures. Returns the detected format name or `null` for text.
37
+ */
38
+ export async function detectBinaryMagicBytes(filePath) {
39
+ let handle;
40
+ try {
41
+ handle = await open(filePath, "r");
42
+ const buf = Buffer.alloc(8);
43
+ const { bytesRead } = await handle.read(buf, 0, 8, 0);
44
+ const header = buf.subarray(0, bytesRead);
45
+ for (const { bytes, format } of BINARY_SIGNATURES) {
46
+ if (bytes.every((b, i) => header[i] === b))
47
+ return format;
48
+ }
49
+ return null;
50
+ }
51
+ catch {
52
+ return null; // Unreadable file — let the extractor deal with it
53
+ }
54
+ finally {
55
+ await handle?.close();
56
+ }
57
+ }
58
+ export function binaryExtractionFailureMessage(format) {
59
+ return `_Binary file could not be converted to markdown (detected format: ${format}).\nCapture a text-based version or a URL pointing to readable content instead._\n`;
60
+ }
61
+ // ---------------------------------------------------------------------------
62
+ const DEFAULT_MARKITDOWN_TIMEOUT_MS = 180_000;
63
+ const DEFAULT_CURL_TIMEOUT_SECONDS = 30;
64
+ const FILE_EXTRACTORS = [
65
+ {
66
+ format: "pdf",
67
+ shouldReadText: false,
68
+ extractorName: "markitdown",
69
+ content_type: "application/pdf",
70
+ matches: hasExtension(".pdf"),
71
+ extract: ({ pi, filePath, signal }) => extractPdf(pi, filePath, signal),
72
+ },
73
+ textFileExtractor("markdown", [".md"], "text/markdown"),
74
+ textFileExtractor("text", [".txt"], "text/plain"),
75
+ textFileExtractor("html", [".html", ".htm"], "text/html"),
76
+ {
77
+ format: "xml",
78
+ shouldReadText: true,
79
+ extractorName: "xmlToMarkdown",
80
+ content_type: "application/xml",
81
+ matches: hasExtension(".xml"),
82
+ extract: ({ content }) => xmlToMarkdown(content),
83
+ },
84
+ {
85
+ format: "json",
86
+ shouldReadText: true,
87
+ extractorName: "jsonToMarkdown",
88
+ content_type: "application/json",
89
+ matches: hasExtension(".json"),
90
+ extract: ({ content }) => jsonToMarkdown(content),
91
+ },
92
+ {
93
+ format: "docx",
94
+ shouldReadText: false,
95
+ extractorName: "markitdown",
96
+ content_type: "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
97
+ matches: hasExtension(".docx"),
98
+ extract: ({ pi, filePath, signal }) => extractDocx(pi, filePath, signal),
99
+ },
100
+ textFileExtractor("file", []),
101
+ ];
102
+ const URL_EXTRACTORS = [
103
+ {
104
+ matches: isPdfUrl,
105
+ extract: ({ pi, url, signal }) => extractPdfUrl(pi, url, signal),
106
+ },
107
+ {
108
+ matches: () => true,
109
+ extract: ({ pi, url, signal }) => extractTextUrl(pi, url, signal),
110
+ },
111
+ ];
112
+ export function fileExtractorFor(filePath) {
113
+ return (FILE_EXTRACTORS.find((extractor) => extractor.matches(filePath)) ?? FILE_EXTRACTORS.at(-1));
114
+ }
115
+ export function extractUrlContent(pi, url, signal) {
116
+ const extractor = URL_EXTRACTORS.find((candidate) => candidate.matches(url)) ?? URL_EXTRACTORS.at(-1);
117
+ return extractor.extract({ pi, url, signal });
118
+ }
119
+ export function pdfExtractionFailureMessage(source) {
120
+ return `_PDF content could not be converted to markdown from ${source}. Try increasing WIKI_MARKITDOWN_TIMEOUT_MS._\n`;
121
+ }
122
+ function textFileExtractor(format, extensions, contentType) {
123
+ return {
124
+ format,
125
+ shouldReadText: true,
126
+ extractorName: "passthrough",
127
+ content_type: contentType,
128
+ matches: extensions.length ? hasAnyExtension(extensions) : () => true,
129
+ extract: ({ content }) => content,
130
+ };
131
+ }
132
+ function hasExtension(extension) {
133
+ return (path) => path.toLowerCase().endsWith(extension);
134
+ }
135
+ function hasAnyExtension(extensions) {
136
+ return (path) => extensions.some((extension) => hasExtension(extension)(path));
137
+ }
138
+ async function extractPdf(pi, source, signal) {
139
+ const extracted = await extractWithMarkItDown(pi, source, signal);
140
+ return extracted || pdfExtractionFailureMessage(source);
141
+ }
142
+ export function docxExtractionFailureMessage(source) {
143
+ return `_DOCX content could not be converted to markdown from ${source}. Ensure uvx and markitdown are installed._\n`;
144
+ }
145
+ async function extractDocx(pi, source, signal) {
146
+ const extracted = await extractWithMarkItDown(pi, source, signal);
147
+ return extracted || docxExtractionFailureMessage(source);
148
+ }
149
+ async function extractPdfUrl(pi, url, signal) {
150
+ const extracted = await extractPdf(pi, url, signal);
151
+ const failed = extracted.includes("could not be converted");
152
+ return {
153
+ extracted,
154
+ title: titleFromMarkdown(extracted),
155
+ extractor: "markitdown",
156
+ extraction_status: failed ? "failed" : "success",
157
+ content_type: "application/pdf",
158
+ };
159
+ }
160
+ async function extractTextUrl(pi, url, signal) {
161
+ const markitdownExtracted = await extractWithMarkItDown(pi, url, signal);
162
+ if (markitdownExtracted) {
163
+ return {
164
+ extracted: markitdownExtracted,
165
+ title: titleFromMarkdown(markitdownExtracted),
166
+ extractor: "markitdown",
167
+ extraction_status: "success",
168
+ };
169
+ }
170
+ const curlExtracted = await fetchTextUrl(pi, url, signal);
171
+ if (!curlExtracted)
172
+ return { extracted: "", extractor: "none", extraction_status: "failed" };
173
+ if (looksLikePdf(curlExtracted)) {
174
+ return {
175
+ extracted: pdfExtractionFailureMessage(url),
176
+ extractor: "curl",
177
+ extraction_status: "failed",
178
+ content_type: "application/pdf",
179
+ };
180
+ }
181
+ const normalized = htmlToMarkdown(curlExtracted);
182
+ return {
183
+ extracted: normalized,
184
+ title: titleFromMarkdown(normalized) ?? titleFromHtml(curlExtracted),
185
+ extractor: "htmlToMarkdown",
186
+ extraction_status: "success",
187
+ };
188
+ }
189
+ async function extractWithMarkItDown(pi, source, signal) {
190
+ try {
191
+ if (!(await hasMarkItDown(pi, signal)))
192
+ return "";
193
+ const mdResult = await exec(pi, "sh", ["-c", `uvx --from 'markitdown[docx,pdf]' markitdown "${source}" 2>/dev/null || echo ""`], { signal, timeout: markitdownTimeoutMs() });
194
+ return mdResult.stdout.trim() ? mdResult.stdout : "";
195
+ }
196
+ catch {
197
+ return "";
198
+ }
199
+ }
200
+ async function hasMarkItDown(pi, signal) {
201
+ const markitdown = await exec(pi, "sh", ["-c", `which uvx >/dev/null 2>&1 && echo "yes" || echo "no"`], { signal });
202
+ return markitdown.stdout.trim() === "yes";
203
+ }
204
+ async function fetchTextUrl(pi, url, signal) {
205
+ try {
206
+ const curlResult = await exec(pi, "curl", ["-sL", "--max-time", String(DEFAULT_CURL_TIMEOUT_SECONDS), url], {
207
+ signal,
208
+ timeout: (DEFAULT_CURL_TIMEOUT_SECONDS + 5) * 1_000,
209
+ });
210
+ return curlResult.stdout || "";
211
+ }
212
+ catch {
213
+ return "";
214
+ }
215
+ }
216
+ function markitdownTimeoutMs() {
217
+ return positiveIntegerFromEnv("WIKI_MARKITDOWN_TIMEOUT_MS", DEFAULT_MARKITDOWN_TIMEOUT_MS);
218
+ }
219
+ function positiveIntegerFromEnv(name, fallback) {
220
+ const raw = process.env[name];
221
+ if (!raw)
222
+ return fallback;
223
+ const parsed = Number.parseInt(raw, 10);
224
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : fallback;
225
+ }
226
+ function isPdfUrl(url) {
227
+ try {
228
+ return new URL(url).pathname.toLowerCase().endsWith(".pdf");
229
+ }
230
+ catch {
231
+ return url.toLowerCase().split(/[?#]/, 1)[0].endsWith(".pdf");
232
+ }
233
+ }
234
+ function looksLikePdf(content) {
235
+ return content.trimStart().startsWith("%PDF-");
236
+ }
237
+ function titleFromMarkdown(markdown) {
238
+ return markdown.match(/^#\s+(.+)$/m)?.[1]?.trim();
239
+ }
240
+ function titleFromHtml(html) {
241
+ return html.match(/<title>([^<]*)<\/title>/i)?.[1]?.trim();
242
+ }
243
+ /** Decode common HTML/XML entities. Shared by xmlToMarkdown and htmlToMarkdown. */
244
+ function decodeHtmlEntities(text) {
245
+ return text.replace(/&(?:amp|lt|gt|quot|apos|#\d+);/gi, (entity) => {
246
+ const map = {
247
+ "&amp;": "&",
248
+ "&lt;": "<",
249
+ "&gt;": ">",
250
+ "&quot;": '"',
251
+ "&apos;": "'",
252
+ };
253
+ const lower = entity.toLowerCase();
254
+ if (map[lower])
255
+ return map[lower];
256
+ if (lower.startsWith("&#"))
257
+ return String.fromCodePoint(Number.parseInt(entity.slice(2, -1)));
258
+ return entity;
259
+ });
260
+ }
261
+ /** Basic XML to markdown conversion: strip tags while preserving text structure. */
262
+ function xmlToMarkdown(xml) {
263
+ let title = "";
264
+ const titleMatch = xml.match(/<title[^>]*>([^<]*)<\/title>/i);
265
+ if (titleMatch)
266
+ title = titleMatch[1].trim();
267
+ let text = xml.replace(/<\?xml[^>]*\?>\s*/gi, "");
268
+ text = text.replace(/<!DOCTYPE[^>]*>\s*/gi, "");
269
+ text = text.replace(/<\/(p|div|section|article|li|h\d|tr|blockquote|pre)>/gi, "\n");
270
+ text = text.replace(/<br\s*\/?>/gi, "\n");
271
+ let prev = "";
272
+ while (prev !== text) {
273
+ prev = text;
274
+ text = text.replace(/<[a-zA-Z\/!?][^>]*>/g, "");
275
+ }
276
+ text = text.replace(/</g, "");
277
+ text = decodeHtmlEntities(text);
278
+ text = text.replace(/\n{3,}/g, "\n\n").trim();
279
+ if (!text)
280
+ return xml;
281
+ const lines = [];
282
+ if (title)
283
+ lines.push(`# ${title}\n`);
284
+ lines.push(text);
285
+ return lines.join("\n\n");
286
+ }
287
+ /**
288
+ * Lightweight HTML-to-markdown normalizer for the curl fallback path.
289
+ *
290
+ * Pre-strips page chrome (nav, header, footer, script, style) that
291
+ * node-html-markdown does not remove, then delegates full conversion —
292
+ * bold, italic, code blocks, tables, ordered lists, image alt text — to
293
+ * node-html-markdown. Prepends the <title> as a # heading when the body
294
+ * has no <h1> of its own.
295
+ *
296
+ * Falls back to the original HTML if conversion yields an empty string.
297
+ */
298
+ export function htmlToMarkdown(input) {
299
+ // 1. Extract <title> from original before stripping head
300
+ const title = input.match(/<title[^>]*>([^<]*)<\/title>/i)?.[1]?.trim() ?? "";
301
+ // 2. Strip <head> and noise blocks that node-html-markdown won't remove
302
+ let html = input.replace(/<head[\s\S]*?<\/head>/gi, "");
303
+ let previousHtml = "";
304
+ while (previousHtml !== html) {
305
+ previousHtml = html;
306
+ html = html.replace(/<(script|style|nav|header|footer|noscript)[\s\S]*?<\/\1>/gi, "");
307
+ }
308
+ // 3. Delegate to node-html-markdown for full semantic conversion
309
+ const converted = NodeHtmlMarkdown.translate(html).trim();
310
+ if (!converted)
311
+ return input;
312
+ // 4. Prepend <title> as # heading only if body has no <h1> of its own
313
+ const hasBodyH1 = /<h1[^>]*>[\s\S]*?<\/h1>/i.test(html);
314
+ const lines = [];
315
+ if (title && !hasBodyH1)
316
+ lines.push(`# ${title}\n`);
317
+ lines.push(converted);
318
+ return lines.join("\n");
319
+ }
320
+ function jsonToMarkdown(json) {
321
+ let value;
322
+ try {
323
+ value = JSON.parse(json);
324
+ }
325
+ catch {
326
+ return json;
327
+ }
328
+ const lines = [];
329
+ const title = titleFromValue(value) || "JSON Extract";
330
+ lines.push(`# ${title}`, "");
331
+ renderJsonValue(value, lines, 0);
332
+ const markdown = lines
333
+ .join("\n")
334
+ .replace(/\n{3,}/g, "\n\n")
335
+ .trim();
336
+ return markdown || json;
337
+ }
338
+ function titleFromValue(value) {
339
+ if (!isRecord(value))
340
+ return undefined;
341
+ for (const key of ["title", "name", "id"]) {
342
+ const candidate = value[key];
343
+ if (typeof candidate === "string" && candidate.trim())
344
+ return candidate.trim();
345
+ }
346
+ return undefined;
347
+ }
348
+ function isRecord(value) {
349
+ return typeof value === "object" && value !== null && !Array.isArray(value);
350
+ }
351
+ function renderJsonValue(value, lines, depth, label) {
352
+ if (Array.isArray(value)) {
353
+ renderJsonArray(value, lines, depth, label);
354
+ return;
355
+ }
356
+ if (isRecord(value)) {
357
+ renderJsonObject(value, lines, depth, label);
358
+ return;
359
+ }
360
+ if (label)
361
+ lines.push(`${indent(depth)}- **${humanizeKey(label)}:** ${formatJsonScalar(value)}`);
362
+ else
363
+ lines.push(`${indent(depth)}- ${formatJsonScalar(value)}`);
364
+ }
365
+ function renderJsonObject(object, lines, depth, label) {
366
+ if (label) {
367
+ lines.push(`${heading(depth)} ${humanizeKey(label)}`, "");
368
+ }
369
+ for (const [key, value] of Object.entries(object)) {
370
+ if (Array.isArray(value) || isRecord(value)) {
371
+ const childDepth = label ? depth + 1 : depth;
372
+ renderJsonValue(value, lines, childDepth, key);
373
+ }
374
+ else {
375
+ lines.push(`${indent(depth)}- **${humanizeKey(key)}:** ${formatJsonScalar(value)}`);
376
+ }
377
+ }
378
+ lines.push("");
379
+ }
380
+ function renderJsonArray(array, lines, depth, label) {
381
+ if (label)
382
+ lines.push(`${heading(depth)} ${humanizeKey(label)}`, "");
383
+ if (array.length === 0) {
384
+ lines.push(`${indent(depth)}- _(empty)_`, "");
385
+ return;
386
+ }
387
+ for (const [index, item] of array.entries()) {
388
+ if (isRecord(item)) {
389
+ const itemTitle = titleFromValue(item) || `Item ${index + 1}`;
390
+ const itemDepth = label ? depth + 1 : depth;
391
+ lines.push(`${heading(itemDepth)} ${itemTitle}`, "");
392
+ renderJsonObject(item, lines, itemDepth);
393
+ }
394
+ else if (Array.isArray(item)) {
395
+ lines.push(`${indent(depth)}- Item ${index + 1}:`);
396
+ renderJsonArray(item, lines, depth + 1);
397
+ }
398
+ else {
399
+ lines.push(`${indent(depth)}- ${formatJsonScalar(item)}`);
400
+ }
401
+ }
402
+ lines.push("");
403
+ }
404
+ function formatJsonScalar(value) {
405
+ if (value === null)
406
+ return "null";
407
+ if (typeof value === "string")
408
+ return value;
409
+ if (typeof value === "number" || typeof value === "boolean")
410
+ return String(value);
411
+ return String(value);
412
+ }
413
+ function humanizeKey(key) {
414
+ return key
415
+ .replace(/[_-]+/g, " ")
416
+ .replace(/([a-z0-9])([A-Z])/g, "$1 $2")
417
+ .replace(/\s+/g, " ")
418
+ .trim()
419
+ .replace(/^./, (char) => char.toUpperCase());
420
+ }
421
+ function heading(depth) {
422
+ return "#".repeat(Math.min(depth + 2, 6));
423
+ }
424
+ function indent(depth) {
425
+ return " ".repeat(Math.max(0, depth));
426
+ }
@@ -0,0 +1,229 @@
1
+ import { mkdirSync, writeFileSync } from "node:fs";
2
+ import { copyFile } from "node:fs/promises";
3
+ import { extname, join } from "node:path";
4
+ import { createKnowledgeDocument, serializeKnowledgeDocument } from "./knowledge-document.js";
5
+ import { appendEvent } from "./metadata.js";
6
+ import { binaryExtractionFailureMessage, detectBinaryMagicBytes, extractUrlContent, fileExtractorFor, } from "./source-extractors.js";
7
+ import { exec, fmtDate, nextSourceId, readText, writeJson, } from "./utils.js";
8
+ import { assertWritableVault } from "./vault-format.js";
9
+ const URL_ORIGINAL_EXTENSIONS = new Set([".html", ".htm", ".md", ".pdf", ".txt", ".xml", ".json"]);
10
+ /** Capture a URL into a source packet. */
11
+ export async function captureUrl(pi, paths, url, signal) {
12
+ return captureSource(paths, urlCaptureSource(pi, url, signal));
13
+ }
14
+ /** Capture a local file into a source packet. */
15
+ export async function captureFile(pi, paths, filePath, signal) {
16
+ return captureSource(paths, fileCaptureSource(pi, filePath, signal));
17
+ }
18
+ /** Capture pasted text into a source packet. */
19
+ export function captureText(paths, text, title) {
20
+ return captureSourceSync(paths, textCaptureSource(text, title));
21
+ }
22
+ async function captureSource(paths, source) {
23
+ assertWritableVault(paths);
24
+ const packet = createSourcePacket(paths, source.needsOriginalDir);
25
+ await source.preserveOriginal?.(packet.packetPath);
26
+ const content = await source.extract();
27
+ return finalizeCapture(paths, packet, source, content);
28
+ }
29
+ function captureSourceSync(paths, source) {
30
+ assertWritableVault(paths);
31
+ const packet = createSourcePacket(paths, source.needsOriginalDir);
32
+ const content = source.extract();
33
+ return finalizeCapture(paths, packet, source, content);
34
+ }
35
+ function urlCaptureSource(pi, url, signal) {
36
+ return {
37
+ needsOriginalDir: true,
38
+ fallbackText: contentExtractionFailureMessage(url),
39
+ preserveOriginal: (packetPath) => preserveUrlOriginal(pi, packetPath, url, signal),
40
+ extract: () => extractUrlContent(pi, url, signal),
41
+ manifest: (content) => ({
42
+ title: content.title || url,
43
+ url,
44
+ format: "web",
45
+ }),
46
+ event: () => ({ url, format: "web" }),
47
+ };
48
+ }
49
+ function fileCaptureSource(pi, filePath, signal) {
50
+ const fileName = filePath.split("/").pop() || "unknown";
51
+ const extractor = fileExtractorFor(filePath);
52
+ const content = extractor.shouldReadText ? readText(filePath) : "";
53
+ return {
54
+ needsOriginalDir: true,
55
+ fallbackText: "",
56
+ preserveOriginal: (packetPath) => preserveFileOriginal(packetPath, filePath, fileName, content),
57
+ extract: async () => {
58
+ // Guard: if we hit the generic catch-all extractor, check for binary magic bytes first
59
+ if (extractor.format === "file") {
60
+ const binaryFormat = await detectBinaryMagicBytes(filePath);
61
+ if (binaryFormat) {
62
+ return {
63
+ extracted: binaryExtractionFailureMessage(binaryFormat),
64
+ extractor: "magicBytes",
65
+ extraction_status: "unsupported",
66
+ };
67
+ }
68
+ }
69
+ const extractedStr = await extractor.extract({ pi, filePath, content, signal });
70
+ const failed = extractedStr.includes("could not be converted");
71
+ return {
72
+ extracted: extractedStr,
73
+ extractor: extractor.extractorName ?? "passthrough",
74
+ extraction_status: (failed ? "failed" : "success"),
75
+ ...(extractor.content_type ? { content_type: extractor.content_type } : {}),
76
+ };
77
+ },
78
+ manifest: () => ({
79
+ title: fileName,
80
+ file_path: filePath,
81
+ format: extractor.format,
82
+ }),
83
+ event: () => ({ file_path: filePath, format: extractor.format }),
84
+ };
85
+ }
86
+ function textCaptureSource(text, title) {
87
+ return {
88
+ needsOriginalDir: false,
89
+ fallbackText: "",
90
+ extract: () => ({ extracted: text }),
91
+ manifest: () => ({
92
+ title: title || `Pasted text — ${fmtDate()}`,
93
+ format: "text",
94
+ }),
95
+ event: () => ({ format: "text" }),
96
+ };
97
+ }
98
+ function createSourcePacket(paths, needsOriginalDir) {
99
+ const sourceId = nextSourceId(paths);
100
+ const packetPath = join(paths.rawSources, sourceId);
101
+ mkdirSync(packetPath, { recursive: true });
102
+ mkdirSync(join(packetPath, "attachments"), { recursive: true });
103
+ if (needsOriginalDir)
104
+ mkdirSync(join(packetPath, "original"), { recursive: true });
105
+ return { sourceId, packetPath };
106
+ }
107
+ function finalizeCapture(paths, packet, source, content) {
108
+ assertWritableVault(paths);
109
+ const extracted = content.extracted || source.fallbackText;
110
+ const manifest = {
111
+ id: packet.sourceId,
112
+ captured: fmtDate(),
113
+ packet_version: "1.0",
114
+ ...source.manifest({ ...content, extracted }),
115
+ extractor: content.extractor ?? "passthrough",
116
+ extraction_status: content.extraction_status ?? "success",
117
+ ...(content.content_type ? { content_type: content.content_type } : {}),
118
+ };
119
+ writeFileSync(join(packet.packetPath, "extracted.md"), extracted, "utf-8");
120
+ writeJson(join(packet.packetPath, "manifest.json"), manifest);
121
+ const sourcePagePath = join(paths.wiki, "sources", `${packet.sourceId}.md`);
122
+ writeFileSync(sourcePagePath, buildSourcePageSkeleton(manifest, extracted), "utf-8");
123
+ appendEvent(paths, {
124
+ kind: "capture",
125
+ source_id: packet.sourceId,
126
+ ...source.event({ ...content, extracted }),
127
+ });
128
+ return {
129
+ sourceId: packet.sourceId,
130
+ packetPath: packet.packetPath,
131
+ sourcePagePath,
132
+ extracted,
133
+ };
134
+ }
135
+ async function preserveFileOriginal(packetPath, filePath, fileName, fallbackContent) {
136
+ try {
137
+ await copyFile(filePath, join(packetPath, "original", fileName));
138
+ }
139
+ catch {
140
+ // If copying fails, preserve whatever text content was available.
141
+ writeFileSync(join(packetPath, "original", fileName), fallbackContent, "utf-8");
142
+ }
143
+ }
144
+ async function preserveUrlOriginal(pi, packetPath, url, signal) {
145
+ const originalPath = join(packetPath, "original", originalFileNameForUrl(url));
146
+ try {
147
+ await exec(pi, "curl", ["-sL", "--max-time", "30", "-o", originalPath, url], {
148
+ signal,
149
+ timeout: 35_000,
150
+ });
151
+ }
152
+ catch {
153
+ // Preserve best-effort extraction behavior even when the original artifact cannot be saved.
154
+ }
155
+ }
156
+ function originalFileNameForUrl(url) {
157
+ try {
158
+ const parsed = new URL(url);
159
+ const ext = extname(parsed.pathname).toLowerCase();
160
+ if (URL_ORIGINAL_EXTENSIONS.has(ext))
161
+ return `source${ext}`;
162
+ }
163
+ catch {
164
+ const path = url.split(/[?#]/, 1)[0] ?? "";
165
+ const ext = extname(path).toLowerCase();
166
+ if (URL_ORIGINAL_EXTENSIONS.has(ext))
167
+ return `source${ext}`;
168
+ }
169
+ return "source.html";
170
+ }
171
+ function contentExtractionFailureMessage(source) {
172
+ return `_Content could not be extracted from ${source}_\n`;
173
+ }
174
+ /** Build a skeleton source page from manifest and extracted text. */
175
+ function buildSourcePageSkeleton(manifest, extracted) {
176
+ const id = String(manifest.id);
177
+ const title = String(manifest.title || id);
178
+ const url = manifest.url ? `\n> _Original: [${manifest.url}](${manifest.url})_` : "";
179
+ const format = String(manifest.format || "unknown");
180
+ const captured = String(manifest.captured || fmtDate());
181
+ // Generate a brief auto-summary (first 500 chars)
182
+ const preview = extracted
183
+ .replace(/[#*_`]/g, "")
184
+ .replace(/\s+/g, " ")
185
+ .trim()
186
+ .slice(0, 500);
187
+ const body = `# ${title}${url}
188
+
189
+ ## Summary
190
+
191
+ [LLM: Replace with 2-3 paragraph summary of key content]
192
+
193
+ > _Auto-preview: ${preview}${extracted.length > 500 ? "..." : ""}_
194
+
195
+ ## Key Takeaways
196
+
197
+ - [LLM: Most important point]
198
+ - [LLM: Second important point]
199
+ - [LLM: Third important point]
200
+
201
+ ## Entities Mentioned
202
+
203
+ - [LLM: Add linked entities after review]
204
+
205
+ ## Concepts Mentioned
206
+
207
+ - [LLM: Add linked concepts after review]
208
+
209
+ ## Notable Quotes
210
+
211
+ > [LLM: Important quote] — attribution
212
+
213
+ ## Source Packet
214
+
215
+ - **ID:** \`sources/${id}\`
216
+ - **Extracted:** \`raw/sources/${id}/extracted.md\`
217
+ - **Manifest:** \`raw/sources/${id}/manifest.json\`
218
+ `;
219
+ const doc = createKnowledgeDocument(`sources/${id}.md`, {
220
+ type: "source",
221
+ title,
222
+ format,
223
+ source_id: id,
224
+ raw_path: `raw/sources/${id}/extracted.md`,
225
+ captured,
226
+ status: "skeleton",
227
+ }, body);
228
+ return serializeKnowledgeDocument(doc);
229
+ }
@@ -0,0 +1,41 @@
1
+ import { agentLoop, } from "@mariozechner/pi-agent-core";
2
+ /**
3
+ * Run a sub-agent loop to completion.
4
+ *
5
+ * Returns nothing useful directly — by design, results are collected by the
6
+ * `tools` the caller passes (their `execute` accumulates into caller-owned
7
+ * state). This keeps the runner generic across every background task type.
8
+ */
9
+ export async function runSubAgent(args) {
10
+ const { model, apiKey, headers, systemPrompt, userPrompt, tools, maxTokens, signal } = args;
11
+ const text = userPrompt.trim();
12
+ if (!text)
13
+ return;
14
+ const prompts = [
15
+ {
16
+ role: "user",
17
+ content: [{ type: "text", text }],
18
+ timestamp: Date.now(),
19
+ },
20
+ ];
21
+ const context = {
22
+ systemPrompt,
23
+ messages: [],
24
+ tools,
25
+ };
26
+ const reasoning = model.reasoning;
27
+ const config = {
28
+ model,
29
+ apiKey,
30
+ headers,
31
+ maxTokens: maxTokens ?? 4096,
32
+ convertToLlm: (msgs) => msgs,
33
+ toolExecution: "sequential",
34
+ ...(reasoning ? { reasoning: "high" } : {}),
35
+ };
36
+ const stream = agentLoop(prompts, context, config, signal);
37
+ for await (const _event of stream) {
38
+ // Drain events; tool `execute` callbacks collect results caller-side.
39
+ }
40
+ await stream.result();
41
+ }