@zosmaai/pi-llm-wiki 0.10.7 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +4 -0
- package/README.de.md +35 -4
- package/README.es.md +260 -170
- package/README.fr.md +35 -4
- package/README.hi.md +35 -4
- package/README.ja.md +35 -4
- package/README.ko.md +35 -4
- package/README.md +38 -3
- package/README.pt.md +35 -4
- package/README.ru.md +35 -4
- package/README.zh.md +260 -170
- package/assets/demo.gif +0 -0
- package/dist/extensions/llm-wiki/lib/bootstrap.js +71 -0
- package/dist/extensions/llm-wiki/lib/embeddings.js +401 -0
- package/dist/extensions/llm-wiki/lib/guardrails.js +232 -0
- package/dist/extensions/llm-wiki/lib/indexing.js +78 -0
- package/dist/extensions/llm-wiki/lib/ingest-worker.js +310 -0
- package/dist/extensions/llm-wiki/lib/inject.js +65 -0
- package/dist/extensions/llm-wiki/lib/knowledge-document.js +442 -0
- package/dist/extensions/llm-wiki/lib/knowledge-links.js +206 -0
- package/dist/extensions/llm-wiki/lib/legacy-repair.js +443 -0
- package/dist/extensions/llm-wiki/lib/metadata.js +499 -0
- package/dist/extensions/llm-wiki/lib/model-command.js +86 -0
- package/dist/extensions/llm-wiki/lib/observation.js +283 -0
- package/dist/extensions/llm-wiki/lib/recall.js +875 -0
- package/dist/extensions/llm-wiki/lib/retro.js +158 -0
- package/dist/extensions/llm-wiki/lib/runtime.js +191 -0
- package/dist/extensions/llm-wiki/lib/source-extractors.js +426 -0
- package/dist/extensions/llm-wiki/lib/source-packet.js +229 -0
- package/dist/extensions/llm-wiki/lib/subagent.js +41 -0
- package/dist/extensions/llm-wiki/lib/task-config.js +172 -0
- package/dist/extensions/llm-wiki/lib/tools.js +1192 -0
- package/dist/extensions/llm-wiki/lib/trajectories-command.js +51 -0
- package/dist/extensions/llm-wiki/lib/trajectory.js +467 -0
- package/dist/extensions/llm-wiki/lib/utils.js +347 -0
- package/dist/extensions/llm-wiki/lib/vault-format.js +247 -0
- package/dist/extensions/llm-wiki/lib/visible-status.js +31 -0
- package/dist/extensions/llm-wiki/lib/wiki-service.js +128 -0
- package/dist/mcp/exec.js +121 -0
- package/dist/mcp/index.js +229 -0
- package/dist/mcp/operations.js +130 -0
- package/dist/package.json +1 -0
- package/docs/superpowers/plans/2026-08-02-okf-foundation.md +1579 -0
- package/docs/superpowers/plans/2026-08-03-okf-foundation-remediation.md +3005 -0
- package/docs/superpowers/plans/2026-08-06-okf-foundation-release-remediation.md +1174 -0
- package/docs/superpowers/specs/2026-08-02-okf-foundation-design.md +578 -0
- package/docs/superpowers/specs/2026-08-02-okf-v0.2-interoperability-design.md +538 -0
- package/extensions/llm-wiki/index.ts +22 -36
- package/extensions/llm-wiki/lib/bootstrap.ts +84 -0
- package/extensions/llm-wiki/lib/embeddings.ts +9 -3
- package/extensions/llm-wiki/lib/guardrails.ts +174 -29
- package/extensions/llm-wiki/lib/indexing.ts +2 -1
- package/extensions/llm-wiki/lib/ingest-worker.ts +170 -29
- package/extensions/llm-wiki/lib/knowledge-document.ts +661 -0
- package/extensions/llm-wiki/lib/knowledge-links.ts +282 -0
- package/extensions/llm-wiki/lib/legacy-repair.ts +572 -0
- package/extensions/llm-wiki/lib/metadata.ts +531 -116
- package/extensions/llm-wiki/lib/observation.ts +37 -43
- package/extensions/llm-wiki/lib/recall.ts +61 -33
- package/extensions/llm-wiki/lib/retro.ts +65 -41
- package/extensions/llm-wiki/lib/source-extractors.ts +12 -17
- package/extensions/llm-wiki/lib/source-packet.ts +44 -31
- package/extensions/llm-wiki/lib/tools.ts +406 -348
- package/extensions/llm-wiki/lib/trajectory.ts +15 -1
- package/extensions/llm-wiki/lib/utils.ts +121 -130
- package/extensions/llm-wiki/lib/vault-format.ts +363 -0
- package/extensions/llm-wiki/lib/wiki-service.ts +183 -0
- package/mcp/exec.ts +122 -0
- package/mcp/index.ts +60 -250
- package/mcp/operations.ts +176 -0
- package/package.json +8 -2
- package/scripts/migrate-llm-wiki.js +801 -0
- package/skills/llm-wiki/SKILL.md +8 -6
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
import { open } from "node:fs/promises";
|
|
2
|
+
import { NodeHtmlMarkdown } from "node-html-markdown";
|
|
3
|
+
import { exec } from "./utils.js";
|
|
4
|
+
// ---------------------------------------------------------------------------
|
|
5
|
+
// Binary magic byte detection
|
|
6
|
+
// ---------------------------------------------------------------------------
|
|
7
|
+
const BINARY_SIGNATURES = [
|
|
8
|
+
// Archives & documents
|
|
9
|
+
{ bytes: [0x50, 0x4b, 0x03, 0x04], format: "zip" }, // ZIP / DOCX / XLSX / PPTX / JAR
|
|
10
|
+
{ bytes: [0x25, 0x50, 0x44, 0x46], format: "pdf" }, // %PDF
|
|
11
|
+
{ bytes: [0x37, 0x7a, 0xbc, 0xaf], format: "7z" }, // 7-Zip
|
|
12
|
+
{ bytes: [0x1f, 0x8b], format: "gzip" }, // gzip / .tar.gz
|
|
13
|
+
// Images
|
|
14
|
+
{ bytes: [0x89, 0x50, 0x4e, 0x47], format: "png" }, // PNG
|
|
15
|
+
{ bytes: [0xff, 0xd8, 0xff], format: "jpeg" }, // JPEG
|
|
16
|
+
{ bytes: [0x47, 0x49, 0x46, 0x38], format: "gif" }, // GIF8
|
|
17
|
+
{ bytes: [0x42, 0x4d], format: "bmp" }, // BMP
|
|
18
|
+
{ bytes: [0x49, 0x49, 0x2a, 0x00], format: "tiff" }, // TIFF (little-endian)
|
|
19
|
+
{ bytes: [0x4d, 0x4d, 0x00, 0x2a], format: "tiff" }, // TIFF (big-endian)
|
|
20
|
+
{ bytes: [0x52, 0x49, 0x46, 0x46], format: "riff" }, // RIFF (WAV / AVI / WebP)
|
|
21
|
+
// Executables & binaries
|
|
22
|
+
{ bytes: [0x4d, 0x5a], format: "exe" }, // Windows PE (EXE / DLL)
|
|
23
|
+
{ bytes: [0xcf, 0xfa, 0xed, 0xfe], format: "macho" }, // Mach-O 64-bit LE
|
|
24
|
+
{ bytes: [0xce, 0xfa, 0xed, 0xfe], format: "macho" }, // Mach-O 32-bit LE
|
|
25
|
+
{ bytes: [0xfe, 0xed, 0xfa, 0xcf], format: "macho" }, // Mach-O 64-bit BE
|
|
26
|
+
{ bytes: [0xfe, 0xed, 0xfa, 0xce], format: "macho" }, // Mach-O 32-bit BE
|
|
27
|
+
{ bytes: [0xca, 0xfe, 0xba, 0xbe], format: "class" }, // Java .class / Mach-O FAT
|
|
28
|
+
{ bytes: [0x7f, 0x45, 0x4c, 0x46], format: "elf" }, // ELF binary
|
|
29
|
+
{ bytes: [0x00, 0x61, 0x73, 0x6d], format: "wasm" }, // WebAssembly
|
|
30
|
+
// Data & media
|
|
31
|
+
{ bytes: [0x53, 0x51, 0x4c, 0x69], format: "sqlite" }, // SQLite
|
|
32
|
+
{ bytes: [0x49, 0x44, 0x33], format: "mp3" }, // MP3 (ID3 tag)
|
|
33
|
+
];
|
|
34
|
+
/**
|
|
35
|
+
* Reads the first 8 bytes of `filePath` and checks them against known binary
|
|
36
|
+
* magic byte signatures. Returns the detected format name or `null` for text.
|
|
37
|
+
*/
|
|
38
|
+
export async function detectBinaryMagicBytes(filePath) {
|
|
39
|
+
let handle;
|
|
40
|
+
try {
|
|
41
|
+
handle = await open(filePath, "r");
|
|
42
|
+
const buf = Buffer.alloc(8);
|
|
43
|
+
const { bytesRead } = await handle.read(buf, 0, 8, 0);
|
|
44
|
+
const header = buf.subarray(0, bytesRead);
|
|
45
|
+
for (const { bytes, format } of BINARY_SIGNATURES) {
|
|
46
|
+
if (bytes.every((b, i) => header[i] === b))
|
|
47
|
+
return format;
|
|
48
|
+
}
|
|
49
|
+
return null;
|
|
50
|
+
}
|
|
51
|
+
catch {
|
|
52
|
+
return null; // Unreadable file — let the extractor deal with it
|
|
53
|
+
}
|
|
54
|
+
finally {
|
|
55
|
+
await handle?.close();
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
export function binaryExtractionFailureMessage(format) {
|
|
59
|
+
return `_Binary file could not be converted to markdown (detected format: ${format}).\nCapture a text-based version or a URL pointing to readable content instead._\n`;
|
|
60
|
+
}
|
|
61
|
+
// ---------------------------------------------------------------------------
|
|
62
|
+
const DEFAULT_MARKITDOWN_TIMEOUT_MS = 180_000;
|
|
63
|
+
const DEFAULT_CURL_TIMEOUT_SECONDS = 30;
|
|
64
|
+
const FILE_EXTRACTORS = [
|
|
65
|
+
{
|
|
66
|
+
format: "pdf",
|
|
67
|
+
shouldReadText: false,
|
|
68
|
+
extractorName: "markitdown",
|
|
69
|
+
content_type: "application/pdf",
|
|
70
|
+
matches: hasExtension(".pdf"),
|
|
71
|
+
extract: ({ pi, filePath, signal }) => extractPdf(pi, filePath, signal),
|
|
72
|
+
},
|
|
73
|
+
textFileExtractor("markdown", [".md"], "text/markdown"),
|
|
74
|
+
textFileExtractor("text", [".txt"], "text/plain"),
|
|
75
|
+
textFileExtractor("html", [".html", ".htm"], "text/html"),
|
|
76
|
+
{
|
|
77
|
+
format: "xml",
|
|
78
|
+
shouldReadText: true,
|
|
79
|
+
extractorName: "xmlToMarkdown",
|
|
80
|
+
content_type: "application/xml",
|
|
81
|
+
matches: hasExtension(".xml"),
|
|
82
|
+
extract: ({ content }) => xmlToMarkdown(content),
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
format: "json",
|
|
86
|
+
shouldReadText: true,
|
|
87
|
+
extractorName: "jsonToMarkdown",
|
|
88
|
+
content_type: "application/json",
|
|
89
|
+
matches: hasExtension(".json"),
|
|
90
|
+
extract: ({ content }) => jsonToMarkdown(content),
|
|
91
|
+
},
|
|
92
|
+
{
|
|
93
|
+
format: "docx",
|
|
94
|
+
shouldReadText: false,
|
|
95
|
+
extractorName: "markitdown",
|
|
96
|
+
content_type: "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
97
|
+
matches: hasExtension(".docx"),
|
|
98
|
+
extract: ({ pi, filePath, signal }) => extractDocx(pi, filePath, signal),
|
|
99
|
+
},
|
|
100
|
+
textFileExtractor("file", []),
|
|
101
|
+
];
|
|
102
|
+
const URL_EXTRACTORS = [
|
|
103
|
+
{
|
|
104
|
+
matches: isPdfUrl,
|
|
105
|
+
extract: ({ pi, url, signal }) => extractPdfUrl(pi, url, signal),
|
|
106
|
+
},
|
|
107
|
+
{
|
|
108
|
+
matches: () => true,
|
|
109
|
+
extract: ({ pi, url, signal }) => extractTextUrl(pi, url, signal),
|
|
110
|
+
},
|
|
111
|
+
];
|
|
112
|
+
export function fileExtractorFor(filePath) {
|
|
113
|
+
return (FILE_EXTRACTORS.find((extractor) => extractor.matches(filePath)) ?? FILE_EXTRACTORS.at(-1));
|
|
114
|
+
}
|
|
115
|
+
export function extractUrlContent(pi, url, signal) {
|
|
116
|
+
const extractor = URL_EXTRACTORS.find((candidate) => candidate.matches(url)) ?? URL_EXTRACTORS.at(-1);
|
|
117
|
+
return extractor.extract({ pi, url, signal });
|
|
118
|
+
}
|
|
119
|
+
export function pdfExtractionFailureMessage(source) {
|
|
120
|
+
return `_PDF content could not be converted to markdown from ${source}. Try increasing WIKI_MARKITDOWN_TIMEOUT_MS._\n`;
|
|
121
|
+
}
|
|
122
|
+
function textFileExtractor(format, extensions, contentType) {
|
|
123
|
+
return {
|
|
124
|
+
format,
|
|
125
|
+
shouldReadText: true,
|
|
126
|
+
extractorName: "passthrough",
|
|
127
|
+
content_type: contentType,
|
|
128
|
+
matches: extensions.length ? hasAnyExtension(extensions) : () => true,
|
|
129
|
+
extract: ({ content }) => content,
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
function hasExtension(extension) {
|
|
133
|
+
return (path) => path.toLowerCase().endsWith(extension);
|
|
134
|
+
}
|
|
135
|
+
function hasAnyExtension(extensions) {
|
|
136
|
+
return (path) => extensions.some((extension) => hasExtension(extension)(path));
|
|
137
|
+
}
|
|
138
|
+
async function extractPdf(pi, source, signal) {
|
|
139
|
+
const extracted = await extractWithMarkItDown(pi, source, signal);
|
|
140
|
+
return extracted || pdfExtractionFailureMessage(source);
|
|
141
|
+
}
|
|
142
|
+
export function docxExtractionFailureMessage(source) {
|
|
143
|
+
return `_DOCX content could not be converted to markdown from ${source}. Ensure uvx and markitdown are installed._\n`;
|
|
144
|
+
}
|
|
145
|
+
async function extractDocx(pi, source, signal) {
|
|
146
|
+
const extracted = await extractWithMarkItDown(pi, source, signal);
|
|
147
|
+
return extracted || docxExtractionFailureMessage(source);
|
|
148
|
+
}
|
|
149
|
+
async function extractPdfUrl(pi, url, signal) {
|
|
150
|
+
const extracted = await extractPdf(pi, url, signal);
|
|
151
|
+
const failed = extracted.includes("could not be converted");
|
|
152
|
+
return {
|
|
153
|
+
extracted,
|
|
154
|
+
title: titleFromMarkdown(extracted),
|
|
155
|
+
extractor: "markitdown",
|
|
156
|
+
extraction_status: failed ? "failed" : "success",
|
|
157
|
+
content_type: "application/pdf",
|
|
158
|
+
};
|
|
159
|
+
}
|
|
160
|
+
async function extractTextUrl(pi, url, signal) {
|
|
161
|
+
const markitdownExtracted = await extractWithMarkItDown(pi, url, signal);
|
|
162
|
+
if (markitdownExtracted) {
|
|
163
|
+
return {
|
|
164
|
+
extracted: markitdownExtracted,
|
|
165
|
+
title: titleFromMarkdown(markitdownExtracted),
|
|
166
|
+
extractor: "markitdown",
|
|
167
|
+
extraction_status: "success",
|
|
168
|
+
};
|
|
169
|
+
}
|
|
170
|
+
const curlExtracted = await fetchTextUrl(pi, url, signal);
|
|
171
|
+
if (!curlExtracted)
|
|
172
|
+
return { extracted: "", extractor: "none", extraction_status: "failed" };
|
|
173
|
+
if (looksLikePdf(curlExtracted)) {
|
|
174
|
+
return {
|
|
175
|
+
extracted: pdfExtractionFailureMessage(url),
|
|
176
|
+
extractor: "curl",
|
|
177
|
+
extraction_status: "failed",
|
|
178
|
+
content_type: "application/pdf",
|
|
179
|
+
};
|
|
180
|
+
}
|
|
181
|
+
const normalized = htmlToMarkdown(curlExtracted);
|
|
182
|
+
return {
|
|
183
|
+
extracted: normalized,
|
|
184
|
+
title: titleFromMarkdown(normalized) ?? titleFromHtml(curlExtracted),
|
|
185
|
+
extractor: "htmlToMarkdown",
|
|
186
|
+
extraction_status: "success",
|
|
187
|
+
};
|
|
188
|
+
}
|
|
189
|
+
async function extractWithMarkItDown(pi, source, signal) {
|
|
190
|
+
try {
|
|
191
|
+
if (!(await hasMarkItDown(pi, signal)))
|
|
192
|
+
return "";
|
|
193
|
+
const mdResult = await exec(pi, "sh", ["-c", `uvx --from 'markitdown[docx,pdf]' markitdown "${source}" 2>/dev/null || echo ""`], { signal, timeout: markitdownTimeoutMs() });
|
|
194
|
+
return mdResult.stdout.trim() ? mdResult.stdout : "";
|
|
195
|
+
}
|
|
196
|
+
catch {
|
|
197
|
+
return "";
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
async function hasMarkItDown(pi, signal) {
|
|
201
|
+
const markitdown = await exec(pi, "sh", ["-c", `which uvx >/dev/null 2>&1 && echo "yes" || echo "no"`], { signal });
|
|
202
|
+
return markitdown.stdout.trim() === "yes";
|
|
203
|
+
}
|
|
204
|
+
async function fetchTextUrl(pi, url, signal) {
|
|
205
|
+
try {
|
|
206
|
+
const curlResult = await exec(pi, "curl", ["-sL", "--max-time", String(DEFAULT_CURL_TIMEOUT_SECONDS), url], {
|
|
207
|
+
signal,
|
|
208
|
+
timeout: (DEFAULT_CURL_TIMEOUT_SECONDS + 5) * 1_000,
|
|
209
|
+
});
|
|
210
|
+
return curlResult.stdout || "";
|
|
211
|
+
}
|
|
212
|
+
catch {
|
|
213
|
+
return "";
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
function markitdownTimeoutMs() {
|
|
217
|
+
return positiveIntegerFromEnv("WIKI_MARKITDOWN_TIMEOUT_MS", DEFAULT_MARKITDOWN_TIMEOUT_MS);
|
|
218
|
+
}
|
|
219
|
+
function positiveIntegerFromEnv(name, fallback) {
|
|
220
|
+
const raw = process.env[name];
|
|
221
|
+
if (!raw)
|
|
222
|
+
return fallback;
|
|
223
|
+
const parsed = Number.parseInt(raw, 10);
|
|
224
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : fallback;
|
|
225
|
+
}
|
|
226
|
+
function isPdfUrl(url) {
|
|
227
|
+
try {
|
|
228
|
+
return new URL(url).pathname.toLowerCase().endsWith(".pdf");
|
|
229
|
+
}
|
|
230
|
+
catch {
|
|
231
|
+
return url.toLowerCase().split(/[?#]/, 1)[0].endsWith(".pdf");
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
function looksLikePdf(content) {
|
|
235
|
+
return content.trimStart().startsWith("%PDF-");
|
|
236
|
+
}
|
|
237
|
+
function titleFromMarkdown(markdown) {
|
|
238
|
+
return markdown.match(/^#\s+(.+)$/m)?.[1]?.trim();
|
|
239
|
+
}
|
|
240
|
+
function titleFromHtml(html) {
|
|
241
|
+
return html.match(/<title>([^<]*)<\/title>/i)?.[1]?.trim();
|
|
242
|
+
}
|
|
243
|
+
/** Decode common HTML/XML entities. Shared by xmlToMarkdown and htmlToMarkdown. */
|
|
244
|
+
function decodeHtmlEntities(text) {
|
|
245
|
+
return text.replace(/&(?:amp|lt|gt|quot|apos|#\d+);/gi, (entity) => {
|
|
246
|
+
const map = {
|
|
247
|
+
"&": "&",
|
|
248
|
+
"<": "<",
|
|
249
|
+
">": ">",
|
|
250
|
+
""": '"',
|
|
251
|
+
"'": "'",
|
|
252
|
+
};
|
|
253
|
+
const lower = entity.toLowerCase();
|
|
254
|
+
if (map[lower])
|
|
255
|
+
return map[lower];
|
|
256
|
+
if (lower.startsWith("&#"))
|
|
257
|
+
return String.fromCodePoint(Number.parseInt(entity.slice(2, -1)));
|
|
258
|
+
return entity;
|
|
259
|
+
});
|
|
260
|
+
}
|
|
261
|
+
/** Basic XML to markdown conversion: strip tags while preserving text structure. */
|
|
262
|
+
function xmlToMarkdown(xml) {
|
|
263
|
+
let title = "";
|
|
264
|
+
const titleMatch = xml.match(/<title[^>]*>([^<]*)<\/title>/i);
|
|
265
|
+
if (titleMatch)
|
|
266
|
+
title = titleMatch[1].trim();
|
|
267
|
+
let text = xml.replace(/<\?xml[^>]*\?>\s*/gi, "");
|
|
268
|
+
text = text.replace(/<!DOCTYPE[^>]*>\s*/gi, "");
|
|
269
|
+
text = text.replace(/<\/(p|div|section|article|li|h\d|tr|blockquote|pre)>/gi, "\n");
|
|
270
|
+
text = text.replace(/<br\s*\/?>/gi, "\n");
|
|
271
|
+
let prev = "";
|
|
272
|
+
while (prev !== text) {
|
|
273
|
+
prev = text;
|
|
274
|
+
text = text.replace(/<[a-zA-Z\/!?][^>]*>/g, "");
|
|
275
|
+
}
|
|
276
|
+
text = text.replace(/</g, "");
|
|
277
|
+
text = decodeHtmlEntities(text);
|
|
278
|
+
text = text.replace(/\n{3,}/g, "\n\n").trim();
|
|
279
|
+
if (!text)
|
|
280
|
+
return xml;
|
|
281
|
+
const lines = [];
|
|
282
|
+
if (title)
|
|
283
|
+
lines.push(`# ${title}\n`);
|
|
284
|
+
lines.push(text);
|
|
285
|
+
return lines.join("\n\n");
|
|
286
|
+
}
|
|
287
|
+
/**
|
|
288
|
+
* Lightweight HTML-to-markdown normalizer for the curl fallback path.
|
|
289
|
+
*
|
|
290
|
+
* Pre-strips page chrome (nav, header, footer, script, style) that
|
|
291
|
+
* node-html-markdown does not remove, then delegates full conversion —
|
|
292
|
+
* bold, italic, code blocks, tables, ordered lists, image alt text — to
|
|
293
|
+
* node-html-markdown. Prepends the <title> as a # heading when the body
|
|
294
|
+
* has no <h1> of its own.
|
|
295
|
+
*
|
|
296
|
+
* Falls back to the original HTML if conversion yields an empty string.
|
|
297
|
+
*/
|
|
298
|
+
export function htmlToMarkdown(input) {
|
|
299
|
+
// 1. Extract <title> from original before stripping head
|
|
300
|
+
const title = input.match(/<title[^>]*>([^<]*)<\/title>/i)?.[1]?.trim() ?? "";
|
|
301
|
+
// 2. Strip <head> and noise blocks that node-html-markdown won't remove
|
|
302
|
+
let html = input.replace(/<head[\s\S]*?<\/head>/gi, "");
|
|
303
|
+
let previousHtml = "";
|
|
304
|
+
while (previousHtml !== html) {
|
|
305
|
+
previousHtml = html;
|
|
306
|
+
html = html.replace(/<(script|style|nav|header|footer|noscript)[\s\S]*?<\/\1>/gi, "");
|
|
307
|
+
}
|
|
308
|
+
// 3. Delegate to node-html-markdown for full semantic conversion
|
|
309
|
+
const converted = NodeHtmlMarkdown.translate(html).trim();
|
|
310
|
+
if (!converted)
|
|
311
|
+
return input;
|
|
312
|
+
// 4. Prepend <title> as # heading only if body has no <h1> of its own
|
|
313
|
+
const hasBodyH1 = /<h1[^>]*>[\s\S]*?<\/h1>/i.test(html);
|
|
314
|
+
const lines = [];
|
|
315
|
+
if (title && !hasBodyH1)
|
|
316
|
+
lines.push(`# ${title}\n`);
|
|
317
|
+
lines.push(converted);
|
|
318
|
+
return lines.join("\n");
|
|
319
|
+
}
|
|
320
|
+
function jsonToMarkdown(json) {
|
|
321
|
+
let value;
|
|
322
|
+
try {
|
|
323
|
+
value = JSON.parse(json);
|
|
324
|
+
}
|
|
325
|
+
catch {
|
|
326
|
+
return json;
|
|
327
|
+
}
|
|
328
|
+
const lines = [];
|
|
329
|
+
const title = titleFromValue(value) || "JSON Extract";
|
|
330
|
+
lines.push(`# ${title}`, "");
|
|
331
|
+
renderJsonValue(value, lines, 0);
|
|
332
|
+
const markdown = lines
|
|
333
|
+
.join("\n")
|
|
334
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
335
|
+
.trim();
|
|
336
|
+
return markdown || json;
|
|
337
|
+
}
|
|
338
|
+
function titleFromValue(value) {
|
|
339
|
+
if (!isRecord(value))
|
|
340
|
+
return undefined;
|
|
341
|
+
for (const key of ["title", "name", "id"]) {
|
|
342
|
+
const candidate = value[key];
|
|
343
|
+
if (typeof candidate === "string" && candidate.trim())
|
|
344
|
+
return candidate.trim();
|
|
345
|
+
}
|
|
346
|
+
return undefined;
|
|
347
|
+
}
|
|
348
|
+
function isRecord(value) {
|
|
349
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
350
|
+
}
|
|
351
|
+
function renderJsonValue(value, lines, depth, label) {
|
|
352
|
+
if (Array.isArray(value)) {
|
|
353
|
+
renderJsonArray(value, lines, depth, label);
|
|
354
|
+
return;
|
|
355
|
+
}
|
|
356
|
+
if (isRecord(value)) {
|
|
357
|
+
renderJsonObject(value, lines, depth, label);
|
|
358
|
+
return;
|
|
359
|
+
}
|
|
360
|
+
if (label)
|
|
361
|
+
lines.push(`${indent(depth)}- **${humanizeKey(label)}:** ${formatJsonScalar(value)}`);
|
|
362
|
+
else
|
|
363
|
+
lines.push(`${indent(depth)}- ${formatJsonScalar(value)}`);
|
|
364
|
+
}
|
|
365
|
+
function renderJsonObject(object, lines, depth, label) {
|
|
366
|
+
if (label) {
|
|
367
|
+
lines.push(`${heading(depth)} ${humanizeKey(label)}`, "");
|
|
368
|
+
}
|
|
369
|
+
for (const [key, value] of Object.entries(object)) {
|
|
370
|
+
if (Array.isArray(value) || isRecord(value)) {
|
|
371
|
+
const childDepth = label ? depth + 1 : depth;
|
|
372
|
+
renderJsonValue(value, lines, childDepth, key);
|
|
373
|
+
}
|
|
374
|
+
else {
|
|
375
|
+
lines.push(`${indent(depth)}- **${humanizeKey(key)}:** ${formatJsonScalar(value)}`);
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
lines.push("");
|
|
379
|
+
}
|
|
380
|
+
function renderJsonArray(array, lines, depth, label) {
|
|
381
|
+
if (label)
|
|
382
|
+
lines.push(`${heading(depth)} ${humanizeKey(label)}`, "");
|
|
383
|
+
if (array.length === 0) {
|
|
384
|
+
lines.push(`${indent(depth)}- _(empty)_`, "");
|
|
385
|
+
return;
|
|
386
|
+
}
|
|
387
|
+
for (const [index, item] of array.entries()) {
|
|
388
|
+
if (isRecord(item)) {
|
|
389
|
+
const itemTitle = titleFromValue(item) || `Item ${index + 1}`;
|
|
390
|
+
const itemDepth = label ? depth + 1 : depth;
|
|
391
|
+
lines.push(`${heading(itemDepth)} ${itemTitle}`, "");
|
|
392
|
+
renderJsonObject(item, lines, itemDepth);
|
|
393
|
+
}
|
|
394
|
+
else if (Array.isArray(item)) {
|
|
395
|
+
lines.push(`${indent(depth)}- Item ${index + 1}:`);
|
|
396
|
+
renderJsonArray(item, lines, depth + 1);
|
|
397
|
+
}
|
|
398
|
+
else {
|
|
399
|
+
lines.push(`${indent(depth)}- ${formatJsonScalar(item)}`);
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
lines.push("");
|
|
403
|
+
}
|
|
404
|
+
function formatJsonScalar(value) {
|
|
405
|
+
if (value === null)
|
|
406
|
+
return "null";
|
|
407
|
+
if (typeof value === "string")
|
|
408
|
+
return value;
|
|
409
|
+
if (typeof value === "number" || typeof value === "boolean")
|
|
410
|
+
return String(value);
|
|
411
|
+
return String(value);
|
|
412
|
+
}
|
|
413
|
+
function humanizeKey(key) {
|
|
414
|
+
return key
|
|
415
|
+
.replace(/[_-]+/g, " ")
|
|
416
|
+
.replace(/([a-z0-9])([A-Z])/g, "$1 $2")
|
|
417
|
+
.replace(/\s+/g, " ")
|
|
418
|
+
.trim()
|
|
419
|
+
.replace(/^./, (char) => char.toUpperCase());
|
|
420
|
+
}
|
|
421
|
+
function heading(depth) {
|
|
422
|
+
return "#".repeat(Math.min(depth + 2, 6));
|
|
423
|
+
}
|
|
424
|
+
function indent(depth) {
|
|
425
|
+
return " ".repeat(Math.max(0, depth));
|
|
426
|
+
}
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
import { mkdirSync, writeFileSync } from "node:fs";
|
|
2
|
+
import { copyFile } from "node:fs/promises";
|
|
3
|
+
import { extname, join } from "node:path";
|
|
4
|
+
import { createKnowledgeDocument, serializeKnowledgeDocument } from "./knowledge-document.js";
|
|
5
|
+
import { appendEvent } from "./metadata.js";
|
|
6
|
+
import { binaryExtractionFailureMessage, detectBinaryMagicBytes, extractUrlContent, fileExtractorFor, } from "./source-extractors.js";
|
|
7
|
+
import { exec, fmtDate, nextSourceId, readText, writeJson, } from "./utils.js";
|
|
8
|
+
import { assertWritableVault } from "./vault-format.js";
|
|
9
|
+
const URL_ORIGINAL_EXTENSIONS = new Set([".html", ".htm", ".md", ".pdf", ".txt", ".xml", ".json"]);
|
|
10
|
+
/** Capture a URL into a source packet. */
|
|
11
|
+
export async function captureUrl(pi, paths, url, signal) {
|
|
12
|
+
return captureSource(paths, urlCaptureSource(pi, url, signal));
|
|
13
|
+
}
|
|
14
|
+
/** Capture a local file into a source packet. */
|
|
15
|
+
export async function captureFile(pi, paths, filePath, signal) {
|
|
16
|
+
return captureSource(paths, fileCaptureSource(pi, filePath, signal));
|
|
17
|
+
}
|
|
18
|
+
/** Capture pasted text into a source packet. */
|
|
19
|
+
export function captureText(paths, text, title) {
|
|
20
|
+
return captureSourceSync(paths, textCaptureSource(text, title));
|
|
21
|
+
}
|
|
22
|
+
async function captureSource(paths, source) {
|
|
23
|
+
assertWritableVault(paths);
|
|
24
|
+
const packet = createSourcePacket(paths, source.needsOriginalDir);
|
|
25
|
+
await source.preserveOriginal?.(packet.packetPath);
|
|
26
|
+
const content = await source.extract();
|
|
27
|
+
return finalizeCapture(paths, packet, source, content);
|
|
28
|
+
}
|
|
29
|
+
function captureSourceSync(paths, source) {
|
|
30
|
+
assertWritableVault(paths);
|
|
31
|
+
const packet = createSourcePacket(paths, source.needsOriginalDir);
|
|
32
|
+
const content = source.extract();
|
|
33
|
+
return finalizeCapture(paths, packet, source, content);
|
|
34
|
+
}
|
|
35
|
+
function urlCaptureSource(pi, url, signal) {
|
|
36
|
+
return {
|
|
37
|
+
needsOriginalDir: true,
|
|
38
|
+
fallbackText: contentExtractionFailureMessage(url),
|
|
39
|
+
preserveOriginal: (packetPath) => preserveUrlOriginal(pi, packetPath, url, signal),
|
|
40
|
+
extract: () => extractUrlContent(pi, url, signal),
|
|
41
|
+
manifest: (content) => ({
|
|
42
|
+
title: content.title || url,
|
|
43
|
+
url,
|
|
44
|
+
format: "web",
|
|
45
|
+
}),
|
|
46
|
+
event: () => ({ url, format: "web" }),
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
function fileCaptureSource(pi, filePath, signal) {
|
|
50
|
+
const fileName = filePath.split("/").pop() || "unknown";
|
|
51
|
+
const extractor = fileExtractorFor(filePath);
|
|
52
|
+
const content = extractor.shouldReadText ? readText(filePath) : "";
|
|
53
|
+
return {
|
|
54
|
+
needsOriginalDir: true,
|
|
55
|
+
fallbackText: "",
|
|
56
|
+
preserveOriginal: (packetPath) => preserveFileOriginal(packetPath, filePath, fileName, content),
|
|
57
|
+
extract: async () => {
|
|
58
|
+
// Guard: if we hit the generic catch-all extractor, check for binary magic bytes first
|
|
59
|
+
if (extractor.format === "file") {
|
|
60
|
+
const binaryFormat = await detectBinaryMagicBytes(filePath);
|
|
61
|
+
if (binaryFormat) {
|
|
62
|
+
return {
|
|
63
|
+
extracted: binaryExtractionFailureMessage(binaryFormat),
|
|
64
|
+
extractor: "magicBytes",
|
|
65
|
+
extraction_status: "unsupported",
|
|
66
|
+
};
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
const extractedStr = await extractor.extract({ pi, filePath, content, signal });
|
|
70
|
+
const failed = extractedStr.includes("could not be converted");
|
|
71
|
+
return {
|
|
72
|
+
extracted: extractedStr,
|
|
73
|
+
extractor: extractor.extractorName ?? "passthrough",
|
|
74
|
+
extraction_status: (failed ? "failed" : "success"),
|
|
75
|
+
...(extractor.content_type ? { content_type: extractor.content_type } : {}),
|
|
76
|
+
};
|
|
77
|
+
},
|
|
78
|
+
manifest: () => ({
|
|
79
|
+
title: fileName,
|
|
80
|
+
file_path: filePath,
|
|
81
|
+
format: extractor.format,
|
|
82
|
+
}),
|
|
83
|
+
event: () => ({ file_path: filePath, format: extractor.format }),
|
|
84
|
+
};
|
|
85
|
+
}
|
|
86
|
+
function textCaptureSource(text, title) {
|
|
87
|
+
return {
|
|
88
|
+
needsOriginalDir: false,
|
|
89
|
+
fallbackText: "",
|
|
90
|
+
extract: () => ({ extracted: text }),
|
|
91
|
+
manifest: () => ({
|
|
92
|
+
title: title || `Pasted text — ${fmtDate()}`,
|
|
93
|
+
format: "text",
|
|
94
|
+
}),
|
|
95
|
+
event: () => ({ format: "text" }),
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
function createSourcePacket(paths, needsOriginalDir) {
|
|
99
|
+
const sourceId = nextSourceId(paths);
|
|
100
|
+
const packetPath = join(paths.rawSources, sourceId);
|
|
101
|
+
mkdirSync(packetPath, { recursive: true });
|
|
102
|
+
mkdirSync(join(packetPath, "attachments"), { recursive: true });
|
|
103
|
+
if (needsOriginalDir)
|
|
104
|
+
mkdirSync(join(packetPath, "original"), { recursive: true });
|
|
105
|
+
return { sourceId, packetPath };
|
|
106
|
+
}
|
|
107
|
+
function finalizeCapture(paths, packet, source, content) {
|
|
108
|
+
assertWritableVault(paths);
|
|
109
|
+
const extracted = content.extracted || source.fallbackText;
|
|
110
|
+
const manifest = {
|
|
111
|
+
id: packet.sourceId,
|
|
112
|
+
captured: fmtDate(),
|
|
113
|
+
packet_version: "1.0",
|
|
114
|
+
...source.manifest({ ...content, extracted }),
|
|
115
|
+
extractor: content.extractor ?? "passthrough",
|
|
116
|
+
extraction_status: content.extraction_status ?? "success",
|
|
117
|
+
...(content.content_type ? { content_type: content.content_type } : {}),
|
|
118
|
+
};
|
|
119
|
+
writeFileSync(join(packet.packetPath, "extracted.md"), extracted, "utf-8");
|
|
120
|
+
writeJson(join(packet.packetPath, "manifest.json"), manifest);
|
|
121
|
+
const sourcePagePath = join(paths.wiki, "sources", `${packet.sourceId}.md`);
|
|
122
|
+
writeFileSync(sourcePagePath, buildSourcePageSkeleton(manifest, extracted), "utf-8");
|
|
123
|
+
appendEvent(paths, {
|
|
124
|
+
kind: "capture",
|
|
125
|
+
source_id: packet.sourceId,
|
|
126
|
+
...source.event({ ...content, extracted }),
|
|
127
|
+
});
|
|
128
|
+
return {
|
|
129
|
+
sourceId: packet.sourceId,
|
|
130
|
+
packetPath: packet.packetPath,
|
|
131
|
+
sourcePagePath,
|
|
132
|
+
extracted,
|
|
133
|
+
};
|
|
134
|
+
}
|
|
135
|
+
async function preserveFileOriginal(packetPath, filePath, fileName, fallbackContent) {
|
|
136
|
+
try {
|
|
137
|
+
await copyFile(filePath, join(packetPath, "original", fileName));
|
|
138
|
+
}
|
|
139
|
+
catch {
|
|
140
|
+
// If copying fails, preserve whatever text content was available.
|
|
141
|
+
writeFileSync(join(packetPath, "original", fileName), fallbackContent, "utf-8");
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
async function preserveUrlOriginal(pi, packetPath, url, signal) {
|
|
145
|
+
const originalPath = join(packetPath, "original", originalFileNameForUrl(url));
|
|
146
|
+
try {
|
|
147
|
+
await exec(pi, "curl", ["-sL", "--max-time", "30", "-o", originalPath, url], {
|
|
148
|
+
signal,
|
|
149
|
+
timeout: 35_000,
|
|
150
|
+
});
|
|
151
|
+
}
|
|
152
|
+
catch {
|
|
153
|
+
// Preserve best-effort extraction behavior even when the original artifact cannot be saved.
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
function originalFileNameForUrl(url) {
|
|
157
|
+
try {
|
|
158
|
+
const parsed = new URL(url);
|
|
159
|
+
const ext = extname(parsed.pathname).toLowerCase();
|
|
160
|
+
if (URL_ORIGINAL_EXTENSIONS.has(ext))
|
|
161
|
+
return `source${ext}`;
|
|
162
|
+
}
|
|
163
|
+
catch {
|
|
164
|
+
const path = url.split(/[?#]/, 1)[0] ?? "";
|
|
165
|
+
const ext = extname(path).toLowerCase();
|
|
166
|
+
if (URL_ORIGINAL_EXTENSIONS.has(ext))
|
|
167
|
+
return `source${ext}`;
|
|
168
|
+
}
|
|
169
|
+
return "source.html";
|
|
170
|
+
}
|
|
171
|
+
function contentExtractionFailureMessage(source) {
|
|
172
|
+
return `_Content could not be extracted from ${source}_\n`;
|
|
173
|
+
}
|
|
174
|
+
/** Build a skeleton source page from manifest and extracted text. */
|
|
175
|
+
function buildSourcePageSkeleton(manifest, extracted) {
|
|
176
|
+
const id = String(manifest.id);
|
|
177
|
+
const title = String(manifest.title || id);
|
|
178
|
+
const url = manifest.url ? `\n> _Original: [${manifest.url}](${manifest.url})_` : "";
|
|
179
|
+
const format = String(manifest.format || "unknown");
|
|
180
|
+
const captured = String(manifest.captured || fmtDate());
|
|
181
|
+
// Generate a brief auto-summary (first 500 chars)
|
|
182
|
+
const preview = extracted
|
|
183
|
+
.replace(/[#*_`]/g, "")
|
|
184
|
+
.replace(/\s+/g, " ")
|
|
185
|
+
.trim()
|
|
186
|
+
.slice(0, 500);
|
|
187
|
+
const body = `# ${title}${url}
|
|
188
|
+
|
|
189
|
+
## Summary
|
|
190
|
+
|
|
191
|
+
[LLM: Replace with 2-3 paragraph summary of key content]
|
|
192
|
+
|
|
193
|
+
> _Auto-preview: ${preview}${extracted.length > 500 ? "..." : ""}_
|
|
194
|
+
|
|
195
|
+
## Key Takeaways
|
|
196
|
+
|
|
197
|
+
- [LLM: Most important point]
|
|
198
|
+
- [LLM: Second important point]
|
|
199
|
+
- [LLM: Third important point]
|
|
200
|
+
|
|
201
|
+
## Entities Mentioned
|
|
202
|
+
|
|
203
|
+
- [LLM: Add linked entities after review]
|
|
204
|
+
|
|
205
|
+
## Concepts Mentioned
|
|
206
|
+
|
|
207
|
+
- [LLM: Add linked concepts after review]
|
|
208
|
+
|
|
209
|
+
## Notable Quotes
|
|
210
|
+
|
|
211
|
+
> [LLM: Important quote] — attribution
|
|
212
|
+
|
|
213
|
+
## Source Packet
|
|
214
|
+
|
|
215
|
+
- **ID:** \`sources/${id}\`
|
|
216
|
+
- **Extracted:** \`raw/sources/${id}/extracted.md\`
|
|
217
|
+
- **Manifest:** \`raw/sources/${id}/manifest.json\`
|
|
218
|
+
`;
|
|
219
|
+
const doc = createKnowledgeDocument(`sources/${id}.md`, {
|
|
220
|
+
type: "source",
|
|
221
|
+
title,
|
|
222
|
+
format,
|
|
223
|
+
source_id: id,
|
|
224
|
+
raw_path: `raw/sources/${id}/extracted.md`,
|
|
225
|
+
captured,
|
|
226
|
+
status: "skeleton",
|
|
227
|
+
}, body);
|
|
228
|
+
return serializeKnowledgeDocument(doc);
|
|
229
|
+
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import { agentLoop, } from "@mariozechner/pi-agent-core";
|
|
2
|
+
/**
|
|
3
|
+
* Run a sub-agent loop to completion.
|
|
4
|
+
*
|
|
5
|
+
* Returns nothing useful directly — by design, results are collected by the
|
|
6
|
+
* `tools` the caller passes (their `execute` accumulates into caller-owned
|
|
7
|
+
* state). This keeps the runner generic across every background task type.
|
|
8
|
+
*/
|
|
9
|
+
export async function runSubAgent(args) {
|
|
10
|
+
const { model, apiKey, headers, systemPrompt, userPrompt, tools, maxTokens, signal } = args;
|
|
11
|
+
const text = userPrompt.trim();
|
|
12
|
+
if (!text)
|
|
13
|
+
return;
|
|
14
|
+
const prompts = [
|
|
15
|
+
{
|
|
16
|
+
role: "user",
|
|
17
|
+
content: [{ type: "text", text }],
|
|
18
|
+
timestamp: Date.now(),
|
|
19
|
+
},
|
|
20
|
+
];
|
|
21
|
+
const context = {
|
|
22
|
+
systemPrompt,
|
|
23
|
+
messages: [],
|
|
24
|
+
tools,
|
|
25
|
+
};
|
|
26
|
+
const reasoning = model.reasoning;
|
|
27
|
+
const config = {
|
|
28
|
+
model,
|
|
29
|
+
apiKey,
|
|
30
|
+
headers,
|
|
31
|
+
maxTokens: maxTokens ?? 4096,
|
|
32
|
+
convertToLlm: (msgs) => msgs,
|
|
33
|
+
toolExecution: "sequential",
|
|
34
|
+
...(reasoning ? { reasoning: "high" } : {}),
|
|
35
|
+
};
|
|
36
|
+
const stream = agentLoop(prompts, context, config, signal);
|
|
37
|
+
for await (const _event of stream) {
|
|
38
|
+
// Drain events; tool `execute` callbacks collect results caller-side.
|
|
39
|
+
}
|
|
40
|
+
await stream.result();
|
|
41
|
+
}
|