pi-quiver 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/fetch.ts ADDED
@@ -0,0 +1,541 @@
1
+ /**
2
+ * Fetch Extension
3
+ *
4
+ * Registers a `fetch` tool that retrieves URLs with context-safe output routing.
5
+ * HTML is extracted to structured Markdown via readability + turndown (boilerplate
6
+ * stripped, headings/lists/tables/code fences preserved). Binary content (images,
7
+ * PDFs, archives, etc.) is saved untouched to a temp file and only the path is
8
+ * returned. Text/Markdown/JSON over 32 KB or 1000 lines is written to a temp file
9
+ * with a 60-line preview; smaller content is returned inline. Parsable downloads
10
+ * are capped at 1 MB; binary downloads at 50 MB.
11
+ */
12
+
13
+ import { mkdirSync, writeFileSync, createWriteStream } from "node:fs";
14
+ import { rm } from "node:fs/promises";
15
+ import { tmpdir } from "node:os";
16
+ import { join } from "node:path";
17
+ import { createHash } from "node:crypto";
18
+ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
19
+ import { formatSize, keyHint } from "@earendil-works/pi-coding-agent";
20
+ import { Text } from "@earendil-works/pi-tui";
21
+ import { Type } from "@sinclair/typebox";
22
+ import { JSDOM } from "jsdom";
23
+ import { Readability } from "@mozilla/readability";
24
+ import TurndownService from "turndown";
25
+ import { gfm } from "turndown-plugin-gfm";
26
+
27
+ interface FetchToolDetails {
28
+ url?: string;
29
+ status?: number;
30
+ contentType?: string;
31
+ charset?: string;
32
+ bytes?: number;
33
+ truncated?: boolean;
34
+ category?: "binary" | "markdown" | "json" | "text";
35
+ spilled?: boolean;
36
+ file?: string;
37
+ lines?: number;
38
+ }
39
+
40
+ const PARSABLE_MAX_BYTES = 1_000_000; // text/markdown/json download ceiling
41
+ const BINARY_MAX_BYTES = 50_000_000; // file-destined download ceiling
42
+ const SNIFF_MAX_BYTES = 64_000; // classification window
43
+ const DEFAULT_TIMEOUT_MS = 20_000;
44
+ const INLINE_MAX_BYTES = 32_000;
45
+ const INLINE_MAX_LINES = 1_000;
46
+ const PREVIEW_LINES = 60;
47
+ const PREVIEW_MAX_BYTES = 4_000;
48
+ const FIREFOX_UA =
49
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 14.7; rv:135.0) Gecko/20100101 Firefox/135.0";
50
+ const DEFAULT_ACCEPT =
51
+ "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8";
52
+
53
+ function parseCharset(contentType: string): string {
54
+ const m = /charset\s*=\s*"?([^";\s]+)"?/i.exec(contentType);
55
+ return (m?.[1] ?? "utf-8").trim().toLowerCase();
56
+ }
57
+
58
+ function decodeBuffer(buf: Buffer, charset: string): string {
59
+ try {
60
+ return new TextDecoder(charset, { fatal: false }).decode(buf);
61
+ } catch {
62
+ return new TextDecoder("utf-8", { fatal: false }).decode(buf);
63
+ }
64
+ }
65
+
66
+ function buildPreview(body: string): string {
67
+ let preview = body.split("\n").slice(0, PREVIEW_LINES).join("\n");
68
+ if (preview.length > PREVIEW_MAX_BYTES) {
69
+ preview = `${preview.slice(0, PREVIEW_MAX_BYTES)}\n…[preview truncated]`;
70
+ }
71
+ return preview;
72
+ }
73
+
74
+ // --- Turndown singleton ---
75
+
76
+ const turndownService = new TurndownService({
77
+ headingStyle: "atx",
78
+ codeBlockStyle: "fenced",
79
+ bulletListMarker: "-",
80
+ });
81
+ turndownService.use(gfm);
82
+
83
+ // --- Content classification ---
84
+
85
+ function mimeType(contentType: string): string {
86
+ return contentType.split(";")[0].trim().toLowerCase();
87
+ }
88
+
89
+ const TEXT_ALLOWLIST: RegExp[] = [
90
+ /^text\//,
91
+ /^application\/(json|xml|xhtml\+xml|javascript)$/,
92
+ /\+json$/,
93
+ /\+xml$/,
94
+ ];
95
+ // octet-stream is intentionally absent — it falls to the NUL-sniff branch.
96
+ const KNOWN_BINARY: RegExp[] = [
97
+ /^audio\//,
98
+ /^video\//,
99
+ /^font\//,
100
+ /^application\/(pdf|zip|gzip|x-tar|x-7z-compressed|x-rar-compressed|wasm)$/,
101
+ ];
102
+
103
+ export function categorize(contentType: string, sniff: Buffer, raw: boolean): "binary" | "markdown" | "json" | "text" {
104
+ const mime = mimeType(contentType);
105
+ if (/^image\//.test(mime)) return "binary"; // includes image/svg+xml
106
+ const isText = TEXT_ALLOWLIST.some((re) => re.test(mime));
107
+ const isBinary = KNOWN_BINARY.some((re) => re.test(mime));
108
+ if (!isText && !isBinary) return sniff.includes(0) ? "binary" : "text";
109
+ if (isBinary && !isText) return "binary";
110
+ if (sniff.includes(0)) return "binary"; // NUL downgrade of a text candidate
111
+ if (raw) return "text"; // raw=true skips all transformations (markdown + JSON pretty-print)
112
+ if (mime === "text/html" || mime === "application/xhtml+xml") return "markdown";
113
+ if (mime === "application/json" || /\+json$/.test(mime)) return "json";
114
+ return "text";
115
+ }
116
+
117
+ export function htmlToMarkdown(html: string, url: string): string | null {
118
+ let doc: Document;
119
+ try {
120
+ doc = new JSDOM(html, { url }).window.document;
121
+ } catch {
122
+ return null;
123
+ }
124
+ let article: { title?: string | null; content?: string | null } | null = null;
125
+ try {
126
+ article = new Readability(doc).parse();
127
+ } catch {
128
+ return null;
129
+ }
130
+ if (!article?.content) return null;
131
+ let md: string;
132
+ try {
133
+ md = turndownService.turndown(article.content).trim();
134
+ } catch {
135
+ return null;
136
+ }
137
+ if (!md) return null;
138
+ if (article.title) md = `# ${article.title}\n\n${md}`;
139
+ return md;
140
+ }
141
+
142
+ export function prettyJson(text: string): string {
143
+ try {
144
+ return JSON.stringify(JSON.parse(text), null, 2);
145
+ } catch {
146
+ return text;
147
+ }
148
+ }
149
+
150
+ export function applyGate(body: string): { spill: boolean; bytes: number; lines: number } {
151
+ const bytes = Buffer.byteLength(body, "utf8");
152
+ const lines = body.length ? body.split("\n").length : 0;
153
+ const spill = body.length > 0 && (bytes > INLINE_MAX_BYTES || lines > INLINE_MAX_LINES);
154
+ return { spill, bytes, lines };
155
+ }
156
+
157
+ // --- Temp file helpers ---
158
+
159
+ function tempFilePath(url: string, ext: string): string {
160
+ const dir = join(tmpdir(), "pi-fetch");
161
+ mkdirSync(dir, { recursive: true });
162
+ let host = "page";
163
+ try {
164
+ host = new URL(url).hostname.replace(/[^a-z0-9.-]/gi, "_") || "page";
165
+ } catch {
166
+ // keep default
167
+ }
168
+ const hash = createHash("sha1").update(url).digest("hex").slice(0, 8);
169
+ const stamp = new Date().toISOString().replace(/[:.]/g, "-");
170
+ return join(dir, `${stamp}-${host}-${hash}.${ext}`);
171
+ }
172
+
173
+ function spillToFile(url: string, body: string, ext: string): string {
174
+ const file = tempFilePath(url, ext);
175
+ writeFileSync(file, body, "utf8");
176
+ return file;
177
+ }
178
+
179
+ function textExtension(category: "markdown" | "json" | "text", contentType: string): string {
180
+ if (category === "markdown") return "md";
181
+ if (category === "json") return "json";
182
+ return mimeType(contentType).includes("xml") ? "xml" : "txt";
183
+ }
184
+
185
+ const BINARY_EXT: Record<string, string> = {
186
+ "application/pdf": "pdf",
187
+ "application/zip": "zip",
188
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document": "docx",
189
+ "application/vnd.openxmlformats-officedocument.presentationml.presentation": "pptx",
190
+ "application/gzip": "gz",
191
+ "image/png": "png",
192
+ "image/jpeg": "jpg",
193
+ "image/gif": "gif",
194
+ "image/webp": "webp",
195
+ "image/svg+xml": "svg",
196
+ };
197
+
198
+ export function binaryExtension(contentType: string): string {
199
+ const mime = mimeType(contentType);
200
+ if (BINARY_EXT[mime]) return BINARY_EXT[mime];
201
+ const sub = (mime.split("/")[1] ?? "").replace(/^x-/, "").replace(/[^a-z0-9]+/g, "").slice(0, 8);
202
+ return sub || "bin";
203
+ }
204
+
205
+ // --- Streaming body collection ---
206
+
207
+ type Category = "binary" | "markdown" | "json" | "text";
208
+
209
+ interface CollectedBody {
210
+ category: Category;
211
+ buffer?: Buffer; // text/markdown/json (raw, pre-transform)
212
+ file?: string; // binary
213
+ bytes: number; // bytes kept (post-cap)
214
+ truncated: boolean;
215
+ }
216
+
217
+ function writeChunk(stream: ReturnType<typeof createWriteStream>, b: Buffer): Promise<void> {
218
+ return new Promise((resolve, reject) => {
219
+ stream.write(b, (err) => (err ? reject(err) : resolve()));
220
+ });
221
+ }
222
+
223
+ async function pumpToFile(
224
+ stream: ReturnType<typeof createWriteStream>,
225
+ reader: ReadableStreamDefaultReader<Uint8Array>,
226
+ prefix: Buffer,
227
+ exhausted: boolean,
228
+ ): Promise<{ bytes: number; truncated: boolean }> {
229
+ let bytes = 0;
230
+ let truncated = false;
231
+ let head = prefix;
232
+ if (head.length > BINARY_MAX_BYTES) {
233
+ head = head.subarray(0, BINARY_MAX_BYTES);
234
+ truncated = true;
235
+ }
236
+ await writeChunk(stream, head);
237
+ bytes += head.length;
238
+ while (!exhausted && !truncated) {
239
+ const { done, value } = await reader.read();
240
+ if (done) break;
241
+ let chunk = Buffer.from(value);
242
+ if (bytes + chunk.length > BINARY_MAX_BYTES) {
243
+ chunk = chunk.subarray(0, BINARY_MAX_BYTES - bytes);
244
+ truncated = true;
245
+ }
246
+ await writeChunk(stream, chunk);
247
+ bytes += chunk.length;
248
+ }
249
+ await new Promise<void>((resolve, reject) => stream.end((err?: Error | null) => (err ? reject(err) : resolve())));
250
+ return { bytes, truncated };
251
+ }
252
+
253
+ export async function collectBody(res: Response, contentType: string, raw: boolean): Promise<CollectedBody> {
254
+ const reader = res.body!.getReader();
255
+ const prefixParts: Buffer[] = [];
256
+ let prefixLen = 0;
257
+ let exhausted = false;
258
+ while (prefixLen < SNIFF_MAX_BYTES) {
259
+ const { done, value } = await reader.read();
260
+ if (done) {
261
+ exhausted = true;
262
+ break;
263
+ }
264
+ const chunk = Buffer.from(value);
265
+ prefixParts.push(chunk);
266
+ prefixLen += chunk.length;
267
+ }
268
+ const prefix = Buffer.concat(prefixParts);
269
+ const category = categorize(contentType, prefix.subarray(0, SNIFF_MAX_BYTES), raw);
270
+
271
+ if (category === "binary") {
272
+ const file = tempFilePath(res.url, binaryExtension(contentType));
273
+ const stream = createWriteStream(file);
274
+ try {
275
+ const { bytes, truncated } = await pumpToFile(stream, reader, prefix, exhausted);
276
+ if (truncated) await reader.cancel().catch(() => {});
277
+ return { category, file, bytes, truncated };
278
+ } catch (err) {
279
+ stream.destroy();
280
+ await rm(file, { force: true });
281
+ await reader.cancel().catch(() => {});
282
+ throw err;
283
+ }
284
+ }
285
+
286
+ const parts = [prefix];
287
+ let bytes = prefix.length;
288
+ let streamDone = exhausted;
289
+ while (!streamDone && bytes < PARSABLE_MAX_BYTES) {
290
+ const { done, value } = await reader.read();
291
+ if (done) { streamDone = true; break; }
292
+ const chunk = Buffer.from(value);
293
+ parts.push(chunk);
294
+ bytes += chunk.length;
295
+ }
296
+ let buffer = Buffer.concat(parts);
297
+ if (buffer.length > PARSABLE_MAX_BYTES) {
298
+ buffer = buffer.subarray(0, PARSABLE_MAX_BYTES);
299
+ }
300
+ const truncated = !streamDone;
301
+ if (truncated) await reader.cancel().catch(() => {});
302
+ return { category, buffer, bytes: buffer.length, truncated };
303
+ }
304
+
305
+ export default function fetchExtension(pi: ExtensionAPI) {
306
+ pi.registerTool({
307
+ name: "fetch",
308
+ label: "Fetch URL",
309
+ description:
310
+ "Fetch a URL over HTTP(S). HTML is extracted to Markdown (readability + turndown). Binary content (images, PDFs, archives) is saved untouched to a temp file and only a path is returned. Text/Markdown/JSON over 32KB or 1000 lines is written to a temp file with a 60-line preview; smaller content is returned inline. Parsable downloads are capped at 1MB, binary at 50MB.",
311
+ promptSnippet: "Fetch the contents of a URL",
312
+ promptGuidelines: [
313
+ "Use fetch when the user provides a URL or asks to read web content.",
314
+ "Binary responses return a file path only — pass that path to a tool that can process the bytes; do not expect inline content.",
315
+ "When the body is written to a file, grep it or read with offset/limit. Converted Markdown is grep-able by heading (^#).",
316
+ "Pass raw=true to skip Markdown/JSON conversion and get the decoded body as-is (still subject to the size gate).",
317
+ ],
318
+ parameters: Type.Object({
319
+ url: Type.String({ description: "Absolute http(s) URL" }),
320
+ method: Type.Optional(
321
+ Type.Union(
322
+ [Type.Literal("GET"), Type.Literal("HEAD"), Type.Literal("POST")],
323
+ { default: "GET" },
324
+ ),
325
+ ),
326
+ headers: Type.Optional(
327
+ Type.Record(Type.String(), Type.String(), {
328
+ description: "Extra request headers (override defaults like UA)",
329
+ }),
330
+ ),
331
+ body: Type.Optional(Type.String({ description: "Request body for POST" })),
332
+ raw: Type.Optional(
333
+ Type.Boolean({ description: "Skip HTML→Markdown and JSON pretty-printing; return the decoded body as-is" }),
334
+ ),
335
+ timeoutMs: Type.Optional(Type.Number({ default: DEFAULT_TIMEOUT_MS })),
336
+ }),
337
+ async execute(_toolCallId, params, signal) {
338
+ const url = new URL(params.url);
339
+ if (url.protocol !== "http:" && url.protocol !== "https:") {
340
+ throw new Error(`Unsupported protocol: ${url.protocol}`);
341
+ }
342
+
343
+ const headers = new Headers(params.headers ?? {});
344
+ if (!headers.has("user-agent")) headers.set("user-agent", FIREFOX_UA);
345
+ if (!headers.has("accept")) headers.set("accept", DEFAULT_ACCEPT);
346
+ if (!headers.has("accept-language"))
347
+ headers.set("accept-language", "en-US,en;q=0.5");
348
+
349
+ const controller = new AbortController();
350
+ const onAbort = () => controller.abort();
351
+ signal?.addEventListener("abort", onAbort);
352
+ const timer = setTimeout(
353
+ () => controller.abort(new Error("fetch timeout")),
354
+ params.timeoutMs ?? DEFAULT_TIMEOUT_MS,
355
+ );
356
+
357
+ try {
358
+ const res = await fetch(url, {
359
+ method: params.method ?? "GET",
360
+ headers,
361
+ body: params.body,
362
+ signal: controller.signal,
363
+ redirect: "follow",
364
+ });
365
+
366
+ const ct = res.headers.get("content-type") ?? "";
367
+ const charset = parseCharset(ct);
368
+ const header = [
369
+ `HTTP ${res.status} ${res.statusText}`,
370
+ `Content-Type: ${ct}`,
371
+ `Charset: ${charset}`,
372
+ ];
373
+
374
+ // HEAD or bodyless response: headers only.
375
+ if (!res.body || (params.method ?? "GET") === "HEAD") {
376
+ return {
377
+ content: [{ type: "text", text: [...header, "Length: 0 (no body)"].join("\n") }],
378
+ details: { url: res.url, status: res.status, contentType: ct, charset, bytes: 0 } as FetchToolDetails,
379
+ };
380
+ }
381
+
382
+ const collected = await collectBody(res, ct, params.raw ?? false);
383
+ const baseDetails: FetchToolDetails = {
384
+ url: res.url,
385
+ status: res.status,
386
+ contentType: ct,
387
+ charset,
388
+ bytes: collected.bytes,
389
+ truncated: collected.truncated,
390
+ category: collected.category,
391
+ };
392
+
393
+ if (collected.category === "binary") {
394
+ const note = collected.truncated ? " (truncated to 50MB)" : "";
395
+ return {
396
+ content: [{
397
+ type: "text",
398
+ text: [
399
+ ...header,
400
+ `Body: ${formatSize(collected.bytes)}${note} binary (${mimeType(ct) || "unknown"}) — saved untouched for processing`,
401
+ `Saved-To: ${collected.file}`,
402
+ "",
403
+ "Binary content is not decoded. Use the appropriate tool to process the file at the path above.",
404
+ ].join("\n"),
405
+ }],
406
+ details: { ...baseDetails, spilled: true, file: collected.file },
407
+ };
408
+ }
409
+
410
+ const decoded = decodeBuffer(collected.buffer!, charset);
411
+ let body: string;
412
+ let effectiveCategory: "markdown" | "json" | "text" = collected.category;
413
+ if (collected.category === "markdown") {
414
+ const md = htmlToMarkdown(decoded, res.url);
415
+ if (md !== null) {
416
+ body = md;
417
+ } else {
418
+ body = decoded; // raw HTML text fallback
419
+ effectiveCategory = "text";
420
+ }
421
+ } else if (collected.category === "json") {
422
+ body = prettyJson(decoded);
423
+ } else {
424
+ body = decoded;
425
+ }
426
+
427
+ baseDetails.category = effectiveCategory;
428
+ const truncNote = collected.truncated ? "\n[Note: source truncated at 1MB — content may be partial]" : "";
429
+ const lengthLine = `Length: ${collected.bytes}${collected.truncated ? " (truncated to 1MB)" : ""}`;
430
+ const { spill, bytes: bodyBytes, lines: lineCount } = applyGate(body);
431
+ baseDetails.lines = lineCount;
432
+
433
+ if (!spill) {
434
+ return {
435
+ content: [{ type: "text", text: [...header, lengthLine, "", body + truncNote].join("\n") }],
436
+ details: { ...baseDetails, spilled: false },
437
+ };
438
+ }
439
+
440
+ const ext = textExtension(effectiveCategory, ct);
441
+ const file = spillToFile(res.url, body, ext);
442
+ const grepHint = effectiveCategory === "markdown"
443
+ ? "Read slices of this file with the read tool (offset/limit) or grep it; do not read the whole file unless you must. Markdown is grep-able by heading (^#)."
444
+ : "Read slices of this file with the read tool (offset/limit) or grep it; do not read the whole file unless you must.";
445
+ return {
446
+ content: [{
447
+ type: "text",
448
+ text: [
449
+ ...header,
450
+ lengthLine,
451
+ `Body: ${formatSize(bodyBytes)} across ${lineCount} lines — written to file (too large to inline)`,
452
+ `Saved-To: ${file}`,
453
+ ...(collected.truncated ? ["[Note: source truncated at 1MB — content may be partial]"] : []),
454
+ "",
455
+ grepHint,
456
+ "",
457
+ `----- preview (first ${PREVIEW_LINES} lines) -----`,
458
+ buildPreview(body),
459
+ ].join("\n"),
460
+ }],
461
+ details: { ...baseDetails, spilled: true, file },
462
+ };
463
+ } finally {
464
+ clearTimeout(timer);
465
+ signal?.removeEventListener("abort", onAbort);
466
+ }
467
+ },
468
+
469
+ renderCall(args, theme, _context) {
470
+ let text = theme.fg("toolTitle", theme.bold("fetch "));
471
+ const method = args.method ?? "GET";
472
+ if (method !== "GET") {
473
+ text += theme.fg("warning", `${method} `);
474
+ }
475
+ text += theme.fg("accent", args.url ?? "");
476
+ if (args.raw) {
477
+ text += theme.fg("dim", " (raw)");
478
+ }
479
+ return new Text(text, 0, 0);
480
+ },
481
+
482
+ renderResult(result, { expanded, isPartial }, theme, context) {
483
+ if (isPartial) {
484
+ return new Text(theme.fg("warning", "Fetching..."), 0, 0);
485
+ }
486
+
487
+ const details = result.details as FetchToolDetails | undefined;
488
+ const content = result.content[0];
489
+ const fullText = content?.type === "text" ? content.text : "";
490
+
491
+ if (context.isError) {
492
+ const firstLine = fullText.split("\n")[0] || "fetch failed";
493
+ return new Text(theme.fg("error", firstLine), 0, 0);
494
+ }
495
+
496
+ const status = details?.status;
497
+ const statusStyled =
498
+ status === undefined
499
+ ? theme.fg("muted", "HTTP ?")
500
+ : status >= 200 && status < 300
501
+ ? theme.fg("success", `HTTP ${status}`)
502
+ : status >= 300 && status < 400
503
+ ? theme.fg("warning", `HTTP ${status}`)
504
+ : theme.fg("error", `HTTP ${status}`);
505
+
506
+ const sep = theme.fg("dim", " · ");
507
+ const parts: string[] = [statusStyled];
508
+ if (details?.contentType) {
509
+ parts.push(theme.fg("muted", details.contentType.split(";")[0].trim()));
510
+ }
511
+ if (typeof details?.bytes === "number") {
512
+ let sizeText = formatSize(details.bytes);
513
+ if (details.truncated) sizeText += " (truncated)";
514
+ parts.push(theme.fg("dim", sizeText));
515
+ }
516
+ if (details?.category === "binary") {
517
+ parts.push(theme.fg("warning", "binary → file"));
518
+ } else if (details?.spilled) {
519
+ parts.push(theme.fg("warning", "→ file"));
520
+ }
521
+
522
+ let text = parts.join(sep);
523
+
524
+ if (!expanded) {
525
+ const lineCount = details?.lines ?? (fullText ? fullText.split("\n").length : 0);
526
+ if (lineCount > 0) {
527
+ text += sep + theme.fg("dim", `${lineCount} lines`);
528
+ }
529
+ text += " " + theme.fg("dim", `(${keyHint("app.tools.expand", "to expand")})`);
530
+ return new Text(text, 0, 0);
531
+ }
532
+
533
+ if (fullText) {
534
+ for (const line of fullText.split("\n")) {
535
+ text += `\n${theme.fg("toolOutput", line)}`;
536
+ }
537
+ }
538
+ return new Text(text, 0, 0);
539
+ },
540
+ });
541
+ }
package/package.json ADDED
@@ -0,0 +1,89 @@
1
+ {
2
+ "name": "pi-quiver",
3
+ "version": "3.0.0",
4
+ "description": "Personal pack of Pi coding-agent extensions. First-party-quality tools that keep context clean. Ships a context-safe fetch tool and a doc_to_md PDF/DOCX/PPTX-to-Markdown converter.",
5
+ "author": "Jacek Juraszek",
6
+ "license": "MIT",
7
+ "type": "module",
8
+ "repository": {
9
+ "type": "git",
10
+ "url": "git+https://github.com/jjuraszek/pi-quiver.git"
11
+ },
12
+ "homepage": "https://github.com/jjuraszek/pi-quiver#readme",
13
+ "bugs": {
14
+ "url": "https://github.com/jjuraszek/pi-quiver/issues"
15
+ },
16
+ "keywords": [
17
+ "pi-package",
18
+ "pi",
19
+ "pi-coding-agent",
20
+ "fetch",
21
+ "markdown",
22
+ "pdf",
23
+ "cli"
24
+ ],
25
+ "engines": {
26
+ "node": ">=20"
27
+ },
28
+ "files": [
29
+ "fetch.ts",
30
+ "doc_to_md.ts",
31
+ "session-name.ts",
32
+ "sword-header.ts",
33
+ "extension-config.ts",
34
+ "scripts/pdf_to_md.py",
35
+ "types/**/*.d.ts",
36
+ "README.md",
37
+ "CHANGELOG.md"
38
+ ],
39
+ "scripts": {
40
+ "test": "node --test \"*.test.ts\"",
41
+ "typecheck": "npx -y tsc --noEmit --allowImportingTsExtensions --target es2022 --module nodenext --moduleResolution nodenext --strict --skipLibCheck --esModuleInterop --resolveJsonModule --lib es2022 --types node fetch.ts fetch.test.ts doc_to_md.ts doc_to_md.test.ts session-name.ts session-name.test.ts sword-header.ts sword-header.test.ts extension-config.ts types/turndown-plugin-gfm.d.ts",
42
+ "test:all": "npm run test && npm run typecheck"
43
+ },
44
+ "pi": {
45
+ "extensions": [
46
+ "./fetch.ts",
47
+ "./doc_to_md.ts",
48
+ "./session-name.ts",
49
+ "./sword-header.ts"
50
+ ]
51
+ },
52
+ "peerDependencies": {
53
+ "@earendil-works/pi-ai": "*",
54
+ "@earendil-works/pi-coding-agent": "*",
55
+ "@earendil-works/pi-tui": "*",
56
+ "@sinclair/typebox": "*"
57
+ },
58
+ "peerDependenciesMeta": {
59
+ "@earendil-works/pi-ai": {
60
+ "optional": true
61
+ },
62
+ "@earendil-works/pi-coding-agent": {
63
+ "optional": true
64
+ },
65
+ "@earendil-works/pi-tui": {
66
+ "optional": true
67
+ },
68
+ "@sinclair/typebox": {
69
+ "optional": true
70
+ }
71
+ },
72
+ "dependencies": {
73
+ "@mozilla/readability": "^0.6.0",
74
+ "jsdom": "^26.0.0",
75
+ "turndown": "^7.2.0",
76
+ "turndown-plugin-gfm": "^1.0.2",
77
+ "unpdf": "^1.6.0"
78
+ },
79
+ "devDependencies": {
80
+ "@earendil-works/pi-ai": "^0.80.3",
81
+ "@earendil-works/pi-coding-agent": "^0.80.3",
82
+ "@earendil-works/pi-tui": "^0.80.3",
83
+ "@sinclair/typebox": "^0.34.49",
84
+ "@types/jsdom": "^28.0.3",
85
+ "@types/node": "^26.1.0",
86
+ "@types/turndown": "^5.0.6",
87
+ "typescript": "^6.0.3"
88
+ }
89
+ }
@@ -0,0 +1,30 @@
1
+ #!/usr/bin/env python3
2
+ """Convert a PDF to Markdown via pymupdf4llm. One argv: the PDF path.
3
+ Writes Markdown to stdout (exit 0) or an error to stderr (non-zero).
4
+
5
+ pymupdf4llm/PyMuPDF emit diagnostic chatter to stdout; the conversion result
6
+ is the RETURN value of to_markdown(). We redirect library stdout into a sink so
7
+ stdout carries only the Markdown, honoring the caller's verbatim contract."""
8
+ import contextlib
9
+ import io
10
+ import sys
11
+
12
+
13
+ def main() -> int:
14
+ if len(sys.argv) != 2:
15
+ print("usage: pdf_to_md.py <pdf-path>", file=sys.stderr)
16
+ return 2
17
+ try:
18
+ sink = io.StringIO()
19
+ with contextlib.redirect_stdout(sink):
20
+ import pymupdf4llm
21
+ md = pymupdf4llm.to_markdown(sys.argv[1])
22
+ except Exception as exc: # noqa: BLE001 — surface any failure to the caller
23
+ print(f"pymupdf4llm conversion failed: {exc}", file=sys.stderr)
24
+ return 1
25
+ sys.stdout.write(md)
26
+ return 0
27
+
28
+
29
+ if __name__ == "__main__":
30
+ sys.exit(main())