pi-quiver 6.4.0 → 6.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,12 +1,39 @@
1
1
  #!/usr/bin/env node
2
-
3
- // bin/pi-quiver.ts
4
- import { existsSync as existsSync3, readFileSync, realpathSync } from "node:fs";
5
- import { homedir as homedir2 } from "node:os";
6
- import { join as join4, resolve as resolve3 } from "node:path";
7
- import { fileURLToPath as fileURLToPath2 } from "node:url";
2
+ var __defProp = Object.defineProperty;
3
+ var __getOwnPropNames = Object.getOwnPropertyNames;
4
+ var __esm = (fn, res, err) => function __init() {
5
+ if (err) throw err[0];
6
+ try {
7
+ return fn && (res = (0, fn[__getOwnPropNames(fn)[0]])(fn = 0)), res;
8
+ } catch (e) {
9
+ throw err = [e], e;
10
+ }
11
+ };
12
+ var __export = (target, all) => {
13
+ for (var name in all)
14
+ __defProp(target, name, { get: all[name], enumerable: true });
15
+ };
8
16
 
9
17
  // lib/fetch-core.ts
18
+ var fetch_core_exports = {};
19
+ __export(fetch_core_exports, {
20
+ DEFAULT_TIMEOUT_MS: () => DEFAULT_TIMEOUT_MS,
21
+ applyGate: () => applyGate,
22
+ binaryExtension: () => binaryExtension,
23
+ buildGhArgs: () => buildGhArgs,
24
+ buildGhLogArgs: () => buildGhLogArgs,
25
+ categorize: () => categorize,
26
+ classifyGitHubTarget: () => classifyGitHubTarget,
27
+ collectBody: () => collectBody,
28
+ executeGhRouting: () => executeGhRouting,
29
+ fetchUrl: () => fetchUrl,
30
+ formatSize: () => formatSize,
31
+ htmlToMarkdown: () => htmlToMarkdown,
32
+ htmlToMarkdownRaw: () => htmlToMarkdownRaw,
33
+ planGhRouting: () => planGhRouting,
34
+ prettyJson: () => prettyJson,
35
+ runGh: () => runGh
36
+ });
10
37
  import { mkdirSync, writeFileSync, createWriteStream } from "node:fs";
11
38
  import { rm } from "node:fs/promises";
12
39
  import { tmpdir } from "node:os";
@@ -27,24 +54,6 @@ function formatSize(bytes) {
27
54
  return `${(bytes / (1024 * 1024)).toFixed(1)}MB`;
28
55
  }
29
56
  }
30
- var RESERVED_OWNERS = /* @__PURE__ */ new Set([
31
- "orgs",
32
- "users",
33
- "sponsors",
34
- "topics",
35
- "marketplace",
36
- "apps",
37
- "collections",
38
- "stars",
39
- "settings",
40
- "notifications",
41
- "codespaces",
42
- "features",
43
- "trending",
44
- "security",
45
- "customer-stories"
46
- ]);
47
- var GH_NAME = /^[A-Za-z0-9._-]+$/;
48
57
  function classifyGitHubTarget(url) {
49
58
  const host = url.hostname.toLowerCase();
50
59
  if (host !== "github.com" && host !== "www.github.com") return null;
@@ -82,22 +91,6 @@ function buildGhLogArgs(target) {
82
91
  if (target.kind === "job") return ["run", "view", "--job", target.jobId, "--log-failed", "--repo", target.slug];
83
92
  return null;
84
93
  }
85
- var GH_MAX_BUFFER = 1e7;
86
- var execFileAsync = promisify(execFile);
87
- var runGh = async (args, timeoutMs, signal) => {
88
- try {
89
- const { stdout } = await execFileAsync("gh", args, {
90
- timeout: timeoutMs,
91
- signal,
92
- maxBuffer: GH_MAX_BUFFER,
93
- encoding: "utf8"
94
- });
95
- if (!stdout.trim()) return { ok: false };
96
- return { ok: true, stdout };
97
- } catch {
98
- return { ok: false };
99
- }
100
- };
101
94
  function planGhRouting(params, url) {
102
95
  if (params.raw) return null;
103
96
  if ((params.method ?? "GET") !== "GET") return null;
@@ -176,16 +169,6 @@ async function executeGhRouting(params, url, signal, runner = runGh) {
176
169
  }
177
170
  return renderGhResult(target, gh.stdout, failedLogs);
178
171
  }
179
- var PARSABLE_MAX_BYTES = 1e6;
180
- var BINARY_MAX_BYTES = 5e7;
181
- var SNIFF_MAX_BYTES = 64e3;
182
- var DEFAULT_TIMEOUT_MS = 2e4;
183
- var INLINE_MAX_BYTES = 32e3;
184
- var INLINE_MAX_LINES = 1e3;
185
- var PREVIEW_LINES = 60;
186
- var PREVIEW_MAX_BYTES = 4e3;
187
- var FIREFOX_UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 14.7; rv:135.0) Gecko/20100101 Firefox/135.0";
188
- var DEFAULT_ACCEPT = "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8";
189
172
  function parseCharset(contentType) {
190
173
  const m = /charset\s*=\s*"?([^";\s]+)"?/i.exec(contentType);
191
174
  return (m?.[1] ?? "utf-8").trim().toLowerCase();
@@ -205,27 +188,9 @@ function buildPreview(body) {
205
188
  }
206
189
  return preview;
207
190
  }
208
- var turndownService = new TurndownService({
209
- headingStyle: "atx",
210
- codeBlockStyle: "fenced",
211
- bulletListMarker: "-"
212
- });
213
- turndownService.use(gfm);
214
191
  function mimeType(contentType) {
215
192
  return contentType.split(";")[0].trim().toLowerCase();
216
193
  }
217
- var TEXT_ALLOWLIST = [
218
- /^text\//,
219
- /^application\/(json|xml|xhtml\+xml|javascript)$/,
220
- /\+json$/,
221
- /\+xml$/
222
- ];
223
- var KNOWN_BINARY = [
224
- /^audio\//,
225
- /^video\//,
226
- /^font\//,
227
- /^application\/(pdf|zip|gzip|x-tar|x-7z-compressed|x-rar-compressed|wasm)$/
228
- ];
229
194
  function categorize(contentType, sniff, raw) {
230
195
  const mime = mimeType(contentType);
231
196
  if (/^image\//.test(mime)) return "binary";
@@ -265,6 +230,9 @@ function htmlToMarkdown(html, url) {
265
230
  ${md}`;
266
231
  return md;
267
232
  }
233
+ function htmlToMarkdownRaw(html) {
234
+ return turndownService.turndown(html).trim();
235
+ }
268
236
  function prettyJson(text) {
269
237
  try {
270
238
  return JSON.stringify(JSON.parse(text), null, 2);
@@ -300,18 +268,6 @@ function textExtension(category, contentType) {
300
268
  if (category === "json") return "json";
301
269
  return mimeType(contentType).includes("xml") ? "xml" : "txt";
302
270
  }
303
- var BINARY_EXT = {
304
- "application/pdf": "pdf",
305
- "application/zip": "zip",
306
- "application/vnd.openxmlformats-officedocument.wordprocessingml.document": "docx",
307
- "application/vnd.openxmlformats-officedocument.presentationml.presentation": "pptx",
308
- "application/gzip": "gz",
309
- "image/png": "png",
310
- "image/jpeg": "jpg",
311
- "image/gif": "gif",
312
- "image/webp": "webp",
313
- "image/svg+xml": "svg"
314
- };
315
271
  function binaryExtension(contentType) {
316
272
  const mime = mimeType(contentType);
317
273
  if (BINARY_EXT[mime]) return BINARY_EXT[mime];
@@ -515,12 +471,99 @@ async function fetchUrl(opts) {
515
471
  opts.signal?.removeEventListener("abort", onAbort);
516
472
  }
517
473
  }
474
+ var RESERVED_OWNERS, GH_NAME, GH_MAX_BUFFER, execFileAsync, runGh, PARSABLE_MAX_BYTES, BINARY_MAX_BYTES, SNIFF_MAX_BYTES, DEFAULT_TIMEOUT_MS, INLINE_MAX_BYTES, INLINE_MAX_LINES, PREVIEW_LINES, PREVIEW_MAX_BYTES, FIREFOX_UA, DEFAULT_ACCEPT, turndownService, TEXT_ALLOWLIST, KNOWN_BINARY, BINARY_EXT;
475
+ var init_fetch_core = __esm({
476
+ "lib/fetch-core.ts"() {
477
+ "use strict";
478
+ RESERVED_OWNERS = /* @__PURE__ */ new Set([
479
+ "orgs",
480
+ "users",
481
+ "sponsors",
482
+ "topics",
483
+ "marketplace",
484
+ "apps",
485
+ "collections",
486
+ "stars",
487
+ "settings",
488
+ "notifications",
489
+ "codespaces",
490
+ "features",
491
+ "trending",
492
+ "security",
493
+ "customer-stories"
494
+ ]);
495
+ GH_NAME = /^[A-Za-z0-9._-]+$/;
496
+ GH_MAX_BUFFER = 1e7;
497
+ execFileAsync = promisify(execFile);
498
+ runGh = async (args, timeoutMs, signal) => {
499
+ try {
500
+ const { stdout } = await execFileAsync("gh", args, {
501
+ timeout: timeoutMs,
502
+ signal,
503
+ maxBuffer: GH_MAX_BUFFER,
504
+ encoding: "utf8"
505
+ });
506
+ if (!stdout.trim()) return { ok: false };
507
+ return { ok: true, stdout };
508
+ } catch {
509
+ return { ok: false };
510
+ }
511
+ };
512
+ PARSABLE_MAX_BYTES = 1e6;
513
+ BINARY_MAX_BYTES = 5e7;
514
+ SNIFF_MAX_BYTES = 64e3;
515
+ DEFAULT_TIMEOUT_MS = 2e4;
516
+ INLINE_MAX_BYTES = 32e3;
517
+ INLINE_MAX_LINES = 1e3;
518
+ PREVIEW_LINES = 60;
519
+ PREVIEW_MAX_BYTES = 4e3;
520
+ FIREFOX_UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 14.7; rv:135.0) Gecko/20100101 Firefox/135.0";
521
+ DEFAULT_ACCEPT = "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8";
522
+ turndownService = new TurndownService({
523
+ headingStyle: "atx",
524
+ codeBlockStyle: "fenced",
525
+ bulletListMarker: "-"
526
+ });
527
+ turndownService.use(gfm);
528
+ TEXT_ALLOWLIST = [
529
+ /^text\//,
530
+ /^application\/(json|xml|xhtml\+xml|javascript)$/,
531
+ /\+json$/,
532
+ /\+xml$/
533
+ ];
534
+ KNOWN_BINARY = [
535
+ /^audio\//,
536
+ /^video\//,
537
+ /^font\//,
538
+ /^application\/(pdf|zip|gzip|x-tar|x-7z-compressed|x-rar-compressed|wasm)$/
539
+ ];
540
+ BINARY_EXT = {
541
+ "application/pdf": "pdf",
542
+ "application/zip": "zip",
543
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document": "docx",
544
+ "application/vnd.openxmlformats-officedocument.presentationml.presentation": "pptx",
545
+ "application/gzip": "gz",
546
+ "image/png": "png",
547
+ "image/jpeg": "jpg",
548
+ "image/gif": "gif",
549
+ "image/webp": "webp",
550
+ "image/svg+xml": "svg"
551
+ };
552
+ }
553
+ });
554
+
555
+ // bin/pi-quiver.ts
556
+ init_fetch_core();
557
+ import { existsSync as existsSync3, readFileSync as readFileSync2, realpathSync } from "node:fs";
558
+ import { homedir as homedir2 } from "node:os";
559
+ import { join as join4, resolve as resolve3 } from "node:path";
560
+ import { fileURLToPath as fileURLToPath2 } from "node:url";
518
561
 
519
562
  // lib/doc-to-md-core.ts
520
- import { existsSync as existsSync2, mkdtempSync, renameSync as renameSync2, rmSync as rmSync2, statSync as statSync2 } from "node:fs";
563
+ import { copyFileSync, existsSync as existsSync2, mkdirSync as mkdirSync3, mkdtempSync, readFileSync, readdirSync as readdirSync2, renameSync as renameSync2, rmSync as rmSync2, statSync as statSync2, writeFileSync as writeFileSync3 } from "node:fs";
521
564
  import { spawn } from "node:child_process";
522
565
  import { homedir, tmpdir as tmpdir3 } from "node:os";
523
- import { basename, dirname, extname as extname3, join as join3, resolve as resolve2 } from "node:path";
566
+ import { basename, dirname, extname as extname3, isAbsolute, join as join3, resolve as resolve2 } from "node:path";
524
567
  import { fileURLToPath, pathToFileURL } from "node:url";
525
568
 
526
569
  // lib/doc-to-md-bundle.ts
@@ -634,9 +677,10 @@ function rewriteLinks(md, sourceMap) {
634
677
  return dest ? whole.split(target).join(dest) : whole;
635
678
  });
636
679
  }
637
- function validateImageLinks(md, manifest, csvManifest = /* @__PURE__ */ new Set()) {
680
+ function validateImageLinks(md, manifest, csvManifest = /* @__PURE__ */ new Set(), html = false) {
638
681
  for (const m of md.matchAll(IMG_LINK_RE)) {
639
682
  const target = imageTarget(m);
683
+ if (html && !target.replace(/^\.\//, "").startsWith("p1/")) continue;
640
684
  if (!target.startsWith("images/") || !manifest.has(target.slice("images/".length))) throw new Error(`unexpected image reference in output: ${target}`);
641
685
  }
642
686
  for (const m of md.matchAll(SHEET_LINK_RE)) {
@@ -697,11 +741,46 @@ function compactRanges(nums, maxEntries = 20) {
697
741
  if (parts.length <= maxEntries) return parts.join(", ");
698
742
  return `${parts.slice(0, maxEntries).join(", ")} (+${parts.length - maxEntries} more)`;
699
743
  }
744
+ var INSTALL_HINT = "install Tesseract (see doc/doc-to-md.md), then rerun with ocr=true";
745
+ var BARE_REASONS = ["fallback tier", "no Python backend"];
746
+ function ocrLine(ocr, type) {
747
+ const image = type === "image";
748
+ const rerun = ocr.tesseract ? "rerun with ocr=true" : INSTALL_HINT;
749
+ switch (ocr.status) {
750
+ case "off":
751
+ return image ? `OCR: off; ${rerun}` : `OCR: off - ${ocr.textless.length} page(s) without a text layer; ${rerun}`;
752
+ case "unavailable": {
753
+ const reason = ocr.reason ?? "";
754
+ const bare = BARE_REASONS.includes(reason) || reason.startsWith("OCR child failed: ");
755
+ return `OCR: unavailable - ${reason}${bare ? "" : " (install Tesseract; see doc/doc-to-md.md)"}`;
756
+ }
757
+ case "skipped":
758
+ return "OCR: skipped - image too small";
759
+ case "ran": {
760
+ const clauses = [`OCR: ${ocr.pages.length} ${image ? "image" : "page(s)"} (${ocr.lang})`];
761
+ if (ocr.noText.length) clauses.push(`${ocr.noText.length} returned no text`);
762
+ if (ocr.ocrFailed.length) clauses.push(`${ocr.ocrFailed.length} failed and were converted without OCR`);
763
+ if (ocr.budgetStopped.length) {
764
+ const r = compactRanges(ocr.budgetStopped, Number.POSITIVE_INFINITY).replaceAll(", ", ",");
765
+ clauses.push(`time budget reached for pages=${r}; rerun with pages=${r} or raise primaryTimeoutMs`);
766
+ }
767
+ return clauses.join("; ");
768
+ }
769
+ }
770
+ }
771
+ var PAGE_MARKER_RE = /^--- end of page\.page_number=(\d+) ---$/;
700
772
  function scanOutline(md, max) {
701
773
  const entries = [];
774
+ let pending = [];
702
775
  let total = 0, inFence = false;
703
776
  md.split("\n").forEach((raw, i) => {
704
777
  const line = raw.replace(/\r$/, "");
778
+ const marker = line.match(PAGE_MARKER_RE);
779
+ if (marker) {
780
+ for (const e of pending) e.page = Number(marker[1]);
781
+ pending = [];
782
+ return;
783
+ }
705
784
  if (/^\s*(```|~~~)/.test(line)) {
706
785
  inFence = !inFence;
707
786
  return;
@@ -710,29 +789,43 @@ function scanOutline(md, max) {
710
789
  const m = line.match(/^(#{1,6}) (.*)$/);
711
790
  if (!m) return;
712
791
  total++;
713
- if (entries.length < max) entries.push({ line: i + 1, level: m[1].length, title: trunc(m[2].trim(), TITLE_MAX) });
792
+ if (entries.length < max) {
793
+ const e = { line: i + 1, level: m[1].length, title: trunc(m[2].trim(), TITLE_MAX), page: null };
794
+ entries.push(e);
795
+ pending.push(e);
796
+ }
714
797
  });
715
798
  return { entries, total };
716
799
  }
717
800
  function outlineLines(entries, total) {
801
+ if (total === 0) return ["Outline: none"];
718
802
  if (entries.length === 0) return [];
719
- const width = Math.max(5, Math.max(...entries.map((e) => `L${e.line}`.length)) + 2);
720
- const out = ["Outline:", ...entries.map((e) => ` ${`L${e.line}`.padEnd(width)}${"#".repeat(e.level)} ${trunc(e.title, TITLE_MAX)}`)];
803
+ const lineWidth = Math.max(3, ...entries.map((e) => `L${e.line}`.length)) + 2;
804
+ const paged = entries.filter((e) => e.page !== null);
805
+ const pageWidth = paged.length ? Math.max(...paged.map((e) => `p${e.page}`.length)) + 2 : 0;
806
+ const out = ["Outline:", ...entries.map((e) => ` ${`L${e.line}`.padEnd(lineWidth)}${pageWidth ? (e.page === null ? "" : `p${e.page}`).padEnd(pageWidth) : ""}${"#".repeat(e.level)} ${trunc(e.title, TITLE_MAX)}`)];
721
807
  if (total > entries.length) out.push(` (+${total - entries.length} more)`);
722
808
  return out;
723
809
  }
810
+ function pageCountLabel(h) {
811
+ const n = h.pageCount ?? "?";
812
+ if (h.type !== "docx") return String(n);
813
+ if (h.tier === "docx") return `${n} (${(h.explicitBreaks ?? 0) > 0 ? "explicit page breaks, not printed pages" : "no explicit page breaks"})`;
814
+ return `${n} (LibreOffice pagination)`;
815
+ }
724
816
  function formatHandle(h) {
725
817
  const lines = [`Saved-To: ${h.savedTo}`];
726
818
  if (h.imagesDir && h.imageCount > 0) lines.push(`Images-Dir: ${h.imagesDir}`);
727
819
  if (h.sheetsDir) lines.push(`Sheets-Dir: ${h.sheetsDir}`);
728
820
  lines.push(`Type: ${h.type} Engine: ${h.engine} Tier: ${h.tier}`);
729
- lines.push(`Page-Count: ${h.pageCount ?? "?"} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize2(h.bytes)} / ${h.lines} lines`);
821
+ lines.push(`Page-Count: ${pageCountLabel(h)} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize2(h.bytes)} / ${h.lines} lines`);
730
822
  if (h.degraded) lines.push(`Degraded: ${h.degraded}`);
731
823
  if (h.fallbackReason) lines.push(`Fallback-Reason: ${h.fallbackReason}`);
732
824
  const fe = [];
733
825
  if (h.failedPages.length) fe.push(`Failed-Pages: ${compactRanges(h.failedPages)}`);
734
826
  if (h.emptyPages.length) fe.push(`Empty-Pages: ${compactRanges(h.emptyPages)}`);
735
827
  if (fe.length) lines.push(fe.join(" "));
828
+ if (h.ocr) lines.push(ocrLine(h.ocr, h.type));
736
829
  h.notes.slice(0, NOTE_MAX_LINES).forEach((n, i) => lines.push(`${i === 0 ? "Notes: " : " "}${trunc(n, NOTE_MAX_CHARS)}`));
737
830
  lines.push(...outlineLines(h.outline, h.outlineTotal));
738
831
  return lines.join("\n");
@@ -752,9 +845,9 @@ function formatInfoHandle(i, max) {
752
845
  const meta = Object.entries(i.metadata).filter(([, v]) => v).map(([k, v]) => `${k[0].toUpperCase()}${k.slice(1)}: ${trunc(v, META_MAX_CHARS)}`);
753
846
  if (meta.length) lines.push(meta.join(" "));
754
847
  if (i.toc.length) {
755
- lines.push("TOC:", ...i.toc.slice(0, max).map((t) => ` L${t.level} ${trunc(t.title, TITLE_MAX)} (p${t.page})`));
848
+ lines.push("TOC:", ...i.toc.slice(0, max).map((t) => ` L${t.level} ${trunc(t.title, TITLE_MAX)} (p${t.page ?? "?"})`));
756
849
  if (i.tocTotal > max) lines.push(` (+${i.tocTotal - max} more)`);
757
- }
850
+ } else if (i.type === "docx") lines.push("TOC: none (no heading styles found)");
758
851
  return lines.join("\n");
759
852
  }
760
853
 
@@ -764,14 +857,16 @@ var UsageError = class extends Error {
764
857
  };
765
858
  var MIN_PYMUPDF4LLM = "1.27.0";
766
859
  var VERSION_RE = /^\d+(\.\d+)*$/;
860
+ var OCR_LANGUAGE_RE = /^[a-z][a-z0-9_]*(\+[a-z][a-z0-9_]*)*$/;
861
+ var IMAGE_EXTS = [".png", ".jpg", ".jpeg", ".tif", ".tiff", ".bmp", ".gif"];
767
862
  var DOC_TO_MD_OPTIONS = [
768
- { key: "path", flag: null, type: "string", default: null, settable: false, help: "Local .pdf .docx .pptx .xlsx .xls file" },
863
+ { key: "path", flag: null, type: "string", default: null, settable: false, help: "Local .pdf .docx .pptx .xlsx .xls .html .htm .png .jpg .jpeg .tif .tiff .bmp .gif file" },
769
864
  { key: "info", flag: "--info", type: "bool", default: false, settable: false, help: "Inspect only (page count, metadata, TOC or sheet inventory); no bundle" },
770
- { key: "pages", flag: "--pages", type: "pages", default: null, settable: false, help: 'Inclusive 1-based pages, e.g. "12-15" or "3,7,10-12" (PDF/DOCX/PPTX only); default all' },
865
+ { key: "pages", flag: "--pages", type: "pages", default: null, settable: false, help: 'Inclusive 1-based pages, e.g. "12-15" or "3,7,10-12" (PDF/DOCX/PPTX only); default all. DOCX: selects explicit-page-break segments; rejected when the file has none' },
771
866
  { key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir" },
772
867
  { key: "overwrite", flag: "--overwrite", type: "bool", default: false, settable: false, help: "Replace an existing completed <stem>.md bundle" },
773
- { key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 6e4, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier; also the unpdf tier" },
774
- { key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier; also info on PDF and Excel rendered views" },
868
+ { key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 6e4, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier and DOCX child (docx mode); also the unpdf tier" },
869
+ { key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier (including DOCX LibreOffice fallback); also PDF and DOCX info and Excel rendered views" },
775
870
  { key: "sofficeTimeoutMs", flag: "--soffice-timeout", type: "int", default: 12e4, settable: true, env: "PI_DOC_TO_MD_SOFFICE_TIMEOUT_MS", help: "DOCX/PPTX -> PDF via LibreOffice; also Excel rendered views" },
776
871
  { key: "excelTimeoutMs", flag: "--excel-timeout", type: "int", default: 6e4, settable: true, help: "Excel child (both openpyxl loads); also info on Excel" },
777
872
  { key: "warmTimeoutMs", flag: "--warm-timeout", type: "int", default: 12e4, settable: true, env: "PI_DOC_TO_MD_WARM_TIMEOUT_MS", help: "Absolute backend discovery/bootstrap deadline (first call per process)" },
@@ -779,7 +874,9 @@ var DOC_TO_MD_OPTIONS = [
779
874
  { key: "imageDpi", flag: "--image-dpi", type: "int", default: 150, settable: true, help: "Render DPI for page images and Excel rendered views" },
780
875
  { key: "imageFormat", flag: "--image-format", type: "enum", default: "png", settable: true, enumValues: ["png", "jpg"], help: "Rendered image format (embedded images keep their native extension)" },
781
876
  { key: "maxOutputBytes", flag: "--max-output-bytes", type: "int", default: 2e7, settable: true, help: "Child stdout cap in bytes" },
782
- { key: "outlineMaxEntries", flag: "--outline-max-entries", type: "int", default: 40, settable: true, help: "Heading outline / TOC / sheet inventory cap in the handle" }
877
+ { key: "outlineMaxEntries", flag: "--outline-max-entries", type: "int", default: 40, settable: true, help: "Heading outline / TOC / sheet inventory cap in the handle" },
878
+ { key: "ocr", flag: "--ocr", type: "bool", default: false, settable: true, help: "Run OCR on pages without a text layer and on image inputs when Tesseract language data is installed; off by default (--no-ocr turns a settings-level true off)" },
879
+ { key: "ocrLanguage", flag: "--ocr-language", type: "lang", default: "eng", settable: true, help: "Tesseract language code(s), +-joined, e.g. deu+eng" }
783
880
  ];
784
881
  var TUNABLE_DEFAULTS = Object.fromEntries(
785
882
  DOC_TO_MD_OPTIONS.filter((d) => d.settable).map((d) => [d.key, d.default])
@@ -806,6 +903,8 @@ function coerceValue(d, raw, fromString = false) {
806
903
  if (typeof raw !== "string" || !VERSION_RE.test(raw)) return { ok: false, reason: "must be digits and dots" };
807
904
  if (!versionAtLeast(raw, MIN_PYMUPDF4LLM)) return { ok: false, reason: `must be >= ${MIN_PYMUPDF4LLM}` };
808
905
  return { ok: true, value: raw };
906
+ case "lang":
907
+ return typeof raw === "string" && OCR_LANGUAGE_RE.test(raw) ? { ok: true, value: raw } : { ok: false, reason: "must be Tesseract language codes joined by + (e.g. eng, deu+eng)" };
809
908
  case "string":
810
909
  return typeof raw === "string" && raw.length > 0 ? { ok: true, value: raw } : { ok: false, reason: "must be a non-empty string" };
811
910
  case "pages":
@@ -844,10 +943,10 @@ function sanitizeStem(base) {
844
943
  const s = base.replace(/[^A-Za-z0-9._-]+/g, "_");
845
944
  return s.length ? s : "document";
846
945
  }
847
- var SUPPORTED = { ".pdf": "pdf", ".docx": "docx", ".pptx": "pptx", ".xlsx": "xlsx", ".xls": "xls" };
946
+ var SUPPORTED = { ".pdf": "pdf", ".docx": "docx", ".pptx": "pptx", ".xlsx": "xlsx", ".xls": "xls", ".html": "html", ".htm": "html", ...Object.fromEntries(IMAGE_EXTS.map((ext) => [ext, "image"])) };
848
947
  function classifyInput(filePath) {
849
948
  const t = SUPPORTED[extname2(filePath).toLowerCase()];
850
- if (!t) throw new Error(`Unsupported file type "${extname2(filePath) || "(none)"}"; supported: .pdf, .docx, .pptx, .xlsx, .xls`);
949
+ if (!t) throw new Error(`Unsupported file type "${extname2(filePath) || "(none)"}"; supported: ${Object.keys(SUPPORTED).join(", ")}`);
851
950
  return t;
852
951
  }
853
952
  function resolveOptions(perCall, settings, env) {
@@ -892,21 +991,22 @@ function renderHelp() {
892
991
  }
893
992
 
894
993
  // lib/doc-to-md-core.ts
895
- var PACKAGE_PINS = { pymupdf4llm: TUNABLE_DEFAULTS.pymupdfVersion, openpyxl: "3.1.5", xlrd: "2.0.2", pillow: "12.3.0" };
994
+ var PACKAGE_PINS = { pymupdf4llm: TUNABLE_DEFAULTS.pymupdfVersion, openpyxl: "3.1.5", xlrd: "2.0.2", pillow: "12.3.0", mammoth: "1.13.0", markdownify: "1.2.3", "python-docx": "1.2.0" };
896
995
  var KILL_GRACE_MS = 2e3;
897
- var VENV_DIR_NAME = "doc-to-md-venv-v2";
898
- var LEGACY_VENV_DIR_NAME = "pymupdf-venv";
996
+ var VENV_DIR_NAME = "doc-to-md-venv-v3";
997
+ var LEGACY_VENV_DIR_NAMES = ["pymupdf-venv", "doc-to-md-venv-v2"];
899
998
  var STDERR_CAP = 1e6;
900
999
  var OUTPUT_MAX_BYTES = 2e7;
901
1000
  var EXCEL_PDF_FILTER = 'pdf:calc_pdf_Export:{"SinglePageSheets":{"type":"boolean","value":"true"}}';
1001
+ var pinSpecs = (cfg) => [`pymupdf4llm==${cfg.pymupdfVersion}`, `openpyxl==${PACKAGE_PINS.openpyxl}`, `xlrd==${PACKAGE_PINS.xlrd}`, `pillow==${PACKAGE_PINS.pillow}`, `mammoth==${PACKAGE_PINS.mammoth}`, `markdownify==${PACKAGE_PINS.markdownify}`, `python-docx==${PACKAGE_PINS["python-docx"]}`];
902
1002
  function withArgs(cfg) {
903
- return ["--with", `pymupdf4llm==${cfg.pymupdfVersion}`, "--with", `openpyxl==${PACKAGE_PINS.openpyxl}`, "--with", `xlrd==${PACKAGE_PINS.xlrd}`, "--with", `pillow==${PACKAGE_PINS.pillow}`];
1003
+ return pinSpecs(cfg).flatMap((spec) => ["--with", spec]);
904
1004
  }
905
1005
  function pipInstallArgs(cfg) {
906
- return ["-m", "pip", "install", `pymupdf4llm==${cfg.pymupdfVersion}`, `openpyxl==${PACKAGE_PINS.openpyxl}`, `xlrd==${PACKAGE_PINS.xlrd}`, `pillow==${PACKAGE_PINS.pillow}`];
1006
+ return ["-m", "pip", "install", ...pinSpecs(cfg)];
907
1007
  }
908
1008
  function warmArgs(cfg) {
909
- return ["run", ...withArgs(cfg), "--python", "3.14", "python", "-c", "import pymupdf4llm, openpyxl, xlrd, PIL"];
1009
+ return ["run", ...withArgs(cfg), "--python", "3.14", "python", "-c", "import pymupdf4llm, openpyxl, xlrd, PIL, mammoth, markdownify, docx"];
910
1010
  }
911
1011
  function uvChildArgs(cfg, script, mode) {
912
1012
  return ["run", ...withArgs(cfg), "--python", "3.14", "python", script, mode];
@@ -1058,6 +1158,11 @@ try:
1058
1158
  print("XLSX", "yes")
1059
1159
  except Exception:
1060
1160
  print("XLSX", "no")
1161
+ try:
1162
+ import mammoth, markdownify, docx
1163
+ print("DOCX", "yes")
1164
+ except Exception:
1165
+ print("DOCX", "no")
1061
1166
  `;
1062
1167
  var PROBE_TIMEOUT_MS = 5e3;
1063
1168
  function probeArgs() {
@@ -1065,8 +1170,8 @@ function probeArgs() {
1065
1170
  }
1066
1171
  var PYTHON_CANDIDATES = ["python3", "python"];
1067
1172
  function parseProbeOutput(stdout) {
1068
- const m = stdout.match(/^PY (\d+) (\d+)\r?\nPDF (yes|no)\r?\nXLSX (yes|no)\s*$/);
1069
- return m ? { major: Number(m[1]), minor: Number(m[2]), pdf: m[3] === "yes", xlsx: m[4] === "yes" } : null;
1173
+ const m = stdout.match(/^PY (\d+) (\d+)\r?\nPDF (yes|no)\r?\nXLSX (yes|no)\r?\nDOCX (yes|no)\s*$/);
1174
+ return m ? { major: Number(m[1]), minor: Number(m[2]), pdf: m[3] === "yes", xlsx: m[4] === "yes", docx: m[5] === "yes" } : null;
1070
1175
  }
1071
1176
  function meetsFloor(p) {
1072
1177
  return p.major > 3 || p.major === 3 && p.minor >= 12;
@@ -1091,7 +1196,7 @@ async function resolveBackend(cfg, deps, signal) {
1091
1196
  };
1092
1197
  const warm = await deps.run("uv", warmArgs(cfg), { timeoutMs: left(), capBytes: OUTPUT_MAX_BYTES, env: deps.env, signal });
1093
1198
  if (signal?.aborted) throw new Error("aborted");
1094
- if (warm.code === 0 && !warm.timedOut) return { kind: "uv", pdf: true, xlsx: true };
1199
+ if (warm.code === 0 && !warm.timedOut) return { kind: "uv", pdf: true, xlsx: true, docx: true };
1095
1200
  const uvAbsent = warm.code === null && !warm.timedOut;
1096
1201
  const isDeadline = (result) => result !== null && "kind" in result;
1097
1202
  const probe = async (exe, tmp) => {
@@ -1107,18 +1212,19 @@ async function resolveBackend(cfg, deps, signal) {
1107
1212
  const p = await probe(exe);
1108
1213
  if (isDeadline(p)) return p;
1109
1214
  if (!p || !meetsFloor(p)) continue;
1110
- if (p.pdf) return { kind: "python", exe, pdf: true, xlsx: p.xlsx };
1215
+ if (p.pdf) return { kind: "python", exe, pdf: true, xlsx: p.xlsx, docx: p.docx };
1111
1216
  eligible ??= { exe, version: `${p.major}.${p.minor}` };
1112
1217
  }
1218
+ const healthy = (p) => meetsFloor(p) && p.pdf && p.xlsx && p.docx;
1113
1219
  const venvDir = join3(deps.cacheRoot, VENV_DIR_NAME);
1114
1220
  const venvExe = venvPython(venvDir, deps.platform);
1115
1221
  const cached = await probe(venvExe);
1116
1222
  if (isDeadline(cached)) return cached;
1117
- if (cached && meetsFloor(cached) && cached.pdf && cached.xlsx) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true };
1223
+ if (cached && healthy(cached)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1118
1224
  if (eligible) {
1119
1225
  const recheck = await probe(venvExe);
1120
1226
  if (isDeadline(recheck)) return recheck;
1121
- if (recheck && meetsFloor(recheck) && recheck.pdf && recheck.xlsx) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true };
1227
+ if (recheck && healthy(recheck)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1122
1228
  const tmp = `${venvDir}.tmp-${deps.pid}`;
1123
1229
  const bootFail = (stderr) => {
1124
1230
  deps.rmrf(tmp);
@@ -1143,19 +1249,19 @@ async function resolveBackend(cfg, deps, signal) {
1143
1249
  }
1144
1250
  };
1145
1251
  if (publish()) {
1146
- deps.rmrf(join3(deps.cacheRoot, LEGACY_VENV_DIR_NAME));
1147
- return { kind: "venv", exe: venvExe, pdf: true, xlsx: true };
1252
+ for (const legacy of LEGACY_VENV_DIR_NAMES) deps.rmrf(join3(deps.cacheRoot, legacy));
1253
+ return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1148
1254
  }
1149
1255
  const winner = await probe(venvExe, tmp);
1150
1256
  if (isDeadline(winner)) return winner;
1151
- if (winner && meetsFloor(winner) && winner.pdf && winner.xlsx) {
1257
+ if (winner && healthy(winner)) {
1152
1258
  deps.rmrf(tmp);
1153
- return { kind: "venv", exe: venvExe, pdf: true, xlsx: true };
1259
+ return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1154
1260
  }
1155
1261
  deps.rmrf(venvDir);
1156
1262
  if (publish()) {
1157
- deps.rmrf(join3(deps.cacheRoot, LEGACY_VENV_DIR_NAME));
1158
- return { kind: "venv", exe: venvExe, pdf: true, xlsx: true };
1263
+ for (const legacy of LEGACY_VENV_DIR_NAMES) deps.rmrf(join3(deps.cacheRoot, legacy));
1264
+ return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1159
1265
  }
1160
1266
  return bootFail("rename after competing bootstrap");
1161
1267
  }
@@ -1213,15 +1319,93 @@ async function tryConvertOffice(sofficeTimeoutMs, src, signal, run = runCapped,
1213
1319
  throw e;
1214
1320
  }
1215
1321
  }
1216
- async function convertOffice(sofficeTimeoutMs, src, signal, run = runCapped) {
1217
- const r = await tryConvertOffice(sofficeTimeoutMs, src, signal, run);
1218
- if (r.ok) return r;
1219
- if (r.kind === "missing") throw new Error("LibreOffice (soffice) is required to convert .docx/.pptx but was not found on PATH. Install LibreOffice or convert the file to PDF first.");
1220
- if (r.kind === "no-pdf") throw new Error("LibreOffice (soffice) ran but produced no usable PDF for this file. Ensure LibreOffice can open the document, or convert it to PDF manually first.");
1221
- throw new Error(`soffice failed (code=${r.code} timedOut=${r.timedOut}): ${r.stderr}`);
1322
+ function officeFailure(r) {
1323
+ if (r.kind === "missing") return new Error("LibreOffice (soffice) is required to convert .docx/.pptx but was not found on PATH. Install LibreOffice or convert the file to PDF first.");
1324
+ if (r.kind === "no-pdf") return new Error("LibreOffice (soffice) ran but produced no usable PDF for this file. Ensure LibreOffice can open the document, or convert it to PDF manually first.");
1325
+ return new Error(`soffice failed (code=${r.code} timedOut=${r.timedOut}): ${r.stderr}`);
1222
1326
  }
1223
1327
  var DEGRADED_TEXT = "PyMuPDF text extraction - layout/tables not preserved";
1328
+ var DEGRADED_HTML_TURNDOWN = "Turndown HTML conversion - definition lists and headerless tables not preserved";
1329
+ var DATA_IMAGE_RE = /^data:image\/(png|jpeg|gif|bmp|tiff);base64,([A-Za-z0-9+/=\s]+)$/i;
1330
+ var DATA_EXT = { png: ".png", jpeg: ".jpg", gif: ".gif", bmp: ".bmp", tiff: ".tif" };
1331
+ var DATA_SIGNATURES = {
1332
+ png: [[137, 80, 78, 71]],
1333
+ jpeg: [[255, 216, 255]],
1334
+ gif: [[71, 73, 70, 56]],
1335
+ bmp: [[66, 77]],
1336
+ tiff: [[73, 73, 42, 0], [77, 77, 0, 42]]
1337
+ };
1338
+ async function prepareHtml(inputPath, stagingDir) {
1339
+ const { JSDOM: JSDOM2 } = await import("jsdom");
1340
+ const bytes = readFileSync(inputPath);
1341
+ let source = bytes;
1342
+ try {
1343
+ source = new TextDecoder("utf-8", { fatal: true }).decode(bytes);
1344
+ } catch {
1345
+ }
1346
+ const dom = new JSDOM2(source);
1347
+ const doc = dom.window.document;
1348
+ const title = doc.querySelector("title")?.textContent?.trim();
1349
+ if (title && !doc.body.querySelector("h1")) {
1350
+ const h1 = doc.createElement("h1");
1351
+ h1.textContent = title;
1352
+ doc.body.prepend(h1);
1353
+ }
1354
+ for (const el of [...doc.querySelectorAll("head, script, style, noscript, template")]) el.remove();
1355
+ const dir = join3(stagingDir, "p1");
1356
+ mkdirSync3(dir, { recursive: true });
1357
+ let k = 0, missing = 0;
1358
+ for (const img of [...doc.querySelectorAll("img")]) {
1359
+ const src = (img.getAttribute("src") ?? "").trim();
1360
+ const alt = img.getAttribute("alt") ?? "";
1361
+ if (/^(?:https?:)?\/\//i.test(src)) {
1362
+ const a = doc.createElement("a");
1363
+ const url = src.startsWith("//") ? `https:${src}` : src;
1364
+ a.setAttribute("href", url);
1365
+ a.textContent = alt || url;
1366
+ img.replaceWith(a);
1367
+ continue;
1368
+ }
1369
+ let target = null;
1370
+ const data = src.match(DATA_IMAGE_RE);
1371
+ if (data) {
1372
+ const buf = Buffer.from(data[2].replace(/\s+/g, ""), "base64");
1373
+ if (DATA_SIGNATURES[data[1].toLowerCase()].some((signature) => signature.every((byte, i) => buf[i] === byte))) {
1374
+ target = `p1/${++k}${DATA_EXT[data[1].toLowerCase()]}`;
1375
+ writeFileSync3(join3(stagingDir, target), buf);
1376
+ }
1377
+ } else if (src && !/^[a-z][a-z0-9+.-]*:/i.test(src) && !isAbsolute(src)) {
1378
+ let file = null;
1379
+ try {
1380
+ file = resolve2(dirname(inputPath), decodeURIComponent(src.split(/[?#]/)[0]));
1381
+ } catch {
1382
+ }
1383
+ const ext = file ? extname3(file).toLowerCase() : "";
1384
+ if (file && IMAGE_EXTS.includes(ext) && statSync2(file, { throwIfNoEntry: false })?.isFile()) {
1385
+ target = `p1/${++k}${ext}`;
1386
+ copyFileSync(file, join3(stagingDir, target));
1387
+ }
1388
+ }
1389
+ if (target) img.setAttribute("src", target);
1390
+ else {
1391
+ missing++;
1392
+ img.replaceWith(doc.createTextNode(alt));
1393
+ }
1394
+ }
1395
+ writeFileSync3(join3(dir, ".done"), "");
1396
+ return { html: dom.serialize(), missing };
1397
+ }
1224
1398
  var DEGRADED_UNPDF = "unpdf text extraction - structure not preserved";
1399
+ var DEGRADED_DOCX_TEXT = "python-docx text extraction - footnotes, hyperlinks, images not preserved";
1400
+ var DEGRADED_DOCX_OFFICE = "LibreOffice PDF route - heading styles and explicit page breaks not preserved; page numbers are LibreOffice pagination";
1401
+ var DOCX_PIP = "pip install mammoth markdownify python-docx";
1402
+ var DOCX_PAGES_OFFICE = `--pages on a DOCX needs the Python DOCX backend (explicit page-break segments); the LibreOffice route has none. Remedy: install uv, or ${DOCX_PIP}`;
1403
+ var docxRemedy = `Remedy: install uv, or ${DOCX_PIP} into a Python that already has pymupdf4llm`;
1404
+ var backendState = (b, missing) => b.kind === "none" ? b.reason : missing;
1405
+ var lacksDocx = (b) => b.kind === "none" || !b.docx;
1406
+ var clearStaging = (b) => {
1407
+ for (const f of readdirSync2(b.stagingDir)) rmSync2(join3(b.stagingDir, f), { recursive: true, force: true });
1408
+ };
1225
1409
  var EXCEL_REMEDY = "Remedy: install uv, or pip install openpyxl xlrd pillow";
1226
1410
  function resolveUnpdfWorker(candidates, exists) {
1227
1411
  for (const candidate of candidates) if (exists(candidate)) return candidate;
@@ -1276,12 +1460,27 @@ function reconcileRenderMarkers(md, renderPages, fmt, sourceMap, reason) {
1276
1460
  if (/<!--rvs?:\d+-->/.test(md)) throw new Error("internal: unresolved render marker");
1277
1461
  return md;
1278
1462
  }
1463
+ var emptyOcr = (lang) => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null });
1464
+ var OCR_SENTINEL_RE = /\x00OCR ([^\x00]*)\x00/g;
1465
+ function resolveOcrLabels(md, sourceMap) {
1466
+ return md.replace(OCR_SENTINEL_RE, (_, key) => {
1467
+ const dest = sourceMap.get(key);
1468
+ return dest ? `> Text recognized in ${dest} (OCR, may contain recognition errors):` : "> Text recognized by OCR (source image missing):";
1469
+ });
1470
+ }
1471
+ function handleOcr(tier, type, o, json) {
1472
+ if (tier === "unpdf") return o.ocr ? { ...emptyOcr(o.ocrLanguage), status: "unavailable", reason: "no Python backend" } : null;
1473
+ const x = json.ocr;
1474
+ if (!x || type !== "image" && !x.textless.length && !x.pages.length && !x.ocrFailed.length) return null;
1475
+ return x;
1476
+ }
1279
1477
  async function convertDocument(o, signal, seams) {
1280
1478
  const s = { backend: (c) => getBackend(c, void 0, signal), runTier: runTierReal, office: tryConvertOffice, ...seams };
1281
1479
  const inputPath = resolve2(o.path);
1282
1480
  const st = statSync2(inputPath, { throwIfNoEntry: false });
1283
1481
  if (!st || !st.isFile()) throw new Error(`Not a readable file: ${o.path}`);
1284
1482
  const type = classifyInput(inputPath);
1483
+ if (o.pages && (type === "html" || type === "image")) throw new Error(`--pages does not apply to ${type === "html" ? "HTML files" : "images"}`);
1285
1484
  const isExcel = type === "xlsx" || type === "xls";
1286
1485
  if (isExcel && o.pages) throw new Error("--pages does not apply to spreadsheets: worksheets have no stable page numbering");
1287
1486
  const backend = await s.backend({ pymupdfVersion: o.pymupdfVersion, warmTimeoutMs: o.warmTimeoutMs });
@@ -1291,81 +1490,175 @@ async function convertDocument(o, signal, seams) {
1291
1490
  let office = null;
1292
1491
  try {
1293
1492
  let pdfPath = inputPath;
1294
- if (type === "docx" || type === "pptx") {
1295
- office = await convertOffice(o.sofficeTimeoutMs, inputPath, signal);
1296
- pdfPath = office.pdfPath;
1297
- }
1298
- const base = { path: pdfPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion };
1493
+ const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
1299
1494
  let tier, engine, json, degraded = null, fallbackReason = null;
1495
+ let explicitBreaks = null;
1300
1496
  let notes = [];
1301
- if (isExcel) {
1302
- const r = await s.runTier("xlsx", base, b, signal, o.excelTimeoutMs, backend);
1303
- if (!r.ok) {
1304
- if ("userError" in r) throw new Error(r.userError);
1305
- const remedy = r.reason.startsWith("timeout after") || r.reason === "output exceeded maxOutputBytes" ? ". Remedy: raise excelTimeoutMs" : "";
1306
- throw new Error(`Excel conversion failed: ${r.reason}${detailSuffix(r)}${remedy}`);
1497
+ let officeRoute = null;
1498
+ if (type === "html") {
1499
+ const prepared = await prepareHtml(inputPath, b.stagingDir);
1500
+ if (signal?.aborted) throw new Error("aborted");
1501
+ if (prepared.missing) notes.push(`${prepared.missing} image(s) not found; replaced with alt text`);
1502
+ if (!lacksDocx(backend)) {
1503
+ const r = await s.runTier("html", { ...base, html: prepared.html }, b, signal, o.primaryTimeoutMs, backend);
1504
+ if (r.ok) {
1505
+ tier = "html";
1506
+ engine = "markdownify";
1507
+ json = r.json;
1508
+ } else if ("userError" in r) throw new Error(r.userError);
1509
+ else if (signal?.aborted || r.reason === "aborted") throw new Error("aborted");
1510
+ else fallbackReason = `html ${r.reason}${detailSuffix(r)}`;
1307
1511
  }
1308
- publishSheetImages(b);
1309
- publishSheetCsvs(b);
1310
- tier = "excel";
1311
- engine = type === "xls" ? "xlrd" : "openpyxl";
1312
- json = r.json;
1313
- notes = [...json.notes ?? []];
1314
- const renderPages = json.renderPages ?? [];
1315
- let skip = null;
1316
- const perSheet = /* @__PURE__ */ new Map();
1317
- if (renderPages.length) {
1318
- const off = await s.office(o.sofficeTimeoutMs, inputPath, signal, runCapped, EXCEL_PDF_FILTER);
1319
- if (!off.ok) skip = off.kind === "missing" ? "LibreOffice not found" : off.kind === "timeout" ? `soffice failed: timeout after ${o.sofficeTimeoutMs}ms` : off.kind === "exit" ? `soffice failed: exit ${off.code}` : "soffice produced no PDF";
1320
- else {
1321
- try {
1322
- const rp = await s.runTier("render-pages", { path: off.pdfPath, sheetIndices: renderPages, expectedPages: json.sheetCount, imageDpi: o.imageDpi, imageFormat: o.imageFormat, stagingDir: b.stagingDir, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.fallbackTimeoutMs, backend);
1323
- if (!rp.ok) skip = `render failed: ${"userError" in rp ? rp.userError : rp.reason}`;
1324
- else if (rp.json.ok === false) skip = rp.json.reason ?? "render failed";
1325
- else {
1326
- publishSheetImages(b);
1327
- for (const f of rp.json.failed ?? []) perSheet.set(f.idx, f.reason);
1328
- for (const d of rp.json.rendered ?? []) if (d.dpi < o.imageDpi) notes.push(`Rendered view s${d.idx}: rendered at ${d.dpi} dpi`);
1512
+ if (json === void 0) {
1513
+ const { htmlToMarkdownRaw: htmlToMarkdownRaw2 } = await Promise.resolve().then(() => (init_fetch_core(), fetch_core_exports));
1514
+ tier = "html";
1515
+ engine = "turndown";
1516
+ degraded = DEGRADED_HTML_TURNDOWN;
1517
+ json = { markdown: `${htmlToMarkdownRaw2(prepared.html)}
1518
+ `, notes: [] };
1519
+ }
1520
+ publishStaged(b);
1521
+ }
1522
+ if (type === "image") {
1523
+ let reason = "no Python backend";
1524
+ if (backend.kind !== "none") {
1525
+ const r = await s.runTier("image", { ...base, stem: b.stem }, b, signal, o.primaryTimeoutMs, backend);
1526
+ if (r.ok) {
1527
+ publishStaged(b);
1528
+ tier = "image";
1529
+ engine = "pymupdf4llm";
1530
+ json = r.json;
1531
+ } else if ("userError" in r) throw new Error(r.userError);
1532
+ else if (signal?.aborted || r.reason === "aborted") throw new Error("aborted");
1533
+ else reason = `OCR child failed: ${r.reason}`;
1534
+ }
1535
+ if (signal?.aborted) throw new Error("aborted");
1536
+ if (json === void 0) {
1537
+ clearStaging(b);
1538
+ const file = `original${extname3(inputPath).toLowerCase()}`;
1539
+ const dir = join3(b.stagingDir, "p1");
1540
+ mkdirSync3(dir, { recursive: true });
1541
+ copyFileSync(inputPath, join3(dir, file));
1542
+ writeFileSync3(join3(dir, ".done"), "");
1543
+ publishStaged(b);
1544
+ tier = "image";
1545
+ engine = "copy";
1546
+ json = { markdown: `![${b.stem}](p1/${file})
1547
+ `, pageCount: 1, notes: [], ocr: { ...emptyOcr(o.ocrLanguage), status: "unavailable", reason } };
1548
+ }
1549
+ }
1550
+ if (type === "docx" && !lacksDocx(backend)) {
1551
+ const d = await s.runTier("docx", base, b, signal, o.primaryTimeoutMs, backend);
1552
+ if (d.ok) {
1553
+ publishStaged(b);
1554
+ tier = "docx";
1555
+ engine = d.json.engine === "python-docx" ? "python-docx" : "mammoth";
1556
+ json = d.json;
1557
+ explicitBreaks = d.json.explicitBreaks ?? 0;
1558
+ if (d.json.degraded) {
1559
+ degraded = DEGRADED_DOCX_TEXT;
1560
+ fallbackReason = d.json.fallbackReason ?? null;
1561
+ }
1562
+ } else if ("userError" in d) throw new Error(d.userError);
1563
+ else if (d.reason === "exit 1") {
1564
+ officeRoute = `docx ${d.reason}${detailSuffix(d)}`;
1565
+ clearStaging(b);
1566
+ } else throw new Error(`Conversion failed: docx ${d.reason}${detailSuffix(d)}`);
1567
+ } else if (type === "docx") officeRoute = backend.kind === "none" ? backend.reason : "python backend lacks DOCX packages";
1568
+ if (type === "docx" && officeRoute !== null) {
1569
+ if (o.pages) throw new Error(officeRoute.startsWith("docx exit") ? `${DOCX_PAGES_OFFICE} (${officeRoute})` : DOCX_PAGES_OFFICE);
1570
+ const r = await s.office(o.sofficeTimeoutMs, inputPath, signal);
1571
+ if (!r.ok && r.kind === "missing") {
1572
+ if (officeRoute.startsWith("docx exit")) throw new Error(`Conversion failed: ${officeRoute}; LibreOffice (soffice) not found on PATH`);
1573
+ throw new Error(`DOCX conversion needs the Python DOCX packages or LibreOffice. Python backend: ${backendState(backend, "found without mammoth/markdownify/python-docx")}. ${docxRemedy}, or install LibreOffice (soffice)`);
1574
+ }
1575
+ if (!r.ok) throw officeRoute.startsWith("docx exit") ? new Error(`Conversion failed: ${officeRoute}; ${officeFailure(r).message}`) : officeFailure(r);
1576
+ office = r;
1577
+ pdfPath = r.pdfPath;
1578
+ }
1579
+ if (type === "pptx") {
1580
+ const r = await s.office(o.sofficeTimeoutMs, inputPath, signal);
1581
+ if (!r.ok && r.kind === "missing") throw new Error(`PPTX conversion needs LibreOffice (soffice); direct conversion is not available. Python backend: ${backendState(backend, "available")}. Remedy: install LibreOffice`);
1582
+ if (!r.ok) throw officeFailure(r);
1583
+ office = r;
1584
+ pdfPath = r.pdfPath;
1585
+ }
1586
+ const pdfBase = { ...base, path: pdfPath };
1587
+ if (json === void 0) {
1588
+ if (isExcel) {
1589
+ const r = await s.runTier("xlsx", base, b, signal, o.excelTimeoutMs, backend);
1590
+ if (!r.ok) {
1591
+ if ("userError" in r) throw new Error(r.userError);
1592
+ const remedy = r.reason.startsWith("timeout after") || r.reason === "output exceeded maxOutputBytes" ? ". Remedy: raise excelTimeoutMs" : "";
1593
+ throw new Error(`Excel conversion failed: ${r.reason}${detailSuffix(r)}${remedy}`);
1594
+ }
1595
+ publishSheetImages(b);
1596
+ publishSheetCsvs(b);
1597
+ tier = "excel";
1598
+ engine = type === "xls" ? "xlrd" : "openpyxl";
1599
+ json = r.json;
1600
+ notes = [...json.notes ?? []];
1601
+ const renderPages = json.renderPages ?? [];
1602
+ let skip = null;
1603
+ const perSheet = /* @__PURE__ */ new Map();
1604
+ if (renderPages.length) {
1605
+ const off = await s.office(o.sofficeTimeoutMs, inputPath, signal, runCapped, EXCEL_PDF_FILTER);
1606
+ if (!off.ok) skip = off.kind === "missing" ? "LibreOffice not found" : off.kind === "timeout" ? `soffice failed: timeout after ${o.sofficeTimeoutMs}ms` : off.kind === "exit" ? `soffice failed: exit ${off.code}` : "soffice produced no PDF";
1607
+ else {
1608
+ try {
1609
+ const rp = await s.runTier("render-pages", { path: off.pdfPath, sheetIndices: renderPages, expectedPages: json.sheetCount, imageDpi: o.imageDpi, imageFormat: o.imageFormat, stagingDir: b.stagingDir, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.fallbackTimeoutMs, backend);
1610
+ if (!rp.ok) skip = `render failed: ${"userError" in rp ? rp.userError : rp.reason}`;
1611
+ else if (rp.json.ok === false) skip = rp.json.reason ?? "render failed";
1612
+ else {
1613
+ publishSheetImages(b);
1614
+ for (const f of rp.json.failed ?? []) perSheet.set(f.idx, f.reason);
1615
+ for (const d of rp.json.rendered ?? []) if (d.dpi < o.imageDpi) notes.push(`Rendered view s${d.idx}: rendered at ${d.dpi} dpi`);
1616
+ }
1617
+ } finally {
1618
+ off.cleanup();
1329
1619
  }
1330
- } finally {
1331
- off.cleanup();
1332
1620
  }
1621
+ if (skip) notes.push(`Rendered views skipped: ${skip}`);
1622
+ else if (perSheet.size) notes.push(`Rendered views: ${perSheet.size} of ${renderPages.length} unavailable`);
1623
+ }
1624
+ json = { ...json, markdown: reconcileRenderMarkers(json.markdown ?? "", renderPages, o.imageFormat, b.sourceMap, (idx) => perSheet.get(idx) ?? skip ?? "render failed") };
1625
+ } else if (backend.kind === "none") {
1626
+ const r = await s.runTier("pdf-text", pdfBase, b, signal, o.primaryTimeoutMs, backend);
1627
+ if (!r.ok) throw new Error("userError" in r ? r.userError : `Conversion failed: unpdf ${r.reason}${detailSuffix(r)}`);
1628
+ tier = "unpdf";
1629
+ engine = "unpdf";
1630
+ json = r.json;
1631
+ degraded = DEGRADED_UNPDF;
1632
+ } else {
1633
+ const p = await s.runTier("pdf-primary", pdfBase, b, signal, o.primaryTimeoutMs, backend);
1634
+ const kept = publishStaged(b);
1635
+ if (p.ok) {
1636
+ tier = "primary";
1637
+ engine = "pymupdf4llm";
1638
+ json = p.json;
1639
+ } else if ("userError" in p) throw new Error(p.userError);
1640
+ else {
1641
+ if (signal?.aborted) throw new Error("aborted");
1642
+ const keepPages = Object.fromEntries([...kept.entries()].map(([k, v]) => [String(k), v]));
1643
+ const f = await s.runTier("pdf-fallback", { ...pdfBase, keepPages }, b, signal, o.fallbackTimeoutMs, backend);
1644
+ publishStaged(b);
1645
+ if (!f.ok) throw new Error("userError" in f ? f.userError : `Conversion failed: primary ${p.reason}; fallback ${f.reason}${detailSuffix(f)}`);
1646
+ tier = "fallback";
1647
+ engine = "pymupdf-text";
1648
+ json = f.json;
1649
+ degraded = DEGRADED_TEXT;
1650
+ fallbackReason = `primary ${p.reason}`;
1333
1651
  }
1334
- if (skip) notes.push(`Rendered views skipped: ${skip}`);
1335
- else if (perSheet.size) notes.push(`Rendered views: ${perSheet.size} of ${renderPages.length} unavailable`);
1336
- }
1337
- json = { ...json, markdown: reconcileRenderMarkers(json.markdown ?? "", renderPages, o.imageFormat, b.sourceMap, (idx) => perSheet.get(idx) ?? skip ?? "render failed") };
1338
- } else if (backend.kind === "none") {
1339
- const r = await s.runTier("pdf-text", base, b, signal, o.primaryTimeoutMs, backend);
1340
- if (!r.ok) throw new Error("userError" in r ? r.userError : `Conversion failed: unpdf ${r.reason}${detailSuffix(r)}`);
1341
- tier = "unpdf";
1342
- engine = "unpdf";
1343
- json = r.json;
1344
- degraded = DEGRADED_UNPDF;
1345
- } else {
1346
- const p = await s.runTier("pdf-primary", base, b, signal, o.primaryTimeoutMs, backend);
1347
- const kept = publishStaged(b);
1348
- if (p.ok) {
1349
- tier = "primary";
1350
- engine = "pymupdf4llm";
1351
- json = p.json;
1352
- } else if ("userError" in p) throw new Error(p.userError);
1353
- else {
1354
- if (signal?.aborted) throw new Error("aborted");
1355
- const keepPages = Object.fromEntries([...kept.entries()].map(([k, v]) => [String(k), v]));
1356
- const f = await s.runTier("pdf-fallback", { ...base, keepPages }, b, signal, o.fallbackTimeoutMs, backend);
1357
- publishStaged(b);
1358
- if (!f.ok) throw new Error("userError" in f ? f.userError : `Conversion failed: primary ${p.reason}; fallback ${f.reason}${detailSuffix(f)}`);
1359
- tier = "fallback";
1360
- engine = "pymupdf-text";
1361
- json = f.json;
1362
- degraded = DEGRADED_TEXT;
1363
- fallbackReason = `primary ${p.reason}`;
1364
1652
  }
1365
1653
  }
1366
- if (!isExcel) notes = json.notes ?? [];
1367
- const body = rewriteLinks(json.markdown ?? "", b.sourceMap);
1368
- validateImageLinks(body, b.manifest, b.csvManifest);
1654
+ if (officeRoute !== null) {
1655
+ degraded = DEGRADED_DOCX_OFFICE;
1656
+ fallbackReason = fallbackReason ? `${officeRoute}; ${fallbackReason}` : officeRoute;
1657
+ }
1658
+ if (tier === void 0 || engine === void 0 || json === void 0) throw new Error("internal: no tier produced output");
1659
+ if (!isExcel) notes = [...notes, ...json.notes ?? []];
1660
+ const body = resolveOcrLabels(rewriteLinks(json.markdown ?? "", b.sourceMap), b.sourceMap);
1661
+ validateImageLinks(body, b.manifest, b.csvManifest, type === "html");
1369
1662
  const head = [];
1370
1663
  if (degraded) head.push(`Degraded: ${degraded}`);
1371
1664
  if (fallbackReason) head.push(`Fallback-Reason: ${fallbackReason}`);
@@ -1377,7 +1670,7 @@ async function convertDocument(o, signal, seams) {
1377
1670
  ` : "") + body;
1378
1671
  commitBundle(b, markdown);
1379
1672
  const outline = scanOutline(markdown, o.outlineMaxEntries);
1380
- const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total };
1673
+ const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr: handleOcr(tier, type, o, json) };
1381
1674
  return { output: formatHandle(details), details };
1382
1675
  } catch (e) {
1383
1676
  abortBundle(b);
@@ -1392,15 +1685,20 @@ async function inspectDocument(o, signal, seams) {
1392
1685
  const st = statSync2(inputPath, { throwIfNoEntry: false });
1393
1686
  if (!st || !st.isFile()) throw new Error(`Not a readable file: ${o.path}`);
1394
1687
  const type = classifyInput(inputPath);
1688
+ if (type === "html" || type === "image") throw new Error(`info does not apply to ${type === "html" ? "HTML files" : "images"}; convert directly`);
1395
1689
  const isExcel = type === "xlsx" || type === "xls";
1396
1690
  const backend = await s.backend({ pymupdfVersion: o.pymupdfVersion, warmTimeoutMs: o.warmTimeoutMs });
1397
1691
  if (isExcel && (backend.kind === "none" || !backend.xlsx)) throw new Error(`Excel inspection needs a Python backend with openpyxl, xlrd and pillow. ${EXCEL_REMEDY}`);
1398
1692
  let office = null;
1399
1693
  try {
1400
1694
  let path = inputPath;
1401
- if (type === "docx" || type === "pptx") {
1402
- office = await convertOffice(o.sofficeTimeoutMs, inputPath, signal);
1403
- path = office.pdfPath;
1695
+ if (type === "docx") {
1696
+ if (lacksDocx(backend)) throw new Error(`DOCX inspection needs the Python DOCX packages. Python backend: ${backendState(backend, "found without mammoth/markdownify/python-docx")}. ${docxRemedy}`);
1697
+ } else if (type === "pptx") {
1698
+ const r2 = await s.office(o.sofficeTimeoutMs, inputPath, signal);
1699
+ if (!r2.ok) throw officeFailure(r2);
1700
+ office = r2;
1701
+ path = r2.pdfPath;
1404
1702
  }
1405
1703
  const r = await s.runTier("info", { path, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, { stagingDir: "" }, signal, isExcel ? o.excelTimeoutMs : o.fallbackTimeoutMs, backend);
1406
1704
  if (!r.ok) {
@@ -1429,6 +1727,11 @@ function parseDocToMd(rest) {
1429
1727
  path = arg;
1430
1728
  continue;
1431
1729
  }
1730
+ const negated = arg.startsWith("--no-") ? DOC_TO_MD_OPTIONS.find((o) => o.type === "bool" && o.settable && o.flag === `--${arg.slice(5)}`) : void 0;
1731
+ if (negated) {
1732
+ perCall[negated.key] = false;
1733
+ continue;
1734
+ }
1432
1735
  const d = DOC_TO_MD_OPTIONS.find((o) => o.flag === arg);
1433
1736
  if (!d) return { ok: false, error: `unknown flag: ${arg}` };
1434
1737
  if (d.type === "bool") {
@@ -1462,7 +1765,7 @@ function readCliSettings(cwd, env, warn) {
1462
1765
  if (!existsSync3(file)) continue;
1463
1766
  let raw;
1464
1767
  try {
1465
- raw = JSON.parse(readFileSync(file, "utf8")).quiver?.docToMd;
1768
+ raw = JSON.parse(readFileSync2(file, "utf8")).quiver?.docToMd;
1466
1769
  } catch {
1467
1770
  warn(`pi-quiver: ${file} is not valid JSON; ignored.`);
1468
1771
  continue;