pi-quiver 6.6.0 → 6.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -554,20 +554,20 @@ var init_fetch_core = __esm({
554
554
 
555
555
  // bin/pi-quiver.ts
556
556
  init_fetch_core();
557
- import { existsSync as existsSync3, readFileSync as readFileSync2, realpathSync } from "node:fs";
557
+ import { existsSync as existsSync3, readFileSync as readFileSync3, realpathSync } from "node:fs";
558
558
  import { homedir as homedir2 } from "node:os";
559
559
  import { join as join4, resolve as resolve3 } from "node:path";
560
560
  import { fileURLToPath as fileURLToPath2 } from "node:url";
561
561
 
562
562
  // lib/doc-to-md-core.ts
563
- import { copyFileSync, existsSync as existsSync2, mkdirSync as mkdirSync3, mkdtempSync, readFileSync, readdirSync as readdirSync2, renameSync as renameSync2, rmSync as rmSync2, statSync as statSync2, writeFileSync as writeFileSync3 } from "node:fs";
563
+ import { copyFileSync, existsSync as existsSync2, mkdirSync as mkdirSync3, mkdtempSync, readFileSync as readFileSync2, readdirSync as readdirSync2, renameSync as renameSync2, rmSync as rmSync2, statSync as statSync2, writeFileSync as writeFileSync3 } from "node:fs";
564
564
  import { spawn } from "node:child_process";
565
565
  import { homedir, tmpdir as tmpdir3 } from "node:os";
566
566
  import { basename, dirname, extname as extname3, isAbsolute, join as join3, resolve as resolve2 } from "node:path";
567
567
  import { fileURLToPath, pathToFileURL } from "node:url";
568
568
 
569
569
  // lib/doc-to-md-bundle.ts
570
- import fs, { closeSync, existsSync, mkdirSync as mkdirSync2, openSync, readdirSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync as writeFileSync2 } from "node:fs";
570
+ import fs, { closeSync, existsSync, mkdirSync as mkdirSync2, openSync, readdirSync, readFileSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync as writeFileSync2 } from "node:fs";
571
571
  import { tmpdir as tmpdir2 } from "node:os";
572
572
  import { extname, join as join2, resolve } from "node:path";
573
573
  import { randomBytes } from "node:crypto";
@@ -578,31 +578,59 @@ function ownedPattern(stem) {
578
578
  function ownedCsvPattern(stem) {
579
579
  return new RegExp(`^${escRe(stem)}-s\\d+-[a-z0-9-]+\\.csv$`);
580
580
  }
581
- var SHEET_LINK_RE = /\[[^\]]*\]\(\s*(sheets\/[^)\s]+)\s*\)/g;
581
+ function ownedPagePattern(stem) {
582
+ return new RegExp(`^${escRe(stem)}-p\\d+\\.[a-z0-9]+$`);
583
+ }
584
+ var FILE_LINK_RE = /\[[^\]]*\]\(\s*((?:sheets|attachments)\/[^)\s]+)\s*\)/g;
582
585
  var IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
583
586
  function imageTarget(m) {
584
587
  const target = (m[1] ?? m[2] ?? m[3] ?? m[4] ?? m[5] ?? "").trim();
585
588
  const destination = target.replace(/^<([\s\S]*)>$/, "$1").trim();
586
589
  return m[2] === void 0 ? destination : destination.replace(/\s+(?:"[^"]*"|'[^']*'|\([^)]*\))$/, "");
587
590
  }
588
- function openBundle(root, stem, overwrite) {
591
+ function openBundle(root, requested, overwrite) {
589
592
  mkdirSync2(root, { recursive: true });
590
- const mdPath = join2(root, `${stem}.md`);
591
- const lockPath = `${mdPath}.lock`;
592
- let fd;
593
- try {
594
- fd = openSync(lockPath, "wx");
595
- } catch (e) {
596
- if (e.code === "EEXIST") throw new Error(`Another conversion owns ${mdPath} (lock: ${lockPath}); if no conversion is running, delete the lock`);
597
- throw new Error(`Output directory not writable: ${root} (${e.message})`);
593
+ let stem, mdPath, lockPath;
594
+ let renameReason = null;
595
+ const exists = `${requested}.md exists`;
596
+ const held = `${requested}.md.lock held; delete it if no conversion is running`;
597
+ for (let i = 1; ; i++) {
598
+ stem = i === 1 ? requested : `${requested}-${i}`;
599
+ mdPath = join2(root, `${stem}.md`);
600
+ lockPath = `${mdPath}.lock`;
601
+ if (!overwrite) {
602
+ const mdExists = existsSync(mdPath);
603
+ if (mdExists || existsSync(lockPath)) {
604
+ if (i === 1) renameReason = mdExists ? exists : held;
605
+ continue;
606
+ }
607
+ }
608
+ let fd;
609
+ try {
610
+ fd = openSync(lockPath, "wx");
611
+ } catch (e) {
612
+ if (e.code !== "EEXIST") throw new Error(`Output directory not writable: ${root} (${e.message})`);
613
+ if (overwrite) throw new Error(`Another conversion owns ${mdPath} (lock: ${lockPath}); if no conversion is running, delete the lock`);
614
+ if (i === 1) renameReason = held;
615
+ continue;
616
+ }
617
+ closeSync(fd);
618
+ if (!overwrite && existsSync(mdPath)) {
619
+ unlinkSync(lockPath);
620
+ if (i === 1) renameReason = exists;
621
+ continue;
622
+ }
623
+ break;
598
624
  }
599
- closeSync(fd);
625
+ const renamedFrom = stem === requested ? null : requested;
600
626
  const imagesDir = join2(root, "images");
601
627
  const sheetsDir = join2(root, "sheets");
628
+ const pagesDir = join2(root, "pages");
629
+ const attachmentsDir = join2(root, "attachments");
602
630
  try {
603
- if (existsSync(mdPath)) {
604
- if (!overwrite) throw new Error(`Output exists: ${mdPath} (pass overwrite)`);
605
- const owned = ownedPattern(stem), ownedCsv = ownedCsvPattern(stem);
631
+ if (existsSync(mdPath) && overwrite) {
632
+ const owned = ownedPattern(stem), ownedCsv = ownedCsvPattern(stem), ownedPage = ownedPagePattern(stem);
633
+ const attachments = new Set([...readFileSync(mdPath, "utf8").matchAll(FILE_LINK_RE)].filter((m) => m[1].startsWith("attachments/")).map((m) => m[1].slice("attachments/".length)).filter((f) => f.startsWith(`${stem}-`)));
606
634
  unlinkSync(mdPath);
607
635
  if (existsSync(imagesDir)) {
608
636
  for (const f of readdirSync(imagesDir)) if (owned.test(f)) rmSync(join2(imagesDir, f), { force: true });
@@ -610,12 +638,20 @@ function openBundle(root, stem, overwrite) {
610
638
  if (existsSync(sheetsDir)) {
611
639
  for (const f of readdirSync(sheetsDir)) if (ownedCsv.test(f)) rmSync(join2(sheetsDir, f), { force: true });
612
640
  }
641
+ if (existsSync(pagesDir)) {
642
+ for (const f of readdirSync(pagesDir)) if (ownedPage.test(f)) rmSync(join2(pagesDir, f), { force: true });
643
+ }
644
+ if (existsSync(attachmentsDir)) {
645
+ for (const f of readdirSync(attachmentsDir)) if (attachments.has(f)) rmSync(join2(attachmentsDir, f), { force: true });
646
+ }
613
647
  }
614
648
  const lockId = randomBytes(6).toString("hex");
615
649
  const stagingDir = join2(imagesDir, `.stage-${lockId}`);
616
650
  const sheetsStagingDir = join2(sheetsDir, `.stage-${lockId}`);
617
651
  mkdirSync2(stagingDir, { recursive: true });
618
- return { root, stem, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
652
+ const pagesStagingDir = join2(pagesDir, `.stage-${lockId}`);
653
+ const attachmentsStagingDir = join2(attachmentsDir, `.stage-${lockId}`);
654
+ return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), pageManifest: /* @__PURE__ */ new Set(), attachmentManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
619
655
  } catch (e) {
620
656
  rmSync(lockPath, { force: true });
621
657
  throw e;
@@ -666,25 +702,54 @@ function publishSheetCsvs(b) {
666
702
  b.sourceMap.set(`sheets/${f}`, `sheets/${name}`);
667
703
  }
668
704
  }
705
+ function publishPageImages(b, pageCount) {
706
+ if (!existsSync(b.pagesStagingDir)) return;
707
+ for (const f of readdirSync(b.pagesStagingDir).sort()) {
708
+ const m = f.match(/^p(\d+)(\.[a-z0-9]+)$/i);
709
+ if (!m || !statSync(join2(b.pagesStagingDir, f)).isFile()) continue;
710
+ const name = `${b.stem}-p${m[1].padStart(String(pageCount).length, "0")}${m[2]}`;
711
+ mkdirSync2(b.pagesDir, { recursive: true });
712
+ renameSync(join2(b.pagesStagingDir, f), join2(b.pagesDir, name));
713
+ b.pageManifest.add(name);
714
+ b.sourceMap.set(`pages/${f}`, `pages/${name}`);
715
+ }
716
+ rmSync(b.pagesStagingDir, { recursive: true, force: true });
717
+ }
718
+ function publishAttachments(b) {
719
+ if (!existsSync(b.attachmentsStagingDir)) return;
720
+ for (const f of readdirSync(b.attachmentsStagingDir).sort()) {
721
+ if (!statSync(join2(b.attachmentsStagingDir, f)).isFile()) continue;
722
+ const name = `${b.stem}-${f}`;
723
+ mkdirSync2(b.attachmentsDir, { recursive: true });
724
+ renameSync(join2(b.attachmentsStagingDir, f), join2(b.attachmentsDir, name));
725
+ b.attachmentManifest.add(name);
726
+ b.sourceMap.set(`attachments/${f}`, `attachments/${name}`);
727
+ }
728
+ rmSync(b.attachmentsStagingDir, { recursive: true, force: true });
729
+ }
669
730
  function rewriteLinks(md, sourceMap) {
670
731
  const images = md.replace(IMG_LINK_RE, (whole, mdAngle, mdPlain, htmlDouble, htmlSingle, htmlUnquoted) => {
671
732
  const target = imageTarget([whole, mdAngle, mdPlain, htmlDouble, htmlSingle, htmlUnquoted]);
672
733
  const dest = sourceMap.get(target) ?? sourceMap.get(target.replace(/^\.\//, ""));
673
734
  return dest ? whole.replace(mdAngle === void 0 ? target : `<${mdAngle}>`, dest) : whole;
674
735
  });
675
- return images.replace(SHEET_LINK_RE, (whole, target) => {
736
+ return images.replace(FILE_LINK_RE, (whole, target) => {
676
737
  const dest = sourceMap.get(target);
677
738
  return dest ? whole.split(target).join(dest) : whole;
678
739
  });
679
740
  }
680
- function validateImageLinks(md, manifest, csvManifest = /* @__PURE__ */ new Set(), html = false) {
741
+ function validateImageLinks(md, manifest, csvManifest = /* @__PURE__ */ new Set(), html = false, pageManifest = /* @__PURE__ */ new Set(), attachmentManifest = /* @__PURE__ */ new Set()) {
681
742
  for (const m of md.matchAll(IMG_LINK_RE)) {
682
743
  const target = imageTarget(m);
683
744
  if (html && !target.replace(/^\.\//, "").startsWith("p1/")) continue;
684
- if (!target.startsWith("images/") || !manifest.has(target.slice("images/".length))) throw new Error(`unexpected image reference in output: ${target}`);
745
+ if (target.startsWith("pages/")) {
746
+ if (!pageManifest.has(target.slice("pages/".length))) throw new Error(`unexpected image reference in output: ${target}`);
747
+ } else if (!target.startsWith("images/") || !manifest.has(target.slice("images/".length))) throw new Error(`unexpected image reference in output: ${target}`);
685
748
  }
686
- for (const m of md.matchAll(SHEET_LINK_RE)) {
687
- if (!csvManifest.has(m[1].slice("sheets/".length))) throw new Error(`unexpected sheet reference in output: ${m[1]}`);
749
+ for (const m of md.matchAll(FILE_LINK_RE)) {
750
+ if (m[1].startsWith("attachments/")) {
751
+ if (!attachmentManifest.has(m[1].slice("attachments/".length))) throw new Error(`unexpected attachment reference in output: ${m[1]}`);
752
+ } else if (!csvManifest.has(m[1].slice("sheets/".length))) throw new Error(`unexpected sheet reference in output: ${m[1]}`);
688
753
  }
689
754
  }
690
755
  function commitBundle(b, markdown) {
@@ -699,6 +764,14 @@ function commitBundle(b, markdown) {
699
764
  fs.rmSync(b.sheetsStagingDir, { recursive: true, force: true });
700
765
  } catch {
701
766
  }
767
+ try {
768
+ fs.rmSync(b.pagesStagingDir, { recursive: true, force: true });
769
+ } catch {
770
+ }
771
+ try {
772
+ fs.rmSync(b.attachmentsStagingDir, { recursive: true, force: true });
773
+ } catch {
774
+ }
702
775
  try {
703
776
  fs.rmSync(b.lockPath, { force: true });
704
777
  } catch {
@@ -707,9 +780,13 @@ function commitBundle(b, markdown) {
707
780
  function abortBundle(b) {
708
781
  for (const f of b.manifest) rmSync(join2(b.imagesDir, f), { force: true });
709
782
  for (const f of b.csvManifest) rmSync(join2(b.sheetsDir, f), { force: true });
783
+ for (const f of b.pageManifest) rmSync(join2(b.pagesDir, f), { force: true });
784
+ for (const f of b.attachmentManifest) rmSync(join2(b.attachmentsDir, f), { force: true });
710
785
  rmSync(`${b.mdPath}.tmp`, { force: true });
711
786
  rmSync(b.stagingDir, { recursive: true, force: true });
712
787
  rmSync(b.sheetsStagingDir, { recursive: true, force: true });
788
+ rmSync(b.pagesStagingDir, { recursive: true, force: true });
789
+ rmSync(b.attachmentsStagingDir, { recursive: true, force: true });
713
790
  rmSync(b.lockPath, { force: true });
714
791
  }
715
792
  function tempBundleRoot() {
@@ -758,7 +835,7 @@ function ocrLine(ocr, type) {
758
835
  return "OCR: skipped - image too small";
759
836
  case "ran": {
760
837
  const clauses = [`OCR: ${ocr.pages.length} ${image ? "image" : "page(s)"} (${ocr.lang})`];
761
- if (ocr.noText.length) clauses.push(`${ocr.noText.length} returned no text`);
838
+ if (ocr.noText.length) clauses.push(`no text on pages ${compactRanges(ocr.noText)}`);
762
839
  if (ocr.ocrFailed.length) clauses.push(`${ocr.ocrFailed.length} failed and were converted without OCR`);
763
840
  if (ocr.budgetStopped.length) {
764
841
  const r = compactRanges(ocr.budgetStopped, Number.POSITIVE_INFINITY).replaceAll(", ", ",");
@@ -810,13 +887,15 @@ function outlineLines(entries, total) {
810
887
  function pageCountLabel(h) {
811
888
  const n = h.pageCount ?? "?";
812
889
  if (h.type !== "docx") return String(n);
813
- if (h.tier === "docx") return `${n} (${(h.explicitBreaks ?? 0) > 0 ? "explicit page breaks, not printed pages" : "no explicit page breaks"})`;
890
+ if (h.tier === "docx") return (h.explicitBreaks ?? 0) > 0 ? `${n} (explicit page breaks, not printed pages)` : `${n} (no explicit page breaks) - no page markers; cite by Outline line`;
814
891
  return `${n} (LibreOffice pagination)`;
815
892
  }
816
893
  function formatHandle(h) {
817
894
  const lines = [`Saved-To: ${h.savedTo}`];
818
895
  if (h.imagesDir && h.imageCount > 0) lines.push(`Images-Dir: ${h.imagesDir}`);
819
896
  if (h.sheetsDir) lines.push(`Sheets-Dir: ${h.sheetsDir}`);
897
+ if (h.pagesDir && h.pageImageCount > 0) lines.push(`Pages-Dir: ${h.pagesDir} (${h.pageImageCount} pages)`);
898
+ else if (h.pageImagesReason) lines.push(`Pages-Dir: none - ${h.pageImagesReason}`);
820
899
  lines.push(`Type: ${h.type} Engine: ${h.engine} Tier: ${h.tier}`);
821
900
  lines.push(`Page-Count: ${pageCountLabel(h)} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize2(h.bytes)} / ${h.lines} lines`);
822
901
  if (h.degraded) lines.push(`Degraded: ${h.degraded}`);
@@ -826,7 +905,7 @@ function formatHandle(h) {
826
905
  if (h.emptyPages.length) fe.push(`Empty-Pages: ${compactRanges(h.emptyPages)}`);
827
906
  if (fe.length) lines.push(fe.join(" "));
828
907
  if (h.ocr) lines.push(ocrLine(h.ocr, h.type));
829
- h.notes.slice(0, NOTE_MAX_LINES).forEach((n, i) => lines.push(`${i === 0 ? "Notes: " : " "}${trunc(n, NOTE_MAX_CHARS)}`));
908
+ h.notes.slice(0, NOTE_MAX_LINES).forEach((n, i) => lines.push(`${i === 0 ? "Notes: " : " "}${n.startsWith("preview truncated:") ? n : trunc(n, NOTE_MAX_CHARS)}`));
830
909
  lines.push(...outlineLines(h.outline, h.outlineTotal));
831
910
  return lines.join("\n");
832
911
  }
@@ -860,11 +939,12 @@ var VERSION_RE = /^\d+(\.\d+)*$/;
860
939
  var OCR_LANGUAGE_RE = /^[a-z][a-z0-9_]*(\+[a-z][a-z0-9_]*)*$/;
861
940
  var IMAGE_EXTS = [".png", ".jpg", ".jpeg", ".tif", ".tiff", ".bmp", ".gif"];
862
941
  var DOC_TO_MD_OPTIONS = [
863
- { key: "path", flag: null, type: "string", default: null, settable: false, help: "Local .pdf .docx .pptx .xlsx .xls .html .htm .png .jpg .jpeg .tif .tiff .bmp .gif file" },
942
+ { key: "path", flag: null, type: "string", default: null, settable: false, help: "Local .pdf .docx .doc .pptx .xlsx .xlsm .xls .msg .eml .html .htm .png .jpg .jpeg .tif .tiff .bmp .gif file" },
864
943
  { key: "info", flag: "--info", type: "bool", default: false, settable: false, help: "Inspect only (page count, metadata, TOC or sheet inventory); no bundle" },
865
- { key: "pages", flag: "--pages", type: "pages", default: null, settable: false, help: 'Inclusive 1-based pages, e.g. "12-15" or "3,7,10-12" (PDF/DOCX/PPTX only); default all. DOCX: selects explicit-page-break segments; rejected when the file has none' },
866
- { key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir" },
944
+ { key: "pages", flag: "--pages", type: "pages", default: null, settable: false, help: 'Inclusive 1-based pages, e.g. "12-15" or "3,7,10-12" (PDF/DOCX/DOC/PPTX only); default all; "" means all pages. DOCX: selects explicit-page-break segments; rejected when the file has none' },
945
+ { key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir. <stem> = basename without extension with [^A-Za-z0-9._-]+ -> _ (empty -> document); a second call on the same stem writes <stem>-2.md" },
867
946
  { key: "overwrite", flag: "--overwrite", type: "bool", default: false, settable: false, help: "Replace an existing completed <stem>.md bundle" },
947
+ { key: "pageImages", flag: "--page-images", type: "bool", default: false, settable: false, help: "Also render every selected page to pages/<stem>-pNNN.<imageFormat> at imageDpi (PDF, PPTX, .doc, DOCX via LibreOffice); off by default" },
868
948
  { key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 6e4, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier and DOCX child (docx mode); also the unpdf tier" },
869
949
  { key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier (including DOCX LibreOffice fallback); also PDF and DOCX info and Excel rendered views" },
870
950
  { key: "sofficeTimeoutMs", flag: "--soffice-timeout", type: "int", default: 12e4, settable: true, env: "PI_DOC_TO_MD_SOFFICE_TIMEOUT_MS", help: "DOCX/PPTX -> PDF via LibreOffice; also Excel rendered views" },
@@ -927,6 +1007,7 @@ function coerceDocToMdSettings(raw, warn = console.warn) {
927
1007
  return out;
928
1008
  }
929
1009
  function parsePages(spec) {
1010
+ if (spec.trim() === "") return null;
930
1011
  const out = /* @__PURE__ */ new Set();
931
1012
  const parts = spec.split(",").map((s) => s.trim());
932
1013
  if (parts.length === 0 || parts.some((p) => p === "")) throw new UsageError(`invalid --pages "${spec}": expected e.g. "12-15" or "3,7,10-12"`);
@@ -943,7 +1024,8 @@ function sanitizeStem(base) {
943
1024
  const s = base.replace(/[^A-Za-z0-9._-]+/g, "_");
944
1025
  return s.length ? s : "document";
945
1026
  }
946
- var SUPPORTED = { ".pdf": "pdf", ".docx": "docx", ".pptx": "pptx", ".xlsx": "xlsx", ".xls": "xls", ".html": "html", ".htm": "html", ...Object.fromEntries(IMAGE_EXTS.map((ext) => [ext, "image"])) };
1027
+ var SUPPORTED = { ".pdf": "pdf", ".docx": "docx", ".pptx": "pptx", ".xlsx": "xlsx", ".xls": "xls", ".xlsm": "xlsm", ".doc": "doc", ".msg": "email", ".eml": "email", ".html": "html", ".htm": "html", ...Object.fromEntries(IMAGE_EXTS.map((ext) => [ext, "image"])) };
1028
+ var SUPPORTED_EXTENSIONS = Object.keys(SUPPORTED);
947
1029
  function classifyInput(filePath) {
948
1030
  const t = SUPPORTED[extname2(filePath).toLowerCase()];
949
1031
  if (!t) throw new Error(`Unsupported file type "${extname2(filePath) || "(none)"}"; supported: ${Object.keys(SUPPORTED).join(", ")}`);
@@ -969,8 +1051,8 @@ function resolveOptions(perCall, settings, env) {
969
1051
  out[d.key] = value;
970
1052
  }
971
1053
  const o = out;
972
- if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite)) {
973
- throw new UsageError("--info cannot be combined with --pages, --output-dir or --overwrite");
1054
+ if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages)) {
1055
+ throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite or --page-images");
974
1056
  }
975
1057
  return o;
976
1058
  }
@@ -991,14 +1073,14 @@ function renderHelp() {
991
1073
  }
992
1074
 
993
1075
  // lib/doc-to-md-core.ts
994
- var PACKAGE_PINS = { pymupdf4llm: TUNABLE_DEFAULTS.pymupdfVersion, openpyxl: "3.1.5", xlrd: "2.0.2", pillow: "12.3.0", mammoth: "1.13.0", markdownify: "1.2.3", "python-docx": "1.2.0" };
1076
+ var PACKAGE_PINS = { pymupdf4llm: TUNABLE_DEFAULTS.pymupdfVersion, openpyxl: "3.1.5", xlrd: "2.0.2", pillow: "12.3.0", mammoth: "1.13.0", markdownify: "1.2.3", "python-docx": "1.2.0", "extract-msg": "0.56.1" };
995
1077
  var KILL_GRACE_MS = 2e3;
996
- var VENV_DIR_NAME = "doc-to-md-venv-v3";
997
- var LEGACY_VENV_DIR_NAMES = ["pymupdf-venv", "doc-to-md-venv-v2"];
1078
+ var VENV_DIR_NAME = "doc-to-md-venv-v4";
1079
+ var LEGACY_VENV_DIR_NAMES = ["pymupdf-venv", "doc-to-md-venv-v2", "doc-to-md-venv-v3"];
998
1080
  var STDERR_CAP = 1e6;
999
1081
  var OUTPUT_MAX_BYTES = 2e7;
1000
1082
  var EXCEL_PDF_FILTER = 'pdf:calc_pdf_Export:{"SinglePageSheets":{"type":"boolean","value":"true"}}';
1001
- var pinSpecs = (cfg) => [`pymupdf4llm==${cfg.pymupdfVersion}`, `openpyxl==${PACKAGE_PINS.openpyxl}`, `xlrd==${PACKAGE_PINS.xlrd}`, `pillow==${PACKAGE_PINS.pillow}`, `mammoth==${PACKAGE_PINS.mammoth}`, `markdownify==${PACKAGE_PINS.markdownify}`, `python-docx==${PACKAGE_PINS["python-docx"]}`];
1083
+ var pinSpecs = (cfg) => [`pymupdf4llm==${cfg.pymupdfVersion}`, `openpyxl==${PACKAGE_PINS.openpyxl}`, `xlrd==${PACKAGE_PINS.xlrd}`, `pillow==${PACKAGE_PINS.pillow}`, `mammoth==${PACKAGE_PINS.mammoth}`, `markdownify==${PACKAGE_PINS.markdownify}`, `python-docx==${PACKAGE_PINS["python-docx"]}`, `extract-msg==${PACKAGE_PINS["extract-msg"]}`];
1002
1084
  function withArgs(cfg) {
1003
1085
  return pinSpecs(cfg).flatMap((spec) => ["--with", spec]);
1004
1086
  }
@@ -1006,7 +1088,7 @@ function pipInstallArgs(cfg) {
1006
1088
  return ["-m", "pip", "install", ...pinSpecs(cfg)];
1007
1089
  }
1008
1090
  function warmArgs(cfg) {
1009
- return ["run", ...withArgs(cfg), "--python", "3.14", "python", "-c", "import pymupdf4llm, openpyxl, xlrd, PIL, mammoth, markdownify, docx"];
1091
+ return ["run", ...withArgs(cfg), "--python", "3.14", "python", "-c", "import pymupdf4llm, openpyxl, xlrd, PIL, mammoth, markdownify, docx, extract_msg"];
1010
1092
  }
1011
1093
  function uvChildArgs(cfg, script, mode) {
1012
1094
  return ["run", ...withArgs(cfg), "--python", "3.14", "python", script, mode];
@@ -1163,6 +1245,11 @@ try:
1163
1245
  print("DOCX", "yes")
1164
1246
  except Exception:
1165
1247
  print("DOCX", "no")
1248
+ try:
1249
+ import extract_msg, markdownify
1250
+ print("EMAIL", "yes")
1251
+ except Exception:
1252
+ print("EMAIL", "no")
1166
1253
  `;
1167
1254
  var PROBE_TIMEOUT_MS = 5e3;
1168
1255
  function probeArgs() {
@@ -1170,8 +1257,8 @@ function probeArgs() {
1170
1257
  }
1171
1258
  var PYTHON_CANDIDATES = ["python3", "python"];
1172
1259
  function parseProbeOutput(stdout) {
1173
- const m = stdout.match(/^PY (\d+) (\d+)\r?\nPDF (yes|no)\r?\nXLSX (yes|no)\r?\nDOCX (yes|no)\s*$/);
1174
- return m ? { major: Number(m[1]), minor: Number(m[2]), pdf: m[3] === "yes", xlsx: m[4] === "yes", docx: m[5] === "yes" } : null;
1260
+ const m = stdout.match(/^PY (\d+) (\d+)\r?\nPDF (yes|no)\r?\nXLSX (yes|no)\r?\nDOCX (yes|no)\r?\nEMAIL (yes|no)\s*$/);
1261
+ return m ? { major: Number(m[1]), minor: Number(m[2]), pdf: m[3] === "yes", xlsx: m[4] === "yes", docx: m[5] === "yes", email: m[6] === "yes" } : null;
1175
1262
  }
1176
1263
  function meetsFloor(p) {
1177
1264
  return p.major > 3 || p.major === 3 && p.minor >= 12;
@@ -1196,7 +1283,7 @@ async function resolveBackend(cfg, deps, signal) {
1196
1283
  };
1197
1284
  const warm = await deps.run("uv", warmArgs(cfg), { timeoutMs: left(), capBytes: OUTPUT_MAX_BYTES, env: deps.env, signal });
1198
1285
  if (signal?.aborted) throw new Error("aborted");
1199
- if (warm.code === 0 && !warm.timedOut) return { kind: "uv", pdf: true, xlsx: true, docx: true };
1286
+ if (warm.code === 0 && !warm.timedOut) return { kind: "uv", pdf: true, xlsx: true, docx: true, email: true };
1200
1287
  const uvAbsent = warm.code === null && !warm.timedOut;
1201
1288
  const isDeadline = (result) => result !== null && "kind" in result;
1202
1289
  const probe = async (exe, tmp) => {
@@ -1212,19 +1299,19 @@ async function resolveBackend(cfg, deps, signal) {
1212
1299
  const p = await probe(exe);
1213
1300
  if (isDeadline(p)) return p;
1214
1301
  if (!p || !meetsFloor(p)) continue;
1215
- if (p.pdf) return { kind: "python", exe, pdf: true, xlsx: p.xlsx, docx: p.docx };
1302
+ if (p.pdf) return { kind: "python", exe, pdf: true, xlsx: p.xlsx, docx: p.docx, email: p.email };
1216
1303
  eligible ??= { exe, version: `${p.major}.${p.minor}` };
1217
1304
  }
1218
- const healthy = (p) => meetsFloor(p) && p.pdf && p.xlsx && p.docx;
1305
+ const healthy = (p) => meetsFloor(p) && p.pdf && p.xlsx && p.docx && p.email;
1219
1306
  const venvDir = join3(deps.cacheRoot, VENV_DIR_NAME);
1220
1307
  const venvExe = venvPython(venvDir, deps.platform);
1221
1308
  const cached = await probe(venvExe);
1222
1309
  if (isDeadline(cached)) return cached;
1223
- if (cached && healthy(cached)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1310
+ if (cached && healthy(cached)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
1224
1311
  if (eligible) {
1225
1312
  const recheck = await probe(venvExe);
1226
1313
  if (isDeadline(recheck)) return recheck;
1227
- if (recheck && healthy(recheck)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1314
+ if (recheck && healthy(recheck)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
1228
1315
  const tmp = `${venvDir}.tmp-${deps.pid}`;
1229
1316
  const bootFail = (stderr) => {
1230
1317
  deps.rmrf(tmp);
@@ -1250,18 +1337,18 @@ async function resolveBackend(cfg, deps, signal) {
1250
1337
  };
1251
1338
  if (publish()) {
1252
1339
  for (const legacy of LEGACY_VENV_DIR_NAMES) deps.rmrf(join3(deps.cacheRoot, legacy));
1253
- return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1340
+ return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
1254
1341
  }
1255
1342
  const winner = await probe(venvExe, tmp);
1256
1343
  if (isDeadline(winner)) return winner;
1257
1344
  if (winner && healthy(winner)) {
1258
1345
  deps.rmrf(tmp);
1259
- return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1346
+ return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
1260
1347
  }
1261
1348
  deps.rmrf(venvDir);
1262
1349
  if (publish()) {
1263
1350
  for (const legacy of LEGACY_VENV_DIR_NAMES) deps.rmrf(join3(deps.cacheRoot, legacy));
1264
- return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
1351
+ return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
1265
1352
  }
1266
1353
  return bootFail("rename after competing bootstrap");
1267
1354
  }
@@ -1337,7 +1424,7 @@ var DATA_SIGNATURES = {
1337
1424
  };
1338
1425
  async function prepareHtml(inputPath, stagingDir) {
1339
1426
  const { JSDOM: JSDOM2 } = await import("jsdom");
1340
- const bytes = readFileSync(inputPath);
1427
+ const bytes = readFileSync2(inputPath);
1341
1428
  let source = bytes;
1342
1429
  try {
1343
1430
  source = new TextDecoder("utf-8", { fatal: true }).decode(bytes);
@@ -1479,18 +1566,21 @@ async function convertDocument(o, signal, seams) {
1479
1566
  const inputPath = resolve2(o.path);
1480
1567
  const st = statSync2(inputPath, { throwIfNoEntry: false });
1481
1568
  if (!st || !st.isFile()) throw new Error(`Not a readable file: ${o.path}`);
1569
+ if (st.size === 0) throw new Error(`empty file: ${o.path}`);
1482
1570
  const type = classifyInput(inputPath);
1483
- if (o.pages && (type === "html" || type === "image")) throw new Error(`--pages does not apply to ${type === "html" ? "HTML files" : "images"}`);
1484
- const isExcel = type === "xlsx" || type === "xls";
1571
+ if (o.pages && (type === "html" || type === "image" || type === "email")) throw new Error(`--pages does not apply to ${type === "html" ? "HTML files" : type === "image" ? "images" : "email"}`);
1572
+ const isExcel = type === "xlsx" || type === "xlsm" || type === "xls";
1485
1573
  if (isExcel && o.pages) throw new Error("--pages does not apply to spreadsheets: worksheets have no stable page numbering");
1486
1574
  const backend = await s.backend({ pymupdfVersion: o.pymupdfVersion, warmTimeoutMs: o.warmTimeoutMs });
1487
1575
  if (isExcel && (backend.kind === "none" || !backend.xlsx)) throw new Error(`Excel conversion needs a Python backend with openpyxl, xlrd and pillow (${backend.kind === "none" ? backend.reason : `${backend.kind} lacks the Excel packages`}). ${EXCEL_REMEDY}`);
1576
+ if (type === "email" && extname3(inputPath).toLowerCase() === ".msg" && (backend.kind === "none" || !backend.email)) throw new Error(`MSG conversion needs the extract-msg package. Python backend: ${backendState(backend, "found without extract-msg")}. Remedy: install uv, or pip install extract-msg markdownify into that Python`);
1577
+ if (type === "email" && extname3(inputPath).toLowerCase() !== ".msg" && lacksDocx(backend)) throw new Error(`EML conversion needs the Python DOCX/HTML packages (mammoth, markdownify, python-docx). Python backend: ${backendState(backend, "found without mammoth/markdownify/python-docx")}. ${docxRemedy}`);
1488
1578
  const stem = sanitizeStem(basename(inputPath, extname3(inputPath)));
1489
1579
  const b = openBundle(o.outputDir ? resolve2(o.outputDir) : tempBundleRoot(), stem, o.overwrite);
1490
1580
  let office = null;
1491
1581
  try {
1492
1582
  let pdfPath = inputPath;
1493
- const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
1583
+ const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
1494
1584
  let tier, engine, json, degraded = null, fallbackReason = null;
1495
1585
  let explicitBreaks = null;
1496
1586
  let notes = [];
@@ -1576,13 +1666,23 @@ async function convertDocument(o, signal, seams) {
1576
1666
  office = r;
1577
1667
  pdfPath = r.pdfPath;
1578
1668
  }
1579
- if (type === "pptx") {
1669
+ if (type === "doc" || type === "pptx") {
1580
1670
  const r = await s.office(o.sofficeTimeoutMs, inputPath, signal);
1581
- if (!r.ok && r.kind === "missing") throw new Error(`PPTX conversion needs LibreOffice (soffice); direct conversion is not available. Python backend: ${backendState(backend, "available")}. Remedy: install LibreOffice`);
1671
+ if (!r.ok && r.kind === "missing") throw new Error(`${type.toUpperCase()} conversion needs LibreOffice (soffice); direct conversion is not available. Python backend: ${backendState(backend, "available")}. Remedy: install LibreOffice`);
1582
1672
  if (!r.ok) throw officeFailure(r);
1583
1673
  office = r;
1584
1674
  pdfPath = r.pdfPath;
1585
1675
  }
1676
+ if (type === "email") {
1677
+ const isMsg = extname3(inputPath).toLowerCase() === ".msg";
1678
+ const r = await s.runTier("email", { ...base, stem: b.stem, attachmentsStagingDir: b.attachmentsStagingDir }, b, signal, o.primaryTimeoutMs, backend);
1679
+ if (!r.ok) throw new Error("userError" in r ? r.userError : `Conversion failed: email ${r.reason}${detailSuffix(r)}`);
1680
+ publishStaged(b);
1681
+ publishAttachments(b);
1682
+ tier = "email";
1683
+ engine = isMsg ? "extract-msg" : "email";
1684
+ json = r.json;
1685
+ }
1586
1686
  const pdfBase = { ...base, path: pdfPath };
1587
1687
  if (json === void 0) {
1588
1688
  if (isExcel) {
@@ -1597,7 +1697,7 @@ async function convertDocument(o, signal, seams) {
1597
1697
  tier = "excel";
1598
1698
  engine = type === "xls" ? "xlrd" : "openpyxl";
1599
1699
  json = r.json;
1600
- notes = [...json.notes ?? []];
1700
+ notes = (json.notes ?? []).map((n) => n.replace(/sheets\/[^\s,;]+/g, (m) => b.sourceMap.get(m) ?? m));
1601
1701
  const renderPages = json.renderPages ?? [];
1602
1702
  let skip = null;
1603
1703
  const perSheet = /* @__PURE__ */ new Map();
@@ -1636,12 +1736,15 @@ async function convertDocument(o, signal, seams) {
1636
1736
  tier = "primary";
1637
1737
  engine = "pymupdf4llm";
1638
1738
  json = p.json;
1739
+ if (json.pageImages?.length) publishPageImages(b, json.pageCount ?? 0);
1639
1740
  } else if ("userError" in p) throw new Error(p.userError);
1640
1741
  else {
1641
1742
  if (signal?.aborted) throw new Error("aborted");
1642
1743
  const keepPages = Object.fromEntries([...kept.entries()].map(([k, v]) => [String(k), v]));
1744
+ rmSync2(b.pagesStagingDir, { recursive: true, force: true });
1643
1745
  const f = await s.runTier("pdf-fallback", { ...pdfBase, keepPages }, b, signal, o.fallbackTimeoutMs, backend);
1644
1746
  publishStaged(b);
1747
+ if (f.ok && f.json.pageImages?.length) publishPageImages(b, f.json.pageCount ?? 0);
1645
1748
  if (!f.ok) throw new Error("userError" in f ? f.userError : `Conversion failed: primary ${p.reason}; fallback ${f.reason}${detailSuffix(f)}`);
1646
1749
  tier = "fallback";
1647
1750
  engine = "pymupdf-text";
@@ -1651,14 +1754,14 @@ async function convertDocument(o, signal, seams) {
1651
1754
  }
1652
1755
  }
1653
1756
  }
1654
- if (officeRoute !== null) {
1655
- degraded = DEGRADED_DOCX_OFFICE;
1656
- fallbackReason = fallbackReason ? `${officeRoute}; ${fallbackReason}` : officeRoute;
1657
- }
1757
+ if (type === "doc" || officeRoute !== null) degraded = DEGRADED_DOCX_OFFICE;
1758
+ if (officeRoute !== null) fallbackReason = fallbackReason ? `${officeRoute}; ${fallbackReason}` : officeRoute;
1658
1759
  if (tier === void 0 || engine === void 0 || json === void 0) throw new Error("internal: no tier produced output");
1659
1760
  if (!isExcel) notes = [...notes, ...json.notes ?? []];
1761
+ if (b.renamedFrom) notes.splice(notes[0]?.startsWith("preview truncated:") ? 1 : 0, 0, `renamed to ${b.stem} (${b.renameReason})`);
1762
+ const pageImagesReason = !o.pageImages || b.pageManifest.size || tier === "primary" || tier === "fallback" ? null : tier === "unpdf" ? "page images need the Python backend" : `${type} has no page geometry`;
1660
1763
  const body = resolveOcrLabels(rewriteLinks(json.markdown ?? "", b.sourceMap), b.sourceMap);
1661
- validateImageLinks(body, b.manifest, b.csvManifest, type === "html");
1764
+ validateImageLinks(body, b.manifest, b.csvManifest, type === "html" || type === "email", b.pageManifest, b.attachmentManifest);
1662
1765
  const head = [];
1663
1766
  if (degraded) head.push(`Degraded: ${degraded}`);
1664
1767
  if (fallbackReason) head.push(`Fallback-Reason: ${fallbackReason}`);
@@ -1670,7 +1773,7 @@ async function convertDocument(o, signal, seams) {
1670
1773
  ` : "") + body;
1671
1774
  commitBundle(b, markdown);
1672
1775
  const outline = scanOutline(markdown, o.outlineMaxEntries);
1673
- const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr: handleOcr(tier, type, o, json) };
1776
+ const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr: handleOcr(tier, type, o, json) };
1674
1777
  return { output: formatHandle(details), details };
1675
1778
  } catch (e) {
1676
1779
  abortBundle(b);
@@ -1684,9 +1787,10 @@ async function inspectDocument(o, signal, seams) {
1684
1787
  const inputPath = resolve2(o.path);
1685
1788
  const st = statSync2(inputPath, { throwIfNoEntry: false });
1686
1789
  if (!st || !st.isFile()) throw new Error(`Not a readable file: ${o.path}`);
1790
+ if (st.size === 0) throw new Error(`empty file: ${o.path}`);
1687
1791
  const type = classifyInput(inputPath);
1688
- if (type === "html" || type === "image") throw new Error(`info does not apply to ${type === "html" ? "HTML files" : "images"}; convert directly`);
1689
- const isExcel = type === "xlsx" || type === "xls";
1792
+ if (type === "html" || type === "image" || type === "email") throw new Error(`info does not apply to ${type === "html" ? "HTML files" : type === "image" ? "images" : "email"}; convert directly`);
1793
+ const isExcel = type === "xlsx" || type === "xlsm" || type === "xls";
1690
1794
  const backend = await s.backend({ pymupdfVersion: o.pymupdfVersion, warmTimeoutMs: o.warmTimeoutMs });
1691
1795
  if (isExcel && (backend.kind === "none" || !backend.xlsx)) throw new Error(`Excel inspection needs a Python backend with openpyxl, xlrd and pillow. ${EXCEL_REMEDY}`);
1692
1796
  let office = null;
@@ -1694,7 +1798,7 @@ async function inspectDocument(o, signal, seams) {
1694
1798
  let path = inputPath;
1695
1799
  if (type === "docx") {
1696
1800
  if (lacksDocx(backend)) throw new Error(`DOCX inspection needs the Python DOCX packages. Python backend: ${backendState(backend, "found without mammoth/markdownify/python-docx")}. ${docxRemedy}`);
1697
- } else if (type === "pptx") {
1801
+ } else if (type === "pptx" || type === "doc") {
1698
1802
  const r2 = await s.office(o.sofficeTimeoutMs, inputPath, signal);
1699
1803
  if (!r2.ok) throw officeFailure(r2);
1700
1804
  office = r2;
@@ -1715,13 +1819,15 @@ async function inspectDocument(o, signal, seams) {
1715
1819
  }
1716
1820
 
1717
1821
  // bin/pi-quiver.ts
1718
- var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--info] [--pages <spec>] [--output-dir <dir>] [--overwrite] [tunable flags] <path> (--help for all flags)';
1822
+ var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--json] [--info] [--page-images] [--pages <spec>] [--output-dir <dir>] [--overwrite] [tunable flags] <path> (--help for all flags)';
1719
1823
  function parseDocToMd(rest) {
1720
1824
  if (rest.includes("--help") || rest.includes("-h")) return { ok: true, cmd: "doc-to-md-help" };
1825
+ const json = rest.includes("--json");
1826
+ const args = rest.filter((a) => a !== "--json");
1721
1827
  const perCall = {};
1722
1828
  let path;
1723
- for (let i = 0; i < rest.length; i++) {
1724
- const arg = rest[i];
1829
+ for (let i = 0; i < args.length; i++) {
1830
+ const arg = args[i];
1725
1831
  if (!arg.startsWith("--")) {
1726
1832
  if (path !== void 0) return { ok: false, error: `unexpected argument: ${arg}` };
1727
1833
  path = arg;
@@ -1738,7 +1844,7 @@ function parseDocToMd(rest) {
1738
1844
  perCall[d.key] = true;
1739
1845
  continue;
1740
1846
  }
1741
- const value = rest[++i];
1847
+ const value = args[++i];
1742
1848
  if (value === void 0) return { ok: false, error: `${arg} requires a value` };
1743
1849
  if (d.type === "int") {
1744
1850
  const n = Number(value);
@@ -1747,7 +1853,7 @@ function parseDocToMd(rest) {
1747
1853
  } else perCall[d.key] = value;
1748
1854
  }
1749
1855
  if (!path) return { ok: false, error: "missing <path>" };
1750
- return { ok: true, cmd: "doc-to-md", perCall: { ...perCall, path } };
1856
+ return { ok: true, cmd: "doc-to-md", perCall: { ...perCall, path }, json };
1751
1857
  }
1752
1858
  function cliAgentDir(env) {
1753
1859
  const raw = env.PI_CODING_AGENT_DIR;
@@ -1765,7 +1871,7 @@ function readCliSettings(cwd, env, warn) {
1765
1871
  if (!existsSync3(file)) continue;
1766
1872
  let raw;
1767
1873
  try {
1768
- raw = JSON.parse(readFileSync2(file, "utf8")).quiver?.docToMd;
1874
+ raw = JSON.parse(readFileSync3(file, "utf8")).quiver?.docToMd;
1769
1875
  } catch {
1770
1876
  warn(`pi-quiver: ${file} is not valid JSON; ignored.`);
1771
1877
  continue;
@@ -1860,7 +1966,11 @@ ${USAGE}
1860
1966
  }
1861
1967
  try {
1862
1968
  const r = o.info ? await inspectDocument(o) : await convertDocument(o);
1863
- process.stdout.write(`${r.output}
1969
+ if (parsed.json) {
1970
+ const payload = o.info ? r.details : (({ path, backend, pymupdfVersion, inputType, file, outputDir, ...handle }) => handle)(r.details);
1971
+ process.stdout.write(`${JSON.stringify(payload)}
1972
+ `);
1973
+ } else process.stdout.write(`${r.output}
1864
1974
  `);
1865
1975
  return 0;
1866
1976
  } catch (err) {
@@ -35,8 +35,8 @@ export default function docToMdExtension(pi: ExtensionAPI) {
35
35
  name: "doc_to_md",
36
36
  label: "Convert doc to Markdown bundle",
37
37
  description:
38
- "Convert a local PDF/DOCX/PPTX/XLSX/XLS, HTML (.html/.htm) or image (.png .jpg .jpeg .tif .tiff .bmp .gif) to a Markdown bundle on disk and return a handle (Saved-To, Images-Dir, Page-Count, Outline, diagnostics) - the Markdown itself is never inlined; read the Saved-To file (offset/limit) for content. `info: true` returns page count, metadata and TOC (or the sheet inventory) without converting - use it to pick `pages`. `pages` selects inclusive 1-based pages (PDF/PPTX) or explicit-page-break segments (DOCX; rejected when the file has none); every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Images are always extracted into images/ with relative links. Primary engine pymupdf4llm, fallback PyMuPDF text (degraded, marked), pure-JS unpdf only when no Python backend exists. Excel yields a sheet inventory (worksheets + chartsheets), a full CSV per non-empty worksheet under sheets/, a bounded preview with formulas and cached values, merged/hidden disclosure, and rendered chart views when LibreOffice is available. DOCX converts directly (mammoth; python-docx text fallback) and keeps heading styles, so the handle Outline lists `L<line>` and `p<segment>` per heading; LibreOffice (soffice) is optional for DOCX (fallback route, degraded) and Excel rendered views, and required for PPTX. Use `outputDir` for a durable bundle; without it the bundle lands in a per-call temp dir. Input must be a local file path (use fetch first for URLs). HTML converts without Readability (markdownify; Turndown fallback): local and data: images are copied into images/, remote images stay links. Pages without a text layer and image inputs always keep their picture in images/. OCR is off by default: `ocr: true` adds recognized text (labeled as OCR) when Tesseract language data is installed; the handle's OCR: line says what ran.",
39
- promptSnippet: "Convert a local PDF/DOCX/PPTX/XLSX/HTML/image to a Markdown bundle (handle returned; read Saved-To)",
38
+ "Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS, email (.msg/.eml), HTML (.html/.htm) or image (.png .jpg .jpeg .tif .tiff .bmp .gif) to a Markdown bundle on disk and return a handle (Saved-To, Images-Dir, Page-Count, Outline, diagnostics) - the Markdown itself is never inlined; read the Saved-To file (offset/limit) for content. `info: true` returns page count, metadata and TOC (or the sheet inventory) without converting - use it to pick `pages`. `pages` selects inclusive 1-based pages (PDF/PPTX) or explicit-page-break segments (DOCX; rejected when the file has none); every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Images are always extracted into images/ with relative links. Primary engine pymupdf4llm, fallback PyMuPDF text (degraded, marked), pure-JS unpdf only when no Python backend exists. Excel yields a sheet inventory (worksheets + chartsheets), a full CSV per non-empty worksheet under sheets/, a bounded preview with formulas and cached values, merged/hidden disclosure, and rendered chart views when LibreOffice is available. DOCX converts directly (mammoth; python-docx text fallback) and keeps heading styles, so the handle Outline lists `L<line>` and `p<segment>` per heading; LibreOffice (soffice) is optional for DOCX (fallback route, degraded) and Excel rendered views, and required for PPTX. Use `outputDir` for a durable bundle; without it the bundle lands in a per-call temp dir. Input must be a local file path (use fetch first for URLs). HTML converts without Readability (markdownify; Turndown fallback): local and data: images are copied into images/, remote images stay links. Pages without a text layer keep their picture in images/, or pages/ with `pageImages: true`; image inputs keep their picture in images/. `pageImages: true` writes page renders to pages/<stem>-pNNN, and email attachments are saved under attachments/. OCR is off by default: `ocr: true` adds recognized text (labeled as OCR) when Tesseract language data is installed; the handle's OCR: line says what ran.",
39
+ promptSnippet: "Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS/MSG/EML/HTML/image to a Markdown bundle (handle returned; read Saved-To)",
40
40
  parameters: buildSchema(),
41
41
 
42
42
  async execute(_toolCallId, params, signal, _onUpdate, ctx) {