pi-quiver 6.6.0 → 6.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +8 -6
- package/dist/bin/pi-quiver.js +180 -70
- package/extensions/doc_to_md.ts +2 -2
- package/lib/doc-to-md-bundle.ts +93 -23
- package/lib/doc-to-md-core.ts +57 -38
- package/lib/doc-to-md-handle.ts +9 -7
- package/lib/doc-to-md-options.ts +13 -8
- package/package.json +2 -1
- package/scripts/doc_to_md.py +286 -66
- package/scripts/docx_numbering.py +171 -0
package/dist/bin/pi-quiver.js
CHANGED
|
@@ -554,20 +554,20 @@ var init_fetch_core = __esm({
|
|
|
554
554
|
|
|
555
555
|
// bin/pi-quiver.ts
|
|
556
556
|
init_fetch_core();
|
|
557
|
-
import { existsSync as existsSync3, readFileSync as
|
|
557
|
+
import { existsSync as existsSync3, readFileSync as readFileSync3, realpathSync } from "node:fs";
|
|
558
558
|
import { homedir as homedir2 } from "node:os";
|
|
559
559
|
import { join as join4, resolve as resolve3 } from "node:path";
|
|
560
560
|
import { fileURLToPath as fileURLToPath2 } from "node:url";
|
|
561
561
|
|
|
562
562
|
// lib/doc-to-md-core.ts
|
|
563
|
-
import { copyFileSync, existsSync as existsSync2, mkdirSync as mkdirSync3, mkdtempSync, readFileSync, readdirSync as readdirSync2, renameSync as renameSync2, rmSync as rmSync2, statSync as statSync2, writeFileSync as writeFileSync3 } from "node:fs";
|
|
563
|
+
import { copyFileSync, existsSync as existsSync2, mkdirSync as mkdirSync3, mkdtempSync, readFileSync as readFileSync2, readdirSync as readdirSync2, renameSync as renameSync2, rmSync as rmSync2, statSync as statSync2, writeFileSync as writeFileSync3 } from "node:fs";
|
|
564
564
|
import { spawn } from "node:child_process";
|
|
565
565
|
import { homedir, tmpdir as tmpdir3 } from "node:os";
|
|
566
566
|
import { basename, dirname, extname as extname3, isAbsolute, join as join3, resolve as resolve2 } from "node:path";
|
|
567
567
|
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
568
568
|
|
|
569
569
|
// lib/doc-to-md-bundle.ts
|
|
570
|
-
import fs, { closeSync, existsSync, mkdirSync as mkdirSync2, openSync, readdirSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync as writeFileSync2 } from "node:fs";
|
|
570
|
+
import fs, { closeSync, existsSync, mkdirSync as mkdirSync2, openSync, readdirSync, readFileSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync as writeFileSync2 } from "node:fs";
|
|
571
571
|
import { tmpdir as tmpdir2 } from "node:os";
|
|
572
572
|
import { extname, join as join2, resolve } from "node:path";
|
|
573
573
|
import { randomBytes } from "node:crypto";
|
|
@@ -578,31 +578,59 @@ function ownedPattern(stem) {
|
|
|
578
578
|
function ownedCsvPattern(stem) {
|
|
579
579
|
return new RegExp(`^${escRe(stem)}-s\\d+-[a-z0-9-]+\\.csv$`);
|
|
580
580
|
}
|
|
581
|
-
|
|
581
|
+
function ownedPagePattern(stem) {
|
|
582
|
+
return new RegExp(`^${escRe(stem)}-p\\d+\\.[a-z0-9]+$`);
|
|
583
|
+
}
|
|
584
|
+
var FILE_LINK_RE = /\[[^\]]*\]\(\s*((?:sheets|attachments)\/[^)\s]+)\s*\)/g;
|
|
582
585
|
var IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
|
|
583
586
|
function imageTarget(m) {
|
|
584
587
|
const target = (m[1] ?? m[2] ?? m[3] ?? m[4] ?? m[5] ?? "").trim();
|
|
585
588
|
const destination = target.replace(/^<([\s\S]*)>$/, "$1").trim();
|
|
586
589
|
return m[2] === void 0 ? destination : destination.replace(/\s+(?:"[^"]*"|'[^']*'|\([^)]*\))$/, "");
|
|
587
590
|
}
|
|
588
|
-
function openBundle(root,
|
|
591
|
+
function openBundle(root, requested, overwrite) {
|
|
589
592
|
mkdirSync2(root, { recursive: true });
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
593
|
+
let stem, mdPath, lockPath;
|
|
594
|
+
let renameReason = null;
|
|
595
|
+
const exists = `${requested}.md exists`;
|
|
596
|
+
const held = `${requested}.md.lock held; delete it if no conversion is running`;
|
|
597
|
+
for (let i = 1; ; i++) {
|
|
598
|
+
stem = i === 1 ? requested : `${requested}-${i}`;
|
|
599
|
+
mdPath = join2(root, `${stem}.md`);
|
|
600
|
+
lockPath = `${mdPath}.lock`;
|
|
601
|
+
if (!overwrite) {
|
|
602
|
+
const mdExists = existsSync(mdPath);
|
|
603
|
+
if (mdExists || existsSync(lockPath)) {
|
|
604
|
+
if (i === 1) renameReason = mdExists ? exists : held;
|
|
605
|
+
continue;
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
let fd;
|
|
609
|
+
try {
|
|
610
|
+
fd = openSync(lockPath, "wx");
|
|
611
|
+
} catch (e) {
|
|
612
|
+
if (e.code !== "EEXIST") throw new Error(`Output directory not writable: ${root} (${e.message})`);
|
|
613
|
+
if (overwrite) throw new Error(`Another conversion owns ${mdPath} (lock: ${lockPath}); if no conversion is running, delete the lock`);
|
|
614
|
+
if (i === 1) renameReason = held;
|
|
615
|
+
continue;
|
|
616
|
+
}
|
|
617
|
+
closeSync(fd);
|
|
618
|
+
if (!overwrite && existsSync(mdPath)) {
|
|
619
|
+
unlinkSync(lockPath);
|
|
620
|
+
if (i === 1) renameReason = exists;
|
|
621
|
+
continue;
|
|
622
|
+
}
|
|
623
|
+
break;
|
|
598
624
|
}
|
|
599
|
-
|
|
625
|
+
const renamedFrom = stem === requested ? null : requested;
|
|
600
626
|
const imagesDir = join2(root, "images");
|
|
601
627
|
const sheetsDir = join2(root, "sheets");
|
|
628
|
+
const pagesDir = join2(root, "pages");
|
|
629
|
+
const attachmentsDir = join2(root, "attachments");
|
|
602
630
|
try {
|
|
603
|
-
if (existsSync(mdPath)) {
|
|
604
|
-
|
|
605
|
-
const
|
|
631
|
+
if (existsSync(mdPath) && overwrite) {
|
|
632
|
+
const owned = ownedPattern(stem), ownedCsv = ownedCsvPattern(stem), ownedPage = ownedPagePattern(stem);
|
|
633
|
+
const attachments = new Set([...readFileSync(mdPath, "utf8").matchAll(FILE_LINK_RE)].filter((m) => m[1].startsWith("attachments/")).map((m) => m[1].slice("attachments/".length)).filter((f) => f.startsWith(`${stem}-`)));
|
|
606
634
|
unlinkSync(mdPath);
|
|
607
635
|
if (existsSync(imagesDir)) {
|
|
608
636
|
for (const f of readdirSync(imagesDir)) if (owned.test(f)) rmSync(join2(imagesDir, f), { force: true });
|
|
@@ -610,12 +638,20 @@ function openBundle(root, stem, overwrite) {
|
|
|
610
638
|
if (existsSync(sheetsDir)) {
|
|
611
639
|
for (const f of readdirSync(sheetsDir)) if (ownedCsv.test(f)) rmSync(join2(sheetsDir, f), { force: true });
|
|
612
640
|
}
|
|
641
|
+
if (existsSync(pagesDir)) {
|
|
642
|
+
for (const f of readdirSync(pagesDir)) if (ownedPage.test(f)) rmSync(join2(pagesDir, f), { force: true });
|
|
643
|
+
}
|
|
644
|
+
if (existsSync(attachmentsDir)) {
|
|
645
|
+
for (const f of readdirSync(attachmentsDir)) if (attachments.has(f)) rmSync(join2(attachmentsDir, f), { force: true });
|
|
646
|
+
}
|
|
613
647
|
}
|
|
614
648
|
const lockId = randomBytes(6).toString("hex");
|
|
615
649
|
const stagingDir = join2(imagesDir, `.stage-${lockId}`);
|
|
616
650
|
const sheetsStagingDir = join2(sheetsDir, `.stage-${lockId}`);
|
|
617
651
|
mkdirSync2(stagingDir, { recursive: true });
|
|
618
|
-
|
|
652
|
+
const pagesStagingDir = join2(pagesDir, `.stage-${lockId}`);
|
|
653
|
+
const attachmentsStagingDir = join2(attachmentsDir, `.stage-${lockId}`);
|
|
654
|
+
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), pageManifest: /* @__PURE__ */ new Set(), attachmentManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
|
|
619
655
|
} catch (e) {
|
|
620
656
|
rmSync(lockPath, { force: true });
|
|
621
657
|
throw e;
|
|
@@ -666,25 +702,54 @@ function publishSheetCsvs(b) {
|
|
|
666
702
|
b.sourceMap.set(`sheets/${f}`, `sheets/${name}`);
|
|
667
703
|
}
|
|
668
704
|
}
|
|
705
|
+
function publishPageImages(b, pageCount) {
|
|
706
|
+
if (!existsSync(b.pagesStagingDir)) return;
|
|
707
|
+
for (const f of readdirSync(b.pagesStagingDir).sort()) {
|
|
708
|
+
const m = f.match(/^p(\d+)(\.[a-z0-9]+)$/i);
|
|
709
|
+
if (!m || !statSync(join2(b.pagesStagingDir, f)).isFile()) continue;
|
|
710
|
+
const name = `${b.stem}-p${m[1].padStart(String(pageCount).length, "0")}${m[2]}`;
|
|
711
|
+
mkdirSync2(b.pagesDir, { recursive: true });
|
|
712
|
+
renameSync(join2(b.pagesStagingDir, f), join2(b.pagesDir, name));
|
|
713
|
+
b.pageManifest.add(name);
|
|
714
|
+
b.sourceMap.set(`pages/${f}`, `pages/${name}`);
|
|
715
|
+
}
|
|
716
|
+
rmSync(b.pagesStagingDir, { recursive: true, force: true });
|
|
717
|
+
}
|
|
718
|
+
function publishAttachments(b) {
|
|
719
|
+
if (!existsSync(b.attachmentsStagingDir)) return;
|
|
720
|
+
for (const f of readdirSync(b.attachmentsStagingDir).sort()) {
|
|
721
|
+
if (!statSync(join2(b.attachmentsStagingDir, f)).isFile()) continue;
|
|
722
|
+
const name = `${b.stem}-${f}`;
|
|
723
|
+
mkdirSync2(b.attachmentsDir, { recursive: true });
|
|
724
|
+
renameSync(join2(b.attachmentsStagingDir, f), join2(b.attachmentsDir, name));
|
|
725
|
+
b.attachmentManifest.add(name);
|
|
726
|
+
b.sourceMap.set(`attachments/${f}`, `attachments/${name}`);
|
|
727
|
+
}
|
|
728
|
+
rmSync(b.attachmentsStagingDir, { recursive: true, force: true });
|
|
729
|
+
}
|
|
669
730
|
function rewriteLinks(md, sourceMap) {
|
|
670
731
|
const images = md.replace(IMG_LINK_RE, (whole, mdAngle, mdPlain, htmlDouble, htmlSingle, htmlUnquoted) => {
|
|
671
732
|
const target = imageTarget([whole, mdAngle, mdPlain, htmlDouble, htmlSingle, htmlUnquoted]);
|
|
672
733
|
const dest = sourceMap.get(target) ?? sourceMap.get(target.replace(/^\.\//, ""));
|
|
673
734
|
return dest ? whole.replace(mdAngle === void 0 ? target : `<${mdAngle}>`, dest) : whole;
|
|
674
735
|
});
|
|
675
|
-
return images.replace(
|
|
736
|
+
return images.replace(FILE_LINK_RE, (whole, target) => {
|
|
676
737
|
const dest = sourceMap.get(target);
|
|
677
738
|
return dest ? whole.split(target).join(dest) : whole;
|
|
678
739
|
});
|
|
679
740
|
}
|
|
680
|
-
function validateImageLinks(md, manifest, csvManifest = /* @__PURE__ */ new Set(), html = false) {
|
|
741
|
+
function validateImageLinks(md, manifest, csvManifest = /* @__PURE__ */ new Set(), html = false, pageManifest = /* @__PURE__ */ new Set(), attachmentManifest = /* @__PURE__ */ new Set()) {
|
|
681
742
|
for (const m of md.matchAll(IMG_LINK_RE)) {
|
|
682
743
|
const target = imageTarget(m);
|
|
683
744
|
if (html && !target.replace(/^\.\//, "").startsWith("p1/")) continue;
|
|
684
|
-
if (
|
|
745
|
+
if (target.startsWith("pages/")) {
|
|
746
|
+
if (!pageManifest.has(target.slice("pages/".length))) throw new Error(`unexpected image reference in output: ${target}`);
|
|
747
|
+
} else if (!target.startsWith("images/") || !manifest.has(target.slice("images/".length))) throw new Error(`unexpected image reference in output: ${target}`);
|
|
685
748
|
}
|
|
686
|
-
for (const m of md.matchAll(
|
|
687
|
-
if (
|
|
749
|
+
for (const m of md.matchAll(FILE_LINK_RE)) {
|
|
750
|
+
if (m[1].startsWith("attachments/")) {
|
|
751
|
+
if (!attachmentManifest.has(m[1].slice("attachments/".length))) throw new Error(`unexpected attachment reference in output: ${m[1]}`);
|
|
752
|
+
} else if (!csvManifest.has(m[1].slice("sheets/".length))) throw new Error(`unexpected sheet reference in output: ${m[1]}`);
|
|
688
753
|
}
|
|
689
754
|
}
|
|
690
755
|
function commitBundle(b, markdown) {
|
|
@@ -699,6 +764,14 @@ function commitBundle(b, markdown) {
|
|
|
699
764
|
fs.rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|
|
700
765
|
} catch {
|
|
701
766
|
}
|
|
767
|
+
try {
|
|
768
|
+
fs.rmSync(b.pagesStagingDir, { recursive: true, force: true });
|
|
769
|
+
} catch {
|
|
770
|
+
}
|
|
771
|
+
try {
|
|
772
|
+
fs.rmSync(b.attachmentsStagingDir, { recursive: true, force: true });
|
|
773
|
+
} catch {
|
|
774
|
+
}
|
|
702
775
|
try {
|
|
703
776
|
fs.rmSync(b.lockPath, { force: true });
|
|
704
777
|
} catch {
|
|
@@ -707,9 +780,13 @@ function commitBundle(b, markdown) {
|
|
|
707
780
|
function abortBundle(b) {
|
|
708
781
|
for (const f of b.manifest) rmSync(join2(b.imagesDir, f), { force: true });
|
|
709
782
|
for (const f of b.csvManifest) rmSync(join2(b.sheetsDir, f), { force: true });
|
|
783
|
+
for (const f of b.pageManifest) rmSync(join2(b.pagesDir, f), { force: true });
|
|
784
|
+
for (const f of b.attachmentManifest) rmSync(join2(b.attachmentsDir, f), { force: true });
|
|
710
785
|
rmSync(`${b.mdPath}.tmp`, { force: true });
|
|
711
786
|
rmSync(b.stagingDir, { recursive: true, force: true });
|
|
712
787
|
rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|
|
788
|
+
rmSync(b.pagesStagingDir, { recursive: true, force: true });
|
|
789
|
+
rmSync(b.attachmentsStagingDir, { recursive: true, force: true });
|
|
713
790
|
rmSync(b.lockPath, { force: true });
|
|
714
791
|
}
|
|
715
792
|
function tempBundleRoot() {
|
|
@@ -758,7 +835,7 @@ function ocrLine(ocr, type) {
|
|
|
758
835
|
return "OCR: skipped - image too small";
|
|
759
836
|
case "ran": {
|
|
760
837
|
const clauses = [`OCR: ${ocr.pages.length} ${image ? "image" : "page(s)"} (${ocr.lang})`];
|
|
761
|
-
if (ocr.noText.length) clauses.push(
|
|
838
|
+
if (ocr.noText.length) clauses.push(`no text on pages ${compactRanges(ocr.noText)}`);
|
|
762
839
|
if (ocr.ocrFailed.length) clauses.push(`${ocr.ocrFailed.length} failed and were converted without OCR`);
|
|
763
840
|
if (ocr.budgetStopped.length) {
|
|
764
841
|
const r = compactRanges(ocr.budgetStopped, Number.POSITIVE_INFINITY).replaceAll(", ", ",");
|
|
@@ -810,13 +887,15 @@ function outlineLines(entries, total) {
|
|
|
810
887
|
function pageCountLabel(h) {
|
|
811
888
|
const n = h.pageCount ?? "?";
|
|
812
889
|
if (h.type !== "docx") return String(n);
|
|
813
|
-
if (h.tier === "docx") return
|
|
890
|
+
if (h.tier === "docx") return (h.explicitBreaks ?? 0) > 0 ? `${n} (explicit page breaks, not printed pages)` : `${n} (no explicit page breaks) - no page markers; cite by Outline line`;
|
|
814
891
|
return `${n} (LibreOffice pagination)`;
|
|
815
892
|
}
|
|
816
893
|
function formatHandle(h) {
|
|
817
894
|
const lines = [`Saved-To: ${h.savedTo}`];
|
|
818
895
|
if (h.imagesDir && h.imageCount > 0) lines.push(`Images-Dir: ${h.imagesDir}`);
|
|
819
896
|
if (h.sheetsDir) lines.push(`Sheets-Dir: ${h.sheetsDir}`);
|
|
897
|
+
if (h.pagesDir && h.pageImageCount > 0) lines.push(`Pages-Dir: ${h.pagesDir} (${h.pageImageCount} pages)`);
|
|
898
|
+
else if (h.pageImagesReason) lines.push(`Pages-Dir: none - ${h.pageImagesReason}`);
|
|
820
899
|
lines.push(`Type: ${h.type} Engine: ${h.engine} Tier: ${h.tier}`);
|
|
821
900
|
lines.push(`Page-Count: ${pageCountLabel(h)} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize2(h.bytes)} / ${h.lines} lines`);
|
|
822
901
|
if (h.degraded) lines.push(`Degraded: ${h.degraded}`);
|
|
@@ -826,7 +905,7 @@ function formatHandle(h) {
|
|
|
826
905
|
if (h.emptyPages.length) fe.push(`Empty-Pages: ${compactRanges(h.emptyPages)}`);
|
|
827
906
|
if (fe.length) lines.push(fe.join(" "));
|
|
828
907
|
if (h.ocr) lines.push(ocrLine(h.ocr, h.type));
|
|
829
|
-
h.notes.slice(0, NOTE_MAX_LINES).forEach((n, i) => lines.push(`${i === 0 ? "Notes: " : " "}${trunc(n, NOTE_MAX_CHARS)}`));
|
|
908
|
+
h.notes.slice(0, NOTE_MAX_LINES).forEach((n, i) => lines.push(`${i === 0 ? "Notes: " : " "}${n.startsWith("preview truncated:") ? n : trunc(n, NOTE_MAX_CHARS)}`));
|
|
830
909
|
lines.push(...outlineLines(h.outline, h.outlineTotal));
|
|
831
910
|
return lines.join("\n");
|
|
832
911
|
}
|
|
@@ -860,11 +939,12 @@ var VERSION_RE = /^\d+(\.\d+)*$/;
|
|
|
860
939
|
var OCR_LANGUAGE_RE = /^[a-z][a-z0-9_]*(\+[a-z][a-z0-9_]*)*$/;
|
|
861
940
|
var IMAGE_EXTS = [".png", ".jpg", ".jpeg", ".tif", ".tiff", ".bmp", ".gif"];
|
|
862
941
|
var DOC_TO_MD_OPTIONS = [
|
|
863
|
-
{ key: "path", flag: null, type: "string", default: null, settable: false, help: "Local .pdf .docx .pptx .xlsx .xls .html .htm .png .jpg .jpeg .tif .tiff .bmp .gif file" },
|
|
942
|
+
{ key: "path", flag: null, type: "string", default: null, settable: false, help: "Local .pdf .docx .doc .pptx .xlsx .xlsm .xls .msg .eml .html .htm .png .jpg .jpeg .tif .tiff .bmp .gif file" },
|
|
864
943
|
{ key: "info", flag: "--info", type: "bool", default: false, settable: false, help: "Inspect only (page count, metadata, TOC or sheet inventory); no bundle" },
|
|
865
|
-
{ key: "pages", flag: "--pages", type: "pages", default: null, settable: false, help: 'Inclusive 1-based pages, e.g. "12-15" or "3,7,10-12" (PDF/DOCX/PPTX only); default all. DOCX: selects explicit-page-break segments; rejected when the file has none' },
|
|
866
|
-
{ key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir" },
|
|
944
|
+
{ key: "pages", flag: "--pages", type: "pages", default: null, settable: false, help: 'Inclusive 1-based pages, e.g. "12-15" or "3,7,10-12" (PDF/DOCX/DOC/PPTX only); default all; "" means all pages. DOCX: selects explicit-page-break segments; rejected when the file has none' },
|
|
945
|
+
{ key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir. <stem> = basename without extension with [^A-Za-z0-9._-]+ -> _ (empty -> document); a second call on the same stem writes <stem>-2.md" },
|
|
867
946
|
{ key: "overwrite", flag: "--overwrite", type: "bool", default: false, settable: false, help: "Replace an existing completed <stem>.md bundle" },
|
|
947
|
+
{ key: "pageImages", flag: "--page-images", type: "bool", default: false, settable: false, help: "Also render every selected page to pages/<stem>-pNNN.<imageFormat> at imageDpi (PDF, PPTX, .doc, DOCX via LibreOffice); off by default" },
|
|
868
948
|
{ key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 6e4, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier and DOCX child (docx mode); also the unpdf tier" },
|
|
869
949
|
{ key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier (including DOCX LibreOffice fallback); also PDF and DOCX info and Excel rendered views" },
|
|
870
950
|
{ key: "sofficeTimeoutMs", flag: "--soffice-timeout", type: "int", default: 12e4, settable: true, env: "PI_DOC_TO_MD_SOFFICE_TIMEOUT_MS", help: "DOCX/PPTX -> PDF via LibreOffice; also Excel rendered views" },
|
|
@@ -927,6 +1007,7 @@ function coerceDocToMdSettings(raw, warn = console.warn) {
|
|
|
927
1007
|
return out;
|
|
928
1008
|
}
|
|
929
1009
|
function parsePages(spec) {
|
|
1010
|
+
if (spec.trim() === "") return null;
|
|
930
1011
|
const out = /* @__PURE__ */ new Set();
|
|
931
1012
|
const parts = spec.split(",").map((s) => s.trim());
|
|
932
1013
|
if (parts.length === 0 || parts.some((p) => p === "")) throw new UsageError(`invalid --pages "${spec}": expected e.g. "12-15" or "3,7,10-12"`);
|
|
@@ -943,7 +1024,8 @@ function sanitizeStem(base) {
|
|
|
943
1024
|
const s = base.replace(/[^A-Za-z0-9._-]+/g, "_");
|
|
944
1025
|
return s.length ? s : "document";
|
|
945
1026
|
}
|
|
946
|
-
var SUPPORTED = { ".pdf": "pdf", ".docx": "docx", ".pptx": "pptx", ".xlsx": "xlsx", ".xls": "xls", ".html": "html", ".htm": "html", ...Object.fromEntries(IMAGE_EXTS.map((ext) => [ext, "image"])) };
|
|
1027
|
+
var SUPPORTED = { ".pdf": "pdf", ".docx": "docx", ".pptx": "pptx", ".xlsx": "xlsx", ".xls": "xls", ".xlsm": "xlsm", ".doc": "doc", ".msg": "email", ".eml": "email", ".html": "html", ".htm": "html", ...Object.fromEntries(IMAGE_EXTS.map((ext) => [ext, "image"])) };
|
|
1028
|
+
var SUPPORTED_EXTENSIONS = Object.keys(SUPPORTED);
|
|
947
1029
|
function classifyInput(filePath) {
|
|
948
1030
|
const t = SUPPORTED[extname2(filePath).toLowerCase()];
|
|
949
1031
|
if (!t) throw new Error(`Unsupported file type "${extname2(filePath) || "(none)"}"; supported: ${Object.keys(SUPPORTED).join(", ")}`);
|
|
@@ -969,8 +1051,8 @@ function resolveOptions(perCall, settings, env) {
|
|
|
969
1051
|
out[d.key] = value;
|
|
970
1052
|
}
|
|
971
1053
|
const o = out;
|
|
972
|
-
if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite)) {
|
|
973
|
-
throw new UsageError("--info cannot be combined with --pages, --output-dir or --
|
|
1054
|
+
if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages)) {
|
|
1055
|
+
throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite or --page-images");
|
|
974
1056
|
}
|
|
975
1057
|
return o;
|
|
976
1058
|
}
|
|
@@ -991,14 +1073,14 @@ function renderHelp() {
|
|
|
991
1073
|
}
|
|
992
1074
|
|
|
993
1075
|
// lib/doc-to-md-core.ts
|
|
994
|
-
var PACKAGE_PINS = { pymupdf4llm: TUNABLE_DEFAULTS.pymupdfVersion, openpyxl: "3.1.5", xlrd: "2.0.2", pillow: "12.3.0", mammoth: "1.13.0", markdownify: "1.2.3", "python-docx": "1.2.0" };
|
|
1076
|
+
var PACKAGE_PINS = { pymupdf4llm: TUNABLE_DEFAULTS.pymupdfVersion, openpyxl: "3.1.5", xlrd: "2.0.2", pillow: "12.3.0", mammoth: "1.13.0", markdownify: "1.2.3", "python-docx": "1.2.0", "extract-msg": "0.56.1" };
|
|
995
1077
|
var KILL_GRACE_MS = 2e3;
|
|
996
|
-
var VENV_DIR_NAME = "doc-to-md-venv-
|
|
997
|
-
var LEGACY_VENV_DIR_NAMES = ["pymupdf-venv", "doc-to-md-venv-v2"];
|
|
1078
|
+
var VENV_DIR_NAME = "doc-to-md-venv-v4";
|
|
1079
|
+
var LEGACY_VENV_DIR_NAMES = ["pymupdf-venv", "doc-to-md-venv-v2", "doc-to-md-venv-v3"];
|
|
998
1080
|
var STDERR_CAP = 1e6;
|
|
999
1081
|
var OUTPUT_MAX_BYTES = 2e7;
|
|
1000
1082
|
var EXCEL_PDF_FILTER = 'pdf:calc_pdf_Export:{"SinglePageSheets":{"type":"boolean","value":"true"}}';
|
|
1001
|
-
var pinSpecs = (cfg) => [`pymupdf4llm==${cfg.pymupdfVersion}`, `openpyxl==${PACKAGE_PINS.openpyxl}`, `xlrd==${PACKAGE_PINS.xlrd}`, `pillow==${PACKAGE_PINS.pillow}`, `mammoth==${PACKAGE_PINS.mammoth}`, `markdownify==${PACKAGE_PINS.markdownify}`, `python-docx==${PACKAGE_PINS["python-docx"]}`];
|
|
1083
|
+
var pinSpecs = (cfg) => [`pymupdf4llm==${cfg.pymupdfVersion}`, `openpyxl==${PACKAGE_PINS.openpyxl}`, `xlrd==${PACKAGE_PINS.xlrd}`, `pillow==${PACKAGE_PINS.pillow}`, `mammoth==${PACKAGE_PINS.mammoth}`, `markdownify==${PACKAGE_PINS.markdownify}`, `python-docx==${PACKAGE_PINS["python-docx"]}`, `extract-msg==${PACKAGE_PINS["extract-msg"]}`];
|
|
1002
1084
|
function withArgs(cfg) {
|
|
1003
1085
|
return pinSpecs(cfg).flatMap((spec) => ["--with", spec]);
|
|
1004
1086
|
}
|
|
@@ -1006,7 +1088,7 @@ function pipInstallArgs(cfg) {
|
|
|
1006
1088
|
return ["-m", "pip", "install", ...pinSpecs(cfg)];
|
|
1007
1089
|
}
|
|
1008
1090
|
function warmArgs(cfg) {
|
|
1009
|
-
return ["run", ...withArgs(cfg), "--python", "3.14", "python", "-c", "import pymupdf4llm, openpyxl, xlrd, PIL, mammoth, markdownify, docx"];
|
|
1091
|
+
return ["run", ...withArgs(cfg), "--python", "3.14", "python", "-c", "import pymupdf4llm, openpyxl, xlrd, PIL, mammoth, markdownify, docx, extract_msg"];
|
|
1010
1092
|
}
|
|
1011
1093
|
function uvChildArgs(cfg, script, mode) {
|
|
1012
1094
|
return ["run", ...withArgs(cfg), "--python", "3.14", "python", script, mode];
|
|
@@ -1163,6 +1245,11 @@ try:
|
|
|
1163
1245
|
print("DOCX", "yes")
|
|
1164
1246
|
except Exception:
|
|
1165
1247
|
print("DOCX", "no")
|
|
1248
|
+
try:
|
|
1249
|
+
import extract_msg, markdownify
|
|
1250
|
+
print("EMAIL", "yes")
|
|
1251
|
+
except Exception:
|
|
1252
|
+
print("EMAIL", "no")
|
|
1166
1253
|
`;
|
|
1167
1254
|
var PROBE_TIMEOUT_MS = 5e3;
|
|
1168
1255
|
function probeArgs() {
|
|
@@ -1170,8 +1257,8 @@ function probeArgs() {
|
|
|
1170
1257
|
}
|
|
1171
1258
|
var PYTHON_CANDIDATES = ["python3", "python"];
|
|
1172
1259
|
function parseProbeOutput(stdout) {
|
|
1173
|
-
const m = stdout.match(/^PY (\d+) (\d+)\r?\nPDF (yes|no)\r?\nXLSX (yes|no)\r?\nDOCX (yes|no)\s*$/);
|
|
1174
|
-
return m ? { major: Number(m[1]), minor: Number(m[2]), pdf: m[3] === "yes", xlsx: m[4] === "yes", docx: m[5] === "yes" } : null;
|
|
1260
|
+
const m = stdout.match(/^PY (\d+) (\d+)\r?\nPDF (yes|no)\r?\nXLSX (yes|no)\r?\nDOCX (yes|no)\r?\nEMAIL (yes|no)\s*$/);
|
|
1261
|
+
return m ? { major: Number(m[1]), minor: Number(m[2]), pdf: m[3] === "yes", xlsx: m[4] === "yes", docx: m[5] === "yes", email: m[6] === "yes" } : null;
|
|
1175
1262
|
}
|
|
1176
1263
|
function meetsFloor(p) {
|
|
1177
1264
|
return p.major > 3 || p.major === 3 && p.minor >= 12;
|
|
@@ -1196,7 +1283,7 @@ async function resolveBackend(cfg, deps, signal) {
|
|
|
1196
1283
|
};
|
|
1197
1284
|
const warm = await deps.run("uv", warmArgs(cfg), { timeoutMs: left(), capBytes: OUTPUT_MAX_BYTES, env: deps.env, signal });
|
|
1198
1285
|
if (signal?.aborted) throw new Error("aborted");
|
|
1199
|
-
if (warm.code === 0 && !warm.timedOut) return { kind: "uv", pdf: true, xlsx: true, docx: true };
|
|
1286
|
+
if (warm.code === 0 && !warm.timedOut) return { kind: "uv", pdf: true, xlsx: true, docx: true, email: true };
|
|
1200
1287
|
const uvAbsent = warm.code === null && !warm.timedOut;
|
|
1201
1288
|
const isDeadline = (result) => result !== null && "kind" in result;
|
|
1202
1289
|
const probe = async (exe, tmp) => {
|
|
@@ -1212,19 +1299,19 @@ async function resolveBackend(cfg, deps, signal) {
|
|
|
1212
1299
|
const p = await probe(exe);
|
|
1213
1300
|
if (isDeadline(p)) return p;
|
|
1214
1301
|
if (!p || !meetsFloor(p)) continue;
|
|
1215
|
-
if (p.pdf) return { kind: "python", exe, pdf: true, xlsx: p.xlsx, docx: p.docx };
|
|
1302
|
+
if (p.pdf) return { kind: "python", exe, pdf: true, xlsx: p.xlsx, docx: p.docx, email: p.email };
|
|
1216
1303
|
eligible ??= { exe, version: `${p.major}.${p.minor}` };
|
|
1217
1304
|
}
|
|
1218
|
-
const healthy = (p) => meetsFloor(p) && p.pdf && p.xlsx && p.docx;
|
|
1305
|
+
const healthy = (p) => meetsFloor(p) && p.pdf && p.xlsx && p.docx && p.email;
|
|
1219
1306
|
const venvDir = join3(deps.cacheRoot, VENV_DIR_NAME);
|
|
1220
1307
|
const venvExe = venvPython(venvDir, deps.platform);
|
|
1221
1308
|
const cached = await probe(venvExe);
|
|
1222
1309
|
if (isDeadline(cached)) return cached;
|
|
1223
|
-
if (cached && healthy(cached)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
|
|
1310
|
+
if (cached && healthy(cached)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
|
|
1224
1311
|
if (eligible) {
|
|
1225
1312
|
const recheck = await probe(venvExe);
|
|
1226
1313
|
if (isDeadline(recheck)) return recheck;
|
|
1227
|
-
if (recheck && healthy(recheck)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
|
|
1314
|
+
if (recheck && healthy(recheck)) return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
|
|
1228
1315
|
const tmp = `${venvDir}.tmp-${deps.pid}`;
|
|
1229
1316
|
const bootFail = (stderr) => {
|
|
1230
1317
|
deps.rmrf(tmp);
|
|
@@ -1250,18 +1337,18 @@ async function resolveBackend(cfg, deps, signal) {
|
|
|
1250
1337
|
};
|
|
1251
1338
|
if (publish()) {
|
|
1252
1339
|
for (const legacy of LEGACY_VENV_DIR_NAMES) deps.rmrf(join3(deps.cacheRoot, legacy));
|
|
1253
|
-
return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
|
|
1340
|
+
return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
|
|
1254
1341
|
}
|
|
1255
1342
|
const winner = await probe(venvExe, tmp);
|
|
1256
1343
|
if (isDeadline(winner)) return winner;
|
|
1257
1344
|
if (winner && healthy(winner)) {
|
|
1258
1345
|
deps.rmrf(tmp);
|
|
1259
|
-
return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
|
|
1346
|
+
return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
|
|
1260
1347
|
}
|
|
1261
1348
|
deps.rmrf(venvDir);
|
|
1262
1349
|
if (publish()) {
|
|
1263
1350
|
for (const legacy of LEGACY_VENV_DIR_NAMES) deps.rmrf(join3(deps.cacheRoot, legacy));
|
|
1264
|
-
return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true };
|
|
1351
|
+
return { kind: "venv", exe: venvExe, pdf: true, xlsx: true, docx: true, email: true };
|
|
1265
1352
|
}
|
|
1266
1353
|
return bootFail("rename after competing bootstrap");
|
|
1267
1354
|
}
|
|
@@ -1337,7 +1424,7 @@ var DATA_SIGNATURES = {
|
|
|
1337
1424
|
};
|
|
1338
1425
|
async function prepareHtml(inputPath, stagingDir) {
|
|
1339
1426
|
const { JSDOM: JSDOM2 } = await import("jsdom");
|
|
1340
|
-
const bytes =
|
|
1427
|
+
const bytes = readFileSync2(inputPath);
|
|
1341
1428
|
let source = bytes;
|
|
1342
1429
|
try {
|
|
1343
1430
|
source = new TextDecoder("utf-8", { fatal: true }).decode(bytes);
|
|
@@ -1479,18 +1566,21 @@ async function convertDocument(o, signal, seams) {
|
|
|
1479
1566
|
const inputPath = resolve2(o.path);
|
|
1480
1567
|
const st = statSync2(inputPath, { throwIfNoEntry: false });
|
|
1481
1568
|
if (!st || !st.isFile()) throw new Error(`Not a readable file: ${o.path}`);
|
|
1569
|
+
if (st.size === 0) throw new Error(`empty file: ${o.path}`);
|
|
1482
1570
|
const type = classifyInput(inputPath);
|
|
1483
|
-
if (o.pages && (type === "html" || type === "image")) throw new Error(`--pages does not apply to ${type === "html" ? "HTML files" : "images"}`);
|
|
1484
|
-
const isExcel = type === "xlsx" || type === "xls";
|
|
1571
|
+
if (o.pages && (type === "html" || type === "image" || type === "email")) throw new Error(`--pages does not apply to ${type === "html" ? "HTML files" : type === "image" ? "images" : "email"}`);
|
|
1572
|
+
const isExcel = type === "xlsx" || type === "xlsm" || type === "xls";
|
|
1485
1573
|
if (isExcel && o.pages) throw new Error("--pages does not apply to spreadsheets: worksheets have no stable page numbering");
|
|
1486
1574
|
const backend = await s.backend({ pymupdfVersion: o.pymupdfVersion, warmTimeoutMs: o.warmTimeoutMs });
|
|
1487
1575
|
if (isExcel && (backend.kind === "none" || !backend.xlsx)) throw new Error(`Excel conversion needs a Python backend with openpyxl, xlrd and pillow (${backend.kind === "none" ? backend.reason : `${backend.kind} lacks the Excel packages`}). ${EXCEL_REMEDY}`);
|
|
1576
|
+
if (type === "email" && extname3(inputPath).toLowerCase() === ".msg" && (backend.kind === "none" || !backend.email)) throw new Error(`MSG conversion needs the extract-msg package. Python backend: ${backendState(backend, "found without extract-msg")}. Remedy: install uv, or pip install extract-msg markdownify into that Python`);
|
|
1577
|
+
if (type === "email" && extname3(inputPath).toLowerCase() !== ".msg" && lacksDocx(backend)) throw new Error(`EML conversion needs the Python DOCX/HTML packages (mammoth, markdownify, python-docx). Python backend: ${backendState(backend, "found without mammoth/markdownify/python-docx")}. ${docxRemedy}`);
|
|
1488
1578
|
const stem = sanitizeStem(basename(inputPath, extname3(inputPath)));
|
|
1489
1579
|
const b = openBundle(o.outputDir ? resolve2(o.outputDir) : tempBundleRoot(), stem, o.overwrite);
|
|
1490
1580
|
let office = null;
|
|
1491
1581
|
try {
|
|
1492
1582
|
let pdfPath = inputPath;
|
|
1493
|
-
const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
|
|
1583
|
+
const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
|
|
1494
1584
|
let tier, engine, json, degraded = null, fallbackReason = null;
|
|
1495
1585
|
let explicitBreaks = null;
|
|
1496
1586
|
let notes = [];
|
|
@@ -1576,13 +1666,23 @@ async function convertDocument(o, signal, seams) {
|
|
|
1576
1666
|
office = r;
|
|
1577
1667
|
pdfPath = r.pdfPath;
|
|
1578
1668
|
}
|
|
1579
|
-
if (type === "pptx") {
|
|
1669
|
+
if (type === "doc" || type === "pptx") {
|
|
1580
1670
|
const r = await s.office(o.sofficeTimeoutMs, inputPath, signal);
|
|
1581
|
-
if (!r.ok && r.kind === "missing") throw new Error(
|
|
1671
|
+
if (!r.ok && r.kind === "missing") throw new Error(`${type.toUpperCase()} conversion needs LibreOffice (soffice); direct conversion is not available. Python backend: ${backendState(backend, "available")}. Remedy: install LibreOffice`);
|
|
1582
1672
|
if (!r.ok) throw officeFailure(r);
|
|
1583
1673
|
office = r;
|
|
1584
1674
|
pdfPath = r.pdfPath;
|
|
1585
1675
|
}
|
|
1676
|
+
if (type === "email") {
|
|
1677
|
+
const isMsg = extname3(inputPath).toLowerCase() === ".msg";
|
|
1678
|
+
const r = await s.runTier("email", { ...base, stem: b.stem, attachmentsStagingDir: b.attachmentsStagingDir }, b, signal, o.primaryTimeoutMs, backend);
|
|
1679
|
+
if (!r.ok) throw new Error("userError" in r ? r.userError : `Conversion failed: email ${r.reason}${detailSuffix(r)}`);
|
|
1680
|
+
publishStaged(b);
|
|
1681
|
+
publishAttachments(b);
|
|
1682
|
+
tier = "email";
|
|
1683
|
+
engine = isMsg ? "extract-msg" : "email";
|
|
1684
|
+
json = r.json;
|
|
1685
|
+
}
|
|
1586
1686
|
const pdfBase = { ...base, path: pdfPath };
|
|
1587
1687
|
if (json === void 0) {
|
|
1588
1688
|
if (isExcel) {
|
|
@@ -1597,7 +1697,7 @@ async function convertDocument(o, signal, seams) {
|
|
|
1597
1697
|
tier = "excel";
|
|
1598
1698
|
engine = type === "xls" ? "xlrd" : "openpyxl";
|
|
1599
1699
|
json = r.json;
|
|
1600
|
-
notes =
|
|
1700
|
+
notes = (json.notes ?? []).map((n) => n.replace(/sheets\/[^\s,;]+/g, (m) => b.sourceMap.get(m) ?? m));
|
|
1601
1701
|
const renderPages = json.renderPages ?? [];
|
|
1602
1702
|
let skip = null;
|
|
1603
1703
|
const perSheet = /* @__PURE__ */ new Map();
|
|
@@ -1636,12 +1736,15 @@ async function convertDocument(o, signal, seams) {
|
|
|
1636
1736
|
tier = "primary";
|
|
1637
1737
|
engine = "pymupdf4llm";
|
|
1638
1738
|
json = p.json;
|
|
1739
|
+
if (json.pageImages?.length) publishPageImages(b, json.pageCount ?? 0);
|
|
1639
1740
|
} else if ("userError" in p) throw new Error(p.userError);
|
|
1640
1741
|
else {
|
|
1641
1742
|
if (signal?.aborted) throw new Error("aborted");
|
|
1642
1743
|
const keepPages = Object.fromEntries([...kept.entries()].map(([k, v]) => [String(k), v]));
|
|
1744
|
+
rmSync2(b.pagesStagingDir, { recursive: true, force: true });
|
|
1643
1745
|
const f = await s.runTier("pdf-fallback", { ...pdfBase, keepPages }, b, signal, o.fallbackTimeoutMs, backend);
|
|
1644
1746
|
publishStaged(b);
|
|
1747
|
+
if (f.ok && f.json.pageImages?.length) publishPageImages(b, f.json.pageCount ?? 0);
|
|
1645
1748
|
if (!f.ok) throw new Error("userError" in f ? f.userError : `Conversion failed: primary ${p.reason}; fallback ${f.reason}${detailSuffix(f)}`);
|
|
1646
1749
|
tier = "fallback";
|
|
1647
1750
|
engine = "pymupdf-text";
|
|
@@ -1651,14 +1754,14 @@ async function convertDocument(o, signal, seams) {
|
|
|
1651
1754
|
}
|
|
1652
1755
|
}
|
|
1653
1756
|
}
|
|
1654
|
-
if (officeRoute !== null)
|
|
1655
|
-
|
|
1656
|
-
fallbackReason = fallbackReason ? `${officeRoute}; ${fallbackReason}` : officeRoute;
|
|
1657
|
-
}
|
|
1757
|
+
if (type === "doc" || officeRoute !== null) degraded = DEGRADED_DOCX_OFFICE;
|
|
1758
|
+
if (officeRoute !== null) fallbackReason = fallbackReason ? `${officeRoute}; ${fallbackReason}` : officeRoute;
|
|
1658
1759
|
if (tier === void 0 || engine === void 0 || json === void 0) throw new Error("internal: no tier produced output");
|
|
1659
1760
|
if (!isExcel) notes = [...notes, ...json.notes ?? []];
|
|
1761
|
+
if (b.renamedFrom) notes.splice(notes[0]?.startsWith("preview truncated:") ? 1 : 0, 0, `renamed to ${b.stem} (${b.renameReason})`);
|
|
1762
|
+
const pageImagesReason = !o.pageImages || b.pageManifest.size || tier === "primary" || tier === "fallback" ? null : tier === "unpdf" ? "page images need the Python backend" : `${type} has no page geometry`;
|
|
1660
1763
|
const body = resolveOcrLabels(rewriteLinks(json.markdown ?? "", b.sourceMap), b.sourceMap);
|
|
1661
|
-
validateImageLinks(body, b.manifest, b.csvManifest, type === "html");
|
|
1764
|
+
validateImageLinks(body, b.manifest, b.csvManifest, type === "html" || type === "email", b.pageManifest, b.attachmentManifest);
|
|
1662
1765
|
const head = [];
|
|
1663
1766
|
if (degraded) head.push(`Degraded: ${degraded}`);
|
|
1664
1767
|
if (fallbackReason) head.push(`Fallback-Reason: ${fallbackReason}`);
|
|
@@ -1670,7 +1773,7 @@ async function convertDocument(o, signal, seams) {
|
|
|
1670
1773
|
` : "") + body;
|
|
1671
1774
|
commitBundle(b, markdown);
|
|
1672
1775
|
const outline = scanOutline(markdown, o.outlineMaxEntries);
|
|
1673
|
-
const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr: handleOcr(tier, type, o, json) };
|
|
1776
|
+
const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr: handleOcr(tier, type, o, json) };
|
|
1674
1777
|
return { output: formatHandle(details), details };
|
|
1675
1778
|
} catch (e) {
|
|
1676
1779
|
abortBundle(b);
|
|
@@ -1684,9 +1787,10 @@ async function inspectDocument(o, signal, seams) {
|
|
|
1684
1787
|
const inputPath = resolve2(o.path);
|
|
1685
1788
|
const st = statSync2(inputPath, { throwIfNoEntry: false });
|
|
1686
1789
|
if (!st || !st.isFile()) throw new Error(`Not a readable file: ${o.path}`);
|
|
1790
|
+
if (st.size === 0) throw new Error(`empty file: ${o.path}`);
|
|
1687
1791
|
const type = classifyInput(inputPath);
|
|
1688
|
-
if (type === "html" || type === "image") throw new Error(`info does not apply to ${type === "html" ? "HTML files" : "images"}; convert directly`);
|
|
1689
|
-
const isExcel = type === "xlsx" || type === "xls";
|
|
1792
|
+
if (type === "html" || type === "image" || type === "email") throw new Error(`info does not apply to ${type === "html" ? "HTML files" : type === "image" ? "images" : "email"}; convert directly`);
|
|
1793
|
+
const isExcel = type === "xlsx" || type === "xlsm" || type === "xls";
|
|
1690
1794
|
const backend = await s.backend({ pymupdfVersion: o.pymupdfVersion, warmTimeoutMs: o.warmTimeoutMs });
|
|
1691
1795
|
if (isExcel && (backend.kind === "none" || !backend.xlsx)) throw new Error(`Excel inspection needs a Python backend with openpyxl, xlrd and pillow. ${EXCEL_REMEDY}`);
|
|
1692
1796
|
let office = null;
|
|
@@ -1694,7 +1798,7 @@ async function inspectDocument(o, signal, seams) {
|
|
|
1694
1798
|
let path = inputPath;
|
|
1695
1799
|
if (type === "docx") {
|
|
1696
1800
|
if (lacksDocx(backend)) throw new Error(`DOCX inspection needs the Python DOCX packages. Python backend: ${backendState(backend, "found without mammoth/markdownify/python-docx")}. ${docxRemedy}`);
|
|
1697
|
-
} else if (type === "pptx") {
|
|
1801
|
+
} else if (type === "pptx" || type === "doc") {
|
|
1698
1802
|
const r2 = await s.office(o.sofficeTimeoutMs, inputPath, signal);
|
|
1699
1803
|
if (!r2.ok) throw officeFailure(r2);
|
|
1700
1804
|
office = r2;
|
|
@@ -1715,13 +1819,15 @@ async function inspectDocument(o, signal, seams) {
|
|
|
1715
1819
|
}
|
|
1716
1820
|
|
|
1717
1821
|
// bin/pi-quiver.ts
|
|
1718
|
-
var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--info] [--pages <spec>] [--output-dir <dir>] [--overwrite] [tunable flags] <path> (--help for all flags)';
|
|
1822
|
+
var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--json] [--info] [--page-images] [--pages <spec>] [--output-dir <dir>] [--overwrite] [tunable flags] <path> (--help for all flags)';
|
|
1719
1823
|
function parseDocToMd(rest) {
|
|
1720
1824
|
if (rest.includes("--help") || rest.includes("-h")) return { ok: true, cmd: "doc-to-md-help" };
|
|
1825
|
+
const json = rest.includes("--json");
|
|
1826
|
+
const args = rest.filter((a) => a !== "--json");
|
|
1721
1827
|
const perCall = {};
|
|
1722
1828
|
let path;
|
|
1723
|
-
for (let i = 0; i <
|
|
1724
|
-
const arg =
|
|
1829
|
+
for (let i = 0; i < args.length; i++) {
|
|
1830
|
+
const arg = args[i];
|
|
1725
1831
|
if (!arg.startsWith("--")) {
|
|
1726
1832
|
if (path !== void 0) return { ok: false, error: `unexpected argument: ${arg}` };
|
|
1727
1833
|
path = arg;
|
|
@@ -1738,7 +1844,7 @@ function parseDocToMd(rest) {
|
|
|
1738
1844
|
perCall[d.key] = true;
|
|
1739
1845
|
continue;
|
|
1740
1846
|
}
|
|
1741
|
-
const value =
|
|
1847
|
+
const value = args[++i];
|
|
1742
1848
|
if (value === void 0) return { ok: false, error: `${arg} requires a value` };
|
|
1743
1849
|
if (d.type === "int") {
|
|
1744
1850
|
const n = Number(value);
|
|
@@ -1747,7 +1853,7 @@ function parseDocToMd(rest) {
|
|
|
1747
1853
|
} else perCall[d.key] = value;
|
|
1748
1854
|
}
|
|
1749
1855
|
if (!path) return { ok: false, error: "missing <path>" };
|
|
1750
|
-
return { ok: true, cmd: "doc-to-md", perCall: { ...perCall, path } };
|
|
1856
|
+
return { ok: true, cmd: "doc-to-md", perCall: { ...perCall, path }, json };
|
|
1751
1857
|
}
|
|
1752
1858
|
function cliAgentDir(env) {
|
|
1753
1859
|
const raw = env.PI_CODING_AGENT_DIR;
|
|
@@ -1765,7 +1871,7 @@ function readCliSettings(cwd, env, warn) {
|
|
|
1765
1871
|
if (!existsSync3(file)) continue;
|
|
1766
1872
|
let raw;
|
|
1767
1873
|
try {
|
|
1768
|
-
raw = JSON.parse(
|
|
1874
|
+
raw = JSON.parse(readFileSync3(file, "utf8")).quiver?.docToMd;
|
|
1769
1875
|
} catch {
|
|
1770
1876
|
warn(`pi-quiver: ${file} is not valid JSON; ignored.`);
|
|
1771
1877
|
continue;
|
|
@@ -1860,7 +1966,11 @@ ${USAGE}
|
|
|
1860
1966
|
}
|
|
1861
1967
|
try {
|
|
1862
1968
|
const r = o.info ? await inspectDocument(o) : await convertDocument(o);
|
|
1863
|
-
|
|
1969
|
+
if (parsed.json) {
|
|
1970
|
+
const payload = o.info ? r.details : (({ path, backend, pymupdfVersion, inputType, file, outputDir, ...handle }) => handle)(r.details);
|
|
1971
|
+
process.stdout.write(`${JSON.stringify(payload)}
|
|
1972
|
+
`);
|
|
1973
|
+
} else process.stdout.write(`${r.output}
|
|
1864
1974
|
`);
|
|
1865
1975
|
return 0;
|
|
1866
1976
|
} catch (err) {
|
package/extensions/doc_to_md.ts
CHANGED
|
@@ -35,8 +35,8 @@ export default function docToMdExtension(pi: ExtensionAPI) {
|
|
|
35
35
|
name: "doc_to_md",
|
|
36
36
|
label: "Convert doc to Markdown bundle",
|
|
37
37
|
description:
|
|
38
|
-
"Convert a local PDF/DOCX/PPTX/XLSX/XLS, HTML (.html/.htm) or image (.png .jpg .jpeg .tif .tiff .bmp .gif) to a Markdown bundle on disk and return a handle (Saved-To, Images-Dir, Page-Count, Outline, diagnostics) - the Markdown itself is never inlined; read the Saved-To file (offset/limit) for content. `info: true` returns page count, metadata and TOC (or the sheet inventory) without converting - use it to pick `pages`. `pages` selects inclusive 1-based pages (PDF/PPTX) or explicit-page-break segments (DOCX; rejected when the file has none); every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Images are always extracted into images/ with relative links. Primary engine pymupdf4llm, fallback PyMuPDF text (degraded, marked), pure-JS unpdf only when no Python backend exists. Excel yields a sheet inventory (worksheets + chartsheets), a full CSV per non-empty worksheet under sheets/, a bounded preview with formulas and cached values, merged/hidden disclosure, and rendered chart views when LibreOffice is available. DOCX converts directly (mammoth; python-docx text fallback) and keeps heading styles, so the handle Outline lists `L<line>` and `p<segment>` per heading; LibreOffice (soffice) is optional for DOCX (fallback route, degraded) and Excel rendered views, and required for PPTX. Use `outputDir` for a durable bundle; without it the bundle lands in a per-call temp dir. Input must be a local file path (use fetch first for URLs). HTML converts without Readability (markdownify; Turndown fallback): local and data: images are copied into images/, remote images stay links. Pages without a text layer
|
|
39
|
-
promptSnippet: "Convert a local PDF/DOCX/PPTX/XLSX/HTML/image to a Markdown bundle (handle returned; read Saved-To)",
|
|
38
|
+
"Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS, email (.msg/.eml), HTML (.html/.htm) or image (.png .jpg .jpeg .tif .tiff .bmp .gif) to a Markdown bundle on disk and return a handle (Saved-To, Images-Dir, Page-Count, Outline, diagnostics) - the Markdown itself is never inlined; read the Saved-To file (offset/limit) for content. `info: true` returns page count, metadata and TOC (or the sheet inventory) without converting - use it to pick `pages`. `pages` selects inclusive 1-based pages (PDF/PPTX) or explicit-page-break segments (DOCX; rejected when the file has none); every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Images are always extracted into images/ with relative links. Primary engine pymupdf4llm, fallback PyMuPDF text (degraded, marked), pure-JS unpdf only when no Python backend exists. Excel yields a sheet inventory (worksheets + chartsheets), a full CSV per non-empty worksheet under sheets/, a bounded preview with formulas and cached values, merged/hidden disclosure, and rendered chart views when LibreOffice is available. DOCX converts directly (mammoth; python-docx text fallback) and keeps heading styles, so the handle Outline lists `L<line>` and `p<segment>` per heading; LibreOffice (soffice) is optional for DOCX (fallback route, degraded) and Excel rendered views, and required for PPTX. Use `outputDir` for a durable bundle; without it the bundle lands in a per-call temp dir. Input must be a local file path (use fetch first for URLs). HTML converts without Readability (markdownify; Turndown fallback): local and data: images are copied into images/, remote images stay links. Pages without a text layer keep their picture in images/, or pages/ with `pageImages: true`; image inputs keep their picture in images/. `pageImages: true` writes page renders to pages/<stem>-pNNN, and email attachments are saved under attachments/. OCR is off by default: `ocr: true` adds recognized text (labeled as OCR) when Tesseract language data is installed; the handle's OCR: line says what ran.",
|
|
39
|
+
promptSnippet: "Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS/MSG/EML/HTML/image to a Markdown bundle (handle returned; read Saved-To)",
|
|
40
40
|
parameters: buildSchema(),
|
|
41
41
|
|
|
42
42
|
async execute(_toolCallId, params, signal, _onUpdate, ctx) {
|