pi-quiver 6.9.0 → 6.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +5 -0
- package/README.md +1 -1
- package/dist/bin/pi-quiver.js +75 -15
- package/lib/doc-to-md-bundle.ts +30 -9
- package/lib/doc-to-md-core.ts +27 -8
- package/lib/doc-to-md-handle.ts +6 -0
- package/lib/doc-to-md-options.ts +21 -3
- package/package.json +1 -1
- package/scripts/doc_to_md.py +134 -6
package/CHANGELOG.md
CHANGED
|
@@ -8,6 +8,11 @@ Published to npm as `pi-quiver` (`pi install npm:pi-quiver`). Pushing a
|
|
|
8
8
|
via OIDC trusted publishing. The release helper at
|
|
9
9
|
`.agents/skills/release/scripts/release.sh` cuts the tag; CI publishes.
|
|
10
10
|
|
|
11
|
+
## v6.10.0 - 2026-10-02
|
|
12
|
+
|
|
13
|
+
- `doc_to_md`: new per-call `words` option (`--words`; not settable) writes `<stem>.words.json` - per selected PDF/image page, every text-layer word and every word inline OCR recognized in the same run, with display-space bbox (points; source pixels for images) and `source` `text`/`ocr`. Under `--ocr-mode all` each sidecar gets `ocr/<stem>-pNNN.words.json`. Never triggers OCR; Markdown, page stats and OCR outcome are unchanged. Handle gains `Words:`, `--json` gains `wordsPath`, `wordsReason`, `wordsErrors`, `ocr.wordSidecars`; `--info --words` is a usage error (#28).
|
|
14
|
+
- `doc_to_md`: `--help` and the generated skill list every bundle artifact under `Bundle layout` (#28).
|
|
15
|
+
|
|
11
16
|
## v6.9.0 - 2026-10-01
|
|
12
17
|
|
|
13
18
|
- `doc_to_md`: every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page `chars`, `images`, `imageCoverage`) and the handle prints `Page-Stats:`; CLI `--json` gains `pageStats`, `pageStatsPath`, `ocrDir`. The unpdf tier writes no stats (#26).
|
package/README.md
CHANGED
|
@@ -65,7 +65,7 @@ A 300 KB changelog page never touches your context window - you get a preview an
|
|
|
65
65
|
| Extension | Tool | What it does |
|
|
66
66
|
| --- | --- | --- |
|
|
67
67
|
| `extensions/fetch.ts` | `fetch` | Retrieve URLs over HTTP(S). HTML -> Markdown (Readability extraction, Turndown conversion). Binary saved untouched to a temp file. GitHub issue/PR/repo/actions-run/actions-job URLs auto-route through `gh` (falls back to HTTP); failed runs/jobs include failed-step logs (best-effort, summary-only otherwise). Same size gate as `fetch`. Behavior lives in `lib/fetch-core.ts`; also exposed as the `pi-quiver fetch` CLI (see [Claude Code support](#claude-code-support)). |
|
|
68
|
-
| `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS, MSG/EML, HTML, or image file to a Markdown bundle on disk (`<stem>.md` + `images/`, spreadsheet `sheets/`, optional `pageImages` in `pages/`, and email `attachments/`) and return a handle (paths, page count, outline, diagnostics) - never inline Markdown. `info` inspects PDF/Office inputs; `pages` selects 1-based PDF/Office pages (DOCX: explicit-page-break segments); HTML, image, and email inputs reject both; every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Tiers: pymupdf4llm -> PyMuPDF text (degraded) -> unpdf worker (no Python only). Excel -> sheet inventory, full CSVs, bounded previews, and optional rendered views. HTML keeps local and `data:` images in the bundle and remote images as links; image inputs keep the original image. Scanned PDF pages keep a page picture on both Python tiers; opt-in OCR needs Tesseract language data. DOCX converts directly (mammoth, python-docx fallback); LibreOffice pagination marks the degraded route. The Outline lists `L<line>` and `p<page>` per heading. Settings under `quiver.docToMd`. Every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page chars, image count, image coverage; handle line `Page-Stats:`); unpdf writes no stats. `ocrMode: "all"` with `ocr: true` and an explicit `pages` forces OCR on those pages into `ocr/<stem>-pNNN.md` sidecars, leaving the Markdown byte-identical to the same selected-page call without OCR. |
|
|
68
|
+
| `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS, MSG/EML, HTML, or image file to a Markdown bundle on disk (`<stem>.md` + `images/`, spreadsheet `sheets/`, optional `pageImages` in `pages/`, and email `attachments/`) and return a handle (paths, page count, outline, diagnostics) - never inline Markdown. `info` inspects PDF/Office inputs; `pages` selects 1-based PDF/Office pages (DOCX: explicit-page-break segments); HTML, image, and email inputs reject both; every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Tiers: pymupdf4llm -> PyMuPDF text (degraded) -> unpdf worker (no Python only). Excel -> sheet inventory, full CSVs, bounded previews, and optional rendered views. HTML keeps local and `data:` images in the bundle and remote images as links; image inputs keep the original image. Scanned PDF pages keep a page picture on both Python tiers; opt-in OCR needs Tesseract language data. DOCX converts directly (mammoth, python-docx fallback); LibreOffice pagination marks the degraded route. The Outline lists `L<line>` and `p<page>` per heading. Settings under `quiver.docToMd`. Every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page chars, image count, image coverage; handle line `Page-Stats:`); unpdf writes no stats. `ocrMode: "all"` with `ocr: true` and an explicit `pages` forces OCR on those pages into `ocr/<stem>-pNNN.md` sidecars, leaving the Markdown byte-identical to the same selected-page call without OCR. `words: true` (CLI `--words`, per-call only - no settings key) writes `<stem>.words.json` with every word's bbox per selected PDF/image page (`source` `text` or `ocr`), plus `ocr/<stem>-pNNN.words.json` beside each forced-OCR sidecar; it never triggers OCR. |
|
|
69
69
|
| `extensions/session-name.ts` | `/session-name` | Manual + opt-in automatic session naming, naming rules and deny list, long-session revisits, and Ghostty/Herdr tab rename. OFF by default. |
|
|
70
70
|
| `extensions/sword-header.ts` | `/builtin-header` | Themed ASCII startup header replacing pi's default logo. OFF by default. |
|
|
71
71
|
| `extensions/fast-mode.ts` | `/fast` | Inject Anthropic fast-mode (`speed: "fast"` + `anthropic-beta: fast-mode-2026-02-01`) into every Claude Opus 4.8 / Opus 5 request, any thinking level. `--fast` flag + `/fast [on\|off\|status]`. OFF by default. |
|
package/dist/bin/pi-quiver.js
CHANGED
|
@@ -582,7 +582,7 @@ function ownedPagePattern(stem) {
|
|
|
582
582
|
return new RegExp(`^${escRe(stem)}-p\\d+\\.[a-z0-9]+$`);
|
|
583
583
|
}
|
|
584
584
|
function ownedOcrPattern(stem) {
|
|
585
|
-
return new RegExp(`^${escRe(stem)}-p\\d
|
|
585
|
+
return new RegExp(`^${escRe(stem)}-p\\d+(?:\\.words\\.json|\\.md)$`);
|
|
586
586
|
}
|
|
587
587
|
var FILE_LINK_RE = /\[[^\]]*\]\(\s*((?:sheets|attachments)\/[^)\s]+)\s*\)/g;
|
|
588
588
|
var IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
|
|
@@ -649,6 +649,7 @@ function openBundle(root, requested, overwrite) {
|
|
|
649
649
|
for (const f of readdirSync(attachmentsDir)) if (attachments.has(f)) rmSync(join2(attachmentsDir, f), { force: true });
|
|
650
650
|
}
|
|
651
651
|
rmSync(join2(root, `${stem}.pages.json`), { force: true });
|
|
652
|
+
rmSync(join2(root, `${stem}.words.json`), { force: true });
|
|
652
653
|
const ownedOcr = ownedOcrPattern(stem);
|
|
653
654
|
if (existsSync(ocrDir)) {
|
|
654
655
|
for (const f of readdirSync(ocrDir)) if (ownedOcr.test(f)) rmSync(join2(ocrDir, f), { force: true });
|
|
@@ -661,7 +662,7 @@ function openBundle(root, requested, overwrite) {
|
|
|
661
662
|
const pagesStagingDir = join2(pagesDir, `.stage-${lockId}`);
|
|
662
663
|
const attachmentsStagingDir = join2(attachmentsDir, `.stage-${lockId}`);
|
|
663
664
|
const ocrStagingDir = join2(ocrDir, `.stage-${lockId}`);
|
|
664
|
-
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join2(root, `${stem}.pages.json`), ocrManifest: /* @__PURE__ */ new Set(), manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), pageManifest: /* @__PURE__ */ new Set(), attachmentManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
|
|
665
|
+
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join2(root, `${stem}.pages.json`), wordsPath: join2(root, `${stem}.words.json`), ocrManifest: /* @__PURE__ */ new Set(), manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), pageManifest: /* @__PURE__ */ new Set(), attachmentManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
|
|
665
666
|
} catch (e) {
|
|
666
667
|
rmSync(lockPath, { force: true });
|
|
667
668
|
throw e;
|
|
@@ -729,9 +730,20 @@ function writePageStats(b, stats) {
|
|
|
729
730
|
writeFileSync2(b.pageStatsPath, `${JSON.stringify(stats, null, 2)}
|
|
730
731
|
`, "utf8");
|
|
731
732
|
}
|
|
733
|
+
function publishWords(b) {
|
|
734
|
+
const staged = join2(b.stagingDir, "words.json");
|
|
735
|
+
try {
|
|
736
|
+
if (!existsSync(staged)) throw new Error("child staged no words.json");
|
|
737
|
+
renameSync(staged, b.wordsPath);
|
|
738
|
+
return null;
|
|
739
|
+
} catch (e) {
|
|
740
|
+
rmSync(b.wordsPath, { force: true });
|
|
741
|
+
return `write failed - ${e.message}`;
|
|
742
|
+
}
|
|
743
|
+
}
|
|
732
744
|
function publishSidecars(b) {
|
|
733
|
-
const
|
|
734
|
-
if (!existsSync(b.ocrStagingDir)) return
|
|
745
|
+
const sidecars = /* @__PURE__ */ new Map(), wordSidecars = /* @__PURE__ */ new Map();
|
|
746
|
+
if (!existsSync(b.ocrStagingDir)) return { sidecars, wordSidecars };
|
|
735
747
|
for (const dir of readdirSync(b.ocrStagingDir).sort()) {
|
|
736
748
|
const m = dir.match(/^p(\d+)$/);
|
|
737
749
|
if (!m) continue;
|
|
@@ -742,10 +754,16 @@ function publishSidecars(b) {
|
|
|
742
754
|
mkdirSync2(b.ocrDir, { recursive: true });
|
|
743
755
|
renameSync(join2(pageDir, file), join2(b.ocrDir, file));
|
|
744
756
|
b.ocrManifest.add(file);
|
|
745
|
-
|
|
757
|
+
sidecars.set(Number(m[1]), join2(b.ocrDir, file));
|
|
758
|
+
const wfile = `${b.stem}-${dir}.words.json`;
|
|
759
|
+
if (existsSync(join2(pageDir, wfile))) {
|
|
760
|
+
renameSync(join2(pageDir, wfile), join2(b.ocrDir, wfile));
|
|
761
|
+
b.ocrManifest.add(wfile);
|
|
762
|
+
wordSidecars.set(Number(m[1]), join2(b.ocrDir, wfile));
|
|
763
|
+
}
|
|
746
764
|
}
|
|
747
765
|
rmSync(b.ocrStagingDir, { recursive: true, force: true });
|
|
748
|
-
return
|
|
766
|
+
return { sidecars, wordSidecars };
|
|
749
767
|
}
|
|
750
768
|
function publishAttachments(b) {
|
|
751
769
|
if (!existsSync(b.attachmentsStagingDir)) return;
|
|
@@ -821,6 +839,7 @@ function abortBundle(b) {
|
|
|
821
839
|
for (const f of b.ocrManifest) rmSync(join2(b.ocrDir, f), { force: true });
|
|
822
840
|
rmSync(b.ocrStagingDir, { recursive: true, force: true });
|
|
823
841
|
rmSync(b.pageStatsPath, { force: true });
|
|
842
|
+
rmSync(b.wordsPath, { force: true });
|
|
824
843
|
rmSync(`${b.mdPath}.tmp`, { force: true });
|
|
825
844
|
rmSync(b.stagingDir, { recursive: true, force: true });
|
|
826
845
|
rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|
|
@@ -950,6 +969,10 @@ function formatHandle(h) {
|
|
|
950
969
|
if (h.pagesDir && h.pageImageCount > 0) lines.push(`Pages-Dir: ${h.pagesDir} (${h.pageImageCount} pages)`);
|
|
951
970
|
else if (h.pageImagesReason) lines.push(`Pages-Dir: none - ${h.pageImagesReason}`);
|
|
952
971
|
if (h.pageStatsPath) lines.push(`Page-Stats: ${h.pageStatsPath}`);
|
|
972
|
+
if (h.wordsPath) {
|
|
973
|
+
const bad = Object.keys(h.wordsErrors).map(Number).sort((a, b) => a - b);
|
|
974
|
+
lines.push(`Words: ${h.wordsPath}${bad.length ? ` (extraction failed for pages ${compactRanges(bad)}: ${h.wordsErrors[bad[0]]})` : ""}`);
|
|
975
|
+
} else if (h.wordsReason) lines.push(`Words: ${h.wordsReason}`);
|
|
953
976
|
if (h.ocrDir) lines.push(`OCR-Dir: ${h.ocrDir}`);
|
|
954
977
|
lines.push(`Type: ${h.type} Engine: ${h.engine} Tier: ${h.tier}`);
|
|
955
978
|
lines.push(`Page-Count: ${pageCountLabel(h)} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize2(h.bytes)} / ${h.lines} lines`);
|
|
@@ -1000,6 +1023,7 @@ var DOC_TO_MD_OPTIONS = [
|
|
|
1000
1023
|
{ key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir. <stem> = basename without extension with [^A-Za-z0-9._-]+ -> _ (empty -> document); a second call on the same stem writes <stem>-2.md" },
|
|
1001
1024
|
{ key: "overwrite", flag: "--overwrite", type: "bool", default: false, settable: false, help: "Replace an existing completed <stem>.md bundle" },
|
|
1002
1025
|
{ key: "pageImages", flag: "--page-images", type: "bool", default: false, settable: false, help: "Also render every selected page to pages/<stem>-pNNN.<imageFormat> at imageDpi (PDF, PPTX, .doc, DOCX via LibreOffice); off by default" },
|
|
1026
|
+
{ key: "words", flag: "--words", type: "bool", default: false, settable: false, help: 'Write word positions: <stem>.words.json beside the Markdown lists every text-layer word of each selected page with its bbox (PDF points, top-left origin, display orientation; image inputs in source pixels) and the words inline OCR recognized, tagged source "text" or "ocr"; under --ocr-mode all the OCR words go to ocr/<stem>-pNNN.words.json beside each sidecar. Never triggers OCR. PDF and image inputs only.' },
|
|
1003
1027
|
{ key: "ocrMode", flag: "--ocr-mode", type: "enum", default: "textless", settable: false, enumValues: ["textless", "all"], help: "OCR policy: textless (default) OCRs only pages with an empty text layer, inline; all OCRs every selected page and writes the recognized text to ocr/<stem>-pNNN.md sidecars, leaving the Markdown untouched. all requires --ocr and an explicit --pages selection (PDF, PPTX, DOC)." },
|
|
1004
1028
|
{ key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 6e4, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier and DOCX child (docx mode); also the unpdf tier" },
|
|
1005
1029
|
{ key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier (including DOCX LibreOffice fallback); also PDF and DOCX info and Excel rendered views" },
|
|
@@ -1107,8 +1131,8 @@ function resolveOptions(perCall, settings, env) {
|
|
|
1107
1131
|
out[d.key] = value;
|
|
1108
1132
|
}
|
|
1109
1133
|
const o = out;
|
|
1110
|
-
if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages || o.ocrMode === "all")) {
|
|
1111
|
-
throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite, --page-images or --ocr-mode all");
|
|
1134
|
+
if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages || o.words || o.ocrMode === "all")) {
|
|
1135
|
+
throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite, --page-images, --words or --ocr-mode all");
|
|
1112
1136
|
}
|
|
1113
1137
|
return o;
|
|
1114
1138
|
}
|
|
@@ -1129,6 +1153,18 @@ var USAGE_PATTERNS = [
|
|
|
1129
1153
|
" Details: doc/doc-to-md.md (bundle contract, failure buckets)."
|
|
1130
1154
|
].join("\n");
|
|
1131
1155
|
var usagePatterns = (cmd) => USAGE_PATTERNS.replaceAll("<cmd>", cmd);
|
|
1156
|
+
var BUNDLE_LAYOUT = [
|
|
1157
|
+
{ artifact: "<stem>.md", trigger: "always", content: "the Markdown", namedBy: "Saved-To: / savedTo" },
|
|
1158
|
+
{ artifact: "images/", trigger: "embedded or extracted figures", content: "image files linked from the Markdown", namedBy: "Images-Dir: / imagesDir" },
|
|
1159
|
+
{ artifact: "pages/<stem>-pNNN.<fmt>", trigger: "--page-images", content: "page renders at --image-dpi", namedBy: "Pages-Dir: / pagesDir" },
|
|
1160
|
+
{ artifact: "sheets/", trigger: "Excel input", content: "one CSV per non-empty worksheet", namedBy: "Sheets-Dir: / sheetsDir" },
|
|
1161
|
+
{ artifact: "attachments/", trigger: "email input", content: "saved attachments", namedBy: "Markdown attachment list" },
|
|
1162
|
+
{ artifact: "<stem>.pages.json", trigger: "Python PDF tiers (PDF, PPTX, DOC, DOCX via LibreOffice; not unpdf)", content: "per-page chars, image count, image coverage", namedBy: "Page-Stats: / pageStatsPath" },
|
|
1163
|
+
{ artifact: "<stem>.words.json", trigger: "--words", content: "per-page word boxes, source text/ocr", namedBy: "Words: / wordsPath" },
|
|
1164
|
+
{ artifact: "ocr/<stem>-pNNN.md", trigger: "--ocr --ocr-mode all", content: "recognized text of a forced page", namedBy: "OCR-Dir: / ocr.sidecars" },
|
|
1165
|
+
{ artifact: "ocr/<stem>-pNNN.words.json", trigger: "--ocr --ocr-mode all --words", content: "word boxes of that OCR", namedBy: "ocr.wordSidecars" }
|
|
1166
|
+
];
|
|
1167
|
+
var bundleLayoutText = () => BUNDLE_LAYOUT.map((r) => ` ${r.artifact.padEnd(28)} ${r.trigger}; ${r.content}; named by ${r.namedBy}`).join("\n");
|
|
1132
1168
|
function renderHelp() {
|
|
1133
1169
|
const row = (d) => ` ${(d.flag ?? "<path>").padEnd(26)} ${d.help}${d.default !== null && d.key !== "info" && d.key !== "overwrite" ? ` (default ${d.default})` : ""}`;
|
|
1134
1170
|
return [
|
|
@@ -1143,6 +1179,9 @@ function renderHelp() {
|
|
|
1143
1179
|
"Result: a handle (Saved-To, Images-Dir, Page-Stats, Page-Count, Outline ...). Read the Saved-To file for the Markdown.",
|
|
1144
1180
|
"Exit codes: 0 success, 1 runtime error, 2 usage error.",
|
|
1145
1181
|
"",
|
|
1182
|
+
"Bundle layout:",
|
|
1183
|
+
bundleLayoutText(),
|
|
1184
|
+
"",
|
|
1146
1185
|
usagePatterns("pi-quiver doc-to-md")
|
|
1147
1186
|
].join("\n");
|
|
1148
1187
|
}
|
|
@@ -1622,7 +1661,7 @@ function reconcileRenderMarkers(md, renderPages, fmt, sourceMap, reason) {
|
|
|
1622
1661
|
if (/<!--rvs?:\d+-->/.test(md)) throw new Error("internal: unresolved render marker");
|
|
1623
1662
|
return md;
|
|
1624
1663
|
}
|
|
1625
|
-
var emptyOcr = (lang) => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null, mode: "textless", sidecars: {}, ocrErrors: {}, killed: null, notAttempted: [], childError: null });
|
|
1664
|
+
var emptyOcr = (lang) => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null, mode: "textless", sidecars: {}, wordSidecars: {}, ocrErrors: {}, killed: null, notAttempted: [], childError: null });
|
|
1626
1665
|
var emptyOutcome = () => ({ written: [], noText: [], ocrFailed: [], ocrErrors: {}, budgetStopped: [], killed: null, notAttempted: [], childError: null });
|
|
1627
1666
|
var ocrPagesTag = (n) => `p${String(n).padStart(3, "0")}`;
|
|
1628
1667
|
var SIDE_MARKER_RE = /^--- end of page\.page_number=\d+ ---$/;
|
|
@@ -1693,11 +1732,12 @@ async function convertDocument(o, signal, seams) {
|
|
|
1693
1732
|
let office = null;
|
|
1694
1733
|
try {
|
|
1695
1734
|
let pdfPath = inputPath;
|
|
1696
|
-
const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr && !forced, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
|
|
1735
|
+
const base = { path: inputPath, pages: o.pages, ...o.words && (type === "pdf" || type === "image") ? { words: true } : {}, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr && !forced, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
|
|
1697
1736
|
let tier, engine, json, degraded = null, fallbackReason = null;
|
|
1698
1737
|
let explicitBreaks = null;
|
|
1699
1738
|
let notes = [];
|
|
1700
1739
|
let officeRoute = null;
|
|
1740
|
+
let copyReason = null;
|
|
1701
1741
|
if (type === "html") {
|
|
1702
1742
|
const prepared = await prepareHtml(inputPath, b.stagingDir);
|
|
1703
1743
|
if (signal?.aborted) throw new Error("aborted");
|
|
@@ -1737,6 +1777,7 @@ async function convertDocument(o, signal, seams) {
|
|
|
1737
1777
|
}
|
|
1738
1778
|
if (signal?.aborted) throw new Error("aborted");
|
|
1739
1779
|
if (json === void 0) {
|
|
1780
|
+
copyReason = reason;
|
|
1740
1781
|
clearStaging(b);
|
|
1741
1782
|
const file = `original${extname3(inputPath).toLowerCase()}`;
|
|
1742
1783
|
const dir = join3(b.stagingDir, "p1");
|
|
@@ -1872,14 +1913,33 @@ async function convertDocument(o, signal, seams) {
|
|
|
1872
1913
|
if (tier === void 0 || engine === void 0 || json === void 0) throw new Error("internal: no tier produced output");
|
|
1873
1914
|
const pageStats = json.pageStats ?? null;
|
|
1874
1915
|
if (pageStats) writePageStats(b, pageStats);
|
|
1916
|
+
let wordsPath = null, wordsReason = null;
|
|
1917
|
+
const wordsErrors = {};
|
|
1918
|
+
const takeWordsErrors = (j) => {
|
|
1919
|
+
for (const [k, v] of Object.entries(j?.wordsErrors ?? {})) {
|
|
1920
|
+
const page = Number(k);
|
|
1921
|
+
if (Number.isFinite(page)) wordsErrors[page] = v;
|
|
1922
|
+
}
|
|
1923
|
+
};
|
|
1924
|
+
if (o.words) {
|
|
1925
|
+
if (type !== "pdf" && type !== "image") wordsReason = `none - word positions apply to PDF and image inputs only (${type})`;
|
|
1926
|
+
else if (tier === "unpdf") wordsReason = "none - unpdf tier has no page geometry";
|
|
1927
|
+
else if (engine === "copy") wordsReason = `none - image copied without conversion (${copyReason})`;
|
|
1928
|
+
else if (json?.words === true) {
|
|
1929
|
+
wordsReason = publishWords(b);
|
|
1930
|
+
if (wordsReason === null) wordsPath = b.wordsPath;
|
|
1931
|
+
} else wordsReason = `write failed - ${json?.wordsErrors?.file ?? "child reported no words document"}`;
|
|
1932
|
+
takeWordsErrors(json);
|
|
1933
|
+
}
|
|
1875
1934
|
let ocr = handleOcr(tier, type, o, json);
|
|
1876
1935
|
if (forced) {
|
|
1877
|
-
const r = await s.runTier("ocr-pages", { path: pdfPath, pages: o.pages, stem: b.stem, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, stagingDir: b.ocrStagingDir, dpi: o.imageDpi, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.primaryTimeoutMs, backend);
|
|
1936
|
+
const r = await s.runTier("ocr-pages", { path: pdfPath, pages: o.pages, ...o.words && type === "pdf" ? { words: true } : {}, stem: b.stem, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, stagingDir: b.ocrStagingDir, dpi: o.imageDpi, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.primaryTimeoutMs, backend);
|
|
1878
1937
|
if (signal?.aborted || !r.ok && "reason" in r && r.reason === "aborted") throw new Error("aborted");
|
|
1879
1938
|
if (r.ok && r.json.status === "unavailable") throw new Error(`OCR unavailable: ${r.json.reason} (install Tesseract; see doc/doc-to-md.md)`);
|
|
1880
1939
|
const outcome = r.ok && r.json.status === "ran" ? outcomeFromChild(r.json) : recoverOcrPages(b.ocrStagingDir, o.pages, !r.ok ? "userError" in r ? r.userError : `${r.reason}${detailSuffix(r)}` : "malformed child output");
|
|
1881
|
-
const sidecars = publishSidecars(b);
|
|
1882
|
-
ocr = { ...emptyOcr(o.ocrLanguage), ...json.ocr, status: "ran", reason: null, mode: "all", pages: outcome.written, noText: outcome.noText, ocrFailed: outcome.ocrFailed, ocrErrors: outcome.ocrErrors, budgetStopped: outcome.budgetStopped, killed: outcome.killed, notAttempted: outcome.notAttempted, childError: outcome.childError, sidecars: Object.fromEntries(sidecars) };
|
|
1940
|
+
const { sidecars, wordSidecars } = publishSidecars(b);
|
|
1941
|
+
ocr = { ...emptyOcr(o.ocrLanguage), ...json.ocr, status: "ran", reason: null, mode: "all", pages: outcome.written, noText: outcome.noText, ocrFailed: outcome.ocrFailed, ocrErrors: outcome.ocrErrors, budgetStopped: outcome.budgetStopped, killed: outcome.killed, notAttempted: outcome.notAttempted, childError: outcome.childError, sidecars: Object.fromEntries(sidecars), wordSidecars: Object.fromEntries(wordSidecars) };
|
|
1942
|
+
if (o.words && r.ok) takeWordsErrors(r.json);
|
|
1883
1943
|
}
|
|
1884
1944
|
if (!isExcel) notes = [...notes, ...json.notes ?? []];
|
|
1885
1945
|
if (b.renamedFrom) notes.splice(notes[0]?.startsWith("preview truncated:") ? 1 : 0, 0, `renamed to ${b.stem} (${b.renameReason})`);
|
|
@@ -1897,7 +1957,7 @@ async function convertDocument(o, signal, seams) {
|
|
|
1897
1957
|
` : "") + body;
|
|
1898
1958
|
commitBundle(b, markdown);
|
|
1899
1959
|
const outline = scanOutline(markdown, o.outlineMaxEntries);
|
|
1900
|
-
const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr, pageStats, pageStatsPath: pageStats ? b.pageStatsPath : null, ocrDir: b.ocrManifest.size ? b.ocrDir : null };
|
|
1960
|
+
const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr, pageStats, pageStatsPath: pageStats ? b.pageStatsPath : null, ocrDir: b.ocrManifest.size ? b.ocrDir : null, wordsPath, wordsReason, wordsErrors };
|
|
1901
1961
|
return { output: formatHandle(details), details };
|
|
1902
1962
|
} catch (e) {
|
|
1903
1963
|
abortBundle(b);
|
|
@@ -1943,7 +2003,7 @@ async function inspectDocument(o, signal, seams) {
|
|
|
1943
2003
|
}
|
|
1944
2004
|
|
|
1945
2005
|
// bin/pi-quiver.ts
|
|
1946
|
-
var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--json] [--info] [--page-images] [--pages <spec>] [--output-dir <dir>] [--overwrite] [--ocr] [--ocr-mode textless|all] [tunable flags] <path> (--help for all flags)';
|
|
2006
|
+
var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--json] [--info] [--page-images] [--words] [--pages <spec>] [--output-dir <dir>] [--overwrite] [--ocr] [--ocr-mode textless|all] [tunable flags] <path> (--help for all flags)';
|
|
1947
2007
|
function parseDocToMd(rest) {
|
|
1948
2008
|
if (rest.includes("--help") || rest.includes("-h")) return { ok: true, cmd: "doc-to-md-help" };
|
|
1949
2009
|
const json = rest.includes("--json");
|
package/lib/doc-to-md-bundle.ts
CHANGED
|
@@ -15,7 +15,7 @@ export interface Bundle {
|
|
|
15
15
|
sheetsDir: string; sheetsStagingDir: string;
|
|
16
16
|
pagesDir: string; pagesStagingDir: string;
|
|
17
17
|
attachmentsDir: string; attachmentsStagingDir: string;
|
|
18
|
-
ocrDir: string; ocrStagingDir: string; pageStatsPath: string;
|
|
18
|
+
ocrDir: string; ocrStagingDir: string; pageStatsPath: string; wordsPath: string;
|
|
19
19
|
manifest: Set<string>;
|
|
20
20
|
csvManifest: Set<string>;
|
|
21
21
|
pageManifest: Set<string>;
|
|
@@ -36,7 +36,7 @@ export function ownedCsvPattern(stem: string): RegExp {
|
|
|
36
36
|
|
|
37
37
|
export function ownedPagePattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d+\\.[a-z0-9]+$`); }
|
|
38
38
|
|
|
39
|
-
export function ownedOcrPattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d
|
|
39
|
+
export function ownedOcrPattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d+(?:\\.words\\.json|\\.md)$`); }
|
|
40
40
|
|
|
41
41
|
const FILE_LINK_RE = /\[[^\]]*\]\(\s*((?:sheets|attachments)\/[^)\s]+)\s*\)/g;
|
|
42
42
|
const IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
|
|
@@ -94,6 +94,7 @@ export function openBundle(root: string, requested: string, overwrite: boolean):
|
|
|
94
94
|
if (existsSync(pagesDir)) for (const f of readdirSync(pagesDir)) if (ownedPage.test(f)) rmSync(join(pagesDir, f), { force: true });
|
|
95
95
|
if (existsSync(attachmentsDir)) for (const f of readdirSync(attachmentsDir)) if (attachments.has(f)) rmSync(join(attachmentsDir, f), { force: true });
|
|
96
96
|
rmSync(join(root, `${stem}.pages.json`), { force: true });
|
|
97
|
+
rmSync(join(root, `${stem}.words.json`), { force: true });
|
|
97
98
|
const ownedOcr = ownedOcrPattern(stem);
|
|
98
99
|
if (existsSync(ocrDir)) for (const f of readdirSync(ocrDir)) if (ownedOcr.test(f)) rmSync(join(ocrDir, f), { force: true });
|
|
99
100
|
}
|
|
@@ -104,7 +105,7 @@ export function openBundle(root: string, requested: string, overwrite: boolean):
|
|
|
104
105
|
const pagesStagingDir = join(pagesDir, `.stage-${lockId}`);
|
|
105
106
|
const attachmentsStagingDir = join(attachmentsDir, `.stage-${lockId}`);
|
|
106
107
|
const ocrStagingDir = join(ocrDir, `.stage-${lockId}`);
|
|
107
|
-
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join(root, `${stem}.pages.json`), ocrManifest: new Set(), manifest: new Set(), csvManifest: new Set(), pageManifest: new Set(), attachmentManifest: new Set(), sourceMap: new Map() };
|
|
108
|
+
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join(root, `${stem}.pages.json`), wordsPath: join(root, `${stem}.words.json`), ocrManifest: new Set(), manifest: new Set(), csvManifest: new Set(), pageManifest: new Set(), attachmentManifest: new Set(), sourceMap: new Map() };
|
|
108
109
|
} catch (e) { rmSync(lockPath, { force: true }); throw e; }
|
|
109
110
|
}
|
|
110
111
|
|
|
@@ -174,10 +175,23 @@ export function writePageStats(b: Bundle, stats: PageStat[]): void {
|
|
|
174
175
|
writeFileSync(b.pageStatsPath, `${JSON.stringify(stats, null, 2)}\n`, "utf8");
|
|
175
176
|
}
|
|
176
177
|
|
|
177
|
-
/**
|
|
178
|
-
export function
|
|
179
|
-
const
|
|
180
|
-
|
|
178
|
+
/** Rename within the bundle root keeps the complete words document atomic. */
|
|
179
|
+
export function publishWords(b: Bundle): string | null {
|
|
180
|
+
const staged = join(b.stagingDir, "words.json");
|
|
181
|
+
try {
|
|
182
|
+
if (!existsSync(staged)) throw new Error("child staged no words.json");
|
|
183
|
+
renameSync(staged, b.wordsPath);
|
|
184
|
+
return null;
|
|
185
|
+
} catch (e) {
|
|
186
|
+
rmSync(b.wordsPath, { force: true });
|
|
187
|
+
return `write failed - ${(e as Error).message}`;
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** Move `.done`-gated Markdown and words sidecars into `ocr/`; drop partial dirs and the checkpoint. Returns page -> absolute paths for each kind. */
|
|
192
|
+
export function publishSidecars(b: Bundle): { sidecars: Map<number, string>; wordSidecars: Map<number, string> } {
|
|
193
|
+
const sidecars = new Map<number, string>(), wordSidecars = new Map<number, string>();
|
|
194
|
+
if (!existsSync(b.ocrStagingDir)) return { sidecars, wordSidecars };
|
|
181
195
|
for (const dir of readdirSync(b.ocrStagingDir).sort()) {
|
|
182
196
|
const m = dir.match(/^p(\d+)$/);
|
|
183
197
|
if (!m) continue;
|
|
@@ -188,10 +202,16 @@ export function publishSidecars(b: Bundle): Map<number, string> {
|
|
|
188
202
|
mkdirSync(b.ocrDir, { recursive: true });
|
|
189
203
|
renameSync(join(pageDir, file), join(b.ocrDir, file));
|
|
190
204
|
b.ocrManifest.add(file);
|
|
191
|
-
|
|
205
|
+
sidecars.set(Number(m[1]), join(b.ocrDir, file));
|
|
206
|
+
const wfile = `${b.stem}-${dir}.words.json`;
|
|
207
|
+
if (existsSync(join(pageDir, wfile))) {
|
|
208
|
+
renameSync(join(pageDir, wfile), join(b.ocrDir, wfile));
|
|
209
|
+
b.ocrManifest.add(wfile);
|
|
210
|
+
wordSidecars.set(Number(m[1]), join(b.ocrDir, wfile));
|
|
211
|
+
}
|
|
192
212
|
}
|
|
193
213
|
rmSync(b.ocrStagingDir, { recursive: true, force: true });
|
|
194
|
-
return
|
|
214
|
+
return { sidecars, wordSidecars };
|
|
195
215
|
}
|
|
196
216
|
|
|
197
217
|
export function publishAttachments(b: Bundle): void {
|
|
@@ -254,6 +274,7 @@ export function abortBundle(b: Bundle): void {
|
|
|
254
274
|
for (const f of b.ocrManifest) rmSync(join(b.ocrDir, f), { force: true });
|
|
255
275
|
rmSync(b.ocrStagingDir, { recursive: true, force: true });
|
|
256
276
|
rmSync(b.pageStatsPath, { force: true });
|
|
277
|
+
rmSync(b.wordsPath, { force: true });
|
|
257
278
|
rmSync(`${b.mdPath}.tmp`, { force: true });
|
|
258
279
|
rmSync(b.stagingDir, { recursive: true, force: true });
|
|
259
280
|
rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|
package/lib/doc-to-md-core.ts
CHANGED
|
@@ -12,7 +12,7 @@ import { type ChildProcess, spawn } from "node:child_process";
|
|
|
12
12
|
import { homedir, tmpdir } from "node:os";
|
|
13
13
|
import { basename, dirname, extname, isAbsolute, join, resolve } from "node:path";
|
|
14
14
|
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
15
|
-
import { type Bundle, abortBundle, commitBundle, openBundle, publishAttachments, publishSidecars, writePageStats, publishPageImages, publishSheetCsvs, publishSheetImages, publishStaged, rewriteLinks, tempBundleRoot, validateImageLinks } from "./doc-to-md-bundle.ts";
|
|
15
|
+
import { type Bundle, abortBundle, commitBundle, openBundle, publishAttachments, publishSidecars, publishWords, writePageStats, publishPageImages, publishSheetCsvs, publishSheetImages, publishStaged, rewriteLinks, tempBundleRoot, validateImageLinks } from "./doc-to-md-bundle.ts";
|
|
16
16
|
import { type Engine, type HandleData, type InfoData, type OcrInfo, type PageStat, type Tier, formatHandle, formatInfoHandle, scanOutline, type SheetInfo, type TocEntry } from "./doc-to-md-handle.ts";
|
|
17
17
|
import { type DocToMdOptions, type InputType, IMAGE_EXTS, TUNABLE_DEFAULTS, UsageError, classifyInput, sanitizeStem } from "./doc-to-md-options.ts";
|
|
18
18
|
|
|
@@ -467,7 +467,7 @@ const clearStaging = (b: Pick<Bundle, "stagingDir">) => { for (const f of readdi
|
|
|
467
467
|
export const EXCEL_REMEDY = "Remedy: install uv, or pip install openpyxl xlrd pillow";
|
|
468
468
|
|
|
469
469
|
export type Mode = "html" | "image" | "info" | "pdf-primary" | "pdf-fallback" | "xlsx" | "pdf-text" | "render-pages" | "docx" | "email" | "ocr-pages";
|
|
470
|
-
export interface TierJson { pageStats?: PageStat[]; status?: string; written?: number[]; noText?: number[]; ocrFailed?: number[]; ocrErrors?: Record<string, string>; budgetStopped?: number[]; pageImages?: { page: number; file: string }[]; ocr?: OcrInfo; markdown?: string; pages?: number[]; pageCount?: number; emptyPages?: number[]; failedPages?: { page: number; error: string }[]; notes?: string[]; images?: { sheetIndex: number; file: string }[]; metadata?: Record<string, string>; toc?: [number, string, number | null][]; explicitBreaks?: number; engine?: string; degraded?: boolean; fallbackReason?: string | null; sheets?: SheetInfo[]; renderPages?: number[]; sheetCount?: number; ok?: boolean; reason?: string; rendered?: { idx: number; file: string; dpi: number }[]; failed?: { idx: number; reason: string }[]; }
|
|
470
|
+
export interface TierJson { words?: boolean; wordsErrors?: Record<string, string>; pageStats?: PageStat[]; status?: string; written?: number[]; noText?: number[]; ocrFailed?: number[]; ocrErrors?: Record<string, string>; budgetStopped?: number[]; pageImages?: { page: number; file: string }[]; ocr?: OcrInfo; markdown?: string; pages?: number[]; pageCount?: number; emptyPages?: number[]; failedPages?: { page: number; error: string }[]; notes?: string[]; images?: { sheetIndex: number; file: string }[]; metadata?: Record<string, string>; toc?: [number, string, number | null][]; explicitBreaks?: number; engine?: string; degraded?: boolean; fallbackReason?: string | null; sheets?: SheetInfo[]; renderPages?: number[]; sheetCount?: number; ok?: boolean; reason?: string; rendered?: { idx: number; file: string; dpi: number }[]; failed?: { idx: number; reason: string }[]; }
|
|
471
471
|
export type TierResult = { ok: true; json: TierJson } | { ok: false; reason: string; detail?: string } | { ok: false; userError: string; pageCount?: number };
|
|
472
472
|
|
|
473
473
|
export interface PipelineSeams {
|
|
@@ -536,7 +536,7 @@ export function reconcileRenderMarkers(md: string, renderPages: number[], fmt: s
|
|
|
536
536
|
return md;
|
|
537
537
|
}
|
|
538
538
|
|
|
539
|
-
export const emptyOcr = (lang: string): OcrInfo => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null, mode: "textless", sidecars: {}, ocrErrors: {}, killed: null, notAttempted: [], childError: null });
|
|
539
|
+
export const emptyOcr = (lang: string): OcrInfo => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null, mode: "textless", sidecars: {}, wordSidecars: {}, ocrErrors: {}, killed: null, notAttempted: [], childError: null });
|
|
540
540
|
|
|
541
541
|
export interface OcrPagesOutcome { written: number[]; noText: number[]; ocrFailed: number[]; ocrErrors: Record<number, string>; budgetStopped: number[]; killed: number | null; notAttempted: number[]; childError: string | null; }
|
|
542
542
|
const emptyOutcome = (): OcrPagesOutcome => ({ written: [], noText: [], ocrFailed: [], ocrErrors: {}, budgetStopped: [], killed: null, notAttempted: [], childError: null });
|
|
@@ -612,11 +612,12 @@ export async function convertDocument(o: DocToMdOptions, signal?: AbortSignal, s
|
|
|
612
612
|
let office: { pdfPath: string; cleanup: () => void } | null = null;
|
|
613
613
|
try {
|
|
614
614
|
let pdfPath = inputPath;
|
|
615
|
-
const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr && !forced, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
|
|
615
|
+
const base = { path: inputPath, pages: o.pages, ...(o.words && (type === "pdf" || type === "image") ? { words: true } : {}), stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr && !forced, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
|
|
616
616
|
let tier: Tier | undefined, engine: Engine | undefined, json: TierJson | undefined, degraded: string | null = null, fallbackReason: string | null = null;
|
|
617
617
|
let explicitBreaks: number | null = null;
|
|
618
618
|
let notes: string[] = [];
|
|
619
619
|
let officeRoute: string | null = null;
|
|
620
|
+
let copyReason: string | null = null;
|
|
620
621
|
if (type === "html") {
|
|
621
622
|
const prepared = await prepareHtml(inputPath, b.stagingDir);
|
|
622
623
|
if (signal?.aborted) throw new Error("aborted");
|
|
@@ -642,6 +643,7 @@ export async function convertDocument(o: DocToMdOptions, signal?: AbortSignal, s
|
|
|
642
643
|
}
|
|
643
644
|
if (signal?.aborted) throw new Error("aborted");
|
|
644
645
|
if (json === undefined) {
|
|
646
|
+
copyReason = reason;
|
|
645
647
|
clearStaging(b);
|
|
646
648
|
const file = `original${extname(inputPath).toLowerCase()}`;
|
|
647
649
|
const dir = join(b.stagingDir, "p1");
|
|
@@ -745,15 +747,32 @@ export async function convertDocument(o: DocToMdOptions, signal?: AbortSignal, s
|
|
|
745
747
|
if (tier === undefined || engine === undefined || json === undefined) throw new Error("internal: no tier produced output");
|
|
746
748
|
const pageStats = json.pageStats ?? null;
|
|
747
749
|
if (pageStats) writePageStats(b, pageStats);
|
|
750
|
+
let wordsPath: string | null = null, wordsReason: string | null = null;
|
|
751
|
+
const wordsErrors: Record<number, string> = {};
|
|
752
|
+
const takeWordsErrors = (j: TierJson | undefined) => {
|
|
753
|
+
for (const [k, v] of Object.entries(j?.wordsErrors ?? {})) {
|
|
754
|
+
const page = Number(k);
|
|
755
|
+
if (Number.isFinite(page)) wordsErrors[page] = v;
|
|
756
|
+
}
|
|
757
|
+
};
|
|
758
|
+
if (o.words) {
|
|
759
|
+
if (type !== "pdf" && type !== "image") wordsReason = `none - word positions apply to PDF and image inputs only (${type})`;
|
|
760
|
+
else if (tier === "unpdf") wordsReason = "none - unpdf tier has no page geometry";
|
|
761
|
+
else if (engine === "copy") wordsReason = `none - image copied without conversion (${copyReason})`;
|
|
762
|
+
else if (json?.words === true) { wordsReason = publishWords(b); if (wordsReason === null) wordsPath = b.wordsPath; }
|
|
763
|
+
else wordsReason = `write failed - ${json?.wordsErrors?.file ?? "child reported no words document"}`;
|
|
764
|
+
takeWordsErrors(json);
|
|
765
|
+
}
|
|
748
766
|
let ocr = handleOcr(tier, type, o, json);
|
|
749
767
|
if (forced) {
|
|
750
|
-
const r = await s.runTier("ocr-pages", { path: pdfPath, pages: o.pages, stem: b.stem, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, stagingDir: b.ocrStagingDir, dpi: o.imageDpi, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.primaryTimeoutMs, backend);
|
|
768
|
+
const r = await s.runTier("ocr-pages", { path: pdfPath, pages: o.pages, ...(o.words && type === "pdf" ? { words: true } : {}), stem: b.stem, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, stagingDir: b.ocrStagingDir, dpi: o.imageDpi, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.primaryTimeoutMs, backend);
|
|
751
769
|
if (signal?.aborted || (!r.ok && "reason" in r && r.reason === "aborted")) throw new Error("aborted");
|
|
752
770
|
if (r.ok && r.json.status === "unavailable") throw new Error(`OCR unavailable: ${r.json.reason} (install Tesseract; see doc/doc-to-md.md)`);
|
|
753
771
|
const outcome = r.ok && r.json.status === "ran" ? outcomeFromChild(r.json)
|
|
754
772
|
: recoverOcrPages(b.ocrStagingDir, o.pages!, !r.ok ? ("userError" in r ? r.userError : `${r.reason}${detailSuffix(r)}`) : "malformed child output");
|
|
755
|
-
const sidecars = publishSidecars(b);
|
|
756
|
-
ocr = { ...emptyOcr(o.ocrLanguage), ...json.ocr, status: "ran", reason: null, mode: "all", pages: outcome.written, noText: outcome.noText, ocrFailed: outcome.ocrFailed, ocrErrors: outcome.ocrErrors, budgetStopped: outcome.budgetStopped, killed: outcome.killed, notAttempted: outcome.notAttempted, childError: outcome.childError, sidecars: Object.fromEntries(sidecars) };
|
|
773
|
+
const { sidecars, wordSidecars } = publishSidecars(b);
|
|
774
|
+
ocr = { ...emptyOcr(o.ocrLanguage), ...json.ocr, status: "ran", reason: null, mode: "all", pages: outcome.written, noText: outcome.noText, ocrFailed: outcome.ocrFailed, ocrErrors: outcome.ocrErrors, budgetStopped: outcome.budgetStopped, killed: outcome.killed, notAttempted: outcome.notAttempted, childError: outcome.childError, sidecars: Object.fromEntries(sidecars), wordSidecars: Object.fromEntries(wordSidecars) };
|
|
775
|
+
if (o.words && r.ok) takeWordsErrors(r.json);
|
|
757
776
|
}
|
|
758
777
|
if (!isExcel) notes = [...notes, ...(json.notes ?? [])];
|
|
759
778
|
if (b.renamedFrom) notes.splice(notes[0]?.startsWith("preview truncated:") ? 1 : 0, 0, `renamed to ${b.stem} (${b.renameReason})`);
|
|
@@ -770,7 +789,7 @@ export async function convertDocument(o: DocToMdOptions, signal?: AbortSignal, s
|
|
|
770
789
|
const markdown = (head.length ? `${head.join("\n")}\n\n` : "") + body;
|
|
771
790
|
commitBundle(b, markdown);
|
|
772
791
|
const outline = scanOutline(markdown, o.outlineMaxEntries);
|
|
773
|
-
const details: DocToMdDetails = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr, pageStats, pageStatsPath: pageStats ? b.pageStatsPath : null, ocrDir: b.ocrManifest.size ? b.ocrDir : null };
|
|
792
|
+
const details: DocToMdDetails = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr, pageStats, pageStatsPath: pageStats ? b.pageStatsPath : null, ocrDir: b.ocrManifest.size ? b.ocrDir : null, wordsPath, wordsReason, wordsErrors };
|
|
774
793
|
return { output: formatHandle(details), details };
|
|
775
794
|
} catch (e) { abortBundle(b); throw e; }
|
|
776
795
|
finally { office?.cleanup(); }
|
package/lib/doc-to-md-handle.ts
CHANGED
|
@@ -32,6 +32,7 @@ export interface OcrInfo {
|
|
|
32
32
|
tesseract: boolean | null;
|
|
33
33
|
mode: OcrMode;
|
|
34
34
|
sidecars: Record<number, string>;
|
|
35
|
+
wordSidecars: Record<number, string>;
|
|
35
36
|
ocrErrors: Record<number, string>;
|
|
36
37
|
killed: number | null;
|
|
37
38
|
notAttempted: number[];
|
|
@@ -44,6 +45,7 @@ export interface HandleData {
|
|
|
44
45
|
degraded: string | null; fallbackReason: string | null; failedPages: number[]; emptyPages: number[];
|
|
45
46
|
notes: string[]; outline: OutlineEntry[]; outlineTotal: number; ocr: OcrInfo | null;
|
|
46
47
|
pageStats: PageStat[] | null; pageStatsPath: string | null; ocrDir: string | null;
|
|
48
|
+
wordsPath: string | null; wordsReason: string | null; wordsErrors: Record<number, string>;
|
|
47
49
|
}
|
|
48
50
|
|
|
49
51
|
export interface InfoData {
|
|
@@ -169,6 +171,10 @@ export function formatHandle(h: HandleData): string {
|
|
|
169
171
|
if (h.pagesDir && h.pageImageCount > 0) lines.push(`Pages-Dir: ${h.pagesDir} (${h.pageImageCount} pages)`);
|
|
170
172
|
else if (h.pageImagesReason) lines.push(`Pages-Dir: none - ${h.pageImagesReason}`);
|
|
171
173
|
if (h.pageStatsPath) lines.push(`Page-Stats: ${h.pageStatsPath}`);
|
|
174
|
+
if (h.wordsPath) {
|
|
175
|
+
const bad = Object.keys(h.wordsErrors).map(Number).sort((a, b) => a - b);
|
|
176
|
+
lines.push(`Words: ${h.wordsPath}${bad.length ? ` (extraction failed for pages ${compactRanges(bad)}: ${h.wordsErrors[bad[0]]})` : ""}`);
|
|
177
|
+
} else if (h.wordsReason) lines.push(`Words: ${h.wordsReason}`);
|
|
172
178
|
if (h.ocrDir) lines.push(`OCR-Dir: ${h.ocrDir}`);
|
|
173
179
|
lines.push(`Type: ${h.type} Engine: ${h.engine} Tier: ${h.tier}`);
|
|
174
180
|
lines.push(`Page-Count: ${pageCountLabel(h)} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize(h.bytes)} / ${h.lines} lines`);
|
package/lib/doc-to-md-options.ts
CHANGED
|
@@ -30,6 +30,7 @@ export interface DocToMdOptions extends Tunables {
|
|
|
30
30
|
outputDir: string | null;
|
|
31
31
|
overwrite: boolean;
|
|
32
32
|
pageImages: boolean;
|
|
33
|
+
words: boolean;
|
|
33
34
|
ocrMode: OcrMode;
|
|
34
35
|
}
|
|
35
36
|
|
|
@@ -41,6 +42,7 @@ export interface PerCallInput extends Partial<Tunables> {
|
|
|
41
42
|
outputDir?: string | null;
|
|
42
43
|
overwrite?: boolean;
|
|
43
44
|
pageImages?: boolean;
|
|
45
|
+
words?: boolean;
|
|
44
46
|
ocrMode?: OcrMode;
|
|
45
47
|
}
|
|
46
48
|
|
|
@@ -71,6 +73,7 @@ export const DOC_TO_MD_OPTIONS: readonly OptionDescriptor[] = [
|
|
|
71
73
|
{ key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir. <stem> = basename without extension with [^A-Za-z0-9._-]+ -> _ (empty -> document); a second call on the same stem writes <stem>-2.md" },
|
|
72
74
|
{ key: "overwrite", flag: "--overwrite", type: "bool", default: false, settable: false, help: "Replace an existing completed <stem>.md bundle" },
|
|
73
75
|
{ key: "pageImages", flag: "--page-images", type: "bool", default: false, settable: false, help: "Also render every selected page to pages/<stem>-pNNN.<imageFormat> at imageDpi (PDF, PPTX, .doc, DOCX via LibreOffice); off by default" },
|
|
76
|
+
{ key: "words", flag: "--words", type: "bool", default: false, settable: false, help: "Write word positions: <stem>.words.json beside the Markdown lists every text-layer word of each selected page with its bbox (PDF points, top-left origin, display orientation; image inputs in source pixels) and the words inline OCR recognized, tagged source \"text\" or \"ocr\"; under --ocr-mode all the OCR words go to ocr/<stem>-pNNN.words.json beside each sidecar. Never triggers OCR. PDF and image inputs only." },
|
|
74
77
|
{ key: "ocrMode", flag: "--ocr-mode", type: "enum", default: "textless", settable: false, enumValues: ["textless", "all"], help: "OCR policy: textless (default) OCRs only pages with an empty text layer, inline; all OCRs every selected page and writes the recognized text to ocr/<stem>-pNNN.md sidecars, leaving the Markdown untouched. all requires --ocr and an explicit --pages selection (PDF, PPTX, DOC)." },
|
|
75
78
|
{ key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 60000, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier and DOCX child (docx mode); also the unpdf tier" },
|
|
76
79
|
{ key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 30000, settable: true, help: "PyMuPDF get_text tier (including DOCX LibreOffice fallback); also PDF and DOCX info and Excel rendered views" },
|
|
@@ -181,8 +184,8 @@ export function resolveOptions(perCall: PerCallInput, settings: Partial<Tunables
|
|
|
181
184
|
out[d.key] = value;
|
|
182
185
|
}
|
|
183
186
|
const o = out as unknown as DocToMdOptions;
|
|
184
|
-
if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages || o.ocrMode === "all")) {
|
|
185
|
-
throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite, --page-images or --ocr-mode all");
|
|
187
|
+
if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages || o.words || o.ocrMode === "all")) {
|
|
188
|
+
throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite, --page-images, --words or --ocr-mode all");
|
|
186
189
|
}
|
|
187
190
|
return o;
|
|
188
191
|
}
|
|
@@ -207,6 +210,21 @@ export const USAGE_PATTERNS = [
|
|
|
207
210
|
|
|
208
211
|
export const usagePatterns = (cmd: string): string => USAGE_PATTERNS.replaceAll("<cmd>", cmd);
|
|
209
212
|
|
|
213
|
+
/** Every artifact a bundle can contain; shared by --help and the generated skill. */
|
|
214
|
+
export const BUNDLE_LAYOUT: readonly { artifact: string; trigger: string; content: string; namedBy: string }[] = [
|
|
215
|
+
{ artifact: "<stem>.md", trigger: "always", content: "the Markdown", namedBy: "Saved-To: / savedTo" },
|
|
216
|
+
{ artifact: "images/", trigger: "embedded or extracted figures", content: "image files linked from the Markdown", namedBy: "Images-Dir: / imagesDir" },
|
|
217
|
+
{ artifact: "pages/<stem>-pNNN.<fmt>", trigger: "--page-images", content: "page renders at --image-dpi", namedBy: "Pages-Dir: / pagesDir" },
|
|
218
|
+
{ artifact: "sheets/", trigger: "Excel input", content: "one CSV per non-empty worksheet", namedBy: "Sheets-Dir: / sheetsDir" },
|
|
219
|
+
{ artifact: "attachments/", trigger: "email input", content: "saved attachments", namedBy: "Markdown attachment list" },
|
|
220
|
+
{ artifact: "<stem>.pages.json", trigger: "Python PDF tiers (PDF, PPTX, DOC, DOCX via LibreOffice; not unpdf)", content: "per-page chars, image count, image coverage", namedBy: "Page-Stats: / pageStatsPath" },
|
|
221
|
+
{ artifact: "<stem>.words.json", trigger: "--words", content: "per-page word boxes, source text/ocr", namedBy: "Words: / wordsPath" },
|
|
222
|
+
{ artifact: "ocr/<stem>-pNNN.md", trigger: "--ocr --ocr-mode all", content: "recognized text of a forced page", namedBy: "OCR-Dir: / ocr.sidecars" },
|
|
223
|
+
{ artifact: "ocr/<stem>-pNNN.words.json", trigger: "--ocr --ocr-mode all --words", content: "word boxes of that OCR", namedBy: "ocr.wordSidecars" },
|
|
224
|
+
];
|
|
225
|
+
|
|
226
|
+
const bundleLayoutText = (): string => BUNDLE_LAYOUT.map((r) => ` ${r.artifact.padEnd(28)} ${r.trigger}; ${r.content}; named by ${r.namedBy}`).join("\n");
|
|
227
|
+
|
|
210
228
|
export function renderHelp(): string {
|
|
211
229
|
const row = (d: OptionDescriptor) => ` ${(d.flag ?? "<path>").padEnd(26)} ${d.help}${d.default !== null && d.key !== "info" && d.key !== "overwrite" ? ` (default ${d.default})` : ""}`;
|
|
212
230
|
return [
|
|
@@ -216,6 +234,6 @@ export function renderHelp(): string {
|
|
|
216
234
|
...DOC_TO_MD_OPTIONS.filter((d) => d.settable).map(row),
|
|
217
235
|
"", "Result: a handle (Saved-To, Images-Dir, Page-Stats, Page-Count, Outline ...). Read the Saved-To file for the Markdown.",
|
|
218
236
|
"Exit codes: 0 success, 1 runtime error, 2 usage error.",
|
|
219
|
-
"", usagePatterns("pi-quiver doc-to-md"),
|
|
237
|
+
"", "Bundle layout:", bundleLayoutText(), "", usagePatterns("pi-quiver doc-to-md"),
|
|
220
238
|
].join("\n");
|
|
221
239
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-quiver",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.10.0",
|
|
4
4
|
"description": "Personal pack of Pi coding-agent extensions: context-safe fetch, doc_to_md PDF/DOCX/PPTX-to-Markdown conversion, session naming, a themed ASCII startup header, Opus 4.8 fast mode, and a provider-stall watchdog.",
|
|
5
5
|
"author": "Jacek Juraszek",
|
|
6
6
|
"license": "MIT",
|
package/scripts/doc_to_md.py
CHANGED
|
@@ -205,11 +205,22 @@ def mode_image(o):
|
|
|
205
205
|
info = new_ocr(lang)
|
|
206
206
|
status = ocr_status(bool(o.get("ocr")), lang)
|
|
207
207
|
apply_status(info, status)
|
|
208
|
-
|
|
208
|
+
words_on = bool(o.get("words"))
|
|
209
|
+
word_pages, words_errors = [], {}
|
|
210
|
+
w_px = h_px = None
|
|
211
|
+
if words_on or status["status"] == "ready":
|
|
209
212
|
try:
|
|
210
213
|
pix = pymupdf.Pixmap(path)
|
|
211
214
|
w_px, h_px = pix.width, pix.height
|
|
212
215
|
del pix
|
|
216
|
+
except Exception as exc: # noqa: BLE001
|
|
217
|
+
if words_on:
|
|
218
|
+
words_errors["1"] = words_error(exc)
|
|
219
|
+
if status["status"] == "ready":
|
|
220
|
+
info["ocrFailed"].append(1)
|
|
221
|
+
status = {**status, "status": "failed"}
|
|
222
|
+
if status["status"] == "ready":
|
|
223
|
+
try:
|
|
213
224
|
if min(w_px, h_px) < MIN_OCR_SIDE_PX:
|
|
214
225
|
info["status"], info["reason"] = "skipped", "image too small"
|
|
215
226
|
else:
|
|
@@ -218,6 +229,19 @@ def mode_image(o):
|
|
|
218
229
|
text = pymupdf4llm.to_markdown(pdf, pages=[0], write_images=False, use_ocr=True, force_ocr=True,
|
|
219
230
|
ocr_language=lang, ocr_dpi=image_ocr_dpi(w_px, r.width, r.height),
|
|
220
231
|
page_separators=False).strip()
|
|
232
|
+
if words_on:
|
|
233
|
+
try:
|
|
234
|
+
pg = pdf[0]
|
|
235
|
+
entry = words_page(pg, 1, 0, page_words(pg, True))
|
|
236
|
+
sx, sy = w_px / r.width, h_px / r.height
|
|
237
|
+
for word in entry["words"]:
|
|
238
|
+
b = word["bbox"]
|
|
239
|
+
word["bbox"] = [round(b[0] * sx, 1), round(b[1] * sy, 1),
|
|
240
|
+
round(b[2] * sx, 1), round(b[3] * sy, 1)]
|
|
241
|
+
entry["width"], entry["height"] = w_px, h_px
|
|
242
|
+
word_pages.append(entry)
|
|
243
|
+
except Exception as exc: # noqa: BLE001
|
|
244
|
+
words_errors["1"] = words_error(exc)
|
|
221
245
|
if text:
|
|
222
246
|
info["pages"].append(1)
|
|
223
247
|
md += "\n\n" + ocr_block(f"p1/{name}", text)
|
|
@@ -225,7 +249,17 @@ def mode_image(o):
|
|
|
225
249
|
info["noText"].append(1)
|
|
226
250
|
except Exception: # noqa: BLE001 - OCR never fails the conversion
|
|
227
251
|
info["ocrFailed"].append(1)
|
|
228
|
-
|
|
252
|
+
if words_on and not o.get("ocr") and w_px is not None and "1" not in words_errors:
|
|
253
|
+
try:
|
|
254
|
+
if os.environ.get("DOC_TO_MD_WORDS_FAIL") == "1": # tests only
|
|
255
|
+
raise RuntimeError("words injected failure")
|
|
256
|
+
word_pages.append({"page": 1, "width": w_px, "height": h_px, "rotation": 0, "words": []})
|
|
257
|
+
except Exception as exc: # noqa: BLE001
|
|
258
|
+
words_errors["1"] = words_error(exc)
|
|
259
|
+
result = {"markdown": md + "\n", "pageCount": 1, "emptyPages": [], "failedPages": [], "notes": [], "ocr": info}
|
|
260
|
+
if words_on:
|
|
261
|
+
result.update(write_words(o["stagingDir"], "px", word_pages, words_errors))
|
|
262
|
+
return result
|
|
229
263
|
|
|
230
264
|
|
|
231
265
|
def mode_info(o):
|
|
@@ -266,6 +300,58 @@ def page_stats(doc, n):
|
|
|
266
300
|
return {"page": n, "error": f"{type(exc).__name__}: {exc}"[:300]}
|
|
267
301
|
|
|
268
302
|
|
|
303
|
+
GLYPHLESS_FONT = "GlyphLessFont"
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def word_entries(page, words, ocr_lines):
|
|
307
|
+
import pymupdf
|
|
308
|
+
matrix = page.rotation_matrix
|
|
309
|
+
out = []
|
|
310
|
+
for x0, y0, x1, y1, text, bno, lno, _ in words:
|
|
311
|
+
r = pymupdf.Rect(x0, y0, x1, y1) * matrix
|
|
312
|
+
out.append({"text": text, "bbox": [round(r.x0, 1), round(r.y0, 1), round(r.x1, 1), round(r.y1, 1)],
|
|
313
|
+
"source": "ocr" if ocr_lines is None or (bno, lno) in ocr_lines else "text"})
|
|
314
|
+
return out
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def page_words(page, inline_ocr_ran):
|
|
318
|
+
if not inline_ocr_ran:
|
|
319
|
+
return word_entries(page, page.get_text("words"), set())
|
|
320
|
+
# Shared textpage keeps block/line indexes aligned despite differing default image flags.
|
|
321
|
+
tp = page.get_textpage()
|
|
322
|
+
ocr_lines = {(bi, li) for bi, b in enumerate(page.get_text("dict", textpage=tp)["blocks"])
|
|
323
|
+
for li, line in enumerate(b.get("lines", []))
|
|
324
|
+
if line["spans"] and all(s["font"] == GLYPHLESS_FONT for s in line["spans"])}
|
|
325
|
+
return word_entries(page, page.get_text("words", textpage=tp), ocr_lines)
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def ocr_words(page, tp):
|
|
329
|
+
return word_entries(page, page.get_text("words", textpage=tp), None)
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def words_page(page, n, rotation, words):
|
|
333
|
+
if os.environ.get("DOC_TO_MD_WORDS_FAIL") == str(n): # tests only
|
|
334
|
+
raise RuntimeError("words injected failure")
|
|
335
|
+
return {"page": n, "width": round(page.rect.width, 1), "height": round(page.rect.height, 1),
|
|
336
|
+
"rotation": rotation, "words": words}
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def words_error(exc):
|
|
340
|
+
return f"{type(exc).__name__}: {exc}"[:300]
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def write_words(staging, unit, pages, errors):
|
|
344
|
+
try:
|
|
345
|
+
os.makedirs(staging, exist_ok=True)
|
|
346
|
+
with open(os.path.join(staging, "words.json"), "w", encoding="utf-8") as fh:
|
|
347
|
+
json.dump({"unit": unit, "pages": pages}, fh, indent=2)
|
|
348
|
+
fh.write("\n")
|
|
349
|
+
return {"words": True, "wordsErrors": errors}
|
|
350
|
+
except Exception as exc: # noqa: BLE001 - geometry never fails the conversion
|
|
351
|
+
errors["file"] = words_error(exc)
|
|
352
|
+
return {"words": False, "wordsErrors": errors}
|
|
353
|
+
|
|
354
|
+
|
|
269
355
|
def mode_pdf_primary(o):
|
|
270
356
|
import time
|
|
271
357
|
start = time.monotonic()
|
|
@@ -278,11 +364,13 @@ def mode_pdf_primary(o):
|
|
|
278
364
|
info = new_ocr(lang)
|
|
279
365
|
status = ocr_status(True, lang) if o.get("ocr") else None
|
|
280
366
|
ocr_ms, plain_ms = [], []
|
|
367
|
+
word_pages, words_errors = [], {}
|
|
281
368
|
for i, n in enumerate(pages):
|
|
282
369
|
stats.append(page_stats(doc, n))
|
|
283
370
|
d = page_dir(staging, n)
|
|
284
371
|
try:
|
|
285
372
|
page = doc[n - 1]
|
|
373
|
+
rotation = page.rotation
|
|
286
374
|
textless = not page.get_text("text").strip()
|
|
287
375
|
if textless:
|
|
288
376
|
info["textless"].append(n)
|
|
@@ -326,6 +414,11 @@ def mode_pdf_primary(o):
|
|
|
326
414
|
if page_pic:
|
|
327
415
|
md = "\n\n".join(x for x in [md.rstrip(), f""] if x)
|
|
328
416
|
out.append(md.rstrip())
|
|
417
|
+
if o.get("words") and not (textless and o.get("ocr") and not kw["use_ocr"]):
|
|
418
|
+
try:
|
|
419
|
+
word_pages.append(words_page(page, n, rotation, page_words(page, kw["use_ocr"])))
|
|
420
|
+
except Exception as exc: # noqa: BLE001
|
|
421
|
+
words_errors[str(n)] = words_error(exc)
|
|
329
422
|
except Exception as exc: # noqa: BLE001
|
|
330
423
|
shutil.rmtree(d, ignore_errors=True)
|
|
331
424
|
failed.append({"page": n, "error": f"{type(exc).__name__}: {exc}"[:300]})
|
|
@@ -339,8 +432,11 @@ def mode_pdf_primary(o):
|
|
|
339
432
|
missing = len(pages) - len(page_images) if o.get("pageImages") else 0
|
|
340
433
|
if missing:
|
|
341
434
|
notes.append(f"Page images: {missing} of {len(pages)} unavailable")
|
|
342
|
-
|
|
343
|
-
|
|
435
|
+
result = {"markdown": "\n\n".join(out) + "\n", "pages": pages, "pageCount": doc.page_count,
|
|
436
|
+
"emptyPages": empty, "failedPages": failed, "notes": notes, "ocr": info, "pageImages": page_images, "pageStats": stats}
|
|
437
|
+
if o.get("words"):
|
|
438
|
+
result.update(write_words(staging, "pt", word_pages, words_errors))
|
|
439
|
+
return result
|
|
344
440
|
|
|
345
441
|
|
|
346
442
|
def mode_pdf_fallback(o):
|
|
@@ -353,14 +449,17 @@ def mode_pdf_fallback(o):
|
|
|
353
449
|
stats = []
|
|
354
450
|
lang = o.get("ocrLanguage", "eng")
|
|
355
451
|
ocr_info = new_ocr(lang)
|
|
452
|
+
word_pages, words_errors = [], {}
|
|
356
453
|
status = {"status": "unavailable", "reason": "fallback tier", "tesseract": None} if o.get("ocr") else None
|
|
357
454
|
for n in pages:
|
|
358
455
|
stats.append(page_stats(doc, n))
|
|
359
456
|
links = [f"" for f in keep.get(n, [])]
|
|
360
457
|
text = ""
|
|
361
458
|
page_pic = None
|
|
459
|
+
page_ok = False
|
|
362
460
|
try:
|
|
363
461
|
page = doc[n - 1]
|
|
462
|
+
rotation = page.rotation
|
|
364
463
|
text = page.get_text("text").strip()
|
|
365
464
|
if not text:
|
|
366
465
|
ocr_info["textless"].append(n)
|
|
@@ -388,11 +487,17 @@ def mode_pdf_fallback(o):
|
|
|
388
487
|
links.append(f"")
|
|
389
488
|
mark_done(d)
|
|
390
489
|
page_pic = render_page_image(page, n, o, page_images) if o.get("pageImages") else None
|
|
490
|
+
page_ok = True
|
|
391
491
|
except Exception as exc: # noqa: BLE001
|
|
392
492
|
shutil.rmtree(os.path.join(staging, f"p{n}"), ignore_errors=True)
|
|
393
493
|
text = ""
|
|
394
494
|
links = [f"" for f in keep.get(n, [])]
|
|
395
495
|
failed.append({"page": n, "error": f"{type(exc).__name__}: {exc}"[:300]})
|
|
496
|
+
if o.get("words") and page_ok and not (not text and o.get("ocr")):
|
|
497
|
+
try:
|
|
498
|
+
word_pages.append(words_page(page, n, rotation, page_words(page, False)))
|
|
499
|
+
except Exception as exc: # noqa: BLE001
|
|
500
|
+
words_errors[str(n)] = words_error(exc)
|
|
396
501
|
if not text:
|
|
397
502
|
empty.append(n)
|
|
398
503
|
out.append("\n\n".join(x for x in [text, "\n".join(links), f"" if page_pic else ""] if x))
|
|
@@ -405,8 +510,11 @@ def mode_pdf_fallback(o):
|
|
|
405
510
|
missing = len(pages) - len(page_images) if o.get("pageImages") else 0
|
|
406
511
|
if missing:
|
|
407
512
|
notes.append(f"Page images: {missing} of {len(pages)} unavailable")
|
|
408
|
-
|
|
409
|
-
|
|
513
|
+
result = {"markdown": "\n\n".join(out) + "\n", "pages": pages, "pageCount": doc.page_count,
|
|
514
|
+
"emptyPages": empty, "failedPages": failed, "notes": notes, "ocr": ocr_info, "pageImages": page_images, "pageStats": stats}
|
|
515
|
+
if o.get("words"):
|
|
516
|
+
result.update(write_words(staging, "pt", word_pages, words_errors))
|
|
517
|
+
return result
|
|
410
518
|
|
|
411
519
|
|
|
412
520
|
def mode_ocr_pages(o):
|
|
@@ -422,6 +530,8 @@ def mode_ocr_pages(o):
|
|
|
422
530
|
os.makedirs(staging, exist_ok=True)
|
|
423
531
|
active = os.path.join(staging, "active")
|
|
424
532
|
out = {"status": "ran", "written": [], "noText": [], "ocrFailed": [], "ocrErrors": {}, "budgetStopped": []}
|
|
533
|
+
if o.get("words"):
|
|
534
|
+
out["wordsErrors"] = {}
|
|
425
535
|
ocr_ms = []
|
|
426
536
|
for i, n in enumerate(pages):
|
|
427
537
|
with open(active, "w") as fh:
|
|
@@ -449,6 +559,20 @@ def mode_ocr_pages(o):
|
|
|
449
559
|
header = f"<!-- OCR of page {n} (tesseract {lang}); recognized text, not the text layer -->"
|
|
450
560
|
with open(sidecar, "w", encoding="utf-8") as fh:
|
|
451
561
|
fh.write(header + "\n\n" + (text + "\n\n" if text else "") + SEP.format(n=n).strip("\n") + "\n")
|
|
562
|
+
if o.get("words"):
|
|
563
|
+
wpath = os.path.join(d, f"{stem}-{tag}.words.json")
|
|
564
|
+
try:
|
|
565
|
+
entry = words_page(page, n, page.rotation, ocr_words(page, tp))
|
|
566
|
+
entry["unit"] = "pt"
|
|
567
|
+
with open(wpath, "w", encoding="utf-8") as fh:
|
|
568
|
+
json.dump(entry, fh, indent=2)
|
|
569
|
+
fh.write("\n")
|
|
570
|
+
except Exception as exc: # noqa: BLE001 - geometry never changes the OCR outcome
|
|
571
|
+
try:
|
|
572
|
+
os.remove(wpath)
|
|
573
|
+
except OSError:
|
|
574
|
+
pass
|
|
575
|
+
out["wordsErrors"][str(n)] = words_error(exc)
|
|
452
576
|
mark_done(d)
|
|
453
577
|
(out["written"] if text else out["noText"]).append(n)
|
|
454
578
|
except Exception as exc: # noqa: BLE001 - one page never stops the pass
|
|
@@ -459,6 +583,10 @@ def mode_ocr_pages(o):
|
|
|
459
583
|
os.remove(sidecar)
|
|
460
584
|
except OSError:
|
|
461
585
|
pass
|
|
586
|
+
try:
|
|
587
|
+
os.remove(os.path.join(d, f"{stem}-{tag}.words.json"))
|
|
588
|
+
except OSError:
|
|
589
|
+
pass
|
|
462
590
|
out["ocrFailed"].append(n)
|
|
463
591
|
out["ocrErrors"][str(n)] = msg
|
|
464
592
|
ocr_ms.append((time.monotonic() - t0) * 1000)
|