pi-quiver 6.9.0 → 6.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/README.md +4 -3
- package/dist/bin/pi-quiver.js +107 -18
- package/lib/doc-to-md-bundle.ts +44 -13
- package/lib/doc-to-md-core.ts +51 -11
- package/lib/doc-to-md-handle.ts +10 -0
- package/lib/doc-to-md-options.ts +23 -3
- package/package.json +1 -1
- package/scripts/doc_to_md.py +511 -117
package/CHANGELOG.md
CHANGED
|
@@ -8,6 +8,19 @@ Published to npm as `pi-quiver` (`pi install npm:pi-quiver`). Pushing a
|
|
|
8
8
|
via OIDC trusted publishing. The release helper at
|
|
9
9
|
`.agents/skills/release/scripts/release.sh` cuts the tag; CI publishes.
|
|
10
10
|
|
|
11
|
+
## v6.11.0 - 2026-10-02
|
|
12
|
+
|
|
13
|
+
- `doc_to_md`: textless PDF pages holding one eligible full-page image deliver the embedded stream without re-encoding; the handle prints `Native-Images:` and tool details / CLI JSON carry `nativeImages` (#27).
|
|
14
|
+
- `doc_to_md`: native probing, PDF page renders, and textless-page inline OCR run in a spawned raster worker with per-job budgets; a toxic page becomes a `Failed pages` note instead of failing the conversion when other pages succeed (#27).
|
|
15
|
+
- `doc_to_md`: the render ceiling rises from 16 to 50 Mpx; clamped PDF renders add Notes and report effective `pageImages[].dpi` plus `requestedDpi` (#27).
|
|
16
|
+
- `doc_to_md`: `hideAnnotations` / `--hide-annotations` hides PDF annotations and form widgets on renders and lets annotated scans use native delivery; default runs keep their painted render (#27).
|
|
17
|
+
- `doc_to_md`: `.done` markers retain native-image and clamp metadata for completed pages across primary-tier failure and fallback; corrupt metadata does not discard completed images (#27).
|
|
18
|
+
|
|
19
|
+
## v6.10.0 - 2026-10-02
|
|
20
|
+
|
|
21
|
+
- `doc_to_md`: new per-call `words` option (`--words`; not settable) writes `<stem>.words.json` - per selected PDF/image page, every text-layer word and every word inline OCR recognized in the same run, with display-space bbox (points; source pixels for images) and `source` `text`/`ocr`. Under `--ocr-mode all` each sidecar gets `ocr/<stem>-pNNN.words.json`. Never triggers OCR; Markdown, page stats and OCR outcome are unchanged. Handle gains `Words:`, `--json` gains `wordsPath`, `wordsReason`, `wordsErrors`, `ocr.wordSidecars`; `--info --words` is a usage error (#28).
|
|
22
|
+
- `doc_to_md`: `--help` and the generated skill list every bundle artifact under `Bundle layout` (#28).
|
|
23
|
+
|
|
11
24
|
## v6.9.0 - 2026-10-01
|
|
12
25
|
|
|
13
26
|
- `doc_to_md`: every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page `chars`, `images`, `imageCoverage`) and the handle prints `Page-Stats:`; CLI `--json` gains `pageStats`, `pageStatsPath`, `ocrDir`. The unpdf tier writes no stats (#26).
|
package/README.md
CHANGED
|
@@ -65,7 +65,7 @@ A 300 KB changelog page never touches your context window - you get a preview an
|
|
|
65
65
|
| Extension | Tool | What it does |
|
|
66
66
|
| --- | --- | --- |
|
|
67
67
|
| `extensions/fetch.ts` | `fetch` | Retrieve URLs over HTTP(S). HTML -> Markdown (Readability extraction, Turndown conversion). Binary saved untouched to a temp file. GitHub issue/PR/repo/actions-run/actions-job URLs auto-route through `gh` (falls back to HTTP); failed runs/jobs include failed-step logs (best-effort, summary-only otherwise). Same size gate as `fetch`. Behavior lives in `lib/fetch-core.ts`; also exposed as the `pi-quiver fetch` CLI (see [Claude Code support](#claude-code-support)). |
|
|
68
|
-
| `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS, MSG/EML, HTML, or image file to a Markdown bundle on disk (`<stem>.md` + `images/`, spreadsheet `sheets/`, optional `pageImages` in `pages/`, and email `attachments/`) and return a handle (paths, page count, outline, diagnostics) - never inline Markdown. `info` inspects PDF/Office inputs; `pages` selects 1-based PDF/Office pages (DOCX: explicit-page-break segments); HTML, image, and email inputs reject both; every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Tiers: pymupdf4llm -> PyMuPDF text (degraded) -> unpdf worker (no Python only). Excel -> sheet inventory, full CSVs, bounded previews, and optional rendered views. HTML keeps local and `data:` images in the bundle and remote images as links; image inputs keep the original image. Scanned PDF pages keep a page picture on both Python tiers; opt-in OCR needs Tesseract language data. DOCX converts directly (mammoth, python-docx fallback); LibreOffice pagination marks the degraded route. The Outline lists `L<line>` and `p<page>` per heading. Settings under `quiver.docToMd`. Every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page chars, image count, image coverage; handle line `Page-Stats:`); unpdf writes no stats. `ocrMode: "all"` with `ocr: true` and an explicit `pages` forces OCR on those pages into `ocr/<stem>-pNNN.md` sidecars, leaving the Markdown byte-identical to the same selected-page call without OCR. |
|
|
68
|
+
| `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS, MSG/EML, HTML, or image file to a Markdown bundle on disk (`<stem>.md` + `images/`, spreadsheet `sheets/`, optional `pageImages` in `pages/`, and email `attachments/`) and return a handle (paths, page count, outline, diagnostics) - never inline Markdown. `info` inspects PDF/Office inputs; `pages` selects 1-based PDF/Office pages (DOCX: explicit-page-break segments); HTML, image, and email inputs reject both; every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Tiers: pymupdf4llm -> PyMuPDF text (degraded) -> unpdf worker (no Python only). Excel -> sheet inventory, full CSVs, bounded previews, and optional rendered views. HTML keeps local and `data:` images in the bundle and remote images as links; image inputs keep the original image. Scanned PDF pages keep a page picture on both Python tiers; eligible single full-page images are delivered as their embedded stream, without re-encoding or render DPI; opt-in OCR needs Tesseract language data. DOCX converts directly (mammoth, python-docx fallback); LibreOffice pagination marks the degraded route. The Outline lists `L<line>` and `p<page>` per heading. Settings under `quiver.docToMd`. Every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page chars, image count, image coverage; handle line `Page-Stats:`); unpdf writes no stats. `ocrMode: "all"` with `ocr: true` and an explicit `pages` forces OCR on those pages into `ocr/<stem>-pNNN.md` sidecars, leaving the Markdown byte-identical to the same selected-page call without OCR. `words: true` (CLI `--words`, per-call only - no settings key) writes `<stem>.words.json` with every word's bbox per selected PDF/image page (`source` `text` or `ocr`), plus `ocr/<stem>-pNNN.words.json` beside each forced-OCR sidecar; it never triggers OCR. |
|
|
69
69
|
| `extensions/session-name.ts` | `/session-name` | Manual + opt-in automatic session naming, naming rules and deny list, long-session revisits, and Ghostty/Herdr tab rename. OFF by default. |
|
|
70
70
|
| `extensions/sword-header.ts` | `/builtin-header` | Themed ASCII startup header replacing pi's default logo. OFF by default. |
|
|
71
71
|
| `extensions/fast-mode.ts` | `/fast` | Inject Anthropic fast-mode (`speed: "fast"` + `anthropic-beta: fast-mode-2026-02-01`) into every Claude Opus 4.8 / Opus 5 request, any thinking level. `--fast` flag + `/fast [on\|off\|status]`. OFF by default. |
|
|
@@ -303,8 +303,9 @@ Each setting can also be overridden per-process via `PI_QUIVER_SLACK_ENABLED`, `
|
|
|
303
303
|
| `excelTimeoutMs` | `60000` | Excel child and Excel info deadline. |
|
|
304
304
|
| `warmTimeoutMs` | `120000` | Absolute first-call backend discovery/bootstrap deadline. |
|
|
305
305
|
| `pymupdfVersion` | `1.27.2.3` | pymupdf4llm pin, minimum `1.27.0`. |
|
|
306
|
-
| `imageDpi` | `150` | Render DPI for page images and Excel rendered views (capped by a
|
|
306
|
+
| `imageDpi` | `150` | Render DPI for page images and Excel rendered views (capped by a 50 Mpx budget). |
|
|
307
307
|
| `imageFormat` | `png` | Rendered image format: `png` or `jpg`; embedded images retain their extension. |
|
|
308
|
+
| `hideAnnotations` | `false` | Hide annotations on PDF textless-page and `pages/` renders, including form-field widgets (filled values disappear). Default paints them. Does not affect OCR text or embedded images; also lets an annotated scan be delivered as its embedded image. |
|
|
308
309
|
| `maxOutputBytes` | `20000000` | Child stdout cap in bytes. |
|
|
309
310
|
| `outlineMaxEntries` | `40` | Heading outline, TOC, or sheet inventory entries in the handle. |
|
|
310
311
|
| `ocr` | `false` | Opt-in OCR for scanned pages and images when Tesseract language data is installed. |
|
|
@@ -314,7 +315,7 @@ Each setting can also be overridden per-process via `PI_QUIVER_SLACK_ENABLED`, `
|
|
|
314
315
|
|
|
315
316
|
A bundle is `<outputDir>/<stem>.md` plus `<outputDir>/<stem>.pages.json` on Python PDF tiers, optional OCR sidecars under `<outputDir>/ocr/`, and `<outputDir>/images/` and, for spreadsheets with data, `<outputDir>/sheets/`; without `outputDir`, the tool creates a per-call temp root. A conversion owns `<stem>.md.lock` until it atomically publishes the Markdown. An existing `<stem>.md` fails the call unless `overwrite` is set; `overwrite` replaces that Markdown and the files it owns (`images/<stem>-p<N>-<n>.*`, `images/<stem>-s<idx>[-<n>].*`, `sheets/<stem>-s<idx>-<slug>.csv`, `<stem>.pages.json`, `ocr/<stem>-pNNN.md`), nothing else. Temp bundles are caller-owned - the tool never deletes a bundle it produced.
|
|
316
317
|
|
|
317
|
-
Excel needs a Python backend with openpyxl, xlrd and pillow - otherwise the call fails with `Remedy: install uv, or pip install openpyxl xlrd pillow`. The Markdown opens with a `## Sheets` table listing every sheet in workbook order (0-based `#`, `worksheet`/`chartsheet`, size, hidden, chart and image counts, rendered view, CSV link for non-empty worksheets), then one section per sheet: a `Data:` line linking the full-content CSV under `sheets/` for non-empty worksheets, chart metadata from the workbook model, embedded images, an optional rendered view, a preview of at most the first 100 rows x 50 columns, and - only when the preview is truncated - a `Columns:` profile (type, non-empty count, min/max, distinct up to 50). Sizes are the extent of non-empty cells (the `info` handle reports the raw worksheet dimensions instead, which may be larger). Rendered views (`images/<stem>-s<idx>.<fmt>`) are produced for sheets carrying charts or images when LibreOffice is on `PATH`: the workbook is exported one PDF page per sheet and rasterized under a
|
|
318
|
+
Excel needs a Python backend with openpyxl, xlrd and pillow - otherwise the call fails with `Remedy: install uv, or pip install openpyxl xlrd pillow`. The Markdown opens with a `## Sheets` table listing every sheet in workbook order (0-based `#`, `worksheet`/`chartsheet`, size, hidden, chart and image counts, rendered view, CSV link for non-empty worksheets), then one section per sheet: a `Data:` line linking the full-content CSV under `sheets/` for non-empty worksheets, chart metadata from the workbook model, embedded images, an optional rendered view, a preview of at most the first 100 rows x 50 columns, and - only when the preview is truncated - a `Columns:` profile (type, non-empty count, min/max, distinct up to 50). Sizes are the extent of non-empty cells (the `info` handle reports the raw worksheet dimensions instead, which may be larger). Rendered views (`images/<stem>-s<idx>.<fmt>`) are produced for sheets carrying charts or images when LibreOffice is on `PATH`: the workbook is exported one PDF page per sheet and rasterized under a 50 Mpx budget. Any LibreOffice or rasterization failure degrades to `Rendered view: unavailable (<reason>)` and a handle note; it never fails the conversion. Workbooks whose chartsheet drawings carry a zero-size anchor (openpyxl-authored files; Excel-authored files are unaffected) render as a degenerate page and are reported as such. `.xls` gets the inventory, CSVs and previews but no visual detection or rendering.
|
|
318
319
|
|
|
319
320
|
Worst-case wall time: PDF `warmTimeoutMs (first call) + primaryTimeoutMs + fallbackTimeoutMs`; PPTX adds `sofficeTimeoutMs`; DOCX on the Python path `warmTimeoutMs + primaryTimeoutMs` (success or a terminal child failure), DOCX child exit 1 then LibreOffice `warmTimeoutMs + primaryTimeoutMs + sofficeTimeoutMs + primaryTimeoutMs + fallbackTimeoutMs`, DOCX without a DOCX-capable backend `warmTimeoutMs + sofficeTimeoutMs + primaryTimeoutMs + fallbackTimeoutMs`; Excel `warmTimeoutMs + excelTimeoutMs + sofficeTimeoutMs + fallbackTimeoutMs`. Add `KILL_GRACE_MS` (2000 ms) per kill. There is no cap on image count, image bytes, cell count or workbook memory - deliberately; the per-tier timeouts, the rendered-view pixel budget and `maxOutputBytes` are the bounds.
|
|
320
321
|
|
package/dist/bin/pi-quiver.js
CHANGED
|
@@ -582,7 +582,7 @@ function ownedPagePattern(stem) {
|
|
|
582
582
|
return new RegExp(`^${escRe(stem)}-p\\d+\\.[a-z0-9]+$`);
|
|
583
583
|
}
|
|
584
584
|
function ownedOcrPattern(stem) {
|
|
585
|
-
return new RegExp(`^${escRe(stem)}-p\\d
|
|
585
|
+
return new RegExp(`^${escRe(stem)}-p\\d+(?:\\.words\\.json|\\.md)$`);
|
|
586
586
|
}
|
|
587
587
|
var FILE_LINK_RE = /\[[^\]]*\]\(\s*((?:sheets|attachments)\/[^)\s]+)\s*\)/g;
|
|
588
588
|
var IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
|
|
@@ -649,6 +649,7 @@ function openBundle(root, requested, overwrite) {
|
|
|
649
649
|
for (const f of readdirSync(attachmentsDir)) if (attachments.has(f)) rmSync(join2(attachmentsDir, f), { force: true });
|
|
650
650
|
}
|
|
651
651
|
rmSync(join2(root, `${stem}.pages.json`), { force: true });
|
|
652
|
+
rmSync(join2(root, `${stem}.words.json`), { force: true });
|
|
652
653
|
const ownedOcr = ownedOcrPattern(stem);
|
|
653
654
|
if (existsSync(ocrDir)) {
|
|
654
655
|
for (const f of readdirSync(ocrDir)) if (ownedOcr.test(f)) rmSync(join2(ocrDir, f), { force: true });
|
|
@@ -661,7 +662,7 @@ function openBundle(root, requested, overwrite) {
|
|
|
661
662
|
const pagesStagingDir = join2(pagesDir, `.stage-${lockId}`);
|
|
662
663
|
const attachmentsStagingDir = join2(attachmentsDir, `.stage-${lockId}`);
|
|
663
664
|
const ocrStagingDir = join2(ocrDir, `.stage-${lockId}`);
|
|
664
|
-
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join2(root, `${stem}.pages.json`), ocrManifest: /* @__PURE__ */ new Set(), manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), pageManifest: /* @__PURE__ */ new Set(), attachmentManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
|
|
665
|
+
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join2(root, `${stem}.pages.json`), wordsPath: join2(root, `${stem}.words.json`), ocrManifest: /* @__PURE__ */ new Set(), manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), pageManifest: /* @__PURE__ */ new Set(), attachmentManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
|
|
665
666
|
} catch (e) {
|
|
666
667
|
rmSync(lockPath, { force: true });
|
|
667
668
|
throw e;
|
|
@@ -679,6 +680,13 @@ function publishStaged(b) {
|
|
|
679
680
|
continue;
|
|
680
681
|
}
|
|
681
682
|
const page = Number(m[1]);
|
|
683
|
+
const raw = readFileSync(join2(pageDir, ".done"), "utf8").trim();
|
|
684
|
+
let meta = {};
|
|
685
|
+
try {
|
|
686
|
+
const v = raw ? JSON.parse(raw) : {};
|
|
687
|
+
if (v && typeof v === "object" && !Array.isArray(v)) meta = v;
|
|
688
|
+
} catch {
|
|
689
|
+
}
|
|
682
690
|
const files = readdirSync(pageDir).filter((f) => f !== ".done" && statSync(join2(pageDir, f)).isFile()).sort();
|
|
683
691
|
const names = [];
|
|
684
692
|
files.forEach((f, i) => {
|
|
@@ -688,8 +696,9 @@ function publishStaged(b) {
|
|
|
688
696
|
b.sourceMap.set(`${dir}/${f}`, `images/${name}`);
|
|
689
697
|
names.push(name);
|
|
690
698
|
});
|
|
699
|
+
if (meta.native) meta.native = { ...meta.native, file: names[files.indexOf(meta.native.file)] ?? meta.native.file };
|
|
691
700
|
rmSync(pageDir, { recursive: true, force: true });
|
|
692
|
-
out.set(page, names);
|
|
701
|
+
out.set(page, { files: names, meta });
|
|
693
702
|
}
|
|
694
703
|
return out;
|
|
695
704
|
}
|
|
@@ -729,9 +738,20 @@ function writePageStats(b, stats) {
|
|
|
729
738
|
writeFileSync2(b.pageStatsPath, `${JSON.stringify(stats, null, 2)}
|
|
730
739
|
`, "utf8");
|
|
731
740
|
}
|
|
741
|
+
function publishWords(b) {
|
|
742
|
+
const staged = join2(b.stagingDir, "words.json");
|
|
743
|
+
try {
|
|
744
|
+
if (!existsSync(staged)) throw new Error("child staged no words.json");
|
|
745
|
+
renameSync(staged, b.wordsPath);
|
|
746
|
+
return null;
|
|
747
|
+
} catch (e) {
|
|
748
|
+
rmSync(b.wordsPath, { force: true });
|
|
749
|
+
return `write failed - ${e.message}`;
|
|
750
|
+
}
|
|
751
|
+
}
|
|
732
752
|
function publishSidecars(b) {
|
|
733
|
-
const
|
|
734
|
-
if (!existsSync(b.ocrStagingDir)) return
|
|
753
|
+
const sidecars = /* @__PURE__ */ new Map(), wordSidecars = /* @__PURE__ */ new Map();
|
|
754
|
+
if (!existsSync(b.ocrStagingDir)) return { sidecars, wordSidecars };
|
|
735
755
|
for (const dir of readdirSync(b.ocrStagingDir).sort()) {
|
|
736
756
|
const m = dir.match(/^p(\d+)$/);
|
|
737
757
|
if (!m) continue;
|
|
@@ -742,10 +762,16 @@ function publishSidecars(b) {
|
|
|
742
762
|
mkdirSync2(b.ocrDir, { recursive: true });
|
|
743
763
|
renameSync(join2(pageDir, file), join2(b.ocrDir, file));
|
|
744
764
|
b.ocrManifest.add(file);
|
|
745
|
-
|
|
765
|
+
sidecars.set(Number(m[1]), join2(b.ocrDir, file));
|
|
766
|
+
const wfile = `${b.stem}-${dir}.words.json`;
|
|
767
|
+
if (existsSync(join2(pageDir, wfile))) {
|
|
768
|
+
renameSync(join2(pageDir, wfile), join2(b.ocrDir, wfile));
|
|
769
|
+
b.ocrManifest.add(wfile);
|
|
770
|
+
wordSidecars.set(Number(m[1]), join2(b.ocrDir, wfile));
|
|
771
|
+
}
|
|
746
772
|
}
|
|
747
773
|
rmSync(b.ocrStagingDir, { recursive: true, force: true });
|
|
748
|
-
return
|
|
774
|
+
return { sidecars, wordSidecars };
|
|
749
775
|
}
|
|
750
776
|
function publishAttachments(b) {
|
|
751
777
|
if (!existsSync(b.attachmentsStagingDir)) return;
|
|
@@ -821,6 +847,7 @@ function abortBundle(b) {
|
|
|
821
847
|
for (const f of b.ocrManifest) rmSync(join2(b.ocrDir, f), { force: true });
|
|
822
848
|
rmSync(b.ocrStagingDir, { recursive: true, force: true });
|
|
823
849
|
rmSync(b.pageStatsPath, { force: true });
|
|
850
|
+
rmSync(b.wordsPath, { force: true });
|
|
824
851
|
rmSync(`${b.mdPath}.tmp`, { force: true });
|
|
825
852
|
rmSync(b.stagingDir, { recursive: true, force: true });
|
|
826
853
|
rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|
|
@@ -950,6 +977,11 @@ function formatHandle(h) {
|
|
|
950
977
|
if (h.pagesDir && h.pageImageCount > 0) lines.push(`Pages-Dir: ${h.pagesDir} (${h.pageImageCount} pages)`);
|
|
951
978
|
else if (h.pageImagesReason) lines.push(`Pages-Dir: none - ${h.pageImagesReason}`);
|
|
952
979
|
if (h.pageStatsPath) lines.push(`Page-Stats: ${h.pageStatsPath}`);
|
|
980
|
+
if (h.nativeImages.length) lines.push(`Native-Images: ${h.nativeImages.length === 1 ? "page" : "pages"} ${compactRanges(h.nativeImages.map((n) => n.page))} (embedded image streams, no render DPI)`);
|
|
981
|
+
if (h.wordsPath) {
|
|
982
|
+
const bad = Object.keys(h.wordsErrors).map(Number).sort((a, b) => a - b);
|
|
983
|
+
lines.push(`Words: ${h.wordsPath}${bad.length ? ` (extraction failed for pages ${compactRanges(bad)}: ${h.wordsErrors[bad[0]]})` : ""}`);
|
|
984
|
+
} else if (h.wordsReason) lines.push(`Words: ${h.wordsReason}`);
|
|
953
985
|
if (h.ocrDir) lines.push(`OCR-Dir: ${h.ocrDir}`);
|
|
954
986
|
lines.push(`Type: ${h.type} Engine: ${h.engine} Tier: ${h.tier}`);
|
|
955
987
|
lines.push(`Page-Count: ${pageCountLabel(h)} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize2(h.bytes)} / ${h.lines} lines`);
|
|
@@ -1000,6 +1032,7 @@ var DOC_TO_MD_OPTIONS = [
|
|
|
1000
1032
|
{ key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir. <stem> = basename without extension with [^A-Za-z0-9._-]+ -> _ (empty -> document); a second call on the same stem writes <stem>-2.md" },
|
|
1001
1033
|
{ key: "overwrite", flag: "--overwrite", type: "bool", default: false, settable: false, help: "Replace an existing completed <stem>.md bundle" },
|
|
1002
1034
|
{ key: "pageImages", flag: "--page-images", type: "bool", default: false, settable: false, help: "Also render every selected page to pages/<stem>-pNNN.<imageFormat> at imageDpi (PDF, PPTX, .doc, DOCX via LibreOffice); off by default" },
|
|
1035
|
+
{ key: "words", flag: "--words", type: "bool", default: false, settable: false, help: 'Write word positions: <stem>.words.json beside the Markdown lists every text-layer word of each selected page with its bbox (PDF points, top-left origin, display orientation; image inputs in source pixels) and the words inline OCR recognized, tagged source "text" or "ocr"; under --ocr-mode all the OCR words go to ocr/<stem>-pNNN.words.json beside each sidecar. Never triggers OCR. PDF and image inputs only.' },
|
|
1003
1036
|
{ key: "ocrMode", flag: "--ocr-mode", type: "enum", default: "textless", settable: false, enumValues: ["textless", "all"], help: "OCR policy: textless (default) OCRs only pages with an empty text layer, inline; all OCRs every selected page and writes the recognized text to ocr/<stem>-pNNN.md sidecars, leaving the Markdown untouched. all requires --ocr and an explicit --pages selection (PDF, PPTX, DOC)." },
|
|
1004
1037
|
{ key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 6e4, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier and DOCX child (docx mode); also the unpdf tier" },
|
|
1005
1038
|
{ key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier (including DOCX LibreOffice fallback); also PDF and DOCX info and Excel rendered views" },
|
|
@@ -1012,7 +1045,8 @@ var DOC_TO_MD_OPTIONS = [
|
|
|
1012
1045
|
{ key: "maxOutputBytes", flag: "--max-output-bytes", type: "int", default: 2e7, settable: true, help: "Child stdout cap in bytes" },
|
|
1013
1046
|
{ key: "outlineMaxEntries", flag: "--outline-max-entries", type: "int", default: 40, settable: true, help: "Heading outline / TOC / sheet inventory cap in the handle" },
|
|
1014
1047
|
{ key: "ocr", flag: "--ocr", type: "bool", default: false, settable: true, help: "Run OCR on pages without a text layer and on image inputs when Tesseract language data is installed; off by default (--no-ocr turns a settings-level true off)" },
|
|
1015
|
-
{ key: "ocrLanguage", flag: "--ocr-language", type: "lang", default: "eng", settable: true, help: "Tesseract language code(s), +-joined, e.g. deu+eng" }
|
|
1048
|
+
{ key: "ocrLanguage", flag: "--ocr-language", type: "lang", default: "eng", settable: true, help: "Tesseract language code(s), +-joined, e.g. deu+eng" },
|
|
1049
|
+
{ key: "hideAnnotations", flag: "--hide-annotations", type: "bool", default: false, settable: true, help: "Render PDF pages without annotations (sticky notes, highlights, stamps - and form-field widgets, so filled form values disappear); default paints them, as PyMuPDF does. Applies to pages/ renders and textless-page renders, not to OCR text or embedded images; also lets an annotated scan be delivered as its embedded image." }
|
|
1016
1050
|
];
|
|
1017
1051
|
var TUNABLE_DEFAULTS = Object.fromEntries(
|
|
1018
1052
|
DOC_TO_MD_OPTIONS.filter((d) => d.settable).map((d) => [d.key, d.default])
|
|
@@ -1107,8 +1141,8 @@ function resolveOptions(perCall, settings, env) {
|
|
|
1107
1141
|
out[d.key] = value;
|
|
1108
1142
|
}
|
|
1109
1143
|
const o = out;
|
|
1110
|
-
if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages || o.ocrMode === "all")) {
|
|
1111
|
-
throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite, --page-images or --ocr-mode all");
|
|
1144
|
+
if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages || o.words || o.ocrMode === "all")) {
|
|
1145
|
+
throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite, --page-images, --words or --ocr-mode all");
|
|
1112
1146
|
}
|
|
1113
1147
|
return o;
|
|
1114
1148
|
}
|
|
@@ -1129,6 +1163,18 @@ var USAGE_PATTERNS = [
|
|
|
1129
1163
|
" Details: doc/doc-to-md.md (bundle contract, failure buckets)."
|
|
1130
1164
|
].join("\n");
|
|
1131
1165
|
var usagePatterns = (cmd) => USAGE_PATTERNS.replaceAll("<cmd>", cmd);
|
|
1166
|
+
var BUNDLE_LAYOUT = [
|
|
1167
|
+
{ artifact: "<stem>.md", trigger: "always", content: "the Markdown", namedBy: "Saved-To: / savedTo" },
|
|
1168
|
+
{ artifact: "images/", trigger: "embedded or extracted figures", content: "image files linked from the Markdown", namedBy: "Images-Dir: / imagesDir" },
|
|
1169
|
+
{ artifact: "pages/<stem>-pNNN.<fmt>", trigger: "--page-images", content: "page renders at --image-dpi", namedBy: "Pages-Dir: / pagesDir" },
|
|
1170
|
+
{ artifact: "sheets/", trigger: "Excel input", content: "one CSV per non-empty worksheet", namedBy: "Sheets-Dir: / sheetsDir" },
|
|
1171
|
+
{ artifact: "attachments/", trigger: "email input", content: "saved attachments", namedBy: "Markdown attachment list" },
|
|
1172
|
+
{ artifact: "<stem>.pages.json", trigger: "Python PDF tiers (PDF, PPTX, DOC, DOCX via LibreOffice; not unpdf)", content: "per-page chars, image count, image coverage", namedBy: "Page-Stats: / pageStatsPath" },
|
|
1173
|
+
{ artifact: "<stem>.words.json", trigger: "--words", content: "per-page word boxes, source text/ocr", namedBy: "Words: / wordsPath" },
|
|
1174
|
+
{ artifact: "ocr/<stem>-pNNN.md", trigger: "--ocr --ocr-mode all", content: "recognized text of a forced page", namedBy: "OCR-Dir: / ocr.sidecars" },
|
|
1175
|
+
{ artifact: "ocr/<stem>-pNNN.words.json", trigger: "--ocr --ocr-mode all --words", content: "word boxes of that OCR", namedBy: "ocr.wordSidecars" }
|
|
1176
|
+
];
|
|
1177
|
+
var bundleLayoutText = () => BUNDLE_LAYOUT.map((r) => ` ${r.artifact.padEnd(28)} ${r.trigger}; ${r.content}; named by ${r.namedBy}`).join("\n");
|
|
1132
1178
|
function renderHelp() {
|
|
1133
1179
|
const row = (d) => ` ${(d.flag ?? "<path>").padEnd(26)} ${d.help}${d.default !== null && d.key !== "info" && d.key !== "overwrite" ? ` (default ${d.default})` : ""}`;
|
|
1134
1180
|
return [
|
|
@@ -1143,6 +1189,9 @@ function renderHelp() {
|
|
|
1143
1189
|
"Result: a handle (Saved-To, Images-Dir, Page-Stats, Page-Count, Outline ...). Read the Saved-To file for the Markdown.",
|
|
1144
1190
|
"Exit codes: 0 success, 1 runtime error, 2 usage error.",
|
|
1145
1191
|
"",
|
|
1192
|
+
"Bundle layout:",
|
|
1193
|
+
bundleLayoutText(),
|
|
1194
|
+
"",
|
|
1146
1195
|
usagePatterns("pi-quiver doc-to-md")
|
|
1147
1196
|
].join("\n");
|
|
1148
1197
|
}
|
|
@@ -1622,7 +1671,7 @@ function reconcileRenderMarkers(md, renderPages, fmt, sourceMap, reason) {
|
|
|
1622
1671
|
if (/<!--rvs?:\d+-->/.test(md)) throw new Error("internal: unresolved render marker");
|
|
1623
1672
|
return md;
|
|
1624
1673
|
}
|
|
1625
|
-
var emptyOcr = (lang) => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null, mode: "textless", sidecars: {}, ocrErrors: {}, killed: null, notAttempted: [], childError: null });
|
|
1674
|
+
var emptyOcr = (lang) => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null, mode: "textless", sidecars: {}, wordSidecars: {}, ocrErrors: {}, killed: null, notAttempted: [], childError: null });
|
|
1626
1675
|
var emptyOutcome = () => ({ written: [], noText: [], ocrFailed: [], ocrErrors: {}, budgetStopped: [], killed: null, notAttempted: [], childError: null });
|
|
1627
1676
|
var ocrPagesTag = (n) => `p${String(n).padStart(3, "0")}`;
|
|
1628
1677
|
var SIDE_MARKER_RE = /^--- end of page\.page_number=\d+ ---$/;
|
|
@@ -1667,6 +1716,22 @@ function handleOcr(tier, type, o, json) {
|
|
|
1667
1716
|
if (!x || type !== "image" && !x.textless.length && !x.pages.length && !x.ocrFailed.length) return null;
|
|
1668
1717
|
return { ...emptyOcr(o.ocrLanguage), ...x };
|
|
1669
1718
|
}
|
|
1719
|
+
var nativeFromChild = (b, json, notes) => (json.nativeImages ?? []).flatMap((e) => {
|
|
1720
|
+
const file = b.sourceMap.get(`p${e.page}/${e.file}`);
|
|
1721
|
+
if (!file) {
|
|
1722
|
+
notes.push(`Native image p${e.page}/${e.file} not published`);
|
|
1723
|
+
return [];
|
|
1724
|
+
}
|
|
1725
|
+
return [{ ...e, file: join3(b.root, file) }];
|
|
1726
|
+
});
|
|
1727
|
+
function retainedNative(b, kept, notes) {
|
|
1728
|
+
const out = [];
|
|
1729
|
+
for (const [page, k] of kept) {
|
|
1730
|
+
if (k.meta.native) out.push({ page, ...k.meta.native, file: join3(b.imagesDir, k.meta.native.file) });
|
|
1731
|
+
if (k.meta.dpi !== void 0 && k.meta.requestedDpi !== void 0 && k.meta.dpi < k.meta.requestedDpi) notes.push(`Page ${page} rendered at ${k.meta.dpi} dpi (requested ${k.meta.requestedDpi}; 50 Mpx ceiling)`);
|
|
1732
|
+
}
|
|
1733
|
+
return out;
|
|
1734
|
+
}
|
|
1670
1735
|
async function convertDocument(o, signal, seams) {
|
|
1671
1736
|
const s = { backend: (c) => getBackend(c, void 0, signal), runTier: runTierReal, office: tryConvertOffice, ...seams };
|
|
1672
1737
|
const inputPath = resolve2(o.path);
|
|
@@ -1693,11 +1758,13 @@ async function convertDocument(o, signal, seams) {
|
|
|
1693
1758
|
let office = null;
|
|
1694
1759
|
try {
|
|
1695
1760
|
let pdfPath = inputPath;
|
|
1696
|
-
const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr && !forced, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
|
|
1761
|
+
const base = { path: inputPath, pages: o.pages, ...o.words && (type === "pdf" || type === "image") ? { words: true } : {}, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr && !forced, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, hideAnnotations: o.hideAnnotations };
|
|
1697
1762
|
let tier, engine, json, degraded = null, fallbackReason = null;
|
|
1698
1763
|
let explicitBreaks = null;
|
|
1699
1764
|
let notes = [];
|
|
1765
|
+
let nativeImages = [];
|
|
1700
1766
|
let officeRoute = null;
|
|
1767
|
+
let copyReason = null;
|
|
1701
1768
|
if (type === "html") {
|
|
1702
1769
|
const prepared = await prepareHtml(inputPath, b.stagingDir);
|
|
1703
1770
|
if (signal?.aborted) throw new Error("aborted");
|
|
@@ -1737,6 +1804,7 @@ async function convertDocument(o, signal, seams) {
|
|
|
1737
1804
|
}
|
|
1738
1805
|
if (signal?.aborted) throw new Error("aborted");
|
|
1739
1806
|
if (json === void 0) {
|
|
1807
|
+
copyReason = reason;
|
|
1740
1808
|
clearStaging(b);
|
|
1741
1809
|
const file = `original${extname3(inputPath).toLowerCase()}`;
|
|
1742
1810
|
const dir = join3(b.stagingDir, "p1");
|
|
@@ -1849,11 +1917,12 @@ async function convertDocument(o, signal, seams) {
|
|
|
1849
1917
|
tier = "primary";
|
|
1850
1918
|
engine = "pymupdf4llm";
|
|
1851
1919
|
json = p.json;
|
|
1920
|
+
nativeImages = nativeFromChild(b, json, notes);
|
|
1852
1921
|
if (json.pageImages?.length) publishPageImages(b, json.pageCount ?? 0);
|
|
1853
1922
|
} else if ("userError" in p) throw new Error(p.userError);
|
|
1854
1923
|
else {
|
|
1855
1924
|
if (signal?.aborted) throw new Error("aborted");
|
|
1856
|
-
const keepPages = Object.fromEntries([...kept.entries()].map(([k, v]) => [String(k), v]));
|
|
1925
|
+
const keepPages = Object.fromEntries([...kept.entries()].map(([k, v]) => [String(k), v.files]));
|
|
1857
1926
|
rmSync2(b.pagesStagingDir, { recursive: true, force: true });
|
|
1858
1927
|
const f = await s.runTier("pdf-fallback", { ...pdfBase, keepPages }, b, signal, o.fallbackTimeoutMs, backend);
|
|
1859
1928
|
publishStaged(b);
|
|
@@ -1864,6 +1933,7 @@ async function convertDocument(o, signal, seams) {
|
|
|
1864
1933
|
json = f.json;
|
|
1865
1934
|
degraded = DEGRADED_TEXT;
|
|
1866
1935
|
fallbackReason = `primary ${p.reason}`;
|
|
1936
|
+
nativeImages = [...retainedNative(b, kept, notes), ...nativeFromChild(b, json, notes)].sort((x, y) => x.page - y.page);
|
|
1867
1937
|
}
|
|
1868
1938
|
}
|
|
1869
1939
|
}
|
|
@@ -1872,14 +1942,33 @@ async function convertDocument(o, signal, seams) {
|
|
|
1872
1942
|
if (tier === void 0 || engine === void 0 || json === void 0) throw new Error("internal: no tier produced output");
|
|
1873
1943
|
const pageStats = json.pageStats ?? null;
|
|
1874
1944
|
if (pageStats) writePageStats(b, pageStats);
|
|
1945
|
+
let wordsPath = null, wordsReason = null;
|
|
1946
|
+
const wordsErrors = {};
|
|
1947
|
+
const takeWordsErrors = (j) => {
|
|
1948
|
+
for (const [k, v] of Object.entries(j?.wordsErrors ?? {})) {
|
|
1949
|
+
const page = Number(k);
|
|
1950
|
+
if (Number.isFinite(page)) wordsErrors[page] = v;
|
|
1951
|
+
}
|
|
1952
|
+
};
|
|
1953
|
+
if (o.words) {
|
|
1954
|
+
if (type !== "pdf" && type !== "image") wordsReason = `none - word positions apply to PDF and image inputs only (${type})`;
|
|
1955
|
+
else if (tier === "unpdf") wordsReason = "none - unpdf tier has no page geometry";
|
|
1956
|
+
else if (engine === "copy") wordsReason = `none - image copied without conversion (${copyReason})`;
|
|
1957
|
+
else if (json?.words === true) {
|
|
1958
|
+
wordsReason = publishWords(b);
|
|
1959
|
+
if (wordsReason === null) wordsPath = b.wordsPath;
|
|
1960
|
+
} else wordsReason = `write failed - ${json?.wordsErrors?.file ?? "child reported no words document"}`;
|
|
1961
|
+
takeWordsErrors(json);
|
|
1962
|
+
}
|
|
1875
1963
|
let ocr = handleOcr(tier, type, o, json);
|
|
1876
1964
|
if (forced) {
|
|
1877
|
-
const r = await s.runTier("ocr-pages", { path: pdfPath, pages: o.pages, stem: b.stem, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, stagingDir: b.ocrStagingDir, dpi: o.imageDpi, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.primaryTimeoutMs, backend);
|
|
1965
|
+
const r = await s.runTier("ocr-pages", { path: pdfPath, pages: o.pages, ...o.words && type === "pdf" ? { words: true } : {}, stem: b.stem, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, stagingDir: b.ocrStagingDir, dpi: o.imageDpi, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.primaryTimeoutMs, backend);
|
|
1878
1966
|
if (signal?.aborted || !r.ok && "reason" in r && r.reason === "aborted") throw new Error("aborted");
|
|
1879
1967
|
if (r.ok && r.json.status === "unavailable") throw new Error(`OCR unavailable: ${r.json.reason} (install Tesseract; see doc/doc-to-md.md)`);
|
|
1880
1968
|
const outcome = r.ok && r.json.status === "ran" ? outcomeFromChild(r.json) : recoverOcrPages(b.ocrStagingDir, o.pages, !r.ok ? "userError" in r ? r.userError : `${r.reason}${detailSuffix(r)}` : "malformed child output");
|
|
1881
|
-
const sidecars = publishSidecars(b);
|
|
1882
|
-
ocr = { ...emptyOcr(o.ocrLanguage), ...json.ocr, status: "ran", reason: null, mode: "all", pages: outcome.written, noText: outcome.noText, ocrFailed: outcome.ocrFailed, ocrErrors: outcome.ocrErrors, budgetStopped: outcome.budgetStopped, killed: outcome.killed, notAttempted: outcome.notAttempted, childError: outcome.childError, sidecars: Object.fromEntries(sidecars) };
|
|
1969
|
+
const { sidecars, wordSidecars } = publishSidecars(b);
|
|
1970
|
+
ocr = { ...emptyOcr(o.ocrLanguage), ...json.ocr, status: "ran", reason: null, mode: "all", pages: outcome.written, noText: outcome.noText, ocrFailed: outcome.ocrFailed, ocrErrors: outcome.ocrErrors, budgetStopped: outcome.budgetStopped, killed: outcome.killed, notAttempted: outcome.notAttempted, childError: outcome.childError, sidecars: Object.fromEntries(sidecars), wordSidecars: Object.fromEntries(wordSidecars) };
|
|
1971
|
+
if (o.words && r.ok) takeWordsErrors(r.json);
|
|
1883
1972
|
}
|
|
1884
1973
|
if (!isExcel) notes = [...notes, ...json.notes ?? []];
|
|
1885
1974
|
if (b.renamedFrom) notes.splice(notes[0]?.startsWith("preview truncated:") ? 1 : 0, 0, `renamed to ${b.stem} (${b.renameReason})`);
|
|
@@ -1897,7 +1986,7 @@ async function convertDocument(o, signal, seams) {
|
|
|
1897
1986
|
` : "") + body;
|
|
1898
1987
|
commitBundle(b, markdown);
|
|
1899
1988
|
const outline = scanOutline(markdown, o.outlineMaxEntries);
|
|
1900
|
-
const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr, pageStats, pageStatsPath: pageStats ? b.pageStatsPath : null, ocrDir: b.ocrManifest.size ? b.ocrDir : null };
|
|
1989
|
+
const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr, pageStats, pageStatsPath: pageStats ? b.pageStatsPath : null, ocrDir: b.ocrManifest.size ? b.ocrDir : null, nativeImages, wordsPath, wordsReason, wordsErrors };
|
|
1901
1990
|
return { output: formatHandle(details), details };
|
|
1902
1991
|
} catch (e) {
|
|
1903
1992
|
abortBundle(b);
|
|
@@ -1943,7 +2032,7 @@ async function inspectDocument(o, signal, seams) {
|
|
|
1943
2032
|
}
|
|
1944
2033
|
|
|
1945
2034
|
// bin/pi-quiver.ts
|
|
1946
|
-
var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--json] [--info] [--page-images] [--pages <spec>] [--output-dir <dir>] [--overwrite] [--ocr] [--ocr-mode textless|all] [tunable flags] <path> (--help for all flags)';
|
|
2035
|
+
var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--json] [--info] [--page-images] [--words] [--pages <spec>] [--output-dir <dir>] [--overwrite] [--ocr] [--ocr-mode textless|all] [tunable flags] <path> (--help for all flags)';
|
|
1947
2036
|
function parseDocToMd(rest) {
|
|
1948
2037
|
if (rest.includes("--help") || rest.includes("-h")) return { ok: true, cmd: "doc-to-md-help" };
|
|
1949
2038
|
const json = rest.includes("--json");
|
package/lib/doc-to-md-bundle.ts
CHANGED
|
@@ -15,7 +15,7 @@ export interface Bundle {
|
|
|
15
15
|
sheetsDir: string; sheetsStagingDir: string;
|
|
16
16
|
pagesDir: string; pagesStagingDir: string;
|
|
17
17
|
attachmentsDir: string; attachmentsStagingDir: string;
|
|
18
|
-
ocrDir: string; ocrStagingDir: string; pageStatsPath: string;
|
|
18
|
+
ocrDir: string; ocrStagingDir: string; pageStatsPath: string; wordsPath: string;
|
|
19
19
|
manifest: Set<string>;
|
|
20
20
|
csvManifest: Set<string>;
|
|
21
21
|
pageManifest: Set<string>;
|
|
@@ -36,7 +36,7 @@ export function ownedCsvPattern(stem: string): RegExp {
|
|
|
36
36
|
|
|
37
37
|
export function ownedPagePattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d+\\.[a-z0-9]+$`); }
|
|
38
38
|
|
|
39
|
-
export function ownedOcrPattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d
|
|
39
|
+
export function ownedOcrPattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d+(?:\\.words\\.json|\\.md)$`); }
|
|
40
40
|
|
|
41
41
|
const FILE_LINK_RE = /\[[^\]]*\]\(\s*((?:sheets|attachments)\/[^)\s]+)\s*\)/g;
|
|
42
42
|
const IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
|
|
@@ -94,6 +94,7 @@ export function openBundle(root: string, requested: string, overwrite: boolean):
|
|
|
94
94
|
if (existsSync(pagesDir)) for (const f of readdirSync(pagesDir)) if (ownedPage.test(f)) rmSync(join(pagesDir, f), { force: true });
|
|
95
95
|
if (existsSync(attachmentsDir)) for (const f of readdirSync(attachmentsDir)) if (attachments.has(f)) rmSync(join(attachmentsDir, f), { force: true });
|
|
96
96
|
rmSync(join(root, `${stem}.pages.json`), { force: true });
|
|
97
|
+
rmSync(join(root, `${stem}.words.json`), { force: true });
|
|
97
98
|
const ownedOcr = ownedOcrPattern(stem);
|
|
98
99
|
if (existsSync(ocrDir)) for (const f of readdirSync(ocrDir)) if (ownedOcr.test(f)) rmSync(join(ocrDir, f), { force: true });
|
|
99
100
|
}
|
|
@@ -104,13 +105,16 @@ export function openBundle(root: string, requested: string, overwrite: boolean):
|
|
|
104
105
|
const pagesStagingDir = join(pagesDir, `.stage-${lockId}`);
|
|
105
106
|
const attachmentsStagingDir = join(attachmentsDir, `.stage-${lockId}`);
|
|
106
107
|
const ocrStagingDir = join(ocrDir, `.stage-${lockId}`);
|
|
107
|
-
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join(root, `${stem}.pages.json`), ocrManifest: new Set(), manifest: new Set(), csvManifest: new Set(), pageManifest: new Set(), attachmentManifest: new Set(), sourceMap: new Map() };
|
|
108
|
+
return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join(root, `${stem}.pages.json`), wordsPath: join(root, `${stem}.words.json`), ocrManifest: new Set(), manifest: new Set(), csvManifest: new Set(), pageManifest: new Set(), attachmentManifest: new Set(), sourceMap: new Map() };
|
|
108
109
|
} catch (e) { rmSync(lockPath, { force: true }); throw e; }
|
|
109
110
|
}
|
|
110
111
|
|
|
111
|
-
|
|
112
|
-
export
|
|
113
|
-
|
|
112
|
+
export interface DoneMeta { native?: { file: string; width: number; height: number }; dpi?: number; requestedDpi?: number }
|
|
113
|
+
export interface StagedPage { files: string[]; meta: DoneMeta }
|
|
114
|
+
|
|
115
|
+
/** Publish completed pages, retaining metadata with native filenames renamed; discard partial pages. */
|
|
116
|
+
export function publishStaged(b: Bundle): Map<number, StagedPage> {
|
|
117
|
+
const out = new Map<number, StagedPage>();
|
|
114
118
|
if (!existsSync(b.stagingDir)) return out;
|
|
115
119
|
for (const dir of readdirSync(b.stagingDir).sort()) {
|
|
116
120
|
const m = dir.match(/^p(\d+)$/);
|
|
@@ -118,6 +122,12 @@ export function publishStaged(b: Bundle): Map<number, string[]> {
|
|
|
118
122
|
const pageDir = join(b.stagingDir, dir);
|
|
119
123
|
if (!existsSync(join(pageDir, ".done"))) { rmSync(pageDir, { recursive: true, force: true }); continue; }
|
|
120
124
|
const page = Number(m[1]);
|
|
125
|
+
const raw = readFileSync(join(pageDir, ".done"), "utf8").trim();
|
|
126
|
+
let meta: DoneMeta = {};
|
|
127
|
+
try {
|
|
128
|
+
const v = raw ? JSON.parse(raw) : {};
|
|
129
|
+
if (v && typeof v === "object" && !Array.isArray(v)) meta = v;
|
|
130
|
+
} catch { /* Completed images survive truncated metadata. */ }
|
|
121
131
|
const files = readdirSync(pageDir).filter((f) => f !== ".done" && statSync(join(pageDir, f)).isFile()).sort();
|
|
122
132
|
const names: string[] = [];
|
|
123
133
|
files.forEach((f, i) => {
|
|
@@ -127,8 +137,9 @@ export function publishStaged(b: Bundle): Map<number, string[]> {
|
|
|
127
137
|
b.sourceMap.set(`${dir}/${f}`, `images/${name}`);
|
|
128
138
|
names.push(name);
|
|
129
139
|
});
|
|
140
|
+
if (meta.native) meta.native = { ...meta.native, file: names[files.indexOf(meta.native.file)] ?? meta.native.file };
|
|
130
141
|
rmSync(pageDir, { recursive: true, force: true });
|
|
131
|
-
out.set(page, names);
|
|
142
|
+
out.set(page, { files: names, meta });
|
|
132
143
|
}
|
|
133
144
|
return out;
|
|
134
145
|
}
|
|
@@ -174,10 +185,23 @@ export function writePageStats(b: Bundle, stats: PageStat[]): void {
|
|
|
174
185
|
writeFileSync(b.pageStatsPath, `${JSON.stringify(stats, null, 2)}\n`, "utf8");
|
|
175
186
|
}
|
|
176
187
|
|
|
177
|
-
/**
|
|
178
|
-
export function
|
|
179
|
-
const
|
|
180
|
-
|
|
188
|
+
/** Rename within the bundle root keeps the complete words document atomic. */
|
|
189
|
+
export function publishWords(b: Bundle): string | null {
|
|
190
|
+
const staged = join(b.stagingDir, "words.json");
|
|
191
|
+
try {
|
|
192
|
+
if (!existsSync(staged)) throw new Error("child staged no words.json");
|
|
193
|
+
renameSync(staged, b.wordsPath);
|
|
194
|
+
return null;
|
|
195
|
+
} catch (e) {
|
|
196
|
+
rmSync(b.wordsPath, { force: true });
|
|
197
|
+
return `write failed - ${(e as Error).message}`;
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
/** Move `.done`-gated Markdown and words sidecars into `ocr/`; drop partial dirs and the checkpoint. Returns page -> absolute paths for each kind. */
|
|
202
|
+
export function publishSidecars(b: Bundle): { sidecars: Map<number, string>; wordSidecars: Map<number, string> } {
|
|
203
|
+
const sidecars = new Map<number, string>(), wordSidecars = new Map<number, string>();
|
|
204
|
+
if (!existsSync(b.ocrStagingDir)) return { sidecars, wordSidecars };
|
|
181
205
|
for (const dir of readdirSync(b.ocrStagingDir).sort()) {
|
|
182
206
|
const m = dir.match(/^p(\d+)$/);
|
|
183
207
|
if (!m) continue;
|
|
@@ -188,10 +212,16 @@ export function publishSidecars(b: Bundle): Map<number, string> {
|
|
|
188
212
|
mkdirSync(b.ocrDir, { recursive: true });
|
|
189
213
|
renameSync(join(pageDir, file), join(b.ocrDir, file));
|
|
190
214
|
b.ocrManifest.add(file);
|
|
191
|
-
|
|
215
|
+
sidecars.set(Number(m[1]), join(b.ocrDir, file));
|
|
216
|
+
const wfile = `${b.stem}-${dir}.words.json`;
|
|
217
|
+
if (existsSync(join(pageDir, wfile))) {
|
|
218
|
+
renameSync(join(pageDir, wfile), join(b.ocrDir, wfile));
|
|
219
|
+
b.ocrManifest.add(wfile);
|
|
220
|
+
wordSidecars.set(Number(m[1]), join(b.ocrDir, wfile));
|
|
221
|
+
}
|
|
192
222
|
}
|
|
193
223
|
rmSync(b.ocrStagingDir, { recursive: true, force: true });
|
|
194
|
-
return
|
|
224
|
+
return { sidecars, wordSidecars };
|
|
195
225
|
}
|
|
196
226
|
|
|
197
227
|
export function publishAttachments(b: Bundle): void {
|
|
@@ -254,6 +284,7 @@ export function abortBundle(b: Bundle): void {
|
|
|
254
284
|
for (const f of b.ocrManifest) rmSync(join(b.ocrDir, f), { force: true });
|
|
255
285
|
rmSync(b.ocrStagingDir, { recursive: true, force: true });
|
|
256
286
|
rmSync(b.pageStatsPath, { force: true });
|
|
287
|
+
rmSync(b.wordsPath, { force: true });
|
|
257
288
|
rmSync(`${b.mdPath}.tmp`, { force: true });
|
|
258
289
|
rmSync(b.stagingDir, { recursive: true, force: true });
|
|
259
290
|
rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|