pi-quiver 6.9.0 → 6.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -8,6 +8,19 @@ Published to npm as `pi-quiver` (`pi install npm:pi-quiver`). Pushing a
8
8
  via OIDC trusted publishing. The release helper at
9
9
  `.agents/skills/release/scripts/release.sh` cuts the tag; CI publishes.
10
10
 
11
+ ## v6.11.0 - 2026-10-02
12
+
13
+ - `doc_to_md`: textless PDF pages holding one eligible full-page image deliver the embedded stream without re-encoding; the handle prints `Native-Images:` and tool details / CLI JSON carry `nativeImages` (#27).
14
+ - `doc_to_md`: native probing, PDF page renders, and textless-page inline OCR run in a spawned raster worker with per-job budgets; a toxic page becomes a `Failed pages` note instead of failing the conversion when other pages succeed (#27).
15
+ - `doc_to_md`: the render ceiling rises from 16 to 50 Mpx; clamped PDF renders add Notes and report effective `pageImages[].dpi` plus `requestedDpi` (#27).
16
+ - `doc_to_md`: `hideAnnotations` / `--hide-annotations` hides PDF annotations and form widgets on renders and lets annotated scans use native delivery; default runs keep their painted render (#27).
17
+ - `doc_to_md`: `.done` markers retain native-image and clamp metadata for completed pages across primary-tier failure and fallback; corrupt metadata does not discard completed images (#27).
18
+
19
+ ## v6.10.0 - 2026-10-02
20
+
21
+ - `doc_to_md`: new per-call `words` option (`--words`; not settable) writes `<stem>.words.json` - per selected PDF/image page, every text-layer word and every word inline OCR recognized in the same run, with display-space bbox (points; source pixels for images) and `source` `text`/`ocr`. Under `--ocr-mode all` each sidecar gets `ocr/<stem>-pNNN.words.json`. Never triggers OCR; Markdown, page stats and OCR outcome are unchanged. Handle gains `Words:`, `--json` gains `wordsPath`, `wordsReason`, `wordsErrors`, `ocr.wordSidecars`; `--info --words` is a usage error (#28).
22
+ - `doc_to_md`: `--help` and the generated skill list every bundle artifact under `Bundle layout` (#28).
23
+
11
24
  ## v6.9.0 - 2026-10-01
12
25
 
13
26
  - `doc_to_md`: every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page `chars`, `images`, `imageCoverage`) and the handle prints `Page-Stats:`; CLI `--json` gains `pageStats`, `pageStatsPath`, `ocrDir`. The unpdf tier writes no stats (#26).
package/README.md CHANGED
@@ -65,7 +65,7 @@ A 300 KB changelog page never touches your context window - you get a preview an
65
65
  | Extension | Tool | What it does |
66
66
  | --- | --- | --- |
67
67
  | `extensions/fetch.ts` | `fetch` | Retrieve URLs over HTTP(S). HTML -> Markdown (Readability extraction, Turndown conversion). Binary saved untouched to a temp file. GitHub issue/PR/repo/actions-run/actions-job URLs auto-route through `gh` (falls back to HTTP); failed runs/jobs include failed-step logs (best-effort, summary-only otherwise). Same size gate as `fetch`. Behavior lives in `lib/fetch-core.ts`; also exposed as the `pi-quiver fetch` CLI (see [Claude Code support](#claude-code-support)). |
68
- | `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS, MSG/EML, HTML, or image file to a Markdown bundle on disk (`<stem>.md` + `images/`, spreadsheet `sheets/`, optional `pageImages` in `pages/`, and email `attachments/`) and return a handle (paths, page count, outline, diagnostics) - never inline Markdown. `info` inspects PDF/Office inputs; `pages` selects 1-based PDF/Office pages (DOCX: explicit-page-break segments); HTML, image, and email inputs reject both; every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Tiers: pymupdf4llm -> PyMuPDF text (degraded) -> unpdf worker (no Python only). Excel -> sheet inventory, full CSVs, bounded previews, and optional rendered views. HTML keeps local and `data:` images in the bundle and remote images as links; image inputs keep the original image. Scanned PDF pages keep a page picture on both Python tiers; opt-in OCR needs Tesseract language data. DOCX converts directly (mammoth, python-docx fallback); LibreOffice pagination marks the degraded route. The Outline lists `L<line>` and `p<page>` per heading. Settings under `quiver.docToMd`. Every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page chars, image count, image coverage; handle line `Page-Stats:`); unpdf writes no stats. `ocrMode: "all"` with `ocr: true` and an explicit `pages` forces OCR on those pages into `ocr/<stem>-pNNN.md` sidecars, leaving the Markdown byte-identical to the same selected-page call without OCR. |
68
+ | `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/DOC/PPTX/XLSX/XLSM/XLS, MSG/EML, HTML, or image file to a Markdown bundle on disk (`<stem>.md` + `images/`, spreadsheet `sheets/`, optional `pageImages` in `pages/`, and email `attachments/`) and return a handle (paths, page count, outline, diagnostics) - never inline Markdown. `info` inspects PDF/Office inputs; `pages` selects 1-based PDF/Office pages (DOCX: explicit-page-break segments); HTML, image, and email inputs reject both; every selected PDF/PPTX page or DOCX segment when `pageCount > 1` ends with `--- end of page.page_number=N ---`. Tiers: pymupdf4llm -> PyMuPDF text (degraded) -> unpdf worker (no Python only). Excel -> sheet inventory, full CSVs, bounded previews, and optional rendered views. HTML keeps local and `data:` images in the bundle and remote images as links; image inputs keep the original image. Scanned PDF pages keep a page picture on both Python tiers; eligible single full-page images are delivered as their embedded stream, without re-encoding or render DPI; opt-in OCR needs Tesseract language data. DOCX converts directly (mammoth, python-docx fallback); LibreOffice pagination marks the degraded route. The Outline lists `L<line>` and `p<page>` per heading. Settings under `quiver.docToMd`. Every conversion that goes through the Python PDF tiers (PDF, PPTX, `.doc`, DOCX via LibreOffice) writes `<stem>.pages.json` (per-page chars, image count, image coverage; handle line `Page-Stats:`); unpdf writes no stats. `ocrMode: "all"` with `ocr: true` and an explicit `pages` forces OCR on those pages into `ocr/<stem>-pNNN.md` sidecars, leaving the Markdown byte-identical to the same selected-page call without OCR. `words: true` (CLI `--words`, per-call only - no settings key) writes `<stem>.words.json` with every word's bbox per selected PDF/image page (`source` `text` or `ocr`), plus `ocr/<stem>-pNNN.words.json` beside each forced-OCR sidecar; it never triggers OCR. |
69
69
  | `extensions/session-name.ts` | `/session-name` | Manual + opt-in automatic session naming, naming rules and deny list, long-session revisits, and Ghostty/Herdr tab rename. OFF by default. |
70
70
  | `extensions/sword-header.ts` | `/builtin-header` | Themed ASCII startup header replacing pi's default logo. OFF by default. |
71
71
  | `extensions/fast-mode.ts` | `/fast` | Inject Anthropic fast-mode (`speed: "fast"` + `anthropic-beta: fast-mode-2026-02-01`) into every Claude Opus 4.8 / Opus 5 request, any thinking level. `--fast` flag + `/fast [on\|off\|status]`. OFF by default. |
@@ -303,8 +303,9 @@ Each setting can also be overridden per-process via `PI_QUIVER_SLACK_ENABLED`, `
303
303
  | `excelTimeoutMs` | `60000` | Excel child and Excel info deadline. |
304
304
  | `warmTimeoutMs` | `120000` | Absolute first-call backend discovery/bootstrap deadline. |
305
305
  | `pymupdfVersion` | `1.27.2.3` | pymupdf4llm pin, minimum `1.27.0`. |
306
- | `imageDpi` | `150` | Render DPI for page images and Excel rendered views (capped by a 16 Mpx budget). |
306
+ | `imageDpi` | `150` | Render DPI for page images and Excel rendered views (capped by a 50 Mpx budget). |
307
307
  | `imageFormat` | `png` | Rendered image format: `png` or `jpg`; embedded images retain their extension. |
308
+ | `hideAnnotations` | `false` | Hide annotations on PDF textless-page and `pages/` renders, including form-field widgets (filled values disappear). Default paints them. Does not affect OCR text or embedded images; also lets an annotated scan be delivered as its embedded image. |
308
309
  | `maxOutputBytes` | `20000000` | Child stdout cap in bytes. |
309
310
  | `outlineMaxEntries` | `40` | Heading outline, TOC, or sheet inventory entries in the handle. |
310
311
  | `ocr` | `false` | Opt-in OCR for scanned pages and images when Tesseract language data is installed. |
@@ -314,7 +315,7 @@ Each setting can also be overridden per-process via `PI_QUIVER_SLACK_ENABLED`, `
314
315
 
315
316
  A bundle is `<outputDir>/<stem>.md` plus `<outputDir>/<stem>.pages.json` on Python PDF tiers, optional OCR sidecars under `<outputDir>/ocr/`, and `<outputDir>/images/` and, for spreadsheets with data, `<outputDir>/sheets/`; without `outputDir`, the tool creates a per-call temp root. A conversion owns `<stem>.md.lock` until it atomically publishes the Markdown. An existing `<stem>.md` fails the call unless `overwrite` is set; `overwrite` replaces that Markdown and the files it owns (`images/<stem>-p<N>-<n>.*`, `images/<stem>-s<idx>[-<n>].*`, `sheets/<stem>-s<idx>-<slug>.csv`, `<stem>.pages.json`, `ocr/<stem>-pNNN.md`), nothing else. Temp bundles are caller-owned - the tool never deletes a bundle it produced.
316
317
 
317
- Excel needs a Python backend with openpyxl, xlrd and pillow - otherwise the call fails with `Remedy: install uv, or pip install openpyxl xlrd pillow`. The Markdown opens with a `## Sheets` table listing every sheet in workbook order (0-based `#`, `worksheet`/`chartsheet`, size, hidden, chart and image counts, rendered view, CSV link for non-empty worksheets), then one section per sheet: a `Data:` line linking the full-content CSV under `sheets/` for non-empty worksheets, chart metadata from the workbook model, embedded images, an optional rendered view, a preview of at most the first 100 rows x 50 columns, and - only when the preview is truncated - a `Columns:` profile (type, non-empty count, min/max, distinct up to 50). Sizes are the extent of non-empty cells (the `info` handle reports the raw worksheet dimensions instead, which may be larger). Rendered views (`images/<stem>-s<idx>.<fmt>`) are produced for sheets carrying charts or images when LibreOffice is on `PATH`: the workbook is exported one PDF page per sheet and rasterized under a 16 Mpx budget. Any LibreOffice or rasterization failure degrades to `Rendered view: unavailable (<reason>)` and a handle note; it never fails the conversion. Workbooks whose chartsheet drawings carry a zero-size anchor (openpyxl-authored files; Excel-authored files are unaffected) render as a degenerate page and are reported as such. `.xls` gets the inventory, CSVs and previews but no visual detection or rendering.
318
+ Excel needs a Python backend with openpyxl, xlrd and pillow - otherwise the call fails with `Remedy: install uv, or pip install openpyxl xlrd pillow`. The Markdown opens with a `## Sheets` table listing every sheet in workbook order (0-based `#`, `worksheet`/`chartsheet`, size, hidden, chart and image counts, rendered view, CSV link for non-empty worksheets), then one section per sheet: a `Data:` line linking the full-content CSV under `sheets/` for non-empty worksheets, chart metadata from the workbook model, embedded images, an optional rendered view, a preview of at most the first 100 rows x 50 columns, and - only when the preview is truncated - a `Columns:` profile (type, non-empty count, min/max, distinct up to 50). Sizes are the extent of non-empty cells (the `info` handle reports the raw worksheet dimensions instead, which may be larger). Rendered views (`images/<stem>-s<idx>.<fmt>`) are produced for sheets carrying charts or images when LibreOffice is on `PATH`: the workbook is exported one PDF page per sheet and rasterized under a 50 Mpx budget. Any LibreOffice or rasterization failure degrades to `Rendered view: unavailable (<reason>)` and a handle note; it never fails the conversion. Workbooks whose chartsheet drawings carry a zero-size anchor (openpyxl-authored files; Excel-authored files are unaffected) render as a degenerate page and are reported as such. `.xls` gets the inventory, CSVs and previews but no visual detection or rendering.
318
319
 
319
320
  Worst-case wall time: PDF `warmTimeoutMs (first call) + primaryTimeoutMs + fallbackTimeoutMs`; PPTX adds `sofficeTimeoutMs`; DOCX on the Python path `warmTimeoutMs + primaryTimeoutMs` (success or a terminal child failure), DOCX child exit 1 then LibreOffice `warmTimeoutMs + primaryTimeoutMs + sofficeTimeoutMs + primaryTimeoutMs + fallbackTimeoutMs`, DOCX without a DOCX-capable backend `warmTimeoutMs + sofficeTimeoutMs + primaryTimeoutMs + fallbackTimeoutMs`; Excel `warmTimeoutMs + excelTimeoutMs + sofficeTimeoutMs + fallbackTimeoutMs`. Add `KILL_GRACE_MS` (2000 ms) per kill. There is no cap on image count, image bytes, cell count or workbook memory - deliberately; the per-tier timeouts, the rendered-view pixel budget and `maxOutputBytes` are the bounds.
320
321
 
@@ -582,7 +582,7 @@ function ownedPagePattern(stem) {
582
582
  return new RegExp(`^${escRe(stem)}-p\\d+\\.[a-z0-9]+$`);
583
583
  }
584
584
  function ownedOcrPattern(stem) {
585
- return new RegExp(`^${escRe(stem)}-p\\d+\\.md$`);
585
+ return new RegExp(`^${escRe(stem)}-p\\d+(?:\\.words\\.json|\\.md)$`);
586
586
  }
587
587
  var FILE_LINK_RE = /\[[^\]]*\]\(\s*((?:sheets|attachments)\/[^)\s]+)\s*\)/g;
588
588
  var IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
@@ -649,6 +649,7 @@ function openBundle(root, requested, overwrite) {
649
649
  for (const f of readdirSync(attachmentsDir)) if (attachments.has(f)) rmSync(join2(attachmentsDir, f), { force: true });
650
650
  }
651
651
  rmSync(join2(root, `${stem}.pages.json`), { force: true });
652
+ rmSync(join2(root, `${stem}.words.json`), { force: true });
652
653
  const ownedOcr = ownedOcrPattern(stem);
653
654
  if (existsSync(ocrDir)) {
654
655
  for (const f of readdirSync(ocrDir)) if (ownedOcr.test(f)) rmSync(join2(ocrDir, f), { force: true });
@@ -661,7 +662,7 @@ function openBundle(root, requested, overwrite) {
661
662
  const pagesStagingDir = join2(pagesDir, `.stage-${lockId}`);
662
663
  const attachmentsStagingDir = join2(attachmentsDir, `.stage-${lockId}`);
663
664
  const ocrStagingDir = join2(ocrDir, `.stage-${lockId}`);
664
- return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join2(root, `${stem}.pages.json`), ocrManifest: /* @__PURE__ */ new Set(), manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), pageManifest: /* @__PURE__ */ new Set(), attachmentManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
665
+ return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join2(root, `${stem}.pages.json`), wordsPath: join2(root, `${stem}.words.json`), ocrManifest: /* @__PURE__ */ new Set(), manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), pageManifest: /* @__PURE__ */ new Set(), attachmentManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
665
666
  } catch (e) {
666
667
  rmSync(lockPath, { force: true });
667
668
  throw e;
@@ -679,6 +680,13 @@ function publishStaged(b) {
679
680
  continue;
680
681
  }
681
682
  const page = Number(m[1]);
683
+ const raw = readFileSync(join2(pageDir, ".done"), "utf8").trim();
684
+ let meta = {};
685
+ try {
686
+ const v = raw ? JSON.parse(raw) : {};
687
+ if (v && typeof v === "object" && !Array.isArray(v)) meta = v;
688
+ } catch {
689
+ }
682
690
  const files = readdirSync(pageDir).filter((f) => f !== ".done" && statSync(join2(pageDir, f)).isFile()).sort();
683
691
  const names = [];
684
692
  files.forEach((f, i) => {
@@ -688,8 +696,9 @@ function publishStaged(b) {
688
696
  b.sourceMap.set(`${dir}/${f}`, `images/${name}`);
689
697
  names.push(name);
690
698
  });
699
+ if (meta.native) meta.native = { ...meta.native, file: names[files.indexOf(meta.native.file)] ?? meta.native.file };
691
700
  rmSync(pageDir, { recursive: true, force: true });
692
- out.set(page, names);
701
+ out.set(page, { files: names, meta });
693
702
  }
694
703
  return out;
695
704
  }
@@ -729,9 +738,20 @@ function writePageStats(b, stats) {
729
738
  writeFileSync2(b.pageStatsPath, `${JSON.stringify(stats, null, 2)}
730
739
  `, "utf8");
731
740
  }
741
+ function publishWords(b) {
742
+ const staged = join2(b.stagingDir, "words.json");
743
+ try {
744
+ if (!existsSync(staged)) throw new Error("child staged no words.json");
745
+ renameSync(staged, b.wordsPath);
746
+ return null;
747
+ } catch (e) {
748
+ rmSync(b.wordsPath, { force: true });
749
+ return `write failed - ${e.message}`;
750
+ }
751
+ }
732
752
  function publishSidecars(b) {
733
- const out = /* @__PURE__ */ new Map();
734
- if (!existsSync(b.ocrStagingDir)) return out;
753
+ const sidecars = /* @__PURE__ */ new Map(), wordSidecars = /* @__PURE__ */ new Map();
754
+ if (!existsSync(b.ocrStagingDir)) return { sidecars, wordSidecars };
735
755
  for (const dir of readdirSync(b.ocrStagingDir).sort()) {
736
756
  const m = dir.match(/^p(\d+)$/);
737
757
  if (!m) continue;
@@ -742,10 +762,16 @@ function publishSidecars(b) {
742
762
  mkdirSync2(b.ocrDir, { recursive: true });
743
763
  renameSync(join2(pageDir, file), join2(b.ocrDir, file));
744
764
  b.ocrManifest.add(file);
745
- out.set(Number(m[1]), join2(b.ocrDir, file));
765
+ sidecars.set(Number(m[1]), join2(b.ocrDir, file));
766
+ const wfile = `${b.stem}-${dir}.words.json`;
767
+ if (existsSync(join2(pageDir, wfile))) {
768
+ renameSync(join2(pageDir, wfile), join2(b.ocrDir, wfile));
769
+ b.ocrManifest.add(wfile);
770
+ wordSidecars.set(Number(m[1]), join2(b.ocrDir, wfile));
771
+ }
746
772
  }
747
773
  rmSync(b.ocrStagingDir, { recursive: true, force: true });
748
- return out;
774
+ return { sidecars, wordSidecars };
749
775
  }
750
776
  function publishAttachments(b) {
751
777
  if (!existsSync(b.attachmentsStagingDir)) return;
@@ -821,6 +847,7 @@ function abortBundle(b) {
821
847
  for (const f of b.ocrManifest) rmSync(join2(b.ocrDir, f), { force: true });
822
848
  rmSync(b.ocrStagingDir, { recursive: true, force: true });
823
849
  rmSync(b.pageStatsPath, { force: true });
850
+ rmSync(b.wordsPath, { force: true });
824
851
  rmSync(`${b.mdPath}.tmp`, { force: true });
825
852
  rmSync(b.stagingDir, { recursive: true, force: true });
826
853
  rmSync(b.sheetsStagingDir, { recursive: true, force: true });
@@ -950,6 +977,11 @@ function formatHandle(h) {
950
977
  if (h.pagesDir && h.pageImageCount > 0) lines.push(`Pages-Dir: ${h.pagesDir} (${h.pageImageCount} pages)`);
951
978
  else if (h.pageImagesReason) lines.push(`Pages-Dir: none - ${h.pageImagesReason}`);
952
979
  if (h.pageStatsPath) lines.push(`Page-Stats: ${h.pageStatsPath}`);
980
+ if (h.nativeImages.length) lines.push(`Native-Images: ${h.nativeImages.length === 1 ? "page" : "pages"} ${compactRanges(h.nativeImages.map((n) => n.page))} (embedded image streams, no render DPI)`);
981
+ if (h.wordsPath) {
982
+ const bad = Object.keys(h.wordsErrors).map(Number).sort((a, b) => a - b);
983
+ lines.push(`Words: ${h.wordsPath}${bad.length ? ` (extraction failed for pages ${compactRanges(bad)}: ${h.wordsErrors[bad[0]]})` : ""}`);
984
+ } else if (h.wordsReason) lines.push(`Words: ${h.wordsReason}`);
953
985
  if (h.ocrDir) lines.push(`OCR-Dir: ${h.ocrDir}`);
954
986
  lines.push(`Type: ${h.type} Engine: ${h.engine} Tier: ${h.tier}`);
955
987
  lines.push(`Page-Count: ${pageCountLabel(h)} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize2(h.bytes)} / ${h.lines} lines`);
@@ -1000,6 +1032,7 @@ var DOC_TO_MD_OPTIONS = [
1000
1032
  { key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir. <stem> = basename without extension with [^A-Za-z0-9._-]+ -> _ (empty -> document); a second call on the same stem writes <stem>-2.md" },
1001
1033
  { key: "overwrite", flag: "--overwrite", type: "bool", default: false, settable: false, help: "Replace an existing completed <stem>.md bundle" },
1002
1034
  { key: "pageImages", flag: "--page-images", type: "bool", default: false, settable: false, help: "Also render every selected page to pages/<stem>-pNNN.<imageFormat> at imageDpi (PDF, PPTX, .doc, DOCX via LibreOffice); off by default" },
1035
+ { key: "words", flag: "--words", type: "bool", default: false, settable: false, help: 'Write word positions: <stem>.words.json beside the Markdown lists every text-layer word of each selected page with its bbox (PDF points, top-left origin, display orientation; image inputs in source pixels) and the words inline OCR recognized, tagged source "text" or "ocr"; under --ocr-mode all the OCR words go to ocr/<stem>-pNNN.words.json beside each sidecar. Never triggers OCR. PDF and image inputs only.' },
1003
1036
  { key: "ocrMode", flag: "--ocr-mode", type: "enum", default: "textless", settable: false, enumValues: ["textless", "all"], help: "OCR policy: textless (default) OCRs only pages with an empty text layer, inline; all OCRs every selected page and writes the recognized text to ocr/<stem>-pNNN.md sidecars, leaving the Markdown untouched. all requires --ocr and an explicit --pages selection (PDF, PPTX, DOC)." },
1004
1037
  { key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 6e4, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier and DOCX child (docx mode); also the unpdf tier" },
1005
1038
  { key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier (including DOCX LibreOffice fallback); also PDF and DOCX info and Excel rendered views" },
@@ -1012,7 +1045,8 @@ var DOC_TO_MD_OPTIONS = [
1012
1045
  { key: "maxOutputBytes", flag: "--max-output-bytes", type: "int", default: 2e7, settable: true, help: "Child stdout cap in bytes" },
1013
1046
  { key: "outlineMaxEntries", flag: "--outline-max-entries", type: "int", default: 40, settable: true, help: "Heading outline / TOC / sheet inventory cap in the handle" },
1014
1047
  { key: "ocr", flag: "--ocr", type: "bool", default: false, settable: true, help: "Run OCR on pages without a text layer and on image inputs when Tesseract language data is installed; off by default (--no-ocr turns a settings-level true off)" },
1015
- { key: "ocrLanguage", flag: "--ocr-language", type: "lang", default: "eng", settable: true, help: "Tesseract language code(s), +-joined, e.g. deu+eng" }
1048
+ { key: "ocrLanguage", flag: "--ocr-language", type: "lang", default: "eng", settable: true, help: "Tesseract language code(s), +-joined, e.g. deu+eng" },
1049
+ { key: "hideAnnotations", flag: "--hide-annotations", type: "bool", default: false, settable: true, help: "Render PDF pages without annotations (sticky notes, highlights, stamps - and form-field widgets, so filled form values disappear); default paints them, as PyMuPDF does. Applies to pages/ renders and textless-page renders, not to OCR text or embedded images; also lets an annotated scan be delivered as its embedded image." }
1016
1050
  ];
1017
1051
  var TUNABLE_DEFAULTS = Object.fromEntries(
1018
1052
  DOC_TO_MD_OPTIONS.filter((d) => d.settable).map((d) => [d.key, d.default])
@@ -1107,8 +1141,8 @@ function resolveOptions(perCall, settings, env) {
1107
1141
  out[d.key] = value;
1108
1142
  }
1109
1143
  const o = out;
1110
- if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages || o.ocrMode === "all")) {
1111
- throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite, --page-images or --ocr-mode all");
1144
+ if (o.info && (o.pages !== null || o.outputDir !== null || o.overwrite || o.pageImages || o.words || o.ocrMode === "all")) {
1145
+ throw new UsageError("--info cannot be combined with --pages, --output-dir, --overwrite, --page-images, --words or --ocr-mode all");
1112
1146
  }
1113
1147
  return o;
1114
1148
  }
@@ -1129,6 +1163,18 @@ var USAGE_PATTERNS = [
1129
1163
  " Details: doc/doc-to-md.md (bundle contract, failure buckets)."
1130
1164
  ].join("\n");
1131
1165
  var usagePatterns = (cmd) => USAGE_PATTERNS.replaceAll("<cmd>", cmd);
1166
+ var BUNDLE_LAYOUT = [
1167
+ { artifact: "<stem>.md", trigger: "always", content: "the Markdown", namedBy: "Saved-To: / savedTo" },
1168
+ { artifact: "images/", trigger: "embedded or extracted figures", content: "image files linked from the Markdown", namedBy: "Images-Dir: / imagesDir" },
1169
+ { artifact: "pages/<stem>-pNNN.<fmt>", trigger: "--page-images", content: "page renders at --image-dpi", namedBy: "Pages-Dir: / pagesDir" },
1170
+ { artifact: "sheets/", trigger: "Excel input", content: "one CSV per non-empty worksheet", namedBy: "Sheets-Dir: / sheetsDir" },
1171
+ { artifact: "attachments/", trigger: "email input", content: "saved attachments", namedBy: "Markdown attachment list" },
1172
+ { artifact: "<stem>.pages.json", trigger: "Python PDF tiers (PDF, PPTX, DOC, DOCX via LibreOffice; not unpdf)", content: "per-page chars, image count, image coverage", namedBy: "Page-Stats: / pageStatsPath" },
1173
+ { artifact: "<stem>.words.json", trigger: "--words", content: "per-page word boxes, source text/ocr", namedBy: "Words: / wordsPath" },
1174
+ { artifact: "ocr/<stem>-pNNN.md", trigger: "--ocr --ocr-mode all", content: "recognized text of a forced page", namedBy: "OCR-Dir: / ocr.sidecars" },
1175
+ { artifact: "ocr/<stem>-pNNN.words.json", trigger: "--ocr --ocr-mode all --words", content: "word boxes of that OCR", namedBy: "ocr.wordSidecars" }
1176
+ ];
1177
+ var bundleLayoutText = () => BUNDLE_LAYOUT.map((r) => ` ${r.artifact.padEnd(28)} ${r.trigger}; ${r.content}; named by ${r.namedBy}`).join("\n");
1132
1178
  function renderHelp() {
1133
1179
  const row = (d) => ` ${(d.flag ?? "<path>").padEnd(26)} ${d.help}${d.default !== null && d.key !== "info" && d.key !== "overwrite" ? ` (default ${d.default})` : ""}`;
1134
1180
  return [
@@ -1143,6 +1189,9 @@ function renderHelp() {
1143
1189
  "Result: a handle (Saved-To, Images-Dir, Page-Stats, Page-Count, Outline ...). Read the Saved-To file for the Markdown.",
1144
1190
  "Exit codes: 0 success, 1 runtime error, 2 usage error.",
1145
1191
  "",
1192
+ "Bundle layout:",
1193
+ bundleLayoutText(),
1194
+ "",
1146
1195
  usagePatterns("pi-quiver doc-to-md")
1147
1196
  ].join("\n");
1148
1197
  }
@@ -1622,7 +1671,7 @@ function reconcileRenderMarkers(md, renderPages, fmt, sourceMap, reason) {
1622
1671
  if (/<!--rvs?:\d+-->/.test(md)) throw new Error("internal: unresolved render marker");
1623
1672
  return md;
1624
1673
  }
1625
- var emptyOcr = (lang) => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null, mode: "textless", sidecars: {}, ocrErrors: {}, killed: null, notAttempted: [], childError: null });
1674
+ var emptyOcr = (lang) => ({ status: "off", lang, textless: [], pages: [], noText: [], ocrFailed: [], budgetStopped: [], reason: null, tesseract: null, mode: "textless", sidecars: {}, wordSidecars: {}, ocrErrors: {}, killed: null, notAttempted: [], childError: null });
1626
1675
  var emptyOutcome = () => ({ written: [], noText: [], ocrFailed: [], ocrErrors: {}, budgetStopped: [], killed: null, notAttempted: [], childError: null });
1627
1676
  var ocrPagesTag = (n) => `p${String(n).padStart(3, "0")}`;
1628
1677
  var SIDE_MARKER_RE = /^--- end of page\.page_number=\d+ ---$/;
@@ -1667,6 +1716,22 @@ function handleOcr(tier, type, o, json) {
1667
1716
  if (!x || type !== "image" && !x.textless.length && !x.pages.length && !x.ocrFailed.length) return null;
1668
1717
  return { ...emptyOcr(o.ocrLanguage), ...x };
1669
1718
  }
1719
+ var nativeFromChild = (b, json, notes) => (json.nativeImages ?? []).flatMap((e) => {
1720
+ const file = b.sourceMap.get(`p${e.page}/${e.file}`);
1721
+ if (!file) {
1722
+ notes.push(`Native image p${e.page}/${e.file} not published`);
1723
+ return [];
1724
+ }
1725
+ return [{ ...e, file: join3(b.root, file) }];
1726
+ });
1727
+ function retainedNative(b, kept, notes) {
1728
+ const out = [];
1729
+ for (const [page, k] of kept) {
1730
+ if (k.meta.native) out.push({ page, ...k.meta.native, file: join3(b.imagesDir, k.meta.native.file) });
1731
+ if (k.meta.dpi !== void 0 && k.meta.requestedDpi !== void 0 && k.meta.dpi < k.meta.requestedDpi) notes.push(`Page ${page} rendered at ${k.meta.dpi} dpi (requested ${k.meta.requestedDpi}; 50 Mpx ceiling)`);
1732
+ }
1733
+ return out;
1734
+ }
1670
1735
  async function convertDocument(o, signal, seams) {
1671
1736
  const s = { backend: (c) => getBackend(c, void 0, signal), runTier: runTierReal, office: tryConvertOffice, ...seams };
1672
1737
  const inputPath = resolve2(o.path);
@@ -1693,11 +1758,13 @@ async function convertDocument(o, signal, seams) {
1693
1758
  let office = null;
1694
1759
  try {
1695
1760
  let pdfPath = inputPath;
1696
- const base = { path: inputPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr && !forced, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs };
1761
+ const base = { path: inputPath, pages: o.pages, ...o.words && (type === "pdf" || type === "image") ? { words: true } : {}, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, pageImages: o.pageImages, pagesStagingDir: b.pagesStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion, ocr: o.ocr && !forced, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, hideAnnotations: o.hideAnnotations };
1697
1762
  let tier, engine, json, degraded = null, fallbackReason = null;
1698
1763
  let explicitBreaks = null;
1699
1764
  let notes = [];
1765
+ let nativeImages = [];
1700
1766
  let officeRoute = null;
1767
+ let copyReason = null;
1701
1768
  if (type === "html") {
1702
1769
  const prepared = await prepareHtml(inputPath, b.stagingDir);
1703
1770
  if (signal?.aborted) throw new Error("aborted");
@@ -1737,6 +1804,7 @@ async function convertDocument(o, signal, seams) {
1737
1804
  }
1738
1805
  if (signal?.aborted) throw new Error("aborted");
1739
1806
  if (json === void 0) {
1807
+ copyReason = reason;
1740
1808
  clearStaging(b);
1741
1809
  const file = `original${extname3(inputPath).toLowerCase()}`;
1742
1810
  const dir = join3(b.stagingDir, "p1");
@@ -1849,11 +1917,12 @@ async function convertDocument(o, signal, seams) {
1849
1917
  tier = "primary";
1850
1918
  engine = "pymupdf4llm";
1851
1919
  json = p.json;
1920
+ nativeImages = nativeFromChild(b, json, notes);
1852
1921
  if (json.pageImages?.length) publishPageImages(b, json.pageCount ?? 0);
1853
1922
  } else if ("userError" in p) throw new Error(p.userError);
1854
1923
  else {
1855
1924
  if (signal?.aborted) throw new Error("aborted");
1856
- const keepPages = Object.fromEntries([...kept.entries()].map(([k, v]) => [String(k), v]));
1925
+ const keepPages = Object.fromEntries([...kept.entries()].map(([k, v]) => [String(k), v.files]));
1857
1926
  rmSync2(b.pagesStagingDir, { recursive: true, force: true });
1858
1927
  const f = await s.runTier("pdf-fallback", { ...pdfBase, keepPages }, b, signal, o.fallbackTimeoutMs, backend);
1859
1928
  publishStaged(b);
@@ -1864,6 +1933,7 @@ async function convertDocument(o, signal, seams) {
1864
1933
  json = f.json;
1865
1934
  degraded = DEGRADED_TEXT;
1866
1935
  fallbackReason = `primary ${p.reason}`;
1936
+ nativeImages = [...retainedNative(b, kept, notes), ...nativeFromChild(b, json, notes)].sort((x, y) => x.page - y.page);
1867
1937
  }
1868
1938
  }
1869
1939
  }
@@ -1872,14 +1942,33 @@ async function convertDocument(o, signal, seams) {
1872
1942
  if (tier === void 0 || engine === void 0 || json === void 0) throw new Error("internal: no tier produced output");
1873
1943
  const pageStats = json.pageStats ?? null;
1874
1944
  if (pageStats) writePageStats(b, pageStats);
1945
+ let wordsPath = null, wordsReason = null;
1946
+ const wordsErrors = {};
1947
+ const takeWordsErrors = (j) => {
1948
+ for (const [k, v] of Object.entries(j?.wordsErrors ?? {})) {
1949
+ const page = Number(k);
1950
+ if (Number.isFinite(page)) wordsErrors[page] = v;
1951
+ }
1952
+ };
1953
+ if (o.words) {
1954
+ if (type !== "pdf" && type !== "image") wordsReason = `none - word positions apply to PDF and image inputs only (${type})`;
1955
+ else if (tier === "unpdf") wordsReason = "none - unpdf tier has no page geometry";
1956
+ else if (engine === "copy") wordsReason = `none - image copied without conversion (${copyReason})`;
1957
+ else if (json?.words === true) {
1958
+ wordsReason = publishWords(b);
1959
+ if (wordsReason === null) wordsPath = b.wordsPath;
1960
+ } else wordsReason = `write failed - ${json?.wordsErrors?.file ?? "child reported no words document"}`;
1961
+ takeWordsErrors(json);
1962
+ }
1875
1963
  let ocr = handleOcr(tier, type, o, json);
1876
1964
  if (forced) {
1877
- const r = await s.runTier("ocr-pages", { path: pdfPath, pages: o.pages, stem: b.stem, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, stagingDir: b.ocrStagingDir, dpi: o.imageDpi, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.primaryTimeoutMs, backend);
1965
+ const r = await s.runTier("ocr-pages", { path: pdfPath, pages: o.pages, ...o.words && type === "pdf" ? { words: true } : {}, stem: b.stem, ocrLanguage: o.ocrLanguage, ocrBudgetMs: o.primaryTimeoutMs, stagingDir: b.ocrStagingDir, dpi: o.imageDpi, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.primaryTimeoutMs, backend);
1878
1966
  if (signal?.aborted || !r.ok && "reason" in r && r.reason === "aborted") throw new Error("aborted");
1879
1967
  if (r.ok && r.json.status === "unavailable") throw new Error(`OCR unavailable: ${r.json.reason} (install Tesseract; see doc/doc-to-md.md)`);
1880
1968
  const outcome = r.ok && r.json.status === "ran" ? outcomeFromChild(r.json) : recoverOcrPages(b.ocrStagingDir, o.pages, !r.ok ? "userError" in r ? r.userError : `${r.reason}${detailSuffix(r)}` : "malformed child output");
1881
- const sidecars = publishSidecars(b);
1882
- ocr = { ...emptyOcr(o.ocrLanguage), ...json.ocr, status: "ran", reason: null, mode: "all", pages: outcome.written, noText: outcome.noText, ocrFailed: outcome.ocrFailed, ocrErrors: outcome.ocrErrors, budgetStopped: outcome.budgetStopped, killed: outcome.killed, notAttempted: outcome.notAttempted, childError: outcome.childError, sidecars: Object.fromEntries(sidecars) };
1969
+ const { sidecars, wordSidecars } = publishSidecars(b);
1970
+ ocr = { ...emptyOcr(o.ocrLanguage), ...json.ocr, status: "ran", reason: null, mode: "all", pages: outcome.written, noText: outcome.noText, ocrFailed: outcome.ocrFailed, ocrErrors: outcome.ocrErrors, budgetStopped: outcome.budgetStopped, killed: outcome.killed, notAttempted: outcome.notAttempted, childError: outcome.childError, sidecars: Object.fromEntries(sidecars), wordSidecars: Object.fromEntries(wordSidecars) };
1971
+ if (o.words && r.ok) takeWordsErrors(r.json);
1883
1972
  }
1884
1973
  if (!isExcel) notes = [...notes, ...json.notes ?? []];
1885
1974
  if (b.renamedFrom) notes.splice(notes[0]?.startsWith("preview truncated:") ? 1 : 0, 0, `renamed to ${b.stem} (${b.renameReason})`);
@@ -1897,7 +1986,7 @@ async function convertDocument(o, signal, seams) {
1897
1986
  ` : "") + body;
1898
1987
  commitBundle(b, markdown);
1899
1988
  const outline = scanOutline(markdown, o.outlineMaxEntries);
1900
- const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr, pageStats, pageStatsPath: pageStats ? b.pageStatsPath : null, ocrDir: b.ocrManifest.size ? b.ocrDir : null };
1989
+ const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, pagesDir: b.pageManifest.size ? b.pagesDir : null, pageImageCount: b.pageManifest.size, pageImagesReason, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, explicitBreaks, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total, ocr, pageStats, pageStatsPath: pageStats ? b.pageStatsPath : null, ocrDir: b.ocrManifest.size ? b.ocrDir : null, nativeImages, wordsPath, wordsReason, wordsErrors };
1901
1990
  return { output: formatHandle(details), details };
1902
1991
  } catch (e) {
1903
1992
  abortBundle(b);
@@ -1943,7 +2032,7 @@ async function inspectDocument(o, signal, seams) {
1943
2032
  }
1944
2033
 
1945
2034
  // bin/pi-quiver.ts
1946
- var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--json] [--info] [--page-images] [--pages <spec>] [--output-dir <dir>] [--overwrite] [--ocr] [--ocr-mode textless|all] [tunable flags] <path> (--help for all flags)';
2035
+ var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]\n pi-quiver doc-to-md [--json] [--info] [--page-images] [--words] [--pages <spec>] [--output-dir <dir>] [--overwrite] [--ocr] [--ocr-mode textless|all] [tunable flags] <path> (--help for all flags)';
1947
2036
  function parseDocToMd(rest) {
1948
2037
  if (rest.includes("--help") || rest.includes("-h")) return { ok: true, cmd: "doc-to-md-help" };
1949
2038
  const json = rest.includes("--json");
@@ -15,7 +15,7 @@ export interface Bundle {
15
15
  sheetsDir: string; sheetsStagingDir: string;
16
16
  pagesDir: string; pagesStagingDir: string;
17
17
  attachmentsDir: string; attachmentsStagingDir: string;
18
- ocrDir: string; ocrStagingDir: string; pageStatsPath: string;
18
+ ocrDir: string; ocrStagingDir: string; pageStatsPath: string; wordsPath: string;
19
19
  manifest: Set<string>;
20
20
  csvManifest: Set<string>;
21
21
  pageManifest: Set<string>;
@@ -36,7 +36,7 @@ export function ownedCsvPattern(stem: string): RegExp {
36
36
 
37
37
  export function ownedPagePattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d+\\.[a-z0-9]+$`); }
38
38
 
39
- export function ownedOcrPattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d+\\.md$`); }
39
+ export function ownedOcrPattern(stem: string): RegExp { return new RegExp(`^${escRe(stem)}-p\\d+(?:\\.words\\.json|\\.md)$`); }
40
40
 
41
41
  const FILE_LINK_RE = /\[[^\]]*\]\(\s*((?:sheets|attachments)\/[^)\s]+)\s*\)/g;
42
42
  const IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
@@ -94,6 +94,7 @@ export function openBundle(root: string, requested: string, overwrite: boolean):
94
94
  if (existsSync(pagesDir)) for (const f of readdirSync(pagesDir)) if (ownedPage.test(f)) rmSync(join(pagesDir, f), { force: true });
95
95
  if (existsSync(attachmentsDir)) for (const f of readdirSync(attachmentsDir)) if (attachments.has(f)) rmSync(join(attachmentsDir, f), { force: true });
96
96
  rmSync(join(root, `${stem}.pages.json`), { force: true });
97
+ rmSync(join(root, `${stem}.words.json`), { force: true });
97
98
  const ownedOcr = ownedOcrPattern(stem);
98
99
  if (existsSync(ocrDir)) for (const f of readdirSync(ocrDir)) if (ownedOcr.test(f)) rmSync(join(ocrDir, f), { force: true });
99
100
  }
@@ -104,13 +105,16 @@ export function openBundle(root: string, requested: string, overwrite: boolean):
104
105
  const pagesStagingDir = join(pagesDir, `.stage-${lockId}`);
105
106
  const attachmentsStagingDir = join(attachmentsDir, `.stage-${lockId}`);
106
107
  const ocrStagingDir = join(ocrDir, `.stage-${lockId}`);
107
- return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join(root, `${stem}.pages.json`), ocrManifest: new Set(), manifest: new Set(), csvManifest: new Set(), pageManifest: new Set(), attachmentManifest: new Set(), sourceMap: new Map() };
108
+ return { root, stem, renamedFrom, renameReason, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, pagesDir, pagesStagingDir, attachmentsDir, attachmentsStagingDir, ocrDir, ocrStagingDir, pageStatsPath: join(root, `${stem}.pages.json`), wordsPath: join(root, `${stem}.words.json`), ocrManifest: new Set(), manifest: new Set(), csvManifest: new Set(), pageManifest: new Set(), attachmentManifest: new Set(), sourceMap: new Map() };
108
109
  } catch (e) { rmSync(lockPath, { force: true }); throw e; }
109
110
  }
110
111
 
111
- /** Publish every `p<N>/` staging dir carrying `.done`; discard partial ones. Returns page -> published filenames. */
112
- export function publishStaged(b: Bundle): Map<number, string[]> {
113
- const out = new Map<number, string[]>();
112
+ export interface DoneMeta { native?: { file: string; width: number; height: number }; dpi?: number; requestedDpi?: number }
113
+ export interface StagedPage { files: string[]; meta: DoneMeta }
114
+
115
+ /** Publish completed pages, retaining metadata with native filenames renamed; discard partial pages. */
116
+ export function publishStaged(b: Bundle): Map<number, StagedPage> {
117
+ const out = new Map<number, StagedPage>();
114
118
  if (!existsSync(b.stagingDir)) return out;
115
119
  for (const dir of readdirSync(b.stagingDir).sort()) {
116
120
  const m = dir.match(/^p(\d+)$/);
@@ -118,6 +122,12 @@ export function publishStaged(b: Bundle): Map<number, string[]> {
118
122
  const pageDir = join(b.stagingDir, dir);
119
123
  if (!existsSync(join(pageDir, ".done"))) { rmSync(pageDir, { recursive: true, force: true }); continue; }
120
124
  const page = Number(m[1]);
125
+ const raw = readFileSync(join(pageDir, ".done"), "utf8").trim();
126
+ let meta: DoneMeta = {};
127
+ try {
128
+ const v = raw ? JSON.parse(raw) : {};
129
+ if (v && typeof v === "object" && !Array.isArray(v)) meta = v;
130
+ } catch { /* Completed images survive truncated metadata. */ }
121
131
  const files = readdirSync(pageDir).filter((f) => f !== ".done" && statSync(join(pageDir, f)).isFile()).sort();
122
132
  const names: string[] = [];
123
133
  files.forEach((f, i) => {
@@ -127,8 +137,9 @@ export function publishStaged(b: Bundle): Map<number, string[]> {
127
137
  b.sourceMap.set(`${dir}/${f}`, `images/${name}`);
128
138
  names.push(name);
129
139
  });
140
+ if (meta.native) meta.native = { ...meta.native, file: names[files.indexOf(meta.native.file)] ?? meta.native.file };
130
141
  rmSync(pageDir, { recursive: true, force: true });
131
- out.set(page, names);
142
+ out.set(page, { files: names, meta });
132
143
  }
133
144
  return out;
134
145
  }
@@ -174,10 +185,23 @@ export function writePageStats(b: Bundle, stats: PageStat[]): void {
174
185
  writeFileSync(b.pageStatsPath, `${JSON.stringify(stats, null, 2)}\n`, "utf8");
175
186
  }
176
187
 
177
- /** Move every `.done`-gated `pNNN/<stem>-pNNN.md` into `ocr/`; partial page dirs and the checkpoint are dropped with the staging dir. Returns page -> absolute sidecar path. */
178
- export function publishSidecars(b: Bundle): Map<number, string> {
179
- const out = new Map<number, string>();
180
- if (!existsSync(b.ocrStagingDir)) return out;
188
+ /** Rename within the bundle root keeps the complete words document atomic. */
189
+ export function publishWords(b: Bundle): string | null {
190
+ const staged = join(b.stagingDir, "words.json");
191
+ try {
192
+ if (!existsSync(staged)) throw new Error("child staged no words.json");
193
+ renameSync(staged, b.wordsPath);
194
+ return null;
195
+ } catch (e) {
196
+ rmSync(b.wordsPath, { force: true });
197
+ return `write failed - ${(e as Error).message}`;
198
+ }
199
+ }
200
+
201
+ /** Move `.done`-gated Markdown and words sidecars into `ocr/`; drop partial dirs and the checkpoint. Returns page -> absolute paths for each kind. */
202
+ export function publishSidecars(b: Bundle): { sidecars: Map<number, string>; wordSidecars: Map<number, string> } {
203
+ const sidecars = new Map<number, string>(), wordSidecars = new Map<number, string>();
204
+ if (!existsSync(b.ocrStagingDir)) return { sidecars, wordSidecars };
181
205
  for (const dir of readdirSync(b.ocrStagingDir).sort()) {
182
206
  const m = dir.match(/^p(\d+)$/);
183
207
  if (!m) continue;
@@ -188,10 +212,16 @@ export function publishSidecars(b: Bundle): Map<number, string> {
188
212
  mkdirSync(b.ocrDir, { recursive: true });
189
213
  renameSync(join(pageDir, file), join(b.ocrDir, file));
190
214
  b.ocrManifest.add(file);
191
- out.set(Number(m[1]), join(b.ocrDir, file));
215
+ sidecars.set(Number(m[1]), join(b.ocrDir, file));
216
+ const wfile = `${b.stem}-${dir}.words.json`;
217
+ if (existsSync(join(pageDir, wfile))) {
218
+ renameSync(join(pageDir, wfile), join(b.ocrDir, wfile));
219
+ b.ocrManifest.add(wfile);
220
+ wordSidecars.set(Number(m[1]), join(b.ocrDir, wfile));
221
+ }
192
222
  }
193
223
  rmSync(b.ocrStagingDir, { recursive: true, force: true });
194
- return out;
224
+ return { sidecars, wordSidecars };
195
225
  }
196
226
 
197
227
  export function publishAttachments(b: Bundle): void {
@@ -254,6 +284,7 @@ export function abortBundle(b: Bundle): void {
254
284
  for (const f of b.ocrManifest) rmSync(join(b.ocrDir, f), { force: true });
255
285
  rmSync(b.ocrStagingDir, { recursive: true, force: true });
256
286
  rmSync(b.pageStatsPath, { force: true });
287
+ rmSync(b.wordsPath, { force: true });
257
288
  rmSync(`${b.mdPath}.tmp`, { force: true });
258
289
  rmSync(b.stagingDir, { recursive: true, force: true });
259
290
  rmSync(b.sheetsStagingDir, { recursive: true, force: true });