pi-quiver 5.5.0 → 6.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7 -0
- package/README.md +8 -9
- package/dist/bin/pi-quiver.js +121 -56
- package/extensions/doc_to_md.ts +1 -1
- package/lib/doc-to-md-bundle.ts +46 -25
- package/lib/doc-to-md-core.ts +75 -38
- package/lib/doc-to-md-handle.ts +14 -3
- package/lib/doc-to-md-options.ts +3 -5
- package/package.json +1 -1
- package/scripts/doc_to_md.py +272 -88
package/CHANGELOG.md
CHANGED
|
@@ -8,6 +8,13 @@ Published to npm as `pi-quiver` (`pi install npm:pi-quiver`). Pushing a
|
|
|
8
8
|
via OIDC trusted publishing. The release helper at
|
|
9
9
|
`.agents/skills/release/scripts/release.sh` cuts the tag; CI publishes.
|
|
10
10
|
|
|
11
|
+
## v6.0.0 - 2026-09-10
|
|
12
|
+
|
|
13
|
+
- doc_to_md (Excel, breaking): `maxCellsPerSheet` is removed from the tool schema, CLI (`--max-cells-per-sheet`), and `quiver.docToMd` settings; a leftover key is reported by the settings lint. Sheet indices are now 0-based workbook positions covering worksheets and chartsheets, so embedded-image files move from `<stem>-s1-<n>.*` to `<stem>-s0-<n>.*` and `SheetInfo.index` is rebased.
|
|
14
|
+
- doc_to_md (Excel): the Markdown opens with a `## Sheets` inventory (kind, size, hidden, chart/image counts, rendered view, CSV link); every non-empty worksheet's full content is exported to `sheets/<stem>-s<idx>-<slug>.csv`; the in-Markdown preview is capped at 100 rows x 50 columns and, when truncated, followed by a per-column profile (type, non-empty, min/max, distinct). Chartsheets and embedded charts are listed with type, title, and series refs.
|
|
15
|
+
- doc_to_md (Excel): sheets carrying charts or images get a rendered view (`images/<stem>-s<idx>.<fmt>`) when LibreOffice is on `PATH` - one PDF page per sheet via `calc_pdf_Export` `SinglePageSheets`, rasterized by a new PyMuPDF `render-pages` child under a 16 Mpx budget. Any soffice or render failure degrades to `Rendered view: unavailable (<reason>)` plus a handle note; the conversion never fails. `.xls` keeps inventory/CSV/preview but has no visual detection.
|
|
16
|
+
- doc_to_md: handle prints `Sheets-Dir` when CSVs were written; `info` lists chartsheets with kind and chart/image counts. Bundle overwrite now removes only this stem's owned-pattern files in `images/` and `sheets/`.
|
|
17
|
+
|
|
11
18
|
## v5.5.0 - 2026-09-10
|
|
12
19
|
|
|
13
20
|
- doc_to_md: rewrite primary-PDF image links correctly on Windows.
|
package/README.md
CHANGED
|
@@ -65,7 +65,7 @@ A 300 KB changelog page never touches your context window - you get a preview an
|
|
|
65
65
|
| Extension | Tool | What it does |
|
|
66
66
|
| --- | --- | --- |
|
|
67
67
|
| `extensions/fetch.ts` | `fetch` | Retrieve URLs over HTTP(S). HTML -> Markdown (Readability extraction, Turndown conversion). Binary saved untouched to a temp file. GitHub issue/PR/repo/actions-run/actions-job URLs auto-route through `gh` (falls back to HTTP); failed runs/jobs include failed-step logs (best-effort, summary-only otherwise). Same size gate as `fetch`. Behavior lives in `lib/fetch-core.ts`; also exposed as the `pi-quiver fetch` CLI (see [Claude Code support](#claude-code-support)). |
|
|
68
|
-
| `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/PPTX/XLSX/XLS to a Markdown bundle on disk (`<stem>.md` + `images/`) and return a handle (paths, page count, outline, diagnostics) - never inline Markdown. `info` mode inspects first; `pages` selects 1-based pages; every page ends with `--- end of page.page_number=N ---`. Tiers: pymupdf4llm -> PyMuPDF text (degraded) -> unpdf worker (no Python only). Excel ->
|
|
68
|
+
| `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/PPTX/XLSX/XLS to a Markdown bundle on disk (`<stem>.md` + `images/` and spreadsheet `sheets/`) and return a handle (paths, page count, outline, diagnostics) - never inline Markdown. `info` mode inspects first; `pages` selects 1-based pages; every page ends with `--- end of page.page_number=N ---`. Tiers: pymupdf4llm -> PyMuPDF text (degraded) -> unpdf worker (no Python only). Excel -> sheet inventory, full CSVs, bounded previews, and optional rendered views. Settings under `quiver.docToMd`. |
|
|
69
69
|
| `extensions/session-name.ts` | `/session-name` | Manual + opt-in automatic session naming, naming rules and deny list, long-session revisits, and Ghostty/Herdr tab rename. OFF by default. |
|
|
70
70
|
| `extensions/sword-header.ts` | `/builtin-header` | Themed ASCII startup header replacing pi's default logo. OFF by default. |
|
|
71
71
|
| `extensions/fast-mode.ts` | `/fast` | Inject Anthropic fast-mode (`speed: "fast"` + `anthropic-beta: fast-mode-2026-02-01`) into every Claude Opus 4.8 / Opus 5 request, any thinking level. `--fast` flag + `/fast [on\|off\|status]`. OFF by default. |
|
|
@@ -93,7 +93,7 @@ Full routing rules, size-gate mechanics, and config: [doc/fetch.md](doc/fetch.md
|
|
|
93
93
|
## When NOT to use
|
|
94
94
|
|
|
95
95
|
- You need a general-purpose web scraper (JS-rendered pages, pagination, auth flows) - `fetch` does plain HTTP + Readability extraction, nothing more.
|
|
96
|
-
- You need `.xlsm`,
|
|
96
|
+
- You need in-grid chart/image placement, `.xls` visual detection, `.xlsm`, or range selection - out of scope; chart/image association is per sheet only.
|
|
97
97
|
- You want automatic session naming, a custom header, fast mode, or stall recovery without opting in - all stay off until you flip the config.
|
|
98
98
|
- You need *mid-stream* stall recovery in JSON, RPC, or print runs - only the pre-first-event tier arms there; mid-stream silence falls through to pi's transport timeout.
|
|
99
99
|
|
|
@@ -289,22 +289,21 @@ Each setting can also be overridden per-process via `PI_QUIVER_SLACK_ENABLED`, `
|
|
|
289
289
|
| Key | Default | Meaning |
|
|
290
290
|
|---|---|---|
|
|
291
291
|
| `primaryTimeoutMs` | `60000` | pymupdf4llm tier and unpdf tier deadline. |
|
|
292
|
-
| `fallbackTimeoutMs` | `30000` | PyMuPDF text tier
|
|
293
|
-
| `sofficeTimeoutMs` | `120000` | DOCX/PPTX -> PDF deadline. |
|
|
292
|
+
| `fallbackTimeoutMs` | `30000` | PyMuPDF text tier, PDF info, and Excel rendered-view rasterization deadline. |
|
|
293
|
+
| `sofficeTimeoutMs` | `120000` | DOCX/PPTX -> PDF deadline; also the Excel rendered-view export. |
|
|
294
294
|
| `excelTimeoutMs` | `60000` | Excel child and Excel info deadline. |
|
|
295
295
|
| `warmTimeoutMs` | `120000` | Absolute first-call backend discovery/bootstrap deadline. |
|
|
296
296
|
| `pymupdfVersion` | `1.27.2.3` | pymupdf4llm pin, minimum `1.27.0`. |
|
|
297
|
-
| `imageDpi` | `150` | Render DPI for page images. |
|
|
297
|
+
| `imageDpi` | `150` | Render DPI for page images and Excel rendered views (capped by a 16 Mpx budget). |
|
|
298
298
|
| `imageFormat` | `png` | Rendered image format: `png` or `jpg`; embedded images retain their extension. |
|
|
299
|
-
| `maxCellsPerSheet` | `50000` | Rows x columns budget per worksheet. |
|
|
300
299
|
| `maxOutputBytes` | `20000000` | Child stdout cap in bytes. |
|
|
301
300
|
| `outlineMaxEntries` | `40` | Heading outline, TOC, or sheet inventory entries in the handle. |
|
|
302
301
|
|
|
303
|
-
A bundle is `<outputDir>/<stem>.md` plus `<outputDir>/images/`; without `outputDir`, the tool creates a per-call temp root. A conversion owns `<stem>.md.lock` until it atomically publishes the Markdown. An existing `<stem>.md` fails the call unless `overwrite` is set; `overwrite` replaces that Markdown and the
|
|
302
|
+
A bundle is `<outputDir>/<stem>.md` plus `<outputDir>/images/` and, for spreadsheets with data, `<outputDir>/sheets/`; without `outputDir`, the tool creates a per-call temp root. A conversion owns `<stem>.md.lock` until it atomically publishes the Markdown. An existing `<stem>.md` fails the call unless `overwrite` is set; `overwrite` replaces that Markdown and the files it owns (`images/<stem>-p<N>-<n>.*`, `images/<stem>-s<idx>[-<n>].*`, `sheets/<stem>-s<idx>-<slug>.csv`), nothing else. Temp bundles are caller-owned - the tool never deletes a bundle it produced.
|
|
304
303
|
|
|
305
|
-
Excel needs a Python backend with openpyxl, xlrd and pillow - otherwise the call fails with `Remedy: install uv, or pip install openpyxl xlrd pillow`.
|
|
304
|
+
Excel needs a Python backend with openpyxl, xlrd and pillow - otherwise the call fails with `Remedy: install uv, or pip install openpyxl xlrd pillow`. The Markdown opens with a `## Sheets` table listing every sheet in workbook order (0-based `#`, `worksheet`/`chartsheet`, size, hidden, chart and image counts, rendered view, CSV link for non-empty worksheets), then one section per sheet: a `Data:` line linking the full-content CSV under `sheets/` for non-empty worksheets, chart metadata from the workbook model, embedded images, an optional rendered view, a preview of at most the first 100 rows x 50 columns, and - only when the preview is truncated - a `Columns:` profile (type, non-empty count, min/max, distinct up to 50). Sizes are the extent of non-empty cells (the `info` handle reports the raw worksheet dimensions instead, which may be larger). Rendered views (`images/<stem>-s<idx>.<fmt>`) are produced for sheets carrying charts or images when LibreOffice is on `PATH`: the workbook is exported one PDF page per sheet and rasterized under a 16 Mpx budget. Any LibreOffice or rasterization failure degrades to `Rendered view: unavailable (<reason>)` and a handle note; it never fails the conversion. Workbooks whose chartsheet drawings carry a zero-size anchor (openpyxl-authored files; Excel-authored files are unaffected) render as a degenerate page and are reported as such. `.xls` gets the inventory, CSVs and previews but no visual detection or rendering.
|
|
306
305
|
|
|
307
|
-
Worst-case wall time is `warmTimeoutMs (first call) + sofficeTimeoutMs (Office only) + primaryTimeoutMs + fallbackTimeoutMs + KILL_GRACE_MS x kills` (Excel: `warmTimeoutMs + excelTimeoutMs + KILL_GRACE_MS`). There is no cap on image count, image bytes or workbook memory - deliberately; the per-tier timeouts and `maxOutputBytes` are the bounds.
|
|
306
|
+
Worst-case wall time is `warmTimeoutMs (first call) + sofficeTimeoutMs (Office only) + primaryTimeoutMs + fallbackTimeoutMs + KILL_GRACE_MS x kills` (Excel: `warmTimeoutMs + excelTimeoutMs + sofficeTimeoutMs + fallbackTimeoutMs + 2 * KILL_GRACE_MS`). There is no cap on image count, image bytes, cell count or workbook memory - deliberately; the per-tier timeouts, the rendered-view pixel budget and `maxOutputBytes` are the bounds.
|
|
308
307
|
|
|
309
308
|
### Migrating from flat keys
|
|
310
309
|
|
package/dist/bin/pi-quiver.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
3
|
// bin/pi-quiver.ts
|
|
4
|
-
import { existsSync as existsSync3, readFileSync
|
|
4
|
+
import { existsSync as existsSync3, readFileSync, realpathSync } from "node:fs";
|
|
5
5
|
import { homedir as homedir2 } from "node:os";
|
|
6
6
|
import { join as join4, resolve as resolve3 } from "node:path";
|
|
7
7
|
import { fileURLToPath as fileURLToPath2 } from "node:url";
|
|
@@ -524,30 +524,24 @@ import { basename, dirname, extname as extname3, join as join3, resolve as resol
|
|
|
524
524
|
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
525
525
|
|
|
526
526
|
// lib/doc-to-md-bundle.ts
|
|
527
|
-
import fs, { closeSync, existsSync, mkdirSync as mkdirSync2, openSync, readdirSync,
|
|
527
|
+
import fs, { closeSync, existsSync, mkdirSync as mkdirSync2, openSync, readdirSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync as writeFileSync2 } from "node:fs";
|
|
528
528
|
import { tmpdir as tmpdir2 } from "node:os";
|
|
529
529
|
import { extname, join as join2, resolve } from "node:path";
|
|
530
530
|
import { randomBytes } from "node:crypto";
|
|
531
|
+
var escRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
531
532
|
function ownedPattern(stem) {
|
|
532
|
-
|
|
533
|
-
return new RegExp(`^${esc}-(p|s)\\d+-\\d+\\.[a-z0-9]+$`);
|
|
533
|
+
return new RegExp(`^${escRe(stem)}-(p|s)\\d+(-\\d+)?\\.[a-z0-9]+$`);
|
|
534
534
|
}
|
|
535
|
+
function ownedCsvPattern(stem) {
|
|
536
|
+
return new RegExp(`^${escRe(stem)}-s\\d+-[a-z0-9-]+\\.csv$`);
|
|
537
|
+
}
|
|
538
|
+
var SHEET_LINK_RE = /\[[^\]]*\]\(\s*(sheets\/[^)\s]+)\s*\)/g;
|
|
535
539
|
var IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
|
|
536
540
|
function imageTarget(m) {
|
|
537
541
|
const target = (m[1] ?? m[2] ?? m[3] ?? m[4] ?? m[5] ?? "").trim();
|
|
538
542
|
const destination = target.replace(/^<([\s\S]*)>$/, "$1").trim();
|
|
539
543
|
return m[2] === void 0 ? destination : destination.replace(/\s+(?:"[^"]*"|'[^']*'|\([^)]*\))$/, "");
|
|
540
544
|
}
|
|
541
|
-
function linkedFiles(md) {
|
|
542
|
-
const out = [];
|
|
543
|
-
for (const m of md.matchAll(IMG_LINK_RE)) {
|
|
544
|
-
const target = imageTarget(m);
|
|
545
|
-
if (!target.startsWith("images/")) continue;
|
|
546
|
-
const file = target.slice("images/".length);
|
|
547
|
-
if (file && file !== "." && file !== ".." && /^[^/\\]+$/.test(file)) out.push(file);
|
|
548
|
-
}
|
|
549
|
-
return out;
|
|
550
|
-
}
|
|
551
545
|
function openBundle(root, stem, overwrite) {
|
|
552
546
|
mkdirSync2(root, { recursive: true });
|
|
553
547
|
const mdPath = join2(root, `${stem}.md`);
|
|
@@ -561,20 +555,24 @@ function openBundle(root, stem, overwrite) {
|
|
|
561
555
|
}
|
|
562
556
|
closeSync(fd);
|
|
563
557
|
const imagesDir = join2(root, "images");
|
|
558
|
+
const sheetsDir = join2(root, "sheets");
|
|
564
559
|
try {
|
|
565
560
|
if (existsSync(mdPath)) {
|
|
566
561
|
if (!overwrite) throw new Error(`Output exists: ${mdPath} (pass overwrite)`);
|
|
567
|
-
const owned = ownedPattern(stem);
|
|
568
|
-
const linked = new Set(linkedFiles(readFileSync(mdPath, "utf8")));
|
|
562
|
+
const owned = ownedPattern(stem), ownedCsv = ownedCsvPattern(stem);
|
|
569
563
|
unlinkSync(mdPath);
|
|
570
564
|
if (existsSync(imagesDir)) {
|
|
571
|
-
for (const f of readdirSync(imagesDir)) if (owned.test(f)
|
|
565
|
+
for (const f of readdirSync(imagesDir)) if (owned.test(f)) rmSync(join2(imagesDir, f), { force: true });
|
|
566
|
+
}
|
|
567
|
+
if (existsSync(sheetsDir)) {
|
|
568
|
+
for (const f of readdirSync(sheetsDir)) if (ownedCsv.test(f)) rmSync(join2(sheetsDir, f), { force: true });
|
|
572
569
|
}
|
|
573
570
|
}
|
|
574
571
|
const lockId = randomBytes(6).toString("hex");
|
|
575
572
|
const stagingDir = join2(imagesDir, `.stage-${lockId}`);
|
|
573
|
+
const sheetsStagingDir = join2(sheetsDir, `.stage-${lockId}`);
|
|
576
574
|
mkdirSync2(stagingDir, { recursive: true });
|
|
577
|
-
return { root, stem, mdPath, lockPath, imagesDir, stagingDir, lockId, manifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
|
|
575
|
+
return { root, stem, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, manifest: /* @__PURE__ */ new Set(), csvManifest: /* @__PURE__ */ new Set(), sourceMap: /* @__PURE__ */ new Map() };
|
|
578
576
|
} catch (e) {
|
|
579
577
|
rmSync(lockPath, { force: true });
|
|
580
578
|
throw e;
|
|
@@ -608,25 +606,42 @@ function publishStaged(b) {
|
|
|
608
606
|
}
|
|
609
607
|
function publishSheetImages(b) {
|
|
610
608
|
for (const f of readdirSync(b.stagingDir).sort()) {
|
|
611
|
-
if (!/^s\d
|
|
609
|
+
if (!/^s\d+(-\d+)?\.[a-z0-9]+$/i.test(f)) continue;
|
|
612
610
|
const name = `${b.stem}-${f.toLowerCase()}`;
|
|
613
611
|
renameSync(join2(b.stagingDir, f), join2(b.imagesDir, name));
|
|
614
612
|
b.manifest.add(name);
|
|
615
613
|
b.sourceMap.set(f, `images/${name}`);
|
|
616
614
|
}
|
|
617
615
|
}
|
|
618
|
-
function
|
|
619
|
-
|
|
616
|
+
function publishSheetCsvs(b) {
|
|
617
|
+
if (!existsSync(b.sheetsStagingDir)) return;
|
|
618
|
+
for (const f of readdirSync(b.sheetsStagingDir).sort()) {
|
|
619
|
+
if (!/^s\d+-[a-z0-9-]+\.csv$/.test(f)) continue;
|
|
620
|
+
const name = `${b.stem}-${f}`;
|
|
621
|
+
renameSync(join2(b.sheetsStagingDir, f), join2(b.sheetsDir, name));
|
|
622
|
+
b.csvManifest.add(name);
|
|
623
|
+
b.sourceMap.set(`sheets/${f}`, `sheets/${name}`);
|
|
624
|
+
}
|
|
625
|
+
}
|
|
626
|
+
function rewriteLinks(md, sourceMap) {
|
|
627
|
+
const images = md.replace(IMG_LINK_RE, (whole, mdAngle, mdPlain, htmlDouble, htmlSingle, htmlUnquoted) => {
|
|
620
628
|
const target = imageTarget([whole, mdAngle, mdPlain, htmlDouble, htmlSingle, htmlUnquoted]);
|
|
621
629
|
const dest = sourceMap.get(target) ?? sourceMap.get(target.replace(/^\.\//, ""));
|
|
622
630
|
return dest ? whole.replace(mdAngle === void 0 ? target : `<${mdAngle}>`, dest) : whole;
|
|
623
631
|
});
|
|
632
|
+
return images.replace(SHEET_LINK_RE, (whole, target) => {
|
|
633
|
+
const dest = sourceMap.get(target);
|
|
634
|
+
return dest ? whole.split(target).join(dest) : whole;
|
|
635
|
+
});
|
|
624
636
|
}
|
|
625
|
-
function validateImageLinks(md, manifest) {
|
|
637
|
+
function validateImageLinks(md, manifest, csvManifest = /* @__PURE__ */ new Set()) {
|
|
626
638
|
for (const m of md.matchAll(IMG_LINK_RE)) {
|
|
627
639
|
const target = imageTarget(m);
|
|
628
640
|
if (!target.startsWith("images/") || !manifest.has(target.slice("images/".length))) throw new Error(`unexpected image reference in output: ${target}`);
|
|
629
641
|
}
|
|
642
|
+
for (const m of md.matchAll(SHEET_LINK_RE)) {
|
|
643
|
+
if (!csvManifest.has(m[1].slice("sheets/".length))) throw new Error(`unexpected sheet reference in output: ${m[1]}`);
|
|
644
|
+
}
|
|
630
645
|
}
|
|
631
646
|
function commitBundle(b, markdown) {
|
|
632
647
|
const tmp = `${b.mdPath}.tmp`;
|
|
@@ -636,6 +651,10 @@ function commitBundle(b, markdown) {
|
|
|
636
651
|
fs.rmSync(b.stagingDir, { recursive: true, force: true });
|
|
637
652
|
} catch {
|
|
638
653
|
}
|
|
654
|
+
try {
|
|
655
|
+
fs.rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|
|
656
|
+
} catch {
|
|
657
|
+
}
|
|
639
658
|
try {
|
|
640
659
|
fs.rmSync(b.lockPath, { force: true });
|
|
641
660
|
} catch {
|
|
@@ -643,8 +662,10 @@ function commitBundle(b, markdown) {
|
|
|
643
662
|
}
|
|
644
663
|
function abortBundle(b) {
|
|
645
664
|
for (const f of b.manifest) rmSync(join2(b.imagesDir, f), { force: true });
|
|
665
|
+
for (const f of b.csvManifest) rmSync(join2(b.sheetsDir, f), { force: true });
|
|
646
666
|
rmSync(`${b.mdPath}.tmp`, { force: true });
|
|
647
667
|
rmSync(b.stagingDir, { recursive: true, force: true });
|
|
668
|
+
rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|
|
648
669
|
rmSync(b.lockPath, { force: true });
|
|
649
670
|
}
|
|
650
671
|
function tempBundleRoot() {
|
|
@@ -703,6 +724,7 @@ function outlineLines(entries, total) {
|
|
|
703
724
|
function formatHandle(h) {
|
|
704
725
|
const lines = [`Saved-To: ${h.savedTo}`];
|
|
705
726
|
if (h.imagesDir && h.imageCount > 0) lines.push(`Images-Dir: ${h.imagesDir}`);
|
|
727
|
+
if (h.sheetsDir) lines.push(`Sheets-Dir: ${h.sheetsDir}`);
|
|
706
728
|
lines.push(`Type: ${h.type} Engine: ${h.engine} Tier: ${h.tier}`);
|
|
707
729
|
lines.push(`Page-Count: ${h.pageCount ?? "?"} Pages: ${h.pages ? compactRanges(h.pages) : "all"} Images: ${h.imageCount} Size: ${formatSize2(h.bytes)} / ${h.lines} lines`);
|
|
708
730
|
if (h.degraded) lines.push(`Degraded: ${h.degraded}`);
|
|
@@ -719,7 +741,7 @@ function formatInfoHandle(i, max) {
|
|
|
719
741
|
if (i.sheets) {
|
|
720
742
|
const lines2 = [`Type: ${i.type} Sheets: ${i.sheetsTotal}`];
|
|
721
743
|
for (const s of i.sheets.slice(0, max)) {
|
|
722
|
-
const dims =
|
|
744
|
+
const dims = `${s.kind} rows=${s.rows ?? "-"} cols=${s.cols ?? "-"} charts=${s.charts} images=${s.images}`;
|
|
723
745
|
const hidden = s.hiddenRows || s.hiddenCols ? ` hiddenRows=${s.hiddenRows} hiddenCols=${s.hiddenCols}` : "";
|
|
724
746
|
lines2.push(` ${trunc(s.name, TITLE_MAX)} ${s.hidden ? "hidden " : ""}${dims}${hidden}`);
|
|
725
747
|
}
|
|
@@ -749,14 +771,13 @@ var DOC_TO_MD_OPTIONS = [
|
|
|
749
771
|
{ key: "outputDir", flag: "--output-dir", type: "string", default: null, settable: false, help: "Bundle root for <stem>.md + images/; default a per-call temp dir" },
|
|
750
772
|
{ key: "overwrite", flag: "--overwrite", type: "bool", default: false, settable: false, help: "Replace an existing completed <stem>.md bundle" },
|
|
751
773
|
{ key: "primaryTimeoutMs", flag: "--primary-timeout", type: "int", default: 6e4, settable: true, env: "PI_DOC_TO_MD_CONVERT_TIMEOUT_MS", help: "pymupdf4llm tier; also the unpdf tier" },
|
|
752
|
-
{ key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier; also info on PDF" },
|
|
753
|
-
{ key: "sofficeTimeoutMs", flag: "--soffice-timeout", type: "int", default: 12e4, settable: true, env: "PI_DOC_TO_MD_SOFFICE_TIMEOUT_MS", help: "DOCX/PPTX -> PDF via LibreOffice" },
|
|
774
|
+
{ key: "fallbackTimeoutMs", flag: "--fallback-timeout", type: "int", default: 3e4, settable: true, help: "PyMuPDF get_text tier; also info on PDF and Excel rendered views" },
|
|
775
|
+
{ key: "sofficeTimeoutMs", flag: "--soffice-timeout", type: "int", default: 12e4, settable: true, env: "PI_DOC_TO_MD_SOFFICE_TIMEOUT_MS", help: "DOCX/PPTX -> PDF via LibreOffice; also Excel rendered views" },
|
|
754
776
|
{ key: "excelTimeoutMs", flag: "--excel-timeout", type: "int", default: 6e4, settable: true, help: "Excel child (both openpyxl loads); also info on Excel" },
|
|
755
777
|
{ key: "warmTimeoutMs", flag: "--warm-timeout", type: "int", default: 12e4, settable: true, env: "PI_DOC_TO_MD_WARM_TIMEOUT_MS", help: "Absolute backend discovery/bootstrap deadline (first call per process)" },
|
|
756
778
|
{ key: "pymupdfVersion", flag: "--pymupdf-version", type: "version", default: "1.27.2.3", settable: true, env: "PI_DOC_TO_MD_PYMUPDF_VERSION", help: "pymupdf4llm pin (>= 1.27.0)" },
|
|
757
|
-
{ key: "imageDpi", flag: "--image-dpi", type: "int", default: 150, settable: true, help: "Render DPI for page images" },
|
|
779
|
+
{ key: "imageDpi", flag: "--image-dpi", type: "int", default: 150, settable: true, help: "Render DPI for page images and Excel rendered views" },
|
|
758
780
|
{ key: "imageFormat", flag: "--image-format", type: "enum", default: "png", settable: true, enumValues: ["png", "jpg"], help: "Rendered image format (embedded images keep their native extension)" },
|
|
759
|
-
{ key: "maxCellsPerSheet", flag: "--max-cells-per-sheet", type: "int", default: 5e4, settable: true, help: "rows x cols budget per worksheet" },
|
|
760
781
|
{ key: "maxOutputBytes", flag: "--max-output-bytes", type: "int", default: 2e7, settable: true, help: "Child stdout cap in bytes" },
|
|
761
782
|
{ key: "outlineMaxEntries", flag: "--outline-max-entries", type: "int", default: 40, settable: true, help: "Heading outline / TOC / sheet inventory cap in the handle" }
|
|
762
783
|
];
|
|
@@ -877,6 +898,7 @@ var VENV_DIR_NAME = "doc-to-md-venv-v2";
|
|
|
877
898
|
var LEGACY_VENV_DIR_NAME = "pymupdf-venv";
|
|
878
899
|
var STDERR_CAP = 1e6;
|
|
879
900
|
var OUTPUT_MAX_BYTES = 2e7;
|
|
901
|
+
var EXCEL_PDF_FILTER = 'pdf:calc_pdf_Export:{"SinglePageSheets":{"type":"boolean","value":"true"}}';
|
|
880
902
|
function withArgs(cfg) {
|
|
881
903
|
return ["--with", `pymupdf4llm==${cfg.pymupdfVersion}`, "--with", `openpyxl==${PACKAGE_PINS.openpyxl}`, "--with", `xlrd==${PACKAGE_PINS.xlrd}`, "--with", `pillow==${PACKAGE_PINS.pillow}`];
|
|
882
904
|
}
|
|
@@ -892,7 +914,7 @@ function uvChildArgs(cfg, script, mode) {
|
|
|
892
914
|
function pythonChildArgs(script, mode) {
|
|
893
915
|
return [script, mode];
|
|
894
916
|
}
|
|
895
|
-
function soffArgs(src, profileDir, outDir) {
|
|
917
|
+
function soffArgs(src, profileDir, outDir, filter = "pdf") {
|
|
896
918
|
return [
|
|
897
919
|
"--headless",
|
|
898
920
|
"--invisible",
|
|
@@ -905,7 +927,7 @@ function soffArgs(src, profileDir, outDir) {
|
|
|
905
927
|
"--quickstart=no",
|
|
906
928
|
`-env:UserInstallation=${pathToFileURL(profileDir).href}`,
|
|
907
929
|
"--convert-to",
|
|
908
|
-
|
|
930
|
+
filter,
|
|
909
931
|
"--outdir",
|
|
910
932
|
outDir,
|
|
911
933
|
src
|
|
@@ -1162,37 +1184,42 @@ function getBackend(cfg, deps, signal) {
|
|
|
1162
1184
|
}
|
|
1163
1185
|
return backendPromise;
|
|
1164
1186
|
}
|
|
1165
|
-
async function
|
|
1187
|
+
async function tryConvertOffice(sofficeTimeoutMs, src, signal, run = runCapped, filter = "pdf") {
|
|
1166
1188
|
const profileDir = mkdtempSync(join3(tmpdir3(), "pi-doc-soffice-prof-"));
|
|
1167
1189
|
const outDir = mkdtempSync(join3(tmpdir3(), "pi-doc-soffice-out-"));
|
|
1168
1190
|
const cleanup = () => {
|
|
1169
|
-
for (const d of [profileDir, outDir]) {
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
} catch {
|
|
1173
|
-
}
|
|
1191
|
+
for (const d of [profileDir, outDir]) try {
|
|
1192
|
+
rmSync2(d, { recursive: true, force: true });
|
|
1193
|
+
} catch {
|
|
1174
1194
|
}
|
|
1175
1195
|
};
|
|
1176
1196
|
try {
|
|
1177
1197
|
const env = { ...process.env, SAL_USE_VCLPLUGIN: "svp", OOO_DISABLE_RECOVERY: "1", SAL_NO_MOUSEGRABS: "1" };
|
|
1178
|
-
const r = await run("soffice", soffArgs(src, profileDir, outDir), { timeoutMs: sofficeTimeoutMs, capBytes: OUTPUT_MAX_BYTES, env, signal });
|
|
1198
|
+
const r = await run("soffice", soffArgs(src, profileDir, outDir, filter), { timeoutMs: sofficeTimeoutMs, capBytes: OUTPUT_MAX_BYTES, env, signal });
|
|
1179
1199
|
if (signal?.aborted) throw new Error("aborted");
|
|
1180
|
-
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1200
|
+
const fail = (kind) => {
|
|
1201
|
+
cleanup();
|
|
1202
|
+
return { ok: false, kind, code: r.code, timedOut: r.timedOut, stderr: r.stderr.slice(0, 500) };
|
|
1203
|
+
};
|
|
1204
|
+
if (r.code === null && !r.timedOut) return fail("missing");
|
|
1205
|
+
if (r.timedOut) return fail("timeout");
|
|
1206
|
+
if (r.code !== 0) return fail("exit");
|
|
1207
|
+
const pdfPath = join3(outDir, `${basename(src).replace(/\.[^.]+$/, "")}.pdf`);
|
|
1186
1208
|
const st = statSync2(pdfPath, { throwIfNoEntry: false });
|
|
1187
|
-
if (!st || !st.isFile() || st.size === 0)
|
|
1188
|
-
|
|
1189
|
-
}
|
|
1190
|
-
return { pdfPath, cleanup };
|
|
1209
|
+
if (!st || !st.isFile() || st.size === 0) return fail("no-pdf");
|
|
1210
|
+
return { ok: true, pdfPath, cleanup };
|
|
1191
1211
|
} catch (e) {
|
|
1192
1212
|
cleanup();
|
|
1193
1213
|
throw e;
|
|
1194
1214
|
}
|
|
1195
1215
|
}
|
|
1216
|
+
async function convertOffice(sofficeTimeoutMs, src, signal, run = runCapped) {
|
|
1217
|
+
const r = await tryConvertOffice(sofficeTimeoutMs, src, signal, run);
|
|
1218
|
+
if (r.ok) return r;
|
|
1219
|
+
if (r.kind === "missing") throw new Error("LibreOffice (soffice) is required to convert .docx/.pptx but was not found on PATH. Install LibreOffice or convert the file to PDF first.");
|
|
1220
|
+
if (r.kind === "no-pdf") throw new Error("LibreOffice (soffice) ran but produced no usable PDF for this file. Ensure LibreOffice can open the document, or convert it to PDF manually first.");
|
|
1221
|
+
throw new Error(`soffice failed (code=${r.code} timedOut=${r.timedOut}): ${r.stderr}`);
|
|
1222
|
+
}
|
|
1196
1223
|
var DEGRADED_TEXT = "PyMuPDF text extraction - layout/tables not preserved";
|
|
1197
1224
|
var DEGRADED_UNPDF = "unpdf text extraction - structure not preserved";
|
|
1198
1225
|
var EXCEL_REMEDY = "Remedy: install uv, or pip install openpyxl xlrd pillow";
|
|
@@ -1239,8 +1266,18 @@ async function runTierReal(mode, childOptions, _b, signal, timeoutMs, backend) {
|
|
|
1239
1266
|
function detailSuffix(r) {
|
|
1240
1267
|
return r.detail ? ` (${r.detail})` : "";
|
|
1241
1268
|
}
|
|
1269
|
+
function reconcileRenderMarkers(md, renderPages, fmt, sourceMap, reason) {
|
|
1270
|
+
for (const idx of renderPages) {
|
|
1271
|
+
const file = `s${idx}.${fmt}`;
|
|
1272
|
+
const ok = sourceMap.has(file);
|
|
1273
|
+
md = md.replace(`<!--rv:${idx}-->`, ok ? `Rendered view: ` : `Rendered view: unavailable (${reason(idx)})`);
|
|
1274
|
+
md = md.replace(`<!--rvs:${idx}-->`, ok ? "yes" : "no");
|
|
1275
|
+
}
|
|
1276
|
+
if (/<!--rvs?:\d+-->/.test(md)) throw new Error("internal: unresolved render marker");
|
|
1277
|
+
return md;
|
|
1278
|
+
}
|
|
1242
1279
|
async function convertDocument(o, signal, seams) {
|
|
1243
|
-
const s = { backend: (c) => getBackend(c, void 0, signal), runTier: runTierReal, ...seams };
|
|
1280
|
+
const s = { backend: (c) => getBackend(c, void 0, signal), runTier: runTierReal, office: tryConvertOffice, ...seams };
|
|
1244
1281
|
const inputPath = resolve2(o.path);
|
|
1245
1282
|
const st = statSync2(inputPath, { throwIfNoEntry: false });
|
|
1246
1283
|
if (!st || !st.isFile()) throw new Error(`Not a readable file: ${o.path}`);
|
|
@@ -1258,19 +1295,46 @@ async function convertDocument(o, signal, seams) {
|
|
|
1258
1295
|
office = await convertOffice(o.sofficeTimeoutMs, inputPath, signal);
|
|
1259
1296
|
pdfPath = office.pdfPath;
|
|
1260
1297
|
}
|
|
1261
|
-
const base = { path: pdfPath, pages: o.pages, stagingDir: b.stagingDir,
|
|
1298
|
+
const base = { path: pdfPath, pages: o.pages, stagingDir: b.stagingDir, sheetsStagingDir: b.sheetsStagingDir, imageDpi: o.imageDpi, imageFormat: o.imageFormat, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion };
|
|
1262
1299
|
let tier, engine, json, degraded = null, fallbackReason = null;
|
|
1300
|
+
let notes = [];
|
|
1263
1301
|
if (isExcel) {
|
|
1264
1302
|
const r = await s.runTier("xlsx", base, b, signal, o.excelTimeoutMs, backend);
|
|
1265
1303
|
if (!r.ok) {
|
|
1266
1304
|
if ("userError" in r) throw new Error(r.userError);
|
|
1267
|
-
const remedy = r.reason.startsWith("timeout after") || r.reason === "output exceeded maxOutputBytes" ? ". Remedy: raise excelTimeoutMs
|
|
1305
|
+
const remedy = r.reason.startsWith("timeout after") || r.reason === "output exceeded maxOutputBytes" ? ". Remedy: raise excelTimeoutMs" : "";
|
|
1268
1306
|
throw new Error(`Excel conversion failed: ${r.reason}${detailSuffix(r)}${remedy}`);
|
|
1269
1307
|
}
|
|
1270
1308
|
publishSheetImages(b);
|
|
1309
|
+
publishSheetCsvs(b);
|
|
1271
1310
|
tier = "excel";
|
|
1272
1311
|
engine = type === "xls" ? "xlrd" : "openpyxl";
|
|
1273
1312
|
json = r.json;
|
|
1313
|
+
notes = [...json.notes ?? []];
|
|
1314
|
+
const renderPages = json.renderPages ?? [];
|
|
1315
|
+
let skip = null;
|
|
1316
|
+
const perSheet = /* @__PURE__ */ new Map();
|
|
1317
|
+
if (renderPages.length) {
|
|
1318
|
+
const off = await s.office(o.sofficeTimeoutMs, inputPath, signal, runCapped, EXCEL_PDF_FILTER);
|
|
1319
|
+
if (!off.ok) skip = off.kind === "missing" ? "LibreOffice not found" : off.kind === "timeout" ? `soffice failed: timeout after ${o.sofficeTimeoutMs}ms` : off.kind === "exit" ? `soffice failed: exit ${off.code}` : "soffice produced no PDF";
|
|
1320
|
+
else {
|
|
1321
|
+
try {
|
|
1322
|
+
const rp = await s.runTier("render-pages", { path: off.pdfPath, sheetIndices: renderPages, expectedPages: json.sheetCount, imageDpi: o.imageDpi, imageFormat: o.imageFormat, stagingDir: b.stagingDir, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, b, signal, o.fallbackTimeoutMs, backend);
|
|
1323
|
+
if (!rp.ok) skip = `render failed: ${"userError" in rp ? rp.userError : rp.reason}`;
|
|
1324
|
+
else if (rp.json.ok === false) skip = rp.json.reason ?? "render failed";
|
|
1325
|
+
else {
|
|
1326
|
+
publishSheetImages(b);
|
|
1327
|
+
for (const f of rp.json.failed ?? []) perSheet.set(f.idx, f.reason);
|
|
1328
|
+
for (const d of rp.json.rendered ?? []) if (d.dpi < o.imageDpi) notes.push(`Rendered view s${d.idx}: rendered at ${d.dpi} dpi`);
|
|
1329
|
+
}
|
|
1330
|
+
} finally {
|
|
1331
|
+
off.cleanup();
|
|
1332
|
+
}
|
|
1333
|
+
}
|
|
1334
|
+
if (skip) notes.push(`Rendered views skipped: ${skip}`);
|
|
1335
|
+
else if (perSheet.size) notes.push(`Rendered views: ${perSheet.size} of ${renderPages.length} unavailable`);
|
|
1336
|
+
}
|
|
1337
|
+
json = { ...json, markdown: reconcileRenderMarkers(json.markdown ?? "", renderPages, o.imageFormat, b.sourceMap, (idx) => perSheet.get(idx) ?? skip ?? "render failed") };
|
|
1274
1338
|
} else if (backend.kind === "none") {
|
|
1275
1339
|
const r = await s.runTier("pdf-text", base, b, signal, o.primaryTimeoutMs, backend);
|
|
1276
1340
|
if (!r.ok) throw new Error("userError" in r ? r.userError : `Conversion failed: unpdf ${r.reason}${detailSuffix(r)}`);
|
|
@@ -1299,20 +1363,21 @@ async function convertDocument(o, signal, seams) {
|
|
|
1299
1363
|
fallbackReason = `primary ${p.reason}`;
|
|
1300
1364
|
}
|
|
1301
1365
|
}
|
|
1302
|
-
|
|
1303
|
-
|
|
1366
|
+
if (!isExcel) notes = json.notes ?? [];
|
|
1367
|
+
const body = rewriteLinks(json.markdown ?? "", b.sourceMap);
|
|
1368
|
+
validateImageLinks(body, b.manifest, b.csvManifest);
|
|
1304
1369
|
const head = [];
|
|
1305
1370
|
if (degraded) head.push(`Degraded: ${degraded}`);
|
|
1306
1371
|
if (fallbackReason) head.push(`Fallback-Reason: ${fallbackReason}`);
|
|
1307
1372
|
if (json.failedPages?.length) head.push(`Failed pages: ${json.failedPages.map((f) => `${f.page} (${f.error})`).join("; ")}`);
|
|
1308
1373
|
if (json.emptyPages?.length) head.push(`Empty pages: ${json.emptyPages.join(", ")}`);
|
|
1309
|
-
for (const n of
|
|
1374
|
+
for (const n of notes) head.push(`Notes: ${n}`);
|
|
1310
1375
|
const markdown = (head.length ? `${head.join("\n")}
|
|
1311
1376
|
|
|
1312
1377
|
` : "") + body;
|
|
1313
1378
|
commitBundle(b, markdown);
|
|
1314
1379
|
const outline = scanOutline(markdown, o.outlineMaxEntries);
|
|
1315
|
-
const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes
|
|
1380
|
+
const details = { path: inputPath, backend: backend.kind, pymupdfVersion: o.pymupdfVersion, inputType: type, file: b.mdPath, outputDir: b.root, savedTo: b.mdPath, imagesDir: b.imagesDir, sheetsDir: b.csvManifest.size ? b.sheetsDir : null, type, engine, tier, pageCount: json.pageCount ?? null, pages: o.pages, imageCount: b.manifest.size, bytes: Buffer.byteLength(markdown, "utf8"), lines: markdown.split("\n").length, degraded, fallbackReason, failedPages: (json.failedPages ?? []).map((f) => f.page), emptyPages: json.emptyPages ?? [], notes, outline: outline.entries, outlineTotal: outline.total };
|
|
1316
1381
|
return { output: formatHandle(details), details };
|
|
1317
1382
|
} catch (e) {
|
|
1318
1383
|
abortBundle(b);
|
|
@@ -1322,7 +1387,7 @@ async function convertDocument(o, signal, seams) {
|
|
|
1322
1387
|
}
|
|
1323
1388
|
}
|
|
1324
1389
|
async function inspectDocument(o, signal, seams) {
|
|
1325
|
-
const s = { backend: (c) => getBackend(c, void 0, signal), runTier: runTierReal, ...seams };
|
|
1390
|
+
const s = { backend: (c) => getBackend(c, void 0, signal), runTier: runTierReal, office: tryConvertOffice, ...seams };
|
|
1326
1391
|
const inputPath = resolve2(o.path);
|
|
1327
1392
|
const st = statSync2(inputPath, { throwIfNoEntry: false });
|
|
1328
1393
|
if (!st || !st.isFile()) throw new Error(`Not a readable file: ${o.path}`);
|
|
@@ -1340,7 +1405,7 @@ async function inspectDocument(o, signal, seams) {
|
|
|
1340
1405
|
const r = await s.runTier("info", { path, maxOutputBytes: o.maxOutputBytes, pymupdfVersion: o.pymupdfVersion }, { stagingDir: "" }, signal, isExcel ? o.excelTimeoutMs : o.fallbackTimeoutMs, backend);
|
|
1341
1406
|
if (!r.ok) {
|
|
1342
1407
|
if ("userError" in r) throw new Error(r.userError);
|
|
1343
|
-
if (isExcel && (r.reason.startsWith("timeout after") || r.reason === "output exceeded maxOutputBytes")) throw new Error(`Excel inspection failed: ${r.reason}. Remedy: raise excelTimeoutMs
|
|
1408
|
+
if (isExcel && (r.reason.startsWith("timeout after") || r.reason === "output exceeded maxOutputBytes")) throw new Error(`Excel inspection failed: ${r.reason}. Remedy: raise excelTimeoutMs${detailSuffix(r)}`);
|
|
1344
1409
|
throw new Error(`Inspection failed: ${r.reason}${detailSuffix(r)}`);
|
|
1345
1410
|
}
|
|
1346
1411
|
const toc = (r.json.toc ?? []).map(([level, title, page]) => ({ level, title, page }));
|
|
@@ -1397,7 +1462,7 @@ function readCliSettings(cwd, env, warn) {
|
|
|
1397
1462
|
if (!existsSync3(file)) continue;
|
|
1398
1463
|
let raw;
|
|
1399
1464
|
try {
|
|
1400
|
-
raw = JSON.parse(
|
|
1465
|
+
raw = JSON.parse(readFileSync(file, "utf8")).quiver?.docToMd;
|
|
1401
1466
|
} catch {
|
|
1402
1467
|
warn(`pi-quiver: ${file} is not valid JSON; ignored.`);
|
|
1403
1468
|
continue;
|
package/extensions/doc_to_md.ts
CHANGED
|
@@ -35,7 +35,7 @@ export default function docToMdExtension(pi: ExtensionAPI) {
|
|
|
35
35
|
name: "doc_to_md",
|
|
36
36
|
label: "Convert doc to Markdown bundle",
|
|
37
37
|
description:
|
|
38
|
-
"Convert a local PDF/DOCX/PPTX/XLSX/XLS to a Markdown bundle on disk and return a handle (Saved-To, Images-Dir, Page-Count, Outline, diagnostics) - the Markdown itself is never inlined; read the Saved-To file (offset/limit) for content. `info: true` returns page count, metadata and TOC (or the sheet inventory) without converting - use it to pick `pages`. `pages` selects inclusive 1-based pages (PDF/DOCX/PPTX); every page ends with `--- end of page.page_number=N ---`. Images are always extracted into images/ with relative links. Primary engine pymupdf4llm, fallback PyMuPDF text (degraded, marked), pure-JS unpdf only when no Python backend exists. Excel yields per-worksheet
|
|
38
|
+
"Convert a local PDF/DOCX/PPTX/XLSX/XLS to a Markdown bundle on disk and return a handle (Saved-To, Images-Dir, Page-Count, Outline, diagnostics) - the Markdown itself is never inlined; read the Saved-To file (offset/limit) for content. `info: true` returns page count, metadata and TOC (or the sheet inventory) without converting - use it to pick `pages`. `pages` selects inclusive 1-based pages (PDF/DOCX/PPTX); every page ends with `--- end of page.page_number=N ---`. Images are always extracted into images/ with relative links. Primary engine pymupdf4llm, fallback PyMuPDF text (degraded, marked), pure-JS unpdf only when no Python backend exists. Excel yields a sheet inventory (worksheets + chartsheets), a full CSV per non-empty worksheet under sheets/, a bounded preview with formulas and cached values, merged/hidden disclosure, and rendered chart views when LibreOffice is available. DOCX/PPTX need LibreOffice (soffice). Use `outputDir` for a durable bundle; without it the bundle lands in a per-call temp dir. Input must be a local file path (use fetch first for URLs).",
|
|
39
39
|
promptSnippet: "Convert a local PDF/DOCX/PPTX/XLSX to a Markdown bundle (handle returned; read Saved-To)",
|
|
40
40
|
parameters: buildSchema(),
|
|
41
41
|
|
package/lib/doc-to-md-bundle.ts
CHANGED
|
@@ -1,25 +1,33 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Bundle protocol: a call owns `<stem>` for its whole duration via `<stem>.md.lock`;
|
|
3
|
-
* children stage images under `images/.stage-<lockId>/p<N>/` and
|
|
4
|
-
* Node publishes
|
|
3
|
+
* children stage images under `images/.stage-<lockId>/p<N>/` and CSVs under `sheets/.stage-<lockId>/s<idx>-<slug>.csv`;
|
|
4
|
+
* Node publishes them to `images/<stem>-p<N>-<n>.<ext>` and `sheets/<stem>-s<idx>-<slug>.csv`, and records every file it
|
|
5
5
|
* wrote in a manifest, and commits `<stem>.md` atomically (tmp + rename).
|
|
6
6
|
*/
|
|
7
|
-
import fs, { closeSync, existsSync, mkdirSync, openSync, readdirSync,
|
|
7
|
+
import fs, { closeSync, existsSync, mkdirSync, openSync, readdirSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync } from "node:fs";
|
|
8
8
|
import { tmpdir } from "node:os";
|
|
9
9
|
import { extname, join, resolve } from "node:path";
|
|
10
10
|
import { randomBytes } from "node:crypto";
|
|
11
11
|
|
|
12
12
|
export interface Bundle {
|
|
13
13
|
root: string; stem: string; mdPath: string; lockPath: string; imagesDir: string; stagingDir: string; lockId: string;
|
|
14
|
+
sheetsDir: string; sheetsStagingDir: string;
|
|
14
15
|
manifest: Set<string>;
|
|
16
|
+
csvManifest: Set<string>;
|
|
15
17
|
sourceMap: Map<string, string>;
|
|
16
18
|
}
|
|
17
19
|
|
|
20
|
+
const escRe = (s: string) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
21
|
+
|
|
18
22
|
export function ownedPattern(stem: string): RegExp {
|
|
19
|
-
|
|
20
|
-
|
|
23
|
+
return new RegExp(`^${escRe(stem)}-(p|s)\\d+(-\\d+)?\\.[a-z0-9]+$`);
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export function ownedCsvPattern(stem: string): RegExp {
|
|
27
|
+
return new RegExp(`^${escRe(stem)}-s\\d+-[a-z0-9-]+\\.csv$`);
|
|
21
28
|
}
|
|
22
29
|
|
|
30
|
+
const SHEET_LINK_RE = /\[[^\]]*\]\(\s*(sheets\/[^)\s]+)\s*\)/g;
|
|
23
31
|
const IMG_LINK_RE = /!\[[^\]]*\]\(\s*(?:<([^>]*)>|([^)]*?))\s*\)|<img\b[^>]*\bsrc\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>"']+))/gi;
|
|
24
32
|
|
|
25
33
|
function imageTarget(m: RegExpMatchArray): string {
|
|
@@ -28,17 +36,6 @@ function imageTarget(m: RegExpMatchArray): string {
|
|
|
28
36
|
return m[2] === undefined ? destination : destination.replace(/\s+(?:"[^"]*"|'[^']*'|\([^)]*\))$/, "");
|
|
29
37
|
}
|
|
30
38
|
|
|
31
|
-
function linkedFiles(md: string): string[] {
|
|
32
|
-
const out: string[] = [];
|
|
33
|
-
for (const m of md.matchAll(IMG_LINK_RE)) {
|
|
34
|
-
const target = imageTarget(m);
|
|
35
|
-
if (!target.startsWith("images/")) continue;
|
|
36
|
-
const file = target.slice("images/".length);
|
|
37
|
-
if (file && file !== "." && file !== ".." && /^[^/\\]+$/.test(file)) out.push(file);
|
|
38
|
-
}
|
|
39
|
-
return out;
|
|
40
|
-
}
|
|
41
|
-
|
|
42
39
|
export function openBundle(root: string, stem: string, overwrite: boolean): Bundle {
|
|
43
40
|
mkdirSync(root, { recursive: true });
|
|
44
41
|
const mdPath = join(root, `${stem}.md`);
|
|
@@ -51,18 +48,20 @@ export function openBundle(root: string, stem: string, overwrite: boolean): Bund
|
|
|
51
48
|
}
|
|
52
49
|
closeSync(fd);
|
|
53
50
|
const imagesDir = join(root, "images");
|
|
51
|
+
const sheetsDir = join(root, "sheets");
|
|
54
52
|
try {
|
|
55
53
|
if (existsSync(mdPath)) {
|
|
56
54
|
if (!overwrite) throw new Error(`Output exists: ${mdPath} (pass overwrite)`);
|
|
57
|
-
const owned = ownedPattern(stem);
|
|
58
|
-
const linked = new Set(linkedFiles(readFileSync(mdPath, "utf8")));
|
|
55
|
+
const owned = ownedPattern(stem), ownedCsv = ownedCsvPattern(stem);
|
|
59
56
|
unlinkSync(mdPath);
|
|
60
|
-
if (existsSync(imagesDir)) for (const f of readdirSync(imagesDir)) if (owned.test(f)
|
|
57
|
+
if (existsSync(imagesDir)) for (const f of readdirSync(imagesDir)) if (owned.test(f)) rmSync(join(imagesDir, f), { force: true });
|
|
58
|
+
if (existsSync(sheetsDir)) for (const f of readdirSync(sheetsDir)) if (ownedCsv.test(f)) rmSync(join(sheetsDir, f), { force: true });
|
|
61
59
|
}
|
|
62
60
|
const lockId = randomBytes(6).toString("hex");
|
|
63
61
|
const stagingDir = join(imagesDir, `.stage-${lockId}`);
|
|
62
|
+
const sheetsStagingDir = join(sheetsDir, `.stage-${lockId}`);
|
|
64
63
|
mkdirSync(stagingDir, { recursive: true });
|
|
65
|
-
return { root, stem, mdPath, lockPath, imagesDir, stagingDir, lockId, manifest: new Set(), sourceMap: new Map() };
|
|
64
|
+
return { root, stem, mdPath, lockPath, imagesDir, stagingDir, lockId, sheetsDir, sheetsStagingDir, manifest: new Set(), csvManifest: new Set(), sourceMap: new Map() };
|
|
66
65
|
} catch (e) { rmSync(lockPath, { force: true }); throw e; }
|
|
67
66
|
}
|
|
68
67
|
|
|
@@ -91,10 +90,10 @@ export function publishStaged(b: Bundle): Map<number, string[]> {
|
|
|
91
90
|
return out;
|
|
92
91
|
}
|
|
93
92
|
|
|
94
|
-
/** Excel:
|
|
93
|
+
/** Excel: `s<idx>-<n>.<ext>` (embedded) and `s<idx>.<fmt>` (rendered view) staged flat -> `images/<stem>-<file>`. */
|
|
95
94
|
export function publishSheetImages(b: Bundle): void {
|
|
96
95
|
for (const f of readdirSync(b.stagingDir).sort()) {
|
|
97
|
-
if (!/^s\d
|
|
96
|
+
if (!/^s\d+(-\d+)?\.[a-z0-9]+$/i.test(f)) continue;
|
|
98
97
|
const name = `${b.stem}-${f.toLowerCase()}`;
|
|
99
98
|
renameSync(join(b.stagingDir, f), join(b.imagesDir, name));
|
|
100
99
|
b.manifest.add(name);
|
|
@@ -102,19 +101,38 @@ export function publishSheetImages(b: Bundle): void {
|
|
|
102
101
|
}
|
|
103
102
|
}
|
|
104
103
|
|
|
105
|
-
|
|
106
|
-
|
|
104
|
+
/** Excel: `sheetsStagingDir/s<idx>-<slug>.csv` -> `sheets/<stem>-s<idx>-<slug>.csv`; `sheets/` is created only when a CSV exists. */
|
|
105
|
+
export function publishSheetCsvs(b: Bundle): void {
|
|
106
|
+
if (!existsSync(b.sheetsStagingDir)) return;
|
|
107
|
+
for (const f of readdirSync(b.sheetsStagingDir).sort()) {
|
|
108
|
+
if (!/^s\d+-[a-z0-9-]+\.csv$/.test(f)) continue;
|
|
109
|
+
const name = `${b.stem}-${f}`;
|
|
110
|
+
renameSync(join(b.sheetsStagingDir, f), join(b.sheetsDir, name));
|
|
111
|
+
b.csvManifest.add(name);
|
|
112
|
+
b.sourceMap.set(`sheets/${f}`, `sheets/${name}`);
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
export function rewriteLinks(md: string, sourceMap: Map<string, string>): string {
|
|
117
|
+
const images = md.replace(IMG_LINK_RE, (whole, mdAngle, mdPlain, htmlDouble, htmlSingle, htmlUnquoted) => {
|
|
107
118
|
const target = imageTarget([whole, mdAngle, mdPlain, htmlDouble, htmlSingle, htmlUnquoted] as unknown as RegExpMatchArray);
|
|
108
119
|
const dest = sourceMap.get(target) ?? sourceMap.get(target.replace(/^\.\//, ""));
|
|
109
120
|
return dest ? whole.replace(mdAngle === undefined ? target : `<${mdAngle}>`, dest) : whole;
|
|
110
121
|
});
|
|
122
|
+
return images.replace(SHEET_LINK_RE, (whole, target: string) => {
|
|
123
|
+
const dest = sourceMap.get(target);
|
|
124
|
+
return dest ? whole.split(target).join(dest) : whole;
|
|
125
|
+
});
|
|
111
126
|
}
|
|
112
127
|
|
|
113
|
-
export function validateImageLinks(md: string, manifest: Set<string>): void {
|
|
128
|
+
export function validateImageLinks(md: string, manifest: Set<string>, csvManifest: Set<string> = new Set()): void {
|
|
114
129
|
for (const m of md.matchAll(IMG_LINK_RE)) {
|
|
115
130
|
const target = imageTarget(m);
|
|
116
131
|
if (!target.startsWith("images/") || !manifest.has(target.slice("images/".length))) throw new Error(`unexpected image reference in output: ${target}`);
|
|
117
132
|
}
|
|
133
|
+
for (const m of md.matchAll(SHEET_LINK_RE)) {
|
|
134
|
+
if (!csvManifest.has(m[1].slice("sheets/".length))) throw new Error(`unexpected sheet reference in output: ${m[1]}`);
|
|
135
|
+
}
|
|
118
136
|
}
|
|
119
137
|
|
|
120
138
|
export function commitBundle(b: Bundle, markdown: string): void {
|
|
@@ -122,13 +140,16 @@ export function commitBundle(b: Bundle, markdown: string): void {
|
|
|
122
140
|
writeFileSync(tmp, markdown, "utf8");
|
|
123
141
|
renameSync(tmp, b.mdPath);
|
|
124
142
|
try { fs.rmSync(b.stagingDir, { recursive: true, force: true }); } catch { /* Markdown is published; cleanup is best-effort. */ }
|
|
143
|
+
try { fs.rmSync(b.sheetsStagingDir, { recursive: true, force: true }); } catch { /* best-effort */ }
|
|
125
144
|
try { fs.rmSync(b.lockPath, { force: true }); } catch { /* Markdown is published; cleanup is best-effort. */ }
|
|
126
145
|
}
|
|
127
146
|
|
|
128
147
|
export function abortBundle(b: Bundle): void {
|
|
129
148
|
for (const f of b.manifest) rmSync(join(b.imagesDir, f), { force: true });
|
|
149
|
+
for (const f of b.csvManifest) rmSync(join(b.sheetsDir, f), { force: true });
|
|
130
150
|
rmSync(`${b.mdPath}.tmp`, { force: true });
|
|
131
151
|
rmSync(b.stagingDir, { recursive: true, force: true });
|
|
152
|
+
rmSync(b.sheetsStagingDir, { recursive: true, force: true });
|
|
132
153
|
rmSync(b.lockPath, { force: true });
|
|
133
154
|
}
|
|
134
155
|
|