documents.js 1.102.2 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/README.md +42 -33
  2. package/dist/codecs/registry.cjs +7 -0
  3. package/dist/codecs/registry.js +7 -0
  4. package/dist/convert/capability.cjs +5 -0
  5. package/dist/convert/capability.js +5 -0
  6. package/dist/convert/codec.cjs +20 -0
  7. package/dist/convert/codec.d.cts +5 -1
  8. package/dist/convert/codec.d.ts +5 -1
  9. package/dist/convert/codec.js +19 -3
  10. package/dist/convert/composition.cjs +47 -11
  11. package/dist/convert/composition.d.cts +9 -5
  12. package/dist/convert/composition.d.ts +9 -5
  13. package/dist/convert/composition.js +47 -11
  14. package/dist/convert/convert.cjs +38 -3
  15. package/dist/convert/convert.d.cts +18 -1
  16. package/dist/convert/convert.d.ts +18 -1
  17. package/dist/convert/convert.js +31 -4
  18. package/dist/convert/from-package.cjs +179 -2
  19. package/dist/convert/from-package.d.cts +3 -2
  20. package/dist/convert/from-package.d.ts +3 -2
  21. package/dist/convert/from-package.js +179 -3
  22. package/dist/convert/local.cjs +2 -0
  23. package/dist/convert/local.js +2 -0
  24. package/dist/convert/port.cjs +1 -0
  25. package/dist/convert/port.d.cts +3 -0
  26. package/dist/convert/port.d.ts +3 -0
  27. package/dist/convert/port.js +1 -0
  28. package/dist/csv/read.cjs +85 -0
  29. package/dist/csv/read.d.cts +10 -0
  30. package/dist/csv/read.d.ts +10 -0
  31. package/dist/csv/read.js +84 -0
  32. package/dist/csv/records.cjs +85 -0
  33. package/dist/csv/records.d.cts +10 -0
  34. package/dist/csv/records.d.ts +10 -0
  35. package/dist/csv/records.js +80 -0
  36. package/dist/csv/text.cjs +22 -0
  37. package/dist/csv/text.d.cts +8 -0
  38. package/dist/csv/text.d.ts +8 -0
  39. package/dist/csv/text.js +19 -0
  40. package/dist/csv/write.cjs +75 -0
  41. package/dist/csv/write.d.cts +22 -0
  42. package/dist/csv/write.d.ts +22 -0
  43. package/dist/csv/write.js +71 -0
  44. package/dist/index.cjs +28 -1
  45. package/dist/index.d.cts +11 -7
  46. package/dist/index.d.ts +11 -7
  47. package/dist/index.js +10 -6
  48. package/dist/layout/drawing.cjs +71 -9
  49. package/dist/layout/drawing.d.cts +8 -3
  50. package/dist/layout/drawing.d.ts +8 -3
  51. package/dist/layout/drawing.js +71 -10
  52. package/dist/layout/engine.cjs +55 -43
  53. package/dist/layout/engine.d.cts +2 -1
  54. package/dist/layout/engine.d.ts +2 -1
  55. package/dist/layout/engine.js +57 -45
  56. package/dist/layout/reconstruct.cjs +260 -105
  57. package/dist/layout/reconstruct.js +260 -105
  58. package/dist/layout/shared.cjs +47 -2
  59. package/dist/layout/shared.d.cts +15 -4
  60. package/dist/layout/shared.d.ts +15 -4
  61. package/dist/layout/shared.js +44 -4
  62. package/dist/layout/sheets.cjs +18 -13
  63. package/dist/layout/sheets.d.cts +4 -2
  64. package/dist/layout/sheets.d.ts +4 -2
  65. package/dist/layout/sheets.js +20 -16
  66. package/dist/layout/slides.cjs +35 -31
  67. package/dist/layout/slides.d.cts +3 -2
  68. package/dist/layout/slides.d.ts +3 -2
  69. package/dist/layout/slides.js +37 -33
  70. package/dist/layout/text-layout.cjs +2 -1
  71. package/dist/layout/text-layout.d.cts +14 -3
  72. package/dist/layout/text-layout.d.ts +14 -3
  73. package/dist/layout/text-layout.js +2 -1
  74. package/dist/metadata/write.cjs +1 -0
  75. package/dist/metadata/write.js +1 -0
  76. package/dist/model/bytes.cjs +2 -0
  77. package/dist/model/bytes.d.cts +2 -1
  78. package/dist/model/bytes.d.ts +2 -1
  79. package/dist/model/bytes.js +2 -1
  80. package/dist/odb/csv.cjs +3 -6
  81. package/dist/odb/csv.js +3 -6
  82. package/package.json +5 -5
package/README.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  [![GitHub](https://img.shields.io/badge/GitHub-181717?logo=github&logoColor=white)](https://github.com/ExaDev/documents.js) [![npm](https://img.shields.io/badge/npm-CB3837?logo=npm&logoColor=white)](https://www.npmjs.com/package/documents.js) [![Release](https://img.shields.io/github/v/release/ExaDev/documents.js)](https://github.com/ExaDev/documents.js/releases/latest) [![CI](https://img.shields.io/github/actions/workflow/status/ExaDev/documents.js/ci.yml?branch=main)](https://github.com/ExaDev/documents.js/actions)
4
4
 
5
- > Converts between any two compatible document formats through a shared content/layout pivot. docx, pptx, odt, odp, ods, odg, xlsx, and markdown all read into and build from the same `ContentDocument`/`LayoutDocument` model, with PDF as the one format every variant can reach. A composition engine (`convertDocument`) routes 73 (source, target) pairs across the eight content formats and PDF, including fourteen PDF-pivot round trips, sixteen cross-format bridges (same-variant direct copies, cross-variant semantic transforms, and PDF-composed), plus special-case conversions for `.odm` master documents, `.odb` database front-ends (HSQLDB and Firebird, four storage tiers), standalone `.odf` formula documents, and a bounded SQL/rpt-formula engine for `.odb` reports. Also includes: read-and-write live-view editors for all six editable formats, docx comment/footnote/header-footer exposure via `readDocxExtras`, real font resolution (source-embedded faces ahead of caller-supplied, vendored substitutes, and the standard 14), a hand-written MathML typesetting engine with embedded-font PDF rendering and a matching MathML ⇄ OMML translator, and a fully hand-written PDF codec. Built on [ooxml.js](https://github.com/ExaDev/ooxml.js), [odf.js](https://github.com/ExaDev/odf.js), [pdf-codec](https://github.com/ExaDev/pdf-codec), [markdown-codec](https://github.com/ExaDev/markdown-codec), and [document-schema.js](https://github.com/ExaDev/document-schema.js).
5
+ > Converts between any two compatible document formats through a shared content/layout pivot. docx, pptx, odt, odp, ods, odg, xlsx, csv (TSV is the same format with a tab delimiter), and markdown all read into and build from the same `ContentDocument`/`LayoutDocument` model, with PDF as the one format every variant can reach. A composition engine (`convertDocument`) routes 91 (source, target) pairs across the nine content formats and PDF, including eighteen PDF-pivot round trips (the seven layout-engine formats, plus xlsx and csv composing through ods), twenty-two cross-format bridge functions (same-variant direct copies, cross-variant semantic transforms, and PDF-composed), plus special-case conversions for `.odm` master documents, `.odb` database front-ends (HSQLDB and Firebird, four storage tiers), standalone `.odf` formula documents, and a bounded SQL/rpt-formula engine for `.odb` reports. Also includes: read-and-write live-view editors for all six editable formats, docx comment/footnote/header-footer exposure via `readDocxExtras`, real font resolution (source-embedded faces ahead of caller-supplied, vendored substitutes, and the standard 14), a hand-written MathML typesetting engine with embedded-font PDF rendering and a matching MathML ⇄ OMML translator, and a fully hand-written PDF codec. Built on [ooxml.js](https://github.com/ExaDev/ooxml.js), [odf.js](https://github.com/ExaDev/odf.js), [pdf-codec](https://github.com/ExaDev/pdf-codec), [markdown-codec](https://github.com/ExaDev/markdown-codec), and [document-schema.js](https://github.com/ExaDev/document-schema.js).
6
6
 
7
7
  `documents.js` extends `ooxml.js` in two directions `ooxml.js` deliberately does not cover: full PDF support (parsing and generating, via `pdf-codec`), and a read-**and-write** manipulation API for docx/pptx content — `ooxml.js`'s own typed readers are one-way. The PDF codec is hand-written against ISO 32000-1, with no external PDF library as a dependency — see [Fidelity](#fidelity) and pdf-codec's own README for the honest trade-off (not as robust against adversarial PDFs as a 15+-year-hardened library; fully auditable and dependency-free instead). `src/mathml/` (the MathML typesetting engine) stays in this package and is hand-written too, for the same supply-chain reason.
8
8
 
@@ -72,7 +72,7 @@ npm install documents.js
72
72
 
73
73
  ### The generic entry point: `convertDocument`
74
74
 
75
- A single function, `convertDocument`, sits behind every named conversion and reaches every pair the composition engine can route — all 73 supported (source, target) combinations. The named functions below are thin one-line forwarders to it; they remain the ergonomic layer for a caller who wants a fixed pair and autocomplete discovery, while `convertDocument` is the first-class entry point for a caller working from a runtime format pair (CLI, MCP tool, matrix enumeration).
75
+ A single function, `convertDocument`, sits behind every named conversion and reaches every pair the composition engine can route — all 91 supported (source, target) combinations. The named functions below are thin one-line forwarders to it; they remain the ergonomic layer for a caller who wants a fixed pair and autocomplete discovery, while `convertDocument` is the first-class entry point for a caller working from a runtime format pair (CLI, MCP tool, matrix enumeration).
76
76
 
77
77
  ```ts
78
78
  import { convertDocument } from 'documents.js';
@@ -89,10 +89,10 @@ const odtBytes = convertDocument('docx', 'odt', docxBytes, { onMathDiagnostic: (
89
89
 
90
90
  ### PDF-pivot conversions
91
91
 
92
- The fourteen round-trip ergonomic conversions between the formats with their own layout engine and PDF (docx/pptx/odt/odp/ods/odg/markdown ⇄ PDF, all round-tripping both ways), plus `xlsxToPdf`/`pdfToXlsx` (composing the ods⇄xlsx bridge with the ods⇄pdf layout pair internally):
92
+ The fourteen round-trip ergonomic conversions between the formats with their own layout engine and PDF (docx/pptx/odt/odp/ods/odg/markdown ⇄ PDF, all round-tripping both ways), plus `xlsxToPdf`/`pdfToXlsx` and `csvToPdf`/`pdfToCsv` (each composing its ods bridge with the ods⇄pdf layout pair internally — neither xlsx nor csv has a layout engine of its own):
93
93
 
94
94
  ```ts
95
- import { docxToPdf, markdownToPdf, odgToPdf, odpToPdf, odsToPdf, odtToPdf, pdfToDocx, pdfToMarkdown, pdfToOdg, pdfToOdp, pdfToOds, pdfToOdt, pdfToPptx, pdfToXlsx, pptxToPdf, xlsxToPdf } from 'documents.js';
95
+ import { csvToPdf, docxToPdf, markdownToPdf, odgToPdf, odpToPdf, odsToPdf, odtToPdf, pdfToCsv, pdfToDocx, pdfToMarkdown, pdfToOdg, pdfToOdp, pdfToOds, pdfToOdt, pdfToPptx, pdfToXlsx, pptxToPdf, xlsxToPdf } from 'documents.js';
96
96
 
97
97
  const pdfBytes = docxToPdf(docxBytes);
98
98
  const docxBytes2 = pdfToDocx(pdfBytes);
@@ -117,13 +117,16 @@ const xlsxBytes2 = pdfToXlsx(pdfFromXlsx); // composes pdfToOds -> odsToXlsx int
117
117
 
118
118
  const pdfFromMarkdown = markdownToPdf(markdownBytes);
119
119
  const markdownBytes2 = pdfToMarkdown(pdfFromMarkdown); // the lossiest conversion in the whole package -- see Fidelity
120
+
121
+ const pdfFromCsv = csvToPdf(csvBytes); // composes csvToOds -> odsToPdf internally
122
+ const csvBytes2 = pdfToCsv(pdfFromCsv); // composes pdfToOds -> odsToCsv internally; recovers what was printed, then heuristically re-types it
120
123
  ```
121
124
 
122
125
  Each accepts an optional `signal` (`AbortSignal`) and either `onSubstitution` (X → PDF, called per character not representable in a standard-14 font) or `sink` (PDF → X, called per recoverable parse diagnostic). Every X → PDF conversion additionally accepts `fonts` (extra `ProvidedFont` faces) and `onFontSubstitution` (per family+weight+style that resolved to something else). Neither is needed for the common case — see [Fonts](#fonts).
123
126
 
124
127
  ### Cross-format bridges
125
128
 
126
- Sixteen bridge functions across eight pairs bypass the PDF pivot entirely. Five same-variant direct-copy pairs (`odtToDocx`/`docxToOdt`, `odpToPptx`/`pptxToOdp`, `odsToXlsx`/`xlsxToOds`, `markdownToDocx`/`docxToMarkdown`, `markdownToOdt`/`odtToMarkdown`) compose a direct `readXContent` → `buildYPackage` pivot copy. Two cross-variant semantic-transform pairs (`docxToPptx`/`pptxToDocx`, `odtToOdp`/`odpToOdt`) go through `src/convert/variant-bridges.ts`. One PDF-composed pair (`xlsxToMarkdown`/`markdownToXlsx`) routes through PDF internally — the single lossiest conversion in the package.
129
+ Twenty-two bridge functions across eleven pairs bypass the PDF pivot where a direct path exists. Seven same-variant direct-copy pairs (`odtToDocx`/`docxToOdt`, `odpToPptx`/`pptxToOdp`, `odsToXlsx`/`xlsxToOds`, `csvToOds`/`odsToCsv`, `csvToXlsx`/`xlsxToCsv`, `markdownToDocx`/`docxToMarkdown`, `markdownToOdt`/`odtToMarkdown`) compose a direct `readXContent` → `buildYPackage` pivot copy — the csv pairs are one hop to its spreadsheet siblings, so csv never needs PDF to reach ods or xlsx. Two cross-variant semantic-transform pairs (`docxToPptx`/`pptxToDocx`, `odtToOdp`/`odpToOdt`) go through `src/convert/variant-bridges.ts`. Two PDF-composed pairs (`xlsxToMarkdown`/`markdownToXlsx`, `csvToMarkdown`/`markdownToCsv`) route through PDF internally — the lossiest conversions in the package.
127
130
 
128
131
  ```ts
129
132
  import { odtToDocx, docxToOdt, markdownToDocx, docxToMarkdown } from 'documents.js';
@@ -135,7 +138,7 @@ const docxFromMarkdown = markdownToDocx(markdownBytes);
135
138
  const markdownBytes3 = docxToMarkdown(docxFromMarkdown); // colour, font family/size, and explicit alignment have no markdown source construct -- dropped on this hop
136
139
  ```
137
140
 
138
- Each takes an optional `{ signal }` — no `onSubstitution`/`sink`, since there is no font substitution or PDF-parse degradation. `odtToDocx`/`markdownToDocx`/`docxToOdt`/`docxToMarkdown` additionally take `onMathDiagnostic`, called per formula construct that degraded crossing the bridge.
141
+ Each takes an optional `{ signal }` — no `onSubstitution`/`sink`, since there is no font substitution or PDF-parse degradation. `odtToDocx`/`markdownToDocx`/`docxToOdt`/`docxToMarkdown` additionally take `onMathDiagnostic`, called per formula construct that degraded crossing the bridge. The csv-sourced bridges (`csvToOds`, `csvToXlsx`, `csvToMarkdown`, `csvToPdf`) take `{ delimiter }` — `'\t'` parses the same format as TSV, since a delimiter is a parse option, not a different document format — and `onCellTypeInference`, the per-decision audit channel the read shares with `pdfToOds`. The csv-target bridges (`odsToCsv`, `xlsxToCsv`, `markdownToCsv`, `pdfToCsv`) take `{ delimiter, sheet }`: csv has no second sheet, so writing a multi-sheet source refuses with `CsvSheetNotSpecifiedError` naming every sheet until a caller selects one.
139
142
 
140
143
  ### The `DocumentConverter` port
141
144
 
@@ -151,18 +154,18 @@ const { document, diagnostics } = await converter.convert(
151
154
  );
152
155
  ```
153
156
 
154
- `DocumentFormat` includes `docx`/`pptx`/`xlsx`/`odt`/`odp`/`ods`/`odg`/`odf`/`markdown`/`pdf` — ten members. The port's `conversions` list is derived from `resolveCompositionPlan` plus the `odf`→`pdf` special case — 73 pairs total. `DocumentFormat` is inferred from `DocumentFormatSchema` (a real Zod schema); `DOCUMENT_FORMATS` is exported as a plain array derived from the same schema:
157
+ `DocumentFormat` includes `docx`/`pptx`/`xlsx`/`odt`/`odp`/`ods`/`odg`/`odf`/`csv`/`markdown`/`pdf` — eleven members. The port's `conversions` list is derived from `resolveCompositionPlan` plus the `odf`→`pdf` special case — 91 pairs total. `DocumentFormat` is inferred from `DocumentFormatSchema` (a real Zod schema); `DOCUMENT_FORMATS` is exported as a plain array derived from the same schema:
155
158
 
156
159
  ```ts
157
160
  import { DOCUMENT_FORMATS, DocumentFormatSchema } from 'documents.js';
158
161
 
159
- console.log(DOCUMENT_FORMATS); // ['docx', 'pptx', 'xlsx', 'odt', 'odp', 'ods', 'odg', 'odf', 'markdown', 'pdf']
162
+ console.log(DOCUMENT_FORMATS); // ['docx', 'pptx', 'xlsx', 'odt', 'odp', 'ods', 'odg', 'odf', 'csv', 'markdown', 'pdf']
160
163
  DocumentFormatSchema.parse(userSuppliedFormat); // throws a ZodError for anything outside that list
161
164
  ```
162
165
 
163
166
  ### Intermediate `DocumentPackage`, JSON, and bytes
164
167
 
165
- Every conversion function accepts an `onDocument` callback receiving the intermediate `DocumentPackage` (content + layout). The port surfaces the same value as `package` on `ConversionResult`. For PDF-bypassing bridges, `pkg.layout` is always `undefined`.
168
+ Every conversion function accepts an `onDocument` callback receiving the intermediate `DocumentPackage` — the fused unified tree of document-schema.js 3: `content` (whose own nodes carry `frames`, the rendered page positions the layout pass stamped onto them, in PDF user-space) plus `pages` (each rendered page's size, indexed to match every `frames[].pageIndex`). The port surfaces the same value as `package` on `ConversionResult`. For PDF-bypassing bridges, `pkg.pages` is always `undefined` and no node carries frames — no layout pass ran.
166
169
 
167
170
  ```ts
168
171
  import { docxToPdf } from 'documents.js';
@@ -170,7 +173,9 @@ import { docxToPdf } from 'documents.js';
170
173
  const pdfBytes = docxToPdf(docxBytes, {
171
174
  onDocument: (pkg) => {
172
175
  console.log(pkg.content.kind); // 'wordprocessing'
173
- console.log(pkg.layout?.pages.length); // populated for every X-to-PDF/PDF-to-X conversion
176
+ console.log(pkg.pages?.length); // populated for every X-to-PDF/PDF-to-X conversion
177
+ const block = pkg.content.kind === 'wordprocessing' ? pkg.content.sections[0]?.blocks[0] : undefined;
178
+ console.log(block?.kind === 'paragraph' ? block.runs[0]?.frames : 'no paragraph'); // that run's rendered placements
174
179
  },
175
180
  });
176
181
  ```
@@ -187,7 +192,7 @@ const { kind, value } = documentFromJson(JSON.parse(readFileSync('converted.doc.
187
192
  // kind: 'DocumentPackage' (here) | 'ContentDocument' | 'LayoutDocument'
188
193
  ```
189
194
 
190
- `buildDocumentBytes` rebuilds any `DocumentFormat`'s bytes from a `DocumentPackage` — `'pdf'` writes the `LayoutDocument` half directly (throwing if the package carries none), `'odf'` has no builder and throws, everything else rebuilds from the `ContentDocument` half:
195
+ `buildDocumentBytes` rebuilds any `DocumentFormat`'s bytes from a `DocumentPackage` — `'pdf'` rebuilds the pdf-codec view from the package's own frames+pages (`layoutDocumentFromPackage`, a mechanical inverse walking the content tree and emitting `LayoutItem`s from each node's recorded placements; throwing if the package carries no `pages`), `'odf'` has no builder and throws, everything else rebuilds from the `ContentDocument` half. `layoutDocumentFromPackage` is exported too, for a caller wanting the rebuilt `LayoutDocument` without writing bytes. Two honest limits on the pdf rebuild, both structural properties of what a package records: a run's frames carry positions, not the wrap decisions that distributed its text across them, so a wrapped run re-renders once, whole, at its first recorded placement; and no font registry or positioned formula survives a bare package (a formula block's frame records where it sat while its glyphs render as nothing):
191
196
 
192
197
  ```ts
193
198
  import { buildDocumentBytes, docxToPdf } from 'documents.js';
@@ -200,7 +205,7 @@ const docxBytesAgain = buildDocumentBytes(captured, 'docx');
200
205
 
201
206
  ### Package decode/encode, metadata, and deep imports
202
207
 
203
- `decodeDocumentPackage`/`encodeDocumentPackage` dispatch docx/pptx/xlsx through `ooxml.js`'s OPC codec and odt/odp/ods/odg/odf through `odf.js`'s ODF codec, throwing `UnsupportedPackageFormatError` for `markdown`/`pdf`. `decodeOdbPackage` is the `.odb`-specific sibling (`.odb` is not a `DocumentFormat` member):
208
+ `decodeDocumentPackage`/`encodeDocumentPackage` dispatch docx/pptx/xlsx through `ooxml.js`'s OPC codec and odt/odp/ods/odg/odf through `odf.js`'s ODF codec, throwing `UnsupportedPackageFormatError` for `markdown`/`csv`/`pdf` (none of the three is a package — they are plain text and bytes respectively). `decodeOdbPackage` is the `.odb`-specific sibling (`.odb` is not a `DocumentFormat` member):
204
209
 
205
210
  ```ts
206
211
  import { decodeDocumentPackage, decodeOdbPackage, encodeDocumentPackage } from 'documents.js';
@@ -210,7 +215,7 @@ const docxBytesAgain = encodeDocumentPackage('docx', pkg);
210
215
  const odbPkg = decodeOdbPackage(odbBytes);
211
216
  ```
212
217
 
213
- `readDocumentMetadata`/`setDocumentMetadata` read or patch metadata across any `DocumentFormat`. `setDocumentMetadata` patches in place (source/target formats must match); `odf` is rejected in both directions. `readDocumentMetadata('xlsx', ...)` is a named exception: it renders via `xlsxToPdf` and reads the PDF's metadata, because a direct read and the PDF-preview path genuinely disagree on `createdIso`/`modifiedIso`/`producer`.
218
+ `readDocumentMetadata`/`setDocumentMetadata` read or patch metadata across any `DocumentFormat`. `setDocumentMetadata` patches in place (source/target formats must match); `odf` is rejected in both directions, and `csv` is rejected in both directions too (RFC 4180 text has no metadata container) — `readDocumentMetadata('csv', ...)` answers an empty `LayoutMetadata` for the same reason. `readDocumentMetadata('xlsx', ...)` is a named exception: it renders via `xlsxToPdf` and reads the PDF's metadata, because a direct read and the PDF-preview path genuinely disagree on `createdIso`/`modifiedIso`/`producer`.
214
219
 
215
220
  ```ts
216
221
  import { readDocumentMetadata, setDocumentMetadata } from 'documents.js';
@@ -228,7 +233,7 @@ import { buildOdtPackage } from 'documents.js/edit/odt/content';
228
233
 
229
234
  ### Reading and building xlsx content directly
230
235
 
231
- Every other content format has its own standalone `readXContent`-shaped entry point (`readDocxContent`, `readPptxContent`, `readOdtContent`, `readOdpContent`, `readOdsContent`, `readOdgContent`) — xlsx is no longer the exception. `readXlsxContent`/`buildXlsxPackage` are `ooxml.js`'s own spreadsheet `ContentDocument` read/build pair — the same one the `ods⇄xlsx` bridge and every xlsx metadata-rebuild path already use internally — re-exported here directly rather than wrapped, since `readXlsxContent` already produces the right shape on its own:
236
+ Every other content format has its own standalone `readXContent`-shaped entry point (`readDocxContent`, `readPptxContent`, `readOdtContent`, `readOdpContent`, `readOdsContent`, `readOdgContent`) — xlsx is no longer the exception. `readXlsxContent`/`buildXlsxPackage` are `ooxml.js`'s own spreadsheet `ContentDocument` read/build pair — the same one the `ods⇄xlsx` bridge and every xlsx metadata-rebuild path already use internally — re-exported here directly rather than wrapped, since `readXlsxContent` already produces the right shape on its own. csv's `readCsvContent`/`buildCsvText` are the same kind of directly-exported stage pair, one level further in: they operate on RFC 4180 text rather than a decoded package (see `src/csv/` under Architecture).
232
237
 
233
238
  ```ts
234
239
  import { buildXlsxPackage, decodeDocumentPackage, encodeDocumentPackage, readXlsxContent } from 'documents.js';
@@ -324,7 +329,7 @@ const layout = readPdf(pdfBytes); // -> LayoutDocument: pages of positioned text
324
329
  const bytes = writePdf(layout);
325
330
  ```
326
331
 
327
- The nine PDF round trips and ten PDF-bypassing bridges are also available as schema-validated [`z.codec()`](https://zod.dev) pairs (`pdfCodec`, `docxPdfCodec`, `pptxPdfCodec`, `odtPdfCodec`, `odpPdfCodec`, `odsPdfCodec`, `odgPdfCodec`, `xlsxPdfCodec`, `markdownPdfCodec`, `odtDocxCodec`, `odpPptxCodec`, `odsXlsxCodec`, `markdownDocxCodec`, `markdownOdtCodec`) — the no-options form, adding automatic two-way schema validation:
332
+ The ten PDF round trips and fourteen PDF-bypassing bridge directions are also available as schema-validated [`z.codec()`](https://zod.dev) pairs (`pdfCodec`, `docxPdfCodec`, `pptxPdfCodec`, `odtPdfCodec`, `odpPdfCodec`, `odsPdfCodec`, `odgPdfCodec`, `xlsxPdfCodec`, `csvPdfCodec`, `markdownPdfCodec`, `odtDocxCodec`, `odpPptxCodec`, `odsXlsxCodec`, `odsCsvCodec`, `xlsxCsvCodec`, `markdownDocxCodec`, `markdownOdtCodec`) — the no-options form, adding automatic two-way schema validation. The two PDF-composed pairs have codec forms too (`xlsxMarkdownCodec`, `csvMarkdownCodec`):
328
333
 
329
334
  ```ts
330
335
  import { z } from 'zod';
@@ -502,6 +507,7 @@ The package is layered from generic primitives outward to the two conversion dir
502
507
  - **`src/ooxml/`** — thin adapters over `ooxml.js`'s own `readDocx`/`readPptx`, wrapping results into `ContentDocument`. `docx/formula.ts` is the one local reading pass (splicing OOXML math equations). `docx/extras.ts`'s `readDocxExtras` returns comments/footnotes/headers/footers/numbering.
503
508
  - **`src/odf/`** — ODF-side counterparts: `readOdtContent`/`readOdpContent`/`readOdsContent`/`readOdgContent` are thin adapters over `odf.js`. `formula/read.ts`/`formula/detect.ts` handle embedded formula detection (genuinely new work with no `odf.js`-side equivalent).
504
509
  - **`src/markdown/`** — third adapter family, via `markdown-codec`. `readMarkdownContent` passes `readMarkdown`'s result straight through (it already produces a full `ContentDocument`). `buildMarkdownText` wraps `writeMarkdown`. `text.ts` is the byte↔text boundary. `MarkdownEditor` holds a mutable in-memory `ContentDocument`.
510
+ - **`src/csv/`** — fourth adapter family, sharing the spreadsheet variant with xlsx/ods. `records.ts` is the RFC 4180 record parser/writer (one shared `quoteCsvField`, also used by the `.odb` CSV exporter); `text.ts` is the byte↔text boundary, rejecting malformed UTF-8; `read.ts` turns records into a spreadsheet `ContentDocument` (first record as verbatim string header, data cells through the same cell-typing heuristic `pdfToOds` uses); `write.ts` turns one sheet of a spreadsheet `ContentDocument` back into records via each cell's `displayText`. TSV is the same format with `{ delimiter: '\t' }` on either side.
505
511
  - **`src/layout/`** — the pure conversion algorithms: `engine.ts` (wordprocessing → layout: flow, line-breaking, pagination), `slides.ts` (presentation → layout: direct placement), `sheets.ts` (spreadsheet → layout: grid, print settings, the first algorithm accepting `AbortSignal`), `drawing.ts` (drawing → layout: vector primitives + shape reuse), `reconstruct.ts` (layout → content: baseline clustering for wordprocessing/presentation, near-1:1 mapping for drawing, gridline-lattice-or-text-clustering for spreadsheet).
506
512
  - **`src/hsqldb/`** — `.odb` decoders, four tiers: `script.ts` (TEXT-script DDL/DML parser), `rowformat.ts`/`cache.ts` (CACHED binary row-store), `binary-script.ts` (BINARY/COMPRESSED whole-script). All import only `document-schema.js` — no odf.js knowledge.
507
513
  - **`src/firebird/`** — Tier 3: gbak logical-backup reader. `reader.ts` (attribute framing + RLE decompression + XDR decoding), `schema.ts`/`data.ts` (table/row walking). No ratified spec — built against Firebird's own engine source.
@@ -543,10 +549,12 @@ To run a single test file: `pnpm vitest run src/path/to/file.test.ts`.
543
549
  - **`ooxml.js`'s typed readers are the basis for conversion** — `readDocxContent`/`readPptxContent` are thin wrappers, not independent walks. They are deliberately not re-exported (exposing both would invite using the wrong one). `readDocx`'s `comments`/`footnotes`/`headers`/`footers`/`numbering` are exposed via `readDocxExtras`. `readPptx` has no extras reader yet. xlsx is the one exception: `ooxml.js`'s `readXlsxContent`/`buildXlsxPackage` already read/write a spreadsheet `ContentDocument` directly (unlike `readDocx`/`readPptx`, which `readDocxContent`/`readPptxContent` wrap), so they're re-exported as-is rather than given a documents.js-local wrapper of their own — `readXlsx`, the separate lossy cell-values-only view, stays unexported for the same reason `readDocx`/`readPptx` do.
544
550
  - **ODF text content is not a plain string.** ODF represents runs of spaces as `<text:s>`, tabs as `<text:tab/>`, line breaks as `<text:line-break/>` — all elements, not text nodes. Every ODF text getter MUST call `decodeOdfText`, never `textContent()` — which silently drops them (no error, just shorter text).
545
551
  - **docx⇄PDF and pptx⇄PDF are explicitly not round-trip-lossless** — see [Fidelity](#fidelity). The cross-format bridge pairs are a genuinely different case.
546
- - **A `DocumentPackage` from `onDocument`/`ConversionResult.package` is a snapshot, not a live view** — mutating `content` afterwards leaves `layout` stale; nothing detects or rejects that.
552
+ - **A `DocumentPackage` from `onDocument`/`ConversionResult.package` is a snapshot, not a live view** — mutating `content` after the layout pass leaves its nodes' `frames` stale; nothing detects or rejects that, and the schema keeps `content`'s populated `frames` and `pages` in sync with nothing.
553
+ - **`frames` are stamped in place onto the caller's own content tree** — `convertXToLayout` mutates its `ContentDocument` argument (each node's placements are appended to its own `frames` array, one frame per rendered placement: per wrapped fragment on a run, the cell box on a cell, the emitted item's box on an image/vector/shape) and returns `pages` alongside the internal `LayoutDocument`. A run wrapped across three lines carries three frames; a repeat-row spreadsheet cell carries one per page it re-renders on. Reconstructors attach frames from the exact items each reconstructed node was clustered from, so every PDF-to-X conversion's content carries genuine positions too.
547
554
  - **ODF text getters must call `decodeOdfText`.** See the dedicated gotcha above.
548
555
  - **`readPdf` recovers rect/ellipse/line as their own `LayoutRect`/`LayoutEllipse`/`LayoutLine` kinds** via pdf-codec's shape-pattern detection — an axis-aligned closed four-corner subpath is a rect, four kappa-ratio cubics at cardinal points is an ellipse, an open single straight stroke is a line. A false positive changes kind, never geometry. Off-axis rotations, freeform curves, and multi-subpath figures narrow to `LayoutPath`.
549
556
  - **`pdfToOds` re-types cells heuristically — this is probabilistic, not a fidelity guarantee.** A rendered PDF never carries a cell's typed value, only the printed string. Re-typing fires only where the string has exactly one defensible reading: the decimal must be exactly representable as a JS number; separators must be unambiguous (`"1,234"` is declined — competing European reading is 1.234); leading zeros decline (`"007"`); dates must self-state their component roles (ISO or named month accepted; `"01/02/2024"` declined). `TRUE`/`FALSE` re-type as booleans; `Yes`/`No` are declined. `displayText` always carries the rendered string verbatim. `onCellTypeInference` reports every decision. A formula is never claimed.
557
+ - **The csv read shares `pdfToOds`'s cell-typing heuristic, with the same decision-only audit channel.** The first record is a verbatim string header (never re-typed, even when it looks like data); data cells re-type through `inferCellValue` exactly as the PDF reconstructor does — declines keep the plain string, `displayText` always carries the raw field text, and `onCellTypeInference` fires per decision, staying silent for header cells and no-candidate text. The parser drops blank records, so a record of one empty field alone cannot round-trip. Writing csv takes exactly one sheet: a multi-sheet source refuses with `CsvSheetNotSpecifiedError` naming every sheet until `{ sheet }` selects one. TSV is not a separate format — `{ delimiter: '\t' }` on either side parses or writes the same grid.
550
558
  - **`reconstructWordprocessing`/`reconstructPresentation` recover vector primitives too**, in a nested drawing document — a rule under a heading, an underline, a cell background are all recovered as vectors (intended — discarding real content because it might be incidental is ruled out). A table's gridlines are excluded from vector recovery when the lattice claims them.
551
559
  - **Recovered vectors round-trip through all four readers** — `buildDocxPackage`/`buildPptxPackage` write real DrawingML; `buildOdtPackage`/`buildOdpPackage` write real `draw:rect`/`draw:ellipse`/`draw:line`/`draw:path`. The six PDF-bypassing bridges carry vector geometry across too.
552
560
  - **Each format wraps a vector shape differently.** OOXML: pptx gets a plain `p:sp`; docx gets a `w:drawing`/`wp:anchor` with `behindDoc="1"`/`wp:wrapNone` carrying a `wps:wsp`. ODF: odp appends to `draw:page`; odt anchors in a `text:p` with `style:horizontal-rel`/`style:vertical-rel="page"` (page-absolute coordinates) and `style:run-through="background"`.
@@ -604,7 +612,7 @@ To run a single test file: `pnpm vitest run src/path/to/file.test.ts`.
604
612
  - **A formula that cannot typeset degrades to its plain-text stand-in, never to nothing.** `buildDocxPackage` writes real OMML; `buildOdtPackage` writes real embedded formula sub-documents. The markdown writer is the only stand-in-only path. `odmToPdf` carries formulas through as ordinary blocks.
605
613
  - **OMML read/write are deliberately asymmetric** — the reader covers more (`m:d`, `m:nary`, `m:acc`, `m:bar`, `m:func`, `m:sPre`) because it must read what Word wrote. `docx → odt → docx` round trips keep the mathematics but may change the OMML construct.
606
614
  - **The OMML translator covers exactly what `src/mathml/layout.ts` typesets.** A stretchy fence diverges: PDF stretches it, docx writes it at base size. `munderover` becomes nested `m:limUpp`/`m:limLow` (no operand scope in MathML).
607
- - **`sourcePath` traces a `LayoutItem` to its `ContentDocument` origin, but only within one read+layout pass** — not an edit-tracking mechanism.
615
+ - **`sourcePath` traces a `LayoutItem` to its `ContentDocument` origin, but only within one read+layout pass** — not an edit-tracking mechanism. Since the frames fusion it survives as traceability only: the authoritative node↔position association is each content node's own `frames`, stamped at the moment of layout (or of reconstruction) rather than re-matched by string afterwards.
608
616
  - **`readMarkdownContent` passes `readMarkdown`'s result straight through** — `markdown-codec` already produces a full `ContentDocument`.
609
617
  - **Every markdown construct-mapping gap is a documented `MarkdownDiagnosticCodes` entry** (`md/invented-page-geometry`, `md/nested-emphasis-flattened`, `md/link-title-dropped`, `md/code-block-info-string-dropped`, `md/blockquote-nested-depth`, `md/list-item-block-unlisted`, `md/list-item-multi-block-flattened`, `md/image-unresolved`, `md/raw-html-preserved-as-text`/`md/raw-html-dropped`, `md/front-matter-key-unmapped`, `md/heading-level-clamped`, `md/adjacent-links-merged`, `md/code-span-as-monospace-run`, `md/paragraph-indent-dropped`, `md/list-numid-fallback`, `md/table-cell-formatting-dropped`, `md/table-cell-multi-paragraph-joined`) — never a silent approximation.
610
618
  - **`buildMarkdownText` throws for non-`'wordprocessing'` `ContentDocument`.**
@@ -615,20 +623,21 @@ To run a single test file: `pnpm vitest run src/path/to/file.test.ts`.
615
623
 
616
624
  Read as **row → column**. `✓` lossless, `~` bounded, `✗` lossy, `✗✗` severe, `→` one-way, `–` no conversion. `.odm`/`.odb` sit outside this table.
617
625
 
618
- | ↓ from \ to → | docx | pptx | xlsx | odt | odp | ods | odg | odf | markdown | pdf |
619
- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
620
- | **docx** | — | ~ | – | ✓ | – | – | – | – | ✗ | ~ |
621
- | **pptx** | ~ | — | – | – | ✓ | – | – | – | – | ~ |
622
- | **xlsx** | – | – | — | – | – | ~ | – | – | ✗✗ | ~ |
623
- | **odt** | ✓ | – | – | — | ~ | – | – | – | ✗ | ~ |
624
- | **odp** | – | ✓ | – | ~ | — | – | – | – | – | ~ |
625
- | **ods** | – | – | ~ | – | – | — | – | – | – | ~ |
626
- | **odg** | – | – | – | – | – | – | — | – | – | ~ |
627
- | **odf** | – | – | – | – | – | – | – | — | – | → |
628
- | **markdown** | ~ | – | ✗✗ | ~ | – | – | – | – | — | ~ |
629
- | **pdf** | ✗ | ✗ | | ✗ | ✗ | | ✗ | – | ✗✗ | — |
630
-
631
- 73 of 90 directional pairs are routable. The `ContentDocument`/`LayoutDocument` pivots are the hub, not PDF — fourteen bridges bypass PDF entirely.
626
+ | ↓ from \ to → | docx | pptx | xlsx | odt | odp | ods | odg | odf | markdown | csv | pdf |
627
+ | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
628
+ | **docx** | — | ~ | – | ✓ | – | – | – | – | ✗ | ✗ | ~ |
629
+ | **pptx** | ~ | — | – | – | ✓ | – | – | – | – | ✗ | ~ |
630
+ | **xlsx** | – | – | — | – | – | ~ | – | – | ✗✗ | ~ | ~ |
631
+ | **odt** | ✓ | – | – | — | ~ | – | – | – | ✗ | ✗ | ~ |
632
+ | **odp** | – | ✓ | – | ~ | — | – | – | – | – | ✗ | ~ |
633
+ | **ods** | – | – | ~ | – | – | — | – | – | – | ~ | ~ |
634
+ | **odg** | – | – | – | – | – | – | — | – | – | ✗ | ~ |
635
+ | **odf** | – | – | – | – | – | – | – | — | – | – | → |
636
+ | **markdown** | ~ | – | ✗✗ | ~ | – | – | – | – | — | ✗✗ | ~ |
637
+ | **csv** | ✗ | ✗ | | ✗ | ✗ | | ✗ | – | ✗✗ | — | ~ |
638
+ | **pdf** | ✗ | ✗ | ✗ | ✗ | ✗ | ✗ | ✗ | – | ✗✗ | ✗ | — |
639
+
640
+ 91 of 110 directional pairs are routable. The `ContentDocument`/`LayoutDocument` pivots are the hub, not PDF — eighteen bridges bypass PDF entirely.
632
641
 
633
642
  **X → PDF** is a genuine layout render: positioned text, images, tables, lists, vector primitives, styled through the full cascade. It is a faithful visual approximation, not pixel-identical — closeness depends on font availability.
634
643
 
@@ -640,9 +649,9 @@ Read as **row → column**. `✓` lossless, `~` bounded, `✗` lossy, `✗✗` s
640
649
 
641
650
  **PDF → ods** recovers what was printed, not what was entered. The printed string always survives in `displayText`; re-typed `value` is explicitly probabilistic inference.
642
651
 
643
- **`markdownToPdf`/`pdfToMarkdown`** is the lossiest round trip: `markdownToPdf` is faithful, but `pdfToMarkdown` stacks reconstruction lossiness PLUS markdown's coarser vocabulary (no colour, font, size, alignment).
652
+ **`markdownToPdf`/`pdfToMarkdown`** is the lossiest round trip: `markdownToPdf` is faithful, but `pdfToMarkdown` stacks reconstruction lossiness PLUS markdown's coarser vocabulary (no colour, font, size, alignment). The PDF-composed markdown bridges (`xlsxToMarkdown`/`markdownToXlsx`, `csvToMarkdown`/`markdownToCsv`) stack the same two losses in both directions — hence their `✗✗` cells.
644
653
 
645
- **The first three bridge pairs** (odt⇄docx, odp⇄pptx, ods⇄xlsx) bypass PDF entirely — no layout engine, no reconstruction. Text, styling, tables, lists, rotated shapes survive completely. `ods⇄xlsx` has small format-boundary limits (time cells, formula dialects). Embedded formulas survive `odtToDocx` as real OOXML math.
654
+ **The five same-variant bridge pairs** (odt⇄docx, odp⇄pptx, ods⇄xlsx, csv⇄ods, csv⇄xlsx) bypass PDF entirely — no layout engine, no reconstruction. Text, styling, tables, lists, rotated shapes survive completely. `ods⇄xlsx` has small format-boundary limits (time cells, formula dialects). Embedded formulas survive `odtToDocx` as real OOXML math. The csv pairs are bounded by what csv itself carries: toward ods/xlsx nothing the csv had is lost, while writing to csv collapses each cell to its `displayText` — formulas become their rendered values, formatting disappears, and a multi-sheet source must name the sheet it wants.
646
655
 
647
656
  **The two markdown bridge pairs** bypass PDF too, but markdown's grammar has no construct for colour/font/size/alignment — `docxToMarkdown`/`odtToMarkdown` drop them (format-boundary loss, not approximation).
648
657
 
@@ -15,6 +15,9 @@ const require_odf_odp_read = require("../odf/odp/read.cjs");
15
15
  const require_odf_ods_read = require("../odf/ods/read.cjs");
16
16
  const require_odf_odg_read = require("../odf/odg/read.cjs");
17
17
  const require_markdown_text = require("../markdown/text.cjs");
18
+ const require_csv_text = require("../csv/text.cjs");
19
+ const require_csv_read = require("../csv/read.cjs");
20
+ const require_csv_write = require("../csv/write.cjs");
18
21
  const require_ports_abort = require("../ports/abort.cjs");
19
22
  const require_package_codec = require("../package-codec.cjs");
20
23
  let ooxml_js = require("ooxml.js");
@@ -81,6 +84,10 @@ const DOCUMENT_FORMAT_CODECS = {
81
84
  }),
82
85
  write: (content) => require_markdown_text.encodeMarkdownText(require_markdown_write.buildMarkdownText(content))
83
86
  } },
87
+ csv: { content: {
88
+ read: (bytes) => require_csv_read.readCsvContent(require_csv_text.decodeCsvText(bytes)),
89
+ write: (content) => require_csv_text.encodeCsvText(require_csv_write.buildCsvText(content))
90
+ } },
84
91
  pdf: { layout: {
85
92
  read: (bytes, options) => (0, pdf_codec.readPdf)(requireArrayBufferBytes(bytes), { signal: options?.signal }),
86
93
  write: (layout, options) => (0, pdf_codec.writePdf)(layout, { signal: options?.signal })
@@ -14,6 +14,9 @@ import { readOdpContent } from "../odf/odp/read.js";
14
14
  import { readOdsContent } from "../odf/ods/read.js";
15
15
  import { readOdgContent } from "../odf/odg/read.js";
16
16
  import { decodeMarkdownText, encodeMarkdownText } from "../markdown/text.js";
17
+ import { decodeCsvText, encodeCsvText } from "../csv/text.js";
18
+ import { readCsvContent } from "../csv/read.js";
19
+ import { buildCsvText } from "../csv/write.js";
17
20
  import { throwIfAborted } from "../ports/abort.js";
18
21
  import { decodeDocumentPackage, encodeDocumentPackage } from "../package-codec.js";
19
22
  import { buildXlsxPackage, readXlsxContent } from "ooxml.js";
@@ -80,6 +83,10 @@ const DOCUMENT_FORMAT_CODECS = {
80
83
  }),
81
84
  write: (content) => encodeMarkdownText(buildMarkdownText(content))
82
85
  } },
86
+ csv: { content: {
87
+ read: (bytes) => readCsvContent(decodeCsvText(bytes)),
88
+ write: (content) => encodeCsvText(buildCsvText(content))
89
+ } },
83
90
  pdf: { layout: {
84
91
  read: (bytes, options) => readPdf(requireArrayBufferBytes(bytes), { signal: options?.signal }),
85
92
  write: (layout, options) => writePdf(layout, { signal: options?.signal })
@@ -31,6 +31,11 @@ const FORMAT_CAPABILITIES = {
31
31
  variant: "spreadsheet",
32
32
  hasLayoutPath: false
33
33
  },
34
+ csv: {
35
+ format: "csv",
36
+ variant: "spreadsheet",
37
+ hasLayoutPath: false
38
+ },
34
39
  odg: {
35
40
  format: "odg",
36
41
  variant: "drawing",
@@ -30,6 +30,11 @@ const FORMAT_CAPABILITIES = {
30
30
  variant: "spreadsheet",
31
31
  hasLayoutPath: false
32
32
  },
33
+ csv: {
34
+ format: "csv",
35
+ variant: "spreadsheet",
36
+ hasLayoutPath: false
37
+ },
33
38
  odg: {
34
39
  format: "odg",
35
40
  variant: "drawing",
@@ -59,7 +59,25 @@ const xlsxMarkdownCodec = zod.z.codec(require_model_bytes.XlsxBytesSchema, requi
59
59
  decode: (xlsxBytes) => require_convert_convert.xlsxToMarkdown(xlsxBytes),
60
60
  encode: (markdownBytes) => require_convert_convert.markdownToXlsx(markdownBytes)
61
61
  });
62
+ const csvPdfCodec = zod.z.codec(require_model_bytes.CsvBytesSchema, require_model_bytes.PdfBytesSchema, {
63
+ decode: (csvBytes) => require_convert_convert.csvToPdf(csvBytes),
64
+ encode: (pdfBytes) => require_convert_convert.pdfToCsv(pdfBytes)
65
+ });
66
+ const odsCsvCodec = zod.z.codec(require_model_bytes.OdsBytesSchema, require_model_bytes.CsvBytesSchema, {
67
+ decode: (odsBytes) => require_convert_convert.odsToCsv(odsBytes),
68
+ encode: (csvBytes) => require_convert_convert.csvToOds(csvBytes)
69
+ });
70
+ const xlsxCsvCodec = zod.z.codec(require_model_bytes.XlsxBytesSchema, require_model_bytes.CsvBytesSchema, {
71
+ decode: (xlsxBytes) => require_convert_convert.xlsxToCsv(xlsxBytes),
72
+ encode: (csvBytes) => require_convert_convert.csvToXlsx(csvBytes)
73
+ });
74
+ const csvMarkdownCodec = zod.z.codec(require_model_bytes.CsvBytesSchema, require_model_bytes.MarkdownBytesSchema, {
75
+ decode: (csvBytes) => require_convert_convert.csvToMarkdown(csvBytes),
76
+ encode: (markdownBytes) => require_convert_convert.markdownToCsv(markdownBytes)
77
+ });
62
78
  //#endregion
79
+ exports.csvMarkdownCodec = csvMarkdownCodec;
80
+ exports.csvPdfCodec = csvPdfCodec;
63
81
  exports.docxPdfCodec = docxPdfCodec;
64
82
  exports.markdownDocxCodec = markdownDocxCodec;
65
83
  exports.markdownOdtCodec = markdownOdtCodec;
@@ -67,10 +85,12 @@ exports.markdownPdfCodec = markdownPdfCodec;
67
85
  exports.odgPdfCodec = odgPdfCodec;
68
86
  exports.odpPdfCodec = odpPdfCodec;
69
87
  exports.odpPptxCodec = odpPptxCodec;
88
+ exports.odsCsvCodec = odsCsvCodec;
70
89
  exports.odsPdfCodec = odsPdfCodec;
71
90
  exports.odsXlsxCodec = odsXlsxCodec;
72
91
  exports.odtDocxCodec = odtDocxCodec;
73
92
  exports.odtPdfCodec = odtPdfCodec;
74
93
  exports.pptxPdfCodec = pptxPdfCodec;
94
+ exports.xlsxCsvCodec = xlsxCsvCodec;
75
95
  exports.xlsxMarkdownCodec = xlsxMarkdownCodec;
76
96
  exports.xlsxPdfCodec = xlsxPdfCodec;
@@ -14,5 +14,9 @@ declare const odsXlsxCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint
14
14
  declare const markdownDocxCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
15
15
  declare const markdownOdtCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
16
16
  declare const xlsxMarkdownCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
17
+ declare const csvPdfCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
18
+ declare const odsCsvCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
19
+ declare const xlsxCsvCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
20
+ declare const csvMarkdownCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
17
21
  //#endregion
18
- export { docxPdfCodec, markdownDocxCodec, markdownOdtCodec, markdownPdfCodec, odgPdfCodec, odpPdfCodec, odpPptxCodec, odsPdfCodec, odsXlsxCodec, odtDocxCodec, odtPdfCodec, pptxPdfCodec, xlsxMarkdownCodec, xlsxPdfCodec };
22
+ export { csvMarkdownCodec, csvPdfCodec, docxPdfCodec, markdownDocxCodec, markdownOdtCodec, markdownPdfCodec, odgPdfCodec, odpPdfCodec, odpPptxCodec, odsCsvCodec, odsPdfCodec, odsXlsxCodec, odtDocxCodec, odtPdfCodec, pptxPdfCodec, xlsxCsvCodec, xlsxMarkdownCodec, xlsxPdfCodec };
@@ -14,5 +14,9 @@ declare const odsXlsxCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint
14
14
  declare const markdownDocxCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
15
15
  declare const markdownOdtCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
16
16
  declare const xlsxMarkdownCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
17
+ declare const csvPdfCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
18
+ declare const odsCsvCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
19
+ declare const xlsxCsvCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
20
+ declare const csvMarkdownCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>, z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Array<ArrayBuffer>>>;
17
21
  //#endregion
18
- export { docxPdfCodec, markdownDocxCodec, markdownOdtCodec, markdownPdfCodec, odgPdfCodec, odpPdfCodec, odpPptxCodec, odsPdfCodec, odsXlsxCodec, odtDocxCodec, odtPdfCodec, pptxPdfCodec, xlsxMarkdownCodec, xlsxPdfCodec };
22
+ export { csvMarkdownCodec, csvPdfCodec, docxPdfCodec, markdownDocxCodec, markdownOdtCodec, markdownPdfCodec, odgPdfCodec, odpPdfCodec, odpPptxCodec, odsCsvCodec, odsPdfCodec, odsXlsxCodec, odtDocxCodec, odtPdfCodec, pptxPdfCodec, xlsxCsvCodec, xlsxMarkdownCodec, xlsxPdfCodec };
@@ -1,5 +1,5 @@
1
- import { DocxBytesSchema, MarkdownBytesSchema, OdgBytesSchema, OdpBytesSchema, OdsBytesSchema, OdtBytesSchema, PdfBytesSchema, PptxBytesSchema, XlsxBytesSchema } from "../model/bytes.js";
2
- import { docxToMarkdown, docxToOdt, docxToPdf, markdownToDocx, markdownToOdt, markdownToPdf, markdownToXlsx, odgToPdf, odpToPdf, odpToPptx, odsToPdf, odsToXlsx, odtToDocx, odtToMarkdown, odtToPdf, pdfToDocx, pdfToMarkdown, pdfToOdg, pdfToOdp, pdfToOds, pdfToOdt, pdfToPptx, pdfToXlsx, pptxToOdp, pptxToPdf, xlsxToMarkdown, xlsxToOds, xlsxToPdf } from "./convert.js";
1
+ import { CsvBytesSchema, DocxBytesSchema, MarkdownBytesSchema, OdgBytesSchema, OdpBytesSchema, OdsBytesSchema, OdtBytesSchema, PdfBytesSchema, PptxBytesSchema, XlsxBytesSchema } from "../model/bytes.js";
2
+ import { csvToMarkdown, csvToOds, csvToPdf, csvToXlsx, docxToMarkdown, docxToOdt, docxToPdf, markdownToCsv, markdownToDocx, markdownToOdt, markdownToPdf, markdownToXlsx, odgToPdf, odpToPdf, odpToPptx, odsToCsv, odsToPdf, odsToXlsx, odtToDocx, odtToMarkdown, odtToPdf, pdfToCsv, pdfToDocx, pdfToMarkdown, pdfToOdg, pdfToOdp, pdfToOds, pdfToOdt, pdfToPptx, pdfToXlsx, pptxToOdp, pptxToPdf, xlsxToCsv, xlsxToMarkdown, xlsxToOds, xlsxToPdf } from "./convert.js";
3
3
  import { z } from "zod";
4
4
  //#region src/convert/codec.ts
5
5
  const docxPdfCodec = z.codec(DocxBytesSchema, PdfBytesSchema, {
@@ -58,5 +58,21 @@ const xlsxMarkdownCodec = z.codec(XlsxBytesSchema, MarkdownBytesSchema, {
58
58
  decode: (xlsxBytes) => xlsxToMarkdown(xlsxBytes),
59
59
  encode: (markdownBytes) => markdownToXlsx(markdownBytes)
60
60
  });
61
+ const csvPdfCodec = z.codec(CsvBytesSchema, PdfBytesSchema, {
62
+ decode: (csvBytes) => csvToPdf(csvBytes),
63
+ encode: (pdfBytes) => pdfToCsv(pdfBytes)
64
+ });
65
+ const odsCsvCodec = z.codec(OdsBytesSchema, CsvBytesSchema, {
66
+ decode: (odsBytes) => odsToCsv(odsBytes),
67
+ encode: (csvBytes) => csvToOds(csvBytes)
68
+ });
69
+ const xlsxCsvCodec = z.codec(XlsxBytesSchema, CsvBytesSchema, {
70
+ decode: (xlsxBytes) => xlsxToCsv(xlsxBytes),
71
+ encode: (csvBytes) => csvToXlsx(csvBytes)
72
+ });
73
+ const csvMarkdownCodec = z.codec(CsvBytesSchema, MarkdownBytesSchema, {
74
+ decode: (csvBytes) => csvToMarkdown(csvBytes),
75
+ encode: (markdownBytes) => markdownToCsv(markdownBytes)
76
+ });
61
77
  //#endregion
62
- export { docxPdfCodec, markdownDocxCodec, markdownOdtCodec, markdownPdfCodec, odgPdfCodec, odpPdfCodec, odpPptxCodec, odsPdfCodec, odsXlsxCodec, odtDocxCodec, odtPdfCodec, pptxPdfCodec, xlsxMarkdownCodec, xlsxPdfCodec };
78
+ export { csvMarkdownCodec, csvPdfCodec, docxPdfCodec, markdownDocxCodec, markdownOdtCodec, markdownPdfCodec, odgPdfCodec, odpPdfCodec, odpPptxCodec, odsCsvCodec, odsPdfCodec, odsXlsxCodec, odtDocxCodec, odtPdfCodec, pptxPdfCodec, xlsxCsvCodec, xlsxMarkdownCodec, xlsxPdfCodec };
@@ -16,6 +16,9 @@ const require_odf_odp_read = require("../odf/odp/read.cjs");
16
16
  const require_odf_ods_read = require("../odf/ods/read.cjs");
17
17
  const require_odf_odg_read = require("../odf/odg/read.cjs");
18
18
  const require_markdown_text = require("../markdown/text.cjs");
19
+ const require_csv_text = require("../csv/text.cjs");
20
+ const require_csv_read = require("../csv/read.cjs");
21
+ const require_csv_write = require("../csv/write.cjs");
19
22
  const require_layout_engine = require("../layout/engine.cjs");
20
23
  const require_layout_slides = require("../layout/slides.cjs");
21
24
  const require_ports_abort = require("../ports/abort.cjs");
@@ -39,8 +42,12 @@ const CONTENT_FORMATS = [
39
42
  "odp",
40
43
  "ods",
41
44
  "odg",
45
+ "csv",
42
46
  "markdown"
43
47
  ];
48
+ function isTextFormatNode(node) {
49
+ return !node.hasSourcePackage;
50
+ }
44
51
  const FORMAT_NODES = {
45
52
  docx: {
46
53
  variant: "wordprocessing",
@@ -116,6 +123,21 @@ const FORMAT_NODES = {
116
123
  build: (content) => require_markdown_write.buildMarkdownText(content),
117
124
  encode: (text) => require_markdown_text.encodeMarkdownText(text),
118
125
  hasSourcePackage: false
126
+ },
127
+ csv: {
128
+ variant: "spreadsheet",
129
+ family: "csv",
130
+ decode: (bytes) => require_csv_text.decodeCsvText(bytes),
131
+ read: (text, options) => require_csv_read.readCsvContent(text, {
132
+ delimiter: options?.delimiter,
133
+ onCellTypeInference: options?.onCellTypeInference
134
+ }),
135
+ build: (content, options) => require_csv_write.buildCsvText(content, {
136
+ delimiter: options?.delimiter,
137
+ sheet: options?.sheet
138
+ }),
139
+ encode: (text) => require_csv_text.encodeCsvText(text),
140
+ hasSourcePackage: false
119
141
  }
120
142
  };
121
143
  const LAYOUT_CAPABLE = /* @__PURE__ */ new Set([
@@ -162,7 +184,7 @@ function executeBridge(source, target, bytes, options) {
162
184
  const sourceNode = FORMAT_NODES[source];
163
185
  const targetNode = FORMAT_NODES[target];
164
186
  let content;
165
- if (sourceNode.family === "markdown") {
187
+ if (isTextFormatNode(sourceNode)) {
166
188
  const text = sourceNode.decode(bytes);
167
189
  content = sourceNode.read(text, options);
168
190
  } else {
@@ -182,8 +204,8 @@ function executeBridge(source, target, bytes, options) {
182
204
  formatVersion: document_schema_js.DOCUMENT_PACKAGE_FORMAT_VERSION,
183
205
  content: buildContent
184
206
  });
185
- if (targetNode.family === "markdown") {
186
- const text = targetNode.build(buildContent);
207
+ if (isTextFormatNode(targetNode)) {
208
+ const text = targetNode.build(buildContent, options);
187
209
  return targetNode.encode(text);
188
210
  }
189
211
  const pkg = targetNode.build(buildContent, options);
@@ -194,7 +216,7 @@ function executeToPdf(format, bytes, options) {
194
216
  const node = FORMAT_NODES[format];
195
217
  let content;
196
218
  let fonts;
197
- if (node.family === "markdown") {
219
+ if (isTextFormatNode(node)) {
198
220
  require_ports_abort.throwIfAborted(options?.signal);
199
221
  const text = node.decode(bytes);
200
222
  const read = node.read(text, options);
@@ -230,6 +252,7 @@ function executeToPdf(format, bytes, options) {
230
252
  }
231
253
  const measurer = (0, pdf_codec.createFontMeasurer)(fonts);
232
254
  let layout;
255
+ let pages;
233
256
  let formulas;
234
257
  switch (content.kind) {
235
258
  case "wordprocessing": {
@@ -238,6 +261,7 @@ function executeToPdf(format, bytes, options) {
238
261
  mathMetricsAt
239
262
  });
240
263
  layout = result.document;
264
+ pages = result.pages;
241
265
  formulas = result.formulas;
242
266
  break;
243
267
  }
@@ -247,6 +271,7 @@ function executeToPdf(format, bytes, options) {
247
271
  mathMetricsAt
248
272
  });
249
273
  layout = result.document;
274
+ pages = result.pages;
250
275
  formulas = result.formulas;
251
276
  break;
252
277
  }
@@ -257,18 +282,22 @@ function executeToPdf(format, bytes, options) {
257
282
  signal: options?.signal
258
283
  });
259
284
  layout = result.document;
285
+ pages = result.pages;
260
286
  formulas = result.formulas;
261
287
  break;
262
288
  }
263
- case "drawing":
264
- layout = LAYOUT_ENGINES.drawing(content, { measurer });
289
+ case "drawing": {
290
+ const result = LAYOUT_ENGINES.drawing(content, { measurer });
291
+ layout = result.document;
292
+ pages = result.pages;
265
293
  break;
294
+ }
266
295
  default: throw new Error(`executeToPdf: cannot lay out a '${content.kind}' document`);
267
296
  }
268
297
  options?.onDocument?.({
269
298
  formatVersion: document_schema_js.DOCUMENT_PACKAGE_FORMAT_VERSION,
270
299
  content,
271
- layout
300
+ pages: [...pages]
272
301
  });
273
302
  if (formulas === void 0) return (0, pdf_codec.writePdf)(layout, {
274
303
  signal: options?.signal,
@@ -288,14 +317,21 @@ function executeFromPdf(target, bytes, options) {
288
317
  signal: options?.signal,
289
318
  sink: options?.sink
290
319
  });
291
- const content = RECONSTRUCTORS[node.variant](layout, { signal: options?.signal });
320
+ const content = RECONSTRUCTORS[node.variant](layout, {
321
+ signal: options?.signal,
322
+ onCellTypeInference: options?.onCellTypeInference
323
+ });
324
+ const pages = layout.pages.map((page) => ({
325
+ widthPt: page.widthPt,
326
+ heightPt: page.heightPt
327
+ }));
292
328
  options?.onDocument?.({
293
329
  formatVersion: document_schema_js.DOCUMENT_PACKAGE_FORMAT_VERSION,
294
330
  content,
295
- layout
331
+ pages
296
332
  });
297
- if (node.family === "markdown") {
298
- const text = node.build(content);
333
+ if (isTextFormatNode(node)) {
334
+ const text = node.build(content, options);
299
335
  return node.encode(text);
300
336
  }
301
337
  const pkg = node.build(content);
@@ -1,6 +1,7 @@
1
1
  import { t as ClockPort } from "../clock-C7SUuYN0.cjs";
2
2
  import { DocumentFormat } from "./port.cjs";
3
3
  import { t as ContentVariant } from "../capability-an5gSNsu.cjs";
4
+ import { CellTypeInferenceSink } from "../layout/cell-typing.cjs";
4
5
  import { t as OmmlDiagnostic } from "../shared-DLZ3IQUl.cjs";
5
6
  import { ContentDocument, DocumentPackage, FontSubstitution, ProvidedFont } from "document-schema.js";
6
7
  import { MarkdownImageResolver } from "markdown-codec";
@@ -21,9 +22,12 @@ interface UnifiedConversionOptions {
21
22
  readonly sourcePath?: string;
22
23
  }) => void;
23
24
  readonly images?: MarkdownImageResolver;
25
+ readonly delimiter?: string;
26
+ readonly sheet?: string;
27
+ readonly onCellTypeInference?: CellTypeInferenceSink;
24
28
  readonly clock?: ClockPort;
25
29
  }
26
- type ContentFormat = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'odg' | 'markdown';
30
+ type ContentFormat = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'odg' | 'csv' | 'markdown';
27
31
  type LayoutVariant = Exclude<ContentVariant, 'formula'>;
28
32
  interface PackageFormatNode {
29
33
  readonly variant: LayoutVariant;
@@ -34,16 +38,16 @@ interface PackageFormatNode {
34
38
  readonly encode: (pkg: SourcePackage) => Uint8Array<ArrayBuffer>;
35
39
  readonly hasSourcePackage: true;
36
40
  }
37
- interface MarkdownFormatNode {
41
+ interface TextFormatNode {
38
42
  readonly variant: LayoutVariant;
39
- readonly family: 'markdown';
43
+ readonly family: 'markdown' | 'csv';
40
44
  readonly decode: (bytes: Uint8Array<ArrayBuffer>) => string;
41
45
  readonly read: (text: string, options?: UnifiedConversionOptions) => ContentDocument;
42
- readonly build: (content: ContentDocument) => string;
46
+ readonly build: (content: ContentDocument, options?: UnifiedConversionOptions) => string;
43
47
  readonly encode: (text: string) => Uint8Array<ArrayBuffer>;
44
48
  readonly hasSourcePackage: false;
45
49
  }
46
- type FormatNode = PackageFormatNode | MarkdownFormatNode;
50
+ type FormatNode = PackageFormatNode | TextFormatNode;
47
51
  declare const FORMAT_NODES: Readonly<Record<ContentFormat, FormatNode>>;
48
52
  type HopExecutor = 'bridge' | 'toPdf' | 'fromPdf';
49
53
  interface CompositionHop {