@bendyline/squisq-formats 2.0.1 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/LICENSE +21 -0
  2. package/NOTICE.md +20 -0
  3. package/README.md +1 -1
  4. package/dist/{chunk-CRAVSMPZ.js → chunk-26ISNJ7Y.js} +356 -114
  5. package/dist/{chunk-NKAJPJ4G.js → chunk-2JJ5RFDZ.js} +0 -1
  6. package/dist/{chunk-HTW2M27H.js → chunk-3NKXBZSR.js} +193 -42
  7. package/dist/{chunk-U32AG3G3.js → chunk-4V3KCHAP.js} +3 -4
  8. package/dist/{chunk-MLX2BOJC.js → chunk-6RQOV3B3.js} +1 -2
  9. package/dist/{chunk-QRVN6A6E.js → chunk-6S6GU3ZG.js} +5 -6
  10. package/dist/{chunk-FE6OJV6O.js → chunk-7AWFHP5U.js} +1 -1
  11. package/dist/{chunk-RFAPOKHJ.js → chunk-AD2WT564.js} +59 -9
  12. package/dist/{chunk-O3GVVND4.js → chunk-AONELFLA.js} +0 -1
  13. package/dist/{chunk-XKUMNGBW.js → chunk-EJTNGKEA.js} +5 -8
  14. package/dist/chunk-GX7RAUME.js +121 -0
  15. package/dist/{chunk-SSUPBUF5.js → chunk-IIQYS2YH.js} +0 -1
  16. package/dist/{chunk-ABVI556T.js → chunk-IPN56VLW.js} +83 -58
  17. package/dist/{chunk-LXYLOOST.js → chunk-JE6LSIHE.js} +83 -22
  18. package/dist/{chunk-U4MRIFKL.js → chunk-JU2RHXUB.js} +0 -1
  19. package/dist/{chunk-4VUWTSGM.js → chunk-K6XRMVPW.js} +64 -31
  20. package/dist/{chunk-ODL3SSPT.js → chunk-KXOZMWBS.js} +0 -1
  21. package/dist/chunk-OGS5VCGJ.js +446 -0
  22. package/dist/{chunk-GVS2XXV6.js → chunk-PJXJI2LY.js} +449 -57
  23. package/dist/{chunk-2KPARF2P.js → chunk-PU7REGWV.js} +5 -8
  24. package/dist/{chunk-PN52A5AA.js → chunk-SBUW7NHR.js} +0 -1
  25. package/dist/{chunk-VSYHZECT.js → chunk-TAAENIRB.js} +5 -8
  26. package/dist/{chunk-WC7WULGV.js → chunk-X2DEAXNK.js} +62 -2
  27. package/dist/container/index.js +1 -2
  28. package/dist/csv/index.d.ts +27 -2
  29. package/dist/csv/index.js +1 -2
  30. package/dist/docx/index.d.ts +5 -1
  31. package/dist/docx/index.js +9 -10
  32. package/dist/epub/index.d.ts +2 -0
  33. package/dist/epub/index.js +5 -6
  34. package/dist/{export-D2NkylDT.d.ts → export-D9msROJS.d.ts} +18 -6
  35. package/dist/extract-MN7LA3NL.js +13 -0
  36. package/dist/html/index.d.ts +11 -4
  37. package/dist/html/index.js +3 -4
  38. package/dist/images-ESPQKVTW.js +6 -0
  39. package/dist/{import-K8mfc0fz.d.ts → import-C3htUTss.d.ts} +5 -1
  40. package/dist/{import-DTkDxHmZ.d.ts → import-C8whCC7_.d.ts} +6 -0
  41. package/dist/index.d.ts +7 -7
  42. package/dist/index.js +28 -26
  43. package/dist/infer/index.d.ts +3 -3
  44. package/dist/infer/index.js +7 -9
  45. package/dist/{layouts-BHrgZ5FS.d.ts → layouts-CTdPlB-u.d.ts} +1 -1
  46. package/dist/layouts-DRWZGSPD.js +10 -0
  47. package/dist/{mapTheme-IR27S6IV.js → mapTheme-4TWH25FT.js} +1 -2
  48. package/dist/ooxml/index.d.ts +3 -3
  49. package/dist/ooxml/index.js +14 -13
  50. package/dist/pdf/index.d.ts +18 -0
  51. package/dist/pdf/index.js +2 -3
  52. package/dist/pptx/index.d.ts +4 -4
  53. package/dist/pptx/index.js +11 -13
  54. package/dist/{reader-B9L8Ucbj.d.ts → reader-B_m1aKZC.d.ts} +30 -1
  55. package/dist/registry/index.d.ts +21 -5
  56. package/dist/registry/index.js +9 -6
  57. package/dist/{themeReader-DJKErl_j.d.ts → themeReader-DCtwC83Q.d.ts} +1 -1
  58. package/dist/xlsx/index.d.ts +3 -3
  59. package/dist/xlsx/index.js +6 -7
  60. package/package.json +6 -3
  61. package/dist/chunk-2KPARF2P.js.map +0 -1
  62. package/dist/chunk-4VUWTSGM.js.map +0 -1
  63. package/dist/chunk-6M7Z25LA.js +0 -46
  64. package/dist/chunk-6M7Z25LA.js.map +0 -1
  65. package/dist/chunk-ABVI556T.js.map +0 -1
  66. package/dist/chunk-CRAVSMPZ.js.map +0 -1
  67. package/dist/chunk-FE6OJV6O.js.map +0 -1
  68. package/dist/chunk-GVS2XXV6.js.map +0 -1
  69. package/dist/chunk-HTW2M27H.js.map +0 -1
  70. package/dist/chunk-LXYLOOST.js.map +0 -1
  71. package/dist/chunk-MLX2BOJC.js.map +0 -1
  72. package/dist/chunk-NKAJPJ4G.js.map +0 -1
  73. package/dist/chunk-O3GVVND4.js.map +0 -1
  74. package/dist/chunk-ODL3SSPT.js.map +0 -1
  75. package/dist/chunk-PN52A5AA.js.map +0 -1
  76. package/dist/chunk-QRVN6A6E.js.map +0 -1
  77. package/dist/chunk-RFAPOKHJ.js.map +0 -1
  78. package/dist/chunk-SSUPBUF5.js.map +0 -1
  79. package/dist/chunk-U32AG3G3.js.map +0 -1
  80. package/dist/chunk-U4MRIFKL.js.map +0 -1
  81. package/dist/chunk-VJJM2SSH.js +0 -275
  82. package/dist/chunk-VJJM2SSH.js.map +0 -1
  83. package/dist/chunk-VSYHZECT.js.map +0 -1
  84. package/dist/chunk-WC7WULGV.js.map +0 -1
  85. package/dist/chunk-XKUMNGBW.js.map +0 -1
  86. package/dist/chunk-YRT7GQ5Y.js +0 -28
  87. package/dist/chunk-YRT7GQ5Y.js.map +0 -1
  88. package/dist/container/index.js.map +0 -1
  89. package/dist/csv/index.js.map +0 -1
  90. package/dist/docx/index.js.map +0 -1
  91. package/dist/epub/index.js.map +0 -1
  92. package/dist/extract-H6RXJMHP.js +0 -15
  93. package/dist/extract-H6RXJMHP.js.map +0 -1
  94. package/dist/html/index.js.map +0 -1
  95. package/dist/images-7FBWPKE3.js +0 -7
  96. package/dist/images-7FBWPKE3.js.map +0 -1
  97. package/dist/index.js.map +0 -1
  98. package/dist/infer/index.js.map +0 -1
  99. package/dist/layouts-QVPK3ZCU.js +0 -12
  100. package/dist/layouts-QVPK3ZCU.js.map +0 -1
  101. package/dist/mapTheme-IR27S6IV.js.map +0 -1
  102. package/dist/ooxml/index.js.map +0 -1
  103. package/dist/pdf/index.js.map +0 -1
  104. package/dist/pptx/index.js.map +0 -1
  105. package/dist/registry/index.js.map +0 -1
  106. package/dist/xlsx/index.js.map +0 -1
  107. package/src/__tests__/container.test.ts +0 -230
  108. package/src/__tests__/convert.test.ts +0 -495
  109. package/src/__tests__/csvImport.test.ts +0 -84
  110. package/src/__tests__/docxExport.test.ts +0 -457
  111. package/src/__tests__/docxImport.test.ts +0 -410
  112. package/src/__tests__/epub.test.ts +0 -649
  113. package/src/__tests__/exportThemeReconciliation.test.ts +0 -87
  114. package/src/__tests__/formatRegistry.test.ts +0 -174
  115. package/src/__tests__/html.test.ts +0 -435
  116. package/src/__tests__/htmlImport.test.ts +0 -57
  117. package/src/__tests__/inferTheme.test.ts +0 -135
  118. package/src/__tests__/lossyWarnings.test.ts +0 -146
  119. package/src/__tests__/ooxml.test.ts +0 -271
  120. package/src/__tests__/ooxmlCancellation.test.ts +0 -113
  121. package/src/__tests__/ooxmlThemeReader.test.ts +0 -92
  122. package/src/__tests__/pdfExport.test.ts +0 -322
  123. package/src/__tests__/pdfImport.test.ts +0 -384
  124. package/src/__tests__/plainHtml.test.ts +0 -417
  125. package/src/__tests__/plainHtmlBundle.test.ts +0 -253
  126. package/src/__tests__/pptxExport.test.ts +0 -138
  127. package/src/__tests__/pptxImport.test.ts +0 -145
  128. package/src/__tests__/pptxInferFixtures.ts +0 -314
  129. package/src/__tests__/pptxLayoutInfer.test.ts +0 -395
  130. package/src/__tests__/roundTrip.test.ts +0 -201
  131. package/src/__tests__/roundTripAssets.test.ts +0 -50
  132. package/src/__tests__/roundTripMatrix.fixtures.ts +0 -86
  133. package/src/__tests__/roundTripMatrix.helpers.ts +0 -154
  134. package/src/__tests__/roundTripMatrix.test.ts +0 -142
  135. package/src/__tests__/sharedContainer.test.ts +0 -41
  136. package/src/__tests__/sharedImages.test.ts +0 -61
  137. package/src/__tests__/xlsxExport.test.ts +0 -164
  138. package/src/__tests__/xlsxImport.test.ts +0 -80
  139. package/src/__tests__/zipSafety.test.ts +0 -317
  140. package/src/container/index.ts +0 -94
  141. package/src/csv/index.ts +0 -188
  142. package/src/docx/export.ts +0 -1267
  143. package/src/docx/import.ts +0 -995
  144. package/src/docx/index.ts +0 -26
  145. package/src/docx/styles.ts +0 -145
  146. package/src/epub/export.ts +0 -968
  147. package/src/epub/index.ts +0 -20
  148. package/src/html/docsHtmlBundle.ts +0 -373
  149. package/src/html/htmlTemplate.ts +0 -385
  150. package/src/html/imageUtils.ts +0 -61
  151. package/src/html/import.ts +0 -297
  152. package/src/html/index.ts +0 -212
  153. package/src/html/plainHtml.ts +0 -790
  154. package/src/html/plainHtmlBundle.ts +0 -421
  155. package/src/index.ts +0 -109
  156. package/src/infer/extract.ts +0 -127
  157. package/src/infer/index.ts +0 -199
  158. package/src/infer/mapTheme.ts +0 -176
  159. package/src/infer/types.ts +0 -27
  160. package/src/ooxml/index.ts +0 -111
  161. package/src/ooxml/namespaces.ts +0 -196
  162. package/src/ooxml/readUtils.ts +0 -44
  163. package/src/ooxml/reader.ts +0 -318
  164. package/src/ooxml/themeReader.ts +0 -197
  165. package/src/ooxml/types.ts +0 -103
  166. package/src/ooxml/writer.ts +0 -339
  167. package/src/ooxml/xmlUtils.ts +0 -123
  168. package/src/pdf/export.ts +0 -1084
  169. package/src/pdf/import.ts +0 -1164
  170. package/src/pdf/index.ts +0 -29
  171. package/src/pdf/styles.ts +0 -180
  172. package/src/pptx/export.ts +0 -1184
  173. package/src/pptx/import.ts +0 -455
  174. package/src/pptx/index.ts +0 -52
  175. package/src/pptx/layouts.ts +0 -1222
  176. package/src/pptx/styles.ts +0 -96
  177. package/src/pptx/templates.ts +0 -187
  178. package/src/registry/convert.ts +0 -433
  179. package/src/registry/defaultFormats.ts +0 -413
  180. package/src/registry/errors.ts +0 -46
  181. package/src/registry/index.ts +0 -43
  182. package/src/registry/registry.ts +0 -48
  183. package/src/registry/types.ts +0 -170
  184. package/src/shared/boundedZipArchive.ts +0 -383
  185. package/src/shared/container.ts +0 -28
  186. package/src/shared/fidelity.ts +0 -130
  187. package/src/shared/images.ts +0 -44
  188. package/src/shared/inlineRuns.ts +0 -99
  189. package/src/shared/text.ts +0 -41
  190. package/src/shared/zipEntryCount.ts +0 -151
  191. package/src/shared/zipLimits.ts +0 -296
  192. package/src/shared/zipSafety.ts +0 -19
  193. package/src/xlsx/export.ts +0 -253
  194. package/src/xlsx/import.ts +0 -160
  195. package/src/xlsx/index.ts +0 -35
@@ -1,995 +0,0 @@
1
- /**
2
- * DOCX Import
3
- *
4
- * Parses a .docx file (Office Open XML WordprocessingML) and converts
5
- * its content into a squisq MarkdownDocument (or Doc).
6
- *
7
- * Uses JSZip + DOMParser to read the archive and parse the XML — no
8
- * third-party docx library. Handles headings, paragraphs, inline
9
- * formatting (bold, italic, strikethrough), hyperlinks, lists, tables,
10
- * blockquotes, code blocks, images, and footnotes.
11
- *
12
- * @example
13
- * ```ts
14
- * import { docxToMarkdownDoc } from '@bendyline/squisq-formats/docx';
15
- *
16
- * const response = await fetch('document.docx');
17
- * const data = await response.arrayBuffer();
18
- * const doc = await docxToMarkdownDoc(data);
19
- * ```
20
- */
21
-
22
- import type { Doc } from '@bendyline/squisq/schemas';
23
- import { markdownToDoc } from '@bendyline/squisq/doc';
24
- import { stringifyMarkdown } from '@bendyline/squisq/markdown';
25
- import type {
26
- MarkdownDocument,
27
- MarkdownBlockNode,
28
- MarkdownInlineNode,
29
- MarkdownHeading,
30
- MarkdownParagraph,
31
- MarkdownBlockquote,
32
- MarkdownList,
33
- MarkdownListItem,
34
- MarkdownCodeBlock,
35
- MarkdownTable,
36
- MarkdownTableRow,
37
- MarkdownTableCell,
38
- MarkdownText,
39
- MarkdownEmphasis,
40
- MarkdownStrong,
41
- MarkdownStrikethrough,
42
- MarkdownInlineCode,
43
- MarkdownLink,
44
- MarkdownImage,
45
- MarkdownBreak,
46
- MarkdownFootnoteReference,
47
- MarkdownFootnoteDefinition,
48
- } from '@bendyline/squisq/markdown';
49
-
50
- import { openPackage, getPartXml, getPartBinary, getPartRelationships } from '../ooxml/reader.js';
51
- import type { OoxmlOpenOptions } from '../ooxml/reader.js';
52
- import type { OoxmlPackage, Relationship } from '../ooxml/types.js';
53
- import { NS_WML, NS_R } from '../ooxml/namespaces.js';
54
- import type { ContentContainer } from '@bendyline/squisq/storage';
55
- import { buildContainer } from '../shared/container.js';
56
- import { extToMime } from '../shared/images.js';
57
- import {
58
- HEADING_STYLE_MAP,
59
- QUOTE_STYLE_IDS,
60
- CODE_STYLE_IDS,
61
- INLINE_CODE_STYLE_IDS,
62
- BULLET_NUM_FORMATS,
63
- } from './styles.js';
64
-
65
- // ============================================
66
- // Public API
67
- // ============================================
68
-
69
- /**
70
- * Options for DOCX import.
71
- */
72
- export interface DocxImportOptions extends OoxmlOpenOptions {
73
- /**
74
- * Whether to extract embedded images as base64 data URIs.
75
- * When false, images are represented as `[Image]` placeholders.
76
- * Default: false
77
- */
78
- extractImages?: boolean;
79
- }
80
-
81
- /**
82
- * Convert a .docx file to a MarkdownDocument.
83
- *
84
- * @param data - The raw .docx file as ArrayBuffer or Blob
85
- * @param options - Import options
86
- * @returns A MarkdownDocument representing the document content
87
- */
88
- export async function docxToMarkdownDoc(
89
- data: ArrayBuffer | Blob,
90
- options: DocxImportOptions = {},
91
- ): Promise<MarkdownDocument> {
92
- const pkg = await openPackage(data, options);
93
- const ctx = await buildImportContext(pkg, options);
94
-
95
- const documentXml = await getPartXml(pkg, 'word/document.xml');
96
- if (!documentXml) {
97
- return { type: 'document', children: [] };
98
- }
99
-
100
- const body = getFirstElement(documentXml, 'body');
101
- if (!body) {
102
- return { type: 'document', children: [] };
103
- }
104
-
105
- const blocks = await convertBody(body, ctx);
106
-
107
- return { type: 'document', children: blocks };
108
- }
109
-
110
- /**
111
- * Convert a .docx file to a squisq Doc.
112
- *
113
- * Convenience wrapper: DOCX → MarkdownDocument → Doc.
114
- *
115
- * @param data - The raw .docx file as ArrayBuffer or Blob
116
- * @param options - Import options
117
- * @returns A squisq Doc
118
- */
119
- export async function docxToDoc(
120
- data: ArrayBuffer | Blob,
121
- options: DocxImportOptions = {},
122
- ): Promise<Doc> {
123
- const markdownDoc = await docxToMarkdownDoc(data, options);
124
- return markdownToDoc(markdownDoc);
125
- }
126
-
127
- /**
128
- * Convert a .docx file to a ContentContainer with markdown + extracted images.
129
- *
130
- * The container will contain:
131
- * - The primary markdown document (index.md)
132
- * - Any embedded images under images/ (e.g., images/image1.png)
133
- *
134
- * @param data - The raw .docx file as ArrayBuffer or Blob
135
- * @param options - Import options
136
- * @returns A ContentContainer with the document and its media
137
- */
138
- export async function docxToContainer(
139
- data: ArrayBuffer | Blob,
140
- options: DocxImportOptions = {},
141
- ): Promise<ContentContainer> {
142
- const pkg = await openPackage(data, options);
143
- const ctx = await buildImportContext(pkg, { ...options, extractImages: true });
144
-
145
- const documentXml = await getPartXml(pkg, 'word/document.xml');
146
- if (!documentXml) return buildContainer('', []);
147
-
148
- const body = getFirstElement(documentXml, 'body');
149
- if (!body) return buildContainer('', []);
150
-
151
- const blocks = await convertBody(body, ctx);
152
- const markdownDoc: MarkdownDocument = { type: 'document', children: blocks };
153
-
154
- return buildContainer(stringifyMarkdown(markdownDoc), ctx.extractedImages);
155
- }
156
-
157
- // ============================================
158
- // Import Context
159
- // ============================================
160
-
161
- interface ImportContext {
162
- /** Style ID → heading depth mapping (from styles.xml) */
163
- headingStyles: Map<string, number>;
164
- /** Style IDs that represent blockquotes */
165
- quoteStyles: Set<string>;
166
- /** Style IDs that represent code blocks */
167
- codeStyles: Set<string>;
168
- /** Character style IDs that represent inline code */
169
- inlineCodeStyles: Set<string>;
170
- /** Document relationship map: rId → Relationship */
171
- documentRels: Map<string, Relationship>;
172
- /** Numbering definitions: numId → { levels: Map<ilvl, isOrdered> } */
173
- numbering: Map<string, NumberingInfo>;
174
- /** Footnote bodies: footnoteId → Element */
175
- footnotes: Map<string, Element>;
176
- /** Reference to the OOXML package (for extracting images) */
177
- pkg: OoxmlPackage;
178
- /** Import options */
179
- options: DocxImportOptions;
180
- /** Collected image files: relative path → { data, mimeType } */
181
- extractedImages: Map<string, { data: ArrayBuffer; mimeType: string }>;
182
- /** Counter for generating unique image filenames */
183
- imageCounter: number;
184
- }
185
-
186
- interface NumberingInfo {
187
- levels: Map<number, boolean>; // ilvl → isOrdered
188
- }
189
-
190
- async function buildImportContext(
191
- pkg: OoxmlPackage,
192
- options: DocxImportOptions,
193
- ): Promise<ImportContext> {
194
- const ctx: ImportContext = {
195
- headingStyles: new Map(),
196
- quoteStyles: new Set(),
197
- codeStyles: new Set(),
198
- inlineCodeStyles: new Set(),
199
- documentRels: new Map(),
200
- numbering: new Map(),
201
- footnotes: new Map(),
202
- pkg,
203
- options,
204
- extractedImages: new Map(),
205
- imageCounter: 0,
206
- };
207
-
208
- // Initialize with built-in defaults
209
- for (const [id, depth] of Object.entries(HEADING_STYLE_MAP)) {
210
- ctx.headingStyles.set(id, depth);
211
- }
212
- for (const id of QUOTE_STYLE_IDS) {
213
- ctx.quoteStyles.add(id);
214
- }
215
- for (const id of CODE_STYLE_IDS) {
216
- ctx.codeStyles.add(id);
217
- }
218
- for (const id of INLINE_CODE_STYLE_IDS) {
219
- ctx.inlineCodeStyles.add(id);
220
- }
221
-
222
- // Parse styles.xml for custom heading/quote/code mappings
223
- await parseStyles(pkg, ctx);
224
-
225
- // Parse document relationships
226
- const rels = await getPartRelationships(pkg, 'word/document.xml');
227
- for (const rel of rels) {
228
- ctx.documentRels.set(rel.id, rel);
229
- }
230
-
231
- // Parse numbering.xml
232
- await parseNumbering(pkg, ctx);
233
-
234
- // Parse footnotes.xml
235
- await parseFootnotes(pkg, ctx);
236
-
237
- return ctx;
238
- }
239
-
240
- // ============================================
241
- // Styles Parsing
242
- // ============================================
243
-
244
- async function parseStyles(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
245
- const doc = await getPartXml(pkg, 'word/styles.xml');
246
- if (!doc) return;
247
-
248
- const styles = doc.getElementsByTagNameNS(NS_WML, 'style');
249
- // Fallback for documents that don't use namespace prefixes properly
250
- const stylesList = styles.length > 0 ? styles : doc.getElementsByTagName('style');
251
-
252
- for (let i = 0; i < stylesList.length; i++) {
253
- const style = stylesList[i];
254
- const styleId = style.getAttributeNS(NS_WML, 'styleId') ?? style.getAttribute('w:styleId');
255
- if (!styleId) continue;
256
-
257
- const nameEl = getFirstChildElement(style, 'name');
258
- const styleName = nameEl?.getAttributeNS(NS_WML, 'val') ?? nameEl?.getAttribute('w:val') ?? '';
259
-
260
- // Check if this is a heading style by name
261
- const headingMatch = styleName.match(/^heading\s+(\d+)$/i);
262
- if (headingMatch) {
263
- const depth = parseInt(headingMatch[1], 10);
264
- if (depth >= 1 && depth <= 6) {
265
- ctx.headingStyles.set(styleId, depth);
266
- }
267
- }
268
-
269
- // Check pPr > outlineLvl for heading detection
270
- const pPr = getFirstChildElement(style, 'pPr');
271
- if (pPr) {
272
- const outlineLvl = getFirstChildElement(pPr, 'outlineLvl');
273
- if (outlineLvl) {
274
- const val = outlineLvl.getAttributeNS(NS_WML, 'val') ?? outlineLvl.getAttribute('w:val');
275
- if (val !== null) {
276
- const depth = parseInt(val, 10) + 1;
277
- if (depth >= 1 && depth <= 6) {
278
- ctx.headingStyles.set(styleId, depth);
279
- }
280
- }
281
- }
282
- }
283
- }
284
- }
285
-
286
- // ============================================
287
- // Numbering Parsing
288
- // ============================================
289
-
290
- async function parseNumbering(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
291
- const doc = await getPartXml(pkg, 'word/numbering.xml');
292
- if (!doc) return;
293
-
294
- // Parse abstract numbering definitions
295
- const abstractNums = new Map<string, Map<number, boolean>>(); // abstractNumId → levels(ilvl → isOrdered)
296
-
297
- const abstractNumEls = getAllElements(doc, 'abstractNum');
298
- for (const absNum of abstractNumEls) {
299
- const absId = getAttr(absNum, 'abstractNumId');
300
- if (!absId) continue;
301
-
302
- const levels = new Map<number, boolean>();
303
- const lvlEls = getAllChildElements(absNum, 'lvl');
304
- for (const lvl of lvlEls) {
305
- const ilvlStr = getAttr(lvl, 'ilvl');
306
- if (ilvlStr === null) continue;
307
- const ilvl = parseInt(ilvlStr, 10);
308
-
309
- const numFmtEl = getFirstChildElement(lvl, 'numFmt');
310
- const numFmt = numFmtEl ? getAttr(numFmtEl, 'val') : null;
311
-
312
- const isOrdered = numFmt !== null && !BULLET_NUM_FORMATS.has(numFmt);
313
- levels.set(ilvl, isOrdered);
314
- }
315
-
316
- abstractNums.set(absId, levels);
317
- }
318
-
319
- // Parse concrete num → abstractNum mappings
320
- const numEls = getAllElements(doc, 'num');
321
- for (const num of numEls) {
322
- const numId = getAttr(num, 'numId');
323
- if (!numId) continue;
324
-
325
- const abstractNumIdEl = getFirstChildElement(num, 'abstractNumId');
326
- const absId = abstractNumIdEl ? getAttr(abstractNumIdEl, 'val') : null;
327
- if (!absId) continue;
328
-
329
- const levels = abstractNums.get(absId);
330
- if (levels) {
331
- ctx.numbering.set(numId, { levels });
332
- }
333
- }
334
- }
335
-
336
- // ============================================
337
- // Footnotes Parsing
338
- // ============================================
339
-
340
- async function parseFootnotes(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
341
- const doc = await getPartXml(pkg, 'word/footnotes.xml');
342
- if (!doc) return;
343
-
344
- const footnoteEls = getAllElements(doc, 'footnote');
345
- for (const fn of footnoteEls) {
346
- const id = getAttr(fn, 'id');
347
- const type = getAttr(fn, 'type');
348
- // Skip separator and continuation separator footnotes
349
- if (!id || type === 'separator' || type === 'continuationSeparator') continue;
350
- ctx.footnotes.set(id, fn);
351
- }
352
- }
353
-
354
- // ============================================
355
- // Body Conversion
356
- // ============================================
357
-
358
- async function convertBody(body: Element, ctx: ImportContext): Promise<MarkdownBlockNode[]> {
359
- const result: MarkdownBlockNode[] = [];
360
- const children = Array.from(body.children);
361
-
362
- let i = 0;
363
- while (i < children.length) {
364
- const el = children[i];
365
- const localName = el.localName;
366
-
367
- if (localName === 'p') {
368
- // Check if this is part of a list
369
- const numPr = getNumPr(el);
370
- if (numPr) {
371
- // Collect consecutive list paragraphs
372
- const { node, consumed } = await collectList(children, i, ctx);
373
- result.push(node);
374
- i += consumed;
375
- continue;
376
- }
377
-
378
- const block = await convertParagraph(el, ctx);
379
- if (block) {
380
- result.push(block);
381
- }
382
- i++;
383
- } else if (localName === 'tbl') {
384
- const table = await convertTable(el, ctx);
385
- if (table) result.push(table);
386
- i++;
387
- } else {
388
- // Skip unknown elements (sectPr, bookmarkStart, etc.)
389
- i++;
390
- }
391
- }
392
-
393
- // Append footnote definitions at the end
394
- const footnoteNodes = await convertFootnoteDefinitions(ctx);
395
- result.push(...footnoteNodes);
396
-
397
- return result;
398
- }
399
-
400
- // ============================================
401
- // Paragraph Conversion
402
- // ============================================
403
-
404
- async function convertParagraph(
405
- el: Element,
406
- ctx: ImportContext,
407
- ): Promise<MarkdownBlockNode | null> {
408
- const pPr = getFirstChildElement(el, 'pPr');
409
- const styleId = getParagraphStyleId(pPr);
410
-
411
- // Check for heading
412
- if (styleId && ctx.headingStyles.has(styleId)) {
413
- const depth = ctx.headingStyles.get(styleId)!;
414
- const inlines = await convertRuns(el, ctx);
415
- if (inlines.length === 0) return null;
416
- return {
417
- type: 'heading',
418
- depth: Math.min(Math.max(depth, 1), 6) as 1 | 2 | 3 | 4 | 5 | 6,
419
- children: inlines,
420
- } satisfies MarkdownHeading;
421
- }
422
-
423
- // Check for blockquote
424
- if (styleId && ctx.quoteStyles.has(styleId)) {
425
- const inlines = await convertRuns(el, ctx);
426
- if (inlines.length === 0) return null;
427
- const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
428
- return { type: 'blockquote', children: [paragraph] } satisfies MarkdownBlockquote;
429
- }
430
-
431
- // Check for code block
432
- if (styleId && ctx.codeStyles.has(styleId)) {
433
- const text = getElementTextContent(el);
434
- return { type: 'code', value: text } satisfies MarkdownCodeBlock;
435
- }
436
-
437
- // Regular paragraph
438
- const inlines = await convertRuns(el, ctx);
439
- if (inlines.length === 0) return null;
440
- return { type: 'paragraph', children: inlines } satisfies MarkdownParagraph;
441
- }
442
-
443
- // ============================================
444
- // Run (Inline) Conversion
445
- // ============================================
446
-
447
- async function convertRuns(
448
- paragraphEl: Element,
449
- ctx: ImportContext,
450
- ): Promise<MarkdownInlineNode[]> {
451
- const result: MarkdownInlineNode[] = [];
452
- const children = Array.from(paragraphEl.children);
453
-
454
- for (const child of children) {
455
- const localName = child.localName;
456
-
457
- if (localName === 'r') {
458
- const inlines = await convertRun(child, ctx);
459
- result.push(...inlines);
460
- } else if (localName === 'hyperlink') {
461
- const link = await convertHyperlink(child, ctx);
462
- if (link) result.push(link);
463
- }
464
- // Skip pPr, bookmarkStart, bookmarkEnd, etc.
465
- }
466
-
467
- return mergeAdjacentText(result);
468
- }
469
-
470
- async function convertRun(runEl: Element, ctx: ImportContext): Promise<MarkdownInlineNode[]> {
471
- const result: MarkdownInlineNode[] = [];
472
- const rPr = getFirstChildElement(runEl, 'rPr');
473
- const format = parseRunFormat(rPr, ctx);
474
-
475
- for (const child of Array.from(runEl.children)) {
476
- const localName = child.localName;
477
-
478
- if (localName === 't') {
479
- const text = child.textContent ?? '';
480
- if (!text) continue;
481
-
482
- if (format.code) {
483
- result.push({ type: 'inlineCode', value: text } satisfies MarkdownInlineCode);
484
- } else {
485
- let node: MarkdownInlineNode = { type: 'text', value: text } satisfies MarkdownText;
486
- if (format.strike) {
487
- node = { type: 'delete', children: [node] } satisfies MarkdownStrikethrough;
488
- }
489
- if (format.italic) {
490
- node = { type: 'emphasis', children: [node] } satisfies MarkdownEmphasis;
491
- }
492
- if (format.bold) {
493
- node = { type: 'strong', children: [node] } satisfies MarkdownStrong;
494
- }
495
- result.push(node);
496
- }
497
- } else if (localName === 'br') {
498
- result.push({ type: 'break' } satisfies MarkdownBreak);
499
- } else if (localName === 'footnoteReference') {
500
- const fnId = getAttr(child, 'id');
501
- if (fnId && fnId !== '0' && fnId !== '-1') {
502
- result.push({
503
- type: 'footnoteReference',
504
- identifier: `fn${fnId}`,
505
- } satisfies MarkdownFootnoteReference);
506
- }
507
- } else if (localName === 'drawing' || localName === 'pict') {
508
- const img = await extractImage(child, ctx);
509
- if (img) result.push(img);
510
- }
511
- }
512
-
513
- return result;
514
- }
515
-
516
- interface RunFormat {
517
- bold: boolean;
518
- italic: boolean;
519
- strike: boolean;
520
- code: boolean;
521
- }
522
-
523
- function parseRunFormat(rPr: Element | null, ctx: ImportContext): RunFormat {
524
- if (!rPr) return { bold: false, italic: false, strike: false, code: false };
525
-
526
- const bold = hasChildElement(rPr, 'b') && !isFalseToggle(getFirstChildElement(rPr, 'b')!);
527
- const italic = hasChildElement(rPr, 'i') && !isFalseToggle(getFirstChildElement(rPr, 'i')!);
528
- const strike =
529
- hasChildElement(rPr, 'strike') && !isFalseToggle(getFirstChildElement(rPr, 'strike')!);
530
-
531
- // Check for inline code via character style
532
- const rStyle = getFirstChildElement(rPr, 'rStyle');
533
- const charStyleId = rStyle ? getAttr(rStyle, 'val') : null;
534
- const isCodeStyle = charStyleId ? ctx.inlineCodeStyles.has(charStyleId) : false;
535
-
536
- // Check for monospace font as a code indicator
537
- const rFonts = getFirstChildElement(rPr, 'rFonts');
538
- const fontName = rFonts ? (getAttr(rFonts, 'ascii') ?? getAttr(rFonts, 'hAnsi') ?? '') : '';
539
- const isMonospace = /consolas|courier|mono/i.test(fontName);
540
-
541
- return { bold, italic, strike, code: isCodeStyle || isMonospace };
542
- }
543
-
544
- function isFalseToggle(el: Element): boolean {
545
- const val = getAttr(el, 'val');
546
- return val === '0' || val === 'false';
547
- }
548
-
549
- // ============================================
550
- // Hyperlink Conversion
551
- // ============================================
552
-
553
- async function convertHyperlink(el: Element, ctx: ImportContext): Promise<MarkdownLink | null> {
554
- const rId = el.getAttributeNS(NS_R, 'id') ?? el.getAttribute('r:id');
555
-
556
- let url = '';
557
- if (rId) {
558
- const rel = ctx.documentRels.get(rId);
559
- if (rel) url = rel.target;
560
- }
561
-
562
- // Also check for w:anchor (internal bookmarks)
563
- if (!url) {
564
- const anchor = el.getAttributeNS(NS_WML, 'anchor') ?? el.getAttribute('w:anchor');
565
- if (anchor) url = `#${anchor}`;
566
- }
567
-
568
- const inlines: MarkdownInlineNode[] = [];
569
- for (const child of Array.from(el.children)) {
570
- if (child.localName === 'r') {
571
- inlines.push(...(await convertRun(child, ctx)));
572
- }
573
- }
574
-
575
- if (inlines.length === 0) return null;
576
-
577
- return {
578
- type: 'link',
579
- url,
580
- children: mergeAdjacentText(inlines),
581
- };
582
- }
583
-
584
- // ============================================
585
- // Image Extraction
586
- // ============================================
587
-
588
- async function extractImage(el: Element, ctx: ImportContext): Promise<MarkdownImage | null> {
589
- // Find <a:blip r:embed="rIdX"/> anywhere in the drawing tree
590
- const blip = findDescendant(el, 'blip');
591
- if (!blip) {
592
- return { type: 'image', url: '', alt: 'Image' };
593
- }
594
-
595
- const rId = blip.getAttributeNS(NS_R, 'embed') ?? blip.getAttribute('r:embed');
596
- if (!rId) {
597
- return { type: 'image', url: '', alt: 'Image' };
598
- }
599
-
600
- const rel = ctx.documentRels.get(rId);
601
- if (!rel) {
602
- return { type: 'image', url: '', alt: 'Image' };
603
- }
604
-
605
- // Resolve the target path relative to word/
606
- const target = rel.target.startsWith('/') ? rel.target.slice(1) : `word/${rel.target}`;
607
-
608
- // Extract binary data from the zip
609
- const data = await getPartBinary(ctx.pkg, target);
610
- if (!data) {
611
- return { type: 'image', url: '', alt: 'Image' };
612
- }
613
-
614
- // Determine extension and MIME type
615
- const dot = target.lastIndexOf('.');
616
- const ext = dot !== -1 ? target.slice(dot).toLowerCase() : '.png';
617
- const mimeType = extToMime(ext);
618
-
619
- // Generate a unique image path
620
- ctx.imageCounter++;
621
- const imagePath = `images/image${ctx.imageCounter}${ext}`;
622
-
623
- // Store the extracted image data
624
- ctx.extractedImages.set(imagePath, { data, mimeType });
625
-
626
- // Try to extract alt text from the drawing's docPr element
627
- const docPr = findDescendant(el, 'docPr');
628
- const alt = docPr?.getAttribute('descr') || docPr?.getAttribute('title') || 'Image';
629
-
630
- return {
631
- type: 'image',
632
- url: imagePath,
633
- alt,
634
- };
635
- }
636
-
637
- /** Recursively find the first descendant element with the given local name. */
638
- function findDescendant(el: Element, localName: string): Element | null {
639
- for (const child of Array.from(el.children)) {
640
- if (child.localName === localName) return child;
641
- const found = findDescendant(child, localName);
642
- if (found) return found;
643
- }
644
- return null;
645
- }
646
-
647
- // ============================================
648
- // List Collection
649
- // ============================================
650
-
651
- interface ListResult {
652
- node: MarkdownList;
653
- consumed: number;
654
- }
655
-
656
- interface NumPrInfo {
657
- numId: string;
658
- ilvl: number;
659
- }
660
-
661
- async function collectList(
662
- elements: Element[],
663
- startIdx: number,
664
- ctx: ImportContext,
665
- ): Promise<ListResult> {
666
- const firstNumPr = getNumPr(elements[startIdx])!;
667
- const { numId } = firstNumPr;
668
-
669
- // Determine if ordered from numbering definition
670
- const numInfo = ctx.numbering.get(numId);
671
- const isOrdered = numInfo?.levels.get(0) ?? false;
672
-
673
- const items: MarkdownListItem[] = [];
674
- let consumed = 0;
675
-
676
- let i = startIdx;
677
- while (i < elements.length) {
678
- const el = elements[i];
679
- if (el.localName !== 'p') break;
680
-
681
- const numPr = getNumPr(el);
682
- if (!numPr || numPr.numId !== numId) break;
683
-
684
- // Convert this paragraph's inline content
685
- const inlines = await convertRuns(el, ctx);
686
- if (inlines.length > 0) {
687
- const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
688
-
689
- // Check if this is a nested list item
690
- if (numPr.ilvl > firstNumPr.ilvl) {
691
- // Collect nested list items
692
- const nested = await collectNestedList(elements, i, ctx, firstNumPr.ilvl);
693
- if (items.length > 0) {
694
- // Attach nested list to the last item
695
- const lastItem = items[items.length - 1];
696
- const nestedIsOrdered = numInfo?.levels.get(numPr.ilvl) ?? false;
697
- const nestedList: MarkdownList = {
698
- type: 'list',
699
- ordered: nestedIsOrdered,
700
- children: nested.items,
701
- };
702
- lastItem.children.push(nestedList);
703
- }
704
- i += nested.consumed;
705
- consumed += nested.consumed;
706
- continue;
707
- }
708
-
709
- items.push({
710
- type: 'listItem',
711
- children: [paragraph],
712
- });
713
- }
714
-
715
- i++;
716
- consumed++;
717
- }
718
-
719
- return {
720
- node: {
721
- type: 'list',
722
- ordered: isOrdered,
723
- children: items,
724
- },
725
- consumed,
726
- };
727
- }
728
-
729
- interface NestedListResult {
730
- items: MarkdownListItem[];
731
- consumed: number;
732
- }
733
-
734
- async function collectNestedList(
735
- elements: Element[],
736
- startIdx: number,
737
- ctx: ImportContext,
738
- parentIlvl: number,
739
- ): Promise<NestedListResult> {
740
- const items: MarkdownListItem[] = [];
741
- let consumed = 0;
742
-
743
- let i = startIdx;
744
- while (i < elements.length) {
745
- const el = elements[i];
746
- if (el.localName !== 'p') break;
747
-
748
- const numPr = getNumPr(el);
749
- if (!numPr) break;
750
- if (numPr.ilvl <= parentIlvl) break;
751
-
752
- const inlines = await convertRuns(el, ctx);
753
- if (inlines.length > 0) {
754
- const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
755
- items.push({ type: 'listItem', children: [paragraph] });
756
- }
757
-
758
- i++;
759
- consumed++;
760
- }
761
-
762
- return { items, consumed };
763
- }
764
-
765
- function getNumPr(el: Element): NumPrInfo | null {
766
- const pPr = getFirstChildElement(el, 'pPr');
767
- if (!pPr) return null;
768
-
769
- const numPr = getFirstChildElement(pPr, 'numPr');
770
- if (!numPr) return null;
771
-
772
- const ilvlEl = getFirstChildElement(numPr, 'ilvl');
773
- const numIdEl = getFirstChildElement(numPr, 'numId');
774
-
775
- const ilvlVal = ilvlEl ? getAttr(ilvlEl, 'val') : null;
776
- const numIdVal = numIdEl ? getAttr(numIdEl, 'val') : null;
777
-
778
- if (!numIdVal || numIdVal === '0') return null; // numId 0 means "no list"
779
-
780
- return {
781
- numId: numIdVal,
782
- ilvl: ilvlVal ? parseInt(ilvlVal, 10) : 0,
783
- };
784
- }
785
-
786
- // ============================================
787
- // Table Conversion
788
- // ============================================
789
-
790
- async function convertTable(tblEl: Element, ctx: ImportContext): Promise<MarkdownTable | null> {
791
- const rows: MarkdownTableRow[] = [];
792
-
793
- const trEls = getAllChildElements(tblEl, 'tr');
794
- for (let ri = 0; ri < trEls.length; ri++) {
795
- const row = await convertTableRow(trEls[ri], ctx, ri === 0);
796
- rows.push(row);
797
- }
798
-
799
- if (rows.length === 0) return null;
800
-
801
- // If first row isn't explicitly a header, treat it as one anyway
802
- // (Markdown tables require a header row)
803
- return {
804
- type: 'table',
805
- children: rows,
806
- };
807
- }
808
-
809
- async function convertTableRow(
810
- trEl: Element,
811
- ctx: ImportContext,
812
- isHeader: boolean,
813
- ): Promise<MarkdownTableRow> {
814
- const cells: MarkdownTableCell[] = [];
815
- const tcEls = getAllChildElements(trEl, 'tc');
816
-
817
- for (const tc of tcEls) {
818
- const cell = await convertTableCell(tc, ctx, isHeader);
819
- cells.push(cell);
820
- }
821
-
822
- return { type: 'tableRow', children: cells };
823
- }
824
-
825
- async function convertTableCell(
826
- tcEl: Element,
827
- ctx: ImportContext,
828
- isHeader: boolean,
829
- ): Promise<MarkdownTableCell> {
830
- const inlines: MarkdownInlineNode[] = [];
831
-
832
- // A cell can contain multiple paragraphs; flatten them with breaks
833
- const paragraphs = getAllChildElements(tcEl, 'p');
834
- for (let pi = 0; pi < paragraphs.length; pi++) {
835
- if (pi > 0) {
836
- inlines.push({ type: 'break' } as MarkdownBreak);
837
- }
838
- const runs = await convertRuns(paragraphs[pi], ctx);
839
- inlines.push(...runs);
840
- }
841
-
842
- return {
843
- type: 'tableCell',
844
- isHeader,
845
- children: mergeAdjacentText(inlines),
846
- };
847
- }
848
-
849
- // ============================================
850
- // Footnote Definition Conversion
851
- // ============================================
852
-
853
- async function convertFootnoteDefinitions(
854
- ctx: ImportContext,
855
- ): Promise<MarkdownFootnoteDefinition[]> {
856
- const results: MarkdownFootnoteDefinition[] = [];
857
-
858
- for (const [id, el] of ctx.footnotes) {
859
- const children: MarkdownBlockNode[] = [];
860
- const paragraphs = getAllChildElements(el, 'p');
861
-
862
- for (const p of paragraphs) {
863
- const inlines = await convertRuns(p, ctx);
864
- if (inlines.length > 0) {
865
- children.push({
866
- type: 'paragraph',
867
- children: inlines,
868
- } satisfies MarkdownParagraph);
869
- }
870
- }
871
-
872
- if (children.length > 0) {
873
- results.push({
874
- type: 'footnoteDefinition',
875
- identifier: `fn${id}`,
876
- children,
877
- });
878
- }
879
- }
880
-
881
- return results;
882
- }
883
-
884
- // ============================================
885
- // XML Helper Utilities
886
- // ============================================
887
-
888
- /**
889
- * Get the first element child with a given local name.
890
- * Handles both namespaced and non-namespaced elements.
891
- */
892
- function getFirstChildElement(parent: Element | Document, localName: string): Element | null {
893
- for (const child of Array.from(parent.children ?? [])) {
894
- if (child.localName === localName) return child;
895
- }
896
- return null;
897
- }
898
-
899
- /**
900
- * Get all direct child elements with a given local name.
901
- */
902
- function getAllChildElements(parent: Element, localName: string): Element[] {
903
- const result: Element[] = [];
904
- for (const child of Array.from(parent.children)) {
905
- if (child.localName === localName) result.push(child);
906
- }
907
- return result;
908
- }
909
-
910
- /**
911
- * Get all elements with a given local name in the document.
912
- */
913
- function getAllElements(doc: Document | Element, localName: string): Element[] {
914
- // Try namespace-aware first
915
- const nsEls =
916
- 'getElementsByTagNameNS' in doc ? doc.getElementsByTagNameNS(NS_WML, localName) : null;
917
- if (nsEls && nsEls.length > 0) return Array.from(nsEls);
918
-
919
- // Fallback: try with w: prefix
920
- const prefixed = doc.getElementsByTagName(`w:${localName}`);
921
- if (prefixed.length > 0) return Array.from(prefixed);
922
-
923
- // Final fallback: bare name
924
- return Array.from(doc.getElementsByTagName(localName));
925
- }
926
-
927
- /**
928
- * Get the first element with given local name in the document or subtree.
929
- */
930
- function getFirstElement(doc: Document | Element, localName: string): Element | null {
931
- const els = getAllElements(doc, localName);
932
- return els.length > 0 ? els[0] : null;
933
- }
934
-
935
- /**
936
- * Get a w:-prefixed attribute, trying namespace-aware first then fallback.
937
- */
938
- function getAttr(el: Element, localName: string): string | null {
939
- return el.getAttributeNS(NS_WML, localName) || el.getAttribute(`w:${localName}`) || null;
940
- }
941
-
942
- /**
943
- * Check if an element has a direct child with the given local name.
944
- */
945
- function hasChildElement(parent: Element, localName: string): boolean {
946
- return getFirstChildElement(parent, localName) !== null;
947
- }
948
-
949
- /**
950
- * Get the paragraph style ID from a pPr element.
951
- */
952
- function getParagraphStyleId(pPr: Element | null): string | null {
953
- if (!pPr) return null;
954
- const pStyle = getFirstChildElement(pPr, 'pStyle');
955
- if (!pStyle) return null;
956
- return getAttr(pStyle, 'val');
957
- }
958
-
959
- /**
960
- * Get all text content from an element (concatenating all w:t descendants).
961
- */
962
- function getElementTextContent(el: Element): string {
963
- const parts: string[] = [];
964
-
965
- function walk(node: Element): void {
966
- if (node.localName === 't') {
967
- parts.push(node.textContent ?? '');
968
- }
969
- for (const child of Array.from(node.children)) {
970
- walk(child);
971
- }
972
- }
973
-
974
- walk(el);
975
- return parts.join('');
976
- }
977
-
978
- /**
979
- * Merge adjacent MarkdownText nodes to reduce fragmentation.
980
- */
981
- function mergeAdjacentText(nodes: MarkdownInlineNode[]): MarkdownInlineNode[] {
982
- if (nodes.length <= 1) return nodes;
983
-
984
- const result: MarkdownInlineNode[] = [];
985
- for (const node of nodes) {
986
- const prev = result[result.length - 1];
987
- if (node.type === 'text' && prev?.type === 'text') {
988
- // Merge into previous text node
989
- (prev as MarkdownText).value += (node as MarkdownText).value;
990
- } else {
991
- result.push(node);
992
- }
993
- }
994
- return result;
995
- }