@bendyline/squisq-formats 2.1.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/LICENSE +21 -0
  2. package/NOTICE.md +20 -0
  3. package/README.md +1 -1
  4. package/dist/{chunk-NNHKUXKA.js → chunk-26ISNJ7Y.js} +85 -65
  5. package/dist/{chunk-NKAJPJ4G.js → chunk-2JJ5RFDZ.js} +0 -1
  6. package/dist/{chunk-WQSHGBLN.js → chunk-3NKXBZSR.js} +193 -42
  7. package/dist/{chunk-KURGXM4I.js → chunk-4V3KCHAP.js} +3 -4
  8. package/dist/{chunk-MLX2BOJC.js → chunk-6RQOV3B3.js} +1 -2
  9. package/dist/{chunk-EW54IRRS.js → chunk-6S6GU3ZG.js} +5 -6
  10. package/dist/{chunk-FE6OJV6O.js → chunk-7AWFHP5U.js} +1 -1
  11. package/dist/{chunk-RFAPOKHJ.js → chunk-AD2WT564.js} +59 -9
  12. package/dist/{chunk-O3GVVND4.js → chunk-AONELFLA.js} +0 -1
  13. package/dist/{chunk-SC67HYQJ.js → chunk-EJTNGKEA.js} +5 -8
  14. package/dist/chunk-GX7RAUME.js +121 -0
  15. package/dist/{chunk-SSUPBUF5.js → chunk-IIQYS2YH.js} +0 -1
  16. package/dist/{chunk-RLU7UFYU.js → chunk-IPN56VLW.js} +83 -58
  17. package/dist/{chunk-DTDF6QDP.js → chunk-JE6LSIHE.js} +81 -20
  18. package/dist/{chunk-U4MRIFKL.js → chunk-JU2RHXUB.js} +0 -1
  19. package/dist/{chunk-4VUWTSGM.js → chunk-K6XRMVPW.js} +64 -31
  20. package/dist/{chunk-ODL3SSPT.js → chunk-KXOZMWBS.js} +0 -1
  21. package/dist/chunk-OGS5VCGJ.js +446 -0
  22. package/dist/{chunk-GVS2XXV6.js → chunk-PJXJI2LY.js} +449 -57
  23. package/dist/{chunk-74GO3FVS.js → chunk-PU7REGWV.js} +5 -8
  24. package/dist/{chunk-PN52A5AA.js → chunk-SBUW7NHR.js} +0 -1
  25. package/dist/{chunk-QFLDYKCR.js → chunk-TAAENIRB.js} +5 -8
  26. package/dist/{chunk-7ARKUCQT.js → chunk-X2DEAXNK.js} +54 -2
  27. package/dist/container/index.js +1 -2
  28. package/dist/csv/index.d.ts +27 -2
  29. package/dist/csv/index.js +1 -2
  30. package/dist/docx/index.d.ts +5 -1
  31. package/dist/docx/index.js +9 -11
  32. package/dist/epub/index.d.ts +2 -0
  33. package/dist/epub/index.js +5 -6
  34. package/dist/{export-D2NkylDT.d.ts → export-D9msROJS.d.ts} +18 -6
  35. package/dist/extract-MN7LA3NL.js +13 -0
  36. package/dist/html/index.d.ts +11 -4
  37. package/dist/html/index.js +3 -4
  38. package/dist/images-ESPQKVTW.js +6 -0
  39. package/dist/{import-K8mfc0fz.d.ts → import-C3htUTss.d.ts} +5 -1
  40. package/dist/{import-DTkDxHmZ.d.ts → import-C8whCC7_.d.ts} +6 -0
  41. package/dist/index.d.ts +7 -7
  42. package/dist/index.js +28 -26
  43. package/dist/infer/index.d.ts +3 -3
  44. package/dist/infer/index.js +7 -9
  45. package/dist/{layouts-BHrgZ5FS.d.ts → layouts-CTdPlB-u.d.ts} +1 -1
  46. package/dist/layouts-DRWZGSPD.js +10 -0
  47. package/dist/{mapTheme-IR27S6IV.js → mapTheme-4TWH25FT.js} +1 -2
  48. package/dist/ooxml/index.d.ts +3 -3
  49. package/dist/ooxml/index.js +14 -13
  50. package/dist/pdf/index.d.ts +18 -0
  51. package/dist/pdf/index.js +2 -3
  52. package/dist/pptx/index.d.ts +4 -4
  53. package/dist/pptx/index.js +11 -13
  54. package/dist/{reader-B9L8Ucbj.d.ts → reader-B_m1aKZC.d.ts} +30 -1
  55. package/dist/registry/index.d.ts +21 -5
  56. package/dist/registry/index.js +9 -6
  57. package/dist/{themeReader-DJKErl_j.d.ts → themeReader-DCtwC83Q.d.ts} +1 -1
  58. package/dist/xlsx/index.d.ts +3 -3
  59. package/dist/xlsx/index.js +6 -7
  60. package/package.json +6 -3
  61. package/dist/chunk-4VUWTSGM.js.map +0 -1
  62. package/dist/chunk-6M7Z25LA.js +0 -46
  63. package/dist/chunk-6M7Z25LA.js.map +0 -1
  64. package/dist/chunk-74GO3FVS.js.map +0 -1
  65. package/dist/chunk-7ARKUCQT.js.map +0 -1
  66. package/dist/chunk-DTDF6QDP.js.map +0 -1
  67. package/dist/chunk-EW54IRRS.js.map +0 -1
  68. package/dist/chunk-FE6OJV6O.js.map +0 -1
  69. package/dist/chunk-GVS2XXV6.js.map +0 -1
  70. package/dist/chunk-KURGXM4I.js.map +0 -1
  71. package/dist/chunk-MLX2BOJC.js.map +0 -1
  72. package/dist/chunk-NKAJPJ4G.js.map +0 -1
  73. package/dist/chunk-NNHKUXKA.js.map +0 -1
  74. package/dist/chunk-O3GVVND4.js.map +0 -1
  75. package/dist/chunk-ODL3SSPT.js.map +0 -1
  76. package/dist/chunk-PN52A5AA.js.map +0 -1
  77. package/dist/chunk-QFLDYKCR.js.map +0 -1
  78. package/dist/chunk-RFAPOKHJ.js.map +0 -1
  79. package/dist/chunk-RLU7UFYU.js.map +0 -1
  80. package/dist/chunk-SC67HYQJ.js.map +0 -1
  81. package/dist/chunk-SSUPBUF5.js.map +0 -1
  82. package/dist/chunk-U4MRIFKL.js.map +0 -1
  83. package/dist/chunk-UGYF5AZE.js +0 -275
  84. package/dist/chunk-UGYF5AZE.js.map +0 -1
  85. package/dist/chunk-WQSHGBLN.js.map +0 -1
  86. package/dist/chunk-YRT7GQ5Y.js +0 -28
  87. package/dist/chunk-YRT7GQ5Y.js.map +0 -1
  88. package/dist/container/index.js.map +0 -1
  89. package/dist/csv/index.js.map +0 -1
  90. package/dist/docx/index.js.map +0 -1
  91. package/dist/epub/index.js.map +0 -1
  92. package/dist/extract-OJ7ZQV6P.js +0 -15
  93. package/dist/extract-OJ7ZQV6P.js.map +0 -1
  94. package/dist/html/index.js.map +0 -1
  95. package/dist/images-7FBWPKE3.js +0 -7
  96. package/dist/images-7FBWPKE3.js.map +0 -1
  97. package/dist/index.js.map +0 -1
  98. package/dist/infer/index.js.map +0 -1
  99. package/dist/layouts-5VDIRPIJ.js +0 -12
  100. package/dist/layouts-5VDIRPIJ.js.map +0 -1
  101. package/dist/mapTheme-IR27S6IV.js.map +0 -1
  102. package/dist/ooxml/index.js.map +0 -1
  103. package/dist/pdf/index.js.map +0 -1
  104. package/dist/pptx/index.js.map +0 -1
  105. package/dist/registry/index.js.map +0 -1
  106. package/dist/xlsx/index.js.map +0 -1
  107. package/src/__tests__/container.test.ts +0 -230
  108. package/src/__tests__/convert.test.ts +0 -495
  109. package/src/__tests__/csvImport.test.ts +0 -84
  110. package/src/__tests__/docxExport.test.ts +0 -491
  111. package/src/__tests__/docxImport.test.ts +0 -531
  112. package/src/__tests__/epub.test.ts +0 -649
  113. package/src/__tests__/exportThemeReconciliation.test.ts +0 -87
  114. package/src/__tests__/formatRegistry.test.ts +0 -174
  115. package/src/__tests__/html.test.ts +0 -439
  116. package/src/__tests__/htmlImport.test.ts +0 -57
  117. package/src/__tests__/inferTheme.test.ts +0 -135
  118. package/src/__tests__/lossyWarnings.test.ts +0 -146
  119. package/src/__tests__/ooxml.test.ts +0 -271
  120. package/src/__tests__/ooxmlCancellation.test.ts +0 -113
  121. package/src/__tests__/ooxmlThemeReader.test.ts +0 -92
  122. package/src/__tests__/pdfExport.test.ts +0 -322
  123. package/src/__tests__/pdfImport.test.ts +0 -384
  124. package/src/__tests__/plainHtml.test.ts +0 -417
  125. package/src/__tests__/plainHtmlBundle.test.ts +0 -253
  126. package/src/__tests__/pptxExport.test.ts +0 -138
  127. package/src/__tests__/pptxImport.test.ts +0 -145
  128. package/src/__tests__/pptxInferFixtures.ts +0 -314
  129. package/src/__tests__/pptxLayoutInfer.test.ts +0 -395
  130. package/src/__tests__/roundTrip.test.ts +0 -201
  131. package/src/__tests__/roundTripAssets.test.ts +0 -50
  132. package/src/__tests__/roundTripMatrix.fixtures.ts +0 -86
  133. package/src/__tests__/roundTripMatrix.helpers.ts +0 -154
  134. package/src/__tests__/roundTripMatrix.test.ts +0 -142
  135. package/src/__tests__/sharedContainer.test.ts +0 -41
  136. package/src/__tests__/sharedImages.test.ts +0 -61
  137. package/src/__tests__/xlsxExport.test.ts +0 -164
  138. package/src/__tests__/xlsxImport.test.ts +0 -80
  139. package/src/__tests__/zipSafety.test.ts +0 -317
  140. package/src/container/index.ts +0 -94
  141. package/src/csv/index.ts +0 -188
  142. package/src/docx/export.ts +0 -1375
  143. package/src/docx/import.ts +0 -1250
  144. package/src/docx/index.ts +0 -26
  145. package/src/docx/styles.ts +0 -145
  146. package/src/epub/export.ts +0 -968
  147. package/src/epub/index.ts +0 -20
  148. package/src/html/docsHtmlBundle.ts +0 -373
  149. package/src/html/htmlTemplate.ts +0 -385
  150. package/src/html/imageUtils.ts +0 -61
  151. package/src/html/import.ts +0 -297
  152. package/src/html/index.ts +0 -212
  153. package/src/html/plainHtml.ts +0 -790
  154. package/src/html/plainHtmlBundle.ts +0 -421
  155. package/src/index.ts +0 -109
  156. package/src/infer/extract.ts +0 -127
  157. package/src/infer/index.ts +0 -199
  158. package/src/infer/mapTheme.ts +0 -176
  159. package/src/infer/types.ts +0 -27
  160. package/src/ooxml/index.ts +0 -111
  161. package/src/ooxml/namespaces.ts +0 -217
  162. package/src/ooxml/readUtils.ts +0 -44
  163. package/src/ooxml/reader.ts +0 -318
  164. package/src/ooxml/themeReader.ts +0 -197
  165. package/src/ooxml/types.ts +0 -103
  166. package/src/ooxml/writer.ts +0 -339
  167. package/src/ooxml/xmlUtils.ts +0 -123
  168. package/src/pdf/export.ts +0 -1084
  169. package/src/pdf/import.ts +0 -1164
  170. package/src/pdf/index.ts +0 -29
  171. package/src/pdf/styles.ts +0 -180
  172. package/src/pptx/export.ts +0 -1184
  173. package/src/pptx/import.ts +0 -455
  174. package/src/pptx/index.ts +0 -52
  175. package/src/pptx/layouts.ts +0 -1222
  176. package/src/pptx/styles.ts +0 -96
  177. package/src/pptx/templates.ts +0 -187
  178. package/src/registry/convert.ts +0 -433
  179. package/src/registry/defaultFormats.ts +0 -413
  180. package/src/registry/errors.ts +0 -46
  181. package/src/registry/index.ts +0 -43
  182. package/src/registry/registry.ts +0 -48
  183. package/src/registry/types.ts +0 -170
  184. package/src/shared/boundedZipArchive.ts +0 -383
  185. package/src/shared/container.ts +0 -28
  186. package/src/shared/fidelity.ts +0 -130
  187. package/src/shared/images.ts +0 -44
  188. package/src/shared/inlineRuns.ts +0 -99
  189. package/src/shared/text.ts +0 -41
  190. package/src/shared/zipEntryCount.ts +0 -151
  191. package/src/shared/zipLimits.ts +0 -296
  192. package/src/shared/zipSafety.ts +0 -19
  193. package/src/xlsx/export.ts +0 -253
  194. package/src/xlsx/import.ts +0 -160
  195. package/src/xlsx/index.ts +0 -35
@@ -1,1250 +0,0 @@
1
- /**
2
- * DOCX Import
3
- *
4
- * Parses a .docx file (Office Open XML WordprocessingML) and converts
5
- * its content into a squisq MarkdownDocument (or Doc).
6
- *
7
- * Uses JSZip + DOMParser to read the archive and parse the XML — no
8
- * third-party docx library. Handles headings, paragraphs, inline
9
- * formatting (bold, italic, strikethrough), hyperlinks, lists, tables,
10
- * blockquotes, code blocks, images, and footnotes.
11
- *
12
- * @example
13
- * ```ts
14
- * import { docxToMarkdownDoc } from '@bendyline/squisq-formats/docx';
15
- *
16
- * const response = await fetch('document.docx');
17
- * const data = await response.arrayBuffer();
18
- * const doc = await docxToMarkdownDoc(data);
19
- * ```
20
- */
21
-
22
- import type { Doc } from '@bendyline/squisq/schemas';
23
- import { markdownToDoc } from '@bendyline/squisq/doc';
24
- import { stringifyMarkdown } from '@bendyline/squisq/markdown';
25
- import type {
26
- MarkdownDocument,
27
- MarkdownBlockNode,
28
- MarkdownInlineNode,
29
- MarkdownHeading,
30
- MarkdownParagraph,
31
- MarkdownBlockquote,
32
- MarkdownList,
33
- MarkdownListItem,
34
- MarkdownCodeBlock,
35
- MarkdownTable,
36
- MarkdownTableRow,
37
- MarkdownTableCell,
38
- MarkdownText,
39
- MarkdownEmphasis,
40
- MarkdownStrong,
41
- MarkdownStrikethrough,
42
- MarkdownInlineCode,
43
- MarkdownLink,
44
- MarkdownImage,
45
- MarkdownBreak,
46
- MarkdownFootnoteReference,
47
- MarkdownFootnoteDefinition,
48
- MarkdownContainerDirective,
49
- } from '@bendyline/squisq/markdown';
50
-
51
- import { openPackage, getPartXml, getPartBinary, getPartRelationships } from '../ooxml/reader.js';
52
- import type { OoxmlOpenOptions } from '../ooxml/reader.js';
53
- import type { OoxmlPackage, Relationship } from '../ooxml/types.js';
54
- import { NS_WML, NS_R } from '../ooxml/namespaces.js';
55
- import { baseDirOf, resolveTarget } from '../ooxml/readUtils.js';
56
- import type { ContentContainer } from '@bendyline/squisq/storage';
57
- import { buildContainer } from '../shared/container.js';
58
- import { extToMime } from '../shared/images.js';
59
- import {
60
- HEADING_STYLE_MAP,
61
- QUOTE_STYLE_IDS,
62
- CODE_STYLE_IDS,
63
- INLINE_CODE_STYLE_IDS,
64
- BULLET_NUM_FORMATS,
65
- } from './styles.js';
66
-
67
- // ============================================
68
- // Public API
69
- // ============================================
70
-
71
- /**
72
- * Options for DOCX import.
73
- */
74
- export interface DocxImportOptions extends OoxmlOpenOptions {
75
- /**
76
- * Whether to extract embedded images as base64 data URIs.
77
- * When false, images are represented as `[Image]` placeholders.
78
- * Default: false
79
- */
80
- extractImages?: boolean;
81
- }
82
-
83
- /**
84
- * Convert a .docx file to a MarkdownDocument.
85
- *
86
- * @param data - The raw .docx file as ArrayBuffer or Blob
87
- * @param options - Import options
88
- * @returns A MarkdownDocument representing the document content
89
- */
90
- export async function docxToMarkdownDoc(
91
- data: ArrayBuffer | Blob,
92
- options: DocxImportOptions = {},
93
- ): Promise<MarkdownDocument> {
94
- const pkg = await openPackage(data, options);
95
- const ctx = await buildImportContext(pkg, options);
96
-
97
- const documentXml = await getPartXml(pkg, 'word/document.xml');
98
- if (!documentXml) {
99
- return { type: 'document', children: [] };
100
- }
101
-
102
- const body = getFirstElement(documentXml, 'body');
103
- if (!body) {
104
- return { type: 'document', children: [] };
105
- }
106
-
107
- const blocks = await convertDocumentStories(body, ctx);
108
-
109
- return { type: 'document', children: blocks };
110
- }
111
-
112
- /**
113
- * Convert a .docx file to a squisq Doc.
114
- *
115
- * Convenience wrapper: DOCX → MarkdownDocument → Doc.
116
- *
117
- * @param data - The raw .docx file as ArrayBuffer or Blob
118
- * @param options - Import options
119
- * @returns A squisq Doc
120
- */
121
- export async function docxToDoc(
122
- data: ArrayBuffer | Blob,
123
- options: DocxImportOptions = {},
124
- ): Promise<Doc> {
125
- const markdownDoc = await docxToMarkdownDoc(data, options);
126
- return markdownToDoc(markdownDoc);
127
- }
128
-
129
- /**
130
- * Convert a .docx file to a ContentContainer with markdown + extracted images.
131
- *
132
- * The container will contain:
133
- * - The primary markdown document (index.md)
134
- * - Any embedded images under images/ (e.g., images/image1.png)
135
- *
136
- * @param data - The raw .docx file as ArrayBuffer or Blob
137
- * @param options - Import options
138
- * @returns A ContentContainer with the document and its media
139
- */
140
- export async function docxToContainer(
141
- data: ArrayBuffer | Blob,
142
- options: DocxImportOptions = {},
143
- ): Promise<ContentContainer> {
144
- const pkg = await openPackage(data, options);
145
- const ctx = await buildImportContext(pkg, { ...options, extractImages: true });
146
-
147
- const documentXml = await getPartXml(pkg, 'word/document.xml');
148
- if (!documentXml) return buildContainer('', []);
149
-
150
- const body = getFirstElement(documentXml, 'body');
151
- if (!body) return buildContainer('', []);
152
-
153
- const blocks = await convertDocumentStories(body, ctx);
154
- const markdownDoc: MarkdownDocument = { type: 'document', children: blocks };
155
-
156
- return buildContainer(stringifyMarkdown(markdownDoc), ctx.extractedImages);
157
- }
158
-
159
- // ============================================
160
- // Import Context
161
- // ============================================
162
-
163
- interface ImportContext {
164
- /** Style ID → heading depth mapping (from styles.xml) */
165
- headingStyles: Map<string, number>;
166
- /** Style IDs that represent blockquotes */
167
- quoteStyles: Set<string>;
168
- /** Style IDs that represent code blocks */
169
- codeStyles: Set<string>;
170
- /** Character style IDs that represent inline code */
171
- inlineCodeStyles: Set<string>;
172
- /** Document relationship map: rId → Relationship */
173
- documentRels: Map<string, Relationship>;
174
- /** Numbering definitions: numId → { levels: Map<ilvl, isOrdered> } */
175
- numbering: Map<string, NumberingInfo>;
176
- /** Footnote bodies: footnoteId → Element */
177
- footnotes: Map<string, Element>;
178
- /** Endnote bodies: endnoteId → Element */
179
- endnotes: Map<string, Element>;
180
- /** Part whose relationships are active while converting runs. */
181
- currentPartPath: string;
182
- /** Reference to the OOXML package (for extracting images) */
183
- pkg: OoxmlPackage;
184
- /** Import options */
185
- options: DocxImportOptions;
186
- /** Collected image files: relative path → { data, mimeType } */
187
- extractedImages: Map<string, { data: ArrayBuffer; mimeType: string }>;
188
- /** Counter for generating unique image filenames */
189
- imageCounter: number;
190
- }
191
-
192
- interface NumberingInfo {
193
- levels: Map<number, boolean>; // ilvl → isOrdered
194
- }
195
-
196
- async function buildImportContext(
197
- pkg: OoxmlPackage,
198
- options: DocxImportOptions,
199
- ): Promise<ImportContext> {
200
- const ctx: ImportContext = {
201
- headingStyles: new Map(),
202
- quoteStyles: new Set(),
203
- codeStyles: new Set(),
204
- inlineCodeStyles: new Set(),
205
- documentRels: new Map(),
206
- numbering: new Map(),
207
- footnotes: new Map(),
208
- endnotes: new Map(),
209
- currentPartPath: 'word/document.xml',
210
- pkg,
211
- options,
212
- extractedImages: new Map(),
213
- imageCounter: 0,
214
- };
215
-
216
- // Initialize with built-in defaults
217
- for (const [id, depth] of Object.entries(HEADING_STYLE_MAP)) {
218
- ctx.headingStyles.set(id, depth);
219
- }
220
- for (const id of QUOTE_STYLE_IDS) {
221
- ctx.quoteStyles.add(id);
222
- }
223
- for (const id of CODE_STYLE_IDS) {
224
- ctx.codeStyles.add(id);
225
- }
226
- for (const id of INLINE_CODE_STYLE_IDS) {
227
- ctx.inlineCodeStyles.add(id);
228
- }
229
-
230
- // Parse styles.xml for custom heading/quote/code mappings
231
- await parseStyles(pkg, ctx);
232
-
233
- // Parse document relationships
234
- const rels = await getPartRelationships(pkg, 'word/document.xml');
235
- for (const rel of rels) {
236
- ctx.documentRels.set(rel.id, rel);
237
- }
238
-
239
- // Parse numbering.xml
240
- await parseNumbering(pkg, ctx);
241
-
242
- // Parse footnotes.xml
243
- await parseFootnotes(pkg, ctx);
244
-
245
- // Parse endnotes.xml
246
- await parseEndnotes(pkg, ctx);
247
-
248
- return ctx;
249
- }
250
-
251
- // ============================================
252
- // Styles Parsing
253
- // ============================================
254
-
255
- async function parseStyles(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
256
- const doc = await getPartXml(pkg, 'word/styles.xml');
257
- if (!doc) return;
258
-
259
- const styles = doc.getElementsByTagNameNS(NS_WML, 'style');
260
- // Fallback for documents that don't use namespace prefixes properly
261
- const stylesList = styles.length > 0 ? styles : doc.getElementsByTagName('style');
262
-
263
- for (let i = 0; i < stylesList.length; i++) {
264
- const style = stylesList[i];
265
- const styleId = style.getAttributeNS(NS_WML, 'styleId') ?? style.getAttribute('w:styleId');
266
- if (!styleId) continue;
267
-
268
- const nameEl = getFirstChildElement(style, 'name');
269
- const styleName = nameEl?.getAttributeNS(NS_WML, 'val') ?? nameEl?.getAttribute('w:val') ?? '';
270
-
271
- // Check if this is a heading style by name
272
- const headingMatch = styleName.match(/^heading\s+(\d+)$/i);
273
- if (headingMatch) {
274
- const depth = parseInt(headingMatch[1], 10);
275
- if (depth >= 1 && depth <= 6) {
276
- ctx.headingStyles.set(styleId, depth);
277
- }
278
- }
279
-
280
- // Check pPr > outlineLvl for heading detection
281
- const pPr = getFirstChildElement(style, 'pPr');
282
- if (pPr) {
283
- const outlineLvl = getFirstChildElement(pPr, 'outlineLvl');
284
- if (outlineLvl) {
285
- const val = outlineLvl.getAttributeNS(NS_WML, 'val') ?? outlineLvl.getAttribute('w:val');
286
- if (val !== null) {
287
- const depth = parseInt(val, 10) + 1;
288
- if (depth >= 1 && depth <= 6) {
289
- ctx.headingStyles.set(styleId, depth);
290
- }
291
- }
292
- }
293
- }
294
- }
295
- }
296
-
297
- // ============================================
298
- // Numbering Parsing
299
- // ============================================
300
-
301
- async function parseNumbering(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
302
- const doc = await getPartXml(pkg, 'word/numbering.xml');
303
- if (!doc) return;
304
-
305
- // Parse abstract numbering definitions
306
- const abstractNums = new Map<string, Map<number, boolean>>(); // abstractNumId → levels(ilvl → isOrdered)
307
-
308
- const abstractNumEls = getAllElements(doc, 'abstractNum');
309
- for (const absNum of abstractNumEls) {
310
- const absId = getAttr(absNum, 'abstractNumId');
311
- if (!absId) continue;
312
-
313
- const levels = new Map<number, boolean>();
314
- const lvlEls = getAllChildElements(absNum, 'lvl');
315
- for (const lvl of lvlEls) {
316
- const ilvlStr = getAttr(lvl, 'ilvl');
317
- if (ilvlStr === null) continue;
318
- const ilvl = parseInt(ilvlStr, 10);
319
-
320
- const numFmtEl = getFirstChildElement(lvl, 'numFmt');
321
- const numFmt = numFmtEl ? getAttr(numFmtEl, 'val') : null;
322
-
323
- const isOrdered = numFmt !== null && !BULLET_NUM_FORMATS.has(numFmt);
324
- levels.set(ilvl, isOrdered);
325
- }
326
-
327
- abstractNums.set(absId, levels);
328
- }
329
-
330
- // Parse concrete num → abstractNum mappings
331
- const numEls = getAllElements(doc, 'num');
332
- for (const num of numEls) {
333
- const numId = getAttr(num, 'numId');
334
- if (!numId) continue;
335
-
336
- const abstractNumIdEl = getFirstChildElement(num, 'abstractNumId');
337
- const absId = abstractNumIdEl ? getAttr(abstractNumIdEl, 'val') : null;
338
- if (!absId) continue;
339
-
340
- const levels = abstractNums.get(absId);
341
- if (levels) {
342
- ctx.numbering.set(numId, { levels });
343
- }
344
- }
345
- }
346
-
347
- // ============================================
348
- // Footnotes Parsing
349
- // ============================================
350
-
351
- async function parseFootnotes(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
352
- const doc = await getPartXml(pkg, 'word/footnotes.xml');
353
- if (!doc) return;
354
-
355
- const footnoteEls = getAllElements(doc, 'footnote');
356
- for (const fn of footnoteEls) {
357
- const id = getAttr(fn, 'id');
358
- const type = getAttr(fn, 'type');
359
- // Skip separator and continuation separator footnotes
360
- if (!id || type === 'separator' || type === 'continuationSeparator') continue;
361
- ctx.footnotes.set(id, fn);
362
- }
363
- }
364
-
365
- async function parseEndnotes(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
366
- const doc = await getPartXml(pkg, 'word/endnotes.xml');
367
- if (!doc) return;
368
-
369
- const endnoteEls = getAllElements(doc, 'endnote');
370
- for (const note of endnoteEls) {
371
- const id = getAttr(note, 'id');
372
- const type = getAttr(note, 'type');
373
- if (!id || type === 'separator' || type === 'continuationSeparator') continue;
374
- ctx.endnotes.set(id, note);
375
- }
376
- }
377
-
378
- // ============================================
379
- // Body Conversion
380
- // ============================================
381
-
382
- /**
383
- * Convert the main story plus the related header/footer stories. Header and
384
- * footer blocks live in named directives so Markdown consumers can distinguish
385
- * them from body content and the DOCX exporter can put them back in the right
386
- * OOXML parts.
387
- */
388
- async function convertDocumentStories(
389
- body: Element,
390
- ctx: ImportContext,
391
- ): Promise<MarkdownBlockNode[]> {
392
- const bodyBlocks = await convertBody(body, ctx);
393
- const headers = await convertRelatedStories(body, ctx, 'header');
394
- const footers = await convertRelatedStories(body, ctx, 'footer');
395
- const noteDefinitions = await convertNoteDefinitions(ctx);
396
- return [...bodyBlocks, ...headers, ...footers, ...noteDefinitions];
397
- }
398
-
399
- async function convertRelatedStories(
400
- body: Element,
401
- ctx: ImportContext,
402
- kind: 'header' | 'footer',
403
- ): Promise<MarkdownContainerDirective[]> {
404
- const relationshipTypeSuffix = `/${kind}`;
405
- const referenceName = `${kind}Reference`;
406
- const referenceTypes = new Map<string, string>();
407
- for (const reference of getAllElements(body, referenceName)) {
408
- const id = reference.getAttributeNS(NS_R, 'id') ?? reference.getAttribute('r:id');
409
- if (id) referenceTypes.set(id, getAttr(reference, 'type') ?? 'default');
410
- }
411
-
412
- const documentRelationships = ctx.documentRels;
413
- const previousPartPath = ctx.currentPartPath;
414
- const results: MarkdownContainerDirective[] = [];
415
-
416
- for (const relationship of documentRelationships.values()) {
417
- if (!relationship.type.endsWith(relationshipTypeSuffix)) continue;
418
-
419
- const partPath = resolveTarget(baseDirOf('word/document.xml'), relationship.target);
420
- const part = await getPartXml(ctx.pkg, partPath);
421
- if (!part?.documentElement) continue;
422
-
423
- const partRelationships = await getPartRelationships(ctx.pkg, partPath);
424
- ctx.documentRels = new Map(partRelationships.map((rel) => [rel.id, rel]));
425
- ctx.currentPartPath = partPath;
426
- try {
427
- const children = await convertBody(part.documentElement, ctx);
428
- if (children.length === 0) continue;
429
- results.push({
430
- type: 'containerDirective',
431
- name: `docx-${kind}`,
432
- attributes: {
433
- type: referenceTypes.get(relationship.id) ?? 'default',
434
- source: partPath,
435
- },
436
- children,
437
- });
438
- } finally {
439
- ctx.documentRels = documentRelationships;
440
- ctx.currentPartPath = previousPartPath;
441
- }
442
- }
443
-
444
- return results;
445
- }
446
-
447
- async function convertBody(body: Element, ctx: ImportContext): Promise<MarkdownBlockNode[]> {
448
- return convertBlockElements(Array.from(body.children), ctx);
449
- }
450
-
451
- async function convertBlockElements(
452
- children: Element[],
453
- ctx: ImportContext,
454
- ): Promise<MarkdownBlockNode[]> {
455
- const result: MarkdownBlockNode[] = [];
456
-
457
- let i = 0;
458
- while (i < children.length) {
459
- const el = children[i];
460
- const localName = el.localName;
461
-
462
- if (localName === 'p') {
463
- // Check if this is part of a list
464
- const numPr = getNumPr(el);
465
- if (numPr) {
466
- // Collect consecutive list paragraphs
467
- const { node, consumed } = await collectList(children, i, ctx);
468
- result.push(node);
469
- i += consumed;
470
- continue;
471
- }
472
-
473
- const block = await convertParagraph(el, ctx);
474
- if (block) {
475
- result.push(block);
476
- }
477
- i++;
478
- } else if (localName === 'tbl') {
479
- const table = await convertTable(el, ctx);
480
- if (table) result.push(table);
481
- i++;
482
- } else if (localName === 'sdt') {
483
- const content = getFirstChildElement(el, 'sdtContent');
484
- if (content) result.push(...(await convertBlockElements(Array.from(content.children), ctx)));
485
- i++;
486
- } else if (
487
- localName === 'ins' ||
488
- localName === 'moveTo' ||
489
- localName === 'customXml' ||
490
- localName === 'smartTag' ||
491
- localName === 'fldSimple'
492
- ) {
493
- result.push(...(await convertBlockElements(Array.from(el.children), ctx)));
494
- i++;
495
- } else if (localName === 'AlternateContent') {
496
- const selected = getFirstChildElement(el, 'Choice') ?? getFirstChildElement(el, 'Fallback');
497
- if (selected)
498
- result.push(...(await convertBlockElements(Array.from(selected.children), ctx)));
499
- i++;
500
- } else {
501
- // Skip unknown elements (sectPr, bookmarkStart, etc.)
502
- i++;
503
- }
504
- }
505
-
506
- return result;
507
- }
508
-
509
- // ============================================
510
- // Paragraph Conversion
511
- // ============================================
512
-
513
- async function convertParagraph(
514
- el: Element,
515
- ctx: ImportContext,
516
- ): Promise<MarkdownBlockNode | null> {
517
- const pPr = getFirstChildElement(el, 'pPr');
518
- const styleId = getParagraphStyleId(pPr);
519
-
520
- // Check for heading
521
- if (styleId && ctx.headingStyles.has(styleId)) {
522
- const depth = ctx.headingStyles.get(styleId)!;
523
- const inlines = await convertRuns(el, ctx);
524
- if (inlines.length === 0) return null;
525
- return {
526
- type: 'heading',
527
- depth: Math.min(Math.max(depth, 1), 6) as 1 | 2 | 3 | 4 | 5 | 6,
528
- children: inlines,
529
- } satisfies MarkdownHeading;
530
- }
531
-
532
- // Check for blockquote
533
- if (styleId && ctx.quoteStyles.has(styleId)) {
534
- const inlines = await convertRuns(el, ctx);
535
- if (inlines.length === 0) return null;
536
- const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
537
- return { type: 'blockquote', children: [paragraph] } satisfies MarkdownBlockquote;
538
- }
539
-
540
- // Check for code block
541
- if (styleId && ctx.codeStyles.has(styleId)) {
542
- const text = getElementTextContent(el);
543
- return { type: 'code', value: text } satisfies MarkdownCodeBlock;
544
- }
545
-
546
- // Regular paragraph
547
- const inlines = await convertRuns(el, ctx);
548
- if (inlines.length === 0) return null;
549
- return { type: 'paragraph', children: inlines } satisfies MarkdownParagraph;
550
- }
551
-
552
- // ============================================
553
- // Run (Inline) Conversion
554
- // ============================================
555
-
556
- async function convertRuns(
557
- paragraphEl: Element,
558
- ctx: ImportContext,
559
- ): Promise<MarkdownInlineNode[]> {
560
- return mergeAdjacentText(await convertInlineElements(Array.from(paragraphEl.children), ctx));
561
- }
562
-
563
- async function convertInlineElements(
564
- children: Element[],
565
- ctx: ImportContext,
566
- ): Promise<MarkdownInlineNode[]> {
567
- const result: MarkdownInlineNode[] = [];
568
-
569
- for (const child of children) {
570
- const localName = child.localName;
571
-
572
- if (localName === 'r') {
573
- const inlines = await convertRun(child, ctx);
574
- result.push(...inlines);
575
- } else if (localName === 'hyperlink') {
576
- const link = await convertHyperlink(child, ctx);
577
- if (link) result.push(link);
578
- } else if (localName === 'sdt') {
579
- const content = getFirstChildElement(child, 'sdtContent');
580
- if (content) {
581
- result.push(...(await convertInlineElements(Array.from(content.children), ctx)));
582
- }
583
- } else if (
584
- localName === 'ins' ||
585
- localName === 'moveTo' ||
586
- localName === 'customXml' ||
587
- localName === 'smartTag' ||
588
- localName === 'fldSimple'
589
- ) {
590
- result.push(...(await convertInlineElements(Array.from(child.children), ctx)));
591
- } else if (localName === 'AlternateContent') {
592
- // Choice and Fallback represent the same content for different Word
593
- // versions. Reading both duplicates every text box and legacy image.
594
- const selected =
595
- getFirstChildElement(child, 'Choice') ?? getFirstChildElement(child, 'Fallback');
596
- if (selected) {
597
- result.push(...(await convertInlineElements(Array.from(selected.children), ctx)));
598
- }
599
- } else if (localName === 'drawing' || localName === 'pict') {
600
- result.push(...(await convertDrawingContent(child, ctx)));
601
- }
602
- // Skip pPr, bookmarkStart, bookmarkEnd, etc.
603
- }
604
-
605
- return result;
606
- }
607
-
608
- async function convertRun(runEl: Element, ctx: ImportContext): Promise<MarkdownInlineNode[]> {
609
- const result: MarkdownInlineNode[] = [];
610
- const rPr = getFirstChildElement(runEl, 'rPr');
611
- const format = parseRunFormat(rPr, ctx);
612
-
613
- for (const child of Array.from(runEl.children)) {
614
- const localName = child.localName;
615
-
616
- if (localName === 't') {
617
- const text = child.textContent ?? '';
618
- if (!text) continue;
619
-
620
- if (format.code) {
621
- result.push({ type: 'inlineCode', value: text } satisfies MarkdownInlineCode);
622
- } else {
623
- let node: MarkdownInlineNode = { type: 'text', value: text } satisfies MarkdownText;
624
- if (format.strike) {
625
- node = { type: 'delete', children: [node] } satisfies MarkdownStrikethrough;
626
- }
627
- if (format.italic) {
628
- node = { type: 'emphasis', children: [node] } satisfies MarkdownEmphasis;
629
- }
630
- if (format.bold) {
631
- node = { type: 'strong', children: [node] } satisfies MarkdownStrong;
632
- }
633
- result.push(node);
634
- }
635
- } else if (localName === 'br' || localName === 'cr') {
636
- result.push({ type: 'break' } satisfies MarkdownBreak);
637
- } else if (localName === 'tab') {
638
- // Tabs are visible separators in Word. A literal tab is unstable when
639
- // serialized through Markdown (it may become indentation), so retain the
640
- // word boundary as a regular space.
641
- result.push({ type: 'text', value: ' ' } satisfies MarkdownText);
642
- } else if (localName === 'footnoteReference') {
643
- const fnId = getAttr(child, 'id');
644
- if (fnId && fnId !== '0' && fnId !== '-1') {
645
- result.push({
646
- type: 'footnoteReference',
647
- identifier: `fn${fnId}`,
648
- } satisfies MarkdownFootnoteReference);
649
- }
650
- } else if (localName === 'endnoteReference') {
651
- const noteId = getAttr(child, 'id');
652
- if (noteId && noteId !== '0' && noteId !== '-1') {
653
- result.push({
654
- type: 'footnoteReference',
655
- identifier: `endnote${noteId}`,
656
- } satisfies MarkdownFootnoteReference);
657
- }
658
- } else if (localName === 'drawing' || localName === 'pict' || localName === 'object') {
659
- result.push(...(await convertDrawingContent(child, ctx)));
660
- } else if (localName === 'AlternateContent') {
661
- const selected =
662
- getFirstChildElement(child, 'Choice') ?? getFirstChildElement(child, 'Fallback');
663
- if (selected) {
664
- result.push(...(await convertInlineElements(Array.from(selected.children), ctx)));
665
- }
666
- }
667
- }
668
-
669
- return result;
670
- }
671
-
672
- async function convertDrawingContent(
673
- el: Element,
674
- ctx: ImportContext,
675
- ): Promise<MarkdownInlineNode[]> {
676
- const result: MarkdownInlineNode[] = [];
677
-
678
- // A positioned Word shape may be a text box, an image, or both. Text box
679
- // paragraphs are nested inside the drawing rather than being paragraph
680
- // siblings, so the normal body walker never sees them.
681
- const textBoxes = findDescendants(el, 'txbxContent');
682
- for (const textBox of textBoxes) {
683
- const inlines = await flattenContainerToInlines(textBox, ctx);
684
- appendInlineGroup(result, inlines);
685
- }
686
-
687
- const image = await extractImage(el, ctx);
688
- if (image) result.push(image);
689
- if (textBoxes.length > 0 && result.length > 0) {
690
- // Positioned text boxes are independent visual regions. Several can be
691
- // anchored in the same otherwise-empty paragraph (for example, labels on
692
- // a number line); hard boundaries prevent their text from collapsing into
693
- // one synthetic word during Markdown serialization.
694
- if (result[0]?.type !== 'break') result.unshift({ type: 'break' } satisfies MarkdownBreak);
695
- if (result[result.length - 1]?.type !== 'break') {
696
- result.push({ type: 'break' } satisfies MarkdownBreak);
697
- }
698
- }
699
- return result;
700
- }
701
-
702
- async function flattenContainerToInlines(
703
- container: Element,
704
- ctx: ImportContext,
705
- ): Promise<MarkdownInlineNode[]> {
706
- const result: MarkdownInlineNode[] = [];
707
-
708
- for (const child of Array.from(container.children)) {
709
- if (child.localName === 'p') {
710
- appendInlineGroup(result, await convertRuns(child, ctx));
711
- } else if (child.localName === 'tbl') {
712
- appendInlineGroup(result, await flattenTableToInlines(child, ctx));
713
- } else if (child.localName === 'sdt') {
714
- const content = getFirstChildElement(child, 'sdtContent');
715
- if (content) appendInlineGroup(result, await flattenContainerToInlines(content, ctx));
716
- } else if (
717
- child.localName === 'ins' ||
718
- child.localName === 'moveTo' ||
719
- child.localName === 'customXml' ||
720
- child.localName === 'smartTag' ||
721
- child.localName === 'fldSimple'
722
- ) {
723
- appendInlineGroup(result, await flattenContainerToInlines(child, ctx));
724
- } else if (child.localName === 'AlternateContent') {
725
- const selected =
726
- getFirstChildElement(child, 'Choice') ?? getFirstChildElement(child, 'Fallback');
727
- if (selected) appendInlineGroup(result, await flattenContainerToInlines(selected, ctx));
728
- } else if (child.localName === 'r' || child.localName === 'hyperlink') {
729
- appendInlineGroup(result, await convertInlineElements([child], ctx));
730
- }
731
- }
732
-
733
- return mergeAdjacentText(result);
734
- }
735
-
736
- async function flattenTableToInlines(
737
- table: Element,
738
- ctx: ImportContext,
739
- ): Promise<MarkdownInlineNode[]> {
740
- const result: MarkdownInlineNode[] = [];
741
- for (const row of getAllChildElements(table, 'tr')) {
742
- for (const cell of getAllChildElements(row, 'tc')) {
743
- appendInlineGroup(result, await flattenContainerToInlines(cell, ctx));
744
- }
745
- }
746
- return mergeAdjacentText(result);
747
- }
748
-
749
- function appendInlineGroup(target: MarkdownInlineNode[], group: MarkdownInlineNode[]): void {
750
- if (group.length === 0) return;
751
- if (target.length > 0 && target[target.length - 1]?.type !== 'break') {
752
- target.push({ type: 'break' } satisfies MarkdownBreak);
753
- }
754
- target.push(...group);
755
- }
756
-
757
- interface RunFormat {
758
- bold: boolean;
759
- italic: boolean;
760
- strike: boolean;
761
- code: boolean;
762
- }
763
-
764
- function parseRunFormat(rPr: Element | null, ctx: ImportContext): RunFormat {
765
- if (!rPr) return { bold: false, italic: false, strike: false, code: false };
766
-
767
- const bold = hasChildElement(rPr, 'b') && !isFalseToggle(getFirstChildElement(rPr, 'b')!);
768
- const italic = hasChildElement(rPr, 'i') && !isFalseToggle(getFirstChildElement(rPr, 'i')!);
769
- const strike =
770
- hasChildElement(rPr, 'strike') && !isFalseToggle(getFirstChildElement(rPr, 'strike')!);
771
-
772
- // Check for inline code via character style
773
- const rStyle = getFirstChildElement(rPr, 'rStyle');
774
- const charStyleId = rStyle ? getAttr(rStyle, 'val') : null;
775
- const isCodeStyle = charStyleId ? ctx.inlineCodeStyles.has(charStyleId) : false;
776
-
777
- // Check for monospace font as a code indicator
778
- const rFonts = getFirstChildElement(rPr, 'rFonts');
779
- const fontName = rFonts ? (getAttr(rFonts, 'ascii') ?? getAttr(rFonts, 'hAnsi') ?? '') : '';
780
- const isMonospace = /consolas|courier|mono/i.test(fontName);
781
-
782
- return { bold, italic, strike, code: isCodeStyle || isMonospace };
783
- }
784
-
785
- function isFalseToggle(el: Element): boolean {
786
- const val = getAttr(el, 'val');
787
- return val === '0' || val === 'false';
788
- }
789
-
790
- // ============================================
791
- // Hyperlink Conversion
792
- // ============================================
793
-
794
- async function convertHyperlink(el: Element, ctx: ImportContext): Promise<MarkdownLink | null> {
795
- const rId = el.getAttributeNS(NS_R, 'id') ?? el.getAttribute('r:id');
796
-
797
- let url = '';
798
- if (rId) {
799
- const rel = ctx.documentRels.get(rId);
800
- if (rel) url = rel.target;
801
- }
802
-
803
- // Also check for w:anchor (internal bookmarks)
804
- if (!url) {
805
- const anchor = el.getAttributeNS(NS_WML, 'anchor') ?? el.getAttribute('w:anchor');
806
- if (anchor) url = `#${anchor}`;
807
- }
808
-
809
- const inlines = await convertInlineElements(Array.from(el.children), ctx);
810
-
811
- if (inlines.length === 0) return null;
812
-
813
- return {
814
- type: 'link',
815
- url,
816
- children: mergeAdjacentText(inlines),
817
- };
818
- }
819
-
820
- // ============================================
821
- // Image Extraction
822
- // ============================================
823
-
824
- async function extractImage(el: Element, ctx: ImportContext): Promise<MarkdownImage | null> {
825
- // DrawingML uses <a:blip r:embed="...">; older Word/VML documents use
826
- // <v:imagedata r:id="...">. Supporting both also recovers many scanned
827
- // forms and older templates in the corpus.
828
- const blip = findDescendant(el, 'blip');
829
- const imageData = findDescendant(el, 'imagedata');
830
- const rId = blip
831
- ? (blip.getAttributeNS(NS_R, 'embed') ?? blip.getAttribute('r:embed'))
832
- : (imageData?.getAttributeNS(NS_R, 'id') ??
833
- imageData?.getAttribute('r:id') ??
834
- imageData?.getAttribute('o:relid'));
835
- if (!rId) return null;
836
-
837
- const rel = ctx.documentRels.get(rId);
838
- if (!rel || rel.targetMode === 'External') return null;
839
-
840
- const target = resolveTarget(baseDirOf(ctx.currentPartPath), rel.target);
841
-
842
- // Extract binary data from the zip
843
- const data = await getPartBinary(ctx.pkg, target);
844
- if (!data) return null;
845
-
846
- // Determine extension and MIME type
847
- const dot = target.lastIndexOf('.');
848
- const ext = dot !== -1 ? target.slice(dot).toLowerCase() : '.png';
849
- const mimeType = extToMime(ext);
850
-
851
- // Generate a unique image path
852
- ctx.imageCounter++;
853
- const imagePath = `images/image${ctx.imageCounter}${ext}`;
854
-
855
- // Store the extracted image data
856
- ctx.extractedImages.set(imagePath, { data, mimeType });
857
-
858
- // Try to extract alt text from the drawing's docPr element
859
- const docPr = findDescendant(el, 'docPr');
860
- const shape = findDescendant(el, 'shape');
861
- const alt =
862
- docPr?.getAttribute('descr') ||
863
- docPr?.getAttribute('title') ||
864
- shape?.getAttribute('alt') ||
865
- shape?.getAttribute('title') ||
866
- 'Image';
867
-
868
- return {
869
- type: 'image',
870
- url: imagePath,
871
- alt,
872
- };
873
- }
874
-
875
- /** Recursively find the first descendant element with the given local name. */
876
- function findDescendant(el: Element, localName: string): Element | null {
877
- for (const child of Array.from(el.children)) {
878
- if (child.localName === localName) return child;
879
- const found = findDescendant(child, localName);
880
- if (found) return found;
881
- }
882
- return null;
883
- }
884
-
885
- /** Recursively find every descendant element with the given local name. */
886
- function findDescendants(el: Element, localName: string): Element[] {
887
- const results: Element[] = [];
888
- for (const child of Array.from(el.children)) {
889
- if (child.localName === localName) results.push(child);
890
- results.push(...findDescendants(child, localName));
891
- }
892
- return results;
893
- }
894
-
895
- // ============================================
896
- // List Collection
897
- // ============================================
898
-
899
- interface ListResult {
900
- node: MarkdownList;
901
- consumed: number;
902
- }
903
-
904
- interface NumPrInfo {
905
- numId: string;
906
- ilvl: number;
907
- }
908
-
909
- async function collectList(
910
- elements: Element[],
911
- startIdx: number,
912
- ctx: ImportContext,
913
- ): Promise<ListResult> {
914
- const firstNumPr = getNumPr(elements[startIdx])!;
915
- const { numId } = firstNumPr;
916
-
917
- // Determine if ordered from numbering definition
918
- const numInfo = ctx.numbering.get(numId);
919
- const isOrdered = numInfo?.levels.get(0) ?? false;
920
-
921
- const items: MarkdownListItem[] = [];
922
- let consumed = 0;
923
-
924
- let i = startIdx;
925
- while (i < elements.length) {
926
- const el = elements[i];
927
- if (el.localName !== 'p') break;
928
-
929
- const numPr = getNumPr(el);
930
- if (!numPr || numPr.numId !== numId) break;
931
-
932
- // Convert this paragraph's inline content
933
- const inlines = await convertRuns(el, ctx);
934
- if (inlines.length > 0) {
935
- const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
936
-
937
- // Check if this is a nested list item
938
- if (numPr.ilvl > firstNumPr.ilvl) {
939
- // Collect nested list items
940
- const nested = await collectNestedList(elements, i, ctx, firstNumPr.ilvl);
941
- if (items.length > 0) {
942
- // Attach nested list to the last item
943
- const lastItem = items[items.length - 1];
944
- const nestedIsOrdered = numInfo?.levels.get(numPr.ilvl) ?? false;
945
- const nestedList: MarkdownList = {
946
- type: 'list',
947
- ordered: nestedIsOrdered,
948
- children: nested.items,
949
- };
950
- lastItem.children.push(nestedList);
951
- } else {
952
- // Some Word producers emit an empty parent-level numbering
953
- // paragraph before the first real (more deeply indented) item. With
954
- // no parent item to attach to, retain those items at this level
955
- // instead of silently discarding the entire nested subtree.
956
- items.push(...nested.items);
957
- }
958
- i += nested.consumed;
959
- consumed += nested.consumed;
960
- continue;
961
- }
962
-
963
- items.push({
964
- type: 'listItem',
965
- children: [paragraph],
966
- });
967
- }
968
-
969
- i++;
970
- consumed++;
971
- }
972
-
973
- return {
974
- node: {
975
- type: 'list',
976
- ordered: isOrdered,
977
- children: items,
978
- },
979
- consumed,
980
- };
981
- }
982
-
983
- interface NestedListResult {
984
- items: MarkdownListItem[];
985
- consumed: number;
986
- }
987
-
988
- async function collectNestedList(
989
- elements: Element[],
990
- startIdx: number,
991
- ctx: ImportContext,
992
- parentIlvl: number,
993
- ): Promise<NestedListResult> {
994
- const items: MarkdownListItem[] = [];
995
- let consumed = 0;
996
-
997
- let i = startIdx;
998
- while (i < elements.length) {
999
- const el = elements[i];
1000
- if (el.localName !== 'p') break;
1001
-
1002
- const numPr = getNumPr(el);
1003
- if (!numPr) break;
1004
- if (numPr.ilvl <= parentIlvl) break;
1005
-
1006
- const inlines = await convertRuns(el, ctx);
1007
- if (inlines.length > 0) {
1008
- const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
1009
- items.push({ type: 'listItem', children: [paragraph] });
1010
- }
1011
-
1012
- i++;
1013
- consumed++;
1014
- }
1015
-
1016
- return { items, consumed };
1017
- }
1018
-
1019
- function getNumPr(el: Element): NumPrInfo | null {
1020
- const pPr = getFirstChildElement(el, 'pPr');
1021
- if (!pPr) return null;
1022
-
1023
- const numPr = getFirstChildElement(pPr, 'numPr');
1024
- if (!numPr) return null;
1025
-
1026
- const ilvlEl = getFirstChildElement(numPr, 'ilvl');
1027
- const numIdEl = getFirstChildElement(numPr, 'numId');
1028
-
1029
- const ilvlVal = ilvlEl ? getAttr(ilvlEl, 'val') : null;
1030
- const numIdVal = numIdEl ? getAttr(numIdEl, 'val') : null;
1031
-
1032
- if (!numIdVal || numIdVal === '0') return null; // numId 0 means "no list"
1033
-
1034
- return {
1035
- numId: numIdVal,
1036
- ilvl: ilvlVal ? parseInt(ilvlVal, 10) : 0,
1037
- };
1038
- }
1039
-
1040
- // ============================================
1041
- // Table Conversion
1042
- // ============================================
1043
-
1044
- async function convertTable(tblEl: Element, ctx: ImportContext): Promise<MarkdownTable | null> {
1045
- const rows: MarkdownTableRow[] = [];
1046
-
1047
- const trEls = getAllChildElements(tblEl, 'tr');
1048
- for (let ri = 0; ri < trEls.length; ri++) {
1049
- const row = await convertTableRow(trEls[ri], ctx, ri === 0);
1050
- rows.push(row);
1051
- }
1052
-
1053
- if (rows.length === 0) return null;
1054
-
1055
- // If first row isn't explicitly a header, treat it as one anyway
1056
- // (Markdown tables require a header row)
1057
- return {
1058
- type: 'table',
1059
- children: rows,
1060
- };
1061
- }
1062
-
1063
- async function convertTableRow(
1064
- trEl: Element,
1065
- ctx: ImportContext,
1066
- isHeader: boolean,
1067
- ): Promise<MarkdownTableRow> {
1068
- const cells: MarkdownTableCell[] = [];
1069
- const tcEls = getAllChildElements(trEl, 'tc');
1070
-
1071
- for (const tc of tcEls) {
1072
- const cell = await convertTableCell(tc, ctx, isHeader);
1073
- cells.push(cell);
1074
- }
1075
-
1076
- return { type: 'tableRow', children: cells };
1077
- }
1078
-
1079
- async function convertTableCell(
1080
- tcEl: Element,
1081
- ctx: ImportContext,
1082
- isHeader: boolean,
1083
- ): Promise<MarkdownTableCell> {
1084
- // Cells may contain paragraphs, content controls, and recursively nested
1085
- // tables (a common layout technique in Word forms). Markdown tables cannot
1086
- // nest, so preserve all cell text in reading order separated by hard breaks.
1087
- const inlines = await flattenContainerToInlines(tcEl, ctx);
1088
-
1089
- return {
1090
- type: 'tableCell',
1091
- isHeader,
1092
- children: mergeAdjacentText(inlines),
1093
- };
1094
- }
1095
-
1096
- // ============================================
1097
- // Footnote Definition Conversion
1098
- // ============================================
1099
-
1100
- async function convertNoteDefinitions(ctx: ImportContext): Promise<MarkdownFootnoteDefinition[]> {
1101
- return [
1102
- ...(await convertNotePartDefinitions(ctx, ctx.footnotes, 'fn', 'word/footnotes.xml')),
1103
- ...(await convertNotePartDefinitions(ctx, ctx.endnotes, 'endnote', 'word/endnotes.xml')),
1104
- ];
1105
- }
1106
-
1107
- async function convertNotePartDefinitions(
1108
- ctx: ImportContext,
1109
- notes: Map<string, Element>,
1110
- identifierPrefix: string,
1111
- partPath: string,
1112
- ): Promise<MarkdownFootnoteDefinition[]> {
1113
- const results: MarkdownFootnoteDefinition[] = [];
1114
- const previousRelationships = ctx.documentRels;
1115
- const previousPartPath = ctx.currentPartPath;
1116
- const partRelationships = await getPartRelationships(ctx.pkg, partPath);
1117
- ctx.documentRels = new Map(partRelationships.map((rel) => [rel.id, rel]));
1118
- ctx.currentPartPath = partPath;
1119
-
1120
- try {
1121
- for (const [id, el] of notes) {
1122
- const children = await convertBody(el, ctx);
1123
- if (children.length > 0) {
1124
- results.push({
1125
- type: 'footnoteDefinition',
1126
- identifier: `${identifierPrefix}${id}`,
1127
- children,
1128
- });
1129
- }
1130
- }
1131
- } finally {
1132
- ctx.documentRels = previousRelationships;
1133
- ctx.currentPartPath = previousPartPath;
1134
- }
1135
-
1136
- return results;
1137
- }
1138
-
1139
- // ============================================
1140
- // XML Helper Utilities
1141
- // ============================================
1142
-
1143
- /**
1144
- * Get the first element child with a given local name.
1145
- * Handles both namespaced and non-namespaced elements.
1146
- */
1147
- function getFirstChildElement(parent: Element | Document, localName: string): Element | null {
1148
- for (const child of Array.from(parent.children ?? [])) {
1149
- if (child.localName === localName) return child;
1150
- }
1151
- return null;
1152
- }
1153
-
1154
- /**
1155
- * Get all direct child elements with a given local name.
1156
- */
1157
- function getAllChildElements(parent: Element, localName: string): Element[] {
1158
- const result: Element[] = [];
1159
- for (const child of Array.from(parent.children)) {
1160
- if (child.localName === localName) result.push(child);
1161
- }
1162
- return result;
1163
- }
1164
-
1165
- /**
1166
- * Get all elements with a given local name in the document.
1167
- */
1168
- function getAllElements(doc: Document | Element, localName: string): Element[] {
1169
- // Try namespace-aware first
1170
- const nsEls =
1171
- 'getElementsByTagNameNS' in doc ? doc.getElementsByTagNameNS(NS_WML, localName) : null;
1172
- if (nsEls && nsEls.length > 0) return Array.from(nsEls);
1173
-
1174
- // Fallback: try with w: prefix
1175
- const prefixed = doc.getElementsByTagName(`w:${localName}`);
1176
- if (prefixed.length > 0) return Array.from(prefixed);
1177
-
1178
- // Final fallback: bare name
1179
- return Array.from(doc.getElementsByTagName(localName));
1180
- }
1181
-
1182
- /**
1183
- * Get the first element with given local name in the document or subtree.
1184
- */
1185
- function getFirstElement(doc: Document | Element, localName: string): Element | null {
1186
- const els = getAllElements(doc, localName);
1187
- return els.length > 0 ? els[0] : null;
1188
- }
1189
-
1190
- /**
1191
- * Get a w:-prefixed attribute, trying namespace-aware first then fallback.
1192
- */
1193
- function getAttr(el: Element, localName: string): string | null {
1194
- return el.getAttributeNS(NS_WML, localName) || el.getAttribute(`w:${localName}`) || null;
1195
- }
1196
-
1197
- /**
1198
- * Check if an element has a direct child with the given local name.
1199
- */
1200
- function hasChildElement(parent: Element, localName: string): boolean {
1201
- return getFirstChildElement(parent, localName) !== null;
1202
- }
1203
-
1204
- /**
1205
- * Get the paragraph style ID from a pPr element.
1206
- */
1207
- function getParagraphStyleId(pPr: Element | null): string | null {
1208
- if (!pPr) return null;
1209
- const pStyle = getFirstChildElement(pPr, 'pStyle');
1210
- if (!pStyle) return null;
1211
- return getAttr(pStyle, 'val');
1212
- }
1213
-
1214
- /**
1215
- * Get all text content from an element (concatenating all w:t descendants).
1216
- */
1217
- function getElementTextContent(el: Element): string {
1218
- const parts: string[] = [];
1219
-
1220
- function walk(node: Element): void {
1221
- if (node.localName === 't') {
1222
- parts.push(node.textContent ?? '');
1223
- }
1224
- for (const child of Array.from(node.children)) {
1225
- walk(child);
1226
- }
1227
- }
1228
-
1229
- walk(el);
1230
- return parts.join('');
1231
- }
1232
-
1233
- /**
1234
- * Merge adjacent MarkdownText nodes to reduce fragmentation.
1235
- */
1236
- function mergeAdjacentText(nodes: MarkdownInlineNode[]): MarkdownInlineNode[] {
1237
- if (nodes.length <= 1) return nodes;
1238
-
1239
- const result: MarkdownInlineNode[] = [];
1240
- for (const node of nodes) {
1241
- const prev = result[result.length - 1];
1242
- if (node.type === 'text' && prev?.type === 'text') {
1243
- // Merge into previous text node
1244
- (prev as MarkdownText).value += (node as MarkdownText).value;
1245
- } else {
1246
- result.push(node);
1247
- }
1248
- }
1249
- return result;
1250
- }