@bendyline/squisq-formats 2.1.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/NOTICE.md +20 -0
- package/README.md +1 -1
- package/dist/{chunk-NNHKUXKA.js → chunk-26ISNJ7Y.js} +85 -65
- package/dist/{chunk-NKAJPJ4G.js → chunk-2JJ5RFDZ.js} +0 -1
- package/dist/{chunk-WQSHGBLN.js → chunk-3NKXBZSR.js} +193 -42
- package/dist/{chunk-KURGXM4I.js → chunk-4V3KCHAP.js} +3 -4
- package/dist/{chunk-MLX2BOJC.js → chunk-6RQOV3B3.js} +1 -2
- package/dist/{chunk-EW54IRRS.js → chunk-6S6GU3ZG.js} +5 -6
- package/dist/{chunk-FE6OJV6O.js → chunk-7AWFHP5U.js} +1 -1
- package/dist/{chunk-RFAPOKHJ.js → chunk-AD2WT564.js} +59 -9
- package/dist/{chunk-O3GVVND4.js → chunk-AONELFLA.js} +0 -1
- package/dist/{chunk-SC67HYQJ.js → chunk-EJTNGKEA.js} +5 -8
- package/dist/chunk-GX7RAUME.js +121 -0
- package/dist/{chunk-SSUPBUF5.js → chunk-IIQYS2YH.js} +0 -1
- package/dist/{chunk-RLU7UFYU.js → chunk-IPN56VLW.js} +83 -58
- package/dist/{chunk-DTDF6QDP.js → chunk-JE6LSIHE.js} +81 -20
- package/dist/{chunk-U4MRIFKL.js → chunk-JU2RHXUB.js} +0 -1
- package/dist/{chunk-4VUWTSGM.js → chunk-K6XRMVPW.js} +64 -31
- package/dist/{chunk-ODL3SSPT.js → chunk-KXOZMWBS.js} +0 -1
- package/dist/chunk-OGS5VCGJ.js +446 -0
- package/dist/{chunk-GVS2XXV6.js → chunk-PJXJI2LY.js} +449 -57
- package/dist/{chunk-74GO3FVS.js → chunk-PU7REGWV.js} +5 -8
- package/dist/{chunk-PN52A5AA.js → chunk-SBUW7NHR.js} +0 -1
- package/dist/{chunk-QFLDYKCR.js → chunk-TAAENIRB.js} +5 -8
- package/dist/{chunk-7ARKUCQT.js → chunk-X2DEAXNK.js} +54 -2
- package/dist/container/index.js +1 -2
- package/dist/csv/index.d.ts +27 -2
- package/dist/csv/index.js +1 -2
- package/dist/docx/index.d.ts +5 -1
- package/dist/docx/index.js +9 -11
- package/dist/epub/index.d.ts +2 -0
- package/dist/epub/index.js +5 -6
- package/dist/{export-D2NkylDT.d.ts → export-D9msROJS.d.ts} +18 -6
- package/dist/extract-MN7LA3NL.js +13 -0
- package/dist/html/index.d.ts +11 -4
- package/dist/html/index.js +3 -4
- package/dist/images-ESPQKVTW.js +6 -0
- package/dist/{import-K8mfc0fz.d.ts → import-C3htUTss.d.ts} +5 -1
- package/dist/{import-DTkDxHmZ.d.ts → import-C8whCC7_.d.ts} +6 -0
- package/dist/index.d.ts +7 -7
- package/dist/index.js +28 -26
- package/dist/infer/index.d.ts +3 -3
- package/dist/infer/index.js +7 -9
- package/dist/{layouts-BHrgZ5FS.d.ts → layouts-CTdPlB-u.d.ts} +1 -1
- package/dist/layouts-DRWZGSPD.js +10 -0
- package/dist/{mapTheme-IR27S6IV.js → mapTheme-4TWH25FT.js} +1 -2
- package/dist/ooxml/index.d.ts +3 -3
- package/dist/ooxml/index.js +14 -13
- package/dist/pdf/index.d.ts +18 -0
- package/dist/pdf/index.js +2 -3
- package/dist/pptx/index.d.ts +4 -4
- package/dist/pptx/index.js +11 -13
- package/dist/{reader-B9L8Ucbj.d.ts → reader-B_m1aKZC.d.ts} +30 -1
- package/dist/registry/index.d.ts +21 -5
- package/dist/registry/index.js +9 -6
- package/dist/{themeReader-DJKErl_j.d.ts → themeReader-DCtwC83Q.d.ts} +1 -1
- package/dist/xlsx/index.d.ts +3 -3
- package/dist/xlsx/index.js +6 -7
- package/package.json +6 -3
- package/dist/chunk-4VUWTSGM.js.map +0 -1
- package/dist/chunk-6M7Z25LA.js +0 -46
- package/dist/chunk-6M7Z25LA.js.map +0 -1
- package/dist/chunk-74GO3FVS.js.map +0 -1
- package/dist/chunk-7ARKUCQT.js.map +0 -1
- package/dist/chunk-DTDF6QDP.js.map +0 -1
- package/dist/chunk-EW54IRRS.js.map +0 -1
- package/dist/chunk-FE6OJV6O.js.map +0 -1
- package/dist/chunk-GVS2XXV6.js.map +0 -1
- package/dist/chunk-KURGXM4I.js.map +0 -1
- package/dist/chunk-MLX2BOJC.js.map +0 -1
- package/dist/chunk-NKAJPJ4G.js.map +0 -1
- package/dist/chunk-NNHKUXKA.js.map +0 -1
- package/dist/chunk-O3GVVND4.js.map +0 -1
- package/dist/chunk-ODL3SSPT.js.map +0 -1
- package/dist/chunk-PN52A5AA.js.map +0 -1
- package/dist/chunk-QFLDYKCR.js.map +0 -1
- package/dist/chunk-RFAPOKHJ.js.map +0 -1
- package/dist/chunk-RLU7UFYU.js.map +0 -1
- package/dist/chunk-SC67HYQJ.js.map +0 -1
- package/dist/chunk-SSUPBUF5.js.map +0 -1
- package/dist/chunk-U4MRIFKL.js.map +0 -1
- package/dist/chunk-UGYF5AZE.js +0 -275
- package/dist/chunk-UGYF5AZE.js.map +0 -1
- package/dist/chunk-WQSHGBLN.js.map +0 -1
- package/dist/chunk-YRT7GQ5Y.js +0 -28
- package/dist/chunk-YRT7GQ5Y.js.map +0 -1
- package/dist/container/index.js.map +0 -1
- package/dist/csv/index.js.map +0 -1
- package/dist/docx/index.js.map +0 -1
- package/dist/epub/index.js.map +0 -1
- package/dist/extract-OJ7ZQV6P.js +0 -15
- package/dist/extract-OJ7ZQV6P.js.map +0 -1
- package/dist/html/index.js.map +0 -1
- package/dist/images-7FBWPKE3.js +0 -7
- package/dist/images-7FBWPKE3.js.map +0 -1
- package/dist/index.js.map +0 -1
- package/dist/infer/index.js.map +0 -1
- package/dist/layouts-5VDIRPIJ.js +0 -12
- package/dist/layouts-5VDIRPIJ.js.map +0 -1
- package/dist/mapTheme-IR27S6IV.js.map +0 -1
- package/dist/ooxml/index.js.map +0 -1
- package/dist/pdf/index.js.map +0 -1
- package/dist/pptx/index.js.map +0 -1
- package/dist/registry/index.js.map +0 -1
- package/dist/xlsx/index.js.map +0 -1
- package/src/__tests__/container.test.ts +0 -230
- package/src/__tests__/convert.test.ts +0 -495
- package/src/__tests__/csvImport.test.ts +0 -84
- package/src/__tests__/docxExport.test.ts +0 -491
- package/src/__tests__/docxImport.test.ts +0 -531
- package/src/__tests__/epub.test.ts +0 -649
- package/src/__tests__/exportThemeReconciliation.test.ts +0 -87
- package/src/__tests__/formatRegistry.test.ts +0 -174
- package/src/__tests__/html.test.ts +0 -439
- package/src/__tests__/htmlImport.test.ts +0 -57
- package/src/__tests__/inferTheme.test.ts +0 -135
- package/src/__tests__/lossyWarnings.test.ts +0 -146
- package/src/__tests__/ooxml.test.ts +0 -271
- package/src/__tests__/ooxmlCancellation.test.ts +0 -113
- package/src/__tests__/ooxmlThemeReader.test.ts +0 -92
- package/src/__tests__/pdfExport.test.ts +0 -322
- package/src/__tests__/pdfImport.test.ts +0 -384
- package/src/__tests__/plainHtml.test.ts +0 -417
- package/src/__tests__/plainHtmlBundle.test.ts +0 -253
- package/src/__tests__/pptxExport.test.ts +0 -138
- package/src/__tests__/pptxImport.test.ts +0 -145
- package/src/__tests__/pptxInferFixtures.ts +0 -314
- package/src/__tests__/pptxLayoutInfer.test.ts +0 -395
- package/src/__tests__/roundTrip.test.ts +0 -201
- package/src/__tests__/roundTripAssets.test.ts +0 -50
- package/src/__tests__/roundTripMatrix.fixtures.ts +0 -86
- package/src/__tests__/roundTripMatrix.helpers.ts +0 -154
- package/src/__tests__/roundTripMatrix.test.ts +0 -142
- package/src/__tests__/sharedContainer.test.ts +0 -41
- package/src/__tests__/sharedImages.test.ts +0 -61
- package/src/__tests__/xlsxExport.test.ts +0 -164
- package/src/__tests__/xlsxImport.test.ts +0 -80
- package/src/__tests__/zipSafety.test.ts +0 -317
- package/src/container/index.ts +0 -94
- package/src/csv/index.ts +0 -188
- package/src/docx/export.ts +0 -1375
- package/src/docx/import.ts +0 -1250
- package/src/docx/index.ts +0 -26
- package/src/docx/styles.ts +0 -145
- package/src/epub/export.ts +0 -968
- package/src/epub/index.ts +0 -20
- package/src/html/docsHtmlBundle.ts +0 -373
- package/src/html/htmlTemplate.ts +0 -385
- package/src/html/imageUtils.ts +0 -61
- package/src/html/import.ts +0 -297
- package/src/html/index.ts +0 -212
- package/src/html/plainHtml.ts +0 -790
- package/src/html/plainHtmlBundle.ts +0 -421
- package/src/index.ts +0 -109
- package/src/infer/extract.ts +0 -127
- package/src/infer/index.ts +0 -199
- package/src/infer/mapTheme.ts +0 -176
- package/src/infer/types.ts +0 -27
- package/src/ooxml/index.ts +0 -111
- package/src/ooxml/namespaces.ts +0 -217
- package/src/ooxml/readUtils.ts +0 -44
- package/src/ooxml/reader.ts +0 -318
- package/src/ooxml/themeReader.ts +0 -197
- package/src/ooxml/types.ts +0 -103
- package/src/ooxml/writer.ts +0 -339
- package/src/ooxml/xmlUtils.ts +0 -123
- package/src/pdf/export.ts +0 -1084
- package/src/pdf/import.ts +0 -1164
- package/src/pdf/index.ts +0 -29
- package/src/pdf/styles.ts +0 -180
- package/src/pptx/export.ts +0 -1184
- package/src/pptx/import.ts +0 -455
- package/src/pptx/index.ts +0 -52
- package/src/pptx/layouts.ts +0 -1222
- package/src/pptx/styles.ts +0 -96
- package/src/pptx/templates.ts +0 -187
- package/src/registry/convert.ts +0 -433
- package/src/registry/defaultFormats.ts +0 -413
- package/src/registry/errors.ts +0 -46
- package/src/registry/index.ts +0 -43
- package/src/registry/registry.ts +0 -48
- package/src/registry/types.ts +0 -170
- package/src/shared/boundedZipArchive.ts +0 -383
- package/src/shared/container.ts +0 -28
- package/src/shared/fidelity.ts +0 -130
- package/src/shared/images.ts +0 -44
- package/src/shared/inlineRuns.ts +0 -99
- package/src/shared/text.ts +0 -41
- package/src/shared/zipEntryCount.ts +0 -151
- package/src/shared/zipLimits.ts +0 -296
- package/src/shared/zipSafety.ts +0 -19
- package/src/xlsx/export.ts +0 -253
- package/src/xlsx/import.ts +0 -160
- package/src/xlsx/index.ts +0 -35
package/src/docx/import.ts
DELETED
|
@@ -1,1250 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* DOCX Import
|
|
3
|
-
*
|
|
4
|
-
* Parses a .docx file (Office Open XML WordprocessingML) and converts
|
|
5
|
-
* its content into a squisq MarkdownDocument (or Doc).
|
|
6
|
-
*
|
|
7
|
-
* Uses JSZip + DOMParser to read the archive and parse the XML — no
|
|
8
|
-
* third-party docx library. Handles headings, paragraphs, inline
|
|
9
|
-
* formatting (bold, italic, strikethrough), hyperlinks, lists, tables,
|
|
10
|
-
* blockquotes, code blocks, images, and footnotes.
|
|
11
|
-
*
|
|
12
|
-
* @example
|
|
13
|
-
* ```ts
|
|
14
|
-
* import { docxToMarkdownDoc } from '@bendyline/squisq-formats/docx';
|
|
15
|
-
*
|
|
16
|
-
* const response = await fetch('document.docx');
|
|
17
|
-
* const data = await response.arrayBuffer();
|
|
18
|
-
* const doc = await docxToMarkdownDoc(data);
|
|
19
|
-
* ```
|
|
20
|
-
*/
|
|
21
|
-
|
|
22
|
-
import type { Doc } from '@bendyline/squisq/schemas';
|
|
23
|
-
import { markdownToDoc } from '@bendyline/squisq/doc';
|
|
24
|
-
import { stringifyMarkdown } from '@bendyline/squisq/markdown';
|
|
25
|
-
import type {
|
|
26
|
-
MarkdownDocument,
|
|
27
|
-
MarkdownBlockNode,
|
|
28
|
-
MarkdownInlineNode,
|
|
29
|
-
MarkdownHeading,
|
|
30
|
-
MarkdownParagraph,
|
|
31
|
-
MarkdownBlockquote,
|
|
32
|
-
MarkdownList,
|
|
33
|
-
MarkdownListItem,
|
|
34
|
-
MarkdownCodeBlock,
|
|
35
|
-
MarkdownTable,
|
|
36
|
-
MarkdownTableRow,
|
|
37
|
-
MarkdownTableCell,
|
|
38
|
-
MarkdownText,
|
|
39
|
-
MarkdownEmphasis,
|
|
40
|
-
MarkdownStrong,
|
|
41
|
-
MarkdownStrikethrough,
|
|
42
|
-
MarkdownInlineCode,
|
|
43
|
-
MarkdownLink,
|
|
44
|
-
MarkdownImage,
|
|
45
|
-
MarkdownBreak,
|
|
46
|
-
MarkdownFootnoteReference,
|
|
47
|
-
MarkdownFootnoteDefinition,
|
|
48
|
-
MarkdownContainerDirective,
|
|
49
|
-
} from '@bendyline/squisq/markdown';
|
|
50
|
-
|
|
51
|
-
import { openPackage, getPartXml, getPartBinary, getPartRelationships } from '../ooxml/reader.js';
|
|
52
|
-
import type { OoxmlOpenOptions } from '../ooxml/reader.js';
|
|
53
|
-
import type { OoxmlPackage, Relationship } from '../ooxml/types.js';
|
|
54
|
-
import { NS_WML, NS_R } from '../ooxml/namespaces.js';
|
|
55
|
-
import { baseDirOf, resolveTarget } from '../ooxml/readUtils.js';
|
|
56
|
-
import type { ContentContainer } from '@bendyline/squisq/storage';
|
|
57
|
-
import { buildContainer } from '../shared/container.js';
|
|
58
|
-
import { extToMime } from '../shared/images.js';
|
|
59
|
-
import {
|
|
60
|
-
HEADING_STYLE_MAP,
|
|
61
|
-
QUOTE_STYLE_IDS,
|
|
62
|
-
CODE_STYLE_IDS,
|
|
63
|
-
INLINE_CODE_STYLE_IDS,
|
|
64
|
-
BULLET_NUM_FORMATS,
|
|
65
|
-
} from './styles.js';
|
|
66
|
-
|
|
67
|
-
// ============================================
|
|
68
|
-
// Public API
|
|
69
|
-
// ============================================
|
|
70
|
-
|
|
71
|
-
/**
|
|
72
|
-
* Options for DOCX import.
|
|
73
|
-
*/
|
|
74
|
-
export interface DocxImportOptions extends OoxmlOpenOptions {
|
|
75
|
-
/**
|
|
76
|
-
* Whether to extract embedded images as base64 data URIs.
|
|
77
|
-
* When false, images are represented as `[Image]` placeholders.
|
|
78
|
-
* Default: false
|
|
79
|
-
*/
|
|
80
|
-
extractImages?: boolean;
|
|
81
|
-
}
|
|
82
|
-
|
|
83
|
-
/**
|
|
84
|
-
* Convert a .docx file to a MarkdownDocument.
|
|
85
|
-
*
|
|
86
|
-
* @param data - The raw .docx file as ArrayBuffer or Blob
|
|
87
|
-
* @param options - Import options
|
|
88
|
-
* @returns A MarkdownDocument representing the document content
|
|
89
|
-
*/
|
|
90
|
-
export async function docxToMarkdownDoc(
|
|
91
|
-
data: ArrayBuffer | Blob,
|
|
92
|
-
options: DocxImportOptions = {},
|
|
93
|
-
): Promise<MarkdownDocument> {
|
|
94
|
-
const pkg = await openPackage(data, options);
|
|
95
|
-
const ctx = await buildImportContext(pkg, options);
|
|
96
|
-
|
|
97
|
-
const documentXml = await getPartXml(pkg, 'word/document.xml');
|
|
98
|
-
if (!documentXml) {
|
|
99
|
-
return { type: 'document', children: [] };
|
|
100
|
-
}
|
|
101
|
-
|
|
102
|
-
const body = getFirstElement(documentXml, 'body');
|
|
103
|
-
if (!body) {
|
|
104
|
-
return { type: 'document', children: [] };
|
|
105
|
-
}
|
|
106
|
-
|
|
107
|
-
const blocks = await convertDocumentStories(body, ctx);
|
|
108
|
-
|
|
109
|
-
return { type: 'document', children: blocks };
|
|
110
|
-
}
|
|
111
|
-
|
|
112
|
-
/**
|
|
113
|
-
* Convert a .docx file to a squisq Doc.
|
|
114
|
-
*
|
|
115
|
-
* Convenience wrapper: DOCX → MarkdownDocument → Doc.
|
|
116
|
-
*
|
|
117
|
-
* @param data - The raw .docx file as ArrayBuffer or Blob
|
|
118
|
-
* @param options - Import options
|
|
119
|
-
* @returns A squisq Doc
|
|
120
|
-
*/
|
|
121
|
-
export async function docxToDoc(
|
|
122
|
-
data: ArrayBuffer | Blob,
|
|
123
|
-
options: DocxImportOptions = {},
|
|
124
|
-
): Promise<Doc> {
|
|
125
|
-
const markdownDoc = await docxToMarkdownDoc(data, options);
|
|
126
|
-
return markdownToDoc(markdownDoc);
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
/**
|
|
130
|
-
* Convert a .docx file to a ContentContainer with markdown + extracted images.
|
|
131
|
-
*
|
|
132
|
-
* The container will contain:
|
|
133
|
-
* - The primary markdown document (index.md)
|
|
134
|
-
* - Any embedded images under images/ (e.g., images/image1.png)
|
|
135
|
-
*
|
|
136
|
-
* @param data - The raw .docx file as ArrayBuffer or Blob
|
|
137
|
-
* @param options - Import options
|
|
138
|
-
* @returns A ContentContainer with the document and its media
|
|
139
|
-
*/
|
|
140
|
-
export async function docxToContainer(
|
|
141
|
-
data: ArrayBuffer | Blob,
|
|
142
|
-
options: DocxImportOptions = {},
|
|
143
|
-
): Promise<ContentContainer> {
|
|
144
|
-
const pkg = await openPackage(data, options);
|
|
145
|
-
const ctx = await buildImportContext(pkg, { ...options, extractImages: true });
|
|
146
|
-
|
|
147
|
-
const documentXml = await getPartXml(pkg, 'word/document.xml');
|
|
148
|
-
if (!documentXml) return buildContainer('', []);
|
|
149
|
-
|
|
150
|
-
const body = getFirstElement(documentXml, 'body');
|
|
151
|
-
if (!body) return buildContainer('', []);
|
|
152
|
-
|
|
153
|
-
const blocks = await convertDocumentStories(body, ctx);
|
|
154
|
-
const markdownDoc: MarkdownDocument = { type: 'document', children: blocks };
|
|
155
|
-
|
|
156
|
-
return buildContainer(stringifyMarkdown(markdownDoc), ctx.extractedImages);
|
|
157
|
-
}
|
|
158
|
-
|
|
159
|
-
// ============================================
|
|
160
|
-
// Import Context
|
|
161
|
-
// ============================================
|
|
162
|
-
|
|
163
|
-
interface ImportContext {
|
|
164
|
-
/** Style ID → heading depth mapping (from styles.xml) */
|
|
165
|
-
headingStyles: Map<string, number>;
|
|
166
|
-
/** Style IDs that represent blockquotes */
|
|
167
|
-
quoteStyles: Set<string>;
|
|
168
|
-
/** Style IDs that represent code blocks */
|
|
169
|
-
codeStyles: Set<string>;
|
|
170
|
-
/** Character style IDs that represent inline code */
|
|
171
|
-
inlineCodeStyles: Set<string>;
|
|
172
|
-
/** Document relationship map: rId → Relationship */
|
|
173
|
-
documentRels: Map<string, Relationship>;
|
|
174
|
-
/** Numbering definitions: numId → { levels: Map<ilvl, isOrdered> } */
|
|
175
|
-
numbering: Map<string, NumberingInfo>;
|
|
176
|
-
/** Footnote bodies: footnoteId → Element */
|
|
177
|
-
footnotes: Map<string, Element>;
|
|
178
|
-
/** Endnote bodies: endnoteId → Element */
|
|
179
|
-
endnotes: Map<string, Element>;
|
|
180
|
-
/** Part whose relationships are active while converting runs. */
|
|
181
|
-
currentPartPath: string;
|
|
182
|
-
/** Reference to the OOXML package (for extracting images) */
|
|
183
|
-
pkg: OoxmlPackage;
|
|
184
|
-
/** Import options */
|
|
185
|
-
options: DocxImportOptions;
|
|
186
|
-
/** Collected image files: relative path → { data, mimeType } */
|
|
187
|
-
extractedImages: Map<string, { data: ArrayBuffer; mimeType: string }>;
|
|
188
|
-
/** Counter for generating unique image filenames */
|
|
189
|
-
imageCounter: number;
|
|
190
|
-
}
|
|
191
|
-
|
|
192
|
-
interface NumberingInfo {
|
|
193
|
-
levels: Map<number, boolean>; // ilvl → isOrdered
|
|
194
|
-
}
|
|
195
|
-
|
|
196
|
-
async function buildImportContext(
|
|
197
|
-
pkg: OoxmlPackage,
|
|
198
|
-
options: DocxImportOptions,
|
|
199
|
-
): Promise<ImportContext> {
|
|
200
|
-
const ctx: ImportContext = {
|
|
201
|
-
headingStyles: new Map(),
|
|
202
|
-
quoteStyles: new Set(),
|
|
203
|
-
codeStyles: new Set(),
|
|
204
|
-
inlineCodeStyles: new Set(),
|
|
205
|
-
documentRels: new Map(),
|
|
206
|
-
numbering: new Map(),
|
|
207
|
-
footnotes: new Map(),
|
|
208
|
-
endnotes: new Map(),
|
|
209
|
-
currentPartPath: 'word/document.xml',
|
|
210
|
-
pkg,
|
|
211
|
-
options,
|
|
212
|
-
extractedImages: new Map(),
|
|
213
|
-
imageCounter: 0,
|
|
214
|
-
};
|
|
215
|
-
|
|
216
|
-
// Initialize with built-in defaults
|
|
217
|
-
for (const [id, depth] of Object.entries(HEADING_STYLE_MAP)) {
|
|
218
|
-
ctx.headingStyles.set(id, depth);
|
|
219
|
-
}
|
|
220
|
-
for (const id of QUOTE_STYLE_IDS) {
|
|
221
|
-
ctx.quoteStyles.add(id);
|
|
222
|
-
}
|
|
223
|
-
for (const id of CODE_STYLE_IDS) {
|
|
224
|
-
ctx.codeStyles.add(id);
|
|
225
|
-
}
|
|
226
|
-
for (const id of INLINE_CODE_STYLE_IDS) {
|
|
227
|
-
ctx.inlineCodeStyles.add(id);
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
// Parse styles.xml for custom heading/quote/code mappings
|
|
231
|
-
await parseStyles(pkg, ctx);
|
|
232
|
-
|
|
233
|
-
// Parse document relationships
|
|
234
|
-
const rels = await getPartRelationships(pkg, 'word/document.xml');
|
|
235
|
-
for (const rel of rels) {
|
|
236
|
-
ctx.documentRels.set(rel.id, rel);
|
|
237
|
-
}
|
|
238
|
-
|
|
239
|
-
// Parse numbering.xml
|
|
240
|
-
await parseNumbering(pkg, ctx);
|
|
241
|
-
|
|
242
|
-
// Parse footnotes.xml
|
|
243
|
-
await parseFootnotes(pkg, ctx);
|
|
244
|
-
|
|
245
|
-
// Parse endnotes.xml
|
|
246
|
-
await parseEndnotes(pkg, ctx);
|
|
247
|
-
|
|
248
|
-
return ctx;
|
|
249
|
-
}
|
|
250
|
-
|
|
251
|
-
// ============================================
|
|
252
|
-
// Styles Parsing
|
|
253
|
-
// ============================================
|
|
254
|
-
|
|
255
|
-
async function parseStyles(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
|
|
256
|
-
const doc = await getPartXml(pkg, 'word/styles.xml');
|
|
257
|
-
if (!doc) return;
|
|
258
|
-
|
|
259
|
-
const styles = doc.getElementsByTagNameNS(NS_WML, 'style');
|
|
260
|
-
// Fallback for documents that don't use namespace prefixes properly
|
|
261
|
-
const stylesList = styles.length > 0 ? styles : doc.getElementsByTagName('style');
|
|
262
|
-
|
|
263
|
-
for (let i = 0; i < stylesList.length; i++) {
|
|
264
|
-
const style = stylesList[i];
|
|
265
|
-
const styleId = style.getAttributeNS(NS_WML, 'styleId') ?? style.getAttribute('w:styleId');
|
|
266
|
-
if (!styleId) continue;
|
|
267
|
-
|
|
268
|
-
const nameEl = getFirstChildElement(style, 'name');
|
|
269
|
-
const styleName = nameEl?.getAttributeNS(NS_WML, 'val') ?? nameEl?.getAttribute('w:val') ?? '';
|
|
270
|
-
|
|
271
|
-
// Check if this is a heading style by name
|
|
272
|
-
const headingMatch = styleName.match(/^heading\s+(\d+)$/i);
|
|
273
|
-
if (headingMatch) {
|
|
274
|
-
const depth = parseInt(headingMatch[1], 10);
|
|
275
|
-
if (depth >= 1 && depth <= 6) {
|
|
276
|
-
ctx.headingStyles.set(styleId, depth);
|
|
277
|
-
}
|
|
278
|
-
}
|
|
279
|
-
|
|
280
|
-
// Check pPr > outlineLvl for heading detection
|
|
281
|
-
const pPr = getFirstChildElement(style, 'pPr');
|
|
282
|
-
if (pPr) {
|
|
283
|
-
const outlineLvl = getFirstChildElement(pPr, 'outlineLvl');
|
|
284
|
-
if (outlineLvl) {
|
|
285
|
-
const val = outlineLvl.getAttributeNS(NS_WML, 'val') ?? outlineLvl.getAttribute('w:val');
|
|
286
|
-
if (val !== null) {
|
|
287
|
-
const depth = parseInt(val, 10) + 1;
|
|
288
|
-
if (depth >= 1 && depth <= 6) {
|
|
289
|
-
ctx.headingStyles.set(styleId, depth);
|
|
290
|
-
}
|
|
291
|
-
}
|
|
292
|
-
}
|
|
293
|
-
}
|
|
294
|
-
}
|
|
295
|
-
}
|
|
296
|
-
|
|
297
|
-
// ============================================
|
|
298
|
-
// Numbering Parsing
|
|
299
|
-
// ============================================
|
|
300
|
-
|
|
301
|
-
async function parseNumbering(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
|
|
302
|
-
const doc = await getPartXml(pkg, 'word/numbering.xml');
|
|
303
|
-
if (!doc) return;
|
|
304
|
-
|
|
305
|
-
// Parse abstract numbering definitions
|
|
306
|
-
const abstractNums = new Map<string, Map<number, boolean>>(); // abstractNumId → levels(ilvl → isOrdered)
|
|
307
|
-
|
|
308
|
-
const abstractNumEls = getAllElements(doc, 'abstractNum');
|
|
309
|
-
for (const absNum of abstractNumEls) {
|
|
310
|
-
const absId = getAttr(absNum, 'abstractNumId');
|
|
311
|
-
if (!absId) continue;
|
|
312
|
-
|
|
313
|
-
const levels = new Map<number, boolean>();
|
|
314
|
-
const lvlEls = getAllChildElements(absNum, 'lvl');
|
|
315
|
-
for (const lvl of lvlEls) {
|
|
316
|
-
const ilvlStr = getAttr(lvl, 'ilvl');
|
|
317
|
-
if (ilvlStr === null) continue;
|
|
318
|
-
const ilvl = parseInt(ilvlStr, 10);
|
|
319
|
-
|
|
320
|
-
const numFmtEl = getFirstChildElement(lvl, 'numFmt');
|
|
321
|
-
const numFmt = numFmtEl ? getAttr(numFmtEl, 'val') : null;
|
|
322
|
-
|
|
323
|
-
const isOrdered = numFmt !== null && !BULLET_NUM_FORMATS.has(numFmt);
|
|
324
|
-
levels.set(ilvl, isOrdered);
|
|
325
|
-
}
|
|
326
|
-
|
|
327
|
-
abstractNums.set(absId, levels);
|
|
328
|
-
}
|
|
329
|
-
|
|
330
|
-
// Parse concrete num → abstractNum mappings
|
|
331
|
-
const numEls = getAllElements(doc, 'num');
|
|
332
|
-
for (const num of numEls) {
|
|
333
|
-
const numId = getAttr(num, 'numId');
|
|
334
|
-
if (!numId) continue;
|
|
335
|
-
|
|
336
|
-
const abstractNumIdEl = getFirstChildElement(num, 'abstractNumId');
|
|
337
|
-
const absId = abstractNumIdEl ? getAttr(abstractNumIdEl, 'val') : null;
|
|
338
|
-
if (!absId) continue;
|
|
339
|
-
|
|
340
|
-
const levels = abstractNums.get(absId);
|
|
341
|
-
if (levels) {
|
|
342
|
-
ctx.numbering.set(numId, { levels });
|
|
343
|
-
}
|
|
344
|
-
}
|
|
345
|
-
}
|
|
346
|
-
|
|
347
|
-
// ============================================
|
|
348
|
-
// Footnotes Parsing
|
|
349
|
-
// ============================================
|
|
350
|
-
|
|
351
|
-
async function parseFootnotes(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
|
|
352
|
-
const doc = await getPartXml(pkg, 'word/footnotes.xml');
|
|
353
|
-
if (!doc) return;
|
|
354
|
-
|
|
355
|
-
const footnoteEls = getAllElements(doc, 'footnote');
|
|
356
|
-
for (const fn of footnoteEls) {
|
|
357
|
-
const id = getAttr(fn, 'id');
|
|
358
|
-
const type = getAttr(fn, 'type');
|
|
359
|
-
// Skip separator and continuation separator footnotes
|
|
360
|
-
if (!id || type === 'separator' || type === 'continuationSeparator') continue;
|
|
361
|
-
ctx.footnotes.set(id, fn);
|
|
362
|
-
}
|
|
363
|
-
}
|
|
364
|
-
|
|
365
|
-
async function parseEndnotes(pkg: OoxmlPackage, ctx: ImportContext): Promise<void> {
|
|
366
|
-
const doc = await getPartXml(pkg, 'word/endnotes.xml');
|
|
367
|
-
if (!doc) return;
|
|
368
|
-
|
|
369
|
-
const endnoteEls = getAllElements(doc, 'endnote');
|
|
370
|
-
for (const note of endnoteEls) {
|
|
371
|
-
const id = getAttr(note, 'id');
|
|
372
|
-
const type = getAttr(note, 'type');
|
|
373
|
-
if (!id || type === 'separator' || type === 'continuationSeparator') continue;
|
|
374
|
-
ctx.endnotes.set(id, note);
|
|
375
|
-
}
|
|
376
|
-
}
|
|
377
|
-
|
|
378
|
-
// ============================================
|
|
379
|
-
// Body Conversion
|
|
380
|
-
// ============================================
|
|
381
|
-
|
|
382
|
-
/**
|
|
383
|
-
* Convert the main story plus the related header/footer stories. Header and
|
|
384
|
-
* footer blocks live in named directives so Markdown consumers can distinguish
|
|
385
|
-
* them from body content and the DOCX exporter can put them back in the right
|
|
386
|
-
* OOXML parts.
|
|
387
|
-
*/
|
|
388
|
-
async function convertDocumentStories(
|
|
389
|
-
body: Element,
|
|
390
|
-
ctx: ImportContext,
|
|
391
|
-
): Promise<MarkdownBlockNode[]> {
|
|
392
|
-
const bodyBlocks = await convertBody(body, ctx);
|
|
393
|
-
const headers = await convertRelatedStories(body, ctx, 'header');
|
|
394
|
-
const footers = await convertRelatedStories(body, ctx, 'footer');
|
|
395
|
-
const noteDefinitions = await convertNoteDefinitions(ctx);
|
|
396
|
-
return [...bodyBlocks, ...headers, ...footers, ...noteDefinitions];
|
|
397
|
-
}
|
|
398
|
-
|
|
399
|
-
async function convertRelatedStories(
|
|
400
|
-
body: Element,
|
|
401
|
-
ctx: ImportContext,
|
|
402
|
-
kind: 'header' | 'footer',
|
|
403
|
-
): Promise<MarkdownContainerDirective[]> {
|
|
404
|
-
const relationshipTypeSuffix = `/${kind}`;
|
|
405
|
-
const referenceName = `${kind}Reference`;
|
|
406
|
-
const referenceTypes = new Map<string, string>();
|
|
407
|
-
for (const reference of getAllElements(body, referenceName)) {
|
|
408
|
-
const id = reference.getAttributeNS(NS_R, 'id') ?? reference.getAttribute('r:id');
|
|
409
|
-
if (id) referenceTypes.set(id, getAttr(reference, 'type') ?? 'default');
|
|
410
|
-
}
|
|
411
|
-
|
|
412
|
-
const documentRelationships = ctx.documentRels;
|
|
413
|
-
const previousPartPath = ctx.currentPartPath;
|
|
414
|
-
const results: MarkdownContainerDirective[] = [];
|
|
415
|
-
|
|
416
|
-
for (const relationship of documentRelationships.values()) {
|
|
417
|
-
if (!relationship.type.endsWith(relationshipTypeSuffix)) continue;
|
|
418
|
-
|
|
419
|
-
const partPath = resolveTarget(baseDirOf('word/document.xml'), relationship.target);
|
|
420
|
-
const part = await getPartXml(ctx.pkg, partPath);
|
|
421
|
-
if (!part?.documentElement) continue;
|
|
422
|
-
|
|
423
|
-
const partRelationships = await getPartRelationships(ctx.pkg, partPath);
|
|
424
|
-
ctx.documentRels = new Map(partRelationships.map((rel) => [rel.id, rel]));
|
|
425
|
-
ctx.currentPartPath = partPath;
|
|
426
|
-
try {
|
|
427
|
-
const children = await convertBody(part.documentElement, ctx);
|
|
428
|
-
if (children.length === 0) continue;
|
|
429
|
-
results.push({
|
|
430
|
-
type: 'containerDirective',
|
|
431
|
-
name: `docx-${kind}`,
|
|
432
|
-
attributes: {
|
|
433
|
-
type: referenceTypes.get(relationship.id) ?? 'default',
|
|
434
|
-
source: partPath,
|
|
435
|
-
},
|
|
436
|
-
children,
|
|
437
|
-
});
|
|
438
|
-
} finally {
|
|
439
|
-
ctx.documentRels = documentRelationships;
|
|
440
|
-
ctx.currentPartPath = previousPartPath;
|
|
441
|
-
}
|
|
442
|
-
}
|
|
443
|
-
|
|
444
|
-
return results;
|
|
445
|
-
}
|
|
446
|
-
|
|
447
|
-
async function convertBody(body: Element, ctx: ImportContext): Promise<MarkdownBlockNode[]> {
|
|
448
|
-
return convertBlockElements(Array.from(body.children), ctx);
|
|
449
|
-
}
|
|
450
|
-
|
|
451
|
-
async function convertBlockElements(
|
|
452
|
-
children: Element[],
|
|
453
|
-
ctx: ImportContext,
|
|
454
|
-
): Promise<MarkdownBlockNode[]> {
|
|
455
|
-
const result: MarkdownBlockNode[] = [];
|
|
456
|
-
|
|
457
|
-
let i = 0;
|
|
458
|
-
while (i < children.length) {
|
|
459
|
-
const el = children[i];
|
|
460
|
-
const localName = el.localName;
|
|
461
|
-
|
|
462
|
-
if (localName === 'p') {
|
|
463
|
-
// Check if this is part of a list
|
|
464
|
-
const numPr = getNumPr(el);
|
|
465
|
-
if (numPr) {
|
|
466
|
-
// Collect consecutive list paragraphs
|
|
467
|
-
const { node, consumed } = await collectList(children, i, ctx);
|
|
468
|
-
result.push(node);
|
|
469
|
-
i += consumed;
|
|
470
|
-
continue;
|
|
471
|
-
}
|
|
472
|
-
|
|
473
|
-
const block = await convertParagraph(el, ctx);
|
|
474
|
-
if (block) {
|
|
475
|
-
result.push(block);
|
|
476
|
-
}
|
|
477
|
-
i++;
|
|
478
|
-
} else if (localName === 'tbl') {
|
|
479
|
-
const table = await convertTable(el, ctx);
|
|
480
|
-
if (table) result.push(table);
|
|
481
|
-
i++;
|
|
482
|
-
} else if (localName === 'sdt') {
|
|
483
|
-
const content = getFirstChildElement(el, 'sdtContent');
|
|
484
|
-
if (content) result.push(...(await convertBlockElements(Array.from(content.children), ctx)));
|
|
485
|
-
i++;
|
|
486
|
-
} else if (
|
|
487
|
-
localName === 'ins' ||
|
|
488
|
-
localName === 'moveTo' ||
|
|
489
|
-
localName === 'customXml' ||
|
|
490
|
-
localName === 'smartTag' ||
|
|
491
|
-
localName === 'fldSimple'
|
|
492
|
-
) {
|
|
493
|
-
result.push(...(await convertBlockElements(Array.from(el.children), ctx)));
|
|
494
|
-
i++;
|
|
495
|
-
} else if (localName === 'AlternateContent') {
|
|
496
|
-
const selected = getFirstChildElement(el, 'Choice') ?? getFirstChildElement(el, 'Fallback');
|
|
497
|
-
if (selected)
|
|
498
|
-
result.push(...(await convertBlockElements(Array.from(selected.children), ctx)));
|
|
499
|
-
i++;
|
|
500
|
-
} else {
|
|
501
|
-
// Skip unknown elements (sectPr, bookmarkStart, etc.)
|
|
502
|
-
i++;
|
|
503
|
-
}
|
|
504
|
-
}
|
|
505
|
-
|
|
506
|
-
return result;
|
|
507
|
-
}
|
|
508
|
-
|
|
509
|
-
// ============================================
|
|
510
|
-
// Paragraph Conversion
|
|
511
|
-
// ============================================
|
|
512
|
-
|
|
513
|
-
async function convertParagraph(
|
|
514
|
-
el: Element,
|
|
515
|
-
ctx: ImportContext,
|
|
516
|
-
): Promise<MarkdownBlockNode | null> {
|
|
517
|
-
const pPr = getFirstChildElement(el, 'pPr');
|
|
518
|
-
const styleId = getParagraphStyleId(pPr);
|
|
519
|
-
|
|
520
|
-
// Check for heading
|
|
521
|
-
if (styleId && ctx.headingStyles.has(styleId)) {
|
|
522
|
-
const depth = ctx.headingStyles.get(styleId)!;
|
|
523
|
-
const inlines = await convertRuns(el, ctx);
|
|
524
|
-
if (inlines.length === 0) return null;
|
|
525
|
-
return {
|
|
526
|
-
type: 'heading',
|
|
527
|
-
depth: Math.min(Math.max(depth, 1), 6) as 1 | 2 | 3 | 4 | 5 | 6,
|
|
528
|
-
children: inlines,
|
|
529
|
-
} satisfies MarkdownHeading;
|
|
530
|
-
}
|
|
531
|
-
|
|
532
|
-
// Check for blockquote
|
|
533
|
-
if (styleId && ctx.quoteStyles.has(styleId)) {
|
|
534
|
-
const inlines = await convertRuns(el, ctx);
|
|
535
|
-
if (inlines.length === 0) return null;
|
|
536
|
-
const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
|
|
537
|
-
return { type: 'blockquote', children: [paragraph] } satisfies MarkdownBlockquote;
|
|
538
|
-
}
|
|
539
|
-
|
|
540
|
-
// Check for code block
|
|
541
|
-
if (styleId && ctx.codeStyles.has(styleId)) {
|
|
542
|
-
const text = getElementTextContent(el);
|
|
543
|
-
return { type: 'code', value: text } satisfies MarkdownCodeBlock;
|
|
544
|
-
}
|
|
545
|
-
|
|
546
|
-
// Regular paragraph
|
|
547
|
-
const inlines = await convertRuns(el, ctx);
|
|
548
|
-
if (inlines.length === 0) return null;
|
|
549
|
-
return { type: 'paragraph', children: inlines } satisfies MarkdownParagraph;
|
|
550
|
-
}
|
|
551
|
-
|
|
552
|
-
// ============================================
|
|
553
|
-
// Run (Inline) Conversion
|
|
554
|
-
// ============================================
|
|
555
|
-
|
|
556
|
-
async function convertRuns(
|
|
557
|
-
paragraphEl: Element,
|
|
558
|
-
ctx: ImportContext,
|
|
559
|
-
): Promise<MarkdownInlineNode[]> {
|
|
560
|
-
return mergeAdjacentText(await convertInlineElements(Array.from(paragraphEl.children), ctx));
|
|
561
|
-
}
|
|
562
|
-
|
|
563
|
-
async function convertInlineElements(
|
|
564
|
-
children: Element[],
|
|
565
|
-
ctx: ImportContext,
|
|
566
|
-
): Promise<MarkdownInlineNode[]> {
|
|
567
|
-
const result: MarkdownInlineNode[] = [];
|
|
568
|
-
|
|
569
|
-
for (const child of children) {
|
|
570
|
-
const localName = child.localName;
|
|
571
|
-
|
|
572
|
-
if (localName === 'r') {
|
|
573
|
-
const inlines = await convertRun(child, ctx);
|
|
574
|
-
result.push(...inlines);
|
|
575
|
-
} else if (localName === 'hyperlink') {
|
|
576
|
-
const link = await convertHyperlink(child, ctx);
|
|
577
|
-
if (link) result.push(link);
|
|
578
|
-
} else if (localName === 'sdt') {
|
|
579
|
-
const content = getFirstChildElement(child, 'sdtContent');
|
|
580
|
-
if (content) {
|
|
581
|
-
result.push(...(await convertInlineElements(Array.from(content.children), ctx)));
|
|
582
|
-
}
|
|
583
|
-
} else if (
|
|
584
|
-
localName === 'ins' ||
|
|
585
|
-
localName === 'moveTo' ||
|
|
586
|
-
localName === 'customXml' ||
|
|
587
|
-
localName === 'smartTag' ||
|
|
588
|
-
localName === 'fldSimple'
|
|
589
|
-
) {
|
|
590
|
-
result.push(...(await convertInlineElements(Array.from(child.children), ctx)));
|
|
591
|
-
} else if (localName === 'AlternateContent') {
|
|
592
|
-
// Choice and Fallback represent the same content for different Word
|
|
593
|
-
// versions. Reading both duplicates every text box and legacy image.
|
|
594
|
-
const selected =
|
|
595
|
-
getFirstChildElement(child, 'Choice') ?? getFirstChildElement(child, 'Fallback');
|
|
596
|
-
if (selected) {
|
|
597
|
-
result.push(...(await convertInlineElements(Array.from(selected.children), ctx)));
|
|
598
|
-
}
|
|
599
|
-
} else if (localName === 'drawing' || localName === 'pict') {
|
|
600
|
-
result.push(...(await convertDrawingContent(child, ctx)));
|
|
601
|
-
}
|
|
602
|
-
// Skip pPr, bookmarkStart, bookmarkEnd, etc.
|
|
603
|
-
}
|
|
604
|
-
|
|
605
|
-
return result;
|
|
606
|
-
}
|
|
607
|
-
|
|
608
|
-
async function convertRun(runEl: Element, ctx: ImportContext): Promise<MarkdownInlineNode[]> {
|
|
609
|
-
const result: MarkdownInlineNode[] = [];
|
|
610
|
-
const rPr = getFirstChildElement(runEl, 'rPr');
|
|
611
|
-
const format = parseRunFormat(rPr, ctx);
|
|
612
|
-
|
|
613
|
-
for (const child of Array.from(runEl.children)) {
|
|
614
|
-
const localName = child.localName;
|
|
615
|
-
|
|
616
|
-
if (localName === 't') {
|
|
617
|
-
const text = child.textContent ?? '';
|
|
618
|
-
if (!text) continue;
|
|
619
|
-
|
|
620
|
-
if (format.code) {
|
|
621
|
-
result.push({ type: 'inlineCode', value: text } satisfies MarkdownInlineCode);
|
|
622
|
-
} else {
|
|
623
|
-
let node: MarkdownInlineNode = { type: 'text', value: text } satisfies MarkdownText;
|
|
624
|
-
if (format.strike) {
|
|
625
|
-
node = { type: 'delete', children: [node] } satisfies MarkdownStrikethrough;
|
|
626
|
-
}
|
|
627
|
-
if (format.italic) {
|
|
628
|
-
node = { type: 'emphasis', children: [node] } satisfies MarkdownEmphasis;
|
|
629
|
-
}
|
|
630
|
-
if (format.bold) {
|
|
631
|
-
node = { type: 'strong', children: [node] } satisfies MarkdownStrong;
|
|
632
|
-
}
|
|
633
|
-
result.push(node);
|
|
634
|
-
}
|
|
635
|
-
} else if (localName === 'br' || localName === 'cr') {
|
|
636
|
-
result.push({ type: 'break' } satisfies MarkdownBreak);
|
|
637
|
-
} else if (localName === 'tab') {
|
|
638
|
-
// Tabs are visible separators in Word. A literal tab is unstable when
|
|
639
|
-
// serialized through Markdown (it may become indentation), so retain the
|
|
640
|
-
// word boundary as a regular space.
|
|
641
|
-
result.push({ type: 'text', value: ' ' } satisfies MarkdownText);
|
|
642
|
-
} else if (localName === 'footnoteReference') {
|
|
643
|
-
const fnId = getAttr(child, 'id');
|
|
644
|
-
if (fnId && fnId !== '0' && fnId !== '-1') {
|
|
645
|
-
result.push({
|
|
646
|
-
type: 'footnoteReference',
|
|
647
|
-
identifier: `fn${fnId}`,
|
|
648
|
-
} satisfies MarkdownFootnoteReference);
|
|
649
|
-
}
|
|
650
|
-
} else if (localName === 'endnoteReference') {
|
|
651
|
-
const noteId = getAttr(child, 'id');
|
|
652
|
-
if (noteId && noteId !== '0' && noteId !== '-1') {
|
|
653
|
-
result.push({
|
|
654
|
-
type: 'footnoteReference',
|
|
655
|
-
identifier: `endnote${noteId}`,
|
|
656
|
-
} satisfies MarkdownFootnoteReference);
|
|
657
|
-
}
|
|
658
|
-
} else if (localName === 'drawing' || localName === 'pict' || localName === 'object') {
|
|
659
|
-
result.push(...(await convertDrawingContent(child, ctx)));
|
|
660
|
-
} else if (localName === 'AlternateContent') {
|
|
661
|
-
const selected =
|
|
662
|
-
getFirstChildElement(child, 'Choice') ?? getFirstChildElement(child, 'Fallback');
|
|
663
|
-
if (selected) {
|
|
664
|
-
result.push(...(await convertInlineElements(Array.from(selected.children), ctx)));
|
|
665
|
-
}
|
|
666
|
-
}
|
|
667
|
-
}
|
|
668
|
-
|
|
669
|
-
return result;
|
|
670
|
-
}
|
|
671
|
-
|
|
672
|
-
async function convertDrawingContent(
|
|
673
|
-
el: Element,
|
|
674
|
-
ctx: ImportContext,
|
|
675
|
-
): Promise<MarkdownInlineNode[]> {
|
|
676
|
-
const result: MarkdownInlineNode[] = [];
|
|
677
|
-
|
|
678
|
-
// A positioned Word shape may be a text box, an image, or both. Text box
|
|
679
|
-
// paragraphs are nested inside the drawing rather than being paragraph
|
|
680
|
-
// siblings, so the normal body walker never sees them.
|
|
681
|
-
const textBoxes = findDescendants(el, 'txbxContent');
|
|
682
|
-
for (const textBox of textBoxes) {
|
|
683
|
-
const inlines = await flattenContainerToInlines(textBox, ctx);
|
|
684
|
-
appendInlineGroup(result, inlines);
|
|
685
|
-
}
|
|
686
|
-
|
|
687
|
-
const image = await extractImage(el, ctx);
|
|
688
|
-
if (image) result.push(image);
|
|
689
|
-
if (textBoxes.length > 0 && result.length > 0) {
|
|
690
|
-
// Positioned text boxes are independent visual regions. Several can be
|
|
691
|
-
// anchored in the same otherwise-empty paragraph (for example, labels on
|
|
692
|
-
// a number line); hard boundaries prevent their text from collapsing into
|
|
693
|
-
// one synthetic word during Markdown serialization.
|
|
694
|
-
if (result[0]?.type !== 'break') result.unshift({ type: 'break' } satisfies MarkdownBreak);
|
|
695
|
-
if (result[result.length - 1]?.type !== 'break') {
|
|
696
|
-
result.push({ type: 'break' } satisfies MarkdownBreak);
|
|
697
|
-
}
|
|
698
|
-
}
|
|
699
|
-
return result;
|
|
700
|
-
}
|
|
701
|
-
|
|
702
|
-
async function flattenContainerToInlines(
|
|
703
|
-
container: Element,
|
|
704
|
-
ctx: ImportContext,
|
|
705
|
-
): Promise<MarkdownInlineNode[]> {
|
|
706
|
-
const result: MarkdownInlineNode[] = [];
|
|
707
|
-
|
|
708
|
-
for (const child of Array.from(container.children)) {
|
|
709
|
-
if (child.localName === 'p') {
|
|
710
|
-
appendInlineGroup(result, await convertRuns(child, ctx));
|
|
711
|
-
} else if (child.localName === 'tbl') {
|
|
712
|
-
appendInlineGroup(result, await flattenTableToInlines(child, ctx));
|
|
713
|
-
} else if (child.localName === 'sdt') {
|
|
714
|
-
const content = getFirstChildElement(child, 'sdtContent');
|
|
715
|
-
if (content) appendInlineGroup(result, await flattenContainerToInlines(content, ctx));
|
|
716
|
-
} else if (
|
|
717
|
-
child.localName === 'ins' ||
|
|
718
|
-
child.localName === 'moveTo' ||
|
|
719
|
-
child.localName === 'customXml' ||
|
|
720
|
-
child.localName === 'smartTag' ||
|
|
721
|
-
child.localName === 'fldSimple'
|
|
722
|
-
) {
|
|
723
|
-
appendInlineGroup(result, await flattenContainerToInlines(child, ctx));
|
|
724
|
-
} else if (child.localName === 'AlternateContent') {
|
|
725
|
-
const selected =
|
|
726
|
-
getFirstChildElement(child, 'Choice') ?? getFirstChildElement(child, 'Fallback');
|
|
727
|
-
if (selected) appendInlineGroup(result, await flattenContainerToInlines(selected, ctx));
|
|
728
|
-
} else if (child.localName === 'r' || child.localName === 'hyperlink') {
|
|
729
|
-
appendInlineGroup(result, await convertInlineElements([child], ctx));
|
|
730
|
-
}
|
|
731
|
-
}
|
|
732
|
-
|
|
733
|
-
return mergeAdjacentText(result);
|
|
734
|
-
}
|
|
735
|
-
|
|
736
|
-
async function flattenTableToInlines(
|
|
737
|
-
table: Element,
|
|
738
|
-
ctx: ImportContext,
|
|
739
|
-
): Promise<MarkdownInlineNode[]> {
|
|
740
|
-
const result: MarkdownInlineNode[] = [];
|
|
741
|
-
for (const row of getAllChildElements(table, 'tr')) {
|
|
742
|
-
for (const cell of getAllChildElements(row, 'tc')) {
|
|
743
|
-
appendInlineGroup(result, await flattenContainerToInlines(cell, ctx));
|
|
744
|
-
}
|
|
745
|
-
}
|
|
746
|
-
return mergeAdjacentText(result);
|
|
747
|
-
}
|
|
748
|
-
|
|
749
|
-
function appendInlineGroup(target: MarkdownInlineNode[], group: MarkdownInlineNode[]): void {
|
|
750
|
-
if (group.length === 0) return;
|
|
751
|
-
if (target.length > 0 && target[target.length - 1]?.type !== 'break') {
|
|
752
|
-
target.push({ type: 'break' } satisfies MarkdownBreak);
|
|
753
|
-
}
|
|
754
|
-
target.push(...group);
|
|
755
|
-
}
|
|
756
|
-
|
|
757
|
-
interface RunFormat {
|
|
758
|
-
bold: boolean;
|
|
759
|
-
italic: boolean;
|
|
760
|
-
strike: boolean;
|
|
761
|
-
code: boolean;
|
|
762
|
-
}
|
|
763
|
-
|
|
764
|
-
function parseRunFormat(rPr: Element | null, ctx: ImportContext): RunFormat {
|
|
765
|
-
if (!rPr) return { bold: false, italic: false, strike: false, code: false };
|
|
766
|
-
|
|
767
|
-
const bold = hasChildElement(rPr, 'b') && !isFalseToggle(getFirstChildElement(rPr, 'b')!);
|
|
768
|
-
const italic = hasChildElement(rPr, 'i') && !isFalseToggle(getFirstChildElement(rPr, 'i')!);
|
|
769
|
-
const strike =
|
|
770
|
-
hasChildElement(rPr, 'strike') && !isFalseToggle(getFirstChildElement(rPr, 'strike')!);
|
|
771
|
-
|
|
772
|
-
// Check for inline code via character style
|
|
773
|
-
const rStyle = getFirstChildElement(rPr, 'rStyle');
|
|
774
|
-
const charStyleId = rStyle ? getAttr(rStyle, 'val') : null;
|
|
775
|
-
const isCodeStyle = charStyleId ? ctx.inlineCodeStyles.has(charStyleId) : false;
|
|
776
|
-
|
|
777
|
-
// Check for monospace font as a code indicator
|
|
778
|
-
const rFonts = getFirstChildElement(rPr, 'rFonts');
|
|
779
|
-
const fontName = rFonts ? (getAttr(rFonts, 'ascii') ?? getAttr(rFonts, 'hAnsi') ?? '') : '';
|
|
780
|
-
const isMonospace = /consolas|courier|mono/i.test(fontName);
|
|
781
|
-
|
|
782
|
-
return { bold, italic, strike, code: isCodeStyle || isMonospace };
|
|
783
|
-
}
|
|
784
|
-
|
|
785
|
-
function isFalseToggle(el: Element): boolean {
|
|
786
|
-
const val = getAttr(el, 'val');
|
|
787
|
-
return val === '0' || val === 'false';
|
|
788
|
-
}
|
|
789
|
-
|
|
790
|
-
// ============================================
|
|
791
|
-
// Hyperlink Conversion
|
|
792
|
-
// ============================================
|
|
793
|
-
|
|
794
|
-
async function convertHyperlink(el: Element, ctx: ImportContext): Promise<MarkdownLink | null> {
|
|
795
|
-
const rId = el.getAttributeNS(NS_R, 'id') ?? el.getAttribute('r:id');
|
|
796
|
-
|
|
797
|
-
let url = '';
|
|
798
|
-
if (rId) {
|
|
799
|
-
const rel = ctx.documentRels.get(rId);
|
|
800
|
-
if (rel) url = rel.target;
|
|
801
|
-
}
|
|
802
|
-
|
|
803
|
-
// Also check for w:anchor (internal bookmarks)
|
|
804
|
-
if (!url) {
|
|
805
|
-
const anchor = el.getAttributeNS(NS_WML, 'anchor') ?? el.getAttribute('w:anchor');
|
|
806
|
-
if (anchor) url = `#${anchor}`;
|
|
807
|
-
}
|
|
808
|
-
|
|
809
|
-
const inlines = await convertInlineElements(Array.from(el.children), ctx);
|
|
810
|
-
|
|
811
|
-
if (inlines.length === 0) return null;
|
|
812
|
-
|
|
813
|
-
return {
|
|
814
|
-
type: 'link',
|
|
815
|
-
url,
|
|
816
|
-
children: mergeAdjacentText(inlines),
|
|
817
|
-
};
|
|
818
|
-
}
|
|
819
|
-
|
|
820
|
-
// ============================================
|
|
821
|
-
// Image Extraction
|
|
822
|
-
// ============================================
|
|
823
|
-
|
|
824
|
-
async function extractImage(el: Element, ctx: ImportContext): Promise<MarkdownImage | null> {
|
|
825
|
-
// DrawingML uses <a:blip r:embed="...">; older Word/VML documents use
|
|
826
|
-
// <v:imagedata r:id="...">. Supporting both also recovers many scanned
|
|
827
|
-
// forms and older templates in the corpus.
|
|
828
|
-
const blip = findDescendant(el, 'blip');
|
|
829
|
-
const imageData = findDescendant(el, 'imagedata');
|
|
830
|
-
const rId = blip
|
|
831
|
-
? (blip.getAttributeNS(NS_R, 'embed') ?? blip.getAttribute('r:embed'))
|
|
832
|
-
: (imageData?.getAttributeNS(NS_R, 'id') ??
|
|
833
|
-
imageData?.getAttribute('r:id') ??
|
|
834
|
-
imageData?.getAttribute('o:relid'));
|
|
835
|
-
if (!rId) return null;
|
|
836
|
-
|
|
837
|
-
const rel = ctx.documentRels.get(rId);
|
|
838
|
-
if (!rel || rel.targetMode === 'External') return null;
|
|
839
|
-
|
|
840
|
-
const target = resolveTarget(baseDirOf(ctx.currentPartPath), rel.target);
|
|
841
|
-
|
|
842
|
-
// Extract binary data from the zip
|
|
843
|
-
const data = await getPartBinary(ctx.pkg, target);
|
|
844
|
-
if (!data) return null;
|
|
845
|
-
|
|
846
|
-
// Determine extension and MIME type
|
|
847
|
-
const dot = target.lastIndexOf('.');
|
|
848
|
-
const ext = dot !== -1 ? target.slice(dot).toLowerCase() : '.png';
|
|
849
|
-
const mimeType = extToMime(ext);
|
|
850
|
-
|
|
851
|
-
// Generate a unique image path
|
|
852
|
-
ctx.imageCounter++;
|
|
853
|
-
const imagePath = `images/image${ctx.imageCounter}${ext}`;
|
|
854
|
-
|
|
855
|
-
// Store the extracted image data
|
|
856
|
-
ctx.extractedImages.set(imagePath, { data, mimeType });
|
|
857
|
-
|
|
858
|
-
// Try to extract alt text from the drawing's docPr element
|
|
859
|
-
const docPr = findDescendant(el, 'docPr');
|
|
860
|
-
const shape = findDescendant(el, 'shape');
|
|
861
|
-
const alt =
|
|
862
|
-
docPr?.getAttribute('descr') ||
|
|
863
|
-
docPr?.getAttribute('title') ||
|
|
864
|
-
shape?.getAttribute('alt') ||
|
|
865
|
-
shape?.getAttribute('title') ||
|
|
866
|
-
'Image';
|
|
867
|
-
|
|
868
|
-
return {
|
|
869
|
-
type: 'image',
|
|
870
|
-
url: imagePath,
|
|
871
|
-
alt,
|
|
872
|
-
};
|
|
873
|
-
}
|
|
874
|
-
|
|
875
|
-
/** Recursively find the first descendant element with the given local name. */
|
|
876
|
-
function findDescendant(el: Element, localName: string): Element | null {
|
|
877
|
-
for (const child of Array.from(el.children)) {
|
|
878
|
-
if (child.localName === localName) return child;
|
|
879
|
-
const found = findDescendant(child, localName);
|
|
880
|
-
if (found) return found;
|
|
881
|
-
}
|
|
882
|
-
return null;
|
|
883
|
-
}
|
|
884
|
-
|
|
885
|
-
/** Recursively find every descendant element with the given local name. */
|
|
886
|
-
function findDescendants(el: Element, localName: string): Element[] {
|
|
887
|
-
const results: Element[] = [];
|
|
888
|
-
for (const child of Array.from(el.children)) {
|
|
889
|
-
if (child.localName === localName) results.push(child);
|
|
890
|
-
results.push(...findDescendants(child, localName));
|
|
891
|
-
}
|
|
892
|
-
return results;
|
|
893
|
-
}
|
|
894
|
-
|
|
895
|
-
// ============================================
|
|
896
|
-
// List Collection
|
|
897
|
-
// ============================================
|
|
898
|
-
|
|
899
|
-
interface ListResult {
|
|
900
|
-
node: MarkdownList;
|
|
901
|
-
consumed: number;
|
|
902
|
-
}
|
|
903
|
-
|
|
904
|
-
interface NumPrInfo {
|
|
905
|
-
numId: string;
|
|
906
|
-
ilvl: number;
|
|
907
|
-
}
|
|
908
|
-
|
|
909
|
-
async function collectList(
|
|
910
|
-
elements: Element[],
|
|
911
|
-
startIdx: number,
|
|
912
|
-
ctx: ImportContext,
|
|
913
|
-
): Promise<ListResult> {
|
|
914
|
-
const firstNumPr = getNumPr(elements[startIdx])!;
|
|
915
|
-
const { numId } = firstNumPr;
|
|
916
|
-
|
|
917
|
-
// Determine if ordered from numbering definition
|
|
918
|
-
const numInfo = ctx.numbering.get(numId);
|
|
919
|
-
const isOrdered = numInfo?.levels.get(0) ?? false;
|
|
920
|
-
|
|
921
|
-
const items: MarkdownListItem[] = [];
|
|
922
|
-
let consumed = 0;
|
|
923
|
-
|
|
924
|
-
let i = startIdx;
|
|
925
|
-
while (i < elements.length) {
|
|
926
|
-
const el = elements[i];
|
|
927
|
-
if (el.localName !== 'p') break;
|
|
928
|
-
|
|
929
|
-
const numPr = getNumPr(el);
|
|
930
|
-
if (!numPr || numPr.numId !== numId) break;
|
|
931
|
-
|
|
932
|
-
// Convert this paragraph's inline content
|
|
933
|
-
const inlines = await convertRuns(el, ctx);
|
|
934
|
-
if (inlines.length > 0) {
|
|
935
|
-
const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
|
|
936
|
-
|
|
937
|
-
// Check if this is a nested list item
|
|
938
|
-
if (numPr.ilvl > firstNumPr.ilvl) {
|
|
939
|
-
// Collect nested list items
|
|
940
|
-
const nested = await collectNestedList(elements, i, ctx, firstNumPr.ilvl);
|
|
941
|
-
if (items.length > 0) {
|
|
942
|
-
// Attach nested list to the last item
|
|
943
|
-
const lastItem = items[items.length - 1];
|
|
944
|
-
const nestedIsOrdered = numInfo?.levels.get(numPr.ilvl) ?? false;
|
|
945
|
-
const nestedList: MarkdownList = {
|
|
946
|
-
type: 'list',
|
|
947
|
-
ordered: nestedIsOrdered,
|
|
948
|
-
children: nested.items,
|
|
949
|
-
};
|
|
950
|
-
lastItem.children.push(nestedList);
|
|
951
|
-
} else {
|
|
952
|
-
// Some Word producers emit an empty parent-level numbering
|
|
953
|
-
// paragraph before the first real (more deeply indented) item. With
|
|
954
|
-
// no parent item to attach to, retain those items at this level
|
|
955
|
-
// instead of silently discarding the entire nested subtree.
|
|
956
|
-
items.push(...nested.items);
|
|
957
|
-
}
|
|
958
|
-
i += nested.consumed;
|
|
959
|
-
consumed += nested.consumed;
|
|
960
|
-
continue;
|
|
961
|
-
}
|
|
962
|
-
|
|
963
|
-
items.push({
|
|
964
|
-
type: 'listItem',
|
|
965
|
-
children: [paragraph],
|
|
966
|
-
});
|
|
967
|
-
}
|
|
968
|
-
|
|
969
|
-
i++;
|
|
970
|
-
consumed++;
|
|
971
|
-
}
|
|
972
|
-
|
|
973
|
-
return {
|
|
974
|
-
node: {
|
|
975
|
-
type: 'list',
|
|
976
|
-
ordered: isOrdered,
|
|
977
|
-
children: items,
|
|
978
|
-
},
|
|
979
|
-
consumed,
|
|
980
|
-
};
|
|
981
|
-
}
|
|
982
|
-
|
|
983
|
-
interface NestedListResult {
|
|
984
|
-
items: MarkdownListItem[];
|
|
985
|
-
consumed: number;
|
|
986
|
-
}
|
|
987
|
-
|
|
988
|
-
async function collectNestedList(
|
|
989
|
-
elements: Element[],
|
|
990
|
-
startIdx: number,
|
|
991
|
-
ctx: ImportContext,
|
|
992
|
-
parentIlvl: number,
|
|
993
|
-
): Promise<NestedListResult> {
|
|
994
|
-
const items: MarkdownListItem[] = [];
|
|
995
|
-
let consumed = 0;
|
|
996
|
-
|
|
997
|
-
let i = startIdx;
|
|
998
|
-
while (i < elements.length) {
|
|
999
|
-
const el = elements[i];
|
|
1000
|
-
if (el.localName !== 'p') break;
|
|
1001
|
-
|
|
1002
|
-
const numPr = getNumPr(el);
|
|
1003
|
-
if (!numPr) break;
|
|
1004
|
-
if (numPr.ilvl <= parentIlvl) break;
|
|
1005
|
-
|
|
1006
|
-
const inlines = await convertRuns(el, ctx);
|
|
1007
|
-
if (inlines.length > 0) {
|
|
1008
|
-
const paragraph: MarkdownParagraph = { type: 'paragraph', children: inlines };
|
|
1009
|
-
items.push({ type: 'listItem', children: [paragraph] });
|
|
1010
|
-
}
|
|
1011
|
-
|
|
1012
|
-
i++;
|
|
1013
|
-
consumed++;
|
|
1014
|
-
}
|
|
1015
|
-
|
|
1016
|
-
return { items, consumed };
|
|
1017
|
-
}
|
|
1018
|
-
|
|
1019
|
-
function getNumPr(el: Element): NumPrInfo | null {
|
|
1020
|
-
const pPr = getFirstChildElement(el, 'pPr');
|
|
1021
|
-
if (!pPr) return null;
|
|
1022
|
-
|
|
1023
|
-
const numPr = getFirstChildElement(pPr, 'numPr');
|
|
1024
|
-
if (!numPr) return null;
|
|
1025
|
-
|
|
1026
|
-
const ilvlEl = getFirstChildElement(numPr, 'ilvl');
|
|
1027
|
-
const numIdEl = getFirstChildElement(numPr, 'numId');
|
|
1028
|
-
|
|
1029
|
-
const ilvlVal = ilvlEl ? getAttr(ilvlEl, 'val') : null;
|
|
1030
|
-
const numIdVal = numIdEl ? getAttr(numIdEl, 'val') : null;
|
|
1031
|
-
|
|
1032
|
-
if (!numIdVal || numIdVal === '0') return null; // numId 0 means "no list"
|
|
1033
|
-
|
|
1034
|
-
return {
|
|
1035
|
-
numId: numIdVal,
|
|
1036
|
-
ilvl: ilvlVal ? parseInt(ilvlVal, 10) : 0,
|
|
1037
|
-
};
|
|
1038
|
-
}
|
|
1039
|
-
|
|
1040
|
-
// ============================================
|
|
1041
|
-
// Table Conversion
|
|
1042
|
-
// ============================================
|
|
1043
|
-
|
|
1044
|
-
async function convertTable(tblEl: Element, ctx: ImportContext): Promise<MarkdownTable | null> {
|
|
1045
|
-
const rows: MarkdownTableRow[] = [];
|
|
1046
|
-
|
|
1047
|
-
const trEls = getAllChildElements(tblEl, 'tr');
|
|
1048
|
-
for (let ri = 0; ri < trEls.length; ri++) {
|
|
1049
|
-
const row = await convertTableRow(trEls[ri], ctx, ri === 0);
|
|
1050
|
-
rows.push(row);
|
|
1051
|
-
}
|
|
1052
|
-
|
|
1053
|
-
if (rows.length === 0) return null;
|
|
1054
|
-
|
|
1055
|
-
// If first row isn't explicitly a header, treat it as one anyway
|
|
1056
|
-
// (Markdown tables require a header row)
|
|
1057
|
-
return {
|
|
1058
|
-
type: 'table',
|
|
1059
|
-
children: rows,
|
|
1060
|
-
};
|
|
1061
|
-
}
|
|
1062
|
-
|
|
1063
|
-
async function convertTableRow(
|
|
1064
|
-
trEl: Element,
|
|
1065
|
-
ctx: ImportContext,
|
|
1066
|
-
isHeader: boolean,
|
|
1067
|
-
): Promise<MarkdownTableRow> {
|
|
1068
|
-
const cells: MarkdownTableCell[] = [];
|
|
1069
|
-
const tcEls = getAllChildElements(trEl, 'tc');
|
|
1070
|
-
|
|
1071
|
-
for (const tc of tcEls) {
|
|
1072
|
-
const cell = await convertTableCell(tc, ctx, isHeader);
|
|
1073
|
-
cells.push(cell);
|
|
1074
|
-
}
|
|
1075
|
-
|
|
1076
|
-
return { type: 'tableRow', children: cells };
|
|
1077
|
-
}
|
|
1078
|
-
|
|
1079
|
-
async function convertTableCell(
|
|
1080
|
-
tcEl: Element,
|
|
1081
|
-
ctx: ImportContext,
|
|
1082
|
-
isHeader: boolean,
|
|
1083
|
-
): Promise<MarkdownTableCell> {
|
|
1084
|
-
// Cells may contain paragraphs, content controls, and recursively nested
|
|
1085
|
-
// tables (a common layout technique in Word forms). Markdown tables cannot
|
|
1086
|
-
// nest, so preserve all cell text in reading order separated by hard breaks.
|
|
1087
|
-
const inlines = await flattenContainerToInlines(tcEl, ctx);
|
|
1088
|
-
|
|
1089
|
-
return {
|
|
1090
|
-
type: 'tableCell',
|
|
1091
|
-
isHeader,
|
|
1092
|
-
children: mergeAdjacentText(inlines),
|
|
1093
|
-
};
|
|
1094
|
-
}
|
|
1095
|
-
|
|
1096
|
-
// ============================================
|
|
1097
|
-
// Footnote Definition Conversion
|
|
1098
|
-
// ============================================
|
|
1099
|
-
|
|
1100
|
-
async function convertNoteDefinitions(ctx: ImportContext): Promise<MarkdownFootnoteDefinition[]> {
|
|
1101
|
-
return [
|
|
1102
|
-
...(await convertNotePartDefinitions(ctx, ctx.footnotes, 'fn', 'word/footnotes.xml')),
|
|
1103
|
-
...(await convertNotePartDefinitions(ctx, ctx.endnotes, 'endnote', 'word/endnotes.xml')),
|
|
1104
|
-
];
|
|
1105
|
-
}
|
|
1106
|
-
|
|
1107
|
-
async function convertNotePartDefinitions(
|
|
1108
|
-
ctx: ImportContext,
|
|
1109
|
-
notes: Map<string, Element>,
|
|
1110
|
-
identifierPrefix: string,
|
|
1111
|
-
partPath: string,
|
|
1112
|
-
): Promise<MarkdownFootnoteDefinition[]> {
|
|
1113
|
-
const results: MarkdownFootnoteDefinition[] = [];
|
|
1114
|
-
const previousRelationships = ctx.documentRels;
|
|
1115
|
-
const previousPartPath = ctx.currentPartPath;
|
|
1116
|
-
const partRelationships = await getPartRelationships(ctx.pkg, partPath);
|
|
1117
|
-
ctx.documentRels = new Map(partRelationships.map((rel) => [rel.id, rel]));
|
|
1118
|
-
ctx.currentPartPath = partPath;
|
|
1119
|
-
|
|
1120
|
-
try {
|
|
1121
|
-
for (const [id, el] of notes) {
|
|
1122
|
-
const children = await convertBody(el, ctx);
|
|
1123
|
-
if (children.length > 0) {
|
|
1124
|
-
results.push({
|
|
1125
|
-
type: 'footnoteDefinition',
|
|
1126
|
-
identifier: `${identifierPrefix}${id}`,
|
|
1127
|
-
children,
|
|
1128
|
-
});
|
|
1129
|
-
}
|
|
1130
|
-
}
|
|
1131
|
-
} finally {
|
|
1132
|
-
ctx.documentRels = previousRelationships;
|
|
1133
|
-
ctx.currentPartPath = previousPartPath;
|
|
1134
|
-
}
|
|
1135
|
-
|
|
1136
|
-
return results;
|
|
1137
|
-
}
|
|
1138
|
-
|
|
1139
|
-
// ============================================
|
|
1140
|
-
// XML Helper Utilities
|
|
1141
|
-
// ============================================
|
|
1142
|
-
|
|
1143
|
-
/**
|
|
1144
|
-
* Get the first element child with a given local name.
|
|
1145
|
-
* Handles both namespaced and non-namespaced elements.
|
|
1146
|
-
*/
|
|
1147
|
-
function getFirstChildElement(parent: Element | Document, localName: string): Element | null {
|
|
1148
|
-
for (const child of Array.from(parent.children ?? [])) {
|
|
1149
|
-
if (child.localName === localName) return child;
|
|
1150
|
-
}
|
|
1151
|
-
return null;
|
|
1152
|
-
}
|
|
1153
|
-
|
|
1154
|
-
/**
|
|
1155
|
-
* Get all direct child elements with a given local name.
|
|
1156
|
-
*/
|
|
1157
|
-
function getAllChildElements(parent: Element, localName: string): Element[] {
|
|
1158
|
-
const result: Element[] = [];
|
|
1159
|
-
for (const child of Array.from(parent.children)) {
|
|
1160
|
-
if (child.localName === localName) result.push(child);
|
|
1161
|
-
}
|
|
1162
|
-
return result;
|
|
1163
|
-
}
|
|
1164
|
-
|
|
1165
|
-
/**
|
|
1166
|
-
* Get all elements with a given local name in the document.
|
|
1167
|
-
*/
|
|
1168
|
-
function getAllElements(doc: Document | Element, localName: string): Element[] {
|
|
1169
|
-
// Try namespace-aware first
|
|
1170
|
-
const nsEls =
|
|
1171
|
-
'getElementsByTagNameNS' in doc ? doc.getElementsByTagNameNS(NS_WML, localName) : null;
|
|
1172
|
-
if (nsEls && nsEls.length > 0) return Array.from(nsEls);
|
|
1173
|
-
|
|
1174
|
-
// Fallback: try with w: prefix
|
|
1175
|
-
const prefixed = doc.getElementsByTagName(`w:${localName}`);
|
|
1176
|
-
if (prefixed.length > 0) return Array.from(prefixed);
|
|
1177
|
-
|
|
1178
|
-
// Final fallback: bare name
|
|
1179
|
-
return Array.from(doc.getElementsByTagName(localName));
|
|
1180
|
-
}
|
|
1181
|
-
|
|
1182
|
-
/**
|
|
1183
|
-
* Get the first element with given local name in the document or subtree.
|
|
1184
|
-
*/
|
|
1185
|
-
function getFirstElement(doc: Document | Element, localName: string): Element | null {
|
|
1186
|
-
const els = getAllElements(doc, localName);
|
|
1187
|
-
return els.length > 0 ? els[0] : null;
|
|
1188
|
-
}
|
|
1189
|
-
|
|
1190
|
-
/**
|
|
1191
|
-
* Get a w:-prefixed attribute, trying namespace-aware first then fallback.
|
|
1192
|
-
*/
|
|
1193
|
-
function getAttr(el: Element, localName: string): string | null {
|
|
1194
|
-
return el.getAttributeNS(NS_WML, localName) || el.getAttribute(`w:${localName}`) || null;
|
|
1195
|
-
}
|
|
1196
|
-
|
|
1197
|
-
/**
|
|
1198
|
-
* Check if an element has a direct child with the given local name.
|
|
1199
|
-
*/
|
|
1200
|
-
function hasChildElement(parent: Element, localName: string): boolean {
|
|
1201
|
-
return getFirstChildElement(parent, localName) !== null;
|
|
1202
|
-
}
|
|
1203
|
-
|
|
1204
|
-
/**
|
|
1205
|
-
* Get the paragraph style ID from a pPr element.
|
|
1206
|
-
*/
|
|
1207
|
-
function getParagraphStyleId(pPr: Element | null): string | null {
|
|
1208
|
-
if (!pPr) return null;
|
|
1209
|
-
const pStyle = getFirstChildElement(pPr, 'pStyle');
|
|
1210
|
-
if (!pStyle) return null;
|
|
1211
|
-
return getAttr(pStyle, 'val');
|
|
1212
|
-
}
|
|
1213
|
-
|
|
1214
|
-
/**
|
|
1215
|
-
* Get all text content from an element (concatenating all w:t descendants).
|
|
1216
|
-
*/
|
|
1217
|
-
function getElementTextContent(el: Element): string {
|
|
1218
|
-
const parts: string[] = [];
|
|
1219
|
-
|
|
1220
|
-
function walk(node: Element): void {
|
|
1221
|
-
if (node.localName === 't') {
|
|
1222
|
-
parts.push(node.textContent ?? '');
|
|
1223
|
-
}
|
|
1224
|
-
for (const child of Array.from(node.children)) {
|
|
1225
|
-
walk(child);
|
|
1226
|
-
}
|
|
1227
|
-
}
|
|
1228
|
-
|
|
1229
|
-
walk(el);
|
|
1230
|
-
return parts.join('');
|
|
1231
|
-
}
|
|
1232
|
-
|
|
1233
|
-
/**
|
|
1234
|
-
* Merge adjacent MarkdownText nodes to reduce fragmentation.
|
|
1235
|
-
*/
|
|
1236
|
-
function mergeAdjacentText(nodes: MarkdownInlineNode[]): MarkdownInlineNode[] {
|
|
1237
|
-
if (nodes.length <= 1) return nodes;
|
|
1238
|
-
|
|
1239
|
-
const result: MarkdownInlineNode[] = [];
|
|
1240
|
-
for (const node of nodes) {
|
|
1241
|
-
const prev = result[result.length - 1];
|
|
1242
|
-
if (node.type === 'text' && prev?.type === 'text') {
|
|
1243
|
-
// Merge into previous text node
|
|
1244
|
-
(prev as MarkdownText).value += (node as MarkdownText).value;
|
|
1245
|
-
} else {
|
|
1246
|
-
result.push(node);
|
|
1247
|
-
}
|
|
1248
|
-
}
|
|
1249
|
-
return result;
|
|
1250
|
-
}
|