@stll/folio-core 0.44.0 → 0.46.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (263) hide show
  1. package/dist/ai-edits/__fixtures__/paragraphs.js +2 -2
  2. package/dist/ai-edits/apply.js +59 -19
  3. package/dist/ai-edits/headless.js +2 -1
  4. package/dist/compare/compare.js +2 -5
  5. package/dist/compare/content-alignment.js +78 -53
  6. package/dist/compare/inline-atoms.js +34 -20
  7. package/dist/compare/scenario.js +2 -5
  8. package/dist/compare/types.d.ts +9 -1
  9. package/dist/compare/types.js +12 -1
  10. package/dist/compat/eigenpal.d.ts +2 -1
  11. package/dist/compat/eigenpal.js +2 -1
  12. package/dist/content-controls/checkboxDisplay.d.ts +11 -0
  13. package/dist/content-controls/checkboxDisplay.js +40 -0
  14. package/dist/content-controls/findContentControls.js +3 -1
  15. package/dist/content-controls/mutateContentControls.js +21 -4
  16. package/dist/controller/contentControlWidgetController.d.ts +1 -0
  17. package/dist/controller/contentControlWidgetController.js +3 -0
  18. package/dist/controller/hiddenEditorApi.js +3 -1
  19. package/dist/controller/hiddenEditorManager.d.ts +1 -1
  20. package/dist/controller/hiddenEditorManager.js +6 -3
  21. package/dist/display-list/build/buildDisplayList.js +20 -1
  22. package/dist/display-list/build/imagePrimitives.d.ts +1 -1
  23. package/dist/display-list/build/imagePrimitives.js +3 -1
  24. package/dist/display-list/build/paragraphPrimitives.js +21 -2
  25. package/dist/display-list/dom/renderDisplayListToDom.js +18 -10
  26. package/dist/display-list/types.d.ts +17 -0
  27. package/dist/document-operations.js +14 -3
  28. package/dist/docx/attributeRemainder.d.ts +43 -0
  29. package/dist/docx/attributeRemainder.js +65 -0
  30. package/dist/docx/blockContentParser.d.ts +1 -1
  31. package/dist/docx/blockContentParser.js +87 -63
  32. package/dist/docx/blockPlainText.d.ts +3 -4
  33. package/dist/docx/blockPlainText.js +1 -0
  34. package/dist/docx/bookmarkPlacement.js +2 -0
  35. package/dist/docx/borderParser.js +5 -5
  36. package/dist/docx/commentAnchorIndex.d.ts +30 -0
  37. package/dist/docx/commentAnchorIndex.js +50 -0
  38. package/dist/docx/commentParser.d.ts +1 -1
  39. package/dist/docx/commentParser.js +78 -50
  40. package/dist/docx/commentRangeIntegrity.d.ts +16 -1
  41. package/dist/docx/commentRangeIntegrity.js +42 -1
  42. package/dist/docx/commentRangeJoin.d.ts +9 -0
  43. package/dist/docx/commentRangeJoin.js +33 -0
  44. package/dist/docx/commentReferenceCompletion.d.ts +9 -0
  45. package/dist/docx/commentReferenceCompletion.js +47 -0
  46. package/dist/docx/commentReferenceNormalization.js +1 -0
  47. package/dist/docx/commentReplyMarkers.js +11 -19
  48. package/dist/docx/commentThreadKey.d.ts +18 -0
  49. package/dist/docx/commentThreadKey.js +22 -0
  50. package/dist/docx/compatibility.js +1 -0
  51. package/dist/docx/containerChildren.d.ts +111 -0
  52. package/dist/docx/containerChildren.gen.d.ts +28 -0
  53. package/dist/docx/containerChildren.gen.js +246 -0
  54. package/dist/docx/containerChildren.js +103 -0
  55. package/dist/docx/diagramPreview.js +87 -27
  56. package/dist/docx/documentParser.d.ts +1 -1
  57. package/dist/docx/documentParser.js +3 -3
  58. package/dist/docx/ensureParaIds.js +17 -7
  59. package/dist/docx/fieldParser.d.ts +1 -1
  60. package/dist/docx/fieldParser.js +4 -4
  61. package/dist/docx/fieldState.d.ts +26 -0
  62. package/dist/docx/fieldState.js +53 -0
  63. package/dist/docx/footnoteParser.d.ts +1 -1
  64. package/dist/docx/graphicFrameLocks.d.ts +16 -4
  65. package/dist/docx/graphicFrameLocks.js +19 -6
  66. package/dist/docx/groupDrawingParser.js +3 -3
  67. package/dist/docx/headerFooterParser.js +1 -1
  68. package/dist/docx/headerFooterReferenceNormalization.js +1 -0
  69. package/dist/docx/hyperlinkParser.d.ts +27 -8
  70. package/dist/docx/hyperlinkParser.js +78 -20
  71. package/dist/docx/imageParser.d.ts +9 -1
  72. package/dist/docx/imageParser.js +91 -20
  73. package/dist/docx/imageRawXml.d.ts +14 -1
  74. package/dist/docx/imageRawXml.js +30 -6
  75. package/dist/docx/inlineWrapperContent.d.ts +85 -0
  76. package/dist/docx/inlineWrapperContent.js +84 -0
  77. package/dist/docx/mathToMathml.js +12 -14
  78. package/dist/docx/nonVisualDrawingProps.d.ts +34 -0
  79. package/dist/docx/nonVisualDrawingProps.js +46 -0
  80. package/dist/docx/normalizeBaseDirection.js +10 -1
  81. package/dist/docx/paraIdAttribute.d.ts +21 -0
  82. package/dist/docx/paraIdAttribute.js +64 -0
  83. package/dist/docx/paragraphParser.d.ts +1 -1
  84. package/dist/docx/paragraphParser.js +277 -168
  85. package/dist/docx/paragraphPropertySource.js +5 -1
  86. package/dist/docx/paragraphTextBoxEnrichment.d.ts +1 -1
  87. package/dist/docx/paragraphTextBoxEnrichment.js +9 -52
  88. package/dist/docx/paragraphTraversal.js +7 -4
  89. package/dist/docx/parseWarningMessage.js +2 -1
  90. package/dist/docx/parser.js +2 -2
  91. package/dist/docx/preservedRunContent.d.ts +32 -0
  92. package/dist/docx/preservedRunContent.js +86 -0
  93. package/dist/docx/previewBudget.d.ts +64 -0
  94. package/dist/docx/previewBudget.js +88 -0
  95. package/dist/docx/renderedPageBreakNormalization.js +2 -4
  96. package/dist/docx/revisionIdNormalization.js +17 -5
  97. package/dist/docx/rezip.js +91 -36
  98. package/dist/docx/runParser.d.ts +1 -1
  99. package/dist/docx/runParser.js +16 -12
  100. package/dist/docx/sdtPropertiesPatch.js +24 -18
  101. package/dist/docx/sectionParser.js +8 -1
  102. package/dist/docx/sectionReferenceHistory.js +2 -2
  103. package/dist/docx/selectiveSave.js +6 -6
  104. package/dist/docx/selectiveXmlPatch.d.ts +46 -2
  105. package/dist/docx/selectiveXmlPatch.js +86 -39
  106. package/dist/docx/serializer/blockSdtSerializer.js +38 -26
  107. package/dist/docx/serializer/borderSerializer.d.ts +1 -1
  108. package/dist/docx/serializer/borderSerializer.js +15 -14
  109. package/dist/docx/serializer/commentSerializer.d.ts +41 -16
  110. package/dist/docx/serializer/commentSerializer.js +87 -92
  111. package/dist/docx/serializer/documentSerializer.js +8 -11
  112. package/dist/docx/serializer/fontTableSerializer.js +6 -6
  113. package/dist/docx/serializer/headerFooterSerializer.js +12 -14
  114. package/dist/docx/serializer/markupRangeAttributes.js +2 -2
  115. package/dist/docx/serializer/noteSerializer.js +8 -8
  116. package/dist/docx/serializer/numberingSerializer.js +7 -6
  117. package/dist/docx/serializer/paragraphSerializer.js +93 -72
  118. package/dist/docx/serializer/partNamespaces.js +2 -2
  119. package/dist/docx/serializer/runSerializer.js +73 -44
  120. package/dist/docx/serializer/sectionPropertiesSerializer.js +19 -14
  121. package/dist/docx/serializer/settingsSerializer.js +4 -3
  122. package/dist/docx/serializer/stylesSerializer.js +6 -6
  123. package/dist/docx/serializer/tableSerializer.js +28 -17
  124. package/dist/docx/serializer/textFormattingSerializer.js +29 -28
  125. package/dist/docx/serializer/themeSerializer.js +6 -6
  126. package/dist/docx/serializer/trackedChangeAttributes.js +2 -2
  127. package/dist/docx/serializer/xmlUtils.d.ts +1 -2
  128. package/dist/docx/serializer/xmlUtils.js +1 -13
  129. package/dist/docx/server/boundedArchive.d.ts +12 -0
  130. package/dist/docx/server/boundedArchive.js +20 -1
  131. package/dist/docx/server/createBilingualDocument.js +3 -1
  132. package/dist/docx/server/materializeYjsDocx.d.ts +1 -1
  133. package/dist/docx/server/materializeYjsDocx.js +9 -1
  134. package/dist/docx/server/migrateYjsAttrSchema.d.ts +55 -0
  135. package/dist/docx/server/migrateYjsAttrSchema.js +95 -0
  136. package/dist/docx/server/validateDocxConformance.js +22 -1
  137. package/dist/docx/shapeParser.js +12 -10
  138. package/dist/docx/tableParser.d.ts +1 -1
  139. package/dist/docx/tableParser.js +206 -85
  140. package/dist/docx/textBoxParser.d.ts +25 -2
  141. package/dist/docx/textBoxParser.js +59 -4
  142. package/dist/docx/unzip.d.ts +23 -0
  143. package/dist/docx/unzip.js +37 -26
  144. package/dist/docx/verbatimCapture.js +1 -1
  145. package/dist/docx/vmlImageParser.d.ts +17 -1
  146. package/dist/docx/vmlImageParser.js +20 -3
  147. package/dist/docx/vmlPreview.d.ts +1 -3
  148. package/dist/docx/vmlPreview.js +2 -30
  149. package/dist/docx/xmlEncoding.d.ts +5 -0
  150. package/dist/docx/xmlEncoding.js +12 -0
  151. package/dist/docx/xmlParser.d.ts +26 -2
  152. package/dist/docx/xmlParser.js +70 -27
  153. package/dist/docx/xmlResourceLimits.d.ts +89 -9
  154. package/dist/docx/xmlResourceLimits.js +105 -24
  155. package/dist/headless-layout.js +10 -1
  156. package/dist/index.d.ts +2 -1
  157. package/dist/index.js +2 -1
  158. package/dist/internal/pageBreakRunSourceDescendantIndex.js +5 -2
  159. package/dist/internal/paragraphFormattingSerialization.js +3 -2
  160. package/dist/layout-bridge/convert/toFlowBlocks.js +102 -12
  161. package/dist/layout-engine/measure/tableCellFloating.d.ts +2 -0
  162. package/dist/layout-engine/measure/tableCellFloating.js +2 -0
  163. package/dist/layout-engine/types.d.ts +23 -1
  164. package/dist/layout-painter/renderImage.d.ts +5 -1
  165. package/dist/layout-painter/renderImage.js +14 -6
  166. package/dist/layout-painter/renderPage.d.ts +2 -0
  167. package/dist/layout-painter/renderPage.js +4 -0
  168. package/dist/layout-painter/renderParagraph.d.ts +13 -1
  169. package/dist/layout-painter/renderParagraph.js +26 -6
  170. package/dist/managers/autoSaveCodec.d.ts +9 -1
  171. package/dist/managers/autoSaveCodec.js +14 -12
  172. package/dist/markdown/images.js +1 -4
  173. package/dist/markdown/renderBlock.js +1 -0
  174. package/dist/markdown/renderRuns.d.ts +1 -1
  175. package/dist/markdown/renderRuns.js +28 -5
  176. package/dist/markdown/renderTable.js +22 -5
  177. package/dist/markdown/trailers.js +4 -1
  178. package/dist/pdf/images.js +142 -6
  179. package/dist/prosemirror/attrs/index.d.ts +14 -3
  180. package/dist/prosemirror/attrs/index.js +224 -2
  181. package/dist/prosemirror/authoredTransformAttrs.d.ts +28 -0
  182. package/dist/prosemirror/authoredTransformAttrs.js +64 -0
  183. package/dist/prosemirror/commands/contentControls.js +10 -2
  184. package/dist/prosemirror/commands/image.d.ts +9 -2
  185. package/dist/prosemirror/commands/image.js +43 -28
  186. package/dist/prosemirror/commands/pageBreak.js +1 -1
  187. package/dist/prosemirror/commentReferenceAttrs.d.ts +8 -0
  188. package/dist/prosemirror/commentReferenceAttrs.js +33 -0
  189. package/dist/prosemirror/commentReferenceIntegrity.d.ts +47 -0
  190. package/dist/prosemirror/commentReferenceIntegrity.js +43 -0
  191. package/dist/prosemirror/conversion/fromProseDoc.d.ts +2 -13
  192. package/dist/prosemirror/conversion/fromProseDoc.js +338 -99
  193. package/dist/prosemirror/conversion/index.d.ts +2 -2
  194. package/dist/prosemirror/conversion/toProseDoc.d.ts +7 -9
  195. package/dist/prosemirror/conversion/toProseDoc.js +551 -327
  196. package/dist/prosemirror/extensions/StarterKit.js +8 -58
  197. package/dist/prosemirror/extensions/core/DocExtension.js +1 -1
  198. package/dist/prosemirror/extensions/core/ParagraphExtension.js +2 -1
  199. package/dist/prosemirror/extensions/features/BaseKeymapExtension.js +45 -27
  200. package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -1
  201. package/dist/prosemirror/extensions/features/ImagePasteExtension.js +5 -3
  202. package/dist/prosemirror/extensions/features/ParaIdAllocatorExtension.js +1 -0
  203. package/dist/prosemirror/extensions/markRegistry.d.ts +41 -0
  204. package/dist/prosemirror/extensions/markRegistry.js +75 -0
  205. package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +2 -3
  206. package/dist/prosemirror/extensions/marks/InlineWrapperExtension.d.ts +18 -0
  207. package/dist/prosemirror/extensions/marks/InlineWrapperExtension.js +63 -0
  208. package/dist/prosemirror/extensions/marks/markUtils.d.ts +2 -1
  209. package/dist/prosemirror/extensions/marks/markUtils.js +47 -4
  210. package/dist/prosemirror/extensions/nodes/CommentReferenceExtension.d.ts +11 -0
  211. package/dist/prosemirror/extensions/nodes/CommentReferenceExtension.js +87 -0
  212. package/dist/prosemirror/extensions/nodes/FieldExtension.js +9 -7
  213. package/dist/prosemirror/extensions/nodes/ImageExtension.js +20 -0
  214. package/dist/prosemirror/extensions/nodes/PreservedBlockExtension.d.ts +32 -0
  215. package/dist/prosemirror/extensions/nodes/PreservedBlockExtension.js +67 -0
  216. package/dist/prosemirror/extensions/nodes/PreservedXmlExtension.d.ts +16 -0
  217. package/dist/prosemirror/extensions/nodes/PreservedXmlExtension.js +60 -0
  218. package/dist/prosemirror/extensions/nodes/ShapeExtension.js +10 -2
  219. package/dist/prosemirror/extensions/nodes/TableExtension.js +3 -2
  220. package/dist/prosemirror/extensions/nodes/TextBoxExtension.js +13 -5
  221. package/dist/prosemirror/inlineWrapperStack.d.ts +39 -0
  222. package/dist/prosemirror/inlineWrapperStack.js +75 -0
  223. package/dist/prosemirror/pageBreakRunProjection.d.ts +17 -7
  224. package/dist/prosemirror/pageBreakRunProjection.js +19 -9
  225. package/dist/prosemirror/paragraphFormattingProvenance.d.ts +6 -3
  226. package/dist/prosemirror/paragraphFormattingProvenance.js +12 -3
  227. package/dist/prosemirror/replacedAnnotations.d.ts +59 -0
  228. package/dist/prosemirror/replacedAnnotations.js +165 -0
  229. package/dist/prosemirror/runFormattingInlineCarriers.d.ts +3 -1
  230. package/dist/prosemirror/runFormattingInlineCarriers.js +5 -1
  231. package/dist/prosemirror/schema/index.d.ts +3 -3
  232. package/dist/prosemirror/schema/marks.d.ts +25 -1
  233. package/dist/prosemirror/schema/nodes.d.ts +139 -3
  234. package/dist/prosemirror/schema/nodes.js +16 -1
  235. package/dist/prosemirror/trackedRunInlineAtoms.d.ts +2 -0
  236. package/dist/prosemirror/trackedRunInlineAtoms.js +2 -0
  237. package/dist/prosemirror/validation.js +16 -2
  238. package/dist/prosemirror/yjsDocumentMetadata.d.ts +73 -0
  239. package/dist/prosemirror/yjsDocumentMetadata.js +171 -0
  240. package/dist/prosemirror/zeroWidthAnchors.js +2 -0
  241. package/dist/render-dom/commentAnchorAttributes.d.ts +37 -0
  242. package/dist/render-dom/commentAnchorAttributes.js +54 -0
  243. package/dist/server.d.ts +3 -1
  244. package/dist/server.js +3 -1
  245. package/dist/types/content.d.ts +2 -2
  246. package/dist/utils/base64.d.ts +36 -0
  247. package/dist/utils/base64.js +40 -0
  248. package/dist/utils/clipboard.d.ts +9 -1
  249. package/dist/utils/clipboard.js +29 -2
  250. package/dist/utils/findReplace.js +1 -1
  251. package/dist/utils/imageLuminance.d.ts +15 -0
  252. package/dist/utils/imageLuminance.js +32 -0
  253. package/dist/utils/mergeDocumentContent.js +7 -17
  254. package/dist/utils/replaceText.js +1 -0
  255. package/dist/utils/units.d.ts +10 -1
  256. package/dist/utils/units.js +12 -1
  257. package/dist/utils/urlSecurity.d.ts +8 -2
  258. package/dist/utils/urlSecurity.js +21 -3
  259. package/package.json +2 -2
  260. package/dist/docx/blockRangeMarkers.d.ts +0 -36
  261. package/dist/docx/blockRangeMarkers.js +0 -59
  262. package/dist/prosemirror/yjsParagraphSourceContract.d.ts +0 -9
  263. package/dist/prosemirror/yjsParagraphSourceContract.js +0 -26
@@ -1,6 +1,7 @@
1
1
  import { emuToPixels } from "../utils/units.js";
2
2
  import { parseAnchorPosition, parseAnchorWrap, parseFill, parseOutline, resolveColorValueToHex } from "./drawingUtils.js";
3
- import { findByFullName, findChildByLocalName, findChildByNamespaceUri, findChildrenByLocalName, findChildrenByNamespaceUri, getAttribute, getChildElements, parseNumericAttribute, parseOnOffAttribute } from "./xmlParser.js";
3
+ import { parseNonVisualDrawingNames } from "./nonVisualDrawingProps.js";
4
+ import { findByFullName, findChildByLocalName, findChildByNamespaceUri, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getChildElements, getLocalName, parseNumericAttribute, parseOnOffAttribute } from "./xmlParser.js";
4
5
  //#region src/docx/textBoxParser.ts
5
6
  const DRAWINGML_MAIN_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.openxmlformats.org/drawingml/2006/main", "http://purl.oclc.org/ooxml/drawingml/main"]);
6
7
  const WORDPROCESSING_SHAPE_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.microsoft.com/office/word/2010/wordprocessingShape"]);
@@ -148,13 +149,15 @@ function parseTextBox(drawingEl) {
148
149
  };
149
150
  const docPr = findByFullName(container, "wp:docPr");
150
151
  const id = docPr ? getAttribute(docPr, null, "id") ?? void 0 : void 0;
152
+ const names = parseNonVisualDrawingNames(docPr);
151
153
  const fill = parseFill(spPr ?? null);
152
154
  const outline = parseOutline(spPr ?? null);
153
155
  const bodyProps = parseBodyProperties(bodyPr ?? null);
154
156
  const textBox = {
155
157
  type: "textBox",
156
158
  size,
157
- content: []
159
+ content: [],
160
+ ...names
158
161
  };
159
162
  if (id) textBox.id = id;
160
163
  if (fill) textBox.fill = fill;
@@ -193,13 +196,15 @@ function parseTextBoxFromShape(wsp, size, position, wrap) {
193
196
  const bodyPr = findChildByNamespaceUri(wsp, WORDPROCESSING_SHAPE_NAMESPACE_URIS, "bodyPr");
194
197
  const cNvPr = wspChildren.find((el) => el.name === "wps:cNvPr");
195
198
  const id = cNvPr ? getAttribute(cNvPr, null, "id") ?? void 0 : void 0;
199
+ const names = parseNonVisualDrawingNames(cNvPr);
196
200
  const fill = parseFill(spPr ?? null);
197
201
  const outline = parseOutline(spPr ?? null);
198
202
  const bodyProps = parseBodyProperties(bodyPr ?? null);
199
203
  const textBox = {
200
204
  type: "textBox",
201
205
  size,
202
- content: []
206
+ content: [],
207
+ ...names
203
208
  };
204
209
  if (id) textBox.id = id;
205
210
  if (fill) textBox.fill = fill;
@@ -285,6 +290,7 @@ const getTextBoxBlockText = (block) => {
285
290
  return runTexts.join("");
286
291
  }
287
292
  if (block.type === "table") return block.rows.map((row) => row.cells.map((cell) => cell.content.map(getTextBoxBlockText).join("\n")).join(" ")).join("\n");
293
+ if (block.type === "preservedBlock") return "";
288
294
  return block.content.map(getTextBoxBlockText).join("\n");
289
295
  };
290
296
  /**
@@ -308,5 +314,54 @@ function getTextBoxOutlineWidthPx(textBox) {
308
314
  if (!textBox.outline?.width) return 0;
309
315
  return emuToPixels(textBox.outline.width);
310
316
  }
317
+ const scanRunForTextBoxDrawings = ({ xmlRun, claimedByRunParser }) => {
318
+ const textBoxDrawings = [];
319
+ const vmlTextBoxes = [];
320
+ let hasNonTextBoxContent = false;
321
+ const visitDrawing = (drawingEl) => {
322
+ if (isTextBoxDrawing(drawingEl)) {
323
+ textBoxDrawings.push(drawingEl);
324
+ return;
325
+ }
326
+ hasNonTextBoxContent = true;
327
+ };
328
+ for (const el of getChildElements(xmlRun)) {
329
+ const name = getLocalName(el.name ?? "");
330
+ if (name === "rPr") continue;
331
+ if (name === "drawing") {
332
+ visitDrawing(el);
333
+ continue;
334
+ }
335
+ if (name === "pict") {
336
+ if (findDeep(el, "v", "textbox") && !claimedByRunParser(el)) vmlTextBoxes.push(el);
337
+ else hasNonTextBoxContent = true;
338
+ continue;
339
+ }
340
+ if (name === "AlternateContent") {
341
+ const branches = getChildElements(el);
342
+ const choice = branches.find((branch) => getLocalName(branch.name ?? "") === "Choice");
343
+ const fallback = branches.find((branch) => getLocalName(branch.name ?? "") === "Fallback");
344
+ const tryBranch = (branch) => {
345
+ if (!branch) return false;
346
+ let found = false;
347
+ for (const innerEl of getChildElements(branch)) if (getLocalName(innerEl.name ?? "") === "drawing") {
348
+ visitDrawing(innerEl);
349
+ found = true;
350
+ }
351
+ return found;
352
+ };
353
+ let foundInBranch = tryBranch(choice);
354
+ if (!foundInBranch) foundInBranch = tryBranch(fallback);
355
+ if (!foundInBranch) hasNonTextBoxContent = true;
356
+ continue;
357
+ }
358
+ hasNonTextBoxContent = true;
359
+ }
360
+ return {
361
+ textBoxDrawings,
362
+ vmlTextBoxes,
363
+ hasNonTextBoxContent
364
+ };
365
+ };
311
366
  //#endregion
312
- export { extractTextBoxContentElements, getTextBoxContentElement, getTextBoxDimensionsPx, getTextBoxHeightPx, getTextBoxMarginsPx, getTextBoxOutlineWidthPx, getTextBoxText, getTextBoxWidthPx, hasTextBoxContent, hasTextBoxFill, hasTextBoxOutline, isFloatingTextBox, isShapeTextBox, isTextBoxDrawing, parseTextBox, parseTextBoxContent, parseTextBoxFromShape, resolveTextBoxFillColor, resolveTextBoxOutlineColor };
367
+ export { extractTextBoxContentElements, getTextBoxContentElement, getTextBoxDimensionsPx, getTextBoxHeightPx, getTextBoxMarginsPx, getTextBoxOutlineWidthPx, getTextBoxText, getTextBoxWidthPx, hasTextBoxContent, hasTextBoxFill, hasTextBoxOutline, isFloatingTextBox, isShapeTextBox, isTextBoxDrawing, parseTextBox, parseTextBoxContent, parseTextBoxFromShape, resolveTextBoxFillColor, resolveTextBoxOutlineColor, scanRunForTextBoxDrawings };
@@ -6,10 +6,33 @@ declare class DocxSecurityError extends Error {
6
6
  type DocxUnzipLimits = {
7
7
  maxInputBytes: number;
8
8
  maxFiles: number;
9
+ /**
10
+ * Inflated bytes allowed in one XML part.
11
+ *
12
+ * A byte ceiling bounds the markup, not what parsing it allocates: a parsed
13
+ * tree costs 3x to 15x the part's bytes, and the cheapest element to write
14
+ * is the most expensive per byte. Use `maxXmlElementsPerPart` and the
15
+ * package bounds to cap memory; this one caps text-dominated parts, where
16
+ * bytes and cost do track each other.
17
+ */
9
18
  maxXmlBytes: number;
10
19
  maxMediaBytes: number;
11
20
  maxFontBytes: number;
21
+ /**
22
+ * Inflated bytes allowed across every entry in the package.
23
+ *
24
+ * Same caveat as `maxXmlBytes`: this is a ceiling on what passes through
25
+ * memory as bytes, not on what the parsed structure retains.
26
+ */
12
27
  maxTotalUncompressedBytes: number;
28
+ /** Elements allowed in one XML part, counted before any tree is built. */
29
+ maxXmlElementsPerPart: number;
30
+ /** Attributes allowed in one XML part, counted before any tree is built. */
31
+ maxXmlAttributesPerPart: number;
32
+ /** Elements allowed across every XML part in the package. */
33
+ maxXmlElementsPerPackage: number;
34
+ /** Attributes allowed across every XML part in the package. */
35
+ maxXmlAttributesPerPackage: number;
13
36
  allowedMediaMimeTypes: ReadonlySet<string>;
14
37
  };
15
38
  type DocxUnzipOptions = Partial<Omit<DocxUnzipLimits, "allowedMediaMimeTypes">> & {
@@ -1,6 +1,8 @@
1
+ import { bytesToDataUrl } from "../utils/base64.js";
1
2
  import { DOCX_CONTAINER_TYPES, detectDocxContainerType } from "./encryption/containerFormat.js";
2
3
  import { openDocxBuffer } from "./encryption/openEncryptedDocx.js";
3
- import { FOLIO_XML_RESOURCE_LIMITS, assertXmlResourceLimits } from "./xmlResourceLimits.js";
4
+ import { decodeXmlBytes } from "./xmlEncoding.js";
5
+ import { FOLIO_XML_RESOURCE_LIMITS, assertXmlResourceLimits, createXmlPackageBudget } from "./xmlResourceLimits.js";
4
6
  import JSZip from "jszip";
5
7
  //#region src/docx/unzip.ts
6
8
  /**
@@ -58,8 +60,21 @@ const DEFAULT_UNZIP_LIMITS = {
58
60
  maxMediaBytes: 25 * MEBIBYTE,
59
61
  maxFontBytes: 10 * MEBIBYTE,
60
62
  maxTotalUncompressedBytes: 250 * MEBIBYTE,
63
+ maxXmlElementsPerPart: FOLIO_XML_RESOURCE_LIMITS.maxElementsPerPart,
64
+ maxXmlAttributesPerPart: FOLIO_XML_RESOURCE_LIMITS.maxAttributesPerPart,
65
+ maxXmlElementsPerPackage: FOLIO_XML_RESOURCE_LIMITS.maxElementsPerPackage,
66
+ maxXmlAttributesPerPackage: FOLIO_XML_RESOURCE_LIMITS.maxAttributesPerPackage,
61
67
  allowedMediaMimeTypes: DEFAULT_ALLOWED_MEDIA_MIME_TYPES
62
68
  };
69
+ /** The XML bounds an unzip enforces, in the shape the preflight takes. */
70
+ const xmlResourceLimitsFor = (limits) => ({
71
+ maxBytes: limits.maxXmlBytes,
72
+ maxDepth: FOLIO_XML_RESOURCE_LIMITS.maxDepth,
73
+ maxElementsPerPart: limits.maxXmlElementsPerPart,
74
+ maxAttributesPerPart: limits.maxXmlAttributesPerPart,
75
+ maxElementsPerPackage: limits.maxXmlElementsPerPackage,
76
+ maxAttributesPerPackage: limits.maxXmlAttributesPerPackage
77
+ });
63
78
  const PARSED_XML_PARTS = /* @__PURE__ */ new Set([
64
79
  "[content_types].xml",
65
80
  "_rels/.rels",
@@ -147,13 +162,13 @@ async function unzipDocx(buffer, options = {}) {
147
162
  if (lowerPath.endsWith(".xml") || lowerPath.endsWith(".rels")) {
148
163
  assertEntrySize(path, declaredSize, limits.maxXmlBytes);
149
164
  if (options.extractAllXml === false && !shouldExtractXmlPart(lowerPath)) continue;
150
- extractionTasks.push(() => file.async("text").then((xmlContent) => {
151
- assertExtractedSize(path, xmlContent.length, limits.maxXmlBytes);
165
+ extractionTasks.push(() => file.async("uint8array").then((xmlBytes) => {
166
+ assertExtractedSize(path, xmlBytes.byteLength, limits.maxXmlBytes);
152
167
  return {
153
168
  type: "xml",
154
169
  path,
155
170
  lowerPath,
156
- content: xmlContent
171
+ content: decodeXmlBytes(xmlBytes)
157
172
  };
158
173
  }));
159
174
  } else if (lowerPath.startsWith("word/media/")) {
@@ -188,10 +203,11 @@ async function unzipDocx(buffer, options = {}) {
188
203
  }));
189
204
  }
190
205
  }
206
+ const xmlBudget = createXmlPackageBudget();
191
207
  for (const extracted of await Promise.all(extractionTasks.map((extract) => extract()))) {
192
208
  if (!extracted) continue;
193
209
  if (extracted.type === "xml") {
194
- assignXmlContent(content, extracted, limits);
210
+ assignXmlContent(content, extracted, limits, xmlBudget);
195
211
  continue;
196
212
  }
197
213
  if (extracted.type === "media") {
@@ -203,19 +219,21 @@ async function unzipDocx(buffer, options = {}) {
203
219
  return content;
204
220
  }
205
221
  /**
206
- * Parts that every consumer expands into an object tree. Preflighting them once
207
- * here puts the bound on the unzip, so `parseDocx`, the selective save and the
208
- * repack path all share it instead of each entry point carrying its own.
222
+ * Preflight every XML part the unzip retains, not a named few.
223
+ *
224
+ * Naming the parts to bound is the bug: `word/document.xml`, `word/styles.xml`
225
+ * and `word/numbering.xml` were counted, and headers, footers, footnotes,
226
+ * endnotes, comments and every other `word/*.xml` part were parsed into trees
227
+ * unbounded. Bounding the reader instead of the part list means a part added
228
+ * later is bounded by construction, and the package budget makes the ceiling
229
+ * the package's rather than each part's.
209
230
  */
210
- const PREFLIGHT_XML_PARTS = /* @__PURE__ */ new Set([
211
- "word/document.xml",
212
- "word/styles.xml",
213
- "word/numbering.xml"
214
- ]);
215
- function assignXmlContent(content, { path, lowerPath, content: xmlContent }, limits) {
216
- if (PREFLIGHT_XML_PARTS.has(lowerPath)) assertXmlResourceLimits(xmlContent, {
217
- ...FOLIO_XML_RESOURCE_LIMITS,
218
- maxBytes: limits.maxXmlBytes
231
+ function assignXmlContent(content, { path, lowerPath, content: xmlContent }, limits, budget) {
232
+ assertXmlResourceLimits({
233
+ xml: xmlContent,
234
+ limits: xmlResourceLimitsFor(limits),
235
+ partPath: path,
236
+ budget
219
237
  });
220
238
  content.allXml.set(path, xmlContent);
221
239
  if (lowerPath === "word/document.xml") content.documentXml = xmlContent;
@@ -426,14 +444,7 @@ function getMediaMimeType(path) {
426
444
  * @returns Data URL string
427
445
  */
428
446
  function mediaToDataUrl(data, mimeType) {
429
- const bytes = new Uint8Array(data);
430
- const chunks = [];
431
- const chunkSize = 32768;
432
- for (let offset = 0; offset < bytes.length; offset += chunkSize) {
433
- const chunk = bytes.subarray(offset, offset + chunkSize);
434
- chunks.push(String.fromCodePoint(...chunk));
435
- }
436
- return `data:${mimeType};base64,${btoa(chunks.join(""))}`;
447
+ return bytesToDataUrl(new Uint8Array(data), mimeType);
437
448
  }
438
449
  /**
439
450
  * Extract a specific file from the original ZIP
@@ -446,7 +457,7 @@ function extractFile(content, path) {
446
457
  const file = content.originalZip.file(path);
447
458
  if (!file || !isParsedDocxEntry(path)) return Promise.resolve(null);
448
459
  const lowerPath = path.toLowerCase();
449
- if (lowerPath.endsWith(".xml") || lowerPath.endsWith(".rels")) return file.async("text");
460
+ if (lowerPath.endsWith(".xml") || lowerPath.endsWith(".rels")) return file.async("uint8array").then(decodeXmlBytes);
450
461
  return file.async("arraybuffer");
451
462
  }
452
463
  /**
@@ -270,7 +270,7 @@ const parsedReplayEnvelope = (xml, inheritedNamespaceScope) => {
270
270
  };
271
271
  const withinXmlResourceLimits = (xml) => {
272
272
  try {
273
- assertXmlResourceLimits(xml);
273
+ assertXmlResourceLimits({ xml });
274
274
  return true;
275
275
  } catch {
276
276
  return false;
@@ -11,6 +11,22 @@ import { XmlElement } from "./xmlParser.js";
11
11
  * same artwork twice.
12
12
  */
13
13
  declare function shouldPreserveRawVmlPict(pictElement: XmlElement): boolean;
14
+ /**
15
+ * Whether the run parser turns this `w:pict` into run content of its own.
16
+ *
17
+ * Exactly one owner may represent a `w:pict`: the run parser, whose drawing
18
+ * carries the whole element as captured XML, or the text-box enrichment pass,
19
+ * which rebuilds one `v:textbox` as an editable shape. Two owners write the
20
+ * same artwork twice, and when the pict holds text, the saved document says it
21
+ * twice.
22
+ *
23
+ * The enrichment asks this rather than re-deriving the answer from the markup:
24
+ * its own reading (`v:imagedata` present) named only the picture path, so a
25
+ * `v:group` claimed by the preview path was represented by both owners. The
26
+ * result is a pure function of the element, so asking it a second time costs
27
+ * only the work the run parser already did.
28
+ */
29
+ declare const isVmlPictParsedByRunParser: (pictElement: XmlElement, rels: document_d_exports.RelationshipMap | null, media: Map<string, document_d_exports.MediaFile> | null) => boolean;
14
30
  /**
15
31
  * Parse a `w:pict` element into an inline image, or null when it carries no
16
32
  * ordinary VML picture (no resolvable `<v:imagedata>`, or a watermark shape).
@@ -22,4 +38,4 @@ declare function shouldPreserveRawVmlPict(pictElement: XmlElement): boolean;
22
38
  */
23
39
  declare function parseVmlImageContent(pictElement: XmlElement, rels: document_d_exports.RelationshipMap | null, media: Map<string, document_d_exports.MediaFile> | null, rootXmlns?: Record<string, string>): document_d_exports.DrawingContent | null;
24
40
  //#endregion
25
- export { parseVmlImageContent, shouldPreserveRawVmlPict };
41
+ export { isVmlPictParsedByRunParser, parseVmlImageContent, shouldPreserveRawVmlPict };
@@ -1,6 +1,7 @@
1
1
  import { sanitizeImageSrc } from "../utils/sanitizeImageSrc.js";
2
2
  import { pixelsToEmu } from "../utils/units.js";
3
3
  import { resolveImageData } from "./imageParser.js";
4
+ import { PREVIEW_KINDS } from "./previewBudget.js";
4
5
  import { captureVerbatimXml } from "./verbatimCapture.js";
5
6
  import { isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl } from "./vmlPreview.js";
6
7
  import { isWatermarkShape } from "./watermarkParser.js";
@@ -73,8 +74,8 @@ const previewImage = (pictElement, svg, widthPx, heightPx, style, rootXmlns) =>
73
74
  type: "image",
74
75
  rId: "",
75
76
  src,
76
- mimeType: "image/svg+xml",
77
- filename: "vml-shape-preview.svg",
77
+ mimeType: PREVIEW_KINDS.vmlShape.mimeType,
78
+ filename: PREVIEW_KINDS.vmlShape.filename,
78
79
  size: {
79
80
  width: pixelsToEmu(widthPx),
80
81
  height: pixelsToEmu(heightPx)
@@ -130,6 +131,22 @@ function shouldPreserveRawVmlPict(pictElement) {
130
131
  return shapes.length > 0 && !shapes.some((shape) => isWatermarkShape(shape));
131
132
  }
132
133
  /**
134
+ * Whether the run parser turns this `w:pict` into run content of its own.
135
+ *
136
+ * Exactly one owner may represent a `w:pict`: the run parser, whose drawing
137
+ * carries the whole element as captured XML, or the text-box enrichment pass,
138
+ * which rebuilds one `v:textbox` as an editable shape. Two owners write the
139
+ * same artwork twice, and when the pict holds text, the saved document says it
140
+ * twice.
141
+ *
142
+ * The enrichment asks this rather than re-deriving the answer from the markup:
143
+ * its own reading (`v:imagedata` present) named only the picture path, so a
144
+ * `v:group` claimed by the preview path was represented by both owners. The
145
+ * result is a pure function of the element, so asking it a second time costs
146
+ * only the work the run parser already did.
147
+ */
148
+ const isVmlPictParsedByRunParser = (pictElement, rels, media) => parseVmlImageContent(pictElement, rels, media) !== null || shouldPreserveRawVmlPict(pictElement);
149
+ /**
133
150
  * Read the relationship id off a `v:imagedata` element. Word writes `r:id`;
134
151
  * some legacy / third-party generators use `r:embed` or the office-namespace
135
152
  * `o:relid` instead, so fall back through those before the bare `id`.
@@ -200,4 +217,4 @@ function parseVmlImageContent(pictElement, rels, media, rootXmlns = {}) {
200
217
  return null;
201
218
  }
202
219
  //#endregion
203
- export { parseVmlImageContent, shouldPreserveRawVmlPict };
220
+ export { isVmlPictParsedByRunParser, parseVmlImageContent, shouldPreserveRawVmlPict };
@@ -25,7 +25,5 @@ declare const vmlSvgDataUrl: (svg: string) => string | undefined;
25
25
  declare const renderStandaloneVmlPreview: (shape: XmlElement) => VmlPreviewResult;
26
26
  /** Render a VML group through bounded, non-clipping local-coordinate transforms. */
27
27
  declare const renderVmlGroupPreview: (group: XmlElement) => VmlPreviewResult;
28
- /** Bound retained synthetic VML previews while preserving their raw replay nodes. */
29
- declare const enforcePackageVmlPreviewBudget: (root: unknown, maxCharacters?: number) => void;
30
28
  //#endregion
31
- export { VmlPreviewResult, enforcePackageVmlPreviewBudget, isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
29
+ export { VmlPreviewResult, isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
@@ -1,3 +1,4 @@
1
+ import { VML_PREVIEW_DATA_URL_PREFIX } from "./previewBudget.js";
1
2
  import { findChild, findDeep, getAttribute, getChildElements, getLocalName } from "./xmlParser.js";
2
3
  //#region src/docx/vmlPreview.ts
3
4
  const MAX_VML_PREVIEW_DEPTH = 16;
@@ -6,10 +7,6 @@ const MAX_VML_PREVIEW_PATH_POINTS = 2e4;
6
7
  const MAX_VML_PREVIEW_COORDINATE = 1e6;
7
8
  const MAX_VML_PREVIEW_DIMENSION_PX = 2e4;
8
9
  const MAX_VML_SVG_CHARACTERS = 1e6;
9
- const MAX_PACKAGE_VML_PREVIEW_CHARACTERS = 8 * 1024 * 1024;
10
- const VML_PREVIEW_DATA_URL_PREFIX = "data:image/svg+xml;charset=utf-8,";
11
- const VML_PREVIEW_FILENAME = "vml-shape-preview.svg";
12
- const VML_PREVIEW_MIME_TYPE = "image/svg+xml";
13
10
  const SAFE_VML_COLORS = /* @__PURE__ */ new Set([
14
11
  "black",
15
12
  "white",
@@ -506,30 +503,5 @@ const renderVmlGroupPreview = (group) => {
506
503
  style
507
504
  };
508
505
  };
509
- /** Bound retained synthetic VML previews while preserving their raw replay nodes. */
510
- const enforcePackageVmlPreviewBudget = (root, maxCharacters = MAX_PACKAGE_VML_PREVIEW_CHARACTERS) => {
511
- let remainingCharacters = Math.max(0, maxCharacters);
512
- const visited = /* @__PURE__ */ new WeakSet();
513
- const visit = (value) => {
514
- if (value === null || typeof value !== "object" || visited.has(value)) return;
515
- visited.add(value);
516
- if (value instanceof ArrayBuffer || ArrayBuffer.isView(value)) return;
517
- if (value instanceof Map) {
518
- for (const child of value.values()) visit(child);
519
- return;
520
- }
521
- if (Array.isArray(value)) {
522
- for (const child of value) visit(child);
523
- return;
524
- }
525
- if ("type" in value && value.type === "image" && "rId" in value && value.rId === "" && "mimeType" in value && value.mimeType === VML_PREVIEW_MIME_TYPE && "filename" in value && value.filename === VML_PREVIEW_FILENAME && "src" in value && typeof value.src === "string" && value.src.startsWith(VML_PREVIEW_DATA_URL_PREFIX)) if (value.src.length <= remainingCharacters) remainingCharacters -= value.src.length;
526
- else {
527
- remainingCharacters = 0;
528
- delete value.src;
529
- }
530
- for (const child of Object.values(value)) visit(child);
531
- };
532
- visit(root);
533
- };
534
506
  //#endregion
535
- export { enforcePackageVmlPreviewBudget, isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
507
+ export { isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
@@ -0,0 +1,5 @@
1
+ //#region src/docx/xmlEncoding.d.ts
2
+ /** Decode the UTF-8 and UTF-16 encodings XML processors must recognize. */
3
+ declare const decodeXmlBytes: (bytes: Uint8Array) => string;
4
+ //#endregion
5
+ export { decodeXmlBytes };
@@ -0,0 +1,12 @@
1
+ //#region src/docx/xmlEncoding.ts
2
+ /** Decode the UTF-8 and UTF-16 encodings XML processors must recognize. */
3
+ const decodeXmlBytes = (bytes) => {
4
+ if (bytes[0] === 255 && bytes[1] === 254) return new TextDecoder("utf-16le").decode(bytes.subarray(2));
5
+ if (bytes[0] === 254 && bytes[1] === 255) return new TextDecoder("utf-16be").decode(bytes.subarray(2));
6
+ if (bytes[0] === 60 && bytes[1] === 0 && bytes[2] === 63 && bytes[3] === 0) return new TextDecoder("utf-16le").decode(bytes);
7
+ if (bytes[0] === 0 && bytes[1] === 60 && bytes[2] === 0 && bytes[3] === 63) return new TextDecoder("utf-16be").decode(bytes);
8
+ const utf8Offset = bytes[0] === 239 && bytes[1] === 187 && bytes[2] === 191 ? 3 : 0;
9
+ return new TextDecoder("utf-8").decode(bytes.subarray(utf8Offset));
10
+ };
11
+ //#endregion
12
+ export { decodeXmlBytes };
@@ -71,7 +71,22 @@ declare const NAMESPACES: {
71
71
  declare const OOXML_NAMESPACE_SCOPE: XmlNamespaceScope;
72
72
  declare function parseXml(xml: string, inheritedNamespaceScope?: XmlNamespaceScope): XmlElement;
73
73
  /**
74
- * Serialize an XmlElement back to an XML string
74
+ * Serialize an XmlElement back to an XML string.
75
+ *
76
+ * Written here rather than handed to `fast-xml-parser`'s builder for two
77
+ * reasons. The builder writes a tab or a newline inside an attribute value
78
+ * literally, and XML 1.0 §3.3.3 has every conformant reader normalise those to
79
+ * a space: a `descr="two lines&#xA;"` a source file wrote came back as
80
+ * `descr="two lines "`. `escapeXmlAttribute` writes the character references
81
+ * that survive that step. And the builder returned a string per node, so a
82
+ * subtree's bytes were copied into its parent's answer, its grandparent's and
83
+ * so on, which `writeElement` below replaces with one buffer.
84
+ *
85
+ * Those two are the whole difference from the builder: the attribute
86
+ * character references, an attribute the model dropped written as absent
87
+ * rather than as the word "undefined", and the characters XML 1.0 §2.2 admits
88
+ * no spelling for dropped. Every other rule reproduces its output byte for
89
+ * byte, so replaying a capture is unchanged wherever it was already correct.
75
90
  */
76
91
  declare function elementToXml(element: XmlElement): string;
77
92
  /**
@@ -90,6 +105,15 @@ declare function getLocalName(name: string | undefined): string;
90
105
  declare function getNamespacePrefix(name: string): string | null;
91
106
  /** Namespace URI resolved from the element's in-scope XML declarations. */
92
107
  declare const getNamespaceUri: (element: XmlElement) => string | undefined;
108
+ /**
109
+ * Namespace URI of one of the element's attributes, by its source spelling.
110
+ *
111
+ * An unprefixed attribute has no namespace — it is not in the element's,
112
+ * which is why this is not {@link getNamespaceUri} with a different argument —
113
+ * and `undefined` is that answer as well as "the prefix resolves to nothing".
114
+ * A caller that must tell the two apart checks the spelling for a colon.
115
+ */
116
+ declare const resolveAttributeNamespaceUri: (element: XmlElement, attributeName: string) => string | undefined;
93
117
  /** WordprocessingML main namespace, Transitional and Strict (ECMA-376 Parts 1 and 4). */
94
118
  declare const WORDPROCESSINGML_NAMESPACE_URIS: ReadonlySet<string>;
95
119
  /** Office document relationship attributes, Transitional and Strict. */
@@ -368,4 +392,4 @@ declare function cloneWithXmlnsDeclarations(element: XmlElement, xmlnsDecls: Rec
368
392
  */
369
393
  declare function cloneElement(element: XmlElement, overrides: Partial<XmlElement>): XmlElement;
370
394
  //#endregion
371
- export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, XmlAttributeMatch, XmlElement, XmlNamespaceScope, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, selectAlternateContentBranch };
395
+ export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, XmlAttributeMatch, XmlElement, XmlNamespaceScope, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, resolveAttributeNamespaceUri, selectAlternateContentBranch };
@@ -1,6 +1,7 @@
1
1
  import { NUMBERS_PER_PERCENT, percentageSpelling, transitionalSlotEncoding } from "./transitionalSpelling.js";
2
2
  import { universalMeasureAs } from "./universalMeasure.js";
3
- import { XMLBuilder, XMLParser } from "fast-xml-parser";
3
+ import { escapeXmlAttribute, escapeXmlText } from "@stll/docx-core";
4
+ import { XMLParser } from "fast-xml-parser";
4
5
  import { OOXML_NS } from "@stll/docx-utils";
5
6
  import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
6
7
  //#region src/docx/xmlParser.ts
@@ -42,22 +43,12 @@ const fxpParserOptionsWithStopNodes = {
42
43
  ...fxpParserOptions,
43
44
  stopNodes: ["*.w:binData"]
44
45
  };
45
- const fxpBuilderOptions = {
46
- preserveOrder: true,
47
- ignoreAttributes: false,
48
- attributeNamePrefix: "",
49
- textNodeName: "#text",
50
- suppressEmptyNode: true
51
- };
52
46
  const fxpParser = new XMLParser(fxpParserOptions);
53
47
  const fxpParserWithStopNodes = new XMLParser(fxpParserOptionsWithStopNodes);
54
- const fxpBuilder = new XMLBuilder(fxpBuilderOptions);
55
48
  /** Text node key used by fast-xml-parser in preserveOrder mode. */
56
49
  const TEXT_KEY = "#text";
57
50
  /** Attribute group key used by fast-xml-parser in preserveOrder mode. */
58
51
  const ATTR_KEY = ":@";
59
- /** Character reference required to keep carriage returns through XML end-of-line normalization. */
60
- const XML_CARRIAGE_RETURN_REFERENCE = "&#13;";
61
52
  const EMPTY_NAMESPACE_SCOPE = { bindings: /* @__PURE__ */ new Map() };
62
53
  const resolveNamespaceUri = (scope, prefix) => {
63
54
  let current = scope;
@@ -138,18 +129,6 @@ function fxpToRootElement(nodes, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE
138
129
  return { elements: nodes };
139
130
  }
140
131
  /**
141
- * Convert an XmlElement back into the fast-xml-parser preserveOrder format
142
- * so we can feed it to XMLBuilder.
143
- */
144
- function elementToFxpNode(el) {
145
- if (el.type === "text") return { [TEXT_KEY]: el.text ?? "" };
146
- const name = el.name ?? "";
147
- const children = el.elements ? el.elements.map(elementToFxpNode) : [];
148
- const node = { [name]: children };
149
- if (el.attributes && Object.keys(el.attributes).length > 0) node[ATTR_KEY] = el.attributes;
150
- return node;
151
- }
152
- /**
153
132
  * Common OOXML namespace URIs — re-exported from @stll/docx-utils.
154
133
  */
155
134
  const NAMESPACES = OOXML_NS;
@@ -174,13 +153,64 @@ function parseXml(xml, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE) {
174
153
  return fxpToRootElement((xml.includes("binData") ? fxpParserWithStopNodes : fxpParser).parse(xml), inheritedNamespaceScope);
175
154
  }
176
155
  /**
177
- * Serialize an XmlElement back to an XML string
156
+ * Serialize an XmlElement back to an XML string.
157
+ *
158
+ * Written here rather than handed to `fast-xml-parser`'s builder for two
159
+ * reasons. The builder writes a tab or a newline inside an attribute value
160
+ * literally, and XML 1.0 §3.3.3 has every conformant reader normalise those to
161
+ * a space: a `descr="two lines&#xA;"` a source file wrote came back as
162
+ * `descr="two lines "`. `escapeXmlAttribute` writes the character references
163
+ * that survive that step. And the builder returned a string per node, so a
164
+ * subtree's bytes were copied into its parent's answer, its grandparent's and
165
+ * so on, which `writeElement` below replaces with one buffer.
166
+ *
167
+ * Those two are the whole difference from the builder: the attribute
168
+ * character references, an attribute the model dropped written as absent
169
+ * rather than as the word "undefined", and the characters XML 1.0 §2.2 admits
170
+ * no spelling for dropped. Every other rule reproduces its output byte for
171
+ * byte, so replaying a capture is unchanged wherever it was already correct.
178
172
  */
179
173
  function elementToXml(element) {
180
- const fxpNode = elementToFxpNode(element);
181
- return fxpBuilder.build([fxpNode]).replaceAll("\r", XML_CARRIAGE_RETURN_REFERENCE);
174
+ const out = [];
175
+ writeElement(element, out);
176
+ return out.join("");
182
177
  }
183
178
  /**
179
+ * Serialize into one shared buffer rather than a string per node.
180
+ *
181
+ * Returning a string per element makes a subtree's bytes a substring of its
182
+ * parent's, its grandparent's and so on, so a table pays for its rows, its
183
+ * rows pay for their cells, and a part that nests four levels deep is copied
184
+ * four times before anything is written. Appending into one array of chunks
185
+ * and joining once costs each node its own text and nothing for its ancestors.
186
+ *
187
+ * Whether an element self-closes is not known until its children are written,
188
+ * so the opening tag reserves a slot in the buffer and fills it afterwards:
189
+ * `>` when something was appended, `/>` when nothing was.
190
+ */
191
+ const writeElement = (element, out) => {
192
+ if (element.type === "text") {
193
+ const text = String(element.text ?? "");
194
+ if (text !== "") out.push(escapeXmlText(text));
195
+ return;
196
+ }
197
+ const name = element.name ?? "";
198
+ out.push(`<${name}`);
199
+ if (element.attributes) for (const [attribute, value] of Object.entries(element.attributes)) {
200
+ if (value === void 0) continue;
201
+ out.push(` ${attribute}="${escapeXmlAttribute(String(value))}"`);
202
+ }
203
+ const openingSlot = out.push("") - 1;
204
+ const contentStart = out.length;
205
+ for (const child of element.elements ?? []) writeElement(child, out);
206
+ if (out.length === contentStart) {
207
+ out[openingSlot] = "/>";
208
+ return;
209
+ }
210
+ out[openingSlot] = ">";
211
+ out.push(`</${name}>`);
212
+ };
213
+ /**
184
214
  * Parse XML string to a more convenient format
185
215
  */
186
216
  function parseXmlDocument(xml) {
@@ -216,6 +246,19 @@ function getNamespacePrefix(name) {
216
246
  }
217
247
  /** Namespace URI resolved from the element's in-scope XML declarations. */
218
248
  const getNamespaceUri = (element) => element.namespaceUri;
249
+ /**
250
+ * Namespace URI of one of the element's attributes, by its source spelling.
251
+ *
252
+ * An unprefixed attribute has no namespace — it is not in the element's,
253
+ * which is why this is not {@link getNamespaceUri} with a different argument —
254
+ * and `undefined` is that answer as well as "the prefix resolves to nothing".
255
+ * A caller that must tell the two apart checks the spelling for a colon.
256
+ */
257
+ const resolveAttributeNamespaceUri = (element, attributeName) => {
258
+ const colonIndex = attributeName.indexOf(":");
259
+ if (colonIndex === -1) return;
260
+ return resolveNamespaceUri(element.namespaceScope, attributeName.slice(0, colonIndex));
261
+ };
219
262
  /** WordprocessingML main namespace, Transitional and Strict (ECMA-376 Parts 1 and 4). */
220
263
  const WORDPROCESSINGML_NAMESPACE_URIS = /* @__PURE__ */ new Set([NAMESPACES.w, "http://purl.oclc.org/ooxml/wordprocessingml/main"]);
221
264
  /** Office document relationship attributes, Transitional and Strict. */
@@ -825,4 +868,4 @@ function cloneElement(element, overrides) {
825
868
  return clone;
826
869
  }
827
870
  //#endregion
828
- export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, selectAlternateContentBranch };
871
+ export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, resolveAttributeNamespaceUri, selectAlternateContentBranch };