@stll/folio-core 0.44.0 → 0.46.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ai-edits/__fixtures__/paragraphs.js +2 -2
- package/dist/ai-edits/apply.js +59 -19
- package/dist/ai-edits/headless.js +2 -1
- package/dist/compare/compare.js +2 -5
- package/dist/compare/content-alignment.js +78 -53
- package/dist/compare/inline-atoms.js +34 -20
- package/dist/compare/scenario.js +2 -5
- package/dist/compare/types.d.ts +9 -1
- package/dist/compare/types.js +12 -1
- package/dist/compat/eigenpal.d.ts +2 -1
- package/dist/compat/eigenpal.js +2 -1
- package/dist/content-controls/checkboxDisplay.d.ts +11 -0
- package/dist/content-controls/checkboxDisplay.js +40 -0
- package/dist/content-controls/findContentControls.js +3 -1
- package/dist/content-controls/mutateContentControls.js +21 -4
- package/dist/controller/contentControlWidgetController.d.ts +1 -0
- package/dist/controller/contentControlWidgetController.js +3 -0
- package/dist/controller/hiddenEditorApi.js +3 -1
- package/dist/controller/hiddenEditorManager.d.ts +1 -1
- package/dist/controller/hiddenEditorManager.js +6 -3
- package/dist/display-list/build/buildDisplayList.js +20 -1
- package/dist/display-list/build/imagePrimitives.d.ts +1 -1
- package/dist/display-list/build/imagePrimitives.js +3 -1
- package/dist/display-list/build/paragraphPrimitives.js +21 -2
- package/dist/display-list/dom/renderDisplayListToDom.js +18 -10
- package/dist/display-list/types.d.ts +17 -0
- package/dist/document-operations.js +14 -3
- package/dist/docx/attributeRemainder.d.ts +43 -0
- package/dist/docx/attributeRemainder.js +65 -0
- package/dist/docx/blockContentParser.d.ts +1 -1
- package/dist/docx/blockContentParser.js +87 -63
- package/dist/docx/blockPlainText.d.ts +3 -4
- package/dist/docx/blockPlainText.js +1 -0
- package/dist/docx/bookmarkPlacement.js +2 -0
- package/dist/docx/borderParser.js +5 -5
- package/dist/docx/commentAnchorIndex.d.ts +30 -0
- package/dist/docx/commentAnchorIndex.js +50 -0
- package/dist/docx/commentParser.d.ts +1 -1
- package/dist/docx/commentParser.js +78 -50
- package/dist/docx/commentRangeIntegrity.d.ts +16 -1
- package/dist/docx/commentRangeIntegrity.js +42 -1
- package/dist/docx/commentRangeJoin.d.ts +9 -0
- package/dist/docx/commentRangeJoin.js +33 -0
- package/dist/docx/commentReferenceCompletion.d.ts +9 -0
- package/dist/docx/commentReferenceCompletion.js +47 -0
- package/dist/docx/commentReferenceNormalization.js +1 -0
- package/dist/docx/commentReplyMarkers.js +11 -19
- package/dist/docx/commentThreadKey.d.ts +18 -0
- package/dist/docx/commentThreadKey.js +22 -0
- package/dist/docx/compatibility.js +1 -0
- package/dist/docx/containerChildren.d.ts +111 -0
- package/dist/docx/containerChildren.gen.d.ts +28 -0
- package/dist/docx/containerChildren.gen.js +246 -0
- package/dist/docx/containerChildren.js +103 -0
- package/dist/docx/diagramPreview.js +87 -27
- package/dist/docx/documentParser.d.ts +1 -1
- package/dist/docx/documentParser.js +3 -3
- package/dist/docx/ensureParaIds.js +17 -7
- package/dist/docx/fieldParser.d.ts +1 -1
- package/dist/docx/fieldParser.js +4 -4
- package/dist/docx/fieldState.d.ts +26 -0
- package/dist/docx/fieldState.js +53 -0
- package/dist/docx/footnoteParser.d.ts +1 -1
- package/dist/docx/graphicFrameLocks.d.ts +16 -4
- package/dist/docx/graphicFrameLocks.js +19 -6
- package/dist/docx/groupDrawingParser.js +3 -3
- package/dist/docx/headerFooterParser.js +1 -1
- package/dist/docx/headerFooterReferenceNormalization.js +1 -0
- package/dist/docx/hyperlinkParser.d.ts +27 -8
- package/dist/docx/hyperlinkParser.js +78 -20
- package/dist/docx/imageParser.d.ts +9 -1
- package/dist/docx/imageParser.js +91 -20
- package/dist/docx/imageRawXml.d.ts +14 -1
- package/dist/docx/imageRawXml.js +30 -6
- package/dist/docx/inlineWrapperContent.d.ts +85 -0
- package/dist/docx/inlineWrapperContent.js +84 -0
- package/dist/docx/mathToMathml.js +12 -14
- package/dist/docx/nonVisualDrawingProps.d.ts +34 -0
- package/dist/docx/nonVisualDrawingProps.js +46 -0
- package/dist/docx/normalizeBaseDirection.js +10 -1
- package/dist/docx/paraIdAttribute.d.ts +21 -0
- package/dist/docx/paraIdAttribute.js +64 -0
- package/dist/docx/paragraphParser.d.ts +1 -1
- package/dist/docx/paragraphParser.js +277 -168
- package/dist/docx/paragraphPropertySource.js +5 -1
- package/dist/docx/paragraphTextBoxEnrichment.d.ts +1 -1
- package/dist/docx/paragraphTextBoxEnrichment.js +9 -52
- package/dist/docx/paragraphTraversal.js +7 -4
- package/dist/docx/parseWarningMessage.js +2 -1
- package/dist/docx/parser.js +2 -2
- package/dist/docx/preservedRunContent.d.ts +32 -0
- package/dist/docx/preservedRunContent.js +86 -0
- package/dist/docx/previewBudget.d.ts +64 -0
- package/dist/docx/previewBudget.js +88 -0
- package/dist/docx/renderedPageBreakNormalization.js +2 -4
- package/dist/docx/revisionIdNormalization.js +17 -5
- package/dist/docx/rezip.js +91 -36
- package/dist/docx/runParser.d.ts +1 -1
- package/dist/docx/runParser.js +16 -12
- package/dist/docx/sdtPropertiesPatch.js +24 -18
- package/dist/docx/sectionParser.js +8 -1
- package/dist/docx/sectionReferenceHistory.js +2 -2
- package/dist/docx/selectiveSave.js +6 -6
- package/dist/docx/selectiveXmlPatch.d.ts +46 -2
- package/dist/docx/selectiveXmlPatch.js +86 -39
- package/dist/docx/serializer/blockSdtSerializer.js +38 -26
- package/dist/docx/serializer/borderSerializer.d.ts +1 -1
- package/dist/docx/serializer/borderSerializer.js +15 -14
- package/dist/docx/serializer/commentSerializer.d.ts +41 -16
- package/dist/docx/serializer/commentSerializer.js +87 -92
- package/dist/docx/serializer/documentSerializer.js +8 -11
- package/dist/docx/serializer/fontTableSerializer.js +6 -6
- package/dist/docx/serializer/headerFooterSerializer.js +12 -14
- package/dist/docx/serializer/markupRangeAttributes.js +2 -2
- package/dist/docx/serializer/noteSerializer.js +8 -8
- package/dist/docx/serializer/numberingSerializer.js +7 -6
- package/dist/docx/serializer/paragraphSerializer.js +93 -72
- package/dist/docx/serializer/partNamespaces.js +2 -2
- package/dist/docx/serializer/runSerializer.js +73 -44
- package/dist/docx/serializer/sectionPropertiesSerializer.js +19 -14
- package/dist/docx/serializer/settingsSerializer.js +4 -3
- package/dist/docx/serializer/stylesSerializer.js +6 -6
- package/dist/docx/serializer/tableSerializer.js +28 -17
- package/dist/docx/serializer/textFormattingSerializer.js +29 -28
- package/dist/docx/serializer/themeSerializer.js +6 -6
- package/dist/docx/serializer/trackedChangeAttributes.js +2 -2
- package/dist/docx/serializer/xmlUtils.d.ts +1 -2
- package/dist/docx/serializer/xmlUtils.js +1 -13
- package/dist/docx/server/boundedArchive.d.ts +12 -0
- package/dist/docx/server/boundedArchive.js +20 -1
- package/dist/docx/server/createBilingualDocument.js +3 -1
- package/dist/docx/server/materializeYjsDocx.d.ts +1 -1
- package/dist/docx/server/materializeYjsDocx.js +9 -1
- package/dist/docx/server/migrateYjsAttrSchema.d.ts +55 -0
- package/dist/docx/server/migrateYjsAttrSchema.js +95 -0
- package/dist/docx/server/validateDocxConformance.js +22 -1
- package/dist/docx/shapeParser.js +12 -10
- package/dist/docx/tableParser.d.ts +1 -1
- package/dist/docx/tableParser.js +206 -85
- package/dist/docx/textBoxParser.d.ts +25 -2
- package/dist/docx/textBoxParser.js +59 -4
- package/dist/docx/unzip.d.ts +23 -0
- package/dist/docx/unzip.js +37 -26
- package/dist/docx/verbatimCapture.js +1 -1
- package/dist/docx/vmlImageParser.d.ts +17 -1
- package/dist/docx/vmlImageParser.js +20 -3
- package/dist/docx/vmlPreview.d.ts +1 -3
- package/dist/docx/vmlPreview.js +2 -30
- package/dist/docx/xmlEncoding.d.ts +5 -0
- package/dist/docx/xmlEncoding.js +12 -0
- package/dist/docx/xmlParser.d.ts +26 -2
- package/dist/docx/xmlParser.js +70 -27
- package/dist/docx/xmlResourceLimits.d.ts +89 -9
- package/dist/docx/xmlResourceLimits.js +105 -24
- package/dist/headless-layout.js +10 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/internal/pageBreakRunSourceDescendantIndex.js +5 -2
- package/dist/internal/paragraphFormattingSerialization.js +3 -2
- package/dist/layout-bridge/convert/toFlowBlocks.js +102 -12
- package/dist/layout-engine/measure/tableCellFloating.d.ts +2 -0
- package/dist/layout-engine/measure/tableCellFloating.js +2 -0
- package/dist/layout-engine/types.d.ts +23 -1
- package/dist/layout-painter/renderImage.d.ts +5 -1
- package/dist/layout-painter/renderImage.js +14 -6
- package/dist/layout-painter/renderPage.d.ts +2 -0
- package/dist/layout-painter/renderPage.js +4 -0
- package/dist/layout-painter/renderParagraph.d.ts +13 -1
- package/dist/layout-painter/renderParagraph.js +26 -6
- package/dist/managers/autoSaveCodec.d.ts +9 -1
- package/dist/managers/autoSaveCodec.js +14 -12
- package/dist/markdown/images.js +1 -4
- package/dist/markdown/renderBlock.js +1 -0
- package/dist/markdown/renderRuns.d.ts +1 -1
- package/dist/markdown/renderRuns.js +28 -5
- package/dist/markdown/renderTable.js +22 -5
- package/dist/markdown/trailers.js +4 -1
- package/dist/pdf/images.js +142 -6
- package/dist/prosemirror/attrs/index.d.ts +14 -3
- package/dist/prosemirror/attrs/index.js +224 -2
- package/dist/prosemirror/authoredTransformAttrs.d.ts +28 -0
- package/dist/prosemirror/authoredTransformAttrs.js +64 -0
- package/dist/prosemirror/commands/contentControls.js +10 -2
- package/dist/prosemirror/commands/image.d.ts +9 -2
- package/dist/prosemirror/commands/image.js +43 -28
- package/dist/prosemirror/commands/pageBreak.js +1 -1
- package/dist/prosemirror/commentReferenceAttrs.d.ts +8 -0
- package/dist/prosemirror/commentReferenceAttrs.js +33 -0
- package/dist/prosemirror/commentReferenceIntegrity.d.ts +47 -0
- package/dist/prosemirror/commentReferenceIntegrity.js +43 -0
- package/dist/prosemirror/conversion/fromProseDoc.d.ts +2 -13
- package/dist/prosemirror/conversion/fromProseDoc.js +338 -99
- package/dist/prosemirror/conversion/index.d.ts +2 -2
- package/dist/prosemirror/conversion/toProseDoc.d.ts +7 -9
- package/dist/prosemirror/conversion/toProseDoc.js +551 -327
- package/dist/prosemirror/extensions/StarterKit.js +8 -58
- package/dist/prosemirror/extensions/core/DocExtension.js +1 -1
- package/dist/prosemirror/extensions/core/ParagraphExtension.js +2 -1
- package/dist/prosemirror/extensions/features/BaseKeymapExtension.js +45 -27
- package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -1
- package/dist/prosemirror/extensions/features/ImagePasteExtension.js +5 -3
- package/dist/prosemirror/extensions/features/ParaIdAllocatorExtension.js +1 -0
- package/dist/prosemirror/extensions/markRegistry.d.ts +41 -0
- package/dist/prosemirror/extensions/markRegistry.js +75 -0
- package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +2 -3
- package/dist/prosemirror/extensions/marks/InlineWrapperExtension.d.ts +18 -0
- package/dist/prosemirror/extensions/marks/InlineWrapperExtension.js +63 -0
- package/dist/prosemirror/extensions/marks/markUtils.d.ts +2 -1
- package/dist/prosemirror/extensions/marks/markUtils.js +47 -4
- package/dist/prosemirror/extensions/nodes/CommentReferenceExtension.d.ts +11 -0
- package/dist/prosemirror/extensions/nodes/CommentReferenceExtension.js +87 -0
- package/dist/prosemirror/extensions/nodes/FieldExtension.js +9 -7
- package/dist/prosemirror/extensions/nodes/ImageExtension.js +20 -0
- package/dist/prosemirror/extensions/nodes/PreservedBlockExtension.d.ts +32 -0
- package/dist/prosemirror/extensions/nodes/PreservedBlockExtension.js +67 -0
- package/dist/prosemirror/extensions/nodes/PreservedXmlExtension.d.ts +16 -0
- package/dist/prosemirror/extensions/nodes/PreservedXmlExtension.js +60 -0
- package/dist/prosemirror/extensions/nodes/ShapeExtension.js +10 -2
- package/dist/prosemirror/extensions/nodes/TableExtension.js +3 -2
- package/dist/prosemirror/extensions/nodes/TextBoxExtension.js +13 -5
- package/dist/prosemirror/inlineWrapperStack.d.ts +39 -0
- package/dist/prosemirror/inlineWrapperStack.js +75 -0
- package/dist/prosemirror/pageBreakRunProjection.d.ts +17 -7
- package/dist/prosemirror/pageBreakRunProjection.js +19 -9
- package/dist/prosemirror/paragraphFormattingProvenance.d.ts +6 -3
- package/dist/prosemirror/paragraphFormattingProvenance.js +12 -3
- package/dist/prosemirror/replacedAnnotations.d.ts +59 -0
- package/dist/prosemirror/replacedAnnotations.js +165 -0
- package/dist/prosemirror/runFormattingInlineCarriers.d.ts +3 -1
- package/dist/prosemirror/runFormattingInlineCarriers.js +5 -1
- package/dist/prosemirror/schema/index.d.ts +3 -3
- package/dist/prosemirror/schema/marks.d.ts +25 -1
- package/dist/prosemirror/schema/nodes.d.ts +139 -3
- package/dist/prosemirror/schema/nodes.js +16 -1
- package/dist/prosemirror/trackedRunInlineAtoms.d.ts +2 -0
- package/dist/prosemirror/trackedRunInlineAtoms.js +2 -0
- package/dist/prosemirror/validation.js +16 -2
- package/dist/prosemirror/yjsDocumentMetadata.d.ts +73 -0
- package/dist/prosemirror/yjsDocumentMetadata.js +171 -0
- package/dist/prosemirror/zeroWidthAnchors.js +2 -0
- package/dist/render-dom/commentAnchorAttributes.d.ts +37 -0
- package/dist/render-dom/commentAnchorAttributes.js +54 -0
- package/dist/server.d.ts +3 -1
- package/dist/server.js +3 -1
- package/dist/types/content.d.ts +2 -2
- package/dist/utils/base64.d.ts +36 -0
- package/dist/utils/base64.js +40 -0
- package/dist/utils/clipboard.d.ts +9 -1
- package/dist/utils/clipboard.js +29 -2
- package/dist/utils/findReplace.js +1 -1
- package/dist/utils/imageLuminance.d.ts +15 -0
- package/dist/utils/imageLuminance.js +32 -0
- package/dist/utils/mergeDocumentContent.js +7 -17
- package/dist/utils/replaceText.js +1 -0
- package/dist/utils/units.d.ts +10 -1
- package/dist/utils/units.js +12 -1
- package/dist/utils/urlSecurity.d.ts +8 -2
- package/dist/utils/urlSecurity.js +21 -3
- package/package.json +2 -2
- package/dist/docx/blockRangeMarkers.d.ts +0 -36
- package/dist/docx/blockRangeMarkers.js +0 -59
- package/dist/prosemirror/yjsParagraphSourceContract.d.ts +0 -9
- package/dist/prosemirror/yjsParagraphSourceContract.js +0 -26
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { emuToPixels } from "../utils/units.js";
|
|
2
2
|
import { parseAnchorPosition, parseAnchorWrap, parseFill, parseOutline, resolveColorValueToHex } from "./drawingUtils.js";
|
|
3
|
-
import {
|
|
3
|
+
import { parseNonVisualDrawingNames } from "./nonVisualDrawingProps.js";
|
|
4
|
+
import { findByFullName, findChildByLocalName, findChildByNamespaceUri, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getChildElements, getLocalName, parseNumericAttribute, parseOnOffAttribute } from "./xmlParser.js";
|
|
4
5
|
//#region src/docx/textBoxParser.ts
|
|
5
6
|
const DRAWINGML_MAIN_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.openxmlformats.org/drawingml/2006/main", "http://purl.oclc.org/ooxml/drawingml/main"]);
|
|
6
7
|
const WORDPROCESSING_SHAPE_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.microsoft.com/office/word/2010/wordprocessingShape"]);
|
|
@@ -148,13 +149,15 @@ function parseTextBox(drawingEl) {
|
|
|
148
149
|
};
|
|
149
150
|
const docPr = findByFullName(container, "wp:docPr");
|
|
150
151
|
const id = docPr ? getAttribute(docPr, null, "id") ?? void 0 : void 0;
|
|
152
|
+
const names = parseNonVisualDrawingNames(docPr);
|
|
151
153
|
const fill = parseFill(spPr ?? null);
|
|
152
154
|
const outline = parseOutline(spPr ?? null);
|
|
153
155
|
const bodyProps = parseBodyProperties(bodyPr ?? null);
|
|
154
156
|
const textBox = {
|
|
155
157
|
type: "textBox",
|
|
156
158
|
size,
|
|
157
|
-
content: []
|
|
159
|
+
content: [],
|
|
160
|
+
...names
|
|
158
161
|
};
|
|
159
162
|
if (id) textBox.id = id;
|
|
160
163
|
if (fill) textBox.fill = fill;
|
|
@@ -193,13 +196,15 @@ function parseTextBoxFromShape(wsp, size, position, wrap) {
|
|
|
193
196
|
const bodyPr = findChildByNamespaceUri(wsp, WORDPROCESSING_SHAPE_NAMESPACE_URIS, "bodyPr");
|
|
194
197
|
const cNvPr = wspChildren.find((el) => el.name === "wps:cNvPr");
|
|
195
198
|
const id = cNvPr ? getAttribute(cNvPr, null, "id") ?? void 0 : void 0;
|
|
199
|
+
const names = parseNonVisualDrawingNames(cNvPr);
|
|
196
200
|
const fill = parseFill(spPr ?? null);
|
|
197
201
|
const outline = parseOutline(spPr ?? null);
|
|
198
202
|
const bodyProps = parseBodyProperties(bodyPr ?? null);
|
|
199
203
|
const textBox = {
|
|
200
204
|
type: "textBox",
|
|
201
205
|
size,
|
|
202
|
-
content: []
|
|
206
|
+
content: [],
|
|
207
|
+
...names
|
|
203
208
|
};
|
|
204
209
|
if (id) textBox.id = id;
|
|
205
210
|
if (fill) textBox.fill = fill;
|
|
@@ -285,6 +290,7 @@ const getTextBoxBlockText = (block) => {
|
|
|
285
290
|
return runTexts.join("");
|
|
286
291
|
}
|
|
287
292
|
if (block.type === "table") return block.rows.map((row) => row.cells.map((cell) => cell.content.map(getTextBoxBlockText).join("\n")).join(" ")).join("\n");
|
|
293
|
+
if (block.type === "preservedBlock") return "";
|
|
288
294
|
return block.content.map(getTextBoxBlockText).join("\n");
|
|
289
295
|
};
|
|
290
296
|
/**
|
|
@@ -308,5 +314,54 @@ function getTextBoxOutlineWidthPx(textBox) {
|
|
|
308
314
|
if (!textBox.outline?.width) return 0;
|
|
309
315
|
return emuToPixels(textBox.outline.width);
|
|
310
316
|
}
|
|
317
|
+
const scanRunForTextBoxDrawings = ({ xmlRun, claimedByRunParser }) => {
|
|
318
|
+
const textBoxDrawings = [];
|
|
319
|
+
const vmlTextBoxes = [];
|
|
320
|
+
let hasNonTextBoxContent = false;
|
|
321
|
+
const visitDrawing = (drawingEl) => {
|
|
322
|
+
if (isTextBoxDrawing(drawingEl)) {
|
|
323
|
+
textBoxDrawings.push(drawingEl);
|
|
324
|
+
return;
|
|
325
|
+
}
|
|
326
|
+
hasNonTextBoxContent = true;
|
|
327
|
+
};
|
|
328
|
+
for (const el of getChildElements(xmlRun)) {
|
|
329
|
+
const name = getLocalName(el.name ?? "");
|
|
330
|
+
if (name === "rPr") continue;
|
|
331
|
+
if (name === "drawing") {
|
|
332
|
+
visitDrawing(el);
|
|
333
|
+
continue;
|
|
334
|
+
}
|
|
335
|
+
if (name === "pict") {
|
|
336
|
+
if (findDeep(el, "v", "textbox") && !claimedByRunParser(el)) vmlTextBoxes.push(el);
|
|
337
|
+
else hasNonTextBoxContent = true;
|
|
338
|
+
continue;
|
|
339
|
+
}
|
|
340
|
+
if (name === "AlternateContent") {
|
|
341
|
+
const branches = getChildElements(el);
|
|
342
|
+
const choice = branches.find((branch) => getLocalName(branch.name ?? "") === "Choice");
|
|
343
|
+
const fallback = branches.find((branch) => getLocalName(branch.name ?? "") === "Fallback");
|
|
344
|
+
const tryBranch = (branch) => {
|
|
345
|
+
if (!branch) return false;
|
|
346
|
+
let found = false;
|
|
347
|
+
for (const innerEl of getChildElements(branch)) if (getLocalName(innerEl.name ?? "") === "drawing") {
|
|
348
|
+
visitDrawing(innerEl);
|
|
349
|
+
found = true;
|
|
350
|
+
}
|
|
351
|
+
return found;
|
|
352
|
+
};
|
|
353
|
+
let foundInBranch = tryBranch(choice);
|
|
354
|
+
if (!foundInBranch) foundInBranch = tryBranch(fallback);
|
|
355
|
+
if (!foundInBranch) hasNonTextBoxContent = true;
|
|
356
|
+
continue;
|
|
357
|
+
}
|
|
358
|
+
hasNonTextBoxContent = true;
|
|
359
|
+
}
|
|
360
|
+
return {
|
|
361
|
+
textBoxDrawings,
|
|
362
|
+
vmlTextBoxes,
|
|
363
|
+
hasNonTextBoxContent
|
|
364
|
+
};
|
|
365
|
+
};
|
|
311
366
|
//#endregion
|
|
312
|
-
export { extractTextBoxContentElements, getTextBoxContentElement, getTextBoxDimensionsPx, getTextBoxHeightPx, getTextBoxMarginsPx, getTextBoxOutlineWidthPx, getTextBoxText, getTextBoxWidthPx, hasTextBoxContent, hasTextBoxFill, hasTextBoxOutline, isFloatingTextBox, isShapeTextBox, isTextBoxDrawing, parseTextBox, parseTextBoxContent, parseTextBoxFromShape, resolveTextBoxFillColor, resolveTextBoxOutlineColor };
|
|
367
|
+
export { extractTextBoxContentElements, getTextBoxContentElement, getTextBoxDimensionsPx, getTextBoxHeightPx, getTextBoxMarginsPx, getTextBoxOutlineWidthPx, getTextBoxText, getTextBoxWidthPx, hasTextBoxContent, hasTextBoxFill, hasTextBoxOutline, isFloatingTextBox, isShapeTextBox, isTextBoxDrawing, parseTextBox, parseTextBoxContent, parseTextBoxFromShape, resolveTextBoxFillColor, resolveTextBoxOutlineColor, scanRunForTextBoxDrawings };
|
package/dist/docx/unzip.d.ts
CHANGED
|
@@ -6,10 +6,33 @@ declare class DocxSecurityError extends Error {
|
|
|
6
6
|
type DocxUnzipLimits = {
|
|
7
7
|
maxInputBytes: number;
|
|
8
8
|
maxFiles: number;
|
|
9
|
+
/**
|
|
10
|
+
* Inflated bytes allowed in one XML part.
|
|
11
|
+
*
|
|
12
|
+
* A byte ceiling bounds the markup, not what parsing it allocates: a parsed
|
|
13
|
+
* tree costs 3x to 15x the part's bytes, and the cheapest element to write
|
|
14
|
+
* is the most expensive per byte. Use `maxXmlElementsPerPart` and the
|
|
15
|
+
* package bounds to cap memory; this one caps text-dominated parts, where
|
|
16
|
+
* bytes and cost do track each other.
|
|
17
|
+
*/
|
|
9
18
|
maxXmlBytes: number;
|
|
10
19
|
maxMediaBytes: number;
|
|
11
20
|
maxFontBytes: number;
|
|
21
|
+
/**
|
|
22
|
+
* Inflated bytes allowed across every entry in the package.
|
|
23
|
+
*
|
|
24
|
+
* Same caveat as `maxXmlBytes`: this is a ceiling on what passes through
|
|
25
|
+
* memory as bytes, not on what the parsed structure retains.
|
|
26
|
+
*/
|
|
12
27
|
maxTotalUncompressedBytes: number;
|
|
28
|
+
/** Elements allowed in one XML part, counted before any tree is built. */
|
|
29
|
+
maxXmlElementsPerPart: number;
|
|
30
|
+
/** Attributes allowed in one XML part, counted before any tree is built. */
|
|
31
|
+
maxXmlAttributesPerPart: number;
|
|
32
|
+
/** Elements allowed across every XML part in the package. */
|
|
33
|
+
maxXmlElementsPerPackage: number;
|
|
34
|
+
/** Attributes allowed across every XML part in the package. */
|
|
35
|
+
maxXmlAttributesPerPackage: number;
|
|
13
36
|
allowedMediaMimeTypes: ReadonlySet<string>;
|
|
14
37
|
};
|
|
15
38
|
type DocxUnzipOptions = Partial<Omit<DocxUnzipLimits, "allowedMediaMimeTypes">> & {
|
package/dist/docx/unzip.js
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
import { bytesToDataUrl } from "../utils/base64.js";
|
|
1
2
|
import { DOCX_CONTAINER_TYPES, detectDocxContainerType } from "./encryption/containerFormat.js";
|
|
2
3
|
import { openDocxBuffer } from "./encryption/openEncryptedDocx.js";
|
|
3
|
-
import {
|
|
4
|
+
import { decodeXmlBytes } from "./xmlEncoding.js";
|
|
5
|
+
import { FOLIO_XML_RESOURCE_LIMITS, assertXmlResourceLimits, createXmlPackageBudget } from "./xmlResourceLimits.js";
|
|
4
6
|
import JSZip from "jszip";
|
|
5
7
|
//#region src/docx/unzip.ts
|
|
6
8
|
/**
|
|
@@ -58,8 +60,21 @@ const DEFAULT_UNZIP_LIMITS = {
|
|
|
58
60
|
maxMediaBytes: 25 * MEBIBYTE,
|
|
59
61
|
maxFontBytes: 10 * MEBIBYTE,
|
|
60
62
|
maxTotalUncompressedBytes: 250 * MEBIBYTE,
|
|
63
|
+
maxXmlElementsPerPart: FOLIO_XML_RESOURCE_LIMITS.maxElementsPerPart,
|
|
64
|
+
maxXmlAttributesPerPart: FOLIO_XML_RESOURCE_LIMITS.maxAttributesPerPart,
|
|
65
|
+
maxXmlElementsPerPackage: FOLIO_XML_RESOURCE_LIMITS.maxElementsPerPackage,
|
|
66
|
+
maxXmlAttributesPerPackage: FOLIO_XML_RESOURCE_LIMITS.maxAttributesPerPackage,
|
|
61
67
|
allowedMediaMimeTypes: DEFAULT_ALLOWED_MEDIA_MIME_TYPES
|
|
62
68
|
};
|
|
69
|
+
/** The XML bounds an unzip enforces, in the shape the preflight takes. */
|
|
70
|
+
const xmlResourceLimitsFor = (limits) => ({
|
|
71
|
+
maxBytes: limits.maxXmlBytes,
|
|
72
|
+
maxDepth: FOLIO_XML_RESOURCE_LIMITS.maxDepth,
|
|
73
|
+
maxElementsPerPart: limits.maxXmlElementsPerPart,
|
|
74
|
+
maxAttributesPerPart: limits.maxXmlAttributesPerPart,
|
|
75
|
+
maxElementsPerPackage: limits.maxXmlElementsPerPackage,
|
|
76
|
+
maxAttributesPerPackage: limits.maxXmlAttributesPerPackage
|
|
77
|
+
});
|
|
63
78
|
const PARSED_XML_PARTS = /* @__PURE__ */ new Set([
|
|
64
79
|
"[content_types].xml",
|
|
65
80
|
"_rels/.rels",
|
|
@@ -147,13 +162,13 @@ async function unzipDocx(buffer, options = {}) {
|
|
|
147
162
|
if (lowerPath.endsWith(".xml") || lowerPath.endsWith(".rels")) {
|
|
148
163
|
assertEntrySize(path, declaredSize, limits.maxXmlBytes);
|
|
149
164
|
if (options.extractAllXml === false && !shouldExtractXmlPart(lowerPath)) continue;
|
|
150
|
-
extractionTasks.push(() => file.async("
|
|
151
|
-
assertExtractedSize(path,
|
|
165
|
+
extractionTasks.push(() => file.async("uint8array").then((xmlBytes) => {
|
|
166
|
+
assertExtractedSize(path, xmlBytes.byteLength, limits.maxXmlBytes);
|
|
152
167
|
return {
|
|
153
168
|
type: "xml",
|
|
154
169
|
path,
|
|
155
170
|
lowerPath,
|
|
156
|
-
content:
|
|
171
|
+
content: decodeXmlBytes(xmlBytes)
|
|
157
172
|
};
|
|
158
173
|
}));
|
|
159
174
|
} else if (lowerPath.startsWith("word/media/")) {
|
|
@@ -188,10 +203,11 @@ async function unzipDocx(buffer, options = {}) {
|
|
|
188
203
|
}));
|
|
189
204
|
}
|
|
190
205
|
}
|
|
206
|
+
const xmlBudget = createXmlPackageBudget();
|
|
191
207
|
for (const extracted of await Promise.all(extractionTasks.map((extract) => extract()))) {
|
|
192
208
|
if (!extracted) continue;
|
|
193
209
|
if (extracted.type === "xml") {
|
|
194
|
-
assignXmlContent(content, extracted, limits);
|
|
210
|
+
assignXmlContent(content, extracted, limits, xmlBudget);
|
|
195
211
|
continue;
|
|
196
212
|
}
|
|
197
213
|
if (extracted.type === "media") {
|
|
@@ -203,19 +219,21 @@ async function unzipDocx(buffer, options = {}) {
|
|
|
203
219
|
return content;
|
|
204
220
|
}
|
|
205
221
|
/**
|
|
206
|
-
*
|
|
207
|
-
*
|
|
208
|
-
*
|
|
222
|
+
* Preflight every XML part the unzip retains, not a named few.
|
|
223
|
+
*
|
|
224
|
+
* Naming the parts to bound is the bug: `word/document.xml`, `word/styles.xml`
|
|
225
|
+
* and `word/numbering.xml` were counted, and headers, footers, footnotes,
|
|
226
|
+
* endnotes, comments and every other `word/*.xml` part were parsed into trees
|
|
227
|
+
* unbounded. Bounding the reader instead of the part list means a part added
|
|
228
|
+
* later is bounded by construction, and the package budget makes the ceiling
|
|
229
|
+
* the package's rather than each part's.
|
|
209
230
|
*/
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
if (PREFLIGHT_XML_PARTS.has(lowerPath)) assertXmlResourceLimits(xmlContent, {
|
|
217
|
-
...FOLIO_XML_RESOURCE_LIMITS,
|
|
218
|
-
maxBytes: limits.maxXmlBytes
|
|
231
|
+
function assignXmlContent(content, { path, lowerPath, content: xmlContent }, limits, budget) {
|
|
232
|
+
assertXmlResourceLimits({
|
|
233
|
+
xml: xmlContent,
|
|
234
|
+
limits: xmlResourceLimitsFor(limits),
|
|
235
|
+
partPath: path,
|
|
236
|
+
budget
|
|
219
237
|
});
|
|
220
238
|
content.allXml.set(path, xmlContent);
|
|
221
239
|
if (lowerPath === "word/document.xml") content.documentXml = xmlContent;
|
|
@@ -426,14 +444,7 @@ function getMediaMimeType(path) {
|
|
|
426
444
|
* @returns Data URL string
|
|
427
445
|
*/
|
|
428
446
|
function mediaToDataUrl(data, mimeType) {
|
|
429
|
-
|
|
430
|
-
const chunks = [];
|
|
431
|
-
const chunkSize = 32768;
|
|
432
|
-
for (let offset = 0; offset < bytes.length; offset += chunkSize) {
|
|
433
|
-
const chunk = bytes.subarray(offset, offset + chunkSize);
|
|
434
|
-
chunks.push(String.fromCodePoint(...chunk));
|
|
435
|
-
}
|
|
436
|
-
return `data:${mimeType};base64,${btoa(chunks.join(""))}`;
|
|
447
|
+
return bytesToDataUrl(new Uint8Array(data), mimeType);
|
|
437
448
|
}
|
|
438
449
|
/**
|
|
439
450
|
* Extract a specific file from the original ZIP
|
|
@@ -446,7 +457,7 @@ function extractFile(content, path) {
|
|
|
446
457
|
const file = content.originalZip.file(path);
|
|
447
458
|
if (!file || !isParsedDocxEntry(path)) return Promise.resolve(null);
|
|
448
459
|
const lowerPath = path.toLowerCase();
|
|
449
|
-
if (lowerPath.endsWith(".xml") || lowerPath.endsWith(".rels")) return file.async("
|
|
460
|
+
if (lowerPath.endsWith(".xml") || lowerPath.endsWith(".rels")) return file.async("uint8array").then(decodeXmlBytes);
|
|
450
461
|
return file.async("arraybuffer");
|
|
451
462
|
}
|
|
452
463
|
/**
|
|
@@ -11,6 +11,22 @@ import { XmlElement } from "./xmlParser.js";
|
|
|
11
11
|
* same artwork twice.
|
|
12
12
|
*/
|
|
13
13
|
declare function shouldPreserveRawVmlPict(pictElement: XmlElement): boolean;
|
|
14
|
+
/**
|
|
15
|
+
* Whether the run parser turns this `w:pict` into run content of its own.
|
|
16
|
+
*
|
|
17
|
+
* Exactly one owner may represent a `w:pict`: the run parser, whose drawing
|
|
18
|
+
* carries the whole element as captured XML, or the text-box enrichment pass,
|
|
19
|
+
* which rebuilds one `v:textbox` as an editable shape. Two owners write the
|
|
20
|
+
* same artwork twice, and when the pict holds text, the saved document says it
|
|
21
|
+
* twice.
|
|
22
|
+
*
|
|
23
|
+
* The enrichment asks this rather than re-deriving the answer from the markup:
|
|
24
|
+
* its own reading (`v:imagedata` present) named only the picture path, so a
|
|
25
|
+
* `v:group` claimed by the preview path was represented by both owners. The
|
|
26
|
+
* result is a pure function of the element, so asking it a second time costs
|
|
27
|
+
* only the work the run parser already did.
|
|
28
|
+
*/
|
|
29
|
+
declare const isVmlPictParsedByRunParser: (pictElement: XmlElement, rels: document_d_exports.RelationshipMap | null, media: Map<string, document_d_exports.MediaFile> | null) => boolean;
|
|
14
30
|
/**
|
|
15
31
|
* Parse a `w:pict` element into an inline image, or null when it carries no
|
|
16
32
|
* ordinary VML picture (no resolvable `<v:imagedata>`, or a watermark shape).
|
|
@@ -22,4 +38,4 @@ declare function shouldPreserveRawVmlPict(pictElement: XmlElement): boolean;
|
|
|
22
38
|
*/
|
|
23
39
|
declare function parseVmlImageContent(pictElement: XmlElement, rels: document_d_exports.RelationshipMap | null, media: Map<string, document_d_exports.MediaFile> | null, rootXmlns?: Record<string, string>): document_d_exports.DrawingContent | null;
|
|
24
40
|
//#endregion
|
|
25
|
-
export { parseVmlImageContent, shouldPreserveRawVmlPict };
|
|
41
|
+
export { isVmlPictParsedByRunParser, parseVmlImageContent, shouldPreserveRawVmlPict };
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { sanitizeImageSrc } from "../utils/sanitizeImageSrc.js";
|
|
2
2
|
import { pixelsToEmu } from "../utils/units.js";
|
|
3
3
|
import { resolveImageData } from "./imageParser.js";
|
|
4
|
+
import { PREVIEW_KINDS } from "./previewBudget.js";
|
|
4
5
|
import { captureVerbatimXml } from "./verbatimCapture.js";
|
|
5
6
|
import { isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl } from "./vmlPreview.js";
|
|
6
7
|
import { isWatermarkShape } from "./watermarkParser.js";
|
|
@@ -73,8 +74,8 @@ const previewImage = (pictElement, svg, widthPx, heightPx, style, rootXmlns) =>
|
|
|
73
74
|
type: "image",
|
|
74
75
|
rId: "",
|
|
75
76
|
src,
|
|
76
|
-
mimeType:
|
|
77
|
-
filename:
|
|
77
|
+
mimeType: PREVIEW_KINDS.vmlShape.mimeType,
|
|
78
|
+
filename: PREVIEW_KINDS.vmlShape.filename,
|
|
78
79
|
size: {
|
|
79
80
|
width: pixelsToEmu(widthPx),
|
|
80
81
|
height: pixelsToEmu(heightPx)
|
|
@@ -130,6 +131,22 @@ function shouldPreserveRawVmlPict(pictElement) {
|
|
|
130
131
|
return shapes.length > 0 && !shapes.some((shape) => isWatermarkShape(shape));
|
|
131
132
|
}
|
|
132
133
|
/**
|
|
134
|
+
* Whether the run parser turns this `w:pict` into run content of its own.
|
|
135
|
+
*
|
|
136
|
+
* Exactly one owner may represent a `w:pict`: the run parser, whose drawing
|
|
137
|
+
* carries the whole element as captured XML, or the text-box enrichment pass,
|
|
138
|
+
* which rebuilds one `v:textbox` as an editable shape. Two owners write the
|
|
139
|
+
* same artwork twice, and when the pict holds text, the saved document says it
|
|
140
|
+
* twice.
|
|
141
|
+
*
|
|
142
|
+
* The enrichment asks this rather than re-deriving the answer from the markup:
|
|
143
|
+
* its own reading (`v:imagedata` present) named only the picture path, so a
|
|
144
|
+
* `v:group` claimed by the preview path was represented by both owners. The
|
|
145
|
+
* result is a pure function of the element, so asking it a second time costs
|
|
146
|
+
* only the work the run parser already did.
|
|
147
|
+
*/
|
|
148
|
+
const isVmlPictParsedByRunParser = (pictElement, rels, media) => parseVmlImageContent(pictElement, rels, media) !== null || shouldPreserveRawVmlPict(pictElement);
|
|
149
|
+
/**
|
|
133
150
|
* Read the relationship id off a `v:imagedata` element. Word writes `r:id`;
|
|
134
151
|
* some legacy / third-party generators use `r:embed` or the office-namespace
|
|
135
152
|
* `o:relid` instead, so fall back through those before the bare `id`.
|
|
@@ -200,4 +217,4 @@ function parseVmlImageContent(pictElement, rels, media, rootXmlns = {}) {
|
|
|
200
217
|
return null;
|
|
201
218
|
}
|
|
202
219
|
//#endregion
|
|
203
|
-
export { parseVmlImageContent, shouldPreserveRawVmlPict };
|
|
220
|
+
export { isVmlPictParsedByRunParser, parseVmlImageContent, shouldPreserveRawVmlPict };
|
|
@@ -25,7 +25,5 @@ declare const vmlSvgDataUrl: (svg: string) => string | undefined;
|
|
|
25
25
|
declare const renderStandaloneVmlPreview: (shape: XmlElement) => VmlPreviewResult;
|
|
26
26
|
/** Render a VML group through bounded, non-clipping local-coordinate transforms. */
|
|
27
27
|
declare const renderVmlGroupPreview: (group: XmlElement) => VmlPreviewResult;
|
|
28
|
-
/** Bound retained synthetic VML previews while preserving their raw replay nodes. */
|
|
29
|
-
declare const enforcePackageVmlPreviewBudget: (root: unknown, maxCharacters?: number) => void;
|
|
30
28
|
//#endregion
|
|
31
|
-
export { VmlPreviewResult,
|
|
29
|
+
export { VmlPreviewResult, isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
|
package/dist/docx/vmlPreview.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { VML_PREVIEW_DATA_URL_PREFIX } from "./previewBudget.js";
|
|
1
2
|
import { findChild, findDeep, getAttribute, getChildElements, getLocalName } from "./xmlParser.js";
|
|
2
3
|
//#region src/docx/vmlPreview.ts
|
|
3
4
|
const MAX_VML_PREVIEW_DEPTH = 16;
|
|
@@ -6,10 +7,6 @@ const MAX_VML_PREVIEW_PATH_POINTS = 2e4;
|
|
|
6
7
|
const MAX_VML_PREVIEW_COORDINATE = 1e6;
|
|
7
8
|
const MAX_VML_PREVIEW_DIMENSION_PX = 2e4;
|
|
8
9
|
const MAX_VML_SVG_CHARACTERS = 1e6;
|
|
9
|
-
const MAX_PACKAGE_VML_PREVIEW_CHARACTERS = 8 * 1024 * 1024;
|
|
10
|
-
const VML_PREVIEW_DATA_URL_PREFIX = "data:image/svg+xml;charset=utf-8,";
|
|
11
|
-
const VML_PREVIEW_FILENAME = "vml-shape-preview.svg";
|
|
12
|
-
const VML_PREVIEW_MIME_TYPE = "image/svg+xml";
|
|
13
10
|
const SAFE_VML_COLORS = /* @__PURE__ */ new Set([
|
|
14
11
|
"black",
|
|
15
12
|
"white",
|
|
@@ -506,30 +503,5 @@ const renderVmlGroupPreview = (group) => {
|
|
|
506
503
|
style
|
|
507
504
|
};
|
|
508
505
|
};
|
|
509
|
-
/** Bound retained synthetic VML previews while preserving their raw replay nodes. */
|
|
510
|
-
const enforcePackageVmlPreviewBudget = (root, maxCharacters = MAX_PACKAGE_VML_PREVIEW_CHARACTERS) => {
|
|
511
|
-
let remainingCharacters = Math.max(0, maxCharacters);
|
|
512
|
-
const visited = /* @__PURE__ */ new WeakSet();
|
|
513
|
-
const visit = (value) => {
|
|
514
|
-
if (value === null || typeof value !== "object" || visited.has(value)) return;
|
|
515
|
-
visited.add(value);
|
|
516
|
-
if (value instanceof ArrayBuffer || ArrayBuffer.isView(value)) return;
|
|
517
|
-
if (value instanceof Map) {
|
|
518
|
-
for (const child of value.values()) visit(child);
|
|
519
|
-
return;
|
|
520
|
-
}
|
|
521
|
-
if (Array.isArray(value)) {
|
|
522
|
-
for (const child of value) visit(child);
|
|
523
|
-
return;
|
|
524
|
-
}
|
|
525
|
-
if ("type" in value && value.type === "image" && "rId" in value && value.rId === "" && "mimeType" in value && value.mimeType === VML_PREVIEW_MIME_TYPE && "filename" in value && value.filename === VML_PREVIEW_FILENAME && "src" in value && typeof value.src === "string" && value.src.startsWith(VML_PREVIEW_DATA_URL_PREFIX)) if (value.src.length <= remainingCharacters) remainingCharacters -= value.src.length;
|
|
526
|
-
else {
|
|
527
|
-
remainingCharacters = 0;
|
|
528
|
-
delete value.src;
|
|
529
|
-
}
|
|
530
|
-
for (const child of Object.values(value)) visit(child);
|
|
531
|
-
};
|
|
532
|
-
visit(root);
|
|
533
|
-
};
|
|
534
506
|
//#endregion
|
|
535
|
-
export {
|
|
507
|
+
export { isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
//#region src/docx/xmlEncoding.ts
|
|
2
|
+
/** Decode the UTF-8 and UTF-16 encodings XML processors must recognize. */
|
|
3
|
+
const decodeXmlBytes = (bytes) => {
|
|
4
|
+
if (bytes[0] === 255 && bytes[1] === 254) return new TextDecoder("utf-16le").decode(bytes.subarray(2));
|
|
5
|
+
if (bytes[0] === 254 && bytes[1] === 255) return new TextDecoder("utf-16be").decode(bytes.subarray(2));
|
|
6
|
+
if (bytes[0] === 60 && bytes[1] === 0 && bytes[2] === 63 && bytes[3] === 0) return new TextDecoder("utf-16le").decode(bytes);
|
|
7
|
+
if (bytes[0] === 0 && bytes[1] === 60 && bytes[2] === 0 && bytes[3] === 63) return new TextDecoder("utf-16be").decode(bytes);
|
|
8
|
+
const utf8Offset = bytes[0] === 239 && bytes[1] === 187 && bytes[2] === 191 ? 3 : 0;
|
|
9
|
+
return new TextDecoder("utf-8").decode(bytes.subarray(utf8Offset));
|
|
10
|
+
};
|
|
11
|
+
//#endregion
|
|
12
|
+
export { decodeXmlBytes };
|
package/dist/docx/xmlParser.d.ts
CHANGED
|
@@ -71,7 +71,22 @@ declare const NAMESPACES: {
|
|
|
71
71
|
declare const OOXML_NAMESPACE_SCOPE: XmlNamespaceScope;
|
|
72
72
|
declare function parseXml(xml: string, inheritedNamespaceScope?: XmlNamespaceScope): XmlElement;
|
|
73
73
|
/**
|
|
74
|
-
* Serialize an XmlElement back to an XML string
|
|
74
|
+
* Serialize an XmlElement back to an XML string.
|
|
75
|
+
*
|
|
76
|
+
* Written here rather than handed to `fast-xml-parser`'s builder for two
|
|
77
|
+
* reasons. The builder writes a tab or a newline inside an attribute value
|
|
78
|
+
* literally, and XML 1.0 §3.3.3 has every conformant reader normalise those to
|
|
79
|
+
* a space: a `descr="two lines
"` a source file wrote came back as
|
|
80
|
+
* `descr="two lines "`. `escapeXmlAttribute` writes the character references
|
|
81
|
+
* that survive that step. And the builder returned a string per node, so a
|
|
82
|
+
* subtree's bytes were copied into its parent's answer, its grandparent's and
|
|
83
|
+
* so on, which `writeElement` below replaces with one buffer.
|
|
84
|
+
*
|
|
85
|
+
* Those two are the whole difference from the builder: the attribute
|
|
86
|
+
* character references, an attribute the model dropped written as absent
|
|
87
|
+
* rather than as the word "undefined", and the characters XML 1.0 §2.2 admits
|
|
88
|
+
* no spelling for dropped. Every other rule reproduces its output byte for
|
|
89
|
+
* byte, so replaying a capture is unchanged wherever it was already correct.
|
|
75
90
|
*/
|
|
76
91
|
declare function elementToXml(element: XmlElement): string;
|
|
77
92
|
/**
|
|
@@ -90,6 +105,15 @@ declare function getLocalName(name: string | undefined): string;
|
|
|
90
105
|
declare function getNamespacePrefix(name: string): string | null;
|
|
91
106
|
/** Namespace URI resolved from the element's in-scope XML declarations. */
|
|
92
107
|
declare const getNamespaceUri: (element: XmlElement) => string | undefined;
|
|
108
|
+
/**
|
|
109
|
+
* Namespace URI of one of the element's attributes, by its source spelling.
|
|
110
|
+
*
|
|
111
|
+
* An unprefixed attribute has no namespace — it is not in the element's,
|
|
112
|
+
* which is why this is not {@link getNamespaceUri} with a different argument —
|
|
113
|
+
* and `undefined` is that answer as well as "the prefix resolves to nothing".
|
|
114
|
+
* A caller that must tell the two apart checks the spelling for a colon.
|
|
115
|
+
*/
|
|
116
|
+
declare const resolveAttributeNamespaceUri: (element: XmlElement, attributeName: string) => string | undefined;
|
|
93
117
|
/** WordprocessingML main namespace, Transitional and Strict (ECMA-376 Parts 1 and 4). */
|
|
94
118
|
declare const WORDPROCESSINGML_NAMESPACE_URIS: ReadonlySet<string>;
|
|
95
119
|
/** Office document relationship attributes, Transitional and Strict. */
|
|
@@ -368,4 +392,4 @@ declare function cloneWithXmlnsDeclarations(element: XmlElement, xmlnsDecls: Rec
|
|
|
368
392
|
*/
|
|
369
393
|
declare function cloneElement(element: XmlElement, overrides: Partial<XmlElement>): XmlElement;
|
|
370
394
|
//#endregion
|
|
371
|
-
export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, XmlAttributeMatch, XmlElement, XmlNamespaceScope, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, selectAlternateContentBranch };
|
|
395
|
+
export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, XmlAttributeMatch, XmlElement, XmlNamespaceScope, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, resolveAttributeNamespaceUri, selectAlternateContentBranch };
|
package/dist/docx/xmlParser.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { NUMBERS_PER_PERCENT, percentageSpelling, transitionalSlotEncoding } from "./transitionalSpelling.js";
|
|
2
2
|
import { universalMeasureAs } from "./universalMeasure.js";
|
|
3
|
-
import {
|
|
3
|
+
import { escapeXmlAttribute, escapeXmlText } from "@stll/docx-core";
|
|
4
|
+
import { XMLParser } from "fast-xml-parser";
|
|
4
5
|
import { OOXML_NS } from "@stll/docx-utils";
|
|
5
6
|
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
6
7
|
//#region src/docx/xmlParser.ts
|
|
@@ -42,22 +43,12 @@ const fxpParserOptionsWithStopNodes = {
|
|
|
42
43
|
...fxpParserOptions,
|
|
43
44
|
stopNodes: ["*.w:binData"]
|
|
44
45
|
};
|
|
45
|
-
const fxpBuilderOptions = {
|
|
46
|
-
preserveOrder: true,
|
|
47
|
-
ignoreAttributes: false,
|
|
48
|
-
attributeNamePrefix: "",
|
|
49
|
-
textNodeName: "#text",
|
|
50
|
-
suppressEmptyNode: true
|
|
51
|
-
};
|
|
52
46
|
const fxpParser = new XMLParser(fxpParserOptions);
|
|
53
47
|
const fxpParserWithStopNodes = new XMLParser(fxpParserOptionsWithStopNodes);
|
|
54
|
-
const fxpBuilder = new XMLBuilder(fxpBuilderOptions);
|
|
55
48
|
/** Text node key used by fast-xml-parser in preserveOrder mode. */
|
|
56
49
|
const TEXT_KEY = "#text";
|
|
57
50
|
/** Attribute group key used by fast-xml-parser in preserveOrder mode. */
|
|
58
51
|
const ATTR_KEY = ":@";
|
|
59
|
-
/** Character reference required to keep carriage returns through XML end-of-line normalization. */
|
|
60
|
-
const XML_CARRIAGE_RETURN_REFERENCE = " ";
|
|
61
52
|
const EMPTY_NAMESPACE_SCOPE = { bindings: /* @__PURE__ */ new Map() };
|
|
62
53
|
const resolveNamespaceUri = (scope, prefix) => {
|
|
63
54
|
let current = scope;
|
|
@@ -138,18 +129,6 @@ function fxpToRootElement(nodes, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE
|
|
|
138
129
|
return { elements: nodes };
|
|
139
130
|
}
|
|
140
131
|
/**
|
|
141
|
-
* Convert an XmlElement back into the fast-xml-parser preserveOrder format
|
|
142
|
-
* so we can feed it to XMLBuilder.
|
|
143
|
-
*/
|
|
144
|
-
function elementToFxpNode(el) {
|
|
145
|
-
if (el.type === "text") return { [TEXT_KEY]: el.text ?? "" };
|
|
146
|
-
const name = el.name ?? "";
|
|
147
|
-
const children = el.elements ? el.elements.map(elementToFxpNode) : [];
|
|
148
|
-
const node = { [name]: children };
|
|
149
|
-
if (el.attributes && Object.keys(el.attributes).length > 0) node[ATTR_KEY] = el.attributes;
|
|
150
|
-
return node;
|
|
151
|
-
}
|
|
152
|
-
/**
|
|
153
132
|
* Common OOXML namespace URIs — re-exported from @stll/docx-utils.
|
|
154
133
|
*/
|
|
155
134
|
const NAMESPACES = OOXML_NS;
|
|
@@ -174,13 +153,64 @@ function parseXml(xml, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE) {
|
|
|
174
153
|
return fxpToRootElement((xml.includes("binData") ? fxpParserWithStopNodes : fxpParser).parse(xml), inheritedNamespaceScope);
|
|
175
154
|
}
|
|
176
155
|
/**
|
|
177
|
-
* Serialize an XmlElement back to an XML string
|
|
156
|
+
* Serialize an XmlElement back to an XML string.
|
|
157
|
+
*
|
|
158
|
+
* Written here rather than handed to `fast-xml-parser`'s builder for two
|
|
159
|
+
* reasons. The builder writes a tab or a newline inside an attribute value
|
|
160
|
+
* literally, and XML 1.0 §3.3.3 has every conformant reader normalise those to
|
|
161
|
+
* a space: a `descr="two lines
"` a source file wrote came back as
|
|
162
|
+
* `descr="two lines "`. `escapeXmlAttribute` writes the character references
|
|
163
|
+
* that survive that step. And the builder returned a string per node, so a
|
|
164
|
+
* subtree's bytes were copied into its parent's answer, its grandparent's and
|
|
165
|
+
* so on, which `writeElement` below replaces with one buffer.
|
|
166
|
+
*
|
|
167
|
+
* Those two are the whole difference from the builder: the attribute
|
|
168
|
+
* character references, an attribute the model dropped written as absent
|
|
169
|
+
* rather than as the word "undefined", and the characters XML 1.0 §2.2 admits
|
|
170
|
+
* no spelling for dropped. Every other rule reproduces its output byte for
|
|
171
|
+
* byte, so replaying a capture is unchanged wherever it was already correct.
|
|
178
172
|
*/
|
|
179
173
|
function elementToXml(element) {
|
|
180
|
-
const
|
|
181
|
-
|
|
174
|
+
const out = [];
|
|
175
|
+
writeElement(element, out);
|
|
176
|
+
return out.join("");
|
|
182
177
|
}
|
|
183
178
|
/**
|
|
179
|
+
* Serialize into one shared buffer rather than a string per node.
|
|
180
|
+
*
|
|
181
|
+
* Returning a string per element makes a subtree's bytes a substring of its
|
|
182
|
+
* parent's, its grandparent's and so on, so a table pays for its rows, its
|
|
183
|
+
* rows pay for their cells, and a part that nests four levels deep is copied
|
|
184
|
+
* four times before anything is written. Appending into one array of chunks
|
|
185
|
+
* and joining once costs each node its own text and nothing for its ancestors.
|
|
186
|
+
*
|
|
187
|
+
* Whether an element self-closes is not known until its children are written,
|
|
188
|
+
* so the opening tag reserves a slot in the buffer and fills it afterwards:
|
|
189
|
+
* `>` when something was appended, `/>` when nothing was.
|
|
190
|
+
*/
|
|
191
|
+
const writeElement = (element, out) => {
|
|
192
|
+
if (element.type === "text") {
|
|
193
|
+
const text = String(element.text ?? "");
|
|
194
|
+
if (text !== "") out.push(escapeXmlText(text));
|
|
195
|
+
return;
|
|
196
|
+
}
|
|
197
|
+
const name = element.name ?? "";
|
|
198
|
+
out.push(`<${name}`);
|
|
199
|
+
if (element.attributes) for (const [attribute, value] of Object.entries(element.attributes)) {
|
|
200
|
+
if (value === void 0) continue;
|
|
201
|
+
out.push(` ${attribute}="${escapeXmlAttribute(String(value))}"`);
|
|
202
|
+
}
|
|
203
|
+
const openingSlot = out.push("") - 1;
|
|
204
|
+
const contentStart = out.length;
|
|
205
|
+
for (const child of element.elements ?? []) writeElement(child, out);
|
|
206
|
+
if (out.length === contentStart) {
|
|
207
|
+
out[openingSlot] = "/>";
|
|
208
|
+
return;
|
|
209
|
+
}
|
|
210
|
+
out[openingSlot] = ">";
|
|
211
|
+
out.push(`</${name}>`);
|
|
212
|
+
};
|
|
213
|
+
/**
|
|
184
214
|
* Parse XML string to a more convenient format
|
|
185
215
|
*/
|
|
186
216
|
function parseXmlDocument(xml) {
|
|
@@ -216,6 +246,19 @@ function getNamespacePrefix(name) {
|
|
|
216
246
|
}
|
|
217
247
|
/** Namespace URI resolved from the element's in-scope XML declarations. */
|
|
218
248
|
const getNamespaceUri = (element) => element.namespaceUri;
|
|
249
|
+
/**
|
|
250
|
+
* Namespace URI of one of the element's attributes, by its source spelling.
|
|
251
|
+
*
|
|
252
|
+
* An unprefixed attribute has no namespace — it is not in the element's,
|
|
253
|
+
* which is why this is not {@link getNamespaceUri} with a different argument —
|
|
254
|
+
* and `undefined` is that answer as well as "the prefix resolves to nothing".
|
|
255
|
+
* A caller that must tell the two apart checks the spelling for a colon.
|
|
256
|
+
*/
|
|
257
|
+
const resolveAttributeNamespaceUri = (element, attributeName) => {
|
|
258
|
+
const colonIndex = attributeName.indexOf(":");
|
|
259
|
+
if (colonIndex === -1) return;
|
|
260
|
+
return resolveNamespaceUri(element.namespaceScope, attributeName.slice(0, colonIndex));
|
|
261
|
+
};
|
|
219
262
|
/** WordprocessingML main namespace, Transitional and Strict (ECMA-376 Parts 1 and 4). */
|
|
220
263
|
const WORDPROCESSINGML_NAMESPACE_URIS = /* @__PURE__ */ new Set([NAMESPACES.w, "http://purl.oclc.org/ooxml/wordprocessingml/main"]);
|
|
221
264
|
/** Office document relationship attributes, Transitional and Strict. */
|
|
@@ -825,4 +868,4 @@ function cloneElement(element, overrides) {
|
|
|
825
868
|
return clone;
|
|
826
869
|
}
|
|
827
870
|
//#endregion
|
|
828
|
-
export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, selectAlternateContentBranch };
|
|
871
|
+
export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, resolveAttributeNamespaceUri, selectAlternateContentBranch };
|