@stll/folio-core 0.43.0 → 0.45.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ai-edits/__fixtures__/paragraphs.js +2 -2
- package/dist/ai-edits/headless.js +7 -5
- package/dist/ai-edits/index.d.ts +2 -2
- package/dist/ai-edits/index.js +2 -2
- package/dist/ai-edits/snapshot.js +13 -9
- package/dist/compare/content-alignment.js +94 -54
- package/dist/compare/inline-atoms.js +34 -20
- package/dist/compare/style-resources.js +6 -0
- package/dist/content-controls/mutateContentControls.js +4 -2
- package/dist/display-list/dom/renderDisplayListToDom.js +8 -8
- package/dist/document-operations.js +14 -3
- package/dist/docx/appVersionNormalization.d.ts +0 -18
- package/dist/docx/blockContentParser.js +8 -0
- package/dist/docx/blockRangeMarkers.d.ts +36 -0
- package/dist/docx/blockRangeMarkers.js +59 -0
- package/dist/docx/bookmarkParser.d.ts +2 -20
- package/dist/docx/bookmarkParser.js +6 -30
- package/dist/docx/borderParser.d.ts +13 -0
- package/dist/docx/borderParser.js +71 -0
- package/dist/docx/builtInStyles.d.ts +165 -0
- package/dist/docx/builtInStyles.js +239 -0
- package/dist/docx/commentIdNormalization.d.ts +3 -1
- package/dist/docx/commentIdNormalization.js +18 -1
- package/dist/docx/commentParser.d.ts +2 -1
- package/dist/docx/commentParser.js +80 -42
- package/dist/docx/commentReferenceNormalization.d.ts +4 -1
- package/dist/docx/commentReferenceNormalization.js +23 -14
- package/dist/docx/commentThreadKey.d.ts +18 -0
- package/dist/docx/commentThreadKey.js +22 -0
- package/dist/docx/danglingRelationshipReferences.d.ts +15 -0
- package/dist/docx/danglingRelationshipReferences.js +30 -0
- package/dist/docx/defaultParagraphStyle.d.ts +18 -1
- package/dist/docx/defaultParagraphStyle.js +23 -1
- package/dist/docx/diagramPreview.js +87 -27
- package/dist/docx/documentParser.d.ts +2 -1
- package/dist/docx/documentParser.js +2 -2
- package/dist/docx/drawingUtils.d.ts +8 -1
- package/dist/docx/drawingUtils.js +12 -3
- package/dist/docx/fieldParser.js +3 -5
- package/dist/docx/footnoteParser.d.ts +3 -2
- package/dist/docx/footnoteParser.js +19 -4
- package/dist/docx/groupDrawingParser.js +4 -4
- package/dist/docx/headerFooterRefParser.d.ts +4 -3
- package/dist/docx/headerFooterRefParser.js +42 -12
- package/dist/docx/headerFooterReferenceNormalization.d.ts +4 -1
- package/dist/docx/headerFooterReferenceNormalization.js +5 -1
- package/dist/docx/hyperlinkParser.js +13 -17
- package/dist/docx/imageParser.d.ts +10 -2
- package/dist/docx/imageParser.js +80 -30
- package/dist/docx/imageRawXml.d.ts +14 -1
- package/dist/docx/imageRawXml.js +35 -11
- package/dist/docx/markupRangeMarker.d.ts +15 -0
- package/dist/docx/markupRangeMarker.js +44 -0
- package/dist/docx/mathToMathml.js +12 -14
- package/dist/docx/nonVisualDrawingProps.d.ts +34 -0
- package/dist/docx/nonVisualDrawingProps.js +46 -0
- package/dist/docx/noteReferenceStyles.d.ts +29 -0
- package/dist/docx/noteReferenceStyles.js +70 -0
- package/dist/docx/numberingReferenceNormalization.d.ts +4 -1
- package/dist/docx/numberingReferenceNormalization.js +20 -1
- package/dist/docx/paraIdRangeNormalization.d.ts +0 -19
- package/dist/docx/paragraphParser.js +66 -99
- package/dist/docx/paragraphPropertySource.js +1 -0
- package/dist/docx/paragraphTextBoxEnrichment.js +3 -0
- package/dist/docx/paragraphTraversal.d.ts +37 -1
- package/dist/docx/paragraphTraversal.js +84 -1
- package/dist/docx/parseContext.d.ts +37 -0
- package/dist/docx/parseContext.js +67 -0
- package/dist/docx/parseWarningMessage.d.ts +6 -0
- package/dist/docx/parseWarningMessage.js +44 -0
- package/dist/docx/parser.js +83 -29
- package/dist/docx/previewBudget.d.ts +64 -0
- package/dist/docx/previewBudget.js +88 -0
- package/dist/docx/relsParser.d.ts +28 -11
- package/dist/docx/relsParser.js +26 -13
- package/dist/docx/revisionIdNormalization.js +96 -10
- package/dist/docx/rezip.js +80 -40
- package/dist/docx/runConsolidator.js +1 -2
- package/dist/docx/runParser.d.ts +8 -1
- package/dist/docx/runParser.js +30 -48
- package/dist/docx/sdtPropertiesPatch.js +24 -18
- package/dist/docx/sectionParser.d.ts +2 -1
- package/dist/docx/sectionParser.js +21 -65
- package/dist/docx/sectionReferenceHistory.js +2 -2
- package/dist/docx/selectiveSave.js +6 -6
- package/dist/docx/serializer/blockSdtSerializer.js +38 -26
- package/dist/docx/serializer/borderSerializer.d.ts +2 -3
- package/dist/docx/serializer/borderSerializer.js +13 -12
- package/dist/docx/serializer/commentSerializer.d.ts +41 -16
- package/dist/docx/serializer/commentSerializer.js +82 -72
- package/dist/docx/serializer/documentSerializer.d.ts +1 -5
- package/dist/docx/serializer/documentSerializer.js +6 -16
- package/dist/docx/serializer/fontTableSerializer.js +6 -6
- package/dist/docx/serializer/headerFooterSerializer.js +10 -5
- package/dist/docx/serializer/markupRangeAttributes.d.ts +8 -0
- package/dist/docx/serializer/markupRangeAttributes.js +24 -0
- package/dist/docx/serializer/noteSerializer.js +5 -0
- package/dist/docx/serializer/numberingSerializer.js +7 -6
- package/dist/docx/serializer/paragraphSerializer.d.ts +1 -5
- package/dist/docx/serializer/paragraphSerializer.js +47 -52
- package/dist/docx/serializer/partNamespaces.js +2 -2
- package/dist/docx/serializer/runSerializer.js +57 -31
- package/dist/docx/serializer/sectionPropertiesSerializer.js +11 -10
- package/dist/docx/serializer/settingsSerializer.js +4 -3
- package/dist/docx/serializer/stylesSerializer.js +6 -6
- package/dist/docx/serializer/tableSerializer.js +37 -21
- package/dist/docx/serializer/textFormattingSerializer.d.ts +2 -3
- package/dist/docx/serializer/textFormattingSerializer.js +29 -28
- package/dist/docx/serializer/themeSerializer.js +6 -6
- package/dist/docx/serializer/trackedChangeAttributes.js +2 -2
- package/dist/docx/serializer/xmlUtils.d.ts +1 -2
- package/dist/docx/serializer/xmlUtils.js +1 -13
- package/dist/docx/server/boundedArchive.d.ts +12 -0
- package/dist/docx/server/boundedArchive.js +20 -1
- package/dist/docx/server/build.js +8 -1
- package/dist/docx/server/createBilingualDocument.js +10 -18
- package/dist/docx/server/extractDocxText.js +3 -4
- package/dist/docx/server/validateDocxConformance.js +22 -1
- package/dist/docx/shadingParser.d.ts +6 -0
- package/dist/docx/shadingParser.js +32 -0
- package/dist/docx/shapeParser.js +10 -8
- package/dist/docx/styleParser.js +13 -87
- package/dist/docx/styleReferenceResolution.d.ts +36 -0
- package/dist/docx/styleReferenceResolution.js +51 -0
- package/dist/docx/tableLook.d.ts +57 -0
- package/dist/docx/tableLook.js +63 -0
- package/dist/docx/tableParser.d.ts +7 -9
- package/dist/docx/tableParser.js +64 -110
- package/dist/docx/textBoxParser.js +11 -6
- package/dist/docx/trackedMoveRangeNormalization.d.ts +3 -1
- package/dist/docx/trackedMoveRangeNormalization.js +11 -21
- package/dist/docx/transitionalSpelling.d.ts +13 -2
- package/dist/docx/transitionalSpelling.js +23 -1
- package/dist/docx/unzip.d.ts +23 -0
- package/dist/docx/unzip.js +32 -22
- package/dist/docx/verbatimCapture.js +5 -12
- package/dist/docx/vmlImageParser.js +5 -4
- package/dist/docx/vmlPreview.d.ts +1 -3
- package/dist/docx/vmlPreview.js +2 -30
- package/dist/docx/watermarkParser.js +2 -2
- package/dist/docx/xmlParser.d.ts +38 -33
- package/dist/docx/xmlParser.js +92 -47
- package/dist/docx/xmlResourceLimits.d.ts +89 -9
- package/dist/docx/xmlResourceLimits.js +105 -24
- package/dist/internal/pageBreakRunSourceDescendantIndex.js +2 -1
- package/dist/internal/paragraphFormattingSerialization.d.ts +2 -3
- package/dist/internal/paragraphFormattingSerialization.js +29 -8
- package/dist/layout-bridge/convert/footnoteLayout.js +2 -7
- package/dist/layout-engine/index.d.ts +2 -2
- package/dist/layout-engine/index.js +2 -2
- package/dist/layout-engine/measure/measureBlocks.js +1 -6
- package/dist/layout-engine/types.d.ts +8 -2
- package/dist/layout-engine/types.js +35 -2
- package/dist/layout-painter/renderImage.js +4 -3
- package/dist/layout-painter/renderParagraph.js +4 -3
- package/dist/managers/autoSaveCodec.js +2 -8
- package/dist/markdown/images.js +1 -4
- package/dist/markdown/index.js +1 -1
- package/dist/markdown/internals.d.ts +6 -1
- package/dist/markdown/internals.js +14 -1
- package/dist/markdown/renderBlock.js +35 -21
- package/dist/markdown/renderParagraph.js +14 -5
- package/dist/markdown/renderRuns.js +4 -3
- package/dist/markdown/renderTable.js +4 -3
- package/dist/markdown/trailers.js +41 -7
- package/dist/markdown/types.d.ts +3 -7
- package/dist/prosemirror/attrs/index.js +71 -5
- package/dist/prosemirror/bookmarkBoundaryAttrs.d.ts +11 -1
- package/dist/prosemirror/bookmarkBoundaryAttrs.js +18 -3
- package/dist/prosemirror/commands/image.js +1 -0
- package/dist/prosemirror/commands/index.d.ts +3 -3
- package/dist/prosemirror/commands/index.js +2 -2
- package/dist/prosemirror/commands/paragraph.d.ts +3 -3
- package/dist/prosemirror/commands/paragraph.js +2 -2
- package/dist/prosemirror/commentIdAllocator.js +2 -7
- package/dist/prosemirror/conversion/fromProseDoc.js +197 -68
- package/dist/prosemirror/conversion/toProseDoc.d.ts +1 -14
- package/dist/prosemirror/conversion/toProseDoc.js +458 -335
- package/dist/prosemirror/extensions/core/ParagraphExtension.d.ts +14 -1
- package/dist/prosemirror/extensions/core/ParagraphExtension.js +11 -6
- package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -3
- package/dist/prosemirror/extensions/features/PasteCleanupExtension.d.ts +4 -1
- package/dist/prosemirror/extensions/features/PasteCleanupExtension.js +6 -2
- package/dist/prosemirror/extensions/features/pastedHeadingStyles.d.ts +7 -0
- package/dist/prosemirror/extensions/features/pastedHeadingStyles.js +74 -0
- package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +2 -3
- package/dist/prosemirror/extensions/marks/markUtils.d.ts +11 -3
- package/dist/prosemirror/extensions/marks/markUtils.js +98 -19
- package/dist/prosemirror/extensions/nodes/BookmarkBoundaryExtension.js +7 -3
- package/dist/prosemirror/extensions/nodes/ImageExtension.js +6 -1
- package/dist/prosemirror/extensions/nodes/ShapeExtension.js +8 -2
- package/dist/prosemirror/extensions/nodes/TableExtension.js +15 -1
- package/dist/prosemirror/extensions/nodes/TextBoxExtension.js +8 -4
- package/dist/prosemirror/extensions/types.d.ts +2 -2
- package/dist/prosemirror/index.d.ts +3 -3
- package/dist/prosemirror/index.js +3 -3
- package/dist/prosemirror/insertOperations.d.ts +9 -2
- package/dist/prosemirror/insertOperations.js +9 -4
- package/dist/prosemirror/paragraphFormattingProvenance.d.ts +162 -0
- package/dist/prosemirror/paragraphFormattingProvenance.js +115 -0
- package/dist/prosemirror/plugins/documentStyles.d.ts +9 -1
- package/dist/prosemirror/plugins/documentStyles.js +11 -1
- package/dist/prosemirror/plugins/index.d.ts +2 -2
- package/dist/prosemirror/plugins/index.js +2 -2
- package/dist/prosemirror/plugins/revisionIds.d.ts +11 -2
- package/dist/prosemirror/plugins/revisionIds.js +21 -6
- package/dist/prosemirror/runFormattingReconciliation.js +3 -2
- package/dist/prosemirror/runStyleFormatting.d.ts +1 -1
- package/dist/prosemirror/schema/nodes.d.ts +81 -1
- package/dist/prosemirror/styles/resolvedStyleAttrs.js +2 -0
- package/dist/prosemirror/styles/styleResolver.d.ts +9 -0
- package/dist/prosemirror/styles/styleResolver.js +12 -0
- package/dist/style-engine/styleEngine.d.ts +3 -0
- package/dist/style-engine/styleEngine.js +3 -0
- package/dist/style-sets/extract.js +1 -23
- package/dist/style-sets/stellaStyle.js +46 -39
- package/dist/style-sets/styleSetNormalization.d.ts +19 -0
- package/dist/style-sets/styleSetNormalization.js +99 -0
- package/dist/types/content.d.ts +2 -2
- package/dist/utils/base64.d.ts +36 -0
- package/dist/utils/base64.js +40 -0
- package/dist/utils/clipboard.js +2 -1
- package/dist/utils/createDocument.js +145 -20
- package/dist/utils/headingCollector.d.ts +8 -5
- package/dist/utils/headingCollector.js +23 -25
- package/dist/utils/tableOfContentsStyle.js +9 -2
- package/dist/utils/units.d.ts +10 -1
- package/dist/utils/units.js +12 -1
- package/dist/utils/urlSecurity.d.ts +8 -2
- package/dist/utils/urlSecurity.js +21 -3
- package/package.json +2 -2
- package/dist/docx/textWhitespace.d.ts +0 -4
- package/dist/docx/textWhitespace.js +0 -4
- package/dist/layout-bridge/engine/tableWidthUtils.d.ts +0 -6
- package/dist/layout-bridge/engine/tableWidthUtils.js +0 -25
- package/dist/markdown/headings.d.ts +0 -13
- package/dist/markdown/headings.js +0 -20
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { sanitizeImageSrc } from "../utils/sanitizeImageSrc.js";
|
|
2
2
|
import { pixelsToEmu } from "../utils/units.js";
|
|
3
3
|
import { resolveImageData } from "./imageParser.js";
|
|
4
|
+
import { PREVIEW_KINDS } from "./previewBudget.js";
|
|
4
5
|
import { captureVerbatimXml } from "./verbatimCapture.js";
|
|
5
6
|
import { isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl } from "./vmlPreview.js";
|
|
6
7
|
import { isWatermarkShape } from "./watermarkParser.js";
|
|
@@ -73,8 +74,8 @@ const previewImage = (pictElement, svg, widthPx, heightPx, style, rootXmlns) =>
|
|
|
73
74
|
type: "image",
|
|
74
75
|
rId: "",
|
|
75
76
|
src,
|
|
76
|
-
mimeType:
|
|
77
|
-
filename:
|
|
77
|
+
mimeType: PREVIEW_KINDS.vmlShape.mimeType,
|
|
78
|
+
filename: PREVIEW_KINDS.vmlShape.filename,
|
|
78
79
|
size: {
|
|
79
80
|
width: pixelsToEmu(widthPx),
|
|
80
81
|
height: pixelsToEmu(heightPx)
|
|
@@ -135,7 +136,7 @@ function shouldPreserveRawVmlPict(pictElement) {
|
|
|
135
136
|
* `o:relid` instead, so fall back through those before the bare `id`.
|
|
136
137
|
*/
|
|
137
138
|
function readImageDataRId(imagedata) {
|
|
138
|
-
return getAttribute(imagedata, "r", "id") ?? getAttribute(imagedata, "r", "embed") ?? getAttribute(imagedata, "o", "relid") ?? getAttribute(imagedata, null, "id") ??
|
|
139
|
+
return getAttribute(imagedata, "r", "id") ?? getAttribute(imagedata, "r", "embed") ?? getAttribute(imagedata, "o", "relid") ?? getAttribute(imagedata, null, "id") ?? void 0;
|
|
139
140
|
}
|
|
140
141
|
/**
|
|
141
142
|
* Parse a `w:pict` element into an inline image, or null when it carries no
|
|
@@ -157,7 +158,7 @@ function parseVmlImageContent(pictElement, rels, media, rootXmlns = {}) {
|
|
|
157
158
|
const imagedata = findChild(shape, "v", "imagedata");
|
|
158
159
|
if (!imagedata) continue;
|
|
159
160
|
const rId = readImageDataRId(imagedata);
|
|
160
|
-
if (
|
|
161
|
+
if (rId === void 0 || rId.length === 0) continue;
|
|
161
162
|
if (isWatermarkShape(shape)) continue;
|
|
162
163
|
const { src, mimeType, filename } = resolveImageData(rId, rels ?? void 0, media ?? void 0);
|
|
163
164
|
const shapeStyle = parseVmlStyle(getAttribute(shape, null, "style"));
|
|
@@ -25,7 +25,5 @@ declare const vmlSvgDataUrl: (svg: string) => string | undefined;
|
|
|
25
25
|
declare const renderStandaloneVmlPreview: (shape: XmlElement) => VmlPreviewResult;
|
|
26
26
|
/** Render a VML group through bounded, non-clipping local-coordinate transforms. */
|
|
27
27
|
declare const renderVmlGroupPreview: (group: XmlElement) => VmlPreviewResult;
|
|
28
|
-
/** Bound retained synthetic VML previews while preserving their raw replay nodes. */
|
|
29
|
-
declare const enforcePackageVmlPreviewBudget: (root: unknown, maxCharacters?: number) => void;
|
|
30
28
|
//#endregion
|
|
31
|
-
export { VmlPreviewResult,
|
|
29
|
+
export { VmlPreviewResult, isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
|
package/dist/docx/vmlPreview.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { VML_PREVIEW_DATA_URL_PREFIX } from "./previewBudget.js";
|
|
1
2
|
import { findChild, findDeep, getAttribute, getChildElements, getLocalName } from "./xmlParser.js";
|
|
2
3
|
//#region src/docx/vmlPreview.ts
|
|
3
4
|
const MAX_VML_PREVIEW_DEPTH = 16;
|
|
@@ -6,10 +7,6 @@ const MAX_VML_PREVIEW_PATH_POINTS = 2e4;
|
|
|
6
7
|
const MAX_VML_PREVIEW_COORDINATE = 1e6;
|
|
7
8
|
const MAX_VML_PREVIEW_DIMENSION_PX = 2e4;
|
|
8
9
|
const MAX_VML_SVG_CHARACTERS = 1e6;
|
|
9
|
-
const MAX_PACKAGE_VML_PREVIEW_CHARACTERS = 8 * 1024 * 1024;
|
|
10
|
-
const VML_PREVIEW_DATA_URL_PREFIX = "data:image/svg+xml;charset=utf-8,";
|
|
11
|
-
const VML_PREVIEW_FILENAME = "vml-shape-preview.svg";
|
|
12
|
-
const VML_PREVIEW_MIME_TYPE = "image/svg+xml";
|
|
13
10
|
const SAFE_VML_COLORS = /* @__PURE__ */ new Set([
|
|
14
11
|
"black",
|
|
15
12
|
"white",
|
|
@@ -506,30 +503,5 @@ const renderVmlGroupPreview = (group) => {
|
|
|
506
503
|
style
|
|
507
504
|
};
|
|
508
505
|
};
|
|
509
|
-
/** Bound retained synthetic VML previews while preserving their raw replay nodes. */
|
|
510
|
-
const enforcePackageVmlPreviewBudget = (root, maxCharacters = MAX_PACKAGE_VML_PREVIEW_CHARACTERS) => {
|
|
511
|
-
let remainingCharacters = Math.max(0, maxCharacters);
|
|
512
|
-
const visited = /* @__PURE__ */ new WeakSet();
|
|
513
|
-
const visit = (value) => {
|
|
514
|
-
if (value === null || typeof value !== "object" || visited.has(value)) return;
|
|
515
|
-
visited.add(value);
|
|
516
|
-
if (value instanceof ArrayBuffer || ArrayBuffer.isView(value)) return;
|
|
517
|
-
if (value instanceof Map) {
|
|
518
|
-
for (const child of value.values()) visit(child);
|
|
519
|
-
return;
|
|
520
|
-
}
|
|
521
|
-
if (Array.isArray(value)) {
|
|
522
|
-
for (const child of value) visit(child);
|
|
523
|
-
return;
|
|
524
|
-
}
|
|
525
|
-
if ("type" in value && value.type === "image" && "rId" in value && value.rId === "" && "mimeType" in value && value.mimeType === VML_PREVIEW_MIME_TYPE && "filename" in value && value.filename === VML_PREVIEW_FILENAME && "src" in value && typeof value.src === "string" && value.src.startsWith(VML_PREVIEW_DATA_URL_PREFIX)) if (value.src.length <= remainingCharacters) remainingCharacters -= value.src.length;
|
|
526
|
-
else {
|
|
527
|
-
remainingCharacters = 0;
|
|
528
|
-
delete value.src;
|
|
529
|
-
}
|
|
530
|
-
for (const child of Object.values(value)) visit(child);
|
|
531
|
-
};
|
|
532
|
-
visit(root);
|
|
533
|
-
};
|
|
534
506
|
//#endregion
|
|
535
|
-
export {
|
|
507
|
+
export { isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { parseAnchorBehindDoc } from "./drawingUtils.js";
|
|
1
2
|
import { captureVerbatimXml } from "./verbatimCapture.js";
|
|
2
3
|
import { cloneWithXmlnsDeclarations, collectXmlnsDeclarations, findChild, findChildren, findDeep, getAttribute, getLocalName, getTextContent } from "./xmlParser.js";
|
|
3
4
|
//#region src/docx/watermarkParser.ts
|
|
@@ -258,8 +259,7 @@ function collectDrawingMlBehindContentAnchors(header) {
|
|
|
258
259
|
for (const drawing of collectByLocalName(header, "drawing")) {
|
|
259
260
|
const anchor = findDeep(drawing, "wp", "anchor");
|
|
260
261
|
if (!anchor) continue;
|
|
261
|
-
|
|
262
|
-
if (behindDoc === "1" || behindDoc === "true") out.push(anchor);
|
|
262
|
+
if (parseAnchorBehindDoc(anchor)) out.push(anchor);
|
|
263
263
|
}
|
|
264
264
|
return out;
|
|
265
265
|
}
|
package/dist/docx/xmlParser.d.ts
CHANGED
|
@@ -1,24 +1,5 @@
|
|
|
1
|
+
import { ParseContext } from "./parseContext.js";
|
|
1
2
|
//#region src/docx/xmlParser.d.ts
|
|
2
|
-
/**
|
|
3
|
-
* XML Parser Utilities for OOXML
|
|
4
|
-
*
|
|
5
|
-
* Provides helper functions for parsing Office Open XML (OOXML) content
|
|
6
|
-
* with proper namespace handling.
|
|
7
|
-
*
|
|
8
|
-
* OOXML uses many namespaces:
|
|
9
|
-
* - w: WordprocessingML (main document content)
|
|
10
|
-
* - a: DrawingML (graphics)
|
|
11
|
-
* - r: Relationships
|
|
12
|
-
* - wp: Word Drawing positioning
|
|
13
|
-
* - wps: Word Drawing shapes
|
|
14
|
-
* - wpc: Word Drawing canvas
|
|
15
|
-
* - wpg: Word Drawing group
|
|
16
|
-
* - m: Math
|
|
17
|
-
* - mc: Markup Compatibility
|
|
18
|
-
* - v: VML (legacy vector graphics)
|
|
19
|
-
* - o: Office (extensions)
|
|
20
|
-
* - pic: Pictures
|
|
21
|
-
*/
|
|
22
3
|
/**
|
|
23
4
|
* XML element tree node — drop-in replacement for the `Element` type
|
|
24
5
|
* previously imported from `xml-js`. Every consumer imports this from
|
|
@@ -90,7 +71,22 @@ declare const NAMESPACES: {
|
|
|
90
71
|
declare const OOXML_NAMESPACE_SCOPE: XmlNamespaceScope;
|
|
91
72
|
declare function parseXml(xml: string, inheritedNamespaceScope?: XmlNamespaceScope): XmlElement;
|
|
92
73
|
/**
|
|
93
|
-
* Serialize an XmlElement back to an XML string
|
|
74
|
+
* Serialize an XmlElement back to an XML string.
|
|
75
|
+
*
|
|
76
|
+
* Written here rather than handed to `fast-xml-parser`'s builder for two
|
|
77
|
+
* reasons. The builder writes a tab or a newline inside an attribute value
|
|
78
|
+
* literally, and XML 1.0 §3.3.3 has every conformant reader normalise those to
|
|
79
|
+
* a space: a `descr="two lines
"` a source file wrote came back as
|
|
80
|
+
* `descr="two lines "`. `escapeXmlAttribute` writes the character references
|
|
81
|
+
* that survive that step. And the builder returned a string per node, so a
|
|
82
|
+
* subtree's bytes were copied into its parent's answer, its grandparent's and
|
|
83
|
+
* so on, which `writeElement` below replaces with one buffer.
|
|
84
|
+
*
|
|
85
|
+
* Those two are the whole difference from the builder: the attribute
|
|
86
|
+
* character references, an attribute the model dropped written as absent
|
|
87
|
+
* rather than as the word "undefined", and the characters XML 1.0 §2.2 admits
|
|
88
|
+
* no spelling for dropped. Every other rule reproduces its output byte for
|
|
89
|
+
* byte, so replaying a capture is unchanged wherever it was already correct.
|
|
94
90
|
*/
|
|
95
91
|
declare function elementToXml(element: XmlElement): string;
|
|
96
92
|
/**
|
|
@@ -249,14 +245,18 @@ declare function getAttributes(element: XmlElement | null | undefined): Record<s
|
|
|
249
245
|
*/
|
|
250
246
|
declare function getTextContent(element: XmlElement | null | undefined): string;
|
|
251
247
|
/**
|
|
252
|
-
*
|
|
248
|
+
* Read an `ST_OnOff` attribute.
|
|
253
249
|
*
|
|
254
|
-
*
|
|
255
|
-
*
|
|
256
|
-
*
|
|
257
|
-
*
|
|
250
|
+
* The type has three spellings per polarity — `1`/`true`/`on` and
|
|
251
|
+
* `0`/`false`/`off` — and producers use all of them. `undefined` means the
|
|
252
|
+
* author said nothing (absent, or a value outside the type), so the caller
|
|
253
|
+
* still owns what absence means for its own slot.
|
|
254
|
+
*
|
|
255
|
+
* Every on/off attribute goes through here: a hand-rolled `=== "1"` reads
|
|
256
|
+
* `w:beforeAutospacing="on"` as false, and the save path then writes `"0"`,
|
|
257
|
+
* inverting what the document said.
|
|
258
258
|
*/
|
|
259
|
-
declare function
|
|
259
|
+
declare function parseOnOffAttribute(element: XmlElement | null | undefined, namespace: string | null, name: string, context?: ParseContext): boolean | undefined;
|
|
260
260
|
/**
|
|
261
261
|
* Check if a child element exists (used for boolean flags in OOXML)
|
|
262
262
|
*
|
|
@@ -297,15 +297,20 @@ declare function parseNumericAttribute(element: XmlElement | null | undefined, n
|
|
|
297
297
|
*/
|
|
298
298
|
declare function parseNumberingLevelAttribute(element: XmlElement | null | undefined): number | undefined;
|
|
299
299
|
/**
|
|
300
|
-
* Parse `w:w` on a table width/height element.
|
|
301
|
-
*
|
|
302
|
-
* (`5000`);
|
|
300
|
+
* Parse `w:w` on a table width/height element. A percentage may be spelled the
|
|
301
|
+
* way it reads (`100%`) or as the number its slot counts in
|
|
302
|
+
* (50ths-of-percent, `5000`); the generated slot table says which unit that
|
|
303
|
+
* is, so the two spellings are not decided here.
|
|
303
304
|
*/
|
|
304
305
|
declare function parseTableMeasurementValue(element: XmlElement | null | undefined, widthType: string): number | undefined;
|
|
305
306
|
/**
|
|
306
307
|
* Parse an OOXML `ST_OnOff` lexical value.
|
|
308
|
+
*
|
|
309
|
+
* `context` is optional and trailing on purpose: every call site that has one
|
|
310
|
+
* reports the value folio declined to read, and the hundred that do not yet
|
|
311
|
+
* thread one keep compiling and keep their behaviour.
|
|
307
312
|
*/
|
|
308
|
-
declare function parseOnOffValue(value: string | null | undefined): boolean | undefined;
|
|
313
|
+
declare function parseOnOffValue(value: string | null | undefined, context?: ParseContext, element?: string): boolean | undefined;
|
|
309
314
|
/**
|
|
310
315
|
* Parse a boolean value from an attribute or element presence
|
|
311
316
|
*
|
|
@@ -318,7 +323,7 @@ declare function parseOnOffValue(value: string | null | undefined): boolean | un
|
|
|
318
323
|
* @param namespace - Namespace for val attribute
|
|
319
324
|
* @returns boolean value
|
|
320
325
|
*/
|
|
321
|
-
declare function parseBooleanElement(element: XmlElement | null | undefined, namespace?: string): boolean;
|
|
326
|
+
declare function parseBooleanElement(element: XmlElement | null | undefined, namespace?: string, context?: ParseContext): boolean;
|
|
322
327
|
/**
|
|
323
328
|
* Deep find - search recursively for an element
|
|
324
329
|
*
|
|
@@ -378,4 +383,4 @@ declare function cloneWithXmlnsDeclarations(element: XmlElement, xmlnsDecls: Rec
|
|
|
378
383
|
*/
|
|
379
384
|
declare function cloneElement(element: XmlElement, overrides: Partial<XmlElement>): XmlElement;
|
|
380
385
|
//#endregion
|
|
381
|
-
export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, XmlAttributeMatch, XmlElement, XmlNamespaceScope, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild,
|
|
386
|
+
export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, XmlAttributeMatch, XmlElement, XmlNamespaceScope, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, selectAlternateContentBranch };
|
package/dist/docx/xmlParser.js
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
|
-
import { transitionalSlotEncoding } from "./transitionalSpelling.js";
|
|
1
|
+
import { NUMBERS_PER_PERCENT, percentageSpelling, transitionalSlotEncoding } from "./transitionalSpelling.js";
|
|
2
2
|
import { universalMeasureAs } from "./universalMeasure.js";
|
|
3
|
-
import {
|
|
3
|
+
import { escapeXmlAttribute, escapeXmlText } from "@stll/docx-core";
|
|
4
|
+
import { XMLParser } from "fast-xml-parser";
|
|
4
5
|
import { OOXML_NS } from "@stll/docx-utils";
|
|
6
|
+
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
5
7
|
//#region src/docx/xmlParser.ts
|
|
6
8
|
/**
|
|
7
9
|
* XML Parser Utilities for OOXML
|
|
@@ -41,22 +43,12 @@ const fxpParserOptionsWithStopNodes = {
|
|
|
41
43
|
...fxpParserOptions,
|
|
42
44
|
stopNodes: ["*.w:binData"]
|
|
43
45
|
};
|
|
44
|
-
const fxpBuilderOptions = {
|
|
45
|
-
preserveOrder: true,
|
|
46
|
-
ignoreAttributes: false,
|
|
47
|
-
attributeNamePrefix: "",
|
|
48
|
-
textNodeName: "#text",
|
|
49
|
-
suppressEmptyNode: true
|
|
50
|
-
};
|
|
51
46
|
const fxpParser = new XMLParser(fxpParserOptions);
|
|
52
47
|
const fxpParserWithStopNodes = new XMLParser(fxpParserOptionsWithStopNodes);
|
|
53
|
-
const fxpBuilder = new XMLBuilder(fxpBuilderOptions);
|
|
54
48
|
/** Text node key used by fast-xml-parser in preserveOrder mode. */
|
|
55
49
|
const TEXT_KEY = "#text";
|
|
56
50
|
/** Attribute group key used by fast-xml-parser in preserveOrder mode. */
|
|
57
51
|
const ATTR_KEY = ":@";
|
|
58
|
-
/** Character reference required to keep carriage returns through XML end-of-line normalization. */
|
|
59
|
-
const XML_CARRIAGE_RETURN_REFERENCE = " ";
|
|
60
52
|
const EMPTY_NAMESPACE_SCOPE = { bindings: /* @__PURE__ */ new Map() };
|
|
61
53
|
const resolveNamespaceUri = (scope, prefix) => {
|
|
62
54
|
let current = scope;
|
|
@@ -137,18 +129,6 @@ function fxpToRootElement(nodes, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE
|
|
|
137
129
|
return { elements: nodes };
|
|
138
130
|
}
|
|
139
131
|
/**
|
|
140
|
-
* Convert an XmlElement back into the fast-xml-parser preserveOrder format
|
|
141
|
-
* so we can feed it to XMLBuilder.
|
|
142
|
-
*/
|
|
143
|
-
function elementToFxpNode(el) {
|
|
144
|
-
if (el.type === "text") return { [TEXT_KEY]: el.text ?? "" };
|
|
145
|
-
const name = el.name ?? "";
|
|
146
|
-
const children = el.elements ? el.elements.map(elementToFxpNode) : [];
|
|
147
|
-
const node = { [name]: children };
|
|
148
|
-
if (el.attributes && Object.keys(el.attributes).length > 0) node[ATTR_KEY] = el.attributes;
|
|
149
|
-
return node;
|
|
150
|
-
}
|
|
151
|
-
/**
|
|
152
132
|
* Common OOXML namespace URIs — re-exported from @stll/docx-utils.
|
|
153
133
|
*/
|
|
154
134
|
const NAMESPACES = OOXML_NS;
|
|
@@ -173,13 +153,64 @@ function parseXml(xml, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE) {
|
|
|
173
153
|
return fxpToRootElement((xml.includes("binData") ? fxpParserWithStopNodes : fxpParser).parse(xml), inheritedNamespaceScope);
|
|
174
154
|
}
|
|
175
155
|
/**
|
|
176
|
-
* Serialize an XmlElement back to an XML string
|
|
156
|
+
* Serialize an XmlElement back to an XML string.
|
|
157
|
+
*
|
|
158
|
+
* Written here rather than handed to `fast-xml-parser`'s builder for two
|
|
159
|
+
* reasons. The builder writes a tab or a newline inside an attribute value
|
|
160
|
+
* literally, and XML 1.0 §3.3.3 has every conformant reader normalise those to
|
|
161
|
+
* a space: a `descr="two lines
"` a source file wrote came back as
|
|
162
|
+
* `descr="two lines "`. `escapeXmlAttribute` writes the character references
|
|
163
|
+
* that survive that step. And the builder returned a string per node, so a
|
|
164
|
+
* subtree's bytes were copied into its parent's answer, its grandparent's and
|
|
165
|
+
* so on, which `writeElement` below replaces with one buffer.
|
|
166
|
+
*
|
|
167
|
+
* Those two are the whole difference from the builder: the attribute
|
|
168
|
+
* character references, an attribute the model dropped written as absent
|
|
169
|
+
* rather than as the word "undefined", and the characters XML 1.0 §2.2 admits
|
|
170
|
+
* no spelling for dropped. Every other rule reproduces its output byte for
|
|
171
|
+
* byte, so replaying a capture is unchanged wherever it was already correct.
|
|
177
172
|
*/
|
|
178
173
|
function elementToXml(element) {
|
|
179
|
-
const
|
|
180
|
-
|
|
174
|
+
const out = [];
|
|
175
|
+
writeElement(element, out);
|
|
176
|
+
return out.join("");
|
|
181
177
|
}
|
|
182
178
|
/**
|
|
179
|
+
* Serialize into one shared buffer rather than a string per node.
|
|
180
|
+
*
|
|
181
|
+
* Returning a string per element makes a subtree's bytes a substring of its
|
|
182
|
+
* parent's, its grandparent's and so on, so a table pays for its rows, its
|
|
183
|
+
* rows pay for their cells, and a part that nests four levels deep is copied
|
|
184
|
+
* four times before anything is written. Appending into one array of chunks
|
|
185
|
+
* and joining once costs each node its own text and nothing for its ancestors.
|
|
186
|
+
*
|
|
187
|
+
* Whether an element self-closes is not known until its children are written,
|
|
188
|
+
* so the opening tag reserves a slot in the buffer and fills it afterwards:
|
|
189
|
+
* `>` when something was appended, `/>` when nothing was.
|
|
190
|
+
*/
|
|
191
|
+
const writeElement = (element, out) => {
|
|
192
|
+
if (element.type === "text") {
|
|
193
|
+
const text = String(element.text ?? "");
|
|
194
|
+
if (text !== "") out.push(escapeXmlText(text));
|
|
195
|
+
return;
|
|
196
|
+
}
|
|
197
|
+
const name = element.name ?? "";
|
|
198
|
+
out.push(`<${name}`);
|
|
199
|
+
if (element.attributes) for (const [attribute, value] of Object.entries(element.attributes)) {
|
|
200
|
+
if (value === void 0) continue;
|
|
201
|
+
out.push(` ${attribute}="${escapeXmlAttribute(String(value))}"`);
|
|
202
|
+
}
|
|
203
|
+
const openingSlot = out.push("") - 1;
|
|
204
|
+
const contentStart = out.length;
|
|
205
|
+
for (const child of element.elements ?? []) writeElement(child, out);
|
|
206
|
+
if (out.length === contentStart) {
|
|
207
|
+
out[openingSlot] = "/>";
|
|
208
|
+
return;
|
|
209
|
+
}
|
|
210
|
+
out[openingSlot] = ">";
|
|
211
|
+
out.push(`</${name}>`);
|
|
212
|
+
};
|
|
213
|
+
/**
|
|
183
214
|
* Parse XML string to a more convenient format
|
|
184
215
|
*/
|
|
185
216
|
function parseXmlDocument(xml) {
|
|
@@ -481,17 +512,19 @@ function getTextContent(element) {
|
|
|
481
512
|
return text;
|
|
482
513
|
}
|
|
483
514
|
/**
|
|
484
|
-
*
|
|
515
|
+
* Read an `ST_OnOff` attribute.
|
|
485
516
|
*
|
|
486
|
-
*
|
|
487
|
-
*
|
|
488
|
-
*
|
|
489
|
-
*
|
|
517
|
+
* The type has three spellings per polarity — `1`/`true`/`on` and
|
|
518
|
+
* `0`/`false`/`off` — and producers use all of them. `undefined` means the
|
|
519
|
+
* author said nothing (absent, or a value outside the type), so the caller
|
|
520
|
+
* still owns what absence means for its own slot.
|
|
521
|
+
*
|
|
522
|
+
* Every on/off attribute goes through here: a hand-rolled `=== "1"` reads
|
|
523
|
+
* `w:beforeAutospacing="on"` as false, and the save path then writes `"0"`,
|
|
524
|
+
* inverting what the document said.
|
|
490
525
|
*/
|
|
491
|
-
function
|
|
492
|
-
|
|
493
|
-
if (value === null) return false;
|
|
494
|
-
return parseOnOffValue(value) ?? true;
|
|
526
|
+
function parseOnOffAttribute(element, namespace, name, context) {
|
|
527
|
+
return parseOnOffValue(getAttribute(element, namespace, name), context, element?.name ?? name);
|
|
495
528
|
}
|
|
496
529
|
/**
|
|
497
530
|
* Check if a child element exists (used for boolean flags in OOXML)
|
|
@@ -554,17 +587,19 @@ function parseNumberingLevelAttribute(element) {
|
|
|
554
587
|
return level !== void 0 && level >= 0 ? level : void 0;
|
|
555
588
|
}
|
|
556
589
|
/**
|
|
557
|
-
* Parse `w:w` on a table width/height element.
|
|
558
|
-
*
|
|
559
|
-
* (`5000`);
|
|
590
|
+
* Parse `w:w` on a table width/height element. A percentage may be spelled the
|
|
591
|
+
* way it reads (`100%`) or as the number its slot counts in
|
|
592
|
+
* (50ths-of-percent, `5000`); the generated slot table says which unit that
|
|
593
|
+
* is, so the two spellings are not decided here.
|
|
560
594
|
*/
|
|
561
595
|
function parseTableMeasurementValue(element, widthType) {
|
|
562
596
|
const raw = getAttribute(element, "w", "w");
|
|
563
597
|
if (raw === null) return;
|
|
564
598
|
const trimmed = raw.trim();
|
|
565
|
-
if (widthType === "pct"
|
|
566
|
-
const
|
|
567
|
-
|
|
599
|
+
if (widthType === "pct") {
|
|
600
|
+
const percent = percentageSpelling(trimmed);
|
|
601
|
+
const unit = element === null || element === void 0 ? void 0 : transitionalSlotEncoding(element.namespaceUri, getLocalName(element.name), "w")?.percent;
|
|
602
|
+
if (percent !== void 0 && unit !== void 0) return Math.round(percent * NUMBERS_PER_PERCENT[unit]);
|
|
568
603
|
}
|
|
569
604
|
const measure = universalMeasureAs(trimmed, "twips");
|
|
570
605
|
if (measure !== void 0) return measure;
|
|
@@ -573,8 +608,12 @@ function parseTableMeasurementValue(element, widthType) {
|
|
|
573
608
|
}
|
|
574
609
|
/**
|
|
575
610
|
* Parse an OOXML `ST_OnOff` lexical value.
|
|
611
|
+
*
|
|
612
|
+
* `context` is optional and trailing on purpose: every call site that has one
|
|
613
|
+
* reports the value folio declined to read, and the hundred that do not yet
|
|
614
|
+
* thread one keep compiling and keep their behaviour.
|
|
576
615
|
*/
|
|
577
|
-
function parseOnOffValue(value) {
|
|
616
|
+
function parseOnOffValue(value, context, element) {
|
|
578
617
|
if (value === null || value === void 0) return;
|
|
579
618
|
switch (value) {
|
|
580
619
|
case "1":
|
|
@@ -583,7 +622,13 @@ function parseOnOffValue(value) {
|
|
|
583
622
|
case "0":
|
|
584
623
|
case "false":
|
|
585
624
|
case "off": return false;
|
|
586
|
-
default:
|
|
625
|
+
default:
|
|
626
|
+
context?.warn({
|
|
627
|
+
code: PARSE_WARNING_CODES.unrecognisedOnOffValue,
|
|
628
|
+
value,
|
|
629
|
+
...element === void 0 ? {} : { element }
|
|
630
|
+
});
|
|
631
|
+
return;
|
|
587
632
|
}
|
|
588
633
|
}
|
|
589
634
|
/**
|
|
@@ -598,7 +643,7 @@ function parseOnOffValue(value) {
|
|
|
598
643
|
* @param namespace - Namespace for val attribute
|
|
599
644
|
* @returns boolean value
|
|
600
645
|
*/
|
|
601
|
-
function parseBooleanElement(element, namespace = "w") {
|
|
646
|
+
function parseBooleanElement(element, namespace = "w", context) {
|
|
602
647
|
if (!element) return false;
|
|
603
648
|
let val = null;
|
|
604
649
|
const elementName = element.name ?? "";
|
|
@@ -613,7 +658,7 @@ function parseBooleanElement(element, namespace = "w") {
|
|
|
613
658
|
}
|
|
614
659
|
}
|
|
615
660
|
if (val === null) return true;
|
|
616
|
-
return parseOnOffValue(val) ??
|
|
661
|
+
return parseOnOffValue(val, context, elementName) ?? false;
|
|
617
662
|
}
|
|
618
663
|
/**
|
|
619
664
|
* Deep find - search recursively for an element
|
|
@@ -810,4 +855,4 @@ function cloneElement(element, overrides) {
|
|
|
810
855
|
return clone;
|
|
811
856
|
}
|
|
812
857
|
//#endregion
|
|
813
|
-
export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild,
|
|
858
|
+
export { NAMESPACES, OFFICE_RELATIONSHIP_NAMESPACE_URIS, OOXML_NAMESPACE_SCOPE, WORDPROCESSINGML_NAMESPACE_URIS, attachXmlNamespaceContext, cloneElement, cloneWithXmlnsDeclarations, collectXmlnsDeclarations, elementToXml, findAllDeep, findAttributeByNamespaceUri, findByFullName, findChild, findChildByLocalName, findChildByNamespaceUri, findChildren, findChildrenByLocalName, findChildrenByNamespaceUri, findDeep, getAttribute, getAttributeAny, getAttributeAnyPrefix, getAttributeByNamespaceUri, getAttributes, getChildElements, getLocalName, getNamespacePrefix, getNamespaceUri, getTextContent, hasChild, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseColorElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue, parseTableMeasurementValue, parseXml, parseXmlDocument, selectAlternateContentBranch };
|
|
@@ -1,27 +1,107 @@
|
|
|
1
1
|
//#region src/docx/xmlResourceLimits.d.ts
|
|
2
|
-
/**
|
|
2
|
+
/**
|
|
3
|
+
* Shared bounds for XML parts parsed by Folio.
|
|
4
|
+
*
|
|
5
|
+
* `maxBytes` and the package expansion ceiling bound the *markup*; they do not
|
|
6
|
+
* bound what parsing that markup allocates. A parsed element retains far more
|
|
7
|
+
* than the bytes it was written as, so a part that satisfies a byte bound can
|
|
8
|
+
* still cost multiples of it in tree. Measured on this repository's corpus
|
|
9
|
+
* generators (Bun 1.4, arm64, heap retained after a forced GC):
|
|
10
|
+
*
|
|
11
|
+
* shape bytes/element tree heap / part bytes
|
|
12
|
+
* `<w:r/>` 49.5 B 8.3x
|
|
13
|
+
* `<w:r a=".." x4/>` 105-124 B 2.8-3.3x
|
|
14
|
+
* `<w:r><w:t>x</w:t></w:r>` 162-171 B 14.1-14.8x
|
|
15
|
+
*
|
|
16
|
+
* An element is therefore the unit a memory bound has to count, because the
|
|
17
|
+
* adversarial shape is the cheap one: `<w:r/>` is 7 bytes of markup and 49.5
|
|
18
|
+
* bytes of tree, so 128 MiB of markup buys ~19M elements and ~950 MB of tree
|
|
19
|
+
* inside a byte budget that never trips.
|
|
20
|
+
*
|
|
21
|
+
* Corpus distribution (5,314 readable packages, every XML and .rels part):
|
|
22
|
+
*
|
|
23
|
+
* metric p50 p90 p99 p99.9 max
|
|
24
|
+
* elements / part 16 227 1,590 24,719 596,668
|
|
25
|
+
* elements / package 917 3,230 29,106 71,975 602,212
|
|
26
|
+
* attributes / part 27 577 1,993 21,037 630,374
|
|
27
|
+
* attributes / package 1,783 4,711 25,896 91,946 639,110
|
|
28
|
+
* depth / part 3 9 15 22 15,005
|
|
29
|
+
* elements per byte 0.012 0.027 0.036 0.043 0.111
|
|
30
|
+
*
|
|
31
|
+
* The defaults below reject no corpus package that today's bounds accept. The
|
|
32
|
+
* package budget is the one that matters: it caps a whole package at ~2.5M
|
|
33
|
+
* elements and ~3M attributes, which is at most ~425 MB of tree at the densest
|
|
34
|
+
* measured shape and ~124 MB at the cheapest. Before it existed, only
|
|
35
|
+
* `word/document.xml`, `word/styles.xml` and `word/numbering.xml` were counted
|
|
36
|
+
* at all, so a package of many merely-large parts could reach the expansion
|
|
37
|
+
* ceiling of 250 MiB, ~37M elements and well past 1.8 GB of tree while passing
|
|
38
|
+
* every bound. Lower these to trade format reach for a smaller ceiling.
|
|
39
|
+
*/
|
|
3
40
|
declare const FOLIO_XML_RESOURCE_LIMITS: {
|
|
4
41
|
readonly maxBytes: number;
|
|
5
42
|
readonly maxDepth: 100;
|
|
6
|
-
|
|
43
|
+
/** 1.68x the corpus maximum (596,668). Unchanged; a shipped bound is not loosened. */
|
|
44
|
+
readonly maxElementsPerPart: 1000000;
|
|
45
|
+
/** 3.97x the corpus maximum (630,374). */
|
|
46
|
+
readonly maxAttributesPerPart: 2500000;
|
|
47
|
+
/** 4.15x the corpus maximum (602,212). */
|
|
48
|
+
readonly maxElementsPerPackage: 2500000;
|
|
49
|
+
/** 4.69x the corpus maximum (639,110). */
|
|
50
|
+
readonly maxAttributesPerPackage: 3000000;
|
|
7
51
|
};
|
|
8
52
|
type XmlResourceLimits = {
|
|
9
53
|
maxBytes: number;
|
|
10
54
|
maxDepth: number;
|
|
11
|
-
|
|
55
|
+
maxElementsPerPart: number;
|
|
56
|
+
maxAttributesPerPart: number;
|
|
57
|
+
maxElementsPerPackage: number;
|
|
58
|
+
maxAttributesPerPackage: number;
|
|
12
59
|
};
|
|
13
|
-
type XmlResourceLimitKind = "bytes" | "depth" | "
|
|
60
|
+
type XmlResourceLimitKind = "bytes" | "depth" | "elements" | "attributes" | "package-elements" | "package-attributes" | "syntax";
|
|
14
61
|
declare const XmlResourceLimitError_base: import("better-result").TaggedErrorClass<"XmlResourceLimitError">;
|
|
15
62
|
/** XML input exceeded a parser resource bound or could not be scanned safely. */
|
|
16
63
|
declare class XmlResourceLimitError extends XmlResourceLimitError_base<{
|
|
17
64
|
message: string;
|
|
18
65
|
limit: XmlResourceLimitKind;
|
|
66
|
+
/** The package part being scanned, when the caller named one. */
|
|
67
|
+
partPath?: string;
|
|
68
|
+
/** The count reached at the point of refusal, not the count of the whole input. */
|
|
69
|
+
observed: number;
|
|
70
|
+
allowed: number;
|
|
19
71
|
}> {}
|
|
20
72
|
/**
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
73
|
+
* What a package has spent so far, shared by every part in one package.
|
|
74
|
+
*
|
|
75
|
+
* Per-part bounds alone do not bound a package: a package may hold hundreds of
|
|
76
|
+
* parts, each individually modest. The budget is the accumulator the readers
|
|
77
|
+
* carry across parts so the ceiling is the package's, not each part's.
|
|
78
|
+
*/
|
|
79
|
+
type XmlPackageBudget = {
|
|
80
|
+
elements: number;
|
|
81
|
+
attributes: number;
|
|
82
|
+
};
|
|
83
|
+
declare const createXmlPackageBudget: () => XmlPackageBudget;
|
|
84
|
+
type XmlResourceScanOptions = {
|
|
85
|
+
xml: string;
|
|
86
|
+
limits?: XmlResourceLimits;
|
|
87
|
+
/** The package path scanned, carried on any refusal so a host can name it. */
|
|
88
|
+
partPath?: string;
|
|
89
|
+
/** Charged as the scan proceeds; omit for a part with no package context. */
|
|
90
|
+
budget?: XmlPackageBudget;
|
|
91
|
+
};
|
|
92
|
+
/** What the preflight counted, for callers that reconcile it against the tree. */
|
|
93
|
+
type XmlResourceScanResult = {
|
|
94
|
+
elements: number;
|
|
95
|
+
attributes: number;
|
|
96
|
+
maxDepth: number;
|
|
97
|
+
};
|
|
98
|
+
/**
|
|
99
|
+
* Bound XML bytes, element count, attribute count and nesting before building
|
|
100
|
+
* an object tree. The lexical scan is iterative, so deeply nested input cannot
|
|
101
|
+
* consume the JS call stack before the depth limit is enforced, and every
|
|
102
|
+
* bound is checked at the point it is crossed, so the work done before a
|
|
103
|
+
* refusal is proportional to the limit rather than to the input.
|
|
24
104
|
*/
|
|
25
|
-
declare const assertXmlResourceLimits: (xml
|
|
105
|
+
declare const assertXmlResourceLimits: ({ xml, limits, partPath, budget }: XmlResourceScanOptions) => XmlResourceScanResult;
|
|
26
106
|
//#endregion
|
|
27
|
-
export { FOLIO_XML_RESOURCE_LIMITS, XmlResourceLimitError, assertXmlResourceLimits };
|
|
107
|
+
export { FOLIO_XML_RESOURCE_LIMITS, XmlPackageBudget, XmlResourceLimitError, XmlResourceLimits, XmlResourceScanOptions, XmlResourceScanResult, assertXmlResourceLimits, createXmlPackageBudget };
|