@stll/folio-core 0.44.0 → 0.45.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/dist/ai-edits/__fixtures__/paragraphs.js +2 -2
  2. package/dist/ai-edits/headless.js +1 -0
  3. package/dist/compare/content-alignment.js +78 -53
  4. package/dist/compare/inline-atoms.js +34 -20
  5. package/dist/content-controls/mutateContentControls.js +4 -2
  6. package/dist/display-list/dom/renderDisplayListToDom.js +8 -8
  7. package/dist/document-operations.js +14 -3
  8. package/dist/docx/borderParser.js +5 -5
  9. package/dist/docx/commentParser.js +47 -36
  10. package/dist/docx/commentThreadKey.d.ts +18 -0
  11. package/dist/docx/commentThreadKey.js +22 -0
  12. package/dist/docx/diagramPreview.js +87 -27
  13. package/dist/docx/groupDrawingParser.js +3 -3
  14. package/dist/docx/hyperlinkParser.js +2 -2
  15. package/dist/docx/imageParser.d.ts +9 -1
  16. package/dist/docx/imageParser.js +58 -12
  17. package/dist/docx/imageRawXml.d.ts +14 -1
  18. package/dist/docx/imageRawXml.js +30 -6
  19. package/dist/docx/mathToMathml.js +12 -14
  20. package/dist/docx/nonVisualDrawingProps.d.ts +34 -0
  21. package/dist/docx/nonVisualDrawingProps.js +46 -0
  22. package/dist/docx/paragraphTextBoxEnrichment.js +3 -0
  23. package/dist/docx/parser.js +2 -2
  24. package/dist/docx/previewBudget.d.ts +64 -0
  25. package/dist/docx/previewBudget.js +88 -0
  26. package/dist/docx/revisionIdNormalization.js +17 -5
  27. package/dist/docx/rezip.js +17 -18
  28. package/dist/docx/sdtPropertiesPatch.js +24 -18
  29. package/dist/docx/sectionReferenceHistory.js +2 -2
  30. package/dist/docx/selectiveSave.js +6 -6
  31. package/dist/docx/serializer/blockSdtSerializer.js +38 -26
  32. package/dist/docx/serializer/borderSerializer.d.ts +1 -1
  33. package/dist/docx/serializer/borderSerializer.js +13 -12
  34. package/dist/docx/serializer/commentSerializer.d.ts +41 -16
  35. package/dist/docx/serializer/commentSerializer.js +82 -85
  36. package/dist/docx/serializer/fontTableSerializer.js +6 -6
  37. package/dist/docx/serializer/headerFooterSerializer.js +5 -5
  38. package/dist/docx/serializer/markupRangeAttributes.js +2 -2
  39. package/dist/docx/serializer/numberingSerializer.js +7 -6
  40. package/dist/docx/serializer/paragraphSerializer.js +19 -18
  41. package/dist/docx/serializer/partNamespaces.js +2 -2
  42. package/dist/docx/serializer/runSerializer.js +48 -28
  43. package/dist/docx/serializer/sectionPropertiesSerializer.js +11 -10
  44. package/dist/docx/serializer/settingsSerializer.js +4 -3
  45. package/dist/docx/serializer/stylesSerializer.js +6 -6
  46. package/dist/docx/serializer/tableSerializer.js +10 -9
  47. package/dist/docx/serializer/textFormattingSerializer.js +29 -28
  48. package/dist/docx/serializer/themeSerializer.js +6 -6
  49. package/dist/docx/serializer/trackedChangeAttributes.js +2 -2
  50. package/dist/docx/serializer/xmlUtils.d.ts +1 -2
  51. package/dist/docx/serializer/xmlUtils.js +1 -13
  52. package/dist/docx/server/boundedArchive.d.ts +12 -0
  53. package/dist/docx/server/boundedArchive.js +20 -1
  54. package/dist/docx/server/validateDocxConformance.js +22 -1
  55. package/dist/docx/shapeParser.js +7 -5
  56. package/dist/docx/textBoxParser.js +7 -2
  57. package/dist/docx/unzip.d.ts +23 -0
  58. package/dist/docx/unzip.js +32 -22
  59. package/dist/docx/verbatimCapture.js +1 -1
  60. package/dist/docx/vmlImageParser.js +3 -2
  61. package/dist/docx/vmlPreview.d.ts +1 -3
  62. package/dist/docx/vmlPreview.js +2 -30
  63. package/dist/docx/xmlParser.d.ts +16 -1
  64. package/dist/docx/xmlParser.js +56 -26
  65. package/dist/docx/xmlResourceLimits.d.ts +89 -9
  66. package/dist/docx/xmlResourceLimits.js +105 -24
  67. package/dist/internal/paragraphFormattingSerialization.js +3 -2
  68. package/dist/layout-painter/renderImage.js +4 -3
  69. package/dist/layout-painter/renderParagraph.js +4 -3
  70. package/dist/managers/autoSaveCodec.js +2 -8
  71. package/dist/markdown/images.js +1 -4
  72. package/dist/prosemirror/attrs/index.js +69 -0
  73. package/dist/prosemirror/commands/image.js +1 -0
  74. package/dist/prosemirror/conversion/fromProseDoc.js +69 -29
  75. package/dist/prosemirror/conversion/toProseDoc.js +83 -19
  76. package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +2 -3
  77. package/dist/prosemirror/extensions/nodes/ImageExtension.js +4 -0
  78. package/dist/prosemirror/extensions/nodes/ShapeExtension.js +7 -2
  79. package/dist/prosemirror/extensions/nodes/TextBoxExtension.js +8 -4
  80. package/dist/prosemirror/paragraphFormattingProvenance.d.ts +6 -3
  81. package/dist/prosemirror/paragraphFormattingProvenance.js +12 -3
  82. package/dist/prosemirror/schema/nodes.d.ts +50 -1
  83. package/dist/utils/base64.d.ts +36 -0
  84. package/dist/utils/base64.js +40 -0
  85. package/dist/utils/clipboard.js +2 -1
  86. package/dist/utils/units.d.ts +10 -1
  87. package/dist/utils/units.js +12 -1
  88. package/dist/utils/urlSecurity.d.ts +8 -2
  89. package/dist/utils/urlSecurity.js +21 -3
  90. package/package.json +2 -2
@@ -1,8 +1,8 @@
1
1
  import { serializePartElement } from "./partNamespaces.js";
2
- import { escapeXml } from "./xmlUtils.js";
2
+ import { escapeXmlAttribute } from "@stll/docx-core";
3
3
  //#region src/docx/serializer/themeSerializer.ts
4
4
  const serializeThemeXml = (theme) => {
5
- const name = escapeXml(theme.name ?? "Folio Theme");
5
+ const name = escapeXmlAttribute(theme.name ?? "Folio Theme");
6
6
  const elements = `${serializeColorScheme(theme.colorScheme)}${serializeFontScheme(theme)}${serializeFormatScheme(theme)}`;
7
7
  return "<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"yes\"?>\n" + serializePartElement({
8
8
  partPath: "word/theme/theme1.xml",
@@ -28,17 +28,17 @@ const serializeColorScheme = (colors) => {
28
28
  hlink: colors?.hlink ?? "0563C1",
29
29
  folHlink: colors?.folHlink ?? "954F72"
30
30
  };
31
- return `<a:clrScheme name="Folio">${Object.entries(values).map(([slot, value]) => `<a:${slot}><a:srgbClr val="${escapeXml(value)}"/></a:${slot}>`).join("")}</a:clrScheme>`;
31
+ return `<a:clrScheme name="Folio">${Object.entries(values).map(([slot, value]) => `<a:${slot}><a:srgbClr val="${escapeXmlAttribute(value)}"/></a:${slot}>`).join("")}</a:clrScheme>`;
32
32
  };
33
33
  const serializeFontScheme = (theme) => {
34
34
  return `<a:fontScheme name="Folio">${serializeThemeFont(theme.fontScheme?.majorFont, "majorFont", "Arial")}${serializeThemeFont(theme.fontScheme?.minorFont, "minorFont", "Arial")}</a:fontScheme>`;
35
35
  };
36
36
  const serializeThemeFont = (font, element, fallback) => {
37
- const scriptFonts = Object.entries(font?.fonts ?? {}).map(([script, typeface]) => `<a:font script="${escapeXml(script)}" typeface="${escapeXml(typeface)}"/>`).join("");
38
- return `<a:${element}><a:latin typeface="${escapeXml(font?.latin ?? fallback)}"/><a:ea typeface="${escapeXml(font?.ea ?? "")}"/><a:cs typeface="${escapeXml(font?.cs ?? "")}"/>${scriptFonts}</a:${element}>`;
37
+ const scriptFonts = Object.entries(font?.fonts ?? {}).map(([script, typeface]) => `<a:font script="${escapeXmlAttribute(script)}" typeface="${escapeXmlAttribute(typeface)}"/>`).join("");
38
+ return `<a:${element}><a:latin typeface="${escapeXmlAttribute(font?.latin ?? fallback)}"/><a:ea typeface="${escapeXmlAttribute(font?.ea ?? "")}"/><a:cs typeface="${escapeXmlAttribute(font?.cs ?? "")}"/>${scriptFonts}</a:${element}>`;
39
39
  };
40
40
  const serializeFormatScheme = (theme) => {
41
- return `<a:fmtScheme name="${escapeXml(theme.formatScheme?.name ?? "Folio")}"><a:fillStyleLst><a:solidFill><a:schemeClr val="phClr"/></a:solidFill></a:fillStyleLst><a:lnStyleLst><a:ln w="6350" cap="flat" cmpd="sng" algn="ctr"><a:solidFill><a:schemeClr val="phClr"/></a:solidFill><a:prstDash val="solid"/></a:ln></a:lnStyleLst><a:effectStyleLst><a:effectStyle><a:effectLst/></a:effectStyle></a:effectStyleLst><a:bgFillStyleLst><a:solidFill><a:schemeClr val="phClr"/></a:solidFill></a:bgFillStyleLst></a:fmtScheme>`;
41
+ return `<a:fmtScheme name="${escapeXmlAttribute(theme.formatScheme?.name ?? "Folio")}"><a:fillStyleLst><a:solidFill><a:schemeClr val="phClr"/></a:solidFill></a:fillStyleLst><a:lnStyleLst><a:ln w="6350" cap="flat" cmpd="sng" algn="ctr"><a:solidFill><a:schemeClr val="phClr"/></a:solidFill><a:prstDash val="solid"/></a:ln></a:lnStyleLst><a:effectStyleLst><a:effectStyle><a:effectLst/></a:effectStyle></a:effectStyleLst><a:bgFillStyleLst><a:solidFill><a:schemeClr val="phClr"/></a:solidFill></a:bgFillStyleLst></a:fmtScheme>`;
42
42
  };
43
43
  //#endregion
44
44
  export { serializeThemeXml };
@@ -1,6 +1,6 @@
1
1
  import { DATE_UTC_ATTRIBUTE } from "../trackedChangeInfo.js";
2
- import { escapeXml } from "./xmlUtils.js";
3
2
  import { panic } from "better-result";
3
+ import { escapeXmlAttribute } from "@stll/docx-core";
4
4
  import { normalizeRevisionId } from "@stll/docx-core/model";
5
5
  //#region src/docx/serializer/trackedChangeAttributes.ts
6
6
  /** Enforces the singular `w:rPrChange` child in a run-property container. */
@@ -26,6 +26,6 @@ const trackedChangeAttributeEntries = (info) => {
26
26
  return entries;
27
27
  };
28
28
  const trackedChangeAttributeRecord = (info) => Object.fromEntries(trackedChangeAttributeEntries(info));
29
- const serializeTrackedChangeAttributes = (info) => trackedChangeAttributeEntries(info).map(([name, value]) => `${name}="${escapeXml(value)}"`).join(" ");
29
+ const serializeTrackedChangeAttributes = (info) => trackedChangeAttributeEntries(info).map(([name, value]) => `${name}="${escapeXmlAttribute(value)}"`).join(" ");
30
30
  //#endregion
31
31
  export { getSingularRunPropertyChange, serializeTrackedChangeAttributes, trackedChangeAttributeEntries, trackedChangeAttributeRecord };
@@ -2,7 +2,6 @@
2
2
  /**
3
3
  * Shared XML utility functions for serializers.
4
4
  */
5
- declare function escapeXml(text: string): string;
6
5
  /**
7
6
  * Format a numeric value as an integer XML attribute.
8
7
  *
@@ -29,4 +28,4 @@ declare function intAttr(value: number | undefined | null): string;
29
28
  */
30
29
  declare function isSingleWellFormedElement(xml: string, expectedLocalName: string): boolean;
31
30
  //#endregion
32
- export { escapeXml, intAttr, isSingleWellFormedElement };
31
+ export { intAttr, isSingleWellFormedElement };
@@ -3,18 +3,6 @@ import { getLocalName, parseXml } from "../xmlParser.js";
3
3
  /**
4
4
  * Shared XML utility functions for serializers.
5
5
  */
6
- const XML_SPECIAL_CHARACTER_PATTERN = /[&<>"']/u;
7
- const XML_SPECIAL_CHARACTER_GLOBAL_PATTERN = /[&<>"']/gu;
8
- function escapeXml(text) {
9
- if (!XML_SPECIAL_CHARACTER_PATTERN.test(text)) return text;
10
- return text.replace(XML_SPECIAL_CHARACTER_GLOBAL_PATTERN, (character) => {
11
- if (character === "&") return "&amp;";
12
- if (character === "<") return "&lt;";
13
- if (character === ">") return "&gt;";
14
- if (character === "\"") return "&quot;";
15
- return "&apos;";
16
- });
17
- }
18
6
  /**
19
7
  * Format a numeric value as an integer XML attribute.
20
8
  *
@@ -57,4 +45,4 @@ function isSingleWellFormedElement(xml, expectedLocalName) {
57
45
  return root?.name !== void 0 && getLocalName(root.name) === expectedLocalName;
58
46
  }
59
47
  //#endregion
60
- export { escapeXml, intAttr, isSingleWellFormedElement };
48
+ export { intAttr, isSingleWellFormedElement };
@@ -1,3 +1,4 @@
1
+ import { XmlResourceLimits } from "../xmlResourceLimits.js";
1
2
  import JSZip from "jszip";
2
3
  //#region src/docx/server/boundedArchive.d.ts
3
4
  declare module "jszip" {
@@ -27,6 +28,17 @@ type DocxArchiveOptions = {
27
28
  maxEntryBytes?: number;
28
29
  maxTotalBytes?: number;
29
30
  maxEntries?: number;
31
+ /**
32
+ * Bounds on parsed XML structure, applied to every XML part this archive
33
+ * hands out as a string.
34
+ *
35
+ * Every consumer of `readEntryString` parses the result into an object tree,
36
+ * and the byte caps above bound the markup, not the tree. Enforcing the
37
+ * element and attribute bounds here rather than at each parse site means a
38
+ * new consumer is bounded by construction: there is no way to obtain a part
39
+ * string from this archive that has not been counted.
40
+ */
41
+ xmlLimits?: Partial<XmlResourceLimits>;
30
42
  };
31
43
  type DocxArchiveEntry = {
32
44
  readonly path: string;
@@ -1,3 +1,4 @@
1
+ import { FOLIO_XML_RESOURCE_LIMITS, assertXmlResourceLimits, createXmlPackageBudget } from "../xmlResourceLimits.js";
1
2
  import { TaggedError } from "better-result";
2
3
  import JSZip from "jszip";
3
4
  //#region src/docx/server/boundedArchive.ts
@@ -117,6 +118,12 @@ const loadDocxArchive = async (bytes, options = {}) => {
117
118
  message: `DOCX archive declares ${declaredTotalBytes} cumulative bytes (max ${maxTotalBytes})`,
118
119
  reason: "total-too-large"
119
120
  });
121
+ const xmlLimits = {
122
+ ...FOLIO_XML_RESOURCE_LIMITS,
123
+ ...options.xmlLimits
124
+ };
125
+ const xmlBudget = createXmlPackageBudget();
126
+ const countedParts = /* @__PURE__ */ new Set();
120
127
  let totalBytesRead = 0;
121
128
  let readChain = Promise.resolve();
122
129
  const readEntry = async (path, readOptions = {}) => {
@@ -151,7 +158,19 @@ const loadDocxArchive = async (bytes, options = {}) => {
151
158
  }))),
152
159
  async readEntryString(path) {
153
160
  const content = await readEntry(path);
154
- return content === null ? null : new TextDecoder("utf-8", { ignoreBOM: true }).decode(content);
161
+ if (content === null) return null;
162
+ const xml = new TextDecoder("utf-8", { ignoreBOM: true }).decode(content);
163
+ const lower = path.toLowerCase();
164
+ if (!lower.endsWith(".xml") && !lower.endsWith(".rels")) return xml;
165
+ const charged = countedParts.has(path);
166
+ countedParts.add(path);
167
+ assertXmlResourceLimits({
168
+ xml,
169
+ limits: xmlLimits,
170
+ partPath: path,
171
+ ...charged ? {} : { budget: xmlBudget }
172
+ });
173
+ return xml;
155
174
  },
156
175
  readEntryUint8: readEntry
157
176
  };
@@ -3,6 +3,7 @@ import { DOCX_CONTAINER_TYPES, detectDocxContainerType } from "../encryption/con
3
3
  import { DocxModelValidationError, validateFolioDocumentModel } from "../modelValidation.js";
4
4
  import { DocxParseError, parseDocx } from "../parser.js";
5
5
  import { findChildByLocalName, getAttributeAnyPrefix, getLocalName, getNamespacePrefix, parseXmlDocument } from "../xmlParser.js";
6
+ import { XmlResourceLimitError } from "../xmlResourceLimits.js";
6
7
  import { getDocxXmlSafetyIssue } from "../xmlSafety.js";
7
8
  import { DocxArchiveError, loadDocxArchive } from "./boundedArchive.js";
8
9
  import { DOCX_CONFORMANCE_CLASSES } from "@stll/docx-core/model";
@@ -112,11 +113,31 @@ const XML_SAFETY_ISSUE = {
112
113
  message: "An XML package part is not well formed."
113
114
  }
114
115
  };
116
+ /**
117
+ * Read a part for inspection, including one the safe reader declines.
118
+ *
119
+ * `readEntryString` runs the XML structure preflight, and the preflight
120
+ * refuses to scan markup it cannot count — a document type declaration, an
121
+ * unterminated tag — which is exactly the markup this check exists to report.
122
+ * Such a part is read raw and classified below, so the report names why the
123
+ * package is invalid instead of saying it could not tell. A part that overran
124
+ * a real bound still refuses, and the caller reports that as indeterminate,
125
+ * because a package too large to scan is not a package known to be wrong.
126
+ */
127
+ const readPartForInspection = async (archive, part) => {
128
+ try {
129
+ return await archive.readEntryString(part);
130
+ } catch (error) {
131
+ if (!(error instanceof XmlResourceLimitError) || error.limit !== "syntax") throw error;
132
+ const bytes = await archive.readEntryUint8(part);
133
+ return bytes === null ? null : new TextDecoder("utf-8", { ignoreBOM: true }).decode(bytes);
134
+ }
135
+ };
115
136
  const validateXmlParts = async (archive, report) => {
116
137
  const xmlByPath = /* @__PURE__ */ new Map();
117
138
  let hasInvalidXml = false;
118
139
  for (const part of archive.entries.filter((path) => XML_PART_PATTERN.test(path)).toSorted()) {
119
- const xml = await archive.readEntryString(part);
140
+ const xml = await readPartForInspection(archive, part);
120
141
  if (xml === null) continue;
121
142
  xmlByPath.set(part, xml);
122
143
  const safetyIssue = getDocxXmlSafetyIssue(xml);
@@ -1,4 +1,5 @@
1
1
  import { parseAnchorPosition, parseAnchorWrap, parseFill, parseOutline } from "./drawingUtils.js";
2
+ import { parseNonVisualDrawingNames } from "./nonVisualDrawingProps.js";
2
3
  import { ShapeTypeSchema, narrowEnum } from "./parserEnums.js";
3
4
  import { findAllDeep, findChildByLocalName, findChildren, getAttribute, getChildElements, parseNumericAttribute, parseOnOffAttribute } from "./xmlParser.js";
4
5
  //#region src/docx/shapeParser.ts
@@ -139,16 +140,15 @@ function parseShape(node) {
139
140
  const fill = parseShapeFill(spPr);
140
141
  const outline = parseOutline(spPr);
141
142
  const id = cNvPr ? getAttribute(cNvPr, null, "id") ?? void 0 : void 0;
142
- const name = cNvPr ? getAttribute(cNvPr, null, "name") ?? void 0 : void 0;
143
143
  const shape = {
144
144
  type: "shape",
145
145
  shapeType,
146
- size
146
+ size,
147
+ ...parseNonVisualDrawingNames(cNvPr)
147
148
  };
148
149
  const geometryAdjustments = parseGeometryAdjustments(spPr);
149
150
  if (geometryAdjustments !== void 0) shape.geometryAdjustments = geometryAdjustments;
150
151
  if (id !== void 0) shape.id = id;
151
- if (name !== void 0) shape.name = name;
152
152
  if (fill !== void 0) shape.fill = fill;
153
153
  if (outline !== void 0) shape.outline = outline;
154
154
  if (transform !== void 0) shape.transform = transform;
@@ -193,9 +193,11 @@ function parseShapeFromDrawing(drawingEl) {
193
193
  const docPr = findChildByLocalName(container, "docPr");
194
194
  if (docPr) {
195
195
  const id = getAttribute(docPr, null, "id");
196
- const name = getAttribute(docPr, null, "name");
197
196
  if (id !== null) shape.id = id;
198
- if (name !== null) shape.name = name;
197
+ const names = parseNonVisualDrawingNames(docPr);
198
+ if (names.name !== void 0) shape.name = names.name;
199
+ if (names.alt !== void 0) shape.alt = names.alt;
200
+ if (names.title !== void 0) shape.title = names.title;
199
201
  }
200
202
  return shape;
201
203
  }
@@ -1,5 +1,6 @@
1
1
  import { emuToPixels } from "../utils/units.js";
2
2
  import { parseAnchorPosition, parseAnchorWrap, parseFill, parseOutline, resolveColorValueToHex } from "./drawingUtils.js";
3
+ import { parseNonVisualDrawingNames } from "./nonVisualDrawingProps.js";
3
4
  import { findByFullName, findChildByLocalName, findChildByNamespaceUri, findChildrenByLocalName, findChildrenByNamespaceUri, getAttribute, getChildElements, parseNumericAttribute, parseOnOffAttribute } from "./xmlParser.js";
4
5
  //#region src/docx/textBoxParser.ts
5
6
  const DRAWINGML_MAIN_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.openxmlformats.org/drawingml/2006/main", "http://purl.oclc.org/ooxml/drawingml/main"]);
@@ -148,13 +149,15 @@ function parseTextBox(drawingEl) {
148
149
  };
149
150
  const docPr = findByFullName(container, "wp:docPr");
150
151
  const id = docPr ? getAttribute(docPr, null, "id") ?? void 0 : void 0;
152
+ const names = parseNonVisualDrawingNames(docPr);
151
153
  const fill = parseFill(spPr ?? null);
152
154
  const outline = parseOutline(spPr ?? null);
153
155
  const bodyProps = parseBodyProperties(bodyPr ?? null);
154
156
  const textBox = {
155
157
  type: "textBox",
156
158
  size,
157
- content: []
159
+ content: [],
160
+ ...names
158
161
  };
159
162
  if (id) textBox.id = id;
160
163
  if (fill) textBox.fill = fill;
@@ -193,13 +196,15 @@ function parseTextBoxFromShape(wsp, size, position, wrap) {
193
196
  const bodyPr = findChildByNamespaceUri(wsp, WORDPROCESSING_SHAPE_NAMESPACE_URIS, "bodyPr");
194
197
  const cNvPr = wspChildren.find((el) => el.name === "wps:cNvPr");
195
198
  const id = cNvPr ? getAttribute(cNvPr, null, "id") ?? void 0 : void 0;
199
+ const names = parseNonVisualDrawingNames(cNvPr);
196
200
  const fill = parseFill(spPr ?? null);
197
201
  const outline = parseOutline(spPr ?? null);
198
202
  const bodyProps = parseBodyProperties(bodyPr ?? null);
199
203
  const textBox = {
200
204
  type: "textBox",
201
205
  size,
202
- content: []
206
+ content: [],
207
+ ...names
203
208
  };
204
209
  if (id) textBox.id = id;
205
210
  if (fill) textBox.fill = fill;
@@ -6,10 +6,33 @@ declare class DocxSecurityError extends Error {
6
6
  type DocxUnzipLimits = {
7
7
  maxInputBytes: number;
8
8
  maxFiles: number;
9
+ /**
10
+ * Inflated bytes allowed in one XML part.
11
+ *
12
+ * A byte ceiling bounds the markup, not what parsing it allocates: a parsed
13
+ * tree costs 3x to 15x the part's bytes, and the cheapest element to write
14
+ * is the most expensive per byte. Use `maxXmlElementsPerPart` and the
15
+ * package bounds to cap memory; this one caps text-dominated parts, where
16
+ * bytes and cost do track each other.
17
+ */
9
18
  maxXmlBytes: number;
10
19
  maxMediaBytes: number;
11
20
  maxFontBytes: number;
21
+ /**
22
+ * Inflated bytes allowed across every entry in the package.
23
+ *
24
+ * Same caveat as `maxXmlBytes`: this is a ceiling on what passes through
25
+ * memory as bytes, not on what the parsed structure retains.
26
+ */
12
27
  maxTotalUncompressedBytes: number;
28
+ /** Elements allowed in one XML part, counted before any tree is built. */
29
+ maxXmlElementsPerPart: number;
30
+ /** Attributes allowed in one XML part, counted before any tree is built. */
31
+ maxXmlAttributesPerPart: number;
32
+ /** Elements allowed across every XML part in the package. */
33
+ maxXmlElementsPerPackage: number;
34
+ /** Attributes allowed across every XML part in the package. */
35
+ maxXmlAttributesPerPackage: number;
13
36
  allowedMediaMimeTypes: ReadonlySet<string>;
14
37
  };
15
38
  type DocxUnzipOptions = Partial<Omit<DocxUnzipLimits, "allowedMediaMimeTypes">> & {
@@ -1,6 +1,7 @@
1
+ import { bytesToDataUrl } from "../utils/base64.js";
1
2
  import { DOCX_CONTAINER_TYPES, detectDocxContainerType } from "./encryption/containerFormat.js";
2
3
  import { openDocxBuffer } from "./encryption/openEncryptedDocx.js";
3
- import { FOLIO_XML_RESOURCE_LIMITS, assertXmlResourceLimits } from "./xmlResourceLimits.js";
4
+ import { FOLIO_XML_RESOURCE_LIMITS, assertXmlResourceLimits, createXmlPackageBudget } from "./xmlResourceLimits.js";
4
5
  import JSZip from "jszip";
5
6
  //#region src/docx/unzip.ts
6
7
  /**
@@ -58,8 +59,21 @@ const DEFAULT_UNZIP_LIMITS = {
58
59
  maxMediaBytes: 25 * MEBIBYTE,
59
60
  maxFontBytes: 10 * MEBIBYTE,
60
61
  maxTotalUncompressedBytes: 250 * MEBIBYTE,
62
+ maxXmlElementsPerPart: FOLIO_XML_RESOURCE_LIMITS.maxElementsPerPart,
63
+ maxXmlAttributesPerPart: FOLIO_XML_RESOURCE_LIMITS.maxAttributesPerPart,
64
+ maxXmlElementsPerPackage: FOLIO_XML_RESOURCE_LIMITS.maxElementsPerPackage,
65
+ maxXmlAttributesPerPackage: FOLIO_XML_RESOURCE_LIMITS.maxAttributesPerPackage,
61
66
  allowedMediaMimeTypes: DEFAULT_ALLOWED_MEDIA_MIME_TYPES
62
67
  };
68
+ /** The XML bounds an unzip enforces, in the shape the preflight takes. */
69
+ const xmlResourceLimitsFor = (limits) => ({
70
+ maxBytes: limits.maxXmlBytes,
71
+ maxDepth: FOLIO_XML_RESOURCE_LIMITS.maxDepth,
72
+ maxElementsPerPart: limits.maxXmlElementsPerPart,
73
+ maxAttributesPerPart: limits.maxXmlAttributesPerPart,
74
+ maxElementsPerPackage: limits.maxXmlElementsPerPackage,
75
+ maxAttributesPerPackage: limits.maxXmlAttributesPerPackage
76
+ });
63
77
  const PARSED_XML_PARTS = /* @__PURE__ */ new Set([
64
78
  "[content_types].xml",
65
79
  "_rels/.rels",
@@ -188,10 +202,11 @@ async function unzipDocx(buffer, options = {}) {
188
202
  }));
189
203
  }
190
204
  }
205
+ const xmlBudget = createXmlPackageBudget();
191
206
  for (const extracted of await Promise.all(extractionTasks.map((extract) => extract()))) {
192
207
  if (!extracted) continue;
193
208
  if (extracted.type === "xml") {
194
- assignXmlContent(content, extracted, limits);
209
+ assignXmlContent(content, extracted, limits, xmlBudget);
195
210
  continue;
196
211
  }
197
212
  if (extracted.type === "media") {
@@ -203,19 +218,21 @@ async function unzipDocx(buffer, options = {}) {
203
218
  return content;
204
219
  }
205
220
  /**
206
- * Parts that every consumer expands into an object tree. Preflighting them once
207
- * here puts the bound on the unzip, so `parseDocx`, the selective save and the
208
- * repack path all share it instead of each entry point carrying its own.
221
+ * Preflight every XML part the unzip retains, not a named few.
222
+ *
223
+ * Naming the parts to bound is the bug: `word/document.xml`, `word/styles.xml`
224
+ * and `word/numbering.xml` were counted, and headers, footers, footnotes,
225
+ * endnotes, comments and every other `word/*.xml` part were parsed into trees
226
+ * unbounded. Bounding the reader instead of the part list means a part added
227
+ * later is bounded by construction, and the package budget makes the ceiling
228
+ * the package's rather than each part's.
209
229
  */
210
- const PREFLIGHT_XML_PARTS = /* @__PURE__ */ new Set([
211
- "word/document.xml",
212
- "word/styles.xml",
213
- "word/numbering.xml"
214
- ]);
215
- function assignXmlContent(content, { path, lowerPath, content: xmlContent }, limits) {
216
- if (PREFLIGHT_XML_PARTS.has(lowerPath)) assertXmlResourceLimits(xmlContent, {
217
- ...FOLIO_XML_RESOURCE_LIMITS,
218
- maxBytes: limits.maxXmlBytes
230
+ function assignXmlContent(content, { path, lowerPath, content: xmlContent }, limits, budget) {
231
+ assertXmlResourceLimits({
232
+ xml: xmlContent,
233
+ limits: xmlResourceLimitsFor(limits),
234
+ partPath: path,
235
+ budget
219
236
  });
220
237
  content.allXml.set(path, xmlContent);
221
238
  if (lowerPath === "word/document.xml") content.documentXml = xmlContent;
@@ -426,14 +443,7 @@ function getMediaMimeType(path) {
426
443
  * @returns Data URL string
427
444
  */
428
445
  function mediaToDataUrl(data, mimeType) {
429
- const bytes = new Uint8Array(data);
430
- const chunks = [];
431
- const chunkSize = 32768;
432
- for (let offset = 0; offset < bytes.length; offset += chunkSize) {
433
- const chunk = bytes.subarray(offset, offset + chunkSize);
434
- chunks.push(String.fromCodePoint(...chunk));
435
- }
436
- return `data:${mimeType};base64,${btoa(chunks.join(""))}`;
446
+ return bytesToDataUrl(new Uint8Array(data), mimeType);
437
447
  }
438
448
  /**
439
449
  * Extract a specific file from the original ZIP
@@ -270,7 +270,7 @@ const parsedReplayEnvelope = (xml, inheritedNamespaceScope) => {
270
270
  };
271
271
  const withinXmlResourceLimits = (xml) => {
272
272
  try {
273
- assertXmlResourceLimits(xml);
273
+ assertXmlResourceLimits({ xml });
274
274
  return true;
275
275
  } catch {
276
276
  return false;
@@ -1,6 +1,7 @@
1
1
  import { sanitizeImageSrc } from "../utils/sanitizeImageSrc.js";
2
2
  import { pixelsToEmu } from "../utils/units.js";
3
3
  import { resolveImageData } from "./imageParser.js";
4
+ import { PREVIEW_KINDS } from "./previewBudget.js";
4
5
  import { captureVerbatimXml } from "./verbatimCapture.js";
5
6
  import { isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl } from "./vmlPreview.js";
6
7
  import { isWatermarkShape } from "./watermarkParser.js";
@@ -73,8 +74,8 @@ const previewImage = (pictElement, svg, widthPx, heightPx, style, rootXmlns) =>
73
74
  type: "image",
74
75
  rId: "",
75
76
  src,
76
- mimeType: "image/svg+xml",
77
- filename: "vml-shape-preview.svg",
77
+ mimeType: PREVIEW_KINDS.vmlShape.mimeType,
78
+ filename: PREVIEW_KINDS.vmlShape.filename,
78
79
  size: {
79
80
  width: pixelsToEmu(widthPx),
80
81
  height: pixelsToEmu(heightPx)
@@ -25,7 +25,5 @@ declare const vmlSvgDataUrl: (svg: string) => string | undefined;
25
25
  declare const renderStandaloneVmlPreview: (shape: XmlElement) => VmlPreviewResult;
26
26
  /** Render a VML group through bounded, non-clipping local-coordinate transforms. */
27
27
  declare const renderVmlGroupPreview: (group: XmlElement) => VmlPreviewResult;
28
- /** Bound retained synthetic VML previews while preserving their raw replay nodes. */
29
- declare const enforcePackageVmlPreviewBudget: (root: unknown, maxCharacters?: number) => void;
30
28
  //#endregion
31
- export { VmlPreviewResult, enforcePackageVmlPreviewBudget, isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
29
+ export { VmlPreviewResult, isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
@@ -1,3 +1,4 @@
1
+ import { VML_PREVIEW_DATA_URL_PREFIX } from "./previewBudget.js";
1
2
  import { findChild, findDeep, getAttribute, getChildElements, getLocalName } from "./xmlParser.js";
2
3
  //#region src/docx/vmlPreview.ts
3
4
  const MAX_VML_PREVIEW_DEPTH = 16;
@@ -6,10 +7,6 @@ const MAX_VML_PREVIEW_PATH_POINTS = 2e4;
6
7
  const MAX_VML_PREVIEW_COORDINATE = 1e6;
7
8
  const MAX_VML_PREVIEW_DIMENSION_PX = 2e4;
8
9
  const MAX_VML_SVG_CHARACTERS = 1e6;
9
- const MAX_PACKAGE_VML_PREVIEW_CHARACTERS = 8 * 1024 * 1024;
10
- const VML_PREVIEW_DATA_URL_PREFIX = "data:image/svg+xml;charset=utf-8,";
11
- const VML_PREVIEW_FILENAME = "vml-shape-preview.svg";
12
- const VML_PREVIEW_MIME_TYPE = "image/svg+xml";
13
10
  const SAFE_VML_COLORS = /* @__PURE__ */ new Set([
14
11
  "black",
15
12
  "white",
@@ -506,30 +503,5 @@ const renderVmlGroupPreview = (group) => {
506
503
  style
507
504
  };
508
505
  };
509
- /** Bound retained synthetic VML previews while preserving their raw replay nodes. */
510
- const enforcePackageVmlPreviewBudget = (root, maxCharacters = MAX_PACKAGE_VML_PREVIEW_CHARACTERS) => {
511
- let remainingCharacters = Math.max(0, maxCharacters);
512
- const visited = /* @__PURE__ */ new WeakSet();
513
- const visit = (value) => {
514
- if (value === null || typeof value !== "object" || visited.has(value)) return;
515
- visited.add(value);
516
- if (value instanceof ArrayBuffer || ArrayBuffer.isView(value)) return;
517
- if (value instanceof Map) {
518
- for (const child of value.values()) visit(child);
519
- return;
520
- }
521
- if (Array.isArray(value)) {
522
- for (const child of value) visit(child);
523
- return;
524
- }
525
- if ("type" in value && value.type === "image" && "rId" in value && value.rId === "" && "mimeType" in value && value.mimeType === VML_PREVIEW_MIME_TYPE && "filename" in value && value.filename === VML_PREVIEW_FILENAME && "src" in value && typeof value.src === "string" && value.src.startsWith(VML_PREVIEW_DATA_URL_PREFIX)) if (value.src.length <= remainingCharacters) remainingCharacters -= value.src.length;
526
- else {
527
- remainingCharacters = 0;
528
- delete value.src;
529
- }
530
- for (const child of Object.values(value)) visit(child);
531
- };
532
- visit(root);
533
- };
534
506
  //#endregion
535
- export { enforcePackageVmlPreviewBudget, isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
507
+ export { isValidVmlPreviewDimension, parseVmlNumber, parseVmlStyle, renderStandaloneVmlPreview, renderVmlGroupPreview, vmlCssLengthToPx, vmlSvgDataUrl };
@@ -71,7 +71,22 @@ declare const NAMESPACES: {
71
71
  declare const OOXML_NAMESPACE_SCOPE: XmlNamespaceScope;
72
72
  declare function parseXml(xml: string, inheritedNamespaceScope?: XmlNamespaceScope): XmlElement;
73
73
  /**
74
- * Serialize an XmlElement back to an XML string
74
+ * Serialize an XmlElement back to an XML string.
75
+ *
76
+ * Written here rather than handed to `fast-xml-parser`'s builder for two
77
+ * reasons. The builder writes a tab or a newline inside an attribute value
78
+ * literally, and XML 1.0 §3.3.3 has every conformant reader normalise those to
79
+ * a space: a `descr="two lines&#xA;"` a source file wrote came back as
80
+ * `descr="two lines "`. `escapeXmlAttribute` writes the character references
81
+ * that survive that step. And the builder returned a string per node, so a
82
+ * subtree's bytes were copied into its parent's answer, its grandparent's and
83
+ * so on, which `writeElement` below replaces with one buffer.
84
+ *
85
+ * Those two are the whole difference from the builder: the attribute
86
+ * character references, an attribute the model dropped written as absent
87
+ * rather than as the word "undefined", and the characters XML 1.0 §2.2 admits
88
+ * no spelling for dropped. Every other rule reproduces its output byte for
89
+ * byte, so replaying a capture is unchanged wherever it was already correct.
75
90
  */
76
91
  declare function elementToXml(element: XmlElement): string;
77
92
  /**
@@ -1,6 +1,7 @@
1
1
  import { NUMBERS_PER_PERCENT, percentageSpelling, transitionalSlotEncoding } from "./transitionalSpelling.js";
2
2
  import { universalMeasureAs } from "./universalMeasure.js";
3
- import { XMLBuilder, XMLParser } from "fast-xml-parser";
3
+ import { escapeXmlAttribute, escapeXmlText } from "@stll/docx-core";
4
+ import { XMLParser } from "fast-xml-parser";
4
5
  import { OOXML_NS } from "@stll/docx-utils";
5
6
  import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
6
7
  //#region src/docx/xmlParser.ts
@@ -42,22 +43,12 @@ const fxpParserOptionsWithStopNodes = {
42
43
  ...fxpParserOptions,
43
44
  stopNodes: ["*.w:binData"]
44
45
  };
45
- const fxpBuilderOptions = {
46
- preserveOrder: true,
47
- ignoreAttributes: false,
48
- attributeNamePrefix: "",
49
- textNodeName: "#text",
50
- suppressEmptyNode: true
51
- };
52
46
  const fxpParser = new XMLParser(fxpParserOptions);
53
47
  const fxpParserWithStopNodes = new XMLParser(fxpParserOptionsWithStopNodes);
54
- const fxpBuilder = new XMLBuilder(fxpBuilderOptions);
55
48
  /** Text node key used by fast-xml-parser in preserveOrder mode. */
56
49
  const TEXT_KEY = "#text";
57
50
  /** Attribute group key used by fast-xml-parser in preserveOrder mode. */
58
51
  const ATTR_KEY = ":@";
59
- /** Character reference required to keep carriage returns through XML end-of-line normalization. */
60
- const XML_CARRIAGE_RETURN_REFERENCE = "&#13;";
61
52
  const EMPTY_NAMESPACE_SCOPE = { bindings: /* @__PURE__ */ new Map() };
62
53
  const resolveNamespaceUri = (scope, prefix) => {
63
54
  let current = scope;
@@ -138,18 +129,6 @@ function fxpToRootElement(nodes, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE
138
129
  return { elements: nodes };
139
130
  }
140
131
  /**
141
- * Convert an XmlElement back into the fast-xml-parser preserveOrder format
142
- * so we can feed it to XMLBuilder.
143
- */
144
- function elementToFxpNode(el) {
145
- if (el.type === "text") return { [TEXT_KEY]: el.text ?? "" };
146
- const name = el.name ?? "";
147
- const children = el.elements ? el.elements.map(elementToFxpNode) : [];
148
- const node = { [name]: children };
149
- if (el.attributes && Object.keys(el.attributes).length > 0) node[ATTR_KEY] = el.attributes;
150
- return node;
151
- }
152
- /**
153
132
  * Common OOXML namespace URIs — re-exported from @stll/docx-utils.
154
133
  */
155
134
  const NAMESPACES = OOXML_NS;
@@ -174,13 +153,64 @@ function parseXml(xml, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE) {
174
153
  return fxpToRootElement((xml.includes("binData") ? fxpParserWithStopNodes : fxpParser).parse(xml), inheritedNamespaceScope);
175
154
  }
176
155
  /**
177
- * Serialize an XmlElement back to an XML string
156
+ * Serialize an XmlElement back to an XML string.
157
+ *
158
+ * Written here rather than handed to `fast-xml-parser`'s builder for two
159
+ * reasons. The builder writes a tab or a newline inside an attribute value
160
+ * literally, and XML 1.0 §3.3.3 has every conformant reader normalise those to
161
+ * a space: a `descr="two lines&#xA;"` a source file wrote came back as
162
+ * `descr="two lines "`. `escapeXmlAttribute` writes the character references
163
+ * that survive that step. And the builder returned a string per node, so a
164
+ * subtree's bytes were copied into its parent's answer, its grandparent's and
165
+ * so on, which `writeElement` below replaces with one buffer.
166
+ *
167
+ * Those two are the whole difference from the builder: the attribute
168
+ * character references, an attribute the model dropped written as absent
169
+ * rather than as the word "undefined", and the characters XML 1.0 §2.2 admits
170
+ * no spelling for dropped. Every other rule reproduces its output byte for
171
+ * byte, so replaying a capture is unchanged wherever it was already correct.
178
172
  */
179
173
  function elementToXml(element) {
180
- const fxpNode = elementToFxpNode(element);
181
- return fxpBuilder.build([fxpNode]).replaceAll("\r", XML_CARRIAGE_RETURN_REFERENCE);
174
+ const out = [];
175
+ writeElement(element, out);
176
+ return out.join("");
182
177
  }
183
178
  /**
179
+ * Serialize into one shared buffer rather than a string per node.
180
+ *
181
+ * Returning a string per element makes a subtree's bytes a substring of its
182
+ * parent's, its grandparent's and so on, so a table pays for its rows, its
183
+ * rows pay for their cells, and a part that nests four levels deep is copied
184
+ * four times before anything is written. Appending into one array of chunks
185
+ * and joining once costs each node its own text and nothing for its ancestors.
186
+ *
187
+ * Whether an element self-closes is not known until its children are written,
188
+ * so the opening tag reserves a slot in the buffer and fills it afterwards:
189
+ * `>` when something was appended, `/>` when nothing was.
190
+ */
191
+ const writeElement = (element, out) => {
192
+ if (element.type === "text") {
193
+ const text = String(element.text ?? "");
194
+ if (text !== "") out.push(escapeXmlText(text));
195
+ return;
196
+ }
197
+ const name = element.name ?? "";
198
+ out.push(`<${name}`);
199
+ if (element.attributes) for (const [attribute, value] of Object.entries(element.attributes)) {
200
+ if (value === void 0) continue;
201
+ out.push(` ${attribute}="${escapeXmlAttribute(String(value))}"`);
202
+ }
203
+ const openingSlot = out.push("") - 1;
204
+ const contentStart = out.length;
205
+ for (const child of element.elements ?? []) writeElement(child, out);
206
+ if (out.length === contentStart) {
207
+ out[openingSlot] = "/>";
208
+ return;
209
+ }
210
+ out[openingSlot] = ">";
211
+ out.push(`</${name}>`);
212
+ };
213
+ /**
184
214
  * Parse XML string to a more convenient format
185
215
  */
186
216
  function parseXmlDocument(xml) {