@stll/folio-core 0.44.0 → 0.45.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/dist/ai-edits/__fixtures__/paragraphs.js +2 -2
  2. package/dist/ai-edits/headless.js +1 -0
  3. package/dist/compare/content-alignment.js +78 -53
  4. package/dist/compare/inline-atoms.js +34 -20
  5. package/dist/content-controls/mutateContentControls.js +4 -2
  6. package/dist/display-list/dom/renderDisplayListToDom.js +8 -8
  7. package/dist/document-operations.js +14 -3
  8. package/dist/docx/borderParser.js +5 -5
  9. package/dist/docx/commentParser.js +47 -36
  10. package/dist/docx/commentThreadKey.d.ts +18 -0
  11. package/dist/docx/commentThreadKey.js +22 -0
  12. package/dist/docx/diagramPreview.js +87 -27
  13. package/dist/docx/groupDrawingParser.js +3 -3
  14. package/dist/docx/hyperlinkParser.js +2 -2
  15. package/dist/docx/imageParser.d.ts +9 -1
  16. package/dist/docx/imageParser.js +58 -12
  17. package/dist/docx/imageRawXml.d.ts +14 -1
  18. package/dist/docx/imageRawXml.js +30 -6
  19. package/dist/docx/mathToMathml.js +12 -14
  20. package/dist/docx/nonVisualDrawingProps.d.ts +34 -0
  21. package/dist/docx/nonVisualDrawingProps.js +46 -0
  22. package/dist/docx/paragraphTextBoxEnrichment.js +3 -0
  23. package/dist/docx/parser.js +2 -2
  24. package/dist/docx/previewBudget.d.ts +64 -0
  25. package/dist/docx/previewBudget.js +88 -0
  26. package/dist/docx/revisionIdNormalization.js +17 -5
  27. package/dist/docx/rezip.js +17 -18
  28. package/dist/docx/sdtPropertiesPatch.js +24 -18
  29. package/dist/docx/sectionReferenceHistory.js +2 -2
  30. package/dist/docx/selectiveSave.js +6 -6
  31. package/dist/docx/serializer/blockSdtSerializer.js +38 -26
  32. package/dist/docx/serializer/borderSerializer.d.ts +1 -1
  33. package/dist/docx/serializer/borderSerializer.js +13 -12
  34. package/dist/docx/serializer/commentSerializer.d.ts +41 -16
  35. package/dist/docx/serializer/commentSerializer.js +82 -85
  36. package/dist/docx/serializer/fontTableSerializer.js +6 -6
  37. package/dist/docx/serializer/headerFooterSerializer.js +5 -5
  38. package/dist/docx/serializer/markupRangeAttributes.js +2 -2
  39. package/dist/docx/serializer/numberingSerializer.js +7 -6
  40. package/dist/docx/serializer/paragraphSerializer.js +19 -18
  41. package/dist/docx/serializer/partNamespaces.js +2 -2
  42. package/dist/docx/serializer/runSerializer.js +48 -28
  43. package/dist/docx/serializer/sectionPropertiesSerializer.js +11 -10
  44. package/dist/docx/serializer/settingsSerializer.js +4 -3
  45. package/dist/docx/serializer/stylesSerializer.js +6 -6
  46. package/dist/docx/serializer/tableSerializer.js +10 -9
  47. package/dist/docx/serializer/textFormattingSerializer.js +29 -28
  48. package/dist/docx/serializer/themeSerializer.js +6 -6
  49. package/dist/docx/serializer/trackedChangeAttributes.js +2 -2
  50. package/dist/docx/serializer/xmlUtils.d.ts +1 -2
  51. package/dist/docx/serializer/xmlUtils.js +1 -13
  52. package/dist/docx/server/boundedArchive.d.ts +12 -0
  53. package/dist/docx/server/boundedArchive.js +20 -1
  54. package/dist/docx/server/validateDocxConformance.js +22 -1
  55. package/dist/docx/shapeParser.js +7 -5
  56. package/dist/docx/textBoxParser.js +7 -2
  57. package/dist/docx/unzip.d.ts +23 -0
  58. package/dist/docx/unzip.js +32 -22
  59. package/dist/docx/verbatimCapture.js +1 -1
  60. package/dist/docx/vmlImageParser.js +3 -2
  61. package/dist/docx/vmlPreview.d.ts +1 -3
  62. package/dist/docx/vmlPreview.js +2 -30
  63. package/dist/docx/xmlParser.d.ts +16 -1
  64. package/dist/docx/xmlParser.js +56 -26
  65. package/dist/docx/xmlResourceLimits.d.ts +89 -9
  66. package/dist/docx/xmlResourceLimits.js +105 -24
  67. package/dist/internal/paragraphFormattingSerialization.js +3 -2
  68. package/dist/layout-painter/renderImage.js +4 -3
  69. package/dist/layout-painter/renderParagraph.js +4 -3
  70. package/dist/managers/autoSaveCodec.js +2 -8
  71. package/dist/markdown/images.js +1 -4
  72. package/dist/prosemirror/attrs/index.js +69 -0
  73. package/dist/prosemirror/commands/image.js +1 -0
  74. package/dist/prosemirror/conversion/fromProseDoc.js +69 -29
  75. package/dist/prosemirror/conversion/toProseDoc.js +83 -19
  76. package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +2 -3
  77. package/dist/prosemirror/extensions/nodes/ImageExtension.js +4 -0
  78. package/dist/prosemirror/extensions/nodes/ShapeExtension.js +7 -2
  79. package/dist/prosemirror/extensions/nodes/TextBoxExtension.js +8 -4
  80. package/dist/prosemirror/paragraphFormattingProvenance.d.ts +6 -3
  81. package/dist/prosemirror/paragraphFormattingProvenance.js +12 -3
  82. package/dist/prosemirror/schema/nodes.d.ts +50 -1
  83. package/dist/utils/base64.d.ts +36 -0
  84. package/dist/utils/base64.js +40 -0
  85. package/dist/utils/clipboard.js +2 -1
  86. package/dist/utils/units.d.ts +10 -1
  87. package/dist/utils/units.js +12 -1
  88. package/dist/utils/urlSecurity.d.ts +8 -2
  89. package/dist/utils/urlSecurity.js +21 -3
  90. package/package.json +2 -2
@@ -1,7 +1,8 @@
1
+ import { commentThreadParaId } from "./commentThreadKey.js";
1
2
  import { parseParagraph } from "./paragraphParser.js";
2
3
  import { cloneParagraphWithPropertySource } from "./paragraphPropertySource.js";
3
4
  import { parseRunProperties } from "./runParser.js";
4
- import { findChild, getAttribute, getChildElements, getLocalName, parseOnOffValue, parseXml } from "./xmlParser.js";
5
+ import { NAMESPACES, WORDPROCESSINGML_NAMESPACE_URIS, findChild, getAttribute, getAttributeByNamespaceUri, getChildElements, getLocalName, parseOnOffValue, parseXml } from "./xmlParser.js";
5
6
  import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
6
7
  //#region src/docx/commentParser.ts
7
8
  /**
@@ -25,6 +26,26 @@ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
25
26
  /** The whole lexical form of `ST_DecimalNumber`: an optional sign and digits. */
26
27
  const DECIMAL_NUMBER = /^[+-]?\d+$/u;
27
28
  const DEFAULT_ANNOTATION_REFERENCE_STYLE_ID = "CommentReference";
29
+ /**
30
+ * The namespaces a paraId may be written in: Word 2010 and Word 2012 wordml.
31
+ *
32
+ * Resolved by URI rather than by prefix, because the value is a thread key: an
33
+ * unrelated `vendor:paraId` picked up by local name alone would key a comment
34
+ * to the wrong thread and carry that thread's date, parent and resolved state.
35
+ */
36
+ const PARA_ID_NAMESPACE_URIS = /* @__PURE__ */ new Set([NAMESPACES.w14, NAMESPACES.w15]);
37
+ /**
38
+ * A paraId join key, wherever an exporter writes it: on `w:comment` or `w:p`.
39
+ *
40
+ * One reading for both, so the wrapper and the paragraphs cannot come to
41
+ * disagree about what counts as a key. The literal reads are the fallback for
42
+ * a part that writes a conventional prefix without binding it, which no URI
43
+ * lookup can resolve.
44
+ */
45
+ const paraIdAttribute = (element) => {
46
+ const raw = getAttributeByNamespaceUri(element, PARA_ID_NAMESPACE_URIS, "paraId") ?? element.attributes?.["w14:paraId"] ?? element.attributes?.["w15:paraId"] ?? getAttributeByNamespaceUri(element, WORDPROCESSINGML_NAMESPACE_URIS, "paraId") ?? element.attributes?.["w:paraId"];
47
+ return raw === null || raw === void 0 ? void 0 : String(raw);
48
+ };
28
49
  const normalizeAnnotationReferenceFormatting = (formatting) => {
29
50
  if (formatting?.styleId === DEFAULT_ANNOTATION_REFERENCE_STYLE_ID && Object.keys(formatting).length === 1) return;
30
51
  return formatting;
@@ -56,7 +77,8 @@ function parseCommentsExtensible(xml) {
56
77
  if ((child.name?.replace(/^.*:/u, "") ?? "") !== "comment") continue;
57
78
  const paraId = getAttribute(child, "w16cex", "paraId") ?? getAttribute(child, "w15", "paraId") ?? child.attributes?.["w16cex:paraId"] ?? child.attributes?.["w15:paraId"];
58
79
  const dateUtc = getAttribute(child, "w16cex", "dateUtc") ?? getAttribute(child, "w15", "dateUtc") ?? child.attributes?.["w16cex:dateUtc"] ?? child.attributes?.["w15:dateUtc"];
59
- if (paraId && dateUtc) dateUtcByParaId.set(String(paraId).toUpperCase(), String(dateUtc));
80
+ const key = paraId === null || paraId === void 0 ? null : String(paraId).toUpperCase();
81
+ if (key && dateUtc && !dateUtcByParaId.has(key)) dateUtcByParaId.set(key, String(dateUtc));
60
82
  }
61
83
  return dateUtcByParaId;
62
84
  }
@@ -82,12 +104,14 @@ function parseCommentsExtended(xml) {
82
104
  if ((child.name?.replace(/^.*:/u, "") ?? "") !== "commentEx") continue;
83
105
  const paraId = getAttribute(child, "w15", "paraId") ?? child.attributes?.["w15:paraId"];
84
106
  if (!paraId) continue;
107
+ const key = String(paraId).toUpperCase();
108
+ if (infoByParaId.has(key)) continue;
85
109
  const parentParaId = getAttribute(child, "w15", "paraIdParent") ?? child.attributes?.["w15:paraIdParent"];
86
110
  const doneAttr = getAttribute(child, "w15", "done") ?? child.attributes?.["w15:done"];
87
111
  const info = {};
88
112
  if (parentParaId) info.parentParaId = String(parentParaId).toUpperCase();
89
113
  if (doneAttr !== void 0) info.done = parseOnOffValue(String(doneAttr).toLowerCase()) ?? false;
90
- infoByParaId.set(String(paraId).toUpperCase(), info);
114
+ infoByParaId.set(key, info);
91
115
  }
92
116
  return infoByParaId;
93
117
  }
@@ -105,9 +129,8 @@ function parseComments(commentsXml, styles, theme, rels, media, commentsExtensib
105
129
  const dateUtcByParaId = commentsExtensibleXml ? parseCommentsExtensible(commentsExtensibleXml) : /* @__PURE__ */ new Map();
106
130
  const extendedByParaId = commentsExtendedXml ? parseCommentsExtended(commentsExtendedXml) : /* @__PURE__ */ new Map();
107
131
  const children = getChildElements(findChild(root, "w", "comments") ?? root);
108
- const comments = [];
132
+ const parsed = [];
109
133
  const commentIdByParaId = /* @__PURE__ */ new Map();
110
- const paraIdByCommentIndex = /* @__PURE__ */ new Map();
111
134
  for (const child of children) {
112
135
  if ((child.name?.replace(/^.*:/u, "") ?? "") !== "comment") continue;
113
136
  const rawId = getAttribute(child, "w", "id");
@@ -126,12 +149,7 @@ function parseComments(commentsXml, styles, theme, rels, media, commentsExtensib
126
149
  const initials = rawInitials !== null ? String(rawInitials) : void 0;
127
150
  const rawDate = getAttribute(child, "w", "date");
128
151
  const localDate = rawDate !== null ? String(rawDate) : void 0;
129
- let rawParaId = getAttribute(child, "w14", "paraId") ?? child.attributes?.["w14:paraId"] ?? getAttribute(child, "w15", "paraId") ?? child.attributes?.["w15:paraId"] ?? getAttribute(child, "w", "paraId");
130
- if (!rawParaId) for (const sub of getChildElements(child)) {
131
- if ((sub.name?.replace(/^.*:/u, "") ?? "") !== "p") continue;
132
- const subParaId = getAttribute(sub, "w14", "paraId") ?? sub.attributes?.["w14:paraId"] ?? getAttribute(sub, "w15", "paraId") ?? sub.attributes?.["w15:paraId"] ?? getAttribute(sub, "w", "paraId");
133
- if (subParaId) rawParaId = subParaId;
134
- }
152
+ const rawParaId = paraIdAttribute(child) ?? commentThreadParaId(getChildElements(child).filter((sub) => (sub.name?.replace(/^.*:/u, "") ?? "") === "p").map(paraIdAttribute));
135
153
  const paraId = rawParaId ? String(rawParaId).toUpperCase() : null;
136
154
  const date = (paraId ? dateUtcByParaId.get(paraId) : void 0) ?? localDate;
137
155
  const done = (paraId ? extendedByParaId.get(paraId) : void 0)?.done;
@@ -147,33 +165,26 @@ function parseComments(commentsXml, styles, theme, rels, media, commentsExtensib
147
165
  annotationReferenceFormatting = normalized.annotationReferenceFormatting;
148
166
  paragraphs.push(normalized.paragraph);
149
167
  }
150
- const commentIndex = comments.length;
151
- if (paraId) {
152
- commentIdByParaId.set(paraId, id);
153
- paraIdByCommentIndex.set(commentIndex, paraId);
154
- }
155
- comments.push({
156
- id,
157
- author,
158
- ...initials !== void 0 ? { initials } : {},
159
- ...date !== void 0 ? { date } : {},
160
- ...done !== void 0 ? { done } : {},
161
- ...annotationReferenceFormatting !== void 0 ? { annotationReferenceFormatting } : {},
162
- content: paragraphs
168
+ if (paraId && !commentIdByParaId.has(paraId)) commentIdByParaId.set(paraId, id);
169
+ parsed.push({
170
+ comment: {
171
+ id,
172
+ author,
173
+ ...initials !== void 0 ? { initials } : {},
174
+ ...date !== void 0 ? { date } : {},
175
+ ...done !== void 0 ? { done } : {},
176
+ ...annotationReferenceFormatting !== void 0 ? { annotationReferenceFormatting } : {},
177
+ content: paragraphs
178
+ },
179
+ threadParaId: paraId
163
180
  });
164
181
  }
165
- for (let i = 0; i < comments.length; i++) {
166
- const comment = comments[i];
167
- if (!comment) continue;
168
- const paraId = paraIdByCommentIndex.get(i);
169
- if (!paraId) continue;
170
- const parentParaId = extendedByParaId.get(paraId)?.parentParaId;
171
- if (!parentParaId) continue;
172
- const parentId = commentIdByParaId.get(parentParaId);
173
- if (parentId !== void 0 && parentId !== comment.id) comments[i] = {
174
- ...comment,
175
- parentId
176
- };
182
+ const comments = [];
183
+ for (const { comment, threadParaId } of parsed) {
184
+ const parentParaId = threadParaId ? extendedByParaId.get(threadParaId)?.parentParaId : void 0;
185
+ const parentId = parentParaId ? commentIdByParaId.get(parentParaId) : void 0;
186
+ if (parentId !== void 0 && parentId !== comment.id) comment.parentId = parentId;
187
+ comments.push(comment);
177
188
  }
178
189
  return comments;
179
190
  }
@@ -0,0 +1,18 @@
1
+ //#region src/docx/commentThreadKey.d.ts
2
+ /**
3
+ * The one rule for "which `w14:paraId` a comment is threaded by".
4
+ *
5
+ * `w15:commentEx/@w15:paraId` and `w15:paraIdParent` name a comment through the
6
+ * `w14:paraId` of its LAST paragraph, never through its `w:id`. Files in the
7
+ * wild leave that paragraph's id off and carry one on an earlier paragraph, so
8
+ * the rule folio reads by is the last paragraph that carries an id.
9
+ *
10
+ * The rule has two readers over two representations: the parser applies it to
11
+ * the `w:p` children of a `w:comment`, the serializer to the model's
12
+ * `Comment.content`. They must not drift — a save that keyed on a different
13
+ * paragraph than the parse would write a `commentsExtended.xml` whose entries
14
+ * belong to other comments — so both call this instead of restating it.
15
+ */
16
+ declare const commentThreadParaId: (paragraphParaIds: Iterable<string | null | undefined>) => string | undefined;
17
+ //#endregion
18
+ export { commentThreadParaId };
@@ -0,0 +1,22 @@
1
+ //#region src/docx/commentThreadKey.ts
2
+ /**
3
+ * The one rule for "which `w14:paraId` a comment is threaded by".
4
+ *
5
+ * `w15:commentEx/@w15:paraId` and `w15:paraIdParent` name a comment through the
6
+ * `w14:paraId` of its LAST paragraph, never through its `w:id`. Files in the
7
+ * wild leave that paragraph's id off and carry one on an earlier paragraph, so
8
+ * the rule folio reads by is the last paragraph that carries an id.
9
+ *
10
+ * The rule has two readers over two representations: the parser applies it to
11
+ * the `w:p` children of a `w:comment`, the serializer to the model's
12
+ * `Comment.content`. They must not drift — a save that keyed on a different
13
+ * paragraph than the parse would write a `commentsExtended.xml` whose entries
14
+ * belong to other comments — so both call this instead of restating it.
15
+ */
16
+ const commentThreadParaId = (paragraphParaIds) => {
17
+ let key;
18
+ for (const paraId of paragraphParaIds) if (paraId) key = paraId;
19
+ return key;
20
+ };
21
+ //#endregion
22
+ export { commentThreadParaId };
@@ -1,5 +1,7 @@
1
- import { resolveRelativePath } from "./relsParser.js";
2
- import { findChildByNamespaceUri, getAttribute, getLocalName, parseNumericAttribute, parseXmlDocument } from "./xmlParser.js";
1
+ import { bytesToDataUrl } from "../utils/base64.js";
2
+ import { PREVIEW_KINDS } from "./previewBudget.js";
3
+ import { resolveRelationshipIdOfType, resolveRelativePath } from "./relsParser.js";
4
+ import { OFFICE_RELATIONSHIP_NAMESPACE_URIS, findChildByNamespaceUri, getAttribute, getAttributeByNamespaceUri, getLocalName, parseNumericAttribute, parseXmlDocument } from "./xmlParser.js";
3
5
  //#region src/docx/diagramPreview.ts
4
6
  const MAX_PREVIEW_SHAPES = 128;
5
7
  const MAX_PREVIEW_PIXELS = 144e4;
@@ -7,21 +9,47 @@ const MAX_PREVIEW_PAINT_PIXELS = MAX_PREVIEW_PIXELS * 4;
7
9
  const DRAWINGML_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.openxmlformats.org/drawingml/2006/main", "http://purl.oclc.org/ooxml/drawingml/main"]);
8
10
  const DIAGRAM_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.openxmlformats.org/drawingml/2006/diagram", "http://purl.oclc.org/ooxml/drawingml/diagram"]);
9
11
  const DIAGRAM_DRAWING_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.microsoft.com/office/drawing/2008/diagram"]);
12
+ const DIAGRAM_DATA_RELATIONSHIP_TYPE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/diagramData";
13
+ const DIAGRAM_DRAWING_RELATIONSHIP_TYPE = "http://schemas.microsoft.com/office/2007/relationships/diagramDrawing";
14
+ const DOCUMENT_PART_PATH = "word/document.xml";
10
15
  const WORD_DRAWING_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing", "http://purl.oclc.org/ooxml/drawingml/wordprocessingDrawing"]);
16
+ /**
17
+ * The preview is a megapixel raster, so both checksums run over megabytes.
18
+ * `for (const byte of bytes)` drives the array iterator protocol once per
19
+ * byte, which profiles as the dominant cost of parsing a SmartArt document;
20
+ * indexed loops and a table-driven CRC produce the same numbers without it.
21
+ */
22
+ const CRC32_TABLE = (() => {
23
+ const table = /* @__PURE__ */ new Uint32Array(256);
24
+ for (let index = 0; index < 256; index += 1) {
25
+ let value = index;
26
+ for (let bit = 0; bit < 8; bit += 1) value = value >>> 1 ^ (value & 1 ? 3988292384 : 0);
27
+ table[index] = value >>> 0;
28
+ }
29
+ return table;
30
+ })();
11
31
  const crc32 = (bytes) => {
12
32
  let crc = 4294967295;
13
- for (const byte of bytes) {
14
- crc ^= byte;
15
- for (let bit = 0; bit < 8; bit += 1) crc = crc >>> 1 ^ (crc & 1 ? 3988292384 : 0);
16
- }
33
+ for (let index = 0; index < bytes.length; index += 1) crc = crc >>> 8 ^ CRC32_TABLE[(crc ^ bytes[index]) & 255];
17
34
  return (crc ^ 4294967295) >>> 0;
18
35
  };
36
+ /**
37
+ * `a` and `b` stay below 2^31 for 5552 iterations from any legal state, so the
38
+ * modulo runs per block rather than per byte.
39
+ */
40
+ const ADLER32_BLOCK = 5552;
19
41
  const adler32 = (bytes) => {
20
42
  let a = 1;
21
43
  let b = 0;
22
- for (const byte of bytes) {
23
- a = (a + byte) % 65521;
24
- b = (b + a) % 65521;
44
+ let index = 0;
45
+ while (index < bytes.length) {
46
+ const end = Math.min(index + ADLER32_BLOCK, bytes.length);
47
+ for (; index < end; index += 1) {
48
+ a += bytes[index];
49
+ b += a;
50
+ }
51
+ a %= 65521;
52
+ b %= 65521;
25
53
  }
26
54
  return b << 16 | a;
27
55
  };
@@ -88,9 +116,8 @@ const previewPng = (width, height, shapes) => {
88
116
  cursor += length;
89
117
  offset += length;
90
118
  }
91
- const output = new Uint8Array(cursor + 4);
92
- output.set(compressed.subarray(0, cursor));
93
- new DataView(output.buffer).setUint32(cursor, adler32(pixels));
119
+ new DataView(compressed.buffer).setUint32(cursor, adler32(pixels));
120
+ const output = compressed;
94
121
  const signature = new Uint8Array([
95
122
  137,
96
123
  80,
@@ -127,13 +154,38 @@ const extent = (drawing) => {
127
154
  height: parseNumericAttribute(value, null, "cy") ?? 0
128
155
  };
129
156
  };
130
- const cachedDiagramShapes = (rels, media) => {
131
- const matches = [...rels.values()].filter((relationship) => relationship.type === "http://schemas.microsoft.com/office/2007/relationships/diagramDrawing");
132
- if (matches.length !== 1 || !matches[0]?.target) return [];
133
- const path = resolveRelativePath("word/document.xml", matches[0].target);
134
- const file = media.get(path);
135
- if (!file?.data) return [];
136
- const root = parseXmlDocument(new TextDecoder().decode(file.data));
157
+ /** Parse the part a relationship of the given type names, by that relationship's id. */
158
+ const partByRelationshipId = ({ rels, media, rId, type }) => {
159
+ const resolved = resolveRelationshipIdOfType(rels, rId ?? void 0, type);
160
+ if (resolved.status !== "resolved" || !resolved.relationship.target) return null;
161
+ const file = media.get(resolveRelativePath(DOCUMENT_PART_PATH, resolved.relationship.target));
162
+ return file?.data ? parseXmlDocument(new TextDecoder().decode(file.data)) : null;
163
+ };
164
+ /**
165
+ * The drawing cache this diagram points at, reached through its own ids.
166
+ *
167
+ * `dgm:relIds/@r:dm` names the data part, and that part's `dsp:dataModelExt`
168
+ * extension names the cached drawing. Scanning the relationship map for a type
169
+ * instead of following the ids reads the wrong diagram whenever a document has
170
+ * more than one, which is why the scan refused outright on a second match:
171
+ * both diagrams in a two-diagram document then got no preview at all.
172
+ */
173
+ const cachedDiagramShapes = (graphicData, rels, media) => {
174
+ const relIds = findChildByNamespaceUri(graphicData, DIAGRAM_NAMESPACE_URIS, "relIds");
175
+ const data = partByRelationshipId({
176
+ rels,
177
+ media,
178
+ rId: getAttributeByNamespaceUri(relIds, OFFICE_RELATIONSHIP_NAMESPACE_URIS, "dm"),
179
+ type: DIAGRAM_DATA_RELATIONSHIP_TYPE
180
+ });
181
+ if (!data) return [];
182
+ const dataModelExt = descendantsByNamespace(data, DIAGRAM_DRAWING_NAMESPACE_URIS, "dataModelExt").at(0);
183
+ const root = partByRelationshipId({
184
+ rels,
185
+ media,
186
+ rId: getAttribute(dataModelExt, null, "relId"),
187
+ type: DIAGRAM_DRAWING_RELATIONSHIP_TYPE
188
+ });
137
189
  if (!root) return [];
138
190
  const shapes = [];
139
191
  for (const shape of descendantsByNamespace(root, DIAGRAM_DRAWING_NAMESPACE_URIS, "sp").slice(0, MAX_PREVIEW_SHAPES)) {
@@ -155,31 +207,39 @@ const cachedDiagramShapes = (rels, media) => {
155
207
  }
156
208
  return shapes;
157
209
  };
210
+ const matchesNamespace = (element, namespaceUris, localName) => element.namespaceUri !== void 0 && namespaceUris.has(element.namespaceUri) && getLocalName(element.name ?? "") === localName;
158
211
  const descendantsByNamespace = (root, namespaceUris, localName) => {
159
212
  const result = [];
160
213
  const visit = (element) => {
161
- if (element.namespaceUri && namespaceUris.has(element.namespaceUri) && getLocalName(element.name ?? "") === localName) result.push(element);
214
+ if (matchesNamespace(element, namespaceUris, localName)) result.push(element);
162
215
  for (const child of element.elements ?? []) if (child.type === "element") visit(child);
163
216
  };
164
217
  visit(root);
165
218
  return result;
166
219
  };
220
+ /** The first match in document order, without walking the rest of the subtree. */
221
+ const firstDescendantByNamespace = (root, namespaceUris, localName) => {
222
+ if (matchesNamespace(root, namespaceUris, localName)) return root;
223
+ for (const child of root.elements ?? []) {
224
+ if (child.type !== "element") continue;
225
+ const found = firstDescendantByNamespace(child, namespaceUris, localName);
226
+ if (found) return found;
227
+ }
228
+ return null;
229
+ };
167
230
  /** Create a deliberately simple, bounded preview; it is never an editable diagram projection. */
168
231
  const parseDiagramPreview = (drawing, rels, media) => {
169
232
  if (!rels || !media) return null;
170
- const graphicData = descendantsByNamespace(drawing, DRAWINGML_NAMESPACE_URIS, "graphicData").at(0);
233
+ const graphicData = firstDescendantByNamespace(drawing, DRAWINGML_NAMESPACE_URIS, "graphicData");
171
234
  if (!graphicData || !DIAGRAM_NAMESPACE_URIS.has(getAttribute(graphicData, null, "uri") ?? "")) return null;
172
235
  const { width, height } = extent(drawing);
173
236
  if (width <= 0 || height <= 0) return null;
174
- const png = previewPng(width, height, cachedDiagramShapes(rels, media));
175
- let binary = "";
176
- for (let offset = 0; offset < png.length; offset += 32768) binary += String.fromCodePoint(...png.subarray(offset, offset + 32768));
177
237
  return {
178
238
  type: "image",
179
239
  rId: "",
180
- src: `data:image/png;base64,${btoa(binary)}`,
181
- mimeType: "image/png",
182
- filename: "smartart-preview.png",
240
+ src: bytesToDataUrl(previewPng(width, height, cachedDiagramShapes(graphicData, rels, media)), PREVIEW_KINDS.smartArt.mimeType),
241
+ mimeType: PREVIEW_KINDS.smartArt.mimeType,
242
+ filename: PREVIEW_KINDS.smartArt.filename,
183
243
  size: {
184
244
  width,
185
245
  height
@@ -1,6 +1,7 @@
1
1
  import { emuToPixels } from "../utils/units.js";
2
2
  import { parseImage, resolveImageData } from "./imageParser.js";
3
3
  import { findAllDeep, findChildByLocalName, findChildrenByLocalName, getAttribute, getChildElements, getLocalName, getTextContent, parseNumericAttribute } from "./xmlParser.js";
4
+ import { escapeXmlAttribute, escapeXmlText } from "@stll/docx-core";
4
5
  //#region src/docx/groupDrawingParser.ts
5
6
  const HEX_COLOR = /^[0-9A-Fa-f]{6}$/u;
6
7
  const DEFAULT_TEXT_COLOR = "000000";
@@ -12,7 +13,6 @@ const CROP_SCALE = 1e5;
12
13
  const MAX_PATH_COMMANDS = 1e4;
13
14
  const MAX_TEXT_CHARACTERS = 2e4;
14
15
  const MAX_SVG_CHARACTERS = 1e6;
15
- const escapeXml = (value) => value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll("\"", "&quot;").replaceAll("'", "&apos;");
16
16
  const numericAttr = (element, name) => {
17
17
  const direct = parseNumericAttribute(element, null, name);
18
18
  if (direct !== void 0) return direct;
@@ -108,7 +108,7 @@ const renderTextBox = (wsp) => {
108
108
  const color = colorFrom(findAllDeep(wsp, "w", "color").at(0) ?? null, DEFAULT_TEXT_COLOR);
109
109
  const fontSize = halfPoints * HALF_POINT_TO_EMU;
110
110
  const maxCharacters = Math.max(1, Math.floor(width / (fontSize * .38)));
111
- const lines = paragraphs.flatMap((paragraph) => wrapLine(getTextContent(paragraph).slice(0, MAX_TEXT_CHARACTERS), maxCharacters)).map(escapeXml);
111
+ const lines = paragraphs.flatMap((paragraph) => wrapLine(getTextContent(paragraph).slice(0, MAX_TEXT_CHARACTERS), maxCharacters)).map(escapeXmlText);
112
112
  if (lines.length === 0) return "";
113
113
  const lineHeight = fontSize * 1.15;
114
114
  const svgFontSize = 1e3;
@@ -131,7 +131,7 @@ const renderPicture = (picture, index, rels, media) => {
131
131
  const visibleWidth = 1 - left - right;
132
132
  const visibleHeight = 1 - top - bottom;
133
133
  if (visibleWidth <= 0 || visibleHeight <= 0) return "";
134
- const image = `<image x="${x - width * left / visibleWidth}" y="${y - height * top / visibleHeight}" width="${width / visibleWidth}" height="${height / visibleHeight}" href="${escapeXml(src)}" preserveAspectRatio="none"/>`;
134
+ const image = `<image x="${x - width * left / visibleWidth}" y="${y - height * top / visibleHeight}" width="${width / visibleWidth}" height="${height / visibleHeight}" href="${escapeXmlAttribute(src)}" preserveAspectRatio="none"/>`;
135
135
  if (left === 0 && top === 0 && right === 0 && bottom === 0) return image;
136
136
  const clipId = `group-picture-${index}`;
137
137
  return `<defs><clipPath id="${clipId}"><rect x="${x}" y="${y}" width="${width}" height="${height}"/></clipPath></defs><g clip-path="url(#${clipId})">${image}</g>`;
@@ -1,4 +1,4 @@
1
- import { sanitizeExternalUrl, sanitizeLinkTarget } from "../utils/urlSecurity.js";
1
+ import { sanitizeExternalUrl } from "../utils/urlSecurity.js";
2
2
  import { RELATIONSHIP_TYPES, resolveRelationshipIdOfType } from "./relsParser.js";
3
3
  import { parseRun } from "./runParser.js";
4
4
  import { getAttribute, getChildElements, getLocalName, mergeXmlnsDeclarations, parseNumericAttribute, parseOnOffAttribute } from "./xmlParser.js";
@@ -63,7 +63,7 @@ function parseHyperlink(node, rels, styles = null, theme = null, media = null, r
63
63
  const tooltip = getAttribute(node, "w", "tooltip");
64
64
  if (tooltip) hyperlink.tooltip = tooltip;
65
65
  const tgtFrame = getAttribute(node, "w", "tgtFrame");
66
- if (tgtFrame) hyperlink.target = sanitizeLinkTarget(tgtFrame);
66
+ if (tgtFrame) hyperlink.target = tgtFrame;
67
67
  if (parseOnOffAttribute(node, "w", "history") === true) hyperlink.history = true;
68
68
  const docLocation = getAttribute(node, "w", "docLocation");
69
69
  if (docLocation) hyperlink.docLocation = docLocation;
@@ -1,6 +1,14 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
2
  import { XmlElement } from "./xmlParser.js";
3
3
  //#region src/docx/imageParser.d.ts
4
+ /**
5
+ * The `a:ext` uri under which Word records that a drawing is decorative. The
6
+ * extension holds `<adec:decorative val="…"/>`; `CT_NonVisualDrawingProps` has
7
+ * no `@decorative` attribute for it to be written as.
8
+ */
9
+ declare const DECORATIVE_EXTENSION_URI = "{C183D7F6-B498-43B3-948B-1728B52AA6E4}";
10
+ /** The namespace the decorative extension's element is bound to. */
11
+ declare const DECORATIVE_NAMESPACE = "http://schemas.microsoft.com/office/drawing/2017/decorative";
4
12
  /**
5
13
  * Resolve image data from relationships and media map
6
14
  *
@@ -89,4 +97,4 @@ declare function getWrapDistancesPx(image: document_d_exports.Image): {
89
97
  */
90
98
  declare function needsTextWrapping(image: document_d_exports.Image): boolean;
91
99
  //#endregion
92
- export { getImageDimensionsPx, getImageHeightPx, getImageWidthPx, getWrapDistancesPx, hasAltText, isBehindText, isDecorativeImage, isFloatingImage, isInFrontOfText, isInlineImage, needsTextWrapping, parseDrawing, parseImage, resolveImageData };
100
+ export { DECORATIVE_EXTENSION_URI, DECORATIVE_NAMESPACE, getImageDimensionsPx, getImageHeightPx, getImageWidthPx, getWrapDistancesPx, hasAltText, isBehindText, isDecorativeImage, isFloatingImage, isInFrontOfText, isInlineImage, needsTextWrapping, parseDrawing, parseImage, resolveImageData };
@@ -3,9 +3,11 @@ import { emuToPixels } from "../utils/units.js";
3
3
  import { sanitizeExternalUrl } from "../utils/urlSecurity.js";
4
4
  import { WRAP_ELEMENT_NAMES, parseAnchorBehindDoc, parsePositionH, parsePositionV, parseWrapElement } from "./drawingUtils.js";
5
5
  import { parseGraphicFrameLocks } from "./graphicFrameLocks.js";
6
+ import { parseNonVisualDrawingNames } from "./nonVisualDrawingProps.js";
6
7
  import { RELATIONSHIP_TYPES, resolveRelationshipIdOfType } from "./relsParser.js";
7
8
  import { isTextBoxDrawing } from "./textBoxParser.js";
8
- import { findByFullName, findChild, getAttribute, getChildElements, parseNumericAttribute, parseOnOffValue } from "./xmlParser.js";
9
+ import { captureVerbatimXml } from "./verbatimCapture.js";
10
+ import { findByFullName, findChild, findChildByNamespaceUri, findChildrenByNamespaceUri, getAttribute, getChildElements, getLocalName, getNamespaceUri, parseNumericAttribute, parseOnOffAttribute, parseOnOffValue } from "./xmlParser.js";
9
11
  //#region src/docx/imageParser.ts
10
12
  /**
11
13
  * Convert rotation value (1/60000 of a degree) to degrees
@@ -64,6 +66,47 @@ function parseEffectExtent(effectExtent) {
64
66
  };
65
67
  }
66
68
  /**
69
+ * The `a:ext` uri under which Word records that a drawing is decorative. The
70
+ * extension holds `<adec:decorative val="…"/>`; `CT_NonVisualDrawingProps` has
71
+ * no `@decorative` attribute for it to be written as.
72
+ */
73
+ const DECORATIVE_EXTENSION_URI = "{C183D7F6-B498-43B3-948B-1728B52AA6E4}";
74
+ /** The namespace the decorative extension's element is bound to. */
75
+ const DECORATIVE_NAMESPACE = "http://schemas.microsoft.com/office/drawing/2017/decorative";
76
+ /**
77
+ * `wp:docPr`'s extension list is DrawingML's (`a:extLst` holding `a:ext`), in
78
+ * the Transitional or the Strict namespace. Matching on the local name alone
79
+ * would take an `extLst` some other namespace owns and replay its children
80
+ * inside an `a:extLst`, which is a different container than the source wrote.
81
+ */
82
+ const DRAWINGML_NAMESPACE_URIS = /* @__PURE__ */ new Set(["http://schemas.openxmlformats.org/drawingml/2006/main", "http://purl.oclc.org/ooxml/drawingml/main"]);
83
+ /**
84
+ * Read the `wp:docPr` extension list.
85
+ *
86
+ * Only the decorative extension is modeled. The rest — a creation id, a
87
+ * local-DPI hint, whatever a later Word writes — are captured as they were
88
+ * written so the save path can put them back: an extension folio drops is a
89
+ * fact the document had and no longer does.
90
+ */
91
+ const parseDocPropsExtensions = (docPr) => {
92
+ const extLst = findChildByNamespaceUri(docPr, DRAWINGML_NAMESPACE_URIS, "extLst");
93
+ if (!extLst) return { other: [] };
94
+ const result = { other: [] };
95
+ for (const ext of findChildrenByNamespaceUri(extLst, DRAWINGML_NAMESPACE_URIS, "ext")) {
96
+ if (getAttribute(ext, null, "uri")?.toUpperCase() !== "{C183D7F6-B498-43B3-948B-1728B52AA6E4}") {
97
+ result.other.push(captureVerbatimXml(ext));
98
+ continue;
99
+ }
100
+ const flag = getChildElements(ext).find((child) => getLocalName(child.name) === "decorative" && getNamespaceUri(child) === "http://schemas.microsoft.com/office/drawing/2017/decorative");
101
+ if (!flag) {
102
+ result.other.push(captureVerbatimXml(ext));
103
+ continue;
104
+ }
105
+ result.decorative = parseOnOffValue(getAttribute(flag, null, "val")) ?? true;
106
+ }
107
+ return result;
108
+ };
109
+ /**
67
110
  * Parse document properties (wp:docPr)
68
111
  *
69
112
  * @param docPr - wp:docPr element
@@ -72,18 +115,17 @@ function parseEffectExtent(effectExtent) {
72
115
  function parseDocProps(docPr) {
73
116
  if (!docPr) return {};
74
117
  const id = getAttribute(docPr, null, "id");
75
- const name = getAttribute(docPr, null, "name");
76
- const descr = getAttribute(docPr, null, "descr");
77
- const title = getAttribute(docPr, null, "title");
78
- const decorative = parseOnOffValue(getAttribute(docPr, null, "decorative")) === true;
118
+ const names = parseNonVisualDrawingNames(docPr);
119
+ const hidden = parseOnOffAttribute(docPr, null, "hidden");
120
+ const { decorative, other: docPrExtensions } = parseDocPropsExtensions(docPr);
79
121
  const hlinkClickEl = findChild(docPr, "a", "hlinkClick");
80
122
  const hlinkRId = hlinkClickEl ? getAttribute(hlinkClickEl, "r", "id") : null;
81
123
  return {
82
124
  ...id != null ? { id } : {},
83
- ...name != null ? { name } : {},
84
- ...descr != null ? { alt: descr } : {},
85
- ...title != null ? { title } : {},
86
- ...decorative ? { decorative } : {},
125
+ ...names,
126
+ ...decorative === void 0 ? {} : { decorative },
127
+ ...hidden === void 0 ? {} : { hidden },
128
+ ...docPrExtensions.length > 0 ? { docPrExtensions } : {},
87
129
  ...hlinkRId != null ? { hlinkRId } : {}
88
130
  };
89
131
  }
@@ -329,7 +371,9 @@ function parseInline(inlineEl, rels, media) {
329
371
  if (props.name !== void 0) image.docPrName = props.name;
330
372
  if (props.alt !== void 0) image.alt = props.alt;
331
373
  if (props.title !== void 0) image.title = props.title;
332
- if (props.decorative) image.decorative = true;
374
+ if (props.decorative !== void 0) image.decorative = props.decorative;
375
+ if (props.hidden !== void 0) image.hidden = props.hidden;
376
+ if (props.docPrExtensions !== void 0) image.docPrExtensions = props.docPrExtensions;
333
377
  const safeSrc = sanitizeImageSrc(imageData.src);
334
378
  if (safeSrc) image.src = safeSrc;
335
379
  if (imageData.mimeType) image.mimeType = imageData.mimeType;
@@ -402,7 +446,9 @@ function parseAnchor(anchorEl, rels, media) {
402
446
  if (props.name !== void 0) image.docPrName = props.name;
403
447
  if (props.alt !== void 0) image.alt = props.alt;
404
448
  if (props.title !== void 0) image.title = props.title;
405
- if (props.decorative) image.decorative = true;
449
+ if (props.decorative !== void 0) image.decorative = props.decorative;
450
+ if (props.hidden !== void 0) image.hidden = props.hidden;
451
+ if (props.docPrExtensions !== void 0) image.docPrExtensions = props.docPrExtensions;
406
452
  const safeSrc = sanitizeImageSrc(imageData.src);
407
453
  if (safeSrc) image.src = safeSrc;
408
454
  if (imageData.mimeType) image.mimeType = imageData.mimeType;
@@ -537,4 +583,4 @@ function needsTextWrapping(image) {
537
583
  ].includes(image.wrap.type);
538
584
  }
539
585
  //#endregion
540
- export { getImageDimensionsPx, getImageHeightPx, getImageWidthPx, getWrapDistancesPx, hasAltText, isBehindText, isDecorativeImage, isFloatingImage, isInFrontOfText, isInlineImage, needsTextWrapping, parseDrawing, parseImage, resolveImageData };
586
+ export { DECORATIVE_EXTENSION_URI, DECORATIVE_NAMESPACE, getImageDimensionsPx, getImageHeightPx, getImageWidthPx, getWrapDistancesPx, hasAltText, isBehindText, isDecorativeImage, isFloatingImage, isInFrontOfText, isInlineImage, needsTextWrapping, parseDrawing, parseImage, resolveImageData };
@@ -12,7 +12,20 @@ declare const isDrawingRawXmlMode: (value: unknown) => value is document_d_expor
12
12
  declare const EDITED_PREVIEW_FINGERPRINT = "editedPreview";
13
13
  /** Fingerprints modeled image fields which make raw DrawingML stale when edited. */
14
14
  declare const imageRawXmlFingerprint: (image: document_d_exports.Image) => string;
15
- /** Editable raw drawing XML can replay only while its modeled projection is unchanged. */
15
+ /**
16
+ * Whether the serializer writes the captured raw XML back.
17
+ *
18
+ * For an unclassified drawing the capture is a cache of the model, so it
19
+ * replays only while the two agree. For a classified one the capture *is* the
20
+ * drawing: a preserve-only drawing has a placeholder image, and a preview-only
21
+ * group has a raster render of shapes the model cannot hold. Regenerating
22
+ * either writes something the source never contained — for the group, at best
23
+ * the one child picture the rasterizer saw first under that child's
24
+ * relationship, in place of the whole group. So a classified capture is always
25
+ * written back; a stale fingerprint means the edit that made it stale is lost,
26
+ * which {@link classifyDrawingSafety} reports as `opaque`, and never that
27
+ * folio may build a replacement.
28
+ */
16
29
  declare const canReplayEditableImageRawXml: (drawing: document_d_exports.DrawingContent) => boolean;
17
30
  /**
18
31
  * What a drawing costs the document when Folio writes it back.
@@ -26,10 +26,29 @@ const EDITED_PREVIEW_FINGERPRINT = "editedPreview";
26
26
  const editableImageProjection = ({ id: _id, rId: _rId, src: _src, mimeType: _mimeType, filename: _filename, ...image }) => image;
27
27
  /** Fingerprints modeled image fields which make raw DrawingML stale when edited. */
28
28
  const imageRawXmlFingerprint = (image) => canonicalJson(editableImageProjection(image));
29
- /** Editable raw drawing XML can replay only while its modeled projection is unchanged. */
29
+ /** Whether the modeled image still says what the captured raw drawing says. */
30
+ const modelAgreesWithRawXml = (drawing) => drawing.rawImageFingerprint === void 0 || drawing.rawImageFingerprint === imageRawXmlFingerprint(drawing.image);
31
+ /**
32
+ * Whether the serializer writes the captured raw XML back.
33
+ *
34
+ * For an unclassified drawing the capture is a cache of the model, so it
35
+ * replays only while the two agree. For a classified one the capture *is* the
36
+ * drawing: a preserve-only drawing has a placeholder image, and a preview-only
37
+ * group has a raster render of shapes the model cannot hold. Regenerating
38
+ * either writes something the source never contained — for the group, at best
39
+ * the one child picture the rasterizer saw first under that child's
40
+ * relationship, in place of the whole group. So a classified capture is always
41
+ * written back; a stale fingerprint means the edit that made it stale is lost,
42
+ * which {@link classifyDrawingSafety} reports as `opaque`, and never that
43
+ * folio may build a replacement.
44
+ */
30
45
  const canReplayEditableImageRawXml = (drawing) => {
31
- if (drawing.rawXmlMode === DRAWING_RAW_XML_MODES.PRESERVE_ONLY) return true;
32
- return drawing.rawImageFingerprint === void 0 || drawing.rawImageFingerprint === imageRawXmlFingerprint(drawing.image);
46
+ switch (drawing.rawXmlMode) {
47
+ case DRAWING_RAW_XML_MODES.PRESERVE_ONLY:
48
+ case DRAWING_RAW_XML_MODES.PREVIEW_ONLY: return true;
49
+ case void 0: return modelAgreesWithRawXml(drawing);
50
+ default: return drawing;
51
+ }
33
52
  };
34
53
  /**
35
54
  * What a drawing costs the document when Folio writes it back.
@@ -60,9 +79,14 @@ const canRegenerateDrawing = (drawing) => drawing.image.rId !== void 0;
60
79
  */
61
80
  const classifyDrawingSafety = (drawing) => {
62
81
  if (drawing.rawXml === void 0) return DRAWING_SAFETY_CLASSES.NATIVE;
63
- if (canReplayEditableImageRawXml(drawing)) return DRAWING_SAFETY_CLASSES.REPLAYABLE;
64
- if (drawing.rawXmlMode === DRAWING_RAW_XML_MODES.PREVIEW_ONLY) return DRAWING_SAFETY_CLASSES.OPAQUE;
65
- return canRegenerateDrawing(drawing) ? DRAWING_SAFETY_CLASSES.NATIVE : DRAWING_SAFETY_CLASSES.OPAQUE;
82
+ switch (drawing.rawXmlMode) {
83
+ case DRAWING_RAW_XML_MODES.PRESERVE_ONLY: return DRAWING_SAFETY_CLASSES.REPLAYABLE;
84
+ case DRAWING_RAW_XML_MODES.PREVIEW_ONLY: return modelAgreesWithRawXml(drawing) ? DRAWING_SAFETY_CLASSES.REPLAYABLE : DRAWING_SAFETY_CLASSES.OPAQUE;
85
+ case void 0:
86
+ if (modelAgreesWithRawXml(drawing)) return DRAWING_SAFETY_CLASSES.REPLAYABLE;
87
+ return canRegenerateDrawing(drawing) ? DRAWING_SAFETY_CLASSES.NATIVE : DRAWING_SAFETY_CLASSES.OPAQUE;
88
+ default: return drawing;
89
+ }
66
90
  };
67
91
  //#endregion
68
92
  export { DRAWING_SAFETY_CLASSES, EDITED_PREVIEW_FINGERPRINT, allowsDirectDrawingEdit, canReplayEditableImageRawXml, classifyDrawingSafety, imageRawXmlFingerprint, isDrawingRawXmlMode };