@stll/folio-core 0.43.0 → 0.45.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. package/dist/ai-edits/__fixtures__/paragraphs.js +2 -2
  2. package/dist/ai-edits/headless.js +7 -5
  3. package/dist/ai-edits/index.d.ts +2 -2
  4. package/dist/ai-edits/index.js +2 -2
  5. package/dist/ai-edits/snapshot.js +13 -9
  6. package/dist/compare/content-alignment.js +94 -54
  7. package/dist/compare/inline-atoms.js +34 -20
  8. package/dist/compare/style-resources.js +6 -0
  9. package/dist/content-controls/mutateContentControls.js +4 -2
  10. package/dist/display-list/dom/renderDisplayListToDom.js +8 -8
  11. package/dist/document-operations.js +14 -3
  12. package/dist/docx/appVersionNormalization.d.ts +0 -18
  13. package/dist/docx/blockContentParser.js +8 -0
  14. package/dist/docx/blockRangeMarkers.d.ts +36 -0
  15. package/dist/docx/blockRangeMarkers.js +59 -0
  16. package/dist/docx/bookmarkParser.d.ts +2 -20
  17. package/dist/docx/bookmarkParser.js +6 -30
  18. package/dist/docx/borderParser.d.ts +13 -0
  19. package/dist/docx/borderParser.js +71 -0
  20. package/dist/docx/builtInStyles.d.ts +165 -0
  21. package/dist/docx/builtInStyles.js +239 -0
  22. package/dist/docx/commentIdNormalization.d.ts +3 -1
  23. package/dist/docx/commentIdNormalization.js +18 -1
  24. package/dist/docx/commentParser.d.ts +2 -1
  25. package/dist/docx/commentParser.js +80 -42
  26. package/dist/docx/commentReferenceNormalization.d.ts +4 -1
  27. package/dist/docx/commentReferenceNormalization.js +23 -14
  28. package/dist/docx/commentThreadKey.d.ts +18 -0
  29. package/dist/docx/commentThreadKey.js +22 -0
  30. package/dist/docx/danglingRelationshipReferences.d.ts +15 -0
  31. package/dist/docx/danglingRelationshipReferences.js +30 -0
  32. package/dist/docx/defaultParagraphStyle.d.ts +18 -1
  33. package/dist/docx/defaultParagraphStyle.js +23 -1
  34. package/dist/docx/diagramPreview.js +87 -27
  35. package/dist/docx/documentParser.d.ts +2 -1
  36. package/dist/docx/documentParser.js +2 -2
  37. package/dist/docx/drawingUtils.d.ts +8 -1
  38. package/dist/docx/drawingUtils.js +12 -3
  39. package/dist/docx/fieldParser.js +3 -5
  40. package/dist/docx/footnoteParser.d.ts +3 -2
  41. package/dist/docx/footnoteParser.js +19 -4
  42. package/dist/docx/groupDrawingParser.js +4 -4
  43. package/dist/docx/headerFooterRefParser.d.ts +4 -3
  44. package/dist/docx/headerFooterRefParser.js +42 -12
  45. package/dist/docx/headerFooterReferenceNormalization.d.ts +4 -1
  46. package/dist/docx/headerFooterReferenceNormalization.js +5 -1
  47. package/dist/docx/hyperlinkParser.js +13 -17
  48. package/dist/docx/imageParser.d.ts +10 -2
  49. package/dist/docx/imageParser.js +80 -30
  50. package/dist/docx/imageRawXml.d.ts +14 -1
  51. package/dist/docx/imageRawXml.js +35 -11
  52. package/dist/docx/markupRangeMarker.d.ts +15 -0
  53. package/dist/docx/markupRangeMarker.js +44 -0
  54. package/dist/docx/mathToMathml.js +12 -14
  55. package/dist/docx/nonVisualDrawingProps.d.ts +34 -0
  56. package/dist/docx/nonVisualDrawingProps.js +46 -0
  57. package/dist/docx/noteReferenceStyles.d.ts +29 -0
  58. package/dist/docx/noteReferenceStyles.js +70 -0
  59. package/dist/docx/numberingReferenceNormalization.d.ts +4 -1
  60. package/dist/docx/numberingReferenceNormalization.js +20 -1
  61. package/dist/docx/paraIdRangeNormalization.d.ts +0 -19
  62. package/dist/docx/paragraphParser.js +66 -99
  63. package/dist/docx/paragraphPropertySource.js +1 -0
  64. package/dist/docx/paragraphTextBoxEnrichment.js +3 -0
  65. package/dist/docx/paragraphTraversal.d.ts +37 -1
  66. package/dist/docx/paragraphTraversal.js +84 -1
  67. package/dist/docx/parseContext.d.ts +37 -0
  68. package/dist/docx/parseContext.js +67 -0
  69. package/dist/docx/parseWarningMessage.d.ts +6 -0
  70. package/dist/docx/parseWarningMessage.js +44 -0
  71. package/dist/docx/parser.js +83 -29
  72. package/dist/docx/previewBudget.d.ts +64 -0
  73. package/dist/docx/previewBudget.js +88 -0
  74. package/dist/docx/relsParser.d.ts +28 -11
  75. package/dist/docx/relsParser.js +26 -13
  76. package/dist/docx/revisionIdNormalization.js +96 -10
  77. package/dist/docx/rezip.js +80 -40
  78. package/dist/docx/runConsolidator.js +1 -2
  79. package/dist/docx/runParser.d.ts +8 -1
  80. package/dist/docx/runParser.js +30 -48
  81. package/dist/docx/sdtPropertiesPatch.js +24 -18
  82. package/dist/docx/sectionParser.d.ts +2 -1
  83. package/dist/docx/sectionParser.js +21 -65
  84. package/dist/docx/sectionReferenceHistory.js +2 -2
  85. package/dist/docx/selectiveSave.js +6 -6
  86. package/dist/docx/serializer/blockSdtSerializer.js +38 -26
  87. package/dist/docx/serializer/borderSerializer.d.ts +2 -3
  88. package/dist/docx/serializer/borderSerializer.js +13 -12
  89. package/dist/docx/serializer/commentSerializer.d.ts +41 -16
  90. package/dist/docx/serializer/commentSerializer.js +82 -72
  91. package/dist/docx/serializer/documentSerializer.d.ts +1 -5
  92. package/dist/docx/serializer/documentSerializer.js +6 -16
  93. package/dist/docx/serializer/fontTableSerializer.js +6 -6
  94. package/dist/docx/serializer/headerFooterSerializer.js +10 -5
  95. package/dist/docx/serializer/markupRangeAttributes.d.ts +8 -0
  96. package/dist/docx/serializer/markupRangeAttributes.js +24 -0
  97. package/dist/docx/serializer/noteSerializer.js +5 -0
  98. package/dist/docx/serializer/numberingSerializer.js +7 -6
  99. package/dist/docx/serializer/paragraphSerializer.d.ts +1 -5
  100. package/dist/docx/serializer/paragraphSerializer.js +47 -52
  101. package/dist/docx/serializer/partNamespaces.js +2 -2
  102. package/dist/docx/serializer/runSerializer.js +57 -31
  103. package/dist/docx/serializer/sectionPropertiesSerializer.js +11 -10
  104. package/dist/docx/serializer/settingsSerializer.js +4 -3
  105. package/dist/docx/serializer/stylesSerializer.js +6 -6
  106. package/dist/docx/serializer/tableSerializer.js +37 -21
  107. package/dist/docx/serializer/textFormattingSerializer.d.ts +2 -3
  108. package/dist/docx/serializer/textFormattingSerializer.js +29 -28
  109. package/dist/docx/serializer/themeSerializer.js +6 -6
  110. package/dist/docx/serializer/trackedChangeAttributes.js +2 -2
  111. package/dist/docx/serializer/xmlUtils.d.ts +1 -2
  112. package/dist/docx/serializer/xmlUtils.js +1 -13
  113. package/dist/docx/server/boundedArchive.d.ts +12 -0
  114. package/dist/docx/server/boundedArchive.js +20 -1
  115. package/dist/docx/server/build.js +8 -1
  116. package/dist/docx/server/createBilingualDocument.js +10 -18
  117. package/dist/docx/server/extractDocxText.js +3 -4
  118. package/dist/docx/server/validateDocxConformance.js +22 -1
  119. package/dist/docx/shadingParser.d.ts +6 -0
  120. package/dist/docx/shadingParser.js +32 -0
  121. package/dist/docx/shapeParser.js +10 -8
  122. package/dist/docx/styleParser.js +13 -87
  123. package/dist/docx/styleReferenceResolution.d.ts +36 -0
  124. package/dist/docx/styleReferenceResolution.js +51 -0
  125. package/dist/docx/tableLook.d.ts +57 -0
  126. package/dist/docx/tableLook.js +63 -0
  127. package/dist/docx/tableParser.d.ts +7 -9
  128. package/dist/docx/tableParser.js +64 -110
  129. package/dist/docx/textBoxParser.js +11 -6
  130. package/dist/docx/trackedMoveRangeNormalization.d.ts +3 -1
  131. package/dist/docx/trackedMoveRangeNormalization.js +11 -21
  132. package/dist/docx/transitionalSpelling.d.ts +13 -2
  133. package/dist/docx/transitionalSpelling.js +23 -1
  134. package/dist/docx/unzip.d.ts +23 -0
  135. package/dist/docx/unzip.js +32 -22
  136. package/dist/docx/verbatimCapture.js +5 -12
  137. package/dist/docx/vmlImageParser.js +5 -4
  138. package/dist/docx/vmlPreview.d.ts +1 -3
  139. package/dist/docx/vmlPreview.js +2 -30
  140. package/dist/docx/watermarkParser.js +2 -2
  141. package/dist/docx/xmlParser.d.ts +38 -33
  142. package/dist/docx/xmlParser.js +92 -47
  143. package/dist/docx/xmlResourceLimits.d.ts +89 -9
  144. package/dist/docx/xmlResourceLimits.js +105 -24
  145. package/dist/internal/pageBreakRunSourceDescendantIndex.js +2 -1
  146. package/dist/internal/paragraphFormattingSerialization.d.ts +2 -3
  147. package/dist/internal/paragraphFormattingSerialization.js +29 -8
  148. package/dist/layout-bridge/convert/footnoteLayout.js +2 -7
  149. package/dist/layout-engine/index.d.ts +2 -2
  150. package/dist/layout-engine/index.js +2 -2
  151. package/dist/layout-engine/measure/measureBlocks.js +1 -6
  152. package/dist/layout-engine/types.d.ts +8 -2
  153. package/dist/layout-engine/types.js +35 -2
  154. package/dist/layout-painter/renderImage.js +4 -3
  155. package/dist/layout-painter/renderParagraph.js +4 -3
  156. package/dist/managers/autoSaveCodec.js +2 -8
  157. package/dist/markdown/images.js +1 -4
  158. package/dist/markdown/index.js +1 -1
  159. package/dist/markdown/internals.d.ts +6 -1
  160. package/dist/markdown/internals.js +14 -1
  161. package/dist/markdown/renderBlock.js +35 -21
  162. package/dist/markdown/renderParagraph.js +14 -5
  163. package/dist/markdown/renderRuns.js +4 -3
  164. package/dist/markdown/renderTable.js +4 -3
  165. package/dist/markdown/trailers.js +41 -7
  166. package/dist/markdown/types.d.ts +3 -7
  167. package/dist/prosemirror/attrs/index.js +71 -5
  168. package/dist/prosemirror/bookmarkBoundaryAttrs.d.ts +11 -1
  169. package/dist/prosemirror/bookmarkBoundaryAttrs.js +18 -3
  170. package/dist/prosemirror/commands/image.js +1 -0
  171. package/dist/prosemirror/commands/index.d.ts +3 -3
  172. package/dist/prosemirror/commands/index.js +2 -2
  173. package/dist/prosemirror/commands/paragraph.d.ts +3 -3
  174. package/dist/prosemirror/commands/paragraph.js +2 -2
  175. package/dist/prosemirror/commentIdAllocator.js +2 -7
  176. package/dist/prosemirror/conversion/fromProseDoc.js +197 -68
  177. package/dist/prosemirror/conversion/toProseDoc.d.ts +1 -14
  178. package/dist/prosemirror/conversion/toProseDoc.js +458 -335
  179. package/dist/prosemirror/extensions/core/ParagraphExtension.d.ts +14 -1
  180. package/dist/prosemirror/extensions/core/ParagraphExtension.js +11 -6
  181. package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -3
  182. package/dist/prosemirror/extensions/features/PasteCleanupExtension.d.ts +4 -1
  183. package/dist/prosemirror/extensions/features/PasteCleanupExtension.js +6 -2
  184. package/dist/prosemirror/extensions/features/pastedHeadingStyles.d.ts +7 -0
  185. package/dist/prosemirror/extensions/features/pastedHeadingStyles.js +74 -0
  186. package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +2 -3
  187. package/dist/prosemirror/extensions/marks/markUtils.d.ts +11 -3
  188. package/dist/prosemirror/extensions/marks/markUtils.js +98 -19
  189. package/dist/prosemirror/extensions/nodes/BookmarkBoundaryExtension.js +7 -3
  190. package/dist/prosemirror/extensions/nodes/ImageExtension.js +6 -1
  191. package/dist/prosemirror/extensions/nodes/ShapeExtension.js +8 -2
  192. package/dist/prosemirror/extensions/nodes/TableExtension.js +15 -1
  193. package/dist/prosemirror/extensions/nodes/TextBoxExtension.js +8 -4
  194. package/dist/prosemirror/extensions/types.d.ts +2 -2
  195. package/dist/prosemirror/index.d.ts +3 -3
  196. package/dist/prosemirror/index.js +3 -3
  197. package/dist/prosemirror/insertOperations.d.ts +9 -2
  198. package/dist/prosemirror/insertOperations.js +9 -4
  199. package/dist/prosemirror/paragraphFormattingProvenance.d.ts +162 -0
  200. package/dist/prosemirror/paragraphFormattingProvenance.js +115 -0
  201. package/dist/prosemirror/plugins/documentStyles.d.ts +9 -1
  202. package/dist/prosemirror/plugins/documentStyles.js +11 -1
  203. package/dist/prosemirror/plugins/index.d.ts +2 -2
  204. package/dist/prosemirror/plugins/index.js +2 -2
  205. package/dist/prosemirror/plugins/revisionIds.d.ts +11 -2
  206. package/dist/prosemirror/plugins/revisionIds.js +21 -6
  207. package/dist/prosemirror/runFormattingReconciliation.js +3 -2
  208. package/dist/prosemirror/runStyleFormatting.d.ts +1 -1
  209. package/dist/prosemirror/schema/nodes.d.ts +81 -1
  210. package/dist/prosemirror/styles/resolvedStyleAttrs.js +2 -0
  211. package/dist/prosemirror/styles/styleResolver.d.ts +9 -0
  212. package/dist/prosemirror/styles/styleResolver.js +12 -0
  213. package/dist/style-engine/styleEngine.d.ts +3 -0
  214. package/dist/style-engine/styleEngine.js +3 -0
  215. package/dist/style-sets/extract.js +1 -23
  216. package/dist/style-sets/stellaStyle.js +46 -39
  217. package/dist/style-sets/styleSetNormalization.d.ts +19 -0
  218. package/dist/style-sets/styleSetNormalization.js +99 -0
  219. package/dist/types/content.d.ts +2 -2
  220. package/dist/utils/base64.d.ts +36 -0
  221. package/dist/utils/base64.js +40 -0
  222. package/dist/utils/clipboard.js +2 -1
  223. package/dist/utils/createDocument.js +145 -20
  224. package/dist/utils/headingCollector.d.ts +8 -5
  225. package/dist/utils/headingCollector.js +23 -25
  226. package/dist/utils/tableOfContentsStyle.js +9 -2
  227. package/dist/utils/units.d.ts +10 -1
  228. package/dist/utils/units.js +12 -1
  229. package/dist/utils/urlSecurity.d.ts +8 -2
  230. package/dist/utils/urlSecurity.js +21 -3
  231. package/package.json +2 -2
  232. package/dist/docx/textWhitespace.d.ts +0 -4
  233. package/dist/docx/textWhitespace.js +0 -4
  234. package/dist/layout-bridge/engine/tableWidthUtils.d.ts +0 -6
  235. package/dist/layout-bridge/engine/tableWidthUtils.js +0 -25
  236. package/dist/markdown/headings.d.ts +0 -13
  237. package/dist/markdown/headings.js +0 -20
@@ -0,0 +1,239 @@
1
+ import { BUILT_IN_DEFAULT_PARAGRAPH_STYLE_NAME, resolveDefaultParagraphStyle } from "./defaultParagraphStyle.js";
2
+ //#region src/docx/builtInStyles.ts
3
+ /**
4
+ * The tenth `w:outlineLvl` value. 17.3.1.20: "the val attribute … can be from
5
+ * 0 to 9, where 9 specifically indicates that there is no outline level
6
+ * specifically applied to this paragraph." It is a deliberate "not a heading",
7
+ * not a tenth level. Every range test goes through
8
+ * {@link isHeadingOutlineLevel} so the reserved value keeps one meaning across
9
+ * the codebase.
10
+ *
11
+ * The same clause adds that an omitted element "is assumed to be 9". That
12
+ * default cannot be applied to a *style* definition, because 17.7.1 tells
13
+ * producers not to write a property "already been set by a previous level of
14
+ * the style hierarchy": a document that names a style `heading 1` and omits
15
+ * the level is inheriting the consumer's built-in definition, which carries
16
+ * level 0. An absent level therefore means "unspecified, ask the name", and
17
+ * only a written 9 means body text.
18
+ */
19
+ const BODY_TEXT_OUTLINE_LEVEL = 9;
20
+ /** The highest `w:outlineLvl` that still names a heading (outline level nine). */
21
+ const MAX_HEADING_OUTLINE_LEVEL = 8;
22
+ /** True when an outline level names a heading rather than body text. */
23
+ const isHeadingOutlineLevel = (level) => typeof level === "number" && Number.isInteger(level) && level >= 0 && level <= MAX_HEADING_OUTLINE_LEVEL;
24
+ /**
25
+ * Compare style names the way producers actually write them. The corpus shows
26
+ * both `heading 1` (Annex L, 94.8%) and `Heading 1` (5.2%), and LibreOffice
27
+ * drops the space entirely (`Heading1`, `IntenseQuote`), so case and
28
+ * whitespace are the tolerance. A name is otherwise matched whole: a style a
29
+ * Czech template calls `Nadpis 1` stays a custom style.
30
+ */
31
+ const normalizeStyleName = (name) => name.trim().toLowerCase().replace(/\s+/gu, "");
32
+ /**
33
+ * The `w:name` Word itself writes for each built-in, and therefore the
34
+ * spelling every style table folio authors must use. One owner: a style set and
35
+ * the classifier that reads it cannot drift apart if both name the same
36
+ * constant.
37
+ *
38
+ * Word is not uniformly cased and guessing gets it wrong, so each value is the
39
+ * spelling that dominates Microsoft Word output in the public corpus:
40
+ * `footnote text` 375 against 49 `Footnote Text`, `footer` 1145 against 2,
41
+ * `caption` 390 against 60 — but `Body Text` 424 against 2, `Title` 621
42
+ * against 2, and the auto-generated linked character styles (`Footnote Text
43
+ * Char` 248, `Endnote Text Char` 116) title-cased without exception.
44
+ * {@link normalizeStyleName} makes matching tolerant of all of it; this map is
45
+ * about what folio *writes*.
46
+ */
47
+ const BUILT_IN_STYLE_NAME = {
48
+ /** 4,515 Word occurrences against 3 lowercase. */
49
+ normal: BUILT_IN_DEFAULT_PARAGRAPH_STYLE_NAME,
50
+ bodyText: "Body Text",
51
+ title: "Title",
52
+ subtitle: "Subtitle",
53
+ quote: "Quote",
54
+ intenseQuote: "Intense Quote",
55
+ listParagraph: "List Paragraph",
56
+ tocHeading: "TOC Heading",
57
+ caption: "caption",
58
+ header: "header",
59
+ footer: "footer",
60
+ footnoteText: "footnote text",
61
+ commentReference: "annotation reference",
62
+ footnoteReference: "footnote reference",
63
+ footnoteTextChar: "Footnote Text Char",
64
+ endnoteText: "endnote text",
65
+ endnoteReference: "endnote reference",
66
+ endnoteTextChar: "Endnote Text Char",
67
+ hyperlink: "Hyperlink",
68
+ defaultParagraphFont: "Default Paragraph Font",
69
+ noList: "No List",
70
+ normalTable: "Normal Table",
71
+ tableGrid: "Table Grid"
72
+ };
73
+ /**
74
+ * The name of a built-in heading, from its zero-based outline level.
75
+ * Lowercase: 1,066 Word occurrences of `heading 1` against 13 `Heading 1`, and
76
+ * Annex L writes the latent-style exceptions the same way.
77
+ */
78
+ const builtInHeadingStyleName = (outlineLevel) => `heading ${outlineLevel + 1}`;
79
+ /** The name of a built-in TOC entry style, from its one-based level (`toc 1`). */
80
+ const builtInTableOfContentsStyleName = (level) => `toc ${level}`;
81
+ /**
82
+ * `heading 1`…`heading 9` against an already-normalised name. Word's built-in
83
+ * names stop at nine, matching `w:outlineLvl`'s nine heading values, so
84
+ * `Cmsor10` → `Címsor 10` is a user style rather than a tenth built-in.
85
+ */
86
+ const BUILT_IN_HEADING_NAME = /^heading(?<level>[1-9])$/u;
87
+ /** `toc 1`…`toc 9`, the styles a `TOC` field writes its entries in. */
88
+ const BUILT_IN_TABLE_OF_CONTENTS_NAME = /^toc(?<level>[1-9])$/u;
89
+ /**
90
+ * The outline level a built-in heading *name* implies (zero-based, so
91
+ * `heading 1` is 0), or undefined when the name is not a built-in heading.
92
+ */
93
+ const headingOutlineLevelFromStyleName = (name) => {
94
+ if (name === void 0) return;
95
+ const level = BUILT_IN_HEADING_NAME.exec(normalizeStyleName(name))?.groups?.["level"];
96
+ return level === void 0 ? void 0 : Number.parseInt(level, 10) - 1;
97
+ };
98
+ /**
99
+ * Every spelling a producer might write, mapped back to the canonical one, so
100
+ * a caller compares against {@link BUILT_IN_STYLE_NAME} rather than against a
101
+ * normalised form it would have to spell a second time.
102
+ */
103
+ const CANONICAL_BY_NORMALIZED = new Map(Object.values(BUILT_IN_STYLE_NAME).map((name) => [normalizeStyleName(name), name]));
104
+ const asBuiltInName = (normalized) => CANONICAL_BY_NORMALIZED.get(normalized);
105
+ /**
106
+ * The outline level a style chain sets: the style's own `w:outlineLvl`, else
107
+ * the nearest ancestor's (17.7.1). `styleParser` already flattens `w:basedOn`
108
+ * for a parsed package, but a style set built in memory carries the raw chain,
109
+ * so the walk keeps both kinds of input on the same answer. `seen` guards the
110
+ * circular `basedOn` a malformed package can contain.
111
+ */
112
+ const inheritedOutlineLevel = (style, styleById) => {
113
+ const seen = /* @__PURE__ */ new Set();
114
+ let current = style;
115
+ while (current && !seen.has(current.styleId)) {
116
+ seen.add(current.styleId);
117
+ if (current.pPr?.outlineLevel !== void 0) return current.pPr.outlineLevel;
118
+ current = current.basedOn === void 0 ? void 0 : styleById.get(current.basedOn);
119
+ }
120
+ };
121
+ const createBuiltInStyleIndex = (styles, docDefaults) => {
122
+ const styleIdByHeadingLevel = /* @__PURE__ */ new Map();
123
+ const styleIdByTableOfContentsLevel = /* @__PURE__ */ new Map();
124
+ const styleIdByBuiltInName = /* @__PURE__ */ new Map();
125
+ const paragraphStyles = [];
126
+ const paragraphStyleById = /* @__PURE__ */ new Map();
127
+ const styleById = /* @__PURE__ */ new Map();
128
+ for (const style of styles) {
129
+ styleById.set(style.styleId, style);
130
+ if (style.type === "paragraph") {
131
+ paragraphStyles.push(style);
132
+ paragraphStyleById.set(style.styleId, style);
133
+ }
134
+ }
135
+ const docDefaultOutlineLevel = docDefaults?.pPr?.outlineLevel;
136
+ for (const style of paragraphStyles) {
137
+ if (style.name === void 0) continue;
138
+ const namedLevel = headingOutlineLevelFromStyleName(style.name);
139
+ const headingLevel = namedLevel === void 0 ? void 0 : inheritedOutlineLevel(style, styleById) ?? docDefaultOutlineLevel ?? namedLevel;
140
+ if (headingLevel !== void 0 && isHeadingOutlineLevel(headingLevel) && !styleIdByHeadingLevel.has(headingLevel)) {
141
+ styleIdByHeadingLevel.set(headingLevel, style.styleId);
142
+ continue;
143
+ }
144
+ const normalized = normalizeStyleName(style.name);
145
+ const tocLevel = BUILT_IN_TABLE_OF_CONTENTS_NAME.exec(normalized)?.groups?.["level"];
146
+ if (tocLevel !== void 0) {
147
+ const level = Number.parseInt(tocLevel, 10);
148
+ if (!styleIdByTableOfContentsLevel.has(level)) styleIdByTableOfContentsLevel.set(level, style.styleId);
149
+ continue;
150
+ }
151
+ const builtInName = asBuiltInName(normalized);
152
+ if (builtInName !== void 0 && !styleIdByBuiltInName.has(builtInName)) styleIdByBuiltInName.set(builtInName, style.styleId);
153
+ }
154
+ /**
155
+ * The style a paragraph actually resolves against. ECMA-376 17.7.2 layer 3
156
+ * is "the paragraph's own style chain", and a paragraph with no `w:pStyle`
157
+ * (or one naming a style the package never defines, or one naming a
158
+ * character style) takes the default paragraph style — the same fallback
159
+ * `StyleResolver` applies, so the model and the editor cannot drift apart.
160
+ */
161
+ const defaultStyle = resolveDefaultParagraphStyle(paragraphStyles);
162
+ const styleFor = (styleId) => (styleId === null || styleId === void 0 ? void 0 : paragraphStyleById.get(styleId)) ?? defaultStyle;
163
+ const outlineLevelCache = /* @__PURE__ */ new Map();
164
+ return {
165
+ outlineLevelOf: (styleId) => {
166
+ if (outlineLevelCache.has(styleId)) return outlineLevelCache.get(styleId);
167
+ const style = styleFor(styleId);
168
+ const level = (style === void 0 ? void 0 : inheritedOutlineLevel(style, styleById)) ?? docDefaultOutlineLevel;
169
+ outlineLevelCache.set(styleId, level);
170
+ return level;
171
+ },
172
+ headingLevelFromNameOf: (styleId) => headingOutlineLevelFromStyleName(styleFor(styleId)?.name),
173
+ builtInNameOf: (styleId) => {
174
+ const name = styleFor(styleId)?.name;
175
+ return name === void 0 ? void 0 : asBuiltInName(normalizeStyleName(name));
176
+ },
177
+ undefinedBuiltInHeadingLevelOf: (styleId) => styleId === null || styleId === void 0 || paragraphStyleById.has(styleId) ? void 0 : headingOutlineLevelFromStyleName(styleId),
178
+ styleIdForHeadingLevel: (level) => styleIdByHeadingLevel.get(level),
179
+ styleIdForTableOfContentsLevel: (level) => styleIdByTableOfContentsLevel.get(level),
180
+ styleIdForBuiltInName: (name) => styleIdByBuiltInName.get(name)
181
+ };
182
+ };
183
+ /** An index over a document that defines no styles: every lookup misses. */
184
+ const EMPTY_BUILT_IN_STYLE_INDEX = createBuiltInStyleIndex([]);
185
+ /**
186
+ * The heading level a paragraph carries, zero-based (`heading 1` is 0), or
187
+ * undefined when it is not a heading.
188
+ *
189
+ * Precedence:
190
+ *
191
+ * 1. An effective outline level decides on its own, including
192
+ * {@link BODY_TEXT_OUTLINE_LEVEL}, which means "not a heading". A style
193
+ * named `heading 5` whose outline level is 0 is a level-1 heading; a style
194
+ * named `heading 3` reset to 9 is body text. The format gives the outline
195
+ * level to field calculation (17.3.1.20) and leaves the name to the UI
196
+ * (17.7.4.9), so the level is the one the document asserts.
197
+ * 2. Only when no outline level is set anywhere does the built-in `w:name`
198
+ * decide — the style is then inheriting the consumer's own built-in
199
+ * definition, which supplies the level.
200
+ * 3. Last resort, and only for a `w:pStyle` the package defines no style for:
201
+ * the id itself, read as the English built-in id. 17.7.4.17 makes a style
202
+ * without `w:customStyle` a built-in and lets an application recognise it
203
+ * "if the associated style ID is known", which is the one case where the id
204
+ * is all the information left. `defaultParagraphStyle.ts` keeps the same
205
+ * last tier for `Normal`. A document that defines its heading styles never
206
+ * reaches this, so it cannot override a name or an outline level — and a
207
+ * localized package never writes an English id to begin with.
208
+ *
209
+ * Two consequences of rule 1 are deliberate, not oversights.
210
+ *
211
+ * **An outline level on a style that is not a heading still makes a heading.**
212
+ * The corpus has 680 such occurrences across 161 files, including `Title` at
213
+ * level 0 (42×) and `Subtitle` at level 1 (23×), plus `H1`, `Sub-heading`,
214
+ * `index heading` and a `DSTOC1-1`…`DSTOC8-8` family. Setting the level is how
215
+ * a document asks for a paragraph to be outlined, and Word's navigation pane
216
+ * and a `TOC \u` field both honour it, so folio does not second-guess a
217
+ * document that asked. Suppressing `Title` here would mean folio deciding a
218
+ * document's outline differs from Word's.
219
+ *
220
+ * **An outline level that disagrees with a built-in heading name wins.** 26
221
+ * corpus styles do this (`heading 5` at level 0, `heading 3` at level 1, and
222
+ * so on), all from non-Word producers or hand-authored fixtures. The format
223
+ * gives the level to field calculation (17.3.1.20) and the name to the user
224
+ * interface (17.7.4.9), so the level is the machine-readable claim and the
225
+ * name is a label. This is the one rule below that was not confirmed against
226
+ * Word itself.
227
+ */
228
+ const resolveHeadingLevel = (paragraph, index) => {
229
+ const effective = paragraph.outlineLevel ?? index.outlineLevelOf(paragraph.styleId);
230
+ if (effective !== null && effective !== void 0) return isHeadingOutlineLevel(effective) ? effective : void 0;
231
+ return index.headingLevelFromNameOf(paragraph.styleId) ?? index.undefinedBuiltInHeadingLevelOf(paragraph.styleId);
232
+ };
233
+ /** True when the paragraph's style is Word's `Quote` or `Intense Quote`. */
234
+ const isQuoteStyle = (styleId, index) => {
235
+ const name = index.builtInNameOf(styleId);
236
+ return name === BUILT_IN_STYLE_NAME.quote || name === BUILT_IN_STYLE_NAME.intenseQuote;
237
+ };
238
+ //#endregion
239
+ export { BODY_TEXT_OUTLINE_LEVEL, BUILT_IN_STYLE_NAME, EMPTY_BUILT_IN_STYLE_INDEX, builtInHeadingStyleName, builtInTableOfContentsStyleName, createBuiltInStyleIndex, isHeadingOutlineLevel, isQuoteStyle, normalizeStyleName, resolveHeadingLevel };
@@ -1,8 +1,10 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
2
  //#region src/docx/commentIdNormalization.d.ts
3
+ /** The code this normalisation is reported under, owned here, not at the caller. */
4
+ declare const DUPLICATE_COMMENT_ID_WARNING: "duplicate-comment-id";
3
5
  type NormalizeCommentIdsResult = {
4
6
  droppedDuplicateComments: number;
5
7
  };
6
8
  declare const normalizeCommentIds: (comments: document_d_exports.Comment[]) => NormalizeCommentIdsResult;
7
9
  //#endregion
8
- export { NormalizeCommentIdsResult, normalizeCommentIds };
10
+ export { DUPLICATE_COMMENT_ID_WARNING, NormalizeCommentIdsResult, normalizeCommentIds };
@@ -1,4 +1,21 @@
1
+ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
1
2
  //#region src/docx/commentIdNormalization.ts
3
+ /**
4
+ * One comment per `w:id`, because that is all the body can address.
5
+ *
6
+ * `w:commentReference`, `w:commentRangeStart` and `w:commentRangeEnd` name a
7
+ * comment by its `w:id`, so two `w:comment` elements sharing an id make every
8
+ * marker that names it ambiguous. Word opens such a package and resolves each
9
+ * marker to the first definition, so folio keeps the first and drops the rest:
10
+ * no marker can address a later one, and re-numbering it would invent a
11
+ * comment nothing anchors. Microsoft's own conformance corpus ships a file
12
+ * that does this.
13
+ *
14
+ * Reply links survive unchanged. `w15:paraIdParent` resolves to a comment id
15
+ * before this runs, and that id still belongs to the definition that stayed.
16
+ */
17
+ /** The code this normalisation is reported under, owned here, not at the caller. */
18
+ const DUPLICATE_COMMENT_ID_WARNING = PARSE_WARNING_CODES.duplicateCommentId;
2
19
  const normalizeCommentIds = (comments) => {
3
20
  const seen = /* @__PURE__ */ new Set();
4
21
  let kept = 0;
@@ -13,4 +30,4 @@ const normalizeCommentIds = (comments) => {
13
30
  return { droppedDuplicateComments };
14
31
  };
15
32
  //#endregion
16
- export { normalizeCommentIds };
33
+ export { DUPLICATE_COMMENT_ID_WARNING, normalizeCommentIds };
@@ -1,5 +1,6 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
2
  import { StyleMap } from "./styleParser.js";
3
+ import { ParseContext } from "./parseContext.js";
3
4
  //#region src/docx/commentParser.d.ts
4
5
  type CommentExtendedInfo = {
5
6
  parentParaId?: string;
@@ -28,6 +29,6 @@ declare function parseCommentsExtended(xml: string): Map<string, CommentExtended
28
29
  * local time. If `commentsExtendedXml` is provided, reply-thread
29
30
  * parent links (`parentId`) and resolved state (`done`) are populated.
30
31
  */
31
- declare function parseComments(commentsXml: string | null, styles: StyleMap | null, theme: document_d_exports.Theme | null, rels: document_d_exports.RelationshipMap, media: Map<string, document_d_exports.MediaFile>, commentsExtensibleXml?: string | null, commentsExtendedXml?: string | null): document_d_exports.Comment[];
32
+ declare function parseComments(commentsXml: string | null, styles: StyleMap | null, theme: document_d_exports.Theme | null, rels: document_d_exports.RelationshipMap, media: Map<string, document_d_exports.MediaFile>, commentsExtensibleXml?: string | null, commentsExtendedXml?: string | null, context?: ParseContext): document_d_exports.Comment[];
32
33
  //#endregion
33
34
  export { CommentExtendedInfo, parseComments, parseCommentsExtended };
@@ -1,9 +1,51 @@
1
+ import { commentThreadParaId } from "./commentThreadKey.js";
1
2
  import { parseParagraph } from "./paragraphParser.js";
2
3
  import { cloneParagraphWithPropertySource } from "./paragraphPropertySource.js";
3
4
  import { parseRunProperties } from "./runParser.js";
4
- import { findChild, getAttribute, getChildElements, getLocalName, parseXml } from "./xmlParser.js";
5
+ import { NAMESPACES, WORDPROCESSINGML_NAMESPACE_URIS, findChild, getAttribute, getAttributeByNamespaceUri, getChildElements, getLocalName, parseOnOffValue, parseXml } from "./xmlParser.js";
6
+ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
5
7
  //#region src/docx/commentParser.ts
8
+ /**
9
+ * Comment Parser - Parse comments.xml, commentsExtensible.xml, and
10
+ * commentsExtended.xml.
11
+ *
12
+ * - `comments.xml` (w) carries the comment author, local date, and body.
13
+ * - `commentsExtensible.xml` (w16cex, Word 2016+) carries reliable UTC
14
+ * timestamps via `w16cex:dateUtc` — Word's `w:date` is local time
15
+ * without an offset and so is ambiguous.
16
+ * - `commentsExtended.xml` (w15, Word 2013+) carries reply-thread
17
+ * parent links via `w15:paraIdParent` and the resolved/done state
18
+ * via `w15:done`. Cross-referenced via the `w14:paraId` on
19
+ * `w:comment` and the matching `w15:paraId` on `w15:commentEx`.
20
+ *
21
+ * OOXML Reference:
22
+ * - Comments: w:comments
23
+ * - Comment: w:comment (w:id, w:author, w:date, w:initials, w14:paraId)
24
+ * - Comment content: child w:p elements
25
+ */
26
+ /** The whole lexical form of `ST_DecimalNumber`: an optional sign and digits. */
27
+ const DECIMAL_NUMBER = /^[+-]?\d+$/u;
6
28
  const DEFAULT_ANNOTATION_REFERENCE_STYLE_ID = "CommentReference";
29
+ /**
30
+ * The namespaces a paraId may be written in: Word 2010 and Word 2012 wordml.
31
+ *
32
+ * Resolved by URI rather than by prefix, because the value is a thread key: an
33
+ * unrelated `vendor:paraId` picked up by local name alone would key a comment
34
+ * to the wrong thread and carry that thread's date, parent and resolved state.
35
+ */
36
+ const PARA_ID_NAMESPACE_URIS = /* @__PURE__ */ new Set([NAMESPACES.w14, NAMESPACES.w15]);
37
+ /**
38
+ * A paraId join key, wherever an exporter writes it: on `w:comment` or `w:p`.
39
+ *
40
+ * One reading for both, so the wrapper and the paragraphs cannot come to
41
+ * disagree about what counts as a key. The literal reads are the fallback for
42
+ * a part that writes a conventional prefix without binding it, which no URI
43
+ * lookup can resolve.
44
+ */
45
+ const paraIdAttribute = (element) => {
46
+ const raw = getAttributeByNamespaceUri(element, PARA_ID_NAMESPACE_URIS, "paraId") ?? element.attributes?.["w14:paraId"] ?? element.attributes?.["w15:paraId"] ?? getAttributeByNamespaceUri(element, WORDPROCESSINGML_NAMESPACE_URIS, "paraId") ?? element.attributes?.["w:paraId"];
47
+ return raw === null || raw === void 0 ? void 0 : String(raw);
48
+ };
7
49
  const normalizeAnnotationReferenceFormatting = (formatting) => {
8
50
  if (formatting?.styleId === DEFAULT_ANNOTATION_REFERENCE_STYLE_ID && Object.keys(formatting).length === 1) return;
9
51
  return formatting;
@@ -35,7 +77,8 @@ function parseCommentsExtensible(xml) {
35
77
  if ((child.name?.replace(/^.*:/u, "") ?? "") !== "comment") continue;
36
78
  const paraId = getAttribute(child, "w16cex", "paraId") ?? getAttribute(child, "w15", "paraId") ?? child.attributes?.["w16cex:paraId"] ?? child.attributes?.["w15:paraId"];
37
79
  const dateUtc = getAttribute(child, "w16cex", "dateUtc") ?? getAttribute(child, "w15", "dateUtc") ?? child.attributes?.["w16cex:dateUtc"] ?? child.attributes?.["w15:dateUtc"];
38
- if (paraId && dateUtc) dateUtcByParaId.set(String(paraId).toUpperCase(), String(dateUtc));
80
+ const key = paraId === null || paraId === void 0 ? null : String(paraId).toUpperCase();
81
+ if (key && dateUtc && !dateUtcByParaId.has(key)) dateUtcByParaId.set(key, String(dateUtc));
39
82
  }
40
83
  return dateUtcByParaId;
41
84
  }
@@ -61,15 +104,14 @@ function parseCommentsExtended(xml) {
61
104
  if ((child.name?.replace(/^.*:/u, "") ?? "") !== "commentEx") continue;
62
105
  const paraId = getAttribute(child, "w15", "paraId") ?? child.attributes?.["w15:paraId"];
63
106
  if (!paraId) continue;
107
+ const key = String(paraId).toUpperCase();
108
+ if (infoByParaId.has(key)) continue;
64
109
  const parentParaId = getAttribute(child, "w15", "paraIdParent") ?? child.attributes?.["w15:paraIdParent"];
65
110
  const doneAttr = getAttribute(child, "w15", "done") ?? child.attributes?.["w15:done"];
66
111
  const info = {};
67
112
  if (parentParaId) info.parentParaId = String(parentParaId).toUpperCase();
68
- if (doneAttr !== void 0) {
69
- const v = String(doneAttr).toLowerCase();
70
- info.done = v === "1" || v === "true";
71
- }
72
- infoByParaId.set(String(paraId).toUpperCase(), info);
113
+ if (doneAttr !== void 0) info.done = parseOnOffValue(String(doneAttr).toLowerCase()) ?? false;
114
+ infoByParaId.set(key, info);
73
115
  }
74
116
  return infoByParaId;
75
117
  }
@@ -81,30 +123,33 @@ function parseCommentsExtended(xml) {
81
123
  * local time. If `commentsExtendedXml` is provided, reply-thread
82
124
  * parent links (`parentId`) and resolved state (`done`) are populated.
83
125
  */
84
- function parseComments(commentsXml, styles, theme, rels, media, commentsExtensibleXml, commentsExtendedXml) {
126
+ function parseComments(commentsXml, styles, theme, rels, media, commentsExtensibleXml, commentsExtendedXml, context) {
85
127
  if (!commentsXml) return [];
86
128
  const root = parseXml(commentsXml);
87
129
  const dateUtcByParaId = commentsExtensibleXml ? parseCommentsExtensible(commentsExtensibleXml) : /* @__PURE__ */ new Map();
88
130
  const extendedByParaId = commentsExtendedXml ? parseCommentsExtended(commentsExtendedXml) : /* @__PURE__ */ new Map();
89
131
  const children = getChildElements(findChild(root, "w", "comments") ?? root);
90
- const comments = [];
132
+ const parsed = [];
91
133
  const commentIdByParaId = /* @__PURE__ */ new Map();
92
- const paraIdByCommentIndex = /* @__PURE__ */ new Map();
93
134
  for (const child of children) {
94
135
  if ((child.name?.replace(/^.*:/u, "") ?? "") !== "comment") continue;
95
- const id = Number.parseInt(getAttribute(child, "w", "id") ?? "0", 10);
136
+ const rawId = getAttribute(child, "w", "id");
137
+ const id = rawId !== null && DECIMAL_NUMBER.test(rawId) ? Number.parseInt(rawId, 10) : NaN;
138
+ if (Number.isNaN(id)) {
139
+ context?.warn({
140
+ code: PARSE_WARNING_CODES.missingCommentId,
141
+ element: "w:comment",
142
+ ...rawId === null ? {} : { value: rawId }
143
+ });
144
+ continue;
145
+ }
96
146
  const rawAuthor = getAttribute(child, "w", "author");
97
147
  const author = parseCommentAuthor(rawAuthor);
98
148
  const rawInitials = getAttribute(child, "w", "initials");
99
149
  const initials = rawInitials !== null ? String(rawInitials) : void 0;
100
150
  const rawDate = getAttribute(child, "w", "date");
101
151
  const localDate = rawDate !== null ? String(rawDate) : void 0;
102
- let rawParaId = getAttribute(child, "w14", "paraId") ?? child.attributes?.["w14:paraId"] ?? getAttribute(child, "w15", "paraId") ?? child.attributes?.["w15:paraId"] ?? getAttribute(child, "w", "paraId");
103
- if (!rawParaId) for (const sub of getChildElements(child)) {
104
- if ((sub.name?.replace(/^.*:/u, "") ?? "") !== "p") continue;
105
- const subParaId = getAttribute(sub, "w14", "paraId") ?? sub.attributes?.["w14:paraId"] ?? getAttribute(sub, "w15", "paraId") ?? sub.attributes?.["w15:paraId"] ?? getAttribute(sub, "w", "paraId");
106
- if (subParaId) rawParaId = subParaId;
107
- }
152
+ const rawParaId = paraIdAttribute(child) ?? commentThreadParaId(getChildElements(child).filter((sub) => (sub.name?.replace(/^.*:/u, "") ?? "") === "p").map(paraIdAttribute));
108
153
  const paraId = rawParaId ? String(rawParaId).toUpperCase() : null;
109
154
  const date = (paraId ? dateUtcByParaId.get(paraId) : void 0) ?? localDate;
110
155
  const done = (paraId ? extendedByParaId.get(paraId) : void 0)?.done;
@@ -120,33 +165,26 @@ function parseComments(commentsXml, styles, theme, rels, media, commentsExtensib
120
165
  annotationReferenceFormatting = normalized.annotationReferenceFormatting;
121
166
  paragraphs.push(normalized.paragraph);
122
167
  }
123
- const commentIndex = comments.length;
124
- if (paraId) {
125
- commentIdByParaId.set(paraId, id);
126
- paraIdByCommentIndex.set(commentIndex, paraId);
127
- }
128
- comments.push({
129
- id,
130
- author,
131
- ...initials !== void 0 ? { initials } : {},
132
- ...date !== void 0 ? { date } : {},
133
- ...done !== void 0 ? { done } : {},
134
- ...annotationReferenceFormatting !== void 0 ? { annotationReferenceFormatting } : {},
135
- content: paragraphs
168
+ if (paraId && !commentIdByParaId.has(paraId)) commentIdByParaId.set(paraId, id);
169
+ parsed.push({
170
+ comment: {
171
+ id,
172
+ author,
173
+ ...initials !== void 0 ? { initials } : {},
174
+ ...date !== void 0 ? { date } : {},
175
+ ...done !== void 0 ? { done } : {},
176
+ ...annotationReferenceFormatting !== void 0 ? { annotationReferenceFormatting } : {},
177
+ content: paragraphs
178
+ },
179
+ threadParaId: paraId
136
180
  });
137
181
  }
138
- for (let i = 0; i < comments.length; i++) {
139
- const comment = comments[i];
140
- if (!comment) continue;
141
- const paraId = paraIdByCommentIndex.get(i);
142
- if (!paraId) continue;
143
- const parentParaId = extendedByParaId.get(paraId)?.parentParaId;
144
- if (!parentParaId) continue;
145
- const parentId = commentIdByParaId.get(parentParaId);
146
- if (parentId !== void 0 && parentId !== comment.id) comments[i] = {
147
- ...comment,
148
- parentId
149
- };
182
+ const comments = [];
183
+ for (const { comment, threadParaId } of parsed) {
184
+ const parentParaId = threadParaId ? extendedByParaId.get(threadParaId)?.parentParaId : void 0;
185
+ const parentId = parentParaId ? commentIdByParaId.get(parentParaId) : void 0;
186
+ if (parentId !== void 0 && parentId !== comment.id) comment.parentId = parentId;
187
+ comments.push(comment);
150
188
  }
151
189
  return comments;
152
190
  }
@@ -1,5 +1,8 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
2
  //#region src/docx/commentReferenceNormalization.d.ts
3
+ /** The codes this normalisation is reported under, owned here, not at the caller. */
4
+ declare const DANGLING_COMMENT_REFERENCE_WARNING: "dangling-comment-reference";
5
+ declare const UNBALANCED_COMMENT_RANGE_WARNING: "unbalanced-comment-range";
3
6
  type NormalizeCommentReferencesInput = {
4
7
  documentBody: document_d_exports.DocumentBody;
5
8
  comments: readonly document_d_exports.Comment[];
@@ -14,4 +17,4 @@ type NormalizeCommentReferencesResult = {
14
17
  };
15
18
  declare const normalizeCommentReferences: ({ documentBody, comments, headers, footers, footnotes, endnotes }: NormalizeCommentReferencesInput) => NormalizeCommentReferencesResult;
16
19
  //#endregion
17
- export { normalizeCommentReferences };
20
+ export { DANGLING_COMMENT_REFERENCE_WARNING, UNBALANCED_COMMENT_RANGE_WARNING, normalizeCommentReferences };
@@ -1,24 +1,31 @@
1
+ import { InlineContentRemovals, visitInlineContentSlots } from "./paragraphTraversal.js";
2
+ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
1
3
  //#region src/docx/commentReferenceNormalization.ts
4
+ /** The codes this normalisation is reported under, owned here, not at the caller. */
5
+ const DANGLING_COMMENT_REFERENCE_WARNING = PARSE_WARNING_CODES.danglingCommentReference;
6
+ const UNBALANCED_COMMENT_RANGE_WARNING = PARSE_WARNING_CODES.unbalancedCommentRange;
2
7
  const normalizeCommentReferences = ({ documentBody, comments, headers, footers, footnotes, endnotes }) => {
3
8
  const validCommentIds = new Set(comments.map((comment) => comment.id));
4
9
  const rangeMarkers = [];
10
+ const dangling = new InlineContentRemovals();
5
11
  let removedDanglingReferences = 0;
6
12
  const normalizeParagraph = (paragraph) => {
7
- const nextContent = [];
8
- for (const content of paragraph.content) {
9
- if (isCommentMarker(content) && !validCommentIds.has(content.id)) {
13
+ visitInlineContentSlots(paragraph, ({ content, index, item }) => {
14
+ if (isCommentMarker(item) && !validCommentIds.has(item.id)) {
15
+ dangling.mark({
16
+ content,
17
+ index
18
+ });
10
19
  removedDanglingReferences += 1;
11
- continue;
20
+ return;
12
21
  }
13
- nextContent.push(content);
14
- if (isCommentRangeMarker(content)) rangeMarkers.push({
15
- content: nextContent,
16
- index: nextContent.length - 1,
17
- id: content.id,
18
- type: content.type
22
+ if (isCommentRangeMarker(item)) rangeMarkers.push({
23
+ content,
24
+ index,
25
+ id: item.id,
26
+ type: item.type
19
27
  });
20
- }
21
- paragraph.content = nextContent;
28
+ });
22
29
  };
23
30
  const normalizeTable = (table) => {
24
31
  for (const row of table.rows) for (const cell of row.cells) normalizeBlocks(cell.content);
@@ -43,9 +50,11 @@ const normalizeCommentReferences = ({ documentBody, comments, headers, footers,
43
50
  for (const footnote of footnotes ?? []) normalizeBlocks(footnote.content);
44
51
  for (const endnote of endnotes ?? []) normalizeBlocks(endnote.content);
45
52
  for (const comment of comments) for (const paragraph of comment.content) normalizeParagraph(paragraph);
53
+ const reanchoredUnbalancedRanges = reanchorUnbalancedCommentRanges(rangeMarkers);
54
+ dangling.apply();
46
55
  return {
47
56
  removedDanglingReferences,
48
- reanchoredUnbalancedRanges: reanchorUnbalancedCommentRanges(rangeMarkers)
57
+ reanchoredUnbalancedRanges
49
58
  };
50
59
  };
51
60
  const reanchorUnbalancedCommentRanges = (rangeMarkers) => {
@@ -94,4 +103,4 @@ const reanchorCommentRangeMarker = (marker) => {
94
103
  const isCommentMarker = (content) => content.type === "commentRangeStart" || content.type === "commentRangeEnd" || content.type === "commentReference";
95
104
  const isCommentRangeMarker = (content) => content.type === "commentRangeStart" || content.type === "commentRangeEnd";
96
105
  //#endregion
97
- export { normalizeCommentReferences };
106
+ export { DANGLING_COMMENT_REFERENCE_WARNING, UNBALANCED_COMMENT_RANGE_WARNING, normalizeCommentReferences };
@@ -0,0 +1,18 @@
1
+ //#region src/docx/commentThreadKey.d.ts
2
+ /**
3
+ * The one rule for "which `w14:paraId` a comment is threaded by".
4
+ *
5
+ * `w15:commentEx/@w15:paraId` and `w15:paraIdParent` name a comment through the
6
+ * `w14:paraId` of its LAST paragraph, never through its `w:id`. Files in the
7
+ * wild leave that paragraph's id off and carry one on an earlier paragraph, so
8
+ * the rule folio reads by is the last paragraph that carries an id.
9
+ *
10
+ * The rule has two readers over two representations: the parser applies it to
11
+ * the `w:p` children of a `w:comment`, the serializer to the model's
12
+ * `Comment.content`. They must not drift — a save that keyed on a different
13
+ * paragraph than the parse would write a `commentsExtended.xml` whose entries
14
+ * belong to other comments — so both call this instead of restating it.
15
+ */
16
+ declare const commentThreadParaId: (paragraphParaIds: Iterable<string | null | undefined>) => string | undefined;
17
+ //#endregion
18
+ export { commentThreadParaId };
@@ -0,0 +1,22 @@
1
+ //#region src/docx/commentThreadKey.ts
2
+ /**
3
+ * The one rule for "which `w14:paraId` a comment is threaded by".
4
+ *
5
+ * `w15:commentEx/@w15:paraId` and `w15:paraIdParent` name a comment through the
6
+ * `w14:paraId` of its LAST paragraph, never through its `w:id`. Files in the
7
+ * wild leave that paragraph's id off and carry one on an earlier paragraph, so
8
+ * the rule folio reads by is the last paragraph that carries an id.
9
+ *
10
+ * The rule has two readers over two representations: the parser applies it to
11
+ * the `w:p` children of a `w:comment`, the serializer to the model's
12
+ * `Comment.content`. They must not drift — a save that keyed on a different
13
+ * paragraph than the parse would write a `commentsExtended.xml` whose entries
14
+ * belong to other comments — so both call this instead of restating it.
15
+ */
16
+ const commentThreadParaId = (paragraphParaIds) => {
17
+ let key;
18
+ for (const paraId of paragraphParaIds) if (paraId) key = paraId;
19
+ return key;
20
+ };
21
+ //#endregion
22
+ export { commentThreadParaId };
@@ -0,0 +1,15 @@
1
+ import { document_d_exports } from "../types/document.js";
2
+ //#region src/docx/danglingRelationshipReferences.d.ts
3
+ type DanglingRelationshipReferences = {
4
+ /** Drawings whose `r:embed` names no relationship. */
5
+ drawings: number;
6
+ /** Hyperlinks whose `r:id` names no relationship. */
7
+ hyperlinks: number;
8
+ };
9
+ type CountDanglingRelationshipReferencesOptions = {
10
+ content: readonly document_d_exports.BlockContent[];
11
+ relationships: document_d_exports.RelationshipMap | undefined;
12
+ };
13
+ declare const countDanglingRelationshipReferences: ({ content, relationships }: CountDanglingRelationshipReferencesOptions) => DanglingRelationshipReferences;
14
+ //#endregion
15
+ export { DanglingRelationshipReferences, countDanglingRelationshipReferences };
@@ -0,0 +1,30 @@
1
+ import { resolveRelationshipId } from "./relsParser.js";
2
+ //#region src/docx/danglingRelationshipReferences.ts
3
+ const countDanglingRelationshipReferences = ({ content, relationships }) => {
4
+ const counts = {
5
+ drawings: 0,
6
+ hyperlinks: 0
7
+ };
8
+ const isDangling = (rId) => resolveRelationshipId(relationships, rId).status === "dangling";
9
+ const visitBlocks = (blocks) => {
10
+ for (const block of blocks) {
11
+ if (block.type === "table") {
12
+ for (const row of block.rows) for (const cell of row.cells) visitBlocks(cell.content);
13
+ continue;
14
+ }
15
+ if (block.type !== "paragraph") continue;
16
+ for (const item of block.content) {
17
+ if (item.type === "hyperlink") {
18
+ if (isDangling(item.rId)) counts.hyperlinks += 1;
19
+ continue;
20
+ }
21
+ if (item.type !== "run") continue;
22
+ for (const child of item.content) if (child.type === "drawing" && isDangling(child.image.rId)) counts.drawings += 1;
23
+ }
24
+ }
25
+ };
26
+ visitBlocks(content);
27
+ return counts;
28
+ };
29
+ //#endregion
30
+ export { countDanglingRelationshipReferences };