@stll/folio-core 0.43.0 → 0.44.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (183) hide show
  1. package/dist/ai-edits/headless.js +6 -5
  2. package/dist/ai-edits/index.d.ts +2 -2
  3. package/dist/ai-edits/index.js +2 -2
  4. package/dist/ai-edits/snapshot.js +13 -9
  5. package/dist/compare/content-alignment.js +16 -1
  6. package/dist/compare/inline-atoms.js +1 -1
  7. package/dist/compare/style-resources.js +6 -0
  8. package/dist/docx/appVersionNormalization.d.ts +0 -18
  9. package/dist/docx/blockContentParser.js +8 -0
  10. package/dist/docx/blockRangeMarkers.d.ts +36 -0
  11. package/dist/docx/blockRangeMarkers.js +59 -0
  12. package/dist/docx/bookmarkParser.d.ts +2 -20
  13. package/dist/docx/bookmarkParser.js +6 -30
  14. package/dist/docx/borderParser.d.ts +13 -0
  15. package/dist/docx/borderParser.js +71 -0
  16. package/dist/docx/builtInStyles.d.ts +165 -0
  17. package/dist/docx/builtInStyles.js +239 -0
  18. package/dist/docx/commentIdNormalization.d.ts +3 -1
  19. package/dist/docx/commentIdNormalization.js +18 -1
  20. package/dist/docx/commentParser.d.ts +2 -1
  21. package/dist/docx/commentParser.js +34 -7
  22. package/dist/docx/commentReferenceNormalization.d.ts +4 -1
  23. package/dist/docx/commentReferenceNormalization.js +23 -14
  24. package/dist/docx/danglingRelationshipReferences.d.ts +15 -0
  25. package/dist/docx/danglingRelationshipReferences.js +30 -0
  26. package/dist/docx/defaultParagraphStyle.d.ts +18 -1
  27. package/dist/docx/defaultParagraphStyle.js +23 -1
  28. package/dist/docx/documentParser.d.ts +2 -1
  29. package/dist/docx/documentParser.js +2 -2
  30. package/dist/docx/drawingUtils.d.ts +8 -1
  31. package/dist/docx/drawingUtils.js +12 -3
  32. package/dist/docx/fieldParser.js +3 -5
  33. package/dist/docx/footnoteParser.d.ts +3 -2
  34. package/dist/docx/footnoteParser.js +19 -4
  35. package/dist/docx/groupDrawingParser.js +1 -1
  36. package/dist/docx/headerFooterRefParser.d.ts +4 -3
  37. package/dist/docx/headerFooterRefParser.js +42 -12
  38. package/dist/docx/headerFooterReferenceNormalization.d.ts +4 -1
  39. package/dist/docx/headerFooterReferenceNormalization.js +5 -1
  40. package/dist/docx/hyperlinkParser.js +11 -15
  41. package/dist/docx/imageParser.d.ts +1 -1
  42. package/dist/docx/imageParser.js +22 -18
  43. package/dist/docx/imageRawXml.js +5 -5
  44. package/dist/docx/markupRangeMarker.d.ts +15 -0
  45. package/dist/docx/markupRangeMarker.js +44 -0
  46. package/dist/docx/noteReferenceStyles.d.ts +29 -0
  47. package/dist/docx/noteReferenceStyles.js +70 -0
  48. package/dist/docx/numberingReferenceNormalization.d.ts +4 -1
  49. package/dist/docx/numberingReferenceNormalization.js +20 -1
  50. package/dist/docx/paraIdRangeNormalization.d.ts +0 -19
  51. package/dist/docx/paragraphParser.js +66 -99
  52. package/dist/docx/paragraphPropertySource.js +1 -0
  53. package/dist/docx/paragraphTraversal.d.ts +37 -1
  54. package/dist/docx/paragraphTraversal.js +84 -1
  55. package/dist/docx/parseContext.d.ts +37 -0
  56. package/dist/docx/parseContext.js +67 -0
  57. package/dist/docx/parseWarningMessage.d.ts +6 -0
  58. package/dist/docx/parseWarningMessage.js +44 -0
  59. package/dist/docx/parser.js +81 -27
  60. package/dist/docx/relsParser.d.ts +28 -11
  61. package/dist/docx/relsParser.js +26 -13
  62. package/dist/docx/revisionIdNormalization.js +81 -7
  63. package/dist/docx/rezip.js +63 -22
  64. package/dist/docx/runConsolidator.js +1 -2
  65. package/dist/docx/runParser.d.ts +8 -1
  66. package/dist/docx/runParser.js +30 -48
  67. package/dist/docx/sectionParser.d.ts +2 -1
  68. package/dist/docx/sectionParser.js +21 -65
  69. package/dist/docx/serializer/borderSerializer.d.ts +1 -2
  70. package/dist/docx/serializer/commentSerializer.js +22 -9
  71. package/dist/docx/serializer/documentSerializer.d.ts +1 -5
  72. package/dist/docx/serializer/documentSerializer.js +6 -16
  73. package/dist/docx/serializer/headerFooterSerializer.js +5 -0
  74. package/dist/docx/serializer/markupRangeAttributes.d.ts +8 -0
  75. package/dist/docx/serializer/markupRangeAttributes.js +24 -0
  76. package/dist/docx/serializer/noteSerializer.js +5 -0
  77. package/dist/docx/serializer/paragraphSerializer.d.ts +1 -5
  78. package/dist/docx/serializer/paragraphSerializer.js +29 -35
  79. package/dist/docx/serializer/runSerializer.js +13 -7
  80. package/dist/docx/serializer/tableSerializer.js +28 -13
  81. package/dist/docx/serializer/textFormattingSerializer.d.ts +2 -3
  82. package/dist/docx/server/build.js +8 -1
  83. package/dist/docx/server/createBilingualDocument.js +10 -18
  84. package/dist/docx/server/extractDocxText.js +3 -4
  85. package/dist/docx/shadingParser.d.ts +6 -0
  86. package/dist/docx/shadingParser.js +32 -0
  87. package/dist/docx/shapeParser.js +3 -3
  88. package/dist/docx/styleParser.js +13 -87
  89. package/dist/docx/styleReferenceResolution.d.ts +36 -0
  90. package/dist/docx/styleReferenceResolution.js +51 -0
  91. package/dist/docx/tableLook.d.ts +57 -0
  92. package/dist/docx/tableLook.js +63 -0
  93. package/dist/docx/tableParser.d.ts +7 -9
  94. package/dist/docx/tableParser.js +64 -110
  95. package/dist/docx/textBoxParser.js +4 -4
  96. package/dist/docx/trackedMoveRangeNormalization.d.ts +3 -1
  97. package/dist/docx/trackedMoveRangeNormalization.js +11 -21
  98. package/dist/docx/transitionalSpelling.d.ts +13 -2
  99. package/dist/docx/transitionalSpelling.js +23 -1
  100. package/dist/docx/verbatimCapture.js +4 -11
  101. package/dist/docx/vmlImageParser.js +2 -2
  102. package/dist/docx/watermarkParser.js +2 -2
  103. package/dist/docx/xmlParser.d.ts +22 -32
  104. package/dist/docx/xmlParser.js +36 -21
  105. package/dist/internal/pageBreakRunSourceDescendantIndex.js +2 -1
  106. package/dist/internal/paragraphFormattingSerialization.d.ts +2 -3
  107. package/dist/internal/paragraphFormattingSerialization.js +26 -6
  108. package/dist/layout-bridge/convert/footnoteLayout.js +2 -7
  109. package/dist/layout-engine/index.d.ts +2 -2
  110. package/dist/layout-engine/index.js +2 -2
  111. package/dist/layout-engine/measure/measureBlocks.js +1 -6
  112. package/dist/layout-engine/types.d.ts +8 -2
  113. package/dist/layout-engine/types.js +35 -2
  114. package/dist/markdown/index.js +1 -1
  115. package/dist/markdown/internals.d.ts +6 -1
  116. package/dist/markdown/internals.js +14 -1
  117. package/dist/markdown/renderBlock.js +35 -21
  118. package/dist/markdown/renderParagraph.js +14 -5
  119. package/dist/markdown/renderRuns.js +4 -3
  120. package/dist/markdown/renderTable.js +4 -3
  121. package/dist/markdown/trailers.js +41 -7
  122. package/dist/markdown/types.d.ts +3 -7
  123. package/dist/prosemirror/attrs/index.js +2 -5
  124. package/dist/prosemirror/bookmarkBoundaryAttrs.d.ts +11 -1
  125. package/dist/prosemirror/bookmarkBoundaryAttrs.js +18 -3
  126. package/dist/prosemirror/commands/index.d.ts +3 -3
  127. package/dist/prosemirror/commands/index.js +2 -2
  128. package/dist/prosemirror/commands/paragraph.d.ts +3 -3
  129. package/dist/prosemirror/commands/paragraph.js +2 -2
  130. package/dist/prosemirror/commentIdAllocator.js +2 -7
  131. package/dist/prosemirror/conversion/fromProseDoc.js +129 -40
  132. package/dist/prosemirror/conversion/toProseDoc.d.ts +1 -14
  133. package/dist/prosemirror/conversion/toProseDoc.js +375 -316
  134. package/dist/prosemirror/extensions/core/ParagraphExtension.d.ts +14 -1
  135. package/dist/prosemirror/extensions/core/ParagraphExtension.js +11 -6
  136. package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -3
  137. package/dist/prosemirror/extensions/features/PasteCleanupExtension.d.ts +4 -1
  138. package/dist/prosemirror/extensions/features/PasteCleanupExtension.js +6 -2
  139. package/dist/prosemirror/extensions/features/pastedHeadingStyles.d.ts +7 -0
  140. package/dist/prosemirror/extensions/features/pastedHeadingStyles.js +74 -0
  141. package/dist/prosemirror/extensions/marks/markUtils.d.ts +11 -3
  142. package/dist/prosemirror/extensions/marks/markUtils.js +98 -19
  143. package/dist/prosemirror/extensions/nodes/BookmarkBoundaryExtension.js +7 -3
  144. package/dist/prosemirror/extensions/nodes/ImageExtension.js +2 -1
  145. package/dist/prosemirror/extensions/nodes/ShapeExtension.js +1 -0
  146. package/dist/prosemirror/extensions/nodes/TableExtension.js +15 -1
  147. package/dist/prosemirror/extensions/types.d.ts +2 -2
  148. package/dist/prosemirror/index.d.ts +3 -3
  149. package/dist/prosemirror/index.js +3 -3
  150. package/dist/prosemirror/insertOperations.d.ts +9 -2
  151. package/dist/prosemirror/insertOperations.js +9 -4
  152. package/dist/prosemirror/paragraphFormattingProvenance.d.ts +159 -0
  153. package/dist/prosemirror/paragraphFormattingProvenance.js +106 -0
  154. package/dist/prosemirror/plugins/documentStyles.d.ts +9 -1
  155. package/dist/prosemirror/plugins/documentStyles.js +11 -1
  156. package/dist/prosemirror/plugins/index.d.ts +2 -2
  157. package/dist/prosemirror/plugins/index.js +2 -2
  158. package/dist/prosemirror/plugins/revisionIds.d.ts +11 -2
  159. package/dist/prosemirror/plugins/revisionIds.js +21 -6
  160. package/dist/prosemirror/runFormattingReconciliation.js +3 -2
  161. package/dist/prosemirror/runStyleFormatting.d.ts +1 -1
  162. package/dist/prosemirror/schema/nodes.d.ts +31 -0
  163. package/dist/prosemirror/styles/resolvedStyleAttrs.js +2 -0
  164. package/dist/prosemirror/styles/styleResolver.d.ts +9 -0
  165. package/dist/prosemirror/styles/styleResolver.js +12 -0
  166. package/dist/style-engine/styleEngine.d.ts +3 -0
  167. package/dist/style-engine/styleEngine.js +3 -0
  168. package/dist/style-sets/extract.js +1 -23
  169. package/dist/style-sets/stellaStyle.js +46 -39
  170. package/dist/style-sets/styleSetNormalization.d.ts +19 -0
  171. package/dist/style-sets/styleSetNormalization.js +99 -0
  172. package/dist/types/content.d.ts +2 -2
  173. package/dist/utils/createDocument.js +145 -20
  174. package/dist/utils/headingCollector.d.ts +8 -5
  175. package/dist/utils/headingCollector.js +23 -25
  176. package/dist/utils/tableOfContentsStyle.js +9 -2
  177. package/package.json +2 -2
  178. package/dist/docx/textWhitespace.d.ts +0 -4
  179. package/dist/docx/textWhitespace.js +0 -4
  180. package/dist/layout-bridge/engine/tableWidthUtils.d.ts +0 -6
  181. package/dist/layout-bridge/engine/tableWidthUtils.js +0 -25
  182. package/dist/markdown/headings.d.ts +0 -13
  183. package/dist/markdown/headings.js +0 -20
@@ -0,0 +1,165 @@
1
+ import { document_d_exports } from "../types/document.js";
2
+ //#region src/docx/builtInStyles.d.ts
3
+ /**
4
+ * The tenth `w:outlineLvl` value. 17.3.1.20: "the val attribute … can be from
5
+ * 0 to 9, where 9 specifically indicates that there is no outline level
6
+ * specifically applied to this paragraph." It is a deliberate "not a heading",
7
+ * not a tenth level. Every range test goes through
8
+ * {@link isHeadingOutlineLevel} so the reserved value keeps one meaning across
9
+ * the codebase.
10
+ *
11
+ * The same clause adds that an omitted element "is assumed to be 9". That
12
+ * default cannot be applied to a *style* definition, because 17.7.1 tells
13
+ * producers not to write a property "already been set by a previous level of
14
+ * the style hierarchy": a document that names a style `heading 1` and omits
15
+ * the level is inheriting the consumer's built-in definition, which carries
16
+ * level 0. An absent level therefore means "unspecified, ask the name", and
17
+ * only a written 9 means body text.
18
+ */
19
+ declare const BODY_TEXT_OUTLINE_LEVEL = 9;
20
+ /** True when an outline level names a heading rather than body text. */
21
+ declare const isHeadingOutlineLevel: (level: number | null | undefined) => level is number;
22
+ /**
23
+ * Compare style names the way producers actually write them. The corpus shows
24
+ * both `heading 1` (Annex L, 94.8%) and `Heading 1` (5.2%), and LibreOffice
25
+ * drops the space entirely (`Heading1`, `IntenseQuote`), so case and
26
+ * whitespace are the tolerance. A name is otherwise matched whole: a style a
27
+ * Czech template calls `Nadpis 1` stays a custom style.
28
+ */
29
+ declare const normalizeStyleName: (name: string) => string;
30
+ /**
31
+ * The `w:name` Word itself writes for each built-in, and therefore the
32
+ * spelling every style table folio authors must use. One owner: a style set and
33
+ * the classifier that reads it cannot drift apart if both name the same
34
+ * constant.
35
+ *
36
+ * Word is not uniformly cased and guessing gets it wrong, so each value is the
37
+ * spelling that dominates Microsoft Word output in the public corpus:
38
+ * `footnote text` 375 against 49 `Footnote Text`, `footer` 1145 against 2,
39
+ * `caption` 390 against 60 — but `Body Text` 424 against 2, `Title` 621
40
+ * against 2, and the auto-generated linked character styles (`Footnote Text
41
+ * Char` 248, `Endnote Text Char` 116) title-cased without exception.
42
+ * {@link normalizeStyleName} makes matching tolerant of all of it; this map is
43
+ * about what folio *writes*.
44
+ */
45
+ declare const BUILT_IN_STYLE_NAME: {
46
+ /** 4,515 Word occurrences against 3 lowercase. */
47
+ readonly normal: "Normal";
48
+ readonly bodyText: "Body Text";
49
+ readonly title: "Title";
50
+ readonly subtitle: "Subtitle";
51
+ readonly quote: "Quote";
52
+ readonly intenseQuote: "Intense Quote";
53
+ readonly listParagraph: "List Paragraph";
54
+ readonly tocHeading: "TOC Heading";
55
+ readonly caption: "caption";
56
+ readonly header: "header";
57
+ readonly footer: "footer";
58
+ readonly footnoteText: "footnote text";
59
+ readonly commentReference: "annotation reference";
60
+ readonly footnoteReference: "footnote reference";
61
+ readonly footnoteTextChar: "Footnote Text Char";
62
+ readonly endnoteText: "endnote text";
63
+ readonly endnoteReference: "endnote reference";
64
+ readonly endnoteTextChar: "Endnote Text Char";
65
+ readonly hyperlink: "Hyperlink";
66
+ readonly defaultParagraphFont: "Default Paragraph Font";
67
+ readonly noList: "No List";
68
+ readonly normalTable: "Normal Table";
69
+ readonly tableGrid: "Table Grid";
70
+ };
71
+ type BuiltInStyleName = (typeof BUILT_IN_STYLE_NAME)[keyof typeof BUILT_IN_STYLE_NAME];
72
+ /**
73
+ * The name of a built-in heading, from its zero-based outline level.
74
+ * Lowercase: 1,066 Word occurrences of `heading 1` against 13 `Heading 1`, and
75
+ * Annex L writes the latent-style exceptions the same way.
76
+ */
77
+ declare const builtInHeadingStyleName: (outlineLevel: number) => string;
78
+ /** The name of a built-in TOC entry style, from its one-based level (`toc 1`). */
79
+ declare const builtInTableOfContentsStyleName: (level: number) => string;
80
+ /**
81
+ * A document's styles indexed by what they *are* rather than by what they are
82
+ * called. Built once per document: the consumers below classify every
83
+ * paragraph, and rebuilding the maps per paragraph would make each of them
84
+ * quadratic.
85
+ */
86
+ type BuiltInStyleIndex = {
87
+ /** The style's effective `w:outlineLvl`, including 9, or undefined. */
88
+ outlineLevelOf: (styleId: string | null | undefined) => number | undefined;
89
+ /** The zero-based level a built-in heading *name* implies, or undefined. */
90
+ headingLevelFromNameOf: (styleId: string | null | undefined) => number | undefined;
91
+ /** The built-in this style is, by name, or undefined for a custom style. */
92
+ builtInNameOf: (styleId: string | null | undefined) => BuiltInStyleName | undefined;
93
+ /**
94
+ * The level an English built-in heading *id* implies, and only when the
95
+ * package defines no style under it. See {@link resolveHeadingLevel} tier 3.
96
+ */
97
+ undefinedBuiltInHeadingLevelOf: (styleId: string | null | undefined) => number | undefined;
98
+ /** The document's style id for a built-in heading level (zero-based). */
99
+ styleIdForHeadingLevel: (level: number) => string | undefined;
100
+ /** The document's style id for a built-in TOC entry level (one-based, `toc 1`). */
101
+ styleIdForTableOfContentsLevel: (level: number) => string | undefined;
102
+ /** The document's style id for a named built-in, e.g. `TOC Heading`. */
103
+ styleIdForBuiltInName: (name: BuiltInStyleName) => string | undefined;
104
+ };
105
+ declare const createBuiltInStyleIndex: (styles: Iterable<document_d_exports.Style>, docDefaults?: document_d_exports.DocDefaults | undefined) => BuiltInStyleIndex;
106
+ /** An index over a document that defines no styles: every lookup misses. */
107
+ declare const EMPTY_BUILT_IN_STYLE_INDEX: BuiltInStyleIndex;
108
+ /**
109
+ * What a consumer knows about a paragraph: its style id and whatever
110
+ * `w:outlineLvl` applies to it. The ProseMirror `outlineLevel` attr already
111
+ * holds direct-else-style resolution, so passing it here agrees with passing
112
+ * direct formatting from the DOCX model.
113
+ */
114
+ type ParagraphOutlineSource = {
115
+ styleId?: string | null | undefined;
116
+ outlineLevel?: number | null | undefined;
117
+ };
118
+ /**
119
+ * The heading level a paragraph carries, zero-based (`heading 1` is 0), or
120
+ * undefined when it is not a heading.
121
+ *
122
+ * Precedence:
123
+ *
124
+ * 1. An effective outline level decides on its own, including
125
+ * {@link BODY_TEXT_OUTLINE_LEVEL}, which means "not a heading". A style
126
+ * named `heading 5` whose outline level is 0 is a level-1 heading; a style
127
+ * named `heading 3` reset to 9 is body text. The format gives the outline
128
+ * level to field calculation (17.3.1.20) and leaves the name to the UI
129
+ * (17.7.4.9), so the level is the one the document asserts.
130
+ * 2. Only when no outline level is set anywhere does the built-in `w:name`
131
+ * decide — the style is then inheriting the consumer's own built-in
132
+ * definition, which supplies the level.
133
+ * 3. Last resort, and only for a `w:pStyle` the package defines no style for:
134
+ * the id itself, read as the English built-in id. 17.7.4.17 makes a style
135
+ * without `w:customStyle` a built-in and lets an application recognise it
136
+ * "if the associated style ID is known", which is the one case where the id
137
+ * is all the information left. `defaultParagraphStyle.ts` keeps the same
138
+ * last tier for `Normal`. A document that defines its heading styles never
139
+ * reaches this, so it cannot override a name or an outline level — and a
140
+ * localized package never writes an English id to begin with.
141
+ *
142
+ * Two consequences of rule 1 are deliberate, not oversights.
143
+ *
144
+ * **An outline level on a style that is not a heading still makes a heading.**
145
+ * The corpus has 680 such occurrences across 161 files, including `Title` at
146
+ * level 0 (42×) and `Subtitle` at level 1 (23×), plus `H1`, `Sub-heading`,
147
+ * `index heading` and a `DSTOC1-1`…`DSTOC8-8` family. Setting the level is how
148
+ * a document asks for a paragraph to be outlined, and Word's navigation pane
149
+ * and a `TOC \u` field both honour it, so folio does not second-guess a
150
+ * document that asked. Suppressing `Title` here would mean folio deciding a
151
+ * document's outline differs from Word's.
152
+ *
153
+ * **An outline level that disagrees with a built-in heading name wins.** 26
154
+ * corpus styles do this (`heading 5` at level 0, `heading 3` at level 1, and
155
+ * so on), all from non-Word producers or hand-authored fixtures. The format
156
+ * gives the level to field calculation (17.3.1.20) and the name to the user
157
+ * interface (17.7.4.9), so the level is the machine-readable claim and the
158
+ * name is a label. This is the one rule below that was not confirmed against
159
+ * Word itself.
160
+ */
161
+ declare const resolveHeadingLevel: (paragraph: ParagraphOutlineSource, index: BuiltInStyleIndex) => number | undefined;
162
+ /** True when the paragraph's style is Word's `Quote` or `Intense Quote`. */
163
+ declare const isQuoteStyle: (styleId: string | null | undefined, index: BuiltInStyleIndex) => boolean;
164
+ //#endregion
165
+ export { BODY_TEXT_OUTLINE_LEVEL, BUILT_IN_STYLE_NAME, BuiltInStyleIndex, EMPTY_BUILT_IN_STYLE_INDEX, ParagraphOutlineSource, builtInHeadingStyleName, builtInTableOfContentsStyleName, createBuiltInStyleIndex, isHeadingOutlineLevel, isQuoteStyle, normalizeStyleName, resolveHeadingLevel };
@@ -0,0 +1,239 @@
1
+ import { BUILT_IN_DEFAULT_PARAGRAPH_STYLE_NAME, resolveDefaultParagraphStyle } from "./defaultParagraphStyle.js";
2
+ //#region src/docx/builtInStyles.ts
3
+ /**
4
+ * The tenth `w:outlineLvl` value. 17.3.1.20: "the val attribute … can be from
5
+ * 0 to 9, where 9 specifically indicates that there is no outline level
6
+ * specifically applied to this paragraph." It is a deliberate "not a heading",
7
+ * not a tenth level. Every range test goes through
8
+ * {@link isHeadingOutlineLevel} so the reserved value keeps one meaning across
9
+ * the codebase.
10
+ *
11
+ * The same clause adds that an omitted element "is assumed to be 9". That
12
+ * default cannot be applied to a *style* definition, because 17.7.1 tells
13
+ * producers not to write a property "already been set by a previous level of
14
+ * the style hierarchy": a document that names a style `heading 1` and omits
15
+ * the level is inheriting the consumer's built-in definition, which carries
16
+ * level 0. An absent level therefore means "unspecified, ask the name", and
17
+ * only a written 9 means body text.
18
+ */
19
+ const BODY_TEXT_OUTLINE_LEVEL = 9;
20
+ /** The highest `w:outlineLvl` that still names a heading (outline level nine). */
21
+ const MAX_HEADING_OUTLINE_LEVEL = 8;
22
+ /** True when an outline level names a heading rather than body text. */
23
+ const isHeadingOutlineLevel = (level) => typeof level === "number" && Number.isInteger(level) && level >= 0 && level <= MAX_HEADING_OUTLINE_LEVEL;
24
+ /**
25
+ * Compare style names the way producers actually write them. The corpus shows
26
+ * both `heading 1` (Annex L, 94.8%) and `Heading 1` (5.2%), and LibreOffice
27
+ * drops the space entirely (`Heading1`, `IntenseQuote`), so case and
28
+ * whitespace are the tolerance. A name is otherwise matched whole: a style a
29
+ * Czech template calls `Nadpis 1` stays a custom style.
30
+ */
31
+ const normalizeStyleName = (name) => name.trim().toLowerCase().replace(/\s+/gu, "");
32
+ /**
33
+ * The `w:name` Word itself writes for each built-in, and therefore the
34
+ * spelling every style table folio authors must use. One owner: a style set and
35
+ * the classifier that reads it cannot drift apart if both name the same
36
+ * constant.
37
+ *
38
+ * Word is not uniformly cased and guessing gets it wrong, so each value is the
39
+ * spelling that dominates Microsoft Word output in the public corpus:
40
+ * `footnote text` 375 against 49 `Footnote Text`, `footer` 1145 against 2,
41
+ * `caption` 390 against 60 — but `Body Text` 424 against 2, `Title` 621
42
+ * against 2, and the auto-generated linked character styles (`Footnote Text
43
+ * Char` 248, `Endnote Text Char` 116) title-cased without exception.
44
+ * {@link normalizeStyleName} makes matching tolerant of all of it; this map is
45
+ * about what folio *writes*.
46
+ */
47
+ const BUILT_IN_STYLE_NAME = {
48
+ /** 4,515 Word occurrences against 3 lowercase. */
49
+ normal: BUILT_IN_DEFAULT_PARAGRAPH_STYLE_NAME,
50
+ bodyText: "Body Text",
51
+ title: "Title",
52
+ subtitle: "Subtitle",
53
+ quote: "Quote",
54
+ intenseQuote: "Intense Quote",
55
+ listParagraph: "List Paragraph",
56
+ tocHeading: "TOC Heading",
57
+ caption: "caption",
58
+ header: "header",
59
+ footer: "footer",
60
+ footnoteText: "footnote text",
61
+ commentReference: "annotation reference",
62
+ footnoteReference: "footnote reference",
63
+ footnoteTextChar: "Footnote Text Char",
64
+ endnoteText: "endnote text",
65
+ endnoteReference: "endnote reference",
66
+ endnoteTextChar: "Endnote Text Char",
67
+ hyperlink: "Hyperlink",
68
+ defaultParagraphFont: "Default Paragraph Font",
69
+ noList: "No List",
70
+ normalTable: "Normal Table",
71
+ tableGrid: "Table Grid"
72
+ };
73
+ /**
74
+ * The name of a built-in heading, from its zero-based outline level.
75
+ * Lowercase: 1,066 Word occurrences of `heading 1` against 13 `Heading 1`, and
76
+ * Annex L writes the latent-style exceptions the same way.
77
+ */
78
+ const builtInHeadingStyleName = (outlineLevel) => `heading ${outlineLevel + 1}`;
79
+ /** The name of a built-in TOC entry style, from its one-based level (`toc 1`). */
80
+ const builtInTableOfContentsStyleName = (level) => `toc ${level}`;
81
+ /**
82
+ * `heading 1`…`heading 9` against an already-normalised name. Word's built-in
83
+ * names stop at nine, matching `w:outlineLvl`'s nine heading values, so
84
+ * `Cmsor10` → `Címsor 10` is a user style rather than a tenth built-in.
85
+ */
86
+ const BUILT_IN_HEADING_NAME = /^heading(?<level>[1-9])$/u;
87
+ /** `toc 1`…`toc 9`, the styles a `TOC` field writes its entries in. */
88
+ const BUILT_IN_TABLE_OF_CONTENTS_NAME = /^toc(?<level>[1-9])$/u;
89
+ /**
90
+ * The outline level a built-in heading *name* implies (zero-based, so
91
+ * `heading 1` is 0), or undefined when the name is not a built-in heading.
92
+ */
93
+ const headingOutlineLevelFromStyleName = (name) => {
94
+ if (name === void 0) return;
95
+ const level = BUILT_IN_HEADING_NAME.exec(normalizeStyleName(name))?.groups?.["level"];
96
+ return level === void 0 ? void 0 : Number.parseInt(level, 10) - 1;
97
+ };
98
+ /**
99
+ * Every spelling a producer might write, mapped back to the canonical one, so
100
+ * a caller compares against {@link BUILT_IN_STYLE_NAME} rather than against a
101
+ * normalised form it would have to spell a second time.
102
+ */
103
+ const CANONICAL_BY_NORMALIZED = new Map(Object.values(BUILT_IN_STYLE_NAME).map((name) => [normalizeStyleName(name), name]));
104
+ const asBuiltInName = (normalized) => CANONICAL_BY_NORMALIZED.get(normalized);
105
+ /**
106
+ * The outline level a style chain sets: the style's own `w:outlineLvl`, else
107
+ * the nearest ancestor's (17.7.1). `styleParser` already flattens `w:basedOn`
108
+ * for a parsed package, but a style set built in memory carries the raw chain,
109
+ * so the walk keeps both kinds of input on the same answer. `seen` guards the
110
+ * circular `basedOn` a malformed package can contain.
111
+ */
112
+ const inheritedOutlineLevel = (style, styleById) => {
113
+ const seen = /* @__PURE__ */ new Set();
114
+ let current = style;
115
+ while (current && !seen.has(current.styleId)) {
116
+ seen.add(current.styleId);
117
+ if (current.pPr?.outlineLevel !== void 0) return current.pPr.outlineLevel;
118
+ current = current.basedOn === void 0 ? void 0 : styleById.get(current.basedOn);
119
+ }
120
+ };
121
+ const createBuiltInStyleIndex = (styles, docDefaults) => {
122
+ const styleIdByHeadingLevel = /* @__PURE__ */ new Map();
123
+ const styleIdByTableOfContentsLevel = /* @__PURE__ */ new Map();
124
+ const styleIdByBuiltInName = /* @__PURE__ */ new Map();
125
+ const paragraphStyles = [];
126
+ const paragraphStyleById = /* @__PURE__ */ new Map();
127
+ const styleById = /* @__PURE__ */ new Map();
128
+ for (const style of styles) {
129
+ styleById.set(style.styleId, style);
130
+ if (style.type === "paragraph") {
131
+ paragraphStyles.push(style);
132
+ paragraphStyleById.set(style.styleId, style);
133
+ }
134
+ }
135
+ const docDefaultOutlineLevel = docDefaults?.pPr?.outlineLevel;
136
+ for (const style of paragraphStyles) {
137
+ if (style.name === void 0) continue;
138
+ const namedLevel = headingOutlineLevelFromStyleName(style.name);
139
+ const headingLevel = namedLevel === void 0 ? void 0 : inheritedOutlineLevel(style, styleById) ?? docDefaultOutlineLevel ?? namedLevel;
140
+ if (headingLevel !== void 0 && isHeadingOutlineLevel(headingLevel) && !styleIdByHeadingLevel.has(headingLevel)) {
141
+ styleIdByHeadingLevel.set(headingLevel, style.styleId);
142
+ continue;
143
+ }
144
+ const normalized = normalizeStyleName(style.name);
145
+ const tocLevel = BUILT_IN_TABLE_OF_CONTENTS_NAME.exec(normalized)?.groups?.["level"];
146
+ if (tocLevel !== void 0) {
147
+ const level = Number.parseInt(tocLevel, 10);
148
+ if (!styleIdByTableOfContentsLevel.has(level)) styleIdByTableOfContentsLevel.set(level, style.styleId);
149
+ continue;
150
+ }
151
+ const builtInName = asBuiltInName(normalized);
152
+ if (builtInName !== void 0 && !styleIdByBuiltInName.has(builtInName)) styleIdByBuiltInName.set(builtInName, style.styleId);
153
+ }
154
+ /**
155
+ * The style a paragraph actually resolves against. ECMA-376 17.7.2 layer 3
156
+ * is "the paragraph's own style chain", and a paragraph with no `w:pStyle`
157
+ * (or one naming a style the package never defines, or one naming a
158
+ * character style) takes the default paragraph style — the same fallback
159
+ * `StyleResolver` applies, so the model and the editor cannot drift apart.
160
+ */
161
+ const defaultStyle = resolveDefaultParagraphStyle(paragraphStyles);
162
+ const styleFor = (styleId) => (styleId === null || styleId === void 0 ? void 0 : paragraphStyleById.get(styleId)) ?? defaultStyle;
163
+ const outlineLevelCache = /* @__PURE__ */ new Map();
164
+ return {
165
+ outlineLevelOf: (styleId) => {
166
+ if (outlineLevelCache.has(styleId)) return outlineLevelCache.get(styleId);
167
+ const style = styleFor(styleId);
168
+ const level = (style === void 0 ? void 0 : inheritedOutlineLevel(style, styleById)) ?? docDefaultOutlineLevel;
169
+ outlineLevelCache.set(styleId, level);
170
+ return level;
171
+ },
172
+ headingLevelFromNameOf: (styleId) => headingOutlineLevelFromStyleName(styleFor(styleId)?.name),
173
+ builtInNameOf: (styleId) => {
174
+ const name = styleFor(styleId)?.name;
175
+ return name === void 0 ? void 0 : asBuiltInName(normalizeStyleName(name));
176
+ },
177
+ undefinedBuiltInHeadingLevelOf: (styleId) => styleId === null || styleId === void 0 || paragraphStyleById.has(styleId) ? void 0 : headingOutlineLevelFromStyleName(styleId),
178
+ styleIdForHeadingLevel: (level) => styleIdByHeadingLevel.get(level),
179
+ styleIdForTableOfContentsLevel: (level) => styleIdByTableOfContentsLevel.get(level),
180
+ styleIdForBuiltInName: (name) => styleIdByBuiltInName.get(name)
181
+ };
182
+ };
183
+ /** An index over a document that defines no styles: every lookup misses. */
184
+ const EMPTY_BUILT_IN_STYLE_INDEX = createBuiltInStyleIndex([]);
185
+ /**
186
+ * The heading level a paragraph carries, zero-based (`heading 1` is 0), or
187
+ * undefined when it is not a heading.
188
+ *
189
+ * Precedence:
190
+ *
191
+ * 1. An effective outline level decides on its own, including
192
+ * {@link BODY_TEXT_OUTLINE_LEVEL}, which means "not a heading". A style
193
+ * named `heading 5` whose outline level is 0 is a level-1 heading; a style
194
+ * named `heading 3` reset to 9 is body text. The format gives the outline
195
+ * level to field calculation (17.3.1.20) and leaves the name to the UI
196
+ * (17.7.4.9), so the level is the one the document asserts.
197
+ * 2. Only when no outline level is set anywhere does the built-in `w:name`
198
+ * decide — the style is then inheriting the consumer's own built-in
199
+ * definition, which supplies the level.
200
+ * 3. Last resort, and only for a `w:pStyle` the package defines no style for:
201
+ * the id itself, read as the English built-in id. 17.7.4.17 makes a style
202
+ * without `w:customStyle` a built-in and lets an application recognise it
203
+ * "if the associated style ID is known", which is the one case where the id
204
+ * is all the information left. `defaultParagraphStyle.ts` keeps the same
205
+ * last tier for `Normal`. A document that defines its heading styles never
206
+ * reaches this, so it cannot override a name or an outline level — and a
207
+ * localized package never writes an English id to begin with.
208
+ *
209
+ * Two consequences of rule 1 are deliberate, not oversights.
210
+ *
211
+ * **An outline level on a style that is not a heading still makes a heading.**
212
+ * The corpus has 680 such occurrences across 161 files, including `Title` at
213
+ * level 0 (42×) and `Subtitle` at level 1 (23×), plus `H1`, `Sub-heading`,
214
+ * `index heading` and a `DSTOC1-1`…`DSTOC8-8` family. Setting the level is how
215
+ * a document asks for a paragraph to be outlined, and Word's navigation pane
216
+ * and a `TOC \u` field both honour it, so folio does not second-guess a
217
+ * document that asked. Suppressing `Title` here would mean folio deciding a
218
+ * document's outline differs from Word's.
219
+ *
220
+ * **An outline level that disagrees with a built-in heading name wins.** 26
221
+ * corpus styles do this (`heading 5` at level 0, `heading 3` at level 1, and
222
+ * so on), all from non-Word producers or hand-authored fixtures. The format
223
+ * gives the level to field calculation (17.3.1.20) and the name to the user
224
+ * interface (17.7.4.9), so the level is the machine-readable claim and the
225
+ * name is a label. This is the one rule below that was not confirmed against
226
+ * Word itself.
227
+ */
228
+ const resolveHeadingLevel = (paragraph, index) => {
229
+ const effective = paragraph.outlineLevel ?? index.outlineLevelOf(paragraph.styleId);
230
+ if (effective !== null && effective !== void 0) return isHeadingOutlineLevel(effective) ? effective : void 0;
231
+ return index.headingLevelFromNameOf(paragraph.styleId) ?? index.undefinedBuiltInHeadingLevelOf(paragraph.styleId);
232
+ };
233
+ /** True when the paragraph's style is Word's `Quote` or `Intense Quote`. */
234
+ const isQuoteStyle = (styleId, index) => {
235
+ const name = index.builtInNameOf(styleId);
236
+ return name === BUILT_IN_STYLE_NAME.quote || name === BUILT_IN_STYLE_NAME.intenseQuote;
237
+ };
238
+ //#endregion
239
+ export { BODY_TEXT_OUTLINE_LEVEL, BUILT_IN_STYLE_NAME, EMPTY_BUILT_IN_STYLE_INDEX, builtInHeadingStyleName, builtInTableOfContentsStyleName, createBuiltInStyleIndex, isHeadingOutlineLevel, isQuoteStyle, normalizeStyleName, resolveHeadingLevel };
@@ -1,8 +1,10 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
2
  //#region src/docx/commentIdNormalization.d.ts
3
+ /** The code this normalisation is reported under, owned here, not at the caller. */
4
+ declare const DUPLICATE_COMMENT_ID_WARNING: "duplicate-comment-id";
3
5
  type NormalizeCommentIdsResult = {
4
6
  droppedDuplicateComments: number;
5
7
  };
6
8
  declare const normalizeCommentIds: (comments: document_d_exports.Comment[]) => NormalizeCommentIdsResult;
7
9
  //#endregion
8
- export { NormalizeCommentIdsResult, normalizeCommentIds };
10
+ export { DUPLICATE_COMMENT_ID_WARNING, NormalizeCommentIdsResult, normalizeCommentIds };
@@ -1,4 +1,21 @@
1
+ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
1
2
  //#region src/docx/commentIdNormalization.ts
3
+ /**
4
+ * One comment per `w:id`, because that is all the body can address.
5
+ *
6
+ * `w:commentReference`, `w:commentRangeStart` and `w:commentRangeEnd` name a
7
+ * comment by its `w:id`, so two `w:comment` elements sharing an id make every
8
+ * marker that names it ambiguous. Word opens such a package and resolves each
9
+ * marker to the first definition, so folio keeps the first and drops the rest:
10
+ * no marker can address a later one, and re-numbering it would invent a
11
+ * comment nothing anchors. Microsoft's own conformance corpus ships a file
12
+ * that does this.
13
+ *
14
+ * Reply links survive unchanged. `w15:paraIdParent` resolves to a comment id
15
+ * before this runs, and that id still belongs to the definition that stayed.
16
+ */
17
+ /** The code this normalisation is reported under, owned here, not at the caller. */
18
+ const DUPLICATE_COMMENT_ID_WARNING = PARSE_WARNING_CODES.duplicateCommentId;
2
19
  const normalizeCommentIds = (comments) => {
3
20
  const seen = /* @__PURE__ */ new Set();
4
21
  let kept = 0;
@@ -13,4 +30,4 @@ const normalizeCommentIds = (comments) => {
13
30
  return { droppedDuplicateComments };
14
31
  };
15
32
  //#endregion
16
- export { normalizeCommentIds };
33
+ export { DUPLICATE_COMMENT_ID_WARNING, normalizeCommentIds };
@@ -1,5 +1,6 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
2
  import { StyleMap } from "./styleParser.js";
3
+ import { ParseContext } from "./parseContext.js";
3
4
  //#region src/docx/commentParser.d.ts
4
5
  type CommentExtendedInfo = {
5
6
  parentParaId?: string;
@@ -28,6 +29,6 @@ declare function parseCommentsExtended(xml: string): Map<string, CommentExtended
28
29
  * local time. If `commentsExtendedXml` is provided, reply-thread
29
30
  * parent links (`parentId`) and resolved state (`done`) are populated.
30
31
  */
31
- declare function parseComments(commentsXml: string | null, styles: StyleMap | null, theme: document_d_exports.Theme | null, rels: document_d_exports.RelationshipMap, media: Map<string, document_d_exports.MediaFile>, commentsExtensibleXml?: string | null, commentsExtendedXml?: string | null): document_d_exports.Comment[];
32
+ declare function parseComments(commentsXml: string | null, styles: StyleMap | null, theme: document_d_exports.Theme | null, rels: document_d_exports.RelationshipMap, media: Map<string, document_d_exports.MediaFile>, commentsExtensibleXml?: string | null, commentsExtendedXml?: string | null, context?: ParseContext): document_d_exports.Comment[];
32
33
  //#endregion
33
34
  export { CommentExtendedInfo, parseComments, parseCommentsExtended };
@@ -1,8 +1,29 @@
1
1
  import { parseParagraph } from "./paragraphParser.js";
2
2
  import { cloneParagraphWithPropertySource } from "./paragraphPropertySource.js";
3
3
  import { parseRunProperties } from "./runParser.js";
4
- import { findChild, getAttribute, getChildElements, getLocalName, parseXml } from "./xmlParser.js";
4
+ import { findChild, getAttribute, getChildElements, getLocalName, parseOnOffValue, parseXml } from "./xmlParser.js";
5
+ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
5
6
  //#region src/docx/commentParser.ts
7
+ /**
8
+ * Comment Parser - Parse comments.xml, commentsExtensible.xml, and
9
+ * commentsExtended.xml.
10
+ *
11
+ * - `comments.xml` (w) carries the comment author, local date, and body.
12
+ * - `commentsExtensible.xml` (w16cex, Word 2016+) carries reliable UTC
13
+ * timestamps via `w16cex:dateUtc` — Word's `w:date` is local time
14
+ * without an offset and so is ambiguous.
15
+ * - `commentsExtended.xml` (w15, Word 2013+) carries reply-thread
16
+ * parent links via `w15:paraIdParent` and the resolved/done state
17
+ * via `w15:done`. Cross-referenced via the `w14:paraId` on
18
+ * `w:comment` and the matching `w15:paraId` on `w15:commentEx`.
19
+ *
20
+ * OOXML Reference:
21
+ * - Comments: w:comments
22
+ * - Comment: w:comment (w:id, w:author, w:date, w:initials, w14:paraId)
23
+ * - Comment content: child w:p elements
24
+ */
25
+ /** The whole lexical form of `ST_DecimalNumber`: an optional sign and digits. */
26
+ const DECIMAL_NUMBER = /^[+-]?\d+$/u;
6
27
  const DEFAULT_ANNOTATION_REFERENCE_STYLE_ID = "CommentReference";
7
28
  const normalizeAnnotationReferenceFormatting = (formatting) => {
8
29
  if (formatting?.styleId === DEFAULT_ANNOTATION_REFERENCE_STYLE_ID && Object.keys(formatting).length === 1) return;
@@ -65,10 +86,7 @@ function parseCommentsExtended(xml) {
65
86
  const doneAttr = getAttribute(child, "w15", "done") ?? child.attributes?.["w15:done"];
66
87
  const info = {};
67
88
  if (parentParaId) info.parentParaId = String(parentParaId).toUpperCase();
68
- if (doneAttr !== void 0) {
69
- const v = String(doneAttr).toLowerCase();
70
- info.done = v === "1" || v === "true";
71
- }
89
+ if (doneAttr !== void 0) info.done = parseOnOffValue(String(doneAttr).toLowerCase()) ?? false;
72
90
  infoByParaId.set(String(paraId).toUpperCase(), info);
73
91
  }
74
92
  return infoByParaId;
@@ -81,7 +99,7 @@ function parseCommentsExtended(xml) {
81
99
  * local time. If `commentsExtendedXml` is provided, reply-thread
82
100
  * parent links (`parentId`) and resolved state (`done`) are populated.
83
101
  */
84
- function parseComments(commentsXml, styles, theme, rels, media, commentsExtensibleXml, commentsExtendedXml) {
102
+ function parseComments(commentsXml, styles, theme, rels, media, commentsExtensibleXml, commentsExtendedXml, context) {
85
103
  if (!commentsXml) return [];
86
104
  const root = parseXml(commentsXml);
87
105
  const dateUtcByParaId = commentsExtensibleXml ? parseCommentsExtensible(commentsExtensibleXml) : /* @__PURE__ */ new Map();
@@ -92,7 +110,16 @@ function parseComments(commentsXml, styles, theme, rels, media, commentsExtensib
92
110
  const paraIdByCommentIndex = /* @__PURE__ */ new Map();
93
111
  for (const child of children) {
94
112
  if ((child.name?.replace(/^.*:/u, "") ?? "") !== "comment") continue;
95
- const id = Number.parseInt(getAttribute(child, "w", "id") ?? "0", 10);
113
+ const rawId = getAttribute(child, "w", "id");
114
+ const id = rawId !== null && DECIMAL_NUMBER.test(rawId) ? Number.parseInt(rawId, 10) : NaN;
115
+ if (Number.isNaN(id)) {
116
+ context?.warn({
117
+ code: PARSE_WARNING_CODES.missingCommentId,
118
+ element: "w:comment",
119
+ ...rawId === null ? {} : { value: rawId }
120
+ });
121
+ continue;
122
+ }
96
123
  const rawAuthor = getAttribute(child, "w", "author");
97
124
  const author = parseCommentAuthor(rawAuthor);
98
125
  const rawInitials = getAttribute(child, "w", "initials");
@@ -1,5 +1,8 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
2
  //#region src/docx/commentReferenceNormalization.d.ts
3
+ /** The codes this normalisation is reported under, owned here, not at the caller. */
4
+ declare const DANGLING_COMMENT_REFERENCE_WARNING: "dangling-comment-reference";
5
+ declare const UNBALANCED_COMMENT_RANGE_WARNING: "unbalanced-comment-range";
3
6
  type NormalizeCommentReferencesInput = {
4
7
  documentBody: document_d_exports.DocumentBody;
5
8
  comments: readonly document_d_exports.Comment[];
@@ -14,4 +17,4 @@ type NormalizeCommentReferencesResult = {
14
17
  };
15
18
  declare const normalizeCommentReferences: ({ documentBody, comments, headers, footers, footnotes, endnotes }: NormalizeCommentReferencesInput) => NormalizeCommentReferencesResult;
16
19
  //#endregion
17
- export { normalizeCommentReferences };
20
+ export { DANGLING_COMMENT_REFERENCE_WARNING, UNBALANCED_COMMENT_RANGE_WARNING, normalizeCommentReferences };
@@ -1,24 +1,31 @@
1
+ import { InlineContentRemovals, visitInlineContentSlots } from "./paragraphTraversal.js";
2
+ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
1
3
  //#region src/docx/commentReferenceNormalization.ts
4
+ /** The codes this normalisation is reported under, owned here, not at the caller. */
5
+ const DANGLING_COMMENT_REFERENCE_WARNING = PARSE_WARNING_CODES.danglingCommentReference;
6
+ const UNBALANCED_COMMENT_RANGE_WARNING = PARSE_WARNING_CODES.unbalancedCommentRange;
2
7
  const normalizeCommentReferences = ({ documentBody, comments, headers, footers, footnotes, endnotes }) => {
3
8
  const validCommentIds = new Set(comments.map((comment) => comment.id));
4
9
  const rangeMarkers = [];
10
+ const dangling = new InlineContentRemovals();
5
11
  let removedDanglingReferences = 0;
6
12
  const normalizeParagraph = (paragraph) => {
7
- const nextContent = [];
8
- for (const content of paragraph.content) {
9
- if (isCommentMarker(content) && !validCommentIds.has(content.id)) {
13
+ visitInlineContentSlots(paragraph, ({ content, index, item }) => {
14
+ if (isCommentMarker(item) && !validCommentIds.has(item.id)) {
15
+ dangling.mark({
16
+ content,
17
+ index
18
+ });
10
19
  removedDanglingReferences += 1;
11
- continue;
20
+ return;
12
21
  }
13
- nextContent.push(content);
14
- if (isCommentRangeMarker(content)) rangeMarkers.push({
15
- content: nextContent,
16
- index: nextContent.length - 1,
17
- id: content.id,
18
- type: content.type
22
+ if (isCommentRangeMarker(item)) rangeMarkers.push({
23
+ content,
24
+ index,
25
+ id: item.id,
26
+ type: item.type
19
27
  });
20
- }
21
- paragraph.content = nextContent;
28
+ });
22
29
  };
23
30
  const normalizeTable = (table) => {
24
31
  for (const row of table.rows) for (const cell of row.cells) normalizeBlocks(cell.content);
@@ -43,9 +50,11 @@ const normalizeCommentReferences = ({ documentBody, comments, headers, footers,
43
50
  for (const footnote of footnotes ?? []) normalizeBlocks(footnote.content);
44
51
  for (const endnote of endnotes ?? []) normalizeBlocks(endnote.content);
45
52
  for (const comment of comments) for (const paragraph of comment.content) normalizeParagraph(paragraph);
53
+ const reanchoredUnbalancedRanges = reanchorUnbalancedCommentRanges(rangeMarkers);
54
+ dangling.apply();
46
55
  return {
47
56
  removedDanglingReferences,
48
- reanchoredUnbalancedRanges: reanchorUnbalancedCommentRanges(rangeMarkers)
57
+ reanchoredUnbalancedRanges
49
58
  };
50
59
  };
51
60
  const reanchorUnbalancedCommentRanges = (rangeMarkers) => {
@@ -94,4 +103,4 @@ const reanchorCommentRangeMarker = (marker) => {
94
103
  const isCommentMarker = (content) => content.type === "commentRangeStart" || content.type === "commentRangeEnd" || content.type === "commentReference";
95
104
  const isCommentRangeMarker = (content) => content.type === "commentRangeStart" || content.type === "commentRangeEnd";
96
105
  //#endregion
97
- export { normalizeCommentReferences };
106
+ export { DANGLING_COMMENT_REFERENCE_WARNING, UNBALANCED_COMMENT_RANGE_WARNING, normalizeCommentReferences };
@@ -0,0 +1,15 @@
1
+ import { document_d_exports } from "../types/document.js";
2
+ //#region src/docx/danglingRelationshipReferences.d.ts
3
+ type DanglingRelationshipReferences = {
4
+ /** Drawings whose `r:embed` names no relationship. */
5
+ drawings: number;
6
+ /** Hyperlinks whose `r:id` names no relationship. */
7
+ hyperlinks: number;
8
+ };
9
+ type CountDanglingRelationshipReferencesOptions = {
10
+ content: readonly document_d_exports.BlockContent[];
11
+ relationships: document_d_exports.RelationshipMap | undefined;
12
+ };
13
+ declare const countDanglingRelationshipReferences: ({ content, relationships }: CountDanglingRelationshipReferencesOptions) => DanglingRelationshipReferences;
14
+ //#endregion
15
+ export { DanglingRelationshipReferences, countDanglingRelationshipReferences };
@@ -0,0 +1,30 @@
1
+ import { resolveRelationshipId } from "./relsParser.js";
2
+ //#region src/docx/danglingRelationshipReferences.ts
3
+ const countDanglingRelationshipReferences = ({ content, relationships }) => {
4
+ const counts = {
5
+ drawings: 0,
6
+ hyperlinks: 0
7
+ };
8
+ const isDangling = (rId) => resolveRelationshipId(relationships, rId).status === "dangling";
9
+ const visitBlocks = (blocks) => {
10
+ for (const block of blocks) {
11
+ if (block.type === "table") {
12
+ for (const row of block.rows) for (const cell of row.cells) visitBlocks(cell.content);
13
+ continue;
14
+ }
15
+ if (block.type !== "paragraph") continue;
16
+ for (const item of block.content) {
17
+ if (item.type === "hyperlink") {
18
+ if (isDangling(item.rId)) counts.hyperlinks += 1;
19
+ continue;
20
+ }
21
+ if (item.type !== "run") continue;
22
+ for (const child of item.content) if (child.type === "drawing" && isDangling(child.image.rId)) counts.drawings += 1;
23
+ }
24
+ }
25
+ };
26
+ visitBlocks(content);
27
+ return counts;
28
+ };
29
+ //#endregion
30
+ export { countDanglingRelationshipReferences };