@stll/folio-core 0.43.0 → 0.44.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (183) hide show
  1. package/dist/ai-edits/headless.js +6 -5
  2. package/dist/ai-edits/index.d.ts +2 -2
  3. package/dist/ai-edits/index.js +2 -2
  4. package/dist/ai-edits/snapshot.js +13 -9
  5. package/dist/compare/content-alignment.js +16 -1
  6. package/dist/compare/inline-atoms.js +1 -1
  7. package/dist/compare/style-resources.js +6 -0
  8. package/dist/docx/appVersionNormalization.d.ts +0 -18
  9. package/dist/docx/blockContentParser.js +8 -0
  10. package/dist/docx/blockRangeMarkers.d.ts +36 -0
  11. package/dist/docx/blockRangeMarkers.js +59 -0
  12. package/dist/docx/bookmarkParser.d.ts +2 -20
  13. package/dist/docx/bookmarkParser.js +6 -30
  14. package/dist/docx/borderParser.d.ts +13 -0
  15. package/dist/docx/borderParser.js +71 -0
  16. package/dist/docx/builtInStyles.d.ts +165 -0
  17. package/dist/docx/builtInStyles.js +239 -0
  18. package/dist/docx/commentIdNormalization.d.ts +3 -1
  19. package/dist/docx/commentIdNormalization.js +18 -1
  20. package/dist/docx/commentParser.d.ts +2 -1
  21. package/dist/docx/commentParser.js +34 -7
  22. package/dist/docx/commentReferenceNormalization.d.ts +4 -1
  23. package/dist/docx/commentReferenceNormalization.js +23 -14
  24. package/dist/docx/danglingRelationshipReferences.d.ts +15 -0
  25. package/dist/docx/danglingRelationshipReferences.js +30 -0
  26. package/dist/docx/defaultParagraphStyle.d.ts +18 -1
  27. package/dist/docx/defaultParagraphStyle.js +23 -1
  28. package/dist/docx/documentParser.d.ts +2 -1
  29. package/dist/docx/documentParser.js +2 -2
  30. package/dist/docx/drawingUtils.d.ts +8 -1
  31. package/dist/docx/drawingUtils.js +12 -3
  32. package/dist/docx/fieldParser.js +3 -5
  33. package/dist/docx/footnoteParser.d.ts +3 -2
  34. package/dist/docx/footnoteParser.js +19 -4
  35. package/dist/docx/groupDrawingParser.js +1 -1
  36. package/dist/docx/headerFooterRefParser.d.ts +4 -3
  37. package/dist/docx/headerFooterRefParser.js +42 -12
  38. package/dist/docx/headerFooterReferenceNormalization.d.ts +4 -1
  39. package/dist/docx/headerFooterReferenceNormalization.js +5 -1
  40. package/dist/docx/hyperlinkParser.js +11 -15
  41. package/dist/docx/imageParser.d.ts +1 -1
  42. package/dist/docx/imageParser.js +22 -18
  43. package/dist/docx/imageRawXml.js +5 -5
  44. package/dist/docx/markupRangeMarker.d.ts +15 -0
  45. package/dist/docx/markupRangeMarker.js +44 -0
  46. package/dist/docx/noteReferenceStyles.d.ts +29 -0
  47. package/dist/docx/noteReferenceStyles.js +70 -0
  48. package/dist/docx/numberingReferenceNormalization.d.ts +4 -1
  49. package/dist/docx/numberingReferenceNormalization.js +20 -1
  50. package/dist/docx/paraIdRangeNormalization.d.ts +0 -19
  51. package/dist/docx/paragraphParser.js +66 -99
  52. package/dist/docx/paragraphPropertySource.js +1 -0
  53. package/dist/docx/paragraphTraversal.d.ts +37 -1
  54. package/dist/docx/paragraphTraversal.js +84 -1
  55. package/dist/docx/parseContext.d.ts +37 -0
  56. package/dist/docx/parseContext.js +67 -0
  57. package/dist/docx/parseWarningMessage.d.ts +6 -0
  58. package/dist/docx/parseWarningMessage.js +44 -0
  59. package/dist/docx/parser.js +81 -27
  60. package/dist/docx/relsParser.d.ts +28 -11
  61. package/dist/docx/relsParser.js +26 -13
  62. package/dist/docx/revisionIdNormalization.js +81 -7
  63. package/dist/docx/rezip.js +63 -22
  64. package/dist/docx/runConsolidator.js +1 -2
  65. package/dist/docx/runParser.d.ts +8 -1
  66. package/dist/docx/runParser.js +30 -48
  67. package/dist/docx/sectionParser.d.ts +2 -1
  68. package/dist/docx/sectionParser.js +21 -65
  69. package/dist/docx/serializer/borderSerializer.d.ts +1 -2
  70. package/dist/docx/serializer/commentSerializer.js +22 -9
  71. package/dist/docx/serializer/documentSerializer.d.ts +1 -5
  72. package/dist/docx/serializer/documentSerializer.js +6 -16
  73. package/dist/docx/serializer/headerFooterSerializer.js +5 -0
  74. package/dist/docx/serializer/markupRangeAttributes.d.ts +8 -0
  75. package/dist/docx/serializer/markupRangeAttributes.js +24 -0
  76. package/dist/docx/serializer/noteSerializer.js +5 -0
  77. package/dist/docx/serializer/paragraphSerializer.d.ts +1 -5
  78. package/dist/docx/serializer/paragraphSerializer.js +29 -35
  79. package/dist/docx/serializer/runSerializer.js +13 -7
  80. package/dist/docx/serializer/tableSerializer.js +28 -13
  81. package/dist/docx/serializer/textFormattingSerializer.d.ts +2 -3
  82. package/dist/docx/server/build.js +8 -1
  83. package/dist/docx/server/createBilingualDocument.js +10 -18
  84. package/dist/docx/server/extractDocxText.js +3 -4
  85. package/dist/docx/shadingParser.d.ts +6 -0
  86. package/dist/docx/shadingParser.js +32 -0
  87. package/dist/docx/shapeParser.js +3 -3
  88. package/dist/docx/styleParser.js +13 -87
  89. package/dist/docx/styleReferenceResolution.d.ts +36 -0
  90. package/dist/docx/styleReferenceResolution.js +51 -0
  91. package/dist/docx/tableLook.d.ts +57 -0
  92. package/dist/docx/tableLook.js +63 -0
  93. package/dist/docx/tableParser.d.ts +7 -9
  94. package/dist/docx/tableParser.js +64 -110
  95. package/dist/docx/textBoxParser.js +4 -4
  96. package/dist/docx/trackedMoveRangeNormalization.d.ts +3 -1
  97. package/dist/docx/trackedMoveRangeNormalization.js +11 -21
  98. package/dist/docx/transitionalSpelling.d.ts +13 -2
  99. package/dist/docx/transitionalSpelling.js +23 -1
  100. package/dist/docx/verbatimCapture.js +4 -11
  101. package/dist/docx/vmlImageParser.js +2 -2
  102. package/dist/docx/watermarkParser.js +2 -2
  103. package/dist/docx/xmlParser.d.ts +22 -32
  104. package/dist/docx/xmlParser.js +36 -21
  105. package/dist/internal/pageBreakRunSourceDescendantIndex.js +2 -1
  106. package/dist/internal/paragraphFormattingSerialization.d.ts +2 -3
  107. package/dist/internal/paragraphFormattingSerialization.js +26 -6
  108. package/dist/layout-bridge/convert/footnoteLayout.js +2 -7
  109. package/dist/layout-engine/index.d.ts +2 -2
  110. package/dist/layout-engine/index.js +2 -2
  111. package/dist/layout-engine/measure/measureBlocks.js +1 -6
  112. package/dist/layout-engine/types.d.ts +8 -2
  113. package/dist/layout-engine/types.js +35 -2
  114. package/dist/markdown/index.js +1 -1
  115. package/dist/markdown/internals.d.ts +6 -1
  116. package/dist/markdown/internals.js +14 -1
  117. package/dist/markdown/renderBlock.js +35 -21
  118. package/dist/markdown/renderParagraph.js +14 -5
  119. package/dist/markdown/renderRuns.js +4 -3
  120. package/dist/markdown/renderTable.js +4 -3
  121. package/dist/markdown/trailers.js +41 -7
  122. package/dist/markdown/types.d.ts +3 -7
  123. package/dist/prosemirror/attrs/index.js +2 -5
  124. package/dist/prosemirror/bookmarkBoundaryAttrs.d.ts +11 -1
  125. package/dist/prosemirror/bookmarkBoundaryAttrs.js +18 -3
  126. package/dist/prosemirror/commands/index.d.ts +3 -3
  127. package/dist/prosemirror/commands/index.js +2 -2
  128. package/dist/prosemirror/commands/paragraph.d.ts +3 -3
  129. package/dist/prosemirror/commands/paragraph.js +2 -2
  130. package/dist/prosemirror/commentIdAllocator.js +2 -7
  131. package/dist/prosemirror/conversion/fromProseDoc.js +129 -40
  132. package/dist/prosemirror/conversion/toProseDoc.d.ts +1 -14
  133. package/dist/prosemirror/conversion/toProseDoc.js +375 -316
  134. package/dist/prosemirror/extensions/core/ParagraphExtension.d.ts +14 -1
  135. package/dist/prosemirror/extensions/core/ParagraphExtension.js +11 -6
  136. package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -3
  137. package/dist/prosemirror/extensions/features/PasteCleanupExtension.d.ts +4 -1
  138. package/dist/prosemirror/extensions/features/PasteCleanupExtension.js +6 -2
  139. package/dist/prosemirror/extensions/features/pastedHeadingStyles.d.ts +7 -0
  140. package/dist/prosemirror/extensions/features/pastedHeadingStyles.js +74 -0
  141. package/dist/prosemirror/extensions/marks/markUtils.d.ts +11 -3
  142. package/dist/prosemirror/extensions/marks/markUtils.js +98 -19
  143. package/dist/prosemirror/extensions/nodes/BookmarkBoundaryExtension.js +7 -3
  144. package/dist/prosemirror/extensions/nodes/ImageExtension.js +2 -1
  145. package/dist/prosemirror/extensions/nodes/ShapeExtension.js +1 -0
  146. package/dist/prosemirror/extensions/nodes/TableExtension.js +15 -1
  147. package/dist/prosemirror/extensions/types.d.ts +2 -2
  148. package/dist/prosemirror/index.d.ts +3 -3
  149. package/dist/prosemirror/index.js +3 -3
  150. package/dist/prosemirror/insertOperations.d.ts +9 -2
  151. package/dist/prosemirror/insertOperations.js +9 -4
  152. package/dist/prosemirror/paragraphFormattingProvenance.d.ts +159 -0
  153. package/dist/prosemirror/paragraphFormattingProvenance.js +106 -0
  154. package/dist/prosemirror/plugins/documentStyles.d.ts +9 -1
  155. package/dist/prosemirror/plugins/documentStyles.js +11 -1
  156. package/dist/prosemirror/plugins/index.d.ts +2 -2
  157. package/dist/prosemirror/plugins/index.js +2 -2
  158. package/dist/prosemirror/plugins/revisionIds.d.ts +11 -2
  159. package/dist/prosemirror/plugins/revisionIds.js +21 -6
  160. package/dist/prosemirror/runFormattingReconciliation.js +3 -2
  161. package/dist/prosemirror/runStyleFormatting.d.ts +1 -1
  162. package/dist/prosemirror/schema/nodes.d.ts +31 -0
  163. package/dist/prosemirror/styles/resolvedStyleAttrs.js +2 -0
  164. package/dist/prosemirror/styles/styleResolver.d.ts +9 -0
  165. package/dist/prosemirror/styles/styleResolver.js +12 -0
  166. package/dist/style-engine/styleEngine.d.ts +3 -0
  167. package/dist/style-engine/styleEngine.js +3 -0
  168. package/dist/style-sets/extract.js +1 -23
  169. package/dist/style-sets/stellaStyle.js +46 -39
  170. package/dist/style-sets/styleSetNormalization.d.ts +19 -0
  171. package/dist/style-sets/styleSetNormalization.js +99 -0
  172. package/dist/types/content.d.ts +2 -2
  173. package/dist/utils/createDocument.js +145 -20
  174. package/dist/utils/headingCollector.d.ts +8 -5
  175. package/dist/utils/headingCollector.js +23 -25
  176. package/dist/utils/tableOfContentsStyle.js +9 -2
  177. package/package.json +2 -2
  178. package/dist/docx/textWhitespace.d.ts +0 -4
  179. package/dist/docx/textWhitespace.js +0 -4
  180. package/dist/layout-bridge/engine/tableWidthUtils.d.ts +0 -6
  181. package/dist/layout-bridge/engine/tableWidthUtils.js +0 -25
  182. package/dist/markdown/headings.d.ts +0 -13
  183. package/dist/markdown/headings.js +0 -20
@@ -0,0 +1,37 @@
1
+ import { ParseWarning, ParseWarningCode } from "@stll/docx-core/model";
2
+ //#region src/docx/parseContext.d.ts
3
+ /** What a call site states; the collector supplies the rest. */
4
+ type ParseWarningReport = {
5
+ code: ParseWarningCode;
6
+ /** The element as written, prefix included. */
7
+ element?: string;
8
+ /** The best position this part can name, e.g. `style "Heading1"`. */
9
+ at?: string;
10
+ /** The value folio declined to read, as written. */
11
+ value?: string;
12
+ /** Text passed through from another owner, for the pass-through codes. */
13
+ detail?: string;
14
+ /** How many occurrences this one report stands for; defaults to 1. */
15
+ count?: number;
16
+ };
17
+ type ParseContext = {
18
+ /** Record one normalisation applied to input folio accepted. */
19
+ warn: (report: ParseWarningReport) => void;
20
+ /** The same collector, reporting against another part or position. */
21
+ scoped: (location: {
22
+ part?: string;
23
+ at?: string;
24
+ }) => ParseContext;
25
+ };
26
+ type ParseWarningCollector = {
27
+ context: ParseContext;
28
+ /**
29
+ * Everything recorded, in the order it was recorded, followed by one entry
30
+ * per code whose occurrences ran past the cap. Deterministic: the same
31
+ * document yields the same list.
32
+ */
33
+ warnings: () => ParseWarning[];
34
+ };
35
+ declare const createParseWarningCollector: (part?: string) => ParseWarningCollector;
36
+ //#endregion
37
+ export { ParseContext, ParseWarningCollector, ParseWarningReport, createParseWarningCollector };
@@ -0,0 +1,67 @@
1
+ import { MAX_RETAINED_PARSE_WARNINGS_PER_CODE } from "@stll/docx-core/model";
2
+ //#region src/docx/parseContext.ts
3
+ /**
4
+ * The channel a parser reports a normalisation through.
5
+ *
6
+ * Folio accepts input Word accepts, which means normalising at the parse
7
+ * boundary rather than refusing. Every such decision has to be visible, and
8
+ * before this the only place that could say so was `parseDocx` itself: the
9
+ * leaf readers are pure functions with no way to report, so a value outside
10
+ * `ST_OnOff` or a `w:type` outside `ST_HdrFtr` was normalised in silence.
11
+ *
12
+ * The context is an explicit parameter, never a module-level accumulator and
13
+ * never async-local storage. Parsers stay re-entrant, a test can hand one in
14
+ * and read what a single reader reported, and two documents parsed at once
15
+ * cannot write into each other's list. The cost is a threaded argument, which
16
+ * is also the thing that makes the reporting greppable.
17
+ */
18
+ const PACKAGE_PART = "package";
19
+ const createParseWarningCollector = (part = PACKAGE_PART) => {
20
+ const retained = [];
21
+ const retainedByCode = /* @__PURE__ */ new Map();
22
+ const suppressedByCode = /* @__PURE__ */ new Map();
23
+ const record = (location, report) => {
24
+ const count = report.count ?? 1;
25
+ const kept = retainedByCode.get(report.code) ?? 0;
26
+ if (kept >= MAX_RETAINED_PARSE_WARNINGS_PER_CODE) {
27
+ suppressedByCode.set(report.code, (suppressedByCode.get(report.code) ?? 0) + count);
28
+ return;
29
+ }
30
+ retainedByCode.set(report.code, kept + 1);
31
+ retained.push({
32
+ code: report.code,
33
+ location: {
34
+ ...location,
35
+ ...report.element === void 0 ? {} : { element: report.element },
36
+ ...report.at === void 0 ? {} : { at: report.at }
37
+ },
38
+ ...report.value === void 0 ? {} : { value: report.value },
39
+ ...report.detail === void 0 ? {} : { detail: report.detail },
40
+ count
41
+ });
42
+ };
43
+ const contextAt = (location) => ({
44
+ warn: (report) => {
45
+ record(location, report);
46
+ },
47
+ scoped: (next) => {
48
+ const at = next.at ?? location.at;
49
+ return contextAt({
50
+ part: next.part ?? location.part,
51
+ ...location.element === void 0 ? {} : { element: location.element },
52
+ ...at === void 0 ? {} : { at }
53
+ });
54
+ }
55
+ });
56
+ return {
57
+ context: contextAt({ part }),
58
+ warnings: () => [...retained, ...[...suppressedByCode.entries()].sort(([left], [right]) => left < right ? -1 : 1).map(([code, count]) => ({
59
+ code,
60
+ location: { part },
61
+ count,
62
+ detail: `${String(count)} further occurrence(s) were counted but not retained.`
63
+ }))]
64
+ };
65
+ };
66
+ //#endregion
67
+ export { createParseWarningCollector };
@@ -0,0 +1,6 @@
1
+ import { ParseWarning } from "@stll/docx-core/model";
2
+ //#region src/docx/parseWarningMessage.d.ts
3
+ declare const formatParseWarning: (warning: ParseWarning) => string;
4
+ declare const formatParseWarnings: (warnings: readonly ParseWarning[]) => string[];
5
+ //#endregion
6
+ export { formatParseWarning, formatParseWarnings };
@@ -0,0 +1,44 @@
1
+ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
2
+ //#region src/docx/parseWarningMessage.ts
3
+ /**
4
+ * The one place a parse warning becomes a sentence.
5
+ *
6
+ * `Document.warnings` is the string list hosts have always read, and it is now
7
+ * rendered from `Document.parseWarnings` rather than written at the call site,
8
+ * so a warning cannot say one thing in the structured list and another in the
9
+ * prose. The map is total over the code union: a new code does not compile
10
+ * until it has a message.
11
+ */
12
+ const plural = (count, singular, pluralForm = `${singular}s`) => `${String(count)} ${count === 1 ? singular : pluralForm}`;
13
+ /** The location suffix, omitted when the part is all we know. */
14
+ const where = ({ location }) => {
15
+ if (location.at === void 0) return "";
16
+ return ` at ${location.at} in ${location.part}`;
17
+ };
18
+ const quoted = (value) => value === void 0 ? "" : ` "${value}"`;
19
+ const PARSE_WARNING_MESSAGES = {
20
+ [PARSE_WARNING_CODES.packageDecrypted]: () => "Document was opened from password-protected storage; saving writes an unencrypted .docx file.",
21
+ [PARSE_WARNING_CODES.packageArchive]: (warning) => warning.detail ?? "The archive reader reported an issue.",
22
+ [PARSE_WARNING_CODES.documentPartMissing]: () => "No document.xml found in DOCX",
23
+ [PARSE_WARNING_CODES.documentModelIssue]: (warning) => warning.detail ?? "The document model reported an issue.",
24
+ [PARSE_WARNING_CODES.duplicateCommentId]: (warning) => `Dropped ${plural(warning.count, "comment")} repeating a w:id another comment already defines.`,
25
+ [PARSE_WARNING_CODES.missingCommentId]: (warning) => `Dropped ${plural(warning.count, "comment")} with no readable w:id.`,
26
+ [PARSE_WARNING_CODES.duplicateNoteId]: (warning) => `Dropped ${plural(warning.count, "note")} repeating a w:id another note already defines.`,
27
+ [PARSE_WARNING_CODES.danglingCommentReference]: (warning) => `Removed ${plural(warning.count, "dangling comment reference marker")} whose comments.xml entries are missing.`,
28
+ [PARSE_WARNING_CODES.unbalancedCommentRange]: (warning) => `Re-anchored ${plural(warning.count, "unbalanced comment range marker")} as point comments.`,
29
+ [PARSE_WARNING_CODES.danglingHeaderReference]: (warning) => `Removed ${plural(warning.count, "dangling header reference")} whose header parts are missing.`,
30
+ [PARSE_WARNING_CODES.danglingFooterReference]: (warning) => `Removed ${plural(warning.count, "dangling footer reference")} whose footer parts are missing.`,
31
+ [PARSE_WARNING_CODES.danglingRelationshipId]: (warning) => `Left relationship id${quoted(warning.value)} unresolved${where(warning)}; the part defines no such relationship.`,
32
+ [PARSE_WARNING_CODES.unnumberedParagraph]: (warning) => `Unnumbered ${plural(warning.count, "paragraph")} whose numbering definitions are missing.`,
33
+ [PARSE_WARNING_CODES.unnumberedStyle]: (warning) => `Unnumbered style${quoted(warning.value)} whose numbering definition is missing.`,
34
+ [PARSE_WARNING_CODES.unbalancedMoveRange]: (warning) => `Removed ${plural(warning.count, "unbalanced tracked move range marker")}.`,
35
+ [PARSE_WARNING_CODES.headerFooterTypeOutsideEnum]: (warning) => `Read header/footer type${quoted(warning.value)} as "default"${where(warning)}; ST_HdrFtr is even, default or first.`,
36
+ [PARSE_WARNING_CODES.unrecognisedOnOffValue]: (warning) => `Ignored on/off value${quoted(warning.value)}${where(warning)}; ST_OnOff is 1, 0, true, false, on or off.`,
37
+ [PARSE_WARNING_CODES.borderWithoutValue]: (warning) => `Read a border with no w:val${where(warning)} as having no border style.`,
38
+ [PARSE_WARNING_CODES.styleSetDuplicateStyleId]: (warning) => `Dropped a style repeating the id${quoted(warning.value)} another style in the set already defines.`,
39
+ [PARSE_WARNING_CODES.styleSetInitialStyleMissing]: (warning) => `The style set names initial paragraph style${quoted(warning.value)}, which it does not contain; used the set's default instead${where(warning)}.`
40
+ };
41
+ const formatParseWarning = (warning) => PARSE_WARNING_MESSAGES[warning.code](warning);
42
+ const formatParseWarnings = (warnings) => warnings.map(formatParseWarning);
43
+ //#endregion
44
+ export { formatParseWarning, formatParseWarnings };
@@ -1,35 +1,39 @@
1
1
  import { toArrayBuffer } from "../utils/docxInput.js";
2
2
  import { loadFontsWithMapping } from "../utils/fontLoader.js";
3
3
  import { MAX_PACKAGE_TIFF_PIXELS, convertTiffToPngDataUrl, isTiffMimeType } from "../utils/tiffConverter.js";
4
- import { normalizeCommentIds } from "./commentIdNormalization.js";
4
+ import { DUPLICATE_COMMENT_ID_WARNING, normalizeCommentIds } from "./commentIdNormalization.js";
5
5
  import { parseComments } from "./commentParser.js";
6
- import { normalizeCommentReferences } from "./commentReferenceNormalization.js";
6
+ import { DANGLING_COMMENT_REFERENCE_WARNING, UNBALANCED_COMMENT_RANGE_WARNING, normalizeCommentReferences } from "./commentReferenceNormalization.js";
7
7
  import { detectDocxConformanceClass } from "./conformance.js";
8
8
  import { parseCoreProperties } from "./corePropertiesParser.js";
9
+ import { countDanglingRelationshipReferences } from "./danglingRelationshipReferences.js";
9
10
  import { extractAllTemplateVariables, parseDocumentBody } from "./documentParser.js";
10
11
  import { normalizeDrawingIds } from "./drawingIdNormalization.js";
11
12
  import { DocxEncryptionError } from "./encryption/errors.js";
12
13
  import { parseFontTable } from "./fontTableParser.js";
13
14
  import { parseEndnotes, parseFootnotes } from "./footnoteParser.js";
14
15
  import { parseFooter, parseHeader } from "./headerFooterParser.js";
15
- import { normalizeHeaderFooterReferences } from "./headerFooterReferenceNormalization.js";
16
+ import { DANGLING_FOOTER_REFERENCE_WARNING, DANGLING_HEADER_REFERENCE_WARNING, normalizeHeaderFooterReferences } from "./headerFooterReferenceNormalization.js";
16
17
  import { assignHeaderFooterVerbatimXml, refreshHeaderFooterVerbatimFingerprint } from "./headerFooterVerbatim.js";
17
18
  import { extractMetafileRaster, isMetafileMimeType } from "./metafileRaster.js";
18
19
  import { renderEmfSvg } from "./metafileSvg.js";
19
20
  import { DocxModelValidationError, formatDocumentModelIssues, validateFolioDocumentModel } from "./modelValidation.js";
20
21
  import { parseNumbering } from "./numberingParser.js";
21
- import { normalizeNumberingReferences, normalizeStyleNumberingReferences } from "./numberingReferenceNormalization.js";
22
+ import { UNNUMBERED_PARAGRAPH_WARNING, UNNUMBERED_STYLE_WARNING, normalizeNumberingReferences, normalizeStyleNumberingReferences } from "./numberingReferenceNormalization.js";
22
23
  import { assignDocumentParagraphPropertySourceContract } from "./paragraphPropertySource.js";
24
+ import { createParseWarningCollector } from "./parseContext.js";
25
+ import { formatParseWarnings } from "./parseWarningMessage.js";
23
26
  import { RELATIONSHIP_TYPES, parseRelationships, resolveRelativePath } from "./relsParser.js";
24
27
  import { normalizeRenderedPageBreakHints } from "./renderedPageBreakNormalization.js";
25
28
  import { parseSettings } from "./settingsParser.js";
26
29
  import { parseStylesPackage } from "./styleParser.js";
27
30
  import { applyThemeFontLang, parseTheme } from "./themeParser.js";
28
- import { normalizeTrackedMoveRanges } from "./trackedMoveRangeNormalization.js";
31
+ import { UNBALANCED_MOVE_RANGE_WARNING, normalizeTrackedMoveRanges } from "./trackedMoveRangeNormalization.js";
29
32
  import { getMediaMimeType, mediaToDataUrl, unzipDocx } from "./unzip.js";
30
33
  import { enforcePackageVmlPreviewBudget } from "./vmlPreview.js";
31
34
  import { FOLIO_XML_RESOURCE_LIMITS } from "./xmlResourceLimits.js";
32
35
  import { TaggedError } from "better-result";
36
+ import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
33
37
  //#region src/docx/parser.ts
34
38
  /**
35
39
  * Main Parser Orchestrator - Unified parseDocx function
@@ -70,7 +74,7 @@ const sha256Hex = async (buffer) => {
70
74
  async function parseDocx(input, options = {}) {
71
75
  const buffer = input instanceof ArrayBuffer ? input : await toArrayBuffer(input);
72
76
  const { onProgress = () => {}, preloadFonts = true, parseHeadersFooters = true, parseNotes = true, detectVariables = true, password, unzipLimits, mediaResolver } = options;
73
- const warnings = [];
77
+ const { context: parseContext, warnings: collectedWarnings } = createParseWarningCollector();
74
78
  try {
75
79
  const timeStage = (_name, fn) => fn();
76
80
  const timeStageAsync = async (_name, fn) => await fn();
@@ -81,8 +85,11 @@ async function parseDocx(input, options = {}) {
81
85
  password,
82
86
  extractAllXml: false
83
87
  }));
84
- if (raw.wasEncrypted) warnings.push("Document was opened from password-protected storage; saving writes an unencrypted .docx file.");
85
- warnings.push(...raw.warnings);
88
+ if (raw.wasEncrypted) parseContext.warn({ code: PARSE_WARNING_CODES.packageDecrypted });
89
+ for (const message of raw.warnings) parseContext.warn({
90
+ code: PARSE_WARNING_CODES.packageArchive,
91
+ detail: message
92
+ });
86
93
  onProgress("Extracted DOCX", 10);
87
94
  onProgress("Parsing relationships...", 10);
88
95
  const rels = timeStage("relationships", () => raw.documentRels ? parseRelationships(raw.documentRels) : /* @__PURE__ */ new Map());
@@ -114,8 +121,8 @@ async function parseDocx(input, options = {}) {
114
121
  onProgress("Parsing document body...", 40);
115
122
  let documentBody = { content: [] };
116
123
  timeStage("documentBody", () => {
117
- if (raw.documentXml) documentBody = parseDocumentBody(raw.documentXml, styles, theme, numbering, rels, media);
118
- else warnings.push("No document.xml found in DOCX");
124
+ if (raw.documentXml) documentBody = parseDocumentBody(raw.documentXml, styles, theme, numbering, rels, media, parseContext.scoped({ part: "word/document.xml" }));
125
+ else parseContext.warn({ code: PARSE_WARNING_CODES.documentPartMissing });
119
126
  });
120
127
  onProgress("Parsed document body", 55);
121
128
  let headers;
@@ -131,15 +138,19 @@ async function parseDocx(input, options = {}) {
131
138
  let endnotes;
132
139
  if (parseNotes) {
133
140
  onProgress("Parsing footnotes/endnotes...", 65);
134
- const notes = timeStage("footnotesEndnotes", () => parseNotesContent(raw, styles, theme, numbering, rels, media));
141
+ const notes = timeStage("footnotesEndnotes", () => parseNotesContent(raw, styles, theme, numbering, rels, media, parseContext));
135
142
  footnotes = notes.footnotes;
136
143
  endnotes = notes.endnotes;
137
144
  onProgress("Parsed footnotes/endnotes", 75);
138
145
  } else onProgress("Skipping footnotes/endnotes", 75);
139
146
  onProgress("Parsing comments...", 75);
140
- const comments = timeStage("comments", () => parseComments(raw.commentsXml, styles, theme, rels, media, raw.commentsExtensibleXml, raw.commentsExtendedXml));
147
+ const commentsContext = parseContext.scoped({ part: "word/comments.xml" });
148
+ const comments = timeStage("comments", () => parseComments(raw.commentsXml, styles, theme, rels, media, raw.commentsExtensibleXml, raw.commentsExtendedXml, commentsContext));
141
149
  const commentIdNormalization = normalizeCommentIds(comments);
142
- if (commentIdNormalization.droppedDuplicateComments > 0) warnings.push(`Dropped ${commentIdNormalization.droppedDuplicateComments} comment(s) repeating a w:id another comment already defines.`);
150
+ if (commentIdNormalization.droppedDuplicateComments > 0) commentsContext.warn({
151
+ code: DUPLICATE_COMMENT_ID_WARNING,
152
+ count: commentIdNormalization.droppedDuplicateComments
153
+ });
143
154
  if (comments.length > 0) documentBody.comments = comments;
144
155
  normalizeDrawingIds({
145
156
  documentBody,
@@ -165,15 +176,27 @@ async function parseDocx(input, options = {}) {
165
176
  ...footnotes !== void 0 ? { footnotes } : {},
166
177
  ...endnotes !== void 0 ? { endnotes } : {}
167
178
  });
168
- if (commentReferenceNormalization.removedDanglingReferences > 0) warnings.push(`Removed ${commentReferenceNormalization.removedDanglingReferences} dangling comment reference marker(s) whose comments.xml entries are missing.`);
169
- if (commentReferenceNormalization.reanchoredUnbalancedRanges > 0) warnings.push(`Re-anchored ${commentReferenceNormalization.reanchoredUnbalancedRanges} unbalanced comment range marker(s) as point comments.`);
179
+ if (commentReferenceNormalization.removedDanglingReferences > 0) parseContext.warn({
180
+ code: DANGLING_COMMENT_REFERENCE_WARNING,
181
+ count: commentReferenceNormalization.removedDanglingReferences
182
+ });
183
+ if (commentReferenceNormalization.reanchoredUnbalancedRanges > 0) parseContext.warn({
184
+ code: UNBALANCED_COMMENT_RANGE_WARNING,
185
+ count: commentReferenceNormalization.reanchoredUnbalancedRanges
186
+ });
170
187
  const headerFooterReferenceNormalization = normalizeHeaderFooterReferences({
171
188
  documentBody,
172
189
  ...headers !== void 0 ? { headers } : {},
173
190
  ...footers !== void 0 ? { footers } : {}
174
191
  });
175
- if (headerFooterReferenceNormalization.removedDanglingHeaderReferences > 0) warnings.push(`Removed ${headerFooterReferenceNormalization.removedDanglingHeaderReferences} dangling header reference(s) whose header parts are missing.`);
176
- if (headerFooterReferenceNormalization.removedDanglingFooterReferences > 0) warnings.push(`Removed ${headerFooterReferenceNormalization.removedDanglingFooterReferences} dangling footer reference(s) whose footer parts are missing.`);
192
+ if (headerFooterReferenceNormalization.removedDanglingHeaderReferences > 0) parseContext.warn({
193
+ code: DANGLING_HEADER_REFERENCE_WARNING,
194
+ count: headerFooterReferenceNormalization.removedDanglingHeaderReferences
195
+ });
196
+ if (headerFooterReferenceNormalization.removedDanglingFooterReferences > 0) parseContext.warn({
197
+ code: DANGLING_FOOTER_REFERENCE_WARNING,
198
+ count: headerFooterReferenceNormalization.removedDanglingFooterReferences
199
+ });
177
200
  const numberingReferenceNormalization = normalizeNumberingReferences({
178
201
  documentBody,
179
202
  numbering,
@@ -182,12 +205,33 @@ async function parseDocx(input, options = {}) {
182
205
  ...footnotes !== void 0 ? { footnotes } : {},
183
206
  ...endnotes !== void 0 ? { endnotes } : {}
184
207
  });
185
- if (numberingReferenceNormalization.unnumberedDanglingReferences > 0) warnings.push(`Unnumbered ${numberingReferenceNormalization.unnumberedDanglingReferences} paragraph(s) whose numbering definitions are missing.`);
208
+ if (numberingReferenceNormalization.unnumberedDanglingReferences > 0) parseContext.warn({
209
+ code: UNNUMBERED_PARAGRAPH_WARNING,
210
+ count: numberingReferenceNormalization.unnumberedDanglingReferences
211
+ });
186
212
  const styleNumberingNormalization = normalizeStyleNumberingReferences({
187
213
  styles: styleDefinitions?.styles ?? [],
188
214
  numbering
189
215
  });
190
- for (const styleId of styleNumberingNormalization.unnumberedStyleIds) warnings.push(`Unnumbered style "${styleId}" whose numbering definition is missing.`);
216
+ for (const styleId of styleNumberingNormalization.unnumberedStyleIds) parseContext.warn({
217
+ code: UNNUMBERED_STYLE_WARNING,
218
+ value: styleId,
219
+ at: `style "${styleId}"`
220
+ });
221
+ const danglingReferences = countDanglingRelationshipReferences({
222
+ content: documentBody.content,
223
+ relationships: rels
224
+ });
225
+ if (danglingReferences.drawings > 0) parseContext.warn({
226
+ code: PARSE_WARNING_CODES.danglingRelationshipId,
227
+ element: "w:drawing",
228
+ count: danglingReferences.drawings
229
+ });
230
+ if (danglingReferences.hyperlinks > 0) parseContext.warn({
231
+ code: PARSE_WARNING_CODES.danglingRelationshipId,
232
+ element: "w:hyperlink",
233
+ count: danglingReferences.hyperlinks
234
+ });
191
235
  const trackedMoveRangeNormalization = normalizeTrackedMoveRanges({
192
236
  documentBody,
193
237
  ...headers !== void 0 ? { headers } : {},
@@ -195,7 +239,10 @@ async function parseDocx(input, options = {}) {
195
239
  ...footnotes !== void 0 ? { footnotes } : {},
196
240
  ...endnotes !== void 0 ? { endnotes } : {}
197
241
  });
198
- if (trackedMoveRangeNormalization.removedUnbalancedMoveRangeMarkers > 0) warnings.push(`Removed ${trackedMoveRangeNormalization.removedUnbalancedMoveRangeMarkers} unbalanced tracked move range marker(s).`);
242
+ if (trackedMoveRangeNormalization.removedUnbalancedMoveRangeMarkers > 0) parseContext.warn({
243
+ code: UNBALANCED_MOVE_RANGE_WARNING,
244
+ count: trackedMoveRangeNormalization.removedUnbalancedMoveRangeMarkers
245
+ });
199
246
  let templateVariables;
200
247
  if (detectVariables) {
201
248
  onProgress("Detecting template variables...", 75);
@@ -238,8 +285,15 @@ async function parseDocx(input, options = {}) {
238
285
  const validation = validateFolioDocumentModel(document);
239
286
  const parsedCompleteModel = parseHeadersFooters && parseNotes;
240
287
  if (!validation.valid && parsedCompleteModel) throw new DocxModelValidationError("Parsed DOCX produced an invalid document model", validation.issues);
241
- warnings.push(...formatDocumentModelIssues(validation.issues));
242
- if (warnings.length > 0) document.warnings = warnings;
288
+ for (const issue of formatDocumentModelIssues(validation.issues)) parseContext.warn({
289
+ code: PARSE_WARNING_CODES.documentModelIssue,
290
+ detail: issue
291
+ });
292
+ const parseWarnings = collectedWarnings();
293
+ if (parseWarnings.length > 0) {
294
+ document.parseWarnings = parseWarnings;
295
+ document.warnings = formatParseWarnings(parseWarnings);
296
+ }
243
297
  onProgress("Complete", 100);
244
298
  return document;
245
299
  } catch (error) {
@@ -418,7 +472,7 @@ function parseHeadersAndFooters(raw, styles, theme, numbering, rels, media) {
418
472
  if (headerXml) {
419
473
  const headerRelsPath = getRelationshipsPathForPart(partPath);
420
474
  const headerRelsXml = getMapCaseInsensitive(raw.allXml, headerRelsPath);
421
- const headerRels = headerRelsXml ? parseRelationships(headerRelsXml) : rels;
475
+ const headerRels = headerRelsXml ? parseRelationships(headerRelsXml) : /* @__PURE__ */ new Map();
422
476
  const header = parseHeader(headerXml, "default", styles, theme, numbering, headerRels, media);
423
477
  const watermark = header.watermark;
424
478
  if (watermark?.kind === "picture") {
@@ -437,7 +491,7 @@ function parseHeadersAndFooters(raw, styles, theme, numbering, rels, media) {
437
491
  if (footerXml) {
438
492
  const footerRelsPath = getRelationshipsPathForPart(partPath);
439
493
  const footerRelsXml = getMapCaseInsensitive(raw.allXml, footerRelsPath);
440
- const footer = parseFooter(footerXml, "default", styles, theme, numbering, footerRelsXml ? parseRelationships(footerRelsXml) : rels, media);
494
+ const footer = parseFooter(footerXml, "default", styles, theme, numbering, footerRelsXml ? parseRelationships(footerRelsXml) : /* @__PURE__ */ new Map(), media);
441
495
  footers.set(rId, footer);
442
496
  }
443
497
  }
@@ -449,13 +503,13 @@ function parseHeadersAndFooters(raw, styles, theme, numbering, rels, media) {
449
503
  /**
450
504
  * Parse footnotes and endnotes from raw content
451
505
  */
452
- function parseNotesContent(raw, styles, theme, numbering, rels, media) {
506
+ function parseNotesContent(raw, styles, theme, numbering, rels, media, context) {
453
507
  const relsForNotePart = (partPath) => {
454
508
  const xml = getMapCaseInsensitive(raw.allXml, getRelationshipsPathForPart(partPath));
455
509
  return xml ? parseRelationships(xml) : rels;
456
510
  };
457
- const footnoteMap = parseFootnotes(raw.footnotesXml, styles, theme, numbering, relsForNotePart("word/footnotes.xml"), media);
458
- const endnoteMap = parseEndnotes(raw.endnotesXml, styles, theme, numbering, relsForNotePart("word/endnotes.xml"), media);
511
+ const footnoteMap = parseFootnotes(raw.footnotesXml, styles, theme, numbering, relsForNotePart("word/footnotes.xml"), media, context?.scoped({ part: "word/footnotes.xml" }));
512
+ const endnoteMap = parseEndnotes(raw.endnotesXml, styles, theme, numbering, relsForNotePart("word/endnotes.xml"), media, context?.scoped({ part: "word/endnotes.xml" }));
459
513
  return {
460
514
  footnotes: footnoteMap.getNormalFootnotes(),
461
515
  endnotes: endnoteMap.getNormalEndnotes()
@@ -107,21 +107,38 @@ declare function getHeaders(map: document_d_exports.RelationshipMap): document_d
107
107
  */
108
108
  declare function getFooters(map: document_d_exports.RelationshipMap): document_d_exports.Relationship[];
109
109
  /**
110
- * Resolve a relationship ID to a target path
110
+ * What a relationship id names.
111
+ *
112
+ * The three cases are kept apart because collapsing any of them into a string
113
+ * turns absence into a lookup key: an id the author never wrote, an id whose
114
+ * target the package no longer holds, and an id that resolves are different
115
+ * facts, and only the last one may be read as a part.
116
+ */
117
+ type RelationshipResolution = {
118
+ status: "resolved";
119
+ relationship: document_d_exports.Relationship;
120
+ } | {
121
+ status: "absent";
122
+ } | {
123
+ status: "dangling";
124
+ id: string;
125
+ };
126
+ /**
127
+ * Resolve a relationship id against a relationship map.
111
128
  *
112
- * @param map - RelationshipMap to search
113
- * @param rId - Relationship ID (e.g., "rId1")
114
- * @returns Target path or undefined if not found
129
+ * The only sanctioned way to turn an `r:id`, `r:embed` or `r:link` into a part.
130
+ * An empty attribute is absence, not an id: `ST_RelationshipId` is an NCName,
131
+ * so `""` can never name a relationship, and no map can hold it as a key.
115
132
  */
116
- declare function resolveTarget(map: document_d_exports.RelationshipMap, rId: string): string | undefined;
133
+ declare function resolveRelationshipId(map: document_d_exports.RelationshipMap | null | undefined, rId: string | undefined): RelationshipResolution;
117
134
  /**
118
- * Resolve a relationship ID to a full relationship
135
+ * Resolve a relationship id and require it to name a relationship of one type.
119
136
  *
120
- * @param map - RelationshipMap to search
121
- * @param rId - Relationship ID (e.g., "rId1")
122
- * @returns Relationship or undefined if not found
137
+ * A reference of the wrong type is reported as dangling: the part it names
138
+ * exists, but not as the thing the reference asked for, and reading it anyway
139
+ * is how a missing image comes back as `styles.xml`.
123
140
  */
124
- declare function resolveRelationship(map: document_d_exports.RelationshipMap, rId: string): document_d_exports.Relationship | undefined;
141
+ declare function resolveRelationshipIdOfType(map: document_d_exports.RelationshipMap | null | undefined, rId: string | undefined, type: document_d_exports.RelationshipType): RelationshipResolution;
125
142
  /**
126
143
  * Resolve a relative target path to an absolute path within the DOCX
127
144
  *
@@ -158,4 +175,4 @@ declare function parsePackageRelationships(relsXml: string): document_d_exports.
158
175
  */
159
176
  declare function formatRelationships(map: document_d_exports.RelationshipMap): string;
160
177
  //#endregion
161
- export { RELATIONSHIP_TYPES, filterByType, formatRelationships, getFooters, getHeaders, getHyperlinks, getImages, getRelationshipTypeName, isExternalHyperlink, isFooterRelationship, isHeaderRelationship, isImageRelationship, parseDocumentRelationships, parsePackageRelationships, parseRelationships, resolveRelationship, resolveRelativePath, resolveTarget };
178
+ export { RELATIONSHIP_TYPES, RelationshipResolution, filterByType, formatRelationships, getFooters, getHeaders, getHyperlinks, getImages, getRelationshipTypeName, isExternalHyperlink, isFooterRelationship, isHeaderRelationship, isImageRelationship, parseDocumentRelationships, parsePackageRelationships, parseRelationships, resolveRelationshipId, resolveRelationshipIdOfType, resolveRelativePath };
@@ -155,24 +155,37 @@ function getFooters(map) {
155
155
  return filterByType(map, RELATIONSHIP_TYPES.footer);
156
156
  }
157
157
  /**
158
- * Resolve a relationship ID to a target path
158
+ * Resolve a relationship id against a relationship map.
159
159
  *
160
- * @param map - RelationshipMap to search
161
- * @param rId - Relationship ID (e.g., "rId1")
162
- * @returns Target path or undefined if not found
160
+ * The only sanctioned way to turn an `r:id`, `r:embed` or `r:link` into a part.
161
+ * An empty attribute is absence, not an id: `ST_RelationshipId` is an NCName,
162
+ * so `""` can never name a relationship, and no map can hold it as a key.
163
163
  */
164
- function resolveTarget(map, rId) {
165
- return map.get(rId)?.target;
164
+ function resolveRelationshipId(map, rId) {
165
+ if (rId === void 0 || rId.length === 0) return { status: "absent" };
166
+ const relationship = map?.get(rId);
167
+ return relationship === void 0 ? {
168
+ status: "dangling",
169
+ id: rId
170
+ } : {
171
+ status: "resolved",
172
+ relationship
173
+ };
166
174
  }
167
175
  /**
168
- * Resolve a relationship ID to a full relationship
176
+ * Resolve a relationship id and require it to name a relationship of one type.
169
177
  *
170
- * @param map - RelationshipMap to search
171
- * @param rId - Relationship ID (e.g., "rId1")
172
- * @returns Relationship or undefined if not found
178
+ * A reference of the wrong type is reported as dangling: the part it names
179
+ * exists, but not as the thing the reference asked for, and reading it anyway
180
+ * is how a missing image comes back as `styles.xml`.
173
181
  */
174
- function resolveRelationship(map, rId) {
175
- return map.get(rId);
182
+ function resolveRelationshipIdOfType(map, rId, type) {
183
+ const resolved = resolveRelationshipId(map, rId);
184
+ if (resolved.status !== "resolved" || resolved.relationship.type === type) return resolved;
185
+ return {
186
+ status: "dangling",
187
+ id: resolved.relationship.id
188
+ };
176
189
  }
177
190
  /**
178
191
  * Resolve a relative target path to an absolute path within the DOCX
@@ -232,4 +245,4 @@ function formatRelationships(map) {
232
245
  return lines.join("\n");
233
246
  }
234
247
  //#endregion
235
- export { RELATIONSHIP_TYPES, filterByType, formatRelationships, getFooters, getHeaders, getHyperlinks, getImages, getRelationshipTypeName, isExternalHyperlink, isFooterRelationship, isHeaderRelationship, isImageRelationship, parseDocumentRelationships, parsePackageRelationships, parseRelationships, resolveRelationship, resolveRelativePath, resolveTarget };
248
+ export { RELATIONSHIP_TYPES, filterByType, formatRelationships, getFooters, getHeaders, getHyperlinks, getImages, getRelationshipTypeName, isExternalHyperlink, isFooterRelationship, isHeaderRelationship, isImageRelationship, parseDocumentRelationships, parsePackageRelationships, parseRelationships, resolveRelationshipId, resolveRelationshipIdOfType, resolveRelativePath };
@@ -31,16 +31,79 @@ const REVISION_ELEMENT_NAMES = /* @__PURE__ */ new Set([
31
31
  "trPrChange"
32
32
  ]);
33
33
  const REVISION_ELEMENT_CANDIDATE = new RegExp(`<(?:[^\\s<>/:]+:)?(?:${[...REVISION_ELEMENT_NAMES].join("|")})(?:[\\s/>])`, "u");
34
- const revisionAttribute = (element) => {
35
- if (!element.name || !WORDPROCESSINGML_NAMESPACE_URIS.has(getNamespaceUri(element) ?? "") || !REVISION_ELEMENT_NAMES.has(getLocalName(element.name))) return null;
34
+ /**
35
+ * The rest of the annotation id space.
36
+ *
37
+ * A comment, a bookmark, a protected range and a tracked change all draw their
38
+ * `w:id` from one space: Word allocates from a single counter, which is why a
39
+ * package carrying several kinds almost never repeats a value across them. So
40
+ * an id this pass mints must avoid these as well, or a renumbered `w:ins`
41
+ * lands on a live comment.
42
+ *
43
+ * They are only ever reserved, never claimed. A comment id legitimately
44
+ * appears four times (`w:comment`, both range markers and the reference) and a
45
+ * bookmark id twice, so feeding them to the uniqueness machinery would reject
46
+ * a package Word wrote. Their pairing is also why they are not revision
47
+ * elements: renumbering one end of a range would unpair it.
48
+ */
49
+ const ANNOTATION_ELEMENT_NAMES = /* @__PURE__ */ new Set([
50
+ "bookmarkEnd",
51
+ "bookmarkStart",
52
+ "comment",
53
+ "commentRangeEnd",
54
+ "commentRangeStart",
55
+ "commentReference",
56
+ "customXmlDelRangeEnd",
57
+ "customXmlDelRangeStart",
58
+ "customXmlInsRangeEnd",
59
+ "customXmlInsRangeStart",
60
+ "customXmlMoveFromRangeEnd",
61
+ "customXmlMoveFromRangeStart",
62
+ "customXmlMoveToRangeEnd",
63
+ "customXmlMoveToRangeStart",
64
+ "moveFromRangeEnd",
65
+ "moveFromRangeStart",
66
+ "moveToRangeEnd",
67
+ "moveToRangeStart",
68
+ "permEnd",
69
+ "permStart"
70
+ ]);
71
+ const ANNOTATION_ELEMENT_CANDIDATE = new RegExp(`<(?:[^\\s<>/:]+:)?(?:${[...ANNOTATION_ELEMENT_NAMES].join("|")})(?:[\\s/>])`, "u");
72
+ const ID_KINDS = {
73
+ revision: "revision",
74
+ annotation: "annotation"
75
+ };
76
+ /** One lookup for both halves of the space, so an element is classified once. */
77
+ const ID_KIND_BY_ELEMENT_NAME = new Map([...[...REVISION_ELEMENT_NAMES].map((name) => [name, ID_KINDS.revision]), ...[...ANNOTATION_ELEMENT_NAMES].map((name) => [name, ID_KINDS.annotation])]);
78
+ /**
79
+ * An element's `w:id` and which half of the annotation space it belongs to.
80
+ *
81
+ * One classifier rather than two, because it runs on every element of every
82
+ * scanned part: resolving the namespace and the local name twice to ask two
83
+ * questions measured 18% on a 17.6 MiB package.
84
+ *
85
+ * `w:permStart` types its id as a string, so a protected range named
86
+ * `everyone` yields nothing. That is correct: a value the allocator can never
87
+ * mint is not one it has to avoid.
88
+ */
89
+ const identifiedElement = (element) => {
90
+ if (!element.name || !WORDPROCESSINGML_NAMESPACE_URIS.has(getNamespaceUri(element) ?? "")) return null;
91
+ const localName = getLocalName(element.name);
92
+ const kind = ID_KIND_BY_ELEMENT_NAME.get(localName);
93
+ if (kind === void 0) return null;
36
94
  const attribute = findAttributeByNamespaceUri(element, WORDPROCESSINGML_NAMESPACE_URIS, "id");
37
95
  if (!attribute) return null;
38
96
  const id = Number(attribute.value);
39
97
  return Number.isSafeInteger(id) && id >= 0 ? {
98
+ kind,
40
99
  name: attribute.name,
41
100
  id
42
101
  } : null;
43
102
  };
103
+ const revisionAttribute = (element) => {
104
+ const identified = identifiedElement(element);
105
+ return identified?.kind === ID_KINDS.revision ? identified : null;
106
+ };
44
107
  /**
45
108
  * Keep physical tracked-change element ids unique across a package.
46
109
  *
@@ -57,11 +120,10 @@ const normalizeRevisionIdsInXmlParts = (parts) => {
57
120
  assertXmlResourceLimits(xml);
58
121
  const ids = [];
59
122
  if (rewriteStreamingXmlDecimalAttributes(xml, (element) => {
60
- const attribute = revisionAttribute(element);
61
- if (attribute) {
62
- ids.push(attribute.id);
63
- reserved.add(attribute.id);
64
- }
123
+ const identified = identifiedElement(element);
124
+ if (identified === null) return null;
125
+ if (identified.kind === ID_KINDS.revision) ids.push(identified.id);
126
+ reserved.add(identified.id);
65
127
  return null;
66
128
  }).status === "unsupported") throw new XmlResourceLimitError({
67
129
  message: `Revision-id normalization could not safely scan ${path}`,
@@ -73,6 +135,18 @@ const normalizeRevisionIdsInXmlParts = (parts) => {
73
135
  const firstSeen = /* @__PURE__ */ new Set();
74
136
  for (const [path, ids] of occurrencesByPath) for (const id of ids) if (firstSeen.has(id)) repeatedPaths.add(path);
75
137
  else firstSeen.add(id);
138
+ if (repeatedPaths.size > 0) for (const [path, xml] of parts) {
139
+ if (occurrencesByPath.has(path) || !ANNOTATION_ELEMENT_CANDIDATE.test(xml)) continue;
140
+ assertXmlResourceLimits(xml);
141
+ if (rewriteStreamingXmlDecimalAttributes(xml, (element) => {
142
+ const identified = identifiedElement(element);
143
+ if (identified !== null) reserved.add(identified.id);
144
+ return null;
145
+ }).status === "unsupported") throw new XmlResourceLimitError({
146
+ message: `Revision-id normalization could not safely scan ${path}`,
147
+ limit: "syntax"
148
+ });
149
+ }
76
150
  let nextId = 0;
77
151
  const allocate = () => {
78
152
  while (reserved.has(nextId)) nextId += 1;