@stll/folio-core 0.43.0 → 0.44.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ai-edits/headless.js +6 -5
- package/dist/ai-edits/index.d.ts +2 -2
- package/dist/ai-edits/index.js +2 -2
- package/dist/ai-edits/snapshot.js +13 -9
- package/dist/compare/content-alignment.js +16 -1
- package/dist/compare/inline-atoms.js +1 -1
- package/dist/compare/style-resources.js +6 -0
- package/dist/docx/appVersionNormalization.d.ts +0 -18
- package/dist/docx/blockContentParser.js +8 -0
- package/dist/docx/blockRangeMarkers.d.ts +36 -0
- package/dist/docx/blockRangeMarkers.js +59 -0
- package/dist/docx/bookmarkParser.d.ts +2 -20
- package/dist/docx/bookmarkParser.js +6 -30
- package/dist/docx/borderParser.d.ts +13 -0
- package/dist/docx/borderParser.js +71 -0
- package/dist/docx/builtInStyles.d.ts +165 -0
- package/dist/docx/builtInStyles.js +239 -0
- package/dist/docx/commentIdNormalization.d.ts +3 -1
- package/dist/docx/commentIdNormalization.js +18 -1
- package/dist/docx/commentParser.d.ts +2 -1
- package/dist/docx/commentParser.js +34 -7
- package/dist/docx/commentReferenceNormalization.d.ts +4 -1
- package/dist/docx/commentReferenceNormalization.js +23 -14
- package/dist/docx/danglingRelationshipReferences.d.ts +15 -0
- package/dist/docx/danglingRelationshipReferences.js +30 -0
- package/dist/docx/defaultParagraphStyle.d.ts +18 -1
- package/dist/docx/defaultParagraphStyle.js +23 -1
- package/dist/docx/documentParser.d.ts +2 -1
- package/dist/docx/documentParser.js +2 -2
- package/dist/docx/drawingUtils.d.ts +8 -1
- package/dist/docx/drawingUtils.js +12 -3
- package/dist/docx/fieldParser.js +3 -5
- package/dist/docx/footnoteParser.d.ts +3 -2
- package/dist/docx/footnoteParser.js +19 -4
- package/dist/docx/groupDrawingParser.js +1 -1
- package/dist/docx/headerFooterRefParser.d.ts +4 -3
- package/dist/docx/headerFooterRefParser.js +42 -12
- package/dist/docx/headerFooterReferenceNormalization.d.ts +4 -1
- package/dist/docx/headerFooterReferenceNormalization.js +5 -1
- package/dist/docx/hyperlinkParser.js +11 -15
- package/dist/docx/imageParser.d.ts +1 -1
- package/dist/docx/imageParser.js +22 -18
- package/dist/docx/imageRawXml.js +5 -5
- package/dist/docx/markupRangeMarker.d.ts +15 -0
- package/dist/docx/markupRangeMarker.js +44 -0
- package/dist/docx/noteReferenceStyles.d.ts +29 -0
- package/dist/docx/noteReferenceStyles.js +70 -0
- package/dist/docx/numberingReferenceNormalization.d.ts +4 -1
- package/dist/docx/numberingReferenceNormalization.js +20 -1
- package/dist/docx/paraIdRangeNormalization.d.ts +0 -19
- package/dist/docx/paragraphParser.js +66 -99
- package/dist/docx/paragraphPropertySource.js +1 -0
- package/dist/docx/paragraphTraversal.d.ts +37 -1
- package/dist/docx/paragraphTraversal.js +84 -1
- package/dist/docx/parseContext.d.ts +37 -0
- package/dist/docx/parseContext.js +67 -0
- package/dist/docx/parseWarningMessage.d.ts +6 -0
- package/dist/docx/parseWarningMessage.js +44 -0
- package/dist/docx/parser.js +81 -27
- package/dist/docx/relsParser.d.ts +28 -11
- package/dist/docx/relsParser.js +26 -13
- package/dist/docx/revisionIdNormalization.js +81 -7
- package/dist/docx/rezip.js +63 -22
- package/dist/docx/runConsolidator.js +1 -2
- package/dist/docx/runParser.d.ts +8 -1
- package/dist/docx/runParser.js +30 -48
- package/dist/docx/sectionParser.d.ts +2 -1
- package/dist/docx/sectionParser.js +21 -65
- package/dist/docx/serializer/borderSerializer.d.ts +1 -2
- package/dist/docx/serializer/commentSerializer.js +22 -9
- package/dist/docx/serializer/documentSerializer.d.ts +1 -5
- package/dist/docx/serializer/documentSerializer.js +6 -16
- package/dist/docx/serializer/headerFooterSerializer.js +5 -0
- package/dist/docx/serializer/markupRangeAttributes.d.ts +8 -0
- package/dist/docx/serializer/markupRangeAttributes.js +24 -0
- package/dist/docx/serializer/noteSerializer.js +5 -0
- package/dist/docx/serializer/paragraphSerializer.d.ts +1 -5
- package/dist/docx/serializer/paragraphSerializer.js +29 -35
- package/dist/docx/serializer/runSerializer.js +13 -7
- package/dist/docx/serializer/tableSerializer.js +28 -13
- package/dist/docx/serializer/textFormattingSerializer.d.ts +2 -3
- package/dist/docx/server/build.js +8 -1
- package/dist/docx/server/createBilingualDocument.js +10 -18
- package/dist/docx/server/extractDocxText.js +3 -4
- package/dist/docx/shadingParser.d.ts +6 -0
- package/dist/docx/shadingParser.js +32 -0
- package/dist/docx/shapeParser.js +3 -3
- package/dist/docx/styleParser.js +13 -87
- package/dist/docx/styleReferenceResolution.d.ts +36 -0
- package/dist/docx/styleReferenceResolution.js +51 -0
- package/dist/docx/tableLook.d.ts +57 -0
- package/dist/docx/tableLook.js +63 -0
- package/dist/docx/tableParser.d.ts +7 -9
- package/dist/docx/tableParser.js +64 -110
- package/dist/docx/textBoxParser.js +4 -4
- package/dist/docx/trackedMoveRangeNormalization.d.ts +3 -1
- package/dist/docx/trackedMoveRangeNormalization.js +11 -21
- package/dist/docx/transitionalSpelling.d.ts +13 -2
- package/dist/docx/transitionalSpelling.js +23 -1
- package/dist/docx/verbatimCapture.js +4 -11
- package/dist/docx/vmlImageParser.js +2 -2
- package/dist/docx/watermarkParser.js +2 -2
- package/dist/docx/xmlParser.d.ts +22 -32
- package/dist/docx/xmlParser.js +36 -21
- package/dist/internal/pageBreakRunSourceDescendantIndex.js +2 -1
- package/dist/internal/paragraphFormattingSerialization.d.ts +2 -3
- package/dist/internal/paragraphFormattingSerialization.js +26 -6
- package/dist/layout-bridge/convert/footnoteLayout.js +2 -7
- package/dist/layout-engine/index.d.ts +2 -2
- package/dist/layout-engine/index.js +2 -2
- package/dist/layout-engine/measure/measureBlocks.js +1 -6
- package/dist/layout-engine/types.d.ts +8 -2
- package/dist/layout-engine/types.js +35 -2
- package/dist/markdown/index.js +1 -1
- package/dist/markdown/internals.d.ts +6 -1
- package/dist/markdown/internals.js +14 -1
- package/dist/markdown/renderBlock.js +35 -21
- package/dist/markdown/renderParagraph.js +14 -5
- package/dist/markdown/renderRuns.js +4 -3
- package/dist/markdown/renderTable.js +4 -3
- package/dist/markdown/trailers.js +41 -7
- package/dist/markdown/types.d.ts +3 -7
- package/dist/prosemirror/attrs/index.js +2 -5
- package/dist/prosemirror/bookmarkBoundaryAttrs.d.ts +11 -1
- package/dist/prosemirror/bookmarkBoundaryAttrs.js +18 -3
- package/dist/prosemirror/commands/index.d.ts +3 -3
- package/dist/prosemirror/commands/index.js +2 -2
- package/dist/prosemirror/commands/paragraph.d.ts +3 -3
- package/dist/prosemirror/commands/paragraph.js +2 -2
- package/dist/prosemirror/commentIdAllocator.js +2 -7
- package/dist/prosemirror/conversion/fromProseDoc.js +129 -40
- package/dist/prosemirror/conversion/toProseDoc.d.ts +1 -14
- package/dist/prosemirror/conversion/toProseDoc.js +375 -316
- package/dist/prosemirror/extensions/core/ParagraphExtension.d.ts +14 -1
- package/dist/prosemirror/extensions/core/ParagraphExtension.js +11 -6
- package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -3
- package/dist/prosemirror/extensions/features/PasteCleanupExtension.d.ts +4 -1
- package/dist/prosemirror/extensions/features/PasteCleanupExtension.js +6 -2
- package/dist/prosemirror/extensions/features/pastedHeadingStyles.d.ts +7 -0
- package/dist/prosemirror/extensions/features/pastedHeadingStyles.js +74 -0
- package/dist/prosemirror/extensions/marks/markUtils.d.ts +11 -3
- package/dist/prosemirror/extensions/marks/markUtils.js +98 -19
- package/dist/prosemirror/extensions/nodes/BookmarkBoundaryExtension.js +7 -3
- package/dist/prosemirror/extensions/nodes/ImageExtension.js +2 -1
- package/dist/prosemirror/extensions/nodes/ShapeExtension.js +1 -0
- package/dist/prosemirror/extensions/nodes/TableExtension.js +15 -1
- package/dist/prosemirror/extensions/types.d.ts +2 -2
- package/dist/prosemirror/index.d.ts +3 -3
- package/dist/prosemirror/index.js +3 -3
- package/dist/prosemirror/insertOperations.d.ts +9 -2
- package/dist/prosemirror/insertOperations.js +9 -4
- package/dist/prosemirror/paragraphFormattingProvenance.d.ts +159 -0
- package/dist/prosemirror/paragraphFormattingProvenance.js +106 -0
- package/dist/prosemirror/plugins/documentStyles.d.ts +9 -1
- package/dist/prosemirror/plugins/documentStyles.js +11 -1
- package/dist/prosemirror/plugins/index.d.ts +2 -2
- package/dist/prosemirror/plugins/index.js +2 -2
- package/dist/prosemirror/plugins/revisionIds.d.ts +11 -2
- package/dist/prosemirror/plugins/revisionIds.js +21 -6
- package/dist/prosemirror/runFormattingReconciliation.js +3 -2
- package/dist/prosemirror/runStyleFormatting.d.ts +1 -1
- package/dist/prosemirror/schema/nodes.d.ts +31 -0
- package/dist/prosemirror/styles/resolvedStyleAttrs.js +2 -0
- package/dist/prosemirror/styles/styleResolver.d.ts +9 -0
- package/dist/prosemirror/styles/styleResolver.js +12 -0
- package/dist/style-engine/styleEngine.d.ts +3 -0
- package/dist/style-engine/styleEngine.js +3 -0
- package/dist/style-sets/extract.js +1 -23
- package/dist/style-sets/stellaStyle.js +46 -39
- package/dist/style-sets/styleSetNormalization.d.ts +19 -0
- package/dist/style-sets/styleSetNormalization.js +99 -0
- package/dist/types/content.d.ts +2 -2
- package/dist/utils/createDocument.js +145 -20
- package/dist/utils/headingCollector.d.ts +8 -5
- package/dist/utils/headingCollector.js +23 -25
- package/dist/utils/tableOfContentsStyle.js +9 -2
- package/package.json +2 -2
- package/dist/docx/textWhitespace.d.ts +0 -4
- package/dist/docx/textWhitespace.js +0 -4
- package/dist/layout-bridge/engine/tableWidthUtils.d.ts +0 -6
- package/dist/layout-bridge/engine/tableWidthUtils.js +0 -25
- package/dist/markdown/headings.d.ts +0 -13
- package/dist/markdown/headings.js +0 -20
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
import { getAttribute, parseNumericAttribute } from "./xmlParser.js";
|
|
2
|
+
//#region src/docx/markupRangeMarker.ts
|
|
3
|
+
/** ST_DisplacedByCustomXml. A value outside it is not a placement folio can honour. */
|
|
4
|
+
const DISPLACED_BY_CUSTOM_XML = /* @__PURE__ */ new Set(["next", "prev"]);
|
|
5
|
+
const parseDisplacedByCustomXml = (node) => {
|
|
6
|
+
const value = getAttribute(node, "w", "displacedByCustomXml") ?? "";
|
|
7
|
+
return DISPLACED_BY_CUSTOM_XML.has(value) ? value : void 0;
|
|
8
|
+
};
|
|
9
|
+
const parseMarkupRangeMarker = (node) => {
|
|
10
|
+
const marker = { id: parseNumericAttribute(node, "w", "id") ?? 0 };
|
|
11
|
+
const displacedByCustomXml = parseDisplacedByCustomXml(node);
|
|
12
|
+
if (displacedByCustomXml !== void 0) marker.displacedByCustomXml = displacedByCustomXml;
|
|
13
|
+
return marker;
|
|
14
|
+
};
|
|
15
|
+
const parseBookmarkRangeMarker = (node) => {
|
|
16
|
+
const marker = {
|
|
17
|
+
...parseMarkupRangeMarker(node),
|
|
18
|
+
name: getAttribute(node, "w", "name") ?? ""
|
|
19
|
+
};
|
|
20
|
+
const colFirst = parseNumericAttribute(node, "w", "colFirst");
|
|
21
|
+
if (colFirst !== void 0) marker.colFirst = colFirst;
|
|
22
|
+
const colLast = parseNumericAttribute(node, "w", "colLast");
|
|
23
|
+
if (colLast !== void 0) marker.colLast = colLast;
|
|
24
|
+
return marker;
|
|
25
|
+
};
|
|
26
|
+
/**
|
|
27
|
+
* `w:author` is required on `CT_MoveBookmark`, so an absent one takes the same
|
|
28
|
+
* `Unknown` fallback a tracked change takes rather than leaving the saved
|
|
29
|
+
* package short of an attribute the schema demands. `w:date` is required too,
|
|
30
|
+
* but inventing a timestamp would state a fact about the document that is not
|
|
31
|
+
* true, so an absent date stays absent.
|
|
32
|
+
*/
|
|
33
|
+
const parseMoveBookmarkMarker = (node) => {
|
|
34
|
+
const author = (getAttribute(node, "w", "author") ?? "").trim();
|
|
35
|
+
const marker = {
|
|
36
|
+
...parseBookmarkRangeMarker(node),
|
|
37
|
+
author: author.length > 0 ? author : "Unknown"
|
|
38
|
+
};
|
|
39
|
+
const date = (getAttribute(node, "w", "date") ?? "").trim();
|
|
40
|
+
if (date.length > 0) marker.date = date;
|
|
41
|
+
return marker;
|
|
42
|
+
};
|
|
43
|
+
//#endregion
|
|
44
|
+
export { parseBookmarkRangeMarker, parseMarkupRangeMarker, parseMoveBookmarkMarker };
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import { document_d_exports } from "../types/document.js";
|
|
2
|
+
//#region src/docx/noteReferenceStyles.d.ts
|
|
3
|
+
/** The id `commentSerializer` and `paragraphSerializer` write. */
|
|
4
|
+
declare const COMMENT_REFERENCE_STYLE_ID = "CommentReference";
|
|
5
|
+
/** The ids `noteSerializer` writes for the two note kinds. */
|
|
6
|
+
declare const FOOTNOTE_REFERENCE_STYLE_ID = "FootnoteReference";
|
|
7
|
+
declare const ENDNOTE_REFERENCE_STYLE_ID = "EndnoteReference";
|
|
8
|
+
/** Which reference marks a package actually contains. */
|
|
9
|
+
type NoteReferenceNeeds = {
|
|
10
|
+
comments: boolean;
|
|
11
|
+
footnotes: boolean;
|
|
12
|
+
endnotes: boolean;
|
|
13
|
+
};
|
|
14
|
+
/** What a package's content requires, read off the model rather than guessed. */
|
|
15
|
+
declare const noteReferenceNeeds: (pkg: {
|
|
16
|
+
comments?: unknown[] | undefined;
|
|
17
|
+
footnotes?: unknown[] | undefined;
|
|
18
|
+
endnotes?: unknown[] | undefined;
|
|
19
|
+
}) => NoteReferenceNeeds;
|
|
20
|
+
/**
|
|
21
|
+
* The reference styles a package needs and does not already define, by id.
|
|
22
|
+
*
|
|
23
|
+
* Returns the definitions to append rather than a mutated style table: the
|
|
24
|
+
* caller owns the document, and a document that already defines the style
|
|
25
|
+
* (under any id folio recognises) keeps its own.
|
|
26
|
+
*/
|
|
27
|
+
declare const missingNoteReferenceStyles: (styles: document_d_exports.StyleDefinitions | undefined, needs: NoteReferenceNeeds) => document_d_exports.Style[];
|
|
28
|
+
//#endregion
|
|
29
|
+
export { COMMENT_REFERENCE_STYLE_ID, ENDNOTE_REFERENCE_STYLE_ID, FOOTNOTE_REFERENCE_STYLE_ID, NoteReferenceNeeds, missingNoteReferenceStyles, noteReferenceNeeds };
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { BUILT_IN_STYLE_NAME } from "./builtInStyles.js";
|
|
2
|
+
//#region src/docx/noteReferenceStyles.ts
|
|
3
|
+
/**
|
|
4
|
+
* The character styles folio's serializers write into every document that
|
|
5
|
+
* carries a comment or a note, and the definitions that keep those references
|
|
6
|
+
* from dangling.
|
|
7
|
+
*
|
|
8
|
+
* `commentSerializer`, `paragraphSerializer` and `noteSerializer` each emit
|
|
9
|
+
* `<w:rStyle w:val="…"/>` for the reference mark, because that is what Word
|
|
10
|
+
* writes and what makes the mark superscript. Nothing guaranteed the document
|
|
11
|
+
* *defined* those styles: a package folio assembled from the generic style set
|
|
12
|
+
* referenced `FootnoteReference` while declaring only seven paragraph styles,
|
|
13
|
+
* so the mark rendered as body text. One owner here, so a serializer and the
|
|
14
|
+
* style table cannot disagree about the id.
|
|
15
|
+
*/
|
|
16
|
+
/** The id `commentSerializer` and `paragraphSerializer` write. */
|
|
17
|
+
const COMMENT_REFERENCE_STYLE_ID = "CommentReference";
|
|
18
|
+
/** The ids `noteSerializer` writes for the two note kinds. */
|
|
19
|
+
const FOOTNOTE_REFERENCE_STYLE_ID = "FootnoteReference";
|
|
20
|
+
const ENDNOTE_REFERENCE_STYLE_ID = "EndnoteReference";
|
|
21
|
+
/** The default character style Word bases these on, when the package has one. */
|
|
22
|
+
const DEFAULT_CHARACTER_STYLE_ID = "DefaultParagraphFont";
|
|
23
|
+
const referenceStyle = (styleId, name, superscript) => ({
|
|
24
|
+
styleId,
|
|
25
|
+
type: "character",
|
|
26
|
+
name,
|
|
27
|
+
uiPriority: 99,
|
|
28
|
+
semiHidden: true,
|
|
29
|
+
unhideWhenUsed: true,
|
|
30
|
+
rPr: superscript ? { vertAlign: "superscript" } : { fontSize: 16 }
|
|
31
|
+
});
|
|
32
|
+
/**
|
|
33
|
+
* The definition for each reference style, keyed by the need that requires it.
|
|
34
|
+
* Word's comment reference is 8pt body text; the note references are
|
|
35
|
+
* superscript.
|
|
36
|
+
*/
|
|
37
|
+
const REFERENCE_STYLES = {
|
|
38
|
+
comments: () => referenceStyle(COMMENT_REFERENCE_STYLE_ID, BUILT_IN_STYLE_NAME.commentReference, false),
|
|
39
|
+
footnotes: () => referenceStyle(FOOTNOTE_REFERENCE_STYLE_ID, BUILT_IN_STYLE_NAME.footnoteReference, true),
|
|
40
|
+
endnotes: () => referenceStyle(ENDNOTE_REFERENCE_STYLE_ID, BUILT_IN_STYLE_NAME.endnoteReference, true)
|
|
41
|
+
};
|
|
42
|
+
/** What a package's content requires, read off the model rather than guessed. */
|
|
43
|
+
const noteReferenceNeeds = (pkg) => ({
|
|
44
|
+
comments: (pkg.comments?.length ?? 0) > 0,
|
|
45
|
+
footnotes: (pkg.footnotes?.length ?? 0) > 0,
|
|
46
|
+
endnotes: (pkg.endnotes?.length ?? 0) > 0
|
|
47
|
+
});
|
|
48
|
+
/**
|
|
49
|
+
* The reference styles a package needs and does not already define, by id.
|
|
50
|
+
*
|
|
51
|
+
* Returns the definitions to append rather than a mutated style table: the
|
|
52
|
+
* caller owns the document, and a document that already defines the style
|
|
53
|
+
* (under any id folio recognises) keeps its own.
|
|
54
|
+
*/
|
|
55
|
+
const missingNoteReferenceStyles = (styles, needs) => {
|
|
56
|
+
const defined = new Set((styles?.styles ?? []).map((style) => style.styleId));
|
|
57
|
+
const basedOn = defined.has(DEFAULT_CHARACTER_STYLE_ID) ? DEFAULT_CHARACTER_STYLE_ID : void 0;
|
|
58
|
+
const missing = [];
|
|
59
|
+
for (const [need, create] of Object.entries(REFERENCE_STYLES)) {
|
|
60
|
+
if (!needs[need]) continue;
|
|
61
|
+
const style = create();
|
|
62
|
+
if (!defined.has(style.styleId)) missing.push(basedOn === void 0 ? style : {
|
|
63
|
+
...style,
|
|
64
|
+
basedOn
|
|
65
|
+
});
|
|
66
|
+
}
|
|
67
|
+
return missing;
|
|
68
|
+
};
|
|
69
|
+
//#endregion
|
|
70
|
+
export { COMMENT_REFERENCE_STYLE_ID, ENDNOTE_REFERENCE_STYLE_ID, FOOTNOTE_REFERENCE_STYLE_ID, missingNoteReferenceStyles, noteReferenceNeeds };
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
import { document_d_exports } from "../types/document.js";
|
|
2
2
|
import { NumberingMap } from "./numberingParser.js";
|
|
3
3
|
//#region src/docx/numberingReferenceNormalization.d.ts
|
|
4
|
+
/** The codes this normalisation is reported under, owned here, not at the caller. */
|
|
5
|
+
declare const UNNUMBERED_PARAGRAPH_WARNING: "unnumbered-paragraph";
|
|
6
|
+
declare const UNNUMBERED_STYLE_WARNING: "unnumbered-style";
|
|
4
7
|
type NormalizeNumberingReferencesInput = {
|
|
5
8
|
documentBody: document_d_exports.DocumentBody;
|
|
6
9
|
numbering: NumberingMap;
|
|
@@ -23,4 +26,4 @@ type NormalizeStyleNumberingReferencesResult = {
|
|
|
23
26
|
};
|
|
24
27
|
declare const normalizeStyleNumberingReferences: ({ styles, numbering }: NormalizeStyleNumberingReferencesInput) => NormalizeStyleNumberingReferencesResult;
|
|
25
28
|
//#endregion
|
|
26
|
-
export { normalizeNumberingReferences, normalizeStyleNumberingReferences };
|
|
29
|
+
export { UNNUMBERED_PARAGRAPH_WARNING, UNNUMBERED_STYLE_WARNING, normalizeNumberingReferences, normalizeStyleNumberingReferences };
|
|
@@ -1,7 +1,23 @@
|
|
|
1
1
|
import { isNumberingReference } from "./numberingReference.js";
|
|
2
2
|
import { visitDocxParagraphs } from "./paragraphTraversal.js";
|
|
3
|
+
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
3
4
|
//#region src/docx/numberingReferenceNormalization.ts
|
|
4
5
|
/**
|
|
6
|
+
* Parse-boundary tolerance for `w:numPr` numbering references.
|
|
7
|
+
*
|
|
8
|
+
* A file Word opens must never reach a panic, so a reference that resolves to
|
|
9
|
+
* nothing is repaired here, on both tiers that can carry one: the paragraph and
|
|
10
|
+
* the paragraph style. Downstream code (and `assertStyleNumberingReferences` in
|
|
11
|
+
* particular) can then treat a numbering reference as resolvable.
|
|
12
|
+
*
|
|
13
|
+
* The repair writes the sentinel rather than deleting the `w:numPr`. ECMA-376
|
|
14
|
+
* 17.9.18 reserves `w:numId w:val="0"` for "the removal of numbering properties
|
|
15
|
+
* at a particular level in the style hierarchy"; deleting the element instead
|
|
16
|
+
* removes nothing, it only uncovers the tier below, handing the paragraph its
|
|
17
|
+
* `w:pStyle` numbering or the style its `w:basedOn` numbering. A source that
|
|
18
|
+
* shows no number would come back numbered.
|
|
19
|
+
*/
|
|
20
|
+
/**
|
|
5
21
|
* Whether a `w:numId` still reaches a level definition: a `w:num` that names a
|
|
6
22
|
* `w:abstractNum` the numbering part defines. Both hops can dangle, and a
|
|
7
23
|
* `w:num` with no `w:abstractNum` numbers nothing just as a missing one does.
|
|
@@ -11,6 +27,9 @@ const resolvesNumbering = (numId, numbering) => {
|
|
|
11
27
|
const abstractNumId = numbering.getAbstractNumId(numId);
|
|
12
28
|
return abstractNumId !== null && numbering.getAbstract(abstractNumId) !== null;
|
|
13
29
|
};
|
|
30
|
+
/** The codes this normalisation is reported under, owned here, not at the caller. */
|
|
31
|
+
const UNNUMBERED_PARAGRAPH_WARNING = PARSE_WARNING_CODES.unnumberedParagraph;
|
|
32
|
+
const UNNUMBERED_STYLE_WARNING = PARSE_WARNING_CODES.unnumberedStyle;
|
|
14
33
|
const normalizeNumberingReferences = ({ documentBody, numbering, headers, footers, footnotes, endnotes }) => {
|
|
15
34
|
let unnumberedDanglingReferences = 0;
|
|
16
35
|
visitDocxParagraphs({
|
|
@@ -42,4 +61,4 @@ const normalizeStyleNumberingReferences = ({ styles, numbering }) => {
|
|
|
42
61
|
return { unnumberedStyleIds };
|
|
43
62
|
};
|
|
44
63
|
//#endregion
|
|
45
|
-
export { normalizeNumberingReferences, normalizeStyleNumberingReferences };
|
|
64
|
+
export { UNNUMBERED_PARAGRAPH_WARNING, UNNUMBERED_STYLE_WARNING, normalizeNumberingReferences, normalizeStyleNumberingReferences };
|
|
@@ -1,23 +1,4 @@
|
|
|
1
1
|
//#region src/docx/paraIdRangeNormalization.d.ts
|
|
2
|
-
/**
|
|
3
|
-
* Keep every paragraph id a package carries inside the range the schema gives
|
|
4
|
-
* it.
|
|
5
|
-
*
|
|
6
|
-
* `w14:paraId`, `w14:textId` and the comment-part ids that reference a
|
|
7
|
-
* paragraph are `ST_LongHexNumber` with a maximum: the value has to be below
|
|
8
|
-
* `0x80000000`, so the ids are 31-bit. Producers exist that write eight hex
|
|
9
|
-
* digits without that bound, and folio preserves the ids a document arrives
|
|
10
|
-
* with — so a package can carry an out-of-range id in, and a save that copies
|
|
11
|
-
* it through hands a consumer a package it will refuse.
|
|
12
|
-
*
|
|
13
|
-
* {@link paraIdInRange} is the one mapping, and it is a pure function of the
|
|
14
|
-
* value alone. That is what lets the parser and the save agree without
|
|
15
|
-
* consulting each other: a paragraph's id in the model is the id the file gets,
|
|
16
|
-
* so bringing an id into range does not make a document's own identity move
|
|
17
|
-
* under it between reading and writing. The package pass below is the same
|
|
18
|
-
* mapping applied to every attribute that carries such an id, so a paragraph
|
|
19
|
-
* and every reference to it move together.
|
|
20
|
-
*/
|
|
21
2
|
/**
|
|
22
3
|
* `value` when it already fits, and a value derived from it when it does not.
|
|
23
4
|
*
|
|
@@ -1,22 +1,24 @@
|
|
|
1
|
-
import { isValidHexColor } from "../utils/colorResolver.js";
|
|
2
1
|
import { isValidHexId } from "../utils/hexId.js";
|
|
3
2
|
import { parseBookmarkEnd as parseBookmarkEnd$1, parseBookmarkStart as parseBookmarkStart$1 } from "./bookmarkParser.js";
|
|
3
|
+
import { parseBorderSpec } from "./borderParser.js";
|
|
4
4
|
import { parseFieldType } from "./fieldParser.js";
|
|
5
5
|
import { parseHyperlink as parseHyperlink$1, parseHyperlinkChild } from "./hyperlinkParser.js";
|
|
6
|
+
import { parseMarkupRangeMarker, parseMoveBookmarkMarker } from "./markupRangeMarker.js";
|
|
6
7
|
import { markerFormattingFromLevel, numberingLevelHasMarkerSlot } from "./numberingParser.js";
|
|
7
8
|
import { isNumberingReference } from "./numberingReference.js";
|
|
8
9
|
import { paraIdInRange } from "./paraIdRangeNormalization.js";
|
|
9
10
|
import { assignParagraphPropertySource } from "./paragraphPropertySource.js";
|
|
10
|
-
import {
|
|
11
|
+
import { FrameWrapSchema, FrameXAlignSchema, FrameYAlignSchema, LineSpacingRuleSchema, ParagraphAlignmentSchema, TabLeaderSchema, TabStopAlignmentSchema, narrowEnum } from "./parserEnums.js";
|
|
11
12
|
import { consolidateParagraphContent } from "./runConsolidator.js";
|
|
12
13
|
import { parseRun, parseRunProperties } from "./runParser.js";
|
|
13
14
|
import { parseSdtProperties } from "./sdtProperties.js";
|
|
14
15
|
import { parseSectionProperties } from "./sectionParser.js";
|
|
16
|
+
import { parseShading } from "./shadingParser.js";
|
|
15
17
|
import { parsePropertyChangeInfo, parseTrackedChangeInfo } from "./trackedChangeInfo.js";
|
|
16
18
|
import { captureVerbatimXml } from "./verbatimCapture.js";
|
|
17
|
-
import { WORDPROCESSINGML_NAMESPACE_URIS, cloneElement, findChild, findChildByNamespaceUri, findChildren, findChildrenByNamespaceUri, getAttribute, getAttributeByNamespaceUri, getChildElements, getLocalName, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseNumberingLevelAttribute, parseNumericAttribute, selectAlternateContentBranch } from "./xmlParser.js";
|
|
19
|
+
import { WORDPROCESSINGML_NAMESPACE_URIS, cloneElement, findChild, findChildByNamespaceUri, findChildren, findChildrenByNamespaceUri, getAttribute, getAttributeByNamespaceUri, getChildElements, getLocalName, getNamespaceUri, matchesName, mergeXmlnsDeclarations, parseBooleanElement, parseNumberingLevelAttribute, parseNumericAttribute, parseOnOffAttribute, selectAlternateContentBranch } from "./xmlParser.js";
|
|
18
20
|
import { panic } from "better-result";
|
|
19
|
-
import { PARAGRAPH_MARK_CHANGE_KINDS, REVIEW_CARRIERS } from "@stll/docx-core/model";
|
|
21
|
+
import { BIDI_CONTROLS, PARAGRAPH_MARK_CHANGE_KINDS, REVIEW_CARRIERS } from "@stll/docx-core/model";
|
|
20
22
|
//#region src/docx/paragraphParser.ts
|
|
21
23
|
const FOLIO_REVIEW_HISTORY_NAMESPACES = /* @__PURE__ */ new Set(["urn:stella:folio:review-history:1"]);
|
|
22
24
|
/**
|
|
@@ -36,63 +38,6 @@ function extractMathText(el) {
|
|
|
36
38
|
return text;
|
|
37
39
|
}
|
|
38
40
|
/**
|
|
39
|
-
* Parse color value from attributes
|
|
40
|
-
*/
|
|
41
|
-
function parseColorValue(rgb, themeColor, themeTint, themeShade) {
|
|
42
|
-
const color = {};
|
|
43
|
-
if (rgb && rgb !== "auto") color.rgb = rgb;
|
|
44
|
-
else if (rgb === "auto") color.auto = true;
|
|
45
|
-
const validatedThemeColor = narrowEnum(themeColor, ThemeColorSlotSchema);
|
|
46
|
-
if (validatedThemeColor) color.themeColor = validatedThemeColor;
|
|
47
|
-
if (themeTint) color.themeTint = themeTint;
|
|
48
|
-
if (themeShade) color.themeShade = themeShade;
|
|
49
|
-
return color;
|
|
50
|
-
}
|
|
51
|
-
/**
|
|
52
|
-
* Parse shading properties (w:shd)
|
|
53
|
-
*/
|
|
54
|
-
function parseShadingProperties(shd) {
|
|
55
|
-
if (!shd) return;
|
|
56
|
-
const props = {};
|
|
57
|
-
const color = getAttribute(shd, "w", "color");
|
|
58
|
-
if (color && color !== "auto" && isValidHexColor(color)) props.color = { rgb: color };
|
|
59
|
-
const fill = getAttribute(shd, "w", "fill");
|
|
60
|
-
if (fill && fill !== "auto" && isValidHexColor(fill)) props.fill = { rgb: fill };
|
|
61
|
-
const validatedThemeFill = narrowEnum(getAttribute(shd, "w", "themeFill"), ThemeColorSlotSchema);
|
|
62
|
-
if (validatedThemeFill) {
|
|
63
|
-
props.fill = props.fill || {};
|
|
64
|
-
props.fill.themeColor = validatedThemeFill;
|
|
65
|
-
}
|
|
66
|
-
const themeFillTint = getAttribute(shd, "w", "themeFillTint");
|
|
67
|
-
if (themeFillTint && props.fill) props.fill.themeTint = themeFillTint;
|
|
68
|
-
const themeFillShade = getAttribute(shd, "w", "themeFillShade");
|
|
69
|
-
if (themeFillShade && props.fill) props.fill.themeShade = themeFillShade;
|
|
70
|
-
const pattern = narrowEnum(getAttribute(shd, "w", "val"), ShadingPatternSchema);
|
|
71
|
-
if (pattern) props.pattern = pattern;
|
|
72
|
-
return Object.keys(props).length > 0 ? props : void 0;
|
|
73
|
-
}
|
|
74
|
-
/**
|
|
75
|
-
* Parse border specification (w:top, w:bottom, w:left, w:right, etc.)
|
|
76
|
-
*/
|
|
77
|
-
function parseBorderSpec(border) {
|
|
78
|
-
if (!border) return;
|
|
79
|
-
const rawStyle = getAttribute(border, "w", "val");
|
|
80
|
-
if (!rawStyle) return;
|
|
81
|
-
const spec = { style: narrowEnum(rawStyle, BorderStyleSchema) ?? rawStyle };
|
|
82
|
-
const colorVal = getAttribute(border, "w", "color");
|
|
83
|
-
const themeColor = getAttribute(border, "w", "themeColor");
|
|
84
|
-
if (colorVal || themeColor) spec.color = parseColorValue(colorVal, themeColor, getAttribute(border, "w", "themeTint"), getAttribute(border, "w", "themeShade"));
|
|
85
|
-
const sz = parseNumericAttribute(border, "w", "sz");
|
|
86
|
-
if (sz !== void 0) spec.size = sz;
|
|
87
|
-
const space = parseNumericAttribute(border, "w", "space");
|
|
88
|
-
if (space !== void 0) spec.space = space;
|
|
89
|
-
const shadowAttr = getAttribute(border, "w", "shadow");
|
|
90
|
-
if (shadowAttr) spec.shadow = shadowAttr === "1" || shadowAttr === "true";
|
|
91
|
-
const frame = getAttribute(border, "w", "frame");
|
|
92
|
-
if (frame) spec.frame = frame === "1" || frame === "true";
|
|
93
|
-
return spec;
|
|
94
|
-
}
|
|
95
|
-
/**
|
|
96
41
|
* Parse tab stops (w:tabs)
|
|
97
42
|
*/
|
|
98
43
|
function parseTabStops(tabs) {
|
|
@@ -273,10 +218,10 @@ function parseParagraphProperties(pPr, theme, styles) {
|
|
|
273
218
|
if (spacingExplicit.before || spacingExplicit.after) formatting.spacingExplicit = spacingExplicit;
|
|
274
219
|
const lineRule = narrowEnum(getAttribute(spacing, "w", "lineRule"), LineSpacingRuleSchema);
|
|
275
220
|
if (lineRule) formatting.lineSpacingRule = lineRule;
|
|
276
|
-
const
|
|
277
|
-
if (
|
|
278
|
-
const
|
|
279
|
-
if (
|
|
221
|
+
const beforeAutospacing = parseOnOffAttribute(spacing, "w", "beforeAutospacing");
|
|
222
|
+
if (beforeAutospacing !== void 0) formatting.beforeAutospacing = beforeAutospacing;
|
|
223
|
+
const afterAutospacing = parseOnOffAttribute(spacing, "w", "afterAutospacing");
|
|
224
|
+
if (afterAutospacing !== void 0) formatting.afterAutospacing = afterAutospacing;
|
|
280
225
|
}
|
|
281
226
|
const ind = propertyChildren.ind;
|
|
282
227
|
if (ind) {
|
|
@@ -315,7 +260,7 @@ function parseParagraphProperties(pPr, theme, styles) {
|
|
|
315
260
|
}
|
|
316
261
|
const shd = propertyChildren.shd;
|
|
317
262
|
if (shd) {
|
|
318
|
-
const shadingResult =
|
|
263
|
+
const shadingResult = parseShading(shd);
|
|
319
264
|
if (shadingResult !== void 0) formatting.shading = shadingResult;
|
|
320
265
|
}
|
|
321
266
|
const tabs = propertyChildren.tabs;
|
|
@@ -348,6 +293,8 @@ function parseParagraphProperties(pPr, theme, styles) {
|
|
|
348
293
|
if (val !== void 0) formatting.numPr.ilvl = val;
|
|
349
294
|
}
|
|
350
295
|
}
|
|
296
|
+
const numberingChange = findChild(numPr, "w", "numberingChange");
|
|
297
|
+
if (numberingChange) formatting.numberingChangeXml = captureVerbatimXml(numberingChange);
|
|
351
298
|
}
|
|
352
299
|
const outlineLvl = propertyChildren.outlineLvl;
|
|
353
300
|
if (outlineLvl) {
|
|
@@ -653,6 +600,27 @@ const hyperlinkRevisionWrapperType = (node) => {
|
|
|
653
600
|
default: return;
|
|
654
601
|
}
|
|
655
602
|
};
|
|
603
|
+
const OMML_NAMESPACE = "http://schemas.openxmlformats.org/officeDocument/2006/math";
|
|
604
|
+
/**
|
|
605
|
+
* A bare OMML element as an inline equation, or nothing when the child is not one.
|
|
606
|
+
*
|
|
607
|
+
* `m:oMath` and `m:oMathPara` have branches of their own. This is for the rest
|
|
608
|
+
* of `m:EG_OMathMathElements` — `m:f`, `m:acc`, `m:rad` and their siblings —
|
|
609
|
+
* which the schema admits wherever `m:oMath` is admitted. They carry no
|
|
610
|
+
* structure the editable model holds, so they travel as the markup they
|
|
611
|
+
* arrived as, exactly like the equations that do have a wrapper.
|
|
612
|
+
*/
|
|
613
|
+
const mathContentOf = (child) => {
|
|
614
|
+
if (getNamespaceUri(child) !== OMML_NAMESPACE) return;
|
|
615
|
+
const equation = {
|
|
616
|
+
type: "mathEquation",
|
|
617
|
+
display: "inline",
|
|
618
|
+
ommlXml: captureVerbatimXml(child)
|
|
619
|
+
};
|
|
620
|
+
const plainText = extractMathText(child);
|
|
621
|
+
if (plainText) equation.plainText = plainText;
|
|
622
|
+
return equation;
|
|
623
|
+
};
|
|
656
624
|
const isHyperlinkChildContent = (content) => content.type === "run" || content.type === "bookmarkStart" || content.type === "bookmarkEnd";
|
|
657
625
|
/**
|
|
658
626
|
* A `w:hyperlink` as paragraph content, with any revision wrapper it holds
|
|
@@ -742,10 +710,8 @@ function parseSimpleField(node, styles, theme, rels, media, rootXmlns = {}) {
|
|
|
742
710
|
fieldType: parseFieldType(instruction),
|
|
743
711
|
content: []
|
|
744
712
|
};
|
|
745
|
-
|
|
746
|
-
if (
|
|
747
|
-
const dirty = getAttribute(node, "w", "dirty");
|
|
748
|
-
if (dirty === "1" || dirty === "true") field.dirty = true;
|
|
713
|
+
if (parseOnOffAttribute(node, "w", "fldLock") === true) field.fldLock = true;
|
|
714
|
+
if (parseOnOffAttribute(node, "w", "dirty") === true) field.dirty = true;
|
|
749
715
|
const inScopeXmlns = mergeXmlnsDeclarations(rootXmlns, node);
|
|
750
716
|
const children = getChildElements(node);
|
|
751
717
|
for (const child of children) {
|
|
@@ -986,57 +952,53 @@ function parseParagraphContents(paraElement, styles, theme, _numbering, rels, me
|
|
|
986
952
|
contents.push(...inner);
|
|
987
953
|
break;
|
|
988
954
|
}
|
|
989
|
-
case "moveFromRangeStart":
|
|
990
|
-
const id = Number.parseInt(getAttribute(child, "w", "id") ?? "0", 10);
|
|
991
|
-
const name = getAttribute(child, "w", "name") ?? "";
|
|
955
|
+
case "moveFromRangeStart":
|
|
992
956
|
contents.push({
|
|
993
957
|
type: "moveFromRangeStart",
|
|
994
|
-
|
|
995
|
-
name
|
|
958
|
+
...parseMoveBookmarkMarker(child)
|
|
996
959
|
});
|
|
997
960
|
break;
|
|
998
|
-
|
|
999
|
-
case "moveFromRangeEnd": {
|
|
1000
|
-
const id = Number.parseInt(getAttribute(child, "w", "id") ?? "0", 10);
|
|
961
|
+
case "moveFromRangeEnd":
|
|
1001
962
|
contents.push({
|
|
1002
963
|
type: "moveFromRangeEnd",
|
|
1003
|
-
|
|
964
|
+
...parseMarkupRangeMarker(child)
|
|
1004
965
|
});
|
|
1005
966
|
break;
|
|
1006
|
-
|
|
1007
|
-
case "moveToRangeStart": {
|
|
1008
|
-
const id = Number.parseInt(getAttribute(child, "w", "id") ?? "0", 10);
|
|
1009
|
-
const name = getAttribute(child, "w", "name") ?? "";
|
|
967
|
+
case "moveToRangeStart":
|
|
1010
968
|
contents.push({
|
|
1011
969
|
type: "moveToRangeStart",
|
|
1012
|
-
|
|
1013
|
-
name
|
|
970
|
+
...parseMoveBookmarkMarker(child)
|
|
1014
971
|
});
|
|
1015
972
|
break;
|
|
1016
|
-
|
|
1017
|
-
case "moveToRangeEnd": {
|
|
1018
|
-
const id = Number.parseInt(getAttribute(child, "w", "id") ?? "0", 10);
|
|
973
|
+
case "moveToRangeEnd":
|
|
1019
974
|
contents.push({
|
|
1020
975
|
type: "moveToRangeEnd",
|
|
1021
|
-
|
|
976
|
+
...parseMarkupRangeMarker(child)
|
|
1022
977
|
});
|
|
1023
978
|
break;
|
|
1024
|
-
|
|
1025
|
-
case "commentRangeStart": {
|
|
1026
|
-
const commentId = Number.parseInt(getAttribute(child, "w", "id") ?? "0", 10);
|
|
979
|
+
case "commentRangeStart":
|
|
1027
980
|
contents.push({
|
|
1028
981
|
type: "commentRangeStart",
|
|
1029
|
-
|
|
982
|
+
...parseMarkupRangeMarker(child)
|
|
1030
983
|
});
|
|
1031
984
|
break;
|
|
1032
|
-
|
|
1033
|
-
case "commentRangeEnd": {
|
|
1034
|
-
const commentId = Number.parseInt(getAttribute(child, "w", "id") ?? "0", 10);
|
|
985
|
+
case "commentRangeEnd":
|
|
1035
986
|
contents.push({
|
|
1036
987
|
type: "commentRangeEnd",
|
|
1037
|
-
|
|
988
|
+
...parseMarkupRangeMarker(child)
|
|
1038
989
|
});
|
|
1039
990
|
break;
|
|
991
|
+
case "bdo":
|
|
992
|
+
case "dir": {
|
|
993
|
+
const direction = getAttribute(child, "w", "val");
|
|
994
|
+
const wrapper = {
|
|
995
|
+
type: "bidiWrapper",
|
|
996
|
+
control: localName === "bdo" ? BIDI_CONTROLS.override : BIDI_CONTROLS.embedding,
|
|
997
|
+
content: parseParagraphContents(child, styles, theme, null, rels, media, trackedContext, inScopeXmlns)
|
|
998
|
+
};
|
|
999
|
+
if (direction === "ltr" || direction === "rtl") wrapper.direction = direction;
|
|
1000
|
+
contents.push(wrapper);
|
|
1001
|
+
break;
|
|
1040
1002
|
}
|
|
1041
1003
|
case "oMath":
|
|
1042
1004
|
case "oMathPara": {
|
|
@@ -1052,7 +1014,11 @@ function parseParagraphContents(paraElement, styles, theme, _numbering, rels, me
|
|
|
1052
1014
|
contents.push(mathEq);
|
|
1053
1015
|
break;
|
|
1054
1016
|
}
|
|
1055
|
-
default:
|
|
1017
|
+
default: {
|
|
1018
|
+
const mathElement = mathContentOf(child);
|
|
1019
|
+
if (mathElement !== void 0) contents.push(mathElement);
|
|
1020
|
+
break;
|
|
1021
|
+
}
|
|
1056
1022
|
}
|
|
1057
1023
|
}
|
|
1058
1024
|
if (inComplexField && afterSeparator) contents.push(...complexFieldResultRuns);
|
|
@@ -1270,7 +1236,8 @@ const getParagraphContentText = (content) => {
|
|
|
1270
1236
|
case "complexField": return content.fieldResult.map(getRunText).join("");
|
|
1271
1237
|
case "inlineSdt": return content.content.map(getParagraphContentText).join("");
|
|
1272
1238
|
case "insertion":
|
|
1273
|
-
case "moveTo":
|
|
1239
|
+
case "moveTo":
|
|
1240
|
+
case "bidiWrapper": return content.content.map(getParagraphContentText).join("");
|
|
1274
1241
|
case "deletion":
|
|
1275
1242
|
case "moveFrom": return "";
|
|
1276
1243
|
case "mathEquation": return content.plainText ?? "";
|
|
@@ -9,6 +9,42 @@ type DocxParagraphSurfaces = {
|
|
|
9
9
|
};
|
|
10
10
|
/** Visit every run directly owned by a paragraph's inline-content tree. */
|
|
11
11
|
declare const visitParagraphRuns: (paragraph: document_d_exports.Paragraph, visit: (run: document_d_exports.Run) => void) => void;
|
|
12
|
+
/**
|
|
13
|
+
* One position in an inline-content array: the array itself, the index, and
|
|
14
|
+
* what sits there. A normaliser that drops or rewrites a marker needs the
|
|
15
|
+
* array and the index, not only the value.
|
|
16
|
+
*/
|
|
17
|
+
type InlineContentSlot = {
|
|
18
|
+
content: document_d_exports.ParagraphContent[];
|
|
19
|
+
index: number;
|
|
20
|
+
item: document_d_exports.ParagraphContent;
|
|
21
|
+
};
|
|
22
|
+
/**
|
|
23
|
+
* Visit every inline-content position a paragraph owns, in document order,
|
|
24
|
+
* descending through every wrapper that is transparent to a range marker.
|
|
25
|
+
*
|
|
26
|
+
* The model validator walks the whole inline tree; a normaliser that walks
|
|
27
|
+
* only `paragraph.content` sees a different document from the one the
|
|
28
|
+
* validator judges, and a marker inside `w:ins`, `w:hyperlink`, `w:sdt`,
|
|
29
|
+
* `w:bdo` or `w:dir` then reaches the validator unnormalised. Both sides read
|
|
30
|
+
* the tree through this one traversal so they cannot disagree again.
|
|
31
|
+
*
|
|
32
|
+
* A complex field's runs are skipped: `fieldCode` and `fieldResult` hold runs
|
|
33
|
+
* only, and a run is not a marker position.
|
|
34
|
+
*/
|
|
35
|
+
declare const visitInlineContentSlots: (paragraph: document_d_exports.Paragraph, visit: (slot: InlineContentSlot) => void) => void;
|
|
36
|
+
/**
|
|
37
|
+
* Positions marked for removal, per inline-content array.
|
|
38
|
+
*
|
|
39
|
+
* Removal shifts every later index in that array, so a normaliser records the
|
|
40
|
+
* positions while it reads and drops them once, after it has finished reading.
|
|
41
|
+
*/
|
|
42
|
+
declare class InlineContentRemovals {
|
|
43
|
+
#private;
|
|
44
|
+
mark({ content, index }: Pick<InlineContentSlot, "content" | "index">): void;
|
|
45
|
+
/** Applies every marked removal and answers how many items were dropped. */
|
|
46
|
+
apply(): number;
|
|
47
|
+
}
|
|
12
48
|
declare const visitDocxParagraphs: ({ documentBody, headers, footers, footnotes, endnotes }: DocxParagraphSurfaces, visit: (paragraph: document_d_exports.Paragraph) => void) => void;
|
|
13
49
|
//#endregion
|
|
14
|
-
export { DocxParagraphSurfaces, visitDocxParagraphs, visitParagraphRuns };
|
|
50
|
+
export { DocxParagraphSurfaces, InlineContentRemovals, InlineContentSlot, visitDocxParagraphs, visitInlineContentSlots, visitParagraphRuns };
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { panic } from "better-result";
|
|
1
2
|
//#region src/docx/paragraphTraversal.ts
|
|
2
3
|
/** Visit every run directly owned by a paragraph's inline-content tree. */
|
|
3
4
|
const visitParagraphRuns = (paragraph, visit) => {
|
|
@@ -15,6 +16,7 @@ const visitParagraphRuns = (paragraph, visit) => {
|
|
|
15
16
|
case "moveFrom":
|
|
16
17
|
case "moveTo":
|
|
17
18
|
case "inlineSdt":
|
|
19
|
+
case "bidiWrapper":
|
|
18
20
|
for (const child of content.content) visitParagraphContent(child);
|
|
19
21
|
return;
|
|
20
22
|
case "complexField":
|
|
@@ -36,6 +38,87 @@ const visitParagraphRuns = (paragraph, visit) => {
|
|
|
36
38
|
};
|
|
37
39
|
for (const content of paragraph.content) visitParagraphContent(content);
|
|
38
40
|
};
|
|
41
|
+
/**
|
|
42
|
+
* Visit every inline-content position a paragraph owns, in document order,
|
|
43
|
+
* descending through every wrapper that is transparent to a range marker.
|
|
44
|
+
*
|
|
45
|
+
* The model validator walks the whole inline tree; a normaliser that walks
|
|
46
|
+
* only `paragraph.content` sees a different document from the one the
|
|
47
|
+
* validator judges, and a marker inside `w:ins`, `w:hyperlink`, `w:sdt`,
|
|
48
|
+
* `w:bdo` or `w:dir` then reaches the validator unnormalised. Both sides read
|
|
49
|
+
* the tree through this one traversal so they cannot disagree again.
|
|
50
|
+
*
|
|
51
|
+
* A complex field's runs are skipped: `fieldCode` and `fieldResult` hold runs
|
|
52
|
+
* only, and a run is not a marker position.
|
|
53
|
+
*/
|
|
54
|
+
const visitInlineContentSlots = (paragraph, visit) => {
|
|
55
|
+
const visitContent = (content) => {
|
|
56
|
+
for (const [index, item] of content.entries()) {
|
|
57
|
+
visit({
|
|
58
|
+
content,
|
|
59
|
+
index,
|
|
60
|
+
item
|
|
61
|
+
});
|
|
62
|
+
switch (item.type) {
|
|
63
|
+
case "hyperlink":
|
|
64
|
+
visitContent(item.children);
|
|
65
|
+
break;
|
|
66
|
+
case "simpleField":
|
|
67
|
+
case "inlineSdt":
|
|
68
|
+
case "insertion":
|
|
69
|
+
case "deletion":
|
|
70
|
+
case "moveFrom":
|
|
71
|
+
case "moveTo":
|
|
72
|
+
case "bidiWrapper":
|
|
73
|
+
visitContent(item.content);
|
|
74
|
+
break;
|
|
75
|
+
case "run":
|
|
76
|
+
case "complexField":
|
|
77
|
+
case "bookmarkStart":
|
|
78
|
+
case "bookmarkEnd":
|
|
79
|
+
case "commentRangeStart":
|
|
80
|
+
case "commentRangeEnd":
|
|
81
|
+
case "commentReference":
|
|
82
|
+
case "moveFromRangeStart":
|
|
83
|
+
case "moveFromRangeEnd":
|
|
84
|
+
case "moveToRangeStart":
|
|
85
|
+
case "moveToRangeEnd":
|
|
86
|
+
case "mathEquation": break;
|
|
87
|
+
default: panic(`Unsupported paragraph content: ${JSON.stringify(item)}`);
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
};
|
|
91
|
+
visitContent(paragraph.content);
|
|
92
|
+
};
|
|
93
|
+
/**
|
|
94
|
+
* Positions marked for removal, per inline-content array.
|
|
95
|
+
*
|
|
96
|
+
* Removal shifts every later index in that array, so a normaliser records the
|
|
97
|
+
* positions while it reads and drops them once, after it has finished reading.
|
|
98
|
+
*/
|
|
99
|
+
var InlineContentRemovals = class {
|
|
100
|
+
#byContent = /* @__PURE__ */ new Map();
|
|
101
|
+
mark({ content, index }) {
|
|
102
|
+
const indexes = this.#byContent.get(content);
|
|
103
|
+
if (indexes) {
|
|
104
|
+
indexes.add(index);
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
this.#byContent.set(content, /* @__PURE__ */ new Set([index]));
|
|
108
|
+
}
|
|
109
|
+
/** Applies every marked removal and answers how many items were dropped. */
|
|
110
|
+
apply() {
|
|
111
|
+
let removed = 0;
|
|
112
|
+
for (const [content, indexes] of this.#byContent) {
|
|
113
|
+
if (indexes.size === 0) continue;
|
|
114
|
+
const next = content.filter((_, index) => !indexes.has(index));
|
|
115
|
+
removed += content.length - next.length;
|
|
116
|
+
content.length = 0;
|
|
117
|
+
content.push(...next);
|
|
118
|
+
}
|
|
119
|
+
return removed;
|
|
120
|
+
}
|
|
121
|
+
};
|
|
39
122
|
const visitDocxParagraphs = ({ documentBody, headers, footers, footnotes, endnotes }, visit) => {
|
|
40
123
|
const seenParagraphs = /* @__PURE__ */ new WeakSet();
|
|
41
124
|
const visitParagraph = (paragraph) => {
|
|
@@ -76,4 +159,4 @@ const visitDocxParagraphs = ({ documentBody, headers, footers, footnotes, endnot
|
|
|
76
159
|
for (const comment of documentBody.comments ?? []) for (const paragraph of comment.content) visitParagraph(paragraph);
|
|
77
160
|
};
|
|
78
161
|
//#endregion
|
|
79
|
-
export { visitDocxParagraphs, visitParagraphRuns };
|
|
162
|
+
export { InlineContentRemovals, visitDocxParagraphs, visitInlineContentSlots, visitParagraphRuns };
|