@stll/folio-core 0.42.0 → 0.44.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ai-edits/headless.js +6 -5
- package/dist/ai-edits/index.d.ts +2 -2
- package/dist/ai-edits/index.js +2 -2
- package/dist/ai-edits/snapshot.js +13 -9
- package/dist/compare/content-alignment.js +16 -1
- package/dist/compare/inline-atoms.js +1 -1
- package/dist/compare/style-resources.js +6 -0
- package/dist/compat/eigenpal.d.ts +2 -2
- package/dist/controller/layoutPipeline.d.ts +2 -1
- package/dist/controller/layoutPipeline.js +5 -2
- package/dist/controller/layoutSession.d.ts +2 -1
- package/dist/controller/layoutSession.js +1 -0
- package/dist/docx/appVersionNormalization.d.ts +0 -18
- package/dist/docx/blockContentParser.js +10 -1
- package/dist/docx/blockRangeMarkers.d.ts +36 -0
- package/dist/docx/blockRangeMarkers.js +59 -0
- package/dist/docx/bookmarkParser.d.ts +2 -20
- package/dist/docx/bookmarkParser.js +6 -30
- package/dist/docx/borderParser.d.ts +13 -0
- package/dist/docx/borderParser.js +71 -0
- package/dist/docx/builtInStyles.d.ts +165 -0
- package/dist/docx/builtInStyles.js +239 -0
- package/dist/docx/commentIdNormalization.d.ts +10 -0
- package/dist/docx/commentIdNormalization.js +33 -0
- package/dist/docx/commentParser.d.ts +2 -1
- package/dist/docx/commentParser.js +34 -7
- package/dist/docx/commentReferenceNormalization.d.ts +4 -1
- package/dist/docx/commentReferenceNormalization.js +23 -14
- package/dist/docx/danglingRelationshipReferences.d.ts +15 -0
- package/dist/docx/danglingRelationshipReferences.js +30 -0
- package/dist/docx/defaultParagraphStyle.d.ts +39 -0
- package/dist/docx/defaultParagraphStyle.js +54 -0
- package/dist/docx/documentParser.d.ts +2 -1
- package/dist/docx/documentParser.js +2 -2
- package/dist/docx/drawingUtils.d.ts +8 -1
- package/dist/docx/drawingUtils.js +12 -3
- package/dist/docx/fieldParser.js +3 -5
- package/dist/docx/footnoteParser.d.ts +3 -2
- package/dist/docx/footnoteParser.js +19 -2
- package/dist/docx/groupDrawingParser.js +1 -1
- package/dist/docx/headerFooterRefParser.d.ts +15 -3
- package/dist/docx/headerFooterRefParser.js +51 -14
- package/dist/docx/headerFooterReferenceNormalization.d.ts +4 -1
- package/dist/docx/headerFooterReferenceNormalization.js +5 -1
- package/dist/docx/hyperlinkParser.js +11 -15
- package/dist/docx/imageParser.d.ts +1 -1
- package/dist/docx/imageParser.js +22 -18
- package/dist/docx/imageRawXml.js +5 -5
- package/dist/docx/markupRangeMarker.d.ts +15 -0
- package/dist/docx/markupRangeMarker.js +44 -0
- package/dist/docx/noteReferenceStyles.d.ts +29 -0
- package/dist/docx/noteReferenceStyles.js +70 -0
- package/dist/docx/numberingParser.js +2 -1
- package/dist/docx/numberingReference.d.ts +21 -0
- package/dist/docx/numberingReference.js +21 -0
- package/dist/docx/numberingReferenceNormalization.d.ts +14 -2
- package/dist/docx/numberingReferenceNormalization.js +51 -9
- package/dist/docx/paraIdRangeNormalization.d.ts +0 -19
- package/dist/docx/paragraphParser.js +69 -101
- package/dist/docx/paragraphPropertySource.js +1 -0
- package/dist/docx/paragraphTraversal.d.ts +37 -1
- package/dist/docx/paragraphTraversal.js +84 -1
- package/dist/docx/parseContext.d.ts +37 -0
- package/dist/docx/parseContext.js +67 -0
- package/dist/docx/parseWarningMessage.d.ts +6 -0
- package/dist/docx/parseWarningMessage.js +44 -0
- package/dist/docx/parser.js +86 -24
- package/dist/docx/relsParser.d.ts +28 -11
- package/dist/docx/relsParser.js +26 -13
- package/dist/docx/revisionIdNormalization.js +81 -7
- package/dist/docx/rezip.js +85 -27
- package/dist/docx/runConsolidator.js +1 -2
- package/dist/docx/runParser.d.ts +8 -1
- package/dist/docx/runParser.js +30 -48
- package/dist/docx/sectionParser.d.ts +2 -1
- package/dist/docx/sectionParser.js +21 -65
- package/dist/docx/serializer/borderSerializer.d.ts +1 -2
- package/dist/docx/serializer/commentSerializer.js +22 -9
- package/dist/docx/serializer/documentSerializer.d.ts +1 -5
- package/dist/docx/serializer/documentSerializer.js +6 -16
- package/dist/docx/serializer/headerFooterSerializer.js +5 -0
- package/dist/docx/serializer/markupRangeAttributes.d.ts +8 -0
- package/dist/docx/serializer/markupRangeAttributes.js +24 -0
- package/dist/docx/serializer/noteSerializer.js +5 -0
- package/dist/docx/serializer/paragraphSerializer.d.ts +1 -5
- package/dist/docx/serializer/paragraphSerializer.js +29 -35
- package/dist/docx/serializer/runSerializer.js +13 -7
- package/dist/docx/serializer/tableSerializer.js +28 -13
- package/dist/docx/serializer/textFormattingSerializer.d.ts +2 -3
- package/dist/docx/server/build.js +8 -1
- package/dist/docx/server/createBilingualDocument.js +15 -22
- package/dist/docx/server/extractDocxText.js +3 -4
- package/dist/docx/shadingParser.d.ts +6 -0
- package/dist/docx/shadingParser.js +32 -0
- package/dist/docx/shapeParser.js +3 -3
- package/dist/docx/styleParser.js +15 -89
- package/dist/docx/styleReferenceResolution.d.ts +36 -0
- package/dist/docx/styleReferenceResolution.js +51 -0
- package/dist/docx/tableLook.d.ts +57 -0
- package/dist/docx/tableLook.js +63 -0
- package/dist/docx/tableParser.d.ts +7 -9
- package/dist/docx/tableParser.js +64 -110
- package/dist/docx/textBoxParser.js +4 -4
- package/dist/docx/trackedMoveRangeNormalization.d.ts +3 -1
- package/dist/docx/trackedMoveRangeNormalization.js +11 -21
- package/dist/docx/transitionalSpelling.d.ts +13 -2
- package/dist/docx/transitionalSpelling.js +23 -1
- package/dist/docx/verbatimCapture.js +4 -11
- package/dist/docx/vmlImageParser.js +2 -2
- package/dist/docx/watermarkParser.js +2 -2
- package/dist/docx/xmlParser.d.ts +22 -32
- package/dist/docx/xmlParser.js +36 -21
- package/dist/index.d.ts +2 -2
- package/dist/internal/pageBreakRunSourceDescendantIndex.d.ts +2 -0
- package/dist/internal/pageBreakRunSourceDescendantIndex.js +9 -6
- package/dist/internal/paragraphFormattingSerialization.d.ts +2 -3
- package/dist/internal/paragraphFormattingSerialization.js +26 -6
- package/dist/layout-bridge/convert/footnoteLayout.js +2 -7
- package/dist/layout-bridge/convert/templatePreviewFlow.d.ts +19 -11
- package/dist/layout-bridge/convert/templatePreviewFlow.js +103 -38
- package/dist/layout-bridge/convert/toFlowBlocks.js +12 -3
- package/dist/layout-engine/index.d.ts +2 -2
- package/dist/layout-engine/index.js +2 -2
- package/dist/layout-engine/measure/measureBlocks.js +1 -6
- package/dist/layout-engine/types.d.ts +8 -2
- package/dist/layout-engine/types.js +35 -2
- package/dist/markdown/index.js +1 -1
- package/dist/markdown/internals.d.ts +6 -1
- package/dist/markdown/internals.js +14 -1
- package/dist/markdown/renderBlock.js +35 -21
- package/dist/markdown/renderParagraph.js +14 -5
- package/dist/markdown/renderRuns.js +4 -3
- package/dist/markdown/renderTable.js +4 -3
- package/dist/markdown/trailers.js +41 -7
- package/dist/markdown/types.d.ts +3 -7
- package/dist/prosemirror/attrs/index.js +2 -5
- package/dist/prosemirror/bookmarkBoundaryAttrs.d.ts +11 -1
- package/dist/prosemirror/bookmarkBoundaryAttrs.js +18 -3
- package/dist/prosemirror/commands/index.d.ts +3 -3
- package/dist/prosemirror/commands/index.js +2 -2
- package/dist/prosemirror/commands/pageBreak.js +12 -1
- package/dist/prosemirror/commands/paragraph.d.ts +3 -3
- package/dist/prosemirror/commands/paragraph.js +2 -2
- package/dist/prosemirror/commentIdAllocator.js +2 -7
- package/dist/prosemirror/conversion/fromProseDoc.js +131 -41
- package/dist/prosemirror/conversion/toProseDoc.d.ts +1 -14
- package/dist/prosemirror/conversion/toProseDoc.js +402 -328
- package/dist/prosemirror/extensions/core/ParagraphExtension.d.ts +14 -1
- package/dist/prosemirror/extensions/core/ParagraphExtension.js +11 -6
- package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -3
- package/dist/prosemirror/extensions/features/ListExtension.js +42 -4
- package/dist/prosemirror/extensions/features/PasteCleanupExtension.d.ts +4 -1
- package/dist/prosemirror/extensions/features/PasteCleanupExtension.js +6 -2
- package/dist/prosemirror/extensions/features/pastedHeadingStyles.d.ts +7 -0
- package/dist/prosemirror/extensions/features/pastedHeadingStyles.js +74 -0
- package/dist/prosemirror/extensions/marks/markUtils.d.ts +11 -3
- package/dist/prosemirror/extensions/marks/markUtils.js +98 -19
- package/dist/prosemirror/extensions/nodes/BookmarkBoundaryExtension.js +7 -3
- package/dist/prosemirror/extensions/nodes/ImageExtension.js +2 -1
- package/dist/prosemirror/extensions/nodes/ShapeExtension.js +1 -0
- package/dist/prosemirror/extensions/nodes/TableExtension.js +15 -1
- package/dist/prosemirror/extensions/types.d.ts +2 -2
- package/dist/prosemirror/index.d.ts +3 -3
- package/dist/prosemirror/index.js +3 -3
- package/dist/prosemirror/insertOperations.d.ts +9 -2
- package/dist/prosemirror/insertOperations.js +9 -4
- package/dist/prosemirror/listMarker.js +2 -1
- package/dist/prosemirror/numberedRefFields.js +2 -1
- package/dist/prosemirror/pageBreakRunProjection.d.ts +11 -3
- package/dist/prosemirror/pageBreakRunProjection.js +16 -8
- package/dist/prosemirror/paragraphFormattingProvenance.d.ts +159 -0
- package/dist/prosemirror/paragraphFormattingProvenance.js +106 -0
- package/dist/prosemirror/plugins/documentStyles.d.ts +9 -1
- package/dist/prosemirror/plugins/documentStyles.js +11 -1
- package/dist/prosemirror/plugins/index.d.ts +2 -2
- package/dist/prosemirror/plugins/index.js +2 -2
- package/dist/prosemirror/plugins/revisionIds.d.ts +11 -2
- package/dist/prosemirror/plugins/revisionIds.js +21 -6
- package/dist/prosemirror/plugins/templatePreviewValues.d.ts +42 -1
- package/dist/prosemirror/plugins/templatePreviewValues.js +217 -14
- package/dist/prosemirror/runFormattingReconciliation.js +3 -2
- package/dist/prosemirror/runStyleFormatting.d.ts +1 -1
- package/dist/prosemirror/schema/nodes.d.ts +31 -0
- package/dist/prosemirror/styles/resolvedStyleAttrs.js +4 -1
- package/dist/prosemirror/styles/styleResolver.d.ts +9 -0
- package/dist/prosemirror/styles/styleResolver.js +15 -6
- package/dist/prosemirror/utils/visualLineNavigation.d.ts +22 -2
- package/dist/prosemirror/utils/visualLineNavigation.js +80 -63
- package/dist/style-engine/styleEngine.d.ts +4 -1
- package/dist/style-engine/styleEngine.js +3 -0
- package/dist/style-sets/extract.js +35 -11
- package/dist/style-sets/stellaStyle.js +46 -39
- package/dist/style-sets/styleSetNormalization.d.ts +19 -0
- package/dist/style-sets/styleSetNormalization.js +99 -0
- package/dist/types/content.d.ts +2 -2
- package/dist/utils/createDocument.js +145 -20
- package/dist/utils/headingCollector.d.ts +8 -5
- package/dist/utils/headingCollector.js +23 -25
- package/dist/utils/tableOfContentsStyle.js +9 -2
- package/package.json +3 -2
- package/dist/docx/textWhitespace.d.ts +0 -4
- package/dist/docx/textWhitespace.js +0 -4
- package/dist/layout-bridge/engine/tableWidthUtils.d.ts +0 -6
- package/dist/layout-bridge/engine/tableWidthUtils.js +0 -25
- package/dist/markdown/headings.d.ts +0 -13
- package/dist/markdown/headings.js +0 -20
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import { BorderStyleSchema, ThemeColorSlotSchema, narrowEnum } from "./parserEnums.js";
|
|
2
|
+
import { getAttribute, parseNumericAttribute, parseOnOffAttribute } from "./xmlParser.js";
|
|
3
|
+
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
4
|
+
//#region src/docx/borderParser.ts
|
|
5
|
+
/**
|
|
6
|
+
* The one reader for `CT_Border`, shared by the paragraph (`w:pBdr`), style,
|
|
7
|
+
* table (`w:tblBorders`/`w:tcBorders`) and page (`w:pgBorders`) tiers.
|
|
8
|
+
*
|
|
9
|
+
* `w:val` is `ST_Border`, whose 193 members include two distinct "no border"
|
|
10
|
+
* tokens: `nil` and `none`. They are not interchangeable downstream, and an
|
|
11
|
+
* explicit one overrides a border inherited from the container, so the member
|
|
12
|
+
* the author wrote is preserved exactly. Members outside the model's known
|
|
13
|
+
* union survive verbatim rather than collapsing to a default, which is how the
|
|
14
|
+
* repo already treats `w:numFmt`, `w:suff` and `w:tab`.
|
|
15
|
+
*/
|
|
16
|
+
const parseBorderColor = (border) => {
|
|
17
|
+
const rgb = getAttribute(border, "w", "color");
|
|
18
|
+
const themeColor = getAttribute(border, "w", "themeColor");
|
|
19
|
+
const themeTint = getAttribute(border, "w", "themeTint");
|
|
20
|
+
const themeShade = getAttribute(border, "w", "themeShade");
|
|
21
|
+
if (rgb === null && themeColor === null && themeTint === null && themeShade === null) return;
|
|
22
|
+
const color = {};
|
|
23
|
+
if (rgb === "auto") color.auto = true;
|
|
24
|
+
else if (rgb) color.rgb = rgb;
|
|
25
|
+
const validatedThemeColor = narrowEnum(themeColor, ThemeColorSlotSchema);
|
|
26
|
+
if (validatedThemeColor) color.themeColor = validatedThemeColor;
|
|
27
|
+
if (themeTint) color.themeTint = themeTint;
|
|
28
|
+
if (themeShade) color.themeShade = themeShade;
|
|
29
|
+
return color;
|
|
30
|
+
};
|
|
31
|
+
/**
|
|
32
|
+
* `w:val` is `use="required"` on `CT_Border`. An element without it states no
|
|
33
|
+
* style at all, so the border is dropped rather than invented as `none`:
|
|
34
|
+
* `none` is an authored token that cancels an inherited border, and a
|
|
35
|
+
* malformed element is not evidence the author wanted that.
|
|
36
|
+
*/
|
|
37
|
+
function parseBorderSpec(border, context) {
|
|
38
|
+
if (!border) return;
|
|
39
|
+
const rawStyle = getAttribute(border, "w", "val");
|
|
40
|
+
if (!rawStyle) {
|
|
41
|
+
context?.warn({
|
|
42
|
+
code: PARSE_WARNING_CODES.borderWithoutValue,
|
|
43
|
+
element: border.name ?? "border"
|
|
44
|
+
});
|
|
45
|
+
return;
|
|
46
|
+
}
|
|
47
|
+
const spec = { style: narrowEnum(rawStyle, BorderStyleSchema) ?? rawStyle };
|
|
48
|
+
const color = parseBorderColor(border);
|
|
49
|
+
if (color) spec.color = color;
|
|
50
|
+
const size = parseNumericAttribute(border, "w", "sz");
|
|
51
|
+
if (size !== void 0) spec.size = size;
|
|
52
|
+
const space = parseNumericAttribute(border, "w", "space");
|
|
53
|
+
if (space !== void 0) spec.space = space;
|
|
54
|
+
const shadow = parseOnOffAttribute(border, "w", "shadow");
|
|
55
|
+
if (shadow !== void 0) spec.shadow = shadow;
|
|
56
|
+
const frame = parseOnOffAttribute(border, "w", "frame");
|
|
57
|
+
if (frame !== void 0) spec.frame = frame;
|
|
58
|
+
const artRelationshipId = getAttribute(border, "w", "id")?.trim();
|
|
59
|
+
if (artRelationshipId) spec.artRelationshipId = artRelationshipId;
|
|
60
|
+
const topLeftArtRelationshipId = getAttribute(border, "w", "topLeft")?.trim();
|
|
61
|
+
if (topLeftArtRelationshipId) spec.topLeftArtRelationshipId = topLeftArtRelationshipId;
|
|
62
|
+
const topRightArtRelationshipId = getAttribute(border, "w", "topRight")?.trim();
|
|
63
|
+
if (topRightArtRelationshipId) spec.topRightArtRelationshipId = topRightArtRelationshipId;
|
|
64
|
+
const bottomLeftArtRelationshipId = getAttribute(border, "w", "bottomLeft")?.trim();
|
|
65
|
+
if (bottomLeftArtRelationshipId) spec.bottomLeftArtRelationshipId = bottomLeftArtRelationshipId;
|
|
66
|
+
const bottomRightArtRelationshipId = getAttribute(border, "w", "bottomRight")?.trim();
|
|
67
|
+
if (bottomRightArtRelationshipId) spec.bottomRightArtRelationshipId = bottomRightArtRelationshipId;
|
|
68
|
+
return spec;
|
|
69
|
+
}
|
|
70
|
+
//#endregion
|
|
71
|
+
export { parseBorderSpec };
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
import { document_d_exports } from "../types/document.js";
|
|
2
|
+
//#region src/docx/builtInStyles.d.ts
|
|
3
|
+
/**
|
|
4
|
+
* The tenth `w:outlineLvl` value. 17.3.1.20: "the val attribute … can be from
|
|
5
|
+
* 0 to 9, where 9 specifically indicates that there is no outline level
|
|
6
|
+
* specifically applied to this paragraph." It is a deliberate "not a heading",
|
|
7
|
+
* not a tenth level. Every range test goes through
|
|
8
|
+
* {@link isHeadingOutlineLevel} so the reserved value keeps one meaning across
|
|
9
|
+
* the codebase.
|
|
10
|
+
*
|
|
11
|
+
* The same clause adds that an omitted element "is assumed to be 9". That
|
|
12
|
+
* default cannot be applied to a *style* definition, because 17.7.1 tells
|
|
13
|
+
* producers not to write a property "already been set by a previous level of
|
|
14
|
+
* the style hierarchy": a document that names a style `heading 1` and omits
|
|
15
|
+
* the level is inheriting the consumer's built-in definition, which carries
|
|
16
|
+
* level 0. An absent level therefore means "unspecified, ask the name", and
|
|
17
|
+
* only a written 9 means body text.
|
|
18
|
+
*/
|
|
19
|
+
declare const BODY_TEXT_OUTLINE_LEVEL = 9;
|
|
20
|
+
/** True when an outline level names a heading rather than body text. */
|
|
21
|
+
declare const isHeadingOutlineLevel: (level: number | null | undefined) => level is number;
|
|
22
|
+
/**
|
|
23
|
+
* Compare style names the way producers actually write them. The corpus shows
|
|
24
|
+
* both `heading 1` (Annex L, 94.8%) and `Heading 1` (5.2%), and LibreOffice
|
|
25
|
+
* drops the space entirely (`Heading1`, `IntenseQuote`), so case and
|
|
26
|
+
* whitespace are the tolerance. A name is otherwise matched whole: a style a
|
|
27
|
+
* Czech template calls `Nadpis 1` stays a custom style.
|
|
28
|
+
*/
|
|
29
|
+
declare const normalizeStyleName: (name: string) => string;
|
|
30
|
+
/**
|
|
31
|
+
* The `w:name` Word itself writes for each built-in, and therefore the
|
|
32
|
+
* spelling every style table folio authors must use. One owner: a style set and
|
|
33
|
+
* the classifier that reads it cannot drift apart if both name the same
|
|
34
|
+
* constant.
|
|
35
|
+
*
|
|
36
|
+
* Word is not uniformly cased and guessing gets it wrong, so each value is the
|
|
37
|
+
* spelling that dominates Microsoft Word output in the public corpus:
|
|
38
|
+
* `footnote text` 375 against 49 `Footnote Text`, `footer` 1145 against 2,
|
|
39
|
+
* `caption` 390 against 60 — but `Body Text` 424 against 2, `Title` 621
|
|
40
|
+
* against 2, and the auto-generated linked character styles (`Footnote Text
|
|
41
|
+
* Char` 248, `Endnote Text Char` 116) title-cased without exception.
|
|
42
|
+
* {@link normalizeStyleName} makes matching tolerant of all of it; this map is
|
|
43
|
+
* about what folio *writes*.
|
|
44
|
+
*/
|
|
45
|
+
declare const BUILT_IN_STYLE_NAME: {
|
|
46
|
+
/** 4,515 Word occurrences against 3 lowercase. */
|
|
47
|
+
readonly normal: "Normal";
|
|
48
|
+
readonly bodyText: "Body Text";
|
|
49
|
+
readonly title: "Title";
|
|
50
|
+
readonly subtitle: "Subtitle";
|
|
51
|
+
readonly quote: "Quote";
|
|
52
|
+
readonly intenseQuote: "Intense Quote";
|
|
53
|
+
readonly listParagraph: "List Paragraph";
|
|
54
|
+
readonly tocHeading: "TOC Heading";
|
|
55
|
+
readonly caption: "caption";
|
|
56
|
+
readonly header: "header";
|
|
57
|
+
readonly footer: "footer";
|
|
58
|
+
readonly footnoteText: "footnote text";
|
|
59
|
+
readonly commentReference: "annotation reference";
|
|
60
|
+
readonly footnoteReference: "footnote reference";
|
|
61
|
+
readonly footnoteTextChar: "Footnote Text Char";
|
|
62
|
+
readonly endnoteText: "endnote text";
|
|
63
|
+
readonly endnoteReference: "endnote reference";
|
|
64
|
+
readonly endnoteTextChar: "Endnote Text Char";
|
|
65
|
+
readonly hyperlink: "Hyperlink";
|
|
66
|
+
readonly defaultParagraphFont: "Default Paragraph Font";
|
|
67
|
+
readonly noList: "No List";
|
|
68
|
+
readonly normalTable: "Normal Table";
|
|
69
|
+
readonly tableGrid: "Table Grid";
|
|
70
|
+
};
|
|
71
|
+
type BuiltInStyleName = (typeof BUILT_IN_STYLE_NAME)[keyof typeof BUILT_IN_STYLE_NAME];
|
|
72
|
+
/**
|
|
73
|
+
* The name of a built-in heading, from its zero-based outline level.
|
|
74
|
+
* Lowercase: 1,066 Word occurrences of `heading 1` against 13 `Heading 1`, and
|
|
75
|
+
* Annex L writes the latent-style exceptions the same way.
|
|
76
|
+
*/
|
|
77
|
+
declare const builtInHeadingStyleName: (outlineLevel: number) => string;
|
|
78
|
+
/** The name of a built-in TOC entry style, from its one-based level (`toc 1`). */
|
|
79
|
+
declare const builtInTableOfContentsStyleName: (level: number) => string;
|
|
80
|
+
/**
|
|
81
|
+
* A document's styles indexed by what they *are* rather than by what they are
|
|
82
|
+
* called. Built once per document: the consumers below classify every
|
|
83
|
+
* paragraph, and rebuilding the maps per paragraph would make each of them
|
|
84
|
+
* quadratic.
|
|
85
|
+
*/
|
|
86
|
+
type BuiltInStyleIndex = {
|
|
87
|
+
/** The style's effective `w:outlineLvl`, including 9, or undefined. */
|
|
88
|
+
outlineLevelOf: (styleId: string | null | undefined) => number | undefined;
|
|
89
|
+
/** The zero-based level a built-in heading *name* implies, or undefined. */
|
|
90
|
+
headingLevelFromNameOf: (styleId: string | null | undefined) => number | undefined;
|
|
91
|
+
/** The built-in this style is, by name, or undefined for a custom style. */
|
|
92
|
+
builtInNameOf: (styleId: string | null | undefined) => BuiltInStyleName | undefined;
|
|
93
|
+
/**
|
|
94
|
+
* The level an English built-in heading *id* implies, and only when the
|
|
95
|
+
* package defines no style under it. See {@link resolveHeadingLevel} tier 3.
|
|
96
|
+
*/
|
|
97
|
+
undefinedBuiltInHeadingLevelOf: (styleId: string | null | undefined) => number | undefined;
|
|
98
|
+
/** The document's style id for a built-in heading level (zero-based). */
|
|
99
|
+
styleIdForHeadingLevel: (level: number) => string | undefined;
|
|
100
|
+
/** The document's style id for a built-in TOC entry level (one-based, `toc 1`). */
|
|
101
|
+
styleIdForTableOfContentsLevel: (level: number) => string | undefined;
|
|
102
|
+
/** The document's style id for a named built-in, e.g. `TOC Heading`. */
|
|
103
|
+
styleIdForBuiltInName: (name: BuiltInStyleName) => string | undefined;
|
|
104
|
+
};
|
|
105
|
+
declare const createBuiltInStyleIndex: (styles: Iterable<document_d_exports.Style>, docDefaults?: document_d_exports.DocDefaults | undefined) => BuiltInStyleIndex;
|
|
106
|
+
/** An index over a document that defines no styles: every lookup misses. */
|
|
107
|
+
declare const EMPTY_BUILT_IN_STYLE_INDEX: BuiltInStyleIndex;
|
|
108
|
+
/**
|
|
109
|
+
* What a consumer knows about a paragraph: its style id and whatever
|
|
110
|
+
* `w:outlineLvl` applies to it. The ProseMirror `outlineLevel` attr already
|
|
111
|
+
* holds direct-else-style resolution, so passing it here agrees with passing
|
|
112
|
+
* direct formatting from the DOCX model.
|
|
113
|
+
*/
|
|
114
|
+
type ParagraphOutlineSource = {
|
|
115
|
+
styleId?: string | null | undefined;
|
|
116
|
+
outlineLevel?: number | null | undefined;
|
|
117
|
+
};
|
|
118
|
+
/**
|
|
119
|
+
* The heading level a paragraph carries, zero-based (`heading 1` is 0), or
|
|
120
|
+
* undefined when it is not a heading.
|
|
121
|
+
*
|
|
122
|
+
* Precedence:
|
|
123
|
+
*
|
|
124
|
+
* 1. An effective outline level decides on its own, including
|
|
125
|
+
* {@link BODY_TEXT_OUTLINE_LEVEL}, which means "not a heading". A style
|
|
126
|
+
* named `heading 5` whose outline level is 0 is a level-1 heading; a style
|
|
127
|
+
* named `heading 3` reset to 9 is body text. The format gives the outline
|
|
128
|
+
* level to field calculation (17.3.1.20) and leaves the name to the UI
|
|
129
|
+
* (17.7.4.9), so the level is the one the document asserts.
|
|
130
|
+
* 2. Only when no outline level is set anywhere does the built-in `w:name`
|
|
131
|
+
* decide — the style is then inheriting the consumer's own built-in
|
|
132
|
+
* definition, which supplies the level.
|
|
133
|
+
* 3. Last resort, and only for a `w:pStyle` the package defines no style for:
|
|
134
|
+
* the id itself, read as the English built-in id. 17.7.4.17 makes a style
|
|
135
|
+
* without `w:customStyle` a built-in and lets an application recognise it
|
|
136
|
+
* "if the associated style ID is known", which is the one case where the id
|
|
137
|
+
* is all the information left. `defaultParagraphStyle.ts` keeps the same
|
|
138
|
+
* last tier for `Normal`. A document that defines its heading styles never
|
|
139
|
+
* reaches this, so it cannot override a name or an outline level — and a
|
|
140
|
+
* localized package never writes an English id to begin with.
|
|
141
|
+
*
|
|
142
|
+
* Two consequences of rule 1 are deliberate, not oversights.
|
|
143
|
+
*
|
|
144
|
+
* **An outline level on a style that is not a heading still makes a heading.**
|
|
145
|
+
* The corpus has 680 such occurrences across 161 files, including `Title` at
|
|
146
|
+
* level 0 (42×) and `Subtitle` at level 1 (23×), plus `H1`, `Sub-heading`,
|
|
147
|
+
* `index heading` and a `DSTOC1-1`…`DSTOC8-8` family. Setting the level is how
|
|
148
|
+
* a document asks for a paragraph to be outlined, and Word's navigation pane
|
|
149
|
+
* and a `TOC \u` field both honour it, so folio does not second-guess a
|
|
150
|
+
* document that asked. Suppressing `Title` here would mean folio deciding a
|
|
151
|
+
* document's outline differs from Word's.
|
|
152
|
+
*
|
|
153
|
+
* **An outline level that disagrees with a built-in heading name wins.** 26
|
|
154
|
+
* corpus styles do this (`heading 5` at level 0, `heading 3` at level 1, and
|
|
155
|
+
* so on), all from non-Word producers or hand-authored fixtures. The format
|
|
156
|
+
* gives the level to field calculation (17.3.1.20) and the name to the user
|
|
157
|
+
* interface (17.7.4.9), so the level is the machine-readable claim and the
|
|
158
|
+
* name is a label. This is the one rule below that was not confirmed against
|
|
159
|
+
* Word itself.
|
|
160
|
+
*/
|
|
161
|
+
declare const resolveHeadingLevel: (paragraph: ParagraphOutlineSource, index: BuiltInStyleIndex) => number | undefined;
|
|
162
|
+
/** True when the paragraph's style is Word's `Quote` or `Intense Quote`. */
|
|
163
|
+
declare const isQuoteStyle: (styleId: string | null | undefined, index: BuiltInStyleIndex) => boolean;
|
|
164
|
+
//#endregion
|
|
165
|
+
export { BODY_TEXT_OUTLINE_LEVEL, BUILT_IN_STYLE_NAME, BuiltInStyleIndex, EMPTY_BUILT_IN_STYLE_INDEX, ParagraphOutlineSource, builtInHeadingStyleName, builtInTableOfContentsStyleName, createBuiltInStyleIndex, isHeadingOutlineLevel, isQuoteStyle, normalizeStyleName, resolveHeadingLevel };
|
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
import { BUILT_IN_DEFAULT_PARAGRAPH_STYLE_NAME, resolveDefaultParagraphStyle } from "./defaultParagraphStyle.js";
|
|
2
|
+
//#region src/docx/builtInStyles.ts
|
|
3
|
+
/**
|
|
4
|
+
* The tenth `w:outlineLvl` value. 17.3.1.20: "the val attribute … can be from
|
|
5
|
+
* 0 to 9, where 9 specifically indicates that there is no outline level
|
|
6
|
+
* specifically applied to this paragraph." It is a deliberate "not a heading",
|
|
7
|
+
* not a tenth level. Every range test goes through
|
|
8
|
+
* {@link isHeadingOutlineLevel} so the reserved value keeps one meaning across
|
|
9
|
+
* the codebase.
|
|
10
|
+
*
|
|
11
|
+
* The same clause adds that an omitted element "is assumed to be 9". That
|
|
12
|
+
* default cannot be applied to a *style* definition, because 17.7.1 tells
|
|
13
|
+
* producers not to write a property "already been set by a previous level of
|
|
14
|
+
* the style hierarchy": a document that names a style `heading 1` and omits
|
|
15
|
+
* the level is inheriting the consumer's built-in definition, which carries
|
|
16
|
+
* level 0. An absent level therefore means "unspecified, ask the name", and
|
|
17
|
+
* only a written 9 means body text.
|
|
18
|
+
*/
|
|
19
|
+
const BODY_TEXT_OUTLINE_LEVEL = 9;
|
|
20
|
+
/** The highest `w:outlineLvl` that still names a heading (outline level nine). */
|
|
21
|
+
const MAX_HEADING_OUTLINE_LEVEL = 8;
|
|
22
|
+
/** True when an outline level names a heading rather than body text. */
|
|
23
|
+
const isHeadingOutlineLevel = (level) => typeof level === "number" && Number.isInteger(level) && level >= 0 && level <= MAX_HEADING_OUTLINE_LEVEL;
|
|
24
|
+
/**
|
|
25
|
+
* Compare style names the way producers actually write them. The corpus shows
|
|
26
|
+
* both `heading 1` (Annex L, 94.8%) and `Heading 1` (5.2%), and LibreOffice
|
|
27
|
+
* drops the space entirely (`Heading1`, `IntenseQuote`), so case and
|
|
28
|
+
* whitespace are the tolerance. A name is otherwise matched whole: a style a
|
|
29
|
+
* Czech template calls `Nadpis 1` stays a custom style.
|
|
30
|
+
*/
|
|
31
|
+
const normalizeStyleName = (name) => name.trim().toLowerCase().replace(/\s+/gu, "");
|
|
32
|
+
/**
|
|
33
|
+
* The `w:name` Word itself writes for each built-in, and therefore the
|
|
34
|
+
* spelling every style table folio authors must use. One owner: a style set and
|
|
35
|
+
* the classifier that reads it cannot drift apart if both name the same
|
|
36
|
+
* constant.
|
|
37
|
+
*
|
|
38
|
+
* Word is not uniformly cased and guessing gets it wrong, so each value is the
|
|
39
|
+
* spelling that dominates Microsoft Word output in the public corpus:
|
|
40
|
+
* `footnote text` 375 against 49 `Footnote Text`, `footer` 1145 against 2,
|
|
41
|
+
* `caption` 390 against 60 — but `Body Text` 424 against 2, `Title` 621
|
|
42
|
+
* against 2, and the auto-generated linked character styles (`Footnote Text
|
|
43
|
+
* Char` 248, `Endnote Text Char` 116) title-cased without exception.
|
|
44
|
+
* {@link normalizeStyleName} makes matching tolerant of all of it; this map is
|
|
45
|
+
* about what folio *writes*.
|
|
46
|
+
*/
|
|
47
|
+
const BUILT_IN_STYLE_NAME = {
|
|
48
|
+
/** 4,515 Word occurrences against 3 lowercase. */
|
|
49
|
+
normal: BUILT_IN_DEFAULT_PARAGRAPH_STYLE_NAME,
|
|
50
|
+
bodyText: "Body Text",
|
|
51
|
+
title: "Title",
|
|
52
|
+
subtitle: "Subtitle",
|
|
53
|
+
quote: "Quote",
|
|
54
|
+
intenseQuote: "Intense Quote",
|
|
55
|
+
listParagraph: "List Paragraph",
|
|
56
|
+
tocHeading: "TOC Heading",
|
|
57
|
+
caption: "caption",
|
|
58
|
+
header: "header",
|
|
59
|
+
footer: "footer",
|
|
60
|
+
footnoteText: "footnote text",
|
|
61
|
+
commentReference: "annotation reference",
|
|
62
|
+
footnoteReference: "footnote reference",
|
|
63
|
+
footnoteTextChar: "Footnote Text Char",
|
|
64
|
+
endnoteText: "endnote text",
|
|
65
|
+
endnoteReference: "endnote reference",
|
|
66
|
+
endnoteTextChar: "Endnote Text Char",
|
|
67
|
+
hyperlink: "Hyperlink",
|
|
68
|
+
defaultParagraphFont: "Default Paragraph Font",
|
|
69
|
+
noList: "No List",
|
|
70
|
+
normalTable: "Normal Table",
|
|
71
|
+
tableGrid: "Table Grid"
|
|
72
|
+
};
|
|
73
|
+
/**
|
|
74
|
+
* The name of a built-in heading, from its zero-based outline level.
|
|
75
|
+
* Lowercase: 1,066 Word occurrences of `heading 1` against 13 `Heading 1`, and
|
|
76
|
+
* Annex L writes the latent-style exceptions the same way.
|
|
77
|
+
*/
|
|
78
|
+
const builtInHeadingStyleName = (outlineLevel) => `heading ${outlineLevel + 1}`;
|
|
79
|
+
/** The name of a built-in TOC entry style, from its one-based level (`toc 1`). */
|
|
80
|
+
const builtInTableOfContentsStyleName = (level) => `toc ${level}`;
|
|
81
|
+
/**
|
|
82
|
+
* `heading 1`…`heading 9` against an already-normalised name. Word's built-in
|
|
83
|
+
* names stop at nine, matching `w:outlineLvl`'s nine heading values, so
|
|
84
|
+
* `Cmsor10` → `Címsor 10` is a user style rather than a tenth built-in.
|
|
85
|
+
*/
|
|
86
|
+
const BUILT_IN_HEADING_NAME = /^heading(?<level>[1-9])$/u;
|
|
87
|
+
/** `toc 1`…`toc 9`, the styles a `TOC` field writes its entries in. */
|
|
88
|
+
const BUILT_IN_TABLE_OF_CONTENTS_NAME = /^toc(?<level>[1-9])$/u;
|
|
89
|
+
/**
|
|
90
|
+
* The outline level a built-in heading *name* implies (zero-based, so
|
|
91
|
+
* `heading 1` is 0), or undefined when the name is not a built-in heading.
|
|
92
|
+
*/
|
|
93
|
+
const headingOutlineLevelFromStyleName = (name) => {
|
|
94
|
+
if (name === void 0) return;
|
|
95
|
+
const level = BUILT_IN_HEADING_NAME.exec(normalizeStyleName(name))?.groups?.["level"];
|
|
96
|
+
return level === void 0 ? void 0 : Number.parseInt(level, 10) - 1;
|
|
97
|
+
};
|
|
98
|
+
/**
|
|
99
|
+
* Every spelling a producer might write, mapped back to the canonical one, so
|
|
100
|
+
* a caller compares against {@link BUILT_IN_STYLE_NAME} rather than against a
|
|
101
|
+
* normalised form it would have to spell a second time.
|
|
102
|
+
*/
|
|
103
|
+
const CANONICAL_BY_NORMALIZED = new Map(Object.values(BUILT_IN_STYLE_NAME).map((name) => [normalizeStyleName(name), name]));
|
|
104
|
+
const asBuiltInName = (normalized) => CANONICAL_BY_NORMALIZED.get(normalized);
|
|
105
|
+
/**
|
|
106
|
+
* The outline level a style chain sets: the style's own `w:outlineLvl`, else
|
|
107
|
+
* the nearest ancestor's (17.7.1). `styleParser` already flattens `w:basedOn`
|
|
108
|
+
* for a parsed package, but a style set built in memory carries the raw chain,
|
|
109
|
+
* so the walk keeps both kinds of input on the same answer. `seen` guards the
|
|
110
|
+
* circular `basedOn` a malformed package can contain.
|
|
111
|
+
*/
|
|
112
|
+
const inheritedOutlineLevel = (style, styleById) => {
|
|
113
|
+
const seen = /* @__PURE__ */ new Set();
|
|
114
|
+
let current = style;
|
|
115
|
+
while (current && !seen.has(current.styleId)) {
|
|
116
|
+
seen.add(current.styleId);
|
|
117
|
+
if (current.pPr?.outlineLevel !== void 0) return current.pPr.outlineLevel;
|
|
118
|
+
current = current.basedOn === void 0 ? void 0 : styleById.get(current.basedOn);
|
|
119
|
+
}
|
|
120
|
+
};
|
|
121
|
+
const createBuiltInStyleIndex = (styles, docDefaults) => {
|
|
122
|
+
const styleIdByHeadingLevel = /* @__PURE__ */ new Map();
|
|
123
|
+
const styleIdByTableOfContentsLevel = /* @__PURE__ */ new Map();
|
|
124
|
+
const styleIdByBuiltInName = /* @__PURE__ */ new Map();
|
|
125
|
+
const paragraphStyles = [];
|
|
126
|
+
const paragraphStyleById = /* @__PURE__ */ new Map();
|
|
127
|
+
const styleById = /* @__PURE__ */ new Map();
|
|
128
|
+
for (const style of styles) {
|
|
129
|
+
styleById.set(style.styleId, style);
|
|
130
|
+
if (style.type === "paragraph") {
|
|
131
|
+
paragraphStyles.push(style);
|
|
132
|
+
paragraphStyleById.set(style.styleId, style);
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
const docDefaultOutlineLevel = docDefaults?.pPr?.outlineLevel;
|
|
136
|
+
for (const style of paragraphStyles) {
|
|
137
|
+
if (style.name === void 0) continue;
|
|
138
|
+
const namedLevel = headingOutlineLevelFromStyleName(style.name);
|
|
139
|
+
const headingLevel = namedLevel === void 0 ? void 0 : inheritedOutlineLevel(style, styleById) ?? docDefaultOutlineLevel ?? namedLevel;
|
|
140
|
+
if (headingLevel !== void 0 && isHeadingOutlineLevel(headingLevel) && !styleIdByHeadingLevel.has(headingLevel)) {
|
|
141
|
+
styleIdByHeadingLevel.set(headingLevel, style.styleId);
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
const normalized = normalizeStyleName(style.name);
|
|
145
|
+
const tocLevel = BUILT_IN_TABLE_OF_CONTENTS_NAME.exec(normalized)?.groups?.["level"];
|
|
146
|
+
if (tocLevel !== void 0) {
|
|
147
|
+
const level = Number.parseInt(tocLevel, 10);
|
|
148
|
+
if (!styleIdByTableOfContentsLevel.has(level)) styleIdByTableOfContentsLevel.set(level, style.styleId);
|
|
149
|
+
continue;
|
|
150
|
+
}
|
|
151
|
+
const builtInName = asBuiltInName(normalized);
|
|
152
|
+
if (builtInName !== void 0 && !styleIdByBuiltInName.has(builtInName)) styleIdByBuiltInName.set(builtInName, style.styleId);
|
|
153
|
+
}
|
|
154
|
+
/**
|
|
155
|
+
* The style a paragraph actually resolves against. ECMA-376 17.7.2 layer 3
|
|
156
|
+
* is "the paragraph's own style chain", and a paragraph with no `w:pStyle`
|
|
157
|
+
* (or one naming a style the package never defines, or one naming a
|
|
158
|
+
* character style) takes the default paragraph style — the same fallback
|
|
159
|
+
* `StyleResolver` applies, so the model and the editor cannot drift apart.
|
|
160
|
+
*/
|
|
161
|
+
const defaultStyle = resolveDefaultParagraphStyle(paragraphStyles);
|
|
162
|
+
const styleFor = (styleId) => (styleId === null || styleId === void 0 ? void 0 : paragraphStyleById.get(styleId)) ?? defaultStyle;
|
|
163
|
+
const outlineLevelCache = /* @__PURE__ */ new Map();
|
|
164
|
+
return {
|
|
165
|
+
outlineLevelOf: (styleId) => {
|
|
166
|
+
if (outlineLevelCache.has(styleId)) return outlineLevelCache.get(styleId);
|
|
167
|
+
const style = styleFor(styleId);
|
|
168
|
+
const level = (style === void 0 ? void 0 : inheritedOutlineLevel(style, styleById)) ?? docDefaultOutlineLevel;
|
|
169
|
+
outlineLevelCache.set(styleId, level);
|
|
170
|
+
return level;
|
|
171
|
+
},
|
|
172
|
+
headingLevelFromNameOf: (styleId) => headingOutlineLevelFromStyleName(styleFor(styleId)?.name),
|
|
173
|
+
builtInNameOf: (styleId) => {
|
|
174
|
+
const name = styleFor(styleId)?.name;
|
|
175
|
+
return name === void 0 ? void 0 : asBuiltInName(normalizeStyleName(name));
|
|
176
|
+
},
|
|
177
|
+
undefinedBuiltInHeadingLevelOf: (styleId) => styleId === null || styleId === void 0 || paragraphStyleById.has(styleId) ? void 0 : headingOutlineLevelFromStyleName(styleId),
|
|
178
|
+
styleIdForHeadingLevel: (level) => styleIdByHeadingLevel.get(level),
|
|
179
|
+
styleIdForTableOfContentsLevel: (level) => styleIdByTableOfContentsLevel.get(level),
|
|
180
|
+
styleIdForBuiltInName: (name) => styleIdByBuiltInName.get(name)
|
|
181
|
+
};
|
|
182
|
+
};
|
|
183
|
+
/** An index over a document that defines no styles: every lookup misses. */
|
|
184
|
+
const EMPTY_BUILT_IN_STYLE_INDEX = createBuiltInStyleIndex([]);
|
|
185
|
+
/**
|
|
186
|
+
* The heading level a paragraph carries, zero-based (`heading 1` is 0), or
|
|
187
|
+
* undefined when it is not a heading.
|
|
188
|
+
*
|
|
189
|
+
* Precedence:
|
|
190
|
+
*
|
|
191
|
+
* 1. An effective outline level decides on its own, including
|
|
192
|
+
* {@link BODY_TEXT_OUTLINE_LEVEL}, which means "not a heading". A style
|
|
193
|
+
* named `heading 5` whose outline level is 0 is a level-1 heading; a style
|
|
194
|
+
* named `heading 3` reset to 9 is body text. The format gives the outline
|
|
195
|
+
* level to field calculation (17.3.1.20) and leaves the name to the UI
|
|
196
|
+
* (17.7.4.9), so the level is the one the document asserts.
|
|
197
|
+
* 2. Only when no outline level is set anywhere does the built-in `w:name`
|
|
198
|
+
* decide — the style is then inheriting the consumer's own built-in
|
|
199
|
+
* definition, which supplies the level.
|
|
200
|
+
* 3. Last resort, and only for a `w:pStyle` the package defines no style for:
|
|
201
|
+
* the id itself, read as the English built-in id. 17.7.4.17 makes a style
|
|
202
|
+
* without `w:customStyle` a built-in and lets an application recognise it
|
|
203
|
+
* "if the associated style ID is known", which is the one case where the id
|
|
204
|
+
* is all the information left. `defaultParagraphStyle.ts` keeps the same
|
|
205
|
+
* last tier for `Normal`. A document that defines its heading styles never
|
|
206
|
+
* reaches this, so it cannot override a name or an outline level — and a
|
|
207
|
+
* localized package never writes an English id to begin with.
|
|
208
|
+
*
|
|
209
|
+
* Two consequences of rule 1 are deliberate, not oversights.
|
|
210
|
+
*
|
|
211
|
+
* **An outline level on a style that is not a heading still makes a heading.**
|
|
212
|
+
* The corpus has 680 such occurrences across 161 files, including `Title` at
|
|
213
|
+
* level 0 (42×) and `Subtitle` at level 1 (23×), plus `H1`, `Sub-heading`,
|
|
214
|
+
* `index heading` and a `DSTOC1-1`…`DSTOC8-8` family. Setting the level is how
|
|
215
|
+
* a document asks for a paragraph to be outlined, and Word's navigation pane
|
|
216
|
+
* and a `TOC \u` field both honour it, so folio does not second-guess a
|
|
217
|
+
* document that asked. Suppressing `Title` here would mean folio deciding a
|
|
218
|
+
* document's outline differs from Word's.
|
|
219
|
+
*
|
|
220
|
+
* **An outline level that disagrees with a built-in heading name wins.** 26
|
|
221
|
+
* corpus styles do this (`heading 5` at level 0, `heading 3` at level 1, and
|
|
222
|
+
* so on), all from non-Word producers or hand-authored fixtures. The format
|
|
223
|
+
* gives the level to field calculation (17.3.1.20) and the name to the user
|
|
224
|
+
* interface (17.7.4.9), so the level is the machine-readable claim and the
|
|
225
|
+
* name is a label. This is the one rule below that was not confirmed against
|
|
226
|
+
* Word itself.
|
|
227
|
+
*/
|
|
228
|
+
const resolveHeadingLevel = (paragraph, index) => {
|
|
229
|
+
const effective = paragraph.outlineLevel ?? index.outlineLevelOf(paragraph.styleId);
|
|
230
|
+
if (effective !== null && effective !== void 0) return isHeadingOutlineLevel(effective) ? effective : void 0;
|
|
231
|
+
return index.headingLevelFromNameOf(paragraph.styleId) ?? index.undefinedBuiltInHeadingLevelOf(paragraph.styleId);
|
|
232
|
+
};
|
|
233
|
+
/** True when the paragraph's style is Word's `Quote` or `Intense Quote`. */
|
|
234
|
+
const isQuoteStyle = (styleId, index) => {
|
|
235
|
+
const name = index.builtInNameOf(styleId);
|
|
236
|
+
return name === BUILT_IN_STYLE_NAME.quote || name === BUILT_IN_STYLE_NAME.intenseQuote;
|
|
237
|
+
};
|
|
238
|
+
//#endregion
|
|
239
|
+
export { BODY_TEXT_OUTLINE_LEVEL, BUILT_IN_STYLE_NAME, EMPTY_BUILT_IN_STYLE_INDEX, builtInHeadingStyleName, builtInTableOfContentsStyleName, createBuiltInStyleIndex, isHeadingOutlineLevel, isQuoteStyle, normalizeStyleName, resolveHeadingLevel };
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { document_d_exports } from "../types/document.js";
|
|
2
|
+
//#region src/docx/commentIdNormalization.d.ts
|
|
3
|
+
/** The code this normalisation is reported under, owned here, not at the caller. */
|
|
4
|
+
declare const DUPLICATE_COMMENT_ID_WARNING: "duplicate-comment-id";
|
|
5
|
+
type NormalizeCommentIdsResult = {
|
|
6
|
+
droppedDuplicateComments: number;
|
|
7
|
+
};
|
|
8
|
+
declare const normalizeCommentIds: (comments: document_d_exports.Comment[]) => NormalizeCommentIdsResult;
|
|
9
|
+
//#endregion
|
|
10
|
+
export { DUPLICATE_COMMENT_ID_WARNING, NormalizeCommentIdsResult, normalizeCommentIds };
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
2
|
+
//#region src/docx/commentIdNormalization.ts
|
|
3
|
+
/**
|
|
4
|
+
* One comment per `w:id`, because that is all the body can address.
|
|
5
|
+
*
|
|
6
|
+
* `w:commentReference`, `w:commentRangeStart` and `w:commentRangeEnd` name a
|
|
7
|
+
* comment by its `w:id`, so two `w:comment` elements sharing an id make every
|
|
8
|
+
* marker that names it ambiguous. Word opens such a package and resolves each
|
|
9
|
+
* marker to the first definition, so folio keeps the first and drops the rest:
|
|
10
|
+
* no marker can address a later one, and re-numbering it would invent a
|
|
11
|
+
* comment nothing anchors. Microsoft's own conformance corpus ships a file
|
|
12
|
+
* that does this.
|
|
13
|
+
*
|
|
14
|
+
* Reply links survive unchanged. `w15:paraIdParent` resolves to a comment id
|
|
15
|
+
* before this runs, and that id still belongs to the definition that stayed.
|
|
16
|
+
*/
|
|
17
|
+
/** The code this normalisation is reported under, owned here, not at the caller. */
|
|
18
|
+
const DUPLICATE_COMMENT_ID_WARNING = PARSE_WARNING_CODES.duplicateCommentId;
|
|
19
|
+
const normalizeCommentIds = (comments) => {
|
|
20
|
+
const seen = /* @__PURE__ */ new Set();
|
|
21
|
+
let kept = 0;
|
|
22
|
+
for (const comment of comments) {
|
|
23
|
+
if (seen.has(comment.id)) continue;
|
|
24
|
+
seen.add(comment.id);
|
|
25
|
+
comments[kept] = comment;
|
|
26
|
+
kept += 1;
|
|
27
|
+
}
|
|
28
|
+
const droppedDuplicateComments = comments.length - kept;
|
|
29
|
+
comments.length = kept;
|
|
30
|
+
return { droppedDuplicateComments };
|
|
31
|
+
};
|
|
32
|
+
//#endregion
|
|
33
|
+
export { DUPLICATE_COMMENT_ID_WARNING, normalizeCommentIds };
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { document_d_exports } from "../types/document.js";
|
|
2
2
|
import { StyleMap } from "./styleParser.js";
|
|
3
|
+
import { ParseContext } from "./parseContext.js";
|
|
3
4
|
//#region src/docx/commentParser.d.ts
|
|
4
5
|
type CommentExtendedInfo = {
|
|
5
6
|
parentParaId?: string;
|
|
@@ -28,6 +29,6 @@ declare function parseCommentsExtended(xml: string): Map<string, CommentExtended
|
|
|
28
29
|
* local time. If `commentsExtendedXml` is provided, reply-thread
|
|
29
30
|
* parent links (`parentId`) and resolved state (`done`) are populated.
|
|
30
31
|
*/
|
|
31
|
-
declare function parseComments(commentsXml: string | null, styles: StyleMap | null, theme: document_d_exports.Theme | null, rels: document_d_exports.RelationshipMap, media: Map<string, document_d_exports.MediaFile>, commentsExtensibleXml?: string | null, commentsExtendedXml?: string | null): document_d_exports.Comment[];
|
|
32
|
+
declare function parseComments(commentsXml: string | null, styles: StyleMap | null, theme: document_d_exports.Theme | null, rels: document_d_exports.RelationshipMap, media: Map<string, document_d_exports.MediaFile>, commentsExtensibleXml?: string | null, commentsExtendedXml?: string | null, context?: ParseContext): document_d_exports.Comment[];
|
|
32
33
|
//#endregion
|
|
33
34
|
export { CommentExtendedInfo, parseComments, parseCommentsExtended };
|
|
@@ -1,8 +1,29 @@
|
|
|
1
1
|
import { parseParagraph } from "./paragraphParser.js";
|
|
2
2
|
import { cloneParagraphWithPropertySource } from "./paragraphPropertySource.js";
|
|
3
3
|
import { parseRunProperties } from "./runParser.js";
|
|
4
|
-
import { findChild, getAttribute, getChildElements, getLocalName, parseXml } from "./xmlParser.js";
|
|
4
|
+
import { findChild, getAttribute, getChildElements, getLocalName, parseOnOffValue, parseXml } from "./xmlParser.js";
|
|
5
|
+
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
5
6
|
//#region src/docx/commentParser.ts
|
|
7
|
+
/**
|
|
8
|
+
* Comment Parser - Parse comments.xml, commentsExtensible.xml, and
|
|
9
|
+
* commentsExtended.xml.
|
|
10
|
+
*
|
|
11
|
+
* - `comments.xml` (w) carries the comment author, local date, and body.
|
|
12
|
+
* - `commentsExtensible.xml` (w16cex, Word 2016+) carries reliable UTC
|
|
13
|
+
* timestamps via `w16cex:dateUtc` — Word's `w:date` is local time
|
|
14
|
+
* without an offset and so is ambiguous.
|
|
15
|
+
* - `commentsExtended.xml` (w15, Word 2013+) carries reply-thread
|
|
16
|
+
* parent links via `w15:paraIdParent` and the resolved/done state
|
|
17
|
+
* via `w15:done`. Cross-referenced via the `w14:paraId` on
|
|
18
|
+
* `w:comment` and the matching `w15:paraId` on `w15:commentEx`.
|
|
19
|
+
*
|
|
20
|
+
* OOXML Reference:
|
|
21
|
+
* - Comments: w:comments
|
|
22
|
+
* - Comment: w:comment (w:id, w:author, w:date, w:initials, w14:paraId)
|
|
23
|
+
* - Comment content: child w:p elements
|
|
24
|
+
*/
|
|
25
|
+
/** The whole lexical form of `ST_DecimalNumber`: an optional sign and digits. */
|
|
26
|
+
const DECIMAL_NUMBER = /^[+-]?\d+$/u;
|
|
6
27
|
const DEFAULT_ANNOTATION_REFERENCE_STYLE_ID = "CommentReference";
|
|
7
28
|
const normalizeAnnotationReferenceFormatting = (formatting) => {
|
|
8
29
|
if (formatting?.styleId === DEFAULT_ANNOTATION_REFERENCE_STYLE_ID && Object.keys(formatting).length === 1) return;
|
|
@@ -65,10 +86,7 @@ function parseCommentsExtended(xml) {
|
|
|
65
86
|
const doneAttr = getAttribute(child, "w15", "done") ?? child.attributes?.["w15:done"];
|
|
66
87
|
const info = {};
|
|
67
88
|
if (parentParaId) info.parentParaId = String(parentParaId).toUpperCase();
|
|
68
|
-
if (doneAttr !== void 0)
|
|
69
|
-
const v = String(doneAttr).toLowerCase();
|
|
70
|
-
info.done = v === "1" || v === "true";
|
|
71
|
-
}
|
|
89
|
+
if (doneAttr !== void 0) info.done = parseOnOffValue(String(doneAttr).toLowerCase()) ?? false;
|
|
72
90
|
infoByParaId.set(String(paraId).toUpperCase(), info);
|
|
73
91
|
}
|
|
74
92
|
return infoByParaId;
|
|
@@ -81,7 +99,7 @@ function parseCommentsExtended(xml) {
|
|
|
81
99
|
* local time. If `commentsExtendedXml` is provided, reply-thread
|
|
82
100
|
* parent links (`parentId`) and resolved state (`done`) are populated.
|
|
83
101
|
*/
|
|
84
|
-
function parseComments(commentsXml, styles, theme, rels, media, commentsExtensibleXml, commentsExtendedXml) {
|
|
102
|
+
function parseComments(commentsXml, styles, theme, rels, media, commentsExtensibleXml, commentsExtendedXml, context) {
|
|
85
103
|
if (!commentsXml) return [];
|
|
86
104
|
const root = parseXml(commentsXml);
|
|
87
105
|
const dateUtcByParaId = commentsExtensibleXml ? parseCommentsExtensible(commentsExtensibleXml) : /* @__PURE__ */ new Map();
|
|
@@ -92,7 +110,16 @@ function parseComments(commentsXml, styles, theme, rels, media, commentsExtensib
|
|
|
92
110
|
const paraIdByCommentIndex = /* @__PURE__ */ new Map();
|
|
93
111
|
for (const child of children) {
|
|
94
112
|
if ((child.name?.replace(/^.*:/u, "") ?? "") !== "comment") continue;
|
|
95
|
-
const
|
|
113
|
+
const rawId = getAttribute(child, "w", "id");
|
|
114
|
+
const id = rawId !== null && DECIMAL_NUMBER.test(rawId) ? Number.parseInt(rawId, 10) : NaN;
|
|
115
|
+
if (Number.isNaN(id)) {
|
|
116
|
+
context?.warn({
|
|
117
|
+
code: PARSE_WARNING_CODES.missingCommentId,
|
|
118
|
+
element: "w:comment",
|
|
119
|
+
...rawId === null ? {} : { value: rawId }
|
|
120
|
+
});
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
96
123
|
const rawAuthor = getAttribute(child, "w", "author");
|
|
97
124
|
const author = parseCommentAuthor(rawAuthor);
|
|
98
125
|
const rawInitials = getAttribute(child, "w", "initials");
|
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
import { document_d_exports } from "../types/document.js";
|
|
2
2
|
//#region src/docx/commentReferenceNormalization.d.ts
|
|
3
|
+
/** The codes this normalisation is reported under, owned here, not at the caller. */
|
|
4
|
+
declare const DANGLING_COMMENT_REFERENCE_WARNING: "dangling-comment-reference";
|
|
5
|
+
declare const UNBALANCED_COMMENT_RANGE_WARNING: "unbalanced-comment-range";
|
|
3
6
|
type NormalizeCommentReferencesInput = {
|
|
4
7
|
documentBody: document_d_exports.DocumentBody;
|
|
5
8
|
comments: readonly document_d_exports.Comment[];
|
|
@@ -14,4 +17,4 @@ type NormalizeCommentReferencesResult = {
|
|
|
14
17
|
};
|
|
15
18
|
declare const normalizeCommentReferences: ({ documentBody, comments, headers, footers, footnotes, endnotes }: NormalizeCommentReferencesInput) => NormalizeCommentReferencesResult;
|
|
16
19
|
//#endregion
|
|
17
|
-
export { normalizeCommentReferences };
|
|
20
|
+
export { DANGLING_COMMENT_REFERENCE_WARNING, UNBALANCED_COMMENT_RANGE_WARNING, normalizeCommentReferences };
|