@stll/folio-core 0.43.0 → 0.45.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ai-edits/__fixtures__/paragraphs.js +2 -2
- package/dist/ai-edits/headless.js +7 -5
- package/dist/ai-edits/index.d.ts +2 -2
- package/dist/ai-edits/index.js +2 -2
- package/dist/ai-edits/snapshot.js +13 -9
- package/dist/compare/content-alignment.js +94 -54
- package/dist/compare/inline-atoms.js +34 -20
- package/dist/compare/style-resources.js +6 -0
- package/dist/content-controls/mutateContentControls.js +4 -2
- package/dist/display-list/dom/renderDisplayListToDom.js +8 -8
- package/dist/document-operations.js +14 -3
- package/dist/docx/appVersionNormalization.d.ts +0 -18
- package/dist/docx/blockContentParser.js +8 -0
- package/dist/docx/blockRangeMarkers.d.ts +36 -0
- package/dist/docx/blockRangeMarkers.js +59 -0
- package/dist/docx/bookmarkParser.d.ts +2 -20
- package/dist/docx/bookmarkParser.js +6 -30
- package/dist/docx/borderParser.d.ts +13 -0
- package/dist/docx/borderParser.js +71 -0
- package/dist/docx/builtInStyles.d.ts +165 -0
- package/dist/docx/builtInStyles.js +239 -0
- package/dist/docx/commentIdNormalization.d.ts +3 -1
- package/dist/docx/commentIdNormalization.js +18 -1
- package/dist/docx/commentParser.d.ts +2 -1
- package/dist/docx/commentParser.js +80 -42
- package/dist/docx/commentReferenceNormalization.d.ts +4 -1
- package/dist/docx/commentReferenceNormalization.js +23 -14
- package/dist/docx/commentThreadKey.d.ts +18 -0
- package/dist/docx/commentThreadKey.js +22 -0
- package/dist/docx/danglingRelationshipReferences.d.ts +15 -0
- package/dist/docx/danglingRelationshipReferences.js +30 -0
- package/dist/docx/defaultParagraphStyle.d.ts +18 -1
- package/dist/docx/defaultParagraphStyle.js +23 -1
- package/dist/docx/diagramPreview.js +87 -27
- package/dist/docx/documentParser.d.ts +2 -1
- package/dist/docx/documentParser.js +2 -2
- package/dist/docx/drawingUtils.d.ts +8 -1
- package/dist/docx/drawingUtils.js +12 -3
- package/dist/docx/fieldParser.js +3 -5
- package/dist/docx/footnoteParser.d.ts +3 -2
- package/dist/docx/footnoteParser.js +19 -4
- package/dist/docx/groupDrawingParser.js +4 -4
- package/dist/docx/headerFooterRefParser.d.ts +4 -3
- package/dist/docx/headerFooterRefParser.js +42 -12
- package/dist/docx/headerFooterReferenceNormalization.d.ts +4 -1
- package/dist/docx/headerFooterReferenceNormalization.js +5 -1
- package/dist/docx/hyperlinkParser.js +13 -17
- package/dist/docx/imageParser.d.ts +10 -2
- package/dist/docx/imageParser.js +80 -30
- package/dist/docx/imageRawXml.d.ts +14 -1
- package/dist/docx/imageRawXml.js +35 -11
- package/dist/docx/markupRangeMarker.d.ts +15 -0
- package/dist/docx/markupRangeMarker.js +44 -0
- package/dist/docx/mathToMathml.js +12 -14
- package/dist/docx/nonVisualDrawingProps.d.ts +34 -0
- package/dist/docx/nonVisualDrawingProps.js +46 -0
- package/dist/docx/noteReferenceStyles.d.ts +29 -0
- package/dist/docx/noteReferenceStyles.js +70 -0
- package/dist/docx/numberingReferenceNormalization.d.ts +4 -1
- package/dist/docx/numberingReferenceNormalization.js +20 -1
- package/dist/docx/paraIdRangeNormalization.d.ts +0 -19
- package/dist/docx/paragraphParser.js +66 -99
- package/dist/docx/paragraphPropertySource.js +1 -0
- package/dist/docx/paragraphTextBoxEnrichment.js +3 -0
- package/dist/docx/paragraphTraversal.d.ts +37 -1
- package/dist/docx/paragraphTraversal.js +84 -1
- package/dist/docx/parseContext.d.ts +37 -0
- package/dist/docx/parseContext.js +67 -0
- package/dist/docx/parseWarningMessage.d.ts +6 -0
- package/dist/docx/parseWarningMessage.js +44 -0
- package/dist/docx/parser.js +83 -29
- package/dist/docx/previewBudget.d.ts +64 -0
- package/dist/docx/previewBudget.js +88 -0
- package/dist/docx/relsParser.d.ts +28 -11
- package/dist/docx/relsParser.js +26 -13
- package/dist/docx/revisionIdNormalization.js +96 -10
- package/dist/docx/rezip.js +80 -40
- package/dist/docx/runConsolidator.js +1 -2
- package/dist/docx/runParser.d.ts +8 -1
- package/dist/docx/runParser.js +30 -48
- package/dist/docx/sdtPropertiesPatch.js +24 -18
- package/dist/docx/sectionParser.d.ts +2 -1
- package/dist/docx/sectionParser.js +21 -65
- package/dist/docx/sectionReferenceHistory.js +2 -2
- package/dist/docx/selectiveSave.js +6 -6
- package/dist/docx/serializer/blockSdtSerializer.js +38 -26
- package/dist/docx/serializer/borderSerializer.d.ts +2 -3
- package/dist/docx/serializer/borderSerializer.js +13 -12
- package/dist/docx/serializer/commentSerializer.d.ts +41 -16
- package/dist/docx/serializer/commentSerializer.js +82 -72
- package/dist/docx/serializer/documentSerializer.d.ts +1 -5
- package/dist/docx/serializer/documentSerializer.js +6 -16
- package/dist/docx/serializer/fontTableSerializer.js +6 -6
- package/dist/docx/serializer/headerFooterSerializer.js +10 -5
- package/dist/docx/serializer/markupRangeAttributes.d.ts +8 -0
- package/dist/docx/serializer/markupRangeAttributes.js +24 -0
- package/dist/docx/serializer/noteSerializer.js +5 -0
- package/dist/docx/serializer/numberingSerializer.js +7 -6
- package/dist/docx/serializer/paragraphSerializer.d.ts +1 -5
- package/dist/docx/serializer/paragraphSerializer.js +47 -52
- package/dist/docx/serializer/partNamespaces.js +2 -2
- package/dist/docx/serializer/runSerializer.js +57 -31
- package/dist/docx/serializer/sectionPropertiesSerializer.js +11 -10
- package/dist/docx/serializer/settingsSerializer.js +4 -3
- package/dist/docx/serializer/stylesSerializer.js +6 -6
- package/dist/docx/serializer/tableSerializer.js +37 -21
- package/dist/docx/serializer/textFormattingSerializer.d.ts +2 -3
- package/dist/docx/serializer/textFormattingSerializer.js +29 -28
- package/dist/docx/serializer/themeSerializer.js +6 -6
- package/dist/docx/serializer/trackedChangeAttributes.js +2 -2
- package/dist/docx/serializer/xmlUtils.d.ts +1 -2
- package/dist/docx/serializer/xmlUtils.js +1 -13
- package/dist/docx/server/boundedArchive.d.ts +12 -0
- package/dist/docx/server/boundedArchive.js +20 -1
- package/dist/docx/server/build.js +8 -1
- package/dist/docx/server/createBilingualDocument.js +10 -18
- package/dist/docx/server/extractDocxText.js +3 -4
- package/dist/docx/server/validateDocxConformance.js +22 -1
- package/dist/docx/shadingParser.d.ts +6 -0
- package/dist/docx/shadingParser.js +32 -0
- package/dist/docx/shapeParser.js +10 -8
- package/dist/docx/styleParser.js +13 -87
- package/dist/docx/styleReferenceResolution.d.ts +36 -0
- package/dist/docx/styleReferenceResolution.js +51 -0
- package/dist/docx/tableLook.d.ts +57 -0
- package/dist/docx/tableLook.js +63 -0
- package/dist/docx/tableParser.d.ts +7 -9
- package/dist/docx/tableParser.js +64 -110
- package/dist/docx/textBoxParser.js +11 -6
- package/dist/docx/trackedMoveRangeNormalization.d.ts +3 -1
- package/dist/docx/trackedMoveRangeNormalization.js +11 -21
- package/dist/docx/transitionalSpelling.d.ts +13 -2
- package/dist/docx/transitionalSpelling.js +23 -1
- package/dist/docx/unzip.d.ts +23 -0
- package/dist/docx/unzip.js +32 -22
- package/dist/docx/verbatimCapture.js +5 -12
- package/dist/docx/vmlImageParser.js +5 -4
- package/dist/docx/vmlPreview.d.ts +1 -3
- package/dist/docx/vmlPreview.js +2 -30
- package/dist/docx/watermarkParser.js +2 -2
- package/dist/docx/xmlParser.d.ts +38 -33
- package/dist/docx/xmlParser.js +92 -47
- package/dist/docx/xmlResourceLimits.d.ts +89 -9
- package/dist/docx/xmlResourceLimits.js +105 -24
- package/dist/internal/pageBreakRunSourceDescendantIndex.js +2 -1
- package/dist/internal/paragraphFormattingSerialization.d.ts +2 -3
- package/dist/internal/paragraphFormattingSerialization.js +29 -8
- package/dist/layout-bridge/convert/footnoteLayout.js +2 -7
- package/dist/layout-engine/index.d.ts +2 -2
- package/dist/layout-engine/index.js +2 -2
- package/dist/layout-engine/measure/measureBlocks.js +1 -6
- package/dist/layout-engine/types.d.ts +8 -2
- package/dist/layout-engine/types.js +35 -2
- package/dist/layout-painter/renderImage.js +4 -3
- package/dist/layout-painter/renderParagraph.js +4 -3
- package/dist/managers/autoSaveCodec.js +2 -8
- package/dist/markdown/images.js +1 -4
- package/dist/markdown/index.js +1 -1
- package/dist/markdown/internals.d.ts +6 -1
- package/dist/markdown/internals.js +14 -1
- package/dist/markdown/renderBlock.js +35 -21
- package/dist/markdown/renderParagraph.js +14 -5
- package/dist/markdown/renderRuns.js +4 -3
- package/dist/markdown/renderTable.js +4 -3
- package/dist/markdown/trailers.js +41 -7
- package/dist/markdown/types.d.ts +3 -7
- package/dist/prosemirror/attrs/index.js +71 -5
- package/dist/prosemirror/bookmarkBoundaryAttrs.d.ts +11 -1
- package/dist/prosemirror/bookmarkBoundaryAttrs.js +18 -3
- package/dist/prosemirror/commands/image.js +1 -0
- package/dist/prosemirror/commands/index.d.ts +3 -3
- package/dist/prosemirror/commands/index.js +2 -2
- package/dist/prosemirror/commands/paragraph.d.ts +3 -3
- package/dist/prosemirror/commands/paragraph.js +2 -2
- package/dist/prosemirror/commentIdAllocator.js +2 -7
- package/dist/prosemirror/conversion/fromProseDoc.js +197 -68
- package/dist/prosemirror/conversion/toProseDoc.d.ts +1 -14
- package/dist/prosemirror/conversion/toProseDoc.js +458 -335
- package/dist/prosemirror/extensions/core/ParagraphExtension.d.ts +14 -1
- package/dist/prosemirror/extensions/core/ParagraphExtension.js +11 -6
- package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +3 -3
- package/dist/prosemirror/extensions/features/PasteCleanupExtension.d.ts +4 -1
- package/dist/prosemirror/extensions/features/PasteCleanupExtension.js +6 -2
- package/dist/prosemirror/extensions/features/pastedHeadingStyles.d.ts +7 -0
- package/dist/prosemirror/extensions/features/pastedHeadingStyles.js +74 -0
- package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +2 -3
- package/dist/prosemirror/extensions/marks/markUtils.d.ts +11 -3
- package/dist/prosemirror/extensions/marks/markUtils.js +98 -19
- package/dist/prosemirror/extensions/nodes/BookmarkBoundaryExtension.js +7 -3
- package/dist/prosemirror/extensions/nodes/ImageExtension.js +6 -1
- package/dist/prosemirror/extensions/nodes/ShapeExtension.js +8 -2
- package/dist/prosemirror/extensions/nodes/TableExtension.js +15 -1
- package/dist/prosemirror/extensions/nodes/TextBoxExtension.js +8 -4
- package/dist/prosemirror/extensions/types.d.ts +2 -2
- package/dist/prosemirror/index.d.ts +3 -3
- package/dist/prosemirror/index.js +3 -3
- package/dist/prosemirror/insertOperations.d.ts +9 -2
- package/dist/prosemirror/insertOperations.js +9 -4
- package/dist/prosemirror/paragraphFormattingProvenance.d.ts +162 -0
- package/dist/prosemirror/paragraphFormattingProvenance.js +115 -0
- package/dist/prosemirror/plugins/documentStyles.d.ts +9 -1
- package/dist/prosemirror/plugins/documentStyles.js +11 -1
- package/dist/prosemirror/plugins/index.d.ts +2 -2
- package/dist/prosemirror/plugins/index.js +2 -2
- package/dist/prosemirror/plugins/revisionIds.d.ts +11 -2
- package/dist/prosemirror/plugins/revisionIds.js +21 -6
- package/dist/prosemirror/runFormattingReconciliation.js +3 -2
- package/dist/prosemirror/runStyleFormatting.d.ts +1 -1
- package/dist/prosemirror/schema/nodes.d.ts +81 -1
- package/dist/prosemirror/styles/resolvedStyleAttrs.js +2 -0
- package/dist/prosemirror/styles/styleResolver.d.ts +9 -0
- package/dist/prosemirror/styles/styleResolver.js +12 -0
- package/dist/style-engine/styleEngine.d.ts +3 -0
- package/dist/style-engine/styleEngine.js +3 -0
- package/dist/style-sets/extract.js +1 -23
- package/dist/style-sets/stellaStyle.js +46 -39
- package/dist/style-sets/styleSetNormalization.d.ts +19 -0
- package/dist/style-sets/styleSetNormalization.js +99 -0
- package/dist/types/content.d.ts +2 -2
- package/dist/utils/base64.d.ts +36 -0
- package/dist/utils/base64.js +40 -0
- package/dist/utils/clipboard.js +2 -1
- package/dist/utils/createDocument.js +145 -20
- package/dist/utils/headingCollector.d.ts +8 -5
- package/dist/utils/headingCollector.js +23 -25
- package/dist/utils/tableOfContentsStyle.js +9 -2
- package/dist/utils/units.d.ts +10 -1
- package/dist/utils/units.js +12 -1
- package/dist/utils/urlSecurity.d.ts +8 -2
- package/dist/utils/urlSecurity.js +21 -3
- package/package.json +2 -2
- package/dist/docx/textWhitespace.d.ts +0 -4
- package/dist/docx/textWhitespace.js +0 -4
- package/dist/layout-bridge/engine/tableWidthUtils.d.ts +0 -6
- package/dist/layout-bridge/engine/tableWidthUtils.js +0 -25
- package/dist/markdown/headings.d.ts +0 -13
- package/dist/markdown/headings.js +0 -20
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { MAX_RETAINED_PARSE_WARNINGS_PER_CODE } from "@stll/docx-core/model";
|
|
2
|
+
//#region src/docx/parseContext.ts
|
|
3
|
+
/**
|
|
4
|
+
* The channel a parser reports a normalisation through.
|
|
5
|
+
*
|
|
6
|
+
* Folio accepts input Word accepts, which means normalising at the parse
|
|
7
|
+
* boundary rather than refusing. Every such decision has to be visible, and
|
|
8
|
+
* before this the only place that could say so was `parseDocx` itself: the
|
|
9
|
+
* leaf readers are pure functions with no way to report, so a value outside
|
|
10
|
+
* `ST_OnOff` or a `w:type` outside `ST_HdrFtr` was normalised in silence.
|
|
11
|
+
*
|
|
12
|
+
* The context is an explicit parameter, never a module-level accumulator and
|
|
13
|
+
* never async-local storage. Parsers stay re-entrant, a test can hand one in
|
|
14
|
+
* and read what a single reader reported, and two documents parsed at once
|
|
15
|
+
* cannot write into each other's list. The cost is a threaded argument, which
|
|
16
|
+
* is also the thing that makes the reporting greppable.
|
|
17
|
+
*/
|
|
18
|
+
const PACKAGE_PART = "package";
|
|
19
|
+
const createParseWarningCollector = (part = PACKAGE_PART) => {
|
|
20
|
+
const retained = [];
|
|
21
|
+
const retainedByCode = /* @__PURE__ */ new Map();
|
|
22
|
+
const suppressedByCode = /* @__PURE__ */ new Map();
|
|
23
|
+
const record = (location, report) => {
|
|
24
|
+
const count = report.count ?? 1;
|
|
25
|
+
const kept = retainedByCode.get(report.code) ?? 0;
|
|
26
|
+
if (kept >= MAX_RETAINED_PARSE_WARNINGS_PER_CODE) {
|
|
27
|
+
suppressedByCode.set(report.code, (suppressedByCode.get(report.code) ?? 0) + count);
|
|
28
|
+
return;
|
|
29
|
+
}
|
|
30
|
+
retainedByCode.set(report.code, kept + 1);
|
|
31
|
+
retained.push({
|
|
32
|
+
code: report.code,
|
|
33
|
+
location: {
|
|
34
|
+
...location,
|
|
35
|
+
...report.element === void 0 ? {} : { element: report.element },
|
|
36
|
+
...report.at === void 0 ? {} : { at: report.at }
|
|
37
|
+
},
|
|
38
|
+
...report.value === void 0 ? {} : { value: report.value },
|
|
39
|
+
...report.detail === void 0 ? {} : { detail: report.detail },
|
|
40
|
+
count
|
|
41
|
+
});
|
|
42
|
+
};
|
|
43
|
+
const contextAt = (location) => ({
|
|
44
|
+
warn: (report) => {
|
|
45
|
+
record(location, report);
|
|
46
|
+
},
|
|
47
|
+
scoped: (next) => {
|
|
48
|
+
const at = next.at ?? location.at;
|
|
49
|
+
return contextAt({
|
|
50
|
+
part: next.part ?? location.part,
|
|
51
|
+
...location.element === void 0 ? {} : { element: location.element },
|
|
52
|
+
...at === void 0 ? {} : { at }
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
});
|
|
56
|
+
return {
|
|
57
|
+
context: contextAt({ part }),
|
|
58
|
+
warnings: () => [...retained, ...[...suppressedByCode.entries()].sort(([left], [right]) => left < right ? -1 : 1).map(([code, count]) => ({
|
|
59
|
+
code,
|
|
60
|
+
location: { part },
|
|
61
|
+
count,
|
|
62
|
+
detail: `${String(count)} further occurrence(s) were counted but not retained.`
|
|
63
|
+
}))]
|
|
64
|
+
};
|
|
65
|
+
};
|
|
66
|
+
//#endregion
|
|
67
|
+
export { createParseWarningCollector };
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { ParseWarning } from "@stll/docx-core/model";
|
|
2
|
+
//#region src/docx/parseWarningMessage.d.ts
|
|
3
|
+
declare const formatParseWarning: (warning: ParseWarning) => string;
|
|
4
|
+
declare const formatParseWarnings: (warnings: readonly ParseWarning[]) => string[];
|
|
5
|
+
//#endregion
|
|
6
|
+
export { formatParseWarning, formatParseWarnings };
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
2
|
+
//#region src/docx/parseWarningMessage.ts
|
|
3
|
+
/**
|
|
4
|
+
* The one place a parse warning becomes a sentence.
|
|
5
|
+
*
|
|
6
|
+
* `Document.warnings` is the string list hosts have always read, and it is now
|
|
7
|
+
* rendered from `Document.parseWarnings` rather than written at the call site,
|
|
8
|
+
* so a warning cannot say one thing in the structured list and another in the
|
|
9
|
+
* prose. The map is total over the code union: a new code does not compile
|
|
10
|
+
* until it has a message.
|
|
11
|
+
*/
|
|
12
|
+
const plural = (count, singular, pluralForm = `${singular}s`) => `${String(count)} ${count === 1 ? singular : pluralForm}`;
|
|
13
|
+
/** The location suffix, omitted when the part is all we know. */
|
|
14
|
+
const where = ({ location }) => {
|
|
15
|
+
if (location.at === void 0) return "";
|
|
16
|
+
return ` at ${location.at} in ${location.part}`;
|
|
17
|
+
};
|
|
18
|
+
const quoted = (value) => value === void 0 ? "" : ` "${value}"`;
|
|
19
|
+
const PARSE_WARNING_MESSAGES = {
|
|
20
|
+
[PARSE_WARNING_CODES.packageDecrypted]: () => "Document was opened from password-protected storage; saving writes an unencrypted .docx file.",
|
|
21
|
+
[PARSE_WARNING_CODES.packageArchive]: (warning) => warning.detail ?? "The archive reader reported an issue.",
|
|
22
|
+
[PARSE_WARNING_CODES.documentPartMissing]: () => "No document.xml found in DOCX",
|
|
23
|
+
[PARSE_WARNING_CODES.documentModelIssue]: (warning) => warning.detail ?? "The document model reported an issue.",
|
|
24
|
+
[PARSE_WARNING_CODES.duplicateCommentId]: (warning) => `Dropped ${plural(warning.count, "comment")} repeating a w:id another comment already defines.`,
|
|
25
|
+
[PARSE_WARNING_CODES.missingCommentId]: (warning) => `Dropped ${plural(warning.count, "comment")} with no readable w:id.`,
|
|
26
|
+
[PARSE_WARNING_CODES.duplicateNoteId]: (warning) => `Dropped ${plural(warning.count, "note")} repeating a w:id another note already defines.`,
|
|
27
|
+
[PARSE_WARNING_CODES.danglingCommentReference]: (warning) => `Removed ${plural(warning.count, "dangling comment reference marker")} whose comments.xml entries are missing.`,
|
|
28
|
+
[PARSE_WARNING_CODES.unbalancedCommentRange]: (warning) => `Re-anchored ${plural(warning.count, "unbalanced comment range marker")} as point comments.`,
|
|
29
|
+
[PARSE_WARNING_CODES.danglingHeaderReference]: (warning) => `Removed ${plural(warning.count, "dangling header reference")} whose header parts are missing.`,
|
|
30
|
+
[PARSE_WARNING_CODES.danglingFooterReference]: (warning) => `Removed ${plural(warning.count, "dangling footer reference")} whose footer parts are missing.`,
|
|
31
|
+
[PARSE_WARNING_CODES.danglingRelationshipId]: (warning) => `Left relationship id${quoted(warning.value)} unresolved${where(warning)}; the part defines no such relationship.`,
|
|
32
|
+
[PARSE_WARNING_CODES.unnumberedParagraph]: (warning) => `Unnumbered ${plural(warning.count, "paragraph")} whose numbering definitions are missing.`,
|
|
33
|
+
[PARSE_WARNING_CODES.unnumberedStyle]: (warning) => `Unnumbered style${quoted(warning.value)} whose numbering definition is missing.`,
|
|
34
|
+
[PARSE_WARNING_CODES.unbalancedMoveRange]: (warning) => `Removed ${plural(warning.count, "unbalanced tracked move range marker")}.`,
|
|
35
|
+
[PARSE_WARNING_CODES.headerFooterTypeOutsideEnum]: (warning) => `Read header/footer type${quoted(warning.value)} as "default"${where(warning)}; ST_HdrFtr is even, default or first.`,
|
|
36
|
+
[PARSE_WARNING_CODES.unrecognisedOnOffValue]: (warning) => `Ignored on/off value${quoted(warning.value)}${where(warning)}; ST_OnOff is 1, 0, true, false, on or off.`,
|
|
37
|
+
[PARSE_WARNING_CODES.borderWithoutValue]: (warning) => `Read a border with no w:val${where(warning)} as having no border style.`,
|
|
38
|
+
[PARSE_WARNING_CODES.styleSetDuplicateStyleId]: (warning) => `Dropped a style repeating the id${quoted(warning.value)} another style in the set already defines.`,
|
|
39
|
+
[PARSE_WARNING_CODES.styleSetInitialStyleMissing]: (warning) => `The style set names initial paragraph style${quoted(warning.value)}, which it does not contain; used the set's default instead${where(warning)}.`
|
|
40
|
+
};
|
|
41
|
+
const formatParseWarning = (warning) => PARSE_WARNING_MESSAGES[warning.code](warning);
|
|
42
|
+
const formatParseWarnings = (warnings) => warnings.map(formatParseWarning);
|
|
43
|
+
//#endregion
|
|
44
|
+
export { formatParseWarning, formatParseWarnings };
|
package/dist/docx/parser.js
CHANGED
|
@@ -1,35 +1,39 @@
|
|
|
1
1
|
import { toArrayBuffer } from "../utils/docxInput.js";
|
|
2
2
|
import { loadFontsWithMapping } from "../utils/fontLoader.js";
|
|
3
3
|
import { MAX_PACKAGE_TIFF_PIXELS, convertTiffToPngDataUrl, isTiffMimeType } from "../utils/tiffConverter.js";
|
|
4
|
-
import { normalizeCommentIds } from "./commentIdNormalization.js";
|
|
4
|
+
import { DUPLICATE_COMMENT_ID_WARNING, normalizeCommentIds } from "./commentIdNormalization.js";
|
|
5
5
|
import { parseComments } from "./commentParser.js";
|
|
6
|
-
import { normalizeCommentReferences } from "./commentReferenceNormalization.js";
|
|
6
|
+
import { DANGLING_COMMENT_REFERENCE_WARNING, UNBALANCED_COMMENT_RANGE_WARNING, normalizeCommentReferences } from "./commentReferenceNormalization.js";
|
|
7
7
|
import { detectDocxConformanceClass } from "./conformance.js";
|
|
8
8
|
import { parseCoreProperties } from "./corePropertiesParser.js";
|
|
9
|
+
import { countDanglingRelationshipReferences } from "./danglingRelationshipReferences.js";
|
|
9
10
|
import { extractAllTemplateVariables, parseDocumentBody } from "./documentParser.js";
|
|
10
11
|
import { normalizeDrawingIds } from "./drawingIdNormalization.js";
|
|
11
12
|
import { DocxEncryptionError } from "./encryption/errors.js";
|
|
12
13
|
import { parseFontTable } from "./fontTableParser.js";
|
|
13
14
|
import { parseEndnotes, parseFootnotes } from "./footnoteParser.js";
|
|
14
15
|
import { parseFooter, parseHeader } from "./headerFooterParser.js";
|
|
15
|
-
import { normalizeHeaderFooterReferences } from "./headerFooterReferenceNormalization.js";
|
|
16
|
+
import { DANGLING_FOOTER_REFERENCE_WARNING, DANGLING_HEADER_REFERENCE_WARNING, normalizeHeaderFooterReferences } from "./headerFooterReferenceNormalization.js";
|
|
16
17
|
import { assignHeaderFooterVerbatimXml, refreshHeaderFooterVerbatimFingerprint } from "./headerFooterVerbatim.js";
|
|
17
18
|
import { extractMetafileRaster, isMetafileMimeType } from "./metafileRaster.js";
|
|
18
19
|
import { renderEmfSvg } from "./metafileSvg.js";
|
|
19
20
|
import { DocxModelValidationError, formatDocumentModelIssues, validateFolioDocumentModel } from "./modelValidation.js";
|
|
20
21
|
import { parseNumbering } from "./numberingParser.js";
|
|
21
|
-
import { normalizeNumberingReferences, normalizeStyleNumberingReferences } from "./numberingReferenceNormalization.js";
|
|
22
|
+
import { UNNUMBERED_PARAGRAPH_WARNING, UNNUMBERED_STYLE_WARNING, normalizeNumberingReferences, normalizeStyleNumberingReferences } from "./numberingReferenceNormalization.js";
|
|
22
23
|
import { assignDocumentParagraphPropertySourceContract } from "./paragraphPropertySource.js";
|
|
24
|
+
import { createParseWarningCollector } from "./parseContext.js";
|
|
25
|
+
import { formatParseWarnings } from "./parseWarningMessage.js";
|
|
26
|
+
import { enforcePackagePreviewBudget } from "./previewBudget.js";
|
|
23
27
|
import { RELATIONSHIP_TYPES, parseRelationships, resolveRelativePath } from "./relsParser.js";
|
|
24
28
|
import { normalizeRenderedPageBreakHints } from "./renderedPageBreakNormalization.js";
|
|
25
29
|
import { parseSettings } from "./settingsParser.js";
|
|
26
30
|
import { parseStylesPackage } from "./styleParser.js";
|
|
27
31
|
import { applyThemeFontLang, parseTheme } from "./themeParser.js";
|
|
28
|
-
import { normalizeTrackedMoveRanges } from "./trackedMoveRangeNormalization.js";
|
|
32
|
+
import { UNBALANCED_MOVE_RANGE_WARNING, normalizeTrackedMoveRanges } from "./trackedMoveRangeNormalization.js";
|
|
29
33
|
import { getMediaMimeType, mediaToDataUrl, unzipDocx } from "./unzip.js";
|
|
30
|
-
import { enforcePackageVmlPreviewBudget } from "./vmlPreview.js";
|
|
31
34
|
import { FOLIO_XML_RESOURCE_LIMITS } from "./xmlResourceLimits.js";
|
|
32
35
|
import { TaggedError } from "better-result";
|
|
36
|
+
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
33
37
|
//#region src/docx/parser.ts
|
|
34
38
|
/**
|
|
35
39
|
* Main Parser Orchestrator - Unified parseDocx function
|
|
@@ -70,7 +74,7 @@ const sha256Hex = async (buffer) => {
|
|
|
70
74
|
async function parseDocx(input, options = {}) {
|
|
71
75
|
const buffer = input instanceof ArrayBuffer ? input : await toArrayBuffer(input);
|
|
72
76
|
const { onProgress = () => {}, preloadFonts = true, parseHeadersFooters = true, parseNotes = true, detectVariables = true, password, unzipLimits, mediaResolver } = options;
|
|
73
|
-
const warnings =
|
|
77
|
+
const { context: parseContext, warnings: collectedWarnings } = createParseWarningCollector();
|
|
74
78
|
try {
|
|
75
79
|
const timeStage = (_name, fn) => fn();
|
|
76
80
|
const timeStageAsync = async (_name, fn) => await fn();
|
|
@@ -81,8 +85,11 @@ async function parseDocx(input, options = {}) {
|
|
|
81
85
|
password,
|
|
82
86
|
extractAllXml: false
|
|
83
87
|
}));
|
|
84
|
-
if (raw.wasEncrypted)
|
|
85
|
-
|
|
88
|
+
if (raw.wasEncrypted) parseContext.warn({ code: PARSE_WARNING_CODES.packageDecrypted });
|
|
89
|
+
for (const message of raw.warnings) parseContext.warn({
|
|
90
|
+
code: PARSE_WARNING_CODES.packageArchive,
|
|
91
|
+
detail: message
|
|
92
|
+
});
|
|
86
93
|
onProgress("Extracted DOCX", 10);
|
|
87
94
|
onProgress("Parsing relationships...", 10);
|
|
88
95
|
const rels = timeStage("relationships", () => raw.documentRels ? parseRelationships(raw.documentRels) : /* @__PURE__ */ new Map());
|
|
@@ -114,8 +121,8 @@ async function parseDocx(input, options = {}) {
|
|
|
114
121
|
onProgress("Parsing document body...", 40);
|
|
115
122
|
let documentBody = { content: [] };
|
|
116
123
|
timeStage("documentBody", () => {
|
|
117
|
-
if (raw.documentXml) documentBody = parseDocumentBody(raw.documentXml, styles, theme, numbering, rels, media);
|
|
118
|
-
else
|
|
124
|
+
if (raw.documentXml) documentBody = parseDocumentBody(raw.documentXml, styles, theme, numbering, rels, media, parseContext.scoped({ part: "word/document.xml" }));
|
|
125
|
+
else parseContext.warn({ code: PARSE_WARNING_CODES.documentPartMissing });
|
|
119
126
|
});
|
|
120
127
|
onProgress("Parsed document body", 55);
|
|
121
128
|
let headers;
|
|
@@ -131,15 +138,19 @@ async function parseDocx(input, options = {}) {
|
|
|
131
138
|
let endnotes;
|
|
132
139
|
if (parseNotes) {
|
|
133
140
|
onProgress("Parsing footnotes/endnotes...", 65);
|
|
134
|
-
const notes = timeStage("footnotesEndnotes", () => parseNotesContent(raw, styles, theme, numbering, rels, media));
|
|
141
|
+
const notes = timeStage("footnotesEndnotes", () => parseNotesContent(raw, styles, theme, numbering, rels, media, parseContext));
|
|
135
142
|
footnotes = notes.footnotes;
|
|
136
143
|
endnotes = notes.endnotes;
|
|
137
144
|
onProgress("Parsed footnotes/endnotes", 75);
|
|
138
145
|
} else onProgress("Skipping footnotes/endnotes", 75);
|
|
139
146
|
onProgress("Parsing comments...", 75);
|
|
140
|
-
const
|
|
147
|
+
const commentsContext = parseContext.scoped({ part: "word/comments.xml" });
|
|
148
|
+
const comments = timeStage("comments", () => parseComments(raw.commentsXml, styles, theme, rels, media, raw.commentsExtensibleXml, raw.commentsExtendedXml, commentsContext));
|
|
141
149
|
const commentIdNormalization = normalizeCommentIds(comments);
|
|
142
|
-
if (commentIdNormalization.droppedDuplicateComments > 0)
|
|
150
|
+
if (commentIdNormalization.droppedDuplicateComments > 0) commentsContext.warn({
|
|
151
|
+
code: DUPLICATE_COMMENT_ID_WARNING,
|
|
152
|
+
count: commentIdNormalization.droppedDuplicateComments
|
|
153
|
+
});
|
|
143
154
|
if (comments.length > 0) documentBody.comments = comments;
|
|
144
155
|
normalizeDrawingIds({
|
|
145
156
|
documentBody,
|
|
@@ -165,15 +176,27 @@ async function parseDocx(input, options = {}) {
|
|
|
165
176
|
...footnotes !== void 0 ? { footnotes } : {},
|
|
166
177
|
...endnotes !== void 0 ? { endnotes } : {}
|
|
167
178
|
});
|
|
168
|
-
if (commentReferenceNormalization.removedDanglingReferences > 0)
|
|
169
|
-
|
|
179
|
+
if (commentReferenceNormalization.removedDanglingReferences > 0) parseContext.warn({
|
|
180
|
+
code: DANGLING_COMMENT_REFERENCE_WARNING,
|
|
181
|
+
count: commentReferenceNormalization.removedDanglingReferences
|
|
182
|
+
});
|
|
183
|
+
if (commentReferenceNormalization.reanchoredUnbalancedRanges > 0) parseContext.warn({
|
|
184
|
+
code: UNBALANCED_COMMENT_RANGE_WARNING,
|
|
185
|
+
count: commentReferenceNormalization.reanchoredUnbalancedRanges
|
|
186
|
+
});
|
|
170
187
|
const headerFooterReferenceNormalization = normalizeHeaderFooterReferences({
|
|
171
188
|
documentBody,
|
|
172
189
|
...headers !== void 0 ? { headers } : {},
|
|
173
190
|
...footers !== void 0 ? { footers } : {}
|
|
174
191
|
});
|
|
175
|
-
if (headerFooterReferenceNormalization.removedDanglingHeaderReferences > 0)
|
|
176
|
-
|
|
192
|
+
if (headerFooterReferenceNormalization.removedDanglingHeaderReferences > 0) parseContext.warn({
|
|
193
|
+
code: DANGLING_HEADER_REFERENCE_WARNING,
|
|
194
|
+
count: headerFooterReferenceNormalization.removedDanglingHeaderReferences
|
|
195
|
+
});
|
|
196
|
+
if (headerFooterReferenceNormalization.removedDanglingFooterReferences > 0) parseContext.warn({
|
|
197
|
+
code: DANGLING_FOOTER_REFERENCE_WARNING,
|
|
198
|
+
count: headerFooterReferenceNormalization.removedDanglingFooterReferences
|
|
199
|
+
});
|
|
177
200
|
const numberingReferenceNormalization = normalizeNumberingReferences({
|
|
178
201
|
documentBody,
|
|
179
202
|
numbering,
|
|
@@ -182,12 +205,33 @@ async function parseDocx(input, options = {}) {
|
|
|
182
205
|
...footnotes !== void 0 ? { footnotes } : {},
|
|
183
206
|
...endnotes !== void 0 ? { endnotes } : {}
|
|
184
207
|
});
|
|
185
|
-
if (numberingReferenceNormalization.unnumberedDanglingReferences > 0)
|
|
208
|
+
if (numberingReferenceNormalization.unnumberedDanglingReferences > 0) parseContext.warn({
|
|
209
|
+
code: UNNUMBERED_PARAGRAPH_WARNING,
|
|
210
|
+
count: numberingReferenceNormalization.unnumberedDanglingReferences
|
|
211
|
+
});
|
|
186
212
|
const styleNumberingNormalization = normalizeStyleNumberingReferences({
|
|
187
213
|
styles: styleDefinitions?.styles ?? [],
|
|
188
214
|
numbering
|
|
189
215
|
});
|
|
190
|
-
for (const styleId of styleNumberingNormalization.unnumberedStyleIds)
|
|
216
|
+
for (const styleId of styleNumberingNormalization.unnumberedStyleIds) parseContext.warn({
|
|
217
|
+
code: UNNUMBERED_STYLE_WARNING,
|
|
218
|
+
value: styleId,
|
|
219
|
+
at: `style "${styleId}"`
|
|
220
|
+
});
|
|
221
|
+
const danglingReferences = countDanglingRelationshipReferences({
|
|
222
|
+
content: documentBody.content,
|
|
223
|
+
relationships: rels
|
|
224
|
+
});
|
|
225
|
+
if (danglingReferences.drawings > 0) parseContext.warn({
|
|
226
|
+
code: PARSE_WARNING_CODES.danglingRelationshipId,
|
|
227
|
+
element: "w:drawing",
|
|
228
|
+
count: danglingReferences.drawings
|
|
229
|
+
});
|
|
230
|
+
if (danglingReferences.hyperlinks > 0) parseContext.warn({
|
|
231
|
+
code: PARSE_WARNING_CODES.danglingRelationshipId,
|
|
232
|
+
element: "w:hyperlink",
|
|
233
|
+
count: danglingReferences.hyperlinks
|
|
234
|
+
});
|
|
191
235
|
const trackedMoveRangeNormalization = normalizeTrackedMoveRanges({
|
|
192
236
|
documentBody,
|
|
193
237
|
...headers !== void 0 ? { headers } : {},
|
|
@@ -195,7 +239,10 @@ async function parseDocx(input, options = {}) {
|
|
|
195
239
|
...footnotes !== void 0 ? { footnotes } : {},
|
|
196
240
|
...endnotes !== void 0 ? { endnotes } : {}
|
|
197
241
|
});
|
|
198
|
-
if (trackedMoveRangeNormalization.removedUnbalancedMoveRangeMarkers > 0)
|
|
242
|
+
if (trackedMoveRangeNormalization.removedUnbalancedMoveRangeMarkers > 0) parseContext.warn({
|
|
243
|
+
code: UNBALANCED_MOVE_RANGE_WARNING,
|
|
244
|
+
count: trackedMoveRangeNormalization.removedUnbalancedMoveRangeMarkers
|
|
245
|
+
});
|
|
199
246
|
let templateVariables;
|
|
200
247
|
if (detectVariables) {
|
|
201
248
|
onProgress("Detecting template variables...", 75);
|
|
@@ -234,12 +281,19 @@ async function parseDocx(input, options = {}) {
|
|
|
234
281
|
...requiredFonts.length > 0 ? { requiredFonts } : {}
|
|
235
282
|
};
|
|
236
283
|
assignDocumentParagraphPropertySourceContract(document, await paragraphPropertySourceDigest);
|
|
237
|
-
|
|
284
|
+
enforcePackagePreviewBudget(document.package);
|
|
238
285
|
const validation = validateFolioDocumentModel(document);
|
|
239
286
|
const parsedCompleteModel = parseHeadersFooters && parseNotes;
|
|
240
287
|
if (!validation.valid && parsedCompleteModel) throw new DocxModelValidationError("Parsed DOCX produced an invalid document model", validation.issues);
|
|
241
|
-
|
|
242
|
-
|
|
288
|
+
for (const issue of formatDocumentModelIssues(validation.issues)) parseContext.warn({
|
|
289
|
+
code: PARSE_WARNING_CODES.documentModelIssue,
|
|
290
|
+
detail: issue
|
|
291
|
+
});
|
|
292
|
+
const parseWarnings = collectedWarnings();
|
|
293
|
+
if (parseWarnings.length > 0) {
|
|
294
|
+
document.parseWarnings = parseWarnings;
|
|
295
|
+
document.warnings = formatParseWarnings(parseWarnings);
|
|
296
|
+
}
|
|
243
297
|
onProgress("Complete", 100);
|
|
244
298
|
return document;
|
|
245
299
|
} catch (error) {
|
|
@@ -418,7 +472,7 @@ function parseHeadersAndFooters(raw, styles, theme, numbering, rels, media) {
|
|
|
418
472
|
if (headerXml) {
|
|
419
473
|
const headerRelsPath = getRelationshipsPathForPart(partPath);
|
|
420
474
|
const headerRelsXml = getMapCaseInsensitive(raw.allXml, headerRelsPath);
|
|
421
|
-
const headerRels = headerRelsXml ? parseRelationships(headerRelsXml) :
|
|
475
|
+
const headerRels = headerRelsXml ? parseRelationships(headerRelsXml) : /* @__PURE__ */ new Map();
|
|
422
476
|
const header = parseHeader(headerXml, "default", styles, theme, numbering, headerRels, media);
|
|
423
477
|
const watermark = header.watermark;
|
|
424
478
|
if (watermark?.kind === "picture") {
|
|
@@ -437,7 +491,7 @@ function parseHeadersAndFooters(raw, styles, theme, numbering, rels, media) {
|
|
|
437
491
|
if (footerXml) {
|
|
438
492
|
const footerRelsPath = getRelationshipsPathForPart(partPath);
|
|
439
493
|
const footerRelsXml = getMapCaseInsensitive(raw.allXml, footerRelsPath);
|
|
440
|
-
const footer = parseFooter(footerXml, "default", styles, theme, numbering, footerRelsXml ? parseRelationships(footerRelsXml) :
|
|
494
|
+
const footer = parseFooter(footerXml, "default", styles, theme, numbering, footerRelsXml ? parseRelationships(footerRelsXml) : /* @__PURE__ */ new Map(), media);
|
|
441
495
|
footers.set(rId, footer);
|
|
442
496
|
}
|
|
443
497
|
}
|
|
@@ -449,13 +503,13 @@ function parseHeadersAndFooters(raw, styles, theme, numbering, rels, media) {
|
|
|
449
503
|
/**
|
|
450
504
|
* Parse footnotes and endnotes from raw content
|
|
451
505
|
*/
|
|
452
|
-
function parseNotesContent(raw, styles, theme, numbering, rels, media) {
|
|
506
|
+
function parseNotesContent(raw, styles, theme, numbering, rels, media, context) {
|
|
453
507
|
const relsForNotePart = (partPath) => {
|
|
454
508
|
const xml = getMapCaseInsensitive(raw.allXml, getRelationshipsPathForPart(partPath));
|
|
455
509
|
return xml ? parseRelationships(xml) : rels;
|
|
456
510
|
};
|
|
457
|
-
const footnoteMap = parseFootnotes(raw.footnotesXml, styles, theme, numbering, relsForNotePart("word/footnotes.xml"), media);
|
|
458
|
-
const endnoteMap = parseEndnotes(raw.endnotesXml, styles, theme, numbering, relsForNotePart("word/endnotes.xml"), media);
|
|
511
|
+
const footnoteMap = parseFootnotes(raw.footnotesXml, styles, theme, numbering, relsForNotePart("word/footnotes.xml"), media, context?.scoped({ part: "word/footnotes.xml" }));
|
|
512
|
+
const endnoteMap = parseEndnotes(raw.endnotesXml, styles, theme, numbering, relsForNotePart("word/endnotes.xml"), media, context?.scoped({ part: "word/endnotes.xml" }));
|
|
459
513
|
return {
|
|
460
514
|
footnotes: footnoteMap.getNormalFootnotes(),
|
|
461
515
|
endnotes: endnoteMap.getNormalEndnotes()
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
//#region src/docx/previewBudget.d.ts
|
|
2
|
+
/**
|
|
3
|
+
* Every synthetic preview a parse attaches to the model, and the budget each
|
|
4
|
+
* one answers to.
|
|
5
|
+
*
|
|
6
|
+
* A preview is not document content. It is a drawing folio makes up so a shape
|
|
7
|
+
* it cannot project still occupies the page, and the package round-trips from
|
|
8
|
+
* its preserved XML whether the preview exists or not. So a preview is the one
|
|
9
|
+
* thing in the model that may be dropped, and a package that would retain more
|
|
10
|
+
* of it than it is worth has it dropped rather than being refused.
|
|
11
|
+
*
|
|
12
|
+
* Producer and budget have to agree on what a preview looks like. They used to
|
|
13
|
+
* agree by coincidence: the VML producer wrote its filename as a literal and
|
|
14
|
+
* the budget recognized it as a constant in another file, so renaming one
|
|
15
|
+
* would have stopped the other charging for it without anything failing. The
|
|
16
|
+
* table below is that agreement written once, and a producer builds its image
|
|
17
|
+
* out of the entry the budget matches against.
|
|
18
|
+
*/
|
|
19
|
+
declare const VML_PREVIEW_DATA_URL_PREFIX = "data:image/svg+xml;charset=utf-8,";
|
|
20
|
+
declare const PREVIEW_KINDS: {
|
|
21
|
+
/** A VML shape folio renders rather than projects (`v:shape`, `v:rect`, ...). */
|
|
22
|
+
readonly vmlShape: {
|
|
23
|
+
readonly mimeType: "image/svg+xml";
|
|
24
|
+
readonly filename: "vml-shape-preview.svg";
|
|
25
|
+
readonly srcPrefix: "data:image/svg+xml;charset=utf-8,";
|
|
26
|
+
readonly maxPackageCharacters: number;
|
|
27
|
+
};
|
|
28
|
+
/**
|
|
29
|
+
* A SmartArt diagram: its extent filled with one flat rectangle per shape.
|
|
30
|
+
*
|
|
31
|
+
* A raster rather than a vector because the display list decodes only base64
|
|
32
|
+
* PNG and JPEG, so a vector preview would be missing from every display-list
|
|
33
|
+
* backend (PDF among them) while still showing in the DOM. That is what
|
|
34
|
+
* makes this preview expensive: one of them is 7.3 MB of data URL whatever
|
|
35
|
+
* the package weighs, because its cost follows the extent the author chose
|
|
36
|
+
* rather than anything the drawing contains.
|
|
37
|
+
*
|
|
38
|
+
* Across the public corpus, the fifty packages that produce one retain a
|
|
39
|
+
* median of 7.3 MB and a maximum of 51.3 MB (ten previews, from a package
|
|
40
|
+
* under a megabyte). The cap is set above that maximum: it refuses no
|
|
41
|
+
* legitimate file in the corpus while bounding what had no bound at all, and
|
|
42
|
+
* it is a ceiling rather than a fix. The fix is for the preview to be a
|
|
43
|
+
* descriptor the renderer rasterizes, which needs the display-list contract
|
|
44
|
+
* to carry one.
|
|
45
|
+
*/
|
|
46
|
+
readonly smartArt: {
|
|
47
|
+
readonly mimeType: "image/png";
|
|
48
|
+
readonly filename: "smartart-preview.png";
|
|
49
|
+
readonly srcPrefix: "data:image/png;base64,";
|
|
50
|
+
readonly maxPackageCharacters: number;
|
|
51
|
+
};
|
|
52
|
+
};
|
|
53
|
+
type PreviewKindName = keyof typeof PREVIEW_KINDS;
|
|
54
|
+
/** Per-kind allowances for one package, defaulting to the table's caps. */
|
|
55
|
+
type PreviewBudgetOverrides = Partial<Record<PreviewKindName, number>>;
|
|
56
|
+
/**
|
|
57
|
+
* Charge every generated preview in the model against its kind's allowance and
|
|
58
|
+
* drop the `src` of those past it. Dropping leaves the image in place with its
|
|
59
|
+
* size and wrap, so the page still reserves the space the drawing occupies,
|
|
60
|
+
* and never touches the preserved XML the package saves from.
|
|
61
|
+
*/
|
|
62
|
+
declare const enforcePackagePreviewBudget: (root: unknown, overrides?: PreviewBudgetOverrides) => void;
|
|
63
|
+
//#endregion
|
|
64
|
+
export { PREVIEW_KINDS, PreviewBudgetOverrides, VML_PREVIEW_DATA_URL_PREFIX, enforcePackagePreviewBudget };
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
//#region src/docx/previewBudget.ts
|
|
2
|
+
const VML_PREVIEW_DATA_URL_PREFIX = "data:image/svg+xml;charset=utf-8,";
|
|
3
|
+
const MEBIBYTE = 1024 * 1024;
|
|
4
|
+
const PREVIEW_KINDS = {
|
|
5
|
+
/** A VML shape folio renders rather than projects (`v:shape`, `v:rect`, ...). */
|
|
6
|
+
vmlShape: {
|
|
7
|
+
mimeType: "image/svg+xml",
|
|
8
|
+
filename: "vml-shape-preview.svg",
|
|
9
|
+
srcPrefix: VML_PREVIEW_DATA_URL_PREFIX,
|
|
10
|
+
maxPackageCharacters: 8 * MEBIBYTE
|
|
11
|
+
},
|
|
12
|
+
/**
|
|
13
|
+
* A SmartArt diagram: its extent filled with one flat rectangle per shape.
|
|
14
|
+
*
|
|
15
|
+
* A raster rather than a vector because the display list decodes only base64
|
|
16
|
+
* PNG and JPEG, so a vector preview would be missing from every display-list
|
|
17
|
+
* backend (PDF among them) while still showing in the DOM. That is what
|
|
18
|
+
* makes this preview expensive: one of them is 7.3 MB of data URL whatever
|
|
19
|
+
* the package weighs, because its cost follows the extent the author chose
|
|
20
|
+
* rather than anything the drawing contains.
|
|
21
|
+
*
|
|
22
|
+
* Across the public corpus, the fifty packages that produce one retain a
|
|
23
|
+
* median of 7.3 MB and a maximum of 51.3 MB (ten previews, from a package
|
|
24
|
+
* under a megabyte). The cap is set above that maximum: it refuses no
|
|
25
|
+
* legitimate file in the corpus while bounding what had no bound at all, and
|
|
26
|
+
* it is a ceiling rather than a fix. The fix is for the preview to be a
|
|
27
|
+
* descriptor the renderer rasterizes, which needs the display-list contract
|
|
28
|
+
* to carry one.
|
|
29
|
+
*/
|
|
30
|
+
smartArt: {
|
|
31
|
+
mimeType: "image/png",
|
|
32
|
+
filename: "smartart-preview.png",
|
|
33
|
+
srcPrefix: "data:image/png;base64,",
|
|
34
|
+
maxPackageCharacters: 64 * MEBIBYTE
|
|
35
|
+
}
|
|
36
|
+
};
|
|
37
|
+
const KIND_NAMES = Object.keys(PREVIEW_KINDS);
|
|
38
|
+
/**
|
|
39
|
+
* The kind a model image was generated as, or `undefined` for one the package
|
|
40
|
+
* actually carries. A generated preview has no relationship behind it, so
|
|
41
|
+
* `rId` is empty; the filename and data-URL prefix name which producer made it.
|
|
42
|
+
*/
|
|
43
|
+
const previewKindOf = (value) => {
|
|
44
|
+
if (!("type" in value) || value.type !== "image" || !("rId" in value) || value.rId !== "" || !("src" in value) || typeof value.src !== "string" || !("mimeType" in value) || !("filename" in value)) return;
|
|
45
|
+
const { src, mimeType, filename } = value;
|
|
46
|
+
return KIND_NAMES.find((name) => {
|
|
47
|
+
const kind = PREVIEW_KINDS[name];
|
|
48
|
+
return mimeType === kind.mimeType && filename === kind.filename && src.startsWith(kind.srcPrefix);
|
|
49
|
+
});
|
|
50
|
+
};
|
|
51
|
+
/**
|
|
52
|
+
* Charge every generated preview in the model against its kind's allowance and
|
|
53
|
+
* drop the `src` of those past it. Dropping leaves the image in place with its
|
|
54
|
+
* size and wrap, so the page still reserves the space the drawing occupies,
|
|
55
|
+
* and never touches the preserved XML the package saves from.
|
|
56
|
+
*/
|
|
57
|
+
const enforcePackagePreviewBudget = (root, overrides = {}) => {
|
|
58
|
+
const remaining = new Map(KIND_NAMES.map((name) => [name, Math.max(0, overrides[name] ?? PREVIEW_KINDS[name].maxPackageCharacters)]));
|
|
59
|
+
const visited = /* @__PURE__ */ new WeakSet();
|
|
60
|
+
const visit = (value) => {
|
|
61
|
+
if (value === null || typeof value !== "object" || visited.has(value)) return;
|
|
62
|
+
visited.add(value);
|
|
63
|
+
if (value instanceof ArrayBuffer || ArrayBuffer.isView(value)) return;
|
|
64
|
+
if (value instanceof Map) {
|
|
65
|
+
for (const child of value.values()) visit(child);
|
|
66
|
+
return;
|
|
67
|
+
}
|
|
68
|
+
if (Array.isArray(value)) {
|
|
69
|
+
for (const child of value) visit(child);
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
const kind = previewKindOf(value);
|
|
73
|
+
if (kind !== void 0) {
|
|
74
|
+
const image = value;
|
|
75
|
+
const length = image.src.length;
|
|
76
|
+
const left = remaining.get(kind);
|
|
77
|
+
if (length <= left) remaining.set(kind, left - length);
|
|
78
|
+
else {
|
|
79
|
+
remaining.set(kind, 0);
|
|
80
|
+
delete image.src;
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
for (const child of Object.values(value)) visit(child);
|
|
84
|
+
};
|
|
85
|
+
visit(root);
|
|
86
|
+
};
|
|
87
|
+
//#endregion
|
|
88
|
+
export { PREVIEW_KINDS, VML_PREVIEW_DATA_URL_PREFIX, enforcePackagePreviewBudget };
|
|
@@ -107,21 +107,38 @@ declare function getHeaders(map: document_d_exports.RelationshipMap): document_d
|
|
|
107
107
|
*/
|
|
108
108
|
declare function getFooters(map: document_d_exports.RelationshipMap): document_d_exports.Relationship[];
|
|
109
109
|
/**
|
|
110
|
-
*
|
|
110
|
+
* What a relationship id names.
|
|
111
|
+
*
|
|
112
|
+
* The three cases are kept apart because collapsing any of them into a string
|
|
113
|
+
* turns absence into a lookup key: an id the author never wrote, an id whose
|
|
114
|
+
* target the package no longer holds, and an id that resolves are different
|
|
115
|
+
* facts, and only the last one may be read as a part.
|
|
116
|
+
*/
|
|
117
|
+
type RelationshipResolution = {
|
|
118
|
+
status: "resolved";
|
|
119
|
+
relationship: document_d_exports.Relationship;
|
|
120
|
+
} | {
|
|
121
|
+
status: "absent";
|
|
122
|
+
} | {
|
|
123
|
+
status: "dangling";
|
|
124
|
+
id: string;
|
|
125
|
+
};
|
|
126
|
+
/**
|
|
127
|
+
* Resolve a relationship id against a relationship map.
|
|
111
128
|
*
|
|
112
|
-
*
|
|
113
|
-
*
|
|
114
|
-
*
|
|
129
|
+
* The only sanctioned way to turn an `r:id`, `r:embed` or `r:link` into a part.
|
|
130
|
+
* An empty attribute is absence, not an id: `ST_RelationshipId` is an NCName,
|
|
131
|
+
* so `""` can never name a relationship, and no map can hold it as a key.
|
|
115
132
|
*/
|
|
116
|
-
declare function
|
|
133
|
+
declare function resolveRelationshipId(map: document_d_exports.RelationshipMap | null | undefined, rId: string | undefined): RelationshipResolution;
|
|
117
134
|
/**
|
|
118
|
-
* Resolve a relationship
|
|
135
|
+
* Resolve a relationship id and require it to name a relationship of one type.
|
|
119
136
|
*
|
|
120
|
-
*
|
|
121
|
-
*
|
|
122
|
-
*
|
|
137
|
+
* A reference of the wrong type is reported as dangling: the part it names
|
|
138
|
+
* exists, but not as the thing the reference asked for, and reading it anyway
|
|
139
|
+
* is how a missing image comes back as `styles.xml`.
|
|
123
140
|
*/
|
|
124
|
-
declare function
|
|
141
|
+
declare function resolveRelationshipIdOfType(map: document_d_exports.RelationshipMap | null | undefined, rId: string | undefined, type: document_d_exports.RelationshipType): RelationshipResolution;
|
|
125
142
|
/**
|
|
126
143
|
* Resolve a relative target path to an absolute path within the DOCX
|
|
127
144
|
*
|
|
@@ -158,4 +175,4 @@ declare function parsePackageRelationships(relsXml: string): document_d_exports.
|
|
|
158
175
|
*/
|
|
159
176
|
declare function formatRelationships(map: document_d_exports.RelationshipMap): string;
|
|
160
177
|
//#endregion
|
|
161
|
-
export { RELATIONSHIP_TYPES, filterByType, formatRelationships, getFooters, getHeaders, getHyperlinks, getImages, getRelationshipTypeName, isExternalHyperlink, isFooterRelationship, isHeaderRelationship, isImageRelationship, parseDocumentRelationships, parsePackageRelationships, parseRelationships,
|
|
178
|
+
export { RELATIONSHIP_TYPES, RelationshipResolution, filterByType, formatRelationships, getFooters, getHeaders, getHyperlinks, getImages, getRelationshipTypeName, isExternalHyperlink, isFooterRelationship, isHeaderRelationship, isImageRelationship, parseDocumentRelationships, parsePackageRelationships, parseRelationships, resolveRelationshipId, resolveRelationshipIdOfType, resolveRelativePath };
|
package/dist/docx/relsParser.js
CHANGED
|
@@ -155,24 +155,37 @@ function getFooters(map) {
|
|
|
155
155
|
return filterByType(map, RELATIONSHIP_TYPES.footer);
|
|
156
156
|
}
|
|
157
157
|
/**
|
|
158
|
-
* Resolve a relationship
|
|
158
|
+
* Resolve a relationship id against a relationship map.
|
|
159
159
|
*
|
|
160
|
-
*
|
|
161
|
-
*
|
|
162
|
-
*
|
|
160
|
+
* The only sanctioned way to turn an `r:id`, `r:embed` or `r:link` into a part.
|
|
161
|
+
* An empty attribute is absence, not an id: `ST_RelationshipId` is an NCName,
|
|
162
|
+
* so `""` can never name a relationship, and no map can hold it as a key.
|
|
163
163
|
*/
|
|
164
|
-
function
|
|
165
|
-
|
|
164
|
+
function resolveRelationshipId(map, rId) {
|
|
165
|
+
if (rId === void 0 || rId.length === 0) return { status: "absent" };
|
|
166
|
+
const relationship = map?.get(rId);
|
|
167
|
+
return relationship === void 0 ? {
|
|
168
|
+
status: "dangling",
|
|
169
|
+
id: rId
|
|
170
|
+
} : {
|
|
171
|
+
status: "resolved",
|
|
172
|
+
relationship
|
|
173
|
+
};
|
|
166
174
|
}
|
|
167
175
|
/**
|
|
168
|
-
* Resolve a relationship
|
|
176
|
+
* Resolve a relationship id and require it to name a relationship of one type.
|
|
169
177
|
*
|
|
170
|
-
*
|
|
171
|
-
*
|
|
172
|
-
*
|
|
178
|
+
* A reference of the wrong type is reported as dangling: the part it names
|
|
179
|
+
* exists, but not as the thing the reference asked for, and reading it anyway
|
|
180
|
+
* is how a missing image comes back as `styles.xml`.
|
|
173
181
|
*/
|
|
174
|
-
function
|
|
175
|
-
|
|
182
|
+
function resolveRelationshipIdOfType(map, rId, type) {
|
|
183
|
+
const resolved = resolveRelationshipId(map, rId);
|
|
184
|
+
if (resolved.status !== "resolved" || resolved.relationship.type === type) return resolved;
|
|
185
|
+
return {
|
|
186
|
+
status: "dangling",
|
|
187
|
+
id: resolved.relationship.id
|
|
188
|
+
};
|
|
176
189
|
}
|
|
177
190
|
/**
|
|
178
191
|
* Resolve a relative target path to an absolute path within the DOCX
|
|
@@ -232,4 +245,4 @@ function formatRelationships(map) {
|
|
|
232
245
|
return lines.join("\n");
|
|
233
246
|
}
|
|
234
247
|
//#endregion
|
|
235
|
-
export { RELATIONSHIP_TYPES, filterByType, formatRelationships, getFooters, getHeaders, getHyperlinks, getImages, getRelationshipTypeName, isExternalHyperlink, isFooterRelationship, isHeaderRelationship, isImageRelationship, parseDocumentRelationships, parsePackageRelationships, parseRelationships,
|
|
248
|
+
export { RELATIONSHIP_TYPES, filterByType, formatRelationships, getFooters, getHeaders, getHyperlinks, getImages, getRelationshipTypeName, isExternalHyperlink, isFooterRelationship, isHeaderRelationship, isImageRelationship, parseDocumentRelationships, parsePackageRelationships, parseRelationships, resolveRelationshipId, resolveRelationshipIdOfType, resolveRelativePath };
|