@stll/folio-core 0.53.0 → 0.54.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (148) hide show
  1. package/dist/ai-edits/apply.js +390 -166
  2. package/dist/ai-edits/batch-claims.d.ts +21 -6
  3. package/dist/ai-edits/batch-claims.js +47 -16
  4. package/dist/ai-edits/headless.d.ts +20 -0
  5. package/dist/ai-edits/headless.js +140 -11
  6. package/dist/ai-edits/pending-suggestions.js +1 -1
  7. package/dist/ai-edits/read.d.ts +5 -3
  8. package/dist/ai-edits/read.js +97 -47
  9. package/dist/ai-edits/revisionStretches.d.ts +17 -0
  10. package/dist/ai-edits/revisionStretches.js +80 -0
  11. package/dist/ai-edits/snapshot.js +3 -2
  12. package/dist/ai-edits/table-cell-mutations.d.ts +3 -1
  13. package/dist/ai-edits/table-cell-mutations.js +2 -1
  14. package/dist/ai-edits/table-row-column-mutations.d.ts +6 -1
  15. package/dist/ai-edits/table-row-column-mutations.js +27 -10
  16. package/dist/compare/content-alignment.d.ts +5 -1
  17. package/dist/compare/content-alignment.js +39 -32
  18. package/dist/compare/inline-provenance.js +1 -1
  19. package/dist/controller/layoutPipeline.js +3 -2
  20. package/dist/docx/archiveInflation.d.ts +127 -0
  21. package/dist/docx/archiveInflation.js +181 -0
  22. package/dist/docx/metadataPrivacy.js +24 -1
  23. package/dist/docx/noteReferenceMark.d.ts +15 -0
  24. package/dist/docx/noteReferenceMark.js +48 -0
  25. package/dist/docx/paragraphParser.js +5 -3
  26. package/dist/docx/paragraphPropertySource.d.ts +1 -1
  27. package/dist/docx/paragraphPropertySource.js +3 -1
  28. package/dist/docx/parser.js +2 -2
  29. package/dist/docx/rezip.js +4 -2
  30. package/dist/docx/selectiveXmlPatch.d.ts +3 -1
  31. package/dist/docx/selectiveXmlPatch.js +81 -3
  32. package/dist/docx/server/boundedArchive.d.ts +8 -13
  33. package/dist/docx/server/boundedArchive.js +70 -59
  34. package/dist/docx/server/extractDocxText.d.ts +8 -2
  35. package/dist/docx/server/extractDocxText.js +2 -2
  36. package/dist/docx/server/validateDocxConformance.d.ts +1 -1
  37. package/dist/docx/server/validateDocxConformance.js +1 -0
  38. package/dist/docx/unzip.d.ts +19 -2
  39. package/dist/docx/unzip.js +135 -24
  40. package/dist/fonts/embeddedFonts.js +1 -1
  41. package/dist/i18n/messages/catalogs.gen.d.ts +34 -34
  42. package/dist/i18n/messages/catalogs.gen.js +34 -34
  43. package/dist/i18n/messages/messages.gen.d.ts +2 -2
  44. package/dist/internal/wholeStoryRevisionResolution.js +95 -25
  45. package/dist/layout-engine/measure/measureParagraph.js +2 -2
  46. package/dist/layout-engine/measure/paragraphMeasureShared.d.ts +12 -1
  47. package/dist/layout-engine/measure/paragraphMeasureShared.js +14 -1
  48. package/dist/markdown/escape.d.ts +2 -2
  49. package/dist/markdown/escape.js +4 -5
  50. package/dist/markdown/renderBlock.js +91 -25
  51. package/dist/markdown/renderRuns.d.ts +10 -1
  52. package/dist/markdown/renderRuns.js +320 -59
  53. package/dist/prosemirror/clearRunColor.d.ts +6 -0
  54. package/dist/prosemirror/clearRunColor.js +46 -0
  55. package/dist/prosemirror/commands/comments.js +56 -21
  56. package/dist/prosemirror/commands/formatPainter.js +1 -1
  57. package/dist/prosemirror/commands/hyperlink.js +2 -1
  58. package/dist/prosemirror/commands/propertyChangeScope.d.ts +3 -2
  59. package/dist/prosemirror/commands/propertyChangeScope.js +24 -3
  60. package/dist/prosemirror/commands/resolveAllTableChanges.js +46 -41
  61. package/dist/prosemirror/commands/resolveParagraphProperties.js +20 -6
  62. package/dist/prosemirror/commands/tableCellMergeResolution.d.ts +8 -1
  63. package/dist/prosemirror/commands/tableCellMergeResolution.js +23 -7
  64. package/dist/prosemirror/containerFinalParagraph.d.ts +11 -5
  65. package/dist/prosemirror/containerFinalParagraph.js +7 -6
  66. package/dist/prosemirror/conversion/fromProseDoc.d.ts +2 -13
  67. package/dist/prosemirror/conversion/fromProseDoc.js +12 -590
  68. package/dist/prosemirror/conversion/toProseDoc.d.ts +4 -11
  69. package/dist/prosemirror/conversion/toProseDoc.js +31 -150
  70. package/dist/prosemirror/documentSchema.d.ts +8 -0
  71. package/dist/prosemirror/documentSchema.js +18 -0
  72. package/dist/prosemirror/extensions/StarterKit.js +2 -0
  73. package/dist/prosemirror/extensions/core/ParagraphExtension.js +104 -63
  74. package/dist/prosemirror/extensions/features/BaseKeymapExtension.js +1 -1
  75. package/dist/prosemirror/extensions/features/EmptyParagraphFormatExtension.js +24 -3
  76. package/dist/prosemirror/extensions/features/JoinedRunStyleExtension.d.ts +20 -0
  77. package/dist/prosemirror/extensions/features/JoinedRunStyleExtension.js +43 -0
  78. package/dist/prosemirror/extensions/features/ListExtension.js +10 -9
  79. package/dist/prosemirror/extensions/features/pastedHeadingStyles.js +1 -1
  80. package/dist/prosemirror/extensions/features/pastedHtmlLists.js +2 -4
  81. package/dist/prosemirror/extensions/marks/FootnoteRefExtension.js +14 -8
  82. package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +8 -4
  83. package/dist/prosemirror/extensions/marks/StrikeExtension.js +2 -2
  84. package/dist/prosemirror/extensions/marks/SubscriptExtension.js +2 -2
  85. package/dist/prosemirror/extensions/marks/SuperscriptExtension.js +2 -2
  86. package/dist/prosemirror/extensions/marks/TextColorExtension.js +4 -3
  87. package/dist/prosemirror/extensions/marks/markUtils.d.ts +8 -6
  88. package/dist/prosemirror/extensions/marks/markUtils.js +17 -8
  89. package/dist/prosemirror/extensions/marks/noteReferenceDeletion.d.ts +3 -1
  90. package/dist/prosemirror/extensions/marks/noteReferenceDeletion.js +12 -1
  91. package/dist/prosemirror/extensions/nodes/CommentReferenceExtension.js +2 -2
  92. package/dist/prosemirror/extensions/nodes/HardBreakExtension.js +5 -3
  93. package/dist/prosemirror/extensions/nodes/TableExtension.js +104 -210
  94. package/dist/prosemirror/extensions/types.d.ts +7 -0
  95. package/dist/prosemirror/hyperlinkRemoval.d.ts +9 -0
  96. package/dist/prosemirror/hyperlinkRemoval.js +129 -0
  97. package/dist/prosemirror/index.d.ts +2 -1
  98. package/dist/prosemirror/index.js +2 -1
  99. package/dist/prosemirror/listNumbering.d.ts +3 -21
  100. package/dist/prosemirror/listNumbering.js +10 -25
  101. package/dist/prosemirror/listRendering.d.ts +27 -0
  102. package/dist/prosemirror/listRendering.js +27 -0
  103. package/dist/prosemirror/markupViewNotes.d.ts +8 -0
  104. package/dist/prosemirror/markupViewNotes.js +49 -0
  105. package/dist/prosemirror/markupViewProjection.d.ts +3 -1
  106. package/dist/prosemirror/markupViewProjection.js +7 -2
  107. package/dist/prosemirror/noteReferenceReview.d.ts +77 -0
  108. package/dist/prosemirror/noteReferenceReview.js +332 -0
  109. package/dist/prosemirror/paragraphIndentation.js +5 -2
  110. package/dist/prosemirror/paragraphMarkJoin.d.ts +6 -4
  111. package/dist/prosemirror/paragraphMarkJoin.js +21 -19
  112. package/dist/prosemirror/paragraphPropertyCarry.d.ts +64 -0
  113. package/dist/prosemirror/paragraphPropertyCarry.js +158 -0
  114. package/dist/prosemirror/plugins/documentStyleState.d.ts +26 -0
  115. package/dist/prosemirror/plugins/documentStyleState.js +32 -0
  116. package/dist/prosemirror/plugins/documentStyles.d.ts +2 -21
  117. package/dist/prosemirror/plugins/documentStyles.js +66 -29
  118. package/dist/prosemirror/plugins/index.d.ts +2 -1
  119. package/dist/prosemirror/plugins/index.js +2 -1
  120. package/dist/prosemirror/plugins/paragraphStyleResolution.d.ts +15 -0
  121. package/dist/prosemirror/plugins/paragraphStyleResolution.js +230 -0
  122. package/dist/prosemirror/plugins/suggestionMode.d.ts +12 -2
  123. package/dist/prosemirror/plugins/suggestionMode.js +391 -28
  124. package/dist/prosemirror/rebaseParagraphRunFormatting.d.ts +22 -2
  125. package/dist/prosemirror/rebaseParagraphRunFormatting.js +39 -13
  126. package/dist/prosemirror/rebaseParagraphRuns.d.ts +36 -0
  127. package/dist/prosemirror/rebaseParagraphRuns.js +99 -0
  128. package/dist/prosemirror/runFormattingFromMarks.d.ts +18 -0
  129. package/dist/prosemirror/runFormattingFromMarks.js +567 -0
  130. package/dist/prosemirror/runFormattingReconciliation.js +10 -9
  131. package/dist/prosemirror/schema/nodes.d.ts +2 -0
  132. package/dist/prosemirror/styles/paragraphStyleCascade.d.ts +59 -0
  133. package/dist/prosemirror/styles/paragraphStyleCascade.js +145 -0
  134. package/dist/prosemirror/styles/resolvedStyleAttrs.d.ts +21 -2
  135. package/dist/prosemirror/styles/resolvedStyleAttrs.js +38 -1
  136. package/dist/prosemirror/styles/tableStyleRegions.d.ts +31 -0
  137. package/dist/prosemirror/styles/tableStyleRegions.js +55 -0
  138. package/dist/prosemirror/tableCellPaste.d.ts +74 -0
  139. package/dist/prosemirror/tableCellPaste.js +525 -0
  140. package/dist/prosemirror/tableGridMutation.d.ts +53 -14
  141. package/dist/prosemirror/tableGridMutation.js +166 -18
  142. package/dist/prosemirror/tableRunIn.d.ts +51 -0
  143. package/dist/prosemirror/tableRunIn.js +165 -0
  144. package/dist/prosemirror/textInput.js +11 -2
  145. package/dist/prosemirror/trackedRevisionPath.d.ts +17 -0
  146. package/dist/prosemirror/trackedRevisionPath.js +32 -0
  147. package/dist/server.d.ts +2 -2
  148. package/package.json +5 -5
@@ -1,17 +1,5 @@
1
1
  import { XmlResourceLimits } from "../xmlResourceLimits.js";
2
- import JSZip from "jszip";
3
2
  //#region src/docx/server/boundedArchive.d.ts
4
- declare module "jszip" {
5
- interface JSZipObject {
6
- /**
7
- * Chunked read of the entry content, missing from the published typings.
8
- * `nodeStream` is this stream wrapped in a Node.js `Readable`, which
9
- * browsers and web workers cannot provide; the stream itself is
10
- * platform-neutral.
11
- */
12
- internalStream(type: "uint8array"): JSZip.JSZipStreamHelper<Uint8Array>;
13
- }
14
- }
15
3
  declare const DOCX_MAX_ENTRY_BYTES: number;
16
4
  declare const DOCX_MAX_TOTAL_BYTES: number;
17
5
  declare const DOCX_MAX_ENTRIES = 4096;
@@ -20,7 +8,7 @@ declare const DocxArchiveError_base: import("better-result").TaggedErrorClass<"D
20
8
  /** Error raised when a DOCX archive cannot be loaded within configured limits. */
21
9
  declare class DocxArchiveError extends DocxArchiveError_base<{
22
10
  message: string;
23
- reason: "load-failed" | "input-too-large" | "too-many-entries" | "entry-too-large" | "total-too-large" | "invalid-options";
11
+ reason: "load-failed" | "input-too-large" | "too-many-entries" | "entry-too-large" | "total-too-large" | "compression-ratio-exceeded" | "invalid-options";
24
12
  cause?: unknown;
25
13
  }> {}
26
14
  type DocxArchiveOptions = {
@@ -28,6 +16,13 @@ type DocxArchiveOptions = {
28
16
  maxEntryBytes?: number;
29
17
  maxTotalBytes?: number;
30
18
  maxEntries?: number;
19
+ /**
20
+ * Most inflated bytes allowed per compressed byte, for each markup or text
21
+ * entry and for the package as a whole. Binary entries are bounded by the
22
+ * byte limits and the package ratio only, and entries and packages under
23
+ * 4 MiB inflated are exempt. Defaults to 200.
24
+ */
25
+ maxCompressionRatio?: number;
31
26
  /**
32
27
  * Bounds on parsed XML structure, applied to every XML part this archive
33
28
  * hands out as a string.
@@ -1,3 +1,4 @@
1
+ import { compressionRatioLimitFor, countCentralDirectoryRecords, createInflationBudget, exceedsCompressionRatio, getZipEntrySizes, inflateEntryWithinLimits } from "../archiveInflation.js";
1
2
  import { FOLIO_XML_RESOURCE_LIMITS, assertXmlResourceLimits, createXmlPackageBudget } from "../xmlResourceLimits.js";
2
3
  import { TaggedError } from "better-result";
3
4
  import JSZip from "jszip";
@@ -8,49 +9,26 @@ const DOCX_MAX_ENTRIES = 4096;
8
9
  const DOCX_MAX_INPUT_BYTES = 50 * 1024 * 1024;
9
10
  /** Error raised when a DOCX archive cannot be loaded within configured limits. */
10
11
  var DocxArchiveError = class extends TaggedError("DocxArchiveError") {};
11
- const concatChunks = (chunks) => {
12
- let totalBytes = 0;
13
- for (const chunk of chunks) totalBytes += chunk.length;
14
- const merged = new Uint8Array(totalBytes);
15
- let offset = 0;
16
- for (const chunk of chunks) {
17
- merged.set(chunk, offset);
18
- offset += chunk.length;
12
+ const readLimitError = ({ limit, path, maxEntryBytes, maxTotalBytes, maxCompressionRatio }) => {
13
+ switch (limit) {
14
+ case "declared-size": return new DocxArchiveError({
15
+ message: `DOCX entry "${path}" inflated past its declared size`,
16
+ reason: "entry-too-large"
17
+ });
18
+ case "compression-ratio": return new DocxArchiveError({
19
+ message: `DOCX entry "${path}" exceeded the ${maxCompressionRatio}:1 compression ratio limit`,
20
+ reason: "compression-ratio-exceeded"
21
+ });
22
+ case "entry": return new DocxArchiveError({
23
+ message: `DOCX entry "${path}" exceeded the ${maxEntryBytes}-byte limit`,
24
+ reason: "entry-too-large"
25
+ });
26
+ case "total":
27
+ case "aborted": return new DocxArchiveError({
28
+ message: `DOCX archive exceeded the ${maxTotalBytes}-byte cumulative limit while reading "${path}"`,
29
+ reason: "total-too-large"
30
+ });
19
31
  }
20
- return merged;
21
- };
22
- /**
23
- * Accumulate an entry chunk by chunk, checking both caps before each chunk is
24
- * retained. Pausing the stream abandons a decompression bomb mid-inflate, so
25
- * the caps bound memory instead of merely reporting the overrun afterwards.
26
- */
27
- const collectStream = async ({ stream, maxEntryBytes, remainingBytes, maxTotalBytes, path }) => await new Promise((resolve, reject) => {
28
- const chunks = [];
29
- let entryBytes = 0;
30
- const fail = (reason, message) => {
31
- stream.pause();
32
- reject(new DocxArchiveError({
33
- message,
34
- reason
35
- }));
36
- };
37
- stream.on("data", (chunk) => {
38
- entryBytes += chunk.length;
39
- if (entryBytes > maxEntryBytes) {
40
- fail("entry-too-large", `DOCX entry "${path}" exceeded the ${maxEntryBytes}-byte limit`);
41
- return;
42
- }
43
- if (entryBytes > remainingBytes) {
44
- fail("total-too-large", `DOCX archive exceeded the ${maxTotalBytes}-byte cumulative limit while reading "${path}"`);
45
- return;
46
- }
47
- chunks.push(chunk);
48
- }).on("end", () => resolve(concatChunks(chunks))).on("error", reject).resume();
49
- });
50
- const getDeclaredUncompressedBytes = (entry) => {
51
- const data = "_data" in entry ? entry._data : void 0;
52
- const declaredBytes = typeof data === "object" && data !== null && "uncompressedSize" in data ? data.uncompressedSize : void 0;
53
- return typeof declaredBytes === "number" && Number.isFinite(declaredBytes) ? declaredBytes : null;
54
32
  };
55
33
  const resolveByteLimit = ({ value, fallback, name }) => {
56
34
  const limit = value ?? fallback;
@@ -60,6 +38,7 @@ const resolveByteLimit = ({ value, fallback, name }) => {
60
38
  });
61
39
  return limit;
62
40
  };
41
+ const asBytes = (bytes) => bytes instanceof Uint8Array ? bytes : new Uint8Array(bytes);
63
42
  const loadDocxArchive = async (bytes, options = {}) => {
64
43
  const maxInputBytes = resolveByteLimit({
65
44
  value: options.maxInputBytes,
@@ -81,10 +60,19 @@ const loadDocxArchive = async (bytes, options = {}) => {
81
60
  fallback: DOCX_MAX_ENTRIES,
82
61
  name: "DOCX entry limit"
83
62
  });
63
+ const maxCompressionRatio = resolveByteLimit({
64
+ value: options.maxCompressionRatio,
65
+ fallback: 200,
66
+ name: "DOCX compression ratio limit"
67
+ });
84
68
  if (bytes.byteLength > maxInputBytes) throw new DocxArchiveError({
85
69
  message: `DOCX input contains ${bytes.byteLength} bytes (max ${maxInputBytes})`,
86
70
  reason: "input-too-large"
87
71
  });
72
+ if (countCentralDirectoryRecords(asBytes(bytes), maxEntries) > maxEntries) throw new DocxArchiveError({
73
+ message: `DOCX archive holds more than ${maxEntries} entries`,
74
+ reason: "too-many-entries"
75
+ });
88
76
  let zip;
89
77
  try {
90
78
  zip = await JSZip.loadAsync(bytes);
@@ -101,30 +89,47 @@ const loadDocxArchive = async (bytes, options = {}) => {
101
89
  reason: "too-many-entries"
102
90
  });
103
91
  let declaredTotalBytes = 0;
92
+ let declaredTotalKnown = true;
104
93
  for (const entry of archiveEntries) {
105
94
  if (entry.dir) continue;
106
- const declaredBytes = getDeclaredUncompressedBytes(entry);
107
- if (declaredBytes === null) {
108
- declaredTotalBytes = NaN;
109
- break;
95
+ const { compressedBytes, uncompressedBytes } = getZipEntrySizes(entry);
96
+ if (uncompressedBytes === null) {
97
+ declaredTotalKnown = false;
98
+ continue;
110
99
  }
111
- if (declaredBytes > maxEntryBytes) throw new DocxArchiveError({
112
- message: `DOCX entry "${entry.name}" declares ${declaredBytes} bytes (max ${maxEntryBytes})`,
100
+ if (uncompressedBytes > maxEntryBytes) throw new DocxArchiveError({
101
+ message: `DOCX entry "${entry.name}" declares ${uncompressedBytes} bytes (max ${maxEntryBytes})`,
113
102
  reason: "entry-too-large"
114
103
  });
115
- declaredTotalBytes += declaredBytes;
104
+ if (exceedsCompressionRatio({
105
+ inflatedBytes: uncompressedBytes,
106
+ compressedBytes,
107
+ maxRatio: compressionRatioLimitFor(entry.name, maxCompressionRatio)
108
+ })) throw new DocxArchiveError({
109
+ message: `DOCX entry "${entry.name}" declares more than ${maxCompressionRatio} bytes per compressed byte`,
110
+ reason: "compression-ratio-exceeded"
111
+ });
112
+ declaredTotalBytes += uncompressedBytes;
116
113
  }
117
- if (Number.isFinite(declaredTotalBytes) && declaredTotalBytes > maxTotalBytes) throw new DocxArchiveError({
114
+ if (declaredTotalKnown && declaredTotalBytes > maxTotalBytes) throw new DocxArchiveError({
118
115
  message: `DOCX archive declares ${declaredTotalBytes} cumulative bytes (max ${maxTotalBytes})`,
119
116
  reason: "total-too-large"
120
117
  });
118
+ if (exceedsCompressionRatio({
119
+ inflatedBytes: declaredTotalBytes,
120
+ compressedBytes: bytes.byteLength,
121
+ maxRatio: maxCompressionRatio
122
+ })) throw new DocxArchiveError({
123
+ message: `DOCX archive declares more than ${maxCompressionRatio} bytes per archive byte`,
124
+ reason: "compression-ratio-exceeded"
125
+ });
121
126
  const xmlLimits = {
122
127
  ...FOLIO_XML_RESOURCE_LIMITS,
123
128
  ...options.xmlLimits
124
129
  };
125
130
  const xmlBudget = createXmlPackageBudget();
126
131
  const countedParts = /* @__PURE__ */ new Set();
127
- let totalBytesRead = 0;
132
+ const inflationBudget = createInflationBudget(maxTotalBytes);
128
133
  let readChain = Promise.resolve();
129
134
  const readEntry = async (path, readOptions = {}) => {
130
135
  const requestedMaxBytes = resolveByteLimit({
@@ -135,15 +140,21 @@ const loadDocxArchive = async (bytes, options = {}) => {
135
140
  const work = async () => {
136
141
  const entry = zip.file(path);
137
142
  if (!entry) return null;
138
- const content = await collectStream({
139
- stream: entry.internalStream("uint8array"),
140
- maxEntryBytes: Math.min(requestedMaxBytes, maxEntryBytes),
141
- remainingBytes: maxTotalBytes - totalBytesRead,
143
+ const entryLimit = Math.min(requestedMaxBytes, maxEntryBytes);
144
+ const result = await inflateEntryWithinLimits({
145
+ entry,
146
+ maxEntryBytes: entryLimit,
147
+ maxCompressionRatio: compressionRatioLimitFor(path, maxCompressionRatio),
148
+ budget: inflationBudget
149
+ });
150
+ if (!result.ok) throw readLimitError({
151
+ limit: result.limit,
152
+ path,
153
+ maxEntryBytes: entryLimit,
142
154
  maxTotalBytes,
143
- path
155
+ maxCompressionRatio
144
156
  });
145
- totalBytesRead += content.length;
146
- return content;
157
+ return result.bytes;
147
158
  };
148
159
  const next = readChain.then(work, work);
149
160
  readChain = next.then(() => void 0, () => void 0);
@@ -154,7 +165,7 @@ const loadDocxArchive = async (bytes, options = {}) => {
154
165
  entryMetadata: Object.freeze(archiveEntries.map((entry) => ({
155
166
  path: entry.name,
156
167
  directory: entry.dir,
157
- declaredUncompressedBytes: getDeclaredUncompressedBytes(entry)
168
+ declaredUncompressedBytes: getZipEntrySizes(entry).uncompressedBytes
158
169
  }))),
159
170
  async readEntryString(path) {
160
171
  const content = await readEntry(path);
@@ -1,3 +1,4 @@
1
+ import { DocxArchiveOptions } from "./boundedArchive.js";
1
2
  //#region src/docx/server/extractDocxText.d.ts
2
3
  /** Document part containing an extracted paragraph. */
3
4
  type DocxParagraphSource = "header" | "body" | "footer";
@@ -59,7 +60,12 @@ type ExtractedDocxText = {
59
60
  charCount: number;
60
61
  view: "accepted";
61
62
  };
63
+ /** Options for {@link extractDocxText}. */
64
+ type ExtractDocxTextOptions = {
65
+ /** Archive limits, applied before and while each part is inflated. */
66
+ readonly archive?: DocxArchiveOptions;
67
+ };
62
68
  /** Extract paragraph text and formatting metadata from a DOCX archive. */
63
- declare const extractDocxText: (bytes: ArrayBuffer | Uint8Array) => Promise<ExtractedDocxText>;
69
+ declare const extractDocxText: (bytes: ArrayBuffer | Uint8Array, options?: ExtractDocxTextOptions) => Promise<ExtractedDocxText>;
64
70
  //#endregion
65
- export { DocxParagraphSource, DocxTableRowKind, DocxTableRowPosition, ExtractedDocxParagraph, ExtractedDocxTableCell, ExtractedDocxTableCellParagraph, ExtractedDocxText, extractDocxText };
71
+ export { DocxParagraphSource, DocxTableRowKind, DocxTableRowPosition, ExtractDocxTextOptions, ExtractedDocxParagraph, ExtractedDocxTableCell, ExtractedDocxTableCellParagraph, ExtractedDocxText, extractDocxText };
@@ -504,8 +504,8 @@ const createEmptyResult = () => ({
504
504
  view: "accepted"
505
505
  });
506
506
  /** Extract paragraph text and formatting metadata from a DOCX archive. */
507
- const extractDocxText = async (bytes) => {
508
- const archive = await loadDocxArchive(bytes);
507
+ const extractDocxText = async (bytes, options = {}) => {
508
+ const archive = await loadDocxArchive(bytes, options.archive);
509
509
  const documentXml = await archive.readEntryString("word/document.xml");
510
510
  if (documentXml === null) return createEmptyResult();
511
511
  const root = parseXml(documentXml);
@@ -4,7 +4,7 @@ import { DocxArchiveOptions } from "./boundedArchive.js";
4
4
  declare const FOLIO_DOCX_CONFORMANCE_REPORT_VERSION: 1;
5
5
  declare const FOLIO_DOCX_CONFORMANCE_PROFILE: "folio-supported-v1";
6
6
  declare const FOLIO_DOCX_CONFORMANCE_CHECKS: readonly ["archive-safety", "required-parts", "xml-well-formedness", "package-roots", "conformance-class", "canonical-model"];
7
- declare const FOLIO_DOCX_CONFORMANCE_ISSUE_CODES: readonly ["archive-load-failed", "archive-invalid-options", "archive-input-too-large", "archive-too-many-entries", "archive-entry-too-large", "archive-total-too-large", "required-part-missing", "xml-doctype-forbidden", "xml-not-well-formed", "xml-read-failed", "required-xml-unreadable", "package-root-invalid", "conformance-class-unknown", "model-invalid", "model-warning", "parser-recovery", "parser-unsupported", "parser-failed", "encrypted-container", "container-not-zip"];
7
+ declare const FOLIO_DOCX_CONFORMANCE_ISSUE_CODES: readonly ["archive-load-failed", "archive-invalid-options", "archive-input-too-large", "archive-too-many-entries", "archive-entry-too-large", "archive-total-too-large", "archive-compression-ratio-exceeded", "required-part-missing", "xml-doctype-forbidden", "xml-not-well-formed", "xml-read-failed", "required-xml-unreadable", "package-root-invalid", "conformance-class-unknown", "model-invalid", "model-warning", "parser-recovery", "parser-unsupported", "parser-failed", "encrypted-container", "container-not-zip"];
8
8
  type FolioDocxConformanceCheckId = (typeof FOLIO_DOCX_CONFORMANCE_CHECKS)[number];
9
9
  type FolioDocxConformanceIssueCode = (typeof FOLIO_DOCX_CONFORMANCE_ISSUE_CODES)[number];
10
10
  type FolioDocxConformanceCheckStatus = "passed" | "failed" | "indeterminate" | "not-run";
@@ -25,6 +25,7 @@ const FOLIO_DOCX_CONFORMANCE_ISSUE_CODES = Object.freeze([
25
25
  "archive-too-many-entries",
26
26
  "archive-entry-too-large",
27
27
  "archive-total-too-large",
28
+ "archive-compression-ratio-exceeded",
28
29
  "required-part-missing",
29
30
  "xml-doctype-forbidden",
30
31
  "xml-not-well-formed",
@@ -25,6 +25,13 @@ type DocxUnzipLimits = {
25
25
  * memory as bytes, not on what the parsed structure retains.
26
26
  */
27
27
  maxTotalUncompressedBytes: number;
28
+ /**
29
+ * Inflated bytes allowed per compressed byte, for each markup or text entry
30
+ * and for the package as a whole. Binary entries are bounded by the byte
31
+ * limits and the package ratio only, and entries and packages under 4 MiB
32
+ * inflated are exempt.
33
+ */
34
+ maxCompressionRatio: number;
28
35
  /** Elements allowed in one XML part, counted before any tree is built. */
29
36
  maxXmlElementsPerPart: number;
30
37
  /** Attributes allowed in one XML part, counted before any tree is built. */
@@ -75,13 +82,23 @@ type RawDocxContent = {
75
82
  /** True when the input was a password-protected CFB container. */
76
83
  wasEncrypted: boolean;
77
84
  };
85
+ /** How {@link unzipDocx} treats the parts it does not extract. */
86
+ type UnzipDocxBehavior = {
87
+ /**
88
+ * Inflate every part the unzip does not extract, within its declared size,
89
+ * so a later read of the returned package can trust that size. Callers
90
+ * that discard the package after reading the extracted parts can skip it.
91
+ * Defaults to `true`.
92
+ */
93
+ verifyUnreadEntries?: boolean;
94
+ };
78
95
  /**
79
96
  * Extract all content from a DOCX file
80
97
  *
81
98
  * @param buffer - DOCX file as ArrayBuffer
82
99
  * @returns Promise resolving to extracted content
83
100
  */
84
- declare function unzipDocx(buffer: ArrayBuffer, options?: DocxUnzipOptions): Promise<RawDocxContent>;
101
+ declare function unzipDocx(buffer: ArrayBuffer, options?: DocxUnzipOptions, { verifyUnreadEntries }?: UnzipDocxBehavior): Promise<RawDocxContent>;
85
102
  declare function getEntryUncompressedSize(file: JSZip.JSZipObject): number | null;
86
103
  /**
87
104
  * Get a list of all files in the DOCX
@@ -143,4 +160,4 @@ declare function getContentSummary(content: RawDocxContent): {
143
160
  totalFiles: number;
144
161
  };
145
162
  //#endregion
146
- export { DocxSecurityError, DocxUnzipLimits, DocxUnzipOptions, RawDocxContent, extractFile, getContentSummary, getEntryUncompressedSize, getFileList, getMediaMimeType, hasFile, mediaToDataUrl, unzipDocx };
163
+ export { DocxSecurityError, DocxUnzipLimits, DocxUnzipOptions, RawDocxContent, UnzipDocxBehavior, extractFile, getContentSummary, getEntryUncompressedSize, getFileList, getMediaMimeType, hasFile, mediaToDataUrl, unzipDocx };
@@ -1,4 +1,5 @@
1
1
  import { bytesToDataUrl } from "../utils/base64.js";
2
+ import { compressionRatioLimitFor, countCentralDirectoryRecords, createInflationBudget, exceedsCompressionRatio, getZipEntrySizes, inflateEntryWithinLimits, isStoredZipEntry } from "./archiveInflation.js";
2
3
  import { DOCX_CONTAINER_TYPES, detectDocxContainerType } from "./encryption/containerFormat.js";
3
4
  import { openDocxBuffer } from "./encryption/openEncryptedDocx.js";
4
5
  import { decodeXmlBytes } from "./xmlEncoding.js";
@@ -60,6 +61,7 @@ const DEFAULT_UNZIP_LIMITS = {
60
61
  maxMediaBytes: 25 * MEBIBYTE,
61
62
  maxFontBytes: 10 * MEBIBYTE,
62
63
  maxTotalUncompressedBytes: 250 * MEBIBYTE,
64
+ maxCompressionRatio: 200,
63
65
  maxXmlElementsPerPart: FOLIO_XML_RESOURCE_LIMITS.maxElementsPerPart,
64
66
  maxXmlAttributesPerPart: FOLIO_XML_RESOURCE_LIMITS.maxAttributesPerPart,
65
67
  maxXmlElementsPerPackage: FOLIO_XML_RESOURCE_LIMITS.maxElementsPerPackage,
@@ -101,19 +103,23 @@ const ZIP_CENTRAL_DIRECTORY_FILE_HEADER_SIGNATURE = 33639248;
101
103
  const ZIP_END_OF_CENTRAL_DIRECTORY_SIZE = 22;
102
104
  const ZIP_END_OF_CENTRAL_DIRECTORY_WITH_COUNTS_SIZE = 12;
103
105
  const ZIP_CENTRAL_DIRECTORY_FILE_HEADER_SIZE = 46;
106
+ /** Entries inflated at once; the rest wait for a slot. */
107
+ const EXTRACTION_CONCURRENCY = 6;
104
108
  /**
105
109
  * Extract all content from a DOCX file
106
110
  *
107
111
  * @param buffer - DOCX file as ArrayBuffer
108
112
  * @returns Promise resolving to extracted content
109
113
  */
110
- async function unzipDocx(buffer, options = {}) {
114
+ async function unzipDocx(buffer, options = {}, { verifyUnreadEntries = true } = {}) {
111
115
  const limits = createUnzipLimits(options);
112
116
  if (buffer.byteLength > limits.maxInputBytes) throw new DocxSecurityError("DOCX file exceeds the maximum allowed size");
113
117
  const containerType = detectDocxContainerType(buffer);
114
118
  const zipBuffer = containerType === DOCX_CONTAINER_TYPES.ZIP ? buffer : await openDocxBuffer(buffer, { password: options.password });
115
119
  if (zipBuffer.byteLength > limits.maxInputBytes) throw new DocxSecurityError("DOCX file exceeds the maximum allowed size");
116
120
  const wasEncrypted = containerType === DOCX_CONTAINER_TYPES.CFB;
121
+ const maxRecords = limits.maxFiles * 2;
122
+ if (countCentralDirectoryRecords(new Uint8Array(zipBuffer), maxRecords) > maxRecords) throw new DocxSecurityError("DOCX file contains too many entries");
117
123
  const loaded = await loadDocxZip(zipBuffer, limits.maxFiles);
118
124
  if (loaded.buffer.byteLength > limits.maxInputBytes) throw new DocxSecurityError("DOCX file exceeds the maximum allowed size");
119
125
  const { zip } = loaded;
@@ -149,40 +155,78 @@ async function unzipDocx(buffer, options = {}) {
149
155
  wasEncrypted
150
156
  };
151
157
  let totalUncompressedBytes = 0;
158
+ const inflationBudget = createInflationBudget(limits.maxTotalUncompressedBytes);
152
159
  const extractionTasks = [];
160
+ const verifyEntry = (file, path) => {
161
+ if (!verifyUnreadEntries || isStoredZipEntry(file)) return;
162
+ extractionTasks.push(async () => {
163
+ await verifyEntryWithinLimits({
164
+ file,
165
+ path,
166
+ limits,
167
+ budget: inflationBudget
168
+ });
169
+ return null;
170
+ });
171
+ };
172
+ const inflate = async (file, maxEntryBytes) => await inflateEntryWithinLimits({
173
+ entry: file,
174
+ maxEntryBytes,
175
+ maxCompressionRatio: compressionRatioLimitFor(file.name, limits.maxCompressionRatio),
176
+ budget: inflationBudget
177
+ });
153
178
  for (const [path, file] of entries) {
154
179
  if (!isSafeDocxPath(path)) throw new DocxSecurityError("DOCX file contains an unsafe entry path");
155
- const declaredSize = getEntryUncompressedSize(file);
180
+ const { compressedBytes, uncompressedBytes: declaredSize } = getZipEntrySizes(file);
156
181
  if (declaredSize !== null) {
157
182
  totalUncompressedBytes += declaredSize;
158
183
  if (totalUncompressedBytes > limits.maxTotalUncompressedBytes) throw new DocxSecurityError("DOCX file expands beyond the maximum allowed size");
184
+ if (exceedsCompressionRatio({
185
+ inflatedBytes: declaredSize,
186
+ compressedBytes,
187
+ maxRatio: compressionRatioLimitFor(path, limits.maxCompressionRatio)
188
+ })) throw new DocxSecurityError(`DOCX entry exceeds the maximum compression ratio: ${path}`);
189
+ }
190
+ if (!isParsedDocxEntry(path)) {
191
+ verifyEntry(file, path);
192
+ continue;
159
193
  }
160
- if (!isParsedDocxEntry(path)) continue;
161
194
  const lowerPath = path.toLowerCase();
162
195
  if (lowerPath.endsWith(".xml") || lowerPath.endsWith(".rels")) {
163
196
  assertEntrySize(path, declaredSize, limits.maxXmlBytes);
164
- if (options.extractAllXml === false && !shouldExtractXmlPart(lowerPath)) continue;
165
- extractionTasks.push(() => file.async("uint8array").then((xmlBytes) => {
166
- assertExtractedSize(path, xmlBytes.byteLength, limits.maxXmlBytes);
197
+ if (options.extractAllXml === false && !shouldExtractXmlPart(lowerPath)) {
198
+ verifyEntry(file, path);
199
+ continue;
200
+ }
201
+ extractionTasks.push(async () => {
202
+ const result = await inflate(file, limits.maxXmlBytes);
203
+ if (!result.ok) throw inflationLimitError(result.limit, path);
167
204
  return {
168
205
  type: "xml",
169
206
  path,
170
207
  lowerPath,
171
- content: decodeXmlBytes(xmlBytes)
208
+ content: decodeXmlBytes(result.bytes)
172
209
  };
173
- }));
210
+ });
174
211
  } else if (lowerPath.startsWith("word/media/")) {
175
212
  const mimeType = getMediaMimeType(path);
176
- if (!limits.allowedMediaMimeTypes.has(mimeType)) continue;
213
+ if (!limits.allowedMediaMimeTypes.has(mimeType)) {
214
+ verifyEntry(file, path);
215
+ continue;
216
+ }
177
217
  if (isEntryTooLarge(declaredSize, limits.maxMediaBytes)) {
178
218
  content.warnings.push(`Skipped oversized media file: ${path}; original entry preserved for round-trip.`);
219
+ verifyEntry(file, path);
179
220
  continue;
180
221
  }
181
- extractionTasks.push(() => file.async("arraybuffer").then((binaryContent) => {
182
- if (binaryContent.byteLength > limits.maxMediaBytes) {
222
+ extractionTasks.push(async () => {
223
+ const result = await inflate(file, limits.maxMediaBytes);
224
+ if (!result.ok) {
225
+ if (result.limit !== "entry") throw inflationLimitError(result.limit, path);
183
226
  content.warnings.push(`Skipped oversized media file: ${path}; original entry preserved for round-trip.`);
184
227
  return null;
185
228
  }
229
+ const binaryContent = result.bytes.buffer;
186
230
  if (!isMediaContentAllowed(binaryContent, mimeType)) return null;
187
231
  return {
188
232
  type: "media",
@@ -190,21 +234,33 @@ async function unzipDocx(buffer, options = {}) {
190
234
  mimeType,
191
235
  content: binaryContent
192
236
  };
193
- }));
237
+ });
194
238
  } else if (lowerPath.startsWith("word/fonts/")) {
195
- if (isEntryTooLarge(declaredSize, limits.maxFontBytes)) continue;
196
- extractionTasks.push(() => file.async("arraybuffer").then((binaryContent) => {
197
- if (binaryContent.byteLength > limits.maxFontBytes) return null;
239
+ if (isEntryTooLarge(declaredSize, limits.maxFontBytes)) {
240
+ verifyEntry(file, path);
241
+ continue;
242
+ }
243
+ extractionTasks.push(async () => {
244
+ const result = await inflate(file, limits.maxFontBytes);
245
+ if (!result.ok) {
246
+ if (result.limit !== "entry") throw inflationLimitError(result.limit, path);
247
+ return null;
248
+ }
198
249
  return {
199
250
  type: "font",
200
251
  path,
201
- content: binaryContent
252
+ content: result.bytes.buffer
202
253
  };
203
- }));
204
- }
254
+ });
255
+ } else verifyEntry(file, path);
205
256
  }
257
+ if (exceedsCompressionRatio({
258
+ inflatedBytes: totalUncompressedBytes,
259
+ compressedBytes: loaded.buffer.byteLength,
260
+ maxRatio: limits.maxCompressionRatio
261
+ })) throw new DocxSecurityError("DOCX file exceeds the maximum compression ratio");
206
262
  const xmlBudget = createXmlPackageBudget();
207
- for (const extracted of await Promise.all(extractionTasks.map((extract) => extract()))) {
263
+ for (const extracted of await runExtractionTasks(extractionTasks, inflationBudget)) {
208
264
  if (!extracted) continue;
209
265
  if (extracted.type === "xml") {
210
266
  assignXmlContent(content, extracted, limits, xmlBudget);
@@ -262,6 +318,62 @@ function assignXmlContent(content, { path, lowerPath, content: xmlContent }, lim
262
318
  content.footers.set(filename, xmlContent);
263
319
  }
264
320
  }
321
+ function inflationLimitError(limit, path) {
322
+ switch (limit) {
323
+ case "declared-size": return new DocxSecurityError(`DOCX entry inflates past its declared size: ${path}`);
324
+ case "compression-ratio": return new DocxSecurityError(`DOCX entry exceeds the maximum compression ratio: ${path}`);
325
+ case "entry": return new DocxSecurityError(`DOCX entry exceeds maximum size: ${path}`);
326
+ case "total": return new DocxSecurityError("DOCX file expands beyond the maximum allowed size");
327
+ case "aborted": return new DocxSecurityError(`DOCX entry was not read after another entry failed: ${path}`);
328
+ }
329
+ }
330
+ /**
331
+ * Run extraction tasks a few at a time, in order, keeping their results in
332
+ * task order. The first failure marks the shared budget aborted, which stops
333
+ * the streams still running at their next chunk and leaves the queued tasks
334
+ * unstarted.
335
+ */
336
+ async function runExtractionTasks(tasks, budget) {
337
+ const results = Array.from({ length: tasks.length }, () => null);
338
+ let next = 0;
339
+ const worker = async () => {
340
+ while (next < tasks.length && !budget.aborted) {
341
+ const index = next;
342
+ next += 1;
343
+ const task = tasks[index];
344
+ if (!task) return;
345
+ try {
346
+ results[index] = await task();
347
+ } catch (error) {
348
+ budget.aborted = true;
349
+ throw error;
350
+ }
351
+ }
352
+ };
353
+ await Promise.all(Array.from({ length: Math.min(EXTRACTION_CONCURRENCY, tasks.length) }, async () => await worker()));
354
+ return results;
355
+ }
356
+ /**
357
+ * Inflate an entry without keeping it, to bound what a later read of it can
358
+ * produce. A limit refuses the package. A corrupt entry does not: it was never
359
+ * going to be read here, a later read stops at the same byte, and refusing
360
+ * the whole document over a part folio does not model would be a regression.
361
+ */
362
+ async function verifyEntryWithinLimits({ file, path, limits, budget }) {
363
+ let result;
364
+ try {
365
+ result = await inflateEntryWithinLimits({
366
+ entry: file,
367
+ maxEntryBytes: Number.POSITIVE_INFINITY,
368
+ maxCompressionRatio: compressionRatioLimitFor(path, limits.maxCompressionRatio),
369
+ budget,
370
+ retain: false
371
+ });
372
+ } catch {
373
+ return;
374
+ }
375
+ if (!result.ok) throw inflationLimitError(result.limit, path);
376
+ }
265
377
  async function loadDocxZip(buffer, maxFiles) {
266
378
  try {
267
379
  return {
@@ -336,9 +448,12 @@ function findLastSignature(view, byteLength, signature) {
336
448
  return -1;
337
449
  }
338
450
  function createUnzipLimits(options) {
451
+ const { maxCompressionRatio } = options;
452
+ if (maxCompressionRatio !== void 0 && (!Number.isSafeInteger(maxCompressionRatio) || maxCompressionRatio < 0)) throw new RangeError("DOCX compression ratio limit must be a non-negative safe integer");
339
453
  return {
340
454
  ...DEFAULT_UNZIP_LIMITS,
341
455
  ...options,
456
+ maxCompressionRatio: maxCompressionRatio ?? DEFAULT_UNZIP_LIMITS.maxCompressionRatio,
342
457
  allowedMediaMimeTypes: options.allowedMediaMimeTypes ? new Set(options.allowedMediaMimeTypes) : DEFAULT_UNZIP_LIMITS.allowedMediaMimeTypes
343
458
  };
344
459
  }
@@ -375,8 +490,7 @@ function isParsedDocxEntry(path) {
375
490
  return lowerPath === "[content_types].xml" || lowerPath.startsWith("_rels/") || lowerPath.startsWith("docprops/") || lowerPath.startsWith("word/") || lowerPath.startsWith("customxml/");
376
491
  }
377
492
  function getEntryUncompressedSize(file) {
378
- const metadata = file._data;
379
- return typeof metadata?.uncompressedSize === "number" ? metadata.uncompressedSize : null;
493
+ return getZipEntrySizes(file).uncompressedBytes;
380
494
  }
381
495
  function assertEntrySize(path, declaredSize, maxBytes) {
382
496
  if (isEntryTooLarge(declaredSize, maxBytes)) throw new DocxSecurityError(`DOCX entry exceeds maximum size: ${path}`);
@@ -384,9 +498,6 @@ function assertEntrySize(path, declaredSize, maxBytes) {
384
498
  function isEntryTooLarge(declaredSize, maxBytes) {
385
499
  return declaredSize !== null && declaredSize > maxBytes;
386
500
  }
387
- function assertExtractedSize(path, byteLength, maxBytes) {
388
- if (byteLength > maxBytes) throw new DocxSecurityError(`DOCX entry exceeds maximum size: ${path}`);
389
- }
390
501
  function isMediaContentAllowed(data, mimeType) {
391
502
  const bytes = new Uint8Array(data);
392
503
  switch (mimeType) {
@@ -130,7 +130,7 @@ function findFontTableRels(allXml) {
130
130
  * embedded fonts. See {@link getEmbeddedFontFaces} for `docNonce`.
131
131
  */
132
132
  async function extractEmbeddedFonts(buffer, docNonce = generateHexId()) {
133
- const raw = await unzipDocx(buffer);
133
+ const raw = await unzipDocx(buffer, {}, { verifyUnreadEntries: false });
134
134
  return getEmbeddedFontFaces({
135
135
  fontTableXml: raw.fontTableXml,
136
136
  fontTableRelsXml: findFontTableRels(raw.allXml),