@stll/folio-core 0.54.1 → 0.54.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/dist/ai-edits/apply.d.ts +1 -1
  2. package/dist/ai-edits/apply.js +18 -31
  3. package/dist/ai-edits/headless.d.ts +1 -2
  4. package/dist/ai-edits/headless.js +12 -27
  5. package/dist/ai-edits/read.d.ts +1 -0
  6. package/dist/ai-edits/read.js +7 -2
  7. package/dist/ai-edits/table-mutation-plan.d.ts +2 -1
  8. package/dist/ai-edits/table-mutation-plan.js +39 -1
  9. package/dist/ai-suggestions/apply.d.ts +1 -1
  10. package/dist/ai-suggestions/apply.js +3 -2
  11. package/dist/compare/compare.js +2 -1
  12. package/dist/docx/bookmarkIds.d.ts +11 -0
  13. package/dist/docx/bookmarkIds.js +33 -0
  14. package/dist/docx/documentParser.d.ts +14 -1
  15. package/dist/docx/documentParser.js +17 -4
  16. package/dist/docx/drawingIdNormalization.js +23 -8
  17. package/dist/docx/ensureParaIds.d.ts +5 -0
  18. package/dist/docx/ensureParaIds.js +10 -2
  19. package/dist/docx/listNumberingInstances.js +11 -9
  20. package/dist/docx/noteIds.d.ts +4 -0
  21. package/dist/docx/noteIds.js +13 -0
  22. package/dist/docx/numberingIds.d.ts +14 -0
  23. package/dist/docx/numberingIds.js +16 -0
  24. package/dist/docx/numericIdAllocator.d.ts +12 -0
  25. package/dist/docx/numericIdAllocator.js +33 -0
  26. package/dist/docx/numericIdNormalization.d.ts +16 -0
  27. package/dist/docx/numericIdNormalization.js +174 -0
  28. package/dist/docx/paragraphTextBoxEnrichment.js +2 -1
  29. package/dist/docx/parser.js +17 -5
  30. package/dist/docx/rasterMime.d.ts +64 -0
  31. package/dist/docx/rasterMime.js +116 -0
  32. package/dist/docx/replyToComment.js +13 -6
  33. package/dist/docx/rezip.js +14 -7
  34. package/dist/docx/selectiveSave.js +10 -0
  35. package/dist/docx/serializer/partNamespaces.js +4 -2
  36. package/dist/docx/serializer/runSerializer.js +5 -4
  37. package/dist/docx/serializer/trackedChangeAttributes.js +7 -1
  38. package/dist/docx/server/build.js +14 -3
  39. package/dist/docx/server/createBilingualDocument.js +10 -9
  40. package/dist/docx/streamingXmlParser.d.ts +10 -1
  41. package/dist/docx/streamingXmlParser.js +88 -19
  42. package/dist/docx/unzip.d.ts +3 -1
  43. package/dist/docx/unzip.js +36 -16
  44. package/dist/layout-bridge/convert/fixedTableColumnWidths.d.ts +6 -0
  45. package/dist/layout-bridge/convert/fixedTableColumnWidths.js +23 -0
  46. package/dist/layout-bridge/convert/tableConversion.js +4 -1
  47. package/dist/layout-painter/index.js +2 -5
  48. package/dist/paged-layout/editorScrollRoot.d.ts +15 -0
  49. package/dist/paged-layout/editorScrollRoot.js +72 -0
  50. package/dist/paged-layout/scrollToPmPosition.js +8 -22
  51. package/dist/prosemirror/commands/clearParagraphIndent.d.ts +6 -0
  52. package/dist/prosemirror/commands/clearParagraphIndent.js +25 -0
  53. package/dist/prosemirror/commentIdAllocator.d.ts +15 -22
  54. package/dist/prosemirror/commentIdAllocator.js +50 -25
  55. package/dist/prosemirror/extensions/core/HistoryExtension.js +97 -5
  56. package/dist/prosemirror/extensions/core/ParagraphExtension.js +3 -1
  57. package/dist/prosemirror/extensions/features/BaseKeymapExtension.js +22 -28
  58. package/dist/prosemirror/extensions/nodes/ShapeExtension.js +8 -6
  59. package/dist/prosemirror/plugins/revisionIds.d.ts +23 -19
  60. package/dist/prosemirror/plugins/revisionIds.js +110 -29
  61. package/dist/prosemirror/plugins/suggestionMode.d.ts +2 -2
  62. package/dist/prosemirror/plugins/suggestionMode.js +196 -40
  63. package/dist/prosemirror/storyListNumbering.js +11 -9
  64. package/dist/prosemirror/textInput.d.ts +6 -1
  65. package/dist/prosemirror/textInput.js +9 -2
  66. package/dist/prosemirror/utils/visualLineNavigation.js +6 -8
  67. package/dist/utils/mergeDocumentContent.d.ts +2 -2
  68. package/dist/utils/mergeDocumentContent.js +11 -8
  69. package/package.json +2 -2
@@ -1,10 +1,16 @@
1
+ import { allocateCommentId, seedCommentIdAbove } from "../prosemirror/commentIdAllocator.js";
1
2
  import { generateHexId } from "../utils/hexId.js";
2
3
  //#region src/docx/replyToComment.ts
3
- const nextCommentId = (comments) => {
4
- let max = 0;
5
- for (const comment of comments) if (comment.id > max) max = comment.id;
6
- return max + 1;
7
- };
4
+ /**
5
+ * Create a threaded reply to an existing comment.
6
+ *
7
+ * Produces a new `Comment` whose `parentId` points at the target comment and
8
+ * whose last paragraph carries a fresh `w14:paraId`, and guarantees the parent
9
+ * has a paraId too — the two ids Word uses to link a reply to its thread in
10
+ * `commentsExtended.xml` (`w15:paraId` / `w15:paraIdParent`). The reply's
11
+ * `document.xml` range markers are synthesized at save time by
12
+ * {@link applyReplyThreadMarkers}, so no anchor work is needed here.
13
+ */
8
14
  const usedParaIds = (comments) => {
9
15
  const ids = /* @__PURE__ */ new Set();
10
16
  for (const comment of comments) for (const paragraph of comment.content) if (paragraph.paraId) ids.add(paragraph.paraId.toUpperCase());
@@ -32,6 +38,7 @@ const createReply = (comments, parentCommentId, input) => {
32
38
  const threadRootId = parent.parentId ?? parent.id;
33
39
  const rootLastParagraph = ((comments.find((comment) => comment.id === threadRootId) ?? parent).content ?? []).at(-1);
34
40
  if (!rootLastParagraph) return null;
41
+ for (const { id } of comments) seedCommentIdAbove(id);
35
42
  const used = usedParaIds(comments);
36
43
  if (!rootLastParagraph.paraId) rootLastParagraph.paraId = freshParaId(used);
37
44
  const replyParagraph = {
@@ -48,7 +55,7 @@ const createReply = (comments, parentCommentId, input) => {
48
55
  }]
49
56
  };
50
57
  return {
51
- id: nextCommentId(comments),
58
+ id: allocateCommentId(),
52
59
  author: input.author,
53
60
  date: input.date ?? (/* @__PURE__ */ new Date()).toISOString(),
54
61
  parentId: threadRootId,
@@ -33,7 +33,7 @@ import { serializeThemeXml } from "./serializer/themeSerializer.js";
33
33
  import { hasCanonicalWordprocessingPrefixes } from "./wordprocessingPrefixes.js";
34
34
  import { WORDPROCESSINGML_NAMESPACE_URIS, findChild, findChildByNamespaceUri, findChildrenByNamespaceUri, getAttribute, getAttributeByNamespaceUri, getChildElements, getLocalName, getNamespaceUri, matchesName, parseXml, parseXmlDocument } from "./xmlParser.js";
35
35
  import { TaggedError, panic } from "better-result";
36
- import { escapeXmlAttribute, escapeXmlText, validateDocxPackage } from "@stll/docx-core";
36
+ import { assertValidOoxmlNumericIds, escapeXmlAttribute, escapeXmlText, validateDocxPackage } from "@stll/docx-core";
37
37
  import { mintRelationshipId } from "@stll/docx-core/model";
38
38
  import JSZip from "jszip";
39
39
  //#region src/docx/rezip.ts
@@ -529,12 +529,19 @@ async function processNewHyperlinks(parts, zip, compressionLevel) {
529
529
  */
530
530
  const normalizePackageIdsInZip = async (zip, compressionLevel) => {
531
531
  const xmlParts = /* @__PURE__ */ new Map();
532
- for (const [path, file] of Object.entries(zip.files)) if (!file.dir && path.startsWith("word/") && path.endsWith(".xml")) xmlParts.set(path, await file.async("text"));
532
+ for (const [path, file] of Object.entries(zip.files)) if (!file.dir && path.startsWith("word/") && path.endsWith(".xml")) {
533
+ const xml = await file.async("text");
534
+ assertValidOoxmlNumericIds(xml, path);
535
+ xmlParts.set(path, xml);
536
+ }
533
537
  const normalizedParts = normalizeParaIdRangeInXmlParts(normalizeRevisionIdsInXmlParts(xmlParts));
534
- for (const [path, xml] of normalizedParts) if (xml !== xmlParts.get(path)) zip.file(path, xml, {
535
- compression: "DEFLATE",
536
- compressionOptions: { level: compressionLevel }
537
- });
538
+ for (const [path, xml] of normalizedParts) if (xml !== xmlParts.get(path)) {
539
+ assertValidOoxmlNumericIds(xml, path);
540
+ zip.file(path, xml, {
541
+ compression: "DEFLATE",
542
+ compressionOptions: { level: compressionLevel }
543
+ });
544
+ }
538
545
  };
539
546
  const EXTENDED_PROPERTIES_PATH = "docProps/app.xml";
540
547
  /**
@@ -1600,7 +1607,7 @@ const notePartXmlFor = ({ patch, originalXml, replacementXml, hasDirtyParagraph,
1600
1607
  return null;
1601
1608
  };
1602
1609
  switch (patch.type) {
1603
- case "patched": return patch.xml === originalXml ? keepOriginal() : patch.xml;
1610
+ case "patched": return patch.xml === originalXml ? null : patch.xml;
1604
1611
  case "refused": switch (patch.reason) {
1605
1612
  case "comment-range-balance": return replacementXml;
1606
1613
  case "unroutable-paragraph": return keepOriginal();
@@ -4,6 +4,7 @@ import { hasUnsynthesizedReplyRanges } from "./commentReplyMarkers.js";
4
4
  import { detectDocxConformanceClass } from "./conformance.js";
5
5
  import { validateFolioDocumentModel } from "./modelValidation.js";
6
6
  import { parseNumbering } from "./numberingParser.js";
7
+ import { normalizeImportedNumericIds } from "./numericIdNormalization.js";
7
8
  import { isUnsafePackagePath } from "./packageParts.js";
8
9
  import { RELATIONSHIP_TYPES } from "./relsParser.js";
9
10
  import { COMMENTS_CONTENT_TYPE, COMMENTS_EXTENDED_PART_LOWER, addCommentsExtendedOverride, addCommentsExtendedRelationship, applyUpdatesToZip, collectHeaderFooterUpdates, findMaxRId, hasModelDrivenPictureWatermark, hasUnmaterializedHeaderFooter, hasUnmaterializedInlineResources, updateCoreProperties, withoutAttachedTemplate } from "./rezip.js";
@@ -203,6 +204,15 @@ async function attemptSelectiveSave(doc, originalBuffer, options) {
203
204
  try {
204
205
  const zip = await (await import("jszip")).default.loadAsync(originalBuffer);
205
206
  for (const [path, file] of Object.entries(zip.files)) if (!file.dir && isUnsafePackagePath(path)) return null;
207
+ if (originalBuffer !== doc.originalBuffer) {
208
+ const sourceParts = /* @__PURE__ */ new Map();
209
+ for (const [path, file] of Object.entries(zip.files)) {
210
+ const lowerPath = path.toLowerCase();
211
+ if (file.dir || !lowerPath.startsWith("word/") || !lowerPath.endsWith(".xml")) continue;
212
+ sourceParts.set(path, await file.async("text"));
213
+ }
214
+ for (const [path, xml] of normalizeImportedNumericIds(sourceParts)) if (xml !== sourceParts.get(path)) zip.file(path, xml);
215
+ }
206
216
  const updates = /* @__PURE__ */ new Map();
207
217
  if (changedParaIds.size > 0 || structuralChange) {
208
218
  const docXmlFile = zip.file("word/document.xml");
@@ -1,6 +1,6 @@
1
1
  import { toTransitionalNamespaceUri } from "../transitionalSpelling.js";
2
2
  import { TaggedError } from "better-result";
3
- import { escapeXmlAttribute } from "@stll/docx-core";
3
+ import { assertValidOoxmlNumericIds, escapeXmlAttribute } from "@stll/docx-core";
4
4
  import { OOXML_NS } from "@stll/docx-utils";
5
5
  //#region src/docx/serializer/partNamespaces.ts
6
6
  /**
@@ -322,7 +322,9 @@ const serializePartElement = ({ partPath, rootName, rootAttributes, baselinePref
322
322
  const attributes = orderedPrefixes.map((prefix) => `xmlns:${prefix}="${escapeXmlAttribute(bindings.get(prefix))}"`);
323
323
  const ignorable = orderedPrefixes.filter((prefix) => NAMESPACE_TABLE.get(prefix)?.ignorable === true);
324
324
  if (ignorable.length > 0) attributes.push(`mc:Ignorable="${ignorable.join(" ")}"`);
325
- return `${openingTag} ${attributes.join(" ")}>${body}</${rootName}>`;
325
+ const xml = `${openingTag} ${attributes.join(" ")}>${body}</${rootName}>`;
326
+ assertValidOoxmlNumericIds(xml, partPath);
327
+ return xml;
326
328
  };
327
329
  //#endregion
328
330
  export { OOXML_NAMESPACES, UnboundNamespacePrefixError, readRootNamespaceBindings, serializePartElement };
@@ -16,7 +16,7 @@ import { serializeTextFormatting } from "./textFormattingSerializer.js";
16
16
  import { getSingularRunPropertyChange, serializeTrackedChangeAttributes } from "./trackedChangeAttributes.js";
17
17
  import { intAttr } from "./xmlUtils.js";
18
18
  import { panic } from "better-result";
19
- import { escapeXmlAttribute, escapeXmlText, requiresXmlSpacePreserve } from "@stll/docx-core";
19
+ import { escapeXmlAttribute, escapeXmlText, isValidOoxmlNumericId, requiresXmlSpacePreserve } from "@stll/docx-core";
20
20
  import { SCHEME_COLOR_VALUE_BY_THEME_COLOR, knownThemeColor, presetLineDashToken } from "@stll/docx-core/model";
21
21
  //#region src/docx/serializer/runSerializer.ts
22
22
  /**
@@ -32,7 +32,7 @@ import { SCHEME_COLOR_VALUE_BY_THEME_COLOR, knownThemeColor, presetLineDashToken
32
32
  */
33
33
  /**
34
34
  * Auto-incrementing counter for generating unique image/shape IDs.
35
- * Used as a fallback when `image.id` or `shape.id` is undefined (e.g., pasted images).
35
+ * Used when a model id cannot identify a DrawingML object (e.g., a VML lexical id).
36
36
  * Starts high (100000) to avoid collisions with IDs parsed from existing DOCX content.
37
37
  */
38
38
  let nextAutoId = 1e5;
@@ -43,9 +43,10 @@ let nextAutoId = 1e5;
43
43
  function resetAutoIdCounter() {
44
44
  nextAutoId = 1e5;
45
45
  }
46
- /** Get a unique positive integer ID, using the provided value or generating one */
46
+ /** Get an unsigned integer ID, using the provided value or generating one */
47
47
  function getUniqueId(id) {
48
- if (id !== void 0 && id !== "" && id !== 0) return String(id);
48
+ if (id !== void 0 && isValidOoxmlNumericId(id, "unsigned32")) return String(id);
49
+ if (!isValidOoxmlNumericId(nextAutoId, "unsigned32")) panic("DrawingML generated id space is exhausted.");
49
50
  return String(nextAutoId++);
50
51
  }
51
52
  function serializeRunPropertyChange(change) {
@@ -1,6 +1,6 @@
1
1
  import { DATE_UTC_ATTRIBUTE } from "../trackedChangeInfo.js";
2
2
  import { panic } from "better-result";
3
- import { escapeXmlAttribute } from "@stll/docx-core";
3
+ import { assertValidOoxmlNumericId, escapeXmlAttribute } from "@stll/docx-core";
4
4
  import { normalizeRevisionId } from "@stll/docx-core/model";
5
5
  //#region src/docx/serializer/trackedChangeAttributes.ts
6
6
  /** Enforces the singular `w:rPrChange` child in a run-property container. */
@@ -14,6 +14,12 @@ const getSingularRunPropertyChange = (propertyChanges) => {
14
14
  };
15
15
  /** Normalized, unescaped attributes in schema order for an XML element or string serializer. */
16
16
  const trackedChangeAttributeEntries = (info) => {
17
+ assertValidOoxmlNumericId({
18
+ value: info.id,
19
+ partPath: "word/*.xml",
20
+ elementName: "tracked change",
21
+ attributeName: "w:id"
22
+ });
17
23
  const author = typeof info.author === "string" ? info.author.trim() : "";
18
24
  const date = typeof info.date === "string" ? info.date.trim() : "";
19
25
  const rawUtcDate = info.utcDate;
@@ -1,4 +1,7 @@
1
+ import { mintBookmarkId, reserveBookmarkIds } from "../bookmarkIds.js";
2
+ import { mintEndnoteId } from "../noteIds.js";
1
3
  import { TaggedError } from "better-result";
4
+ import { assertValidOoxmlNumericId } from "@stll/docx-core";
2
5
  //#region src/docx/server/build.ts
3
6
  /**
4
7
  * Headless document builders.
@@ -103,10 +106,18 @@ const hyperlink = ({ text, formatting, tooltip, href, anchor }) => ({
103
106
  ...formatting
104
107
  })]
105
108
  });
106
- let nextBookmarkId = 0;
107
109
  /** `content` wrapped in a named bookmark, the target of `hyperlink({ anchor })`. */
108
110
  const bookmark = ({ name, content, id }) => {
109
- const bookmarkId = id ?? nextBookmarkId++;
111
+ if (id !== void 0) {
112
+ assertValidOoxmlNumericId({
113
+ value: id,
114
+ partPath: "word/document.xml",
115
+ elementName: "w:bookmarkStart",
116
+ attributeName: "w:id"
117
+ });
118
+ reserveBookmarkIds([id]);
119
+ }
120
+ const bookmarkId = id ?? mintBookmarkId();
110
121
  const start = {
111
122
  type: "bookmarkStart",
112
123
  id: bookmarkId,
@@ -225,7 +236,7 @@ const table = ({ header, rows, columnWidths, headerShading, repeatHeader = true
225
236
  const endnote = (doc, content) => {
226
237
  const endnotes = doc.package.endnotes ?? [];
227
238
  doc.package.endnotes = endnotes;
228
- const id = Math.max(0, ...endnotes.map((note) => note.id)) + 1;
239
+ const id = mintEndnoteId(endnotes.map((note) => note.id));
229
240
  const note = {
230
241
  type: "endnote",
231
242
  id,
@@ -1,5 +1,7 @@
1
1
  import { deterministicHexId } from "../../utils/hexId.js";
2
+ import { createBookmarkIdAllocator } from "../bookmarkIds.js";
2
3
  import { createBuiltInStyleIndex, resolveHeadingLevel } from "../builtInStyles.js";
4
+ import { createNumberingIdAllocator } from "../numberingIds.js";
3
5
  import { mergeParagraphNumbering, paragraphNumberingReference, paragraphNumberingReferenceId } from "../numberingReference.js";
4
6
  import { getParagraphText } from "../paragraphParser.js";
5
7
  import { cloneParagraphWithoutPropertySource } from "../paragraphPropertySource.js";
@@ -271,8 +273,8 @@ const createNumberingCloner = ({ numbering, styleById, warnings }) => {
271
273
  const nums = numbering?.nums ?? [];
272
274
  const abstractById = new Map(abstractNums.map((item) => [item.abstractNumId, item]));
273
275
  const numById = new Map(nums.map((item) => [item.numId, item]));
274
- let nextAbstractNumId = Math.max(0, ...abstractNums.map((item) => item.abstractNumId)) + 1;
275
- let nextNumId = Math.max(0, ...nums.map((item) => item.numId)) + 1;
276
+ const abstractIds = createNumberingIdAllocator("abstract", abstractById.keys());
277
+ const instanceIds = createNumberingIdAllocator("num", numById.keys());
276
278
  const clonedAbstract = /* @__PURE__ */ new Map();
277
279
  const clonedNum = /* @__PURE__ */ new Map();
278
280
  /**
@@ -291,10 +293,9 @@ const createNumberingCloner = ({ numbering, styleById, warnings }) => {
291
293
  const { numStyleLink: _numStyleLink, styleLink: _styleLink, ...rest } = resolved;
292
294
  const clone = {
293
295
  ...rest,
294
- abstractNumId: nextAbstractNumId,
296
+ abstractNumId: abstractIds.next(),
295
297
  levels: structuredClone(resolved.levels)
296
298
  };
297
- nextAbstractNumId += 1;
298
299
  clonedAbstract.set(sourceId, clone);
299
300
  return clone;
300
301
  };
@@ -326,10 +327,9 @@ const createNumberingCloner = ({ numbering, styleById, warnings }) => {
326
327
  if (!abstract) warnings.push(`Numbering instance ${numId} references abstractNum ${source.abstractNumId}, which is not defined; its copy shares the source counters.`);
327
328
  const clone = {
328
329
  ...source,
329
- numId: nextNumId,
330
+ numId: instanceIds.next(),
330
331
  abstractNumId: abstract ? abstract.abstractNumId : source.abstractNumId
331
332
  };
332
- nextNumId += 1;
333
333
  clonedNum.set(numId, clone);
334
334
  return clone.numId;
335
335
  };
@@ -411,7 +411,7 @@ const cloneParagraphForTarget = (paragraph, targetParaId, styleCloner, cloner, b
411
411
  return cloned;
412
412
  };
413
413
  const createBookmarkIdMinter = (source) => {
414
- let nextId = 0;
414
+ const reservedIds = [];
415
415
  const remapped = /* @__PURE__ */ new Map();
416
416
  const visit = (value, seen) => {
417
417
  if (typeof value !== "object" || value === null || seen.has(value)) return;
@@ -423,15 +423,16 @@ const createBookmarkIdMinter = (source) => {
423
423
  const record = value;
424
424
  if (record["type"] === "bookmarkStart" || record["type"] === "bookmarkEnd") {
425
425
  const id = record["id"];
426
- if (typeof id === "number") nextId = Math.max(nextId, id + 1);
426
+ if (typeof id === "number") reservedIds.push(id);
427
427
  }
428
428
  Object.values(record).forEach((item) => visit(item, seen));
429
429
  };
430
430
  visit(source, /* @__PURE__ */ new Set());
431
+ const allocator = createBookmarkIdAllocator(reservedIds);
431
432
  return { mint: (sourceId) => {
432
433
  const existing = remapped.get(sourceId);
433
434
  if (existing !== void 0) return existing;
434
- const id = nextId++;
435
+ const id = allocator.next();
435
436
  remapped.set(sourceId, id);
436
437
  return id;
437
438
  } };
@@ -16,11 +16,14 @@ type AttributeValueSpan = {
16
16
  end: number;
17
17
  };
18
18
  type OpenTagVisitor = (element: XmlElement, attributeValueSpans: ReadonlyMap<string, AttributeValueSpan>) => ReadonlyMap<string, string> | null;
19
+ type OpenTagScanVisitor = (...args: Parameters<OpenTagVisitor>) => null;
19
20
  /**
20
21
  * @param inheritedNamespaceScope the bindings in scope around `xml`, for a
21
22
  * fragment captured out of a part whose root declared them.
22
23
  */
23
24
  declare const parseStreamingXml: (xml: string, inheritedNamespaceScope?: XmlNamespaceScope) => ParseXmlResult;
25
+ /** Retain the body tree while collecting identities for import repair. */
26
+ declare const parseStreamingXmlWithIdentityVisitor: (xml: string, visitOpenTag: OpenTagScanVisitor) => ParseXmlResult;
24
27
  /** Rewrite selected decimal attribute values while preserving every other source byte. */
25
28
  declare const rewriteStreamingXmlDecimalAttributes: (xml: string, visitOpenTag: OpenTagVisitor) => {
26
29
  status: "rewritten";
@@ -28,5 +31,11 @@ declare const rewriteStreamingXmlDecimalAttributes: (xml: string, visitOpenTag:
28
31
  } | {
29
32
  status: "unsupported";
30
33
  };
34
+ /** Visit identity attributes and namespace bindings without building an XML tree. */
35
+ declare const scanStreamingXmlNumericIdAttributes: (xml: string, visitOpenTag: OpenTagScanVisitor) => {
36
+ status: "unsupported";
37
+ } | {
38
+ status: "scanned";
39
+ };
31
40
  //#endregion
32
- export { parseStreamingXml, rewriteStreamingXmlDecimalAttributes };
41
+ export { parseStreamingXml, parseStreamingXmlWithIdentityVisitor, rewriteStreamingXmlDecimalAttributes, scanStreamingXmlNumericIdAttributes };
@@ -1,5 +1,6 @@
1
1
  import { EMPTY_NAMESPACE_SCOPE, attachXmlNamespaceContext } from "./xmlNamespaceContext.js";
2
2
  import { FOLIO_XML_RESOURCE_LIMITS } from "./xmlResourceLimits.js";
3
+ import { isOoxmlNumericIdAttributeName } from "@stll/docx-core";
3
4
  //#region src/docx/streamingXmlParser.ts
4
5
  const BUILT_IN_ENTITIES = {
5
6
  amp: "&",
@@ -8,19 +9,35 @@ const BUILT_IN_ENTITIES = {
8
9
  lt: "<",
9
10
  quot: "\""
10
11
  };
11
- const parseStreamingXmlInternal = (xml, visitOpenTag, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE) => {
12
+ const parseStreamingXmlInternal = (options) => {
13
+ const { xml, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE } = options;
14
+ const visitOpenTag = options.visitOpenTag;
15
+ let spanMode = "none";
16
+ if (visitOpenTag !== void 0) spanMode = options.mode === "tree" ? "identities" : "all";
17
+ const openTagOptions = {
18
+ xml,
19
+ start: 0,
20
+ close: 0,
21
+ spanMode,
22
+ retainAttribute: options.mode === "attributes" ? options.retainAttribute : void 0
23
+ };
12
24
  const root = { elements: [] };
13
25
  const stack = [];
14
26
  const replacements = [];
15
27
  let cursor = 0;
16
28
  let mergeAdjacentText = false;
29
+ const acceptText = (text, merge) => {
30
+ if (options.mode === "tree") return appendText(text, stack, merge);
31
+ const decoded = decodeXmlEntities(normalizeLineEndings(text));
32
+ return decoded !== null && (stack.length > 0 || decoded.trim().length === 0);
33
+ };
17
34
  while (cursor < xml.length) {
18
35
  const open = xml.indexOf("<", cursor);
19
36
  if (open === -1) {
20
- if (!appendText(xml.slice(cursor), stack, mergeAdjacentText)) return { status: "unsupported" };
37
+ if (!acceptText(xml.slice(cursor), mergeAdjacentText)) return { status: "unsupported" };
21
38
  break;
22
39
  }
23
- if (open > cursor && !appendText(xml.slice(cursor, open), stack, mergeAdjacentText)) return { status: "unsupported" };
40
+ if (open > cursor && !acceptText(xml.slice(cursor, open), mergeAdjacentText)) return { status: "unsupported" };
24
41
  if (open > cursor) mergeAdjacentText = true;
25
42
  if (xml.startsWith("<!--", open)) {
26
43
  const close = xml.indexOf("-->", open + 4);
@@ -30,7 +47,8 @@ const parseStreamingXmlInternal = (xml, visitOpenTag, inheritedNamespaceScope =
30
47
  }
31
48
  if (xml.startsWith("<![CDATA[", open)) {
32
49
  const close = xml.indexOf("]]>", open + 9);
33
- if (close === -1 || !appendRawText(normalizeLineEndings(xml.slice(open + 9, close)), stack)) return { status: "unsupported" };
50
+ const text = normalizeLineEndings(xml.slice(open + 9, close));
51
+ if (close === -1 || !(options.mode === "tree" ? appendRawText(text, stack) : stack.length > 0 || text.trim().length === 0)) return { status: "unsupported" };
34
52
  cursor = close + 3;
35
53
  mergeAdjacentText = false;
36
54
  continue;
@@ -53,11 +71,15 @@ const parseStreamingXmlInternal = (xml, visitOpenTag, inheritedNamespaceScope =
53
71
  mergeAdjacentText = false;
54
72
  continue;
55
73
  }
56
- const parsedTag = parseOpenTag(xml, open + 1, close, visitOpenTag !== void 0);
74
+ openTagOptions.start = open + 1;
75
+ openTagOptions.close = close;
76
+ const parsedTag = parseOpenTag(openTagOptions);
57
77
  if (parsedTag.status === "unsupported") return parsedTag;
58
78
  const parent = stack.at(-1)?.element;
59
- attachXmlNamespaceContext(parsedTag.element, parent === void 0 ? inheritedNamespaceScope : parent.namespaceScope);
60
- const rewritten = visitOpenTag?.(parsedTag.element, parsedTag.attributeValueSpans ?? /* @__PURE__ */ new Map());
79
+ const scope = parent?.namespaceScope ?? inheritedNamespaceScope;
80
+ if (options.mode === "attributes" && options.retainAttribute !== void 0 && parsedTag.element.attributes === void 0) parsedTag.element.namespaceScope = scope;
81
+ else attachXmlNamespaceContext(parsedTag.element, scope);
82
+ const rewritten = (options.mode === "tree" ? parsedTag.attributeValueSpans !== void 0 : options.retainAttribute === void 0 || parsedTag.element.attributes !== void 0) ? visitOpenTag?.(parsedTag.element, parsedTag.attributeValueSpans ?? /* @__PURE__ */ new Map()) : null;
61
83
  if (rewritten) for (const [attributeName, value] of rewritten) {
62
84
  const span = parsedTag.attributeValueSpans?.get(attributeName);
63
85
  if (!span) return { status: "unsupported" };
@@ -66,7 +88,7 @@ const parseStreamingXmlInternal = (xml, visitOpenTag, inheritedNamespaceScope =
66
88
  value
67
89
  });
68
90
  }
69
- appendElement(parent ?? root, parsedTag.element);
91
+ if (options.mode === "tree") appendElement(parent ?? root, parsedTag.element);
70
92
  if (!parsedTag.selfClosing) {
71
93
  if (stack.length >= FOLIO_XML_RESOURCE_LIMITS.maxDepth) return { status: "unsupported" };
72
94
  stack.push({
@@ -89,15 +111,35 @@ const parseStreamingXmlInternal = (xml, visitOpenTag, inheritedNamespaceScope =
89
111
  * fragment captured out of a part whose root declared them.
90
112
  */
91
113
  const parseStreamingXml = (xml, inheritedNamespaceScope = EMPTY_NAMESPACE_SCOPE) => {
92
- const parsed = parseStreamingXmlInternal(xml, void 0, inheritedNamespaceScope);
114
+ const parsed = parseStreamingXmlInternal({
115
+ xml,
116
+ mode: "tree",
117
+ inheritedNamespaceScope
118
+ });
93
119
  return parsed.status === "parsed" ? {
94
120
  status: "parsed",
95
121
  value: parsed.value
96
122
  } : { status: "unsupported" };
97
123
  };
124
+ /** Retain the body tree while collecting identities for import repair. */
125
+ const parseStreamingXmlWithIdentityVisitor = (xml, visitOpenTag) => {
126
+ const parsed = parseStreamingXmlInternal({
127
+ xml,
128
+ mode: "tree",
129
+ visitOpenTag
130
+ });
131
+ return parsed.status === "parsed" ? {
132
+ status: "parsed",
133
+ value: parsed.value
134
+ } : parsed;
135
+ };
98
136
  /** Rewrite selected decimal attribute values while preserving every other source byte. */
99
137
  const rewriteStreamingXmlDecimalAttributes = (xml, visitOpenTag) => {
100
- const parsed = parseStreamingXmlInternal(xml, visitOpenTag);
138
+ const parsed = parseStreamingXmlInternal({
139
+ xml,
140
+ mode: "attributes",
141
+ visitOpenTag
142
+ });
101
143
  if (parsed.status === "unsupported") return parsed;
102
144
  const chunks = [];
103
145
  let cursor = 0;
@@ -112,7 +154,23 @@ const rewriteStreamingXmlDecimalAttributes = (xml, visitOpenTag) => {
112
154
  value: chunks.join("")
113
155
  };
114
156
  };
115
- const parseOpenTag = (xml, start, close, captureAttributeSpans) => {
157
+ /** Visit identity attributes and namespace bindings without building an XML tree. */
158
+ const scanStreamingXmlNumericIdAttributes = (xml, visitOpenTag) => {
159
+ const parsed = parseStreamingXmlInternal({
160
+ xml,
161
+ mode: "attributes",
162
+ visitOpenTag,
163
+ retainAttribute: ({ attributeName, elementName }) => {
164
+ if (attributeName === "xmlns" || attributeName.startsWith("xmlns:")) return true;
165
+ return isOoxmlNumericIdAttributeName({
166
+ elementName,
167
+ attributeName
168
+ });
169
+ }
170
+ });
171
+ return parsed.status === "parsed" ? { status: "scanned" } : parsed;
172
+ };
173
+ const parseOpenTag = ({ xml, start, close, spanMode, retainAttribute }) => {
116
174
  let cursor = skipWhitespace(xml, start, close);
117
175
  const nameStart = cursor;
118
176
  cursor = scanName(xml, cursor, close);
@@ -120,7 +178,7 @@ const parseOpenTag = (xml, start, close, captureAttributeSpans) => {
120
178
  const name = xml.slice(nameStart, cursor);
121
179
  if (isUnsafePropertyName(name)) return { status: "unsupported" };
122
180
  let attributes;
123
- const attributeValueSpans = captureAttributeSpans ? /* @__PURE__ */ new Map() : void 0;
181
+ let attributeValueSpans;
124
182
  let selfClosing = false;
125
183
  while (cursor < close) {
126
184
  cursor = skipWhitespace(xml, cursor, close);
@@ -146,12 +204,23 @@ const parseOpenTag = (xml, start, close, captureAttributeSpans) => {
146
204
  if (cursor === -1 || cursor > close) return { status: "unsupported" };
147
205
  const decoded = decodeXmlEntities(normalizeLineEndings(xml.slice(valueStart, cursor)));
148
206
  if (decoded === null) return { status: "unsupported" };
149
- attributes ??= {};
150
- attributes[attributeName] = decoded;
151
- attributeValueSpans?.set(attributeName, {
152
- start: valueStart,
153
- end: cursor
154
- });
207
+ if (retainAttribute === void 0 || retainAttribute({
208
+ attributeName,
209
+ elementName: name
210
+ })) {
211
+ attributes ??= {};
212
+ attributes[attributeName] = decoded;
213
+ if (spanMode === "all" || spanMode === "identities" && isOoxmlNumericIdAttributeName({
214
+ elementName: name,
215
+ attributeName
216
+ })) {
217
+ attributeValueSpans ??= /* @__PURE__ */ new Map();
218
+ attributeValueSpans.set(attributeName, {
219
+ start: valueStart,
220
+ end: cursor
221
+ });
222
+ }
223
+ }
155
224
  cursor += 1;
156
225
  }
157
226
  const element = {
@@ -289,4 +358,4 @@ const isUnsafePropertyName = (name) => {
289
358
  }
290
359
  };
291
360
  //#endregion
292
- export { parseStreamingXml, rewriteStreamingXmlDecimalAttributes };
361
+ export { parseStreamingXml, parseStreamingXmlWithIdentityVisitor, rewriteStreamingXmlDecimalAttributes, scanStreamingXmlNumericIdAttributes };
@@ -99,6 +99,8 @@ type UnzipDocxBehavior = {
99
99
  * @returns Promise resolving to extracted content
100
100
  */
101
101
  declare function unzipDocx(buffer: ArrayBuffer, options?: DocxUnzipOptions, { verifyUnreadEntries }?: UnzipDocxBehavior): Promise<RawDocxContent>;
102
+ /** Keep extracted projections and the source ZIP aligned after parser normalization. */
103
+ declare const replaceRawDocxXmlParts: (content: RawDocxContent, parts: ReadonlyMap<string, string>) => Promise<void>;
102
104
  declare function getEntryUncompressedSize(file: JSZip.JSZipObject): number | null;
103
105
  /**
104
106
  * Get a list of all files in the DOCX
@@ -160,4 +162,4 @@ declare function getContentSummary(content: RawDocxContent): {
160
162
  totalFiles: number;
161
163
  };
162
164
  //#endregion
163
- export { DocxSecurityError, DocxUnzipLimits, DocxUnzipOptions, RawDocxContent, UnzipDocxBehavior, extractFile, getContentSummary, getEntryUncompressedSize, getFileList, getMediaMimeType, hasFile, mediaToDataUrl, unzipDocx };
165
+ export { DocxSecurityError, DocxUnzipLimits, DocxUnzipOptions, RawDocxContent, UnzipDocxBehavior, extractFile, getContentSummary, getEntryUncompressedSize, getFileList, getMediaMimeType, hasFile, mediaToDataUrl, replaceRawDocxXmlParts, unzipDocx };
@@ -2,6 +2,7 @@ import { bytesToDataUrl } from "../utils/base64.js";
2
2
  import { compressionRatioLimitFor, countCentralDirectoryRecords, createInflationBudget, exceedsCompressionRatio, getZipEntrySizes, inflateEntryWithinLimits, isStoredZipEntry } from "./archiveInflation.js";
3
3
  import { DOCX_CONTAINER_TYPES, detectDocxContainerType } from "./encryption/containerFormat.js";
4
4
  import { openDocxBuffer } from "./encryption/openEncryptedDocx.js";
5
+ import { RASTER_MIME_TYPES, detectRasterMimeType } from "./rasterMime.js";
5
6
  import { decodeXmlBytes } from "./xmlEncoding.js";
6
7
  import { FOLIO_XML_RESOURCE_LIMITS, assertXmlResourceLimits, createXmlPackageBudget } from "./xmlResourceLimits.js";
7
8
  import JSZip from "jszip";
@@ -38,12 +39,7 @@ var DocxSecurityError = class extends Error {
38
39
  }
39
40
  };
40
41
  const DEFAULT_ALLOWED_MEDIA_MIME_TYPES = /* @__PURE__ */ new Set([
41
- "image/png",
42
- "image/jpeg",
43
- "image/gif",
44
- "image/bmp",
45
- "image/tiff",
46
- "image/webp",
42
+ ...RASTER_MIME_TYPES,
47
43
  "image/x-emf",
48
44
  "image/x-wmf"
49
45
  ]);
@@ -227,11 +223,12 @@ async function unzipDocx(buffer, options = {}, { verifyUnreadEntries = true } =
227
223
  return null;
228
224
  }
229
225
  const binaryContent = result.bytes.buffer;
230
- if (!isMediaContentAllowed(binaryContent, mimeType)) return null;
226
+ const effectiveMimeType = RASTER_MIME_TYPES.has(mimeType) ? detectRasterMimeType(binaryContent) : mimeType;
227
+ if (!effectiveMimeType || !limits.allowedMediaMimeTypes.has(effectiveMimeType) || !isMediaContentAllowed(binaryContent, effectiveMimeType)) return null;
231
228
  return {
232
229
  type: "media",
233
230
  path,
234
- mimeType,
231
+ mimeType: effectiveMimeType,
235
232
  content: binaryContent
236
233
  };
237
234
  });
@@ -274,6 +271,26 @@ async function unzipDocx(buffer, options = {}, { verifyUnreadEntries = true } =
274
271
  }
275
272
  return content;
276
273
  }
274
+ /** Keep extracted projections and the source ZIP aligned after parser normalization. */
275
+ const replaceRawDocxXmlParts = async (content, parts) => {
276
+ let identitiesChanged = false;
277
+ for (const [path, xmlContent] of parts) {
278
+ if (content.allXml.get(path) === xmlContent) continue;
279
+ indexXmlContent(content, {
280
+ type: "xml",
281
+ path,
282
+ lowerPath: path.toLowerCase(),
283
+ content: xmlContent
284
+ });
285
+ content.originalZip.file(path, xmlContent);
286
+ identitiesChanged = true;
287
+ }
288
+ if (identitiesChanged) content.originalBuffer = await content.originalZip.generateAsync({
289
+ type: "arraybuffer",
290
+ compression: "DEFLATE",
291
+ compressionOptions: { level: 1 }
292
+ });
293
+ };
277
294
  /**
278
295
  * Preflight every XML part the unzip retains, not a named few.
279
296
  *
@@ -291,6 +308,14 @@ function assignXmlContent(content, { path, lowerPath, content: xmlContent }, lim
291
308
  partPath: path,
292
309
  budget
293
310
  });
311
+ indexXmlContent(content, {
312
+ type: "xml",
313
+ path,
314
+ lowerPath,
315
+ content: xmlContent
316
+ });
317
+ }
318
+ const indexXmlContent = (content, { path, lowerPath, content: xmlContent }) => {
294
319
  content.allXml.set(path, xmlContent);
295
320
  if (lowerPath === "word/document.xml") content.documentXml = xmlContent;
296
321
  else if (lowerPath === "word/styles.xml") content.stylesXml = xmlContent;
@@ -317,7 +342,7 @@ function assignXmlContent(content, { path, lowerPath, content: xmlContent }, lim
317
342
  const filename = path.split("/").pop() || path;
318
343
  content.footers.set(filename, xmlContent);
319
344
  }
320
- }
345
+ };
321
346
  function inflationLimitError(limit, path) {
322
347
  switch (limit) {
323
348
  case "declared-size": return new DocxSecurityError(`DOCX entry inflates past its declared size: ${path}`);
@@ -499,14 +524,9 @@ function isEntryTooLarge(declaredSize, maxBytes) {
499
524
  return declaredSize !== null && declaredSize > maxBytes;
500
525
  }
501
526
  function isMediaContentAllowed(data, mimeType) {
527
+ if (RASTER_MIME_TYPES.has(mimeType)) return detectRasterMimeType(data) === mimeType;
502
528
  const bytes = new Uint8Array(data);
503
529
  switch (mimeType) {
504
- case "image/png": return bytes[0] === 137 && bytes[1] === 80 && bytes[2] === 78 && bytes[3] === 71;
505
- case "image/jpeg": return bytes[0] === 255 && bytes[1] === 216;
506
- case "image/gif": return bytes[0] === 71 && bytes[1] === 73 && bytes[2] === 70 && bytes[3] === 56;
507
- case "image/bmp": return bytes[0] === 66 && bytes[1] === 77;
508
- case "image/webp": return bytes[0] === 82 && bytes[1] === 73 && bytes[2] === 70 && bytes[3] === 70 && bytes[8] === 87 && bytes[9] === 69 && bytes[10] === 66 && bytes[11] === 80;
509
- case "image/tiff": return bytes[0] === 73 && bytes[1] === 73 || bytes[0] === 77 && bytes[1] === 77;
510
530
  case "image/x-emf":
511
531
  case "image/emf": return bytes.length >= 44 && bytes[0] === 1 && bytes[40] === 32 && bytes[41] === 69 && bytes[42] === 77 && bytes[43] === 70;
512
532
  case "image/x-wmf":
@@ -605,4 +625,4 @@ function getContentSummary(content) {
605
625
  };
606
626
  }
607
627
  //#endregion
608
- export { DocxSecurityError, extractFile, getContentSummary, getEntryUncompressedSize, getFileList, getMediaMimeType, hasFile, mediaToDataUrl, unzipDocx };
628
+ export { DocxSecurityError, extractFile, getContentSummary, getEntryUncompressedSize, getFileList, getMediaMimeType, hasFile, mediaToDataUrl, replaceRawDocxXmlParts, unzipDocx };
@@ -0,0 +1,6 @@
1
+ import { Node } from "prosemirror-model";
2
+ //#region src/layout-bridge/convert/fixedTableColumnWidths.d.ts
3
+ /** Resolve absolute cell preferences for an unbounded fixed-layout grid (§17.4.53). */
4
+ declare const fixedTableColumnWidths: (table: Node, grid: readonly number[]) => readonly number[];
5
+ //#endregion
6
+ export { fixedTableColumnWidths };