@stll/folio-core 0.55.0 → 0.56.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/dist/ai-edits/apply.js +169 -38
  2. package/dist/ai-edits/headless.js +2 -2
  3. package/dist/ai-edits/types.d.ts +7 -1
  4. package/dist/compare/compare.js +1 -0
  5. package/dist/compare/reproducible-package.d.ts +0 -17
  6. package/dist/compare/reproducible-package.js +12 -18
  7. package/dist/controller/canonicalClipboard.d.ts +23 -0
  8. package/dist/controller/canonicalClipboard.js +476 -0
  9. package/dist/controller/canonicalClipboardResources.d.ts +21 -0
  10. package/dist/controller/canonicalClipboardResources.js +382 -0
  11. package/dist/controller/canonicalComposition.d.ts +20 -3
  12. package/dist/controller/canonicalComposition.js +166 -38
  13. package/dist/controller/canonicalInlineProjection.d.ts +113 -0
  14. package/dist/controller/canonicalInlineProjection.js +393 -0
  15. package/dist/controller/canonicalInput.d.ts +13 -4
  16. package/dist/controller/canonicalInput.js +52 -13
  17. package/dist/controller/canonicalInputTimer.d.ts +7 -0
  18. package/dist/controller/canonicalInputTimer.js +11 -0
  19. package/dist/controller/canonicalPublicOperations.d.ts +48 -0
  20. package/dist/controller/canonicalPublicOperations.js +556 -0
  21. package/dist/controller/canonicalSession.d.ts +24 -6
  22. package/dist/controller/canonicalSession.js +259 -111
  23. package/dist/controller/canonicalStoryEditor.d.ts +5 -3
  24. package/dist/controller/canonicalStoryEditor.js +7 -2
  25. package/dist/controller/canonicalStructure.js +57 -4
  26. package/dist/controller/canonicalTocSelection.d.ts +11 -0
  27. package/dist/controller/canonicalTocSelection.js +43 -0
  28. package/dist/controller/folioEditor.d.ts +2 -9
  29. package/dist/controller/folioEditor.js +5 -4
  30. package/dist/controller/headerFooterEditorManager.d.ts +2 -1
  31. package/dist/controller/headerFooterEditorManager.js +8 -7
  32. package/dist/controller/hiddenEditorApi.d.ts +12 -2
  33. package/dist/controller/hiddenEditorApi.js +4 -0
  34. package/dist/controller/hiddenEditorManager.d.ts +3 -1
  35. package/dist/controller/hiddenEditorManager.js +216 -34
  36. package/dist/controller/noteEditorManager.d.ts +2 -1
  37. package/dist/controller/noteEditorManager.js +8 -7
  38. package/dist/document-operations.d.ts +8 -1
  39. package/dist/document-operations.js +6 -4
  40. package/dist/docx/blockPlainText.d.ts +1 -1
  41. package/dist/docx/blockPlainText.js +11 -3
  42. package/dist/docx/canonicalResourceSave.d.ts +16 -0
  43. package/dist/docx/canonicalResourceSave.js +27 -0
  44. package/dist/docx/canonicalSave.d.ts +29 -0
  45. package/dist/docx/canonicalSave.js +101 -0
  46. package/dist/docx/canonicalStoryRepack.d.ts +1 -1
  47. package/dist/docx/canonicalStoryRepack.js +38 -5
  48. package/dist/docx/documentParser.d.ts +1 -13
  49. package/dist/docx/documentParser.js +1 -47
  50. package/dist/docx/ensureParaIds.js +10 -4
  51. package/dist/docx/footnoteParser.d.ts +1 -11
  52. package/dist/docx/footnoteParser.js +1 -16
  53. package/dist/docx/headerFooterParser.d.ts +1 -9
  54. package/dist/docx/headerFooterParser.js +1 -12
  55. package/dist/docx/headerFooterVerbatim.d.ts +26 -1
  56. package/dist/docx/headerFooterVerbatim.js +81 -3
  57. package/dist/docx/imageParser.d.ts +13 -1
  58. package/dist/docx/imageParser.js +20 -11
  59. package/dist/docx/metadataPrivacy.js +6 -1
  60. package/dist/docx/packageParts.js +9 -3
  61. package/dist/docx/paragraphPropertySource.js +2 -1
  62. package/dist/docx/parser.js +2 -1
  63. package/dist/docx/removeHeaderFooterParts.js +17 -6
  64. package/dist/docx/rezip.d.ts +49 -5
  65. package/dist/docx/rezip.js +745 -229
  66. package/dist/docx/saveDiagnostics.d.ts +11 -0
  67. package/dist/docx/selectiveSave.d.ts +2 -1
  68. package/dist/docx/selectiveSave.js +32 -3
  69. package/dist/docx/selectiveXmlPatch.js +7 -5
  70. package/dist/docx/serializer/documentSerializer.d.ts +9 -2
  71. package/dist/docx/serializer/documentSerializer.js +29 -10
  72. package/dist/docx/server/applyDocxXmlPatchProposal.js +14 -9
  73. package/dist/docx/storyBlockReplay.d.ts +21 -1
  74. package/dist/docx/storyBlockReplay.js +113 -9
  75. package/dist/docx/storyPlainText.d.ts +20 -0
  76. package/dist/docx/storyPlainText.js +23 -0
  77. package/dist/docx/structuralXmlPatch.js +10 -22
  78. package/dist/docx/styleParser.d.ts +5 -1
  79. package/dist/docx/styleParser.js +1 -1
  80. package/dist/docx/tableParser.d.ts +1 -8
  81. package/dist/docx/tableParser.js +1 -28
  82. package/dist/docx/textBoxParser.d.ts +1 -5
  83. package/dist/docx/textBoxParser.js +1 -20
  84. package/dist/docx/unzip.js +6 -1
  85. package/dist/internal/acceptedBlockProjection.d.ts +12 -0
  86. package/dist/internal/acceptedBlockProjection.js +45 -0
  87. package/dist/managers/DocumentLoaderManager.js +2 -1
  88. package/dist/markdown/escape.d.ts +3 -1
  89. package/dist/markdown/escape.js +9 -2
  90. package/dist/markdown/renderBlock.js +5 -120
  91. package/dist/markdown/renderParagraph.js +4 -2
  92. package/dist/markdown/renderRuns.d.ts +7 -1
  93. package/dist/markdown/renderRuns.js +32 -13
  94. package/dist/prosemirror/canonicalCommands.d.ts +28 -4
  95. package/dist/prosemirror/canonicalCommands.js +11 -4
  96. package/dist/prosemirror/canonicalSelectionRange.d.ts +9 -0
  97. package/dist/prosemirror/canonicalSelectionRange.js +15 -0
  98. package/dist/prosemirror/clipboardIntent.d.ts +7 -0
  99. package/dist/prosemirror/clipboardIntent.js +9 -0
  100. package/dist/prosemirror/commands/pageBreak.js +2 -2
  101. package/dist/prosemirror/commands/pastePlainText.js +2 -1
  102. package/dist/prosemirror/conversion/hyphenTextCarriers.d.ts +8 -0
  103. package/dist/prosemirror/conversion/hyphenTextCarriers.js +8 -0
  104. package/dist/prosemirror/conversion/toProseDoc.d.ts +7 -1
  105. package/dist/prosemirror/conversion/toProseDoc.js +5 -4
  106. package/dist/prosemirror/extensions/core/ParagraphExtension.js +39 -30
  107. package/dist/prosemirror/extensions/marks/HyperlinkExtension.js +56 -28
  108. package/dist/prosemirror/extensions/marks/markUtils.js +3 -1
  109. package/dist/prosemirror/insertOperations.js +2 -1
  110. package/dist/prosemirror/paragraphPropertyCarry.d.ts +7 -1
  111. package/dist/prosemirror/paragraphPropertyCarry.js +8 -3
  112. package/dist/prosemirror/plugins/suggestionMode.js +7 -5
  113. package/dist/types/canonicalCapabilities.d.ts +215 -0
  114. package/dist/types/canonicalCapabilities.js +212 -0
  115. package/dist/types/canonicalSave.d.ts +11 -0
  116. package/dist/types/canonicalSave.js +0 -0
  117. package/dist/types/docxSerialization.d.ts +14 -0
  118. package/dist/types/docxSerialization.js +7 -0
  119. package/package.json +4 -4
@@ -1,8 +1,19 @@
1
+ import { CANONICAL_GAP } from "../types/canonicalCapabilities.js";
1
2
  //#region src/docx/saveDiagnostics.d.ts
2
3
  /** Save diagnostics describe fidelity fallbacks without including document content. */
3
4
  type SaveDiagnostic = {
4
5
  type: "sourceReplayMismatch";
5
6
  part: string;
7
+ } | {
8
+ type: "selectiveSaveRefused";
9
+ part: string;
10
+ } | {
11
+ type: "sourceReplayUnavailable";
12
+ part: string;
13
+ } | {
14
+ type: "canonicalResourceReplacement";
15
+ gap: typeof CANONICAL_GAP.resourceReplacement;
16
+ part: string;
6
17
  };
7
18
  type SaveDiagnosticOptions = {
8
19
  onDiagnostic?: ((diagnostic: SaveDiagnostic) => void) | undefined;
@@ -1,7 +1,8 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
2
  import { SaveDiagnosticOptions } from "./saveDiagnostics.js";
3
+ import { DocumentBodyAuthorityOptions } from "./serializer/documentSerializer.js";
3
4
  //#region src/docx/selectiveSave.d.ts
4
- type SelectiveSaveOptions = SaveDiagnosticOptions & {
5
+ type SelectiveSaveOptions = SaveDiagnosticOptions & DocumentBodyAuthorityOptions & {
5
6
  /** Changed paragraph IDs to selectively patch */
6
7
  changedParaIds: Set<string>;
7
8
  /** Whether paragraph membership, order, or block structure changed. */
@@ -7,7 +7,7 @@ import { parseNumbering } from "./numberingParser.js";
7
7
  import { normalizeImportedNumericIds } from "./numericIdNormalization.js";
8
8
  import { isUnsafePackagePath } from "./packageParts.js";
9
9
  import { RELATIONSHIP_TYPES } from "./relsParser.js";
10
- import { COMMENTS_CONTENT_TYPE, COMMENTS_EXTENDED_PART_LOWER, addCommentsExtendedOverride, addCommentsExtendedRelationship, applyUpdatesToZip, collectHeaderFooterUpdates, findMaxRId, hasModelDrivenPictureWatermark, hasUnmaterializedHeaderFooter, hasUnmaterializedInlineResources, updateCoreProperties, withoutAttachedTemplate } from "./rezip.js";
10
+ import { COMMENTS_CONTENT_TYPE, COMMENTS_EXTENDED_PART_LOWER, addCommentsExtendedOverride, addCommentsExtendedRelationship, applyUpdatesToZip, collectHeaderFooterUpdates, findMaxRId, hasModelDrivenPictureWatermark, hasUnmaterializedHeaderFooter, hasUnmaterializedHyperlinkBindings, hasUnmaterializedInlineResources, planAddedStyles, publishCanonicalImageResources, updateCoreProperties, withoutAttachedTemplate } from "./rezip.js";
11
11
  import "./selectiveSaveFlags.js";
12
12
  import { buildPatchedDocumentXml, buildPatchedNoteXml, collectAddedNumberingDefs, collectParaIds, patchNumberingDefinitions } from "./selectiveXmlPatch.js";
13
13
  import { planCommentParts, serializeComments, serializeCommentsExtended } from "./serializer/commentSerializer.js";
@@ -17,6 +17,7 @@ import { serializeNumberingXml } from "./serializer/numberingSerializer.js";
17
17
  import { readRootNamespaceBindings } from "./serializer/partNamespaces.js";
18
18
  import { buildStructuralDocumentPatch } from "./structuralXmlPatch.js";
19
19
  import { DOCX_CONFORMANCE_CLASSES } from "@stll/docx-core/model";
20
+ import { writeZipPart } from "@stll/docx-core/zip";
20
21
  //#region src/docx/selectiveSave.ts
21
22
  /**
22
23
  * Splice edited footnote/endnote paragraphs into their parts and return the
@@ -203,6 +204,7 @@ async function attemptSelectiveSave(doc, originalBuffer, options) {
203
204
  const commentPlan = planCommentParts(comments);
204
205
  try {
205
206
  const zip = await (await import("jszip")).default.loadAsync(originalBuffer);
207
+ if (await hasUnmaterializedHyperlinkBindings(doc, zip)) return null;
206
208
  for (const [path, file] of Object.entries(zip.files)) if (!file.dir && isUnsafePackagePath(path)) return null;
207
209
  if (originalBuffer !== doc.originalBuffer) {
208
210
  const sourceParts = /* @__PURE__ */ new Map();
@@ -211,14 +213,27 @@ async function attemptSelectiveSave(doc, originalBuffer, options) {
211
213
  if (file.dir || !lowerPath.startsWith("word/") || !lowerPath.endsWith(".xml")) continue;
212
214
  sourceParts.set(path, await file.async("text"));
213
215
  }
214
- for (const [path, xml] of normalizeImportedNumericIds(sourceParts)) if (xml !== sourceParts.get(path)) zip.file(path, xml);
216
+ for (const [path, xml] of normalizeImportedNumericIds(sourceParts)) if (xml !== sourceParts.get(path)) writeZipPart({
217
+ zip,
218
+ path,
219
+ data: xml
220
+ });
215
221
  }
222
+ if (options.bodyAuthority === "canonical") await publishCanonicalImageResources({
223
+ document: doc,
224
+ zip,
225
+ compressionLevel: 6
226
+ });
216
227
  const updates = /* @__PURE__ */ new Map();
217
228
  if (changedParaIds.size > 0 || structuralChange) {
218
229
  const docXmlFile = zip.file("word/document.xml");
219
230
  if (!docXmlFile) return null;
220
231
  const originalDocXml = await docXmlFile.async("text");
221
- const serializedDocXml = serializeDocument(doc, readRootNamespaceBindings(originalDocXml));
232
+ const serializedDocXml = serializeDocument(doc, readRootNamespaceBindings(originalDocXml), {
233
+ xml: originalDocXml,
234
+ bodyAuthority: options.bodyAuthority,
235
+ onDiagnostic: options.onDiagnostic
236
+ });
222
237
  const bodyParaIds = collectParaIds(serializedDocXml);
223
238
  const originalBodyParaIds = structuralChange ? collectParaIds(originalDocXml) : void 0;
224
239
  if (structuralChange && doc.package.numbering) {
@@ -268,6 +283,20 @@ async function attemptSelectiveSave(doc, originalBuffer, options) {
268
283
  }
269
284
  if (!await patchCommentsExtended(zip, commentPlan, updates)) return null;
270
285
  if (!await patchNumberingPart(zip, doc, updates)) return null;
286
+ const styles = await planAddedStyles(doc, zip);
287
+ switch (styles.type) {
288
+ case "unchanged": break;
289
+ case "patch":
290
+ updates.set(styles.path, styles.xml);
291
+ break;
292
+ case "materialize":
293
+ options.onDiagnostic?.({
294
+ type: "selectiveSaveRefused",
295
+ part: "word/styles.xml"
296
+ });
297
+ return null;
298
+ default: return styles;
299
+ }
271
300
  for (const [path, xml] of await collectHeaderFooterUpdates(doc, {
272
301
  sourceZip: zip,
273
302
  onDiagnostic: options.onDiagnostic
@@ -35,7 +35,7 @@ function isXmlNameBoundary(char) {
35
35
  */
36
36
  function findParagraphOffsets(xml, paraId) {
37
37
  const escaped = escapeRegExp(paraId);
38
- const pattern = new RegExp(`<w:p[\\s][^>]*w14:paraId="${escaped}"`, "gu");
38
+ const pattern = new RegExp(`<w:p[\\s][^>]*w14:paraId\\s*=\\s*(["'])${escaped}\\1`, "gu");
39
39
  const matches = [];
40
40
  let match;
41
41
  while ((match = pattern.exec(xml)) !== null) matches.push(match.index);
@@ -191,7 +191,7 @@ function scanParagraphs(xml) {
191
191
  const tagEnd = xml.indexOf(">", tagStart);
192
192
  if (tagEnd === -1) break;
193
193
  const openTag = xml.slice(tagStart, tagEnd + 1);
194
- const paraId = /\bw14:paraId="(?<id>[^"]+)"/u.exec(openTag)?.groups?.["id"];
194
+ const paraId = /\bw14:paraId\s*=\s*(?<quote>["'])(?<id>[^"']+)\k<quote>/u.exec(openTag)?.groups?.["id"];
195
195
  const container = {
196
196
  inFallback: depth.fallback > 0,
197
197
  inAlternateContent: depth.alternateContent > 0,
@@ -236,7 +236,7 @@ function countParagraphElements(xml) {
236
236
  */
237
237
  function collectParaIds(xml) {
238
238
  const ids = /* @__PURE__ */ new Map();
239
- const pattern = /w14:paraId="(?<id>[^"]+)"/gu;
239
+ const pattern = /w14:paraId\s*=\s*(?<quote>["'])(?<id>[^"']+)\k<quote>/gu;
240
240
  let match;
241
241
  while ((match = pattern.exec(xml)) !== null) {
242
242
  const id = match.groups["id"];
@@ -245,7 +245,7 @@ function collectParaIds(xml) {
245
245
  return ids;
246
246
  }
247
247
  /** Paragraph ids folio writes and this module may have to remove again. */
248
- const MINTED_PARA_ID_ATTRIBUTE = /\sw14:(?:para|text)Id="[^"]*"/gu;
248
+ const MINTED_PARA_ID_ATTRIBUTE = /\sw14:(paraId|textId)="[^"]*"/gu;
249
249
  /**
250
250
  * The replacement for a paragraph whose source open tag carries no paraId.
251
251
  *
@@ -262,7 +262,9 @@ const MINTED_PARA_ID_ATTRIBUTE = /\sw14:(?:para|text)Id="[^"]*"/gu;
262
262
  const withoutMintedIds = (paragraphXml, sourceOpenTag) => {
263
263
  const tagEnd = paragraphXml.indexOf(">");
264
264
  if (tagEnd === -1) return paragraphXml;
265
- return paragraphXml.slice(0, tagEnd + 1).replace(MINTED_PARA_ID_ATTRIBUTE, (attribute) => sourceOpenTag.includes(attribute.trim()) ? attribute : "") + paragraphXml.slice(tagEnd + 1);
265
+ return paragraphXml.slice(0, tagEnd + 1).replace(MINTED_PARA_ID_ATTRIBUTE, (attribute, name) => {
266
+ return new RegExp(`\\bw14:${name}\\s*=\\s*(["'])[^"']*\\1`, "u").test(sourceOpenTag) ? attribute : "";
267
+ }) + paragraphXml.slice(tagEnd + 1);
266
268
  };
267
269
  const storyOf = ({ textBoxDepth }) => textBoxDepth > 0 ? "text-box" : "main";
268
270
  const sameContainer = (a, b) => a.inAlternateContent === b.inAlternateContent && a.textBoxDepth === b.textBoxDepth && a.tableDepth === b.tableDepth;
@@ -1,4 +1,5 @@
1
1
  import { document_d_exports } from "../../types/document.js";
2
+ import { SaveDiagnosticOptions } from "../saveDiagnostics.js";
2
3
  //#region src/docx/serializer/documentSerializer.d.ts
3
4
  /**
4
5
  * Serialize a DocumentBody to document.xml body content
@@ -15,7 +16,13 @@ declare function serializeDocumentBody(body: document_d_exports.DocumentBody): s
15
16
  * prefix only the source document bound keeps its URI
16
17
  * @returns Complete XML string for document.xml
17
18
  */
18
- declare function serializeDocument(doc: document_d_exports.Document, sourceBindings?: ReadonlyMap<string, string>): string;
19
+ type DocumentBodyAuthorityOptions = {
20
+ /** Canonical saves preserve trusted source blocks; default saves serialize the model. */
21
+ bodyAuthority?: "canonical" | "model" | undefined;
22
+ };
23
+ declare function serializeDocument(doc: document_d_exports.Document, sourceBindings?: ReadonlyMap<string, string>, source?: SaveDiagnosticOptions & DocumentBodyAuthorityOptions & {
24
+ xml?: string | undefined;
25
+ }): string;
19
26
  /**
20
27
  * Serialize just the document body (useful for partial updates)
21
28
  *
@@ -59,4 +66,4 @@ declare function createSimpleDocument(paragraphs: {
59
66
  styleId?: string;
60
67
  }[]): document_d_exports.Document;
61
68
  //#endregion
62
- export { createEmptyDocument, createSimpleDocument, getDocumentContentCount, getDocumentParagraphCount, getDocumentTableCount, hasDocumentContent, hasDocumentSections, hasSectionProperties, serializeDocument, serializeDocumentBody, serializeDocumentBodyElement };
69
+ export { DocumentBodyAuthorityOptions, createEmptyDocument, createSimpleDocument, getDocumentContentCount, getDocumentParagraphCount, getDocumentTableCount, hasDocumentContent, hasDocumentSections, hasSectionProperties, serializeDocument, serializeDocumentBody, serializeDocumentBodyElement };
@@ -1,3 +1,5 @@
1
+ import { getDocumentSourceBaseline } from "../headerFooterVerbatim.js";
2
+ import { buildDocumentBlockReplay, canReplayDocumentSource } from "../storyBlockReplay.js";
1
3
  import { serializeBlockCustomXml } from "./blockCustomXmlSerializer.js";
2
4
  import { serializeBlockSdt } from "./blockSdtSerializer.js";
3
5
  import { serializeBookmarkMarker } from "./markupRangeAttributes.js";
@@ -84,24 +86,41 @@ function serializeDocumentBody(body) {
84
86
  if (body.finalSectionProperties) parts.push(serializeSectionProperties(body.finalSectionProperties));
85
87
  return parts.join("");
86
88
  }
87
- /**
88
- * Serialize a complete Document to valid document.xml
89
- *
90
- * @param doc - The document to serialize
91
- * @param sourceBindings - Root `xmlns:*` of the part being replaced, so a
92
- * prefix only the source document bound keeps its URI
93
- * @returns Complete XML string for document.xml
94
- */
95
- function serializeDocument(doc, sourceBindings) {
89
+ const replayableDocumentSources = /* @__PURE__ */ new WeakMap();
90
+ function serializeDocument(doc, sourceBindings, source) {
96
91
  resetAutoIdCounter();
92
+ const baseline = source?.bodyAuthority === "canonical" ? getDocumentSourceBaseline(doc) : { type: "missing" };
93
+ const sourceMatches = baseline.type === "captured" && (source?.xml === void 0 || source.xml === baseline.xml);
94
+ if (baseline.type === "captured" && sourceMatches && JSON.stringify(doc.package.document) === baseline.fingerprint) {
95
+ if (replayableDocumentSources.get(baseline.body) === baseline.xml) return baseline.xml;
96
+ if (canReplayDocumentSource(baseline.xml, baseline.body)) {
97
+ replayableDocumentSources.set(baseline.body, baseline.xml);
98
+ return baseline.xml;
99
+ }
100
+ }
97
101
  const body = serializeDocumentBackground(doc.package.document.background) + `<w:body>${serializeDocumentBody(doc.package.document)}</w:body>`;
98
- return "<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"yes\"?>" + serializePartElement({
102
+ const serializedXml = "<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"yes\"?>" + serializePartElement({
99
103
  partPath: "word/document.xml",
100
104
  rootName: "w:document",
101
105
  baselinePrefixes: DOCUMENT_BASELINE_PREFIXES,
102
106
  sourceBindings,
103
107
  body
104
108
  });
109
+ if (baseline.type === "missing") return serializedXml;
110
+ if (baseline.type === "captured" && sourceMatches) {
111
+ const replay = buildDocumentBlockReplay({
112
+ sourceXml: baseline.xml,
113
+ baseline: baseline.body,
114
+ current: doc.package.document,
115
+ serializedXml
116
+ });
117
+ if (replay !== null) return replay;
118
+ }
119
+ source?.onDiagnostic?.({
120
+ type: "sourceReplayMismatch",
121
+ part: "word/document.xml"
122
+ });
123
+ return serializedXml;
105
124
  }
106
125
  /**
107
126
  * Serialize just the document body (useful for partial updates)
@@ -3,6 +3,7 @@ import { FOLIO_DOCX_XML_PATCH_PROPOSAL_PROFILE, InvalidFolioDocxXmlPatchProposal
3
3
  import { FOLIO_DOCX_CONFORMANCE_PROFILE, validateDocxConformance } from "./validateDocxConformance.js";
4
4
  import { TaggedError, panic } from "better-result";
5
5
  import { createHash } from "node:crypto";
6
+ import { writeZipPart } from "@stll/docx-core/zip";
6
7
  import JSZip from "jszip";
7
8
  //#region src/docx/server/applyDocxXmlPatchProposal.ts
8
9
  const FOLIO_DOCX_XML_PATCH_APPLICATION_VERSION = 1;
@@ -92,15 +93,19 @@ const applyReplacements = async ({ bytes, proposal }) => {
92
93
  message: "An evaluated package part was unavailable during replacement.",
93
94
  stage: "replace"
94
95
  });
95
- zip.file(replacement.path, new TextEncoder().encode(replacement.replacementXml), {
96
- binary: true,
97
- date: current.date,
98
- comment: current.comment,
99
- createFolders: false,
100
- unixPermissions: current.unixPermissions,
101
- dosPermissions: current.dosPermissions,
102
- compression: "DEFLATE",
103
- compressionOptions: { level: 6 }
96
+ writeZipPart({
97
+ zip,
98
+ path: replacement.path,
99
+ data: new TextEncoder().encode(replacement.replacementXml),
100
+ options: {
101
+ binary: true,
102
+ date: current.date,
103
+ comment: current.comment,
104
+ unixPermissions: current.unixPermissions,
105
+ dosPermissions: current.dosPermissions,
106
+ compression: "DEFLATE",
107
+ compressionOptions: { level: 6 }
108
+ }
104
109
  });
105
110
  }
106
111
  try {
@@ -1,11 +1,31 @@
1
1
  import { document_d_exports } from "../types/document.js";
2
+ import { XmlElement } from "./xmlParser.js";
2
3
  //#region src/docx/storyBlockReplay.d.ts
4
+ type MaterializeReplayFragmentOptions = {
5
+ xml: string;
6
+ element: XmlElement;
7
+ sourceNamespace: string;
8
+ generatedRoot?: XmlElement;
9
+ };
10
+ /** Bind a replay fragment once, retaining existing local declarations and compatibility tokens. */
11
+ declare const materializeReplayFragment: ({ xml, element, sourceNamespace, generatedRoot }: MaterializeReplayFragmentOptions) => string | null;
3
12
  type StoryBlockReplayOptions = {
4
13
  sourceXml: string;
5
14
  baselineContent: readonly document_d_exports.BlockContent[];
6
15
  currentContent: readonly document_d_exports.BlockContent[];
7
16
  serializedXml: string;
17
+ ignoredChild?: string;
8
18
  };
9
19
  declare const buildStoryBlockReplay: (options: StoryBlockReplayOptions) => string | null;
20
+ /** Validate source/model correspondence before replaying an unchanged complete part. */
21
+ declare const canReplayDocumentSource: (sourceXml: string, baseline: document_d_exports.DocumentBody) => boolean;
22
+ type DocumentBlockReplayOptions = {
23
+ sourceXml: string;
24
+ serializedXml: string;
25
+ baseline: document_d_exports.DocumentBody;
26
+ current: document_d_exports.DocumentBody;
27
+ };
28
+ /** Replay body blocks and modeled document metadata inside the authored part shell. */
29
+ declare const buildDocumentBlockReplay: ({ sourceXml, serializedXml, baseline, current }: DocumentBlockReplayOptions) => string | null;
10
30
  //#endregion
11
- export { buildStoryBlockReplay };
31
+ export { buildDocumentBlockReplay, buildStoryBlockReplay, canReplayDocumentSource, materializeReplayFragment };
@@ -31,7 +31,7 @@ const modelMatchesElement = ({ value, element, mode }) => {
31
31
  };
32
32
  const XML_TOKEN = /<!--[\s\S]*?-->|<!\[CDATA\[[\s\S]*?\]\]>|<\?[\s\S]*?\?>|<(?:"[^"]*"|'[^']*'|[^'">])*>/gu;
33
33
  const XML_ATTRIBUTE = /([^\s=<>/]+)(\s*=\s*)(["'])(.*?)\3/gsu;
34
- const readStory = (xml, scope) => {
34
+ const readStory = (xml, scope, ignoredChild) => {
35
35
  const parsed = parseStreamingXmlWithSourceRanges(xml, scope);
36
36
  if (parsed.status !== "parsed") return null;
37
37
  const roots = getChildElements(parsed.value);
@@ -41,6 +41,7 @@ const readStory = (xml, scope) => {
41
41
  const children = getChildElements(root);
42
42
  const blocks = [];
43
43
  for (const element of children) {
44
+ if (getLocalName(element.name) === ignoredChild && WORDPROCESSINGML_NAMESPACE_URIS.has(getNamespaceUri(element) ?? "")) continue;
44
45
  const range = getXmlSourceRange(element);
45
46
  if (!range) return null;
46
47
  blocks.push({
@@ -64,7 +65,8 @@ const effectiveBindings = (element) => {
64
65
  }
65
66
  return bindings;
66
67
  };
67
- const generatedFragment = ({ xml, element, sourceNamespace, generatedRoot }) => {
68
+ /** Bind a replay fragment once, retaining existing local declarations and compatibility tokens. */
69
+ const materializeReplayFragment = ({ xml, element, sourceNamespace, generatedRoot }) => {
68
70
  const strict = TRANSITIONAL_NAMESPACE_BY_STRICT_URI.has(sourceNamespace);
69
71
  const first = [...xml.matchAll(XML_TOKEN)].at(0);
70
72
  if (!first || first.index !== 0 || first[0].startsWith("<!") || first[0].startsWith("<?")) return null;
@@ -121,9 +123,9 @@ const takeCandidate = (queue, used) => {
121
123
  }
122
124
  };
123
125
  /** Replace only mapped block ranges, retaining authored root syntax and every intervening gap. */
124
- const replayBlocks = ({ sourceXml, baselineContent, currentContent, serializedXml }, scope) => {
125
- const source = readStory(sourceXml, scope?.source);
126
- const generated = readStory(serializedXml, scope?.generated);
126
+ const replayBlocks = ({ sourceXml, baselineContent, currentContent, serializedXml, ignoredChild }, scope) => {
127
+ const source = readStory(sourceXml, scope?.source, ignoredChild);
128
+ const generated = readStory(serializedXml, scope?.generated, ignoredChild);
127
129
  if (!source || !generated || getLocalName(source.root.name) !== getLocalName(generated.root.name) || source.blocks.length !== baselineContent.length || generated.blocks.length !== currentContent.length) return null;
128
130
  if (source.blocks.some((block, index) => !modelMatchesElement({
129
131
  value: baselineContent[index],
@@ -185,7 +187,7 @@ const replayBlocks = ({ sourceXml, baselineContent, currentContent, serializedXm
185
187
  continue;
186
188
  }
187
189
  }
188
- const fragment = generatedFragment({
190
+ const fragment = materializeReplayFragment({
189
191
  xml: serializedXml.slice(replacement.start, replacement.end),
190
192
  element: replacement.element,
191
193
  sourceNamespace: getNamespaceUri(source.root) ?? "",
@@ -208,7 +210,8 @@ const replayBlocks = ({ sourceXml, baselineContent, currentContent, serializedXm
208
210
  newXml: `${opening.slice(0, -2)}>${fragments.join("")}</${source.root.name}>`
209
211
  });
210
212
  else {
211
- const close = sourceXml.lastIndexOf("</", source.range.end);
213
+ const ignored = ignoredChild === void 0 ? void 0 : getChildElements(source.root).find((child) => getLocalName(child.name) === ignoredChild && WORDPROCESSINGML_NAMESPACE_URIS.has(getNamespaceUri(child) ?? ""));
214
+ const close = (ignored && getXmlSourceRange(ignored)?.start) ?? sourceXml.lastIndexOf("</", source.range.end);
212
215
  if (close < source.range.start) return null;
213
216
  splices.push({
214
217
  start: close,
@@ -322,7 +325,7 @@ const replayCellContent = ({ sourceXml, serializedXml, sourceCell, generatedCell
322
325
  if (getChildElements(sourceCell).at(0) !== sourceProperty || !generatedProperty || getChildElements(generatedCell).at(0) !== generatedProperty) return false;
323
326
  const range = getXmlSourceRange(sourceProperty);
324
327
  if (!range) return false;
325
- const propertyXml = generatedFragment({
328
+ const propertyXml = materializeReplayFragment({
326
329
  xml: sourceXml.slice(range.start, range.end),
327
330
  element: sourceProperty,
328
331
  sourceNamespace: getNamespaceUri(sourceCell) ?? ""
@@ -363,5 +366,106 @@ const replayCellContent = ({ sourceXml, serializedXml, sourceCell, generatedCell
363
366
  });
364
367
  return true;
365
368
  };
369
+ const readDocumentStory = (xml) => {
370
+ const document = readStory(xml);
371
+ if (!document || getLocalName(document.root.name) !== "document") return null;
372
+ const bodies = wordChildren(document.root, "body");
373
+ const body = bodies.at(0);
374
+ if (bodies.length !== 1 || !body || wordChildren(body, "sectPr").length > 1 || wordChildren(document.root, "background").length > 1) return null;
375
+ const range = getXmlSourceRange(body);
376
+ return range ? {
377
+ document,
378
+ body,
379
+ range
380
+ } : null;
381
+ };
382
+ /** Validate source/model correspondence before replaying an unchanged complete part. */
383
+ const canReplayDocumentSource = (sourceXml, baseline) => {
384
+ const source = readDocumentStory(sourceXml);
385
+ if (!source) return false;
386
+ const body = readStory(sourceXml.slice(source.range.start, source.range.end), source.body.namespaceScope, "sectPr");
387
+ return body !== null && body.blocks.length === baseline.content.length && body.blocks.every((block, index) => modelMatchesElement({
388
+ value: baseline.content[index],
389
+ element: block.element,
390
+ mode: "source"
391
+ }));
392
+ };
393
+ /** Replay body blocks and modeled document metadata inside the authored part shell. */
394
+ const buildDocumentBlockReplay = ({ sourceXml, serializedXml, baseline, current }) => {
395
+ const sourceStory = readDocumentStory(sourceXml);
396
+ const generatedStory = readDocumentStory(serializedXml);
397
+ if (!sourceStory || !generatedStory) return null;
398
+ const { body: sourceBody, range: sourceRange } = sourceStory;
399
+ const { document: generated, body: generatedBody, range: generatedRange } = generatedStory;
400
+ let bodyXml = replayBlocks({
401
+ sourceXml: sourceXml.slice(sourceRange.start, sourceRange.end),
402
+ serializedXml: serializedXml.slice(generatedRange.start, generatedRange.end),
403
+ baselineContent: baseline.content,
404
+ currentContent: current.content,
405
+ ignoredChild: "sectPr"
406
+ }, {
407
+ source: sourceBody.namespaceScope,
408
+ generated: generatedBody.namespaceScope,
409
+ generatedRoot: generated.root
410
+ });
411
+ if (bodyXml === null) return null;
412
+ const replaceMetadata = (xml, sourceContainer, generatedContainer, name) => {
413
+ const before = wordChildren(sourceContainer, name);
414
+ const after = wordChildren(generatedContainer, name);
415
+ if (before.length > 1 || after.length > 1) return null;
416
+ const old = before.at(0);
417
+ const next = after.at(0);
418
+ const containerRange = getXmlSourceRange(sourceContainer);
419
+ const oldRange = old && getXmlSourceRange(old);
420
+ const nextRange = next && getXmlSourceRange(next);
421
+ if (!containerRange || old && !oldRange || next && !nextRange) return null;
422
+ const replacement = next && nextRange ? materializeReplayFragment({
423
+ xml: serializedXml.slice(nextRange.start, nextRange.end),
424
+ element: next,
425
+ sourceNamespace: getNamespaceUri(sourceContainer) ?? "",
426
+ generatedRoot: generated.root
427
+ }) : "";
428
+ if (replacement === null) return null;
429
+ if (oldRange) return spliceXml(xml, [{
430
+ start: oldRange.start - containerRange.start,
431
+ end: oldRange.end - containerRange.start,
432
+ newXml: replacement
433
+ }]);
434
+ if (!next) return xml;
435
+ const insertion = name === "background" ? [...xml.matchAll(XML_TOKEN)].at(0)?.[0].length ?? -1 : xml.lastIndexOf("</");
436
+ if (insertion < 0) return null;
437
+ return spliceXml(xml, [{
438
+ start: insertion,
439
+ end: insertion,
440
+ newXml: replacement
441
+ }]);
442
+ };
443
+ if (canonicalJson(baseline.finalSectionProperties) !== canonicalJson(current.finalSectionProperties)) {
444
+ const reparsed = parseStreamingXmlWithSourceRanges(bodyXml, sourceBody.namespaceScope);
445
+ if (reparsed.status !== "parsed") return null;
446
+ const root = getChildElements(reparsed.value).at(0);
447
+ if (!root) return null;
448
+ bodyXml = replaceMetadata(bodyXml, root, generatedBody, "sectPr");
449
+ if (bodyXml === null) return null;
450
+ }
451
+ let result = spliceXml(sourceXml, [{
452
+ start: sourceRange.start,
453
+ end: sourceRange.end,
454
+ newXml: bodyXml
455
+ }]);
456
+ if (result === null) return null;
457
+ if (canonicalJson(baseline.background) !== canonicalJson(current.background)) {
458
+ const reparsed = readStory(result);
459
+ if (!reparsed) return null;
460
+ const replaced = replaceMetadata(result.slice(reparsed.range.start, reparsed.range.end), reparsed.root, generated.root, "background");
461
+ if (replaced === null) return null;
462
+ result = spliceXml(result, [{
463
+ start: reparsed.range.start,
464
+ end: reparsed.range.end,
465
+ newXml: replaced
466
+ }]);
467
+ }
468
+ return result;
469
+ };
366
470
  //#endregion
367
- export { buildStoryBlockReplay };
471
+ export { buildDocumentBlockReplay, buildStoryBlockReplay, canReplayDocumentSource, materializeReplayFragment };
@@ -0,0 +1,20 @@
1
+ import { document_d_exports } from "../types/document.js";
2
+ //#region src/docx/storyPlainText.d.ts
3
+ /** Read the document body with pending revisions accepted. */
4
+ declare const getDocumentText: (body: document_d_exports.DocumentBody) => string;
5
+ /** Approximate word count of the accepted body text. */
6
+ declare const getWordCount: (body: document_d_exports.DocumentBody) => number;
7
+ /** Character count of the accepted body text. */
8
+ declare const getCharacterCount: (body: document_d_exports.DocumentBody) => number;
9
+ /** Read an accepted table, including nested blocks and tracked structure. */
10
+ declare const getTableText: (table: document_d_exports.Table) => string;
11
+ /** Read accepted text-box content for search and indexing. */
12
+ declare const getTextBoxText: (textBox: document_d_exports.TextBox) => string;
13
+ /** Read an accepted header or footer through the same walk as body and notes. */
14
+ declare const getHeaderFooterText: (headerFooter: document_d_exports.HeaderFooter) => string;
15
+ /** Read accepted footnote content, including nested blocks. */
16
+ declare const getFootnoteText: (footnote: document_d_exports.Footnote) => string;
17
+ /** Read accepted endnote content, including nested blocks. */
18
+ declare const getEndnoteText: (endnote: document_d_exports.Endnote) => string;
19
+ //#endregion
20
+ export { getCharacterCount, getDocumentText, getEndnoteText, getFootnoteText, getHeaderFooterText, getTableText, getTextBoxText, getWordCount };
@@ -0,0 +1,23 @@
1
+ import { blockPlainText } from "./blockPlainText.js";
2
+ //#region src/docx/storyPlainText.ts
3
+ /** Read the document body with pending revisions accepted. */
4
+ const getDocumentText = (body) => blockPlainText(body.content);
5
+ /** Approximate word count of the accepted body text. */
6
+ const getWordCount = (body) => {
7
+ const words = getDocumentText(body).trim().split(/\s+/u);
8
+ return words.length > 0 && words[0] !== "" ? words.length : 0;
9
+ };
10
+ /** Character count of the accepted body text. */
11
+ const getCharacterCount = (body) => getDocumentText(body).length;
12
+ /** Read an accepted table, including nested blocks and tracked structure. */
13
+ const getTableText = (table) => blockPlainText([table]);
14
+ /** Read accepted text-box content for search and indexing. */
15
+ const getTextBoxText = (textBox) => blockPlainText(textBox.content);
16
+ /** Read an accepted header or footer through the same walk as body and notes. */
17
+ const getHeaderFooterText = (headerFooter) => blockPlainText(headerFooter.content);
18
+ /** Read accepted footnote content, including nested blocks. */
19
+ const getFootnoteText = (footnote) => blockPlainText(footnote.content);
20
+ /** Read accepted endnote content, including nested blocks. */
21
+ const getEndnoteText = (endnote) => blockPlainText(endnote.content);
22
+ //#endregion
23
+ export { getCharacterCount, getDocumentText, getEndnoteText, getFootnoteText, getHeaderFooterText, getTableText, getTextBoxText, getWordCount };
@@ -2,7 +2,7 @@ import { canonicalJson } from "../utils/canonicalJson.js";
2
2
  import { parseDocumentBody } from "./documentParser.js";
3
3
  import { paraIdAttribute } from "./paraIdAttribute.js";
4
4
  import { spliceSupportsRootNamespaces, spliceXml } from "./selectiveXmlPatch.js";
5
- import { readRootNamespaceBindings, serializePartElement } from "./serializer/partNamespaces.js";
5
+ import { materializeReplayFragment } from "./storyBlockReplay.js";
6
6
  import { NAMESPACES, WORDPROCESSINGML_NAMESPACE_URIS, getChildElements, getLocalName, getNamespaceUri, parseXmlDocument } from "./xmlParser.js";
7
7
  //#region src/docx/structuralXmlPatch.ts
8
8
  /** Conservative body-paragraph splices; source offsets never come from reserialization. */
@@ -111,21 +111,6 @@ const paragraphIds = (root) => {
111
111
  }
112
112
  return ids;
113
113
  };
114
- /** Bind fragment prefixes locally: the source root may use entirely different aliases. */
115
- const paragraphFragment = ({ xml, block, bindings }) => {
116
- const fragment = xml.slice(block.start, block.end);
117
- const openEnd = fragment.indexOf(">");
118
- const name = block.element.name ?? "w:p";
119
- const selfClosing = fragment[openEnd - 1] === "/";
120
- return serializePartElement({
121
- partPath: "word/document.xml",
122
- rootName: name,
123
- rootAttributes: fragment.slice(name.length + 1, selfClosing ? openEnd - 1 : openEnd).trim(),
124
- baselinePrefixes: [],
125
- sourceBindings: bindings,
126
- body: selfClosing ? "" : fragment.slice(openEnd + 1, fragment.lastIndexOf("</"))
127
- });
128
- };
129
114
  /**
130
115
  * Insert/delete direct body paragraphs between surviving paragraph/table anchors.
131
116
  * Tables are opaque barriers: their modeled content and order must stay identical.
@@ -168,11 +153,11 @@ const buildStructuralDocumentPatch = ({ originalXml, serializedXml, changedIds }
168
153
  const survivingBefore = [...before.keys()].filter((key) => after.has(key));
169
154
  const survivingAfter = [...after.keys()].filter((key) => before.has(key));
170
155
  if (canonicalJson(survivingBefore) !== canonicalJson(survivingAfter)) return null;
171
- const bindings = readRootNamespaceBindings(serializedXml);
172
- const fragmentFor = (block) => paragraphFragment({
173
- xml: serializedXml,
174
- block,
175
- bindings
156
+ const fragmentFor = (block) => materializeReplayFragment({
157
+ xml: serializedXml.slice(block.start, block.end),
158
+ element: block.element,
159
+ sourceNamespace: getNamespaceUri(source.root) ?? "",
160
+ generatedRoot: current.root
176
161
  });
177
162
  const splices = [];
178
163
  for (const [key, block] of before) if (!after.has(key)) splices.push({
@@ -186,11 +171,14 @@ const buildStructuralDocumentPatch = ({ originalXml, serializedXml, changedIds }
186
171
  if (!original) {
187
172
  const id = paraIdAttribute(block.element)?.toUpperCase();
188
173
  if (!id || allSourceIds.has(id)) return null;
189
- pending.push(fragmentFor(block));
174
+ const fragment = fragmentFor(block);
175
+ if (fragment === null) return null;
176
+ pending.push(fragment);
190
177
  continue;
191
178
  }
192
179
  const id = paraIdAttribute(block.element)?.toUpperCase();
193
180
  const replacement = id !== void 0 && changed.has(id) ? fragmentFor(block) : originalXml.slice(original.start, original.end);
181
+ if (replacement === null) return null;
194
182
  if (pending.length > 0 || replacement !== originalXml.slice(original.start, original.end)) splices.push({
195
183
  start: original.start,
196
184
  end: original.end,
@@ -8,6 +8,10 @@ type ParsedStylesPackage = {
8
8
  styleDefinitions: document_d_exports.StyleDefinitions;
9
9
  styles: StyleMap;
10
10
  };
11
+ /**
12
+ * Resolve style inheritance chain
13
+ */
14
+ declare function resolveStyleInheritance(style: document_d_exports.Style, styleMap: StyleMap, visited?: Set<string>): document_d_exports.Style;
11
15
  /**
12
16
  * Parse styles.xml content
13
17
  *
@@ -46,4 +50,4 @@ declare function getDefaultCharacterStyle(styleMap: StyleMap): document_d_export
46
50
  */
47
51
  declare function getStylesByType(styleMap: StyleMap, type: document_d_exports.StyleType): document_d_exports.Style[];
48
52
  //#endregion
49
- export { ParsedStylesPackage, StyleMap, getDefaultCharacterStyle, getDefaultParagraphStyle, getResolvedStyle, getStylesByType, parseStyleDefinitions, parseStyles, parseStylesPackage };
53
+ export { ParsedStylesPackage, StyleMap, getDefaultCharacterStyle, getDefaultParagraphStyle, getResolvedStyle, getStylesByType, parseStyleDefinitions, parseStyles, parseStylesPackage, resolveStyleInheritance };
@@ -554,4 +554,4 @@ function getStylesByType(styleMap, type) {
554
554
  return result;
555
555
  }
556
556
  //#endregion
557
- export { getDefaultCharacterStyle, getDefaultParagraphStyle, getResolvedStyle, getStylesByType, parseStyleDefinitions, parseStyles, parseStylesPackage };
557
+ export { getDefaultCharacterStyle, getDefaultParagraphStyle, getResolvedStyle, getStylesByType, parseStyleDefinitions, parseStyles, parseStylesPackage, resolveStyleInheritance };
@@ -342,13 +342,6 @@ declare function isCellMergeStart(cell: document_d_exports.TableCell): boolean;
342
342
  * @returns true if cell spans multiple columns
343
343
  */
344
344
  declare function isCellHorizontallyMerged(cell: document_d_exports.TableCell): boolean;
345
- /**
346
- * Get the plain text content of a table
347
- *
348
- * @param table - The table to extract text from
349
- * @returns Plain text content
350
- */
351
- declare function getTableText(table: document_d_exports.Table): string;
352
345
  /**
353
346
  * Check if table has header row
354
347
  *
@@ -371,4 +364,4 @@ declare function getHeaderRows(table: document_d_exports.Table): document_d_expo
371
364
  */
372
365
  declare function isFloatingTable(table: document_d_exports.Table): boolean;
373
366
  //#endregion
374
- export { CELL_CONTENT_HANDLERS, MAX_TABLE_COLUMNS, ROW_CONTENT_HANDLERS, TABLE_CONTENT_HANDLERS, getHeaderRows, getTableColumnCount, getTableRowCount, getTableText, hasHeaderRow, isCellHorizontallyMerged, isCellMergeContinuation, isCellMergeStart, isFloatingTable, isTableCellMergeRevisionContinuation, isTableCellMergeRevisionValue, parseCellMargins, parseConditionalFormatStyle, parseFloatingTableProperties, parseTable, parseTableBorders, parseTableCell, parseTableCellProperties, parseTableGrid, parseTableLook, parseTableMeasurement, parseTableProperties, parseTablePropertyExceptions, parseTableRow, parseTableRowProperties };
367
+ export { CELL_CONTENT_HANDLERS, MAX_TABLE_COLUMNS, ROW_CONTENT_HANDLERS, TABLE_CONTENT_HANDLERS, getHeaderRows, getTableColumnCount, getTableRowCount, hasHeaderRow, isCellHorizontallyMerged, isCellMergeContinuation, isCellMergeStart, isFloatingTable, isTableCellMergeRevisionContinuation, isTableCellMergeRevisionValue, parseCellMargins, parseConditionalFormatStyle, parseFloatingTableProperties, parseTable, parseTableBorders, parseTableCell, parseTableCellProperties, parseTableGrid, parseTableLook, parseTableMeasurement, parseTableProperties, parseTablePropertyExceptions, parseTableRow, parseTableRowProperties };