docxodus 12.3.0 → 12.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. package/README.md +31 -6
  2. package/dist/core.d.ts +924 -0
  3. package/dist/core.d.ts.map +1 -0
  4. package/dist/core.js +2094 -0
  5. package/dist/core.js.map +1 -0
  6. package/dist/editor.bundle.js +1206 -27
  7. package/dist/editor.d.ts +10 -1
  8. package/dist/editor.d.ts.map +1 -1
  9. package/dist/editor.js +48 -16
  10. package/dist/editor.js.map +1 -1
  11. package/dist/embed.bundle.js +12371 -11779
  12. package/dist/embed.d.ts +5 -2
  13. package/dist/embed.d.ts.map +1 -1
  14. package/dist/embed.iife.js +12282 -11690
  15. package/dist/embed.js +10 -3
  16. package/dist/embed.js.map +1 -1
  17. package/dist/export-assets.json +39 -39
  18. package/dist/history-checkpoints.d.ts +2 -0
  19. package/dist/history-checkpoints.d.ts.map +1 -1
  20. package/dist/history-checkpoints.js +2 -0
  21. package/dist/history-checkpoints.js.map +1 -1
  22. package/dist/history-controls.d.ts +6 -3
  23. package/dist/history-controls.d.ts.map +1 -1
  24. package/dist/history-controls.js +71 -29
  25. package/dist/history-controls.js.map +1 -1
  26. package/dist/index.d.ts +3 -923
  27. package/dist/index.d.ts.map +1 -1
  28. package/dist/index.js +2 -2093
  29. package/dist/index.js.map +1 -1
  30. package/dist/ribbon-chrome.d.ts +1 -1
  31. package/dist/ribbon-chrome.d.ts.map +1 -1
  32. package/dist/ribbon-chrome.js +6 -3
  33. package/dist/ribbon-chrome.js.map +1 -1
  34. package/dist/ribbon-history.d.ts +85 -0
  35. package/dist/ribbon-history.d.ts.map +1 -0
  36. package/dist/ribbon-history.js +508 -0
  37. package/dist/ribbon-history.js.map +1 -0
  38. package/dist/ribbon.d.ts +6 -0
  39. package/dist/ribbon.d.ts.map +1 -1
  40. package/dist/ribbon.js +47 -18
  41. package/dist/ribbon.js.map +1 -1
  42. package/dist/viewport.d.ts +3 -1
  43. package/dist/viewport.d.ts.map +1 -1
  44. package/dist/viewport.js +14 -0
  45. package/dist/viewport.js.map +1 -1
  46. package/dist/wasm/_framework/Docxodus.wasm +0 -0
  47. package/dist/wasm/_framework/Docxodus.wasm.br +0 -0
  48. package/dist/wasm/_framework/DocxodusWasm.wasm +0 -0
  49. package/dist/wasm/_framework/DocxodusWasm.wasm.br +0 -0
  50. package/dist/wasm/_framework/System.Collections.Concurrent.wasm +0 -0
  51. package/dist/wasm/_framework/System.Collections.Concurrent.wasm.br +0 -0
  52. package/dist/wasm/_framework/System.Collections.Immutable.wasm +0 -0
  53. package/dist/wasm/_framework/System.Collections.Immutable.wasm.br +0 -0
  54. package/dist/wasm/_framework/System.Collections.NonGeneric.wasm +0 -0
  55. package/dist/wasm/_framework/System.Collections.NonGeneric.wasm.br +0 -0
  56. package/dist/wasm/_framework/System.Collections.Specialized.wasm +0 -0
  57. package/dist/wasm/_framework/System.Collections.Specialized.wasm.br +0 -0
  58. package/dist/wasm/_framework/System.Collections.wasm +0 -0
  59. package/dist/wasm/_framework/System.Collections.wasm.br +0 -0
  60. package/dist/wasm/_framework/System.ComponentModel.Primitives.wasm +0 -0
  61. package/dist/wasm/_framework/System.ComponentModel.Primitives.wasm.br +0 -0
  62. package/dist/wasm/_framework/System.ComponentModel.TypeConverter.wasm +0 -0
  63. package/dist/wasm/_framework/System.ComponentModel.TypeConverter.wasm.br +0 -0
  64. package/dist/wasm/_framework/System.ComponentModel.wasm +0 -0
  65. package/dist/wasm/_framework/System.ComponentModel.wasm.br +0 -0
  66. package/dist/wasm/_framework/System.Console.wasm +0 -0
  67. package/dist/wasm/_framework/System.Console.wasm.br +0 -0
  68. package/dist/wasm/_framework/System.IO.Compression.wasm +0 -0
  69. package/dist/wasm/_framework/System.IO.Compression.wasm.br +0 -0
  70. package/dist/wasm/_framework/System.IO.Pipelines.wasm +0 -0
  71. package/dist/wasm/_framework/System.IO.Pipelines.wasm.br +0 -0
  72. package/dist/wasm/_framework/System.Linq.Expressions.wasm +0 -0
  73. package/dist/wasm/_framework/System.Linq.Expressions.wasm.br +0 -0
  74. package/dist/wasm/_framework/System.Linq.wasm +0 -0
  75. package/dist/wasm/_framework/System.Linq.wasm.br +0 -0
  76. package/dist/wasm/_framework/System.Memory.wasm +0 -0
  77. package/dist/wasm/_framework/System.Memory.wasm.br +0 -0
  78. package/dist/wasm/_framework/System.Net.Http.wasm +0 -0
  79. package/dist/wasm/_framework/System.Net.Http.wasm.br +0 -0
  80. package/dist/wasm/_framework/System.Net.Primitives.wasm +0 -0
  81. package/dist/wasm/_framework/System.Net.Primitives.wasm.br +0 -0
  82. package/dist/wasm/_framework/System.ObjectModel.wasm +0 -0
  83. package/dist/wasm/_framework/System.ObjectModel.wasm.br +0 -0
  84. package/dist/wasm/_framework/System.Private.CoreLib.wasm +0 -0
  85. package/dist/wasm/_framework/System.Private.CoreLib.wasm.br +0 -0
  86. package/dist/wasm/_framework/System.Private.Uri.wasm +0 -0
  87. package/dist/wasm/_framework/System.Private.Uri.wasm.br +0 -0
  88. package/dist/wasm/_framework/System.Private.Xml.Linq.wasm +0 -0
  89. package/dist/wasm/_framework/System.Private.Xml.Linq.wasm.br +0 -0
  90. package/dist/wasm/_framework/System.Private.Xml.wasm +0 -0
  91. package/dist/wasm/_framework/System.Private.Xml.wasm.br +0 -0
  92. package/dist/wasm/_framework/System.Runtime.InteropServices.JavaScript.wasm +0 -0
  93. package/dist/wasm/_framework/System.Runtime.InteropServices.JavaScript.wasm.br +0 -0
  94. package/dist/wasm/_framework/System.Runtime.wasm +0 -0
  95. package/dist/wasm/_framework/System.Runtime.wasm.br +0 -0
  96. package/dist/wasm/_framework/System.Security.Cryptography.wasm +0 -0
  97. package/dist/wasm/_framework/System.Security.Cryptography.wasm.br +0 -0
  98. package/dist/wasm/_framework/System.Text.Encodings.Web.wasm +0 -0
  99. package/dist/wasm/_framework/System.Text.Encodings.Web.wasm.br +0 -0
  100. package/dist/wasm/_framework/System.Text.Json.wasm +0 -0
  101. package/dist/wasm/_framework/System.Text.Json.wasm.br +0 -0
  102. package/dist/wasm/_framework/System.Text.RegularExpressions.wasm +0 -0
  103. package/dist/wasm/_framework/System.Text.RegularExpressions.wasm.br +0 -0
  104. package/dist/wasm/_framework/System.Xml.Linq.wasm +0 -0
  105. package/dist/wasm/_framework/System.Xml.Linq.wasm.br +0 -0
  106. package/dist/wasm/_framework/System.Xml.XDocument.wasm +0 -0
  107. package/dist/wasm/_framework/System.Xml.XDocument.wasm.br +0 -0
  108. package/dist/wasm/_framework/System.wasm +0 -0
  109. package/dist/wasm/_framework/System.wasm.br +0 -0
  110. package/dist/wasm/_framework/dotnet.boot.js +34 -34
  111. package/dist/wasm/_framework/dotnet.boot.js.br +0 -0
  112. package/dist/wasm/_framework/dotnet.js +1 -1
  113. package/dist/wasm/_framework/dotnet.js.br +0 -0
  114. package/dist/wasm/_framework/dotnet.native.js +3 -3
  115. package/dist/wasm/_framework/dotnet.native.js.br +0 -0
  116. package/dist/wasm/_framework/dotnet.native.wasm +0 -0
  117. package/dist/wasm/_framework/dotnet.native.wasm.br +0 -0
  118. package/dist/wasm/_framework/dotnet.runtime.js +1 -1
  119. package/dist/wasm/_framework/dotnet.runtime.js.br +0 -0
  120. package/package.json +14 -8
package/dist/index.js CHANGED
@@ -1,50 +1,5 @@
1
- import { openDocxSession as openDocxSessionImpl } from "./session.js";
2
- import { DocxHistoryArchive, DocxHistoryClient, installHistoryStorageImports } from './history.js';
3
- export * from './history.js';
4
- export * from './history-checkpoints.js';
5
- export * from './history-indexeddb.js';
6
- export * from './history-controls.js';
7
- /** Open history over host-owned storage after initialize(). No network layer is installed. */
8
- export function openDocxHistory(storage) {
9
- const bridge = ensureInitialized().HistoryBridge;
10
- if (!bridge)
11
- throw new Error('This WASM build does not include history bindings.');
12
- return new DocxHistoryClient(bridge, storage);
13
- }
14
- /** Open a self-contained readonly .docxhistory file after initialize(); no storage adapter needed. */
15
- export function openDocxHistoryArchive(bytes) {
16
- const bridge = ensureInitialized().HistoryBridge;
17
- if (!bridge)
18
- throw new Error('This WASM build does not include history bindings.');
19
- return DocxHistoryArchive.open(bridge, bytes);
20
- }
21
- export { DocxSession } from "./session.js";
22
- export { PlaceholderKinds, ContextBoundary } from "./types.js";
23
- export { DiffFormat } from "./types.js";
24
- /**
25
- * Open a {@link DocxSession} for surgical mutation of a DOCX. Requires
26
- * {@link initialize} to have been called and awaited.
27
- *
28
- * The returned session keeps the document in WASM memory; call
29
- * {@link DocxSession.close} when done.
30
- */
31
- export function openDocxSession(bytes, settings) {
32
- const wasm = ensureInitialized();
33
- return openDocxSessionImpl(bytes, wasm, settings);
34
- }
35
- /**
36
- * Mint a complete, blank single-paragraph DOCX (Normal style, US-Letter section) as bytes —
37
- * a "New document" seed for editors that draft from scratch. Requires {@link initialize}.
38
- */
39
- export function createBlankDocx() {
40
- return ensureInitialized().DocxSessionBridge.CreateBlankDocx();
41
- }
42
- import { CommentRenderMode, PaginationMode, AnnotationLabelMode, RevisionType, DocxDiffRevisionGranularity, DocxDiffFormatComparison, ConflictResolution, ProjectionScopes, AnchorRenderMode, TableRenderMode, TrackedChangeMode, EmptyParagraphMode, AnchorIdRendering, ProjectionDepth, DocumentElementType, ComparisonLogLevel, ComparisonLogCodes, isInsertion, isDeletion, isMove, isFormatChange, findElementById, findElementsByType, getParagraphs, getTables, getTableColumns, targetElement, targetParagraph, targetParagraphRange, targetRun, targetTable, targetTableRow, targetTableCell, targetTableColumn, targetSearch, targetSearchInElement, } from "./types.js";
43
- export { PaginationEngine, clearPageCitationHighlight, createUnavailablePageMap, navigateToPageCitation, paginateHtml, } from "./pagination.js";
44
- // Page geometry is the document's own page setup (w:sectPr), read off the section wrappers
45
- // the converter stamps in every render mode, plus the fit-to-width zoom a view applies to it.
46
- export { DEFAULT_MARGIN, DEFAULT_PAGE_HEIGHT, DEFAULT_PAGE_WIDTH, fitScale, parseSectionDimensions, ptToPx, pxToPt, } from "./page-geometry.js";
47
- export { DocumentViewport } from "./viewport.js";
1
+ /** Full browser API. Use docxodus/core for the engine without editor dependencies. */
2
+ export * from "./core.js";
48
3
  export { DocxEditor } from "./editor.js";
49
4
  // The comment gutter and header/footer region are the editor's own; exported for hosts that
50
5
  // build their own chrome and want to drive them directly.
@@ -53,2050 +8,4 @@ export { CommentGutter } from "./editor-comments.js";
53
8
  // loading overlay wired onto DocxEditor's command surface. `createRibbonEditor`
54
9
  // (docxodus/embed) is the one-call version that also boots WASM.
55
10
  export { mountRibbon } from "./ribbon.js";
56
- export { CommentRenderMode, PaginationMode, AnnotationLabelMode, RevisionType, DocxDiffRevisionGranularity, DocxDiffFormatComparison, ConflictResolution, ProjectionScopes, AnchorRenderMode, TableRenderMode, TrackedChangeMode, EmptyParagraphMode, AnchorIdRendering, ProjectionDepth, DocumentElementType, ComparisonLogLevel, ComparisonLogCodes, isInsertion, isDeletion, isMove, isFormatChange,
57
- // Document structure helpers
58
- findElementById, findElementsByType, getParagraphs, getTables, getTableColumns,
59
- // Annotation target factory functions
60
- targetElement, targetParagraph, targetParagraphRange, targetRun, targetTable, targetTableRow, targetTableCell, targetTableColumn, targetSearch, targetSearchInElement, };
61
- let wasmExports = null;
62
- let initPromise = null;
63
- /**
64
- * Yields to the browser's main thread, allowing pending UI updates to render.
65
- *
66
- * This is critical for WASM operations: since WASM runs synchronously on the
67
- * main thread, React state updates (like loading spinners) won't paint unless
68
- * we yield before the blocking work begins.
69
- *
70
- * Uses requestAnimationFrame which fires just before the next paint, ensuring
71
- * any queued state updates are committed to the DOM.
72
- *
73
- * @internal
74
- */
75
- async function yieldToMain() {
76
- // In non-browser environments (SSR, tests), skip yielding
77
- if (typeof requestAnimationFrame === "undefined") {
78
- return;
79
- }
80
- // Double-rAF ensures the browser has fully painted before we continue
81
- // First rAF: scheduled for next frame
82
- // Second rAF: ensures first frame actually painted
83
- await new Promise((resolve) => {
84
- requestAnimationFrame(() => {
85
- requestAnimationFrame(() => resolve());
86
- });
87
- });
88
- }
89
- /**
90
- * Derive the WASM base path from this module's URL.
91
- * Works whether loaded from node_modules, CDN, or bundled.
92
- */
93
- function getDefaultWasmBasePath() {
94
- try {
95
- // import.meta.url gives us the URL of this module
96
- // e.g., "https://cdn.jsdelivr.net/npm/docxodus@12.2.0/dist/index.js"
97
- // or "file:///path/to/node_modules/docxodus/dist/index.js"
98
- const moduleUrl = import.meta.url;
99
- // Remove the filename to get the directory
100
- const baseDir = moduleUrl.substring(0, moduleUrl.lastIndexOf('/') + 1);
101
- // WASM files are in ./wasm/ relative to dist/
102
- return baseDir + "wasm/";
103
- }
104
- catch {
105
- // Fallback if import.meta.url is not available
106
- return "";
107
- }
108
- }
109
- /**
110
- * Current base path for WASM files.
111
- * Empty string means auto-detect from module URL.
112
- */
113
- export let wasmBasePath = "";
114
- /**
115
- * Set custom base path for WASM files.
116
- * Pass empty string or don't call this to auto-detect from module location.
117
- *
118
- * @param path - Custom path to WASM files, or empty string for auto-detection
119
- */
120
- export function setWasmBasePath(path) {
121
- wasmBasePath = path && !path.endsWith("/") ? path + "/" : path;
122
- }
123
- /**
124
- * Initialize the Docxodus WASM runtime.
125
- * Must be called before using any conversion/comparison functions.
126
- * Safe to call multiple times - will only initialize once.
127
- *
128
- * By default, WASM files are auto-detected from the module's location
129
- * (works with CDN, npm, or local hosting).
130
- * Pass a basePath to load from a custom location instead.
131
- *
132
- * @param basePath - Optional custom path to WASM files. Leave empty for auto-detection.
133
- */
134
- export async function initialize(basePath) {
135
- if (wasmExports)
136
- return;
137
- if (initPromise) {
138
- return initPromise;
139
- }
140
- if (basePath !== undefined) {
141
- setWasmBasePath(basePath);
142
- }
143
- // Clear the cached promise on failure so a caller can retry with a
144
- // different base path (a rejected initialize() used to be permanent).
145
- initPromise = loadWasm().catch((e) => {
146
- initPromise = null;
147
- throw e;
148
- });
149
- return initPromise;
150
- }
151
- /**
152
- * Try to load WASM from a specific base path
153
- */
154
- async function tryLoadFromPath(basePath) {
155
- try {
156
- const dotnetPath = basePath + "_framework/dotnet.js";
157
- const { dotnet } = await import(/* webpackIgnore: true */ /* @vite-ignore */ dotnetPath);
158
- const { getAssemblyExports, getConfig, setModuleImports } = await dotnet
159
- .withDiagnosticTracing(false)
160
- .create();
161
- installHistoryStorageImports(setModuleImports);
162
- const config = getConfig();
163
- const exports = await getAssemblyExports(config.mainAssemblyName);
164
- wasmExports = {
165
- DocumentConverter: exports.DocxodusWasm.DocumentConverter,
166
- DocumentComparer: exports.DocxodusWasm.DocumentComparer,
167
- DocxDiffBridge: exports.DocxodusWasm.DocxDiffBridge,
168
- DocxSessionBridge: exports.DocxodusWasm.DocxSessionBridge,
169
- HistoryBridge: exports.DocxodusWasm.HistoryBridge,
170
- };
171
- return true;
172
- }
173
- catch {
174
- return false;
175
- }
176
- }
177
- async function loadWasm() {
178
- // If a custom path is set, use it directly
179
- if (wasmBasePath) {
180
- const success = await tryLoadFromPath(wasmBasePath);
181
- if (success)
182
- return;
183
- throw new Error(`Failed to load WASM from custom path: ${wasmBasePath}. ` +
184
- `Ensure the WASM files are served at this location.`);
185
- }
186
- // Try to auto-detect from module URL (works for CDN and local imports)
187
- const autoDetectedPath = getDefaultWasmBasePath();
188
- if (autoDetectedPath) {
189
- const success = await tryLoadFromPath(autoDetectedPath);
190
- if (success) {
191
- wasmBasePath = autoDetectedPath;
192
- return;
193
- }
194
- }
195
- // Auto-detection failed
196
- throw new Error(`Failed to load WASM files. ` +
197
- `Auto-detected path: ${autoDetectedPath || "(none)"}. ` +
198
- `You can specify a custom path by calling initialize("/path/to/wasm/").`);
199
- }
200
- function ensureInitialized() {
201
- if (!wasmExports) {
202
- throw new Error("Docxodus not initialized. Call initialize() first and await it.");
203
- }
204
- return wasmExports;
205
- }
206
- function isErrorResponse(result) {
207
- try {
208
- const parsed = JSON.parse(result);
209
- return typeof parsed === "object" && "Error" in parsed;
210
- }
211
- catch {
212
- return false;
213
- }
214
- }
215
- function parseError(result) {
216
- const parsed = JSON.parse(result);
217
- return {
218
- error: parsed.Error || parsed.error,
219
- type: parsed.Type || parsed.type,
220
- stackTrace: parsed.StackTrace || parsed.stackTrace,
221
- };
222
- }
223
- /**
224
- * Convert a File or Uint8Array to Uint8Array
225
- */
226
- async function toBytes(input) {
227
- if (input instanceof Uint8Array) {
228
- return input;
229
- }
230
- const buffer = await input.arrayBuffer();
231
- return new Uint8Array(buffer);
232
- }
233
- /**
234
- * Generate a deterministic, non-mutating verification manifest from DOCX bytes.
235
- * Invalid, malformed, and encrypted packages are represented by structured findings.
236
- */
237
- export async function generatePackageManifest(document) {
238
- const exports = ensureInitialized();
239
- const bytes = await toBytes(document);
240
- await yieldToMain();
241
- return JSON.parse(exports.DocumentConverter.GeneratePackageManifest(bytes));
242
- }
243
- /**
244
- * Run the default bounded deliverable-verification policy on exact DOCX bytes.
245
- * Invalid, malformed, encrypted, and safety-limited packages are returned as
246
- * structured report findings rather than editable-session errors. When supplied,
247
- * the exact baseline bytes are used to classify pre-existing, new, and resolved findings.
248
- */
249
- export async function verifyDeliverable(document, baseline) {
250
- const exports = ensureInitialized();
251
- const bytes = await toBytes(document);
252
- const baselineBytes = baseline === undefined ? undefined : await toBytes(baseline);
253
- await yieldToMain();
254
- return JSON.parse(baselineBytes === undefined
255
- ? exports.DocumentConverter.VerifyDeliverable(bytes)
256
- : exports.DocumentConverter.VerifyDeliverableWithBaseline(baselineBytes, bytes));
257
- }
258
- /**
259
- * Verify a portable JSON delivery change receipt against supplied artifact bytes.
260
- *
261
- * The receipt travels as its JSON envelope string; `artifacts` maps each artifact id
262
- * the receipt records to the exact bytes to independently re-hash against it. Omitted
263
- * artifacts report `"missing"`. Malformed input yields a structured invalid verdict
264
- * whose findings carry the reason — never a thrown error.
265
- */
266
- export async function verifyDeliveryReceipt(receiptJson, artifacts) {
267
- const exports = ensureInitialized();
268
- const artifactsJson = artifacts === undefined
269
- ? ""
270
- : JSON.stringify(Object.fromEntries(Object.entries(artifacts).map(([artifactId, bytes]) => [
271
- artifactId,
272
- bytesToBase64(bytes),
273
- ])));
274
- await yieldToMain();
275
- return JSON.parse(exports.DocumentConverter.VerifyDeliveryReceipt(receiptJson, artifactsJson));
276
- }
277
- /**
278
- * Prove that a redline's generated revisions accept to the intended final and reject to the
279
- * selected baseline without consuming pre-existing review state.
280
- *
281
- * Three packages are inspected and two rebuilt, so on a UI thread prefer the worker proxy's
282
- * `proveRedlineReversibility`. Malformed, encrypted, and safety-limited packages are reported as
283
- * structured proof findings rather than thrown errors. The rebuilt packages are not returned:
284
- * the proof carries their digests and the divergences between them and each expected document.
285
- *
286
- * @param baseline - The document the redline was generated against
287
- * @param intendedFinal - The document accepting the generated revisions must reproduce
288
- * @param redline - The generated redline under proof
289
- */
290
- export async function proveRedlineReversibility(baseline, intendedFinal, redline) {
291
- const exports = ensureInitialized();
292
- const baselineBytes = await toBytes(baseline);
293
- const intendedFinalBytes = await toBytes(intendedFinal);
294
- const redlineBytes = await toBytes(redline);
295
- await yieldToMain();
296
- return JSON.parse(exports.DocumentConverter.ProveRedlineReversibility(baselineBytes, intendedFinalBytes, redlineBytes));
297
- }
298
- /**
299
- * Convert a DOCX document to HTML.
300
- *
301
- * @param document - DOCX file as File object or Uint8Array
302
- * @param options - Conversion options
303
- * @returns HTML string
304
- * @throws Error if conversion fails
305
- *
306
- * @example
307
- * ```typescript
308
- * // Basic conversion
309
- * const html = await convertDocxToHtml(docxFile);
310
- *
311
- * // With pagination (PDF.js-style page view)
312
- * const html = await convertDocxToHtml(docxFile, {
313
- * paginationMode: PaginationMode.Paginated,
314
- * paginationScale: 0.8
315
- * });
316
- *
317
- * // With annotations rendered
318
- * const html = await convertDocxToHtml(docxFile, {
319
- * renderAnnotations: true,
320
- * annotationLabelMode: AnnotationLabelMode.Above
321
- * });
322
- *
323
- * // With footnotes and endnotes
324
- * const html = await convertDocxToHtml(docxFile, {
325
- * renderFootnotesAndEndnotes: true
326
- * });
327
- *
328
- * // With headers and footers
329
- * const html = await convertDocxToHtml(docxFile, {
330
- * renderHeadersAndFooters: true
331
- * });
332
- *
333
- * // With tracked changes (redlines visible)
334
- * const html = await convertDocxToHtml(docxFile, {
335
- * renderTrackedChanges: true,
336
- * showDeletedContent: true,
337
- * renderMoveOperations: true
338
- * });
339
- * ```
340
- */
341
- /**
342
- * Render a single document block to faithful HTML, addressed by its anchor.
343
- *
344
- * The anchor is the `data-anchor` value stamped on a block during a full
345
- * conversion (a bare 32-hex Unid), or a full `kind:scope:unid` anchor — either
346
- * form works. Powers the editor's incremental per-block re-render: apply an edit
347
- * to a DocxSession, then re-render only the changed block instead of the whole
348
- * document. Returns the block's HTML element (no `<html>`/`<head>` wrapper).
349
- */
350
- export async function renderBlockHtml(document, anchorId, options) {
351
- const exports = ensureInitialized();
352
- const bytes = await toBytes(document);
353
- await yieldToMain();
354
- const result = exports.DocumentConverter.RenderBlockHtml(bytes, anchorId, options?.cssPrefix ?? "docx-", options?.fabricateClasses ?? false);
355
- if (isErrorResponse(result)) {
356
- throw new Error(`Block rendering failed: ${parseError(result).error}`);
357
- }
358
- return result;
359
- }
360
- export async function convertDocxToHtml(document, options) {
361
- const exports = ensureInitialized();
362
- const bytes = await toBytes(document);
363
- // Yield to browser before heavy WASM work - allows loading states to render
364
- await yieldToMain();
365
- let result;
366
- // Check if any of the new complete options are specified
367
- const needsCompleteMethod = options?.renderFootnotesAndEndnotes !== undefined ||
368
- options?.renderHeadersAndFooters !== undefined ||
369
- options?.renderTrackedChanges !== undefined ||
370
- options?.showDeletedContent !== undefined ||
371
- options?.renderMoveOperations !== undefined ||
372
- options?.renderUnsupportedContentPlaceholders !== undefined ||
373
- options?.documentLanguage !== undefined ||
374
- options?.stampAnchors !== undefined;
375
- // Use complete method when any new options are specified (most comprehensive)
376
- if (needsCompleteMethod || options?.renderAnnotations) {
377
- result = exports.DocumentConverter.ConvertDocxToHtmlComplete(bytes, options?.pageTitle ?? "Document", options?.cssPrefix ?? "docx-", options?.fabricateClasses ?? true, options?.additionalCss ?? "", options?.commentRenderMode ?? CommentRenderMode.Disabled, options?.commentCssClassPrefix ?? "comment-", options?.paginationMode ?? PaginationMode.None, options?.paginationScale ?? 1.0, options?.paginationCssClassPrefix ?? "page-", options?.renderAnnotations ?? false, options?.annotationLabelMode ?? AnnotationLabelMode.Above, options?.annotationCssClassPrefix ?? "annot-", options?.renderFootnotesAndEndnotes ?? false, options?.renderHeadersAndFooters ?? false, options?.renderTrackedChanges ?? false, options?.showDeletedContent ?? true, options?.renderMoveOperations ?? true, options?.renderUnsupportedContentPlaceholders ?? false, options?.documentLanguage ?? null, options?.stampAnchors ?? false);
378
- }
379
- // Use pagination-aware method when pagination is requested
380
- else if (options?.paginationMode !== undefined && options.paginationMode !== PaginationMode.None) {
381
- result = exports.DocumentConverter.ConvertDocxToHtmlWithPagination(bytes, options.pageTitle ?? "Document", options.cssPrefix ?? "docx-", options.fabricateClasses ?? true, options.additionalCss ?? "", options.commentRenderMode ?? CommentRenderMode.Disabled, options.commentCssClassPrefix ?? "comment-", options.paginationMode, options.paginationScale ?? 1.0, options.paginationCssClassPrefix ?? "page-");
382
- }
383
- else if (options) {
384
- result = exports.DocumentConverter.ConvertDocxToHtmlWithOptions(bytes, options.pageTitle ?? "Document", options.cssPrefix ?? "docx-", options.fabricateClasses ?? true, options.additionalCss ?? "", options.commentRenderMode ?? CommentRenderMode.Disabled, options.commentCssClassPrefix ?? "comment-");
385
- }
386
- else {
387
- result = exports.DocumentConverter.ConvertDocxToHtml(bytes);
388
- }
389
- if (isErrorResponse(result)) {
390
- const error = parseError(result);
391
- throw new Error(`Conversion failed: ${error.error}`);
392
- }
393
- return result;
394
- }
395
- /**
396
- * Compare two DOCX documents and return the redlined result as a DOCX.
397
- *
398
- * @param original - Original DOCX document
399
- * @param modified - Modified DOCX document
400
- * @param options - Comparison options
401
- * @returns Redlined DOCX as Uint8Array
402
- * @throws Error if comparison fails
403
- */
404
- export async function compareDocuments(original, modified, options) {
405
- const exports = ensureInitialized();
406
- const originalBytes = await toBytes(original);
407
- const modifiedBytes = await toBytes(modified);
408
- // Yield to browser before heavy WASM work - allows loading states to render
409
- await yieldToMain();
410
- let result;
411
- if (options?.caseInsensitive) {
412
- result = exports.DocumentComparer.CompareDocumentsWithOptions(originalBytes, modifiedBytes, options?.authorName ?? "Docxodus", options.caseInsensitive);
413
- }
414
- else {
415
- result = exports.DocumentComparer.CompareDocuments(originalBytes, modifiedBytes, options?.authorName ?? "Docxodus");
416
- }
417
- if (result.length === 0) {
418
- throw new Error("Comparison failed - empty result");
419
- }
420
- return result;
421
- }
422
- /**
423
- * Compare two DOCX documents and return the result as HTML.
424
- *
425
- * @param original - Original DOCX document
426
- * @param modified - Modified DOCX document
427
- * @param options - Comparison options
428
- * @returns HTML string with redlined content
429
- * @throws Error if comparison fails
430
- */
431
- export async function compareDocumentsToHtml(original, modified, options) {
432
- const exports = ensureInitialized();
433
- const originalBytes = await toBytes(original);
434
- const modifiedBytes = await toBytes(modified);
435
- // Yield to browser before heavy WASM work - allows loading states to render
436
- await yieldToMain();
437
- const renderTrackedChanges = options?.renderTrackedChanges ?? true;
438
- let result;
439
- if (options?.caseInsensitive !== undefined) {
440
- result = exports.DocumentComparer.CompareDocumentsToHtmlFull(originalBytes, modifiedBytes, options?.authorName ?? "Docxodus", options.caseInsensitive, renderTrackedChanges);
441
- }
442
- else {
443
- result = exports.DocumentComparer.CompareDocumentsToHtmlWithOptions(originalBytes, modifiedBytes, options?.authorName ?? "Docxodus", renderTrackedChanges);
444
- }
445
- if (isErrorResponse(result)) {
446
- const error = parseError(result);
447
- throw new Error(`Comparison failed: ${error.error}`);
448
- }
449
- return result;
450
- }
451
- /**
452
- * Get revisions from a compared document.
453
- *
454
- * @param document - A document that has been through comparison (has tracked changes)
455
- * @param options - Optional move detection configuration
456
- * @returns Array of revisions
457
- * @throws Error if operation fails
458
- *
459
- * @example
460
- * ```typescript
461
- * // Default settings (move detection enabled, 80% threshold)
462
- * const revisions = await getRevisions(comparedDoc);
463
- *
464
- * // Custom move detection settings
465
- * const revisions = await getRevisions(comparedDoc, {
466
- * detectMoves: true,
467
- * moveSimilarityThreshold: 0.9, // Require 90% word overlap
468
- * moveMinimumWordCount: 5, // Only consider phrases of 5+ words
469
- * caseInsensitive: true // Ignore case when matching
470
- * });
471
- *
472
- * // Disable move detection entirely
473
- * const revisions = await getRevisions(comparedDoc, { detectMoves: false });
474
- * ```
475
- */
476
- export async function getRevisions(document) {
477
- const exports = ensureInitialized();
478
- const bytes = await toBytes(document);
479
- // Yield to browser before WASM work - allows loading states to render
480
- await yieldToMain();
481
- const result = exports.DocumentComparer.GetRevisionsJson(bytes);
482
- if (isErrorResponse(result)) {
483
- const error = parseError(result);
484
- throw new Error(`Failed to get revisions: ${error.error}`);
485
- }
486
- // The payload is the session's own revision wire shape, so it needs no remapping.
487
- return JSON.parse(result);
488
- }
489
- // ─── DocxDiff (IR diff engine) ──────────────────────────────────────────────
490
- //
491
- // The structure-aware default comparison engine. Its specialized APIs add
492
- // anchor-addressed revisions and the diff-as-data edit script. Settings flow as
493
- // a JSON object; an empty `{}` (or omitted options) uses the engine defaults.
494
- /** Serialize DocxDiffSettings to the wire JSON the bridge parses (empty string when undefined). */
495
- function docxDiffSettingsJson(settings) {
496
- return settings ? JSON.stringify(settings) : "";
497
- }
498
- /**
499
- * Compare two DOCX documents with the IR diff engine and return the redlined
500
- * result as a DOCX (native w:ins/w:del/w:moveFrom/w:moveTo/w:rPrChange markup).
501
- *
502
- * @param left - The earlier/original document.
503
- * @param right - The later/revised document.
504
- * @param settings - Optional {@link DocxDiffSettings}; omit for engine defaults.
505
- * @returns Redlined DOCX as Uint8Array.
506
- * @throws Error if comparison fails.
507
- */
508
- export async function docxDiffCompare(left, right, settings) {
509
- const exports = ensureInitialized();
510
- const leftBytes = await toBytes(left);
511
- const rightBytes = await toBytes(right);
512
- await yieldToMain();
513
- const result = exports.DocxDiffBridge.Compare(leftBytes, rightBytes, docxDiffSettingsJson(settings));
514
- if (result.length === 0) {
515
- throw new Error("DocxDiff comparison failed - empty result");
516
- }
517
- return result;
518
- }
519
- /**
520
- * Compare two DOCX documents with the IR diff engine and return the
521
- * anchor-addressed revision list.
522
- *
523
- * @param left - The earlier/original document.
524
- * @param right - The later/revised document.
525
- * @param settings - Optional {@link DocxDiffSettings}; omit for engine defaults.
526
- * @returns Array of {@link DocxDiffRevision} (each carrying its left/right block anchors).
527
- * @throws Error if the operation fails.
528
- */
529
- export async function docxDiffGetRevisions(left, right, settings) {
530
- const exports = ensureInitialized();
531
- const leftBytes = await toBytes(left);
532
- const rightBytes = await toBytes(right);
533
- await yieldToMain();
534
- const result = exports.DocxDiffBridge.GetRevisionsJson(leftBytes, rightBytes, docxDiffSettingsJson(settings));
535
- if (isErrorResponse(result)) {
536
- const error = parseError(result);
537
- throw new Error(`Failed to get DocxDiff revisions: ${error.error}`);
538
- }
539
- const parsed = JSON.parse(result);
540
- return (parsed.revisions || parsed.Revisions || []).map(mapDocxDiffRevision);
541
- }
542
- /**
543
- * Compare two DOCX documents ONCE and return every requested data product from
544
- * that single memoized pass (issue #594). Where a review pipeline calling
545
- * {@link docxDiffCompare}, {@link docxDiffGetRevisions}, and
546
- * {@link docxDiffGetEditScript} separately pays for the alignment per call, this
547
- * runs it once — each product identical to its standalone counterpart (the edit
548
- * script is handed over parsed rather than as the serialized string).
549
- *
550
- * @param left - The earlier/original document.
551
- * @param right - The later/revised document.
552
- * @param settings - Optional {@link DocxDiffSettings}; omit for engine defaults.
553
- * @param products - Products to compute; omit for all four.
554
- * @throws Error if the operation fails.
555
- */
556
- export async function docxDiffCompareProducts(left, right, settings, products) {
557
- const exports = ensureInitialized();
558
- const leftBytes = await toBytes(left);
559
- const rightBytes = await toBytes(right);
560
- await yieldToMain();
561
- const result = exports.DocxDiffBridge.CompareProductsJson(leftBytes, rightBytes, docxDiffSettingsJson(settings), products ? JSON.stringify(products) : "");
562
- if (isErrorResponse(result)) {
563
- const error = parseError(result);
564
- throw new Error(`Failed to compare products: ${error.error}`);
565
- }
566
- const parsed = JSON.parse(result);
567
- return {
568
- redline: typeof parsed.redlineB64 === "string"
569
- ? Uint8Array.from(atob(parsed.redlineB64), c => c.charCodeAt(0))
570
- : undefined,
571
- revisions: Array.isArray(parsed.revisions)
572
- ? parsed.revisions.map(mapDocxDiffRevision)
573
- : undefined,
574
- editScript: parsed.editScript !== undefined
575
- ? parsed.editScript
576
- : undefined,
577
- semanticChanges: parsed.semanticChanges !== undefined
578
- ? parsed.semanticChanges
579
- : undefined,
580
- };
581
- }
582
- /**
583
- * Compare ONE baseline against MANY candidates, reading the baseline once (issue #617).
584
- *
585
- * The read is the single largest stage of a comparison, and a fan-out — one negotiated
586
- * draft against every counterparty's markup — otherwise re-reads the baseline for each
587
- * one. This reads it once and compares every candidate against that snapshot; each
588
- * result is identical to what {@link docxDiffCompareProducts} returns for the same pair.
589
- *
590
- * A candidate that fails carries an `error` instead of products; the rest of the batch
591
- * still comes back, because one malformed markup should not cost the other ninety-nine.
592
- *
593
- * @param baseline - The shared left-hand document.
594
- * @param candidates - The documents to compare against it, in order.
595
- * @param settings - Optional {@link DocxDiffSettings}; omit for engine defaults.
596
- * @param products - Products to compute; omit for all four.
597
- * @throws Error if the batch itself fails (a bad baseline, malformed settings).
598
- */
599
- export async function docxDiffCompareBatch(baseline, candidates, settings, products) {
600
- const exports = ensureInitialized();
601
- const baselineBytes = await toBytes(baseline);
602
- const payload = [];
603
- for (let i = 0; i < candidates.length; i++) {
604
- const bytes = await toBytes(candidates[i].document);
605
- payload.push({
606
- name: candidates[i].name ?? String(i),
607
- docB64: bytesToBase64(bytes),
608
- });
609
- }
610
- await yieldToMain();
611
- const result = exports.DocxDiffBridge.CompareBatchJson(baselineBytes, JSON.stringify(payload), docxDiffSettingsJson(settings), products ? JSON.stringify(products) : "");
612
- if (isErrorResponse(result)) {
613
- const error = parseError(result);
614
- throw new Error(`Failed to compare batch: ${error.error}`);
615
- }
616
- const parsed = JSON.parse(result);
617
- return (parsed.results ?? []).map((entry) => ({
618
- name: String(entry.name ?? ""),
619
- error: typeof entry.error === "string" ? entry.error : undefined,
620
- redline: typeof entry.redlineB64 === "string"
621
- ? Uint8Array.from(atob(entry.redlineB64), c => c.charCodeAt(0))
622
- : undefined,
623
- revisions: Array.isArray(entry.revisions)
624
- ? entry.revisions.map(mapDocxDiffRevision)
625
- : undefined,
626
- editScript: entry.editScript !== undefined
627
- ? entry.editScript
628
- : undefined,
629
- semanticChanges: entry.semanticChanges !== undefined
630
- ? entry.semanticChanges
631
- : undefined,
632
- }));
633
- }
634
- /**
635
- * Compare two DOCX documents with the IR diff engine and return the edit script
636
- * as a JSON string — the diff-as-data differentiator. The script is the
637
- * anchor-addressed list of block operations the markup and revision renderers
638
- * both consume: stable and machine-readable for storage, transport, and audit.
639
- *
640
- * @param left - The earlier/original document.
641
- * @param right - The later/revised document.
642
- * @param settings - Optional {@link DocxDiffSettings}; omit for engine defaults.
643
- * @returns The edit script serialized as indented JSON.
644
- * @throws Error if the operation fails.
645
- */
646
- export async function docxDiffGetEditScript(left, right, settings) {
647
- const exports = ensureInitialized();
648
- const leftBytes = await toBytes(left);
649
- const rightBytes = await toBytes(right);
650
- await yieldToMain();
651
- const result = exports.DocxDiffBridge.GetEditScriptJson(leftBytes, rightBytes, docxDiffSettingsJson(settings));
652
- if (isErrorResponse(result)) {
653
- const error = parseError(result);
654
- throw new Error(`Failed to get DocxDiff edit script: ${error.error}`);
655
- }
656
- return result;
657
- }
658
- /**
659
- * Compare two DOCX documents and return the stable, versioned semantic-change
660
- * schema. This is the audit/verification surface; it classifies document meaning
661
- * beyond the renderer's internal edit script and preserves unknown package changes.
662
- */
663
- export async function docxDiffGetSemanticChanges(left, right, settings) {
664
- const exports = ensureInitialized();
665
- const leftBytes = await toBytes(left);
666
- const rightBytes = await toBytes(right);
667
- await yieldToMain();
668
- const result = exports.DocxDiffBridge.GetSemanticChangesJson(leftBytes, rightBytes, docxDiffSettingsJson(settings));
669
- if (isErrorResponse(result)) {
670
- const error = parseError(result);
671
- throw new Error(`Failed to get semantic changes: ${error.error}`);
672
- }
673
- return JSON.parse(result);
674
- }
675
- /**
676
- * Accept every tracked revision in a redlined DOCX and return the resulting bytes
677
- * (materializes the "right"/revised side). The byte-in, byte-out counterpart of
678
- * {@link docxDiffCompare}: `docxDiffAcceptRevisions(await docxDiffCompare(left, right))`
679
- * equals `right` at the per-block text level — so callers can verify the round-trip
680
- * contract of a redline, not just inspect its shape.
681
- *
682
- * @param redline - A DOCX carrying tracked-changes markup (e.g. {@link docxDiffCompare} output).
683
- * @returns The DOCX bytes with all revisions accepted.
684
- * @throws Error if the operation fails.
685
- */
686
- export async function docxDiffAcceptRevisions(redline) {
687
- const exports = ensureInitialized();
688
- const bytes = await toBytes(redline);
689
- await yieldToMain();
690
- const result = exports.DocxDiffBridge.AcceptRevisions(bytes);
691
- if (result.length === 0) {
692
- throw new Error("DocxDiff accept-revisions failed - empty result");
693
- }
694
- return result;
695
- }
696
- /**
697
- * Reject every tracked revision in a redlined DOCX and return the resulting bytes
698
- * (materializes the "left"/original side): `docxDiffRejectRevisions(await
699
- * docxDiffCompare(left, right))` equals `left` at the per-block text level.
700
- *
701
- * @param redline - A DOCX carrying tracked-changes markup (e.g. {@link docxDiffCompare} output).
702
- * @returns The DOCX bytes with all revisions rejected.
703
- * @throws Error if the operation fails.
704
- */
705
- export async function docxDiffRejectRevisions(redline) {
706
- const exports = ensureInitialized();
707
- const bytes = await toBytes(redline);
708
- await yieldToMain();
709
- const result = exports.DocxDiffBridge.RejectRevisions(bytes);
710
- if (result.length === 0) {
711
- throw new Error("DocxDiff reject-revisions failed - empty result");
712
- }
713
- return result;
714
- }
715
- // ─── DocxDiff consolidate (composite N-way) ─────────────────────────────────
716
- //
717
- // Merge several reviewers' edits against one shared base DOCX. Each reviewer is
718
- // base64-encoded into the `[{author,docB64}]` wire shape the host base64-DECODES
719
- // (standard base64, not url-safe). Settings flow as the diff-settings JSON object
720
- // extended with an optional integer `conflictResolution`.
721
- /**
722
- * Encode a Uint8Array to a standard (non-url-safe) base64 string. Uses a chunked
723
- * binary string so large documents don't blow the call-stack limit of
724
- * `String.fromCharCode(...bytes)`, and works in both browser (`btoa`) and Node
725
- * (`Buffer`) hosts.
726
- */
727
- function bytesToBase64(bytes) {
728
- if (typeof btoa === "function") {
729
- let binary = "";
730
- const chunkSize = 0x8000; // 32 KB per chunk keeps the spread small
731
- for (let i = 0; i < bytes.length; i += chunkSize) {
732
- const chunk = bytes.subarray(i, i + chunkSize);
733
- binary += String.fromCharCode.apply(null, chunk);
734
- }
735
- return btoa(binary);
736
- }
737
- // Node fallback (e.g. unit tests outside a browser).
738
- return Buffer.from(bytes).toString("base64");
739
- }
740
- /** Serialize reviewers to the `[{author,docB64}]` wire JSON the host expects. */
741
- async function reviewersJson(reviewers) {
742
- const arr = await Promise.all(reviewers.map(async (r) => ({
743
- author: r.author,
744
- docB64: bytesToBase64(await toBytes(r.document)),
745
- })));
746
- return JSON.stringify(arr);
747
- }
748
- /**
749
- * Serialize DocxDiffConsolidateSettings to the wire JSON the bridge parses. Same
750
- * shape as {@link docxDiffSettingsJson} plus the integer `conflictResolution`
751
- * when present (empty string when undefined).
752
- */
753
- function docxDiffConsolidateSettingsJson(settings) {
754
- return settings ? JSON.stringify(settings) : "";
755
- }
756
- /** Map a single revision wire object (camelCase or PascalCase) to {@link DocxDiffRevision}. */
757
- function mapDocxDiffRevision(r) {
758
- return {
759
- revisionType: r.revisionType ?? r.RevisionType,
760
- text: r.text ?? r.Text,
761
- author: r.author ?? r.Author,
762
- date: r.date ?? r.Date,
763
- moveGroupId: r.moveGroupId ?? r.MoveGroupId ?? undefined,
764
- isMoveSource: r.isMoveSource ?? r.IsMoveSource ?? undefined,
765
- formatChange: (r.formatChange || r.FormatChange) ? {
766
- oldProperties: r.formatChange?.oldProperties ?? r.FormatChange?.OldProperties,
767
- newProperties: r.formatChange?.newProperties ?? r.FormatChange?.NewProperties,
768
- changedPropertyNames: r.formatChange?.changedPropertyNames ?? r.FormatChange?.ChangedPropertyNames,
769
- } : undefined,
770
- leftAnchor: r.leftAnchor ?? r.LeftAnchor ?? undefined,
771
- rightAnchor: r.rightAnchor ?? r.RightAnchor ?? undefined,
772
- };
773
- }
774
- /**
775
- * Consolidate several reviewers' edits against a shared base DOCX and return the
776
- * merged redlined result as a DOCX (native multi-author tracked-changes markup).
777
- *
778
- * @param base - The shared base document all reviewers edited from.
779
- * @param reviewers - The reviewers' edited copies + author names.
780
- * @param settings - Optional {@link DocxDiffConsolidateSettings}; omit for engine defaults.
781
- * @returns Consolidated redlined DOCX as Uint8Array.
782
- * @throws Error if consolidation fails.
783
- */
784
- export async function docxDiffConsolidate(base, reviewers, settings) {
785
- const exports = ensureInitialized();
786
- const baseBytes = await toBytes(base);
787
- const reviewersJsonStr = await reviewersJson(reviewers);
788
- await yieldToMain();
789
- const result = exports.DocxDiffBridge.Consolidate(baseBytes, reviewersJsonStr, docxDiffConsolidateSettingsJson(settings));
790
- if (result.length === 0) {
791
- throw new Error("DocxDiff consolidation failed - empty result");
792
- }
793
- return result;
794
- }
795
- /**
796
- * Consolidate several reviewers' edits against a shared base DOCX and return the
797
- * per-token conflict report — every base span two or more reviewers edited
798
- * incompatibly, with each reviewer's competing variant.
799
- *
800
- * @param base - The shared base document all reviewers edited from.
801
- * @param reviewers - The reviewers' edited copies + author names.
802
- * @param settings - Optional {@link DocxDiffConsolidateSettings}; omit for engine defaults.
803
- * @returns Array of {@link DocxDiffConflict}.
804
- * @throws Error if the operation fails.
805
- */
806
- export async function docxDiffGetConflicts(base, reviewers, settings) {
807
- const exports = ensureInitialized();
808
- const baseBytes = await toBytes(base);
809
- const reviewersJsonStr = await reviewersJson(reviewers);
810
- await yieldToMain();
811
- const result = exports.DocxDiffBridge.GetConflictsJson(baseBytes, reviewersJsonStr, docxDiffConsolidateSettingsJson(settings));
812
- if (isErrorResponse(result)) {
813
- const error = parseError(result);
814
- throw new Error(`Failed to get DocxDiff conflicts: ${error.error}`);
815
- }
816
- const parsed = JSON.parse(result);
817
- return (parsed.conflicts || parsed.Conflicts || []).map((c) => ({
818
- id: c.id ?? c.Id,
819
- baseAnchor: c.baseAnchor ?? c.BaseAnchor,
820
- tokenStart: c.tokenStart ?? c.TokenStart,
821
- tokenEnd: c.tokenEnd ?? c.TokenEnd,
822
- policy: c.policy ?? c.Policy,
823
- competitors: (c.competitors || c.Competitors || []).map((comp) => ({
824
- author: comp.author ?? comp.Author,
825
- resultText: comp.resultText ?? comp.ResultText,
826
- })),
827
- }));
828
- }
829
- /**
830
- * Consolidate several reviewers' edits against a shared base DOCX and return the
831
- * merged revision list — each revision carrying its author, block anchors, and
832
- * (when contested) the {@link DocxDiffConsolidatedRevision.conflictId} linking it
833
- * to a {@link DocxDiffConflict}.
834
- *
835
- * @param base - The shared base document all reviewers edited from.
836
- * @param reviewers - The reviewers' edited copies + author names.
837
- * @param settings - Optional {@link DocxDiffConsolidateSettings}; omit for engine defaults.
838
- * @returns Array of {@link DocxDiffConsolidatedRevision}.
839
- * @throws Error if the operation fails.
840
- */
841
- export async function docxDiffGetConsolidatedRevisions(base, reviewers, settings) {
842
- const exports = ensureInitialized();
843
- const baseBytes = await toBytes(base);
844
- const reviewersJsonStr = await reviewersJson(reviewers);
845
- await yieldToMain();
846
- const result = exports.DocxDiffBridge.GetConsolidatedRevisionsJson(baseBytes, reviewersJsonStr, docxDiffConsolidateSettingsJson(settings));
847
- if (isErrorResponse(result)) {
848
- const error = parseError(result);
849
- throw new Error(`Failed to get DocxDiff consolidated revisions: ${error.error}`);
850
- }
851
- const parsed = JSON.parse(result);
852
- return (parsed.revisions || parsed.Revisions || []).map((r) => ({
853
- ...mapDocxDiffRevision(r),
854
- conflictId: r.conflictId ?? r.ConflictId ?? undefined,
855
- }));
856
- }
857
- /**
858
- * Consolidate several reviewers' edits against a shared base DOCX and return the
859
- * merged edit script as a JSON string — the diff-as-data view of the
860
- * consolidation (the anchor-addressed list of composite block operations).
861
- *
862
- * @param base - The shared base document all reviewers edited from.
863
- * @param reviewers - The reviewers' edited copies + author names.
864
- * @param settings - Optional {@link DocxDiffConsolidateSettings}; omit for engine defaults.
865
- * @returns The consolidated edit script serialized as indented JSON.
866
- * @throws Error if the operation fails.
867
- */
868
- export async function docxDiffGetConsolidatedEditScript(base, reviewers, settings) {
869
- const exports = ensureInitialized();
870
- const baseBytes = await toBytes(base);
871
- const reviewersJsonStr = await reviewersJson(reviewers);
872
- await yieldToMain();
873
- const result = exports.DocxDiffBridge.GetConsolidatedEditScriptJson(baseBytes, reviewersJsonStr, docxDiffConsolidateSettingsJson(settings));
874
- if (isErrorResponse(result)) {
875
- const error = parseError(result);
876
- throw new Error(`Failed to get DocxDiff consolidated edit script: ${error.error}`);
877
- }
878
- return result;
879
- }
880
- /**
881
- * Get version information about the library.
882
- */
883
- export function getVersion() {
884
- const exports = ensureInitialized();
885
- const result = exports.DocumentConverter.GetVersion();
886
- const parsed = JSON.parse(result);
887
- return {
888
- library: parsed.Library || parsed.library,
889
- dotnetVersion: parsed.DotnetVersion || parsed.dotnetVersion,
890
- platform: parsed.Platform || parsed.platform,
891
- };
892
- }
893
- /**
894
- * Check if the WASM runtime is initialized.
895
- */
896
- export function isInitialized() {
897
- return wasmExports !== null;
898
- }
899
- /**
900
- * The raw WASM bridge exports (DocumentConverter, DocxSessionBridge, ...).
901
- *
902
- * For consumers that drive a bridge class directly — most notably
903
- * `DocxEditor.open(container, bytes, exports)`, which needs the exports object
904
- * rather than the wrapped functions in this module. Requires `initialize()` to
905
- * have completed; throws otherwise.
906
- */
907
- export function getWasmExports() {
908
- return ensureInitialized();
909
- }
910
- /**
911
- * Get all annotations from a document.
912
- *
913
- * @param document - DOCX file as File object or Uint8Array
914
- * @returns Array of annotations
915
- * @throws Error if operation fails
916
- *
917
- * @example
918
- * ```typescript
919
- * const annotations = await getAnnotations(docxFile);
920
- * for (const annot of annotations) {
921
- * console.log(`${annot.label}: "${annot.annotatedText}"`);
922
- * }
923
- * ```
924
- */
925
- export async function getAnnotations(document) {
926
- const exports = ensureInitialized();
927
- const bytes = await toBytes(document);
928
- const result = exports.DocumentConverter.GetAnnotations(bytes);
929
- if (isErrorResponse(result)) {
930
- const error = parseError(result);
931
- throw new Error(`Failed to get annotations: ${error.error}`);
932
- }
933
- const parsed = JSON.parse(result);
934
- return (parsed.Annotations || parsed.annotations || []).map((a) => ({
935
- id: a.Id || a.id,
936
- labelId: a.LabelId || a.labelId,
937
- label: a.Label || a.label,
938
- color: a.Color || a.color,
939
- author: a.Author || a.author,
940
- created: a.Created || a.created,
941
- bookmarkName: a.BookmarkName || a.bookmarkName,
942
- startPage: a.StartPage ?? a.startPage,
943
- endPage: a.EndPage ?? a.endPage,
944
- annotatedText: a.AnnotatedText || a.annotatedText,
945
- metadata: a.Metadata || a.metadata,
946
- }));
947
- }
948
- /**
949
- * Add an annotation to a document.
950
- *
951
- * @param document - DOCX file as File object or Uint8Array
952
- * @param request - Annotation details including search text or paragraph indices
953
- * @returns Response with modified document bytes and annotation info
954
- * @throws Error if operation fails
955
- *
956
- * @example
957
- * ```typescript
958
- * // Annotate by searching for text
959
- * const result = await addAnnotation(docxFile, {
960
- * id: "annot-1",
961
- * labelId: "CLAUSE_A",
962
- * label: "Important Clause",
963
- * color: "#FFEB3B",
964
- * searchText: "shall not be liable",
965
- * occurrence: 1
966
- * });
967
- *
968
- * // Annotate by paragraph range
969
- * const result = await addAnnotation(docxFile, {
970
- * id: "annot-2",
971
- * labelId: "SECTION_1",
972
- * label: "Introduction",
973
- * color: "#4CAF50",
974
- * startParagraphIndex: 0,
975
- * endParagraphIndex: 2
976
- * });
977
- *
978
- * // Get modified document
979
- * const modifiedDocBytes = base64ToBytes(result.documentBytes);
980
- * ```
981
- */
982
- export async function addAnnotation(document, request) {
983
- const exports = ensureInitialized();
984
- const bytes = await toBytes(document);
985
- // Yield to browser before WASM work - allows loading states to render
986
- await yieldToMain();
987
- const requestJson = JSON.stringify({
988
- Id: request.id,
989
- LabelId: request.labelId,
990
- Label: request.label,
991
- Color: request.color ?? "#FFEB3B",
992
- Author: request.author,
993
- SearchText: request.searchText,
994
- Occurrence: request.occurrence ?? 1,
995
- StartParagraphIndex: request.startParagraphIndex,
996
- EndParagraphIndex: request.endParagraphIndex,
997
- Metadata: request.metadata,
998
- });
999
- const result = exports.DocumentConverter.AddAnnotation(bytes, requestJson);
1000
- if (isErrorResponse(result)) {
1001
- const error = parseError(result);
1002
- throw new Error(`Failed to add annotation: ${error.error}`);
1003
- }
1004
- const parsed = JSON.parse(result);
1005
- const annotation = parsed.Annotation || parsed.annotation;
1006
- return {
1007
- success: parsed.Success ?? parsed.success ?? true,
1008
- documentBytes: parsed.DocumentBytes || parsed.documentBytes,
1009
- annotation: annotation ? {
1010
- id: annotation.Id || annotation.id,
1011
- labelId: annotation.LabelId || annotation.labelId,
1012
- label: annotation.Label || annotation.label,
1013
- color: annotation.Color || annotation.color,
1014
- author: annotation.Author || annotation.author,
1015
- created: annotation.Created || annotation.created,
1016
- bookmarkName: annotation.BookmarkName || annotation.bookmarkName,
1017
- annotatedText: annotation.AnnotatedText || annotation.annotatedText,
1018
- } : undefined,
1019
- };
1020
- }
1021
- /**
1022
- * Remove an annotation from a document.
1023
- *
1024
- * @param document - DOCX file as File object or Uint8Array
1025
- * @param annotationId - The ID of the annotation to remove
1026
- * @returns Response with modified document bytes
1027
- * @throws Error if operation fails
1028
- *
1029
- * @example
1030
- * ```typescript
1031
- * const result = await removeAnnotation(docxFile, "annot-1");
1032
- * const modifiedDocBytes = base64ToBytes(result.documentBytes);
1033
- * ```
1034
- */
1035
- export async function removeAnnotation(document, annotationId) {
1036
- const exports = ensureInitialized();
1037
- const bytes = await toBytes(document);
1038
- const result = exports.DocumentConverter.RemoveAnnotation(bytes, annotationId);
1039
- if (isErrorResponse(result)) {
1040
- const error = parseError(result);
1041
- throw new Error(`Failed to remove annotation: ${error.error}`);
1042
- }
1043
- const parsed = JSON.parse(result);
1044
- return {
1045
- success: parsed.Success ?? parsed.success ?? true,
1046
- documentBytes: parsed.DocumentBytes || parsed.documentBytes,
1047
- };
1048
- }
1049
- /**
1050
- * Check if a document has any annotations.
1051
- *
1052
- * @param document - DOCX file as File object or Uint8Array
1053
- * @returns true if the document has annotations
1054
- * @throws Error if operation fails
1055
- *
1056
- * @example
1057
- * ```typescript
1058
- * if (await hasAnnotations(docxFile)) {
1059
- * const annotations = await getAnnotations(docxFile);
1060
- * console.log(`Document has ${annotations.length} annotations`);
1061
- * }
1062
- * ```
1063
- */
1064
- export async function hasAnnotations(document) {
1065
- const exports = ensureInitialized();
1066
- const bytes = await toBytes(document);
1067
- const result = exports.DocumentConverter.HasAnnotations(bytes);
1068
- if (isErrorResponse(result)) {
1069
- const error = parseError(result);
1070
- throw new Error(`Failed to check annotations: ${error.error}`);
1071
- }
1072
- const parsed = JSON.parse(result);
1073
- return parsed.HasAnnotations ?? parsed.hasAnnotations ?? false;
1074
- }
1075
- /**
1076
- * Get the document structure for element-based annotation targeting.
1077
- *
1078
- * @param document - DOCX file as File object or Uint8Array
1079
- * @returns Document structure with element tree
1080
- * @throws Error if operation fails
1081
- *
1082
- * @example
1083
- * ```typescript
1084
- * const structure = await getDocumentStructure(docxFile);
1085
- *
1086
- * // Navigate the structure tree
1087
- * console.log(`Document has ${structure.root.children.length} top-level elements`);
1088
- *
1089
- * // Find all paragraphs
1090
- * const paragraphs = getParagraphs(structure);
1091
- * console.log(`Found ${paragraphs.length} paragraphs`);
1092
- *
1093
- * // Find all tables
1094
- * const tables = getTables(structure);
1095
- * for (const table of tables) {
1096
- * const columns = getTableColumns(structure, table.id);
1097
- * console.log(`Table ${table.id} has ${columns.length} columns`);
1098
- * }
1099
- *
1100
- * // Look up element by ID
1101
- * const element = findElementById(structure, "doc/p-0");
1102
- * if (element) {
1103
- * console.log(`First paragraph: "${element.textPreview}"`);
1104
- * }
1105
- * ```
1106
- */
1107
- export async function getDocumentStructure(document) {
1108
- const exports = ensureInitialized();
1109
- const bytes = await toBytes(document);
1110
- // Yield to browser before WASM work - allows loading states to render
1111
- await yieldToMain();
1112
- const result = exports.DocumentConverter.GetDocumentStructure(bytes);
1113
- if (isErrorResponse(result)) {
1114
- const error = parseError(result);
1115
- throw new Error(`Failed to get document structure: ${error.error}`);
1116
- }
1117
- const parsed = JSON.parse(result);
1118
- // Convert from PascalCase to camelCase
1119
- const convertElement = (el) => ({
1120
- id: el.Id || el.id,
1121
- anchorId: el.AnchorId || el.anchorId,
1122
- type: el.Type || el.type,
1123
- textPreview: el.TextPreview || el.textPreview,
1124
- index: el.Index ?? el.index,
1125
- rowIndex: el.RowIndex ?? el.rowIndex,
1126
- columnIndex: el.ColumnIndex ?? el.columnIndex,
1127
- rowSpan: el.RowSpan ?? el.rowSpan,
1128
- columnSpan: el.ColumnSpan ?? el.columnSpan,
1129
- children: (el.Children || el.children || []).map(convertElement),
1130
- });
1131
- const convertTableColumn = (col) => ({
1132
- tableId: col.TableId || col.tableId,
1133
- anchorId: col.AnchorId || col.anchorId,
1134
- tableAnchorId: col.TableAnchorId || col.tableAnchorId,
1135
- isVirtual: col.IsVirtual ?? col.isVirtual ?? false,
1136
- columnIndex: col.ColumnIndex ?? col.columnIndex,
1137
- cellIds: col.CellIds || col.cellIds || [],
1138
- cellAnchorIds: col.CellAnchorIds || col.cellAnchorIds || [],
1139
- rowCount: col.RowCount ?? col.rowCount,
1140
- });
1141
- const root = convertElement(parsed.Root || parsed.root);
1142
- // Convert elementsById dictionary
1143
- const elementsById = {};
1144
- const rawElementsById = parsed.ElementsById || parsed.elementsById || {};
1145
- for (const [key, el] of Object.entries(rawElementsById)) {
1146
- elementsById[key] = convertElement(el);
1147
- }
1148
- // Convert tableColumns dictionary
1149
- const tableColumns = {};
1150
- const rawTableColumns = parsed.TableColumns || parsed.tableColumns || {};
1151
- for (const [key, col] of Object.entries(rawTableColumns)) {
1152
- tableColumns[key] = convertTableColumn(col);
1153
- }
1154
- return {
1155
- root,
1156
- elementsById,
1157
- tableColumns,
1158
- };
1159
- }
1160
- /**
1161
- * Get document metadata for lazy loading pagination.
1162
- * This is a fast operation that extracts structure information without full HTML rendering.
1163
- *
1164
- * @param document - DOCX file as File object or Uint8Array
1165
- * @returns Document metadata including sections, dimensions, and content counts
1166
- * @throws Error if operation fails
1167
- *
1168
- * @example
1169
- * ```typescript
1170
- * const metadata = await getDocumentMetadata(docxFile);
1171
- *
1172
- * // Check document overview
1173
- * console.log(`Document has ${metadata.totalParagraphs} paragraphs`);
1174
- * console.log(`Document has ${metadata.sections.length} sections`);
1175
- * console.log(`Estimated ${metadata.estimatedPageCount} pages`);
1176
- *
1177
- * // Check section properties
1178
- * for (const section of metadata.sections) {
1179
- * console.log(`Section ${section.sectionIndex}: ${section.pageWidthPt}x${section.pageHeightPt}pt`);
1180
- * console.log(` Paragraphs: ${section.paragraphCount}, Tables: ${section.tableCount}`);
1181
- * console.log(` Has header: ${section.hasHeader}, Has footer: ${section.hasFooter}`);
1182
- * }
1183
- *
1184
- * // Check document features
1185
- * if (metadata.hasTrackedChanges) {
1186
- * console.log("Document has tracked changes");
1187
- * }
1188
- * if (metadata.hasFootnotes) {
1189
- * console.log("Document has footnotes");
1190
- * }
1191
- * ```
1192
- */
1193
- export async function getDocumentMetadata(document) {
1194
- const exports = ensureInitialized();
1195
- const bytes = await toBytes(document);
1196
- // Yield to browser before WASM work - allows loading states to render
1197
- await yieldToMain();
1198
- const result = exports.DocumentConverter.GetDocumentMetadata(bytes);
1199
- if (isErrorResponse(result)) {
1200
- const error = parseError(result);
1201
- throw new Error(`Failed to get document metadata: ${error.error}`);
1202
- }
1203
- const parsed = JSON.parse(result);
1204
- // Convert from PascalCase to camelCase
1205
- const convertSection = (s) => ({
1206
- sectionIndex: s.SectionIndex ?? s.sectionIndex,
1207
- pageWidthPt: s.PageWidthPt ?? s.pageWidthPt,
1208
- pageHeightPt: s.PageHeightPt ?? s.pageHeightPt,
1209
- marginTopPt: s.MarginTopPt ?? s.marginTopPt,
1210
- marginRightPt: s.MarginRightPt ?? s.marginRightPt,
1211
- marginBottomPt: s.MarginBottomPt ?? s.marginBottomPt,
1212
- marginLeftPt: s.MarginLeftPt ?? s.marginLeftPt,
1213
- contentWidthPt: s.ContentWidthPt ?? s.contentWidthPt,
1214
- contentHeightPt: s.ContentHeightPt ?? s.contentHeightPt,
1215
- headerPt: s.HeaderPt ?? s.headerPt,
1216
- footerPt: s.FooterPt ?? s.footerPt,
1217
- paragraphCount: s.ParagraphCount ?? s.paragraphCount,
1218
- tableCount: s.TableCount ?? s.tableCount,
1219
- hasHeader: s.HasHeader ?? s.hasHeader,
1220
- hasFooter: s.HasFooter ?? s.hasFooter,
1221
- hasFirstPageHeader: s.HasFirstPageHeader ?? s.hasFirstPageHeader,
1222
- hasFirstPageFooter: s.HasFirstPageFooter ?? s.hasFirstPageFooter,
1223
- hasEvenPageHeader: s.HasEvenPageHeader ?? s.hasEvenPageHeader,
1224
- hasEvenPageFooter: s.HasEvenPageFooter ?? s.hasEvenPageFooter,
1225
- startParagraphIndex: s.StartParagraphIndex ?? s.startParagraphIndex,
1226
- endParagraphIndex: s.EndParagraphIndex ?? s.endParagraphIndex,
1227
- startTableIndex: s.StartTableIndex ?? s.startTableIndex,
1228
- endTableIndex: s.EndTableIndex ?? s.endTableIndex,
1229
- });
1230
- return {
1231
- sections: (parsed.Sections || parsed.sections || []).map(convertSection),
1232
- totalParagraphs: parsed.TotalParagraphs ?? parsed.totalParagraphs,
1233
- totalTables: parsed.TotalTables ?? parsed.totalTables,
1234
- hasFootnotes: parsed.HasFootnotes ?? parsed.hasFootnotes,
1235
- hasEndnotes: parsed.HasEndnotes ?? parsed.hasEndnotes,
1236
- hasTrackedChanges: parsed.HasTrackedChanges ?? parsed.hasTrackedChanges,
1237
- hasComments: parsed.HasComments ?? parsed.hasComments,
1238
- estimatedPageCount: parsed.EstimatedPageCount ?? parsed.estimatedPageCount,
1239
- estimatedPageCountSource: parsed.EstimatedPageCountSource ?? parsed.estimatedPageCountSource ?? "heuristic",
1240
- };
1241
- }
1242
- /**
1243
- * Export document to OpenContracts format.
1244
- *
1245
- * This provides complete document text, structure, and layout information
1246
- * compatible with the OpenContracts ecosystem for document analysis.
1247
- *
1248
- * @param document - DOCX file as File object or Uint8Array
1249
- * @returns OpenContractDocExport with complete document data
1250
- * @throws Error if export fails
1251
- *
1252
- * @example
1253
- * ```typescript
1254
- * const result = await exportToOpenContract(docxFile);
1255
- *
1256
- * // Access complete document text
1257
- * console.log(`Content length: ${result.content.length} characters`);
1258
- *
1259
- * // Get document structure
1260
- * console.log(`Pages: ${result.pageCount}`);
1261
- * console.log(`Structural annotations: ${result.labelledText.filter(a => a.structural).length}`);
1262
- *
1263
- * // Access PAWLS layout data
1264
- * for (const page of result.pawlsFileContent) {
1265
- * console.log(`Page ${page.page.index}: ${page.tokens.length} tokens`);
1266
- * }
1267
- * ```
1268
- */
1269
- export async function exportToOpenContract(document) {
1270
- const exports = ensureInitialized();
1271
- const bytes = await toBytes(document);
1272
- // Yield to browser before WASM work - allows loading states to render
1273
- await yieldToMain();
1274
- const result = exports.DocumentConverter.ExportToOpenContract(bytes);
1275
- if (isErrorResponse(result)) {
1276
- const error = parseError(result);
1277
- throw new Error(`Failed to export to OpenContract format: ${error.error}`);
1278
- }
1279
- const parsed = JSON.parse(result);
1280
- // Convert from PascalCase to camelCase
1281
- const convertPawlsPage = (p) => ({
1282
- page: {
1283
- width: p.Page?.Width ?? p.page?.width,
1284
- height: p.Page?.Height ?? p.page?.height,
1285
- index: p.Page?.Index ?? p.page?.index,
1286
- },
1287
- tokens: (p.Tokens || p.tokens || []).map((t) => ({
1288
- x: t.X ?? t.x,
1289
- y: t.Y ?? t.y,
1290
- width: t.Width ?? t.width,
1291
- height: t.Height ?? t.height,
1292
- text: t.Text ?? t.text,
1293
- })),
1294
- });
1295
- const convertAnnotation = (a) => ({
1296
- id: a.Id ?? a.id,
1297
- annotationLabel: a.AnnotationLabel ?? a.annotationLabel,
1298
- rawText: a.RawText ?? a.rawText,
1299
- page: a.Page ?? a.page,
1300
- annotationJson: convertAnnotationJson(a.AnnotationJson ?? a.annotationJson),
1301
- parentId: a.ParentId ?? a.parentId,
1302
- annotationType: a.AnnotationType ?? a.annotationType,
1303
- structural: a.Structural ?? a.structural,
1304
- });
1305
- const convertAnnotationJson = (json) => {
1306
- if (!json)
1307
- return undefined;
1308
- // Check if it's a TextSpan
1309
- if (json.Start !== undefined || json.start !== undefined) {
1310
- return {
1311
- id: json.Id ?? json.id,
1312
- start: json.Start ?? json.start,
1313
- end: json.End ?? json.end,
1314
- text: json.Text ?? json.text,
1315
- };
1316
- }
1317
- // Otherwise it's a dictionary of single-page annotations
1318
- const result = {};
1319
- for (const [key, value] of Object.entries(json)) {
1320
- const v = value;
1321
- result[key] = {
1322
- bounds: {
1323
- top: v.Bounds?.Top ?? v.bounds?.top,
1324
- bottom: v.Bounds?.Bottom ?? v.bounds?.bottom,
1325
- left: v.Bounds?.Left ?? v.bounds?.left,
1326
- right: v.Bounds?.Right ?? v.bounds?.right,
1327
- },
1328
- tokensJsons: (v.TokensJsons || v.tokensJsons || []).map((t) => ({
1329
- pageIndex: t.PageIndex ?? t.pageIndex,
1330
- tokenIndex: t.TokenIndex ?? t.tokenIndex,
1331
- })),
1332
- rawText: v.RawText ?? v.rawText,
1333
- };
1334
- }
1335
- return result;
1336
- };
1337
- const convertRelationship = (r) => ({
1338
- id: r.Id ?? r.id,
1339
- relationshipLabel: r.RelationshipLabel ?? r.relationshipLabel,
1340
- sourceAnnotationIds: r.SourceAnnotationIds ?? r.sourceAnnotationIds ?? [],
1341
- targetAnnotationIds: r.TargetAnnotationIds ?? r.targetAnnotationIds ?? [],
1342
- structural: r.Structural ?? r.structural,
1343
- });
1344
- return {
1345
- title: parsed.Title ?? parsed.title,
1346
- content: parsed.Content ?? parsed.content,
1347
- description: parsed.Description ?? parsed.description,
1348
- pageCount: parsed.PageCount ?? parsed.pageCount,
1349
- pawlsFileContent: (parsed.PawlsFileContent || parsed.pawlsFileContent || []).map(convertPawlsPage),
1350
- docLabels: parsed.DocLabels ?? parsed.docLabels ?? [],
1351
- labelledText: (parsed.LabelledText || parsed.labelledText || []).map(convertAnnotation),
1352
- relationships: (parsed.Relationships || parsed.relationships)?.map(convertRelationship),
1353
- };
1354
- }
1355
- /**
1356
- * Convert a DOCX file to an anchor-addressed Markdown projection.
1357
- *
1358
- * The projection is a deterministic, anchor-keyed Markdown rendering of the document,
1359
- * suitable for LLM editing pipelines, structured search indexers, and diff/review UIs.
1360
- * Every paragraph, heading, list item, table, table cell, footnote, endnote, and
1361
- * comment is addressable by an `{#kind:scope:unid}` anchor that survives reformatting.
1362
- *
1363
- * See `docs/architecture/markdown_projection.md` for the projection spec.
1364
- *
1365
- * @param document - DOCX file as `File` or `Uint8Array`
1366
- * @param settings - Optional projection settings (defaults: all scopes, anchor blocks, accept tracked changes)
1367
- * @throws Error if conversion fails
1368
- *
1369
- * @example
1370
- * ```typescript
1371
- * const result = await convertWmlToMarkdown(docxFile);
1372
- * console.log(result.markdown);
1373
- * for (const [id, target] of Object.entries(result.anchorIndex)) {
1374
- * console.log(id, target.partUri);
1375
- * }
1376
- * ```
1377
- */
1378
- export async function convertWmlToMarkdown(document, settings = {}) {
1379
- const exports = ensureInitialized();
1380
- const bytes = await toBytes(document);
1381
- await yieldToMain();
1382
- const settingsJson = JSON.stringify({
1383
- Scopes: settings.scopes ?? ProjectionScopes.All,
1384
- HeadingLevelOffset: settings.headingLevelOffset ?? 0,
1385
- AnchorMode: settings.anchorMode ?? AnchorRenderMode.Block,
1386
- TableMode: settings.tableMode ?? TableRenderMode.GfmWithOpaqueFallback,
1387
- TableInlineCellMax: settings.tableInlineCellMax ?? 80,
1388
- TrackedChanges: settings.trackedChanges ?? TrackedChangeMode.Accept,
1389
- ResolveNumbering: settings.resolveNumbering ?? true,
1390
- EmptyParagraphs: settings.emptyParagraphs ?? EmptyParagraphMode.AnchorOnly,
1391
- });
1392
- const result = exports.DocumentConverter.ConvertWmlToMarkdown(bytes, settingsJson);
1393
- if (isErrorResponse(result)) {
1394
- const error = parseError(result);
1395
- throw new Error(`Failed to convert document to markdown: ${error.error}`);
1396
- }
1397
- const parsed = JSON.parse(result);
1398
- const rawIndex = parsed.AnchorIndex ?? parsed.anchorIndex ?? {};
1399
- const anchorIndex = {};
1400
- for (const [key, value] of Object.entries(rawIndex)) {
1401
- const v = value;
1402
- anchorIndex[key] = {
1403
- id: v.Id ?? v.id,
1404
- kind: v.Kind ?? v.kind,
1405
- scope: v.Scope ?? v.scope,
1406
- unid: v.Unid ?? v.unid,
1407
- partUri: v.PartUri ?? v.partUri,
1408
- textPreview: v.TextPreview ?? v.textPreview ?? "",
1409
- };
1410
- }
1411
- return {
1412
- markdown: parsed.Markdown ?? parsed.markdown ?? "",
1413
- anchorIndex,
1414
- };
1415
- }
1416
- /**
1417
- * Add an annotation using flexible targeting (element ID, indices, or text search).
1418
- *
1419
- * @param document - DOCX file as File object or Uint8Array
1420
- * @param request - Annotation details with target specification
1421
- * @returns Response with modified document bytes and annotation info
1422
- * @throws Error if operation fails
1423
- *
1424
- * @example
1425
- * ```typescript
1426
- * // First get the document structure to find target elements
1427
- * const structure = await getDocumentStructure(docxFile);
1428
- *
1429
- * // Annotate a specific paragraph by element ID
1430
- * const result1 = await addAnnotationWithTarget(docxFile, {
1431
- * id: "annot-1",
1432
- * labelId: "INTRO",
1433
- * label: "Introduction",
1434
- * color: "#4CAF50",
1435
- * target: targetElement("doc/p-0")
1436
- * });
1437
- *
1438
- * // Annotate a table cell
1439
- * const result2 = await addAnnotationWithTarget(docxFile, {
1440
- * id: "annot-2",
1441
- * labelId: "CELL_HIGHLIGHT",
1442
- * label: "Important Cell",
1443
- * color: "#FFEB3B",
1444
- * target: targetTableCell(0, 1, 2) // Table 0, Row 1, Cell 2
1445
- * });
1446
- *
1447
- * // Annotate a table column
1448
- * const result3 = await addAnnotationWithTarget(docxFile, {
1449
- * id: "annot-3",
1450
- * labelId: "COLUMN_DATA",
1451
- * label: "Values Column",
1452
- * color: "#2196F3",
1453
- * target: targetTableColumn(0, 1) // Table 0, Column 1
1454
- * });
1455
- *
1456
- * // Search for text within a specific element
1457
- * const result4 = await addAnnotationWithTarget(docxFile, {
1458
- * id: "annot-4",
1459
- * labelId: "KEYWORD",
1460
- * label: "Keyword",
1461
- * color: "#FF5722",
1462
- * target: targetSearchInElement("doc/p-2", "important", 1)
1463
- * });
1464
- * ```
1465
- */
1466
- export async function addAnnotationWithTarget(document, request) {
1467
- const exports = ensureInitialized();
1468
- const bytes = await toBytes(document);
1469
- // Yield to browser before WASM work - allows loading states to render
1470
- await yieldToMain();
1471
- const requestJson = JSON.stringify({
1472
- Id: request.id,
1473
- LabelId: request.labelId,
1474
- Label: request.label,
1475
- Color: request.color ?? "#FFEB3B",
1476
- Author: request.author,
1477
- Metadata: request.metadata,
1478
- ElementId: request.target.elementId,
1479
- ElementType: request.target.elementType,
1480
- ParagraphIndex: request.target.paragraphIndex,
1481
- RunIndex: request.target.runIndex,
1482
- TableIndex: request.target.tableIndex,
1483
- RowIndex: request.target.rowIndex,
1484
- CellIndex: request.target.cellIndex,
1485
- ColumnIndex: request.target.columnIndex,
1486
- SearchText: request.target.searchText,
1487
- Occurrence: request.target.occurrence ?? 1,
1488
- RangeEndParagraphIndex: request.target.rangeEndParagraphIndex,
1489
- });
1490
- const result = exports.DocumentConverter.AddAnnotationWithTarget(bytes, requestJson);
1491
- if (isErrorResponse(result)) {
1492
- const error = parseError(result);
1493
- throw new Error(`Failed to add annotation: ${error.error}`);
1494
- }
1495
- const parsed = JSON.parse(result);
1496
- const annotation = parsed.Annotation || parsed.annotation;
1497
- return {
1498
- success: parsed.Success ?? parsed.success ?? true,
1499
- documentBytes: parsed.DocumentBytes || parsed.documentBytes,
1500
- annotation: annotation ? {
1501
- id: annotation.Id || annotation.id,
1502
- labelId: annotation.LabelId || annotation.labelId,
1503
- label: annotation.Label || annotation.label,
1504
- color: annotation.Color || annotation.color,
1505
- author: annotation.Author || annotation.author,
1506
- created: annotation.Created || annotation.created,
1507
- bookmarkName: annotation.BookmarkName || annotation.bookmarkName,
1508
- annotatedText: annotation.AnnotatedText || annotation.annotatedText,
1509
- metadata: annotation.Metadata || annotation.metadata,
1510
- } : undefined,
1511
- };
1512
- }
1513
- // ============================================================================
1514
- // External Annotation Functions (Issue #57)
1515
- // ============================================================================
1516
- /**
1517
- * Compute the SHA256 hash of a document for integrity validation.
1518
- *
1519
- * @param document - DOCX file as File object or Uint8Array
1520
- * @returns SHA256 hash as lowercase hex string
1521
- * @throws Error if operation fails
1522
- *
1523
- * @example
1524
- * ```typescript
1525
- * const hash = await computeDocumentHash(docxFile);
1526
- * console.log(`Document hash: ${hash}`);
1527
- *
1528
- * // Later, verify the document hasn't changed
1529
- * const currentHash = await computeDocumentHash(docxFile);
1530
- * if (currentHash !== storedHash) {
1531
- * console.log("Document has been modified");
1532
- * }
1533
- * ```
1534
- */
1535
- export async function computeDocumentHash(document) {
1536
- const exports = ensureInitialized();
1537
- const bytes = await toBytes(document);
1538
- const result = exports.DocumentConverter.ComputeDocumentHash(bytes);
1539
- if (isErrorResponse(result)) {
1540
- const error = parseError(result);
1541
- throw new Error(`Failed to compute document hash: ${error.error}`);
1542
- }
1543
- const parsed = JSON.parse(result);
1544
- return parsed.Hash ?? parsed.hash;
1545
- }
1546
- /**
1547
- * Create an ExternalAnnotationSet from a document.
1548
- * This extracts the document structure and computes the hash for integrity validation.
1549
- *
1550
- * @param document - DOCX file as File object or Uint8Array
1551
- * @param documentId - Unique identifier for the document (filename, UUID, etc.)
1552
- * @returns ExternalAnnotationSet ready for adding annotations
1553
- * @throws Error if operation fails
1554
- *
1555
- * @example
1556
- * ```typescript
1557
- * // Create an annotation set
1558
- * const set = await createExternalAnnotationSet(docxFile, "contract-v1.0");
1559
- *
1560
- * // Access document text for searching
1561
- * console.log(`Document length: ${set.content.length} chars`);
1562
- *
1563
- * // Add label definitions
1564
- * set.textLabels["IMPORTANT"] = {
1565
- * id: "IMPORTANT",
1566
- * text: "Important",
1567
- * color: "#FF0000",
1568
- * description: "Important text",
1569
- * icon: "",
1570
- * labelType: "text"
1571
- * };
1572
- *
1573
- * // Create annotations using the content
1574
- * const annotation = createAnnotationFromSearch(
1575
- * "ann-001", "IMPORTANT", set.content, "shall not be liable"
1576
- * );
1577
- * if (annotation) {
1578
- * set.labelledText.push(annotation);
1579
- * }
1580
- *
1581
- * // Serialize for storage
1582
- * const json = JSON.stringify(set);
1583
- * ```
1584
- */
1585
- export async function createExternalAnnotationSet(document, documentId) {
1586
- const exports = ensureInitialized();
1587
- const bytes = await toBytes(document);
1588
- // Yield to browser before WASM work - allows loading states to render
1589
- await yieldToMain();
1590
- const result = exports.DocumentConverter.CreateExternalAnnotationSet(bytes, documentId);
1591
- if (isErrorResponse(result)) {
1592
- const error = parseError(result);
1593
- throw new Error(`Failed to create external annotation set: ${error.error}`);
1594
- }
1595
- const parsed = JSON.parse(result);
1596
- // Convert from PascalCase to camelCase
1597
- return convertExternalAnnotationSet(parsed);
1598
- }
1599
- /**
1600
- * Validate an external annotation set against a document.
1601
- * Checks hash match and verifies each annotation's text still matches.
1602
- *
1603
- * @param document - DOCX file as File object or Uint8Array
1604
- * @param annotationSet - The annotation set to validate
1605
- * @returns Validation result with any issues found
1606
- * @throws Error if operation fails
1607
- *
1608
- * @example
1609
- * ```typescript
1610
- * const result = await validateExternalAnnotations(docxFile, annotationSet);
1611
- *
1612
- * if (!result.isValid) {
1613
- * if (result.hashMismatch) {
1614
- * console.log("Document has been modified since annotations were created");
1615
- * }
1616
- * for (const issue of result.issues) {
1617
- * console.log(`${issue.issueType}: ${issue.description}`);
1618
- * }
1619
- * }
1620
- * ```
1621
- */
1622
- export async function validateExternalAnnotations(document, annotationSet) {
1623
- const exports = ensureInitialized();
1624
- const bytes = await toBytes(document);
1625
- // Yield to browser before WASM work - allows loading states to render
1626
- await yieldToMain();
1627
- const annotationSetJson = JSON.stringify(annotationSet);
1628
- const result = exports.DocumentConverter.ValidateExternalAnnotations(bytes, annotationSetJson);
1629
- if (isErrorResponse(result)) {
1630
- const error = parseError(result);
1631
- throw new Error(`Failed to validate external annotations: ${error.error}`);
1632
- }
1633
- const parsed = JSON.parse(result);
1634
- return {
1635
- isValid: parsed.IsValid ?? parsed.isValid,
1636
- hashMismatch: parsed.HashMismatch ?? parsed.hashMismatch,
1637
- issues: (parsed.Issues || parsed.issues || []).map((i) => ({
1638
- annotationId: i.AnnotationId ?? i.annotationId,
1639
- issueType: i.IssueType ?? i.issueType,
1640
- description: i.Description ?? i.description,
1641
- expectedText: i.ExpectedText ?? i.expectedText,
1642
- actualText: i.ActualText ?? i.actualText,
1643
- })),
1644
- };
1645
- }
1646
- /**
1647
- * Convert a DOCX document to HTML with external annotations projected.
1648
- *
1649
- * @param document - DOCX file as File object or Uint8Array
1650
- * @param annotationSet - The external annotation set to project
1651
- * @param conversionOptions - HTML conversion options
1652
- * @param projectionOptions - Annotation projection options
1653
- * @returns HTML string with annotations projected
1654
- * @throws Error if operation fails
1655
- *
1656
- * @example
1657
- * ```typescript
1658
- * // Basic usage
1659
- * const html = await convertDocxToHtmlWithExternalAnnotations(
1660
- * docxFile,
1661
- * annotationSet
1662
- * );
1663
- *
1664
- * // With custom options
1665
- * const html = await convertDocxToHtmlWithExternalAnnotations(
1666
- * docxFile,
1667
- * annotationSet,
1668
- * { pageTitle: "Annotated Document" },
1669
- * { labelMode: AnnotationLabelMode.Inline, cssClassPrefix: "my-annot-" }
1670
- * );
1671
- * ```
1672
- */
1673
- export async function convertDocxToHtmlWithExternalAnnotations(document, annotationSet, conversionOptions, projectionOptions) {
1674
- const exports = ensureInitialized();
1675
- const bytes = await toBytes(document);
1676
- // Yield to browser before WASM work - allows loading states to render
1677
- await yieldToMain();
1678
- const annotationSetJson = JSON.stringify(annotationSet);
1679
- const result = exports.DocumentConverter.ConvertDocxToHtmlWithExternalAnnotations(bytes, annotationSetJson, conversionOptions?.pageTitle ?? "Document", conversionOptions?.cssPrefix ?? "docx-", conversionOptions?.fabricateClasses ?? true, conversionOptions?.additionalCss ?? "", projectionOptions?.cssClassPrefix ?? "ext-annot-", projectionOptions?.labelMode ?? AnnotationLabelMode.Above);
1680
- if (isErrorResponse(result)) {
1681
- const error = parseError(result);
1682
- throw new Error(`Failed to convert with external annotations: ${error.error}`);
1683
- }
1684
- const parsed = JSON.parse(result);
1685
- return parsed.Html ?? parsed.html;
1686
- }
1687
- /**
1688
- * Search for text in a document and return character offsets.
1689
- * Useful for finding text locations to create annotations.
1690
- *
1691
- * @param document - DOCX file as File object or Uint8Array
1692
- * @param searchText - Text to search for
1693
- * @param maxResults - Maximum number of results (default: 100)
1694
- * @returns Array of TextSpan objects with offsets
1695
- * @throws Error if operation fails
1696
- *
1697
- * @example
1698
- * ```typescript
1699
- * const occurrences = await searchTextOffsets(docxFile, "liability");
1700
- * console.log(`Found ${occurrences.length} occurrences`);
1701
- *
1702
- * for (const span of occurrences) {
1703
- * console.log(`"${span.text}" at offset ${span.start}-${span.end}`);
1704
- * }
1705
- * ```
1706
- */
1707
- export async function searchTextOffsets(document, searchText, maxResults = 100) {
1708
- const exports = ensureInitialized();
1709
- const bytes = await toBytes(document);
1710
- // Yield to browser before WASM work - allows loading states to render
1711
- await yieldToMain();
1712
- const result = exports.DocumentConverter.SearchTextOffsets(bytes, searchText, maxResults);
1713
- if (isErrorResponse(result)) {
1714
- const error = parseError(result);
1715
- throw new Error(`Failed to search text: ${error.error}`);
1716
- }
1717
- const parsed = JSON.parse(result);
1718
- return (parsed.Results || parsed.results || []).map((r) => ({
1719
- id: r.Id ?? r.id,
1720
- start: r.Start ?? r.start,
1721
- end: r.End ?? r.end,
1722
- text: r.Text ?? r.text,
1723
- }));
1724
- }
1725
- /**
1726
- * Create an annotation from character offsets.
1727
- * This is a client-side helper - no WASM call needed.
1728
- *
1729
- * @param id - Unique identifier for the annotation
1730
- * @param labelId - Label/category ID for the annotation
1731
- * @param documentText - Full document text (from annotationSet.content)
1732
- * @param startOffset - Start character offset (0-indexed, inclusive)
1733
- * @param endOffset - End character offset (exclusive)
1734
- * @returns OpenContractsAnnotation ready to add to an annotation set
1735
- * @throws Error if offsets are invalid
1736
- *
1737
- * @example
1738
- * ```typescript
1739
- * const set = await createExternalAnnotationSet(docxFile, "doc-1");
1740
- * const annotation = createAnnotation("ann-001", "IMPORTANT", set.content, 100, 150);
1741
- * set.labelledText.push(annotation);
1742
- * ```
1743
- */
1744
- export function createAnnotation(id, labelId, documentText, startOffset, endOffset) {
1745
- if (startOffset < 0) {
1746
- throw new Error("Start offset must be non-negative");
1747
- }
1748
- if (endOffset < startOffset) {
1749
- throw new Error("End offset must be >= start offset");
1750
- }
1751
- if (endOffset > documentText.length) {
1752
- throw new Error("End offset exceeds document length");
1753
- }
1754
- const rawText = documentText.substring(startOffset, endOffset);
1755
- return {
1756
- id,
1757
- annotationLabel: labelId,
1758
- rawText,
1759
- page: 0,
1760
- annotationJson: {
1761
- id,
1762
- start: startOffset,
1763
- end: endOffset,
1764
- text: rawText,
1765
- },
1766
- annotationType: "text",
1767
- structural: false,
1768
- };
1769
- }
1770
- /**
1771
- * Create an annotation by searching for text in the document.
1772
- * This is a client-side helper - no WASM call needed.
1773
- *
1774
- * @param id - Unique identifier for the annotation
1775
- * @param labelId - Label/category ID for the annotation
1776
- * @param documentText - Full document text (from annotationSet.content)
1777
- * @param searchText - Text to search for
1778
- * @param occurrence - Which occurrence to use (1-based, default: 1)
1779
- * @returns OpenContractsAnnotation, or null if text not found
1780
- *
1781
- * @example
1782
- * ```typescript
1783
- * const set = await createExternalAnnotationSet(docxFile, "doc-1");
1784
- *
1785
- * // Find first occurrence
1786
- * const ann1 = createAnnotationFromSearch("ann-001", "LIABILITY", set.content, "shall not be liable");
1787
- * if (ann1) set.labelledText.push(ann1);
1788
- *
1789
- * // Find second occurrence
1790
- * const ann2 = createAnnotationFromSearch("ann-002", "LIABILITY", set.content, "shall not be liable", 2);
1791
- * if (ann2) set.labelledText.push(ann2);
1792
- * ```
1793
- */
1794
- export function createAnnotationFromSearch(id, labelId, documentText, searchText, occurrence = 1) {
1795
- if (occurrence < 1) {
1796
- throw new Error("Occurrence must be >= 1");
1797
- }
1798
- const offsets = findTextOccurrences(documentText, searchText);
1799
- if (occurrence > offsets.length) {
1800
- return null;
1801
- }
1802
- const { start, end } = offsets[occurrence - 1];
1803
- return createAnnotation(id, labelId, documentText, start, end);
1804
- }
1805
- /**
1806
- * Find all occurrences of a text string in the document.
1807
- * This is a client-side helper - no WASM call needed.
1808
- *
1809
- * @param documentText - Full document text
1810
- * @param searchText - Text to search for
1811
- * @param maxResults - Maximum number of results (default: 100)
1812
- * @returns Array of { start, end } offsets
1813
- *
1814
- * @example
1815
- * ```typescript
1816
- * const occurrences = findTextOccurrences(set.content, "the");
1817
- * console.log(`Found ${occurrences.length} occurrences of "the"`);
1818
- * ```
1819
- */
1820
- export function findTextOccurrences(documentText, searchText, maxResults = 100) {
1821
- if (!searchText)
1822
- return [];
1823
- const results = [];
1824
- let index = 0;
1825
- while (results.length < maxResults) {
1826
- index = documentText.indexOf(searchText, index);
1827
- if (index < 0)
1828
- break;
1829
- results.push({ start: index, end: index + searchText.length });
1830
- index += 1; // Move past start to find overlapping matches
1831
- }
1832
- return results;
1833
- }
1834
- // Helper function to convert PascalCase response to camelCase ExternalAnnotationSet
1835
- function convertExternalAnnotationSet(parsed) {
1836
- const convertLabel = (l) => ({
1837
- id: l.Id ?? l.id,
1838
- color: l.Color ?? l.color,
1839
- description: l.Description ?? l.description ?? "",
1840
- icon: l.Icon ?? l.icon ?? "",
1841
- text: l.Text ?? l.text,
1842
- labelType: l.LabelType ?? l.labelType ?? "text",
1843
- });
1844
- const convertPawlsPage = (p) => ({
1845
- page: {
1846
- width: p.Page?.Width ?? p.page?.width,
1847
- height: p.Page?.Height ?? p.page?.height,
1848
- index: p.Page?.Index ?? p.page?.index,
1849
- },
1850
- tokens: (p.Tokens || p.tokens || []).map((t) => ({
1851
- x: t.X ?? t.x,
1852
- y: t.Y ?? t.y,
1853
- width: t.Width ?? t.width,
1854
- height: t.Height ?? t.height,
1855
- text: t.Text ?? t.text,
1856
- })),
1857
- });
1858
- const convertAnnotation = (a) => ({
1859
- id: a.Id ?? a.id,
1860
- annotationLabel: a.AnnotationLabel ?? a.annotationLabel,
1861
- rawText: a.RawText ?? a.rawText,
1862
- page: a.Page ?? a.page,
1863
- annotationJson: convertAnnotationJson(a.AnnotationJson ?? a.annotationJson),
1864
- parentId: a.ParentId ?? a.parentId,
1865
- annotationType: a.AnnotationType ?? a.annotationType,
1866
- structural: a.Structural ?? a.structural,
1867
- });
1868
- const convertAnnotationJson = (json) => {
1869
- if (!json)
1870
- return undefined;
1871
- // Check if it's a TextSpan
1872
- if (json.Start !== undefined || json.start !== undefined) {
1873
- return {
1874
- id: json.Id ?? json.id,
1875
- start: json.Start ?? json.start,
1876
- end: json.End ?? json.end,
1877
- text: json.Text ?? json.text,
1878
- };
1879
- }
1880
- // Otherwise it's a dictionary of single-page annotations
1881
- const result = {};
1882
- for (const [key, value] of Object.entries(json)) {
1883
- const v = value;
1884
- result[key] = {
1885
- bounds: {
1886
- top: v.Bounds?.Top ?? v.bounds?.top,
1887
- bottom: v.Bounds?.Bottom ?? v.bounds?.bottom,
1888
- left: v.Bounds?.Left ?? v.bounds?.left,
1889
- right: v.Bounds?.Right ?? v.bounds?.right,
1890
- },
1891
- tokensJsons: (v.TokensJsons || v.tokensJsons || []).map((t) => ({
1892
- pageIndex: t.PageIndex ?? t.pageIndex,
1893
- tokenIndex: t.TokenIndex ?? t.tokenIndex,
1894
- })),
1895
- rawText: v.RawText ?? v.rawText,
1896
- };
1897
- }
1898
- return result;
1899
- };
1900
- const convertRelationship = (r) => ({
1901
- id: r.Id ?? r.id,
1902
- relationshipLabel: r.RelationshipLabel ?? r.relationshipLabel,
1903
- sourceAnnotationIds: r.SourceAnnotationIds ?? r.sourceAnnotationIds ?? [],
1904
- targetAnnotationIds: r.TargetAnnotationIds ?? r.targetAnnotationIds ?? [],
1905
- structural: r.Structural ?? r.structural,
1906
- });
1907
- // Convert label dictionaries
1908
- const textLabels = {};
1909
- const rawTextLabels = parsed.TextLabels || parsed.textLabels || {};
1910
- for (const [key, value] of Object.entries(rawTextLabels)) {
1911
- textLabels[key] = convertLabel(value);
1912
- }
1913
- const docLabelDefinitions = {};
1914
- const rawDocLabelDefs = parsed.DocLabelDefinitions || parsed.docLabelDefinitions || {};
1915
- for (const [key, value] of Object.entries(rawDocLabelDefs)) {
1916
- docLabelDefinitions[key] = convertLabel(value);
1917
- }
1918
- return {
1919
- documentId: parsed.DocumentId ?? parsed.documentId,
1920
- documentHash: parsed.DocumentHash ?? parsed.documentHash,
1921
- createdAt: parsed.CreatedAt ?? parsed.createdAt,
1922
- updatedAt: parsed.UpdatedAt ?? parsed.updatedAt,
1923
- version: parsed.Version ?? parsed.version,
1924
- title: parsed.Title ?? parsed.title,
1925
- content: parsed.Content ?? parsed.content,
1926
- description: parsed.Description ?? parsed.description,
1927
- pageCount: parsed.PageCount ?? parsed.pageCount,
1928
- pawlsFileContent: (parsed.PawlsFileContent || parsed.pawlsFileContent || []).map(convertPawlsPage),
1929
- docLabels: parsed.DocLabels ?? parsed.docLabels ?? [],
1930
- labelledText: (parsed.LabelledText || parsed.labelledText || []).map(convertAnnotation),
1931
- relationships: (parsed.Relationships || parsed.relationships)?.map(convertRelationship),
1932
- textLabels,
1933
- docLabelDefinitions,
1934
- };
1935
- }
1936
- // ============================================================================
1937
- // Incremental Annotation Overlay API (Issue #106)
1938
- // ============================================================================
1939
- /**
1940
- * Project external annotations onto already-converted HTML.
1941
- * This avoids full DOCX re-conversion when only annotations change.
1942
- *
1943
- * Workflow:
1944
- * 1. Convert DOCX to HTML once using `convertDocxToHtml()`
1945
- * 2. Use this function to overlay annotations on the cached HTML
1946
- * 3. When annotations change, call this again with the same base HTML
1947
- *
1948
- * @param html - HTML string (previously converted via convertDocxToHtml)
1949
- * @param annotationSet - The external annotation set to project
1950
- * @param projectionOptions - Projection settings (CSS prefix, label mode, etc.)
1951
- * @returns HTML string with annotations projected
1952
- * @throws Error if projection fails
1953
- *
1954
- * @example
1955
- * ```typescript
1956
- * // Step 1: Convert once
1957
- * const baseHtml = await convertDocxToHtml(docxFile);
1958
- *
1959
- * // Step 2: Project annotations (fast, no DOCX re-conversion)
1960
- * const annotatedHtml = await projectAnnotationsOntoHtml(baseHtml, annotationSet);
1961
- *
1962
- * // Step 3: When annotations change, project again on the same base HTML
1963
- * annotationSet.labelledText.push(newAnnotation);
1964
- * const updatedHtml = await projectAnnotationsOntoHtml(baseHtml, annotationSet);
1965
- * ```
1966
- */
1967
- export async function projectAnnotationsOntoHtml(html, annotationSet, projectionOptions) {
1968
- const exports = ensureInitialized();
1969
- await yieldToMain();
1970
- const annotationSetJson = JSON.stringify(annotationSet);
1971
- const result = exports.DocumentConverter.ProjectAnnotationsOntoHtml(html, annotationSetJson, projectionOptions?.cssClassPrefix ?? "ext-annot-", projectionOptions?.labelMode ?? AnnotationLabelMode.Above);
1972
- if (isErrorResponse(result)) {
1973
- const error = parseError(result);
1974
- throw new Error(`Failed to project annotations: ${error.error}`);
1975
- }
1976
- const parsed = JSON.parse(result);
1977
- return parsed.Html ?? parsed.html;
1978
- }
1979
- /**
1980
- * Add a single annotation to existing HTML without re-converting the document.
1981
- * This is the fastest way to add one annotation to already-rendered HTML.
1982
- *
1983
- * @param html - HTML string (with or without existing annotations)
1984
- * @param annotation - The annotation to add
1985
- * @param label - Label definition for the annotation (optional, for color/text)
1986
- * @param projectionOptions - Projection settings
1987
- * @returns HTML string with the annotation added
1988
- * @throws Error if operation fails
1989
- *
1990
- * @example
1991
- * ```typescript
1992
- * const annotation = createAnnotation("ann-new", "CLAUSE", set.content, 100, 150);
1993
- * const label = { id: "CLAUSE", text: "Clause", color: "#FF5722" };
1994
- * const updatedHtml = await addAnnotationToHtml(currentHtml, annotation, label);
1995
- * ```
1996
- */
1997
- export async function addAnnotationToHtml(html, annotation, label, projectionOptions) {
1998
- const exports = ensureInitialized();
1999
- await yieldToMain();
2000
- const annotationJson = JSON.stringify(annotation);
2001
- const labelJson = label ? JSON.stringify(label) : "";
2002
- const result = exports.DocumentConverter.AddAnnotationToHtml(html, annotationJson, labelJson, projectionOptions?.cssClassPrefix ?? "ext-annot-", projectionOptions?.labelMode ?? AnnotationLabelMode.Above);
2003
- if (isErrorResponse(result)) {
2004
- const error = parseError(result);
2005
- throw new Error(`Failed to add annotation to HTML: ${error.error}`);
2006
- }
2007
- const parsed = JSON.parse(result);
2008
- return parsed.Html ?? parsed.html;
2009
- }
2010
- /**
2011
- * Remove a single annotation from HTML by annotation ID.
2012
- * Unwraps annotation spans back to plain text.
2013
- *
2014
- * @param html - HTML string with annotations
2015
- * @param annotationId - ID of the annotation to remove
2016
- * @param cssClassPrefix - CSS class prefix used for annotations (default: "ext-annot-")
2017
- * @returns HTML string with the annotation removed
2018
- * @throws Error if operation fails
2019
- *
2020
- * @example
2021
- * ```typescript
2022
- * const updatedHtml = await removeAnnotationFromHtml(currentHtml, "ann-001");
2023
- * ```
2024
- */
2025
- export async function removeAnnotationFromHtml(html, annotationId, cssClassPrefix) {
2026
- const exports = ensureInitialized();
2027
- await yieldToMain();
2028
- const result = exports.DocumentConverter.RemoveAnnotationFromHtml(html, annotationId, cssClassPrefix ?? "ext-annot-");
2029
- if (isErrorResponse(result)) {
2030
- const error = parseError(result);
2031
- throw new Error(`Failed to remove annotation from HTML: ${error.error}`);
2032
- }
2033
- const parsed = JSON.parse(result);
2034
- return parsed.Html ?? parsed.html;
2035
- }
2036
- /**
2037
- * Generate CSS to hide annotations with specific label IDs.
2038
- * Enables CSS-based label filtering without re-rendering HTML.
2039
- *
2040
- * Apply the returned CSS to your document (e.g., via a `<style>` element)
2041
- * to hide/show annotations by label. This is much faster than re-projecting
2042
- * all annotations.
2043
- *
2044
- * @param hiddenLabelIds - Array of label IDs to hide
2045
- * @param cssClassPrefix - CSS class prefix (default: "ext-annot-")
2046
- * @returns CSS string that hides the specified labels
2047
- * @throws Error if operation fails
2048
- *
2049
- * @example
2050
- * ```typescript
2051
- * // Hide annotations with label "DRAFT" and "INTERNAL"
2052
- * const css = await generateAnnotationVisibilityCss(["DRAFT", "INTERNAL"]);
2053
- *
2054
- * // Apply to a <style> element in the DOM
2055
- * const styleEl = document.getElementById("visibility-overrides");
2056
- * styleEl.textContent = css;
2057
- *
2058
- * // To show all annotations again, clear the style:
2059
- * styleEl.textContent = "";
2060
- * ```
2061
- */
2062
- export async function generateAnnotationVisibilityCss(hiddenLabelIds, cssClassPrefix) {
2063
- const exports = ensureInitialized();
2064
- await yieldToMain();
2065
- const result = exports.DocumentConverter.GenerateAnnotationVisibilityCss(JSON.stringify(hiddenLabelIds), cssClassPrefix ?? "ext-annot-");
2066
- if (isErrorResponse(result)) {
2067
- const error = parseError(result);
2068
- throw new Error(`Failed to generate visibility CSS: ${error.error}`);
2069
- }
2070
- const parsed = JSON.parse(result);
2071
- return parsed.Css ?? parsed.css;
2072
- }
2073
- /**
2074
- * Generate annotation CSS for a set of labels.
2075
- * Useful when managing CSS separately from HTML content.
2076
- *
2077
- * @param labels - Label definitions (keyed by label ID)
2078
- * @param projectionOptions - Projection settings
2079
- * @returns CSS string for the annotation styles
2080
- * @throws Error if operation fails
2081
- *
2082
- * @example
2083
- * ```typescript
2084
- * const labels = {
2085
- * "CLAUSE": { id: "CLAUSE", text: "Clause", color: "#FF5722" },
2086
- * "TERM": { id: "TERM", text: "Term", color: "#2196F3" },
2087
- * };
2088
- * const css = await generateAnnotationCss(labels);
2089
- * ```
2090
- */
2091
- export async function generateAnnotationCss(labels, projectionOptions) {
2092
- const exports = ensureInitialized();
2093
- await yieldToMain();
2094
- const result = exports.DocumentConverter.GenerateAnnotationCss(JSON.stringify(labels), projectionOptions?.cssClassPrefix ?? "ext-annot-", projectionOptions?.labelMode ?? AnnotationLabelMode.Above);
2095
- if (isErrorResponse(result)) {
2096
- const error = parseError(result);
2097
- throw new Error(`Failed to generate annotation CSS: ${error.error}`);
2098
- }
2099
- const parsed = JSON.parse(result);
2100
- return parsed.Css ?? parsed.css;
2101
- }
2102
11
  //# sourceMappingURL=index.js.map