web-doc 0.6.1 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. package/THIRD_PARTY_NOTICES.md +26 -1
  2. package/dist/adapters/docx-images.d.ts +13 -4
  3. package/dist/adapters/docx-images.js +3 -0
  4. package/dist/adapters/docx-paragraphs.d.ts +65 -0
  5. package/dist/adapters/docx-paragraphs.js +165 -0
  6. package/dist/adapters/docx-prepass.d.ts +17 -0
  7. package/dist/adapters/docx-prepass.js +230 -0
  8. package/dist/adapters/office.d.ts +31 -5
  9. package/dist/adapters/office.js +59 -28
  10. package/dist/adapters/pdf.d.ts +9 -0
  11. package/dist/adapters/pdf.js +10 -0
  12. package/dist/assets/pdfium/pdfium.wasm +0 -0
  13. package/dist/contracts.d.ts +47 -2
  14. package/dist/edit/assets.d.ts +23 -0
  15. package/dist/edit/assets.js +75 -0
  16. package/dist/edit/docx/elements.d.ts +6 -0
  17. package/dist/edit/docx/elements.js +90 -0
  18. package/dist/edit/docx/engine.d.ts +51 -0
  19. package/dist/edit/docx/engine.js +473 -0
  20. package/dist/edit/docx/handlers.d.ts +3 -0
  21. package/dist/edit/docx/handlers.js +15 -0
  22. package/dist/edit/docx/ids.d.ts +45 -0
  23. package/dist/edit/docx/ids.js +101 -0
  24. package/dist/edit/docx/model.d.ts +92 -0
  25. package/dist/edit/docx/model.js +339 -0
  26. package/dist/edit/docx/operations.d.ts +37 -0
  27. package/dist/edit/docx/operations.js +3 -0
  28. package/dist/edit/docx/provider.d.ts +13 -0
  29. package/dist/edit/docx/provider.js +34 -0
  30. package/dist/edit/docx/schemas.d.ts +6 -0
  31. package/dist/edit/docx/schemas.js +192 -0
  32. package/dist/edit/docx/session.d.ts +37 -0
  33. package/dist/edit/docx/session.js +426 -0
  34. package/dist/edit/docx/structure-ops.d.ts +6 -0
  35. package/dist/edit/docx/structure-ops.js +494 -0
  36. package/dist/edit/docx/style.d.ts +45 -0
  37. package/dist/edit/docx/style.js +375 -0
  38. package/dist/edit/docx/table-ops.d.ts +4 -0
  39. package/dist/edit/docx/table-ops.js +219 -0
  40. package/dist/edit/docx/text-ops.d.ts +18 -0
  41. package/dist/edit/docx/text-ops.js +458 -0
  42. package/dist/edit/docx/text.d.ts +34 -0
  43. package/dist/edit/docx/text.js +263 -0
  44. package/dist/edit/docx/types.d.ts +191 -0
  45. package/dist/edit/docx/types.js +1 -0
  46. package/dist/edit/docx/write.d.ts +50 -0
  47. package/dist/edit/docx/write.js +352 -0
  48. package/dist/edit/engine.d.ts +117 -0
  49. package/dist/edit/engine.js +1 -0
  50. package/dist/edit/history.d.ts +53 -0
  51. package/dist/edit/history.js +101 -0
  52. package/dist/edit/ooxml/names.d.ts +20 -0
  53. package/dist/edit/ooxml/names.js +61 -0
  54. package/dist/edit/ooxml/opc.d.ts +50 -0
  55. package/dist/edit/ooxml/opc.js +150 -0
  56. package/dist/edit/ooxml/package.d.ts +82 -0
  57. package/dist/edit/ooxml/package.js +233 -0
  58. package/dist/edit/ooxml/patch.d.ts +51 -0
  59. package/dist/edit/ooxml/patch.js +250 -0
  60. package/dist/edit/ooxml/transaction.d.ts +39 -0
  61. package/dist/edit/ooxml/transaction.js +316 -0
  62. package/dist/edit/ooxml/worker.d.ts +8 -0
  63. package/dist/edit/ooxml/worker.js +12 -0
  64. package/dist/edit/ooxml/writer.d.ts +21 -0
  65. package/dist/edit/ooxml/writer.js +187 -0
  66. package/dist/edit/ooxml/xml.d.ts +74 -0
  67. package/dist/edit/ooxml/xml.js +451 -0
  68. package/dist/edit/ooxml/zip.d.ts +54 -0
  69. package/dist/edit/ooxml/zip.js +280 -0
  70. package/dist/edit/operations.d.ts +19 -0
  71. package/dist/edit/operations.js +137 -0
  72. package/dist/edit/pdf/engine/compact.d.ts +6 -0
  73. package/dist/edit/pdf/engine/compact.js +442 -0
  74. package/dist/edit/pdf/engine/document.d.ts +95 -0
  75. package/dist/edit/pdf/engine/document.js +868 -0
  76. package/dist/edit/pdf/engine/elements.d.ts +45 -0
  77. package/dist/edit/pdf/engine/elements.js +313 -0
  78. package/dist/edit/pdf/engine/existing-text.d.ts +4 -0
  79. package/dist/edit/pdf/engine/existing-text.js +424 -0
  80. package/dist/edit/pdf/engine/fonts.d.ts +78 -0
  81. package/dist/edit/pdf/engine/fonts.js +466 -0
  82. package/dist/edit/pdf/engine/geometry.d.ts +36 -0
  83. package/dist/edit/pdf/engine/geometry.js +97 -0
  84. package/dist/edit/pdf/engine/handler.d.ts +21 -0
  85. package/dist/edit/pdf/engine/handler.js +108 -0
  86. package/dist/edit/pdf/engine/images.d.ts +33 -0
  87. package/dist/edit/pdf/engine/images.js +188 -0
  88. package/dist/edit/pdf/engine/layout.d.ts +23 -0
  89. package/dist/edit/pdf/engine/layout.js +177 -0
  90. package/dist/edit/pdf/engine/operations.d.ts +67 -0
  91. package/dist/edit/pdf/engine/operations.js +3 -0
  92. package/dist/edit/pdf/engine/pages.d.ts +6 -0
  93. package/dist/edit/pdf/engine/pages.js +98 -0
  94. package/dist/edit/pdf/engine/pdfium.d.ts +195 -0
  95. package/dist/edit/pdf/engine/pdfium.js +249 -0
  96. package/dist/edit/pdf/engine/shapes.d.ts +4 -0
  97. package/dist/edit/pdf/engine/shapes.js +183 -0
  98. package/dist/edit/pdf/engine/tables.d.ts +44 -0
  99. package/dist/edit/pdf/engine/tables.js +315 -0
  100. package/dist/edit/pdf/engine/text-box.d.ts +71 -0
  101. package/dist/edit/pdf/engine/text-box.js +317 -0
  102. package/dist/edit/pdf/engine/text-layout.d.ts +28 -0
  103. package/dist/edit/pdf/engine/text-layout.js +67 -0
  104. package/dist/edit/pdf/engine/transform.d.ts +5 -0
  105. package/dist/edit/pdf/engine/transform.js +137 -0
  106. package/dist/edit/pdf/provider.d.ts +35 -0
  107. package/dist/edit/pdf/provider.js +92 -0
  108. package/dist/edit/pdf/range-map.d.ts +14 -0
  109. package/dist/edit/pdf/range-map.js +107 -0
  110. package/dist/edit/pdf/schemas.d.ts +11 -0
  111. package/dist/edit/pdf/schemas.js +291 -0
  112. package/dist/edit/pdf/selection.d.ts +5 -0
  113. package/dist/edit/pdf/selection.js +152 -0
  114. package/dist/edit/pdf/session.d.ts +53 -0
  115. package/dist/edit/pdf/session.js +273 -0
  116. package/dist/edit/pdf/types.d.ts +336 -0
  117. package/dist/edit/pdf/types.js +1 -0
  118. package/dist/edit/pptx/elements.d.ts +74 -0
  119. package/dist/edit/pptx/elements.js +301 -0
  120. package/dist/edit/pptx/engine.d.ts +37 -0
  121. package/dist/edit/pptx/engine.js +466 -0
  122. package/dist/edit/pptx/geometry.d.ts +72 -0
  123. package/dist/edit/pptx/geometry.js +193 -0
  124. package/dist/edit/pptx/handler.d.ts +7 -0
  125. package/dist/edit/pptx/handler.js +91 -0
  126. package/dist/edit/pptx/handlers.d.ts +3 -0
  127. package/dist/edit/pptx/handlers.js +21 -0
  128. package/dist/edit/pptx/image-table-ops.d.ts +5 -0
  129. package/dist/edit/pptx/image-table-ops.js +233 -0
  130. package/dist/edit/pptx/model.d.ts +69 -0
  131. package/dist/edit/pptx/model.js +170 -0
  132. package/dist/edit/pptx/operations.d.ts +50 -0
  133. package/dist/edit/pptx/operations.js +11 -0
  134. package/dist/edit/pptx/provider.d.ts +17 -0
  135. package/dist/edit/pptx/provider.js +40 -0
  136. package/dist/edit/pptx/schemas.d.ts +6 -0
  137. package/dist/edit/pptx/schemas.js +220 -0
  138. package/dist/edit/pptx/session.d.ts +44 -0
  139. package/dist/edit/pptx/session.js +114 -0
  140. package/dist/edit/pptx/shape-ops.d.ts +10 -0
  141. package/dist/edit/pptx/shape-ops.js +486 -0
  142. package/dist/edit/pptx/slide-ops.d.ts +6 -0
  143. package/dist/edit/pptx/slide-ops.js +353 -0
  144. package/dist/edit/pptx/style.d.ts +31 -0
  145. package/dist/edit/pptx/style.js +181 -0
  146. package/dist/edit/pptx/text-ops.d.ts +13 -0
  147. package/dist/edit/pptx/text-ops.js +267 -0
  148. package/dist/edit/pptx/text-write.d.ts +67 -0
  149. package/dist/edit/pptx/text-write.js +293 -0
  150. package/dist/edit/pptx/text.d.ts +41 -0
  151. package/dist/edit/pptx/text.js +99 -0
  152. package/dist/edit/pptx/types.d.ts +232 -0
  153. package/dist/edit/pptx/types.js +1 -0
  154. package/dist/edit/schema.d.ts +11 -0
  155. package/dist/edit/schema.js +275 -0
  156. package/dist/edit/session.d.ts +66 -0
  157. package/dist/edit/session.js +701 -0
  158. package/dist/edit/sessions.d.ts +8 -0
  159. package/dist/edit/sessions.js +1 -0
  160. package/dist/edit/types.d.ts +276 -0
  161. package/dist/edit/types.js +1 -0
  162. package/dist/edit/worker-engine.d.ts +31 -0
  163. package/dist/edit/worker-engine.js +91 -0
  164. package/dist/fonts/THIRD_PARTY_NOTICES.md +3 -0
  165. package/dist/fonts/manifest.json +6 -1
  166. package/dist/fonts/noto-sans-latin-cyrillic.ttf +0 -0
  167. package/dist/fonts.d.ts +2 -0
  168. package/dist/fonts.js +4 -0
  169. package/dist/headless.d.ts +4 -0
  170. package/dist/headless.js +4 -0
  171. package/dist/index.d.ts +5 -0
  172. package/dist/index.js +5 -0
  173. package/dist/limits.js +3 -0
  174. package/dist/ooxml-edit-worker.d.ts +1 -0
  175. package/dist/ooxml-edit-worker.js +8 -0
  176. package/dist/pdf-edit-worker.d.ts +1 -0
  177. package/dist/pdf-edit-worker.js +35 -0
  178. package/dist/spreadsheet-viewport.d.ts +5 -0
  179. package/dist/spreadsheet-viewport.js +11 -0
  180. package/dist/ui.js +7 -0
  181. package/dist/viewer.d.ts +6 -0
  182. package/dist/viewer.js +293 -19
  183. package/dist/viewport.d.ts +13 -0
  184. package/dist/viewport.js +126 -1
  185. package/dist/worker-protocol.d.ts +39 -2
  186. package/dist/workers/ooxml-edit-worker.js +8671 -0
  187. package/dist/workers/pdf-edit-worker.js +10844 -0
  188. package/package.json +3 -2
@@ -0,0 +1,473 @@
1
+ import { ViewerError } from "../../errors.js";
2
+ import { AssetStore } from "../assets.js";
3
+ import { OoxmlPackage } from "../ooxml/package.js";
4
+ import { patches } from "../ooxml/patch.js";
5
+ import { invalidOperationError, parseReference } from "../operations.js";
6
+ import { NO_PAGE, toElement } from "./elements.js";
7
+ import { docxHandlers } from "./handlers.js";
8
+ import { freshParagraphId, paragraphsOf } from "./ids.js";
9
+ import { DocxModel } from "./model.js";
10
+ import { issueCollector } from "./operations.js";
11
+ import { docxOperationSchemas } from "./schemas.js";
12
+ import { namespacePatches } from "./write.js";
13
+ /*
14
+ * The DOCX edit engine: the package layer under a block index of the body
15
+ * story. It runs inside the OOXML edit worker in the browser and directly
16
+ * in Node tests. It never lays out: elements carry no geometry, and the
17
+ * session joins the renderer's runs on the main thread.
18
+ *
19
+ * Paragraph ids: a paragraph with a `w14:paraId` keeps it; one without is
20
+ * numbered from its position when the document opens, and the engine then
21
+ * tracks those ids by document order (`#unauthored`), writing a
22
+ * `w14:paraId` on every paragraph it rebuilds or creates. The shown copy
23
+ * carries an id on every paragraph, so the viewer's runs name the same
24
+ * paragraphs; the saved file carries ids only where the session wrote.
25
+ */
26
+ /**
27
+ * The part a shown copy carries with the ids the engine owns, so a copy
28
+ * restored as a base knows which `w14:paraId` values to leave out of a
29
+ * saved file. An XML part: its content type is the one every package
30
+ * declares for the extension, so `[Content_Types].xml` stays untouched.
31
+ */
32
+ const UNAUTHORED_PART = "/webdoc/unauthored.xml";
33
+ const UNAUTHORED_NS = "urn:web-doc:docx-edit";
34
+ function unauthoredPartXml(ids) {
35
+ const items = [...ids].map((id) => `<p id="${id}"/>`).join("");
36
+ return new TextEncoder().encode(`<?xml version="1.0" encoding="UTF-8" standalone="yes"?><unauthored xmlns="${UNAUTHORED_NS}">${items}</unauthored>`);
37
+ }
38
+ function unauthoredIdsOf(bytes) {
39
+ const text = new TextDecoder().decode(bytes);
40
+ return [...text.matchAll(/<p id="([0-9A-F]{8})"\/>/g)].map((m) => m[1]);
41
+ }
42
+ export class DocxEditEngine {
43
+ schemas = docxOperationSchemas;
44
+ #original;
45
+ #limits;
46
+ #assets = new AssetStore();
47
+ #pkg;
48
+ #model;
49
+ /** The styles and theme, read once: no operation changes them. */
50
+ #styles;
51
+ /** Ids of the paragraphs without `w14:paraId`, in document order; undefined until read from the bytes. */
52
+ #unauthored;
53
+ #disposed = false;
54
+ #stateId = 0;
55
+ constructor(original, pkg, limits) {
56
+ this.#original = original;
57
+ this.#pkg = pkg;
58
+ this.#limits = limits;
59
+ }
60
+ /** Opens the package and reads the block index, so a broken document fails here. */
61
+ static async open(bytes, limits, signal) {
62
+ const pkg = await OoxmlPackage.open(bytes, {
63
+ limits,
64
+ ...(signal ? { signal } : {}),
65
+ });
66
+ const engine = new DocxEditEngine(bytes, pkg, limits);
67
+ await engine.model(signal);
68
+ return engine;
69
+ }
70
+ /** The package behind the engine, for tests and the operations. */
71
+ get package() {
72
+ return this.#pkg;
73
+ }
74
+ /** The engine has no pages of its own; the renderer counts them. */
75
+ get pageCount() {
76
+ return 0;
77
+ }
78
+ /** The block index at the current revision. */
79
+ model(signal) {
80
+ this.#assertAlive();
81
+ const revision = this.#pkg.revision;
82
+ if (!this.#model)
83
+ return this.#startModel(signal);
84
+ return this.#model.then((current) => current.revision === revision ? current : this.#startModel(signal));
85
+ }
86
+ #startModel(signal) {
87
+ const pending = DocxModel.load(this.#pkg, signal, this.#unauthored, this.#styles).then((model) => {
88
+ this.#unauthored ??= [...model.unauthoredIds];
89
+ this.#styles ??= model.styles;
90
+ return model;
91
+ });
92
+ this.#model = pending;
93
+ pending.catch(() => {
94
+ if (this.#model === pending)
95
+ this.#model = undefined;
96
+ });
97
+ return pending;
98
+ }
99
+ async validate(operations, signal) {
100
+ const issues = [];
101
+ const base = await this.#context(0, signal);
102
+ for (const [index, operation] of operations.entries()) {
103
+ const context = { ...base, operationIndex: index };
104
+ const handler = docxHandlers.get(operation.op);
105
+ if (!handler) {
106
+ issues.push({
107
+ operationIndex: index,
108
+ path: "/op",
109
+ code: "unknown-operation",
110
+ message: `Unknown operation ${operation.op}`,
111
+ });
112
+ continue;
113
+ }
114
+ // A same-batch reference names an element that does not exist yet:
115
+ // its target is checked when the batch is applied, the rest now.
116
+ const referenced = referenceFields(operation);
117
+ const forward = referenced.find((entry) => entry.reference >= index);
118
+ if (forward) {
119
+ issues.push({
120
+ operationIndex: index,
121
+ path: `/${forward.field}`,
122
+ code: "unknown-target",
123
+ message: `"${forward.value}" must refer to an earlier operation`,
124
+ });
125
+ continue;
126
+ }
127
+ const collect = issueCollector(index, issues);
128
+ const skipped = referenced.map((entry) => `/${entry.field}`);
129
+ await handler.validate(operation, context, referenced.length === 0
130
+ ? collect
131
+ : (path, code, message) => {
132
+ if (!skipped.some((prefix) => path === prefix || path.startsWith(`${prefix}/`)) &&
133
+ !path.startsWith("/range"))
134
+ collect(path, code, message);
135
+ });
136
+ }
137
+ return issues;
138
+ }
139
+ /** Plain operation arrays, as the unit tests pass them, become the next batch. */
140
+ async apply(input, signal) {
141
+ const batch = Array.isArray(input)
142
+ ? { stateId: this.#nextStateId(), operations: input }
143
+ : input;
144
+ this.#stateId = Math.max(this.#stateId, batch.stateId);
145
+ const snapshot = this.#pkg.snapshot();
146
+ const unauthored = this.#unauthored ? [...this.#unauthored] : undefined;
147
+ const createdIds = [];
148
+ const removedIds = [];
149
+ const warnings = [];
150
+ const createdByOperation = [];
151
+ const issued = new Set();
152
+ const remappedIds = {};
153
+ let reflowFrom;
154
+ try {
155
+ for (const [index, raw] of batch.operations.entries()) {
156
+ throwIfAborted(signal);
157
+ const operation = resolveReferences(raw, createdByOperation, index);
158
+ const handler = docxHandlers.get(operation.op);
159
+ if (!handler)
160
+ throw new ViewerError("invalid-operation", `Unknown operation ${operation.op}`);
161
+ const context = await this.#context(batch.stateId, signal, index, issued);
162
+ const issues = [];
163
+ await handler.validate(operation, context, issueCollector(index, issues));
164
+ if (issues.length > 0)
165
+ throw invalidOperationError(issues);
166
+ const result = await handler.apply(operation, context);
167
+ this.#model = undefined;
168
+ const gone = new Set([
169
+ ...(result.stamped ?? []),
170
+ ...(result.removedParagraphIds ?? []),
171
+ ]);
172
+ if (gone.size > 0 && this.#unauthored)
173
+ this.#unauthored = this.#unauthored.filter((id) => !gone.has(id));
174
+ createdByOperation.push([...result.createdIds]);
175
+ createdIds.push(...result.createdIds);
176
+ removedIds.push(...(result.removedIds ?? []));
177
+ warnings.push(...result.warnings);
178
+ reflowFrom ??= result.reflowFrom;
179
+ // A table renamed twice in one batch maps its first id to its last.
180
+ for (const [from, to] of Object.entries(result.remappedIds ?? {})) {
181
+ const origin = Object.entries(remappedIds).find(([, value]) => value === from)?.[0] ?? from;
182
+ remappedIds[origin] = to;
183
+ }
184
+ }
185
+ // The index after the batch; an inconsistency surfaces here and
186
+ // rolls the batch back like any other failure.
187
+ await this.model(signal);
188
+ }
189
+ catch (error) {
190
+ // Everything the batch did, ids included, is undone.
191
+ this.#pkg.restore(snapshot);
192
+ this.#unauthored = unauthored;
193
+ this.#model = undefined;
194
+ throw error;
195
+ }
196
+ this.#pkg.release(snapshot);
197
+ return {
198
+ createdIds,
199
+ removedIds,
200
+ // A flow document reflows from the first changed paragraph; the host
201
+ // turns it into pages.
202
+ changedPages: [],
203
+ ...(reflowFrom === undefined ? {} : { reflowFrom }),
204
+ ...(Object.keys(remappedIds).length > 0 ? { remappedIds } : {}),
205
+ warnings,
206
+ };
207
+ }
208
+ #nextStateId() {
209
+ this.#stateId += 1;
210
+ return this.#stateId;
211
+ }
212
+ async #context(stateId, signal, operationIndex = 0, issued = new Set()) {
213
+ const model = await this.model(signal);
214
+ const taken = new Set([...model.takenIds, ...issued]);
215
+ let count = 0;
216
+ return {
217
+ pkg: this.#pkg,
218
+ model,
219
+ limits: this.#limits,
220
+ assets: this.#assets,
221
+ stateId,
222
+ operationIndex,
223
+ freshParagraphId: () => {
224
+ const id = freshParagraphId(stateId, operationIndex, count, taken);
225
+ count += 1;
226
+ issued.add(id);
227
+ taken.add(id);
228
+ return id;
229
+ },
230
+ };
231
+ }
232
+ async materialize(purposeOrSignal = "show", options = {}, signal = new AbortController().signal) {
233
+ return (await this.materializeDocument(purposeOrSignal, options, signal))
234
+ .bytes;
235
+ }
236
+ /**
237
+ * `save` is the package as the session changed it: ids written only on
238
+ * the paragraphs the session rebuilt or created. `show` also stamps every
239
+ * other paragraph with the id the engine knows it by, in a copy, so the
240
+ * display pre-pass and the viewer's runs name the engine's paragraphs
241
+ * after edits that moved paragraphs around. Without changes both are
242
+ * the original bytes.
243
+ */
244
+ async materializeDocument(purposeOrSignal = "show", _options = {}, signal = new AbortController().signal) {
245
+ this.#assertAlive();
246
+ const purpose = purposeOrSignal instanceof AbortSignal ? "show" : purposeOrSignal;
247
+ const own = purposeOrSignal instanceof AbortSignal ? purposeOrSignal : signal;
248
+ const bytes = await this.#pkg.save({}, own);
249
+ const fromShownBase = this.#pkg.has(UNAUTHORED_PART);
250
+ if (purpose === "save")
251
+ return {
252
+ bytes: fromShownBase ? await this.#unstamped(bytes, own) : bytes,
253
+ warnings: [],
254
+ };
255
+ if (this.#pkg.changedParts.length === 0 && !fromShownBase)
256
+ return { bytes, warnings: [] };
257
+ return { bytes: await this.#stamped(bytes, own), warnings: [] };
258
+ }
259
+ /**
260
+ * A copy of `bytes` with a `w14:paraId` on every paragraph of the main
261
+ * part and the engine's own ids listed in a part of their own.
262
+ */
263
+ async #stamped(bytes, signal) {
264
+ const model = await this.model(signal);
265
+ const copy = await OoxmlPackage.open(bytes, {
266
+ limits: this.#limits,
267
+ signal,
268
+ });
269
+ const part = await copy.xml(model.mainPart, signal);
270
+ const owned = this.#unauthored ?? [];
271
+ // Ids already written (by an edit, or by the shown base this state
272
+ // came from) stay; the rest go on the unmarked paragraphs in order.
273
+ const attributed = new Set();
274
+ for (const paragraph of paragraphsOf(part)) {
275
+ const id = part.attribute(paragraph, "w14:paraId");
276
+ if (id)
277
+ attributed.add(id.toUpperCase());
278
+ }
279
+ const queue = owned.filter((id) => !attributed.has(id));
280
+ const items = [];
281
+ for (const paragraph of paragraphsOf(part)) {
282
+ if (part.attribute(paragraph, "w14:paraId"))
283
+ continue;
284
+ const id = queue.shift();
285
+ if (id === undefined)
286
+ throw new ViewerError("internal", "The document has more unmarked paragraphs than the session knows");
287
+ items.push(patches.setAttribute(part, paragraph, "w14:paraId", id));
288
+ }
289
+ const list = owned.length > 0 ? unauthoredPartXml(owned) : undefined;
290
+ if (items.length === 0) {
291
+ // Already the shown form of this state: nothing to write.
292
+ const current = copy.has(UNAUTHORED_PART)
293
+ ? await copy.part(UNAUTHORED_PART, signal)
294
+ : undefined;
295
+ if (current === undefined
296
+ ? list === undefined
297
+ : list !== undefined && sameBytes(current, list))
298
+ return bytes;
299
+ }
300
+ const transaction = copy.transaction();
301
+ if (items.length > 0)
302
+ transaction.patch(part, [...namespacePatches(part), ...items]);
303
+ if (list)
304
+ transaction.setPart(UNAUTHORED_PART, list, "application/xml");
305
+ else if (copy.has(UNAUTHORED_PART))
306
+ transaction.removePart(UNAUTHORED_PART);
307
+ await transaction.commit(signal);
308
+ return copy.save({}, signal);
309
+ }
310
+ /**
311
+ * A copy of `bytes` without the ids a shown base brought: the engine's
312
+ * own `w14:paraId` values go, the ones an edit wrote stay, and the list
313
+ * part goes with them.
314
+ */
315
+ async #unstamped(bytes, signal) {
316
+ const model = await this.model(signal);
317
+ const copy = await OoxmlPackage.open(bytes, {
318
+ limits: this.#limits,
319
+ signal,
320
+ });
321
+ const part = await copy.xml(model.mainPart, signal);
322
+ const owned = model.unauthoredSet;
323
+ const items = [];
324
+ for (const paragraph of paragraphsOf(part)) {
325
+ const id = part.attribute(paragraph, "w14:paraId")?.toUpperCase();
326
+ if (id && owned.has(id))
327
+ items.push(patches.removeAttribute(part, paragraph, "w14:paraId"));
328
+ }
329
+ const transaction = copy.transaction();
330
+ if (items.length > 0)
331
+ transaction.patch(part, items);
332
+ if (copy.has(UNAUTHORED_PART))
333
+ transaction.removePart(UNAUTHORED_PART);
334
+ await transaction.commit(signal);
335
+ return copy.save({}, signal);
336
+ }
337
+ async restore(input, signal) {
338
+ const target = Array.isArray(input)
339
+ ? {
340
+ batches: input.map((operations, index) => ({ stateId: index + 1, operations })),
341
+ }
342
+ : input;
343
+ this.#assertAlive();
344
+ this.#pkg = await OoxmlPackage.open(target.base ?? this.#original, {
345
+ limits: this.#limits,
346
+ signal,
347
+ });
348
+ // The original has the ids its positions give; a base is a shown copy
349
+ // with every paragraph marked and the engine's own ids listed.
350
+ this.#unauthored = undefined;
351
+ if (this.#pkg.has(UNAUTHORED_PART))
352
+ this.#unauthored = unauthoredIdsOf(await this.#pkg.part(UNAUTHORED_PART, signal));
353
+ this.#model = undefined;
354
+ await this.model(signal);
355
+ for (const batch of target.batches)
356
+ await this.apply(batch, signal);
357
+ }
358
+ async putAsset(id, data, _signal) {
359
+ this.#assets.set(id, data);
360
+ }
361
+ /**
362
+ * Every element of the body story, in document order, without geometry.
363
+ * `pageIndex` and `intersects` are the session's to apply once the runs
364
+ * are joined; `kinds` filters here.
365
+ */
366
+ async getElements(query, signal) {
367
+ const model = await this.model(signal);
368
+ const out = [];
369
+ for (const record of model.records) {
370
+ if (query.kinds && !query.kinds.includes(record.kind))
371
+ continue;
372
+ out.push(toElement(model, record));
373
+ }
374
+ return out;
375
+ }
376
+ async getElement(id, signal) {
377
+ const record = await this.locate(id, signal);
378
+ return record ? toElement(await this.model(signal), record) : undefined;
379
+ }
380
+ /** The record an element id names, at the current revision. */
381
+ async locate(id, signal) {
382
+ return (await this.model(signal)).byId.get(id);
383
+ }
384
+ /** Hit-testing needs the renderer's geometry; the session answers it. */
385
+ async elementsAt(_pageIndex, _point, _signal) {
386
+ return [];
387
+ }
388
+ /**
389
+ * Matches in paragraph text, in document order, without rectangles or
390
+ * pages: the session adds those from the renderer's runs.
391
+ */
392
+ async findText(query, options, signal) {
393
+ if (query.length === 0)
394
+ return [];
395
+ const model = await this.model(signal);
396
+ // A case-insensitive regular expression keeps offsets on the original
397
+ // text, where a lower-cased copy can change length.
398
+ const needle = new RegExp(query.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), options.caseSensitive ? "g" : "gi");
399
+ const limit = options.maxResults ?? Number.POSITIVE_INFINITY;
400
+ const targets = [];
401
+ for (const record of model.records) {
402
+ if (record.kind !== "paragraph")
403
+ continue;
404
+ const text = record.text.text;
405
+ if (!text)
406
+ continue;
407
+ needle.lastIndex = 0;
408
+ while (targets.length < limit) {
409
+ const match = needle.exec(text);
410
+ if (!match)
411
+ break;
412
+ const at = match.index;
413
+ const end = at + match[0].length;
414
+ if (match[0].length === 0)
415
+ needle.lastIndex += 1;
416
+ targets.push({
417
+ pageIndex: NO_PAGE,
418
+ text: text.slice(at, end),
419
+ rects: [],
420
+ elementIds: [record.elementId],
421
+ ranges: [
422
+ {
423
+ start: { elementId: record.elementId, offset: at },
424
+ end: { elementId: record.elementId, offset: end },
425
+ },
426
+ ],
427
+ });
428
+ }
429
+ if (targets.length >= limit)
430
+ break;
431
+ }
432
+ return targets;
433
+ }
434
+ async dispose() {
435
+ this.#disposed = true;
436
+ this.#model = undefined;
437
+ }
438
+ #assertAlive() {
439
+ if (this.#disposed)
440
+ throw new ViewerError("lifecycle-error", "The DOCX engine was disposed");
441
+ }
442
+ }
443
+ const REFERENCE_FIELDS = ["target", "before", "after"];
444
+ function referenceFields(operation) {
445
+ const out = [];
446
+ for (const field of REFERENCE_FIELDS) {
447
+ const value = operation[field];
448
+ if (typeof value !== "string")
449
+ continue;
450
+ const reference = parseReference(value);
451
+ if (reference !== undefined)
452
+ out.push({ field, value, reference });
453
+ }
454
+ return out;
455
+ }
456
+ /** Replaces `"$<n>"` references with the first id operation `n` created. */
457
+ function resolveReferences(operation, created, index) {
458
+ let resolved = operation;
459
+ for (const { field, value, reference } of referenceFields(operation)) {
460
+ const id = reference < index ? created[reference]?.[0] : undefined;
461
+ if (!id)
462
+ throw new ViewerError("invalid-operation", `Operation ${index} refers to "${value}", which created nothing`, { details: { operationIndex: index, target: value } });
463
+ resolved = { ...resolved, [field]: id };
464
+ }
465
+ return resolved;
466
+ }
467
+ function sameBytes(a, b) {
468
+ return a.length === b.length && a.every((byte, index) => byte === b[index]);
469
+ }
470
+ function throwIfAborted(signal) {
471
+ if (signal?.aborted)
472
+ throw new ViewerError("aborted", "The operation was aborted");
473
+ }
@@ -0,0 +1,3 @@
1
+ import type { DocxOperationHandler } from "./operations.js";
2
+ /** Handlers by operation name; `IMPLEMENTED_OPERATIONS` in schemas.ts lists the same names. */
3
+ export declare const docxHandlers: ReadonlyMap<string, DocxOperationHandler>;
@@ -0,0 +1,15 @@
1
+ import { deleteElementHandler, insertImageHandler, insertParagraphHandler, moveElementHandler, } from "./structure-ops.js";
2
+ import { insertTableHandler, setTableCellHandler } from "./table-ops.js";
3
+ import { replaceTextHandler, setParagraphStyleHandler, setTextStyleHandler, } from "./text-ops.js";
4
+ /** Handlers by operation name; `IMPLEMENTED_OPERATIONS` in schemas.ts lists the same names. */
5
+ export const docxHandlers = new Map([
6
+ ["replaceText", replaceTextHandler],
7
+ ["setTextStyle", setTextStyleHandler],
8
+ ["setParagraphStyle", setParagraphStyleHandler],
9
+ ["insertParagraph", insertParagraphHandler],
10
+ ["deleteElement", deleteElementHandler],
11
+ ["moveElement", moveElementHandler],
12
+ ["insertImage", insertImageHandler],
13
+ ["insertTable", insertTableHandler],
14
+ ["setTableCell", setTableCellHandler],
15
+ ]);
@@ -0,0 +1,45 @@
1
+ import type { XmlElement, XmlPart } from "../ooxml/xml.js";
2
+ export declare const W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main";
3
+ export declare const W14_NS = "http://schemas.microsoft.com/office/word/2010/wordml";
4
+ export declare const OFFICE_RELATIONSHIPS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/";
5
+ /** Story parts whose paragraphs share the id sequence, in the order they are numbered. */
6
+ export declare const STORY_RELATIONSHIP_TYPES: readonly ["http://schemas.openxmlformats.org/officeDocument/2006/relationships/header", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/footer", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/footnotes", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/endnotes"];
7
+ /** Prefix of the hidden bookmarks that carry paragraph ids. */
8
+ export declare const PARAGRAPH_BOOKMARK_PREFIX = "_wd";
9
+ /** First bookmark id the pre-pass uses, above what Word writes. */
10
+ export declare const BOOKMARK_ID_BASE = 7000000;
11
+ /** Ids and bookmark numbers shared by every story part of one document. */
12
+ export interface DocxIdState {
13
+ readonly taken: Set<string>;
14
+ generated: number;
15
+ bookmarkId: number;
16
+ }
17
+ export declare function newIdState(): DocxIdState;
18
+ /**
19
+ * The paragraph id generated for the `index`-th paragraph without one, in
20
+ * document order, skipping ids the document already uses.
21
+ */
22
+ export declare function generatedParagraphId(index: number, taken: ReadonlySet<string>): string;
23
+ /** Every `w:p` of a part in document order, nested ones included. */
24
+ export declare function paragraphsOf(part: XmlPart): readonly XmlElement[];
25
+ /** Collects the ids and bookmark numbers a part already uses. */
26
+ export declare function collectIds(part: XmlPart, state: DocxIdState): void;
27
+ export interface ParagraphId {
28
+ readonly paragraph: XmlElement;
29
+ /** Eight upper-case hex digits. */
30
+ readonly id: string;
31
+ /** True when the file carries the id as `w14:paraId`. */
32
+ readonly authored: boolean;
33
+ }
34
+ /**
35
+ * The id of every paragraph of a part, numbering the unmarked ones from
36
+ * where `state` stands. Call it for the parts in numbering order after
37
+ * `collectIds` has seen all of them.
38
+ */
39
+ export declare function assignParagraphIds(part: XmlPart, state: DocxIdState): readonly ParagraphId[];
40
+ /**
41
+ * A fresh id for the `index`-th paragraph an operation creates: derived
42
+ * from the batch's state id and the operation's position, so a replay of
43
+ * the same batches issues the same ids, and skipping any id in use.
44
+ */
45
+ export declare function freshParagraphId(stateId: number, operationIndex: number, index: number, taken: ReadonlySet<string>): string;
@@ -0,0 +1,101 @@
1
+ /*
2
+ * Paragraph ids shared by the display pre-pass and the DOCX edit engine.
3
+ * A paragraph's id is its `w14:paraId` when the file has one; otherwise a
4
+ * deterministic id from its position among the unmarked paragraphs of the
5
+ * document, numbered part by part in one sequence (the main part first,
6
+ * then headers, footers, footnotes and endnotes in relationship order) and
7
+ * skipping every id the document already uses. Both sides walk the same
8
+ * paragraphs in the same order, so the pre-pass's bookmarks and the
9
+ * engine's element ids agree on the original bytes.
10
+ */
11
+ export const W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main";
12
+ export const W14_NS = "http://schemas.microsoft.com/office/word/2010/wordml";
13
+ export const OFFICE_RELATIONSHIPS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/";
14
+ /** Story parts whose paragraphs share the id sequence, in the order they are numbered. */
15
+ export const STORY_RELATIONSHIP_TYPES = [
16
+ `${OFFICE_RELATIONSHIPS}header`,
17
+ `${OFFICE_RELATIONSHIPS}footer`,
18
+ `${OFFICE_RELATIONSHIPS}footnotes`,
19
+ `${OFFICE_RELATIONSHIPS}endnotes`,
20
+ ];
21
+ /** Prefix of the hidden bookmarks that carry paragraph ids. */
22
+ export const PARAGRAPH_BOOKMARK_PREFIX = "_wd";
23
+ /** First bookmark id the pre-pass uses, above what Word writes. */
24
+ export const BOOKMARK_ID_BASE = 7_000_000;
25
+ /** First generated id; Word keeps `w14:paraId` below 0x80000000. */
26
+ const ID_BASE = 0x1a000000;
27
+ const ID_STEP = 0x9e37;
28
+ export function newIdState() {
29
+ return { taken: new Set(), generated: 0, bookmarkId: BOOKMARK_ID_BASE };
30
+ }
31
+ /**
32
+ * The paragraph id generated for the `index`-th paragraph without one, in
33
+ * document order, skipping ids the document already uses.
34
+ */
35
+ export function generatedParagraphId(index, taken) {
36
+ let candidate = (ID_BASE + index * ID_STEP) % 0x80000000;
37
+ for (;;) {
38
+ const value = candidate.toString(16).toUpperCase().padStart(8, "0");
39
+ if (!taken.has(value))
40
+ return value;
41
+ candidate = (candidate + 1) % 0x80000000;
42
+ }
43
+ }
44
+ /** Every `w:p` of a part in document order, nested ones included. */
45
+ export function paragraphsOf(part) {
46
+ return part.findAll("p").filter((node) => node.namespace === W_NS);
47
+ }
48
+ /** Collects the ids and bookmark numbers a part already uses. */
49
+ export function collectIds(part, state) {
50
+ for (const node of part.findAll("bookmarkStart")) {
51
+ if (node.namespace !== W_NS)
52
+ continue;
53
+ const id = Number(part.attribute(node, "w:id") ?? NaN);
54
+ if (Number.isInteger(id) && id >= state.bookmarkId)
55
+ state.bookmarkId = id + 1;
56
+ const name = part.attribute(node, "w:name") ?? "";
57
+ if (name.startsWith(PARAGRAPH_BOOKMARK_PREFIX))
58
+ state.taken.add(name.slice(PARAGRAPH_BOOKMARK_PREFIX.length).toUpperCase());
59
+ }
60
+ for (const paragraph of paragraphsOf(part)) {
61
+ const existing = part.attribute(paragraph, "w14:paraId");
62
+ if (existing)
63
+ state.taken.add(existing.toUpperCase());
64
+ }
65
+ }
66
+ /**
67
+ * The id of every paragraph of a part, numbering the unmarked ones from
68
+ * where `state` stands. Call it for the parts in numbering order after
69
+ * `collectIds` has seen all of them.
70
+ */
71
+ export function assignParagraphIds(part, state) {
72
+ const out = [];
73
+ for (const paragraph of paragraphsOf(part)) {
74
+ const authored = part.attribute(paragraph, "w14:paraId");
75
+ if (authored) {
76
+ out.push({ paragraph, id: authored.toUpperCase(), authored: true });
77
+ continue;
78
+ }
79
+ const id = generatedParagraphId(state.generated, state.taken);
80
+ state.taken.add(id);
81
+ state.generated += 1;
82
+ out.push({ paragraph, id, authored: false });
83
+ }
84
+ return out;
85
+ }
86
+ /**
87
+ * A fresh id for the `index`-th paragraph an operation creates: derived
88
+ * from the batch's state id and the operation's position, so a replay of
89
+ * the same batches issues the same ids, and skipping any id in use.
90
+ */
91
+ export function freshParagraphId(stateId, operationIndex, index, taken) {
92
+ const base = (0x2a000000 + stateId * 0x10000 + operationIndex * 0x100 + index) %
93
+ 0x80000000;
94
+ let candidate = base;
95
+ for (;;) {
96
+ const value = candidate.toString(16).toUpperCase().padStart(8, "0");
97
+ if (!taken.has(value))
98
+ return value;
99
+ candidate = (candidate + 1) % 0x80000000;
100
+ }
101
+ }