web-doc 0.6.2 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. package/THIRD_PARTY_NOTICES.md +26 -1
  2. package/dist/adapters/docx-images.d.ts +13 -4
  3. package/dist/adapters/docx-images.js +3 -0
  4. package/dist/adapters/docx-paragraphs.d.ts +65 -0
  5. package/dist/adapters/docx-paragraphs.js +165 -0
  6. package/dist/adapters/docx-prepass.d.ts +17 -0
  7. package/dist/adapters/docx-prepass.js +230 -0
  8. package/dist/adapters/office.d.ts +31 -5
  9. package/dist/adapters/office.js +60 -29
  10. package/dist/adapters/pdf.d.ts +9 -0
  11. package/dist/adapters/pdf.js +10 -0
  12. package/dist/assets/pdfium/pdfium.wasm +0 -0
  13. package/dist/contracts.d.ts +47 -2
  14. package/dist/edit/assets.d.ts +23 -0
  15. package/dist/edit/assets.js +75 -0
  16. package/dist/edit/docx/elements.d.ts +6 -0
  17. package/dist/edit/docx/elements.js +90 -0
  18. package/dist/edit/docx/engine.d.ts +51 -0
  19. package/dist/edit/docx/engine.js +473 -0
  20. package/dist/edit/docx/handlers.d.ts +3 -0
  21. package/dist/edit/docx/handlers.js +15 -0
  22. package/dist/edit/docx/ids.d.ts +45 -0
  23. package/dist/edit/docx/ids.js +101 -0
  24. package/dist/edit/docx/model.d.ts +92 -0
  25. package/dist/edit/docx/model.js +339 -0
  26. package/dist/edit/docx/operations.d.ts +37 -0
  27. package/dist/edit/docx/operations.js +3 -0
  28. package/dist/edit/docx/provider.d.ts +13 -0
  29. package/dist/edit/docx/provider.js +34 -0
  30. package/dist/edit/docx/schemas.d.ts +6 -0
  31. package/dist/edit/docx/schemas.js +192 -0
  32. package/dist/edit/docx/session.d.ts +37 -0
  33. package/dist/edit/docx/session.js +426 -0
  34. package/dist/edit/docx/structure-ops.d.ts +6 -0
  35. package/dist/edit/docx/structure-ops.js +494 -0
  36. package/dist/edit/docx/style.d.ts +45 -0
  37. package/dist/edit/docx/style.js +375 -0
  38. package/dist/edit/docx/table-ops.d.ts +4 -0
  39. package/dist/edit/docx/table-ops.js +219 -0
  40. package/dist/edit/docx/text-ops.d.ts +18 -0
  41. package/dist/edit/docx/text-ops.js +458 -0
  42. package/dist/edit/docx/text.d.ts +34 -0
  43. package/dist/edit/docx/text.js +263 -0
  44. package/dist/edit/docx/types.d.ts +191 -0
  45. package/dist/edit/docx/types.js +1 -0
  46. package/dist/edit/docx/write.d.ts +50 -0
  47. package/dist/edit/docx/write.js +352 -0
  48. package/dist/edit/engine.d.ts +117 -0
  49. package/dist/edit/engine.js +1 -0
  50. package/dist/edit/history.d.ts +53 -0
  51. package/dist/edit/history.js +101 -0
  52. package/dist/edit/ooxml/names.d.ts +20 -0
  53. package/dist/edit/ooxml/names.js +61 -0
  54. package/dist/edit/ooxml/opc.d.ts +50 -0
  55. package/dist/edit/ooxml/opc.js +150 -0
  56. package/dist/edit/ooxml/package.d.ts +82 -0
  57. package/dist/edit/ooxml/package.js +233 -0
  58. package/dist/edit/ooxml/patch.d.ts +51 -0
  59. package/dist/edit/ooxml/patch.js +250 -0
  60. package/dist/edit/ooxml/transaction.d.ts +39 -0
  61. package/dist/edit/ooxml/transaction.js +316 -0
  62. package/dist/edit/ooxml/worker.d.ts +8 -0
  63. package/dist/edit/ooxml/worker.js +12 -0
  64. package/dist/edit/ooxml/writer.d.ts +21 -0
  65. package/dist/edit/ooxml/writer.js +187 -0
  66. package/dist/edit/ooxml/xml.d.ts +74 -0
  67. package/dist/edit/ooxml/xml.js +451 -0
  68. package/dist/edit/ooxml/zip.d.ts +54 -0
  69. package/dist/edit/ooxml/zip.js +280 -0
  70. package/dist/edit/operations.d.ts +19 -0
  71. package/dist/edit/operations.js +137 -0
  72. package/dist/edit/pdf/engine/compact.d.ts +6 -0
  73. package/dist/edit/pdf/engine/compact.js +442 -0
  74. package/dist/edit/pdf/engine/document.d.ts +95 -0
  75. package/dist/edit/pdf/engine/document.js +868 -0
  76. package/dist/edit/pdf/engine/elements.d.ts +45 -0
  77. package/dist/edit/pdf/engine/elements.js +313 -0
  78. package/dist/edit/pdf/engine/existing-text.d.ts +4 -0
  79. package/dist/edit/pdf/engine/existing-text.js +424 -0
  80. package/dist/edit/pdf/engine/fonts.d.ts +78 -0
  81. package/dist/edit/pdf/engine/fonts.js +466 -0
  82. package/dist/edit/pdf/engine/geometry.d.ts +36 -0
  83. package/dist/edit/pdf/engine/geometry.js +97 -0
  84. package/dist/edit/pdf/engine/handler.d.ts +21 -0
  85. package/dist/edit/pdf/engine/handler.js +108 -0
  86. package/dist/edit/pdf/engine/images.d.ts +33 -0
  87. package/dist/edit/pdf/engine/images.js +188 -0
  88. package/dist/edit/pdf/engine/layout.d.ts +23 -0
  89. package/dist/edit/pdf/engine/layout.js +177 -0
  90. package/dist/edit/pdf/engine/operations.d.ts +67 -0
  91. package/dist/edit/pdf/engine/operations.js +3 -0
  92. package/dist/edit/pdf/engine/pages.d.ts +6 -0
  93. package/dist/edit/pdf/engine/pages.js +98 -0
  94. package/dist/edit/pdf/engine/pdfium.d.ts +195 -0
  95. package/dist/edit/pdf/engine/pdfium.js +249 -0
  96. package/dist/edit/pdf/engine/shapes.d.ts +4 -0
  97. package/dist/edit/pdf/engine/shapes.js +183 -0
  98. package/dist/edit/pdf/engine/tables.d.ts +44 -0
  99. package/dist/edit/pdf/engine/tables.js +315 -0
  100. package/dist/edit/pdf/engine/text-box.d.ts +71 -0
  101. package/dist/edit/pdf/engine/text-box.js +317 -0
  102. package/dist/edit/pdf/engine/text-layout.d.ts +28 -0
  103. package/dist/edit/pdf/engine/text-layout.js +67 -0
  104. package/dist/edit/pdf/engine/transform.d.ts +5 -0
  105. package/dist/edit/pdf/engine/transform.js +137 -0
  106. package/dist/edit/pdf/provider.d.ts +35 -0
  107. package/dist/edit/pdf/provider.js +92 -0
  108. package/dist/edit/pdf/range-map.d.ts +14 -0
  109. package/dist/edit/pdf/range-map.js +107 -0
  110. package/dist/edit/pdf/schemas.d.ts +11 -0
  111. package/dist/edit/pdf/schemas.js +291 -0
  112. package/dist/edit/pdf/selection.d.ts +5 -0
  113. package/dist/edit/pdf/selection.js +152 -0
  114. package/dist/edit/pdf/session.d.ts +53 -0
  115. package/dist/edit/pdf/session.js +273 -0
  116. package/dist/edit/pdf/types.d.ts +336 -0
  117. package/dist/edit/pdf/types.js +1 -0
  118. package/dist/edit/pptx/elements.d.ts +74 -0
  119. package/dist/edit/pptx/elements.js +301 -0
  120. package/dist/edit/pptx/engine.d.ts +37 -0
  121. package/dist/edit/pptx/engine.js +466 -0
  122. package/dist/edit/pptx/geometry.d.ts +72 -0
  123. package/dist/edit/pptx/geometry.js +193 -0
  124. package/dist/edit/pptx/handler.d.ts +7 -0
  125. package/dist/edit/pptx/handler.js +91 -0
  126. package/dist/edit/pptx/handlers.d.ts +3 -0
  127. package/dist/edit/pptx/handlers.js +21 -0
  128. package/dist/edit/pptx/image-table-ops.d.ts +5 -0
  129. package/dist/edit/pptx/image-table-ops.js +233 -0
  130. package/dist/edit/pptx/model.d.ts +69 -0
  131. package/dist/edit/pptx/model.js +170 -0
  132. package/dist/edit/pptx/operations.d.ts +50 -0
  133. package/dist/edit/pptx/operations.js +11 -0
  134. package/dist/edit/pptx/provider.d.ts +17 -0
  135. package/dist/edit/pptx/provider.js +40 -0
  136. package/dist/edit/pptx/schemas.d.ts +6 -0
  137. package/dist/edit/pptx/schemas.js +220 -0
  138. package/dist/edit/pptx/session.d.ts +44 -0
  139. package/dist/edit/pptx/session.js +114 -0
  140. package/dist/edit/pptx/shape-ops.d.ts +10 -0
  141. package/dist/edit/pptx/shape-ops.js +486 -0
  142. package/dist/edit/pptx/slide-ops.d.ts +6 -0
  143. package/dist/edit/pptx/slide-ops.js +353 -0
  144. package/dist/edit/pptx/style.d.ts +31 -0
  145. package/dist/edit/pptx/style.js +181 -0
  146. package/dist/edit/pptx/text-ops.d.ts +13 -0
  147. package/dist/edit/pptx/text-ops.js +267 -0
  148. package/dist/edit/pptx/text-write.d.ts +67 -0
  149. package/dist/edit/pptx/text-write.js +293 -0
  150. package/dist/edit/pptx/text.d.ts +41 -0
  151. package/dist/edit/pptx/text.js +99 -0
  152. package/dist/edit/pptx/types.d.ts +232 -0
  153. package/dist/edit/pptx/types.js +1 -0
  154. package/dist/edit/schema.d.ts +11 -0
  155. package/dist/edit/schema.js +275 -0
  156. package/dist/edit/session.d.ts +66 -0
  157. package/dist/edit/session.js +701 -0
  158. package/dist/edit/sessions.d.ts +8 -0
  159. package/dist/edit/sessions.js +1 -0
  160. package/dist/edit/types.d.ts +276 -0
  161. package/dist/edit/types.js +1 -0
  162. package/dist/edit/worker-engine.d.ts +31 -0
  163. package/dist/edit/worker-engine.js +91 -0
  164. package/dist/fonts/THIRD_PARTY_NOTICES.md +3 -0
  165. package/dist/fonts/manifest.json +6 -1
  166. package/dist/fonts/noto-sans-latin-cyrillic.ttf +0 -0
  167. package/dist/fonts.d.ts +2 -0
  168. package/dist/fonts.js +4 -0
  169. package/dist/headless.d.ts +4 -0
  170. package/dist/headless.js +4 -0
  171. package/dist/index.d.ts +5 -0
  172. package/dist/index.js +5 -0
  173. package/dist/limits.js +3 -0
  174. package/dist/ooxml-edit-worker.d.ts +1 -0
  175. package/dist/ooxml-edit-worker.js +8 -0
  176. package/dist/pdf-edit-worker.d.ts +1 -0
  177. package/dist/pdf-edit-worker.js +35 -0
  178. package/dist/spreadsheet-viewport.d.ts +5 -0
  179. package/dist/spreadsheet-viewport.js +11 -0
  180. package/dist/ui.js +7 -0
  181. package/dist/viewer.d.ts +6 -0
  182. package/dist/viewer.js +293 -19
  183. package/dist/viewport.d.ts +13 -0
  184. package/dist/viewport.js +126 -1
  185. package/dist/worker-protocol.d.ts +39 -2
  186. package/dist/workers/ooxml-edit-worker.js +8671 -0
  187. package/dist/workers/pdf-edit-worker.js +10844 -0
  188. package/package.json +3 -3
@@ -0,0 +1,442 @@
1
+ import { ViewerError } from "../../../errors.js";
2
+ export class PdfCompactionError extends ViewerError {
3
+ constructor(message) {
4
+ super("edit-failed", `The full save could not be compacted: ${message}`, {
5
+ details: { stage: "materialize", reason: "pdf-compaction" },
6
+ });
7
+ }
8
+ }
9
+ /** The reachable objects of `bytes` with a fresh cross-reference table. */
10
+ export function compactPdf(bytes) {
11
+ const { objects, trailer, trailerReferences, headerEnd } = parseFile(bytes);
12
+ const byNumber = new Map(objects.map((object) => [object.number, object]));
13
+ const reachable = new Set();
14
+ const queue = [...trailerReferences];
15
+ while (queue.length > 0) {
16
+ const number = queue.pop();
17
+ if (reachable.has(number))
18
+ continue;
19
+ const object = byNumber.get(number);
20
+ if (!object)
21
+ continue;
22
+ reachable.add(number);
23
+ queue.push(...object.references);
24
+ }
25
+ const kept = objects.filter((object) => reachable.has(object.number));
26
+ const parts = [bytes.subarray(0, headerEnd)];
27
+ const offsets = new Map();
28
+ let position = headerEnd;
29
+ for (const object of kept) {
30
+ offsets.set(object.number, position);
31
+ const slice = bytes.subarray(object.start, object.end);
32
+ parts.push(slice, NEWLINE);
33
+ position += slice.byteLength + 1;
34
+ }
35
+ // One subsection per run of consecutive numbers, as PDFium writes it, so
36
+ // sparse numbering does not cost twenty bytes per gap.
37
+ const numbers = kept.map((object) => object.number).sort((a, b) => a - b);
38
+ const size = (numbers.at(-1) ?? 0) + 1;
39
+ let xref = "xref\n0 1\n0000000000 65535 f \n";
40
+ for (let index = 0; index < numbers.length;) {
41
+ let count = 1;
42
+ while (numbers[index + count] === numbers[index] + count)
43
+ count += 1;
44
+ xref += `${numbers[index]} ${count}\n`;
45
+ for (const number of numbers.slice(index, index + count))
46
+ xref += `${String(offsets.get(number)).padStart(10, "0")} ${String(byNumber.get(number).generation).padStart(5, "0")} n \n`;
47
+ index += count;
48
+ }
49
+ const dictionary = trailer.replace(/\/Size\s+\d+/, `/Size ${size}`);
50
+ parts.push(ascii(`${xref}trailer\n${dictionary}\nstartxref\n${position}\n%%EOF\n`));
51
+ return concat(parts);
52
+ }
53
+ function parseFile(bytes) {
54
+ const lexer = new Lexer(bytes);
55
+ const objects = [];
56
+ let trailer;
57
+ let trailerReferences = [];
58
+ let headerEnd;
59
+ // Two integers followed by `obj` open an object; a dangling pair is kept
60
+ // until the next token tells what it is.
61
+ let pending = [];
62
+ for (;;) {
63
+ const token = lexer.next();
64
+ if (token.kind === "end")
65
+ break;
66
+ if (token.kind === "number") {
67
+ pending.push(token);
68
+ if (pending.length > 2)
69
+ pending.shift();
70
+ continue;
71
+ }
72
+ if (token.kind === "keyword" && token.value === "obj") {
73
+ const [first, second] = pending;
74
+ if (pending.length !== 2 ||
75
+ first.kind !== "number" ||
76
+ second.kind !== "number")
77
+ throw new PdfCompactionError("an object header is malformed");
78
+ // The object starts at its number: the cross-reference offset must
79
+ // point there, not at the whitespace before it.
80
+ const object = parseObject(lexer, bytes, first.value, second.value, first.start);
81
+ headerEnd ??= object.start;
82
+ objects.push(object);
83
+ pending = [];
84
+ continue;
85
+ }
86
+ pending = [];
87
+ if (token.kind === "keyword" && token.value === "trailer") {
88
+ const dictionaryStart = lexer.skipWhitespace();
89
+ const references = [];
90
+ const dictionaryEnd = skipValue(lexer, references);
91
+ trailer = latin1(bytes.subarray(dictionaryStart, dictionaryEnd));
92
+ trailerReferences = references;
93
+ continue;
94
+ }
95
+ if (token.kind === "keyword" && token.value === "xref") {
96
+ // A classic table: `xref`, subsections of `start count` and entries.
97
+ // PDFium always writes one; its entries are skipped as plain tokens.
98
+ continue;
99
+ }
100
+ if (token.kind === "keyword" && token.value === "startxref") {
101
+ lexer.next();
102
+ continue;
103
+ }
104
+ if (token.kind === "delimiter" && token.value === "<<")
105
+ // A dictionary outside any object: an xref stream's or a stray one.
106
+ throw new PdfCompactionError("the file uses a cross-reference stream instead of a trailer");
107
+ }
108
+ if (trailer === undefined)
109
+ throw new PdfCompactionError("the file has no trailer dictionary");
110
+ if (headerEnd === undefined)
111
+ throw new PdfCompactionError("the file has no objects");
112
+ // Indirect stream lengths were resolved while parsing where the length
113
+ // object came first; the rest were measured from `endstream`.
114
+ return { objects, trailer, trailerReferences, headerEnd };
115
+ }
116
+ function parseObject(lexer, bytes, number, generation, start) {
117
+ const references = [];
118
+ let integer;
119
+ let lengthValue;
120
+ let lengthReference;
121
+ // The object's value: a dictionary, an array, a scalar or nothing.
122
+ const first = lexer.peek();
123
+ if (first.kind === "number") {
124
+ lexer.next();
125
+ const second = lexer.peek();
126
+ if (second.kind === "number") {
127
+ lexer.next();
128
+ const third = lexer.peek();
129
+ if (third.kind === "keyword" && third.value === "R") {
130
+ lexer.next();
131
+ references.push(first.value);
132
+ }
133
+ }
134
+ else
135
+ integer = first.value;
136
+ }
137
+ else if (first.kind !== "keyword" || first.value !== "endobj") {
138
+ const dictionaryStart = lexer.position;
139
+ const dictionaryEnd = skipValue(lexer, references);
140
+ if (first.kind === "delimiter" && first.value === "<<") {
141
+ const length = findLength(bytes, dictionaryStart, dictionaryEnd);
142
+ lengthValue = length?.value;
143
+ lengthReference = length?.reference;
144
+ }
145
+ }
146
+ let next = lexer.next();
147
+ if (next.kind === "keyword" && next.value === "stream") {
148
+ let dataStart = next.end;
149
+ if (bytes[dataStart] === CR)
150
+ dataStart += 1;
151
+ if (bytes[dataStart] === LF)
152
+ dataStart += 1;
153
+ let dataEnd;
154
+ if (lengthValue !== undefined)
155
+ dataEnd = dataStart + lengthValue;
156
+ else if (lengthReference !== undefined) {
157
+ const known = lexer.integerObjects.get(lengthReference);
158
+ if (known !== undefined)
159
+ dataEnd = dataStart + known;
160
+ }
161
+ if (dataEnd === undefined ||
162
+ dataEnd > bytes.length ||
163
+ !followedByEndstream(bytes, dataEnd)) {
164
+ // Length unknown or wrong: measure to the `endstream` that closes the
165
+ // object. Stream data that spells "endstream" would mislead this, but
166
+ // a wrong direct length is the more common fault.
167
+ dataEnd = findEndstream(bytes, dataStart);
168
+ if (dataEnd === undefined)
169
+ throw new PdfCompactionError(`object ${number} has no endstream`);
170
+ }
171
+ lexer.seek(dataEnd);
172
+ const endstream = lexer.next();
173
+ if (endstream.kind !== "keyword" || endstream.value !== "endstream")
174
+ throw new PdfCompactionError(`object ${number} is not terminated`);
175
+ next = lexer.next();
176
+ }
177
+ if (next.kind !== "keyword" || next.value !== "endobj")
178
+ throw new PdfCompactionError(`object ${number} has no endobj`);
179
+ if (integer !== undefined)
180
+ lexer.integerObjects.set(number, integer);
181
+ return {
182
+ number,
183
+ generation,
184
+ start,
185
+ end: next.end,
186
+ references,
187
+ ...(integer === undefined ? {} : { integer }),
188
+ };
189
+ }
190
+ /** Skips one value (dictionary, array, scalar), collecting `n g R` references. */
191
+ function skipValue(lexer, references) {
192
+ let depth = 0;
193
+ let numbers = [];
194
+ let end = lexer.position;
195
+ do {
196
+ const token = lexer.next();
197
+ end = token.end;
198
+ switch (token.kind) {
199
+ case "end":
200
+ throw new PdfCompactionError("a value is not terminated");
201
+ case "delimiter":
202
+ if (token.value === "<<" || token.value === "[")
203
+ depth += 1;
204
+ else if (token.value === ">>" || token.value === "]")
205
+ depth -= 1;
206
+ numbers = [];
207
+ break;
208
+ case "number":
209
+ numbers.push(token.value);
210
+ if (numbers.length > 2)
211
+ numbers.shift();
212
+ break;
213
+ case "keyword":
214
+ if (token.value === "R" && numbers.length === 2)
215
+ references.push(numbers[0]);
216
+ else if (depth === 0 && !SCALAR_KEYWORDS.has(token.value))
217
+ throw new PdfCompactionError(`unexpected keyword ${token.value}`);
218
+ numbers = [];
219
+ break;
220
+ default:
221
+ numbers = [];
222
+ }
223
+ if (depth < 0)
224
+ throw new PdfCompactionError("unbalanced delimiters");
225
+ } while (depth > 0);
226
+ return end;
227
+ }
228
+ /** `/Length` of a dictionary's bytes: a direct integer or an indirect reference. */
229
+ function findLength(bytes, start, end) {
230
+ const lexer = new Lexer(bytes.subarray(start, end));
231
+ for (;;) {
232
+ const token = lexer.next();
233
+ if (token.kind === "end")
234
+ return undefined;
235
+ if (token.kind !== "name" || !lexer.lastNameIs("Length"))
236
+ continue;
237
+ const value = lexer.next();
238
+ if (value.kind !== "number")
239
+ return undefined;
240
+ const generation = lexer.peek();
241
+ if (generation.kind !== "number")
242
+ return { value: value.value };
243
+ lexer.next();
244
+ const reference = lexer.peek();
245
+ if (reference.kind === "keyword" && reference.value === "R")
246
+ return { reference: value.value };
247
+ return { value: value.value };
248
+ }
249
+ }
250
+ function followedByEndstream(bytes, at) {
251
+ let index = at;
252
+ while (index < bytes.length && isWhitespace(bytes[index]))
253
+ index += 1;
254
+ return matchesAscii(bytes, index, "endstream");
255
+ }
256
+ function findEndstream(bytes, from) {
257
+ for (let index = from; index <= bytes.length - 9; index += 1) {
258
+ if (bytes[index] !== 0x65 || !matchesAscii(bytes, index, "endstream"))
259
+ continue;
260
+ const after = index + 9;
261
+ if (after < bytes.length && !isWhitespace(bytes[after]))
262
+ continue;
263
+ // Trim the end-of-line that precedes `endstream`.
264
+ let end = index;
265
+ if (end > from && bytes[end - 1] === LF)
266
+ end -= 1;
267
+ if (end > from && bytes[end - 1] === CR)
268
+ end -= 1;
269
+ return end;
270
+ }
271
+ return undefined;
272
+ }
273
+ /** A PDF tokenizer that skips comments, strings and names as units. */
274
+ class Lexer {
275
+ #bytes;
276
+ #position = 0;
277
+ #lastName = "";
278
+ /** Objects whose value is a bare integer, as seen so far: indirect lengths. */
279
+ integerObjects = new Map();
280
+ constructor(bytes) {
281
+ this.#bytes = bytes;
282
+ }
283
+ get position() {
284
+ return this.#position;
285
+ }
286
+ seek(position) {
287
+ this.#position = position;
288
+ }
289
+ lastNameIs(name) {
290
+ return this.#lastName === name;
291
+ }
292
+ peek() {
293
+ const position = this.#position;
294
+ const token = this.next();
295
+ this.#position = position;
296
+ return token;
297
+ }
298
+ /** Skips whitespace and comments; returns where the next token starts. */
299
+ skipWhitespace() {
300
+ const bytes = this.#bytes;
301
+ for (;;) {
302
+ while (this.#position < bytes.length &&
303
+ isWhitespace(bytes[this.#position]))
304
+ this.#position += 1;
305
+ if (bytes[this.#position] !== 0x25)
306
+ return this.#position; // %
307
+ while (this.#position < bytes.length &&
308
+ bytes[this.#position] !== LF &&
309
+ bytes[this.#position] !== CR)
310
+ this.#position += 1;
311
+ }
312
+ }
313
+ next() {
314
+ const bytes = this.#bytes;
315
+ const start = this.skipWhitespace();
316
+ if (start >= bytes.length)
317
+ return { kind: "end", start, end: start };
318
+ const byte = bytes[start];
319
+ if (byte === 0x28) {
320
+ // ( literal string with nesting and escapes
321
+ let depth = 0;
322
+ let index = start;
323
+ for (; index < bytes.length; index += 1) {
324
+ const current = bytes[index];
325
+ if (current === 0x5c)
326
+ index += 1; // backslash escapes the next byte
327
+ else if (current === 0x28)
328
+ depth += 1;
329
+ else if (current === 0x29 && --depth === 0)
330
+ break;
331
+ }
332
+ this.#position = Math.min(index + 1, bytes.length);
333
+ return { kind: "string", start, end: this.#position };
334
+ }
335
+ if (byte === 0x3c) {
336
+ if (bytes[start + 1] === 0x3c) {
337
+ this.#position = start + 2;
338
+ return {
339
+ kind: "delimiter",
340
+ value: "<<",
341
+ start,
342
+ end: this.#position,
343
+ };
344
+ }
345
+ let index = start + 1;
346
+ while (index < bytes.length && bytes[index] !== 0x3e)
347
+ index += 1;
348
+ this.#position = Math.min(index + 1, bytes.length);
349
+ return { kind: "string", start, end: this.#position };
350
+ }
351
+ if (byte === 0x3e && bytes[start + 1] === 0x3e) {
352
+ this.#position = start + 2;
353
+ return {
354
+ kind: "delimiter",
355
+ value: ">>",
356
+ start,
357
+ end: this.#position,
358
+ };
359
+ }
360
+ if (byte === 0x5b || byte === 0x5d || byte === 0x7b || byte === 0x7d) {
361
+ this.#position = start + 1;
362
+ return {
363
+ kind: "delimiter",
364
+ value: String.fromCharCode(byte),
365
+ start,
366
+ end: this.#position,
367
+ };
368
+ }
369
+ if (byte === 0x2f) {
370
+ let index = start + 1;
371
+ while (index < bytes.length && isRegular(bytes[index]))
372
+ index += 1;
373
+ this.#position = index;
374
+ this.#lastName = latin1(bytes.subarray(start + 1, index));
375
+ return { kind: "name", start, end: index };
376
+ }
377
+ let index = start;
378
+ while (index < bytes.length && isRegular(bytes[index]))
379
+ index += 1;
380
+ if (index === start)
381
+ index += 1; // a stray delimiter byte
382
+ this.#position = index;
383
+ const text = latin1(bytes.subarray(start, index));
384
+ const value = Number(text);
385
+ if (/^[+-]?(\d+\.?\d*|\.\d+)$/.test(text) && Number.isFinite(value))
386
+ return { kind: "number", value, start, end: index };
387
+ return { kind: "keyword", value: text, start, end: index };
388
+ }
389
+ }
390
+ /** Keywords that are whole values on their own: `null`, `true`, `false`. */
391
+ const SCALAR_KEYWORDS = new Set(["null", "true", "false"]);
392
+ const CR = 0x0d;
393
+ const LF = 0x0a;
394
+ const NEWLINE = Uint8Array.of(LF);
395
+ function isWhitespace(byte) {
396
+ return (byte === 0x20 ||
397
+ byte === LF ||
398
+ byte === CR ||
399
+ byte === 0x09 ||
400
+ byte === 0x0c ||
401
+ byte === 0x00);
402
+ }
403
+ /** Regular characters: anything but whitespace and the delimiters. */
404
+ function isRegular(byte) {
405
+ return (!isWhitespace(byte) &&
406
+ byte !== 0x28 &&
407
+ byte !== 0x29 &&
408
+ byte !== 0x3c &&
409
+ byte !== 0x3e &&
410
+ byte !== 0x5b &&
411
+ byte !== 0x5d &&
412
+ byte !== 0x7b &&
413
+ byte !== 0x7d &&
414
+ byte !== 0x2f &&
415
+ byte !== 0x25);
416
+ }
417
+ function matchesAscii(bytes, at, text) {
418
+ if (at + text.length > bytes.length)
419
+ return false;
420
+ for (let index = 0; index < text.length; index += 1)
421
+ if (bytes[at + index] !== text.charCodeAt(index))
422
+ return false;
423
+ return true;
424
+ }
425
+ function latin1(bytes) {
426
+ let text = "";
427
+ for (let offset = 0; offset < bytes.length; offset += 0x8000)
428
+ text += String.fromCharCode(...bytes.subarray(offset, offset + 0x8000));
429
+ return text;
430
+ }
431
+ function ascii(text) {
432
+ return Uint8Array.from(text, (character) => character.charCodeAt(0) & 0xff);
433
+ }
434
+ function concat(parts) {
435
+ const out = new Uint8Array(parts.reduce((total, part) => total + part.byteLength, 0));
436
+ let offset = 0;
437
+ for (const part of parts) {
438
+ out.set(part, offset);
439
+ offset += part.byteLength;
440
+ }
441
+ return out;
442
+ }
@@ -0,0 +1,95 @@
1
+ import type { EngineBatch, EngineChange, MaterializedDocument, RestoreTarget } from "../../engine.js";
2
+ import type { EditFindOptions, EditOperation, ElementQuery, OperationIssue, PagePoint, PageRect, TextPosition, TextRange, TextTarget } from "../../types.js";
3
+ import type { PageLayout, PdfElement, PdfOperation, TextLayout } from "../types.js";
4
+ import type { EditWorkerBitmap } from "../../../worker-protocol.js";
5
+ import { FontLibrary } from "./fonts.js";
6
+ import { ImageCache } from "./images.js";
7
+ import { type AssetSource } from "../../assets.js";
8
+ import type { ResourceLimits } from "../../../contracts.js";
9
+ import type { Pdfium } from "./pdfium.js";
10
+ /**
11
+ * The working copy of one PDF: a PDFium document rebuilt from the original
12
+ * bytes plus the applied batches, with a stable id for every page and object.
13
+ * Everything a query answers comes from PDFium; the model only remembers
14
+ * identities and caches.
15
+ */
16
+ /** Batches arrive as plain JSON; unknown operations are reported, not typed away. */
17
+ type PdfOrUnknownOperation = PdfOperation | EditOperation;
18
+ /** Properties of the original that a change invalidates or leaves behind. */
19
+ export type DocumentFeature = "docmdp" | "tagged" | "pdfa";
20
+ export declare class PdfEditDocument {
21
+ #private;
22
+ readonly images: ImageCache;
23
+ constructor(pdfium: Pdfium, original: Uint8Array, fonts?: FontLibrary, limits?: ResourceLimits, assets?: AssetSource, compact?: (bytes: Uint8Array) => Uint8Array);
24
+ /** Features of the original a change affects, see `DocumentFeature`. */
25
+ get features(): readonly DocumentFeature[];
26
+ /** Signature fields in the document. */
27
+ get signatureCount(): number;
28
+ /**
29
+ * The families and texts a batch will draw, so the fonts can be fetched
30
+ * before the synchronous validation and application run.
31
+ */
32
+ fontRequests(operations: readonly PdfOrUnknownOperation[]): {
33
+ readonly family: string;
34
+ readonly text: string;
35
+ }[];
36
+ get pageCount(): number;
37
+ get pdfium(): Pdfium;
38
+ /**
39
+ * Checks a batch against the current document. Operations are checked one
40
+ * after another against the state before the batch, which is exact for
41
+ * everything but a page count another operation of the batch changes.
42
+ */
43
+ validate(operations: readonly PdfOrUnknownOperation[]): OperationIssue[];
44
+ /**
45
+ * Applies a batch. Plain operation arrays, as the unit tests pass them, get
46
+ * the next batch number as their state id.
47
+ */
48
+ apply(input: readonly PdfOrUnknownOperation[] | EngineBatch): EngineChange;
49
+ /**
50
+ * Bytes of the current state. The viewer reopens the incremental form,
51
+ * which is cheap to produce and read. A save defaults to a full rewrite —
52
+ * compacted, so deleted content is gone and the bytes do not depend on
53
+ * which pages were read — unless the file is signed, where the incremental
54
+ * form keeps the signed revision intact. Without changes either form is
55
+ * the bytes the document was opened from.
56
+ */
57
+ materialize(purpose?: "show" | "save", mode?: "incremental" | "full"): Uint8Array;
58
+ /**
59
+ * The bytes of the current state. A full save goes through the compaction
60
+ * pass; when the pass cannot read PDFium's output, the uncompacted full
61
+ * save stands in — silently for the display copy, with a
62
+ * `privacy-not-guaranteed` warning for a save, since deleted content may
63
+ * then remain recoverable in the file (decided 2026-10-02). Signed files
64
+ * take the incremental form for both purposes so the signed revision
65
+ * stays intact in every base the session may restore from.
66
+ */
67
+ materializeDocument(purpose?: "show" | "save", mode?: "incremental" | "full"): MaterializedDocument;
68
+ /**
69
+ * Rebuilds a state from its base (the original, or a checkpoint) and the
70
+ * batches after it. Plain arrays of batches, as the unit tests pass them,
71
+ * get state ids 1, 2, 3…
72
+ */
73
+ restore(input: readonly (readonly PdfOrUnknownOperation[])[] | RestoreTarget): void;
74
+ getElements(query: ElementQuery): PdfElement[];
75
+ getElement(id: string): PdfElement | undefined;
76
+ /** Elements under a point, top-most (drawn last) first. */
77
+ elementsAt(pageIndex: number, point: PagePoint): PdfElement[];
78
+ findText(query: string, options: EditFindOptions): TextTarget[];
79
+ /** Lines, glyph boxes and styles of a text, text box or table element. */
80
+ textLayout(elementId: string): TextLayout | undefined;
81
+ /** The layouts of every text element on a page, with the page's displayed size. */
82
+ pageLayout(pageIndex: number): PageLayout | undefined;
83
+ /** The caret position nearest to a page-space point; none on a page without text. */
84
+ positionAt(pageIndex: number, point: PagePoint): TextPosition | undefined;
85
+ /**
86
+ * Renders a page with the listed elements inactive, as RGBA over white at
87
+ * `scale` device pixels per point. The objects are reactivated before the
88
+ * call returns, so nothing about the document changes.
89
+ */
90
+ renderPageWithout(pageIndex: number, elementIds: readonly string[], scale: number): EditWorkerBitmap;
91
+ /** The rectangles a range covers, one per line fragment, in reading order. */
92
+ rangeRects(range: TextRange): PageRect[];
93
+ dispose(): void;
94
+ }
95
+ export {};