web-doc 0.6.2 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (201) hide show
  1. package/THIRD_PARTY_NOTICES.md +33 -1
  2. package/dist/adapters/docx-images.d.ts +13 -4
  3. package/dist/adapters/docx-images.js +3 -0
  4. package/dist/adapters/docx-paragraphs.d.ts +65 -0
  5. package/dist/adapters/docx-paragraphs.js +165 -0
  6. package/dist/adapters/docx-prepass.d.ts +17 -0
  7. package/dist/adapters/docx-prepass.js +230 -0
  8. package/dist/adapters/office.d.ts +31 -5
  9. package/dist/adapters/office.js +60 -29
  10. package/dist/adapters/pdf.d.ts +9 -0
  11. package/dist/adapters/pdf.js +10 -0
  12. package/dist/assets/pdfium/pdfium.wasm +0 -0
  13. package/dist/contracts.d.ts +53 -2
  14. package/dist/edit/ai/outline.d.ts +47 -0
  15. package/dist/edit/ai/outline.js +338 -0
  16. package/dist/edit/ai/targets.d.ts +16 -0
  17. package/dist/edit/ai/targets.js +309 -0
  18. package/dist/edit/ai/tools.d.ts +28 -0
  19. package/dist/edit/ai/tools.js +605 -0
  20. package/dist/edit/ai/types.d.ts +175 -0
  21. package/dist/edit/ai/types.js +1 -0
  22. package/dist/edit/assets.d.ts +23 -0
  23. package/dist/edit/assets.js +75 -0
  24. package/dist/edit/docx/elements.d.ts +6 -0
  25. package/dist/edit/docx/elements.js +90 -0
  26. package/dist/edit/docx/engine.d.ts +57 -0
  27. package/dist/edit/docx/engine.js +547 -0
  28. package/dist/edit/docx/handlers.d.ts +3 -0
  29. package/dist/edit/docx/handlers.js +15 -0
  30. package/dist/edit/docx/ids.d.ts +45 -0
  31. package/dist/edit/docx/ids.js +101 -0
  32. package/dist/edit/docx/model.d.ts +94 -0
  33. package/dist/edit/docx/model.js +350 -0
  34. package/dist/edit/docx/operations.d.ts +47 -0
  35. package/dist/edit/docx/operations.js +3 -0
  36. package/dist/edit/docx/provider.d.ts +16 -0
  37. package/dist/edit/docx/provider.js +37 -0
  38. package/dist/edit/docx/schemas.d.ts +6 -0
  39. package/dist/edit/docx/schemas.js +192 -0
  40. package/dist/edit/docx/session.d.ts +48 -0
  41. package/dist/edit/docx/session.js +467 -0
  42. package/dist/edit/docx/structure-ops.d.ts +6 -0
  43. package/dist/edit/docx/structure-ops.js +529 -0
  44. package/dist/edit/docx/style.d.ts +45 -0
  45. package/dist/edit/docx/style.js +375 -0
  46. package/dist/edit/docx/table-ops.d.ts +4 -0
  47. package/dist/edit/docx/table-ops.js +241 -0
  48. package/dist/edit/docx/text-ops.d.ts +40 -0
  49. package/dist/edit/docx/text-ops.js +491 -0
  50. package/dist/edit/docx/text.d.ts +34 -0
  51. package/dist/edit/docx/text.js +284 -0
  52. package/dist/edit/docx/tracked.d.ts +52 -0
  53. package/dist/edit/docx/tracked.js +347 -0
  54. package/dist/edit/docx/types.d.ts +206 -0
  55. package/dist/edit/docx/types.js +1 -0
  56. package/dist/edit/docx/write.d.ts +64 -0
  57. package/dist/edit/docx/write.js +375 -0
  58. package/dist/edit/engine.d.ts +128 -0
  59. package/dist/edit/engine.js +1 -0
  60. package/dist/edit/history.d.ts +77 -0
  61. package/dist/edit/history.js +123 -0
  62. package/dist/edit/ooxml/names.d.ts +20 -0
  63. package/dist/edit/ooxml/names.js +61 -0
  64. package/dist/edit/ooxml/opc.d.ts +50 -0
  65. package/dist/edit/ooxml/opc.js +150 -0
  66. package/dist/edit/ooxml/package.d.ts +82 -0
  67. package/dist/edit/ooxml/package.js +233 -0
  68. package/dist/edit/ooxml/patch.d.ts +51 -0
  69. package/dist/edit/ooxml/patch.js +250 -0
  70. package/dist/edit/ooxml/transaction.d.ts +39 -0
  71. package/dist/edit/ooxml/transaction.js +316 -0
  72. package/dist/edit/ooxml/worker.d.ts +8 -0
  73. package/dist/edit/ooxml/worker.js +12 -0
  74. package/dist/edit/ooxml/writer.d.ts +21 -0
  75. package/dist/edit/ooxml/writer.js +187 -0
  76. package/dist/edit/ooxml/xml.d.ts +74 -0
  77. package/dist/edit/ooxml/xml.js +451 -0
  78. package/dist/edit/ooxml/zip.d.ts +54 -0
  79. package/dist/edit/ooxml/zip.js +280 -0
  80. package/dist/edit/operations.d.ts +19 -0
  81. package/dist/edit/operations.js +137 -0
  82. package/dist/edit/pdf/engine/compact.d.ts +6 -0
  83. package/dist/edit/pdf/engine/compact.js +442 -0
  84. package/dist/edit/pdf/engine/document.d.ts +95 -0
  85. package/dist/edit/pdf/engine/document.js +868 -0
  86. package/dist/edit/pdf/engine/elements.d.ts +45 -0
  87. package/dist/edit/pdf/engine/elements.js +313 -0
  88. package/dist/edit/pdf/engine/existing-text.d.ts +4 -0
  89. package/dist/edit/pdf/engine/existing-text.js +424 -0
  90. package/dist/edit/pdf/engine/fonts.d.ts +78 -0
  91. package/dist/edit/pdf/engine/fonts.js +466 -0
  92. package/dist/edit/pdf/engine/geometry.d.ts +36 -0
  93. package/dist/edit/pdf/engine/geometry.js +97 -0
  94. package/dist/edit/pdf/engine/handler.d.ts +21 -0
  95. package/dist/edit/pdf/engine/handler.js +108 -0
  96. package/dist/edit/pdf/engine/images.d.ts +33 -0
  97. package/dist/edit/pdf/engine/images.js +188 -0
  98. package/dist/edit/pdf/engine/layout.d.ts +23 -0
  99. package/dist/edit/pdf/engine/layout.js +177 -0
  100. package/dist/edit/pdf/engine/operations.d.ts +67 -0
  101. package/dist/edit/pdf/engine/operations.js +3 -0
  102. package/dist/edit/pdf/engine/pages.d.ts +6 -0
  103. package/dist/edit/pdf/engine/pages.js +98 -0
  104. package/dist/edit/pdf/engine/pdfium.d.ts +195 -0
  105. package/dist/edit/pdf/engine/pdfium.js +249 -0
  106. package/dist/edit/pdf/engine/shapes.d.ts +4 -0
  107. package/dist/edit/pdf/engine/shapes.js +183 -0
  108. package/dist/edit/pdf/engine/tables.d.ts +44 -0
  109. package/dist/edit/pdf/engine/tables.js +315 -0
  110. package/dist/edit/pdf/engine/text-box.d.ts +71 -0
  111. package/dist/edit/pdf/engine/text-box.js +317 -0
  112. package/dist/edit/pdf/engine/text-layout.d.ts +28 -0
  113. package/dist/edit/pdf/engine/text-layout.js +67 -0
  114. package/dist/edit/pdf/engine/transform.d.ts +5 -0
  115. package/dist/edit/pdf/engine/transform.js +137 -0
  116. package/dist/edit/pdf/provider.d.ts +35 -0
  117. package/dist/edit/pdf/provider.js +92 -0
  118. package/dist/edit/pdf/range-map.d.ts +16 -0
  119. package/dist/edit/pdf/range-map.js +107 -0
  120. package/dist/edit/pdf/schemas.d.ts +11 -0
  121. package/dist/edit/pdf/schemas.js +291 -0
  122. package/dist/edit/pdf/selection.d.ts +5 -0
  123. package/dist/edit/pdf/selection.js +152 -0
  124. package/dist/edit/pdf/session.d.ts +63 -0
  125. package/dist/edit/pdf/session.js +312 -0
  126. package/dist/edit/pdf/types.d.ts +336 -0
  127. package/dist/edit/pdf/types.js +1 -0
  128. package/dist/edit/pptx/elements.d.ts +74 -0
  129. package/dist/edit/pptx/elements.js +301 -0
  130. package/dist/edit/pptx/engine.d.ts +37 -0
  131. package/dist/edit/pptx/engine.js +466 -0
  132. package/dist/edit/pptx/geometry.d.ts +72 -0
  133. package/dist/edit/pptx/geometry.js +193 -0
  134. package/dist/edit/pptx/handler.d.ts +7 -0
  135. package/dist/edit/pptx/handler.js +99 -0
  136. package/dist/edit/pptx/handlers.d.ts +3 -0
  137. package/dist/edit/pptx/handlers.js +21 -0
  138. package/dist/edit/pptx/image-table-ops.d.ts +5 -0
  139. package/dist/edit/pptx/image-table-ops.js +233 -0
  140. package/dist/edit/pptx/model.d.ts +69 -0
  141. package/dist/edit/pptx/model.js +170 -0
  142. package/dist/edit/pptx/operations.d.ts +50 -0
  143. package/dist/edit/pptx/operations.js +11 -0
  144. package/dist/edit/pptx/provider.d.ts +17 -0
  145. package/dist/edit/pptx/provider.js +40 -0
  146. package/dist/edit/pptx/schemas.d.ts +6 -0
  147. package/dist/edit/pptx/schemas.js +220 -0
  148. package/dist/edit/pptx/session.d.ts +54 -0
  149. package/dist/edit/pptx/session.js +145 -0
  150. package/dist/edit/pptx/shape-ops.d.ts +10 -0
  151. package/dist/edit/pptx/shape-ops.js +486 -0
  152. package/dist/edit/pptx/slide-ops.d.ts +6 -0
  153. package/dist/edit/pptx/slide-ops.js +353 -0
  154. package/dist/edit/pptx/style.d.ts +31 -0
  155. package/dist/edit/pptx/style.js +181 -0
  156. package/dist/edit/pptx/text-ops.d.ts +13 -0
  157. package/dist/edit/pptx/text-ops.js +267 -0
  158. package/dist/edit/pptx/text-write.d.ts +67 -0
  159. package/dist/edit/pptx/text-write.js +293 -0
  160. package/dist/edit/pptx/text.d.ts +41 -0
  161. package/dist/edit/pptx/text.js +99 -0
  162. package/dist/edit/pptx/types.d.ts +232 -0
  163. package/dist/edit/pptx/types.js +1 -0
  164. package/dist/edit/schema.d.ts +11 -0
  165. package/dist/edit/schema.js +275 -0
  166. package/dist/edit/session.d.ts +77 -0
  167. package/dist/edit/session.js +943 -0
  168. package/dist/edit/sessions.d.ts +8 -0
  169. package/dist/edit/sessions.js +1 -0
  170. package/dist/edit/types.d.ts +290 -0
  171. package/dist/edit/types.js +1 -0
  172. package/dist/edit/worker-engine.d.ts +31 -0
  173. package/dist/edit/worker-engine.js +91 -0
  174. package/dist/fonts/THIRD_PARTY_NOTICES.md +3 -0
  175. package/dist/fonts/manifest.json +6 -1
  176. package/dist/fonts/noto-sans-latin-cyrillic.ttf +0 -0
  177. package/dist/fonts.d.ts +2 -0
  178. package/dist/fonts.js +4 -0
  179. package/dist/fuzzy-alignment.d.ts +11 -4
  180. package/dist/fuzzy-alignment.js +3 -9
  181. package/dist/headless.d.ts +5 -1
  182. package/dist/headless.js +8 -1
  183. package/dist/index.d.ts +10 -2
  184. package/dist/index.js +16 -2
  185. package/dist/limits.js +6 -0
  186. package/dist/ooxml-edit-worker.d.ts +1 -0
  187. package/dist/ooxml-edit-worker.js +8 -0
  188. package/dist/pdf-edit-worker.d.ts +1 -0
  189. package/dist/pdf-edit-worker.js +35 -0
  190. package/dist/spreadsheet-viewport.d.ts +5 -0
  191. package/dist/spreadsheet-viewport.js +11 -0
  192. package/dist/ui.js +7 -0
  193. package/dist/viewer.d.ts +6 -0
  194. package/dist/viewer.js +293 -19
  195. package/dist/viewport.d.ts +13 -0
  196. package/dist/viewport.js +126 -1
  197. package/dist/worker-protocol.d.ts +39 -2
  198. package/dist/workers/fuzzy-search-worker.js +1 -1
  199. package/dist/workers/ooxml-edit-worker.js +9212 -0
  200. package/dist/workers/pdf-edit-worker.js +10847 -0
  201. package/package.json +3 -3
@@ -2,13 +2,22 @@
2
2
 
3
3
  The release artifact contains or depends on the following principal components. The generated SPDX SBOM in `artifacts/sbom.spdx.json` is the complete machine-readable inventory for the pinned lockfiles.
4
4
 
5
- - [`@silurus/ooxml`](https://github.com/yukiyokotani/office-open-xml-viewer) — MIT; modern Office parsing/rendering.
5
+ - [`@silurus/ooxml`](https://github.com/yukiyokotani/office-open-xml-viewer) — MIT; modern Office parsing/rendering (0.88.0 for DOCX, XLSX and PPTX). The package ships its own `THIRD_PARTY_NOTICES.md` for what the engine bundles; its optional region-map renderer carries a public-domain Natural Earth dataset and its optional math bundle MathJax, neither of which Zrimo imports.
6
6
  - [`office_oxide`](https://github.com/yfedoseev/office_oxide) — MIT OR Apache-2.0; compound-file handling, Office IR/writer utilities and legacy XLS/PPT conversion.
7
7
  - [`Fuse.js`](https://github.com/krisk/Fuse) — Apache-2.0; fuzzy matching behind the opt-in `search()` fallback.
8
8
  - [`pdfjs-dist` / Mozilla PDF.js](https://github.com/mozilla/pdf.js) — Apache-2.0; browser PDF parsing,
9
9
  font/CMap handling, canvas rendering, and text extraction. The packaged
10
10
  standard-font, ICC, CMap, OpenJPEG, JBIG2, and QCMS assets retain the license
11
11
  files distributed with PDF.js under `dist/assets/pdfjs/`.
12
+ - [`@embedpdf/pdfium`](https://github.com/embedpdf/embed-pdf-viewer/tree/main/packages/pdfium) — MIT
13
+ per the license file and `package.json` of the pinned 2.15.1 tarball (the
14
+ upstream repository moved to Apache-2.0 on 2026-07-20); loaded only by the
15
+ PDF edit worker. It bundles [PDFium](https://pdfium.googlesource.com/pdfium/)
16
+ compiled to WebAssembly — BSD-3-Clause, see `LICENSE.pdfium` in that package —
17
+ together with the third-party libraries PDFium builds in, such as FreeType
18
+ (FreeType License), OpenJPEG (BSD-2-Clause), libpng (libpng License), zlib
19
+ (Zlib License) and Anti-Grain Geometry 2.3, whose notices ship with the
20
+ PDFium source tree.
12
21
  - [`core-js`](https://github.com/zloirock/core-js) compatibility modules embedded in the PDF.js legacy browser build —
13
22
  MIT; polyfills required by the supported browser matrix, including the PDF
14
23
  worker realm.
@@ -17,5 +26,28 @@ The release artifact contains or depends on the following principal components.
17
26
  - [`wasm-bindgen`](https://github.com/wasm-bindgen/wasm-bindgen) — MIT OR Apache-2.0; browser bindings for project-owned Rust/WASM modules.
18
27
  - [`zip`](https://github.com/zip-rs/zip2) — MIT; bounded in-memory ZIP/OOXML package manipulation.
19
28
  - [Serde](https://github.com/serde-rs/serde), [`serde_json`](https://github.com/serde-rs/json) and [`thiserror`](https://github.com/dtolnay/thiserror) — MIT OR Apache-2.0; serialization and structured Rust errors.
29
+ - [GenOffice](https://github.com/genspark-ai/genoffice) — Apache-2.0; the
30
+ selection-to-object matching ladder of the PDF overlay primitives (rectangle
31
+ overlap, then a single containing object, then a text match after NFKC
32
+ folding) follows the technique of its `apps/pdf/src/main/text-edit.ts`, and
33
+ the PPTX text editing of `packages/viewer/src/edit/pptx/` follows its
34
+ published ideas of tracing edited runs to their source so formatting
35
+ survives, rebuilding only the paragraphs an edit touches, keeping
36
+ `a:bodyPr` and `a:lstStyle`, and dropping a stale autofit scale. The DOCX
37
+ editing of `packages/viewer/src/edit/docx/` follows its published ideas of
38
+ a flat block index over the original bytes with the raw paragraph and run
39
+ properties kept, property merges that keep the original bytes where a
40
+ value does not change and place new children in schema order, and
41
+ section properties that an edited paragraph never duplicates. The AI
42
+ tooling of `packages/viewer/src/edit/ai/` follows its published ideas of
43
+ a numbered document skeleton the model reads first with full content
44
+ pulled on demand within a character budget, writes made only through a
45
+ small validated tool set with a dry run, index addressing guarded by an
46
+ optimistic "document seen" check, a per-turn snapshot to roll back, and
47
+ Word tracked changes authored by the AI as the review channel. The
48
+ implementations in `packages/viewer/src/edit/pdf/selection.ts`,
49
+ `packages/viewer/src/edit/pptx/`, `packages/viewer/src/edit/docx/` and
50
+ `packages/viewer/src/edit/ai/` are web-doc's own; no GenOffice code is
51
+ included.
20
52
  - [Noto Sans](https://github.com/notofonts/noto-fonts) and [Noto Sans CJK](https://github.com/notofonts/noto-cjk) subset fonts — SIL Open Font License 1.1. The font manifest, complete OFL text, pinned source commits and SHA-256 hashes are included in `dist/fonts/`.
21
53
  No Microsoft proprietary font or copyleft runtime component is bundled. Transitive notices and license expressions are verified by `npm run licenses`.
@@ -2,10 +2,16 @@
2
2
  * Word draws an inline picture at its declared extent even when that is wider
3
3
  * than the text area, so a generated document that embeds a 21-inch chart on
4
4
  * a 6.5-inch column shows a clipped picture. A viewer has no margin to spill
5
- * into, so before the page layout runs the oversized inline pictures are
6
- * scaled down to fit the section's content box, aspect ratio preserved.
7
- * Anchored (floating) pictures keep their geometry: their position is part of
8
- * the author's layout.
5
+ * into, so the oversized inline pictures are scaled down to fit the section's
6
+ * content box, aspect ratio preserved. Anchored (floating) pictures keep
7
+ * their geometry: their position is part of the author's layout.
8
+ *
9
+ * @deprecated The viewer no longer calls this helper: since `@silurus/ooxml`
10
+ * 0.88 lays a document out inside `load()`, adjusting the parsed model
11
+ * changes nothing on the page. The fitting now happens in the XML before the
12
+ * engine reads it (`prepareDocxForDisplay`). The helper stays exported for
13
+ * hosts that applied it to their own `main`-mode models and is removed in
14
+ * the next major version.
9
15
  */
10
16
  export interface DocxSectionGeometry {
11
17
  readonly pageWidth: number;
@@ -22,5 +28,8 @@ export interface DocxModelLike {
22
28
  /**
23
29
  * Shrink every inline picture that would not fit its section's content box.
24
30
  * Mutates the model in place and returns how many pictures were scaled.
31
+ *
32
+ * @deprecated See the module note: the viewer fits pictures in the XML
33
+ * pre-pass instead; this helper is removed in the next major version.
25
34
  */
26
35
  export declare function fitInlineImagesToPage(model: DocxModelLike): number;
@@ -1,6 +1,9 @@
1
1
  /**
2
2
  * Shrink every inline picture that would not fit its section's content box.
3
3
  * Mutates the model in place and returns how many pictures were scaled.
4
+ *
5
+ * @deprecated See the module note: the viewer fits pictures in the XML
6
+ * pre-pass instead; this helper is removed in the next major version.
4
7
  */
5
8
  export function fitInlineImagesToPage(model) {
6
9
  let scaled = 0;
@@ -0,0 +1,65 @@
1
+ export type DocxStory = "body" | "header" | "footer" | "footnote" | "endnote" | "textbox";
2
+ export interface DocxRunSource {
3
+ readonly story: DocxStory | string;
4
+ readonly storyInstance: string;
5
+ readonly path: readonly number[];
6
+ }
7
+ export interface DocxModelParagraph {
8
+ readonly type: "paragraph";
9
+ readonly paragraphId?: string;
10
+ readonly bookmarks?: readonly string[];
11
+ readonly runs?: readonly DocxModelRun[];
12
+ }
13
+ export interface DocxModelRun {
14
+ readonly type?: string;
15
+ /** The blocks a text box shape holds. */
16
+ readonly textBoxContent?: readonly DocxModelBlock[];
17
+ }
18
+ export interface DocxModelTable {
19
+ readonly type: "table";
20
+ readonly rows: readonly {
21
+ readonly cells: readonly {
22
+ readonly content: readonly DocxModelBlock[];
23
+ }[];
24
+ }[];
25
+ }
26
+ export interface DocxModelHeaderFooter {
27
+ readonly body: readonly DocxModelBlock[];
28
+ }
29
+ export interface DocxModelHeadersFooters {
30
+ readonly default?: DocxModelHeaderFooter | null;
31
+ readonly first?: DocxModelHeaderFooter | null;
32
+ readonly even?: DocxModelHeaderFooter | null;
33
+ }
34
+ export interface DocxModelSectionBreak {
35
+ readonly type: "sectionBreak";
36
+ readonly headers?: DocxModelHeadersFooters;
37
+ readonly footers?: DocxModelHeadersFooters;
38
+ }
39
+ export interface DocxModelPageBreak {
40
+ readonly type: "pageBreak";
41
+ }
42
+ export type DocxModelBlock = DocxModelParagraph | DocxModelTable | DocxModelSectionBreak | DocxModelPageBreak | {
43
+ readonly type: string;
44
+ };
45
+ export interface DocxModelNote {
46
+ readonly id: string;
47
+ readonly content: readonly DocxModelBlock[];
48
+ }
49
+ /** The part of the engine's document model the bridge walks. */
50
+ export interface DocxModelDocument {
51
+ readonly body: readonly DocxModelBlock[];
52
+ readonly headers?: DocxModelHeadersFooters;
53
+ readonly footers?: DocxModelHeadersFooters;
54
+ readonly footnotes?: readonly DocxModelNote[];
55
+ readonly endnotes?: readonly DocxModelNote[];
56
+ }
57
+ /** Resolves a run's source to the id of its `w:p`, or `undefined`. */
58
+ export type DocxParagraphIdResolver = (source: DocxRunSource | undefined) => string | undefined;
59
+ /**
60
+ * A resolver over one document model. Results are cached per source, so a
61
+ * page's runs (many per paragraph) cost one walk per paragraph.
62
+ */
63
+ export declare function createDocxParagraphIdResolver(model: DocxModelDocument | undefined): DocxParagraphIdResolver;
64
+ /** The id a model paragraph carries: its own `paragraphId` or its bookmark. */
65
+ export declare function paragraphIdOf(paragraph: DocxModelParagraph): string | undefined;
@@ -0,0 +1,165 @@
1
+ /*
2
+ * The paragraph bridge for DOCX text runs. The renderer lays out from its
3
+ * own model and tags every run with a `source`: the story it came from
4
+ * (`body`, a header or footer, a note, a text box), the story instance and
5
+ * the path of block indices that leads to the paragraph (a table adds its
6
+ * row and cell indices, the last entry is the run index). The display
7
+ * pre-pass marks every `w:p` of the file with a hidden `_wd<id>` bookmark,
8
+ * and the engine keeps bookmark names on its model paragraphs, so a run's
9
+ * source leads to a model paragraph whose bookmark names its `w:p`.
10
+ *
11
+ * Instance names follow the engine: `body` for the body, `default`, `first`
12
+ * or `even` for the document's headers and footers and `section:<i>:<kind>`
13
+ * for those a section break carries (`i` being the break's index in the
14
+ * body), the note id for footnotes and endnotes, and for a text box the
15
+ * `story:instance:path` key of the shape run that holds it.
16
+ */
17
+ /** The name prefix of the pre-pass bookmarks that carry paragraph ids. */
18
+ const BOOKMARK_PREFIX = "_wd";
19
+ const HEADER_KINDS = ["default", "first", "even"];
20
+ /**
21
+ * A resolver over one document model. Results are cached per source, so a
22
+ * page's runs (many per paragraph) cost one walk per paragraph.
23
+ */
24
+ export function createDocxParagraphIdResolver(model) {
25
+ if (!model || !Array.isArray(model.body))
26
+ return () => undefined;
27
+ const cache = new Map();
28
+ return (source) => {
29
+ if (!source || !Array.isArray(source.path))
30
+ return undefined;
31
+ const key = `${source.story}\u0000${source.storyInstance}\u0000${source.path.join(",")}`;
32
+ if (cache.has(key))
33
+ return cache.get(key);
34
+ let id;
35
+ try {
36
+ const container = containerOf(model, source.story, source.storyInstance);
37
+ const located = container && paragraphAt(container, source.path);
38
+ id = located && paragraphIdAt(located.blocks, located.index);
39
+ }
40
+ catch {
41
+ id = undefined;
42
+ }
43
+ cache.set(key, id);
44
+ return id;
45
+ };
46
+ }
47
+ /** The id a model paragraph carries: its own `paragraphId` or its bookmark. */
48
+ export function paragraphIdOf(paragraph) {
49
+ if (typeof paragraph.paragraphId === "string" && paragraph.paragraphId)
50
+ return paragraph.paragraphId;
51
+ for (const name of paragraph.bookmarks ?? [])
52
+ if (typeof name === "string" && name.startsWith(BOOKMARK_PREFIX))
53
+ return name.slice(BOOKMARK_PREFIX.length);
54
+ return undefined;
55
+ }
56
+ function containerOf(model, story, instance) {
57
+ switch (story) {
58
+ case "body":
59
+ return model.body;
60
+ case "header":
61
+ case "footer":
62
+ return headerFooterBody(model, story, instance);
63
+ case "footnote":
64
+ return model.footnotes?.find((note) => note.id === instance)?.content;
65
+ case "endnote":
66
+ return model.endnotes?.find((note) => note.id === instance)?.content;
67
+ case "textbox":
68
+ return textBoxContent(model, instance);
69
+ default:
70
+ return undefined;
71
+ }
72
+ }
73
+ function headerFooterBody(model, story, instance) {
74
+ const kindOf = (value) => HEADER_KINDS.find((kind) => kind === value);
75
+ const sectionMatch = /^section:(\d+):([a-z]+)$/.exec(instance);
76
+ if (sectionMatch) {
77
+ const block = model.body[Number(sectionMatch[1])];
78
+ const kind = kindOf(sectionMatch[2]);
79
+ if (!block || block.type !== "sectionBreak" || !kind)
80
+ return undefined;
81
+ const set = block[story === "header" ? "headers" : "footers"];
82
+ return set?.[kind]?.body;
83
+ }
84
+ const kind = kindOf(instance);
85
+ if (!kind)
86
+ return undefined;
87
+ const set = story === "header" ? model.headers : model.footers;
88
+ return set?.[kind]?.body;
89
+ }
90
+ /**
91
+ * A text box's instance is the `story:instance:path` key of the shape run
92
+ * that holds it; the path's last entry is that run's index in its paragraph.
93
+ */
94
+ function textBoxContent(model, instance) {
95
+ const first = instance.indexOf(":");
96
+ const last = instance.lastIndexOf(":");
97
+ if (first < 0 || last <= first)
98
+ return undefined;
99
+ const story = instance.slice(0, first);
100
+ const hostInstance = instance.slice(first + 1, last);
101
+ const path = instance
102
+ .slice(last + 1)
103
+ .split(".")
104
+ .map((entry) => Number(entry));
105
+ if (path.length === 0 || path.some((entry) => !Number.isInteger(entry)))
106
+ return undefined;
107
+ const container = containerOf(model, story, hostInstance);
108
+ const located = container && paragraphAt(container, path);
109
+ const runIndex = located?.rest[0];
110
+ if (!located || runIndex === undefined)
111
+ return undefined;
112
+ const run = located.paragraph.runs?.[runIndex];
113
+ return Array.isArray(run?.textBoxContent) ? run.textBoxContent : undefined;
114
+ }
115
+ /** Follows a source path through blocks and table cells to its paragraph. */
116
+ function paragraphAt(container, path) {
117
+ let blocks = container;
118
+ let at = 0;
119
+ while (at < path.length) {
120
+ const index = path[at];
121
+ const node = blocks[index];
122
+ if (!node)
123
+ return undefined;
124
+ if (node.type === "paragraph")
125
+ return {
126
+ blocks,
127
+ index,
128
+ paragraph: node,
129
+ rest: path.slice(at + 1),
130
+ };
131
+ if (node.type !== "table")
132
+ return undefined;
133
+ const cell = node.rows[path[at + 1] ?? -1]?.cells[path[at + 2] ?? -1];
134
+ if (!cell || !Array.isArray(cell.content))
135
+ return undefined;
136
+ blocks = cell.content;
137
+ at += 3;
138
+ }
139
+ return undefined;
140
+ }
141
+ /**
142
+ * The id of the paragraph at `index`. The engine splits a paragraph that
143
+ * holds a page break into two model paragraphs around a hoisted page break,
144
+ * and only the first keeps the bookmark, so an unmarked paragraph right
145
+ * after a page break takes the id of the paragraph before that break.
146
+ * Any other unmarked paragraph has no id: guessing would name the wrong
147
+ * `w:p`.
148
+ */
149
+ function paragraphIdAt(blocks, index) {
150
+ let at = index;
151
+ for (;;) {
152
+ const node = blocks[at];
153
+ if (!node || node.type !== "paragraph")
154
+ return undefined;
155
+ const id = paragraphIdOf(node);
156
+ if (id !== undefined)
157
+ return id;
158
+ if (at >= 2 &&
159
+ blocks[at - 1]?.type === "pageBreak" &&
160
+ blocks[at - 2]?.type === "paragraph")
161
+ at -= 2;
162
+ else
163
+ return undefined;
164
+ }
165
+ }
@@ -0,0 +1,17 @@
1
+ import type { ResourceLimits } from "../contracts.js";
2
+ export { generatedParagraphId, PARAGRAPH_BOOKMARK_PREFIX, } from "../edit/docx/ids.js";
3
+ export interface DocxPrepassResult {
4
+ /** The bytes to render; the input when nothing had to change. */
5
+ readonly bytes: Uint8Array;
6
+ /** Inline pictures scaled down. */
7
+ readonly scaledImages: number;
8
+ /** Paragraphs marked with an id bookmark. */
9
+ readonly markedParagraphs: number;
10
+ /** Of those, paragraphs whose id was generated (no `w14:paraId` in the file). */
11
+ readonly generatedIds: number;
12
+ }
13
+ /**
14
+ * Prepares DOCX bytes for display. A document the package layer cannot
15
+ * open or scan is returned as it is: the renderer reports its own errors.
16
+ */
17
+ export declare function prepareDocxForDisplay(bytes: Uint8Array, limits: ResourceLimits, signal?: AbortSignal): Promise<DocxPrepassResult>;
@@ -0,0 +1,230 @@
1
+ import { assignParagraphIds, collectIds, newIdState, OFFICE_RELATIONSHIPS, PARAGRAPH_BOOKMARK_PREFIX, STORY_RELATIONSHIP_TYPES, W_NS, } from "../edit/docx/ids.js";
2
+ import { OoxmlPackage } from "../edit/ooxml/package.js";
3
+ import { patches } from "../edit/ooxml/patch.js";
4
+ export { generatedParagraphId, PARAGRAPH_BOOKMARK_PREFIX, } from "../edit/docx/ids.js";
5
+ /*
6
+ * The DOCX display pre-pass: what the renderer should see instead of the
7
+ * file as stored, written as patches of the XML parts through the package
8
+ * layer and never into what an editor saves.
9
+ *
10
+ * 1. Oversized inline pictures are scaled down to their section's content
11
+ * box, aspect ratio kept, as Word's own layout would not: Word draws a
12
+ * 21-inch picture on a 6.5-inch column clipped, a viewer has no margin
13
+ * to spill into. Anchored pictures keep their geometry.
14
+ * 2. Every paragraph is marked with a hidden bookmark that carries its id —
15
+ * the file's own `w14:paraId` when it has one, else a deterministic id
16
+ * from its position — because the renderer keeps bookmark names on its
17
+ * model paragraphs but does not read `w14:paraId`. An editor that walks
18
+ * the same paragraphs in the same order computes the same ids from the
19
+ * original bytes.
20
+ */
21
+ const WP_NS = "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing";
22
+ const A_NS = "http://schemas.openxmlformats.org/drawingml/2006/main";
23
+ /** EMU per twentieth of a point, the unit of page sizes and margins. */
24
+ const EMU_PER_TWIP = 635;
25
+ /** US Letter with one-inch margins, Word's default when a section says nothing. */
26
+ const DEFAULT_SECTION = {
27
+ pageWidth: 12240,
28
+ pageHeight: 15840,
29
+ marginLeft: 1440,
30
+ marginRight: 1440,
31
+ marginTop: 1440,
32
+ marginBottom: 1440,
33
+ };
34
+ /**
35
+ * Prepares DOCX bytes for display. A document the package layer cannot
36
+ * open or scan is returned as it is: the renderer reports its own errors.
37
+ */
38
+ export async function prepareDocxForDisplay(bytes, limits, signal) {
39
+ const unchanged = {
40
+ bytes,
41
+ scaledImages: 0,
42
+ markedParagraphs: 0,
43
+ generatedIds: 0,
44
+ };
45
+ let pkg;
46
+ try {
47
+ pkg = await OoxmlPackage.open(bytes, {
48
+ limits,
49
+ ...(signal ? { signal } : {}),
50
+ });
51
+ }
52
+ catch {
53
+ return unchanged;
54
+ }
55
+ try {
56
+ const root = await pkg.relationships("/", signal);
57
+ const main = root.byType(`${OFFICE_RELATIONSHIPS}officeDocument`)[0]
58
+ ?.targetPart;
59
+ if (!main || !pkg.has(main))
60
+ return unchanged;
61
+ const transaction = pkg.transaction();
62
+ let scaledImages = 0;
63
+ let markedParagraphs = 0;
64
+ let generatedIds = 0;
65
+ const counted = (marked, generated) => {
66
+ markedParagraphs += marked;
67
+ generatedIds += generated;
68
+ };
69
+ const document = await pkg.xml(main, signal);
70
+ const relationships = await pkg.relationships(main, signal);
71
+ const stories = [];
72
+ for (const type of STORY_RELATIONSHIP_TYPES)
73
+ for (const item of relationships.byType(type))
74
+ if (item.targetPart && pkg.has(item.targetPart))
75
+ stories.push(await pkg.xml(item.targetPart, signal));
76
+ // Ids are unique across the document: every part is read before any is marked.
77
+ const state = newIdState();
78
+ for (const part of [document, ...stories])
79
+ collectIds(part, state);
80
+ const documentPatches = [
81
+ ...fitInlinePictures(document, (count) => (scaledImages += count)),
82
+ ];
83
+ documentPatches.push(...paragraphIdPatches(document, state, counted));
84
+ if (documentPatches.length > 0)
85
+ transaction.patch(document, documentPatches);
86
+ for (const part of stories) {
87
+ const items = paragraphIdPatches(part, state, counted);
88
+ if (items.length > 0)
89
+ transaction.patch(part, items);
90
+ }
91
+ if (scaledImages === 0 && markedParagraphs === 0)
92
+ return unchanged;
93
+ await transaction.commit(signal);
94
+ return {
95
+ bytes: await pkg.save({}, signal),
96
+ scaledImages,
97
+ markedParagraphs,
98
+ generatedIds,
99
+ };
100
+ }
101
+ catch {
102
+ return unchanged;
103
+ }
104
+ }
105
+ function sectionGeometry(part, sectPr) {
106
+ const read = (node, name, fallback) => {
107
+ const value = node ? Number(part.attribute(node, name) ?? NaN) : NaN;
108
+ return Number.isFinite(value) && value >= 0 ? value : fallback;
109
+ };
110
+ const pgSz = sectPr?.children.find((child) => child.local === "pgSz" && child.namespace === W_NS);
111
+ const pgMar = sectPr?.children.find((child) => child.local === "pgMar" && child.namespace === W_NS);
112
+ return {
113
+ pageWidth: read(pgSz, "w:w", DEFAULT_SECTION.pageWidth),
114
+ pageHeight: read(pgSz, "w:h", DEFAULT_SECTION.pageHeight),
115
+ marginLeft: read(pgMar, "w:left", DEFAULT_SECTION.marginLeft),
116
+ marginRight: read(pgMar, "w:right", DEFAULT_SECTION.marginRight),
117
+ marginTop: read(pgMar, "w:top", DEFAULT_SECTION.marginTop),
118
+ marginBottom: read(pgMar, "w:bottom", DEFAULT_SECTION.marginBottom),
119
+ };
120
+ }
121
+ /** Patches that scale the oversized inline pictures of the body, section by section. */
122
+ function fitInlinePictures(part, count) {
123
+ const body = part.root.children.find((child) => child.local === "body" && child.namespace === W_NS);
124
+ if (!body)
125
+ return [];
126
+ const items = [];
127
+ // A w:sectPr closes the section that ends at it: the geometry of a run of
128
+ // body elements is known once the paragraph carrying the break, or the
129
+ // body's own w:sectPr, is reached.
130
+ let pending = [];
131
+ const flush = (sectPr) => {
132
+ const geometry = sectionGeometry(part, sectPr);
133
+ for (const element of pending)
134
+ items.push(...fitPicturesUnder(part, element, geometry, count));
135
+ pending = [];
136
+ };
137
+ let bodySectPr;
138
+ for (const element of body.children) {
139
+ if (element.namespace === W_NS && element.local === "sectPr") {
140
+ bodySectPr = element;
141
+ continue;
142
+ }
143
+ pending.push(element);
144
+ const sectPr = element.namespace === W_NS && element.local === "p"
145
+ ? element.children
146
+ .find((child) => child.local === "pPr" && child.namespace === W_NS)
147
+ ?.children.find((child) => child.local === "sectPr" && child.namespace === W_NS)
148
+ : undefined;
149
+ if (sectPr)
150
+ flush(sectPr);
151
+ }
152
+ flush(bodySectPr);
153
+ return items;
154
+ }
155
+ function fitPicturesUnder(part, element, geometry, count) {
156
+ const contentWidth = (geometry.pageWidth - geometry.marginLeft - geometry.marginRight) *
157
+ EMU_PER_TWIP;
158
+ const contentHeight = (geometry.pageHeight - geometry.marginTop - geometry.marginBottom) *
159
+ EMU_PER_TWIP;
160
+ if (!(contentWidth > 0 && contentHeight > 0))
161
+ return [];
162
+ const items = [];
163
+ for (const inline of part.findAll("inline", element)) {
164
+ if (inline.namespace !== WP_NS)
165
+ continue;
166
+ const extent = inline.children.find((child) => child.local === "extent" && child.namespace === WP_NS);
167
+ if (!extent)
168
+ continue;
169
+ const cx = Number(part.attribute(extent, "cx") ?? NaN);
170
+ const cy = Number(part.attribute(extent, "cy") ?? NaN);
171
+ if (!(cx > 0 && cy > 0))
172
+ continue;
173
+ const scale = Math.min(1, contentWidth / cx, contentHeight / cy);
174
+ if (scale >= 1)
175
+ continue;
176
+ const width = String(Math.round(cx * scale));
177
+ const height = String(Math.round(cy * scale));
178
+ items.push(patches.setAttribute(part, extent, "cx", width), patches.setAttribute(part, extent, "cy", height));
179
+ // The picture's own transform repeats the extent; keep both in step.
180
+ const ext = part
181
+ .findAll("ext", inline)
182
+ .find((candidate) => candidate.namespace === A_NS &&
183
+ candidate.parent?.local === "xfrm" &&
184
+ part.attribute(candidate, "cx") === String(cx) &&
185
+ part.attribute(candidate, "cy") === String(cy));
186
+ if (ext)
187
+ items.push(patches.setAttribute(part, ext, "cx", width), patches.setAttribute(part, ext, "cy", height));
188
+ count(1);
189
+ }
190
+ return items;
191
+ }
192
+ /**
193
+ * Patches that mark every `w:p` of a part with a hidden bookmark carrying
194
+ * its id: `_wd` plus the file's `w14:paraId`, or a generated id for a
195
+ * paragraph without one, unique across the document's story parts. The
196
+ * bookmark pair goes right after `w:pPr`, where the schema allows it, with
197
+ * ids above any the document already uses.
198
+ */
199
+ function paragraphIdPatches(part, state, count) {
200
+ const paragraphs = assignParagraphIds(part, state);
201
+ if (paragraphs.length === 0)
202
+ return [];
203
+ const items = [];
204
+ let generated = 0;
205
+ for (const { paragraph, id, authored } of paragraphs) {
206
+ if (!authored)
207
+ generated += 1;
208
+ const name = `${PARAGRAPH_BOOKMARK_PREFIX}${id}`;
209
+ const start = `<w:bookmarkStart w:id="${state.bookmarkId}" w:name="${name}"/>`;
210
+ const end = `<w:bookmarkEnd w:id="${state.bookmarkId}"/>`;
211
+ state.bookmarkId += 1;
212
+ const pPr = paragraph.children.find((child) => child.local === "pPr" && child.namespace === W_NS);
213
+ if (pPr)
214
+ items.push(patches.insertAfter(part, pPr, start), patches.insertAfter(part, pPr, end));
215
+ else if (paragraph.selfClosing) {
216
+ // `<w:p …/>` opened around the bookmarks, as one element.
217
+ const xml = part.text.slice(paragraph.start, paragraph.end);
218
+ items.push(patches.replaceElement(part, paragraph, `${xml.slice(0, -2)}>${start}${end}</w:p>`));
219
+ }
220
+ else if (paragraph.children.length === 0) {
221
+ const xml = part.text.slice(paragraph.start, paragraph.end);
222
+ const open = paragraph.contentStart - paragraph.start;
223
+ items.push(patches.replaceElement(part, paragraph, `${xml.slice(0, open)}${start}${end}${xml.slice(open)}`));
224
+ }
225
+ else
226
+ items.push(patches.insertBefore(part, paragraph.children[0], start), patches.insertBefore(part, paragraph.children[0], end));
227
+ }
228
+ count(paragraphs.length, generated);
229
+ return items;
230
+ }
@@ -1,11 +1,15 @@
1
1
  import type { AdapterOpenContext, DocumentAdapter, DocumentFormat, DocumentInfo, HyperlinkTarget, RenderViewport, SpreadsheetSheetInfo, TextRun, ViewerWarning } from "../contracts.js";
2
- import { type DocxModelLike } from "./docx-images.js";
2
+ import type { EditEngineProvider } from "../edit/engine.js";
3
+ import type { OoxmlEditProviderOptions } from "../edit/ooxml/worker.js";
4
+ import { type DocxModelDocument, type DocxParagraphIdResolver, type DocxRunSource } from "./docx-paragraphs.js";
3
5
  declare const LEGACY_FORMATS: readonly ["doc", "xls", "ppt"];
4
6
  type LegacyFormat = (typeof LEGACY_FORMATS)[number];
5
7
  interface EngineLoadOptions {
6
8
  readonly useGoogleFonts: false;
7
9
  readonly maxZipEntryBytes: number;
8
10
  readonly mode: "main";
11
+ /** Presentations: lay slides out in the background after the first ones. */
12
+ readonly progressiveLayout?: boolean;
9
13
  }
10
14
  interface EngineHyperlink {
11
15
  readonly kind: "external" | "internal";
@@ -21,6 +25,10 @@ interface DocxRun {
21
25
  readonly h: number;
22
26
  readonly fontSize: number;
23
27
  readonly font: string;
28
+ /** The `w14:paraId` of the run's paragraph, when the engine reads one. */
29
+ readonly paragraphId?: string;
30
+ /** Where the run came from in the engine's model; the paragraph bridge. */
31
+ readonly source?: DocxRunSource;
24
32
  readonly letterSpacingPx?: number;
25
33
  readonly transform?: string;
26
34
  readonly eastAsianVert?: boolean;
@@ -28,10 +36,11 @@ interface DocxRun {
28
36
  }
29
37
  interface DocxBackend {
30
38
  readonly pageCount: number;
31
- /** Render mode; the parsed model is only reachable in `main` mode. */
32
- readonly mode?: "main" | "worker";
33
- /** Parsed document model (main mode). Read lazily by the page layout. */
34
- readonly document?: DocxModelLike;
39
+ /**
40
+ * The parsed model (`main` mode). Read once at open for the paragraph
41
+ * bridge: its paragraphs keep the pre-pass bookmarks that name each `w:p`.
42
+ */
43
+ readonly document?: DocxModelDocument;
35
44
  pageSize(pageIndex: number): {
36
45
  widthPt: number;
37
46
  heightPt: number;
@@ -178,12 +187,16 @@ export interface LegacyConversionOptions {
178
187
  export interface OfficeAdapterOptions {
179
188
  readonly engines?: OfficeEngineLoaders;
180
189
  readonly legacy?: LegacyConversionOptions;
190
+ /** Where the OOXML edit worker is served from; the package's own by default. */
191
+ readonly edit?: OoxmlEditProviderOptions;
181
192
  }
182
193
  interface DocumentHandle {
183
194
  readonly kind: "document";
184
195
  readonly format: DocumentFormat;
185
196
  readonly backend: DocxBackend;
186
197
  readonly warnings: readonly ViewerWarning[];
198
+ /** Maps a run's source to the id of its `w:p` (cached per paragraph). */
199
+ readonly paragraphIdOf: DocxParagraphIdResolver;
187
200
  }
188
201
  interface PresentationHandle {
189
202
  readonly kind: "presentation";
@@ -203,9 +216,22 @@ type OfficeHandle = DocumentHandle | PresentationHandle | SpreadsheetHandle;
203
216
  export declare class OfficeDocumentAdapter implements DocumentAdapter<OfficeHandle> {
204
217
  #private;
205
218
  readonly id = "office";
219
+ /**
220
+ * PPTX and DOCX editing on the OOXML package layer. The worker and the
221
+ * engine client are imported on the first `edit()`; viewing never loads
222
+ * them.
223
+ */
224
+ readonly edit: EditEngineProvider;
206
225
  readonly formats: readonly ["docx", "docm", "xlsx", "xlsm", "pptx", "pptm", "ppsx", "doc", "xls", "ppt"];
207
226
  constructor(options?: OfficeAdapterOptions);
208
227
  open(input: Uint8Array, context: AdapterOpenContext): Promise<OfficeHandle>;
228
+ /**
229
+ * Edited bytes open as a fresh document (the engine owns no reusable
230
+ * state); a presentation lays its slides out progressively so the slide
231
+ * on screen paints without waiting for the whole deck — Firefox takes
232
+ * seconds for a 500-slide preflight. The viewer closes `previous`.
233
+ */
234
+ reopen(_previous: OfficeHandle, data: Uint8Array, context: AdapterOpenContext): Promise<OfficeHandle>;
209
235
  getInfo(handle: OfficeHandle): Promise<DocumentInfo>;
210
236
  render(handle: OfficeHandle, target: HTMLCanvasElement | OffscreenCanvas, viewport: RenderViewport, signal?: AbortSignal): Promise<void>;
211
237
  getTextMap(handle: OfficeHandle, pageIndex: number, signal?: AbortSignal): Promise<readonly TextRun[]>;