web-doc 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,26 @@
1
+ /**
2
+ * Word draws an inline picture at its declared extent even when that is wider
3
+ * than the text area, so a generated document that embeds a 21-inch chart on
4
+ * a 6.5-inch column shows a clipped picture. A viewer has no margin to spill
5
+ * into, so before the page layout runs the oversized inline pictures are
6
+ * scaled down to fit the section's content box, aspect ratio preserved.
7
+ * Anchored (floating) pictures keep their geometry: their position is part of
8
+ * the author's layout.
9
+ */
10
+ export interface DocxSectionGeometry {
11
+ readonly pageWidth: number;
12
+ readonly pageHeight: number;
13
+ readonly marginLeft: number;
14
+ readonly marginRight: number;
15
+ readonly marginTop: number;
16
+ readonly marginBottom: number;
17
+ }
18
+ export interface DocxModelLike {
19
+ readonly section: DocxSectionGeometry;
20
+ readonly body: readonly unknown[];
21
+ }
22
+ /**
23
+ * Shrink every inline picture that would not fit its section's content box.
24
+ * Mutates the model in place and returns how many pictures were scaled.
25
+ */
26
+ export declare function fitInlineImagesToPage(model: DocxModelLike): number;
@@ -0,0 +1,74 @@
1
+ /**
2
+ * Shrink every inline picture that would not fit its section's content box.
3
+ * Mutates the model in place and returns how many pictures were scaled.
4
+ */
5
+ export function fitInlineImagesToPage(model) {
6
+ let scaled = 0;
7
+ // A `<w:sectPr>` closes the section that ENDS at it, so the geometry for a
8
+ // run of body elements is only known once the break after them is reached.
9
+ let pending = [];
10
+ const flush = (geometry) => {
11
+ for (const element of pending)
12
+ scaled += fitElement(element, geometry);
13
+ pending = [];
14
+ };
15
+ for (const element of model.body) {
16
+ if (isSectionBreak(element)) {
17
+ flush(element.geom ?? model.section);
18
+ continue;
19
+ }
20
+ pending.push(element);
21
+ }
22
+ flush(model.section);
23
+ return scaled;
24
+ }
25
+ function fitElement(element, geometry) {
26
+ if (isParagraph(element)) {
27
+ let scaled = 0;
28
+ for (const run of element.runs ?? [])
29
+ if (isInlineImage(run) && fitImage(run, geometry))
30
+ scaled += 1;
31
+ return scaled;
32
+ }
33
+ if (isTable(element)) {
34
+ let scaled = 0;
35
+ for (const row of element.rows ?? [])
36
+ for (const cell of row.cells ?? [])
37
+ for (const child of cell.content ?? [])
38
+ scaled += fitElement(child, geometry);
39
+ return scaled;
40
+ }
41
+ return 0;
42
+ }
43
+ function fitImage(image, geometry) {
44
+ const contentWidth = geometry.pageWidth - geometry.marginLeft - geometry.marginRight;
45
+ const contentHeight = geometry.pageHeight - geometry.marginTop - geometry.marginBottom;
46
+ if (!(contentWidth > 0 && contentHeight > 0) ||
47
+ !(image.widthPt > 0 && image.heightPt > 0))
48
+ return false;
49
+ const scale = Math.min(1, contentWidth / image.widthPt, contentHeight / image.heightPt);
50
+ if (scale >= 1)
51
+ return false;
52
+ image.widthPt *= scale;
53
+ image.heightPt *= scale;
54
+ return true;
55
+ }
56
+ function isRecord(value) {
57
+ return typeof value === "object" && value !== null;
58
+ }
59
+ function isParagraph(value) {
60
+ return isRecord(value) && value.type === "paragraph";
61
+ }
62
+ function isTable(value) {
63
+ return isRecord(value) && value.type === "table";
64
+ }
65
+ function isSectionBreak(value) {
66
+ return isRecord(value) && value.type === "sectionBreak";
67
+ }
68
+ function isInlineImage(value) {
69
+ return (isRecord(value) &&
70
+ value.type === "image" &&
71
+ value.anchor !== true &&
72
+ typeof value.widthPt === "number" &&
73
+ typeof value.heightPt === "number");
74
+ }
@@ -1,4 +1,5 @@
1
1
  import type { AdapterOpenContext, DocumentAdapter, DocumentFormat, DocumentInfo, HyperlinkTarget, RenderViewport, SpreadsheetSheetInfo, TextRun, ViewerWarning } from "../contracts.js";
2
+ import { type DocxModelLike } from "./docx-images.js";
2
3
  declare const LEGACY_FORMATS: readonly ["doc", "xls", "ppt"];
3
4
  type LegacyFormat = (typeof LEGACY_FORMATS)[number];
4
5
  interface EngineLoadOptions {
@@ -27,6 +28,10 @@ interface DocxRun {
27
28
  }
28
29
  interface DocxBackend {
29
30
  readonly pageCount: number;
31
+ /** Render mode; the parsed model is only reachable in `main` mode. */
32
+ readonly mode?: "main" | "worker";
33
+ /** Parsed document model (main mode). Read lazily by the page layout. */
34
+ readonly document?: DocxModelLike;
30
35
  pageSize(pageIndex: number): {
31
36
  widthPt: number;
32
37
  heightPt: number;
@@ -1,4 +1,5 @@
1
1
  import { abortError, ViewerError } from "../errors.js";
2
+ import { fitInlineImagesToPage } from "./docx-images.js";
2
3
  import { enforceContainerLimits } from "../limits.js";
3
4
  const MODERN_FORMATS = [
4
5
  "docx",
@@ -325,10 +326,11 @@ export class OfficeDocumentAdapter {
325
326
  return convertInWorker(data, format, workerUrl, moduleUrl, context.signal, context.limits.maxOperationMs);
326
327
  }
327
328
  async #loadDocx(data, options) {
328
- if (this.#options.engines?.docx)
329
- return this.#options.engines.docx(data, options);
330
- const { DocxDocument } = await import("@silurus/ooxml/docx");
331
- return DocxDocument.load(data, options);
329
+ const backend = this.#options.engines?.docx
330
+ ? await this.#options.engines.docx(data, options)
331
+ : await (await import("@silurus/ooxml/docx")).DocxDocument.load(data, options);
332
+ fitDocxInlineImages(backend);
333
+ return backend;
332
334
  }
333
335
  async #loadXlsx(data, options) {
334
336
  if (this.#options.engines?.xlsx)
@@ -670,3 +672,22 @@ function normalizeOfficeError(error) {
670
672
  function textDirection(text) {
671
673
  return /[\u0590-\u08ff\ufb1d-\ufefc]/u.test(text) ? "rtl" : "ltr";
672
674
  }
675
+ /**
676
+ * Oversized inline pictures are shrunk to the section's content box before
677
+ * the engine paginates (the layout is built lazily on first page access).
678
+ * The model is reachable in `main` mode only; a worker-mode engine keeps
679
+ * Word's geometry.
680
+ */
681
+ function fitDocxInlineImages(backend) {
682
+ if (backend.mode === "worker")
683
+ return;
684
+ let model;
685
+ try {
686
+ model = backend.document;
687
+ }
688
+ catch {
689
+ return;
690
+ }
691
+ if (model)
692
+ fitInlineImagesToPage(model);
693
+ }
@@ -236,16 +236,27 @@ export interface FuzzySearchOptions {
236
236
  * partly survives on a page, such as a citation that spans a page break.
237
237
  */
238
238
  readonly maxScore?: number;
239
- /** Query characters considered. Default `2000`. */
239
+ /**
240
+ * Query characters considered. The matcher's cost grows with the query and
241
+ * a passage is identified well before its end, so the default `600` keeps
242
+ * a page under about 100 ms; the highlight covers the matched prefix.
243
+ */
240
244
  readonly maxQueryLength?: number;
241
245
  /** Characters of each page's text considered. Default `20000`. */
242
246
  readonly maxPageTextLength?: number;
243
247
  /**
244
248
  * Pages compared per batch. The scan proceeds nearest to `nearPage` first
245
249
  * and stops after the first batch with a match, yielding to the event loop
246
- * between batches. Default `4`.
250
+ * between batches. Default `2`.
247
251
  */
248
252
  readonly pagesPerBatch?: number;
253
+ /**
254
+ * With `nearPage`, how many pages nearest to the hint the fallback scans
255
+ * before giving up; the passage a hint points at sits within a few pages
256
+ * of it, and the rest of a long document is not worth the cost. Without a
257
+ * hint every page in range is scanned. Default `12`.
258
+ */
259
+ readonly pageWindow?: number;
249
260
  }
250
261
  export interface SearchOptions {
251
262
  readonly caseSensitive?: boolean;
@@ -13,6 +13,7 @@ export interface ResolvedFuzzySearchOptions {
13
13
  readonly maxQueryLength: number;
14
14
  readonly maxPageTextLength: number;
15
15
  readonly pagesPerBatch: number;
16
+ readonly pageWindow: number;
16
17
  }
17
18
  export declare const DEFAULT_FUZZY_SEARCH_OPTIONS: ResolvedFuzzySearchOptions;
18
19
  /**
@@ -2,9 +2,12 @@ import Fuse from "fuse.js";
2
2
  export const DEFAULT_FUZZY_SEARCH_OPTIONS = Object.freeze({
3
3
  threshold: 0.3,
4
4
  maxScore: 0.4,
5
- maxQueryLength: 2000,
5
+ // Bitap cost grows with the query; 600 characters still identifies a
6
+ // passage while keeping a page under ~100 ms on a laptop.
7
+ maxQueryLength: 600,
6
8
  maxPageTextLength: 20_000,
7
- pagesPerBatch: 4,
9
+ pagesPerBatch: 2,
10
+ pageWindow: 12,
8
11
  });
9
12
  /**
10
13
  * Layer fuzzy settings in precedence order (viewer defaults first, then the
@@ -29,6 +32,7 @@ export function resolveFuzzySearchOptions(...layers) {
29
32
  maxQueryLength: positiveInteger(layer.maxQueryLength, resolved.maxQueryLength),
30
33
  maxPageTextLength: positiveInteger(layer.maxPageTextLength, resolved.maxPageTextLength),
31
34
  pagesPerBatch: positiveInteger(layer.pagesPerBatch, resolved.pagesPerBatch),
35
+ pageWindow: positiveInteger(layer.pageWindow, resolved.pageWindow),
32
36
  };
33
37
  }
34
38
  return enabled ? resolved : undefined;
package/dist/index.d.ts CHANGED
@@ -6,6 +6,8 @@ export * from "./format.js";
6
6
  export * from "./limits.js";
7
7
  export * from "./interaction.js";
8
8
  export * from "./fuzzy-search.js";
9
+ export * from "./search-reveal.js";
10
+ export * from "./adapters/docx-images.js";
9
11
  export * from "./render-scheduler.js";
10
12
  export * from "./i18n.js";
11
13
  export * from "./font-manifest.js";
package/dist/index.js CHANGED
@@ -6,6 +6,8 @@ export * from "./format.js";
6
6
  export * from "./limits.js";
7
7
  export * from "./interaction.js";
8
8
  export * from "./fuzzy-search.js";
9
+ export * from "./search-reveal.js";
10
+ export * from "./adapters/docx-images.js";
9
11
  export * from "./render-scheduler.js";
10
12
  export * from "./i18n.js";
11
13
  export * from "./font-manifest.js";
@@ -0,0 +1,21 @@
1
+ import type { SearchMatch, TextRun } from "./contracts.js";
2
+ /**
3
+ * Vertical position of a match inside its page, in the coordinate units of
4
+ * the page's text layer: the `y` of the first run the match overlaps.
5
+ * Offsets follow the same convention as the highlight layers (a run's
6
+ * explicit `logicalStart`/`logicalEnd`, else the running text length), so
7
+ * the position agrees with where the highlight is painted. `undefined` when
8
+ * no run overlaps the match.
9
+ */
10
+ export declare function matchTopInPage(runs: readonly TextRun[], match: SearchMatch): number | undefined;
11
+ /**
12
+ * Scroll offset that places a match about a third of the way down the
13
+ * viewport — far enough from the top edge to read the line before it, never
14
+ * above the page's own top so a match near the top of a page still shows the
15
+ * page head. `matchTop` is the match's offset from the page top in CSS px.
16
+ */
17
+ export declare function revealScrollTop(options: {
18
+ readonly pageTop: number;
19
+ readonly matchTop: number;
20
+ readonly viewportHeight: number;
21
+ }): number;
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Vertical position of a match inside its page, in the coordinate units of
3
+ * the page's text layer: the `y` of the first run the match overlaps.
4
+ * Offsets follow the same convention as the highlight layers (a run's
5
+ * explicit `logicalStart`/`logicalEnd`, else the running text length), so
6
+ * the position agrees with where the highlight is painted. `undefined` when
7
+ * no run overlaps the match.
8
+ */
9
+ export function matchTopInPage(runs, match) {
10
+ let offset = 0;
11
+ for (const run of runs) {
12
+ const start = run.logicalStart ?? offset;
13
+ const end = run.logicalEnd ?? start + run.text.length;
14
+ offset = end;
15
+ if (match.start < end && match.end > start)
16
+ return run.y;
17
+ }
18
+ return undefined;
19
+ }
20
+ /**
21
+ * Scroll offset that places a match about a third of the way down the
22
+ * viewport — far enough from the top edge to read the line before it, never
23
+ * above the page's own top so a match near the top of a page still shows the
24
+ * page head. `matchTop` is the match's offset from the page top in CSS px.
25
+ */
26
+ export function revealScrollTop(options) {
27
+ const lead = Math.max(0, options.viewportHeight) / 3;
28
+ return Math.max(options.pageTop, options.pageTop + options.matchTop - lead);
29
+ }
@@ -24,6 +24,8 @@ export declare class SpreadsheetViewport {
24
24
  setDocument(info: DocumentInfo | undefined): void;
25
25
  update(): void;
26
26
  panBy(deltaX: number, deltaY: number): void;
27
+ /** Sheet matches are cell-addressed; the sheet viewport scrolls per cell already. */
28
+ revealMatch(): Promise<void>;
27
29
  goToPage(pageIndex: number): void;
28
30
  fitWidth(): number;
29
31
  fitPage(): number;
@@ -241,6 +241,8 @@ export class SpreadsheetViewport {
241
241
  this.#reportPan();
242
242
  this.schedule();
243
243
  }
244
+ /** Sheet matches are cell-addressed; the sheet viewport scrolls per cell already. */
245
+ async revealMatch() { }
244
246
  goToPage(pageIndex) {
245
247
  if (pageIndex === this.#sheetIndex)
246
248
  return;
package/dist/viewer.js CHANGED
@@ -415,7 +415,9 @@ export class DocumentViewer {
415
415
  // Pages are compared nearest to the hint first, a batch at a time,
416
416
  // and the scan stops at the first batch that holds the passage; a
417
417
  // yield between batches keeps a long document from freezing the UI.
418
- const order = pagesNearestFirst(firstPage, lastPage, nearPage);
418
+ // With a hint the passage sits near it, so only that neighbourhood is
419
+ // worth the fuzzy cost; without one every page is a candidate.
420
+ const order = pagesNearestFirst(firstPage, lastPage, nearPage).slice(0, nearPage === undefined ? undefined : fuzzy.pageWindow);
419
421
  for (let offset = 0; offset < order.length && matches.length === 0; offset += fuzzy.pagesPerBatch) {
420
422
  if (offset > 0)
421
423
  await yieldToEventLoop();
@@ -442,7 +444,7 @@ export class DocumentViewer {
442
444
  });
443
445
  this.#searchResult = result;
444
446
  if (result.activeIndex >= 0)
445
- this.#goToPage(result.matches[result.activeIndex].pageIndex, true);
447
+ this.#revealSearchMatch(result.matches[result.activeIndex]);
446
448
  this.#emit("searchchange", result);
447
449
  this.#viewport?.update();
448
450
  return result;
@@ -683,11 +685,16 @@ export class DocumentViewer {
683
685
  current.matches.length;
684
686
  const result = immutableSearchResult({ ...current, activeIndex });
685
687
  this.#searchResult = result;
686
- this.#goToPage(result.matches[activeIndex].pageIndex, true);
688
+ this.#revealSearchMatch(result.matches[activeIndex]);
687
689
  this.#emit("searchchange", result);
688
690
  this.#viewport?.update();
689
691
  return result;
690
692
  }
693
+ /** Land on the match's page, then bring the match itself into view. */
694
+ #revealSearchMatch(match) {
695
+ this.#goToPage(match.pageIndex, true);
696
+ void this.#viewport?.revealMatch(match);
697
+ }
691
698
  #goToPage(pageIndex, scrollViewport) {
692
699
  this.#assertAlive();
693
700
  const upperBound = Math.max(0, this.#state.pageCount - 1);
@@ -18,6 +18,7 @@ interface ViewportStrategy {
18
18
  update(): void;
19
19
  panBy(deltaX: number, deltaY: number): void;
20
20
  goToPage(pageIndex: number): void;
21
+ revealMatch(match: SearchMatch): Promise<void>;
21
22
  fitWidth(): number;
22
23
  fitPage(): number;
23
24
  destroy(): void;
@@ -32,6 +33,7 @@ export declare class AdaptiveViewport implements ViewportStrategy {
32
33
  update(): void;
33
34
  panBy(deltaX: number, deltaY: number): void;
34
35
  goToPage(pageIndex: number): void;
36
+ revealMatch(match: SearchMatch): Promise<void>;
35
37
  fitWidth(): number;
36
38
  fitPage(): number;
37
39
  destroy(): void;
@@ -46,6 +48,13 @@ export declare class ViewerViewport {
46
48
  update(): void;
47
49
  panBy(deltaX: number, deltaY: number): void;
48
50
  goToPage(pageIndex: number): void;
51
+ /**
52
+ * Scroll so the match itself is in view, not just its page: a page can be
53
+ * taller than the viewport, and a search that only lands on the page top
54
+ * leaves a match further down invisible until the reader scrolls. Resolves
55
+ * once the text runs are known; a navigation that happened meanwhile wins.
56
+ */
57
+ revealMatch(match: SearchMatch): Promise<void>;
49
58
  fitWidth(): number;
50
59
  fitPage(): number;
51
60
  schedule(): void;
package/dist/viewport.js CHANGED
@@ -1,8 +1,11 @@
1
+ import { matchTopInPage, revealScrollTop } from "./search-reveal.js";
1
2
  import { snapGraphemeOffset } from "./interaction.js";
2
3
  import { SpreadsheetViewport } from "./spreadsheet-viewport.js";
3
4
  const BASE_WIDTH = 816;
4
5
  const BASE_HEIGHT = 1056;
5
6
  const PAGE_GAP = 24;
7
+ /** Page slots sit this far inside the spacer (top and left). */
8
+ const SLOT_INSET = 12;
6
9
  export class AdaptiveViewport {
7
10
  #container;
8
11
  #host;
@@ -36,6 +39,9 @@ export class AdaptiveViewport {
36
39
  goToPage(pageIndex) {
37
40
  this.#strategy.goToPage(pageIndex);
38
41
  }
42
+ revealMatch(match) {
43
+ return this.#strategy.revealMatch(match);
44
+ }
39
45
  fitWidth() {
40
46
  return this.#strategy.fitWidth();
41
47
  }
@@ -154,6 +160,36 @@ export class ViewerViewport {
154
160
  }
155
161
  this.schedule();
156
162
  }
163
+ /**
164
+ * Scroll so the match itself is in view, not just its page: a page can be
165
+ * taller than the viewport, and a search that only lands on the page top
166
+ * leaves a match further down invisible until the reader scrolls. Resolves
167
+ * once the text runs are known; a navigation that happened meanwhile wins.
168
+ */
169
+ async revealMatch(match) {
170
+ if (this.#layout !== "continuous" || !this.#info)
171
+ return;
172
+ let runs;
173
+ try {
174
+ runs = await this.#host.getTextRuns(match.pageIndex);
175
+ }
176
+ catch {
177
+ return;
178
+ }
179
+ if (this.#destroyed || this.#host.state.pageIndex !== match.pageIndex)
180
+ return;
181
+ const top = matchTopInPage(runs, match);
182
+ if (top === undefined)
183
+ return;
184
+ const zoom = this.#host.state.zoom;
185
+ const metrics = pageMetrics(this.#info, zoom);
186
+ this.#root.scrollTop = revealScrollTop({
187
+ pageTop: metrics.offsets[match.pageIndex] ?? 0,
188
+ matchTop: SLOT_INSET + top * zoom,
189
+ viewportHeight: this.#root.clientHeight,
190
+ });
191
+ this.schedule();
192
+ }
157
193
  fitWidth() {
158
194
  const size = naturalPageSize(this.#info, this.#host.state.pageIndex);
159
195
  return Math.max(0.1, Math.min(8, (this.#root.clientWidth - 24) / size.width));
@@ -217,7 +253,9 @@ export class ViewerViewport {
217
253
  const slot = this.#slots.get(pageIndex) ?? this.#createSlot(pageIndex);
218
254
  const width = metrics.widths[pageIndex] ?? BASE_WIDTH * state.zoom;
219
255
  const height = metrics.heights[pageIndex] ?? BASE_HEIGHT * state.zoom;
220
- const top = this.#layout === "single" ? 12 : (metrics.offsets[pageIndex] ?? 0) + 12;
256
+ const top = this.#layout === "single"
257
+ ? SLOT_INSET
258
+ : (metrics.offsets[pageIndex] ?? 0) + SLOT_INSET;
221
259
  const contentWidth = Math.max(this.#root.clientWidth, metrics.maxWidth + 24);
222
260
  slot.root.style.top = `${top}px`;
223
261
  slot.root.style.left = `${Math.max(12, (contentWidth - width) / 2)}px`;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "web-doc",
3
- "version": "0.4.0",
3
+ "version": "0.5.0",
4
4
  "description": "web-doc — embeddable browser-only document viewer with Rust/WASM adapters (a fork of Zrimo)",
5
5
  "keywords": [
6
6
  "document-viewer",