web-doc 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -9,7 +9,8 @@
9
9
 
10
10
  > web-doc is a fork of [Zrimo](https://github.com/bnku/zrimo) maintained at
11
11
  > [leonidkuznetsov18/web-doc](https://github.com/leonidkuznetsov18/web-doc).
12
- > It adds page-scoped search (`SearchOptions.pageRange`) and is released
12
+ > It adds page-scoped search (`SearchOptions.pageRange`), an approximate page
13
+ > hint with a fuzzy fallback (`SearchOptions.nearPage` / `fuzzy`) and is released
13
14
  > independently; the runtime API, CSS hooks (`.zrimo-ui`, `--zrimo-*`) and
14
15
  > asset layout are the same as upstream's `@zrimo/viewer`.
15
16
 
@@ -7,3 +7,12 @@ export declare function drawEncodedImage(data: Uint8Array, mimeType: string, tar
7
7
  width: number;
8
8
  height: number;
9
9
  }>;
10
+ /**
11
+ * Decode an encoded image only to read its natural size, in CSS pixels at
12
+ * zoom 1. Uses the same decoder as {@link drawEncodedImage}, so EXIF
13
+ * orientation is applied and the result matches what a render will draw.
14
+ */
15
+ export declare function measureEncodedImage(data: Uint8Array, mimeType: string): Promise<{
16
+ width: number;
17
+ height: number;
18
+ }>;
@@ -27,6 +27,20 @@ export async function drawEncodedImage(data, mimeType, target, options) {
27
27
  decoded.close();
28
28
  }
29
29
  }
30
+ /**
31
+ * Decode an encoded image only to read its natural size, in CSS pixels at
32
+ * zoom 1. Uses the same decoder as {@link drawEncodedImage}, so EXIF
33
+ * orientation is applied and the result matches what a render will draw.
34
+ */
35
+ export async function measureEncodedImage(data, mimeType) {
36
+ const decoded = await decodeImage(new Blob([data.slice()], { type: mimeType }));
37
+ try {
38
+ return { width: decoded.width, height: decoded.height };
39
+ }
40
+ finally {
41
+ decoded.close();
42
+ }
43
+ }
30
44
  async function decodeImage(blob) {
31
45
  if (typeof createImageBitmap === "function") {
32
46
  const bitmap = await createImageBitmap(blob, {
@@ -0,0 +1,26 @@
1
+ /**
2
+ * Word draws an inline picture at its declared extent even when that is wider
3
+ * than the text area, so a generated document that embeds a 21-inch chart on
4
+ * a 6.5-inch column shows a clipped picture. A viewer has no margin to spill
5
+ * into, so before the page layout runs the oversized inline pictures are
6
+ * scaled down to fit the section's content box, aspect ratio preserved.
7
+ * Anchored (floating) pictures keep their geometry: their position is part of
8
+ * the author's layout.
9
+ */
10
+ export interface DocxSectionGeometry {
11
+ readonly pageWidth: number;
12
+ readonly pageHeight: number;
13
+ readonly marginLeft: number;
14
+ readonly marginRight: number;
15
+ readonly marginTop: number;
16
+ readonly marginBottom: number;
17
+ }
18
+ export interface DocxModelLike {
19
+ readonly section: DocxSectionGeometry;
20
+ readonly body: readonly unknown[];
21
+ }
22
+ /**
23
+ * Shrink every inline picture that would not fit its section's content box.
24
+ * Mutates the model in place and returns how many pictures were scaled.
25
+ */
26
+ export declare function fitInlineImagesToPage(model: DocxModelLike): number;
@@ -0,0 +1,74 @@
1
+ /**
2
+ * Shrink every inline picture that would not fit its section's content box.
3
+ * Mutates the model in place and returns how many pictures were scaled.
4
+ */
5
+ export function fitInlineImagesToPage(model) {
6
+ let scaled = 0;
7
+ // A `<w:sectPr>` closes the section that ENDS at it, so the geometry for a
8
+ // run of body elements is only known once the break after them is reached.
9
+ let pending = [];
10
+ const flush = (geometry) => {
11
+ for (const element of pending)
12
+ scaled += fitElement(element, geometry);
13
+ pending = [];
14
+ };
15
+ for (const element of model.body) {
16
+ if (isSectionBreak(element)) {
17
+ flush(element.geom ?? model.section);
18
+ continue;
19
+ }
20
+ pending.push(element);
21
+ }
22
+ flush(model.section);
23
+ return scaled;
24
+ }
25
+ function fitElement(element, geometry) {
26
+ if (isParagraph(element)) {
27
+ let scaled = 0;
28
+ for (const run of element.runs ?? [])
29
+ if (isInlineImage(run) && fitImage(run, geometry))
30
+ scaled += 1;
31
+ return scaled;
32
+ }
33
+ if (isTable(element)) {
34
+ let scaled = 0;
35
+ for (const row of element.rows ?? [])
36
+ for (const cell of row.cells ?? [])
37
+ for (const child of cell.content ?? [])
38
+ scaled += fitElement(child, geometry);
39
+ return scaled;
40
+ }
41
+ return 0;
42
+ }
43
+ function fitImage(image, geometry) {
44
+ const contentWidth = geometry.pageWidth - geometry.marginLeft - geometry.marginRight;
45
+ const contentHeight = geometry.pageHeight - geometry.marginTop - geometry.marginBottom;
46
+ if (!(contentWidth > 0 && contentHeight > 0) ||
47
+ !(image.widthPt > 0 && image.heightPt > 0))
48
+ return false;
49
+ const scale = Math.min(1, contentWidth / image.widthPt, contentHeight / image.heightPt);
50
+ if (scale >= 1)
51
+ return false;
52
+ image.widthPt *= scale;
53
+ image.heightPt *= scale;
54
+ return true;
55
+ }
56
+ function isRecord(value) {
57
+ return typeof value === "object" && value !== null;
58
+ }
59
+ function isParagraph(value) {
60
+ return isRecord(value) && value.type === "paragraph";
61
+ }
62
+ function isTable(value) {
63
+ return isRecord(value) && value.type === "table";
64
+ }
65
+ function isSectionBreak(value) {
66
+ return isRecord(value) && value.type === "sectionBreak";
67
+ }
68
+ function isInlineImage(value) {
69
+ return (isRecord(value) &&
70
+ value.type === "image" &&
71
+ value.anchor !== true &&
72
+ typeof value.widthPt === "number" &&
73
+ typeof value.heightPt === "number");
74
+ }
@@ -1,9 +1,11 @@
1
- import type { AdapterOpenContext, DocumentAdapter, DocumentInfo, RenderViewport, ViewerWarning } from "../contracts.js";
1
+ import type { AdapterOpenContext, DocumentAdapter, DocumentInfo, PageSize, RenderViewport, ViewerWarning } from "../contracts.js";
2
2
  type NativeImageFormat = "png" | "jpeg" | "gif" | "webp" | "bmp";
3
3
  interface NativeHandle {
4
4
  readonly kind: "native";
5
5
  readonly format: NativeImageFormat;
6
6
  readonly data: Uint8Array;
7
+ /** Natural size when the browser could decode the image at open time. */
8
+ readonly size: PageSize | undefined;
7
9
  readonly warnings: readonly ViewerWarning[];
8
10
  }
9
11
  interface TiffHandle {
@@ -1,5 +1,5 @@
1
1
  import { abortError, ViewerError } from "../errors.js";
2
- import { drawEncodedImage } from "./bitmap.js";
2
+ import { drawEncodedImage, measureEncodedImage } from "./bitmap.js";
3
3
  export class ImageDocumentAdapter {
4
4
  id = "image";
5
5
  formats = ["png", "jpeg", "gif", "webp", "bmp", "tiff"];
@@ -35,10 +35,14 @@ export class ImageDocumentAdapter {
35
35
  }
36
36
  const format = context.format;
37
37
  const animated = isAnimated(data, format);
38
+ const size = await measureNativeImage(data, format);
39
+ if (context.signal.aborted)
40
+ throw abortError();
38
41
  return {
39
42
  kind: "native",
40
43
  format,
41
44
  data,
45
+ size,
42
46
  warnings: animated
43
47
  ? [
44
48
  {
@@ -51,10 +55,12 @@ export class ImageDocumentAdapter {
51
55
  };
52
56
  }
53
57
  async getInfo(handle) {
58
+ const pageSizes = pageSizesOf(handle);
54
59
  return {
55
60
  format: handle.format,
56
61
  unit: "image",
57
62
  pageCount: handle.kind === "tiff" ? handle.backend.pages.length : 1,
63
+ ...(pageSizes ? { pageSizes } : {}),
58
64
  warnings: handle.warnings,
59
65
  };
60
66
  }
@@ -180,6 +186,30 @@ function requestWorker(worker, payload, transfer, signal, timeoutMs = 30_000) {
180
186
  worker.postMessage(payload, transfer);
181
187
  });
182
188
  }
189
+ /**
190
+ * Natural page geometry for the viewport layout and fit modes. Without it the
191
+ * viewport falls back to a letter-sized page, so fit-to-width scales a photo
192
+ * against a size it does not have and the bitmap overflows its slot.
193
+ */
194
+ function pageSizesOf(handle) {
195
+ if (handle.kind === "tiff")
196
+ return handle.backend.pages.map(({ width, height }) => ({ width, height }));
197
+ return handle.size ? [handle.size] : undefined;
198
+ }
199
+ /**
200
+ * Read the natural size once at open time. A decode failure is not fatal here:
201
+ * the render path reports it with its own error, and the viewport keeps its
202
+ * fallback geometry until then.
203
+ */
204
+ async function measureNativeImage(data, format) {
205
+ try {
206
+ const { width, height } = await measureEncodedImage(data, mimeType(format));
207
+ return width > 0 && height > 0 ? { width, height } : undefined;
208
+ }
209
+ catch {
210
+ return undefined;
211
+ }
212
+ }
183
213
  function mimeType(format) {
184
214
  return format === "jpeg" ? "image/jpeg" : `image/${format}`;
185
215
  }
@@ -1,4 +1,5 @@
1
1
  import type { AdapterOpenContext, DocumentAdapter, DocumentFormat, DocumentInfo, HyperlinkTarget, RenderViewport, SpreadsheetSheetInfo, TextRun, ViewerWarning } from "../contracts.js";
2
+ import { type DocxModelLike } from "./docx-images.js";
2
3
  declare const LEGACY_FORMATS: readonly ["doc", "xls", "ppt"];
3
4
  type LegacyFormat = (typeof LEGACY_FORMATS)[number];
4
5
  interface EngineLoadOptions {
@@ -27,6 +28,10 @@ interface DocxRun {
27
28
  }
28
29
  interface DocxBackend {
29
30
  readonly pageCount: number;
31
+ /** Render mode; the parsed model is only reachable in `main` mode. */
32
+ readonly mode?: "main" | "worker";
33
+ /** Parsed document model (main mode). Read lazily by the page layout. */
34
+ readonly document?: DocxModelLike;
30
35
  pageSize(pageIndex: number): {
31
36
  widthPt: number;
32
37
  heightPt: number;
@@ -1,4 +1,5 @@
1
1
  import { abortError, ViewerError } from "../errors.js";
2
+ import { fitInlineImagesToPage } from "./docx-images.js";
2
3
  import { enforceContainerLimits } from "../limits.js";
3
4
  const MODERN_FORMATS = [
4
5
  "docx",
@@ -325,10 +326,11 @@ export class OfficeDocumentAdapter {
325
326
  return convertInWorker(data, format, workerUrl, moduleUrl, context.signal, context.limits.maxOperationMs);
326
327
  }
327
328
  async #loadDocx(data, options) {
328
- if (this.#options.engines?.docx)
329
- return this.#options.engines.docx(data, options);
330
- const { DocxDocument } = await import("@silurus/ooxml/docx");
331
- return DocxDocument.load(data, options);
329
+ const backend = this.#options.engines?.docx
330
+ ? await this.#options.engines.docx(data, options)
331
+ : await (await import("@silurus/ooxml/docx")).DocxDocument.load(data, options);
332
+ fitDocxInlineImages(backend);
333
+ return backend;
332
334
  }
333
335
  async #loadXlsx(data, options) {
334
336
  if (this.#options.engines?.xlsx)
@@ -670,3 +672,22 @@ function normalizeOfficeError(error) {
670
672
  function textDirection(text) {
671
673
  return /[\u0590-\u08ff\ufb1d-\ufefc]/u.test(text) ? "rtl" : "ltr";
672
674
  }
675
+ /**
676
+ * Oversized inline pictures are shrunk to the section's content box before
677
+ * the engine paginates (the layout is built lazily on first page access).
678
+ * The model is reachable in `main` mode only; a worker-mode engine keeps
679
+ * Word's geometry.
680
+ */
681
+ function fitDocxInlineImages(backend) {
682
+ if (backend.mode === "worker")
683
+ return;
684
+ let model;
685
+ try {
686
+ model = backend.document;
687
+ }
688
+ catch {
689
+ return;
690
+ }
691
+ if (model)
692
+ fitInlineImagesToPage(model);
693
+ }
@@ -1,6 +1,8 @@
1
- import type { AdapterOpenContext, DocumentAdapter, DocumentInfo, RenderViewport, ViewerWarning } from "../contracts.js";
1
+ import type { AdapterOpenContext, DocumentAdapter, DocumentInfo, PageSize, RenderViewport, ViewerWarning } from "../contracts.js";
2
2
  interface SvgHandle {
3
3
  readonly data: Uint8Array;
4
+ /** Intrinsic size from the root `width`/`height` or `viewBox`, when declared. */
5
+ readonly size: PageSize | undefined;
4
6
  readonly warnings: readonly ViewerWarning[];
5
7
  }
6
8
  export declare class SvgDocumentAdapter implements DocumentAdapter<SvgHandle> {
@@ -13,5 +15,12 @@ export declare class SvgDocumentAdapter implements DocumentAdapter<SvgHandle> {
13
15
  close(): void;
14
16
  }
15
17
  export declare function sanitizeSvg(source: string): string;
18
+ /**
19
+ * Intrinsic size of the root `<svg>` in CSS pixels at zoom 1: explicit
20
+ * `width`/`height` win, a `viewBox` stands in for a missing pair, and a
21
+ * document that declares neither has no natural size (the viewport keeps its
22
+ * fallback page geometry).
23
+ */
24
+ export declare function parseSvgSize(source: string): PageSize | undefined;
16
25
  export declare function createSvgAdapter(): SvgDocumentAdapter;
17
26
  export {};
@@ -19,6 +19,7 @@ export class SvgDocumentAdapter {
19
19
  const changed = sanitized !== source;
20
20
  return {
21
21
  data: new TextEncoder().encode(sanitized),
22
+ size: parseSvgSize(sanitized),
22
23
  warnings: changed
23
24
  ? [
24
25
  {
@@ -34,6 +35,7 @@ export class SvgDocumentAdapter {
34
35
  format: "svg",
35
36
  unit: "image",
36
37
  pageCount: 1,
38
+ ...(handle.size ? { pageSizes: [handle.size] } : {}),
37
39
  warnings: handle.warnings,
38
40
  };
39
41
  }
@@ -93,6 +95,61 @@ export function sanitizeSvg(source) {
93
95
  }
94
96
  return new XMLSerializer().serializeToString(document.documentElement);
95
97
  }
98
+ /** CSS absolute units → CSS pixels; percentages and unknown units are ignored. */
99
+ const SVG_LENGTH_UNITS = {
100
+ "": 1,
101
+ px: 1,
102
+ pt: 96 / 72,
103
+ pc: 16,
104
+ in: 96,
105
+ cm: 96 / 2.54,
106
+ mm: 96 / 25.4,
107
+ };
108
+ /**
109
+ * Intrinsic size of the root `<svg>` in CSS pixels at zoom 1: explicit
110
+ * `width`/`height` win, a `viewBox` stands in for a missing pair, and a
111
+ * document that declares neither has no natural size (the viewport keeps its
112
+ * fallback page geometry).
113
+ */
114
+ export function parseSvgSize(source) {
115
+ const root = /<svg\b([^>]*)>/i.exec(source);
116
+ if (!root)
117
+ return undefined;
118
+ const width = parseSvgLength(svgAttribute(root[1] ?? "", "width"));
119
+ const height = parseSvgLength(svgAttribute(root[1] ?? "", "height"));
120
+ if (width && height)
121
+ return { width, height };
122
+ const viewBox = svgAttribute(root[1] ?? "", "viewBox")
123
+ ?.trim()
124
+ .split(/[\s,]+/)
125
+ .map(Number);
126
+ if (viewBox?.length === 4 &&
127
+ viewBox.every(Number.isFinite) &&
128
+ viewBox[2] > 0 &&
129
+ viewBox[3] > 0) {
130
+ // One declared side scales the viewBox aspect; none means viewBox units.
131
+ if (width)
132
+ return { width, height: (width * viewBox[3]) / viewBox[2] };
133
+ if (height)
134
+ return { width: (height * viewBox[2]) / viewBox[3], height };
135
+ return { width: viewBox[2], height: viewBox[3] };
136
+ }
137
+ return undefined;
138
+ }
139
+ function svgAttribute(attributes, name) {
140
+ const match = new RegExp(`(?:^|\\s)${name}\\s*=\\s*(?:"([^"]*)"|'([^']*)')`, "i").exec(attributes);
141
+ return match?.[1] ?? match?.[2];
142
+ }
143
+ function parseSvgLength(value) {
144
+ const match = /^\s*([0-9]*\.?[0-9]+(?:e[+-]?[0-9]+)?)\s*([a-z%]*)\s*$/i.exec(value ?? "");
145
+ if (!match)
146
+ return undefined;
147
+ const factor = SVG_LENGTH_UNITS[match[2].toLowerCase()];
148
+ if (factor === undefined)
149
+ return undefined;
150
+ const length = Number(match[1]) * factor;
151
+ return Number.isFinite(length) && length > 0 ? length : undefined;
152
+ }
96
153
  export function createSvgAdapter() {
97
154
  return new SvgDocumentAdapter();
98
155
  }
@@ -112,6 +112,7 @@ export interface ViewerOptions {
112
112
  readonly layout?: "continuous" | "single";
113
113
  readonly overscan?: number;
114
114
  readonly translations?: Partial<ViewerTranslations>;
115
+ readonly search?: SearchDefaults;
115
116
  }
116
117
  export interface ViewerClientOptions {
117
118
  readonly fetch?: ViewerFetch;
@@ -216,6 +217,47 @@ export interface ViewerEventMap {
216
217
  readonly viewchange: ViewerState;
217
218
  readonly searchchange: SearchResult | null;
218
219
  }
220
+ /**
221
+ * Tuning for the fuzzy fallback that runs when the exact search finds
222
+ * nothing. Matching is delegated to Fuse.js: the query is compared to each
223
+ * page's text with a bounded edit budget, so spacing, line breaks, list
224
+ * bullets, table separators and typographic punctuation may differ from the
225
+ * source, and every hit maps back to the verbatim page text.
226
+ */
227
+ export interface FuzzySearchOptions {
228
+ /**
229
+ * Fuse.js `threshold`: the edit budget per 32-character chunk of the query,
230
+ * `0` exact to `1` anything. Default `0.3`.
231
+ */
232
+ readonly threshold?: number;
233
+ /**
234
+ * Highest Fuse.js score (`0` perfect, `1` no resemblance) a page may have to
235
+ * count as a match. Default `0.4`; raise it to accept a passage that only
236
+ * partly survives on a page, such as a citation that spans a page break.
237
+ */
238
+ readonly maxScore?: number;
239
+ /**
240
+ * Query characters considered. The matcher's cost grows with the query and
241
+ * a passage is identified well before its end, so the default `600` keeps
242
+ * a page under about 100 ms; the highlight covers the matched prefix.
243
+ */
244
+ readonly maxQueryLength?: number;
245
+ /** Characters of each page's text considered. Default `20000`. */
246
+ readonly maxPageTextLength?: number;
247
+ /**
248
+ * Pages compared per batch. The scan proceeds nearest to `nearPage` first
249
+ * and stops after the first batch with a match, yielding to the event loop
250
+ * between batches. Default `2`.
251
+ */
252
+ readonly pagesPerBatch?: number;
253
+ /**
254
+ * With `nearPage`, how many pages nearest to the hint the fallback scans
255
+ * before giving up; the passage a hint points at sits within a few pages
256
+ * of it, and the rest of a long document is not worth the cost. Without a
257
+ * hint every page in range is scanned. Default `12`.
258
+ */
259
+ readonly pageWindow?: number;
260
+ }
219
261
  export interface SearchOptions {
220
262
  readonly caseSensitive?: boolean;
221
263
  /**
@@ -225,6 +267,24 @@ export interface SearchOptions {
225
267
  * is rejected.
226
268
  */
227
269
  readonly pageRange?: readonly [number, number];
270
+ /**
271
+ * Approximate 0-based page the passage is expected on, for callers whose
272
+ * page numbers come from another pagination (a citation produced from a
273
+ * server-side render of the same file). Clamped to the document. The
274
+ * result's `activeIndex` becomes the match closest to it, and the fuzzy
275
+ * fallback scans pages nearest to it first.
276
+ */
277
+ readonly nearPage?: number;
278
+ /**
279
+ * Fall back to Fuse.js fuzzy matching when the exact search finds nothing.
280
+ * `true` uses the viewer's `search.fuzzy` defaults; an object enables it
281
+ * and overrides them; `false` disables it for this call.
282
+ */
283
+ readonly fuzzy?: boolean | FuzzySearchOptions;
284
+ }
285
+ /** Per-viewer defaults applied to every `search()` call. */
286
+ export interface SearchDefaults {
287
+ readonly fuzzy?: boolean | FuzzySearchOptions;
228
288
  }
229
289
  export interface SearchMatch {
230
290
  readonly pageIndex: number;
@@ -232,10 +292,14 @@ export interface SearchMatch {
232
292
  readonly end: number;
233
293
  readonly text: string;
234
294
  }
295
+ /** How the matches of a result were found. */
296
+ export type SearchStrategy = "exact" | "fuzzy";
235
297
  export interface SearchResult {
236
298
  readonly query: string;
237
299
  readonly matches: readonly SearchMatch[];
238
300
  readonly activeIndex: number;
301
+ /** Present when the result has matches. */
302
+ readonly strategy?: SearchStrategy;
239
303
  }
240
304
  export interface HeadlessRenderOptions {
241
305
  readonly zoom?: number;
@@ -0,0 +1,39 @@
1
+ import type { FuzzySearchOptions, SearchMatch } from "./contracts.js";
2
+ /**
3
+ * Fuzzy matching is delegated to Fuse.js (Bitap with a bounded edit budget per
4
+ * 32-character chunk). It bridges the gaps NFKC + case folding leave open when
5
+ * a query was not copied verbatim from the document: AI-generated citations
6
+ * and OCR'd passages differ from the source in spacing, line breaks, list
7
+ * bullets, table separators and typographic punctuation. Every hit maps back
8
+ * to original UTF-16 offsets, so the viewport highlights the verbatim text.
9
+ */
10
+ export interface ResolvedFuzzySearchOptions {
11
+ readonly threshold: number;
12
+ readonly maxScore: number;
13
+ readonly maxQueryLength: number;
14
+ readonly maxPageTextLength: number;
15
+ readonly pagesPerBatch: number;
16
+ readonly pageWindow: number;
17
+ }
18
+ export declare const DEFAULT_FUZZY_SEARCH_OPTIONS: ResolvedFuzzySearchOptions;
19
+ /**
20
+ * Layer fuzzy settings in precedence order (viewer defaults first, then the
21
+ * per-call option). `true` enables the defaults, an object enables and
22
+ * overrides them, `false` disables, `undefined` leaves the previous layer in
23
+ * place. Returns `undefined` when fuzzy matching ends up disabled.
24
+ */
25
+ export declare function resolveFuzzySearchOptions(...layers: readonly (boolean | FuzzySearchOptions | undefined)[]): ResolvedFuzzySearchOptions | undefined;
26
+ export interface FuzzyPageText {
27
+ readonly pageIndex: number;
28
+ readonly text: string;
29
+ }
30
+ /**
31
+ * One match per page whose text holds the query within the edit budget: the
32
+ * span from the first to the last matched character, so the highlight covers
33
+ * the passage as one block. Pages are returned in page order.
34
+ */
35
+ export declare function findFuzzyPageMatches(pages: readonly FuzzyPageText[], query: string, options: ResolvedFuzzySearchOptions, caseSensitive?: boolean): readonly SearchMatch[];
36
+ /** Page indices of `[first, last]` ordered by distance from `nearPage`, ties earlier-first. */
37
+ export declare function pagesNearestFirst(first: number, last: number, nearPage: number | undefined): number[];
38
+ /** Index of the match closest to `nearPage`; the first match when there is no hint. */
39
+ export declare function nearestMatchIndex(matches: readonly SearchMatch[], nearPage: number | undefined): number;
@@ -0,0 +1,118 @@
1
+ import Fuse from "fuse.js";
2
+ export const DEFAULT_FUZZY_SEARCH_OPTIONS = Object.freeze({
3
+ threshold: 0.3,
4
+ maxScore: 0.4,
5
+ // Bitap cost grows with the query; 600 characters still identifies a
6
+ // passage while keeping a page under ~100 ms on a laptop.
7
+ maxQueryLength: 600,
8
+ maxPageTextLength: 20_000,
9
+ pagesPerBatch: 2,
10
+ pageWindow: 12,
11
+ });
12
+ /**
13
+ * Layer fuzzy settings in precedence order (viewer defaults first, then the
14
+ * per-call option). `true` enables the defaults, an object enables and
15
+ * overrides them, `false` disables, `undefined` leaves the previous layer in
16
+ * place. Returns `undefined` when fuzzy matching ends up disabled.
17
+ */
18
+ export function resolveFuzzySearchOptions(...layers) {
19
+ let enabled = false;
20
+ let resolved = DEFAULT_FUZZY_SEARCH_OPTIONS;
21
+ for (const layer of layers) {
22
+ if (layer === undefined)
23
+ continue;
24
+ if (typeof layer === "boolean") {
25
+ enabled = layer;
26
+ continue;
27
+ }
28
+ enabled = true;
29
+ resolved = {
30
+ threshold: unitInterval(layer.threshold, resolved.threshold),
31
+ maxScore: unitInterval(layer.maxScore, resolved.maxScore),
32
+ maxQueryLength: positiveInteger(layer.maxQueryLength, resolved.maxQueryLength),
33
+ maxPageTextLength: positiveInteger(layer.maxPageTextLength, resolved.maxPageTextLength),
34
+ pagesPerBatch: positiveInteger(layer.pagesPerBatch, resolved.pagesPerBatch),
35
+ pageWindow: positiveInteger(layer.pageWindow, resolved.pageWindow),
36
+ };
37
+ }
38
+ return enabled ? resolved : undefined;
39
+ }
40
+ /**
41
+ * One match per page whose text holds the query within the edit budget: the
42
+ * span from the first to the last matched character, so the highlight covers
43
+ * the passage as one block. Pages are returned in page order.
44
+ */
45
+ export function findFuzzyPageMatches(pages, query, options, caseSensitive = false) {
46
+ const pattern = query.slice(0, options.maxQueryLength);
47
+ if (!pattern.trim())
48
+ return [];
49
+ const items = pages.map((page) => ({
50
+ pageIndex: page.pageIndex,
51
+ text: page.text.slice(0, options.maxPageTextLength),
52
+ }));
53
+ const fuse = new Fuse(items, {
54
+ keys: ["text"],
55
+ isCaseSensitive: caseSensitive,
56
+ ignoreDiacritics: false,
57
+ includeMatches: true,
58
+ includeScore: true,
59
+ // A citation can sit anywhere on the page; Fuse's location bias would
60
+ // otherwise penalize matches far from the start of the text.
61
+ ignoreLocation: true,
62
+ // Long page text must not dilute the score of a match inside it.
63
+ ignoreFieldNorm: true,
64
+ threshold: options.threshold,
65
+ minMatchCharLength: 3,
66
+ shouldSort: false,
67
+ });
68
+ const matches = [];
69
+ for (const result of fuse.search(pattern)) {
70
+ if ((result.score ?? 1) > options.maxScore)
71
+ continue;
72
+ const indices = result.matches?.[0]?.indices ?? [];
73
+ if (indices.length === 0)
74
+ continue;
75
+ let start = Number.POSITIVE_INFINITY;
76
+ let end = 0;
77
+ for (const [first, last] of indices) {
78
+ start = Math.min(start, first);
79
+ end = Math.max(end, last + 1);
80
+ }
81
+ const text = result.item.text.slice(start, end);
82
+ matches.push({ pageIndex: result.item.pageIndex, start, end, text });
83
+ }
84
+ return matches.sort((a, b) => a.pageIndex - b.pageIndex);
85
+ }
86
+ /** Page indices of `[first, last]` ordered by distance from `nearPage`, ties earlier-first. */
87
+ export function pagesNearestFirst(first, last, nearPage) {
88
+ const pages = Array.from({ length: last - first + 1 }, (_, i) => first + i);
89
+ if (nearPage === undefined)
90
+ return pages;
91
+ return pages.sort((a, b) => Math.abs(a - nearPage) - Math.abs(b - nearPage) || a - b);
92
+ }
93
+ /** Index of the match closest to `nearPage`; the first match when there is no hint. */
94
+ export function nearestMatchIndex(matches, nearPage) {
95
+ if (matches.length === 0)
96
+ return -1;
97
+ if (nearPage === undefined)
98
+ return 0;
99
+ let best = 0;
100
+ for (let index = 1; index < matches.length; index += 1)
101
+ if (Math.abs(matches[index].pageIndex - nearPage) <
102
+ Math.abs(matches[best].pageIndex - nearPage))
103
+ best = index;
104
+ return best;
105
+ }
106
+ function unitInterval(value, fallback) {
107
+ return value !== undefined &&
108
+ Number.isFinite(value) &&
109
+ value >= 0 &&
110
+ value <= 1
111
+ ? value
112
+ : fallback;
113
+ }
114
+ function positiveInteger(value, fallback) {
115
+ return value !== undefined && Number.isInteger(value) && value > 0
116
+ ? value
117
+ : fallback;
118
+ }
package/dist/index.d.ts CHANGED
@@ -5,6 +5,9 @@ export * from "./errors.js";
5
5
  export * from "./format.js";
6
6
  export * from "./limits.js";
7
7
  export * from "./interaction.js";
8
+ export * from "./fuzzy-search.js";
9
+ export * from "./search-reveal.js";
10
+ export * from "./adapters/docx-images.js";
8
11
  export * from "./render-scheduler.js";
9
12
  export * from "./i18n.js";
10
13
  export * from "./font-manifest.js";
package/dist/index.js CHANGED
@@ -5,6 +5,9 @@ export * from "./errors.js";
5
5
  export * from "./format.js";
6
6
  export * from "./limits.js";
7
7
  export * from "./interaction.js";
8
+ export * from "./fuzzy-search.js";
9
+ export * from "./search-reveal.js";
10
+ export * from "./adapters/docx-images.js";
8
11
  export * from "./render-scheduler.js";
9
12
  export * from "./i18n.js";
10
13
  export * from "./font-manifest.js";
@@ -0,0 +1,21 @@
1
+ import type { SearchMatch, TextRun } from "./contracts.js";
2
+ /**
3
+ * Vertical position of a match inside its page, in the coordinate units of
4
+ * the page's text layer: the `y` of the first run the match overlaps.
5
+ * Offsets follow the same convention as the highlight layers (a run's
6
+ * explicit `logicalStart`/`logicalEnd`, else the running text length), so
7
+ * the position agrees with where the highlight is painted. `undefined` when
8
+ * no run overlaps the match.
9
+ */
10
+ export declare function matchTopInPage(runs: readonly TextRun[], match: SearchMatch): number | undefined;
11
+ /**
12
+ * Scroll offset that places a match about a third of the way down the
13
+ * viewport — far enough from the top edge to read the line before it, never
14
+ * above the page's own top so a match near the top of a page still shows the
15
+ * page head. `matchTop` is the match's offset from the page top in CSS px.
16
+ */
17
+ export declare function revealScrollTop(options: {
18
+ readonly pageTop: number;
19
+ readonly matchTop: number;
20
+ readonly viewportHeight: number;
21
+ }): number;
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Vertical position of a match inside its page, in the coordinate units of
3
+ * the page's text layer: the `y` of the first run the match overlaps.
4
+ * Offsets follow the same convention as the highlight layers (a run's
5
+ * explicit `logicalStart`/`logicalEnd`, else the running text length), so
6
+ * the position agrees with where the highlight is painted. `undefined` when
7
+ * no run overlaps the match.
8
+ */
9
+ export function matchTopInPage(runs, match) {
10
+ let offset = 0;
11
+ for (const run of runs) {
12
+ const start = run.logicalStart ?? offset;
13
+ const end = run.logicalEnd ?? start + run.text.length;
14
+ offset = end;
15
+ if (match.start < end && match.end > start)
16
+ return run.y;
17
+ }
18
+ return undefined;
19
+ }
20
+ /**
21
+ * Scroll offset that places a match about a third of the way down the
22
+ * viewport — far enough from the top edge to read the line before it, never
23
+ * above the page's own top so a match near the top of a page still shows the
24
+ * page head. `matchTop` is the match's offset from the page top in CSS px.
25
+ */
26
+ export function revealScrollTop(options) {
27
+ const lead = Math.max(0, options.viewportHeight) / 3;
28
+ return Math.max(options.pageTop, options.pageTop + options.matchTop - lead);
29
+ }
@@ -24,6 +24,8 @@ export declare class SpreadsheetViewport {
24
24
  setDocument(info: DocumentInfo | undefined): void;
25
25
  update(): void;
26
26
  panBy(deltaX: number, deltaY: number): void;
27
+ /** Sheet matches are cell-addressed; the sheet viewport scrolls per cell already. */
28
+ revealMatch(): Promise<void>;
27
29
  goToPage(pageIndex: number): void;
28
30
  fitWidth(): number;
29
31
  fitPage(): number;
@@ -241,6 +241,8 @@ export class SpreadsheetViewport {
241
241
  this.#reportPan();
242
242
  this.schedule();
243
243
  }
244
+ /** Sheet matches are cell-addressed; the sheet viewport scrolls per cell already. */
245
+ async revealMatch() { }
244
246
  goToPage(pageIndex) {
245
247
  if (pageIndex === this.#sheetIndex)
246
248
  return;
package/dist/viewer.js CHANGED
@@ -2,6 +2,7 @@ import { linkedAbortController } from "./abort.js";
2
2
  import { detectFormat } from "./detect.js";
3
3
  import { abortError, normalizeError, ViewerError } from "./errors.js";
4
4
  import { cellRangeToTsv, cellRangesToTsv, findNormalizedMatches, normalizeCellRange, } from "./interaction.js";
5
+ import { findFuzzyPageMatches, nearestMatchIndex, pagesNearestFirst, resolveFuzzySearchOptions, } from "./fuzzy-search.js";
5
6
  import { enforceContainerLimits, resolveLimits } from "./limits.js";
6
7
  import { loadDocumentSource } from "./source.js";
7
8
  import { AdaptiveViewport } from "./viewport.js";
@@ -386,6 +387,9 @@ export class DocumentViewer {
386
387
  // Validated before the in-flight search is cancelled so a rejected range
387
388
  // leaves the current result and highlights untouched.
388
389
  const [firstPage, lastPage] = resolveSearchPageRange(info.pageCount, options.pageRange);
390
+ const nearPage = resolveNearPage(firstPage, lastPage, options.nearPage);
391
+ const fuzzy = resolveFuzzySearchOptions(this.#options.search?.fuzzy, options.fuzzy);
392
+ const caseSensitive = options.caseSensitive ?? false;
389
393
  this.#activeSearch?.abort();
390
394
  const controller = new AbortController();
391
395
  this.#activeSearch = controller;
@@ -396,23 +400,51 @@ export class DocumentViewer {
396
400
  return immutableSearchResult({ query, matches: [], activeIndex: -1 });
397
401
  }
398
402
  const matches = [];
403
+ let strategy = "exact";
399
404
  try {
405
+ const texts = new Map();
400
406
  for (let pageIndex = firstPage; pageIndex <= lastPage; pageIndex += 1) {
401
407
  if (controller.signal.aborted)
402
408
  throw abortError();
403
409
  const text = await this.getPageText(pageIndex, controller.signal);
404
- matches.push(...findNormalizedMatches(text, cleanQuery, pageIndex, options.caseSensitive ?? false));
410
+ texts.set(pageIndex, text);
411
+ matches.push(...findNormalizedMatches(text, cleanQuery, pageIndex, caseSensitive));
412
+ }
413
+ if (matches.length === 0 && fuzzy) {
414
+ // The page texts are already in memory, so the fallback is CPU only.
415
+ // Pages are compared nearest to the hint first, a batch at a time,
416
+ // and the scan stops at the first batch that holds the passage; a
417
+ // yield between batches keeps a long document from freezing the UI.
418
+ // With a hint the passage sits near it, so only that neighbourhood is
419
+ // worth the fuzzy cost; without one every page is a candidate.
420
+ const order = pagesNearestFirst(firstPage, lastPage, nearPage).slice(0, nearPage === undefined ? undefined : fuzzy.pageWindow);
421
+ for (let offset = 0; offset < order.length && matches.length === 0; offset += fuzzy.pagesPerBatch) {
422
+ if (offset > 0)
423
+ await yieldToEventLoop();
424
+ if (controller.signal.aborted)
425
+ throw abortError();
426
+ const batch = order
427
+ .slice(offset, offset + fuzzy.pagesPerBatch)
428
+ .map((pageIndex) => ({
429
+ pageIndex,
430
+ text: texts.get(pageIndex) ?? "",
431
+ }));
432
+ matches.push(...findFuzzyPageMatches(batch, cleanQuery, fuzzy, caseSensitive));
433
+ }
434
+ if (matches.length > 0)
435
+ strategy = "fuzzy";
405
436
  }
406
437
  if (generation !== this.#searchGeneration)
407
438
  throw abortError();
408
439
  const result = immutableSearchResult({
409
440
  query,
410
441
  matches,
411
- activeIndex: matches.length > 0 ? 0 : -1,
442
+ activeIndex: nearestMatchIndex(matches, nearPage),
443
+ ...(matches.length > 0 ? { strategy } : {}),
412
444
  });
413
445
  this.#searchResult = result;
414
446
  if (result.activeIndex >= 0)
415
- this.#goToPage(result.matches[result.activeIndex].pageIndex, true);
447
+ this.#revealSearchMatch(result.matches[result.activeIndex]);
416
448
  this.#emit("searchchange", result);
417
449
  this.#viewport?.update();
418
450
  return result;
@@ -653,11 +685,16 @@ export class DocumentViewer {
653
685
  current.matches.length;
654
686
  const result = immutableSearchResult({ ...current, activeIndex });
655
687
  this.#searchResult = result;
656
- this.#goToPage(result.matches[activeIndex].pageIndex, true);
688
+ this.#revealSearchMatch(result.matches[activeIndex]);
657
689
  this.#emit("searchchange", result);
658
690
  this.#viewport?.update();
659
691
  return result;
660
692
  }
693
+ /** Land on the match's page, then bring the match itself into view. */
694
+ #revealSearchMatch(match) {
695
+ this.#goToPage(match.pageIndex, true);
696
+ void this.#viewport?.revealMatch(match);
697
+ }
661
698
  #goToPage(pageIndex, scrollViewport) {
662
699
  this.#assertAlive();
663
700
  const upperBound = Math.max(0, this.#state.pageCount - 1);
@@ -783,6 +820,19 @@ function resolveSearchPageRange(pageCount, pageRange) {
783
820
  assertPageIndex(last, pageCount);
784
821
  return first <= last ? [first, last] : [last, first];
785
822
  }
823
+ /**
824
+ * Clamp the approximate page hint into the scanned window. A hint that comes
825
+ * from another pagination of the same file is allowed to overshoot; rejecting
826
+ * it would defeat its purpose.
827
+ */
828
+ function resolveNearPage(firstPage, lastPage, nearPage) {
829
+ if (nearPage === undefined || !Number.isFinite(nearPage))
830
+ return undefined;
831
+ return Math.min(lastPage, Math.max(firstPage, Math.trunc(nearPage)));
832
+ }
833
+ function yieldToEventLoop() {
834
+ return new Promise((resolve) => setTimeout(resolve, 0));
835
+ }
786
836
  function assertPageIndex(pageIndex, pageCount) {
787
837
  if (!Number.isInteger(pageIndex) || pageIndex < 0 || pageIndex >= pageCount)
788
838
  throw new ViewerError("render-failed", "Page index is out of range", {
@@ -18,6 +18,7 @@ interface ViewportStrategy {
18
18
  update(): void;
19
19
  panBy(deltaX: number, deltaY: number): void;
20
20
  goToPage(pageIndex: number): void;
21
+ revealMatch(match: SearchMatch): Promise<void>;
21
22
  fitWidth(): number;
22
23
  fitPage(): number;
23
24
  destroy(): void;
@@ -32,6 +33,7 @@ export declare class AdaptiveViewport implements ViewportStrategy {
32
33
  update(): void;
33
34
  panBy(deltaX: number, deltaY: number): void;
34
35
  goToPage(pageIndex: number): void;
36
+ revealMatch(match: SearchMatch): Promise<void>;
35
37
  fitWidth(): number;
36
38
  fitPage(): number;
37
39
  destroy(): void;
@@ -46,6 +48,13 @@ export declare class ViewerViewport {
46
48
  update(): void;
47
49
  panBy(deltaX: number, deltaY: number): void;
48
50
  goToPage(pageIndex: number): void;
51
+ /**
52
+ * Scroll so the match itself is in view, not just its page: a page can be
53
+ * taller than the viewport, and a search that only lands on the page top
54
+ * leaves a match further down invisible until the reader scrolls. Resolves
55
+ * once the text runs are known; a navigation that happened meanwhile wins.
56
+ */
57
+ revealMatch(match: SearchMatch): Promise<void>;
49
58
  fitWidth(): number;
50
59
  fitPage(): number;
51
60
  schedule(): void;
package/dist/viewport.js CHANGED
@@ -1,8 +1,11 @@
1
+ import { matchTopInPage, revealScrollTop } from "./search-reveal.js";
1
2
  import { snapGraphemeOffset } from "./interaction.js";
2
3
  import { SpreadsheetViewport } from "./spreadsheet-viewport.js";
3
4
  const BASE_WIDTH = 816;
4
5
  const BASE_HEIGHT = 1056;
5
6
  const PAGE_GAP = 24;
7
+ /** Page slots sit this far inside the spacer (top and left). */
8
+ const SLOT_INSET = 12;
6
9
  export class AdaptiveViewport {
7
10
  #container;
8
11
  #host;
@@ -36,6 +39,9 @@ export class AdaptiveViewport {
36
39
  goToPage(pageIndex) {
37
40
  this.#strategy.goToPage(pageIndex);
38
41
  }
42
+ revealMatch(match) {
43
+ return this.#strategy.revealMatch(match);
44
+ }
39
45
  fitWidth() {
40
46
  return this.#strategy.fitWidth();
41
47
  }
@@ -154,6 +160,36 @@ export class ViewerViewport {
154
160
  }
155
161
  this.schedule();
156
162
  }
163
+ /**
164
+ * Scroll so the match itself is in view, not just its page: a page can be
165
+ * taller than the viewport, and a search that only lands on the page top
166
+ * leaves a match further down invisible until the reader scrolls. Resolves
167
+ * once the text runs are known; a navigation that happened meanwhile wins.
168
+ */
169
+ async revealMatch(match) {
170
+ if (this.#layout !== "continuous" || !this.#info)
171
+ return;
172
+ let runs;
173
+ try {
174
+ runs = await this.#host.getTextRuns(match.pageIndex);
175
+ }
176
+ catch {
177
+ return;
178
+ }
179
+ if (this.#destroyed || this.#host.state.pageIndex !== match.pageIndex)
180
+ return;
181
+ const top = matchTopInPage(runs, match);
182
+ if (top === undefined)
183
+ return;
184
+ const zoom = this.#host.state.zoom;
185
+ const metrics = pageMetrics(this.#info, zoom);
186
+ this.#root.scrollTop = revealScrollTop({
187
+ pageTop: metrics.offsets[match.pageIndex] ?? 0,
188
+ matchTop: SLOT_INSET + top * zoom,
189
+ viewportHeight: this.#root.clientHeight,
190
+ });
191
+ this.schedule();
192
+ }
157
193
  fitWidth() {
158
194
  const size = naturalPageSize(this.#info, this.#host.state.pageIndex);
159
195
  return Math.max(0.1, Math.min(8, (this.#root.clientWidth - 24) / size.width));
@@ -217,7 +253,9 @@ export class ViewerViewport {
217
253
  const slot = this.#slots.get(pageIndex) ?? this.#createSlot(pageIndex);
218
254
  const width = metrics.widths[pageIndex] ?? BASE_WIDTH * state.zoom;
219
255
  const height = metrics.heights[pageIndex] ?? BASE_HEIGHT * state.zoom;
220
- const top = this.#layout === "single" ? 12 : (metrics.offsets[pageIndex] ?? 0) + 12;
256
+ const top = this.#layout === "single"
257
+ ? SLOT_INSET
258
+ : (metrics.offsets[pageIndex] ?? 0) + SLOT_INSET;
221
259
  const contentWidth = Math.max(this.#root.clientWidth, metrics.maxWidth + 24);
222
260
  slot.root.style.top = `${top}px`;
223
261
  slot.root.style.left = `${Math.max(12, (contentWidth - width) / 2)}px`;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "web-doc",
3
- "version": "0.3.0",
3
+ "version": "0.5.0",
4
4
  "description": "web-doc — embeddable browser-only document viewer with Rust/WASM adapters (a fork of Zrimo)",
5
5
  "keywords": [
6
6
  "document-viewer",
@@ -69,6 +69,7 @@
69
69
  },
70
70
  "dependencies": {
71
71
  "@silurus/ooxml": "0.72.2",
72
+ "fuse.js": "7.5.0",
72
73
  "pdfjs-dist": "6.2.108"
73
74
  }
74
75
  }