web-doc 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_NOTICES.md +1 -0
- package/dist/adapters/docx-images.d.ts +26 -0
- package/dist/adapters/docx-images.js +74 -0
- package/dist/adapters/office.d.ts +5 -0
- package/dist/adapters/office.js +25 -4
- package/dist/contracts.d.ts +20 -2
- package/dist/fuzzy-search-worker.d.ts +1 -0
- package/dist/fuzzy-search-worker.js +6 -0
- package/dist/fuzzy-search.d.ts +26 -3
- package/dist/fuzzy-search.js +78 -34
- package/dist/fuzzy-worker-client.d.ts +28 -0
- package/dist/fuzzy-worker-client.js +75 -0
- package/dist/fuzzy-worker-protocol.d.ts +39 -0
- package/dist/fuzzy-worker-protocol.js +35 -0
- package/dist/index.d.ts +4 -0
- package/dist/index.js +4 -0
- package/dist/search-reveal.d.ts +21 -0
- package/dist/search-reveal.js +29 -0
- package/dist/spreadsheet-viewport.d.ts +2 -0
- package/dist/spreadsheet-viewport.js +2 -0
- package/dist/viewer.js +96 -20
- package/dist/viewport.d.ts +9 -0
- package/dist/viewport.js +39 -1
- package/dist/workers/fuzzy-search-worker.js +1680 -0
- package/package.json +1 -1
package/THIRD_PARTY_NOTICES.md
CHANGED
|
@@ -4,6 +4,7 @@ The release artifact contains or depends on the following principal components.
|
|
|
4
4
|
|
|
5
5
|
- [`@silurus/ooxml`](https://github.com/yukiyokotani/office-open-xml-viewer) — MIT; modern Office parsing/rendering.
|
|
6
6
|
- [`office_oxide`](https://github.com/yfedoseev/office_oxide) — MIT OR Apache-2.0; compound-file handling, Office IR/writer utilities and legacy XLS/PPT conversion.
|
|
7
|
+
- [`Fuse.js`](https://github.com/krisk/Fuse) — Apache-2.0; fuzzy matching behind the opt-in `search()` fallback.
|
|
7
8
|
- [`pdfjs-dist` / Mozilla PDF.js](https://github.com/mozilla/pdf.js) — Apache-2.0; browser PDF parsing,
|
|
8
9
|
font/CMap handling, canvas rendering, and text extraction. The packaged
|
|
9
10
|
standard-font, ICC, CMap, OpenJPEG, JBIG2, and QCMS assets retain the license
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Word draws an inline picture at its declared extent even when that is wider
|
|
3
|
+
* than the text area, so a generated document that embeds a 21-inch chart on
|
|
4
|
+
* a 6.5-inch column shows a clipped picture. A viewer has no margin to spill
|
|
5
|
+
* into, so before the page layout runs the oversized inline pictures are
|
|
6
|
+
* scaled down to fit the section's content box, aspect ratio preserved.
|
|
7
|
+
* Anchored (floating) pictures keep their geometry: their position is part of
|
|
8
|
+
* the author's layout.
|
|
9
|
+
*/
|
|
10
|
+
export interface DocxSectionGeometry {
|
|
11
|
+
readonly pageWidth: number;
|
|
12
|
+
readonly pageHeight: number;
|
|
13
|
+
readonly marginLeft: number;
|
|
14
|
+
readonly marginRight: number;
|
|
15
|
+
readonly marginTop: number;
|
|
16
|
+
readonly marginBottom: number;
|
|
17
|
+
}
|
|
18
|
+
export interface DocxModelLike {
|
|
19
|
+
readonly section: DocxSectionGeometry;
|
|
20
|
+
readonly body: readonly unknown[];
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* Shrink every inline picture that would not fit its section's content box.
|
|
24
|
+
* Mutates the model in place and returns how many pictures were scaled.
|
|
25
|
+
*/
|
|
26
|
+
export declare function fitInlineImagesToPage(model: DocxModelLike): number;
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shrink every inline picture that would not fit its section's content box.
|
|
3
|
+
* Mutates the model in place and returns how many pictures were scaled.
|
|
4
|
+
*/
|
|
5
|
+
export function fitInlineImagesToPage(model) {
|
|
6
|
+
let scaled = 0;
|
|
7
|
+
// A `<w:sectPr>` closes the section that ENDS at it, so the geometry for a
|
|
8
|
+
// run of body elements is only known once the break after them is reached.
|
|
9
|
+
let pending = [];
|
|
10
|
+
const flush = (geometry) => {
|
|
11
|
+
for (const element of pending)
|
|
12
|
+
scaled += fitElement(element, geometry);
|
|
13
|
+
pending = [];
|
|
14
|
+
};
|
|
15
|
+
for (const element of model.body) {
|
|
16
|
+
if (isSectionBreak(element)) {
|
|
17
|
+
flush(element.geom ?? model.section);
|
|
18
|
+
continue;
|
|
19
|
+
}
|
|
20
|
+
pending.push(element);
|
|
21
|
+
}
|
|
22
|
+
flush(model.section);
|
|
23
|
+
return scaled;
|
|
24
|
+
}
|
|
25
|
+
function fitElement(element, geometry) {
|
|
26
|
+
if (isParagraph(element)) {
|
|
27
|
+
let scaled = 0;
|
|
28
|
+
for (const run of element.runs ?? [])
|
|
29
|
+
if (isInlineImage(run) && fitImage(run, geometry))
|
|
30
|
+
scaled += 1;
|
|
31
|
+
return scaled;
|
|
32
|
+
}
|
|
33
|
+
if (isTable(element)) {
|
|
34
|
+
let scaled = 0;
|
|
35
|
+
for (const row of element.rows ?? [])
|
|
36
|
+
for (const cell of row.cells ?? [])
|
|
37
|
+
for (const child of cell.content ?? [])
|
|
38
|
+
scaled += fitElement(child, geometry);
|
|
39
|
+
return scaled;
|
|
40
|
+
}
|
|
41
|
+
return 0;
|
|
42
|
+
}
|
|
43
|
+
function fitImage(image, geometry) {
|
|
44
|
+
const contentWidth = geometry.pageWidth - geometry.marginLeft - geometry.marginRight;
|
|
45
|
+
const contentHeight = geometry.pageHeight - geometry.marginTop - geometry.marginBottom;
|
|
46
|
+
if (!(contentWidth > 0 && contentHeight > 0) ||
|
|
47
|
+
!(image.widthPt > 0 && image.heightPt > 0))
|
|
48
|
+
return false;
|
|
49
|
+
const scale = Math.min(1, contentWidth / image.widthPt, contentHeight / image.heightPt);
|
|
50
|
+
if (scale >= 1)
|
|
51
|
+
return false;
|
|
52
|
+
image.widthPt *= scale;
|
|
53
|
+
image.heightPt *= scale;
|
|
54
|
+
return true;
|
|
55
|
+
}
|
|
56
|
+
function isRecord(value) {
|
|
57
|
+
return typeof value === "object" && value !== null;
|
|
58
|
+
}
|
|
59
|
+
function isParagraph(value) {
|
|
60
|
+
return isRecord(value) && value.type === "paragraph";
|
|
61
|
+
}
|
|
62
|
+
function isTable(value) {
|
|
63
|
+
return isRecord(value) && value.type === "table";
|
|
64
|
+
}
|
|
65
|
+
function isSectionBreak(value) {
|
|
66
|
+
return isRecord(value) && value.type === "sectionBreak";
|
|
67
|
+
}
|
|
68
|
+
function isInlineImage(value) {
|
|
69
|
+
return (isRecord(value) &&
|
|
70
|
+
value.type === "image" &&
|
|
71
|
+
value.anchor !== true &&
|
|
72
|
+
typeof value.widthPt === "number" &&
|
|
73
|
+
typeof value.heightPt === "number");
|
|
74
|
+
}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { AdapterOpenContext, DocumentAdapter, DocumentFormat, DocumentInfo, HyperlinkTarget, RenderViewport, SpreadsheetSheetInfo, TextRun, ViewerWarning } from "../contracts.js";
|
|
2
|
+
import { type DocxModelLike } from "./docx-images.js";
|
|
2
3
|
declare const LEGACY_FORMATS: readonly ["doc", "xls", "ppt"];
|
|
3
4
|
type LegacyFormat = (typeof LEGACY_FORMATS)[number];
|
|
4
5
|
interface EngineLoadOptions {
|
|
@@ -27,6 +28,10 @@ interface DocxRun {
|
|
|
27
28
|
}
|
|
28
29
|
interface DocxBackend {
|
|
29
30
|
readonly pageCount: number;
|
|
31
|
+
/** Render mode; the parsed model is only reachable in `main` mode. */
|
|
32
|
+
readonly mode?: "main" | "worker";
|
|
33
|
+
/** Parsed document model (main mode). Read lazily by the page layout. */
|
|
34
|
+
readonly document?: DocxModelLike;
|
|
30
35
|
pageSize(pageIndex: number): {
|
|
31
36
|
widthPt: number;
|
|
32
37
|
heightPt: number;
|
package/dist/adapters/office.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { abortError, ViewerError } from "../errors.js";
|
|
2
|
+
import { fitInlineImagesToPage } from "./docx-images.js";
|
|
2
3
|
import { enforceContainerLimits } from "../limits.js";
|
|
3
4
|
const MODERN_FORMATS = [
|
|
4
5
|
"docx",
|
|
@@ -325,10 +326,11 @@ export class OfficeDocumentAdapter {
|
|
|
325
326
|
return convertInWorker(data, format, workerUrl, moduleUrl, context.signal, context.limits.maxOperationMs);
|
|
326
327
|
}
|
|
327
328
|
async #loadDocx(data, options) {
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
329
|
+
const backend = this.#options.engines?.docx
|
|
330
|
+
? await this.#options.engines.docx(data, options)
|
|
331
|
+
: await (await import("@silurus/ooxml/docx")).DocxDocument.load(data, options);
|
|
332
|
+
fitDocxInlineImages(backend);
|
|
333
|
+
return backend;
|
|
332
334
|
}
|
|
333
335
|
async #loadXlsx(data, options) {
|
|
334
336
|
if (this.#options.engines?.xlsx)
|
|
@@ -670,3 +672,22 @@ function normalizeOfficeError(error) {
|
|
|
670
672
|
function textDirection(text) {
|
|
671
673
|
return /[\u0590-\u08ff\ufb1d-\ufefc]/u.test(text) ? "rtl" : "ltr";
|
|
672
674
|
}
|
|
675
|
+
/**
|
|
676
|
+
* Oversized inline pictures are shrunk to the section's content box before
|
|
677
|
+
* the engine paginates (the layout is built lazily on first page access).
|
|
678
|
+
* The model is reachable in `main` mode only; a worker-mode engine keeps
|
|
679
|
+
* Word's geometry.
|
|
680
|
+
*/
|
|
681
|
+
function fitDocxInlineImages(backend) {
|
|
682
|
+
if (backend.mode === "worker")
|
|
683
|
+
return;
|
|
684
|
+
let model;
|
|
685
|
+
try {
|
|
686
|
+
model = backend.document;
|
|
687
|
+
}
|
|
688
|
+
catch {
|
|
689
|
+
return;
|
|
690
|
+
}
|
|
691
|
+
if (model)
|
|
692
|
+
fitInlineImagesToPage(model);
|
|
693
|
+
}
|
package/dist/contracts.d.ts
CHANGED
|
@@ -236,16 +236,34 @@ export interface FuzzySearchOptions {
|
|
|
236
236
|
* partly survives on a page, such as a citation that spans a page break.
|
|
237
237
|
*/
|
|
238
238
|
readonly maxScore?: number;
|
|
239
|
-
/**
|
|
239
|
+
/**
|
|
240
|
+
* Query characters considered. The matcher's cost grows with the query and
|
|
241
|
+
* a passage is identified well before its end, so the default `600` keeps
|
|
242
|
+
* a page under about 100 ms; the highlight covers the matched prefix.
|
|
243
|
+
*/
|
|
240
244
|
readonly maxQueryLength?: number;
|
|
241
245
|
/** Characters of each page's text considered. Default `20000`. */
|
|
242
246
|
readonly maxPageTextLength?: number;
|
|
243
247
|
/**
|
|
244
248
|
* Pages compared per batch. The scan proceeds nearest to `nearPage` first
|
|
245
249
|
* and stops after the first batch with a match, yielding to the event loop
|
|
246
|
-
* between batches. Default `
|
|
250
|
+
* between batches. Default `2`.
|
|
247
251
|
*/
|
|
248
252
|
readonly pagesPerBatch?: number;
|
|
253
|
+
/**
|
|
254
|
+
* With `nearPage`, how many pages nearest to the hint the fallback scans
|
|
255
|
+
* before giving up; the passage a hint points at sits within a few pages
|
|
256
|
+
* of it, and the rest of a long document is not worth the cost. Without a
|
|
257
|
+
* hint every page in range is scanned. Default `12`.
|
|
258
|
+
*/
|
|
259
|
+
readonly pageWindow?: number;
|
|
260
|
+
/**
|
|
261
|
+
* Run the matcher in a Web Worker that keeps the document's index, so a
|
|
262
|
+
* long scan never blocks the page and repeated citation lookups reuse the
|
|
263
|
+
* index. Falls back to the main thread where workers are unavailable or
|
|
264
|
+
* the worker script cannot be loaded. Default `true`.
|
|
265
|
+
*/
|
|
266
|
+
readonly worker?: boolean;
|
|
249
267
|
}
|
|
250
268
|
export interface SearchOptions {
|
|
251
269
|
readonly caseSensitive?: boolean;
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/fuzzy-search.d.ts
CHANGED
|
@@ -13,6 +13,8 @@ export interface ResolvedFuzzySearchOptions {
|
|
|
13
13
|
readonly maxQueryLength: number;
|
|
14
14
|
readonly maxPageTextLength: number;
|
|
15
15
|
readonly pagesPerBatch: number;
|
|
16
|
+
readonly pageWindow: number;
|
|
17
|
+
readonly worker: boolean;
|
|
16
18
|
}
|
|
17
19
|
export declare const DEFAULT_FUZZY_SEARCH_OPTIONS: ResolvedFuzzySearchOptions;
|
|
18
20
|
/**
|
|
@@ -27,9 +29,30 @@ export interface FuzzyPageText {
|
|
|
27
29
|
readonly text: string;
|
|
28
30
|
}
|
|
29
31
|
/**
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
32
|
+
* A document's pages, indexed once for Fuse.js so that every fuzzy search
|
|
33
|
+
* reuses the normalized records instead of rebuilding them: a citation
|
|
34
|
+
* lookup tries several anchors in a row, and each used to pay for the index
|
|
35
|
+
* again. A scan restricted to a page window reuses the same records through
|
|
36
|
+
* `Fuse.parseIndex`, so only the Bitap pass runs for those pages.
|
|
37
|
+
*/
|
|
38
|
+
export declare class FuzzyPageIndex {
|
|
39
|
+
#private;
|
|
40
|
+
constructor(pages: readonly FuzzyPageText[], options: {
|
|
41
|
+
threshold: number;
|
|
42
|
+
maxPageTextLength: number;
|
|
43
|
+
}, caseSensitive?: boolean);
|
|
44
|
+
get pageCount(): number;
|
|
45
|
+
/**
|
|
46
|
+
* One match per page whose text holds the query within the edit budget:
|
|
47
|
+
* the span from the first to the last matched character, so the highlight
|
|
48
|
+
* covers the passage as one block. `pageIndices` restricts the scan; the
|
|
49
|
+
* result is in page order either way.
|
|
50
|
+
*/
|
|
51
|
+
search(query: string, maxScore: number, pageIndices?: readonly number[]): readonly SearchMatch[];
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* One-off scan of a few pages, for callers without a standing index. The
|
|
55
|
+
* viewer keeps a {@link FuzzyPageIndex} per document instead.
|
|
33
56
|
*/
|
|
34
57
|
export declare function findFuzzyPageMatches(pages: readonly FuzzyPageText[], query: string, options: ResolvedFuzzySearchOptions, caseSensitive?: boolean): readonly SearchMatch[];
|
|
35
58
|
/** Page indices of `[first, last]` ordered by distance from `nearPage`, ties earlier-first. */
|
package/dist/fuzzy-search.js
CHANGED
|
@@ -1,10 +1,14 @@
|
|
|
1
|
-
import Fuse from "fuse.js";
|
|
1
|
+
import Fuse, {} from "fuse.js";
|
|
2
2
|
export const DEFAULT_FUZZY_SEARCH_OPTIONS = Object.freeze({
|
|
3
3
|
threshold: 0.3,
|
|
4
4
|
maxScore: 0.4,
|
|
5
|
-
|
|
5
|
+
// Bitap cost grows with the query; 600 characters still identifies a
|
|
6
|
+
// passage while keeping a page under ~100 ms on a laptop.
|
|
7
|
+
maxQueryLength: 600,
|
|
6
8
|
maxPageTextLength: 20_000,
|
|
7
|
-
pagesPerBatch:
|
|
9
|
+
pagesPerBatch: 2,
|
|
10
|
+
pageWindow: 12,
|
|
11
|
+
worker: true,
|
|
8
12
|
});
|
|
9
13
|
/**
|
|
10
14
|
* Layer fuzzy settings in precedence order (viewer defaults first, then the
|
|
@@ -29,24 +33,15 @@ export function resolveFuzzySearchOptions(...layers) {
|
|
|
29
33
|
maxQueryLength: positiveInteger(layer.maxQueryLength, resolved.maxQueryLength),
|
|
30
34
|
maxPageTextLength: positiveInteger(layer.maxPageTextLength, resolved.maxPageTextLength),
|
|
31
35
|
pagesPerBatch: positiveInteger(layer.pagesPerBatch, resolved.pagesPerBatch),
|
|
36
|
+
pageWindow: positiveInteger(layer.pageWindow, resolved.pageWindow),
|
|
37
|
+
worker: layer.worker ?? resolved.worker,
|
|
32
38
|
};
|
|
33
39
|
}
|
|
34
40
|
return enabled ? resolved : undefined;
|
|
35
41
|
}
|
|
36
|
-
/**
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
* the passage as one block. Pages are returned in page order.
|
|
40
|
-
*/
|
|
41
|
-
export function findFuzzyPageMatches(pages, query, options, caseSensitive = false) {
|
|
42
|
-
const pattern = query.slice(0, options.maxQueryLength);
|
|
43
|
-
if (!pattern.trim())
|
|
44
|
-
return [];
|
|
45
|
-
const items = pages.map((page) => ({
|
|
46
|
-
pageIndex: page.pageIndex,
|
|
47
|
-
text: page.text.slice(0, options.maxPageTextLength),
|
|
48
|
-
}));
|
|
49
|
-
const fuse = new Fuse(items, {
|
|
42
|
+
/** Fuse.js options shared by every fuzzy scan; `keys` and `threshold` are fixed per index. */
|
|
43
|
+
function fuseOptions(threshold, caseSensitive) {
|
|
44
|
+
return {
|
|
50
45
|
keys: ["text"],
|
|
51
46
|
isCaseSensitive: caseSensitive,
|
|
52
47
|
ignoreDiacritics: false,
|
|
@@ -57,27 +52,76 @@ export function findFuzzyPageMatches(pages, query, options, caseSensitive = fals
|
|
|
57
52
|
ignoreLocation: true,
|
|
58
53
|
// Long page text must not dilute the score of a match inside it.
|
|
59
54
|
ignoreFieldNorm: true,
|
|
60
|
-
threshold
|
|
55
|
+
threshold,
|
|
61
56
|
minMatchCharLength: 3,
|
|
62
57
|
shouldSort: false,
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* A document's pages, indexed once for Fuse.js so that every fuzzy search
|
|
62
|
+
* reuses the normalized records instead of rebuilding them: a citation
|
|
63
|
+
* lookup tries several anchors in a row, and each used to pay for the index
|
|
64
|
+
* again. A scan restricted to a page window reuses the same records through
|
|
65
|
+
* `Fuse.parseIndex`, so only the Bitap pass runs for those pages.
|
|
66
|
+
*/
|
|
67
|
+
export class FuzzyPageIndex {
|
|
68
|
+
#pages;
|
|
69
|
+
#index;
|
|
70
|
+
#options;
|
|
71
|
+
#fuse;
|
|
72
|
+
constructor(pages, options, caseSensitive = false) {
|
|
73
|
+
this.#pages = pages.map((page) => ({
|
|
74
|
+
pageIndex: page.pageIndex,
|
|
75
|
+
text: page.text.slice(0, options.maxPageTextLength),
|
|
76
|
+
}));
|
|
77
|
+
this.#options = fuseOptions(options.threshold, caseSensitive);
|
|
78
|
+
this.#index = Fuse.createIndex(["text"], this.#pages);
|
|
79
|
+
this.#fuse = new Fuse(this.#pages, this.#options, this.#index);
|
|
80
|
+
}
|
|
81
|
+
get pageCount() {
|
|
82
|
+
return this.#pages.length;
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* One match per page whose text holds the query within the edit budget:
|
|
86
|
+
* the span from the first to the last matched character, so the highlight
|
|
87
|
+
* covers the passage as one block. `pageIndices` restricts the scan; the
|
|
88
|
+
* result is in page order either way.
|
|
89
|
+
*/
|
|
90
|
+
search(query, maxScore, pageIndices) {
|
|
91
|
+
if (!query.trim())
|
|
92
|
+
return [];
|
|
93
|
+
const fuse = pageIndices ? this.#subset(pageIndices) : this.#fuse;
|
|
94
|
+
const matches = [];
|
|
95
|
+
for (const result of fuse.search(query)) {
|
|
96
|
+
if ((result.score ?? 1) > maxScore)
|
|
97
|
+
continue;
|
|
98
|
+
const indices = result.matches?.[0]?.indices ?? [];
|
|
99
|
+
if (indices.length === 0)
|
|
100
|
+
continue;
|
|
101
|
+
let start = Number.POSITIVE_INFINITY;
|
|
102
|
+
let end = 0;
|
|
103
|
+
for (const [first, last] of indices) {
|
|
104
|
+
start = Math.min(start, first);
|
|
105
|
+
end = Math.max(end, last + 1);
|
|
106
|
+
}
|
|
107
|
+
const text = result.item.text.slice(start, end);
|
|
108
|
+
matches.push({ pageIndex: result.item.pageIndex, start, end, text });
|
|
76
109
|
}
|
|
77
|
-
|
|
78
|
-
matches.push({ pageIndex: result.item.pageIndex, start, end, text });
|
|
110
|
+
return matches.sort((a, b) => a.pageIndex - b.pageIndex);
|
|
79
111
|
}
|
|
80
|
-
|
|
112
|
+
#subset(pageIndices) {
|
|
113
|
+
const wanted = new Set(pageIndices);
|
|
114
|
+
const { keys, records } = this.#index.toJSON();
|
|
115
|
+
const kept = records.filter((record) => wanted.has(this.#pages[record.i]?.pageIndex ?? -1));
|
|
116
|
+
return new Fuse(this.#pages, this.#options, Fuse.parseIndex({ keys, records: kept }));
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* One-off scan of a few pages, for callers without a standing index. The
|
|
121
|
+
* viewer keeps a {@link FuzzyPageIndex} per document instead.
|
|
122
|
+
*/
|
|
123
|
+
export function findFuzzyPageMatches(pages, query, options, caseSensitive = false) {
|
|
124
|
+
return new FuzzyPageIndex(pages, options, caseSensitive).search(query.slice(0, options.maxQueryLength), options.maxScore);
|
|
81
125
|
}
|
|
82
126
|
/** Page indices of `[first, last]` ordered by distance from `nearPage`, ties earlier-first. */
|
|
83
127
|
export function pagesNearestFirst(first, last, nearPage) {
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import type { SearchMatch } from "./contracts.js";
|
|
2
|
+
import type { FuzzyPageText } from "./fuzzy-search.js";
|
|
3
|
+
import type { FuzzyWorkerReply, FuzzyWorkerRequest } from "./fuzzy-worker-protocol.js";
|
|
4
|
+
/** The slice of `Worker` the client needs; tests pass a fake. */
|
|
5
|
+
export interface FuzzyWorkerTransport {
|
|
6
|
+
postMessage(message: FuzzyWorkerRequest): void;
|
|
7
|
+
addEventListener(type: "message", listener: (event: MessageEvent<FuzzyWorkerReply>) => void): void;
|
|
8
|
+
addEventListener(type: "error", listener: (event: Event) => void): void;
|
|
9
|
+
terminate(): void;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* Main-thread side of the fuzzy-search worker. Bitap over a long document
|
|
13
|
+
* is CPU work that used to run on the main thread and freeze the page; the
|
|
14
|
+
* worker keeps the document's index and answers each search off-thread.
|
|
15
|
+
*/
|
|
16
|
+
export declare class FuzzyWorkerClient {
|
|
17
|
+
#private;
|
|
18
|
+
constructor(worker: FuzzyWorkerTransport);
|
|
19
|
+
/** Spawn the bundled worker script; `undefined` where workers do not exist. */
|
|
20
|
+
static create(workerUrl: URL): FuzzyWorkerClient | undefined;
|
|
21
|
+
index(pages: readonly FuzzyPageText[], options: {
|
|
22
|
+
readonly threshold: number;
|
|
23
|
+
readonly maxPageTextLength: number;
|
|
24
|
+
readonly caseSensitive: boolean;
|
|
25
|
+
}): Promise<number>;
|
|
26
|
+
search(query: string, maxScore: number, pageIndices?: readonly number[]): Promise<readonly SearchMatch[]>;
|
|
27
|
+
terminate(): void;
|
|
28
|
+
}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import { ViewerError } from "./errors.js";
|
|
2
|
+
/**
|
|
3
|
+
* Main-thread side of the fuzzy-search worker. Bitap over a long document
|
|
4
|
+
* is CPU work that used to run on the main thread and freeze the page; the
|
|
5
|
+
* worker keeps the document's index and answers each search off-thread.
|
|
6
|
+
*/
|
|
7
|
+
export class FuzzyWorkerClient {
|
|
8
|
+
#worker;
|
|
9
|
+
#pending = new Map();
|
|
10
|
+
#nextId = 1;
|
|
11
|
+
#dead;
|
|
12
|
+
constructor(worker) {
|
|
13
|
+
this.#worker = worker;
|
|
14
|
+
worker.addEventListener("message", (event) => {
|
|
15
|
+
const pending = this.#pending.get(event.data.id);
|
|
16
|
+
if (!pending)
|
|
17
|
+
return;
|
|
18
|
+
this.#pending.delete(event.data.id);
|
|
19
|
+
pending.resolve(event.data);
|
|
20
|
+
});
|
|
21
|
+
worker.addEventListener("error", (event) => {
|
|
22
|
+
this.#fail(new ViewerError("worker-crashed", "Fuzzy search worker crashed", {
|
|
23
|
+
details: { message: event.message },
|
|
24
|
+
}));
|
|
25
|
+
});
|
|
26
|
+
}
|
|
27
|
+
/** Spawn the bundled worker script; `undefined` where workers do not exist. */
|
|
28
|
+
static create(workerUrl) {
|
|
29
|
+
if (typeof Worker === "undefined")
|
|
30
|
+
return undefined;
|
|
31
|
+
return new FuzzyWorkerClient(new Worker(workerUrl, { type: "module", name: "fuzzy-search" }));
|
|
32
|
+
}
|
|
33
|
+
async index(pages, options) {
|
|
34
|
+
const reply = await this.#send({ kind: "index", id: 0, pages, ...options });
|
|
35
|
+
if (reply.kind !== "indexed")
|
|
36
|
+
throw unexpected(reply);
|
|
37
|
+
return reply.pageCount;
|
|
38
|
+
}
|
|
39
|
+
async search(query, maxScore, pageIndices) {
|
|
40
|
+
const reply = await this.#send({
|
|
41
|
+
kind: "search",
|
|
42
|
+
id: 0,
|
|
43
|
+
query,
|
|
44
|
+
maxScore,
|
|
45
|
+
...(pageIndices ? { pageIndices } : {}),
|
|
46
|
+
});
|
|
47
|
+
if (reply.kind !== "matches")
|
|
48
|
+
throw unexpected(reply);
|
|
49
|
+
return reply.matches;
|
|
50
|
+
}
|
|
51
|
+
terminate() {
|
|
52
|
+
this.#fail(new ViewerError("lifecycle-error", "Fuzzy search worker was terminated"));
|
|
53
|
+
this.#worker.terminate();
|
|
54
|
+
}
|
|
55
|
+
#send(request) {
|
|
56
|
+
if (this.#dead)
|
|
57
|
+
return Promise.reject(this.#dead);
|
|
58
|
+
const id = this.#nextId++;
|
|
59
|
+
return new Promise((resolve, reject) => {
|
|
60
|
+
this.#pending.set(id, { resolve, reject });
|
|
61
|
+
this.#worker.postMessage({ ...request, id });
|
|
62
|
+
});
|
|
63
|
+
}
|
|
64
|
+
#fail(error) {
|
|
65
|
+
this.#dead = error;
|
|
66
|
+
for (const pending of this.#pending.values())
|
|
67
|
+
pending.reject(error);
|
|
68
|
+
this.#pending.clear();
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
function unexpected(reply) {
|
|
72
|
+
return new ViewerError("render-failed", reply.kind === "failure"
|
|
73
|
+
? reply.message
|
|
74
|
+
: `Unexpected worker reply: ${reply.kind}`);
|
|
75
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import type { SearchMatch } from "./contracts.js";
|
|
2
|
+
import { FuzzyPageIndex, type FuzzyPageText } from "./fuzzy-search.js";
|
|
3
|
+
/**
|
|
4
|
+
* Messages between the viewer and the fuzzy-search worker. The worker holds
|
|
5
|
+
* one {@link FuzzyPageIndex} at a time: `index` replaces it, `search` runs
|
|
6
|
+
* against it. Every request is answered by exactly one reply with its `id`.
|
|
7
|
+
*/
|
|
8
|
+
export type FuzzyWorkerRequest = {
|
|
9
|
+
readonly kind: "index";
|
|
10
|
+
readonly id: number;
|
|
11
|
+
readonly pages: readonly FuzzyPageText[];
|
|
12
|
+
readonly threshold: number;
|
|
13
|
+
readonly maxPageTextLength: number;
|
|
14
|
+
readonly caseSensitive: boolean;
|
|
15
|
+
} | {
|
|
16
|
+
readonly kind: "search";
|
|
17
|
+
readonly id: number;
|
|
18
|
+
readonly query: string;
|
|
19
|
+
readonly maxScore: number;
|
|
20
|
+
readonly pageIndices?: readonly number[];
|
|
21
|
+
};
|
|
22
|
+
export type FuzzyWorkerReply = {
|
|
23
|
+
readonly kind: "indexed";
|
|
24
|
+
readonly id: number;
|
|
25
|
+
readonly pageCount: number;
|
|
26
|
+
} | {
|
|
27
|
+
readonly kind: "matches";
|
|
28
|
+
readonly id: number;
|
|
29
|
+
readonly matches: readonly SearchMatch[];
|
|
30
|
+
} | {
|
|
31
|
+
readonly kind: "failure";
|
|
32
|
+
readonly id: number;
|
|
33
|
+
readonly message: string;
|
|
34
|
+
};
|
|
35
|
+
export interface FuzzyWorkerState {
|
|
36
|
+
index?: FuzzyPageIndex;
|
|
37
|
+
}
|
|
38
|
+
/** Apply one request to the worker's state and produce its reply. */
|
|
39
|
+
export declare function handleFuzzyWorkerRequest(state: FuzzyWorkerState, request: FuzzyWorkerRequest): FuzzyWorkerReply;
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { FuzzyPageIndex } from "./fuzzy-search.js";
|
|
2
|
+
/** Apply one request to the worker's state and produce its reply. */
|
|
3
|
+
export function handleFuzzyWorkerRequest(state, request) {
|
|
4
|
+
try {
|
|
5
|
+
if (request.kind === "index") {
|
|
6
|
+
state.index = new FuzzyPageIndex(request.pages, {
|
|
7
|
+
threshold: request.threshold,
|
|
8
|
+
maxPageTextLength: request.maxPageTextLength,
|
|
9
|
+
}, request.caseSensitive);
|
|
10
|
+
return {
|
|
11
|
+
kind: "indexed",
|
|
12
|
+
id: request.id,
|
|
13
|
+
pageCount: state.index.pageCount,
|
|
14
|
+
};
|
|
15
|
+
}
|
|
16
|
+
if (!state.index)
|
|
17
|
+
return {
|
|
18
|
+
kind: "failure",
|
|
19
|
+
id: request.id,
|
|
20
|
+
message: "No document is indexed",
|
|
21
|
+
};
|
|
22
|
+
return {
|
|
23
|
+
kind: "matches",
|
|
24
|
+
id: request.id,
|
|
25
|
+
matches: state.index.search(request.query, request.maxScore, request.pageIndices),
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
catch (error) {
|
|
29
|
+
return {
|
|
30
|
+
kind: "failure",
|
|
31
|
+
id: request.id,
|
|
32
|
+
message: error instanceof Error ? error.message : String(error),
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -6,6 +6,10 @@ export * from "./format.js";
|
|
|
6
6
|
export * from "./limits.js";
|
|
7
7
|
export * from "./interaction.js";
|
|
8
8
|
export * from "./fuzzy-search.js";
|
|
9
|
+
export * from "./fuzzy-worker-protocol.js";
|
|
10
|
+
export * from "./fuzzy-worker-client.js";
|
|
11
|
+
export * from "./search-reveal.js";
|
|
12
|
+
export * from "./adapters/docx-images.js";
|
|
9
13
|
export * from "./render-scheduler.js";
|
|
10
14
|
export * from "./i18n.js";
|
|
11
15
|
export * from "./font-manifest.js";
|
package/dist/index.js
CHANGED
|
@@ -6,6 +6,10 @@ export * from "./format.js";
|
|
|
6
6
|
export * from "./limits.js";
|
|
7
7
|
export * from "./interaction.js";
|
|
8
8
|
export * from "./fuzzy-search.js";
|
|
9
|
+
export * from "./fuzzy-worker-protocol.js";
|
|
10
|
+
export * from "./fuzzy-worker-client.js";
|
|
11
|
+
export * from "./search-reveal.js";
|
|
12
|
+
export * from "./adapters/docx-images.js";
|
|
9
13
|
export * from "./render-scheduler.js";
|
|
10
14
|
export * from "./i18n.js";
|
|
11
15
|
export * from "./font-manifest.js";
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import type { SearchMatch, TextRun } from "./contracts.js";
|
|
2
|
+
/**
|
|
3
|
+
* Vertical position of a match inside its page, in the coordinate units of
|
|
4
|
+
* the page's text layer: the `y` of the first run the match overlaps.
|
|
5
|
+
* Offsets follow the same convention as the highlight layers (a run's
|
|
6
|
+
* explicit `logicalStart`/`logicalEnd`, else the running text length), so
|
|
7
|
+
* the position agrees with where the highlight is painted. `undefined` when
|
|
8
|
+
* no run overlaps the match.
|
|
9
|
+
*/
|
|
10
|
+
export declare function matchTopInPage(runs: readonly TextRun[], match: SearchMatch): number | undefined;
|
|
11
|
+
/**
|
|
12
|
+
* Scroll offset that places a match about a third of the way down the
|
|
13
|
+
* viewport — far enough from the top edge to read the line before it, never
|
|
14
|
+
* above the page's own top so a match near the top of a page still shows the
|
|
15
|
+
* page head. `matchTop` is the match's offset from the page top in CSS px.
|
|
16
|
+
*/
|
|
17
|
+
export declare function revealScrollTop(options: {
|
|
18
|
+
readonly pageTop: number;
|
|
19
|
+
readonly matchTop: number;
|
|
20
|
+
readonly viewportHeight: number;
|
|
21
|
+
}): number;
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Vertical position of a match inside its page, in the coordinate units of
|
|
3
|
+
* the page's text layer: the `y` of the first run the match overlaps.
|
|
4
|
+
* Offsets follow the same convention as the highlight layers (a run's
|
|
5
|
+
* explicit `logicalStart`/`logicalEnd`, else the running text length), so
|
|
6
|
+
* the position agrees with where the highlight is painted. `undefined` when
|
|
7
|
+
* no run overlaps the match.
|
|
8
|
+
*/
|
|
9
|
+
export function matchTopInPage(runs, match) {
|
|
10
|
+
let offset = 0;
|
|
11
|
+
for (const run of runs) {
|
|
12
|
+
const start = run.logicalStart ?? offset;
|
|
13
|
+
const end = run.logicalEnd ?? start + run.text.length;
|
|
14
|
+
offset = end;
|
|
15
|
+
if (match.start < end && match.end > start)
|
|
16
|
+
return run.y;
|
|
17
|
+
}
|
|
18
|
+
return undefined;
|
|
19
|
+
}
|
|
20
|
+
/**
|
|
21
|
+
* Scroll offset that places a match about a third of the way down the
|
|
22
|
+
* viewport — far enough from the top edge to read the line before it, never
|
|
23
|
+
* above the page's own top so a match near the top of a page still shows the
|
|
24
|
+
* page head. `matchTop` is the match's offset from the page top in CSS px.
|
|
25
|
+
*/
|
|
26
|
+
export function revealScrollTop(options) {
|
|
27
|
+
const lead = Math.max(0, options.viewportHeight) / 3;
|
|
28
|
+
return Math.max(options.pageTop, options.pageTop + options.matchTop - lead);
|
|
29
|
+
}
|