web-doc 0.5.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_NOTICES.md +1 -0
- package/dist/contracts.d.ts +7 -0
- package/dist/fuzzy-search-worker.d.ts +1 -0
- package/dist/fuzzy-search-worker.js +6 -0
- package/dist/fuzzy-search.d.ts +25 -3
- package/dist/fuzzy-search.js +72 -32
- package/dist/fuzzy-worker-client.d.ts +28 -0
- package/dist/fuzzy-worker-client.js +75 -0
- package/dist/fuzzy-worker-protocol.d.ts +39 -0
- package/dist/fuzzy-worker-protocol.js +35 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.js +2 -0
- package/dist/viewer.js +86 -17
- package/dist/workers/fuzzy-search-worker.js +1680 -0
- package/package.json +1 -1
package/THIRD_PARTY_NOTICES.md
CHANGED
|
@@ -4,6 +4,7 @@ The release artifact contains or depends on the following principal components.
|
|
|
4
4
|
|
|
5
5
|
- [`@silurus/ooxml`](https://github.com/yukiyokotani/office-open-xml-viewer) — MIT; modern Office parsing/rendering.
|
|
6
6
|
- [`office_oxide`](https://github.com/yfedoseev/office_oxide) — MIT OR Apache-2.0; compound-file handling, Office IR/writer utilities and legacy XLS/PPT conversion.
|
|
7
|
+
- [`Fuse.js`](https://github.com/krisk/Fuse) — Apache-2.0; fuzzy matching behind the opt-in `search()` fallback.
|
|
7
8
|
- [`pdfjs-dist` / Mozilla PDF.js](https://github.com/mozilla/pdf.js) — Apache-2.0; browser PDF parsing,
|
|
8
9
|
font/CMap handling, canvas rendering, and text extraction. The packaged
|
|
9
10
|
standard-font, ICC, CMap, OpenJPEG, JBIG2, and QCMS assets retain the license
|
package/dist/contracts.d.ts
CHANGED
|
@@ -257,6 +257,13 @@ export interface FuzzySearchOptions {
|
|
|
257
257
|
* hint every page in range is scanned. Default `12`.
|
|
258
258
|
*/
|
|
259
259
|
readonly pageWindow?: number;
|
|
260
|
+
/**
|
|
261
|
+
* Run the matcher in a Web Worker that keeps the document's index, so a
|
|
262
|
+
* long scan never blocks the page and repeated citation lookups reuse the
|
|
263
|
+
* index. Falls back to the main thread where workers are unavailable or
|
|
264
|
+
* the worker script cannot be loaded. Default `true`.
|
|
265
|
+
*/
|
|
266
|
+
readonly worker?: boolean;
|
|
260
267
|
}
|
|
261
268
|
export interface SearchOptions {
|
|
262
269
|
readonly caseSensitive?: boolean;
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/fuzzy-search.d.ts
CHANGED
|
@@ -14,6 +14,7 @@ export interface ResolvedFuzzySearchOptions {
|
|
|
14
14
|
readonly maxPageTextLength: number;
|
|
15
15
|
readonly pagesPerBatch: number;
|
|
16
16
|
readonly pageWindow: number;
|
|
17
|
+
readonly worker: boolean;
|
|
17
18
|
}
|
|
18
19
|
export declare const DEFAULT_FUZZY_SEARCH_OPTIONS: ResolvedFuzzySearchOptions;
|
|
19
20
|
/**
|
|
@@ -28,9 +29,30 @@ export interface FuzzyPageText {
|
|
|
28
29
|
readonly text: string;
|
|
29
30
|
}
|
|
30
31
|
/**
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
32
|
+
* A document's pages, indexed once for Fuse.js so that every fuzzy search
|
|
33
|
+
* reuses the normalized records instead of rebuilding them: a citation
|
|
34
|
+
* lookup tries several anchors in a row, and each used to pay for the index
|
|
35
|
+
* again. A scan restricted to a page window reuses the same records through
|
|
36
|
+
* `Fuse.parseIndex`, so only the Bitap pass runs for those pages.
|
|
37
|
+
*/
|
|
38
|
+
export declare class FuzzyPageIndex {
|
|
39
|
+
#private;
|
|
40
|
+
constructor(pages: readonly FuzzyPageText[], options: {
|
|
41
|
+
threshold: number;
|
|
42
|
+
maxPageTextLength: number;
|
|
43
|
+
}, caseSensitive?: boolean);
|
|
44
|
+
get pageCount(): number;
|
|
45
|
+
/**
|
|
46
|
+
* One match per page whose text holds the query within the edit budget:
|
|
47
|
+
* the span from the first to the last matched character, so the highlight
|
|
48
|
+
* covers the passage as one block. `pageIndices` restricts the scan; the
|
|
49
|
+
* result is in page order either way.
|
|
50
|
+
*/
|
|
51
|
+
search(query: string, maxScore: number, pageIndices?: readonly number[]): readonly SearchMatch[];
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* One-off scan of a few pages, for callers without a standing index. The
|
|
55
|
+
* viewer keeps a {@link FuzzyPageIndex} per document instead.
|
|
34
56
|
*/
|
|
35
57
|
export declare function findFuzzyPageMatches(pages: readonly FuzzyPageText[], query: string, options: ResolvedFuzzySearchOptions, caseSensitive?: boolean): readonly SearchMatch[];
|
|
36
58
|
/** Page indices of `[first, last]` ordered by distance from `nearPage`, ties earlier-first. */
|
package/dist/fuzzy-search.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import Fuse from "fuse.js";
|
|
1
|
+
import Fuse, {} from "fuse.js";
|
|
2
2
|
export const DEFAULT_FUZZY_SEARCH_OPTIONS = Object.freeze({
|
|
3
3
|
threshold: 0.3,
|
|
4
4
|
maxScore: 0.4,
|
|
@@ -8,6 +8,7 @@ export const DEFAULT_FUZZY_SEARCH_OPTIONS = Object.freeze({
|
|
|
8
8
|
maxPageTextLength: 20_000,
|
|
9
9
|
pagesPerBatch: 2,
|
|
10
10
|
pageWindow: 12,
|
|
11
|
+
worker: true,
|
|
11
12
|
});
|
|
12
13
|
/**
|
|
13
14
|
* Layer fuzzy settings in precedence order (viewer defaults first, then the
|
|
@@ -33,24 +34,14 @@ export function resolveFuzzySearchOptions(...layers) {
|
|
|
33
34
|
maxPageTextLength: positiveInteger(layer.maxPageTextLength, resolved.maxPageTextLength),
|
|
34
35
|
pagesPerBatch: positiveInteger(layer.pagesPerBatch, resolved.pagesPerBatch),
|
|
35
36
|
pageWindow: positiveInteger(layer.pageWindow, resolved.pageWindow),
|
|
37
|
+
worker: layer.worker ?? resolved.worker,
|
|
36
38
|
};
|
|
37
39
|
}
|
|
38
40
|
return enabled ? resolved : undefined;
|
|
39
41
|
}
|
|
40
|
-
/**
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
* the passage as one block. Pages are returned in page order.
|
|
44
|
-
*/
|
|
45
|
-
export function findFuzzyPageMatches(pages, query, options, caseSensitive = false) {
|
|
46
|
-
const pattern = query.slice(0, options.maxQueryLength);
|
|
47
|
-
if (!pattern.trim())
|
|
48
|
-
return [];
|
|
49
|
-
const items = pages.map((page) => ({
|
|
50
|
-
pageIndex: page.pageIndex,
|
|
51
|
-
text: page.text.slice(0, options.maxPageTextLength),
|
|
52
|
-
}));
|
|
53
|
-
const fuse = new Fuse(items, {
|
|
42
|
+
/** Fuse.js options shared by every fuzzy scan; `keys` and `threshold` are fixed per index. */
|
|
43
|
+
function fuseOptions(threshold, caseSensitive) {
|
|
44
|
+
return {
|
|
54
45
|
keys: ["text"],
|
|
55
46
|
isCaseSensitive: caseSensitive,
|
|
56
47
|
ignoreDiacritics: false,
|
|
@@ -61,27 +52,76 @@ export function findFuzzyPageMatches(pages, query, options, caseSensitive = fals
|
|
|
61
52
|
ignoreLocation: true,
|
|
62
53
|
// Long page text must not dilute the score of a match inside it.
|
|
63
54
|
ignoreFieldNorm: true,
|
|
64
|
-
threshold
|
|
55
|
+
threshold,
|
|
65
56
|
minMatchCharLength: 3,
|
|
66
57
|
shouldSort: false,
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* A document's pages, indexed once for Fuse.js so that every fuzzy search
|
|
62
|
+
* reuses the normalized records instead of rebuilding them: a citation
|
|
63
|
+
* lookup tries several anchors in a row, and each used to pay for the index
|
|
64
|
+
* again. A scan restricted to a page window reuses the same records through
|
|
65
|
+
* `Fuse.parseIndex`, so only the Bitap pass runs for those pages.
|
|
66
|
+
*/
|
|
67
|
+
export class FuzzyPageIndex {
|
|
68
|
+
#pages;
|
|
69
|
+
#index;
|
|
70
|
+
#options;
|
|
71
|
+
#fuse;
|
|
72
|
+
constructor(pages, options, caseSensitive = false) {
|
|
73
|
+
this.#pages = pages.map((page) => ({
|
|
74
|
+
pageIndex: page.pageIndex,
|
|
75
|
+
text: page.text.slice(0, options.maxPageTextLength),
|
|
76
|
+
}));
|
|
77
|
+
this.#options = fuseOptions(options.threshold, caseSensitive);
|
|
78
|
+
this.#index = Fuse.createIndex(["text"], this.#pages);
|
|
79
|
+
this.#fuse = new Fuse(this.#pages, this.#options, this.#index);
|
|
80
|
+
}
|
|
81
|
+
get pageCount() {
|
|
82
|
+
return this.#pages.length;
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* One match per page whose text holds the query within the edit budget:
|
|
86
|
+
* the span from the first to the last matched character, so the highlight
|
|
87
|
+
* covers the passage as one block. `pageIndices` restricts the scan; the
|
|
88
|
+
* result is in page order either way.
|
|
89
|
+
*/
|
|
90
|
+
search(query, maxScore, pageIndices) {
|
|
91
|
+
if (!query.trim())
|
|
92
|
+
return [];
|
|
93
|
+
const fuse = pageIndices ? this.#subset(pageIndices) : this.#fuse;
|
|
94
|
+
const matches = [];
|
|
95
|
+
for (const result of fuse.search(query)) {
|
|
96
|
+
if ((result.score ?? 1) > maxScore)
|
|
97
|
+
continue;
|
|
98
|
+
const indices = result.matches?.[0]?.indices ?? [];
|
|
99
|
+
if (indices.length === 0)
|
|
100
|
+
continue;
|
|
101
|
+
let start = Number.POSITIVE_INFINITY;
|
|
102
|
+
let end = 0;
|
|
103
|
+
for (const [first, last] of indices) {
|
|
104
|
+
start = Math.min(start, first);
|
|
105
|
+
end = Math.max(end, last + 1);
|
|
106
|
+
}
|
|
107
|
+
const text = result.item.text.slice(start, end);
|
|
108
|
+
matches.push({ pageIndex: result.item.pageIndex, start, end, text });
|
|
80
109
|
}
|
|
81
|
-
|
|
82
|
-
matches.push({ pageIndex: result.item.pageIndex, start, end, text });
|
|
110
|
+
return matches.sort((a, b) => a.pageIndex - b.pageIndex);
|
|
83
111
|
}
|
|
84
|
-
|
|
112
|
+
#subset(pageIndices) {
|
|
113
|
+
const wanted = new Set(pageIndices);
|
|
114
|
+
const { keys, records } = this.#index.toJSON();
|
|
115
|
+
const kept = records.filter((record) => wanted.has(this.#pages[record.i]?.pageIndex ?? -1));
|
|
116
|
+
return new Fuse(this.#pages, this.#options, Fuse.parseIndex({ keys, records: kept }));
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* One-off scan of a few pages, for callers without a standing index. The
|
|
121
|
+
* viewer keeps a {@link FuzzyPageIndex} per document instead.
|
|
122
|
+
*/
|
|
123
|
+
export function findFuzzyPageMatches(pages, query, options, caseSensitive = false) {
|
|
124
|
+
return new FuzzyPageIndex(pages, options, caseSensitive).search(query.slice(0, options.maxQueryLength), options.maxScore);
|
|
85
125
|
}
|
|
86
126
|
/** Page indices of `[first, last]` ordered by distance from `nearPage`, ties earlier-first. */
|
|
87
127
|
export function pagesNearestFirst(first, last, nearPage) {
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import type { SearchMatch } from "./contracts.js";
|
|
2
|
+
import type { FuzzyPageText } from "./fuzzy-search.js";
|
|
3
|
+
import type { FuzzyWorkerReply, FuzzyWorkerRequest } from "./fuzzy-worker-protocol.js";
|
|
4
|
+
/** The slice of `Worker` the client needs; tests pass a fake. */
|
|
5
|
+
export interface FuzzyWorkerTransport {
|
|
6
|
+
postMessage(message: FuzzyWorkerRequest): void;
|
|
7
|
+
addEventListener(type: "message", listener: (event: MessageEvent<FuzzyWorkerReply>) => void): void;
|
|
8
|
+
addEventListener(type: "error", listener: (event: Event) => void): void;
|
|
9
|
+
terminate(): void;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* Main-thread side of the fuzzy-search worker. Bitap over a long document
|
|
13
|
+
* is CPU work that used to run on the main thread and freeze the page; the
|
|
14
|
+
* worker keeps the document's index and answers each search off-thread.
|
|
15
|
+
*/
|
|
16
|
+
export declare class FuzzyWorkerClient {
|
|
17
|
+
#private;
|
|
18
|
+
constructor(worker: FuzzyWorkerTransport);
|
|
19
|
+
/** Spawn the bundled worker script; `undefined` where workers do not exist. */
|
|
20
|
+
static create(workerUrl: URL): FuzzyWorkerClient | undefined;
|
|
21
|
+
index(pages: readonly FuzzyPageText[], options: {
|
|
22
|
+
readonly threshold: number;
|
|
23
|
+
readonly maxPageTextLength: number;
|
|
24
|
+
readonly caseSensitive: boolean;
|
|
25
|
+
}): Promise<number>;
|
|
26
|
+
search(query: string, maxScore: number, pageIndices?: readonly number[]): Promise<readonly SearchMatch[]>;
|
|
27
|
+
terminate(): void;
|
|
28
|
+
}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import { ViewerError } from "./errors.js";
|
|
2
|
+
/**
|
|
3
|
+
* Main-thread side of the fuzzy-search worker. Bitap over a long document
|
|
4
|
+
* is CPU work that used to run on the main thread and freeze the page; the
|
|
5
|
+
* worker keeps the document's index and answers each search off-thread.
|
|
6
|
+
*/
|
|
7
|
+
export class FuzzyWorkerClient {
|
|
8
|
+
#worker;
|
|
9
|
+
#pending = new Map();
|
|
10
|
+
#nextId = 1;
|
|
11
|
+
#dead;
|
|
12
|
+
constructor(worker) {
|
|
13
|
+
this.#worker = worker;
|
|
14
|
+
worker.addEventListener("message", (event) => {
|
|
15
|
+
const pending = this.#pending.get(event.data.id);
|
|
16
|
+
if (!pending)
|
|
17
|
+
return;
|
|
18
|
+
this.#pending.delete(event.data.id);
|
|
19
|
+
pending.resolve(event.data);
|
|
20
|
+
});
|
|
21
|
+
worker.addEventListener("error", (event) => {
|
|
22
|
+
this.#fail(new ViewerError("worker-crashed", "Fuzzy search worker crashed", {
|
|
23
|
+
details: { message: event.message },
|
|
24
|
+
}));
|
|
25
|
+
});
|
|
26
|
+
}
|
|
27
|
+
/** Spawn the bundled worker script; `undefined` where workers do not exist. */
|
|
28
|
+
static create(workerUrl) {
|
|
29
|
+
if (typeof Worker === "undefined")
|
|
30
|
+
return undefined;
|
|
31
|
+
return new FuzzyWorkerClient(new Worker(workerUrl, { type: "module", name: "fuzzy-search" }));
|
|
32
|
+
}
|
|
33
|
+
async index(pages, options) {
|
|
34
|
+
const reply = await this.#send({ kind: "index", id: 0, pages, ...options });
|
|
35
|
+
if (reply.kind !== "indexed")
|
|
36
|
+
throw unexpected(reply);
|
|
37
|
+
return reply.pageCount;
|
|
38
|
+
}
|
|
39
|
+
async search(query, maxScore, pageIndices) {
|
|
40
|
+
const reply = await this.#send({
|
|
41
|
+
kind: "search",
|
|
42
|
+
id: 0,
|
|
43
|
+
query,
|
|
44
|
+
maxScore,
|
|
45
|
+
...(pageIndices ? { pageIndices } : {}),
|
|
46
|
+
});
|
|
47
|
+
if (reply.kind !== "matches")
|
|
48
|
+
throw unexpected(reply);
|
|
49
|
+
return reply.matches;
|
|
50
|
+
}
|
|
51
|
+
terminate() {
|
|
52
|
+
this.#fail(new ViewerError("lifecycle-error", "Fuzzy search worker was terminated"));
|
|
53
|
+
this.#worker.terminate();
|
|
54
|
+
}
|
|
55
|
+
#send(request) {
|
|
56
|
+
if (this.#dead)
|
|
57
|
+
return Promise.reject(this.#dead);
|
|
58
|
+
const id = this.#nextId++;
|
|
59
|
+
return new Promise((resolve, reject) => {
|
|
60
|
+
this.#pending.set(id, { resolve, reject });
|
|
61
|
+
this.#worker.postMessage({ ...request, id });
|
|
62
|
+
});
|
|
63
|
+
}
|
|
64
|
+
#fail(error) {
|
|
65
|
+
this.#dead = error;
|
|
66
|
+
for (const pending of this.#pending.values())
|
|
67
|
+
pending.reject(error);
|
|
68
|
+
this.#pending.clear();
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
function unexpected(reply) {
|
|
72
|
+
return new ViewerError("render-failed", reply.kind === "failure"
|
|
73
|
+
? reply.message
|
|
74
|
+
: `Unexpected worker reply: ${reply.kind}`);
|
|
75
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import type { SearchMatch } from "./contracts.js";
|
|
2
|
+
import { FuzzyPageIndex, type FuzzyPageText } from "./fuzzy-search.js";
|
|
3
|
+
/**
|
|
4
|
+
* Messages between the viewer and the fuzzy-search worker. The worker holds
|
|
5
|
+
* one {@link FuzzyPageIndex} at a time: `index` replaces it, `search` runs
|
|
6
|
+
* against it. Every request is answered by exactly one reply with its `id`.
|
|
7
|
+
*/
|
|
8
|
+
export type FuzzyWorkerRequest = {
|
|
9
|
+
readonly kind: "index";
|
|
10
|
+
readonly id: number;
|
|
11
|
+
readonly pages: readonly FuzzyPageText[];
|
|
12
|
+
readonly threshold: number;
|
|
13
|
+
readonly maxPageTextLength: number;
|
|
14
|
+
readonly caseSensitive: boolean;
|
|
15
|
+
} | {
|
|
16
|
+
readonly kind: "search";
|
|
17
|
+
readonly id: number;
|
|
18
|
+
readonly query: string;
|
|
19
|
+
readonly maxScore: number;
|
|
20
|
+
readonly pageIndices?: readonly number[];
|
|
21
|
+
};
|
|
22
|
+
export type FuzzyWorkerReply = {
|
|
23
|
+
readonly kind: "indexed";
|
|
24
|
+
readonly id: number;
|
|
25
|
+
readonly pageCount: number;
|
|
26
|
+
} | {
|
|
27
|
+
readonly kind: "matches";
|
|
28
|
+
readonly id: number;
|
|
29
|
+
readonly matches: readonly SearchMatch[];
|
|
30
|
+
} | {
|
|
31
|
+
readonly kind: "failure";
|
|
32
|
+
readonly id: number;
|
|
33
|
+
readonly message: string;
|
|
34
|
+
};
|
|
35
|
+
export interface FuzzyWorkerState {
|
|
36
|
+
index?: FuzzyPageIndex;
|
|
37
|
+
}
|
|
38
|
+
/** Apply one request to the worker's state and produce its reply. */
|
|
39
|
+
export declare function handleFuzzyWorkerRequest(state: FuzzyWorkerState, request: FuzzyWorkerRequest): FuzzyWorkerReply;
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { FuzzyPageIndex } from "./fuzzy-search.js";
|
|
2
|
+
/** Apply one request to the worker's state and produce its reply. */
|
|
3
|
+
export function handleFuzzyWorkerRequest(state, request) {
|
|
4
|
+
try {
|
|
5
|
+
if (request.kind === "index") {
|
|
6
|
+
state.index = new FuzzyPageIndex(request.pages, {
|
|
7
|
+
threshold: request.threshold,
|
|
8
|
+
maxPageTextLength: request.maxPageTextLength,
|
|
9
|
+
}, request.caseSensitive);
|
|
10
|
+
return {
|
|
11
|
+
kind: "indexed",
|
|
12
|
+
id: request.id,
|
|
13
|
+
pageCount: state.index.pageCount,
|
|
14
|
+
};
|
|
15
|
+
}
|
|
16
|
+
if (!state.index)
|
|
17
|
+
return {
|
|
18
|
+
kind: "failure",
|
|
19
|
+
id: request.id,
|
|
20
|
+
message: "No document is indexed",
|
|
21
|
+
};
|
|
22
|
+
return {
|
|
23
|
+
kind: "matches",
|
|
24
|
+
id: request.id,
|
|
25
|
+
matches: state.index.search(request.query, request.maxScore, request.pageIndices),
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
catch (error) {
|
|
29
|
+
return {
|
|
30
|
+
kind: "failure",
|
|
31
|
+
id: request.id,
|
|
32
|
+
message: error instanceof Error ? error.message : String(error),
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -6,6 +6,8 @@ export * from "./format.js";
|
|
|
6
6
|
export * from "./limits.js";
|
|
7
7
|
export * from "./interaction.js";
|
|
8
8
|
export * from "./fuzzy-search.js";
|
|
9
|
+
export * from "./fuzzy-worker-protocol.js";
|
|
10
|
+
export * from "./fuzzy-worker-client.js";
|
|
9
11
|
export * from "./search-reveal.js";
|
|
10
12
|
export * from "./adapters/docx-images.js";
|
|
11
13
|
export * from "./render-scheduler.js";
|
package/dist/index.js
CHANGED
|
@@ -6,6 +6,8 @@ export * from "./format.js";
|
|
|
6
6
|
export * from "./limits.js";
|
|
7
7
|
export * from "./interaction.js";
|
|
8
8
|
export * from "./fuzzy-search.js";
|
|
9
|
+
export * from "./fuzzy-worker-protocol.js";
|
|
10
|
+
export * from "./fuzzy-worker-client.js";
|
|
9
11
|
export * from "./search-reveal.js";
|
|
10
12
|
export * from "./adapters/docx-images.js";
|
|
11
13
|
export * from "./render-scheduler.js";
|
package/dist/viewer.js
CHANGED
|
@@ -3,6 +3,7 @@ import { detectFormat } from "./detect.js";
|
|
|
3
3
|
import { abortError, normalizeError, ViewerError } from "./errors.js";
|
|
4
4
|
import { cellRangeToTsv, cellRangesToTsv, findNormalizedMatches, normalizeCellRange, } from "./interaction.js";
|
|
5
5
|
import { findFuzzyPageMatches, nearestMatchIndex, pagesNearestFirst, resolveFuzzySearchOptions, } from "./fuzzy-search.js";
|
|
6
|
+
import { FuzzyWorkerClient } from "./fuzzy-worker-client.js";
|
|
6
7
|
import { enforceContainerLimits, resolveLimits } from "./limits.js";
|
|
7
8
|
import { loadDocumentSource } from "./source.js";
|
|
8
9
|
import { AdaptiveViewport } from "./viewport.js";
|
|
@@ -31,6 +32,12 @@ export class DocumentViewer {
|
|
|
31
32
|
#activeSearch;
|
|
32
33
|
#selection = null;
|
|
33
34
|
#searchResult = null;
|
|
35
|
+
/** Off-thread fuzzy matcher holding the current document's index. */
|
|
36
|
+
#fuzzyWorker;
|
|
37
|
+
/** Which pages/options the worker's index was built from; rebuilt on change. */
|
|
38
|
+
#fuzzyIndexKey;
|
|
39
|
+
/** Set once the worker failed to start, so the main thread takes over for good. */
|
|
40
|
+
#fuzzyWorkerUnavailable = false;
|
|
34
41
|
#generation = 0;
|
|
35
42
|
#searchGeneration = 0;
|
|
36
43
|
#viewEventScheduled = false;
|
|
@@ -411,26 +418,37 @@ export class DocumentViewer {
|
|
|
411
418
|
matches.push(...findNormalizedMatches(text, cleanQuery, pageIndex, caseSensitive));
|
|
412
419
|
}
|
|
413
420
|
if (matches.length === 0 && fuzzy) {
|
|
414
|
-
// The page texts are already in memory, so the fallback is CPU only.
|
|
415
|
-
// Pages are compared nearest to the hint first, a batch at a time,
|
|
416
|
-
// and the scan stops at the first batch that holds the passage; a
|
|
417
|
-
// yield between batches keeps a long document from freezing the UI.
|
|
418
421
|
// With a hint the passage sits near it, so only that neighbourhood is
|
|
419
422
|
// worth the fuzzy cost; without one every page is a candidate.
|
|
420
423
|
const order = pagesNearestFirst(firstPage, lastPage, nearPage).slice(0, nearPage === undefined ? undefined : fuzzy.pageWindow);
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
424
|
+
const pattern = cleanQuery.slice(0, fuzzy.maxQueryLength);
|
|
425
|
+
const pages = [];
|
|
426
|
+
for (const [pageIndex, text] of texts)
|
|
427
|
+
pages.push({ pageIndex, text });
|
|
428
|
+
const offThread = fuzzy.worker
|
|
429
|
+
? await this.#searchFuzzyInWorker(pages, pattern, order, fuzzy, caseSensitive, controller.signal)
|
|
430
|
+
: undefined;
|
|
431
|
+
if (offThread)
|
|
432
|
+
matches.push(...offThread);
|
|
433
|
+
else
|
|
434
|
+
for (let offset = 0; offset < order.length && matches.length === 0; offset += fuzzy.pagesPerBatch) {
|
|
435
|
+
// The page texts are already in memory, so the fallback is CPU
|
|
436
|
+
// only. Pages are compared nearest to the hint first, a batch at
|
|
437
|
+
// a time, and the scan stops at the first batch that holds the
|
|
438
|
+
// passage; a yield between batches keeps a long document from
|
|
439
|
+
// freezing the UI.
|
|
440
|
+
if (offset > 0)
|
|
441
|
+
await yieldToEventLoop();
|
|
442
|
+
if (controller.signal.aborted)
|
|
443
|
+
throw abortError();
|
|
444
|
+
const batch = order
|
|
445
|
+
.slice(offset, offset + fuzzy.pagesPerBatch)
|
|
446
|
+
.map((pageIndex) => ({
|
|
447
|
+
pageIndex,
|
|
448
|
+
text: texts.get(pageIndex) ?? "",
|
|
449
|
+
}));
|
|
450
|
+
matches.push(...findFuzzyPageMatches(batch, pattern, fuzzy, caseSensitive));
|
|
451
|
+
}
|
|
434
452
|
if (matches.length > 0)
|
|
435
453
|
strategy = "fuzzy";
|
|
436
454
|
}
|
|
@@ -454,6 +472,56 @@ export class DocumentViewer {
|
|
|
454
472
|
this.#activeSearch = undefined;
|
|
455
473
|
}
|
|
456
474
|
}
|
|
475
|
+
/**
|
|
476
|
+
* Fuzzy scan in the worker. The document's pages are indexed once per
|
|
477
|
+
* (range, options) and reused by every search until the document closes;
|
|
478
|
+
* `undefined` hands the scan back to the main thread when the worker is
|
|
479
|
+
* unavailable or failed, so a missing worker asset degrades to slowness,
|
|
480
|
+
* never to a lost match.
|
|
481
|
+
*/
|
|
482
|
+
async #searchFuzzyInWorker(pages, pattern, pageIndices, fuzzy, caseSensitive, signal) {
|
|
483
|
+
if (this.#fuzzyWorkerUnavailable)
|
|
484
|
+
return undefined;
|
|
485
|
+
try {
|
|
486
|
+
this.#fuzzyWorker ??= FuzzyWorkerClient.create(this.#fuzzyWorkerUrl());
|
|
487
|
+
if (!this.#fuzzyWorker) {
|
|
488
|
+
this.#fuzzyWorkerUnavailable = true;
|
|
489
|
+
return undefined;
|
|
490
|
+
}
|
|
491
|
+
const key = `${pages.map((page) => page.pageIndex).join(",")}|${fuzzy.threshold}|${fuzzy.maxPageTextLength}|${caseSensitive}`;
|
|
492
|
+
if (this.#fuzzyIndexKey !== key) {
|
|
493
|
+
this.#fuzzyIndexKey = undefined;
|
|
494
|
+
await this.#fuzzyWorker.index(pages, {
|
|
495
|
+
threshold: fuzzy.threshold,
|
|
496
|
+
maxPageTextLength: fuzzy.maxPageTextLength,
|
|
497
|
+
caseSensitive,
|
|
498
|
+
});
|
|
499
|
+
this.#fuzzyIndexKey = key;
|
|
500
|
+
}
|
|
501
|
+
const matches = await this.#fuzzyWorker.search(pattern, fuzzy.maxScore, pageIndices);
|
|
502
|
+
if (signal.aborted)
|
|
503
|
+
throw abortError();
|
|
504
|
+
return matches;
|
|
505
|
+
}
|
|
506
|
+
catch (error) {
|
|
507
|
+
if (signal.aborted)
|
|
508
|
+
throw error;
|
|
509
|
+
this.#runtime.logger?.warn?.("Fuzzy search worker unavailable; matching on the main thread", { message: error instanceof Error ? error.message : String(error) });
|
|
510
|
+
this.#dropFuzzyWorker();
|
|
511
|
+
this.#fuzzyWorkerUnavailable = true;
|
|
512
|
+
return undefined;
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
#fuzzyWorkerUrl() {
|
|
516
|
+
return this.#runtime.assetBaseUrl
|
|
517
|
+
? new URL("workers/fuzzy-search-worker.js", this.#runtime.assetBaseUrl)
|
|
518
|
+
: new URL("./workers/fuzzy-search-worker.js", import.meta.url);
|
|
519
|
+
}
|
|
520
|
+
#dropFuzzyWorker() {
|
|
521
|
+
this.#fuzzyWorker?.terminate();
|
|
522
|
+
this.#fuzzyWorker = undefined;
|
|
523
|
+
this.#fuzzyIndexKey = undefined;
|
|
524
|
+
}
|
|
457
525
|
searchNext() {
|
|
458
526
|
return this.#moveSearch(1);
|
|
459
527
|
}
|
|
@@ -721,6 +789,7 @@ export class DocumentViewer {
|
|
|
721
789
|
this.#activeSearch?.abort();
|
|
722
790
|
this.#activeSearch = undefined;
|
|
723
791
|
this.#searchResult = null;
|
|
792
|
+
this.#dropFuzzyWorker();
|
|
724
793
|
this.#selection = null;
|
|
725
794
|
this.#textMaps.clear();
|
|
726
795
|
this.#textMapBytes = 0;
|