web-doc 0.5.0 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_NOTICES.md +1 -0
- package/dist/contracts.d.ts +14 -7
- package/dist/fuzzy-alignment.d.ts +12 -0
- package/dist/fuzzy-alignment.js +63 -0
- package/dist/fuzzy-search-worker.d.ts +1 -0
- package/dist/fuzzy-search-worker.js +6 -0
- package/dist/fuzzy-search.d.ts +28 -5
- package/dist/fuzzy-search.js +71 -34
- package/dist/fuzzy-worker-client.d.ts +28 -0
- package/dist/fuzzy-worker-client.js +75 -0
- package/dist/fuzzy-worker-protocol.d.ts +39 -0
- package/dist/fuzzy-worker-protocol.js +35 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.js +2 -0
- package/dist/interaction.d.ts +1 -1
- package/dist/interaction.js +2 -44
- package/dist/search-text.d.ts +11 -0
- package/dist/search-text.js +44 -0
- package/dist/viewer.js +86 -17
- package/dist/workers/fuzzy-search-worker.js +1773 -0
- package/package.json +1 -1
package/THIRD_PARTY_NOTICES.md
CHANGED
|
@@ -4,6 +4,7 @@ The release artifact contains or depends on the following principal components.
|
|
|
4
4
|
|
|
5
5
|
- [`@silurus/ooxml`](https://github.com/yukiyokotani/office-open-xml-viewer) — MIT; modern Office parsing/rendering.
|
|
6
6
|
- [`office_oxide`](https://github.com/yfedoseev/office_oxide) — MIT OR Apache-2.0; compound-file handling, Office IR/writer utilities and legacy XLS/PPT conversion.
|
|
7
|
+
- [`Fuse.js`](https://github.com/krisk/Fuse) — Apache-2.0; fuzzy matching behind the opt-in `search()` fallback.
|
|
7
8
|
- [`pdfjs-dist` / Mozilla PDF.js](https://github.com/mozilla/pdf.js) — Apache-2.0; browser PDF parsing,
|
|
8
9
|
font/CMap handling, canvas rendering, and text extraction. The packaged
|
|
9
10
|
standard-font, ICC, CMap, OpenJPEG, JBIG2, and QCMS assets retain the license
|
package/dist/contracts.d.ts
CHANGED
|
@@ -219,9 +219,9 @@ export interface ViewerEventMap {
|
|
|
219
219
|
}
|
|
220
220
|
/**
|
|
221
221
|
* Tuning for the fuzzy fallback that runs when the exact search finds
|
|
222
|
-
* nothing.
|
|
223
|
-
*
|
|
224
|
-
* bullets, table separators and
|
|
222
|
+
* nothing. Fuse.js selects candidate pages; a contiguous edit alignment
|
|
223
|
+
* determines the original-text boundaries with a bounded edit budget. Spacing,
|
|
224
|
+
* line breaks, list bullets, table separators and punctuation may differ from the
|
|
225
225
|
* source, and every hit maps back to the verbatim page text.
|
|
226
226
|
*/
|
|
227
227
|
export interface FuzzySearchOptions {
|
|
@@ -231,15 +231,15 @@ export interface FuzzySearchOptions {
|
|
|
231
231
|
*/
|
|
232
232
|
readonly threshold?: number;
|
|
233
233
|
/**
|
|
234
|
-
* Highest Fuse.js score (`0` perfect, `1` no
|
|
235
|
-
*
|
|
236
|
-
* partly survives on a page, such as
|
|
234
|
+
* Highest Fuse.js score and whole-passage edit ratio (`0` perfect, `1` no
|
|
235
|
+
* resemblance) a match may have. Default `0.4`; raise it to accept a passage
|
|
236
|
+
* that only partly survives on a page, such as one spanning a page break.
|
|
237
237
|
*/
|
|
238
238
|
readonly maxScore?: number;
|
|
239
239
|
/**
|
|
240
240
|
* Query characters considered. The matcher's cost grows with the query and
|
|
241
241
|
* a passage is identified well before its end, so the default `600` keeps
|
|
242
|
-
*
|
|
242
|
+
* the work bounded; the highlight covers the matched prefix.
|
|
243
243
|
*/
|
|
244
244
|
readonly maxQueryLength?: number;
|
|
245
245
|
/** Characters of each page's text considered. Default `20000`. */
|
|
@@ -257,6 +257,13 @@ export interface FuzzySearchOptions {
|
|
|
257
257
|
* hint every page in range is scanned. Default `12`.
|
|
258
258
|
*/
|
|
259
259
|
readonly pageWindow?: number;
|
|
260
|
+
/**
|
|
261
|
+
* Run the matcher in a Web Worker that keeps the document's index, so a
|
|
262
|
+
* long scan never blocks the page and repeated citation lookups reuse the
|
|
263
|
+
* index. Falls back to the main thread where workers are unavailable or
|
|
264
|
+
* the worker script cannot be loaded. Default `true`.
|
|
265
|
+
*/
|
|
266
|
+
readonly worker?: boolean;
|
|
260
267
|
}
|
|
261
268
|
export interface SearchOptions {
|
|
262
269
|
readonly caseSensitive?: boolean;
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fuse's indices describe character masks, not a contiguous edit alignment.
|
|
3
|
+
* Refine an accepted page with semi-global Levenshtein alignment: consume the
|
|
4
|
+
* whole query, allowing free text before and after one occurrence. Tracking
|
|
5
|
+
* the start alongside each cost needs O(query length) working memory rather
|
|
6
|
+
* than a page × query traceback matrix. Equal-cost occurrences prefer the
|
|
7
|
+
* earliest end, so an unmatched character after a passage cannot extend it.
|
|
8
|
+
*/
|
|
9
|
+
export declare function alignFuzzyPassage(text: string, query: string, caseSensitive: boolean, maxScore: number): {
|
|
10
|
+
start: number;
|
|
11
|
+
end: number;
|
|
12
|
+
} | undefined;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import { normalizeSearchText, normalizeWithMap } from "./search-text.js";
|
|
2
|
+
/**
|
|
3
|
+
* Fuse's indices describe character masks, not a contiguous edit alignment.
|
|
4
|
+
* Refine an accepted page with semi-global Levenshtein alignment: consume the
|
|
5
|
+
* whole query, allowing free text before and after one occurrence. Tracking
|
|
6
|
+
* the start alongside each cost needs O(query length) working memory rather
|
|
7
|
+
* than a page × query traceback matrix. Equal-cost occurrences prefer the
|
|
8
|
+
* earliest end, so an unmatched character after a passage cannot extend it.
|
|
9
|
+
*/
|
|
10
|
+
export function alignFuzzyPassage(text, query, caseSensitive, maxScore) {
|
|
11
|
+
const source = normalizeWithMap(text, caseSensitive);
|
|
12
|
+
const pattern = normalizeSearchText(query, caseSensitive);
|
|
13
|
+
const length = pattern.length;
|
|
14
|
+
if (!length || !source.text.length)
|
|
15
|
+
return undefined;
|
|
16
|
+
const costs = new Uint32Array(length + 1);
|
|
17
|
+
const starts = new Uint32Array(length + 1);
|
|
18
|
+
for (let i = 0; i <= length; i += 1)
|
|
19
|
+
costs[i] = i;
|
|
20
|
+
let bestCost = length;
|
|
21
|
+
let bestStart = 0;
|
|
22
|
+
let bestEnd = 0;
|
|
23
|
+
for (let end = 1; end <= source.text.length; end += 1) {
|
|
24
|
+
let diagonalCost = costs[0];
|
|
25
|
+
let diagonalStart = starts[0];
|
|
26
|
+
costs[0] = 0;
|
|
27
|
+
starts[0] = end;
|
|
28
|
+
for (let i = 1; i <= length; i += 1) {
|
|
29
|
+
const previousCost = costs[i];
|
|
30
|
+
const previousStart = starts[i];
|
|
31
|
+
let cost = diagonalCost + (pattern[i - 1] === source.text[end - 1] ? 0 : 1);
|
|
32
|
+
let start = diagonalStart;
|
|
33
|
+
// Prefer dropping an unmatched query character over substituting a
|
|
34
|
+
// neighbouring source character when both alignments cost the same.
|
|
35
|
+
const deletion = costs[i - 1] + 1;
|
|
36
|
+
if (deletion <= cost) {
|
|
37
|
+
cost = deletion;
|
|
38
|
+
start = starts[i - 1];
|
|
39
|
+
}
|
|
40
|
+
const insertion = previousCost + 1;
|
|
41
|
+
if (insertion < cost) {
|
|
42
|
+
cost = insertion;
|
|
43
|
+
start = previousStart;
|
|
44
|
+
}
|
|
45
|
+
costs[i] = cost;
|
|
46
|
+
starts[i] = start;
|
|
47
|
+
diagonalCost = previousCost;
|
|
48
|
+
diagonalStart = previousStart;
|
|
49
|
+
}
|
|
50
|
+
if (costs[length] < bestCost) {
|
|
51
|
+
bestCost = costs[length];
|
|
52
|
+
bestStart = starts[length];
|
|
53
|
+
bestEnd = end;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
// A page-level Fuse score alone can accept disconnected 32-character
|
|
57
|
+
// chunks. Require the single passage to meet the same score ceiling.
|
|
58
|
+
if (bestEnd <= bestStart || bestCost / length > maxScore)
|
|
59
|
+
return undefined;
|
|
60
|
+
const start = source.starts[bestStart];
|
|
61
|
+
const end = source.ends[bestEnd - 1];
|
|
62
|
+
return start === undefined || end === undefined ? undefined : { start, end };
|
|
63
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/fuzzy-search.d.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import type { FuzzySearchOptions, SearchMatch } from "./contracts.js";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
* 32-character chunk)
|
|
3
|
+
* Fuse.js selects candidate pages (Bitap with a bounded edit budget per
|
|
4
|
+
* 32-character chunk); contiguous alignment determines highlight boundaries.
|
|
5
|
+
* This bridges the gaps NFKC + case folding leave open when
|
|
5
6
|
* a query was not copied verbatim from the document: AI-generated citations
|
|
6
7
|
* and OCR'd passages differ from the source in spacing, line breaks, list
|
|
7
8
|
* bullets, table separators and typographic punctuation. Every hit maps back
|
|
@@ -14,6 +15,7 @@ export interface ResolvedFuzzySearchOptions {
|
|
|
14
15
|
readonly maxPageTextLength: number;
|
|
15
16
|
readonly pagesPerBatch: number;
|
|
16
17
|
readonly pageWindow: number;
|
|
18
|
+
readonly worker: boolean;
|
|
17
19
|
}
|
|
18
20
|
export declare const DEFAULT_FUZZY_SEARCH_OPTIONS: ResolvedFuzzySearchOptions;
|
|
19
21
|
/**
|
|
@@ -28,9 +30,30 @@ export interface FuzzyPageText {
|
|
|
28
30
|
readonly text: string;
|
|
29
31
|
}
|
|
30
32
|
/**
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
33
|
+
* A document's pages, indexed once for Fuse.js so that every fuzzy search
|
|
34
|
+
* reuses the normalized records instead of rebuilding them: a citation
|
|
35
|
+
* lookup tries several anchors in a row, and each used to pay for the index
|
|
36
|
+
* again. A scan restricted to a page window reuses the same records through
|
|
37
|
+
* `Fuse.parseIndex`, so only scoring and alignment run for those pages.
|
|
38
|
+
*/
|
|
39
|
+
export declare class FuzzyPageIndex {
|
|
40
|
+
#private;
|
|
41
|
+
constructor(pages: readonly FuzzyPageText[], options: {
|
|
42
|
+
threshold: number;
|
|
43
|
+
maxPageTextLength: number;
|
|
44
|
+
}, caseSensitive?: boolean);
|
|
45
|
+
get pageCount(): number;
|
|
46
|
+
/**
|
|
47
|
+
* One match per page whose text holds the query within the edit budget:
|
|
48
|
+
* the best contiguous alignment, so neighbouring text and repeated
|
|
49
|
+
* occurrences cannot expand the highlight. `pageIndices` restricts the scan;
|
|
50
|
+
* the result is in page order either way.
|
|
51
|
+
*/
|
|
52
|
+
search(query: string, maxScore: number, pageIndices?: readonly number[]): readonly SearchMatch[];
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* One-off scan of a few pages, for callers without a standing index. The
|
|
56
|
+
* viewer keeps a {@link FuzzyPageIndex} per document instead.
|
|
34
57
|
*/
|
|
35
58
|
export declare function findFuzzyPageMatches(pages: readonly FuzzyPageText[], query: string, options: ResolvedFuzzySearchOptions, caseSensitive?: boolean): readonly SearchMatch[];
|
|
36
59
|
/** Page indices of `[first, last]` ordered by distance from `nearPage`, ties earlier-first. */
|
package/dist/fuzzy-search.js
CHANGED
|
@@ -1,13 +1,15 @@
|
|
|
1
|
-
import Fuse from "fuse.js";
|
|
1
|
+
import Fuse, {} from "fuse.js";
|
|
2
|
+
import { alignFuzzyPassage } from "./fuzzy-alignment.js";
|
|
2
3
|
export const DEFAULT_FUZZY_SEARCH_OPTIONS = Object.freeze({
|
|
3
4
|
threshold: 0.3,
|
|
4
5
|
maxScore: 0.4,
|
|
5
6
|
// Bitap cost grows with the query; 600 characters still identifies a
|
|
6
|
-
// passage while
|
|
7
|
+
// passage while bounding both candidate scoring and alignment work.
|
|
7
8
|
maxQueryLength: 600,
|
|
8
9
|
maxPageTextLength: 20_000,
|
|
9
10
|
pagesPerBatch: 2,
|
|
10
11
|
pageWindow: 12,
|
|
12
|
+
worker: true,
|
|
11
13
|
});
|
|
12
14
|
/**
|
|
13
15
|
* Layer fuzzy settings in precedence order (viewer defaults first, then the
|
|
@@ -33,55 +35,90 @@ export function resolveFuzzySearchOptions(...layers) {
|
|
|
33
35
|
maxPageTextLength: positiveInteger(layer.maxPageTextLength, resolved.maxPageTextLength),
|
|
34
36
|
pagesPerBatch: positiveInteger(layer.pagesPerBatch, resolved.pagesPerBatch),
|
|
35
37
|
pageWindow: positiveInteger(layer.pageWindow, resolved.pageWindow),
|
|
38
|
+
worker: layer.worker ?? resolved.worker,
|
|
36
39
|
};
|
|
37
40
|
}
|
|
38
41
|
return enabled ? resolved : undefined;
|
|
39
42
|
}
|
|
40
|
-
/**
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
* the passage as one block. Pages are returned in page order.
|
|
44
|
-
*/
|
|
45
|
-
export function findFuzzyPageMatches(pages, query, options, caseSensitive = false) {
|
|
46
|
-
const pattern = query.slice(0, options.maxQueryLength);
|
|
47
|
-
if (!pattern.trim())
|
|
48
|
-
return [];
|
|
49
|
-
const items = pages.map((page) => ({
|
|
50
|
-
pageIndex: page.pageIndex,
|
|
51
|
-
text: page.text.slice(0, options.maxPageTextLength),
|
|
52
|
-
}));
|
|
53
|
-
const fuse = new Fuse(items, {
|
|
43
|
+
/** Fuse.js options shared by every fuzzy scan; `keys` and `threshold` are fixed per index. */
|
|
44
|
+
function fuseOptions(threshold, caseSensitive) {
|
|
45
|
+
return {
|
|
54
46
|
keys: ["text"],
|
|
55
47
|
isCaseSensitive: caseSensitive,
|
|
56
48
|
ignoreDiacritics: false,
|
|
57
|
-
includeMatches: true,
|
|
58
49
|
includeScore: true,
|
|
59
50
|
// A citation can sit anywhere on the page; Fuse's location bias would
|
|
60
51
|
// otherwise penalize matches far from the start of the text.
|
|
61
52
|
ignoreLocation: true,
|
|
62
53
|
// Long page text must not dilute the score of a match inside it.
|
|
63
54
|
ignoreFieldNorm: true,
|
|
64
|
-
threshold
|
|
55
|
+
threshold,
|
|
65
56
|
minMatchCharLength: 3,
|
|
66
57
|
shouldSort: false,
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* A document's pages, indexed once for Fuse.js so that every fuzzy search
|
|
62
|
+
* reuses the normalized records instead of rebuilding them: a citation
|
|
63
|
+
* lookup tries several anchors in a row, and each used to pay for the index
|
|
64
|
+
* again. A scan restricted to a page window reuses the same records through
|
|
65
|
+
* `Fuse.parseIndex`, so only scoring and alignment run for those pages.
|
|
66
|
+
*/
|
|
67
|
+
export class FuzzyPageIndex {
|
|
68
|
+
#caseSensitive;
|
|
69
|
+
#pages;
|
|
70
|
+
#index;
|
|
71
|
+
#options;
|
|
72
|
+
#fuse;
|
|
73
|
+
constructor(pages, options, caseSensitive = false) {
|
|
74
|
+
this.#caseSensitive = caseSensitive;
|
|
75
|
+
this.#pages = pages.map((page) => ({
|
|
76
|
+
pageIndex: page.pageIndex,
|
|
77
|
+
text: page.text.slice(0, options.maxPageTextLength),
|
|
78
|
+
}));
|
|
79
|
+
this.#options = fuseOptions(options.threshold, caseSensitive);
|
|
80
|
+
this.#index = Fuse.createIndex(["text"], this.#pages);
|
|
81
|
+
this.#fuse = new Fuse(this.#pages, this.#options, this.#index);
|
|
82
|
+
}
|
|
83
|
+
get pageCount() {
|
|
84
|
+
return this.#pages.length;
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* One match per page whose text holds the query within the edit budget:
|
|
88
|
+
* the best contiguous alignment, so neighbouring text and repeated
|
|
89
|
+
* occurrences cannot expand the highlight. `pageIndices` restricts the scan;
|
|
90
|
+
* the result is in page order either way.
|
|
91
|
+
*/
|
|
92
|
+
search(query, maxScore, pageIndices) {
|
|
93
|
+
if (!query.trim())
|
|
94
|
+
return [];
|
|
95
|
+
const fuse = pageIndices ? this.#subset(pageIndices) : this.#fuse;
|
|
96
|
+
const matches = [];
|
|
97
|
+
for (const result of fuse.search(query)) {
|
|
98
|
+
if ((result.score ?? 1) > maxScore)
|
|
99
|
+
continue;
|
|
100
|
+
const span = alignFuzzyPassage(result.item.text, query, this.#caseSensitive, maxScore);
|
|
101
|
+
if (!span)
|
|
102
|
+
continue;
|
|
103
|
+
const { start, end } = span;
|
|
104
|
+
const text = result.item.text.slice(start, end);
|
|
105
|
+
matches.push({ pageIndex: result.item.pageIndex, start, end, text });
|
|
80
106
|
}
|
|
81
|
-
|
|
82
|
-
matches.push({ pageIndex: result.item.pageIndex, start, end, text });
|
|
107
|
+
return matches.sort((a, b) => a.pageIndex - b.pageIndex);
|
|
83
108
|
}
|
|
84
|
-
|
|
109
|
+
#subset(pageIndices) {
|
|
110
|
+
const wanted = new Set(pageIndices);
|
|
111
|
+
const { keys, records } = this.#index.toJSON();
|
|
112
|
+
const kept = records.filter((record) => wanted.has(this.#pages[record.i]?.pageIndex ?? -1));
|
|
113
|
+
return new Fuse(this.#pages, this.#options, Fuse.parseIndex({ keys, records: kept }));
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* One-off scan of a few pages, for callers without a standing index. The
|
|
118
|
+
* viewer keeps a {@link FuzzyPageIndex} per document instead.
|
|
119
|
+
*/
|
|
120
|
+
export function findFuzzyPageMatches(pages, query, options, caseSensitive = false) {
|
|
121
|
+
return new FuzzyPageIndex(pages, options, caseSensitive).search(query.slice(0, options.maxQueryLength), options.maxScore);
|
|
85
122
|
}
|
|
86
123
|
/** Page indices of `[first, last]` ordered by distance from `nearPage`, ties earlier-first. */
|
|
87
124
|
export function pagesNearestFirst(first, last, nearPage) {
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import type { SearchMatch } from "./contracts.js";
|
|
2
|
+
import type { FuzzyPageText } from "./fuzzy-search.js";
|
|
3
|
+
import type { FuzzyWorkerReply, FuzzyWorkerRequest } from "./fuzzy-worker-protocol.js";
|
|
4
|
+
/** The slice of `Worker` the client needs; tests pass a fake. */
|
|
5
|
+
export interface FuzzyWorkerTransport {
|
|
6
|
+
postMessage(message: FuzzyWorkerRequest): void;
|
|
7
|
+
addEventListener(type: "message", listener: (event: MessageEvent<FuzzyWorkerReply>) => void): void;
|
|
8
|
+
addEventListener(type: "error", listener: (event: Event) => void): void;
|
|
9
|
+
terminate(): void;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* Main-thread side of the fuzzy-search worker. Bitap over a long document
|
|
13
|
+
* is CPU work that used to run on the main thread and freeze the page; the
|
|
14
|
+
* worker keeps the document's index and answers each search off-thread.
|
|
15
|
+
*/
|
|
16
|
+
export declare class FuzzyWorkerClient {
|
|
17
|
+
#private;
|
|
18
|
+
constructor(worker: FuzzyWorkerTransport);
|
|
19
|
+
/** Spawn the bundled worker script; `undefined` where workers do not exist. */
|
|
20
|
+
static create(workerUrl: URL): FuzzyWorkerClient | undefined;
|
|
21
|
+
index(pages: readonly FuzzyPageText[], options: {
|
|
22
|
+
readonly threshold: number;
|
|
23
|
+
readonly maxPageTextLength: number;
|
|
24
|
+
readonly caseSensitive: boolean;
|
|
25
|
+
}): Promise<number>;
|
|
26
|
+
search(query: string, maxScore: number, pageIndices?: readonly number[]): Promise<readonly SearchMatch[]>;
|
|
27
|
+
terminate(): void;
|
|
28
|
+
}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import { ViewerError } from "./errors.js";
|
|
2
|
+
/**
|
|
3
|
+
* Main-thread side of the fuzzy-search worker. Bitap over a long document
|
|
4
|
+
* is CPU work that used to run on the main thread and freeze the page; the
|
|
5
|
+
* worker keeps the document's index and answers each search off-thread.
|
|
6
|
+
*/
|
|
7
|
+
export class FuzzyWorkerClient {
|
|
8
|
+
#worker;
|
|
9
|
+
#pending = new Map();
|
|
10
|
+
#nextId = 1;
|
|
11
|
+
#dead;
|
|
12
|
+
constructor(worker) {
|
|
13
|
+
this.#worker = worker;
|
|
14
|
+
worker.addEventListener("message", (event) => {
|
|
15
|
+
const pending = this.#pending.get(event.data.id);
|
|
16
|
+
if (!pending)
|
|
17
|
+
return;
|
|
18
|
+
this.#pending.delete(event.data.id);
|
|
19
|
+
pending.resolve(event.data);
|
|
20
|
+
});
|
|
21
|
+
worker.addEventListener("error", (event) => {
|
|
22
|
+
this.#fail(new ViewerError("worker-crashed", "Fuzzy search worker crashed", {
|
|
23
|
+
details: { message: event.message },
|
|
24
|
+
}));
|
|
25
|
+
});
|
|
26
|
+
}
|
|
27
|
+
/** Spawn the bundled worker script; `undefined` where workers do not exist. */
|
|
28
|
+
static create(workerUrl) {
|
|
29
|
+
if (typeof Worker === "undefined")
|
|
30
|
+
return undefined;
|
|
31
|
+
return new FuzzyWorkerClient(new Worker(workerUrl, { type: "module", name: "fuzzy-search" }));
|
|
32
|
+
}
|
|
33
|
+
async index(pages, options) {
|
|
34
|
+
const reply = await this.#send({ kind: "index", id: 0, pages, ...options });
|
|
35
|
+
if (reply.kind !== "indexed")
|
|
36
|
+
throw unexpected(reply);
|
|
37
|
+
return reply.pageCount;
|
|
38
|
+
}
|
|
39
|
+
async search(query, maxScore, pageIndices) {
|
|
40
|
+
const reply = await this.#send({
|
|
41
|
+
kind: "search",
|
|
42
|
+
id: 0,
|
|
43
|
+
query,
|
|
44
|
+
maxScore,
|
|
45
|
+
...(pageIndices ? { pageIndices } : {}),
|
|
46
|
+
});
|
|
47
|
+
if (reply.kind !== "matches")
|
|
48
|
+
throw unexpected(reply);
|
|
49
|
+
return reply.matches;
|
|
50
|
+
}
|
|
51
|
+
terminate() {
|
|
52
|
+
this.#fail(new ViewerError("lifecycle-error", "Fuzzy search worker was terminated"));
|
|
53
|
+
this.#worker.terminate();
|
|
54
|
+
}
|
|
55
|
+
#send(request) {
|
|
56
|
+
if (this.#dead)
|
|
57
|
+
return Promise.reject(this.#dead);
|
|
58
|
+
const id = this.#nextId++;
|
|
59
|
+
return new Promise((resolve, reject) => {
|
|
60
|
+
this.#pending.set(id, { resolve, reject });
|
|
61
|
+
this.#worker.postMessage({ ...request, id });
|
|
62
|
+
});
|
|
63
|
+
}
|
|
64
|
+
#fail(error) {
|
|
65
|
+
this.#dead = error;
|
|
66
|
+
for (const pending of this.#pending.values())
|
|
67
|
+
pending.reject(error);
|
|
68
|
+
this.#pending.clear();
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
function unexpected(reply) {
|
|
72
|
+
return new ViewerError("render-failed", reply.kind === "failure"
|
|
73
|
+
? reply.message
|
|
74
|
+
: `Unexpected worker reply: ${reply.kind}`);
|
|
75
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import type { SearchMatch } from "./contracts.js";
|
|
2
|
+
import { FuzzyPageIndex, type FuzzyPageText } from "./fuzzy-search.js";
|
|
3
|
+
/**
|
|
4
|
+
* Messages between the viewer and the fuzzy-search worker. The worker holds
|
|
5
|
+
* one {@link FuzzyPageIndex} at a time: `index` replaces it, `search` runs
|
|
6
|
+
* against it. Every request is answered by exactly one reply with its `id`.
|
|
7
|
+
*/
|
|
8
|
+
export type FuzzyWorkerRequest = {
|
|
9
|
+
readonly kind: "index";
|
|
10
|
+
readonly id: number;
|
|
11
|
+
readonly pages: readonly FuzzyPageText[];
|
|
12
|
+
readonly threshold: number;
|
|
13
|
+
readonly maxPageTextLength: number;
|
|
14
|
+
readonly caseSensitive: boolean;
|
|
15
|
+
} | {
|
|
16
|
+
readonly kind: "search";
|
|
17
|
+
readonly id: number;
|
|
18
|
+
readonly query: string;
|
|
19
|
+
readonly maxScore: number;
|
|
20
|
+
readonly pageIndices?: readonly number[];
|
|
21
|
+
};
|
|
22
|
+
export type FuzzyWorkerReply = {
|
|
23
|
+
readonly kind: "indexed";
|
|
24
|
+
readonly id: number;
|
|
25
|
+
readonly pageCount: number;
|
|
26
|
+
} | {
|
|
27
|
+
readonly kind: "matches";
|
|
28
|
+
readonly id: number;
|
|
29
|
+
readonly matches: readonly SearchMatch[];
|
|
30
|
+
} | {
|
|
31
|
+
readonly kind: "failure";
|
|
32
|
+
readonly id: number;
|
|
33
|
+
readonly message: string;
|
|
34
|
+
};
|
|
35
|
+
export interface FuzzyWorkerState {
|
|
36
|
+
index?: FuzzyPageIndex;
|
|
37
|
+
}
|
|
38
|
+
/** Apply one request to the worker's state and produce its reply. */
|
|
39
|
+
export declare function handleFuzzyWorkerRequest(state: FuzzyWorkerState, request: FuzzyWorkerRequest): FuzzyWorkerReply;
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { FuzzyPageIndex } from "./fuzzy-search.js";
|
|
2
|
+
/** Apply one request to the worker's state and produce its reply. */
|
|
3
|
+
export function handleFuzzyWorkerRequest(state, request) {
|
|
4
|
+
try {
|
|
5
|
+
if (request.kind === "index") {
|
|
6
|
+
state.index = new FuzzyPageIndex(request.pages, {
|
|
7
|
+
threshold: request.threshold,
|
|
8
|
+
maxPageTextLength: request.maxPageTextLength,
|
|
9
|
+
}, request.caseSensitive);
|
|
10
|
+
return {
|
|
11
|
+
kind: "indexed",
|
|
12
|
+
id: request.id,
|
|
13
|
+
pageCount: state.index.pageCount,
|
|
14
|
+
};
|
|
15
|
+
}
|
|
16
|
+
if (!state.index)
|
|
17
|
+
return {
|
|
18
|
+
kind: "failure",
|
|
19
|
+
id: request.id,
|
|
20
|
+
message: "No document is indexed",
|
|
21
|
+
};
|
|
22
|
+
return {
|
|
23
|
+
kind: "matches",
|
|
24
|
+
id: request.id,
|
|
25
|
+
matches: state.index.search(request.query, request.maxScore, request.pageIndices),
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
catch (error) {
|
|
29
|
+
return {
|
|
30
|
+
kind: "failure",
|
|
31
|
+
id: request.id,
|
|
32
|
+
message: error instanceof Error ? error.message : String(error),
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -6,6 +6,8 @@ export * from "./format.js";
|
|
|
6
6
|
export * from "./limits.js";
|
|
7
7
|
export * from "./interaction.js";
|
|
8
8
|
export * from "./fuzzy-search.js";
|
|
9
|
+
export * from "./fuzzy-worker-protocol.js";
|
|
10
|
+
export * from "./fuzzy-worker-client.js";
|
|
9
11
|
export * from "./search-reveal.js";
|
|
10
12
|
export * from "./adapters/docx-images.js";
|
|
11
13
|
export * from "./render-scheduler.js";
|
package/dist/index.js
CHANGED
|
@@ -6,6 +6,8 @@ export * from "./format.js";
|
|
|
6
6
|
export * from "./limits.js";
|
|
7
7
|
export * from "./interaction.js";
|
|
8
8
|
export * from "./fuzzy-search.js";
|
|
9
|
+
export * from "./fuzzy-worker-protocol.js";
|
|
10
|
+
export * from "./fuzzy-worker-client.js";
|
|
9
11
|
export * from "./search-reveal.js";
|
|
10
12
|
export * from "./adapters/docx-images.js";
|
|
11
13
|
export * from "./render-scheduler.js";
|
package/dist/interaction.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
export { normalizeSearchText } from "./search-text.js";
|
|
1
2
|
import type { CellRange, SearchMatch } from "./contracts.js";
|
|
2
3
|
export interface VirtualRange {
|
|
3
4
|
readonly start: number;
|
|
@@ -6,7 +7,6 @@ export interface VirtualRange {
|
|
|
6
7
|
export declare function visibleRange(scrollOffset: number, viewportExtent: number, itemExtent: number, itemCount: number, overscan?: number): VirtualRange;
|
|
7
8
|
export declare function normalizeCellRange(range: CellRange): CellRange;
|
|
8
9
|
export declare function findNormalizedMatches(text: string, query: string, pageIndex: number, caseSensitive?: boolean): readonly SearchMatch[];
|
|
9
|
-
export declare function normalizeSearchText(text: string, caseSensitive?: boolean): string;
|
|
10
10
|
/** Clamp a UTF-16 DOM offset to a grapheme boundary for native selection UI. */
|
|
11
11
|
export declare function snapGraphemeOffset(text: string, offset: number, edge: "start" | "end"): number;
|
|
12
12
|
export declare function cellRangeToTsv(range: CellRange, cells: ReadonlyMap<string, string>): string;
|
package/dist/interaction.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { graphemeSegments, normalizeSearchText, normalizeWithMap, } from "./search-text.js";
|
|
2
|
+
export { normalizeSearchText } from "./search-text.js";
|
|
1
3
|
export function visibleRange(scrollOffset, viewportExtent, itemExtent, itemCount, overscan = 1) {
|
|
2
4
|
if (itemCount <= 0 || itemExtent <= 0)
|
|
3
5
|
return { start: 0, end: 0 };
|
|
@@ -41,10 +43,6 @@ export function findNormalizedMatches(text, query, pageIndex, caseSensitive = fa
|
|
|
41
43
|
}
|
|
42
44
|
return matches;
|
|
43
45
|
}
|
|
44
|
-
export function normalizeSearchText(text, caseSensitive = false) {
|
|
45
|
-
const normalized = text.normalize("NFKC");
|
|
46
|
-
return caseSensitive ? normalized : unicodeCaseFold(normalized);
|
|
47
|
-
}
|
|
48
46
|
/** Clamp a UTF-16 DOM offset to a grapheme boundary for native selection UI. */
|
|
49
47
|
export function snapGraphemeOffset(text, offset, edge) {
|
|
50
48
|
const safeOffset = Math.max(0, Math.min(text.length, Math.trunc(offset)));
|
|
@@ -126,46 +124,6 @@ export class LruMap {
|
|
|
126
124
|
this.#values.clear();
|
|
127
125
|
}
|
|
128
126
|
}
|
|
129
|
-
function normalizeWithMap(text, caseSensitive) {
|
|
130
|
-
const output = [];
|
|
131
|
-
const starts = [];
|
|
132
|
-
const ends = [];
|
|
133
|
-
const segments = graphemeSegments(text);
|
|
134
|
-
for (const segment of segments) {
|
|
135
|
-
const normalized = normalizeSearchText(segment.value, caseSensitive);
|
|
136
|
-
output.push(normalized);
|
|
137
|
-
for (let index = 0; index < normalized.length; index += 1) {
|
|
138
|
-
starts.push(segment.start);
|
|
139
|
-
ends.push(segment.end);
|
|
140
|
-
}
|
|
141
|
-
}
|
|
142
|
-
return { text: output.join(""), starts, ends };
|
|
143
|
-
}
|
|
144
|
-
function graphemeSegments(text) {
|
|
145
|
-
if (typeof Intl.Segmenter === "function") {
|
|
146
|
-
const segmenter = new Intl.Segmenter(undefined, {
|
|
147
|
-
granularity: "grapheme",
|
|
148
|
-
});
|
|
149
|
-
return [...segmenter.segment(text)].map((segment) => ({
|
|
150
|
-
value: segment.segment,
|
|
151
|
-
start: segment.index,
|
|
152
|
-
end: segment.index + segment.segment.length,
|
|
153
|
-
}));
|
|
154
|
-
}
|
|
155
|
-
const result = [];
|
|
156
|
-
let offset = 0;
|
|
157
|
-
for (const value of text) {
|
|
158
|
-
result.push({ value, start: offset, end: offset + value.length });
|
|
159
|
-
offset += value.length;
|
|
160
|
-
}
|
|
161
|
-
return result;
|
|
162
|
-
}
|
|
163
|
-
function unicodeCaseFold(text) {
|
|
164
|
-
return text
|
|
165
|
-
.toLocaleLowerCase("und")
|
|
166
|
-
.replaceAll("ß", "ss")
|
|
167
|
-
.replaceAll("ς", "σ");
|
|
168
|
-
}
|
|
169
127
|
function escapeTsv(value) {
|
|
170
128
|
return /[\t\n\r"]/.test(value) ? `"${value.replaceAll('"', '""')}"` : value;
|
|
171
129
|
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
export declare function normalizeSearchText(text: string, caseSensitive?: boolean): string;
|
|
2
|
+
export declare function normalizeWithMap(text: string, caseSensitive: boolean): {
|
|
3
|
+
text: string;
|
|
4
|
+
starts: number[];
|
|
5
|
+
ends: number[];
|
|
6
|
+
};
|
|
7
|
+
export declare function graphemeSegments(text: string): readonly {
|
|
8
|
+
value: string;
|
|
9
|
+
start: number;
|
|
10
|
+
end: number;
|
|
11
|
+
}[];
|