web-doc 0.6.0 → 0.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/office.js +1 -1
- package/dist/contracts.d.ts +7 -7
- package/dist/fuzzy-alignment.d.ts +12 -0
- package/dist/fuzzy-alignment.js +63 -0
- package/dist/fuzzy-search.d.ts +7 -6
- package/dist/fuzzy-search.js +11 -14
- package/dist/interaction.d.ts +1 -1
- package/dist/interaction.js +2 -44
- package/dist/search-text.d.ts +11 -0
- package/dist/search-text.js +44 -0
- package/dist/workers/fuzzy-search-worker.js +106 -13
- package/package.json +2 -1
package/dist/adapters/office.js
CHANGED
|
@@ -341,7 +341,7 @@ export class OfficeDocumentAdapter {
|
|
|
341
341
|
async #loadPptx(data, options) {
|
|
342
342
|
if (this.#options.engines?.pptx)
|
|
343
343
|
return this.#options.engines.pptx(data, options);
|
|
344
|
-
const { PptxPresentation } = await import("@silurus/ooxml/pptx");
|
|
344
|
+
const { PptxPresentation } = await import("@silurus/ooxml-pptx/pptx");
|
|
345
345
|
return PptxPresentation.load(data, options);
|
|
346
346
|
}
|
|
347
347
|
}
|
package/dist/contracts.d.ts
CHANGED
|
@@ -219,9 +219,9 @@ export interface ViewerEventMap {
|
|
|
219
219
|
}
|
|
220
220
|
/**
|
|
221
221
|
* Tuning for the fuzzy fallback that runs when the exact search finds
|
|
222
|
-
* nothing.
|
|
223
|
-
*
|
|
224
|
-
* bullets, table separators and
|
|
222
|
+
* nothing. Fuse.js selects candidate pages; a contiguous edit alignment
|
|
223
|
+
* determines the original-text boundaries with a bounded edit budget. Spacing,
|
|
224
|
+
* line breaks, list bullets, table separators and punctuation may differ from the
|
|
225
225
|
* source, and every hit maps back to the verbatim page text.
|
|
226
226
|
*/
|
|
227
227
|
export interface FuzzySearchOptions {
|
|
@@ -231,15 +231,15 @@ export interface FuzzySearchOptions {
|
|
|
231
231
|
*/
|
|
232
232
|
readonly threshold?: number;
|
|
233
233
|
/**
|
|
234
|
-
* Highest Fuse.js score (`0` perfect, `1` no
|
|
235
|
-
*
|
|
236
|
-
* partly survives on a page, such as
|
|
234
|
+
* Highest Fuse.js score and whole-passage edit ratio (`0` perfect, `1` no
|
|
235
|
+
* resemblance) a match may have. Default `0.4`; raise it to accept a passage
|
|
236
|
+
* that only partly survives on a page, such as one spanning a page break.
|
|
237
237
|
*/
|
|
238
238
|
readonly maxScore?: number;
|
|
239
239
|
/**
|
|
240
240
|
* Query characters considered. The matcher's cost grows with the query and
|
|
241
241
|
* a passage is identified well before its end, so the default `600` keeps
|
|
242
|
-
*
|
|
242
|
+
* the work bounded; the highlight covers the matched prefix.
|
|
243
243
|
*/
|
|
244
244
|
readonly maxQueryLength?: number;
|
|
245
245
|
/** Characters of each page's text considered. Default `20000`. */
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fuse's indices describe character masks, not a contiguous edit alignment.
|
|
3
|
+
* Refine an accepted page with semi-global Levenshtein alignment: consume the
|
|
4
|
+
* whole query, allowing free text before and after one occurrence. Tracking
|
|
5
|
+
* the start alongside each cost needs O(query length) working memory rather
|
|
6
|
+
* than a page × query traceback matrix. Equal-cost occurrences prefer the
|
|
7
|
+
* earliest end, so an unmatched character after a passage cannot extend it.
|
|
8
|
+
*/
|
|
9
|
+
export declare function alignFuzzyPassage(text: string, query: string, caseSensitive: boolean, maxScore: number): {
|
|
10
|
+
start: number;
|
|
11
|
+
end: number;
|
|
12
|
+
} | undefined;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import { normalizeSearchText, normalizeWithMap } from "./search-text.js";
|
|
2
|
+
/**
|
|
3
|
+
* Fuse's indices describe character masks, not a contiguous edit alignment.
|
|
4
|
+
* Refine an accepted page with semi-global Levenshtein alignment: consume the
|
|
5
|
+
* whole query, allowing free text before and after one occurrence. Tracking
|
|
6
|
+
* the start alongside each cost needs O(query length) working memory rather
|
|
7
|
+
* than a page × query traceback matrix. Equal-cost occurrences prefer the
|
|
8
|
+
* earliest end, so an unmatched character after a passage cannot extend it.
|
|
9
|
+
*/
|
|
10
|
+
export function alignFuzzyPassage(text, query, caseSensitive, maxScore) {
|
|
11
|
+
const source = normalizeWithMap(text, caseSensitive);
|
|
12
|
+
const pattern = normalizeSearchText(query, caseSensitive);
|
|
13
|
+
const length = pattern.length;
|
|
14
|
+
if (!length || !source.text.length)
|
|
15
|
+
return undefined;
|
|
16
|
+
const costs = new Uint32Array(length + 1);
|
|
17
|
+
const starts = new Uint32Array(length + 1);
|
|
18
|
+
for (let i = 0; i <= length; i += 1)
|
|
19
|
+
costs[i] = i;
|
|
20
|
+
let bestCost = length;
|
|
21
|
+
let bestStart = 0;
|
|
22
|
+
let bestEnd = 0;
|
|
23
|
+
for (let end = 1; end <= source.text.length; end += 1) {
|
|
24
|
+
let diagonalCost = costs[0];
|
|
25
|
+
let diagonalStart = starts[0];
|
|
26
|
+
costs[0] = 0;
|
|
27
|
+
starts[0] = end;
|
|
28
|
+
for (let i = 1; i <= length; i += 1) {
|
|
29
|
+
const previousCost = costs[i];
|
|
30
|
+
const previousStart = starts[i];
|
|
31
|
+
let cost = diagonalCost + (pattern[i - 1] === source.text[end - 1] ? 0 : 1);
|
|
32
|
+
let start = diagonalStart;
|
|
33
|
+
// Prefer dropping an unmatched query character over substituting a
|
|
34
|
+
// neighbouring source character when both alignments cost the same.
|
|
35
|
+
const deletion = costs[i - 1] + 1;
|
|
36
|
+
if (deletion <= cost) {
|
|
37
|
+
cost = deletion;
|
|
38
|
+
start = starts[i - 1];
|
|
39
|
+
}
|
|
40
|
+
const insertion = previousCost + 1;
|
|
41
|
+
if (insertion < cost) {
|
|
42
|
+
cost = insertion;
|
|
43
|
+
start = previousStart;
|
|
44
|
+
}
|
|
45
|
+
costs[i] = cost;
|
|
46
|
+
starts[i] = start;
|
|
47
|
+
diagonalCost = previousCost;
|
|
48
|
+
diagonalStart = previousStart;
|
|
49
|
+
}
|
|
50
|
+
if (costs[length] < bestCost) {
|
|
51
|
+
bestCost = costs[length];
|
|
52
|
+
bestStart = starts[length];
|
|
53
|
+
bestEnd = end;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
// A page-level Fuse score alone can accept disconnected 32-character
|
|
57
|
+
// chunks. Require the single passage to meet the same score ceiling.
|
|
58
|
+
if (bestEnd <= bestStart || bestCost / length > maxScore)
|
|
59
|
+
return undefined;
|
|
60
|
+
const start = source.starts[bestStart];
|
|
61
|
+
const end = source.ends[bestEnd - 1];
|
|
62
|
+
return start === undefined || end === undefined ? undefined : { start, end };
|
|
63
|
+
}
|
package/dist/fuzzy-search.d.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import type { FuzzySearchOptions, SearchMatch } from "./contracts.js";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
* 32-character chunk)
|
|
3
|
+
* Fuse.js selects candidate pages (Bitap with a bounded edit budget per
|
|
4
|
+
* 32-character chunk); contiguous alignment determines highlight boundaries.
|
|
5
|
+
* This bridges the gaps NFKC + case folding leave open when
|
|
5
6
|
* a query was not copied verbatim from the document: AI-generated citations
|
|
6
7
|
* and OCR'd passages differ from the source in spacing, line breaks, list
|
|
7
8
|
* bullets, table separators and typographic punctuation. Every hit maps back
|
|
@@ -33,7 +34,7 @@ export interface FuzzyPageText {
|
|
|
33
34
|
* reuses the normalized records instead of rebuilding them: a citation
|
|
34
35
|
* lookup tries several anchors in a row, and each used to pay for the index
|
|
35
36
|
* again. A scan restricted to a page window reuses the same records through
|
|
36
|
-
* `Fuse.parseIndex`, so only
|
|
37
|
+
* `Fuse.parseIndex`, so only scoring and alignment run for those pages.
|
|
37
38
|
*/
|
|
38
39
|
export declare class FuzzyPageIndex {
|
|
39
40
|
#private;
|
|
@@ -44,9 +45,9 @@ export declare class FuzzyPageIndex {
|
|
|
44
45
|
get pageCount(): number;
|
|
45
46
|
/**
|
|
46
47
|
* One match per page whose text holds the query within the edit budget:
|
|
47
|
-
* the
|
|
48
|
-
*
|
|
49
|
-
* result is in page order either way.
|
|
48
|
+
* the best contiguous alignment, so neighbouring text and repeated
|
|
49
|
+
* occurrences cannot expand the highlight. `pageIndices` restricts the scan;
|
|
50
|
+
* the result is in page order either way.
|
|
50
51
|
*/
|
|
51
52
|
search(query: string, maxScore: number, pageIndices?: readonly number[]): readonly SearchMatch[];
|
|
52
53
|
}
|
package/dist/fuzzy-search.js
CHANGED
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
import Fuse, {} from "fuse.js";
|
|
2
|
+
import { alignFuzzyPassage } from "./fuzzy-alignment.js";
|
|
2
3
|
export const DEFAULT_FUZZY_SEARCH_OPTIONS = Object.freeze({
|
|
3
4
|
threshold: 0.3,
|
|
4
5
|
maxScore: 0.4,
|
|
5
6
|
// Bitap cost grows with the query; 600 characters still identifies a
|
|
6
|
-
// passage while
|
|
7
|
+
// passage while bounding both candidate scoring and alignment work.
|
|
7
8
|
maxQueryLength: 600,
|
|
8
9
|
maxPageTextLength: 20_000,
|
|
9
10
|
pagesPerBatch: 2,
|
|
@@ -45,7 +46,6 @@ function fuseOptions(threshold, caseSensitive) {
|
|
|
45
46
|
keys: ["text"],
|
|
46
47
|
isCaseSensitive: caseSensitive,
|
|
47
48
|
ignoreDiacritics: false,
|
|
48
|
-
includeMatches: true,
|
|
49
49
|
includeScore: true,
|
|
50
50
|
// A citation can sit anywhere on the page; Fuse's location bias would
|
|
51
51
|
// otherwise penalize matches far from the start of the text.
|
|
@@ -62,14 +62,16 @@ function fuseOptions(threshold, caseSensitive) {
|
|
|
62
62
|
* reuses the normalized records instead of rebuilding them: a citation
|
|
63
63
|
* lookup tries several anchors in a row, and each used to pay for the index
|
|
64
64
|
* again. A scan restricted to a page window reuses the same records through
|
|
65
|
-
* `Fuse.parseIndex`, so only
|
|
65
|
+
* `Fuse.parseIndex`, so only scoring and alignment run for those pages.
|
|
66
66
|
*/
|
|
67
67
|
export class FuzzyPageIndex {
|
|
68
|
+
#caseSensitive;
|
|
68
69
|
#pages;
|
|
69
70
|
#index;
|
|
70
71
|
#options;
|
|
71
72
|
#fuse;
|
|
72
73
|
constructor(pages, options, caseSensitive = false) {
|
|
74
|
+
this.#caseSensitive = caseSensitive;
|
|
73
75
|
this.#pages = pages.map((page) => ({
|
|
74
76
|
pageIndex: page.pageIndex,
|
|
75
77
|
text: page.text.slice(0, options.maxPageTextLength),
|
|
@@ -83,9 +85,9 @@ export class FuzzyPageIndex {
|
|
|
83
85
|
}
|
|
84
86
|
/**
|
|
85
87
|
* One match per page whose text holds the query within the edit budget:
|
|
86
|
-
* the
|
|
87
|
-
*
|
|
88
|
-
* result is in page order either way.
|
|
88
|
+
* the best contiguous alignment, so neighbouring text and repeated
|
|
89
|
+
* occurrences cannot expand the highlight. `pageIndices` restricts the scan;
|
|
90
|
+
* the result is in page order either way.
|
|
89
91
|
*/
|
|
90
92
|
search(query, maxScore, pageIndices) {
|
|
91
93
|
if (!query.trim())
|
|
@@ -95,15 +97,10 @@ export class FuzzyPageIndex {
|
|
|
95
97
|
for (const result of fuse.search(query)) {
|
|
96
98
|
if ((result.score ?? 1) > maxScore)
|
|
97
99
|
continue;
|
|
98
|
-
const
|
|
99
|
-
if (
|
|
100
|
+
const span = alignFuzzyPassage(result.item.text, query, this.#caseSensitive, maxScore);
|
|
101
|
+
if (!span)
|
|
100
102
|
continue;
|
|
101
|
-
|
|
102
|
-
let end = 0;
|
|
103
|
-
for (const [first, last] of indices) {
|
|
104
|
-
start = Math.min(start, first);
|
|
105
|
-
end = Math.max(end, last + 1);
|
|
106
|
-
}
|
|
103
|
+
const { start, end } = span;
|
|
107
104
|
const text = result.item.text.slice(start, end);
|
|
108
105
|
matches.push({ pageIndex: result.item.pageIndex, start, end, text });
|
|
109
106
|
}
|
package/dist/interaction.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
export { normalizeSearchText } from "./search-text.js";
|
|
1
2
|
import type { CellRange, SearchMatch } from "./contracts.js";
|
|
2
3
|
export interface VirtualRange {
|
|
3
4
|
readonly start: number;
|
|
@@ -6,7 +7,6 @@ export interface VirtualRange {
|
|
|
6
7
|
export declare function visibleRange(scrollOffset: number, viewportExtent: number, itemExtent: number, itemCount: number, overscan?: number): VirtualRange;
|
|
7
8
|
export declare function normalizeCellRange(range: CellRange): CellRange;
|
|
8
9
|
export declare function findNormalizedMatches(text: string, query: string, pageIndex: number, caseSensitive?: boolean): readonly SearchMatch[];
|
|
9
|
-
export declare function normalizeSearchText(text: string, caseSensitive?: boolean): string;
|
|
10
10
|
/** Clamp a UTF-16 DOM offset to a grapheme boundary for native selection UI. */
|
|
11
11
|
export declare function snapGraphemeOffset(text: string, offset: number, edge: "start" | "end"): number;
|
|
12
12
|
export declare function cellRangeToTsv(range: CellRange, cells: ReadonlyMap<string, string>): string;
|
package/dist/interaction.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { graphemeSegments, normalizeSearchText, normalizeWithMap, } from "./search-text.js";
|
|
2
|
+
export { normalizeSearchText } from "./search-text.js";
|
|
1
3
|
export function visibleRange(scrollOffset, viewportExtent, itemExtent, itemCount, overscan = 1) {
|
|
2
4
|
if (itemCount <= 0 || itemExtent <= 0)
|
|
3
5
|
return { start: 0, end: 0 };
|
|
@@ -41,10 +43,6 @@ export function findNormalizedMatches(text, query, pageIndex, caseSensitive = fa
|
|
|
41
43
|
}
|
|
42
44
|
return matches;
|
|
43
45
|
}
|
|
44
|
-
export function normalizeSearchText(text, caseSensitive = false) {
|
|
45
|
-
const normalized = text.normalize("NFKC");
|
|
46
|
-
return caseSensitive ? normalized : unicodeCaseFold(normalized);
|
|
47
|
-
}
|
|
48
46
|
/** Clamp a UTF-16 DOM offset to a grapheme boundary for native selection UI. */
|
|
49
47
|
export function snapGraphemeOffset(text, offset, edge) {
|
|
50
48
|
const safeOffset = Math.max(0, Math.min(text.length, Math.trunc(offset)));
|
|
@@ -126,46 +124,6 @@ export class LruMap {
|
|
|
126
124
|
this.#values.clear();
|
|
127
125
|
}
|
|
128
126
|
}
|
|
129
|
-
function normalizeWithMap(text, caseSensitive) {
|
|
130
|
-
const output = [];
|
|
131
|
-
const starts = [];
|
|
132
|
-
const ends = [];
|
|
133
|
-
const segments = graphemeSegments(text);
|
|
134
|
-
for (const segment of segments) {
|
|
135
|
-
const normalized = normalizeSearchText(segment.value, caseSensitive);
|
|
136
|
-
output.push(normalized);
|
|
137
|
-
for (let index = 0; index < normalized.length; index += 1) {
|
|
138
|
-
starts.push(segment.start);
|
|
139
|
-
ends.push(segment.end);
|
|
140
|
-
}
|
|
141
|
-
}
|
|
142
|
-
return { text: output.join(""), starts, ends };
|
|
143
|
-
}
|
|
144
|
-
function graphemeSegments(text) {
|
|
145
|
-
if (typeof Intl.Segmenter === "function") {
|
|
146
|
-
const segmenter = new Intl.Segmenter(undefined, {
|
|
147
|
-
granularity: "grapheme",
|
|
148
|
-
});
|
|
149
|
-
return [...segmenter.segment(text)].map((segment) => ({
|
|
150
|
-
value: segment.segment,
|
|
151
|
-
start: segment.index,
|
|
152
|
-
end: segment.index + segment.segment.length,
|
|
153
|
-
}));
|
|
154
|
-
}
|
|
155
|
-
const result = [];
|
|
156
|
-
let offset = 0;
|
|
157
|
-
for (const value of text) {
|
|
158
|
-
result.push({ value, start: offset, end: offset + value.length });
|
|
159
|
-
offset += value.length;
|
|
160
|
-
}
|
|
161
|
-
return result;
|
|
162
|
-
}
|
|
163
|
-
function unicodeCaseFold(text) {
|
|
164
|
-
return text
|
|
165
|
-
.toLocaleLowerCase("und")
|
|
166
|
-
.replaceAll("ß", "ss")
|
|
167
|
-
.replaceAll("ς", "σ");
|
|
168
|
-
}
|
|
169
127
|
function escapeTsv(value) {
|
|
170
128
|
return /[\t\n\r"]/.test(value) ? `"${value.replaceAll('"', '""')}"` : value;
|
|
171
129
|
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
export declare function normalizeSearchText(text: string, caseSensitive?: boolean): string;
|
|
2
|
+
export declare function normalizeWithMap(text: string, caseSensitive: boolean): {
|
|
3
|
+
text: string;
|
|
4
|
+
starts: number[];
|
|
5
|
+
ends: number[];
|
|
6
|
+
};
|
|
7
|
+
export declare function graphemeSegments(text: string): readonly {
|
|
8
|
+
value: string;
|
|
9
|
+
start: number;
|
|
10
|
+
end: number;
|
|
11
|
+
}[];
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
export function normalizeSearchText(text, caseSensitive = false) {
|
|
2
|
+
const normalized = text.normalize("NFKC");
|
|
3
|
+
return caseSensitive ? normalized : unicodeCaseFold(normalized);
|
|
4
|
+
}
|
|
5
|
+
export function normalizeWithMap(text, caseSensitive) {
|
|
6
|
+
const output = [];
|
|
7
|
+
const starts = [];
|
|
8
|
+
const ends = [];
|
|
9
|
+
const segments = graphemeSegments(text);
|
|
10
|
+
for (const segment of segments) {
|
|
11
|
+
const normalized = normalizeSearchText(segment.value, caseSensitive);
|
|
12
|
+
output.push(normalized);
|
|
13
|
+
for (let index = 0; index < normalized.length; index += 1) {
|
|
14
|
+
starts.push(segment.start);
|
|
15
|
+
ends.push(segment.end);
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
return { text: output.join(""), starts, ends };
|
|
19
|
+
}
|
|
20
|
+
export function graphemeSegments(text) {
|
|
21
|
+
if (typeof Intl.Segmenter === "function") {
|
|
22
|
+
const segmenter = new Intl.Segmenter(undefined, {
|
|
23
|
+
granularity: "grapheme",
|
|
24
|
+
});
|
|
25
|
+
return [...segmenter.segment(text)].map((segment) => ({
|
|
26
|
+
value: segment.segment,
|
|
27
|
+
start: segment.index,
|
|
28
|
+
end: segment.index + segment.segment.length,
|
|
29
|
+
}));
|
|
30
|
+
}
|
|
31
|
+
const result = [];
|
|
32
|
+
let offset = 0;
|
|
33
|
+
for (const value of text) {
|
|
34
|
+
result.push({ value, start: offset, end: offset + value.length });
|
|
35
|
+
offset += value.length;
|
|
36
|
+
}
|
|
37
|
+
return result;
|
|
38
|
+
}
|
|
39
|
+
function unicodeCaseFold(text) {
|
|
40
|
+
return text
|
|
41
|
+
.toLocaleLowerCase("und")
|
|
42
|
+
.replaceAll("ß", "ss")
|
|
43
|
+
.replaceAll("ς", "σ");
|
|
44
|
+
}
|
|
@@ -1545,12 +1545,104 @@ Fuse.use = function(...plugins) {
|
|
|
1545
1545
|
};
|
|
1546
1546
|
var entry_default = Fuse;
|
|
1547
1547
|
|
|
1548
|
+
// src/search-text.ts
|
|
1549
|
+
function normalizeSearchText(text, caseSensitive = false) {
|
|
1550
|
+
const normalized = text.normalize("NFKC");
|
|
1551
|
+
return caseSensitive ? normalized : unicodeCaseFold(normalized);
|
|
1552
|
+
}
|
|
1553
|
+
function normalizeWithMap(text, caseSensitive) {
|
|
1554
|
+
const output = [];
|
|
1555
|
+
const starts = [];
|
|
1556
|
+
const ends = [];
|
|
1557
|
+
const segments = graphemeSegments(text);
|
|
1558
|
+
for (const segment of segments) {
|
|
1559
|
+
const normalized = normalizeSearchText(segment.value, caseSensitive);
|
|
1560
|
+
output.push(normalized);
|
|
1561
|
+
for (let index = 0; index < normalized.length; index += 1) {
|
|
1562
|
+
starts.push(segment.start);
|
|
1563
|
+
ends.push(segment.end);
|
|
1564
|
+
}
|
|
1565
|
+
}
|
|
1566
|
+
return { text: output.join(""), starts, ends };
|
|
1567
|
+
}
|
|
1568
|
+
function graphemeSegments(text) {
|
|
1569
|
+
if (typeof Intl.Segmenter === "function") {
|
|
1570
|
+
const segmenter = new Intl.Segmenter(void 0, {
|
|
1571
|
+
granularity: "grapheme"
|
|
1572
|
+
});
|
|
1573
|
+
return [...segmenter.segment(text)].map((segment) => ({
|
|
1574
|
+
value: segment.segment,
|
|
1575
|
+
start: segment.index,
|
|
1576
|
+
end: segment.index + segment.segment.length
|
|
1577
|
+
}));
|
|
1578
|
+
}
|
|
1579
|
+
const result = [];
|
|
1580
|
+
let offset = 0;
|
|
1581
|
+
for (const value of text) {
|
|
1582
|
+
result.push({ value, start: offset, end: offset + value.length });
|
|
1583
|
+
offset += value.length;
|
|
1584
|
+
}
|
|
1585
|
+
return result;
|
|
1586
|
+
}
|
|
1587
|
+
function unicodeCaseFold(text) {
|
|
1588
|
+
return text.toLocaleLowerCase("und").replaceAll("\xDF", "ss").replaceAll("\u03C2", "\u03C3");
|
|
1589
|
+
}
|
|
1590
|
+
|
|
1591
|
+
// src/fuzzy-alignment.ts
|
|
1592
|
+
function alignFuzzyPassage(text, query, caseSensitive, maxScore) {
|
|
1593
|
+
const source = normalizeWithMap(text, caseSensitive);
|
|
1594
|
+
const pattern = normalizeSearchText(query, caseSensitive);
|
|
1595
|
+
const length = pattern.length;
|
|
1596
|
+
if (!length || !source.text.length) return void 0;
|
|
1597
|
+
const costs = new Uint32Array(length + 1);
|
|
1598
|
+
const starts = new Uint32Array(length + 1);
|
|
1599
|
+
for (let i = 0; i <= length; i += 1) costs[i] = i;
|
|
1600
|
+
let bestCost = length;
|
|
1601
|
+
let bestStart = 0;
|
|
1602
|
+
let bestEnd = 0;
|
|
1603
|
+
for (let end2 = 1; end2 <= source.text.length; end2 += 1) {
|
|
1604
|
+
let diagonalCost = costs[0];
|
|
1605
|
+
let diagonalStart = starts[0];
|
|
1606
|
+
costs[0] = 0;
|
|
1607
|
+
starts[0] = end2;
|
|
1608
|
+
for (let i = 1; i <= length; i += 1) {
|
|
1609
|
+
const previousCost = costs[i];
|
|
1610
|
+
const previousStart = starts[i];
|
|
1611
|
+
let cost = diagonalCost + (pattern[i - 1] === source.text[end2 - 1] ? 0 : 1);
|
|
1612
|
+
let start2 = diagonalStart;
|
|
1613
|
+
const deletion = costs[i - 1] + 1;
|
|
1614
|
+
if (deletion <= cost) {
|
|
1615
|
+
cost = deletion;
|
|
1616
|
+
start2 = starts[i - 1];
|
|
1617
|
+
}
|
|
1618
|
+
const insertion = previousCost + 1;
|
|
1619
|
+
if (insertion < cost) {
|
|
1620
|
+
cost = insertion;
|
|
1621
|
+
start2 = previousStart;
|
|
1622
|
+
}
|
|
1623
|
+
costs[i] = cost;
|
|
1624
|
+
starts[i] = start2;
|
|
1625
|
+
diagonalCost = previousCost;
|
|
1626
|
+
diagonalStart = previousStart;
|
|
1627
|
+
}
|
|
1628
|
+
if (costs[length] < bestCost) {
|
|
1629
|
+
bestCost = costs[length];
|
|
1630
|
+
bestStart = starts[length];
|
|
1631
|
+
bestEnd = end2;
|
|
1632
|
+
}
|
|
1633
|
+
}
|
|
1634
|
+
if (bestEnd <= bestStart || bestCost / length > maxScore) return void 0;
|
|
1635
|
+
const start = source.starts[bestStart];
|
|
1636
|
+
const end = source.ends[bestEnd - 1];
|
|
1637
|
+
return start === void 0 || end === void 0 ? void 0 : { start, end };
|
|
1638
|
+
}
|
|
1639
|
+
|
|
1548
1640
|
// src/fuzzy-search.ts
|
|
1549
1641
|
var DEFAULT_FUZZY_SEARCH_OPTIONS = Object.freeze({
|
|
1550
1642
|
threshold: 0.3,
|
|
1551
1643
|
maxScore: 0.4,
|
|
1552
1644
|
// Bitap cost grows with the query; 600 characters still identifies a
|
|
1553
|
-
// passage while
|
|
1645
|
+
// passage while bounding both candidate scoring and alignment work.
|
|
1554
1646
|
maxQueryLength: 600,
|
|
1555
1647
|
maxPageTextLength: 2e4,
|
|
1556
1648
|
pagesPerBatch: 2,
|
|
@@ -1562,7 +1654,6 @@ function fuseOptions(threshold, caseSensitive) {
|
|
|
1562
1654
|
keys: ["text"],
|
|
1563
1655
|
isCaseSensitive: caseSensitive,
|
|
1564
1656
|
ignoreDiacritics: false,
|
|
1565
|
-
includeMatches: true,
|
|
1566
1657
|
includeScore: true,
|
|
1567
1658
|
// A citation can sit anywhere on the page; Fuse's location bias would
|
|
1568
1659
|
// otherwise penalize matches far from the start of the text.
|
|
@@ -1575,11 +1666,13 @@ function fuseOptions(threshold, caseSensitive) {
|
|
|
1575
1666
|
};
|
|
1576
1667
|
}
|
|
1577
1668
|
var FuzzyPageIndex = class {
|
|
1669
|
+
#caseSensitive;
|
|
1578
1670
|
#pages;
|
|
1579
1671
|
#index;
|
|
1580
1672
|
#options;
|
|
1581
1673
|
#fuse;
|
|
1582
1674
|
constructor(pages, options, caseSensitive = false) {
|
|
1675
|
+
this.#caseSensitive = caseSensitive;
|
|
1583
1676
|
this.#pages = pages.map((page) => ({
|
|
1584
1677
|
pageIndex: page.pageIndex,
|
|
1585
1678
|
text: page.text.slice(0, options.maxPageTextLength)
|
|
@@ -1593,9 +1686,9 @@ var FuzzyPageIndex = class {
|
|
|
1593
1686
|
}
|
|
1594
1687
|
/**
|
|
1595
1688
|
* One match per page whose text holds the query within the edit budget:
|
|
1596
|
-
* the
|
|
1597
|
-
*
|
|
1598
|
-
* result is in page order either way.
|
|
1689
|
+
* the best contiguous alignment, so neighbouring text and repeated
|
|
1690
|
+
* occurrences cannot expand the highlight. `pageIndices` restricts the scan;
|
|
1691
|
+
* the result is in page order either way.
|
|
1599
1692
|
*/
|
|
1600
1693
|
search(query, maxScore, pageIndices) {
|
|
1601
1694
|
if (!query.trim()) return [];
|
|
@@ -1603,14 +1696,14 @@ var FuzzyPageIndex = class {
|
|
|
1603
1696
|
const matches = [];
|
|
1604
1697
|
for (const result of fuse.search(query)) {
|
|
1605
1698
|
if ((result.score ?? 1) > maxScore) continue;
|
|
1606
|
-
const
|
|
1607
|
-
|
|
1608
|
-
|
|
1609
|
-
|
|
1610
|
-
|
|
1611
|
-
|
|
1612
|
-
|
|
1613
|
-
}
|
|
1699
|
+
const span = alignFuzzyPassage(
|
|
1700
|
+
result.item.text,
|
|
1701
|
+
query,
|
|
1702
|
+
this.#caseSensitive,
|
|
1703
|
+
maxScore
|
|
1704
|
+
);
|
|
1705
|
+
if (!span) continue;
|
|
1706
|
+
const { start, end } = span;
|
|
1614
1707
|
const text = result.item.text.slice(start, end);
|
|
1615
1708
|
matches.push({ pageIndex: result.item.pageIndex, start, end, text });
|
|
1616
1709
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "web-doc",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.2",
|
|
4
4
|
"description": "web-doc — embeddable browser-only document viewer with Rust/WASM adapters (a fork of Zrimo)",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"document-viewer",
|
|
@@ -69,6 +69,7 @@
|
|
|
69
69
|
},
|
|
70
70
|
"dependencies": {
|
|
71
71
|
"@silurus/ooxml": "0.72.2",
|
|
72
|
+
"@silurus/ooxml-pptx": "npm:@silurus/ooxml@0.88.0",
|
|
72
73
|
"fuse.js": "7.5.0",
|
|
73
74
|
"pdfjs-dist": "6.2.108"
|
|
74
75
|
}
|