web-doc 0.5.0 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,44 @@
1
+ export function normalizeSearchText(text, caseSensitive = false) {
2
+ const normalized = text.normalize("NFKC");
3
+ return caseSensitive ? normalized : unicodeCaseFold(normalized);
4
+ }
5
+ export function normalizeWithMap(text, caseSensitive) {
6
+ const output = [];
7
+ const starts = [];
8
+ const ends = [];
9
+ const segments = graphemeSegments(text);
10
+ for (const segment of segments) {
11
+ const normalized = normalizeSearchText(segment.value, caseSensitive);
12
+ output.push(normalized);
13
+ for (let index = 0; index < normalized.length; index += 1) {
14
+ starts.push(segment.start);
15
+ ends.push(segment.end);
16
+ }
17
+ }
18
+ return { text: output.join(""), starts, ends };
19
+ }
20
+ export function graphemeSegments(text) {
21
+ if (typeof Intl.Segmenter === "function") {
22
+ const segmenter = new Intl.Segmenter(undefined, {
23
+ granularity: "grapheme",
24
+ });
25
+ return [...segmenter.segment(text)].map((segment) => ({
26
+ value: segment.segment,
27
+ start: segment.index,
28
+ end: segment.index + segment.segment.length,
29
+ }));
30
+ }
31
+ const result = [];
32
+ let offset = 0;
33
+ for (const value of text) {
34
+ result.push({ value, start: offset, end: offset + value.length });
35
+ offset += value.length;
36
+ }
37
+ return result;
38
+ }
39
+ function unicodeCaseFold(text) {
40
+ return text
41
+ .toLocaleLowerCase("und")
42
+ .replaceAll("ß", "ss")
43
+ .replaceAll("ς", "σ");
44
+ }
package/dist/viewer.js CHANGED
@@ -3,6 +3,7 @@ import { detectFormat } from "./detect.js";
3
3
  import { abortError, normalizeError, ViewerError } from "./errors.js";
4
4
  import { cellRangeToTsv, cellRangesToTsv, findNormalizedMatches, normalizeCellRange, } from "./interaction.js";
5
5
  import { findFuzzyPageMatches, nearestMatchIndex, pagesNearestFirst, resolveFuzzySearchOptions, } from "./fuzzy-search.js";
6
+ import { FuzzyWorkerClient } from "./fuzzy-worker-client.js";
6
7
  import { enforceContainerLimits, resolveLimits } from "./limits.js";
7
8
  import { loadDocumentSource } from "./source.js";
8
9
  import { AdaptiveViewport } from "./viewport.js";
@@ -31,6 +32,12 @@ export class DocumentViewer {
31
32
  #activeSearch;
32
33
  #selection = null;
33
34
  #searchResult = null;
35
+ /** Off-thread fuzzy matcher holding the current document's index. */
36
+ #fuzzyWorker;
37
+ /** Which pages/options the worker's index was built from; rebuilt on change. */
38
+ #fuzzyIndexKey;
39
+ /** Set once the worker failed to start, so the main thread takes over for good. */
40
+ #fuzzyWorkerUnavailable = false;
34
41
  #generation = 0;
35
42
  #searchGeneration = 0;
36
43
  #viewEventScheduled = false;
@@ -411,26 +418,37 @@ export class DocumentViewer {
411
418
  matches.push(...findNormalizedMatches(text, cleanQuery, pageIndex, caseSensitive));
412
419
  }
413
420
  if (matches.length === 0 && fuzzy) {
414
- // The page texts are already in memory, so the fallback is CPU only.
415
- // Pages are compared nearest to the hint first, a batch at a time,
416
- // and the scan stops at the first batch that holds the passage; a
417
- // yield between batches keeps a long document from freezing the UI.
418
421
  // With a hint the passage sits near it, so only that neighbourhood is
419
422
  // worth the fuzzy cost; without one every page is a candidate.
420
423
  const order = pagesNearestFirst(firstPage, lastPage, nearPage).slice(0, nearPage === undefined ? undefined : fuzzy.pageWindow);
421
- for (let offset = 0; offset < order.length && matches.length === 0; offset += fuzzy.pagesPerBatch) {
422
- if (offset > 0)
423
- await yieldToEventLoop();
424
- if (controller.signal.aborted)
425
- throw abortError();
426
- const batch = order
427
- .slice(offset, offset + fuzzy.pagesPerBatch)
428
- .map((pageIndex) => ({
429
- pageIndex,
430
- text: texts.get(pageIndex) ?? "",
431
- }));
432
- matches.push(...findFuzzyPageMatches(batch, cleanQuery, fuzzy, caseSensitive));
433
- }
424
+ const pattern = cleanQuery.slice(0, fuzzy.maxQueryLength);
425
+ const pages = [];
426
+ for (const [pageIndex, text] of texts)
427
+ pages.push({ pageIndex, text });
428
+ const offThread = fuzzy.worker
429
+ ? await this.#searchFuzzyInWorker(pages, pattern, order, fuzzy, caseSensitive, controller.signal)
430
+ : undefined;
431
+ if (offThread)
432
+ matches.push(...offThread);
433
+ else
434
+ for (let offset = 0; offset < order.length && matches.length === 0; offset += fuzzy.pagesPerBatch) {
435
+ // The page texts are already in memory, so the fallback is CPU
436
+ // only. Pages are compared nearest to the hint first, a batch at
437
+ // a time, and the scan stops at the first batch that holds the
438
+ // passage; a yield between batches keeps a long document from
439
+ // freezing the UI.
440
+ if (offset > 0)
441
+ await yieldToEventLoop();
442
+ if (controller.signal.aborted)
443
+ throw abortError();
444
+ const batch = order
445
+ .slice(offset, offset + fuzzy.pagesPerBatch)
446
+ .map((pageIndex) => ({
447
+ pageIndex,
448
+ text: texts.get(pageIndex) ?? "",
449
+ }));
450
+ matches.push(...findFuzzyPageMatches(batch, pattern, fuzzy, caseSensitive));
451
+ }
434
452
  if (matches.length > 0)
435
453
  strategy = "fuzzy";
436
454
  }
@@ -454,6 +472,56 @@ export class DocumentViewer {
454
472
  this.#activeSearch = undefined;
455
473
  }
456
474
  }
475
+ /**
476
+ * Fuzzy scan in the worker. The document's pages are indexed once per
477
+ * (range, options) and reused by every search until the document closes;
478
+ * `undefined` hands the scan back to the main thread when the worker is
479
+ * unavailable or failed, so a missing worker asset degrades to slowness,
480
+ * never to a lost match.
481
+ */
482
+ async #searchFuzzyInWorker(pages, pattern, pageIndices, fuzzy, caseSensitive, signal) {
483
+ if (this.#fuzzyWorkerUnavailable)
484
+ return undefined;
485
+ try {
486
+ this.#fuzzyWorker ??= FuzzyWorkerClient.create(this.#fuzzyWorkerUrl());
487
+ if (!this.#fuzzyWorker) {
488
+ this.#fuzzyWorkerUnavailable = true;
489
+ return undefined;
490
+ }
491
+ const key = `${pages.map((page) => page.pageIndex).join(",")}|${fuzzy.threshold}|${fuzzy.maxPageTextLength}|${caseSensitive}`;
492
+ if (this.#fuzzyIndexKey !== key) {
493
+ this.#fuzzyIndexKey = undefined;
494
+ await this.#fuzzyWorker.index(pages, {
495
+ threshold: fuzzy.threshold,
496
+ maxPageTextLength: fuzzy.maxPageTextLength,
497
+ caseSensitive,
498
+ });
499
+ this.#fuzzyIndexKey = key;
500
+ }
501
+ const matches = await this.#fuzzyWorker.search(pattern, fuzzy.maxScore, pageIndices);
502
+ if (signal.aborted)
503
+ throw abortError();
504
+ return matches;
505
+ }
506
+ catch (error) {
507
+ if (signal.aborted)
508
+ throw error;
509
+ this.#runtime.logger?.warn?.("Fuzzy search worker unavailable; matching on the main thread", { message: error instanceof Error ? error.message : String(error) });
510
+ this.#dropFuzzyWorker();
511
+ this.#fuzzyWorkerUnavailable = true;
512
+ return undefined;
513
+ }
514
+ }
515
+ #fuzzyWorkerUrl() {
516
+ return this.#runtime.assetBaseUrl
517
+ ? new URL("workers/fuzzy-search-worker.js", this.#runtime.assetBaseUrl)
518
+ : new URL("./workers/fuzzy-search-worker.js", import.meta.url);
519
+ }
520
+ #dropFuzzyWorker() {
521
+ this.#fuzzyWorker?.terminate();
522
+ this.#fuzzyWorker = undefined;
523
+ this.#fuzzyIndexKey = undefined;
524
+ }
457
525
  searchNext() {
458
526
  return this.#moveSearch(1);
459
527
  }
@@ -721,6 +789,7 @@ export class DocumentViewer {
721
789
  this.#activeSearch?.abort();
722
790
  this.#activeSearch = undefined;
723
791
  this.#searchResult = null;
792
+ this.#dropFuzzyWorker();
724
793
  this.#selection = null;
725
794
  this.#textMaps.clear();
726
795
  this.#textMapBytes = 0;