@readium/helpers 1.0.0 → 1.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,883 @@
1
+ /**
2
+ * Copyright 2020 Google LLC
3
+ *
4
+ * Licensed under the Apache License, Version 2.0 (the "License");
5
+ * you may not use this file except in compliance with the License.
6
+ * You may obtain a copy of the License at
7
+ *
8
+ * https://www.apache.org/licenses/LICENSE-2.0
9
+ *
10
+ * Unless required by applicable law or agreed to in writing, software
11
+ * distributed under the License is distributed on an "AS IS" BASIS,
12
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ * See the License for the specific language governing permissions and
14
+ * limitations under the License.
15
+ */
16
+
17
+ // Ported from text-fragments-polyfill@6.7.0
18
+ // (https://github.com/GoogleChromeLabs/text-fragments-polyfill), trimmed to
19
+ // the directive-matching subset (dropped the live-highlighting/marking
20
+ // exports) and converted from JSDoc-typed JavaScript to TypeScript. See
21
+ // README.MD in this directory.
22
+
23
+ export interface TextFragment {
24
+ textStart: string;
25
+ textEnd?: string;
26
+ prefix?: string;
27
+ suffix?: string;
28
+ }
29
+
30
+ interface BoundaryPoint {
31
+ node: Node;
32
+ offset: number;
33
+ }
34
+
35
+ type ElementFilterFunction = (node: Node) => number;
36
+
37
+ // Block elements. elements of a text fragment cannot cross the boundaries of a
38
+ // block element. Source for the list:
39
+ // https://developer.mozilla.org/en-US/docs/Web/HTML/Block-level_elements#Elements
40
+ const BLOCK_ELEMENTS = [
41
+ 'ADDRESS', 'ARTICLE', 'ASIDE', 'BLOCKQUOTE', 'BR', 'DETAILS',
42
+ 'DIALOG', 'DD', 'DIV', 'DL', 'DT', 'FIELDSET',
43
+ 'FIGCAPTION', 'FIGURE', 'FOOTER', 'FORM', 'H1', 'H2',
44
+ 'H3', 'H4', 'H5', 'H6', 'HEADER', 'HGROUP',
45
+ 'HR', 'LI', 'MAIN', 'NAV', 'OL', 'P',
46
+ 'PRE', 'SECTION', 'TABLE', 'UL', 'TR', 'TH',
47
+ 'TD', 'COLGROUP', 'COL', 'CAPTION', 'THEAD', 'TBODY',
48
+ 'TFOOT',
49
+ ];
50
+
51
+ // Characters that indicate a word boundary. Use the script
52
+ // tools/generate-boundary-regex.js if it's necessary to modify or regenerate
53
+ // this. Because it's a hefty regex, this should be used infrequently and only
54
+ // on single-character strings.
55
+ const BOUNDARY_CHARS =
56
+ /[\t-\r -#%-\*,-\/:;\?@\[-\]_\{\}\x85\xA0\xA1\xA7\xAB\xB6\xB7\xBB\xBF\u037E\u0387\u055A-\u055F\u0589\u058A\u05BE\u05C0\u05C3\u05C6\u05F3\u05F4\u0609\u060A\u060C\u060D\u061B\u061E\u061F\u066A-\u066D\u06D4\u0700-\u070D\u07F7-\u07F9\u0830-\u083E\u085E\u0964\u0965\u0970\u0AF0\u0DF4\u0E4F\u0E5A\u0E5B\u0F04-\u0F12\u0F14\u0F3A-\u0F3D\u0F85\u0FD0-\u0FD4\u0FD9\u0FDA\u104A-\u104F\u10FB\u1360-\u1368\u1400\u166D\u166E\u1680\u169B\u169C\u16EB-\u16ED\u1735\u1736\u17D4-\u17D6\u17D8-\u17DA\u1800-\u180A\u1944\u1945\u1A1E\u1A1F\u1AA0-\u1AA6\u1AA8-\u1AAD\u1B5A-\u1B60\u1BFC-\u1BFF\u1C3B-\u1C3F\u1C7E\u1C7F\u1CC0-\u1CC7\u1CD3\u2000-\u200A\u2010-\u2029\u202F-\u2043\u2045-\u2051\u2053-\u205F\u207D\u207E\u208D\u208E\u2308-\u230B\u2329\u232A\u2768-\u2775\u27C5\u27C6\u27E6-\u27EF\u2983-\u2998\u29D8-\u29DB\u29FC\u29FD\u2CF9-\u2CFC\u2CFE\u2CFF\u2D70\u2E00-\u2E2E\u2E30-\u2E44\u3000-\u3003\u3008-\u3011\u3014-\u301F\u3030\u303D\u30A0\u30FB\uA4FE\uA4FF\uA60D-\uA60F\uA673\uA67E\uA6F2-\uA6F7\uA874-\uA877\uA8CE\uA8CF\uA8F8-\uA8FA\uA8FC\uA92E\uA92F\uA95F\uA9C1-\uA9CD\uA9DE\uA9DF\uAA5C-\uAA5F\uAADE\uAADF\uAAF0\uAAF1\uABEB\uFD3E\uFD3F\uFE10-\uFE19\uFE30-\uFE52\uFE54-\uFE61\uFE63\uFE68\uFE6A\uFE6B\uFF01-\uFF03\uFF05-\uFF0A\uFF0C-\uFF0F\uFF1A\uFF1B\uFF1F\uFF20\uFF3B-\uFF3D\uFF3F\uFF5B\uFF5D\uFF5F-\uFF65]|\uD800[\uDD00-\uDD02\uDF9F\uDFD0]|\uD801\uDD6F|\uD802[\uDC57\uDD1F\uDD3F\uDE50-\uDE58\uDE7F\uDEF0-\uDEF6\uDF39-\uDF3F\uDF99-\uDF9C]|\uD804[\uDC47-\uDC4D\uDCBB\uDCBC\uDCBE-\uDCC1\uDD40-\uDD43\uDD74\uDD75\uDDC5-\uDDC9\uDDCD\uDDDB\uDDDD-\uDDDF\uDE38-\uDE3D\uDEA9]|\uD805[\uDC4B-\uDC4F\uDC5B\uDC5D\uDCC6\uDDC1-\uDDD7\uDE41-\uDE43\uDE60-\uDE6C\uDF3C-\uDF3E]|\uD807[\uDC41-\uDC45\uDC70\uDC71]|\uD809[\uDC70-\uDC74]|\uD81A[\uDE6E\uDE6F\uDEF5\uDF37-\uDF3B\uDF44]|\uD82F\uDC9F|\uD836[\uDE87-\uDE8B]|\uD83A[\uDD5E\uDD5F]/u;
57
+
58
+ // The same thing, but with a ^.
59
+ const NON_BOUNDARY_CHARS =
60
+ /[^\t-\r -#%-\*,-\/:;\?@\[-\]_\{\}\x85\xA0\xA1\xA7\xAB\xB6\xB7\xBB\xBF\u037E\u0387\u055A-\u055F\u0589\u058A\u05BE\u05C0\u05C3\u05C6\u05F3\u05F4\u0609\u060A\u060C\u060D\u061B\u061E\u061F\u066A-\u066D\u06D4\u0700-\u070D\u07F7-\u07F9\u0830-\u083E\u085E\u0964\u0965\u0970\u0AF0\u0DF4\u0E4F\u0E5A\u0E5B\u0F04-\u0F12\u0F14\u0F3A-\u0F3D\u0F85\u0FD0-\u0FD4\u0FD9\u0FDA\u104A-\u104F\u10FB\u1360-\u1368\u1400\u166D\u166E\u1680\u169B\u169C\u16EB-\u16ED\u1735\u1736\u17D4-\u17D6\u17D8-\u17DA\u1800-\u180A\u1944\u1945\u1A1E\u1A1F\u1AA0-\u1AA6\u1AA8-\u1AAD\u1B5A-\u1B60\u1BFC-\u1BFF\u1C3B-\u1C3F\u1C7E\u1C7F\u1CC0-\u1CC7\u1CD3\u2000-\u200A\u2010-\u2029\u202F-\u2043\u2045-\u2051\u2053-\u205F\u207D\u207E\u208D\u208E\u2308-\u230B\u2329\u232A\u2768-\u2775\u27C5\u27C6\u27E6-\u27EF\u2983-\u2998\u29D8-\u29DB\u29FC\u29FD\u2CF9-\u2CFC\u2CFE\u2CFF\u2D70\u2E00-\u2E2E\u2E30-\u2E44\u3000-\u3003\u3008-\u3011\u3014-\u301F\u3030\u303D\u30A0\u30FB\uA4FE\uA4FF\uA60D-\uA60F\uA673\uA67E\uA6F2-\uA6F7\uA874-\uA877\uA8CE\uA8CF\uA8F8-\uA8FA\uA8FC\uA92E\uA92F\uA95F\uA9C1-\uA9CD\uA9DE\uA9DF\uAA5C-\uAA5F\uAADE\uAADF\uAAF0\uAAF1\uABEB\uFD3E\uFD3F\uFE10-\uFE19\uFE30-\uFE52\uFE54-\uFE61\uFE63\uFE68\uFE6A\uFE6B\uFF01-\uFF03\uFF05-\uFF0A\uFF0C-\uFF0F\uFF1A\uFF1B\uFF1F\uFF20\uFF3B-\uFF3D\uFF3F\uFF5B\uFF5D\uFF5F-\uFF65]|\uD800[\uDD00-\uDD02\uDF9F\uDFD0]|\uD801\uDD6F|\uD802[\uDC57\uDD1F\uDD3F\uDE50-\uDE58\uDE7F\uDEF0-\uDEF6\uDF39-\uDF3F\uDF99-\uDF9C]|\uD804[\uDC47-\uDC4D\uDCBB\uDCBC\uDCBE-\uDCC1\uDD40-\uDD43\uDD74\uDD75\uDDC5-\uDDC9\uDDCD\uDDDB\uDDDD-\uDDDF\uDE38-\uDE3D\uDEA9]|\uD805[\uDC4B-\uDC4F\uDC5B\uDC5D\uDCC6\uDDC1-\uDDD7\uDE41-\uDE43\uDE60-\uDE6C\uDF3C-\uDF3E]|\uD807[\uDC41-\uDC45\uDC70\uDC71]|\uD809[\uDC70-\uDC74]|\uD81A[\uDE6E\uDE6F\uDEF5\uDF37-\uDF3B\uDF44]|\uD82F\uDC9F|\uD836[\uDE87-\uDE8B]|\uD83A[\uDD5E\uDD5F]/u;
61
+
62
+ /**
63
+ * Searches the document for a given text fragment.
64
+ *
65
+ * @param textFragment - Text Fragment to highlight.
66
+ * @param documentToProcess - document where to extract and mark fragments in.
67
+ * @param root - the root element where to extract and mark fragments in.
68
+ * @return Zero or more ranges within the document corresponding to the
69
+ * fragment. If the fragment corresponds to more than one location in the
70
+ * document (i.e., is ambiguous) then the first two matches will be
71
+ * returned (regardless of how many more matches there may be in the
72
+ * document).
73
+ */
74
+ export const processTextFragmentDirective =
75
+ (textFragment: TextFragment, documentToProcess: Document, root?: Node): Range[] => {
76
+ const results: Range[] = [];
77
+
78
+ const searchRange = documentToProcess.createRange();
79
+ searchRange.selectNodeContents(root ?? documentToProcess);
80
+
81
+ while (!searchRange.collapsed && results.length < 2) {
82
+ let potentialMatch: Range | undefined;
83
+ if (textFragment.prefix) {
84
+ const prefixMatch = findTextInRange(textFragment.prefix, searchRange);
85
+ if (prefixMatch == null) {
86
+ break;
87
+ }
88
+ // Future iterations, if necessary, should start after the first
89
+ // character of the prefix match.
90
+ advanceRangeStartPastOffset(
91
+ searchRange,
92
+ prefixMatch.startContainer,
93
+ prefixMatch.startOffset,
94
+ );
95
+
96
+ // The search space for textStart is everything after the prefix and
97
+ // before the end of the top-level search range, starting at the next
98
+ // non- whitespace position.
99
+ const matchRange = documentToProcess.createRange();
100
+ matchRange.setStart(prefixMatch.endContainer, prefixMatch.endOffset);
101
+ matchRange.setEnd(searchRange.endContainer, searchRange.endOffset);
102
+
103
+ advanceRangeStartToNonWhitespace(matchRange);
104
+ if (matchRange.collapsed) {
105
+ break;
106
+ }
107
+
108
+ potentialMatch = findTextInRange(textFragment.textStart, matchRange);
109
+ // If textStart wasn't found anywhere in the matchRange, then there's
110
+ // no possible match and we can stop early.
111
+ if (potentialMatch == null) {
112
+ break;
113
+ }
114
+
115
+ // If potentialMatch is immediately after the prefix (i.e., its start
116
+ // equals matchRange's start), this is a candidate and we should keep
117
+ // going with this iteration. Otherwise, we'll need to find the next
118
+ // instance (if any) of the prefix.
119
+ if (potentialMatch.compareBoundaryPoints(
120
+ Range.START_TO_START,
121
+ matchRange,
122
+ ) !== 0) {
123
+ continue;
124
+ }
125
+ } else {
126
+ // With no prefix, just look directly for textStart.
127
+ potentialMatch = findTextInRange(textFragment.textStart, searchRange);
128
+ if (potentialMatch == null) {
129
+ break;
130
+ }
131
+ advanceRangeStartPastOffset(
132
+ searchRange,
133
+ potentialMatch.startContainer,
134
+ potentialMatch.startOffset,
135
+ );
136
+ }
137
+
138
+ if (textFragment.textEnd) {
139
+ const textEndRange = documentToProcess.createRange();
140
+ textEndRange.setStart(
141
+ potentialMatch.endContainer, potentialMatch.endOffset);
142
+ textEndRange.setEnd(searchRange.endContainer, searchRange.endOffset);
143
+
144
+ // Keep track of matches of the end term followed by suffix term
145
+ // (if needed).
146
+ // If no matches are found then there's no point in keeping looking
147
+ // for matches of the start term after the current start term
148
+ // occurrence.
149
+ let matchFound = false;
150
+
151
+ // Search through the rest of the document to find a textEnd match.
152
+ // This may take multiple iterations if a suffix needs to be found.
153
+ while (!textEndRange.collapsed && results.length < 2) {
154
+ const textEndMatch =
155
+ findTextInRange(textFragment.textEnd, textEndRange);
156
+ if (textEndMatch == null) {
157
+ break;
158
+ }
159
+
160
+ advanceRangeStartPastOffset(
161
+ textEndRange, textEndMatch.startContainer,
162
+ textEndMatch.startOffset);
163
+
164
+ potentialMatch.setEnd(
165
+ textEndMatch.endContainer, textEndMatch.endOffset);
166
+
167
+ if (textFragment.suffix) {
168
+ // If there's supposed to be a suffix, check if it appears after
169
+ // the textEnd we just found.
170
+ const suffixResult = checkSuffix(
171
+ textFragment.suffix, potentialMatch, searchRange,
172
+ documentToProcess);
173
+ if (suffixResult === CheckSuffixResult.NO_SUFFIX_MATCH) {
174
+ break;
175
+ } else if (suffixResult === CheckSuffixResult.SUFFIX_MATCH) {
176
+ matchFound = true;
177
+ results.push(potentialMatch.cloneRange());
178
+ continue;
179
+ } else if (suffixResult === CheckSuffixResult.MISPLACED_SUFFIX) {
180
+ continue;
181
+ }
182
+ } else {
183
+ // If we've found textEnd and there's no suffix, then it's a
184
+ // match!
185
+ matchFound = true;
186
+ results.push(potentialMatch.cloneRange());
187
+ }
188
+ }
189
+ // Stopping match search because suffix or textEnd are missing from
190
+ // the rest of the search space.
191
+ if (!matchFound) {
192
+ break;
193
+ }
194
+
195
+ } else if (textFragment.suffix) {
196
+ // If there's no textEnd but there is a suffix, search for the suffix
197
+ // after potentialMatch
198
+ const suffixResult = checkSuffix(
199
+ textFragment.suffix, potentialMatch, searchRange,
200
+ documentToProcess);
201
+ if (suffixResult === CheckSuffixResult.NO_SUFFIX_MATCH) {
202
+ break;
203
+ } else if (suffixResult === CheckSuffixResult.SUFFIX_MATCH) {
204
+ results.push(potentialMatch.cloneRange());
205
+ advanceRangeStartPastOffset(
206
+ searchRange, searchRange.startContainer,
207
+ searchRange.startOffset);
208
+ continue;
209
+ } else if (suffixResult === CheckSuffixResult.MISPLACED_SUFFIX) {
210
+ continue;
211
+ }
212
+ } else {
213
+ results.push(potentialMatch.cloneRange());
214
+ }
215
+ }
216
+ return results;
217
+ };
218
+
219
+ /**
220
+ * Enum indicating the result of the checkSuffix function.
221
+ */
222
+ const CheckSuffixResult = {
223
+ NO_SUFFIX_MATCH: 0, // Suffix wasn't found at all. Search should halt.
224
+ SUFFIX_MATCH: 1, // The suffix matches the expectation.
225
+ MISPLACED_SUFFIX: 2, // The suffix was found, but not in the right place.
226
+ } as const;
227
+
228
+ /**
229
+ * Checks to see if potentialMatch satisfies the suffix conditions of this
230
+ * Text Fragment.
231
+ * @param suffix - the suffix text to find
232
+ * @param potentialMatch - the Range containing the match text.
233
+ * @param searchRange - the Range in which to search for |suffix|.
234
+ * Regardless of the start boundary of this Range, nothing appearing before
235
+ * |potentialMatch| will be considered.
236
+ * @param documentToProcess - document where to extract and mark fragments in.
237
+ * @return enum value indicating that potentialMatch should be accepted, that
238
+ * the search should continue, or that the search should halt.
239
+ */
240
+ const checkSuffix =
241
+ (suffix: string, potentialMatch: Range, searchRange: Range, documentToProcess: Document): number => {
242
+ const suffixRange = documentToProcess.createRange();
243
+ suffixRange.setStart(
244
+ potentialMatch.endContainer,
245
+ potentialMatch.endOffset,
246
+ );
247
+ suffixRange.setEnd(searchRange.endContainer, searchRange.endOffset);
248
+ advanceRangeStartToNonWhitespace(suffixRange);
249
+
250
+ const suffixMatch = findTextInRange(suffix, suffixRange);
251
+ // If suffix wasn't found anywhere in the suffixRange, then there's no
252
+ // possible match and we can stop early.
253
+ if (suffixMatch == null) {
254
+ return CheckSuffixResult.NO_SUFFIX_MATCH;
255
+ }
256
+
257
+ // If suffixMatch is immediately after potentialMatch (i.e., its start
258
+ // equals suffixRange's start), this is a match. If not, we have to
259
+ // start over from the beginning.
260
+ if (suffixMatch.compareBoundaryPoints(
261
+ Range.START_TO_START, suffixRange) !== 0) {
262
+ return CheckSuffixResult.MISPLACED_SUFFIX;
263
+ }
264
+
265
+ return CheckSuffixResult.SUFFIX_MATCH;
266
+ };
267
+
268
+ /**
269
+ * Sets the start of |range| to be the first boundary point after |offset| in
270
+ * |node|--either at offset+1, or after the node.
271
+ * @param range - the range to mutate
272
+ * @param node - the node used to determine the new range start
273
+ * @param offset - the offset immediately before the desired new boundary point
274
+ */
275
+ const advanceRangeStartPastOffset = (range: Range, node: Node, offset: number): void => {
276
+ try {
277
+ range.setStart(node, offset + 1);
278
+ } catch (err) {
279
+ range.setStartAfter(node);
280
+ }
281
+ };
282
+
283
+ /**
284
+ * Modifies |range| to start at the next non-whitespace position.
285
+ * @param range - the range to mutate
286
+ */
287
+ const advanceRangeStartToNonWhitespace = (range: Range): void => {
288
+ const walker = makeTextNodeWalker(range);
289
+
290
+ let node = walker.nextNode() as Text | null;
291
+ while (!range.collapsed && node != null) {
292
+ if (node !== range.startContainer) {
293
+ range.setStart(node, 0);
294
+ }
295
+
296
+ if (node.textContent!.length > range.startOffset) {
297
+ const firstChar = node.textContent![range.startOffset];
298
+ if (!firstChar.match(/\s/)) {
299
+ return;
300
+ }
301
+ }
302
+
303
+ try {
304
+ range.setStart(node, range.startOffset + 1);
305
+ } catch (err) {
306
+ node = walker.nextNode() as Text | null;
307
+ if (node == null) {
308
+ range.collapse();
309
+ } else {
310
+ range.setStart(node, 0);
311
+ }
312
+ }
313
+ }
314
+ };
315
+
316
+ /**
317
+ * Creates a TreeWalker that traverses a range and emits visible text nodes in
318
+ * the range.
319
+ * @param range - Range to be traversed by the walker
320
+ */
321
+ const makeTextNodeWalker = (range: Range): TreeWalker => {
322
+ const doc = range.commonAncestorContainer.ownerDocument ?? (range.commonAncestorContainer as unknown as Document);
323
+ return doc.createTreeWalker(
324
+ range.commonAncestorContainer,
325
+ NodeFilter.SHOW_TEXT | NodeFilter.SHOW_ELEMENT,
326
+ {
327
+ acceptNode: (node: Node) => acceptTextNodeIfVisibleInRange(node, range),
328
+ },
329
+ );
330
+ };
331
+
332
+ /**
333
+ * Helper function to check if the element has attribute `hidden="until-found"`.
334
+ * @param elt - the element to evaluate
335
+ * @return true if the element has attribute `hidden="until-found"`
336
+ */
337
+ const isHiddenUntilFound = (elt: Element): boolean => {
338
+ if ((elt as unknown as { hidden: unknown }).hidden === 'until-found') {
339
+ return true;
340
+ }
341
+ // Workaround for WebKit. See https://bugs.webkit.org/show_bug.cgi?id=238266
342
+ const attributes = elt.attributes as unknown as Record<string, { value: string } | undefined>;
343
+ if (attributes && attributes['hidden']) {
344
+ const value = attributes['hidden']!.value;
345
+ if (value === 'until-found') {
346
+ return true;
347
+ }
348
+ }
349
+ return false;
350
+ };
351
+
352
+ /**
353
+ * Helper function to calculate the visibility of a Node based on its CSS
354
+ * computed style. This function does not take into account the visibility of
355
+ * the node's ancestors so even if the node is visible according to its style
356
+ * it might not be visible on the page if one of its ancestors is not visible.
357
+ * @param node - the Node to evaluate
358
+ * @return true if the node is visible. A node will be visible if
359
+ * its computed style meets all of the following criteria:
360
+ * - non zero height, width, height and opacity
361
+ * - visibility not hidden
362
+ * - display not none
363
+ */
364
+ const isNodeVisible = (node: Node): boolean => {
365
+ // Find an HTMLElement (this node or an ancestor) so we can check
366
+ // visibility.
367
+ let elt: Node | null = node;
368
+ while (elt != null && !(elt instanceof HTMLElement)) elt = elt.parentNode;
369
+ // A document parsed via DOMParser (as opposed to one attached to a
370
+ // real browsing context) has no defaultView, and therefore no
371
+ // rendering/layout at all — nothing to check visibility against, so
372
+ // every node in it counts as visible.
373
+ const win = elt?.ownerDocument?.defaultView;
374
+ if (elt != null && win != null) {
375
+ if (isHiddenUntilFound(elt)) {
376
+ return true;
377
+ }
378
+ const nodeStyle = win.getComputedStyle(elt);
379
+ // If the node is not rendered, just skip it.
380
+ if (nodeStyle.visibility === 'hidden' || nodeStyle.display === 'none' ||
381
+ parseInt(nodeStyle.height, 10) === 0 &&
382
+ nodeStyle.overflowY != 'visible' ||
383
+ parseInt(nodeStyle.width, 10) === 0 &&
384
+ nodeStyle.overflowX != 'visible' ||
385
+ parseInt(nodeStyle.opacity, 10) === 0) {
386
+ return false;
387
+ }
388
+ }
389
+ return true;
390
+ };
391
+
392
+ /**
393
+ * Filter function for use with TreeWalkers. Rejects nodes that aren't in the
394
+ * given range or aren't visible.
395
+ * @param node - the Node to evaluate
396
+ * @param range - the range in which node must fall. Optional; if null, the
397
+ * range check is skipped.
398
+ * @return FILTER_ACCEPT or FILTER_REJECT, to be passed along to a TreeWalker.
399
+ */
400
+ const acceptNodeIfVisibleInRange = (node: Node, range?: Range): number => {
401
+ if (range != null && !range.intersectsNode(node))
402
+ return NodeFilter.FILTER_REJECT;
403
+
404
+ return isNodeVisible(node) ? NodeFilter.FILTER_ACCEPT :
405
+ NodeFilter.FILTER_REJECT;
406
+ };
407
+
408
+ /**
409
+ * Filter function for use with TreeWalkers. Accepts only visible text nodes
410
+ * that are in the given range. Other types of nodes visible in the given range
411
+ * are skipped so a TreeWalker using this filter function still visits text
412
+ * nodes in the node's subtree.
413
+ * @param node - the Node to evaluate
414
+ * @param range - the range in which node must fall. Optional; if null, the
415
+ * range check is skipped/
416
+ * @return NodeFilter value to be passed along to a TreeWalker.
417
+ * Values returned:
418
+ * - FILTER_REJECT: Node not in range or not visible.
419
+ * - FILTER_SKIP: Non Text Node visible and in range
420
+ * - FILTER_ACCEPT: Text Node visible and in range
421
+ */
422
+ const acceptTextNodeIfVisibleInRange = (node: Node, range?: Range): number => {
423
+ if (range != null && !range.intersectsNode(node))
424
+ return NodeFilter.FILTER_REJECT;
425
+
426
+ if (!isNodeVisible(node)) {
427
+ return NodeFilter.FILTER_REJECT;
428
+ }
429
+
430
+ return node.nodeType === Node.TEXT_NODE ? NodeFilter.FILTER_ACCEPT :
431
+ NodeFilter.FILTER_SKIP;
432
+ };
433
+
434
+ /**
435
+ * Extracts all the text nodes within the given range.
436
+ * @param root - the root node in which to search
437
+ * @param range - a range restricting the scope of extraction
438
+ * @return a list of lists of text nodes, in document order. Lists represent
439
+ * block boundaries; i.e., two nodes appear in the same list iff there are
440
+ * no block element starts or ends in between them.
441
+ */
442
+ const getAllTextNodes = (root: Node, range?: Range): Text[][] => {
443
+ const blocks: Text[][] = [];
444
+ let tmp: Text[] = [];
445
+
446
+ const nodes = Array.from(
447
+ getElementsIn(
448
+ root,
449
+ (node) => {
450
+ return acceptNodeIfVisibleInRange(node, range);
451
+ }),
452
+ );
453
+
454
+ for (const node of nodes) {
455
+ if (node.nodeType === Node.TEXT_NODE) {
456
+ tmp.push(node as Text);
457
+ } else if (
458
+ node instanceof HTMLElement &&
459
+ BLOCK_ELEMENTS.includes(node.tagName.toUpperCase()) && tmp.length > 0) {
460
+ // If this is a block element, the current set of text nodes in |tmp| is
461
+ // complete, and we need to move on to a new one.
462
+ blocks.push(tmp);
463
+ tmp = [];
464
+ }
465
+ }
466
+ if (tmp.length > 0) blocks.push(tmp);
467
+
468
+ return blocks;
469
+ };
470
+
471
+ /**
472
+ * Returns the textContent of all the textNodes and normalizes strings by
473
+ * replacing duplicated spaces with single space.
474
+ * @param nodes - TextNodes to get the textContent from.
475
+ * @param startOffset - Where to start in the first TextNode.
476
+ * @param endOffset - Where to end in the last TextNode.
477
+ * @return Entire text content of all the nodes, with spaces normalized.
478
+ */
479
+ const getTextContent = (nodes: Text[], startOffset: number, endOffset?: number): string => {
480
+ let str = '';
481
+ if (nodes.length === 1) {
482
+ str = nodes[0]!.textContent!.substring(startOffset, endOffset);
483
+ } else {
484
+ str = nodes[0]!.textContent!.substring(startOffset) +
485
+ nodes.slice(1, -1).reduce((s, n) => s + n.textContent, '') +
486
+ nodes.slice(-1)[0]!.textContent!.substring(0, endOffset);
487
+ }
488
+ return str.replace(/[\t\n\r ]+/g, ' ');
489
+ };
490
+
491
+ /**
492
+ * Returns all nodes inside root using the provided filter.
493
+ * @param root - Node where to start the TreeWalker.
494
+ * @param filter - Filter provided to the TreeWalker's acceptNode filter.
495
+ * @yield All elements that were accepted by filter.
496
+ */
497
+ function* getElementsIn(root: Node, filter: ElementFilterFunction): Generator<Node> {
498
+ const doc = root.ownerDocument ?? (root as unknown as Document);
499
+ const treeWalker = doc.createTreeWalker(
500
+ root,
501
+ NodeFilter.SHOW_ELEMENT | NodeFilter.SHOW_TEXT,
502
+ {acceptNode: filter},
503
+ );
504
+
505
+ const finishedSubtrees = new Set<Node>();
506
+ while (forwardTraverse(treeWalker, finishedSubtrees) !== null) {
507
+ yield treeWalker.currentNode;
508
+ }
509
+ }
510
+
511
+ /**
512
+ * Returns a range pointing to the first instance of |query| within |range|.
513
+ * @param query - the string to find
514
+ * @param range - the range in which to search
515
+ * @return The first found instance of |query| within |range|.
516
+ */
517
+ const findTextInRange = (query: string, range: Range): Range | undefined => {
518
+ const textNodeLists = getAllTextNodes(range.commonAncestorContainer, range);
519
+ const segmenter = makeNewSegmenter(range.commonAncestorContainer.ownerDocument ?? undefined);
520
+
521
+ for (const list of textNodeLists) {
522
+ const found = findRangeFromNodeList(query, range, list, segmenter);
523
+ if (found !== undefined) return found;
524
+ }
525
+ return undefined;
526
+ };
527
+
528
+ /**
529
+ * Finds a range pointing to the first instance of |query| within |range|,
530
+ * searching over the text contained in a list |nodeList| of relevant textNodes.
531
+ * @param query - the string to find
532
+ * @param range - the range in which to search
533
+ * @param textNodes - the visible text nodes within |range|
534
+ * @param segmenter - a segmenter to be used for finding word boundaries, if
535
+ * supported
536
+ * @return the found range, or undefined if no such range could be found
537
+ */
538
+ const findRangeFromNodeList = (query: string, range: Range, textNodes: Text[], segmenter?: Intl.Segmenter): Range | undefined => {
539
+ if (!query || !range || !(textNodes || []).length) return undefined;
540
+ const startOffset =
541
+ textNodes[0] === range.startContainer ? range.startOffset : 0;
542
+ const data =
543
+ normalizeString(getTextContent(textNodes, startOffset, undefined));
544
+ const normalizedQuery = normalizeString(query);
545
+ let searchStart = 0;
546
+ let start: BoundaryPoint | undefined;
547
+ let end: BoundaryPoint | undefined;
548
+ while (searchStart < data.length) {
549
+ const matchIndex = data.indexOf(normalizedQuery, searchStart);
550
+ if (matchIndex === -1) return undefined;
551
+ if (isWordBounded(data, matchIndex, normalizedQuery.length, segmenter)) {
552
+ const normalizedStartOffset =
553
+ normalizeString(textNodes[0]!.data.slice(0, startOffset)).length;
554
+ start = getBoundaryPointAtIndex(
555
+ normalizedStartOffset + matchIndex, textNodes, /* isEnd=*/ false);
556
+ end = getBoundaryPointAtIndex(
557
+ normalizedStartOffset + matchIndex + normalizedQuery.length,
558
+ textNodes,
559
+ /* isEnd=*/ true,
560
+ );
561
+ }
562
+
563
+ if (start != null && end != null) {
564
+ const foundRange = new Range();
565
+ foundRange.setStart(start.node, start.offset);
566
+ foundRange.setEnd(end.node, end.offset);
567
+
568
+ // Verify that |foundRange| is a subrange of |range|
569
+ if (range.compareBoundaryPoints(Range.START_TO_START, foundRange) <= 0 &&
570
+ range.compareBoundaryPoints(Range.END_TO_END, foundRange) >= 0) {
571
+ return foundRange;
572
+ }
573
+ }
574
+ searchStart = matchIndex + 1;
575
+ }
576
+ return undefined;
577
+ };
578
+
579
+ /**
580
+ * Generates a boundary point pointing to the given text position.
581
+ * @param index - the text offset indicating the start/end of a substring of
582
+ * the concatenated, normalized text in |textNodes|
583
+ * @param textNodes - the text Nodes whose contents make up the search space
584
+ * @param isEnd - indicates whether the offset is the start or end of the
585
+ * substring
586
+ * @return a boundary point suitable for setting as the start or end of a
587
+ * Range, or undefined if it couldn't be computed.
588
+ */
589
+ const getBoundaryPointAtIndex = (index: number, textNodes: Text[], isEnd: boolean): BoundaryPoint | undefined => {
590
+ let counted = 0;
591
+ let normalizedData: string | undefined;
592
+ for (let i = 0; i < textNodes.length; i++) {
593
+ const node = textNodes[i]!;
594
+ if (!normalizedData) normalizedData = normalizeString(node.data);
595
+ let nodeEnd = counted + normalizedData.length;
596
+ if (isEnd) nodeEnd += 1;
597
+ if (nodeEnd > index) {
598
+ // |index| falls within this node, but we need to turn the offset in the
599
+ // normalized data into an offset in the real node data.
600
+ const normalizedOffset = index - counted;
601
+ let denormalizedOffset = Math.min(index - counted, node.data.length);
602
+
603
+ // Walk through the string until denormalizedOffset produces a substring
604
+ // that corresponds to the target from the normalized data.
605
+ const targetSubstring = isEnd ?
606
+ normalizedData.substring(0, normalizedOffset) :
607
+ normalizedData.substring(normalizedOffset);
608
+
609
+ let candidateSubstring = isEnd ?
610
+ normalizeString(node.data.substring(0, denormalizedOffset)) :
611
+ normalizeString(node.data.substring(denormalizedOffset));
612
+
613
+ // We will either lengthen or shrink the candidate string to approach the
614
+ // length of the target string. If we're looking for the start, adding 1
615
+ // makes the candidate shorter; if we're looking for the end, it makes the
616
+ // candidate longer.
617
+ const direction = (isEnd ? -1 : 1) *
618
+ (targetSubstring.length > candidateSubstring.length ? -1 : 1);
619
+
620
+ while (denormalizedOffset >= 0 &&
621
+ denormalizedOffset <= node.data.length) {
622
+ if (candidateSubstring.length === targetSubstring.length) {
623
+ return {node: node, offset: denormalizedOffset};
624
+ }
625
+
626
+ denormalizedOffset += direction;
627
+
628
+ candidateSubstring = isEnd ?
629
+ normalizeString(node.data.substring(0, denormalizedOffset)) :
630
+ normalizeString(node.data.substring(denormalizedOffset));
631
+ }
632
+ }
633
+ counted += normalizedData.length;
634
+
635
+ if (i + 1 < textNodes.length) {
636
+ // Edge case: if this node ends with a whitespace character and the next
637
+ // node starts with one, they'll be double-counted relative to the
638
+ // normalized version. Subtract 1 from |counted| to compensate.
639
+ const nextNormalizedData = normalizeString(textNodes[i + 1]!.data);
640
+ if (normalizedData.slice(-1) === ' ' &&
641
+ nextNormalizedData.slice(0, 1) === ' ') {
642
+ counted -= 1;
643
+ }
644
+ // Since we already normalized the next node's data, hold on to it for the
645
+ // next iteration.
646
+ normalizedData = nextNormalizedData;
647
+ }
648
+ }
649
+ return undefined;
650
+ };
651
+
652
+ /**
653
+ * Checks if a substring is word-bounded in the context of a longer string.
654
+ *
655
+ * If an Intl.Segmenter is provided for locale-specific segmenting, it will be
656
+ * used for this check. This is the most desirable option, but not supported in
657
+ * all browsers.
658
+ *
659
+ * If one is not provided, a heuristic will be applied,
660
+ * returning true iff:
661
+ * - startPos == 0 OR char before start is a boundary char, AND
662
+ * - length indicates end of string OR char after end is a boundary char
663
+ * Where boundary chars are whitespace/punctuation defined in the const above.
664
+ * This causes the known issue that some languages, notably Japanese, only match
665
+ * at the level of roughly a full clause or sentence, rather than a word.
666
+ *
667
+ * @param text - the text to search
668
+ * @param startPos - the index of the start of the substring
669
+ * @param length - the length of the substring
670
+ * @param segmenter - a segmenter to be used for finding word boundaries, if
671
+ * supported
672
+ * @return true iff startPos and length point to a word-bounded substring of
673
+ * |text|.
674
+ */
675
+ const isWordBounded = (text: string, startPos: number, length: number, segmenter?: Intl.Segmenter): boolean => {
676
+ if (startPos < 0 || startPos >= text.length || length <= 0 ||
677
+ startPos + length > text.length) {
678
+ return false;
679
+ }
680
+
681
+ if (segmenter) {
682
+ // If the Intl.Segmenter API is available on this client, use it for more
683
+ // reliable word boundary checking.
684
+
685
+ const segments = segmenter.segment(text);
686
+ const startSegment = segments.containing(startPos);
687
+ if (!startSegment) return false;
688
+ // If the start index is inside a word segment but not the first character
689
+ // in that segment, it's not word-bounded. If it's not a word segment, then
690
+ // it's punctuation, etc., so that counts for word bounding.
691
+ if (startSegment.isWordLike && startSegment.index != startPos) return false;
692
+
693
+ // |endPos| points to the first character outside the target substring.
694
+ const endPos = startPos + length;
695
+ const endSegment = segments.containing(endPos);
696
+
697
+ // If there's no end segment found, it's because we're at the end of the
698
+ // text, which is a valid boundary. (Because of the preconditions we
699
+ // checked above, we know we aren't out of range.)
700
+ // If there's an end segment found but it's non-word-like, that's also OK,
701
+ // since punctuation and whitespace are acceptable boundaries.
702
+ // Lastly, if there's an end segment and it is word-like, then |endPos|
703
+ // needs to point to the start of that new word, or |endSegment.index|.
704
+ if (endSegment && endSegment.isWordLike && endSegment.index != endPos)
705
+ return false;
706
+ } else {
707
+ // We don't have Intl.Segmenter support, so fall back to checking whether or
708
+ // not the substring is flanked by boundary characters.
709
+
710
+ // If the first character is already a boundary, move it once.
711
+ if (text[startPos]!.match(BOUNDARY_CHARS)) {
712
+ ++startPos;
713
+ --length;
714
+ if (!length) {
715
+ return false;
716
+ }
717
+ }
718
+
719
+ // If the last character is already a boundary, move it once.
720
+ if (text[startPos + length - 1]!.match(BOUNDARY_CHARS)) {
721
+ --length;
722
+ if (!length) {
723
+ return false;
724
+ }
725
+ }
726
+
727
+ if (startPos !== 0 && (!text[startPos - 1]!.match(BOUNDARY_CHARS)))
728
+ return false;
729
+
730
+ if (startPos + length !== text.length &&
731
+ !text[startPos + length]!.match(BOUNDARY_CHARS))
732
+ return false;
733
+ }
734
+
735
+ return true;
736
+ };
737
+
738
+ /**
739
+ * @param str - a string to be normalized
740
+ * @return a normalized version of |str| with all consecutive whitespace chars
741
+ * converted to a single ' ' and all diacriticals removed (e.g., 'é' ->
742
+ * 'e').
743
+ */
744
+ const normalizeString = (str?: string): string => {
745
+ // First, decompose any characters with diacriticals. Then, turn all
746
+ // consecutive whitespace characters into a standard " ", and strip out
747
+ // anything in the Unicode U+0300..U+036F (Combining Diacritical Marks) range.
748
+ // This may change the length of the string.
749
+ return (str || '')
750
+ .normalize('NFKD')
751
+ .replace(/\s+/g, ' ')
752
+ .replace(/[\u0300-\u036f]/g, '')
753
+ .toLowerCase();
754
+ };
755
+
756
+ /**
757
+ * @param doc - document whose language governs segmentation.
758
+ * @return a segmenter object suitable for finding word boundaries. Returns
759
+ * undefined on browsers/platforms that do not yet support the
760
+ * Intl.Segmenter API.
761
+ */
762
+ const makeNewSegmenter = (doc?: Document): Intl.Segmenter | undefined => {
763
+ if (Intl.Segmenter) {
764
+ // Falls back to the runtime's default locale (undefined) rather than
765
+ // upstream's navigator.language — doc.defaultView is null for a
766
+ // detached, DOMParser-produced document, which is the common case here.
767
+ const lang = doc?.documentElement?.lang || doc?.defaultView?.navigator?.language || undefined;
768
+ return new Intl.Segmenter(lang, {granularity: 'word'});
769
+ }
770
+ return undefined;
771
+ };
772
+
773
+ /**
774
+ * Performs traversal on a TreeWalker, visiting each subtree in document order.
775
+ * When visiting a subtree not already visited (its root not in finishedSubtrees
776
+ * ), first the root is emitted then the subtree is traversed, then the root is
777
+ * emitted again and then the next subtree in document order is visited.
778
+ *
779
+ * Subtree's roots are emitted twice to signal the beginning and ending of
780
+ * element nodes. This is useful for ensuring the ends of block boundaries are
781
+ * found.
782
+ * @param walker - the TreeWalker to be traversed
783
+ * @param finishedSubtrees - set of subtree roots already visited
784
+ * @return next node in the traversal
785
+ */
786
+ const forwardTraverse = (walker: TreeWalker, finishedSubtrees: Set<Node>): Node | null => {
787
+ // If current node's subtree is not already finished
788
+ // try to go first down the subtree.
789
+ if (!finishedSubtrees.has(walker.currentNode)) {
790
+ const firstChild = walker.firstChild();
791
+ if (firstChild !== null) {
792
+ return firstChild;
793
+ }
794
+ }
795
+
796
+ // If no subtree go to next sibling if any.
797
+ const nextSibling = walker.nextSibling();
798
+ if (nextSibling !== null) {
799
+ return nextSibling;
800
+ }
801
+
802
+ // If no sibling go back to parent and mark it as finished.
803
+ const parent = walker.parentNode();
804
+
805
+ if (parent !== null) {
806
+ finishedSubtrees.add(parent);
807
+ }
808
+
809
+ return parent;
810
+ };
811
+
812
+ /**
813
+ * Performs backwards traversal on a TreeWalker, visiting each subtree in
814
+ * backwards document order. When visiting a subtree not already visited (its
815
+ * root not in finishedSubtrees ), first the root is emitted then the subtree is
816
+ * backward traversed, then the root is emitted again and then the previous
817
+ * subtree in document order is visited.
818
+ *
819
+ * Subtree's roots are emitted twice to signal the beginning and ending of
820
+ * element nodes. This is useful for ensuring block boundaries are found.
821
+ * @param walker - the TreeWalker to be traversed
822
+ * @param finishedSubtrees - set of subtree roots already visited
823
+ * @return next node in the backwards traversal
824
+ */
825
+ const backwardTraverse = (walker: TreeWalker, finishedSubtrees: Set<Node>): Node | null => {
826
+ // If current node's subtree is not already finished
827
+ // try to go first down the subtree.
828
+ if (!finishedSubtrees.has(walker.currentNode)) {
829
+ const lastChild = walker.lastChild();
830
+ if (lastChild !== null) {
831
+ return lastChild;
832
+ }
833
+ }
834
+
835
+ // If no subtree go to previous sibling if any.
836
+ const previousSibling = walker.previousSibling();
837
+ if (previousSibling !== null) {
838
+ return previousSibling;
839
+ }
840
+
841
+ // If no sibling go back to parent and mark it as finished.
842
+ const parent = walker.parentNode();
843
+
844
+ if (parent !== null) {
845
+ finishedSubtrees.add(parent);
846
+ }
847
+
848
+ return parent;
849
+ };
850
+
851
+ /**
852
+ * Should not be referenced except in the /test directory.
853
+ */
854
+ export const forTesting = {
855
+ advanceRangeStartPastOffset,
856
+ advanceRangeStartToNonWhitespace,
857
+ findRangeFromNodeList,
858
+ findTextInRange,
859
+ getBoundaryPointAtIndex,
860
+ isWordBounded,
861
+ makeNewSegmenter,
862
+ normalizeString,
863
+ forwardTraverse,
864
+ backwardTraverse,
865
+ getAllTextNodes,
866
+ acceptTextNodeIfVisibleInRange,
867
+ };
868
+
869
+ /**
870
+ * Should only be used by other files in this directory.
871
+ */
872
+ export const internal = {
873
+ BLOCK_ELEMENTS,
874
+ BOUNDARY_CHARS,
875
+ NON_BOUNDARY_CHARS,
876
+ acceptNodeIfVisibleInRange,
877
+ normalizeString,
878
+ makeNewSegmenter,
879
+ forwardTraverse,
880
+ backwardTraverse,
881
+ makeTextNodeWalker,
882
+ isNodeVisible,
883
+ };