@readium/helpers 1.0.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -3
- package/dist/index.js +1 -1
- package/dist/textFragmentDirective.js +1 -0
- package/dist/vendor/text-fragments-polyfill/textFragmentGenerator.js +1 -0
- package/dist/vendor/text-fragments-polyfill/textFragmentMatcher.js +1 -0
- package/package.json +6 -6
- package/src/index.ts +3 -0
- package/src/textFragmentDirective.ts +44 -0
- package/src/vendor/text-fragments-polyfill/LICENSE +201 -0
- package/src/vendor/text-fragments-polyfill/README.MD +8 -0
- package/src/vendor/text-fragments-polyfill/textFragmentGenerator.ts +1744 -0
- package/src/vendor/text-fragments-polyfill/textFragmentMatcher.ts +883 -0
- package/types/src/index.d.ts +3 -0
- package/types/src/textFragmentDirective.d.ts +7 -0
- package/types/src/vendor/text-fragments-polyfill/textFragmentGenerator.d.ts +60 -0
- package/types/src/vendor/text-fragments-polyfill/textFragmentMatcher.d.ts +71 -0
|
@@ -0,0 +1,883 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Copyright 2020 Google LLC
|
|
3
|
+
*
|
|
4
|
+
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
* you may not use this file except in compliance with the License.
|
|
6
|
+
* You may obtain a copy of the License at
|
|
7
|
+
*
|
|
8
|
+
* https://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
*
|
|
10
|
+
* Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
* See the License for the specific language governing permissions and
|
|
14
|
+
* limitations under the License.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
// Ported from text-fragments-polyfill@6.7.0
|
|
18
|
+
// (https://github.com/GoogleChromeLabs/text-fragments-polyfill), trimmed to
|
|
19
|
+
// the directive-matching subset (dropped the live-highlighting/marking
|
|
20
|
+
// exports) and converted from JSDoc-typed JavaScript to TypeScript. See
|
|
21
|
+
// README.MD in this directory.
|
|
22
|
+
|
|
23
|
+
export interface TextFragment {
|
|
24
|
+
textStart: string;
|
|
25
|
+
textEnd?: string;
|
|
26
|
+
prefix?: string;
|
|
27
|
+
suffix?: string;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
interface BoundaryPoint {
|
|
31
|
+
node: Node;
|
|
32
|
+
offset: number;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
type ElementFilterFunction = (node: Node) => number;
|
|
36
|
+
|
|
37
|
+
// Block elements. elements of a text fragment cannot cross the boundaries of a
|
|
38
|
+
// block element. Source for the list:
|
|
39
|
+
// https://developer.mozilla.org/en-US/docs/Web/HTML/Block-level_elements#Elements
|
|
40
|
+
const BLOCK_ELEMENTS = [
|
|
41
|
+
'ADDRESS', 'ARTICLE', 'ASIDE', 'BLOCKQUOTE', 'BR', 'DETAILS',
|
|
42
|
+
'DIALOG', 'DD', 'DIV', 'DL', 'DT', 'FIELDSET',
|
|
43
|
+
'FIGCAPTION', 'FIGURE', 'FOOTER', 'FORM', 'H1', 'H2',
|
|
44
|
+
'H3', 'H4', 'H5', 'H6', 'HEADER', 'HGROUP',
|
|
45
|
+
'HR', 'LI', 'MAIN', 'NAV', 'OL', 'P',
|
|
46
|
+
'PRE', 'SECTION', 'TABLE', 'UL', 'TR', 'TH',
|
|
47
|
+
'TD', 'COLGROUP', 'COL', 'CAPTION', 'THEAD', 'TBODY',
|
|
48
|
+
'TFOOT',
|
|
49
|
+
];
|
|
50
|
+
|
|
51
|
+
// Characters that indicate a word boundary. Use the script
|
|
52
|
+
// tools/generate-boundary-regex.js if it's necessary to modify or regenerate
|
|
53
|
+
// this. Because it's a hefty regex, this should be used infrequently and only
|
|
54
|
+
// on single-character strings.
|
|
55
|
+
const BOUNDARY_CHARS =
|
|
56
|
+
/[\t-\r -#%-\*,-\/:;\?@\[-\]_\{\}\x85\xA0\xA1\xA7\xAB\xB6\xB7\xBB\xBF\u037E\u0387\u055A-\u055F\u0589\u058A\u05BE\u05C0\u05C3\u05C6\u05F3\u05F4\u0609\u060A\u060C\u060D\u061B\u061E\u061F\u066A-\u066D\u06D4\u0700-\u070D\u07F7-\u07F9\u0830-\u083E\u085E\u0964\u0965\u0970\u0AF0\u0DF4\u0E4F\u0E5A\u0E5B\u0F04-\u0F12\u0F14\u0F3A-\u0F3D\u0F85\u0FD0-\u0FD4\u0FD9\u0FDA\u104A-\u104F\u10FB\u1360-\u1368\u1400\u166D\u166E\u1680\u169B\u169C\u16EB-\u16ED\u1735\u1736\u17D4-\u17D6\u17D8-\u17DA\u1800-\u180A\u1944\u1945\u1A1E\u1A1F\u1AA0-\u1AA6\u1AA8-\u1AAD\u1B5A-\u1B60\u1BFC-\u1BFF\u1C3B-\u1C3F\u1C7E\u1C7F\u1CC0-\u1CC7\u1CD3\u2000-\u200A\u2010-\u2029\u202F-\u2043\u2045-\u2051\u2053-\u205F\u207D\u207E\u208D\u208E\u2308-\u230B\u2329\u232A\u2768-\u2775\u27C5\u27C6\u27E6-\u27EF\u2983-\u2998\u29D8-\u29DB\u29FC\u29FD\u2CF9-\u2CFC\u2CFE\u2CFF\u2D70\u2E00-\u2E2E\u2E30-\u2E44\u3000-\u3003\u3008-\u3011\u3014-\u301F\u3030\u303D\u30A0\u30FB\uA4FE\uA4FF\uA60D-\uA60F\uA673\uA67E\uA6F2-\uA6F7\uA874-\uA877\uA8CE\uA8CF\uA8F8-\uA8FA\uA8FC\uA92E\uA92F\uA95F\uA9C1-\uA9CD\uA9DE\uA9DF\uAA5C-\uAA5F\uAADE\uAADF\uAAF0\uAAF1\uABEB\uFD3E\uFD3F\uFE10-\uFE19\uFE30-\uFE52\uFE54-\uFE61\uFE63\uFE68\uFE6A\uFE6B\uFF01-\uFF03\uFF05-\uFF0A\uFF0C-\uFF0F\uFF1A\uFF1B\uFF1F\uFF20\uFF3B-\uFF3D\uFF3F\uFF5B\uFF5D\uFF5F-\uFF65]|\uD800[\uDD00-\uDD02\uDF9F\uDFD0]|\uD801\uDD6F|\uD802[\uDC57\uDD1F\uDD3F\uDE50-\uDE58\uDE7F\uDEF0-\uDEF6\uDF39-\uDF3F\uDF99-\uDF9C]|\uD804[\uDC47-\uDC4D\uDCBB\uDCBC\uDCBE-\uDCC1\uDD40-\uDD43\uDD74\uDD75\uDDC5-\uDDC9\uDDCD\uDDDB\uDDDD-\uDDDF\uDE38-\uDE3D\uDEA9]|\uD805[\uDC4B-\uDC4F\uDC5B\uDC5D\uDCC6\uDDC1-\uDDD7\uDE41-\uDE43\uDE60-\uDE6C\uDF3C-\uDF3E]|\uD807[\uDC41-\uDC45\uDC70\uDC71]|\uD809[\uDC70-\uDC74]|\uD81A[\uDE6E\uDE6F\uDEF5\uDF37-\uDF3B\uDF44]|\uD82F\uDC9F|\uD836[\uDE87-\uDE8B]|\uD83A[\uDD5E\uDD5F]/u;
|
|
57
|
+
|
|
58
|
+
// The same thing, but with a ^.
|
|
59
|
+
const NON_BOUNDARY_CHARS =
|
|
60
|
+
/[^\t-\r -#%-\*,-\/:;\?@\[-\]_\{\}\x85\xA0\xA1\xA7\xAB\xB6\xB7\xBB\xBF\u037E\u0387\u055A-\u055F\u0589\u058A\u05BE\u05C0\u05C3\u05C6\u05F3\u05F4\u0609\u060A\u060C\u060D\u061B\u061E\u061F\u066A-\u066D\u06D4\u0700-\u070D\u07F7-\u07F9\u0830-\u083E\u085E\u0964\u0965\u0970\u0AF0\u0DF4\u0E4F\u0E5A\u0E5B\u0F04-\u0F12\u0F14\u0F3A-\u0F3D\u0F85\u0FD0-\u0FD4\u0FD9\u0FDA\u104A-\u104F\u10FB\u1360-\u1368\u1400\u166D\u166E\u1680\u169B\u169C\u16EB-\u16ED\u1735\u1736\u17D4-\u17D6\u17D8-\u17DA\u1800-\u180A\u1944\u1945\u1A1E\u1A1F\u1AA0-\u1AA6\u1AA8-\u1AAD\u1B5A-\u1B60\u1BFC-\u1BFF\u1C3B-\u1C3F\u1C7E\u1C7F\u1CC0-\u1CC7\u1CD3\u2000-\u200A\u2010-\u2029\u202F-\u2043\u2045-\u2051\u2053-\u205F\u207D\u207E\u208D\u208E\u2308-\u230B\u2329\u232A\u2768-\u2775\u27C5\u27C6\u27E6-\u27EF\u2983-\u2998\u29D8-\u29DB\u29FC\u29FD\u2CF9-\u2CFC\u2CFE\u2CFF\u2D70\u2E00-\u2E2E\u2E30-\u2E44\u3000-\u3003\u3008-\u3011\u3014-\u301F\u3030\u303D\u30A0\u30FB\uA4FE\uA4FF\uA60D-\uA60F\uA673\uA67E\uA6F2-\uA6F7\uA874-\uA877\uA8CE\uA8CF\uA8F8-\uA8FA\uA8FC\uA92E\uA92F\uA95F\uA9C1-\uA9CD\uA9DE\uA9DF\uAA5C-\uAA5F\uAADE\uAADF\uAAF0\uAAF1\uABEB\uFD3E\uFD3F\uFE10-\uFE19\uFE30-\uFE52\uFE54-\uFE61\uFE63\uFE68\uFE6A\uFE6B\uFF01-\uFF03\uFF05-\uFF0A\uFF0C-\uFF0F\uFF1A\uFF1B\uFF1F\uFF20\uFF3B-\uFF3D\uFF3F\uFF5B\uFF5D\uFF5F-\uFF65]|\uD800[\uDD00-\uDD02\uDF9F\uDFD0]|\uD801\uDD6F|\uD802[\uDC57\uDD1F\uDD3F\uDE50-\uDE58\uDE7F\uDEF0-\uDEF6\uDF39-\uDF3F\uDF99-\uDF9C]|\uD804[\uDC47-\uDC4D\uDCBB\uDCBC\uDCBE-\uDCC1\uDD40-\uDD43\uDD74\uDD75\uDDC5-\uDDC9\uDDCD\uDDDB\uDDDD-\uDDDF\uDE38-\uDE3D\uDEA9]|\uD805[\uDC4B-\uDC4F\uDC5B\uDC5D\uDCC6\uDDC1-\uDDD7\uDE41-\uDE43\uDE60-\uDE6C\uDF3C-\uDF3E]|\uD807[\uDC41-\uDC45\uDC70\uDC71]|\uD809[\uDC70-\uDC74]|\uD81A[\uDE6E\uDE6F\uDEF5\uDF37-\uDF3B\uDF44]|\uD82F\uDC9F|\uD836[\uDE87-\uDE8B]|\uD83A[\uDD5E\uDD5F]/u;
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Searches the document for a given text fragment.
|
|
64
|
+
*
|
|
65
|
+
* @param textFragment - Text Fragment to highlight.
|
|
66
|
+
* @param documentToProcess - document where to extract and mark fragments in.
|
|
67
|
+
* @param root - the root element where to extract and mark fragments in.
|
|
68
|
+
* @return Zero or more ranges within the document corresponding to the
|
|
69
|
+
* fragment. If the fragment corresponds to more than one location in the
|
|
70
|
+
* document (i.e., is ambiguous) then the first two matches will be
|
|
71
|
+
* returned (regardless of how many more matches there may be in the
|
|
72
|
+
* document).
|
|
73
|
+
*/
|
|
74
|
+
export const processTextFragmentDirective =
|
|
75
|
+
(textFragment: TextFragment, documentToProcess: Document, root?: Node): Range[] => {
|
|
76
|
+
const results: Range[] = [];
|
|
77
|
+
|
|
78
|
+
const searchRange = documentToProcess.createRange();
|
|
79
|
+
searchRange.selectNodeContents(root ?? documentToProcess);
|
|
80
|
+
|
|
81
|
+
while (!searchRange.collapsed && results.length < 2) {
|
|
82
|
+
let potentialMatch: Range | undefined;
|
|
83
|
+
if (textFragment.prefix) {
|
|
84
|
+
const prefixMatch = findTextInRange(textFragment.prefix, searchRange);
|
|
85
|
+
if (prefixMatch == null) {
|
|
86
|
+
break;
|
|
87
|
+
}
|
|
88
|
+
// Future iterations, if necessary, should start after the first
|
|
89
|
+
// character of the prefix match.
|
|
90
|
+
advanceRangeStartPastOffset(
|
|
91
|
+
searchRange,
|
|
92
|
+
prefixMatch.startContainer,
|
|
93
|
+
prefixMatch.startOffset,
|
|
94
|
+
);
|
|
95
|
+
|
|
96
|
+
// The search space for textStart is everything after the prefix and
|
|
97
|
+
// before the end of the top-level search range, starting at the next
|
|
98
|
+
// non- whitespace position.
|
|
99
|
+
const matchRange = documentToProcess.createRange();
|
|
100
|
+
matchRange.setStart(prefixMatch.endContainer, prefixMatch.endOffset);
|
|
101
|
+
matchRange.setEnd(searchRange.endContainer, searchRange.endOffset);
|
|
102
|
+
|
|
103
|
+
advanceRangeStartToNonWhitespace(matchRange);
|
|
104
|
+
if (matchRange.collapsed) {
|
|
105
|
+
break;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
potentialMatch = findTextInRange(textFragment.textStart, matchRange);
|
|
109
|
+
// If textStart wasn't found anywhere in the matchRange, then there's
|
|
110
|
+
// no possible match and we can stop early.
|
|
111
|
+
if (potentialMatch == null) {
|
|
112
|
+
break;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// If potentialMatch is immediately after the prefix (i.e., its start
|
|
116
|
+
// equals matchRange's start), this is a candidate and we should keep
|
|
117
|
+
// going with this iteration. Otherwise, we'll need to find the next
|
|
118
|
+
// instance (if any) of the prefix.
|
|
119
|
+
if (potentialMatch.compareBoundaryPoints(
|
|
120
|
+
Range.START_TO_START,
|
|
121
|
+
matchRange,
|
|
122
|
+
) !== 0) {
|
|
123
|
+
continue;
|
|
124
|
+
}
|
|
125
|
+
} else {
|
|
126
|
+
// With no prefix, just look directly for textStart.
|
|
127
|
+
potentialMatch = findTextInRange(textFragment.textStart, searchRange);
|
|
128
|
+
if (potentialMatch == null) {
|
|
129
|
+
break;
|
|
130
|
+
}
|
|
131
|
+
advanceRangeStartPastOffset(
|
|
132
|
+
searchRange,
|
|
133
|
+
potentialMatch.startContainer,
|
|
134
|
+
potentialMatch.startOffset,
|
|
135
|
+
);
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
if (textFragment.textEnd) {
|
|
139
|
+
const textEndRange = documentToProcess.createRange();
|
|
140
|
+
textEndRange.setStart(
|
|
141
|
+
potentialMatch.endContainer, potentialMatch.endOffset);
|
|
142
|
+
textEndRange.setEnd(searchRange.endContainer, searchRange.endOffset);
|
|
143
|
+
|
|
144
|
+
// Keep track of matches of the end term followed by suffix term
|
|
145
|
+
// (if needed).
|
|
146
|
+
// If no matches are found then there's no point in keeping looking
|
|
147
|
+
// for matches of the start term after the current start term
|
|
148
|
+
// occurrence.
|
|
149
|
+
let matchFound = false;
|
|
150
|
+
|
|
151
|
+
// Search through the rest of the document to find a textEnd match.
|
|
152
|
+
// This may take multiple iterations if a suffix needs to be found.
|
|
153
|
+
while (!textEndRange.collapsed && results.length < 2) {
|
|
154
|
+
const textEndMatch =
|
|
155
|
+
findTextInRange(textFragment.textEnd, textEndRange);
|
|
156
|
+
if (textEndMatch == null) {
|
|
157
|
+
break;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
advanceRangeStartPastOffset(
|
|
161
|
+
textEndRange, textEndMatch.startContainer,
|
|
162
|
+
textEndMatch.startOffset);
|
|
163
|
+
|
|
164
|
+
potentialMatch.setEnd(
|
|
165
|
+
textEndMatch.endContainer, textEndMatch.endOffset);
|
|
166
|
+
|
|
167
|
+
if (textFragment.suffix) {
|
|
168
|
+
// If there's supposed to be a suffix, check if it appears after
|
|
169
|
+
// the textEnd we just found.
|
|
170
|
+
const suffixResult = checkSuffix(
|
|
171
|
+
textFragment.suffix, potentialMatch, searchRange,
|
|
172
|
+
documentToProcess);
|
|
173
|
+
if (suffixResult === CheckSuffixResult.NO_SUFFIX_MATCH) {
|
|
174
|
+
break;
|
|
175
|
+
} else if (suffixResult === CheckSuffixResult.SUFFIX_MATCH) {
|
|
176
|
+
matchFound = true;
|
|
177
|
+
results.push(potentialMatch.cloneRange());
|
|
178
|
+
continue;
|
|
179
|
+
} else if (suffixResult === CheckSuffixResult.MISPLACED_SUFFIX) {
|
|
180
|
+
continue;
|
|
181
|
+
}
|
|
182
|
+
} else {
|
|
183
|
+
// If we've found textEnd and there's no suffix, then it's a
|
|
184
|
+
// match!
|
|
185
|
+
matchFound = true;
|
|
186
|
+
results.push(potentialMatch.cloneRange());
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
// Stopping match search because suffix or textEnd are missing from
|
|
190
|
+
// the rest of the search space.
|
|
191
|
+
if (!matchFound) {
|
|
192
|
+
break;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
} else if (textFragment.suffix) {
|
|
196
|
+
// If there's no textEnd but there is a suffix, search for the suffix
|
|
197
|
+
// after potentialMatch
|
|
198
|
+
const suffixResult = checkSuffix(
|
|
199
|
+
textFragment.suffix, potentialMatch, searchRange,
|
|
200
|
+
documentToProcess);
|
|
201
|
+
if (suffixResult === CheckSuffixResult.NO_SUFFIX_MATCH) {
|
|
202
|
+
break;
|
|
203
|
+
} else if (suffixResult === CheckSuffixResult.SUFFIX_MATCH) {
|
|
204
|
+
results.push(potentialMatch.cloneRange());
|
|
205
|
+
advanceRangeStartPastOffset(
|
|
206
|
+
searchRange, searchRange.startContainer,
|
|
207
|
+
searchRange.startOffset);
|
|
208
|
+
continue;
|
|
209
|
+
} else if (suffixResult === CheckSuffixResult.MISPLACED_SUFFIX) {
|
|
210
|
+
continue;
|
|
211
|
+
}
|
|
212
|
+
} else {
|
|
213
|
+
results.push(potentialMatch.cloneRange());
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
return results;
|
|
217
|
+
};
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* Enum indicating the result of the checkSuffix function.
|
|
221
|
+
*/
|
|
222
|
+
const CheckSuffixResult = {
|
|
223
|
+
NO_SUFFIX_MATCH: 0, // Suffix wasn't found at all. Search should halt.
|
|
224
|
+
SUFFIX_MATCH: 1, // The suffix matches the expectation.
|
|
225
|
+
MISPLACED_SUFFIX: 2, // The suffix was found, but not in the right place.
|
|
226
|
+
} as const;
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* Checks to see if potentialMatch satisfies the suffix conditions of this
|
|
230
|
+
* Text Fragment.
|
|
231
|
+
* @param suffix - the suffix text to find
|
|
232
|
+
* @param potentialMatch - the Range containing the match text.
|
|
233
|
+
* @param searchRange - the Range in which to search for |suffix|.
|
|
234
|
+
* Regardless of the start boundary of this Range, nothing appearing before
|
|
235
|
+
* |potentialMatch| will be considered.
|
|
236
|
+
* @param documentToProcess - document where to extract and mark fragments in.
|
|
237
|
+
* @return enum value indicating that potentialMatch should be accepted, that
|
|
238
|
+
* the search should continue, or that the search should halt.
|
|
239
|
+
*/
|
|
240
|
+
const checkSuffix =
|
|
241
|
+
(suffix: string, potentialMatch: Range, searchRange: Range, documentToProcess: Document): number => {
|
|
242
|
+
const suffixRange = documentToProcess.createRange();
|
|
243
|
+
suffixRange.setStart(
|
|
244
|
+
potentialMatch.endContainer,
|
|
245
|
+
potentialMatch.endOffset,
|
|
246
|
+
);
|
|
247
|
+
suffixRange.setEnd(searchRange.endContainer, searchRange.endOffset);
|
|
248
|
+
advanceRangeStartToNonWhitespace(suffixRange);
|
|
249
|
+
|
|
250
|
+
const suffixMatch = findTextInRange(suffix, suffixRange);
|
|
251
|
+
// If suffix wasn't found anywhere in the suffixRange, then there's no
|
|
252
|
+
// possible match and we can stop early.
|
|
253
|
+
if (suffixMatch == null) {
|
|
254
|
+
return CheckSuffixResult.NO_SUFFIX_MATCH;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
// If suffixMatch is immediately after potentialMatch (i.e., its start
|
|
258
|
+
// equals suffixRange's start), this is a match. If not, we have to
|
|
259
|
+
// start over from the beginning.
|
|
260
|
+
if (suffixMatch.compareBoundaryPoints(
|
|
261
|
+
Range.START_TO_START, suffixRange) !== 0) {
|
|
262
|
+
return CheckSuffixResult.MISPLACED_SUFFIX;
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
return CheckSuffixResult.SUFFIX_MATCH;
|
|
266
|
+
};
|
|
267
|
+
|
|
268
|
+
/**
|
|
269
|
+
* Sets the start of |range| to be the first boundary point after |offset| in
|
|
270
|
+
* |node|--either at offset+1, or after the node.
|
|
271
|
+
* @param range - the range to mutate
|
|
272
|
+
* @param node - the node used to determine the new range start
|
|
273
|
+
* @param offset - the offset immediately before the desired new boundary point
|
|
274
|
+
*/
|
|
275
|
+
const advanceRangeStartPastOffset = (range: Range, node: Node, offset: number): void => {
|
|
276
|
+
try {
|
|
277
|
+
range.setStart(node, offset + 1);
|
|
278
|
+
} catch (err) {
|
|
279
|
+
range.setStartAfter(node);
|
|
280
|
+
}
|
|
281
|
+
};
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Modifies |range| to start at the next non-whitespace position.
|
|
285
|
+
* @param range - the range to mutate
|
|
286
|
+
*/
|
|
287
|
+
const advanceRangeStartToNonWhitespace = (range: Range): void => {
|
|
288
|
+
const walker = makeTextNodeWalker(range);
|
|
289
|
+
|
|
290
|
+
let node = walker.nextNode() as Text | null;
|
|
291
|
+
while (!range.collapsed && node != null) {
|
|
292
|
+
if (node !== range.startContainer) {
|
|
293
|
+
range.setStart(node, 0);
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
if (node.textContent!.length > range.startOffset) {
|
|
297
|
+
const firstChar = node.textContent![range.startOffset];
|
|
298
|
+
if (!firstChar.match(/\s/)) {
|
|
299
|
+
return;
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
try {
|
|
304
|
+
range.setStart(node, range.startOffset + 1);
|
|
305
|
+
} catch (err) {
|
|
306
|
+
node = walker.nextNode() as Text | null;
|
|
307
|
+
if (node == null) {
|
|
308
|
+
range.collapse();
|
|
309
|
+
} else {
|
|
310
|
+
range.setStart(node, 0);
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
};
|
|
315
|
+
|
|
316
|
+
/**
|
|
317
|
+
* Creates a TreeWalker that traverses a range and emits visible text nodes in
|
|
318
|
+
* the range.
|
|
319
|
+
* @param range - Range to be traversed by the walker
|
|
320
|
+
*/
|
|
321
|
+
const makeTextNodeWalker = (range: Range): TreeWalker => {
|
|
322
|
+
const doc = range.commonAncestorContainer.ownerDocument ?? (range.commonAncestorContainer as unknown as Document);
|
|
323
|
+
return doc.createTreeWalker(
|
|
324
|
+
range.commonAncestorContainer,
|
|
325
|
+
NodeFilter.SHOW_TEXT | NodeFilter.SHOW_ELEMENT,
|
|
326
|
+
{
|
|
327
|
+
acceptNode: (node: Node) => acceptTextNodeIfVisibleInRange(node, range),
|
|
328
|
+
},
|
|
329
|
+
);
|
|
330
|
+
};
|
|
331
|
+
|
|
332
|
+
/**
|
|
333
|
+
* Helper function to check if the element has attribute `hidden="until-found"`.
|
|
334
|
+
* @param elt - the element to evaluate
|
|
335
|
+
* @return true if the element has attribute `hidden="until-found"`
|
|
336
|
+
*/
|
|
337
|
+
const isHiddenUntilFound = (elt: Element): boolean => {
|
|
338
|
+
if ((elt as unknown as { hidden: unknown }).hidden === 'until-found') {
|
|
339
|
+
return true;
|
|
340
|
+
}
|
|
341
|
+
// Workaround for WebKit. See https://bugs.webkit.org/show_bug.cgi?id=238266
|
|
342
|
+
const attributes = elt.attributes as unknown as Record<string, { value: string } | undefined>;
|
|
343
|
+
if (attributes && attributes['hidden']) {
|
|
344
|
+
const value = attributes['hidden']!.value;
|
|
345
|
+
if (value === 'until-found') {
|
|
346
|
+
return true;
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
return false;
|
|
350
|
+
};
|
|
351
|
+
|
|
352
|
+
/**
|
|
353
|
+
* Helper function to calculate the visibility of a Node based on its CSS
|
|
354
|
+
* computed style. This function does not take into account the visibility of
|
|
355
|
+
* the node's ancestors so even if the node is visible according to its style
|
|
356
|
+
* it might not be visible on the page if one of its ancestors is not visible.
|
|
357
|
+
* @param node - the Node to evaluate
|
|
358
|
+
* @return true if the node is visible. A node will be visible if
|
|
359
|
+
* its computed style meets all of the following criteria:
|
|
360
|
+
* - non zero height, width, height and opacity
|
|
361
|
+
* - visibility not hidden
|
|
362
|
+
* - display not none
|
|
363
|
+
*/
|
|
364
|
+
const isNodeVisible = (node: Node): boolean => {
|
|
365
|
+
// Find an HTMLElement (this node or an ancestor) so we can check
|
|
366
|
+
// visibility.
|
|
367
|
+
let elt: Node | null = node;
|
|
368
|
+
while (elt != null && !(elt instanceof HTMLElement)) elt = elt.parentNode;
|
|
369
|
+
// A document parsed via DOMParser (as opposed to one attached to a
|
|
370
|
+
// real browsing context) has no defaultView, and therefore no
|
|
371
|
+
// rendering/layout at all — nothing to check visibility against, so
|
|
372
|
+
// every node in it counts as visible.
|
|
373
|
+
const win = elt?.ownerDocument?.defaultView;
|
|
374
|
+
if (elt != null && win != null) {
|
|
375
|
+
if (isHiddenUntilFound(elt)) {
|
|
376
|
+
return true;
|
|
377
|
+
}
|
|
378
|
+
const nodeStyle = win.getComputedStyle(elt);
|
|
379
|
+
// If the node is not rendered, just skip it.
|
|
380
|
+
if (nodeStyle.visibility === 'hidden' || nodeStyle.display === 'none' ||
|
|
381
|
+
parseInt(nodeStyle.height, 10) === 0 &&
|
|
382
|
+
nodeStyle.overflowY != 'visible' ||
|
|
383
|
+
parseInt(nodeStyle.width, 10) === 0 &&
|
|
384
|
+
nodeStyle.overflowX != 'visible' ||
|
|
385
|
+
parseInt(nodeStyle.opacity, 10) === 0) {
|
|
386
|
+
return false;
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
return true;
|
|
390
|
+
};
|
|
391
|
+
|
|
392
|
+
/**
|
|
393
|
+
* Filter function for use with TreeWalkers. Rejects nodes that aren't in the
|
|
394
|
+
* given range or aren't visible.
|
|
395
|
+
* @param node - the Node to evaluate
|
|
396
|
+
* @param range - the range in which node must fall. Optional; if null, the
|
|
397
|
+
* range check is skipped.
|
|
398
|
+
* @return FILTER_ACCEPT or FILTER_REJECT, to be passed along to a TreeWalker.
|
|
399
|
+
*/
|
|
400
|
+
const acceptNodeIfVisibleInRange = (node: Node, range?: Range): number => {
|
|
401
|
+
if (range != null && !range.intersectsNode(node))
|
|
402
|
+
return NodeFilter.FILTER_REJECT;
|
|
403
|
+
|
|
404
|
+
return isNodeVisible(node) ? NodeFilter.FILTER_ACCEPT :
|
|
405
|
+
NodeFilter.FILTER_REJECT;
|
|
406
|
+
};
|
|
407
|
+
|
|
408
|
+
/**
|
|
409
|
+
* Filter function for use with TreeWalkers. Accepts only visible text nodes
|
|
410
|
+
* that are in the given range. Other types of nodes visible in the given range
|
|
411
|
+
* are skipped so a TreeWalker using this filter function still visits text
|
|
412
|
+
* nodes in the node's subtree.
|
|
413
|
+
* @param node - the Node to evaluate
|
|
414
|
+
* @param range - the range in which node must fall. Optional; if null, the
|
|
415
|
+
* range check is skipped/
|
|
416
|
+
* @return NodeFilter value to be passed along to a TreeWalker.
|
|
417
|
+
* Values returned:
|
|
418
|
+
* - FILTER_REJECT: Node not in range or not visible.
|
|
419
|
+
* - FILTER_SKIP: Non Text Node visible and in range
|
|
420
|
+
* - FILTER_ACCEPT: Text Node visible and in range
|
|
421
|
+
*/
|
|
422
|
+
const acceptTextNodeIfVisibleInRange = (node: Node, range?: Range): number => {
|
|
423
|
+
if (range != null && !range.intersectsNode(node))
|
|
424
|
+
return NodeFilter.FILTER_REJECT;
|
|
425
|
+
|
|
426
|
+
if (!isNodeVisible(node)) {
|
|
427
|
+
return NodeFilter.FILTER_REJECT;
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
return node.nodeType === Node.TEXT_NODE ? NodeFilter.FILTER_ACCEPT :
|
|
431
|
+
NodeFilter.FILTER_SKIP;
|
|
432
|
+
};
|
|
433
|
+
|
|
434
|
+
/**
|
|
435
|
+
* Extracts all the text nodes within the given range.
|
|
436
|
+
* @param root - the root node in which to search
|
|
437
|
+
* @param range - a range restricting the scope of extraction
|
|
438
|
+
* @return a list of lists of text nodes, in document order. Lists represent
|
|
439
|
+
* block boundaries; i.e., two nodes appear in the same list iff there are
|
|
440
|
+
* no block element starts or ends in between them.
|
|
441
|
+
*/
|
|
442
|
+
const getAllTextNodes = (root: Node, range?: Range): Text[][] => {
|
|
443
|
+
const blocks: Text[][] = [];
|
|
444
|
+
let tmp: Text[] = [];
|
|
445
|
+
|
|
446
|
+
const nodes = Array.from(
|
|
447
|
+
getElementsIn(
|
|
448
|
+
root,
|
|
449
|
+
(node) => {
|
|
450
|
+
return acceptNodeIfVisibleInRange(node, range);
|
|
451
|
+
}),
|
|
452
|
+
);
|
|
453
|
+
|
|
454
|
+
for (const node of nodes) {
|
|
455
|
+
if (node.nodeType === Node.TEXT_NODE) {
|
|
456
|
+
tmp.push(node as Text);
|
|
457
|
+
} else if (
|
|
458
|
+
node instanceof HTMLElement &&
|
|
459
|
+
BLOCK_ELEMENTS.includes(node.tagName.toUpperCase()) && tmp.length > 0) {
|
|
460
|
+
// If this is a block element, the current set of text nodes in |tmp| is
|
|
461
|
+
// complete, and we need to move on to a new one.
|
|
462
|
+
blocks.push(tmp);
|
|
463
|
+
tmp = [];
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
if (tmp.length > 0) blocks.push(tmp);
|
|
467
|
+
|
|
468
|
+
return blocks;
|
|
469
|
+
};
|
|
470
|
+
|
|
471
|
+
/**
|
|
472
|
+
* Returns the textContent of all the textNodes and normalizes strings by
|
|
473
|
+
* replacing duplicated spaces with single space.
|
|
474
|
+
* @param nodes - TextNodes to get the textContent from.
|
|
475
|
+
* @param startOffset - Where to start in the first TextNode.
|
|
476
|
+
* @param endOffset - Where to end in the last TextNode.
|
|
477
|
+
* @return Entire text content of all the nodes, with spaces normalized.
|
|
478
|
+
*/
|
|
479
|
+
const getTextContent = (nodes: Text[], startOffset: number, endOffset?: number): string => {
|
|
480
|
+
let str = '';
|
|
481
|
+
if (nodes.length === 1) {
|
|
482
|
+
str = nodes[0]!.textContent!.substring(startOffset, endOffset);
|
|
483
|
+
} else {
|
|
484
|
+
str = nodes[0]!.textContent!.substring(startOffset) +
|
|
485
|
+
nodes.slice(1, -1).reduce((s, n) => s + n.textContent, '') +
|
|
486
|
+
nodes.slice(-1)[0]!.textContent!.substring(0, endOffset);
|
|
487
|
+
}
|
|
488
|
+
return str.replace(/[\t\n\r ]+/g, ' ');
|
|
489
|
+
};
|
|
490
|
+
|
|
491
|
+
/**
|
|
492
|
+
* Returns all nodes inside root using the provided filter.
|
|
493
|
+
* @param root - Node where to start the TreeWalker.
|
|
494
|
+
* @param filter - Filter provided to the TreeWalker's acceptNode filter.
|
|
495
|
+
* @yield All elements that were accepted by filter.
|
|
496
|
+
*/
|
|
497
|
+
function* getElementsIn(root: Node, filter: ElementFilterFunction): Generator<Node> {
|
|
498
|
+
const doc = root.ownerDocument ?? (root as unknown as Document);
|
|
499
|
+
const treeWalker = doc.createTreeWalker(
|
|
500
|
+
root,
|
|
501
|
+
NodeFilter.SHOW_ELEMENT | NodeFilter.SHOW_TEXT,
|
|
502
|
+
{acceptNode: filter},
|
|
503
|
+
);
|
|
504
|
+
|
|
505
|
+
const finishedSubtrees = new Set<Node>();
|
|
506
|
+
while (forwardTraverse(treeWalker, finishedSubtrees) !== null) {
|
|
507
|
+
yield treeWalker.currentNode;
|
|
508
|
+
}
|
|
509
|
+
}
|
|
510
|
+
|
|
511
|
+
/**
|
|
512
|
+
* Returns a range pointing to the first instance of |query| within |range|.
|
|
513
|
+
* @param query - the string to find
|
|
514
|
+
* @param range - the range in which to search
|
|
515
|
+
* @return The first found instance of |query| within |range|.
|
|
516
|
+
*/
|
|
517
|
+
const findTextInRange = (query: string, range: Range): Range | undefined => {
|
|
518
|
+
const textNodeLists = getAllTextNodes(range.commonAncestorContainer, range);
|
|
519
|
+
const segmenter = makeNewSegmenter(range.commonAncestorContainer.ownerDocument ?? undefined);
|
|
520
|
+
|
|
521
|
+
for (const list of textNodeLists) {
|
|
522
|
+
const found = findRangeFromNodeList(query, range, list, segmenter);
|
|
523
|
+
if (found !== undefined) return found;
|
|
524
|
+
}
|
|
525
|
+
return undefined;
|
|
526
|
+
};
|
|
527
|
+
|
|
528
|
+
/**
|
|
529
|
+
* Finds a range pointing to the first instance of |query| within |range|,
|
|
530
|
+
* searching over the text contained in a list |nodeList| of relevant textNodes.
|
|
531
|
+
* @param query - the string to find
|
|
532
|
+
* @param range - the range in which to search
|
|
533
|
+
* @param textNodes - the visible text nodes within |range|
|
|
534
|
+
* @param segmenter - a segmenter to be used for finding word boundaries, if
|
|
535
|
+
* supported
|
|
536
|
+
* @return the found range, or undefined if no such range could be found
|
|
537
|
+
*/
|
|
538
|
+
const findRangeFromNodeList = (query: string, range: Range, textNodes: Text[], segmenter?: Intl.Segmenter): Range | undefined => {
|
|
539
|
+
if (!query || !range || !(textNodes || []).length) return undefined;
|
|
540
|
+
const startOffset =
|
|
541
|
+
textNodes[0] === range.startContainer ? range.startOffset : 0;
|
|
542
|
+
const data =
|
|
543
|
+
normalizeString(getTextContent(textNodes, startOffset, undefined));
|
|
544
|
+
const normalizedQuery = normalizeString(query);
|
|
545
|
+
let searchStart = 0;
|
|
546
|
+
let start: BoundaryPoint | undefined;
|
|
547
|
+
let end: BoundaryPoint | undefined;
|
|
548
|
+
while (searchStart < data.length) {
|
|
549
|
+
const matchIndex = data.indexOf(normalizedQuery, searchStart);
|
|
550
|
+
if (matchIndex === -1) return undefined;
|
|
551
|
+
if (isWordBounded(data, matchIndex, normalizedQuery.length, segmenter)) {
|
|
552
|
+
const normalizedStartOffset =
|
|
553
|
+
normalizeString(textNodes[0]!.data.slice(0, startOffset)).length;
|
|
554
|
+
start = getBoundaryPointAtIndex(
|
|
555
|
+
normalizedStartOffset + matchIndex, textNodes, /* isEnd=*/ false);
|
|
556
|
+
end = getBoundaryPointAtIndex(
|
|
557
|
+
normalizedStartOffset + matchIndex + normalizedQuery.length,
|
|
558
|
+
textNodes,
|
|
559
|
+
/* isEnd=*/ true,
|
|
560
|
+
);
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
if (start != null && end != null) {
|
|
564
|
+
const foundRange = new Range();
|
|
565
|
+
foundRange.setStart(start.node, start.offset);
|
|
566
|
+
foundRange.setEnd(end.node, end.offset);
|
|
567
|
+
|
|
568
|
+
// Verify that |foundRange| is a subrange of |range|
|
|
569
|
+
if (range.compareBoundaryPoints(Range.START_TO_START, foundRange) <= 0 &&
|
|
570
|
+
range.compareBoundaryPoints(Range.END_TO_END, foundRange) >= 0) {
|
|
571
|
+
return foundRange;
|
|
572
|
+
}
|
|
573
|
+
}
|
|
574
|
+
searchStart = matchIndex + 1;
|
|
575
|
+
}
|
|
576
|
+
return undefined;
|
|
577
|
+
};
|
|
578
|
+
|
|
579
|
+
/**
|
|
580
|
+
* Generates a boundary point pointing to the given text position.
|
|
581
|
+
* @param index - the text offset indicating the start/end of a substring of
|
|
582
|
+
* the concatenated, normalized text in |textNodes|
|
|
583
|
+
* @param textNodes - the text Nodes whose contents make up the search space
|
|
584
|
+
* @param isEnd - indicates whether the offset is the start or end of the
|
|
585
|
+
* substring
|
|
586
|
+
* @return a boundary point suitable for setting as the start or end of a
|
|
587
|
+
* Range, or undefined if it couldn't be computed.
|
|
588
|
+
*/
|
|
589
|
+
const getBoundaryPointAtIndex = (index: number, textNodes: Text[], isEnd: boolean): BoundaryPoint | undefined => {
|
|
590
|
+
let counted = 0;
|
|
591
|
+
let normalizedData: string | undefined;
|
|
592
|
+
for (let i = 0; i < textNodes.length; i++) {
|
|
593
|
+
const node = textNodes[i]!;
|
|
594
|
+
if (!normalizedData) normalizedData = normalizeString(node.data);
|
|
595
|
+
let nodeEnd = counted + normalizedData.length;
|
|
596
|
+
if (isEnd) nodeEnd += 1;
|
|
597
|
+
if (nodeEnd > index) {
|
|
598
|
+
// |index| falls within this node, but we need to turn the offset in the
|
|
599
|
+
// normalized data into an offset in the real node data.
|
|
600
|
+
const normalizedOffset = index - counted;
|
|
601
|
+
let denormalizedOffset = Math.min(index - counted, node.data.length);
|
|
602
|
+
|
|
603
|
+
// Walk through the string until denormalizedOffset produces a substring
|
|
604
|
+
// that corresponds to the target from the normalized data.
|
|
605
|
+
const targetSubstring = isEnd ?
|
|
606
|
+
normalizedData.substring(0, normalizedOffset) :
|
|
607
|
+
normalizedData.substring(normalizedOffset);
|
|
608
|
+
|
|
609
|
+
let candidateSubstring = isEnd ?
|
|
610
|
+
normalizeString(node.data.substring(0, denormalizedOffset)) :
|
|
611
|
+
normalizeString(node.data.substring(denormalizedOffset));
|
|
612
|
+
|
|
613
|
+
// We will either lengthen or shrink the candidate string to approach the
|
|
614
|
+
// length of the target string. If we're looking for the start, adding 1
|
|
615
|
+
// makes the candidate shorter; if we're looking for the end, it makes the
|
|
616
|
+
// candidate longer.
|
|
617
|
+
const direction = (isEnd ? -1 : 1) *
|
|
618
|
+
(targetSubstring.length > candidateSubstring.length ? -1 : 1);
|
|
619
|
+
|
|
620
|
+
while (denormalizedOffset >= 0 &&
|
|
621
|
+
denormalizedOffset <= node.data.length) {
|
|
622
|
+
if (candidateSubstring.length === targetSubstring.length) {
|
|
623
|
+
return {node: node, offset: denormalizedOffset};
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
denormalizedOffset += direction;
|
|
627
|
+
|
|
628
|
+
candidateSubstring = isEnd ?
|
|
629
|
+
normalizeString(node.data.substring(0, denormalizedOffset)) :
|
|
630
|
+
normalizeString(node.data.substring(denormalizedOffset));
|
|
631
|
+
}
|
|
632
|
+
}
|
|
633
|
+
counted += normalizedData.length;
|
|
634
|
+
|
|
635
|
+
if (i + 1 < textNodes.length) {
|
|
636
|
+
// Edge case: if this node ends with a whitespace character and the next
|
|
637
|
+
// node starts with one, they'll be double-counted relative to the
|
|
638
|
+
// normalized version. Subtract 1 from |counted| to compensate.
|
|
639
|
+
const nextNormalizedData = normalizeString(textNodes[i + 1]!.data);
|
|
640
|
+
if (normalizedData.slice(-1) === ' ' &&
|
|
641
|
+
nextNormalizedData.slice(0, 1) === ' ') {
|
|
642
|
+
counted -= 1;
|
|
643
|
+
}
|
|
644
|
+
// Since we already normalized the next node's data, hold on to it for the
|
|
645
|
+
// next iteration.
|
|
646
|
+
normalizedData = nextNormalizedData;
|
|
647
|
+
}
|
|
648
|
+
}
|
|
649
|
+
return undefined;
|
|
650
|
+
};
|
|
651
|
+
|
|
652
|
+
/**
|
|
653
|
+
* Checks if a substring is word-bounded in the context of a longer string.
|
|
654
|
+
*
|
|
655
|
+
* If an Intl.Segmenter is provided for locale-specific segmenting, it will be
|
|
656
|
+
* used for this check. This is the most desirable option, but not supported in
|
|
657
|
+
* all browsers.
|
|
658
|
+
*
|
|
659
|
+
* If one is not provided, a heuristic will be applied,
|
|
660
|
+
* returning true iff:
|
|
661
|
+
* - startPos == 0 OR char before start is a boundary char, AND
|
|
662
|
+
* - length indicates end of string OR char after end is a boundary char
|
|
663
|
+
* Where boundary chars are whitespace/punctuation defined in the const above.
|
|
664
|
+
* This causes the known issue that some languages, notably Japanese, only match
|
|
665
|
+
* at the level of roughly a full clause or sentence, rather than a word.
|
|
666
|
+
*
|
|
667
|
+
* @param text - the text to search
|
|
668
|
+
* @param startPos - the index of the start of the substring
|
|
669
|
+
* @param length - the length of the substring
|
|
670
|
+
* @param segmenter - a segmenter to be used for finding word boundaries, if
|
|
671
|
+
* supported
|
|
672
|
+
* @return true iff startPos and length point to a word-bounded substring of
|
|
673
|
+
* |text|.
|
|
674
|
+
*/
|
|
675
|
+
const isWordBounded = (text: string, startPos: number, length: number, segmenter?: Intl.Segmenter): boolean => {
|
|
676
|
+
if (startPos < 0 || startPos >= text.length || length <= 0 ||
|
|
677
|
+
startPos + length > text.length) {
|
|
678
|
+
return false;
|
|
679
|
+
}
|
|
680
|
+
|
|
681
|
+
if (segmenter) {
|
|
682
|
+
// If the Intl.Segmenter API is available on this client, use it for more
|
|
683
|
+
// reliable word boundary checking.
|
|
684
|
+
|
|
685
|
+
const segments = segmenter.segment(text);
|
|
686
|
+
const startSegment = segments.containing(startPos);
|
|
687
|
+
if (!startSegment) return false;
|
|
688
|
+
// If the start index is inside a word segment but not the first character
|
|
689
|
+
// in that segment, it's not word-bounded. If it's not a word segment, then
|
|
690
|
+
// it's punctuation, etc., so that counts for word bounding.
|
|
691
|
+
if (startSegment.isWordLike && startSegment.index != startPos) return false;
|
|
692
|
+
|
|
693
|
+
// |endPos| points to the first character outside the target substring.
|
|
694
|
+
const endPos = startPos + length;
|
|
695
|
+
const endSegment = segments.containing(endPos);
|
|
696
|
+
|
|
697
|
+
// If there's no end segment found, it's because we're at the end of the
|
|
698
|
+
// text, which is a valid boundary. (Because of the preconditions we
|
|
699
|
+
// checked above, we know we aren't out of range.)
|
|
700
|
+
// If there's an end segment found but it's non-word-like, that's also OK,
|
|
701
|
+
// since punctuation and whitespace are acceptable boundaries.
|
|
702
|
+
// Lastly, if there's an end segment and it is word-like, then |endPos|
|
|
703
|
+
// needs to point to the start of that new word, or |endSegment.index|.
|
|
704
|
+
if (endSegment && endSegment.isWordLike && endSegment.index != endPos)
|
|
705
|
+
return false;
|
|
706
|
+
} else {
|
|
707
|
+
// We don't have Intl.Segmenter support, so fall back to checking whether or
|
|
708
|
+
// not the substring is flanked by boundary characters.
|
|
709
|
+
|
|
710
|
+
// If the first character is already a boundary, move it once.
|
|
711
|
+
if (text[startPos]!.match(BOUNDARY_CHARS)) {
|
|
712
|
+
++startPos;
|
|
713
|
+
--length;
|
|
714
|
+
if (!length) {
|
|
715
|
+
return false;
|
|
716
|
+
}
|
|
717
|
+
}
|
|
718
|
+
|
|
719
|
+
// If the last character is already a boundary, move it once.
|
|
720
|
+
if (text[startPos + length - 1]!.match(BOUNDARY_CHARS)) {
|
|
721
|
+
--length;
|
|
722
|
+
if (!length) {
|
|
723
|
+
return false;
|
|
724
|
+
}
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
if (startPos !== 0 && (!text[startPos - 1]!.match(BOUNDARY_CHARS)))
|
|
728
|
+
return false;
|
|
729
|
+
|
|
730
|
+
if (startPos + length !== text.length &&
|
|
731
|
+
!text[startPos + length]!.match(BOUNDARY_CHARS))
|
|
732
|
+
return false;
|
|
733
|
+
}
|
|
734
|
+
|
|
735
|
+
return true;
|
|
736
|
+
};
|
|
737
|
+
|
|
738
|
+
/**
|
|
739
|
+
* @param str - a string to be normalized
|
|
740
|
+
* @return a normalized version of |str| with all consecutive whitespace chars
|
|
741
|
+
* converted to a single ' ' and all diacriticals removed (e.g., 'é' ->
|
|
742
|
+
* 'e').
|
|
743
|
+
*/
|
|
744
|
+
const normalizeString = (str?: string): string => {
|
|
745
|
+
// First, decompose any characters with diacriticals. Then, turn all
|
|
746
|
+
// consecutive whitespace characters into a standard " ", and strip out
|
|
747
|
+
// anything in the Unicode U+0300..U+036F (Combining Diacritical Marks) range.
|
|
748
|
+
// This may change the length of the string.
|
|
749
|
+
return (str || '')
|
|
750
|
+
.normalize('NFKD')
|
|
751
|
+
.replace(/\s+/g, ' ')
|
|
752
|
+
.replace(/[\u0300-\u036f]/g, '')
|
|
753
|
+
.toLowerCase();
|
|
754
|
+
};
|
|
755
|
+
|
|
756
|
+
/**
|
|
757
|
+
* @param doc - document whose language governs segmentation.
|
|
758
|
+
* @return a segmenter object suitable for finding word boundaries. Returns
|
|
759
|
+
* undefined on browsers/platforms that do not yet support the
|
|
760
|
+
* Intl.Segmenter API.
|
|
761
|
+
*/
|
|
762
|
+
const makeNewSegmenter = (doc?: Document): Intl.Segmenter | undefined => {
|
|
763
|
+
if (Intl.Segmenter) {
|
|
764
|
+
// Falls back to the runtime's default locale (undefined) rather than
|
|
765
|
+
// upstream's navigator.language — doc.defaultView is null for a
|
|
766
|
+
// detached, DOMParser-produced document, which is the common case here.
|
|
767
|
+
const lang = doc?.documentElement?.lang || doc?.defaultView?.navigator?.language || undefined;
|
|
768
|
+
return new Intl.Segmenter(lang, {granularity: 'word'});
|
|
769
|
+
}
|
|
770
|
+
return undefined;
|
|
771
|
+
};
|
|
772
|
+
|
|
773
|
+
/**
|
|
774
|
+
* Performs traversal on a TreeWalker, visiting each subtree in document order.
|
|
775
|
+
* When visiting a subtree not already visited (its root not in finishedSubtrees
|
|
776
|
+
* ), first the root is emitted then the subtree is traversed, then the root is
|
|
777
|
+
* emitted again and then the next subtree in document order is visited.
|
|
778
|
+
*
|
|
779
|
+
* Subtree's roots are emitted twice to signal the beginning and ending of
|
|
780
|
+
* element nodes. This is useful for ensuring the ends of block boundaries are
|
|
781
|
+
* found.
|
|
782
|
+
* @param walker - the TreeWalker to be traversed
|
|
783
|
+
* @param finishedSubtrees - set of subtree roots already visited
|
|
784
|
+
* @return next node in the traversal
|
|
785
|
+
*/
|
|
786
|
+
const forwardTraverse = (walker: TreeWalker, finishedSubtrees: Set<Node>): Node | null => {
|
|
787
|
+
// If current node's subtree is not already finished
|
|
788
|
+
// try to go first down the subtree.
|
|
789
|
+
if (!finishedSubtrees.has(walker.currentNode)) {
|
|
790
|
+
const firstChild = walker.firstChild();
|
|
791
|
+
if (firstChild !== null) {
|
|
792
|
+
return firstChild;
|
|
793
|
+
}
|
|
794
|
+
}
|
|
795
|
+
|
|
796
|
+
// If no subtree go to next sibling if any.
|
|
797
|
+
const nextSibling = walker.nextSibling();
|
|
798
|
+
if (nextSibling !== null) {
|
|
799
|
+
return nextSibling;
|
|
800
|
+
}
|
|
801
|
+
|
|
802
|
+
// If no sibling go back to parent and mark it as finished.
|
|
803
|
+
const parent = walker.parentNode();
|
|
804
|
+
|
|
805
|
+
if (parent !== null) {
|
|
806
|
+
finishedSubtrees.add(parent);
|
|
807
|
+
}
|
|
808
|
+
|
|
809
|
+
return parent;
|
|
810
|
+
};
|
|
811
|
+
|
|
812
|
+
/**
|
|
813
|
+
* Performs backwards traversal on a TreeWalker, visiting each subtree in
|
|
814
|
+
* backwards document order. When visiting a subtree not already visited (its
|
|
815
|
+
* root not in finishedSubtrees ), first the root is emitted then the subtree is
|
|
816
|
+
* backward traversed, then the root is emitted again and then the previous
|
|
817
|
+
* subtree in document order is visited.
|
|
818
|
+
*
|
|
819
|
+
* Subtree's roots are emitted twice to signal the beginning and ending of
|
|
820
|
+
* element nodes. This is useful for ensuring block boundaries are found.
|
|
821
|
+
* @param walker - the TreeWalker to be traversed
|
|
822
|
+
* @param finishedSubtrees - set of subtree roots already visited
|
|
823
|
+
* @return next node in the backwards traversal
|
|
824
|
+
*/
|
|
825
|
+
const backwardTraverse = (walker: TreeWalker, finishedSubtrees: Set<Node>): Node | null => {
|
|
826
|
+
// If current node's subtree is not already finished
|
|
827
|
+
// try to go first down the subtree.
|
|
828
|
+
if (!finishedSubtrees.has(walker.currentNode)) {
|
|
829
|
+
const lastChild = walker.lastChild();
|
|
830
|
+
if (lastChild !== null) {
|
|
831
|
+
return lastChild;
|
|
832
|
+
}
|
|
833
|
+
}
|
|
834
|
+
|
|
835
|
+
// If no subtree go to previous sibling if any.
|
|
836
|
+
const previousSibling = walker.previousSibling();
|
|
837
|
+
if (previousSibling !== null) {
|
|
838
|
+
return previousSibling;
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
// If no sibling go back to parent and mark it as finished.
|
|
842
|
+
const parent = walker.parentNode();
|
|
843
|
+
|
|
844
|
+
if (parent !== null) {
|
|
845
|
+
finishedSubtrees.add(parent);
|
|
846
|
+
}
|
|
847
|
+
|
|
848
|
+
return parent;
|
|
849
|
+
};
|
|
850
|
+
|
|
851
|
+
/**
|
|
852
|
+
* Should not be referenced except in the /test directory.
|
|
853
|
+
*/
|
|
854
|
+
export const forTesting = {
|
|
855
|
+
advanceRangeStartPastOffset,
|
|
856
|
+
advanceRangeStartToNonWhitespace,
|
|
857
|
+
findRangeFromNodeList,
|
|
858
|
+
findTextInRange,
|
|
859
|
+
getBoundaryPointAtIndex,
|
|
860
|
+
isWordBounded,
|
|
861
|
+
makeNewSegmenter,
|
|
862
|
+
normalizeString,
|
|
863
|
+
forwardTraverse,
|
|
864
|
+
backwardTraverse,
|
|
865
|
+
getAllTextNodes,
|
|
866
|
+
acceptTextNodeIfVisibleInRange,
|
|
867
|
+
};
|
|
868
|
+
|
|
869
|
+
/**
|
|
870
|
+
* Should only be used by other files in this directory.
|
|
871
|
+
*/
|
|
872
|
+
export const internal = {
|
|
873
|
+
BLOCK_ELEMENTS,
|
|
874
|
+
BOUNDARY_CHARS,
|
|
875
|
+
NON_BOUNDARY_CHARS,
|
|
876
|
+
acceptNodeIfVisibleInRange,
|
|
877
|
+
normalizeString,
|
|
878
|
+
makeNewSegmenter,
|
|
879
|
+
forwardTraverse,
|
|
880
|
+
backwardTraverse,
|
|
881
|
+
makeTextNodeWalker,
|
|
882
|
+
isNodeVisible,
|
|
883
|
+
};
|