@readium/helpers 1.0.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1744 @@
1
+ /**
2
+ * Copyright 2020 Google LLC
3
+ *
4
+ * Licensed under the Apache License, Version 2.0 (the "License");
5
+ * you may not use this file except in compliance with the License.
6
+ * You may obtain a copy of the License at
7
+ *
8
+ * https://www.apache.org/licenses/LICENSE-2.0
9
+ *
10
+ * Unless required by applicable law or agreed to in writing, software
11
+ * distributed under the License is distributed on an "AS IS" BASIS,
12
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ * See the License for the specific language governing permissions and
14
+ * limitations under the License.
15
+ */
16
+
17
+ // Ported from text-fragments-polyfill@6.7.0
18
+ // (https://github.com/GoogleChromeLabs/text-fragments-polyfill), converted
19
+ // from JSDoc-typed JavaScript to TypeScript. See README.MD in this directory.
20
+
21
+ import * as fragments from './textFragmentMatcher.ts';
22
+ import type { TextFragment } from './textFragmentMatcher.ts';
23
+
24
+ const MAX_EXACT_MATCH_LENGTH = 300;
25
+ const MIN_LENGTH_WITHOUT_CONTEXT = 20;
26
+ const ITERATIONS_BEFORE_ADDING_CONTEXT = 1;
27
+ const WORDS_TO_ADD_FIRST_ITERATION = 3;
28
+ const WORDS_TO_ADD_SUBSEQUENT_ITERATIONS = 1;
29
+ const TRUNCATE_RANGE_CHECK_CHARS = 10000;
30
+ const MAX_DEPTH = 500;
31
+
32
+ // Desired max run time, in ms. Can be overwritten.
33
+ let timeoutDurationMs: number | null = 500;
34
+ let t0: number; // Start timestamp for fragment generation
35
+
36
+ class FragmentTimeoutError extends Error {
37
+ readonly isTimeout = true;
38
+ }
39
+
40
+ /**
41
+ * Allows overriding the max runtime to specify a different interval. Fragment
42
+ * generation will halt and throw an error after this amount of time.
43
+ * @param newTimeoutDurationMs - the desired timeout length, in ms.
44
+ */
45
+ export const setFragmentGenerationTimeout = (newTimeoutDurationMs: number | null): void => {
46
+ timeoutDurationMs = newTimeoutDurationMs;
47
+ };
48
+
49
+ /**
50
+ * Enum indicating the success, or failure reason, of generateFragment.
51
+ */
52
+ export const GenerateFragmentStatus = {
53
+ SUCCESS: 0, // A fragment was generated.
54
+ INVALID_SELECTION: 1, // The selection provided could not be used.
55
+ AMBIGUOUS: 2, // No unique fragment could be identified for this selection.
56
+ TIMEOUT: 3, // Computation could not complete in time.
57
+ EXECUTION_FAILED: 4, // An exception was raised during generation.
58
+ } as const;
59
+
60
+ export type GenerateFragmentStatusValue = typeof GenerateFragmentStatus[keyof typeof GenerateFragmentStatus];
61
+
62
+ export interface GenerateFragmentResult {
63
+ status: GenerateFragmentStatusValue;
64
+ fragment?: TextFragment;
65
+ }
66
+
67
+ /**
68
+ * Attempts to generate a fragment, suitable for formatting and including in a
69
+ * URL, which will highlight the given selection upon opening.
70
+ * @param selection - a Selection object, the result of window.getSelection
71
+ * @param startTime - the time when generation began, for timeout purposes.
72
+ * Defaults to current timestamp.
73
+ */
74
+ export const generateFragment = (selection: Selection, startTime: number = Date.now()): GenerateFragmentResult => {
75
+ return doGenerateFragment(selection, startTime);
76
+ };
77
+
78
+ /**
79
+ * Attampts to generate a fragment using a given range. @see {@link generateFragment}
80
+ * @param startTime - the time when generation began, for timeout purposes.
81
+ * Defaults to current timestamp.
82
+ */
83
+ export const generateFragmentFromRange =
84
+ (range: Range, startTime: number = Date.now()): GenerateFragmentResult => {
85
+ try {
86
+ return doGenerateFragmentFromRange(range, startTime);
87
+ } catch (err) {
88
+ if (err instanceof FragmentTimeoutError) {
89
+ return {status: GenerateFragmentStatus.TIMEOUT};
90
+ } else {
91
+ return {status: GenerateFragmentStatus.EXECUTION_FAILED};
92
+ }
93
+ }
94
+ };
95
+
96
+ /**
97
+ * Checks whether fragment generation can be attempted for a given range. This
98
+ * checks a handful of simple conditions: the range must be nonempty, not inside
99
+ * an <input>, etc. A true return is not a guarantee that fragment generation
100
+ * will succeed; instead, this is a way to quickly rule out generation in cases
101
+ * where a failure is predictable.
102
+ * @return true if fragment generation may proceed; false otherwise.
103
+ */
104
+ // Not called by textFragmentGenerator.ts's own callers — its iframe/window.top
105
+ // check assumes a live page, which doesn't apply to the detached documents
106
+ // this is typically used against. Left as upstream for reference/future use.
107
+ export const isValidRangeForFragmentGeneration = (range: Range): boolean => {
108
+ // Check that the range isn't just punctuation and whitespace. Only check the
109
+ // first |TRUNCATE_RANGE_CHECK_CHARS| to put an upper bound on runtime; ranges
110
+ // that start with (e.g.) thousands of periods should be rare.
111
+ // This also implicitly ensures the selection isn't in an input or textarea
112
+ // field, as document.selection contains an empty range in these cases.
113
+ if (!range.toString()
114
+ .substring(0, TRUNCATE_RANGE_CHECK_CHARS)
115
+ .match(fragments.internal.NON_BOUNDARY_CHARS)) {
116
+ return false;
117
+ }
118
+
119
+ // Check for iframe
120
+ try {
121
+ if ((range.startContainer.ownerDocument as Document).defaultView !== window.top) {
122
+ return false;
123
+ }
124
+ } catch {
125
+ // If accessing window.top throws an error, this is in a cross-origin
126
+ // iframe.
127
+ return false;
128
+ }
129
+
130
+ // Walk up the DOM to ensure that the range isn't inside an editable. Limit
131
+ // the search depth to |MAX_DEPTH| to constrain runtime.
132
+ let node: Node | null = range.commonAncestorContainer;
133
+ let numIterations = 0;
134
+ while (node) {
135
+ if (node.nodeType == Node.ELEMENT_NODE) {
136
+ const element = node as Element;
137
+ if (['TEXTAREA', 'INPUT'].includes(element.tagName.toUpperCase())) {
138
+ return false;
139
+ }
140
+
141
+ const editable = element.attributes.getNamedItem('contenteditable');
142
+ if (editable && editable.value !== 'false') {
143
+ return false;
144
+ }
145
+
146
+ // Cap the number of iterations at |MAX_PRECONDITION_DEPTH| to put an
147
+ // upper bound on runtime.
148
+ numIterations++;
149
+ if (numIterations >= MAX_DEPTH) {
150
+ return false;
151
+ }
152
+ }
153
+ node = node.parentNode;
154
+ }
155
+
156
+ return true;
157
+ };
158
+
159
+ /**
160
+ * @see {@link generateFragment} - this method wraps the error-throwing portions
161
+ * of that method.
162
+ * @throws {Error} - Will throw if computation takes longer than the accepted
163
+ * timeout length.
164
+ */
165
+ const doGenerateFragment =
166
+ (selection: Selection, startTime: number): GenerateFragmentResult => {
167
+ let range: Range;
168
+ try {
169
+ range = selection.getRangeAt(0);
170
+ } catch {
171
+ return {status: GenerateFragmentStatus.INVALID_SELECTION};
172
+ }
173
+
174
+ return doGenerateFragmentFromRange(range, startTime);
175
+ };
176
+
177
+ /**
178
+ * @see {@link doGenerateFragment}
179
+ */
180
+ const doGenerateFragmentFromRange = (range: Range, startTime: number): GenerateFragmentResult => {
181
+ recordStartTime(startTime);
182
+ // Upstream derives all of this from the global `document`/`document.body`
183
+ // (the live page it's attached to). This is also used against arbitrary,
184
+ // often-detached parsed documents, so every document/root lookup below is
185
+ // derived from the range's own nodes.
186
+ const doc = range.startContainer.ownerDocument as Document;
187
+ const root = documentBody(doc);
188
+
189
+ expandRangeStartToWordBound(range);
190
+ expandRangeEndToWordBound(range);
191
+ // Keep a copy of the range before we try to shrink it to make it start and
192
+ // end in text nodes. We need to use the range edges as starting points
193
+ // for context term building, so it makes sense to start from the original
194
+ // edges instead of the edges after shrinking. This way we don't have to
195
+ // traverse all the non-text nodes that are between the edges after shrinking
196
+ // and the original ones.
197
+ const rangeBeforeShrinking = range.cloneRange();
198
+
199
+ moveRangeEdgesToTextNodes(range);
200
+
201
+ if (range.collapsed) {
202
+ return {status: GenerateFragmentStatus.INVALID_SELECTION};
203
+ }
204
+
205
+ let factory: FragmentFactory;
206
+
207
+ if (canUseExactMatch(range)) {
208
+ const exactText = fragments.internal.normalizeString(range.toString());
209
+ const fragment: TextFragment = {
210
+ textStart: exactText,
211
+ };
212
+ // If the exact text is long enough to be used on its own, try this and skip
213
+ // the longer process below.
214
+ if (exactText.length >= MIN_LENGTH_WITHOUT_CONTEXT &&
215
+ isUniquelyIdentifying(fragment, doc, root)) {
216
+ return {
217
+ status: GenerateFragmentStatus.SUCCESS,
218
+ fragment: fragment,
219
+ };
220
+ }
221
+
222
+ factory = new FragmentFactory(doc, root).setExactTextMatch(exactText);
223
+ } else {
224
+ // We have to use textStart and textEnd to identify a range. First, break
225
+ // the range up based on block boundaries, as textStart/textEnd can't cross
226
+ // these.
227
+ const startSearchSpace = getSearchSpaceForStart(range);
228
+ const endSearchSpace = getSearchSpaceForEnd(range);
229
+
230
+ if (startSearchSpace && endSearchSpace) {
231
+ // If the search spaces are truthy, then there's a block boundary between
232
+ // them.
233
+ factory = new FragmentFactory(doc, root).setStartAndEndSearchSpace(
234
+ startSearchSpace, endSearchSpace);
235
+ } else {
236
+ // If the search space was empty/undefined, it's because no block boundary
237
+ // was found. That means textStart and textEnd *share* a search space, so
238
+ // our approach must ensure the substrings chosen as candidates don't
239
+ // overlap.
240
+ factory = new FragmentFactory(doc, root)
241
+ .setSharedSearchSpace(range.toString().trim());
242
+ }
243
+ }
244
+
245
+ const prefixRange = doc.createRange();
246
+ prefixRange.selectNodeContents(root);
247
+ const suffixRange = prefixRange.cloneRange();
248
+
249
+ prefixRange.setEnd(
250
+ rangeBeforeShrinking.startContainer, rangeBeforeShrinking.startOffset);
251
+ suffixRange.setStart(
252
+ rangeBeforeShrinking.endContainer, rangeBeforeShrinking.endOffset);
253
+
254
+ const prefixSearchSpace = getSearchSpaceForEnd(prefixRange);
255
+ const suffixSearchSpace = getSearchSpaceForStart(suffixRange);
256
+
257
+ if (prefixSearchSpace || suffixSearchSpace) {
258
+ factory.setPrefixAndSuffixSearchSpace(prefixSearchSpace, suffixSearchSpace);
259
+ }
260
+
261
+ factory.useSegmenter(fragments.internal.makeNewSegmenter(doc));
262
+
263
+ let didEmbiggen = false;
264
+ do {
265
+ checkTimeout();
266
+ didEmbiggen = factory.embiggen();
267
+ const fragment = factory.tryToMakeUniqueFragment();
268
+ if (fragment != null) {
269
+ return {
270
+ status: GenerateFragmentStatus.SUCCESS,
271
+ fragment: fragment,
272
+ };
273
+ }
274
+ } while (didEmbiggen);
275
+
276
+ return {status: GenerateFragmentStatus.AMBIGUOUS};
277
+ };
278
+
279
+ /**
280
+ * @throws {Error} - if the timeout duration has been exceeded, an error will
281
+ * be thrown so that execution can be halted.
282
+ */
283
+ // docRoot.body can return a synthesized, still-empty node under some
284
+ // parsers' still-in-progress HTML5 tree construction for a body-less
285
+ // fragment — querying for the real, content-bearing <body> in document
286
+ // order sidesteps that.
287
+ const documentBody = (doc: Document): Element => doc.querySelector('body') ?? doc.documentElement;
288
+
289
+ const checkTimeout = (): void => {
290
+ // disable check when no timeout duration specified
291
+ if (timeoutDurationMs === null) {
292
+ return;
293
+ }
294
+ const delta = Date.now() - t0;
295
+ if (delta > timeoutDurationMs) {
296
+ throw new FragmentTimeoutError(`Fragment generation timed out after ${delta} ms.`);
297
+ }
298
+ };
299
+
300
+ /**
301
+ * Call at the start of fragment generation to set the baseline for timeout
302
+ * checking.
303
+ * @param newStartTime - the timestamp when fragment generation began
304
+ */
305
+ const recordStartTime = (newStartTime: number): void => {
306
+ t0 = newStartTime;
307
+ };
308
+
309
+ /**
310
+ * Finds the search space for parameters when using range or suffix match.
311
+ * This is the text from the start of the range to the first block boundary,
312
+ * trimmed to remove any leading/trailing whitespace characters.
313
+ * @param range - the range which will be highlighted.
314
+ * @return the text which may be used for constructing a textStart parameter
315
+ * identifying this range. Will return undefined if no block boundaries
316
+ * are found inside this range, or if all the candidate ranges were empty
317
+ * (or included only whitespace characters).
318
+ */
319
+ const getSearchSpaceForStart = (range: Range): string | undefined => {
320
+ let node: Node | null = getFirstNodeForBlockSearch(range);
321
+ const walker = makeWalkerForNode(node, range.endContainer);
322
+ if (!walker) {
323
+ return undefined;
324
+ }
325
+
326
+ const finishedSubtrees = new Set<Node>();
327
+ // If the range starts after the last child of an element node
328
+ // don't visit its subtree because it's not included in the range.
329
+ if (range.startContainer.nodeType === Node.ELEMENT_NODE &&
330
+ range.startOffset === range.startContainer.childNodes.length) {
331
+ finishedSubtrees.add(range.startContainer);
332
+ }
333
+ const origin = node;
334
+ const textAccumulator = new BlockTextAccumulator(range, true);
335
+ // tempRange monitors whether we've exhausted our search space yet.
336
+ const tempRange = range.cloneRange();
337
+ while (!tempRange.collapsed && node != null) {
338
+ checkTimeout();
339
+ // Depending on whether |node| is an ancestor of the start of our
340
+ // search, we use either its leading or trailing edge as our start.
341
+ if ((node as Element).contains?.(origin)) {
342
+ tempRange.setStartAfter(node);
343
+ } else {
344
+ tempRange.setStartBefore(node);
345
+ }
346
+ // Add node to accumulator to keep track of text inside the current block
347
+ // boundaries
348
+ textAccumulator.appendNode(node);
349
+
350
+ // If the accumulator found a non empty block boundary we've got our search
351
+ // space.
352
+ if (textAccumulator.textInBlock !== null) {
353
+ return textAccumulator.textInBlock;
354
+ }
355
+ node = fragments.internal.forwardTraverse(walker, finishedSubtrees);
356
+ }
357
+ return undefined;
358
+ };
359
+
360
+ /**
361
+ * Finds the search space for parameters when using range or prefix match.
362
+ * This is the text from the last block boundary to the end of the range,
363
+ * trimmed to remove any leading/trailing whitespace characters.
364
+ * @param range - the range which will be highlighted.
365
+ * @return the text which may be used for constructing a textEnd parameter
366
+ * identifying this range. Will return undefined if no block boundaries
367
+ * are found inside this range, or if all the candidate ranges were empty
368
+ * (or included only whitespace characters).
369
+ */
370
+ const getSearchSpaceForEnd = (range: Range): string | undefined => {
371
+ let node: Node | null = getLastNodeForBlockSearch(range);
372
+ const walker = makeWalkerForNode(node, range.startContainer);
373
+ if (!walker) {
374
+ return undefined;
375
+ }
376
+ const finishedSubtrees = new Set<Node>();
377
+ // If the range ends before the first child of an element node
378
+ // don't visit its subtree because it's not included in the range.
379
+ if (range.endContainer.nodeType === Node.ELEMENT_NODE &&
380
+ range.endOffset === 0) {
381
+ finishedSubtrees.add(range.endContainer);
382
+ }
383
+
384
+ const origin = node;
385
+ const textAccumulator = new BlockTextAccumulator(range, false);
386
+
387
+ // tempRange monitors whether we've exhausted our search space yet.
388
+ const tempRange = range.cloneRange();
389
+ while (!tempRange.collapsed && node != null) {
390
+ checkTimeout();
391
+ // Depending on whether |node| is an ancestor of the start of our
392
+ // search, we use either its leading or trailing edge as our end.
393
+ if ((node as Element).contains?.(origin)) {
394
+ tempRange.setEnd(node, 0);
395
+ } else {
396
+ tempRange.setEndAfter(node);
397
+ }
398
+
399
+ // Add node to accumulator to keep track of text inside the current block
400
+ // boundaries.
401
+ textAccumulator.appendNode(node);
402
+
403
+ // If the accumulator found a non empty block boundary we've got our search
404
+ // space.
405
+ if (textAccumulator.textInBlock !== null) {
406
+ return textAccumulator.textInBlock;
407
+ }
408
+
409
+ node = fragments.internal.backwardTraverse(walker, finishedSubtrees);
410
+ }
411
+ return undefined;
412
+ };
413
+
414
+ const FactoryMode = {
415
+ ALL_PARTS: 1,
416
+ SHARED_START_AND_END: 2,
417
+ CONTEXT_ONLY: 3,
418
+ } as const;
419
+ type FactoryModeValue = typeof FactoryMode[keyof typeof FactoryMode];
420
+
421
+ /**
422
+ * Helper class for constructing range-based fragments for selections that cross
423
+ * block boundaries.
424
+ */
425
+ class FragmentFactory {
426
+ private readonly doc: Document;
427
+ private readonly root: Element;
428
+ private readonly Mode = FactoryMode;
429
+
430
+ private mode?: FactoryModeValue;
431
+
432
+ private startOffset: number | null = null;
433
+ private endOffset: number | null = null;
434
+ private prefixOffset: number | null = null;
435
+ private suffixOffset: number | null = null;
436
+
437
+ private prefixSearchSpace = '';
438
+ private backwardsPrefixSearchSpace = '';
439
+ private suffixSearchSpace = '';
440
+
441
+ private startSearchSpace?: string;
442
+ private endSearchSpace?: string;
443
+ private backwardsEndSearchSpace?: string;
444
+ private sharedSearchSpace?: string;
445
+ private backwardsSharedSearchSpace?: string;
446
+ private exactTextMatch?: string;
447
+
448
+ private startSegments?: Intl.Segments;
449
+ private endSegments?: Intl.Segments;
450
+ private sharedSegments?: Intl.Segments;
451
+ private prefixSegments?: Intl.Segments;
452
+ private suffixSegments?: Intl.Segments;
453
+
454
+ private numIterations = 0;
455
+
456
+ /**
457
+ * Initializes the basic state of the factory. Users should then call exactly
458
+ * one of setStartAndEndSearchSpace, setSharedSearchSpace, or
459
+ * setExactTextMatch, and optionally setPrefixAndSuffixSearchSpace.
460
+ * @param doc - document to check uniqueness against.
461
+ * @param root - root element to check uniqueness against.
462
+ */
463
+ constructor(doc: Document, root: Element) {
464
+ this.doc = doc;
465
+ this.root = root;
466
+ }
467
+
468
+ /**
469
+ * Generates a fragment based on the current state, then tests it for
470
+ * uniqueness.
471
+ * @return a text fragment if the current state is uniquely identifying, or
472
+ * undefined if the current state is ambiguous.
473
+ */
474
+ tryToMakeUniqueFragment(): TextFragment | undefined {
475
+ let fragment: TextFragment;
476
+ if (this.mode === this.Mode.CONTEXT_ONLY) {
477
+ fragment = {textStart: this.exactTextMatch!};
478
+ } else {
479
+ fragment = {
480
+ textStart:
481
+ this.getStartSearchSpace().substring(0, this.startOffset!).trim(),
482
+ textEnd: this.getEndSearchSpace().substring(this.endOffset!).trim(),
483
+ };
484
+ }
485
+ if (this.prefixOffset != null) {
486
+ const prefix =
487
+ this.getPrefixSearchSpace().substring(this.prefixOffset).trim();
488
+ if (prefix) {
489
+ fragment.prefix = prefix;
490
+ }
491
+ }
492
+ if (this.suffixOffset != null) {
493
+ const suffix =
494
+ this.getSuffixSearchSpace().substring(0, this.suffixOffset).trim();
495
+ if (suffix) {
496
+ fragment.suffix = suffix;
497
+ }
498
+ }
499
+ return isUniquelyIdentifying(fragment, this.doc, this.root) ? fragment :
500
+ undefined;
501
+ }
502
+
503
+ /**
504
+ * Shifts the current state such that the candidates for textStart and textEnd
505
+ * represent more of the possible search spaces.
506
+ * @return true if the desired expansion occurred; false if the entire search
507
+ * space has been consumed and no further attempts can be made.
508
+ */
509
+ embiggen(): boolean {
510
+ let canExpandRange = true;
511
+
512
+ if (this.mode === this.Mode.SHARED_START_AND_END) {
513
+ if (this.startOffset! >= this.endOffset!) {
514
+ // If the search space is shared between textStart and textEnd, then
515
+ // stop expanding when textStart overlaps textEnd.
516
+ canExpandRange = false;
517
+ }
518
+ } else if (this.mode === this.Mode.ALL_PARTS) {
519
+ // Stop expanding if both start and end have already consumed their full
520
+ // search spaces.
521
+ if (this.startOffset === this.getStartSearchSpace().length &&
522
+ this.backwardsEndOffset() === this.getEndSearchSpace().length) {
523
+ canExpandRange = false;
524
+ }
525
+ } else if (this.mode === this.Mode.CONTEXT_ONLY) {
526
+ canExpandRange = false;
527
+ }
528
+
529
+ if (canExpandRange) {
530
+ const desiredIterations = this.getNumberOfRangeWordsToAdd();
531
+ if (this.startOffset! < this.getStartSearchSpace().length) {
532
+ let i = 0;
533
+ if (this.getStartSegments() != null) {
534
+ while (i < desiredIterations &&
535
+ this.startOffset! < this.getStartSearchSpace().length) {
536
+ this.startOffset = this.getNextOffsetForwards(
537
+ this.getStartSegments()!, this.startOffset!,
538
+ this.getStartSearchSpace());
539
+ i++;
540
+ }
541
+ } else {
542
+ // We don't have a segmenter, so find the next boundary character
543
+ // instead. Shift to the next boundary char, and repeat until we've
544
+ // added a word char.
545
+ let oldStartOffset = this.startOffset!;
546
+ do {
547
+ checkTimeout();
548
+ const newStartOffset =
549
+ this.getStartSearchSpace()
550
+ .substring(this.startOffset! + 1)
551
+ .search(fragments.internal.BOUNDARY_CHARS);
552
+ if (newStartOffset === -1) {
553
+ this.startOffset = this.getStartSearchSpace().length;
554
+ } else {
555
+ this.startOffset = this.startOffset! + 1 + newStartOffset;
556
+ }
557
+ // Only count as an iteration if a word character was added.
558
+ if (this.getStartSearchSpace()
559
+ .substring(oldStartOffset, this.startOffset)
560
+ .search(fragments.internal.NON_BOUNDARY_CHARS) !== -1) {
561
+ oldStartOffset = this.startOffset;
562
+ i++;
563
+ }
564
+ } while (this.startOffset! < this.getStartSearchSpace().length &&
565
+ i < desiredIterations);
566
+ }
567
+
568
+ // Ensure we don't have overlapping start and end offsets.
569
+ if (this.mode === this.Mode.SHARED_START_AND_END) {
570
+ this.startOffset = Math.min(this.startOffset!, this.endOffset!);
571
+ }
572
+ }
573
+
574
+ if (this.backwardsEndOffset() < this.getEndSearchSpace().length) {
575
+ let i = 0;
576
+ if (this.getEndSegments() != null) {
577
+ while (i < desiredIterations && this.endOffset! > 0) {
578
+ this.endOffset = this.getNextOffsetBackwards(
579
+ this.getEndSegments()!, this.endOffset!);
580
+ i++;
581
+ }
582
+ } else {
583
+ // No segmenter, so shift to the next boundary char, and repeat until
584
+ // we've added a word char.
585
+ let oldBackwardsEndOffset = this.backwardsEndOffset();
586
+ do {
587
+ checkTimeout();
588
+ const newBackwardsOffset =
589
+ this.getBackwardsEndSearchSpace()
590
+ .substring(this.backwardsEndOffset() + 1)
591
+ .search(fragments.internal.BOUNDARY_CHARS);
592
+ if (newBackwardsOffset === -1) {
593
+ this.setBackwardsEndOffset(this.getEndSearchSpace().length);
594
+ } else {
595
+ this.setBackwardsEndOffset(
596
+ this.backwardsEndOffset() + 1 + newBackwardsOffset);
597
+ }
598
+ // Only count as an iteration if a word character was added.
599
+ if (this.getBackwardsEndSearchSpace()
600
+ .substring(oldBackwardsEndOffset, this.backwardsEndOffset())
601
+ .search(fragments.internal.NON_BOUNDARY_CHARS) !== -1) {
602
+ oldBackwardsEndOffset = this.backwardsEndOffset();
603
+ i++;
604
+ }
605
+ } while (this.backwardsEndOffset() <
606
+ this.getEndSearchSpace().length &&
607
+ i < desiredIterations);
608
+ }
609
+ // Ensure we don't have overlapping start and end offsets.
610
+ if (this.mode === this.Mode.SHARED_START_AND_END) {
611
+ this.endOffset = Math.max(this.startOffset!, this.endOffset!);
612
+ }
613
+ }
614
+ }
615
+
616
+ let canExpandContext = false;
617
+ if (!canExpandRange ||
618
+ this.startOffset! + this.backwardsEndOffset() <
619
+ MIN_LENGTH_WITHOUT_CONTEXT ||
620
+ this.numIterations >= ITERATIONS_BEFORE_ADDING_CONTEXT) {
621
+ // Check if there's any unused search space left.
622
+ if ((this.backwardsPrefixOffset() != null &&
623
+ this.backwardsPrefixOffset() !==
624
+ this.getPrefixSearchSpace().length) ||
625
+ (this.suffixOffset != null &&
626
+ this.suffixOffset !== this.getSuffixSearchSpace().length)) {
627
+ canExpandContext = true;
628
+ }
629
+ }
630
+
631
+ if (canExpandContext) {
632
+ const desiredIterations = this.getNumberOfContextWordsToAdd();
633
+ if (this.backwardsPrefixOffset()! < this.getPrefixSearchSpace().length) {
634
+ let i = 0;
635
+ if (this.getPrefixSegments() != null) {
636
+ while (i < desiredIterations && this.prefixOffset! > 0) {
637
+ this.prefixOffset = this.getNextOffsetBackwards(
638
+ this.getPrefixSegments()!, this.prefixOffset!);
639
+ i++;
640
+ }
641
+ } else {
642
+ // Shift to the next boundary char, and repeat until we've added a
643
+ // word char.
644
+ let oldBackwardsPrefixOffset = this.backwardsPrefixOffset()!;
645
+ do {
646
+ checkTimeout();
647
+ const newBackwardsPrefixOffset =
648
+ this.getBackwardsPrefixSearchSpace()
649
+ .substring(this.backwardsPrefixOffset()! + 1)
650
+ .search(fragments.internal.BOUNDARY_CHARS);
651
+ if (newBackwardsPrefixOffset === -1) {
652
+ this.setBackwardsPrefixOffset(
653
+ this.getBackwardsPrefixSearchSpace().length);
654
+ } else {
655
+ this.setBackwardsPrefixOffset(
656
+ this.backwardsPrefixOffset()! + 1 + newBackwardsPrefixOffset);
657
+ }
658
+ // Only count as an iteration if a word character was added.
659
+ if (this.getBackwardsPrefixSearchSpace()
660
+ .substring(
661
+ oldBackwardsPrefixOffset, this.backwardsPrefixOffset()!)
662
+ .search(fragments.internal.NON_BOUNDARY_CHARS) !== -1) {
663
+ oldBackwardsPrefixOffset = this.backwardsPrefixOffset()!;
664
+ i++;
665
+ }
666
+ } while (this.backwardsPrefixOffset()! <
667
+ this.getPrefixSearchSpace().length &&
668
+ i < desiredIterations);
669
+ }
670
+ }
671
+ if (this.suffixOffset! < this.getSuffixSearchSpace().length) {
672
+ let i = 0;
673
+ if (this.getSuffixSegments() != null) {
674
+ while (i < desiredIterations &&
675
+ this.suffixOffset! < this.getSuffixSearchSpace().length) {
676
+ this.suffixOffset = this.getNextOffsetForwards(
677
+ this.getSuffixSegments()!, this.suffixOffset!,
678
+ this.getSuffixSearchSpace());
679
+ i++;
680
+ }
681
+ } else {
682
+ let oldSuffixOffset = this.suffixOffset!;
683
+ do {
684
+ checkTimeout();
685
+ const newSuffixOffset =
686
+ this.getSuffixSearchSpace()
687
+ .substring(this.suffixOffset! + 1)
688
+ .search(fragments.internal.BOUNDARY_CHARS);
689
+ if (newSuffixOffset === -1) {
690
+ this.suffixOffset = this.getSuffixSearchSpace().length;
691
+ } else {
692
+ this.suffixOffset = this.suffixOffset! + 1 + newSuffixOffset;
693
+ }
694
+ // Only count as an iteration if a word character was added.
695
+ if (this.getSuffixSearchSpace()
696
+ .substring(oldSuffixOffset, this.suffixOffset)
697
+ .search(fragments.internal.NON_BOUNDARY_CHARS) !== -1) {
698
+ oldSuffixOffset = this.suffixOffset;
699
+ i++;
700
+ }
701
+ } while (this.suffixOffset! < this.getSuffixSearchSpace().length &&
702
+ i < desiredIterations);
703
+ }
704
+ }
705
+ }
706
+ this.numIterations++;
707
+
708
+ // TODO: check if this exceeds the total length limit
709
+ return canExpandRange || canExpandContext;
710
+ }
711
+
712
+ /**
713
+ * Sets up the factory for a range-based match with a highlight that crosses
714
+ * block boundaries.
715
+ *
716
+ * Exactly one of this, setSharedSearchSpace, or setExactTextMatch should be
717
+ * called so the factory can identify the fragment.
718
+ *
719
+ * @param startSearchSpace - the maximum possible string which can be used to
720
+ * identify the start of the fragment
721
+ * @param endSearchSpace - the maximum possible string which can be used to
722
+ * identify the end of the fragment
723
+ * @return returns |this| to allow call chaining and assignment
724
+ */
725
+ setStartAndEndSearchSpace(startSearchSpace: string, endSearchSpace: string): this {
726
+ this.startSearchSpace = startSearchSpace;
727
+ this.endSearchSpace = endSearchSpace;
728
+ this.backwardsEndSearchSpace = reverseString(endSearchSpace);
729
+
730
+ this.startOffset = 0;
731
+ this.endOffset = endSearchSpace.length;
732
+
733
+ this.mode = this.Mode.ALL_PARTS;
734
+ return this;
735
+ }
736
+
737
+ /**
738
+ * Sets up the factory for a range-based match with a highlight that doesn't
739
+ * cross block boundaries.
740
+ *
741
+ * Exactly one of this, setStartAndEndSearchSpace, or setExactTextMatch should
742
+ * be called so the factory can identify the fragment.
743
+ *
744
+ * @param sharedSearchSpace - the full text of the highlight
745
+ * @return returns |this| to allow call chaining and assignment
746
+ */
747
+ setSharedSearchSpace(sharedSearchSpace: string): this {
748
+ this.sharedSearchSpace = sharedSearchSpace;
749
+ this.backwardsSharedSearchSpace = reverseString(sharedSearchSpace);
750
+
751
+ this.startOffset = 0;
752
+ this.endOffset = sharedSearchSpace.length;
753
+
754
+ this.mode = this.Mode.SHARED_START_AND_END;
755
+ return this;
756
+ }
757
+
758
+ /**
759
+ * Sets up the factory for an exact text match.
760
+ *
761
+ * Exactly one of this, setStartAndEndSearchSpace, or setSharedSearchSpace
762
+ * should be called so the factory can identify the fragment.
763
+ *
764
+ * @param exactTextMatch - the full text of the highlight
765
+ * @return returns |this| to allow call chaining and assignment
766
+ */
767
+ setExactTextMatch(exactTextMatch: string): this {
768
+ this.exactTextMatch = exactTextMatch;
769
+
770
+ this.mode = this.Mode.CONTEXT_ONLY;
771
+ return this;
772
+ }
773
+
774
+ /**
775
+ * Sets up the factory for context-based matches.
776
+ * @param prefixSearchSpace - the string to be used as the search space for
777
+ * prefix
778
+ * @param suffixSearchSpace - the string to be used as the search space for
779
+ * suffix
780
+ * @return returns |this| to allow call chaining and assignment
781
+ */
782
+ setPrefixAndSuffixSearchSpace(prefixSearchSpace: string | undefined, suffixSearchSpace: string | undefined): this {
783
+ if (prefixSearchSpace) {
784
+ this.prefixSearchSpace = prefixSearchSpace;
785
+ this.backwardsPrefixSearchSpace = reverseString(prefixSearchSpace);
786
+ this.prefixOffset = prefixSearchSpace.length;
787
+ }
788
+
789
+ if (suffixSearchSpace) {
790
+ this.suffixSearchSpace = suffixSearchSpace;
791
+ this.suffixOffset = 0;
792
+ }
793
+
794
+ return this;
795
+ }
796
+
797
+ /**
798
+ * Sets up the factory to use an instance of Intl.Segmenter when identifying
799
+ * the start/end of words. |segmenter| is not actually retained; instead it is
800
+ * used to create segment objects which are cached.
801
+ *
802
+ * This must be called AFTER any calls to setStartAndEndSearchSpace,
803
+ * setSharedSearchSpace, and/or setPrefixAndSuffixSearchSpace, as these search
804
+ * spaces will be segmented immediately.
805
+ */
806
+ useSegmenter(segmenter: Intl.Segmenter | undefined): this {
807
+ if (segmenter == null) {
808
+ return this;
809
+ }
810
+
811
+ if (this.mode === this.Mode.ALL_PARTS) {
812
+ this.startSegments = segmenter.segment(this.startSearchSpace!);
813
+ this.endSegments = segmenter.segment(this.endSearchSpace!);
814
+ } else if (this.mode === this.Mode.SHARED_START_AND_END) {
815
+ this.sharedSegments = segmenter.segment(this.sharedSearchSpace!);
816
+ }
817
+
818
+ if (this.prefixSearchSpace) {
819
+ this.prefixSegments = segmenter.segment(this.prefixSearchSpace);
820
+ }
821
+ if (this.suffixSearchSpace) {
822
+ this.suffixSegments = segmenter.segment(this.suffixSearchSpace);
823
+ }
824
+
825
+ return this;
826
+ }
827
+
828
+ /**
829
+ * @return how many words should be added to the prefix and suffix when
830
+ * embiggening. This changes depending on the current state of the
831
+ * prefix/suffix, so it should be invoked once per embiggen, before either
832
+ * is modified.
833
+ */
834
+ private getNumberOfContextWordsToAdd(): number {
835
+ return (this.backwardsPrefixOffset() === 0 && this.suffixOffset === 0) ?
836
+ WORDS_TO_ADD_FIRST_ITERATION :
837
+ WORDS_TO_ADD_SUBSEQUENT_ITERATIONS;
838
+ }
839
+
840
+ /**
841
+ * @return how many words should be added to textStart and textEnd when
842
+ * embiggening. This changes depending on the current state of
843
+ * textStart/textEnd, so it should be invoked once per embiggen, before
844
+ * either is modified.
845
+ */
846
+ private getNumberOfRangeWordsToAdd(): number {
847
+ return (this.startOffset === 0 && this.backwardsEndOffset() === 0) ?
848
+ WORDS_TO_ADD_FIRST_ITERATION :
849
+ WORDS_TO_ADD_SUBSEQUENT_ITERATIONS;
850
+ }
851
+
852
+ /**
853
+ * Helper method for embiggening using Intl.Segmenter. Finds the next offset
854
+ * to be tried in the forwards direction (i.e., a prefix of the search space).
855
+ */
856
+ private getNextOffsetForwards(segments: Intl.Segments, offset: number, searchSpace: string): number {
857
+ // Find the nearest wordlike segment and move to the end of it.
858
+ let currentSegment = segments.containing(offset);
859
+ while (currentSegment != null) {
860
+ checkTimeout();
861
+ const currentSegmentEnd =
862
+ currentSegment.index + currentSegment.segment.length;
863
+ if (currentSegment.isWordLike) {
864
+ return currentSegmentEnd;
865
+ }
866
+ currentSegment = segments.containing(currentSegmentEnd);
867
+ }
868
+ // If we didn't find a wordlike segment by the end of the string, set the
869
+ // offset to the full search space.
870
+ return searchSpace.length;
871
+ }
872
+
873
+ /**
874
+ * Helper method for embiggening using Intl.Segmenter. Finds the next offset
875
+ * to be tried in the backwards direction (i.e., a suffix of the search
876
+ * space).
877
+ */
878
+ private getNextOffsetBackwards(segments: Intl.Segments, offset: number): number {
879
+ // Find the nearest wordlike segment and move to the start of it.
880
+ let currentSegment = segments.containing(offset);
881
+
882
+ // Handle two edge cases:
883
+ // 1. |offset| is at the end of the search space, so |currentSegment|
884
+ // is undefined
885
+ // 2. We're already at the start of a segment, so moving to the start of
886
+ // |currentSegment| would be a no-op.
887
+ // In both cases, the solution is to grab the segment immediately
888
+ // prior to this offset.
889
+ if (!currentSegment || offset == currentSegment.index) {
890
+ // If offset is 0, this will return null, which is handled below.
891
+ currentSegment = segments.containing(offset - 1);
892
+ }
893
+
894
+ while (currentSegment != null) {
895
+ checkTimeout();
896
+ if (currentSegment.isWordLike) {
897
+ return currentSegment.index;
898
+ }
899
+ currentSegment = segments.containing(currentSegment.index - 1);
900
+ }
901
+ // If we didn't find a wordlike segment by the start of the string,
902
+ // set the offset to the full search space.
903
+ return 0;
904
+ }
905
+
906
+ /** @return the string to be used as the search space for textStart */
907
+ private getStartSearchSpace(): string {
908
+ return this.mode === this.Mode.SHARED_START_AND_END ?
909
+ this.sharedSearchSpace! :
910
+ this.startSearchSpace!;
911
+ }
912
+
913
+ /**
914
+ * @return the result of segmenting the start search space using
915
+ * Intl.Segmenter, or undefined if a segmenter was not provided.
916
+ */
917
+ private getStartSegments(): Intl.Segments | undefined {
918
+ return this.mode === this.Mode.SHARED_START_AND_END ? this.sharedSegments :
919
+ this.startSegments;
920
+ }
921
+
922
+ /** @return the string to be used as the search space for textEnd */
923
+ private getEndSearchSpace(): string {
924
+ return this.mode === this.Mode.SHARED_START_AND_END ?
925
+ this.sharedSearchSpace! :
926
+ this.endSearchSpace!;
927
+ }
928
+
929
+ /**
930
+ * @return the result of segmenting the end search space using
931
+ * Intl.Segmenter, or undefined if a segmenter was not provided.
932
+ */
933
+ private getEndSegments(): Intl.Segments | undefined {
934
+ return this.mode === this.Mode.SHARED_START_AND_END ? this.sharedSegments :
935
+ this.endSegments;
936
+ }
937
+
938
+ /** @return the string to be used as the search space for textEnd, backwards. */
939
+ private getBackwardsEndSearchSpace(): string {
940
+ return this.mode === this.Mode.SHARED_START_AND_END ?
941
+ this.backwardsSharedSearchSpace! :
942
+ this.backwardsEndSearchSpace!;
943
+ }
944
+
945
+ /** @return the string to be used as the search space for prefix */
946
+ private getPrefixSearchSpace(): string {
947
+ return this.prefixSearchSpace;
948
+ }
949
+
950
+ /**
951
+ * @return the result of segmenting the prefix search space using
952
+ * Intl.Segmenter, or undefined if a segmenter was not provided.
953
+ */
954
+ private getPrefixSegments(): Intl.Segments | undefined {
955
+ return this.prefixSegments;
956
+ }
957
+
958
+ /** @return the string to be used as the search space for prefix, backwards. */
959
+ private getBackwardsPrefixSearchSpace(): string {
960
+ return this.backwardsPrefixSearchSpace;
961
+ }
962
+
963
+ /** @return the string to be used as the search space for suffix */
964
+ private getSuffixSearchSpace(): string {
965
+ return this.suffixSearchSpace;
966
+ }
967
+
968
+ /**
969
+ * @return the result of segmenting the suffix search space using
970
+ * Intl.Segmenter, or undefined if a segmenter was not provided.
971
+ */
972
+ private getSuffixSegments(): Intl.Segments | undefined {
973
+ return this.suffixSegments;
974
+ }
975
+
976
+ /**
977
+ * Helper method for doing arithmetic in the backwards search space.
978
+ * @return the current end offset, as a start offset in the backwards search
979
+ * space
980
+ */
981
+ private backwardsEndOffset(): number {
982
+ return this.getEndSearchSpace().length - this.endOffset!;
983
+ }
984
+
985
+ /**
986
+ * Helper method for doing arithmetic in the backwards search space.
987
+ * @param backwardsEndOffset - the desired new value of the start offset in
988
+ * the backwards search space
989
+ */
990
+ private setBackwardsEndOffset(backwardsEndOffset: number): void {
991
+ this.endOffset = this.getEndSearchSpace().length - backwardsEndOffset;
992
+ }
993
+
994
+ /**
995
+ * Helper method for doing arithmetic in the backwards search space.
996
+ * @return the current prefix offset, as a start offset in the backwards
997
+ * search space
998
+ */
999
+ private backwardsPrefixOffset(): number | null {
1000
+ if (this.prefixOffset == null) return null;
1001
+ return this.getPrefixSearchSpace().length - this.prefixOffset;
1002
+ }
1003
+
1004
+ /**
1005
+ * Helper method for doing arithmetic in the backwards search space.
1006
+ * @param backwardsPrefixOffset - the desired new value of the prefix offset
1007
+ * in the backwards search space
1008
+ */
1009
+ private setBackwardsPrefixOffset(backwardsPrefixOffset: number): void {
1010
+ if (this.prefixOffset == null) return;
1011
+ this.prefixOffset =
1012
+ this.getPrefixSearchSpace().length - backwardsPrefixOffset;
1013
+ }
1014
+ }
1015
+
1016
+ type TextNodeLike = Node | { textContent: string };
1017
+
1018
+ /**
1019
+ * Helper class to calculate visible text from the start or end of a range
1020
+ * until a block boundary is reached or the range is exhausted.
1021
+ */
1022
+ class BlockTextAccumulator {
1023
+ private readonly searchRange: Range;
1024
+ private readonly isForwardTraversal: boolean;
1025
+ private textFound = false;
1026
+ private textNodes: TextNodeLike[] = [];
1027
+ textInBlock: string | null = null;
1028
+
1029
+ /**
1030
+ * @param searchRange - the range for which the text in the last or first
1031
+ * non empty block boundary will be calculated
1032
+ * @param isForwardTraversal - true if nodes in searchRange will be forward
1033
+ * traversed
1034
+ */
1035
+ constructor(searchRange: Range, isForwardTraversal: boolean) {
1036
+ this.searchRange = searchRange;
1037
+ this.isForwardTraversal = isForwardTraversal;
1038
+ }
1039
+
1040
+ /**
1041
+ * Adds the next node in the search space range traversal to the accumulator.
1042
+ * The accumulator then will keep track of the text nodes in the range until a
1043
+ * block boundary is found. Once a block boundary is found and the content of
1044
+ * the text nodes in the boundary is non empty, the property textInBlock will
1045
+ * be set with the content of the text nodes, trimmed of leading and trailing
1046
+ * whitespaces.
1047
+ * @param node - next node in the traversal of the searchRange
1048
+ */
1049
+ appendNode(node: Node): void {
1050
+ // If we already calculated the text in the block boundary just ignore any
1051
+ // calls to append nodes.
1052
+ if (this.textInBlock !== null) {
1053
+ return;
1054
+ }
1055
+ // We found a block boundary, check if there's text inside and set it to
1056
+ // textInBlock or keep going to the next block boundary.
1057
+ if (isBlock(node)) {
1058
+ if (this.textFound) {
1059
+ // When traversing backwards the nodes are pushed in reverse order.
1060
+ // Reversing them to get them in the right order.
1061
+ if (!this.isForwardTraversal) {
1062
+ this.textNodes.reverse();
1063
+ }
1064
+ // Concatenate all the text nodes in the block boundary and trim any
1065
+ // trailing and leading whitespaces.
1066
+ this.textInBlock = this.textNodes.map(textNode => textNode.textContent)
1067
+ .join('')
1068
+ .trim();
1069
+ } else {
1070
+ // Discard the text nodes visited so far since they are empty and we'll
1071
+ // continue searching in the next block boundary.
1072
+ this.textNodes = [];
1073
+ }
1074
+ return;
1075
+ }
1076
+
1077
+ // Ignore non text nodes.
1078
+ if (!isText(node)) return;
1079
+
1080
+ // Get the part of node inside the search range. This is to avoid
1081
+ // accumulating text that's not inside the range.
1082
+ const nodeToInsert = this.getNodeIntersectionWithRange(node);
1083
+
1084
+ // Keep track of any text found in the block boundary.
1085
+ this.textFound = this.textFound || (nodeToInsert.textContent ?? '').trim() !== '';
1086
+
1087
+ this.textNodes.push(nodeToInsert);
1088
+ }
1089
+
1090
+ /**
1091
+ * Calculates the intersection of a node with searchRange and returns a Text
1092
+ * Node with the intersection
1093
+ * @param node - the node to intercept with searchRange
1094
+ * @return node if node is fully within searchRange or a Text Node with the
1095
+ * substring of the content of node inside the search range
1096
+ */
1097
+ private getNodeIntersectionWithRange(node: Node): TextNodeLike {
1098
+ let startOffset: number | null = null;
1099
+ let endOffset: number | null = null;
1100
+
1101
+ const textLength = (node.textContent ?? '').length;
1102
+
1103
+ if (node === this.searchRange.startContainer &&
1104
+ this.searchRange.startOffset !== 0) {
1105
+ startOffset = this.searchRange.startOffset;
1106
+ }
1107
+
1108
+ if (node === this.searchRange.endContainer &&
1109
+ this.searchRange.endOffset !== textLength) {
1110
+ endOffset = this.searchRange.endOffset;
1111
+ }
1112
+ if (startOffset !== null || endOffset !== null) {
1113
+ return {
1114
+ textContent: (node.textContent ?? '').substring(
1115
+ startOffset ?? 0, endOffset ?? textLength),
1116
+ };
1117
+ }
1118
+
1119
+ return node;
1120
+ }
1121
+ }
1122
+
1123
+ /**
1124
+ * @param fragment - the candidate fragment
1125
+ * @param doc - document to check uniqueness against.
1126
+ * @param root - root element to check uniqueness against.
1127
+ * @return true iff the candidate fragment identifies exactly one portion of
1128
+ * the document.
1129
+ */
1130
+ const isUniquelyIdentifying = (fragment: TextFragment, doc: Document, root: Element): boolean => {
1131
+ return fragments.processTextFragmentDirective(fragment, doc, root).length ===
1132
+ 1;
1133
+ };
1134
+
1135
+ /**
1136
+ * Reverses a string. Compound unicode characters are preserved.
1137
+ * @param string - the string to reverse
1138
+ * @return sdrawkcab |gnirts|
1139
+ */
1140
+ const reverseString = (string: string): string => {
1141
+ // Spread operator (...) splits full characters, rather than code points, to
1142
+ // avoid breaking compound unicode characters upon reverse.
1143
+ return [...(string || '')].reverse().join('');
1144
+ };
1145
+
1146
+ /**
1147
+ * Determines whether the conditions for an exact match are met.
1148
+ * @param range - the range for which a fragment is being generated.
1149
+ * @return true if exact matching (i.e., only textStart) can be used; false if
1150
+ * range matching (i.e., both textStart and textEnd) must be used.
1151
+ */
1152
+ const canUseExactMatch = (range: Range): boolean => {
1153
+ if (range.toString().length > MAX_EXACT_MATCH_LENGTH) return false;
1154
+ return !containsBlockBoundary(range);
1155
+ };
1156
+
1157
+ /**
1158
+ * Finds the node at which a forward traversal through |range| should begin,
1159
+ * based on the range's start container and offset values.
1160
+ * @param range - the range which will be traversed
1161
+ * @return the node where traversal should begin
1162
+ */
1163
+ const getFirstNodeForBlockSearch = (range: Range): Node => {
1164
+ // Get a handle on the first node inside the range. For text nodes, this
1165
+ // is the start container; for element nodes, we use the offset to find
1166
+ // where it actually starts.
1167
+ let node: Node = range.startContainer;
1168
+ if (node.nodeType == Node.ELEMENT_NODE &&
1169
+ range.startOffset < node.childNodes.length) {
1170
+ node = node.childNodes[range.startOffset]!;
1171
+ }
1172
+ return node;
1173
+ };
1174
+
1175
+ /**
1176
+ * Finds the node at which a backward traversal through |range| should begin,
1177
+ * based on the range's end container and offset values.
1178
+ * @param range - the range which will be traversed
1179
+ * @return the node where traversal should begin
1180
+ */
1181
+ const getLastNodeForBlockSearch = (range: Range): Node => {
1182
+ // Get a handle on the last node inside the range. For text nodes, this
1183
+ // is the end container; for element nodes, we use the offset to find
1184
+ // where it actually ends. If the offset is 0, the node itself is returned.
1185
+ let node: Node = range.endContainer;
1186
+ if (node.nodeType == Node.ELEMENT_NODE && range.endOffset > 0) {
1187
+ node = node.childNodes[range.endOffset - 1]!;
1188
+ }
1189
+ return node;
1190
+ };
1191
+
1192
+ /**
1193
+ * Finds the first visible text node within a given range.
1194
+ * @param range - range in which to find the first visible text node
1195
+ * @return first visible text node within |range| or null if there are no
1196
+ * visible text nodes within |range|
1197
+ */
1198
+ const getFirstTextNode = (range: Range): Node | null => {
1199
+ // Check if first node in the range is a visible text node.
1200
+ const firstNode = getFirstNodeForBlockSearch(range);
1201
+ if (isText(firstNode) && fragments.internal.isNodeVisible(firstNode)) {
1202
+ return firstNode;
1203
+ }
1204
+
1205
+ // First node is not visible text, use a tree walker to find the first visible
1206
+ // text node.
1207
+ const walker = fragments.internal.makeTextNodeWalker(range);
1208
+ walker.currentNode = firstNode;
1209
+
1210
+ return walker.nextNode();
1211
+ };
1212
+
1213
+ /**
1214
+ * Finds the last visible text node within a given range.
1215
+ * @param range - range in which to find the last visible text node
1216
+ * @return last visible text node within |range| or null if there are no
1217
+ * visible text nodes within |range|
1218
+ */
1219
+ const getLastTextNode = (range: Range): Node | null => {
1220
+ // Check if last node in the range is a visible text node.
1221
+ const lastNode = getLastNodeForBlockSearch(range);
1222
+ if (isText(lastNode) && fragments.internal.isNodeVisible(lastNode)) {
1223
+ return lastNode;
1224
+ }
1225
+
1226
+ // Last node is not visible text, traverse the range backwards to find the
1227
+ // last visible text node.
1228
+ const walker = fragments.internal.makeTextNodeWalker(range);
1229
+ walker.currentNode = lastNode;
1230
+
1231
+ return fragments.internal.backwardTraverse(walker, new Set());
1232
+ };
1233
+
1234
+ /**
1235
+ * Determines whether or not a range crosses a block boundary.
1236
+ * @param range - the range to investigate
1237
+ * @return true if a block boundary was found, false if no such boundary was
1238
+ * found.
1239
+ */
1240
+ const containsBlockBoundary = (range: Range): boolean => {
1241
+ const tempRange = range.cloneRange();
1242
+ let node: Node | null = getFirstNodeForBlockSearch(tempRange);
1243
+ const walker = makeWalkerForNode(node);
1244
+ if (!walker) {
1245
+ return false;
1246
+ }
1247
+ const finishedSubtrees = new Set<Node>();
1248
+
1249
+ while (!tempRange.collapsed && node != null) {
1250
+ if (isBlock(node)) return true;
1251
+ if (node != null) tempRange.setStartAfter(node);
1252
+ node = fragments.internal.forwardTraverse(walker, finishedSubtrees);
1253
+ checkTimeout();
1254
+ }
1255
+ return false;
1256
+ };
1257
+
1258
+ /**
1259
+ * Attempts to find a word start within the given text node, starting at
1260
+ * |offset| and working backwards.
1261
+ *
1262
+ * @param node - a node to be searched
1263
+ * @param startOffset - the character offset within |node| where the selected
1264
+ * text begins. If undefined, the entire node will be searched.
1265
+ * @return the number indicating the offset to which a range should be set to
1266
+ * ensure it starts on a word bound. Returns -1 if the node is not a text
1267
+ * node, or if no word boundary character could be found.
1268
+ */
1269
+ const findWordStartBoundInTextNode = (node: Node, startOffset?: number | null): number => {
1270
+ if (node.nodeType !== Node.TEXT_NODE) return -1;
1271
+ const textNode = node as Text;
1272
+
1273
+ const offset = startOffset != null ? startOffset : textNode.data.length;
1274
+
1275
+ // If the first character in the range is a boundary character, we don't
1276
+ // need to do anything.
1277
+ if (offset < textNode.data.length &&
1278
+ fragments.internal.BOUNDARY_CHARS.test(textNode.data[offset]!))
1279
+ return offset;
1280
+
1281
+ const precedingText = textNode.data.substring(0, offset);
1282
+ const boundaryIndex =
1283
+ reverseString(precedingText).search(fragments.internal.BOUNDARY_CHARS);
1284
+
1285
+ if (boundaryIndex !== -1) {
1286
+ // Because we did a backwards search, the found index counts backwards
1287
+ // from offset, so we subtract to find the start of the word.
1288
+ return offset - boundaryIndex;
1289
+ }
1290
+ return -1;
1291
+ };
1292
+
1293
+ /**
1294
+ * Attempts to find a word end within the given text node, starting at |offset|.
1295
+ *
1296
+ * @param node - a node to be searched
1297
+ * @param endOffset - the character offset within |node| where the selected
1298
+ * text end. If undefined, the entire node will be searched.
1299
+ * @return the number indicating the offset to which a range should be set to
1300
+ * ensure it ends on a word bound. Returns -1 if the node is not a text
1301
+ * node, or if no word boundary character could be found.
1302
+ */
1303
+ const findWordEndBoundInTextNode = (node: Node, endOffset?: number | null): number => {
1304
+ if (node.nodeType !== Node.TEXT_NODE) return -1;
1305
+ const textNode = node as Text;
1306
+
1307
+ const offset = endOffset != null ? endOffset : 0;
1308
+
1309
+ // If the last character in the range is a boundary character, we don't
1310
+ // need to do anything.
1311
+ if (offset < textNode.data.length && offset > 0 &&
1312
+ fragments.internal.BOUNDARY_CHARS.test(textNode.data[offset - 1]!)) {
1313
+ return offset;
1314
+ }
1315
+
1316
+ const followingText = textNode.data.substring(offset);
1317
+ const boundaryIndex = followingText.search(fragments.internal.BOUNDARY_CHARS);
1318
+
1319
+ if (boundaryIndex !== -1) {
1320
+ return offset + boundaryIndex;
1321
+ }
1322
+ return -1;
1323
+ };
1324
+
1325
+ /**
1326
+ * Helper method to create a TreeWalker useful for finding a block boundary near
1327
+ * a given node.
1328
+ * @param node - the node where the search should start
1329
+ * @param endNode - optional; if included, the root of the walker will be
1330
+ * chosen to ensure it can traverse at least as far as this node.
1331
+ * @return a TreeWalker, rooted in a block ancestor of |node|, currently
1332
+ * pointing to |node|, which will traverse only visible text and element
1333
+ * nodes.
1334
+ */
1335
+ const makeWalkerForNode = (node: Node | null, endNode?: Node): TreeWalker | undefined => {
1336
+ if (!node) {
1337
+ return undefined;
1338
+ }
1339
+
1340
+ // Find a block-level ancestor of the node by walking up the tree. This
1341
+ // will be used as the root of the tree walker.
1342
+ let blockAncestor: Node = node;
1343
+ const endNodeNotNull = endNode != null ? endNode : node;
1344
+ while (!(blockAncestor as Element).contains?.(endNodeNotNull) || !isBlock(blockAncestor)) {
1345
+ if (blockAncestor.parentNode) {
1346
+ blockAncestor = blockAncestor.parentNode;
1347
+ }
1348
+ }
1349
+
1350
+ const doc = blockAncestor.ownerDocument ?? (blockAncestor as unknown as Document);
1351
+ const walker = doc.createTreeWalker(
1352
+ blockAncestor, NodeFilter.SHOW_ELEMENT | NodeFilter.SHOW_TEXT, {
1353
+ acceptNode: (node: Node) => fragments.internal.acceptNodeIfVisibleInRange(node),
1354
+ });
1355
+
1356
+ walker.currentNode = node;
1357
+ return walker;
1358
+ };
1359
+
1360
+ /**
1361
+ * Modifies the start of the range, if necessary, to ensure the selection text
1362
+ * starts after a boundary char (whitespace, etc.) or a block boundary. Can only
1363
+ * expand the range, not shrink it.
1364
+ * @param range - the range to be modified
1365
+ */
1366
+ const expandRangeStartToWordBound = (range: Range): void => {
1367
+ const segmenter =
1368
+ fragments.internal.makeNewSegmenter(range.startContainer.ownerDocument ?? undefined);
1369
+ if (segmenter) {
1370
+ // Find the starting text node and offset (since the range may start with a
1371
+ // non-text node).
1372
+ const startNode = getFirstNodeForBlockSearch(range);
1373
+ if (startNode !== range.startContainer) {
1374
+ range.setStartBefore(startNode);
1375
+ }
1376
+
1377
+ expandToNearestWordBoundaryPointUsingSegments(
1378
+ segmenter, /* isRangeEnd= */ false, range);
1379
+ } else {
1380
+ // Simplest case: If we're in a text node, try to find a boundary char in
1381
+ // the same text node.
1382
+ const newOffset =
1383
+ findWordStartBoundInTextNode(range.startContainer, range.startOffset);
1384
+ if (newOffset !== -1) {
1385
+ range.setStart(range.startContainer, newOffset);
1386
+ return;
1387
+ }
1388
+
1389
+ // Also, skip doing any traversal if we're already at the inside edge of
1390
+ // a block node.
1391
+ if (isBlock(range.startContainer) && range.startOffset === 0) {
1392
+ return;
1393
+ }
1394
+
1395
+ const walker = makeWalkerForNode(range.startContainer);
1396
+ if (!walker) {
1397
+ return;
1398
+ }
1399
+ const finishedSubtrees = new Set<Node>();
1400
+
1401
+ let node = fragments.internal.backwardTraverse(walker, finishedSubtrees);
1402
+ while (node != null) {
1403
+ const newOffset = findWordStartBoundInTextNode(node);
1404
+ if (newOffset !== -1) {
1405
+ range.setStart(node, newOffset);
1406
+ return;
1407
+ }
1408
+
1409
+ // If |node| is a block node, then we've hit a block boundary, which
1410
+ // counts as a word boundary.
1411
+ if (isBlock(node)) {
1412
+ if ((node as Element).contains?.(range.startContainer)) {
1413
+ // If the selection starts inside |node|, then the correct range
1414
+ // boundary is the *leading* edge of |node|.
1415
+ range.setStart(node, 0);
1416
+ } else {
1417
+ // Otherwise, |node| is before the selection, so the correct boundary
1418
+ // is the *trailing* edge of |node|.
1419
+ range.setStartAfter(node);
1420
+ }
1421
+ return;
1422
+ }
1423
+
1424
+ node = fragments.internal.backwardTraverse(walker, finishedSubtrees);
1425
+ // We should never get here; the walker should eventually hit a block node
1426
+ // or the root of the document. Collapse range so the caller can handle
1427
+ // this as an error.
1428
+ range.collapse();
1429
+ }
1430
+ }
1431
+ };
1432
+
1433
+ /**
1434
+ * Moves the range edges to the first and last visible text nodes inside of it.
1435
+ * If there are no visible text nodes in the range then it is collapsed.
1436
+ * @param range - the range to be modified
1437
+ */
1438
+ const moveRangeEdgesToTextNodes = (range: Range): void => {
1439
+ const firstTextNode = getFirstTextNode(range);
1440
+ // No text nodes in range. Collapsing the range and early return.
1441
+ if (firstTextNode == null) {
1442
+ range.collapse();
1443
+ return;
1444
+ }
1445
+
1446
+ const firstNode = getFirstNodeForBlockSearch(range);
1447
+
1448
+ // Making sure the range starts with visible text.
1449
+ if (firstNode !== firstTextNode) {
1450
+ range.setStart(firstTextNode, 0);
1451
+ }
1452
+
1453
+ const lastNode = getLastNodeForBlockSearch(range);
1454
+ const lastTextNode = getLastTextNode(range)!;
1455
+ // No need for no text node checks here because we know at there's at least
1456
+ // firstTextNode in the range.
1457
+
1458
+ // Making sure the range ends with visible text.
1459
+ if (lastNode !== lastTextNode) {
1460
+ range.setEnd(lastTextNode, (lastTextNode.textContent ?? '').length);
1461
+ }
1462
+ };
1463
+
1464
+ /**
1465
+ * Uses Intl.Segmenter to shift the start or end of a range to a word boundary.
1466
+ * Helper method for expandWord*ToWordBound methods.
1467
+ * @param segmenter - object to use for word segmenting
1468
+ * @param isRangeEnd - true if the range end should be modified, false if the
1469
+ * range start should be modified
1470
+ * @param range - the range to modify
1471
+ */
1472
+ const expandToNearestWordBoundaryPointUsingSegments =
1473
+ (segmenter: Intl.Segmenter, isRangeEnd: boolean, range: Range): void => {
1474
+ // Find the index as an offset in the full text of the block in which
1475
+ // boundary occurs.
1476
+ const boundary = isRangeEnd ?
1477
+ {node: range.endContainer, offset: range.endOffset} :
1478
+ {node: range.startContainer, offset: range.startOffset};
1479
+
1480
+ const nodes = getTextNodesInSameBlock(boundary.node);
1481
+ if (!nodes) return;
1482
+ const preNodeText = nodes.preNodes.reduce((prev, cur) => {
1483
+ return prev.concat(cur.textContent ?? '');
1484
+ }, '');
1485
+
1486
+ const innerNodeText = nodes.innerNodes.reduce((prev, cur) => {
1487
+ return prev.concat(cur.textContent ?? '');
1488
+ }, '');
1489
+
1490
+ let offsetInText = preNodeText.length;
1491
+ if (boundary.node.nodeType === Node.TEXT_NODE) {
1492
+ offsetInText += boundary.offset;
1493
+ } else if (isRangeEnd) {
1494
+ offsetInText += innerNodeText.length;
1495
+ }
1496
+
1497
+ // Find the segment of the full block text containing the range start.
1498
+ const postNodeText = nodes.postNodes.reduce((prev, cur) => {
1499
+ return prev.concat(cur.textContent ?? '');
1500
+ }, '');
1501
+
1502
+ const allNodes =
1503
+ [...nodes.preNodes, ...nodes.innerNodes, ...nodes.postNodes];
1504
+
1505
+ // Edge case: There's no text nodes in the block.
1506
+ // In that case there's nothing to do because there is no word boundary
1507
+ // to find.
1508
+ if (allNodes.length == 0) {
1509
+ return;
1510
+ }
1511
+
1512
+ const text = preNodeText.concat(innerNodeText, postNodeText);
1513
+
1514
+ const segments = segmenter.segment(text);
1515
+ const foundSegment = segments.containing(offsetInText);
1516
+
1517
+ if (!foundSegment) {
1518
+ if (isRangeEnd) {
1519
+ range.setEndAfter(allNodes[allNodes.length - 1]!);
1520
+ } else {
1521
+ range.setEndBefore(allNodes[0]!);
1522
+ }
1523
+ return;
1524
+ }
1525
+
1526
+ // Easy case: if the segment is not word-like (i.e., contains whitespace,
1527
+ // punctuation, etc.) then nothing needs to be done because this
1528
+ // boundary point is between words.
1529
+ if (!foundSegment.isWordLike) {
1530
+ return;
1531
+ }
1532
+
1533
+ // Another easy case: if we are at the first/last character of the
1534
+ // segment, then we're done.
1535
+ if (offsetInText === foundSegment.index ||
1536
+ offsetInText === foundSegment.index + foundSegment.segment.length) {
1537
+ return;
1538
+ }
1539
+
1540
+ // We're inside a word. Based on |isRangeEnd|, the target offset will
1541
+ // either be the start or the end of the found segment.
1542
+ const desiredOffsetInText = isRangeEnd ?
1543
+ foundSegment.index + foundSegment.segment.length :
1544
+ foundSegment.index;
1545
+ let newNodeIndexInText = 0;
1546
+ for (const node of allNodes) {
1547
+ const nodeTextLength = (node.textContent ?? '').length;
1548
+ if (newNodeIndexInText <= desiredOffsetInText &&
1549
+ desiredOffsetInText <
1550
+ newNodeIndexInText + nodeTextLength) {
1551
+ const offsetInNode = desiredOffsetInText - newNodeIndexInText;
1552
+ if (isRangeEnd) {
1553
+ if (offsetInNode >= nodeTextLength) {
1554
+ range.setEndAfter(node);
1555
+ } else {
1556
+ range.setEnd(node, offsetInNode);
1557
+ }
1558
+ } else {
1559
+ if (offsetInNode >= nodeTextLength) {
1560
+ range.setStartAfter(node);
1561
+ } else {
1562
+ range.setStart(node, offsetInNode);
1563
+ }
1564
+ }
1565
+ return;
1566
+ }
1567
+ newNodeIndexInText += nodeTextLength;
1568
+ }
1569
+
1570
+ // If we got here, then somehow the offset didn't fall within a node. As a
1571
+ // fallback, move the range to the start/end of the block.
1572
+ if (isRangeEnd) {
1573
+ range.setEndAfter(allNodes[allNodes.length - 1]!);
1574
+ } else {
1575
+ range.setStartBefore(allNodes[0]!);
1576
+ }
1577
+ };
1578
+
1579
+ interface TextNodeLists {
1580
+ preNodes: Text[];
1581
+ innerNodes: Text[];
1582
+ postNodes: Text[];
1583
+ }
1584
+
1585
+ /**
1586
+ * Traverses the DOM to extract all TextNodes appearing in the same block level
1587
+ * as |node| (i.e., those that are descendents of a common ancestor of |node|
1588
+ * with no other block elements in between.)
1589
+ */
1590
+ const getTextNodesInSameBlock = (node: Node): TextNodeLists | undefined => {
1591
+ const preNodes: Text[] = [];
1592
+ // First, backtraverse to get to a block boundary
1593
+ const backWalker = makeWalkerForNode(node);
1594
+ if (!backWalker) {
1595
+ return undefined;
1596
+ }
1597
+ const finishedSubtrees = new Set<Node>();
1598
+ let backNode: Node | null =
1599
+ fragments.internal.backwardTraverse(backWalker, finishedSubtrees);
1600
+ while (backNode != null && !isBlock(backNode)) {
1601
+ checkTimeout();
1602
+ if (backNode.nodeType === Node.TEXT_NODE) {
1603
+ preNodes.push(backNode as Text);
1604
+ }
1605
+ backNode =
1606
+ fragments.internal.backwardTraverse(backWalker, finishedSubtrees);
1607
+ }
1608
+ preNodes.reverse();
1609
+
1610
+ const innerNodes: Text[] = [];
1611
+ if (node.nodeType === Node.TEXT_NODE) {
1612
+ innerNodes.push(node as Text);
1613
+ } else {
1614
+ const doc = node.ownerDocument ?? (node as unknown as Document);
1615
+ const walker = doc.createTreeWalker(
1616
+ node, NodeFilter.SHOW_ELEMENT | NodeFilter.SHOW_TEXT, {
1617
+ acceptNode: (n: Node) => fragments.internal.acceptNodeIfVisibleInRange(n),
1618
+ });
1619
+ walker.currentNode = node;
1620
+ let child = walker.nextNode();
1621
+ while (child != null) {
1622
+ checkTimeout();
1623
+ if (child.nodeType === Node.TEXT_NODE) {
1624
+ innerNodes.push(child as Text);
1625
+ }
1626
+ child = walker.nextNode();
1627
+ }
1628
+ }
1629
+
1630
+ const postNodes: Text[] = [];
1631
+ const forwardWalker = makeWalkerForNode(node);
1632
+ if (!forwardWalker) {
1633
+ return undefined;
1634
+ }
1635
+ // Forward traverse from node after having finished its subtree
1636
+ // to get text nodes after it until we find a block boundary.
1637
+ const finishedSubtreesForward = new Set<Node>([node]);
1638
+ let forwardNode: Node | null = fragments.internal.forwardTraverse(
1639
+ forwardWalker, finishedSubtreesForward);
1640
+ while (forwardNode != null && !isBlock(forwardNode)) {
1641
+ checkTimeout();
1642
+ if (forwardNode.nodeType === Node.TEXT_NODE) {
1643
+ postNodes.push(forwardNode as Text);
1644
+ }
1645
+ forwardNode = fragments.internal.forwardTraverse(
1646
+ forwardWalker, finishedSubtreesForward);
1647
+ }
1648
+
1649
+ return {preNodes, innerNodes, postNodes};
1650
+ };
1651
+
1652
+ /**
1653
+ * Modifies the end of the range, if necessary, to ensure the selection text
1654
+ * ends before a boundary char (whitespace, etc.) or a block boundary. Can only
1655
+ * expand the range, not shrink it.
1656
+ * @param range - the range to be modified
1657
+ */
1658
+ const expandRangeEndToWordBound = (range: Range): void => {
1659
+ const segmenter =
1660
+ fragments.internal.makeNewSegmenter(range.endContainer.ownerDocument ?? undefined);
1661
+ if (segmenter) {
1662
+ // Find the ending text node and offset (since the range may end with a
1663
+ // non-text node).
1664
+ const endNode = getLastNodeForBlockSearch(range);
1665
+ if (endNode !== range.endContainer) {
1666
+ range.setEndAfter(endNode);
1667
+ }
1668
+ expandToNearestWordBoundaryPointUsingSegments(
1669
+ segmenter, /* isRangeEnd= */ true, range);
1670
+ } else {
1671
+ let initialOffset: number | null = range.endOffset;
1672
+
1673
+ let node: Node | null = range.endContainer;
1674
+ if (node.nodeType === Node.ELEMENT_NODE) {
1675
+ if (range.endOffset < node.childNodes.length) {
1676
+ node = node.childNodes[range.endOffset]!;
1677
+ }
1678
+ }
1679
+
1680
+ const walker = makeWalkerForNode(node);
1681
+ if (!walker) {
1682
+ return;
1683
+ }
1684
+ // We'll traverse the dom after node's subtree to try to find
1685
+ // either a word or block boundary.
1686
+ const finishedSubtrees = new Set<Node>([node]);
1687
+
1688
+ while (node != null) {
1689
+ checkTimeout();
1690
+
1691
+ const newOffset = findWordEndBoundInTextNode(node, initialOffset);
1692
+ // Future iterations should not use initialOffset; null it out so it is
1693
+ // discarded.
1694
+ initialOffset = null;
1695
+
1696
+ if (newOffset !== -1) {
1697
+ range.setEnd(node, newOffset);
1698
+ return;
1699
+ }
1700
+
1701
+ // If |node| is a block node, then we've hit a block boundary, which
1702
+ // counts as a word boundary.
1703
+ if (isBlock(node)) {
1704
+ if ((node as Element).contains?.(range.endContainer)) {
1705
+ // If the selection starts inside |node|, then the correct range
1706
+ // boundary is the *trailing* edge of |node|.
1707
+ range.setEnd(node, node.childNodes.length);
1708
+ } else {
1709
+ // Otherwise, |node| is after the selection, so the correct boundary
1710
+ // is the *leading* edge of |node|.
1711
+ range.setEndBefore(node);
1712
+ }
1713
+ return;
1714
+ }
1715
+
1716
+ node = fragments.internal.forwardTraverse(walker, finishedSubtrees);
1717
+ }
1718
+ // We should never get here; the walker should eventually hit a block node
1719
+ // or the root of the document. Collapse range so the caller can handle this
1720
+ // as an error.
1721
+ range.collapse();
1722
+ }
1723
+ };
1724
+
1725
+ /**
1726
+ * Helper to determine if a node is a block element or not.
1727
+ * @param node - the node to evaluate
1728
+ * @return true if the node is an element classified as block-level
1729
+ */
1730
+ const isBlock = (node: Node): boolean => {
1731
+ return node.nodeType === Node.ELEMENT_NODE &&
1732
+ (fragments.internal.BLOCK_ELEMENTS.includes((node as Element).tagName.toUpperCase()) ||
1733
+ (node as Element).tagName.toUpperCase() === 'HTML' ||
1734
+ (node as Element).tagName.toUpperCase() === 'BODY');
1735
+ };
1736
+
1737
+ /**
1738
+ * Helper to determine if a node is a Text Node or not
1739
+ * @param node - the node to evaluate
1740
+ * @return true if the node is a Text Node
1741
+ */
1742
+ const isText = (node: Node): boolean => {
1743
+ return node.nodeType === Node.TEXT_NODE;
1744
+ };