extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,614 @@
1
+ /**
2
+ * @fileoverview
3
+ * Splits text into sentences, handling 220+ common abbreviations,
4
+ * and inferring acronyms, numbers, URLs, times, names, etc.
5
+ *
6
+ * @param inputText - The text to be split into sentences.
7
+ * @param options - Configuration options for sentence splitting.
8
+ * @returns An array of sentences.
9
+ * @author [vtempest (2025)](https://github.com/vtempest)
10
+ * @license MIT
11
+ * @example
12
+ * ```ts
13
+ * const text = "Dr. Smith went to the U.S. He met Mr. Jones.";
14
+ * const sentences = splitTextToSentences(text);
15
+ * ["Dr. Smith went to the U.S.", "He met Mr. Jones."]
16
+ * ```
17
+ */
18
+ export function splitTextToSentences(
19
+ inputText: string,
20
+ options: SplitSentencesOptions = {},
21
+ ): string[] {
22
+ const { splitOnHtmlTags = true, minSize = 20, maxSize = 500 } = options;
23
+
24
+ // Validate input
25
+ if (!isValidInput(inputText)) {
26
+ return [];
27
+ }
28
+
29
+ // Preprocess and tokenize
30
+ const LINEBREAK_MARKER = " @~@ ";
31
+ const processedText = preprocessText(
32
+ inputText,
33
+ splitOnHtmlTags,
34
+ LINEBREAK_MARKER,
35
+ );
36
+ const tokens = tokenizeText(processedText);
37
+
38
+ if (!tokens || tokens.length === 0) {
39
+ return [];
40
+ }
41
+
42
+ // Detect sentence boundaries
43
+ const sentenceGroups = detectSentenceBoundaries(
44
+ tokens,
45
+ LINEBREAK_MARKER.trim(),
46
+ );
47
+
48
+ // Post-process and merge short abbreviations
49
+ const mergedGroups = mergeSplitAbbreviations(sentenceGroups);
50
+
51
+ // Apply size constraints and convert to strings
52
+ return applySizeConstraints(mergedGroups, minSize, maxSize);
53
+ }
54
+
55
+ export type SplitSentencesOptions = {
56
+ /**
57
+ * Split on HTML tags like P, DIV, UL, OL.
58
+ * @default true
59
+ */
60
+ splitOnHtmlTags?: boolean;
61
+ /**
62
+ * Minimum size for a sentence.
63
+ * @default 20
64
+ */
65
+ minSize?: number;
66
+ /**
67
+ * Maximum size for a sentence.
68
+ * @default 500
69
+ */
70
+ maxSize?: number;
71
+ };
72
+
73
+ /**
74
+ * Validates if the input text is non-empty and contains non-whitespace characters.
75
+ *
76
+ * @param text - The text to validate.
77
+ * @returns True if the text is valid, false otherwise.
78
+ */
79
+ function isValidInput(text: string): boolean {
80
+ const NON_EMPTY_REGEX = /\S/;
81
+ return (
82
+ !!text &&
83
+ typeof text === "string" &&
84
+ text.length > 0 &&
85
+ NON_EMPTY_REGEX.test(text)
86
+ );
87
+ }
88
+
89
+ /**
90
+ * Preprocesses text by replacing line breaks and optionally splitting on HTML tags.
91
+ *
92
+ * @param text - The text to preprocess.
93
+ * @param splitOnHtmlTags - Whether to split on HTML tags.
94
+ * @param linebreakMarker - The marker to use for line breaks.
95
+ * @returns The preprocessed text.
96
+ */
97
+ function preprocessText(
98
+ text: string,
99
+ splitOnHtmlTags: boolean,
100
+ linebreakMarker: string,
101
+ ): string {
102
+ const LINEBREAK_BOUNDARY_REGEX = /\n+|[-#=_+*]{4,}/g;
103
+ let processed = text.replace(LINEBREAK_BOUNDARY_REGEX, linebreakMarker);
104
+
105
+ if (splitOnHtmlTags) {
106
+ const htmlTagsToSplit = ["p", "div", "ul", "ol"];
107
+ const htmlSplitRegex = new RegExp(
108
+ `(<br\\s*\\/?>|<\\/(${htmlTagsToSplit.join("|")})>)`,
109
+ "g",
110
+ );
111
+ processed = processed.replace(htmlSplitRegex, `$1${linebreakMarker}`);
112
+ }
113
+
114
+ return processed;
115
+ }
116
+
117
+ /**
118
+ * Tokenizes text into words and newlines.
119
+ *
120
+ * @param text - The text to tokenize.
121
+ * @returns Array of tokens.
122
+ */
123
+ function tokenizeText(text: string): string[] | null {
124
+ const WORD_TOKENIZE_REGEX = /\S+|\n/g;
125
+ return text.trim().match(WORD_TOKENIZE_REGEX);
126
+ }
127
+
128
+ /**
129
+ * Detects sentence boundaries in a token array.
130
+ *
131
+ * @param tokens - Array of tokens to process.
132
+ * @param linebreakMarker - The marker used for line breaks.
133
+ * @returns Array of sentence groups (each group is an array of tokens).
134
+ */
135
+ function detectSentenceBoundaries(
136
+ tokens: string[],
137
+ linebreakMarker: string,
138
+ ): string[][] {
139
+ const sentenceGroups: string[][] = [];
140
+ let currentGroup: string[] = [];
141
+ let wordCounter = 0;
142
+
143
+ for (let i = 0; i < tokens.length; i++) {
144
+ const token = tokens[i];
145
+
146
+ wordCounter++;
147
+ currentGroup.push(token);
148
+
149
+ if (token.includes(",")) {
150
+ wordCounter = 0;
151
+ }
152
+
153
+ // Check for definite sentence endings
154
+ if (shouldEndSentence(token, linebreakMarker)) {
155
+ if (token === linebreakMarker) {
156
+ currentGroup.pop();
157
+ }
158
+ sentenceGroups.push(currentGroup);
159
+ currentGroup = [];
160
+ wordCounter = 0;
161
+ continue;
162
+ }
163
+
164
+ // Handle quotes at end of token
165
+ if (hasEndPunctuation(token, '"') || hasEndPunctuation(token, "'")) {
166
+ tokens[i] = token.slice(0, -1);
167
+ }
168
+
169
+ // Handle periods
170
+ if (hasEndPunctuation(token, ".")) {
171
+ const shouldContinue = shouldContinueAfterPeriod(
172
+ token,
173
+ tokens,
174
+ i,
175
+ wordCounter,
176
+ );
177
+
178
+ if (!shouldContinue) {
179
+ sentenceGroups.push(currentGroup);
180
+ currentGroup = [];
181
+ wordCounter = 0;
182
+ continue;
183
+ } else {
184
+ continue;
185
+ }
186
+ }
187
+
188
+ // Check for periods within tokens (decimals, URLs, etc.)
189
+ const periodIndex = token.indexOf(".");
190
+ if (periodIndex > -1) {
191
+ if (
192
+ isNumeric(token, periodIndex) ||
193
+ isComplexAbbreviation(token) ||
194
+ isValidUrl(token) ||
195
+ isPhoneNumber(token)
196
+ ) {
197
+ continue;
198
+ }
199
+ }
200
+
201
+ // Handle punctuation within words (e.g., "word.Another")
202
+ const maybeSplit = maybeSplitAtInternalPunctuation(
203
+ token,
204
+ currentGroup,
205
+ sentenceGroups,
206
+ );
207
+ if (maybeSplit) {
208
+ currentGroup = maybeSplit.currentGroup;
209
+ wordCounter = 0;
210
+ }
211
+ }
212
+
213
+ if (currentGroup.length > 0) {
214
+ sentenceGroups.push(currentGroup);
215
+ }
216
+
217
+ return sentenceGroups;
218
+ }
219
+
220
+ /**
221
+ * Checks if a token should end the current sentence.
222
+ *
223
+ * @param token - The token to check.
224
+ * @param linebreakMarker - The marker used for line breaks.
225
+ * @returns True if the sentence should end, false otherwise.
226
+ */
227
+ function shouldEndSentence(token: string, linebreakMarker: string): boolean {
228
+ return (
229
+ [".", "!", "?"].includes(token) ||
230
+ hasEndPunctuation(token, "?!") ||
231
+ token === linebreakMarker
232
+ );
233
+ }
234
+
235
+ /**
236
+ * Determines if processing should continue after encountering a period.
237
+ *
238
+ * @param token - The current token.
239
+ * @param tokens - All tokens.
240
+ * @param index - Current index in tokens array.
241
+ * @param wordCounter - Current word count in sentence.
242
+ * @returns True if should continue (not end sentence), false if should end sentence.
243
+ */
244
+ function shouldContinueAfterPeriod(
245
+ token: string,
246
+ tokens: string[],
247
+ index: number,
248
+ wordCounter: number,
249
+ ): boolean {
250
+ if (index + 1 >= tokens.length) {
251
+ return false; // End of tokens, should end sentence
252
+ }
253
+
254
+ // Special case: single letter abbreviations
255
+ if (token.length === 2 && isNaN(Number(token.charAt(0)))) {
256
+ return true;
257
+ }
258
+
259
+ // Check common abbreviations
260
+ if (isInCommonAbbreviationList(token)) {
261
+ return true;
262
+ }
263
+
264
+ const nextToken = tokens[index + 1];
265
+
266
+ if (isBeginsNewSentence(nextToken)) {
267
+ // Could be start of new sentence
268
+ if (isAbbreviatedTime(token, nextToken)) {
269
+ return true;
270
+ }
271
+
272
+ if (isAbbreviatedName(wordCounter, tokens.slice(index, index + 6))) {
273
+ return true;
274
+ }
275
+
276
+ if (isNumeric(nextToken) && isCustomAbbreviation(token)) {
277
+ return true;
278
+ }
279
+ } else {
280
+ // Doesn't look like new sentence
281
+ if (token.endsWith("..")) {
282
+ return true; // Ellipsis
283
+ }
284
+
285
+ if (isComplexAbbreviation(token)) {
286
+ return true;
287
+ }
288
+
289
+ if (isAbbreviatedName(wordCounter, tokens.slice(index, index + 5))) {
290
+ return true;
291
+ }
292
+ }
293
+
294
+ return false; // Default: end the sentence
295
+ }
296
+
297
+ /**
298
+ * Attempts to split a token that contains punctuation followed by a letter.
299
+ *
300
+ * @param token - The token to potentially split.
301
+ * @param currentGroup - The current sentence group.
302
+ * @param sentenceGroups - All sentence groups.
303
+ * @returns Object with new currentGroup if split occurred, null otherwise.
304
+ */
305
+ function maybeSplitAtInternalPunctuation(
306
+ token: string,
307
+ currentGroup: string[],
308
+ sentenceGroups: string[][],
309
+ ): { currentGroup: string[] } | null {
310
+ const boundaryIndex = token.search(/[.!?]/);
311
+
312
+ if (boundaryIndex > -1 && boundaryIndex < token.length - 1) {
313
+ const nextChar = token.charAt(boundaryIndex + 1);
314
+ if (nextChar.match(/[a-zA-Z]/)) {
315
+ const splitResult = [
316
+ token.slice(0, boundaryIndex + 1),
317
+ token.slice(boundaryIndex + 1),
318
+ ];
319
+
320
+ currentGroup.pop();
321
+ currentGroup.push(splitResult[0]);
322
+ sentenceGroups.push(currentGroup);
323
+
324
+ return { currentGroup: [splitResult[1]] };
325
+ }
326
+ }
327
+
328
+ return null;
329
+ }
330
+
331
+ /**
332
+ * Merges sentence groups that were incorrectly split on short abbreviations.
333
+ *
334
+ * @param sentenceGroups - Array of sentence groups to process.
335
+ * @returns Merged sentence groups.
336
+ */
337
+ function mergeSplitAbbreviations(sentenceGroups: string[][]): string[][] {
338
+ return sentenceGroups
339
+ .filter((group) => group.length > 0)
340
+ .reduce((output, group, index) => {
341
+ if (index === 0) {
342
+ output.push(group);
343
+ return output;
344
+ }
345
+
346
+ const previousGroup = output[output.length - 1];
347
+
348
+ // Check for short abbreviations that might have been split incorrectly
349
+ if (
350
+ previousGroup.length === 1 &&
351
+ /^.{1,2}[.]$/.test(previousGroup[0]) &&
352
+ !/[.]/.test(group[0])
353
+ ) {
354
+ output[output.length - 1] = previousGroup.concat(group);
355
+ } else {
356
+ output.push(group);
357
+ }
358
+
359
+ return output;
360
+ }, [] as string[][]);
361
+ }
362
+
363
+ /**
364
+ * Applies minimum and maximum size constraints to sentence groups.
365
+ *
366
+ * @param sentenceGroups - Array of sentence groups (token arrays).
367
+ * @param minSize - Minimum sentence length.
368
+ * @param maxSize - Maximum sentence length.
369
+ * @returns Array of final sentences as strings.
370
+ */
371
+ function applySizeConstraints(
372
+ sentenceGroups: string[][],
373
+ minSize: number,
374
+ maxSize: number,
375
+ ): string[] {
376
+ return sentenceGroups
377
+ .map((group, index) => {
378
+ let sentence = group.join(" ");
379
+
380
+ // Apply minSize constraint
381
+ if (sentence.length < minSize) {
382
+ if (index < sentenceGroups.length - 1) {
383
+ // Merge with next sentence
384
+ sentenceGroups[index + 1] = group.concat(sentenceGroups[index + 1]);
385
+ return null;
386
+ }
387
+ }
388
+ // Apply maxSize constraint
389
+ else if (sentence.length > maxSize) {
390
+ const lastPunctuation = sentence.lastIndexOf(".", maxSize);
391
+
392
+ if (lastPunctuation === -1) {
393
+ return sliceIntoChunks(sentence, maxSize);
394
+ }
395
+
396
+ if (lastPunctuation > minSize) {
397
+ // Split sentence and add remainder to next position
398
+ sentenceGroups.splice(
399
+ index + 1,
400
+ 0,
401
+ sentence
402
+ .slice(lastPunctuation + 1)
403
+ .trim()
404
+ .split(" "),
405
+ );
406
+ return sentence.slice(0, lastPunctuation + 1);
407
+ }
408
+ }
409
+
410
+ return sentence;
411
+ })
412
+ .flat()
413
+ .filter(Boolean) as string[];
414
+ }
415
+
416
+ /**
417
+ * Slices a string into chunks of a maximum size, attempting to split at word
418
+ * boundaries within a 20-character window of the maximum size.
419
+ *
420
+ * @param str - The string to be sliced.
421
+ * @param maxSize - The maximum length of each chunk.
422
+ * @returns An array of string chunks.
423
+ */
424
+ function sliceIntoChunks(str: string, maxSize: number): string[] {
425
+ const chunks: string[] = [];
426
+ let startIndex = 0;
427
+
428
+ while (startIndex < str.length) {
429
+ let endIndex = startIndex + maxSize;
430
+
431
+ if (endIndex < str.length) {
432
+ // Look for the last space within the last 20 characters
433
+ const lastSpaceIndex = str.lastIndexOf(" ", endIndex);
434
+ const searchStartIndex = Math.max(startIndex, endIndex - 20);
435
+
436
+ if (lastSpaceIndex >= searchStartIndex) {
437
+ endIndex = lastSpaceIndex;
438
+ }
439
+ } else {
440
+ endIndex = str.length;
441
+ }
442
+
443
+ chunks.push(str.slice(startIndex, endIndex).trim());
444
+ startIndex = endIndex + 1; // Skip the space
445
+ }
446
+
447
+ return chunks;
448
+ }
449
+
450
+ /**
451
+ * Checks if a word ends with a specific character or characters.
452
+ *
453
+ * @param word - The word to check.
454
+ * @param char - The character(s) to check for at the end of the word.
455
+ * @returns True if the word ends with the specified character(s), false otherwise.
456
+ */
457
+ function hasEndPunctuation(word: string, char: string): boolean {
458
+ return char.length > 1
459
+ ? char.includes(word.slice(-1))
460
+ : word.slice(-1) === char;
461
+ }
462
+
463
+ /**
464
+ * Checks if a string is capitalized or a number.
465
+ *
466
+ * @param str - The string to check.
467
+ * @returns True if the string is capitalized or a number, false otherwise.
468
+ */
469
+ function isCapitalizedOrNumeric(str: string): boolean {
470
+ return /^[A-Z][a-z].*/.test(str) || isNumeric(str);
471
+ }
472
+
473
+ /**
474
+ * Checks if a string is likely to begin a new sentence.
475
+ *
476
+ * @param str - The string to check.
477
+ * @returns True if the string is likely to begin a new sentence, false otherwise.
478
+ */
479
+ function isBeginsNewSentence(str: string): boolean {
480
+ return isCapitalizedOrNumeric(str) || /``|"|'/.test(str.substring(0, 2));
481
+ }
482
+
483
+ /**
484
+ * Checks if a word is in the list of 222 common abbreviations for various categories.
485
+ *
486
+ * @param str - The word to check, which gets cleaned and lowercased.
487
+ * @returns True if the string is a common abbreviation, false otherwise.
488
+ */
489
+ function isInCommonAbbreviationList(str: string): boolean {
490
+ const COMMON_ABBR_LIST = (
491
+ "adj,adm,adv,al,ala,alta,apr,arc,ariz,ark,art,assn,asst,attys,aug,ave,ba," +
492
+ "bart,bld,bldg,blvd,brig,bros,bsc,btw,cal,calif,capt,cc,cell,ch,cl,cmdr,co,col,colo,comdr,con," +
493
+ "conn,corp,cpl,cres,ct,dak,dec,del,dem,dept,det,dist,dphil,dr,drs,ed,eg,ens,eq,eqs,esp,esq,est," +
494
+ "etc,ex,exp,expy,ext,feb,fed,fig,figs,fla,fri,ft,fwy,fy,ga,gen,gov,hon,hosp,hr,hrs,hway,hwy,ia,id," +
495
+ "ida,ie,ill,inc,ind,ing,insp,is,jan,jr,jul,jun,kan,kans,ken,kg,km,kmph,ky,la,lt,ltd,ma,maj,man," +
496
+ "mar,mass,may,md,me,med,messrs,mex,mfg,mi,mich,min,minn,miss,mlle,mm,mme,mo,mol,mont,mr,mrs,ms," +
497
+ "msc,msgr,mssrs,mt,mtn,neb,nebr,nev,no,nos,nov,nr,oct,ok,okla,ont,op,ord,ore,p,pa,pd,pde,penn," +
498
+ "penna,pfc,ph,phd,pl,plz,pop,pp,prof,pvt,que,rd,ref,refs,rep,repr,reps,res,rev,rs,rt,sask,sat," +
499
+ "sec,secs,sen,sens,sep,sept,sfc,sgt,sr,st,sun,supt,surg,tce,tenn,tex,th,thu,thur,thurs,trans,tu," +
500
+ "tue,tues,univ,us,usafa,ut,v,va,ver,viz,vol,vs,vt,wash,wed,wis,wisc,wy,wyo"
501
+ ).split(",");
502
+
503
+ const cleaned = str
504
+ .toLowerCase()
505
+ .replace(/[-'`~!@#$%^&*()_|+=?;:'",.<>\{\}\[\]\\\/]/gi, "");
506
+ return COMMON_ABBR_LIST.includes(cleaned);
507
+ }
508
+
509
+ /**
510
+ * Checks if a word is an abbreviated time (a.m. or p.m.) followed by 'day'.
511
+ *
512
+ * @param word - The current word.
513
+ * @param nextWord - The next word in the sequence.
514
+ * @returns True if it's an abbreviated time followed by 'day', false otherwise.
515
+ */
516
+ function isAbbreviatedTime(word: string, nextWord: string): boolean {
517
+ if (word === "a.m." || word === "p.m.") {
518
+ const nextWordEnd = nextWord.replace(/\W+/g, "").slice(-3).toLowerCase();
519
+ return nextWordEnd === "day";
520
+ }
521
+ return false;
522
+ }
523
+
524
+ /**
525
+ * Checks if a word or phrase is a complex abbreviation (e.g., U.S., U.K., Ph.D.).
526
+ *
527
+ * @param word - The word or phrase to check.
528
+ * @returns True if it's a complex abbreviation, false otherwise.
529
+ */
530
+ function isComplexAbbreviation(word: string): boolean {
531
+ // Check for common multi-part abbreviations like U.S. or U.K.
532
+ if (/^([A-Za-z]\.){2,}[A-Za-z]?\.?(\s*\([^)]+\))?$/.test(word)) {
533
+ return true;
534
+ }
535
+
536
+ const cleaned = word.replace(/[\(\)\[\]\{\}]/g, "");
537
+ const matches = cleaned.match(/(.\.)*/);
538
+ return matches !== null && matches[0].length > 0;
539
+ }
540
+
541
+ /**
542
+ * Checks if a string is a custom abbreviation (short or capitalized).
543
+ *
544
+ * @param str - The string to check.
545
+ * @returns True if it's a custom abbreviation, false otherwise.
546
+ */
547
+ function isCustomAbbreviation(str: string): boolean {
548
+ return str.length <= 3 || isCapitalizedOrNumeric(str);
549
+ }
550
+
551
+ /**
552
+ * Checks if a sequence of words represents an abbreviated name (e.g., J. R. R. Tolkien).
553
+ *
554
+ * @param wordCount - The current word count in the sentence.
555
+ * @param words - A sequence of words to check.
556
+ * @returns True if the sequence represents an abbreviated name, false otherwise.
557
+ */
558
+ function isAbbreviatedName(wordCount: number, words: string[]): boolean {
559
+ if (words.length > 0) {
560
+ if (
561
+ wordCount < 5 &&
562
+ words[0].length < 6 &&
563
+ isCapitalizedOrNumeric(words[0])
564
+ ) {
565
+ return true;
566
+ }
567
+ const capitalizedCount = words.filter((str) =>
568
+ /[A-Z]/.test(str.charAt(0)),
569
+ ).length;
570
+ return capitalizedCount >= 3;
571
+ }
572
+ return false;
573
+ }
574
+
575
+ /**
576
+ * Checks if a string is numeric, optionally starting from a specific position.
577
+ *
578
+ * @param str - The string to check.
579
+ * @param startPos - The position to start checking from (optional).
580
+ * @returns True if the string is numeric, false otherwise.
581
+ */
582
+ function isNumeric(str: string, startPos?: number): boolean {
583
+ if (startPos != null) {
584
+ str = str.slice(startPos - 1, startPos + 2);
585
+ }
586
+ return !isNaN(Number(str));
587
+ }
588
+
589
+ /**
590
+ * Checks if a string matches a phone number pattern.
591
+ *
592
+ * @param str - The string to check.
593
+ * @returns True if the string matches a phone number pattern, false otherwise.
594
+ */
595
+ function isPhoneNumber(str: string): boolean {
596
+ return new RegExp(
597
+ "^(?:(?:\\+?1\\s*(?:[.-]\\s*)?)?(?:(\\s*([2-9]1[02-9]|[2-9][02-8]1|[2-9][02-8]" +
598
+ "[02-9])\\s*)|([2-9]1[02-9]|[2-9][02-8]1|[2-9][02-8][02-9]))\\s*(?:[.-]\\s*)?)?([2-9]1" +
599
+ "[02-9]|[2-9][02-9]1|[2-9][02-9]{2})\\s*(?:[.-]\\s*)?([0-9]{4})(?:\\s*(?:#|x\\.?|ext\\.?" +
600
+ "|extension)\\s*(\\d+))?$",
601
+ ).test(str);
602
+ }
603
+
604
+ /**
605
+ * Checks if a string is a valid URL.
606
+ *
607
+ * @param str - The string to check.
608
+ * @returns True if the string is a valid URL, false otherwise.
609
+ */
610
+ function isValidUrl(str: string): boolean {
611
+ return /[-a-zA-Z0-9@:%._\+~#=]{2,256}\.[a-z]{2,6}\b([-a-zA-Z0-9@:%_\+.~#?&//=]*)/.test(
612
+ str,
613
+ );
614
+ }