extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,38 @@
1
+ // import { pipeline, type TextGenerationSingle } from "@huggingface/transformers";
2
+
3
+ // /**
4
+ // * Generates the next words for a given prompt using a text generation model.
5
+ // *
6
+ // * The DistilGPT2 model is a distilled, smaller version of OpenAI's GPT-2, featuring
7
+ // * around 82 million parameters (~80MB), and available in Hugging Face Transformers
8
+ // * for efficient next word (next-token) prediction tasks in English text. It is
9
+ // * well-suited for sequence generation and can be used as a next word prediction
10
+ // * model with code similar to GPT2, but offers faster and lighter inference compared
11
+ // * GPT-2 (124M parameter) model, preserving most capabilities but running
12
+ // * significantly faster and requiring less memory.
13
+ // * @see https://huggingface.co/Xenova/distilgpt2
14
+ // * @param {string} prompt - The input prompt to complete.
15
+ // * @param {Object} [options] - Generation options.
16
+ // * @param {number} [options.maxTokens=16] - Maximum number of new tokens to generate.
17
+ // * @param {string} [options.model='Xenova/distilgpt2'] - The model identifier to use.
18
+ // * @returns {Promise<string>} - The generated text completions.
19
+ // */
20
+ // export async function predictNextWordsWithSmallLocalModel(
21
+ // prompt: string,
22
+ // { model = "Xenova/distilgpt2", maxTokens = 16, modelParams = {} } = {}
23
+ // ): Promise<string> {
24
+ // if (!prompt) throw new Error("Prompt is required");
25
+ // return (
26
+ // await (
27
+ // await pipeline("text-generation", model, {
28
+ // dtype: "fp16",
29
+ // device: "cpu",
30
+ // ...modelParams,
31
+ // })
32
+ // )(prompt, { max_new_tokens: maxTokens })
33
+ // )
34
+ // .map((res) => (res as TextGenerationSingle).generated_text)
35
+ // .join(" ")
36
+ // .replace(prompt, "")
37
+ // .trim();
38
+ // }
@@ -0,0 +1,435 @@
1
+ /**
2
+ * Provides search query autocomplete/suggestions from various search engines.
3
+ */
4
+
5
+ import { parseHTML } from "linkedom";
6
+
7
+ // Some suggest APIs (notably Google) return 403 Forbidden for requests
8
+ // without a browser-like User-Agent, especially from datacenter/Cloudflare
9
+ // egress IPs. Sent by default; per-backend headers can still override it.
10
+ const DEFAULT_USER_AGENT =
11
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36";
12
+
13
+ /**
14
+ * Fetch wrapper compatible with grab-url interface
15
+ */
16
+ async function grab(url: string, options: any = {}): Promise<any> {
17
+ const timeout = options.timeout ? options.timeout * 1000 : 5000;
18
+ const controller = new AbortController();
19
+ const timeoutId = setTimeout(() => controller.abort(), timeout);
20
+
21
+ try {
22
+ const response = await fetch(url, {
23
+ ...options,
24
+ headers: {
25
+ "User-Agent": DEFAULT_USER_AGENT,
26
+ ...options.headers,
27
+ },
28
+ signal: controller.signal,
29
+ });
30
+
31
+ clearTimeout(timeoutId);
32
+
33
+ if (!response.ok) {
34
+ throw new Error(`HTTP ${response.status}: ${response.statusText}`);
35
+ }
36
+
37
+ if (options.responseType === "json") {
38
+ return { data: await response.json() };
39
+ }
40
+ if (options.responseType === "arraybuffer") {
41
+ return await response.arrayBuffer();
42
+ }
43
+ if (options.responseType === "text") {
44
+ return await response.text();
45
+ }
46
+
47
+ // Default to json for backwards compatibility
48
+ try {
49
+ return { data: await response.json() };
50
+ } catch {
51
+ return await response.text();
52
+ }
53
+ } catch (error) {
54
+ clearTimeout(timeoutId);
55
+ throw error;
56
+ }
57
+ }
58
+
59
+ /**
60
+ * Autocomplete function type
61
+ */
62
+ type AutocompleteFunction = (
63
+ query: string,
64
+ locale?: string,
65
+ ) => Promise<string[]>;
66
+
67
+ /**
68
+ * Get autocomplete suggestions with error handling
69
+ */
70
+ async function getSuggestions(url: string, options: any = {}): Promise<any> {
71
+ try {
72
+ const response = await grab(url, {
73
+ timeout: 3,
74
+ responseType: options.responseType || "json",
75
+ ...options,
76
+ });
77
+
78
+ if (typeof response === "object" && "data" in response) {
79
+ return response.data;
80
+ }
81
+ return response;
82
+ } catch (error) {
83
+ console.error("Autocomplete error:", error);
84
+ return null;
85
+ }
86
+ }
87
+
88
+ /**
89
+ * Baidu autocomplete
90
+ */
91
+ export async function baidu(
92
+ query: string,
93
+ _locale?: string,
94
+ ): Promise<string[]> {
95
+ const params = new URLSearchParams({
96
+ ie: "utf-8",
97
+ json: "1",
98
+ prod: "pc",
99
+ wd: query,
100
+ });
101
+
102
+ const url = `https://www.baidu.com/sugrec?${params.toString()}`;
103
+ const data = await getSuggestions(url);
104
+ const results: string[] = [];
105
+
106
+ if (data && data.g) {
107
+ for (const item of data.g) {
108
+ results.push(item.q);
109
+ }
110
+ }
111
+
112
+ return results;
113
+ }
114
+
115
+ /**
116
+ * Brave autocomplete
117
+ */
118
+ export async function brave(
119
+ query: string,
120
+ _locale?: string,
121
+ ): Promise<string[]> {
122
+ const params = new URLSearchParams({ q: query });
123
+ const url = `https://search.brave.com/api/suggest?${params.toString()}`;
124
+
125
+ const data = await getSuggestions(url, {
126
+ headers: {
127
+ Cookie: "country=all",
128
+ },
129
+ });
130
+
131
+ if (data && Array.isArray(data) && data.length > 1) {
132
+ return data[1];
133
+ }
134
+
135
+ return [];
136
+ }
137
+
138
+ /**
139
+ * DuckDuckGo autocomplete
140
+ */
141
+ export async function duckduckgo(
142
+ query: string,
143
+ locale: string = "en-US",
144
+ ): Promise<string[]> {
145
+ // Extract region code (e.g., 'us-en' from 'en-US')
146
+ const region = locale.toLowerCase().split("-").reverse().join("-");
147
+
148
+ const params = new URLSearchParams({
149
+ q: query,
150
+ kl: region,
151
+ });
152
+
153
+ const url = `https://duckduckgo.com/ac/?type=list&${params.toString()}`;
154
+ const data = await getSuggestions(url);
155
+
156
+ if (data && Array.isArray(data) && data.length > 1) {
157
+ return data[1];
158
+ }
159
+
160
+ return [];
161
+ }
162
+
163
+ /**
164
+ * Google autocomplete
165
+ */
166
+ export async function google(
167
+ query: string,
168
+ locale: string = "en",
169
+ ): Promise<string[]> {
170
+ // Map locale to Google subdomain
171
+ const subdomainMap: { [key: string]: string } = {
172
+ de: "google.de",
173
+ fr: "google.fr",
174
+ es: "google.es",
175
+ it: "google.it",
176
+ nl: "google.nl",
177
+ pt: "google.pt",
178
+ ru: "google.ru",
179
+ ja: "google.co.jp",
180
+ zh: "google.com.hk",
181
+ ko: "google.co.kr",
182
+ };
183
+
184
+ const lang = locale.split("-")[0];
185
+ const subdomain = subdomainMap[lang] || "google.com";
186
+
187
+ const params = new URLSearchParams({
188
+ q: query,
189
+ client: "gws-wiz",
190
+ hl: lang,
191
+ });
192
+
193
+ const url = `https://${subdomain}/complete/search?${params.toString()}`;
194
+ const response = await getSuggestions(url, { responseType: "text" });
195
+
196
+ const results: string[] = [];
197
+
198
+ if (response) {
199
+ try {
200
+ // Extract JSON from response
201
+ const jsonStart = response.indexOf("[");
202
+ const jsonEnd = response.lastIndexOf("]") + 1;
203
+
204
+ if (jsonStart >= 0 && jsonEnd > jsonStart) {
205
+ const jsonText = response.substring(jsonStart, jsonEnd);
206
+ const data = JSON.parse(jsonText);
207
+
208
+ if (data[0]) {
209
+ for (const item of data[0]) {
210
+ // Parse HTML entities
211
+ const { document } = parseHTML(item[0]);
212
+ const text = document.body?.textContent?.trim();
213
+ if (text) {
214
+ results.push(text);
215
+ }
216
+ }
217
+ }
218
+ }
219
+ } catch (e) {
220
+ // Ignore parse errors
221
+ }
222
+ }
223
+
224
+ return results;
225
+ }
226
+
227
+ /**
228
+ * Qwant autocomplete
229
+ */
230
+ export async function qwant(
231
+ query: string,
232
+ locale: string = "en_US",
233
+ ): Promise<string[]> {
234
+ const params = new URLSearchParams({
235
+ q: query,
236
+ locale: locale.replace("-", "_"),
237
+ version: "2",
238
+ });
239
+
240
+ const url = `https://api.qwant.com/v3/suggest?${params.toString()}`;
241
+ const data = await getSuggestions(url);
242
+ const results: string[] = [];
243
+
244
+ if (data && data.status === "success" && data.data && data.data.items) {
245
+ for (const item of data.data.items) {
246
+ results.push(item.value);
247
+ }
248
+ }
249
+
250
+ return results;
251
+ }
252
+
253
+ /**
254
+ * Startpage autocomplete
255
+ */
256
+ export async function startpage(
257
+ query: string,
258
+ locale: string = "en",
259
+ ): Promise<string[]> {
260
+ const langMap: { [key: string]: string } = {
261
+ da: "dansk",
262
+ de: "deutsch",
263
+ en: "english",
264
+ es: "espanol",
265
+ fr: "francais",
266
+ nb: "norsk",
267
+ nl: "nederlands",
268
+ pl: "polski",
269
+ pt: "portugues",
270
+ sv: "svenska",
271
+ };
272
+
273
+ const baseLang = locale.split("-")[0];
274
+ const lui = langMap[baseLang] || "english";
275
+
276
+ const params = new URLSearchParams({
277
+ q: query,
278
+ format: "opensearch",
279
+ segment: "startpage.defaultffx",
280
+ lui: lui,
281
+ });
282
+
283
+ const url = `https://www.startpage.com/suggestions?${params.toString()}`;
284
+ const data = await getSuggestions(url, {
285
+ headers: {
286
+ "User-Agent":
287
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
288
+ },
289
+ });
290
+
291
+ if (
292
+ data &&
293
+ Array.isArray(data) &&
294
+ data.length >= 2 &&
295
+ Array.isArray(data[1])
296
+ ) {
297
+ return data[1];
298
+ }
299
+
300
+ return [];
301
+ }
302
+
303
+ /**
304
+ * Wikipedia autocomplete
305
+ */
306
+ export async function wikipedia(
307
+ query: string,
308
+ locale: string = "en",
309
+ ): Promise<string[]> {
310
+ const langMap: { [key: string]: string } = {
311
+ en: "en.wikipedia.org",
312
+ de: "de.wikipedia.org",
313
+ fr: "fr.wikipedia.org",
314
+ es: "es.wikipedia.org",
315
+ it: "it.wikipedia.org",
316
+ nl: "nl.wikipedia.org",
317
+ pt: "pt.wikipedia.org",
318
+ ru: "ru.wikipedia.org",
319
+ ja: "ja.wikipedia.org",
320
+ zh: "zh.wikipedia.org",
321
+ ar: "ar.wikipedia.org",
322
+ ko: "ko.wikipedia.org",
323
+ };
324
+
325
+ const lang = locale.split("-")[0];
326
+ const netloc = langMap[lang] || "en.wikipedia.org";
327
+
328
+ const params = new URLSearchParams({
329
+ action: "opensearch",
330
+ format: "json",
331
+ formatversion: "2",
332
+ search: query,
333
+ namespace: "0",
334
+ limit: "10",
335
+ });
336
+
337
+ const url = `https://${netloc}/w/api.php?${params.toString()}`;
338
+ const data = await getSuggestions(url);
339
+
340
+ if (data && Array.isArray(data) && data.length > 1) {
341
+ return data[1];
342
+ }
343
+
344
+ return [];
345
+ }
346
+
347
+ /**
348
+ * Yandex autocomplete
349
+ */
350
+ export async function yandex(
351
+ query: string,
352
+ _locale?: string,
353
+ ): Promise<string[]> {
354
+ const params = new URLSearchParams({ part: query });
355
+ const url = `https://suggest.yandex.com/suggest-ff.cgi?${params.toString()}`;
356
+
357
+ const data = await getSuggestions(url);
358
+
359
+ if (data && Array.isArray(data) && data.length > 1) {
360
+ return data[1];
361
+ }
362
+
363
+ return [];
364
+ }
365
+
366
+ /**
367
+ * Available autocomplete backends
368
+ */
369
+ export const backends: { [key: string]: AutocompleteFunction } = {
370
+ baidu,
371
+ brave,
372
+ duckduckgo,
373
+ google,
374
+ qwant,
375
+ startpage,
376
+ wikipedia,
377
+ yandex,
378
+ };
379
+
380
+ /**
381
+ * Get autocomplete suggestions from a specific backend
382
+ *
383
+ * @param backendName - Name of the autocomplete backend
384
+ * @param query - Search query
385
+ * @param locale - Locale/language code (e.g., 'en-US', 'de-DE')
386
+ * @returns Array of suggestion strings
387
+ */
388
+ export async function searchAutocomplete(
389
+ backendName: string,
390
+ query: string,
391
+ locale: string = "en-US",
392
+ ): Promise<string[]> {
393
+ const backend = backends[backendName];
394
+
395
+ if (!backend) {
396
+ console.warn(`Autocomplete backend '${backendName}' not found`);
397
+ return [];
398
+ }
399
+
400
+ try {
401
+ return await backend(query, locale);
402
+ } catch (error) {
403
+ console.error(`Autocomplete error for ${backendName}:`, error);
404
+ return [];
405
+ }
406
+ }
407
+
408
+ /**
409
+ * Get autocomplete suggestions from multiple backends and merge them
410
+ *
411
+ * @param backendNames - Array of backend names to query
412
+ * @param query - Search query
413
+ * @param locale - Locale/language code
414
+ * @returns Merged and deduplicated array of suggestions
415
+ */
416
+ export async function searchAutocompleteMulti(
417
+ backendNames: string[],
418
+ query: string,
419
+ locale: string = "en-US",
420
+ ): Promise<string[]> {
421
+ const promises = backendNames.map((name) =>
422
+ searchAutocomplete(name, query, locale),
423
+ );
424
+ const results = await Promise.all(promises);
425
+
426
+ // Merge and deduplicate
427
+ const merged = new Set<string>();
428
+ for (const result of results) {
429
+ for (const suggestion of result) {
430
+ merged.add(suggestion);
431
+ }
432
+ }
433
+
434
+ return Array.from(merged);
435
+ }
@@ -0,0 +1,137 @@
1
+ /**
2
+ * @fileoverview Utility for suggesting word and phrase completions based on a trie model.
3
+ * Used for search autocomplete and real-time query suggestions.
4
+ */
5
+ export interface SuggestCompletionsOptions {
6
+ phrasesModel: Record<string, Record<string, Array<[string | null, number, number]>>>;
7
+ limitMaxResults?: number;
8
+ numberOfLastWordsToCheck?: number;
9
+ optionShowFullQuery?: boolean;
10
+ }
11
+
12
+ export interface SuggestionResult {
13
+ name?: string;
14
+ word?: string;
15
+ phrase?: string;
16
+ }
17
+
18
+ /**
19
+ * ### Autocomplete Topic Phrase Completions
20
+ * <img width="350px" src="https://i.imgur.com/0k5mO76.png" />
21
+ *
22
+ * Completes the query with the most likely next words for phrases.
23
+ * If typing 2+ letters of a word, returns all possible words matching those few letters.
24
+ *
25
+ *
26
+ * @param {string} query - The input query which can be pertial words or phrases.
27
+ * @param {Object} [options]
28
+ * @param {Object} options.phrasesModel - A custom phrases model to use for autocomplete suggestions.
29
+ * @param {number} options.limitMaxResults default=10 - The maximum number of autocomplete suggestions to return.
30
+ * @param {number} options.numberOfLastWordsToCheck default=5 - The number of last words in the query to check for phrase completions.
31
+ * @returns {Promise<Array<Object>>} An array of autocomplete suggestions, each containing either a 'phrase' or 'word' property.
32
+ * @example
33
+ * // Basic usage
34
+ * const suggestions = await suggestNextWordCompletions("self att");
35
+ * // Possible output: [{ phrase: "self attention" }, { phrase: "self attract" }, { phrase: "self attack" }]
36
+ *
37
+ * @example
38
+ * // Using options
39
+ * const customModel = await import("./custom-phrases-model.json");
40
+ * const suggestions = await suggestNextWordCompletions("artificial int", {
41
+ * phrasesModel: customModel,
42
+ * limitMaxResults: 5,
43
+ * numberOfLastWordsToCheck: 3
44
+ * });
45
+ * // Possible output: [{ phrase: "artificial intelligence" }, { phrase: "artificial interpretation" }]
46
+ *
47
+ * @author [vtempest (2025)](https://github.com/vtempest)
48
+ * @category Topics
49
+ */
50
+ export async function suggestNextWordCompletions(
51
+ query: string,
52
+ options: Partial<SuggestCompletionsOptions> = {},
53
+ ): Promise<SuggestionResult[] | undefined> {
54
+ var {
55
+ phrasesModel, //pass in remote model
56
+ limitMaxResults = 10, //limit the number of results
57
+ numberOfLastWordsToCheck = 5, // check last few words for their phrase completions
58
+ optionShowFullQuery = true, //show full query in the result
59
+ } = options;
60
+
61
+ if (!phrasesModel) {
62
+ throw new Error("Missing phrasesModel");
63
+ }
64
+
65
+ // if (!phrasesModel)
66
+ // phrasesModel = await import ( "../wordlists/wiki-phrases-model-240k.json");
67
+
68
+ //strip non-alphanumeric characters from query and -'
69
+ query = query.trim().replace(/[^a-zA-Z0-9\s\-\']/g, "");
70
+
71
+ //split into words
72
+ var words = query.toLowerCase().split(/\W+/);
73
+
74
+ if (query.length < 1) return;
75
+
76
+ const autocompleteGroups: SuggestionResult[][] = [];
77
+
78
+ const lastWords = words.slice(-numberOfLastWordsToCheck);
79
+
80
+ for (var i = 0; i < lastWords.length; i++) {
81
+ // Iterate over the tail-window we just sliced; using the full `words` array here
82
+ // skews matching when `numberOfLastWordsToCheck` is smaller than query length.
83
+ var word = lastWords[i];
84
+
85
+ //Find next word query completion list
86
+ var firstTwoLetters = word.slice(0, 2);
87
+ const possiblePhrases = phrasesModel[firstTwoLetters]
88
+ ? phrasesModel[firstTwoLetters][word]
89
+ : null;
90
+
91
+ if (possiblePhrases) {
92
+ var nextWords = words
93
+ .slice(i + 1)
94
+ .join(" ")
95
+ .trim();
96
+
97
+ const phraseMatches = possiblePhrases
98
+ .filter((phrase) =>
99
+ phrase[0] ? phrase[0].startsWith(nextWords) : false
100
+ )
101
+ .map((phrase) => ({ phrase: word + " " + phrase[0] }));
102
+
103
+ const possiblePhrasesInput = JSON.parse(JSON.stringify(phraseMatches));
104
+ autocompleteGroups.push(possiblePhrasesInput);
105
+ }
106
+ }
107
+
108
+ //if typing 2 letters of first word,
109
+ // return all possible first words matched what is typed in
110
+ let lastWord = words[words.length - 1];
111
+ autocompleteGroups.push(
112
+ Object.keys(phrasesModel[lastWord.slice(0, 2)] || {})
113
+ .filter((phrase) => phrase.startsWith(lastWord))
114
+ .map((phrase) => ({ word: phrase }))
115
+ );
116
+
117
+ let autocompletes: SuggestionResult[] = autocompleteGroups.flat(2).slice(0, limitMaxResults);
118
+
119
+ if (optionShowFullQuery)
120
+ autocompletes = autocompletes
121
+ .map((s) => {
122
+ var name;
123
+ if (s.word) name = words.slice(0, -1).join(" ") + " " + s.word;
124
+ if (s.phrase) {
125
+ name =
126
+ words
127
+ .join(" ")
128
+ .slice(0, words.join(" ").lastIndexOf(s.phrase.split(" ")[0])) +
129
+ " " +
130
+ s.phrase;
131
+ }
132
+ return { name };
133
+ })
134
+ .filter(Boolean);
135
+
136
+ return autocompletes;
137
+ }