extract-webpage 1.2.50 → 1.2.52

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,470 +0,0 @@
1
- /**
2
- * @module research/search/public-searxng
3
- * @description Research library module.
4
- */
5
- import { getDomainWithoutSuffix } from "tldts";
6
- import { parseDate } from "chrono-node";
7
-
8
- /**
9
- * Fetch wrapper compatible with grab-url interface
10
- */
11
- async function grab(url: string, params: any = {}): Promise<any> {
12
- // Build query string from params
13
- const queryParams = new URLSearchParams();
14
- for (const [key, value] of Object.entries(params)) {
15
- if (value !== undefined && value !== null && value !== false) {
16
- queryParams.append(key, String(value));
17
- }
18
- }
19
-
20
- const fullUrl = queryParams.toString() ? `${url}?${queryParams}` : url;
21
- const timeout = params.timeout ? params.timeout * 1000 : 10000;
22
- const controller = new AbortController();
23
- const timeoutId = setTimeout(() => controller.abort(), timeout);
24
-
25
- try {
26
- const response = await fetch(fullUrl, {
27
- signal: controller.signal,
28
- });
29
-
30
- clearTimeout(timeoutId);
31
-
32
- if (!response.ok) {
33
- throw new Error(`HTTP ${response.status}: ${response.statusText}`);
34
- }
35
-
36
- const contentType = response.headers.get('content-type');
37
- if (contentType?.includes('application/json')) {
38
- return { data: await response.json() };
39
- }
40
- return await response.text();
41
- } catch (error) {
42
- clearTimeout(timeoutId);
43
- throw error;
44
- }
45
- }
46
-
47
- /**
48
- * Search Web via SearXNG metasearch of all major search engines.
49
- */
50
- export async function searchWeb(
51
- query: string,
52
- options: SearchOptions = {},
53
- ): Promise<SearxngSearchResult[] | SearchResponse> {
54
- const {
55
- category = "general",
56
- recency,
57
- privateSearxng = null,
58
- maxRetries = 3,
59
- page = 1,
60
- safesearch = false,
61
- lang = "en-US",
62
- proxy = null,
63
- } = options;
64
-
65
- const CATEGORY_LIST = [
66
- "general",
67
- "news",
68
- "videos",
69
- "images",
70
- "science",
71
- "it",
72
- "files",
73
- "social+media",
74
- ];
75
- const RECENCY_ALLOWED_LIST = ["day", "week", "month", "year"];
76
-
77
- const SEARX_DOMAINS = [
78
- "baresearch.org",
79
- "copp.gg",
80
- "darmarit.org",
81
- "etsi.me",
82
- "fairsuch.net",
83
- "nogoo.me",
84
- "northboot.xyz",
85
- "nyc1.sx.ggtyler.dev",
86
- "ooglester.com",
87
- "opnxng.com",
88
- "paulgo.io",
89
- "priv.au",
90
- "s.trung.fun",
91
- "search.blitzw.in",
92
- "search.charliewhiskey.net",
93
- "search.citw.lgbt",
94
- "search.darkness.services",
95
- "search.datura.network",
96
- "search.dotone.nl",
97
- "search.gcomm.ch",
98
- "search.hbubli.cc",
99
- "search.im-in.space",
100
- "search.incogniweb.net",
101
- "search.inetol.net",
102
- "search.leptons.xyz",
103
- "search.nadeko.net",
104
- "search.ngn.tf",
105
- "search.ononoki.org",
106
- "search.privacyredirect.com",
107
- "search.sapti.me",
108
- "search.rowie.at",
109
- "search.projectsegfau.lt",
110
- "search.tommy-tran.com",
111
- "searx.aleteoryx.me",
112
- "searx.ankha.ac",
113
- "searx.be",
114
- "searx.colbster937.dev",
115
- "searx.daetalytica.io",
116
- "searx.dresden.network",
117
- "searx.foss.family",
118
- "searx.hu",
119
- "searx.juancord.xyz",
120
- "searx.lunar.icu",
121
- "searx.mxchange.org",
122
- "searx.namejeff.xyz",
123
- "searx.oakleycord.dev",
124
- "searx.ro",
125
- "searx.sev.monster",
126
- "searx.thefloatinglab.world",
127
- "searx.tiekoetter.com",
128
- "searx.tuxcloud.net",
129
- "searx.work",
130
- "searx.zhenyapav.com",
131
- "searxng.hweeren.com",
132
- "searxng.online",
133
- "searxng.shreven.org",
134
- "searxng.site",
135
- "skyrimhater.com",
136
- "sx.ca.zorby.top",
137
- "sx.catgirl.cloud",
138
- "sx.thatxtreme.dev",
139
- "sx.zorby.top",
140
- "xo.wtf",
141
- ];
142
-
143
- //select a random domain if none is provided
144
- const searchDomain =
145
- privateSearxng ||
146
- "https://" +
147
- SEARX_DOMAINS[Math.floor(Math.random() * SEARX_DOMAINS.length)];
148
-
149
- var categoryName = categoryName == "tech" ? (categoryName = "it") : category;
150
-
151
- let url = `${searchDomain}/search`;
152
-
153
- if (privateSearxng) url += "&format=json";
154
-
155
- //on cloudflare to avoid "Too many redirects" change SSL mode to Full
156
- if (proxy && !privateSearxng) url = proxy + url;
157
-
158
- let resultHTML: any;
159
- try {
160
- resultHTML = await grab(searchDomain + "/search", {
161
- q: encodeURIComponent(query),
162
- ["category_" + categoryName]: 1,
163
- language: lang,
164
- [privateSearxng && "format"]: "json",
165
- [recency && RECENCY_ALLOWED_LIST.includes(recency) ? "time_range" : ""]:
166
- recency,
167
- safesearch: safesearch ? "1" : "0",
168
- pageno: page,
169
- headers: {
170
- "accept-language": lang + ",en;q=0.9",
171
- },
172
- });
173
- } catch (error: any) {
174
- const errorMsg = error instanceof Error ? error.message : String(error);
175
- console.warn(`[searchWeb] Failed to fetch from SearXNG domain "${searchDomain}": ${errorMsg}`);
176
- if (maxRetries > 0) {
177
- console.log(`[searchWeb] Retrying with another instance... (${maxRetries} retries left)`);
178
- return await searchWeb(query, {
179
- ...options,
180
- maxRetries: maxRetries - 1,
181
- });
182
- }
183
- console.error(`[searchWeb] All retries exhausted. Returning empty results.`);
184
- return [];
185
- }
186
-
187
- if (privateSearxng) {
188
- let parsedData: any;
189
-
190
- // Check if resultHTML is already an object (grab-url auto-parsed JSON)
191
- if (typeof resultHTML === "object" && resultHTML !== null) {
192
- parsedData = resultHTML;
193
- } else if (typeof resultHTML === "string") {
194
- // It's a string, try to parse it
195
- if (!resultHTML.startsWith("{")) {
196
- console.warn(
197
- "Private SearXNG instance did not return valid JSON, falling back or returning empty",
198
- );
199
- return { results: [], suggestions: [], infoboxes: [] };
200
- }
201
-
202
- try {
203
- parsedData = JSON.parse(resultHTML);
204
- } catch (e) {
205
- console.error("Failed to parse JSON from private instance", e);
206
- return { results: [], suggestions: [], infoboxes: [] };
207
- }
208
- } else {
209
- console.error("Unexpected resultHTML type:", typeof resultHTML);
210
- return { results: [], suggestions: [], infoboxes: [] };
211
- }
212
-
213
- let { results, suggestions, infoboxes } = parsedData;
214
-
215
- results = results.map((result: any) => {
216
- let title = result.title.replace(/<\/?[^>]+(>|$)/g, "");
217
-
218
- const TITLE_SPLITTERS_RE = /( [|\-\/:\u00bb] )|( - )|(\|)/;
219
-
220
- if (TITLE_SPLITTERS_RE.test(title)) {
221
- const splitTitle = title.split(TITLE_SPLITTERS_RE);
222
- // Handle breadcrumbed titles
223
- if (splitTitle.length >= 2) {
224
- const longestPart = splitTitle.reduce(
225
- (acc: string, part: string) =>
226
- part?.length > acc?.length ? part : acc,
227
- "",
228
- );
229
- if (longestPart.length > 10) {
230
- title = longestPart;
231
- }
232
- }
233
- }
234
-
235
- title = convertURLSafeHTMLToHTML(title);
236
- let urlPtr = result.url.replace(/&amp;/g, "&");
237
-
238
- // Validate URL has path component, not just domain
239
- try {
240
- const parsedUrl = new URL(urlPtr);
241
- // If URL is just domain (path is just "/"), log warning
242
- if (parsedUrl.pathname === "/" || parsedUrl.pathname === "") {
243
- console.warn(`[searchWeb] Result URL is domain-only: ${urlPtr}. Full result:`, JSON.stringify(result).slice(0, 200));
244
- }
245
- } catch (e) {
246
- console.error(`[searchWeb] Invalid URL in result: ${urlPtr}`);
247
- }
248
-
249
- const snippet = result.content?.replace(/<\/?[^>]+(>|$)/g, "");
250
- const thumbnail = result.thumbnail;
251
- const score = Math.round(result.score * 100) / 100;
252
-
253
- const domain = result.url
254
- ?.replace(/(http:\/\/|https:\/\/|www.)/gi, "")
255
- .split("/")[0];
256
-
257
- let date: string | undefined = undefined;
258
- let source: string | undefined = undefined;
259
-
260
- if (typeof result.metadata === "string") {
261
- const parts = result.metadata.split("|").map((s: string) => s.trim());
262
- if (parts.length > 1) {
263
- // Basic check
264
- const dateObj = parseDate(result.metadata);
265
- date = dateObj ? dateObj.toISOString().split("T")[0] : undefined;
266
- const sourcePart = parts[1]; // assuming second part might be source
267
- source = sourcePart || null;
268
- }
269
- }
270
-
271
- if (!source && domain) {
272
- source =
273
- getDomainWithoutSuffix(domain)?.replace(/\b\w/g, (l) =>
274
- l.toUpperCase(),
275
- ) || undefined;
276
- if (source && source.length < 5) source = source.toUpperCase();
277
- }
278
-
279
- const favicon = `https://s2.googleusercontent.com/s2/favicons?domain_url=${result.url}`;
280
- // const favicon =
281
- // "https://www.google.com/s2/favicons?domain=" +
282
- // result.url.match(
283
- // /^(?:https?:\/\/)?(?:www\.)?([^/:?\s]+)(?:[/:?]|$)/i,
284
- // )?.[0] +
285
- // "&sz=16";
286
-
287
- return {
288
- title,
289
- url: urlPtr,
290
- snippet,
291
- score,
292
- ...(date ? { date } : {}),
293
- ...(source ? { source } : {}),
294
- domain,
295
- favicon,
296
- // Compatibility fields
297
- content: snippet,
298
- thumbnail,
299
- ...(result.img_src ? { img_src: result.img_src } : {}),
300
- ...(result.iframe_src ? { iframe_src: result.iframe_src } : {}),
301
- };
302
- });
303
- return { results, suggestions: suggestions || [], infoboxes };
304
- }
305
-
306
- // Public instance scraping (HTML parsing)
307
- let results: SearxngSearchResult[] = [];
308
- const resultRegex = /<article class="result[^>]*>[\s\S]*?<\/article>/g;
309
- const titleUrlRegex = /<h3><a href="([^"]*)"[^>]*>(.*?)<\/a><\/h3>/;
310
- const snippetRegex = /<p class="content">\s*(.*?)\s*<\/p>/;
311
-
312
- // Unused in current logic but kept from original code for potential future use or completeness
313
- // const enginesRegex = /<span>(bing|duckduckgo|yahoo|google)<\/span>/g;
314
- // const linksRegex = /<a href="([^"]*)" class="(cache_link|proxyfied_link)"[^>]*>(cached|proxied)<\/a>/g;
315
-
316
- let match;
317
- while ((match = resultRegex.exec(resultHTML)) !== null) {
318
- const resultHtml = match[0];
319
- const titleUrlMatch = titleUrlRegex.exec(resultHtml);
320
- const snippetMatch = snippetRegex.exec(resultHtml);
321
-
322
- if (titleUrlMatch && titleUrlMatch[1] && titleUrlMatch[2]) {
323
- // const urlFound = convertURLSafeHTMLToHTML(titleUrlMatch[1]); // Not used in original, seemingly
324
- let title = titleUrlMatch[2].replace(/<\/?[^>]+(>|$)/g, "");
325
- let snippet = snippetMatch
326
- ? snippetMatch[1].replace(/<\/?[^>]+(>|$)/g, "")
327
- : "";
328
-
329
- title = convertURLSafeHTMLToHTML(title);
330
- snippet = convertURLSafeHTMLToHTML(snippet);
331
- const urlClean = convertURLSafeHTMLToHTML(titleUrlMatch[1]);
332
-
333
- // Validate URL has path component, not just domain
334
- try {
335
- const parsedUrl = new URL(urlClean);
336
- if (parsedUrl.pathname === "/" || parsedUrl.pathname === "") {
337
- console.warn(`[searchWeb] Public scrape URL is domain-only: ${urlClean}`);
338
- }
339
- } catch (e) {
340
- console.error(`[searchWeb] Invalid URL in public scrape: ${urlClean}`);
341
- }
342
-
343
- results.push({
344
- title,
345
- url: urlClean,
346
- snippet,
347
- content: snippet, // Compatibility
348
- });
349
- }
350
- }
351
-
352
- if (results.length === 0 && maxRetries > 0) {
353
- return (await searchWeb(query, {
354
- ...options,
355
- maxRetries: maxRetries - 1,
356
- useProxy: true,
357
- })) as SearxngSearchResult[];
358
- }
359
-
360
- results = results.map((result) => {
361
- const match = result.url.match(
362
- /^(?:https?:\/\/)?(?:www\.)?([^/:?\s]+)(?:[/:?]|$)/i,
363
- );
364
- const domainStr = match ? match[0] : "";
365
-
366
- const favicon = "https://www.google.com/s2/favicons?domain=" + domainStr;
367
-
368
- const domain = result.url
369
- ?.replace(/(http:\/\/|https:\/\/|www.)/gi, "")
370
- .split("/")[0];
371
-
372
- return {
373
- ...result,
374
- domain,
375
- favicon,
376
- thumbnail: favicon, // Compatibility
377
- };
378
- });
379
-
380
- return results;
381
- }
382
-
383
- // Wrapper to match existing `searchSearxng` signature if needed elsewhere,
384
- // OR the user might want this to be the primary `searchWeb` and we just export `searchSearxng` that calls it.
385
- // The user's request showed `searchWeb` being imported.
386
- // But the application likely calls `searchSearxng`. Let's reimplement `searchSearxng` to use `searchWeb`.
387
-
388
- interface SearxngSearchOptions {
389
- categories?: string[];
390
- engines?: string[];
391
- language?: string;
392
- pageno?: number;
393
- }
394
-
395
- export const searchSearxng = async (
396
- query: string,
397
- opts?: SearxngSearchOptions,
398
- ): Promise<{ results: SearxngSearchResult[]; suggestions: string[] }> => {
399
- // Adapter to call the new searchWeb
400
- const category = opts?.categories?.[0] || "general"; // simplistic mapping
401
- const page = opts?.pageno || 1;
402
- const lang = opts?.language || "en-US";
403
-
404
- const result = await searchWeb(query, {
405
- category,
406
- page,
407
- lang,
408
- // privateSearxng: true // or false? The user code said "use custom or false to use the public instances"
409
- // Let's rely on the default behavior or what `searchWeb` does.
410
- // However, `searchWeb` logic branches on `privateSearxng` significantly.
411
- // If we want JSON, we probably want `privateSearxng` set to a domain if we have one, or handle the array return.
412
-
413
- // IMPORTANT: The user code's `GET` handler passes `privateSearxng: publicInstances ? false : searxngDomain`.
414
- // where `searxngDomain` was imported from `customize-site`.
415
- // Since we don't have that file, we used empty string defaults.
416
- // If `searxngDomain` is falsy, `privateSearxng` becomes falsy (or we should be careful).
417
- });
418
-
419
- if (Array.isArray(result)) {
420
- return { results: result, suggestions: [] };
421
- } else {
422
- return { results: result.results, suggestions: result.suggestions || [] };
423
- }
424
- };
425
-
426
- // Helper function to decode HTML entities
427
- function convertURLSafeHTMLToHTML(html: string): string {
428
- if (!html) return "";
429
- return html
430
- .replace(/&amp;/g, "&")
431
- .replace(/&lt;/g, "<")
432
- .replace(/&gt;/g, ">")
433
- .replace(/&quot;/g, '"')
434
- .replace(/&#39;/g, "'");
435
- }
436
-
437
- interface SearchOptions {
438
- category?: string | number;
439
- recency?: string;
440
- privateSearxng?: string | boolean | null;
441
- maxRetries?: number;
442
- page?: number;
443
- safesearch?: boolean;
444
- lang?: string;
445
- proxy?: string | null;
446
- useProxy?: boolean;
447
- }
448
-
449
- export interface SearxngSearchResult {
450
- title: string;
451
- url: string;
452
- snippet?: string;
453
- domain?: string;
454
- favicon?: string;
455
- score?: number;
456
- source?: string;
457
- date?: string;
458
- img_src?: string; // Added for compatibility with existing interfaces
459
- thumbnail_src?: string; // Added for compatibility
460
- thumbnail?: string; // Added for compatibility
461
- content?: string; // Added for compatibility
462
- author?: string; // Added for compatibility
463
- iframe_src?: string; // Added for compatibility
464
- }
465
-
466
- export interface SearchResponse {
467
- results: SearxngSearchResult[];
468
- suggestions: string[];
469
- infoboxes?: any[];
470
- }