extract-webpage 1.2.49 → 1.2.51
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extract-webpage.cjs.js +1 -1
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js +3 -3
- package/dist/extract-webpage.es.js.map +1 -1
- package/dist/index.d.ts +0 -2
- package/package.json +4 -11
- package/src/html-to-content/html-utils.ts +8 -44
- package/src/index.ts +0 -3
- package/dist/search/index.d.ts +0 -14
- package/dist/search/meta-search-agent-reexport.d.ts +0 -8
- package/dist/search/public-searxng.d.ts +0 -47
- package/dist/search/search-web.d.ts +0 -33
- package/dist/search/tavily.d.ts +0 -20
- package/dist/search/url-to-html.d.ts +0 -62
- package/src/search/__tests__/public-searxng.test.ts +0 -529
- package/src/search/index.ts +0 -45
- package/src/search/meta-search-agent-reexport.ts +0 -38
- package/src/search/public-searxng.ts +0 -470
- package/src/search/search-web.ts +0 -668
- package/src/search/tavily.ts +0 -106
- package/src/search/url-to-html.ts +0 -278
package/src/search/search-web.ts
DELETED
|
@@ -1,668 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @fileoverview Core web searching utilities via SearXNG.
|
|
3
|
-
* Handles metasearch execution, result parsing, and domain-based filtering.
|
|
4
|
-
*/
|
|
5
|
-
import { getDomainWithoutSuffix } from "tldts";
|
|
6
|
-
import { scrapeURL } from "./url-to-html";
|
|
7
|
-
import { parseDate } from "chrono-node";
|
|
8
|
-
import grab from "../utils/grab";
|
|
9
|
-
|
|
10
|
-
/**
|
|
11
|
-
* Search Web via SearXNG metasearch of all major search engines.
|
|
12
|
-
* Options are 10 search categories, recency, and how many
|
|
13
|
-
* times to retry other domains if first time fails.
|
|
14
|
-
* SearXNG is a free internet metasearch engine which aggregates results from
|
|
15
|
-
* more than [180+ search sources](https://docs.searxng.org/user/configured_engines.html).
|
|
16
|
-
*
|
|
17
|
-
* [Searxng Overview](https://medium.com/@elmo92/search-in-peace-with-searxng-an-alternative-search-engine-that-keeps-your-searches-private-accd8cddd6fc)
|
|
18
|
-
* [Searxng Installation Guide](https://github.com/searxng/searxng-docker/tree/master)
|
|
19
|
-
*
|
|
20
|
-
* 
|
|
21
|
-
* @param {string} query - The search query string.
|
|
22
|
-
* @param {Object} [options]
|
|
23
|
-
* @param {string} options.category default=general - ["general", "news", "videos", "images",
|
|
24
|
-
* "science","it", "files", "social+media", "map", "music"]
|
|
25
|
-
* @param {string} options.recency default=all - ["all", "day", "week", "month", "year"]
|
|
26
|
-
* @param {string|boolean} options.privateSearxng default=null - Use your custom domain SearXNG
|
|
27
|
-
* @param {number} options.maxRetries default=3 - Maximum number of retry attempts if the initial search fails.
|
|
28
|
-
* @param {number} options.page default=1 - The page number to retrieve.
|
|
29
|
-
* @param {boolean} options.safesearch default=false - Whether to block adult content.
|
|
30
|
-
* @param {string} options.lang default="en-US" - The language to use for the search.
|
|
31
|
-
* @param {string} options.proxy default=false - Use corsproxy.io to access in frontend JS
|
|
32
|
-
* @returns {Promise<Array<{title: string, url: string, snippet: string, domain: string, favicon: string, path: string, engines: string[]}>>} An array of search result objects.
|
|
33
|
-
* @example const advancedResults = await searchWeb('Node.js', {
|
|
34
|
-
* category: 2,
|
|
35
|
-
* recency: 1,
|
|
36
|
-
* maxRetries: 5
|
|
37
|
-
* });
|
|
38
|
-
* @category Search
|
|
39
|
-
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
40
|
-
* [Heiser, M., Tauber, A., Flament, A., et al. (2014-)](https://github.com/searxng/searxng/graphs/contributors)
|
|
41
|
-
*/
|
|
42
|
-
export async function searchWeb(query, options = {} as any) {
|
|
43
|
-
const {
|
|
44
|
-
category = "general",
|
|
45
|
-
recency,
|
|
46
|
-
privateSearxng = null,
|
|
47
|
-
maxRetries = 3,
|
|
48
|
-
page = 1,
|
|
49
|
-
safesearch = false,
|
|
50
|
-
lang = "en-US",
|
|
51
|
-
proxy = null,
|
|
52
|
-
} = options;
|
|
53
|
-
|
|
54
|
-
const CATEGORY_LIST = [
|
|
55
|
-
"general",
|
|
56
|
-
"news",
|
|
57
|
-
"videos",
|
|
58
|
-
"images",
|
|
59
|
-
"science",
|
|
60
|
-
"it",
|
|
61
|
-
"files",
|
|
62
|
-
"social+media",
|
|
63
|
-
];
|
|
64
|
-
const RECENCY_ALLOWED_LIST = ["day", "week", "month", "year"];
|
|
65
|
-
|
|
66
|
-
const SEARX_DOMAINS = [
|
|
67
|
-
"baresearch.org",
|
|
68
|
-
"copp.gg",
|
|
69
|
-
"darmarit.org",
|
|
70
|
-
"etsi.me",
|
|
71
|
-
"fairsuch.net",
|
|
72
|
-
"nogoo.me",
|
|
73
|
-
"northboot.xyz",
|
|
74
|
-
"nyc1.sx.ggtyler.dev",
|
|
75
|
-
"ooglester.com",
|
|
76
|
-
"opnxng.com",
|
|
77
|
-
"paulgo.io",
|
|
78
|
-
"priv.au",
|
|
79
|
-
"s.trung.fun",
|
|
80
|
-
"search.blitzw.in",
|
|
81
|
-
"search.charliewhiskey.net",
|
|
82
|
-
"search.citw.lgbt",
|
|
83
|
-
"search.darkness.services",
|
|
84
|
-
"search.datura.network",
|
|
85
|
-
"search.dotone.nl",
|
|
86
|
-
"search.gcomm.ch",
|
|
87
|
-
"search.hbubli.cc",
|
|
88
|
-
"search.im-in.space",
|
|
89
|
-
"search.incogniweb.net",
|
|
90
|
-
"search.inetol.net",
|
|
91
|
-
"search.leptons.xyz",
|
|
92
|
-
"search.nadeko.net",
|
|
93
|
-
"search.ngn.tf",
|
|
94
|
-
"search.ononoki.org",
|
|
95
|
-
"search.privacyredirect.com",
|
|
96
|
-
"search.sapti.me",
|
|
97
|
-
"search.rowie.at",
|
|
98
|
-
"search.projectsegfau.lt",
|
|
99
|
-
"search.tommy-tran.com",
|
|
100
|
-
"searx.aleteoryx.me",
|
|
101
|
-
"searx.ankha.ac",
|
|
102
|
-
"searx.be",
|
|
103
|
-
"searx.colbster937.dev",
|
|
104
|
-
"searx.daetalytica.io",
|
|
105
|
-
"searx.dresden.network",
|
|
106
|
-
"searx.foss.family",
|
|
107
|
-
"searx.hu",
|
|
108
|
-
"searx.juancord.xyz",
|
|
109
|
-
"searx.lunar.icu",
|
|
110
|
-
"searx.mxchange.org",
|
|
111
|
-
"searx.namejeff.xyz",
|
|
112
|
-
"searx.oakleycord.dev",
|
|
113
|
-
"searx.ro",
|
|
114
|
-
"searx.sev.monster",
|
|
115
|
-
"searx.thefloatinglab.world",
|
|
116
|
-
"searx.tiekoetter.com",
|
|
117
|
-
"searx.tuxcloud.net",
|
|
118
|
-
"searx.work",
|
|
119
|
-
"searx.zhenyapav.com",
|
|
120
|
-
"searxng.hweeren.com",
|
|
121
|
-
"searxng.online",
|
|
122
|
-
"searxng.shreven.org",
|
|
123
|
-
"searxng.site",
|
|
124
|
-
"skyrimhater.com",
|
|
125
|
-
"sx.ca.zorby.top",
|
|
126
|
-
"sx.catgirl.cloud",
|
|
127
|
-
"sx.thatxtreme.dev",
|
|
128
|
-
"sx.zorby.top",
|
|
129
|
-
"xo.wtf",
|
|
130
|
-
];
|
|
131
|
-
|
|
132
|
-
//select a random domain if none is provided
|
|
133
|
-
const searchDomain =
|
|
134
|
-
privateSearxng ||
|
|
135
|
-
"https://" +
|
|
136
|
-
SEARX_DOMAINS[Math.floor(Math.random() * SEARX_DOMAINS.length)];
|
|
137
|
-
|
|
138
|
-
const categoryName =
|
|
139
|
-
typeof category === "number" ? CATEGORY_LIST[category] : category; // Using the first category as default
|
|
140
|
-
|
|
141
|
-
var url =
|
|
142
|
-
`${searchDomain}/search?q=${encodeURIComponent(query)}` +
|
|
143
|
-
`&category_${categoryName}=1&language=${lang}` +
|
|
144
|
-
`${
|
|
145
|
-
recency in RECENCY_ALLOWED_LIST ? "&time_range=" + recency : ""
|
|
146
|
-
}&safesearch=${safesearch ? "1" : "0"}&pageno=${page}`;
|
|
147
|
-
|
|
148
|
-
if (privateSearxng) url += "&format=json";
|
|
149
|
-
|
|
150
|
-
let resultHTML: string = "";
|
|
151
|
-
try {
|
|
152
|
-
resultHTML = await grab(url, {
|
|
153
|
-
headers: {
|
|
154
|
-
"accept-language": lang + ",en;q=0.9",
|
|
155
|
-
},
|
|
156
|
-
responseType: "text",
|
|
157
|
-
}) as string;
|
|
158
|
-
} catch (error: any) {
|
|
159
|
-
const errorMsg = error instanceof Error ? error.message : String(error);
|
|
160
|
-
console.warn(`[searchWeb] Failed to fetch from SearXNG domain "${searchDomain}": ${errorMsg}`);
|
|
161
|
-
if (maxRetries > 0) {
|
|
162
|
-
console.log(`[searchWeb] Retrying with another instance... (${maxRetries} retries left)`);
|
|
163
|
-
return await searchWeb(query, {
|
|
164
|
-
...options,
|
|
165
|
-
maxRetries: maxRetries - 1,
|
|
166
|
-
});
|
|
167
|
-
}
|
|
168
|
-
console.error(`[searchWeb] All retries exhausted. Returning empty results.`);
|
|
169
|
-
return [];
|
|
170
|
-
}
|
|
171
|
-
|
|
172
|
-
if (privateSearxng) {
|
|
173
|
-
if (!resultHTML.startsWith("{"))
|
|
174
|
-
return { error: "Private SearXNG instance did not return valid JSON" };
|
|
175
|
-
//todo use public
|
|
176
|
-
|
|
177
|
-
var { results, suggestions, infoboxes } = JSON.parse(resultHTML);
|
|
178
|
-
|
|
179
|
-
results = results.map((result) => {
|
|
180
|
-
var title = result.title.replace(/<\/?[^>]+(>|$)/g, "");
|
|
181
|
-
|
|
182
|
-
// Clean and normalize the title
|
|
183
|
-
const TITLE_SPLITTERS_RE = /( [|\-\/:\u00bb] )|( - )|(\|)/;
|
|
184
|
-
|
|
185
|
-
// Handle split titles
|
|
186
|
-
if (TITLE_SPLITTERS_RE.test(title)) {
|
|
187
|
-
const splitTitle = title.split(TITLE_SPLITTERS_RE);
|
|
188
|
-
|
|
189
|
-
// Handle breadcrumbed titles
|
|
190
|
-
if (splitTitle.length >= 2) {
|
|
191
|
-
const longestPart = splitTitle.reduce(
|
|
192
|
-
(acc, part) => (part?.length > acc?.length ? part : acc),
|
|
193
|
-
"",
|
|
194
|
-
);
|
|
195
|
-
if (longestPart.length > 10) {
|
|
196
|
-
title = longestPart;
|
|
197
|
-
}
|
|
198
|
-
}
|
|
199
|
-
}
|
|
200
|
-
|
|
201
|
-
title = convertURLSafeHTMLToHTML(title);
|
|
202
|
-
var url = result.url.replace(/&/g, "&");
|
|
203
|
-
var snippet = result.content?.replace(/<\/?[^>]+(>|$)/g, "");
|
|
204
|
-
var score = Math.round(result.score * 100) / 100;
|
|
205
|
-
// Parse metadata into date and source
|
|
206
|
-
|
|
207
|
-
var domain = result.url
|
|
208
|
-
?.replace(/(http:\/\/|https:\/\/|www.)/gi, "")
|
|
209
|
-
.split("/")[0];
|
|
210
|
-
|
|
211
|
-
let date = null,
|
|
212
|
-
source = null;
|
|
213
|
-
if (typeof result.metadata === "string") {
|
|
214
|
-
const [datePart, sourcePart] = result.metadata
|
|
215
|
-
.split("|")
|
|
216
|
-
.map((s) => s.trim());
|
|
217
|
-
date =
|
|
218
|
-
parseDate(result.metadata)?.toISOString().split("T")[0] || undefined;
|
|
219
|
-
source = sourcePart || null;
|
|
220
|
-
}
|
|
221
|
-
|
|
222
|
-
if (!source) {
|
|
223
|
-
source = getDomainWithoutSuffix(domain).replace(/\b\w/g, (l) =>
|
|
224
|
-
l.toUpperCase(),
|
|
225
|
-
);
|
|
226
|
-
// for small source names like CNN
|
|
227
|
-
if (source.length < 5) source = source.toUpperCase();
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
var favicon =
|
|
231
|
-
"https://www.google.com/s2/favicons?domain=" +
|
|
232
|
-
result.url.match(
|
|
233
|
-
/^(?:https?:\/\/)?(?:www\.)?([^/:?\s]+)(?:[/:?]|$)/i,
|
|
234
|
-
)?.[0] +
|
|
235
|
-
"&sz=16";
|
|
236
|
-
|
|
237
|
-
return {
|
|
238
|
-
title,
|
|
239
|
-
url,
|
|
240
|
-
snippet,
|
|
241
|
-
score,
|
|
242
|
-
...(date ? { date } : {}),
|
|
243
|
-
...(source ? { source } : {}),
|
|
244
|
-
domain,
|
|
245
|
-
favicon,
|
|
246
|
-
};
|
|
247
|
-
});
|
|
248
|
-
return { results, suggestions, infoboxes };
|
|
249
|
-
}
|
|
250
|
-
|
|
251
|
-
results = [];
|
|
252
|
-
const resultRegex = /<article class="result[^>]*>[\s\S]*?<\/article>/g;
|
|
253
|
-
const titleUrlRegex = /<h3><a href="([^"]*)"[^>]*>(.*?)<\/a><\/h3>/;
|
|
254
|
-
const snippetRegex = /<p class="content">\s*(.*?)\s*<\/p>/;
|
|
255
|
-
const enginesRegex = /<span>(bing|duckduckgo|yahoo|google)<\/span>/g;
|
|
256
|
-
const linksRegex =
|
|
257
|
-
/<a href="([^"]*)" class="(cache_link|proxyfied_link)"[^>]*>(cached|proxied)<\/a>/g;
|
|
258
|
-
|
|
259
|
-
let match;
|
|
260
|
-
while ((match = resultRegex.exec(resultHTML)) !== null) {
|
|
261
|
-
const resultHtml = match[0];
|
|
262
|
-
const titleUrlMatch = titleUrlRegex.exec(resultHtml);
|
|
263
|
-
const snippetMatch = snippetRegex.exec(resultHtml);
|
|
264
|
-
|
|
265
|
-
if (titleUrlMatch && titleUrlMatch[1] && titleUrlMatch[2]) {
|
|
266
|
-
const url = convertURLSafeHTMLToHTML(titleUrlMatch[1]);
|
|
267
|
-
let title = titleUrlMatch[2].replace(/<\/?[^>]+(>|$)/g, "");
|
|
268
|
-
let snippet = snippetMatch
|
|
269
|
-
? snippetMatch[1].replace(/<\/?[^>]+(>|$)/g, "")
|
|
270
|
-
: "";
|
|
271
|
-
|
|
272
|
-
let engines = [];
|
|
273
|
-
let engineMatch;
|
|
274
|
-
while ((engineMatch = enginesRegex.exec(resultHtml)) !== null) {
|
|
275
|
-
engines.push(engineMatch[1]);
|
|
276
|
-
}
|
|
277
|
-
|
|
278
|
-
let cached = null;
|
|
279
|
-
let linkMatch;
|
|
280
|
-
// while ((linkMatch = linksRegex.exec(resultHtml)) !== null) {
|
|
281
|
-
// cached = linkMatch[1];
|
|
282
|
-
// }
|
|
283
|
-
|
|
284
|
-
title = convertURLSafeHTMLToHTML(title);
|
|
285
|
-
snippet = convertURLSafeHTMLToHTML(snippet);
|
|
286
|
-
// if (!url.includes(".de/"))
|
|
287
|
-
results.push({ title, url, snippet });
|
|
288
|
-
}
|
|
289
|
-
}
|
|
290
|
-
|
|
291
|
-
if (results.length === 0 && maxRetries > 0) {
|
|
292
|
-
results = await searchWeb(query, {
|
|
293
|
-
...options,
|
|
294
|
-
maxRetries: maxRetries - 1,
|
|
295
|
-
useProxy: true,
|
|
296
|
-
});
|
|
297
|
-
}
|
|
298
|
-
|
|
299
|
-
//filter out url that end with .de
|
|
300
|
-
// results = results.filter((result) => !result.url.includes(".de/"));
|
|
301
|
-
|
|
302
|
-
results = results.map((result) => {
|
|
303
|
-
var favicon =
|
|
304
|
-
"https://www.google.com/s2/favicons?domain=" +
|
|
305
|
-
result.url.match(
|
|
306
|
-
/^(?:https?:\/\/)?(?:www\.)?([^/:?\s]+)(?:[/:?]|$)/i,
|
|
307
|
-
)?.[0];
|
|
308
|
-
|
|
309
|
-
var domain = result.url
|
|
310
|
-
?.replace(/(http:\/\/|https:\/\/|www.)/gi, "")
|
|
311
|
-
.split("/")[0];
|
|
312
|
-
|
|
313
|
-
let source = getDomainWithoutSuffix(domain).replace(/\b\w/g, (l) =>
|
|
314
|
-
l.toUpperCase(),
|
|
315
|
-
);
|
|
316
|
-
// for small source names like CNN
|
|
317
|
-
if (source.length < 5) source = source.toUpperCase();
|
|
318
|
-
|
|
319
|
-
return {
|
|
320
|
-
...result,
|
|
321
|
-
domain,
|
|
322
|
-
favicon,
|
|
323
|
-
source,
|
|
324
|
-
};
|
|
325
|
-
});
|
|
326
|
-
return results;
|
|
327
|
-
// } catch (error) {
|
|
328
|
-
// console.error(`Error fetching search results: ${error.message}`);
|
|
329
|
-
// return [];
|
|
330
|
-
// }
|
|
331
|
-
}
|
|
332
|
-
|
|
333
|
-
/**
|
|
334
|
-
* Converts URL-safe escaped HTML codes like &"'`’ & to standard HTML or in reverse.
|
|
335
|
-
* @param {string} str - The string to process.
|
|
336
|
-
* @param {boolean} toStandardHTML default=true - If true, converts url-safe codes
|
|
337
|
-
* to standard HTML. If false, converts standard HTML to url-safe codes.
|
|
338
|
-
* @return {string} The processed string.
|
|
339
|
-
* @category HTML Utilities
|
|
340
|
-
* @example
|
|
341
|
-
* var normalHTML = convertURLSafeHTMLToHTML('<p>This & that © 2023 '+
|
|
342
|
-
* '"Quotes"'Apostrophes' €100 ☺</p>', true)
|
|
343
|
-
* console.log(normalHTML) // "<p>This & that \u00a9 2023 "Quotes" 'Apostrophes' \u20ac100 \u263a</p>"
|
|
344
|
-
*/
|
|
345
|
-
function convertURLSafeHTMLToHTML(str, toStandardHTML = true) {
|
|
346
|
-
const entityMap = {
|
|
347
|
-
"&": "&",
|
|
348
|
-
"<": "<",
|
|
349
|
-
">": ">",
|
|
350
|
-
'"': """,
|
|
351
|
-
" ": " ",
|
|
352
|
-
"'": "'",
|
|
353
|
-
"`": "`",
|
|
354
|
-
"\u00a2": "¢",
|
|
355
|
-
"\u00a3": "£",
|
|
356
|
-
"\u00a5": "¥",
|
|
357
|
-
"\u20ac": "€",
|
|
358
|
-
"\u00a9": "©",
|
|
359
|
-
"\u00ae": "®",
|
|
360
|
-
"\u2122": "™",
|
|
361
|
-
};
|
|
362
|
-
|
|
363
|
-
// Add numeric character references for Latin-1 Supplement characters
|
|
364
|
-
for (let i = 160; i <= 255; i++) {
|
|
365
|
-
entityMap[String.fromCharCode(i)] = `&#${i};`;
|
|
366
|
-
}
|
|
367
|
-
|
|
368
|
-
if (toStandardHTML) {
|
|
369
|
-
// Create a reverse mapping for unescaping
|
|
370
|
-
const reverseEntityMap = Object.fromEntries(
|
|
371
|
-
Object.entries(entityMap).map(([k, v]) => [v, k]),
|
|
372
|
-
);
|
|
373
|
-
|
|
374
|
-
// Add alternative representations
|
|
375
|
-
reverseEntityMap["'"] = "'";
|
|
376
|
-
reverseEntityMap["«"] = "\u00ab";
|
|
377
|
-
reverseEntityMap["»"] = "\u00bb";
|
|
378
|
-
|
|
379
|
-
// Regex to match all types of HTML entities
|
|
380
|
-
const entityRegex = new RegExp(
|
|
381
|
-
Object.keys(reverseEntityMap).join("|") + "|&#[0-9]+;|&#x[0-9a-fA-F]+;",
|
|
382
|
-
"g",
|
|
383
|
-
);
|
|
384
|
-
|
|
385
|
-
str = str.replace(entityRegex, (entity) => {
|
|
386
|
-
if (entity.startsWith("&#x")) {
|
|
387
|
-
// Convert hexadecimal numeric character reference
|
|
388
|
-
return String.fromCharCode(parseInt(entity.slice(3, -1), 16));
|
|
389
|
-
} else if (entity.startsWith("&#")) {
|
|
390
|
-
// Convert decimal numeric character reference
|
|
391
|
-
return String.fromCharCode(parseInt(entity.slice(2, -1), 10));
|
|
392
|
-
}
|
|
393
|
-
// Convert named entity
|
|
394
|
-
return reverseEntityMap[entity] || entity;
|
|
395
|
-
});
|
|
396
|
-
|
|
397
|
-
str = str.replace(/[\u0300-\u036f]/g, ""); //special chars
|
|
398
|
-
|
|
399
|
-
return str;
|
|
400
|
-
} else {
|
|
401
|
-
// Regex to match all characters that need to be escaped
|
|
402
|
-
const charRegex = new RegExp(`[${Object.keys(entityMap).join("")}]`, "g");
|
|
403
|
-
return str.replace(charRegex, (char) => entityMap[char]);
|
|
404
|
-
}
|
|
405
|
-
}
|
|
406
|
-
|
|
407
|
-
var sources = [
|
|
408
|
-
["google", "go", "The world's most popular search engine."],
|
|
409
|
-
["bing", "bi", "Microsoft's web search engine."],
|
|
410
|
-
[
|
|
411
|
-
"brave",
|
|
412
|
-
"br",
|
|
413
|
-
"Privacy-focused web browser with built-in search functionality.",
|
|
414
|
-
],
|
|
415
|
-
[
|
|
416
|
-
"duckduckgo",
|
|
417
|
-
"ddg",
|
|
418
|
-
"Privacy-oriented search engine that doesn't track users.",
|
|
419
|
-
],
|
|
420
|
-
["mojeek", "mjk", "Independent search engine that builds its own index."],
|
|
421
|
-
[
|
|
422
|
-
"presearch",
|
|
423
|
-
"ps",
|
|
424
|
-
"Decentralized search engine using blockchain technology.",
|
|
425
|
-
],
|
|
426
|
-
["presearch videos", "psvid", "Video search feature of Presearch."],
|
|
427
|
-
["qwant", "qw", "European privacy-focused search engine."],
|
|
428
|
-
[
|
|
429
|
-
"startpage",
|
|
430
|
-
"sp",
|
|
431
|
-
"Search engine that provides Google results with enhanced privacy.",
|
|
432
|
-
],
|
|
433
|
-
["wiby", "wib", "Search engine for older-style, minimal HTML websites."],
|
|
434
|
-
[
|
|
435
|
-
"yahoo",
|
|
436
|
-
"yh",
|
|
437
|
-
"Web services provider known for its search engine and email service.",
|
|
438
|
-
],
|
|
439
|
-
[
|
|
440
|
-
"naver (KO)",
|
|
441
|
-
"nvr",
|
|
442
|
-
"Major South Korean search engine and online platform.",
|
|
443
|
-
],
|
|
444
|
-
["wikibooks", "wb", "Wikimedia project for free textbooks and manuals."],
|
|
445
|
-
["wikiquote", "wq", "Wikimedia project collecting quotations."],
|
|
446
|
-
["wikisource", "ws", "Wikimedia library of source texts."],
|
|
447
|
-
["wikispecies", "wsp", "Wikimedia project cataloging species."],
|
|
448
|
-
[
|
|
449
|
-
"wikiversity",
|
|
450
|
-
"wv",
|
|
451
|
-
"Wikimedia project dedicated to learning resources and activities.",
|
|
452
|
-
],
|
|
453
|
-
["wikivoyage", "wy", "Wikimedia project for travel guides."],
|
|
454
|
-
["ask", "ask", "Question-answering search engine."],
|
|
455
|
-
["cloudflareai", "cfai", "AI services provided by Cloudflare."],
|
|
456
|
-
["crowdview", "cv", "Likely a crowdsourced information or review platform."],
|
|
457
|
-
["curlie", "cl", "Web directory maintained by volunteer editors."],
|
|
458
|
-
["dictzone", "dc", "Online dictionary and translation service."],
|
|
459
|
-
["libretranslate", "lt", "Open-source machine translation tool."],
|
|
460
|
-
[
|
|
461
|
-
"mymemory translated",
|
|
462
|
-
"tl",
|
|
463
|
-
"Translation memory service combining human and machine translations.",
|
|
464
|
-
],
|
|
465
|
-
["currency", "cc", "Currency conversion tool."],
|
|
466
|
-
["ddg definitions", "ddd", "Definition search using DuckDuckGo."],
|
|
467
|
-
["encyclosearch", "es", "Specialized search for encyclopedic content."],
|
|
468
|
-
["searchmysite", "sms", "Custom site search service."],
|
|
469
|
-
["stract", "str", "Likely a specialized or alternative search engine."],
|
|
470
|
-
["tineye", "tin", "Reverse image search engine."],
|
|
471
|
-
["wikidata", "wd", "Wikimedia's collaborative knowledge base."],
|
|
472
|
-
["wikipedia", "wp", "Free online encyclopedia."],
|
|
473
|
-
["wolframalpha", "wa", "Computational knowledge engine."],
|
|
474
|
-
["tagesschau (DE)", "ts", "Search for German news from Tagesschau."],
|
|
475
|
-
["wikimini (FR)", "wkmn", "French-language wiki encyclopedia for children."],
|
|
476
|
-
["bing images", "bii", "Image search by Microsoft's Bing."],
|
|
477
|
-
["brave.images", "brimg", "Image search feature of Brave browser."],
|
|
478
|
-
["duckduckgo images", "ddi", "Image search by DuckDuckGo."],
|
|
479
|
-
["google images", "goi", "Google's image search service."],
|
|
480
|
-
["mojeek images", "mjkimg", "Image search by Mojeek."],
|
|
481
|
-
["presearch images", "psimg", "Image search feature of Presearch."],
|
|
482
|
-
["qwant images", "qwi", "Image search by Qwant."],
|
|
483
|
-
["1x", "1x", "Curated photography platform."],
|
|
484
|
-
["artic", "arc", "Art Institute of Chicago's collection search."],
|
|
485
|
-
["deviantart", "da", "Online art community and platform."],
|
|
486
|
-
["findthatmeme", "ftm", "Meme search engine."],
|
|
487
|
-
["flickr", "fl", "Image and video hosting platform."],
|
|
488
|
-
["frinkiac", "frk", "Search engine for The Simpsons screenshots and quotes."],
|
|
489
|
-
["imgur", "img", "Image hosting and sharing platform."],
|
|
490
|
-
["library of congress", "loc", "Search for Library of Congress resources."],
|
|
491
|
-
["material icons", "mi", "Google's Material Design icon search."],
|
|
492
|
-
["openverse", "opv", "Search engine for openly licensed media."],
|
|
493
|
-
["pinterest", "pin", "Image sharing and social media platform."],
|
|
494
|
-
["svgrepo", "svg", "SVG file and icon repository."],
|
|
495
|
-
["unsplash", "us", "Platform for freely usable images."],
|
|
496
|
-
["wallhaven", "wh", "Wallpaper search engine and community."],
|
|
497
|
-
["wikicommons.images", "wc", "Wikimedia Commons image search."],
|
|
498
|
-
["yacy images", "yai", "Image search using the YaCy network."],
|
|
499
|
-
["yep images", "yepi", "Image search feature of Yep."],
|
|
500
|
-
["seekr images (EN)", "seimg", "Image search by Seekr."],
|
|
501
|
-
["bing videos", "biv", "Video search by Microsoft's Bing."],
|
|
502
|
-
["brave.videos", "brvid", "Video search feature of Brave browser."],
|
|
503
|
-
["duckduckgo videos", "ddv", "Video search by DuckDuckGo."],
|
|
504
|
-
["google videos", "gov", "Google's video search service."],
|
|
505
|
-
["qwant videos", "qwv", "Video search by Qwant."],
|
|
506
|
-
["bilibili", "bil", "Chinese video sharing website."],
|
|
507
|
-
["dailymotion", "dm", "Video-sharing platform."],
|
|
508
|
-
["google play movies", "gpm", "Google's movie and TV show service."],
|
|
509
|
-
["invidious", "iv", "Alternative front-end for YouTube."],
|
|
510
|
-
["livespace", "ls", "Likely a live streaming platform."],
|
|
511
|
-
[
|
|
512
|
-
"media.ccc.de",
|
|
513
|
-
"c3tv",
|
|
514
|
-
"Video platform for Chaos Computer Club conferences.",
|
|
515
|
-
],
|
|
516
|
-
["odysee", "od", "Blockchain-based video platform."],
|
|
517
|
-
["peertube", "ptb", "Decentralized video hosting network."],
|
|
518
|
-
["piped", "ppd", "Alternative privacy-friendly YouTube frontend."],
|
|
519
|
-
["rumble", "ru", "Video sharing platform."],
|
|
520
|
-
["sepiasearch", "sep", "Search engine for PeerTube videos."],
|
|
521
|
-
["vimeo", "vm", "Video hosting and sharing platform."],
|
|
522
|
-
["wikicommons.videos", "wcv", "Wikimedia Commons video search."],
|
|
523
|
-
["youtube", "yt", "Popular video sharing platform."],
|
|
524
|
-
["mediathekviewweb (DE)", "mvw", "German public television archive search."],
|
|
525
|
-
["seekr videos (EN)", "sevid", "Video search by Seekr."],
|
|
526
|
-
["ina (FR)", "in", "French National Audiovisual Institute archive search."],
|
|
527
|
-
["duckduckgo news", "ddn", "News search by DuckDuckGo."],
|
|
528
|
-
["mojeek news", "mjknews", "News search by Mojeek."],
|
|
529
|
-
["presearch news", "psnews", "News search feature of Presearch."],
|
|
530
|
-
["wikinews", "wn", "Wikimedia's collaborative news source."],
|
|
531
|
-
["bing news", "bin", "News search by Microsoft's Bing."],
|
|
532
|
-
["brave.news", "brnews", "News search feature of Brave browser."],
|
|
533
|
-
["google news", "gon", "Google's news aggregation service."],
|
|
534
|
-
["qwant news", "qwn", "News search by Qwant."],
|
|
535
|
-
["yahoo news", "yhn", "Yahoo's news service."],
|
|
536
|
-
["yep news", "yepn", "News search feature of Yep."],
|
|
537
|
-
["tagesschau (DE)", "ts", "German news from Tagesschau."],
|
|
538
|
-
["seekr news (EN)", "senews", "News search by Seekr."],
|
|
539
|
-
["apple maps", "apm", "Apple's mapping service."],
|
|
540
|
-
["openstreetmap", "osm", "Collaborative, open-source map."],
|
|
541
|
-
["photon", "ph", "Search engine for OpenStreetMap."],
|
|
542
|
-
["genius", "gen", "Song lyrics and annotation platform."],
|
|
543
|
-
["radio browser", "rb", "Search engine for radio stations."],
|
|
544
|
-
["bandcamp", "bc", "Music platform for independent artists."],
|
|
545
|
-
["deezer", "dz", "Music streaming service."],
|
|
546
|
-
[
|
|
547
|
-
"invidious",
|
|
548
|
-
"iv",
|
|
549
|
-
"Alternative front-end for YouTube, including music videos.",
|
|
550
|
-
],
|
|
551
|
-
["mixcloud", "mc", "Audio streaming platform for DJs and radio shows."],
|
|
552
|
-
[
|
|
553
|
-
"piped.music",
|
|
554
|
-
"ppdm",
|
|
555
|
-
"Music-focused feature of Piped (YouTube alternative).",
|
|
556
|
-
],
|
|
557
|
-
["soundcloud", "sc", "Audio distribution and music sharing platform."],
|
|
558
|
-
["wikicommons.audio", "wca", "Wikimedia Commons audio search."],
|
|
559
|
-
["youtube", "yt", "Video sharing platform, often used for music."],
|
|
560
|
-
["alpine linux packages", "alp", "Package search for Alpine Linux."],
|
|
561
|
-
["crates.io", "crates", "Registry of Rust packages."],
|
|
562
|
-
["docker hub", "dh", "Repository for Docker container images."],
|
|
563
|
-
["hex", "hex", "Package manager for the Erlang ecosystem."],
|
|
564
|
-
["hoogle", "ho", "Haskell API search engine."],
|
|
565
|
-
["lib.rs", "lrs", "Alternative crates.io front-end and Rust package index."],
|
|
566
|
-
["metacpan", "cpan", "Search engine for Perl modules."],
|
|
567
|
-
["npm", "npm", "Package manager for JavaScript."],
|
|
568
|
-
["packagist", "pack", "Package repository for PHP's Composer."],
|
|
569
|
-
["pkg.go.dev", "pgo", "Go package documentation."],
|
|
570
|
-
["pub.dev", "pd", "Package repository for Dart and Flutter."],
|
|
571
|
-
["pypi", "pypi", "Python Package Index."],
|
|
572
|
-
["rubygems", "rbg", "Package manager for Ruby."],
|
|
573
|
-
["voidlinux", "void", "Package search for Void Linux."],
|
|
574
|
-
["askubuntu", "ubuntu", "Q&A site for Ubuntu users."],
|
|
575
|
-
["caddy.community", "caddy", "Community forum for Caddy web server."],
|
|
576
|
-
["discuss.python", "dpy", "Official Python community discussion forum."],
|
|
577
|
-
[
|
|
578
|
-
"pi-hole.community",
|
|
579
|
-
"pi",
|
|
580
|
-
"Community forum for Pi-hole ad-blocking software.",
|
|
581
|
-
],
|
|
582
|
-
["stackoverflow", "st", "Q&A site for programmers."],
|
|
583
|
-
["superuser", "su", "Q&A site for computer enthusiasts and power users."],
|
|
584
|
-
["bitbucket", "bb", "Web-based version control repository hosting service."],
|
|
585
|
-
["codeberg", "cb", "Open-source code hosting platform."],
|
|
586
|
-
["gitea.com", "gitea", "Self-hosted Git service."],
|
|
587
|
-
["github", "gh", "Web-based hosting service for version control using Git."],
|
|
588
|
-
["gitlab", "gl", "Web-based DevOps lifecycle tool."],
|
|
589
|
-
["sourcehut", "srht", "Suite of open source software development tools."],
|
|
590
|
-
["arch linux wiki", "al", "Comprehensive documentation for Arch Linux."],
|
|
591
|
-
["free software directory", "fsd", "Catalog of free software."],
|
|
592
|
-
["gentoo", "ge", "Wiki for Gentoo Linux distribution."],
|
|
593
|
-
["anaconda", "conda", "Package manager for scientific computing."],
|
|
594
|
-
["cppreference", "cpp", "Reference for the C++ programming language."],
|
|
595
|
-
[
|
|
596
|
-
"habrahabr",
|
|
597
|
-
"habr",
|
|
598
|
-
"Russian collaborative blog about IT and computer science.",
|
|
599
|
-
],
|
|
600
|
-
[
|
|
601
|
-
"hackernews",
|
|
602
|
-
"hn",
|
|
603
|
-
"Social news website focusing on computer science and entrepreneurship.",
|
|
604
|
-
],
|
|
605
|
-
["lobste.rs", "lo", "Technology-focused link-aggregation site."],
|
|
606
|
-
["mankier", "man", "Web-based man page viewer."],
|
|
607
|
-
["mdn", "mdn", "Mozilla Developer Network documentation."],
|
|
608
|
-
["searchcode code", "scc", "Source code search engine."],
|
|
609
|
-
["arxiv", "arx", "Repository of electronic preprints for scientific papers."],
|
|
610
|
-
["crossref", "cr", "Official Digital Object Identifier Registration Agency."],
|
|
611
|
-
[
|
|
612
|
-
"google scholar",
|
|
613
|
-
"gos",
|
|
614
|
-
"Google\'s search engine for scholarly literature.",
|
|
615
|
-
],
|
|
616
|
-
[
|
|
617
|
-
"internetarchivescholar",
|
|
618
|
-
"ias",
|
|
619
|
-
"Search engine for scholarly works in Internet Archive.",
|
|
620
|
-
],
|
|
621
|
-
["pubmed", "pub", "Search engine for biomedical literature."],
|
|
622
|
-
[
|
|
623
|
-
"semantic scholar",
|
|
624
|
-
"se",
|
|
625
|
-
"AI-powered research tool for scientific literature.",
|
|
626
|
-
],
|
|
627
|
-
["wikispecies", "wsp", "Wikimedia project cataloging species."],
|
|
628
|
-
["openairedatasets", "oad", "Search for open access datasets."],
|
|
629
|
-
["openairepublications", "oap", "Search for open access publications."],
|
|
630
|
-
["pdbe", "pdb", "Protein Data Bank in Europe."],
|
|
631
|
-
["apk mirror", "apkm", "Repository of Android APK files."],
|
|
632
|
-
["apple app store", "aps", "Official app store for iOS devices."],
|
|
633
|
-
["fdroid", "fd", "App store for Free and Open Source Software on Android."],
|
|
634
|
-
["google play apps", "gpa", "Official app store for Android devices."],
|
|
635
|
-
["9gag", "9g", "Social media platform for sharing humor content."],
|
|
636
|
-
["lemmy posts", "lepo", "Post search for Lemmy."],
|
|
637
|
-
[
|
|
638
|
-
"mastodon hashtags",
|
|
639
|
-
"mah",
|
|
640
|
-
"Hashtag search for the Mastodon social network.",
|
|
641
|
-
],
|
|
642
|
-
["reddit", "re", "Social news and discussion website."],
|
|
643
|
-
[
|
|
644
|
-
"tootfinder",
|
|
645
|
-
"toot",
|
|
646
|
-
"Search engine for Mastodon and other federated networks.",
|
|
647
|
-
],
|
|
648
|
-
];
|
|
649
|
-
|
|
650
|
-
var sources_copyleft = [
|
|
651
|
-
["1337x", "1337x", "Torrent search engine."],
|
|
652
|
-
["annas archive", "aa", "Search engine for shadow libraries."],
|
|
653
|
-
["bt4g", "bt4g", "Torrent search engine."],
|
|
654
|
-
["btdigg", "bt", "BitTorrent DHT search engine."],
|
|
655
|
-
["kickass", "kc", "Torrent search engine."],
|
|
656
|
-
[
|
|
657
|
-
"library genesis",
|
|
658
|
-
"lg",
|
|
659
|
-
"File-sharing site for scholarly journal articles and books.",
|
|
660
|
-
],
|
|
661
|
-
["nyaa", "nt", "BitTorrent site focused on East Asian media."],
|
|
662
|
-
["openrepos", "or", "Repository for mobile apps."],
|
|
663
|
-
["piratebay", "tpb", "Well-known torrent site."],
|
|
664
|
-
["solidtorrents", "solid", "Decentralized torrent search engine."],
|
|
665
|
-
["tokyotoshokan", "tt", "BitTorrent site focused on Asian media."],
|
|
666
|
-
["wikicommons.files", "wcf", "File search on Wikimedia Commons."],
|
|
667
|
-
["z-library", "zlib", "Shadow library for books and articles."],
|
|
668
|
-
];
|