extract-webpage 1.2.35 → 1.2.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js.map +1 -1
- package/package.json +4 -4
- package/src/fs-mock.js +22 -22
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -1049
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3452 -3452
- package/src/search/index.ts +45 -45
- package/src/search/meta-search-agent-reexport.ts +38 -38
- package/src/search/url-to-html.ts +278 -278
- package/src/seektopic/seektopic-keyphrases.ts +279 -279
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -435
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -332
- package/src/url-to-content/url-to-content.ts +367 -367
- package/src/url-to-content/url-to-html.ts +436 -436
- package/src/utils/grab.ts +51 -51
|
@@ -1,278 +1,278 @@
|
|
|
1
|
-
import grab from "../utils/grab";
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* ### Tardigrade the Web Crawler
|
|
5
|
-
*
|
|
6
|
-
* 1. **Use Fetch API, check for bot detection.
|
|
7
|
-
* Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
|
|
8
|
-
* Scraping internet pages is a [free speech right
|
|
9
|
-
* ](https://blog.apify.com/is-web-scraping-legal/).
|
|
10
|
-
* 2. Features: timeout, redirects, default UA, referer as google, and bot
|
|
11
|
-
* detection checking. <br />
|
|
12
|
-
* 3. If fetch method does not get needed HTML, use Docker proxy as backup.
|
|
13
|
-
*
|
|
14
|
-
* 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
|
|
15
|
-
* container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
|
|
16
|
-
* secondary in-page API requests after the initial page request, including user login and cookie storage.
|
|
17
|
-
* 5. Bypass Cloudflare bot check: A webpage proxy that request
|
|
18
|
-
* through Chromium (puppeteer) - can be used to bypass Cloudflare
|
|
19
|
-
* anti bot using cookie id javascript method.
|
|
20
|
-
* 6. Send your request to the server with the port 3000 and add your URL to the "url"
|
|
21
|
-
* query string like this: `http://localhost:3000/?url=https://example.org`
|
|
22
|
-
*
|
|
23
|
-
* 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
|
|
24
|
-
* and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
|
|
25
|
-
* [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
|
|
26
|
-
* [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
|
|
27
|
-
* [Proxy-Cheap](https://app.proxy-cheap.com/order)
|
|
28
|
-
* [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
|
|
29
|
-
*
|
|
30
|
-
* @param {string} url - any domain's URL
|
|
31
|
-
* @param {Object} [options]
|
|
32
|
-
* @param {number} options.timeout default=5 - abort request if not retrived, in seconds
|
|
33
|
-
* @param {number} options.maxRedirects default=3 - max redirects to follow
|
|
34
|
-
* @param {number} options.checkBotDetection default=true - check for bot detection messages
|
|
35
|
-
* @param {number} options.changeReferer default=true - set referer as google
|
|
36
|
-
* @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
|
|
37
|
-
* @param {string} options.proxy default=false - use proxy url
|
|
38
|
-
* @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
|
|
39
|
-
* @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
|
|
40
|
-
* @category Extract
|
|
41
|
-
* @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
|
|
42
|
-
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
43
|
-
* @license MIT
|
|
44
|
-
*/
|
|
45
|
-
export async function scrapeURL(url, options = {} as any) {
|
|
46
|
-
// try {
|
|
47
|
-
let {
|
|
48
|
-
timeout = 5,
|
|
49
|
-
checkBotDetection = true,
|
|
50
|
-
maxRedirects = 3,
|
|
51
|
-
changeReferer = 0,
|
|
52
|
-
userAgentIndex = 0,
|
|
53
|
-
proxy = null,
|
|
54
|
-
useProxyAsBackup = true,
|
|
55
|
-
checkRobotsAllowed = false,
|
|
56
|
-
} = options;
|
|
57
|
-
|
|
58
|
-
if (checkRobotsAllowed) {
|
|
59
|
-
const rules = await fetchScrapingRules(url);
|
|
60
|
-
if (!isAllowedToScrape(rules, url)) {
|
|
61
|
-
return { error: "Robots.txt forbids to scrape there" };
|
|
62
|
-
}
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
if (proxy) url = proxy + url;
|
|
66
|
-
|
|
67
|
-
var userAgentStrings = [
|
|
68
|
-
"Chrome/41.0.2272.96 Mobile Safari/537.36 (compatible ; Googlebot/2.1 ; +http://www.google.com/bot.html)",
|
|
69
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_4) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/85.0.4183.83 Safari/537.36,gzip(gfe)",
|
|
70
|
-
];
|
|
71
|
-
|
|
72
|
-
var headers = {
|
|
73
|
-
...options,
|
|
74
|
-
"User-Agent": userAgentStrings[userAgentIndex],
|
|
75
|
-
signal: AbortSignal.timeout(timeout * 1000),
|
|
76
|
-
accept:
|
|
77
|
-
"text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7",
|
|
78
|
-
"accept-language": "en-US,en;q=0.9",
|
|
79
|
-
};
|
|
80
|
-
|
|
81
|
-
if (changeReferer) headers["Referer"] = "https://www.google.com/";
|
|
82
|
-
|
|
83
|
-
let response;
|
|
84
|
-
try {
|
|
85
|
-
response = await grab(url, {
|
|
86
|
-
...headers,
|
|
87
|
-
responseType: "raw",
|
|
88
|
-
});
|
|
89
|
-
} catch (e) {
|
|
90
|
-
return { error: "Error in fetch", msg: e.message };
|
|
91
|
-
}
|
|
92
|
-
|
|
93
|
-
if (response.redirected) {
|
|
94
|
-
if (maxRedirects <= 0) return { error: "Max redirects exceeded" };
|
|
95
|
-
maxRedirects--;
|
|
96
|
-
options = { ...options, maxRedirects };
|
|
97
|
-
|
|
98
|
-
return scrapeURL(response.url, options);
|
|
99
|
-
}
|
|
100
|
-
|
|
101
|
-
//return based on content type
|
|
102
|
-
const contentType = response.headers.get("Content-Type");
|
|
103
|
-
|
|
104
|
-
// if (contentType.includes("application/json")) {
|
|
105
|
-
// return await response.json();
|
|
106
|
-
// } else
|
|
107
|
-
// if (contentType.includes("text")) {
|
|
108
|
-
var html = await response.text();
|
|
109
|
-
|
|
110
|
-
if (checkBotDetection && checkHTMLForBotDetection(html)) {
|
|
111
|
-
html = await scrapeJINA(url, timeout);
|
|
112
|
-
|
|
113
|
-
//if all methods fail -- return jina
|
|
114
|
-
if (checkBotDetection && checkHTMLForBotDetection(html))
|
|
115
|
-
return { error: "Bot detected" }; //, html: response.html };
|
|
116
|
-
}
|
|
117
|
-
|
|
118
|
-
return html;
|
|
119
|
-
}
|
|
120
|
-
|
|
121
|
-
/**
|
|
122
|
-
* As backup, scrape with JINA to get html
|
|
123
|
-
* @param {string} url
|
|
124
|
-
* @returns {Promise<string>}
|
|
125
|
-
*/
|
|
126
|
-
export async function scrapeJINA(url, timeout = 15) {
|
|
127
|
-
let articleExtract = "";
|
|
128
|
-
try {
|
|
129
|
-
articleExtract = await grab("https://r.jina.ai/" + url, {
|
|
130
|
-
timeout: timeout * 1000,
|
|
131
|
-
responseType: "text",
|
|
132
|
-
});
|
|
133
|
-
} catch {
|
|
134
|
-
return "";
|
|
135
|
-
}
|
|
136
|
-
|
|
137
|
-
//convert Title: to <title>
|
|
138
|
-
var title = articleExtract.match(/Title: (.*)/)?.[1];
|
|
139
|
-
|
|
140
|
-
if (articleExtract.includes("===============\n"))
|
|
141
|
-
articleExtract = articleExtract
|
|
142
|
-
.split("===============\n")
|
|
143
|
-
.slice(1)
|
|
144
|
-
.join(" ");
|
|
145
|
-
|
|
146
|
-
var match = articleExtract.match(/Markdown Content:([\s\S]*)/);
|
|
147
|
-
articleExtract = match ? match[1] : articleExtract;
|
|
148
|
-
|
|
149
|
-
// articleExtract = convertMarkdownToHTML(articleExtract);
|
|
150
|
-
|
|
151
|
-
if (title) articleExtract = "<title>" + title + "</title>" + articleExtract;
|
|
152
|
-
|
|
153
|
-
return articleExtract;
|
|
154
|
-
}
|
|
155
|
-
|
|
156
|
-
/**
|
|
157
|
-
* Check html for bot block messages
|
|
158
|
-
* @param {string} html
|
|
159
|
-
* @returns {Boolean} true if bot detection message found
|
|
160
|
-
*/
|
|
161
|
-
function checkHTMLForBotDetection(html) {
|
|
162
|
-
var commonBlocks = [
|
|
163
|
-
"Error 403 - Unavailable",
|
|
164
|
-
"The security system for this website has been triggered",
|
|
165
|
-
"You do not have permission to view this page.",
|
|
166
|
-
"Our systems have detected unusual traffic from your computer network.",
|
|
167
|
-
"Your request has been blocked due to a network policy.",
|
|
168
|
-
"Cloudflare Ray ID found ",
|
|
169
|
-
"Please verify you are a human",
|
|
170
|
-
"Our systems have detected unusual traffic activity from your network. Please complete this reCAPTCHA",
|
|
171
|
-
"Sorry, we just need to make sure you're not a robot",
|
|
172
|
-
"Access to this page has been denied",
|
|
173
|
-
"<p>Please enable JS and disable any ad blocker",
|
|
174
|
-
"Please make sure your browser supports JavaScript",
|
|
175
|
-
"Please complete the security check to access",
|
|
176
|
-
"https://errors.edgesuite.net",
|
|
177
|
-
"Please enable JS and disable any ad blocker",
|
|
178
|
-
"The resource you are looking for might have been removed, had its name changed, or is temporarily unavailable.",
|
|
179
|
-
"We\u2019re currently checking your connection. This shouldn\u2019t take long.",
|
|
180
|
-
"Generated by cloudfront (CloudFront)",
|
|
181
|
-
"You don't have permission to access",
|
|
182
|
-
"The request could not be satisfied.",
|
|
183
|
-
"Enable JavaScript and cookies to continue",
|
|
184
|
-
"Something went wrong. Wait a moment and try again.",
|
|
185
|
-
"You\u2019re using a web browser that isn\u2019t supported",
|
|
186
|
-
"403 Forbidden",
|
|
187
|
-
"504 Gateway Timeout",
|
|
188
|
-
"You\u2019re Temporarily Blocked",
|
|
189
|
-
"Our systems have detected unusual activity",
|
|
190
|
-
"Agree & Join LinkedIn",
|
|
191
|
-
"Verifying you are human. This may take a few seconds",
|
|
192
|
-
"500 Internal Server Error",
|
|
193
|
-
"By clicking Continue to join or sign in, you agree to LinkedIn",
|
|
194
|
-
"Enable JS in your browser",
|
|
195
|
-
"Verifying you are human",
|
|
196
|
-
"Your request has been blocked",
|
|
197
|
-
"You've been blocked by network security",
|
|
198
|
-
"You've hit the rate limit.",
|
|
199
|
-
];
|
|
200
|
-
|
|
201
|
-
return (
|
|
202
|
-
html &&
|
|
203
|
-
typeof html?.indexOf !== "undefined" &&
|
|
204
|
-
commonBlocks.filter((m) => html?.indexOf(m) > -1).length > 0
|
|
205
|
-
);
|
|
206
|
-
}
|
|
207
|
-
|
|
208
|
-
/**
|
|
209
|
-
* Fetches and parses the robots.txt file for a given URL.
|
|
210
|
-
* @param {string} url - The base URL to fetch the robots.txt from.
|
|
211
|
-
* @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
|
|
212
|
-
*/
|
|
213
|
-
export async function fetchScrapingRules(url) {
|
|
214
|
-
const robotsUrl = `https://${url.split("//")[1].split("/")[0]}/robots.txt`;
|
|
215
|
-
let content;
|
|
216
|
-
try {
|
|
217
|
-
content = await grab(robotsUrl, { responseType: "text" });
|
|
218
|
-
} catch (e) {
|
|
219
|
-
return { error: "No robots.txt found" };
|
|
220
|
-
}
|
|
221
|
-
|
|
222
|
-
const rules = {
|
|
223
|
-
directives: {},
|
|
224
|
-
crawlDelay: {},
|
|
225
|
-
sitemaps: [],
|
|
226
|
-
preferredHost: null,
|
|
227
|
-
};
|
|
228
|
-
let currentUserAgents = [];
|
|
229
|
-
|
|
230
|
-
const lines = content.split("\n");
|
|
231
|
-
for (const line of lines) {
|
|
232
|
-
const [directive, value] = line.split(":").map((s) => s.trim());
|
|
233
|
-
switch (directive.toLowerCase()) {
|
|
234
|
-
case "user-agent":
|
|
235
|
-
currentUserAgents = [value.toLowerCase()];
|
|
236
|
-
break;
|
|
237
|
-
case "disallow":
|
|
238
|
-
case "allow":
|
|
239
|
-
for (const ua of currentUserAgents) {
|
|
240
|
-
rules.directives[ua] = rules.directives[ua] || [];
|
|
241
|
-
rules.directives[ua].push({
|
|
242
|
-
path: value,
|
|
243
|
-
allow: directive.toLowerCase() === "allow",
|
|
244
|
-
});
|
|
245
|
-
}
|
|
246
|
-
break;
|
|
247
|
-
case "crawl-delay":
|
|
248
|
-
for (const ua of currentUserAgents) {
|
|
249
|
-
rules.crawlDelay[ua] = parseFloat(value);
|
|
250
|
-
}
|
|
251
|
-
break;
|
|
252
|
-
case "sitemap":
|
|
253
|
-
rules.sitemaps.push(value);
|
|
254
|
-
break;
|
|
255
|
-
case "host":
|
|
256
|
-
rules.preferredHost = value.toLowerCase();
|
|
257
|
-
break;
|
|
258
|
-
}
|
|
259
|
-
}
|
|
260
|
-
return rules;
|
|
261
|
-
}
|
|
262
|
-
|
|
263
|
-
/**
|
|
264
|
-
* Checks if a given path is allowed for a specific user agent.
|
|
265
|
-
* //TODO cache rules per domain
|
|
266
|
-
* @param {Object} rules - The parsed rules from robots.txt.
|
|
267
|
-
* @param {string} path - The path to check.
|
|
268
|
-
* @param {string} [userAgent='*'] - The user agent to check for.
|
|
269
|
-
* @returns {boolean} True if the path is allowed, false otherwise.
|
|
270
|
-
*/
|
|
271
|
-
function isAllowedToScrape(rules, path, userAgent = "*") {
|
|
272
|
-
const relevantRules =
|
|
273
|
-
rules.directives[userAgent.toLowerCase()] || rules.directives["*"] || [];
|
|
274
|
-
for (const rule of relevantRules)
|
|
275
|
-
if (path.startsWith(rule.path)) return rule.allow;
|
|
276
|
-
|
|
277
|
-
return true; // If no rules match, it's allowed by default
|
|
278
|
-
}
|
|
1
|
+
import grab from "../utils/grab";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* ### Tardigrade the Web Crawler
|
|
5
|
+
*
|
|
6
|
+
* 1. **Use Fetch API, check for bot detection.
|
|
7
|
+
* Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
|
|
8
|
+
* Scraping internet pages is a [free speech right
|
|
9
|
+
* ](https://blog.apify.com/is-web-scraping-legal/).
|
|
10
|
+
* 2. Features: timeout, redirects, default UA, referer as google, and bot
|
|
11
|
+
* detection checking. <br />
|
|
12
|
+
* 3. If fetch method does not get needed HTML, use Docker proxy as backup.
|
|
13
|
+
*
|
|
14
|
+
* 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
|
|
15
|
+
* container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
|
|
16
|
+
* secondary in-page API requests after the initial page request, including user login and cookie storage.
|
|
17
|
+
* 5. Bypass Cloudflare bot check: A webpage proxy that request
|
|
18
|
+
* through Chromium (puppeteer) - can be used to bypass Cloudflare
|
|
19
|
+
* anti bot using cookie id javascript method.
|
|
20
|
+
* 6. Send your request to the server with the port 3000 and add your URL to the "url"
|
|
21
|
+
* query string like this: `http://localhost:3000/?url=https://example.org`
|
|
22
|
+
*
|
|
23
|
+
* 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
|
|
24
|
+
* and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
|
|
25
|
+
* [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
|
|
26
|
+
* [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
|
|
27
|
+
* [Proxy-Cheap](https://app.proxy-cheap.com/order)
|
|
28
|
+
* [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
|
|
29
|
+
*
|
|
30
|
+
* @param {string} url - any domain's URL
|
|
31
|
+
* @param {Object} [options]
|
|
32
|
+
* @param {number} options.timeout default=5 - abort request if not retrived, in seconds
|
|
33
|
+
* @param {number} options.maxRedirects default=3 - max redirects to follow
|
|
34
|
+
* @param {number} options.checkBotDetection default=true - check for bot detection messages
|
|
35
|
+
* @param {number} options.changeReferer default=true - set referer as google
|
|
36
|
+
* @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
|
|
37
|
+
* @param {string} options.proxy default=false - use proxy url
|
|
38
|
+
* @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
|
|
39
|
+
* @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
|
|
40
|
+
* @category Extract
|
|
41
|
+
* @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
|
|
42
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
43
|
+
* @license MIT
|
|
44
|
+
*/
|
|
45
|
+
export async function scrapeURL(url, options = {} as any) {
|
|
46
|
+
// try {
|
|
47
|
+
let {
|
|
48
|
+
timeout = 5,
|
|
49
|
+
checkBotDetection = true,
|
|
50
|
+
maxRedirects = 3,
|
|
51
|
+
changeReferer = 0,
|
|
52
|
+
userAgentIndex = 0,
|
|
53
|
+
proxy = null,
|
|
54
|
+
useProxyAsBackup = true,
|
|
55
|
+
checkRobotsAllowed = false,
|
|
56
|
+
} = options;
|
|
57
|
+
|
|
58
|
+
if (checkRobotsAllowed) {
|
|
59
|
+
const rules = await fetchScrapingRules(url);
|
|
60
|
+
if (!isAllowedToScrape(rules, url)) {
|
|
61
|
+
return { error: "Robots.txt forbids to scrape there" };
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
if (proxy) url = proxy + url;
|
|
66
|
+
|
|
67
|
+
var userAgentStrings = [
|
|
68
|
+
"Chrome/41.0.2272.96 Mobile Safari/537.36 (compatible ; Googlebot/2.1 ; +http://www.google.com/bot.html)",
|
|
69
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_4) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/85.0.4183.83 Safari/537.36,gzip(gfe)",
|
|
70
|
+
];
|
|
71
|
+
|
|
72
|
+
var headers = {
|
|
73
|
+
...options,
|
|
74
|
+
"User-Agent": userAgentStrings[userAgentIndex],
|
|
75
|
+
signal: AbortSignal.timeout(timeout * 1000),
|
|
76
|
+
accept:
|
|
77
|
+
"text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7",
|
|
78
|
+
"accept-language": "en-US,en;q=0.9",
|
|
79
|
+
};
|
|
80
|
+
|
|
81
|
+
if (changeReferer) headers["Referer"] = "https://www.google.com/";
|
|
82
|
+
|
|
83
|
+
let response;
|
|
84
|
+
try {
|
|
85
|
+
response = await grab(url, {
|
|
86
|
+
...headers,
|
|
87
|
+
responseType: "raw",
|
|
88
|
+
});
|
|
89
|
+
} catch (e) {
|
|
90
|
+
return { error: "Error in fetch", msg: e.message };
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
if (response.redirected) {
|
|
94
|
+
if (maxRedirects <= 0) return { error: "Max redirects exceeded" };
|
|
95
|
+
maxRedirects--;
|
|
96
|
+
options = { ...options, maxRedirects };
|
|
97
|
+
|
|
98
|
+
return scrapeURL(response.url, options);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
//return based on content type
|
|
102
|
+
const contentType = response.headers.get("Content-Type");
|
|
103
|
+
|
|
104
|
+
// if (contentType.includes("application/json")) {
|
|
105
|
+
// return await response.json();
|
|
106
|
+
// } else
|
|
107
|
+
// if (contentType.includes("text")) {
|
|
108
|
+
var html = await response.text();
|
|
109
|
+
|
|
110
|
+
if (checkBotDetection && checkHTMLForBotDetection(html)) {
|
|
111
|
+
html = await scrapeJINA(url, timeout);
|
|
112
|
+
|
|
113
|
+
//if all methods fail -- return jina
|
|
114
|
+
if (checkBotDetection && checkHTMLForBotDetection(html))
|
|
115
|
+
return { error: "Bot detected" }; //, html: response.html };
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
return html;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* As backup, scrape with JINA to get html
|
|
123
|
+
* @param {string} url
|
|
124
|
+
* @returns {Promise<string>}
|
|
125
|
+
*/
|
|
126
|
+
export async function scrapeJINA(url, timeout = 15) {
|
|
127
|
+
let articleExtract = "";
|
|
128
|
+
try {
|
|
129
|
+
articleExtract = await grab("https://r.jina.ai/" + url, {
|
|
130
|
+
timeout: timeout * 1000,
|
|
131
|
+
responseType: "text",
|
|
132
|
+
});
|
|
133
|
+
} catch {
|
|
134
|
+
return "";
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
//convert Title: to <title>
|
|
138
|
+
var title = articleExtract.match(/Title: (.*)/)?.[1];
|
|
139
|
+
|
|
140
|
+
if (articleExtract.includes("===============\n"))
|
|
141
|
+
articleExtract = articleExtract
|
|
142
|
+
.split("===============\n")
|
|
143
|
+
.slice(1)
|
|
144
|
+
.join(" ");
|
|
145
|
+
|
|
146
|
+
var match = articleExtract.match(/Markdown Content:([\s\S]*)/);
|
|
147
|
+
articleExtract = match ? match[1] : articleExtract;
|
|
148
|
+
|
|
149
|
+
// articleExtract = convertMarkdownToHTML(articleExtract);
|
|
150
|
+
|
|
151
|
+
if (title) articleExtract = "<title>" + title + "</title>" + articleExtract;
|
|
152
|
+
|
|
153
|
+
return articleExtract;
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Check html for bot block messages
|
|
158
|
+
* @param {string} html
|
|
159
|
+
* @returns {Boolean} true if bot detection message found
|
|
160
|
+
*/
|
|
161
|
+
function checkHTMLForBotDetection(html) {
|
|
162
|
+
var commonBlocks = [
|
|
163
|
+
"Error 403 - Unavailable",
|
|
164
|
+
"The security system for this website has been triggered",
|
|
165
|
+
"You do not have permission to view this page.",
|
|
166
|
+
"Our systems have detected unusual traffic from your computer network.",
|
|
167
|
+
"Your request has been blocked due to a network policy.",
|
|
168
|
+
"Cloudflare Ray ID found ",
|
|
169
|
+
"Please verify you are a human",
|
|
170
|
+
"Our systems have detected unusual traffic activity from your network. Please complete this reCAPTCHA",
|
|
171
|
+
"Sorry, we just need to make sure you're not a robot",
|
|
172
|
+
"Access to this page has been denied",
|
|
173
|
+
"<p>Please enable JS and disable any ad blocker",
|
|
174
|
+
"Please make sure your browser supports JavaScript",
|
|
175
|
+
"Please complete the security check to access",
|
|
176
|
+
"https://errors.edgesuite.net",
|
|
177
|
+
"Please enable JS and disable any ad blocker",
|
|
178
|
+
"The resource you are looking for might have been removed, had its name changed, or is temporarily unavailable.",
|
|
179
|
+
"We\u2019re currently checking your connection. This shouldn\u2019t take long.",
|
|
180
|
+
"Generated by cloudfront (CloudFront)",
|
|
181
|
+
"You don't have permission to access",
|
|
182
|
+
"The request could not be satisfied.",
|
|
183
|
+
"Enable JavaScript and cookies to continue",
|
|
184
|
+
"Something went wrong. Wait a moment and try again.",
|
|
185
|
+
"You\u2019re using a web browser that isn\u2019t supported",
|
|
186
|
+
"403 Forbidden",
|
|
187
|
+
"504 Gateway Timeout",
|
|
188
|
+
"You\u2019re Temporarily Blocked",
|
|
189
|
+
"Our systems have detected unusual activity",
|
|
190
|
+
"Agree & Join LinkedIn",
|
|
191
|
+
"Verifying you are human. This may take a few seconds",
|
|
192
|
+
"500 Internal Server Error",
|
|
193
|
+
"By clicking Continue to join or sign in, you agree to LinkedIn",
|
|
194
|
+
"Enable JS in your browser",
|
|
195
|
+
"Verifying you are human",
|
|
196
|
+
"Your request has been blocked",
|
|
197
|
+
"You've been blocked by network security",
|
|
198
|
+
"You've hit the rate limit.",
|
|
199
|
+
];
|
|
200
|
+
|
|
201
|
+
return (
|
|
202
|
+
html &&
|
|
203
|
+
typeof html?.indexOf !== "undefined" &&
|
|
204
|
+
commonBlocks.filter((m) => html?.indexOf(m) > -1).length > 0
|
|
205
|
+
);
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Fetches and parses the robots.txt file for a given URL.
|
|
210
|
+
* @param {string} url - The base URL to fetch the robots.txt from.
|
|
211
|
+
* @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
|
|
212
|
+
*/
|
|
213
|
+
export async function fetchScrapingRules(url) {
|
|
214
|
+
const robotsUrl = `https://${url.split("//")[1].split("/")[0]}/robots.txt`;
|
|
215
|
+
let content;
|
|
216
|
+
try {
|
|
217
|
+
content = await grab(robotsUrl, { responseType: "text" });
|
|
218
|
+
} catch (e) {
|
|
219
|
+
return { error: "No robots.txt found" };
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
const rules = {
|
|
223
|
+
directives: {},
|
|
224
|
+
crawlDelay: {},
|
|
225
|
+
sitemaps: [],
|
|
226
|
+
preferredHost: null,
|
|
227
|
+
};
|
|
228
|
+
let currentUserAgents = [];
|
|
229
|
+
|
|
230
|
+
const lines = content.split("\n");
|
|
231
|
+
for (const line of lines) {
|
|
232
|
+
const [directive, value] = line.split(":").map((s) => s.trim());
|
|
233
|
+
switch (directive.toLowerCase()) {
|
|
234
|
+
case "user-agent":
|
|
235
|
+
currentUserAgents = [value.toLowerCase()];
|
|
236
|
+
break;
|
|
237
|
+
case "disallow":
|
|
238
|
+
case "allow":
|
|
239
|
+
for (const ua of currentUserAgents) {
|
|
240
|
+
rules.directives[ua] = rules.directives[ua] || [];
|
|
241
|
+
rules.directives[ua].push({
|
|
242
|
+
path: value,
|
|
243
|
+
allow: directive.toLowerCase() === "allow",
|
|
244
|
+
});
|
|
245
|
+
}
|
|
246
|
+
break;
|
|
247
|
+
case "crawl-delay":
|
|
248
|
+
for (const ua of currentUserAgents) {
|
|
249
|
+
rules.crawlDelay[ua] = parseFloat(value);
|
|
250
|
+
}
|
|
251
|
+
break;
|
|
252
|
+
case "sitemap":
|
|
253
|
+
rules.sitemaps.push(value);
|
|
254
|
+
break;
|
|
255
|
+
case "host":
|
|
256
|
+
rules.preferredHost = value.toLowerCase();
|
|
257
|
+
break;
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
return rules;
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
/**
|
|
264
|
+
* Checks if a given path is allowed for a specific user agent.
|
|
265
|
+
* //TODO cache rules per domain
|
|
266
|
+
* @param {Object} rules - The parsed rules from robots.txt.
|
|
267
|
+
* @param {string} path - The path to check.
|
|
268
|
+
* @param {string} [userAgent='*'] - The user agent to check for.
|
|
269
|
+
* @returns {boolean} True if the path is allowed, false otherwise.
|
|
270
|
+
*/
|
|
271
|
+
function isAllowedToScrape(rules, path, userAgent = "*") {
|
|
272
|
+
const relevantRules =
|
|
273
|
+
rules.directives[userAgent.toLowerCase()] || rules.directives["*"] || [];
|
|
274
|
+
for (const rule of relevantRules)
|
|
275
|
+
if (path.startsWith(rule.path)) return rule.allow;
|
|
276
|
+
|
|
277
|
+
return true; // If no rules match, it's allowed by default
|
|
278
|
+
}
|