extract-webpage 1.2.49 → 1.2.51

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,106 +0,0 @@
1
- /**
2
- * @module research/search/tavily
3
- * @description Research library module.
4
- */
5
- import axios from "axios";
6
- import configManager from "../config";
7
-
8
- interface TavilySearchOptions {
9
- searchDepth?: "basic" | "advanced";
10
- maxResults?: number;
11
- includeDomains?: string[];
12
- excludeDomains?: string[];
13
- }
14
-
15
- interface TavilySearchResult {
16
- title: string;
17
- url: string;
18
- content: string;
19
- score: number;
20
- raw_content?: string;
21
- }
22
-
23
- interface TavilyResponse {
24
- results: TavilySearchResult[];
25
- query: string;
26
- }
27
-
28
- export const searchTavily = async (
29
- query: string,
30
- opts?: TavilySearchOptions,
31
- ): Promise<{ results: TavilySearchResult[]; suggestions: string[] }> => {
32
- const tavilyApiKey =
33
- configManager.getConfig("search.tavilyApiKey", "") ||
34
- (typeof process !== "undefined" ? process.env.TAVILY_API_KEY : "") ||
35
- "";
36
-
37
- if (!tavilyApiKey) {
38
- throw new Error(
39
- "Tavily API key not configured. Please add your API key in Settings > Search.",
40
- );
41
- }
42
-
43
- // Sanitize query - Tavily has a maximum query length limit (~400 characters)
44
- // If the query is too long, truncate it to avoid 400 Bad Request errors
45
- let sanitizedQuery = query.trim();
46
- if (sanitizedQuery.length > 400) {
47
- console.warn(`[Tavily] Query too long (${sanitizedQuery.length} chars), truncating to 400 chars`);
48
- sanitizedQuery = sanitizedQuery.slice(0, 400);
49
- }
50
-
51
- try {
52
- const response = await axios.post<TavilyResponse>(
53
- "https://api.tavily.com/search",
54
- {
55
- api_key: tavilyApiKey,
56
- query: sanitizedQuery,
57
- search_depth: opts?.searchDepth || "basic",
58
- max_results: opts?.maxResults || 10,
59
- include_domains: opts?.includeDomains || [],
60
- exclude_domains: opts?.excludeDomains || [],
61
- include_answer: false,
62
- include_raw_content: false,
63
- },
64
- {
65
- headers: {
66
- "Content-Type": "application/json",
67
- },
68
- },
69
- );
70
-
71
- const results = response.data.results || [];
72
-
73
- return {
74
- results: results.map((r) => ({
75
- title: r.title,
76
- url: r.url,
77
- content: r.content,
78
- score: r.score,
79
- raw_content: r.raw_content,
80
- })),
81
- suggestions: [],
82
- };
83
- } catch (error: any) {
84
- console.error("Tavily search error:", error);
85
-
86
- // Provide more context for debugging
87
- if (error.response?.status === 400) {
88
- console.error("Tavily 400 error - Query length:", sanitizedQuery.length);
89
- console.error("Tavily 400 error - Query preview:", sanitizedQuery.slice(0, 200));
90
- }
91
-
92
- throw new Error(
93
- `Tavily search failed: ${error.response?.data?.error || error.message}`,
94
- );
95
- }
96
- };
97
-
98
- export const getTavilyApiKey = () =>
99
- configManager.getConfig("search.tavilyApiKey", "") ||
100
- (typeof process !== "undefined" ? process.env.TAVILY_API_KEY : "") ||
101
- "";
102
-
103
- export const isTavilyConfigured = () => {
104
- const apiKey = getTavilyApiKey();
105
- return apiKey && apiKey.length > 0;
106
- };
@@ -1,278 +0,0 @@
1
- import grab from "../utils/grab";
2
-
3
- /**
4
- * ### Tardigrade the Web Crawler
5
- *
6
- * 1. **Use Fetch API, check for bot detection.
7
- * Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
8
- * Scraping internet pages is a [free speech right
9
- * ](https://blog.apify.com/is-web-scraping-legal/).
10
- * 2. Features: timeout, redirects, default UA, referer as google, and bot
11
- * detection checking. <br />
12
- * 3. If fetch method does not get needed HTML, use Docker proxy as backup.
13
- *
14
- * 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
15
- * container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
16
- * secondary in-page API requests after the initial page request, including user login and cookie storage.
17
- * 5. Bypass Cloudflare bot check: A webpage proxy that request
18
- * through Chromium (puppeteer) - can be used to bypass Cloudflare
19
- * anti bot using cookie id javascript method.
20
- * 6. Send your request to the server with the port 3000 and add your URL to the "url"
21
- * query string like this: `http://localhost:3000/?url=https://example.org`
22
- *
23
- * 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
24
- * and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
25
- * [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
26
- * [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
27
- * [Proxy-Cheap](https://app.proxy-cheap.com/order)
28
- * [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
29
- *
30
- * @param {string} url - any domain's URL
31
- * @param {Object} [options]
32
- * @param {number} options.timeout default=5 - abort request if not retrived, in seconds
33
- * @param {number} options.maxRedirects default=3 - max redirects to follow
34
- * @param {number} options.checkBotDetection default=true - check for bot detection messages
35
- * @param {number} options.changeReferer default=true - set referer as google
36
- * @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
37
- * @param {string} options.proxy default=false - use proxy url
38
- * @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
39
- * @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
40
- * @category Extract
41
- * @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
42
- * @author [vtempest (2025)](https://github.com/vtempest)
43
- * @license MIT
44
- */
45
- export async function scrapeURL(url, options = {} as any) {
46
- // try {
47
- let {
48
- timeout = 5,
49
- checkBotDetection = true,
50
- maxRedirects = 3,
51
- changeReferer = 0,
52
- userAgentIndex = 0,
53
- proxy = null,
54
- useProxyAsBackup = true,
55
- checkRobotsAllowed = false,
56
- } = options;
57
-
58
- if (checkRobotsAllowed) {
59
- const rules = await fetchScrapingRules(url);
60
- if (!isAllowedToScrape(rules, url)) {
61
- return { error: "Robots.txt forbids to scrape there" };
62
- }
63
- }
64
-
65
- if (proxy) url = proxy + url;
66
-
67
- var userAgentStrings = [
68
- "Chrome/41.0.2272.96 Mobile Safari/537.36 (compatible ; Googlebot/2.1 ; +http://www.google.com/bot.html)",
69
- "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_4) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/85.0.4183.83 Safari/537.36,gzip(gfe)",
70
- ];
71
-
72
- var headers = {
73
- ...options,
74
- "User-Agent": userAgentStrings[userAgentIndex],
75
- signal: AbortSignal.timeout(timeout * 1000),
76
- accept:
77
- "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7",
78
- "accept-language": "en-US,en;q=0.9",
79
- };
80
-
81
- if (changeReferer) headers["Referer"] = "https://www.google.com/";
82
-
83
- let response;
84
- try {
85
- response = await grab(url, {
86
- ...headers,
87
- responseType: "raw",
88
- });
89
- } catch (e) {
90
- return { error: "Error in fetch", msg: e.message };
91
- }
92
-
93
- if (response.redirected) {
94
- if (maxRedirects <= 0) return { error: "Max redirects exceeded" };
95
- maxRedirects--;
96
- options = { ...options, maxRedirects };
97
-
98
- return scrapeURL(response.url, options);
99
- }
100
-
101
- //return based on content type
102
- const contentType = response.headers.get("Content-Type");
103
-
104
- // if (contentType.includes("application/json")) {
105
- // return await response.json();
106
- // } else
107
- // if (contentType.includes("text")) {
108
- var html = await response.text();
109
-
110
- if (checkBotDetection && checkHTMLForBotDetection(html)) {
111
- html = await scrapeJINA(url, timeout);
112
-
113
- //if all methods fail -- return jina
114
- if (checkBotDetection && checkHTMLForBotDetection(html))
115
- return { error: "Bot detected" }; //, html: response.html };
116
- }
117
-
118
- return html;
119
- }
120
-
121
- /**
122
- * As backup, scrape with JINA to get html
123
- * @param {string} url
124
- * @returns {Promise<string>}
125
- */
126
- export async function scrapeJINA(url, timeout = 15) {
127
- let articleExtract = "";
128
- try {
129
- articleExtract = await grab("https://r.jina.ai/" + url, {
130
- timeout: timeout * 1000,
131
- responseType: "text",
132
- });
133
- } catch {
134
- return "";
135
- }
136
-
137
- //convert Title: to <title>
138
- var title = articleExtract.match(/Title: (.*)/)?.[1];
139
-
140
- if (articleExtract.includes("===============\n"))
141
- articleExtract = articleExtract
142
- .split("===============\n")
143
- .slice(1)
144
- .join(" ");
145
-
146
- var match = articleExtract.match(/Markdown Content:([\s\S]*)/);
147
- articleExtract = match ? match[1] : articleExtract;
148
-
149
- // articleExtract = convertMarkdownToHTML(articleExtract);
150
-
151
- if (title) articleExtract = "<title>" + title + "</title>" + articleExtract;
152
-
153
- return articleExtract;
154
- }
155
-
156
- /**
157
- * Check html for bot block messages
158
- * @param {string} html
159
- * @returns {Boolean} true if bot detection message found
160
- */
161
- function checkHTMLForBotDetection(html) {
162
- var commonBlocks = [
163
- "Error 403 - Unavailable",
164
- "The security system for this website has been triggered",
165
- "You do not have permission to view this page.",
166
- "Our systems have detected unusual traffic from your computer network.",
167
- "Your request has been blocked due to a network policy.",
168
- "Cloudflare Ray ID found ",
169
- "Please verify you are a human",
170
- "Our systems have detected unusual traffic activity from your network. Please complete this reCAPTCHA",
171
- "Sorry, we just need to make sure you're not a robot",
172
- "Access to this page has been denied",
173
- "<p>Please enable JS and disable any ad blocker",
174
- "Please make sure your browser supports JavaScript",
175
- "Please complete the security check to access",
176
- "https://errors.edgesuite.net",
177
- "Please enable JS and disable any ad blocker",
178
- "The resource you are looking for might have been removed, had its name changed, or is temporarily unavailable.",
179
- "We\u2019re currently checking your connection. This shouldn\u2019t take long.",
180
- "Generated by cloudfront (CloudFront)",
181
- "You don't have permission to access",
182
- "The request could not be satisfied.",
183
- "Enable JavaScript and cookies to continue",
184
- "Something went wrong. Wait a moment and try again.",
185
- "You\u2019re using a web browser that isn\u2019t supported",
186
- "403 Forbidden",
187
- "504 Gateway Timeout",
188
- "You\u2019re Temporarily Blocked",
189
- "Our systems have detected unusual activity",
190
- "Agree & Join LinkedIn",
191
- "Verifying you are human. This may take a few seconds",
192
- "500 Internal Server Error",
193
- "By clicking Continue to join or sign in, you agree to LinkedIn",
194
- "Enable JS in your browser",
195
- "Verifying you are human",
196
- "Your request has been blocked",
197
- "You've been blocked by network security",
198
- "You've hit the rate limit.",
199
- ];
200
-
201
- return (
202
- html &&
203
- typeof html?.indexOf !== "undefined" &&
204
- commonBlocks.filter((m) => html?.indexOf(m) > -1).length > 0
205
- );
206
- }
207
-
208
- /**
209
- * Fetches and parses the robots.txt file for a given URL.
210
- * @param {string} url - The base URL to fetch the robots.txt from.
211
- * @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
212
- */
213
- export async function fetchScrapingRules(url) {
214
- const robotsUrl = `https://${url.split("//")[1].split("/")[0]}/robots.txt`;
215
- let content;
216
- try {
217
- content = await grab(robotsUrl, { responseType: "text" });
218
- } catch (e) {
219
- return { error: "No robots.txt found" };
220
- }
221
-
222
- const rules = {
223
- directives: {},
224
- crawlDelay: {},
225
- sitemaps: [],
226
- preferredHost: null,
227
- };
228
- let currentUserAgents = [];
229
-
230
- const lines = content.split("\n");
231
- for (const line of lines) {
232
- const [directive, value] = line.split(":").map((s) => s.trim());
233
- switch (directive.toLowerCase()) {
234
- case "user-agent":
235
- currentUserAgents = [value.toLowerCase()];
236
- break;
237
- case "disallow":
238
- case "allow":
239
- for (const ua of currentUserAgents) {
240
- rules.directives[ua] = rules.directives[ua] || [];
241
- rules.directives[ua].push({
242
- path: value,
243
- allow: directive.toLowerCase() === "allow",
244
- });
245
- }
246
- break;
247
- case "crawl-delay":
248
- for (const ua of currentUserAgents) {
249
- rules.crawlDelay[ua] = parseFloat(value);
250
- }
251
- break;
252
- case "sitemap":
253
- rules.sitemaps.push(value);
254
- break;
255
- case "host":
256
- rules.preferredHost = value.toLowerCase();
257
- break;
258
- }
259
- }
260
- return rules;
261
- }
262
-
263
- /**
264
- * Checks if a given path is allowed for a specific user agent.
265
- * //TODO cache rules per domain
266
- * @param {Object} rules - The parsed rules from robots.txt.
267
- * @param {string} path - The path to check.
268
- * @param {string} [userAgent='*'] - The user agent to check for.
269
- * @returns {boolean} True if the path is allowed, false otherwise.
270
- */
271
- function isAllowedToScrape(rules, path, userAgent = "*") {
272
- const relevantRules =
273
- rules.directives[userAgent.toLowerCase()] || rules.directives["*"] || [];
274
- for (const rule of relevantRules)
275
- if (path.startsWith(rule.path)) return rule.allow;
276
-
277
- return true; // If no rules match, it's allowed by default
278
- }