extract-webpage 1.2.115 → 1.2.117
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md
CHANGED
|
@@ -2,11 +2,8 @@
|
|
|
2
2
|
<br />
|
|
3
3
|
<a href="https://www.npmjs.com/package/extract-webpage"><img src="https://img.shields.io/npm/dm/extract-webpage.svg" alt="NPM Monthly Downloads"></a>
|
|
4
4
|
<a href="https://www.npmjs.com/package/extract-webpage"><img src="https://img.shields.io/npm/v/extract-webpage.svg" alt="npm version"></a>
|
|
5
|
-
<a href="https://
|
|
6
|
-
<img
|
|
7
|
-
src="https://img.shields.io/bundlephobia/minzip/extract-webpage?style=flat&label=size"
|
|
8
|
-
alt="npm bundle size "
|
|
9
|
-
/>
|
|
5
|
+
<a href="https://packagephobia.com/result?p=extract-webpage" target="_blank" rel="noopener noreferrer">
|
|
6
|
+
<img src="https://packagephobia.com/badge?p=extract-webpage" alt="install size" />
|
|
10
7
|
</a>
|
|
11
8
|
<a href="https://discord.gg/SJdBqBz3tV">
|
|
12
9
|
<img src="https://img.shields.io/discord/1110227955554209923.svg?label=Chat&logo=Discord&colorB=7289da&style=flat"
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Client for the Cloudflare Puppeteer-based scraper service.
|
|
3
|
+
* Renders JavaScript-heavy pages and bypasses bot detection.
|
|
4
|
+
*
|
|
5
|
+
* This wraps the scraper-cloudflare package deployed as a Cloudflare Worker
|
|
6
|
+
* with Browser Rendering (Puppeteer) support.
|
|
7
|
+
*/
|
|
8
|
+
export interface ScraperOptions {
|
|
9
|
+
/** URL to render */
|
|
10
|
+
url: string;
|
|
11
|
+
/** API key for authentication (optional if SCRAPER_API_KEY not set) */
|
|
12
|
+
apiKey?: string;
|
|
13
|
+
/** Additional wait time after page load (ms) */
|
|
14
|
+
wait?: number;
|
|
15
|
+
/** Block image loading for faster rendering */
|
|
16
|
+
blockImages?: boolean;
|
|
17
|
+
/** Session ID for cookie persistence */
|
|
18
|
+
sessionId?: string;
|
|
19
|
+
/** Navigation timeout (ms) */
|
|
20
|
+
timeout?: number;
|
|
21
|
+
/** Puppeteer waitUntil condition */
|
|
22
|
+
waitUntil?: 'domcontentloaded' | 'load' | 'networkidle0' | 'networkidle2';
|
|
23
|
+
/** Response format */
|
|
24
|
+
format?: 'html' | 'json';
|
|
25
|
+
/** Custom headers */
|
|
26
|
+
headers?: Record<string, string>;
|
|
27
|
+
/** Proxy configuration */
|
|
28
|
+
proxyUrl?: string;
|
|
29
|
+
proxyUser?: string;
|
|
30
|
+
proxyPass?: string;
|
|
31
|
+
/** Bypass Cloudflare challenges and CAPTCHAs */
|
|
32
|
+
bypassCaptcha?: boolean;
|
|
33
|
+
/** Challenge detection pattern */
|
|
34
|
+
challengeMatch?: string;
|
|
35
|
+
/** Max retry attempts for challenges */
|
|
36
|
+
maxRetries?: number;
|
|
37
|
+
/** 2Captcha API key for solving */
|
|
38
|
+
twoCaptchaKey?: string;
|
|
39
|
+
/** Abort signal to bound the request (e.g. an 8s deadline). */
|
|
40
|
+
signal?: AbortSignal;
|
|
41
|
+
}
|
|
42
|
+
export interface ScraperJsonResponse {
|
|
43
|
+
html: string;
|
|
44
|
+
url: string;
|
|
45
|
+
title: string;
|
|
46
|
+
cookies: Array<{
|
|
47
|
+
name: string;
|
|
48
|
+
value: string;
|
|
49
|
+
domain: string;
|
|
50
|
+
path: string;
|
|
51
|
+
expires?: number;
|
|
52
|
+
httpOnly?: boolean;
|
|
53
|
+
secure?: boolean;
|
|
54
|
+
sameSite?: 'Strict' | 'Lax' | 'None';
|
|
55
|
+
}>;
|
|
56
|
+
challengeBypassed: boolean;
|
|
57
|
+
retryCount: number;
|
|
58
|
+
loadTime: number;
|
|
59
|
+
}
|
|
60
|
+
export interface ScraperConfig {
|
|
61
|
+
/** Base URL of the scraper service */
|
|
62
|
+
baseURL: string;
|
|
63
|
+
/** Global API key */
|
|
64
|
+
apiKey?: string;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* Renders a URL using the Cloudflare Puppeteer scraper service.
|
|
68
|
+
* Supports JavaScript rendering, bot detection bypass, and session management.
|
|
69
|
+
*
|
|
70
|
+
* @param options - Scraping configuration
|
|
71
|
+
* @param config - Service configuration (base URL and API key)
|
|
72
|
+
* @returns Rendered HTML or structured JSON response
|
|
73
|
+
*
|
|
74
|
+
* @example
|
|
75
|
+
* ```ts
|
|
76
|
+
* // Basic usage
|
|
77
|
+
* const html = await renderWithCloudflare({ url: 'https://example.com' });
|
|
78
|
+
*
|
|
79
|
+
* // With challenge bypass
|
|
80
|
+
* const result = await renderWithCloudflare({
|
|
81
|
+
* url: 'https://protected-site.com',
|
|
82
|
+
* bypassCaptcha: true,
|
|
83
|
+
* format: 'json'
|
|
84
|
+
* });
|
|
85
|
+
*
|
|
86
|
+
* // With session management
|
|
87
|
+
* const html = await renderWithCloudflare({
|
|
88
|
+
* url: 'https://site-requiring-login.com',
|
|
89
|
+
* sessionId: 'user-123',
|
|
90
|
+
* blockImages: true
|
|
91
|
+
* });
|
|
92
|
+
* ```
|
|
93
|
+
*/
|
|
94
|
+
export declare function renderWithCloudflare(options: ScraperOptions, config?: Partial<ScraperConfig>): Promise<string | ScraperJsonResponse>;
|
|
95
|
+
/**
|
|
96
|
+
* Convenience function to render a URL and return just the HTML content.
|
|
97
|
+
*
|
|
98
|
+
* @param url - URL to render
|
|
99
|
+
* @param options - Additional scraping options
|
|
100
|
+
* @param config - Service configuration
|
|
101
|
+
* @returns Rendered HTML string
|
|
102
|
+
*/
|
|
103
|
+
export declare function renderUrlToHtml(url: string, options?: Omit<ScraperOptions, 'url'>, config?: Partial<ScraperConfig>): Promise<string>;
|
|
104
|
+
/**
|
|
105
|
+
* Renders a URL and returns full metadata including cookies, load time, etc.
|
|
106
|
+
*
|
|
107
|
+
* @param url - URL to render
|
|
108
|
+
* @param options - Additional scraping options
|
|
109
|
+
* @param config - Service configuration
|
|
110
|
+
* @returns Structured response with HTML and metadata
|
|
111
|
+
*/
|
|
112
|
+
export declare function renderUrlWithMetadata(url: string, options?: Omit<ScraperOptions, 'url' | 'format'>, config?: Partial<ScraperConfig>): Promise<ScraperJsonResponse>;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "extract-webpage",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.117",
|
|
4
4
|
"module": "./dist/extract-webpage.es.js",
|
|
5
5
|
"description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
|
|
6
6
|
"author": "vtempest <grokthiscontact@gmail.com>",
|
|
@@ -72,11 +72,11 @@
|
|
|
72
72
|
"dependencies": {
|
|
73
73
|
"@huggingface/transformers": "^3.8.1",
|
|
74
74
|
"ai": "^5.0.0",
|
|
75
|
-
"chat-agent-toolkit": "^1.2.
|
|
75
|
+
"chat-agent-toolkit": "^1.2.117",
|
|
76
76
|
"chrono-node": "^2.9.0",
|
|
77
77
|
"drizzle-orm": "^0.45.1",
|
|
78
|
-
"extract-pdf": "^0.1.
|
|
79
|
-
"extract-youtube": "^1.0.
|
|
78
|
+
"extract-pdf": "^0.1.104",
|
|
79
|
+
"extract-youtube": "^1.0.102",
|
|
80
80
|
"html-entities": "^2.6.0",
|
|
81
81
|
"js-yaml": "^4.1.1",
|
|
82
82
|
"jsdom": "^28.1.0",
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Client for the Cloudflare Puppeteer-based scraper service.
|
|
3
|
+
* Renders JavaScript-heavy pages and bypasses bot detection.
|
|
4
|
+
*
|
|
5
|
+
* This wraps the scraper-cloudflare package deployed as a Cloudflare Worker
|
|
6
|
+
* with Browser Rendering (Puppeteer) support.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
export interface ScraperOptions {
|
|
10
|
+
/** URL to render */
|
|
11
|
+
url: string;
|
|
12
|
+
/** API key for authentication (optional if SCRAPER_API_KEY not set) */
|
|
13
|
+
apiKey?: string;
|
|
14
|
+
/** Additional wait time after page load (ms) */
|
|
15
|
+
wait?: number;
|
|
16
|
+
/** Block image loading for faster rendering */
|
|
17
|
+
blockImages?: boolean;
|
|
18
|
+
/** Session ID for cookie persistence */
|
|
19
|
+
sessionId?: string;
|
|
20
|
+
/** Navigation timeout (ms) */
|
|
21
|
+
timeout?: number;
|
|
22
|
+
/** Puppeteer waitUntil condition */
|
|
23
|
+
waitUntil?: 'domcontentloaded' | 'load' | 'networkidle0' | 'networkidle2';
|
|
24
|
+
/** Response format */
|
|
25
|
+
format?: 'html' | 'json';
|
|
26
|
+
/** Custom headers */
|
|
27
|
+
headers?: Record<string, string>;
|
|
28
|
+
/** Proxy configuration */
|
|
29
|
+
proxyUrl?: string;
|
|
30
|
+
proxyUser?: string;
|
|
31
|
+
proxyPass?: string;
|
|
32
|
+
/** Bypass Cloudflare challenges and CAPTCHAs */
|
|
33
|
+
bypassCaptcha?: boolean;
|
|
34
|
+
/** Challenge detection pattern */
|
|
35
|
+
challengeMatch?: string;
|
|
36
|
+
/** Max retry attempts for challenges */
|
|
37
|
+
maxRetries?: number;
|
|
38
|
+
/** 2Captcha API key for solving */
|
|
39
|
+
twoCaptchaKey?: string;
|
|
40
|
+
/** Abort signal to bound the request (e.g. an 8s deadline). */
|
|
41
|
+
signal?: AbortSignal;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export interface ScraperJsonResponse {
|
|
45
|
+
html: string;
|
|
46
|
+
url: string;
|
|
47
|
+
title: string;
|
|
48
|
+
cookies: Array<{
|
|
49
|
+
name: string;
|
|
50
|
+
value: string;
|
|
51
|
+
domain: string;
|
|
52
|
+
path: string;
|
|
53
|
+
expires?: number;
|
|
54
|
+
httpOnly?: boolean;
|
|
55
|
+
secure?: boolean;
|
|
56
|
+
sameSite?: 'Strict' | 'Lax' | 'None';
|
|
57
|
+
}>;
|
|
58
|
+
challengeBypassed: boolean;
|
|
59
|
+
retryCount: number;
|
|
60
|
+
loadTime: number;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export interface ScraperConfig {
|
|
64
|
+
/** Base URL of the scraper service */
|
|
65
|
+
baseURL: string;
|
|
66
|
+
/** Global API key */
|
|
67
|
+
apiKey?: string;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
const DEFAULT_CONFIG: ScraperConfig = {
|
|
71
|
+
baseURL: typeof process !== 'undefined' && process?.env?.SCRAPER_URL
|
|
72
|
+
? process.env.SCRAPER_URL
|
|
73
|
+
: 'https://proxy.qwksearch.com',
|
|
74
|
+
apiKey: typeof process !== 'undefined' && process?.env?.SCRAPER_API_KEY
|
|
75
|
+
? process.env.SCRAPER_API_KEY
|
|
76
|
+
: undefined,
|
|
77
|
+
};
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Renders a URL using the Cloudflare Puppeteer scraper service.
|
|
81
|
+
* Supports JavaScript rendering, bot detection bypass, and session management.
|
|
82
|
+
*
|
|
83
|
+
* @param options - Scraping configuration
|
|
84
|
+
* @param config - Service configuration (base URL and API key)
|
|
85
|
+
* @returns Rendered HTML or structured JSON response
|
|
86
|
+
*
|
|
87
|
+
* @example
|
|
88
|
+
* ```ts
|
|
89
|
+
* // Basic usage
|
|
90
|
+
* const html = await renderWithCloudflare({ url: 'https://example.com' });
|
|
91
|
+
*
|
|
92
|
+
* // With challenge bypass
|
|
93
|
+
* const result = await renderWithCloudflare({
|
|
94
|
+
* url: 'https://protected-site.com',
|
|
95
|
+
* bypassCaptcha: true,
|
|
96
|
+
* format: 'json'
|
|
97
|
+
* });
|
|
98
|
+
*
|
|
99
|
+
* // With session management
|
|
100
|
+
* const html = await renderWithCloudflare({
|
|
101
|
+
* url: 'https://site-requiring-login.com',
|
|
102
|
+
* sessionId: 'user-123',
|
|
103
|
+
* blockImages: true
|
|
104
|
+
* });
|
|
105
|
+
* ```
|
|
106
|
+
*/
|
|
107
|
+
export async function renderWithCloudflare(
|
|
108
|
+
options: ScraperOptions,
|
|
109
|
+
config: Partial<ScraperConfig> = {}
|
|
110
|
+
): Promise<string | ScraperJsonResponse> {
|
|
111
|
+
const mergedConfig = { ...DEFAULT_CONFIG, ...config };
|
|
112
|
+
const apiKey = options.apiKey || mergedConfig.apiKey;
|
|
113
|
+
|
|
114
|
+
// The deployed scraper worker (proxy.qwksearch.com) accepts GET requests
|
|
115
|
+
// with query-string parameters on `/` and `/api/render`.
|
|
116
|
+
const url = new URL('/api/render', mergedConfig.baseURL);
|
|
117
|
+
|
|
118
|
+
const params: Record<string, string | number | boolean | undefined> = {
|
|
119
|
+
url: options.url,
|
|
120
|
+
wait: options.wait ?? 0,
|
|
121
|
+
blockImages: options.blockImages ?? false,
|
|
122
|
+
sessionId: options.sessionId ?? 'default',
|
|
123
|
+
timeout: options.timeout ?? 30000,
|
|
124
|
+
waitUntil: options.waitUntil ?? 'networkidle2',
|
|
125
|
+
format: options.format ?? 'html',
|
|
126
|
+
proxyUrl: options.proxyUrl,
|
|
127
|
+
proxyUser: options.proxyUser,
|
|
128
|
+
proxyPass: options.proxyPass,
|
|
129
|
+
bypassCaptcha: options.bypassCaptcha ?? true,
|
|
130
|
+
challengeMatch: options.challengeMatch,
|
|
131
|
+
maxRetries: options.maxRetries ?? 10,
|
|
132
|
+
twoCaptchaKey: options.twoCaptchaKey,
|
|
133
|
+
};
|
|
134
|
+
|
|
135
|
+
for (const [key, value] of Object.entries(params)) {
|
|
136
|
+
if (value !== undefined) {
|
|
137
|
+
url.searchParams.set(key, String(value));
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const headers: Record<string, string> = { ...(options.headers ?? {}) };
|
|
142
|
+
|
|
143
|
+
if (apiKey) {
|
|
144
|
+
headers['Authorization'] = `Bearer ${apiKey}`;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
const response = await fetch(url.toString(), {
|
|
148
|
+
method: 'GET',
|
|
149
|
+
headers,
|
|
150
|
+
signal: options.signal,
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
if (!response.ok) {
|
|
154
|
+
const errorText = await response.text();
|
|
155
|
+
throw new Error(
|
|
156
|
+
`Scraper request failed (${response.status}): ${errorText}`
|
|
157
|
+
);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
if (options.format === 'json') {
|
|
161
|
+
return (await response.json()) as ScraperJsonResponse;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
return await response.text();
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Convenience function to render a URL and return just the HTML content.
|
|
169
|
+
*
|
|
170
|
+
* @param url - URL to render
|
|
171
|
+
* @param options - Additional scraping options
|
|
172
|
+
* @param config - Service configuration
|
|
173
|
+
* @returns Rendered HTML string
|
|
174
|
+
*/
|
|
175
|
+
export async function renderUrlToHtml(
|
|
176
|
+
url: string,
|
|
177
|
+
options: Omit<ScraperOptions, 'url'> = {},
|
|
178
|
+
config: Partial<ScraperConfig> = {}
|
|
179
|
+
): Promise<string> {
|
|
180
|
+
const result = await renderWithCloudflare(
|
|
181
|
+
{ ...options, url, format: 'html' },
|
|
182
|
+
config
|
|
183
|
+
);
|
|
184
|
+
return typeof result === 'string' ? result : result.html;
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* Renders a URL and returns full metadata including cookies, load time, etc.
|
|
189
|
+
*
|
|
190
|
+
* @param url - URL to render
|
|
191
|
+
* @param options - Additional scraping options
|
|
192
|
+
* @param config - Service configuration
|
|
193
|
+
* @returns Structured response with HTML and metadata
|
|
194
|
+
*/
|
|
195
|
+
export async function renderUrlWithMetadata(
|
|
196
|
+
url: string,
|
|
197
|
+
options: Omit<ScraperOptions, 'url' | 'format'> = {},
|
|
198
|
+
config: Partial<ScraperConfig> = {}
|
|
199
|
+
): Promise<ScraperJsonResponse> {
|
|
200
|
+
const result = await renderWithCloudflare(
|
|
201
|
+
{ ...options, url, format: 'json' },
|
|
202
|
+
config
|
|
203
|
+
);
|
|
204
|
+
return result as ScraperJsonResponse;
|
|
205
|
+
}
|