extract-webpage 1.2.116 → 1.2.117

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -2,11 +2,8 @@
2
2
  <br />
3
3
  <a href="https://www.npmjs.com/package/extract-webpage"><img src="https://img.shields.io/npm/dm/extract-webpage.svg" alt="NPM Monthly Downloads"></a>
4
4
  <a href="https://www.npmjs.com/package/extract-webpage"><img src="https://img.shields.io/npm/v/extract-webpage.svg" alt="npm version"></a>
5
- <a href="https://bundlephobia.com/package/extract-webpage" target="_blank" rel="noopener noreferrer">
6
- <img
7
- src="https://img.shields.io/bundlephobia/minzip/extract-webpage?style=flat&label=size"
8
- alt="npm bundle size "
9
- />
5
+ <a href="https://packagephobia.com/result?p=extract-webpage" target="_blank" rel="noopener noreferrer">
6
+ <img src="https://packagephobia.com/badge?p=extract-webpage" alt="install size" />
10
7
  </a>
11
8
  <a href="https://discord.gg/SJdBqBz3tV">
12
9
  <img src="https://img.shields.io/discord/1110227955554209923.svg?label=Chat&logo=Discord&colorB=7289da&style=flat"
@@ -0,0 +1,112 @@
1
+ /**
2
+ * @fileoverview Client for the Cloudflare Puppeteer-based scraper service.
3
+ * Renders JavaScript-heavy pages and bypasses bot detection.
4
+ *
5
+ * This wraps the scraper-cloudflare package deployed as a Cloudflare Worker
6
+ * with Browser Rendering (Puppeteer) support.
7
+ */
8
+ export interface ScraperOptions {
9
+ /** URL to render */
10
+ url: string;
11
+ /** API key for authentication (optional if SCRAPER_API_KEY not set) */
12
+ apiKey?: string;
13
+ /** Additional wait time after page load (ms) */
14
+ wait?: number;
15
+ /** Block image loading for faster rendering */
16
+ blockImages?: boolean;
17
+ /** Session ID for cookie persistence */
18
+ sessionId?: string;
19
+ /** Navigation timeout (ms) */
20
+ timeout?: number;
21
+ /** Puppeteer waitUntil condition */
22
+ waitUntil?: 'domcontentloaded' | 'load' | 'networkidle0' | 'networkidle2';
23
+ /** Response format */
24
+ format?: 'html' | 'json';
25
+ /** Custom headers */
26
+ headers?: Record<string, string>;
27
+ /** Proxy configuration */
28
+ proxyUrl?: string;
29
+ proxyUser?: string;
30
+ proxyPass?: string;
31
+ /** Bypass Cloudflare challenges and CAPTCHAs */
32
+ bypassCaptcha?: boolean;
33
+ /** Challenge detection pattern */
34
+ challengeMatch?: string;
35
+ /** Max retry attempts for challenges */
36
+ maxRetries?: number;
37
+ /** 2Captcha API key for solving */
38
+ twoCaptchaKey?: string;
39
+ /** Abort signal to bound the request (e.g. an 8s deadline). */
40
+ signal?: AbortSignal;
41
+ }
42
+ export interface ScraperJsonResponse {
43
+ html: string;
44
+ url: string;
45
+ title: string;
46
+ cookies: Array<{
47
+ name: string;
48
+ value: string;
49
+ domain: string;
50
+ path: string;
51
+ expires?: number;
52
+ httpOnly?: boolean;
53
+ secure?: boolean;
54
+ sameSite?: 'Strict' | 'Lax' | 'None';
55
+ }>;
56
+ challengeBypassed: boolean;
57
+ retryCount: number;
58
+ loadTime: number;
59
+ }
60
+ export interface ScraperConfig {
61
+ /** Base URL of the scraper service */
62
+ baseURL: string;
63
+ /** Global API key */
64
+ apiKey?: string;
65
+ }
66
+ /**
67
+ * Renders a URL using the Cloudflare Puppeteer scraper service.
68
+ * Supports JavaScript rendering, bot detection bypass, and session management.
69
+ *
70
+ * @param options - Scraping configuration
71
+ * @param config - Service configuration (base URL and API key)
72
+ * @returns Rendered HTML or structured JSON response
73
+ *
74
+ * @example
75
+ * ```ts
76
+ * // Basic usage
77
+ * const html = await renderWithCloudflare({ url: 'https://example.com' });
78
+ *
79
+ * // With challenge bypass
80
+ * const result = await renderWithCloudflare({
81
+ * url: 'https://protected-site.com',
82
+ * bypassCaptcha: true,
83
+ * format: 'json'
84
+ * });
85
+ *
86
+ * // With session management
87
+ * const html = await renderWithCloudflare({
88
+ * url: 'https://site-requiring-login.com',
89
+ * sessionId: 'user-123',
90
+ * blockImages: true
91
+ * });
92
+ * ```
93
+ */
94
+ export declare function renderWithCloudflare(options: ScraperOptions, config?: Partial<ScraperConfig>): Promise<string | ScraperJsonResponse>;
95
+ /**
96
+ * Convenience function to render a URL and return just the HTML content.
97
+ *
98
+ * @param url - URL to render
99
+ * @param options - Additional scraping options
100
+ * @param config - Service configuration
101
+ * @returns Rendered HTML string
102
+ */
103
+ export declare function renderUrlToHtml(url: string, options?: Omit<ScraperOptions, 'url'>, config?: Partial<ScraperConfig>): Promise<string>;
104
+ /**
105
+ * Renders a URL and returns full metadata including cookies, load time, etc.
106
+ *
107
+ * @param url - URL to render
108
+ * @param options - Additional scraping options
109
+ * @param config - Service configuration
110
+ * @returns Structured response with HTML and metadata
111
+ */
112
+ export declare function renderUrlWithMetadata(url: string, options?: Omit<ScraperOptions, 'url' | 'format'>, config?: Partial<ScraperConfig>): Promise<ScraperJsonResponse>;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "extract-webpage",
3
- "version": "1.2.116",
3
+ "version": "1.2.117",
4
4
  "module": "./dist/extract-webpage.es.js",
5
5
  "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
6
  "author": "vtempest <grokthiscontact@gmail.com>",
@@ -72,11 +72,11 @@
72
72
  "dependencies": {
73
73
  "@huggingface/transformers": "^3.8.1",
74
74
  "ai": "^5.0.0",
75
- "chat-agent-toolkit": "^1.2.116",
75
+ "chat-agent-toolkit": "^1.2.117",
76
76
  "chrono-node": "^2.9.0",
77
77
  "drizzle-orm": "^0.45.1",
78
- "extract-pdf": "^0.1.103",
79
- "extract-youtube": "^1.0.99",
78
+ "extract-pdf": "^0.1.104",
79
+ "extract-youtube": "^1.0.102",
80
80
  "html-entities": "^2.6.0",
81
81
  "js-yaml": "^4.1.1",
82
82
  "jsdom": "^28.1.0",
@@ -0,0 +1,205 @@
1
+ /**
2
+ * @fileoverview Client for the Cloudflare Puppeteer-based scraper service.
3
+ * Renders JavaScript-heavy pages and bypasses bot detection.
4
+ *
5
+ * This wraps the scraper-cloudflare package deployed as a Cloudflare Worker
6
+ * with Browser Rendering (Puppeteer) support.
7
+ */
8
+
9
+ export interface ScraperOptions {
10
+ /** URL to render */
11
+ url: string;
12
+ /** API key for authentication (optional if SCRAPER_API_KEY not set) */
13
+ apiKey?: string;
14
+ /** Additional wait time after page load (ms) */
15
+ wait?: number;
16
+ /** Block image loading for faster rendering */
17
+ blockImages?: boolean;
18
+ /** Session ID for cookie persistence */
19
+ sessionId?: string;
20
+ /** Navigation timeout (ms) */
21
+ timeout?: number;
22
+ /** Puppeteer waitUntil condition */
23
+ waitUntil?: 'domcontentloaded' | 'load' | 'networkidle0' | 'networkidle2';
24
+ /** Response format */
25
+ format?: 'html' | 'json';
26
+ /** Custom headers */
27
+ headers?: Record<string, string>;
28
+ /** Proxy configuration */
29
+ proxyUrl?: string;
30
+ proxyUser?: string;
31
+ proxyPass?: string;
32
+ /** Bypass Cloudflare challenges and CAPTCHAs */
33
+ bypassCaptcha?: boolean;
34
+ /** Challenge detection pattern */
35
+ challengeMatch?: string;
36
+ /** Max retry attempts for challenges */
37
+ maxRetries?: number;
38
+ /** 2Captcha API key for solving */
39
+ twoCaptchaKey?: string;
40
+ /** Abort signal to bound the request (e.g. an 8s deadline). */
41
+ signal?: AbortSignal;
42
+ }
43
+
44
+ export interface ScraperJsonResponse {
45
+ html: string;
46
+ url: string;
47
+ title: string;
48
+ cookies: Array<{
49
+ name: string;
50
+ value: string;
51
+ domain: string;
52
+ path: string;
53
+ expires?: number;
54
+ httpOnly?: boolean;
55
+ secure?: boolean;
56
+ sameSite?: 'Strict' | 'Lax' | 'None';
57
+ }>;
58
+ challengeBypassed: boolean;
59
+ retryCount: number;
60
+ loadTime: number;
61
+ }
62
+
63
+ export interface ScraperConfig {
64
+ /** Base URL of the scraper service */
65
+ baseURL: string;
66
+ /** Global API key */
67
+ apiKey?: string;
68
+ }
69
+
70
+ const DEFAULT_CONFIG: ScraperConfig = {
71
+ baseURL: typeof process !== 'undefined' && process?.env?.SCRAPER_URL
72
+ ? process.env.SCRAPER_URL
73
+ : 'https://proxy.qwksearch.com',
74
+ apiKey: typeof process !== 'undefined' && process?.env?.SCRAPER_API_KEY
75
+ ? process.env.SCRAPER_API_KEY
76
+ : undefined,
77
+ };
78
+
79
+ /**
80
+ * Renders a URL using the Cloudflare Puppeteer scraper service.
81
+ * Supports JavaScript rendering, bot detection bypass, and session management.
82
+ *
83
+ * @param options - Scraping configuration
84
+ * @param config - Service configuration (base URL and API key)
85
+ * @returns Rendered HTML or structured JSON response
86
+ *
87
+ * @example
88
+ * ```ts
89
+ * // Basic usage
90
+ * const html = await renderWithCloudflare({ url: 'https://example.com' });
91
+ *
92
+ * // With challenge bypass
93
+ * const result = await renderWithCloudflare({
94
+ * url: 'https://protected-site.com',
95
+ * bypassCaptcha: true,
96
+ * format: 'json'
97
+ * });
98
+ *
99
+ * // With session management
100
+ * const html = await renderWithCloudflare({
101
+ * url: 'https://site-requiring-login.com',
102
+ * sessionId: 'user-123',
103
+ * blockImages: true
104
+ * });
105
+ * ```
106
+ */
107
+ export async function renderWithCloudflare(
108
+ options: ScraperOptions,
109
+ config: Partial<ScraperConfig> = {}
110
+ ): Promise<string | ScraperJsonResponse> {
111
+ const mergedConfig = { ...DEFAULT_CONFIG, ...config };
112
+ const apiKey = options.apiKey || mergedConfig.apiKey;
113
+
114
+ // The deployed scraper worker (proxy.qwksearch.com) accepts GET requests
115
+ // with query-string parameters on `/` and `/api/render`.
116
+ const url = new URL('/api/render', mergedConfig.baseURL);
117
+
118
+ const params: Record<string, string | number | boolean | undefined> = {
119
+ url: options.url,
120
+ wait: options.wait ?? 0,
121
+ blockImages: options.blockImages ?? false,
122
+ sessionId: options.sessionId ?? 'default',
123
+ timeout: options.timeout ?? 30000,
124
+ waitUntil: options.waitUntil ?? 'networkidle2',
125
+ format: options.format ?? 'html',
126
+ proxyUrl: options.proxyUrl,
127
+ proxyUser: options.proxyUser,
128
+ proxyPass: options.proxyPass,
129
+ bypassCaptcha: options.bypassCaptcha ?? true,
130
+ challengeMatch: options.challengeMatch,
131
+ maxRetries: options.maxRetries ?? 10,
132
+ twoCaptchaKey: options.twoCaptchaKey,
133
+ };
134
+
135
+ for (const [key, value] of Object.entries(params)) {
136
+ if (value !== undefined) {
137
+ url.searchParams.set(key, String(value));
138
+ }
139
+ }
140
+
141
+ const headers: Record<string, string> = { ...(options.headers ?? {}) };
142
+
143
+ if (apiKey) {
144
+ headers['Authorization'] = `Bearer ${apiKey}`;
145
+ }
146
+
147
+ const response = await fetch(url.toString(), {
148
+ method: 'GET',
149
+ headers,
150
+ signal: options.signal,
151
+ });
152
+
153
+ if (!response.ok) {
154
+ const errorText = await response.text();
155
+ throw new Error(
156
+ `Scraper request failed (${response.status}): ${errorText}`
157
+ );
158
+ }
159
+
160
+ if (options.format === 'json') {
161
+ return (await response.json()) as ScraperJsonResponse;
162
+ }
163
+
164
+ return await response.text();
165
+ }
166
+
167
+ /**
168
+ * Convenience function to render a URL and return just the HTML content.
169
+ *
170
+ * @param url - URL to render
171
+ * @param options - Additional scraping options
172
+ * @param config - Service configuration
173
+ * @returns Rendered HTML string
174
+ */
175
+ export async function renderUrlToHtml(
176
+ url: string,
177
+ options: Omit<ScraperOptions, 'url'> = {},
178
+ config: Partial<ScraperConfig> = {}
179
+ ): Promise<string> {
180
+ const result = await renderWithCloudflare(
181
+ { ...options, url, format: 'html' },
182
+ config
183
+ );
184
+ return typeof result === 'string' ? result : result.html;
185
+ }
186
+
187
+ /**
188
+ * Renders a URL and returns full metadata including cookies, load time, etc.
189
+ *
190
+ * @param url - URL to render
191
+ * @param options - Additional scraping options
192
+ * @param config - Service configuration
193
+ * @returns Structured response with HTML and metadata
194
+ */
195
+ export async function renderUrlWithMetadata(
196
+ url: string,
197
+ options: Omit<ScraperOptions, 'url' | 'format'> = {},
198
+ config: Partial<ScraperConfig> = {}
199
+ ): Promise<ScraperJsonResponse> {
200
+ const result = await renderWithCloudflare(
201
+ { ...options, url, format: 'json' },
202
+ config
203
+ );
204
+ return result as ScraperJsonResponse;
205
+ }