html-renderer-api 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/INTEGRATION.md +319 -0
- package/QUICKSTART.md +270 -0
- package/package.json +19 -0
- package/readme.md +317 -0
- package/src/config/setup.sh +60 -0
- package/src/durable-objects/browser-durable-object.ts +534 -0
- package/src/scrapers/scraper-captcha.ts +479 -0
- package/src/scrapers/scraper-cloudflare.ts +93 -0
- package/src/scrapers/scraper-login.ts +271 -0
- package/src/scrapers/scraper-openapi.ts +363 -0
- package/src/scrapers/scraper-stealth.ts +521 -0
- package/src/utils/scraper-utils.ts +219 -0
package/INTEGRATION.md
ADDED
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
# Cloudflare Puppeteer Scraper Integration
|
|
2
|
+
|
|
3
|
+
## Overview
|
|
4
|
+
|
|
5
|
+
The Cloudflare Puppeteer scraper is now **fully integrated** into qwksearch-web for backend rendering of JavaScript-heavy websites, bot detection bypass, and dynamic content extraction.
|
|
6
|
+
|
|
7
|
+
## Architecture
|
|
8
|
+
|
|
9
|
+
### 1. Scraper Service (Cloudflare Worker)
|
|
10
|
+
**Location**: `packages/render-url-to-html/scraper-cloudflare/`
|
|
11
|
+
|
|
12
|
+
A production-ready Cloudflare Worker that uses Puppeteer with Browser Rendering to:
|
|
13
|
+
- Render JavaScript-heavy pages
|
|
14
|
+
- Bypass Cloudflare challenges and CAPTCHAs
|
|
15
|
+
- Handle session management and cookie persistence
|
|
16
|
+
- Support proxy configuration
|
|
17
|
+
- Reuse browser instances for performance (5-minute persistence)
|
|
18
|
+
|
|
19
|
+
### 2. Client Library
|
|
20
|
+
**Location**: `apps/qwksearch-web/lib/scraper/`
|
|
21
|
+
|
|
22
|
+
- `cloudflare-scraper-client.ts` - TypeScript client for calling the scraper
|
|
23
|
+
- `use-scraper.ts` - React hooks for easy integration
|
|
24
|
+
- `index.ts` - Barrel exports
|
|
25
|
+
|
|
26
|
+
### 3. Next.js API Route
|
|
27
|
+
**Location**: `apps/qwksearch-web/app/api/scraper/route.ts`
|
|
28
|
+
|
|
29
|
+
Edge function that proxies requests to the scraper service:
|
|
30
|
+
- `POST /api/scraper` - Full rendering with JSON body
|
|
31
|
+
- `GET /api/scraper?url=...` - Quick rendering with query params
|
|
32
|
+
|
|
33
|
+
### 4. Integration into URL-to-Content Pipeline
|
|
34
|
+
**Location**: `packages/extract-webpage/src/url-to-content/url-to-html.ts`
|
|
35
|
+
|
|
36
|
+
The scraper is now integrated into the fallback chain:
|
|
37
|
+
1. **First**: Try basic fetch with `grab()`
|
|
38
|
+
2. **Second**: Try Cloudflare Puppeteer scraper (handles JS + bot detection)
|
|
39
|
+
3. **Third**: Try JINA reader (final fallback)
|
|
40
|
+
|
|
41
|
+
Bot detection triggers immediate escalation to Cloudflare scraper.
|
|
42
|
+
|
|
43
|
+
## Deployment Instructions
|
|
44
|
+
|
|
45
|
+
### Step 1: Deploy Cloudflare Worker
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
cd packages/render-url-to-html/scraper-cloudflare
|
|
49
|
+
|
|
50
|
+
# Install dependencies
|
|
51
|
+
npm install @cloudflare/puppeteer
|
|
52
|
+
|
|
53
|
+
# Deploy to Cloudflare
|
|
54
|
+
npx wrangler deploy
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
This will output a URL like: `https://scraper-cloudflare.your-subdomain.workers.dev`
|
|
58
|
+
|
|
59
|
+
### Step 2: Configure Environment Variables
|
|
60
|
+
|
|
61
|
+
Add to `apps/qwksearch-web/.env.local`:
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
# Cloudflare Browser Rendering (Scraper)
|
|
65
|
+
SCRAPER_URL=https://scraper-cloudflare.your-subdomain.workers.dev
|
|
66
|
+
SCRAPER_API_KEY=your-api-key-here # Optional - for authentication
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
### Step 3: Set Scraper API Key (Optional)
|
|
70
|
+
|
|
71
|
+
If you want to protect your scraper with an API key:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
cd packages/render-url-to-html/scraper-cloudflare
|
|
75
|
+
npx wrangler secret put SCRAPER_API_KEY
|
|
76
|
+
# Enter your secret key when prompted
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Then add the same key to your `.env.local` as shown above.
|
|
80
|
+
|
|
81
|
+
## Usage Examples
|
|
82
|
+
|
|
83
|
+
### From React Components
|
|
84
|
+
|
|
85
|
+
```tsx
|
|
86
|
+
import { useScraper } from '@/lib/scraper';
|
|
87
|
+
|
|
88
|
+
function MyComponent() {
|
|
89
|
+
const scraper = useScraper({
|
|
90
|
+
blockImages: true,
|
|
91
|
+
bypassCaptcha: true
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
const handleScrape = async () => {
|
|
95
|
+
await scraper.scrape('https://example.com');
|
|
96
|
+
console.log(scraper.data?.html);
|
|
97
|
+
console.log('Load time:', scraper.data?.loadTime, 'ms');
|
|
98
|
+
};
|
|
99
|
+
|
|
100
|
+
return (
|
|
101
|
+
<button onClick={handleScrape} disabled={scraper.isLoading}>
|
|
102
|
+
{scraper.isLoading ? 'Scraping...' : 'Scrape Page'}
|
|
103
|
+
</button>
|
|
104
|
+
);
|
|
105
|
+
}
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
### From API Routes
|
|
109
|
+
|
|
110
|
+
```typescript
|
|
111
|
+
import { renderUrlWithMetadata } from '@/lib/scraper';
|
|
112
|
+
|
|
113
|
+
const result = await renderUrlWithMetadata('https://example.com', {
|
|
114
|
+
blockImages: true,
|
|
115
|
+
bypassCaptcha: true
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
console.log(result.html);
|
|
119
|
+
console.log(result.title);
|
|
120
|
+
console.log(result.cookies);
|
|
121
|
+
console.log(result.loadTime);
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### Via Next.js API Endpoint
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
# POST request
|
|
128
|
+
curl -X POST http://localhost:3000/api/scraper \
|
|
129
|
+
-H "Content-Type: application/json" \
|
|
130
|
+
-d '{
|
|
131
|
+
"url": "https://example.com",
|
|
132
|
+
"blockImages": true,
|
|
133
|
+
"bypassCaptcha": true
|
|
134
|
+
}'
|
|
135
|
+
|
|
136
|
+
# GET request
|
|
137
|
+
curl "http://localhost:3000/api/scraper?url=https://example.com&blockImages=true"
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
## Features
|
|
141
|
+
|
|
142
|
+
### Core Features
|
|
143
|
+
- ✅ JavaScript rendering with Puppeteer
|
|
144
|
+
- ✅ Bot detection bypass (Cloudflare, Datadome, etc.)
|
|
145
|
+
- ✅ CAPTCHA solving support (2Captcha integration)
|
|
146
|
+
- ✅ Session management with cookie persistence
|
|
147
|
+
- ✅ Browser instance reuse (5-minute persistence)
|
|
148
|
+
- ✅ Resource blocking (images, CSS, fonts) for performance
|
|
149
|
+
- ✅ Proxy support (HTTP/HTTPS/SOCKS5)
|
|
150
|
+
- ✅ Custom headers and user agents
|
|
151
|
+
- ✅ Multiple wait strategies (domcontentloaded, networkidle, etc.)
|
|
152
|
+
|
|
153
|
+
### API Documentation
|
|
154
|
+
- ✅ Swagger UI at `/swagger`
|
|
155
|
+
- ✅ OpenAPI 3.0 spec at `/api/openapi.json`
|
|
156
|
+
- ✅ Authentication via Bearer token, query param, or POST body
|
|
157
|
+
|
|
158
|
+
### Response Formats
|
|
159
|
+
- **HTML**: Just the rendered HTML content
|
|
160
|
+
- **JSON**: Full metadata including cookies, load time, title, etc.
|
|
161
|
+
|
|
162
|
+
## Integration Points
|
|
163
|
+
|
|
164
|
+
### 1. Article Extraction
|
|
165
|
+
The scraper is automatically used when extracting articles that:
|
|
166
|
+
- Return bot detection errors
|
|
167
|
+
- Fail to load with basic fetch
|
|
168
|
+
- Require JavaScript rendering
|
|
169
|
+
|
|
170
|
+
### 2. Search Results
|
|
171
|
+
When search results point to JavaScript-heavy sites, the scraper ensures full content extraction.
|
|
172
|
+
|
|
173
|
+
### 3. Citation Links
|
|
174
|
+
When users click citation links in AI responses, the scraper ensures the article can be loaded even if bot-protected.
|
|
175
|
+
|
|
176
|
+
## Performance Optimization
|
|
177
|
+
|
|
178
|
+
### Resource Blocking
|
|
179
|
+
Block images for 2-3x faster loading:
|
|
180
|
+
|
|
181
|
+
```typescript
|
|
182
|
+
const result = await renderUrlWithMetadata(url, {
|
|
183
|
+
blockImages: true, // Saves ~60% load time
|
|
184
|
+
});
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
### Wait Strategies
|
|
188
|
+
Choose the right strategy for your use case:
|
|
189
|
+
|
|
190
|
+
- `domcontentloaded` - Fastest, for static content
|
|
191
|
+
- `load` - Standard, waits for all resources
|
|
192
|
+
- `networkidle2` - **Default**, balanced approach
|
|
193
|
+
- `networkidle0` - Slowest, waits for complete network silence
|
|
194
|
+
|
|
195
|
+
### Browser Reuse
|
|
196
|
+
Browsers stay alive for 5 minutes between requests, making subsequent requests much faster:
|
|
197
|
+
- First request: ~3-5 seconds
|
|
198
|
+
- Subsequent requests: ~1-2 seconds
|
|
199
|
+
|
|
200
|
+
## Monitoring & Debugging
|
|
201
|
+
|
|
202
|
+
### Response Headers
|
|
203
|
+
The scraper returns useful debug headers:
|
|
204
|
+
|
|
205
|
+
```
|
|
206
|
+
X-Load-Time: 2341 # Page load time in milliseconds
|
|
207
|
+
X-Session-Id: user123 # Session identifier used
|
|
208
|
+
X-Final-URL: https://... # Final URL after redirects
|
|
209
|
+
X-Browser-Reused: true # Whether browser was reused
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
### Logging
|
|
213
|
+
Check Cloudflare Worker logs to see:
|
|
214
|
+
- Browser launch/reuse events
|
|
215
|
+
- Session management activities
|
|
216
|
+
- Performance metrics
|
|
217
|
+
- Error details
|
|
218
|
+
|
|
219
|
+
View logs:
|
|
220
|
+
```bash
|
|
221
|
+
cd packages/render-url-to-html/scraper-cloudflare
|
|
222
|
+
npx wrangler tail
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
## Troubleshooting
|
|
226
|
+
|
|
227
|
+
### Issue: "Scraper request failed (401)"
|
|
228
|
+
**Solution**: Add SCRAPER_API_KEY to both:
|
|
229
|
+
1. Cloudflare Worker secrets: `wrangler secret put SCRAPER_API_KEY`
|
|
230
|
+
2. Local environment: Add to `.env.local`
|
|
231
|
+
|
|
232
|
+
### Issue: "Scraper request failed (500)"
|
|
233
|
+
**Solution**: Check Cloudflare Worker logs with `wrangler tail` to see the actual error.
|
|
234
|
+
|
|
235
|
+
### Issue: Bot detection still triggered
|
|
236
|
+
**Solution**: Enable captcha bypass:
|
|
237
|
+
```typescript
|
|
238
|
+
const result = await renderUrlWithMetadata(url, {
|
|
239
|
+
bypassCaptcha: true,
|
|
240
|
+
maxRetries: 10
|
|
241
|
+
});
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
### Issue: Slow performance
|
|
245
|
+
**Solution**: Enable resource blocking:
|
|
246
|
+
```typescript
|
|
247
|
+
const result = await renderUrlWithMetadata(url, {
|
|
248
|
+
blockImages: true, // Blocks images, CSS, fonts
|
|
249
|
+
waitUntil: 'domcontentloaded' // Don't wait for all resources
|
|
250
|
+
});
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
## Cost Considerations
|
|
254
|
+
|
|
255
|
+
### Cloudflare Browser Rendering Pricing
|
|
256
|
+
- **Free Tier**: 1,000 browser rendering hours/month
|
|
257
|
+
- **Paid**: $5 per million Browser Rendering requests
|
|
258
|
+
- **Durable Objects**: $0.15 per million requests + $0.02 per GB-hour
|
|
259
|
+
|
|
260
|
+
### Optimization Tips
|
|
261
|
+
1. Use browser reuse (already implemented)
|
|
262
|
+
2. Block images when content is the priority
|
|
263
|
+
3. Set appropriate timeouts
|
|
264
|
+
4. Use session management to avoid re-authentication
|
|
265
|
+
|
|
266
|
+
## Security Features
|
|
267
|
+
|
|
268
|
+
### Authentication
|
|
269
|
+
Three methods supported:
|
|
270
|
+
1. Bearer token: `Authorization: Bearer YOUR_KEY`
|
|
271
|
+
2. Query parameter: `?api_key=YOUR_KEY`
|
|
272
|
+
3. POST body: `{"api_key": "YOUR_KEY"}`
|
|
273
|
+
|
|
274
|
+
### Input Validation
|
|
275
|
+
- URL format validation
|
|
276
|
+
- Parameter type checking
|
|
277
|
+
- Timeout and dimension limits
|
|
278
|
+
- XSS prevention in responses
|
|
279
|
+
|
|
280
|
+
### Session Isolation
|
|
281
|
+
- Separate cookie storage per session ID
|
|
282
|
+
- No cross-session data leakage
|
|
283
|
+
- Automatic session cleanup
|
|
284
|
+
|
|
285
|
+
## Next Steps
|
|
286
|
+
|
|
287
|
+
1. **Deploy the scraper** to Cloudflare Workers
|
|
288
|
+
2. **Configure environment variables** in qwksearch-web
|
|
289
|
+
3. **Test the integration** by visiting a JavaScript-heavy site
|
|
290
|
+
4. **Monitor performance** using Cloudflare dashboard
|
|
291
|
+
5. **Adjust settings** based on your needs (blocking, timeouts, etc.)
|
|
292
|
+
|
|
293
|
+
## Files Modified
|
|
294
|
+
|
|
295
|
+
### New/Updated Files
|
|
296
|
+
1. `packages/extract-webpage/src/url-to-content/url-to-html.ts` - Added Cloudflare scraper to fallback chain
|
|
297
|
+
2. `apps/qwksearch-web/lib/scraper/cloudflare-scraper-client.ts` - Client library (already existed)
|
|
298
|
+
3. `apps/qwksearch-web/lib/scraper/use-scraper.ts` - React hooks (already existed)
|
|
299
|
+
4. `apps/qwksearch-web/app/api/scraper/route.ts` - API endpoint (already existed)
|
|
300
|
+
5. `.env.example` - Updated with scraper configuration (already had it)
|
|
301
|
+
|
|
302
|
+
### Deployment Files
|
|
303
|
+
All files in `packages/render-url-to-html/scraper-cloudflare/`:
|
|
304
|
+
- `scraper-cloudflare.ts` - Main worker entry point
|
|
305
|
+
- `browser-durable-object.ts` - Durable Object for browser management
|
|
306
|
+
- `scraper-utils.ts` - Utility functions
|
|
307
|
+
- `scraper-openapi.ts` - API documentation
|
|
308
|
+
- `scraper-captcha.ts` - CAPTCHA solving
|
|
309
|
+
- `scraper-stealth.ts` - Bot detection bypass
|
|
310
|
+
- `wrangler.toml` - Cloudflare Worker configuration
|
|
311
|
+
- `package.json` - Dependencies
|
|
312
|
+
|
|
313
|
+
## Support
|
|
314
|
+
|
|
315
|
+
For issues or questions:
|
|
316
|
+
1. Check Cloudflare Worker logs: `npx wrangler tail`
|
|
317
|
+
2. Review the readme: `packages/render-url-to-html/scraper-cloudflare/readme.md`
|
|
318
|
+
3. Check browser console for client-side errors
|
|
319
|
+
4. Verify environment variables are set correctly
|
package/QUICKSTART.md
ADDED
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
# Cloudflare Scraper Quick Start
|
|
2
|
+
|
|
3
|
+
This guide will help you deploy and use the Cloudflare Puppeteer scraper service.
|
|
4
|
+
|
|
5
|
+
## Prerequisites
|
|
6
|
+
|
|
7
|
+
1. **Cloudflare Account**: Workers Paid plan required for Browser Rendering
|
|
8
|
+
2. **Wrangler CLI**: Install with `npm install -g wrangler`
|
|
9
|
+
3. **Authentication**: Run `wrangler login`
|
|
10
|
+
|
|
11
|
+
## Deployment
|
|
12
|
+
|
|
13
|
+
### 1. Navigate to the scraper directory
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
cd packages/render-url-to-html/scraper-cloudflare
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
### 2. Install dependencies
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
bun install
|
|
23
|
+
# or
|
|
24
|
+
npm install
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
### 3. Configure secrets (optional but recommended)
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
# Set API key for authentication
|
|
31
|
+
wrangler secret put SCRAPER_API_KEY
|
|
32
|
+
# Enter your API key when prompted
|
|
33
|
+
|
|
34
|
+
# Optional: Configure proxy
|
|
35
|
+
wrangler secret put PROXY_URL
|
|
36
|
+
wrangler secret put PROXY_USER
|
|
37
|
+
wrangler secret put PROXY_PASS
|
|
38
|
+
|
|
39
|
+
# Optional: 2Captcha for automated CAPTCHA solving
|
|
40
|
+
wrangler secret put TWO_CAPTCHA_KEY
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
### 4. Deploy to Cloudflare
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
wrangler deploy
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
This will output your Worker URL, e.g., `https://scraper-cloudflare.your-subdomain.workers.dev`
|
|
50
|
+
|
|
51
|
+
### 5. Test the deployment
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
# View Swagger UI
|
|
55
|
+
curl https://scraper-cloudflare.your-subdomain.workers.dev/swagger
|
|
56
|
+
|
|
57
|
+
# Test rendering (replace YOUR_API_KEY if you set one)
|
|
58
|
+
curl -X POST https://scraper-cloudflare.your-subdomain.workers.dev/api/render \
|
|
59
|
+
-H "Content-Type: application/json" \
|
|
60
|
+
-H "Authorization: Bearer YOUR_API_KEY" \
|
|
61
|
+
-d '{
|
|
62
|
+
"url": "https://example.com",
|
|
63
|
+
"format": "json"
|
|
64
|
+
}'
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Using from qwksearch-web
|
|
68
|
+
|
|
69
|
+
### 1. Set environment variables
|
|
70
|
+
|
|
71
|
+
Add to your `.env` or `.env.local`:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
SCRAPER_URL=https://scraper-cloudflare.your-subdomain.workers.dev
|
|
75
|
+
SCRAPER_API_KEY=your-api-key-here
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### 2. Use in code
|
|
79
|
+
|
|
80
|
+
```typescript
|
|
81
|
+
import { renderUrlToHtml } from '@/lib/scraper';
|
|
82
|
+
|
|
83
|
+
// Simple usage
|
|
84
|
+
const html = await renderUrlToHtml('https://example.com');
|
|
85
|
+
|
|
86
|
+
// With options
|
|
87
|
+
const html = await renderUrlToHtml('https://spa-site.com', {
|
|
88
|
+
blockImages: true,
|
|
89
|
+
bypassCaptcha: true,
|
|
90
|
+
timeout: 45000
|
|
91
|
+
});
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
### 3. Use from AI agent
|
|
95
|
+
|
|
96
|
+
The AI agent can automatically use the `render_page_with_javascript` tool:
|
|
97
|
+
|
|
98
|
+
```
|
|
99
|
+
User: "Fetch the content from https://js-heavy-site.com"
|
|
100
|
+
|
|
101
|
+
Agent: [Uses render_page_with_javascript tool automatically]
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## Configuration Options
|
|
105
|
+
|
|
106
|
+
### Request Options
|
|
107
|
+
|
|
108
|
+
| Option | Type | Default | Description |
|
|
109
|
+
|--------|------|---------|-------------|
|
|
110
|
+
| `url` | string | **required** | URL to render |
|
|
111
|
+
| `blockImages` | boolean | `false` | Block image loading for faster rendering |
|
|
112
|
+
| `wait` | number | `0` | Additional wait time after page load (ms) |
|
|
113
|
+
| `timeout` | number | `30000` | Navigation timeout (ms) |
|
|
114
|
+
| `waitUntil` | string | `"networkidle2"` | Puppeteer waitUntil condition |
|
|
115
|
+
| `bypassCaptcha` | boolean | `true` | Attempt to bypass challenges |
|
|
116
|
+
| `sessionId` | string | `"default"` | Session ID for cookie persistence |
|
|
117
|
+
| `format` | string | `"html"` | Response format (`html` or `json`) |
|
|
118
|
+
| `maxRetries` | number | `10` | Max challenge bypass attempts |
|
|
119
|
+
|
|
120
|
+
### Wait Strategies
|
|
121
|
+
|
|
122
|
+
- **`domcontentloaded`**: Fastest, waits for initial HTML
|
|
123
|
+
- **`load`**: Waits for all resources (images, stylesheets)
|
|
124
|
+
- **`networkidle2`**: Balanced, waits for network to be mostly idle
|
|
125
|
+
- **`networkidle0`**: Slowest, waits for complete network silence
|
|
126
|
+
|
|
127
|
+
## Common Use Cases
|
|
128
|
+
|
|
129
|
+
### JavaScript-Heavy Site
|
|
130
|
+
|
|
131
|
+
```typescript
|
|
132
|
+
const html = await renderUrlToHtml('https://react-app.com', {
|
|
133
|
+
waitUntil: 'networkidle2',
|
|
134
|
+
blockImages: true
|
|
135
|
+
});
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
### Behind Cloudflare Protection
|
|
139
|
+
|
|
140
|
+
```typescript
|
|
141
|
+
const result = await renderUrlWithMetadata('https://protected-site.com', {
|
|
142
|
+
bypassCaptcha: true,
|
|
143
|
+
maxRetries: 15
|
|
144
|
+
});
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
### Session-Based Scraping
|
|
148
|
+
|
|
149
|
+
```typescript
|
|
150
|
+
// Login
|
|
151
|
+
await renderUrlToHtml('https://site.com/login', {
|
|
152
|
+
sessionId: 'user-123',
|
|
153
|
+
wait: 3000
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
// Access protected page
|
|
157
|
+
const html = await renderUrlToHtml('https://site.com/protected', {
|
|
158
|
+
sessionId: 'user-123'
|
|
159
|
+
});
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
### With Custom Headers
|
|
163
|
+
|
|
164
|
+
```typescript
|
|
165
|
+
const result = await renderWithCloudflare({
|
|
166
|
+
url: 'https://api-site.com',
|
|
167
|
+
headers: {
|
|
168
|
+
'X-API-Key': 'your-key',
|
|
169
|
+
'User-Agent': 'Custom Bot 1.0'
|
|
170
|
+
},
|
|
171
|
+
format: 'json'
|
|
172
|
+
});
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
## Monitoring
|
|
176
|
+
|
|
177
|
+
View Worker logs:
|
|
178
|
+
|
|
179
|
+
```bash
|
|
180
|
+
wrangler tail
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
Check metrics in the Cloudflare dashboard:
|
|
184
|
+
|
|
185
|
+
1. Go to Workers & Pages
|
|
186
|
+
2. Select your `scraper-cloudflare` worker
|
|
187
|
+
3. View Metrics tab
|
|
188
|
+
|
|
189
|
+
## Troubleshooting
|
|
190
|
+
|
|
191
|
+
### Issue: "Browser timeout"
|
|
192
|
+
|
|
193
|
+
**Solution**: Increase timeout or use faster wait strategy:
|
|
194
|
+
|
|
195
|
+
```typescript
|
|
196
|
+
await renderUrlToHtml(url, {
|
|
197
|
+
timeout: 60000,
|
|
198
|
+
waitUntil: 'domcontentloaded'
|
|
199
|
+
});
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
### Issue: "Invalid API key"
|
|
203
|
+
|
|
204
|
+
**Solution**: Verify environment variable is set:
|
|
205
|
+
|
|
206
|
+
```bash
|
|
207
|
+
wrangler secret list
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
### Issue: "Challenge not bypassed"
|
|
211
|
+
|
|
212
|
+
**Solution**:
|
|
213
|
+
1. Ensure `bypassCaptcha: true`
|
|
214
|
+
2. Set `TWO_CAPTCHA_KEY` for automated solving
|
|
215
|
+
3. Increase `maxRetries`
|
|
216
|
+
|
|
217
|
+
### Issue: High costs
|
|
218
|
+
|
|
219
|
+
**Solution**:
|
|
220
|
+
1. Use `blockImages: true`
|
|
221
|
+
2. Lower `timeout` values
|
|
222
|
+
3. Use faster wait strategies
|
|
223
|
+
4. Implement caching in your application
|
|
224
|
+
5. Use `extract_page` tool when JavaScript isn't needed
|
|
225
|
+
|
|
226
|
+
## Cost Optimization
|
|
227
|
+
|
|
228
|
+
Cloudflare Browser Rendering is billed per request (~$0.50 per 1,000 requests).
|
|
229
|
+
|
|
230
|
+
To minimize costs:
|
|
231
|
+
|
|
232
|
+
1. **Cache rendered pages** at the application level
|
|
233
|
+
2. **Block unnecessary resources**: `blockImages: true`
|
|
234
|
+
3. **Use appropriate timeouts**: Don't wait longer than needed
|
|
235
|
+
4. **Choose the right wait strategy**: `domcontentloaded` is fastest
|
|
236
|
+
5. **Fallback to extract_page**: Use for server-rendered sites
|
|
237
|
+
|
|
238
|
+
Example caching implementation:
|
|
239
|
+
|
|
240
|
+
```typescript
|
|
241
|
+
const cache = new Map<string, { html: string; timestamp: number }>();
|
|
242
|
+
const CACHE_TTL = 5 * 60 * 1000; // 5 minutes
|
|
243
|
+
|
|
244
|
+
async function getCachedOrRender(url: string): Promise<string> {
|
|
245
|
+
const cached = cache.get(url);
|
|
246
|
+
|
|
247
|
+
if (cached && Date.now() - cached.timestamp < CACHE_TTL) {
|
|
248
|
+
return cached.html;
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
const html = await renderUrlToHtml(url);
|
|
252
|
+
cache.set(url, { html, timestamp: Date.now() });
|
|
253
|
+
|
|
254
|
+
return html;
|
|
255
|
+
}
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
## Next Steps
|
|
259
|
+
|
|
260
|
+
- Read the [full integration documentation](../../../docs/SCRAPER_CLOUDFLARE_INTEGRATION.md)
|
|
261
|
+
- View [API reference](readme.md)
|
|
262
|
+
- Check out [example code](../../../apps/qwksearch-web/lib/scraper/cloudflare-scraper-client.ts)
|
|
263
|
+
|
|
264
|
+
## Support
|
|
265
|
+
|
|
266
|
+
For issues or questions:
|
|
267
|
+
|
|
268
|
+
1. Check Worker logs: `wrangler tail`
|
|
269
|
+
2. Review [Cloudflare Browser Rendering docs](https://developers.cloudflare.com/browser-rendering/)
|
|
270
|
+
3. Open an issue on GitHub
|
package/package.json
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "html-renderer-api",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "Server API renders DOM with puppeteer to get HTML & Bypass Cloudflare bot check.",
|
|
5
|
+
"main": "src/scrapers/scraper-cloudflare.js",
|
|
6
|
+
"type": "module",
|
|
7
|
+
"license": "rights.institute/PROSPER",
|
|
8
|
+
"repository": {
|
|
9
|
+
"type": "git",
|
|
10
|
+
"url": "https://github.com/OpenSourceAGI/qwksearch-research-agent"
|
|
11
|
+
},
|
|
12
|
+
"scripts": {
|
|
13
|
+
"start": "bun src/scrapers/scraper-cloudflare.js",
|
|
14
|
+
"deploy": "wrangler deploy"
|
|
15
|
+
},
|
|
16
|
+
"dependencies": {
|
|
17
|
+
"@cloudflare/puppeteer": "*"
|
|
18
|
+
}
|
|
19
|
+
}
|