webcrawlerapi-langchain 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- webcrawlerapi_langchain/__init__.py +4 -0
- webcrawlerapi_langchain/loader.py +310 -0
- webcrawlerapi_langchain-0.1.0.dist-info/METADATA +104 -0
- webcrawlerapi_langchain-0.1.0.dist-info/RECORD +6 -0
- webcrawlerapi_langchain-0.1.0.dist-info/WHEEL +5 -0
- webcrawlerapi_langchain-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
from typing import Dict, Iterator, List, Optional, AsyncIterator, Any, Literal
|
|
2
|
+
import os
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
from langchain_core.documents import Document
|
|
6
|
+
from langchain_core.utils import get_from_env
|
|
7
|
+
from langchain_core.document_loaders import BaseLoader
|
|
8
|
+
from webcrawlerapi import WebCrawlerAPI
|
|
9
|
+
|
|
10
|
+
# Configure logging
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
logger.setLevel(logging.DEBUG)
|
|
13
|
+
|
|
14
|
+
class WebCrawlerAPILoaderError(Exception):
|
|
15
|
+
"""Custom exception class for WebCrawlerAPILoader errors."""
|
|
16
|
+
pass
|
|
17
|
+
|
|
18
|
+
class WebCrawlerAPILoader(BaseLoader):
|
|
19
|
+
"""WebCrawlerAPI document loader integration.
|
|
20
|
+
|
|
21
|
+
This loader uses WebCrawlerAPI to crawl websites and convert them into LangChain Documents.
|
|
22
|
+
Each crawled page becomes a Document with its content and metadata.
|
|
23
|
+
|
|
24
|
+
Raises:
|
|
25
|
+
WebCrawlerAPILoaderError: If there are issues with initialization or crawling
|
|
26
|
+
ValueError: If required parameters are invalid
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
url: str,
|
|
32
|
+
*,
|
|
33
|
+
api_key: Optional[str] = None,
|
|
34
|
+
base_url: Optional[str] = None,
|
|
35
|
+
version: str = "v1",
|
|
36
|
+
scrape_type: Literal["html", "cleaned", "markdown"] = "markdown",
|
|
37
|
+
items_limit: int = 10,
|
|
38
|
+
allow_subdomains: bool = False,
|
|
39
|
+
whitelist_regexp: Optional[str] = None,
|
|
40
|
+
blacklist_regexp: Optional[str] = None,
|
|
41
|
+
max_polls: int = 100
|
|
42
|
+
):
|
|
43
|
+
"""Initialize the WebCrawlerAPI document loader.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
url: The URL to crawl
|
|
47
|
+
api_key: Your WebCrawlerAPI API key. If not provided, will try to get from WEBCRAWLERAPI_API_KEY env var
|
|
48
|
+
base_url: The base URL of the API. If not provided, will try to get from WEBCRAWLERAPI_BASE_URL env var
|
|
49
|
+
version: API version to use (optional)
|
|
50
|
+
scrape_type: Type of scraping (html, cleaned, markdown)
|
|
51
|
+
items_limit: Maximum number of pages to crawl
|
|
52
|
+
allow_subdomains: Whether to crawl subdomains
|
|
53
|
+
whitelist_regexp: Regex pattern for URL whitelist
|
|
54
|
+
blacklist_regexp: Regex pattern for URL blacklist
|
|
55
|
+
max_polls: Maximum number of status checks before returning
|
|
56
|
+
"""
|
|
57
|
+
if not url:
|
|
58
|
+
raise ValueError("URL must be provided")
|
|
59
|
+
|
|
60
|
+
if not url.startswith(('http://', 'https://')):
|
|
61
|
+
raise ValueError("URL must start with http:// or https://")
|
|
62
|
+
|
|
63
|
+
if items_limit < 1:
|
|
64
|
+
raise ValueError("items_limit must be greater than 0")
|
|
65
|
+
|
|
66
|
+
if max_polls < 1:
|
|
67
|
+
raise ValueError("max_polls must be greater than 0")
|
|
68
|
+
|
|
69
|
+
if scrape_type not in ("html", "cleaned", "markdown"):
|
|
70
|
+
raise ValueError(
|
|
71
|
+
f"Invalid scrape_type '{scrape_type}'. "
|
|
72
|
+
"Allowed: 'html', 'cleaned', 'markdown'."
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
# Get API key and base URL from env vars if not provided
|
|
76
|
+
self.api_key = api_key or get_from_env("api_key", "WEBCRAWLERAPI_API_KEY")
|
|
77
|
+
if not self.api_key:
|
|
78
|
+
raise ValueError("API key must be provided either through api_key parameter or WEBCRAWLERAPI_API_KEY environment variable")
|
|
79
|
+
|
|
80
|
+
self.base_url = base_url or get_from_env(
|
|
81
|
+
"base_url",
|
|
82
|
+
"WEBCRAWLERAPI_BASE_URL",
|
|
83
|
+
default="https://api.webcrawlerapi.com"
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
self.url = url
|
|
87
|
+
self.version = version
|
|
88
|
+
self.scrape_type = scrape_type
|
|
89
|
+
self.items_limit = items_limit
|
|
90
|
+
self.allow_subdomains = allow_subdomains
|
|
91
|
+
self.whitelist_regexp = whitelist_regexp
|
|
92
|
+
self.blacklist_regexp = blacklist_regexp
|
|
93
|
+
self.max_polls = max_polls
|
|
94
|
+
|
|
95
|
+
logger.debug(f"Initializing WebCrawlerAPILoader with URL: {url}, base_url: {self.base_url}")
|
|
96
|
+
self.client = WebCrawlerAPI(
|
|
97
|
+
api_key=self.api_key,
|
|
98
|
+
base_url=self.base_url,
|
|
99
|
+
version=version
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
def _create_document(self, item: Any) -> Optional[Document]:
|
|
103
|
+
"""Create a Document from a job item if it's valid.
|
|
104
|
+
|
|
105
|
+
Args:
|
|
106
|
+
item: Job item from the API response
|
|
107
|
+
|
|
108
|
+
Returns:
|
|
109
|
+
Document if item is valid and has content, None otherwise
|
|
110
|
+
"""
|
|
111
|
+
try:
|
|
112
|
+
logger.debug(f"Processing job item: {item}")
|
|
113
|
+
if not (item.status == "done" and item.content):
|
|
114
|
+
logger.debug(f"Skipping item - status: {item.status}, has_content: {bool(item.content)}")
|
|
115
|
+
return None
|
|
116
|
+
|
|
117
|
+
doc = Document(
|
|
118
|
+
page_content=item.content,
|
|
119
|
+
metadata={
|
|
120
|
+
"url": item.original_url,
|
|
121
|
+
"title": item.title,
|
|
122
|
+
"status_code": item.page_status_code,
|
|
123
|
+
"created_at": item.created_at,
|
|
124
|
+
"referred_url": item.referred_url,
|
|
125
|
+
"cost": item.cost
|
|
126
|
+
}
|
|
127
|
+
)
|
|
128
|
+
logger.debug(f"Created document from item: {item.original_url}")
|
|
129
|
+
return doc
|
|
130
|
+
except AttributeError as e:
|
|
131
|
+
logger.error(f"Failed to create document from item: {e}")
|
|
132
|
+
raise WebCrawlerAPILoaderError(f"Invalid job item format: {str(e)}") from e
|
|
133
|
+
|
|
134
|
+
def load(self) -> List[Document]:
|
|
135
|
+
"""Load data into Document objects.
|
|
136
|
+
|
|
137
|
+
Returns:
|
|
138
|
+
List of Document objects, one for each crawled page.
|
|
139
|
+
|
|
140
|
+
Raises:
|
|
141
|
+
WebCrawlerAPILoaderError: If there are issues with the crawling job or API response
|
|
142
|
+
ValueError: If the response format is invalid
|
|
143
|
+
json.JSONDecodeError: If the API response contains invalid JSON
|
|
144
|
+
"""
|
|
145
|
+
logger.info(f"Starting crawl for URL: {self.url}")
|
|
146
|
+
try:
|
|
147
|
+
logger.debug("Making crawl request with params: "
|
|
148
|
+
f"scrape_type={self.scrape_type}, "
|
|
149
|
+
f"items_limit={self.items_limit}, "
|
|
150
|
+
f"allow_subdomains={self.allow_subdomains}")
|
|
151
|
+
|
|
152
|
+
job = self.client.crawl(
|
|
153
|
+
url=self.url,
|
|
154
|
+
scrape_type=self.scrape_type,
|
|
155
|
+
items_limit=self.items_limit,
|
|
156
|
+
allow_subdomains=self.allow_subdomains,
|
|
157
|
+
whitelist_regexp=self.whitelist_regexp,
|
|
158
|
+
blacklist_regexp=self.blacklist_regexp,
|
|
159
|
+
max_polls=self.max_polls
|
|
160
|
+
)
|
|
161
|
+
logger.debug(f"Received crawl response: {job}")
|
|
162
|
+
|
|
163
|
+
except json.JSONDecodeError as e:
|
|
164
|
+
logger.error(f"JSON decode error in crawl response: {e}")
|
|
165
|
+
raise WebCrawlerAPILoaderError(
|
|
166
|
+
f"Failed to parse API response (invalid JSON at position {e.pos}): {str(e)}"
|
|
167
|
+
) from e
|
|
168
|
+
except Exception as e:
|
|
169
|
+
logger.error(f"API request failed: {e}")
|
|
170
|
+
raise WebCrawlerAPILoaderError(f"API request failed: {str(e)}") from e
|
|
171
|
+
|
|
172
|
+
if job.status == "error":
|
|
173
|
+
error_msg = getattr(job, 'error', 'Unknown error')
|
|
174
|
+
logger.error(f"Crawl job failed with error: {error_msg}")
|
|
175
|
+
raise WebCrawlerAPILoaderError(f"Crawling job failed: {error_msg}")
|
|
176
|
+
|
|
177
|
+
documents = []
|
|
178
|
+
logger.info(f"Processing {len(job.job_items)} job items")
|
|
179
|
+
for item in job.job_items:
|
|
180
|
+
try:
|
|
181
|
+
doc = self._create_document(item)
|
|
182
|
+
if doc:
|
|
183
|
+
documents.append(doc)
|
|
184
|
+
except AttributeError as e:
|
|
185
|
+
logger.error(f"Failed to process job item: {e}")
|
|
186
|
+
raise WebCrawlerAPILoaderError(
|
|
187
|
+
f"Invalid job item format - missing required field: {str(e)}"
|
|
188
|
+
) from e
|
|
189
|
+
|
|
190
|
+
logger.info(f"Successfully created {len(documents)} documents")
|
|
191
|
+
return documents
|
|
192
|
+
|
|
193
|
+
def lazy_load(self) -> Iterator[Document]:
|
|
194
|
+
"""A lazy loader for Documents.
|
|
195
|
+
|
|
196
|
+
Yields:
|
|
197
|
+
Document objects one at a time as they are crawled.
|
|
198
|
+
|
|
199
|
+
Raises:
|
|
200
|
+
WebCrawlerAPILoaderError: If there are issues with the crawling job or API response
|
|
201
|
+
ValueError: If the response format is invalid
|
|
202
|
+
json.JSONDecodeError: If the API response contains invalid JSON
|
|
203
|
+
"""
|
|
204
|
+
logger.info(f"Starting async crawl for URL: {self.url}")
|
|
205
|
+
try:
|
|
206
|
+
logger.debug("Making async crawl request")
|
|
207
|
+
response = self.client.crawl_async(
|
|
208
|
+
url=self.url,
|
|
209
|
+
scrape_type=self.scrape_type,
|
|
210
|
+
items_limit=self.items_limit,
|
|
211
|
+
allow_subdomains=self.allow_subdomains,
|
|
212
|
+
whitelist_regexp=self.whitelist_regexp,
|
|
213
|
+
blacklist_regexp=self.blacklist_regexp
|
|
214
|
+
)
|
|
215
|
+
logger.debug(f"Received async crawl response: {response}")
|
|
216
|
+
|
|
217
|
+
except json.JSONDecodeError as e:
|
|
218
|
+
logger.error(f"JSON decode error in async crawl response: {e}")
|
|
219
|
+
raise WebCrawlerAPILoaderError(
|
|
220
|
+
f"Failed to parse API response (invalid JSON at position {e.pos}): {str(e)}"
|
|
221
|
+
) from e
|
|
222
|
+
except Exception as e:
|
|
223
|
+
logger.error(f"Async API request failed: {e}")
|
|
224
|
+
raise WebCrawlerAPILoaderError(f"API request failed: {str(e)}") from e
|
|
225
|
+
|
|
226
|
+
job_id = response.id
|
|
227
|
+
polls = 0
|
|
228
|
+
processed_items = set()
|
|
229
|
+
logger.info(f"Starting to poll job {job_id}")
|
|
230
|
+
|
|
231
|
+
while polls < self.max_polls:
|
|
232
|
+
try:
|
|
233
|
+
logger.debug(f"Polling job {job_id} (attempt {polls + 1}/{self.max_polls})")
|
|
234
|
+
job = self.client.get_job(job_id)
|
|
235
|
+
logger.debug(f"Received job status: {job.status}")
|
|
236
|
+
|
|
237
|
+
except json.JSONDecodeError as e:
|
|
238
|
+
logger.error(f"JSON decode error in job status response: {e}")
|
|
239
|
+
raise WebCrawlerAPILoaderError(
|
|
240
|
+
f"Failed to parse job status response (invalid JSON at position {e.pos}): {str(e)}"
|
|
241
|
+
) from e
|
|
242
|
+
except Exception as e:
|
|
243
|
+
logger.error(f"Failed to fetch job status: {e}")
|
|
244
|
+
raise WebCrawlerAPILoaderError(f"Failed to fetch job status: {str(e)}") from e
|
|
245
|
+
|
|
246
|
+
if job.status == "error":
|
|
247
|
+
error_msg = getattr(job, 'error', 'Unknown error')
|
|
248
|
+
logger.error(f"Job failed with error: {error_msg}")
|
|
249
|
+
raise WebCrawlerAPILoaderError(f"Crawling job failed: {error_msg}")
|
|
250
|
+
|
|
251
|
+
for item in job.job_items:
|
|
252
|
+
if item.id not in processed_items:
|
|
253
|
+
try:
|
|
254
|
+
doc = self._create_document(item)
|
|
255
|
+
if doc:
|
|
256
|
+
processed_items.add(item.id)
|
|
257
|
+
logger.debug(f"Yielding document for URL: {item.original_url}")
|
|
258
|
+
yield doc
|
|
259
|
+
except AttributeError as e:
|
|
260
|
+
logger.error(f"Failed to process job item: {e}")
|
|
261
|
+
raise WebCrawlerAPILoaderError(
|
|
262
|
+
f"Invalid job item format - missing required field: {str(e)}"
|
|
263
|
+
) from e
|
|
264
|
+
|
|
265
|
+
if job.is_terminal:
|
|
266
|
+
logger.info("Job completed successfully")
|
|
267
|
+
break
|
|
268
|
+
|
|
269
|
+
delay_seconds = (
|
|
270
|
+
job.recommended_pull_delay_ms / 1000
|
|
271
|
+
if job.recommended_pull_delay_ms
|
|
272
|
+
else self.client.DEFAULT_POLL_DELAY_SECONDS
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
import time
|
|
276
|
+
logger.debug(f"Waiting {delay_seconds} seconds before next poll")
|
|
277
|
+
time.sleep(delay_seconds)
|
|
278
|
+
polls += 1
|
|
279
|
+
|
|
280
|
+
if not job.is_terminal and polls >= self.max_polls:
|
|
281
|
+
logger.error(f"Job timed out after {self.max_polls} polls")
|
|
282
|
+
raise WebCrawlerAPILoaderError(
|
|
283
|
+
f"Maximum number of polls ({self.max_polls}) reached without job completion"
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
async def aload(self) -> List[Document]:
|
|
287
|
+
"""Asynchronously load data into Document objects.
|
|
288
|
+
|
|
289
|
+
Returns:
|
|
290
|
+
List of Document objects, one for each crawled page.
|
|
291
|
+
|
|
292
|
+
Raises:
|
|
293
|
+
RuntimeError: If the crawling job fails
|
|
294
|
+
"""
|
|
295
|
+
import asyncio
|
|
296
|
+
return await asyncio.to_thread(self.load)
|
|
297
|
+
|
|
298
|
+
async def alazy_load(self) -> AsyncIterator[Document]:
|
|
299
|
+
"""An async lazy loader for Documents.
|
|
300
|
+
|
|
301
|
+
Yields:
|
|
302
|
+
Document objects one at a time as they are crawled.
|
|
303
|
+
|
|
304
|
+
Raises:
|
|
305
|
+
RuntimeError: If the crawling job fails
|
|
306
|
+
"""
|
|
307
|
+
import asyncio
|
|
308
|
+
for doc in self.lazy_load():
|
|
309
|
+
yield doc
|
|
310
|
+
await asyncio.sleep(0)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: webcrawlerapi-langchain
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: LangChain integration for WebCrawlerAPI
|
|
5
|
+
Home-page: https://github.com/webcrawlerapi/webcrawlerapi-langchain
|
|
6
|
+
Author: WebCrawlerAPI
|
|
7
|
+
Author-email: support@webcrawlerapi.com
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Requires-Python: >=3.8
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
Requires-Dist: webcrawlerapi==1.0.6
|
|
14
|
+
Requires-Dist: langchain-core>=0.1.0
|
|
15
|
+
Dynamic: author
|
|
16
|
+
Dynamic: author-email
|
|
17
|
+
Dynamic: classifier
|
|
18
|
+
Dynamic: description
|
|
19
|
+
Dynamic: description-content-type
|
|
20
|
+
Dynamic: home-page
|
|
21
|
+
Dynamic: requires-dist
|
|
22
|
+
Dynamic: requires-python
|
|
23
|
+
Dynamic: summary
|
|
24
|
+
|
|
25
|
+
# WebCrawlerAPI LangChain Integration
|
|
26
|
+
|
|
27
|
+
[WebcrawlerAPI](https://webcrawlerapi.com/) - is a website to LLM data API. It allows to convert websites and webpages markdown or cleaned content.
|
|
28
|
+
|
|
29
|
+
**No subscription required**.
|
|
30
|
+
|
|
31
|
+
This package provides LangChain integration for [WebCrawlerAPI](https://webcrawlerapi.com/), allowing you to easily use web crawling capabilities with [LangChain](https://www.langchain.com/) document processing pipeline.
|
|
32
|
+
|
|
33
|
+
## Installation
|
|
34
|
+
|
|
35
|
+
Get your [API key](https://webcrawlerapi.com/docs/access-key) first
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install webcrawlerapi-langchain
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Usage
|
|
42
|
+
|
|
43
|
+
### Basic Loading
|
|
44
|
+
```python
|
|
45
|
+
from webcrawlerapi_langchain import WebCrawlerAPILoader
|
|
46
|
+
|
|
47
|
+
# Initialize the loader
|
|
48
|
+
loader = WebCrawlerAPILoader(
|
|
49
|
+
url="https://example.com",
|
|
50
|
+
api_key="your-api-key",
|
|
51
|
+
scrape_type="markdown",
|
|
52
|
+
items_limit=10
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
# Load documents
|
|
56
|
+
documents = loader.load()
|
|
57
|
+
|
|
58
|
+
# Use documents in your LangChain pipeline
|
|
59
|
+
for doc in documents:
|
|
60
|
+
print(doc.page_content[:100])
|
|
61
|
+
print(doc.metadata)
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### Async Loading
|
|
65
|
+
```python
|
|
66
|
+
# Async loading
|
|
67
|
+
documents = await loader.aload()
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
### Lazy Loading
|
|
71
|
+
```python
|
|
72
|
+
# Lazy loading
|
|
73
|
+
for doc in loader.lazy_load():
|
|
74
|
+
print(doc.page_content[:100])
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
### Async Lazy Loading
|
|
78
|
+
```python
|
|
79
|
+
# Async lazy loading
|
|
80
|
+
async for doc in loader.alazy_load():
|
|
81
|
+
print(doc.page_content[:100])
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
## Configuration
|
|
85
|
+
|
|
86
|
+
The loader accepts the following parameters:
|
|
87
|
+
|
|
88
|
+
- `url`: The URL to crawl
|
|
89
|
+
- `api_key`: Your WebCrawlerAPI API key
|
|
90
|
+
- `scrape_type`: Type of scraping (html, cleaned, markdown)
|
|
91
|
+
- `items_limit`: Maximum number of pages to crawl
|
|
92
|
+
- `whitelist_regexp`: Regex pattern for URL whitelist
|
|
93
|
+
- `blacklist_regexp`: Regex pattern for URL blacklist
|
|
94
|
+
|
|
95
|
+
### Links
|
|
96
|
+
- [WebCrawlerAPI Python SDK](https://github.com/WebCrawlerAPI/webcrawlerapi-python-sdk)
|
|
97
|
+
- [WebCrawlerAPI Documentation](https://webcrawlerapi.com/docs/getting-started)
|
|
98
|
+
- [WebCrawlerAPI API Reference](https://webcrawlerapi.com/docs/api-reference)
|
|
99
|
+
|
|
100
|
+
If you need help with integration feel free to [contact us](support@webcrawlerapi.com).
|
|
101
|
+
|
|
102
|
+
## License
|
|
103
|
+
|
|
104
|
+
MIT License
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
webcrawlerapi_langchain/__init__.py,sha256=iGJbhONAD-DOCQt1biaMXCRXV6TSZ345il_3yRnFxT8,97
|
|
2
|
+
webcrawlerapi_langchain/loader.py,sha256=-6k91BcxKhWIBK7cAKg37Ad8YOWDZxooJLXT-2Mgdgw,12418
|
|
3
|
+
webcrawlerapi_langchain-0.1.0.dist-info/METADATA,sha256=inCIWMKKJf93VOZ1qKr8k-iyah5nUTU3f1MxcIn2pF0,2778
|
|
4
|
+
webcrawlerapi_langchain-0.1.0.dist-info/WHEEL,sha256=CmyFI0kx5cdEMTLiONQRbGQwjIoR1aIYB7eCAQ4KPJ0,91
|
|
5
|
+
webcrawlerapi_langchain-0.1.0.dist-info/top_level.txt,sha256=Jh7YnCZY0svZUVeItbDfACLakPi1tBwqGb3NNESgKV0,24
|
|
6
|
+
webcrawlerapi_langchain-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
webcrawlerapi_langchain
|