webcrawlerapi-langchain 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,4 @@
1
+ from .loader import WebCrawlerAPILoader
2
+
3
+ __version__ = "0.1.0"
4
+ __all__ = ["WebCrawlerAPILoader"]
@@ -0,0 +1,310 @@
1
+ from typing import Dict, Iterator, List, Optional, AsyncIterator, Any, Literal
2
+ import os
3
+ import json
4
+ import logging
5
+ from langchain_core.documents import Document
6
+ from langchain_core.utils import get_from_env
7
+ from langchain_core.document_loaders import BaseLoader
8
+ from webcrawlerapi import WebCrawlerAPI
9
+
10
+ # Configure logging
11
+ logger = logging.getLogger(__name__)
12
+ logger.setLevel(logging.DEBUG)
13
+
14
+ class WebCrawlerAPILoaderError(Exception):
15
+ """Custom exception class for WebCrawlerAPILoader errors."""
16
+ pass
17
+
18
+ class WebCrawlerAPILoader(BaseLoader):
19
+ """WebCrawlerAPI document loader integration.
20
+
21
+ This loader uses WebCrawlerAPI to crawl websites and convert them into LangChain Documents.
22
+ Each crawled page becomes a Document with its content and metadata.
23
+
24
+ Raises:
25
+ WebCrawlerAPILoaderError: If there are issues with initialization or crawling
26
+ ValueError: If required parameters are invalid
27
+ """
28
+
29
+ def __init__(
30
+ self,
31
+ url: str,
32
+ *,
33
+ api_key: Optional[str] = None,
34
+ base_url: Optional[str] = None,
35
+ version: str = "v1",
36
+ scrape_type: Literal["html", "cleaned", "markdown"] = "markdown",
37
+ items_limit: int = 10,
38
+ allow_subdomains: bool = False,
39
+ whitelist_regexp: Optional[str] = None,
40
+ blacklist_regexp: Optional[str] = None,
41
+ max_polls: int = 100
42
+ ):
43
+ """Initialize the WebCrawlerAPI document loader.
44
+
45
+ Args:
46
+ url: The URL to crawl
47
+ api_key: Your WebCrawlerAPI API key. If not provided, will try to get from WEBCRAWLERAPI_API_KEY env var
48
+ base_url: The base URL of the API. If not provided, will try to get from WEBCRAWLERAPI_BASE_URL env var
49
+ version: API version to use (optional)
50
+ scrape_type: Type of scraping (html, cleaned, markdown)
51
+ items_limit: Maximum number of pages to crawl
52
+ allow_subdomains: Whether to crawl subdomains
53
+ whitelist_regexp: Regex pattern for URL whitelist
54
+ blacklist_regexp: Regex pattern for URL blacklist
55
+ max_polls: Maximum number of status checks before returning
56
+ """
57
+ if not url:
58
+ raise ValueError("URL must be provided")
59
+
60
+ if not url.startswith(('http://', 'https://')):
61
+ raise ValueError("URL must start with http:// or https://")
62
+
63
+ if items_limit < 1:
64
+ raise ValueError("items_limit must be greater than 0")
65
+
66
+ if max_polls < 1:
67
+ raise ValueError("max_polls must be greater than 0")
68
+
69
+ if scrape_type not in ("html", "cleaned", "markdown"):
70
+ raise ValueError(
71
+ f"Invalid scrape_type '{scrape_type}'. "
72
+ "Allowed: 'html', 'cleaned', 'markdown'."
73
+ )
74
+
75
+ # Get API key and base URL from env vars if not provided
76
+ self.api_key = api_key or get_from_env("api_key", "WEBCRAWLERAPI_API_KEY")
77
+ if not self.api_key:
78
+ raise ValueError("API key must be provided either through api_key parameter or WEBCRAWLERAPI_API_KEY environment variable")
79
+
80
+ self.base_url = base_url or get_from_env(
81
+ "base_url",
82
+ "WEBCRAWLERAPI_BASE_URL",
83
+ default="https://api.webcrawlerapi.com"
84
+ )
85
+
86
+ self.url = url
87
+ self.version = version
88
+ self.scrape_type = scrape_type
89
+ self.items_limit = items_limit
90
+ self.allow_subdomains = allow_subdomains
91
+ self.whitelist_regexp = whitelist_regexp
92
+ self.blacklist_regexp = blacklist_regexp
93
+ self.max_polls = max_polls
94
+
95
+ logger.debug(f"Initializing WebCrawlerAPILoader with URL: {url}, base_url: {self.base_url}")
96
+ self.client = WebCrawlerAPI(
97
+ api_key=self.api_key,
98
+ base_url=self.base_url,
99
+ version=version
100
+ )
101
+
102
+ def _create_document(self, item: Any) -> Optional[Document]:
103
+ """Create a Document from a job item if it's valid.
104
+
105
+ Args:
106
+ item: Job item from the API response
107
+
108
+ Returns:
109
+ Document if item is valid and has content, None otherwise
110
+ """
111
+ try:
112
+ logger.debug(f"Processing job item: {item}")
113
+ if not (item.status == "done" and item.content):
114
+ logger.debug(f"Skipping item - status: {item.status}, has_content: {bool(item.content)}")
115
+ return None
116
+
117
+ doc = Document(
118
+ page_content=item.content,
119
+ metadata={
120
+ "url": item.original_url,
121
+ "title": item.title,
122
+ "status_code": item.page_status_code,
123
+ "created_at": item.created_at,
124
+ "referred_url": item.referred_url,
125
+ "cost": item.cost
126
+ }
127
+ )
128
+ logger.debug(f"Created document from item: {item.original_url}")
129
+ return doc
130
+ except AttributeError as e:
131
+ logger.error(f"Failed to create document from item: {e}")
132
+ raise WebCrawlerAPILoaderError(f"Invalid job item format: {str(e)}") from e
133
+
134
+ def load(self) -> List[Document]:
135
+ """Load data into Document objects.
136
+
137
+ Returns:
138
+ List of Document objects, one for each crawled page.
139
+
140
+ Raises:
141
+ WebCrawlerAPILoaderError: If there are issues with the crawling job or API response
142
+ ValueError: If the response format is invalid
143
+ json.JSONDecodeError: If the API response contains invalid JSON
144
+ """
145
+ logger.info(f"Starting crawl for URL: {self.url}")
146
+ try:
147
+ logger.debug("Making crawl request with params: "
148
+ f"scrape_type={self.scrape_type}, "
149
+ f"items_limit={self.items_limit}, "
150
+ f"allow_subdomains={self.allow_subdomains}")
151
+
152
+ job = self.client.crawl(
153
+ url=self.url,
154
+ scrape_type=self.scrape_type,
155
+ items_limit=self.items_limit,
156
+ allow_subdomains=self.allow_subdomains,
157
+ whitelist_regexp=self.whitelist_regexp,
158
+ blacklist_regexp=self.blacklist_regexp,
159
+ max_polls=self.max_polls
160
+ )
161
+ logger.debug(f"Received crawl response: {job}")
162
+
163
+ except json.JSONDecodeError as e:
164
+ logger.error(f"JSON decode error in crawl response: {e}")
165
+ raise WebCrawlerAPILoaderError(
166
+ f"Failed to parse API response (invalid JSON at position {e.pos}): {str(e)}"
167
+ ) from e
168
+ except Exception as e:
169
+ logger.error(f"API request failed: {e}")
170
+ raise WebCrawlerAPILoaderError(f"API request failed: {str(e)}") from e
171
+
172
+ if job.status == "error":
173
+ error_msg = getattr(job, 'error', 'Unknown error')
174
+ logger.error(f"Crawl job failed with error: {error_msg}")
175
+ raise WebCrawlerAPILoaderError(f"Crawling job failed: {error_msg}")
176
+
177
+ documents = []
178
+ logger.info(f"Processing {len(job.job_items)} job items")
179
+ for item in job.job_items:
180
+ try:
181
+ doc = self._create_document(item)
182
+ if doc:
183
+ documents.append(doc)
184
+ except AttributeError as e:
185
+ logger.error(f"Failed to process job item: {e}")
186
+ raise WebCrawlerAPILoaderError(
187
+ f"Invalid job item format - missing required field: {str(e)}"
188
+ ) from e
189
+
190
+ logger.info(f"Successfully created {len(documents)} documents")
191
+ return documents
192
+
193
+ def lazy_load(self) -> Iterator[Document]:
194
+ """A lazy loader for Documents.
195
+
196
+ Yields:
197
+ Document objects one at a time as they are crawled.
198
+
199
+ Raises:
200
+ WebCrawlerAPILoaderError: If there are issues with the crawling job or API response
201
+ ValueError: If the response format is invalid
202
+ json.JSONDecodeError: If the API response contains invalid JSON
203
+ """
204
+ logger.info(f"Starting async crawl for URL: {self.url}")
205
+ try:
206
+ logger.debug("Making async crawl request")
207
+ response = self.client.crawl_async(
208
+ url=self.url,
209
+ scrape_type=self.scrape_type,
210
+ items_limit=self.items_limit,
211
+ allow_subdomains=self.allow_subdomains,
212
+ whitelist_regexp=self.whitelist_regexp,
213
+ blacklist_regexp=self.blacklist_regexp
214
+ )
215
+ logger.debug(f"Received async crawl response: {response}")
216
+
217
+ except json.JSONDecodeError as e:
218
+ logger.error(f"JSON decode error in async crawl response: {e}")
219
+ raise WebCrawlerAPILoaderError(
220
+ f"Failed to parse API response (invalid JSON at position {e.pos}): {str(e)}"
221
+ ) from e
222
+ except Exception as e:
223
+ logger.error(f"Async API request failed: {e}")
224
+ raise WebCrawlerAPILoaderError(f"API request failed: {str(e)}") from e
225
+
226
+ job_id = response.id
227
+ polls = 0
228
+ processed_items = set()
229
+ logger.info(f"Starting to poll job {job_id}")
230
+
231
+ while polls < self.max_polls:
232
+ try:
233
+ logger.debug(f"Polling job {job_id} (attempt {polls + 1}/{self.max_polls})")
234
+ job = self.client.get_job(job_id)
235
+ logger.debug(f"Received job status: {job.status}")
236
+
237
+ except json.JSONDecodeError as e:
238
+ logger.error(f"JSON decode error in job status response: {e}")
239
+ raise WebCrawlerAPILoaderError(
240
+ f"Failed to parse job status response (invalid JSON at position {e.pos}): {str(e)}"
241
+ ) from e
242
+ except Exception as e:
243
+ logger.error(f"Failed to fetch job status: {e}")
244
+ raise WebCrawlerAPILoaderError(f"Failed to fetch job status: {str(e)}") from e
245
+
246
+ if job.status == "error":
247
+ error_msg = getattr(job, 'error', 'Unknown error')
248
+ logger.error(f"Job failed with error: {error_msg}")
249
+ raise WebCrawlerAPILoaderError(f"Crawling job failed: {error_msg}")
250
+
251
+ for item in job.job_items:
252
+ if item.id not in processed_items:
253
+ try:
254
+ doc = self._create_document(item)
255
+ if doc:
256
+ processed_items.add(item.id)
257
+ logger.debug(f"Yielding document for URL: {item.original_url}")
258
+ yield doc
259
+ except AttributeError as e:
260
+ logger.error(f"Failed to process job item: {e}")
261
+ raise WebCrawlerAPILoaderError(
262
+ f"Invalid job item format - missing required field: {str(e)}"
263
+ ) from e
264
+
265
+ if job.is_terminal:
266
+ logger.info("Job completed successfully")
267
+ break
268
+
269
+ delay_seconds = (
270
+ job.recommended_pull_delay_ms / 1000
271
+ if job.recommended_pull_delay_ms
272
+ else self.client.DEFAULT_POLL_DELAY_SECONDS
273
+ )
274
+
275
+ import time
276
+ logger.debug(f"Waiting {delay_seconds} seconds before next poll")
277
+ time.sleep(delay_seconds)
278
+ polls += 1
279
+
280
+ if not job.is_terminal and polls >= self.max_polls:
281
+ logger.error(f"Job timed out after {self.max_polls} polls")
282
+ raise WebCrawlerAPILoaderError(
283
+ f"Maximum number of polls ({self.max_polls}) reached without job completion"
284
+ )
285
+
286
+ async def aload(self) -> List[Document]:
287
+ """Asynchronously load data into Document objects.
288
+
289
+ Returns:
290
+ List of Document objects, one for each crawled page.
291
+
292
+ Raises:
293
+ RuntimeError: If the crawling job fails
294
+ """
295
+ import asyncio
296
+ return await asyncio.to_thread(self.load)
297
+
298
+ async def alazy_load(self) -> AsyncIterator[Document]:
299
+ """An async lazy loader for Documents.
300
+
301
+ Yields:
302
+ Document objects one at a time as they are crawled.
303
+
304
+ Raises:
305
+ RuntimeError: If the crawling job fails
306
+ """
307
+ import asyncio
308
+ for doc in self.lazy_load():
309
+ yield doc
310
+ await asyncio.sleep(0)
@@ -0,0 +1,104 @@
1
+ Metadata-Version: 2.4
2
+ Name: webcrawlerapi-langchain
3
+ Version: 0.1.0
4
+ Summary: LangChain integration for WebCrawlerAPI
5
+ Home-page: https://github.com/webcrawlerapi/webcrawlerapi-langchain
6
+ Author: WebCrawlerAPI
7
+ Author-email: support@webcrawlerapi.com
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: OS Independent
11
+ Requires-Python: >=3.8
12
+ Description-Content-Type: text/markdown
13
+ Requires-Dist: webcrawlerapi==1.0.6
14
+ Requires-Dist: langchain-core>=0.1.0
15
+ Dynamic: author
16
+ Dynamic: author-email
17
+ Dynamic: classifier
18
+ Dynamic: description
19
+ Dynamic: description-content-type
20
+ Dynamic: home-page
21
+ Dynamic: requires-dist
22
+ Dynamic: requires-python
23
+ Dynamic: summary
24
+
25
+ # WebCrawlerAPI LangChain Integration
26
+
27
+ [WebcrawlerAPI](https://webcrawlerapi.com/) - is a website to LLM data API. It allows to convert websites and webpages markdown or cleaned content.
28
+
29
+ **No subscription required**.
30
+
31
+ This package provides LangChain integration for [WebCrawlerAPI](https://webcrawlerapi.com/), allowing you to easily use web crawling capabilities with [LangChain](https://www.langchain.com/) document processing pipeline.
32
+
33
+ ## Installation
34
+
35
+ Get your [API key](https://webcrawlerapi.com/docs/access-key) first
36
+
37
+ ```bash
38
+ pip install webcrawlerapi-langchain
39
+ ```
40
+
41
+ ## Usage
42
+
43
+ ### Basic Loading
44
+ ```python
45
+ from webcrawlerapi_langchain import WebCrawlerAPILoader
46
+
47
+ # Initialize the loader
48
+ loader = WebCrawlerAPILoader(
49
+ url="https://example.com",
50
+ api_key="your-api-key",
51
+ scrape_type="markdown",
52
+ items_limit=10
53
+ )
54
+
55
+ # Load documents
56
+ documents = loader.load()
57
+
58
+ # Use documents in your LangChain pipeline
59
+ for doc in documents:
60
+ print(doc.page_content[:100])
61
+ print(doc.metadata)
62
+ ```
63
+
64
+ ### Async Loading
65
+ ```python
66
+ # Async loading
67
+ documents = await loader.aload()
68
+ ```
69
+
70
+ ### Lazy Loading
71
+ ```python
72
+ # Lazy loading
73
+ for doc in loader.lazy_load():
74
+ print(doc.page_content[:100])
75
+ ```
76
+
77
+ ### Async Lazy Loading
78
+ ```python
79
+ # Async lazy loading
80
+ async for doc in loader.alazy_load():
81
+ print(doc.page_content[:100])
82
+ ```
83
+
84
+ ## Configuration
85
+
86
+ The loader accepts the following parameters:
87
+
88
+ - `url`: The URL to crawl
89
+ - `api_key`: Your WebCrawlerAPI API key
90
+ - `scrape_type`: Type of scraping (html, cleaned, markdown)
91
+ - `items_limit`: Maximum number of pages to crawl
92
+ - `whitelist_regexp`: Regex pattern for URL whitelist
93
+ - `blacklist_regexp`: Regex pattern for URL blacklist
94
+
95
+ ### Links
96
+ - [WebCrawlerAPI Python SDK](https://github.com/WebCrawlerAPI/webcrawlerapi-python-sdk)
97
+ - [WebCrawlerAPI Documentation](https://webcrawlerapi.com/docs/getting-started)
98
+ - [WebCrawlerAPI API Reference](https://webcrawlerapi.com/docs/api-reference)
99
+
100
+ If you need help with integration feel free to [contact us](support@webcrawlerapi.com).
101
+
102
+ ## License
103
+
104
+ MIT License
@@ -0,0 +1,6 @@
1
+ webcrawlerapi_langchain/__init__.py,sha256=iGJbhONAD-DOCQt1biaMXCRXV6TSZ345il_3yRnFxT8,97
2
+ webcrawlerapi_langchain/loader.py,sha256=-6k91BcxKhWIBK7cAKg37Ad8YOWDZxooJLXT-2Mgdgw,12418
3
+ webcrawlerapi_langchain-0.1.0.dist-info/METADATA,sha256=inCIWMKKJf93VOZ1qKr8k-iyah5nUTU3f1MxcIn2pF0,2778
4
+ webcrawlerapi_langchain-0.1.0.dist-info/WHEEL,sha256=CmyFI0kx5cdEMTLiONQRbGQwjIoR1aIYB7eCAQ4KPJ0,91
5
+ webcrawlerapi_langchain-0.1.0.dist-info/top_level.txt,sha256=Jh7YnCZY0svZUVeItbDfACLakPi1tBwqGb3NNESgKV0,24
6
+ webcrawlerapi_langchain-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (78.1.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1 @@
1
+ webcrawlerapi_langchain