crawlerflow 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/CHANGELOG.md +23 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/PKG-INFO +1 -1
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/__init__.py +1 -1
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/browser/base.py +7 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/browser/pydoll.py +62 -1
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/context.py +2 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/steps/utility.py +10 -2
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/plugins.md +50 -0
- crawlerflow-0.3.0/examples/plugins/capsolver/README.md +66 -0
- crawlerflow-0.3.0/examples/plugins/capsolver/pyproject.toml +21 -0
- crawlerflow-0.3.0/examples/plugins/capsolver/src/crawlerflow_capsolver/__init__.py +503 -0
- crawlerflow-0.3.0/examples/plugins/capsolver/workflow.yaml +28 -0
- crawlerflow-0.3.0/examples/plugins/capsolver-webshare/cloudflare-challenge-http.yaml +68 -0
- crawlerflow-0.3.0/examples/plugins/capsolver-webshare/cloudflare-challenge.yaml +74 -0
- crawlerflow-0.3.0/examples/plugins/capsolver-webshare/workflow.yaml +54 -0
- crawlerflow-0.3.0/examples/plugins/webshare/README.md +63 -0
- crawlerflow-0.3.0/examples/plugins/webshare/pyproject.toml +21 -0
- crawlerflow-0.3.0/examples/plugins/webshare/src/crawlerflow_webshare/__init__.py +454 -0
- crawlerflow-0.3.0/examples/plugins/webshare/workflow.yaml +28 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/pyproject.toml +1 -1
- crawlerflow-0.3.0/tests/test_capsolver_plugin.py +96 -0
- crawlerflow-0.3.0/tests/test_combined_plugin_workflow.py +55 -0
- crawlerflow-0.3.0/tests/test_proxy_retry.py +141 -0
- crawlerflow-0.3.0/tests/test_webshare_plugin.py +127 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/.gitignore +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/LICENSE +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/README.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/__main__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/browser/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/browser/factory.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/cli/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/cli/app.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/executor.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/registry.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/runner.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/bus.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/console_logger.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/json_logger.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/models.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/expressions/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/expressions/conditions.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/expressions/engine.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/plugins/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/plugins/base.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/plugins/manager.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/py.typed +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/steps/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/steps/browser.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/steps/control.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/workflow/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/workflow/loader.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/workflow/models.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/adr/0001-core-boundaries.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/architecture.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/control-flow.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/getting-started.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/html-output.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/http-requests.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/index.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/pydoll.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/retries-and-logging.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/runtime-variables.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/basic.yaml +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/control-flow.yaml +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/loops-macros.yaml +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/plugins/example/README.md +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/plugins/example/pyproject.toml +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/plugins/example/src/crawlerflow_example_plugin/__init__.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/plugins/example/workflow.yaml +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/pydoll.yaml +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/retry-logging.yaml +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_browser_steps.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_cli.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_conditions.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_control_flow.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_example_plugin.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_expressions.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_loops_and_macros.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_plugins.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_pydoll_adapter.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_retry_and_logging.py +0 -0
- {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_workflow.py +0 -0
|
@@ -5,6 +5,28 @@ All notable changes to this project are documented in this file.
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and this project
|
|
6
6
|
adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
|
|
8
|
+
## [0.3.0] - 2026-09-17
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- Workflow-wide proxy support. `WorkflowContext` carries a `proxy_url` that survives parallel loop
|
|
13
|
+
forks, and the built-in `http_request`, `resolve_location_url`, `enrich_html_links_http`, and
|
|
14
|
+
`enrich_json_map_locations` steps route their requests through it.
|
|
15
|
+
- `BrowserAdapter.configure_proxy()` extension hook. Adapters that cannot proxy raise a clear
|
|
16
|
+
error, keeping the contract explicit.
|
|
17
|
+
- Pydoll proxy support for HTTP, HTTPS, and SOCKS5 URLs, including authenticated proxies through a
|
|
18
|
+
generated extension. The temporary extension directory is removed when the adapter closes.
|
|
19
|
+
- CapSolver reference plugin that solves captcha tasks and exposes the solution to later steps.
|
|
20
|
+
- Webshare reference plugin that selects proxies and publishes the active proxy to the workflow.
|
|
21
|
+
- Combined CapSolver and Webshare example workflows covering browser and HTTP-only Cloudflare
|
|
22
|
+
challenge flows.
|
|
23
|
+
- Plugin documentation covering proxy-aware plugins and the new adapter hook.
|
|
24
|
+
|
|
25
|
+
### Fixed
|
|
26
|
+
|
|
27
|
+
- Disabled Jekyll processing for the published documentation site so workflow expression examples
|
|
28
|
+
are no longer parsed as Liquid templates.
|
|
29
|
+
|
|
8
30
|
## [0.2.0] - 2026-08-18
|
|
9
31
|
|
|
10
32
|
### Changed
|
|
@@ -36,5 +58,6 @@ adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
|
36
58
|
- Plugin API with typed settings, lifecycle hooks, steps, filters, and subscribers.
|
|
37
59
|
- `run`, `validate`, `list-steps`, `list-plugins`, and `doctor` CLI commands.
|
|
38
60
|
|
|
61
|
+
[0.3.0]: https://github.com/mehmetemineker/crawlerflow/releases/tag/v0.3.0
|
|
39
62
|
[0.2.0]: https://github.com/mehmetemineker/crawlerflow/releases/tag/v0.2.0
|
|
40
63
|
[0.1.0]: https://github.com/mehmetemineker/crawlerflow/releases/tag/v0.1.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: crawlerflow
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Declarative YAML workflow engine for browser automation and web scraping
|
|
5
5
|
Project-URL: Homepage, https://github.com/mehmetemineker/crawlerflow
|
|
6
6
|
Project-URL: Repository, https://github.com/mehmetemineker/crawlerflow
|
|
@@ -66,6 +66,13 @@ class BrowserAdapter(ABC):
|
|
|
66
66
|
@abstractmethod
|
|
67
67
|
async def screenshot(self, path: Path) -> Path: ...
|
|
68
68
|
|
|
69
|
+
def configure_proxy(self, proxy_url: str) -> None:
|
|
70
|
+
"""Configure a proxy before the browser session starts, if supported."""
|
|
71
|
+
|
|
72
|
+
raise RuntimeError(
|
|
73
|
+
f"Browser adapter {type(self).__name__} does not support proxy configuration"
|
|
74
|
+
)
|
|
75
|
+
|
|
69
76
|
async def close(self) -> None:
|
|
70
77
|
"""Release browser resources when an adapter owns them."""
|
|
71
78
|
|
|
@@ -6,11 +6,14 @@ import asyncio
|
|
|
6
6
|
import json
|
|
7
7
|
import logging
|
|
8
8
|
import math
|
|
9
|
+
import shutil
|
|
10
|
+
import tempfile
|
|
9
11
|
import time
|
|
10
12
|
from collections.abc import Mapping
|
|
11
|
-
from dataclasses import dataclass
|
|
13
|
+
from dataclasses import dataclass, replace
|
|
12
14
|
from pathlib import Path
|
|
13
15
|
from typing import Any
|
|
16
|
+
from urllib.parse import unquote, urlsplit
|
|
14
17
|
|
|
15
18
|
from crawlerflow.browser.base import BrowserAdapter, BrowserResponse
|
|
16
19
|
|
|
@@ -40,6 +43,7 @@ class PydollBrowserConfig:
|
|
|
40
43
|
default_wait_timeout: float = 10
|
|
41
44
|
network_idle_period: float = 0.5
|
|
42
45
|
download_directory: Path | None = None
|
|
46
|
+
proxy_url: str | None = None
|
|
43
47
|
|
|
44
48
|
|
|
45
49
|
class PydollBrowserAdapter(BrowserAdapter):
|
|
@@ -61,6 +65,7 @@ class PydollBrowserAdapter(BrowserAdapter):
|
|
|
61
65
|
self._network_callback_ids: list[int] = []
|
|
62
66
|
self._inflight_requests: set[str] = set()
|
|
63
67
|
self._last_network_activity = time.monotonic()
|
|
68
|
+
self._proxy_extension_dir: Path | None = None
|
|
64
69
|
|
|
65
70
|
async def goto(self, url: str) -> None:
|
|
66
71
|
tab = await self._get_tab()
|
|
@@ -174,6 +179,18 @@ class PydollBrowserAdapter(BrowserAdapter):
|
|
|
174
179
|
await tab.take_screenshot(path)
|
|
175
180
|
return path
|
|
176
181
|
|
|
182
|
+
def configure_proxy(self, proxy_url: str) -> None:
|
|
183
|
+
"""Set an HTTP(S)/SOCKS5 proxy before Chromium is launched."""
|
|
184
|
+
|
|
185
|
+
if self._browser is not None or self._tab is not None:
|
|
186
|
+
raise PydollAdapterError("Proxy must be configured before the browser starts")
|
|
187
|
+
parsed = urlsplit(proxy_url)
|
|
188
|
+
if parsed.scheme not in {"http", "https", "socks5"} or not parsed.hostname:
|
|
189
|
+
raise PydollAdapterError("Proxy URL must include an HTTP(S) or SOCKS5 host")
|
|
190
|
+
if parsed.port is None or not 1 <= parsed.port <= 65535:
|
|
191
|
+
raise PydollAdapterError("Proxy URL must include a valid port")
|
|
192
|
+
self.config = replace(self.config, proxy_url=proxy_url)
|
|
193
|
+
|
|
177
194
|
async def close(self) -> None:
|
|
178
195
|
if self._tab is not None:
|
|
179
196
|
for callback_id in self._network_callback_ids:
|
|
@@ -191,6 +208,9 @@ class PydollBrowserAdapter(BrowserAdapter):
|
|
|
191
208
|
self._browser = None
|
|
192
209
|
self._network_tracking_enabled = False
|
|
193
210
|
self._inflight_requests.clear()
|
|
211
|
+
if self._proxy_extension_dir is not None:
|
|
212
|
+
shutil.rmtree(self._proxy_extension_dir, ignore_errors=True)
|
|
213
|
+
self._proxy_extension_dir = None
|
|
194
214
|
|
|
195
215
|
async def _get_tab(self) -> Any:
|
|
196
216
|
if self._tab is None:
|
|
@@ -220,10 +240,51 @@ class PydollBrowserAdapter(BrowserAdapter):
|
|
|
220
240
|
options.set_default_download_directory(str(self.config.download_directory))
|
|
221
241
|
for argument in self.config.arguments:
|
|
222
242
|
options.add_argument(argument)
|
|
243
|
+
if self.config.proxy_url is not None:
|
|
244
|
+
self._configure_proxy_options(options, self.config.proxy_url)
|
|
223
245
|
|
|
224
246
|
self._browser = Chrome(options=options)
|
|
225
247
|
self._tab = await self._browser.start()
|
|
226
248
|
|
|
249
|
+
def _configure_proxy_options(self, options: Any, proxy_url: str) -> None:
|
|
250
|
+
parsed = urlsplit(proxy_url)
|
|
251
|
+
host = parsed.hostname
|
|
252
|
+
port = parsed.port
|
|
253
|
+
if host is None or port is None:
|
|
254
|
+
raise PydollAdapterError("Proxy URL must include a host and port")
|
|
255
|
+
options.add_argument(f"--proxy-server={parsed.scheme}://{host}:{port}")
|
|
256
|
+
if parsed.username is not None or parsed.password is not None:
|
|
257
|
+
if parsed.username is None or parsed.password is None:
|
|
258
|
+
raise PydollAdapterError("Proxy URL must include both username and password")
|
|
259
|
+
self._proxy_extension_dir = self._create_proxy_auth_extension(
|
|
260
|
+
unquote(parsed.username),
|
|
261
|
+
unquote(parsed.password),
|
|
262
|
+
)
|
|
263
|
+
options.add_argument(f"--load-extension={self._proxy_extension_dir}")
|
|
264
|
+
|
|
265
|
+
@staticmethod
|
|
266
|
+
def _create_proxy_auth_extension(username: str, password: str) -> Path:
|
|
267
|
+
extension_dir = Path(tempfile.mkdtemp(prefix="crawlerflow-proxy-auth-"))
|
|
268
|
+
manifest = {
|
|
269
|
+
"manifest_version": 2,
|
|
270
|
+
"name": "Crawlerflow proxy authentication",
|
|
271
|
+
"version": "1.0",
|
|
272
|
+
"permissions": ["webRequest", "webRequestBlocking", "<all_urls>"],
|
|
273
|
+
"background": {"scripts": ["background.js"]},
|
|
274
|
+
}
|
|
275
|
+
background = (
|
|
276
|
+
"chrome.webRequest.onAuthRequired.addListener("
|
|
277
|
+
"function(details) { return {authCredentials: {username: "
|
|
278
|
+
f"{json.dumps(username)}, password: {json.dumps(password)}}}; }}, "
|
|
279
|
+
"{urls: ['<all_urls>']}, ['blocking']);"
|
|
280
|
+
)
|
|
281
|
+
(extension_dir / "manifest.json").write_text(
|
|
282
|
+
json.dumps(manifest),
|
|
283
|
+
encoding="utf-8",
|
|
284
|
+
)
|
|
285
|
+
(extension_dir / "background.js").write_text(background, encoding="utf-8")
|
|
286
|
+
return extension_dir
|
|
287
|
+
|
|
227
288
|
async def _query(self, selector: str, timeout_seconds: float | None = None) -> Any:
|
|
228
289
|
tab = await self._get_tab()
|
|
229
290
|
timeout = timeout_seconds or self.config.default_wait_timeout
|
|
@@ -22,6 +22,7 @@ class WorkflowContext:
|
|
|
22
22
|
workflow_name: str
|
|
23
23
|
base_path: Path
|
|
24
24
|
browser: BrowserAdapter | None = None
|
|
25
|
+
proxy_url: str | None = None
|
|
25
26
|
variables: dict[str, Any] = field(default_factory=dict)
|
|
26
27
|
outputs: dict[str, Any] = field(default_factory=dict)
|
|
27
28
|
cookies: dict[str, str] = field(default_factory=dict)
|
|
@@ -51,6 +52,7 @@ class WorkflowContext:
|
|
|
51
52
|
workflow_name=self.workflow_name,
|
|
52
53
|
base_path=self.base_path,
|
|
53
54
|
browser=self.browser,
|
|
55
|
+
proxy_url=self.proxy_url,
|
|
54
56
|
variables=dict(self.variables),
|
|
55
57
|
outputs=dict(self.outputs),
|
|
56
58
|
cookies=dict(self.cookies),
|
|
@@ -939,7 +939,10 @@ class EnrichHtmlLinksHttpStep(BaseStep[EnrichHtmlLinksHttpConfig]):
|
|
|
939
939
|
cache: dict[str, str | None] = {}
|
|
940
940
|
insertions: dict[int, list[tuple[str, str]]] = {}
|
|
941
941
|
last_finished_at: float | None = None
|
|
942
|
-
async with httpx.AsyncClient(
|
|
942
|
+
async with httpx.AsyncClient(
|
|
943
|
+
timeout=self.config.timeout,
|
|
944
|
+
proxy=context.proxy_url,
|
|
945
|
+
) as client:
|
|
943
946
|
for target_index, url in collector.links:
|
|
944
947
|
if url not in cache:
|
|
945
948
|
if last_finished_at is not None and self.config.delay > 0:
|
|
@@ -1092,6 +1095,7 @@ class ResolveLocationUrlStep(BaseStep[ResolveLocationUrlConfig]):
|
|
|
1092
1095
|
timeout=self.config.timeout,
|
|
1093
1096
|
follow_redirects=True,
|
|
1094
1097
|
max_redirects=self.config.max_redirects,
|
|
1098
|
+
proxy=context.proxy_url,
|
|
1095
1099
|
) as client:
|
|
1096
1100
|
response = await client.request(
|
|
1097
1101
|
"GET",
|
|
@@ -1307,6 +1311,7 @@ class EnrichJsonMapLocationsStep(BaseStep[EnrichJsonMapLocationsConfig]):
|
|
|
1307
1311
|
timeout=self.config.timeout,
|
|
1308
1312
|
follow_redirects=True,
|
|
1309
1313
|
max_redirects=self.config.max_redirects,
|
|
1314
|
+
proxy=context.proxy_url,
|
|
1310
1315
|
) as client:
|
|
1311
1316
|
|
|
1312
1317
|
async def resolve(url: str) -> tuple[str, dict[str, Any] | None]:
|
|
@@ -1435,7 +1440,10 @@ class HttpRequestStep(BaseStep[HttpRequestConfig]):
|
|
|
1435
1440
|
{"transport": "http", "method": method, "url": self.config.url},
|
|
1436
1441
|
)
|
|
1437
1442
|
try:
|
|
1438
|
-
async with httpx.AsyncClient(
|
|
1443
|
+
async with httpx.AsyncClient(
|
|
1444
|
+
timeout=self.config.timeout,
|
|
1445
|
+
proxy=context.proxy_url,
|
|
1446
|
+
) as client:
|
|
1439
1447
|
response = await client.request(
|
|
1440
1448
|
method,
|
|
1441
1449
|
self.config.url,
|
|
@@ -120,6 +120,56 @@ python -m crawlerflow list-plugins
|
|
|
120
120
|
|
|
121
121
|
The command displays each plugin name, import target, and owning Python distribution.
|
|
122
122
|
|
|
123
|
+
## Webshare proxy plugin
|
|
124
|
+
|
|
125
|
+
The repository also contains a separately installable `crawlerflow-webshare` plugin. It reads the
|
|
126
|
+
Webshare proxy list, selects one valid proxy randomly during workflow startup, and keeps that proxy
|
|
127
|
+
for the workflow's browser and direct HTTP requests:
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
python -m pip install -e . -e examples/plugins/webshare
|
|
131
|
+
export WEBSHARE_API_KEY="your-webshare-api-key"
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
```yaml
|
|
135
|
+
plugins:
|
|
136
|
+
- name: webshare
|
|
137
|
+
settings:
|
|
138
|
+
mode: direct
|
|
139
|
+
country_codes: [US]
|
|
140
|
+
valid_only: true
|
|
141
|
+
proxy_retry:
|
|
142
|
+
enabled: true
|
|
143
|
+
attempts_per_proxy: 3
|
|
144
|
+
max_proxies: 5
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
See the complete example in
|
|
148
|
+
[`examples/plugins/webshare/workflow.yaml`](https://github.com/mehmetemineker/crawlerflow/blob/main/examples/plugins/webshare/workflow.yaml).
|
|
149
|
+
|
|
150
|
+
For an HTTP-only variant without Pydoll, use
|
|
151
|
+
[`cloudflare-challenge-http.yaml`](https://github.com/mehmetemineker/crawlerflow/blob/main/examples/plugins/capsolver-webshare/cloudflare-challenge-http.yaml).
|
|
152
|
+
It fetches the challenge HTML and sends the returned `cf_clearance` cookie with a second direct
|
|
153
|
+
HTTP request through the same proxy.
|
|
154
|
+
|
|
155
|
+
## Cloudflare Challenge example
|
|
156
|
+
|
|
157
|
+
The combined CapSolver and Webshare example uses CapSolver's `AntiCloudflareTask`. It opens the
|
|
158
|
+
target with a fixed user agent, passes the same selected Webshare proxy to CapSolver, applies the
|
|
159
|
+
returned clearance cookies, and reloads the target page:
|
|
160
|
+
|
|
161
|
+
```bash
|
|
162
|
+
python -m pip install -e . -e examples/plugins/capsolver -e examples/plugins/webshare
|
|
163
|
+
export CAPSOLVER_API_KEY="your-capsolver-api-key"
|
|
164
|
+
export WEBSHARE_API_KEY="your-webshare-api-key"
|
|
165
|
+
python -m crawlerflow run examples/plugins/capsolver-webshare/cloudflare-challenge.yaml
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
Replace the authorized target URL before running. Because the example enables
|
|
169
|
+
`include_credentials: true` to pass the proxy to CapSolver, protect the generated result file;
|
|
170
|
+
it can contain a `cf_clearance` cookie. See
|
|
171
|
+
[`cloudflare-challenge.yaml`](https://github.com/mehmetemineker/crawlerflow/blob/main/examples/plugins/capsolver-webshare/cloudflare-challenge.yaml).
|
|
172
|
+
|
|
123
173
|
## Example package
|
|
124
174
|
|
|
125
175
|
`examples/plugins/example` is a separately installable reference package. It registers a custom
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# Crawlerflow CapSolver plugin
|
|
2
|
+
|
|
3
|
+
This separately installable plugin exposes CapSolver's generic task API to Crawlerflow. The task
|
|
4
|
+
object is passed through without a task-type-specific schema, so all CapSolver task types can be
|
|
5
|
+
used by setting the documented `task.type` and parameters. New CapSolver task types do not require
|
|
6
|
+
a plugin release.
|
|
7
|
+
|
|
8
|
+
Install it from the repository root:
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
python -m pip install -e . -e examples/plugins/capsolver
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Set the API key outside the workflow file:
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
export CAPSOLVER_API_KEY="your-capsolver-api-key"
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
The complete example is available at [`workflow.yaml`](workflow.yaml). It uses an authorized
|
|
21
|
+
reCAPTCHA v3 test target and writes the raw CapSolver response to `output/capsolver-result.json`.
|
|
22
|
+
|
|
23
|
+
Enable the plugin in YAML:
|
|
24
|
+
|
|
25
|
+
```yaml
|
|
26
|
+
version: 1
|
|
27
|
+
|
|
28
|
+
workflow:
|
|
29
|
+
name: capsolver-example
|
|
30
|
+
|
|
31
|
+
plugins:
|
|
32
|
+
- capsolver
|
|
33
|
+
|
|
34
|
+
steps:
|
|
35
|
+
- capsolver_solve:
|
|
36
|
+
task:
|
|
37
|
+
type: ImageToTextTask
|
|
38
|
+
body: BASE64_IMAGE_DATA
|
|
39
|
+
save_as: solution
|
|
40
|
+
|
|
41
|
+
- save_json:
|
|
42
|
+
path: output/capsolver-result.json
|
|
43
|
+
data: "{{solution}}"
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## Steps
|
|
47
|
+
|
|
48
|
+
| Step | API operation |
|
|
49
|
+
| --- | --- |
|
|
50
|
+
| `capsolver_create_task` | `createTask` |
|
|
51
|
+
| `capsolver_get_task_result` | `getTaskResult` |
|
|
52
|
+
| `capsolver_solve` | `createTask` plus polling `getTaskResult` |
|
|
53
|
+
| `capsolver_get_token` | `getToken` |
|
|
54
|
+
| `capsolver_get_balance` | `getBalance` |
|
|
55
|
+
| `capsolver_get_state` | CapSolver's documented service-state operation |
|
|
56
|
+
| `capsolver_request` | Any documented CapSolver API endpoint |
|
|
57
|
+
|
|
58
|
+
`capsolver_solve` handles both synchronous results, such as recognition tasks, and asynchronous
|
|
59
|
+
token tasks. Polling defaults to three seconds and is capped at 120 attempts, matching the
|
|
60
|
+
documented `getTaskResult` limit. Plugin settings can override `base_url`, `app_id`, `timeout`,
|
|
61
|
+
`poll_interval`, and `max_poll_attempts`; step-level `app_id` and `callback_url` take precedence.
|
|
62
|
+
|
|
63
|
+
The plugin returns CapSolver responses unchanged, including the task-specific `solution` object.
|
|
64
|
+
The generic `task` and `capsolver_request` payloads preserve forward compatibility with all current
|
|
65
|
+
and future task types and API fields. It does not inject solutions into a browser page; browser/page
|
|
66
|
+
integration remains the responsibility of the consuming workflow and target-site authorization.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "crawlerflow-capsolver"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "CapSolver plugin for Crawlerflow"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.12"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"crawlerflow>=0.2.0",
|
|
13
|
+
"httpx>=0.27",
|
|
14
|
+
"pydantic>=2.8",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.entry-points."crawlerflow.plugins"]
|
|
18
|
+
capsolver = "crawlerflow_capsolver:CapSolverPlugin"
|
|
19
|
+
|
|
20
|
+
[tool.hatch.build.targets.wheel]
|
|
21
|
+
packages = ["src/crawlerflow_capsolver"]
|