crawlerflow 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/CHANGELOG.md +23 -0
  2. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/PKG-INFO +1 -1
  3. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/__init__.py +1 -1
  4. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/browser/base.py +7 -0
  5. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/browser/pydoll.py +62 -1
  6. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/context.py +2 -0
  7. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/steps/utility.py +10 -2
  8. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/plugins.md +50 -0
  9. crawlerflow-0.3.0/examples/plugins/capsolver/README.md +66 -0
  10. crawlerflow-0.3.0/examples/plugins/capsolver/pyproject.toml +21 -0
  11. crawlerflow-0.3.0/examples/plugins/capsolver/src/crawlerflow_capsolver/__init__.py +503 -0
  12. crawlerflow-0.3.0/examples/plugins/capsolver/workflow.yaml +28 -0
  13. crawlerflow-0.3.0/examples/plugins/capsolver-webshare/cloudflare-challenge-http.yaml +68 -0
  14. crawlerflow-0.3.0/examples/plugins/capsolver-webshare/cloudflare-challenge.yaml +74 -0
  15. crawlerflow-0.3.0/examples/plugins/capsolver-webshare/workflow.yaml +54 -0
  16. crawlerflow-0.3.0/examples/plugins/webshare/README.md +63 -0
  17. crawlerflow-0.3.0/examples/plugins/webshare/pyproject.toml +21 -0
  18. crawlerflow-0.3.0/examples/plugins/webshare/src/crawlerflow_webshare/__init__.py +454 -0
  19. crawlerflow-0.3.0/examples/plugins/webshare/workflow.yaml +28 -0
  20. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/pyproject.toml +1 -1
  21. crawlerflow-0.3.0/tests/test_capsolver_plugin.py +96 -0
  22. crawlerflow-0.3.0/tests/test_combined_plugin_workflow.py +55 -0
  23. crawlerflow-0.3.0/tests/test_proxy_retry.py +141 -0
  24. crawlerflow-0.3.0/tests/test_webshare_plugin.py +127 -0
  25. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/.gitignore +0 -0
  26. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/LICENSE +0 -0
  27. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/README.md +0 -0
  28. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/__main__.py +0 -0
  29. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/browser/__init__.py +0 -0
  30. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/browser/factory.py +0 -0
  31. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/cli/__init__.py +0 -0
  32. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/cli/app.py +0 -0
  33. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/__init__.py +0 -0
  34. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/executor.py +0 -0
  35. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/registry.py +0 -0
  36. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/engine/runner.py +0 -0
  37. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/__init__.py +0 -0
  38. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/bus.py +0 -0
  39. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/console_logger.py +0 -0
  40. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/json_logger.py +0 -0
  41. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/events/models.py +0 -0
  42. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/expressions/__init__.py +0 -0
  43. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/expressions/conditions.py +0 -0
  44. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/expressions/engine.py +0 -0
  45. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/plugins/__init__.py +0 -0
  46. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/plugins/base.py +0 -0
  47. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/plugins/manager.py +0 -0
  48. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/py.typed +0 -0
  49. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/steps/__init__.py +0 -0
  50. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/steps/browser.py +0 -0
  51. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/steps/control.py +0 -0
  52. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/workflow/__init__.py +0 -0
  53. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/workflow/loader.py +0 -0
  54. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/crawlerflow/workflow/models.py +0 -0
  55. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/adr/0001-core-boundaries.md +0 -0
  56. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/architecture.md +0 -0
  57. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/control-flow.md +0 -0
  58. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/getting-started.md +0 -0
  59. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/html-output.md +0 -0
  60. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/http-requests.md +0 -0
  61. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/index.md +0 -0
  62. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/pydoll.md +0 -0
  63. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/retries-and-logging.md +0 -0
  64. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/docs/runtime-variables.md +0 -0
  65. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/basic.yaml +0 -0
  66. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/control-flow.yaml +0 -0
  67. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/loops-macros.yaml +0 -0
  68. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/plugins/example/README.md +0 -0
  69. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/plugins/example/pyproject.toml +0 -0
  70. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/plugins/example/src/crawlerflow_example_plugin/__init__.py +0 -0
  71. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/plugins/example/workflow.yaml +0 -0
  72. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/pydoll.yaml +0 -0
  73. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/examples/retry-logging.yaml +0 -0
  74. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_browser_steps.py +0 -0
  75. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_cli.py +0 -0
  76. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_conditions.py +0 -0
  77. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_control_flow.py +0 -0
  78. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_example_plugin.py +0 -0
  79. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_expressions.py +0 -0
  80. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_loops_and_macros.py +0 -0
  81. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_plugins.py +0 -0
  82. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_pydoll_adapter.py +0 -0
  83. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_retry_and_logging.py +0 -0
  84. {crawlerflow-0.2.0 → crawlerflow-0.3.0}/tests/test_workflow.py +0 -0
@@ -5,6 +5,28 @@ All notable changes to this project are documented in this file.
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and this project
6
6
  adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
7
 
8
+ ## [0.3.0] - 2026-09-17
9
+
10
+ ### Added
11
+
12
+ - Workflow-wide proxy support. `WorkflowContext` carries a `proxy_url` that survives parallel loop
13
+ forks, and the built-in `http_request`, `resolve_location_url`, `enrich_html_links_http`, and
14
+ `enrich_json_map_locations` steps route their requests through it.
15
+ - `BrowserAdapter.configure_proxy()` extension hook. Adapters that cannot proxy raise a clear
16
+ error, keeping the contract explicit.
17
+ - Pydoll proxy support for HTTP, HTTPS, and SOCKS5 URLs, including authenticated proxies through a
18
+ generated extension. The temporary extension directory is removed when the adapter closes.
19
+ - CapSolver reference plugin that solves captcha tasks and exposes the solution to later steps.
20
+ - Webshare reference plugin that selects proxies and publishes the active proxy to the workflow.
21
+ - Combined CapSolver and Webshare example workflows covering browser and HTTP-only Cloudflare
22
+ challenge flows.
23
+ - Plugin documentation covering proxy-aware plugins and the new adapter hook.
24
+
25
+ ### Fixed
26
+
27
+ - Disabled Jekyll processing for the published documentation site so workflow expression examples
28
+ are no longer parsed as Liquid templates.
29
+
8
30
  ## [0.2.0] - 2026-08-18
9
31
 
10
32
  ### Changed
@@ -36,5 +58,6 @@ adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
36
58
  - Plugin API with typed settings, lifecycle hooks, steps, filters, and subscribers.
37
59
  - `run`, `validate`, `list-steps`, `list-plugins`, and `doctor` CLI commands.
38
60
 
61
+ [0.3.0]: https://github.com/mehmetemineker/crawlerflow/releases/tag/v0.3.0
39
62
  [0.2.0]: https://github.com/mehmetemineker/crawlerflow/releases/tag/v0.2.0
40
63
  [0.1.0]: https://github.com/mehmetemineker/crawlerflow/releases/tag/v0.1.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: crawlerflow
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Declarative YAML workflow engine for browser automation and web scraping
5
5
  Project-URL: Homepage, https://github.com/mehmetemineker/crawlerflow
6
6
  Project-URL: Repository, https://github.com/mehmetemineker/crawlerflow
@@ -5,5 +5,5 @@ from __future__ import annotations
5
5
  from crawlerflow.engine.runner import WorkflowRunner
6
6
 
7
7
  __all__ = ["WorkflowRunner"]
8
- __version__ = "0.2.0"
8
+ __version__ = "0.3.0"
9
9
 
@@ -66,6 +66,13 @@ class BrowserAdapter(ABC):
66
66
  @abstractmethod
67
67
  async def screenshot(self, path: Path) -> Path: ...
68
68
 
69
+ def configure_proxy(self, proxy_url: str) -> None:
70
+ """Configure a proxy before the browser session starts, if supported."""
71
+
72
+ raise RuntimeError(
73
+ f"Browser adapter {type(self).__name__} does not support proxy configuration"
74
+ )
75
+
69
76
  async def close(self) -> None:
70
77
  """Release browser resources when an adapter owns them."""
71
78
 
@@ -6,11 +6,14 @@ import asyncio
6
6
  import json
7
7
  import logging
8
8
  import math
9
+ import shutil
10
+ import tempfile
9
11
  import time
10
12
  from collections.abc import Mapping
11
- from dataclasses import dataclass
13
+ from dataclasses import dataclass, replace
12
14
  from pathlib import Path
13
15
  from typing import Any
16
+ from urllib.parse import unquote, urlsplit
14
17
 
15
18
  from crawlerflow.browser.base import BrowserAdapter, BrowserResponse
16
19
 
@@ -40,6 +43,7 @@ class PydollBrowserConfig:
40
43
  default_wait_timeout: float = 10
41
44
  network_idle_period: float = 0.5
42
45
  download_directory: Path | None = None
46
+ proxy_url: str | None = None
43
47
 
44
48
 
45
49
  class PydollBrowserAdapter(BrowserAdapter):
@@ -61,6 +65,7 @@ class PydollBrowserAdapter(BrowserAdapter):
61
65
  self._network_callback_ids: list[int] = []
62
66
  self._inflight_requests: set[str] = set()
63
67
  self._last_network_activity = time.monotonic()
68
+ self._proxy_extension_dir: Path | None = None
64
69
 
65
70
  async def goto(self, url: str) -> None:
66
71
  tab = await self._get_tab()
@@ -174,6 +179,18 @@ class PydollBrowserAdapter(BrowserAdapter):
174
179
  await tab.take_screenshot(path)
175
180
  return path
176
181
 
182
+ def configure_proxy(self, proxy_url: str) -> None:
183
+ """Set an HTTP(S)/SOCKS5 proxy before Chromium is launched."""
184
+
185
+ if self._browser is not None or self._tab is not None:
186
+ raise PydollAdapterError("Proxy must be configured before the browser starts")
187
+ parsed = urlsplit(proxy_url)
188
+ if parsed.scheme not in {"http", "https", "socks5"} or not parsed.hostname:
189
+ raise PydollAdapterError("Proxy URL must include an HTTP(S) or SOCKS5 host")
190
+ if parsed.port is None or not 1 <= parsed.port <= 65535:
191
+ raise PydollAdapterError("Proxy URL must include a valid port")
192
+ self.config = replace(self.config, proxy_url=proxy_url)
193
+
177
194
  async def close(self) -> None:
178
195
  if self._tab is not None:
179
196
  for callback_id in self._network_callback_ids:
@@ -191,6 +208,9 @@ class PydollBrowserAdapter(BrowserAdapter):
191
208
  self._browser = None
192
209
  self._network_tracking_enabled = False
193
210
  self._inflight_requests.clear()
211
+ if self._proxy_extension_dir is not None:
212
+ shutil.rmtree(self._proxy_extension_dir, ignore_errors=True)
213
+ self._proxy_extension_dir = None
194
214
 
195
215
  async def _get_tab(self) -> Any:
196
216
  if self._tab is None:
@@ -220,10 +240,51 @@ class PydollBrowserAdapter(BrowserAdapter):
220
240
  options.set_default_download_directory(str(self.config.download_directory))
221
241
  for argument in self.config.arguments:
222
242
  options.add_argument(argument)
243
+ if self.config.proxy_url is not None:
244
+ self._configure_proxy_options(options, self.config.proxy_url)
223
245
 
224
246
  self._browser = Chrome(options=options)
225
247
  self._tab = await self._browser.start()
226
248
 
249
+ def _configure_proxy_options(self, options: Any, proxy_url: str) -> None:
250
+ parsed = urlsplit(proxy_url)
251
+ host = parsed.hostname
252
+ port = parsed.port
253
+ if host is None or port is None:
254
+ raise PydollAdapterError("Proxy URL must include a host and port")
255
+ options.add_argument(f"--proxy-server={parsed.scheme}://{host}:{port}")
256
+ if parsed.username is not None or parsed.password is not None:
257
+ if parsed.username is None or parsed.password is None:
258
+ raise PydollAdapterError("Proxy URL must include both username and password")
259
+ self._proxy_extension_dir = self._create_proxy_auth_extension(
260
+ unquote(parsed.username),
261
+ unquote(parsed.password),
262
+ )
263
+ options.add_argument(f"--load-extension={self._proxy_extension_dir}")
264
+
265
+ @staticmethod
266
+ def _create_proxy_auth_extension(username: str, password: str) -> Path:
267
+ extension_dir = Path(tempfile.mkdtemp(prefix="crawlerflow-proxy-auth-"))
268
+ manifest = {
269
+ "manifest_version": 2,
270
+ "name": "Crawlerflow proxy authentication",
271
+ "version": "1.0",
272
+ "permissions": ["webRequest", "webRequestBlocking", "<all_urls>"],
273
+ "background": {"scripts": ["background.js"]},
274
+ }
275
+ background = (
276
+ "chrome.webRequest.onAuthRequired.addListener("
277
+ "function(details) { return {authCredentials: {username: "
278
+ f"{json.dumps(username)}, password: {json.dumps(password)}}}; }}, "
279
+ "{urls: ['<all_urls>']}, ['blocking']);"
280
+ )
281
+ (extension_dir / "manifest.json").write_text(
282
+ json.dumps(manifest),
283
+ encoding="utf-8",
284
+ )
285
+ (extension_dir / "background.js").write_text(background, encoding="utf-8")
286
+ return extension_dir
287
+
227
288
  async def _query(self, selector: str, timeout_seconds: float | None = None) -> Any:
228
289
  tab = await self._get_tab()
229
290
  timeout = timeout_seconds or self.config.default_wait_timeout
@@ -22,6 +22,7 @@ class WorkflowContext:
22
22
  workflow_name: str
23
23
  base_path: Path
24
24
  browser: BrowserAdapter | None = None
25
+ proxy_url: str | None = None
25
26
  variables: dict[str, Any] = field(default_factory=dict)
26
27
  outputs: dict[str, Any] = field(default_factory=dict)
27
28
  cookies: dict[str, str] = field(default_factory=dict)
@@ -51,6 +52,7 @@ class WorkflowContext:
51
52
  workflow_name=self.workflow_name,
52
53
  base_path=self.base_path,
53
54
  browser=self.browser,
55
+ proxy_url=self.proxy_url,
54
56
  variables=dict(self.variables),
55
57
  outputs=dict(self.outputs),
56
58
  cookies=dict(self.cookies),
@@ -939,7 +939,10 @@ class EnrichHtmlLinksHttpStep(BaseStep[EnrichHtmlLinksHttpConfig]):
939
939
  cache: dict[str, str | None] = {}
940
940
  insertions: dict[int, list[tuple[str, str]]] = {}
941
941
  last_finished_at: float | None = None
942
- async with httpx.AsyncClient(timeout=self.config.timeout) as client:
942
+ async with httpx.AsyncClient(
943
+ timeout=self.config.timeout,
944
+ proxy=context.proxy_url,
945
+ ) as client:
943
946
  for target_index, url in collector.links:
944
947
  if url not in cache:
945
948
  if last_finished_at is not None and self.config.delay > 0:
@@ -1092,6 +1095,7 @@ class ResolveLocationUrlStep(BaseStep[ResolveLocationUrlConfig]):
1092
1095
  timeout=self.config.timeout,
1093
1096
  follow_redirects=True,
1094
1097
  max_redirects=self.config.max_redirects,
1098
+ proxy=context.proxy_url,
1095
1099
  ) as client:
1096
1100
  response = await client.request(
1097
1101
  "GET",
@@ -1307,6 +1311,7 @@ class EnrichJsonMapLocationsStep(BaseStep[EnrichJsonMapLocationsConfig]):
1307
1311
  timeout=self.config.timeout,
1308
1312
  follow_redirects=True,
1309
1313
  max_redirects=self.config.max_redirects,
1314
+ proxy=context.proxy_url,
1310
1315
  ) as client:
1311
1316
 
1312
1317
  async def resolve(url: str) -> tuple[str, dict[str, Any] | None]:
@@ -1435,7 +1440,10 @@ class HttpRequestStep(BaseStep[HttpRequestConfig]):
1435
1440
  {"transport": "http", "method": method, "url": self.config.url},
1436
1441
  )
1437
1442
  try:
1438
- async with httpx.AsyncClient(timeout=self.config.timeout) as client:
1443
+ async with httpx.AsyncClient(
1444
+ timeout=self.config.timeout,
1445
+ proxy=context.proxy_url,
1446
+ ) as client:
1439
1447
  response = await client.request(
1440
1448
  method,
1441
1449
  self.config.url,
@@ -120,6 +120,56 @@ python -m crawlerflow list-plugins
120
120
 
121
121
  The command displays each plugin name, import target, and owning Python distribution.
122
122
 
123
+ ## Webshare proxy plugin
124
+
125
+ The repository also contains a separately installable `crawlerflow-webshare` plugin. It reads the
126
+ Webshare proxy list, selects one valid proxy randomly during workflow startup, and keeps that proxy
127
+ for the workflow's browser and direct HTTP requests:
128
+
129
+ ```bash
130
+ python -m pip install -e . -e examples/plugins/webshare
131
+ export WEBSHARE_API_KEY="your-webshare-api-key"
132
+ ```
133
+
134
+ ```yaml
135
+ plugins:
136
+ - name: webshare
137
+ settings:
138
+ mode: direct
139
+ country_codes: [US]
140
+ valid_only: true
141
+ proxy_retry:
142
+ enabled: true
143
+ attempts_per_proxy: 3
144
+ max_proxies: 5
145
+ ```
146
+
147
+ See the complete example in
148
+ [`examples/plugins/webshare/workflow.yaml`](https://github.com/mehmetemineker/crawlerflow/blob/main/examples/plugins/webshare/workflow.yaml).
149
+
150
+ For an HTTP-only variant without Pydoll, use
151
+ [`cloudflare-challenge-http.yaml`](https://github.com/mehmetemineker/crawlerflow/blob/main/examples/plugins/capsolver-webshare/cloudflare-challenge-http.yaml).
152
+ It fetches the challenge HTML and sends the returned `cf_clearance` cookie with a second direct
153
+ HTTP request through the same proxy.
154
+
155
+ ## Cloudflare Challenge example
156
+
157
+ The combined CapSolver and Webshare example uses CapSolver's `AntiCloudflareTask`. It opens the
158
+ target with a fixed user agent, passes the same selected Webshare proxy to CapSolver, applies the
159
+ returned clearance cookies, and reloads the target page:
160
+
161
+ ```bash
162
+ python -m pip install -e . -e examples/plugins/capsolver -e examples/plugins/webshare
163
+ export CAPSOLVER_API_KEY="your-capsolver-api-key"
164
+ export WEBSHARE_API_KEY="your-webshare-api-key"
165
+ python -m crawlerflow run examples/plugins/capsolver-webshare/cloudflare-challenge.yaml
166
+ ```
167
+
168
+ Replace the authorized target URL before running. Because the example enables
169
+ `include_credentials: true` to pass the proxy to CapSolver, protect the generated result file;
170
+ it can contain a `cf_clearance` cookie. See
171
+ [`cloudflare-challenge.yaml`](https://github.com/mehmetemineker/crawlerflow/blob/main/examples/plugins/capsolver-webshare/cloudflare-challenge.yaml).
172
+
123
173
  ## Example package
124
174
 
125
175
  `examples/plugins/example` is a separately installable reference package. It registers a custom
@@ -0,0 +1,66 @@
1
+ # Crawlerflow CapSolver plugin
2
+
3
+ This separately installable plugin exposes CapSolver's generic task API to Crawlerflow. The task
4
+ object is passed through without a task-type-specific schema, so all CapSolver task types can be
5
+ used by setting the documented `task.type` and parameters. New CapSolver task types do not require
6
+ a plugin release.
7
+
8
+ Install it from the repository root:
9
+
10
+ ```bash
11
+ python -m pip install -e . -e examples/plugins/capsolver
12
+ ```
13
+
14
+ Set the API key outside the workflow file:
15
+
16
+ ```bash
17
+ export CAPSOLVER_API_KEY="your-capsolver-api-key"
18
+ ```
19
+
20
+ The complete example is available at [`workflow.yaml`](workflow.yaml). It uses an authorized
21
+ reCAPTCHA v3 test target and writes the raw CapSolver response to `output/capsolver-result.json`.
22
+
23
+ Enable the plugin in YAML:
24
+
25
+ ```yaml
26
+ version: 1
27
+
28
+ workflow:
29
+ name: capsolver-example
30
+
31
+ plugins:
32
+ - capsolver
33
+
34
+ steps:
35
+ - capsolver_solve:
36
+ task:
37
+ type: ImageToTextTask
38
+ body: BASE64_IMAGE_DATA
39
+ save_as: solution
40
+
41
+ - save_json:
42
+ path: output/capsolver-result.json
43
+ data: "{{solution}}"
44
+ ```
45
+
46
+ ## Steps
47
+
48
+ | Step | API operation |
49
+ | --- | --- |
50
+ | `capsolver_create_task` | `createTask` |
51
+ | `capsolver_get_task_result` | `getTaskResult` |
52
+ | `capsolver_solve` | `createTask` plus polling `getTaskResult` |
53
+ | `capsolver_get_token` | `getToken` |
54
+ | `capsolver_get_balance` | `getBalance` |
55
+ | `capsolver_get_state` | CapSolver's documented service-state operation |
56
+ | `capsolver_request` | Any documented CapSolver API endpoint |
57
+
58
+ `capsolver_solve` handles both synchronous results, such as recognition tasks, and asynchronous
59
+ token tasks. Polling defaults to three seconds and is capped at 120 attempts, matching the
60
+ documented `getTaskResult` limit. Plugin settings can override `base_url`, `app_id`, `timeout`,
61
+ `poll_interval`, and `max_poll_attempts`; step-level `app_id` and `callback_url` take precedence.
62
+
63
+ The plugin returns CapSolver responses unchanged, including the task-specific `solution` object.
64
+ The generic `task` and `capsolver_request` payloads preserve forward compatibility with all current
65
+ and future task types and API fields. It does not inject solutions into a browser page; browser/page
66
+ integration remains the responsibility of the consuming workflow and target-site authorization.
@@ -0,0 +1,21 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "crawlerflow-capsolver"
7
+ version = "0.1.0"
8
+ description = "CapSolver plugin for Crawlerflow"
9
+ readme = "README.md"
10
+ requires-python = ">=3.12"
11
+ dependencies = [
12
+ "crawlerflow>=0.2.0",
13
+ "httpx>=0.27",
14
+ "pydantic>=2.8",
15
+ ]
16
+
17
+ [project.entry-points."crawlerflow.plugins"]
18
+ capsolver = "crawlerflow_capsolver:CapSolverPlugin"
19
+
20
+ [tool.hatch.build.targets.wheel]
21
+ packages = ["src/crawlerflow_capsolver"]