searxNcrawl 0.31.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- searxncrawl-0.31.0/LICENSE +21 -0
- searxncrawl-0.31.0/PKG-INFO +390 -0
- searxncrawl-0.31.0/README.md +356 -0
- searxncrawl-0.31.0/crawler/__init__.py +339 -0
- searxncrawl-0.31.0/crawler/auth.py +107 -0
- searxncrawl-0.31.0/crawler/browser_setup.py +96 -0
- searxncrawl-0.31.0/crawler/builder.py +216 -0
- searxncrawl-0.31.0/crawler/cli.py +975 -0
- searxncrawl-0.31.0/crawler/config.py +244 -0
- searxncrawl-0.31.0/crawler/document.py +31 -0
- searxncrawl-0.31.0/crawler/env.py +65 -0
- searxncrawl-0.31.0/crawler/markdown_dedup.py +108 -0
- searxncrawl-0.31.0/crawler/mcp_server.py +753 -0
- searxncrawl-0.31.0/crawler/references.py +62 -0
- searxncrawl-0.31.0/crawler/session_capture.py +364 -0
- searxncrawl-0.31.0/crawler/site.py +269 -0
- searxncrawl-0.31.0/pyproject.toml +81 -0
- searxncrawl-0.31.0/searxNcrawl.egg-info/PKG-INFO +390 -0
- searxncrawl-0.31.0/searxNcrawl.egg-info/SOURCES.txt +48 -0
- searxncrawl-0.31.0/searxNcrawl.egg-info/dependency_links.txt +1 -0
- searxncrawl-0.31.0/searxNcrawl.egg-info/entry_points.txt +6 -0
- searxncrawl-0.31.0/searxNcrawl.egg-info/requires.txt +10 -0
- searxncrawl-0.31.0/searxNcrawl.egg-info/top_level.txt +1 -0
- searxncrawl-0.31.0/setup.cfg +4 -0
- searxncrawl-0.31.0/tests/test_auth_core.py +166 -0
- searxncrawl-0.31.0/tests/test_browser_setup.py +165 -0
- searxncrawl-0.31.0/tests/test_builder.py +83 -0
- searxncrawl-0.31.0/tests/test_cli.py +494 -0
- searxncrawl-0.31.0/tests/test_config.py +12 -0
- searxncrawl-0.31.0/tests/test_config_loading.py +154 -0
- searxncrawl-0.31.0/tests/test_content_wait.py +121 -0
- searxncrawl-0.31.0/tests/test_cors.py +184 -0
- searxncrawl-0.31.0/tests/test_init.py +109 -0
- searxncrawl-0.31.0/tests/test_markdown_dedup.py +253 -0
- searxncrawl-0.31.0/tests/test_mcp_server.py +197 -0
- searxncrawl-0.31.0/tests/test_session_capture.py +297 -0
- searxncrawl-0.31.0/tests/test_timeout.py +220 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 DDM – Das Digitale Momentum GmbH & Co KG
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,390 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: searxNcrawl
|
|
3
|
+
Version: 0.31.0
|
|
4
|
+
Summary: searxNcrawl web crawler with markdown extraction - no DB, no enrichment, just crawling.
|
|
5
|
+
Author: DDM – Das Digitale Momentum GmbH & Co KG
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://www.das-digitale-momentum.de/en/open-source/#searxncrawl
|
|
8
|
+
Project-URL: Repository, https://github.com/DasDigitaleMomentum/searxNcrawl
|
|
9
|
+
Project-URL: Issues, https://github.com/DasDigitaleMomentum/searxNcrawl/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/DasDigitaleMomentum/searxNcrawl/blob/main/CHANGELOG.md
|
|
11
|
+
Keywords: mcp,mcp-server,searxng,crawl4ai,web-crawler,web-search,markdown,llm-tools
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Framework :: AsyncIO
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
20
|
+
Classifier: Topic :: Text Processing :: Markup :: Markdown
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: crawl4ai>=0.7.4
|
|
25
|
+
Requires-Dist: tldextract>=5.1.2
|
|
26
|
+
Requires-Dist: playwright>=1.40.0
|
|
27
|
+
Requires-Dist: fastmcp>=3.4.3
|
|
28
|
+
Requires-Dist: httpx>=0.27.0
|
|
29
|
+
Requires-Dist: python-dotenv>=1.0.1
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: pytest>=8.0.0; extra == "dev"
|
|
32
|
+
Requires-Dist: pytest-asyncio>=0.23.0; extra == "dev"
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
|
|
35
|
+
# searxNcrawl
|
|
36
|
+
|
|
37
|
+
MCP server and CLI toolkit for web search and crawling, built on [Crawl4AI](https://github.com/unclecode/crawl4ai) and [SearXNG](https://github.com/searxng/searxng).
|
|
38
|
+
|
|
39
|
+
Published at [github.com/DasDigitaleMomentum/searxNcrawl](https://github.com/DasDigitaleMomentum/searxNcrawl) — maintained by **DDM – Das Digitale Momentum GmbH & Co KG**. Successor to `searxng-mcp`.
|
|
40
|
+
|
|
41
|
+
<!-- mcp-name: io.github.DasDigitaleMomentum/searxncrawl -->
|
|
42
|
+
|
|
43
|
+
## Quick Start
|
|
44
|
+
|
|
45
|
+
Pick your setup:
|
|
46
|
+
|
|
47
|
+
### Docker Compose
|
|
48
|
+
|
|
49
|
+
MCP server with Playwright/Chromium, ready in one command. SearXNG required separately for search.
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
cp .env.example .env # set SEARXNG_URL to your SearXNG instance
|
|
53
|
+
docker compose up --build
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
➜ MCP server at `http://localhost:9555/mcp`
|
|
57
|
+
|
|
58
|
+
### uvx (no clone)
|
|
59
|
+
|
|
60
|
+
Run the MCP server from [PyPI](https://pypi.org/project/searxncrawl/), no clone or virtualenv. Chromium is downloaded automatically on the first crawl.
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
SEARXNG_URL=http://your-searxng:8888 uvx searxncrawl
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
The latest development version runs straight from GitHub with `uvx --from git+https://github.com/DasDigitaleMomentum/searxNcrawl searxncrawl`.
|
|
67
|
+
|
|
68
|
+
### pip (standalone)
|
|
69
|
+
|
|
70
|
+
CLI tools, Python API, and MCP server. SearXNG required for search.
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
python -m venv .venv && source .venv/bin/activate
|
|
74
|
+
pip install -e .
|
|
75
|
+
playwright install chromium
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### uv (standalone)
|
|
79
|
+
|
|
80
|
+
Same capabilities as pip.
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
uv sync
|
|
84
|
+
uv run playwright install chromium
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
### What you get
|
|
88
|
+
|
|
89
|
+
| Feature | Docker Compose | pip / uv |
|
|
90
|
+
| ----------------------- | -------------- | --------- |
|
|
91
|
+
| MCP Server (STDIO) | — | ✅ |
|
|
92
|
+
| MCP Server (HTTP) | ✅ | ✅ |
|
|
93
|
+
| Web Crawl | ✅ | ✅ |
|
|
94
|
+
| Web Search | ✅¹ | ✅¹ |
|
|
95
|
+
| CLI Tools | via `exec`² | ✅ |
|
|
96
|
+
| Python API | — | ✅ |
|
|
97
|
+
| CORS (HTTP) | ✅ | ✅ |
|
|
98
|
+
|
|
99
|
+
¹ Requires a SearXNG instance. ² `docker compose exec searxncrawl crawl ...`
|
|
100
|
+
|
|
101
|
+
## Features
|
|
102
|
+
|
|
103
|
+
### Crawling
|
|
104
|
+
- Single page, multi-page, and **site crawling** (DFS with depth/page limits)
|
|
105
|
+
- Production-tested extraction config optimized for documentation sites
|
|
106
|
+
- Configurable timeouts with graceful error handling
|
|
107
|
+
|
|
108
|
+
### Content Quality
|
|
109
|
+
- **Markdown deduplication** — `exact` (default) removes repeated blocks, `off` disables it
|
|
110
|
+
- **Link removal** — strip all links for cleaner LLM context (`--remove-links`)
|
|
111
|
+
- **Dedup guardrails** — non-destructive metadata signals when removal is unusually aggressive
|
|
112
|
+
|
|
113
|
+
### Web Search
|
|
114
|
+
- SearXNG metasearch integration (privacy-respecting)
|
|
115
|
+
- Configurable language, time range, categories, engines, safe search
|
|
116
|
+
|
|
117
|
+
### MCP Server
|
|
118
|
+
- **STDIO transport** — for MCP harnesses (Zed, opencode, VS Code, Claude Code, etc.)
|
|
119
|
+
- **HTTP transport** — for remote access and browser clients
|
|
120
|
+
- **CORS support** — configurable origins for browser-based MCP clients
|
|
121
|
+
- Noise-free startup with UTF-8 encoding (cross-platform, incl. Windows)
|
|
122
|
+
|
|
123
|
+
### CLI Tools
|
|
124
|
+
- `crawl` — crawl pages from the command line
|
|
125
|
+
- `search` — search the web via SearXNG
|
|
126
|
+
- `crawl-capture` — session capture for authenticated crawling
|
|
127
|
+
|
|
128
|
+
## Installation
|
|
129
|
+
|
|
130
|
+
### Docker Compose
|
|
131
|
+
|
|
132
|
+
The Compose stack includes searxNcrawl + Playwright/Chromium. SearXNG must be provided separately.
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
cp .env.example .env
|
|
136
|
+
# Edit .env: set SEARXNG_URL to your SearXNG instance
|
|
137
|
+
docker compose up --build
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
| Variable | Default | Description |
|
|
141
|
+
| ----------- | ------------------------- | ------------------------------------------------------------- |
|
|
142
|
+
| `MCP_PORT` | `9555` | MCP server HTTP port |
|
|
143
|
+
| `LOG_LEVEL` | `INFO` | MCP server log level (DEBUG, INFO, WARNING, ERROR, CRITICAL) |
|
|
144
|
+
| `PLAYWRIGHT_AUTO_INSTALL` | `true` | Download Playwright's Chromium automatically before the first browser launch if it is missing. Set to `false` where browsers are provisioned separately |
|
|
145
|
+
| `FASTMCP_HTTP_ALLOWED_HOSTS` | (FastMCP secure defaults) | JSON list of trusted HTTP Host headers, for example `["mcp.example.com"]` |
|
|
146
|
+
|
|
147
|
+
The MCP server is available at `http://localhost:9555/mcp`.
|
|
148
|
+
|
|
149
|
+
### pip
|
|
150
|
+
|
|
151
|
+
```bash
|
|
152
|
+
cd searxNcrawl
|
|
153
|
+
python -m venv .venv
|
|
154
|
+
source .venv/bin/activate
|
|
155
|
+
pip install -e .
|
|
156
|
+
playwright install chromium
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
### uv
|
|
160
|
+
|
|
161
|
+
```bash
|
|
162
|
+
cd searxNcrawl
|
|
163
|
+
uv sync
|
|
164
|
+
uv run playwright install chromium
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
### SearXNG (search feature)
|
|
168
|
+
|
|
169
|
+
The `search` tool and CLI command require a SearXNG instance with **JSON output enabled** (`search.formats` in `settings.yml`). For all setups you need your own instance — self-hosting is recommended over public instances (rate limits).
|
|
170
|
+
|
|
171
|
+
**Environment variables:**
|
|
172
|
+
|
|
173
|
+
| Variable | Example / Recommended | Description |
|
|
174
|
+
| ---------------------- | ------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- |
|
|
175
|
+
| `SEARXNG_URL` | `http://localhost:8888` | SearXNG instance URL |
|
|
176
|
+
| `SEARXNG_USERNAME` | (none) | Optional basic auth user |
|
|
177
|
+
| `SEARXNG_PASSWORD` | (none) | Optional basic auth pass |
|
|
178
|
+
| `SEARCH_RESULT_FIELDS` | `title,url,content,publishedDate` | Comma-separated result fields. Unset = all SearXNG fields. Available: title, url, content, publishedDate, engine, score, category, img_src, thumbnail |
|
|
179
|
+
|
|
180
|
+
Example `.env`:
|
|
181
|
+
```bash
|
|
182
|
+
SEARXNG_URL=http://localhost:8888
|
|
183
|
+
SEARCH_RESULT_FIELDS=title,url,content,publishedDate
|
|
184
|
+
LOG_LEVEL=INFO
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
**Config file search order** (CLI tools only):
|
|
188
|
+
|
|
189
|
+
1. `./.env` — current directory
|
|
190
|
+
2. `~/.config/searxncrawl/.env` — user config
|
|
191
|
+
|
|
192
|
+
If no `.env` exists, `.env.example` is auto-copied to the user config path.
|
|
193
|
+
|
|
194
|
+
## Usage
|
|
195
|
+
|
|
196
|
+
### MCP Server
|
|
197
|
+
|
|
198
|
+
#### Start the server
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
# STDIO transport (for MCP harnesses)
|
|
202
|
+
python -m crawler.mcp_server
|
|
203
|
+
|
|
204
|
+
# HTTP transport
|
|
205
|
+
python -m crawler.mcp_server --transport http --port 8000
|
|
206
|
+
|
|
207
|
+
# HTTP exposed through a specific public hostname
|
|
208
|
+
python -m crawler.mcp_server --transport http --host 0.0.0.0 --allowed-hosts "mcp.example.com"
|
|
209
|
+
|
|
210
|
+
# HTTP with CORS
|
|
211
|
+
python -m crawler.mcp_server --transport http --allowed-hosts "mcp.example.com" --cors-origins "https://app.example.com"
|
|
212
|
+
|
|
213
|
+
# Docker (HTTP only)
|
|
214
|
+
docker compose up --build
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
#### MCP client configuration
|
|
218
|
+
|
|
219
|
+
**With uvx (no clone, no venv):**
|
|
220
|
+
|
|
221
|
+
```json
|
|
222
|
+
{
|
|
223
|
+
"mcpServers": {
|
|
224
|
+
"crawler": {
|
|
225
|
+
"command": "uvx",
|
|
226
|
+
"args": ["searxncrawl"],
|
|
227
|
+
"env": { "SEARXNG_URL": "http://your-searxng:8888" }
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
```
|
|
232
|
+
|
|
233
|
+
The first crawl downloads Playwright's Chromium once (about 550 MB on disk); later starts reuse it. On Linux hosts without the browser's system libraries, run `uvx --from searxncrawl playwright install --with-deps chromium` once.
|
|
234
|
+
|
|
235
|
+
**Python with venv:**
|
|
236
|
+
|
|
237
|
+
```json
|
|
238
|
+
{
|
|
239
|
+
"mcpServers": {
|
|
240
|
+
"crawler": {
|
|
241
|
+
"command": "python",
|
|
242
|
+
"args": ["-m", "crawler.mcp_server"],
|
|
243
|
+
"cwd": "/path/to/searxNcrawl",
|
|
244
|
+
"env": { "SEARXNG_URL": "http://your-searxng:8888" }
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
```
|
|
249
|
+
|
|
250
|
+
**With uv (no manual venv):**
|
|
251
|
+
|
|
252
|
+
```json
|
|
253
|
+
{
|
|
254
|
+
"mcpServers": {
|
|
255
|
+
"crawler": {
|
|
256
|
+
"command": "uv",
|
|
257
|
+
"args": ["run", "--directory", "/path/to/searxNcrawl", "python", "-m", "crawler.mcp_server"],
|
|
258
|
+
"env": { "SEARXNG_URL": "http://your-searxng:8888" }
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
```
|
|
263
|
+
|
|
264
|
+
**Docker (HTTP endpoint):**
|
|
265
|
+
|
|
266
|
+
```json
|
|
267
|
+
{
|
|
268
|
+
"mcpServers": {
|
|
269
|
+
"crawler": {
|
|
270
|
+
"url": "http://localhost:9555/mcp"
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
```
|
|
275
|
+
|
|
276
|
+
#### CORS
|
|
277
|
+
|
|
278
|
+
FastMCP validates the HTTP `Host` header independently of the address on which
|
|
279
|
+
the server listens. For remote access, allow the exact externally visible Host
|
|
280
|
+
header with a comma-separated CLI value:
|
|
281
|
+
|
|
282
|
+
```bash
|
|
283
|
+
crawl-mcp --transport http --host 0.0.0.0 --allowed-hosts "mcp.example.com,mcp.internal.example"
|
|
284
|
+
```
|
|
285
|
+
|
|
286
|
+
Alternatively, use FastMCP's environment setting. It uses JSON-list syntax:
|
|
287
|
+
|
|
288
|
+
```bash
|
|
289
|
+
FASTMCP_HTTP_ALLOWED_HOSTS='["mcp.example.com"]' crawl-mcp --transport http --host 0.0.0.0
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
Browser Origin validation and CORS response headers are separate from Host
|
|
293
|
+
validation. `--cors-origins` configures both FastMCP's Origin guard and the CORS
|
|
294
|
+
middleware using the same normalized, comma-separated values:
|
|
295
|
+
|
|
296
|
+
```bash
|
|
297
|
+
crawl-mcp --transport http --cors-origins "http://localhost:3000,https://myapp.com"
|
|
298
|
+
crawl-mcp --transport http --cors-origins "*" # all origins — local dev only
|
|
299
|
+
```
|
|
300
|
+
|
|
301
|
+
Omitting either allowlist preserves FastMCP's secure defaults (and permits the
|
|
302
|
+
upstream environment setting to apply). A value of `*` for Hosts or Origins is
|
|
303
|
+
an explicit opt-in to broad access and should only be used when that security
|
|
304
|
+
trade-off is intentional. Without `--cors-origins`, no CORS headers are sent.
|
|
305
|
+
|
|
306
|
+
### CLI Tools
|
|
307
|
+
|
|
308
|
+
After `pip install -e .` (or `uv sync`), the following commands are available:
|
|
309
|
+
|
|
310
|
+
```bash
|
|
311
|
+
# Crawl a page
|
|
312
|
+
crawl https://docs.example.com
|
|
313
|
+
|
|
314
|
+
# Site crawl with depth limit
|
|
315
|
+
crawl https://docs.example.com --site --max-depth 2 --max-pages 10 -o docs/
|
|
316
|
+
|
|
317
|
+
# Clean output (no links)
|
|
318
|
+
crawl https://example.com --remove-links
|
|
319
|
+
|
|
320
|
+
# Search
|
|
321
|
+
search "python tutorials"
|
|
322
|
+
search "Rezepte" --language de --max-results 5
|
|
323
|
+
|
|
324
|
+
# Session capture for authenticated crawling
|
|
325
|
+
crawl-capture --start-url https://example.com/login \
|
|
326
|
+
--completion-url 'https://example.com/dashboard.*' \
|
|
327
|
+
--output ./state.json
|
|
328
|
+
```
|
|
329
|
+
|
|
330
|
+
See [Session Capture](docs/usage/session-capture.md) for the full `crawl-capture` guide.
|
|
331
|
+
|
|
332
|
+
### Python API
|
|
333
|
+
|
|
334
|
+
```python
|
|
335
|
+
from crawler import crawl_page, crawl_page_async, crawl_site, crawl_site_async
|
|
336
|
+
|
|
337
|
+
# Single page
|
|
338
|
+
doc = await crawl_page_async("https://docs.example.com/intro", dedup_mode="exact")
|
|
339
|
+
print(doc.markdown)
|
|
340
|
+
|
|
341
|
+
# Site crawl
|
|
342
|
+
result = crawl_site("https://docs.example.com", max_depth=2, max_pages=10)
|
|
343
|
+
for doc in result.documents:
|
|
344
|
+
print(f"{doc.status}: {doc.final_url}")
|
|
345
|
+
|
|
346
|
+
# Authenticated crawl
|
|
347
|
+
doc = await crawl_page_async(
|
|
348
|
+
"https://example.com/private",
|
|
349
|
+
auth={"storage_state": "/path/to/state.json"},
|
|
350
|
+
)
|
|
351
|
+
```
|
|
352
|
+
|
|
353
|
+
## Reference
|
|
354
|
+
|
|
355
|
+
- **[MCP Tools](docs/usage/mcp-tools.md)** — full parameter reference for `crawl`, `crawl_site`, `search`
|
|
356
|
+
- **[Output Formats](docs/usage/output-formats.md)** — Markdown and JSON output structure, including `CrawledDocument`
|
|
357
|
+
- **[Session Capture](docs/usage/session-capture.md)** — manual login flow and CDP session export
|
|
358
|
+
|
|
359
|
+
## Configuration
|
|
360
|
+
|
|
361
|
+
Default config is optimized for documentation sites. Customize via overrides:
|
|
362
|
+
|
|
363
|
+
```python
|
|
364
|
+
from crawler import build_markdown_run_config, RunConfigOverrides
|
|
365
|
+
|
|
366
|
+
config = build_markdown_run_config(
|
|
367
|
+
RunConfigOverrides(
|
|
368
|
+
delay_before_return_html=1.0,
|
|
369
|
+
mean_delay=1.0,
|
|
370
|
+
scan_full_page=True,
|
|
371
|
+
)
|
|
372
|
+
)
|
|
373
|
+
doc = await crawl_page_async("https://example.com", config=config)
|
|
374
|
+
```
|
|
375
|
+
|
|
376
|
+
## Dependencies
|
|
377
|
+
|
|
378
|
+
- `crawl4ai>=0.7.4` — crawler engine
|
|
379
|
+
- `playwright>=1.40.0` — browser automation
|
|
380
|
+
- `fastmcp>=3.4.3` — MCP server framework
|
|
381
|
+
- `httpx>=0.27.0` — HTTP client for SearXNG
|
|
382
|
+
- `tldextract>=5.1.2` — domain parsing for site crawls
|
|
383
|
+
|
|
384
|
+
## License
|
|
385
|
+
|
|
386
|
+
MIT — © 2026 DDM – Das Digitale Momentum GmbH & Co KG
|
|
387
|
+
|
|
388
|
+
---
|
|
389
|
+
|
|
390
|
+
Maintained by [Das Digitale Momentum](https://www.das-digitale-momentum.de/en/open-source/#searxncrawl) · Much, Germany · [All our open source projects](https://github.com/DasDigitaleMomentum)
|