searxNcrawl 0.31.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. searxncrawl-0.31.0/LICENSE +21 -0
  2. searxncrawl-0.31.0/PKG-INFO +390 -0
  3. searxncrawl-0.31.0/README.md +356 -0
  4. searxncrawl-0.31.0/crawler/__init__.py +339 -0
  5. searxncrawl-0.31.0/crawler/auth.py +107 -0
  6. searxncrawl-0.31.0/crawler/browser_setup.py +96 -0
  7. searxncrawl-0.31.0/crawler/builder.py +216 -0
  8. searxncrawl-0.31.0/crawler/cli.py +975 -0
  9. searxncrawl-0.31.0/crawler/config.py +244 -0
  10. searxncrawl-0.31.0/crawler/document.py +31 -0
  11. searxncrawl-0.31.0/crawler/env.py +65 -0
  12. searxncrawl-0.31.0/crawler/markdown_dedup.py +108 -0
  13. searxncrawl-0.31.0/crawler/mcp_server.py +753 -0
  14. searxncrawl-0.31.0/crawler/references.py +62 -0
  15. searxncrawl-0.31.0/crawler/session_capture.py +364 -0
  16. searxncrawl-0.31.0/crawler/site.py +269 -0
  17. searxncrawl-0.31.0/pyproject.toml +81 -0
  18. searxncrawl-0.31.0/searxNcrawl.egg-info/PKG-INFO +390 -0
  19. searxncrawl-0.31.0/searxNcrawl.egg-info/SOURCES.txt +48 -0
  20. searxncrawl-0.31.0/searxNcrawl.egg-info/dependency_links.txt +1 -0
  21. searxncrawl-0.31.0/searxNcrawl.egg-info/entry_points.txt +6 -0
  22. searxncrawl-0.31.0/searxNcrawl.egg-info/requires.txt +10 -0
  23. searxncrawl-0.31.0/searxNcrawl.egg-info/top_level.txt +1 -0
  24. searxncrawl-0.31.0/setup.cfg +4 -0
  25. searxncrawl-0.31.0/tests/test_auth_core.py +166 -0
  26. searxncrawl-0.31.0/tests/test_browser_setup.py +165 -0
  27. searxncrawl-0.31.0/tests/test_builder.py +83 -0
  28. searxncrawl-0.31.0/tests/test_cli.py +494 -0
  29. searxncrawl-0.31.0/tests/test_config.py +12 -0
  30. searxncrawl-0.31.0/tests/test_config_loading.py +154 -0
  31. searxncrawl-0.31.0/tests/test_content_wait.py +121 -0
  32. searxncrawl-0.31.0/tests/test_cors.py +184 -0
  33. searxncrawl-0.31.0/tests/test_init.py +109 -0
  34. searxncrawl-0.31.0/tests/test_markdown_dedup.py +253 -0
  35. searxncrawl-0.31.0/tests/test_mcp_server.py +197 -0
  36. searxncrawl-0.31.0/tests/test_session_capture.py +297 -0
  37. searxncrawl-0.31.0/tests/test_timeout.py +220 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 DDM – Das Digitale Momentum GmbH & Co KG
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,390 @@
1
+ Metadata-Version: 2.4
2
+ Name: searxNcrawl
3
+ Version: 0.31.0
4
+ Summary: searxNcrawl web crawler with markdown extraction - no DB, no enrichment, just crawling.
5
+ Author: DDM – Das Digitale Momentum GmbH & Co KG
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://www.das-digitale-momentum.de/en/open-source/#searxncrawl
8
+ Project-URL: Repository, https://github.com/DasDigitaleMomentum/searxNcrawl
9
+ Project-URL: Issues, https://github.com/DasDigitaleMomentum/searxNcrawl/issues
10
+ Project-URL: Changelog, https://github.com/DasDigitaleMomentum/searxNcrawl/blob/main/CHANGELOG.md
11
+ Keywords: mcp,mcp-server,searxng,crawl4ai,web-crawler,web-search,markdown,llm-tools
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Environment :: Console
14
+ Classifier: Framework :: AsyncIO
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3 :: Only
19
+ Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
20
+ Classifier: Topic :: Text Processing :: Markup :: Markdown
21
+ Requires-Python: >=3.10
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Requires-Dist: crawl4ai>=0.7.4
25
+ Requires-Dist: tldextract>=5.1.2
26
+ Requires-Dist: playwright>=1.40.0
27
+ Requires-Dist: fastmcp>=3.4.3
28
+ Requires-Dist: httpx>=0.27.0
29
+ Requires-Dist: python-dotenv>=1.0.1
30
+ Provides-Extra: dev
31
+ Requires-Dist: pytest>=8.0.0; extra == "dev"
32
+ Requires-Dist: pytest-asyncio>=0.23.0; extra == "dev"
33
+ Dynamic: license-file
34
+
35
+ # searxNcrawl
36
+
37
+ MCP server and CLI toolkit for web search and crawling, built on [Crawl4AI](https://github.com/unclecode/crawl4ai) and [SearXNG](https://github.com/searxng/searxng).
38
+
39
+ Published at [github.com/DasDigitaleMomentum/searxNcrawl](https://github.com/DasDigitaleMomentum/searxNcrawl) — maintained by **DDM – Das Digitale Momentum GmbH & Co KG**. Successor to `searxng-mcp`.
40
+
41
+ <!-- mcp-name: io.github.DasDigitaleMomentum/searxncrawl -->
42
+
43
+ ## Quick Start
44
+
45
+ Pick your setup:
46
+
47
+ ### Docker Compose
48
+
49
+ MCP server with Playwright/Chromium, ready in one command. SearXNG required separately for search.
50
+
51
+ ```bash
52
+ cp .env.example .env # set SEARXNG_URL to your SearXNG instance
53
+ docker compose up --build
54
+ ```
55
+
56
+ ➜ MCP server at `http://localhost:9555/mcp`
57
+
58
+ ### uvx (no clone)
59
+
60
+ Run the MCP server from [PyPI](https://pypi.org/project/searxncrawl/), no clone or virtualenv. Chromium is downloaded automatically on the first crawl.
61
+
62
+ ```bash
63
+ SEARXNG_URL=http://your-searxng:8888 uvx searxncrawl
64
+ ```
65
+
66
+ The latest development version runs straight from GitHub with `uvx --from git+https://github.com/DasDigitaleMomentum/searxNcrawl searxncrawl`.
67
+
68
+ ### pip (standalone)
69
+
70
+ CLI tools, Python API, and MCP server. SearXNG required for search.
71
+
72
+ ```bash
73
+ python -m venv .venv && source .venv/bin/activate
74
+ pip install -e .
75
+ playwright install chromium
76
+ ```
77
+
78
+ ### uv (standalone)
79
+
80
+ Same capabilities as pip.
81
+
82
+ ```bash
83
+ uv sync
84
+ uv run playwright install chromium
85
+ ```
86
+
87
+ ### What you get
88
+
89
+ | Feature | Docker Compose | pip / uv |
90
+ | ----------------------- | -------------- | --------- |
91
+ | MCP Server (STDIO) | — | ✅ |
92
+ | MCP Server (HTTP) | ✅ | ✅ |
93
+ | Web Crawl | ✅ | ✅ |
94
+ | Web Search | ✅¹ | ✅¹ |
95
+ | CLI Tools | via `exec`² | ✅ |
96
+ | Python API | — | ✅ |
97
+ | CORS (HTTP) | ✅ | ✅ |
98
+
99
+ ¹ Requires a SearXNG instance. ² `docker compose exec searxncrawl crawl ...`
100
+
101
+ ## Features
102
+
103
+ ### Crawling
104
+ - Single page, multi-page, and **site crawling** (DFS with depth/page limits)
105
+ - Production-tested extraction config optimized for documentation sites
106
+ - Configurable timeouts with graceful error handling
107
+
108
+ ### Content Quality
109
+ - **Markdown deduplication** — `exact` (default) removes repeated blocks, `off` disables it
110
+ - **Link removal** — strip all links for cleaner LLM context (`--remove-links`)
111
+ - **Dedup guardrails** — non-destructive metadata signals when removal is unusually aggressive
112
+
113
+ ### Web Search
114
+ - SearXNG metasearch integration (privacy-respecting)
115
+ - Configurable language, time range, categories, engines, safe search
116
+
117
+ ### MCP Server
118
+ - **STDIO transport** — for MCP harnesses (Zed, opencode, VS Code, Claude Code, etc.)
119
+ - **HTTP transport** — for remote access and browser clients
120
+ - **CORS support** — configurable origins for browser-based MCP clients
121
+ - Noise-free startup with UTF-8 encoding (cross-platform, incl. Windows)
122
+
123
+ ### CLI Tools
124
+ - `crawl` — crawl pages from the command line
125
+ - `search` — search the web via SearXNG
126
+ - `crawl-capture` — session capture for authenticated crawling
127
+
128
+ ## Installation
129
+
130
+ ### Docker Compose
131
+
132
+ The Compose stack includes searxNcrawl + Playwright/Chromium. SearXNG must be provided separately.
133
+
134
+ ```bash
135
+ cp .env.example .env
136
+ # Edit .env: set SEARXNG_URL to your SearXNG instance
137
+ docker compose up --build
138
+ ```
139
+
140
+ | Variable | Default | Description |
141
+ | ----------- | ------------------------- | ------------------------------------------------------------- |
142
+ | `MCP_PORT` | `9555` | MCP server HTTP port |
143
+ | `LOG_LEVEL` | `INFO` | MCP server log level (DEBUG, INFO, WARNING, ERROR, CRITICAL) |
144
+ | `PLAYWRIGHT_AUTO_INSTALL` | `true` | Download Playwright's Chromium automatically before the first browser launch if it is missing. Set to `false` where browsers are provisioned separately |
145
+ | `FASTMCP_HTTP_ALLOWED_HOSTS` | (FastMCP secure defaults) | JSON list of trusted HTTP Host headers, for example `["mcp.example.com"]` |
146
+
147
+ The MCP server is available at `http://localhost:9555/mcp`.
148
+
149
+ ### pip
150
+
151
+ ```bash
152
+ cd searxNcrawl
153
+ python -m venv .venv
154
+ source .venv/bin/activate
155
+ pip install -e .
156
+ playwright install chromium
157
+ ```
158
+
159
+ ### uv
160
+
161
+ ```bash
162
+ cd searxNcrawl
163
+ uv sync
164
+ uv run playwright install chromium
165
+ ```
166
+
167
+ ### SearXNG (search feature)
168
+
169
+ The `search` tool and CLI command require a SearXNG instance with **JSON output enabled** (`search.formats` in `settings.yml`). For all setups you need your own instance — self-hosting is recommended over public instances (rate limits).
170
+
171
+ **Environment variables:**
172
+
173
+ | Variable | Example / Recommended | Description |
174
+ | ---------------------- | ------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- |
175
+ | `SEARXNG_URL` | `http://localhost:8888` | SearXNG instance URL |
176
+ | `SEARXNG_USERNAME` | (none) | Optional basic auth user |
177
+ | `SEARXNG_PASSWORD` | (none) | Optional basic auth pass |
178
+ | `SEARCH_RESULT_FIELDS` | `title,url,content,publishedDate` | Comma-separated result fields. Unset = all SearXNG fields. Available: title, url, content, publishedDate, engine, score, category, img_src, thumbnail |
179
+
180
+ Example `.env`:
181
+ ```bash
182
+ SEARXNG_URL=http://localhost:8888
183
+ SEARCH_RESULT_FIELDS=title,url,content,publishedDate
184
+ LOG_LEVEL=INFO
185
+ ```
186
+
187
+ **Config file search order** (CLI tools only):
188
+
189
+ 1. `./.env` — current directory
190
+ 2. `~/.config/searxncrawl/.env` — user config
191
+
192
+ If no `.env` exists, `.env.example` is auto-copied to the user config path.
193
+
194
+ ## Usage
195
+
196
+ ### MCP Server
197
+
198
+ #### Start the server
199
+
200
+ ```bash
201
+ # STDIO transport (for MCP harnesses)
202
+ python -m crawler.mcp_server
203
+
204
+ # HTTP transport
205
+ python -m crawler.mcp_server --transport http --port 8000
206
+
207
+ # HTTP exposed through a specific public hostname
208
+ python -m crawler.mcp_server --transport http --host 0.0.0.0 --allowed-hosts "mcp.example.com"
209
+
210
+ # HTTP with CORS
211
+ python -m crawler.mcp_server --transport http --allowed-hosts "mcp.example.com" --cors-origins "https://app.example.com"
212
+
213
+ # Docker (HTTP only)
214
+ docker compose up --build
215
+ ```
216
+
217
+ #### MCP client configuration
218
+
219
+ **With uvx (no clone, no venv):**
220
+
221
+ ```json
222
+ {
223
+ "mcpServers": {
224
+ "crawler": {
225
+ "command": "uvx",
226
+ "args": ["searxncrawl"],
227
+ "env": { "SEARXNG_URL": "http://your-searxng:8888" }
228
+ }
229
+ }
230
+ }
231
+ ```
232
+
233
+ The first crawl downloads Playwright's Chromium once (about 550 MB on disk); later starts reuse it. On Linux hosts without the browser's system libraries, run `uvx --from searxncrawl playwright install --with-deps chromium` once.
234
+
235
+ **Python with venv:**
236
+
237
+ ```json
238
+ {
239
+ "mcpServers": {
240
+ "crawler": {
241
+ "command": "python",
242
+ "args": ["-m", "crawler.mcp_server"],
243
+ "cwd": "/path/to/searxNcrawl",
244
+ "env": { "SEARXNG_URL": "http://your-searxng:8888" }
245
+ }
246
+ }
247
+ }
248
+ ```
249
+
250
+ **With uv (no manual venv):**
251
+
252
+ ```json
253
+ {
254
+ "mcpServers": {
255
+ "crawler": {
256
+ "command": "uv",
257
+ "args": ["run", "--directory", "/path/to/searxNcrawl", "python", "-m", "crawler.mcp_server"],
258
+ "env": { "SEARXNG_URL": "http://your-searxng:8888" }
259
+ }
260
+ }
261
+ }
262
+ ```
263
+
264
+ **Docker (HTTP endpoint):**
265
+
266
+ ```json
267
+ {
268
+ "mcpServers": {
269
+ "crawler": {
270
+ "url": "http://localhost:9555/mcp"
271
+ }
272
+ }
273
+ }
274
+ ```
275
+
276
+ #### CORS
277
+
278
+ FastMCP validates the HTTP `Host` header independently of the address on which
279
+ the server listens. For remote access, allow the exact externally visible Host
280
+ header with a comma-separated CLI value:
281
+
282
+ ```bash
283
+ crawl-mcp --transport http --host 0.0.0.0 --allowed-hosts "mcp.example.com,mcp.internal.example"
284
+ ```
285
+
286
+ Alternatively, use FastMCP's environment setting. It uses JSON-list syntax:
287
+
288
+ ```bash
289
+ FASTMCP_HTTP_ALLOWED_HOSTS='["mcp.example.com"]' crawl-mcp --transport http --host 0.0.0.0
290
+ ```
291
+
292
+ Browser Origin validation and CORS response headers are separate from Host
293
+ validation. `--cors-origins` configures both FastMCP's Origin guard and the CORS
294
+ middleware using the same normalized, comma-separated values:
295
+
296
+ ```bash
297
+ crawl-mcp --transport http --cors-origins "http://localhost:3000,https://myapp.com"
298
+ crawl-mcp --transport http --cors-origins "*" # all origins — local dev only
299
+ ```
300
+
301
+ Omitting either allowlist preserves FastMCP's secure defaults (and permits the
302
+ upstream environment setting to apply). A value of `*` for Hosts or Origins is
303
+ an explicit opt-in to broad access and should only be used when that security
304
+ trade-off is intentional. Without `--cors-origins`, no CORS headers are sent.
305
+
306
+ ### CLI Tools
307
+
308
+ After `pip install -e .` (or `uv sync`), the following commands are available:
309
+
310
+ ```bash
311
+ # Crawl a page
312
+ crawl https://docs.example.com
313
+
314
+ # Site crawl with depth limit
315
+ crawl https://docs.example.com --site --max-depth 2 --max-pages 10 -o docs/
316
+
317
+ # Clean output (no links)
318
+ crawl https://example.com --remove-links
319
+
320
+ # Search
321
+ search "python tutorials"
322
+ search "Rezepte" --language de --max-results 5
323
+
324
+ # Session capture for authenticated crawling
325
+ crawl-capture --start-url https://example.com/login \
326
+ --completion-url 'https://example.com/dashboard.*' \
327
+ --output ./state.json
328
+ ```
329
+
330
+ See [Session Capture](docs/usage/session-capture.md) for the full `crawl-capture` guide.
331
+
332
+ ### Python API
333
+
334
+ ```python
335
+ from crawler import crawl_page, crawl_page_async, crawl_site, crawl_site_async
336
+
337
+ # Single page
338
+ doc = await crawl_page_async("https://docs.example.com/intro", dedup_mode="exact")
339
+ print(doc.markdown)
340
+
341
+ # Site crawl
342
+ result = crawl_site("https://docs.example.com", max_depth=2, max_pages=10)
343
+ for doc in result.documents:
344
+ print(f"{doc.status}: {doc.final_url}")
345
+
346
+ # Authenticated crawl
347
+ doc = await crawl_page_async(
348
+ "https://example.com/private",
349
+ auth={"storage_state": "/path/to/state.json"},
350
+ )
351
+ ```
352
+
353
+ ## Reference
354
+
355
+ - **[MCP Tools](docs/usage/mcp-tools.md)** — full parameter reference for `crawl`, `crawl_site`, `search`
356
+ - **[Output Formats](docs/usage/output-formats.md)** — Markdown and JSON output structure, including `CrawledDocument`
357
+ - **[Session Capture](docs/usage/session-capture.md)** — manual login flow and CDP session export
358
+
359
+ ## Configuration
360
+
361
+ Default config is optimized for documentation sites. Customize via overrides:
362
+
363
+ ```python
364
+ from crawler import build_markdown_run_config, RunConfigOverrides
365
+
366
+ config = build_markdown_run_config(
367
+ RunConfigOverrides(
368
+ delay_before_return_html=1.0,
369
+ mean_delay=1.0,
370
+ scan_full_page=True,
371
+ )
372
+ )
373
+ doc = await crawl_page_async("https://example.com", config=config)
374
+ ```
375
+
376
+ ## Dependencies
377
+
378
+ - `crawl4ai>=0.7.4` — crawler engine
379
+ - `playwright>=1.40.0` — browser automation
380
+ - `fastmcp>=3.4.3` — MCP server framework
381
+ - `httpx>=0.27.0` — HTTP client for SearXNG
382
+ - `tldextract>=5.1.2` — domain parsing for site crawls
383
+
384
+ ## License
385
+
386
+ MIT — © 2026 DDM – Das Digitale Momentum GmbH & Co KG
387
+
388
+ ---
389
+
390
+ Maintained by [Das Digitale Momentum](https://www.das-digitale-momentum.de/en/open-source/#searxncrawl) · Much, Germany · [All our open source projects](https://github.com/DasDigitaleMomentum)