searchfetch 3.2.3__tar.gz → 3.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {searchfetch-3.2.3/searchfetch.egg-info → searchfetch-3.3.1}/PKG-INFO +11 -4
- {searchfetch-3.2.3 → searchfetch-3.3.1}/README.md +7 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/pyproject.toml +5 -16
- {searchfetch-3.2.3 → searchfetch-3.3.1/searchfetch.egg-info}/PKG-INFO +11 -4
- {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/SOURCES.txt +10 -1
- {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/requires.txt +3 -3
- {searchfetch-3.2.3 → searchfetch-3.3.1}/server.py +127 -47
- searchfetch-3.3.1/templates/devto.json +45 -0
- searchfetch-3.3.1/templates/gitlab.json +49 -0
- searchfetch-3.3.1/templates/go-pkg.json +50 -0
- searchfetch-3.3.1/templates/javadoc.json +43 -0
- searchfetch-3.3.1/templates/mdn-web-docs.json +39 -0
- searchfetch-3.3.1/templates/raw.json +14 -0
- searchfetch-3.3.1/templates/reddit.json +65 -0
- searchfetch-3.3.1/templates/wikipedia.json +46 -0
- searchfetch-3.3.1/templates/youtube.json +44 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/LICENSE +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/MANIFEST.in +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/dependency_links.txt +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/entry_points.txt +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/top_level.txt +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/setup.cfg +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/crates-package.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/docker-hub.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/docs-page.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/docs-rs.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/duckduckgo-search.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/github-issue.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/github-repo.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/google-search.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/npm-package.json +0 -0
- {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/pypi-package.json +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: searchfetch
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.3.1
|
|
4
4
|
Summary: A maximum fault-tolerant, stealth-enabled MCP server for web searching and fetching.
|
|
5
5
|
Author: Max
|
|
6
6
|
License-Expression: MIT
|
|
@@ -11,12 +11,12 @@ Requires-Python: >=3.10
|
|
|
11
11
|
Description-Content-Type: text/markdown
|
|
12
12
|
License-File: LICENSE
|
|
13
13
|
Requires-Dist: mcp>=1.28.1
|
|
14
|
-
Requires-Dist: cloakbrowser>=0.4.
|
|
14
|
+
Requires-Dist: cloakbrowser>=0.4.11
|
|
15
15
|
Requires-Dist: beautifulsoup4>=4.15.0
|
|
16
|
-
Requires-Dist: markdownify>=1.2.
|
|
16
|
+
Requires-Dist: markdownify>=1.2.3
|
|
17
17
|
Provides-Extra: dev
|
|
18
18
|
Requires-Dist: pytest>=9.1.1; extra == "dev"
|
|
19
|
-
Requires-Dist: ruff>=0.15.
|
|
19
|
+
Requires-Dist: ruff>=0.15.22; extra == "dev"
|
|
20
20
|
Dynamic: license-file
|
|
21
21
|
|
|
22
22
|
# SearchFetch (MCP Server)
|
|
@@ -105,6 +105,13 @@ Template extraction supports `text`, `markdown`, `attribute`, and `html` formats
|
|
|
105
105
|
|
|
106
106
|
Built-in templates live in `templates/*.json` and are shared by the Node.js and Python implementations.
|
|
107
107
|
|
|
108
|
+
**Available page templates** (auto-detected by URL or selectable by name):
|
|
109
|
+
`wikipedia`, `reddit`, `mdn-web-docs`, `gitlab`, `youtube`, `devto`, `go-pkg`, `javadoc`,
|
|
110
|
+
`github-repo`, `github-issue`, `npm-package`, `pypi-package`, `crates-package`,
|
|
111
|
+
`docker-hub`, `docs-rs`, `docs-page`
|
|
112
|
+
|
|
113
|
+
**`raw`** — special template that applies minimal filtering and returns full body content as markdown. Use when you need the complete page without template-specific extraction.
|
|
114
|
+
|
|
108
115
|
---
|
|
109
116
|
|
|
110
117
|
## Local Development
|
|
@@ -84,6 +84,13 @@ Template extraction supports `text`, `markdown`, `attribute`, and `html` formats
|
|
|
84
84
|
|
|
85
85
|
Built-in templates live in `templates/*.json` and are shared by the Node.js and Python implementations.
|
|
86
86
|
|
|
87
|
+
**Available page templates** (auto-detected by URL or selectable by name):
|
|
88
|
+
`wikipedia`, `reddit`, `mdn-web-docs`, `gitlab`, `youtube`, `devto`, `go-pkg`, `javadoc`,
|
|
89
|
+
`github-repo`, `github-issue`, `npm-package`, `pypi-package`, `crates-package`,
|
|
90
|
+
`docker-hub`, `docs-rs`, `docs-page`
|
|
91
|
+
|
|
92
|
+
**`raw`** — special template that applies minimal filtering and returns full body content as markdown. Use when you need the complete page without template-specific extraction.
|
|
93
|
+
|
|
87
94
|
---
|
|
88
95
|
|
|
89
96
|
## Local Development
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "searchfetch"
|
|
7
|
-
version = "3.
|
|
7
|
+
version = "3.3.1"
|
|
8
8
|
description = "A maximum fault-tolerant, stealth-enabled MCP server for web searching and fetching."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -15,15 +15,15 @@ authors =[
|
|
|
15
15
|
keywords =["mcp", "search", "fetch", "llm", "ai", "agent", "stealth"]
|
|
16
16
|
dependencies =[
|
|
17
17
|
"mcp>=1.28.1",
|
|
18
|
-
"cloakbrowser>=0.4.
|
|
18
|
+
"cloakbrowser>=0.4.11",
|
|
19
19
|
"beautifulsoup4>=4.15.0",
|
|
20
|
-
"markdownify>=1.2.
|
|
20
|
+
"markdownify>=1.2.3"
|
|
21
21
|
]
|
|
22
22
|
|
|
23
23
|
[project.optional-dependencies]
|
|
24
24
|
dev = [
|
|
25
25
|
"pytest>=9.1.1",
|
|
26
|
-
"ruff>=0.15.
|
|
26
|
+
"ruff>=0.15.22",
|
|
27
27
|
]
|
|
28
28
|
|
|
29
29
|
[project.urls]
|
|
@@ -38,18 +38,7 @@ searchfetch = "server:main"
|
|
|
38
38
|
py-modules =["server"]
|
|
39
39
|
|
|
40
40
|
[tool.setuptools.data-files]
|
|
41
|
-
templates = [
|
|
42
|
-
"templates/github-repo.json",
|
|
43
|
-
"templates/github-issue.json",
|
|
44
|
-
"templates/npm-package.json",
|
|
45
|
-
"templates/pypi-package.json",
|
|
46
|
-
"templates/crates-package.json",
|
|
47
|
-
"templates/google-search.json",
|
|
48
|
-
"templates/duckduckgo-search.json",
|
|
49
|
-
"templates/docs-page.json",
|
|
50
|
-
"templates/docs-rs.json",
|
|
51
|
-
"templates/docker-hub.json",
|
|
52
|
-
]
|
|
41
|
+
templates = ["templates/*.json"]
|
|
53
42
|
|
|
54
43
|
[tool.ruff]
|
|
55
44
|
line-length = 100
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: searchfetch
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.3.1
|
|
4
4
|
Summary: A maximum fault-tolerant, stealth-enabled MCP server for web searching and fetching.
|
|
5
5
|
Author: Max
|
|
6
6
|
License-Expression: MIT
|
|
@@ -11,12 +11,12 @@ Requires-Python: >=3.10
|
|
|
11
11
|
Description-Content-Type: text/markdown
|
|
12
12
|
License-File: LICENSE
|
|
13
13
|
Requires-Dist: mcp>=1.28.1
|
|
14
|
-
Requires-Dist: cloakbrowser>=0.4.
|
|
14
|
+
Requires-Dist: cloakbrowser>=0.4.11
|
|
15
15
|
Requires-Dist: beautifulsoup4>=4.15.0
|
|
16
|
-
Requires-Dist: markdownify>=1.2.
|
|
16
|
+
Requires-Dist: markdownify>=1.2.3
|
|
17
17
|
Provides-Extra: dev
|
|
18
18
|
Requires-Dist: pytest>=9.1.1; extra == "dev"
|
|
19
|
-
Requires-Dist: ruff>=0.15.
|
|
19
|
+
Requires-Dist: ruff>=0.15.22; extra == "dev"
|
|
20
20
|
Dynamic: license-file
|
|
21
21
|
|
|
22
22
|
# SearchFetch (MCP Server)
|
|
@@ -105,6 +105,13 @@ Template extraction supports `text`, `markdown`, `attribute`, and `html` formats
|
|
|
105
105
|
|
|
106
106
|
Built-in templates live in `templates/*.json` and are shared by the Node.js and Python implementations.
|
|
107
107
|
|
|
108
|
+
**Available page templates** (auto-detected by URL or selectable by name):
|
|
109
|
+
`wikipedia`, `reddit`, `mdn-web-docs`, `gitlab`, `youtube`, `devto`, `go-pkg`, `javadoc`,
|
|
110
|
+
`github-repo`, `github-issue`, `npm-package`, `pypi-package`, `crates-package`,
|
|
111
|
+
`docker-hub`, `docs-rs`, `docs-page`
|
|
112
|
+
|
|
113
|
+
**`raw`** — special template that applies minimal filtering and returns full body content as markdown. Use when you need the complete page without template-specific extraction.
|
|
114
|
+
|
|
108
115
|
---
|
|
109
116
|
|
|
110
117
|
## Local Development
|
|
@@ -10,12 +10,21 @@ searchfetch.egg-info/entry_points.txt
|
|
|
10
10
|
searchfetch.egg-info/requires.txt
|
|
11
11
|
searchfetch.egg-info/top_level.txt
|
|
12
12
|
templates/crates-package.json
|
|
13
|
+
templates/devto.json
|
|
13
14
|
templates/docker-hub.json
|
|
14
15
|
templates/docs-page.json
|
|
15
16
|
templates/docs-rs.json
|
|
16
17
|
templates/duckduckgo-search.json
|
|
17
18
|
templates/github-issue.json
|
|
18
19
|
templates/github-repo.json
|
|
20
|
+
templates/gitlab.json
|
|
21
|
+
templates/go-pkg.json
|
|
19
22
|
templates/google-search.json
|
|
23
|
+
templates/javadoc.json
|
|
24
|
+
templates/mdn-web-docs.json
|
|
20
25
|
templates/npm-package.json
|
|
21
|
-
templates/pypi-package.json
|
|
26
|
+
templates/pypi-package.json
|
|
27
|
+
templates/raw.json
|
|
28
|
+
templates/reddit.json
|
|
29
|
+
templates/wikipedia.json
|
|
30
|
+
templates/youtube.json
|
|
@@ -11,7 +11,7 @@ from markdownify import markdownify as md
|
|
|
11
11
|
from mcp.server.fastmcp import FastMCP
|
|
12
12
|
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
|
|
13
13
|
|
|
14
|
-
__version__ = "3.
|
|
14
|
+
__version__ = "3.3.1"
|
|
15
15
|
|
|
16
16
|
mcp = FastMCP("searchfetch")
|
|
17
17
|
|
|
@@ -448,13 +448,16 @@ ACCESS_DENIED_TITLE_PATTERNS = [
|
|
|
448
448
|
]
|
|
449
449
|
|
|
450
450
|
ACCESS_DENIED_BODY_TRIGGERS = [
|
|
451
|
-
"captcha",
|
|
452
451
|
"verify you are human",
|
|
453
452
|
"are you a robot",
|
|
454
|
-
"unusual traffic",
|
|
455
453
|
"access denied",
|
|
454
|
+
]
|
|
455
|
+
|
|
456
|
+
ACCESS_DENIED_STRONG_BODY_TRIGGERS = [
|
|
457
|
+
"our systems have detected unusual traffic",
|
|
456
458
|
"sorry, you have been blocked",
|
|
457
459
|
"to continue, please type the characters",
|
|
460
|
+
"bots use duckduckgo too",
|
|
458
461
|
]
|
|
459
462
|
|
|
460
463
|
_ACCESS_DENIED_RE = re.compile("|".join(ACCESS_DENIED_TITLE_PATTERNS), re.IGNORECASE)
|
|
@@ -473,8 +476,10 @@ def _detect_access_failure(http_status: int | None, soup: BeautifulSoup) -> bool
|
|
|
473
476
|
body = soup.find("body")
|
|
474
477
|
if body:
|
|
475
478
|
body_text = body.get_text(" ", strip=True)
|
|
479
|
+
lowered = body_text.lower()
|
|
480
|
+
if any(trigger in lowered for trigger in ACCESS_DENIED_STRONG_BODY_TRIGGERS):
|
|
481
|
+
return True
|
|
476
482
|
if len(body_text) < 800:
|
|
477
|
-
lowered = body_text.lower()
|
|
478
483
|
if any(trigger in lowered for trigger in ACCESS_DENIED_BODY_TRIGGERS):
|
|
479
484
|
return True
|
|
480
485
|
|
|
@@ -606,6 +611,7 @@ async def fetch_html(url: str, template: dict | None, block_media: bool, retry:
|
|
|
606
611
|
or "econnreset" in exc_msg
|
|
607
612
|
or "enotfound" in exc_msg
|
|
608
613
|
or "navigation" in exc_msg
|
|
614
|
+
or "navigating" in exc_msg
|
|
609
615
|
)
|
|
610
616
|
if attempt < attempts - 1 and is_network:
|
|
611
617
|
await asyncio.sleep(0.5)
|
|
@@ -1033,30 +1039,10 @@ def generic_fallback(html_content: str, _url: str = "") -> str:
|
|
|
1033
1039
|
# ---------------------------------------------------------------------------
|
|
1034
1040
|
|
|
1035
1041
|
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
engine: str = "duckduckgo",
|
|
1040
|
-
region: str | None = None,
|
|
1041
|
-
safe_search: bool | None = None,
|
|
1042
|
-
max_results: int = 10,
|
|
1043
|
-
block_media: bool = True,
|
|
1044
|
-
) -> str:
|
|
1045
|
-
"""
|
|
1046
|
-
Search the web using template-driven search engines (Google, DuckDuckGo, or custom).
|
|
1047
|
-
|
|
1048
|
-
Args:
|
|
1049
|
-
query: The search query string.
|
|
1050
|
-
engine: Search engine: 'duckduckgo' or 'google'. Also accepts custom template names.
|
|
1051
|
-
region: Region code. DDG: 'us-en', 'de-de', etc. Google: 'de-de' maps to hl=de, gl=de.
|
|
1052
|
-
safe_search: Enable safe search. Maps to engine-specific params.
|
|
1053
|
-
max_results: Max results to extract. Default: 10.
|
|
1054
|
-
block_media: Block images/media/fonts. Default: True.
|
|
1055
|
-
"""
|
|
1056
|
-
# 1. Resolve engine → template name
|
|
1042
|
+
def _resolve_search_candidate(
|
|
1043
|
+
engine: str, query: str, region: str | None, safe_search: bool | None
|
|
1044
|
+
) -> tuple[str, dict, str]:
|
|
1057
1045
|
template_name = resolve_search_template(engine)
|
|
1058
|
-
|
|
1059
|
-
# 2. Resolve the template object
|
|
1060
1046
|
if template_name.startswith("{"):
|
|
1061
1047
|
try:
|
|
1062
1048
|
template = json.loads(template_name)
|
|
@@ -1070,33 +1056,116 @@ async def websearch(
|
|
|
1070
1056
|
)
|
|
1071
1057
|
template = BUILTIN_TEMPLATES[template_name]
|
|
1072
1058
|
|
|
1073
|
-
# Validate url_template exists
|
|
1074
1059
|
if not template.get("url_template"):
|
|
1075
1060
|
raise ValueError(
|
|
1076
1061
|
f"Template '{template.get('name', template_name)}' is not a search "
|
|
1077
1062
|
"template (no url_template). Use webfetch for page templates."
|
|
1078
1063
|
)
|
|
1079
1064
|
|
|
1080
|
-
# 3. Map universal params to engine-specific url_params
|
|
1081
1065
|
engine_key = engine if engine in ("duckduckgo", "google") else "custom"
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
# 4. Resolve URL template
|
|
1085
|
-
resolved_url = resolve_url_template(template, url_params)
|
|
1086
|
-
|
|
1087
|
-
# Google safe_search: append &safe=active after URL resolution
|
|
1066
|
+
params = map_search_params(query, engine_key, region, safe_search)
|
|
1067
|
+
resolved_url = resolve_url_template(template, params)
|
|
1088
1068
|
if engine_key == "google" and safe_search is True:
|
|
1089
1069
|
resolved_url += "&safe=active"
|
|
1070
|
+
return engine, template, resolved_url
|
|
1071
|
+
|
|
1072
|
+
|
|
1073
|
+
def _resolve_duckduckgo_lite_candidate(
|
|
1074
|
+
query: str, region: str | None, safe_search: bool | None
|
|
1075
|
+
) -> tuple[str, dict, str]:
|
|
1076
|
+
template = dict(BUILTIN_TEMPLATES["duckduckgo-search"])
|
|
1077
|
+
template.update(
|
|
1078
|
+
{
|
|
1079
|
+
"name": "duckduckgo-lite-search",
|
|
1080
|
+
"url_template": "https://lite.duckduckgo.com/lite/?q={query}&kl={kl}&kp={kp}",
|
|
1081
|
+
"sections": [
|
|
1082
|
+
{
|
|
1083
|
+
"name": "Results",
|
|
1084
|
+
"selector": "a.result-link",
|
|
1085
|
+
"multiple": True,
|
|
1086
|
+
"max_items": 10,
|
|
1087
|
+
"children": [
|
|
1088
|
+
{"name": "Title", "selector": "", "format": "text"},
|
|
1089
|
+
{
|
|
1090
|
+
"name": "URL",
|
|
1091
|
+
"selector": "",
|
|
1092
|
+
"format": "attribute",
|
|
1093
|
+
"attribute": "href",
|
|
1094
|
+
"transform": "decode_ddg_url",
|
|
1095
|
+
},
|
|
1096
|
+
],
|
|
1097
|
+
}
|
|
1098
|
+
],
|
|
1099
|
+
}
|
|
1100
|
+
)
|
|
1101
|
+
params = map_search_params(query, "duckduckgo", region, safe_search)
|
|
1102
|
+
return "duckduckgo-lite", template, resolve_url_template(template, params)
|
|
1103
|
+
|
|
1104
|
+
|
|
1105
|
+
def _resolve_search_candidates(
|
|
1106
|
+
engine: str, query: str, region: str | None, safe_search: bool | None
|
|
1107
|
+
) -> list[tuple[str, dict, str]]:
|
|
1108
|
+
primary = _resolve_search_candidate(engine, query, region, safe_search)
|
|
1109
|
+
if engine == "google":
|
|
1110
|
+
return [
|
|
1111
|
+
primary,
|
|
1112
|
+
_resolve_search_candidate("duckduckgo", query, region, safe_search),
|
|
1113
|
+
_resolve_duckduckgo_lite_candidate(query, region, safe_search),
|
|
1114
|
+
]
|
|
1115
|
+
if engine == "duckduckgo":
|
|
1116
|
+
return [primary, _resolve_duckduckgo_lite_candidate(query, region, safe_search)]
|
|
1117
|
+
return [primary]
|
|
1090
1118
|
|
|
1091
|
-
# 5. Fetch
|
|
1092
|
-
html = await fetch_html(resolved_url, template, block_media, retry=True)
|
|
1093
1119
|
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1120
|
+
@mcp.tool()
|
|
1121
|
+
async def websearch(
|
|
1122
|
+
query: str,
|
|
1123
|
+
engine: str = "duckduckgo",
|
|
1124
|
+
region: str | None = None,
|
|
1125
|
+
safe_search: bool | None = None,
|
|
1126
|
+
max_results: int = 10,
|
|
1127
|
+
block_media: bool = True,
|
|
1128
|
+
) -> str:
|
|
1129
|
+
"""
|
|
1130
|
+
Search the web using template-driven search engines (Google, DuckDuckGo, or custom).
|
|
1097
1131
|
|
|
1098
|
-
|
|
1099
|
-
|
|
1132
|
+
Args:
|
|
1133
|
+
query: The search query string.
|
|
1134
|
+
engine: Search engine: 'duckduckgo' or 'google'. Also accepts custom template names.
|
|
1135
|
+
region: Region code. DDG: 'us-en', 'de-de', etc. Google: 'de-de' maps to hl=de, gl=de.
|
|
1136
|
+
safe_search: Enable safe search. Maps to engine-specific params.
|
|
1137
|
+
max_results: Max results to extract. Default: 10.
|
|
1138
|
+
block_media: Block images/media/fonts. Default: True.
|
|
1139
|
+
"""
|
|
1140
|
+
candidates = _resolve_search_candidates(engine, query, region, safe_search)
|
|
1141
|
+
last_exc = None
|
|
1142
|
+
|
|
1143
|
+
for selected_engine, template, resolved_url in candidates:
|
|
1144
|
+
try:
|
|
1145
|
+
html = await fetch_html(resolved_url, template, block_media, retry=True)
|
|
1146
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
1147
|
+
sections_data = extract_template(
|
|
1148
|
+
soup, template, origin="", max_results_override=max_results
|
|
1149
|
+
)
|
|
1150
|
+
result_section = next(
|
|
1151
|
+
(section for section in sections_data if section.get("type") == "search_results"),
|
|
1152
|
+
None,
|
|
1153
|
+
)
|
|
1154
|
+
if result_section and result_section.get("items"):
|
|
1155
|
+
result = compose_search_results(
|
|
1156
|
+
sections_data, template.get("name", selected_engine)
|
|
1157
|
+
)
|
|
1158
|
+
if selected_engine != engine:
|
|
1159
|
+
result = (
|
|
1160
|
+
f"[websearch: {engine} was unavailable; results are from "
|
|
1161
|
+
f"{selected_engine}.]\n\n{result}"
|
|
1162
|
+
)
|
|
1163
|
+
return result
|
|
1164
|
+
last_exc = RuntimeError(f"No results extracted from {selected_engine}.")
|
|
1165
|
+
except Exception as exc:
|
|
1166
|
+
last_exc = exc
|
|
1167
|
+
|
|
1168
|
+
raise last_exc or RuntimeError("No search results found.")
|
|
1100
1169
|
|
|
1101
1170
|
|
|
1102
1171
|
@mcp.tool()
|
|
@@ -1108,14 +1177,25 @@ async def webfetch(
|
|
|
1108
1177
|
block_media: bool = True,
|
|
1109
1178
|
) -> str:
|
|
1110
1179
|
"""
|
|
1111
|
-
Fetch
|
|
1180
|
+
Fetch and extract the main text content from any webpage. Fully executes
|
|
1181
|
+
JavaScript to load React/SPAs and aggressively strips images/media
|
|
1182
|
+
(including base64) to save context tokens.
|
|
1112
1183
|
|
|
1113
1184
|
Args:
|
|
1114
|
-
url: The full URL to fetch.
|
|
1115
|
-
template:
|
|
1185
|
+
url: The full URL of the webpage to fetch (must start with http/https).
|
|
1186
|
+
template: Template to use: 'auto' (auto-detect from URL), a built-in
|
|
1187
|
+
page template name (see list below), 'raw' for
|
|
1188
|
+
minimal-filtering full-page output, or inline JSON.
|
|
1189
|
+
Default: 'auto'.
|
|
1116
1190
|
start_index: Character offset for pagination. Default: 0.
|
|
1117
|
-
max_length:
|
|
1118
|
-
block_media: Block images
|
|
1191
|
+
max_length: Maximum characters to return per request. Default is 10000.
|
|
1192
|
+
block_media: Block images, videos, and fonts entirely at the network
|
|
1193
|
+
layer. Default is true.
|
|
1194
|
+
|
|
1195
|
+
Built-in page templates (use by name to force a specific extractor):
|
|
1196
|
+
wikipedia, reddit, mdn-web-docs, gitlab, youtube, devto, go-pkg,
|
|
1197
|
+
javadoc, github-repo, github-issue, npm-package, pypi-package,
|
|
1198
|
+
crates-package, docker-hub, docs-rs, docs-page, raw
|
|
1119
1199
|
"""
|
|
1120
1200
|
# 1. Resolve template
|
|
1121
1201
|
matched_template, template_name = resolve_page_template(url, template)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "devto",
|
|
3
|
+
"description": "DEV Community article — title, author, and content",
|
|
4
|
+
"order": 20,
|
|
5
|
+
"url_patterns": [
|
|
6
|
+
"^https?://dev\\.to/[^/]+/[^/]+"
|
|
7
|
+
],
|
|
8
|
+
"remove": [
|
|
9
|
+
"script",
|
|
10
|
+
"style",
|
|
11
|
+
"svg",
|
|
12
|
+
"img",
|
|
13
|
+
"nav",
|
|
14
|
+
"footer",
|
|
15
|
+
".sidebar",
|
|
16
|
+
".crayons-layout__sidebar-left",
|
|
17
|
+
".crayons-layout__sidebar-right",
|
|
18
|
+
".crayons-article-actions",
|
|
19
|
+
".reaction-button",
|
|
20
|
+
".crayons-btn",
|
|
21
|
+
".spec__tags",
|
|
22
|
+
".article-engagement-bar",
|
|
23
|
+
".comments-container"
|
|
24
|
+
],
|
|
25
|
+
"sections": [
|
|
26
|
+
{
|
|
27
|
+
"name": "Title",
|
|
28
|
+
"selector": "h1",
|
|
29
|
+
"format": "text",
|
|
30
|
+
"required": true
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
"name": "Author",
|
|
34
|
+
"selector": ".crayons-article__header__meta a, a.crayons-avatar",
|
|
35
|
+
"format": "text",
|
|
36
|
+
"required": false
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"name": "Content",
|
|
40
|
+
"selector": "#article-body, .crayons-article__body, article.crayons-article",
|
|
41
|
+
"format": "markdown",
|
|
42
|
+
"required": true
|
|
43
|
+
}
|
|
44
|
+
]
|
|
45
|
+
}
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "gitlab",
|
|
3
|
+
"description": "GitLab project — description, file listing, and README",
|
|
4
|
+
"order": 12,
|
|
5
|
+
"url_patterns": [
|
|
6
|
+
"^https?://gitlab\\.com/[^/]+/[^/]+/?$",
|
|
7
|
+
"^https?://gitlab\\.com/[^/]+/[^/]+/-/tree/[^/]+/?$"
|
|
8
|
+
],
|
|
9
|
+
"remove": [
|
|
10
|
+
"script",
|
|
11
|
+
"style",
|
|
12
|
+
"svg",
|
|
13
|
+
"nav",
|
|
14
|
+
"footer",
|
|
15
|
+
"header",
|
|
16
|
+
".navbar-gitlab",
|
|
17
|
+
".nav-sidebar",
|
|
18
|
+
".issuable-list",
|
|
19
|
+
".super-sidebar",
|
|
20
|
+
"aside"
|
|
21
|
+
],
|
|
22
|
+
"sections": [
|
|
23
|
+
{
|
|
24
|
+
"name": "Title",
|
|
25
|
+
"selector": "h1",
|
|
26
|
+
"format": "text",
|
|
27
|
+
"required": true
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"name": "Description",
|
|
31
|
+
"selector": "meta[name='description']",
|
|
32
|
+
"format": "attribute",
|
|
33
|
+
"attribute": "content",
|
|
34
|
+
"required": false
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
"name": "Files",
|
|
38
|
+
"selector": "table[data-qa-selector='file_tree_table'], .tree-table",
|
|
39
|
+
"format": "markdown",
|
|
40
|
+
"required": false
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"name": "README",
|
|
44
|
+
"selector": ".readme-holder .md, .file-content .md, .readme-holder",
|
|
45
|
+
"format": "markdown",
|
|
46
|
+
"required": false
|
|
47
|
+
}
|
|
48
|
+
]
|
|
49
|
+
}
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "go-pkg",
|
|
3
|
+
"description": "Go package documentation on pkg.go.dev — name, overview, and docs",
|
|
4
|
+
"order": 22,
|
|
5
|
+
"url_patterns": [
|
|
6
|
+
"^https?://pkg\\.go\\.dev/[^/]+/?$",
|
|
7
|
+
"^https?://pkg\\.go\\.dev/[^/]+/[^/]+/?$",
|
|
8
|
+
"^https?://pkg\\.go\\.dev/[^/]+/[^/]+/[^/]+/?$"
|
|
9
|
+
],
|
|
10
|
+
"remove": [
|
|
11
|
+
"script",
|
|
12
|
+
"style",
|
|
13
|
+
"svg",
|
|
14
|
+
"nav",
|
|
15
|
+
"footer",
|
|
16
|
+
".go-Header",
|
|
17
|
+
".go-SiteHeader",
|
|
18
|
+
".go-Footer",
|
|
19
|
+
".UnitMeta",
|
|
20
|
+
".UnitMeta-repo",
|
|
21
|
+
".UnitFiles",
|
|
22
|
+
".JumpDrawer",
|
|
23
|
+
".go-Main-nav",
|
|
24
|
+
".go-Main-headerSubmenu",
|
|
25
|
+
".Site-header",
|
|
26
|
+
".UnitHeader-version",
|
|
27
|
+
".UnitHeader-breadcrumb",
|
|
28
|
+
".UnitHeader-details"
|
|
29
|
+
],
|
|
30
|
+
"sections": [
|
|
31
|
+
{
|
|
32
|
+
"name": "Title",
|
|
33
|
+
"selector": "h1",
|
|
34
|
+
"format": "text",
|
|
35
|
+
"required": true
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"name": "Overview",
|
|
39
|
+
"selector": ".Documentation-overview, .Overview",
|
|
40
|
+
"format": "markdown",
|
|
41
|
+
"required": false
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"name": "Documentation",
|
|
45
|
+
"selector": ".Documentation-content, .Documentation, #documentation",
|
|
46
|
+
"format": "markdown",
|
|
47
|
+
"required": false
|
|
48
|
+
}
|
|
49
|
+
]
|
|
50
|
+
}
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "javadoc",
|
|
3
|
+
"description": "Java SE API documentation — class/method details",
|
|
4
|
+
"order": 8.5,
|
|
5
|
+
"url_patterns": [
|
|
6
|
+
"^https?://docs\\.oracle\\.com/.*/docs/api/.*"
|
|
7
|
+
],
|
|
8
|
+
"remove": [
|
|
9
|
+
"script",
|
|
10
|
+
"style",
|
|
11
|
+
"svg",
|
|
12
|
+
"nav",
|
|
13
|
+
"footer",
|
|
14
|
+
"header",
|
|
15
|
+
".topNav",
|
|
16
|
+
".bottomNav",
|
|
17
|
+
".subNav",
|
|
18
|
+
".navListSearch",
|
|
19
|
+
".skipNav",
|
|
20
|
+
".searchTag",
|
|
21
|
+
".searchTagResult",
|
|
22
|
+
".useSummary",
|
|
23
|
+
".inheritedList",
|
|
24
|
+
".details",
|
|
25
|
+
".memberSummary",
|
|
26
|
+
".colFirst",
|
|
27
|
+
".colLast"
|
|
28
|
+
],
|
|
29
|
+
"sections": [
|
|
30
|
+
{
|
|
31
|
+
"name": "Title",
|
|
32
|
+
"selector": ".header h2, h2.title, h2:first-of-type",
|
|
33
|
+
"format": "text",
|
|
34
|
+
"required": false
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
"name": "Content",
|
|
38
|
+
"selector": "[role='main'], .document, .rst-content, article, .content, .markdown-body, .prose, main, .contentContainer",
|
|
39
|
+
"format": "markdown",
|
|
40
|
+
"required": true
|
|
41
|
+
}
|
|
42
|
+
]
|
|
43
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "mdn-web-docs",
|
|
3
|
+
"description": "MDN Web Docs article — title and main content",
|
|
4
|
+
"order": 15,
|
|
5
|
+
"url_patterns": [
|
|
6
|
+
"^https?://developer\\.mozilla\\.org/.*"
|
|
7
|
+
],
|
|
8
|
+
"remove": [
|
|
9
|
+
"script",
|
|
10
|
+
"style",
|
|
11
|
+
"svg",
|
|
12
|
+
"nav",
|
|
13
|
+
"footer",
|
|
14
|
+
"header",
|
|
15
|
+
".sidebar",
|
|
16
|
+
".toc",
|
|
17
|
+
".interactive",
|
|
18
|
+
".playground",
|
|
19
|
+
".bc-data",
|
|
20
|
+
".prev-next",
|
|
21
|
+
".notecard",
|
|
22
|
+
".language-menu",
|
|
23
|
+
".breadcrumbs-container"
|
|
24
|
+
],
|
|
25
|
+
"sections": [
|
|
26
|
+
{
|
|
27
|
+
"name": "Title",
|
|
28
|
+
"selector": "h1",
|
|
29
|
+
"format": "text",
|
|
30
|
+
"required": true
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
"name": "Content",
|
|
34
|
+
"selector": "#content, article, main, .main-page-content",
|
|
35
|
+
"format": "markdown",
|
|
36
|
+
"required": true
|
|
37
|
+
}
|
|
38
|
+
]
|
|
39
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "raw",
|
|
3
|
+
"description": "Raw page content — minimal filtering, full body as markdown (use when you need the complete page without template extraction)",
|
|
4
|
+
"order": 999,
|
|
5
|
+
"remove": [],
|
|
6
|
+
"sections": [
|
|
7
|
+
{
|
|
8
|
+
"name": "Content",
|
|
9
|
+
"selector": "body",
|
|
10
|
+
"format": "markdown",
|
|
11
|
+
"required": false
|
|
12
|
+
}
|
|
13
|
+
]
|
|
14
|
+
}
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "reddit",
|
|
3
|
+
"description": "Reddit post — title, content, and comments",
|
|
4
|
+
"order": 13,
|
|
5
|
+
"url_patterns": [
|
|
6
|
+
"^https?://(?:www\\.|old\\.)?reddit\\.com/r/[^/]+/comments/[^/]+"
|
|
7
|
+
],
|
|
8
|
+
"remove": [
|
|
9
|
+
"script",
|
|
10
|
+
"style",
|
|
11
|
+
"svg",
|
|
12
|
+
"img",
|
|
13
|
+
"nav",
|
|
14
|
+
"footer",
|
|
15
|
+
"header",
|
|
16
|
+
".promoted-post",
|
|
17
|
+
"shreddit-ad-post",
|
|
18
|
+
".side",
|
|
19
|
+
".sidebar",
|
|
20
|
+
".footer-parent",
|
|
21
|
+
"#header",
|
|
22
|
+
".sr-header-area",
|
|
23
|
+
".menuarea",
|
|
24
|
+
".linkinfo",
|
|
25
|
+
".panestack-title",
|
|
26
|
+
".report-button",
|
|
27
|
+
"shreddit-join-button",
|
|
28
|
+
"#overlayScrollContainer",
|
|
29
|
+
"#right-sidebar-container",
|
|
30
|
+
"#left-sidebar-container"
|
|
31
|
+
],
|
|
32
|
+
"sections": [
|
|
33
|
+
{
|
|
34
|
+
"name": "Subreddit",
|
|
35
|
+
"selector": "a[href*='/r/'], shreddit-post a[href*='/r/']",
|
|
36
|
+
"format": "text",
|
|
37
|
+
"required": false
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
"name": "Title",
|
|
41
|
+
"selector": "h1, shreddit-post [slot='title']",
|
|
42
|
+
"format": "text",
|
|
43
|
+
"required": true
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"name": "Author",
|
|
47
|
+
"selector": "a[href*='/user/'], shreddit-post a[href*='/user/'], [slot='authorName']",
|
|
48
|
+
"format": "text",
|
|
49
|
+
"required": false
|
|
50
|
+
},
|
|
51
|
+
{
|
|
52
|
+
"name": "Link",
|
|
53
|
+
"selector": "a[slot='external-anchor'], a.outbound-link, a[data-testid='outbound-link']",
|
|
54
|
+
"format": "attribute",
|
|
55
|
+
"attribute": "href",
|
|
56
|
+
"required": false
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
"name": "Content",
|
|
60
|
+
"selector": "[slot='text-body'], .text-body, .expando, .usertext-body .md",
|
|
61
|
+
"format": "markdown",
|
|
62
|
+
"required": false
|
|
63
|
+
}
|
|
64
|
+
]
|
|
65
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "wikipedia",
|
|
3
|
+
"description": "Wikipedia article — title and main content",
|
|
4
|
+
"order": 11,
|
|
5
|
+
"url_patterns": [
|
|
6
|
+
"^https?://[a-z]{2,4}\\.wikipedia\\.org/wiki/[^:]+$"
|
|
7
|
+
],
|
|
8
|
+
"remove": [
|
|
9
|
+
"script",
|
|
10
|
+
"style",
|
|
11
|
+
"svg",
|
|
12
|
+
"nav",
|
|
13
|
+
"footer",
|
|
14
|
+
".mw-editsection",
|
|
15
|
+
".reflist",
|
|
16
|
+
".navbox",
|
|
17
|
+
".sisterproject",
|
|
18
|
+
".metadata",
|
|
19
|
+
".sidebar",
|
|
20
|
+
".infobox",
|
|
21
|
+
"#toc",
|
|
22
|
+
".mw-jump-link",
|
|
23
|
+
".printfooter",
|
|
24
|
+
".catlinks",
|
|
25
|
+
".mw-authority-control",
|
|
26
|
+
".shortdescription",
|
|
27
|
+
".mw-indicators",
|
|
28
|
+
".page-actions",
|
|
29
|
+
".mw-page-title-namespace",
|
|
30
|
+
".mw-heading-subtitle"
|
|
31
|
+
],
|
|
32
|
+
"sections": [
|
|
33
|
+
{
|
|
34
|
+
"name": "Title",
|
|
35
|
+
"selector": "#firstHeading, h1",
|
|
36
|
+
"format": "text",
|
|
37
|
+
"required": true
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
"name": "Content",
|
|
41
|
+
"selector": "#mw-content-text .mw-parser-output",
|
|
42
|
+
"format": "markdown",
|
|
43
|
+
"required": true
|
|
44
|
+
}
|
|
45
|
+
]
|
|
46
|
+
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "youtube",
|
|
3
|
+
"description": "YouTube video — title, channel, description",
|
|
4
|
+
"order": 30,
|
|
5
|
+
"url_patterns": [
|
|
6
|
+
"^https?://(?:www\\.)?(?:youtube\\.com/watch\\?v=|youtu\\.be/).+"
|
|
7
|
+
],
|
|
8
|
+
"remove": [
|
|
9
|
+
"script",
|
|
10
|
+
"style",
|
|
11
|
+
"svg",
|
|
12
|
+
"img",
|
|
13
|
+
"nav",
|
|
14
|
+
"header",
|
|
15
|
+
"ytd-guide-renderer",
|
|
16
|
+
"ytd-mini-guide-renderer",
|
|
17
|
+
"#comments",
|
|
18
|
+
"#related",
|
|
19
|
+
"ytd-comments",
|
|
20
|
+
"ytd-live-chat-frame",
|
|
21
|
+
".ytp-ad-module"
|
|
22
|
+
],
|
|
23
|
+
"sections": [
|
|
24
|
+
{
|
|
25
|
+
"name": "Title",
|
|
26
|
+
"selector": "meta[name='title']",
|
|
27
|
+
"format": "attribute",
|
|
28
|
+
"attribute": "content",
|
|
29
|
+
"required": true
|
|
30
|
+
},
|
|
31
|
+
{
|
|
32
|
+
"name": "Channel",
|
|
33
|
+
"selector": "#owner a, ytd-video-owner-renderer a, #channel-name a",
|
|
34
|
+
"format": "text",
|
|
35
|
+
"required": false
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"name": "Description",
|
|
39
|
+
"selector": "#description-inline-expander, ytd-text-inline-expander, #info-container",
|
|
40
|
+
"format": "markdown",
|
|
41
|
+
"required": false
|
|
42
|
+
}
|
|
43
|
+
]
|
|
44
|
+
}
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|