searchfetch 3.2.3__tar.gz → 3.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {searchfetch-3.2.3/searchfetch.egg-info → searchfetch-3.3.1}/PKG-INFO +11 -4
  2. {searchfetch-3.2.3 → searchfetch-3.3.1}/README.md +7 -0
  3. {searchfetch-3.2.3 → searchfetch-3.3.1}/pyproject.toml +5 -16
  4. {searchfetch-3.2.3 → searchfetch-3.3.1/searchfetch.egg-info}/PKG-INFO +11 -4
  5. {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/SOURCES.txt +10 -1
  6. {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/requires.txt +3 -3
  7. {searchfetch-3.2.3 → searchfetch-3.3.1}/server.py +127 -47
  8. searchfetch-3.3.1/templates/devto.json +45 -0
  9. searchfetch-3.3.1/templates/gitlab.json +49 -0
  10. searchfetch-3.3.1/templates/go-pkg.json +50 -0
  11. searchfetch-3.3.1/templates/javadoc.json +43 -0
  12. searchfetch-3.3.1/templates/mdn-web-docs.json +39 -0
  13. searchfetch-3.3.1/templates/raw.json +14 -0
  14. searchfetch-3.3.1/templates/reddit.json +65 -0
  15. searchfetch-3.3.1/templates/wikipedia.json +46 -0
  16. searchfetch-3.3.1/templates/youtube.json +44 -0
  17. {searchfetch-3.2.3 → searchfetch-3.3.1}/LICENSE +0 -0
  18. {searchfetch-3.2.3 → searchfetch-3.3.1}/MANIFEST.in +0 -0
  19. {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/dependency_links.txt +0 -0
  20. {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/entry_points.txt +0 -0
  21. {searchfetch-3.2.3 → searchfetch-3.3.1}/searchfetch.egg-info/top_level.txt +0 -0
  22. {searchfetch-3.2.3 → searchfetch-3.3.1}/setup.cfg +0 -0
  23. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/crates-package.json +0 -0
  24. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/docker-hub.json +0 -0
  25. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/docs-page.json +0 -0
  26. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/docs-rs.json +0 -0
  27. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/duckduckgo-search.json +0 -0
  28. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/github-issue.json +0 -0
  29. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/github-repo.json +0 -0
  30. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/google-search.json +0 -0
  31. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/npm-package.json +0 -0
  32. {searchfetch-3.2.3 → searchfetch-3.3.1}/templates/pypi-package.json +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: searchfetch
3
- Version: 3.2.3
3
+ Version: 3.3.1
4
4
  Summary: A maximum fault-tolerant, stealth-enabled MCP server for web searching and fetching.
5
5
  Author: Max
6
6
  License-Expression: MIT
@@ -11,12 +11,12 @@ Requires-Python: >=3.10
11
11
  Description-Content-Type: text/markdown
12
12
  License-File: LICENSE
13
13
  Requires-Dist: mcp>=1.28.1
14
- Requires-Dist: cloakbrowser>=0.4.5
14
+ Requires-Dist: cloakbrowser>=0.4.11
15
15
  Requires-Dist: beautifulsoup4>=4.15.0
16
- Requires-Dist: markdownify>=1.2.2
16
+ Requires-Dist: markdownify>=1.2.3
17
17
  Provides-Extra: dev
18
18
  Requires-Dist: pytest>=9.1.1; extra == "dev"
19
- Requires-Dist: ruff>=0.15.20; extra == "dev"
19
+ Requires-Dist: ruff>=0.15.22; extra == "dev"
20
20
  Dynamic: license-file
21
21
 
22
22
  # SearchFetch (MCP Server)
@@ -105,6 +105,13 @@ Template extraction supports `text`, `markdown`, `attribute`, and `html` formats
105
105
 
106
106
  Built-in templates live in `templates/*.json` and are shared by the Node.js and Python implementations.
107
107
 
108
+ **Available page templates** (auto-detected by URL or selectable by name):
109
+ `wikipedia`, `reddit`, `mdn-web-docs`, `gitlab`, `youtube`, `devto`, `go-pkg`, `javadoc`,
110
+ `github-repo`, `github-issue`, `npm-package`, `pypi-package`, `crates-package`,
111
+ `docker-hub`, `docs-rs`, `docs-page`
112
+
113
+ **`raw`** — special template that applies minimal filtering and returns full body content as markdown. Use when you need the complete page without template-specific extraction.
114
+
108
115
  ---
109
116
 
110
117
  ## Local Development
@@ -84,6 +84,13 @@ Template extraction supports `text`, `markdown`, `attribute`, and `html` formats
84
84
 
85
85
  Built-in templates live in `templates/*.json` and are shared by the Node.js and Python implementations.
86
86
 
87
+ **Available page templates** (auto-detected by URL or selectable by name):
88
+ `wikipedia`, `reddit`, `mdn-web-docs`, `gitlab`, `youtube`, `devto`, `go-pkg`, `javadoc`,
89
+ `github-repo`, `github-issue`, `npm-package`, `pypi-package`, `crates-package`,
90
+ `docker-hub`, `docs-rs`, `docs-page`
91
+
92
+ **`raw`** — special template that applies minimal filtering and returns full body content as markdown. Use when you need the complete page without template-specific extraction.
93
+
87
94
  ---
88
95
 
89
96
  ## Local Development
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "searchfetch"
7
- version = "3.2.3"
7
+ version = "3.3.1"
8
8
  description = "A maximum fault-tolerant, stealth-enabled MCP server for web searching and fetching."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -15,15 +15,15 @@ authors =[
15
15
  keywords =["mcp", "search", "fetch", "llm", "ai", "agent", "stealth"]
16
16
  dependencies =[
17
17
  "mcp>=1.28.1",
18
- "cloakbrowser>=0.4.5",
18
+ "cloakbrowser>=0.4.11",
19
19
  "beautifulsoup4>=4.15.0",
20
- "markdownify>=1.2.2"
20
+ "markdownify>=1.2.3"
21
21
  ]
22
22
 
23
23
  [project.optional-dependencies]
24
24
  dev = [
25
25
  "pytest>=9.1.1",
26
- "ruff>=0.15.20",
26
+ "ruff>=0.15.22",
27
27
  ]
28
28
 
29
29
  [project.urls]
@@ -38,18 +38,7 @@ searchfetch = "server:main"
38
38
  py-modules =["server"]
39
39
 
40
40
  [tool.setuptools.data-files]
41
- templates = [
42
- "templates/github-repo.json",
43
- "templates/github-issue.json",
44
- "templates/npm-package.json",
45
- "templates/pypi-package.json",
46
- "templates/crates-package.json",
47
- "templates/google-search.json",
48
- "templates/duckduckgo-search.json",
49
- "templates/docs-page.json",
50
- "templates/docs-rs.json",
51
- "templates/docker-hub.json",
52
- ]
41
+ templates = ["templates/*.json"]
53
42
 
54
43
  [tool.ruff]
55
44
  line-length = 100
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: searchfetch
3
- Version: 3.2.3
3
+ Version: 3.3.1
4
4
  Summary: A maximum fault-tolerant, stealth-enabled MCP server for web searching and fetching.
5
5
  Author: Max
6
6
  License-Expression: MIT
@@ -11,12 +11,12 @@ Requires-Python: >=3.10
11
11
  Description-Content-Type: text/markdown
12
12
  License-File: LICENSE
13
13
  Requires-Dist: mcp>=1.28.1
14
- Requires-Dist: cloakbrowser>=0.4.5
14
+ Requires-Dist: cloakbrowser>=0.4.11
15
15
  Requires-Dist: beautifulsoup4>=4.15.0
16
- Requires-Dist: markdownify>=1.2.2
16
+ Requires-Dist: markdownify>=1.2.3
17
17
  Provides-Extra: dev
18
18
  Requires-Dist: pytest>=9.1.1; extra == "dev"
19
- Requires-Dist: ruff>=0.15.20; extra == "dev"
19
+ Requires-Dist: ruff>=0.15.22; extra == "dev"
20
20
  Dynamic: license-file
21
21
 
22
22
  # SearchFetch (MCP Server)
@@ -105,6 +105,13 @@ Template extraction supports `text`, `markdown`, `attribute`, and `html` formats
105
105
 
106
106
  Built-in templates live in `templates/*.json` and are shared by the Node.js and Python implementations.
107
107
 
108
+ **Available page templates** (auto-detected by URL or selectable by name):
109
+ `wikipedia`, `reddit`, `mdn-web-docs`, `gitlab`, `youtube`, `devto`, `go-pkg`, `javadoc`,
110
+ `github-repo`, `github-issue`, `npm-package`, `pypi-package`, `crates-package`,
111
+ `docker-hub`, `docs-rs`, `docs-page`
112
+
113
+ **`raw`** — special template that applies minimal filtering and returns full body content as markdown. Use when you need the complete page without template-specific extraction.
114
+
108
115
  ---
109
116
 
110
117
  ## Local Development
@@ -10,12 +10,21 @@ searchfetch.egg-info/entry_points.txt
10
10
  searchfetch.egg-info/requires.txt
11
11
  searchfetch.egg-info/top_level.txt
12
12
  templates/crates-package.json
13
+ templates/devto.json
13
14
  templates/docker-hub.json
14
15
  templates/docs-page.json
15
16
  templates/docs-rs.json
16
17
  templates/duckduckgo-search.json
17
18
  templates/github-issue.json
18
19
  templates/github-repo.json
20
+ templates/gitlab.json
21
+ templates/go-pkg.json
19
22
  templates/google-search.json
23
+ templates/javadoc.json
24
+ templates/mdn-web-docs.json
20
25
  templates/npm-package.json
21
- templates/pypi-package.json
26
+ templates/pypi-package.json
27
+ templates/raw.json
28
+ templates/reddit.json
29
+ templates/wikipedia.json
30
+ templates/youtube.json
@@ -1,8 +1,8 @@
1
1
  mcp>=1.28.1
2
- cloakbrowser>=0.4.5
2
+ cloakbrowser>=0.4.11
3
3
  beautifulsoup4>=4.15.0
4
- markdownify>=1.2.2
4
+ markdownify>=1.2.3
5
5
 
6
6
  [dev]
7
7
  pytest>=9.1.1
8
- ruff>=0.15.20
8
+ ruff>=0.15.22
@@ -11,7 +11,7 @@ from markdownify import markdownify as md
11
11
  from mcp.server.fastmcp import FastMCP
12
12
  from playwright.async_api import TimeoutError as PlaywrightTimeoutError
13
13
 
14
- __version__ = "3.2.3"
14
+ __version__ = "3.3.1"
15
15
 
16
16
  mcp = FastMCP("searchfetch")
17
17
 
@@ -448,13 +448,16 @@ ACCESS_DENIED_TITLE_PATTERNS = [
448
448
  ]
449
449
 
450
450
  ACCESS_DENIED_BODY_TRIGGERS = [
451
- "captcha",
452
451
  "verify you are human",
453
452
  "are you a robot",
454
- "unusual traffic",
455
453
  "access denied",
454
+ ]
455
+
456
+ ACCESS_DENIED_STRONG_BODY_TRIGGERS = [
457
+ "our systems have detected unusual traffic",
456
458
  "sorry, you have been blocked",
457
459
  "to continue, please type the characters",
460
+ "bots use duckduckgo too",
458
461
  ]
459
462
 
460
463
  _ACCESS_DENIED_RE = re.compile("|".join(ACCESS_DENIED_TITLE_PATTERNS), re.IGNORECASE)
@@ -473,8 +476,10 @@ def _detect_access_failure(http_status: int | None, soup: BeautifulSoup) -> bool
473
476
  body = soup.find("body")
474
477
  if body:
475
478
  body_text = body.get_text(" ", strip=True)
479
+ lowered = body_text.lower()
480
+ if any(trigger in lowered for trigger in ACCESS_DENIED_STRONG_BODY_TRIGGERS):
481
+ return True
476
482
  if len(body_text) < 800:
477
- lowered = body_text.lower()
478
483
  if any(trigger in lowered for trigger in ACCESS_DENIED_BODY_TRIGGERS):
479
484
  return True
480
485
 
@@ -606,6 +611,7 @@ async def fetch_html(url: str, template: dict | None, block_media: bool, retry:
606
611
  or "econnreset" in exc_msg
607
612
  or "enotfound" in exc_msg
608
613
  or "navigation" in exc_msg
614
+ or "navigating" in exc_msg
609
615
  )
610
616
  if attempt < attempts - 1 and is_network:
611
617
  await asyncio.sleep(0.5)
@@ -1033,30 +1039,10 @@ def generic_fallback(html_content: str, _url: str = "") -> str:
1033
1039
  # ---------------------------------------------------------------------------
1034
1040
 
1035
1041
 
1036
- @mcp.tool()
1037
- async def websearch(
1038
- query: str,
1039
- engine: str = "duckduckgo",
1040
- region: str | None = None,
1041
- safe_search: bool | None = None,
1042
- max_results: int = 10,
1043
- block_media: bool = True,
1044
- ) -> str:
1045
- """
1046
- Search the web using template-driven search engines (Google, DuckDuckGo, or custom).
1047
-
1048
- Args:
1049
- query: The search query string.
1050
- engine: Search engine: 'duckduckgo' or 'google'. Also accepts custom template names.
1051
- region: Region code. DDG: 'us-en', 'de-de', etc. Google: 'de-de' maps to hl=de, gl=de.
1052
- safe_search: Enable safe search. Maps to engine-specific params.
1053
- max_results: Max results to extract. Default: 10.
1054
- block_media: Block images/media/fonts. Default: True.
1055
- """
1056
- # 1. Resolve engine → template name
1042
+ def _resolve_search_candidate(
1043
+ engine: str, query: str, region: str | None, safe_search: bool | None
1044
+ ) -> tuple[str, dict, str]:
1057
1045
  template_name = resolve_search_template(engine)
1058
-
1059
- # 2. Resolve the template object
1060
1046
  if template_name.startswith("{"):
1061
1047
  try:
1062
1048
  template = json.loads(template_name)
@@ -1070,33 +1056,116 @@ async def websearch(
1070
1056
  )
1071
1057
  template = BUILTIN_TEMPLATES[template_name]
1072
1058
 
1073
- # Validate url_template exists
1074
1059
  if not template.get("url_template"):
1075
1060
  raise ValueError(
1076
1061
  f"Template '{template.get('name', template_name)}' is not a search "
1077
1062
  "template (no url_template). Use webfetch for page templates."
1078
1063
  )
1079
1064
 
1080
- # 3. Map universal params to engine-specific url_params
1081
1065
  engine_key = engine if engine in ("duckduckgo", "google") else "custom"
1082
- url_params = map_search_params(query, engine_key, region, safe_search)
1083
-
1084
- # 4. Resolve URL template
1085
- resolved_url = resolve_url_template(template, url_params)
1086
-
1087
- # Google safe_search: append &safe=active after URL resolution
1066
+ params = map_search_params(query, engine_key, region, safe_search)
1067
+ resolved_url = resolve_url_template(template, params)
1088
1068
  if engine_key == "google" and safe_search is True:
1089
1069
  resolved_url += "&safe=active"
1070
+ return engine, template, resolved_url
1071
+
1072
+
1073
+ def _resolve_duckduckgo_lite_candidate(
1074
+ query: str, region: str | None, safe_search: bool | None
1075
+ ) -> tuple[str, dict, str]:
1076
+ template = dict(BUILTIN_TEMPLATES["duckduckgo-search"])
1077
+ template.update(
1078
+ {
1079
+ "name": "duckduckgo-lite-search",
1080
+ "url_template": "https://lite.duckduckgo.com/lite/?q={query}&kl={kl}&kp={kp}",
1081
+ "sections": [
1082
+ {
1083
+ "name": "Results",
1084
+ "selector": "a.result-link",
1085
+ "multiple": True,
1086
+ "max_items": 10,
1087
+ "children": [
1088
+ {"name": "Title", "selector": "", "format": "text"},
1089
+ {
1090
+ "name": "URL",
1091
+ "selector": "",
1092
+ "format": "attribute",
1093
+ "attribute": "href",
1094
+ "transform": "decode_ddg_url",
1095
+ },
1096
+ ],
1097
+ }
1098
+ ],
1099
+ }
1100
+ )
1101
+ params = map_search_params(query, "duckduckgo", region, safe_search)
1102
+ return "duckduckgo-lite", template, resolve_url_template(template, params)
1103
+
1104
+
1105
+ def _resolve_search_candidates(
1106
+ engine: str, query: str, region: str | None, safe_search: bool | None
1107
+ ) -> list[tuple[str, dict, str]]:
1108
+ primary = _resolve_search_candidate(engine, query, region, safe_search)
1109
+ if engine == "google":
1110
+ return [
1111
+ primary,
1112
+ _resolve_search_candidate("duckduckgo", query, region, safe_search),
1113
+ _resolve_duckduckgo_lite_candidate(query, region, safe_search),
1114
+ ]
1115
+ if engine == "duckduckgo":
1116
+ return [primary, _resolve_duckduckgo_lite_candidate(query, region, safe_search)]
1117
+ return [primary]
1090
1118
 
1091
- # 5. Fetch
1092
- html = await fetch_html(resolved_url, template, block_media, retry=True)
1093
1119
 
1094
- # 6. Parse and extract
1095
- soup = BeautifulSoup(html, "html.parser")
1096
- sections_data = extract_template(soup, template, origin="", max_results_override=max_results)
1120
+ @mcp.tool()
1121
+ async def websearch(
1122
+ query: str,
1123
+ engine: str = "duckduckgo",
1124
+ region: str | None = None,
1125
+ safe_search: bool | None = None,
1126
+ max_results: int = 10,
1127
+ block_media: bool = True,
1128
+ ) -> str:
1129
+ """
1130
+ Search the web using template-driven search engines (Google, DuckDuckGo, or custom).
1097
1131
 
1098
- # 7. Compose output
1099
- return compose_search_results(sections_data, template.get("name", template_name))
1132
+ Args:
1133
+ query: The search query string.
1134
+ engine: Search engine: 'duckduckgo' or 'google'. Also accepts custom template names.
1135
+ region: Region code. DDG: 'us-en', 'de-de', etc. Google: 'de-de' maps to hl=de, gl=de.
1136
+ safe_search: Enable safe search. Maps to engine-specific params.
1137
+ max_results: Max results to extract. Default: 10.
1138
+ block_media: Block images/media/fonts. Default: True.
1139
+ """
1140
+ candidates = _resolve_search_candidates(engine, query, region, safe_search)
1141
+ last_exc = None
1142
+
1143
+ for selected_engine, template, resolved_url in candidates:
1144
+ try:
1145
+ html = await fetch_html(resolved_url, template, block_media, retry=True)
1146
+ soup = BeautifulSoup(html, "html.parser")
1147
+ sections_data = extract_template(
1148
+ soup, template, origin="", max_results_override=max_results
1149
+ )
1150
+ result_section = next(
1151
+ (section for section in sections_data if section.get("type") == "search_results"),
1152
+ None,
1153
+ )
1154
+ if result_section and result_section.get("items"):
1155
+ result = compose_search_results(
1156
+ sections_data, template.get("name", selected_engine)
1157
+ )
1158
+ if selected_engine != engine:
1159
+ result = (
1160
+ f"[websearch: {engine} was unavailable; results are from "
1161
+ f"{selected_engine}.]\n\n{result}"
1162
+ )
1163
+ return result
1164
+ last_exc = RuntimeError(f"No results extracted from {selected_engine}.")
1165
+ except Exception as exc:
1166
+ last_exc = exc
1167
+
1168
+ raise last_exc or RuntimeError("No search results found.")
1100
1169
 
1101
1170
 
1102
1171
  @mcp.tool()
@@ -1108,14 +1177,25 @@ async def webfetch(
1108
1177
  block_media: bool = True,
1109
1178
  ) -> str:
1110
1179
  """
1111
- Fetch a webpage and extract structured information using templates.
1180
+ Fetch and extract the main text content from any webpage. Fully executes
1181
+ JavaScript to load React/SPAs and aggressively strips images/media
1182
+ (including base64) to save context tokens.
1112
1183
 
1113
1184
  Args:
1114
- url: The full URL to fetch.
1115
- template: "auto", a built-in name, or inline JSON. Default: "auto".
1185
+ url: The full URL of the webpage to fetch (must start with http/https).
1186
+ template: Template to use: 'auto' (auto-detect from URL), a built-in
1187
+ page template name (see list below), 'raw' for
1188
+ minimal-filtering full-page output, or inline JSON.
1189
+ Default: 'auto'.
1116
1190
  start_index: Character offset for pagination. Default: 0.
1117
- max_length: Max characters to return. Default: 10000.
1118
- block_media: Block images/media/fonts. Default: True.
1191
+ max_length: Maximum characters to return per request. Default is 10000.
1192
+ block_media: Block images, videos, and fonts entirely at the network
1193
+ layer. Default is true.
1194
+
1195
+ Built-in page templates (use by name to force a specific extractor):
1196
+ wikipedia, reddit, mdn-web-docs, gitlab, youtube, devto, go-pkg,
1197
+ javadoc, github-repo, github-issue, npm-package, pypi-package,
1198
+ crates-package, docker-hub, docs-rs, docs-page, raw
1119
1199
  """
1120
1200
  # 1. Resolve template
1121
1201
  matched_template, template_name = resolve_page_template(url, template)
@@ -0,0 +1,45 @@
1
+ {
2
+ "name": "devto",
3
+ "description": "DEV Community article — title, author, and content",
4
+ "order": 20,
5
+ "url_patterns": [
6
+ "^https?://dev\\.to/[^/]+/[^/]+"
7
+ ],
8
+ "remove": [
9
+ "script",
10
+ "style",
11
+ "svg",
12
+ "img",
13
+ "nav",
14
+ "footer",
15
+ ".sidebar",
16
+ ".crayons-layout__sidebar-left",
17
+ ".crayons-layout__sidebar-right",
18
+ ".crayons-article-actions",
19
+ ".reaction-button",
20
+ ".crayons-btn",
21
+ ".spec__tags",
22
+ ".article-engagement-bar",
23
+ ".comments-container"
24
+ ],
25
+ "sections": [
26
+ {
27
+ "name": "Title",
28
+ "selector": "h1",
29
+ "format": "text",
30
+ "required": true
31
+ },
32
+ {
33
+ "name": "Author",
34
+ "selector": ".crayons-article__header__meta a, a.crayons-avatar",
35
+ "format": "text",
36
+ "required": false
37
+ },
38
+ {
39
+ "name": "Content",
40
+ "selector": "#article-body, .crayons-article__body, article.crayons-article",
41
+ "format": "markdown",
42
+ "required": true
43
+ }
44
+ ]
45
+ }
@@ -0,0 +1,49 @@
1
+ {
2
+ "name": "gitlab",
3
+ "description": "GitLab project — description, file listing, and README",
4
+ "order": 12,
5
+ "url_patterns": [
6
+ "^https?://gitlab\\.com/[^/]+/[^/]+/?$",
7
+ "^https?://gitlab\\.com/[^/]+/[^/]+/-/tree/[^/]+/?$"
8
+ ],
9
+ "remove": [
10
+ "script",
11
+ "style",
12
+ "svg",
13
+ "nav",
14
+ "footer",
15
+ "header",
16
+ ".navbar-gitlab",
17
+ ".nav-sidebar",
18
+ ".issuable-list",
19
+ ".super-sidebar",
20
+ "aside"
21
+ ],
22
+ "sections": [
23
+ {
24
+ "name": "Title",
25
+ "selector": "h1",
26
+ "format": "text",
27
+ "required": true
28
+ },
29
+ {
30
+ "name": "Description",
31
+ "selector": "meta[name='description']",
32
+ "format": "attribute",
33
+ "attribute": "content",
34
+ "required": false
35
+ },
36
+ {
37
+ "name": "Files",
38
+ "selector": "table[data-qa-selector='file_tree_table'], .tree-table",
39
+ "format": "markdown",
40
+ "required": false
41
+ },
42
+ {
43
+ "name": "README",
44
+ "selector": ".readme-holder .md, .file-content .md, .readme-holder",
45
+ "format": "markdown",
46
+ "required": false
47
+ }
48
+ ]
49
+ }
@@ -0,0 +1,50 @@
1
+ {
2
+ "name": "go-pkg",
3
+ "description": "Go package documentation on pkg.go.dev — name, overview, and docs",
4
+ "order": 22,
5
+ "url_patterns": [
6
+ "^https?://pkg\\.go\\.dev/[^/]+/?$",
7
+ "^https?://pkg\\.go\\.dev/[^/]+/[^/]+/?$",
8
+ "^https?://pkg\\.go\\.dev/[^/]+/[^/]+/[^/]+/?$"
9
+ ],
10
+ "remove": [
11
+ "script",
12
+ "style",
13
+ "svg",
14
+ "nav",
15
+ "footer",
16
+ ".go-Header",
17
+ ".go-SiteHeader",
18
+ ".go-Footer",
19
+ ".UnitMeta",
20
+ ".UnitMeta-repo",
21
+ ".UnitFiles",
22
+ ".JumpDrawer",
23
+ ".go-Main-nav",
24
+ ".go-Main-headerSubmenu",
25
+ ".Site-header",
26
+ ".UnitHeader-version",
27
+ ".UnitHeader-breadcrumb",
28
+ ".UnitHeader-details"
29
+ ],
30
+ "sections": [
31
+ {
32
+ "name": "Title",
33
+ "selector": "h1",
34
+ "format": "text",
35
+ "required": true
36
+ },
37
+ {
38
+ "name": "Overview",
39
+ "selector": ".Documentation-overview, .Overview",
40
+ "format": "markdown",
41
+ "required": false
42
+ },
43
+ {
44
+ "name": "Documentation",
45
+ "selector": ".Documentation-content, .Documentation, #documentation",
46
+ "format": "markdown",
47
+ "required": false
48
+ }
49
+ ]
50
+ }
@@ -0,0 +1,43 @@
1
+ {
2
+ "name": "javadoc",
3
+ "description": "Java SE API documentation — class/method details",
4
+ "order": 8.5,
5
+ "url_patterns": [
6
+ "^https?://docs\\.oracle\\.com/.*/docs/api/.*"
7
+ ],
8
+ "remove": [
9
+ "script",
10
+ "style",
11
+ "svg",
12
+ "nav",
13
+ "footer",
14
+ "header",
15
+ ".topNav",
16
+ ".bottomNav",
17
+ ".subNav",
18
+ ".navListSearch",
19
+ ".skipNav",
20
+ ".searchTag",
21
+ ".searchTagResult",
22
+ ".useSummary",
23
+ ".inheritedList",
24
+ ".details",
25
+ ".memberSummary",
26
+ ".colFirst",
27
+ ".colLast"
28
+ ],
29
+ "sections": [
30
+ {
31
+ "name": "Title",
32
+ "selector": ".header h2, h2.title, h2:first-of-type",
33
+ "format": "text",
34
+ "required": false
35
+ },
36
+ {
37
+ "name": "Content",
38
+ "selector": "[role='main'], .document, .rst-content, article, .content, .markdown-body, .prose, main, .contentContainer",
39
+ "format": "markdown",
40
+ "required": true
41
+ }
42
+ ]
43
+ }
@@ -0,0 +1,39 @@
1
+ {
2
+ "name": "mdn-web-docs",
3
+ "description": "MDN Web Docs article — title and main content",
4
+ "order": 15,
5
+ "url_patterns": [
6
+ "^https?://developer\\.mozilla\\.org/.*"
7
+ ],
8
+ "remove": [
9
+ "script",
10
+ "style",
11
+ "svg",
12
+ "nav",
13
+ "footer",
14
+ "header",
15
+ ".sidebar",
16
+ ".toc",
17
+ ".interactive",
18
+ ".playground",
19
+ ".bc-data",
20
+ ".prev-next",
21
+ ".notecard",
22
+ ".language-menu",
23
+ ".breadcrumbs-container"
24
+ ],
25
+ "sections": [
26
+ {
27
+ "name": "Title",
28
+ "selector": "h1",
29
+ "format": "text",
30
+ "required": true
31
+ },
32
+ {
33
+ "name": "Content",
34
+ "selector": "#content, article, main, .main-page-content",
35
+ "format": "markdown",
36
+ "required": true
37
+ }
38
+ ]
39
+ }
@@ -0,0 +1,14 @@
1
+ {
2
+ "name": "raw",
3
+ "description": "Raw page content — minimal filtering, full body as markdown (use when you need the complete page without template extraction)",
4
+ "order": 999,
5
+ "remove": [],
6
+ "sections": [
7
+ {
8
+ "name": "Content",
9
+ "selector": "body",
10
+ "format": "markdown",
11
+ "required": false
12
+ }
13
+ ]
14
+ }
@@ -0,0 +1,65 @@
1
+ {
2
+ "name": "reddit",
3
+ "description": "Reddit post — title, content, and comments",
4
+ "order": 13,
5
+ "url_patterns": [
6
+ "^https?://(?:www\\.|old\\.)?reddit\\.com/r/[^/]+/comments/[^/]+"
7
+ ],
8
+ "remove": [
9
+ "script",
10
+ "style",
11
+ "svg",
12
+ "img",
13
+ "nav",
14
+ "footer",
15
+ "header",
16
+ ".promoted-post",
17
+ "shreddit-ad-post",
18
+ ".side",
19
+ ".sidebar",
20
+ ".footer-parent",
21
+ "#header",
22
+ ".sr-header-area",
23
+ ".menuarea",
24
+ ".linkinfo",
25
+ ".panestack-title",
26
+ ".report-button",
27
+ "shreddit-join-button",
28
+ "#overlayScrollContainer",
29
+ "#right-sidebar-container",
30
+ "#left-sidebar-container"
31
+ ],
32
+ "sections": [
33
+ {
34
+ "name": "Subreddit",
35
+ "selector": "a[href*='/r/'], shreddit-post a[href*='/r/']",
36
+ "format": "text",
37
+ "required": false
38
+ },
39
+ {
40
+ "name": "Title",
41
+ "selector": "h1, shreddit-post [slot='title']",
42
+ "format": "text",
43
+ "required": true
44
+ },
45
+ {
46
+ "name": "Author",
47
+ "selector": "a[href*='/user/'], shreddit-post a[href*='/user/'], [slot='authorName']",
48
+ "format": "text",
49
+ "required": false
50
+ },
51
+ {
52
+ "name": "Link",
53
+ "selector": "a[slot='external-anchor'], a.outbound-link, a[data-testid='outbound-link']",
54
+ "format": "attribute",
55
+ "attribute": "href",
56
+ "required": false
57
+ },
58
+ {
59
+ "name": "Content",
60
+ "selector": "[slot='text-body'], .text-body, .expando, .usertext-body .md",
61
+ "format": "markdown",
62
+ "required": false
63
+ }
64
+ ]
65
+ }
@@ -0,0 +1,46 @@
1
+ {
2
+ "name": "wikipedia",
3
+ "description": "Wikipedia article — title and main content",
4
+ "order": 11,
5
+ "url_patterns": [
6
+ "^https?://[a-z]{2,4}\\.wikipedia\\.org/wiki/[^:]+$"
7
+ ],
8
+ "remove": [
9
+ "script",
10
+ "style",
11
+ "svg",
12
+ "nav",
13
+ "footer",
14
+ ".mw-editsection",
15
+ ".reflist",
16
+ ".navbox",
17
+ ".sisterproject",
18
+ ".metadata",
19
+ ".sidebar",
20
+ ".infobox",
21
+ "#toc",
22
+ ".mw-jump-link",
23
+ ".printfooter",
24
+ ".catlinks",
25
+ ".mw-authority-control",
26
+ ".shortdescription",
27
+ ".mw-indicators",
28
+ ".page-actions",
29
+ ".mw-page-title-namespace",
30
+ ".mw-heading-subtitle"
31
+ ],
32
+ "sections": [
33
+ {
34
+ "name": "Title",
35
+ "selector": "#firstHeading, h1",
36
+ "format": "text",
37
+ "required": true
38
+ },
39
+ {
40
+ "name": "Content",
41
+ "selector": "#mw-content-text .mw-parser-output",
42
+ "format": "markdown",
43
+ "required": true
44
+ }
45
+ ]
46
+ }
@@ -0,0 +1,44 @@
1
+ {
2
+ "name": "youtube",
3
+ "description": "YouTube video — title, channel, description",
4
+ "order": 30,
5
+ "url_patterns": [
6
+ "^https?://(?:www\\.)?(?:youtube\\.com/watch\\?v=|youtu\\.be/).+"
7
+ ],
8
+ "remove": [
9
+ "script",
10
+ "style",
11
+ "svg",
12
+ "img",
13
+ "nav",
14
+ "header",
15
+ "ytd-guide-renderer",
16
+ "ytd-mini-guide-renderer",
17
+ "#comments",
18
+ "#related",
19
+ "ytd-comments",
20
+ "ytd-live-chat-frame",
21
+ ".ytp-ad-module"
22
+ ],
23
+ "sections": [
24
+ {
25
+ "name": "Title",
26
+ "selector": "meta[name='title']",
27
+ "format": "attribute",
28
+ "attribute": "content",
29
+ "required": true
30
+ },
31
+ {
32
+ "name": "Channel",
33
+ "selector": "#owner a, ytd-video-owner-renderer a, #channel-name a",
34
+ "format": "text",
35
+ "required": false
36
+ },
37
+ {
38
+ "name": "Description",
39
+ "selector": "#description-inline-expander, ytd-text-inline-expander, #info-container",
40
+ "format": "markdown",
41
+ "required": false
42
+ }
43
+ ]
44
+ }
File without changes
File without changes
File without changes