webget-cli 0.9.0__tar.gz → 0.11.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {webget_cli-0.9.0 → webget_cli-0.11.0}/PKG-INFO +1 -1
- {webget_cli-0.9.0 → webget_cli-0.11.0}/pyproject.toml +3 -2
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_mcp.py +5 -5
- webget_cli-0.11.0/tests/test_discovery_map.py +29 -0
- webget_cli-0.11.0/tests/test_ladder_retry.py +41 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_login_flow.py +18 -7
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_mcp_leak_review.py +2 -2
- webget_cli-0.11.0/tests/test_mcp_map.py +18 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_mcp_profile.py +3 -3
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_mcp_server.py +9 -9
- webget_cli-0.11.0/tests/test_ssrf_dual_dns.py +34 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_webget.py +25 -10
- webget_cli-0.11.0/webget/__init__.py +149 -0
- webget_cli-0.11.0/webget/cache.py +188 -0
- webget_cli-0.11.0/webget/cli.py +407 -0
- webget_cli-0.11.0/webget/discovery.py +94 -0
- webget_cli-0.11.0/webget/firecrawl.py +61 -0
- webget_cli-0.11.0/webget/http.py +174 -0
- webget_cli-0.11.0/webget/ladder.py +613 -0
- webget_cli-0.11.0/webget/profile.py +410 -0
- webget_cli-0.11.0/webget/search.py +38 -0
- webget_cli-0.11.0/webget/ssrf.py +280 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/PKG-INFO +1 -1
- {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/SOURCES.txt +14 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/entry_points.txt +1 -1
- {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/top_level.txt +1 -0
- webget_cli-0.11.0/webget_cli.py +58 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_mcp.py +22 -0
- webget_cli-0.9.0/webget_cli.py +0 -1762
- {webget_cli-0.9.0 → webget_cli-0.11.0}/LICENSE +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/README.md +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/setup.cfg +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_auth.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_cache.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_concurrency.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_http.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_ssrf.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_auth_review.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_browser_ssrf.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_cache_review.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_concurrency_review.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_extraction_markdown.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_firecrawl_policy.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_integration_ladder.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_mcp_smoke.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_security_review.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_size_review.py +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/dependency_links.txt +0 -0
- {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/requires.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: webget-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.11.0
|
|
4
4
|
Summary: Local search + scrape CLI with an HTTP fast path, optional browser fallback, and authenticated session profiles. Zero API keys.
|
|
5
5
|
Author-email: David Tarigan <tarigansdavid@gmail.com>
|
|
6
6
|
License: Apache-2.0
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "webget-cli"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.11.0"
|
|
8
8
|
description = "Local search + scrape CLI with an HTTP fast path, optional browser fallback, and authenticated session profiles. Zero API keys."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11"
|
|
@@ -39,11 +39,12 @@ mcp = ["fastmcp>=2"]
|
|
|
39
39
|
dev = ["pytest>=8", "ruff>=0.6"]
|
|
40
40
|
|
|
41
41
|
[project.scripts]
|
|
42
|
-
webget = "
|
|
42
|
+
webget = "webget.cli:main"
|
|
43
43
|
webget-mcp = "webget_mcp:main"
|
|
44
44
|
|
|
45
45
|
[tool.setuptools]
|
|
46
46
|
py-modules = ["webget_cli", "webget_mcp"]
|
|
47
|
+
packages = ["webget"]
|
|
47
48
|
|
|
48
49
|
[tool.pytest.ini_options]
|
|
49
50
|
testpaths = ["tests"]
|
|
@@ -58,7 +58,7 @@ class TestMalformedArguments:
|
|
|
58
58
|
return res
|
|
59
59
|
|
|
60
60
|
res = _run(run())
|
|
61
|
-
assert res.
|
|
61
|
+
assert res.isError
|
|
62
62
|
|
|
63
63
|
|
|
64
64
|
class TestInvalidURLs:
|
|
@@ -72,7 +72,7 @@ class TestInvalidURLs:
|
|
|
72
72
|
return res
|
|
73
73
|
|
|
74
74
|
res = _run(run())
|
|
75
|
-
assert "error" in res.content[0].text or res.
|
|
75
|
+
assert "error" in res.content[0].text or res.isError
|
|
76
76
|
|
|
77
77
|
|
|
78
78
|
class TestRepeatedCalls:
|
|
@@ -86,7 +86,7 @@ class TestRepeatedCalls:
|
|
|
86
86
|
"fetch",
|
|
87
87
|
{"url": "https://example.com", "strategy": "http", "no_cache": True},
|
|
88
88
|
)
|
|
89
|
-
if res.
|
|
89
|
+
if res.isError:
|
|
90
90
|
return "ERROR"
|
|
91
91
|
return "OK"
|
|
92
92
|
|
|
@@ -106,7 +106,7 @@ class TestRepeatedCalls:
|
|
|
106
106
|
for _ in range(5)
|
|
107
107
|
]
|
|
108
108
|
)
|
|
109
|
-
return [r.
|
|
109
|
+
return [r.isError for r in results]
|
|
110
110
|
|
|
111
111
|
assert _run(run()) == [False] * 5
|
|
112
112
|
|
|
@@ -133,7 +133,7 @@ class TestInputCaps:
|
|
|
133
133
|
return res
|
|
134
134
|
|
|
135
135
|
res = _run(run())
|
|
136
|
-
assert "must be between" in res.content[0].text or res.
|
|
136
|
+
assert "must be between" in res.content[0].text or res.isError
|
|
137
137
|
|
|
138
138
|
|
|
139
139
|
class TestToolFailureIsolation:
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
|
|
3
|
+
from webget.discovery import _extract_sitemap_urls, discover_urls
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_extract_sitemap_urls_standard_xml():
|
|
7
|
+
xml = """<?xml version="1.0" encoding="UTF-8"?>
|
|
8
|
+
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
|
|
9
|
+
<url>
|
|
10
|
+
<loc>https://example.com/page1</loc>
|
|
11
|
+
</url>
|
|
12
|
+
<url>
|
|
13
|
+
<loc>https://example.com/page2</loc>
|
|
14
|
+
</url>
|
|
15
|
+
</urlset>
|
|
16
|
+
"""
|
|
17
|
+
urls = _extract_sitemap_urls(xml)
|
|
18
|
+
assert urls == ["https://example.com/page1", "https://example.com/page2"]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_extract_sitemap_urls_malformed_regex_fallback():
|
|
22
|
+
xml = """<urlset><url><loc>https://example.com/broken1</loc></unclosed>"""
|
|
23
|
+
urls = _extract_sitemap_urls(xml)
|
|
24
|
+
assert "https://example.com/broken1" in urls
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_discover_urls_private_target_blocked():
|
|
28
|
+
urls = asyncio.run(discover_urls("http://127.0.0.1/sitemap.xml", allow_private=False))
|
|
29
|
+
assert urls == []
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
from unittest.mock import patch
|
|
3
|
+
|
|
4
|
+
from webget.ladder import scrape_many
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def test_scrape_many_retry_transient_timeout(fresh_cache, server):
|
|
8
|
+
calls = []
|
|
9
|
+
|
|
10
|
+
async def mock_fetch(url, *args, **kwargs):
|
|
11
|
+
calls.append(url)
|
|
12
|
+
if len(calls) == 1:
|
|
13
|
+
raise TimeoutError("timeout")
|
|
14
|
+
return {
|
|
15
|
+
"title": "Success After Retry",
|
|
16
|
+
"markdown": "Valid content length " * 10,
|
|
17
|
+
"status": "success",
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
async def run_without_retry():
|
|
21
|
+
return await scrape_many(
|
|
22
|
+
[server.url("/test")], strategy="http", per_url_timeout=1, retry_transient=False
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
async def run_with_retry():
|
|
26
|
+
return await scrape_many(
|
|
27
|
+
[server.url("/test")], strategy="http", per_url_timeout=1, retry_transient=True
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
with patch("webget.ladder._resolve_fetch_http", return_value=mock_fetch):
|
|
31
|
+
# Without retry flag -> error on first timeout
|
|
32
|
+
res = asyncio.run(run_without_retry())
|
|
33
|
+
target = server.url("/test")
|
|
34
|
+
assert res[target]["status"] == "error"
|
|
35
|
+
assert res[target]["attempts"] == 1
|
|
36
|
+
|
|
37
|
+
# With retry_transient=True -> retries and succeeds on 2nd attempt
|
|
38
|
+
calls.clear()
|
|
39
|
+
res = asyncio.run(run_with_retry())
|
|
40
|
+
assert res[target]["status"] == "success"
|
|
41
|
+
assert res[target]["attempts"] == 2
|
|
@@ -102,7 +102,9 @@ def _spawn_mcp(profile_root):
|
|
|
102
102
|
"WEBGET_PROFILE_DIR": str(profile_root),
|
|
103
103
|
"WEBGET_ALLOW_PRIVATE": "1",
|
|
104
104
|
}
|
|
105
|
-
return StdioServerParameters(
|
|
105
|
+
return StdioServerParameters(
|
|
106
|
+
command=sys.executable, args=[str(ROOT / "webget_mcp.py")], env=env
|
|
107
|
+
)
|
|
106
108
|
|
|
107
109
|
|
|
108
110
|
def test_mcp_login_tool_persists_and_fetch_uses_it(server, tmp_path):
|
|
@@ -117,14 +119,18 @@ def test_mcp_login_tool_persists_and_fetch_uses_it(server, tmp_path):
|
|
|
117
119
|
gated_url = server.url("/cookie-gated")
|
|
118
120
|
|
|
119
121
|
async def run():
|
|
120
|
-
async with
|
|
122
|
+
async with (
|
|
123
|
+
stdio_client(_spawn_mcp(root)) as (read, write),
|
|
124
|
+
ClientSession(read, write) as session,
|
|
125
|
+
):
|
|
121
126
|
await session.initialize()
|
|
122
127
|
res = await session.call_tool(
|
|
123
128
|
"login",
|
|
124
129
|
{"url": login_url, "profile": "mcplogin", "headless": True, "wait_seconds": 30},
|
|
125
130
|
)
|
|
126
131
|
authed = await session.call_tool(
|
|
127
|
-
"fetch",
|
|
132
|
+
"fetch",
|
|
133
|
+
{"url": gated_url, "strategy": "http", "no_cache": True, "profile": "mcplogin"},
|
|
128
134
|
)
|
|
129
135
|
anon = await session.call_tool(
|
|
130
136
|
"fetch", {"url": gated_url, "strategy": "http", "no_cache": True}
|
|
@@ -133,7 +139,7 @@ def test_mcp_login_tool_persists_and_fetch_uses_it(server, tmp_path):
|
|
|
133
139
|
|
|
134
140
|
res, authed, anon = asyncio.run(asyncio.wait_for(run(), timeout=90))
|
|
135
141
|
res_text = res.content[0].text if res.content else ""
|
|
136
|
-
assert not res.
|
|
142
|
+
assert not res.isError, res_text
|
|
137
143
|
assert '"status":"success"' in res_text, res_text
|
|
138
144
|
assert '"profile":"mcplogin"' in res_text
|
|
139
145
|
|
|
@@ -155,16 +161,21 @@ def test_mcp_login_rejects_bad_url_and_profile(server, tmp_path):
|
|
|
155
161
|
root = _profile_root(tmp_path)
|
|
156
162
|
|
|
157
163
|
async def run():
|
|
158
|
-
async with
|
|
164
|
+
async with (
|
|
165
|
+
stdio_client(_spawn_mcp(root)) as (read, write),
|
|
166
|
+
ClientSession(read, write) as session,
|
|
167
|
+
):
|
|
159
168
|
await session.initialize()
|
|
160
169
|
bad_url = await session.call_tool(
|
|
161
170
|
"login", {"url": "not a url", "profile": "x", "headless": True, "wait_seconds": 5}
|
|
162
171
|
)
|
|
163
172
|
bad_name = await session.call_tool(
|
|
164
|
-
"login",
|
|
173
|
+
"login",
|
|
174
|
+
{"url": server.url("/"), "profile": "../evil", "headless": True, "wait_seconds": 5},
|
|
165
175
|
)
|
|
166
176
|
bad_secs = await session.call_tool(
|
|
167
|
-
"login",
|
|
177
|
+
"login",
|
|
178
|
+
{"url": server.url("/"), "profile": "x", "headless": True, "wait_seconds": 1},
|
|
168
179
|
)
|
|
169
180
|
return bad_url, bad_name, bad_secs
|
|
170
181
|
|
|
@@ -64,7 +64,7 @@ class TestNoSecretLeakage:
|
|
|
64
64
|
return res
|
|
65
65
|
|
|
66
66
|
res = _run(run())
|
|
67
|
-
if res.
|
|
67
|
+
if res.isError:
|
|
68
68
|
return # error path: no payload to leak, still fine
|
|
69
69
|
text = res.content[0].text
|
|
70
70
|
hits = _scan(text)
|
|
@@ -121,7 +121,7 @@ class TestServerRecovery:
|
|
|
121
121
|
bad = await session.call_tool(
|
|
122
122
|
"fetch", {"url": "https://example.com", "strategy": "bogus"}
|
|
123
123
|
)
|
|
124
|
-
assert not bad.
|
|
124
|
+
assert not bad.isError or "error" in bad.content[0].text
|
|
125
125
|
good = await session.call_tool(
|
|
126
126
|
"fetch", {"url": "https://example.com", "strategy": "http", "no_cache": True}
|
|
127
127
|
)
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
from unittest.mock import AsyncMock, patch
|
|
3
|
+
|
|
4
|
+
from webget_mcp import map as mcp_map
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def test_mcp_map_clamps_limit():
|
|
8
|
+
res = asyncio.run(mcp_map("https://example.com", limit=-5))
|
|
9
|
+
assert len(res) == 1
|
|
10
|
+
assert "error: limit must be between" in res[0]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def test_mcp_map_calls_discover_urls():
|
|
14
|
+
with patch("webget_cli.discover_urls", new_callable=AsyncMock) as mock_disc:
|
|
15
|
+
mock_disc.return_value = ["https://example.com/p1", "https://example.com/p2"]
|
|
16
|
+
res = asyncio.run(mcp_map("https://example.com", limit=50))
|
|
17
|
+
assert res == ["https://example.com/p1", "https://example.com/p2"]
|
|
18
|
+
mock_disc.assert_awaited_once_with("https://example.com", limit=50, timeout=15)
|
|
@@ -79,7 +79,7 @@ def test_mcp_fetch_with_profile_uses_session(tmp_path):
|
|
|
79
79
|
ok, anon = asyncio.run(asyncio.wait_for(run(), timeout=60))
|
|
80
80
|
ok_text = ok.content[0].text if ok.content else ""
|
|
81
81
|
anon_text = anon.content[0].text if anon.content else ""
|
|
82
|
-
assert not ok.
|
|
82
|
+
assert not ok.isError, ok_text
|
|
83
83
|
assert '"status":"success"' in ok_text, ok_text
|
|
84
84
|
assert "you are authenticated" in ok_text, ok_text
|
|
85
85
|
assert '"authenticated":true' in ok_text, ok_text # session was used
|
|
@@ -105,7 +105,7 @@ def test_mcp_fetch_unknown_profile_is_hard_error(tmp_path):
|
|
|
105
105
|
return res
|
|
106
106
|
|
|
107
107
|
res = asyncio.run(asyncio.wait_for(run(), timeout=60))
|
|
108
|
-
assert not res.
|
|
108
|
+
assert not res.isError
|
|
109
109
|
assert "profile 'ghost' not found" in res.content[0].text
|
|
110
110
|
|
|
111
111
|
|
|
@@ -150,7 +150,7 @@ def test_mcp_fetch_enriched_output_present(tmp_path):
|
|
|
150
150
|
anon_text = anon.content[0].text if anon.content else ""
|
|
151
151
|
|
|
152
152
|
# Authenticated: standard assertions
|
|
153
|
-
assert not ok.
|
|
153
|
+
assert not ok.isError, ok_text
|
|
154
154
|
assert '"status":"success"' in ok_text
|
|
155
155
|
# auth
|
|
156
156
|
assert '"auth"' in ok_text
|
|
@@ -31,7 +31,7 @@ def test_tools_listed():
|
|
|
31
31
|
|
|
32
32
|
# order is not a contract; membership is
|
|
33
33
|
assert sorted(_run(run())) == sorted(
|
|
34
|
-
["search", "fetch", "search_fetch", "list_profiles", "login"]
|
|
34
|
+
["search", "fetch", "search_fetch", "list_profiles", "login", "map"]
|
|
35
35
|
)
|
|
36
36
|
|
|
37
37
|
|
|
@@ -53,7 +53,7 @@ def test_invalid_strategy_returns_error_not_crash():
|
|
|
53
53
|
return res
|
|
54
54
|
|
|
55
55
|
res = _run(run())
|
|
56
|
-
assert not res.
|
|
56
|
+
assert not res.isError
|
|
57
57
|
assert "error" in res.content[0].text
|
|
58
58
|
|
|
59
59
|
|
|
@@ -72,7 +72,7 @@ def test_firecrawl_without_key_returns_error_not_crash():
|
|
|
72
72
|
return res
|
|
73
73
|
|
|
74
74
|
res = _run(run())
|
|
75
|
-
assert not res.
|
|
75
|
+
assert not res.isError
|
|
76
76
|
assert "error" in res.content[0].text
|
|
77
77
|
|
|
78
78
|
|
|
@@ -95,7 +95,7 @@ def test_server_stays_alive_after_bad_calls():
|
|
|
95
95
|
return res
|
|
96
96
|
|
|
97
97
|
res = _run(run())
|
|
98
|
-
assert not res.
|
|
98
|
+
assert not res.isError
|
|
99
99
|
assert "success" in res.content[0].text
|
|
100
100
|
|
|
101
101
|
|
|
@@ -118,7 +118,7 @@ def test_fetch_invalid_profile_returns_error_not_crash():
|
|
|
118
118
|
return res, ok
|
|
119
119
|
|
|
120
120
|
res, ok = _run(run())
|
|
121
|
-
assert not res.
|
|
121
|
+
assert not res.isError
|
|
122
122
|
assert "invalid profile name" in res.content[0].text
|
|
123
123
|
assert "success" in ok.content[0].text # server alive after the bad call
|
|
124
124
|
|
|
@@ -139,7 +139,7 @@ def test_fetch_nonexistent_profile_returns_error():
|
|
|
139
139
|
return res
|
|
140
140
|
|
|
141
141
|
res = _run(run())
|
|
142
|
-
assert not res.
|
|
142
|
+
assert not res.isError
|
|
143
143
|
assert "profile 'ghost' not found" in res.content[0].text
|
|
144
144
|
|
|
145
145
|
|
|
@@ -161,9 +161,9 @@ def test_search_fetch_invalid_profile_returns_error_not_crash():
|
|
|
161
161
|
return res, ok
|
|
162
162
|
|
|
163
163
|
res, ok = _run(run())
|
|
164
|
-
assert not res.
|
|
164
|
+
assert not res.isError
|
|
165
165
|
assert "invalid profile name" in res.content[0].text
|
|
166
|
-
assert not ok.
|
|
166
|
+
assert not ok.isError # server alive after the bad call
|
|
167
167
|
|
|
168
168
|
|
|
169
169
|
def test_list_profiles_tool_metadata_only(tmp_path):
|
|
@@ -189,7 +189,7 @@ def test_list_profiles_tool_metadata_only(tmp_path):
|
|
|
189
189
|
return res
|
|
190
190
|
|
|
191
191
|
res = _run(run())
|
|
192
|
-
assert not res.
|
|
192
|
+
assert not res.isError
|
|
193
193
|
text = res.content[0].text
|
|
194
194
|
assert "sion" in text
|
|
195
195
|
assert "SUPERSECRET" not in text # cookie values never exposed
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import socket
|
|
2
|
+
from unittest.mock import patch
|
|
3
|
+
|
|
4
|
+
from webget.ssrf import _hostname_private, _resolve_hostname_ips
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def test_resolve_hostname_ips_primary_success():
|
|
8
|
+
ips = _resolve_hostname_ips("localhost")
|
|
9
|
+
assert any(ip.startswith("127.") or ip == "::1" for ip in ips)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def test_resolve_hostname_ips_fallback_on_primary_failure():
|
|
13
|
+
orig_getaddrinfo = socket.getaddrinfo
|
|
14
|
+
|
|
15
|
+
def mock_getaddrinfo(host, port, *args, **kwargs):
|
|
16
|
+
if host == "flaky.example":
|
|
17
|
+
raise socket.gaierror(socket.EAI_NONAME, "Name or service not known")
|
|
18
|
+
return orig_getaddrinfo(host, port, *args, **kwargs)
|
|
19
|
+
|
|
20
|
+
# Secondary resolver gives fallback IP
|
|
21
|
+
with (
|
|
22
|
+
patch("socket.getaddrinfo", side_effect=mock_getaddrinfo),
|
|
23
|
+
patch("webget.ssrf._doh_resolve", return_value=["93.184.216.34"]),
|
|
24
|
+
):
|
|
25
|
+
ips = _resolve_hostname_ips("flaky.example")
|
|
26
|
+
assert "93.184.216.34" in ips
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_hostname_private_uses_fallback_and_detects_private():
|
|
30
|
+
with (
|
|
31
|
+
patch("socket.getaddrinfo", side_effect=socket.gaierror(socket.EAI_NONAME, "Fail")),
|
|
32
|
+
patch("webget.ssrf._doh_resolve", return_value=["192.168.1.1"]),
|
|
33
|
+
):
|
|
34
|
+
assert _hostname_private("router.local") is True
|
|
@@ -22,12 +22,21 @@ def auth_state(md="", html="", status=None, profile=None):
|
|
|
22
22
|
|
|
23
23
|
class TestParseOpts:
|
|
24
24
|
def test_positional(self):
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
25
|
+
(
|
|
26
|
+
remaining,
|
|
27
|
+
*_,
|
|
28
|
+
limit,
|
|
29
|
+
strategy,
|
|
30
|
+
profile,
|
|
31
|
+
no_cache,
|
|
32
|
+
headless,
|
|
33
|
+
concurrency,
|
|
34
|
+
retry_transient,
|
|
35
|
+
) = opts("u", "https://x.com")
|
|
28
36
|
assert remaining == ["u", "https://x.com"]
|
|
29
37
|
assert limit is None and strategy == "auto" and profile is None
|
|
30
38
|
assert no_cache is False and headless is False and concurrency is None
|
|
39
|
+
assert retry_transient is False
|
|
31
40
|
|
|
32
41
|
def test_cookies_short_and_long(self, tmp_path):
|
|
33
42
|
ck = tmp_path / "ck.txt"
|
|
@@ -45,27 +54,31 @@ class TestParseOpts:
|
|
|
45
54
|
assert mc1 == 500 and mc2 == 500
|
|
46
55
|
|
|
47
56
|
def test_limit(self):
|
|
48
|
-
*_, limit, _, _, _, _, _ = opts("s", "q", "--limit", "7")
|
|
57
|
+
*_, limit, _, _, _, _, _, _ = opts("s", "q", "--limit", "7")
|
|
49
58
|
assert limit == 7
|
|
50
59
|
|
|
51
60
|
def test_profile_and_no_cache(self):
|
|
52
|
-
*_, profile, no_cache, _, _ = opts("u", "https://x.com", "--profile", "campus")
|
|
61
|
+
*_, profile, no_cache, _, _, _ = opts("u", "https://x.com", "--profile", "campus")
|
|
53
62
|
assert profile == "campus" and no_cache is False
|
|
54
|
-
*_, profile, no_cache, _, _ = opts("u", "https://x.com", "--no-cache")
|
|
63
|
+
*_, profile, no_cache, _, _, _ = opts("u", "https://x.com", "--no-cache")
|
|
55
64
|
assert profile is None and no_cache is True
|
|
56
65
|
|
|
57
66
|
def test_strategy(self):
|
|
58
|
-
*_, strategy, _, _, _, _ = opts("u", "https://x.com", "--strategy", "crawl4ai")
|
|
67
|
+
*_, strategy, _, _, _, _, _ = opts("u", "https://x.com", "--strategy", "crawl4ai")
|
|
59
68
|
assert strategy == "crawl4ai"
|
|
60
69
|
|
|
61
70
|
def test_concurrency(self):
|
|
62
|
-
*_, concurrency = opts("u", "https://x.com", "--concurrency", "5")
|
|
71
|
+
*_, concurrency, _ = opts("u", "https://x.com", "--concurrency", "5")
|
|
63
72
|
assert concurrency == 5
|
|
64
73
|
|
|
65
74
|
def test_headless(self):
|
|
66
|
-
*_, headless, _ = opts("login", "https://x.com", "--headless")
|
|
75
|
+
*_, headless, _, _ = opts("login", "https://x.com", "--headless")
|
|
67
76
|
assert headless is True
|
|
68
77
|
|
|
78
|
+
def test_retry_flag(self):
|
|
79
|
+
*_, retry_transient = opts("u", "https://x.com", "--retry")
|
|
80
|
+
assert retry_transient is True
|
|
81
|
+
|
|
69
82
|
def test_unknown_flag_passthrough(self):
|
|
70
83
|
remaining, *_ = opts("u", "https://x.com", "--weird")
|
|
71
84
|
assert "--weird" in remaining
|
|
@@ -533,7 +546,9 @@ class TestStrategyMemory:
|
|
|
533
546
|
webget._learn_strategy("old.com", "crawl4ai")
|
|
534
547
|
# Age the entry beyond the TTL by shifting the clock forward.
|
|
535
548
|
real_time = webget.time.time
|
|
536
|
-
monkeypatch.setattr(
|
|
549
|
+
monkeypatch.setattr(
|
|
550
|
+
webget.time, "time", lambda: real_time() + webget._STRATEGY_MEMORY_TTL + 1
|
|
551
|
+
)
|
|
537
552
|
assert webget._load_strategy_memory() == {}
|
|
538
553
|
|
|
539
554
|
def test_fresh_entry_survives(self, isolated_env):
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""webget - local search + scrape, zero API keys, unlimited usage.
|
|
2
|
+
|
|
3
|
+
A package refactor of the original single-file webget_cli.py. Public API
|
|
4
|
+
is re-exported here so callers can do `from webget import fetch_http,
|
|
5
|
+
scrape_many, ...` instead of importing internal modules.
|
|
6
|
+
|
|
7
|
+
Sub-modules:
|
|
8
|
+
- cache - disk cache + cookie/header parsing
|
|
9
|
+
- ssrf - private-IP guard for HTTP + browser paths
|
|
10
|
+
- profile - persistent browser-profile (auth session) management
|
|
11
|
+
- http - fast-path HTTP fetch + markdown extraction
|
|
12
|
+
- firecrawl - optional cloud escape-hatch fetch
|
|
13
|
+
- search - DuckDuckGo text search + atomic JSON helpers
|
|
14
|
+
- ladder - HTTP -> Crawl4AI -> Firecrawl orchestration
|
|
15
|
+
- cli - argparse entry point
|
|
16
|
+
|
|
17
|
+
The legacy single-file import `import webget_cli as webget` keeps
|
|
18
|
+
working via the compatibility shim at ./webget_cli.py.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from .cache import (
|
|
24
|
+
_CACHE_SWEEP_TTL,
|
|
25
|
+
CACHE_DIR,
|
|
26
|
+
_cache_path,
|
|
27
|
+
cache_get,
|
|
28
|
+
cache_put,
|
|
29
|
+
parse_cookie_file,
|
|
30
|
+
parse_headers,
|
|
31
|
+
)
|
|
32
|
+
from .cli import main, parse_opts
|
|
33
|
+
from .discovery import discover_urls
|
|
34
|
+
from .firecrawl import fetch_firecrawl, firecrawl_key
|
|
35
|
+
from .http import MAX_RESPONSE_BYTES, ResponseTooLarge, _extract_markdown, fetch_http
|
|
36
|
+
from .ladder import (
|
|
37
|
+
_DEFAULT_CONCURRENCY,
|
|
38
|
+
_STRATEGY_MEMORY_TTL,
|
|
39
|
+
_auth_message,
|
|
40
|
+
_crawl4ai_once,
|
|
41
|
+
_ladder,
|
|
42
|
+
_learn_strategy,
|
|
43
|
+
_load_strategy_memory,
|
|
44
|
+
_normalize_hit,
|
|
45
|
+
_reorder_steps_by_domain,
|
|
46
|
+
_save_strategy_memory,
|
|
47
|
+
_strategy_memory_path,
|
|
48
|
+
_terminal_state,
|
|
49
|
+
scrape_many,
|
|
50
|
+
)
|
|
51
|
+
from .profile import (
|
|
52
|
+
_PROFILE_NAME_RE,
|
|
53
|
+
PROFILE_DIR,
|
|
54
|
+
_auth_state,
|
|
55
|
+
_cookie_belongs_to,
|
|
56
|
+
_domain_match,
|
|
57
|
+
_effective_cookies,
|
|
58
|
+
_fmt_age,
|
|
59
|
+
_login_flow,
|
|
60
|
+
_logout_domain_regex,
|
|
61
|
+
_logout_flow,
|
|
62
|
+
_profile_meta,
|
|
63
|
+
_profile_root,
|
|
64
|
+
_prune_storage_cookies,
|
|
65
|
+
_valid_site_url,
|
|
66
|
+
_wait_for_session_cookies,
|
|
67
|
+
_warn,
|
|
68
|
+
list_profiles,
|
|
69
|
+
load_profile_cookies,
|
|
70
|
+
profile_dir,
|
|
71
|
+
profile_exists,
|
|
72
|
+
profile_state_path,
|
|
73
|
+
)
|
|
74
|
+
from .search import _read_json, _write_json, search
|
|
75
|
+
from .ssrf import (
|
|
76
|
+
SSRFError,
|
|
77
|
+
_guard_browser_routes,
|
|
78
|
+
_hostname_private,
|
|
79
|
+
_ip_is_private,
|
|
80
|
+
_is_private_target,
|
|
81
|
+
_private_ip_for,
|
|
82
|
+
_request_body_bytes,
|
|
83
|
+
_resolve_hostname_ips,
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
# Sorted to satisfy ruff RUF022; module grouping lives in the imports above.
|
|
87
|
+
__all__ = [
|
|
88
|
+
"CACHE_DIR",
|
|
89
|
+
"MAX_RESPONSE_BYTES",
|
|
90
|
+
"PROFILE_DIR",
|
|
91
|
+
"_CACHE_SWEEP_TTL",
|
|
92
|
+
"_DEFAULT_CONCURRENCY",
|
|
93
|
+
"_PROFILE_NAME_RE",
|
|
94
|
+
"_STRATEGY_MEMORY_TTL",
|
|
95
|
+
"ResponseTooLarge",
|
|
96
|
+
"SSRFError",
|
|
97
|
+
"_auth_message",
|
|
98
|
+
"_auth_state",
|
|
99
|
+
"_cache_path",
|
|
100
|
+
"_cookie_belongs_to",
|
|
101
|
+
"_crawl4ai_once",
|
|
102
|
+
"_domain_match",
|
|
103
|
+
"_effective_cookies",
|
|
104
|
+
"_extract_markdown",
|
|
105
|
+
"_fmt_age",
|
|
106
|
+
"_guard_browser_routes",
|
|
107
|
+
"_hostname_private",
|
|
108
|
+
"_ip_is_private",
|
|
109
|
+
"_is_private_target",
|
|
110
|
+
"_ladder",
|
|
111
|
+
"_learn_strategy",
|
|
112
|
+
"_load_strategy_memory",
|
|
113
|
+
"_login_flow",
|
|
114
|
+
"_logout_domain_regex",
|
|
115
|
+
"_logout_flow",
|
|
116
|
+
"_normalize_hit",
|
|
117
|
+
"_private_ip_for",
|
|
118
|
+
"_profile_meta",
|
|
119
|
+
"_profile_root",
|
|
120
|
+
"_prune_storage_cookies",
|
|
121
|
+
"_read_json",
|
|
122
|
+
"_reorder_steps_by_domain",
|
|
123
|
+
"_request_body_bytes",
|
|
124
|
+
"_resolve_hostname_ips",
|
|
125
|
+
"_save_strategy_memory",
|
|
126
|
+
"_strategy_memory_path",
|
|
127
|
+
"_terminal_state",
|
|
128
|
+
"_valid_site_url",
|
|
129
|
+
"_wait_for_session_cookies",
|
|
130
|
+
"_warn",
|
|
131
|
+
"_write_json",
|
|
132
|
+
"cache_get",
|
|
133
|
+
"cache_put",
|
|
134
|
+
"discover_urls",
|
|
135
|
+
"fetch_firecrawl",
|
|
136
|
+
"fetch_http",
|
|
137
|
+
"firecrawl_key",
|
|
138
|
+
"list_profiles",
|
|
139
|
+
"load_profile_cookies",
|
|
140
|
+
"main",
|
|
141
|
+
"parse_cookie_file",
|
|
142
|
+
"parse_headers",
|
|
143
|
+
"parse_opts",
|
|
144
|
+
"profile_dir",
|
|
145
|
+
"profile_exists",
|
|
146
|
+
"profile_state_path",
|
|
147
|
+
"scrape_many",
|
|
148
|
+
"search",
|
|
149
|
+
]
|