webget-cli 0.9.0__tar.gz → 0.11.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. {webget_cli-0.9.0 → webget_cli-0.11.0}/PKG-INFO +1 -1
  2. {webget_cli-0.9.0 → webget_cli-0.11.0}/pyproject.toml +3 -2
  3. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_mcp.py +5 -5
  4. webget_cli-0.11.0/tests/test_discovery_map.py +29 -0
  5. webget_cli-0.11.0/tests/test_ladder_retry.py +41 -0
  6. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_login_flow.py +18 -7
  7. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_mcp_leak_review.py +2 -2
  8. webget_cli-0.11.0/tests/test_mcp_map.py +18 -0
  9. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_mcp_profile.py +3 -3
  10. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_mcp_server.py +9 -9
  11. webget_cli-0.11.0/tests/test_ssrf_dual_dns.py +34 -0
  12. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_webget.py +25 -10
  13. webget_cli-0.11.0/webget/__init__.py +149 -0
  14. webget_cli-0.11.0/webget/cache.py +188 -0
  15. webget_cli-0.11.0/webget/cli.py +407 -0
  16. webget_cli-0.11.0/webget/discovery.py +94 -0
  17. webget_cli-0.11.0/webget/firecrawl.py +61 -0
  18. webget_cli-0.11.0/webget/http.py +174 -0
  19. webget_cli-0.11.0/webget/ladder.py +613 -0
  20. webget_cli-0.11.0/webget/profile.py +410 -0
  21. webget_cli-0.11.0/webget/search.py +38 -0
  22. webget_cli-0.11.0/webget/ssrf.py +280 -0
  23. {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/PKG-INFO +1 -1
  24. {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/SOURCES.txt +14 -0
  25. {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/entry_points.txt +1 -1
  26. {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/top_level.txt +1 -0
  27. webget_cli-0.11.0/webget_cli.py +58 -0
  28. {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_mcp.py +22 -0
  29. webget_cli-0.9.0/webget_cli.py +0 -1762
  30. {webget_cli-0.9.0 → webget_cli-0.11.0}/LICENSE +0 -0
  31. {webget_cli-0.9.0 → webget_cli-0.11.0}/README.md +0 -0
  32. {webget_cli-0.9.0 → webget_cli-0.11.0}/setup.cfg +0 -0
  33. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_auth.py +0 -0
  34. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_cache.py +0 -0
  35. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_concurrency.py +0 -0
  36. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_http.py +0 -0
  37. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_adversarial_ssrf.py +0 -0
  38. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_auth_review.py +0 -0
  39. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_browser_ssrf.py +0 -0
  40. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_cache_review.py +0 -0
  41. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_concurrency_review.py +0 -0
  42. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_extraction_markdown.py +0 -0
  43. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_firecrawl_policy.py +0 -0
  44. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_integration_ladder.py +0 -0
  45. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_mcp_smoke.py +0 -0
  46. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_security_review.py +0 -0
  47. {webget_cli-0.9.0 → webget_cli-0.11.0}/tests/test_size_review.py +0 -0
  48. {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/dependency_links.txt +0 -0
  49. {webget_cli-0.9.0 → webget_cli-0.11.0}/webget_cli.egg-info/requires.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: webget-cli
3
- Version: 0.9.0
3
+ Version: 0.11.0
4
4
  Summary: Local search + scrape CLI with an HTTP fast path, optional browser fallback, and authenticated session profiles. Zero API keys.
5
5
  Author-email: David Tarigan <tarigansdavid@gmail.com>
6
6
  License: Apache-2.0
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "webget-cli"
7
- version = "0.9.0"
7
+ version = "0.11.0"
8
8
  description = "Local search + scrape CLI with an HTTP fast path, optional browser fallback, and authenticated session profiles. Zero API keys."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -39,11 +39,12 @@ mcp = ["fastmcp>=2"]
39
39
  dev = ["pytest>=8", "ruff>=0.6"]
40
40
 
41
41
  [project.scripts]
42
- webget = "webget_cli:main"
42
+ webget = "webget.cli:main"
43
43
  webget-mcp = "webget_mcp:main"
44
44
 
45
45
  [tool.setuptools]
46
46
  py-modules = ["webget_cli", "webget_mcp"]
47
+ packages = ["webget"]
47
48
 
48
49
  [tool.pytest.ini_options]
49
50
  testpaths = ["tests"]
@@ -58,7 +58,7 @@ class TestMalformedArguments:
58
58
  return res
59
59
 
60
60
  res = _run(run())
61
- assert res.is_error
61
+ assert res.isError
62
62
 
63
63
 
64
64
  class TestInvalidURLs:
@@ -72,7 +72,7 @@ class TestInvalidURLs:
72
72
  return res
73
73
 
74
74
  res = _run(run())
75
- assert "error" in res.content[0].text or res.is_error
75
+ assert "error" in res.content[0].text or res.isError
76
76
 
77
77
 
78
78
  class TestRepeatedCalls:
@@ -86,7 +86,7 @@ class TestRepeatedCalls:
86
86
  "fetch",
87
87
  {"url": "https://example.com", "strategy": "http", "no_cache": True},
88
88
  )
89
- if res.is_error:
89
+ if res.isError:
90
90
  return "ERROR"
91
91
  return "OK"
92
92
 
@@ -106,7 +106,7 @@ class TestRepeatedCalls:
106
106
  for _ in range(5)
107
107
  ]
108
108
  )
109
- return [r.is_error for r in results]
109
+ return [r.isError for r in results]
110
110
 
111
111
  assert _run(run()) == [False] * 5
112
112
 
@@ -133,7 +133,7 @@ class TestInputCaps:
133
133
  return res
134
134
 
135
135
  res = _run(run())
136
- assert "must be between" in res.content[0].text or res.is_error
136
+ assert "must be between" in res.content[0].text or res.isError
137
137
 
138
138
 
139
139
  class TestToolFailureIsolation:
@@ -0,0 +1,29 @@
1
+ import asyncio
2
+
3
+ from webget.discovery import _extract_sitemap_urls, discover_urls
4
+
5
+
6
+ def test_extract_sitemap_urls_standard_xml():
7
+ xml = """<?xml version="1.0" encoding="UTF-8"?>
8
+ <urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
9
+ <url>
10
+ <loc>https://example.com/page1</loc>
11
+ </url>
12
+ <url>
13
+ <loc>https://example.com/page2</loc>
14
+ </url>
15
+ </urlset>
16
+ """
17
+ urls = _extract_sitemap_urls(xml)
18
+ assert urls == ["https://example.com/page1", "https://example.com/page2"]
19
+
20
+
21
+ def test_extract_sitemap_urls_malformed_regex_fallback():
22
+ xml = """<urlset><url><loc>https://example.com/broken1</loc></unclosed>"""
23
+ urls = _extract_sitemap_urls(xml)
24
+ assert "https://example.com/broken1" in urls
25
+
26
+
27
+ def test_discover_urls_private_target_blocked():
28
+ urls = asyncio.run(discover_urls("http://127.0.0.1/sitemap.xml", allow_private=False))
29
+ assert urls == []
@@ -0,0 +1,41 @@
1
+ import asyncio
2
+ from unittest.mock import patch
3
+
4
+ from webget.ladder import scrape_many
5
+
6
+
7
+ def test_scrape_many_retry_transient_timeout(fresh_cache, server):
8
+ calls = []
9
+
10
+ async def mock_fetch(url, *args, **kwargs):
11
+ calls.append(url)
12
+ if len(calls) == 1:
13
+ raise TimeoutError("timeout")
14
+ return {
15
+ "title": "Success After Retry",
16
+ "markdown": "Valid content length " * 10,
17
+ "status": "success",
18
+ }
19
+
20
+ async def run_without_retry():
21
+ return await scrape_many(
22
+ [server.url("/test")], strategy="http", per_url_timeout=1, retry_transient=False
23
+ )
24
+
25
+ async def run_with_retry():
26
+ return await scrape_many(
27
+ [server.url("/test")], strategy="http", per_url_timeout=1, retry_transient=True
28
+ )
29
+
30
+ with patch("webget.ladder._resolve_fetch_http", return_value=mock_fetch):
31
+ # Without retry flag -> error on first timeout
32
+ res = asyncio.run(run_without_retry())
33
+ target = server.url("/test")
34
+ assert res[target]["status"] == "error"
35
+ assert res[target]["attempts"] == 1
36
+
37
+ # With retry_transient=True -> retries and succeeds on 2nd attempt
38
+ calls.clear()
39
+ res = asyncio.run(run_with_retry())
40
+ assert res[target]["status"] == "success"
41
+ assert res[target]["attempts"] == 2
@@ -102,7 +102,9 @@ def _spawn_mcp(profile_root):
102
102
  "WEBGET_PROFILE_DIR": str(profile_root),
103
103
  "WEBGET_ALLOW_PRIVATE": "1",
104
104
  }
105
- return StdioServerParameters(command=sys.executable, args=[str(ROOT / "webget_mcp.py")], env=env)
105
+ return StdioServerParameters(
106
+ command=sys.executable, args=[str(ROOT / "webget_mcp.py")], env=env
107
+ )
106
108
 
107
109
 
108
110
  def test_mcp_login_tool_persists_and_fetch_uses_it(server, tmp_path):
@@ -117,14 +119,18 @@ def test_mcp_login_tool_persists_and_fetch_uses_it(server, tmp_path):
117
119
  gated_url = server.url("/cookie-gated")
118
120
 
119
121
  async def run():
120
- async with stdio_client(_spawn_mcp(root)) as (read, write), ClientSession(read, write) as session:
122
+ async with (
123
+ stdio_client(_spawn_mcp(root)) as (read, write),
124
+ ClientSession(read, write) as session,
125
+ ):
121
126
  await session.initialize()
122
127
  res = await session.call_tool(
123
128
  "login",
124
129
  {"url": login_url, "profile": "mcplogin", "headless": True, "wait_seconds": 30},
125
130
  )
126
131
  authed = await session.call_tool(
127
- "fetch", {"url": gated_url, "strategy": "http", "no_cache": True, "profile": "mcplogin"}
132
+ "fetch",
133
+ {"url": gated_url, "strategy": "http", "no_cache": True, "profile": "mcplogin"},
128
134
  )
129
135
  anon = await session.call_tool(
130
136
  "fetch", {"url": gated_url, "strategy": "http", "no_cache": True}
@@ -133,7 +139,7 @@ def test_mcp_login_tool_persists_and_fetch_uses_it(server, tmp_path):
133
139
 
134
140
  res, authed, anon = asyncio.run(asyncio.wait_for(run(), timeout=90))
135
141
  res_text = res.content[0].text if res.content else ""
136
- assert not res.is_error, res_text
142
+ assert not res.isError, res_text
137
143
  assert '"status":"success"' in res_text, res_text
138
144
  assert '"profile":"mcplogin"' in res_text
139
145
 
@@ -155,16 +161,21 @@ def test_mcp_login_rejects_bad_url_and_profile(server, tmp_path):
155
161
  root = _profile_root(tmp_path)
156
162
 
157
163
  async def run():
158
- async with stdio_client(_spawn_mcp(root)) as (read, write), ClientSession(read, write) as session:
164
+ async with (
165
+ stdio_client(_spawn_mcp(root)) as (read, write),
166
+ ClientSession(read, write) as session,
167
+ ):
159
168
  await session.initialize()
160
169
  bad_url = await session.call_tool(
161
170
  "login", {"url": "not a url", "profile": "x", "headless": True, "wait_seconds": 5}
162
171
  )
163
172
  bad_name = await session.call_tool(
164
- "login", {"url": server.url("/"), "profile": "../evil", "headless": True, "wait_seconds": 5}
173
+ "login",
174
+ {"url": server.url("/"), "profile": "../evil", "headless": True, "wait_seconds": 5},
165
175
  )
166
176
  bad_secs = await session.call_tool(
167
- "login", {"url": server.url("/"), "profile": "x", "headless": True, "wait_seconds": 1}
177
+ "login",
178
+ {"url": server.url("/"), "profile": "x", "headless": True, "wait_seconds": 1},
168
179
  )
169
180
  return bad_url, bad_name, bad_secs
170
181
 
@@ -64,7 +64,7 @@ class TestNoSecretLeakage:
64
64
  return res
65
65
 
66
66
  res = _run(run())
67
- if res.is_error:
67
+ if res.isError:
68
68
  return # error path: no payload to leak, still fine
69
69
  text = res.content[0].text
70
70
  hits = _scan(text)
@@ -121,7 +121,7 @@ class TestServerRecovery:
121
121
  bad = await session.call_tool(
122
122
  "fetch", {"url": "https://example.com", "strategy": "bogus"}
123
123
  )
124
- assert not bad.is_error or "error" in bad.content[0].text
124
+ assert not bad.isError or "error" in bad.content[0].text
125
125
  good = await session.call_tool(
126
126
  "fetch", {"url": "https://example.com", "strategy": "http", "no_cache": True}
127
127
  )
@@ -0,0 +1,18 @@
1
+ import asyncio
2
+ from unittest.mock import AsyncMock, patch
3
+
4
+ from webget_mcp import map as mcp_map
5
+
6
+
7
+ def test_mcp_map_clamps_limit():
8
+ res = asyncio.run(mcp_map("https://example.com", limit=-5))
9
+ assert len(res) == 1
10
+ assert "error: limit must be between" in res[0]
11
+
12
+
13
+ def test_mcp_map_calls_discover_urls():
14
+ with patch("webget_cli.discover_urls", new_callable=AsyncMock) as mock_disc:
15
+ mock_disc.return_value = ["https://example.com/p1", "https://example.com/p2"]
16
+ res = asyncio.run(mcp_map("https://example.com", limit=50))
17
+ assert res == ["https://example.com/p1", "https://example.com/p2"]
18
+ mock_disc.assert_awaited_once_with("https://example.com", limit=50, timeout=15)
@@ -79,7 +79,7 @@ def test_mcp_fetch_with_profile_uses_session(tmp_path):
79
79
  ok, anon = asyncio.run(asyncio.wait_for(run(), timeout=60))
80
80
  ok_text = ok.content[0].text if ok.content else ""
81
81
  anon_text = anon.content[0].text if anon.content else ""
82
- assert not ok.is_error, ok_text
82
+ assert not ok.isError, ok_text
83
83
  assert '"status":"success"' in ok_text, ok_text
84
84
  assert "you are authenticated" in ok_text, ok_text
85
85
  assert '"authenticated":true' in ok_text, ok_text # session was used
@@ -105,7 +105,7 @@ def test_mcp_fetch_unknown_profile_is_hard_error(tmp_path):
105
105
  return res
106
106
 
107
107
  res = asyncio.run(asyncio.wait_for(run(), timeout=60))
108
- assert not res.is_error
108
+ assert not res.isError
109
109
  assert "profile 'ghost' not found" in res.content[0].text
110
110
 
111
111
 
@@ -150,7 +150,7 @@ def test_mcp_fetch_enriched_output_present(tmp_path):
150
150
  anon_text = anon.content[0].text if anon.content else ""
151
151
 
152
152
  # Authenticated: standard assertions
153
- assert not ok.is_error, ok_text
153
+ assert not ok.isError, ok_text
154
154
  assert '"status":"success"' in ok_text
155
155
  # auth
156
156
  assert '"auth"' in ok_text
@@ -31,7 +31,7 @@ def test_tools_listed():
31
31
 
32
32
  # order is not a contract; membership is
33
33
  assert sorted(_run(run())) == sorted(
34
- ["search", "fetch", "search_fetch", "list_profiles", "login"]
34
+ ["search", "fetch", "search_fetch", "list_profiles", "login", "map"]
35
35
  )
36
36
 
37
37
 
@@ -53,7 +53,7 @@ def test_invalid_strategy_returns_error_not_crash():
53
53
  return res
54
54
 
55
55
  res = _run(run())
56
- assert not res.is_error
56
+ assert not res.isError
57
57
  assert "error" in res.content[0].text
58
58
 
59
59
 
@@ -72,7 +72,7 @@ def test_firecrawl_without_key_returns_error_not_crash():
72
72
  return res
73
73
 
74
74
  res = _run(run())
75
- assert not res.is_error
75
+ assert not res.isError
76
76
  assert "error" in res.content[0].text
77
77
 
78
78
 
@@ -95,7 +95,7 @@ def test_server_stays_alive_after_bad_calls():
95
95
  return res
96
96
 
97
97
  res = _run(run())
98
- assert not res.is_error
98
+ assert not res.isError
99
99
  assert "success" in res.content[0].text
100
100
 
101
101
 
@@ -118,7 +118,7 @@ def test_fetch_invalid_profile_returns_error_not_crash():
118
118
  return res, ok
119
119
 
120
120
  res, ok = _run(run())
121
- assert not res.is_error
121
+ assert not res.isError
122
122
  assert "invalid profile name" in res.content[0].text
123
123
  assert "success" in ok.content[0].text # server alive after the bad call
124
124
 
@@ -139,7 +139,7 @@ def test_fetch_nonexistent_profile_returns_error():
139
139
  return res
140
140
 
141
141
  res = _run(run())
142
- assert not res.is_error
142
+ assert not res.isError
143
143
  assert "profile 'ghost' not found" in res.content[0].text
144
144
 
145
145
 
@@ -161,9 +161,9 @@ def test_search_fetch_invalid_profile_returns_error_not_crash():
161
161
  return res, ok
162
162
 
163
163
  res, ok = _run(run())
164
- assert not res.is_error
164
+ assert not res.isError
165
165
  assert "invalid profile name" in res.content[0].text
166
- assert not ok.is_error # server alive after the bad call
166
+ assert not ok.isError # server alive after the bad call
167
167
 
168
168
 
169
169
  def test_list_profiles_tool_metadata_only(tmp_path):
@@ -189,7 +189,7 @@ def test_list_profiles_tool_metadata_only(tmp_path):
189
189
  return res
190
190
 
191
191
  res = _run(run())
192
- assert not res.is_error
192
+ assert not res.isError
193
193
  text = res.content[0].text
194
194
  assert "sion" in text
195
195
  assert "SUPERSECRET" not in text # cookie values never exposed
@@ -0,0 +1,34 @@
1
+ import socket
2
+ from unittest.mock import patch
3
+
4
+ from webget.ssrf import _hostname_private, _resolve_hostname_ips
5
+
6
+
7
+ def test_resolve_hostname_ips_primary_success():
8
+ ips = _resolve_hostname_ips("localhost")
9
+ assert any(ip.startswith("127.") or ip == "::1" for ip in ips)
10
+
11
+
12
+ def test_resolve_hostname_ips_fallback_on_primary_failure():
13
+ orig_getaddrinfo = socket.getaddrinfo
14
+
15
+ def mock_getaddrinfo(host, port, *args, **kwargs):
16
+ if host == "flaky.example":
17
+ raise socket.gaierror(socket.EAI_NONAME, "Name or service not known")
18
+ return orig_getaddrinfo(host, port, *args, **kwargs)
19
+
20
+ # Secondary resolver gives fallback IP
21
+ with (
22
+ patch("socket.getaddrinfo", side_effect=mock_getaddrinfo),
23
+ patch("webget.ssrf._doh_resolve", return_value=["93.184.216.34"]),
24
+ ):
25
+ ips = _resolve_hostname_ips("flaky.example")
26
+ assert "93.184.216.34" in ips
27
+
28
+
29
+ def test_hostname_private_uses_fallback_and_detects_private():
30
+ with (
31
+ patch("socket.getaddrinfo", side_effect=socket.gaierror(socket.EAI_NONAME, "Fail")),
32
+ patch("webget.ssrf._doh_resolve", return_value=["192.168.1.1"]),
33
+ ):
34
+ assert _hostname_private("router.local") is True
@@ -22,12 +22,21 @@ def auth_state(md="", html="", status=None, profile=None):
22
22
 
23
23
  class TestParseOpts:
24
24
  def test_positional(self):
25
- remaining, *_, limit, strategy, profile, no_cache, headless, concurrency = opts(
26
- "u", "https://x.com"
27
- )
25
+ (
26
+ remaining,
27
+ *_,
28
+ limit,
29
+ strategy,
30
+ profile,
31
+ no_cache,
32
+ headless,
33
+ concurrency,
34
+ retry_transient,
35
+ ) = opts("u", "https://x.com")
28
36
  assert remaining == ["u", "https://x.com"]
29
37
  assert limit is None and strategy == "auto" and profile is None
30
38
  assert no_cache is False and headless is False and concurrency is None
39
+ assert retry_transient is False
31
40
 
32
41
  def test_cookies_short_and_long(self, tmp_path):
33
42
  ck = tmp_path / "ck.txt"
@@ -45,27 +54,31 @@ class TestParseOpts:
45
54
  assert mc1 == 500 and mc2 == 500
46
55
 
47
56
  def test_limit(self):
48
- *_, limit, _, _, _, _, _ = opts("s", "q", "--limit", "7")
57
+ *_, limit, _, _, _, _, _, _ = opts("s", "q", "--limit", "7")
49
58
  assert limit == 7
50
59
 
51
60
  def test_profile_and_no_cache(self):
52
- *_, profile, no_cache, _, _ = opts("u", "https://x.com", "--profile", "campus")
61
+ *_, profile, no_cache, _, _, _ = opts("u", "https://x.com", "--profile", "campus")
53
62
  assert profile == "campus" and no_cache is False
54
- *_, profile, no_cache, _, _ = opts("u", "https://x.com", "--no-cache")
63
+ *_, profile, no_cache, _, _, _ = opts("u", "https://x.com", "--no-cache")
55
64
  assert profile is None and no_cache is True
56
65
 
57
66
  def test_strategy(self):
58
- *_, strategy, _, _, _, _ = opts("u", "https://x.com", "--strategy", "crawl4ai")
67
+ *_, strategy, _, _, _, _, _ = opts("u", "https://x.com", "--strategy", "crawl4ai")
59
68
  assert strategy == "crawl4ai"
60
69
 
61
70
  def test_concurrency(self):
62
- *_, concurrency = opts("u", "https://x.com", "--concurrency", "5")
71
+ *_, concurrency, _ = opts("u", "https://x.com", "--concurrency", "5")
63
72
  assert concurrency == 5
64
73
 
65
74
  def test_headless(self):
66
- *_, headless, _ = opts("login", "https://x.com", "--headless")
75
+ *_, headless, _, _ = opts("login", "https://x.com", "--headless")
67
76
  assert headless is True
68
77
 
78
+ def test_retry_flag(self):
79
+ *_, retry_transient = opts("u", "https://x.com", "--retry")
80
+ assert retry_transient is True
81
+
69
82
  def test_unknown_flag_passthrough(self):
70
83
  remaining, *_ = opts("u", "https://x.com", "--weird")
71
84
  assert "--weird" in remaining
@@ -533,7 +546,9 @@ class TestStrategyMemory:
533
546
  webget._learn_strategy("old.com", "crawl4ai")
534
547
  # Age the entry beyond the TTL by shifting the clock forward.
535
548
  real_time = webget.time.time
536
- monkeypatch.setattr(webget.time, "time", lambda: real_time() + webget._STRATEGY_MEMORY_TTL + 1)
549
+ monkeypatch.setattr(
550
+ webget.time, "time", lambda: real_time() + webget._STRATEGY_MEMORY_TTL + 1
551
+ )
537
552
  assert webget._load_strategy_memory() == {}
538
553
 
539
554
  def test_fresh_entry_survives(self, isolated_env):
@@ -0,0 +1,149 @@
1
+ """webget - local search + scrape, zero API keys, unlimited usage.
2
+
3
+ A package refactor of the original single-file webget_cli.py. Public API
4
+ is re-exported here so callers can do `from webget import fetch_http,
5
+ scrape_many, ...` instead of importing internal modules.
6
+
7
+ Sub-modules:
8
+ - cache - disk cache + cookie/header parsing
9
+ - ssrf - private-IP guard for HTTP + browser paths
10
+ - profile - persistent browser-profile (auth session) management
11
+ - http - fast-path HTTP fetch + markdown extraction
12
+ - firecrawl - optional cloud escape-hatch fetch
13
+ - search - DuckDuckGo text search + atomic JSON helpers
14
+ - ladder - HTTP -> Crawl4AI -> Firecrawl orchestration
15
+ - cli - argparse entry point
16
+
17
+ The legacy single-file import `import webget_cli as webget` keeps
18
+ working via the compatibility shim at ./webget_cli.py.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from .cache import (
24
+ _CACHE_SWEEP_TTL,
25
+ CACHE_DIR,
26
+ _cache_path,
27
+ cache_get,
28
+ cache_put,
29
+ parse_cookie_file,
30
+ parse_headers,
31
+ )
32
+ from .cli import main, parse_opts
33
+ from .discovery import discover_urls
34
+ from .firecrawl import fetch_firecrawl, firecrawl_key
35
+ from .http import MAX_RESPONSE_BYTES, ResponseTooLarge, _extract_markdown, fetch_http
36
+ from .ladder import (
37
+ _DEFAULT_CONCURRENCY,
38
+ _STRATEGY_MEMORY_TTL,
39
+ _auth_message,
40
+ _crawl4ai_once,
41
+ _ladder,
42
+ _learn_strategy,
43
+ _load_strategy_memory,
44
+ _normalize_hit,
45
+ _reorder_steps_by_domain,
46
+ _save_strategy_memory,
47
+ _strategy_memory_path,
48
+ _terminal_state,
49
+ scrape_many,
50
+ )
51
+ from .profile import (
52
+ _PROFILE_NAME_RE,
53
+ PROFILE_DIR,
54
+ _auth_state,
55
+ _cookie_belongs_to,
56
+ _domain_match,
57
+ _effective_cookies,
58
+ _fmt_age,
59
+ _login_flow,
60
+ _logout_domain_regex,
61
+ _logout_flow,
62
+ _profile_meta,
63
+ _profile_root,
64
+ _prune_storage_cookies,
65
+ _valid_site_url,
66
+ _wait_for_session_cookies,
67
+ _warn,
68
+ list_profiles,
69
+ load_profile_cookies,
70
+ profile_dir,
71
+ profile_exists,
72
+ profile_state_path,
73
+ )
74
+ from .search import _read_json, _write_json, search
75
+ from .ssrf import (
76
+ SSRFError,
77
+ _guard_browser_routes,
78
+ _hostname_private,
79
+ _ip_is_private,
80
+ _is_private_target,
81
+ _private_ip_for,
82
+ _request_body_bytes,
83
+ _resolve_hostname_ips,
84
+ )
85
+
86
+ # Sorted to satisfy ruff RUF022; module grouping lives in the imports above.
87
+ __all__ = [
88
+ "CACHE_DIR",
89
+ "MAX_RESPONSE_BYTES",
90
+ "PROFILE_DIR",
91
+ "_CACHE_SWEEP_TTL",
92
+ "_DEFAULT_CONCURRENCY",
93
+ "_PROFILE_NAME_RE",
94
+ "_STRATEGY_MEMORY_TTL",
95
+ "ResponseTooLarge",
96
+ "SSRFError",
97
+ "_auth_message",
98
+ "_auth_state",
99
+ "_cache_path",
100
+ "_cookie_belongs_to",
101
+ "_crawl4ai_once",
102
+ "_domain_match",
103
+ "_effective_cookies",
104
+ "_extract_markdown",
105
+ "_fmt_age",
106
+ "_guard_browser_routes",
107
+ "_hostname_private",
108
+ "_ip_is_private",
109
+ "_is_private_target",
110
+ "_ladder",
111
+ "_learn_strategy",
112
+ "_load_strategy_memory",
113
+ "_login_flow",
114
+ "_logout_domain_regex",
115
+ "_logout_flow",
116
+ "_normalize_hit",
117
+ "_private_ip_for",
118
+ "_profile_meta",
119
+ "_profile_root",
120
+ "_prune_storage_cookies",
121
+ "_read_json",
122
+ "_reorder_steps_by_domain",
123
+ "_request_body_bytes",
124
+ "_resolve_hostname_ips",
125
+ "_save_strategy_memory",
126
+ "_strategy_memory_path",
127
+ "_terminal_state",
128
+ "_valid_site_url",
129
+ "_wait_for_session_cookies",
130
+ "_warn",
131
+ "_write_json",
132
+ "cache_get",
133
+ "cache_put",
134
+ "discover_urls",
135
+ "fetch_firecrawl",
136
+ "fetch_http",
137
+ "firecrawl_key",
138
+ "list_profiles",
139
+ "load_profile_cookies",
140
+ "main",
141
+ "parse_cookie_file",
142
+ "parse_headers",
143
+ "parse_opts",
144
+ "profile_dir",
145
+ "profile_exists",
146
+ "profile_state_path",
147
+ "scrape_many",
148
+ "search",
149
+ ]