scrapebadger-cli 0.4.2__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/PKG-INFO +1 -1
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/pyproject.toml +1 -1
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/client.py +13 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/web.py +40 -4
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/.gitignore +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/README.md +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/__init__.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/__init__.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/amazon.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/auth.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/depop.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/ebay.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/google.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/idealista.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/immobiliare.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/leboncoin.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/linkedin.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/loopnet.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/realtor.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/reddit.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/redfin.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/stream.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/tiktok.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/twitter.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/vinted.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/youtube.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/zillow.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/config.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/main.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/output.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/uv.lock +0 -0
|
@@ -43,6 +43,19 @@ def api_post(path: str, json_body: dict[str, Any] | None = None) -> dict | list:
|
|
|
43
43
|
return _handle_response(resp)
|
|
44
44
|
|
|
45
45
|
|
|
46
|
+
def api_post_raw(path: str, json_body: dict[str, Any] | None = None) -> tuple[bytes, str]:
|
|
47
|
+
"""POST and return ``(body_bytes, content_type)`` without JSON-decoding.
|
|
48
|
+
|
|
49
|
+
For `/v1/web/scrape` with `raw_content`, whose response is the scraped body
|
|
50
|
+
itself. Returns bytes, not text: the payload may be an image or a PDF and
|
|
51
|
+
decoding it would corrupt it.
|
|
52
|
+
"""
|
|
53
|
+
resp = _get_client().post(path, json=json_body)
|
|
54
|
+
if resp.status_code >= 400:
|
|
55
|
+
_handle_response(resp)
|
|
56
|
+
return resp.content, resp.headers.get("content-type", "")
|
|
57
|
+
|
|
58
|
+
|
|
46
59
|
def _handle_response(resp: httpx.Response) -> dict | list:
|
|
47
60
|
"""Handle API response, exit on errors."""
|
|
48
61
|
if resp.status_code >= 400:
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
import typer
|
|
4
4
|
|
|
5
|
-
from scrapebadger_cli.client import api_get, api_post
|
|
5
|
+
from scrapebadger_cli.client import api_get, api_post, api_post_raw
|
|
6
6
|
from scrapebadger_cli.output import render
|
|
7
7
|
|
|
8
8
|
app = typer.Typer(no_args_is_help=True)
|
|
@@ -21,22 +21,58 @@ def scrape(
|
|
|
21
21
|
return_format: str | None = typer.Option(
|
|
22
22
|
None, "--format", help="Response format: html, markdown, text"
|
|
23
23
|
),
|
|
24
|
+
save: str | None = typer.Option(
|
|
25
|
+
None,
|
|
26
|
+
"--save",
|
|
27
|
+
help="Write the response body to this file. Required for binary targets "
|
|
28
|
+
"(images, PDFs, archives) — use it to download a file.",
|
|
29
|
+
),
|
|
24
30
|
fmt: str = FMT,
|
|
25
31
|
fields: str | None = FIELDS,
|
|
26
32
|
) -> None:
|
|
27
33
|
"""Scrape a URL and return its content."""
|
|
34
|
+
# These option names are the CLI's contract; the wire names are
|
|
35
|
+
# ScrapeRequest's. Three of them differ, and ScrapeRequest sets no
|
|
36
|
+
# model_config, so pydantic SILENTLY DROPS unknown keys — sending
|
|
37
|
+
# timeout_ms/proxy_country/return_format was a no-op that looked like it
|
|
38
|
+
# worked. Map them explicitly.
|
|
28
39
|
body: dict = {"url": url}
|
|
29
40
|
if render_js:
|
|
30
41
|
body["render_js"] = True
|
|
31
42
|
if wait_for:
|
|
32
43
|
body["wait_for"] = wait_for
|
|
33
44
|
if timeout_ms:
|
|
34
|
-
body["
|
|
45
|
+
body["wait_timeout"] = timeout_ms
|
|
35
46
|
if proxy_country:
|
|
36
|
-
body["
|
|
47
|
+
body["country"] = proxy_country
|
|
37
48
|
if return_format:
|
|
38
|
-
body["
|
|
49
|
+
body["format"] = return_format
|
|
50
|
+
|
|
51
|
+
# --save streams the body straight to disk instead of JSON-wrapping it: no
|
|
52
|
+
# base64 expansion, and binary bytes are never decoded.
|
|
53
|
+
if save:
|
|
54
|
+
body["raw_content"] = True
|
|
55
|
+
payload, content_type = api_post_raw("/v1/web/scrape", body)
|
|
56
|
+
with open(save, "wb") as handle:
|
|
57
|
+
handle.write(payload)
|
|
58
|
+
typer.echo(f"Wrote {len(payload)} bytes to {save} ({content_type or 'unknown type'})")
|
|
59
|
+
return
|
|
60
|
+
|
|
39
61
|
data = api_post("/v1/web/scrape", body)
|
|
62
|
+
|
|
63
|
+
# A binary target returns base64; dumping megabytes of it into a terminal
|
|
64
|
+
# helps nobody. Say what it is and how to get it.
|
|
65
|
+
if isinstance(data, dict) and data.get("is_binary"):
|
|
66
|
+
encoded = data.get("content_base64")
|
|
67
|
+
size = data.get("content_length", 0)
|
|
68
|
+
kind = data.get("content_type") or "binary"
|
|
69
|
+
if encoded:
|
|
70
|
+
data = {**data, "content_base64": f"<{size} bytes of {kind}, omitted>"}
|
|
71
|
+
typer.echo(
|
|
72
|
+
f"Binary response ({kind}, {size} bytes). Re-run with --save PATH to download it.",
|
|
73
|
+
err=True,
|
|
74
|
+
)
|
|
75
|
+
|
|
40
76
|
render(data, fmt, fields)
|
|
41
77
|
|
|
42
78
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/idealista.py
RENAMED
|
File without changes
|
{scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/immobiliare.py
RENAMED
|
File without changes
|
{scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/leboncoin.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|