scrapebadger-cli 0.4.2__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/PKG-INFO +1 -1
  2. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/pyproject.toml +1 -1
  3. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/client.py +13 -0
  4. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/web.py +40 -4
  5. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/.gitignore +0 -0
  6. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/README.md +0 -0
  7. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/__init__.py +0 -0
  8. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/__init__.py +0 -0
  9. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/amazon.py +0 -0
  10. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/auth.py +0 -0
  11. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/depop.py +0 -0
  12. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/ebay.py +0 -0
  13. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/google.py +0 -0
  14. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/idealista.py +0 -0
  15. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/immobiliare.py +0 -0
  16. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/leboncoin.py +0 -0
  17. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/linkedin.py +0 -0
  18. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/loopnet.py +0 -0
  19. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/realtor.py +0 -0
  20. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/reddit.py +0 -0
  21. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/redfin.py +0 -0
  22. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/stream.py +0 -0
  23. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/tiktok.py +0 -0
  24. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/twitter.py +0 -0
  25. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/vinted.py +0 -0
  26. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/youtube.py +0 -0
  27. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/commands/zillow.py +0 -0
  28. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/config.py +0 -0
  29. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/main.py +0 -0
  30. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/src/scrapebadger_cli/output.py +0 -0
  31. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.5.0}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scrapebadger-cli
3
- Version: 0.4.2
3
+ Version: 0.5.0
4
4
  Summary: ScrapeBadger CLI — Twitter, Vinted, Google, and Web Scraping from the command line
5
5
  Author: ScrapeBadger Team
6
6
  License: MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "scrapebadger-cli"
3
- version = "0.4.2"
3
+ version = "0.5.0"
4
4
  description = "ScrapeBadger CLI — Twitter, Vinted, Google, and Web Scraping from the command line"
5
5
  requires-python = ">=3.10"
6
6
  license = {text = "MIT"}
@@ -43,6 +43,19 @@ def api_post(path: str, json_body: dict[str, Any] | None = None) -> dict | list:
43
43
  return _handle_response(resp)
44
44
 
45
45
 
46
+ def api_post_raw(path: str, json_body: dict[str, Any] | None = None) -> tuple[bytes, str]:
47
+ """POST and return ``(body_bytes, content_type)`` without JSON-decoding.
48
+
49
+ For `/v1/web/scrape` with `raw_content`, whose response is the scraped body
50
+ itself. Returns bytes, not text: the payload may be an image or a PDF and
51
+ decoding it would corrupt it.
52
+ """
53
+ resp = _get_client().post(path, json=json_body)
54
+ if resp.status_code >= 400:
55
+ _handle_response(resp)
56
+ return resp.content, resp.headers.get("content-type", "")
57
+
58
+
46
59
  def _handle_response(resp: httpx.Response) -> dict | list:
47
60
  """Handle API response, exit on errors."""
48
61
  if resp.status_code >= 400:
@@ -2,7 +2,7 @@
2
2
 
3
3
  import typer
4
4
 
5
- from scrapebadger_cli.client import api_get, api_post
5
+ from scrapebadger_cli.client import api_get, api_post, api_post_raw
6
6
  from scrapebadger_cli.output import render
7
7
 
8
8
  app = typer.Typer(no_args_is_help=True)
@@ -21,22 +21,58 @@ def scrape(
21
21
  return_format: str | None = typer.Option(
22
22
  None, "--format", help="Response format: html, markdown, text"
23
23
  ),
24
+ save: str | None = typer.Option(
25
+ None,
26
+ "--save",
27
+ help="Write the response body to this file. Required for binary targets "
28
+ "(images, PDFs, archives) — use it to download a file.",
29
+ ),
24
30
  fmt: str = FMT,
25
31
  fields: str | None = FIELDS,
26
32
  ) -> None:
27
33
  """Scrape a URL and return its content."""
34
+ # These option names are the CLI's contract; the wire names are
35
+ # ScrapeRequest's. Three of them differ, and ScrapeRequest sets no
36
+ # model_config, so pydantic SILENTLY DROPS unknown keys — sending
37
+ # timeout_ms/proxy_country/return_format was a no-op that looked like it
38
+ # worked. Map them explicitly.
28
39
  body: dict = {"url": url}
29
40
  if render_js:
30
41
  body["render_js"] = True
31
42
  if wait_for:
32
43
  body["wait_for"] = wait_for
33
44
  if timeout_ms:
34
- body["timeout_ms"] = timeout_ms
45
+ body["wait_timeout"] = timeout_ms
35
46
  if proxy_country:
36
- body["proxy_country"] = proxy_country
47
+ body["country"] = proxy_country
37
48
  if return_format:
38
- body["return_format"] = return_format
49
+ body["format"] = return_format
50
+
51
+ # --save streams the body straight to disk instead of JSON-wrapping it: no
52
+ # base64 expansion, and binary bytes are never decoded.
53
+ if save:
54
+ body["raw_content"] = True
55
+ payload, content_type = api_post_raw("/v1/web/scrape", body)
56
+ with open(save, "wb") as handle:
57
+ handle.write(payload)
58
+ typer.echo(f"Wrote {len(payload)} bytes to {save} ({content_type or 'unknown type'})")
59
+ return
60
+
39
61
  data = api_post("/v1/web/scrape", body)
62
+
63
+ # A binary target returns base64; dumping megabytes of it into a terminal
64
+ # helps nobody. Say what it is and how to get it.
65
+ if isinstance(data, dict) and data.get("is_binary"):
66
+ encoded = data.get("content_base64")
67
+ size = data.get("content_length", 0)
68
+ kind = data.get("content_type") or "binary"
69
+ if encoded:
70
+ data = {**data, "content_base64": f"<{size} bytes of {kind}, omitted>"}
71
+ typer.echo(
72
+ f"Binary response ({kind}, {size} bytes). Re-run with --save PATH to download it.",
73
+ err=True,
74
+ )
75
+
40
76
  render(data, fmt, fields)
41
77
 
42
78