scrapebadger-cli 0.4.2__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/PKG-INFO +1 -1
  2. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/pyproject.toml +1 -1
  3. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/client.py +13 -0
  4. scrapebadger_cli-0.6.0/src/scrapebadger_cli/commands/chatgpt.py +73 -0
  5. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/web.py +40 -4
  6. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/main.py +2 -0
  7. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/output.py +1 -0
  8. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/.gitignore +0 -0
  9. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/README.md +0 -0
  10. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/__init__.py +0 -0
  11. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/__init__.py +0 -0
  12. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/amazon.py +0 -0
  13. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/auth.py +0 -0
  14. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/depop.py +0 -0
  15. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/ebay.py +0 -0
  16. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/google.py +0 -0
  17. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/idealista.py +0 -0
  18. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/immobiliare.py +0 -0
  19. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/leboncoin.py +0 -0
  20. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/linkedin.py +0 -0
  21. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/loopnet.py +0 -0
  22. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/realtor.py +0 -0
  23. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/reddit.py +0 -0
  24. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/redfin.py +0 -0
  25. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/stream.py +0 -0
  26. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/tiktok.py +0 -0
  27. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/twitter.py +0 -0
  28. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/vinted.py +0 -0
  29. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/youtube.py +0 -0
  30. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/zillow.py +0 -0
  31. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/config.py +0 -0
  32. {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scrapebadger-cli
3
- Version: 0.4.2
3
+ Version: 0.6.0
4
4
  Summary: ScrapeBadger CLI — Twitter, Vinted, Google, and Web Scraping from the command line
5
5
  Author: ScrapeBadger Team
6
6
  License: MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "scrapebadger-cli"
3
- version = "0.4.2"
3
+ version = "0.6.0"
4
4
  description = "ScrapeBadger CLI — Twitter, Vinted, Google, and Web Scraping from the command line"
5
5
  requires-python = ">=3.10"
6
6
  license = {text = "MIT"}
@@ -43,6 +43,19 @@ def api_post(path: str, json_body: dict[str, Any] | None = None) -> dict | list:
43
43
  return _handle_response(resp)
44
44
 
45
45
 
46
+ def api_post_raw(path: str, json_body: dict[str, Any] | None = None) -> tuple[bytes, str]:
47
+ """POST and return ``(body_bytes, content_type)`` without JSON-decoding.
48
+
49
+ For `/v1/web/scrape` with `raw_content`, whose response is the scraped body
50
+ itself. Returns bytes, not text: the payload may be an image or a PDF and
51
+ decoding it would corrupt it.
52
+ """
53
+ resp = _get_client().post(path, json=json_body)
54
+ if resp.status_code >= 400:
55
+ _handle_response(resp)
56
+ return resp.content, resp.headers.get("content-type", "")
57
+
58
+
46
59
  def _handle_response(resp: httpx.Response) -> dict | list:
47
60
  """Handle API response, exit on errors."""
48
61
  if resp.status_code >= 400:
@@ -0,0 +1,73 @@
1
+ """ChatGPT LLM scraping commands."""
2
+
3
+ import typer
4
+
5
+ from scrapebadger_cli.client import api_get
6
+ from scrapebadger_cli.output import render
7
+
8
+ app = typer.Typer(no_args_is_help=True)
9
+
10
+ FMT = typer.Option("json", "--output", "-o", help="Output format: json, csv, table, markdown")
11
+ FIELDS = typer.Option(None, "--fields", "-f", help="Comma-separated fields to include")
12
+ COUNTRY = typer.Option("US", "--country", "-c", help="ISO-3166 alpha-2 egress country")
13
+
14
+
15
+ @app.command()
16
+ def ask(
17
+ prompt: str = typer.Argument(..., help="Prompt to send to chatgpt.com (max 4096 chars)"),
18
+ country: str = COUNTRY,
19
+ web_search: str = typer.Option(
20
+ "auto", "--web-search", help="auto, force or off — whether ChatGPT should browse"
21
+ ),
22
+ fmt: str = FMT,
23
+ fields: str | None = FIELDS,
24
+ ) -> None:
25
+ """Ask the real chatgpt.com and get the answer with its cited sources (takes 20-25s ungrounded, 30-70s with web search)."""
26
+ data = api_get(
27
+ "/v1/chatgpt/ask",
28
+ {"prompt": prompt, "country": country, "web_search": web_search},
29
+ )
30
+ render(data, fmt, fields)
31
+
32
+
33
+ @app.command(name="brand")
34
+ def brand_visibility(
35
+ prompt: str = typer.Argument(..., help="Buyer-intent prompt, e.g. 'best CRM for startups'"),
36
+ brand: str = typer.Option(..., "--brand", "-b", help="Brand name to look for"),
37
+ domain: str | None = typer.Option(None, "--domain", "-d", help="Brand domain, for citations"),
38
+ aliases: str | None = typer.Option(
39
+ None, "--aliases", help="Comma-separated alternative brand spellings"
40
+ ),
41
+ competitors: str | None = typer.Option(
42
+ None, "--competitors", help="Comma-separated competitor names"
43
+ ),
44
+ country: str = COUNTRY,
45
+ web_search: str = typer.Option("force", "--web-search", help="auto, force or off"),
46
+ fmt: str = FMT,
47
+ fields: str | None = FIELDS,
48
+ ) -> None:
49
+ """Score how a brand appears in a real ChatGPT answer (AEO/GEO, takes 20-25s ungrounded, 30-70s with web search)."""
50
+ data = api_get(
51
+ "/v1/chatgpt/brand-visibility",
52
+ {
53
+ "prompt": prompt,
54
+ "brand": brand,
55
+ "domain": domain,
56
+ "aliases": aliases,
57
+ "competitors": competitors,
58
+ "country": country,
59
+ "web_search": web_search,
60
+ },
61
+ )
62
+ render(data, fmt, fields)
63
+
64
+
65
+ @app.command()
66
+ def models(
67
+ country: str = COUNTRY,
68
+ fmt: str = FMT,
69
+ fields: str | None = FIELDS,
70
+ ) -> None:
71
+ """List the models chatgpt.com currently offers anonymous users."""
72
+ data = api_get("/v1/chatgpt/models", {"country": country})
73
+ render(data, fmt, fields)
@@ -2,7 +2,7 @@
2
2
 
3
3
  import typer
4
4
 
5
- from scrapebadger_cli.client import api_get, api_post
5
+ from scrapebadger_cli.client import api_get, api_post, api_post_raw
6
6
  from scrapebadger_cli.output import render
7
7
 
8
8
  app = typer.Typer(no_args_is_help=True)
@@ -21,22 +21,58 @@ def scrape(
21
21
  return_format: str | None = typer.Option(
22
22
  None, "--format", help="Response format: html, markdown, text"
23
23
  ),
24
+ save: str | None = typer.Option(
25
+ None,
26
+ "--save",
27
+ help="Write the response body to this file. Required for binary targets "
28
+ "(images, PDFs, archives) — use it to download a file.",
29
+ ),
24
30
  fmt: str = FMT,
25
31
  fields: str | None = FIELDS,
26
32
  ) -> None:
27
33
  """Scrape a URL and return its content."""
34
+ # These option names are the CLI's contract; the wire names are
35
+ # ScrapeRequest's. Three of them differ, and ScrapeRequest sets no
36
+ # model_config, so pydantic SILENTLY DROPS unknown keys — sending
37
+ # timeout_ms/proxy_country/return_format was a no-op that looked like it
38
+ # worked. Map them explicitly.
28
39
  body: dict = {"url": url}
29
40
  if render_js:
30
41
  body["render_js"] = True
31
42
  if wait_for:
32
43
  body["wait_for"] = wait_for
33
44
  if timeout_ms:
34
- body["timeout_ms"] = timeout_ms
45
+ body["wait_timeout"] = timeout_ms
35
46
  if proxy_country:
36
- body["proxy_country"] = proxy_country
47
+ body["country"] = proxy_country
37
48
  if return_format:
38
- body["return_format"] = return_format
49
+ body["format"] = return_format
50
+
51
+ # --save streams the body straight to disk instead of JSON-wrapping it: no
52
+ # base64 expansion, and binary bytes are never decoded.
53
+ if save:
54
+ body["raw_content"] = True
55
+ payload, content_type = api_post_raw("/v1/web/scrape", body)
56
+ with open(save, "wb") as handle:
57
+ handle.write(payload)
58
+ typer.echo(f"Wrote {len(payload)} bytes to {save} ({content_type or 'unknown type'})")
59
+ return
60
+
39
61
  data = api_post("/v1/web/scrape", body)
62
+
63
+ # A binary target returns base64; dumping megabytes of it into a terminal
64
+ # helps nobody. Say what it is and how to get it.
65
+ if isinstance(data, dict) and data.get("is_binary"):
66
+ encoded = data.get("content_base64")
67
+ size = data.get("content_length", 0)
68
+ kind = data.get("content_type") or "binary"
69
+ if encoded:
70
+ data = {**data, "content_base64": f"<{size} bytes of {kind}, omitted>"}
71
+ typer.echo(
72
+ f"Binary response ({kind}, {size} bytes). Re-run with --save PATH to download it.",
73
+ err=True,
74
+ )
75
+
40
76
  render(data, fmt, fields)
41
77
 
42
78
 
@@ -5,6 +5,7 @@ import typer
5
5
  from scrapebadger_cli.commands import (
6
6
  amazon,
7
7
  auth,
8
+ chatgpt,
8
9
  depop,
9
10
  ebay,
10
11
  google,
@@ -50,6 +51,7 @@ app.add_typer(loopnet.app, name="loopnet", help="LoopNet commercial real estate
50
51
  app.add_typer(depop.app, name="depop", help="Depop marketplace commands")
51
52
  app.add_typer(redfin.app, name="redfin", help="Redfin real-estate commands")
52
53
  app.add_typer(linkedin.app, name="linkedin", help="LinkedIn public data commands")
54
+ app.add_typer(chatgpt.app, name="chatgpt", help="ChatGPT LLM scraping commands")
53
55
  app.command()(auth.auth)
54
56
  app.command(name="credits")(auth.credits)
55
57
  app.command(name="logout")(auth.logout)
@@ -55,6 +55,7 @@ def _normalize_rows(data: Any) -> list[dict]:
55
55
  "logs",
56
56
  "tiers",
57
57
  "monitors",
58
+ "models",
58
59
  ):
59
60
  if key in data and isinstance(data[key], list):
60
61
  return [r if isinstance(r, dict) else {"value": r} for r in data[key]]