scrapebadger-cli 0.4.2__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/PKG-INFO +1 -1
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/pyproject.toml +1 -1
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/client.py +13 -0
- scrapebadger_cli-0.6.0/src/scrapebadger_cli/commands/chatgpt.py +73 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/web.py +40 -4
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/main.py +2 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/output.py +1 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/.gitignore +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/README.md +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/__init__.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/__init__.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/amazon.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/auth.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/depop.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/ebay.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/google.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/idealista.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/immobiliare.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/leboncoin.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/linkedin.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/loopnet.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/realtor.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/reddit.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/redfin.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/stream.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/tiktok.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/twitter.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/vinted.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/youtube.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/zillow.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/config.py +0 -0
- {scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/uv.lock +0 -0
|
@@ -43,6 +43,19 @@ def api_post(path: str, json_body: dict[str, Any] | None = None) -> dict | list:
|
|
|
43
43
|
return _handle_response(resp)
|
|
44
44
|
|
|
45
45
|
|
|
46
|
+
def api_post_raw(path: str, json_body: dict[str, Any] | None = None) -> tuple[bytes, str]:
|
|
47
|
+
"""POST and return ``(body_bytes, content_type)`` without JSON-decoding.
|
|
48
|
+
|
|
49
|
+
For `/v1/web/scrape` with `raw_content`, whose response is the scraped body
|
|
50
|
+
itself. Returns bytes, not text: the payload may be an image or a PDF and
|
|
51
|
+
decoding it would corrupt it.
|
|
52
|
+
"""
|
|
53
|
+
resp = _get_client().post(path, json=json_body)
|
|
54
|
+
if resp.status_code >= 400:
|
|
55
|
+
_handle_response(resp)
|
|
56
|
+
return resp.content, resp.headers.get("content-type", "")
|
|
57
|
+
|
|
58
|
+
|
|
46
59
|
def _handle_response(resp: httpx.Response) -> dict | list:
|
|
47
60
|
"""Handle API response, exit on errors."""
|
|
48
61
|
if resp.status_code >= 400:
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""ChatGPT LLM scraping commands."""
|
|
2
|
+
|
|
3
|
+
import typer
|
|
4
|
+
|
|
5
|
+
from scrapebadger_cli.client import api_get
|
|
6
|
+
from scrapebadger_cli.output import render
|
|
7
|
+
|
|
8
|
+
app = typer.Typer(no_args_is_help=True)
|
|
9
|
+
|
|
10
|
+
FMT = typer.Option("json", "--output", "-o", help="Output format: json, csv, table, markdown")
|
|
11
|
+
FIELDS = typer.Option(None, "--fields", "-f", help="Comma-separated fields to include")
|
|
12
|
+
COUNTRY = typer.Option("US", "--country", "-c", help="ISO-3166 alpha-2 egress country")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@app.command()
|
|
16
|
+
def ask(
|
|
17
|
+
prompt: str = typer.Argument(..., help="Prompt to send to chatgpt.com (max 4096 chars)"),
|
|
18
|
+
country: str = COUNTRY,
|
|
19
|
+
web_search: str = typer.Option(
|
|
20
|
+
"auto", "--web-search", help="auto, force or off — whether ChatGPT should browse"
|
|
21
|
+
),
|
|
22
|
+
fmt: str = FMT,
|
|
23
|
+
fields: str | None = FIELDS,
|
|
24
|
+
) -> None:
|
|
25
|
+
"""Ask the real chatgpt.com and get the answer with its cited sources (takes 20-25s ungrounded, 30-70s with web search)."""
|
|
26
|
+
data = api_get(
|
|
27
|
+
"/v1/chatgpt/ask",
|
|
28
|
+
{"prompt": prompt, "country": country, "web_search": web_search},
|
|
29
|
+
)
|
|
30
|
+
render(data, fmt, fields)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@app.command(name="brand")
|
|
34
|
+
def brand_visibility(
|
|
35
|
+
prompt: str = typer.Argument(..., help="Buyer-intent prompt, e.g. 'best CRM for startups'"),
|
|
36
|
+
brand: str = typer.Option(..., "--brand", "-b", help="Brand name to look for"),
|
|
37
|
+
domain: str | None = typer.Option(None, "--domain", "-d", help="Brand domain, for citations"),
|
|
38
|
+
aliases: str | None = typer.Option(
|
|
39
|
+
None, "--aliases", help="Comma-separated alternative brand spellings"
|
|
40
|
+
),
|
|
41
|
+
competitors: str | None = typer.Option(
|
|
42
|
+
None, "--competitors", help="Comma-separated competitor names"
|
|
43
|
+
),
|
|
44
|
+
country: str = COUNTRY,
|
|
45
|
+
web_search: str = typer.Option("force", "--web-search", help="auto, force or off"),
|
|
46
|
+
fmt: str = FMT,
|
|
47
|
+
fields: str | None = FIELDS,
|
|
48
|
+
) -> None:
|
|
49
|
+
"""Score how a brand appears in a real ChatGPT answer (AEO/GEO, takes 20-25s ungrounded, 30-70s with web search)."""
|
|
50
|
+
data = api_get(
|
|
51
|
+
"/v1/chatgpt/brand-visibility",
|
|
52
|
+
{
|
|
53
|
+
"prompt": prompt,
|
|
54
|
+
"brand": brand,
|
|
55
|
+
"domain": domain,
|
|
56
|
+
"aliases": aliases,
|
|
57
|
+
"competitors": competitors,
|
|
58
|
+
"country": country,
|
|
59
|
+
"web_search": web_search,
|
|
60
|
+
},
|
|
61
|
+
)
|
|
62
|
+
render(data, fmt, fields)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@app.command()
|
|
66
|
+
def models(
|
|
67
|
+
country: str = COUNTRY,
|
|
68
|
+
fmt: str = FMT,
|
|
69
|
+
fields: str | None = FIELDS,
|
|
70
|
+
) -> None:
|
|
71
|
+
"""List the models chatgpt.com currently offers anonymous users."""
|
|
72
|
+
data = api_get("/v1/chatgpt/models", {"country": country})
|
|
73
|
+
render(data, fmt, fields)
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
import typer
|
|
4
4
|
|
|
5
|
-
from scrapebadger_cli.client import api_get, api_post
|
|
5
|
+
from scrapebadger_cli.client import api_get, api_post, api_post_raw
|
|
6
6
|
from scrapebadger_cli.output import render
|
|
7
7
|
|
|
8
8
|
app = typer.Typer(no_args_is_help=True)
|
|
@@ -21,22 +21,58 @@ def scrape(
|
|
|
21
21
|
return_format: str | None = typer.Option(
|
|
22
22
|
None, "--format", help="Response format: html, markdown, text"
|
|
23
23
|
),
|
|
24
|
+
save: str | None = typer.Option(
|
|
25
|
+
None,
|
|
26
|
+
"--save",
|
|
27
|
+
help="Write the response body to this file. Required for binary targets "
|
|
28
|
+
"(images, PDFs, archives) — use it to download a file.",
|
|
29
|
+
),
|
|
24
30
|
fmt: str = FMT,
|
|
25
31
|
fields: str | None = FIELDS,
|
|
26
32
|
) -> None:
|
|
27
33
|
"""Scrape a URL and return its content."""
|
|
34
|
+
# These option names are the CLI's contract; the wire names are
|
|
35
|
+
# ScrapeRequest's. Three of them differ, and ScrapeRequest sets no
|
|
36
|
+
# model_config, so pydantic SILENTLY DROPS unknown keys — sending
|
|
37
|
+
# timeout_ms/proxy_country/return_format was a no-op that looked like it
|
|
38
|
+
# worked. Map them explicitly.
|
|
28
39
|
body: dict = {"url": url}
|
|
29
40
|
if render_js:
|
|
30
41
|
body["render_js"] = True
|
|
31
42
|
if wait_for:
|
|
32
43
|
body["wait_for"] = wait_for
|
|
33
44
|
if timeout_ms:
|
|
34
|
-
body["
|
|
45
|
+
body["wait_timeout"] = timeout_ms
|
|
35
46
|
if proxy_country:
|
|
36
|
-
body["
|
|
47
|
+
body["country"] = proxy_country
|
|
37
48
|
if return_format:
|
|
38
|
-
body["
|
|
49
|
+
body["format"] = return_format
|
|
50
|
+
|
|
51
|
+
# --save streams the body straight to disk instead of JSON-wrapping it: no
|
|
52
|
+
# base64 expansion, and binary bytes are never decoded.
|
|
53
|
+
if save:
|
|
54
|
+
body["raw_content"] = True
|
|
55
|
+
payload, content_type = api_post_raw("/v1/web/scrape", body)
|
|
56
|
+
with open(save, "wb") as handle:
|
|
57
|
+
handle.write(payload)
|
|
58
|
+
typer.echo(f"Wrote {len(payload)} bytes to {save} ({content_type or 'unknown type'})")
|
|
59
|
+
return
|
|
60
|
+
|
|
39
61
|
data = api_post("/v1/web/scrape", body)
|
|
62
|
+
|
|
63
|
+
# A binary target returns base64; dumping megabytes of it into a terminal
|
|
64
|
+
# helps nobody. Say what it is and how to get it.
|
|
65
|
+
if isinstance(data, dict) and data.get("is_binary"):
|
|
66
|
+
encoded = data.get("content_base64")
|
|
67
|
+
size = data.get("content_length", 0)
|
|
68
|
+
kind = data.get("content_type") or "binary"
|
|
69
|
+
if encoded:
|
|
70
|
+
data = {**data, "content_base64": f"<{size} bytes of {kind}, omitted>"}
|
|
71
|
+
typer.echo(
|
|
72
|
+
f"Binary response ({kind}, {size} bytes). Re-run with --save PATH to download it.",
|
|
73
|
+
err=True,
|
|
74
|
+
)
|
|
75
|
+
|
|
40
76
|
render(data, fmt, fields)
|
|
41
77
|
|
|
42
78
|
|
|
@@ -5,6 +5,7 @@ import typer
|
|
|
5
5
|
from scrapebadger_cli.commands import (
|
|
6
6
|
amazon,
|
|
7
7
|
auth,
|
|
8
|
+
chatgpt,
|
|
8
9
|
depop,
|
|
9
10
|
ebay,
|
|
10
11
|
google,
|
|
@@ -50,6 +51,7 @@ app.add_typer(loopnet.app, name="loopnet", help="LoopNet commercial real estate
|
|
|
50
51
|
app.add_typer(depop.app, name="depop", help="Depop marketplace commands")
|
|
51
52
|
app.add_typer(redfin.app, name="redfin", help="Redfin real-estate commands")
|
|
52
53
|
app.add_typer(linkedin.app, name="linkedin", help="LinkedIn public data commands")
|
|
54
|
+
app.add_typer(chatgpt.app, name="chatgpt", help="ChatGPT LLM scraping commands")
|
|
53
55
|
app.command()(auth.auth)
|
|
54
56
|
app.command(name="credits")(auth.credits)
|
|
55
57
|
app.command(name="logout")(auth.logout)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/idealista.py
RENAMED
|
File without changes
|
{scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/immobiliare.py
RENAMED
|
File without changes
|
{scrapebadger_cli-0.4.2 → scrapebadger_cli-0.6.0}/src/scrapebadger_cli/commands/leboncoin.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|