scrapebadger-cli 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,117 @@
1
+ """Web scraping commands."""
2
+
3
+ import typer
4
+
5
+ from scrapebadger_cli.client import api_get, api_post
6
+ from scrapebadger_cli.output import render
7
+
8
+ app = typer.Typer(no_args_is_help=True)
9
+
10
+ FMT = typer.Option("json", "--output", "-o", help="Output format: json, csv, table, markdown")
11
+ FIELDS = typer.Option(None, "--fields", "-f", help="Comma-separated fields to include")
12
+
13
+
14
+ @app.command()
15
+ def scrape(
16
+ url: str = typer.Argument(..., help="URL to scrape"),
17
+ render_js: bool = typer.Option(False, "--js", help="Enable JavaScript rendering"),
18
+ wait_for: str | None = typer.Option(None, "--wait-for", help="CSS selector to wait for"),
19
+ timeout_ms: int | None = typer.Option(None, "--timeout", help="Timeout in milliseconds"),
20
+ proxy_country: str | None = typer.Option(None, "--proxy", help="Proxy country code"),
21
+ return_format: str | None = typer.Option(
22
+ None, "--format", help="Response format: html, markdown, text"
23
+ ),
24
+ fmt: str = FMT,
25
+ fields: str | None = FIELDS,
26
+ ) -> None:
27
+ """Scrape a URL and return its content."""
28
+ body: dict = {"url": url}
29
+ if render_js:
30
+ body["render_js"] = True
31
+ if wait_for:
32
+ body["wait_for"] = wait_for
33
+ if timeout_ms:
34
+ body["timeout_ms"] = timeout_ms
35
+ if proxy_country:
36
+ body["proxy_country"] = proxy_country
37
+ if return_format:
38
+ body["return_format"] = return_format
39
+ data = api_post("/v1/web/scrape", body)
40
+ render(data, fmt, fields)
41
+
42
+
43
+ @app.command()
44
+ def detect(
45
+ url: str = typer.Argument(..., help="URL to analyze"),
46
+ fmt: str = FMT,
47
+ fields: str | None = FIELDS,
48
+ ) -> None:
49
+ """Detect anti-bot and CAPTCHA protection on a URL."""
50
+ data = api_post("/v1/web/detect", {"url": url})
51
+ render(data, fmt, fields)
52
+
53
+
54
+ @app.command()
55
+ def screenshot(
56
+ url: str = typer.Argument(..., help="URL to screenshot"),
57
+ full_page: bool = typer.Option(False, "--full-page", help="Capture full scrollable page"),
58
+ width: int | None = typer.Option(None, "--width", help="Viewport width"),
59
+ height: int | None = typer.Option(None, "--height", help="Viewport height"),
60
+ fmt: str = FMT,
61
+ fields: str | None = FIELDS,
62
+ ) -> None:
63
+ """Take a screenshot of a URL."""
64
+ body: dict = {"url": url}
65
+ if full_page:
66
+ body["full_page"] = True
67
+ if width:
68
+ body["width"] = width
69
+ if height:
70
+ body["height"] = height
71
+ data = api_post("/v1/web/screenshot", body)
72
+ render(data, fmt, fields)
73
+
74
+
75
+ @app.command()
76
+ def extract(
77
+ url: str = typer.Argument(..., help="URL to extract data from"),
78
+ rules: str | None = typer.Option(None, "--rules", help="CSS/XPath rules as JSON"),
79
+ ai_rules: str | None = typer.Option(None, "--ai-rules", help="AI extraction rules as JSON"),
80
+ ai_query: str | None = typer.Option(None, "--ai-query", help="Freeform AI query"),
81
+ fmt: str = FMT,
82
+ fields: str | None = FIELDS,
83
+ ) -> None:
84
+ """Extract structured data from a URL."""
85
+ import json as json_mod
86
+
87
+ body: dict = {"url": url}
88
+ if rules:
89
+ body["extract_rules"] = json_mod.loads(rules)
90
+ if ai_rules:
91
+ body["ai_extract_rules"] = json_mod.loads(ai_rules)
92
+ if ai_query:
93
+ body["ai_query"] = ai_query
94
+ data = api_post("/v1/web/extract", body)
95
+ render(data, fmt, fields)
96
+
97
+
98
+ @app.command(name="batch")
99
+ def batch_submit(
100
+ urls: list[str] = typer.Argument(..., help="URLs to scrape (space-separated)"),
101
+ fmt: str = FMT,
102
+ fields: str | None = FIELDS,
103
+ ) -> None:
104
+ """Submit a batch of URLs for scraping."""
105
+ data = api_post("/v1/web/batch", {"urls": urls})
106
+ render(data, fmt, fields)
107
+
108
+
109
+ @app.command(name="batch-status")
110
+ def batch_status(
111
+ job_id: str = typer.Argument(..., help="Batch job ID"),
112
+ fmt: str = FMT,
113
+ fields: str | None = FIELDS,
114
+ ) -> None:
115
+ """Get the status of a batch scraping job."""
116
+ data = api_get(f"/v1/web/batch/{job_id}")
117
+ render(data, fmt, fields)