scrapewise 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scrapewise-0.1.0/.gitignore +8 -0
- scrapewise-0.1.0/LICENSE +21 -0
- scrapewise-0.1.0/PKG-INFO +190 -0
- scrapewise-0.1.0/README.md +157 -0
- scrapewise-0.1.0/pyproject.toml +56 -0
- scrapewise-0.1.0/src/scrapewise/__init__.py +90 -0
- scrapewise-0.1.0/src/scrapewise/client.py +855 -0
- scrapewise-0.1.0/src/scrapewise/errors.py +179 -0
- scrapewise-0.1.0/src/scrapewise/py.typed +0 -0
- scrapewise-0.1.0/src/scrapewise/types.py +235 -0
- scrapewise-0.1.0/tests/conftest.py +23 -0
- scrapewise-0.1.0/tests/test_client_config.py +118 -0
- scrapewise-0.1.0/tests/test_errors.py +127 -0
- scrapewise-0.1.0/tests/test_helpers.py +93 -0
- scrapewise-0.1.0/tests/test_routes.py +336 -0
scrapewise-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 BEBOTECH OÜ
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: scrapewise
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Python client for the ScrapeWise web-scraping and price-monitoring API.
|
|
5
|
+
Project-URL: Homepage, https://scrapewise.ai
|
|
6
|
+
Project-URL: Documentation, https://scrapewise.ai
|
|
7
|
+
Project-URL: Repository, https://github.com/BEBOTECH-OU/scrapewise-python
|
|
8
|
+
Project-URL: Issues, https://github.com/BEBOTECH-OU/scrapewise-python/issues
|
|
9
|
+
Author-email: BEBOTECH OÜ <hello@scrapewise.ai>
|
|
10
|
+
Maintainer-email: BEBOTECH OÜ <hello@scrapewise.ai>
|
|
11
|
+
License: MIT
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Keywords: api-client,competitor-prices,ecommerce,price-monitoring,product-data,scrapewise,web-scraping
|
|
14
|
+
Classifier: Development Status :: 4 - Beta
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
25
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
26
|
+
Classifier: Typing :: Typed
|
|
27
|
+
Requires-Python: >=3.9
|
|
28
|
+
Requires-Dist: httpx<1,>=0.24
|
|
29
|
+
Provides-Extra: test
|
|
30
|
+
Requires-Dist: pytest>=7.4; extra == 'test'
|
|
31
|
+
Requires-Dist: respx>=0.20; extra == 'test'
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
# scrapewise
|
|
35
|
+
|
|
36
|
+
Python client for the [ScrapeWise](https://scrapewise.ai) web-scraping and
|
|
37
|
+
price-monitoring API. One runtime dependency (`httpx`), fully type-hinted,
|
|
38
|
+
synchronous.
|
|
39
|
+
|
|
40
|
+
## Install
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install scrapewise
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## Quickstart
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
from scrapewise import ScrapewiseClient
|
|
50
|
+
|
|
51
|
+
with ScrapewiseClient(api_key="YOUR_SCRAPEWISE_API_KEY") as sw: # or env SCRAPEWISE_API_KEY
|
|
52
|
+
groups = sw.list_groups() # GET /scraper/group/list
|
|
53
|
+
rows = sw.get_sample_data(sw.list_scrapers()[0]["id"]) # GET /scraper/{id}/get-sample-data
|
|
54
|
+
print(len(groups), len(rows))
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Authentication
|
|
58
|
+
|
|
59
|
+
**Where to get a key:** sign in at [portal.scrapewise.ai](https://portal.scrapewise.ai)
|
|
60
|
+
and create one under **Settings → API Keys**. Two scopes exist:
|
|
61
|
+
|
|
62
|
+
| Scope | What it can do |
|
|
63
|
+
|---|---|
|
|
64
|
+
| `LLM_READ` | Lists your groups, reads saved rows and job results. Cannot create, run or delete anything. |
|
|
65
|
+
| `LLM_FULL` | The above, plus creating and running scrapers. |
|
|
66
|
+
|
|
67
|
+
Your sign-in password is not a key, and a key created anywhere else is rejected
|
|
68
|
+
with `401`. If reads work but creating or running a scraper returns `401`, the
|
|
69
|
+
key is `LLM_READ` and you need an `LLM_FULL` one.
|
|
70
|
+
|
|
71
|
+
Pass `api_key=` explicitly, or set `SCRAPEWISE_API_KEY` in the environment and
|
|
72
|
+
construct with no arguments. The key is sent as `Authorization: Bearer <key>`.
|
|
73
|
+
A missing key raises `ScrapewiseConfigurationError` at construction time
|
|
74
|
+
instead of failing later with a 401.
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
import os
|
|
78
|
+
os.environ["SCRAPEWISE_API_KEY"] = "YOUR_SCRAPEWISE_API_KEY"
|
|
79
|
+
sw = ScrapewiseClient()
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Self-hosted or staging deployments: pass `base_url=` or set
|
|
83
|
+
`SCRAPEWISE_BASE_URL`. The default is
|
|
84
|
+
`https://portal.scrapewise.ai/api/scraper-api/api`.
|
|
85
|
+
|
|
86
|
+
## Timeouts
|
|
87
|
+
|
|
88
|
+
The default timeout is **60 seconds**, chosen to match the hard tool-call
|
|
89
|
+
ceiling of the hosted ScrapeWise MCP gateway so that code written against this
|
|
90
|
+
client behaves identically when driven from an agent. The REST API itself has
|
|
91
|
+
no such ceiling, so you can raise it:
|
|
92
|
+
|
|
93
|
+
```python
|
|
94
|
+
sw = ScrapewiseClient(timeout=180.0) # client-wide
|
|
95
|
+
sw.get_sample_data(scraper_id, timeout=120.0) # per call
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
One route is slow by nature: `load_site()` renders a live page and routinely
|
|
99
|
+
exceeds 60 s, so it defaults to its own 180 s timeout
|
|
100
|
+
(`scrapewise.LOAD_SITE_TIMEOUT`) rather than the client default.
|
|
101
|
+
|
|
102
|
+
## What is wrapped
|
|
103
|
+
|
|
104
|
+
Every method maps to exactly one documented REST route. Nothing is inferred.
|
|
105
|
+
|
|
106
|
+
| Method | Route |
|
|
107
|
+
| --- | --- |
|
|
108
|
+
| `list_scrapers()` | `GET /scraper/list` |
|
|
109
|
+
| `get_scraper(id)` | `GET /scraper/{id}` |
|
|
110
|
+
| `upsert_scraper(payload)` | `PUT /scraper` |
|
|
111
|
+
| `run_scraper(id)` | `GET /scraper/{id}/run` |
|
|
112
|
+
| `get_sample_data(id)` | `GET /scraper/{id}/get-sample-data` |
|
|
113
|
+
| `attach_schema(id, schema_id)` | `PATCH /scraper/{id}/schema` |
|
|
114
|
+
| `get_scraper_site(id)` | `GET /scraper/{id}/site` |
|
|
115
|
+
| `list_site_links(site_id)` | `GET /scraper/site/{siteId}/links` |
|
|
116
|
+
| `replace_site_links(...)` | `PUT /scraper/site` |
|
|
117
|
+
| `load_site(url)` | `GET /scraper/load-site?url=` |
|
|
118
|
+
| `list_groups()` | `GET /scraper/group/list` |
|
|
119
|
+
| `upsert_group(payload)` | `PUT /scraper/group` |
|
|
120
|
+
| `set_group_currency(id, ccy)` | `PUT /scraper/group/{id}/currency` |
|
|
121
|
+
| `get_group_data(id)` | `GET /scraper/data/group/{id}` |
|
|
122
|
+
| `get_group_post_process_rules(id)` | `GET /scraper/group/{id}/post-process-rules` |
|
|
123
|
+
| `update_group_post_process_rules(...)` | `PUT /scraper/group/{id}/post-process-rules` |
|
|
124
|
+
| `get_load_history(scraper_id)` | `GET /scraper/load-history?scraperId=` |
|
|
125
|
+
| `get_job_errors(group_id, job_id)` | `GET /scraper/load-history/group/{groupId}/job/{jobId}/errors` |
|
|
126
|
+
| `preview_rule(rule, sample_value)` | `POST /scraper/preview-rule` |
|
|
127
|
+
| `list_customer_schemas()` | `GET /schema/customer` |
|
|
128
|
+
| `get_schema(schema_id)` | `GET /schema/get/{id}` |
|
|
129
|
+
| `publish_customer_schema(payload)` | `PUT /schema/customer` |
|
|
130
|
+
| `list_desktops()` | `GET /scraper/desktop` |
|
|
131
|
+
|
|
132
|
+
Two helpers are built on top of those routes and add no new ones:
|
|
133
|
+
`iter_group_data()` pages through `get_group_data()`, and `latest_run()`
|
|
134
|
+
returns the newest `get_load_history()` entry.
|
|
135
|
+
|
|
136
|
+
## Errors
|
|
137
|
+
|
|
138
|
+
All failures raise a subclass of `ScrapewiseError`, which carries `status_code`,
|
|
139
|
+
the raw `body`, the request `method`/`url`, and a `payload` property that parses
|
|
140
|
+
the body as JSON when possible.
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
from scrapewise import ScrapewiseAuthError, ScrapewiseError
|
|
144
|
+
|
|
145
|
+
try:
|
|
146
|
+
sw.list_scrapers()
|
|
147
|
+
except ScrapewiseAuthError as exc:
|
|
148
|
+
print("bad key:", exc.status_code)
|
|
149
|
+
except ScrapewiseError as exc:
|
|
150
|
+
print(exc.status_code, exc.payload)
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
`400 → ScrapewiseBadRequestError`, `401/403 → ScrapewiseAuthError`,
|
|
154
|
+
`404 → ScrapewiseNotFoundError`, `409 → ScrapewiseConflictError`,
|
|
155
|
+
`429 → ScrapewiseRateLimitError`, `5xx → ScrapewiseServerError`.
|
|
156
|
+
Network failures raise `ScrapewiseTransportError`, timeouts the
|
|
157
|
+
`ScrapewiseTimeoutError` subclass.
|
|
158
|
+
|
|
159
|
+
## Platform behaviour worth knowing
|
|
160
|
+
|
|
161
|
+
These are properties of the ScrapeWise API, not of this client, and are
|
|
162
|
+
repeated in the relevant docstrings:
|
|
163
|
+
|
|
164
|
+
- `get_sample_data()` returns **at most 100 rows**. It is a sample, not an
|
|
165
|
+
export — use `latest_run()["itemsQuantity"]` for the real row count of a run.
|
|
166
|
+
- Paginated responses key rows on `content`, not `items`.
|
|
167
|
+
- `replace_site_links()` **replaces** the whole link list for a site. Links not
|
|
168
|
+
present in the call are removed.
|
|
169
|
+
- `attach_schema()` is effectively one-way: it switches the scraper's extractor
|
|
170
|
+
to the AI/schema path.
|
|
171
|
+
- Writing group post-process rules is optimistically locked. Read returns
|
|
172
|
+
`version`; the write expects `expectedVersion`.
|
|
173
|
+
`update_group_post_process_rules()` does that rename for you.
|
|
174
|
+
- A scraper is scoped to one domain. Posting links from another host fails with
|
|
175
|
+
`LINK_HOST_MISMATCH`.
|
|
176
|
+
|
|
177
|
+
## Development
|
|
178
|
+
|
|
179
|
+
```bash
|
|
180
|
+
python -m venv .venv && . .venv/bin/activate
|
|
181
|
+
pip install -e ".[test]"
|
|
182
|
+
pytest
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
Tests mock HTTP at the transport layer with `respx`; there are no live network
|
|
186
|
+
calls in the suite.
|
|
187
|
+
|
|
188
|
+
## License
|
|
189
|
+
|
|
190
|
+
MIT © BEBOTECH OÜ
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
# scrapewise
|
|
2
|
+
|
|
3
|
+
Python client for the [ScrapeWise](https://scrapewise.ai) web-scraping and
|
|
4
|
+
price-monitoring API. One runtime dependency (`httpx`), fully type-hinted,
|
|
5
|
+
synchronous.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install scrapewise
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Quickstart
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from scrapewise import ScrapewiseClient
|
|
17
|
+
|
|
18
|
+
with ScrapewiseClient(api_key="YOUR_SCRAPEWISE_API_KEY") as sw: # or env SCRAPEWISE_API_KEY
|
|
19
|
+
groups = sw.list_groups() # GET /scraper/group/list
|
|
20
|
+
rows = sw.get_sample_data(sw.list_scrapers()[0]["id"]) # GET /scraper/{id}/get-sample-data
|
|
21
|
+
print(len(groups), len(rows))
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## Authentication
|
|
25
|
+
|
|
26
|
+
**Where to get a key:** sign in at [portal.scrapewise.ai](https://portal.scrapewise.ai)
|
|
27
|
+
and create one under **Settings → API Keys**. Two scopes exist:
|
|
28
|
+
|
|
29
|
+
| Scope | What it can do |
|
|
30
|
+
|---|---|
|
|
31
|
+
| `LLM_READ` | Lists your groups, reads saved rows and job results. Cannot create, run or delete anything. |
|
|
32
|
+
| `LLM_FULL` | The above, plus creating and running scrapers. |
|
|
33
|
+
|
|
34
|
+
Your sign-in password is not a key, and a key created anywhere else is rejected
|
|
35
|
+
with `401`. If reads work but creating or running a scraper returns `401`, the
|
|
36
|
+
key is `LLM_READ` and you need an `LLM_FULL` one.
|
|
37
|
+
|
|
38
|
+
Pass `api_key=` explicitly, or set `SCRAPEWISE_API_KEY` in the environment and
|
|
39
|
+
construct with no arguments. The key is sent as `Authorization: Bearer <key>`.
|
|
40
|
+
A missing key raises `ScrapewiseConfigurationError` at construction time
|
|
41
|
+
instead of failing later with a 401.
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
import os
|
|
45
|
+
os.environ["SCRAPEWISE_API_KEY"] = "YOUR_SCRAPEWISE_API_KEY"
|
|
46
|
+
sw = ScrapewiseClient()
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Self-hosted or staging deployments: pass `base_url=` or set
|
|
50
|
+
`SCRAPEWISE_BASE_URL`. The default is
|
|
51
|
+
`https://portal.scrapewise.ai/api/scraper-api/api`.
|
|
52
|
+
|
|
53
|
+
## Timeouts
|
|
54
|
+
|
|
55
|
+
The default timeout is **60 seconds**, chosen to match the hard tool-call
|
|
56
|
+
ceiling of the hosted ScrapeWise MCP gateway so that code written against this
|
|
57
|
+
client behaves identically when driven from an agent. The REST API itself has
|
|
58
|
+
no such ceiling, so you can raise it:
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
sw = ScrapewiseClient(timeout=180.0) # client-wide
|
|
62
|
+
sw.get_sample_data(scraper_id, timeout=120.0) # per call
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
One route is slow by nature: `load_site()` renders a live page and routinely
|
|
66
|
+
exceeds 60 s, so it defaults to its own 180 s timeout
|
|
67
|
+
(`scrapewise.LOAD_SITE_TIMEOUT`) rather than the client default.
|
|
68
|
+
|
|
69
|
+
## What is wrapped
|
|
70
|
+
|
|
71
|
+
Every method maps to exactly one documented REST route. Nothing is inferred.
|
|
72
|
+
|
|
73
|
+
| Method | Route |
|
|
74
|
+
| --- | --- |
|
|
75
|
+
| `list_scrapers()` | `GET /scraper/list` |
|
|
76
|
+
| `get_scraper(id)` | `GET /scraper/{id}` |
|
|
77
|
+
| `upsert_scraper(payload)` | `PUT /scraper` |
|
|
78
|
+
| `run_scraper(id)` | `GET /scraper/{id}/run` |
|
|
79
|
+
| `get_sample_data(id)` | `GET /scraper/{id}/get-sample-data` |
|
|
80
|
+
| `attach_schema(id, schema_id)` | `PATCH /scraper/{id}/schema` |
|
|
81
|
+
| `get_scraper_site(id)` | `GET /scraper/{id}/site` |
|
|
82
|
+
| `list_site_links(site_id)` | `GET /scraper/site/{siteId}/links` |
|
|
83
|
+
| `replace_site_links(...)` | `PUT /scraper/site` |
|
|
84
|
+
| `load_site(url)` | `GET /scraper/load-site?url=` |
|
|
85
|
+
| `list_groups()` | `GET /scraper/group/list` |
|
|
86
|
+
| `upsert_group(payload)` | `PUT /scraper/group` |
|
|
87
|
+
| `set_group_currency(id, ccy)` | `PUT /scraper/group/{id}/currency` |
|
|
88
|
+
| `get_group_data(id)` | `GET /scraper/data/group/{id}` |
|
|
89
|
+
| `get_group_post_process_rules(id)` | `GET /scraper/group/{id}/post-process-rules` |
|
|
90
|
+
| `update_group_post_process_rules(...)` | `PUT /scraper/group/{id}/post-process-rules` |
|
|
91
|
+
| `get_load_history(scraper_id)` | `GET /scraper/load-history?scraperId=` |
|
|
92
|
+
| `get_job_errors(group_id, job_id)` | `GET /scraper/load-history/group/{groupId}/job/{jobId}/errors` |
|
|
93
|
+
| `preview_rule(rule, sample_value)` | `POST /scraper/preview-rule` |
|
|
94
|
+
| `list_customer_schemas()` | `GET /schema/customer` |
|
|
95
|
+
| `get_schema(schema_id)` | `GET /schema/get/{id}` |
|
|
96
|
+
| `publish_customer_schema(payload)` | `PUT /schema/customer` |
|
|
97
|
+
| `list_desktops()` | `GET /scraper/desktop` |
|
|
98
|
+
|
|
99
|
+
Two helpers are built on top of those routes and add no new ones:
|
|
100
|
+
`iter_group_data()` pages through `get_group_data()`, and `latest_run()`
|
|
101
|
+
returns the newest `get_load_history()` entry.
|
|
102
|
+
|
|
103
|
+
## Errors
|
|
104
|
+
|
|
105
|
+
All failures raise a subclass of `ScrapewiseError`, which carries `status_code`,
|
|
106
|
+
the raw `body`, the request `method`/`url`, and a `payload` property that parses
|
|
107
|
+
the body as JSON when possible.
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
from scrapewise import ScrapewiseAuthError, ScrapewiseError
|
|
111
|
+
|
|
112
|
+
try:
|
|
113
|
+
sw.list_scrapers()
|
|
114
|
+
except ScrapewiseAuthError as exc:
|
|
115
|
+
print("bad key:", exc.status_code)
|
|
116
|
+
except ScrapewiseError as exc:
|
|
117
|
+
print(exc.status_code, exc.payload)
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
`400 → ScrapewiseBadRequestError`, `401/403 → ScrapewiseAuthError`,
|
|
121
|
+
`404 → ScrapewiseNotFoundError`, `409 → ScrapewiseConflictError`,
|
|
122
|
+
`429 → ScrapewiseRateLimitError`, `5xx → ScrapewiseServerError`.
|
|
123
|
+
Network failures raise `ScrapewiseTransportError`, timeouts the
|
|
124
|
+
`ScrapewiseTimeoutError` subclass.
|
|
125
|
+
|
|
126
|
+
## Platform behaviour worth knowing
|
|
127
|
+
|
|
128
|
+
These are properties of the ScrapeWise API, not of this client, and are
|
|
129
|
+
repeated in the relevant docstrings:
|
|
130
|
+
|
|
131
|
+
- `get_sample_data()` returns **at most 100 rows**. It is a sample, not an
|
|
132
|
+
export — use `latest_run()["itemsQuantity"]` for the real row count of a run.
|
|
133
|
+
- Paginated responses key rows on `content`, not `items`.
|
|
134
|
+
- `replace_site_links()` **replaces** the whole link list for a site. Links not
|
|
135
|
+
present in the call are removed.
|
|
136
|
+
- `attach_schema()` is effectively one-way: it switches the scraper's extractor
|
|
137
|
+
to the AI/schema path.
|
|
138
|
+
- Writing group post-process rules is optimistically locked. Read returns
|
|
139
|
+
`version`; the write expects `expectedVersion`.
|
|
140
|
+
`update_group_post_process_rules()` does that rename for you.
|
|
141
|
+
- A scraper is scoped to one domain. Posting links from another host fails with
|
|
142
|
+
`LINK_HOST_MISMATCH`.
|
|
143
|
+
|
|
144
|
+
## Development
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
python -m venv .venv && . .venv/bin/activate
|
|
148
|
+
pip install -e ".[test]"
|
|
149
|
+
pytest
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Tests mock HTTP at the transport layer with `respx`; there are no live network
|
|
153
|
+
calls in the suite.
|
|
154
|
+
|
|
155
|
+
## License
|
|
156
|
+
|
|
157
|
+
MIT © BEBOTECH OÜ
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.21"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "scrapewise"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Python client for the ScrapeWise web-scraping and price-monitoring API."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "BEBOTECH OÜ", email = "hello@scrapewise.ai" }]
|
|
13
|
+
maintainers = [{ name = "BEBOTECH OÜ", email = "hello@scrapewise.ai" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"scrapewise",
|
|
16
|
+
"web-scraping",
|
|
17
|
+
"price-monitoring",
|
|
18
|
+
"competitor-prices",
|
|
19
|
+
"ecommerce",
|
|
20
|
+
"product-data",
|
|
21
|
+
"api-client",
|
|
22
|
+
]
|
|
23
|
+
classifiers = [
|
|
24
|
+
"Development Status :: 4 - Beta",
|
|
25
|
+
"Intended Audience :: Developers",
|
|
26
|
+
"License :: OSI Approved :: MIT License",
|
|
27
|
+
"Operating System :: OS Independent",
|
|
28
|
+
"Programming Language :: Python :: 3",
|
|
29
|
+
"Programming Language :: Python :: 3.9",
|
|
30
|
+
"Programming Language :: Python :: 3.10",
|
|
31
|
+
"Programming Language :: Python :: 3.11",
|
|
32
|
+
"Programming Language :: Python :: 3.12",
|
|
33
|
+
"Programming Language :: Python :: 3.13",
|
|
34
|
+
"Topic :: Internet :: WWW/HTTP",
|
|
35
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
36
|
+
"Typing :: Typed",
|
|
37
|
+
]
|
|
38
|
+
dependencies = ["httpx>=0.24,<1"]
|
|
39
|
+
|
|
40
|
+
[project.optional-dependencies]
|
|
41
|
+
test = ["pytest>=7.4", "respx>=0.20"]
|
|
42
|
+
|
|
43
|
+
[project.urls]
|
|
44
|
+
Homepage = "https://scrapewise.ai"
|
|
45
|
+
Documentation = "https://scrapewise.ai"
|
|
46
|
+
Repository = "https://github.com/BEBOTECH-OU/scrapewise-python"
|
|
47
|
+
Issues = "https://github.com/BEBOTECH-OU/scrapewise-python/issues"
|
|
48
|
+
|
|
49
|
+
[tool.hatch.build.targets.wheel]
|
|
50
|
+
packages = ["src/scrapewise"]
|
|
51
|
+
|
|
52
|
+
[tool.hatch.build.targets.sdist]
|
|
53
|
+
include = ["src/scrapewise", "README.md", "LICENSE", "tests"]
|
|
54
|
+
|
|
55
|
+
[tool.pytest.ini_options]
|
|
56
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""Thin, dependency-light Python client for the ScrapeWise REST API.
|
|
2
|
+
|
|
3
|
+
Quickstart::
|
|
4
|
+
|
|
5
|
+
from scrapewise import ScrapewiseClient
|
|
6
|
+
|
|
7
|
+
with ScrapewiseClient(api_key="YOUR_SCRAPEWISE_API_KEY") as sw:
|
|
8
|
+
for scraper in sw.list_scrapers():
|
|
9
|
+
print(scraper["id"], scraper.get("name"))
|
|
10
|
+
|
|
11
|
+
Every method on :class:`~scrapewise.client.ScrapewiseClient` maps to exactly one
|
|
12
|
+
documented ScrapeWise REST route. No route is synthesised or guessed.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from .client import (
|
|
18
|
+
API_KEY_ENV_VAR,
|
|
19
|
+
BASE_URL_ENV_VAR,
|
|
20
|
+
DEFAULT_BASE_URL,
|
|
21
|
+
DEFAULT_TIMEOUT,
|
|
22
|
+
LOAD_SITE_TIMEOUT,
|
|
23
|
+
ScrapewiseClient,
|
|
24
|
+
)
|
|
25
|
+
from .errors import (
|
|
26
|
+
ScrapewiseAPIError,
|
|
27
|
+
ScrapewiseAuthError,
|
|
28
|
+
ScrapewiseBadRequestError,
|
|
29
|
+
ScrapewiseConfigurationError,
|
|
30
|
+
ScrapewiseConflictError,
|
|
31
|
+
ScrapewiseError,
|
|
32
|
+
ScrapewiseNotFoundError,
|
|
33
|
+
ScrapewiseRateLimitError,
|
|
34
|
+
ScrapewiseServerError,
|
|
35
|
+
ScrapewiseTimeoutError,
|
|
36
|
+
ScrapewiseTransportError,
|
|
37
|
+
)
|
|
38
|
+
from .types import (
|
|
39
|
+
CustomerSchema,
|
|
40
|
+
CustomerSchemaSummary,
|
|
41
|
+
DataRow,
|
|
42
|
+
Desktop,
|
|
43
|
+
GroupPostProcessRules,
|
|
44
|
+
LoadHistoryEntry,
|
|
45
|
+
Page,
|
|
46
|
+
PostProcessRule,
|
|
47
|
+
PreviewRuleResult,
|
|
48
|
+
Scraper,
|
|
49
|
+
ScraperConfig,
|
|
50
|
+
ScraperGroup,
|
|
51
|
+
ScraperSite,
|
|
52
|
+
SiteLink,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
__version__ = "0.1.0"
|
|
56
|
+
|
|
57
|
+
__all__ = [
|
|
58
|
+
"API_KEY_ENV_VAR",
|
|
59
|
+
"BASE_URL_ENV_VAR",
|
|
60
|
+
"DEFAULT_BASE_URL",
|
|
61
|
+
"DEFAULT_TIMEOUT",
|
|
62
|
+
"LOAD_SITE_TIMEOUT",
|
|
63
|
+
"CustomerSchema",
|
|
64
|
+
"CustomerSchemaSummary",
|
|
65
|
+
"DataRow",
|
|
66
|
+
"Desktop",
|
|
67
|
+
"GroupPostProcessRules",
|
|
68
|
+
"LoadHistoryEntry",
|
|
69
|
+
"Page",
|
|
70
|
+
"PostProcessRule",
|
|
71
|
+
"PreviewRuleResult",
|
|
72
|
+
"Scraper",
|
|
73
|
+
"ScraperConfig",
|
|
74
|
+
"ScraperGroup",
|
|
75
|
+
"ScraperSite",
|
|
76
|
+
"ScrapewiseAPIError",
|
|
77
|
+
"ScrapewiseAuthError",
|
|
78
|
+
"ScrapewiseBadRequestError",
|
|
79
|
+
"ScrapewiseClient",
|
|
80
|
+
"ScrapewiseConfigurationError",
|
|
81
|
+
"ScrapewiseConflictError",
|
|
82
|
+
"ScrapewiseError",
|
|
83
|
+
"ScrapewiseNotFoundError",
|
|
84
|
+
"ScrapewiseRateLimitError",
|
|
85
|
+
"ScrapewiseServerError",
|
|
86
|
+
"ScrapewiseTimeoutError",
|
|
87
|
+
"ScrapewiseTransportError",
|
|
88
|
+
"SiteLink",
|
|
89
|
+
"__version__",
|
|
90
|
+
]
|