dataify-sdk 1.0.0__tar.gz → 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/PKG-INFO +19 -7
- dataify_sdk-1.1.0/PYPI_RELEASE.md +136 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/README.md +30 -18
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/pyproject.toml +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/scripts/codegen.py +3 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/__init__.py +16 -6
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/_codegen/generate.py +8 -2
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/client.py +196 -78
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/errors.py +4 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/airbnbproduct.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazoncomment.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazonglobalproduct.py +4 -4
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazonproduct.py +5 -5
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazonproductlist.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazonseller.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingimages.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingmaps.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingnews.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingsearch.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingshopping.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingvideos.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bookinghotellist.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/crunchbasecompany.py +2 -2
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/duckduckgosearch.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/ebayinfo.py +4 -4
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/facebookcomment.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/facebookevent.py +2 -2
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/facebookpost.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/facebookprofile.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/githubrepository.py +3 -3
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/glassdoorcompany.py +4 -4
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/glassdoorjoblistings.py +3 -3
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleaimode.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlefinance.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleflights.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlehotels.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleimages.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlejobs.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlelens.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlelocal.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlemapcomment.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlemapdetails.py +4 -4
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlemaps.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlenews.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlepatents.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleplay.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleplaystoreinformation.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleplaystorereviews.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlescholar.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlesearch.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleshopping.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleshoppinginfo.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googletrends.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlevideos.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/indeedcompaniesinfo.py +4 -4
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/indeedjoblistings.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/instagramcomment.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/instagramprofiles.py +2 -2
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/instagramreel.py +3 -3
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/linkedincompanyinformation.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/linkedinjoblistingsinformation.py +3 -3
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/redditcomment.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/redditposts.py +3 -3
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/tiktokcomment.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/tiktokposts.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/tiktokprofiles.py +2 -2
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/tiktokshop.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/twitterpost.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/twitterprofile.py +2 -2
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/walmartproduct.py +4 -4
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/yandexsearch.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubeaudio.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubecomment.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubeproduct.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubeprofiles.py +2 -2
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubetranscript.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubevideo.py +1 -1
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubevideopost.py +6 -6
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/zillowproduct.py +1 -1
- dataify_sdk-1.1.0/tests/test_client/test_http_client.py +130 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/test_tools/test_tool_modules.py +11 -11
- dataify_sdk-1.0.0/tests/test_client/test_http_client.py +0 -64
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/.gitignore +0 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/LICENSE +0 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/_codegen/__init__.py +0 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/__init__.py +0 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/__init__.py +0 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/conftest.py +0 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/test_client/__init__.py +0 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/test_tools/__init__.py +0 -0
- {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/test_tools/test_codegen.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: dataify-sdk
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.1.0
|
|
4
4
|
Summary: Python SDK that calls the Dataify upstream REST API directly (no MCP layer) — web scrapers and search engines.
|
|
5
5
|
Project-URL: Homepage, https://dashboard.dataify.com
|
|
6
6
|
Project-URL: Repository, https://github.com/dataify/dataify-sdk
|
|
@@ -31,7 +31,7 @@ Python 客户端库,**直接调用** [Dataify](https://dashboard.dataify.com)
|
|
|
31
31
|
## 功能
|
|
32
32
|
|
|
33
33
|
- 🔍 **搜索引擎** — Google(17 种)、Bing(6 种)、Yandex、DuckDuckGo,走 `POST /request`
|
|
34
|
-
- 🛒 **平台抓取器** — Amazon、YouTube、TikTok、Facebook、Instagram、Reddit、Twitter/X、LinkedIn、Glassdoor、Indeed、Walmart、Zillow、Airbnb、Booking、Crunchbase、eBay、GitHub 等 45
|
|
34
|
+
- 🛒 **平台抓取器** — Amazon、YouTube、TikTok、Facebook、Instagram、Reddit、Twitter/X、LinkedIn、Glassdoor、Indeed、Walmart、Zillow、Airbnb、Booking、Crunchbase、eBay、GitHub 等 45 个采集器,先走 `POST /builder?platform=1`,再可查询任务状态并下载结果
|
|
35
35
|
- 📖 **参数全暴露** — 每个工具函数把上游请求参数、类型、是否必填、中文描述都写在签名与 docstring 里;另见 `docs/api_reference.md`
|
|
36
36
|
- 🐍 **零依赖** — 仅用标准库 `urllib`,同步 API
|
|
37
37
|
|
|
@@ -50,17 +50,27 @@ from dataify_sdk import DataifyClient
|
|
|
50
50
|
from dataify_sdk.tools.amazonproduct import amazon_product_by_asin
|
|
51
51
|
from dataify_sdk.tools.googlesearch import google_search
|
|
52
52
|
|
|
53
|
-
# token 也可通过环境变量 DATAIFY_TOKEN
|
|
53
|
+
# token 也可通过环境变量 DATAIFY_API_TOKEN 提供(兼容旧版 DATAIFY_TOKEN)
|
|
54
54
|
client = DataifyClient(token="YOUR_TOKEN")
|
|
55
55
|
|
|
56
|
-
#
|
|
57
|
-
|
|
56
|
+
# 采集类:先提交 Builder 任务
|
|
57
|
+
task = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
|
|
58
|
+
task_id = task["data"]["task_id"]
|
|
59
|
+
|
|
60
|
+
# 再查询任务状态
|
|
61
|
+
status = client.query_scraper_task_status(task_id)
|
|
62
|
+
if status["data"]["status"] == "成功":
|
|
63
|
+
result = client.download_scraper_task_result(task_id, result_type="json")
|
|
58
64
|
|
|
59
65
|
# 搜索类:直接打搜索引擎接口
|
|
60
66
|
result = google_search(q="pizza", client=client)
|
|
61
67
|
```
|
|
62
68
|
|
|
63
|
-
也可以不传 `client`,使用默认 client
|
|
69
|
+
也可以不传 `client`,使用默认 client(优先读取 `DATAIFY_API_TOKEN`,兼容 `DATAIFY_TOKEN`):
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
export DATAIFY_API_TOKEN="..."
|
|
73
|
+
```
|
|
64
74
|
|
|
65
75
|
```python
|
|
66
76
|
from dataify_sdk.tools.googlesearch import google_search
|
|
@@ -68,6 +78,8 @@ from dataify_sdk.tools.googlesearch import google_search
|
|
|
68
78
|
result = google_search(q="pizza")
|
|
69
79
|
```
|
|
70
80
|
|
|
81
|
+
`client.query_scraper_task_status(...)` 会返回任务状态 JSON,`data.status` 常见值为 `处理中`、`成功`、`失败`。任务成功后可用 `client.download_scraper_task_result(task_id, result_type="json")` 下载最终结果;`result_type` 支持 `json`、`csv`、`xlsx`。
|
|
82
|
+
|
|
71
83
|
## API 设计
|
|
72
84
|
|
|
73
85
|
- **每个 `spider_id` = 一个独立函数**(采集类),例如 `amazon_product_by_asin()`、`amazon_product_by_url()`、`amazon_product_by_keywords()`。
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# Dataify Python SDK 更新发布步骤
|
|
2
|
+
|
|
3
|
+
适用于把当前 `dataify-sdk` 新版本发布到 PyPI。
|
|
4
|
+
|
|
5
|
+
## 1. 更新版本号
|
|
6
|
+
|
|
7
|
+
修改两个位置,版本号必须一致:
|
|
8
|
+
|
|
9
|
+
```text
|
|
10
|
+
pyproject.toml
|
|
11
|
+
src/dataify_sdk/__init__.py
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
例如本次新增公开方法,建议改为:
|
|
15
|
+
|
|
16
|
+
```text
|
|
17
|
+
1.1.0
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
## 2. 检查代码状态
|
|
21
|
+
|
|
22
|
+
```powershell
|
|
23
|
+
git status --short
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
确认只包含本次要发布的改动。
|
|
27
|
+
|
|
28
|
+
## 3. 安装开发依赖
|
|
29
|
+
|
|
30
|
+
```powershell
|
|
31
|
+
python -m pip install -e ".[dev]"
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## 4. 运行测试
|
|
35
|
+
|
|
36
|
+
```powershell
|
|
37
|
+
python -m pytest
|
|
38
|
+
python -m compileall src tests
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
测试通过后再继续。
|
|
42
|
+
|
|
43
|
+
## 5. 安装发布工具
|
|
44
|
+
|
|
45
|
+
```powershell
|
|
46
|
+
python -m pip install --upgrade build twine
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## 6. 清理旧构建产物
|
|
50
|
+
|
|
51
|
+
```powershell
|
|
52
|
+
Remove-Item -LiteralPath .\dist -Recurse -Force -ErrorAction SilentlyContinue
|
|
53
|
+
Remove-Item -LiteralPath .\build -Recurse -Force -ErrorAction SilentlyContinue
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## 7. 构建新版本
|
|
57
|
+
|
|
58
|
+
```powershell
|
|
59
|
+
python -m build
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
构建后检查 `dist/` 下是否生成:
|
|
63
|
+
|
|
64
|
+
```text
|
|
65
|
+
*.tar.gz
|
|
66
|
+
*.whl
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
## 8. 检查发布包
|
|
70
|
+
|
|
71
|
+
```powershell
|
|
72
|
+
python -m twine check --strict dist/*
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
必须通过。
|
|
76
|
+
|
|
77
|
+
## 9. 先发到 TestPyPI
|
|
78
|
+
|
|
79
|
+
```powershell
|
|
80
|
+
python -m twine upload --repository testpypi dist/*
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
登录时:
|
|
84
|
+
|
|
85
|
+
```text
|
|
86
|
+
username: __token__
|
|
87
|
+
password: TestPyPI token
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## 10. 从 TestPyPI 安装验证
|
|
91
|
+
|
|
92
|
+
```powershell
|
|
93
|
+
python -m venv .venv-testpypi-check
|
|
94
|
+
.\.venv-testpypi-check\Scripts\python -m pip install --upgrade pip
|
|
95
|
+
.\.venv-testpypi-check\Scripts\python -m pip install --index-url https://test.pypi.org/simple/ --no-deps dataify-sdk==<版本号>
|
|
96
|
+
.\.venv-testpypi-check\Scripts\python -c "import dataify_sdk; print(dataify_sdk.__version__)"
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
确认输出版本号正确。
|
|
100
|
+
|
|
101
|
+
## 11. 发布到正式 PyPI
|
|
102
|
+
|
|
103
|
+
```powershell
|
|
104
|
+
python -m twine upload dist/*
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
登录时:
|
|
108
|
+
|
|
109
|
+
```text
|
|
110
|
+
username: __token__
|
|
111
|
+
password: PyPI token
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
## 12. 从正式 PyPI 安装验证
|
|
115
|
+
|
|
116
|
+
```powershell
|
|
117
|
+
python -m venv .venv-pypi-check
|
|
118
|
+
.\.venv-pypi-check\Scripts\python -m pip install --upgrade pip
|
|
119
|
+
.\.venv-pypi-check\Scripts\python -m pip install --no-cache-dir dataify-sdk==<版本号>
|
|
120
|
+
.\.venv-pypi-check\Scripts\python -c "import dataify_sdk; print(dataify_sdk.__version__)"
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
确认输出版本号正确。
|
|
124
|
+
|
|
125
|
+
## 13. 打 Git Tag
|
|
126
|
+
|
|
127
|
+
```powershell
|
|
128
|
+
git tag v<版本号>
|
|
129
|
+
git push origin v<版本号>
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## 注意
|
|
133
|
+
|
|
134
|
+
- PyPI 已发布的同版本不能覆盖上传。
|
|
135
|
+
- token 不要写进代码、README 或提交记录。
|
|
136
|
+
- 如果发布后发现问题,直接修复后发下一个版本。
|
|
@@ -6,7 +6,7 @@ Python 客户端库,**直接调用** [Dataify](https://dashboard.dataify.com)
|
|
|
6
6
|
## 功能
|
|
7
7
|
|
|
8
8
|
- 🔍 **搜索引擎** — Google(17 种)、Bing(6 种)、Yandex、DuckDuckGo,走 `POST /request`
|
|
9
|
-
- 🛒 **平台抓取器** — Amazon、YouTube、TikTok、Facebook、Instagram、Reddit、Twitter/X、LinkedIn、Glassdoor、Indeed、Walmart、Zillow、Airbnb、Booking、Crunchbase、eBay、GitHub 等 45
|
|
9
|
+
- 🛒 **平台抓取器** — Amazon、YouTube、TikTok、Facebook、Instagram、Reddit、Twitter/X、LinkedIn、Glassdoor、Indeed、Walmart、Zillow、Airbnb、Booking、Crunchbase、eBay、GitHub 等 45 个采集器,先走 `POST /builder?platform=1`,再可查询任务状态并下载结果
|
|
10
10
|
- 📖 **参数全暴露** — 每个工具函数把上游请求参数、类型、是否必填、中文描述都写在签名与 docstring 里;另见 `docs/api_reference.md`
|
|
11
11
|
- 🐍 **零依赖** — 仅用标准库 `urllib`,同步 API
|
|
12
12
|
|
|
@@ -25,23 +25,35 @@ from dataify_sdk import DataifyClient
|
|
|
25
25
|
from dataify_sdk.tools.amazonproduct import amazon_product_by_asin
|
|
26
26
|
from dataify_sdk.tools.googlesearch import google_search
|
|
27
27
|
|
|
28
|
-
# token 也可通过环境变量 DATAIFY_TOKEN
|
|
29
|
-
client = DataifyClient(token="YOUR_TOKEN")
|
|
30
|
-
|
|
31
|
-
#
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
28
|
+
# token 也可通过环境变量 DATAIFY_API_TOKEN 提供(兼容旧版 DATAIFY_TOKEN)
|
|
29
|
+
client = DataifyClient(token="YOUR_TOKEN")
|
|
30
|
+
|
|
31
|
+
# 采集类:先提交 Builder 任务
|
|
32
|
+
task = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
|
|
33
|
+
task_id = task["data"]["task_id"]
|
|
34
|
+
|
|
35
|
+
# 再查询任务状态
|
|
36
|
+
status = client.query_scraper_task_status(task_id)
|
|
37
|
+
if status["data"]["status"] == "成功":
|
|
38
|
+
result = client.download_scraper_task_result(task_id, result_type="json")
|
|
39
|
+
|
|
40
|
+
# 搜索类:直接打搜索引擎接口
|
|
41
|
+
result = google_search(q="pizza", client=client)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
也可以不传 `client`,使用默认 client(优先读取 `DATAIFY_API_TOKEN`,兼容 `DATAIFY_TOKEN`):
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
export DATAIFY_API_TOKEN="..."
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from dataify_sdk.tools.googlesearch import google_search
|
|
52
|
+
|
|
53
|
+
result = google_search(q="pizza")
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
`client.query_scraper_task_status(...)` 会返回任务状态 JSON,`data.status` 常见值为 `处理中`、`成功`、`失败`。任务成功后可用 `client.download_scraper_task_result(task_id, result_type="json")` 下载最终结果;`result_type` 支持 `json`、`csv`、`xlsx`。
|
|
45
57
|
|
|
46
58
|
## API 设计
|
|
47
59
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dataify-sdk"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.1.0"
|
|
8
8
|
description = "Python SDK that calls the Dataify upstream REST API directly (no MCP layer) — web scrapers and search engines."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = {text = "MIT"}
|
|
@@ -7,11 +7,18 @@ Typical usage::
|
|
|
7
7
|
from dataify_sdk.tools.amazonproduct import amazon_product_by_asin
|
|
8
8
|
|
|
9
9
|
client = DataifyClient(token="YOUR_TOKEN")
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
10
|
+
task = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
|
|
11
|
+
task_id = task["data"]["task_id"]
|
|
12
|
+
status = client.query_scraper_task_status(task_id)
|
|
13
|
+
if status["data"]["status"] == "成功":
|
|
14
|
+
result = client.download_scraper_task_result(task_id, result_type="json")
|
|
15
|
+
|
|
16
|
+
or, using the default client (token from ``DATAIFY_API_TOKEN``, with
|
|
17
|
+
``DATAIFY_TOKEN`` still accepted)::
|
|
18
|
+
|
|
19
|
+
task = amazon_product_by_asin(asin="B0BZYCJK89")
|
|
20
|
+
client = DataifyClient()
|
|
21
|
+
status = client.query_scraper_task_status(task["data"]["task_id"])
|
|
15
22
|
"""
|
|
16
23
|
|
|
17
24
|
from __future__ import annotations
|
|
@@ -24,7 +31,10 @@ from dataify_sdk.errors import (
|
|
|
24
31
|
DataifyTimeoutError,
|
|
25
32
|
)
|
|
26
33
|
|
|
27
|
-
__version__ = "1.
|
|
34
|
+
__version__ = "1.1.0"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
28
38
|
|
|
29
39
|
__all__ = [
|
|
30
40
|
"__version__",
|
|
@@ -434,7 +434,10 @@ def render_scraper_function(
|
|
|
434
434
|
]
|
|
435
435
|
for p in sig_params:
|
|
436
436
|
doc.append(_doc_param_line(p, p["upstream"]))
|
|
437
|
-
doc.append(
|
|
437
|
+
doc.append(
|
|
438
|
+
" client: 可选 DataifyClient 实例;不传则使用默认 client("
|
|
439
|
+
"优先读取 DATAIFY_API_TOKEN, 兼容 DATAIFY_TOKEN)。"
|
|
440
|
+
)
|
|
438
441
|
doc.append(' """')
|
|
439
442
|
doc.append(" params_obj = {")
|
|
440
443
|
doc.extend(body_assign)
|
|
@@ -492,7 +495,10 @@ def render_serp_function(
|
|
|
492
495
|
]
|
|
493
496
|
for p in sig_params:
|
|
494
497
|
doc.append(_doc_param_line(p, p["upstream"]))
|
|
495
|
-
doc.append(
|
|
498
|
+
doc.append(
|
|
499
|
+
" client: 可选 DataifyClient 实例;不传则使用默认 client("
|
|
500
|
+
"优先读取 DATAIFY_API_TOKEN, 兼容 DATAIFY_TOKEN)。"
|
|
501
|
+
)
|
|
496
502
|
doc.append(' """')
|
|
497
503
|
doc.append(" form = {")
|
|
498
504
|
doc.extend(body_assign)
|
|
@@ -3,14 +3,19 @@
|
|
|
3
3
|
This client talks **directly** to the Dataify REST endpoints (the same ones the
|
|
4
4
|
Dataify MCP server proxies), instead of going through the MCP protocol:
|
|
5
5
|
|
|
6
|
-
* Scraper / platform tools -> ``POST {base_url}/builder?platform=1``
|
|
7
|
-
(form fields: ``spider_name``, ``spider_id``, ``spider_parameters``,
|
|
8
|
-
``spider_errors``, ``file_name``)
|
|
9
|
-
*
|
|
10
|
-
(
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
6
|
+
* Scraper / platform tools -> ``POST {base_url}/builder?platform=1``
|
|
7
|
+
(form fields: ``spider_name``, ``spider_id``, ``spider_parameters``,
|
|
8
|
+
``spider_errors``, ``file_name``)
|
|
9
|
+
* Scraper task status query -> ``GET {base_url}/task_status``
|
|
10
|
+
(query params: ``api_key``, ``task_id``)
|
|
11
|
+
* Scraper task result download -> ``GET {base_url}/download``
|
|
12
|
+
(query params: ``api_key``, ``task_id``, ``type``)
|
|
13
|
+
* Search engines (Google / Bing / Yandex / DuckDuckGo) -> ``POST {base_url}/request``
|
|
14
|
+
(form fields are the engine-specific parameters plus a fixed ``engine`` value)
|
|
15
|
+
|
|
16
|
+
Authentication is via a Bearer token, supplied either to the constructor or
|
|
17
|
+
through the ``DATAIFY_API_TOKEN`` environment variable. The legacy
|
|
18
|
+
``DATAIFY_TOKEN`` name is still accepted for backward compatibility.
|
|
14
19
|
"""
|
|
15
20
|
|
|
16
21
|
from __future__ import annotations
|
|
@@ -28,23 +33,53 @@ from dataify_sdk.errors import (
|
|
|
28
33
|
DataifyTimeoutError,
|
|
29
34
|
)
|
|
30
35
|
|
|
31
|
-
DEFAULT_BASE_URL = "https://scraperapi.dataify.com"
|
|
32
|
-
DEFAULT_TIMEOUT = 120
|
|
33
|
-
ENV_TOKEN = "
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
36
|
+
DEFAULT_BASE_URL = "https://scraperapi.dataify.com"
|
|
37
|
+
DEFAULT_TIMEOUT = 120
|
|
38
|
+
ENV_TOKEN = "DATAIFY_API_TOKEN"
|
|
39
|
+
LEGACY_ENV_TOKEN = "DATAIFY_TOKEN"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _read_env_token() -> str | None:
|
|
43
|
+
for env_name in (ENV_TOKEN, LEGACY_ENV_TOKEN):
|
|
44
|
+
token = os.environ.get(env_name)
|
|
45
|
+
if token:
|
|
46
|
+
return token
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _redact_url(url: str) -> str:
|
|
51
|
+
try:
|
|
52
|
+
parsed = urllib.parse.urlsplit(url)
|
|
53
|
+
query = [
|
|
54
|
+
(key, "***" if key == "api_key" else value)
|
|
55
|
+
for key, value in urllib.parse.parse_qsl(
|
|
56
|
+
parsed.query, keep_blank_values=True
|
|
57
|
+
)
|
|
58
|
+
]
|
|
59
|
+
return urllib.parse.urlunsplit(
|
|
60
|
+
parsed._replace(query=urllib.parse.urlencode(query))
|
|
61
|
+
)
|
|
62
|
+
except ValueError:
|
|
63
|
+
return url
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
# Endpoints on the upstream.
|
|
67
|
+
_SCRAPER_PATH = "/builder?platform=1"
|
|
68
|
+
_SCRAPER_TASK_STATUS_PATH = "/task_status"
|
|
69
|
+
_SCRAPER_TASK_RESULT_PATH = "/download"
|
|
70
|
+
_SCRAPER_TASK_RESULT_TYPES = {"json", "csv", "xlsx"}
|
|
71
|
+
_SERP_PATH = "/request"
|
|
38
72
|
|
|
39
73
|
|
|
40
74
|
class DataifyClient:
|
|
41
75
|
"""Synchronous client for the Dataify upstream REST API.
|
|
42
76
|
|
|
43
|
-
Parameters
|
|
44
|
-
----------
|
|
45
|
-
token:
|
|
46
|
-
Dataify API token. If omitted, the ``
|
|
47
|
-
variable is used
|
|
77
|
+
Parameters
|
|
78
|
+
----------
|
|
79
|
+
token:
|
|
80
|
+
Dataify API token. If omitted, the ``DATAIFY_API_TOKEN`` environment
|
|
81
|
+
variable is used first, with ``DATAIFY_TOKEN`` kept as a legacy
|
|
82
|
+
fallback. A token is required to make any request.
|
|
48
83
|
base_url:
|
|
49
84
|
Upstream base URL. Fixed to the Dataify production endpoint by default;
|
|
50
85
|
exposed only for testing / private deployments.
|
|
@@ -52,56 +87,84 @@ class DataifyClient:
|
|
|
52
87
|
Per-request timeout in seconds.
|
|
53
88
|
"""
|
|
54
89
|
|
|
55
|
-
def __init__(
|
|
56
|
-
self,
|
|
57
|
-
token: str | None = None,
|
|
58
|
-
base_url: str = DEFAULT_BASE_URL,
|
|
59
|
-
timeout: int = DEFAULT_TIMEOUT,
|
|
60
|
-
) -> None:
|
|
61
|
-
self.token = token or
|
|
62
|
-
if not self.token:
|
|
63
|
-
raise ValueError(
|
|
64
|
-
"A Dataify token is required: pass token=... or set the "
|
|
65
|
-
f"{ENV_TOKEN} environment variable."
|
|
66
|
-
)
|
|
67
|
-
self.base_url = base_url.rstrip("/")
|
|
68
|
-
self.timeout = timeout
|
|
90
|
+
def __init__(
|
|
91
|
+
self,
|
|
92
|
+
token: str | None = None,
|
|
93
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
94
|
+
timeout: int = DEFAULT_TIMEOUT,
|
|
95
|
+
) -> None:
|
|
96
|
+
self.token = token or _read_env_token()
|
|
97
|
+
if not self.token:
|
|
98
|
+
raise ValueError(
|
|
99
|
+
"A Dataify token is required: pass token=... or set the "
|
|
100
|
+
f"{ENV_TOKEN} environment variable (legacy: {LEGACY_ENV_TOKEN})."
|
|
101
|
+
)
|
|
102
|
+
self.base_url = base_url.rstrip("/")
|
|
103
|
+
self.timeout = timeout
|
|
69
104
|
|
|
70
|
-
# ------------------------------------------------------------------
|
|
71
|
-
# Low-level transport
|
|
72
|
-
# ------------------------------------------------------------------
|
|
73
|
-
def
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
105
|
+
# ------------------------------------------------------------------
|
|
106
|
+
# Low-level transport
|
|
107
|
+
# ------------------------------------------------------------------
|
|
108
|
+
def _read_response_bytes(self, req: urllib.request.Request, url: str) -> bytes:
|
|
109
|
+
display_url = _redact_url(url)
|
|
110
|
+
try:
|
|
111
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as resp:
|
|
112
|
+
return resp.read()
|
|
113
|
+
except urllib.error.HTTPError as exc: # upstream returned >= 400
|
|
114
|
+
body = exc.read().decode("utf-8", errors="replace")
|
|
115
|
+
raise DataifyAPIError(
|
|
116
|
+
f"Dataify API request failed: {display_url}", exc.code, body
|
|
117
|
+
) from exc
|
|
118
|
+
except urllib.error.URLError as exc:
|
|
119
|
+
if isinstance(exc.reason, TimeoutError):
|
|
120
|
+
raise DataifyTimeoutError(
|
|
121
|
+
f"Request to {display_url} timed out after {self.timeout}s"
|
|
122
|
+
) from exc
|
|
123
|
+
raise DataifyConnectionError(
|
|
124
|
+
f"Could not connect to Dataify API at {display_url}: {exc.reason}"
|
|
125
|
+
) from exc
|
|
126
|
+
|
|
127
|
+
def _read_json_response(
|
|
128
|
+
self, req: urllib.request.Request, url: str
|
|
129
|
+
) -> Any:
|
|
130
|
+
raw = self._read_response_bytes(req, url).decode(
|
|
131
|
+
"utf-8", errors="replace"
|
|
132
|
+
)
|
|
133
|
+
try:
|
|
134
|
+
return json.loads(raw)
|
|
135
|
+
except json.JSONDecodeError:
|
|
136
|
+
return {"raw": raw}
|
|
137
|
+
|
|
138
|
+
def _post_form(self, path: str, fields: dict[str, str]) -> dict[str, Any]:
|
|
139
|
+
"""POST form-encoded fields and return the parsed JSON response."""
|
|
140
|
+
url = self.base_url + path
|
|
141
|
+
# Only send non-empty values (mirrors the Go upstream behaviour).
|
|
142
|
+
payload = {k: v for k, v in fields.items() if v not in (None, "")}
|
|
78
143
|
data = urllib.parse.urlencode(payload, doseq=False).encode("utf-8")
|
|
79
144
|
|
|
80
|
-
req = urllib.request.Request(url, data=data, method="POST")
|
|
81
|
-
req.add_header("Content-Type", "application/x-www-form-urlencoded")
|
|
82
|
-
req.add_header("Authorization", f"Bearer {self.token}")
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
except json.JSONDecodeError:
|
|
104
|
-
return {"raw": raw}
|
|
145
|
+
req = urllib.request.Request(url, data=data, method="POST")
|
|
146
|
+
req.add_header("Content-Type", "application/x-www-form-urlencoded")
|
|
147
|
+
req.add_header("Authorization", f"Bearer {self.token}")
|
|
148
|
+
return self._read_json_response(req, url)
|
|
149
|
+
|
|
150
|
+
def _get_json(self, path: str, query: dict[str, str]) -> Any:
|
|
151
|
+
"""GET a JSON endpoint with query parameters and return parsed JSON."""
|
|
152
|
+
url, req = self._build_get_request(path, query)
|
|
153
|
+
return self._read_json_response(req, url)
|
|
154
|
+
|
|
155
|
+
def _get_bytes(self, path: str, query: dict[str, str]) -> bytes:
|
|
156
|
+
"""GET an endpoint with query parameters and return response bytes."""
|
|
157
|
+
url, req = self._build_get_request(path, query)
|
|
158
|
+
return self._read_response_bytes(req, url)
|
|
159
|
+
|
|
160
|
+
def _build_get_request(
|
|
161
|
+
self, path: str, query: dict[str, str]
|
|
162
|
+
) -> tuple[str, urllib.request.Request]:
|
|
163
|
+
url = self.base_url + path
|
|
164
|
+
payload = {k: v for k, v in query.items() if v not in (None, "")}
|
|
165
|
+
if payload:
|
|
166
|
+
url = url + "?" + urllib.parse.urlencode(payload, doseq=False)
|
|
167
|
+
return url, urllib.request.Request(url, method="GET")
|
|
105
168
|
|
|
106
169
|
# ------------------------------------------------------------------
|
|
107
170
|
# High-level helpers used by the generated tool functions
|
|
@@ -146,8 +209,8 @@ class DataifyClient:
|
|
|
146
209
|
fields["spider_universal"] = spider_universal
|
|
147
210
|
return self._post_form(_SCRAPER_PATH, fields)
|
|
148
211
|
|
|
149
|
-
def request_serp(self, engine: str, fields: dict[str, str]) -> dict[str, Any]:
|
|
150
|
-
"""Call a search-engine endpoint.
|
|
212
|
+
def request_serp(self, engine: str, fields: dict[str, str]) -> dict[str, Any]:
|
|
213
|
+
"""Call a search-engine endpoint.
|
|
151
214
|
|
|
152
215
|
Parameters
|
|
153
216
|
----------
|
|
@@ -158,20 +221,75 @@ class DataifyClient:
|
|
|
158
221
|
field names / JSON tags). The ``engine`` field is added
|
|
159
222
|
automatically.
|
|
160
223
|
"""
|
|
161
|
-
form = {"engine": engine}
|
|
162
|
-
form.update(fields)
|
|
163
|
-
return self._post_form(_SERP_PATH, form)
|
|
224
|
+
form = {"engine": engine}
|
|
225
|
+
form.update(fields)
|
|
226
|
+
return self._post_form(_SERP_PATH, form)
|
|
227
|
+
|
|
228
|
+
def query_scraper_task_status(self, task_id: str) -> dict[str, Any]:
|
|
229
|
+
"""Query a scraper task status by task ID.
|
|
230
|
+
|
|
231
|
+
Parameters
|
|
232
|
+
----------
|
|
233
|
+
task_id:
|
|
234
|
+
Task ID returned by ``request_scraper``.
|
|
235
|
+
"""
|
|
236
|
+
task_id = task_id.strip()
|
|
237
|
+
if not task_id:
|
|
238
|
+
raise ValueError("task_id is required")
|
|
239
|
+
return self._get_json(
|
|
240
|
+
_SCRAPER_TASK_STATUS_PATH,
|
|
241
|
+
{"api_key": self.token, "task_id": task_id},
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
def download_scraper_task_result(
|
|
245
|
+
self, task_id: str, result_type: str = "json"
|
|
246
|
+
) -> Any:
|
|
247
|
+
"""Download a scraper task result.
|
|
248
|
+
|
|
249
|
+
Parameters
|
|
250
|
+
----------
|
|
251
|
+
task_id:
|
|
252
|
+
Task ID returned by ``request_scraper``.
|
|
253
|
+
result_type:
|
|
254
|
+
Result format. Supported values are ``"json"``, ``"csv"``, and
|
|
255
|
+
``"xlsx"``. JSON results are parsed, CSV results are
|
|
256
|
+
returned as text, and XLSX results are returned as bytes.
|
|
257
|
+
"""
|
|
258
|
+
task_id = task_id.strip()
|
|
259
|
+
if not task_id:
|
|
260
|
+
raise ValueError("task_id is required")
|
|
261
|
+
|
|
262
|
+
result_type = result_type.strip().lower()
|
|
263
|
+
if result_type not in _SCRAPER_TASK_RESULT_TYPES:
|
|
264
|
+
raise ValueError(
|
|
265
|
+
"result_type must be one of: "
|
|
266
|
+
+ ", ".join(sorted(_SCRAPER_TASK_RESULT_TYPES))
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
fields = {
|
|
270
|
+
"api_key": self.token,
|
|
271
|
+
"task_id": task_id,
|
|
272
|
+
"type": result_type,
|
|
273
|
+
}
|
|
274
|
+
if result_type == "json":
|
|
275
|
+
return self._get_json(_SCRAPER_TASK_RESULT_PATH, fields)
|
|
276
|
+
|
|
277
|
+
content = self._get_bytes(_SCRAPER_TASK_RESULT_PATH, fields)
|
|
278
|
+
if result_type == "csv":
|
|
279
|
+
return content.decode("utf-8", errors="replace")
|
|
280
|
+
return content
|
|
164
281
|
|
|
165
282
|
|
|
166
283
|
_DEFAULT_CLIENT: DataifyClient | None = None
|
|
167
284
|
|
|
168
285
|
|
|
169
|
-
def get_default_client() -> DataifyClient:
|
|
170
|
-
"""Return a process-wide default :class:`DataifyClient`.
|
|
171
|
-
|
|
172
|
-
Uses ``
|
|
173
|
-
|
|
174
|
-
|
|
286
|
+
def get_default_client() -> DataifyClient:
|
|
287
|
+
"""Return a process-wide default :class:`DataifyClient`.
|
|
288
|
+
|
|
289
|
+
Uses ``DATAIFY_API_TOKEN`` from the environment, with
|
|
290
|
+
``DATAIFY_TOKEN`` kept as a legacy fallback. Handy for the generated
|
|
291
|
+
tool functions, which accept an optional ``client`` argument.
|
|
292
|
+
"""
|
|
175
293
|
global _DEFAULT_CLIENT
|
|
176
294
|
if _DEFAULT_CLIENT is None:
|
|
177
295
|
_DEFAULT_CLIENT = DataifyClient()
|