dataify-sdk 1.0.0__tar.gz → 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/PKG-INFO +19 -7
  2. dataify_sdk-1.1.0/PYPI_RELEASE.md +136 -0
  3. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/README.md +30 -18
  4. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/pyproject.toml +1 -1
  5. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/scripts/codegen.py +3 -0
  6. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/__init__.py +16 -6
  7. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/_codegen/generate.py +8 -2
  8. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/client.py +196 -78
  9. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/errors.py +4 -0
  10. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/airbnbproduct.py +1 -1
  11. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazoncomment.py +1 -1
  12. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazonglobalproduct.py +4 -4
  13. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazonproduct.py +5 -5
  14. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazonproductlist.py +1 -1
  15. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/amazonseller.py +1 -1
  16. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingimages.py +1 -1
  17. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingmaps.py +1 -1
  18. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingnews.py +1 -1
  19. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingsearch.py +1 -1
  20. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingshopping.py +1 -1
  21. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bingvideos.py +1 -1
  22. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/bookinghotellist.py +1 -1
  23. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/crunchbasecompany.py +2 -2
  24. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/duckduckgosearch.py +1 -1
  25. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/ebayinfo.py +4 -4
  26. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/facebookcomment.py +1 -1
  27. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/facebookevent.py +2 -2
  28. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/facebookpost.py +1 -1
  29. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/facebookprofile.py +1 -1
  30. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/githubrepository.py +3 -3
  31. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/glassdoorcompany.py +4 -4
  32. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/glassdoorjoblistings.py +3 -3
  33. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleaimode.py +1 -1
  34. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlefinance.py +1 -1
  35. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleflights.py +1 -1
  36. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlehotels.py +1 -1
  37. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleimages.py +1 -1
  38. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlejobs.py +1 -1
  39. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlelens.py +1 -1
  40. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlelocal.py +1 -1
  41. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlemapcomment.py +1 -1
  42. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlemapdetails.py +4 -4
  43. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlemaps.py +1 -1
  44. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlenews.py +1 -1
  45. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlepatents.py +1 -1
  46. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleplay.py +1 -1
  47. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleplaystoreinformation.py +1 -1
  48. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleplaystorereviews.py +1 -1
  49. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlescholar.py +1 -1
  50. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlesearch.py +1 -1
  51. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleshopping.py +1 -1
  52. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googleshoppinginfo.py +1 -1
  53. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googletrends.py +1 -1
  54. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/googlevideos.py +1 -1
  55. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/indeedcompaniesinfo.py +4 -4
  56. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/indeedjoblistings.py +1 -1
  57. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/instagramcomment.py +1 -1
  58. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/instagramprofiles.py +2 -2
  59. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/instagramreel.py +3 -3
  60. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/linkedincompanyinformation.py +1 -1
  61. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/linkedinjoblistingsinformation.py +3 -3
  62. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/redditcomment.py +1 -1
  63. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/redditposts.py +3 -3
  64. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/tiktokcomment.py +1 -1
  65. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/tiktokposts.py +1 -1
  66. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/tiktokprofiles.py +2 -2
  67. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/tiktokshop.py +1 -1
  68. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/twitterpost.py +1 -1
  69. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/twitterprofile.py +2 -2
  70. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/walmartproduct.py +4 -4
  71. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/yandexsearch.py +1 -1
  72. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubeaudio.py +1 -1
  73. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubecomment.py +1 -1
  74. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubeproduct.py +1 -1
  75. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubeprofiles.py +2 -2
  76. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubetranscript.py +1 -1
  77. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubevideo.py +1 -1
  78. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/youtubevideopost.py +6 -6
  79. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/zillowproduct.py +1 -1
  80. dataify_sdk-1.1.0/tests/test_client/test_http_client.py +130 -0
  81. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/test_tools/test_tool_modules.py +11 -11
  82. dataify_sdk-1.0.0/tests/test_client/test_http_client.py +0 -64
  83. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/.gitignore +0 -0
  84. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/LICENSE +0 -0
  85. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/_codegen/__init__.py +0 -0
  86. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/src/dataify_sdk/tools/__init__.py +0 -0
  87. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/__init__.py +0 -0
  88. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/conftest.py +0 -0
  89. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/test_client/__init__.py +0 -0
  90. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/test_tools/__init__.py +0 -0
  91. {dataify_sdk-1.0.0 → dataify_sdk-1.1.0}/tests/test_tools/test_codegen.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: dataify-sdk
3
- Version: 1.0.0
3
+ Version: 1.1.0
4
4
  Summary: Python SDK that calls the Dataify upstream REST API directly (no MCP layer) — web scrapers and search engines.
5
5
  Project-URL: Homepage, https://dashboard.dataify.com
6
6
  Project-URL: Repository, https://github.com/dataify/dataify-sdk
@@ -31,7 +31,7 @@ Python 客户端库,**直接调用** [Dataify](https://dashboard.dataify.com)
31
31
  ## 功能
32
32
 
33
33
  - 🔍 **搜索引擎** — Google(17 种)、Bing(6 种)、Yandex、DuckDuckGo,走 `POST /request`
34
- - 🛒 **平台抓取器** — Amazon、YouTube、TikTok、Facebook、Instagram、Reddit、Twitter/X、LinkedIn、Glassdoor、Indeed、Walmart、Zillow、Airbnb、Booking、Crunchbase、eBay、GitHub 等 45 个采集器,走 `POST /builder?platform=1`
34
+ - 🛒 **平台抓取器** — Amazon、YouTube、TikTok、Facebook、Instagram、Reddit、Twitter/X、LinkedIn、Glassdoor、Indeed、Walmart、Zillow、Airbnb、Booking、Crunchbase、eBay、GitHub 等 45 个采集器,先走 `POST /builder?platform=1`,再可查询任务状态并下载结果
35
35
  - 📖 **参数全暴露** — 每个工具函数把上游请求参数、类型、是否必填、中文描述都写在签名与 docstring 里;另见 `docs/api_reference.md`
36
36
  - 🐍 **零依赖** — 仅用标准库 `urllib`,同步 API
37
37
 
@@ -50,17 +50,27 @@ from dataify_sdk import DataifyClient
50
50
  from dataify_sdk.tools.amazonproduct import amazon_product_by_asin
51
51
  from dataify_sdk.tools.googlesearch import google_search
52
52
 
53
- # token 也可通过环境变量 DATAIFY_TOKEN 提供
53
+ # token 也可通过环境变量 DATAIFY_API_TOKEN 提供(兼容旧版 DATAIFY_TOKEN)
54
54
  client = DataifyClient(token="YOUR_TOKEN")
55
55
 
56
- # 采集类:直接提交 Builder 任务
57
- result = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
56
+ # 采集类:先提交 Builder 任务
57
+ task = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
58
+ task_id = task["data"]["task_id"]
59
+
60
+ # 再查询任务状态
61
+ status = client.query_scraper_task_status(task_id)
62
+ if status["data"]["status"] == "成功":
63
+ result = client.download_scraper_task_result(task_id, result_type="json")
58
64
 
59
65
  # 搜索类:直接打搜索引擎接口
60
66
  result = google_search(q="pizza", client=client)
61
67
  ```
62
68
 
63
- 也可以不传 `client`,使用默认 client(读取 `DATAIFY_TOKEN` 环境变量):
69
+ 也可以不传 `client`,使用默认 client(优先读取 `DATAIFY_API_TOKEN`,兼容 `DATAIFY_TOKEN`):
70
+
71
+ ```bash
72
+ export DATAIFY_API_TOKEN="..."
73
+ ```
64
74
 
65
75
  ```python
66
76
  from dataify_sdk.tools.googlesearch import google_search
@@ -68,6 +78,8 @@ from dataify_sdk.tools.googlesearch import google_search
68
78
  result = google_search(q="pizza")
69
79
  ```
70
80
 
81
+ `client.query_scraper_task_status(...)` 会返回任务状态 JSON,`data.status` 常见值为 `处理中`、`成功`、`失败`。任务成功后可用 `client.download_scraper_task_result(task_id, result_type="json")` 下载最终结果;`result_type` 支持 `json`、`csv`、`xlsx`。
82
+
71
83
  ## API 设计
72
84
 
73
85
  - **每个 `spider_id` = 一个独立函数**(采集类),例如 `amazon_product_by_asin()`、`amazon_product_by_url()`、`amazon_product_by_keywords()`。
@@ -0,0 +1,136 @@
1
+ # Dataify Python SDK 更新发布步骤
2
+
3
+ 适用于把当前 `dataify-sdk` 新版本发布到 PyPI。
4
+
5
+ ## 1. 更新版本号
6
+
7
+ 修改两个位置,版本号必须一致:
8
+
9
+ ```text
10
+ pyproject.toml
11
+ src/dataify_sdk/__init__.py
12
+ ```
13
+
14
+ 例如本次新增公开方法,建议改为:
15
+
16
+ ```text
17
+ 1.1.0
18
+ ```
19
+
20
+ ## 2. 检查代码状态
21
+
22
+ ```powershell
23
+ git status --short
24
+ ```
25
+
26
+ 确认只包含本次要发布的改动。
27
+
28
+ ## 3. 安装开发依赖
29
+
30
+ ```powershell
31
+ python -m pip install -e ".[dev]"
32
+ ```
33
+
34
+ ## 4. 运行测试
35
+
36
+ ```powershell
37
+ python -m pytest
38
+ python -m compileall src tests
39
+ ```
40
+
41
+ 测试通过后再继续。
42
+
43
+ ## 5. 安装发布工具
44
+
45
+ ```powershell
46
+ python -m pip install --upgrade build twine
47
+ ```
48
+
49
+ ## 6. 清理旧构建产物
50
+
51
+ ```powershell
52
+ Remove-Item -LiteralPath .\dist -Recurse -Force -ErrorAction SilentlyContinue
53
+ Remove-Item -LiteralPath .\build -Recurse -Force -ErrorAction SilentlyContinue
54
+ ```
55
+
56
+ ## 7. 构建新版本
57
+
58
+ ```powershell
59
+ python -m build
60
+ ```
61
+
62
+ 构建后检查 `dist/` 下是否生成:
63
+
64
+ ```text
65
+ *.tar.gz
66
+ *.whl
67
+ ```
68
+
69
+ ## 8. 检查发布包
70
+
71
+ ```powershell
72
+ python -m twine check --strict dist/*
73
+ ```
74
+
75
+ 必须通过。
76
+
77
+ ## 9. 先发到 TestPyPI
78
+
79
+ ```powershell
80
+ python -m twine upload --repository testpypi dist/*
81
+ ```
82
+
83
+ 登录时:
84
+
85
+ ```text
86
+ username: __token__
87
+ password: TestPyPI token
88
+ ```
89
+
90
+ ## 10. 从 TestPyPI 安装验证
91
+
92
+ ```powershell
93
+ python -m venv .venv-testpypi-check
94
+ .\.venv-testpypi-check\Scripts\python -m pip install --upgrade pip
95
+ .\.venv-testpypi-check\Scripts\python -m pip install --index-url https://test.pypi.org/simple/ --no-deps dataify-sdk==<版本号>
96
+ .\.venv-testpypi-check\Scripts\python -c "import dataify_sdk; print(dataify_sdk.__version__)"
97
+ ```
98
+
99
+ 确认输出版本号正确。
100
+
101
+ ## 11. 发布到正式 PyPI
102
+
103
+ ```powershell
104
+ python -m twine upload dist/*
105
+ ```
106
+
107
+ 登录时:
108
+
109
+ ```text
110
+ username: __token__
111
+ password: PyPI token
112
+ ```
113
+
114
+ ## 12. 从正式 PyPI 安装验证
115
+
116
+ ```powershell
117
+ python -m venv .venv-pypi-check
118
+ .\.venv-pypi-check\Scripts\python -m pip install --upgrade pip
119
+ .\.venv-pypi-check\Scripts\python -m pip install --no-cache-dir dataify-sdk==<版本号>
120
+ .\.venv-pypi-check\Scripts\python -c "import dataify_sdk; print(dataify_sdk.__version__)"
121
+ ```
122
+
123
+ 确认输出版本号正确。
124
+
125
+ ## 13. 打 Git Tag
126
+
127
+ ```powershell
128
+ git tag v<版本号>
129
+ git push origin v<版本号>
130
+ ```
131
+
132
+ ## 注意
133
+
134
+ - PyPI 已发布的同版本不能覆盖上传。
135
+ - token 不要写进代码、README 或提交记录。
136
+ - 如果发布后发现问题,直接修复后发下一个版本。
@@ -6,7 +6,7 @@ Python 客户端库,**直接调用** [Dataify](https://dashboard.dataify.com)
6
6
  ## 功能
7
7
 
8
8
  - 🔍 **搜索引擎** — Google(17 种)、Bing(6 种)、Yandex、DuckDuckGo,走 `POST /request`
9
- - 🛒 **平台抓取器** — Amazon、YouTube、TikTok、Facebook、Instagram、Reddit、Twitter/X、LinkedIn、Glassdoor、Indeed、Walmart、Zillow、Airbnb、Booking、Crunchbase、eBay、GitHub 等 45 个采集器,走 `POST /builder?platform=1`
9
+ - 🛒 **平台抓取器** — Amazon、YouTube、TikTok、Facebook、Instagram、Reddit、Twitter/X、LinkedIn、Glassdoor、Indeed、Walmart、Zillow、Airbnb、Booking、Crunchbase、eBay、GitHub 等 45 个采集器,先走 `POST /builder?platform=1`,再可查询任务状态并下载结果
10
10
  - 📖 **参数全暴露** — 每个工具函数把上游请求参数、类型、是否必填、中文描述都写在签名与 docstring 里;另见 `docs/api_reference.md`
11
11
  - 🐍 **零依赖** — 仅用标准库 `urllib`,同步 API
12
12
 
@@ -25,23 +25,35 @@ from dataify_sdk import DataifyClient
25
25
  from dataify_sdk.tools.amazonproduct import amazon_product_by_asin
26
26
  from dataify_sdk.tools.googlesearch import google_search
27
27
 
28
- # token 也可通过环境变量 DATAIFY_TOKEN 提供
29
- client = DataifyClient(token="YOUR_TOKEN")
30
-
31
- # 采集类:直接提交 Builder 任务
32
- result = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
33
-
34
- # 搜索类:直接打搜索引擎接口
35
- result = google_search(q="pizza", client=client)
36
- ```
37
-
38
- 也可以不传 `client`,使用默认 client(读取 `DATAIFY_TOKEN` 环境变量):
39
-
40
- ```python
41
- from dataify_sdk.tools.googlesearch import google_search
42
-
43
- result = google_search(q="pizza")
44
- ```
28
+ # token 也可通过环境变量 DATAIFY_API_TOKEN 提供(兼容旧版 DATAIFY_TOKEN)
29
+ client = DataifyClient(token="YOUR_TOKEN")
30
+
31
+ # 采集类:先提交 Builder 任务
32
+ task = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
33
+ task_id = task["data"]["task_id"]
34
+
35
+ # 再查询任务状态
36
+ status = client.query_scraper_task_status(task_id)
37
+ if status["data"]["status"] == "成功":
38
+ result = client.download_scraper_task_result(task_id, result_type="json")
39
+
40
+ # 搜索类:直接打搜索引擎接口
41
+ result = google_search(q="pizza", client=client)
42
+ ```
43
+
44
+ 也可以不传 `client`,使用默认 client(优先读取 `DATAIFY_API_TOKEN`,兼容 `DATAIFY_TOKEN`):
45
+
46
+ ```bash
47
+ export DATAIFY_API_TOKEN="..."
48
+ ```
49
+
50
+ ```python
51
+ from dataify_sdk.tools.googlesearch import google_search
52
+
53
+ result = google_search(q="pizza")
54
+ ```
55
+
56
+ `client.query_scraper_task_status(...)` 会返回任务状态 JSON,`data.status` 常见值为 `处理中`、`成功`、`失败`。任务成功后可用 `client.download_scraper_task_result(task_id, result_type="json")` 下载最终结果;`result_type` 支持 `json`、`csv`、`xlsx`。
45
57
 
46
58
  ## API 设计
47
59
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "dataify-sdk"
7
- version = "1.0.0"
7
+ version = "1.1.0"
8
8
  description = "Python SDK that calls the Dataify upstream REST API directly (no MCP layer) — web scrapers and search engines."
9
9
  readme = "README.md"
10
10
  license = {text = "MIT"}
@@ -21,5 +21,8 @@ sys.path.insert(0, str(_ROOT / "src"))
21
21
  from dataify_sdk._codegen.generate import main
22
22
 
23
23
 
24
+
25
+
26
+
24
27
  if __name__ == "__main__":
25
28
  raise SystemExit(main())
@@ -7,11 +7,18 @@ Typical usage::
7
7
  from dataify_sdk.tools.amazonproduct import amazon_product_by_asin
8
8
 
9
9
  client = DataifyClient(token="YOUR_TOKEN")
10
- result = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
11
-
12
- or, using the default client (token from ``DATAIFY_TOKEN``)::
13
-
14
- result = amazon_product_by_asin(asin="B0BZYCJK89")
10
+ task = amazon_product_by_asin(asin="B0BZYCJK89", client=client)
11
+ task_id = task["data"]["task_id"]
12
+ status = client.query_scraper_task_status(task_id)
13
+ if status["data"]["status"] == "成功":
14
+ result = client.download_scraper_task_result(task_id, result_type="json")
15
+
16
+ or, using the default client (token from ``DATAIFY_API_TOKEN``, with
17
+ ``DATAIFY_TOKEN`` still accepted)::
18
+
19
+ task = amazon_product_by_asin(asin="B0BZYCJK89")
20
+ client = DataifyClient()
21
+ status = client.query_scraper_task_status(task["data"]["task_id"])
15
22
  """
16
23
 
17
24
  from __future__ import annotations
@@ -24,7 +31,10 @@ from dataify_sdk.errors import (
24
31
  DataifyTimeoutError,
25
32
  )
26
33
 
27
- __version__ = "1.0.0"
34
+ __version__ = "1.1.0"
35
+
36
+
37
+
28
38
 
29
39
  __all__ = [
30
40
  "__version__",
@@ -434,7 +434,10 @@ def render_scraper_function(
434
434
  ]
435
435
  for p in sig_params:
436
436
  doc.append(_doc_param_line(p, p["upstream"]))
437
- doc.append(" client: 可选 DataifyClient 实例;不传则使用默认 client(读取 DATAIFY_TOKEN)。")
437
+ doc.append(
438
+ " client: 可选 DataifyClient 实例;不传则使用默认 client("
439
+ "优先读取 DATAIFY_API_TOKEN, 兼容 DATAIFY_TOKEN)。"
440
+ )
438
441
  doc.append(' """')
439
442
  doc.append(" params_obj = {")
440
443
  doc.extend(body_assign)
@@ -492,7 +495,10 @@ def render_serp_function(
492
495
  ]
493
496
  for p in sig_params:
494
497
  doc.append(_doc_param_line(p, p["upstream"]))
495
- doc.append(" client: 可选 DataifyClient 实例;不传则使用默认 client(读取 DATAIFY_TOKEN)。")
498
+ doc.append(
499
+ " client: 可选 DataifyClient 实例;不传则使用默认 client("
500
+ "优先读取 DATAIFY_API_TOKEN, 兼容 DATAIFY_TOKEN)。"
501
+ )
496
502
  doc.append(' """')
497
503
  doc.append(" form = {")
498
504
  doc.extend(body_assign)
@@ -3,14 +3,19 @@
3
3
  This client talks **directly** to the Dataify REST endpoints (the same ones the
4
4
  Dataify MCP server proxies), instead of going through the MCP protocol:
5
5
 
6
- * Scraper / platform tools -> ``POST {base_url}/builder?platform=1``
7
- (form fields: ``spider_name``, ``spider_id``, ``spider_parameters``,
8
- ``spider_errors``, ``file_name``)
9
- * Search engines (Google / Bing / Yandex / DuckDuckGo) -> ``POST {base_url}/request``
10
- (form fields are the engine-specific parameters plus a fixed ``engine`` value)
11
-
12
- Authentication is via a Bearer token, supplied either to the constructor or
13
- through the ``DATAIFY_TOKEN`` environment variable.
6
+ * Scraper / platform tools -> ``POST {base_url}/builder?platform=1``
7
+ (form fields: ``spider_name``, ``spider_id``, ``spider_parameters``,
8
+ ``spider_errors``, ``file_name``)
9
+ * Scraper task status query -> ``GET {base_url}/task_status``
10
+ (query params: ``api_key``, ``task_id``)
11
+ * Scraper task result download -> ``GET {base_url}/download``
12
+ (query params: ``api_key``, ``task_id``, ``type``)
13
+ * Search engines (Google / Bing / Yandex / DuckDuckGo) -> ``POST {base_url}/request``
14
+ (form fields are the engine-specific parameters plus a fixed ``engine`` value)
15
+
16
+ Authentication is via a Bearer token, supplied either to the constructor or
17
+ through the ``DATAIFY_API_TOKEN`` environment variable. The legacy
18
+ ``DATAIFY_TOKEN`` name is still accepted for backward compatibility.
14
19
  """
15
20
 
16
21
  from __future__ import annotations
@@ -28,23 +33,53 @@ from dataify_sdk.errors import (
28
33
  DataifyTimeoutError,
29
34
  )
30
35
 
31
- DEFAULT_BASE_URL = "https://scraperapi.dataify.com"
32
- DEFAULT_TIMEOUT = 120
33
- ENV_TOKEN = "DATAIFY_TOKEN"
34
-
35
- # Endpoints on the upstream.
36
- _SCRAPER_PATH = "/builder?platform=1"
37
- _SERP_PATH = "/request"
36
+ DEFAULT_BASE_URL = "https://scraperapi.dataify.com"
37
+ DEFAULT_TIMEOUT = 120
38
+ ENV_TOKEN = "DATAIFY_API_TOKEN"
39
+ LEGACY_ENV_TOKEN = "DATAIFY_TOKEN"
40
+
41
+
42
+ def _read_env_token() -> str | None:
43
+ for env_name in (ENV_TOKEN, LEGACY_ENV_TOKEN):
44
+ token = os.environ.get(env_name)
45
+ if token:
46
+ return token
47
+ return None
48
+
49
+
50
+ def _redact_url(url: str) -> str:
51
+ try:
52
+ parsed = urllib.parse.urlsplit(url)
53
+ query = [
54
+ (key, "***" if key == "api_key" else value)
55
+ for key, value in urllib.parse.parse_qsl(
56
+ parsed.query, keep_blank_values=True
57
+ )
58
+ ]
59
+ return urllib.parse.urlunsplit(
60
+ parsed._replace(query=urllib.parse.urlencode(query))
61
+ )
62
+ except ValueError:
63
+ return url
64
+
65
+
66
+ # Endpoints on the upstream.
67
+ _SCRAPER_PATH = "/builder?platform=1"
68
+ _SCRAPER_TASK_STATUS_PATH = "/task_status"
69
+ _SCRAPER_TASK_RESULT_PATH = "/download"
70
+ _SCRAPER_TASK_RESULT_TYPES = {"json", "csv", "xlsx"}
71
+ _SERP_PATH = "/request"
38
72
 
39
73
 
40
74
  class DataifyClient:
41
75
  """Synchronous client for the Dataify upstream REST API.
42
76
 
43
- Parameters
44
- ----------
45
- token:
46
- Dataify API token. If omitted, the ``DATAIFY_TOKEN`` environment
47
- variable is used. A token is required to make any request.
77
+ Parameters
78
+ ----------
79
+ token:
80
+ Dataify API token. If omitted, the ``DATAIFY_API_TOKEN`` environment
81
+ variable is used first, with ``DATAIFY_TOKEN`` kept as a legacy
82
+ fallback. A token is required to make any request.
48
83
  base_url:
49
84
  Upstream base URL. Fixed to the Dataify production endpoint by default;
50
85
  exposed only for testing / private deployments.
@@ -52,56 +87,84 @@ class DataifyClient:
52
87
  Per-request timeout in seconds.
53
88
  """
54
89
 
55
- def __init__(
56
- self,
57
- token: str | None = None,
58
- base_url: str = DEFAULT_BASE_URL,
59
- timeout: int = DEFAULT_TIMEOUT,
60
- ) -> None:
61
- self.token = token or os.environ.get(ENV_TOKEN)
62
- if not self.token:
63
- raise ValueError(
64
- "A Dataify token is required: pass token=... or set the "
65
- f"{ENV_TOKEN} environment variable."
66
- )
67
- self.base_url = base_url.rstrip("/")
68
- self.timeout = timeout
90
+ def __init__(
91
+ self,
92
+ token: str | None = None,
93
+ base_url: str = DEFAULT_BASE_URL,
94
+ timeout: int = DEFAULT_TIMEOUT,
95
+ ) -> None:
96
+ self.token = token or _read_env_token()
97
+ if not self.token:
98
+ raise ValueError(
99
+ "A Dataify token is required: pass token=... or set the "
100
+ f"{ENV_TOKEN} environment variable (legacy: {LEGACY_ENV_TOKEN})."
101
+ )
102
+ self.base_url = base_url.rstrip("/")
103
+ self.timeout = timeout
69
104
 
70
- # ------------------------------------------------------------------
71
- # Low-level transport
72
- # ------------------------------------------------------------------
73
- def _post_form(self, path: str, fields: dict[str, str]) -> dict[str, Any]:
74
- """POST form-encoded fields and return the parsed JSON response."""
75
- url = self.base_url + path
76
- # Only send non-empty values (mirrors the Go upstream behaviour).
77
- payload = {k: v for k, v in fields.items() if v not in (None, "")}
105
+ # ------------------------------------------------------------------
106
+ # Low-level transport
107
+ # ------------------------------------------------------------------
108
+ def _read_response_bytes(self, req: urllib.request.Request, url: str) -> bytes:
109
+ display_url = _redact_url(url)
110
+ try:
111
+ with urllib.request.urlopen(req, timeout=self.timeout) as resp:
112
+ return resp.read()
113
+ except urllib.error.HTTPError as exc: # upstream returned >= 400
114
+ body = exc.read().decode("utf-8", errors="replace")
115
+ raise DataifyAPIError(
116
+ f"Dataify API request failed: {display_url}", exc.code, body
117
+ ) from exc
118
+ except urllib.error.URLError as exc:
119
+ if isinstance(exc.reason, TimeoutError):
120
+ raise DataifyTimeoutError(
121
+ f"Request to {display_url} timed out after {self.timeout}s"
122
+ ) from exc
123
+ raise DataifyConnectionError(
124
+ f"Could not connect to Dataify API at {display_url}: {exc.reason}"
125
+ ) from exc
126
+
127
+ def _read_json_response(
128
+ self, req: urllib.request.Request, url: str
129
+ ) -> Any:
130
+ raw = self._read_response_bytes(req, url).decode(
131
+ "utf-8", errors="replace"
132
+ )
133
+ try:
134
+ return json.loads(raw)
135
+ except json.JSONDecodeError:
136
+ return {"raw": raw}
137
+
138
+ def _post_form(self, path: str, fields: dict[str, str]) -> dict[str, Any]:
139
+ """POST form-encoded fields and return the parsed JSON response."""
140
+ url = self.base_url + path
141
+ # Only send non-empty values (mirrors the Go upstream behaviour).
142
+ payload = {k: v for k, v in fields.items() if v not in (None, "")}
78
143
  data = urllib.parse.urlencode(payload, doseq=False).encode("utf-8")
79
144
 
80
- req = urllib.request.Request(url, data=data, method="POST")
81
- req.add_header("Content-Type", "application/x-www-form-urlencoded")
82
- req.add_header("Authorization", f"Bearer {self.token}")
83
-
84
- try:
85
- with urllib.request.urlopen(req, timeout=self.timeout) as resp:
86
- raw = resp.read().decode("utf-8", errors="replace")
87
- except urllib.error.HTTPError as exc: # upstream returned >= 400
88
- body = exc.read().decode("utf-8", errors="replace")
89
- raise DataifyAPIError(
90
- f"Dataify API request failed: {url}", exc.code, body
91
- ) from exc
92
- except urllib.error.URLError as exc:
93
- if isinstance(exc.reason, TimeoutError):
94
- raise DataifyTimeoutError(
95
- f"Request to {url} timed out after {self.timeout}s"
96
- ) from exc
97
- raise DataifyConnectionError(
98
- f"Could not connect to Dataify API at {url}: {exc.reason}"
99
- ) from exc
100
-
101
- try:
102
- return json.loads(raw)
103
- except json.JSONDecodeError:
104
- return {"raw": raw}
145
+ req = urllib.request.Request(url, data=data, method="POST")
146
+ req.add_header("Content-Type", "application/x-www-form-urlencoded")
147
+ req.add_header("Authorization", f"Bearer {self.token}")
148
+ return self._read_json_response(req, url)
149
+
150
+ def _get_json(self, path: str, query: dict[str, str]) -> Any:
151
+ """GET a JSON endpoint with query parameters and return parsed JSON."""
152
+ url, req = self._build_get_request(path, query)
153
+ return self._read_json_response(req, url)
154
+
155
+ def _get_bytes(self, path: str, query: dict[str, str]) -> bytes:
156
+ """GET an endpoint with query parameters and return response bytes."""
157
+ url, req = self._build_get_request(path, query)
158
+ return self._read_response_bytes(req, url)
159
+
160
+ def _build_get_request(
161
+ self, path: str, query: dict[str, str]
162
+ ) -> tuple[str, urllib.request.Request]:
163
+ url = self.base_url + path
164
+ payload = {k: v for k, v in query.items() if v not in (None, "")}
165
+ if payload:
166
+ url = url + "?" + urllib.parse.urlencode(payload, doseq=False)
167
+ return url, urllib.request.Request(url, method="GET")
105
168
 
106
169
  # ------------------------------------------------------------------
107
170
  # High-level helpers used by the generated tool functions
@@ -146,8 +209,8 @@ class DataifyClient:
146
209
  fields["spider_universal"] = spider_universal
147
210
  return self._post_form(_SCRAPER_PATH, fields)
148
211
 
149
- def request_serp(self, engine: str, fields: dict[str, str]) -> dict[str, Any]:
150
- """Call a search-engine endpoint.
212
+ def request_serp(self, engine: str, fields: dict[str, str]) -> dict[str, Any]:
213
+ """Call a search-engine endpoint.
151
214
 
152
215
  Parameters
153
216
  ----------
@@ -158,20 +221,75 @@ class DataifyClient:
158
221
  field names / JSON tags). The ``engine`` field is added
159
222
  automatically.
160
223
  """
161
- form = {"engine": engine}
162
- form.update(fields)
163
- return self._post_form(_SERP_PATH, form)
224
+ form = {"engine": engine}
225
+ form.update(fields)
226
+ return self._post_form(_SERP_PATH, form)
227
+
228
+ def query_scraper_task_status(self, task_id: str) -> dict[str, Any]:
229
+ """Query a scraper task status by task ID.
230
+
231
+ Parameters
232
+ ----------
233
+ task_id:
234
+ Task ID returned by ``request_scraper``.
235
+ """
236
+ task_id = task_id.strip()
237
+ if not task_id:
238
+ raise ValueError("task_id is required")
239
+ return self._get_json(
240
+ _SCRAPER_TASK_STATUS_PATH,
241
+ {"api_key": self.token, "task_id": task_id},
242
+ )
243
+
244
+ def download_scraper_task_result(
245
+ self, task_id: str, result_type: str = "json"
246
+ ) -> Any:
247
+ """Download a scraper task result.
248
+
249
+ Parameters
250
+ ----------
251
+ task_id:
252
+ Task ID returned by ``request_scraper``.
253
+ result_type:
254
+ Result format. Supported values are ``"json"``, ``"csv"``, and
255
+ ``"xlsx"``. JSON results are parsed, CSV results are
256
+ returned as text, and XLSX results are returned as bytes.
257
+ """
258
+ task_id = task_id.strip()
259
+ if not task_id:
260
+ raise ValueError("task_id is required")
261
+
262
+ result_type = result_type.strip().lower()
263
+ if result_type not in _SCRAPER_TASK_RESULT_TYPES:
264
+ raise ValueError(
265
+ "result_type must be one of: "
266
+ + ", ".join(sorted(_SCRAPER_TASK_RESULT_TYPES))
267
+ )
268
+
269
+ fields = {
270
+ "api_key": self.token,
271
+ "task_id": task_id,
272
+ "type": result_type,
273
+ }
274
+ if result_type == "json":
275
+ return self._get_json(_SCRAPER_TASK_RESULT_PATH, fields)
276
+
277
+ content = self._get_bytes(_SCRAPER_TASK_RESULT_PATH, fields)
278
+ if result_type == "csv":
279
+ return content.decode("utf-8", errors="replace")
280
+ return content
164
281
 
165
282
 
166
283
  _DEFAULT_CLIENT: DataifyClient | None = None
167
284
 
168
285
 
169
- def get_default_client() -> DataifyClient:
170
- """Return a process-wide default :class:`DataifyClient`.
171
-
172
- Uses ``DATAIFY_TOKEN`` from the environment. Handy for the generated
173
- tool functions, which accept an optional ``client`` argument.
174
- """
286
+ def get_default_client() -> DataifyClient:
287
+ """Return a process-wide default :class:`DataifyClient`.
288
+
289
+ Uses ``DATAIFY_API_TOKEN`` from the environment, with
290
+ ``DATAIFY_TOKEN`` kept as a legacy fallback. Handy for the generated
291
+ tool functions, which accept an optional ``client`` argument.
292
+ """
175
293
  global _DEFAULT_CLIENT
176
294
  if _DEFAULT_CLIENT is None:
177
295
  _DEFAULT_CLIENT = DataifyClient()