dataify-sdk 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,79 @@
1
+ """User management tools.
2
+
3
+ Wraps MCP tools for querying user account info, balance, API keys,
4
+ and credit usage.
5
+
6
+ Auto-generated. Regenerate with: python scripts/codegen.py
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from typing import Any
12
+
13
+ from pydantic import BaseModel
14
+
15
+
16
+ # ---------------------------------------------------------------------------
17
+ # query_user_info — no parameters
18
+ # ---------------------------------------------------------------------------
19
+
20
+
21
+ async def query_user_info(self) -> Any:
22
+ """Call ``query_user_info`` — 查询用户账号信息。
23
+
24
+ 根据链接 token 查询 Dataify 用户账号信息、注册日期、
25
+ 脱敏手机号和实名认证信息。
26
+ """
27
+ return await self.call_tool("query_user_info", {})
28
+
29
+
30
+ # ---------------------------------------------------------------------------
31
+ # query_user_balance — no parameters
32
+ # ---------------------------------------------------------------------------
33
+
34
+
35
+ async def query_user_balance(self) -> Any:
36
+ """Call ``query_user_balance`` — 查询用户余额。
37
+
38
+ 查询 Dataify 用户剩余积分或余额、累计充值和累计使用。
39
+ """
40
+ return await self.call_tool("query_user_balance", {})
41
+
42
+
43
+ # ---------------------------------------------------------------------------
44
+ # query_user_api_keys — no parameters
45
+ # ---------------------------------------------------------------------------
46
+
47
+
48
+ async def query_user_api_keys(self) -> Any:
49
+ """Call ``query_user_api_keys`` — 查询用户 API Token 列表。
50
+
51
+ 根据链接 token 查询当前 Dataify 用户的 API Token 列表。
52
+ """
53
+ return await self.call_tool("query_user_api_keys", {})
54
+
55
+
56
+ # ---------------------------------------------------------------------------
57
+ # query_user_credit_usage — no parameters
58
+ # ---------------------------------------------------------------------------
59
+
60
+
61
+ async def query_user_credit_usage(self) -> Any:
62
+ """Call ``query_user_credit_usage`` — 查询用户积分使用明细。
63
+
64
+ 查询 Dataify 用户的每日积分消耗统计。
65
+ """
66
+ return await self.call_tool("query_user_credit_usage", {})
67
+
68
+
69
+ # ---------------------------------------------------------------------------
70
+ # Attach methods to DataifyClient
71
+ # ---------------------------------------------------------------------------
72
+
73
+
74
+ def _attach(client_cls: type) -> None:
75
+ """Attach all user management methods to the client class."""
76
+ client_cls.query_user_info = query_user_info
77
+ client_cls.query_user_balance = query_user_balance
78
+ client_cls.query_user_api_keys = query_user_api_keys
79
+ client_cls.query_user_credit_usage = query_user_credit_usage
@@ -0,0 +1,53 @@
1
+ """Web unlocker tool.
2
+
3
+ Wraps the ``request_web_unlocker`` MCP tool for bypassing CAPTCHAs
4
+ and extracting rendered page content.
5
+
6
+ Auto-generated. Regenerate with: python scripts/codegen.py
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from typing import Any
12
+
13
+ from pydantic import BaseModel, Field
14
+
15
+
16
+ class RequestWebUnlockerParams(BaseModel):
17
+ """Parameters for the ``request_web_unlocker`` tool.
18
+
19
+ 调用 Dataify 通用采集 API,输入任意 URL,智能识别 CAPTCHA、
20
+ 自动执行 JS 渲染,返回完整 PNG 截图或 HTML 源码。
21
+ """
22
+
23
+ url: str = Field(..., description="解锁网址,必填项")
24
+ type: str | None = Field(default="html", description="输出格式。html 和/或 png,逗号分隔,默认 html")
25
+ js_render: str | None = Field(default="True", description="JS 渲染,建议开启。True/False,默认 True")
26
+ block_resources: str | None = Field(default="", description="阻止加载的资源类型,逗号分隔 (javascript,css)")
27
+ clean_content: str | None = Field(default="", description="清除返回内容中的 JS 或 CSS 代码")
28
+ country: str | None = Field(default="us", description="代理所在国家/地区代码,默认 us")
29
+ headers: str | None = Field(default="", description='自定义请求 Headers,JSON 格式 {"aaa":"bbb"}')
30
+ cookies: str | None = Field(default="", description='自定义请求 Cookies,JSON 格式 {"aaa":"bbb"}')
31
+ wait: str | None = Field(default="", description="页面加载后额外等待的时间(毫秒)")
32
+ wait_for: str | None = Field(default="", description="等待指定的 CSS 选择器出现后再返回内容")
33
+ follow_redirect: str | None = Field(default="True", description="跟随 HTTP 重定向。True/False,默认 True")
34
+
35
+
36
+ async def request_web_unlocker(self, params: RequestWebUnlockerParams | None = None) -> Any:
37
+ """Call ``request_web_unlocker`` — 通用网页解锁。
38
+
39
+ 调用 Dataify 通用采集 API,输入任意 URL,智能识别 CAPTCHA、
40
+ 自动执行 JS 渲染,返回完整 PNG 截图或 HTML 源码。
41
+ """
42
+ arguments = params.model_dump(exclude_none=True) if params else {}
43
+ return await self.call_tool("request_web_unlocker", arguments)
44
+
45
+
46
+ # ---------------------------------------------------------------------------
47
+ # Attach methods to DataifyClient
48
+ # ---------------------------------------------------------------------------
49
+
50
+
51
+ def _attach(client_cls: type) -> None:
52
+ """Attach all web unlocker methods to the client class."""
53
+ client_cls.request_web_unlocker = request_web_unlocker
@@ -0,0 +1,163 @@
1
+ """YouTube scraper tools.
2
+
3
+ Wraps all YouTube-related MCP tools: video, video_post, profiles,
4
+ comment, transcript, product, and audio.
5
+
6
+ Auto-generated. Regenerate with: python scripts/codegen.py
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from typing import Any
12
+
13
+ from pydantic import BaseModel, Field
14
+
15
+
16
+ class _YouTubeFileParams(BaseModel):
17
+ """Base params for YouTube tools with file_name."""
18
+
19
+ file_name: str | None = Field(default="{{TasksID}}", description="Builder file_name")
20
+
21
+
22
+ # ---------------------------------------------------------------------------
23
+ # scrape_youtube_video
24
+ # ---------------------------------------------------------------------------
25
+
26
+
27
+ class ScrapeYouTubeVideoParams(_YouTubeFileParams):
28
+ """Parameters for ``scrape_youtube_video`` — YouTube 视频文件下载。
29
+
30
+ 下载 YouTube 视频文件,支持字幕、分辨率、视频编码、音频格式等配置。
31
+ """
32
+
33
+ url: str = Field(default="https://www.youtube.com/watch?v=_SdpvpvVrLY", description="YouTube 视频 URL")
34
+ subtitles_language: str | None = Field(default="ab", description="字幕语言代码")
35
+ selected_only: str | None = Field(default="false", description="仅下载已选规格: true/false")
36
+ resolution: str | None = Field(default="<=360p", description="分辨率,如 <=1080p")
37
+ video_codec: str | None = Field(default="vp9", description="视频编码: vp9, avc1, av01")
38
+ audio_format: str | None = Field(default="opus", description="音频格式: opus, m4a")
39
+ bitrate: str | None = Field(default="<=320", description="比特率,如 <=320")
40
+
41
+
42
+ async def scrape_youtube_video(self, params: ScrapeYouTubeVideoParams | None = None) -> Any:
43
+ """Call ``scrape_youtube_video`` — YouTube 视频文件下载。"""
44
+ arguments = params.model_dump(exclude_none=True) if params else {}
45
+ return await self.call_tool("scrape_youtube_video", arguments)
46
+
47
+
48
+ # ---------------------------------------------------------------------------
49
+ # scrape_youtube_video_post
50
+ # ---------------------------------------------------------------------------
51
+
52
+
53
+ class ScrapeYouTubeVideoPostParams(_YouTubeFileParams):
54
+ """Parameters for ``scrape_youtube_video_post`` — YouTube 视频帖子采集。"""
55
+
56
+ url: str = Field(..., description="YouTube 视频 URL")
57
+
58
+
59
+ async def scrape_youtube_video_post(self, params: ScrapeYouTubeVideoPostParams | None = None) -> Any:
60
+ """Call ``scrape_youtube_video_post`` — YouTube 视频帖子采集。"""
61
+ arguments = params.model_dump(exclude_none=True) if params else {}
62
+ return await self.call_tool("scrape_youtube_video_post", arguments)
63
+
64
+
65
+ # ---------------------------------------------------------------------------
66
+ # scrape_youtube_profiles
67
+ # ---------------------------------------------------------------------------
68
+
69
+
70
+ class ScrapeYouTubeProfilesParams(_YouTubeFileParams):
71
+ """Parameters for ``scrape_youtube_profiles`` — YouTube 频道信息采集。"""
72
+
73
+ url: str = Field(..., description="YouTube 频道 URL")
74
+
75
+
76
+ async def scrape_youtube_profiles(self, params: ScrapeYouTubeProfilesParams | None = None) -> Any:
77
+ """Call ``scrape_youtube_profiles`` — YouTube 频道信息采集。"""
78
+ arguments = params.model_dump(exclude_none=True) if params else {}
79
+ return await self.call_tool("scrape_youtube_profiles", arguments)
80
+
81
+
82
+ # ---------------------------------------------------------------------------
83
+ # scrape_youtube_comment
84
+ # ---------------------------------------------------------------------------
85
+
86
+
87
+ class ScrapeYouTubeCommentParams(_YouTubeFileParams):
88
+ """Parameters for ``scrape_youtube_comment`` — YouTube 评论采集。"""
89
+
90
+ url: str = Field(..., description="YouTube 视频 URL")
91
+
92
+
93
+ async def scrape_youtube_comment(self, params: ScrapeYouTubeCommentParams | None = None) -> Any:
94
+ """Call ``scrape_youtube_comment`` — YouTube 评论采集。"""
95
+ arguments = params.model_dump(exclude_none=True) if params else {}
96
+ return await self.call_tool("scrape_youtube_comment", arguments)
97
+
98
+
99
+ # ---------------------------------------------------------------------------
100
+ # scrape_youtube_transcript
101
+ # ---------------------------------------------------------------------------
102
+
103
+
104
+ class ScrapeYouTubeTranscriptParams(_YouTubeFileParams):
105
+ """Parameters for ``scrape_youtube_transcript`` — YouTube 字幕采集。"""
106
+
107
+ url: str = Field(..., description="YouTube 视频 URL")
108
+
109
+
110
+ async def scrape_youtube_transcript(self, params: ScrapeYouTubeTranscriptParams | None = None) -> Any:
111
+ """Call ``scrape_youtube_transcript`` — YouTube 字幕采集。"""
112
+ arguments = params.model_dump(exclude_none=True) if params else {}
113
+ return await self.call_tool("scrape_youtube_transcript", arguments)
114
+
115
+
116
+ # ---------------------------------------------------------------------------
117
+ # scrape_youtube_product
118
+ # ---------------------------------------------------------------------------
119
+
120
+
121
+ class ScrapeYouTubeProductParams(_YouTubeFileParams):
122
+ """Parameters for ``scrape_youtube_product`` — YouTube 商品采集。"""
123
+
124
+ url: str = Field(..., description="YouTube 商品相关 URL")
125
+
126
+
127
+ async def scrape_youtube_product(self, params: ScrapeYouTubeProductParams | None = None) -> Any:
128
+ """Call ``scrape_youtube_product`` — YouTube 商品采集。"""
129
+ arguments = params.model_dump(exclude_none=True) if params else {}
130
+ return await self.call_tool("scrape_youtube_product", arguments)
131
+
132
+
133
+ # ---------------------------------------------------------------------------
134
+ # scrape_youtube_audio
135
+ # ---------------------------------------------------------------------------
136
+
137
+
138
+ class ScrapeYouTubeAudioParams(_YouTubeFileParams):
139
+ """Parameters for ``scrape_youtube_audio`` — YouTube 音频采集。"""
140
+
141
+ url: str = Field(..., description="YouTube 视频 URL")
142
+
143
+
144
+ async def scrape_youtube_audio(self, params: ScrapeYouTubeAudioParams | None = None) -> Any:
145
+ """Call ``scrape_youtube_audio`` — YouTube 音频采集。"""
146
+ arguments = params.model_dump(exclude_none=True) if params else {}
147
+ return await self.call_tool("scrape_youtube_audio", arguments)
148
+
149
+
150
+ # ---------------------------------------------------------------------------
151
+ # Attach methods to DataifyClient
152
+ # ---------------------------------------------------------------------------
153
+
154
+
155
+ def _attach(client_cls: type) -> None:
156
+ """Attach all YouTube scraper methods to the client class."""
157
+ client_cls.scrape_youtube_video = scrape_youtube_video
158
+ client_cls.scrape_youtube_video_post = scrape_youtube_video_post
159
+ client_cls.scrape_youtube_profiles = scrape_youtube_profiles
160
+ client_cls.scrape_youtube_comment = scrape_youtube_comment
161
+ client_cls.scrape_youtube_transcript = scrape_youtube_transcript
162
+ client_cls.scrape_youtube_product = scrape_youtube_product
163
+ client_cls.scrape_youtube_audio = scrape_youtube_audio
@@ -0,0 +1,39 @@
1
+ """Type definitions for the Dataify MCP SDK."""
2
+
3
+ from dataify_mcp.types._errors import (
4
+ AuthenticationError,
5
+ ConnectionError,
6
+ DataifyError,
7
+ ProtocolError,
8
+ ServerError,
9
+ TimeoutError,
10
+ ToolAccessError,
11
+ ToolError,
12
+ )
13
+ from dataify_mcp.types._mcp import (
14
+ JSONRPCError,
15
+ JSONRPCRequest,
16
+ JSONRPCResponse,
17
+ MCPServerCapabilities,
18
+ ServerInfo,
19
+ ToolDefinition,
20
+ )
21
+
22
+ __all__ = [
23
+ # MCP types
24
+ "JSONRPCRequest",
25
+ "JSONRPCResponse",
26
+ "JSONRPCError",
27
+ "ToolDefinition",
28
+ "ServerInfo",
29
+ "MCPServerCapabilities",
30
+ # Errors
31
+ "DataifyError",
32
+ "AuthenticationError",
33
+ "ToolAccessError",
34
+ "ConnectionError",
35
+ "ProtocolError",
36
+ "ToolError",
37
+ "TimeoutError",
38
+ "ServerError",
39
+ ]
@@ -0,0 +1,220 @@
1
+ """Enums for common Dataify parameter values.
2
+
3
+ These provide type-safe alternatives to raw string literals when
4
+ constructing tool parameters.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from enum import Enum
10
+
11
+
12
+ # ---------------------------------------------------------------------------
13
+ # Web Unlocker
14
+ # ---------------------------------------------------------------------------
15
+
16
+
17
+ class UnlockerOutputType(str, Enum):
18
+ """Output format for request_web_unlocker 'type' parameter."""
19
+
20
+ HTML = "html"
21
+ PNG = "png"
22
+ HTML_PNG = "html,png"
23
+
24
+
25
+ # ---------------------------------------------------------------------------
26
+ # Google / Bing / Search Engine
27
+ # ---------------------------------------------------------------------------
28
+
29
+
30
+ class JSONOutputFormat(str, Enum):
31
+ """JSON output format for search engine results."""
32
+
33
+ JSON = "1"
34
+ JSON_HTML = "2"
35
+ HTML = "3"
36
+ LIGHT_JSON = "4"
37
+
38
+
39
+ class Device(str, Enum):
40
+ """Device type for search engine emulation."""
41
+
42
+ DESKTOP = "desktop"
43
+ TABLET = "tablet"
44
+ MOBILE = "mobile"
45
+
46
+
47
+ class SafeSearch(str, Enum):
48
+ """SafeSearch / adult content filtering."""
49
+
50
+ ACTIVE = "active"
51
+ OFF = "off"
52
+
53
+
54
+ class BingSafeSearch(str, Enum):
55
+ """Bing SafeSearch levels."""
56
+
57
+ OFF = "Off"
58
+ MODERATE = "Moderate"
59
+ STRICT = "Strict"
60
+
61
+
62
+ class DuckDuckGoSafeSearch(str, Enum):
63
+ """DuckDuckGo SafeSearch levels."""
64
+
65
+ STRICT = "1"
66
+ MODERATE = "-1"
67
+ OFF = "-2"
68
+
69
+
70
+ class DateFilter(str, Enum):
71
+ """Date filter shortcuts for search results."""
72
+
73
+ PAST_DAY = "d"
74
+ PAST_WEEK = "w"
75
+ PAST_MONTH = "m"
76
+ PAST_YEAR = "y"
77
+
78
+
79
+ class TaskStatus(str, Enum):
80
+ """Generic task status codes."""
81
+
82
+ ALL = "-1"
83
+ PROCESSING = "-1"
84
+ SUCCESS = "0"
85
+ FAILED = "1"
86
+
87
+
88
+ class ScraperTaskStatus(str, Enum):
89
+ """Scraper task status codes."""
90
+
91
+ ALL = "0"
92
+ PROCESSING = "-1"
93
+ SUCCESS = "200"
94
+ FAILED = "400"
95
+
96
+
97
+ class ScraperType(str, Enum):
98
+ """Scraper product type."""
99
+
100
+ ALL = "-1"
101
+ WEB_SCRAPER = "0"
102
+ SERP = "1"
103
+
104
+
105
+ class TaskType(str, Enum):
106
+ """Scraper task type."""
107
+
108
+ ALL = "0"
109
+ SERP = "1"
110
+ WEB_SCRAPER = "2"
111
+
112
+
113
+ # ---------------------------------------------------------------------------
114
+ # Yandex
115
+ # ---------------------------------------------------------------------------
116
+
117
+
118
+ class YandexFamilyMode(str, Enum):
119
+ """Yandex SafeSearch / family filter."""
120
+
121
+ OFF = "0"
122
+ MODERATE = "1"
123
+ STRICT = "2"
124
+
125
+
126
+ # ---------------------------------------------------------------------------
127
+ # Amazon
128
+ # ---------------------------------------------------------------------------
129
+
130
+
131
+ class AmazonSpiderID(str, Enum):
132
+ """Amazon product spider identifiers."""
133
+
134
+ BY_ASIN = "amazon_product_by-asin"
135
+ BY_URL = "amazon_product_by-url"
136
+ BY_KEYWORDS = "amazon_product_by-keywords"
137
+ BY_CATEGORY_URL = "amazon_product_by-category-url"
138
+ BY_BEST_SELLERS = "amazon_product_by-best-sellers"
139
+
140
+
141
+ class AmazonDomain(str, Enum):
142
+ """Amazon marketplace domains."""
143
+
144
+ COM = "amazon.com"
145
+ JP = "amazon.co.jp"
146
+ DE = "amazon.de"
147
+ UK = "amazon.co.uk"
148
+ FR = "amazon.fr"
149
+ IT = "amazon.it"
150
+ ES = "amazon.es"
151
+ CA = "amazon.ca"
152
+ IN = "amazon.in"
153
+ BR = "amazon.com.br"
154
+ MX = "amazon.com.mx"
155
+ AU = "amazon.com.au"
156
+
157
+
158
+ class AmazonSortBy(str, Enum):
159
+ """Amazon product sorting options (Chinese)."""
160
+
161
+ BEST_SELLERS = "畅销排行"
162
+ NEWEST_ARRIVALS = "最新上架"
163
+ AVG_CUSTOMER_REVIEW = "平均评价"
164
+ PRICE_HIGH_TO_LOW = "价格:从高到低"
165
+ PRICE_LOW_TO_HIGH = "价格:从低到高"
166
+ FEATURED = "精选推荐"
167
+
168
+
169
+ # ---------------------------------------------------------------------------
170
+ # YouTube
171
+ # ---------------------------------------------------------------------------
172
+
173
+
174
+ class YouTubeVideoCodec(str, Enum):
175
+ """YouTube video codec options."""
176
+
177
+ VP9 = "vp9"
178
+ AVC1 = "avc1" # Also: h264, avc
179
+ AV01 = "av01" # Also: av1
180
+
181
+
182
+ class YouTubeAudioFormat(str, Enum):
183
+ """YouTube audio format options."""
184
+
185
+ OPUS = "opus"
186
+ M4A = "m4a" # Also: aac
187
+
188
+
189
+ class YouTubeResolution(str, Enum):
190
+ """YouTube video resolution options (bare value, no operator)."""
191
+
192
+ P360 = "360p"
193
+ P480 = "480p"
194
+ P720 = "720p"
195
+ P1080 = "1080p"
196
+ P1440 = "1440p"
197
+ P2160 = "2160p"
198
+
199
+
200
+ class YouTubeBitrate(str, Enum):
201
+ """YouTube audio bitrate options (bare value, no operator)."""
202
+
203
+ K48 = "48"
204
+ K64 = "64"
205
+ K128 = "128"
206
+ K160 = "160"
207
+ K256 = "256"
208
+ K320 = "320"
209
+
210
+
211
+ # ---------------------------------------------------------------------------
212
+ # Comparison operators
213
+ # ---------------------------------------------------------------------------
214
+
215
+
216
+ class ComparisonOp(str, Enum):
217
+ """Comparison operators used in YouTube resolution/bitrate."""
218
+
219
+ LE = "<="
220
+ GE = ">="
@@ -0,0 +1,73 @@
1
+ """Error hierarchy for the Dataify MCP SDK.
2
+
3
+ Every exception raised by the SDK inherits from `DataifyError`, making it
4
+ easy to catch any SDK-related error with a single except clause.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+
10
+ class DataifyError(Exception):
11
+ """Base exception for all SDK errors."""
12
+
13
+
14
+ class AuthenticationError(DataifyError):
15
+ """Raised when the API token is missing, invalid, or expired.
16
+
17
+ The user should obtain a valid token from https://dashboard.dataify.com
18
+ and pass it to the client constructor or as a query parameter.
19
+ """
20
+
21
+
22
+ class ToolAccessError(DataifyError):
23
+ """Raised when the token does not have permission to use a specific tool.
24
+
25
+ This typically means the tool is not in the user's plan. The user can
26
+ upgrade via https://dashboard.dataify.com or pass a ``tools=`` query
27
+ parameter to limit which tools are requested.
28
+ """
29
+
30
+
31
+ class ConnectionError(DataifyError):
32
+ """Raised when the SDK cannot connect to the MCP server.
33
+
34
+ This covers DNS resolution failures, TCP connection refused, TLS errors,
35
+ and other transport-level issues.
36
+ """
37
+
38
+
39
+ class ProtocolError(DataifyError):
40
+ """Raised when the server sends an invalid or unexpected JSON-RPC response.
41
+
42
+ Examples include missing ``id`` fields, invalid JSON, or responses that
43
+ don't match any pending request.
44
+ """
45
+
46
+
47
+ class ToolError(DataifyError):
48
+ """Raised when a tool call returns ``isError: true``.
49
+
50
+ The tool executed on the server side but failed (e.g. upstream API
51
+ returned an error, invalid parameters, etc.). The error message
52
+ comes from the server.
53
+ """
54
+
55
+
56
+ class TimeoutError(DataifyError):
57
+ """Raised when a request to the MCP server exceeds the configured timeout."""
58
+
59
+
60
+ class ServerError(DataifyError):
61
+ """Raised when the server returns an HTTP 5xx status code."""
62
+
63
+
64
+ # Maps JSON-RPC error codes to SDK exception types.
65
+ # Reference: https://www.jsonrpc.org/specification#error_object
66
+ ERROR_CODE_MAP: dict[int, type[DataifyError]] = {
67
+ -32000: ServerError, # Server error (generic)
68
+ -32600: ProtocolError, # Invalid Request
69
+ -32601: ProtocolError, # Method not found
70
+ -32602: ProtocolError, # Invalid params
71
+ -32603: ServerError, # Internal error
72
+ -32001: AuthenticationError, # Unauthorized (common MCP convention)
73
+ }