dataify-sdk 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dataify_mcp/__init__.py +37 -0
- dataify_mcp/_version.py +3 -0
- dataify_mcp/client/__init__.py +7 -0
- dataify_mcp/client/_base.py +223 -0
- dataify_mcp/client/_http.py +205 -0
- dataify_mcp/client/_protocol.py +204 -0
- dataify_mcp/client/_sse.py +319 -0
- dataify_mcp/tools/__init__.py +90 -0
- dataify_mcp/tools/amazon.py +137 -0
- dataify_mcp/tools/bing.py +141 -0
- dataify_mcp/tools/facebook.py +63 -0
- dataify_mcp/tools/glassdoor.py +41 -0
- dataify_mcp/tools/google_scraper.py +121 -0
- dataify_mcp/tools/google_serp.py +320 -0
- dataify_mcp/tools/indeed.py +41 -0
- dataify_mcp/tools/instagram.py +52 -0
- dataify_mcp/tools/linkedin.py +41 -0
- dataify_mcp/tools/other_scrapers.py +137 -0
- dataify_mcp/tools/other_search.py +90 -0
- dataify_mcp/tools/reddit.py +41 -0
- dataify_mcp/tools/task_status.py +188 -0
- dataify_mcp/tools/tiktok.py +81 -0
- dataify_mcp/tools/twitter.py +41 -0
- dataify_mcp/tools/user.py +79 -0
- dataify_mcp/tools/web_unlocker.py +53 -0
- dataify_mcp/tools/youtube.py +163 -0
- dataify_mcp/types/__init__.py +39 -0
- dataify_mcp/types/_enums.py +220 -0
- dataify_mcp/types/_errors.py +73 -0
- dataify_mcp/types/_mcp.py +89 -0
- dataify_sdk-0.1.0.dist-info/METADATA +154 -0
- dataify_sdk-0.1.0.dist-info/RECORD +34 -0
- dataify_sdk-0.1.0.dist-info/WHEEL +4 -0
- dataify_sdk-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""LinkedIn scraper tools.
|
|
2
|
+
|
|
3
|
+
Wraps LinkedIn MCP tools: company information and job listings.
|
|
4
|
+
|
|
5
|
+
Auto-generated. Regenerate with: python scripts/codegen.py
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, Field
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class _LinkedInBase(BaseModel):
|
|
16
|
+
file_name: str | None = Field(default="{{TasksID}}", description="Builder file_name")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ScrapeLinkedinCompanyInformationParams(_LinkedInBase):
|
|
20
|
+
"""Parameters for ``scrape_linkedin_company_information`` — LinkedIn 公司信息采集。"""
|
|
21
|
+
url: str = Field(..., description="LinkedIn 公司页面 URL")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ScrapeLinkedinJobListingsInformationParams(_LinkedInBase):
|
|
25
|
+
"""Parameters for ``scrape_linkedin_job_listings_information`` — LinkedIn 职位列表采集。"""
|
|
26
|
+
url: str = Field(..., description="LinkedIn 职位搜索 URL")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
async def scrape_linkedin_company_information(self, params: ScrapeLinkedinCompanyInformationParams | None = None) -> Any: # noqa: E501
|
|
30
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
31
|
+
return await self.call_tool("scrape_linkedin_company_information", arguments)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
async def scrape_linkedin_job_listings_information(self, params: ScrapeLinkedinJobListingsInformationParams | None = None) -> Any: # noqa: E501
|
|
35
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
36
|
+
return await self.call_tool("scrape_linkedin_job_listings_information", arguments)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _attach(client_cls: type) -> None:
|
|
40
|
+
client_cls.scrape_linkedin_company_information = scrape_linkedin_company_information
|
|
41
|
+
client_cls.scrape_linkedin_job_listings_information = scrape_linkedin_job_listings_information
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Miscellaneous platform scraper tools.
|
|
2
|
+
|
|
3
|
+
Wraps Airbnb, Booking, Crunchbase, eBay, GitHub, Walmart, and Zillow MCP tools.
|
|
4
|
+
|
|
5
|
+
Auto-generated. Regenerate with: python scripts/codegen.py
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, Field
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class _ScraperBase(BaseModel):
|
|
16
|
+
file_name: str | None = Field(default="{{TasksID}}", description="Builder file_name")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
# ---------------------------------------------------------------------------
|
|
20
|
+
# scrape_airbnb_product
|
|
21
|
+
# ---------------------------------------------------------------------------
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ScrapeAirbnbProductParams(_ScraperBase):
|
|
25
|
+
"""Parameters for ``scrape_airbnb_product`` — Airbnb 房源采集。"""
|
|
26
|
+
url: str = Field(..., description="Airbnb 房源 URL")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
async def scrape_airbnb_product(self, params: ScrapeAirbnbProductParams | None = None) -> Any:
|
|
30
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
31
|
+
return await self.call_tool("scrape_airbnb_product", arguments)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# ---------------------------------------------------------------------------
|
|
35
|
+
# scrape_booking_hotel_list
|
|
36
|
+
# ---------------------------------------------------------------------------
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class ScrapeBookingHotelListParams(_ScraperBase):
|
|
40
|
+
"""Parameters for ``scrape_booking_hotel_list`` — Booking 酒店列表采集。"""
|
|
41
|
+
url: str = Field(..., description="Booking 酒店搜索 URL")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
async def scrape_booking_hotel_list(self, params: ScrapeBookingHotelListParams | None = None) -> Any:
|
|
45
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
46
|
+
return await self.call_tool("scrape_booking_hotel_list", arguments)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# ---------------------------------------------------------------------------
|
|
50
|
+
# scrape_crunchbase_company
|
|
51
|
+
# ---------------------------------------------------------------------------
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class ScrapeCrunchbaseCompanyParams(_ScraperBase):
|
|
55
|
+
"""Parameters for ``scrape_crunchbase_company`` — Crunchbase 公司采集。"""
|
|
56
|
+
url: str = Field(..., description="Crunchbase 公司页面 URL")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
async def scrape_crunchbase_company(self, params: ScrapeCrunchbaseCompanyParams | None = None) -> Any:
|
|
60
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
61
|
+
return await self.call_tool("scrape_crunchbase_company", arguments)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# ---------------------------------------------------------------------------
|
|
65
|
+
# scrape_ebay_info
|
|
66
|
+
# ---------------------------------------------------------------------------
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class ScrapeEbayInfoParams(_ScraperBase):
|
|
70
|
+
"""Parameters for ``scrape_ebay_info`` — eBay 商品信息采集。"""
|
|
71
|
+
url: str = Field(..., description="eBay 商品页面 URL")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
async def scrape_ebay_info(self, params: ScrapeEbayInfoParams | None = None) -> Any:
|
|
75
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
76
|
+
return await self.call_tool("scrape_ebay_info", arguments)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
# ---------------------------------------------------------------------------
|
|
80
|
+
# scrape_github_repository
|
|
81
|
+
# ---------------------------------------------------------------------------
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class ScrapeGithubRepositoryParams(_ScraperBase):
|
|
85
|
+
"""Parameters for ``scrape_github_repository`` — GitHub 仓库信息采集。"""
|
|
86
|
+
url: str = Field(..., description="GitHub 仓库 URL")
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
async def scrape_github_repository(self, params: ScrapeGithubRepositoryParams | None = None) -> Any:
|
|
90
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
91
|
+
return await self.call_tool("scrape_github_repository", arguments)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# ---------------------------------------------------------------------------
|
|
95
|
+
# scrape_walmart_product
|
|
96
|
+
# ---------------------------------------------------------------------------
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class ScrapeWalmartProductParams(_ScraperBase):
|
|
100
|
+
"""Parameters for ``scrape_walmart_product`` — Walmart 商品采集。"""
|
|
101
|
+
url: str = Field(..., description="Walmart 商品页面 URL")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
async def scrape_walmart_product(self, params: ScrapeWalmartProductParams | None = None) -> Any:
|
|
105
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
106
|
+
return await self.call_tool("scrape_walmart_product", arguments)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# ---------------------------------------------------------------------------
|
|
110
|
+
# scrape_zillow_product
|
|
111
|
+
# ---------------------------------------------------------------------------
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class ScrapeZillowProductParams(_ScraperBase):
|
|
115
|
+
"""Parameters for ``scrape_zillow_product`` — Zillow 房产采集。"""
|
|
116
|
+
url: str = Field(..., description="Zillow 房产页面 URL")
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
async def scrape_zillow_product(self, params: ScrapeZillowProductParams | None = None) -> Any:
|
|
120
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
121
|
+
return await self.call_tool("scrape_zillow_product", arguments)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# ---------------------------------------------------------------------------
|
|
125
|
+
# Attach methods to DataifyClient
|
|
126
|
+
# ---------------------------------------------------------------------------
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _attach(client_cls: type) -> None:
|
|
130
|
+
"""Attach all miscellaneous scraper methods to the client class."""
|
|
131
|
+
client_cls.scrape_airbnb_product = scrape_airbnb_product
|
|
132
|
+
client_cls.scrape_booking_hotel_list = scrape_booking_hotel_list
|
|
133
|
+
client_cls.scrape_crunchbase_company = scrape_crunchbase_company
|
|
134
|
+
client_cls.scrape_ebay_info = scrape_ebay_info
|
|
135
|
+
client_cls.scrape_github_repository = scrape_github_repository
|
|
136
|
+
client_cls.scrape_walmart_product = scrape_walmart_product
|
|
137
|
+
client_cls.scrape_zillow_product = scrape_zillow_product
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""Other search engine tools.
|
|
2
|
+
|
|
3
|
+
Wraps Yandex and DuckDuckGo search tools.
|
|
4
|
+
|
|
5
|
+
Auto-generated. Regenerate with: python scripts/codegen.py
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, Field
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
# ---------------------------------------------------------------------------
|
|
16
|
+
# yandex_search
|
|
17
|
+
# ---------------------------------------------------------------------------
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class YandexSearchParams(BaseModel):
|
|
21
|
+
"""Parameters for ``yandex_search`` — Yandex 网页搜索。
|
|
22
|
+
|
|
23
|
+
通过 Yandex 搜索公开网页信息,按关键词、地区、语言等条件获取结果。
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
model_config = {"protected_namespaces": ()}
|
|
27
|
+
|
|
28
|
+
text: str = Field(default="", description="搜索查询内容")
|
|
29
|
+
json: str | None = Field(default="1", description="输出格式: 1 JSON, 2 JSON+HTML, 3 HTML, 4 Light JSON")
|
|
30
|
+
yandex_domain: str | None = Field(default="yandex.com", description="Yandex 域名")
|
|
31
|
+
lang: str | None = Field(default="en", description="搜索语言")
|
|
32
|
+
lr: str | None = Field(default="", description="限制搜索的国家或地区 ID")
|
|
33
|
+
p: str | None = Field(default="0", description="页码,从 0 开始")
|
|
34
|
+
family_mode: str | None = Field(default="1", description="家庭模式: 0 关闭, 1 中等, 2 严格")
|
|
35
|
+
fix_typo: str | None = Field(default="true", description="自动拼写纠正: true/false")
|
|
36
|
+
groups_on_page: str | None = Field(default="10", description="单页最大群组数")
|
|
37
|
+
no_cache: str | None = Field(default="false", description="是否跳过缓存")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
async def yandex_search(self, params: YandexSearchParams | None = None) -> Any:
|
|
41
|
+
"""Call ``yandex_search`` — Yandex 网页搜索。
|
|
42
|
+
|
|
43
|
+
通过 Yandex 搜索公开网页信息,按关键词、地区、语言等条件获取搜索结果。
|
|
44
|
+
"""
|
|
45
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
46
|
+
return await self.call_tool("yandex_search", arguments)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# ---------------------------------------------------------------------------
|
|
50
|
+
# duckduckgo_search
|
|
51
|
+
# ---------------------------------------------------------------------------
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class DuckDuckGoSearchParams(BaseModel):
|
|
55
|
+
"""Parameters for ``duckduckgo_search`` — DuckDuckGo 网页搜索。
|
|
56
|
+
|
|
57
|
+
通过 DuckDuckGo 搜索公开网页信息,获取隐私搜索结果。
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
model_config = {"protected_namespaces": ()}
|
|
61
|
+
|
|
62
|
+
q: str = Field(default="Pizza", description="搜索查询内容")
|
|
63
|
+
json: str | None = Field(default="1", description="输出格式: 1 JSON, 2 JSON+HTML, 3 HTML, 4 Light JSON")
|
|
64
|
+
kl: str | None = Field(default="", description="地区代码,如 us-en, uk-en, fr-fr")
|
|
65
|
+
search_assist: str | None = Field(default="false", description="AI 搜索辅助: true/false")
|
|
66
|
+
safe: str | None = Field(default="-1", description="成人内容过滤: 1 严格, -1 中等, -2 关闭")
|
|
67
|
+
df: str | None = Field(default="", description="日期过滤: d/w/m/y 或 start_date..end_date")
|
|
68
|
+
start: str | None = Field(default="0", description="结果偏移量")
|
|
69
|
+
m: str | None = Field(default="10", description="最大结果数 (1-50)")
|
|
70
|
+
no_cache: str | None = Field(default="false", description="是否跳过缓存")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
async def duckduckgo_search(self, params: DuckDuckGoSearchParams | None = None) -> Any:
|
|
74
|
+
"""Call ``duckduckgo_search`` — DuckDuckGo 网页搜索。
|
|
75
|
+
|
|
76
|
+
通过 DuckDuckGo 搜索公开网页信息,获取隐私搜索结果。
|
|
77
|
+
"""
|
|
78
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
79
|
+
return await self.call_tool("duckduckgo_search", arguments)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# ---------------------------------------------------------------------------
|
|
83
|
+
# Attach methods to DataifyClient
|
|
84
|
+
# ---------------------------------------------------------------------------
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _attach(client_cls: type) -> None:
|
|
88
|
+
"""Attach all other search engine methods to the client class."""
|
|
89
|
+
client_cls.yandex_search = yandex_search
|
|
90
|
+
client_cls.duckduckgo_search = duckduckgo_search
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Reddit scraper tools.
|
|
2
|
+
|
|
3
|
+
Wraps Reddit MCP tools: posts and comment.
|
|
4
|
+
|
|
5
|
+
Auto-generated. Regenerate with: python scripts/codegen.py
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, Field
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class _RedditBase(BaseModel):
|
|
16
|
+
file_name: str | None = Field(default="{{TasksID}}", description="Builder file_name")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ScrapeRedditPostsParams(_RedditBase):
|
|
20
|
+
"""Parameters for ``scrape_reddit_posts`` — Reddit 帖子采集。"""
|
|
21
|
+
url: str = Field(..., description="Reddit 帖子/子版块 URL")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ScrapeRedditCommentParams(_RedditBase):
|
|
25
|
+
"""Parameters for ``scrape_reddit_comment`` — Reddit 评论采集。"""
|
|
26
|
+
url: str = Field(..., description="Reddit 帖子 URL")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
async def scrape_reddit_posts(self, params: ScrapeRedditPostsParams | None = None) -> Any:
|
|
30
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
31
|
+
return await self.call_tool("scrape_reddit_posts", arguments)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
async def scrape_reddit_comment(self, params: ScrapeRedditCommentParams | None = None) -> Any:
|
|
35
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
36
|
+
return await self.call_tool("scrape_reddit_comment", arguments)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _attach(client_cls: type) -> None:
|
|
40
|
+
client_cls.scrape_reddit_posts = scrape_reddit_posts
|
|
41
|
+
client_cls.scrape_reddit_comment = scrape_reddit_comment
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""Task status and statistics tools.
|
|
2
|
+
|
|
3
|
+
Wraps MCP tools for querying web unlocker and scraper task status,
|
|
4
|
+
statistics, products, and tool lists.
|
|
5
|
+
|
|
6
|
+
Auto-generated. Regenerate with: python scripts/codegen.py
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel, Field
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# ---------------------------------------------------------------------------
|
|
17
|
+
# web_unlock_task
|
|
18
|
+
# ---------------------------------------------------------------------------
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class WebUnlockTaskParams(BaseModel):
|
|
22
|
+
"""Parameters for the ``web_unlock_task`` tool.
|
|
23
|
+
|
|
24
|
+
查询 Dataify 通用采集 API 的任务状态、消耗明细和用户累计统计。
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
keyword: str | None = Field(default="", description="搜索的任务 ID 或域名,可选")
|
|
28
|
+
status: int | None = Field(default=-1, description="任务状态: -1 全部,0 成功,1 失败")
|
|
29
|
+
page: int | None = Field(default=1, description="当前页,默认 1")
|
|
30
|
+
page_size: int | None = Field(default=10, description="每页数据量,默认 10,最大 100")
|
|
31
|
+
start: int | None = Field(default=None, description="查询时间范围开始时间,秒级时间戳")
|
|
32
|
+
end: int | None = Field(default=None, description="查询时间范围结束时间,秒级时间戳")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
async def web_unlock_task(self, params: WebUnlockTaskParams | None = None) -> Any:
|
|
36
|
+
"""Call ``web_unlock_task`` — 查询通用采集 API 任务状态。
|
|
37
|
+
|
|
38
|
+
查询 Dataify 通用采集 API 的任务状态、消耗明细和用户累计统计。
|
|
39
|
+
"""
|
|
40
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
41
|
+
# Map snake_case parameter names back to the server's parameter names
|
|
42
|
+
if "page_size" in arguments:
|
|
43
|
+
arguments["pageSize"] = arguments.pop("page_size")
|
|
44
|
+
return await self.call_tool("web_unlock_task", arguments)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# ---------------------------------------------------------------------------
|
|
48
|
+
# web_unlock_statistics
|
|
49
|
+
# ---------------------------------------------------------------------------
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class WebUnlockStatisticsParams(BaseModel):
|
|
53
|
+
"""Parameters for the ``web_unlock_statistics`` tool.
|
|
54
|
+
|
|
55
|
+
根据时间范围统计 Dataify 通用采集 API 的成功次数、失败次数、成功率等。
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
start: int | None = Field(default=None, description="查询时间范围开始时间,秒级时间戳")
|
|
59
|
+
end: int | None = Field(default=None, description="查询时间范围结束时间,秒级时间戳")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
async def web_unlock_statistics(self, params: WebUnlockStatisticsParams | None = None) -> Any:
|
|
63
|
+
"""Call ``web_unlock_statistics`` — 查询通用采集 API 统计数据。
|
|
64
|
+
|
|
65
|
+
根据时间范围统计 Dataify 通用采集 API 的成功、失败次数、成功率、
|
|
66
|
+
平均响应时间、P50/P90 响应时间和消耗积分。
|
|
67
|
+
"""
|
|
68
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
69
|
+
return await self.call_tool("web_unlock_statistics", arguments)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
# scraper_task_list
|
|
74
|
+
# ---------------------------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class ScraperTaskListParams(BaseModel):
|
|
78
|
+
"""Parameters for the ``scraper_task_list`` tool.
|
|
79
|
+
|
|
80
|
+
根据任务 ID、工具名称、类型、状态和时间范围查询网页采集与 SERP 采集任务。
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
keyword: str | None = Field(default="", description="按任务 ID 或工具名称查询,可选")
|
|
84
|
+
type: int | None = Field(default=0, description="类型: 0 全部,1 SERP 采集,2 网页采集")
|
|
85
|
+
status: int | None = Field(default=0, description="任务状态: 0 全部,-1 处理中,200 成功,400 失败")
|
|
86
|
+
start: int | None = Field(default=None, description="查询时间范围开始时间,秒级时间戳")
|
|
87
|
+
end: int | None = Field(default=None, description="查询时间范围结束时间,秒级时间戳")
|
|
88
|
+
page: int | None = Field(default=1, description="当前页,默认 1")
|
|
89
|
+
page_size: int | None = Field(default=10, description="每页数据量,默认 10,最大 100")
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
async def scraper_task_list(self, params: ScraperTaskListParams | None = None) -> Any:
|
|
93
|
+
"""Call ``scraper_task_list`` — 查询采集任务列表。
|
|
94
|
+
|
|
95
|
+
根据任务 ID、工具名称、类型、状态和时间范围查询 Dataify
|
|
96
|
+
网页采集与 SERP 采集任务。
|
|
97
|
+
"""
|
|
98
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
99
|
+
if "page_size" in arguments:
|
|
100
|
+
arguments["pageSize"] = arguments.pop("page_size")
|
|
101
|
+
return await self.call_tool("scraper_task_list", arguments)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
# ---------------------------------------------------------------------------
|
|
105
|
+
# scraper_statistics
|
|
106
|
+
# ---------------------------------------------------------------------------
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class ScraperStatisticsParams(BaseModel):
|
|
110
|
+
"""Parameters for the ``scraper_statistics`` tool.
|
|
111
|
+
|
|
112
|
+
根据时间范围统计网页采集和 SERP 采集的成功、失败次数和消耗积分。
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
start: int | None = Field(default=None, description="查询时间范围开始时间,秒级时间戳")
|
|
116
|
+
end: int | None = Field(default=None, description="查询时间范围结束时间,秒级时间戳")
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
async def scraper_statistics(self, params: ScraperStatisticsParams | None = None) -> Any:
|
|
120
|
+
"""Call ``scraper_statistics`` — 查询采集统计数据。
|
|
121
|
+
|
|
122
|
+
根据时间范围统计网页采集和 SERP 采集的成功、失败次数和消耗积分。
|
|
123
|
+
"""
|
|
124
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
125
|
+
return await self.call_tool("scraper_statistics", arguments)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
# ---------------------------------------------------------------------------
|
|
129
|
+
# scraper_serp_products
|
|
130
|
+
# ---------------------------------------------------------------------------
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
class ScraperSerpProductsParams(BaseModel):
|
|
134
|
+
"""Parameters for the ``scraper_serp_products`` tool.
|
|
135
|
+
|
|
136
|
+
查询 Dataify 网页采集和 SERP 的产品列表信息。
|
|
137
|
+
"""
|
|
138
|
+
|
|
139
|
+
scraper_type: int | None = Field(default=-1, description="产品类型: -1 全部,0 网页抓取,1 SERP")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
async def scraper_serp_products(self, params: ScraperSerpProductsParams | None = None) -> Any:
|
|
143
|
+
"""Call ``scraper_serp_products`` — 查询采集产品列表。
|
|
144
|
+
|
|
145
|
+
查询 Dataify 网页采集和 SERP 的产品列表信息。
|
|
146
|
+
"""
|
|
147
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
148
|
+
if "scraper_type" in arguments:
|
|
149
|
+
arguments["scraperType"] = arguments.pop("scraper_type")
|
|
150
|
+
return await self.call_tool("scraper_serp_products", arguments)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
# ---------------------------------------------------------------------------
|
|
154
|
+
# scraper_serp_tools
|
|
155
|
+
# ---------------------------------------------------------------------------
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
class ScraperSerpToolsParams(BaseModel):
|
|
159
|
+
"""Parameters for the ``scraper_serp_tools`` tool.
|
|
160
|
+
|
|
161
|
+
查询指定采集产品的工具列表。
|
|
162
|
+
"""
|
|
163
|
+
|
|
164
|
+
product_id: int = Field(..., description="采集产品 ID")
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
async def scraper_serp_tools(self, params: ScraperSerpToolsParams) -> Any:
|
|
168
|
+
"""Call ``scraper_serp_tools`` — 查询采集产品工具列表。
|
|
169
|
+
|
|
170
|
+
查询指定采集产品的工具列表。
|
|
171
|
+
"""
|
|
172
|
+
arguments = {"productId": params.product_id}
|
|
173
|
+
return await self.call_tool("scraper_serp_tools", arguments)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
# ---------------------------------------------------------------------------
|
|
177
|
+
# Attach methods to DataifyClient
|
|
178
|
+
# ---------------------------------------------------------------------------
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _attach(client_cls: type) -> None:
|
|
182
|
+
"""Attach all task status methods to the client class."""
|
|
183
|
+
client_cls.web_unlock_task = web_unlock_task
|
|
184
|
+
client_cls.web_unlock_statistics = web_unlock_statistics
|
|
185
|
+
client_cls.scraper_task_list = scraper_task_list
|
|
186
|
+
client_cls.scraper_statistics = scraper_statistics
|
|
187
|
+
client_cls.scraper_serp_products = scraper_serp_products
|
|
188
|
+
client_cls.scraper_serp_tools = scraper_serp_tools
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""TikTok scraper tools.
|
|
2
|
+
|
|
3
|
+
Wraps TikTok-related MCP tools: posts, profiles, comment, and shop.
|
|
4
|
+
|
|
5
|
+
Auto-generated. Regenerate with: python scripts/codegen.py
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, Field
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class _TikTokBase(BaseModel):
|
|
16
|
+
file_name: str | None = Field(default="{{TasksID}}", description="Builder file_name")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ScrapeTikTokPostsParams(_TikTokBase):
|
|
20
|
+
"""Parameters for ``scrape_tiktok_posts`` — TikTok 帖子信息采集。"""
|
|
21
|
+
|
|
22
|
+
url: str = Field(default="https://www.tiktok.com/discover/dog", description="TikTok 列表 URL")
|
|
23
|
+
num_of_posts: str | None = Field(default="5", description="收集的帖子数量")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class ScrapeTikTokProfilesParams(_TikTokBase):
|
|
27
|
+
"""Parameters for ``scrape_tiktok_profiles`` — TikTok 用户信息采集。"""
|
|
28
|
+
|
|
29
|
+
url: str = Field(..., description="TikTok 用户主页 URL")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class ScrapeTikTokCommentParams(_TikTokBase):
|
|
33
|
+
"""Parameters for ``scrape_tiktok_comment`` — TikTok 评论采集。"""
|
|
34
|
+
|
|
35
|
+
url: str = Field(..., description="TikTok 视频 URL")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ScrapeTikTokShopParams(_TikTokBase):
|
|
39
|
+
"""Parameters for ``scrape_tiktok_shop`` — TikTok Shop 采集。"""
|
|
40
|
+
|
|
41
|
+
url: str = Field(..., description="TikTok Shop 商品/店铺 URL")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _make_method(tool_name: str, desc: str, param_cls: type[BaseModel]):
|
|
45
|
+
async def _method(self, params: param_cls | None = None) -> Any: # type: ignore[valid-type]
|
|
46
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
47
|
+
return await self.call_tool(tool_name, arguments)
|
|
48
|
+
_method.__name__ = tool_name
|
|
49
|
+
_method.__doc__ = f"""Call ``{tool_name}`` — {desc}."""
|
|
50
|
+
return _method
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
async def scrape_tiktok_posts(self, params: ScrapeTikTokPostsParams | None = None) -> Any:
|
|
54
|
+
"""Call ``scrape_tiktok_posts`` — TikTok 帖子信息采集。"""
|
|
55
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
56
|
+
return await self.call_tool("scrape_tiktok_posts", arguments)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
async def scrape_tiktok_profiles(self, params: ScrapeTikTokProfilesParams | None = None) -> Any:
|
|
60
|
+
"""Call ``scrape_tiktok_profiles`` — TikTok 用户信息采集。"""
|
|
61
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
62
|
+
return await self.call_tool("scrape_tiktok_profiles", arguments)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
async def scrape_tiktok_comment(self, params: ScrapeTikTokCommentParams | None = None) -> Any:
|
|
66
|
+
"""Call ``scrape_tiktok_comment`` — TikTok 评论采集。"""
|
|
67
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
68
|
+
return await self.call_tool("scrape_tiktok_comment", arguments)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
async def scrape_tiktok_shop(self, params: ScrapeTikTokShopParams | None = None) -> Any:
|
|
72
|
+
"""Call ``scrape_tiktok_shop`` — TikTok Shop 采集。"""
|
|
73
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
74
|
+
return await self.call_tool("scrape_tiktok_shop", arguments)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _attach(client_cls: type) -> None:
|
|
78
|
+
client_cls.scrape_tiktok_posts = scrape_tiktok_posts
|
|
79
|
+
client_cls.scrape_tiktok_profiles = scrape_tiktok_profiles
|
|
80
|
+
client_cls.scrape_tiktok_comment = scrape_tiktok_comment
|
|
81
|
+
client_cls.scrape_tiktok_shop = scrape_tiktok_shop
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Twitter/X scraper tools.
|
|
2
|
+
|
|
3
|
+
Wraps Twitter/X MCP tools: post and profile.
|
|
4
|
+
|
|
5
|
+
Auto-generated. Regenerate with: python scripts/codegen.py
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, Field
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class _TwitterBase(BaseModel):
|
|
16
|
+
file_name: str | None = Field(default="{{TasksID}}", description="Builder file_name")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ScrapeTwitterPostParams(_TwitterBase):
|
|
20
|
+
"""Parameters for ``scrape_twitter_post`` — Twitter/X 帖子采集。"""
|
|
21
|
+
url: str = Field(..., description="Twitter/X 帖子 URL")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ScrapeTwitterProfileParams(_TwitterBase):
|
|
25
|
+
"""Parameters for ``scrape_twitter_profile`` — Twitter/X 用户信息采集。"""
|
|
26
|
+
url: str = Field(..., description="Twitter/X 用户主页 URL")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
async def scrape_twitter_post(self, params: ScrapeTwitterPostParams | None = None) -> Any:
|
|
30
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
31
|
+
return await self.call_tool("scrape_twitter_post", arguments)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
async def scrape_twitter_profile(self, params: ScrapeTwitterProfileParams | None = None) -> Any:
|
|
35
|
+
arguments = params.model_dump(exclude_none=True) if params else {}
|
|
36
|
+
return await self.call_tool("scrape_twitter_profile", arguments)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _attach(client_cls: type) -> None:
|
|
40
|
+
client_cls.scrape_twitter_post = scrape_twitter_post
|
|
41
|
+
client_cls.scrape_twitter_profile = scrape_twitter_profile
|