goofish-cli 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/.claude-plugin/marketplace.json +1 -1
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/CHANGELOG.md +16 -1
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/PKG-INFO +4 -3
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/README.md +3 -2
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/pyproject.toml +1 -1
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/auth/login.py +33 -23
- goofish_cli-0.4.0/src/goofish_cli/commands/search/search.py +376 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/browser.py +35 -28
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/browser_cookie.py +30 -13
- goofish_cli-0.4.0/src/goofish_cli/core/cookie_types.py +10 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/crypto.py +4 -4
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/qr_login.py +22 -15
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/refresh.py +38 -18
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/session.py +62 -32
- goofish_cli-0.3.0/src/goofish_cli/commands/search/search.py +0 -169
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/.gitignore +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/LICENSE +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/NOTICE +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/examples/README.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-overview/SKILL.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-overview/references/accounts.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-overview/references/mcp-tools-index.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-overview/references/xianyu-concepts.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-publish-item/SKILL.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-publish-item/references/category-selection.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-publish-item/references/common-pitfalls.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-publish-item/references/image-checklist.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-publish-item/references/title-formula.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-reply-buyer/SKILL.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-reply-buyer/references/bargain-ladder.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-reply-buyer/references/intent-classification.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-reply-buyer/references/tone-guide.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-risk-guard/SKILL.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-risk-guard/references/external-contact-keywords.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-risk-guard/references/forbidden-words.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-risk-guard/references/publish-red-lines.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-risk-guard/references/x5sec-recovery.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-shop-diagnosis/SKILL.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-shop-diagnosis/references/diagnosis-workflow.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-shop-diagnosis/references/rank-factors.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/skills/goofish-shop-diagnosis/references/traffic-drop-causes.md +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/cli.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/auth/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/auth/reset_guard.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/auth/status.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/category/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/category/recommend.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/item/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/item/delete.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/item/get.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/item/list.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/item/publish.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/item/view.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/location/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/location/default.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/media/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/media/upload.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/message/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/message/history.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/message/list_chats.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/message/send.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/message/watch.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/search/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/skills/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/commands/skills/install.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/__init__.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/errors.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/guard.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/limiter.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/mtop.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/output.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/registry.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/sign.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/strategy.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/token.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/core/ws.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/mcp_server.py +0 -0
- {goofish_cli-0.3.0 → goofish_cli-0.4.0}/src/goofish_cli/static/goofish_js_version_2.js +0 -0
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"name": "goofish-cli-skills",
|
|
4
4
|
"displayName": "goofish-cli Skills",
|
|
5
5
|
"description": "闲鱼(goofish.com)自动化运营的 Claude Skills 套件。配合 goofish-cli MCP server 使用,让 Agent 在发布商品 / 回复买家 / 风控预检 / 店铺诊断这些场景里上手即会。",
|
|
6
|
-
"version": "0.
|
|
6
|
+
"version": "0.4.0",
|
|
7
7
|
"author": {
|
|
8
8
|
"name": "fancy",
|
|
9
9
|
"url": "https://github.com/fancyboi999"
|
|
@@ -6,6 +6,20 @@ and this project adheres to [Semantic Versioning](https://semver.org/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.4.0] - 2026-09-07
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
- 搜索支持 `goofish search items --pages N`,跨页按商品 ID 去重;默认仍读取一页,
|
|
13
|
+
`--limit` 控制所有页面的结果总数。返回 `pages_fetched`、`pages_total` 和
|
|
14
|
+
`stopped_reason`,说明实际翻页数量与停止原因。翻页过程中发生错误或遇到登录阻拦时,
|
|
15
|
+
保留已经取得的结果。
|
|
16
|
+
|
|
17
|
+
### Fixed
|
|
18
|
+
- 浏览器导入、JSON 导入、扫码登录和 Cookie 刷新保留 Cookie 的 domain/path,
|
|
19
|
+
防止同名 Cookie 覆盖或注入到错误域名、路径;兼容旧版扁平 Cookie 数据。
|
|
20
|
+
- 修复 Cookie 刷新在 `persist=False` 时的异常,并保持内存与持久化记录一致。
|
|
21
|
+
- 恢复搜索成功时的结构化返回;仅在首屏没有结果且明确要求登录时重试一次。
|
|
22
|
+
|
|
9
23
|
## [0.3.0] - 2026-08-19
|
|
10
24
|
|
|
11
25
|
### Added
|
|
@@ -182,7 +196,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/).
|
|
|
182
196
|
- 本版本需要用户手动从浏览器导入 cookie(含 `unb` / `_m_h5_tk` / `x5sec`)
|
|
183
197
|
- 遇到 `RGV587_ERROR` 风控时,需在浏览器完成滑块验证并**重新导出**带 `x5sec` 的 cookie
|
|
184
198
|
|
|
185
|
-
[Unreleased]: https://github.com/fancyboi999/goofish-cli/compare/v0.
|
|
199
|
+
[Unreleased]: https://github.com/fancyboi999/goofish-cli/compare/v0.4.0...HEAD
|
|
200
|
+
[0.4.0]: https://github.com/fancyboi999/goofish-cli/compare/v0.3.0...v0.4.0
|
|
186
201
|
[0.3.0]: https://github.com/fancyboi999/goofish-cli/compare/v0.2.4...v0.3.0
|
|
187
202
|
[0.2.4]: https://github.com/fancyboi999/goofish-cli/compare/v0.2.3...v0.2.4
|
|
188
203
|
[0.2.3]: https://github.com/fancyboi999/goofish-cli/compare/v0.2.2...v0.2.3
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: goofish-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: 闲鱼 CLI — 支持 MCP,未来支持 Claude Skills。为 AI Agent 提供闲鱼自动化基础能力。
|
|
5
5
|
Project-URL: Homepage, https://github.com/fancyboi999/goofish-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/fancyboi999/goofish-cli
|
|
@@ -153,13 +153,14 @@ claude /plugin marketplace add fancyboi999/goofish-cli
|
|
|
153
153
|
|
|
154
154
|
## OpenClaw / ClawHub
|
|
155
155
|
|
|
156
|
-
OpenClaw `2026.6.1` 及以上可把本仓库作为 compatible bundle
|
|
156
|
+
OpenClaw `2026.6.1` 及以上可把本仓库作为 compatible bundle 加载。已发布到
|
|
157
|
+
[ClawHub](https://clawhub.ai/plugins/openclaw-goofish),安装:
|
|
157
158
|
|
|
158
159
|
```bash
|
|
159
160
|
openclaw plugins install clawhub:openclaw-goofish
|
|
160
161
|
|
|
161
162
|
# 登录态由用户在终端初始化,不交给 Agent 覆盖
|
|
162
|
-
uvx --from goofish-cli==0.
|
|
163
|
+
uvx --from goofish-cli==0.4.0 goofish auth login --qr
|
|
163
164
|
|
|
164
165
|
openclaw plugins inspect goofish --json
|
|
165
166
|
openclaw gateway restart
|
|
@@ -113,13 +113,14 @@ claude /plugin marketplace add fancyboi999/goofish-cli
|
|
|
113
113
|
|
|
114
114
|
## OpenClaw / ClawHub
|
|
115
115
|
|
|
116
|
-
OpenClaw `2026.6.1` 及以上可把本仓库作为 compatible bundle
|
|
116
|
+
OpenClaw `2026.6.1` 及以上可把本仓库作为 compatible bundle 加载。已发布到
|
|
117
|
+
[ClawHub](https://clawhub.ai/plugins/openclaw-goofish),安装:
|
|
117
118
|
|
|
118
119
|
```bash
|
|
119
120
|
openclaw plugins install clawhub:openclaw-goofish
|
|
120
121
|
|
|
121
122
|
# 登录态由用户在终端初始化,不交给 Agent 覆盖
|
|
122
|
-
uvx --from goofish-cli==0.
|
|
123
|
+
uvx --from goofish-cli==0.4.0 goofish auth login --qr
|
|
123
124
|
|
|
124
125
|
openclaw plugins inspect goofish --json
|
|
125
126
|
openclaw gateway restart
|
|
@@ -23,8 +23,14 @@ import json
|
|
|
23
23
|
from pathlib import Path
|
|
24
24
|
|
|
25
25
|
from goofish_cli.core import Strategy, command
|
|
26
|
+
from goofish_cli.core.cookie_types import CookieRecord
|
|
26
27
|
from goofish_cli.core.errors import AuthRequiredError
|
|
27
|
-
from goofish_cli.core.session import
|
|
28
|
+
from goofish_cli.core.session import (
|
|
29
|
+
DEFAULT_COOKIE_PATH,
|
|
30
|
+
_coerce_records,
|
|
31
|
+
_records_to_flat,
|
|
32
|
+
write_cookies_json,
|
|
33
|
+
)
|
|
28
34
|
|
|
29
35
|
|
|
30
36
|
@command(
|
|
@@ -56,16 +62,8 @@ def login(
|
|
|
56
62
|
# qr_timeout=None 让 core.qr_login 走统一的 env → 默认值 兜底逻辑;
|
|
57
63
|
# 这里若写 int 默认(例如 120)会把 env 覆盖路径挡掉(CLI 总是显式传值)。
|
|
58
64
|
from goofish_cli.core.qr_login import login_via_qr
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
# 空 dict 可能来自两种失败:超时未扫码,或 Playwright 起不来(Chrome
|
|
62
|
-
# 未装、端口占用等)。文案同时覆盖,让用户知道去翻日志。
|
|
63
|
-
raise AuthRequiredError(
|
|
64
|
-
"QR 扫码登录未完成——可能是超时内未扫码 / 手机未确认,"
|
|
65
|
-
"也可能是 Playwright 浏览器启动失败(Chrome 未装、端口占用等,"
|
|
66
|
-
"详见前面的 warning 日志)。可重试并延长超时:"
|
|
67
|
-
"goofish auth login --qr --qr-timeout 180"
|
|
68
|
-
)
|
|
65
|
+
|
|
66
|
+
records = login_via_qr(timeout=qr_timeout, persist=False)
|
|
69
67
|
source_label = "qr"
|
|
70
68
|
elif source is None:
|
|
71
69
|
if raw:
|
|
@@ -75,33 +73,35 @@ def login(
|
|
|
75
73
|
"--raw 需要配合 cookie 字符串使用,如:"
|
|
76
74
|
"goofish auth login 'unb=...; _m_h5_tk=...' --raw"
|
|
77
75
|
)
|
|
78
|
-
|
|
76
|
+
records, source_label = _pull_from_browser(browser)
|
|
79
77
|
elif raw:
|
|
80
|
-
|
|
78
|
+
records = _coerce_records(_parse_raw(source))
|
|
81
79
|
source_label = "raw"
|
|
82
80
|
else:
|
|
83
81
|
p = Path(source).expanduser()
|
|
84
|
-
|
|
82
|
+
records = _parse_json(p.read_text())
|
|
85
83
|
source_label = f"file:{p}"
|
|
86
84
|
|
|
87
|
-
|
|
85
|
+
# 校验用 flat 视角
|
|
86
|
+
flat = _records_to_flat(records)
|
|
87
|
+
if "unb" not in flat or "_m_h5_tk" not in flat:
|
|
88
88
|
raise AuthRequiredError(
|
|
89
89
|
"cookie 缺失关键字段 unb / _m_h5_tk。"
|
|
90
90
|
"请先在浏览器里登录 https://www.goofish.com 再试。"
|
|
91
91
|
)
|
|
92
92
|
|
|
93
|
-
write_cookies_json(target,
|
|
93
|
+
write_cookies_json(target, records)
|
|
94
94
|
|
|
95
95
|
return {
|
|
96
96
|
"source": source_label,
|
|
97
97
|
"path": str(target),
|
|
98
|
-
"unb":
|
|
99
|
-
"tracknick":
|
|
100
|
-
"cookies_count": len(
|
|
98
|
+
"unb": flat.get("unb", ""),
|
|
99
|
+
"tracknick": flat.get("tracknick", ""),
|
|
100
|
+
"cookies_count": len(flat),
|
|
101
101
|
}
|
|
102
102
|
|
|
103
103
|
|
|
104
|
-
def _pull_from_browser(browser: str) -> tuple[
|
|
104
|
+
def _pull_from_browser(browser: str) -> tuple[list[CookieRecord], str]:
|
|
105
105
|
from goofish_cli.core.browser_cookie import (
|
|
106
106
|
BrowserCookieError,
|
|
107
107
|
available_browsers,
|
|
@@ -133,10 +133,20 @@ def _parse_raw(raw: str) -> dict[str, str]:
|
|
|
133
133
|
return out
|
|
134
134
|
|
|
135
135
|
|
|
136
|
-
def _parse_json(text: str) ->
|
|
136
|
+
def _parse_json(text: str) -> list[CookieRecord]:
|
|
137
|
+
"""解析导入文件。list-form JSON → records(保留 path/domain);dict → coerce。"""
|
|
137
138
|
data = json.loads(text)
|
|
138
139
|
if isinstance(data, list):
|
|
139
|
-
return
|
|
140
|
+
return [
|
|
141
|
+
{
|
|
142
|
+
"name": c["name"],
|
|
143
|
+
"value": c["value"],
|
|
144
|
+
"domain": c.get("domain", ""),
|
|
145
|
+
"path": c.get("path", ""),
|
|
146
|
+
}
|
|
147
|
+
for c in data
|
|
148
|
+
if "name" in c and "value" in c
|
|
149
|
+
]
|
|
140
150
|
if isinstance(data, dict):
|
|
141
|
-
return
|
|
151
|
+
return _coerce_records(data)
|
|
142
152
|
raise AuthRequiredError("cookie JSON 格式不识别(需 list 或 dict)")
|
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
"""search — 搜索闲鱼商品。对标 OpenCLI `xianyu/search.js`。
|
|
2
|
+
|
|
3
|
+
思路:打开 `https://www.goofish.com/search?q=xxx` 让页面自己渲染,autoScroll 触发
|
|
4
|
+
懒加载,再在 page context 里跑 DOM 选择器提卡片。**不走 mtop 直签**:
|
|
5
|
+
- search 没对外 API,只有 HTML 卡片 + 动态加载
|
|
6
|
+
- 浏览器真实渲染天然抗风控
|
|
7
|
+
|
|
8
|
+
分页(2026-08 实测):搜索页**不是无限滚动**——窄查询滚动到底卡片数不再增长;
|
|
9
|
+
翻页靠 DOM 里的 `search-pagination-container`(宽查询下 1..50 页),点击右箭头后
|
|
10
|
+
SPA 内部重渲染、URL 不变,`?page=N` URL 参数被服务端忽略。所以跨页抓取 = 点击
|
|
11
|
+
翻页箭头 + 按 item_id 去重累积,终止条件:达到 --pages 上限 / 右箭头 disabled /
|
|
12
|
+
连续一页无新增(防御)。
|
|
13
|
+
|
|
14
|
+
字段参考 OpenCLI:`item_id / rank / title / price / original_price / condition /
|
|
15
|
+
brand / location / badge / url / extra`。
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import asyncio
|
|
20
|
+
import re
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from goofish_cli.core import Strategy, command
|
|
24
|
+
from goofish_cli.core.browser import auto_scroll, goofish_page
|
|
25
|
+
from goofish_cli.core.errors import AuthRequiredError, GoofishError
|
|
26
|
+
|
|
27
|
+
# limit 是**跨页总上限**(去重后条数)。站点每页 30 卡,50 页满配远超此值,
|
|
28
|
+
# 200 是给 MCP 调用方的运行时护栏(每页 ~4s,200 条 ≈ 7 页 ≈ 40s)。
|
|
29
|
+
MAX_LIMIT = 200
|
|
30
|
+
DEFAULT_LIMIT = 20
|
|
31
|
+
MAX_PAGES = 50 # 与站点分页控件一致(实测 1..50)
|
|
32
|
+
# 0 卡片 + 非 empty/blocked 时判定为疑似瞬时登录墙(同一 cookie 注入实测时好时坏),
|
|
33
|
+
# 总尝试次数(含首次)。重试前短暂退避,避免连续两枪都打在风控瞬间。
|
|
34
|
+
AUTH_WALL_ATTEMPTS = 2
|
|
35
|
+
# 点击翻页后的兜底等待;主等待靠"首卡片 id 变化"事件,这里只兜渲染卡住的底
|
|
36
|
+
PAGE_STABLE_MS = 2000
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _should_retry(payload: dict[str, Any]) -> bool:
|
|
40
|
+
"""零卡片且明确 requiresAuth 时才重试(瞬时登录墙兜底)。
|
|
41
|
+
|
|
42
|
+
#28 合并后的契约:其他零卡片形态(未知页面结构等)不重试,
|
|
43
|
+
直接交给 `_raise_for_failed_page` 按语义抛错。
|
|
44
|
+
"""
|
|
45
|
+
return not payload.get("items") and bool(payload.get("requiresAuth"))
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _normalize_limit(value: Any) -> int:
|
|
49
|
+
try:
|
|
50
|
+
n = int(value)
|
|
51
|
+
except (TypeError, ValueError):
|
|
52
|
+
return DEFAULT_LIMIT
|
|
53
|
+
return min(MAX_LIMIT, max(1, n))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _normalize_pages(value: Any) -> int:
|
|
57
|
+
try:
|
|
58
|
+
n = int(value)
|
|
59
|
+
except (TypeError, ValueError):
|
|
60
|
+
return 1
|
|
61
|
+
return min(MAX_PAGES, max(1, n))
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _item_id_from_url(url: str) -> str:
|
|
65
|
+
m = re.search(r"[?&]id=(\d+)", url or "")
|
|
66
|
+
return m.group(1) if m else ""
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _build_search_url(query: str) -> str:
|
|
70
|
+
from urllib.parse import quote
|
|
71
|
+
return f"https://www.goofish.com/search?q={quote(query)}"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
# 页面上下文里跑的 JS。抽出来做常量方便测试(`__test__` 导出)。
|
|
75
|
+
_EXTRACT_JS = r"""
|
|
76
|
+
(limit) => (async () => {
|
|
77
|
+
const wait = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
|
|
78
|
+
const waitFor = async (predicate, timeoutMs = 8000) => {
|
|
79
|
+
const start = Date.now();
|
|
80
|
+
while (Date.now() - start < timeoutMs) {
|
|
81
|
+
if (predicate()) return true;
|
|
82
|
+
await wait(150);
|
|
83
|
+
}
|
|
84
|
+
return false;
|
|
85
|
+
};
|
|
86
|
+
|
|
87
|
+
const clean = (v) => (v || '').replace(/\s+/g, ' ').trim();
|
|
88
|
+
const sel = {
|
|
89
|
+
card: 'a[href*="/item?id="]',
|
|
90
|
+
title: '[class*="row1-wrap-title"], [class*="main-title"]',
|
|
91
|
+
attrs: '[class*="row2-wrap-cpv"] span[class*="cpv--"]',
|
|
92
|
+
priceWrap: '[class*="price-wrap"]',
|
|
93
|
+
priceNum: '[class*="number"]',
|
|
94
|
+
priceDec: '[class*="decimal"]',
|
|
95
|
+
priceDesc: '[class*="price-desc"] [title], [class*="price-desc"] [style*="line-through"]',
|
|
96
|
+
sellerWrap: '[class*="row4-wrap-seller"]',
|
|
97
|
+
sellerText: '[class*="seller-text"]',
|
|
98
|
+
badge: '[class*="credit-container"] [title], [class*="credit-container"] span',
|
|
99
|
+
};
|
|
100
|
+
|
|
101
|
+
await waitFor(() => {
|
|
102
|
+
const bodyText = document.body?.innerText || '';
|
|
103
|
+
return Boolean(
|
|
104
|
+
document.querySelector(sel.card)
|
|
105
|
+
|| /请先登录|登录后|验证码|安全验证|异常访问/.test(bodyText)
|
|
106
|
+
|| /暂无相关宝贝|未找到相关宝贝|没有找到/.test(bodyText)
|
|
107
|
+
);
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
const bodyText = document.body?.innerText || '';
|
|
111
|
+
const requiresAuth = /请先登录|登录后/.test(bodyText);
|
|
112
|
+
const blocked = /验证码|安全验证|异常访问/.test(bodyText);
|
|
113
|
+
const empty = /暂无相关宝贝|未找到相关宝贝|没有找到/.test(bodyText);
|
|
114
|
+
|
|
115
|
+
const items = Array.from(document.querySelectorAll(sel.card))
|
|
116
|
+
.slice(0, limit)
|
|
117
|
+
.map((card) => {
|
|
118
|
+
const href = card.href || card.getAttribute('href') || '';
|
|
119
|
+
const title = clean(card.querySelector(sel.title)?.textContent || '');
|
|
120
|
+
const attrs = Array.from(card.querySelectorAll(sel.attrs))
|
|
121
|
+
.map((n) => clean(n.textContent || ''))
|
|
122
|
+
.filter(Boolean);
|
|
123
|
+
const priceWrap = card.querySelector(sel.priceWrap);
|
|
124
|
+
const priceNumber = clean(priceWrap?.querySelector(sel.priceNum)?.textContent || '');
|
|
125
|
+
const priceDecimal = clean(priceWrap?.querySelector(sel.priceDec)?.textContent || '');
|
|
126
|
+
const location = clean(card.querySelector(sel.sellerWrap)?.querySelector(sel.sellerText)?.textContent || '');
|
|
127
|
+
const originalPriceNode = card.querySelector(sel.priceDesc);
|
|
128
|
+
const badgeNode = card.querySelector(sel.badge);
|
|
129
|
+
|
|
130
|
+
return {
|
|
131
|
+
title,
|
|
132
|
+
url: href,
|
|
133
|
+
price: clean('¥' + priceNumber + priceDecimal).replace(/^¥\s*$/, ''),
|
|
134
|
+
original_price: clean(originalPriceNode?.getAttribute('title') || originalPriceNode?.textContent || ''),
|
|
135
|
+
condition: attrs[0] || '',
|
|
136
|
+
brand: attrs[1] || '',
|
|
137
|
+
extra: attrs.slice(2).join(' | '),
|
|
138
|
+
location,
|
|
139
|
+
badge: clean(badgeNode?.getAttribute('title') || badgeNode?.textContent || ''),
|
|
140
|
+
};
|
|
141
|
+
})
|
|
142
|
+
.filter((it) => it.title && it.url);
|
|
143
|
+
|
|
144
|
+
return { requiresAuth, blocked, empty, items, bodyPreview: bodyText.slice(0, 500) };
|
|
145
|
+
})()
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
# 翻页控件状态:右箭头是否可用 + 分页盒里的最大页码(实测 1..50,"..." 是非数字盒)
|
|
149
|
+
_PAGINATION_JS = r"""
|
|
150
|
+
() => {
|
|
151
|
+
const boxes = [...document.querySelectorAll('[class*="page-box"]')]
|
|
152
|
+
.map((e) => (e.textContent || '').trim())
|
|
153
|
+
.filter((t) => /^\d+$/.test(t))
|
|
154
|
+
.map(Number);
|
|
155
|
+
const arrows = [...document.querySelectorAll('[class*="pagination-arrow-container"]')];
|
|
156
|
+
const right = arrows.length ? arrows[arrows.length - 1] : null;
|
|
157
|
+
return {
|
|
158
|
+
hasNext: Boolean(right) && !right.hasAttribute('disabled'),
|
|
159
|
+
totalPages: boxes.length ? Math.max(...boxes) : null,
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
"""
|
|
163
|
+
|
|
164
|
+
# 翻页是 SPA 重渲染(URL 不变)。点完右箭头等"首卡片 id 变化",最多 8s;
|
|
165
|
+
# 超时返回 false(大概率是最后一页重渲染失败或内容未变,交给上层去重兜底)。
|
|
166
|
+
_WAIT_PAGE_CHANGE_JS = r"""
|
|
167
|
+
(prevFirstId) => new Promise((resolve) => {
|
|
168
|
+
const start = Date.now();
|
|
169
|
+
const tick = () => {
|
|
170
|
+
const first = document.querySelector('a[href*="/item?id="]');
|
|
171
|
+
const id = first ? ((first.href || '').match(/id=(\d+)/) || [])[1] : null;
|
|
172
|
+
if (id && id !== prevFirstId) return resolve(true);
|
|
173
|
+
if (Date.now() - start > 8000) return resolve(false);
|
|
174
|
+
setTimeout(tick, 200);
|
|
175
|
+
};
|
|
176
|
+
tick();
|
|
177
|
+
})
|
|
178
|
+
"""
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _first_card_id(items: list[dict[str, Any]]) -> str:
|
|
182
|
+
"""当前页首卡 id——翻页等待的"内容已变化"基准。"""
|
|
183
|
+
for it in items:
|
|
184
|
+
item_id = _item_id_from_url(it.get("url", ""))
|
|
185
|
+
if item_id:
|
|
186
|
+
return item_id
|
|
187
|
+
return ""
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _raise_for_failed_page(payload: dict[str, Any]) -> None:
|
|
191
|
+
"""首页级失败(0 卡片)按原语义抛错;翻页途中的失败由调用方优雅终止。"""
|
|
192
|
+
items = payload.get("items") or []
|
|
193
|
+
# "登录后" 在页脚也会出现——只有在"没拿到卡片 && 命中关键词"时才判定 auth 失败
|
|
194
|
+
if not items and payload.get("requiresAuth"):
|
|
195
|
+
raise AuthRequiredError("www.goofish.com 搜索结果页要求登录,cookies 可能失效")
|
|
196
|
+
if not items and payload.get("blocked"):
|
|
197
|
+
raise GoofishError("搜索页返回验证码/安全验证(触发风控),稍后重试或换账号")
|
|
198
|
+
if not items and not payload.get("empty"):
|
|
199
|
+
preview = (payload.get("bodyPreview") or "")[:200]
|
|
200
|
+
raise GoofishError(
|
|
201
|
+
f"未在搜索页上解析到任何卡片,可能 DOM 结构已变。"
|
|
202
|
+
f"页面文案预览:{preview!r}"
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
async def _walk_pages(
|
|
207
|
+
page: Any,
|
|
208
|
+
items: list[dict[str, Any]],
|
|
209
|
+
seen: set[str],
|
|
210
|
+
fetched_pages: int,
|
|
211
|
+
pages: int,
|
|
212
|
+
limit: int,
|
|
213
|
+
) -> tuple[int, int | None, str]:
|
|
214
|
+
"""翻页状态机:点右箭头 → 等重渲染 → 去重累积。
|
|
215
|
+
|
|
216
|
+
返回 (fetched_pages, total_pages, stopped_reason)。任何翻页途中的
|
|
217
|
+
Playwright 异常(SPA 重渲染销毁执行上下文等)都被吞掉并优雅终止——
|
|
218
|
+
调用方拿到已累积的部分结果,stopped_reason 说明终止原因。
|
|
219
|
+
|
|
220
|
+
锚点(cur_first_id)始终取**未过滤**的下一页 payload 首卡:
|
|
221
|
+
fresh 是去重后的列表,若下一页首卡与上一页重复,它会被过滤掉,
|
|
222
|
+
用 fresh 的首卡做锚点会与 DOM 实况脱节,导致 page-change 等待
|
|
223
|
+
假阳性、提前终止(review P1-2)。
|
|
224
|
+
"""
|
|
225
|
+
anchor_id = _first_card_id(items) or ""
|
|
226
|
+
total_pages: int | None = None
|
|
227
|
+
stopped_reason = "pages_reached"
|
|
228
|
+
|
|
229
|
+
while fetched_pages < pages and len(items) < limit:
|
|
230
|
+
try:
|
|
231
|
+
pag = await page.evaluate(_PAGINATION_JS)
|
|
232
|
+
except Exception: # noqa: BLE001 — 执行上下文销毁等,保留已抓结果
|
|
233
|
+
return fetched_pages, total_pages, "error"
|
|
234
|
+
# 结构突变(None / 非dict)同样按优雅终止处理,不能 AttributeError 穿透
|
|
235
|
+
if not isinstance(pag, dict):
|
|
236
|
+
return fetched_pages, total_pages, "error"
|
|
237
|
+
|
|
238
|
+
if total_pages is None and pag.get("totalPages") is not None:
|
|
239
|
+
total_pages = pag["totalPages"]
|
|
240
|
+
if not pag.get("hasNext"):
|
|
241
|
+
return fetched_pages, total_pages, "last_page"
|
|
242
|
+
|
|
243
|
+
try:
|
|
244
|
+
arrow = page.locator('[class*="pagination-arrow-container"]').nth(1)
|
|
245
|
+
await arrow.click(timeout=5000)
|
|
246
|
+
# 等待"首卡 id 不再等于上一页 DOM 首卡"。changed=False 是等待超时
|
|
247
|
+
# (重渲染大概率失败);先看提取结果再定性。
|
|
248
|
+
changed = await page.evaluate(_WAIT_PAGE_CHANGE_JS, anchor_id)
|
|
249
|
+
await page.wait_for_timeout(PAGE_STABLE_MS)
|
|
250
|
+
nxt = await page.evaluate(_EXTRACT_JS, MAX_LIMIT)
|
|
251
|
+
except Exception: # noqa: BLE001 — 同上,部分结果优先
|
|
252
|
+
return fetched_pages, total_pages, "error"
|
|
253
|
+
if not isinstance(nxt, dict):
|
|
254
|
+
return fetched_pages, total_pages, "error"
|
|
255
|
+
|
|
256
|
+
# 锚点更新为**未过滤** payload 的首卡(= 本页 DOM 实际首卡)。
|
|
257
|
+
# fresh 是去重后的列表:若本页首卡与上一页重复,会被过滤掉,
|
|
258
|
+
# 用 fresh 的首卡会让下一轮等待基准与 DOM �脱节(review P1-2)。
|
|
259
|
+
nxt_items = nxt.get("items") or []
|
|
260
|
+
anchor_id = _first_card_id(nxt_items) or anchor_id
|
|
261
|
+
# item_id 是输出的稳定主键(调用方按它去重/取详情)。解析不出数字 id
|
|
262
|
+
# 的卡片(?id=abc 等)跨页会重复追加(seen 只记非空 id),直接跳过。
|
|
263
|
+
nxt_items = [it for it in nxt_items if _item_id_from_url(it.get("url", ""))]
|
|
264
|
+
fresh = [
|
|
265
|
+
it for it in nxt_items if _item_id_from_url(it.get("url", "")) not in seen
|
|
266
|
+
]
|
|
267
|
+
if not fresh:
|
|
268
|
+
if not nxt_items and nxt.get("blocked"):
|
|
269
|
+
return fetched_pages, total_pages, "blocked"
|
|
270
|
+
return fetched_pages, total_pages, "no_new" if changed else "stale"
|
|
271
|
+
|
|
272
|
+
for it in fresh:
|
|
273
|
+
item_id = _item_id_from_url(it.get("url", ""))
|
|
274
|
+
if item_id:
|
|
275
|
+
seen.add(item_id)
|
|
276
|
+
items.append(it)
|
|
277
|
+
fetched_pages += 1
|
|
278
|
+
|
|
279
|
+
if len(items) >= limit:
|
|
280
|
+
return fetched_pages, total_pages, "limit"
|
|
281
|
+
return fetched_pages, total_pages, stopped_reason
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
async def _run(query: str, limit: int, pages: int) -> dict[str, Any]:
|
|
285
|
+
url = _build_search_url(query)
|
|
286
|
+
|
|
287
|
+
items: list[dict[str, Any]] = []
|
|
288
|
+
seen: set[str] = set()
|
|
289
|
+
fetched_pages = 0
|
|
290
|
+
total_pages: int | None = None
|
|
291
|
+
|
|
292
|
+
async with goofish_page() as page:
|
|
293
|
+
# ---- 第 1 页(保留瞬时登录墙重试:整页重新导航)----
|
|
294
|
+
payload: dict[str, Any] | None = None
|
|
295
|
+
for attempt in range(1, AUTH_WALL_ATTEMPTS + 1):
|
|
296
|
+
await page.goto(url, wait_until="domcontentloaded")
|
|
297
|
+
await page.wait_for_timeout(2000)
|
|
298
|
+
await auto_scroll(page, times=2)
|
|
299
|
+
raw = await page.evaluate(_EXTRACT_JS, MAX_LIMIT)
|
|
300
|
+
if not isinstance(raw, dict):
|
|
301
|
+
raise GoofishError("搜索页返回结构非预期")
|
|
302
|
+
payload = raw
|
|
303
|
+
# 拿到卡片 / 明确的空结果 / 明确的风控页都不需要重试
|
|
304
|
+
if not _should_retry(raw):
|
|
305
|
+
break
|
|
306
|
+
if attempt < AUTH_WALL_ATTEMPTS:
|
|
307
|
+
await asyncio.sleep(1.5)
|
|
308
|
+
|
|
309
|
+
assert payload is not None
|
|
310
|
+
_raise_for_failed_page(payload)
|
|
311
|
+
|
|
312
|
+
# item_id 是输出的稳定主键:解析不出数字 id 的卡片直接跳过,
|
|
313
|
+
# 否则同一张坏卡跨页会重复追加(seen 只记非空 id)
|
|
314
|
+
for it in payload.get("items") or []:
|
|
315
|
+
item_id = _item_id_from_url(it.get("url", ""))
|
|
316
|
+
if not item_id:
|
|
317
|
+
continue
|
|
318
|
+
if item_id in seen:
|
|
319
|
+
continue
|
|
320
|
+
seen.add(item_id)
|
|
321
|
+
items.append(it)
|
|
322
|
+
fetched_pages = 1
|
|
323
|
+
|
|
324
|
+
# ---- 翻页:点右箭头 → 等重渲染 → 去重累积(状态机,见 _walk_pages)----
|
|
325
|
+
fetched_pages, total_pages, stopped_reason = await _walk_pages(
|
|
326
|
+
page, items, seen, fetched_pages, pages, limit
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
# 循环提前退出(到 limit / 末页)时补一次终态读取
|
|
330
|
+
if total_pages is None:
|
|
331
|
+
try:
|
|
332
|
+
pag = await page.evaluate(_PAGINATION_JS)
|
|
333
|
+
total_pages = pag.get("totalPages")
|
|
334
|
+
except Exception: # noqa: BLE001 — 终态读取失败不影响部分结果
|
|
335
|
+
total_pages = None
|
|
336
|
+
|
|
337
|
+
items = items[:limit]
|
|
338
|
+
return {
|
|
339
|
+
"items": [
|
|
340
|
+
{"rank": i + 1, "item_id": _item_id_from_url(it.get("url", "")), **it}
|
|
341
|
+
for i, it in enumerate(items)
|
|
342
|
+
],
|
|
343
|
+
"total": len(items),
|
|
344
|
+
"pages_fetched": fetched_pages,
|
|
345
|
+
"pages_total": total_pages,
|
|
346
|
+
"stopped_reason": stopped_reason,
|
|
347
|
+
"query": "",
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
@command(
|
|
352
|
+
namespace="search",
|
|
353
|
+
name="items",
|
|
354
|
+
description="搜索闲鱼商品(浏览器路径,抗风控;--pages 跨页抓取)",
|
|
355
|
+
strategy=Strategy.COOKIE,
|
|
356
|
+
columns=["rank", "item_id", "title", "price", "condition", "brand", "location", "badge", "url"],
|
|
357
|
+
)
|
|
358
|
+
def search(query: str, limit: int = DEFAULT_LIMIT, pages: int = 1) -> dict[str, Any]:
|
|
359
|
+
q = str(query).strip()
|
|
360
|
+
result = asyncio.run(_run(q, _normalize_limit(limit), _normalize_pages(pages)))
|
|
361
|
+
result["query"] = q
|
|
362
|
+
return result
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
__test__ = {
|
|
366
|
+
"MAX_LIMIT": MAX_LIMIT,
|
|
367
|
+
"MAX_PAGES": MAX_PAGES,
|
|
368
|
+
"AUTH_WALL_ATTEMPTS": AUTH_WALL_ATTEMPTS,
|
|
369
|
+
"_normalize_limit": _normalize_limit,
|
|
370
|
+
"_normalize_pages": _normalize_pages,
|
|
371
|
+
"_build_search_url": _build_search_url,
|
|
372
|
+
"_item_id_from_url": _item_id_from_url,
|
|
373
|
+
"_should_retry": _should_retry,
|
|
374
|
+
"_first_card_id": _first_card_id,
|
|
375
|
+
"_walk_pages": _walk_pages,
|
|
376
|
+
}
|