@acedatacloud/skills 2026.719.0 → 2026.719.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/xiaohongshu/SKILL.md +2 -0
- package/skills/xiaohongshu/references/browse.md +4 -0
- package/skills/xiaohongshu/scripts/xhs_contract.py +121 -3
- package/skills/xiaohongshu/tests/fixtures/note.json +12 -0
- package/skills/xiaohongshu/tests/fixtures/profile.json +11 -0
- package/skills/xiaohongshu/tests/test_contract_script.py +53 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@acedatacloud/skills",
|
|
3
|
-
"version": "2026.719.
|
|
3
|
+
"version": "2026.719.1",
|
|
4
4
|
"description": "Agent Skills for AceDataCloud AI services — music, image, video generation, LLM chat, web search. Compatible with Claude Code, GitHub Copilot, Gemini CLI, OpenAI Codex, and 30+ AI coding agents.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"agent-skills",
|
|
@@ -70,6 +70,8 @@ The helper is deterministic and has no network or browser access. Pass JSON thro
|
|
|
70
70
|
- `validate-publish`: validate and normalize a publish preview.
|
|
71
71
|
- `normalize-filters`: validate and normalize search filters.
|
|
72
72
|
- `parse-feed-snapshot`: convert a `www.xiaohongshu.com` semantic snapshot into bounded note cards.
|
|
73
|
+
- `parse-note-snapshot --note-url <url>`: normalize one visible note detail snapshot.
|
|
74
|
+
- `parse-profile-snapshot --profile-url <url>`: normalize one visible profile snapshot.
|
|
73
75
|
|
|
74
76
|
## Completion rules
|
|
75
77
|
|
|
@@ -28,10 +28,14 @@ Read the attached home/recommendation page. Scroll in bounded steps and read aft
|
|
|
28
28
|
|
|
29
29
|
Open a note from its fresh result ref or same-origin canonical URL. Read visible text, media labels, author, engagement, and the first visible comment batch. For more comments/replies, expand and scroll in bounded batches, reading after every transition and stopping at the requested limit. This is not a guaranteed full-comment export.
|
|
30
30
|
|
|
31
|
+
Pass the final semantic snapshot to `parse-note-snapshot --note-url <canonical-url>`. Preserve missing, ambiguous, and truncated states instead of filling absent fields. Only nodes with an explicit `comment` role are returned as structured comments; generic list items are never assumed to be comments.
|
|
32
|
+
|
|
31
33
|
## Profile
|
|
32
34
|
|
|
33
35
|
Open the fresh author link from a result/detail page. Return visible profile text, followers/following/engagement totals, and bounded recent notes. Do not expose unrelated private account data.
|
|
34
36
|
|
|
37
|
+
Pass the final semantic snapshot to `parse-profile-snapshot --profile-url <canonical-url>` before reporting structured profile data. Report only explicitly labeled profile metrics; never reinterpret unrelated page numbers as followers, following, likes, or favorites.
|
|
38
|
+
|
|
35
39
|
## Content planning
|
|
36
40
|
|
|
37
41
|
Search multiple user-approved keywords, compare recent and visibly high-engagement notes, inspect representative details/comments, and synthesize themes, title patterns, formats, audience questions, and tag opportunities. This workflow stays read-only unless the user separately requests publishing.
|
|
@@ -8,7 +8,7 @@ import json
|
|
|
8
8
|
import re
|
|
9
9
|
import sys
|
|
10
10
|
from datetime import datetime, timedelta, timezone
|
|
11
|
-
from typing import Match, Optional, Pattern
|
|
11
|
+
from typing import Dict, List, Match, Optional, Pattern, Set
|
|
12
12
|
from urllib.parse import urlparse
|
|
13
13
|
|
|
14
14
|
|
|
@@ -25,6 +25,8 @@ MAX_MEDIA = 20
|
|
|
25
25
|
NOTE_PATH = re.compile(r"^/explore/([A-Za-z0-9]+)$")
|
|
26
26
|
PROFILE_PATH = re.compile(r"^/user/profile/([A-Za-z0-9]+)$")
|
|
27
27
|
TITLE_UNIT = re.compile(r"[\u3400-\u9fff]|[A-Za-z0-9]+")
|
|
28
|
+
COUNT_VALUE = re.compile(r"^(\d+(?:\.\d+)?)([万千]?)$")
|
|
29
|
+
PROFILE_METRIC = re.compile(r"^(关注|粉丝|获赞与收藏|获赞|收藏)\s*([\d,.]+(?:万|千|w|W|k|K)?)$")
|
|
28
30
|
|
|
29
31
|
|
|
30
32
|
class ContractError(ValueError):
|
|
@@ -211,9 +213,117 @@ def parse_feed_snapshot(snapshot: dict) -> dict:
|
|
|
211
213
|
return {"notes": notes, "truncated": bool(snapshot.get("truncated", False))}
|
|
212
214
|
|
|
213
215
|
|
|
216
|
+
def _valid_nodes(snapshot: dict, origin: str) -> List[Dict]:
|
|
217
|
+
if snapshot.get("origin") != origin:
|
|
218
|
+
raise ContractError(f"snapshot requires the {origin} origin")
|
|
219
|
+
nodes = snapshot.get("nodes")
|
|
220
|
+
if not isinstance(nodes, list):
|
|
221
|
+
raise ContractError("snapshot nodes must be an array")
|
|
222
|
+
return [node for node in nodes if isinstance(node, dict)]
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _named_text(node: Dict) -> str:
|
|
226
|
+
name = node.get("name")
|
|
227
|
+
return name.strip() if isinstance(name, str) else ""
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _first_named(nodes: List[Dict], roles: Set[str]) -> Optional[str]:
|
|
231
|
+
for node in nodes:
|
|
232
|
+
name = _named_text(node)
|
|
233
|
+
if name and node.get("role") in roles:
|
|
234
|
+
return name
|
|
235
|
+
return None
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _visible_counts(nodes: List[Dict]) -> List[str]:
|
|
239
|
+
values = []
|
|
240
|
+
for node in nodes:
|
|
241
|
+
name = _named_text(node)
|
|
242
|
+
if node.get("role") in {"button", "section"} and COUNT_VALUE.fullmatch(name):
|
|
243
|
+
values.append(name)
|
|
244
|
+
return values
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def parse_note_snapshot(snapshot: dict, note_url: str) -> dict:
|
|
248
|
+
nodes = _valid_nodes(snapshot, "https://www.xiaohongshu.com")
|
|
249
|
+
match = _path_match(note_url, NOTE_PATH)
|
|
250
|
+
if not match:
|
|
251
|
+
raise ContractError("note_url must be a canonical Xiaohongshu explore URL")
|
|
252
|
+
canonical_url = f"https://www.xiaohongshu.com/explore/{match.group(1)}"
|
|
253
|
+
profiles = []
|
|
254
|
+
comments = []
|
|
255
|
+
content_parts = []
|
|
256
|
+
comments_started = False
|
|
257
|
+
for node in nodes:
|
|
258
|
+
name = _named_text(node)
|
|
259
|
+
if not name:
|
|
260
|
+
continue
|
|
261
|
+
if node.get("role") == "comment":
|
|
262
|
+
comments_started = True
|
|
263
|
+
comments.append(name)
|
|
264
|
+
continue
|
|
265
|
+
profile = _path_match(node.get("href"), PROFILE_PATH)
|
|
266
|
+
if profile and not comments_started:
|
|
267
|
+
profiles.append({"user_id": profile.group(1), "name": name, "url": f"https://www.xiaohongshu.com/user/profile/{profile.group(1)}"})
|
|
268
|
+
if node.get("role") in {"article", "paragraph", "p"}:
|
|
269
|
+
content_parts.append(name)
|
|
270
|
+
unique_profiles = {item["url"]: item for item in profiles}
|
|
271
|
+
author = next(iter(unique_profiles.values())) if len(unique_profiles) == 1 else None
|
|
272
|
+
return {
|
|
273
|
+
"note_id": match.group(1),
|
|
274
|
+
"canonical_url": canonical_url,
|
|
275
|
+
"title": _first_named(nodes, {"heading"}),
|
|
276
|
+
"author": author,
|
|
277
|
+
"author_state": "available" if author else ("unavailable" if not unique_profiles else "ambiguous"),
|
|
278
|
+
"profile_url": author["url"] if author else None,
|
|
279
|
+
"content": content_parts,
|
|
280
|
+
"visible_engagement": _visible_counts(nodes),
|
|
281
|
+
"comments": comments,
|
|
282
|
+
"truncated": bool(snapshot.get("truncated", False)),
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def parse_profile_snapshot(snapshot: dict, profile_url: str) -> dict:
|
|
287
|
+
nodes = _valid_nodes(snapshot, "https://www.xiaohongshu.com")
|
|
288
|
+
match = _path_match(profile_url, PROFILE_PATH)
|
|
289
|
+
if not match:
|
|
290
|
+
raise ContractError("profile_url must be a canonical Xiaohongshu profile URL")
|
|
291
|
+
canonical_url = f"https://www.xiaohongshu.com/user/profile/{match.group(1)}"
|
|
292
|
+
notes = []
|
|
293
|
+
seen = set()
|
|
294
|
+
metrics = []
|
|
295
|
+
for node in nodes:
|
|
296
|
+
name = _named_text(node)
|
|
297
|
+
if node.get("role") == "section" and PROFILE_METRIC.fullmatch(name):
|
|
298
|
+
metrics.append(name)
|
|
299
|
+
note = _path_match(node.get("href"), NOTE_PATH)
|
|
300
|
+
if note and name and note.group(1) not in seen:
|
|
301
|
+
seen.add(note.group(1))
|
|
302
|
+
notes.append({"note_id": note.group(1), "title": name, "url": f"https://www.xiaohongshu.com/explore/{note.group(1)}"})
|
|
303
|
+
return {
|
|
304
|
+
"user_id": match.group(1),
|
|
305
|
+
"canonical_url": canonical_url,
|
|
306
|
+
"display_name": _first_named(nodes, {"heading"}),
|
|
307
|
+
"visible_metrics": metrics,
|
|
308
|
+
"notes": notes,
|
|
309
|
+
"truncated": bool(snapshot.get("truncated", False)),
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
|
|
214
313
|
def main() -> None:
|
|
215
314
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
216
|
-
parser.add_argument(
|
|
315
|
+
parser.add_argument(
|
|
316
|
+
"command",
|
|
317
|
+
choices=(
|
|
318
|
+
"validate-publish",
|
|
319
|
+
"normalize-filters",
|
|
320
|
+
"parse-feed-snapshot",
|
|
321
|
+
"parse-note-snapshot",
|
|
322
|
+
"parse-profile-snapshot",
|
|
323
|
+
),
|
|
324
|
+
)
|
|
325
|
+
parser.add_argument("--note-url")
|
|
326
|
+
parser.add_argument("--profile-url")
|
|
217
327
|
args = parser.parse_args()
|
|
218
328
|
try:
|
|
219
329
|
payload = _read_json()
|
|
@@ -221,8 +331,16 @@ def main() -> None:
|
|
|
221
331
|
result = validate_publish(payload)
|
|
222
332
|
elif args.command == "normalize-filters":
|
|
223
333
|
result = normalize_filters(payload)
|
|
224
|
-
|
|
334
|
+
elif args.command == "parse-feed-snapshot":
|
|
225
335
|
result = parse_feed_snapshot(payload)
|
|
336
|
+
elif args.command == "parse-note-snapshot":
|
|
337
|
+
if not args.note_url:
|
|
338
|
+
raise ContractError("--note-url is required")
|
|
339
|
+
result = parse_note_snapshot(payload, args.note_url)
|
|
340
|
+
else:
|
|
341
|
+
if not args.profile_url:
|
|
342
|
+
raise ContractError("--profile-url is required")
|
|
343
|
+
result = parse_profile_snapshot(payload, args.profile_url)
|
|
226
344
|
except ContractError as exc:
|
|
227
345
|
print(json.dumps({"ok": False, "error": str(exc)}, ensure_ascii=False))
|
|
228
346
|
raise SystemExit(2) from exc
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
{
|
|
2
|
+
"origin": "https://www.xiaohongshu.com",
|
|
3
|
+
"nodes": [
|
|
4
|
+
{"role": "heading", "name": "示例笔记"},
|
|
5
|
+
{"role": "link", "name": "示例作者", "href": "https://www.xiaohongshu.com/user/profile/user123"},
|
|
6
|
+
{"role": "article", "name": "这是脱敏后的示例正文"},
|
|
7
|
+
{"role": "button", "name": "128"},
|
|
8
|
+
{"role": "comment", "name": "示例评论"},
|
|
9
|
+
{"role": "link", "name": "评论者", "href": "https://www.xiaohongshu.com/user/profile/commenter456"}
|
|
10
|
+
],
|
|
11
|
+
"truncated": false
|
|
12
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
{
|
|
2
|
+
"origin": "https://www.xiaohongshu.com",
|
|
3
|
+
"nodes": [
|
|
4
|
+
{"role": "heading", "name": "示例作者"},
|
|
5
|
+
{"role": "section", "name": "粉丝 128"},
|
|
6
|
+
{"role": "section", "name": "999"},
|
|
7
|
+
{"role": "link", "name": "最近笔记", "href": "https://www.xiaohongshu.com/explore/abc123?source=profile"},
|
|
8
|
+
{"role": "button", "name": "5678"}
|
|
9
|
+
],
|
|
10
|
+
"truncated": false
|
|
11
|
+
}
|
|
@@ -188,4 +188,56 @@ def test_sanitized_home_fixture_matches_parser_contract() -> None:
|
|
|
188
188
|
}
|
|
189
189
|
],
|
|
190
190
|
"truncated": False,
|
|
191
|
-
}
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def test_parse_note_fixture() -> None:
|
|
195
|
+
fixture = Path(__file__).parent / "fixtures" / "note.json"
|
|
196
|
+
result = xhs_contract.parse_note_snapshot(
|
|
197
|
+
json.loads(fixture.read_text(encoding="utf-8")),
|
|
198
|
+
"https://www.xiaohongshu.com/explore/abc123?tracking=removed",
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
assert result["canonical_url"] == "https://www.xiaohongshu.com/explore/abc123"
|
|
202
|
+
assert result["title"] == "示例笔记"
|
|
203
|
+
assert result["author_state"] == "available"
|
|
204
|
+
assert result["profile_url"] == "https://www.xiaohongshu.com/user/profile/user123"
|
|
205
|
+
assert result["visible_engagement"] == ["128"]
|
|
206
|
+
assert result["comments"] == ["示例评论"]
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def test_parse_note_ignores_generic_list_items_before_author() -> None:
|
|
210
|
+
snapshot = {
|
|
211
|
+
"origin": "https://www.xiaohongshu.com",
|
|
212
|
+
"nodes": [
|
|
213
|
+
{"role": "heading", "name": "示例笔记"},
|
|
214
|
+
{"role": "listitem", "name": "导航项"},
|
|
215
|
+
{
|
|
216
|
+
"role": "link",
|
|
217
|
+
"name": "示例作者",
|
|
218
|
+
"href": "https://www.xiaohongshu.com/user/profile/user123",
|
|
219
|
+
},
|
|
220
|
+
],
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
result = xhs_contract.parse_note_snapshot(
|
|
224
|
+
snapshot, "https://www.xiaohongshu.com/explore/abc123"
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
assert result["author_state"] == "available"
|
|
228
|
+
assert result["comments"] == []
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def test_parse_profile_fixture() -> None:
|
|
232
|
+
fixture = Path(__file__).parent / "fixtures" / "profile.json"
|
|
233
|
+
result = xhs_contract.parse_profile_snapshot(
|
|
234
|
+
json.loads(fixture.read_text(encoding="utf-8")),
|
|
235
|
+
"https://www.xiaohongshu.com/user/profile/user123?tracking=removed",
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
assert result["canonical_url"] == "https://www.xiaohongshu.com/user/profile/user123"
|
|
239
|
+
assert result["display_name"] == "示例作者"
|
|
240
|
+
assert result["visible_metrics"] == ["粉丝 128"]
|
|
241
|
+
assert result["notes"] == [
|
|
242
|
+
{"note_id": "abc123", "title": "最近笔记", "url": "https://www.xiaohongshu.com/explore/abc123"}
|
|
243
|
+
]
|