@acedatacloud/skills 2026.719.0 → 2026.719.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/xiaohongshu/SKILL.md +2 -0
- package/skills/xiaohongshu/references/browse.md +4 -0
- package/skills/xiaohongshu/scripts/xhs_contract.py +128 -3
- package/skills/xiaohongshu/tests/fixtures/note.json +13 -0
- package/skills/xiaohongshu/tests/fixtures/profile.json +11 -0
- package/skills/xiaohongshu/tests/test_contract_script.py +80 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@acedatacloud/skills",
|
|
3
|
-
"version": "2026.719.
|
|
3
|
+
"version": "2026.719.2",
|
|
4
4
|
"description": "Agent Skills for AceDataCloud AI services — music, image, video generation, LLM chat, web search. Compatible with Claude Code, GitHub Copilot, Gemini CLI, OpenAI Codex, and 30+ AI coding agents.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"agent-skills",
|
|
@@ -70,6 +70,8 @@ The helper is deterministic and has no network or browser access. Pass JSON thro
|
|
|
70
70
|
- `validate-publish`: validate and normalize a publish preview.
|
|
71
71
|
- `normalize-filters`: validate and normalize search filters.
|
|
72
72
|
- `parse-feed-snapshot`: convert a `www.xiaohongshu.com` semantic snapshot into bounded note cards.
|
|
73
|
+
- `parse-note-snapshot --note-url <url>`: normalize one visible note detail snapshot.
|
|
74
|
+
- `parse-profile-snapshot --profile-url <url>`: normalize one visible profile snapshot.
|
|
73
75
|
|
|
74
76
|
## Completion rules
|
|
75
77
|
|
|
@@ -28,10 +28,14 @@ Read the attached home/recommendation page. Scroll in bounded steps and read aft
|
|
|
28
28
|
|
|
29
29
|
Open a note from its fresh result ref or same-origin canonical URL. Read visible text, media labels, author, engagement, and the first visible comment batch. For more comments/replies, expand and scroll in bounded batches, reading after every transition and stopping at the requested limit. This is not a guaranteed full-comment export.
|
|
30
30
|
|
|
31
|
+
Pass the final semantic snapshot to `parse-note-snapshot --note-url <canonical-url>`. Preserve missing, ambiguous, and truncated states instead of filling absent fields. Only nodes with an explicit `comment` role are returned as structured comments; generic list items are never assumed to be comments.
|
|
32
|
+
|
|
31
33
|
## Profile
|
|
32
34
|
|
|
33
35
|
Open the fresh author link from a result/detail page. Return visible profile text, followers/following/engagement totals, and bounded recent notes. Do not expose unrelated private account data.
|
|
34
36
|
|
|
37
|
+
Pass the final semantic snapshot to `parse-profile-snapshot --profile-url <canonical-url>` before reporting structured profile data. `visible_metrics` contains explicitly labeled profile metrics. Preserve `visible_counts` separately as unlabeled page counters; never reinterpret them as followers, following, likes, or favorites.
|
|
38
|
+
|
|
35
39
|
## Content planning
|
|
36
40
|
|
|
37
41
|
Search multiple user-approved keywords, compare recent and visibly high-engagement notes, inspect representative details/comments, and synthesize themes, title patterns, formats, audience questions, and tag opportunities. This workflow stays read-only unless the user separately requests publishing.
|
|
@@ -8,7 +8,7 @@ import json
|
|
|
8
8
|
import re
|
|
9
9
|
import sys
|
|
10
10
|
from datetime import datetime, timedelta, timezone
|
|
11
|
-
from typing import Match, Optional, Pattern
|
|
11
|
+
from typing import Dict, List, Match, Optional, Pattern, Set
|
|
12
12
|
from urllib.parse import urlparse
|
|
13
13
|
|
|
14
14
|
|
|
@@ -25,6 +25,9 @@ MAX_MEDIA = 20
|
|
|
25
25
|
NOTE_PATH = re.compile(r"^/explore/([A-Za-z0-9]+)$")
|
|
26
26
|
PROFILE_PATH = re.compile(r"^/user/profile/([A-Za-z0-9]+)$")
|
|
27
27
|
TITLE_UNIT = re.compile(r"[\u3400-\u9fff]|[A-Za-z0-9]+")
|
|
28
|
+
COUNT_VALUE = re.compile(r"^(\d+(?:\.\d+)?)([万千]?)$")
|
|
29
|
+
PROFILE_METRIC = re.compile(r"^(关注|粉丝|获赞与收藏|获赞|收藏)\s*([\d,.]+(?:万|千|w|W|k|K)?)$")
|
|
30
|
+
ENGAGEMENT_METRIC = re.compile(r"^(赞|点赞|收藏|评论|分享)\s*[::]?\s*([\d,.]+(?:万|千|w|W|k|K)?)$")
|
|
28
31
|
|
|
29
32
|
|
|
30
33
|
class ContractError(ValueError):
|
|
@@ -211,9 +214,123 @@ def parse_feed_snapshot(snapshot: dict) -> dict:
|
|
|
211
214
|
return {"notes": notes, "truncated": bool(snapshot.get("truncated", False))}
|
|
212
215
|
|
|
213
216
|
|
|
217
|
+
def _valid_nodes(snapshot: dict, origin: str) -> List[Dict]:
|
|
218
|
+
if snapshot.get("origin") != origin:
|
|
219
|
+
raise ContractError(f"snapshot requires the {origin} origin")
|
|
220
|
+
nodes = snapshot.get("nodes")
|
|
221
|
+
if not isinstance(nodes, list):
|
|
222
|
+
raise ContractError("snapshot nodes must be an array")
|
|
223
|
+
return [node for node in nodes if isinstance(node, dict)]
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _named_text(node: Dict) -> str:
|
|
227
|
+
name = node.get("name")
|
|
228
|
+
return name.strip() if isinstance(name, str) else ""
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _first_named(nodes: List[Dict], roles: Set[str]) -> Optional[str]:
|
|
232
|
+
for node in nodes:
|
|
233
|
+
name = _named_text(node)
|
|
234
|
+
if name and node.get("role") in roles:
|
|
235
|
+
return name
|
|
236
|
+
return None
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _visible_counts(nodes: List[Dict]) -> List[str]:
|
|
240
|
+
values = []
|
|
241
|
+
for node in nodes:
|
|
242
|
+
name = _named_text(node)
|
|
243
|
+
if node.get("role") in {"button", "section"} and (
|
|
244
|
+
COUNT_VALUE.fullmatch(name) or ENGAGEMENT_METRIC.fullmatch(name)
|
|
245
|
+
):
|
|
246
|
+
values.append(name)
|
|
247
|
+
return values
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def parse_note_snapshot(snapshot: dict, note_url: str) -> dict:
|
|
251
|
+
nodes = _valid_nodes(snapshot, "https://www.xiaohongshu.com")
|
|
252
|
+
match = _path_match(note_url, NOTE_PATH)
|
|
253
|
+
if not match:
|
|
254
|
+
raise ContractError("note_url must be a canonical Xiaohongshu explore URL")
|
|
255
|
+
canonical_url = f"https://www.xiaohongshu.com/explore/{match.group(1)}"
|
|
256
|
+
profiles = []
|
|
257
|
+
comments = []
|
|
258
|
+
content_parts = []
|
|
259
|
+
comments_started = False
|
|
260
|
+
for node in nodes:
|
|
261
|
+
name = _named_text(node)
|
|
262
|
+
if not name:
|
|
263
|
+
continue
|
|
264
|
+
if node.get("role") == "comment":
|
|
265
|
+
comments_started = True
|
|
266
|
+
comments.append(name)
|
|
267
|
+
continue
|
|
268
|
+
profile = _path_match(node.get("href"), PROFILE_PATH)
|
|
269
|
+
if profile and not comments_started:
|
|
270
|
+
profiles.append({"user_id": profile.group(1), "name": name, "url": f"https://www.xiaohongshu.com/user/profile/{profile.group(1)}"})
|
|
271
|
+
if node.get("role") in {"article", "paragraph", "p"}:
|
|
272
|
+
content_parts.append(name)
|
|
273
|
+
unique_profiles = {item["url"]: item for item in profiles}
|
|
274
|
+
author = next(iter(unique_profiles.values())) if len(unique_profiles) == 1 else None
|
|
275
|
+
return {
|
|
276
|
+
"note_id": match.group(1),
|
|
277
|
+
"canonical_url": canonical_url,
|
|
278
|
+
"title": _first_named(nodes, {"heading"}),
|
|
279
|
+
"author": author,
|
|
280
|
+
"author_state": "available" if author else ("unavailable" if not unique_profiles else "ambiguous"),
|
|
281
|
+
"profile_url": author["url"] if author else None,
|
|
282
|
+
"content": content_parts,
|
|
283
|
+
"visible_engagement": _visible_counts(nodes),
|
|
284
|
+
"comments": comments,
|
|
285
|
+
"truncated": bool(snapshot.get("truncated", False)),
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def parse_profile_snapshot(snapshot: dict, profile_url: str) -> dict:
|
|
290
|
+
nodes = _valid_nodes(snapshot, "https://www.xiaohongshu.com")
|
|
291
|
+
match = _path_match(profile_url, PROFILE_PATH)
|
|
292
|
+
if not match:
|
|
293
|
+
raise ContractError("profile_url must be a canonical Xiaohongshu profile URL")
|
|
294
|
+
canonical_url = f"https://www.xiaohongshu.com/user/profile/{match.group(1)}"
|
|
295
|
+
notes = []
|
|
296
|
+
seen = set()
|
|
297
|
+
metrics = []
|
|
298
|
+
counts = []
|
|
299
|
+
for node in nodes:
|
|
300
|
+
name = _named_text(node)
|
|
301
|
+
if node.get("role") == "section" and PROFILE_METRIC.fullmatch(name):
|
|
302
|
+
metrics.append(name)
|
|
303
|
+
elif node.get("role") in {"button", "section"} and COUNT_VALUE.fullmatch(name):
|
|
304
|
+
counts.append(name)
|
|
305
|
+
note = _path_match(node.get("href"), NOTE_PATH)
|
|
306
|
+
if note and name and note.group(1) not in seen:
|
|
307
|
+
seen.add(note.group(1))
|
|
308
|
+
notes.append({"note_id": note.group(1), "title": name, "url": f"https://www.xiaohongshu.com/explore/{note.group(1)}"})
|
|
309
|
+
return {
|
|
310
|
+
"user_id": match.group(1),
|
|
311
|
+
"canonical_url": canonical_url,
|
|
312
|
+
"display_name": _first_named(nodes, {"heading"}),
|
|
313
|
+
"visible_metrics": metrics,
|
|
314
|
+
"visible_counts": counts,
|
|
315
|
+
"notes": notes,
|
|
316
|
+
"truncated": bool(snapshot.get("truncated", False)),
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
|
|
214
320
|
def main() -> None:
|
|
215
321
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
216
|
-
parser.add_argument(
|
|
322
|
+
parser.add_argument(
|
|
323
|
+
"command",
|
|
324
|
+
choices=(
|
|
325
|
+
"validate-publish",
|
|
326
|
+
"normalize-filters",
|
|
327
|
+
"parse-feed-snapshot",
|
|
328
|
+
"parse-note-snapshot",
|
|
329
|
+
"parse-profile-snapshot",
|
|
330
|
+
),
|
|
331
|
+
)
|
|
332
|
+
parser.add_argument("--note-url")
|
|
333
|
+
parser.add_argument("--profile-url")
|
|
217
334
|
args = parser.parse_args()
|
|
218
335
|
try:
|
|
219
336
|
payload = _read_json()
|
|
@@ -221,8 +338,16 @@ def main() -> None:
|
|
|
221
338
|
result = validate_publish(payload)
|
|
222
339
|
elif args.command == "normalize-filters":
|
|
223
340
|
result = normalize_filters(payload)
|
|
224
|
-
|
|
341
|
+
elif args.command == "parse-feed-snapshot":
|
|
225
342
|
result = parse_feed_snapshot(payload)
|
|
343
|
+
elif args.command == "parse-note-snapshot":
|
|
344
|
+
if not args.note_url:
|
|
345
|
+
raise ContractError("--note-url is required")
|
|
346
|
+
result = parse_note_snapshot(payload, args.note_url)
|
|
347
|
+
else:
|
|
348
|
+
if not args.profile_url:
|
|
349
|
+
raise ContractError("--profile-url is required")
|
|
350
|
+
result = parse_profile_snapshot(payload, args.profile_url)
|
|
226
351
|
except ContractError as exc:
|
|
227
352
|
print(json.dumps({"ok": False, "error": str(exc)}, ensure_ascii=False))
|
|
228
353
|
raise SystemExit(2) from exc
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
{
|
|
2
|
+
"origin": "https://www.xiaohongshu.com",
|
|
3
|
+
"nodes": [
|
|
4
|
+
{"role": "heading", "name": "示例笔记"},
|
|
5
|
+
{"role": "link", "name": "示例作者", "href": "https://www.xiaohongshu.com/user/profile/user123"},
|
|
6
|
+
{"role": "article", "name": "这是脱敏后的示例正文"},
|
|
7
|
+
{"role": "button", "name": "赞:128"},
|
|
8
|
+
{"role": "comment", "name": "示例评论"},
|
|
9
|
+
{"role": "link", "name": "赞过的人", "href": "https://www.xiaohongshu.com/user/profile/liker999"},
|
|
10
|
+
{"role": "link", "name": "评论者", "href": "https://www.xiaohongshu.com/user/profile/commenter456"}
|
|
11
|
+
],
|
|
12
|
+
"truncated": false
|
|
13
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
{
|
|
2
|
+
"origin": "https://www.xiaohongshu.com",
|
|
3
|
+
"nodes": [
|
|
4
|
+
{"role": "heading", "name": "示例作者"},
|
|
5
|
+
{"role": "section", "name": "粉丝 128"},
|
|
6
|
+
{"role": "section", "name": "999"},
|
|
7
|
+
{"role": "link", "name": "最近笔记", "href": "https://www.xiaohongshu.com/explore/abc123?source=profile"},
|
|
8
|
+
{"role": "button", "name": "5678"}
|
|
9
|
+
],
|
|
10
|
+
"truncated": false
|
|
11
|
+
}
|
|
@@ -188,4 +188,83 @@ def test_sanitized_home_fixture_matches_parser_contract() -> None:
|
|
|
188
188
|
}
|
|
189
189
|
],
|
|
190
190
|
"truncated": False,
|
|
191
|
-
}
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def test_parse_note_fixture() -> None:
|
|
195
|
+
fixture = Path(__file__).parent / "fixtures" / "note.json"
|
|
196
|
+
result = xhs_contract.parse_note_snapshot(
|
|
197
|
+
json.loads(fixture.read_text(encoding="utf-8")),
|
|
198
|
+
"https://www.xiaohongshu.com/explore/abc123?tracking=removed",
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
assert result["canonical_url"] == "https://www.xiaohongshu.com/explore/abc123"
|
|
202
|
+
assert result["title"] == "示例笔记"
|
|
203
|
+
assert result["author_state"] == "available"
|
|
204
|
+
assert result["profile_url"] == "https://www.xiaohongshu.com/user/profile/user123"
|
|
205
|
+
assert result["visible_engagement"] == ["赞:128"]
|
|
206
|
+
assert result["comments"] == ["示例评论"]
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def test_parse_note_ignores_generic_list_items_before_author() -> None:
|
|
210
|
+
snapshot = {
|
|
211
|
+
"origin": "https://www.xiaohongshu.com",
|
|
212
|
+
"nodes": [
|
|
213
|
+
{"role": "heading", "name": "示例笔记"},
|
|
214
|
+
{"role": "listitem", "name": "导航项"},
|
|
215
|
+
{
|
|
216
|
+
"role": "link",
|
|
217
|
+
"name": "示例作者",
|
|
218
|
+
"href": "https://www.xiaohongshu.com/user/profile/user123",
|
|
219
|
+
},
|
|
220
|
+
],
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
result = xhs_contract.parse_note_snapshot(
|
|
224
|
+
snapshot, "https://www.xiaohongshu.com/explore/abc123"
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
assert result["author_state"] == "available"
|
|
228
|
+
assert result["comments"] == []
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def test_parse_note_refuses_to_guess_between_pre_comment_profiles() -> None:
|
|
232
|
+
snapshot = {
|
|
233
|
+
"origin": "https://www.xiaohongshu.com",
|
|
234
|
+
"nodes": [
|
|
235
|
+
{"role": "heading", "name": "示例笔记"},
|
|
236
|
+
{
|
|
237
|
+
"role": "link",
|
|
238
|
+
"name": "赞过的人",
|
|
239
|
+
"href": "https://www.xiaohongshu.com/user/profile/liker999",
|
|
240
|
+
},
|
|
241
|
+
{
|
|
242
|
+
"role": "link",
|
|
243
|
+
"name": "示例作者",
|
|
244
|
+
"href": "https://www.xiaohongshu.com/user/profile/user123",
|
|
245
|
+
},
|
|
246
|
+
],
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
result = xhs_contract.parse_note_snapshot(
|
|
250
|
+
snapshot, "https://www.xiaohongshu.com/explore/abc123"
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
assert result["author"] is None
|
|
254
|
+
assert result["author_state"] == "ambiguous"
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def test_parse_profile_fixture() -> None:
|
|
258
|
+
fixture = Path(__file__).parent / "fixtures" / "profile.json"
|
|
259
|
+
result = xhs_contract.parse_profile_snapshot(
|
|
260
|
+
json.loads(fixture.read_text(encoding="utf-8")),
|
|
261
|
+
"https://www.xiaohongshu.com/user/profile/user123?tracking=removed",
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
assert result["canonical_url"] == "https://www.xiaohongshu.com/user/profile/user123"
|
|
265
|
+
assert result["display_name"] == "示例作者"
|
|
266
|
+
assert result["visible_metrics"] == ["粉丝 128"]
|
|
267
|
+
assert result["visible_counts"] == ["999", "5678"]
|
|
268
|
+
assert result["notes"] == [
|
|
269
|
+
{"note_id": "abc123", "title": "最近笔记", "url": "https://www.xiaohongshu.com/explore/abc123"}
|
|
270
|
+
]
|