@acedatacloud/skills 2026.719.0 → 2026.719.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@acedatacloud/skills",
3
- "version": "2026.719.0",
3
+ "version": "2026.719.2",
4
4
  "description": "Agent Skills for AceDataCloud AI services — music, image, video generation, LLM chat, web search. Compatible with Claude Code, GitHub Copilot, Gemini CLI, OpenAI Codex, and 30+ AI coding agents.",
5
5
  "keywords": [
6
6
  "agent-skills",
@@ -70,6 +70,8 @@ The helper is deterministic and has no network or browser access. Pass JSON thro
70
70
  - `validate-publish`: validate and normalize a publish preview.
71
71
  - `normalize-filters`: validate and normalize search filters.
72
72
  - `parse-feed-snapshot`: convert a `www.xiaohongshu.com` semantic snapshot into bounded note cards.
73
+ - `parse-note-snapshot --note-url <url>`: normalize one visible note detail snapshot.
74
+ - `parse-profile-snapshot --profile-url <url>`: normalize one visible profile snapshot.
73
75
 
74
76
  ## Completion rules
75
77
 
@@ -28,10 +28,14 @@ Read the attached home/recommendation page. Scroll in bounded steps and read aft
28
28
 
29
29
  Open a note from its fresh result ref or same-origin canonical URL. Read visible text, media labels, author, engagement, and the first visible comment batch. For more comments/replies, expand and scroll in bounded batches, reading after every transition and stopping at the requested limit. This is not a guaranteed full-comment export.
30
30
 
31
+ Pass the final semantic snapshot to `parse-note-snapshot --note-url <canonical-url>`. Preserve missing, ambiguous, and truncated states instead of filling absent fields. Only nodes with an explicit `comment` role are returned as structured comments; generic list items are never assumed to be comments.
32
+
31
33
  ## Profile
32
34
 
33
35
  Open the fresh author link from a result/detail page. Return visible profile text, followers/following/engagement totals, and bounded recent notes. Do not expose unrelated private account data.
34
36
 
37
+ Pass the final semantic snapshot to `parse-profile-snapshot --profile-url <canonical-url>` before reporting structured profile data. `visible_metrics` contains explicitly labeled profile metrics. Preserve `visible_counts` separately as unlabeled page counters; never reinterpret them as followers, following, likes, or favorites.
38
+
35
39
  ## Content planning
36
40
 
37
41
  Search multiple user-approved keywords, compare recent and visibly high-engagement notes, inspect representative details/comments, and synthesize themes, title patterns, formats, audience questions, and tag opportunities. This workflow stays read-only unless the user separately requests publishing.
@@ -8,7 +8,7 @@ import json
8
8
  import re
9
9
  import sys
10
10
  from datetime import datetime, timedelta, timezone
11
- from typing import Match, Optional, Pattern
11
+ from typing import Dict, List, Match, Optional, Pattern, Set
12
12
  from urllib.parse import urlparse
13
13
 
14
14
 
@@ -25,6 +25,9 @@ MAX_MEDIA = 20
25
25
  NOTE_PATH = re.compile(r"^/explore/([A-Za-z0-9]+)$")
26
26
  PROFILE_PATH = re.compile(r"^/user/profile/([A-Za-z0-9]+)$")
27
27
  TITLE_UNIT = re.compile(r"[\u3400-\u9fff]|[A-Za-z0-9]+")
28
+ COUNT_VALUE = re.compile(r"^(\d+(?:\.\d+)?)([万千]?)$")
29
+ PROFILE_METRIC = re.compile(r"^(关注|粉丝|获赞与收藏|获赞|收藏)\s*([\d,.]+(?:万|千|w|W|k|K)?)$")
30
+ ENGAGEMENT_METRIC = re.compile(r"^(赞|点赞|收藏|评论|分享)\s*[::]?\s*([\d,.]+(?:万|千|w|W|k|K)?)$")
28
31
 
29
32
 
30
33
  class ContractError(ValueError):
@@ -211,9 +214,123 @@ def parse_feed_snapshot(snapshot: dict) -> dict:
211
214
  return {"notes": notes, "truncated": bool(snapshot.get("truncated", False))}
212
215
 
213
216
 
217
+ def _valid_nodes(snapshot: dict, origin: str) -> List[Dict]:
218
+ if snapshot.get("origin") != origin:
219
+ raise ContractError(f"snapshot requires the {origin} origin")
220
+ nodes = snapshot.get("nodes")
221
+ if not isinstance(nodes, list):
222
+ raise ContractError("snapshot nodes must be an array")
223
+ return [node for node in nodes if isinstance(node, dict)]
224
+
225
+
226
+ def _named_text(node: Dict) -> str:
227
+ name = node.get("name")
228
+ return name.strip() if isinstance(name, str) else ""
229
+
230
+
231
+ def _first_named(nodes: List[Dict], roles: Set[str]) -> Optional[str]:
232
+ for node in nodes:
233
+ name = _named_text(node)
234
+ if name and node.get("role") in roles:
235
+ return name
236
+ return None
237
+
238
+
239
+ def _visible_counts(nodes: List[Dict]) -> List[str]:
240
+ values = []
241
+ for node in nodes:
242
+ name = _named_text(node)
243
+ if node.get("role") in {"button", "section"} and (
244
+ COUNT_VALUE.fullmatch(name) or ENGAGEMENT_METRIC.fullmatch(name)
245
+ ):
246
+ values.append(name)
247
+ return values
248
+
249
+
250
+ def parse_note_snapshot(snapshot: dict, note_url: str) -> dict:
251
+ nodes = _valid_nodes(snapshot, "https://www.xiaohongshu.com")
252
+ match = _path_match(note_url, NOTE_PATH)
253
+ if not match:
254
+ raise ContractError("note_url must be a canonical Xiaohongshu explore URL")
255
+ canonical_url = f"https://www.xiaohongshu.com/explore/{match.group(1)}"
256
+ profiles = []
257
+ comments = []
258
+ content_parts = []
259
+ comments_started = False
260
+ for node in nodes:
261
+ name = _named_text(node)
262
+ if not name:
263
+ continue
264
+ if node.get("role") == "comment":
265
+ comments_started = True
266
+ comments.append(name)
267
+ continue
268
+ profile = _path_match(node.get("href"), PROFILE_PATH)
269
+ if profile and not comments_started:
270
+ profiles.append({"user_id": profile.group(1), "name": name, "url": f"https://www.xiaohongshu.com/user/profile/{profile.group(1)}"})
271
+ if node.get("role") in {"article", "paragraph", "p"}:
272
+ content_parts.append(name)
273
+ unique_profiles = {item["url"]: item for item in profiles}
274
+ author = next(iter(unique_profiles.values())) if len(unique_profiles) == 1 else None
275
+ return {
276
+ "note_id": match.group(1),
277
+ "canonical_url": canonical_url,
278
+ "title": _first_named(nodes, {"heading"}),
279
+ "author": author,
280
+ "author_state": "available" if author else ("unavailable" if not unique_profiles else "ambiguous"),
281
+ "profile_url": author["url"] if author else None,
282
+ "content": content_parts,
283
+ "visible_engagement": _visible_counts(nodes),
284
+ "comments": comments,
285
+ "truncated": bool(snapshot.get("truncated", False)),
286
+ }
287
+
288
+
289
+ def parse_profile_snapshot(snapshot: dict, profile_url: str) -> dict:
290
+ nodes = _valid_nodes(snapshot, "https://www.xiaohongshu.com")
291
+ match = _path_match(profile_url, PROFILE_PATH)
292
+ if not match:
293
+ raise ContractError("profile_url must be a canonical Xiaohongshu profile URL")
294
+ canonical_url = f"https://www.xiaohongshu.com/user/profile/{match.group(1)}"
295
+ notes = []
296
+ seen = set()
297
+ metrics = []
298
+ counts = []
299
+ for node in nodes:
300
+ name = _named_text(node)
301
+ if node.get("role") == "section" and PROFILE_METRIC.fullmatch(name):
302
+ metrics.append(name)
303
+ elif node.get("role") in {"button", "section"} and COUNT_VALUE.fullmatch(name):
304
+ counts.append(name)
305
+ note = _path_match(node.get("href"), NOTE_PATH)
306
+ if note and name and note.group(1) not in seen:
307
+ seen.add(note.group(1))
308
+ notes.append({"note_id": note.group(1), "title": name, "url": f"https://www.xiaohongshu.com/explore/{note.group(1)}"})
309
+ return {
310
+ "user_id": match.group(1),
311
+ "canonical_url": canonical_url,
312
+ "display_name": _first_named(nodes, {"heading"}),
313
+ "visible_metrics": metrics,
314
+ "visible_counts": counts,
315
+ "notes": notes,
316
+ "truncated": bool(snapshot.get("truncated", False)),
317
+ }
318
+
319
+
214
320
  def main() -> None:
215
321
  parser = argparse.ArgumentParser(description=__doc__)
216
- parser.add_argument("command", choices=("validate-publish", "normalize-filters", "parse-feed-snapshot"))
322
+ parser.add_argument(
323
+ "command",
324
+ choices=(
325
+ "validate-publish",
326
+ "normalize-filters",
327
+ "parse-feed-snapshot",
328
+ "parse-note-snapshot",
329
+ "parse-profile-snapshot",
330
+ ),
331
+ )
332
+ parser.add_argument("--note-url")
333
+ parser.add_argument("--profile-url")
217
334
  args = parser.parse_args()
218
335
  try:
219
336
  payload = _read_json()
@@ -221,8 +338,16 @@ def main() -> None:
221
338
  result = validate_publish(payload)
222
339
  elif args.command == "normalize-filters":
223
340
  result = normalize_filters(payload)
224
- else:
341
+ elif args.command == "parse-feed-snapshot":
225
342
  result = parse_feed_snapshot(payload)
343
+ elif args.command == "parse-note-snapshot":
344
+ if not args.note_url:
345
+ raise ContractError("--note-url is required")
346
+ result = parse_note_snapshot(payload, args.note_url)
347
+ else:
348
+ if not args.profile_url:
349
+ raise ContractError("--profile-url is required")
350
+ result = parse_profile_snapshot(payload, args.profile_url)
226
351
  except ContractError as exc:
227
352
  print(json.dumps({"ok": False, "error": str(exc)}, ensure_ascii=False))
228
353
  raise SystemExit(2) from exc
@@ -0,0 +1,13 @@
1
+ {
2
+ "origin": "https://www.xiaohongshu.com",
3
+ "nodes": [
4
+ {"role": "heading", "name": "示例笔记"},
5
+ {"role": "link", "name": "示例作者", "href": "https://www.xiaohongshu.com/user/profile/user123"},
6
+ {"role": "article", "name": "这是脱敏后的示例正文"},
7
+ {"role": "button", "name": "赞:128"},
8
+ {"role": "comment", "name": "示例评论"},
9
+ {"role": "link", "name": "赞过的人", "href": "https://www.xiaohongshu.com/user/profile/liker999"},
10
+ {"role": "link", "name": "评论者", "href": "https://www.xiaohongshu.com/user/profile/commenter456"}
11
+ ],
12
+ "truncated": false
13
+ }
@@ -0,0 +1,11 @@
1
+ {
2
+ "origin": "https://www.xiaohongshu.com",
3
+ "nodes": [
4
+ {"role": "heading", "name": "示例作者"},
5
+ {"role": "section", "name": "粉丝 128"},
6
+ {"role": "section", "name": "999"},
7
+ {"role": "link", "name": "最近笔记", "href": "https://www.xiaohongshu.com/explore/abc123?source=profile"},
8
+ {"role": "button", "name": "5678"}
9
+ ],
10
+ "truncated": false
11
+ }
@@ -188,4 +188,83 @@ def test_sanitized_home_fixture_matches_parser_contract() -> None:
188
188
  }
189
189
  ],
190
190
  "truncated": False,
191
- }
191
+ }
192
+
193
+
194
+ def test_parse_note_fixture() -> None:
195
+ fixture = Path(__file__).parent / "fixtures" / "note.json"
196
+ result = xhs_contract.parse_note_snapshot(
197
+ json.loads(fixture.read_text(encoding="utf-8")),
198
+ "https://www.xiaohongshu.com/explore/abc123?tracking=removed",
199
+ )
200
+
201
+ assert result["canonical_url"] == "https://www.xiaohongshu.com/explore/abc123"
202
+ assert result["title"] == "示例笔记"
203
+ assert result["author_state"] == "available"
204
+ assert result["profile_url"] == "https://www.xiaohongshu.com/user/profile/user123"
205
+ assert result["visible_engagement"] == ["赞:128"]
206
+ assert result["comments"] == ["示例评论"]
207
+
208
+
209
+ def test_parse_note_ignores_generic_list_items_before_author() -> None:
210
+ snapshot = {
211
+ "origin": "https://www.xiaohongshu.com",
212
+ "nodes": [
213
+ {"role": "heading", "name": "示例笔记"},
214
+ {"role": "listitem", "name": "导航项"},
215
+ {
216
+ "role": "link",
217
+ "name": "示例作者",
218
+ "href": "https://www.xiaohongshu.com/user/profile/user123",
219
+ },
220
+ ],
221
+ }
222
+
223
+ result = xhs_contract.parse_note_snapshot(
224
+ snapshot, "https://www.xiaohongshu.com/explore/abc123"
225
+ )
226
+
227
+ assert result["author_state"] == "available"
228
+ assert result["comments"] == []
229
+
230
+
231
+ def test_parse_note_refuses_to_guess_between_pre_comment_profiles() -> None:
232
+ snapshot = {
233
+ "origin": "https://www.xiaohongshu.com",
234
+ "nodes": [
235
+ {"role": "heading", "name": "示例笔记"},
236
+ {
237
+ "role": "link",
238
+ "name": "赞过的人",
239
+ "href": "https://www.xiaohongshu.com/user/profile/liker999",
240
+ },
241
+ {
242
+ "role": "link",
243
+ "name": "示例作者",
244
+ "href": "https://www.xiaohongshu.com/user/profile/user123",
245
+ },
246
+ ],
247
+ }
248
+
249
+ result = xhs_contract.parse_note_snapshot(
250
+ snapshot, "https://www.xiaohongshu.com/explore/abc123"
251
+ )
252
+
253
+ assert result["author"] is None
254
+ assert result["author_state"] == "ambiguous"
255
+
256
+
257
+ def test_parse_profile_fixture() -> None:
258
+ fixture = Path(__file__).parent / "fixtures" / "profile.json"
259
+ result = xhs_contract.parse_profile_snapshot(
260
+ json.loads(fixture.read_text(encoding="utf-8")),
261
+ "https://www.xiaohongshu.com/user/profile/user123?tracking=removed",
262
+ )
263
+
264
+ assert result["canonical_url"] == "https://www.xiaohongshu.com/user/profile/user123"
265
+ assert result["display_name"] == "示例作者"
266
+ assert result["visible_metrics"] == ["粉丝 128"]
267
+ assert result["visible_counts"] == ["999", "5678"]
268
+ assert result["notes"] == [
269
+ {"note_id": "abc123", "title": "最近笔记", "url": "https://www.xiaohongshu.com/explore/abc123"}
270
+ ]