@acedatacloud/skills 2026.719.0 → 2026.719.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@acedatacloud/skills",
3
- "version": "2026.719.0",
3
+ "version": "2026.719.1",
4
4
  "description": "Agent Skills for AceDataCloud AI services — music, image, video generation, LLM chat, web search. Compatible with Claude Code, GitHub Copilot, Gemini CLI, OpenAI Codex, and 30+ AI coding agents.",
5
5
  "keywords": [
6
6
  "agent-skills",
@@ -70,6 +70,8 @@ The helper is deterministic and has no network or browser access. Pass JSON thro
70
70
  - `validate-publish`: validate and normalize a publish preview.
71
71
  - `normalize-filters`: validate and normalize search filters.
72
72
  - `parse-feed-snapshot`: convert a `www.xiaohongshu.com` semantic snapshot into bounded note cards.
73
+ - `parse-note-snapshot --note-url <url>`: normalize one visible note detail snapshot.
74
+ - `parse-profile-snapshot --profile-url <url>`: normalize one visible profile snapshot.
73
75
 
74
76
  ## Completion rules
75
77
 
@@ -28,10 +28,14 @@ Read the attached home/recommendation page. Scroll in bounded steps and read aft
28
28
 
29
29
  Open a note from its fresh result ref or same-origin canonical URL. Read visible text, media labels, author, engagement, and the first visible comment batch. For more comments/replies, expand and scroll in bounded batches, reading after every transition and stopping at the requested limit. This is not a guaranteed full-comment export.
30
30
 
31
+ Pass the final semantic snapshot to `parse-note-snapshot --note-url <canonical-url>`. Preserve missing, ambiguous, and truncated states instead of filling absent fields. Only nodes with an explicit `comment` role are returned as structured comments; generic list items are never assumed to be comments.
32
+
31
33
  ## Profile
32
34
 
33
35
  Open the fresh author link from a result/detail page. Return visible profile text, followers/following/engagement totals, and bounded recent notes. Do not expose unrelated private account data.
34
36
 
37
+ Pass the final semantic snapshot to `parse-profile-snapshot --profile-url <canonical-url>` before reporting structured profile data. Report only explicitly labeled profile metrics; never reinterpret unrelated page numbers as followers, following, likes, or favorites.
38
+
35
39
  ## Content planning
36
40
 
37
41
  Search multiple user-approved keywords, compare recent and visibly high-engagement notes, inspect representative details/comments, and synthesize themes, title patterns, formats, audience questions, and tag opportunities. This workflow stays read-only unless the user separately requests publishing.
@@ -8,7 +8,7 @@ import json
8
8
  import re
9
9
  import sys
10
10
  from datetime import datetime, timedelta, timezone
11
- from typing import Match, Optional, Pattern
11
+ from typing import Dict, List, Match, Optional, Pattern, Set
12
12
  from urllib.parse import urlparse
13
13
 
14
14
 
@@ -25,6 +25,8 @@ MAX_MEDIA = 20
25
25
  NOTE_PATH = re.compile(r"^/explore/([A-Za-z0-9]+)$")
26
26
  PROFILE_PATH = re.compile(r"^/user/profile/([A-Za-z0-9]+)$")
27
27
  TITLE_UNIT = re.compile(r"[\u3400-\u9fff]|[A-Za-z0-9]+")
28
+ COUNT_VALUE = re.compile(r"^(\d+(?:\.\d+)?)([万千]?)$")
29
+ PROFILE_METRIC = re.compile(r"^(关注|粉丝|获赞与收藏|获赞|收藏)\s*([\d,.]+(?:万|千|w|W|k|K)?)$")
28
30
 
29
31
 
30
32
  class ContractError(ValueError):
@@ -211,9 +213,117 @@ def parse_feed_snapshot(snapshot: dict) -> dict:
211
213
  return {"notes": notes, "truncated": bool(snapshot.get("truncated", False))}
212
214
 
213
215
 
216
+ def _valid_nodes(snapshot: dict, origin: str) -> List[Dict]:
217
+ if snapshot.get("origin") != origin:
218
+ raise ContractError(f"snapshot requires the {origin} origin")
219
+ nodes = snapshot.get("nodes")
220
+ if not isinstance(nodes, list):
221
+ raise ContractError("snapshot nodes must be an array")
222
+ return [node for node in nodes if isinstance(node, dict)]
223
+
224
+
225
+ def _named_text(node: Dict) -> str:
226
+ name = node.get("name")
227
+ return name.strip() if isinstance(name, str) else ""
228
+
229
+
230
+ def _first_named(nodes: List[Dict], roles: Set[str]) -> Optional[str]:
231
+ for node in nodes:
232
+ name = _named_text(node)
233
+ if name and node.get("role") in roles:
234
+ return name
235
+ return None
236
+
237
+
238
+ def _visible_counts(nodes: List[Dict]) -> List[str]:
239
+ values = []
240
+ for node in nodes:
241
+ name = _named_text(node)
242
+ if node.get("role") in {"button", "section"} and COUNT_VALUE.fullmatch(name):
243
+ values.append(name)
244
+ return values
245
+
246
+
247
+ def parse_note_snapshot(snapshot: dict, note_url: str) -> dict:
248
+ nodes = _valid_nodes(snapshot, "https://www.xiaohongshu.com")
249
+ match = _path_match(note_url, NOTE_PATH)
250
+ if not match:
251
+ raise ContractError("note_url must be a canonical Xiaohongshu explore URL")
252
+ canonical_url = f"https://www.xiaohongshu.com/explore/{match.group(1)}"
253
+ profiles = []
254
+ comments = []
255
+ content_parts = []
256
+ comments_started = False
257
+ for node in nodes:
258
+ name = _named_text(node)
259
+ if not name:
260
+ continue
261
+ if node.get("role") == "comment":
262
+ comments_started = True
263
+ comments.append(name)
264
+ continue
265
+ profile = _path_match(node.get("href"), PROFILE_PATH)
266
+ if profile and not comments_started:
267
+ profiles.append({"user_id": profile.group(1), "name": name, "url": f"https://www.xiaohongshu.com/user/profile/{profile.group(1)}"})
268
+ if node.get("role") in {"article", "paragraph", "p"}:
269
+ content_parts.append(name)
270
+ unique_profiles = {item["url"]: item for item in profiles}
271
+ author = next(iter(unique_profiles.values())) if len(unique_profiles) == 1 else None
272
+ return {
273
+ "note_id": match.group(1),
274
+ "canonical_url": canonical_url,
275
+ "title": _first_named(nodes, {"heading"}),
276
+ "author": author,
277
+ "author_state": "available" if author else ("unavailable" if not unique_profiles else "ambiguous"),
278
+ "profile_url": author["url"] if author else None,
279
+ "content": content_parts,
280
+ "visible_engagement": _visible_counts(nodes),
281
+ "comments": comments,
282
+ "truncated": bool(snapshot.get("truncated", False)),
283
+ }
284
+
285
+
286
+ def parse_profile_snapshot(snapshot: dict, profile_url: str) -> dict:
287
+ nodes = _valid_nodes(snapshot, "https://www.xiaohongshu.com")
288
+ match = _path_match(profile_url, PROFILE_PATH)
289
+ if not match:
290
+ raise ContractError("profile_url must be a canonical Xiaohongshu profile URL")
291
+ canonical_url = f"https://www.xiaohongshu.com/user/profile/{match.group(1)}"
292
+ notes = []
293
+ seen = set()
294
+ metrics = []
295
+ for node in nodes:
296
+ name = _named_text(node)
297
+ if node.get("role") == "section" and PROFILE_METRIC.fullmatch(name):
298
+ metrics.append(name)
299
+ note = _path_match(node.get("href"), NOTE_PATH)
300
+ if note and name and note.group(1) not in seen:
301
+ seen.add(note.group(1))
302
+ notes.append({"note_id": note.group(1), "title": name, "url": f"https://www.xiaohongshu.com/explore/{note.group(1)}"})
303
+ return {
304
+ "user_id": match.group(1),
305
+ "canonical_url": canonical_url,
306
+ "display_name": _first_named(nodes, {"heading"}),
307
+ "visible_metrics": metrics,
308
+ "notes": notes,
309
+ "truncated": bool(snapshot.get("truncated", False)),
310
+ }
311
+
312
+
214
313
  def main() -> None:
215
314
  parser = argparse.ArgumentParser(description=__doc__)
216
- parser.add_argument("command", choices=("validate-publish", "normalize-filters", "parse-feed-snapshot"))
315
+ parser.add_argument(
316
+ "command",
317
+ choices=(
318
+ "validate-publish",
319
+ "normalize-filters",
320
+ "parse-feed-snapshot",
321
+ "parse-note-snapshot",
322
+ "parse-profile-snapshot",
323
+ ),
324
+ )
325
+ parser.add_argument("--note-url")
326
+ parser.add_argument("--profile-url")
217
327
  args = parser.parse_args()
218
328
  try:
219
329
  payload = _read_json()
@@ -221,8 +331,16 @@ def main() -> None:
221
331
  result = validate_publish(payload)
222
332
  elif args.command == "normalize-filters":
223
333
  result = normalize_filters(payload)
224
- else:
334
+ elif args.command == "parse-feed-snapshot":
225
335
  result = parse_feed_snapshot(payload)
336
+ elif args.command == "parse-note-snapshot":
337
+ if not args.note_url:
338
+ raise ContractError("--note-url is required")
339
+ result = parse_note_snapshot(payload, args.note_url)
340
+ else:
341
+ if not args.profile_url:
342
+ raise ContractError("--profile-url is required")
343
+ result = parse_profile_snapshot(payload, args.profile_url)
226
344
  except ContractError as exc:
227
345
  print(json.dumps({"ok": False, "error": str(exc)}, ensure_ascii=False))
228
346
  raise SystemExit(2) from exc
@@ -0,0 +1,12 @@
1
+ {
2
+ "origin": "https://www.xiaohongshu.com",
3
+ "nodes": [
4
+ {"role": "heading", "name": "示例笔记"},
5
+ {"role": "link", "name": "示例作者", "href": "https://www.xiaohongshu.com/user/profile/user123"},
6
+ {"role": "article", "name": "这是脱敏后的示例正文"},
7
+ {"role": "button", "name": "128"},
8
+ {"role": "comment", "name": "示例评论"},
9
+ {"role": "link", "name": "评论者", "href": "https://www.xiaohongshu.com/user/profile/commenter456"}
10
+ ],
11
+ "truncated": false
12
+ }
@@ -0,0 +1,11 @@
1
+ {
2
+ "origin": "https://www.xiaohongshu.com",
3
+ "nodes": [
4
+ {"role": "heading", "name": "示例作者"},
5
+ {"role": "section", "name": "粉丝 128"},
6
+ {"role": "section", "name": "999"},
7
+ {"role": "link", "name": "最近笔记", "href": "https://www.xiaohongshu.com/explore/abc123?source=profile"},
8
+ {"role": "button", "name": "5678"}
9
+ ],
10
+ "truncated": false
11
+ }
@@ -188,4 +188,56 @@ def test_sanitized_home_fixture_matches_parser_contract() -> None:
188
188
  }
189
189
  ],
190
190
  "truncated": False,
191
- }
191
+ }
192
+
193
+
194
+ def test_parse_note_fixture() -> None:
195
+ fixture = Path(__file__).parent / "fixtures" / "note.json"
196
+ result = xhs_contract.parse_note_snapshot(
197
+ json.loads(fixture.read_text(encoding="utf-8")),
198
+ "https://www.xiaohongshu.com/explore/abc123?tracking=removed",
199
+ )
200
+
201
+ assert result["canonical_url"] == "https://www.xiaohongshu.com/explore/abc123"
202
+ assert result["title"] == "示例笔记"
203
+ assert result["author_state"] == "available"
204
+ assert result["profile_url"] == "https://www.xiaohongshu.com/user/profile/user123"
205
+ assert result["visible_engagement"] == ["128"]
206
+ assert result["comments"] == ["示例评论"]
207
+
208
+
209
+ def test_parse_note_ignores_generic_list_items_before_author() -> None:
210
+ snapshot = {
211
+ "origin": "https://www.xiaohongshu.com",
212
+ "nodes": [
213
+ {"role": "heading", "name": "示例笔记"},
214
+ {"role": "listitem", "name": "导航项"},
215
+ {
216
+ "role": "link",
217
+ "name": "示例作者",
218
+ "href": "https://www.xiaohongshu.com/user/profile/user123",
219
+ },
220
+ ],
221
+ }
222
+
223
+ result = xhs_contract.parse_note_snapshot(
224
+ snapshot, "https://www.xiaohongshu.com/explore/abc123"
225
+ )
226
+
227
+ assert result["author_state"] == "available"
228
+ assert result["comments"] == []
229
+
230
+
231
+ def test_parse_profile_fixture() -> None:
232
+ fixture = Path(__file__).parent / "fixtures" / "profile.json"
233
+ result = xhs_contract.parse_profile_snapshot(
234
+ json.loads(fixture.read_text(encoding="utf-8")),
235
+ "https://www.xiaohongshu.com/user/profile/user123?tracking=removed",
236
+ )
237
+
238
+ assert result["canonical_url"] == "https://www.xiaohongshu.com/user/profile/user123"
239
+ assert result["display_name"] == "示例作者"
240
+ assert result["visible_metrics"] == ["粉丝 128"]
241
+ assert result["notes"] == [
242
+ {"note_id": "abc123", "title": "最近笔记", "url": "https://www.xiaohongshu.com/explore/abc123"}
243
+ ]