@acedatacloud/skills 2026.726.6 → 2026.726.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/bilibili/SKILL.md +44 -5
- package/skills/bilibili/scripts/bilibili.py +158 -14
- package/skills/toutiao/SKILL.md +123 -0
- package/skills/toutiao/scripts/toutiao.py +737 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@acedatacloud/skills",
|
|
3
|
-
"version": "2026.726.
|
|
3
|
+
"version": "2026.726.8",
|
|
4
4
|
"description": "Agent Skills for AceDataCloud AI services — music, image, video generation, LLM chat, web search. Compatible with Claude Code, GitHub Copilot, Gemini CLI, OpenAI Codex, and 30+ AI coding agents.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"agent-skills",
|
package/skills/bilibili/SKILL.md
CHANGED
|
@@ -42,6 +42,8 @@ python3 "$BILI" whoami # who is logged in (mid, name)
|
|
|
42
42
|
python3 "$BILI" articles --limit 20 # my 专栏 articles + stats
|
|
43
43
|
python3 "$BILI" article <cvid> # one article's stats (cv id)
|
|
44
44
|
python3 "$BILI" drafts --limit 50 # list saved drafts (aid + title)
|
|
45
|
+
python3 "$BILI" status --limit 10 # review state of recent submissions
|
|
46
|
+
python3 "$BILI" categories # 分类 names accepted by --category
|
|
45
47
|
```
|
|
46
48
|
|
|
47
49
|
Stats come straight from Bilibili: `view` (阅读), `like` (点赞), `reply` (评论),
|
|
@@ -69,12 +71,41 @@ BILI="$SKILL_DIR/scripts/bilibili.py"; [ -f "$BILI" ] || BILI=$(find /tmp -maxde
|
|
|
69
71
|
python3 "$BILI" publish --title "标题" --content-file a.html # dry-run
|
|
70
72
|
python3 "$BILI" publish --title "标题" --content-file a.html --draft-only --confirm # save a draft
|
|
71
73
|
python3 "$BILI" publish --title "标题" --content-file a.html --confirm # save draft + submit (publish)
|
|
74
|
+
python3 "$BILI" publish --title "标题" --content-file a.html --category 数码 --confirm # pick the 分类
|
|
72
75
|
```
|
|
73
76
|
|
|
74
77
|
- `--draft-only` saves a draft (no submit) — safe; finish/publish in the editor.
|
|
78
|
+
Prefer it when the user has not clearly asked to go public: a submitted
|
|
79
|
+
article enters a review queue this CLI cannot withdraw it from.
|
|
75
80
|
- The **submit** (go public) step is frequently rate-limited by Bilibili
|
|
76
81
|
risk-control (HTTP 412). When that happens the CLI reports the saved draft +
|
|
77
|
-
edit URL so the user can publish from the web editor.
|
|
82
|
+
edit URL so the user can publish from the web editor.
|
|
83
|
+
- `--category` picks the 分类 (default **数码**, which suits technical posts).
|
|
84
|
+
Run `categories` for the accepted names. The result echoes the 分类 actually
|
|
85
|
+
used as `category` / `category_id` — report it when it wasn't the user's pick.
|
|
86
|
+
|
|
87
|
+
### Publishing is not instant — it enters a review queue
|
|
88
|
+
|
|
89
|
+
A successful `submit` returns `state: -2` (**待审核**) and `pending_review: true`.
|
|
90
|
+
The returned `url` **404s for everyone until Bilibili approves it** (usually
|
|
91
|
+
minutes to hours). Tell the user it is pending; do not claim it is live, and do
|
|
92
|
+
not re-submit. Check later with `status`, which is the only view that shows
|
|
93
|
+
pending/rejected articles (the public list omits them):
|
|
94
|
+
|
|
95
|
+
```sh
|
|
96
|
+
python3 "$BILI" status --limit 5 # state_desc: 待审核 / 已发布 / 未通过 (+ reason)
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Other states are **not** pending and will never go live on their own — the CLI
|
|
100
|
+
returns them with `ok: false` and `published: false` (`-1` 未通过 rejected, `-3`
|
|
101
|
+
锁定, `-4` 已删除). Read `state_desc`, tell the user plainly, and check `status`
|
|
102
|
+
for the `reason` rather than re-submitting.
|
|
103
|
+
|
|
104
|
+
The `id` in the publish result is the **article id** (use `cv<id>`), which is a
|
|
105
|
+
different number from `draft_aid`. Only ever share the `url` field. If Bilibili
|
|
106
|
+
returned no article id, `url` and `id` are `null` and `id_unverified: true` is
|
|
107
|
+
set — there is nothing shareable yet, so find the article with `status` instead
|
|
108
|
+
of constructing a link from `draft_aid`.
|
|
78
109
|
|
|
79
110
|
## Managing drafts (the 999-draft cap)
|
|
80
111
|
|
|
@@ -108,8 +139,8 @@ skips this.
|
|
|
108
139
|
- **This is the user's real Bilibili account.** Confirm before any publish.
|
|
109
140
|
- **submit may 412** (anti-bot) even when the draft saved fine — the draft is the
|
|
110
141
|
reliable result; don't loop-retry submit.
|
|
111
|
-
- A
|
|
112
|
-
|
|
142
|
+
- A 分类 the account can't post to returns `-17`; without `--category` the CLI
|
|
143
|
+
auto-retries the fallback list.
|
|
113
144
|
- **Never print `BILIBILI_COOKIES`** — it is full account access.
|
|
114
145
|
- **ToS**: acts only on the user's own account with their own captured cookie.
|
|
115
146
|
|
|
@@ -123,6 +154,14 @@ After you successfully publish and obtain the live result URL, call the built-in
|
|
|
123
154
|
publish_artifact(kind="article", channel="bilibili", title="<title>", url="<the REAL returned URL>", status="delivered")
|
|
124
155
|
```
|
|
125
156
|
|
|
126
|
-
Use the real returned URL — never fabricate one.
|
|
127
|
-
|
|
157
|
+
Use the real returned URL — never fabricate one. That is the `url` field of the
|
|
158
|
+
publish result (built from the article `id`, **not** `draft_aid`). Call it once
|
|
159
|
+
per published item, only after delivery is confirmed.
|
|
160
|
+
|
|
161
|
+
- `state: 0` or `1` (已发布, live) → `status="delivered"`.
|
|
162
|
+
- `pending_review: true` (待审核) → `status="draft"`; the URL is not live yet.
|
|
163
|
+
Re-record as delivered later if the user asks you to re-check with `status`.
|
|
164
|
+
- rejected/locked states, a 412-blocked submit, `id_unverified: true`, or
|
|
165
|
+
`state_unknown: true` → `status="failed"` (or skip it).
|
|
166
|
+
|
|
128
167
|
See `_shared/artifacts.md`.
|
|
@@ -277,8 +277,30 @@ def cmd_article(jar, args):
|
|
|
277
277
|
})
|
|
278
278
|
|
|
279
279
|
|
|
280
|
-
#
|
|
281
|
-
|
|
280
|
+
# The article's 分类 (from /x/article/categories). This is the `category` form
|
|
281
|
+
# field — `tid` alone does NOT set it (a hardcoded category=0 always lands in
|
|
282
|
+
# 生活 regardless of tid). Sub-category ids only; a parent id is rejected.
|
|
283
|
+
_CATEGORIES = {
|
|
284
|
+
"数码": "26", "科技": "26", "tech": "26",
|
|
285
|
+
"学习": "34", "人文历史": "25", "自然": "33", "汽车": "27",
|
|
286
|
+
"日常": "15", "生活": "15", "美食": "13", "时尚": "14", "运动": "22", "萌宠": "21",
|
|
287
|
+
"绘画": "23", "手工": "24", "摄影": "38", "音乐舞蹈": "39", "模型手办": "11",
|
|
288
|
+
"动漫杂谈": "4", "动漫资讯": "5", "动画技术": "31",
|
|
289
|
+
"单机游戏": "6", "电子竞技": "7", "手机游戏": "8", "网络游戏": "9", "桌游棋牌": "10",
|
|
290
|
+
"电影": "12", "电视剧": "35", "纪录片": "36", "综艺": "37",
|
|
291
|
+
"原创连载": "18", "同人连载": "19", "短篇小说": "32", "小说杂谈": "20",
|
|
292
|
+
}
|
|
293
|
+
# Fallback order when --category is not given: 数码 first (most of our content
|
|
294
|
+
# is technical), then broad ones. A category the account can't post to → -17.
|
|
295
|
+
_TID_CANDIDATES = ["26", "34", "15", "4", "6", "12", "23"]
|
|
296
|
+
|
|
297
|
+
# A creative article's `state` (what 投稿管理 shows). Only 0/1 are actually live.
|
|
298
|
+
_STATES = {0: "已发布", 1: "已发布", -1: "未通过", -2: "待审核", -3: "锁定", -4: "已删除"}
|
|
299
|
+
|
|
300
|
+
# id → canonical display name (first alias wins; 数码/科技/tech all map to 26).
|
|
301
|
+
_CATEGORY_NAMES = {}
|
|
302
|
+
for _name, _cid in _CATEGORIES.items():
|
|
303
|
+
_CATEGORY_NAMES.setdefault(_cid, _name)
|
|
282
304
|
|
|
283
305
|
|
|
284
306
|
# ── image upload (Bilibili hotlink-blocks external imgs; re-host via
|
|
@@ -545,6 +567,19 @@ def md_to_html(src):
|
|
|
545
567
|
return "\n".join(out)
|
|
546
568
|
|
|
547
569
|
|
|
570
|
+
def _resolve_categories(raw):
|
|
571
|
+
"""--category → the single 分类 id to use, or the fallback list if unset."""
|
|
572
|
+
if not raw:
|
|
573
|
+
return _TID_CANDIDATES, None
|
|
574
|
+
name = raw.strip()
|
|
575
|
+
cid = _CATEGORIES.get(name)
|
|
576
|
+
if not cid:
|
|
577
|
+
if not name.isdigit():
|
|
578
|
+
die(f"unknown --category {raw!r}; known: " + ", ".join(sorted(_CATEGORIES)))
|
|
579
|
+
cid = name
|
|
580
|
+
return [cid], cid # explicit choice: don't silently fall back to another 分类
|
|
581
|
+
|
|
582
|
+
|
|
548
583
|
def cmd_publish(jar, args):
|
|
549
584
|
if not args.title:
|
|
550
585
|
die("--title is required")
|
|
@@ -562,14 +597,22 @@ def cmd_publish(jar, args):
|
|
|
562
597
|
if not csrf:
|
|
563
598
|
die("no bili_jct cookie (CSRF token) — reconnect Bilibili.")
|
|
564
599
|
|
|
600
|
+
# Resolve before the dry-run so a typo'd 分类 is caught in the preview,
|
|
601
|
+
# and before rehost_images so a bad one can't burn uploads then abort.
|
|
602
|
+
cids, explicit = _resolve_categories(args.category)
|
|
603
|
+
|
|
565
604
|
if not CONFIRM:
|
|
566
605
|
out({
|
|
567
606
|
"dry_run": True, "command": "publish", "platform": "bilibili",
|
|
568
607
|
"title": args.title, "draft_only": args.draft_only,
|
|
608
|
+
"category": (_CATEGORY_NAMES.get(explicit, explicit) if explicit
|
|
609
|
+
else "(auto: 数码)"),
|
|
569
610
|
"content_bytes": len(content),
|
|
570
611
|
"note": "Bilibili 专栏 content is HTML. Re-run with --confirm as the "
|
|
571
|
-
"LAST argument to write.
|
|
572
|
-
"
|
|
612
|
+
"LAST argument to write. Published articles enter Bilibili's "
|
|
613
|
+
"review queue (state -2) before going public; the submit step "
|
|
614
|
+
"is also often 412-limited, in which case the saved draft is "
|
|
615
|
+
"the reliable result.",
|
|
573
616
|
})
|
|
574
617
|
return
|
|
575
618
|
|
|
@@ -585,13 +628,15 @@ def cmd_publish(jar, args):
|
|
|
585
628
|
ref = "https://member.bilibili.com/"
|
|
586
629
|
base = {
|
|
587
630
|
"title": args.title, "content": content, "csrf": csrf,
|
|
588
|
-
"
|
|
631
|
+
"list_id": "0", "reprint": "0", "original": "1",
|
|
589
632
|
"media_id": "0", "spoiler": "0", "save": "0", "pgc_id": "0",
|
|
590
633
|
}
|
|
591
|
-
# 1. save draft, retrying
|
|
592
|
-
aid, last = None, None
|
|
593
|
-
for
|
|
594
|
-
|
|
634
|
+
# 1. save draft, retrying the 分类 until one is accepted
|
|
635
|
+
aid, chosen, last = None, None, None
|
|
636
|
+
for cid in cids:
|
|
637
|
+
# `category` is what actually sets 分类; `tid` must match or the article
|
|
638
|
+
# silently lands in 生活.
|
|
639
|
+
body = dict(base, tid=cid, category=cid)
|
|
595
640
|
status, text = request("POST", f"{API}/x/article/creative/draft/addupdate",
|
|
596
641
|
jar, referer=ref, form=body)
|
|
597
642
|
try:
|
|
@@ -601,21 +646,22 @@ def cmd_publish(jar, args):
|
|
|
601
646
|
continue
|
|
602
647
|
if r.get("code") == 0:
|
|
603
648
|
aid = (r.get("data") or {}).get("aid")
|
|
604
|
-
|
|
649
|
+
chosen = cid
|
|
605
650
|
break
|
|
606
651
|
last = f"code={r.get('code')} {r.get('message')}"
|
|
607
652
|
if "分类" not in str(r.get("message", "")) and r.get("code") != -17:
|
|
608
|
-
break # a non-category error won't be fixed by another
|
|
653
|
+
break # a non-category error won't be fixed by another one
|
|
609
654
|
if not aid:
|
|
610
655
|
die(f"save-draft failed: {last}")
|
|
611
656
|
|
|
612
657
|
if args.draft_only:
|
|
613
658
|
out({"ok": True, "draft_only": True, "aid": str(aid),
|
|
659
|
+
"category_id": chosen, "category": _CATEGORY_NAMES.get(chosen, chosen),
|
|
614
660
|
"edit_url": f"https://member.bilibili.com/article-text/home?aid={aid}"})
|
|
615
661
|
return
|
|
616
662
|
|
|
617
663
|
# 2. submit (publish) — may be 412 risk-controlled; report draft if so
|
|
618
|
-
body = dict(base, tid=
|
|
664
|
+
body = dict(base, tid=chosen, category=chosen, aid=aid)
|
|
619
665
|
status, text = request("POST", f"{API}/x/article/creative/article/submit",
|
|
620
666
|
jar, referer=ref, form=body)
|
|
621
667
|
try:
|
|
@@ -623,8 +669,64 @@ def cmd_publish(jar, args):
|
|
|
623
669
|
except json.JSONDecodeError:
|
|
624
670
|
r = None
|
|
625
671
|
if r and r.get("code") == 0:
|
|
626
|
-
|
|
627
|
-
|
|
672
|
+
# The published article gets a NEW id — the draft aid is not a cv id, so
|
|
673
|
+
# cv<draft_aid> 404s. Only trust an id the submit response actually gave.
|
|
674
|
+
data = r.get("data") or {}
|
|
675
|
+
try:
|
|
676
|
+
cvid = int(data.get("aid"))
|
|
677
|
+
except (TypeError, ValueError):
|
|
678
|
+
cvid = None
|
|
679
|
+
if cvid is not None and cvid <= 0:
|
|
680
|
+
cvid = None # article ids are positive; 0 means "not returned"
|
|
681
|
+
res = {"ok": True, "published": True,
|
|
682
|
+
"id": str(cvid) if cvid is not None else None,
|
|
683
|
+
"url": (f"https://www.bilibili.com/read/cv{cvid}"
|
|
684
|
+
if cvid is not None else None),
|
|
685
|
+
"draft_aid": str(aid),
|
|
686
|
+
"category_id": chosen,
|
|
687
|
+
"category": _CATEGORY_NAMES.get(chosen, chosen)}
|
|
688
|
+
notes = []
|
|
689
|
+
if cvid is None:
|
|
690
|
+
# Never hand back a draft-aid URL dressed up as the published one.
|
|
691
|
+
res["id_unverified"] = True
|
|
692
|
+
notes.append("submit succeeded but returned no article id — there is "
|
|
693
|
+
f"no shareable URL yet (draft aid {aid} is NOT a cv id). "
|
|
694
|
+
"Find the article with `status`.")
|
|
695
|
+
raw_state = data.get("state")
|
|
696
|
+
if isinstance(raw_state, bool):
|
|
697
|
+
state = None
|
|
698
|
+
else:
|
|
699
|
+
try:
|
|
700
|
+
state = int(raw_state)
|
|
701
|
+
except (TypeError, ValueError):
|
|
702
|
+
state = None
|
|
703
|
+
if state is None and raw_state is not None:
|
|
704
|
+
res["state_unknown"] = True
|
|
705
|
+
notes.append(f"submit returned an unreadable state ({raw_state!r}); "
|
|
706
|
+
"confirm with `status` before reporting this as live.")
|
|
707
|
+
if state is not None and state not in (0, 1):
|
|
708
|
+
res["state"] = state
|
|
709
|
+
res["state_desc"] = _STATES.get(state, str(state))
|
|
710
|
+
if state == -2:
|
|
711
|
+
# 待审核: the url 404s for everyone until Bilibili approves it.
|
|
712
|
+
res["pending_review"] = True
|
|
713
|
+
notes.append("submitted, now in Bilibili's review queue — the URL "
|
|
714
|
+
"goes live once approved (usually minutes to hours). "
|
|
715
|
+
"Re-check with `status`; do not re-submit.")
|
|
716
|
+
elif state < 0:
|
|
717
|
+
# -1 未通过 / -3 锁定 / -4 已删除 — NOT pending, will never go live.
|
|
718
|
+
res["ok"] = False
|
|
719
|
+
res["published"] = False
|
|
720
|
+
notes.append(f"submit returned state {state} "
|
|
721
|
+
f"({_STATES.get(state, 'unknown')}) — the article is "
|
|
722
|
+
"not live and is not queued. Check `status` for the "
|
|
723
|
+
"reason; do not report this as published.")
|
|
724
|
+
elif state in (0, 1):
|
|
725
|
+
res["state"] = state
|
|
726
|
+
res["state_desc"] = _STATES[state]
|
|
727
|
+
if notes:
|
|
728
|
+
res["note"] = " ".join(notes)
|
|
729
|
+
out(res)
|
|
628
730
|
else:
|
|
629
731
|
out({"ok": False, "published": False, "draft_saved": True, "aid": str(aid),
|
|
630
732
|
"edit_url": f"https://member.bilibili.com/article-text/home?aid={aid}",
|
|
@@ -637,6 +739,41 @@ def _drafts_of(d: dict) -> list:
|
|
|
637
739
|
return al.get("drafts") or []
|
|
638
740
|
|
|
639
741
|
|
|
742
|
+
def cmd_categories(jar, args):
|
|
743
|
+
by_id = {}
|
|
744
|
+
for name, cid in _CATEGORIES.items():
|
|
745
|
+
by_id.setdefault(cid, []).append(name)
|
|
746
|
+
cats = [{"id": cid, "names": names} for cid, names in by_id.items()]
|
|
747
|
+
out({"count": len(cats), "categories": cats,
|
|
748
|
+
"note": "pass any of the names (or the raw id) to `publish --category`; "
|
|
749
|
+
"the default is 数码."})
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
def cmd_status(jar, args):
|
|
753
|
+
# The creative list is the ONLY view that shows pending/rejected articles —
|
|
754
|
+
# the public space list omits anything not yet approved.
|
|
755
|
+
d = get_json(f"{API}/x/article/creative/article/list?pn=1&ps={args.limit}",
|
|
756
|
+
jar, referer="https://member.bilibili.com/")
|
|
757
|
+
if d.get("code"):
|
|
758
|
+
die(f"status error (code={d.get('code')}): {d.get('message')} — cookie expired?")
|
|
759
|
+
al = (d.get("data") or {}).get("artlist") or d.get("artlist") or {}
|
|
760
|
+
rows = []
|
|
761
|
+
for a in (al.get("articles") or [])[: args.limit]:
|
|
762
|
+
st = a.get("state")
|
|
763
|
+
cvid = a.get("id")
|
|
764
|
+
rows.append({
|
|
765
|
+
"id": str(cvid) if cvid else None,
|
|
766
|
+
"title": a.get("title"),
|
|
767
|
+
"state": st,
|
|
768
|
+
"state_desc": _STATES.get(st, str(st)) if isinstance(st, int) else None,
|
|
769
|
+
"live": isinstance(st, int) and st >= 0,
|
|
770
|
+
"reason": a.get("reason") or None,
|
|
771
|
+
"category": (a.get("category") or {}).get("name"),
|
|
772
|
+
"url": f"https://www.bilibili.com/read/cv{cvid}" if cvid else None,
|
|
773
|
+
})
|
|
774
|
+
out({"count": len(rows), "articles": rows})
|
|
775
|
+
|
|
776
|
+
|
|
640
777
|
def cmd_drafts(jar, args):
|
|
641
778
|
# 专栏 drafts are capped at 999; this lists them so they can be pruned.
|
|
642
779
|
d = get_json(f"{API}/x/article/creative/draft/list?pn={args.page}&ps={args.limit}",
|
|
@@ -686,6 +823,8 @@ COMMANDS = {
|
|
|
686
823
|
"publish": cmd_publish,
|
|
687
824
|
"drafts": cmd_drafts,
|
|
688
825
|
"delete-draft": cmd_delete_draft,
|
|
826
|
+
"categories": cmd_categories,
|
|
827
|
+
"status": cmd_status,
|
|
689
828
|
}
|
|
690
829
|
|
|
691
830
|
|
|
@@ -702,8 +841,13 @@ def main() -> None:
|
|
|
702
841
|
sp.add_argument("--content", help="HTML content inline")
|
|
703
842
|
sp.add_argument("--content-file", help="path to an HTML file")
|
|
704
843
|
sp.add_argument("--draft-only", action="store_true", help="save a draft; do NOT submit")
|
|
844
|
+
sp.add_argument("--category", help="分类 name (e.g. 数码, 学习, 日常) or a raw tid; "
|
|
845
|
+
"default tries 数码 first")
|
|
705
846
|
sp.add_argument("--no-rehost-images", action="store_true",
|
|
706
847
|
help="keep external image URLs as-is (skip Bilibili CDN re-host)")
|
|
848
|
+
sp = sub.add_parser("categories", help="list the 分类 names accepted by --category")
|
|
849
|
+
sp = sub.add_parser("status", help="review state of the user's recent submissions")
|
|
850
|
+
sp.add_argument("--limit", type=int, default=10)
|
|
707
851
|
sp = sub.add_parser("drafts", help="list 专栏 drafts (id+title); use to prune the 999-draft cap")
|
|
708
852
|
sp.add_argument("--limit", type=int, default=50)
|
|
709
853
|
sp.add_argument("--page", type=int, default=1)
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: toutiao
|
|
3
|
+
description: Read and publish on 今日头条 / Toutiao (mp.toutiao.com) with the user's own login cookies (BYOC) — list their 头条号 articles with impression/read/comment stats, inspect one article, and publish a new 图文 article or draft. Use when the user mentions 今日头条, 头条号, Toutiao, "我的头条文章", reading their article stats (展现/阅读), or 发头条 / publishing to Toutiao.
|
|
4
|
+
when_to_use: |
|
|
5
|
+
Trigger for anything on the user's 今日头条号 (mp.toutiao.com) account driven by
|
|
6
|
+
their own login cookie: show who they are, list their articles with impression /
|
|
7
|
+
read / comment counts, look at one article's stats, or publish a new article.
|
|
8
|
+
This acts as the user's real account, so writes are gated behind an explicit
|
|
9
|
+
confirmation.
|
|
10
|
+
connections: [toutiao]
|
|
11
|
+
allowed_tools: [Bash]
|
|
12
|
+
license: Apache-2.0
|
|
13
|
+
metadata:
|
|
14
|
+
author: acedatacloud
|
|
15
|
+
version: "1.0"
|
|
16
|
+
---
|
|
17
|
+
|
|
18
|
+
# toutiao — read & publish on 今日头条 via your own cookies
|
|
19
|
+
|
|
20
|
+
Drives the user's **real** 头条号 through the same `mp.toutiao.com` creator APIs
|
|
21
|
+
the web console uses, authenticated by the login cookie they captured with the
|
|
22
|
+
ACE extension. No browser, no third-party deps — just `urllib`.
|
|
23
|
+
|
|
24
|
+
The connector injects the cookie jar as an env var:
|
|
25
|
+
|
|
26
|
+
- `TOUTIAO_COOKIES` — a JSON array of cookies. **Secret — never echo or print
|
|
27
|
+
it.** The CLI reads it for you.
|
|
28
|
+
|
|
29
|
+
> Writes echo the `csrftoken` cookie back as the `X-CSRFToken` header (the CLI
|
|
30
|
+
> does this). Reads and writes are otherwise cookie-only — no request signing.
|
|
31
|
+
|
|
32
|
+
## CLI
|
|
33
|
+
|
|
34
|
+
The skill ships [`scripts/toutiao.py`](scripts/toutiao.py) — self-contained, stdlib only.
|
|
35
|
+
|
|
36
|
+
```sh
|
|
37
|
+
# $SKILL_DIR can point at another skill loaded this turn — anchor on our own
|
|
38
|
+
# script, and re-run this at the top of every Bash block (fresh shell each time).
|
|
39
|
+
TT="$SKILL_DIR/scripts/toutiao.py"; [ -f "$TT" ] || TT=$(find /tmp -maxdepth 8 -path '*/skills/*/scripts/toutiao.py' 2>/dev/null | head -1)
|
|
40
|
+
[ -f "$TT" ] || { echo "toutiao script not found (SKILL_DIR=$SKILL_DIR)" >&2; exit 1; }
|
|
41
|
+
python3 "$TT" whoami # who is logged in (+ total article count)
|
|
42
|
+
python3 "$TT" articles --limit 20 # my articles + stats
|
|
43
|
+
python3 "$TT" articles --status draft # only drafts
|
|
44
|
+
python3 "$TT" article <pgc-id> # one article's stats
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Stats come straight from 头条: `impression_count` (展现), `read_count` (阅读),
|
|
48
|
+
`comment_count` (评论), `digg_count` (点赞).
|
|
49
|
+
|
|
50
|
+
`--status` accepts `all` (default) / `draft` / `published` / `reviewing` / `failed`.
|
|
51
|
+
|
|
52
|
+
## Verify the connection first
|
|
53
|
+
|
|
54
|
+
```sh
|
|
55
|
+
TT="$SKILL_DIR/scripts/toutiao.py"; [ -f "$TT" ] || TT=$(find /tmp -maxdepth 8 -path '*/skills/*/scripts/toutiao.py' 2>/dev/null | head -1)
|
|
56
|
+
python3 "$TT" whoami
|
|
57
|
+
# → {"user_id": ..., "name": "...", "articles_total": 273}
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
On an auth error the cookie is expired — tell the user to reconnect at
|
|
61
|
+
<https://auth.acedata.cloud/user/connections>. Do **not** retry in a loop.
|
|
62
|
+
|
|
63
|
+
## Publishing — GATED (dry-run unless trailing `--confirm`)
|
|
64
|
+
|
|
65
|
+
`publish` writes to the user's real 头条号. Content is **Markdown** (converted to
|
|
66
|
+
HTML for 头条's body field). Without a trailing `--confirm` it dry-runs.
|
|
67
|
+
`--confirm` is honored **only as the last argument**. Always show the dry-run,
|
|
68
|
+
get an explicit "yes", then re-run with `--confirm` last.
|
|
69
|
+
|
|
70
|
+
```sh
|
|
71
|
+
TT="$SKILL_DIR/scripts/toutiao.py"; [ -f "$TT" ] || TT=$(find /tmp -maxdepth 8 -path '*/skills/*/scripts/toutiao.py' 2>/dev/null | head -1)
|
|
72
|
+
python3 "$TT" publish --title "标题" --content-file a.md # dry-run
|
|
73
|
+
python3 "$TT" publish --title "标题" --content-file a.md --draft-only --confirm # private draft
|
|
74
|
+
python3 "$TT" publish --title "标题" --content-file a.md --confirm # PUBLIC, enters 审核
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
- `--draft-only` saves a private draft (`save=1`) — safe, nothing public.
|
|
78
|
+
- Without `--draft-only` the article is **submitted publicly** under the user's
|
|
79
|
+
name and enters 头条's 审核 queue. Default to `--draft-only` unless the user
|
|
80
|
+
clearly asked to go live.
|
|
81
|
+
- **Titles must be 2–30 characters** — 头条 rejects anything outside that range
|
|
82
|
+
(the CLI fails early with a clear message).
|
|
83
|
+
|
|
84
|
+
## Images
|
|
85
|
+
|
|
86
|
+
头条 rejects the **entire article** (`7115 图片uri非法`) if any `<img>` points at
|
|
87
|
+
a non-头条 URL — so external images cannot simply be left alone. `publish`
|
|
88
|
+
uploads every image in the body to 头条's own CDN first and rewrites the tag with
|
|
89
|
+
the CDN attributes 头条 requires.
|
|
90
|
+
|
|
91
|
+
If an image fails to upload, `publish` **aborts and posts nothing**, listing the
|
|
92
|
+
offending URLs — it will not silently publish the user's article with images
|
|
93
|
+
missing. Pass `--drop-failed-images` to publish without them instead.
|
|
94
|
+
`--no-rehost-images` skips the whole step (头条 will then reject the article
|
|
95
|
+
unless the body already carries 头条-hosted images).
|
|
96
|
+
|
|
97
|
+
## Gotchas — surface before the user is surprised
|
|
98
|
+
|
|
99
|
+
- **This is the user's real 头条号.** Confirm before any publish.
|
|
100
|
+
- **审核**: a published article is not instantly live — 头条 reviews it. The
|
|
101
|
+
returned URL goes live once it passes; a rejected article shows up under
|
|
102
|
+
`articles --status failed`.
|
|
103
|
+
- **Daily publish cap**: 头条 caps 图文 posts per day. Hitting it fails the
|
|
104
|
+
publish with 头条's own message — relay it, don't retry in a loop.
|
|
105
|
+
- **14-day edit window**: 头条 refuses edits to articles published more than 14
|
|
106
|
+
days ago, so this skill does not offer an edit command.
|
|
107
|
+
- **Cookie expiry**: reconnect at auth.acedata.cloud/user/connections.
|
|
108
|
+
- **Never print `TOUTIAO_COOKIES`** — it is full account access.
|
|
109
|
+
- **ToS**: cookie automation acts only on the user's own account with their own
|
|
110
|
+
captured cookie; the user owns that risk.
|
|
111
|
+
|
|
112
|
+
## Record the output
|
|
113
|
+
|
|
114
|
+
After you successfully publish and obtain the live result URL, call the built-in
|
|
115
|
+
`publish_artifact` tool ONCE so the user can track this deliverable in **My Outputs**:
|
|
116
|
+
|
|
117
|
+
```
|
|
118
|
+
publish_artifact(kind="article", channel="toutiao", title="<title>", url="<the REAL returned URL>", status="delivered")
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
Use the real returned URL — never fabricate one. Call it once per published item,
|
|
122
|
+
only after delivery is confirmed; skip it (or use `status="failed"`) if publishing failed.
|
|
123
|
+
See `_shared/artifacts.md`.
|
|
@@ -0,0 +1,737 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
toutiao — read & publish on 今日头条 (mp.toutiao.com) with the user's own login
|
|
4
|
+
cookies (BYOC). Standard-library only (urllib), no third-party deps, so it runs
|
|
5
|
+
in the bare sandbox without an image change.
|
|
6
|
+
|
|
7
|
+
The connector injects the user's cookie jar as a JSON env var ``TOUTIAO_COOKIES``
|
|
8
|
+
— a list of ``{name, value, domain, ...}`` dicts captured by the ACE extension.
|
|
9
|
+
|
|
10
|
+
头条's creator APIs are cookie-only (no request signing); writes additionally
|
|
11
|
+
need the ``csrftoken`` cookie echoed back as the ``X-CSRFToken`` header.
|
|
12
|
+
|
|
13
|
+
Read commands run directly. ``publish`` is GATED: without a trailing
|
|
14
|
+
``--confirm`` it only dry-runs. ``--confirm`` is honored ONLY as the last
|
|
15
|
+
argument, so a title/content that merely contains "--confirm" can never silently
|
|
16
|
+
go live. ``--draft-only`` stops after saving a private draft.
|
|
17
|
+
|
|
18
|
+
Examples:
|
|
19
|
+
python3 toutiao.py whoami
|
|
20
|
+
python3 toutiao.py articles --limit 20
|
|
21
|
+
python3 toutiao.py articles --status draft
|
|
22
|
+
python3 toutiao.py article <pgc-id>
|
|
23
|
+
python3 toutiao.py publish --title T --content-file a.md --draft-only --confirm
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import argparse
|
|
29
|
+
import gzip
|
|
30
|
+
import html as _html
|
|
31
|
+
import ipaddress
|
|
32
|
+
import json
|
|
33
|
+
import os
|
|
34
|
+
import random
|
|
35
|
+
import re
|
|
36
|
+
import socket
|
|
37
|
+
import sys
|
|
38
|
+
import urllib.error
|
|
39
|
+
import urllib.parse
|
|
40
|
+
import urllib.request
|
|
41
|
+
from html.parser import HTMLParser as _HTMLParser
|
|
42
|
+
|
|
43
|
+
UA = (
|
|
44
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
|
|
45
|
+
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
|
|
46
|
+
)
|
|
47
|
+
PLATFORM = "toutiao"
|
|
48
|
+
MP = "https://mp.toutiao.com"
|
|
49
|
+
|
|
50
|
+
_RAW = sys.argv[1:]
|
|
51
|
+
CONFIRM = bool(_RAW) and _RAW[-1] == "--confirm"
|
|
52
|
+
ARGV = _RAW[:-1] if CONFIRM else list(_RAW)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def out(obj) -> None:
|
|
56
|
+
print(json.dumps(obj, ensure_ascii=False, indent=2, default=str))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def die(msg: str, code: int = 1) -> None:
|
|
60
|
+
out({"error": msg})
|
|
61
|
+
sys.exit(code)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# ── Cookie jar (shared pattern across the cookie-BYOC skills) ────────
|
|
65
|
+
|
|
66
|
+
def load_cookies() -> list:
|
|
67
|
+
env = f"{PLATFORM.upper()}_COOKIES"
|
|
68
|
+
raw = os.environ.get(env)
|
|
69
|
+
if not raw:
|
|
70
|
+
die(f"{env} is not set — connect 今日头条 at "
|
|
71
|
+
f"https://auth.acedata.cloud/user/connections, then retry.")
|
|
72
|
+
try:
|
|
73
|
+
jar = json.loads(raw)
|
|
74
|
+
except json.JSONDecodeError as e:
|
|
75
|
+
die(f"{env} is not valid JSON: {e}")
|
|
76
|
+
if not isinstance(jar, list):
|
|
77
|
+
die(f"{env} must be a JSON list of cookies, got {type(jar).__name__}")
|
|
78
|
+
return jar
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _domain_matches(host: str, domain: str) -> bool:
|
|
82
|
+
d = domain.lstrip(".").lower()
|
|
83
|
+
h = host.lower()
|
|
84
|
+
return not d or h == d or h.endswith("." + d)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def cookie_header(jar: list, url: str) -> str:
|
|
88
|
+
host = urllib.parse.urlsplit(url).hostname or ""
|
|
89
|
+
host_in_scope = any(
|
|
90
|
+
c.get("domain") and _domain_matches(host, str(c["domain"])) for c in jar
|
|
91
|
+
)
|
|
92
|
+
parts = []
|
|
93
|
+
for c in jar:
|
|
94
|
+
name, value = c.get("name"), c.get("value")
|
|
95
|
+
if not name or value is None:
|
|
96
|
+
continue
|
|
97
|
+
domain = c.get("domain")
|
|
98
|
+
if domain:
|
|
99
|
+
if not _domain_matches(host, str(domain)):
|
|
100
|
+
continue
|
|
101
|
+
elif not host_in_scope:
|
|
102
|
+
continue
|
|
103
|
+
parts.append(f"{name}={value}")
|
|
104
|
+
return "; ".join(parts)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def cookie_value(jar: list, name: str):
|
|
108
|
+
for c in jar:
|
|
109
|
+
if c.get("name") == name:
|
|
110
|
+
return c.get("value")
|
|
111
|
+
return None
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
# ── HTTP ────────────────────────────────────────────────────────────
|
|
115
|
+
|
|
116
|
+
def _headers(jar: list, referer: str) -> dict:
|
|
117
|
+
return {
|
|
118
|
+
"User-Agent": UA,
|
|
119
|
+
"Accept": "application/json, text/plain, */*",
|
|
120
|
+
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
|
121
|
+
"Referer": referer,
|
|
122
|
+
"Origin": MP,
|
|
123
|
+
"sec-ch-ua": '"Chromium";v="124", "Google Chrome";v="124", "Not-A.Brand";v="99"',
|
|
124
|
+
"sec-ch-ua-mobile": "?0",
|
|
125
|
+
"sec-ch-ua-platform": '"macOS"',
|
|
126
|
+
"Sec-Fetch-Dest": "empty",
|
|
127
|
+
"Sec-Fetch-Mode": "cors",
|
|
128
|
+
"Sec-Fetch-Site": "same-origin",
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def request(method: str, url: str, jar: list, *, referer, headers=None, body=None,
|
|
133
|
+
nonfatal=False):
|
|
134
|
+
hdrs = _headers(jar, referer)
|
|
135
|
+
if headers:
|
|
136
|
+
hdrs.update(headers)
|
|
137
|
+
data = body.encode("utf-8") if isinstance(body, str) else body
|
|
138
|
+
req = urllib.request.Request(url, data=data, headers=hdrs, method=method)
|
|
139
|
+
# Unredirected → urllib will NOT re-send these if the API 30x-redirects to a
|
|
140
|
+
# different host (e.g. a login page), so neither the jar nor the CSRF token
|
|
141
|
+
# (which is itself a cookie value) leaks off-site.
|
|
142
|
+
req.add_unredirected_header("Cookie", cookie_header(jar, url))
|
|
143
|
+
# Writes are rejected without the csrftoken cookie echoed as a header.
|
|
144
|
+
req.add_unredirected_header("X-CSRFToken", str(cookie_value(jar, "csrftoken") or ""))
|
|
145
|
+
try:
|
|
146
|
+
with urllib.request.urlopen(req, timeout=60) as resp:
|
|
147
|
+
raw = resp.read()
|
|
148
|
+
if resp.headers.get("Content-Encoding") == "gzip":
|
|
149
|
+
raw = gzip.decompress(raw)
|
|
150
|
+
return resp.status, raw.decode("utf-8", "replace")
|
|
151
|
+
except urllib.error.HTTPError as e:
|
|
152
|
+
raw = e.read()
|
|
153
|
+
try:
|
|
154
|
+
if e.headers.get("Content-Encoding") == "gzip":
|
|
155
|
+
raw = gzip.decompress(raw)
|
|
156
|
+
except Exception:
|
|
157
|
+
pass
|
|
158
|
+
return e.code, raw.decode("utf-8", "replace")
|
|
159
|
+
except urllib.error.URLError as e:
|
|
160
|
+
if nonfatal:
|
|
161
|
+
raise RuntimeError(f"network error reaching {url}: {e.reason}")
|
|
162
|
+
die(f"network error reaching {url}: {e.reason}")
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def api_envelope(method: str, path: str, jar: list, *, referer=None, body=None, headers=None,
|
|
166
|
+
nonfatal=False):
|
|
167
|
+
"""Call an mp.toutiao.com endpoint and return the raw {code, message, ...}
|
|
168
|
+
envelope, dying on a non-zero code. 头条 answers HTTP 200 even for logical
|
|
169
|
+
failures, so the envelope code is the real status."""
|
|
170
|
+
url = f"{MP}{path}"
|
|
171
|
+
ref = referer or f"{MP}/profile_v4/graphic/articles"
|
|
172
|
+
status, text = request(method, url, jar, referer=ref, body=body, headers=headers,
|
|
173
|
+
nonfatal=nonfatal)
|
|
174
|
+
if status in (401, 403) or "/auth/page/login" in text:
|
|
175
|
+
msg = (f"auth failed ({status}) on {path} — cookie likely expired. "
|
|
176
|
+
f"Reconnect 今日头条 at https://auth.acedata.cloud/user/connections.")
|
|
177
|
+
if nonfatal:
|
|
178
|
+
raise RuntimeError(msg)
|
|
179
|
+
die(msg)
|
|
180
|
+
try:
|
|
181
|
+
env = json.loads(text)
|
|
182
|
+
except json.JSONDecodeError:
|
|
183
|
+
if nonfatal:
|
|
184
|
+
raise RuntimeError(f"non-JSON response ({status}) from {path}: {text[:200]}")
|
|
185
|
+
die(f"non-JSON response ({status}) from {path}: {text[:300]}")
|
|
186
|
+
if not isinstance(env, dict):
|
|
187
|
+
if nonfatal:
|
|
188
|
+
raise RuntimeError(f"unexpected response from {path}: {text[:200]}")
|
|
189
|
+
die(f"unexpected response from {path}: {text[:300]}")
|
|
190
|
+
# `.get("code", …)` would return a stored None instead of falling back to
|
|
191
|
+
# err_no, so test explicitly.
|
|
192
|
+
code = env.get("code")
|
|
193
|
+
if code is None:
|
|
194
|
+
code = env.get("err_no")
|
|
195
|
+
if code not in (0, None):
|
|
196
|
+
msg = env.get("message") or env.get("reason") or ""
|
|
197
|
+
if code in (401, 403) or "登录" in str(msg):
|
|
198
|
+
auth_msg = (f"auth failed (code={code}: {msg}) — cookie likely expired. "
|
|
199
|
+
f"Reconnect at https://auth.acedata.cloud/user/connections.")
|
|
200
|
+
if nonfatal:
|
|
201
|
+
raise RuntimeError(auth_msg)
|
|
202
|
+
die(auth_msg)
|
|
203
|
+
if nonfatal:
|
|
204
|
+
raise RuntimeError(f"头条 API error on {path} (code={code}): {msg}")
|
|
205
|
+
die(f"头条 API error on {path} (code={code}): {msg}")
|
|
206
|
+
return env
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def api(method: str, path: str, jar: list, **kw):
|
|
210
|
+
"""Same as api_envelope but unwraps the `data` payload."""
|
|
211
|
+
return api_envelope(method, path, jar, **kw).get("data")
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
# ── commands ────────────────────────────────────────────────────────
|
|
215
|
+
|
|
216
|
+
def media_info(jar):
|
|
217
|
+
d = api("GET", "/mp/agw/media/get_media_info/", jar, referer=f"{MP}/profile_v4/index")
|
|
218
|
+
if not isinstance(d, dict):
|
|
219
|
+
die("could not read 头条号 profile (cookie expired?)")
|
|
220
|
+
return d
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def cmd_whoami(jar, _args):
|
|
224
|
+
d = media_info(jar)
|
|
225
|
+
media = d.get("media") or {}
|
|
226
|
+
user = d.get("user") or {}
|
|
227
|
+
_, total = _list_page(jar, "all", 1, 1)
|
|
228
|
+
out({
|
|
229
|
+
"user_id": user.get("id"),
|
|
230
|
+
"media_id": media.get("id"),
|
|
231
|
+
"name": user.get("screen_name") or media.get("display_name"),
|
|
232
|
+
"url": f"https://www.toutiao.com/c/user/token/{media.get('id')}/"
|
|
233
|
+
if media.get("id") else None,
|
|
234
|
+
"articles_total": total,
|
|
235
|
+
"media_level": media.get("article_media_level"),
|
|
236
|
+
})
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
# status → 头条's own filter value. `all` also returns drafts (is_draft=1).
|
|
240
|
+
_STATUS = {"all": "all", "draft": "draft", "published": "published",
|
|
241
|
+
"reviewing": "verifying", "failed": "unpass"}
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _list_page(jar, status, page, size, nonfatal=False):
|
|
245
|
+
q = urllib.parse.urlencode({
|
|
246
|
+
"status": _STATUS.get(status, "all"), "from_time": 0, "start_time": 0,
|
|
247
|
+
"end_time": 0, "search_word": "", "page": page, "size": size,
|
|
248
|
+
})
|
|
249
|
+
d = api("GET", f"/mp/agw/article/list/?{q}", jar, nonfatal=nonfatal) or {}
|
|
250
|
+
return d.get("content") or [], d.get("total")
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _fmt(a: dict) -> dict:
|
|
254
|
+
pgc = a.get("pgc_id") or a.get("item_id") or a.get("id")
|
|
255
|
+
return {
|
|
256
|
+
"pgc_id": str(pgc) if pgc is not None else None,
|
|
257
|
+
"title": a.get("title"),
|
|
258
|
+
"url": a.get("article_url"),
|
|
259
|
+
"is_draft": bool(a.get("is_draft")),
|
|
260
|
+
"status_desc": a.get("status_desc"),
|
|
261
|
+
"impression_count": a.get("impression_count"),
|
|
262
|
+
# go_detail_count_v2 is the read count; show_go_detail_count is a
|
|
263
|
+
# display-toggle BOOLEAN, never a fallback for it.
|
|
264
|
+
"read_count": a.get("go_detail_count_v2"),
|
|
265
|
+
"comment_count": a.get("comment_count"),
|
|
266
|
+
"digg_count": a.get("digg_count"),
|
|
267
|
+
"create_time": a.get("create_time"),
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _iter_articles(jar, status, hard_cap=2000):
|
|
272
|
+
page, seen = 1, 0
|
|
273
|
+
while seen < hard_cap:
|
|
274
|
+
items, total = _list_page(jar, status, page, 50)
|
|
275
|
+
if not items:
|
|
276
|
+
return
|
|
277
|
+
for it in items:
|
|
278
|
+
yield it
|
|
279
|
+
seen += 1
|
|
280
|
+
if total is not None and seen >= total:
|
|
281
|
+
return
|
|
282
|
+
page += 1
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def cmd_articles(jar, args):
|
|
286
|
+
items = []
|
|
287
|
+
for it in _iter_articles(jar, args.status):
|
|
288
|
+
items.append(it)
|
|
289
|
+
if len(items) >= args.limit:
|
|
290
|
+
break
|
|
291
|
+
_, total = _list_page(jar, args.status, 1, 1)
|
|
292
|
+
out({"total": total, "count": len(items), "status": args.status,
|
|
293
|
+
"articles": [_fmt(a) for a in items]})
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def cmd_article(jar, args):
|
|
297
|
+
# 头条 has no per-article detail endpoint for the creator (article/edit only
|
|
298
|
+
# works within 14 days of publishing), but the list already carries every
|
|
299
|
+
# stat, so resolve by scanning it.
|
|
300
|
+
for it in _iter_articles(jar, "all"):
|
|
301
|
+
if str(it.get("pgc_id")) == str(args.id) or str(it.get("item_id")) == str(args.id):
|
|
302
|
+
res = _fmt(it)
|
|
303
|
+
res["abstract"] = (it.get("abstract") or "")[:200]
|
|
304
|
+
res["word_count"] = it.get("content_word_cnt")
|
|
305
|
+
out(res)
|
|
306
|
+
return
|
|
307
|
+
die(f"article {args.id} not found among your 头条 articles")
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
# ── image re-host (头条 rejects the whole article if an <img> is external) ──
|
|
311
|
+
|
|
312
|
+
MAX_IMG_BYTES = 12 * 1024 * 1024
|
|
313
|
+
_IMG_SKIP = ("toutiao.com", "byteimg.com", "pstatp.com", "toutiaoimg.com")
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
class _ImgFinder(_HTMLParser):
|
|
317
|
+
"""Locate <img> tags with the stdlib parser instead of a regex.
|
|
318
|
+
|
|
319
|
+
Four regex attempts each traded one unparseable shape for another; the
|
|
320
|
+
ambiguity is inherent (`title="unclosed>` is indistinguishable from an
|
|
321
|
+
attribute containing `>`). HTMLParser resolves tag boundaries by the same
|
|
322
|
+
rules the browser and 头条 use, so what we rewrite is what they render.
|
|
323
|
+
"""
|
|
324
|
+
|
|
325
|
+
def __init__(self):
|
|
326
|
+
# convert_charrefs=False → attribute values stay as authored, matching
|
|
327
|
+
# what we re-emit into the HTML body.
|
|
328
|
+
super().__init__(convert_charrefs=False)
|
|
329
|
+
self.spans = [] # (start_offset, end_offset, attrs)
|
|
330
|
+
|
|
331
|
+
def _record(self, tag, attrs):
|
|
332
|
+
if tag.lower() != "img":
|
|
333
|
+
return
|
|
334
|
+
text = self.get_starttag_text() or ""
|
|
335
|
+
line, col = self.getpos()
|
|
336
|
+
start = self._line_starts[line - 1] + col
|
|
337
|
+
self.spans.append((start, start + len(text), attrs))
|
|
338
|
+
|
|
339
|
+
handle_starttag = _record
|
|
340
|
+
handle_startendtag = _record
|
|
341
|
+
|
|
342
|
+
def find(self, html):
|
|
343
|
+
# getpos() is (line, col); precompute line offsets to map to an index.
|
|
344
|
+
self._line_starts, pos = [0], 0
|
|
345
|
+
for ln in html.splitlines(keepends=True):
|
|
346
|
+
pos += len(ln)
|
|
347
|
+
self._line_starts.append(pos)
|
|
348
|
+
self.feed(html)
|
|
349
|
+
self.close()
|
|
350
|
+
return self.spans
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _first_attr(attrs, name):
|
|
354
|
+
"""First occurrence wins, as browsers do with a duplicated attribute."""
|
|
355
|
+
for k, v in attrs:
|
|
356
|
+
if k.lower() == name:
|
|
357
|
+
return v or ""
|
|
358
|
+
return None
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
class _NoRedirect(urllib.request.HTTPRedirectHandler):
|
|
362
|
+
# Refuse redirects on image fetches — a 30x could otherwise reach an internal
|
|
363
|
+
# host that _assert_public_url() never saw (SSRF).
|
|
364
|
+
def redirect_request(self, req, fp, code, msg, headers, newurl):
|
|
365
|
+
raise RuntimeError(f"image redirect blocked ({code}) -> {newurl[:80]}")
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
_IMG_OPENER = urllib.request.build_opener(_NoRedirect)
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _assert_public_url(url):
|
|
372
|
+
parts = urllib.parse.urlsplit(url)
|
|
373
|
+
if parts.scheme not in ("http", "https") or not parts.hostname:
|
|
374
|
+
raise RuntimeError(f"unsupported image URL: {url[:80]}")
|
|
375
|
+
try:
|
|
376
|
+
addrs = socket.getaddrinfo(parts.hostname, None)
|
|
377
|
+
except OSError as e:
|
|
378
|
+
raise RuntimeError(f"cannot resolve {parts.hostname}: {e}")
|
|
379
|
+
for info in addrs:
|
|
380
|
+
ip = ipaddress.ip_address(info[4][0])
|
|
381
|
+
if (ip.is_private or ip.is_loopback or ip.is_link_local
|
|
382
|
+
or ip.is_reserved or ip.is_multicast or ip.is_unspecified):
|
|
383
|
+
raise RuntimeError(f"blocked non-public image host: {parts.hostname}")
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _download_image(url):
|
|
387
|
+
_assert_public_url(url)
|
|
388
|
+
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
|
389
|
+
with _IMG_OPENER.open(req, timeout=30) as r:
|
|
390
|
+
data = r.read(MAX_IMG_BYTES + 1)
|
|
391
|
+
if len(data) > MAX_IMG_BYTES:
|
|
392
|
+
raise RuntimeError(f"image exceeds {MAX_IMG_BYTES} bytes")
|
|
393
|
+
return data
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _same_or_sub(host, suffix):
|
|
397
|
+
return host == suffix or host.endswith("." + suffix)
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _ext_of(url, default="png"):
|
|
401
|
+
tail = url.rsplit("/", 1)[-1].split("?")[0]
|
|
402
|
+
if "." in tail:
|
|
403
|
+
e = tail.rsplit(".", 1)[-1].lower()
|
|
404
|
+
if e in ("jpg", "jpeg", "png", "gif", "webp"):
|
|
405
|
+
return e
|
|
406
|
+
return default
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _multipart(field, filename, blob, ctype):
|
|
410
|
+
boundary = "----acedata" + "".join(random.choice("0123456789abcdef") for _ in range(20))
|
|
411
|
+
body = b"".join([
|
|
412
|
+
(f'--{boundary}\r\nContent-Disposition: form-data; name="{field}";'
|
|
413
|
+
f' filename="{filename}"\r\nContent-Type: {ctype}\r\n\r\n').encode(),
|
|
414
|
+
blob,
|
|
415
|
+
f"\r\n--{boundary}--\r\n".encode(),
|
|
416
|
+
])
|
|
417
|
+
return body, boundary
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def upload_image(jar, src) -> dict:
|
|
421
|
+
"""Upload one image to 头条's CDN, returning its {url, web_uri, width, ...}."""
|
|
422
|
+
img = _download_image(src)
|
|
423
|
+
ext = _ext_of(src)
|
|
424
|
+
mime = "image/jpeg" if ext in ("jpg", "jpeg") else f"image/{ext}"
|
|
425
|
+
# The upload form field MUST be `upfile` — `file`/`image` return 1053.
|
|
426
|
+
body, boundary = _multipart("upfile", f"image.{ext}", img, mime)
|
|
427
|
+
# This endpoint answers with a FLAT envelope (no `data` wrapper), unlike the
|
|
428
|
+
# rest of the creator API, so read the fields off the top level.
|
|
429
|
+
# nonfatal → a per-image failure raises instead of die()ing, so the caller's
|
|
430
|
+
# failure collector (and --drop-failed-images) actually gets to run.
|
|
431
|
+
env = api_envelope("POST", "/mp/agw/article_material/photo/upload_picture/?type=json", jar,
|
|
432
|
+
referer=f"{MP}/profile_v4/graphic/publish", body=body,
|
|
433
|
+
headers={"Content-Type": f"multipart/form-data; boundary={boundary}"},
|
|
434
|
+
nonfatal=True)
|
|
435
|
+
if not env.get("web_url") or not env.get("web_uri"):
|
|
436
|
+
raise RuntimeError(f"upload returned no image URL: {str(env)[:200]}")
|
|
437
|
+
return env
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _img_tag(info: dict, alt: str) -> str:
|
|
441
|
+
"""头条 needs the CDN attributes it returned, not a bare src — a plain
|
|
442
|
+
<img src> (or any external URL) makes publish fail with 7115 图片uri非法."""
|
|
443
|
+
return (
|
|
444
|
+
f'<img src="{_attr_url(info["web_url"])}"'
|
|
445
|
+
f' img_width="{int(info.get("width") or 0)}"'
|
|
446
|
+
f' img_height="{int(info.get("height") or 0)}"'
|
|
447
|
+
f' image_type="{int(info.get("image_type") or 1)}"'
|
|
448
|
+
f' mime_type="{_alt(str(info.get("mime_type") or "image/jpeg"))}"'
|
|
449
|
+
f' web_uri="{_alt(str(info["web_uri"]))}"'
|
|
450
|
+
f' alt="{_alt(alt)}">'
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def rehost_images(jar, html, drop_failed=False):
|
|
455
|
+
"""Rewrite every <img> in the rendered body to a 头条-hosted one.
|
|
456
|
+
|
|
457
|
+
An external <img> is not merely blocked — it makes 头条 reject the WHOLE
|
|
458
|
+
article (7115), so keeping the original is never an option. Default is to
|
|
459
|
+
abort the publish so the user's article is never silently posted without
|
|
460
|
+
its images; ``drop_failed`` opts into dropping them instead.
|
|
461
|
+
"""
|
|
462
|
+
failures = []
|
|
463
|
+
try:
|
|
464
|
+
spans = _ImgFinder().find(html)
|
|
465
|
+
except Exception as e: # noqa: BLE001 — malformed beyond parsing
|
|
466
|
+
die(f"could not parse the article HTML to find its images: {e}")
|
|
467
|
+
|
|
468
|
+
out, cursor = [], 0
|
|
469
|
+
for start, end, attrs in spans:
|
|
470
|
+
out.append(html[cursor:start])
|
|
471
|
+
cursor = end
|
|
472
|
+
tag = html[start:end]
|
|
473
|
+
src = _first_attr(attrs, "src")
|
|
474
|
+
if not src:
|
|
475
|
+
# `data-src`-only or valueless src — we cannot fetch it, and 头条
|
|
476
|
+
# would reject the article, so surface it instead of dropping it.
|
|
477
|
+
failures.append(f"{tag[:80]} (no usable src attribute)")
|
|
478
|
+
continue
|
|
479
|
+
# Attribute values arrive unescaped; re-escape alt on the way back out.
|
|
480
|
+
alt = _first_attr(attrs, "alt") or ""
|
|
481
|
+
host = (urllib.parse.urlsplit(src).hostname or "").lower()
|
|
482
|
+
if any(_same_or_sub(host, s) for s in _IMG_SKIP) and _first_attr(attrs, "web_uri"):
|
|
483
|
+
out.append(tag)
|
|
484
|
+
continue
|
|
485
|
+
try:
|
|
486
|
+
info = upload_image(jar, src)
|
|
487
|
+
sys.stderr.write(f"[img] rehosted {src[:60]} -> {info['web_uri']}\n")
|
|
488
|
+
out.append(_img_tag(info, _html.escape(alt, quote=False)))
|
|
489
|
+
except Exception as e: # noqa: BLE001 — collected, then reported below
|
|
490
|
+
failures.append(f"{src[:80]} ({e})")
|
|
491
|
+
out.append(html[cursor:])
|
|
492
|
+
result = "".join(out)
|
|
493
|
+
if failures and not drop_failed:
|
|
494
|
+
die("could not upload these images to 头条, and 头条 rejects any article "
|
|
495
|
+
"with an external image, so nothing was published:\n - "
|
|
496
|
+
+ "\n - ".join(failures)
|
|
497
|
+
+ "\nFix the image URLs, or re-run with --drop-failed-images to "
|
|
498
|
+
"publish without them.")
|
|
499
|
+
for f in failures:
|
|
500
|
+
sys.stderr.write(f"[img] DROPPED {f}\n")
|
|
501
|
+
return result
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
# ── Markdown → HTML (头条's `content` field is rendered HTML, not source) ──
|
|
505
|
+
|
|
506
|
+
_IMG_RE = re.compile(r"!\[([^\]]{0,500})\]\(([^)\s]{0,2000})\)")
|
|
507
|
+
_LINK_RE = re.compile(r"(?<!!)\[([^\]]{0,500})\]\(([^)\s]{0,2000})\)")
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def _attr_url(u):
|
|
511
|
+
# Strip C0 controls/space FIRST, then allow only http/https/mailto — otherwise
|
|
512
|
+
# `\x01javascript:` would slip past a literal scheme check yet be re-normalised
|
|
513
|
+
# to javascript: by the browser. Finally neutralise attribute-breaking quotes.
|
|
514
|
+
u = re.sub(r"[\x00-\x20\x7f]", "", u or "")
|
|
515
|
+
m = re.match(r"(?i)([a-z][a-z0-9+.\-]*):", u)
|
|
516
|
+
if m and m.group(1).lower() not in ("http", "https", "mailto"):
|
|
517
|
+
return "#"
|
|
518
|
+
return u.replace('"', "%22").replace("'", "%27")
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
def _alt(s):
|
|
522
|
+
return (s or "").replace('"', """).replace("'", "'")
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def _inline_md(t):
|
|
526
|
+
t = _html.escape(t, quote=False).replace("\x00", "")
|
|
527
|
+
# Stash code spans FIRST so emphasis/link/image markup inside `…` stays literal.
|
|
528
|
+
spans = []
|
|
529
|
+
|
|
530
|
+
def _stash(m):
|
|
531
|
+
spans.append("<code>" + m.group(1) + "</code>")
|
|
532
|
+
return f"\x00{len(spans) - 1}\x00"
|
|
533
|
+
|
|
534
|
+
t = re.sub(r"`([^`]+)`", _stash, t)
|
|
535
|
+
t = _IMG_RE.sub(lambda m: f'<img src="{_attr_url(m.group(2))}" alt="{_alt(m.group(1))}">', t)
|
|
536
|
+
t = _LINK_RE.sub(lambda m: f'<a href="{_attr_url(m.group(2))}">{m.group(1)}</a>', t)
|
|
537
|
+
t = re.sub(r"\*\*([^*]+)\*\*", r"<strong>\1</strong>", t)
|
|
538
|
+
t = re.sub(r"(?<!\*)\*([^*\n]+)\*(?!\*)", r"<em>\1</em>", t)
|
|
539
|
+
t = re.sub(r"\x00(\d+)\x00", lambda m: spans[int(m.group(1))], t)
|
|
540
|
+
return t
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _md_block_to_html(b):
|
|
544
|
+
b = b.strip("\n")
|
|
545
|
+
if not b.strip():
|
|
546
|
+
return None
|
|
547
|
+
f = b.lstrip()
|
|
548
|
+
m = re.match(r"(#{1,6})\s+(.*)", f)
|
|
549
|
+
if m:
|
|
550
|
+
lvl = len(m.group(1))
|
|
551
|
+
return f"<h{lvl}>{_inline_md(m.group(2).strip())}</h{lvl}>"
|
|
552
|
+
if f.startswith(">"):
|
|
553
|
+
inner = re.sub(r"^>\s?", "", b, flags=re.M).replace("\n", " ")
|
|
554
|
+
return "<blockquote><p>" + _inline_md(inner) + "</p></blockquote>"
|
|
555
|
+
if re.match(r"[-*+]\s+", f):
|
|
556
|
+
items = [_inline_md(re.sub(r"^[-*+]\s+", "", ln)) for ln in b.split("\n") if ln.strip()]
|
|
557
|
+
return "<ul>" + "".join(f"<li>{x}</li>" for x in items) + "</ul>"
|
|
558
|
+
if re.match(r"\d+\.\s+", f):
|
|
559
|
+
items = [_inline_md(re.sub(r"^\d+\.\s+", "", ln)) for ln in b.split("\n") if ln.strip()]
|
|
560
|
+
return "<ol>" + "".join(f"<li>{x}</li>" for x in items) + "</ol>"
|
|
561
|
+
only = re.fullmatch(r"!\[([^\]]{0,500})\]\(([^)\s]{0,2000})\)", f)
|
|
562
|
+
if only:
|
|
563
|
+
return f'<img src="{_attr_url(only.group(2))}" alt="{_alt(_html.escape(only.group(1), quote=False))}">'
|
|
564
|
+
return "<p>" + _inline_md(b.replace("\n", " ").strip()) + "</p>"
|
|
565
|
+
|
|
566
|
+
|
|
567
|
+
def md_to_html(src):
|
|
568
|
+
"""Minimal stdlib Markdown→HTML — 头条's body field is HTML, so raw markdown
|
|
569
|
+
would publish as literal `##` / `![]()` on one collapsed line."""
|
|
570
|
+
src = (src or "").strip()
|
|
571
|
+
res = []
|
|
572
|
+
# Pull fenced code out FIRST (it may itself contain blank lines) so the
|
|
573
|
+
# blank-line block splitter below can never tear a ``` … ``` fence apart.
|
|
574
|
+
for i, part in enumerate(re.split(r"(?ms)^(```.*?(?:\n```[ \t]*$|\Z))", src)):
|
|
575
|
+
if i % 2 == 1:
|
|
576
|
+
code = re.sub(r"\A```[^\n]*\n?", "", part)
|
|
577
|
+
code = re.sub(r"\n?```[ \t]*\Z", "", code)
|
|
578
|
+
res.append("<pre><code>" + _html.escape(code) + "</code></pre>")
|
|
579
|
+
continue
|
|
580
|
+
for b in re.split(r"\n[ \t]*\n", part):
|
|
581
|
+
block = _md_block_to_html(b)
|
|
582
|
+
if block:
|
|
583
|
+
res.append(block)
|
|
584
|
+
return "\n".join(res)
|
|
585
|
+
|
|
586
|
+
|
|
587
|
+
def _looks_like_markdown(s):
|
|
588
|
+
# Three-way classification (no full parser): block-level HTML ⇒ already an
|
|
589
|
+
# HTML document → pass through; else markdown markers ⇒ render; else
|
|
590
|
+
# inline-only HTML ⇒ pass through; else plain text ⇒ render.
|
|
591
|
+
s = s or ""
|
|
592
|
+
if re.search(
|
|
593
|
+
r"</?(?:p|div|h[1-6]|ul|ol|li|table|thead|tbody|tr|td|th|blockquote|"
|
|
594
|
+
r"pre|figure|figcaption|section|article|header|footer|nav|aside|hr|"
|
|
595
|
+
r"main|details|summary)\b", s, re.I,
|
|
596
|
+
):
|
|
597
|
+
return False
|
|
598
|
+
# ^-anchored (re.M) + bounded {0,N} repeats so the scan can't backtrack
|
|
599
|
+
# across newlines or to EOF (no quadratic scanning).
|
|
600
|
+
if re.search(
|
|
601
|
+
r"^#{1,6}\s|!\[[^\]]{0,500}\]\([^)]{0,2000}\)|^[ \t]*[-*+]\s|^[ \t]*\d+\.\s"
|
|
602
|
+
r"|\[[^\]]{0,500}\]\([^)]{0,2000}\)|`[^`]{1,500}`|\*\*[^*]{1,500}\*\*",
|
|
603
|
+
s, re.M,
|
|
604
|
+
):
|
|
605
|
+
return True
|
|
606
|
+
inline_html = re.search(
|
|
607
|
+
r"</?(?:a|strong|em|b|i|u|s|span|code|br|img|small|mark|sup|sub|"
|
|
608
|
+
r"video|audio|iframe)\b", s, re.I,
|
|
609
|
+
)
|
|
610
|
+
return not inline_html
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
# ── publish ─────────────────────────────────────────────────────────
|
|
614
|
+
|
|
615
|
+
PUBLISH_PATH = "/mp/agw/article/publish/?source=mp&type=article"
|
|
616
|
+
|
|
617
|
+
|
|
618
|
+
def _resolve_url(jar, pgc_id, draft):
|
|
619
|
+
"""Look the freshly-written article up in the list to return its real URL
|
|
620
|
+
(preview URL for a draft, public /item/ URL once published).
|
|
621
|
+
|
|
622
|
+
Runs AFTER the write succeeded, so it must never fail the command — losing
|
|
623
|
+
the pgc_id would read as a failed publish and invite a duplicate post.
|
|
624
|
+
"""
|
|
625
|
+
try:
|
|
626
|
+
items, _ = _list_page(jar, "draft" if draft else "all", 1, 20, nonfatal=True)
|
|
627
|
+
for it in items:
|
|
628
|
+
if str(it.get("pgc_id")) == str(pgc_id):
|
|
629
|
+
return it.get("article_url")
|
|
630
|
+
except (Exception, SystemExit): # noqa: BLE001 — URL is a nicety
|
|
631
|
+
pass
|
|
632
|
+
if draft:
|
|
633
|
+
return f"{MP}/preview_article/?pgc_id={pgc_id}"
|
|
634
|
+
return f"https://www.toutiao.com/item/{pgc_id}/"
|
|
635
|
+
|
|
636
|
+
|
|
637
|
+
def cmd_publish(jar, args):
|
|
638
|
+
if not args.title:
|
|
639
|
+
die("--title is required")
|
|
640
|
+
# 头条 rejects out-of-range titles server-side; fail early with a clear message.
|
|
641
|
+
if not 2 <= len(args.title) <= 30:
|
|
642
|
+
die(f"头条 titles must be 2–30 characters; got {len(args.title)}")
|
|
643
|
+
if not args.content_file and args.content is None:
|
|
644
|
+
die("provide --content-file <path.md> or --content <markdown>")
|
|
645
|
+
content = args.content
|
|
646
|
+
if args.content_file:
|
|
647
|
+
try:
|
|
648
|
+
with open(args.content_file, encoding="utf-8") as f:
|
|
649
|
+
content = f.read()
|
|
650
|
+
except OSError as e:
|
|
651
|
+
die(f"cannot read --content-file: {e}")
|
|
652
|
+
content = content or ""
|
|
653
|
+
|
|
654
|
+
if not CONFIRM:
|
|
655
|
+
out({
|
|
656
|
+
"dry_run": True, "command": "publish", "platform": "toutiao",
|
|
657
|
+
"title": args.title, "draft_only": args.draft_only,
|
|
658
|
+
"content_bytes": len(content),
|
|
659
|
+
"note": "头条 content is Markdown (converted to HTML for the body). "
|
|
660
|
+
"Re-run with --confirm as the LAST argument to actually write. "
|
|
661
|
+
"Without --draft-only it publishes a PUBLIC article on the "
|
|
662
|
+
"user's real 头条号 and goes through 审核.",
|
|
663
|
+
})
|
|
664
|
+
return
|
|
665
|
+
|
|
666
|
+
# Writes need the csrftoken cookie echoed as a header; without it 头条 fails
|
|
667
|
+
# deep in its API with an opaque code instead of "reconnect".
|
|
668
|
+
if not cookie_value(jar, "csrftoken"):
|
|
669
|
+
die("the 今日头条 cookie jar has no `csrftoken` — it is incomplete or "
|
|
670
|
+
"expired. Reconnect at https://auth.acedata.cloud/user/connections.")
|
|
671
|
+
|
|
672
|
+
# Render to HTML FIRST, then rewrite <img> tags — 头条 rejects the whole
|
|
673
|
+
# article (7115) if any image isn't hosted on its own CDN with web_uri.
|
|
674
|
+
body_html = md_to_html(content) if _looks_like_markdown(content) else content
|
|
675
|
+
if not args.no_rehost_images:
|
|
676
|
+
body_html = rehost_images(jar, body_html, drop_failed=args.drop_failed_images)
|
|
677
|
+
|
|
678
|
+
form = urllib.parse.urlencode({
|
|
679
|
+
"title": args.title,
|
|
680
|
+
"content": body_html,
|
|
681
|
+
# save=1 → draft; save=0 → submit for 审核 and publish.
|
|
682
|
+
"save": "1" if args.draft_only else "0",
|
|
683
|
+
# 0 = no ad. Other values need 广告/自营 permissions most accounts lack.
|
|
684
|
+
"article_ad_type": "0",
|
|
685
|
+
})
|
|
686
|
+
d = api("POST", PUBLISH_PATH, jar, referer=f"{MP}/profile_v4/graphic/publish",
|
|
687
|
+
body=form, headers={"Content-Type": "application/x-www-form-urlencoded"})
|
|
688
|
+
pgc_id = (d or {}).get("pgc_id")
|
|
689
|
+
if not pgc_id:
|
|
690
|
+
die(f"publish returned no pgc_id: {str(d)[:300]}")
|
|
691
|
+
out({
|
|
692
|
+
"ok": True,
|
|
693
|
+
"draft_only": bool(args.draft_only),
|
|
694
|
+
"published": not args.draft_only,
|
|
695
|
+
"pgc_id": str(pgc_id),
|
|
696
|
+
"url": _resolve_url(jar, pgc_id, args.draft_only),
|
|
697
|
+
"note": None if args.draft_only else "头条 reviews new articles (审核); "
|
|
698
|
+
"the public URL goes live once it passes.",
|
|
699
|
+
})
|
|
700
|
+
|
|
701
|
+
|
|
702
|
+
COMMANDS = {
|
|
703
|
+
"whoami": cmd_whoami,
|
|
704
|
+
"articles": cmd_articles,
|
|
705
|
+
"article": cmd_article,
|
|
706
|
+
"publish": cmd_publish,
|
|
707
|
+
}
|
|
708
|
+
|
|
709
|
+
|
|
710
|
+
def main() -> None:
|
|
711
|
+
p = argparse.ArgumentParser(prog="toutiao.py", description="今日头条 cookie CLI")
|
|
712
|
+
sub = p.add_subparsers(dest="command", required=True)
|
|
713
|
+
sub.add_parser("whoami", help="show the logged-in 头条号")
|
|
714
|
+
sp = sub.add_parser("articles", help="list the user's articles + stats")
|
|
715
|
+
sp.add_argument("--limit", type=int, default=20)
|
|
716
|
+
sp.add_argument("--status", default="all",
|
|
717
|
+
choices=["all", "draft", "published", "reviewing", "failed"])
|
|
718
|
+
sp = sub.add_parser("article", help="one article's stats")
|
|
719
|
+
sp.add_argument("id", help="pgc_id / item_id")
|
|
720
|
+
sp = sub.add_parser("publish", help="create/publish an article (GATED by trailing --confirm)")
|
|
721
|
+
sp.add_argument("--title", help="2–30 characters")
|
|
722
|
+
sp.add_argument("--content", help="Markdown content inline")
|
|
723
|
+
sp.add_argument("--content-file", help="path to a Markdown file")
|
|
724
|
+
sp.add_argument("--draft-only", action="store_true",
|
|
725
|
+
help="save a private draft; do NOT go public")
|
|
726
|
+
sp.add_argument("--no-rehost-images", action="store_true",
|
|
727
|
+
help="keep external image URLs as-is (头条 will reject the article)")
|
|
728
|
+
sp.add_argument("--drop-failed-images", action="store_true",
|
|
729
|
+
help="publish without any image that fails to upload, "
|
|
730
|
+
"instead of aborting")
|
|
731
|
+
args = p.parse_args(ARGV)
|
|
732
|
+
jar = load_cookies()
|
|
733
|
+
COMMANDS[args.command](jar, args)
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
if __name__ == "__main__":
|
|
737
|
+
main()
|