dsh-ab-wechat-scrape 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +330 -0
- package/cordis.patch.yml +8 -0
- package/lib/clean.d.ts +32 -0
- package/lib/clean.d.ts.map +1 -0
- package/lib/clean.js +193 -0
- package/lib/clean.js.map +1 -0
- package/lib/document.d.ts +27 -0
- package/lib/document.d.ts.map +1 -0
- package/lib/document.js +79 -0
- package/lib/document.js.map +1 -0
- package/lib/fetch.d.ts +84 -0
- package/lib/fetch.d.ts.map +1 -0
- package/lib/fetch.js +238 -0
- package/lib/fetch.js.map +1 -0
- package/lib/filename.d.ts +36 -0
- package/lib/filename.d.ts.map +1 -0
- package/lib/filename.js +92 -0
- package/lib/filename.js.map +1 -0
- package/lib/html.d.ts +114 -0
- package/lib/html.d.ts.map +1 -0
- package/lib/html.js +402 -0
- package/lib/html.js.map +1 -0
- package/lib/index.d.ts +219 -0
- package/lib/index.d.ts.map +1 -0
- package/lib/index.js +2523 -0
- package/lib/index.js.map +1 -0
- package/lib/list.d.ts +55 -0
- package/lib/list.d.ts.map +1 -0
- package/lib/list.js +191 -0
- package/lib/list.js.map +1 -0
- package/lib/markdown.d.ts +20 -0
- package/lib/markdown.d.ts.map +1 -0
- package/lib/markdown.js +250 -0
- package/lib/markdown.js.map +1 -0
- package/lib/render.d.ts +103 -0
- package/lib/render.d.ts.map +1 -0
- package/lib/render.js +209 -0
- package/lib/render.js.map +1 -0
- package/lib/store.d.ts +50 -0
- package/lib/store.d.ts.map +1 -0
- package/lib/store.js +173 -0
- package/lib/store.js.map +1 -0
- package/lib/types.d.ts +133 -0
- package/lib/types.d.ts.map +1 -0
- package/lib/types.js +8 -0
- package/lib/types.js.map +1 -0
- package/package.json +91 -0
- package/tsconfig.json +30 -0
- package/tsdown.config.ts +18 -0
package/lib/index.js
ADDED
|
@@ -0,0 +1,2523 @@
|
|
|
1
|
+
import { isAbsolute, join, resolve } from "node:path";
|
|
2
|
+
import z from "@deepseek-ai/schemastery";
|
|
3
|
+
import { defineTool } from "@deepseek-ai/dsh-tools";
|
|
4
|
+
//#region lib/html.js
|
|
5
|
+
/**
|
|
6
|
+
* Pure page parsing for mp.weixin.qq.com articles: HTML entity and inline-script
|
|
7
|
+
* string decoding, metadata lookup, balanced body extraction, interstitial
|
|
8
|
+
* redirect resolution, and the detection of block, verification, and withdrawn
|
|
9
|
+
* pages. No I/O and no DOM library, so tests drive every rule directly.
|
|
10
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/html
|
|
11
|
+
*/
|
|
12
|
+
const NAMED_ENTITIES = {
|
|
13
|
+
amp: "&",
|
|
14
|
+
lt: "<",
|
|
15
|
+
gt: ">",
|
|
16
|
+
quot: "\"",
|
|
17
|
+
apos: "'",
|
|
18
|
+
nbsp: " ",
|
|
19
|
+
ensp: " ",
|
|
20
|
+
emsp: " ",
|
|
21
|
+
ldquo: "“",
|
|
22
|
+
rdquo: "”",
|
|
23
|
+
lsquo: "‘",
|
|
24
|
+
rsquo: "’",
|
|
25
|
+
hellip: "…",
|
|
26
|
+
mdash: "—",
|
|
27
|
+
ndash: "–",
|
|
28
|
+
middot: "·",
|
|
29
|
+
copy: "©"
|
|
30
|
+
};
|
|
31
|
+
/** Block-page markers WeChat serves instead of an article. */
|
|
32
|
+
const BLOCK_MARKERS = [
|
|
33
|
+
"环境异常",
|
|
34
|
+
"去验证",
|
|
35
|
+
"请在微信客户端打开链接",
|
|
36
|
+
"访问过于频繁",
|
|
37
|
+
"操作频繁",
|
|
38
|
+
"参数错误",
|
|
39
|
+
"完成验证",
|
|
40
|
+
"wappoc_appmsg",
|
|
41
|
+
"poc_token"
|
|
42
|
+
];
|
|
43
|
+
/** Response-URL shapes WeChat redirects a throttled or unverified request to. */
|
|
44
|
+
const BLOCK_URL_PATTERNS = [
|
|
45
|
+
/\/mp\/wappoc_appmsg/i,
|
|
46
|
+
/\/mp\/verify/i,
|
|
47
|
+
/\/mp\/captcha/i,
|
|
48
|
+
/poc_token=/i
|
|
49
|
+
];
|
|
50
|
+
/**
|
|
51
|
+
* Page markers WeChat serves when an article is gone rather than withheld:
|
|
52
|
+
* the publisher withdrew it, a reviewer removed it, or the account itself was
|
|
53
|
+
* closed. Rendering cannot recover any of these, and none of them is an
|
|
54
|
+
* article, so a page carrying one is never saved.
|
|
55
|
+
*/
|
|
56
|
+
const UNAVAILABLE_MARKERS = [
|
|
57
|
+
"该内容已被发布者删除",
|
|
58
|
+
"该内容因违规无法查看",
|
|
59
|
+
"该内容已被删除",
|
|
60
|
+
"此内容因违规无法查看",
|
|
61
|
+
"此内容已被发布者删除",
|
|
62
|
+
"此内容发送失败无法查看",
|
|
63
|
+
"该账号已被屏蔽",
|
|
64
|
+
"该公众号已注销",
|
|
65
|
+
"文章已被删除",
|
|
66
|
+
"内容不存在",
|
|
67
|
+
"链接已失效"
|
|
68
|
+
];
|
|
69
|
+
/** One attribute of an opening tag, double- or single-quoted. */
|
|
70
|
+
const ATTRIBUTE = /([a-zA-Z_:][-a-zA-Z0-9_:.]*)\s*=\s*(?:"([^"]*)"|'([^']*)')/g;
|
|
71
|
+
/** One opening tag, closing tag, comment, or run of text. */
|
|
72
|
+
const TAG = /<([a-zA-Z][a-zA-Z0-9-]*)((?:"[^"]*"|'[^']*'|[^>"'])*?)(\/?)>/g;
|
|
73
|
+
/** The inline-script shapes that hand the browser the real article link. */
|
|
74
|
+
const REDIRECT_PATTERNS = [
|
|
75
|
+
/(?:var\s+)?msg_link\s*[:=]\s*"((?:\\[\s\S]|[^"\\])*)"/,
|
|
76
|
+
/(?:var\s+)?msg_link\s*[:=]\s*'((?:\\[\s\S]|[^'\\])*)'/,
|
|
77
|
+
/window\.location\.replace\(\s*"([^"]+)"\s*\)/,
|
|
78
|
+
/window\.location\.replace\(\s*'([^']+)'\s*\)/,
|
|
79
|
+
/location\.href\s*[:=]\s*"([^"]+)"/,
|
|
80
|
+
/location\.href\s*[:=]\s*'([^']+)'/
|
|
81
|
+
];
|
|
82
|
+
/** One code point as a character, or the original text when it is out of range. */
|
|
83
|
+
function codePoint(code, fallback) {
|
|
84
|
+
return Number.isInteger(code) && code >= 0 && code <= 1114111 ? String.fromCodePoint(code) : fallback;
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Decode named and numeric HTML entities.
|
|
88
|
+
* @param text - text carrying entity references.
|
|
89
|
+
* @returns the same text with every known reference replaced.
|
|
90
|
+
*/
|
|
91
|
+
function decodeEntities(text) {
|
|
92
|
+
return text.replace(/&(#x[0-9a-f]+|#[0-9]+|[a-z]+);/gi, (match, body) => {
|
|
93
|
+
if (/^#x/i.test(body)) return codePoint(Number.parseInt(body.slice(2), 16), match);
|
|
94
|
+
if (body.startsWith("#")) return codePoint(Number.parseInt(body.slice(1), 10), match);
|
|
95
|
+
const named = NAMED_ENTITIES[body.toLowerCase()];
|
|
96
|
+
return named === void 0 ? match : named;
|
|
97
|
+
});
|
|
98
|
+
}
|
|
99
|
+
/**
|
|
100
|
+
* Decode the string-literal escapes WeChat embeds in its inline scripts.
|
|
101
|
+
* @param raw - the literal body, without its surrounding quotes.
|
|
102
|
+
* @returns the decoded text.
|
|
103
|
+
*/
|
|
104
|
+
function decodeJsString(raw) {
|
|
105
|
+
return decodeEntities(raw.replace(/\\(u[0-9a-f]{4}|x[0-9a-f]{2}|[\s\S])/gi, (match, escape) => {
|
|
106
|
+
const kind = escape.charAt(0);
|
|
107
|
+
if (kind === "u" || kind === "x") return codePoint(Number.parseInt(escape.slice(1), 16), match);
|
|
108
|
+
if (escape === "n") return "\n";
|
|
109
|
+
if (escape === "r") return "\r";
|
|
110
|
+
if (escape === "t") return " ";
|
|
111
|
+
return escape;
|
|
112
|
+
}));
|
|
113
|
+
}
|
|
114
|
+
/**
|
|
115
|
+
* Read one opening tag's attributes into a lookup keyed by lower-case name.
|
|
116
|
+
* @param tag - the opening tag text.
|
|
117
|
+
* @returns decoded attribute values; the last occurrence of a name wins.
|
|
118
|
+
*/
|
|
119
|
+
function parseAttributes(tag) {
|
|
120
|
+
const attributes = /* @__PURE__ */ new Map();
|
|
121
|
+
ATTRIBUTE.lastIndex = 0;
|
|
122
|
+
let match;
|
|
123
|
+
while ((match = ATTRIBUTE.exec(tag)) !== null) {
|
|
124
|
+
const name = match[1];
|
|
125
|
+
const value = match[2] ?? match[3];
|
|
126
|
+
if (name === void 0 || value === void 0) continue;
|
|
127
|
+
attributes.set(name.toLowerCase(), decodeEntities(value).trim());
|
|
128
|
+
}
|
|
129
|
+
return attributes;
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* Read one meta tag's content by its property or name.
|
|
133
|
+
* @param html - complete page HTML.
|
|
134
|
+
* @param key - the property or name to match, case-insensitive.
|
|
135
|
+
* @returns the decoded content, or an empty string when no such tag exists.
|
|
136
|
+
*/
|
|
137
|
+
function metaContent(html, key) {
|
|
138
|
+
const wanted = key.toLowerCase();
|
|
139
|
+
const tags = /<meta\b[^>]*>/gi;
|
|
140
|
+
let match;
|
|
141
|
+
while ((match = tags.exec(html)) !== null) {
|
|
142
|
+
const attributes = parseAttributes(match[0]);
|
|
143
|
+
if ((attributes.get("property") ?? attributes.get("name") ?? "").toLowerCase() === wanted) return attributes.get("content") ?? "";
|
|
144
|
+
}
|
|
145
|
+
return "";
|
|
146
|
+
}
|
|
147
|
+
/** Escape one literal so it can be embedded in a regular expression. */
|
|
148
|
+
function escapeRegExp(text) {
|
|
149
|
+
return text.replace(/[.*+?^(){}|[\]\\]/g, "\\$&");
|
|
150
|
+
}
|
|
151
|
+
/** The first quoted literal inside one script expression, decoded. */
|
|
152
|
+
function firstQuoted(expression) {
|
|
153
|
+
const match = /"((?:\\[\s\S]|[^"\\])*)"|'((?:\\[\s\S]|[^'\\])*)'/.exec(expression);
|
|
154
|
+
const value = match?.[1] ?? match?.[2];
|
|
155
|
+
return value === void 0 ? "" : decodeJsString(value).trim();
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Read one inline-script variable by name. The page assigns both bare literals
|
|
159
|
+
* and wrapped ones (htmlDecode("..."), "....html(false)"), so the whole
|
|
160
|
+
* right-hand expression is read and its first quoted literal decoded.
|
|
161
|
+
* @param html - complete page HTML.
|
|
162
|
+
* @param variable - the variable to read, case-sensitive.
|
|
163
|
+
* @returns the decoded value, or an empty string when the page carries none.
|
|
164
|
+
*/
|
|
165
|
+
function jsString(html, variable) {
|
|
166
|
+
const pattern = new RegExp("\\bvar\\s+" + escapeRegExp(variable) + "\\s*=\\s*([^\\n;]*)", "g");
|
|
167
|
+
let match;
|
|
168
|
+
while ((match = pattern.exec(html)) !== null) {
|
|
169
|
+
const value = firstQuoted(match[1] ?? "");
|
|
170
|
+
if (value !== "") return value;
|
|
171
|
+
}
|
|
172
|
+
return "";
|
|
173
|
+
}
|
|
174
|
+
/**
|
|
175
|
+
* Extract the inner HTML of the first element carrying one id. Tag depth is
|
|
176
|
+
* counted, so nested elements of the same name cannot end the slice early.
|
|
177
|
+
* @param html - complete page HTML.
|
|
178
|
+
* @param id - the id attribute value the opening tag must carry.
|
|
179
|
+
* @returns the inner HTML, or an empty string when no such element exists.
|
|
180
|
+
*/
|
|
181
|
+
function elementById(html, id) {
|
|
182
|
+
TAG.lastIndex = 0;
|
|
183
|
+
let match;
|
|
184
|
+
while ((match = TAG.exec(html)) !== null) {
|
|
185
|
+
const name = match[1];
|
|
186
|
+
if (name === void 0) continue;
|
|
187
|
+
if (parseAttributes(match[0]).get("id") !== id) continue;
|
|
188
|
+
const element = name.toLowerCase();
|
|
189
|
+
const contentStart = match.index + match[0].length;
|
|
190
|
+
const walk = new RegExp("<" + element + "\\b|</" + element + "\\s*>", "gi");
|
|
191
|
+
walk.lastIndex = contentStart;
|
|
192
|
+
let depth = 1;
|
|
193
|
+
let step;
|
|
194
|
+
while ((step = walk.exec(html)) !== null) if (step[0].startsWith("</")) {
|
|
195
|
+
depth -= 1;
|
|
196
|
+
if (depth === 0) return html.slice(contentStart, step.index);
|
|
197
|
+
} else depth += 1;
|
|
198
|
+
return html.slice(contentStart);
|
|
199
|
+
}
|
|
200
|
+
return "";
|
|
201
|
+
}
|
|
202
|
+
/**
|
|
203
|
+
* Decode entities, drop tags, and collapse whitespace into readable text.
|
|
204
|
+
* @param html - markup to flatten.
|
|
205
|
+
* @returns the element text with every run of whitespace collapsed to one space.
|
|
206
|
+
*/
|
|
207
|
+
function textOf(html) {
|
|
208
|
+
return decodeEntities(html.replace(/<[^>]*>/g, " ")).replace(/\s+/g, " ").trim();
|
|
209
|
+
}
|
|
210
|
+
/**
|
|
211
|
+
* Resolve the real article URL an interstitial or short link hands the browser.
|
|
212
|
+
* @param html - the page that came back for the requested link.
|
|
213
|
+
* @returns an mp.weixin.qq.com article URL, or an empty string when none is named.
|
|
214
|
+
*/
|
|
215
|
+
function findRedirectUrl(html) {
|
|
216
|
+
for (const pattern of REDIRECT_PATTERNS) {
|
|
217
|
+
const raw = pattern.exec(html)?.[1];
|
|
218
|
+
if (raw === void 0 || raw.length === 0) continue;
|
|
219
|
+
const url = decodeJsString(raw).trim();
|
|
220
|
+
if (isMpArticleUrl(url)) return url;
|
|
221
|
+
}
|
|
222
|
+
return "";
|
|
223
|
+
}
|
|
224
|
+
/**
|
|
225
|
+
* Whether a link names an official-account article host.
|
|
226
|
+
* @param url - the candidate link.
|
|
227
|
+
* @returns true for an http(s) mp.weixin.qq.com link.
|
|
228
|
+
*/
|
|
229
|
+
function isMpArticleUrl(url) {
|
|
230
|
+
let parsed;
|
|
231
|
+
try {
|
|
232
|
+
parsed = new URL(url);
|
|
233
|
+
} catch {
|
|
234
|
+
return false;
|
|
235
|
+
}
|
|
236
|
+
return (parsed.protocol === "https:" || parsed.protocol === "http:") && (parsed.hostname === "mp.weixin.qq.com" || parsed.hostname.endsWith(".mp.weixin.qq.com"));
|
|
237
|
+
}
|
|
238
|
+
/**
|
|
239
|
+
* Whether a link names one official-account article rather than another page on
|
|
240
|
+
* the same host. A collection page lives under /mp/appmsgalbum and an article
|
|
241
|
+
* under /s, so the path decides.
|
|
242
|
+
* @param url - the candidate link.
|
|
243
|
+
* @returns true for an http(s) mp.weixin.qq.com /s link.
|
|
244
|
+
*/
|
|
245
|
+
function isMpArticleLink(url) {
|
|
246
|
+
if (!isMpArticleUrl(url)) return false;
|
|
247
|
+
const path = new URL(url).pathname;
|
|
248
|
+
return path === "/s" || path.startsWith("/s/");
|
|
249
|
+
}
|
|
250
|
+
/**
|
|
251
|
+
* The block-page marker a page carries.
|
|
252
|
+
* @param html - complete page HTML.
|
|
253
|
+
* @returns the marker found in the page head, or an empty string for a normal page.
|
|
254
|
+
*/
|
|
255
|
+
function detectBlock(html) {
|
|
256
|
+
const head = html.slice(0, 4e4);
|
|
257
|
+
for (const marker of BLOCK_MARKERS) if (head.includes(marker)) return marker;
|
|
258
|
+
return "";
|
|
259
|
+
}
|
|
260
|
+
/**
|
|
261
|
+
* The verification redirect a response URL names, which the body may not mention.
|
|
262
|
+
* @param url - the final response URL.
|
|
263
|
+
* @returns the matched URL fragment, or an empty string for a normal article URL.
|
|
264
|
+
*/
|
|
265
|
+
function detectBlockUrl(url) {
|
|
266
|
+
for (const pattern of BLOCK_URL_PATTERNS) {
|
|
267
|
+
const match = pattern.exec(url);
|
|
268
|
+
if (match !== null) return match[0];
|
|
269
|
+
}
|
|
270
|
+
return "";
|
|
271
|
+
}
|
|
272
|
+
/**
|
|
273
|
+
* The withdrawal marker a page carries.
|
|
274
|
+
* @param html - complete page HTML.
|
|
275
|
+
* @returns the marker found in the page head, or an empty string for a page that
|
|
276
|
+
* is not reporting a withdrawn or removed article.
|
|
277
|
+
*/
|
|
278
|
+
function detectUnavailable(html) {
|
|
279
|
+
const head = html.slice(0, 4e4);
|
|
280
|
+
for (const marker of UNAVAILABLE_MARKERS) if (head.includes(marker)) return marker;
|
|
281
|
+
return "";
|
|
282
|
+
}
|
|
283
|
+
/**
|
|
284
|
+
* Collect the content image URLs in document order, preferring lazy data-src.
|
|
285
|
+
* @param bodyHtml - the article body HTML.
|
|
286
|
+
* @returns absolute http(s) image URLs, de-duplicated.
|
|
287
|
+
*/
|
|
288
|
+
function extractImages(bodyHtml) {
|
|
289
|
+
const images = [];
|
|
290
|
+
const seen = /* @__PURE__ */ new Set();
|
|
291
|
+
const tags = /<img\b[^>]*>/gi;
|
|
292
|
+
let match;
|
|
293
|
+
while ((match = tags.exec(bodyHtml)) !== null) {
|
|
294
|
+
const attributes = parseAttributes(match[0]);
|
|
295
|
+
const src = attributes.get("data-src") ?? attributes.get("src") ?? "";
|
|
296
|
+
if (src === "" || src.startsWith("data:") || !/^https?:/i.test(src) || seen.has(src)) continue;
|
|
297
|
+
seen.add(src);
|
|
298
|
+
images.push(src);
|
|
299
|
+
}
|
|
300
|
+
return images;
|
|
301
|
+
}
|
|
302
|
+
/**
|
|
303
|
+
* Normalize a page timestamp to ISO 8601.
|
|
304
|
+
* @param raw - the page's own timestamp text.
|
|
305
|
+
* @returns an ISO timestamp, the unchanged text when it is not a time, or an empty string.
|
|
306
|
+
*/
|
|
307
|
+
function toIsoTime(raw) {
|
|
308
|
+
const trimmed = raw.trim();
|
|
309
|
+
if (trimmed === "") return "";
|
|
310
|
+
if (/^\d{9,11}$/.test(trimmed)) return (/* @__PURE__ */ new Date(Number.parseInt(trimmed, 10) * 1e3)).toISOString();
|
|
311
|
+
const parsed = new Date(trimmed);
|
|
312
|
+
return Number.isNaN(parsed.getTime()) ? trimmed : parsed.toISOString();
|
|
313
|
+
}
|
|
314
|
+
/** The first non-empty value, trimmed. */
|
|
315
|
+
function firstNonEmpty(values) {
|
|
316
|
+
for (const value of values) {
|
|
317
|
+
const trimmed = value.trim();
|
|
318
|
+
if (trimmed !== "") return trimmed;
|
|
319
|
+
}
|
|
320
|
+
return "";
|
|
321
|
+
}
|
|
322
|
+
/**
|
|
323
|
+
* Parse one page into article facts, a block report, or an empty result.
|
|
324
|
+
* @param html - complete page HTML.
|
|
325
|
+
* @returns the extraction outcome the caller turns into the canonical value.
|
|
326
|
+
*/
|
|
327
|
+
function extractArticle(html) {
|
|
328
|
+
const marker = detectBlock(html);
|
|
329
|
+
if (marker !== "") return {
|
|
330
|
+
kind: "blocked",
|
|
331
|
+
marker
|
|
332
|
+
};
|
|
333
|
+
const bodyHtml = elementById(html, "js_content");
|
|
334
|
+
if (bodyHtml.trim() === "") {
|
|
335
|
+
const reason = detectUnavailable(html);
|
|
336
|
+
if (reason !== "") return {
|
|
337
|
+
kind: "unavailable",
|
|
338
|
+
reason
|
|
339
|
+
};
|
|
340
|
+
}
|
|
341
|
+
const title = firstNonEmpty([
|
|
342
|
+
metaContent(html, "og:title"),
|
|
343
|
+
jsString(html, "msg_title"),
|
|
344
|
+
textOf(elementById(html, "activity-name"))
|
|
345
|
+
]);
|
|
346
|
+
const account = firstNonEmpty([
|
|
347
|
+
jsString(html, "nickname"),
|
|
348
|
+
textOf(elementById(html, "js_name")),
|
|
349
|
+
metaContent(html, "og:site_name")
|
|
350
|
+
]);
|
|
351
|
+
const author = firstNonEmpty([jsString(html, "author"), metaContent(html, "author")]);
|
|
352
|
+
const publishTime = firstNonEmpty([toIsoTime(jsString(html, "ct")), textOf(elementById(html, "publish_time"))]);
|
|
353
|
+
const digest = firstNonEmpty([jsString(html, "msg_desc"), metaContent(html, "description")]);
|
|
354
|
+
const cover = firstNonEmpty([jsString(html, "msg_cdn_url"), metaContent(html, "og:image")]);
|
|
355
|
+
if (bodyHtml.trim() === "" && title === "") return { kind: "empty" };
|
|
356
|
+
return {
|
|
357
|
+
kind: "article",
|
|
358
|
+
article: {
|
|
359
|
+
title,
|
|
360
|
+
account,
|
|
361
|
+
author,
|
|
362
|
+
publishTime,
|
|
363
|
+
digest,
|
|
364
|
+
cover,
|
|
365
|
+
bodyHtml,
|
|
366
|
+
images: extractImages(bodyHtml)
|
|
367
|
+
}
|
|
368
|
+
};
|
|
369
|
+
}
|
|
370
|
+
//#endregion
|
|
371
|
+
//#region lib/clean.js
|
|
372
|
+
/**
|
|
373
|
+
* Promotional-block removal for WeChat article bodies. WeChat mixes ad
|
|
374
|
+
* components, follow prompts, QR codes, and trailing "read more" lists into the
|
|
375
|
+
* same body element as the article, so the body is filtered before it is
|
|
376
|
+
* converted. Two rules run over the body HTML: a structural rule drops the
|
|
377
|
+
* elements WeChat marks as advertising, and a text rule drops a short block
|
|
378
|
+
* whose whole text is a follow or promotion prompt. The text rule never drops a
|
|
379
|
+
* long block, so a section that carries the article cannot match it. Both rules
|
|
380
|
+
* are the built-in ones; a deployment widens the text rule through
|
|
381
|
+
* {@link CleanPolicy.patterns} and moves its length ceiling through
|
|
382
|
+
* {@link CleanPolicy.maxPromoChars}.
|
|
383
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/clean
|
|
384
|
+
*/
|
|
385
|
+
/** One opening tag, closing tag, comment, or run of text. */
|
|
386
|
+
const TOKEN$1 = /<([a-zA-Z][a-zA-Z0-9-]*)((?:"[^"]*"|'[^']*'|[^>"'])*?)(\/?)>|<\/([a-zA-Z][a-zA-Z0-9-]*)>|<!--[\s\S]*?-->|[^<]+/g;
|
|
387
|
+
/** Elements that never have children, so they never open a subtree. */
|
|
388
|
+
const VOID_ELEMENTS = /* @__PURE__ */ new Set([
|
|
389
|
+
"area",
|
|
390
|
+
"base",
|
|
391
|
+
"br",
|
|
392
|
+
"col",
|
|
393
|
+
"embed",
|
|
394
|
+
"hr",
|
|
395
|
+
"img",
|
|
396
|
+
"input",
|
|
397
|
+
"link",
|
|
398
|
+
"meta",
|
|
399
|
+
"param",
|
|
400
|
+
"source",
|
|
401
|
+
"track",
|
|
402
|
+
"wbr"
|
|
403
|
+
]);
|
|
404
|
+
/** Elements that start a block of prose, the unit the text rule drops. */
|
|
405
|
+
const BLOCK_ELEMENTS$1 = /* @__PURE__ */ new Set([
|
|
406
|
+
"address",
|
|
407
|
+
"article",
|
|
408
|
+
"aside",
|
|
409
|
+
"blockquote",
|
|
410
|
+
"dd",
|
|
411
|
+
"div",
|
|
412
|
+
"dl",
|
|
413
|
+
"dt",
|
|
414
|
+
"figcaption",
|
|
415
|
+
"figure",
|
|
416
|
+
"footer",
|
|
417
|
+
"h1",
|
|
418
|
+
"h2",
|
|
419
|
+
"h3",
|
|
420
|
+
"h4",
|
|
421
|
+
"h5",
|
|
422
|
+
"h6",
|
|
423
|
+
"header",
|
|
424
|
+
"li",
|
|
425
|
+
"main",
|
|
426
|
+
"nav",
|
|
427
|
+
"ol",
|
|
428
|
+
"p",
|
|
429
|
+
"section",
|
|
430
|
+
"table",
|
|
431
|
+
"tbody",
|
|
432
|
+
"td",
|
|
433
|
+
"tfoot",
|
|
434
|
+
"th",
|
|
435
|
+
"thead",
|
|
436
|
+
"tr",
|
|
437
|
+
"ul"
|
|
438
|
+
]);
|
|
439
|
+
/** Custom elements and frames WeChat fills with advertising. */
|
|
440
|
+
const AD_TAGS = /* @__PURE__ */ new Set([
|
|
441
|
+
"iframe",
|
|
442
|
+
"mp-common-mpa",
|
|
443
|
+
"mp-common-profile",
|
|
444
|
+
"mpcpc",
|
|
445
|
+
"mpa-ad",
|
|
446
|
+
"mp-ad",
|
|
447
|
+
"mp-common-mpa-ad"
|
|
448
|
+
]);
|
|
449
|
+
/** Element ids WeChat gives its follow prompts, QR codes, reward widgets, and ad slots. */
|
|
450
|
+
const AD_IDS = /* @__PURE__ */ new Set([
|
|
451
|
+
"js_pc_qr_code",
|
|
452
|
+
"js_profile_qrcode",
|
|
453
|
+
"js_qr_code",
|
|
454
|
+
"js_reward_area",
|
|
455
|
+
"js_praise_area",
|
|
456
|
+
"js_star_area",
|
|
457
|
+
"js_follow_area",
|
|
458
|
+
"js_mp_ad_area",
|
|
459
|
+
"js_bottom_ad_area",
|
|
460
|
+
"js_ad_area",
|
|
461
|
+
"js_author_reward",
|
|
462
|
+
"js_reward_author",
|
|
463
|
+
"js_related_articles",
|
|
464
|
+
"js_article_bottom_ad",
|
|
465
|
+
"js_immersive_ad",
|
|
466
|
+
"js_ad_link",
|
|
467
|
+
"js_tags_area",
|
|
468
|
+
"js_like_area"
|
|
469
|
+
]);
|
|
470
|
+
/** A class or id token that names an advertising element. */
|
|
471
|
+
const AD_ATTRIBUTE = /(?:^|[-_])ads?(?:[-_]|$)/i;
|
|
472
|
+
/**
|
|
473
|
+
* A class or id token that names a QR-code, reward, or subscription widget.
|
|
474
|
+
* Anchored to the end of the token: an unanchored form would drop ordinary
|
|
475
|
+
* content markup, because `like`, `zan`, and `follow` start English words too.
|
|
476
|
+
*/
|
|
477
|
+
const PROMO_ATTRIBUTE = /(?:^|[-_])(?:qrcode|qr_code|reward|praise|subscribe)(?:[-_](?:area|box|wrap|panel|btn|bar))?$/i;
|
|
478
|
+
/** Element names whose whole content is a widget rather than article prose. */
|
|
479
|
+
const WIDGET_ELEMENTS = /* @__PURE__ */ new Set([
|
|
480
|
+
"mp-common-qrcode",
|
|
481
|
+
"mp-common-reward",
|
|
482
|
+
"mp-common-follow",
|
|
483
|
+
"mp-common-vote"
|
|
484
|
+
]);
|
|
485
|
+
/** Text that makes a short block a follow or promotion prompt. */
|
|
486
|
+
const PROMO_TEXT = [
|
|
487
|
+
/关注(?:我们|公众号|一下|本号|后)/,
|
|
488
|
+
/扫码关注/,
|
|
489
|
+
/长按(?:识别|二维码|关注)/,
|
|
490
|
+
/(?:点击|戳)上方(?:蓝字|关注)/,
|
|
491
|
+
/星标/,
|
|
492
|
+
/(?:点|戳)(?:个|一下)?(?:在看|赞)/,
|
|
493
|
+
/点亮(?:在看|赞)/,
|
|
494
|
+
/(?:分享|转发|收藏)到?(?:朋友圈|好友|群)/,
|
|
495
|
+
/往期(?:回顾|推荐|精选|文章)/,
|
|
496
|
+
/推荐阅读/,
|
|
497
|
+
/精彩(?:回顾|推荐)/,
|
|
498
|
+
/更多(?:精彩|内容)/,
|
|
499
|
+
/商务合作/,
|
|
500
|
+
/投稿/,
|
|
501
|
+
/^广告$/,
|
|
502
|
+
/广告位/,
|
|
503
|
+
/本条广告/,
|
|
504
|
+
/免责声明/,
|
|
505
|
+
/版权(?:归|所有)/,
|
|
506
|
+
/微信号[::]/,
|
|
507
|
+
/阅读原文/,
|
|
508
|
+
/(?:长按|扫描|识别)(?:下方|上面)?二维码/,
|
|
509
|
+
/点击(?:下方|底部)?(?:阅读原文|了解更多)/,
|
|
510
|
+
/本文(?:为|系)(?:广告|推广|软文)/,
|
|
511
|
+
/(?:合作|推广|投稿)邮箱/,
|
|
512
|
+
/转载(?:请)?(?:注明|联系)/,
|
|
513
|
+
/(?:喜欢|觉得)不错[,,]?(?:请)?(?:点|分享)/
|
|
514
|
+
];
|
|
515
|
+
/** The policy a caller that states none gets: the built-in rules alone. */
|
|
516
|
+
const DEFAULT_CLEAN_POLICY = {
|
|
517
|
+
maxPromoChars: 200,
|
|
518
|
+
patterns: []
|
|
519
|
+
};
|
|
520
|
+
/**
|
|
521
|
+
* Remove the promotional elements WeChat embeds in an article body.
|
|
522
|
+
* @param bodyHtml - the inner HTML of the article body element.
|
|
523
|
+
* @param policy - the promotion rules the deployment filters with.
|
|
524
|
+
* @returns the same HTML without the advertising subtrees and promotion blocks.
|
|
525
|
+
*/
|
|
526
|
+
function filterAds(bodyHtml, policy = DEFAULT_CLEAN_POLICY) {
|
|
527
|
+
let output = "";
|
|
528
|
+
const stack = [];
|
|
529
|
+
let skipDepth = -1;
|
|
530
|
+
TOKEN$1.lastIndex = 0;
|
|
531
|
+
let match;
|
|
532
|
+
while ((match = TOKEN$1.exec(bodyHtml)) !== null) {
|
|
533
|
+
const whole = match[0];
|
|
534
|
+
const openName = match[1];
|
|
535
|
+
const closeName = match[4];
|
|
536
|
+
if (closeName !== void 0) {
|
|
537
|
+
const entry = closeElement(stack, closeName.toLowerCase());
|
|
538
|
+
if (skipDepth >= 0) {
|
|
539
|
+
if (stack.length < skipDepth) skipDepth = -1;
|
|
540
|
+
continue;
|
|
541
|
+
}
|
|
542
|
+
if (entry !== void 0 && entry.droppable && isPromo(output.slice(entry.mark), policy)) {
|
|
543
|
+
output = output.slice(0, entry.mark);
|
|
544
|
+
continue;
|
|
545
|
+
}
|
|
546
|
+
output += whole;
|
|
547
|
+
continue;
|
|
548
|
+
}
|
|
549
|
+
if (openName !== void 0) {
|
|
550
|
+
const name = openName.toLowerCase();
|
|
551
|
+
const childless = match[3] === "/" || VOID_ELEMENTS.has(name);
|
|
552
|
+
if (skipDepth >= 0) {
|
|
553
|
+
if (!childless) stack.push({
|
|
554
|
+
name,
|
|
555
|
+
mark: 0,
|
|
556
|
+
droppable: false
|
|
557
|
+
});
|
|
558
|
+
continue;
|
|
559
|
+
}
|
|
560
|
+
if (isAdElement(name, match[2] ?? "")) {
|
|
561
|
+
if (!childless) {
|
|
562
|
+
stack.push({
|
|
563
|
+
name,
|
|
564
|
+
mark: 0,
|
|
565
|
+
droppable: false
|
|
566
|
+
});
|
|
567
|
+
skipDepth = stack.length;
|
|
568
|
+
}
|
|
569
|
+
continue;
|
|
570
|
+
}
|
|
571
|
+
if (childless) {
|
|
572
|
+
output += whole;
|
|
573
|
+
continue;
|
|
574
|
+
}
|
|
575
|
+
stack.push({
|
|
576
|
+
name,
|
|
577
|
+
mark: output.length,
|
|
578
|
+
droppable: BLOCK_ELEMENTS$1.has(name)
|
|
579
|
+
});
|
|
580
|
+
output += whole;
|
|
581
|
+
continue;
|
|
582
|
+
}
|
|
583
|
+
if (skipDepth >= 0 || whole.startsWith("<!--")) continue;
|
|
584
|
+
output += whole;
|
|
585
|
+
}
|
|
586
|
+
return output;
|
|
587
|
+
}
|
|
588
|
+
/**
|
|
589
|
+
* Whether one opening tag names an advertising element.
|
|
590
|
+
* @param name - the lower-case element name.
|
|
591
|
+
* @param rawAttributes - the tag's raw attribute text.
|
|
592
|
+
* @returns true when the tag is one of WeChat's advertising elements.
|
|
593
|
+
*/
|
|
594
|
+
function isAdElement(name, rawAttributes) {
|
|
595
|
+
if (AD_TAGS.has(name) || WIDGET_ELEMENTS.has(name)) return true;
|
|
596
|
+
const attributes = parseAttributes(rawAttributes);
|
|
597
|
+
const id = (attributes.get("id") ?? "").toLowerCase();
|
|
598
|
+
if (AD_IDS.has(id) || id !== "" && (AD_ATTRIBUTE.test(id) || PROMO_ATTRIBUTE.test(id))) return true;
|
|
599
|
+
return (attributes.get("class") ?? "").split(/\s+/).some((token) => token !== "" && (AD_ATTRIBUTE.test(token) || PROMO_ATTRIBUTE.test(token)));
|
|
600
|
+
}
|
|
601
|
+
/**
|
|
602
|
+
* Whether one block's whole text is a follow or promotion prompt.
|
|
603
|
+
* @param blockHtml - the block element's emitted HTML.
|
|
604
|
+
* @param policy - the promotion rules the deployment filters with.
|
|
605
|
+
* @returns true when the block's text is short and matches a promotion rule.
|
|
606
|
+
*/
|
|
607
|
+
function isPromo(blockHtml, policy) {
|
|
608
|
+
const text = textOf(blockHtml);
|
|
609
|
+
if (text === "" || text.length > policy.maxPromoChars) return false;
|
|
610
|
+
if (policy.patterns.some((pattern) => pattern.test(text))) return true;
|
|
611
|
+
return PROMO_TEXT.some((pattern) => pattern.test(text));
|
|
612
|
+
}
|
|
613
|
+
/**
|
|
614
|
+
* Pop the open-element stack back to and including one element name.
|
|
615
|
+
* @param stack - the open elements, innermost last.
|
|
616
|
+
* @param name - the lower-case element name the closing tag carried.
|
|
617
|
+
* @returns the popped entry when the tag closed the innermost element, otherwise undefined.
|
|
618
|
+
*/
|
|
619
|
+
function closeElement(stack, name) {
|
|
620
|
+
if (stack[stack.length - 1]?.name === name) return stack.pop();
|
|
621
|
+
while (stack.length > 0) if (stack.pop()?.name === name) break;
|
|
622
|
+
}
|
|
623
|
+
//#endregion
|
|
624
|
+
//#region lib/document.js
|
|
625
|
+
/**
|
|
626
|
+
* The Markdown record one article becomes: the file that is written and the text
|
|
627
|
+
* the model reads.
|
|
628
|
+
*
|
|
629
|
+
* They are **not** the same text. The file holds the whole body; the model reads a
|
|
630
|
+
* bounded head of it plus the path, because a call that carried whole articles
|
|
631
|
+
* would pay their full length in context and the session log would carry it too.
|
|
632
|
+
* Both render the same facts, so a saved article and the call that produced it
|
|
633
|
+
* never disagree about what was retrieved.
|
|
634
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/document
|
|
635
|
+
*/
|
|
636
|
+
/** Quote one value as a YAML double-quoted scalar, which is a JSON string. */
|
|
637
|
+
function yamlString(value) {
|
|
638
|
+
return JSON.stringify(value);
|
|
639
|
+
}
|
|
640
|
+
/**
|
|
641
|
+
* The facts shared by the saved file and the model-facing text.
|
|
642
|
+
* @param article - the article's canonical value.
|
|
643
|
+
* @param body - the text to place in the body position.
|
|
644
|
+
* @returns the heading, the metadata line, and the body.
|
|
645
|
+
*/
|
|
646
|
+
function articleLines(article, body) {
|
|
647
|
+
const meta = [];
|
|
648
|
+
if (article.account !== "") meta.push("公众号:" + article.account);
|
|
649
|
+
if (article.author !== "") meta.push("作者:" + article.author);
|
|
650
|
+
if (article.publishTime !== "") meta.push("发布:" + article.publishTime);
|
|
651
|
+
meta.push("链接:" + article.resolvedUrl);
|
|
652
|
+
const lines = [
|
|
653
|
+
"# " + (article.title === "" ? "(无标题)" : article.title),
|
|
654
|
+
"",
|
|
655
|
+
meta.join(" | "),
|
|
656
|
+
"",
|
|
657
|
+
body
|
|
658
|
+
];
|
|
659
|
+
if (article.images.length > 0) lines.push("", "## 图片", ...article.images.map((src, index) => String(index + 1) + ". " + src));
|
|
660
|
+
return lines;
|
|
661
|
+
}
|
|
662
|
+
/**
|
|
663
|
+
* Render the Markdown file one article is saved as. This is the complete record:
|
|
664
|
+
* the whole body, never a window of it.
|
|
665
|
+
* @param article - the article as it was built, body included.
|
|
666
|
+
* @returns YAML front matter, the full body, and the image list.
|
|
667
|
+
*/
|
|
668
|
+
function articleDocument(article) {
|
|
669
|
+
return [
|
|
670
|
+
"---",
|
|
671
|
+
"title: " + yamlString(article.title),
|
|
672
|
+
"account: " + yamlString(article.account),
|
|
673
|
+
"author: " + yamlString(article.author),
|
|
674
|
+
"publishTime: " + yamlString(article.publishTime),
|
|
675
|
+
"source: " + yamlString(article.resolvedUrl),
|
|
676
|
+
"cover: " + yamlString(article.cover),
|
|
677
|
+
"chars: " + String(article.chars),
|
|
678
|
+
"rendered: " + String(article.rendered),
|
|
679
|
+
"---",
|
|
680
|
+
...articleLines(article, article.body)
|
|
681
|
+
].join("\n");
|
|
682
|
+
}
|
|
683
|
+
/**
|
|
684
|
+
* Render the model-facing text of one saved article: the facts, a head of the
|
|
685
|
+
* body, how long the body is, and where the whole of it lives.
|
|
686
|
+
* @param article - the article's canonical value, saved path included.
|
|
687
|
+
* @returns the article facts, the excerpt, the body's length, and the saved path.
|
|
688
|
+
*/
|
|
689
|
+
function renderArticle(article) {
|
|
690
|
+
const lines = articleLines(article, article.excerpt);
|
|
691
|
+
lines.push("");
|
|
692
|
+
lines.push(article.excerptClipped ? `(以上为节选,正文共 ${String(article.chars)} 字;完整内容见文件)` : `(正文共 ${String(article.chars)} 字)`);
|
|
693
|
+
lines.push("文件:" + article.file);
|
|
694
|
+
return lines.join("\n");
|
|
695
|
+
}
|
|
696
|
+
//#endregion
|
|
697
|
+
//#region lib/fetch.js
|
|
698
|
+
/**
|
|
699
|
+
* The plugin's HTTP layer: browser-like request headers, a session cookie jar,
|
|
700
|
+
* a minimum interval between requests, a cool-down a refused request imposes on
|
|
701
|
+
* the rest of the instance, bounded retries that honour the server's own retry
|
|
702
|
+
* hint, and the deployment's timeout. Kept separate from page parsing, and
|
|
703
|
+
* driven through injected transport and clock ports so tests never sleep or
|
|
704
|
+
* reach the network.
|
|
705
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/fetch
|
|
706
|
+
*/
|
|
707
|
+
/** Statuses worth another attempt: rate limits and server-side failures. */
|
|
708
|
+
const RETRYABLE_STATUS = /* @__PURE__ */ new Set([
|
|
709
|
+
408,
|
|
710
|
+
425,
|
|
711
|
+
429,
|
|
712
|
+
500,
|
|
713
|
+
502,
|
|
714
|
+
503,
|
|
715
|
+
504
|
|
716
|
+
]);
|
|
717
|
+
/**
|
|
718
|
+
* Headers a browser sends when navigating to a page. The client hints are
|
|
719
|
+
* deliberately absent: they must agree with the configured user agent, and a
|
|
720
|
+
* mismatch is a stronger signal than their absence. `sec-fetch-site` is absent
|
|
721
|
+
* too, because it describes the navigation the request belongs to and is set
|
|
722
|
+
* per request.
|
|
723
|
+
*/
|
|
724
|
+
const NAVIGATION_HEADERS = {
|
|
725
|
+
accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
726
|
+
"accept-language": "zh-CN,zh;q=0.9,en;q=0.8",
|
|
727
|
+
"cache-control": "max-age=0",
|
|
728
|
+
"sec-fetch-dest": "document",
|
|
729
|
+
"sec-fetch-mode": "navigate",
|
|
730
|
+
"sec-fetch-user": "?1",
|
|
731
|
+
"upgrade-insecure-requests": "1"
|
|
732
|
+
};
|
|
733
|
+
/** `sec-fetch-site` for the instance's first navigation, which no page referred. */
|
|
734
|
+
const FIRST_NAVIGATION_SITE = "none";
|
|
735
|
+
/** `sec-fetch-site` for every navigation after the first, which a fetched page referred. */
|
|
736
|
+
const REFERRED_NAVIGATION_SITE = "same-origin";
|
|
737
|
+
const DEFAULT_DEPS = {
|
|
738
|
+
fetchImpl: fetch,
|
|
739
|
+
sleep: (ms) => new Promise((resolve) => {
|
|
740
|
+
setTimeout(resolve, ms);
|
|
741
|
+
}),
|
|
742
|
+
now: () => Date.now(),
|
|
743
|
+
random: () => Math.random()
|
|
744
|
+
};
|
|
745
|
+
/** Whether the caller already cancelled; a function call keeps TS from narrowing the flag. */
|
|
746
|
+
function cancelled(signal) {
|
|
747
|
+
return signal !== void 0 && signal.aborted;
|
|
748
|
+
}
|
|
749
|
+
/** A cancellation error the caller can distinguish from a failed request. */
|
|
750
|
+
function abortError() {
|
|
751
|
+
const error = /* @__PURE__ */ new Error("wechat: the call was cancelled");
|
|
752
|
+
error.name = "AbortError";
|
|
753
|
+
return error;
|
|
754
|
+
}
|
|
755
|
+
/** Milliseconds to wait before one retry, jittered so parallel callers do not sync. */
|
|
756
|
+
function backoffDelay(base, attempt, random) {
|
|
757
|
+
return Math.round(base * 2 ** attempt * (.5 + random()));
|
|
758
|
+
}
|
|
759
|
+
/**
|
|
760
|
+
* Milliseconds to wait before the next attempt. A usable server hint wins,
|
|
761
|
+
* capped by the deployment, because it is the rate the server is willing to
|
|
762
|
+
* serve; otherwise the jittered backoff applies.
|
|
763
|
+
*/
|
|
764
|
+
function retryDelay(policy, attempt, retryAfterMs, random) {
|
|
765
|
+
if (retryAfterMs > 0) return Math.min(retryAfterMs, policy.maxRetryAfterMs);
|
|
766
|
+
return backoffDelay(policy.backoffBaseMs, attempt, random);
|
|
767
|
+
}
|
|
768
|
+
/**
|
|
769
|
+
* Read a Retry-After header, in either of its two legal forms.
|
|
770
|
+
* @param value - the header value, or null when the response carried none.
|
|
771
|
+
* @param now - current time in milliseconds, for the HTTP-date form.
|
|
772
|
+
* @returns milliseconds to wait, or 0 when the header is absent or unusable.
|
|
773
|
+
*/
|
|
774
|
+
function parseRetryAfter(value, now) {
|
|
775
|
+
if (value === null || value.trim() === "") return 0;
|
|
776
|
+
const seconds = Number(value);
|
|
777
|
+
if (Number.isFinite(seconds) && seconds >= 0) return Math.round(seconds * 1e3);
|
|
778
|
+
const date = Date.parse(value);
|
|
779
|
+
return Number.isNaN(date) ? 0 : Math.max(0, date - now);
|
|
780
|
+
}
|
|
781
|
+
/**
|
|
782
|
+
* Every `Set-Cookie` value one response carried. `Headers.getSetCookie` is the
|
|
783
|
+
* only accessor that returns more than one; a runtime without it yields the
|
|
784
|
+
* single folded header, which is still better than dropping the cookies.
|
|
785
|
+
* @param headers - the response headers.
|
|
786
|
+
* @returns the header values, in the order the server sent them.
|
|
787
|
+
*/
|
|
788
|
+
function setCookieValues(headers) {
|
|
789
|
+
const multi = headers.getSetCookie;
|
|
790
|
+
if (typeof multi === "function") return multi.call(headers);
|
|
791
|
+
const single = headers.get("set-cookie");
|
|
792
|
+
return single === null ? [] : [single];
|
|
793
|
+
}
|
|
794
|
+
/**
|
|
795
|
+
* Read the name and value out of one `Set-Cookie` value, ignoring its
|
|
796
|
+
* attributes. A deletion (`Max-Age=0`) is read like any other; the server's own
|
|
797
|
+
* later value replaces it, so a stale entry cannot outlive the response that
|
|
798
|
+
* removed it on any page the walk reaches afterwards.
|
|
799
|
+
* @param value - one `Set-Cookie` header value.
|
|
800
|
+
* @returns the cookie pair, or undefined when the value carries no `name=value`.
|
|
801
|
+
*/
|
|
802
|
+
function cookiePair(value) {
|
|
803
|
+
const pair = value.split(";", 1)[0]?.trim() ?? "";
|
|
804
|
+
const equals = pair.indexOf("=");
|
|
805
|
+
if (equals <= 0) return void 0;
|
|
806
|
+
return [pair.slice(0, equals).trim(), pair.slice(equals + 1).trim()];
|
|
807
|
+
}
|
|
808
|
+
/**
|
|
809
|
+
* The cookie header one request sends: the deployment's own cookies first, in
|
|
810
|
+
* the order it wrote them, then every cookie this instance has been given that
|
|
811
|
+
* the deployment's header does not already name.
|
|
812
|
+
* @param configured - the cookies parsed from the deployment's header.
|
|
813
|
+
* @param jar - cookies harvested from earlier responses, keyed by name.
|
|
814
|
+
* @returns the header value, or an empty string when no cookie is held.
|
|
815
|
+
*/
|
|
816
|
+
function cookieHeader(configured, jar) {
|
|
817
|
+
const named = new Set(configured.map((pair) => pair[0]));
|
|
818
|
+
const parts = configured.map((pair) => pair[0] + "=" + pair[1]);
|
|
819
|
+
for (const [name, value] of jar) if (!named.has(name)) parts.push(name + "=" + value);
|
|
820
|
+
return parts.join("; ");
|
|
821
|
+
}
|
|
822
|
+
/**
|
|
823
|
+
* Parse a Cookie header into ordered pairs.
|
|
824
|
+
* @param header - the deployment's Cookie header.
|
|
825
|
+
* @returns one pair per non-empty entry it carries.
|
|
826
|
+
*/
|
|
827
|
+
function configuredCookies(header) {
|
|
828
|
+
if (header.trim() === "") return [];
|
|
829
|
+
const pairs = [];
|
|
830
|
+
for (const part of header.split(";")) {
|
|
831
|
+
const pair = cookiePair(part);
|
|
832
|
+
if (pair !== void 0) pairs.push(pair);
|
|
833
|
+
}
|
|
834
|
+
return pairs;
|
|
835
|
+
}
|
|
836
|
+
/** Sleep that settles early when the caller cancels. */
|
|
837
|
+
function sleepWithSignal(ms, sleep, signal) {
|
|
838
|
+
if (signal === void 0) return sleep(ms);
|
|
839
|
+
if (signal.aborted) return Promise.reject(abortError());
|
|
840
|
+
return new Promise((resolve, reject) => {
|
|
841
|
+
const onAbort = () => {
|
|
842
|
+
reject(abortError());
|
|
843
|
+
};
|
|
844
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
845
|
+
sleep(ms).then(() => {
|
|
846
|
+
signal.removeEventListener("abort", onAbort);
|
|
847
|
+
resolve();
|
|
848
|
+
}, (error) => {
|
|
849
|
+
signal.removeEventListener("abort", onAbort);
|
|
850
|
+
reject(error);
|
|
851
|
+
});
|
|
852
|
+
});
|
|
853
|
+
}
|
|
854
|
+
/**
|
|
855
|
+
* Build the plugin instance's fetcher.
|
|
856
|
+
* @param policy - the deployment's header, pacing, retry, and timeout values.
|
|
857
|
+
* @param overrides - test ports replacing the default transport, clock, or jitter.
|
|
858
|
+
* @returns the fetcher whose pace() the browser renderer shares.
|
|
859
|
+
*/
|
|
860
|
+
function createFetcher(policy, overrides = {}) {
|
|
861
|
+
const deps = {
|
|
862
|
+
...DEFAULT_DEPS,
|
|
863
|
+
...overrides
|
|
864
|
+
};
|
|
865
|
+
const configured = configuredCookies(policy.cookie);
|
|
866
|
+
const jar = /* @__PURE__ */ new Map();
|
|
867
|
+
let lastRequestAt = Number.NEGATIVE_INFINITY;
|
|
868
|
+
let resumeAt = Number.NEGATIVE_INFINITY;
|
|
869
|
+
let navigations = 0;
|
|
870
|
+
const pace = async (signal) => {
|
|
871
|
+
const jitter = policy.minIntervalJitterMs > 0 ? Math.round(deps.random() * policy.minIntervalJitterMs) : 0;
|
|
872
|
+
const wait = Math.max(lastRequestAt + policy.minIntervalMs + jitter, resumeAt) - deps.now();
|
|
873
|
+
if (wait > 0) await sleepWithSignal(wait, deps.sleep, signal);
|
|
874
|
+
lastRequestAt = deps.now();
|
|
875
|
+
};
|
|
876
|
+
const coolDown = (ms) => {
|
|
877
|
+
if (ms > 0) resumeAt = Math.max(resumeAt, deps.now() + ms);
|
|
878
|
+
};
|
|
879
|
+
const once = async (url, signal) => {
|
|
880
|
+
const timeout = AbortSignal.timeout(policy.timeoutMs);
|
|
881
|
+
const combined = signal === void 0 ? timeout : AbortSignal.any([signal, timeout]);
|
|
882
|
+
const headers = {
|
|
883
|
+
...NAVIGATION_HEADERS,
|
|
884
|
+
"user-agent": policy.userAgent,
|
|
885
|
+
"sec-fetch-site": navigations === 0 ? FIRST_NAVIGATION_SITE : REFERRED_NAVIGATION_SITE
|
|
886
|
+
};
|
|
887
|
+
navigations += 1;
|
|
888
|
+
if (policy.referer !== "") headers.referer = policy.referer;
|
|
889
|
+
const cookie = policy.cookieJar ? cookieHeader(configured, jar) : policy.cookie;
|
|
890
|
+
if (cookie !== "") headers.cookie = cookie;
|
|
891
|
+
const response = await deps.fetchImpl(url, {
|
|
892
|
+
headers,
|
|
893
|
+
redirect: "follow",
|
|
894
|
+
signal: combined
|
|
895
|
+
});
|
|
896
|
+
const setCookie = setCookieValues(response.headers);
|
|
897
|
+
if (policy.cookieJar) for (const value of setCookie) {
|
|
898
|
+
const pair = cookiePair(value);
|
|
899
|
+
if (pair !== void 0) jar.set(pair[0], pair[1]);
|
|
900
|
+
}
|
|
901
|
+
const body = await response.text();
|
|
902
|
+
return {
|
|
903
|
+
response: {
|
|
904
|
+
url: response.url === "" ? url : response.url,
|
|
905
|
+
status: response.status,
|
|
906
|
+
body,
|
|
907
|
+
setCookie
|
|
908
|
+
},
|
|
909
|
+
retryAfterMs: parseRetryAfter(response.headers.get("retry-after"), deps.now())
|
|
910
|
+
};
|
|
911
|
+
};
|
|
912
|
+
const get = async (url, signal) => {
|
|
913
|
+
if (cancelled(signal)) throw abortError();
|
|
914
|
+
let lastError;
|
|
915
|
+
let delayMs = 0;
|
|
916
|
+
for (let attempt = 0; attempt <= policy.maxRetries; attempt += 1) {
|
|
917
|
+
if (attempt > 0) await sleepWithSignal(delayMs, deps.sleep, signal);
|
|
918
|
+
await pace(signal);
|
|
919
|
+
try {
|
|
920
|
+
const { response, retryAfterMs } = await once(url, signal);
|
|
921
|
+
if (RETRYABLE_STATUS.has(response.status) && attempt < policy.maxRetries) {
|
|
922
|
+
lastError = /* @__PURE__ */ new Error("wechat: HTTP " + response.status + " from " + url);
|
|
923
|
+
delayMs = retryDelay(policy, attempt, retryAfterMs, deps.random);
|
|
924
|
+
continue;
|
|
925
|
+
}
|
|
926
|
+
return response;
|
|
927
|
+
} catch (error) {
|
|
928
|
+
if (cancelled(signal)) throw error;
|
|
929
|
+
lastError = error;
|
|
930
|
+
delayMs = backoffDelay(policy.backoffBaseMs, attempt, deps.random);
|
|
931
|
+
}
|
|
932
|
+
}
|
|
933
|
+
throw lastError instanceof Error ? lastError : /* @__PURE__ */ new Error("wechat: the request to " + url + " failed");
|
|
934
|
+
};
|
|
935
|
+
return {
|
|
936
|
+
get,
|
|
937
|
+
pace,
|
|
938
|
+
coolDown
|
|
939
|
+
};
|
|
940
|
+
}
|
|
941
|
+
//#endregion
|
|
942
|
+
//#region lib/list.js
|
|
943
|
+
/**
|
|
944
|
+
* List and collection page parsing. A WeChat collection page
|
|
945
|
+
* (mp/appmsgalbum?action=getalbum) carries its article list in static HTML: one
|
|
946
|
+
* element per article with a data-link, plus the msgid and itemidx cursors that
|
|
947
|
+
* address the next page, and a continue_flag that says whether more follow. A
|
|
948
|
+
* generic list page links its articles with ordinary anchors instead, and
|
|
949
|
+
* addresses its continuation with an ordinary next-page anchor when it has one.
|
|
950
|
+
* This module turns either into an ordered, de-duplicated list of article links
|
|
951
|
+
* plus the cursor that continues the walk.
|
|
952
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/list
|
|
953
|
+
*/
|
|
954
|
+
/** One opening tag, with its attribute text. */
|
|
955
|
+
const OPEN_TAG = /<([a-zA-Z][a-zA-Z0-9-]*)((?:"[^"]*"|'[^']*'|[^>"'])*?)(\/?)>/g;
|
|
956
|
+
/** One anchor tag, whatever attributes it carries. */
|
|
957
|
+
const ANCHOR = /<a\b[^>]*>/gi;
|
|
958
|
+
/** The href of one anchor tag, double- or single-quoted. */
|
|
959
|
+
const HREF = /href\s*=\s*(?:"([^"]*)"|'([^']*)')/i;
|
|
960
|
+
/** The inline flag a collection page sets while further pages exist. */
|
|
961
|
+
const CONTINUE_FLAG = /continue_flag\s*:\s*['"]?1['"]?/;
|
|
962
|
+
/**
|
|
963
|
+
* The whole text of an anchor that addresses the next page of a generic list.
|
|
964
|
+
* Anchored at both ends so a link merely mentioning "更多" elsewhere in a
|
|
965
|
+
* sentence is not mistaken for the continuation.
|
|
966
|
+
*/
|
|
967
|
+
const NEXT_TEXT = /^(?:下一页|下页|下一頁|下一篇|更多|查看更多|next(?:\s*page)?|older|›|»|>>)$/i;
|
|
968
|
+
/**
|
|
969
|
+
* Entries one collection page carries. The server ignores a larger count, so
|
|
970
|
+
* this is the page size the walk must assume rather than a deployment choice.
|
|
971
|
+
*/
|
|
972
|
+
const ALBUM_PAGE_SIZE = "20";
|
|
973
|
+
/**
|
|
974
|
+
* Parse one list or collection page.
|
|
975
|
+
* @param html - the page's HTML.
|
|
976
|
+
* @param requestedUrl - the link the page was fetched from, used to resolve relative hrefs and to drop a self-link.
|
|
977
|
+
* @returns the page's title, entries, and its continuation cursor when it carries one.
|
|
978
|
+
*/
|
|
979
|
+
function parseListPage(html, requestedUrl) {
|
|
980
|
+
const items = [];
|
|
981
|
+
const seen = /* @__PURE__ */ new Set();
|
|
982
|
+
const add = (raw, cursors) => {
|
|
983
|
+
const url = normalizeLink(raw, requestedUrl);
|
|
984
|
+
if (url === "" || url === requestedUrl || seen.has(url)) return;
|
|
985
|
+
seen.add(url);
|
|
986
|
+
items.push({
|
|
987
|
+
url,
|
|
988
|
+
msgid: cursors?.get("data-msgid") ?? "",
|
|
989
|
+
itemidx: cursors?.get("data-itemidx") ?? ""
|
|
990
|
+
});
|
|
991
|
+
};
|
|
992
|
+
OPEN_TAG.lastIndex = 0;
|
|
993
|
+
let match;
|
|
994
|
+
while ((match = OPEN_TAG.exec(html)) !== null) {
|
|
995
|
+
if (!match[0].includes("data-link")) continue;
|
|
996
|
+
const attributes = parseAttributes(match[0]);
|
|
997
|
+
const link = attributes.get("data-link");
|
|
998
|
+
if (link !== void 0) add(link, attributes);
|
|
999
|
+
}
|
|
1000
|
+
while ((match = ANCHOR.exec(html)) !== null) {
|
|
1001
|
+
const href = HREF.exec(match[0]);
|
|
1002
|
+
const raw = href?.[1] ?? href?.[2];
|
|
1003
|
+
if (raw !== void 0) add(raw, parseAttributes(match[0]));
|
|
1004
|
+
}
|
|
1005
|
+
const more = CONTINUE_FLAG.test(html);
|
|
1006
|
+
const cursor = [...items].reverse().find((item) => item.msgid !== "" && item.itemidx !== "");
|
|
1007
|
+
const next = more && cursor !== void 0 ? {
|
|
1008
|
+
msgid: cursor.msgid,
|
|
1009
|
+
itemidx: cursor.itemidx
|
|
1010
|
+
} : void 0;
|
|
1011
|
+
return {
|
|
1012
|
+
title: listTitle(html),
|
|
1013
|
+
items,
|
|
1014
|
+
more,
|
|
1015
|
+
next,
|
|
1016
|
+
nextUrl: nextPageUrl(html, requestedUrl)
|
|
1017
|
+
};
|
|
1018
|
+
}
|
|
1019
|
+
/**
|
|
1020
|
+
* The link a generic list page gives for its own next page.
|
|
1021
|
+
* @param html - the page's HTML.
|
|
1022
|
+
* @param base - the link the page was fetched from.
|
|
1023
|
+
* @returns the absolute same-host next-page link, or an empty string when the page carries none.
|
|
1024
|
+
*/
|
|
1025
|
+
function nextPageUrl(html, base) {
|
|
1026
|
+
const anchors = /<a\b[^>]*>[\s\S]*?<\/a\s*>/gi;
|
|
1027
|
+
let match;
|
|
1028
|
+
while ((match = anchors.exec(html)) !== null) {
|
|
1029
|
+
if (!NEXT_TEXT.test(textOf(match[0]))) continue;
|
|
1030
|
+
const href = HREF.exec(match[0]);
|
|
1031
|
+
const raw = href?.[1] ?? href?.[2];
|
|
1032
|
+
if (raw === void 0) continue;
|
|
1033
|
+
const resolved = absoluteSameHostLink(raw, base);
|
|
1034
|
+
if (resolved !== "" && resolved !== base) return resolved;
|
|
1035
|
+
}
|
|
1036
|
+
return "";
|
|
1037
|
+
}
|
|
1038
|
+
/**
|
|
1039
|
+
* Address the collection page after one that ended on a cursor.
|
|
1040
|
+
* @param pageUrl - the collection page the walk is on.
|
|
1041
|
+
* @param cursor - the msgid and itemidx the page's last entry carried.
|
|
1042
|
+
* @returns the next page's URL, or an empty string when the URL names no collection.
|
|
1043
|
+
*/
|
|
1044
|
+
function albumPageUrl(pageUrl, cursor) {
|
|
1045
|
+
let parsed;
|
|
1046
|
+
try {
|
|
1047
|
+
parsed = new URL(pageUrl);
|
|
1048
|
+
} catch {
|
|
1049
|
+
return "";
|
|
1050
|
+
}
|
|
1051
|
+
if (parsed.searchParams.get("__biz") === null || parsed.searchParams.get("album_id") === null) return "";
|
|
1052
|
+
parsed.hash = "";
|
|
1053
|
+
parsed.searchParams.set("action", "getalbum");
|
|
1054
|
+
parsed.searchParams.set("count", ALBUM_PAGE_SIZE);
|
|
1055
|
+
parsed.searchParams.set("begin_msgid", cursor.msgid);
|
|
1056
|
+
parsed.searchParams.set("begin_itemidx", cursor.itemidx);
|
|
1057
|
+
return parsed.href;
|
|
1058
|
+
}
|
|
1059
|
+
/**
|
|
1060
|
+
* Normalize one page link to an absolute https article URL.
|
|
1061
|
+
* @param raw - the link as the page carried it, possibly relative and entity-encoded.
|
|
1062
|
+
* @param base - the link the page was fetched from, resolving a relative href.
|
|
1063
|
+
* @returns the normalized article link, or an empty string when it names no article page.
|
|
1064
|
+
*/
|
|
1065
|
+
function normalizeLink(raw, base) {
|
|
1066
|
+
const parsed = absoluteLink(raw, base);
|
|
1067
|
+
if (parsed === void 0) return "";
|
|
1068
|
+
if (parsed.protocol === "http:" && parsed.hostname === "mp.weixin.qq.com") parsed.protocol = "https:";
|
|
1069
|
+
return isMpArticleLink(parsed.href) ? parsed.href : "";
|
|
1070
|
+
}
|
|
1071
|
+
/**
|
|
1072
|
+
* Resolve one page link against its base.
|
|
1073
|
+
* @param raw - the link as the page carried it, possibly relative and entity-encoded.
|
|
1074
|
+
* @param base - the link the page was fetched from.
|
|
1075
|
+
* @returns the absolute http(s) link without its fragment, or undefined when it names no such link.
|
|
1076
|
+
*/
|
|
1077
|
+
function absoluteLink(raw, base) {
|
|
1078
|
+
let parsed;
|
|
1079
|
+
try {
|
|
1080
|
+
parsed = new URL(decodeEntities(raw).trim(), base);
|
|
1081
|
+
} catch {
|
|
1082
|
+
return;
|
|
1083
|
+
}
|
|
1084
|
+
if (parsed.protocol !== "http:" && parsed.protocol !== "https:") return void 0;
|
|
1085
|
+
parsed.hash = "";
|
|
1086
|
+
return parsed;
|
|
1087
|
+
}
|
|
1088
|
+
/**
|
|
1089
|
+
* Resolve one next-page link against its base, requiring the same host: a
|
|
1090
|
+
* continuation that leaves the site is not the list's own next page.
|
|
1091
|
+
* @param raw - the link as the page carried it.
|
|
1092
|
+
* @param base - the link the page was fetched from.
|
|
1093
|
+
* @returns the absolute link, or an empty string when it names no same-host page.
|
|
1094
|
+
*/
|
|
1095
|
+
function absoluteSameHostLink(raw, base) {
|
|
1096
|
+
let origin;
|
|
1097
|
+
try {
|
|
1098
|
+
origin = new URL(base);
|
|
1099
|
+
} catch {
|
|
1100
|
+
return "";
|
|
1101
|
+
}
|
|
1102
|
+
const parsed = absoluteLink(raw, base);
|
|
1103
|
+
return parsed !== void 0 && parsed.hostname === origin.hostname ? parsed.href : "";
|
|
1104
|
+
}
|
|
1105
|
+
/**
|
|
1106
|
+
* The title one list page carries.
|
|
1107
|
+
* @param html - the page's HTML.
|
|
1108
|
+
* @returns the page's own title, or an empty string when it carries none.
|
|
1109
|
+
*/
|
|
1110
|
+
function listTitle(html) {
|
|
1111
|
+
const openGraph = metaContent(html, "og:title");
|
|
1112
|
+
if (openGraph !== "") return openGraph;
|
|
1113
|
+
const title = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(html);
|
|
1114
|
+
return title?.[1] === void 0 ? "" : textOf(title[1]);
|
|
1115
|
+
}
|
|
1116
|
+
//#endregion
|
|
1117
|
+
//#region lib/markdown.js
|
|
1118
|
+
/**
|
|
1119
|
+
* HTML-to-Markdown and HTML-to-text conversion for article bodies. A single tag
|
|
1120
|
+
* walker covers the elements WeChat article bodies use: sections, paragraphs,
|
|
1121
|
+
* headings, emphasis, links, lazy images, lists, quotes, and code. It is a
|
|
1122
|
+
* focused converter for that corpus, not a general HTML parser.
|
|
1123
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/markdown
|
|
1124
|
+
*/
|
|
1125
|
+
/** Elements that start a new block of prose. */
|
|
1126
|
+
const BLOCK_ELEMENTS = /* @__PURE__ */ new Set([
|
|
1127
|
+
"address",
|
|
1128
|
+
"article",
|
|
1129
|
+
"aside",
|
|
1130
|
+
"blockquote",
|
|
1131
|
+
"div",
|
|
1132
|
+
"dl",
|
|
1133
|
+
"dd",
|
|
1134
|
+
"dt",
|
|
1135
|
+
"figcaption",
|
|
1136
|
+
"figure",
|
|
1137
|
+
"footer",
|
|
1138
|
+
"form",
|
|
1139
|
+
"h1",
|
|
1140
|
+
"h2",
|
|
1141
|
+
"h3",
|
|
1142
|
+
"h4",
|
|
1143
|
+
"h5",
|
|
1144
|
+
"h6",
|
|
1145
|
+
"header",
|
|
1146
|
+
"li",
|
|
1147
|
+
"main",
|
|
1148
|
+
"nav",
|
|
1149
|
+
"ol",
|
|
1150
|
+
"p",
|
|
1151
|
+
"section",
|
|
1152
|
+
"table",
|
|
1153
|
+
"tbody",
|
|
1154
|
+
"td",
|
|
1155
|
+
"tfoot",
|
|
1156
|
+
"th",
|
|
1157
|
+
"thead",
|
|
1158
|
+
"tr",
|
|
1159
|
+
"ul"
|
|
1160
|
+
]);
|
|
1161
|
+
/** Elements whose content never reaches the article text. */
|
|
1162
|
+
const SKIP_ELEMENTS = /* @__PURE__ */ new Set([
|
|
1163
|
+
"script",
|
|
1164
|
+
"style",
|
|
1165
|
+
"noscript",
|
|
1166
|
+
"svg",
|
|
1167
|
+
"iframe",
|
|
1168
|
+
"video",
|
|
1169
|
+
"audio"
|
|
1170
|
+
]);
|
|
1171
|
+
/** Heading element names, captured with their level. */
|
|
1172
|
+
const HEADING = /^h([1-6])$/;
|
|
1173
|
+
/** One opening tag, closing tag, comment, or run of text. */
|
|
1174
|
+
const TOKEN = /<([a-zA-Z][a-zA-Z0-9-]*)((?:"[^"]*"|'[^']*'|[^>"'])*?)(\/?)>|<\/([a-zA-Z][a-zA-Z0-9-]*)>|<!--[\s\S]*?-->|[^<]+/g;
|
|
1175
|
+
/**
|
|
1176
|
+
* Convert one article body to Markdown.
|
|
1177
|
+
* @param html - the body element's inner HTML.
|
|
1178
|
+
* @returns Markdown text with headings, lists, links, and images preserved.
|
|
1179
|
+
*/
|
|
1180
|
+
function htmlToMarkdown(html) {
|
|
1181
|
+
return convert(html, true);
|
|
1182
|
+
}
|
|
1183
|
+
/**
|
|
1184
|
+
* Convert one article body to plain text.
|
|
1185
|
+
* @param html - the body element's inner HTML.
|
|
1186
|
+
* @returns plain text with block boundaries preserved and inline markers dropped.
|
|
1187
|
+
*/
|
|
1188
|
+
function htmlToText(html) {
|
|
1189
|
+
return convert(html, false);
|
|
1190
|
+
}
|
|
1191
|
+
/**
|
|
1192
|
+
* Walk the body's tags once, emitting block structure and inline markers.
|
|
1193
|
+
* @param html - body HTML.
|
|
1194
|
+
* @param markdown - whether heading, emphasis, link, and fence markers are emitted.
|
|
1195
|
+
* @returns normalized output text.
|
|
1196
|
+
*/
|
|
1197
|
+
function convert(html, markdown) {
|
|
1198
|
+
let result = "";
|
|
1199
|
+
let skipTag;
|
|
1200
|
+
let skipDepth = 0;
|
|
1201
|
+
let preDepth = 0;
|
|
1202
|
+
let linkHref;
|
|
1203
|
+
let codeOpen = false;
|
|
1204
|
+
let cellsInRow = 0;
|
|
1205
|
+
const lists = [];
|
|
1206
|
+
const ensureNewline = () => {
|
|
1207
|
+
if (result !== "" && !result.endsWith("\n")) result += "\n";
|
|
1208
|
+
};
|
|
1209
|
+
const ensureBlank = () => {
|
|
1210
|
+
if (result === "") return;
|
|
1211
|
+
ensureNewline();
|
|
1212
|
+
if (!result.endsWith("\n\n")) result += "\n";
|
|
1213
|
+
};
|
|
1214
|
+
const marker = (text) => {
|
|
1215
|
+
if (markdown) result += text;
|
|
1216
|
+
};
|
|
1217
|
+
const open = (name, rawAttributes) => {
|
|
1218
|
+
if (preDepth > 0) {
|
|
1219
|
+
if (name === "br") result += "\n";
|
|
1220
|
+
return;
|
|
1221
|
+
}
|
|
1222
|
+
if (name === "br") {
|
|
1223
|
+
result += "\n";
|
|
1224
|
+
return;
|
|
1225
|
+
}
|
|
1226
|
+
if (name === "hr") {
|
|
1227
|
+
ensureBlank();
|
|
1228
|
+
marker("---");
|
|
1229
|
+
ensureBlank();
|
|
1230
|
+
return;
|
|
1231
|
+
}
|
|
1232
|
+
if (name === "img") {
|
|
1233
|
+
const attributes = parseAttributes(rawAttributes);
|
|
1234
|
+
const src = attributes.get("data-src") ?? attributes.get("src") ?? "";
|
|
1235
|
+
if (src === "" || src.startsWith("data:")) return;
|
|
1236
|
+
const alt = attributes.get("alt") ?? "图片";
|
|
1237
|
+
result += markdown ? "" : src;
|
|
1238
|
+
return;
|
|
1239
|
+
}
|
|
1240
|
+
if (name === "pre") {
|
|
1241
|
+
ensureBlank();
|
|
1242
|
+
marker("~~~");
|
|
1243
|
+
ensureNewline();
|
|
1244
|
+
preDepth = 1;
|
|
1245
|
+
return;
|
|
1246
|
+
}
|
|
1247
|
+
if (BLOCK_ELEMENTS.has(name)) {
|
|
1248
|
+
ensureBlank();
|
|
1249
|
+
const heading = HEADING.exec(name);
|
|
1250
|
+
if (heading?.[1] !== void 0) marker("#".repeat(Number(heading[1])) + " ");
|
|
1251
|
+
else if (name === "li") {
|
|
1252
|
+
const list = lists[lists.length - 1];
|
|
1253
|
+
if (list?.ordered === true) {
|
|
1254
|
+
list.index += 1;
|
|
1255
|
+
result += list.index + ". ";
|
|
1256
|
+
} else result += "- ";
|
|
1257
|
+
} else if (name === "blockquote") marker("> ");
|
|
1258
|
+
else if (name === "tr") cellsInRow = 0;
|
|
1259
|
+
else if ((name === "td" || name === "th") && cellsInRow > 0) result += " | ";
|
|
1260
|
+
else if (name === "ul" || name === "ol") lists.push({
|
|
1261
|
+
ordered: name === "ol",
|
|
1262
|
+
index: 0
|
|
1263
|
+
});
|
|
1264
|
+
return;
|
|
1265
|
+
}
|
|
1266
|
+
if (name === "a") {
|
|
1267
|
+
linkHref = parseAttributes(rawAttributes).get("href");
|
|
1268
|
+
marker("[");
|
|
1269
|
+
return;
|
|
1270
|
+
}
|
|
1271
|
+
if (name === "strong" || name === "b") {
|
|
1272
|
+
marker("**");
|
|
1273
|
+
return;
|
|
1274
|
+
}
|
|
1275
|
+
if (name === "em" || name === "i") {
|
|
1276
|
+
marker("*");
|
|
1277
|
+
return;
|
|
1278
|
+
}
|
|
1279
|
+
if (name === "del" || name === "s" || name === "strike") {
|
|
1280
|
+
marker("~~");
|
|
1281
|
+
return;
|
|
1282
|
+
}
|
|
1283
|
+
if (name === "code" && !codeOpen) {
|
|
1284
|
+
marker("`");
|
|
1285
|
+
codeOpen = true;
|
|
1286
|
+
}
|
|
1287
|
+
};
|
|
1288
|
+
const close = (name) => {
|
|
1289
|
+
if (name === "pre") {
|
|
1290
|
+
ensureNewline();
|
|
1291
|
+
marker("~~~");
|
|
1292
|
+
ensureBlank();
|
|
1293
|
+
preDepth = 0;
|
|
1294
|
+
return;
|
|
1295
|
+
}
|
|
1296
|
+
if (preDepth > 0) return;
|
|
1297
|
+
if (name === "a") {
|
|
1298
|
+
const href = linkHref;
|
|
1299
|
+
linkHref = void 0;
|
|
1300
|
+
if (markdown && href !== void 0 && href !== "") result += "](" + href + ")";
|
|
1301
|
+
return;
|
|
1302
|
+
}
|
|
1303
|
+
if (name === "strong" || name === "b") {
|
|
1304
|
+
marker("**");
|
|
1305
|
+
return;
|
|
1306
|
+
}
|
|
1307
|
+
if (name === "em" || name === "i") {
|
|
1308
|
+
marker("*");
|
|
1309
|
+
return;
|
|
1310
|
+
}
|
|
1311
|
+
if (name === "del" || name === "s" || name === "strike") {
|
|
1312
|
+
marker("~~");
|
|
1313
|
+
return;
|
|
1314
|
+
}
|
|
1315
|
+
if (name === "code" && codeOpen) {
|
|
1316
|
+
marker("`");
|
|
1317
|
+
codeOpen = false;
|
|
1318
|
+
return;
|
|
1319
|
+
}
|
|
1320
|
+
if (name === "td" || name === "th") {
|
|
1321
|
+
cellsInRow += 1;
|
|
1322
|
+
return;
|
|
1323
|
+
}
|
|
1324
|
+
if (name === "tr") {
|
|
1325
|
+
ensureNewline();
|
|
1326
|
+
return;
|
|
1327
|
+
}
|
|
1328
|
+
if (name === "ul" || name === "ol") {
|
|
1329
|
+
lists.pop();
|
|
1330
|
+
ensureBlank();
|
|
1331
|
+
return;
|
|
1332
|
+
}
|
|
1333
|
+
if (BLOCK_ELEMENTS.has(name)) ensureBlank();
|
|
1334
|
+
};
|
|
1335
|
+
TOKEN.lastIndex = 0;
|
|
1336
|
+
let match;
|
|
1337
|
+
while ((match = TOKEN.exec(html)) !== null) {
|
|
1338
|
+
const whole = match[0];
|
|
1339
|
+
const openName = match[1];
|
|
1340
|
+
const closeName = match[4];
|
|
1341
|
+
if (closeName !== void 0) {
|
|
1342
|
+
const name = closeName.toLowerCase();
|
|
1343
|
+
if (skipTag !== void 0) {
|
|
1344
|
+
if (name === skipTag) {
|
|
1345
|
+
skipDepth -= 1;
|
|
1346
|
+
if (skipDepth === 0) skipTag = void 0;
|
|
1347
|
+
}
|
|
1348
|
+
continue;
|
|
1349
|
+
}
|
|
1350
|
+
close(name);
|
|
1351
|
+
continue;
|
|
1352
|
+
}
|
|
1353
|
+
if (openName !== void 0) {
|
|
1354
|
+
const name = openName.toLowerCase();
|
|
1355
|
+
const selfClosing = match[3] === "/";
|
|
1356
|
+
if (skipTag !== void 0) {
|
|
1357
|
+
if (name === skipTag && !selfClosing) skipDepth += 1;
|
|
1358
|
+
continue;
|
|
1359
|
+
}
|
|
1360
|
+
if (SKIP_ELEMENTS.has(name)) {
|
|
1361
|
+
if (!selfClosing) {
|
|
1362
|
+
skipTag = name;
|
|
1363
|
+
skipDepth = 1;
|
|
1364
|
+
}
|
|
1365
|
+
continue;
|
|
1366
|
+
}
|
|
1367
|
+
open(name, match[2] ?? "");
|
|
1368
|
+
continue;
|
|
1369
|
+
}
|
|
1370
|
+
if (whole.startsWith("<!--")) continue;
|
|
1371
|
+
if (skipTag !== void 0) continue;
|
|
1372
|
+
result += decodeEntities(whole).replace(/[\t\r\n\u00a0]+/g, " ");
|
|
1373
|
+
}
|
|
1374
|
+
return result.replace(/[ \t]+/g, " ").replace(/ *\n */g, "\n").replace(/\n{3,}/g, "\n\n").trim();
|
|
1375
|
+
}
|
|
1376
|
+
//#endregion
|
|
1377
|
+
//#region lib/render.js
|
|
1378
|
+
/**
|
|
1379
|
+
* Browser rendering for pages whose article body is built by client script, and
|
|
1380
|
+
* for pages the plain HTTP path is refused. Playwright is imported lazily
|
|
1381
|
+
* through a variable specifier, so the plugin loads, and every statically
|
|
1382
|
+
* served article still works, when the optional driver is absent. One browser
|
|
1383
|
+
* instance is reused and closed on an idle timer or when the owning fiber
|
|
1384
|
+
* disposes. A render waits for the configured body selector to fill, bounded by
|
|
1385
|
+
* its own window, instead of trusting a fixed settle time.
|
|
1386
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/render
|
|
1387
|
+
*/
|
|
1388
|
+
var __rewriteRelativeImportExtension = function(path, preserveJsx) {
|
|
1389
|
+
if (typeof path === "string" && /^\.\.?\//.test(path)) return path.replace(/\.(tsx)$|((?:\.d)?)((?:\.[^./]+?)?)\.([cm]?)ts$/i, function(m, tsx, d, ext, cm) {
|
|
1390
|
+
return tsx ? preserveJsx ? ".jsx" : ".js" : d && (!ext || !cm) ? m : d + ext + "." + cm.toLowerCase() + "js";
|
|
1391
|
+
});
|
|
1392
|
+
return path;
|
|
1393
|
+
};
|
|
1394
|
+
/** Extra milliseconds lazy images need after the page is scrolled to its end. */
|
|
1395
|
+
const SCROLL_SETTLE_MS = 400;
|
|
1396
|
+
/**
|
|
1397
|
+
* The expression the readiness wait polls: the article body element exists and
|
|
1398
|
+
* carries text or an image. A body that stayed empty is what a page with no
|
|
1399
|
+
* article behind the link looks like, so the caller reads the page as it stands
|
|
1400
|
+
* rather than waiting the whole window out for nothing.
|
|
1401
|
+
* @param selector - the selector the scripted body fills.
|
|
1402
|
+
* @returns a self-contained expression whose value is a boolean.
|
|
1403
|
+
*/
|
|
1404
|
+
function readinessExpression(selector) {
|
|
1405
|
+
return "(() => { const body = document.querySelector(" + JSON.stringify(selector) + "); if (body === null) return false; return (body.textContent || \"\").trim().length > 0 || body.querySelector(\"img\") !== null; })()";
|
|
1406
|
+
}
|
|
1407
|
+
/**
|
|
1408
|
+
* Load the optional Playwright package through a variable specifier, so the
|
|
1409
|
+
* emitted bundle never inlines it and a deployment without it still loads.
|
|
1410
|
+
* @returns the chromium launcher.
|
|
1411
|
+
*/
|
|
1412
|
+
async function loadChromium() {
|
|
1413
|
+
const specifier = "playwright";
|
|
1414
|
+
let loaded;
|
|
1415
|
+
try {
|
|
1416
|
+
loaded = await import(__rewriteRelativeImportExtension(specifier));
|
|
1417
|
+
} catch {
|
|
1418
|
+
throw new Error("wechat: JS rendering needs the optional playwright package installed for this plugin");
|
|
1419
|
+
}
|
|
1420
|
+
const chromium = loaded.chromium;
|
|
1421
|
+
if (chromium === void 0) throw new Error("wechat: the playwright package exposes no chromium launcher");
|
|
1422
|
+
return chromium;
|
|
1423
|
+
}
|
|
1424
|
+
/**
|
|
1425
|
+
* Split one Cookie header into Playwright cookie objects for the target host.
|
|
1426
|
+
* @param header - the deployment's Cookie header.
|
|
1427
|
+
* @param target - the page being rendered.
|
|
1428
|
+
* @returns one cookie object per header entry.
|
|
1429
|
+
*/
|
|
1430
|
+
function cookiesFor(header, target) {
|
|
1431
|
+
let host = "mp.weixin.qq.com";
|
|
1432
|
+
try {
|
|
1433
|
+
host = new URL(target).hostname;
|
|
1434
|
+
} catch {}
|
|
1435
|
+
return header.split(";").map((part) => part.trim()).filter((part) => part !== "").map((part) => {
|
|
1436
|
+
const equals = part.indexOf("=");
|
|
1437
|
+
return {
|
|
1438
|
+
name: equals === -1 ? part : part.slice(0, equals),
|
|
1439
|
+
value: equals === -1 ? "" : part.slice(equals + 1),
|
|
1440
|
+
domain: host,
|
|
1441
|
+
path: "/"
|
|
1442
|
+
};
|
|
1443
|
+
});
|
|
1444
|
+
}
|
|
1445
|
+
/**
|
|
1446
|
+
* Build the plugin instance's renderer.
|
|
1447
|
+
* @param policy - the deployment's driver, browser, and timing values.
|
|
1448
|
+
* @param load - launcher factory, replaced by tests with a structural fake.
|
|
1449
|
+
* @returns the renderer whose dispose() the plugin fiber owns.
|
|
1450
|
+
*/
|
|
1451
|
+
function createRenderer(policy, load = loadChromium) {
|
|
1452
|
+
let browser;
|
|
1453
|
+
let launching;
|
|
1454
|
+
let idleTimer;
|
|
1455
|
+
const closeBrowser = async () => {
|
|
1456
|
+
if (idleTimer !== void 0) {
|
|
1457
|
+
clearTimeout(idleTimer);
|
|
1458
|
+
idleTimer = void 0;
|
|
1459
|
+
}
|
|
1460
|
+
const current = browser;
|
|
1461
|
+
browser = void 0;
|
|
1462
|
+
launching = void 0;
|
|
1463
|
+
if (current !== void 0) await current.close().catch(() => void 0);
|
|
1464
|
+
};
|
|
1465
|
+
const scheduleIdleClose = () => {
|
|
1466
|
+
if (idleTimer !== void 0) clearTimeout(idleTimer);
|
|
1467
|
+
idleTimer = setTimeout(() => {
|
|
1468
|
+
closeBrowser();
|
|
1469
|
+
}, policy.idleMs);
|
|
1470
|
+
};
|
|
1471
|
+
const acquire = async () => {
|
|
1472
|
+
if (idleTimer !== void 0) {
|
|
1473
|
+
clearTimeout(idleTimer);
|
|
1474
|
+
idleTimer = void 0;
|
|
1475
|
+
}
|
|
1476
|
+
if (browser?.isConnected() === true) return browser;
|
|
1477
|
+
if (launching === void 0) {
|
|
1478
|
+
launching = (async () => {
|
|
1479
|
+
const chromium = await load();
|
|
1480
|
+
const options = { headless: policy.headless };
|
|
1481
|
+
if (policy.executablePath !== "") options.executablePath = policy.executablePath;
|
|
1482
|
+
else if (policy.channel !== "") options.channel = policy.channel;
|
|
1483
|
+
return chromium.launch(options);
|
|
1484
|
+
})();
|
|
1485
|
+
launching.catch(() => {
|
|
1486
|
+
launching = void 0;
|
|
1487
|
+
});
|
|
1488
|
+
}
|
|
1489
|
+
const started = await launching;
|
|
1490
|
+
browser = started;
|
|
1491
|
+
return started;
|
|
1492
|
+
};
|
|
1493
|
+
const renderWithBrowser = async (url, signal) => {
|
|
1494
|
+
const active = await acquire();
|
|
1495
|
+
const contextOptions = {
|
|
1496
|
+
userAgent: policy.userAgent,
|
|
1497
|
+
locale: "zh-CN",
|
|
1498
|
+
viewport: {
|
|
1499
|
+
width: 1280,
|
|
1500
|
+
height: 900
|
|
1501
|
+
}
|
|
1502
|
+
};
|
|
1503
|
+
if (policy.referer !== "") contextOptions.extraHTTPHeaders = { referer: policy.referer };
|
|
1504
|
+
const context = await active.newContext(contextOptions);
|
|
1505
|
+
try {
|
|
1506
|
+
if (policy.cookie !== "") await context.addCookies(cookiesFor(policy.cookie, url));
|
|
1507
|
+
const page = await context.newPage();
|
|
1508
|
+
const onAbort = () => {
|
|
1509
|
+
page.close().catch(() => void 0);
|
|
1510
|
+
};
|
|
1511
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
1512
|
+
try {
|
|
1513
|
+
await page.goto(url, {
|
|
1514
|
+
waitUntil: "domcontentloaded",
|
|
1515
|
+
timeout: policy.timeoutMs
|
|
1516
|
+
});
|
|
1517
|
+
if (policy.readySelector !== "") try {
|
|
1518
|
+
await page.waitForFunction(readinessExpression(policy.readySelector), void 0, { timeout: policy.waitMs });
|
|
1519
|
+
} catch {}
|
|
1520
|
+
await page.waitForTimeout(policy.settleMs);
|
|
1521
|
+
await page.evaluate("window.scrollTo(0, document.body.scrollHeight)");
|
|
1522
|
+
await page.waitForTimeout(SCROLL_SETTLE_MS);
|
|
1523
|
+
return await page.content();
|
|
1524
|
+
} finally {
|
|
1525
|
+
signal?.removeEventListener("abort", onAbort);
|
|
1526
|
+
}
|
|
1527
|
+
} finally {
|
|
1528
|
+
await context.close().catch(() => void 0);
|
|
1529
|
+
scheduleIdleClose();
|
|
1530
|
+
}
|
|
1531
|
+
};
|
|
1532
|
+
const renderWithHttp = async (url, signal) => {
|
|
1533
|
+
if (policy.url === "") throw new Error("wechat: the http renderer driver needs a rendererUrl");
|
|
1534
|
+
const timeout = AbortSignal.timeout(policy.timeoutMs);
|
|
1535
|
+
const combined = signal === void 0 ? timeout : AbortSignal.any([signal, timeout]);
|
|
1536
|
+
const response = await fetch(policy.url, {
|
|
1537
|
+
method: "POST",
|
|
1538
|
+
headers: {
|
|
1539
|
+
"content-type": "application/json",
|
|
1540
|
+
accept: "text/html, application/json"
|
|
1541
|
+
},
|
|
1542
|
+
body: JSON.stringify({ url }),
|
|
1543
|
+
signal: combined
|
|
1544
|
+
});
|
|
1545
|
+
if (!response.ok) throw new Error("wechat: the renderer endpoint returned HTTP " + response.status);
|
|
1546
|
+
const text = await response.text();
|
|
1547
|
+
if (!(response.headers.get("content-type") ?? "").includes("json")) return text;
|
|
1548
|
+
const parsed = JSON.parse(text);
|
|
1549
|
+
const html = typeof parsed === "object" && parsed !== null ? parsed.html : void 0;
|
|
1550
|
+
if (typeof html !== "string" || html === "") throw new Error("wechat: the renderer response carried no html string");
|
|
1551
|
+
return html;
|
|
1552
|
+
};
|
|
1553
|
+
const render = async (url, signal) => {
|
|
1554
|
+
await policy.pace(signal);
|
|
1555
|
+
return policy.driver === "http" ? renderWithHttp(url, signal) : renderWithBrowser(url, signal);
|
|
1556
|
+
};
|
|
1557
|
+
return {
|
|
1558
|
+
render,
|
|
1559
|
+
dispose: closeBrowser
|
|
1560
|
+
};
|
|
1561
|
+
}
|
|
1562
|
+
/** Base name used when neither the title nor the link yields anything usable. */
|
|
1563
|
+
const UNTITLED = "untitled";
|
|
1564
|
+
/** Characters no file name may carry: Windows' reserved set, the POSIX separator, and controls. */
|
|
1565
|
+
const ILLEGAL_CHARACTERS = /[\u0000-\u001f\u007f-\u009f<>:"/\\|?*]/g;
|
|
1566
|
+
/**
|
|
1567
|
+
* Characters a file name may carry but a reader cannot see: zero-width spaces,
|
|
1568
|
+
* bidirectional overrides, word joiners, and the byte-order mark. WeChat titles
|
|
1569
|
+
* are full of them, and a name made only of them looks empty in a directory
|
|
1570
|
+
* listing while still being a distinct file.
|
|
1571
|
+
*/
|
|
1572
|
+
const INVISIBLE_CHARACTERS = /[\u200b-\u200f\u202a-\u202e\u2060-\u2064\u206a-\u206f\ufeff]/g;
|
|
1573
|
+
/** Device names Windows reserves in every directory, with or without an extension. */
|
|
1574
|
+
const RESERVED_DEVICE_NAME = /^(?:con|prn|aux|nul|com[1-9]|lpt[1-9])$/i;
|
|
1575
|
+
/** Leading and trailing characters a file name may not carry on Windows. */
|
|
1576
|
+
const EDGE_NOISE = /^[\s.\-]+|[\s.\-]+$/g;
|
|
1577
|
+
/** Message identity a short link carries in its path, e.g. `/s/AbCdEf123`. */
|
|
1578
|
+
const SHORT_LINK_PATH = /^\/s\/([^/?#]+)$/;
|
|
1579
|
+
/**
|
|
1580
|
+
* Fold one article title into a file-name base: Unicode-normalized, with the
|
|
1581
|
+
* characters no file system accepts replaced, and whitespace, separators, and
|
|
1582
|
+
* edge noise collapsed.
|
|
1583
|
+
* @param title - the article title as the page carried it.
|
|
1584
|
+
* @returns the normalized base, or an empty string when nothing survives.
|
|
1585
|
+
*/
|
|
1586
|
+
function normalizeTitle(title) {
|
|
1587
|
+
return title.normalize("NFC").replace(INVISIBLE_CHARACTERS, "").replace(ILLEGAL_CHARACTERS, "-").replace(/\s+/g, " ").replace(/-{2,}/g, "-").replace(EDGE_NOISE, "");
|
|
1588
|
+
}
|
|
1589
|
+
/**
|
|
1590
|
+
* A stable name for an article whose title yielded none, taken from its own
|
|
1591
|
+
* link: the message identity is what makes two untitled articles different, and
|
|
1592
|
+
* a name derived from it separates them without a collision suffix.
|
|
1593
|
+
* @param source - the article's resolved link.
|
|
1594
|
+
* @returns the normalized stem, or an empty string when the link names no identity.
|
|
1595
|
+
*/
|
|
1596
|
+
function linkStem(source) {
|
|
1597
|
+
let parsed;
|
|
1598
|
+
try {
|
|
1599
|
+
parsed = new URL(source);
|
|
1600
|
+
} catch {
|
|
1601
|
+
return "";
|
|
1602
|
+
}
|
|
1603
|
+
const identity = SHORT_LINK_PATH.exec(parsed.pathname)?.[1] ?? parsed.searchParams.get("sn") ?? [parsed.searchParams.get("mid"), parsed.searchParams.get("idx")].filter((part) => part !== null).join("-");
|
|
1604
|
+
return identity === void 0 || identity === "" ? "" : normalizeTitle(identity);
|
|
1605
|
+
}
|
|
1606
|
+
/**
|
|
1607
|
+
* Build the file name one article is saved under.
|
|
1608
|
+
* @param title - the article title as the page carried it.
|
|
1609
|
+
* @param maxChars - longest base name, in code points, the deployment allows.
|
|
1610
|
+
* @param source - the article's resolved link, used when the title yields no name.
|
|
1611
|
+
* @returns a base name carrying {@link FILE_EXTENSION}, creatable on Windows and POSIX.
|
|
1612
|
+
*/
|
|
1613
|
+
function articleFileName(title, maxChars, source = "") {
|
|
1614
|
+
const normalized = normalizeTitle(title);
|
|
1615
|
+
if (normalized !== "") return finish(normalized, maxChars);
|
|
1616
|
+
const stem = linkStem(source);
|
|
1617
|
+
return finish(stem === "" ? UNTITLED : stem, maxChars);
|
|
1618
|
+
}
|
|
1619
|
+
/** Clip one base name and escape the device names Windows reserves. */
|
|
1620
|
+
function finish(base, maxChars) {
|
|
1621
|
+
const clipped = clip$1(base, maxChars);
|
|
1622
|
+
return (RESERVED_DEVICE_NAME.test(clipped) ? "_" + clipped : clipped) + ".md";
|
|
1623
|
+
}
|
|
1624
|
+
/** Cut a base name to the ceiling without leaving edge noise or an empty name behind. */
|
|
1625
|
+
function clip$1(base, maxChars) {
|
|
1626
|
+
const characters = [...base];
|
|
1627
|
+
if (characters.length <= maxChars) return base;
|
|
1628
|
+
const cut = characters.slice(0, maxChars).join("").replace(EDGE_NOISE, "");
|
|
1629
|
+
return cut === "" ? UNTITLED : cut;
|
|
1630
|
+
}
|
|
1631
|
+
//#endregion
|
|
1632
|
+
//#region lib/store.js
|
|
1633
|
+
/**
|
|
1634
|
+
* File persistence for scraped articles: the article's file name, the
|
|
1635
|
+
* same-title collision rule, and the write itself. Every operation goes through
|
|
1636
|
+
* the mounted `ctx.fs` capability rather than the host filesystem directly, so
|
|
1637
|
+
* a deployment's file policy sees the write and its own backend decides how the
|
|
1638
|
+
* bytes are published. The calling session's per-call sandbox policy travels
|
|
1639
|
+
* with the write, which is what lets a confining backend fence it against the
|
|
1640
|
+
* workspace the session actually runs in.
|
|
1641
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/store
|
|
1642
|
+
*/
|
|
1643
|
+
/** Number-suffixed names tried before a colliding name falls back to a timestamp. */
|
|
1644
|
+
const MAX_COLLISION_SUFFIX = 50;
|
|
1645
|
+
/** Opening line of the YAML front matter every article file starts with. */
|
|
1646
|
+
const FRONT_MATTER_FENCE = "---";
|
|
1647
|
+
/** The structured code a sandboxing filesystem reports a fenced write with. */
|
|
1648
|
+
const SANDBOX_DENIED = "FS_SANDBOX_DENIED";
|
|
1649
|
+
/** Front-matter key carrying the article's resolved link. */
|
|
1650
|
+
const SOURCE_KEY = "source: ";
|
|
1651
|
+
/**
|
|
1652
|
+
* Whether one failure is the file sandbox refusing the write, judged by the
|
|
1653
|
+
* stable code the capability publishes rather than by its message text.
|
|
1654
|
+
* @param error - the error the write threw.
|
|
1655
|
+
* @returns true when a confining filesystem denied the write.
|
|
1656
|
+
*/
|
|
1657
|
+
function isSandboxDenial(error) {
|
|
1658
|
+
return typeof error === "object" && error !== null && error.code === SANDBOX_DENIED;
|
|
1659
|
+
}
|
|
1660
|
+
/**
|
|
1661
|
+
* Drop the body, which now lives in the file, from the value the call records.
|
|
1662
|
+
*
|
|
1663
|
+
* Written out field by field rather than spread-and-delete: adding a field to the
|
|
1664
|
+
* article then stops this compiling, instead of silently leaving the new field —
|
|
1665
|
+
* and any body-sized thing it might carry — in the value and the log.
|
|
1666
|
+
* @param drafted - the article as it was built, body included.
|
|
1667
|
+
* @returns the article as the call records it.
|
|
1668
|
+
*/
|
|
1669
|
+
function recordedArticle(drafted) {
|
|
1670
|
+
return {
|
|
1671
|
+
url: drafted.url,
|
|
1672
|
+
resolvedUrl: drafted.resolvedUrl,
|
|
1673
|
+
title: drafted.title,
|
|
1674
|
+
account: drafted.account,
|
|
1675
|
+
author: drafted.author,
|
|
1676
|
+
publishTime: drafted.publishTime,
|
|
1677
|
+
digest: drafted.digest,
|
|
1678
|
+
cover: drafted.cover,
|
|
1679
|
+
chars: drafted.chars,
|
|
1680
|
+
excerpt: drafted.excerpt,
|
|
1681
|
+
excerptClipped: drafted.excerptClipped,
|
|
1682
|
+
images: drafted.images,
|
|
1683
|
+
rendered: drafted.rendered
|
|
1684
|
+
};
|
|
1685
|
+
}
|
|
1686
|
+
/**
|
|
1687
|
+
* Write one article's Markdown file, and return the value the call records.
|
|
1688
|
+
*
|
|
1689
|
+
* The body stops here: it is written to the file, and what the caller gets back is
|
|
1690
|
+
* the article without it. That is the whole point of the split — a call that
|
|
1691
|
+
* returned the body would pay the article's full length in context and in the log,
|
|
1692
|
+
* then have the reader look at a window of it anyway.
|
|
1693
|
+
* @param article - the article as it was built, body included.
|
|
1694
|
+
* @param policy - the deployment's output directory, name ceiling, and per-call sandbox policy.
|
|
1695
|
+
* @param fs - the mounted filesystem capability the write goes through.
|
|
1696
|
+
* @param signal - caller cancellation, forwarded to the write.
|
|
1697
|
+
* @returns the saved path, in the backend's execution world, and the recorded value.
|
|
1698
|
+
*/
|
|
1699
|
+
async function saveArticle(article, policy, fs, signal) {
|
|
1700
|
+
const name = articleFileName(article.title, policy.fileNameMaxChars, article.resolvedUrl);
|
|
1701
|
+
const target = await resolveTarget(fs, policy.dir, name, article.resolvedUrl);
|
|
1702
|
+
await write(fs, target, articleDocument(article) + "\n", policy, signal);
|
|
1703
|
+
return {
|
|
1704
|
+
file: fs.processPath(target),
|
|
1705
|
+
article: recordedArticle(article)
|
|
1706
|
+
};
|
|
1707
|
+
}
|
|
1708
|
+
/**
|
|
1709
|
+
* Publish one article file, turning a sandbox refusal into the one instruction
|
|
1710
|
+
* that resolves it.
|
|
1711
|
+
*
|
|
1712
|
+
* A refusal under a stamped policy is a statement about the output directory
|
|
1713
|
+
* rather than about the article: the confining backend fences the write by the
|
|
1714
|
+
* workspace root the calling session runs in, so an outputDir outside that root
|
|
1715
|
+
* has to move. Reporting the bare provider text would leave that unsaid. A
|
|
1716
|
+
* call that stamped no policy gets the provider's own message, which already
|
|
1717
|
+
* names the mode it was fenced by.
|
|
1718
|
+
* @param fs - the mounted filesystem capability.
|
|
1719
|
+
* @param target - the resolved file target.
|
|
1720
|
+
* @param document - the complete file content.
|
|
1721
|
+
* @param policy - the deployment's output directory and per-call sandbox policy.
|
|
1722
|
+
* @param signal - caller cancellation.
|
|
1723
|
+
* @throws Error the provider raised, or the same failure restated with the remedy.
|
|
1724
|
+
*/
|
|
1725
|
+
async function write(fs, target, document, policy, signal) {
|
|
1726
|
+
try {
|
|
1727
|
+
await fs.writeText(target, document, void 0, signal, policy.sandboxPolicy);
|
|
1728
|
+
} catch (error) {
|
|
1729
|
+
const stamped = policy.sandboxPolicy;
|
|
1730
|
+
if (stamped === void 0 || !isSandboxDenial(error)) throw error;
|
|
1731
|
+
throw new Error("wechat: the file sandbox refused to write " + fs.processPath(target) + " under mode " + stamped.mode + "; this session's workspace is " + stamped.workspaceRoot + ", so set the plugin row's outputDir inside it", { cause: error });
|
|
1732
|
+
}
|
|
1733
|
+
}
|
|
1734
|
+
/**
|
|
1735
|
+
* Choose the file one article is written to.
|
|
1736
|
+
* @param fs - the mounted filesystem capability.
|
|
1737
|
+
* @param dir - the absolute output directory.
|
|
1738
|
+
* @param name - the normalized file name, extension included.
|
|
1739
|
+
* @param source - the article's resolved link, the identity a collision is judged by.
|
|
1740
|
+
* @returns a free target, or the target already holding this same article.
|
|
1741
|
+
*/
|
|
1742
|
+
async function resolveTarget(fs, dir, name, source) {
|
|
1743
|
+
const base = name.slice(0, -3);
|
|
1744
|
+
const first = await fs.resolve(join(dir, name));
|
|
1745
|
+
if (await matchesOrIsFree(fs, first, source)) return first;
|
|
1746
|
+
for (let suffix = 2; suffix <= MAX_COLLISION_SUFFIX; suffix += 1) {
|
|
1747
|
+
const candidate = await fs.resolve(join(dir, base + "-" + suffix + ".md"));
|
|
1748
|
+
if (await matchesOrIsFree(fs, candidate, source)) return candidate;
|
|
1749
|
+
}
|
|
1750
|
+
return fs.resolve(join(dir, base + "-" + Date.now() + ".md"));
|
|
1751
|
+
}
|
|
1752
|
+
/**
|
|
1753
|
+
* Whether one target is free, or already holds the article with this resolved link.
|
|
1754
|
+
* @param fs - the mounted filesystem capability.
|
|
1755
|
+
* @param target - the candidate file target.
|
|
1756
|
+
* @param source - the article's resolved link.
|
|
1757
|
+
* @returns true when the caller may write the article to this target.
|
|
1758
|
+
*/
|
|
1759
|
+
async function matchesOrIsFree(fs, target, source) {
|
|
1760
|
+
if (await fs.stat(target) === void 0) return true;
|
|
1761
|
+
return await recordedSource(fs, target) === source;
|
|
1762
|
+
}
|
|
1763
|
+
/**
|
|
1764
|
+
* The resolved link recorded in one file's front matter.
|
|
1765
|
+
* @param fs - the mounted filesystem capability.
|
|
1766
|
+
* @param target - the existing file target.
|
|
1767
|
+
* @returns the recorded link, or an empty string when the file is unreadable or carries none.
|
|
1768
|
+
*/
|
|
1769
|
+
async function recordedSource(fs, target) {
|
|
1770
|
+
let text;
|
|
1771
|
+
try {
|
|
1772
|
+
text = await fs.readText(target);
|
|
1773
|
+
} catch {
|
|
1774
|
+
return "";
|
|
1775
|
+
}
|
|
1776
|
+
const lines = text.split("\n");
|
|
1777
|
+
if (lines[0] !== FRONT_MATTER_FENCE) return "";
|
|
1778
|
+
for (const line of lines.slice(1)) {
|
|
1779
|
+
if (line === FRONT_MATTER_FENCE) break;
|
|
1780
|
+
if (!line.startsWith(SOURCE_KEY)) continue;
|
|
1781
|
+
try {
|
|
1782
|
+
const parsed = JSON.parse(line.slice(8));
|
|
1783
|
+
return typeof parsed === "string" ? parsed : "";
|
|
1784
|
+
} catch {
|
|
1785
|
+
return "";
|
|
1786
|
+
}
|
|
1787
|
+
}
|
|
1788
|
+
return "";
|
|
1789
|
+
}
|
|
1790
|
+
//#endregion
|
|
1791
|
+
//#region lib/index.js
|
|
1792
|
+
/**
|
|
1793
|
+
* Model-facing wechat_scrape tool. One call retrieves articles from WeChat
|
|
1794
|
+
* official accounts and saves each as its own Markdown file named after the
|
|
1795
|
+
* article title. A link may be an article page, or a list or collection page
|
|
1796
|
+
* whose article links are parsed out and then scraped one by one the same way;
|
|
1797
|
+
* a page that carries no article body is read as a list page. The HTTP layer
|
|
1798
|
+
* paces, retries, carries the session cookies WeChat sets, and reports WeChat's
|
|
1799
|
+
* block pages; a page whose body is built by client script, and a page the plain
|
|
1800
|
+
* path is refused for, fall back to the optional browser renderer. Only a page
|
|
1801
|
+
* that is still an article with a usable body after promotion filtering is ever
|
|
1802
|
+
* written to a file: a block page, a withdrawn article, and an article the
|
|
1803
|
+
* filtering left empty are reported as failures instead. The tool reads no
|
|
1804
|
+
* session state beyond the workspace directory it writes into and returns the
|
|
1805
|
+
* canonical value alone, so model-facing text and any replay read the same
|
|
1806
|
+
* recorded call.
|
|
1807
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape
|
|
1808
|
+
*/
|
|
1809
|
+
const name = "wechat-scrape";
|
|
1810
|
+
const inject = [
|
|
1811
|
+
"tools",
|
|
1812
|
+
"systemPrompt",
|
|
1813
|
+
"fs"
|
|
1814
|
+
];
|
|
1815
|
+
/** Wire name this plugin registers. */
|
|
1816
|
+
const TOOL_NAME = "wechat_scrape";
|
|
1817
|
+
/** Body formats the tool accepts, in the order the model sees them. */
|
|
1818
|
+
const BODY_FORMATS = ["markdown", "text"];
|
|
1819
|
+
/**
|
|
1820
|
+
* Name and order of the routing section this plugin contributes. Repository-owned
|
|
1821
|
+
* tool sections allocate their order centrally; an external contribution states
|
|
1822
|
+
* its own finite order and sits after the built-in tool band (TOOL_REPORT, 2900).
|
|
1823
|
+
*/
|
|
1824
|
+
const SECTION_NAME = "tool:wechat_scrape";
|
|
1825
|
+
/** Order of {@link SECTION_NAME}, after the built-in tool sections. */
|
|
1826
|
+
const SECTION_ORDER = 2970;
|
|
1827
|
+
/**
|
|
1828
|
+
* Schemastery configuration for the tool. Defaults cover a single-machine
|
|
1829
|
+
* deployment that writes into the calling session's workspace; the
|
|
1830
|
+
* request-identity, list-depth, and output values are the ones a restricted or
|
|
1831
|
+
* authenticated deployment overrides from cordis.yml.
|
|
1832
|
+
*/
|
|
1833
|
+
const Config = z.object({
|
|
1834
|
+
userAgent: z.string().default("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"),
|
|
1835
|
+
referer: z.string().default("https://mp.weixin.qq.com/"),
|
|
1836
|
+
cookie: z.string().default(""),
|
|
1837
|
+
cookieJar: z.boolean().default(true),
|
|
1838
|
+
minIntervalMs: z.number().step(1).min(0).default(2e3),
|
|
1839
|
+
minIntervalJitterMs: z.number().step(1).min(0).default(1e3),
|
|
1840
|
+
blockCooldownMs: z.number().step(1).min(0).default(6e4),
|
|
1841
|
+
timeoutMs: z.number().step(1).min(1e3).default(2e4),
|
|
1842
|
+
maxRetries: z.number().step(1).min(0).max(5).default(2),
|
|
1843
|
+
backoffBaseMs: z.number().step(1).min(0).default(800),
|
|
1844
|
+
maxRetryAfterMs: z.number().step(1).min(0).default(6e4),
|
|
1845
|
+
maxChars: z.number().step(1).min(200).default(4e4),
|
|
1846
|
+
maxImages: z.number().step(1).min(0).default(50),
|
|
1847
|
+
maxArticles: z.number().step(1).min(1).max(200).default(10),
|
|
1848
|
+
maxListPages: z.number().step(1).min(0).max(200).default(0),
|
|
1849
|
+
filterAds: z.boolean().default(true),
|
|
1850
|
+
promoMaxChars: z.number().step(1).min(0).default(200),
|
|
1851
|
+
promoPatterns: z.array(z.string()).default([]),
|
|
1852
|
+
minBodyChars: z.number().step(1).min(0).default(10),
|
|
1853
|
+
outputDir: z.string().default("wechat-articles"),
|
|
1854
|
+
fileNameMaxChars: z.number().step(1).min(8).max(200).default(80),
|
|
1855
|
+
renderer: z.union([
|
|
1856
|
+
"auto",
|
|
1857
|
+
"never",
|
|
1858
|
+
"always"
|
|
1859
|
+
]).default("auto"),
|
|
1860
|
+
rendererDriver: z.union(["playwright", "http"]).default("playwright"),
|
|
1861
|
+
rendererUrl: z.string().default(""),
|
|
1862
|
+
browserChannel: z.string().default("msedge"),
|
|
1863
|
+
browserExecutablePath: z.string().default(""),
|
|
1864
|
+
headless: z.boolean().default(true),
|
|
1865
|
+
browserIdleMs: z.number().step(1).min(1e3).default(6e4),
|
|
1866
|
+
renderReadySelector: z.string().default("#js_content"),
|
|
1867
|
+
renderWaitMs: z.number().step(1).min(0).default(8e3),
|
|
1868
|
+
renderSettleMs: z.number().step(1).min(0).default(1200)
|
|
1869
|
+
});
|
|
1870
|
+
const DESCRIPTION = [
|
|
1871
|
+
"Scrape WeChat official-account articles and save each as its own Markdown file named after the article title.",
|
|
1872
|
+
"A link may name one article, or a page that lists articles (a collection or archive page); a listing link is scraped as the articles it points to.",
|
|
1873
|
+
"Returns each article's title, account, author, publish time, body text, image links, and the file it was saved to, plus the article links every listing page produced.",
|
|
1874
|
+
"Accepts a short /s/<id> link, a full mp.weixin.qq.com/s?... link, or an mp.weixin.qq.com collection link."
|
|
1875
|
+
].join(" ");
|
|
1876
|
+
/** Model-facing routing rule: when to make a call, which a schema cannot carry. */
|
|
1877
|
+
const SECTION_TEXT = [
|
|
1878
|
+
"When the reader needs the text of WeChat official-account articles (an mp.weixin.qq.com link), answer with wechat_scrape:",
|
|
1879
|
+
"it saves each article as a Markdown file named after its title and returns the title, account, author, publish time, body text, and image links.",
|
|
1880
|
+
"Give it a collection or list page link to scrape every article that page lists."
|
|
1881
|
+
].join(" ");
|
|
1882
|
+
/** The error text for one WeChat block page. */
|
|
1883
|
+
function blockedError(marker, url) {
|
|
1884
|
+
return /* @__PURE__ */ new Error("wechat: WeChat refused " + url + " (page said \"" + marker + "\"); set a logged-in cookie on the plugin row and raise minIntervalMs");
|
|
1885
|
+
}
|
|
1886
|
+
/** The error text for one page whose article is gone rather than withheld. */
|
|
1887
|
+
function unavailableError(reason, url) {
|
|
1888
|
+
return /* @__PURE__ */ new Error("wechat: " + url + " carries no article (page said \"" + reason + "\"); the publisher withdrew it or WeChat removed it, so nothing was saved");
|
|
1889
|
+
}
|
|
1890
|
+
/** The text one failed link is reported with. */
|
|
1891
|
+
function failureText(error) {
|
|
1892
|
+
return error instanceof Error ? error.message : String(error);
|
|
1893
|
+
}
|
|
1894
|
+
/**
|
|
1895
|
+
* Validate the model-supplied link before any request is made.
|
|
1896
|
+
* @param raw - the link as the model supplied it.
|
|
1897
|
+
* @returns the trimmed link.
|
|
1898
|
+
* @throws Error when the link is empty or names another host.
|
|
1899
|
+
*/
|
|
1900
|
+
function normalizeUrl(raw) {
|
|
1901
|
+
const trimmed = raw.trim();
|
|
1902
|
+
if (trimmed === "") throw new Error("invalid wechat: url must be a non-empty string");
|
|
1903
|
+
if (!isMpArticleUrl(trimmed)) throw new Error("invalid wechat: url must be an mp.weixin.qq.com link");
|
|
1904
|
+
return trimmed;
|
|
1905
|
+
}
|
|
1906
|
+
/**
|
|
1907
|
+
* Collect the links one call asked for.
|
|
1908
|
+
* @param args - the model-supplied call arguments.
|
|
1909
|
+
* @param maxArticles - most links one call may start from.
|
|
1910
|
+
* @returns the distinct links, in request order.
|
|
1911
|
+
* @throws Error when the call names no link, a foreign link, or more links than the ceiling.
|
|
1912
|
+
*/
|
|
1913
|
+
function requestUrls(args, maxArticles) {
|
|
1914
|
+
const requested = [...args.url === void 0 ? [] : [args.url], ...args.urls ?? []];
|
|
1915
|
+
if (requested.length === 0) throw new Error("invalid wechat: give url or urls with at least one link");
|
|
1916
|
+
const urls = [];
|
|
1917
|
+
for (const raw of requested) {
|
|
1918
|
+
const url = normalizeUrl(raw);
|
|
1919
|
+
if (!urls.includes(url)) urls.push(url);
|
|
1920
|
+
}
|
|
1921
|
+
if (urls.length > maxArticles) throw new Error("invalid wechat: one call starts from at most " + maxArticles + " links");
|
|
1922
|
+
return urls;
|
|
1923
|
+
}
|
|
1924
|
+
/**
|
|
1925
|
+
* Resolve the deployment's output directory.
|
|
1926
|
+
*
|
|
1927
|
+
* A relative value resolves against the workspace the calling session runs in.
|
|
1928
|
+
* Under a confining filesystem that workspace is the per-call sandbox policy's
|
|
1929
|
+
* root, not merely the session header's cwd: the policy is what the backend
|
|
1930
|
+
* fences the write by, so a relative outputDir resolved against anything else
|
|
1931
|
+
* would land outside the fence and be refused.
|
|
1932
|
+
* @param configured - the row's outputDir value.
|
|
1933
|
+
* @param cwd - the workspace this call resolves against, absent for a call without a session.
|
|
1934
|
+
* @returns the absolute directory article files are written to.
|
|
1935
|
+
*/
|
|
1936
|
+
function resolveOutputDir(configured, cwd) {
|
|
1937
|
+
return isAbsolute(configured) ? configured : resolve(cwd ?? process.cwd(), configured);
|
|
1938
|
+
}
|
|
1939
|
+
/**
|
|
1940
|
+
* Cut text at the deployment ceiling, reporting whether anything was dropped.
|
|
1941
|
+
* @param text - the whole converted body.
|
|
1942
|
+
* @param maxChars - the deployment's excerpt ceiling.
|
|
1943
|
+
* @returns the head within the ceiling and whether anything was dropped.
|
|
1944
|
+
*/
|
|
1945
|
+
function clip(text, maxChars) {
|
|
1946
|
+
if (text.length <= maxChars) return {
|
|
1947
|
+
text,
|
|
1948
|
+
truncated: false
|
|
1949
|
+
};
|
|
1950
|
+
return {
|
|
1951
|
+
text: text.slice(0, maxChars).replace(/\s+$/, ""),
|
|
1952
|
+
truncated: true
|
|
1953
|
+
};
|
|
1954
|
+
}
|
|
1955
|
+
/**
|
|
1956
|
+
* Retrieve one link and refuse a verification redirect.
|
|
1957
|
+
* @param deps - the instance's fetcher and deployment values.
|
|
1958
|
+
* @param url - the link to retrieve.
|
|
1959
|
+
* @param signal - caller cancellation.
|
|
1960
|
+
* @returns the page's HTML and its final link.
|
|
1961
|
+
* @throws Error when WeChat answered with a verification redirect.
|
|
1962
|
+
*/
|
|
1963
|
+
async function fetchPage(deps, url, signal) {
|
|
1964
|
+
const response = await deps.fetcher.get(url, signal);
|
|
1965
|
+
const resolvedUrl = response.url === "" ? url : response.url;
|
|
1966
|
+
const verification = detectBlockUrl(resolvedUrl);
|
|
1967
|
+
if (verification !== "") {
|
|
1968
|
+
deps.fetcher.coolDown(deps.config.blockCooldownMs);
|
|
1969
|
+
throw blockedError(verification, resolvedUrl);
|
|
1970
|
+
}
|
|
1971
|
+
return {
|
|
1972
|
+
html: response.body,
|
|
1973
|
+
resolvedUrl
|
|
1974
|
+
};
|
|
1975
|
+
}
|
|
1976
|
+
/**
|
|
1977
|
+
* Read one page's HTML as an article, a list, a refusal, or none of those. A
|
|
1978
|
+
* page with an article body is the article; a page with an article title but no
|
|
1979
|
+
* body is an article shell a browser render must fill; a page reporting a
|
|
1980
|
+
* withdrawn article is neither; anything else that links to articles is a list
|
|
1981
|
+
* page.
|
|
1982
|
+
* @param html - the page's HTML.
|
|
1983
|
+
* @param requestedUrl - the link the page was fetched from.
|
|
1984
|
+
* @param forced - whether the call declared the link a list page.
|
|
1985
|
+
* @returns what the page carried.
|
|
1986
|
+
*/
|
|
1987
|
+
function interpret(html, requestedUrl, forced) {
|
|
1988
|
+
const extraction = forced ? void 0 : extractArticle(html);
|
|
1989
|
+
if (extraction?.kind === "article") {
|
|
1990
|
+
const article = extraction.article;
|
|
1991
|
+
if (article.bodyHtml.trim() !== "" || article.title !== "") return {
|
|
1992
|
+
kind: "article",
|
|
1993
|
+
article
|
|
1994
|
+
};
|
|
1995
|
+
}
|
|
1996
|
+
if (extraction?.kind === "unavailable") return {
|
|
1997
|
+
kind: "unavailable",
|
|
1998
|
+
reason: extraction.reason
|
|
1999
|
+
};
|
|
2000
|
+
const list = parseListPage(html, requestedUrl);
|
|
2001
|
+
if (list.items.length > 0) return {
|
|
2002
|
+
kind: "list",
|
|
2003
|
+
list
|
|
2004
|
+
};
|
|
2005
|
+
if (extraction?.kind === "blocked") return {
|
|
2006
|
+
kind: "blocked",
|
|
2007
|
+
marker: extraction.marker
|
|
2008
|
+
};
|
|
2009
|
+
return { kind: "none" };
|
|
2010
|
+
}
|
|
2011
|
+
/**
|
|
2012
|
+
* The promotion rules one instance filters with.
|
|
2013
|
+
* @param config - the deployment's filtering values.
|
|
2014
|
+
* @returns the length ceiling and the compiled extra rules.
|
|
2015
|
+
* @throws Error when an extra rule is not a valid regular expression.
|
|
2016
|
+
*/
|
|
2017
|
+
function cleanPolicy(config) {
|
|
2018
|
+
return {
|
|
2019
|
+
maxPromoChars: config.promoMaxChars,
|
|
2020
|
+
patterns: config.promoPatterns.map((source) => {
|
|
2021
|
+
try {
|
|
2022
|
+
return new RegExp(source, "iu");
|
|
2023
|
+
} catch (error) {
|
|
2024
|
+
throw new Error("invalid wechat: promoPatterns entry " + JSON.stringify(source) + " is not a regular expression: " + failureText(error));
|
|
2025
|
+
}
|
|
2026
|
+
})
|
|
2027
|
+
};
|
|
2028
|
+
}
|
|
2029
|
+
/**
|
|
2030
|
+
* Turn one parsed article into the drafted value, body included.
|
|
2031
|
+
* @param article - the parsed page facts.
|
|
2032
|
+
* @param args - the model-supplied call arguments.
|
|
2033
|
+
* @param deps - the instance's deployment values.
|
|
2034
|
+
* @param url - the link the model supplied.
|
|
2035
|
+
* @param resolvedUrl - the article link after redirects.
|
|
2036
|
+
* @param rendered - whether a browser render supplied the HTML.
|
|
2037
|
+
* @returns the drafted article, before its body is written and dropped.
|
|
2038
|
+
*/
|
|
2039
|
+
function buildArticle(article, args, deps, url, resolvedUrl, rendered) {
|
|
2040
|
+
const bodyHtml = deps.config.filterAds ? filterAds(article.bodyHtml, cleanPolicy(deps.config)) : article.bodyHtml;
|
|
2041
|
+
const body = args.format === "text" ? htmlToText(bodyHtml) : htmlToMarkdown(bodyHtml);
|
|
2042
|
+
const excerpt = clip(body, deps.config.maxChars);
|
|
2043
|
+
return {
|
|
2044
|
+
url,
|
|
2045
|
+
resolvedUrl,
|
|
2046
|
+
title: article.title,
|
|
2047
|
+
account: article.account,
|
|
2048
|
+
author: article.author,
|
|
2049
|
+
publishTime: article.publishTime,
|
|
2050
|
+
digest: article.digest,
|
|
2051
|
+
cover: article.cover,
|
|
2052
|
+
body,
|
|
2053
|
+
chars: body.length,
|
|
2054
|
+
excerpt: excerpt.text,
|
|
2055
|
+
excerptClipped: excerpt.truncated,
|
|
2056
|
+
images: args.includeImages === false ? [] : extractImages(bodyHtml).slice(0, deps.config.maxImages),
|
|
2057
|
+
rendered
|
|
2058
|
+
};
|
|
2059
|
+
}
|
|
2060
|
+
/**
|
|
2061
|
+
* Read one requested link as an article page or as a list of article links. The
|
|
2062
|
+
* short-link interstitial is resolved from its own inline script before any
|
|
2063
|
+
* browser render, and a page that carries neither an article body nor a list is
|
|
2064
|
+
* rendered once before the call fails. Only a page that is still an article
|
|
2065
|
+
* with a usable body after filtering leaves this function; every other outcome
|
|
2066
|
+
* is an error, so no caller can turn one into a file.
|
|
2067
|
+
* @param url - the link to read.
|
|
2068
|
+
* @param args - the model-supplied call arguments.
|
|
2069
|
+
* @param deps - the instance's fetcher, optional renderer, and bounds.
|
|
2070
|
+
* @param signal - caller cancellation.
|
|
2071
|
+
* @returns the article this link is, or the list page it is.
|
|
2072
|
+
* @throws Error for a block page, a withdrawn article, a failed request, a body
|
|
2073
|
+
* the filtering rule left too short to be an article, or a page carrying neither.
|
|
2074
|
+
*/
|
|
2075
|
+
async function loadPage(url, args, deps, signal) {
|
|
2076
|
+
const forced = args.list === true;
|
|
2077
|
+
let page = await fetchPage(deps, url, signal);
|
|
2078
|
+
let reading = interpret(page.html, page.resolvedUrl, forced);
|
|
2079
|
+
if (reading.kind === "none") {
|
|
2080
|
+
const redirect = findRedirectUrl(page.html);
|
|
2081
|
+
if (redirect !== "" && redirect !== page.resolvedUrl) {
|
|
2082
|
+
page = await fetchPage(deps, redirect, signal);
|
|
2083
|
+
reading = interpret(page.html, page.resolvedUrl, forced);
|
|
2084
|
+
}
|
|
2085
|
+
}
|
|
2086
|
+
const articleMissing = reading.kind === "none" || reading.kind === "article" && reading.article.bodyHtml.trim() === "";
|
|
2087
|
+
let rendered = false;
|
|
2088
|
+
if (deps.renderer !== void 0 && (deps.config.renderer === "always" || articleMissing || reading.kind === "blocked")) {
|
|
2089
|
+
const resolvedUrl = page.resolvedUrl;
|
|
2090
|
+
page = {
|
|
2091
|
+
html: await deps.renderer.render(resolvedUrl, signal),
|
|
2092
|
+
resolvedUrl
|
|
2093
|
+
};
|
|
2094
|
+
reading = interpret(page.html, resolvedUrl, forced);
|
|
2095
|
+
rendered = true;
|
|
2096
|
+
}
|
|
2097
|
+
if (reading.kind === "list") return {
|
|
2098
|
+
kind: "list",
|
|
2099
|
+
list: reading.list
|
|
2100
|
+
};
|
|
2101
|
+
if (reading.kind === "blocked") {
|
|
2102
|
+
deps.fetcher.coolDown(deps.config.blockCooldownMs);
|
|
2103
|
+
throw blockedError(reading.marker, page.resolvedUrl);
|
|
2104
|
+
}
|
|
2105
|
+
if (reading.kind === "unavailable") throw unavailableError(reading.reason, page.resolvedUrl);
|
|
2106
|
+
if (reading.kind === "article" && reading.article.bodyHtml.trim() !== "") {
|
|
2107
|
+
const article = buildArticle(reading.article, args, deps, url, page.resolvedUrl, rendered);
|
|
2108
|
+
if (article.body.trim().length < deps.config.minBodyChars) throw new Error("wechat: " + page.resolvedUrl + " left no usable article body after filtering (" + String(article.body.trim().length) + " of " + String(deps.config.minBodyChars) + " characters); lower minBodyChars or set filterAds false to keep it");
|
|
2109
|
+
return {
|
|
2110
|
+
kind: "article",
|
|
2111
|
+
article
|
|
2112
|
+
};
|
|
2113
|
+
}
|
|
2114
|
+
if (forced) throw new Error("wechat: " + page.resolvedUrl + " listed no article links");
|
|
2115
|
+
throw new Error("wechat: " + page.resolvedUrl + " served no article body; check the link or enable the browser renderer");
|
|
2116
|
+
}
|
|
2117
|
+
/**
|
|
2118
|
+
* Walk one list page and, while it addresses a next page, the list pages after
|
|
2119
|
+
* it. Every page read here is paced by the same fetcher as an article, so a walk
|
|
2120
|
+
* costs the same interval as any other request.
|
|
2121
|
+
* @param first - the list page the seed produced.
|
|
2122
|
+
* @param seed - the link that page was fetched from, which addresses its continuation.
|
|
2123
|
+
* @param deps - the instance's fetcher and bounds.
|
|
2124
|
+
* @param signal - caller cancellation.
|
|
2125
|
+
* @returns the article links the walk produced and whether it stopped early.
|
|
2126
|
+
* @throws Error when a page in the walk answers with a verification redirect.
|
|
2127
|
+
*/
|
|
2128
|
+
async function walkList(first, seed, deps, signal) {
|
|
2129
|
+
const links = [];
|
|
2130
|
+
const visited = /* @__PURE__ */ new Set([seed]);
|
|
2131
|
+
const add = (items) => {
|
|
2132
|
+
for (const item of items) if (!links.includes(item.url)) links.push(item.url);
|
|
2133
|
+
};
|
|
2134
|
+
const continuation = (page, pageUrl) => {
|
|
2135
|
+
if (page.next !== void 0) {
|
|
2136
|
+
const album = albumPageUrl(pageUrl, page.next);
|
|
2137
|
+
if (album !== "") return album;
|
|
2138
|
+
}
|
|
2139
|
+
return deps.config.maxListPages > 0 ? page.nextUrl : "";
|
|
2140
|
+
};
|
|
2141
|
+
let page = first;
|
|
2142
|
+
let pageUrl = seed;
|
|
2143
|
+
let pages = 1;
|
|
2144
|
+
for (;;) {
|
|
2145
|
+
add(page.items);
|
|
2146
|
+
const nextUrl = continuation(page, pageUrl);
|
|
2147
|
+
if (nextUrl === "" || visited.has(nextUrl)) return {
|
|
2148
|
+
links,
|
|
2149
|
+
truncated: false
|
|
2150
|
+
};
|
|
2151
|
+
if (deps.config.maxListPages > 0 && pages >= deps.config.maxListPages || links.length >= deps.config.maxArticles) return {
|
|
2152
|
+
links,
|
|
2153
|
+
truncated: true
|
|
2154
|
+
};
|
|
2155
|
+
const response = await deps.fetcher.get(nextUrl, signal);
|
|
2156
|
+
const resolvedUrl = response.url === "" ? nextUrl : response.url;
|
|
2157
|
+
const verification = detectBlockUrl(resolvedUrl);
|
|
2158
|
+
if (verification !== "") {
|
|
2159
|
+
deps.fetcher.coolDown(deps.config.blockCooldownMs);
|
|
2160
|
+
throw blockedError(verification, resolvedUrl);
|
|
2161
|
+
}
|
|
2162
|
+
visited.add(nextUrl);
|
|
2163
|
+
visited.add(resolvedUrl);
|
|
2164
|
+
page = parseListPage(response.body, resolvedUrl);
|
|
2165
|
+
if (page.items.length === 0) return {
|
|
2166
|
+
links,
|
|
2167
|
+
truncated: false
|
|
2168
|
+
};
|
|
2169
|
+
pageUrl = resolvedUrl;
|
|
2170
|
+
pages += 1;
|
|
2171
|
+
}
|
|
2172
|
+
}
|
|
2173
|
+
/**
|
|
2174
|
+
* Retrieve and parse one article.
|
|
2175
|
+
* @param args - the model-supplied call arguments.
|
|
2176
|
+
* @param deps - the instance's fetcher, optional renderer, and bounds.
|
|
2177
|
+
* @param signal - caller cancellation.
|
|
2178
|
+
* @returns the drafted article, before it is saved.
|
|
2179
|
+
* @throws Error for an invalid link, a block page, a failed request, or a link that is a list page.
|
|
2180
|
+
*/
|
|
2181
|
+
async function collectArticle(args, deps, signal) {
|
|
2182
|
+
const url = normalizeUrl(args.url ?? "");
|
|
2183
|
+
const outcome = await loadPage(url, args, deps, signal);
|
|
2184
|
+
if (outcome.kind !== "article") throw new Error("wechat: " + url + " is a list page; scrape it with list set, or scrape the articles it links to");
|
|
2185
|
+
return outcome.article;
|
|
2186
|
+
}
|
|
2187
|
+
/**
|
|
2188
|
+
* Scrape every link one call named, saving each article to its own file. A link
|
|
2189
|
+
* may be an article or a list page; a list page's article links are scraped the
|
|
2190
|
+
* same way as a named article. One failing link or article does not discard
|
|
2191
|
+
* what the other links produced.
|
|
2192
|
+
* @param args - the model-supplied call arguments.
|
|
2193
|
+
* @param deps - the instance's fetcher, renderer, bounds, and output directory.
|
|
2194
|
+
* @param signal - caller cancellation.
|
|
2195
|
+
* @returns the saved articles, the list pages walked, and the links that failed.
|
|
2196
|
+
* @throws Error when the call names no usable link, or when the caller cancels.
|
|
2197
|
+
*/
|
|
2198
|
+
async function scrape(args, deps, signal) {
|
|
2199
|
+
const seeds = requestUrls(args, deps.config.maxArticles);
|
|
2200
|
+
const articles = [];
|
|
2201
|
+
const lists = [];
|
|
2202
|
+
const failures = [];
|
|
2203
|
+
const scraped = /* @__PURE__ */ new Set();
|
|
2204
|
+
const remaining = () => deps.config.maxArticles - articles.length;
|
|
2205
|
+
const save = async (drafted) => {
|
|
2206
|
+
const saved = await saveArticle(drafted, {
|
|
2207
|
+
dir: deps.outputDir,
|
|
2208
|
+
fileNameMaxChars: deps.config.fileNameMaxChars,
|
|
2209
|
+
sandboxPolicy: deps.sandboxPolicy
|
|
2210
|
+
}, deps.fs, signal);
|
|
2211
|
+
articles.push({
|
|
2212
|
+
...saved.article,
|
|
2213
|
+
file: saved.file
|
|
2214
|
+
});
|
|
2215
|
+
};
|
|
2216
|
+
for (const seed of seeds) {
|
|
2217
|
+
if (remaining() <= 0) break;
|
|
2218
|
+
let outcome;
|
|
2219
|
+
try {
|
|
2220
|
+
outcome = await loadPage(seed, args, deps, signal);
|
|
2221
|
+
} catch (error) {
|
|
2222
|
+
if (signal?.aborted === true) throw error;
|
|
2223
|
+
failures.push({
|
|
2224
|
+
url: seed,
|
|
2225
|
+
error: failureText(error)
|
|
2226
|
+
});
|
|
2227
|
+
continue;
|
|
2228
|
+
}
|
|
2229
|
+
if (outcome.kind === "article") {
|
|
2230
|
+
scraped.add(outcome.article.resolvedUrl);
|
|
2231
|
+
try {
|
|
2232
|
+
await save(outcome.article);
|
|
2233
|
+
} catch (error) {
|
|
2234
|
+
failures.push({
|
|
2235
|
+
url: seed,
|
|
2236
|
+
error: failureText(error)
|
|
2237
|
+
});
|
|
2238
|
+
}
|
|
2239
|
+
continue;
|
|
2240
|
+
}
|
|
2241
|
+
let links = [];
|
|
2242
|
+
let truncated = false;
|
|
2243
|
+
try {
|
|
2244
|
+
const walked = await walkList(outcome.list, seed, deps, signal);
|
|
2245
|
+
links = walked.links;
|
|
2246
|
+
truncated = walked.truncated;
|
|
2247
|
+
} catch (error) {
|
|
2248
|
+
if (signal?.aborted === true) throw error;
|
|
2249
|
+
failures.push({
|
|
2250
|
+
url: seed,
|
|
2251
|
+
error: failureText(error)
|
|
2252
|
+
});
|
|
2253
|
+
continue;
|
|
2254
|
+
}
|
|
2255
|
+
lists.push({
|
|
2256
|
+
url: seed,
|
|
2257
|
+
title: outcome.list.title,
|
|
2258
|
+
links,
|
|
2259
|
+
truncated
|
|
2260
|
+
});
|
|
2261
|
+
for (const link of links) {
|
|
2262
|
+
if (remaining() <= 0) break;
|
|
2263
|
+
if (scraped.has(link)) continue;
|
|
2264
|
+
scraped.add(link);
|
|
2265
|
+
try {
|
|
2266
|
+
await save(await collectArticle({
|
|
2267
|
+
...args,
|
|
2268
|
+
url: link,
|
|
2269
|
+
list: false
|
|
2270
|
+
}, deps, signal));
|
|
2271
|
+
} catch (error) {
|
|
2272
|
+
if (signal?.aborted === true) throw error;
|
|
2273
|
+
failures.push({
|
|
2274
|
+
url: link,
|
|
2275
|
+
error: failureText(error)
|
|
2276
|
+
});
|
|
2277
|
+
}
|
|
2278
|
+
}
|
|
2279
|
+
}
|
|
2280
|
+
return {
|
|
2281
|
+
articles,
|
|
2282
|
+
lists,
|
|
2283
|
+
failures
|
|
2284
|
+
};
|
|
2285
|
+
}
|
|
2286
|
+
/**
|
|
2287
|
+
* Compose the model-facing text of one call's result.
|
|
2288
|
+
* @param value - the canonical scrape value.
|
|
2289
|
+
* @returns every list page walked, every saved article, then every failure.
|
|
2290
|
+
*/
|
|
2291
|
+
function renderScrape(value) {
|
|
2292
|
+
const blocks = [];
|
|
2293
|
+
for (const list of value.lists) blocks.push([
|
|
2294
|
+
"列表页:" + (list.title === "" ? list.url : list.title),
|
|
2295
|
+
"链接:" + list.url,
|
|
2296
|
+
"文章数:" + list.links.length + (list.truncated ? "(合集还有更多,已按本次上限截断)" : ""),
|
|
2297
|
+
...list.links.map((link, index) => String(index + 1) + ". " + link)
|
|
2298
|
+
].join("\n"));
|
|
2299
|
+
for (const article of value.articles) blocks.push(renderArticle(article));
|
|
2300
|
+
for (const failure of value.failures) blocks.push("抓取失败:" + failure.url + "\n" + failure.error);
|
|
2301
|
+
return blocks.join("\n\n---\n\n");
|
|
2302
|
+
}
|
|
2303
|
+
/**
|
|
2304
|
+
* Contribute the routing section, own the renderer's lifetime, and register the
|
|
2305
|
+
* wechat_scrape tool. The section is empty wherever the tool is not visible in
|
|
2306
|
+
* that scope, so a restricted composition is not told to call a hidden tool.
|
|
2307
|
+
* @param ctx - registrant context carrying the tool, system-prompt, and filesystem registries.
|
|
2308
|
+
* @param config - the deployment's retrieval, list, filtering, output, and renderer values.
|
|
2309
|
+
* @throws Error when a confining filesystem is mounted with no sandbox-policy service,
|
|
2310
|
+
* because no per-call workspace root could then be resolved for the article writes.
|
|
2311
|
+
*/
|
|
2312
|
+
function apply(ctx, config) {
|
|
2313
|
+
cleanPolicy(config);
|
|
2314
|
+
const sandboxPolicy = ctx.get("sandboxPolicy");
|
|
2315
|
+
if (ctx.fs.sandboxMode !== void 0 && sandboxPolicy === void 0) throw new Error("dsh-ab-wechat-scrape: the mounted filesystem confines but ctx.sandboxPolicy is missing");
|
|
2316
|
+
const fetcher = createFetcher({
|
|
2317
|
+
userAgent: config.userAgent,
|
|
2318
|
+
referer: config.referer,
|
|
2319
|
+
cookie: config.cookie,
|
|
2320
|
+
cookieJar: config.cookieJar,
|
|
2321
|
+
minIntervalMs: config.minIntervalMs,
|
|
2322
|
+
minIntervalJitterMs: config.minIntervalJitterMs,
|
|
2323
|
+
timeoutMs: config.timeoutMs,
|
|
2324
|
+
maxRetries: config.maxRetries,
|
|
2325
|
+
backoffBaseMs: config.backoffBaseMs,
|
|
2326
|
+
maxRetryAfterMs: config.maxRetryAfterMs
|
|
2327
|
+
});
|
|
2328
|
+
const renderer = config.renderer === "never" ? void 0 : createRenderer({
|
|
2329
|
+
driver: config.rendererDriver,
|
|
2330
|
+
url: config.rendererUrl,
|
|
2331
|
+
channel: config.browserChannel,
|
|
2332
|
+
executablePath: config.browserExecutablePath,
|
|
2333
|
+
headless: config.headless,
|
|
2334
|
+
idleMs: config.browserIdleMs,
|
|
2335
|
+
settleMs: config.renderSettleMs,
|
|
2336
|
+
readySelector: config.renderReadySelector,
|
|
2337
|
+
waitMs: config.renderWaitMs,
|
|
2338
|
+
timeoutMs: config.timeoutMs,
|
|
2339
|
+
userAgent: config.userAgent,
|
|
2340
|
+
referer: config.referer,
|
|
2341
|
+
cookie: config.cookie,
|
|
2342
|
+
pace: fetcher.pace
|
|
2343
|
+
});
|
|
2344
|
+
if (renderer !== void 0) ctx.effect(() => () => {
|
|
2345
|
+
renderer.dispose();
|
|
2346
|
+
}, "wechat: browser renderer");
|
|
2347
|
+
ctx.systemPrompt.section({
|
|
2348
|
+
name: SECTION_NAME,
|
|
2349
|
+
order: SECTION_ORDER,
|
|
2350
|
+
text: ({ scope }) => ctx.tools.get("wechat_scrape", scope) === void 0 ? "" : SECTION_TEXT
|
|
2351
|
+
});
|
|
2352
|
+
ctx.tools.register(defineTool({
|
|
2353
|
+
name: TOOL_NAME,
|
|
2354
|
+
description: DESCRIPTION,
|
|
2355
|
+
parameters: {
|
|
2356
|
+
url: {
|
|
2357
|
+
type: "string",
|
|
2358
|
+
description: "One mp.weixin.qq.com link: an article (/s/<id> or /s?...), or a list or collection page."
|
|
2359
|
+
},
|
|
2360
|
+
urls: {
|
|
2361
|
+
type: "array",
|
|
2362
|
+
items: { type: "string" },
|
|
2363
|
+
description: "Several links to scrape in one call; each article found is saved as its own file."
|
|
2364
|
+
},
|
|
2365
|
+
list: {
|
|
2366
|
+
type: "boolean",
|
|
2367
|
+
description: "Whether every link is a list or collection page whose articles are scraped instead of the page itself (default false). A page that carries no article body is read as a list page either way."
|
|
2368
|
+
},
|
|
2369
|
+
format: {
|
|
2370
|
+
type: "string",
|
|
2371
|
+
enum: [...BODY_FORMATS],
|
|
2372
|
+
description: "Body format: markdown (default) keeps headings, lists, and links; text is plain prose."
|
|
2373
|
+
},
|
|
2374
|
+
includeImages: {
|
|
2375
|
+
type: "boolean",
|
|
2376
|
+
description: "Whether the result lists the article image URLs (default true)."
|
|
2377
|
+
}
|
|
2378
|
+
},
|
|
2379
|
+
output: {
|
|
2380
|
+
schema: {
|
|
2381
|
+
type: "object",
|
|
2382
|
+
additionalProperties: false,
|
|
2383
|
+
properties: {
|
|
2384
|
+
articles: {
|
|
2385
|
+
type: "array",
|
|
2386
|
+
required: true,
|
|
2387
|
+
items: {
|
|
2388
|
+
type: "object",
|
|
2389
|
+
additionalProperties: false,
|
|
2390
|
+
properties: {
|
|
2391
|
+
url: {
|
|
2392
|
+
type: "string",
|
|
2393
|
+
required: true
|
|
2394
|
+
},
|
|
2395
|
+
resolvedUrl: {
|
|
2396
|
+
type: "string",
|
|
2397
|
+
required: true
|
|
2398
|
+
},
|
|
2399
|
+
title: {
|
|
2400
|
+
type: "string",
|
|
2401
|
+
required: true
|
|
2402
|
+
},
|
|
2403
|
+
account: {
|
|
2404
|
+
type: "string",
|
|
2405
|
+
required: true
|
|
2406
|
+
},
|
|
2407
|
+
author: {
|
|
2408
|
+
type: "string",
|
|
2409
|
+
required: true
|
|
2410
|
+
},
|
|
2411
|
+
publishTime: {
|
|
2412
|
+
type: "string",
|
|
2413
|
+
required: true
|
|
2414
|
+
},
|
|
2415
|
+
digest: {
|
|
2416
|
+
type: "string",
|
|
2417
|
+
required: true
|
|
2418
|
+
},
|
|
2419
|
+
cover: {
|
|
2420
|
+
type: "string",
|
|
2421
|
+
required: true
|
|
2422
|
+
},
|
|
2423
|
+
chars: {
|
|
2424
|
+
type: "number",
|
|
2425
|
+
required: true
|
|
2426
|
+
},
|
|
2427
|
+
excerpt: {
|
|
2428
|
+
type: "string",
|
|
2429
|
+
required: true
|
|
2430
|
+
},
|
|
2431
|
+
excerptClipped: {
|
|
2432
|
+
type: "boolean",
|
|
2433
|
+
required: true
|
|
2434
|
+
},
|
|
2435
|
+
images: {
|
|
2436
|
+
type: "array",
|
|
2437
|
+
required: true,
|
|
2438
|
+
items: { type: "string" }
|
|
2439
|
+
},
|
|
2440
|
+
rendered: {
|
|
2441
|
+
type: "boolean",
|
|
2442
|
+
required: true
|
|
2443
|
+
},
|
|
2444
|
+
file: {
|
|
2445
|
+
type: "string",
|
|
2446
|
+
required: true
|
|
2447
|
+
}
|
|
2448
|
+
}
|
|
2449
|
+
}
|
|
2450
|
+
},
|
|
2451
|
+
lists: {
|
|
2452
|
+
type: "array",
|
|
2453
|
+
required: true,
|
|
2454
|
+
items: {
|
|
2455
|
+
type: "object",
|
|
2456
|
+
additionalProperties: false,
|
|
2457
|
+
properties: {
|
|
2458
|
+
url: {
|
|
2459
|
+
type: "string",
|
|
2460
|
+
required: true
|
|
2461
|
+
},
|
|
2462
|
+
title: {
|
|
2463
|
+
type: "string",
|
|
2464
|
+
required: true
|
|
2465
|
+
},
|
|
2466
|
+
links: {
|
|
2467
|
+
type: "array",
|
|
2468
|
+
required: true,
|
|
2469
|
+
items: { type: "string" }
|
|
2470
|
+
},
|
|
2471
|
+
truncated: {
|
|
2472
|
+
type: "boolean",
|
|
2473
|
+
required: true
|
|
2474
|
+
}
|
|
2475
|
+
}
|
|
2476
|
+
}
|
|
2477
|
+
},
|
|
2478
|
+
failures: {
|
|
2479
|
+
type: "array",
|
|
2480
|
+
required: true,
|
|
2481
|
+
items: {
|
|
2482
|
+
type: "object",
|
|
2483
|
+
additionalProperties: false,
|
|
2484
|
+
properties: {
|
|
2485
|
+
url: {
|
|
2486
|
+
type: "string",
|
|
2487
|
+
required: true
|
|
2488
|
+
},
|
|
2489
|
+
error: {
|
|
2490
|
+
type: "string",
|
|
2491
|
+
required: true
|
|
2492
|
+
}
|
|
2493
|
+
}
|
|
2494
|
+
}
|
|
2495
|
+
}
|
|
2496
|
+
}
|
|
2497
|
+
},
|
|
2498
|
+
render: (_args, value) => [{
|
|
2499
|
+
type: "text",
|
|
2500
|
+
text: renderScrape(value)
|
|
2501
|
+
}]
|
|
2502
|
+
},
|
|
2503
|
+
execute(args, exec) {
|
|
2504
|
+
const policy = sandboxPolicy?.resolve(exec.agent === void 0 ? {} : { session: exec.agent.session });
|
|
2505
|
+
return scrape(args, {
|
|
2506
|
+
fetcher,
|
|
2507
|
+
renderer,
|
|
2508
|
+
config,
|
|
2509
|
+
fs: ctx.fs,
|
|
2510
|
+
outputDir: resolveOutputDir(config.outputDir, policy?.workspaceRoot ?? exec.agent?.session.header.cwd),
|
|
2511
|
+
sandboxPolicy: policy
|
|
2512
|
+
}, exec.signal);
|
|
2513
|
+
},
|
|
2514
|
+
presentCall: (args) => ({
|
|
2515
|
+
card: "generic",
|
|
2516
|
+
title: args.list === true ? "Scrape WeChat collection" : "Scrape WeChat articles",
|
|
2517
|
+
kind: "fetch",
|
|
2518
|
+
rawInput: args.url ?? (args.urls ?? []).join(", ")
|
|
2519
|
+
})
|
|
2520
|
+
}));
|
|
2521
|
+
}
|
|
2522
|
+
//#endregion
|
|
2523
|
+
export { BODY_FORMATS, Config, SECTION_NAME, SECTION_ORDER, SECTION_TEXT, TOOL_NAME, albumPageUrl, apply, articleDocument, cleanPolicy, clip, collectArticle, inject, name, normalizeUrl, parseListPage, recordedArticle, renderArticle, renderScrape, requestUrls, resolveOutputDir, scrape };
|