dsh-ab-wechat-scrape 0.4.6 → 0.4.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -275,6 +275,20 @@ Arguments and result are data-dependent and resent until compaction; each excerp
275
275
 
276
276
  None beyond the ordinary request prefix.
277
277
 
278
+ ### Auto-invocation
279
+
280
+ #### What the model sees
281
+
282
+ When a reader's inbound message names one or more `mp.weixin.qq.com` links, the plugin's `agent/pre-step` listener detects them (through the pure `detectWechatLinks` over the free text) and, before the model's step, folds in a priming instruction that names the links and asks for `wechat_scrape`. The tool then fires on the model's next action rather than only after the model first notices the link. The detection is self-aware: it acts only when `wechat_scrape` is registered in the assembling scope, and only on the inbound user message — never on a tool result or on its own prime — so it adds no work when no link is present and never double-fires.
283
+
284
+ #### Token effect
285
+
286
+ Negligible: the listener scans only the last user message for links each step; when none are found it returns the step unchanged. When links are found, the prime adds one short instruction whose text is the link list.
287
+
288
+ #### KV Cache effect
289
+
290
+ None beyond the ordinary request prefix; the prime is appended after the cached prefix.
291
+
278
292
  -----
279
293
 
280
294
  <a id="known-limitations-and-deferred-work"></a>
@@ -291,6 +305,7 @@ None beyond the ordinary request prefix.
291
305
  - The minimum-body rule is a character count, not a judgement: a genuine article shorter than minBodyChars is reported instead of saved, and a long page of boilerplate that survived filtering is saved.
292
306
  - A page that carries an article title but no body is treated as an article shell and rendered, not as a list page. A list page that carries such a title needs `list: true`.
293
307
  - A withdrawal marker is read from the page head only, and only on a page with no body element; an article whose body was emptied by filtering is caught by the minimum-body rule rather than by that marker.
308
+ - Auto-invocation recognizes only links the host check accepts (`mp.weixin.qq.com`, article or collection pages). A link the detector cannot isolate from surrounding prose — for example one split across a line break or embedded in an image — is not detected; the layer-1 routing section still covers those cases by asking the model to call the tool.
294
309
  - A block page is reported, not circumvented: when the browser render is refused as well, an expired or absent cookie still fails the call, and the remedy is a deployment cookie plus a lower request rate.
295
310
  - The cookie jar lives for one plugin instance and is never persisted, so each process starts as a fresh visitor; the deployment's configured cookie is what carries an authenticated session across restarts.
296
311
  - The playwright driver depends on a browser the deployment already has (channel msedge by default, or an explicit executable path); it does not download one.
package/lib/html.d.ts CHANGED
@@ -74,6 +74,20 @@ export declare function isMpArticleUrl(url: string): boolean;
74
74
  * @returns true for an http(s) mp.weixin.qq.com /s link.
75
75
  */
76
76
  export declare function isMpArticleLink(url: string): boolean;
77
+ /**
78
+ * Find every mp.weixin.qq.com link named in free text, in first-seen order and
79
+ * without duplicates. This is the tool's auto-invocation trigger: a reader
80
+ * pastes or names a WeChat link and the plugin should retrieve it without the
81
+ * model having to notice the link first. The scan stops at any character that is
82
+ * not part of a URL — crucially including CJK characters, because Chinese text
83
+ * has no word spacing and would otherwise swallow the words after a link into
84
+ * the URL. Trailing punctuation that is still not a URL character (ASCII or
85
+ * full-width) is then trimmed before the host check, and only links the host
86
+ * check accepts reach the result.
87
+ * @param text - the inbound message text to scan.
88
+ * @returns the distinct mp.weixin.qq.com links found, in first-seen order.
89
+ */
90
+ export declare function detectWechatLinks(text: string): string[];
77
91
  /**
78
92
  * The block-page marker a page carries.
79
93
  * @param html - complete page HTML.
package/lib/html.js CHANGED
@@ -271,6 +271,28 @@ export function isMpArticleLink(url) {
271
271
  const path = new URL(url).pathname;
272
272
  return path === '/s' || path.startsWith('/s/');
273
273
  }
274
+ /**
275
+ * Find every mp.weixin.qq.com link named in free text, in first-seen order and
276
+ * without duplicates. This is the tool's auto-invocation trigger: a reader
277
+ * pastes or names a WeChat link and the plugin should retrieve it without the
278
+ * model having to notice the link first. The scan stops at any character that is
279
+ * not part of a URL — crucially including CJK characters, because Chinese text
280
+ * has no word spacing and would otherwise swallow the words after a link into
281
+ * the URL. Trailing punctuation that is still not a URL character (ASCII or
282
+ * full-width) is then trimmed before the host check, and only links the host
283
+ * check accepts reach the result.
284
+ * @param text - the inbound message text to scan.
285
+ * @returns the distinct mp.weixin.qq.com links found, in first-seen order.
286
+ */
287
+ export function detectWechatLinks(text) {
288
+ const found = [];
289
+ for (const token of text.match(/https?:\/\/[A-Za-z0-9\-._~:/?#[\]@!$&*+,;=%]+/gu) ?? []) {
290
+ const url = token.replace(/[.,;:!?')\]。,;:!?、)]+$/u, '');
291
+ if (isMpArticleUrl(url) && !found.includes(url))
292
+ found.push(url);
293
+ }
294
+ return found;
295
+ }
274
296
  /**
275
297
  * The block-page marker a page carries.
276
298
  * @param html - complete page HTML.
package/lib/index.d.ts CHANGED
@@ -17,6 +17,14 @@
17
17
  */
18
18
  import type { Context } from '@deepseek-ai/cordis';
19
19
  import z from '@deepseek-ai/schemastery';
20
+ import type { ContextFormed } from '@deepseek-ai/dsh-llm';
21
+ declare module '@deepseek-ai/dsh-llm' {
22
+ interface MessageSourceMap {
23
+ 'wechat-scrape': {
24
+ kind: 'wechat-scrape';
25
+ } & ContextFormed;
26
+ }
27
+ }
20
28
  import type FileSystem from '@deepseek-ai/dsh-fs';
21
29
  import type { SandboxExecutionPolicy } from '@deepseek-ai/dsh-sandbox';
22
30
  import { type CleanPolicy } from './clean.ts';
package/lib/index.js CHANGED
@@ -1,6 +1,7 @@
1
1
  import { isAbsolute, join, resolve } from "node:path";
2
2
  import z from "@deepseek-ai/schemastery";
3
3
  import { defineTool } from "@deepseek-ai/dsh-tools";
4
+ import { createUserMessage } from "@deepseek-ai/dsh-llm";
4
5
  import { FsError } from "@deepseek-ai/dsh-fs";
5
6
  import { sandboxDenialMarker } from "@deepseek-ai/dsh-sandbox";
6
7
  //#region lib/html.js
@@ -250,6 +251,27 @@ function isMpArticleLink(url) {
250
251
  return path === "/s" || path.startsWith("/s/");
251
252
  }
252
253
  /**
254
+ * Find every mp.weixin.qq.com link named in free text, in first-seen order and
255
+ * without duplicates. This is the tool's auto-invocation trigger: a reader
256
+ * pastes or names a WeChat link and the plugin should retrieve it without the
257
+ * model having to notice the link first. The scan stops at any character that is
258
+ * not part of a URL — crucially including CJK characters, because Chinese text
259
+ * has no word spacing and would otherwise swallow the words after a link into
260
+ * the URL. Trailing punctuation that is still not a URL character (ASCII or
261
+ * full-width) is then trimmed before the host check, and only links the host
262
+ * check accepts reach the result.
263
+ * @param text - the inbound message text to scan.
264
+ * @returns the distinct mp.weixin.qq.com links found, in first-seen order.
265
+ */
266
+ function detectWechatLinks(text) {
267
+ const found = [];
268
+ for (const token of text.match(/https?:\/\/[A-Za-z0-9\-._~:/?#[\]@!$&*+,;=%]+/gu) ?? []) {
269
+ const url = token.replace(/[.,;:!?')\]。,;:!?、)]+$/u, "");
270
+ if (isMpArticleUrl(url) && !found.includes(url)) found.push(url);
271
+ }
272
+ return found;
273
+ }
274
+ /**
253
275
  * The block-page marker a page carries.
254
276
  * @param html - complete page HTML.
255
277
  * @returns the marker found in the page head, or an empty string for a normal page.
@@ -2561,6 +2583,32 @@ function apply(ctx, config) {
2561
2583
  rawInput: args.url ?? (args.urls ?? []).join(", ")
2562
2584
  })
2563
2585
  }));
2586
+ ctx.on("agent/pre-step", async ({ signal }, next) => {
2587
+ const decision = await next();
2588
+ if (decision.kind === "reject" || signal.aborted) return decision;
2589
+ const lastUser = [...decision.messages].reverse().find((message) => message.source.kind === "user");
2590
+ if (lastUser === void 0) return decision;
2591
+ const links = detectWechatLinks(lastUser.content.map((block) => block.type === "text" ? block.text : "").filter((part) => part !== "").join("\n"));
2592
+ if (links.length === 0) return decision;
2593
+ if (ctx.tools.get("wechat_scrape") === void 0) return decision;
2594
+ return {
2595
+ ...decision,
2596
+ messages: [createUserMessage({
2597
+ content: [{
2598
+ type: "text",
2599
+ text: "检测到以下 wechat_scrape 触发链接,请调用 wechat_scrape 处理:\n" + links.join("\n")
2600
+ }],
2601
+ source: {
2602
+ kind: name,
2603
+ form: "snapshot",
2604
+ sections: [{
2605
+ name: SECTION_NAME,
2606
+ text: links.join("\n")
2607
+ }]
2608
+ }
2609
+ }), ...decision.messages]
2610
+ };
2611
+ });
2564
2612
  }
2565
2613
  //#endregion
2566
2614
  export { BODY_FORMATS, Config, SECTION_NAME, SECTION_ORDER, SECTION_TEXT, TOOL_NAME, albumPageUrl, apply, articleDocument, cleanPolicy, clip, collectArticle, inject, name, normalizeUrl, parseListPage, recordedArticle, renderArticle, renderScrape, requestUrls, resolveOutputDir, scrape };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-ab-wechat-scrape",
3
- "version": "0.4.6",
3
+ "version": "0.4.8",
4
4
  "type": "module",
5
5
  "main": "lib/index.js",
6
6
  "types": "lib/index.d.ts",
@@ -66,6 +66,7 @@
66
66
  "peerDependencies": {
67
67
  "@deepseek-ai/cordis": "^4.0.4",
68
68
  "@deepseek-ai/dsh-fs": "^0.1.7-rc.1",
69
+ "@deepseek-ai/dsh-llm": "^0.1.7-rc.1",
69
70
  "@deepseek-ai/dsh-sandbox": "^0.1.7-rc.1",
70
71
  "@deepseek-ai/dsh-tools": "^0.1.7-rc.1"
71
72
  },
@@ -81,6 +82,7 @@
81
82
  "@deepseek-ai/dsh-sandbox-policy": "^0.1.7-rc.1",
82
83
  "@deepseek-ai/dsh-system-prompt": "^0.1.7-rc.1",
83
84
  "@deepseek-ai/dsh-tools": "^0.1.7-rc.1",
85
+ "@deepseek-ai/dsh-llm": "^0.1.7-rc.1",
84
86
  "@types/node": "^22.10.2",
85
87
  "prettier": "^3.3.0",
86
88
  "publint": "^0.2.0",