dsh-ab-wechat-scrape 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +330 -0
- package/cordis.patch.yml +8 -0
- package/lib/clean.d.ts +32 -0
- package/lib/clean.d.ts.map +1 -0
- package/lib/clean.js +193 -0
- package/lib/clean.js.map +1 -0
- package/lib/document.d.ts +27 -0
- package/lib/document.d.ts.map +1 -0
- package/lib/document.js +79 -0
- package/lib/document.js.map +1 -0
- package/lib/fetch.d.ts +84 -0
- package/lib/fetch.d.ts.map +1 -0
- package/lib/fetch.js +238 -0
- package/lib/fetch.js.map +1 -0
- package/lib/filename.d.ts +36 -0
- package/lib/filename.d.ts.map +1 -0
- package/lib/filename.js +92 -0
- package/lib/filename.js.map +1 -0
- package/lib/html.d.ts +114 -0
- package/lib/html.d.ts.map +1 -0
- package/lib/html.js +402 -0
- package/lib/html.js.map +1 -0
- package/lib/index.d.ts +219 -0
- package/lib/index.d.ts.map +1 -0
- package/lib/index.js +2523 -0
- package/lib/index.js.map +1 -0
- package/lib/list.d.ts +55 -0
- package/lib/list.d.ts.map +1 -0
- package/lib/list.js +191 -0
- package/lib/list.js.map +1 -0
- package/lib/markdown.d.ts +20 -0
- package/lib/markdown.d.ts.map +1 -0
- package/lib/markdown.js +250 -0
- package/lib/markdown.js.map +1 -0
- package/lib/render.d.ts +103 -0
- package/lib/render.d.ts.map +1 -0
- package/lib/render.js +209 -0
- package/lib/render.js.map +1 -0
- package/lib/store.d.ts +50 -0
- package/lib/store.d.ts.map +1 -0
- package/lib/store.js +173 -0
- package/lib/store.js.map +1 -0
- package/lib/types.d.ts +133 -0
- package/lib/types.d.ts.map +1 -0
- package/lib/types.js +8 -0
- package/lib/types.js.map +1 -0
- package/package.json +91 -0
- package/tsconfig.json +30 -0
- package/tsdown.config.ts +18 -0
package/lib/types.d.ts
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Vocabulary shared by the wechat_scrape tool and its tests: the canonical
|
|
3
|
+
* value one call records, the call's own arguments, and the parsed page facts
|
|
4
|
+
* the extractor produces.
|
|
5
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/types
|
|
6
|
+
*/
|
|
7
|
+
/** Body formats one call may ask for. */
|
|
8
|
+
export type WechatFormat = 'markdown' | 'text';
|
|
9
|
+
/** Arguments the wechat_scrape parameter schema admits. */
|
|
10
|
+
export interface WechatArgs {
|
|
11
|
+
/** One official-account link: an article, or a list page when list is set. */
|
|
12
|
+
url?: string;
|
|
13
|
+
/** Several links, scraped in order and saved as separate files. */
|
|
14
|
+
urls?: string[];
|
|
15
|
+
/** Whether each link is a list or collection page whose articles are scraped instead of the page itself. */
|
|
16
|
+
list?: boolean;
|
|
17
|
+
/** Body format; markdown keeps headings, lists, links, and images. */
|
|
18
|
+
format?: WechatFormat;
|
|
19
|
+
/** Whether the value carries the article's image URLs. */
|
|
20
|
+
includeImages?: boolean;
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* One article after parsing, promotion filtering, and the deployment ceilings.
|
|
24
|
+
*
|
|
25
|
+
* This is what the call records and the model reads: the facts about the article,
|
|
26
|
+
* the length of its body, and a bounded head of it. The body itself is not here —
|
|
27
|
+
* it is in the file this article was saved to, which {@link WechatArticle.file}
|
|
28
|
+
* names.
|
|
29
|
+
*/
|
|
30
|
+
export interface CollectedArticle {
|
|
31
|
+
/** The link the model supplied. */
|
|
32
|
+
url: string;
|
|
33
|
+
/** The article link after any interstitial redirect was resolved. */
|
|
34
|
+
resolvedUrl: string;
|
|
35
|
+
/** Article title. */
|
|
36
|
+
title: string;
|
|
37
|
+
/** Official-account display name. */
|
|
38
|
+
account: string;
|
|
39
|
+
/** Author byline, empty when the page carries none. */
|
|
40
|
+
author: string;
|
|
41
|
+
/** Publish time as ISO 8601, or the page's own text when it is not a timestamp. */
|
|
42
|
+
publishTime: string;
|
|
43
|
+
/** Article summary shown under the title. */
|
|
44
|
+
digest: string;
|
|
45
|
+
/** Cover image URL, empty when the page carries none. */
|
|
46
|
+
cover: string;
|
|
47
|
+
/** Characters in the whole body, which is in the saved file. */
|
|
48
|
+
chars: number;
|
|
49
|
+
/** Head of the body, capped by the deployment's excerpt ceiling. */
|
|
50
|
+
excerpt: string;
|
|
51
|
+
/** Whether the excerpt is shorter than the body. */
|
|
52
|
+
excerptClipped: boolean;
|
|
53
|
+
/** Content image URLs of the filtered body, in document order. */
|
|
54
|
+
images: string[];
|
|
55
|
+
/** Whether a browser render supplied the HTML. */
|
|
56
|
+
rendered: boolean;
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* One article as it is built, before its body is written and dropped.
|
|
60
|
+
*
|
|
61
|
+
* The body exists only between parsing and the write. It never reaches the
|
|
62
|
+
* canonical value, the model, or the session log: a call that carried it would pay
|
|
63
|
+
* the article's full length in context and then have it discarded again.
|
|
64
|
+
*/
|
|
65
|
+
export interface DraftedArticle extends CollectedArticle {
|
|
66
|
+
/** The whole body, in the requested format. Written to the file and nowhere else. */
|
|
67
|
+
body: string;
|
|
68
|
+
}
|
|
69
|
+
/** One article the call scraped and saved. */
|
|
70
|
+
export interface WechatArticle extends CollectedArticle {
|
|
71
|
+
/** Absolute path of the Markdown file this call wrote for the article. */
|
|
72
|
+
file: string;
|
|
73
|
+
}
|
|
74
|
+
/** One link the call could not scrape. */
|
|
75
|
+
export interface WechatFailure {
|
|
76
|
+
/** The link that failed. */
|
|
77
|
+
url: string;
|
|
78
|
+
/** Why it failed, as the tool reported it. */
|
|
79
|
+
error: string;
|
|
80
|
+
}
|
|
81
|
+
/** One list or collection page the call walked. */
|
|
82
|
+
export interface WechatList {
|
|
83
|
+
/** The list page the model supplied. */
|
|
84
|
+
url: string;
|
|
85
|
+
/** The page's own title, empty when it carried none. */
|
|
86
|
+
title: string;
|
|
87
|
+
/** Distinct article links the walk produced, in page order. */
|
|
88
|
+
links: string[];
|
|
89
|
+
/** Whether the collection carried more entries than the call's ceilings allowed. */
|
|
90
|
+
truncated: boolean;
|
|
91
|
+
}
|
|
92
|
+
/** Canonical value one wechat_scrape call records. */
|
|
93
|
+
export interface WechatScrape {
|
|
94
|
+
/** Articles that were retrieved and saved, in request order. */
|
|
95
|
+
articles: WechatArticle[];
|
|
96
|
+
/** List pages the call walked, in request order. */
|
|
97
|
+
lists: WechatList[];
|
|
98
|
+
/** Links that were requested but produced no article. */
|
|
99
|
+
failures: WechatFailure[];
|
|
100
|
+
}
|
|
101
|
+
/** One parsed article page, before promotion filtering and the deployment ceilings. */
|
|
102
|
+
export interface ExtractedArticle {
|
|
103
|
+
/** Article title. */
|
|
104
|
+
title: string;
|
|
105
|
+
/** Official-account display name. */
|
|
106
|
+
account: string;
|
|
107
|
+
/** Author byline. */
|
|
108
|
+
author: string;
|
|
109
|
+
/** Publish time, already normalized when the page carries a timestamp. */
|
|
110
|
+
publishTime: string;
|
|
111
|
+
/** Article summary. */
|
|
112
|
+
digest: string;
|
|
113
|
+
/** Cover image URL. */
|
|
114
|
+
cover: string;
|
|
115
|
+
/** Inner HTML of the article body element, promotional blocks included. */
|
|
116
|
+
bodyHtml: string;
|
|
117
|
+
/** Content image URLs in document order. */
|
|
118
|
+
images: string[];
|
|
119
|
+
}
|
|
120
|
+
/** What parsing one page produced. */
|
|
121
|
+
export type Extraction = {
|
|
122
|
+
kind: 'article';
|
|
123
|
+
article: ExtractedArticle;
|
|
124
|
+
} | {
|
|
125
|
+
kind: 'blocked';
|
|
126
|
+
marker: string;
|
|
127
|
+
} | {
|
|
128
|
+
kind: 'unavailable';
|
|
129
|
+
reason: string;
|
|
130
|
+
} | {
|
|
131
|
+
kind: 'empty';
|
|
132
|
+
};
|
|
133
|
+
//# sourceMappingURL=types.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,yCAAyC;AACzC,MAAM,MAAM,YAAY,GAAG,UAAU,GAAG,MAAM,CAAA;AAE9C,2DAA2D;AAC3D,MAAM,WAAW,UAAU;IACzB,8EAA8E;IAC9E,GAAG,CAAC,EAAE,MAAM,CAAA;IACZ,mEAAmE;IACnE,IAAI,CAAC,EAAE,MAAM,EAAE,CAAA;IACf,4GAA4G;IAC5G,IAAI,CAAC,EAAE,OAAO,CAAA;IACd,sEAAsE;IACtE,MAAM,CAAC,EAAE,YAAY,CAAA;IACrB,0DAA0D;IAC1D,aAAa,CAAC,EAAE,OAAO,CAAA;CACxB;AAED;;;;;;;GAOG;AACH,MAAM,WAAW,gBAAgB;IAC/B,mCAAmC;IACnC,GAAG,EAAE,MAAM,CAAA;IACX,qEAAqE;IACrE,WAAW,EAAE,MAAM,CAAA;IACnB,qBAAqB;IACrB,KAAK,EAAE,MAAM,CAAA;IACb,qCAAqC;IACrC,OAAO,EAAE,MAAM,CAAA;IACf,uDAAuD;IACvD,MAAM,EAAE,MAAM,CAAA;IACd,mFAAmF;IACnF,WAAW,EAAE,MAAM,CAAA;IACnB,6CAA6C;IAC7C,MAAM,EAAE,MAAM,CAAA;IACd,yDAAyD;IACzD,KAAK,EAAE,MAAM,CAAA;IACb,gEAAgE;IAChE,KAAK,EAAE,MAAM,CAAA;IACb,oEAAoE;IACpE,OAAO,EAAE,MAAM,CAAA;IACf,oDAAoD;IACpD,cAAc,EAAE,OAAO,CAAA;IACvB,kEAAkE;IAClE,MAAM,EAAE,MAAM,EAAE,CAAA;IAChB,kDAAkD;IAClD,QAAQ,EAAE,OAAO,CAAA;CAClB;AAED;;;;;;GAMG;AACH,MAAM,WAAW,cAAe,SAAQ,gBAAgB;IACtD,qFAAqF;IACrF,IAAI,EAAE,MAAM,CAAA;CACb;AAED,8CAA8C;AAC9C,MAAM,WAAW,aAAc,SAAQ,gBAAgB;IACrD,0EAA0E;IAC1E,IAAI,EAAE,MAAM,CAAA;CACb;AAED,0CAA0C;AAC1C,MAAM,WAAW,aAAa;IAC5B,4BAA4B;IAC5B,GAAG,EAAE,MAAM,CAAA;IACX,8CAA8C;IAC9C,KAAK,EAAE,MAAM,CAAA;CACd;AAED,mDAAmD;AACnD,MAAM,WAAW,UAAU;IACzB,wCAAwC;IACxC,GAAG,EAAE,MAAM,CAAA;IACX,wDAAwD;IACxD,KAAK,EAAE,MAAM,CAAA;IACb,+DAA+D;IAC/D,KAAK,EAAE,MAAM,EAAE,CAAA;IACf,oFAAoF;IACpF,SAAS,EAAE,OAAO,CAAA;CACnB;AAED,sDAAsD;AACtD,MAAM,WAAW,YAAY;IAC3B,gEAAgE;IAChE,QAAQ,EAAE,aAAa,EAAE,CAAA;IACzB,oDAAoD;IACpD,KAAK,EAAE,UAAU,EAAE,CAAA;IACnB,yDAAyD;IACzD,QAAQ,EAAE,aAAa,EAAE,CAAA;CAC1B;AAED,uFAAuF;AACvF,MAAM,WAAW,gBAAgB;IAC/B,qBAAqB;IACrB,KAAK,EAAE,MAAM,CAAA;IACb,qCAAqC;IACrC,OAAO,EAAE,MAAM,CAAA;IACf,qBAAqB;IACrB,MAAM,EAAE,MAAM,CAAA;IACd,0EAA0E;IAC1E,WAAW,EAAE,MAAM,CAAA;IACnB,uBAAuB;IACvB,MAAM,EAAE,MAAM,CAAA;IACd,uBAAuB;IACvB,KAAK,EAAE,MAAM,CAAA;IACb,2EAA2E;IAC3E,QAAQ,EAAE,MAAM,CAAA;IAChB,4CAA4C;IAC5C,MAAM,EAAE,MAAM,EAAE,CAAA;CACjB;AAED,sCAAsC;AACtC,MAAM,MAAM,UAAU,GAClB;IAAE,IAAI,EAAE,SAAS,CAAC;IAAC,OAAO,EAAE,gBAAgB,CAAA;CAAE,GAC9C;IAAE,IAAI,EAAE,SAAS,CAAC;IAAC,MAAM,EAAE,MAAM,CAAA;CAAE,GACnC;IAAE,IAAI,EAAE,aAAa,CAAC;IAAC,MAAM,EAAE,MAAM,CAAA;CAAE,GACvC;IAAE,IAAI,EAAE,OAAO,CAAA;CAAE,CAAA"}
|
package/lib/types.js
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Vocabulary shared by the wechat_scrape tool and its tests: the canonical
|
|
3
|
+
* value one call records, the call's own arguments, and the parsed page facts
|
|
4
|
+
* the extractor produces.
|
|
5
|
+
* @module @deepseek-ai/dsh-ab-wechat-scrape/types
|
|
6
|
+
*/
|
|
7
|
+
export {};
|
|
8
|
+
//# sourceMappingURL=types.js.map
|
package/lib/types.js.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.js","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;;;GAKG"}
|
package/package.json
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "dsh-ab-wechat-scrape",
|
|
3
|
+
"version": "0.4.0",
|
|
4
|
+
"type": "module",
|
|
5
|
+
"main": "lib/index.js",
|
|
6
|
+
"types": "lib/index.d.ts",
|
|
7
|
+
"exports": {
|
|
8
|
+
".": {
|
|
9
|
+
"types": "./lib/index.d.ts",
|
|
10
|
+
"default": "./lib/index.js"
|
|
11
|
+
},
|
|
12
|
+
"./package.json": "./package.json"
|
|
13
|
+
},
|
|
14
|
+
"files": [
|
|
15
|
+
"lib/**/*",
|
|
16
|
+
"cordis.patch.yml",
|
|
17
|
+
"README.md",
|
|
18
|
+
"LICENSE",
|
|
19
|
+
"tsconfig.json",
|
|
20
|
+
"tsdown.config.ts"
|
|
21
|
+
],
|
|
22
|
+
"scripts": {
|
|
23
|
+
"build": "tsc -p tsconfig.json && tsdown",
|
|
24
|
+
"clean": "rimraf lib",
|
|
25
|
+
"typecheck": "tsc -p tsconfig.json --noEmit",
|
|
26
|
+
"test": "node --test",
|
|
27
|
+
"lint": "publint && attw --pack . --profile esm-only",
|
|
28
|
+
"format": "prettier --write \"src/**/*.{ts,mjs}\" \"tests/**/*.mjs\" \"*.{json,yml,md}\"",
|
|
29
|
+
"verify": "npm run typecheck && npm run test && npm run build && npm run lint",
|
|
30
|
+
"prepublishOnly": "npm run verify",
|
|
31
|
+
"prepack": "npm run build",
|
|
32
|
+
"release": "npm publish --access public --no-git-checks"
|
|
33
|
+
},
|
|
34
|
+
"dsh": {
|
|
35
|
+
"bundle": {
|
|
36
|
+
"patch": "./cordis.patch.yml"
|
|
37
|
+
}
|
|
38
|
+
},
|
|
39
|
+
"publishConfig": {
|
|
40
|
+
"access": "public"
|
|
41
|
+
},
|
|
42
|
+
"repository": {
|
|
43
|
+
"type": "git",
|
|
44
|
+
"url": "git+https://github.com/deepseek-ai/deepseek-harness.git",
|
|
45
|
+
"directory": ".dsh/plugins/dsh-ab-wechat-scrape"
|
|
46
|
+
},
|
|
47
|
+
"bugs": {
|
|
48
|
+
"url": "https://github.com/deepseek-ai/deepseek-harness/issues"
|
|
49
|
+
},
|
|
50
|
+
"homepage": "https://github.com/deepseek-ai/deepseek-harness/tree/main/.dsh/plugins/dsh-ab-wechat-scrape",
|
|
51
|
+
"license": "MIT",
|
|
52
|
+
"keywords": [
|
|
53
|
+
"deepseek-harness",
|
|
54
|
+
"dsh-plugin",
|
|
55
|
+
"cordis-plugin",
|
|
56
|
+
"wechat",
|
|
57
|
+
"wechat-scrape",
|
|
58
|
+
"scraper"
|
|
59
|
+
],
|
|
60
|
+
"engines": {
|
|
61
|
+
"node": ">=18"
|
|
62
|
+
},
|
|
63
|
+
"dependencies": {
|
|
64
|
+
"@deepseek-ai/dsh-tools": "^0.1.7-rc.1",
|
|
65
|
+
"@deepseek-ai/schemastery": "^3.18.4"
|
|
66
|
+
},
|
|
67
|
+
"peerDependencies": {
|
|
68
|
+
"@deepseek-ai/cordis": "*"
|
|
69
|
+
},
|
|
70
|
+
"devDependencies": {
|
|
71
|
+
"@deepseek-ai/cordis": "^4.0.4",
|
|
72
|
+
"@deepseek-ai/cordis-plugin-include": "^1.0.9",
|
|
73
|
+
"@deepseek-ai/cordis-plugin-loader": "^1.0.5",
|
|
74
|
+
"@deepseek-ai/dsh-fs": "^0.1.7-rc.1",
|
|
75
|
+
"@deepseek-ai/dsh-fs-local": "^0.1.7-rc.1",
|
|
76
|
+
"@deepseek-ai/dsh-fs-sandbox": "^0.1.7-rc.1",
|
|
77
|
+
"@deepseek-ai/dsh-sandbox": "^0.1.7-rc.1",
|
|
78
|
+
"@deepseek-ai/dsh-sandbox-policy": "^0.1.7-rc.1",
|
|
79
|
+
"@deepseek-ai/dsh-system-prompt": "^0.1.7-rc.1",
|
|
80
|
+
"@types/node": "^22.10.2",
|
|
81
|
+
"tsdown": "^0.22.2",
|
|
82
|
+
"typescript": "^6.0.3",
|
|
83
|
+
"rimraf": "^6.0.1",
|
|
84
|
+
"publint": "^0.2.0",
|
|
85
|
+
"@arethetypeswrong/cli": "^0.17.0",
|
|
86
|
+
"prettier": "^3.3.0"
|
|
87
|
+
},
|
|
88
|
+
"optionalDependencies": {
|
|
89
|
+
"playwright": "^1.61.1"
|
|
90
|
+
}
|
|
91
|
+
}
|
package/tsconfig.json
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
{
|
|
2
|
+
"compilerOptions": {
|
|
3
|
+
"target": "es2024",
|
|
4
|
+
"module": "esnext",
|
|
5
|
+
"moduleResolution": "bundler",
|
|
6
|
+
"declaration": true,
|
|
7
|
+
"declarationMap": true,
|
|
8
|
+
"sourceMap": true,
|
|
9
|
+
"skipLibCheck": true,
|
|
10
|
+
"esModuleInterop": true,
|
|
11
|
+
"allowImportingTsExtensions": true,
|
|
12
|
+
"rewriteRelativeImportExtensions": true,
|
|
13
|
+
"strict": true,
|
|
14
|
+
"noUncheckedIndexedAccess": true,
|
|
15
|
+
"exactOptionalPropertyTypes": true,
|
|
16
|
+
"noImplicitOverride": true,
|
|
17
|
+
"noFallthroughCasesInSwitch": true,
|
|
18
|
+
"noUnusedLocals": true,
|
|
19
|
+
"noUnusedParameters": true,
|
|
20
|
+
"jsx": "react-jsx",
|
|
21
|
+
"types": [
|
|
22
|
+
"node"
|
|
23
|
+
],
|
|
24
|
+
"rootDir": "src",
|
|
25
|
+
"outDir": "lib"
|
|
26
|
+
},
|
|
27
|
+
"include": [
|
|
28
|
+
"src"
|
|
29
|
+
]
|
|
30
|
+
}
|
package/tsdown.config.ts
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/** Bundle config: one node entry (host tool). */
|
|
2
|
+
import { defineConfig } from 'tsdown'
|
|
3
|
+
|
|
4
|
+
export default defineConfig({
|
|
5
|
+
name: "dsh-ab-wechat-scrape",
|
|
6
|
+
entry: ['lib/index.js'],
|
|
7
|
+
outDir: 'lib',
|
|
8
|
+
format: ['esm'],
|
|
9
|
+
platform: 'node',
|
|
10
|
+
target: 'es2024',
|
|
11
|
+
fixedExtension: false,
|
|
12
|
+
dts: false,
|
|
13
|
+
clean: false,
|
|
14
|
+
deps: {
|
|
15
|
+
neverBundle: [/^@deepseek-ai\//],
|
|
16
|
+
alwaysBundle: (specifier) => !specifier.startsWith('@deepseek-ai/') && !specifier.startsWith('node:'),
|
|
17
|
+
},
|
|
18
|
+
})
|