@wwkit/harness 1.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +101 -0
  3. package/agents/extract.md +38 -0
  4. package/agents/query.md +54 -0
  5. package/agents/revise.md +40 -0
  6. package/package.json +55 -0
  7. package/plugin.js +19 -0
  8. package/readme/development.md +37 -0
  9. package/readme/publish.md +38 -0
  10. package/readme/testing.md +36 -0
  11. package/skills/extract/SKILL.md +100 -0
  12. package/skills/extract/references/build-xpath.js +31 -0
  13. package/skills/extract/references/clean-html.js +91 -0
  14. package/skills/extract/references/detail.md +60 -0
  15. package/skills/extract/references/detail.schema.json5 +49 -0
  16. package/skills/extract/references/extract-detail.js +72 -0
  17. package/skills/extract/references/extract-regex.js +82 -0
  18. package/skills/extract/references/extract-sample.js +49 -0
  19. package/skills/extract/references/format-aliases.json5 +22 -0
  20. package/skills/extract/references/input.schema.json5 +23 -0
  21. package/skills/extract/references/list-rule-gen.md +109 -0
  22. package/skills/extract/references/list.schema.json5 +34 -0
  23. package/skills/extract/references/list_from_html.md +74 -0
  24. package/skills/extract/references/list_from_json.md +52 -0
  25. package/skills/extract/references/list_from_text.md +60 -0
  26. package/skills/extract/references/navi.md +49 -0
  27. package/skills/extract/references/navi.schema.json5 +23 -0
  28. package/skills/extract/references/text.md +46 -0
  29. package/skills/extract/references/text.schema.json5 +22 -0
  30. package/skills/extract/references/to-text.js +21 -0
  31. package/skills/extract/references/validate-schema.js +40 -0
  32. package/skills/revise/SKILL.md +113 -0
  33. package/skills/revise/references/article.md +44 -0
  34. package/skills/revise/references/article.schema.json5 +40 -0
  35. package/skills/revise/references/format-aliases.json5 +22 -0
  36. package/skills/revise/references/gallery.md +43 -0
  37. package/skills/revise/references/gallery.schema.json5 +40 -0
  38. package/skills/revise/references/input.schema.json5 +28 -0
  39. package/skills/revise/references/question.md +41 -0
  40. package/skills/revise/references/question.schema.json5 +31 -0
  41. package/skills/revise/references/status.md +40 -0
  42. package/skills/revise/references/status.schema.json5 +27 -0
  43. package/skills/revise/references/validate-schema.js +40 -0
@@ -0,0 +1,91 @@
1
+ import { fileURLToPath } from 'url';
2
+ import * as cheerio from 'cheerio';
3
+
4
+ const STRIP_TAGS = new Set([
5
+ 'script', 'style', 'noscript', 'iframe', 'svg', 'canvas', 'template',
6
+ 'form', 'video', 'audio', 'object', 'embed', 'nav', 'footer', 'aside',
7
+ ]);
8
+
9
+ const WRAPPER_TAGS = new Set([
10
+ 'strong', 'b', 'i', 'em', 'span', 'u', 's', 'strike', 'mark',
11
+ 'small', 'sub', 'sup', 'code', 'kbd', 'q', 'abbr', 'cite',
12
+ ]);
13
+
14
+ const KEEP_ATTRS = new Set(['href', 'src', 'alt']);
15
+
16
+ export function cleanHtml(html) {
17
+ if (!html) return '';
18
+ let $ = cheerio.load(html);
19
+
20
+ // 1. Strip blocks
21
+ STRIP_TAGS.forEach(tag => { $(tag).remove(); });
22
+
23
+ // 2. Strip comments
24
+ $.root().find('*').contents().filter((i, el) => el.type === 'comment').remove();
25
+
26
+ // 3. Extract body
27
+ const body = $('body').html();
28
+ if (body) {
29
+ $ = cheerio.load(body);
30
+ }
31
+
32
+ // 4. Strip attributes (keep only href/src/alt)
33
+ $('*').each((i, el) => {
34
+ const attrs = el.attribs;
35
+ Object.keys(attrs).forEach(attr => {
36
+ if (!KEEP_ATTRS.has(attr)) delete attrs[attr];
37
+ });
38
+ });
39
+
40
+ // 5. Strip wrapper tags (unwrap content)
41
+ WRAPPER_TAGS.forEach(tag => {
42
+ $(tag).each((i, el) => {
43
+ const $el = $(el);
44
+ $el.replaceWith($el.contents());
45
+ });
46
+ });
47
+
48
+ // 6. Strip void elements (img, br, hr, etc.) and empty paired elements
49
+ const VOID_TAGS = new Set([
50
+ 'area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input',
51
+ 'link', 'meta', 'param', 'source', 'track', 'wbr',
52
+ ]);
53
+ VOID_TAGS.forEach(tag => { $(tag).remove(); });
54
+
55
+ // Remove empty elements repeatedly
56
+ let prev = '';
57
+ let curr = $.html();
58
+ while (curr !== prev) {
59
+ prev = curr;
60
+ $ = cheerio.load(curr);
61
+ $('*').each((i, el) => {
62
+ const $el = $(el);
63
+ if ($el.text().trim() === '' && !$el.children().length) {
64
+ $el.remove();
65
+ }
66
+ });
67
+ curr = $.html();
68
+ }
69
+
70
+ // 7. Strip base64 images
71
+ $ = cheerio.load(curr);
72
+ $('img[src^="data:image/"]').remove();
73
+
74
+ // 8. Minify
75
+ let result = $.html();
76
+ result = result.replace(/\s+/g, ' ');
77
+ result = result.replace(/>\s+</g, '><');
78
+ return result.trim();
79
+ }
80
+
81
+ function main() {
82
+ const chunks = [];
83
+ process.stdin.on('data', chunk => chunks.push(chunk));
84
+ process.stdin.on('end', () => {
85
+ const html = Buffer.concat(chunks).toString();
86
+ process.stdout.write(cleanHtml(html));
87
+ });
88
+ }
89
+
90
+ export default cleanHtml;
91
+ if (process.argv[1] === fileURLToPath(import.meta.url)) main();
@@ -0,0 +1,60 @@
1
+ # detail 提取流程
2
+
3
+ 从文章详情页的正文容器 HTML 中提取结构化信息(本地提取 + LLM 规范)。
4
+
5
+ ## 输入
6
+
7
+ - `source`:正文容器 HTML(已去除非正文内容)
8
+ - `url`:页面 URL(可选,相对路径转绝对)
9
+
10
+ ## 输出
11
+
12
+ 符合 `detail.schema.json5` 的 JSON 数组(仅含 1 项):
13
+
14
+ ```json
15
+ [
16
+ {
17
+ "title": "文章标题",
18
+ "author": "作者名",
19
+ "date": "2026-06-20",
20
+ "url": "https://example.com/article/123",
21
+ "blocks": [
22
+ {"type": "text", "value": "第一段文本"},
23
+ {"type": "image", "value": "https://example.com/pic1.jpg"},
24
+ {"type": "text", "value": "第二段文本"}
25
+ ]
26
+ }
27
+ ]
28
+ ```
29
+
30
+ 字段:`title`(可选,空则基于全文生成 ≤30 字短标题)、`author` / `date`(可选)、`url`(可选,直接用输入 `{{ url }}`)、`blocks`(必填,`type`/`value` 交替数组)。
31
+
32
+ ## 步骤
33
+
34
+ ### Step 1 — 本地提取(按 img 切分)
35
+
36
+ 用 `write` 工具将正文容器 HTML 写入临时文件,再用 stdin 重定向交给 Node.js 按 `<img>` 标签切分:
37
+
38
+ ```bash
39
+ node '<技能目录>/references/extract-detail.js' < <临时 HTML 文件>
40
+ ```
41
+
42
+ 输出 JSON 包含 `blocks`:按 DOM 序交替排列的 `{"type":"text"/"image", "value":...}`。
43
+
44
+ ### Step 2 — LLM 数据规范
45
+
46
+ 将 Step 1 输出的 `{blocks}` + `{{ url }}` 交给 LLM 进行规范:
47
+
48
+ 1. **提取元数据**:从 `blocks[0]["value"]`(文章首段)中识别并提取 `title`、`author`、`date`
49
+ - 标题通常是首段开头最显著的完整短句
50
+ - 作者通常在标题后或段尾,常见格式:「作者:xxx」「文/xxx」「xxx/文」
51
+ - 日期通常在段首或段尾,格式多样,统一输出 `yyyy-mm-dd`
52
+ 2. **无标题时生成**:如果首段无法提取 title,基于全部 `blocks` 中 `type=text` 的内容生成一个不超过 30 字的短标题
53
+ 3. **清理 blocks**:从 `blocks` 数组中移除已提取的元数据行、广告引导语、文末推荐、版权声明等无效段(仅移除 `type=text` 的无效段,保留图片段)
54
+ 4. 相对路径 URL 基于 `{{ url }}` 转为绝对地址
55
+
56
+ 输出:规范后的 JSON 数组(`blocks` 已清理)。
57
+
58
+ ### Step 3 — 验证
59
+
60
+ 检查 `blocks` 非空,输出规范后的 JSON 数组(供 extract 技能阶段 4 校验)。
@@ -0,0 +1,49 @@
1
+ {
2
+ // 省略 $schema 字段,避免 bash -c "" 中 $ 展开问题
3
+ "title": "详情提取格式",
4
+ "description": "详情页提取(detail)输出校验 schema(数组,仅 1 项)",
5
+ "type": "array",
6
+ "items": {
7
+ "type": "object",
8
+ "properties": {
9
+ "title": {
10
+ "type": "string",
11
+ "description": "文章标题,可选;空则生成短标题"
12
+ },
13
+ "author": {
14
+ "type": "string",
15
+ "description": "作者,可选"
16
+ },
17
+ "date": {
18
+ "type": "string",
19
+ "description": "日期,统一 yyyy-mm-dd,可选"
20
+ },
21
+ "url": {
22
+ "type": "string",
23
+ "description": "文章链接,可选"
24
+ },
25
+ "blocks": {
26
+ "type": "array",
27
+ "description": "正文段落,type/value 交替数组",
28
+ "items": {
29
+ "type": "object",
30
+ "properties": {
31
+ "type": {
32
+ "type": "string",
33
+ "enum": ["text", "image"],
34
+ "description": "段落类型:text 文本 / image 图片"
35
+ },
36
+ "value": {
37
+ "type": "string",
38
+ "description": "文本内容或图片地址"
39
+ }
40
+ },
41
+ "required": ["type", "value"],
42
+ "additionalProperties": false
43
+ }
44
+ }
45
+ },
46
+ "required": ["blocks"],
47
+ "additionalProperties": false
48
+ }
49
+ }
@@ -0,0 +1,72 @@
1
+ import { fileURLToPath } from 'url';
2
+ import * as cheerio from 'cheerio';
3
+
4
+ const IMG_RE = /<img\s[^>]*src=["'](.*?)["'][^>]*>/gi;
5
+
6
+ export function extractDetail(html, maxLength = -1, maxImageCount = -1) {
7
+ if (!html || !html.trim()) return { paras: [] };
8
+
9
+ const parts = html.split(IMG_RE);
10
+ const rawParas = [];
11
+
12
+ for (let i = 0; i < parts.length; i++) {
13
+ if (i % 2 === 0) {
14
+ const text = cleanText(parts[i]);
15
+ if (text) rawParas.push({ type: 'text', value: text });
16
+ } else {
17
+ const imgUrl = (parts[i] || '').trim();
18
+ if (imgUrl) rawParas.push({ type: 'image', value: imgUrl });
19
+ }
20
+ }
21
+
22
+ return { paras: applyLimits(rawParas, maxLength, maxImageCount) };
23
+ }
24
+
25
+ function cleanText(raw) {
26
+ const $ = cheerio.load(raw);
27
+ let text = $.text();
28
+ text = text.replace(/\s+/g, ' ').trim();
29
+ return text;
30
+ }
31
+
32
+ function applyLimits(paras, maxLength, maxImageCount) {
33
+ if (maxLength === -1 && maxImageCount === -1) return paras;
34
+ const result = [];
35
+ let accLen = 0;
36
+ let imgCount = 0;
37
+
38
+ for (const entry of paras) {
39
+ if (entry.type === 'image') {
40
+ if (maxImageCount !== -1 && imgCount >= maxImageCount) break;
41
+ if (maxLength !== -1 && accLen >= maxLength) break;
42
+ imgCount++;
43
+ result.push(entry);
44
+ } else {
45
+ let text = entry.value;
46
+ if (maxLength !== -1) {
47
+ if (accLen >= maxLength) break;
48
+ const remaining = maxLength - accLen;
49
+ if (text.length > remaining) {
50
+ text = text.slice(0, remaining);
51
+ result.push({ type: 'text', value: text });
52
+ break;
53
+ }
54
+ }
55
+ result.push({ type: 'text', value: text });
56
+ accLen += text.length;
57
+ }
58
+ }
59
+ return result;
60
+ }
61
+
62
+ function main() {
63
+ const chunks = [];
64
+ process.stdin.on('data', chunk => chunks.push(chunk));
65
+ process.stdin.on('end', () => {
66
+ const html = Buffer.concat(chunks).toString();
67
+ process.stdout.write(JSON.stringify(extractDetail(html), null, 2));
68
+ });
69
+ }
70
+
71
+ export default extractDetail;
72
+ if (process.argv[1] === fileURLToPath(import.meta.url)) main();
@@ -0,0 +1,82 @@
1
+ import { fileURLToPath } from 'url';
2
+
3
+ export function extractRegex(html, rules) {
4
+ if (!html || !rules) return [];
5
+
6
+ const { container_regex, fields } = rules;
7
+ if (!container_regex || !fields) return [];
8
+
9
+ const containerPattern = new RegExp(container_regex, 'gs');
10
+ const items = [];
11
+ let match;
12
+
13
+ while ((match = containerPattern.exec(html)) !== null) {
14
+ const block = match[1] !== undefined ? match[1] : match[0];
15
+ const item = {};
16
+
17
+ for (const [fieldName, fieldPattern] of Object.entries(fields)) {
18
+ if (fieldPattern) {
19
+ const fm = new RegExp(fieldPattern, 's').exec(block);
20
+ if (fm) {
21
+ item[fieldName] = (fm[1] !== undefined ? fm[1] : fm[0]).trim();
22
+ } else {
23
+ item[fieldName] = '';
24
+ }
25
+ } else {
26
+ item[fieldName] = '';
27
+ }
28
+ }
29
+
30
+ item.selector = makeSelector(item.title || '');
31
+ items.push(item);
32
+ }
33
+
34
+ return items;
35
+ }
36
+
37
+ const CURLY_CHARS = '\u201c\u201d\u2018\u2019';
38
+ const EXCLUDE = ' and not(ancestor::script)';
39
+
40
+ function makeSelector(title) {
41
+ if (!title) return '';
42
+ const hasCurly = [...CURLY_CHARS].some(c => title.includes(c));
43
+
44
+ if (!hasCurly) {
45
+ if (!title.includes("'")) return `//text()[contains(.,'${title}')${EXCLUDE}]/..`;
46
+ if (!title.includes('"')) return `//text()[contains(.,"${title}")${EXCLUDE}]/..`;
47
+ const parts = title.split("'");
48
+ const inner = parts.map(p => `'${p}'`).join(",\"'\",");
49
+ return `//text()[contains(.,concat(${inner}))${EXCLUDE}]/..`;
50
+ }
51
+
52
+ const allQuotes = CURLY_CHARS + '"\'';
53
+ const segs = title.split(new RegExp(`[${allQuotes}]+`)).filter(Boolean);
54
+ if (!segs.length) return '';
55
+
56
+ const clauses = segs.map(s => {
57
+ if (!s.includes("'")) return `contains(.,'${s}')`;
58
+ if (!s.includes('"')) return `contains(.,"${s}")`;
59
+ const sub = s.split("'");
60
+ const inner = sub.map(p => `'${p}'`).join(",\"'\",");
61
+ return `contains(.,concat(${inner}))`;
62
+ });
63
+
64
+ return `//text()[${clauses.join(' and ')}${EXCLUDE}]/..`;
65
+ }
66
+
67
+ function main() {
68
+ const chunks = [];
69
+ process.stdin.on('data', chunk => chunks.push(chunk));
70
+ process.stdin.on('end', () => {
71
+ try {
72
+ const input = JSON.parse(Buffer.concat(chunks).toString());
73
+ const result = extractRegex(input.html, input.rules);
74
+ process.stdout.write(JSON.stringify(result, null, 2));
75
+ } catch (e) {
76
+ process.stdout.write('[]');
77
+ }
78
+ });
79
+ }
80
+
81
+ export { makeSelector };
82
+ if (process.argv[1] === fileURLToPath(import.meta.url)) main();
@@ -0,0 +1,49 @@
1
+ import { fileURLToPath } from 'url';
2
+ import * as cheerio from 'cheerio';
3
+ import { cleanHtml } from './clean-html.js';
4
+
5
+ export function extractSample(html, keepItems = 3) {
6
+ const cleaned = cleanHtml(html);
7
+ const $ = cheerio.load(cleaned);
8
+
9
+ // Find content items: look for list items or repeated block elements
10
+ const items = [];
11
+
12
+ // Try <li> first
13
+ $('li').each((i, el) => {
14
+ if (items.length < keepItems) items.push($(el).html() || '');
15
+ });
16
+
17
+ if (items.length > 0) {
18
+ return `<ul>${items.map(item => `<li>${item}</li>`).join('')}</ul>`;
19
+ }
20
+
21
+ // Fallback: look for repeated block elements
22
+ const blockTags = ['p', 'div', 'section', 'article', 'tr', 'td'];
23
+ for (const tag of blockTags) {
24
+ items.length = 0;
25
+ $(tag).each((i, el) => {
26
+ if (items.length < keepItems) items.push($.html(el));
27
+ });
28
+ if (items.length > 0) break;
29
+ }
30
+
31
+ return items.join('\n');
32
+ }
33
+
34
+ function main() {
35
+ const chunks = [];
36
+ process.stdin.on('data', chunk => chunks.push(chunk));
37
+ process.stdin.on('end', () => {
38
+ try {
39
+ const input = JSON.parse(Buffer.concat(chunks).toString());
40
+ const result = extractSample(input.html, input.keep_items || 3);
41
+ process.stdout.write(result);
42
+ } catch (e) {
43
+ process.stdout.write('');
44
+ }
45
+ });
46
+ }
47
+
48
+ export default extractSample;
49
+ if (process.argv[1] === fileURLToPath(import.meta.url)) main();
@@ -0,0 +1,22 @@
1
+ {
2
+ // format 别名映射表:供 agent 将用户的自然语言描述翻译为 format 标准值。
3
+ // 用户可能用中文/口语而非英文标识,按以下 terms 命中匹配。
4
+ "aliases": {
5
+ "list": {
6
+ "description": "列表提取:从列表页提取多条目(HTML 或结构化 JSON 两种来源)",
7
+ "terms": ["列表", "多条", "集合", "条目", "list"]
8
+ },
9
+ "detail": {
10
+ "description": "详情提取:单条详情/正文内容",
11
+ "terms": ["详情", "正文", "单条", "内容页", "detail"]
12
+ },
13
+ "text": {
14
+ "description": "纯文本提取:HTML 转纯文本",
15
+ "terms": ["纯文本", "文字", "文本", "转文本", "text"]
16
+ },
17
+ "navi": {
18
+ "description": "导航提取:生成栏目/链接导航的 XPath",
19
+ "terms": ["导航", "栏目", "链接导航", "navi"]
20
+ }
21
+ }
22
+ }
@@ -0,0 +1,23 @@
1
+ {
2
+ // extract 技能入参字段清单(纯文档用途,供 agent 解析入参时参照,不做运行时校验)。
3
+ "title": "extract 技能入参字段清单",
4
+ "type": "object",
5
+ "properties": {
6
+ "source": {
7
+ "type": "string",
8
+ "description": "网页 HTML/文本内容,或本地文件路径",
9
+ "required": true
10
+ },
11
+ "format": {
12
+ "type": "string",
13
+ "description": "提取类型。标准值为 list/detail/text/navi 之一",
14
+ "enum": ["list", "detail", "text", "navi"],
15
+ "required": true
16
+ },
17
+ "url": {
18
+ "type": "string",
19
+ "description": "页面 URL;source 为相对路径时基于它转为绝对路径",
20
+ "required": false
21
+ }
22
+ }
23
+ }
@@ -0,0 +1,109 @@
1
+ # 规则生成指导
2
+
3
+ ## 你的角色
4
+
5
+ 你是一个 HTML 结构分析专家。给你一段(仅前几项的)HTML 样本,你需要分析其重复结构并生成正则提取规则。
6
+
7
+ ## 输入数据
8
+
9
+ - **source**:干净的 HTML 片段(移除了 script/style/注释,仅含前若干条记录)。≥ 2 条时通过对比识别容器边界最准确;仅 1 条时需基于标签结构的层级关系推断容器模式
10
+ - **fields**:需要提取的字段及其语义说明,例如 `{"title": "新闻标题", "author": "作者名"}`。LLM 据此在 HTML 中定位对应内容并编写正则
11
+ - **path**(可选,框架级参数,LLM 无需处理):指定时将规则写入该文件,否则直接输出
12
+
13
+ ## 输出格式
14
+
15
+ 必须是严格 JSON,不含任何其他内容。`fields` 的 key 与输入 `fields` 的 key 完全一致:
16
+
17
+ ```json
18
+ {
19
+ "container_regex": "用于匹配每条容器的正则表达式,必须包含至少一个捕获组(group 1)",
20
+ "fields": {
21
+ "<field_name_1>": "对应输入字段的正则提取规则",
22
+ "<field_name_2>": "..."
23
+ }
24
+ }
25
+ ```
26
+
27
+ ## 正则编写规则
28
+
29
+ ### 容器正则(container_regex)
30
+
31
+ - 必须匹配列表中**每一条完整记录**的 HTML 块
32
+ - 必须包含**一个捕获组 `(.*?)`**,代表单条记录的内部内容
33
+ - 使用 `.*?`(非贪婪)避免跨记录匹配
34
+ - 示例:
35
+ - `<li class="item">(.*?)</li>`
36
+ - `<div class="card"[^>]*>(.*?)</div>\s*`
37
+ - `<tr[^>]*>(.*?)</tr>`
38
+
39
+ ### 字段正则(fields)
40
+
41
+ - 基于容器捕获组的内部内容编写
42
+ - 使用具体标签名和 CSS 类名定位
43
+ - 优先用 `<h1~6>`、`<a>` 提取标题
44
+ - 用 `<time[datetime]>`、`<span[class~=date]>` 提取日期
45
+ - 用 `<a[href]>` 提取链接
46
+ - 用 `<img[src]>` 提取图片
47
+ - 字段不存在时值留空字符串 `""`
48
+ - **捕获组规则**:字段正则若有捕获组,提取值取 `group(1)`;建议每个字段正则只含 **1 个捕获组**。
49
+ > 反例:`<a href="(.*?)"[^>]*>(.*?)</a>` 含 2 个捕获组(href、文本),代码会静默取 `group(1)`(href)而非标题文本,与预期不符。应写为 `<a[^>]*>(.*?)</a>` 只捕获标签文本。
50
+
51
+ ### 常见模式(参考示例,非固定字段集)
52
+
53
+ > **重要**:以下仅为正则写法范式。实际 class 名、标签层级必须**基于输入样本的真实结构**编写,禁止照搬表中的 class 名。
54
+
55
+ | 字段语义 | 典型正则 |
56
+ |---------|---------|
57
+ | 标题类 | `<h3[^>]*class="title"[^>]*>(.*?)</h3>` |
58
+ | 作者类 | `<span[^>]*class="author"[^>]*>(.*?)</span>` |
59
+ | 日期类 | `<time[^>]*datetime="(.*?)"` 或 `<span[^>]*class="date"[^>]*>(.*?)</span>` |
60
+ | 链接类 | `<a[^>]*href="(.*?)"` |
61
+ | 图片类 | `<img[^>]*src="(.*?)"` |
62
+
63
+ ## 完整示例(以新闻列表字段为例)
64
+
65
+ ### 输入
66
+
67
+ `fields` = `{"title": "新闻标题", "author": "作者名", "date": "发布日期", "url": "文章链接", "thumb": "缩略图地址"}`
68
+
69
+ ### 输入样本
70
+
71
+ ```html
72
+ <li class="news-item">
73
+ <h3 class="title"><a href="/article/1">Title One</a></h3>
74
+ <span class="author">Alice</span>
75
+ <time datetime="2024-01-15">2024-01-15</time>
76
+ <img src="/thumb/1.jpg" alt="thumb1">
77
+ </li>
78
+ <li class="news-item">
79
+ <h3 class="title"><a href="/article/2">Title Two</a></h3>
80
+ <span class="author">Bob</span>
81
+ <time datetime="2024-01-16">2024-01-16</time>
82
+ <img src="/thumb/2.jpg" alt="thumb2">
83
+ </li>
84
+ ```
85
+
86
+ ### 输出规则
87
+
88
+ ```json
89
+ {
90
+ "container_regex": "<li[^>]*class=\"news-item\"[^>]*>(.*?)</li>",
91
+ "fields": {
92
+ "title": "<h3[^>]*class=\"title\"[^>]*><a[^>]*>(.*?)</a></h3>",
93
+ "author": "<span[^>]*class=\"author\"[^>]*>(.*?)</span>",
94
+ "date": "<time[^>]*datetime=\"(.*?)\"",
95
+ "url": "<a[^>]*href=\"(.*?)\"",
96
+ "thumb": "<img[^>]*src=\"(.*?)\""
97
+ }
98
+ }
99
+ ```
100
+
101
+ ## 注意事项
102
+
103
+ - **正则必须匹配清洗后 HTML**(等同 Step 3 中 `HtmlCleaner.clean()` 的输出口径),包括属性中的引号和空格
104
+ - **自我验证**:产出规则后,在输入的样本上跑一遍——`container_regex` 至少匹配 1 条、每个字段至少有一条非空命中。不满足则调整规则
105
+ - `container_regex` 的捕获组数量必须 >= 1
106
+ - 字段正则可能匹配到空值(如空 `<span></span>`),此时提取为空字符串
107
+ - 日期保留原始格式即可,规范化由下游处理
108
+ - 链接和图片如果相对路径,保留原始值
109
+ - **嵌套容器陷阱**:如果容器标签可能自嵌套(如 `<div>` 内嵌 `<div>`),`.*?` 非贪婪会在第一个闭合标签处提前闭合,导致匹配截断。优先选择不会自嵌套的标签(如 `<li>` 罕见包 `<li>`),或对容器用更具体的 class 锚定
@@ -0,0 +1,34 @@
1
+ {
2
+ // 省略 $schema 字段,避免 bash -c "" 中 $ 展开问题
3
+ "title": "列表提取格式",
4
+ "description": "列表页提取(list)输出校验 schema(数组)",
5
+ "type": "array",
6
+ "items": {
7
+ "type": "object",
8
+ "properties": {
9
+ "title": {
10
+ "type": "string",
11
+ "minLength": 1,
12
+ "description": "标题,非空"
13
+ },
14
+ "author": {
15
+ "type": "string",
16
+ "description": "作者,可选"
17
+ },
18
+ "date": {
19
+ "type": "string",
20
+ "description": "日期,统一 yyyy-mm-dd,可选"
21
+ },
22
+ "url": {
23
+ "type": "string",
24
+ "description": "文章链接,可选"
25
+ },
26
+ "thumb": {
27
+ "type": "string",
28
+ "description": "缩略图地址,可选"
29
+ }
30
+ },
31
+ "required": ["title"],
32
+ "additionalProperties": false
33
+ }
34
+ }
@@ -0,0 +1,74 @@
1
+ # list_from_html 提取流程
2
+
3
+ 从列表页 HTML 中提取结构化信息(混合方案:本地提取 + 规则 + LLM 规范)。
4
+
5
+ ## 输入
6
+
7
+ - `source`:完整网页 HTML 源码
8
+ - `url`:页面 URL(可选,相对路径转绝对)
9
+
10
+ ## 输出
11
+
12
+ 符合 `list.schema.json5` 的 JSON 数组:
13
+
14
+ ```json
15
+ [
16
+ {
17
+ "title": "标题",
18
+ "author": "作者",
19
+ "date": "2026-06-12",
20
+ "url": "https://example.com/article/123",
21
+ "thumb": "https://example.com/thumb.jpg"
22
+ }
23
+ ]
24
+ ```
25
+
26
+ 字段:`title`(必填,空则跳过本条)、`author` / `date` / `url` / `thumb`(可选,无则留空)。
27
+
28
+ ## 步骤
29
+
30
+ ### Step 1 — 样本截取(本地)
31
+
32
+ 用 `write` 工具将完整 HTML 写入临时文件(如 `extract_list.html`),再用 `write` 工具将 `{"html": "<HTML内容>", "keep_items": 3}` 写入 JSON 输入文件,stdin 重定向交给 Node.js 清洗并截取前 3 条:
33
+
34
+ ```bash
35
+ node '<技能目录>/references/extract-sample.js' < <临时 JSON 输入文件>
36
+ ```
37
+
38
+ 输出仅含前 3 条(或更少)的干净 HTML 样本。
39
+
40
+ ### Step 2 — 规则生成(LLM)
41
+
42
+ 先用 `read` 读取 `references/list-rule-gen.md`(HTML 正则生成指导,含编写规则与示例),按其中指导分析 Step 1 输出的样本 HTML,生成提取规则:
43
+
44
+ - 字段集固定为 `{"title": "新闻标题", "author": "作者名", "date": "发布日期", "url": "文章链接", "thumb": "缩略图地址"}`
45
+ - 输出 `{container_regex, fields: {title, author, date, url, thumb}}` 格式的 JSON 规则,**不写入文件**,直接在后续步骤内联使用
46
+
47
+ `container_regex` 用于定义每条记录的外层容器边界(如 `<li>...</li>`),Node.js 先用它切分 HTML 为单条容器,再对每条容器内部应用字段正则——避免跨边界匹配。字段正则若有捕获组,提取值取 `group(1)`;建议每个字段正则只含 1 个捕获组。
48
+
49
+ ### Step 3 — 批量提取(本地)
50
+
51
+ 对完整 HTML 用相同清洗逻辑处理后再提取(与 Step 1 清洗口径一致)。用 `write` 工具将输入 JSON 写入临时文件,再 stdin 重定向交给 Node.js:
52
+
53
+ ```bash
54
+ node '<技能目录>/references/extract-regex.js' < <临时输入文件>
55
+ ```
56
+ ```
57
+
58
+ 输入 JSON:`{"html": "<完整HTML>", "rules": <step2 规则>}`
59
+
60
+ 输出:JSON 数组,每项含 `url`, `date`, `author`, `thumb`, `title` 五个字段(title 为空的条目已移除)。
61
+
62
+ ### Step 4 — LLM 数据规范
63
+
64
+ 将 Step 3 输出的数组交给 LLM 进行数据规范:
65
+
66
+ - URL 相对路径 → 基于 `{{ url }}` 转为绝对地址
67
+ - 日期统一格式 `yyyy-mm-dd`
68
+ - 作者清洗(去多余空白、特殊字符)
69
+ - **保持条目数量和顺序不变**
70
+ - 输出必须是 JSON 数组(不是对象包裹)
71
+
72
+ ### Step 5 — 验证
73
+
74
+ 检查每项 title 非空,输出规范后的 JSON 数组(供 extract 技能阶段 4 校验)。