ppxans-harness 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +265 -0
  3. package/bin/ppx-channels.js +3 -0
  4. package/bin/ppx-serve.js +6 -0
  5. package/bin/ppx.js +3 -0
  6. package/config/identity.md +6 -0
  7. package/config/ishiki.md +16 -0
  8. package/config/ppx.json +151 -0
  9. package/package.json +69 -0
  10. package/src/agent/context.js +179 -0
  11. package/src/agent/index.js +717 -0
  12. package/src/agent/prompts.js +107 -0
  13. package/src/aml-server.js +151 -0
  14. package/src/ans/eviction.js +144 -0
  15. package/src/ans/guard.js +120 -0
  16. package/src/ans/lifecycle.js +93 -0
  17. package/src/ans/proactive.js +129 -0
  18. package/src/ans/reward.js +112 -0
  19. package/src/ans/values.js +15 -0
  20. package/src/audit/audit-chain.js +167 -0
  21. package/src/audit/verifier.js +120 -0
  22. package/src/bus/circuit-breaker.js +115 -0
  23. package/src/bus/runtime-bus.js +94 -0
  24. package/src/channels/base.js +35 -0
  25. package/src/channels/feishu.js +127 -0
  26. package/src/channels/http.js +592 -0
  27. package/src/channels/index.js +110 -0
  28. package/src/channels/log.js +29 -0
  29. package/src/channels/wechat-crypto.js +74 -0
  30. package/src/channels/wechat.js +197 -0
  31. package/src/channels-cli.js +124 -0
  32. package/src/cli.js +120 -0
  33. package/src/config/channels.js +170 -0
  34. package/src/config/index.js +224 -0
  35. package/src/config/providers.js +189 -0
  36. package/src/config/settings.js +182 -0
  37. package/src/core/policy.js +272 -0
  38. package/src/core/trace.js +89 -0
  39. package/src/evolve/playbook.js +194 -0
  40. package/src/llm/client.js +446 -0
  41. package/src/llm/dsml.js +74 -0
  42. package/src/llm/embedder.js +35 -0
  43. package/src/llm/fence.js +105 -0
  44. package/src/llm/index.js +4 -0
  45. package/src/llm/retry.js +73 -0
  46. package/src/llm/router.js +98 -0
  47. package/src/mcp/client.js +375 -0
  48. package/src/mcp/index.js +116 -0
  49. package/src/memory/asset-hub.js +131 -0
  50. package/src/memory/canvas.js +131 -0
  51. package/src/memory/compaction.js +28 -0
  52. package/src/memory/experience.js +122 -0
  53. package/src/memory/fact-store.js +699 -0
  54. package/src/memory/failure-episode.js +99 -0
  55. package/src/memory/fork.js +83 -0
  56. package/src/memory/index.js +7 -0
  57. package/src/memory/l0.js +52 -0
  58. package/src/memory/l2.js +131 -0
  59. package/src/memory/l3.js +112 -0
  60. package/src/memory/memory-ticker.js +240 -0
  61. package/src/memory/session.js +398 -0
  62. package/src/mode/blackboard.js +49 -0
  63. package/src/mode/graph.js +41 -0
  64. package/src/mode/index.js +64 -0
  65. package/src/mode/legion.js +51 -0
  66. package/src/mode/plan-exec.js +50 -0
  67. package/src/mode/router.js +40 -0
  68. package/src/orchestrator/agent-worker.js +70 -0
  69. package/src/orchestrator/dag.js +83 -0
  70. package/src/orchestrator/index.js +2 -0
  71. package/src/orchestrator/legion.js +188 -0
  72. package/src/orchestrator/supervisor.js +177 -0
  73. package/src/persona/index.js +29 -0
  74. package/src/plugin/builtin.js +212 -0
  75. package/src/plugin/context.js +79 -0
  76. package/src/plugin/index.js +62 -0
  77. package/src/seam/registry.js +98 -0
  78. package/src/seam/shell.js +55 -0
  79. package/src/selfheal/evolve.js +68 -0
  80. package/src/selfheal/healer.js +167 -0
  81. package/src/selfheal/run.js +9 -0
  82. package/src/server.js +60 -0
  83. package/src/services/learning-service.js +177 -0
  84. package/src/services/memory-health.js +99 -0
  85. package/src/services/memory-service.js +160 -0
  86. package/src/skills/loader.js +150 -0
  87. package/src/skills/verify.js +100 -0
  88. package/src/tools/advanced.js +353 -0
  89. package/src/tools/builtin.js +298 -0
  90. package/src/tools/catalog.js +159 -0
  91. package/src/tools/command-guard.js +112 -0
  92. package/src/tools/custom.js +47 -0
  93. package/src/tools/delegate.js +297 -0
  94. package/src/tools/document.js +253 -0
  95. package/src/tools/governance.js +260 -0
  96. package/src/tools/index.js +11 -0
  97. package/src/tools/methods.js +178 -0
  98. package/src/tools/ocr.js +59 -0
  99. package/src/tools/seam.js +125 -0
  100. package/src/tools/selfmod.js +176 -0
  101. package/src/utils/logger.js +17 -0
  102. package/src/utils/pii.js +42 -0
  103. package/src/utils/store.js +108 -0
  104. package/src/utils/text.js +16 -0
  105. package/src/utils/trace.js +153 -0
  106. package/src/utils/winutf8.js +15 -0
@@ -0,0 +1,160 @@
1
+ // src/services/memory-service.js - 记忆协调服务 (重构第二刀, 2026-09-14)
2
+ // 目的: 把散在 PPXAgent 上的记忆升降级逻辑 (L1提炼/L2归档/L3画像/经验/检索)
3
+ // 收敛为独立服务, agent 只保留薄委托 (公共 API 兼容, 外部调用点/测试不变)。
4
+ // 设计 (对应方案「记忆升降级协调器」轻量版):
5
+ // - MemoryService 持有四层记忆的协调逻辑, 依赖全部注入 (llm 用 getLlm 闭包,
6
+ // 因 reloadProviders 会热替换 agent.llm, 固定引用会过期)。
7
+ // - afterTurn() 是「升降级协调器」: 一轮对话落盘后, 依次触发 L2 场景归档、
8
+ // 用户主动经验学习、L3 画像跨天刷新 —— 升降级策略集中在此, 可单独调参/替换。
9
+ // - 事件流 (tracer) 埋点保留, 行为与抽取前逐字节等价。
10
+ import { info, warn } from "../utils/logger.js";
11
+ import { logicalDay } from "../utils/store.js";
12
+
13
+ // 辅助 LLM 调用短超时 (提炼/压缩/检索扩展): 模型不可用/网络不通时快速失败降级
14
+ const AUX_LLM_TIMEOUT_MS = 10000;
15
+
16
+ export class MemoryService {
17
+ // deps: { getLlm, facts, scenes, personaStore, experience, lifecycle, tracer }
18
+ constructor(deps) {
19
+ this.getLlm = deps.getLlm; // () => llm (闭包, 实时取当前 provider)
20
+ this.facts = deps.facts; // L1 事实库
21
+ this.scenes = deps.scenes; // L2 场景
22
+ this.personaStore = deps.personaStore; // L3 画像
23
+ this.experience = deps.experience; // 经验库
24
+ this.lifecycle = deps.lifecycle; // ANS 生命周期 (进化计数)
25
+ this.tracer = deps.tracer; // 结构化事件流
26
+ this._personaBuilt = null; // L3 画像上次生成日期 (跨天刷新标记, 原 agent 字段)
27
+ }
28
+
29
+ _llm() { return this.getLlm ? this.getLlm() : null; }
30
+
31
+ // 辅助 LLM 调用前置健康探测: 模型不可用 (本地服务未运行/远端不可达) 时快速跳过
32
+ async _auxLlmReady() {
33
+ const llm = this._llm();
34
+ if (!llm) return false;
35
+ if (typeof llm.health !== "function") return true;
36
+ try { return await llm.health(); } catch { return false; }
37
+ }
38
+
39
+ // L1 提炼: 从一轮对话提取值得长期记忆的事实 (原 agent._extractMemory)
40
+ async extractMemory(user, assistant, existing = []) {
41
+ const llm = this._llm();
42
+ if (!llm) return [];
43
+ if (!(await this._auxLlmReady())) return []; // 模型不可用时跳过提炼 (退回启发式)
44
+ // 噪声治理: 显式跳过寒暄/无信息量/关于系统本身的元讨论
45
+ const sys = "你是记忆提炼器。从对话中提取值得长期记忆的关键事实、用户偏好、待办事项。只输出 JSON 数组, 每项是{content: 一句完整中文记忆}。没有值得记的返回 []。不要解释, 只输出 JSON。\n跳过以下内容: 1) 寒暄/问候/客套话; 2) 无信息量的闲聊; 3) 对助手/系统本身的元讨论与建议 (如任务描述方式、提示词建议等); 4) 已被现有记忆覆盖的内容。";
46
+ let userMsg = "用户: " + String(user).slice(0, 800) + "\n助手: " + String(assistant).slice(0, 800);
47
+ // 感知已有记忆: 若提炼结果与已有记忆含义相同/已被覆盖, 不要输出该条 (避免重复)
48
+ if (existing.length) {
49
+ userMsg += "\n\n【已有记忆】以下记忆已存在, 若你提炼的内容与其中任意一条含义相同或被其覆盖, 则不要输出该条 (避免重复):\n"
50
+ + existing.map((f, i) => `${i + 1}. ${f.content}`).join("\n");
51
+ }
52
+ const r = await llm.chat([
53
+ { role: "system", content: sys },
54
+ { role: "user", content: userMsg },
55
+ ], { timeoutMs: AUX_LLM_TIMEOUT_MS, retryMax: 0 });
56
+ const text = String(r.content || "").trim();
57
+ // 容忍模型把 JSON 包在 markdown 代码块里
58
+ const cleaned = text.replace(/```(?:json|JSON)?\s*/g, "").replace(/```/g, "").trim();
59
+ // 提取第一个最外层 JSON 数组 (贪婪匹配到最后一个 ], 容忍内容里的嵌套方括号)
60
+ const m = cleaned.match(/\[[\s\S]*\]/);
61
+ if (!m) { this.tracer?.event("memory/extract", { count: 0, reason: "no_json" }); return []; }
62
+ try {
63
+ const arr = JSON.parse(m[0]);
64
+ const out = Array.isArray(arr) ? arr.map((x) => String(x.content || x).trim()).filter(Boolean) : [];
65
+ this.tracer?.event("memory/extract", { count: out.length, existing: existing.length });
66
+ return out;
67
+ } catch { this.tracer?.event("memory/extract", { count: 0, reason: "parse_fail" }); return []; }
68
+ }
69
+
70
+ // L0 压缩: 用 LLM 把旧对话浓缩成语义摘要 (原 agent._summarizeMemory)
71
+ async summarizeMemory(raw) {
72
+ const llm = this._llm();
73
+ if (!llm) throw new Error("无 LLM");
74
+ if (!(await this._auxLlmReady())) throw new Error("LLM 不可用, 跳过辅助摘要");
75
+ const r = await llm.chat([
76
+ { role: "system", content: "你是记忆压缩器。把下面这段对话记录压缩成一段简洁的中文摘要(≤200字), 保留关键事实、用户偏好、进展和待办。不要客套, 直接输出摘要。" },
77
+ { role: "user", content: String(raw).slice(0, 4000) },
78
+ ], { timeoutMs: AUX_LLM_TIMEOUT_MS, retryMax: 0 });
79
+ this.tracer?.event("memory/summarize", { chars: String(raw).length, ok: true });
80
+ return r.content;
81
+ }
82
+
83
+ // 检索扩展: 把问题改写成多个词面变体, 补语义召回 (原 agent._expandQuery)
84
+ async expandQuery(q) {
85
+ const llm = this._llm();
86
+ if (!llm) return [];
87
+ if (!(await this._auxLlmReady())) return []; // 模型不可用时跳过扩展 (退回单查询)
88
+ const r = await llm.chat([
89
+ { role: "system", content: "你是查询扩展器。把用户的问题改写成 3 个语义相近但词面不同的检索短语(用于语义记忆检索), 每行一个, 不要序号、不要解释。" },
90
+ { role: "user", content: String(q).slice(0, 300) },
91
+ ], { timeoutMs: AUX_LLM_TIMEOUT_MS, retryMax: 0 });
92
+ return String(r.content || "")
93
+ .split(/\n+/)
94
+ .map((s) => s.replace(/^[\d\.\-、))]\s*/, "").trim())
95
+ .filter((s) => s && s !== String(q).trim())
96
+ .slice(0, 3);
97
+ }
98
+
99
+ // L1 检索: 原始查询 + LLM 扩展变体做 RRF 融合; 无 LLM 时退化为单查询 (原 agent._memoryQuery)
100
+ async query(q, { limit = 5, scope = null } = {}) {
101
+ // 有 embedder 时走 dense 语义检索 (与 BM25 RRF 融合), 否则 LLM 扩展 + RRF
102
+ let hits;
103
+ if (this.facts.embedder) {
104
+ hits = this.facts.querySemantic(q, { limit, scope });
105
+ } else {
106
+ const variants = [q];
107
+ const llm = this._llm();
108
+ if (llm) {
109
+ try { variants.push(...(await this.expandQuery(q))); } catch { /* LLM 失败静默降级 */ }
110
+ }
111
+ hits = variants.length === 1 ? this.facts.query(q, { limit, scope }) : this.facts.queryMulti(variants, { limit, scope });
112
+ }
113
+ this.tracer?.event("memory/query", { q: String(q).slice(0, 80), hits: Array.isArray(hits) ? hits.length : 0, scope: scope || null });
114
+ return hits;
115
+ }
116
+
117
+ // L3 画像刷新: 跨天触发 (原 agent._maybeRefreshPersona, 状态 _personaBuilt 移入本服务)
118
+ refreshPersona() {
119
+ const today = logicalDay();
120
+ if (this._personaBuilt === today) return;
121
+ this._personaBuilt = today;
122
+ try {
123
+ this.personaStore.buildUserPersona(this.facts.list(), { force: true });
124
+ this.personaStore.buildAgentPersona(this.experience.lessons, { force: true });
125
+ this.tracer?.event("memory/persona", { ok: true });
126
+ } catch (e) {
127
+ warn("L3 画像生成失败:", e.message);
128
+ this.tracer?.event("memory/persona", { ok: false }, { error: e?.message });
129
+ }
130
+ }
131
+
132
+ // L2 场景归档: 从新记忆里找需要归档的 (原 agent._archiveScenes)
133
+ archiveScenes() {
134
+ const recent = this.facts.query("", { limit: 5 });
135
+ let assigned = 0;
136
+ for (const f of recent) {
137
+ if (!this.scenes.findByFactId(f.id)) { this.scenes.assign(f); assigned++; }
138
+ }
139
+ if (assigned > 0) this.tracer?.event("memory/scene_assign", { assigned });
140
+ }
141
+
142
+ // 用户主动经验学习: 「经验交给皮皮虾: xxx」指令 (原 agent._learnFromTurn)
143
+ learnFromTurn(userMsg, reply) {
144
+ const m = String(userMsg).match(/经验交给皮皮虾[::]\s*(.+)/i);
145
+ if (m) {
146
+ this.experience.learn({ task: "用户主动分享", lesson: m[1], tags: ["user-shared"] });
147
+ if (this.lifecycle) this.lifecycle.evolve(); // 生命周期: 进化计数 (落盘)
148
+ this.tracer?.event("memory/learn", { lesson: m[1].slice(0, 120), source: "user-shared" });
149
+ info(`学到经验: ${m[1]}`);
150
+ }
151
+ }
152
+
153
+ // 升降级协调器: 一轮对话落盘后统一触发 L2 归档 + 经验学习 + L3 画像刷新
154
+ // (原 agent.chat persist 块里散落的三个调用, 收敛于此)
155
+ afterTurn(userMsg, reply) {
156
+ this.archiveScenes();
157
+ this.learnFromTurn(userMsg, reply);
158
+ this.refreshPersona();
159
+ }
160
+ }
@@ -0,0 +1,150 @@
1
+ // src/skills/loader.js - Skill Catalog + Loader
2
+ // 参考 deepseek-harness 的 skill package(catalog + loader): 可枚举、可发现、可加载的技能注册表
3
+ // 扫描 <skillsDir>/*/SKILL.md, 解析 frontmatter(name/description), 提供 list/get/loadAll
4
+ import fs from "node:fs";
5
+ import path from "node:path";
6
+
7
+ // 解析 SKILL.md 顶部的 --- frontmatter ---
8
+ // 支持简单 `key: value` 和 YAML 折叠块(`>` / `|`),保证多行 description 可被解析。
9
+ export function parseFrontmatter(md) {
10
+ const m = md.match(/^---\r?\n([\s\S]*?)\r?\n---/);
11
+ if (!m) return {};
12
+ const meta = {};
13
+ const lines = m[1].split(/\r?\n/);
14
+ let key = null;
15
+ let block = null;
16
+ let blockLines = [];
17
+ const flush = () => {
18
+ if (key && block) {
19
+ meta[key] = block === ">"
20
+ ? blockLines.join(" ").replace(/\s+/g, " ").trim()
21
+ : blockLines.join("\n").trim();
22
+ }
23
+ key = null;
24
+ block = null;
25
+ blockLines = [];
26
+ };
27
+ for (const line of lines) {
28
+ const mm = line.match(/^([A-Za-z0-9_]+):\s*(.*)$/);
29
+ if (mm) {
30
+ flush();
31
+ key = mm[1];
32
+ const val = mm[2].trim();
33
+ if (val === ">" || val === "|") {
34
+ block = val;
35
+ } else {
36
+ meta[key] = val;
37
+ key = null;
38
+ }
39
+ } else if (block && key) {
40
+ blockLines.push(line.trim());
41
+ }
42
+ }
43
+ flush();
44
+ return meta;
45
+ }
46
+
47
+ // 解析 SKILL.md 正文章节: "## 标题" -> 内容
48
+ // 供按需加载 (渐进式披露): agent 可只读「反合理化」「验证」等特定段, 不用整篇读入
49
+ export function parseSections(md) {
50
+ const sections = {};
51
+ const lines = String(md || "").split(/\r?\n/);
52
+ let cur = null;
53
+ for (const line of lines) {
54
+ const m = line.match(/^##\s+(.+)$/);
55
+ if (m) { cur = m[1].trim(); sections[cur] = ""; continue; }
56
+ if (cur) sections[cur] += line + "\n";
57
+ }
58
+ for (const k of Object.keys(sections)) sections[k] = sections[k].trim();
59
+ return sections;
60
+ }
61
+
62
+ export class SkillLoader {
63
+ constructor(skillsDir) {
64
+ this.dir = skillsDir;
65
+ // 技能使用追踪 (source: Hermes "skill self-improves during use")
66
+ this.usageFile = path.join(skillsDir, ".usage.json");
67
+ this._usage = this._readUsage();
68
+ }
69
+
70
+ _scan() {
71
+ if (!fs.existsSync(this.dir)) return {};
72
+ const out = {};
73
+ for (const entry of fs.readdirSync(this.dir, { withFileTypes: true })) {
74
+ if (!entry.isDirectory()) continue;
75
+ const skillMd = path.join(this.dir, entry.name, "SKILL.md");
76
+ if (!fs.existsSync(skillMd)) continue;
77
+ const raw = fs.readFileSync(skillMd, "utf8");
78
+ const meta = parseFrontmatter(raw);
79
+ out[entry.name] = {
80
+ id: entry.name,
81
+ name: meta.name || entry.name,
82
+ description: meta.description || "",
83
+ dir: path.join(this.dir, entry.name),
84
+ path: skillMd,
85
+ };
86
+ }
87
+ return out;
88
+ }
89
+
90
+ // 可枚举: 全部技能
91
+ list() {
92
+ return Object.values(this._scan());
93
+ }
94
+
95
+ // 可发现: 按 id 查
96
+ get(id) {
97
+ return this._scan()[id] || null;
98
+ }
99
+
100
+ has(id) {
101
+ return !!this._scan()[id];
102
+ }
103
+
104
+ // 读取某个技能的完整 SKILL.md 内容
105
+ read(id) {
106
+ const s = this.get(id);
107
+ if (!s) return null;
108
+ return fs.readFileSync(s.path, "utf8");
109
+ }
110
+
111
+ // 按需读取某个技能的指定章节 (渐进式披露: 只读「反合理化」「验证」等段, 省 token)
112
+ readSection(id, section) {
113
+ const raw = this.read(id);
114
+ if (raw === null) return null;
115
+ const sections = parseSections(raw);
116
+ return sections[section] ?? null;
117
+ }
118
+
119
+ _readUsage() {
120
+ try { if (fs.existsSync(this.usageFile)) { const d = JSON.parse(fs.readFileSync(this.usageFile, "utf8")); if (d && typeof d === "object") return d; } } catch {}
121
+ return {};
122
+ }
123
+
124
+ _saveUsage() {
125
+ try { fs.writeFileSync(this.usageFile, JSON.stringify(this._usage, null, 2), "utf8"); } catch {}
126
+ }
127
+
128
+ // 记录一次技能使用 (load_skill 时调用)
129
+ trackUse(id) {
130
+ this._usage = this._readUsage();
131
+ const u = this._usage[id] || { uses: 0, lastUsed: null };
132
+ u.uses = (u.uses || 0) + 1;
133
+ u.lastUsed = new Date().toISOString();
134
+ this._usage[id] = u;
135
+ this._saveUsage();
136
+ return u;
137
+ }
138
+
139
+ // 某技能使用统计
140
+ useOf(id) { const u = this._readUsage()[id]; return u ? { uses: u.uses || 0, lastUsed: u.lastUsed || null } : { uses: 0, lastUsed: null }; }
141
+
142
+ // 全部技能使用统计
143
+ usageAll() { return this._readUsage(); }
144
+
145
+ // 重置某技能使用计数 (用中自进化升级完成后调用, 防连跑)
146
+ resetUse(id) {
147
+ this._usage = this._readUsage();
148
+ if (this._usage[id]) { this._usage[id].uses = 0; this._usage[id].lastUpgraded = new Date().toISOString(); this._saveUsage(); }
149
+ }
150
+ }
@@ -0,0 +1,100 @@
1
+ // src/skills/verify.js - 技能入库前确定性验证闸门
2
+ // 背景 (DeepSeek-Harness 自进化四拼图之「可靠验证」):
3
+ // refineSkill 由 LLM 提炼技能后直接 create_skill 落盘, 无独立验收环节 —
4
+ // 「完成是自报的, 没有独立验收者」。这会让半成品/幻觉技能污染 skills/。
5
+ // 本模块提供确定性验收 (不依赖 LLM, 静态可测, 缺一即拒):
6
+ // - 结构验收: SKILL.md 必须含 ## 流程 + ## 验证
7
+ // - 落地接地: 内容必须真实引用提炼它的高频工具, 且该工具确有 ≥minFreq 条近期成功轨迹
8
+ // - held-out 回归: (P0② Self-Harness) 若提供 heldOutTraces, 要求接地工具在未见过的子集也有背书, 防过拟合
9
+ // 用法: verifySkill({ name, content, hotTools, okTraces, minFreq, heldOutTraces })
10
+ // -> { ok: boolean, reason?: string, matchedTool?, traceCount?, heldOutCount? }
11
+
12
+ // 结构验收: 验收必需段落 (create_skill 契约为 流程/反合理化/验证; 至少 流程+验证 缺一不可)
13
+ export function requiredSections(content) {
14
+ const c = String(content || "");
15
+ const need = ["## 流程", "## 验证"];
16
+ const missing = need.filter((s) => !c.includes(s));
17
+ return {
18
+ ok: missing.length === 0,
19
+ missing,
20
+ hasFlow: c.includes("## 流程"),
21
+ hasVerify: c.includes("## 验证"),
22
+ };
23
+ }
24
+
25
+ // 落地接地: 技能内容是否真实基于提炼它的高频工具 (反幻觉)
26
+ export function groundedInTools(content, hotTools = []) {
27
+ const c = String(content || "");
28
+ const names = (Array.isArray(hotTools) ? hotTools : []).filter(Boolean);
29
+ if (names.length === 0) return { ok: false, reason: "无高频工具可核对" };
30
+ const hit = names.find((t) => c.toLowerCase().includes(String(t).toLowerCase()));
31
+ return { ok: !!hit, matchedTool: hit || null, hotTools: names.slice(0, 5) };
32
+ }
33
+
34
+ // 轨迹接地: 该工具确有 ≥minFreq 条近期成功轨迹
35
+ export function traceBacked(tool, okTraces = [], minFreq = 2) {
36
+ const arr = Array.isArray(okTraces) ? okTraces : [];
37
+ if (!tool) return { ok: false, reason: "无工具名" };
38
+ const n = arr.filter((t) => t && String(t.tool) === tool).length;
39
+ return { ok: n >= minFreq, count: n, minFreq, tool };
40
+ }
41
+
42
+ // held-out 回归闸门 (Self-Harness: 候选改动必须在 held-out 子集也不退化才合并)
43
+ // 用途: refineSkill 提炼的技能若只在训练轨迹接地、在未见过的 held-out 轨迹里不接地 = 过拟合 → 拒
44
+ export function verifyHeldOut({ tool, heldOutTraces = [], minFreq = 2 } = {}) {
45
+ const need = Math.max(1, Math.floor(minFreq / 2));
46
+ if (!tool) return { ok: false, reason: "无工具名", need };
47
+ const n = (Array.isArray(heldOutTraces) ? heldOutTraces : []).filter((t) => t && String(t.tool) === tool).length;
48
+ return { ok: n >= need, count: n, need, tool };
49
+ }
50
+
51
+ // 总闸门
52
+ export function verifySkill({ name, content, hotTools, okTraces, minFreq = 2, heldOutTraces } = {}) {
53
+ // 1. 结构验收
54
+ const s = requiredSections(content);
55
+ if (!s.ok) {
56
+ return { ok: false, reason: `技能缺必需段落: ${s.missing.join(", ")} (SKILL.md 契约要求 ## 流程 + ## 验证)` };
57
+ }
58
+ // 内容太短 (只够段落标题没实质步骤) 也拒
59
+ const body = String(content || "").replace(/##\s*\S+/g, "").trim();
60
+ if (body.length < 20) {
61
+ return { ok: false, reason: "技能内容太短, 缺实质步骤/检查点" };
62
+ }
63
+ // 2. 落地接地: 内容引用高频工具
64
+ const g = groundedInTools(content, hotTools);
65
+ if (!g.ok) {
66
+ return { ok: false, reason: `技能内容未引用提炼它的高频工具 (${g.hotTools.join(", ")})` };
67
+ }
68
+ // 3. 轨迹接地: 该工具有足够近期成功轨迹背书
69
+ const tb = traceBacked(g.matchedTool, okTraces, minFreq);
70
+ if (!tb.ok) {
71
+ return { ok: false, reason: `工具 ${g.matchedTool} 成功轨迹不足 (${tb.count}/${tb.minFreq})` };
72
+ }
73
+ // 4. held-out 回归: 若提供, 要求接地工具在 held-out 子集也有 ≥ceil(minFreq/2) 条成功轨迹背书 (防过拟合训练集)
74
+ if (heldOutTraces && Array.isArray(heldOutTraces)) {
75
+ const ho = verifyHeldOut({ tool: g.matchedTool, heldOutTraces, minFreq });
76
+ if (!ho.ok) {
77
+ return { ok: false, reason: `工具 ${g.matchedTool} held-out 回归轨迹不足 (${ho.count}/${ho.need}), 疑似过拟合训练集` };
78
+ }
79
+ }
80
+ return { ok: true, matchedTool: g.matchedTool, traceCount: tb.count, heldOutCount: (heldOutTraces && Array.isArray(heldOutTraces)) ? traceBacked(g.matchedTool, heldOutTraces, 1).count : null };
81
+ }
82
+
83
+ // 技能升级专用验收 (用中自进化): 不重新做 grounding (创建时已接地验证),
84
+ // 重点防止退化: 缺段落 / 正文缩水 / 变空。返回 { ok, reason, changed }
85
+ export function verifyUpgradeSkill({ content, prevContent }) {
86
+ const c = String(content || "").replace(/^---[\s\S]*?---\s*/, ""); // 剥 frontmatter (skills.read 返回 ---meta--- 全文)
87
+ const p = String(prevContent || "").replace(/^---[\s\S]*?---\s*/, "");
88
+ // 1. 结构验收 (必须仍含 流程+验证)
89
+ const s = requiredSections(c);
90
+ if (!s.ok) return { ok: false, reason: "升级版缺必需段落: " + s.missing.join(", ") };
91
+ // 2. 正文不缩水 (去段落标题后比旧版短 = 退化)
92
+ const bodyNow = c.replace(/##\s*\S+/g, "").trim();
93
+ const bodyPrev = p.replace(/##\s*\S+/g, "").trim();
94
+ if (bodyNow.length < 20) return { ok: false, reason: "升级版内容太短" };
95
+ if (bodyPrev.length && bodyNow.length < bodyPrev.length * 0.6) {
96
+ return { ok: false, reason: "升级版正文缩水 (" + bodyNow.length + "<" + Math.ceil(bodyPrev.length * 0.6) + "), 疑似退化" };
97
+ }
98
+ const changed = bodyNow !== bodyPrev;
99
+ return { ok: true, changed };
100
+ }