ppxans-harness 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +265 -0
- package/bin/ppx-channels.js +3 -0
- package/bin/ppx-serve.js +6 -0
- package/bin/ppx.js +3 -0
- package/config/identity.md +6 -0
- package/config/ishiki.md +16 -0
- package/config/ppx.json +151 -0
- package/package.json +69 -0
- package/src/agent/context.js +179 -0
- package/src/agent/index.js +717 -0
- package/src/agent/prompts.js +107 -0
- package/src/aml-server.js +151 -0
- package/src/ans/eviction.js +144 -0
- package/src/ans/guard.js +120 -0
- package/src/ans/lifecycle.js +93 -0
- package/src/ans/proactive.js +129 -0
- package/src/ans/reward.js +112 -0
- package/src/ans/values.js +15 -0
- package/src/audit/audit-chain.js +167 -0
- package/src/audit/verifier.js +120 -0
- package/src/bus/circuit-breaker.js +115 -0
- package/src/bus/runtime-bus.js +94 -0
- package/src/channels/base.js +35 -0
- package/src/channels/feishu.js +127 -0
- package/src/channels/http.js +592 -0
- package/src/channels/index.js +110 -0
- package/src/channels/log.js +29 -0
- package/src/channels/wechat-crypto.js +74 -0
- package/src/channels/wechat.js +197 -0
- package/src/channels-cli.js +124 -0
- package/src/cli.js +120 -0
- package/src/config/channels.js +170 -0
- package/src/config/index.js +224 -0
- package/src/config/providers.js +189 -0
- package/src/config/settings.js +182 -0
- package/src/core/policy.js +272 -0
- package/src/core/trace.js +89 -0
- package/src/evolve/playbook.js +194 -0
- package/src/llm/client.js +446 -0
- package/src/llm/dsml.js +74 -0
- package/src/llm/embedder.js +35 -0
- package/src/llm/fence.js +105 -0
- package/src/llm/index.js +4 -0
- package/src/llm/retry.js +73 -0
- package/src/llm/router.js +98 -0
- package/src/mcp/client.js +375 -0
- package/src/mcp/index.js +116 -0
- package/src/memory/asset-hub.js +131 -0
- package/src/memory/canvas.js +131 -0
- package/src/memory/compaction.js +28 -0
- package/src/memory/experience.js +122 -0
- package/src/memory/fact-store.js +699 -0
- package/src/memory/failure-episode.js +99 -0
- package/src/memory/fork.js +83 -0
- package/src/memory/index.js +7 -0
- package/src/memory/l0.js +52 -0
- package/src/memory/l2.js +131 -0
- package/src/memory/l3.js +112 -0
- package/src/memory/memory-ticker.js +240 -0
- package/src/memory/session.js +398 -0
- package/src/mode/blackboard.js +49 -0
- package/src/mode/graph.js +41 -0
- package/src/mode/index.js +64 -0
- package/src/mode/legion.js +51 -0
- package/src/mode/plan-exec.js +50 -0
- package/src/mode/router.js +40 -0
- package/src/orchestrator/agent-worker.js +70 -0
- package/src/orchestrator/dag.js +83 -0
- package/src/orchestrator/index.js +2 -0
- package/src/orchestrator/legion.js +188 -0
- package/src/orchestrator/supervisor.js +177 -0
- package/src/persona/index.js +29 -0
- package/src/plugin/builtin.js +212 -0
- package/src/plugin/context.js +79 -0
- package/src/plugin/index.js +62 -0
- package/src/seam/registry.js +98 -0
- package/src/seam/shell.js +55 -0
- package/src/selfheal/evolve.js +68 -0
- package/src/selfheal/healer.js +167 -0
- package/src/selfheal/run.js +9 -0
- package/src/server.js +60 -0
- package/src/services/learning-service.js +177 -0
- package/src/services/memory-health.js +99 -0
- package/src/services/memory-service.js +160 -0
- package/src/skills/loader.js +150 -0
- package/src/skills/verify.js +100 -0
- package/src/tools/advanced.js +353 -0
- package/src/tools/builtin.js +298 -0
- package/src/tools/catalog.js +159 -0
- package/src/tools/command-guard.js +112 -0
- package/src/tools/custom.js +47 -0
- package/src/tools/delegate.js +297 -0
- package/src/tools/document.js +253 -0
- package/src/tools/governance.js +260 -0
- package/src/tools/index.js +11 -0
- package/src/tools/methods.js +178 -0
- package/src/tools/ocr.js +59 -0
- package/src/tools/seam.js +125 -0
- package/src/tools/selfmod.js +176 -0
- package/src/utils/logger.js +17 -0
- package/src/utils/pii.js +42 -0
- package/src/utils/store.js +108 -0
- package/src/utils/text.js +16 -0
- package/src/utils/trace.js +153 -0
- package/src/utils/winutf8.js +15 -0
|
@@ -0,0 +1,353 @@
|
|
|
1
|
+
// src/tools/advanced.js - 进阶工具集 (搜索 / HTTP / 定时任务)
|
|
2
|
+
// 全部零依赖: 用 Node 原生 fetch + timers
|
|
3
|
+
import fs from "node:fs";
|
|
4
|
+
import net from "node:net";
|
|
5
|
+
import dns from "node:dns/promises";
|
|
6
|
+
import path from "node:path";
|
|
7
|
+
import { ensureDir, readJson, writeJson } from "../utils/store.js";
|
|
8
|
+
|
|
9
|
+
// ---------- 网页搜索 (零依赖, 多引擎兜底: tavily/brave[有key] -> DDG) ----------
|
|
10
|
+
// 有 TAVILY_API_KEY / BRAVE_API_KEY 时优先用官方 API, 否则回退加固后的 DDG 解析
|
|
11
|
+
function _stripTags(h) { return String(h || "").replace(/<[^>]+>/g, "").replace(/&/g, "&").replace(/"/g, '"').replace(/'/g, "'").trim(); }
|
|
12
|
+
function _decodeDDGUrl(u) {
|
|
13
|
+
// DDG 结果链接是 /duckduckgo.html?uddg=<encoded>&rut=...
|
|
14
|
+
const m = String(u || "").match(/[?&]uddg=([^&]+)/);
|
|
15
|
+
if (m) { try { return decodeURIComponent(m[1]); } catch {} }
|
|
16
|
+
return u;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
async function searchWeb(query) {
|
|
20
|
+
const q = encodeURIComponent(query);
|
|
21
|
+
|
|
22
|
+
// 1. Tavily (官方 API, 需 TAVILY_API_KEY)
|
|
23
|
+
const tavilyKey = process.env.TAVILY_API_KEY;
|
|
24
|
+
if (tavilyKey) {
|
|
25
|
+
try {
|
|
26
|
+
const r = await fetch("https://api.tavily.com/search", {
|
|
27
|
+
method: "POST",
|
|
28
|
+
headers: { "Content-Type": "application/json" },
|
|
29
|
+
body: JSON.stringify({ api_key: tavilyKey, query, max_results: 5 }),
|
|
30
|
+
signal: AbortSignal.timeout(15000),
|
|
31
|
+
});
|
|
32
|
+
if (r.ok) {
|
|
33
|
+
const j = await r.json();
|
|
34
|
+
const results = (j.results || []).map(x => ({ title: x.title, url: x.url, snippet: x.content }));
|
|
35
|
+
if (results.length) return results;
|
|
36
|
+
}
|
|
37
|
+
} catch {}
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// 2. Brave (官方 API, 需 BRAVE_API_KEY)
|
|
41
|
+
const braveKey = process.env.BRAVE_API_KEY;
|
|
42
|
+
if (braveKey) {
|
|
43
|
+
try {
|
|
44
|
+
const r = await fetch(`https://api.search.brave.com/res/v1/web/search?q=${q}&count=5`, {
|
|
45
|
+
headers: { "X-Subscription-Token": braveKey, "Accept": "application/json" },
|
|
46
|
+
signal: AbortSignal.timeout(15000),
|
|
47
|
+
});
|
|
48
|
+
if (r.ok) {
|
|
49
|
+
const j = await r.json();
|
|
50
|
+
const results = (j.web?.results || []).map(x => ({ title: x.title, url: x.url, snippet: x.description }));
|
|
51
|
+
if (results.length) return results;
|
|
52
|
+
}
|
|
53
|
+
} catch {}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
// 3. DuckDuckGo HTML (免key兜底, 加固解析)
|
|
57
|
+
try {
|
|
58
|
+
const r = await fetch(`https://html.duckduckgo.com/html/?q=${q}`, {
|
|
59
|
+
headers: { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" },
|
|
60
|
+
signal: AbortSignal.timeout(15000),
|
|
61
|
+
});
|
|
62
|
+
const html = await r.text();
|
|
63
|
+
const results = [];
|
|
64
|
+
// 每次抓一个 result 块: <a class="result__a" href="...">title<\/a> ... <a class="result__snippet"...>snippet<\/a>
|
|
65
|
+
const re = /<a[^>]*class="[^"]*result__a[^"]*"[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>[\s\S]*?<a[^>]*class="[^"]*result__snippet[^"]*"[^>]*>([\s\S]*?)<\/a>/g;
|
|
66
|
+
let m;
|
|
67
|
+
while ((m = re.exec(html)) && results.length < 5) {
|
|
68
|
+
const title = _stripTags(m[2]);
|
|
69
|
+
const snippet = _stripTags(m[3]);
|
|
70
|
+
if (title) results.push({ title, url: _decodeDDGUrl(m[1]), snippet });
|
|
71
|
+
}
|
|
72
|
+
if (results.length) return results;
|
|
73
|
+
} catch {}
|
|
74
|
+
|
|
75
|
+
// 4. DuckDuckGo lite (最终兜底)
|
|
76
|
+
try {
|
|
77
|
+
const r = await fetch(`https://lite.duckduckgo.com/lite/?q=${q}`, {
|
|
78
|
+
headers: { "User-Agent": "Mozilla/5.0" },
|
|
79
|
+
signal: AbortSignal.timeout(15000),
|
|
80
|
+
});
|
|
81
|
+
const html = await r.text();
|
|
82
|
+
const results = [];
|
|
83
|
+
const re = /<a[^>]+class="result-link"[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/g;
|
|
84
|
+
let m;
|
|
85
|
+
while ((m = re.exec(html)) && results.length < 5) {
|
|
86
|
+
const title = _stripTags(m[2]);
|
|
87
|
+
if (title) results.push({ title, url: _decodeDDGUrl(m[1]), snippet: "" });
|
|
88
|
+
}
|
|
89
|
+
if (results.length) return results;
|
|
90
|
+
} catch {}
|
|
91
|
+
|
|
92
|
+
throw new Error("所有搜索源失败: 无结果");
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// ---------- HTTP 请求 (含 SSRF 防护) ----------
|
|
96
|
+
function isPrivateIP(ip) {
|
|
97
|
+
if (!ip) return false;
|
|
98
|
+
const parts = (ip.replace(/^::ffff:/, "")).split(".").map(Number);
|
|
99
|
+
if (parts.length !== 4) return false;
|
|
100
|
+
const [a, b] = parts;
|
|
101
|
+
if (a === 127) return true;
|
|
102
|
+
if (a === 10) return true;
|
|
103
|
+
if (a === 169 && b === 254) return true;
|
|
104
|
+
if (a === 172 && b >= 16 && b <= 31) return true;
|
|
105
|
+
if (a === 192 && b === 168) return true;
|
|
106
|
+
if (a === 0) return true;
|
|
107
|
+
if (a === 100 && b >= 64 && b <= 127) return true;
|
|
108
|
+
return false;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
async function assertPublicUrl(url) {
|
|
112
|
+
const u = new URL(url);
|
|
113
|
+
if (u.protocol !== "http:" && u.protocol !== "https:") throw new Error("仅允许 http/https");
|
|
114
|
+
const hostname = u.hostname.replace(/^\[|\]$/g, "");
|
|
115
|
+
if (net.isIP(hostname)) {
|
|
116
|
+
if (isPrivateIP(hostname)) throw new Error("SSRF 拒绝: 内网地址 " + hostname);
|
|
117
|
+
return;
|
|
118
|
+
}
|
|
119
|
+
const addrs = await dns.lookup(hostname, { all: true, verbatim: true });
|
|
120
|
+
for (const { address } of addrs) {
|
|
121
|
+
if (isPrivateIP(address)) throw new Error("SSRF 拒绝: " + hostname + " 解析到内网 " + address);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
// 带 SSRF 校验的安全 fetch: 手动跟随重定向, 并对每一跳(含首发)都做 assertPublicUrl 校验。
|
|
126
|
+
// fetch 默认 redirect:"follow" 会在每次跳转时重新解析 DNS —— 攻击者用一个公网 URL 302→内网
|
|
127
|
+
// (如 http://127.0.0.1:x 或 http://169.254.169.254/)即可绕过单次 assertPublicUrl。
|
|
128
|
+
// 这里改用 redirect:"manual" 逐跳校验后再继续, 堵住"302 到内网/云元数据"的绕过。
|
|
129
|
+
async function _fetchWithSsrSafe(url, { method = "GET", headers = {}, body, signal, maxRedirects = 5 } = {}) {
|
|
130
|
+
let current = url;
|
|
131
|
+
for (let hop = 0; hop <= maxRedirects; hop++) {
|
|
132
|
+
await assertPublicUrl(current);
|
|
133
|
+
const resp = await fetch(current, {
|
|
134
|
+
method,
|
|
135
|
+
headers,
|
|
136
|
+
body: body !== undefined ? (typeof body === "string" ? body : JSON.stringify(body)) : undefined,
|
|
137
|
+
redirect: "manual",
|
|
138
|
+
signal,
|
|
139
|
+
});
|
|
140
|
+
if (resp.status >= 300 && resp.status < 400 && resp.headers.has("location")) {
|
|
141
|
+
const loc = resp.headers.get("location");
|
|
142
|
+
current = new URL(loc, current).toString(); // 相对 Location 基于 current 解析成绝对 URL
|
|
143
|
+
method = "GET"; // 重定向后不再携带 body, 并退回 GET (302/303 语义)
|
|
144
|
+
body = undefined;
|
|
145
|
+
continue;
|
|
146
|
+
}
|
|
147
|
+
return resp;
|
|
148
|
+
}
|
|
149
|
+
throw new Error("SSRF 拒绝: 重定向次数超过 " + maxRedirects);
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
async function httpRequest({ url, method = "GET", headers = {}, body = null, timeout = 15000 }) {
|
|
153
|
+
const ctrl = new AbortController();
|
|
154
|
+
const timer = setTimeout(() => ctrl.abort(), timeout);
|
|
155
|
+
try {
|
|
156
|
+
const resp = await _fetchWithSsrSafe(url, {
|
|
157
|
+
method,
|
|
158
|
+
headers: { "User-Agent": "PPX-Agent/0.2", ...headers },
|
|
159
|
+
body,
|
|
160
|
+
signal: ctrl.signal,
|
|
161
|
+
});
|
|
162
|
+
const text = await resp.text();
|
|
163
|
+
return { status: resp.status, ok: resp.ok, body: text.slice(0, 20000) };
|
|
164
|
+
} finally {
|
|
165
|
+
clearTimeout(timer);
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
// ---------- 网页正文提取 (fetch_page: 抓正文转纯文本, 配合 web_search 读网页) ----------
|
|
171
|
+
function _htmlToText(html) {
|
|
172
|
+
const s0 = String(html || "");
|
|
173
|
+
// 去掉 script/style/nav/footer/header 噪音
|
|
174
|
+
let out = s0.replace(/<script[\s\S]*?<\/script>/gi, " ")
|
|
175
|
+
.replace(/<style[\s\S]*?<\/style>/gi, " ")
|
|
176
|
+
.replace(/<(script|style|nav|footer|header|iframe|form|button|svg)[^>]*>[\s\S]*?<\/\1>/gi, " ");
|
|
177
|
+
out = out.replace(/<br\s*\/?>|<\/p>|<\/div>|<\/li>|<\/h[1-6]>|<\/tr>/gi, "\n");
|
|
178
|
+
out = out.replace(/<[^>]+>/g, " "); // 去掉所有标签
|
|
179
|
+
out = out.replace(/ /gi, " ").replace(/&/gi, "&")
|
|
180
|
+
.replace(/</gi, "<").replace(/>/gi, ">")
|
|
181
|
+
.replace(/"/gi, '"').replace(/'|'/gi, "'");
|
|
182
|
+
out = out.replace(/\s+/g, " "); // 压缩空白
|
|
183
|
+
return out.trim();
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
// ---------- 定时任务 ----------
|
|
187
|
+
export class Scheduler {
|
|
188
|
+
constructor(dataDir) {
|
|
189
|
+
this.dir = path.join(dataDir, "scheduler");
|
|
190
|
+
ensureDir(this.dir);
|
|
191
|
+
this.file = path.join(this.dir, "jobs.json");
|
|
192
|
+
this.jobs = readJson(this.file, []);
|
|
193
|
+
this.timers = new Map();
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
add({ name, cron, action, type = "once" }) {
|
|
197
|
+
const job = {
|
|
198
|
+
id: "j_" + Math.random().toString(36).slice(2, 8),
|
|
199
|
+
name, cron, action, type, enabled: true, createdAt: new Date().toISOString(),
|
|
200
|
+
};
|
|
201
|
+
this.jobs.push(job);
|
|
202
|
+
writeJson(this.file, this.jobs);
|
|
203
|
+
this._schedule(job);
|
|
204
|
+
return job;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
_schedule(job) {
|
|
208
|
+
if (this.timers.has(job.id)) clearTimeout(this.timers.get(job.id));
|
|
209
|
+
// 简化 cron: 支持 "HH:MM" (每日) 或 "after:Ns" (N秒后)
|
|
210
|
+
let delayMs = null;
|
|
211
|
+
if (typeof job.cron === "string" && job.cron.startsWith("after:")) {
|
|
212
|
+
delayMs = parseInt(job.cron.split(":")[1], 10) * 1000;
|
|
213
|
+
} else if (typeof job.cron === "string" && /^\d{2}:\d{2}$/.test(job.cron)) {
|
|
214
|
+
const [h, m] = job.cron.split(":").map(Number);
|
|
215
|
+
const now = new Date();
|
|
216
|
+
const target = new Date(now); target.setHours(h, m, 0, 0);
|
|
217
|
+
if (target <= now) target.setDate(target.getDate() + 1);
|
|
218
|
+
delayMs = target - now;
|
|
219
|
+
} else if (typeof job.cron === "number") {
|
|
220
|
+
delayMs = job.cron * 1000;
|
|
221
|
+
}
|
|
222
|
+
if (delayMs === null) return;
|
|
223
|
+
const timer = setTimeout(() => this._fire(job), delayMs);
|
|
224
|
+
this.timers.set(job.id, timer);
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
async _fire(job) {
|
|
228
|
+
if (job.type === "once") { this.remove(job.id); }
|
|
229
|
+
else { this._schedule(job); } // 每日任务重新排
|
|
230
|
+
try {
|
|
231
|
+
if (typeof job.action === "function") await job.action();
|
|
232
|
+
} catch (e) { console.error("定时任务失败:", e); }
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
remove(id) {
|
|
236
|
+
if (this.timers.has(id)) { clearTimeout(this.timers.get(id)); this.timers.delete(id); }
|
|
237
|
+
this.jobs = this.jobs.filter((j) => j.id !== id);
|
|
238
|
+
writeJson(this.file, this.jobs);
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
list() { return this.jobs.map(({ id, name, cron, enabled }) => ({ id, name, cron, enabled })); }
|
|
242
|
+
|
|
243
|
+
// 清理所有定时器 (进程关停时调用, 防 daily/repeating 任务把事件循环挂住不退出)
|
|
244
|
+
shutdown() {
|
|
245
|
+
for (const [id, t] of this.timers) { try { clearTimeout(t); } catch {} }
|
|
246
|
+
this.timers.clear();
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
// 注册进阶工具
|
|
251
|
+
export function registerAdvancedTools(catalog, { dataDir, scheduler, onMemoryNote }) {
|
|
252
|
+
catalog.register({
|
|
253
|
+
name: "web_search",
|
|
254
|
+
description: "搜索互联网, 返回网页标题+链接+摘要。",
|
|
255
|
+
parameters: { type: "object", properties: { query: { type: "string" } }, required: ["query"] },
|
|
256
|
+
execute: async (args) => {
|
|
257
|
+
try {
|
|
258
|
+
const results = await searchWeb(args.query);
|
|
259
|
+
return results.map((r, i) => `${i + 1}. ${r.title}\n ${r.url}\n ${r.snippet || ""}`).join("\n");
|
|
260
|
+
} catch (e) {
|
|
261
|
+
return JSON.stringify({ error: `搜索失败: ${e.message}` });
|
|
262
|
+
}
|
|
263
|
+
},
|
|
264
|
+
});
|
|
265
|
+
|
|
266
|
+
catalog.register({
|
|
267
|
+
name: "http_request",
|
|
268
|
+
description: "发送 HTTP 请求 (GET/POST/PUT/DELETE), 返回状态码和响应体。",
|
|
269
|
+
parameters: {
|
|
270
|
+
type: "object",
|
|
271
|
+
properties: {
|
|
272
|
+
url: { type: "string" },
|
|
273
|
+
method: { type: "string", enum: ["GET", "POST", "PUT", "DELETE"] },
|
|
274
|
+
headers: { type: "object" },
|
|
275
|
+
body: { type: "string" },
|
|
276
|
+
},
|
|
277
|
+
required: ["url"],
|
|
278
|
+
},
|
|
279
|
+
execute: async (args) => {
|
|
280
|
+
try {
|
|
281
|
+
const r = await httpRequest(args);
|
|
282
|
+
return JSON.stringify({ status: r.status, ok: r.ok, body: r.body.slice(0, 5000) });
|
|
283
|
+
} catch (e) {
|
|
284
|
+
return JSON.stringify({ error: `HTTP 请求失败: ${e.message}` });
|
|
285
|
+
}
|
|
286
|
+
},
|
|
287
|
+
});
|
|
288
|
+
|
|
289
|
+
catalog.register({
|
|
290
|
+
name: "notify",
|
|
291
|
+
description: "主动向用户通道发送通知消息 (用于长任务或异步完成提醒)。",
|
|
292
|
+
parameters: { type: "object", properties: { message: { type: "string", description: "要发送的通知内容" } }, required: ["message"] },
|
|
293
|
+
execute: async (args, ctx) => {
|
|
294
|
+
const agent = ctx && ctx.agent;
|
|
295
|
+
if (agent && agent.notify) { agent.notify(args && args.message); return "notified"; }
|
|
296
|
+
return "no notify sink registered";
|
|
297
|
+
},
|
|
298
|
+
});
|
|
299
|
+
|
|
300
|
+
catalog.register({
|
|
301
|
+
name: "add_schedule",
|
|
302
|
+
description: "添加定时任务。cron 支持 'HH:MM'(每日) 或 'after:秒数'(N秒后执行一次)。",
|
|
303
|
+
parameters: {
|
|
304
|
+
type: "object",
|
|
305
|
+
properties: {
|
|
306
|
+
name: { type: "string" },
|
|
307
|
+
cron: { type: "string", description: "'HH:MM' 或 'after:60'" },
|
|
308
|
+
},
|
|
309
|
+
required: ["name", "cron"],
|
|
310
|
+
},
|
|
311
|
+
execute: async (args) => {
|
|
312
|
+
if (!scheduler) return JSON.stringify({ error: "调度器未初始化" });
|
|
313
|
+
const job = scheduler.add({ name: args.name, cron: args.cron, type: /^\d{2}:\d{2}$/.test(args.cron) ? "daily" : "once", action: () => onMemoryNote?.(`定时任务触发: ${args.name}`) });
|
|
314
|
+
return JSON.stringify({ ok: true, id: job.id, next: job.cron });
|
|
315
|
+
},
|
|
316
|
+
});
|
|
317
|
+
|
|
318
|
+
catalog.register({
|
|
319
|
+
name: "list_schedules",
|
|
320
|
+
description: "列出所有定时任务。",
|
|
321
|
+
parameters: { type: "object", properties: {} },
|
|
322
|
+
execute: async () => {
|
|
323
|
+
if (!scheduler) return "调度器未初始化";
|
|
324
|
+
return JSON.stringify(scheduler.list());
|
|
325
|
+
},
|
|
326
|
+
});
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
catalog.register({
|
|
330
|
+
name: "fetch_page",
|
|
331
|
+
description: "抓取一个网页的正文并转成纯文本 (截断到 20000 字符)。用于读文章/文档/新闻内容后回答问题。",
|
|
332
|
+
parameters: {
|
|
333
|
+
type: "object",
|
|
334
|
+
properties: { url: { type: "string", description: "要抓取的网页 URL" }, maxChars: { type: "number", description: "返回最大字符数, 默认 20000" } },
|
|
335
|
+
required: ["url"],
|
|
336
|
+
},
|
|
337
|
+
execute: async (args) => {
|
|
338
|
+
try {
|
|
339
|
+
const timeout = 15000;
|
|
340
|
+
const r = await httpRequest({ url: args.url, timeout });
|
|
341
|
+
if (!r.ok) return JSON.stringify({ error: "抓取失败 HTTP " + r.status });
|
|
342
|
+
const text = _htmlToText(r.body);
|
|
343
|
+
const max = Math.min(args.maxChars || 20000, 40000);
|
|
344
|
+
const truncated = text.length > max ? text.slice(0, max) + "\n...[已截断, 共 " + text.length + " 字符]" : text;
|
|
345
|
+
return JSON.stringify({ url: args.url, status: r.status, chars: text.length, content: truncated });
|
|
346
|
+
} catch (e) {
|
|
347
|
+
return JSON.stringify({ error: "fetch_page 失败: " + e.message });
|
|
348
|
+
}
|
|
349
|
+
},
|
|
350
|
+
});
|
|
351
|
+
|
|
352
|
+
return catalog;
|
|
353
|
+
}
|
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
// src/tools/builtin.js - 内置工具集 (皮皮虾的手脚)
|
|
2
|
+
// 文件/命令/时间/记忆查询 — 全部零依赖, 用 Node 原生
|
|
3
|
+
import fs from "node:fs";
|
|
4
|
+
import os from "node:os";
|
|
5
|
+
import path from "node:path";
|
|
6
|
+
import { execFile } from "node:child_process";
|
|
7
|
+
import { promisify } from "node:util";
|
|
8
|
+
import { scrubPII } from "../utils/pii.js";
|
|
9
|
+
import { LocalShellProvider } from "../seam/shell.js";
|
|
10
|
+
import { checkCommand, DENY_HINT } from "./command-guard.js";
|
|
11
|
+
|
|
12
|
+
const execFileP = promisify(execFile);
|
|
13
|
+
|
|
14
|
+
// 默认 shell provider (未通过 seam 注入时的兜底, 供测试/独立工具目录使用)
|
|
15
|
+
const defaultShell = new LocalShellProvider();
|
|
16
|
+
|
|
17
|
+
// 图片 MIME 表 (read_image + 多模态注入共用)
|
|
18
|
+
const IMAGE_MIME = { ".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg", ".gif": "image/gif", ".webp": "image/webp", ".bmp": "image/bmp" };
|
|
19
|
+
|
|
20
|
+
// 读图片文件转 base64 data URL (供 read_image 工具与多模态 user 消息注入复用)
|
|
21
|
+
export function imageFileToDataUrl(rootDir, p, { maxBytes = 8 * 1024 * 1024 } = {}) {
|
|
22
|
+
const fp = safePath(rootDir, p);
|
|
23
|
+
const mime = IMAGE_MIME[path.extname(fp).toLowerCase()];
|
|
24
|
+
if (!mime) throw new Error("不支持的文件类型: " + path.extname(fp));
|
|
25
|
+
if (!fs.existsSync(fp)) throw new Error("文件不存在: " + p);
|
|
26
|
+
const buf = fs.readFileSync(fp);
|
|
27
|
+
if (buf.length > maxBytes) throw new Error(`图片过大 (>${Math.round(maxBytes / 1024 / 1024)}MB)`);
|
|
28
|
+
return `data:${mime};base64,${buf.toString("base64")}`;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// ---- run_command 安全策略 (P0): 统一走命令守卫 (src/tools/command-guard.js) ----
|
|
32
|
+
// 三层防线: 用户 deny(security.deny) -> 硬黑名单(allow_all 也拦) -> 常规高危 + 前缀白名单
|
|
33
|
+
// 命令守卫的 isDeniedCommand/isAllowedCommand 兼容导出见 command-guard.js
|
|
34
|
+
|
|
35
|
+
// 安全路径: 阻止逃出工作目录 (防路径穿越)
|
|
36
|
+
// v1.0.9: 追加 realpath 校验 — 字符串前缀检查可被工作区内 symlink 指向外部绕过 (resolve 后仍在 root 内但实际文件在外部)
|
|
37
|
+
export function safePath(root, p) {
|
|
38
|
+
const resolved = path.resolve(root, p);
|
|
39
|
+
if (resolved !== root && !resolved.startsWith(root + path.sep)) {
|
|
40
|
+
throw new Error(`路径越界拒绝: ${p}`);
|
|
41
|
+
}
|
|
42
|
+
try {
|
|
43
|
+
const realRoot = fs.realpathSync(root);
|
|
44
|
+
let target = resolved;
|
|
45
|
+
if (fs.existsSync(target)) {
|
|
46
|
+
target = fs.realpathSync(target); // 存在: 直接解析真实路径
|
|
47
|
+
} else {
|
|
48
|
+
// 不存在: 用最近已存在父目录的真实路径 + 剩余部分 (新建文件场景)
|
|
49
|
+
let dir = path.dirname(target);
|
|
50
|
+
while (dir !== root && dir !== path.dirname(dir) && !fs.existsSync(dir)) dir = path.dirname(dir);
|
|
51
|
+
target = path.join(fs.realpathSync(fs.existsSync(dir) ? dir : root), path.relative(dir, resolved));
|
|
52
|
+
}
|
|
53
|
+
if (target !== realRoot && !target.startsWith(realRoot + path.sep)) {
|
|
54
|
+
throw new Error(`路径越界拒绝 (符号链接): ${p}`);
|
|
55
|
+
}
|
|
56
|
+
} catch (e) {
|
|
57
|
+
if (e && e.message && e.message.includes("路径越界拒绝")) throw e;
|
|
58
|
+
// root 不存在等边缘: 退回前缀检查 (已通过)
|
|
59
|
+
}
|
|
60
|
+
return resolved;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
// ---- code_act (CodeAct 出口): 一次提交脚本批量操作, 压 N 轮工具往返 → 1 轮 ----
|
|
64
|
+
// 安全: 默认关闭 (security.code_act), 开启后限 python/node 解释器 + 工作目录 + 超时 + PII + 黑名单扫描
|
|
65
|
+
// 相比 run_command 的增量风险: 脚本体绕过命令串黑名单, 故独立开关 + 默认关闭
|
|
66
|
+
// 沙箱加固 (进程级): 干净环境变量(剥离密钥/令牌) + node 内存上限 + 超时强杀进程树 + 输出上限
|
|
67
|
+
// 真正隔离需外部 Docker/MicroVM (见 docs/CONFIG.md), 此处为无依赖下的最大进程级约束
|
|
68
|
+
|
|
69
|
+
// 干净沙箱环境: 只保留运行必需变量, 剥离一切敏感密钥/令牌/凭证, 防脚本窃取宿主凭据
|
|
70
|
+
const SANDBOX_ENV_KEEP = /^(PATH|PATHEXT|SYSTEMROOT|WINDIR|TEMP|TMP|USERPROFILE|HOMEDRIVE|HOMEPATH|COMSPEC|OS|PROCESSOR_ARCHITECTURE|PROCESSOR_IDENTIFIER|NUMBER_OF_PROCESSORS|LANG|LC_|PYTHONIOENCODING|PYTHONPATH)$/i;
|
|
71
|
+
function sandboxEnv() {
|
|
72
|
+
const env = {};
|
|
73
|
+
for (const [k, v] of Object.entries(process.env)) {
|
|
74
|
+
if (v && SANDBOX_ENV_KEEP.test(k)) env[k] = v;
|
|
75
|
+
}
|
|
76
|
+
return env;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const CODE_ACT_MEMORY_MB = 256; // node 解释器内存上限
|
|
80
|
+
const CODE_ACT_OUTPUT_MAX = 512 * 1024; // 输出上限 512KB
|
|
81
|
+
|
|
82
|
+
export async function runCodeAct(rootDir, lang, code, timeoutMs) {
|
|
83
|
+
const isWin = process.platform === "win32";
|
|
84
|
+
const ext = lang === "node" ? "js" : "py";
|
|
85
|
+
const tmp = path.join(os.tmpdir(), `ppx_codeact_${Date.now()}_${Math.random().toString(36).slice(2, 6)}.${ext}`);
|
|
86
|
+
fs.writeFileSync(tmp, code, "utf8");
|
|
87
|
+
const interpreter = lang === "node" ? process.execPath : (isWin ? "python" : "python3");
|
|
88
|
+
// node 加内存上限; python 无等价零依赖参数 (可外接 Docker 时再限制)
|
|
89
|
+
const args = lang === "node" ? [`--max-old-space-size=${CODE_ACT_MEMORY_MB}`, tmp] : [tmp];
|
|
90
|
+
try {
|
|
91
|
+
const { stdout, stderr } = await execFileP(interpreter, args, {
|
|
92
|
+
cwd: rootDir,
|
|
93
|
+
timeout: timeoutMs || 30000,
|
|
94
|
+
maxBuffer: CODE_ACT_OUTPUT_MAX,
|
|
95
|
+
env: sandboxEnv(),
|
|
96
|
+
windowsHide: true,
|
|
97
|
+
});
|
|
98
|
+
const out = (stdout || "") + (stderr ? "\n[stderr] " + stderr : "");
|
|
99
|
+
return scrubPII(out).cleaned.slice(0, 20000) || "(无输出)";
|
|
100
|
+
} catch (e) {
|
|
101
|
+
// 超时/输出超限/内存超限的友好提示
|
|
102
|
+
if (e.killed || /timed out|ETIMEDOUT/i.test(e.message)) {
|
|
103
|
+
return JSON.stringify({ error: `code_act 执行超时 (${timeoutMs || 30000}ms), 已强制终止` });
|
|
104
|
+
}
|
|
105
|
+
return JSON.stringify({ error: e.message, code: e.code });
|
|
106
|
+
} finally {
|
|
107
|
+
try { fs.rmSync(tmp, { force: true }); } catch {}
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// 注册全部内置工具
|
|
112
|
+
export function registerBuiltinTools(catalog, { rootDir, facts, memory }) {
|
|
113
|
+
// 1. 读文件
|
|
114
|
+
catalog.register({
|
|
115
|
+
name: "read_file",
|
|
116
|
+
description: "读取文件内容。返回文件文本。",
|
|
117
|
+
parameters: {
|
|
118
|
+
type: "object",
|
|
119
|
+
properties: { path: { type: "string", description: "文件路径 (相对工作目录)" } },
|
|
120
|
+
required: ["path"],
|
|
121
|
+
},
|
|
122
|
+
execute: async (args) => {
|
|
123
|
+
const p = safePath(rootDir, args.path);
|
|
124
|
+
if (!fs.existsSync(p)) return JSON.stringify({ error: `文件不存在: ${args.path}` });
|
|
125
|
+
const content = fs.readFileSync(p, "utf8");
|
|
126
|
+
// v1.0.9: 输出 PII 脱敏 (与 run_command/code_act 一致, 文件可能含密钥/手机号)
|
|
127
|
+
return scrubPII(content).cleaned.slice(0, 20000);
|
|
128
|
+
},
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
// 2. 写文件
|
|
132
|
+
catalog.register({
|
|
133
|
+
name: "write_file",
|
|
134
|
+
description: "写入文件 (覆盖)。可用于创建/修改文件。",
|
|
135
|
+
parameters: {
|
|
136
|
+
type: "object",
|
|
137
|
+
properties: {
|
|
138
|
+
path: { type: "string", description: "文件路径" },
|
|
139
|
+
content: { type: "string", description: "要写入的内容" },
|
|
140
|
+
},
|
|
141
|
+
required: ["path", "content"],
|
|
142
|
+
},
|
|
143
|
+
execute: async (args) => {
|
|
144
|
+
// v1.0.9: 写入内容上限 512KB (防撑爆磁盘); 路径是目录时给友好错误
|
|
145
|
+
const content = String(args.content ?? "");
|
|
146
|
+
if (content.length > 512 * 1024) return JSON.stringify({ error: `写入内容过大 (>512KB, 当前 ${content.length} 字符)` });
|
|
147
|
+
const p = safePath(rootDir, args.path);
|
|
148
|
+
if (fs.existsSync(p) && fs.statSync(p).isDirectory()) {
|
|
149
|
+
return JSON.stringify({ error: `目标是目录: ${args.path}` });
|
|
150
|
+
}
|
|
151
|
+
try {
|
|
152
|
+
fs.mkdirSync(path.dirname(p), { recursive: true });
|
|
153
|
+
fs.writeFileSync(p, content, "utf8");
|
|
154
|
+
return JSON.stringify({ ok: true, bytes: Buffer.byteLength(content) });
|
|
155
|
+
} catch (e) {
|
|
156
|
+
return JSON.stringify({ error: `写入失败: ${e.message}` });
|
|
157
|
+
}
|
|
158
|
+
},
|
|
159
|
+
});
|
|
160
|
+
|
|
161
|
+
// 3. 列目录
|
|
162
|
+
catalog.register({
|
|
163
|
+
name: "list_dir",
|
|
164
|
+
description: "列出目录内容 (文件名列表)。",
|
|
165
|
+
parameters: {
|
|
166
|
+
type: "object",
|
|
167
|
+
properties: { path: { type: "string", description: "目录路径, 默认工作目录" } },
|
|
168
|
+
},
|
|
169
|
+
execute: async (args) => {
|
|
170
|
+
const p = safePath(rootDir, args.path || ".");
|
|
171
|
+
if (!fs.existsSync(p) || !fs.statSync(p).isDirectory()) {
|
|
172
|
+
return JSON.stringify({ error: `目录不存在: ${args.path || "."}` });
|
|
173
|
+
}
|
|
174
|
+
const items = fs.readdirSync(p).map((f) => {
|
|
175
|
+
const fp = path.join(p, f);
|
|
176
|
+
const st = fs.statSync(fp);
|
|
177
|
+
return `${st.isDirectory() ? "[D]" : "[F]"} ${f}`;
|
|
178
|
+
});
|
|
179
|
+
return items.join("\n");
|
|
180
|
+
},
|
|
181
|
+
});
|
|
182
|
+
|
|
183
|
+
// 4. 执行命令 (安全: 限制在允许目录, 超时)
|
|
184
|
+
catalog.register({
|
|
185
|
+
name: "run_command",
|
|
186
|
+
description: "执行 shell 命令并返回输出。只能在工作目录内执行, 有超时。",
|
|
187
|
+
parameters: {
|
|
188
|
+
type: "object",
|
|
189
|
+
properties: { command: { type: "string", description: "要执行的命令" } },
|
|
190
|
+
required: ["command"],
|
|
191
|
+
},
|
|
192
|
+
execute: async (args, ctx) => {
|
|
193
|
+
const cmd = String(args.command || "").trim();
|
|
194
|
+
if (!cmd) return JSON.stringify({ error: "空命令" });
|
|
195
|
+
const opts = (ctx && ctx.agent && ctx.agent.config && ctx.agent.config.security) || {};
|
|
196
|
+
const guard = checkCommand(cmd, opts);
|
|
197
|
+
if (!guard.ok) {
|
|
198
|
+
return JSON.stringify({ error: guard.reason + DENY_HINT });
|
|
199
|
+
}
|
|
200
|
+
// 通过 shell seam 调用 (可替换 provider: 本地/沙箱/Docker), 换 provider 即换执行环境
|
|
201
|
+
const shell = (ctx?.agent?.ctx && ctx.agent.ctx.consume("shell")) || defaultShell;
|
|
202
|
+
const r = await shell.exec(cmd, { cwd: rootDir, timeoutMs: opts.command_timeout_ms || 30000 });
|
|
203
|
+
if (!r.ok) return JSON.stringify({ error: r.stderr, code: r.code });
|
|
204
|
+
const out = r.stdout + (r.stderr ? "\n[stderr] " + r.stderr : "");
|
|
205
|
+
return scrubPII(out).cleaned.slice(0, 20000) || "(无输出)";
|
|
206
|
+
},
|
|
207
|
+
});
|
|
208
|
+
|
|
209
|
+
// 4.5 code_act (CodeAct 出口): 脚本批量操作, 压 N 轮工具往返 → 1 轮
|
|
210
|
+
catalog.register({
|
|
211
|
+
name: "code_act",
|
|
212
|
+
description: "用 Python/Node 脚本一次性完成多个操作(读文件/处理数据/写结果), 用 print/console.log 输出结果。默认关闭, 需 security.code_act=true。",
|
|
213
|
+
parameters: {
|
|
214
|
+
type: "object",
|
|
215
|
+
properties: {
|
|
216
|
+
language: { type: "string", enum: ["python", "node"], description: "脚本语言" },
|
|
217
|
+
code: { type: "string", description: "脚本内容" },
|
|
218
|
+
},
|
|
219
|
+
required: ["language", "code"],
|
|
220
|
+
},
|
|
221
|
+
execute: async (args, ctx) => {
|
|
222
|
+
const sec = (ctx && ctx.agent && ctx.agent.config && ctx.agent.config.security) || {};
|
|
223
|
+
if (!sec.allow_all && !sec.code_act) {
|
|
224
|
+
return JSON.stringify({ error: "code_act 未开启: 在 security 设置 code_act=true (或 allow_all=true)" });
|
|
225
|
+
}
|
|
226
|
+
const lang = String(args.language || "").toLowerCase();
|
|
227
|
+
if (!["python", "node"].includes(lang)) return JSON.stringify({ error: "language 仅支持 python/node" });
|
|
228
|
+
const code = String(args.code || "");
|
|
229
|
+
if (!code) return JSON.stringify({ error: "空代码" });
|
|
230
|
+
// code_act 是脚本体: 只做 deny 检查 (硬黑名单 + 用户 deny + 常规高危), 不做前缀白名单 (脚本无"命令前缀")
|
|
231
|
+
const guard = checkCommand(code, { ...sec, allowAll: true });
|
|
232
|
+
if (!guard.ok) return JSON.stringify({ error: guard.reason + DENY_HINT });
|
|
233
|
+
return runCodeAct(rootDir, lang, code, sec.command_timeout_ms);
|
|
234
|
+
},
|
|
235
|
+
});
|
|
236
|
+
|
|
237
|
+
// 5. 当前时间
|
|
238
|
+
catalog.register({
|
|
239
|
+
name: "get_time",
|
|
240
|
+
description: "获取当前日期和时间。",
|
|
241
|
+
parameters: { type: "object", properties: {} },
|
|
242
|
+
execute: async () => new Date().toLocaleString("zh-CN", { timeZone: "Asia/Shanghai" }),
|
|
243
|
+
});
|
|
244
|
+
|
|
245
|
+
// 6. 记忆查询 (皮皮虾自己查记忆)
|
|
246
|
+
catalog.register({
|
|
247
|
+
name: "memory_search",
|
|
248
|
+
description: "搜索皮皮虾的记忆库, 返回相关事实。",
|
|
249
|
+
parameters: {
|
|
250
|
+
type: "object",
|
|
251
|
+
properties: { query: { type: "string", description: "要搜的内容" }, limit: { type: "number" } },
|
|
252
|
+
required: ["query"],
|
|
253
|
+
},
|
|
254
|
+
execute: async (args) => {
|
|
255
|
+
if (!facts) return JSON.stringify({ error: "记忆未初始化" });
|
|
256
|
+
const results = facts.query(args.query, { limit: args.limit || 5 });
|
|
257
|
+
return results.length
|
|
258
|
+
? results.map((r) => `- [${r.score}] ${r.content}`).join("\n")
|
|
259
|
+
: "(无匹配记忆)";
|
|
260
|
+
},
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
// 7. 记住新事实
|
|
264
|
+
catalog.register({
|
|
265
|
+
name: "memory_add",
|
|
266
|
+
description: "把一条重要信息写进皮皮虾的长期记忆。",
|
|
267
|
+
parameters: {
|
|
268
|
+
type: "object",
|
|
269
|
+
properties: { content: { type: "string", description: "要记住的内容" } },
|
|
270
|
+
required: ["content"],
|
|
271
|
+
},
|
|
272
|
+
execute: async (args) => {
|
|
273
|
+
if (!facts) return JSON.stringify({ error: "记忆未初始化" });
|
|
274
|
+
const f = facts.add(args.content, { source: "agent-self" });
|
|
275
|
+
return JSON.stringify({ ok: true, id: f.id });
|
|
276
|
+
},
|
|
277
|
+
});
|
|
278
|
+
|
|
279
|
+
// 8. 读图片 (多模态): 返回 base64 data URL, 供多模态模型视觉理解
|
|
280
|
+
catalog.register({
|
|
281
|
+
name: "read_image",
|
|
282
|
+
description: "读取图片文件, 返回 base64 data URL 供多模态模型理解图片内容 (需配置支持视觉的模型, 如 gpt-4o/qwen-vl/glm-4v)。",
|
|
283
|
+
parameters: {
|
|
284
|
+
type: "object",
|
|
285
|
+
properties: { path: { type: "string", description: "图片文件路径 (相对工作目录)" } },
|
|
286
|
+
required: ["path"],
|
|
287
|
+
},
|
|
288
|
+
execute: async (args) => {
|
|
289
|
+
try {
|
|
290
|
+
return imageFileToDataUrl(rootDir, args.path);
|
|
291
|
+
} catch (e) {
|
|
292
|
+
return JSON.stringify({ error: e.message });
|
|
293
|
+
}
|
|
294
|
+
},
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
return catalog;
|
|
298
|
+
}
|