ppxans-harness 2.4.0 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -201
- package/README.md +218 -265
- package/bin/ppx-channels.js +2 -2
- package/bin/ppx-serve.js +5 -5
- package/bin/ppx-setup.js +124 -0
- package/bin/ppx-web.js +140 -0
- package/bin/ppx.js +2 -2
- package/config/identity.md +6 -6
- package/config/ishiki.md +16 -16
- package/config/ppx.json +15 -151
- package/config/ppx.json.example +143 -0
- package/package.json +17 -10
- package/skills/.usage.json +6 -0
- package/skills/agent-professional-training/SKILL.md +94 -0
- package/skills/brainstorm/SKILL.md +24 -0
- package/skills/cupid-lover-comms/SKILL.md +37 -0
- package/skills/debug/SKILL.md +26 -0
- package/skills/plan/SKILL.md +25 -0
- package/skills/ponytail/SKILL.md +25 -0
- package/skills/ppx-memory/SKILL.md +91 -0
- package/skills/ppx-memory/scripts/cli.js +192 -0
- package/skills/ppx-memory/scripts/experience.js +133 -0
- package/skills/ppx-memory/scripts/fact-store.js +842 -0
- package/skills/ppx-memory/scripts/l0.js +52 -0
- package/skills/ppx-memory/scripts/l2.js +146 -0
- package/skills/ppx-memory/scripts/l3.js +112 -0
- package/skills/ppx-memory/scripts/memory-ticker.js +238 -0
- package/skills/ppx-memory/scripts/pii.js +42 -0
- package/skills/ppx-memory/scripts/schema.js +80 -0
- package/skills/ppx-memory/scripts/session.js +398 -0
- package/skills/ppx-memory/scripts/similarity.js +43 -0
- package/skills/ppx-memory/scripts/store.js +116 -0
- package/skills/ppx-memory/scripts/wal.js +38 -0
- package/skills/ppx-selfheal/SKILL.md +24 -0
- package/skills/ppx-selfheal/scripts/cli.js +80 -0
- package/skills/ppx-selfheal/scripts/healer.js +184 -0
- package/skills/ppx-selfheal/scripts/logger.js +17 -0
- package/skills/ppx-selfheal/scripts/store.js +116 -0
- package/skills/prompt-depth-kit/SKILL.md +28 -0
- package/skills/session-naming/SKILL.md +36 -0
- package/skills/verify/SKILL.md +25 -0
- package/src/agent/index.js +1340 -717
- package/src/agent/prompts.js +47 -3
- package/src/aml-server.js +197 -151
- package/src/ans/eviction.js +123 -143
- package/src/ans/guard.js +159 -120
- package/src/ans/lifecycle.js +96 -93
- package/src/ans/proactive.js +112 -129
- package/src/ans/reward.js +95 -111
- package/src/ans/values.js +15 -15
- package/src/audit/audit-chain.js +43 -7
- package/src/audit/verifier.js +157 -120
- package/src/bus/circuit-breaker.js +9 -1
- package/src/bus/runtime-bus.js +107 -93
- package/src/channels/base.js +57 -34
- package/src/channels/feishu.js +118 -126
- package/src/channels/http.js +1090 -592
- package/src/channels/index.js +111 -110
- package/src/channels/log.js +29 -29
- package/src/channels/wechat-crypto.js +73 -74
- package/src/channels/wechat.js +191 -197
- package/src/channels/workspace.js +94 -0
- package/src/channels-cli.js +126 -124
- package/src/cli.js +129 -120
- package/src/commands/index.js +142 -0
- package/src/config/channels.js +137 -170
- package/src/config/index.js +277 -224
- package/src/config/placeholder.js +31 -0
- package/src/config/providers.js +148 -188
- package/src/config/settings.js +156 -182
- package/src/core/policy.js +69 -18
- package/src/core/trace.js +8 -6
- package/src/edit/editblock.js +266 -0
- package/src/edit/snapshot.js +67 -0
- package/src/evidence/index.js +153 -0
- package/src/evolve/playbook.js +9 -10
- package/src/hooks/index.js +113 -0
- package/src/llm/client.js +188 -446
- package/src/llm/dsml.js +74 -74
- package/src/llm/embedder.js +41 -35
- package/src/llm/fence.js +52 -105
- package/src/llm/index.js +4 -4
- package/src/llm/local-embedder.js +94 -0
- package/src/llm/presets.js +113 -0
- package/src/llm/pricing.js +93 -0
- package/src/llm/retry.js +73 -73
- package/src/llm/router.js +92 -97
- package/src/mcp/admin.js +326 -0
- package/src/mcp/client.js +487 -375
- package/src/mcp/http.js +203 -0
- package/src/mcp/index.js +116 -116
- package/src/mcp/server.js +392 -0
- package/src/mcp/tasks.js +133 -0
- package/src/memory/asset-hub.js +6 -11
- package/src/memory/canvas.js +2 -5
- package/src/memory/compaction.js +28 -28
- package/src/memory/experience.js +133 -122
- package/src/memory/fact-store.js +914 -698
- package/src/memory/failure-episode.js +20 -11
- package/src/memory/fork.js +17 -8
- package/src/memory/index.js +8 -6
- package/src/memory/l0.js +53 -52
- package/src/memory/l2.js +145 -130
- package/src/memory/l3.js +111 -111
- package/src/memory/legion-board.js +71 -0
- package/src/memory/memory-ticker.js +239 -240
- package/src/memory/session.js +398 -397
- package/src/memory/sqlite-store.js +581 -0
- package/src/mode/blackboard.js +49 -49
- package/src/mode/graph.js +42 -41
- package/src/mode/index.js +64 -64
- package/src/mode/legion.js +54 -51
- package/src/mode/plan-exec.js +50 -50
- package/src/mode/router.js +28 -40
- package/src/orchestrator/agent-worker.js +69 -69
- package/src/orchestrator/dag.js +90 -83
- package/src/orchestrator/experts.js +76 -0
- package/src/orchestrator/index.js +1 -1
- package/src/orchestrator/legion.js +179 -187
- package/src/orchestrator/supervisor.js +6 -8
- package/src/permissions/index.js +378 -0
- package/src/persona/index.js +28 -29
- package/src/plugin/builtin.js +313 -212
- package/src/plugin/context.js +80 -79
- package/src/plugin/index.js +62 -62
- package/src/plugin/v3.js +73 -0
- package/src/protocol/index.js +148 -0
- package/src/repomap/index.js +309 -0
- package/src/review/index.js +393 -0
- package/src/seam/registry.js +3 -0
- package/src/seam/shell.js +55 -55
- package/src/security/injection.js +79 -0
- package/src/selfheal/evolve.js +67 -67
- package/src/selfheal/healer.js +184 -167
- package/src/selfheal/run.js +9 -9
- package/src/server.js +63 -60
- package/src/services/diagnose.js +180 -0
- package/src/services/learning-service.js +9 -0
- package/src/services/memory-health.js +34 -6
- package/src/services/memory-service.js +42 -9
- package/src/services/triage.js +138 -0
- package/src/session/parts.js +76 -0
- package/src/session/projection.js +73 -0
- package/src/session/rollout.js +54 -0
- package/src/session/turn.js +137 -0
- package/src/skills/lint.js +72 -0
- package/src/skills/loader.js +231 -150
- package/src/skills/search.js +58 -0
- package/src/skills/verify.js +95 -100
- package/src/tools/advanced.js +388 -352
- package/src/tools/builtin.js +384 -297
- package/src/tools/catalog.js +283 -159
- package/src/tools/command-guard.js +112 -112
- package/src/tools/custom.js +47 -47
- package/src/tools/delegate.js +383 -297
- package/src/tools/document.js +254 -253
- package/src/tools/git.js +151 -0
- package/src/tools/governance.js +47 -20
- package/src/tools/index.js +16 -11
- package/src/tools/methods.js +178 -178
- package/src/tools/ocr.js +59 -59
- package/src/tools/sandbox-worker.js +40 -0
- package/src/tools/sandbox.js +92 -0
- package/src/tools/seam.js +162 -125
- package/src/tools/selfmod.js +196 -176
- package/src/tools/v3.js +225 -0
- package/src/tools/vad.js +176 -0
- package/src/tools/voice.js +238 -0
- package/src/utils/async.js +14 -0
- package/src/utils/config-file.js +53 -0
- package/src/utils/crashguard.js +88 -0
- package/src/utils/http.js +53 -0
- package/src/utils/id.js +8 -0
- package/src/utils/json-state.js +33 -0
- package/src/utils/logger.js +17 -17
- package/src/utils/ndjson.js +25 -0
- package/src/utils/pii.js +42 -42
- package/src/utils/rate-limit.js +50 -0
- package/src/utils/schema.js +80 -0
- package/src/utils/similarity.js +43 -0
- package/src/utils/store.js +170 -108
- package/src/utils/text.js +15 -15
- package/src/utils/trace.js +153 -153
- package/src/utils/wal.js +39 -0
- package/src/utils/winutf8.js +16 -15
- package/src/wiki/index.js +170 -0
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
// src/llm/pricing.js - 模型价格表 + 成本折算 (增强框架第 8 条: "预算控制"的最后一环)
|
|
2
|
+
// 背景: usageStats 已聚合 token 数 (calls/tokens/byModel), 但没有金额折算与支出上限 ——
|
|
3
|
+
// token 数不等于成本, 不同模型价差 100 倍, "预算控制"必须落在金额上。
|
|
4
|
+
// 设计:
|
|
5
|
+
// - 内置常用模型价格表 (USD / 1M tokens, prompt + completion), 前缀匹配取最长者
|
|
6
|
+
// - 云厂商价格随时变动: 内置表只是"开箱即用的估算快照", 精确控费用 config.budget.model_prices 覆盖
|
|
7
|
+
// - 未知模型 → 返回 null (cost 记 0), **不编数字**; 想纳入预算控制就显式配置价格
|
|
8
|
+
// - 只有 total_tokens 无拆分时, 全部按 completion 价计 (预算取保守侧, 宁高估不高估)
|
|
9
|
+
// 零依赖, 纯函数, 可独立测试。
|
|
10
|
+
|
|
11
|
+
// USD per 1M tokens。数字为各厂商公开目录价的历史快照 (2026-10), 仅供预算估算。
|
|
12
|
+
// 匹配规则: model 小写后按前缀匹配, 取最长命中 (如 glm-4-flash 命中 flash 行而非 glm-4 行)。
|
|
13
|
+
const PRICES = [
|
|
14
|
+
// 智谱
|
|
15
|
+
{ prefix: "glm-4-flash", prompt: 0, completion: 0 }, // 免费档
|
|
16
|
+
{ prefix: "glm-4.5", prompt: 0.6, completion: 2.2 },
|
|
17
|
+
{ prefix: "glm-4.6", prompt: 0.6, completion: 2.2 },
|
|
18
|
+
{ prefix: "glm-4", prompt: 0.5, completion: 1.5 },
|
|
19
|
+
// DeepSeek
|
|
20
|
+
{ prefix: "deepseek-chat", prompt: 0.27, completion: 1.1 },
|
|
21
|
+
{ prefix: "deepseek-reasoner", prompt: 0.55, completion: 2.19 },
|
|
22
|
+
// OpenAI
|
|
23
|
+
{ prefix: "gpt-4o-mini", prompt: 0.15, completion: 0.6 },
|
|
24
|
+
{ prefix: "gpt-4o", prompt: 2.5, completion: 10 },
|
|
25
|
+
{ prefix: "gpt-4.1-mini", prompt: 0.4, completion: 1.6 },
|
|
26
|
+
{ prefix: "gpt-4.1", prompt: 2, completion: 8 },
|
|
27
|
+
// Anthropic (前缀族: 3.x 与 4.x 同档从宽)
|
|
28
|
+
{ prefix: "claude-opus", prompt: 15, completion: 75 },
|
|
29
|
+
{ prefix: "claude-sonnet", prompt: 3, completion: 15 },
|
|
30
|
+
{ prefix: "claude-haiku", prompt: 1, completion: 5 },
|
|
31
|
+
// Gemini
|
|
32
|
+
{ prefix: "gemini-2.5-pro", prompt: 1.25, completion: 10 },
|
|
33
|
+
{ prefix: "gemini-2.5-flash", prompt: 0.3, completion: 2.5 },
|
|
34
|
+
// 阿里 / 月之暗面
|
|
35
|
+
{ prefix: "qwen-flash", prompt: 0.05, completion: 0.4 },
|
|
36
|
+
{ prefix: "qwen-plus", prompt: 0.4, completion: 1.2 },
|
|
37
|
+
{ prefix: "qwen-max", prompt: 1.6, completion: 6.4 },
|
|
38
|
+
{ prefix: "kimi-k2", prompt: 0.6, completion: 2.5 },
|
|
39
|
+
];
|
|
40
|
+
|
|
41
|
+
// 规范化一条价格: {prompt, completion} 均为非负有限数字, 否则视为无效 (返回 null)
|
|
42
|
+
function normPrice(v) {
|
|
43
|
+
if (!v || typeof v !== "object") return null;
|
|
44
|
+
const p = Number(v.prompt), c = Number(v.completion);
|
|
45
|
+
if (!Number.isFinite(p) || !Number.isFinite(c) || p < 0 || c < 0) return null;
|
|
46
|
+
return { prompt: p, completion: c };
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// 解析某模型的价格。overrides: config.budget?.model_prices, 形如
|
|
50
|
+
// { "glm-4-flash": { prompt: 0, completion: 0 }, "my-private-model": { prompt: 1, completion: 3 } }
|
|
51
|
+
// 优先级: overrides 精确命中 > overrides 前缀命中(最长) > 内置表前缀命中(最长) > null
|
|
52
|
+
export function resolvePrice(model, overrides) {
|
|
53
|
+
const m = String(model || "").toLowerCase().trim();
|
|
54
|
+
if (!m) return null;
|
|
55
|
+
if (overrides && typeof overrides === "object" && !Array.isArray(overrides)) {
|
|
56
|
+
// 精确命中最优先
|
|
57
|
+
const exact = normPrice(overrides[m]) || normPrice(overrides[String(model || "").trim()]);
|
|
58
|
+
if (exact) return exact;
|
|
59
|
+
// 前缀命中取最长
|
|
60
|
+
let best = null, bestLen = -1;
|
|
61
|
+
for (const k of Object.keys(overrides)) {
|
|
62
|
+
const kk = String(k).toLowerCase();
|
|
63
|
+
if (kk && m.startsWith(kk) && kk.length > bestLen) {
|
|
64
|
+
const v = normPrice(overrides[k]);
|
|
65
|
+
if (v) { best = v; bestLen = kk.length; }
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
if (best) return best;
|
|
69
|
+
}
|
|
70
|
+
let hit = null, hitLen = -1;
|
|
71
|
+
for (const e of PRICES) {
|
|
72
|
+
if (m.startsWith(e.prefix) && e.prefix.length > hitLen) { hit = e; hitLen = e.prefix.length; }
|
|
73
|
+
}
|
|
74
|
+
return hit ? { prompt: hit.prompt, completion: hit.completion } : null;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// 折算一笔 usage 的成本 (USD)。usage: { prompt_tokens, completion_tokens } 或仅 { total_tokens }。
|
|
78
|
+
// 无价格 → 0 (调用方无从区分"免费"与"未知", 需要区分时自行调 resolvePrice)。
|
|
79
|
+
export function estimateCost(model, usage, overrides) {
|
|
80
|
+
const p = resolvePrice(model, overrides);
|
|
81
|
+
if (!p) return 0;
|
|
82
|
+
const pt = usage?.prompt_tokens, ct = usage?.completion_tokens;
|
|
83
|
+
let prompt, completion;
|
|
84
|
+
if (pt == null && ct == null) {
|
|
85
|
+
// 只有总量 (部分本地推理后端不拆分): 全按 completion 价计 —— 预算保守侧
|
|
86
|
+
prompt = 0;
|
|
87
|
+
completion = Number(usage?.total_tokens) || 0;
|
|
88
|
+
} else {
|
|
89
|
+
prompt = Number(pt) || 0;
|
|
90
|
+
completion = Number(ct) || 0;
|
|
91
|
+
}
|
|
92
|
+
return (prompt * p.prompt + completion * p.completion) / 1e6;
|
|
93
|
+
}
|
package/src/llm/retry.js
CHANGED
|
@@ -1,73 +1,73 @@
|
|
|
1
|
-
// src/llm/retry.js - 错误重试内核 (吸收 OpenClaw retry/operation-retry 精华)
|
|
2
|
-
// 设计要点:
|
|
3
|
-
// - 瞬态分类: 429/5xx/timeout/网络错误 才重试; 400/401/403/404 等客户端错误立即失败
|
|
4
|
-
// - 指数退避 + full jitter (避免惊群); 尊重 Retry-After 头
|
|
5
|
-
// - 可取消 (AbortSignal)
|
|
6
|
-
// - LLM 失败不 throw 的契约由上层 (provider 回退) 承担, 本模块只负责"单次调用内"的重试
|
|
7
|
-
|
|
8
|
-
// 从错误对象/消息提取 HTTP 状态码
|
|
9
|
-
export function httpStatusOf(err) {
|
|
10
|
-
if (!err) return null;
|
|
11
|
-
if (typeof err.status === "number") return err.status;
|
|
12
|
-
if (typeof err.statusCode === "number") return err.statusCode;
|
|
13
|
-
const m = String(err.message || err).match(/\b(4\d\d|5\d\d)\b/);
|
|
14
|
-
return m ? Number(m[1]) : null;
|
|
15
|
-
}
|
|
16
|
-
|
|
17
|
-
// 瞬态分类: 值得重试的错误 (openclaw operation-retry: 429/5xx/ENOTFOUND/timeout/fetch failed)
|
|
18
|
-
export function isTransientError(err) {
|
|
19
|
-
// v1.0.9: AbortError (用户主动取消 / 内部超时中止) 一律不重试 — 原把 message "aborted" 判瞬态, 取消后仍退避重试
|
|
20
|
-
if (err && (err.name === "AbortError" || err.code === "ABORT_ERR")) return false;
|
|
21
|
-
const status = httpStatusOf(err);
|
|
22
|
-
if (status) {
|
|
23
|
-
if (status === 429 || status >= 500) return true;
|
|
24
|
-
if (status >= 400 && status < 500) return false; // 客户端错误不重试
|
|
25
|
-
}
|
|
26
|
-
const msg = String(err?.message || err || "");
|
|
27
|
-
if (/timeout|timed?\s*out|ETIMEDOUT|ECONNRESET|ECONNREFUSED|ENOTFOUND|EAI_AGAIN|fetch failed|network|socket hang up|undici/i.test(msg)) return true;
|
|
28
|
-
return false;
|
|
29
|
-
}
|
|
30
|
-
|
|
31
|
-
// Retry-After (秒), 无则 null (上限 30s 防恶意长退避)
|
|
32
|
-
export function retryAfterSeconds(err) {
|
|
33
|
-
const raw = err?.headers?.get?.("retry-after") ?? err?.retryAfter ?? err?.retry_after;
|
|
34
|
-
const n = Number(raw);
|
|
35
|
-
if (Number.isFinite(n) && n > 0) return Math.min(n, 30);
|
|
36
|
-
return null;
|
|
37
|
-
}
|
|
38
|
-
|
|
39
|
-
// 指数退避 + full jitter
|
|
40
|
-
export function backoffMs(attempt, { baseMs = 500, factor = 2, maxMs = 10000 } = {}) {
|
|
41
|
-
const exp = Math.min(maxMs, baseMs * Math.pow(factor, attempt));
|
|
42
|
-
return Math.floor(exp * (0.5 + Math.random() * 0.5));
|
|
43
|
-
}
|
|
44
|
-
|
|
45
|
-
// 可取消 sleep (AbortSignal)
|
|
46
|
-
export function sleep(ms, signal) {
|
|
47
|
-
return new Promise((resolve, reject) => {
|
|
48
|
-
if (signal?.aborted) return reject(abortError());
|
|
49
|
-
const t = setTimeout(() => { cleanup(); resolve(); }, ms);
|
|
50
|
-
const onAbort = () => { clearTimeout(t); cleanup(); reject(abortError()); };
|
|
51
|
-
if (signal) signal.addEventListener("abort", onAbort, { once: true });
|
|
52
|
-
function cleanup() { if (signal) signal.removeEventListener("abort", onAbort); }
|
|
53
|
-
});
|
|
54
|
-
}
|
|
55
|
-
|
|
56
|
-
function abortError() { const e = new Error("aborted"); e.name = "AbortError"; return e; }
|
|
57
|
-
|
|
58
|
-
// 重试执行器: fn 抛瞬态错误时按退避重试, 非瞬态直接抛
|
|
59
|
-
export async function withRetry(fn, { maxRetries = 3, baseMs = 500, factor = 2, maxMs = 10000, signal = null, shouldRetry = isTransientError } = {}) {
|
|
60
|
-
let lastErr;
|
|
61
|
-
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
62
|
-
try {
|
|
63
|
-
return await fn();
|
|
64
|
-
} catch (e) {
|
|
65
|
-
lastErr = e;
|
|
66
|
-
if (attempt >= maxRetries || !shouldRetry(e)) throw e;
|
|
67
|
-
const ra = retryAfterSeconds(e);
|
|
68
|
-
const ms = ra != null ? ra * 1000 : backoffMs(attempt, { baseMs, factor, maxMs });
|
|
69
|
-
await sleep(ms, signal);
|
|
70
|
-
}
|
|
71
|
-
}
|
|
72
|
-
throw lastErr;
|
|
73
|
-
}
|
|
1
|
+
// src/llm/retry.js - 错误重试内核 (吸收 OpenClaw retry/operation-retry 精华)
|
|
2
|
+
// 设计要点:
|
|
3
|
+
// - 瞬态分类: 429/5xx/timeout/网络错误 才重试; 400/401/403/404 等客户端错误立即失败
|
|
4
|
+
// - 指数退避 + full jitter (避免惊群); 尊重 Retry-After 头
|
|
5
|
+
// - 可取消 (AbortSignal)
|
|
6
|
+
// - LLM 失败不 throw 的契约由上层 (provider 回退) 承担, 本模块只负责"单次调用内"的重试
|
|
7
|
+
|
|
8
|
+
// 从错误对象/消息提取 HTTP 状态码
|
|
9
|
+
export function httpStatusOf(err) {
|
|
10
|
+
if (!err) return null;
|
|
11
|
+
if (typeof err.status === "number") return err.status;
|
|
12
|
+
if (typeof err.statusCode === "number") return err.statusCode;
|
|
13
|
+
const m = String(err.message || err).match(/\b(4\d\d|5\d\d)\b/);
|
|
14
|
+
return m ? Number(m[1]) : null;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
// 瞬态分类: 值得重试的错误 (openclaw operation-retry: 429/5xx/ENOTFOUND/timeout/fetch failed)
|
|
18
|
+
export function isTransientError(err) {
|
|
19
|
+
// v1.0.9: AbortError (用户主动取消 / 内部超时中止) 一律不重试 — 原把 message "aborted" 判瞬态, 取消后仍退避重试
|
|
20
|
+
if (err && (err.name === "AbortError" || err.code === "ABORT_ERR")) return false;
|
|
21
|
+
const status = httpStatusOf(err);
|
|
22
|
+
if (status) {
|
|
23
|
+
if (status === 429 || status >= 500) return true;
|
|
24
|
+
if (status >= 400 && status < 500) return false; // 客户端错误不重试
|
|
25
|
+
}
|
|
26
|
+
const msg = String(err?.message || err || "");
|
|
27
|
+
if (/timeout|timed?\s*out|ETIMEDOUT|ECONNRESET|ECONNREFUSED|ENOTFOUND|EAI_AGAIN|fetch failed|network|socket hang up|undici/i.test(msg)) return true;
|
|
28
|
+
return false;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// Retry-After (秒), 无则 null (上限 30s 防恶意长退避)
|
|
32
|
+
export function retryAfterSeconds(err) {
|
|
33
|
+
const raw = err?.headers?.get?.("retry-after") ?? err?.retryAfter ?? err?.retry_after;
|
|
34
|
+
const n = Number(raw);
|
|
35
|
+
if (Number.isFinite(n) && n > 0) return Math.min(n, 30);
|
|
36
|
+
return null;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// 指数退避 + full jitter
|
|
40
|
+
export function backoffMs(attempt, { baseMs = 500, factor = 2, maxMs = 10000 } = {}) {
|
|
41
|
+
const exp = Math.min(maxMs, baseMs * Math.pow(factor, attempt));
|
|
42
|
+
return Math.floor(exp * (0.5 + Math.random() * 0.5));
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// 可取消 sleep (AbortSignal)
|
|
46
|
+
export function sleep(ms, signal) {
|
|
47
|
+
return new Promise((resolve, reject) => {
|
|
48
|
+
if (signal?.aborted) return reject(abortError());
|
|
49
|
+
const t = setTimeout(() => { cleanup(); resolve(); }, ms);
|
|
50
|
+
const onAbort = () => { clearTimeout(t); cleanup(); reject(abortError()); };
|
|
51
|
+
if (signal) signal.addEventListener("abort", onAbort, { once: true });
|
|
52
|
+
function cleanup() { if (signal) signal.removeEventListener("abort", onAbort); }
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
function abortError() { const e = new Error("aborted"); e.name = "AbortError"; return e; }
|
|
57
|
+
|
|
58
|
+
// 重试执行器: fn 抛瞬态错误时按退避重试, 非瞬态直接抛
|
|
59
|
+
export async function withRetry(fn, { maxRetries = 3, baseMs = 500, factor = 2, maxMs = 10000, signal = null, shouldRetry = isTransientError } = {}) {
|
|
60
|
+
let lastErr;
|
|
61
|
+
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
62
|
+
try {
|
|
63
|
+
return await fn();
|
|
64
|
+
} catch (e) {
|
|
65
|
+
lastErr = e;
|
|
66
|
+
if (attempt >= maxRetries || !shouldRetry(e)) throw e;
|
|
67
|
+
const ra = retryAfterSeconds(e);
|
|
68
|
+
const ms = ra != null ? ra * 1000 : backoffMs(attempt, { baseMs, factor, maxMs });
|
|
69
|
+
await sleep(ms, signal);
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
throw lastErr;
|
|
73
|
+
}
|
package/src/llm/router.js
CHANGED
|
@@ -1,98 +1,93 @@
|
|
|
1
|
-
// src/llm/router.js - 模型路由 (provider 选择中枢,
|
|
2
|
-
// 目标: "本地默认优先, 云端可自由接入"
|
|
3
|
-
// - 占位死配置过滤: model 含 REPLACE_WITH_YOUR_ENDPOINT 等占位符的 provider 视为不可用(省得误选+报噪音警告)
|
|
4
|
-
// - 本地优先(默认): 本地测试直接用本地模型 (lmstudio/ollama, 127.0.0.1 即零配置可用), 配真实云端 key 也先走本地
|
|
5
|
-
// - 云端优先(可选): 设 agent.model_preference=cloud → 配真 key 的云端排前 (深度/智谱/千问/火山/OpenAI), 本地兜底
|
|
6
|
-
// - 健康排序: 启动时异步探测各 provider /models, 能连的排前, 连不上自动降级
|
|
7
|
-
//
|
|
8
|
-
//
|
|
9
|
-
// const
|
|
10
|
-
// const
|
|
11
|
-
// const
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
//
|
|
15
|
-
|
|
16
|
-
//
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
//
|
|
42
|
-
|
|
43
|
-
if (
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
//
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
const
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
const
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
const
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
}));
|
|
94
|
-
const healthy = states.filter((s) => s.ok).map((s) => s.client);
|
|
95
|
-
const unwell = states.filter((s) => !s.ok).map((s) => s.client);
|
|
96
|
-
// 健康的保持原云/本地顺序, 不健康排最后 (不丢弃, 让运行时 fallback 继续尝试)
|
|
97
|
-
return [...healthy, ...unwell];
|
|
1
|
+
// src/llm/router.js - 模型路由 (provider 选择中枢, 唯一真相源, 自研底座)
|
|
2
|
+
// 目标: "本地默认优先, 云端可自由接入"
|
|
3
|
+
// - 占位死配置过滤: model 含 REPLACE_WITH_YOUR_ENDPOINT 等占位符的 provider 视为不可用(省得误选+报噪音警告)
|
|
4
|
+
// - 本地优先(默认): 本地测试直接用本地模型 (lmstudio/ollama, 127.0.0.1 即零配置可用), 配真实云端 key 也先走本地
|
|
5
|
+
// - 云端优先(可选): 设 agent.model_preference=cloud → 配真 key 的云端排前 (深度/智谱/千问/火山/OpenAI), 本地兜底
|
|
6
|
+
// - 健康排序: 启动时异步探测各 provider /models, 能连的排前, 连不上自动降级
|
|
7
|
+
// 全部 provider 均为自研 http 底座 (OpenAI 兼容 API 直连), 无外部引擎依赖。
|
|
8
|
+
// 用法 (与旧 builtin.resolveLLM 同签名, 向后兼容):
|
|
9
|
+
// const llm = resolveLLM(config); // 同步选择主 LLM
|
|
10
|
+
// const all = resolveAllLLMs(config); // 全量可用 provider
|
|
11
|
+
// const ok = isUsableProvider(prov); // 是否可用(过滤占位符)
|
|
12
|
+
// const candid = await orderByHealth(all); // 异步健康排序(可选, 给启动探测用)
|
|
13
|
+
import { LLMClient } from "./client.js";
|
|
14
|
+
// 占位符判定收敛到 config/placeholder.js (唯一真相源) —— v1.0.8 修复 P2-3:
|
|
15
|
+
// 原先此处自带一份正则且漏了 `YOUR_*_MODEL` 形态, 导致模板里的 lmstudio
|
|
16
|
+
// (model=YOUR_LOCAL_MODEL_NAME, base_url=127.0.0.1) 被判为"零配置可用"并选为主模型。
|
|
17
|
+
import { isPlaceholder } from "../config/placeholder.js";
|
|
18
|
+
|
|
19
|
+
// 本地推理服务地址特征
|
|
20
|
+
const LOCAL_HOST_RE = /127\.0\.0\.1|localhost|lm-studio|ollama/i;
|
|
21
|
+
|
|
22
|
+
function hasRealKey(p) {
|
|
23
|
+
// 显式 api_key 且非占位 -> 真 key
|
|
24
|
+
if (p.api_key && !isPlaceholder(String(p.api_key))) return true;
|
|
25
|
+
// env 引用且 env 里设了非空值 -> 真 key
|
|
26
|
+
if (p.api_key_env && process.env[p.api_key_env]) {
|
|
27
|
+
const v = String(process.env[p.api_key_env]);
|
|
28
|
+
return !!v && !isPlaceholder(v);
|
|
29
|
+
}
|
|
30
|
+
return false;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function isLocal(p) {
|
|
34
|
+
return LOCAL_HOST_RE.test(String(p.base_url || ""));
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
// 占位符过滤 + 可用判定 (与旧 isUsableProvider 同语义, 增加占位符排除)
|
|
38
|
+
export function isUsableProvider(prov) {
|
|
39
|
+
if (!prov) return false;
|
|
40
|
+
// 占位 model/base_url 的死配置直接判不可用
|
|
41
|
+
// (volcengine 的 REPLACE_WITH_YOUR_ENDPOINT, 以及模板里 lmstudio 的 YOUR_LOCAL_MODEL_NAME)
|
|
42
|
+
if (isPlaceholder(prov.model) || isPlaceholder(prov.base_url)) return false;
|
|
43
|
+
if (hasRealKey(prov)) return true; // 云端真 key
|
|
44
|
+
if (isLocal(prov)) return true; // 本地推理零配置可用
|
|
45
|
+
return false;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// 排序: 按 agent.model_preference 决定本地/云端谁优先 (默认 local)
|
|
49
|
+
// - local: 本地优先(lmstudio/ollama) > 云端真key [本地测试默认]
|
|
50
|
+
// - cloud: 云端真key优先 > 本地兜底 [正式发布可配]
|
|
51
|
+
// 健康状态排序由 orderByHealth 异步完成; 这里是"无探测时代理"的基础排序
|
|
52
|
+
function orderProviders(provs, preference) {
|
|
53
|
+
const cloud = provs.filter((p) => hasRealKey(p) && !isLocal(p));
|
|
54
|
+
const local = provs.filter((p) => isLocal(p)); // 本地服务都收 (lmstudio 常带字面 api_key, 仍零配置)
|
|
55
|
+
return preference === "cloud" ? [...cloud, ...local] : [...local, ...cloud];
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export function resolvePreference(config) {
|
|
59
|
+
return (config?.agent?.model_preference === "cloud") ? "cloud" : "local";
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function resolveAllLLMs(config) {
|
|
63
|
+
const provs = (config && config.providers) || [];
|
|
64
|
+
const pref = resolvePreference(config);
|
|
65
|
+
return orderProviders(provs.filter(isUsableProvider), pref).map((p) => new LLMClient(p));
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export function resolveLLM(config) {
|
|
69
|
+
const provs = (config && config.providers) || [];
|
|
70
|
+
// 强制指定: PPX_PROVIDER=<id> (测试/用户显式选择)
|
|
71
|
+
const forced = process.env.PPX_PROVIDER;
|
|
72
|
+
if (forced) {
|
|
73
|
+
const t = provs.find((x) => x.id === forced || x.id === String(forced).toLowerCase());
|
|
74
|
+
if (t && isUsableProvider(t)) return new LLMClient(t);
|
|
75
|
+
}
|
|
76
|
+
const pref = resolvePreference(config);
|
|
77
|
+
const ordered = orderProviders(provs.filter(isUsableProvider), pref);
|
|
78
|
+
return ordered.length ? new LLMClient(ordered[0]) : null;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
// 异步健康排序: 启动时探测各候选 /models, 能连的排前 (只读, 不改配置)
|
|
82
|
+
// returns: [{ client, health }] 按健康排序; 全部失败则保持原顺序 (兜底不报死)
|
|
83
|
+
export async function orderByHealth(clients, { probeMs = 4000 } = {}) {
|
|
84
|
+
if (!clients || !clients.length) return [];
|
|
85
|
+
const states = await Promise.all(clients.map(async (c) => {
|
|
86
|
+
try { return { client: c, ok: typeof c.health === "function" ? await c.health() : true }; }
|
|
87
|
+
catch { return { client: c, ok: false }; }
|
|
88
|
+
}));
|
|
89
|
+
const healthy = states.filter((s) => s.ok).map((s) => s.client);
|
|
90
|
+
const unwell = states.filter((s) => !s.ok).map((s) => s.client);
|
|
91
|
+
// 健康的保持原云/本地顺序, 不健康排最后 (不丢弃, 让运行时 fallback 继续尝试)
|
|
92
|
+
return [...healthy, ...unwell];
|
|
98
93
|
}
|