lume-dsh-plugin 0.7.3 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +460 -0
- package/README.md +245 -275
- package/lib/client.js +242 -181
- package/lib/core/card.js +11 -9
- package/lib/core/citations.js +235 -0
- package/lib/core/coverage.js +149 -0
- package/lib/core/dialogue-mining.js +49 -11
- package/lib/core/knowledge.js +165 -0
- package/lib/core/leak-detector.js +1 -3
- package/lib/core/ledger.js +129 -18
- package/lib/core/manifest.js +3 -1
- package/lib/core/memory-id.js +105 -0
- package/lib/core/metrics.js +270 -0
- package/lib/core/persona-limits.js +25 -0
- package/lib/core/scope.js +124 -0
- package/lib/core/signals.js +285 -5
- package/lib/core/task-memory.js +143 -0
- package/lib/core/text.js +45 -3
- package/lib/host/backfill.js +218 -0
- package/lib/host/bootstrap.js +130 -0
- package/lib/host/boundary.js +1 -3
- package/lib/host/clauses.js +180 -0
- package/lib/host/config.js +7 -0
- package/lib/host/diag.js +44 -7
- package/lib/host/distill-prompt.js +365 -0
- package/lib/host/distill.js +22 -348
- package/lib/host/extraction.js +11 -3
- package/lib/host/host-context.js +1 -0
- package/lib/host/host-events.js +100 -0
- package/lib/host/identity.js +12 -29
- package/lib/host/inbound.js +133 -0
- package/lib/host/injection.js +2 -6
- package/lib/host/llm-aux.js +130 -0
- package/lib/host/llm-route.js +3 -0
- package/lib/host/methods.js +182 -11
- package/lib/host/metrics-log.js +169 -0
- package/lib/host/notices.js +62 -0
- package/lib/host/project-access.js +210 -0
- package/lib/host/project.js +153 -2
- package/lib/host/prompt-blocks.js +113 -0
- package/lib/host/protocol.js +144 -11
- package/lib/host/reflection.js +28 -5
- package/lib/host/registry.js +0 -4
- package/lib/host/requirements-scan.js +108 -0
- package/lib/host/rpc-bridge.js +23 -4
- package/lib/host/rpc.js +1 -1
- package/lib/host/sections.js +48 -0
- package/lib/host/session-deps.js +27 -0
- package/lib/host/session-events.js +371 -0
- package/lib/host/session-runtime.js +18 -5
- package/lib/host/thinking.js +10 -1
- package/lib/host/tools.js +367 -0
- package/lib/host/triggers.js +30 -6
- package/lib/host/turn-boundary.js +116 -0
- package/lib/host/wiring.js +280 -0
- package/lib/host/workspace-map.js +81 -0
- package/lib/index.js +334 -836
- package/package.json +13 -4
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 知识作用域:把「这个仓库怎么干活」和「这个需求特有的结论」分开。
|
|
3
|
+
*
|
|
4
|
+
* 为什么需要:知识按**工作目录**共享(同一仓库的所有需求共用一份)。好处是通用约定一次学会处处可用;
|
|
5
|
+
* 坏处是**需求特有的结论会污染别的需求**——「优惠视图的列名用 PERMISSION_NAME」对退费需求毫无意义,
|
|
6
|
+
* 却会出现在它的注入里,既占额度又误导。
|
|
7
|
+
*
|
|
8
|
+
* 判定全部机械(零 token、可测):
|
|
9
|
+
* - 命中任务指代词(本次/这个需求/该需求/本需求…)→ task;
|
|
10
|
+
* - 命中**需求标题的显著词**(标题里 ≥2 字、且不在通用词表里的词)→ task(并记下归属哪个需求);
|
|
11
|
+
* - 其余 → repo(构建/测试/模块链路/通用约定/环境坑,这些对同仓库所有需求都成立)。
|
|
12
|
+
*
|
|
13
|
+
* 保守取向:拿不准就归 repo —— 少给一条通用知识比多给一条无关知识代价大。
|
|
14
|
+
*/
|
|
15
|
+
/** 任务指代:出现这些词说明句子在讲"当前这个需求"。 */
|
|
16
|
+
const TASK_POINTER_RE = /(本次|这次|当前需求|这个需求|该需求|本需求|此需求|这条需求|本方案|这个方案)/;
|
|
17
|
+
/** 通用词表:太常见,不能作为"需求特有词"(否则任何句子都会被判成 task)。 */
|
|
18
|
+
const GENERIC_WORDS = new Set([
|
|
19
|
+
"新增",
|
|
20
|
+
"修改",
|
|
21
|
+
"删除",
|
|
22
|
+
"字段",
|
|
23
|
+
"接口",
|
|
24
|
+
"页面",
|
|
25
|
+
"列表",
|
|
26
|
+
"导入",
|
|
27
|
+
"导出",
|
|
28
|
+
"查询",
|
|
29
|
+
"数据",
|
|
30
|
+
"脚本",
|
|
31
|
+
"文档",
|
|
32
|
+
"需求",
|
|
33
|
+
"功能",
|
|
34
|
+
"问题",
|
|
35
|
+
"方案",
|
|
36
|
+
"代码",
|
|
37
|
+
"配置",
|
|
38
|
+
"权限",
|
|
39
|
+
"用户",
|
|
40
|
+
"订单",
|
|
41
|
+
"系统",
|
|
42
|
+
"平台",
|
|
43
|
+
"管理",
|
|
44
|
+
"服务",
|
|
45
|
+
"数据库",
|
|
46
|
+
"表结构",
|
|
47
|
+
"测试",
|
|
48
|
+
"构建",
|
|
49
|
+
"部署",
|
|
50
|
+
"上线",
|
|
51
|
+
"回滚",
|
|
52
|
+
"迁移",
|
|
53
|
+
"校验",
|
|
54
|
+
"统计",
|
|
55
|
+
"报表",
|
|
56
|
+
"通知",
|
|
57
|
+
]);
|
|
58
|
+
/**
|
|
59
|
+
* 需求标题的显著词:切出 ≥2 字的中文片段与 ≥3 字的英文词,去掉通用词。
|
|
60
|
+
* 例:「B2I 优惠视图新增字段」→ [b2i, 优惠视图]("新增/字段"是通用词,丢掉)。
|
|
61
|
+
*/
|
|
62
|
+
export function taskKeywords(title) {
|
|
63
|
+
const raw = String(title ?? "").toLowerCase();
|
|
64
|
+
const english = (raw.match(/[a-z][a-z0-9]{2,}/g) ?? []).filter((word) => !GENERIC_WORDS.has(word));
|
|
65
|
+
const chinese = (raw.match(/[\u4e00-\u9fff]{2,}/g) ?? []).flatMap((run) => {
|
|
66
|
+
// 长片段再切 2~4 字滑窗,让"优惠视图新增字段"里的"优惠视图"能被单独识别
|
|
67
|
+
if (run.length <= 4)
|
|
68
|
+
return [run];
|
|
69
|
+
const out = [];
|
|
70
|
+
for (let size = 4; size >= 2; size--)
|
|
71
|
+
for (let i = 0; i + size <= run.length; i++)
|
|
72
|
+
out.push(run.slice(i, i + size));
|
|
73
|
+
return out;
|
|
74
|
+
});
|
|
75
|
+
return [...new Set([...english, ...chinese])].filter((word) => !GENERIC_WORDS.has(word));
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* 判定一条知识的作用域。
|
|
79
|
+
*
|
|
80
|
+
* 现场问题(2026-09-24):判据原先只用**会话标题**,而标题常是"接着优惠视图的任务干活"这种临时话 →
|
|
81
|
+
* 判不出归属 → 40 条知识全归 repo → 任何需求都能看到全部知识 → 模型从"退费 / 优惠视图 / 通用约定"
|
|
82
|
+
* 三条线索里读出了**三个需求**(实际只有两个;WTPF_GOODS_PROPERTY_DEF 就是优惠视图那张表)。
|
|
83
|
+
*
|
|
84
|
+
* 现在优先按**仓库里真实存在的需求名**归属(来自 `<cwd>/doc/*`):命中多个取最长的(更具体)。
|
|
85
|
+
* `taskTitle` 只作为兜底。
|
|
86
|
+
*/
|
|
87
|
+
export function classifyScope(text, input) {
|
|
88
|
+
const options = typeof input === "string" || input == null ? { taskTitle: input ?? null } : input;
|
|
89
|
+
const body = String(text ?? "").toLowerCase();
|
|
90
|
+
let best = null;
|
|
91
|
+
for (const hint of options.requirementHints ?? []) {
|
|
92
|
+
const name = String(hint?.name ?? "").trim();
|
|
93
|
+
if (!name)
|
|
94
|
+
continue;
|
|
95
|
+
// 需求名切词(中文 2~4 字窗口)+ 文档里抽到的标识符(表名/常量/字段名)
|
|
96
|
+
for (const keyword of [...taskKeywords(name), ...(hint.keywords ?? [])]) {
|
|
97
|
+
const token = String(keyword).toLowerCase();
|
|
98
|
+
if (token.length < 2 || !body.includes(token))
|
|
99
|
+
continue;
|
|
100
|
+
if (!best || token.length > best.score)
|
|
101
|
+
best = { name, score: token.length };
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
if (best)
|
|
105
|
+
return { scope: "task", task: best.name };
|
|
106
|
+
const title = String(options.taskTitle ?? "").trim();
|
|
107
|
+
// 标题可能是"首条用户消息"被截断的样子(含方括号/换行/过长)——那种**不是需求名**,
|
|
108
|
+
// 拿它当 task 标签会造出 `[系统背景与诊断事实] 你是 DSH…` 这种垃圾归属(现场见过)。
|
|
109
|
+
if (!title || title.length > 30 || /[[\]\n\r]/.test(title))
|
|
110
|
+
return { scope: "repo" };
|
|
111
|
+
if (TASK_POINTER_RE.test(body))
|
|
112
|
+
return { scope: "task", task: title };
|
|
113
|
+
const hit = taskKeywords(title).some((keyword) => keyword.length >= 2 && body.includes(keyword));
|
|
114
|
+
return hit ? { scope: "task", task: title } : { scope: "repo" };
|
|
115
|
+
}
|
|
116
|
+
/** 这条知识能不能给"当前需求"看:通用知识人人可见,需求知识只给同一需求。 */
|
|
117
|
+
export function visibleForTask(fact, currentTask) {
|
|
118
|
+
if (fact.scope !== "task")
|
|
119
|
+
return true;
|
|
120
|
+
const current = String(currentTask ?? "").trim();
|
|
121
|
+
if (!current)
|
|
122
|
+
return false; // 冷启动/无标题:需求级知识先不给,避免误导
|
|
123
|
+
return fact.task === current;
|
|
124
|
+
}
|
package/lib/core/signals.js
CHANGED
|
@@ -1,12 +1,90 @@
|
|
|
1
1
|
const PLAN_TOKENS = new Set(["todo", "plan", "contract", "change", "ledger", "hypothesis", "note"]);
|
|
2
2
|
const LUME_TOKENS = new Set(["lume"]);
|
|
3
3
|
const VERIFY_TOKENS = new Set([
|
|
4
|
-
"bash",
|
|
5
|
-
"
|
|
6
|
-
"
|
|
4
|
+
"bash",
|
|
5
|
+
"shell",
|
|
6
|
+
"pwsh",
|
|
7
|
+
"powershell",
|
|
8
|
+
"cmd",
|
|
9
|
+
"terminal",
|
|
10
|
+
"run",
|
|
11
|
+
"exec",
|
|
12
|
+
"job",
|
|
13
|
+
"make",
|
|
14
|
+
"mvn",
|
|
15
|
+
"gradle",
|
|
16
|
+
"npm",
|
|
17
|
+
"pnpm",
|
|
18
|
+
"yarn",
|
|
19
|
+
"bun",
|
|
20
|
+
"deno",
|
|
21
|
+
"node",
|
|
22
|
+
"tsc",
|
|
23
|
+
"tsdown",
|
|
24
|
+
"vite",
|
|
25
|
+
"vitest",
|
|
26
|
+
"jest",
|
|
27
|
+
"pytest",
|
|
28
|
+
"cargo",
|
|
29
|
+
"go",
|
|
30
|
+
"dotnet",
|
|
31
|
+
"msbuild",
|
|
32
|
+
"compile",
|
|
33
|
+
"build",
|
|
34
|
+
"test",
|
|
35
|
+
"lint",
|
|
36
|
+
"typecheck",
|
|
37
|
+
"check",
|
|
38
|
+
"verify",
|
|
39
|
+
]);
|
|
40
|
+
const MUTATE_TOKENS = new Set([
|
|
41
|
+
"edit",
|
|
42
|
+
"write",
|
|
43
|
+
"multiedit",
|
|
44
|
+
"patch",
|
|
45
|
+
"apply",
|
|
46
|
+
"replace",
|
|
47
|
+
"create",
|
|
48
|
+
"delete",
|
|
49
|
+
"remove",
|
|
50
|
+
"rename",
|
|
51
|
+
"move",
|
|
52
|
+
"append",
|
|
53
|
+
"insert",
|
|
54
|
+
"mkdir",
|
|
55
|
+
"apply_patch",
|
|
56
|
+
]);
|
|
57
|
+
const INSPECT_TOKENS = new Set([
|
|
58
|
+
"read",
|
|
59
|
+
"view",
|
|
60
|
+
"cat",
|
|
61
|
+
"grep",
|
|
62
|
+
"search",
|
|
63
|
+
"glob",
|
|
64
|
+
"find",
|
|
65
|
+
"ls",
|
|
66
|
+
"list",
|
|
67
|
+
"tree",
|
|
68
|
+
"analyze",
|
|
69
|
+
"symbol",
|
|
70
|
+
"reference",
|
|
71
|
+
"web",
|
|
72
|
+
"fetch",
|
|
73
|
+
"browser",
|
|
74
|
+
"screenshot",
|
|
75
|
+
"image",
|
|
76
|
+
"git",
|
|
77
|
+
"status",
|
|
78
|
+
"diff",
|
|
79
|
+
"log",
|
|
80
|
+
"show",
|
|
81
|
+
"stat",
|
|
82
|
+
"head",
|
|
83
|
+
"tail",
|
|
84
|
+
"query",
|
|
85
|
+
"sql",
|
|
86
|
+
"map",
|
|
7
87
|
]);
|
|
8
|
-
const MUTATE_TOKENS = new Set(["edit", "write", "multiedit", "patch", "apply", "replace", "create", "delete", "remove", "rename", "move", "append", "insert", "mkdir", "apply_patch"]);
|
|
9
|
-
const INSPECT_TOKENS = new Set(["read", "view", "cat", "grep", "search", "glob", "find", "ls", "list", "tree", "analyze", "symbol", "reference", "web", "fetch", "browser", "screenshot", "image", "git", "status", "diff", "log", "show", "stat", "head", "tail", "query", "sql", "map"]);
|
|
10
88
|
/** 把工具名切成小写词元:`lume_contract` → [lume, contract];`mcp__fs__read_file` → [mcp, fs, read, file]。 */
|
|
11
89
|
function tokens(name) {
|
|
12
90
|
return String(name ?? "")
|
|
@@ -14,6 +92,140 @@ function tokens(name) {
|
|
|
14
92
|
.split(/[^a-z0-9]+/)
|
|
15
93
|
.filter(Boolean);
|
|
16
94
|
}
|
|
95
|
+
/**
|
|
96
|
+
* 「真验证」判据(C1 自动推进台账用):命令文本真的在跑编译/测试/检查,
|
|
97
|
+
* 而不是 `git grep`、`ls` 这类同样归入 verify 类的通用命令。
|
|
98
|
+
*
|
|
99
|
+
* 为什么需要它:把 `git grep` 当成一次验证,会把"未验证"洗白——那比不做更糟。
|
|
100
|
+
* 所以取**宁窄勿宽**:宁可少自动推进几条,也不能让台账撒谎。
|
|
101
|
+
*/
|
|
102
|
+
export const REAL_VERIFY_RE = /(?:^|[\s&|;"'])(?:tsc|tsdown|vitest|jest|mocha|pytest|mvn|gradle|gradlew|npm|pnpm|yarn|bun|deno|go|dotnet|cargo|make)\s[^&|]*\b(?:test|build|lint|typecheck|check|verify|compile|package|vitest|tsc)\b|node_modules[\\/]\.bin[\\/]|--noEmit|\btsc\b|\bvitest\b|\btsdown\b/i;
|
|
103
|
+
export function isRealVerifyCommand(args) {
|
|
104
|
+
const body = typeof args === "string" ? args : JSON.stringify(args ?? "");
|
|
105
|
+
return REAL_VERIFY_RE.test(body);
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* 把一次改动工具的入参压成一句可读摘要(自动台账用)。
|
|
109
|
+
*
|
|
110
|
+
* 为什么需要:自动台账原来只写「(自动)由 edit 修改」——交付对账要靠它列条目,
|
|
111
|
+
* 而这种文案对模型没有任何信息量(它不知道改了哪一行、改成了什么),对账就沦为形式。
|
|
112
|
+
*/
|
|
113
|
+
/** 交付物正文(覆盖核对要用全文,不是首行摘要)。 */
|
|
114
|
+
export function toolArtifactText(args) {
|
|
115
|
+
const record = (args && typeof args === "object" ? args : null);
|
|
116
|
+
if (!record)
|
|
117
|
+
return "";
|
|
118
|
+
const parts = [record.content, record.new_string, record.newString, record.new_str, record.text, record.new_text];
|
|
119
|
+
const edits = Array.isArray(record.edits) ? record.edits : [];
|
|
120
|
+
for (const edit of edits) {
|
|
121
|
+
if (edit && typeof edit === "object")
|
|
122
|
+
parts.push(edit.new_string, edit.content);
|
|
123
|
+
}
|
|
124
|
+
return parts.filter((p) => typeof p === "string" && p.trim().length > 0).join("\n");
|
|
125
|
+
}
|
|
126
|
+
export function summarizeToolChange(args, toolName) {
|
|
127
|
+
const record = (args && typeof args === "object" ? args : null);
|
|
128
|
+
if (!record)
|
|
129
|
+
return `由 ${toolName} 修改`;
|
|
130
|
+
const candidates = [
|
|
131
|
+
record.new_string,
|
|
132
|
+
record.newString,
|
|
133
|
+
record.new_str,
|
|
134
|
+
record.content,
|
|
135
|
+
record.text,
|
|
136
|
+
record.new_text,
|
|
137
|
+
record.contents,
|
|
138
|
+
];
|
|
139
|
+
const edits = Array.isArray(record.edits) ? record.edits : [];
|
|
140
|
+
for (const edit of edits) {
|
|
141
|
+
if (edit && typeof edit === "object")
|
|
142
|
+
candidates.push(edit.new_string, edit.content);
|
|
143
|
+
}
|
|
144
|
+
for (const candidate of candidates) {
|
|
145
|
+
if (typeof candidate !== "string" || !candidate.trim())
|
|
146
|
+
continue;
|
|
147
|
+
const firstLine = candidate
|
|
148
|
+
.split(/\r?\n/)
|
|
149
|
+
.map((line) => line.trim())
|
|
150
|
+
.find((line) => line.length > 0) ?? "";
|
|
151
|
+
if (!firstLine)
|
|
152
|
+
continue;
|
|
153
|
+
return firstLine.replace(/\s+/g, " ").slice(0, 70);
|
|
154
|
+
}
|
|
155
|
+
const path = record.path ?? record.file_path ?? record.filePath;
|
|
156
|
+
return path ? `${toolName} 改了 ${String(path)}(未取到内容摘要)` : `由 ${toolName} 修改`;
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
* 数「抛回给用户的待确认清单」有几项(提问纪律的结构化核对)。
|
|
160
|
+
*
|
|
161
|
+
* 现场(2026-09-23 turn 20):用户只问「现在方案是不是都清楚了?」,模型回了 4 条"待你定"
|
|
162
|
+
* (生产库类型 / status 口径 / 权限人下拉来源 / 分页 total),用户直接反问「分页还能有疑问?
|
|
163
|
+
* 不就是改前端的吗」,模型下一轮自己承认「三个是我自己造的,撤」——而那一轮的上下文里
|
|
164
|
+
* **没有**「提问前提必须已核实」这条(它当时只挂在〔需求解读〕上,而需求解读只在"用户给了
|
|
165
|
+
* 新需求"的轮次出现)。规则在不在场,决定它会不会自查。
|
|
166
|
+
*
|
|
167
|
+
* 判据只做**结构计数**,不做语义猜测:出现"待定/待确认/还没定/不确定"这类小节标题后,
|
|
168
|
+
* 紧跟的列表项数量。目的不是判断问题对不对,而是把"你抛了几个问题"这个事实摆出来。
|
|
169
|
+
*/
|
|
170
|
+
export const MAX_OPEN_QUESTIONS = 2;
|
|
171
|
+
const OPEN_QUESTION_HEADING_RE = /待定|待确认|还没定|未定|不确定|需要你|要你定|请你确认/;
|
|
172
|
+
const LIST_ITEM_RE = /^\s*(?:(?:[-*•](?!\*))|\d+\s*[.、)]|[一二三四五六七八九十]+\s*[、.])/;
|
|
173
|
+
const QUESTION_ITEM_RE = /待你定|待定|待确认|要你定|你定|请你确认|需要你确认/;
|
|
174
|
+
const ITEM_EVIDENCE_RE = /[::]\s*\d{1,6}/;
|
|
175
|
+
const ITEM_UNANSWERABLE_RE = /查不到|没有代码|代码库里没有|不在仓库|登录不了|无法访问|环境限制|需要你提供|只有你能|配置中心/;
|
|
176
|
+
/** 抽出「要用户拍板的条目」原文:小节标题下的列表项 + 行内含「待确认/待你定」的句子。 */
|
|
177
|
+
function openQuestionItems(text) {
|
|
178
|
+
const lines = String(text ?? "").split(/\r?\n/);
|
|
179
|
+
const items = [];
|
|
180
|
+
const headingLines = new Set();
|
|
181
|
+
const push = (line) => {
|
|
182
|
+
const trimmed = line.trim();
|
|
183
|
+
if (trimmed && !items.includes(trimmed))
|
|
184
|
+
items.push(trimmed);
|
|
185
|
+
};
|
|
186
|
+
// 「小节标题」才是标题(短、无句读);长句里出现「待定」是条目本身,不能当标题排掉
|
|
187
|
+
const isHeadingLike = (line) => line.trim().length <= 24 && !/[。!?;,,;]/.test(line);
|
|
188
|
+
for (let i = 0; i < lines.length; i++) {
|
|
189
|
+
if (!OPEN_QUESTION_HEADING_RE.test(lines[i]))
|
|
190
|
+
continue;
|
|
191
|
+
if (isHeadingLike(lines[i]))
|
|
192
|
+
headingLines.add(i);
|
|
193
|
+
for (let j = i + 1; j < lines.length && j <= i + 16; j++) {
|
|
194
|
+
const line = lines[j];
|
|
195
|
+
if (!line.trim())
|
|
196
|
+
continue; // 列表项之间允许空行
|
|
197
|
+
if (LIST_ITEM_RE.test(line)) {
|
|
198
|
+
push(line);
|
|
199
|
+
continue;
|
|
200
|
+
}
|
|
201
|
+
break; // 遇到非列表内容即认为小节结束
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
// 行内条目("文档里要标一条待定:status 口径"这种单句也常见);标题行本身不算条目
|
|
205
|
+
for (let i = 0; i < lines.length; i++) {
|
|
206
|
+
if (headingLines.has(i))
|
|
207
|
+
continue;
|
|
208
|
+
if (QUESTION_ITEM_RE.test(lines[i]))
|
|
209
|
+
push(lines[i]);
|
|
210
|
+
}
|
|
211
|
+
return items;
|
|
212
|
+
}
|
|
213
|
+
export function countOpenQuestions(text) {
|
|
214
|
+
return openQuestionItems(text).length;
|
|
215
|
+
}
|
|
216
|
+
/**
|
|
217
|
+
* 提问质量核对(比数量核对更准,因为真实事故常常只有**一条**假问题)。
|
|
218
|
+
*
|
|
219
|
+
* 现场(2026-09-23 turn 22):模型把「status 口径」挂成待确认要用户拍板——而它自己 turn 19
|
|
220
|
+
* 就读过 526-534(导入路径直写 Excel 值),结论早已在手。数量核对(>2 条)抓不到这种一条就
|
|
221
|
+
* 命中的情况,所以这里按**证据**判:条目里有没有行号?没行号又没说明"代码答不了"的,就是
|
|
222
|
+
* 应该自己先核实的那类。
|
|
223
|
+
*/
|
|
224
|
+
export function auditOpenQuestions(text) {
|
|
225
|
+
const items = openQuestionItems(text);
|
|
226
|
+
const unsupported = items.filter((item) => !ITEM_EVIDENCE_RE.test(item) && !ITEM_UNANSWERABLE_RE.test(item));
|
|
227
|
+
return { count: items.length, unsupported: unsupported.slice(0, 3) };
|
|
228
|
+
}
|
|
17
229
|
export function classifyTool(name) {
|
|
18
230
|
const parts = tokens(name);
|
|
19
231
|
if (parts.length === 0)
|
|
@@ -43,6 +255,20 @@ const UNKNOWN_RE = /结果未知|outcome unknown|tool_not_started|tool_outcome_u
|
|
|
43
255
|
* 网络/权限受阻。命中它才给「验证降级阶梯」——普通编译错误该归因到代码,
|
|
44
256
|
* 给环境阶梯反而会误导。
|
|
45
257
|
*/
|
|
258
|
+
/**
|
|
259
|
+
* 失败判据(唯一实现)。
|
|
260
|
+
*
|
|
261
|
+
* 现场(2026-09-24 评审):这条正则原先在 host/turn-boundary.ts 与 host/session-events.ts 各抄一份,
|
|
262
|
+
* 两处都没单测、改一处漏一处——而它决定「红了要不要立刻闭环」。判据属于 core(纯函数 + 单测),不该漏在 host。
|
|
263
|
+
*/
|
|
264
|
+
export function looksLikeFailure(text) {
|
|
265
|
+
return FAILURE_RE.test(String(text ?? "")) || /timed out|not started/i.test(String(text ?? ""));
|
|
266
|
+
}
|
|
267
|
+
/** 助手这轮是否声称「验证过」(交付对账与阶段推进共用)。 */
|
|
268
|
+
const CLAIMS_VERIFICATION_RE = /验证|测试|构建|检查|确认生效|实际结果|已通过|未验证|无法验证/i;
|
|
269
|
+
export function claimsVerification(text) {
|
|
270
|
+
return CLAIMS_VERIFICATION_RE.test(String(text ?? ""));
|
|
271
|
+
}
|
|
46
272
|
const ENV_FAILURE_RE = /could not resolve dependencies|could not find artifact|cannot find module|module_not_found|command not found|not recognized as an internal|不是内部或外部命令|系统找不到指定的路径|no such file or directory|enoent|offline mode|cannot access .* in offline|本地仓库|repository.*(?:empty|missing)|network is unreachable|econnrefused|etimedout|proxy|self-signed certificate|eacces/i;
|
|
47
273
|
/** 从工具结果文本判定成败。`explicitError` 为宿主上报的错误字段。 */
|
|
48
274
|
export function readResultSignals(text, explicitError = false) {
|
|
@@ -62,3 +288,57 @@ export function deadPathKind(envHits, failStreak) {
|
|
|
62
288
|
return null;
|
|
63
289
|
return envHits >= 2 ? "env" : "retry";
|
|
64
290
|
}
|
|
291
|
+
/**
|
|
292
|
+
* 需求漂移检测(词法级、零成本):模型的输出里出现了**需求原话里没有**的变更类型词。
|
|
293
|
+
*
|
|
294
|
+
* 现场样本:需求写「业务类型下拉新增三个选项」,模型却推论出「删除/割接」——用户当场纠正。
|
|
295
|
+
* 这类脑补完全可以用词法检出:动词在模型侧出现、在用户侧从未出现。
|
|
296
|
+
*/
|
|
297
|
+
const CHANGE_TYPE_WORDS = ["删除", "删掉", "下线", "停用", "替换", "割接", "回滚", "重构", "改名", "重命名", "迁移", "拆分", "合并"];
|
|
298
|
+
export function unrequestedChangeWords(requirementText, candidateText, skipWords = []) {
|
|
299
|
+
const requirement = String(requirementText ?? "");
|
|
300
|
+
const candidate = String(candidateText ?? "");
|
|
301
|
+
if (!candidate)
|
|
302
|
+
return [];
|
|
303
|
+
return CHANGE_TYPE_WORDS.filter((word) => {
|
|
304
|
+
if (!candidate.includes(word) || requirement.includes(word))
|
|
305
|
+
return false;
|
|
306
|
+
if (skipWords.includes(word))
|
|
307
|
+
return false;
|
|
308
|
+
return isDriftProposal(candidate, word);
|
|
309
|
+
});
|
|
310
|
+
}
|
|
311
|
+
/**
|
|
312
|
+
* 命中词到底算不算「漂移」:只有当模型把这件事说成**自己要做的变更**时才算。
|
|
313
|
+
*
|
|
314
|
+
* 事故(2026-09-23 现场,b2i-all 会话):旧实现只做词表匹配,于是
|
|
315
|
+
* ① 用户自己问「业务类型删了旧值,旧数据是不是涉及割接」→ 模型的回答被顶「收回」;
|
|
316
|
+
* ② 模型陈述事实「该字段 2025-03-17 引入时没做数据迁移」→ 被顶;
|
|
317
|
+
* ③ 风险分析里的「回滚」「影响面」→ 被顶(最近 5 轮里 3 轮命中,全是误报)。
|
|
318
|
+
* 代价写在模型的推理里:它开始**躲词**(「不提割接/迁移/替换」「为了安全我换措辞」)
|
|
319
|
+
* ——把注意力花在词表上而不是问题上,这才是「变笨」的真实来源。
|
|
320
|
+
*
|
|
321
|
+
* 所以判定分两步:附近有否定/疑问/风险语境 → 不计;变更词前有计划线索 → 才算提议。
|
|
322
|
+
* 取舍:**宁漏报不误报**——漏一次脑补的代价,远小于天天冤枉它、把它训成不敢用词。
|
|
323
|
+
*/
|
|
324
|
+
function isDriftProposal(candidate, word) {
|
|
325
|
+
let from = 0;
|
|
326
|
+
for (;;) {
|
|
327
|
+
const at = candidate.indexOf(word, from);
|
|
328
|
+
if (at < 0)
|
|
329
|
+
return false;
|
|
330
|
+
const before = candidate.slice(Math.max(0, at - DRIFT_BEFORE), at);
|
|
331
|
+
const after = candidate.slice(at + word.length, at + word.length + DRIFT_AFTER);
|
|
332
|
+
if (!DRIFT_EXEMPT_RE.test(before + after) && DRIFT_PLAN_RE.test(before))
|
|
333
|
+
return true;
|
|
334
|
+
from = at + word.length;
|
|
335
|
+
}
|
|
336
|
+
}
|
|
337
|
+
/** 豁免语境:否定、疑问、风险与讨论——出现这些说明它在讨论,不是在动手。 */
|
|
338
|
+
const DRIFT_EXEMPT_RE = /不|没|别|避免|无需|不用|是否|会不会|风险|影响|回滚|留痕|降级|万一|如果|若|讨论|方案|选项|历史|曾经|之前|已经|吗|?|\?/;
|
|
339
|
+
/** 计划线索:变更词之前出现这些,才是在说「我要做的变更」。 */
|
|
340
|
+
const DRIFT_PLAN_RE = /要|会|将|建议|应该|打算|计划|准备|必须|改为|改成|直接|需要/;
|
|
341
|
+
const DRIFT_BEFORE = 24;
|
|
342
|
+
const DRIFT_AFTER = 8;
|
|
343
|
+
/** 每会话最多顶几次〔需求漂移〕:同一条提醒反复出现 = 噪音,模型会学会忽略它。 */
|
|
344
|
+
export const DRIFT_NOTICE_MAX = 2;
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
const CAP = { requirement: 3, decided: 5, changed: 6, open: 4, deadends: 3, locate: 5 };
|
|
2
|
+
const text = (value, max = 120) => String(value ?? "")
|
|
3
|
+
.replace(/\s+/g, " ")
|
|
4
|
+
.trim()
|
|
5
|
+
.slice(0, max);
|
|
6
|
+
/** 台账条目的状态标记:未验证的要显眼——那是接手时最该先做的事。 */
|
|
7
|
+
const STATUS_MARK = { verified: "[已验证]", done: "[已改未验]", planned: "[计划]", skipped: "[跳过]" };
|
|
8
|
+
/**
|
|
9
|
+
* 从结构化状态构建会话记忆。**空白会话返回 null**(不写空记忆,避免把桶塞满"什么都没做"的记录)。
|
|
10
|
+
*/
|
|
11
|
+
export function buildTaskMemory(input) {
|
|
12
|
+
const goal = text(input.goal, 200);
|
|
13
|
+
const requirement = (input.requirement ?? [])
|
|
14
|
+
.map((item) => text(item.text, 200))
|
|
15
|
+
.filter(Boolean)
|
|
16
|
+
.slice(0, CAP.requirement);
|
|
17
|
+
const decided = (input.design ?? [])
|
|
18
|
+
.map((item) => text(`${item.point} → ${item.choice}`, 160))
|
|
19
|
+
.filter(Boolean)
|
|
20
|
+
.slice(-CAP.decided);
|
|
21
|
+
const changed = (input.changes ?? [])
|
|
22
|
+
.slice(-CAP.changed)
|
|
23
|
+
.map((item) => `${STATUS_MARK[item.status] ?? ""}${text(item.target, 70)}:${text(item.change, 80)}`.trim())
|
|
24
|
+
.filter(Boolean);
|
|
25
|
+
const open = [
|
|
26
|
+
...(input.hypotheses ?? [])
|
|
27
|
+
.filter((item) => item.status === "open" || item.status === "unconfirmed")
|
|
28
|
+
.map((item) => text(item.text, 140)),
|
|
29
|
+
]
|
|
30
|
+
.filter(Boolean)
|
|
31
|
+
.slice(0, CAP.open);
|
|
32
|
+
const deadends = (input.deadends ?? [])
|
|
33
|
+
.map((item) => text(item.text, 140))
|
|
34
|
+
.filter(Boolean)
|
|
35
|
+
.slice(0, CAP.deadends);
|
|
36
|
+
const locate = (input.locate ?? [])
|
|
37
|
+
.map((item) => text(item, 100))
|
|
38
|
+
.filter(Boolean)
|
|
39
|
+
.slice(-CAP.locate);
|
|
40
|
+
if (!goal && requirement.length === 0 && decided.length === 0 && changed.length === 0 && open.length === 0 && deadends.length === 0)
|
|
41
|
+
return null;
|
|
42
|
+
return {
|
|
43
|
+
sid: input.sid,
|
|
44
|
+
title: text(input.title, 60) || "(未命名会话)",
|
|
45
|
+
turn: input.turn,
|
|
46
|
+
goal,
|
|
47
|
+
requirement,
|
|
48
|
+
decided,
|
|
49
|
+
changed,
|
|
50
|
+
open,
|
|
51
|
+
deadends,
|
|
52
|
+
locate,
|
|
53
|
+
at: input.now ?? Date.now(),
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
/** 记忆是不是"值得写/值得注入":至少有两类内容,避免只有一句目标的空壳记忆。 */
|
|
57
|
+
export function memoryWeight(memory) {
|
|
58
|
+
return [
|
|
59
|
+
memory.goal,
|
|
60
|
+
memory.requirement.length,
|
|
61
|
+
memory.decided.length,
|
|
62
|
+
memory.changed.length,
|
|
63
|
+
memory.open.length,
|
|
64
|
+
memory.deadends.length,
|
|
65
|
+
memory.locate.length,
|
|
66
|
+
].filter((value) => (typeof value === "number" ? value > 0 : Boolean(value))).length;
|
|
67
|
+
}
|
|
68
|
+
const ageLabel = (at, now) => {
|
|
69
|
+
const minutes = (now - at) / 60_000;
|
|
70
|
+
if (!Number.isFinite(minutes) || minutes < 0)
|
|
71
|
+
return "";
|
|
72
|
+
if (minutes < 90)
|
|
73
|
+
return `约 ${Math.max(1, Math.round(minutes))} 分钟前`;
|
|
74
|
+
if (minutes < 48 * 60)
|
|
75
|
+
return `约 ${Math.round(minutes / 60)} 小时前`;
|
|
76
|
+
return `约 ${Math.round(minutes / 1440)} 天前`;
|
|
77
|
+
};
|
|
78
|
+
/**
|
|
79
|
+
* 注入文本:新会话开局用它"接着上一个会话干"。
|
|
80
|
+
*
|
|
81
|
+
* `recent` 是同一工作目录下的其它会话标题——用户可以直接说「继续 X」,
|
|
82
|
+
* 不必自己回忆"上次那个窗口叫什么"。
|
|
83
|
+
*/
|
|
84
|
+
export function renderTaskMemory(memory, options = {}) {
|
|
85
|
+
if (!memory || memoryWeight(memory) < 2)
|
|
86
|
+
return null;
|
|
87
|
+
const now = options.now ?? Date.now();
|
|
88
|
+
const lines = [`〔上次会话记忆|${memory.title}(${ageLabel(memory.at, now)},第 ${memory.turn} 轮)〕`];
|
|
89
|
+
if (memory.goal)
|
|
90
|
+
lines.push(`目标:${memory.goal}`);
|
|
91
|
+
if (memory.requirement.length > 0)
|
|
92
|
+
lines.push(`需求原话:${memory.requirement.map((item) => `「${item}」`).join(" ")}`);
|
|
93
|
+
if (memory.decided.length > 0)
|
|
94
|
+
lines.push(`已拍板:${memory.decided.map((item) => `- ${item}`).join(" ")}`);
|
|
95
|
+
if (memory.changed.length > 0)
|
|
96
|
+
lines.push(`改动:${memory.changed.join(" ")}`);
|
|
97
|
+
if (memory.open.length > 0)
|
|
98
|
+
lines.push(`未决:${memory.open.join(" ")}`);
|
|
99
|
+
if (memory.deadends.length > 0)
|
|
100
|
+
lines.push(`死路(别再试):${memory.deadends.join(" ")}`);
|
|
101
|
+
if (memory.locate.length > 0)
|
|
102
|
+
lines.push(`关键定位:${memory.locate.join(" ")}`);
|
|
103
|
+
lines.push(`要继续就说「继续 ${memory.title}」;未验证的改动优先补验证,别从头重做。`);
|
|
104
|
+
const others = (options.recent ?? []).filter((item) => item.title !== memory.title).slice(0, 3);
|
|
105
|
+
if (others.length > 0)
|
|
106
|
+
lines.push(`同目录其它会话:${others.map((item) => `${item.title}(${ageLabel(item.at, now)})`).join(" / ")}`);
|
|
107
|
+
return lines.join("\n");
|
|
108
|
+
}
|
|
109
|
+
/** markdown 版本:落到工作区给人看(等价于"手写会话记忆"的自动版)。 */
|
|
110
|
+
export function renderTaskMemoryMarkdown(memory, options = {}) {
|
|
111
|
+
const now = options.now ?? Date.now();
|
|
112
|
+
const section = (title, items) => items.length === 0 ? "" : `\n## ${title}\n\n${items.map((item) => `- ${item}`).join("\n")}\n`;
|
|
113
|
+
return [
|
|
114
|
+
`# 会话记忆 · ${memory.title}`,
|
|
115
|
+
"",
|
|
116
|
+
`> 自动生成(Lume)· 更新于 ${new Date(memory.at).toISOString()}(${ageLabel(memory.at, now)})· 第 ${memory.turn} 轮 · session \`${memory.sid}\``,
|
|
117
|
+
memory.goal ? `\n## 目标\n\n${memory.goal}\n` : "",
|
|
118
|
+
section("需求原话(逐字)", memory.requirement.map((item) => `「${item}」`)),
|
|
119
|
+
section("已拍板", memory.decided),
|
|
120
|
+
section("改动(未验证的优先补验证)", memory.changed),
|
|
121
|
+
section("未决", memory.open),
|
|
122
|
+
section("死路(别再试)", memory.deadends),
|
|
123
|
+
section("关键定位", memory.locate),
|
|
124
|
+
].join("\n");
|
|
125
|
+
}
|
|
126
|
+
/** 会话起点:没有契约、没有台账 —— 这种时候才需要把"上次会话记忆"顶上去。 */
|
|
127
|
+
export function isColdStart(state) {
|
|
128
|
+
return !state.hasContract && state.changes === 0 && state.requirements === 0;
|
|
129
|
+
}
|
|
130
|
+
/** 上下文压力:宿主给了 contextWindow,我们按最近一次用量估占用率。 */
|
|
131
|
+
export function contextPressure(usedTokens, contextWindow) {
|
|
132
|
+
if (!Number.isFinite(usedTokens) || !Number.isFinite(contextWindow) || contextWindow <= 0)
|
|
133
|
+
return { level: "ok", ratio: 0 };
|
|
134
|
+
const ratio = usedTokens / contextWindow;
|
|
135
|
+
return { level: ratio >= 0.9 ? "critical" : ratio >= 0.75 ? "warn" : "ok", ratio };
|
|
136
|
+
}
|
|
137
|
+
/** 上下文预警文案:告诉用户"该换窗口了",并说清记忆不会丢。 */
|
|
138
|
+
export function buildContextPressureDirective(level, ratio, memorySaved) {
|
|
139
|
+
const percent = Math.round(ratio * 100);
|
|
140
|
+
const head = level === "critical" ? `〔上下文接近上限:约 ${percent}%〕` : `〔上下文已用约 ${percent}%〕`;
|
|
141
|
+
const advice = "收尾当前这一步,然后**开一个新会话**继续——同工作目录的新会话会直接带上「上次会话记忆」(目标/已拍板/未决/关键定位)。";
|
|
142
|
+
return `${head}${memorySaved ? "会话记忆已保存:" : ""}${advice}不要再展开新话题,也不要把已有结论重述一遍占额度。`;
|
|
143
|
+
}
|
package/lib/core/text.js
CHANGED
|
@@ -2,16 +2,58 @@
|
|
|
2
2
|
* 消息文本提取(纯函数):从 Cordis 消息对象中提取纯文本。
|
|
3
3
|
*
|
|
4
4
|
* user/message 的 data 即消息内容;assistant/message 的 data.message 即消息内容。
|
|
5
|
+
*
|
|
6
|
+
* **真机形状(2026-09-24 现场取证)**:`tool/result` 的文本比消息本体**深一层**——
|
|
7
|
+
* `data.message.content = [{ type: "tool-result", content: [{ type: "text", text: "…" }] }]`
|
|
8
|
+
* 早期实现只看第一层 `block.text`,于是**工具结果文本永远是空串**:一条 bug 同时打死
|
|
9
|
+
* 自动沉淀(0 候选)、失败识别(信号永远"无失败")、grep 命中的证据记账、否定断言的证据底账。
|
|
10
|
+
* 所以这里统一**递归收集**(限深 3 层,防环)。
|
|
5
11
|
*/
|
|
12
|
+
function collectText(blocks, out, depth = 0) {
|
|
13
|
+
if (!Array.isArray(blocks) || depth > 3)
|
|
14
|
+
return;
|
|
15
|
+
for (const block of blocks) {
|
|
16
|
+
if (!block || typeof block !== "object")
|
|
17
|
+
continue;
|
|
18
|
+
const record = block;
|
|
19
|
+
if (typeof record.text === "string")
|
|
20
|
+
out.push(record.text);
|
|
21
|
+
if (record.content !== undefined)
|
|
22
|
+
collectText(record.content, out, depth + 1);
|
|
23
|
+
}
|
|
24
|
+
}
|
|
6
25
|
export function messageText(message) {
|
|
26
|
+
const content = message?.content;
|
|
27
|
+
const parts = [];
|
|
28
|
+
collectText(content, parts);
|
|
29
|
+
return parts.join(" ").trim();
|
|
30
|
+
}
|
|
31
|
+
/** 内部推理块类型:这些不是「用户看到的话」,不参与漂移/交付类判定。 */
|
|
32
|
+
const NON_VISIBLE_BLOCK_TYPES = new Set(["reasoning", "thinking", "analysis", "chain_of_thought"]);
|
|
33
|
+
/** 工具收发的块:属于「工具说了什么」,不是「助手说了什么」——可见正文里要整块排除。 */
|
|
34
|
+
const TOOL_BLOCK_TYPES = new Set(["tool-call", "tool-result"]);
|
|
35
|
+
/**
|
|
36
|
+
* 只取「用户可见的正文」——排除 reasoning / thinking 这类内部推理块,以及工具收发块。
|
|
37
|
+
*
|
|
38
|
+
* 为什么要分开:漂移检测的现场事故正是**扫到了推理文本**。模型在推理里权衡「要不要删、
|
|
39
|
+
* 会不会割接」,被当成「它要删」并顶了一句「收回」,于是它开始在推理里躲词(实测原文:
|
|
40
|
+
* 「不提割接/迁移/替换」「为了安全我换措辞」)。判断「它打算做什么」看可见回答;
|
|
41
|
+
* 「它想了什么」不归插件管;「工具输出」由 messageText 负责。
|
|
42
|
+
*/
|
|
43
|
+
export function visibleText(message) {
|
|
7
44
|
const content = message?.content;
|
|
8
45
|
if (!Array.isArray(content))
|
|
9
46
|
return "";
|
|
10
47
|
const parts = [];
|
|
11
48
|
for (const block of content) {
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
49
|
+
if (!block || typeof block !== "object")
|
|
50
|
+
continue;
|
|
51
|
+
const record = block;
|
|
52
|
+
const type = typeof record.type === "string" ? record.type : "";
|
|
53
|
+
if (NON_VISIBLE_BLOCK_TYPES.has(type) || TOOL_BLOCK_TYPES.has(type))
|
|
54
|
+
continue;
|
|
55
|
+
if (typeof record.text === "string")
|
|
56
|
+
parts.push(record.text);
|
|
15
57
|
}
|
|
16
58
|
return parts.join(" ").trim();
|
|
17
59
|
}
|