dsh-jev-prune 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/LICENSE +21 -0
- package/README.md +177 -0
- package/README_zh.md +175 -0
- package/assets/banner.png +0 -0
- package/assets/banner.svg +73 -0
- package/assets/demo-poster.png +0 -0
- package/assets/demo.cast +6 -0
- package/assets/demo.gif +0 -0
- package/assets/demo.json +51 -0
- package/assets/two-layers.png +0 -0
- package/assets/two-layers.svg +118 -0
- package/cordis.patch.yml +3 -0
- package/demo/README.md +23 -0
- package/demo/fixtures.mjs +301 -0
- package/demo/render.py +31 -0
- package/demo/run.mjs +85 -0
- package/docs/ARCHITECTURE.md +63 -0
- package/docs/CONTRIBUTING.md +48 -0
- package/docs/PORTING.md +60 -0
- package/docs/RELEASING.md +27 -0
- package/docs/implementation.md +310 -0
- package/docs/measurements.md +16 -0
- package/docs/review-2026-10-05.md +44 -0
- package/examples/README.md +7 -0
- package/examples/minimal.yml +10 -0
- package/index.js +2043 -0
- package/package.json +77 -0
- package/scripts/inspect_session.mjs +235 -0
- package/scripts/verify_layout.mjs +24 -0
- package/scripts/verify_real_shapes.mjs +99 -0
- package/scripts/wire_profile.mjs +173 -0
- package/src/jev.js +320 -0
- package/src/prune.js +383 -0
- package/src/receipt.js +800 -0
- package/src/state.js +532 -0
- package/test/check.js +1971 -0
- package/test/smoke_apply.mjs +1359 -0
package/src/receipt.js
ADDED
|
@@ -0,0 +1,800 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* receipt.js —— 第二层:**整对调用的「回执压缩」**(纯函数,无 DSH 依赖)。
|
|
3
|
+
*
|
|
4
|
+
* 第一层(prune.js)只裁工具结果的**正文**,调用记录与结果的外壳都留着。
|
|
5
|
+
* 这一层把**整个工具调用对**(assistant 消息里的 tool-call + 对应的 tool/result)
|
|
6
|
+
* 移出 surface,用一个**确定性回执**顶替原本由模型写的摘要。
|
|
7
|
+
*
|
|
8
|
+
* 为什么值得做:
|
|
9
|
+
* DSH 自带的压缩会让主模型读原始历史、**写一段摘要**。摘要会幻觉——它可能写出
|
|
10
|
+
* 「我们确认了 bug 在 X」这种历史里并没有的断言。而回执的每一行都是代码算出来的
|
|
11
|
+
* 事实(工具名、命令、路径、字符数、seq),**不可能包含模型推断**。
|
|
12
|
+
* 代价是它不解释、不总结——它是一个**可索引的收据**,不是一个摘要。
|
|
13
|
+
*
|
|
14
|
+
* 为什么敢删整对:
|
|
15
|
+
* ① 原始事件仍在会话日志里(`compactRegion` 只把它们移出 surface,不销毁)
|
|
16
|
+
* ② 回执**逐字记录了命令/路径**,所以"发生过什么副作用"这个事实不会丢
|
|
17
|
+
* ③ 只处理只读探查类工具,且 Jev 必须同时判定"结果可丢"且"没有副作用"
|
|
18
|
+
*
|
|
19
|
+
* 为什么不能随便删(实测教训,见 README):
|
|
20
|
+
* · `Edit`/`Write` 这类改写型调用是承重信息,判错一次就丢失关键改动 → 硬规则排除
|
|
21
|
+
* · 含 error/assert/fail 等**证据词**的结果可能正是根因所在 → 证据守卫生效时不删
|
|
22
|
+
* · 落在最近 compactPreserveRecent 个节点内的(含正在进行的调用)→ 一律不碰
|
|
23
|
+
* · 有**副作用**的调用(写文件、提交、安装、删除)→ 结果可能很短,但"发生过"是承重的
|
|
24
|
+
*
|
|
25
|
+
* 工具配对平衡:镜像 DSH 的 `@deepseek-ai/dsh-compaction/tool-pairing`。
|
|
26
|
+
* `compactRegion` 会在服务端再做一次同样的校验并抛错;我们自己先算一遍,
|
|
27
|
+
* 是为了**在花钱之前**就把非法范围筛掉,而不是等 API 报错。
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import { cachedVerdictForEvent, countChars, isToolIn, normalizeToolName } from './prune.js'
|
|
31
|
+
|
|
32
|
+
// 归一化工具在 prune.js(依赖链最底层)——两层的安全黑名单共用同一套比较。
|
|
33
|
+
// 这里转出以保持既有 import 路径(check.js / 调用方)不变。
|
|
34
|
+
export { cachedVerdictForEvent, isToolIn, normalizeToolName }
|
|
35
|
+
|
|
36
|
+
/** 回执文本的识别前缀(用于判断某个 checkpoint 是不是我们写的)。 */
|
|
37
|
+
export const RECEIPT_MARKER = '[已压缩 · 确定性回执]'
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* 真实 DSH 只读探查类工具名(**取证得到,不是猜的**:解压会话日志读到的是 `read` / `glob` / `grep`)。
|
|
41
|
+
*
|
|
42
|
+
* ⚠️ **刻意不含 shell**(`pwsh` / `bash` / `sh` / `shell`):shell 能做任何事,
|
|
43
|
+
* `pwsh: Remove-Item important.txt` 就是合法调用 —— 外部审查复现过这个案例。
|
|
44
|
+
* "整对移出 shell 调用"不是只读安全操作,哪怕回执逐字记录了命令。
|
|
45
|
+
*
|
|
46
|
+
* 这份清单就是 `DEFAULT_COMPACT_TOOLS` 的内容(默认生效)。
|
|
47
|
+
* 想放宽就把它配成 `[]`(只受 neverCompactTools 约束)—— 那是显式 opt-in 的不安全模式。
|
|
48
|
+
*
|
|
49
|
+
* 历史教训(保留给维护者):我最初把 Claude Code 风格的 PascalCase 名字当默认白名单,
|
|
50
|
+
* 在真实 DSH 里**命中 0/11** —— 第二层静默地永不触发。所以白名单比较必须走
|
|
51
|
+
* `normalizeToolName`,且命中/拦截的名字都要进 `selectReceiptRanges` 的 stats 上报,
|
|
52
|
+
* 让"白名单没配上"可观测。
|
|
53
|
+
*/
|
|
54
|
+
export const DSH_READONLY_TOOLS = [
|
|
55
|
+
'read', 'view', 'cat', 'glob', 'grep', 'list', 'ls', 'search', 'find', 'tree', 'stat',
|
|
56
|
+
'fetch', 'websearch', 'web_search',
|
|
57
|
+
'getcontent', 'getchilditem', 'getitem', 'selectstring', 'testpath', 'measureobject', 'resolvepath',
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
/** 默认白名单 = 上面的只读工具集(安全默认)。设为 `[]` 可放宽为只受黑名单约束。 */
|
|
61
|
+
export const DEFAULT_COMPACT_TOOLS = [...DSH_READONLY_TOOLS]
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* 相对分位模式需要的**最小总体规模**;低于它则降级为绝对下限模式(见 computeEligibleSeqs)。
|
|
65
|
+
* 独立导出是为了让 index.js 的 schema 与 receipt.js 的运行时兜底共用同一个数字——
|
|
66
|
+
* 两处各写一个字面量的话,改一处就会漂移。
|
|
67
|
+
*/
|
|
68
|
+
export const DEFAULT_MIN_CANDIDATES_FOR_RELATIVE = 4
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* 降级模式(总体 < 上面那个数)的**绝对下限阈值**。明显严于 compactThreshold(0.5):
|
|
72
|
+
* 没有相对排序信息时,只能靠"概率本身够不够低"来替代,所以阈值必须更保守。
|
|
73
|
+
*/
|
|
74
|
+
export const DEFAULT_FLOOR_THRESHOLD = 0.2
|
|
75
|
+
|
|
76
|
+
/** 降级模式仍要求的最低样本量:低于它连"分布"都谈不上,宁可不做。 */
|
|
77
|
+
export const DEFAULT_MIN_CANDIDATES_FOR_FLOOR = 2
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* 证据守卫必须扫描当前结果及其完整来源链。
|
|
81
|
+
*
|
|
82
|
+
* 第一层会把超长 tool/result 替换成“头 + 标记 + 尾”;若 error/assert/fail 恰好位于
|
|
83
|
+
* 被截掉的中间,只扫描 replacement 会让第二层的确定性守卫失明。DSH 保留所有旧事件,
|
|
84
|
+
* 且 replacement 通过 sourceEventSeqs 指回它们,所以这里沿来源链取回原文。
|
|
85
|
+
*/
|
|
86
|
+
function resultEvidenceText(eventAt, startSeq) {
|
|
87
|
+
const pending = [startSeq]
|
|
88
|
+
const seen = new Set()
|
|
89
|
+
const texts = []
|
|
90
|
+
while (pending.length > 0) {
|
|
91
|
+
const seq = pending.pop()
|
|
92
|
+
if (seen.has(seq)) continue
|
|
93
|
+
seen.add(seq)
|
|
94
|
+
const event = eventAt(seq)
|
|
95
|
+
if (event == null) continue
|
|
96
|
+
if (event.type === 'tool/result') texts.push(resultText(event))
|
|
97
|
+
for (const sourceSeq of event.sourceEventSeqs ?? []) pending.push(sourceSeq)
|
|
98
|
+
}
|
|
99
|
+
return texts.join('\n')
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* 永不**整对移出**的工具(第二层黑名单):改写型调用是承重信息
|
|
104
|
+
* (第一版实测 Jev 误删过 Edit)。比较时归一化。
|
|
105
|
+
*
|
|
106
|
+
* 为什么必须与第一层的默认黑名单分开(issue #31):两层的破坏性根本不同。
|
|
107
|
+
* 第一层只是**截断**——原文留在会话日志里,仍有损但可逆;
|
|
108
|
+
* 第二层是**整对移出**——调用与结果一起从 surface 上消失。
|
|
109
|
+
*
|
|
110
|
+
* 而这一组里两者兼有:`Write` / `NotebookEdit`(参数含完整内容)确实该两层都守;
|
|
111
|
+
* `Edit` / `MultiEdit` / `ApplyPatch` / `str_replace_*` / `apply_patch` 的参数
|
|
112
|
+
* 只是 **find/replace 差异**,第一层截掉的往往正是那段差异——但截断仍能靠重读文件
|
|
113
|
+
* 恢复;整对移出则会把"我改了什么"这个事实一起抹掉。
|
|
114
|
+
*
|
|
115
|
+
* 所以第二层保留全部,第一层放宽到只守"参数即内容"的那两个。
|
|
116
|
+
* 共用同一个常量会让第一层的意图(尽量多裁)被第二层的意图(尽量少删)绑住。
|
|
117
|
+
*/
|
|
118
|
+
export const DEFAULT_NEVER_COMPACT_TOOLS = [
|
|
119
|
+
'Edit', 'Write', 'MultiEdit', 'ApplyPatch', 'NotebookEdit',
|
|
120
|
+
'str_replace_editor', 'str_replace_based_edit_tool', 'apply_patch',
|
|
121
|
+
]
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* 第一层(截断)永不触碰的工具。比第二层黑名单**窄**:只守住"参数本身就是内容"
|
|
125
|
+
* 的写文件类工具——截掉它们的入参等于把写进去的内容丢了,且日志里未必还有第二份。
|
|
126
|
+
* 差异型编辑工具(Edit 等)的参数是 diff,第一层截断后靠重读文件可以恢复,
|
|
127
|
+
* 因此不在这里(第二层仍然守)。
|
|
128
|
+
*/
|
|
129
|
+
export const DEFAULT_NEVER_PRUNE_TOOLS = ['Write', 'NotebookEdit']
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* 证据守卫的默认词表。命中即**不整对删除**(仍允许第一层截断,那是有损但可逆的)。
|
|
133
|
+
* 方向性说明:这里宁可误守(少省一点)也不能漏守(丢证据)。
|
|
134
|
+
* 命中数量会如实上报到状态里,如果发现守得过宽可以直接改配置。
|
|
135
|
+
* 匹配口径见 scanEvidence —— 要求命中落在标识符段开头,所以 `bug` 不会被 `debug` 触发。
|
|
136
|
+
*/
|
|
137
|
+
export const DEFAULT_EVIDENCE_PATTERNS = [
|
|
138
|
+
'error', 'exception', 'traceback', 'stack trace', 'assertion',
|
|
139
|
+
'failed', 'fail:', 'panic', 'unexpected', 'mismatch',
|
|
140
|
+
'todo', 'fixme', 'bug',
|
|
141
|
+
]
|
|
142
|
+
|
|
143
|
+
// ---------------------------------------------------------------- 工具配对平衡
|
|
144
|
+
|
|
145
|
+
/** 一个 surface 事件对"未闭合工具调用数"的增量(与 DSH 同口径)。 */
|
|
146
|
+
export function eventDelta(event) {
|
|
147
|
+
if (event?.type === 'assistant/message') {
|
|
148
|
+
const content = event.data?.message?.content
|
|
149
|
+
if (!Array.isArray(content)) return 0
|
|
150
|
+
return content.filter((block) => block?.type === 'tool-call').length
|
|
151
|
+
}
|
|
152
|
+
if (event?.type === 'tool/result') return -1
|
|
153
|
+
return 0
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* 计算每个切点的平衡性。N 个 surface 节点有 N+1 个切点:
|
|
158
|
+
* 下标 i 表示"第 i 个节点之前"的切点,下标 N 表示"尾部之后"。
|
|
159
|
+
*
|
|
160
|
+
* @returns {Array<{balanced:boolean, inProgress:number}>} 长度为 N+1
|
|
161
|
+
* @throws 当 surface 损坏(节点的 seq 无对应事件 / 出现无主的 tool/result)时
|
|
162
|
+
*/
|
|
163
|
+
export function computeCuts(surface, eventAt) {
|
|
164
|
+
const cuts = [{ balanced: true, inProgress: 0 }]
|
|
165
|
+
let inProgress = 0
|
|
166
|
+
for (const seq of surface) {
|
|
167
|
+
const event = eventAt(seq)
|
|
168
|
+
if (event == null || event.seq !== seq) {
|
|
169
|
+
throw new Error(`tool-pairing: surface seq ${seq} 没有对应的会话事件(surface 损坏)`)
|
|
170
|
+
}
|
|
171
|
+
inProgress += eventDelta(event)
|
|
172
|
+
if (inProgress < 0) {
|
|
173
|
+
throw new Error(`tool-pairing: surface seq ${seq} 的 tool/result 没有对应的 tool-call(surface 损坏)`)
|
|
174
|
+
}
|
|
175
|
+
cuts.push({ balanced: inProgress === 0, inProgress })
|
|
176
|
+
}
|
|
177
|
+
return cuts
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/** 第 index 个节点之前的切点是否平衡(等价于 DSH 的 toolPairingBalancedBefore)。 */
|
|
181
|
+
export function balancedBefore(cuts, index) {
|
|
182
|
+
return cuts[index]?.balanced === true
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/** 第 index 个节点之后的切点是否平衡(等价于 DSH 的 toolPairingBalancedAfter)。 */
|
|
186
|
+
export function balancedAfter(cuts, index) {
|
|
187
|
+
return cuts[index + 1]?.balanced === true
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
// ---------------------------------------------------------------- 事件解析
|
|
191
|
+
|
|
192
|
+
/** assistant 消息里的 tool-call 块(带 seq 以便定位)。 */
|
|
193
|
+
function toolCallsOf(event) {
|
|
194
|
+
if (event?.type !== 'assistant/message') return []
|
|
195
|
+
const content = event.data?.message?.content
|
|
196
|
+
if (!Array.isArray(content)) return []
|
|
197
|
+
return content.filter((block) => block?.type === 'tool-call')
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
function toolCallIdOf(block) {
|
|
201
|
+
return block?.id ?? block?.callId ?? block?.toolCallId ?? null
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
function resultCallIdOf(event) {
|
|
205
|
+
return event?.data?.message?.source?.callId ?? event?.data?.callId ?? null
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* 按块类型分别统计 assistant 消息的字符数。
|
|
210
|
+
*
|
|
211
|
+
* ⚠️ **text 与 reasoning 必须分开统计、分开设阈值**(这是 issue #26 的修复)。
|
|
212
|
+
*
|
|
213
|
+
* 历史经过(保留给维护者,避免又合并回去):
|
|
214
|
+
* 最初只数 `text` —— 实测 DSH 的 assistant 消息块是 `[reasoning, tool-call]`,
|
|
215
|
+
* 模型把思考写在 reasoning 里、text 常常只有一句"看一下。",于是门控形同虚设,
|
|
216
|
+
* 一条思考了两千字的步骤会被当成纯探查删掉。于是改成"text + reasoning 累加"。
|
|
217
|
+
* 但两者**性质完全不同**:`text` 是用户可见的结论/说明(长 = 有信息量,该守),
|
|
218
|
+
* `reasoning` 是模型的草稿(长 = 想得多,**不代表这一步有承重信息**)。
|
|
219
|
+
* 合并累加的直接后果是阈值被 reasoning 主导:实测 DeepSeek 单步
|
|
220
|
+
* text 0~287、reasoning 0~1207 字符,合并后卡在 1200 的右端——
|
|
221
|
+
* **只要模型在思考上多花点字,第二层整层静默关闭**(实测 reasoning=1300 时
|
|
222
|
+
* 10 步里合格 0 步,且无任何日志提示是这道门导致的)。
|
|
223
|
+
* 分开之后,各自按自己的分布标定,两个信号互不污染。
|
|
224
|
+
*
|
|
225
|
+
* @returns {{text:number, reasoning:number}} 各自独立的字符数(码点计)
|
|
226
|
+
*/
|
|
227
|
+
function assistantBlockChars(event) {
|
|
228
|
+
const content = event.data?.message?.content
|
|
229
|
+
if (!Array.isArray(content)) return { text: 0, reasoning: 0 }
|
|
230
|
+
let text = 0
|
|
231
|
+
let reasoning = 0
|
|
232
|
+
for (const block of content) {
|
|
233
|
+
if (typeof block?.text !== 'string') continue
|
|
234
|
+
const chars = Array.from(block.text).length
|
|
235
|
+
if (block.type === 'text') text += chars
|
|
236
|
+
else if (block.type === 'reasoning') reasoning += chars
|
|
237
|
+
}
|
|
238
|
+
return { text, reasoning }
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
/** tool-call 块的 id(不同版本字段名不同,逐个容忍)。
|
|
242
|
+
* 已删除:callBlockId 此前导出但全仓无引用(issue #14 死代码)。 */
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* 把一次调用的入参渲染成一行可读文本。
|
|
246
|
+
*
|
|
247
|
+
* 先取"最有信息量"的字段(命令 / 路径 / 模式),再补上**取值很短的其它字段**
|
|
248
|
+
* (例如 Bash 的 `cwd`)—— 后者对"能不能重跑这条命令"是必要的,不该被静默丢掉。
|
|
249
|
+
* 体积型字段(正文/替换内容/子任务提示词)刻意跳过:那些是载荷,不是可复现性信息。
|
|
250
|
+
*/
|
|
251
|
+
const ARG_PRIORITY = ['command', 'file_path', 'notebook_path', 'pattern', 'query', 'path', 'url', 'glob']
|
|
252
|
+
const ARG_IGNORED = new Set(['content', 'old_string', 'new_string', 'prompt', 'description'])
|
|
253
|
+
const ARG_SECONDARY_MAX_CHARS = 40
|
|
254
|
+
|
|
255
|
+
export function renderCallArgs(block, maxChars = 120) {
|
|
256
|
+
let args = block?.arguments
|
|
257
|
+
if (typeof args === 'string') {
|
|
258
|
+
try {
|
|
259
|
+
args = JSON.parse(args)
|
|
260
|
+
} catch {
|
|
261
|
+
return clip(args, maxChars)
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
if (args == null) return ''
|
|
265
|
+
if (typeof args !== 'object') return clip(String(args), maxChars)
|
|
266
|
+
|
|
267
|
+
const parts = []
|
|
268
|
+
const used = new Set()
|
|
269
|
+
for (const key of ARG_PRIORITY) {
|
|
270
|
+
if (args[key] == null) continue
|
|
271
|
+
parts.push(String(args[key]))
|
|
272
|
+
used.add(key)
|
|
273
|
+
if (parts.length >= 3) break
|
|
274
|
+
}
|
|
275
|
+
if (parts.length < 3) {
|
|
276
|
+
for (const [key, value] of Object.entries(args)) {
|
|
277
|
+
if (used.has(key) || ARG_IGNORED.has(key) || value == null) continue
|
|
278
|
+
const text = String(value)
|
|
279
|
+
if (text.length > ARG_SECONDARY_MAX_CHARS) continue
|
|
280
|
+
// 次要字段带 key 前缀:不写 key 的话,光看值不知道它是哪个参数
|
|
281
|
+
parts.push(`${key}=${text}`)
|
|
282
|
+
if (parts.length >= 3) break
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
if (parts.length === 0) {
|
|
286
|
+
// 没有任何可用字段(全是载荷型)→ 整体序列化,至少让形状可见
|
|
287
|
+
return clip(JSON.stringify(args), maxChars)
|
|
288
|
+
}
|
|
289
|
+
return clip(parts.join(' '), maxChars)
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
function clip(text, maxChars) {
|
|
293
|
+
const value = String(text ?? '').replace(/\s+/g, ' ').trim()
|
|
294
|
+
const points = Array.from(value)
|
|
295
|
+
if (points.length <= maxChars) return value
|
|
296
|
+
return `${points.slice(0, maxChars).join('')}…`
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
// ---------------------------------------------------------------- 证据守卫
|
|
300
|
+
|
|
301
|
+
/**
|
|
302
|
+
* 找出文本命中的证据词。
|
|
303
|
+
*
|
|
304
|
+
* 匹配口径:命中必须落在**一个标识符段的开头** —— 串首、非字母数字之后,
|
|
305
|
+
* 或者驼峰分界(小写字母之后的大写开头)。这样:
|
|
306
|
+
* · `TypeError` 里的 `error` 命中(真证据)
|
|
307
|
+
* · `debug` / `__debug__` / `this.debug` 里的 `bug` **不**命中(子串误报,issue #1)
|
|
308
|
+
* 为什么不是"整词匹配"(词首词尾都要求边界):那会让 `errors`、`bugfix`、
|
|
309
|
+
* `failures` 这类真证据一起漏掉,而本守卫的方向性是宁可误守也不能漏守。
|
|
310
|
+
* 代价是 `myerror` 这种无分隔符的复合标识符不再命中 —— 与 `debug` 同类,属于有意取舍。
|
|
311
|
+
*
|
|
312
|
+
* @returns {{hit:boolean, matches:string[]}}
|
|
313
|
+
*/
|
|
314
|
+
export function scanEvidence(text, patterns) {
|
|
315
|
+
const haystack = String(text ?? '') // 保持原始大小写:驼峰分界只能对原文判定
|
|
316
|
+
const matches = []
|
|
317
|
+
for (const pattern of patterns ?? []) {
|
|
318
|
+
const needle = String(pattern).toLowerCase()
|
|
319
|
+
if (needle.length === 0) continue
|
|
320
|
+
const escaped = needle.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
|
|
321
|
+
const atSegmentStart = new RegExp(`(?<=^|[^A-Za-z0-9])${escaped}`, 'i')
|
|
322
|
+
const camelFirst = escaped.charAt(0).toUpperCase() + escaped.slice(1)
|
|
323
|
+
const atCamelBoundary = new RegExp(`(?<=[a-z])${camelFirst}`)
|
|
324
|
+
if (atSegmentStart.test(haystack) || atCamelBoundary.test(haystack)) matches.push(needle)
|
|
325
|
+
}
|
|
326
|
+
return { hit: matches.length > 0, matches }
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
// ---------------------------------------------------------------- 相对分位门控
|
|
330
|
+
/**
|
|
331
|
+
* **为什么不能用一个固定的绝对阈值**(开发期实测,数据未随仓库提交,结论与 README 一致):
|
|
332
|
+
*
|
|
333
|
+
* 在 10 个真假已知的调用上(5 个有副作用、5 个纯只读):
|
|
334
|
+
* 轴 有副作用组均值 纯只读组均值 间距 排序正确率
|
|
335
|
+
* result 0.076 0.160 -0.084 0%
|
|
336
|
+
* effect 0.274 0.068 +0.206 100%
|
|
337
|
+
*
|
|
338
|
+
* effect 轴的**排序是完美的**(有副作用组最低 0.220 > 只读组最高 0.080),
|
|
339
|
+
* 但**所有值都落在 0.5 以下** —— 拿 0.5 当阈值会把 10 条全判成"无副作用"。
|
|
340
|
+
* 这是 Jev 概率被压在一个窄带里的必然结果(同 dsh-compact 实测:分布密集区
|
|
341
|
+
* 不能放阈值)。所以第二层用**相对分位**:只在"本次会话已判定集合"的尾部取。
|
|
342
|
+
*
|
|
343
|
+
* 另一条实测结论:`result` 轴被**体量**污染(只读结果往往更大,于是"要保留"的分更高)。
|
|
344
|
+
* 相对分位能抵消掉一部分全局偏置,但不能抵消这种相关性——所以两轴取**交集**
|
|
345
|
+
* (必须同时在两轴的尾部),而不是取并集。
|
|
346
|
+
*
|
|
347
|
+
* @param {Array<{seq:number, prob?:number, effectProb?:number}>} verdicts
|
|
348
|
+
* @param {{quantile:number, minCandidates:number, minCandidatesForAbsolute?:number,
|
|
349
|
+
* floorThreshold?:number, onNote?:(note:string)=>void}} options
|
|
350
|
+
* @returns {Set<number>} 同时落在两轴尾部 quantile 的 seq 集合
|
|
351
|
+
* @throws quantile 非法时抛错(issue #6:此前 NaN/undefined 会静默返回空集——
|
|
352
|
+
* 第二层"功能静默死亡",而 compactPass 还会把它渲染成"样本不够",误导排查)
|
|
353
|
+
*/
|
|
354
|
+
export function computeEligibleSeqs(verdicts, {
|
|
355
|
+
quantile, minCandidates,
|
|
356
|
+
// ⚠️ 必须引用导出常量,不能写字面量。此前这里是硬编码的 `3` / `0.2`,
|
|
357
|
+
// 于是改 DEFAULT_MIN_CANDIDATES_FOR_FLOOR 时**这个函数的默认值不会跟着变** ——
|
|
358
|
+
// 插件实体走 resolveConfig(读常量,行为会变),而直接调本函数的地方(含 check.js)
|
|
359
|
+
// 吃的是旧字面量,两边悄悄分叉。实测:常量改成 2 后 check.js 仍全绿,
|
|
360
|
+
// 因为它绕开了配置路径,测的是另一个数。
|
|
361
|
+
minCandidatesForAbsolute = DEFAULT_MIN_CANDIDATES_FOR_FLOOR,
|
|
362
|
+
floorThreshold = DEFAULT_FLOOR_THRESHOLD,
|
|
363
|
+
onNote,
|
|
364
|
+
} = {}) {
|
|
365
|
+
if (!Number.isFinite(quantile) || quantile < 0 || quantile > 1) {
|
|
366
|
+
throw new Error(`compactQuantile 非法:${quantile}(必须是 0~1 的有限数值;配置留空/解析成 null 都会走到这里)`)
|
|
367
|
+
}
|
|
368
|
+
// 0 是显式关闭值;必须先于小样本降级,否则低分候选会绕过开关。
|
|
369
|
+
if (quantile === 0) {
|
|
370
|
+
onNote?.('compactQuantile=0 → 相对分位选择已显式关闭')
|
|
371
|
+
return new Set()
|
|
372
|
+
}
|
|
373
|
+
const usable = (verdicts ?? []).filter(
|
|
374
|
+
// 必须用 Number.isFinite:typeof NaN === 'number',用 typeof 会让 NaN 混进总体,
|
|
375
|
+
// 而排序比较 (a-b) 返回 NaN 时被 V8 当作"相等"不换位 → NaN 项按**数组位置**
|
|
376
|
+
// 混进尾部,可能被选中做整对移出。口径与上面的 quantile 校验统一。
|
|
377
|
+
(v) => Number.isFinite(v?.prob) && Number.isFinite(v?.effectProb),
|
|
378
|
+
)
|
|
379
|
+
if (usable.length === 0) return new Set()
|
|
380
|
+
|
|
381
|
+
// ── 小总体分支(issue #27 修复)────────────────────────────────────────────
|
|
382
|
+
// 此前 `usable.length < minCandidates` 直接返回空集:只读工具在写/执行密集会话里
|
|
383
|
+
// 常常只占极少数(实测只读 1/6 → 总体仅 2 条),于是**第二层在绝大多数真实会话里
|
|
384
|
+
// 静默不工作**,而报错文案只说"需要 ≥4 个",看起来像"样本确实不够",不像 bug。
|
|
385
|
+
//
|
|
386
|
+
// 修法:小总体时降级为**绝对下限兜底**,而不是放弃。安全性论证——
|
|
387
|
+
// 分位排序携带的是"会话内相对位置"信息,需要总体;绝对下限携带的是"这个值本身
|
|
388
|
+
// 够不够低",不需要比较。两者正交,所以降级不是放宽安全标准,而是**换一个更弱的
|
|
389
|
+
// 信号**。为补偿这个信号更弱,这里加三重约束:
|
|
390
|
+
// ① floorThreshold 比 compactThreshold(0.5) 严格得多(默认 0.2)
|
|
391
|
+
// ② 仍然要求**两轴同时**满足(与相对模式口径一致——实测 result 轴被体量污染,
|
|
392
|
+
// 单看它会放过"结果很大但确实无用"的调用)
|
|
393
|
+
// ③ 样本比 minCandidatesForAbsolute 还少时仍然不做(1 条根本谈不上分布)
|
|
394
|
+
// 注意这在数学上**不是绝对禁止**:当相对阈值取到 0.5 附近时,尾部就是 prob<0.5,
|
|
395
|
+
// 与绝对阈值同构。降级分支只是把阈值收紧到 0.2 来补足"没有相对信息"这个缺口。
|
|
396
|
+
if (usable.length < Math.max(2, minCandidates ?? 4)) {
|
|
397
|
+
if (usable.length < Math.max(1, minCandidatesForAbsolute)) {
|
|
398
|
+
onNote?.(`总体仅 ${usable.length} 条,低于绝对下限模式的最低样本 ${minCandidatesForAbsolute} → 不做整对移出`)
|
|
399
|
+
return new Set()
|
|
400
|
+
}
|
|
401
|
+
const floor = usable.filter((v) => v.prob < floorThreshold && v.effectProb < floorThreshold)
|
|
402
|
+
onNote?.(`总体仅 ${usable.length} 条(< ${minCandidates})→ 降级为绝对下限模式:`
|
|
403
|
+
+ `要求两轴同时 < ${floorThreshold},命中 ${floor.length}/${usable.length} 条`)
|
|
404
|
+
return new Set(floor.map((v) => v.seq))
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
const take = Math.max(1, Math.floor(usable.length * quantile))
|
|
408
|
+
const tailOf = (key) => new Set(
|
|
409
|
+
[...usable]
|
|
410
|
+
.sort((a, b) => (a[key] - b[key]) || (a.seq - b.seq))
|
|
411
|
+
.slice(0, take)
|
|
412
|
+
.map((v) => v.seq),
|
|
413
|
+
)
|
|
414
|
+
const byResult = tailOf('prob')
|
|
415
|
+
const byEffect = tailOf('effectProb')
|
|
416
|
+
const intersection = new Set([...byResult].filter((seq) => byEffect.has(seq)))
|
|
417
|
+
if (intersection.size > 0) return intersection
|
|
418
|
+
|
|
419
|
+
const floor = usable.filter((v) => v.prob < floorThreshold && v.effectProb < floorThreshold)
|
|
420
|
+
onNote?.(`两轴尾部交集为空 → 降级为绝对下限模式:要求两轴同时 < ${floorThreshold},命中 ${floor.length}/${usable.length} 条`)
|
|
421
|
+
return new Set(floor.map((v) => v.seq))
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
// ---------------------------------------------------------------- 范围选择
|
|
425
|
+
|
|
426
|
+
/** 一个"步骤"= assistant 消息(含 ≥1 个 tool-call)+ 紧随其后的 1:1 tool/result。 */
|
|
427
|
+
function readStep(surface, eventAt, headIdx) {
|
|
428
|
+
const headSeq = surface[headIdx]
|
|
429
|
+
const head = eventAt(headSeq)
|
|
430
|
+
const calls = toolCallsOf(head)
|
|
431
|
+
if (calls.length === 0) return null
|
|
432
|
+
|
|
433
|
+
const resultIdx = []
|
|
434
|
+
for (let offset = 1; offset <= calls.length; offset += 1) {
|
|
435
|
+
const idx = headIdx + offset
|
|
436
|
+
if (idx >= surface.length) return { headIdx, headSeq, head, calls, resultIdx, complete: false }
|
|
437
|
+
if (eventAt(surface[idx])?.type !== 'tool/result') {
|
|
438
|
+
return { headIdx, headSeq, head, calls, resultIdx, complete: false }
|
|
439
|
+
}
|
|
440
|
+
resultIdx.push(idx)
|
|
441
|
+
}
|
|
442
|
+
const callIds = calls.map(toolCallIdOf)
|
|
443
|
+
const resultIds = resultIdx.map((idx) => resultCallIdOf(eventAt(surface[idx])))
|
|
444
|
+
const anyId = [...callIds, ...resultIds].some((id) => id != null)
|
|
445
|
+
let pairedResultIdx = [...resultIdx]
|
|
446
|
+
if (anyId) {
|
|
447
|
+
if (callIds.some((id) => id == null) || resultIds.some((id) => id == null)
|
|
448
|
+
|| new Set(callIds).size !== callIds.length || new Set(resultIds).size !== resultIds.length) {
|
|
449
|
+
return { headIdx, headSeq, head, calls, resultIdx, pairedResultIdx: [], complete: false }
|
|
450
|
+
}
|
|
451
|
+
const resultByCallId = new Map(resultIds.map((id, offset) => [id, resultIdx[offset]]))
|
|
452
|
+
pairedResultIdx = callIds.map((id) => resultByCallId.get(id))
|
|
453
|
+
if (pairedResultIdx.some((idx) => idx == null)) {
|
|
454
|
+
return { headIdx, headSeq, head, calls, resultIdx, pairedResultIdx: [], complete: false }
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
return { headIdx, headSeq, head, calls, resultIdx, pairedResultIdx, complete: true }
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
/**
|
|
461
|
+
* 挑出可以整对移出的范围(连续步骤合并成一段)。
|
|
462
|
+
*
|
|
463
|
+
* 依赖全部注入,所以能脱离 DSH 单测。
|
|
464
|
+
*
|
|
465
|
+
* @param {object} input
|
|
466
|
+
* @param {number[]} input.surface surface 上的 seq 列表
|
|
467
|
+
* @param {(seq:number)=>object} input.eventAt seq → 事件
|
|
468
|
+
* @param {Map<number,object>} input.cache 判定缓存:**按工具结果 seq** 索引
|
|
469
|
+
* @param {(seq:number, verdict:object)=>boolean} input.dropVerdict 判定"这条可以整对移出"
|
|
470
|
+
* @param {(event:object)=>string} input.toolNameOf
|
|
471
|
+
* @param {object} input.cfg 见下面用到的字段
|
|
472
|
+
* @returns {{ranges:Array, stats:object}}
|
|
473
|
+
*/
|
|
474
|
+
export function selectReceiptRanges({ surface, eventAt, cache, dropVerdict, cfg }) {
|
|
475
|
+
const stats = {
|
|
476
|
+
scanned: 0,
|
|
477
|
+
steps: 0,
|
|
478
|
+
eligibleSteps: 0,
|
|
479
|
+
skippedTail: 0,
|
|
480
|
+
skippedTool: 0,
|
|
481
|
+
/**
|
|
482
|
+
* 被工具规则排除的**调用名统计**。
|
|
483
|
+
*
|
|
484
|
+
* 为什么必须记这个:第二层用的是**白名单**,一旦真实工具名与默认白名单对不上
|
|
485
|
+
* (各宿主的命名风格差很多:Read / read / fs_read / Get-Content…),
|
|
486
|
+
* 整个功能会**静默地永不触发**——第一层用黑名单所以一直没暴露这个问题。
|
|
487
|
+
* 把名字如实统计出来并上报,才能一眼看出"是白名单没配上"而不是"模型判断不对"。
|
|
488
|
+
*/
|
|
489
|
+
blockedToolNames: {},
|
|
490
|
+
allowedToolNames: {},
|
|
491
|
+
skippedToolUnknown: 0,
|
|
492
|
+
skippedVerdict: 0,
|
|
493
|
+
skippedGuard: 0,
|
|
494
|
+
/** 用户可见文本过长的步骤数(text 轴) */
|
|
495
|
+
skippedText: 0,
|
|
496
|
+
/** 思考草稿过长的步骤数(reasoning 轴)——与 skippedText 分开计数,
|
|
497
|
+
* 否则排查时无法区分"是结论在守"还是"是草稿在守"(issue #26) */
|
|
498
|
+
skippedReasoning: 0,
|
|
499
|
+
skippedIncomplete: 0,
|
|
500
|
+
skippedShort: 0,
|
|
501
|
+
/** 同一 assistant 批次里仅部分 result 合格时,安全降级为逐结果回执的批次数。 */
|
|
502
|
+
partialSteps: 0,
|
|
503
|
+
/** 逐结果回执实际选中的 tool/result 数。 */
|
|
504
|
+
partialResults: 0,
|
|
505
|
+
/** 已经是确定性回执的结果,防止低阈值配置下重复压缩。 */
|
|
506
|
+
skippedReceipt: 0,
|
|
507
|
+
guardHits: [],
|
|
508
|
+
}
|
|
509
|
+
const cuts = computeCuts(surface, eventAt)
|
|
510
|
+
const lastAllowed = surface.length - 1 - cfg.preserveRecent
|
|
511
|
+
|
|
512
|
+
const steps = []
|
|
513
|
+
let index = 0
|
|
514
|
+
while (index < surface.length) {
|
|
515
|
+
stats.scanned += 1
|
|
516
|
+
const step = readStep(surface, eventAt, index)
|
|
517
|
+
if (step == null) {
|
|
518
|
+
index += 1
|
|
519
|
+
continue // 不是步骤头(user 消息 / 纯文本 assistant / checkpoint)
|
|
520
|
+
}
|
|
521
|
+
stats.steps += 1
|
|
522
|
+
steps.push(step)
|
|
523
|
+
// 无论是否合格都跳到这一步的末尾,避免把同一个调用数两遍。
|
|
524
|
+
// ⚠️ 这里必须用**位置**推进,不能用 seq 值 —— 一旦发生过 surface 替换
|
|
525
|
+
// (我们的第一层裁剪就会替换),seq 就不再等于「位置+1」了。
|
|
526
|
+
index = step.resultIdx.length > 0 ? step.resultIdx[step.resultIdx.length - 1] + 1 : index + 1
|
|
527
|
+
}
|
|
528
|
+
|
|
529
|
+
const eligible = []
|
|
530
|
+
const partial = []
|
|
531
|
+
for (const step of steps) {
|
|
532
|
+
const lastResultIdx = step.resultIdx[step.resultIdx.length - 1] ?? step.headIdx
|
|
533
|
+
const resultSeqs = (step.pairedResultIdx ?? step.resultIdx).map((idx) => surface[idx])
|
|
534
|
+
let reason = null
|
|
535
|
+
|
|
536
|
+
if (!step.complete) reason = 'incomplete'
|
|
537
|
+
else if (!balancedBefore(cuts, step.headIdx) || !balancedAfter(cuts, lastResultIdx)) reason = 'incomplete'
|
|
538
|
+
else {
|
|
539
|
+
// 两轴分开判(issue #26):text = 结论/说明(长则守),reasoning = 草稿(长则守)。
|
|
540
|
+
// 拆开的重点是**阈值各自标定**——合并累加会让 reasoning 的分布盖住 text 的语义。
|
|
541
|
+
const { text, reasoning } = assistantBlockChars(step.head)
|
|
542
|
+
if (text > cfg.maxStepTextChars) reason = 'text'
|
|
543
|
+
else if (reasoning > cfg.maxStepReasoningChars) reason = 'reasoning'
|
|
544
|
+
}
|
|
545
|
+
let hits = []
|
|
546
|
+
|
|
547
|
+
// 完整性和 assistant 文本门是整批属性;通过后,其余规则逐调用判断。
|
|
548
|
+
// 这样一个 6-read 批次中 5 条合格、1 条不合格时,5 条仍可各自替换为回执。
|
|
549
|
+
const pairs = reason == null
|
|
550
|
+
? step.calls.map((call, offset) => {
|
|
551
|
+
const resultIdx = (step.pairedResultIdx ?? step.resultIdx)[offset]
|
|
552
|
+
const seq = surface[resultIdx]
|
|
553
|
+
const event = eventAt(seq)
|
|
554
|
+
let pairReason = null
|
|
555
|
+
if (resultIdx > lastAllowed) pairReason = 'tail'
|
|
556
|
+
else if ((cfg.compactTools.length > 0 && !isToolIn(cfg.compactTools, call.name))
|
|
557
|
+
|| isToolIn(cfg.neverCompactTools, call.name)) pairReason = 'tool'
|
|
558
|
+
else if (resultText(event).includes(RECEIPT_MARKER)) pairReason = 'receipt'
|
|
559
|
+
else {
|
|
560
|
+
const verdict = cachedVerdictForEvent(cache, event)
|
|
561
|
+
if (verdict == null || !dropVerdict(seq, verdict)) pairReason = 'verdict'
|
|
562
|
+
}
|
|
563
|
+
let pairHits = []
|
|
564
|
+
if (pairReason == null && cfg.evidenceGuard) {
|
|
565
|
+
const scan = scanEvidence(resultEvidenceText(eventAt, seq), cfg.evidencePatterns)
|
|
566
|
+
if (scan.hit) {
|
|
567
|
+
pairHits = scan.matches
|
|
568
|
+
pairReason = 'guard'
|
|
569
|
+
}
|
|
570
|
+
}
|
|
571
|
+
return { call, resultIdx, seq, event, chars: resultCharsOf(event), reason: pairReason, hits: pairHits }
|
|
572
|
+
})
|
|
573
|
+
: []
|
|
574
|
+
|
|
575
|
+
if (reason == null) {
|
|
576
|
+
const selected = pairs.filter((pair) => pair.reason == null)
|
|
577
|
+
const rejected = pairs.filter((pair) => pair.reason != null)
|
|
578
|
+
if (selected.length === pairs.length && step.headIdx <= lastAllowed) {
|
|
579
|
+
// 原路径:整批合格,交给 compactRegion 一次移出 assistant + 全部 results。
|
|
580
|
+
} else if (selected.length > 0) {
|
|
581
|
+
const selectedInSurfaceOrder = [...selected].sort((a, b) => a.resultIdx - b.resultIdx)
|
|
582
|
+
const selectedStep = {
|
|
583
|
+
...step,
|
|
584
|
+
calls: selectedInSurfaceOrder.map((pair) => pair.call),
|
|
585
|
+
resultIdx: selectedInSurfaceOrder.map((pair) => pair.resultIdx),
|
|
586
|
+
resultSeqs: selectedInSurfaceOrder.map((pair) => pair.seq),
|
|
587
|
+
pairs: selectedInSurfaceOrder,
|
|
588
|
+
lastResultIdx: selectedInSurfaceOrder[selectedInSurfaceOrder.length - 1].resultIdx,
|
|
589
|
+
}
|
|
590
|
+
const chars = selected.reduce((sum, pair) => sum + pair.chars, 0)
|
|
591
|
+
partial.push({
|
|
592
|
+
kind: 'partial',
|
|
593
|
+
startIdx: selectedInSurfaceOrder[0].resultIdx,
|
|
594
|
+
endIdx: selectedInSurfaceOrder[selectedInSurfaceOrder.length - 1].resultIdx,
|
|
595
|
+
start: selectedInSurfaceOrder[0].seq,
|
|
596
|
+
end: selectedInSurfaceOrder[selectedInSurfaceOrder.length - 1].seq,
|
|
597
|
+
steps: [selectedStep],
|
|
598
|
+
chars,
|
|
599
|
+
})
|
|
600
|
+
stats.eligibleSteps += 1
|
|
601
|
+
stats.partialSteps += 1
|
|
602
|
+
stats.partialResults += selected.length
|
|
603
|
+
for (const pair of selected) {
|
|
604
|
+
const name = pair.call.name ?? 'unknown'
|
|
605
|
+
stats.allowedToolNames[name] = (stats.allowedToolNames[name] ?? 0) + 1
|
|
606
|
+
}
|
|
607
|
+
for (const pair of rejected) {
|
|
608
|
+
if (pair.reason === 'tail') stats.skippedTail += 1
|
|
609
|
+
else if (pair.reason === 'verdict') stats.skippedVerdict += 1
|
|
610
|
+
else if (pair.reason === 'guard') {
|
|
611
|
+
stats.skippedGuard += 1
|
|
612
|
+
stats.guardHits.push({ headSeq: step.headSeq, resultSeq: pair.seq, matches: pair.hits })
|
|
613
|
+
} else if (pair.reason === 'receipt') stats.skippedReceipt += 1
|
|
614
|
+
else if (pair.reason === 'tool') {
|
|
615
|
+
stats.skippedTool += 1
|
|
616
|
+
const name = pair.call.name ?? 'unknown'
|
|
617
|
+
if (name === 'unknown' || name === '') stats.skippedToolUnknown += 1
|
|
618
|
+
stats.blockedToolNames[name] = (stats.blockedToolNames[name] ?? 0) + 1
|
|
619
|
+
}
|
|
620
|
+
}
|
|
621
|
+
continue
|
|
622
|
+
} else {
|
|
623
|
+
reason = pairs[0]?.reason ?? (step.headIdx > lastAllowed ? 'tail' : 'verdict')
|
|
624
|
+
hits = pairs.find((pair) => pair.reason === 'guard')?.hits ?? []
|
|
625
|
+
}
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
if (reason != null) {
|
|
629
|
+
if (reason === 'tail') stats.skippedTail += 1
|
|
630
|
+
else if (reason === 'tool') {
|
|
631
|
+
stats.skippedTool += 1
|
|
632
|
+
// 只记**真正没通过工具门**的那个名字(issue #7):此前把整步的所有调用名
|
|
633
|
+
// 都记进 blockedToolNames,通过白名单的名字也被列为"被拦下",用户按
|
|
634
|
+
// jev_probe_shapes 的提示去"补配"一个本来就在白名单里的名字,而真正的
|
|
635
|
+
// 元凶淹没在同一份名单里——这个诊断字段就完成不了它的设计任务。
|
|
636
|
+
const offending = cfg.compactTools.length > 0
|
|
637
|
+
? step.calls.filter((call) => !isToolIn(cfg.compactTools, call.name))
|
|
638
|
+
: step.calls.filter((call) => isToolIn(cfg.neverCompactTools, call.name))
|
|
639
|
+
for (const call of offending) {
|
|
640
|
+
const name = call.name ?? 'unknown'
|
|
641
|
+
if (name === 'unknown' || name === '') stats.skippedToolUnknown += 1
|
|
642
|
+
stats.blockedToolNames[name] = (stats.blockedToolNames[name] ?? 0) + 1
|
|
643
|
+
}
|
|
644
|
+
} else if (reason === 'verdict') stats.skippedVerdict += 1
|
|
645
|
+
else if (reason === 'receipt') stats.skippedReceipt += 1
|
|
646
|
+
else if (reason === 'guard') {
|
|
647
|
+
stats.skippedGuard += 1
|
|
648
|
+
stats.guardHits.push({ headSeq: step.headSeq, matches: hits })
|
|
649
|
+
} else if (reason === 'text') stats.skippedText += 1
|
|
650
|
+
else if (reason === 'reasoning') stats.skippedReasoning += 1
|
|
651
|
+
else stats.skippedIncomplete += 1
|
|
652
|
+
continue
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
stats.eligibleSteps += 1
|
|
656
|
+
for (const call of step.calls) {
|
|
657
|
+
const name = call.name ?? 'unknown'
|
|
658
|
+
stats.allowedToolNames[name] = (stats.allowedToolNames[name] ?? 0) + 1
|
|
659
|
+
}
|
|
660
|
+
eligible.push({
|
|
661
|
+
kind: 'full',
|
|
662
|
+
...step,
|
|
663
|
+
resultSeqs,
|
|
664
|
+
lastResultIdx,
|
|
665
|
+
resultChars: resultSeqs.reduce((sum, seq) => sum + resultCharsOf(eventAt(seq)), 0),
|
|
666
|
+
tools: step.calls.map((call) => call.name ?? 'unknown'),
|
|
667
|
+
})
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
// 相邻的合格步骤合并成一段(合并后两端仍然平衡:步骤之间没有别的东西)
|
|
671
|
+
const merged = []
|
|
672
|
+
for (const step of eligible) {
|
|
673
|
+
const previous = merged[merged.length - 1]
|
|
674
|
+
if (previous != null && previous.endIdx + 1 === step.headIdx) {
|
|
675
|
+
previous.steps.push(step)
|
|
676
|
+
previous.endIdx = step.lastResultIdx
|
|
677
|
+
previous.end = surface[step.lastResultIdx]
|
|
678
|
+
previous.chars += step.resultChars
|
|
679
|
+
continue
|
|
680
|
+
}
|
|
681
|
+
merged.push({
|
|
682
|
+
kind: 'full',
|
|
683
|
+
startIdx: step.headIdx,
|
|
684
|
+
endIdx: step.lastResultIdx,
|
|
685
|
+
start: surface[step.headIdx],
|
|
686
|
+
end: surface[step.lastResultIdx],
|
|
687
|
+
steps: [step],
|
|
688
|
+
chars: step.resultChars,
|
|
689
|
+
})
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
// 省得不够多的范围不值得开一次压缩事务
|
|
693
|
+
const ranges = []
|
|
694
|
+
for (const range of [...merged, ...partial].sort((a, b) => a.startIdx - b.startIdx)) {
|
|
695
|
+
if (range.chars < cfg.compactMinChars) {
|
|
696
|
+
stats.skippedShort += 1
|
|
697
|
+
continue
|
|
698
|
+
}
|
|
699
|
+
ranges.push(range)
|
|
700
|
+
}
|
|
701
|
+
return { ranges, stats }
|
|
702
|
+
}
|
|
703
|
+
|
|
704
|
+
function resultText(event) {
|
|
705
|
+
const content = event?.data?.message?.content
|
|
706
|
+
if (!Array.isArray(content)) return ''
|
|
707
|
+
const block = content.find((item) => item?.type === 'tool-result')
|
|
708
|
+
const inner = block?.content
|
|
709
|
+
if (!Array.isArray(inner)) return ''
|
|
710
|
+
return inner
|
|
711
|
+
.filter((item) => item?.type === 'text' && typeof item.text === 'string')
|
|
712
|
+
.map((item) => item.text)
|
|
713
|
+
.join('\n')
|
|
714
|
+
}
|
|
715
|
+
|
|
716
|
+
function resultCharsOf(event) {
|
|
717
|
+
const content = event?.data?.message?.content
|
|
718
|
+
if (!Array.isArray(content)) return 0
|
|
719
|
+
const block = content.find((item) => item?.type === 'tool-result')
|
|
720
|
+
return countChars(block?.content)
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
// ---------------------------------------------------------------- 回执渲染
|
|
724
|
+
|
|
725
|
+
/**
|
|
726
|
+
* 取 assistant 消息的**可见文本**(`text` 块),折叠空白并按码点截断,不做任何推断。
|
|
727
|
+
*
|
|
728
|
+
* 为什么必须保留原文摘录(修复:长任务提前收工):
|
|
729
|
+
* 第二层把「调用 + 结果」整段移出 surface。若回执只留调用事实,模型自己写下的
|
|
730
|
+
* **结论与进度**会一并消失。`maxStepTextChars` 只拦「长文本」,而「每步一句短结论」
|
|
731
|
+
* 这类工作流的文本很短(实测 10–20 字),拦不住。
|
|
732
|
+
* 实测证据:37 步逐文件读取任务中,模型的 12/13 条中间输出被逐步抹掉,
|
|
733
|
+
* 随后在第 14 步只输出一句概括、不再发起工具调用,任务中途终止。
|
|
734
|
+
* 把模型原话摘录进回执,**不引入任何模型生成**,但保住模型的进行中状态。
|
|
735
|
+
*
|
|
736
|
+
* 注意:这里只折叠空白并截断,不改写措辞,也不作推断。测试中「回执不得含推断性表述」的断言
|
|
737
|
+
* 针对的是**插件生成**的内容;原话属于摘录,且由“模型原话(原文摘录)”标签标明来源。
|
|
738
|
+
*/
|
|
739
|
+
function assistantVisibleText(event, maxChars) {
|
|
740
|
+
if (maxChars <= 0) return ''
|
|
741
|
+
const content = event?.data?.message?.content
|
|
742
|
+
if (!Array.isArray(content)) return ''
|
|
743
|
+
const parts = []
|
|
744
|
+
for (const block of content) {
|
|
745
|
+
if (block?.type !== 'text' || typeof block.text !== 'string') continue
|
|
746
|
+
parts.push(block.text)
|
|
747
|
+
}
|
|
748
|
+
const text = parts.join(' ').replace(/\s+/g, ' ').trim()
|
|
749
|
+
if (text.length === 0) return ''
|
|
750
|
+
const chars = Array.from(text)
|
|
751
|
+
return chars.length > maxChars ? `${chars.slice(0, maxChars).join('')}…` : text
|
|
752
|
+
}
|
|
753
|
+
|
|
754
|
+
/**
|
|
755
|
+
* 渲染确定性回执。
|
|
756
|
+
*
|
|
757
|
+
* 每一行都是可以核对的**事实**:工具名、命令/路径(逐字截断)、输出字符数、seq。
|
|
758
|
+
* assistant 文本只做空白折叠与定长截断,不改写措辞;插件不生成结论。
|
|
759
|
+
*/
|
|
760
|
+
export function renderReceipt(range, { eventAt, argChars = 120, textChars = 400 } = {}) {
|
|
761
|
+
const lines = []
|
|
762
|
+
const count = range.steps.reduce((sum, step) => sum + step.calls.length, 0)
|
|
763
|
+
lines.push(
|
|
764
|
+
`${RECEIPT_MARKER} 原历史 s${range.start}–s${range.end} 是 ${count} 次工具调用`
|
|
765
|
+
+ `(共约 ${range.chars} 字符输出),为释放上下文已移出。以下为事实清单`
|
|
766
|
+
+ `(工具名/入参/字符数/seq 由代码算出;模型原话仅作原文摘录,插件不新增推断):`,
|
|
767
|
+
)
|
|
768
|
+
for (const step of range.steps) {
|
|
769
|
+
// ★ 修复:把该步模型自己的可见文本摘录带上,避免把模型的结论/进度一并抹掉。
|
|
770
|
+
const said = assistantVisibleText(step.head, textChars)
|
|
771
|
+
if (said) lines.push(`· s${step.headSeq} 模型原话(原文摘录):${said}`)
|
|
772
|
+
step.calls.forEach((call, offset) => {
|
|
773
|
+
const seq = step.resultSeqs[offset] ?? step.headSeq
|
|
774
|
+
const args = renderCallArgs(call, argChars)
|
|
775
|
+
const chars = resultCharsOf(eventAt(step.resultSeqs[offset]))
|
|
776
|
+
lines.push(`· s${seq} ${call.name ?? 'unknown'}${args ? `:${args}` : ''} → ${chars} 字符输出`)
|
|
777
|
+
})
|
|
778
|
+
}
|
|
779
|
+
lines.push(
|
|
780
|
+
`原始事件仍完整保存在会话日志中(seqs ${range.start}–${range.end})。`
|
|
781
|
+
+ '需要内容时重跑相同命令/读取相同文件即可;本回执不含对内容的解释。',
|
|
782
|
+
)
|
|
783
|
+
return lines.join('\n')
|
|
784
|
+
}
|
|
785
|
+
|
|
786
|
+
/**
|
|
787
|
+
* 为批量步骤中的单个合格结果生成短回执。
|
|
788
|
+
*
|
|
789
|
+
* 这里不删除 assistant/tool-call 外壳,只替换对应 tool/result 的正文。因此同批次
|
|
790
|
+
* 其他不合格结果可以原样留在 surface,且每次单节点 replace 前后都保持配对完整。
|
|
791
|
+
*/
|
|
792
|
+
export function renderPartialResultReceipt(pair, { argChars = 120 } = {}) {
|
|
793
|
+
const call = pair?.call ?? {}
|
|
794
|
+
const seq = pair?.seq ?? pair?.event?.seq ?? '?'
|
|
795
|
+
const chars = resultCharsOf(pair?.event)
|
|
796
|
+
const args = renderCallArgs(call, argChars)
|
|
797
|
+
return `${RECEIPT_MARKER} s${seq} ${call.name ?? 'unknown'}${args ? `:${args}` : ''}`
|
|
798
|
+
+ ` 的 ${chars} 字符输出已从当前上下文移出;原始事件仍保存在会话日志中,`
|
|
799
|
+
+ '需要内容时请重跑相同命令或重新读取相同文件。本回执不含对内容的解释。'
|
|
800
|
+
}
|