dsh-jev-prune 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/LICENSE +21 -0
- package/README.md +177 -0
- package/README_zh.md +175 -0
- package/assets/banner.png +0 -0
- package/assets/banner.svg +73 -0
- package/assets/demo-poster.png +0 -0
- package/assets/demo.cast +6 -0
- package/assets/demo.gif +0 -0
- package/assets/demo.json +51 -0
- package/assets/two-layers.png +0 -0
- package/assets/two-layers.svg +118 -0
- package/cordis.patch.yml +3 -0
- package/demo/README.md +23 -0
- package/demo/fixtures.mjs +301 -0
- package/demo/render.py +31 -0
- package/demo/run.mjs +85 -0
- package/docs/ARCHITECTURE.md +63 -0
- package/docs/CONTRIBUTING.md +48 -0
- package/docs/PORTING.md +60 -0
- package/docs/RELEASING.md +27 -0
- package/docs/implementation.md +310 -0
- package/docs/measurements.md +16 -0
- package/docs/review-2026-10-05.md +44 -0
- package/examples/README.md +7 -0
- package/examples/minimal.yml +10 -0
- package/index.js +2043 -0
- package/package.json +77 -0
- package/scripts/inspect_session.mjs +235 -0
- package/scripts/verify_layout.mjs +24 -0
- package/scripts/verify_real_shapes.mjs +99 -0
- package/scripts/wire_profile.mjs +173 -0
- package/src/jev.js +320 -0
- package/src/prune.js +383 -0
- package/src/receipt.js +800 -0
- package/src/state.js +532 -0
- package/test/check.js +1971 -0
- package/test/smoke_apply.mjs +1359 -0
package/src/state.js
ADDED
|
@@ -0,0 +1,532 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DSH 会话 → Jev state 的组装。纯函数,无 DSH 运行时依赖。
|
|
3
|
+
*
|
|
4
|
+
* 事件结构依据 DSH 源码与参考插件(aerince/dsh-active-context-pruning)的读法:
|
|
5
|
+
* user/message → data.content[] (blocks)
|
|
6
|
+
* assistant/message → data.message.content[] (blocks)
|
|
7
|
+
* tool/result → data.message.content[] ;callId 在 data.message.source.callId
|
|
8
|
+
* compaction/summary→ data.summary / shadowedRange / shadowedSeqs
|
|
9
|
+
* compaction/prune → data.shadowedRange / shadowedSeqs / shadowedTokenCount
|
|
10
|
+
* block:{ type:'text', text } 或 { type:'tool-call', name, arguments }
|
|
11
|
+
*
|
|
12
|
+
* state 头部必须含【任务目标】—— 实测缺它会让判断概率整体悬在阈值附近、
|
|
13
|
+
* 阈值摆动 41.4%(见 dsh-compact README 的 A/B 实验)。
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import { estimateTokens } from './jev.js'
|
|
17
|
+
// 工具名黑名单必须与两层裁决用同一套归一化比较(prune.js 是依赖链最底层)。
|
|
18
|
+
// 外部审查(issue #2):selectCandidates 此前用字面 includes,默认 PascalCase 黑名单
|
|
19
|
+
// 对真实小写工具名(edit/write)静默失效——改写类节点照进候选、照花判定钱。
|
|
20
|
+
import { isToolIn } from './prune.js'
|
|
21
|
+
// 摘录优先带"命中证据词的行",词表与第二层的证据守卫共用一份(口径一致)。
|
|
22
|
+
import { DEFAULT_EVIDENCE_PATTERNS } from './receipt.js'
|
|
23
|
+
|
|
24
|
+
export const STATE_CONTEXT =
|
|
25
|
+
'一个编码助手的对话正被压缩以释放上下文。history 是当前模型可见的全部历史(surface),' +
|
|
26
|
+
'最老在前;工具输出已被替换为简短的 result 注记。每个问题问的是:某一次工具调用的输出,' +
|
|
27
|
+
'是否仍需逐字留在历史里。没被保留的内容会被裁剪,但助手总是可以重新运行工具或重新读取文件。'
|
|
28
|
+
|
|
29
|
+
export function blockText(block) {
|
|
30
|
+
if (block == null || typeof block !== 'object') return ''
|
|
31
|
+
if (typeof block.text === 'string') return block.text
|
|
32
|
+
if (block.type === 'tool-call') {
|
|
33
|
+
const args = typeof block.arguments === 'string' ? block.arguments : JSON.stringify(block.arguments ?? {})
|
|
34
|
+
return `[工具调用] ${block.name ?? block.tool ?? '?'} ${args}`
|
|
35
|
+
}
|
|
36
|
+
return ''
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function contentOf(event) {
|
|
40
|
+
if (event == null || typeof event !== 'object') return null
|
|
41
|
+
if (event.type === 'user/message') return event.data?.content ?? null
|
|
42
|
+
if (event.type === 'assistant/message' || event.type === 'tool/result') return event.data?.message?.content ?? null
|
|
43
|
+
return null
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
export function eventText(event) {
|
|
47
|
+
if (event?.type === 'compaction/summary') {
|
|
48
|
+
const summary = event.data?.summary
|
|
49
|
+
if (typeof summary === 'string') return summary
|
|
50
|
+
if (Array.isArray(summary)) return summary.map(blockText).join('\n')
|
|
51
|
+
}
|
|
52
|
+
if (event?.type === 'tool/result') {
|
|
53
|
+
const blocks = resultContent(event)
|
|
54
|
+
if (!Array.isArray(blocks)) return ''
|
|
55
|
+
return blocks.map(blockText).join('\n')
|
|
56
|
+
}
|
|
57
|
+
const content = contentOf(event)
|
|
58
|
+
if (!Array.isArray(content)) return ''
|
|
59
|
+
return content.map(blockText).join('\n')
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function isCheckpointEvent(event) {
|
|
63
|
+
return event?.type === 'user/message'
|
|
64
|
+
&& event.data?.source?.kind === 'plugin'
|
|
65
|
+
&& event.data.source.plugin === 'compact'
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export function callIdOf(event) {
|
|
69
|
+
return event?.data?.message?.source?.callId ?? event?.data?.callId ?? null
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** 工具名解析:不同 DSH 版本字段名可能不同,逐个容忍探测。 */
|
|
73
|
+
export function toolNameOf(event, nameByCallId) {
|
|
74
|
+
const source = event?.data?.message?.source
|
|
75
|
+
const direct = source?.name ?? source?.tool ?? source?.toolName ?? event?.data?.name ?? event?.data?.tool
|
|
76
|
+
if (typeof direct === 'string' && direct.length > 0) return direct
|
|
77
|
+
const callId = callIdOf(event)
|
|
78
|
+
if (callId != null && nameByCallId?.has(callId)) return nameByCallId.get(callId)
|
|
79
|
+
return 'unknown'
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* 取会话的"全部事件"数组,带回退链。
|
|
84
|
+
*
|
|
85
|
+
* ⚠️ 实测活的 DSH 会话对象上 **`session.events` 是 undefined**(`isArray:false, ctor:"null"`),
|
|
86
|
+
* 只有 `session.eventAt(seq)` 是确认存在的。之前所有 `session.events ?? []` 的地方
|
|
87
|
+
* 都拿到空数组 → 工具名索引建出 0 条、任务目标也丢了。这里按可靠度依次回退:
|
|
88
|
+
* ① session.events(数组)
|
|
89
|
+
* ② session.snapshotEvents()
|
|
90
|
+
* ③ session.ownEvents()
|
|
91
|
+
* ④ 遍历 session.surface.nodes 逐个 eventAt()(surface 是确认存在的访问器)
|
|
92
|
+
*/
|
|
93
|
+
export function sessionEvents(session) {
|
|
94
|
+
if (Array.isArray(session?.events)) return session.events
|
|
95
|
+
for (const method of ['snapshotEvents', 'ownEvents']) {
|
|
96
|
+
if (typeof session?.[method] === 'function') {
|
|
97
|
+
try {
|
|
98
|
+
const result = session[method]()
|
|
99
|
+
if (Array.isArray(result)) return result
|
|
100
|
+
} catch {
|
|
101
|
+
// 继续尝试下一种
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
const surface = session?.surface?.nodes
|
|
106
|
+
if (Array.isArray(surface) && typeof session?.eventAt === 'function') {
|
|
107
|
+
const out = []
|
|
108
|
+
for (const seq of surface) {
|
|
109
|
+
const event = session.eventAt(seq)
|
|
110
|
+
if (event != null) out.push(event)
|
|
111
|
+
}
|
|
112
|
+
return out
|
|
113
|
+
}
|
|
114
|
+
return []
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* 从会话事件里建立 callId → 工具名 索引。
|
|
119
|
+
*
|
|
120
|
+
* 两条来源都要吃(实测真实 DSH 会话里两者同时存在):
|
|
121
|
+
* ① `tool/call` 事件:`data.callId` + `data.name` —— **最直接、最权威**(扁平字段,不用挖块)
|
|
122
|
+
* ② `assistant/message` 里的 `tool-call` 块:`block.id` + `block.name`
|
|
123
|
+
* 先扫 ① 再扫 ②(② 不覆盖 ①)。
|
|
124
|
+
*/
|
|
125
|
+
export function buildToolNameIndex(events) {
|
|
126
|
+
const index = new Map()
|
|
127
|
+
for (const event of events) {
|
|
128
|
+
if (event?.type !== 'tool/call') continue
|
|
129
|
+
const id = event.data?.callId ?? event.data?.id
|
|
130
|
+
const name = event.data?.name ?? event.data?.tool
|
|
131
|
+
if (id != null && typeof name === 'string' && name.length > 0) index.set(id, name)
|
|
132
|
+
}
|
|
133
|
+
for (const event of events) {
|
|
134
|
+
const content = contentOf(event)
|
|
135
|
+
if (!Array.isArray(content)) continue
|
|
136
|
+
for (const block of content) {
|
|
137
|
+
if (block?.type !== 'tool-call') continue
|
|
138
|
+
const id = block.id ?? block.callId ?? block.toolCallId
|
|
139
|
+
const name = block.name ?? block.tool ?? block.toolName
|
|
140
|
+
if (id == null || typeof name !== 'string' || name.length === 0) continue
|
|
141
|
+
if (!index.has(id)) index.set(id, name)
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
return index
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/** 工具结果的正文块数组。
|
|
148
|
+
* DSH 的层次是 message.content[0] = { type:'tool-result', content:[区块…] },
|
|
149
|
+
* 正文在**内层** content —— 与裁剪器源码 `original.content[0].content` 一致。 */
|
|
150
|
+
export function resultContent(event) {
|
|
151
|
+
const block = firstResultBlock(event)
|
|
152
|
+
return block?.content ?? null
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
function firstResultBlock(event) {
|
|
156
|
+
const content = contentOf(event)
|
|
157
|
+
if (!Array.isArray(content)) return null
|
|
158
|
+
return content.find((block) => block?.type === 'tool-result') ?? null
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/** 工具结果的正文长度(Unicode 码点数),与 DSH 裁剪器口径一致。 */
|
|
162
|
+
export function resultChars(event) {
|
|
163
|
+
const blocks = resultContent(event)
|
|
164
|
+
if (!Array.isArray(blocks)) return 0
|
|
165
|
+
let chars = 0
|
|
166
|
+
for (const block of blocks) {
|
|
167
|
+
if (block?.type === 'text' && typeof block.text === 'string') chars += Array.from(block.text).length
|
|
168
|
+
}
|
|
169
|
+
return chars
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/** 是否已经被裁过(正文里带裁剪标记)→ 不再重复判定。 */
|
|
173
|
+
export function looksPruned(event, marker) {
|
|
174
|
+
const blocks = resultContent(event)
|
|
175
|
+
if (!Array.isArray(blocks)) return false
|
|
176
|
+
return blocks.some((block) => typeof block?.text === 'string' && block.text.includes(marker))
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* 工具结果的**关键摘录**(P0-2,判盲缓解)。
|
|
181
|
+
*
|
|
182
|
+
* 问题:state 里工具结果此前只有 `ok, N chars`——判断者只知道"有这么一条很大的东西",
|
|
183
|
+
* 看不到里面是什么。Claude 版的历史教训(issue #26:256 条结果无一过 0.5 阈值、
|
|
184
|
+
* 与假评分器打平)证明**盲判 ≈ 抛硬币**;我们自己的 in-vivo 实测也是 42/42 判"过期",
|
|
185
|
+
* 其中一个会话里被误判的正是后续修 bug 必须用的结果。
|
|
186
|
+
*
|
|
187
|
+
* 设计:拿**最便宜的位置线索**补上"这是什么",不展开全文——
|
|
188
|
+
* · 头 2 行:文件/命令回显通常在这里("Step 2/7 : RUN apt-get…"、"// 模块:payment")
|
|
189
|
+
* · 命中证据词的行(error/fail/assert/timeout…,词表与第二层守卫共用):
|
|
190
|
+
* 报错栈、失败断言、超时配置**往往在结果中段**——恰是"掐中间"策略丢掉的位置
|
|
191
|
+
* 总预算 `resultExcerptChars` 字符(默认 240,0 = 关闭回到旧行为)。
|
|
192
|
+
*
|
|
193
|
+
* @returns {string} 摘录文本;无内容/预算为 0 时返回 ''
|
|
194
|
+
*/
|
|
195
|
+
export function resultExcerpt(event, budget = 240) {
|
|
196
|
+
if (!(budget > 0)) return ''
|
|
197
|
+
const blocks = resultContent(event)
|
|
198
|
+
if (!Array.isArray(blocks)) return ''
|
|
199
|
+
const text = blocks
|
|
200
|
+
.filter((block) => block?.type === 'text' && typeof block.text === 'string')
|
|
201
|
+
.map((block) => block.text)
|
|
202
|
+
.join('\n')
|
|
203
|
+
if (text.trim().length === 0) return ''
|
|
204
|
+
const lines = text.split('\n').map((line) => line.trim()).filter((line) => line.length > 0)
|
|
205
|
+
if (lines.length === 0) return ''
|
|
206
|
+
|
|
207
|
+
const clip = (line, max = 120) => {
|
|
208
|
+
const points = Array.from(line)
|
|
209
|
+
return points.length <= max ? line : points.slice(0, max).join('') + '…'
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* **按信息量打分选行**(而不是先到先得)。
|
|
214
|
+
*
|
|
215
|
+
* 实测教训:最初是"头 2 行 + 错误词行",结果 4 行的预算被**头部元数据**吃光——
|
|
216
|
+
* DSH 的 read 结果开头是 `<path>…</path>` / `<type>file</type>`,而它们的信息
|
|
217
|
+
* 在 state 的 tool_call 行里已经有了。真正有区分度的中段配置行
|
|
218
|
+
* (`THRESHOLD_DISCOUNT_PCT = 15`)根本挤不进来,于是摘录等于没摘。
|
|
219
|
+
*
|
|
220
|
+
* 现在给每行打分再取前 4:
|
|
221
|
+
* 4.0 证据词(error/fail/assert/timeout…)——报错栈/失败断言
|
|
222
|
+
* 3.5 常量标识符(ALL_CAPS)——配置项、错误码(如 THRESHOLD_DISCOUNT_PCT)
|
|
223
|
+
* 3.0 赋值/键值(`x = 15` / `"k": v`)——具体数值
|
|
224
|
+
* 2.0 路径/文件名
|
|
225
|
+
* 1.0 头部两行(提供"这是什么"的上下文)
|
|
226
|
+
* 0.2 纯元数据行(`<path>`/`<type>`/`</…>`)——与 tool_call 行重复
|
|
227
|
+
*/
|
|
228
|
+
const escaped = DEFAULT_EVIDENCE_PATTERNS.map((p) => String(p).replace(/[.*+?^${}()|[\]\\]/g, '\\$&'))
|
|
229
|
+
const evidenceRe = new RegExp(escaped.join('|'), 'i')
|
|
230
|
+
const allCapsRe = /\b[A-Z][A-Z0-9_]{4,}\b/
|
|
231
|
+
const assignRe = /[\w.$-]+\s*[=:]\s*\S+/
|
|
232
|
+
const pathRe = /[\w./\\-]+\.(js|ts|mjs|cjs|json|py|go|rs|java|rb|md|conf|ini|cfg|log|ya?ml|toml|sql)\b/i
|
|
233
|
+
const metaRe = /^<\/?[a-z][a-z0-9-]*>?$|^<type>[^<]*<\/type>$|^<\/?(path|type|content)>/i
|
|
234
|
+
|
|
235
|
+
const scored = lines.map((line, idx) => {
|
|
236
|
+
let score = 0.5
|
|
237
|
+
if (metaRe.test(line)) score = 0.2
|
|
238
|
+
else if (evidenceRe.test(line)) score = 4
|
|
239
|
+
else if (allCapsRe.test(line)) score = 3.5
|
|
240
|
+
else if (assignRe.test(line)) score = 3
|
|
241
|
+
else if (pathRe.test(line)) score = 2
|
|
242
|
+
if (idx === 0) score = Math.max(score, 1)
|
|
243
|
+
if (idx === 1) score = Math.max(score, 0.8)
|
|
244
|
+
if (idx === Math.floor(lines.length / 2)) score = Math.max(score, 1.5) // 中段兜底:掐中间丢的就是这里
|
|
245
|
+
return { line, idx, score }
|
|
246
|
+
})
|
|
247
|
+
|
|
248
|
+
// 先按评分降序选行,再**按评分顺序**分配预算(高分先占,低分行不够就被截/丢)。
|
|
249
|
+
// 修 #2:旧实现按 idx 排序后 join 再整体截断,从尾部切——于是位置靠后的高分行(报错行往往在中段)
|
|
250
|
+
// 被截掉,而与 tool_call 重复的低分元数据行(<path>)却完整保留,正好反了。
|
|
251
|
+
// 修 #3:旧实现 slice(0,budget)+'…' 是 budget+1 字符,这里改用 avail-1 留省略号,严格不超。
|
|
252
|
+
const byScore = scored
|
|
253
|
+
.slice()
|
|
254
|
+
.sort((a, b) => (b.score - a.score) || (a.idx - b.idx))
|
|
255
|
+
.slice(0, 4)
|
|
256
|
+
const allocated = new Map()
|
|
257
|
+
let budgetLeft = budget
|
|
258
|
+
for (const entry of byScore) {
|
|
259
|
+
if (budgetLeft <= 0) break
|
|
260
|
+
const sepLen = allocated.size > 0 ? 3 : 0 // ' | ' 长度 3
|
|
261
|
+
const avail = budgetLeft - sepLen
|
|
262
|
+
if (avail < 2) break // 放不下「至少 1 字符 + 省略号」
|
|
263
|
+
const clipped = clip(entry.line)
|
|
264
|
+
const points = Array.from(clipped)
|
|
265
|
+
if (points.length <= avail) {
|
|
266
|
+
allocated.set(entry.idx, clipped)
|
|
267
|
+
budgetLeft = avail - points.length
|
|
268
|
+
} else {
|
|
269
|
+
allocated.set(entry.idx, points.slice(0, avail - 1).join('') + '…')
|
|
270
|
+
budgetLeft = 0
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
// 按原位置(idx)排序输出,保持可读性
|
|
274
|
+
const picked = [...allocated.entries()].sort((a, b) => a[0] - b[0]).map(([, t]) => t)
|
|
275
|
+
if (picked.length === 0) return ''
|
|
276
|
+
return picked.join(' | ')
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/** 最近若干条「纯文本用户消息」作为任务目标(对齐上游 goalFromMessages)。 */
|
|
280
|
+
export function recentGoal(events, limit = 3, maxChars = 500) {
|
|
281
|
+
const picked = []
|
|
282
|
+
for (const event of events) {
|
|
283
|
+
if (event?.type !== 'user/message') continue
|
|
284
|
+
if (isCheckpointEvent(event)) continue
|
|
285
|
+
if (event.data?.source?.kind !== 'user' && event.data?.source?.kind != null) continue
|
|
286
|
+
const content = contentOf(event)
|
|
287
|
+
if (!Array.isArray(content)) continue
|
|
288
|
+
const text = content.filter((b) => b?.type === 'text').map((b) => b.text).join(' ').trim()
|
|
289
|
+
if (text.length > 0) picked.push(text.slice(0, maxChars))
|
|
290
|
+
}
|
|
291
|
+
if (picked.length === 0) return ''
|
|
292
|
+
return picked.slice(-limit).join('\n')
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* 挑出候选工具结果节点。
|
|
297
|
+
*
|
|
298
|
+
* 排除:最近 preserveRecent 个 surface 节点(含正在进行的调用)、
|
|
299
|
+
* 永不裁剪工具、以及索引解析不出的。已裁过的默认排除;第二层在缓存丢失后
|
|
300
|
+
* 可用 includePruned 重新判定 replacement,以便恢复/重启后的会话仍能做回执压缩。
|
|
301
|
+
* @returns {Array<{seq:number, index:number, chars:number, callId:string|null, tool:string}>}
|
|
302
|
+
*/
|
|
303
|
+
export function selectCandidates({ surface, eventAt, events, preserveRecent, neverPruneTools, marker, nameByCallId, includePruned = false }) {
|
|
304
|
+
const lastAllowed = surface.length - 1 - preserveRecent
|
|
305
|
+
const out = []
|
|
306
|
+
for (let index = 0; index <= lastAllowed; index += 1) {
|
|
307
|
+
const seq = surface[index]
|
|
308
|
+
const event = eventAt(seq)
|
|
309
|
+
if (event?.type !== 'tool/result') continue
|
|
310
|
+
if (!includePruned && looksPruned(event, marker)) continue
|
|
311
|
+
const tool = toolNameOf(event, nameByCallId)
|
|
312
|
+
// 归一化比较(issue #1/#2):字面 includes 对小写工具名永远不命中
|
|
313
|
+
if (isToolIn(neverPruneTools, tool)) continue
|
|
314
|
+
out.push({ seq, index, chars: resultChars(event), callId: callIdOf(event), tool })
|
|
315
|
+
}
|
|
316
|
+
return out
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* 候选 → **两道** noul 陈述(陈述为"应当保留/是承重的",noul 即该概率)。
|
|
321
|
+
*
|
|
322
|
+
* result_sN 结果本身是否还需逐字留在历史里 → 第一层(裁正文)用它
|
|
323
|
+
* effect_sN 这次调用是否**改变了会话之外的状态**(有副作用)→ 第二层(移出整对)用它
|
|
324
|
+
*
|
|
325
|
+
* 为什么要第二道:整对删除连"调用发生过"这个事实一起删掉。有些调用的**输出很短**
|
|
326
|
+
* (`Successfully installed` / 12 字符),按结果判分必然"可丢",但它做过的事是承重的。
|
|
327
|
+
* 这道问句专门抓这类:**高 = 有副作用 = 不许整对删除**(仍可由第一层截断正文)。
|
|
328
|
+
*
|
|
329
|
+
* ⚠️ 措辞是这套方案里**最敏感的一环**,实测(probe_phrasing.js,5 个候选同一份 state):
|
|
330
|
+
*
|
|
331
|
+
* 变体 极差 标准差
|
|
332
|
+
* 上游原味(陈述式) 0.070 0.023 ← 几乎无区分度,等于对每条都说"裁"
|
|
333
|
+
* 带后果(丢弃后缺什么) 0.080 0.028
|
|
334
|
+
* 带相对对比 0.380 0.136 ← 区分度 6 倍,但措辞里出现"体积大"时
|
|
335
|
+
* 带目标引用 0.340 0.116 大结果会被字面匹配抬分(污染)
|
|
336
|
+
*
|
|
337
|
+
* 所以默认用 goal 版,并且:
|
|
338
|
+
* · **不在问题里写字符数** —— 写"约 15599 字符"会诱导模型按体积作答
|
|
339
|
+
* · **不出现"大/小/体积"这类词** —— 会被字面匹配,而不是语义理解
|
|
340
|
+
* · 显式锚定【任务目标】—— 判断"还有用吗"本质是"相对于目标还有用吗"
|
|
341
|
+
*
|
|
342
|
+
* 传 'legacy' 可切回上游原味(做对照用)。
|
|
343
|
+
* 注意:区分度已验证,**排序正确性尚未验证**(没有标注数据)。
|
|
344
|
+
*/
|
|
345
|
+
export function questionsFor(candidates, wording = 'goal') {
|
|
346
|
+
const out = {}
|
|
347
|
+
for (const candidate of candidates) {
|
|
348
|
+
out[`result_s${candidate.seq}`] = phrase(candidate, wording)
|
|
349
|
+
out[`effect_s${candidate.seq}`] = effectPhrase(candidate, wording)
|
|
350
|
+
}
|
|
351
|
+
return out
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
function phrase(c, wording) {
|
|
355
|
+
if (wording === 'legacy') {
|
|
356
|
+
return `第 s${c.seq} 号工具结果(${c.tool},约 ${c.chars} 字符)应当逐字留在历史里:` +
|
|
357
|
+
`助手接下来的动作仍然需要它的内容,且重跑一次该工具无法替代。`
|
|
358
|
+
}
|
|
359
|
+
if (wording === 'contrast') {
|
|
360
|
+
return `在本次会话的所有工具结果中,第 s${c.seq} 号(${c.tool})属于` +
|
|
361
|
+
`"结论已经被别处记录下来、可以安全丢弃"的那一类。`
|
|
362
|
+
}
|
|
363
|
+
if (wording === 'consequence') {
|
|
364
|
+
return `第 s${c.seq} 号工具结果(${c.tool})被裁掉后,助手在完成当前任务时会缺少必要的信息,` +
|
|
365
|
+
`且无法通过重跑该工具廉价地拿回来。`
|
|
366
|
+
}
|
|
367
|
+
// 默认:目标锚定版
|
|
368
|
+
return `第 s${c.seq} 号工具结果(${c.tool}):在【任务目标】接下来还要进行的步骤里,` +
|
|
369
|
+
`助手仍需要直接引用它的内容,重跑该工具不能替代。`
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
/**
|
|
373
|
+
* 副作用问句。**方向必须显式写清楚**(高 = 有副作用 = 承重),
|
|
374
|
+
* 因为实测里我第一版把方向读反过一次。
|
|
375
|
+
*
|
|
376
|
+
* 刻意**不写字符数、不写工具类别** —— 那些会诱导模型按表面的体量/类型作答,
|
|
377
|
+
* 而这道题的答案是"这次动作有没有改变外部世界"。
|
|
378
|
+
*/
|
|
379
|
+
function effectPhrase(c, wording) {
|
|
380
|
+
if (wording === 'legacy' || wording === 'contrast' || wording === 'consequence') {
|
|
381
|
+
// 对照臂统一用同一句,避免把措辞变量混进方向验证
|
|
382
|
+
return `第 s${c.seq} 号工具调用(${c.tool}):它改变了会话之外的持久状态` +
|
|
383
|
+
`(写文件、改配置、安装、提交、删除、对外发起操作等),后续步骤需要知道它发生过。`
|
|
384
|
+
}
|
|
385
|
+
return `第 s${c.seq} 号工具调用(${c.tool}):在【任务目标】接下来的步骤里,` +
|
|
386
|
+
`助手仍需要知道这次调用**发生过、并改变了会话之外的状态**(而不是仅仅读过/看过)。`
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
/**
|
|
390
|
+
* 组装 Jev state:头部(上下文 + 任务目标)+ history 行。
|
|
391
|
+
*
|
|
392
|
+
* 预算压制保留一个**行数地板**(minHistoryLines):把历史丢光虽然能塞进预算,
|
|
393
|
+
* 但判断者看不到上下文就没法判断了。地板之下仍超预算时如实返回 `fitted: false`,
|
|
394
|
+
* 由调用方决定(宁可少判几个节点,也不要发一个会 400 的请求)。
|
|
395
|
+
*
|
|
396
|
+
* @returns {{state:string, lines:number, omitted:number, fitted:boolean, stateTokens:number}}
|
|
397
|
+
*/
|
|
398
|
+
export function buildJevState({ surface, eventAt, goal, context = STATE_CONTEXT, options }) {
|
|
399
|
+
const { textHead, textTail, maxStateTokens, inputChars } = options
|
|
400
|
+
const excerptBudget = Number.isFinite(options.resultExcerptChars) ? options.resultExcerptChars : 240
|
|
401
|
+
const minLines = Math.max(1, options.minHistoryLines ?? 8)
|
|
402
|
+
const entries = []
|
|
403
|
+
for (const seq of surface) {
|
|
404
|
+
const event = eventAt(seq)
|
|
405
|
+
if (event == null) continue
|
|
406
|
+
const type = event.type
|
|
407
|
+
if (type === 'tool/result') {
|
|
408
|
+
// P0-2:带上关键摘录(头 2 行 + 证据词行),让判断者知道"里面是什么"再决定留不留
|
|
409
|
+
const excerpt = resultExcerpt(event, excerptBudget)
|
|
410
|
+
entries.push([`[s${seq}][tool_result] ok, ${resultChars(event)} chars${excerpt.length > 0 ? `;摘录: ${excerpt}` : ' (内容省略)'}`])
|
|
411
|
+
continue
|
|
412
|
+
}
|
|
413
|
+
if (type === 'compaction/summary') {
|
|
414
|
+
entries.push([`[s${seq}][checkpoint] ${abridge(eventText(event), textHead, textTail)}`])
|
|
415
|
+
continue
|
|
416
|
+
}
|
|
417
|
+
const content = contentOf(event)
|
|
418
|
+
if (!Array.isArray(content)) continue
|
|
419
|
+
const lines = []
|
|
420
|
+
for (const block of content) {
|
|
421
|
+
if (block?.type === 'text' && typeof block.text === 'string') {
|
|
422
|
+
lines.push(`[s${seq}][${type === 'user/message' ? 'user' : 'assistant'}] ${abridge(block.text, textHead, textTail)}`)
|
|
423
|
+
} else if (block?.type === 'reasoning' && typeof block.text === 'string') {
|
|
424
|
+
// 实测 DSH 的 assistant 消息块是 [reasoning, tool-call]:模型的思考在 reasoning 里。
|
|
425
|
+
// 不把正文塞进 state(太贵),但要让判断者知道"这一步思考了多少"——它和纯只读探查不是一回事。
|
|
426
|
+
lines.push(`[s${seq}][reasoning] ~${Array.from(block.text).length} 字符的思考过程(未展开)`)
|
|
427
|
+
} else if (block?.type === 'tool-call') {
|
|
428
|
+
const args = typeof block.arguments === 'string' ? block.arguments : JSON.stringify(block.arguments ?? {})
|
|
429
|
+
lines.push(`[s${seq}][tool_call] ${block.name ?? '?'} ${args.slice(0, inputChars)}${args.length > inputChars ? '…' : ''}`)
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
if (lines.length > 0) entries.push(lines)
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
let header = `【上下文】\n${context}\n`
|
|
436
|
+
if (goal && goal.length > 0) header += `\n【任务目标】\n${goal}\n`
|
|
437
|
+
header += '\n【history】\n'
|
|
438
|
+
|
|
439
|
+
const budget = maxStateTokens - estimateTokens(header)
|
|
440
|
+
const all = entries.flatMap((lines) => lines)
|
|
441
|
+
let omitted = 0
|
|
442
|
+
// O(n) 预算压制(修 #4):旧实现每次 shift 都重新 join + tokenize 整个剩余串,O(n²);
|
|
443
|
+
// 摘录让每行从 ~40 字符涨到 ~280,N=800 时实测 992ms。这里逐行算一次 token、
|
|
444
|
+
// 增量丢头部(丢掉的从总量里减),把整轮压到 O(n)。逐行 ceil 会略高估总量,
|
|
445
|
+
// 属于安全方向(宁可多丢几行也要塞进预算)。
|
|
446
|
+
const tokens = all.map((line) => estimateTokens(line))
|
|
447
|
+
let total = tokens.reduce((s, t) => s + t, 0)
|
|
448
|
+
let start = 0
|
|
449
|
+
while (all.length - start > minLines && total > budget) {
|
|
450
|
+
total -= tokens[start]
|
|
451
|
+
start += 1
|
|
452
|
+
omitted += 1
|
|
453
|
+
}
|
|
454
|
+
const kept = all.slice(start)
|
|
455
|
+
const state = header + kept.join('\n')
|
|
456
|
+
const stateTokens = estimateTokens(state)
|
|
457
|
+
return { state, lines: kept.length, omitted, fitted: stateTokens <= maxStateTokens, stateTokens }
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
/**
|
|
461
|
+
* 按 **Unicode 码点**切片截断(issue #4/#5)。
|
|
462
|
+
*
|
|
463
|
+
* 两个此前都真实存在的坑:
|
|
464
|
+
* · head/tail 为 undefined(config 未经 schemastery 归一化时)→ `head + tail + 40` 是
|
|
465
|
+
* NaN、比较恒假、slice 返回整段原文 —— 同一段文本输出两遍、state 里带 "NaN" 字面量。
|
|
466
|
+
* 这里做数值兜底,但根因是 resolveConfig 漏键(见 index.js)。
|
|
467
|
+
* · 用 UTF-16 的 .length/.slice 会把代理对劈成半个字符 —— 与 README「按码点切片」的
|
|
468
|
+
* 承诺相反,也与 prune.js 的口径不一致。统一改为码点数组。
|
|
469
|
+
*/
|
|
470
|
+
function abridge(text, head, tail) {
|
|
471
|
+
const headChars = Number.isFinite(head) ? head : 400
|
|
472
|
+
const tailChars = Number.isFinite(tail) ? tail : 150
|
|
473
|
+
const points = Array.from(String(text ?? ''))
|
|
474
|
+
if (points.length <= headChars + tailChars + 40) return points.join('')
|
|
475
|
+
const omitted = points.length - headChars - tailChars
|
|
476
|
+
return `${points.slice(0, headChars).join('')}\n… ${omitted} 字符省略 …\n${points.slice(-tailChars).join('')}`
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
/**
|
|
480
|
+
* 首次运行探针:把实际事件形状打出来,用来校正字段假设。
|
|
481
|
+
*
|
|
482
|
+
* 只打"事件级"形状(type / 块类型 / 键名)是不够的 —— 实测真正会错的是**块内部**的
|
|
483
|
+
* 字段名与取值(`tool-call.id` / `tool-call.name` / `tool-result.toolCallId` /
|
|
484
|
+
* `source.callId`)。所以这里额外打:
|
|
485
|
+
* · 每个 tool-call 块的实际 `name` 与 `id` 前缀
|
|
486
|
+
* · 每个 tool-result 块的 `toolCallId` 与 `source.callId`
|
|
487
|
+
* · **id → name 的配对结果**(能不能解析出工具名)
|
|
488
|
+
* 有它就不必再猜"为什么白名单不命中"。
|
|
489
|
+
*/
|
|
490
|
+
export function probeShapes(surface, eventAt, limit = 40) {
|
|
491
|
+
const seen = new Map()
|
|
492
|
+
const rows = []
|
|
493
|
+
for (const seq of surface.slice(0, limit)) {
|
|
494
|
+
const event = eventAt(seq)
|
|
495
|
+
if (event == null) continue
|
|
496
|
+
const content = contentOf(event)
|
|
497
|
+
const blocks = Array.isArray(content) ? content.map((b) => b?.type ?? typeof b) : []
|
|
498
|
+
const sourceKeys = event.data?.message?.source != null ? Object.keys(event.data.message.source) : []
|
|
499
|
+
const dataKeys = event.data != null ? Object.keys(event.data) : []
|
|
500
|
+
rows.push({ seq, type: event.type, blocks, dataKeys, sourceKeys })
|
|
501
|
+
const key = `${event.type}|${blocks.join(',')}|${dataKeys.join(',')}|${sourceKeys.join(',')}`
|
|
502
|
+
seen.set(key, (seen.get(key) ?? 0) + 1)
|
|
503
|
+
}
|
|
504
|
+
return { rows, summary: [...seen.entries()].map(([key, count]) => ({ key, count })) }
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
/**
|
|
508
|
+
* 工具名解析的取证块:把真实的 callId↔name 配对、以及**解析结果**打出来。
|
|
509
|
+
* `jev_probe_shapes` 与 `jev_prune_status` 共用它。
|
|
510
|
+
*/
|
|
511
|
+
export function probeToolNames({ surface, eventAt, events, limit = 12 }) {
|
|
512
|
+
const nameByCallId = buildToolNameIndex(events ?? [])
|
|
513
|
+
const rows = []
|
|
514
|
+
let unresolved = 0
|
|
515
|
+
for (const seq of surface.slice(0, limit)) {
|
|
516
|
+
const event = eventAt(seq)
|
|
517
|
+
if (event?.type !== 'tool/result') continue
|
|
518
|
+
const callId = callIdOf(event)
|
|
519
|
+
const name = toolNameOf(event, nameByCallId)
|
|
520
|
+
if (name === 'unknown') unresolved += 1
|
|
521
|
+
rows.push({ seq, callId, tool: name })
|
|
522
|
+
}
|
|
523
|
+
return {
|
|
524
|
+
indexSize: nameByCallId.size,
|
|
525
|
+
// 名字集合取自索引(同时含 tool/call 事件与 assistant 消息块两种来源)——
|
|
526
|
+
// 这是用户配 compactTools 时真正需要的那份清单
|
|
527
|
+
names: [...new Set(nameByCallId.values())].sort(),
|
|
528
|
+
rows,
|
|
529
|
+
unresolved,
|
|
530
|
+
resolved: rows.length - unresolved,
|
|
531
|
+
}
|
|
532
|
+
}
|