dsh-jev-prune 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/LICENSE +21 -0
- package/README.md +177 -0
- package/README_zh.md +175 -0
- package/assets/banner.png +0 -0
- package/assets/banner.svg +73 -0
- package/assets/demo-poster.png +0 -0
- package/assets/demo.cast +6 -0
- package/assets/demo.gif +0 -0
- package/assets/demo.json +51 -0
- package/assets/two-layers.png +0 -0
- package/assets/two-layers.svg +118 -0
- package/cordis.patch.yml +3 -0
- package/demo/README.md +23 -0
- package/demo/fixtures.mjs +301 -0
- package/demo/render.py +31 -0
- package/demo/run.mjs +85 -0
- package/docs/ARCHITECTURE.md +63 -0
- package/docs/CONTRIBUTING.md +48 -0
- package/docs/PORTING.md +60 -0
- package/docs/RELEASING.md +27 -0
- package/docs/implementation.md +310 -0
- package/docs/measurements.md +16 -0
- package/docs/review-2026-10-05.md +44 -0
- package/examples/README.md +7 -0
- package/examples/minimal.yml +10 -0
- package/index.js +2043 -0
- package/package.json +77 -0
- package/scripts/inspect_session.mjs +235 -0
- package/scripts/verify_layout.mjs +24 -0
- package/scripts/verify_real_shapes.mjs +99 -0
- package/scripts/wire_profile.mjs +173 -0
- package/src/jev.js +320 -0
- package/src/prune.js +383 -0
- package/src/receipt.js +800 -0
- package/src/state.js +532 -0
- package/test/check.js +1971 -0
- package/test/smoke_apply.mjs +1359 -0
package/test/check.js
ADDED
|
@@ -0,0 +1,1971 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* check.js —— 纯函数自检(不碰 DSH 运行时,node test/check.js 直接跑)。
|
|
3
|
+
*
|
|
4
|
+
* 覆盖三件最容易写错的事:
|
|
5
|
+
* 1. token 估算(上游校正算法的移植是否正确)
|
|
6
|
+
* 2. state 组装是否带【任务目标】(缺它会让判断整体悬在阈值附近)
|
|
7
|
+
* 3. 候选筛选的三条排除规则(最近区 / 永不裁剪工具 / 已裁过)
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import assert from 'node:assert/strict'
|
|
11
|
+
import { existsSync, readFileSync, readdirSync, statSync } from 'node:fs'
|
|
12
|
+
import { dirname, join, resolve } from 'node:path'
|
|
13
|
+
import { fileURLToPath } from 'node:url'
|
|
14
|
+
|
|
15
|
+
import { JevClient, JevError, estimateTokens } from '../src/jev.js'
|
|
16
|
+
import { JEV_PRUNE_MARKER, countChars, decideAction, parseLimit, planTrims, pruneSessionWithJev, sliceWithBudget } from '../src/prune.js'
|
|
17
|
+
import {
|
|
18
|
+
DEFAULT_COMPACT_TOOLS,
|
|
19
|
+
DEFAULT_EVIDENCE_PATTERNS,
|
|
20
|
+
DEFAULT_FLOOR_THRESHOLD,
|
|
21
|
+
DEFAULT_MIN_CANDIDATES_FOR_RELATIVE,
|
|
22
|
+
DEFAULT_NEVER_COMPACT_TOOLS,
|
|
23
|
+
DSH_READONLY_TOOLS,
|
|
24
|
+
RECEIPT_MARKER,
|
|
25
|
+
balancedAfter,
|
|
26
|
+
balancedBefore,
|
|
27
|
+
cachedVerdictForEvent,
|
|
28
|
+
computeCuts,
|
|
29
|
+
computeEligibleSeqs,
|
|
30
|
+
isToolIn,
|
|
31
|
+
normalizeToolName,
|
|
32
|
+
renderCallArgs,
|
|
33
|
+
renderReceipt,
|
|
34
|
+
scanEvidence,
|
|
35
|
+
selectReceiptRanges,
|
|
36
|
+
} from '../src/receipt.js'
|
|
37
|
+
import {
|
|
38
|
+
buildJevState,
|
|
39
|
+
buildToolNameIndex,
|
|
40
|
+
callIdOf,
|
|
41
|
+
looksPruned,
|
|
42
|
+
probeToolNames,
|
|
43
|
+
questionsFor,
|
|
44
|
+
recentGoal,
|
|
45
|
+
resultChars,
|
|
46
|
+
resultExcerpt,
|
|
47
|
+
selectCandidates,
|
|
48
|
+
sessionEvents,
|
|
49
|
+
toolNameOf,
|
|
50
|
+
} from '../src/state.js'
|
|
51
|
+
import {
|
|
52
|
+
CONFIG_WARNINGS,
|
|
53
|
+
Config,
|
|
54
|
+
clampConfigNumber,
|
|
55
|
+
isCompactableTool,
|
|
56
|
+
resolveConfig,
|
|
57
|
+
} from '../index.js'
|
|
58
|
+
|
|
59
|
+
const here = dirname(dirname(fileURLToPath(import.meta.url)))
|
|
60
|
+
|
|
61
|
+
// ---------------------------------------------------------------- token 估算
|
|
62
|
+
// 常数由真实 BPE 标定(见 jev.js 的 TOKEN_ESTIMATE_CONSTANTS 注释)。
|
|
63
|
+
// 英文词 ≤ 6 字母 → 0.9 片,向上取整后是 1
|
|
64
|
+
assert.equal(estimateTokens('abcdef'), 1)
|
|
65
|
+
// 12 字母 → 0.9 + 6×0.16 = 1.86 → ceil = 2
|
|
66
|
+
assert.equal(estimateTokens('abcdefghijkl'), 2)
|
|
67
|
+
// 数字串按 1.8 片/位分组:4 位 → 4/1.8 = 2.22 → ceil = 3
|
|
68
|
+
assert.equal(estimateTokens('1234'), 3)
|
|
69
|
+
// 符号成串按 0.65/字符:4 个 → 2.6 → ceil = 3
|
|
70
|
+
assert.equal(estimateTokens('....'), 3)
|
|
71
|
+
// 混排的实际量级:JSON 密集文本不能被明显低估
|
|
72
|
+
const jsonish = '{"file_path": "/server/src/game/betting.ts", "limit": 1000}'
|
|
73
|
+
const got = estimateTokens(jsonish)
|
|
74
|
+
assert.ok(got >= 20 && got <= 60, `JSON 估算应在合理区间,实际 ${got}`)
|
|
75
|
+
assert.equal(estimateTokens(''), 0)
|
|
76
|
+
|
|
77
|
+
// 空串与纯空白都要落到 0/1 的边界,不能变成 0 除或 NaN
|
|
78
|
+
assert.ok(estimateTokens(' ') >= 1, '纯空白至少要算 1(下游会拿它当除数)')
|
|
79
|
+
|
|
80
|
+
// 标定效果(issue #33 回归):对真实 BPE 的平均绝对偏差必须压住。
|
|
81
|
+
//
|
|
82
|
+
// ⚠️ 留出集原则(PR #28 review):这些样本**不得**参与常数拟合。
|
|
83
|
+
// 早期版本直接复用了标定数据本身,于是断言退化成同义反复——它必然通过,
|
|
84
|
+
// 只能防"手改常数",完全防不了"过拟合到拟合集"。
|
|
85
|
+
// 下面这组是标定时**留出**的样本(gpt-tokenizer 实测值),常数没见过它们,
|
|
86
|
+
// 所以 MAE 超过阈值真的说明泛化坏了。
|
|
87
|
+
const HOLDOUT_CASES = [
|
|
88
|
+
// [文本, 真实 token 数(gpt-tokenizer 实测,未参与拟合)]
|
|
89
|
+
['There is a substantial difference between a plausible-sounding explanation and a verified one; the former is cheap, the latter is not. '.repeat(3), 79],
|
|
90
|
+
['DEFAULT_NEVER_PRUNE_TOOLS DEFAULT_NEVER_COMPACT_TOOLS resolveConfig clampConfigNumber CONFIG_RANGES CONFIG_WARNINGS '.repeat(3), 76],
|
|
91
|
+
['这个插件的第一层只做截断(可逆),第二层会把调用与结果整对移出(破坏性),所以两层的黑名单必须分开维护。'.repeat(3), 120],
|
|
92
|
+
['await client.ask(state, questions, { signal }) 之后要检查 lastRetries lastError 与 requests,不能用累计量判断"这一次"是否重试过。'.repeat(3), 105],
|
|
93
|
+
['2026-09-22T02:58:00.579Z WARN meter=missing threshold=3000 measured=false action=proceed\n'.repeat(5), 150],
|
|
94
|
+
]
|
|
95
|
+
let tokenAbsDrift = 0
|
|
96
|
+
for (const [text, real] of HOLDOUT_CASES) {
|
|
97
|
+
tokenAbsDrift += Math.abs(estimateTokens(text) - real) / real
|
|
98
|
+
}
|
|
99
|
+
const tokenMae = tokenAbsDrift / HOLDOUT_CASES.length
|
|
100
|
+
if (!(tokenMae <= 0.15)) {
|
|
101
|
+
throw new Error(`token 估算在留出集上的平均绝对偏差 ${(tokenMae * 100).toFixed(1)}% 超过 15%`
|
|
102
|
+
+ `(标定过拟合或常数被手改?)`)
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// 方向性:估算不能系统性偏大(把两层的门都收紧)。
|
|
106
|
+
// 旧实现在英文与路径上 +37%/+44%,正是这条要防的。
|
|
107
|
+
const englishProse = HOLDOUT_CASES[0][0]
|
|
108
|
+
assert.ok(estimateTokens(englishProse) / 79 - 1 <= 0.15,
|
|
109
|
+
`纯英文必须不显著高估(实际 ${estimateTokens(englishProse)} vs 真实 79)`)
|
|
110
|
+
|
|
111
|
+
// ---------------------------------------------------------------- 事件构造
|
|
112
|
+
function userEvent(seq, text) {
|
|
113
|
+
return { seq, type: 'user/message', data: { content: [{ type: 'text', text }], source: { kind: 'user' } } }
|
|
114
|
+
}
|
|
115
|
+
function assistantWithCall(seq, callId, tool, args) {
|
|
116
|
+
return {
|
|
117
|
+
seq,
|
|
118
|
+
type: 'assistant/message',
|
|
119
|
+
data: { message: { content: [{ type: 'text', text: '继续' }, { type: 'tool-call', id: callId, name: tool, arguments: args }] } },
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
function toolResult(seq, callId, text) {
|
|
123
|
+
return {
|
|
124
|
+
seq,
|
|
125
|
+
type: 'tool/result',
|
|
126
|
+
data: { message: { source: { callId }, content: [{ type: 'tool-result', content: [{ type: 'text', text }] }] } },
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const BIG = 'x'.repeat(5000)
|
|
131
|
+
const SMALL = 'y'.repeat(50)
|
|
132
|
+
|
|
133
|
+
const events = [
|
|
134
|
+
userEvent(1, '联机德州扑克:河牌阶段加注校验对不上,不要动 handEvaluator。'),
|
|
135
|
+
assistantWithCall(2, 'c1', 'Read', { file_path: 'server/src/game/betting.ts' }),
|
|
136
|
+
toolResult(3, 'c1', BIG),
|
|
137
|
+
assistantWithCall(4, 'c2', 'Grep', { pattern: 'pots.flop' }),
|
|
138
|
+
toolResult(5, 'c2', SMALL),
|
|
139
|
+
assistantWithCall(6, 'c3', 'Edit', { file_path: 'a.ts' }),
|
|
140
|
+
toolResult(7, 'c3', 'Edit applied: a.ts (+1 -1)'),
|
|
141
|
+
assistantWithCall(8, 'c4', 'Bash', { command: 'find . -type f' }),
|
|
142
|
+
toolResult(9, 'c4', BIG),
|
|
143
|
+
toolResult(10, 'c5', `已裁剪${'z'.repeat(10)}`),
|
|
144
|
+
]
|
|
145
|
+
const eventAt = (seq) => events.find((event) => event.seq === seq)
|
|
146
|
+
const surface = events.map((event) => event.seq)
|
|
147
|
+
|
|
148
|
+
// ---------------------------------------------------------------- callId / 工具名
|
|
149
|
+
assert.equal(callIdOf(eventAt(3)), 'c1')
|
|
150
|
+
const nameIndex = buildToolNameIndex(events)
|
|
151
|
+
assert.equal(nameIndex.get('c1'), 'Read')
|
|
152
|
+
assert.equal(nameIndex.get('c4'), 'Bash')
|
|
153
|
+
|
|
154
|
+
// ---------------------------------------------------------------- resultChars
|
|
155
|
+
assert.equal(resultChars(eventAt(3)), 5000)
|
|
156
|
+
assert.equal(resultChars(eventAt(5)), 50)
|
|
157
|
+
|
|
158
|
+
// ---------------------------------------------------------------- 候选筛选
|
|
159
|
+
const candidates = selectCandidates({
|
|
160
|
+
surface,
|
|
161
|
+
eventAt,
|
|
162
|
+
events,
|
|
163
|
+
preserveRecent: 2, // 排除 s9 / s10
|
|
164
|
+
// 故意用小写(issue #1/#2):默认黑名单是 PascalCase,真实 DSH 工具名是全小写——
|
|
165
|
+
// 字面 includes 会让这条排除静默失效。比较必须走 isToolIn 的归一化。
|
|
166
|
+
neverPruneTools: ['edit', 'write'],
|
|
167
|
+
marker: '已裁剪',
|
|
168
|
+
nameByCallId: nameIndex,
|
|
169
|
+
})
|
|
170
|
+
const seqs = candidates.map((c) => c.seq)
|
|
171
|
+
assert.deepEqual(seqs, [3, 5], `候选应为 [3,5],实际 ${JSON.stringify(seqs)}`)
|
|
172
|
+
assert.equal(candidates[0].tool, 'Read')
|
|
173
|
+
assert.equal(candidates[0].chars, 5000)
|
|
174
|
+
// s7 是 Edit 结果 → 被 neverPruneTools 排除('Edit' 必须能匹配小写黑名单 'edit')
|
|
175
|
+
assert.equal(seqs.includes(7), false, 'Edit 结果不应进候选')
|
|
176
|
+
// s9 是 Bash 且落在最近 2 个节点里 → 排除
|
|
177
|
+
assert.equal(seqs.includes(9), false, '最近区不应进候选')
|
|
178
|
+
|
|
179
|
+
// 已经裁过的节点要能识别
|
|
180
|
+
assert.equal(looksPruned(eventAt(10), '已裁剪'), true)
|
|
181
|
+
assert.equal(looksPruned(eventAt(3), '已裁剪'), false)
|
|
182
|
+
const layer2Candidates = selectCandidates({
|
|
183
|
+
surface,
|
|
184
|
+
eventAt,
|
|
185
|
+
events,
|
|
186
|
+
preserveRecent: 0,
|
|
187
|
+
neverPruneTools: [],
|
|
188
|
+
marker: '已裁剪',
|
|
189
|
+
nameByCallId: nameIndex,
|
|
190
|
+
includePruned: true,
|
|
191
|
+
})
|
|
192
|
+
assert.equal(layer2Candidates.some((candidate) => candidate.seq === 10), true,
|
|
193
|
+
'第二层必须能在缓存丢失后重新判定已裁 replacement')
|
|
194
|
+
|
|
195
|
+
// ---------------------------------------------------------------- 问题措辞
|
|
196
|
+
const questions = questionsFor(candidates)
|
|
197
|
+
assert.deepEqual(Object.keys(questions), ['result_s3', 'effect_s3', 'result_s5', 'effect_s5'])
|
|
198
|
+
// 默认(goal 版)必须锚定任务目标,且**不得出现字符数/体积词**——否则模型会按体积作答
|
|
199
|
+
assert.match(questions.result_s3, /s3 号工具结果/)
|
|
200
|
+
assert.match(questions.result_s3, /【任务目标】/)
|
|
201
|
+
assert.equal(/字符/.test(questions.result_s3), false, 'goal 版措辞不应暴露字符数(会诱导按体积判断)')
|
|
202
|
+
assert.equal(/体积/.test(questions.result_s3), false, 'goal 版措辞不应出现体积词(会被字面匹配)')
|
|
203
|
+
// 副作用问句:方向必须显式写成"高 = 有副作用 = 承重"
|
|
204
|
+
assert.match(questions.effect_s3, /s3 号工具调用/)
|
|
205
|
+
assert.match(questions.effect_s3, /改变了会话之外的状态/)
|
|
206
|
+
// legacy 版保留上游原味,供对照
|
|
207
|
+
const legacy = questionsFor(candidates, 'legacy')
|
|
208
|
+
assert.match(legacy.result_s3, /5000 字符/)
|
|
209
|
+
assert.match(legacy.result_s3, /重跑一次该工具无法替代/)
|
|
210
|
+
// 四种措辞都要能生成(每题两个轴)
|
|
211
|
+
for (const w of ['goal', 'legacy', 'contrast', 'consequence']) {
|
|
212
|
+
const q = questionsFor(candidates, w)
|
|
213
|
+
assert.equal(Object.keys(q).length, candidates.length * 2, `措辞 ${w} 应生成两倍于候选数的问题`)
|
|
214
|
+
assert.equal(typeof q.result_s3, 'string')
|
|
215
|
+
assert.equal(typeof q.effect_s3, 'string')
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
// ---------------------------------------------------------------- 任务目标
|
|
219
|
+
const goal = recentGoal(events)
|
|
220
|
+
assert.match(goal, /河牌阶段加注校验/)
|
|
221
|
+
assert.match(goal, /handEvaluator/)
|
|
222
|
+
|
|
223
|
+
// ---------------------------------------------------------------- state 组装
|
|
224
|
+
const built = buildJevState({
|
|
225
|
+
surface,
|
|
226
|
+
eventAt,
|
|
227
|
+
goal,
|
|
228
|
+
options: { textHead: 400, textTail: 150, maxStateTokens: 25000, inputChars: 300 },
|
|
229
|
+
})
|
|
230
|
+
assert.match(built.state, /【上下文】/)
|
|
231
|
+
assert.match(built.state, /【任务目标】/, 'state 必须带任务目标 —— 缺它会让概率悬在阈值附近')
|
|
232
|
+
assert.match(built.state, /【history】/)
|
|
233
|
+
assert.match(built.state, /\[s1\]\[user\]/)
|
|
234
|
+
// tool/result 只给注记、体积与**有界摘录**——不给全文。
|
|
235
|
+
// P0-2 起不变量变了:从"正文一律不进 state"改为"正文只能以 ≤ resultExcerptChars 的摘录出现"。
|
|
236
|
+
// 判盲的代价是实测过的(Claude 版 256 条无一过阈值、我们 42/42 判过期),所以摘录默认开启。
|
|
237
|
+
assert.match(built.state, /\[s3\]\[tool_result\] ok, 5000 chars/)
|
|
238
|
+
assert.match(built.state, /摘录: /, 'P0-2:结果应带关键摘录(判盲缓解)')
|
|
239
|
+
assert.equal(/x{1000}/.test(built.state), false, '工具结果正文不得整段进 state(只允许有界摘录)')
|
|
240
|
+
assert.equal(/x{200}/.test(built.state), false, '单行超长正文只允许取头部一小段(摘录受预算截断)')
|
|
241
|
+
|
|
242
|
+
// P0-2:摘录要带**对的线索**——头几行 + 命中证据词的行(报错/失败**往往在结果中段**,
|
|
243
|
+
// 正是"掐中间"策略会丢掉的位置)
|
|
244
|
+
{
|
|
245
|
+
const body = [
|
|
246
|
+
'Step 2/7 : RUN apt-get update && apt-get install -y curl',
|
|
247
|
+
...Array.from({ length: 40 }, (_, i) => `filler line ${i} lorem ipsum dolor sit amet`),
|
|
248
|
+
'ERROR E2001_BASE_IMAGE: base image node:18-broken does not exist; use node:20-alpine instead',
|
|
249
|
+
...Array.from({ length: 40 }, (_, i) => `tail filler ${i}`),
|
|
250
|
+
].join('\n')
|
|
251
|
+
const ev = {
|
|
252
|
+
type: 'tool/result',
|
|
253
|
+
data: { message: { source: { callId: 'cX' }, content: [{ type: 'tool-result', content: [{ type: 'text', text: body }] }] } },
|
|
254
|
+
}
|
|
255
|
+
const excerpt = resultExcerpt(ev, 240)
|
|
256
|
+
assert.match(excerpt, /Step 2\/7/, '头行应进摘录("这是什么文件/命令")')
|
|
257
|
+
assert.match(excerpt, /E2001_BASE_IMAGE/, '中段的证据行应进摘录(这是掐中间会丢的那一段)')
|
|
258
|
+
assert.equal(excerpt.includes('tail filler 39'), false, '尾部无证据的填充不该占摘录预算')
|
|
259
|
+
assert.ok(Array.from(excerpt).length <= 240, `摘录必须受预算约束,实际 ${Array.from(excerpt).length}`)
|
|
260
|
+
assert.equal(resultExcerpt(ev, 0), '', '预算 0 = 关闭摘录')
|
|
261
|
+
// 关闭时必须回到旧行为(可配置回退,别把判盲当成不可逆)
|
|
262
|
+
const off = buildJevState({
|
|
263
|
+
surface,
|
|
264
|
+
eventAt,
|
|
265
|
+
goal,
|
|
266
|
+
options: { textHead: 400, textTail: 150, maxStateTokens: 25000, inputChars: 300, resultExcerptChars: 0 },
|
|
267
|
+
})
|
|
268
|
+
assert.equal(/x{100}/.test(off.state), false, 'resultExcerptChars=0 时必须回到"只给体积"的旧行为')
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
// 预算压制:预算很紧时应从最老开始丢行,但保留行数地板,并如实报告是否装下
|
|
272
|
+
const squeezed = buildJevState({
|
|
273
|
+
surface,
|
|
274
|
+
eventAt,
|
|
275
|
+
goal,
|
|
276
|
+
options: { textHead: 10, textTail: 5, maxStateTokens: 200, inputChars: 20, minHistoryLines: 8 },
|
|
277
|
+
})
|
|
278
|
+
assert.ok(squeezed.omitted > 0, '预算不足时应从最老开始丢行')
|
|
279
|
+
assert.ok(squeezed.lines >= 8, `行数不应低于地板 8,实际 ${squeezed.lines}`)
|
|
280
|
+
assert.ok(squeezed.lines < surface.length, '确实丢了行')
|
|
281
|
+
assert.equal(typeof squeezed.fitted, 'boolean')
|
|
282
|
+
assert.equal(squeezed.stateTokens > 0, true)
|
|
283
|
+
|
|
284
|
+
// abridge 的两个回归(issue #4/#5):
|
|
285
|
+
// ① head/tail undefined(config 未经 schemastery 归一化)→ 不得产生 NaN 或重复原文
|
|
286
|
+
// ② 按 Unicode 码点切片,不得劈开代理对(README 的承诺在 state 侧同样成立)
|
|
287
|
+
{
|
|
288
|
+
const evs2 = [userEvent(1, 'a'.repeat(600))] // 600 码点 > 400+150+40 → 必须触发截断
|
|
289
|
+
const built2 = buildJevState({
|
|
290
|
+
surface: [1],
|
|
291
|
+
eventAt: (s) => evs2.find((e) => e.seq === s),
|
|
292
|
+
goal: '',
|
|
293
|
+
options: {},
|
|
294
|
+
})
|
|
295
|
+
assert.equal(built2.state.includes('NaN'), false, '缺 textHead/textTail 时不得产生 NaN')
|
|
296
|
+
// NaN 路径下 slice(0,undefined) 会返回整段原文 → 600 个 a 出现两遍;
|
|
297
|
+
// 正常路径只有头部 400 个 a + 尾部 150 个 a,"400 连 a"恰好出现 1 次
|
|
298
|
+
assert.equal(built2.state.split('a'.repeat(400)).length - 1, 1, '同一段文本不得被输出两遍')
|
|
299
|
+
|
|
300
|
+
const emojiText = 'a'.repeat(399) + '😀' + 'b'.repeat(400)
|
|
301
|
+
const evs3 = [userEvent(1, emojiText)]
|
|
302
|
+
const built3 = buildJevState({
|
|
303
|
+
surface: [1],
|
|
304
|
+
eventAt: (s) => evs3.find((e) => e.seq === s),
|
|
305
|
+
goal: '',
|
|
306
|
+
options: { textHead: 400, textTail: 150 },
|
|
307
|
+
})
|
|
308
|
+
const lonely = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]/.test(built3.state)
|
|
309
|
+
assert.equal(lonely, false, 'state 不得包含孤立代理项(abridge 必须按码点切)')
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
// ---------------------------------------------------------------- 裁剪机制(prune.js)
|
|
313
|
+
// 这段逻辑以前困在 index.js 里(要 import DSH 的包 → 在 DSH 外跑不起来),
|
|
314
|
+
// 而它恰恰最容易写错:按码点切、标记只插一次、非文本块保序、必须真的更短。
|
|
315
|
+
|
|
316
|
+
const marker = ' MARK '
|
|
317
|
+
const big = 'a'.repeat(2000)
|
|
318
|
+
|
|
319
|
+
// 正常裁剪:留头 + 标记 + 留尾,且必须真的更短
|
|
320
|
+
{
|
|
321
|
+
const out = sliceWithBudget([{ type: 'text', text: big }], 600, 200, marker)
|
|
322
|
+
assert.ok(out != null, '2000 字符配 600/200 的预算应当动手')
|
|
323
|
+
const text = out[0].text
|
|
324
|
+
assert.ok(text.startsWith('a'.repeat(600)), '应保留完整的头 600')
|
|
325
|
+
assert.ok(text.endsWith('a'.repeat(200)), '应保留完整的尾 200')
|
|
326
|
+
assert.equal(text.includes(marker.trim()), true, '应插入标记')
|
|
327
|
+
assert.ok(countChars(out) < countChars([{ type: 'text', text: big }]), '结果必须更短')
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
// 不值得裁:省下的还不够标记占的地方 → 返回 null
|
|
331
|
+
{
|
|
332
|
+
const out = sliceWithBudget([{ type: 'text', text: 'a'.repeat(650) }], 600, 200, marker)
|
|
333
|
+
assert.equal(out, null, '头+尾就超过原文长度时不该动手')
|
|
334
|
+
}
|
|
335
|
+
assert.equal(sliceWithBudget([], 600, 200, marker), null, '空数组返回 null')
|
|
336
|
+
assert.equal(sliceWithBudget(null, 600, 200, marker), null, '非数组返回 null')
|
|
337
|
+
|
|
338
|
+
// 标记只插一次(多文本块时最易出错的点)
|
|
339
|
+
{
|
|
340
|
+
const blocks = [
|
|
341
|
+
{ type: 'text', text: 'a'.repeat(1000) },
|
|
342
|
+
{ type: 'text', text: 'b'.repeat(1000) },
|
|
343
|
+
{ type: 'text', text: 'c'.repeat(1000) },
|
|
344
|
+
]
|
|
345
|
+
const out = sliceWithBudget(blocks, 500, 300, marker)
|
|
346
|
+
assert.ok(out != null)
|
|
347
|
+
const joined = out.map((b) => b.text).join('')
|
|
348
|
+
assert.equal(joined.split(marker.trim()).length - 1, 1, '标记应恰好出现一次')
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
// 非文本块保序原样保留
|
|
352
|
+
{
|
|
353
|
+
const blocks = [
|
|
354
|
+
{ type: 'text', text: 'x'.repeat(30) },
|
|
355
|
+
{ type: 'image', data: 'zzz' },
|
|
356
|
+
{ type: 'text', text: 'y'.repeat(3000) },
|
|
357
|
+
]
|
|
358
|
+
const out = sliceWithBudget(blocks, 20, 20, marker)
|
|
359
|
+
assert.ok(out != null)
|
|
360
|
+
const imageAt = out.findIndex((b) => b.type === 'image')
|
|
361
|
+
assert.ok(imageAt >= 0, '非文本块不该消失')
|
|
362
|
+
assert.deepEqual(out[imageAt], { type: 'image', data: 'zzz' }, '非文本块应原样保留')
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
// 按码点切:不能劈开代理对(emoji / 增补平面字符)
|
|
366
|
+
{
|
|
367
|
+
const emoji = '🎲'.repeat(500) // 每个是 2 个 UTF-16 单元、1 个码点
|
|
368
|
+
const out = sliceWithBudget([{ type: 'text', text: emoji }], 100, 50, marker)
|
|
369
|
+
assert.ok(out != null)
|
|
370
|
+
for (const block of out) {
|
|
371
|
+
if (block.type !== 'text') continue
|
|
372
|
+
// 不含"半个"代理对:孤立的代理项会被 JSON 序列化成 \udXXX
|
|
373
|
+
const lonely = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]/.test(block.text)
|
|
374
|
+
assert.equal(lonely, false, '不应出现孤立代理项(说明切在了码点边界上)')
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
// parseLimit(pressureLevel 已删除:有测试无调用的死代码,issue #14)
|
|
379
|
+
assert.deepEqual(parseLimit('55%'), { kind: 'ratio', value: 0.55 })
|
|
380
|
+
assert.deepEqual(parseLimit('154000'), { kind: 'tokens', value: 154000 })
|
|
381
|
+
|
|
382
|
+
// ---------------------------------------------------------------- 逐节点裁决(含 append 协议)
|
|
383
|
+
// 用假的 pruner / session 复刻 DSH 的接口形状(形状取自 DSH 源码,不是猜的):
|
|
384
|
+
// pruner.measureContent / pruneContent / ctx.tokenMeter.estimateMessage
|
|
385
|
+
// session.surface.nodes / eventAt / deriveEventMessage / append
|
|
386
|
+
|
|
387
|
+
function fakeSession(eventsArr, tailSeqs = []) {
|
|
388
|
+
const appended = []
|
|
389
|
+
const events = new Map(eventsArr.map((e) => [e.seq, e]))
|
|
390
|
+
return {
|
|
391
|
+
appended,
|
|
392
|
+
get events() { return [...events.values()] },
|
|
393
|
+
surface: { nodes: eventsArr.map((e) => e.seq) },
|
|
394
|
+
eventAt: (seq) => events.get(seq) ?? null,
|
|
395
|
+
deriveEventMessage: (event) => event.data.message,
|
|
396
|
+
append(type, data, options) {
|
|
397
|
+
const seq = Math.max(...[...events.keys()]) + 1 + appended.length
|
|
398
|
+
appended.push({ seq, type, data, options })
|
|
399
|
+
events.set(seq, { seq, type, data })
|
|
400
|
+
return { seq }
|
|
401
|
+
},
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
function fakePruner(thresholdChars) {
|
|
406
|
+
return {
|
|
407
|
+
ctx: { tokenMeter: { estimateMessage: () => 42 } },
|
|
408
|
+
measureContent: (blocks) => countChars(blocks),
|
|
409
|
+
// 与 DSH 同口径:只有超过 thresholdChars 才裁
|
|
410
|
+
pruneContent(blocks) {
|
|
411
|
+
if (countChars(blocks) <= thresholdChars) return null
|
|
412
|
+
return sliceWithBudget(blocks, 40, 20, ' VOLUME ')
|
|
413
|
+
},
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
const resultEvent = (seq, callId, text) => ({
|
|
418
|
+
seq,
|
|
419
|
+
type: 'tool/result',
|
|
420
|
+
data: { message: { source: { callId }, content: [{ type: 'tool-result', content: [{ type: 'text', text }] }] } },
|
|
421
|
+
})
|
|
422
|
+
|
|
423
|
+
const baseCfg = {
|
|
424
|
+
preserveRecent: 0,
|
|
425
|
+
neverPruneTools: ['Edit'],
|
|
426
|
+
keepThreshold: 0.5,
|
|
427
|
+
minCharsToPrune: 400,
|
|
428
|
+
headChars: 50,
|
|
429
|
+
tailChars: 30,
|
|
430
|
+
marker: marker,
|
|
431
|
+
dryRun: false,
|
|
432
|
+
}
|
|
433
|
+
const freshStats = () => ({
|
|
434
|
+
judged: 0, requests: 0, prunedByJev: 0, prunedByVolume: 0,
|
|
435
|
+
savedChars: 0, keptByJev: 0, keptByTail: 0, keptByBlacklist: 0,
|
|
436
|
+
skipped: 0, errors: 0, lastNote: '',
|
|
437
|
+
})
|
|
438
|
+
const run = ({ events, cache, cfg, threshold }) => {
|
|
439
|
+
const session = fakeSession(events)
|
|
440
|
+
const pruner = fakePruner(threshold ?? 999999)
|
|
441
|
+
const stats = freshStats()
|
|
442
|
+
const out = pruneSessionWithJev({
|
|
443
|
+
pruner, session, cache, cfg: { ...baseCfg, ...cfg }, stats,
|
|
444
|
+
freeze: (m) => m,
|
|
445
|
+
toolNameOf: () => 'Bash',
|
|
446
|
+
callIdOf: (e) => e.data.message.source.callId,
|
|
447
|
+
})
|
|
448
|
+
return { out, session, stats }
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
// ① Jev 说「还要」→ 不裁,哪怕它巨大(这是修 DSH 的误伤)
|
|
452
|
+
{
|
|
453
|
+
const events = [resultEvent(10, 'c1', 'a'.repeat(5000))]
|
|
454
|
+
const { out, session, stats } = run({ events, cache: new Map([[10, { keep: true, prob: 0.9 }]]) })
|
|
455
|
+
assert.equal(out.pruned.length, 0, 'Jev 说还要就不该裁')
|
|
456
|
+
assert.equal(session.appended.length, 0, '不该有任何 append')
|
|
457
|
+
assert.equal(stats.keptByJev, 1)
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
// ② Jev 说「过期」→ 裁,哪怕它远低于体积阈值(这是纯增量)
|
|
461
|
+
{
|
|
462
|
+
const events = [resultEvent(11, 'c2', 'b'.repeat(800))]
|
|
463
|
+
const { out, session, stats } = run({
|
|
464
|
+
events, cache: new Map([[11, { keep: false, prob: 0.1 }]]), threshold: 999999,
|
|
465
|
+
})
|
|
466
|
+
assert.equal(out.pruned.length, 1, 'Jev 说过期就该裁,与体积无关')
|
|
467
|
+
assert.equal(stats.prunedByJev, 1)
|
|
468
|
+
assert.equal(stats.prunedByVolume, 0)
|
|
469
|
+
// append 协议:先 compaction/prune(带 shadowedRange/Seqs/tokenCount),再 tool/result replace
|
|
470
|
+
const [price, replace] = session.appended
|
|
471
|
+
assert.equal(price.type, 'compaction/prune')
|
|
472
|
+
assert.deepEqual(price.data.shadowedRange, { start: 11, end: 11 })
|
|
473
|
+
assert.deepEqual(price.data.shadowedSeqs, [11])
|
|
474
|
+
assert.equal(price.data.shadowedTokenCount, 42, '应调用 pruner.ctx.tokenMeter.estimateMessage')
|
|
475
|
+
assert.equal(replace.type, 'tool/result')
|
|
476
|
+
assert.deepEqual(replace.options.surfaceOp, { op: 'replace', startSeq: 11, endSeq: 11 })
|
|
477
|
+
assert.deepEqual(replace.options.sourceEventSeqs, [11])
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
// ③ 没有判定 → 退回按体积(且体积不够大时什么都不做)
|
|
481
|
+
{
|
|
482
|
+
const events = [resultEvent(12, 'c3', 'c'.repeat(5000))]
|
|
483
|
+
const { out, session, stats } = run({ events, cache: new Map(), threshold: 1000 })
|
|
484
|
+
assert.equal(out.pruned.length, 1, '无判定时应退回按体积,且大块会被裁')
|
|
485
|
+
assert.equal(stats.prunedByVolume, 1)
|
|
486
|
+
assert.equal(stats.prunedByJev, 0)
|
|
487
|
+
assert.equal(session.appended.length, 2)
|
|
488
|
+
}
|
|
489
|
+
{
|
|
490
|
+
const events = [resultEvent(13, 'c4', 'd'.repeat(100))]
|
|
491
|
+
const { out } = run({ events, cache: new Map(), threshold: 1000 })
|
|
492
|
+
assert.equal(out.pruned.length, 0, '无判定 + 体积未超 → 不动手')
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
// ④ 落在最近区 → 一律不碰(哪怕 Jev 说过期)
|
|
496
|
+
{
|
|
497
|
+
const events = [resultEvent(14, 'c5', 'e'.repeat(5000))]
|
|
498
|
+
const { out, session, stats } = run({
|
|
499
|
+
events, cache: new Map([[14, { keep: false, prob: 0.05 }]]), cfg: { preserveRecent: 1 },
|
|
500
|
+
})
|
|
501
|
+
assert.equal(out.pruned.length, 0, '最近区不该被裁')
|
|
502
|
+
assert.equal(session.appended.length, 0)
|
|
503
|
+
// issue #8:最近区保护此前被记成"Jev 保留"——三个 keep 来源必须分开计数
|
|
504
|
+
assert.equal(stats.keptByTail, 1, '最近区保护应记入 keptByTail')
|
|
505
|
+
assert.equal(stats.keptByJev, 0, '最近区保护不应记入 keptByJev')
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
// ⑤ 永不裁剪工具 → 不碰
|
|
509
|
+
{
|
|
510
|
+
const events = [resultEvent(15, 'c6', 'f'.repeat(5000))]
|
|
511
|
+
const { out, stats } = run({
|
|
512
|
+
events, cache: new Map([[15, { keep: false, prob: 0.05 }]]),
|
|
513
|
+
cfg: { neverPruneTools: ['Bash'] }, // toolNameOf 固定返回 Bash
|
|
514
|
+
})
|
|
515
|
+
assert.equal(out.pruned.length, 0, 'neverPruneTools 里的工具不该被裁')
|
|
516
|
+
// issue #8:黑名单保护此前在统计里完全不可见(三个计数器全 0)
|
|
517
|
+
assert.equal(stats.keptByBlacklist, 1, '黑名单保护应记入 keptByBlacklist')
|
|
518
|
+
assert.equal(stats.keptByJev, 0)
|
|
519
|
+
}
|
|
520
|
+
|
|
521
|
+
// ⑤b 外部审查回归:第一层黑名单必须**归一化**比较。
|
|
522
|
+
// 真实 DSH 的改写类工具名是小写 edit/write,而默认黑名单历史上是 PascalCase ——
|
|
523
|
+
// 字面 includes 永远不命中,"改写类永不裁剪"的承诺在第一层静默失效。
|
|
524
|
+
{
|
|
525
|
+
const base = { inTail: false, verdict: { keep: false, prob: 0.05 }, charsBefore: 5000, minCharsToPrune: 400 }
|
|
526
|
+
const pascal = ['Edit', 'Write', 'MultiEdit', 'ApplyPatch']
|
|
527
|
+
assert.equal(decideAction({ ...base, tool: 'edit', neverPruneTools: pascal }), 'keep', '小写 edit 要命中 Edit')
|
|
528
|
+
assert.equal(decideAction({ ...base, tool: 'write', neverPruneTools: pascal }), 'keep', '小写 write 要命中 Write')
|
|
529
|
+
assert.equal(decideAction({ ...base, tool: 'multi_edit', neverPruneTools: pascal }), 'keep', 'multi_edit 要命中 MultiEdit')
|
|
530
|
+
assert.equal(decideAction({ ...base, tool: 'apply_patch', neverPruneTools: pascal }), 'keep', 'apply_patch 要命中 ApplyPatch')
|
|
531
|
+
assert.equal(decideAction({ ...base, tool: 'str_replace_editor', neverPruneTools: DEFAULT_NEVER_COMPACT_TOOLS }), 'keep')
|
|
532
|
+
// 反事实:只读工具不受黑名单保护,正常进入后续裁决分支
|
|
533
|
+
assert.equal(decideAction({ ...base, tool: 'read', neverPruneTools: pascal }), 'prune', 'read 不在黑名单里,Jev 说过期就裁')
|
|
534
|
+
|
|
535
|
+
// 整链验证:toolNameOf 解析出小写 edit + PascalCase 黑名单 → 整条管线不碰它
|
|
536
|
+
const events = [resultEvent(15, 'c6', 'f'.repeat(5000))]
|
|
537
|
+
const session = fakeSession(events)
|
|
538
|
+
const pruner = fakePruner(999999)
|
|
539
|
+
const stats = freshStats()
|
|
540
|
+
const out = pruneSessionWithJev({
|
|
541
|
+
pruner, session, cache: new Map([[15, { keep: false, prob: 0.05 }]]),
|
|
542
|
+
cfg: { ...baseCfg, neverPruneTools: pascal },
|
|
543
|
+
stats, freeze: (m) => m,
|
|
544
|
+
toolNameOf: () => 'edit',
|
|
545
|
+
callIdOf: (e) => e.data.message.source.callId,
|
|
546
|
+
})
|
|
547
|
+
assert.equal(out.pruned.length, 0, '整条管线:小写 edit 不被裁')
|
|
548
|
+
assert.equal(session.appended.length, 0)
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
// ⑥ Jev 说过期但太短 → 不裁(省不到东西还丢信息)
|
|
552
|
+
{
|
|
553
|
+
const events = [resultEvent(16, 'c7', 'g'.repeat(200))]
|
|
554
|
+
const { out, stats } = run({ events, cache: new Map([[16, { keep: false, prob: 0.02 }]]) })
|
|
555
|
+
assert.equal(out.pruned.length, 0, `短于 minCharsToPrune(${baseCfg.minCharsToPrune}) 不该裁`)
|
|
556
|
+
assert.equal(stats.prunedByJev, 0)
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
// ⑦ dryRun:只记账、不 append、不改内容
|
|
560
|
+
{
|
|
561
|
+
const events = [resultEvent(17, 'c8', 'h'.repeat(5000))]
|
|
562
|
+
const { out, session, stats } = run({
|
|
563
|
+
events, cache: new Map([[17, { keep: false, prob: 0.05 }]]), cfg: { dryRun: true },
|
|
564
|
+
})
|
|
565
|
+
assert.equal(session.appended.length, 0, 'dryRun 不该写任何事件')
|
|
566
|
+
assert.equal(out.pruned.length, 0)
|
|
567
|
+
assert.ok(out.charsRemoved > 0, 'dryRun 仍应如实记账省下多少')
|
|
568
|
+
assert.equal(stats.savedChars, out.charsRemoved)
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
// ⑧ 非 tool/result 节点与结构异常节点要被安全跳过
|
|
572
|
+
{
|
|
573
|
+
const events = [
|
|
574
|
+
{ seq: 20, type: 'user/message', data: { content: [{ type: 'text', text: 'x' }] } },
|
|
575
|
+
{ seq: 21, type: 'tool/result', data: { message: { source: {}, content: [{ type: 'text', text: 'no tool-result block' }] } } },
|
|
576
|
+
resultEvent(22, 'c9', 'i'.repeat(5000)),
|
|
577
|
+
]
|
|
578
|
+
const { out } = run({ events, cache: new Map([[22, { keep: false, prob: 0.05 }]]) })
|
|
579
|
+
assert.equal(out.pruned.length, 1, '只有结构完整的那条被处理')
|
|
580
|
+
assert.equal(out.pruned[0].originalSeq, 22)
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
// ⑨ 第一层 replacement 的新 seq 必须继承旧判定,且已经裁过的内容不得再次走 fallback。
|
|
584
|
+
// 真实日志里出现过新 seq + chars=848 + no-verdict-fallback,根因就是这里只按新 seq 直查。
|
|
585
|
+
{
|
|
586
|
+
const original = resultEvent(72, 'c10', 'j'.repeat(5000))
|
|
587
|
+
const replacement = {
|
|
588
|
+
...resultEvent(90, 'c10', `j`.repeat(400) + marker + `j`.repeat(300)),
|
|
589
|
+
sourceEventSeqs: [72],
|
|
590
|
+
}
|
|
591
|
+
const session = fakeSession([original, replacement])
|
|
592
|
+
session.surface.nodes = [90]
|
|
593
|
+
let fallbackCalls = 0
|
|
594
|
+
const pruner = fakePruner(100)
|
|
595
|
+
const originalPruneContent = pruner.pruneContent
|
|
596
|
+
pruner.pruneContent = (...args) => {
|
|
597
|
+
fallbackCalls += 1
|
|
598
|
+
return originalPruneContent(...args)
|
|
599
|
+
}
|
|
600
|
+
const stats = freshStats()
|
|
601
|
+
const out = pruneSessionWithJev({
|
|
602
|
+
pruner,
|
|
603
|
+
session,
|
|
604
|
+
cache: new Map([[72, { keep: false, prob: 0.05, effectProb: 0.05 }]]),
|
|
605
|
+
cfg: { ...baseCfg, keepMode: 'budget', pressureRatio: 1 },
|
|
606
|
+
stats,
|
|
607
|
+
freeze: (message) => message,
|
|
608
|
+
toolNameOf: () => 'Read',
|
|
609
|
+
callIdOf: (event) => event.data.message.source.callId,
|
|
610
|
+
})
|
|
611
|
+
assert.equal(out.pruned.length, 0, '已经裁过的 replacement 不应再次裁剪')
|
|
612
|
+
assert.equal(fallbackCalls, 0, 'replacement 应沿 sourceEventSeqs 命中判定,不得走体积 fallback')
|
|
613
|
+
assert.equal(out.decisions[0]?.prob, 0.05, '第一层计划应读到旧 seq 的 Jev 概率')
|
|
614
|
+
assert.equal(out.decisions[0]?.reason, 'already-pruned')
|
|
615
|
+
assert.equal(out.plan?.selected.includes(90), false, '已裁 replacement 不得占用本轮裁剪预算')
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
// ================================================================ P0-1:压力自适应分位的裁剪选择
|
|
619
|
+
// 为什么需要这一层:Jev 概率是**窄带**的(真实会话实测 42/42 条低于 0.5、P50=0.13),
|
|
620
|
+
// 固定 0.5 阈值会把每一轮判定都读成"可裁";而纯相对分位又会"每轮必裁固定比例"。
|
|
621
|
+
// 所以拆成正交的两件事:**裁多少**由压力缺口比例定(ratio × 池子总增益)、**裁哪些**由概率排序定。
|
|
622
|
+
{
|
|
623
|
+
const node = (seq, chars, prob, extra = {}) => ({
|
|
624
|
+
seq, index: seq, tool: 'read', chars,
|
|
625
|
+
gain: chars - 50 - 30 - 6, // 与 baseCfg 的 head/tail 一致(marker=' MARK ' 6 字符)
|
|
626
|
+
prob, effectProb: prob, verdict: { keep: prob >= 0.5, prob }, inTail: false, blacklisted: false,
|
|
627
|
+
...extra,
|
|
628
|
+
})
|
|
629
|
+
|
|
630
|
+
// ① 压力缺口为 0(ratio=0)→ **一条都不裁**
|
|
631
|
+
const small = [node(1, 800, 0.05), node(2, 900, 0.06), node(3, 1000, 0.07), node(4, 1100, 0.08)]
|
|
632
|
+
const zeroPlan = planTrims(small, { keepMode: 'budget', pressureRatio: 0 })
|
|
633
|
+
assert.equal(zeroPlan.mode, 'budget')
|
|
634
|
+
assert.equal(zeroPlan.budget, 0, '无压力缺口时预算应为 0')
|
|
635
|
+
assert.equal(zeroPlan.selected.length, 0, '预算为 0 时一条都不裁')
|
|
636
|
+
|
|
637
|
+
// ② ratio=0.5 → 预算 = 池子总增益的一半;按概率升序裁,裁够就停
|
|
638
|
+
const mixed = [
|
|
639
|
+
node(10, 12000, 0.09),
|
|
640
|
+
node(11, 900, 0.03), // 概率最低
|
|
641
|
+
node(12, 900, 0.04),
|
|
642
|
+
node(13, 900, 0.20),
|
|
643
|
+
]
|
|
644
|
+
const plan = planTrims(mixed, { keepMode: 'budget', pressureRatio: 0.5 })
|
|
645
|
+
assert.equal(plan.mode, 'budget')
|
|
646
|
+
const totalGain = mixed.reduce((s, n) => s + n.gain, 0)
|
|
647
|
+
assert.equal(plan.budget, totalGain * 0.5, '预算应为池子总增益 × 压力比例')
|
|
648
|
+
// 顺序即设计:按概率升序裁,先裁低概率的小结果,省不够预算时必须动到那条大的(10)
|
|
649
|
+
assert.deepEqual(plan.selected, [11, 12, 10], '按概率升序裁,裁到省够预算为止')
|
|
650
|
+
assert.ok(plan.spent >= plan.budget, '裁完必须至少省到预算量')
|
|
651
|
+
|
|
652
|
+
// ③ 保护上限:prob ≥ keepThreshold 的一律不进候选池
|
|
653
|
+
const protectedSet = [node(20, 12000, 0.9), node(21, 900, 0.05), node(22, 900, 0.06), node(23, 900, 0.07)]
|
|
654
|
+
const guarded = planTrims(protectedSet, { keepMode: 'budget', pressureRatio: 0.5 })
|
|
655
|
+
assert.equal(guarded.keptByCeiling, 1, 'prob 0.9 的应计入保护上限')
|
|
656
|
+
assert.equal(guarded.selected.includes(20), false, '保护上限之上的结果绝不能被选中')
|
|
657
|
+
|
|
658
|
+
// ④ 小样本(< minCandidatesForBudget)→ 降级绝对下限,只裁 prob < floorThreshold 的
|
|
659
|
+
const tiny = [node(30, 12000, 0.05), node(31, 12000, 0.30)]
|
|
660
|
+
const floored = planTrims(tiny, { keepMode: 'budget', pressureRatio: 0.5 })
|
|
661
|
+
assert.equal(floored.mode, 'floor', '候选太少应降级为绝对下限')
|
|
662
|
+
assert.deepEqual(floored.selected, [30], '降级模式只裁 prob < 0.2 的')
|
|
663
|
+
|
|
664
|
+
// ⑤ keepMode 不是 budget 时返回 null(调用方退回逐节点 absolute 裁决,旧行为不变)
|
|
665
|
+
assert.equal(planTrims(mixed, { keepMode: 'absolute' }), null)
|
|
666
|
+
assert.equal(planTrims(mixed, undefined), null, '缺省时不得改变旧行为')
|
|
667
|
+
|
|
668
|
+
// ⑤b 修 null 排序 bug:prob=null 的节点(result 轴批次失败)不得进 selected,更不该被当成 0 优先裁
|
|
669
|
+
{
|
|
670
|
+
const withNull = [node(60, 12000, null), node(61, 900, 0.05), node(62, 900, 0.06), node(63, 900, 0.07)]
|
|
671
|
+
const p = planTrims(withNull, { keepMode: 'budget', pressureRatio: 0.5 })
|
|
672
|
+
assert.equal(p.selected.includes(60), false, 'prob=null 的节点不得进 selected(失败方向:未知不裁)')
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
// ⑥ 整链:budget 模式下按 ratio 只裁"预算内"的那条,其余如实记为"预算用尽"
|
|
676
|
+
{
|
|
677
|
+
const events = [
|
|
678
|
+
resultEvent(40, 'c1', 'x'.repeat(12000)),
|
|
679
|
+
resultEvent(41, 'c2', 'y'.repeat(5000)),
|
|
680
|
+
resultEvent(42, 'c3', 'z'.repeat(900)),
|
|
681
|
+
resultEvent(43, 'c4', 'w'.repeat(900)),
|
|
682
|
+
]
|
|
683
|
+
const { out, stats } = run({
|
|
684
|
+
events,
|
|
685
|
+
cache: new Map([
|
|
686
|
+
[40, { keep: false, prob: 0.05 }], [41, { keep: false, prob: 0.06 }],
|
|
687
|
+
[42, { keep: false, prob: 0.07 }], [43, { keep: false, prob: 0.08 }],
|
|
688
|
+
]),
|
|
689
|
+
cfg: { keepMode: 'budget', pressureRatio: 0.25 },
|
|
690
|
+
})
|
|
691
|
+
assert.equal(out.pruned.length, 1, '只裁预算内的那一条')
|
|
692
|
+
assert.equal(out.pruned[0].originalSeq, 40)
|
|
693
|
+
assert.equal(stats.prunedByJev, 1)
|
|
694
|
+
assert.equal(stats.keptByBudget, 3, '概率同样低但预算已用尽的那三条应记入 keptByBudget')
|
|
695
|
+
assert.equal(out.plan.mode, 'budget')
|
|
696
|
+
assert.equal(out.decisions[0].reason, 'selected(budget)')
|
|
697
|
+
assert.equal(out.decisions[1].reason, 'budget-exhausted')
|
|
698
|
+
}
|
|
699
|
+
|
|
700
|
+
// ⑦ 对照:`absolute` 模式(旧行为)下四条都会被裁 —— 证明省下来的是"分位"在起作用
|
|
701
|
+
{
|
|
702
|
+
const events = [
|
|
703
|
+
resultEvent(50, 'c1', 'x'.repeat(12000)),
|
|
704
|
+
resultEvent(51, 'c2', 'y'.repeat(5000)),
|
|
705
|
+
resultEvent(52, 'c3', 'z'.repeat(900)),
|
|
706
|
+
resultEvent(53, 'c4', 'w'.repeat(900)),
|
|
707
|
+
]
|
|
708
|
+
const { out } = run({
|
|
709
|
+
events,
|
|
710
|
+
cache: new Map([
|
|
711
|
+
[50, { keep: false, prob: 0.05 }], [51, { keep: false, prob: 0.06 }],
|
|
712
|
+
[52, { keep: false, prob: 0.07 }], [53, { keep: false, prob: 0.08 }],
|
|
713
|
+
]),
|
|
714
|
+
cfg: { keepMode: 'absolute' },
|
|
715
|
+
})
|
|
716
|
+
assert.equal(out.pruned.length, 4, 'absolute 模式会四条都裁(这是被替换掉的旧行为)')
|
|
717
|
+
}
|
|
718
|
+
}
|
|
719
|
+
|
|
720
|
+
// ================================================================ 第二层:回执压缩
|
|
721
|
+
// 这一层做的是**破坏性**动作(整对移出 surface),所以每条门控都要单独测。
|
|
722
|
+
// 全部用纯函数,不需要 DSH 运行时。
|
|
723
|
+
|
|
724
|
+
// ---------------------------------------------------------------- 工具配对平衡
|
|
725
|
+
{
|
|
726
|
+
const evs = [
|
|
727
|
+
{ seq: 1, type: 'user/message', data: { content: [{ type: 'text', text: 'go' }] } },
|
|
728
|
+
{ seq: 2, type: 'assistant/message', data: { message: { content: [{ type: 'tool-call', id: 'a', name: 'Read', arguments: {} }] } } },
|
|
729
|
+
{ seq: 3, type: 'tool/result', data: { message: { content: [{ type: 'tool-result', content: [] }] } } },
|
|
730
|
+
{ seq: 4, type: 'assistant/message', data: { message: { content: [{ type: 'text', text: 'done' }] } } },
|
|
731
|
+
]
|
|
732
|
+
const at = (s) => evs.find((e) => e.seq === s)
|
|
733
|
+
const cuts = computeCuts(evs.map((e) => e.seq), at)
|
|
734
|
+
assert.equal(cuts.length, evs.length + 1, 'N 个节点应有 N+1 个切点')
|
|
735
|
+
assert.equal(balancedBefore(cuts, 0), true)
|
|
736
|
+
assert.equal(balancedBefore(cuts, 1), true, '调用之前是平衡的')
|
|
737
|
+
assert.equal(balancedAfter(cuts, 1), false, '调用之后(结果未到)不平衡 —— 这正是不能切的地方')
|
|
738
|
+
assert.equal(balancedAfter(cuts, 2), true, '结果之后重新平衡')
|
|
739
|
+
assert.equal(balancedAfter(cuts, 3), true)
|
|
740
|
+
|
|
741
|
+
// 多个 tool-call 在一条 assistant 消息里
|
|
742
|
+
const many = [
|
|
743
|
+
{ seq: 1, type: 'assistant/message', data: { message: { content: [{ type: 'tool-call' }, { type: 'tool-call' }] } } },
|
|
744
|
+
{ seq: 2, type: 'tool/result', data: { message: { content: [] } } },
|
|
745
|
+
{ seq: 3, type: 'tool/result', data: { message: { content: [] } } },
|
|
746
|
+
]
|
|
747
|
+
const mcuts = computeCuts(many.map((e) => e.seq), (s) => many.find((e) => e.seq === s))
|
|
748
|
+
assert.equal(balancedAfter(mcuts, 0), false)
|
|
749
|
+
assert.equal(balancedAfter(mcuts, 1), false, '两个调用只回来一个结果时仍不平衡')
|
|
750
|
+
assert.equal(balancedAfter(mcuts, 2), true)
|
|
751
|
+
|
|
752
|
+
// surface 损坏要抛错,而不是静默算错
|
|
753
|
+
assert.throws(() => computeCuts([1, 99], at), /没有对应的会话事件/)
|
|
754
|
+
assert.throws(
|
|
755
|
+
() => computeCuts([3], at),
|
|
756
|
+
/没有对应的 tool-call/,
|
|
757
|
+
'无主的 tool/result 必须抛错',
|
|
758
|
+
)
|
|
759
|
+
}
|
|
760
|
+
|
|
761
|
+
// ---------------------------------------------------------------- 相对分位门控
|
|
762
|
+
{
|
|
763
|
+
// 只读 3 条(两轴都低)+ 有副作用 3 条(effect 高)→ 交集应只含只读那 3 条
|
|
764
|
+
const verdicts = [
|
|
765
|
+
{ seq: 10, prob: 0.10, effectProb: 0.05 },
|
|
766
|
+
{ seq: 12, prob: 0.12, effectProb: 0.06 },
|
|
767
|
+
{ seq: 14, prob: 0.11, effectProb: 0.07 },
|
|
768
|
+
{ seq: 20, prob: 0.13, effectProb: 0.30 },
|
|
769
|
+
{ seq: 22, prob: 0.14, effectProb: 0.33 },
|
|
770
|
+
{ seq: 24, prob: 0.15, effectProb: 0.28 },
|
|
771
|
+
]
|
|
772
|
+
const half = computeEligibleSeqs(verdicts, { quantile: 0.5, minCandidates: 4 })
|
|
773
|
+
assert.deepEqual([...half].sort((a, b) => a - b), [10, 12, 14], 'q=0.5 时两轴各取 3 条,交集为 3 条只读')
|
|
774
|
+
assert.equal([...half].some((seq) => seq >= 20), false, '有副作用的调用**绝不能**被选中')
|
|
775
|
+
|
|
776
|
+
// 交集语义:只落在**单轴**尾部的必须被排除(若是并集,12 和 14 都会被选中)
|
|
777
|
+
const strict = computeEligibleSeqs(verdicts, { quantile: 0.34, minCandidates: 4 })
|
|
778
|
+
assert.deepEqual([...strict], [10], '必须取交集:只在 result 尾部(14)或只在 effect 尾部(12)的都不算')
|
|
779
|
+
|
|
780
|
+
// 小总体分支(issue #27 修复):此前样本不足**直接返回空集** → 第二层在只读占比低的
|
|
781
|
+
// 会话里静默不工作。现在改为降级到绝对下限模式,但下限阈值明显更严(0.2)。
|
|
782
|
+
//
|
|
783
|
+
// ⚠️ 这里必须分成**两个口径**测。原断言只用默认配置测 k=2 → 期望 0,而
|
|
784
|
+
// `computeEligibleSeqs` 的默认参数曾是硬编码字面量 `3` / `0.2`,与导出的
|
|
785
|
+
// `DEFAULT_MIN_CANDIDATES_FOR_FLOOR` / `DEFAULT_FLOOR_THRESHOLD` **分叉**:
|
|
786
|
+
// 常量改成 2 之后插件实体(走 resolveConfig,读常量)行为已变,而这行断言吃的是旧字面量、
|
|
787
|
+
// 依然全绿 —— 它测的是一个真实配置路径上不存在的数。现在默认值引用常量,
|
|
788
|
+
// 于是两个口径都必须显式写出来。
|
|
789
|
+
assert.equal(computeEligibleSeqs(verdicts.slice(0, 2), { quantile: 0.5, minCandidates: 4, minCandidatesForAbsolute: 3 }).size, 0,
|
|
790
|
+
'显式把地板设成 3 时,2 条样本仍不得做整对移出')
|
|
791
|
+
assert.equal(computeEligibleSeqs(verdicts.slice(0, 2), { quantile: 0.5, minCandidates: 4 }).size, 2,
|
|
792
|
+
'默认地板=2 时,2 条样本进绝对下限模式并全选(与 DEFAULT_MIN_CANDIDATES_FOR_FLOOR 一致)')
|
|
793
|
+
assert.equal(computeEligibleSeqs(verdicts.slice(0, 1), { quantile: 0.5, minCandidates: 4 }).size, 0,
|
|
794
|
+
'1 条样本在任何配置下都不得动作(分布的下限)')
|
|
795
|
+
|
|
796
|
+
// 缺任何一轴的概率都不参与
|
|
797
|
+
assert.equal(computeEligibleSeqs(
|
|
798
|
+
Array.from({ length: 6 }, (_, i) => ({ seq: i + 1, prob: 0.1 })),
|
|
799
|
+
{ quantile: 0.5, minCandidates: 4 },
|
|
800
|
+
).size, 0, '缺 effectProb 的判定不能用于第二层')
|
|
801
|
+
|
|
802
|
+
// quantile 非法必须 fail loud(issue #6):此前 NaN/undefined 静默返回空集,
|
|
803
|
+
// 第二层"功能静默死亡"且报错文案被误读成"样本不够"
|
|
804
|
+
for (const bad of [undefined, NaN, -0.1, 1.5, '0.34']) {
|
|
805
|
+
assert.throws(
|
|
806
|
+
() => computeEligibleSeqs(verdicts, { quantile: bad, minCandidates: 4 }),
|
|
807
|
+
(e) => /compactQuantile 非法/.test(e.message),
|
|
808
|
+
`quantile=${String(bad)} 应抛错`,
|
|
809
|
+
)
|
|
810
|
+
}
|
|
811
|
+
// quantile=0 的语义是字面意义"一条不取"(此前 Math.max(1,…) 反而取 1 条)
|
|
812
|
+
let disabledNote = ''
|
|
813
|
+
assert.equal(computeEligibleSeqs(verdicts, {
|
|
814
|
+
quantile: 0,
|
|
815
|
+
minCandidates: 4,
|
|
816
|
+
onNote: (note) => { disabledNote = note },
|
|
817
|
+
}).size, 0)
|
|
818
|
+
assert.match(disabledNote, /compactQuantile=0.*关闭/, '显式关闭不能误报成“分位交集为空”')
|
|
819
|
+
assert.equal(computeEligibleSeqs(verdicts.slice(0, 2), { quantile: 0, minCandidates: 4 }).size, 0,
|
|
820
|
+
'quantile=0 必须在小样本降级之前生效,2 条低分候选也不得被重新选中')
|
|
821
|
+
|
|
822
|
+
// 两轴各取 1 条但不是同一节点时,交集为空;严格绝对下限仍可救回两轴都很低的节点。
|
|
823
|
+
let fallbackNote = ''
|
|
824
|
+
const disjoint = computeEligibleSeqs([
|
|
825
|
+
{ seq: 1, prob: 0.01, effectProb: 0.90 },
|
|
826
|
+
{ seq: 2, prob: 0.90, effectProb: 0.01 },
|
|
827
|
+
{ seq: 3, prob: 0.10, effectProb: 0.10 },
|
|
828
|
+
{ seq: 4, prob: 0.80, effectProb: 0.80 },
|
|
829
|
+
], { quantile: 0.25, minCandidates: 4, onNote: (note) => { fallbackNote = note } })
|
|
830
|
+
assert.deepEqual([...disjoint], [3], '相对尾部交集为空时应降级到两轴绝对下限')
|
|
831
|
+
assert.match(fallbackNote, /交集为空.*绝对下限/, '降级必须通过 onNote 对外可见')
|
|
832
|
+
|
|
833
|
+
const cached = { keep: false, prob: 0.1, effectProb: 0.1 }
|
|
834
|
+
const cache = new Map([[7, cached]])
|
|
835
|
+
assert.equal(cachedVerdictForEvent(cache, { seq: 7 }), cached, '当前 seq 直接命中优先')
|
|
836
|
+
assert.equal(cachedVerdictForEvent(cache, { seq: 70, sourceEventSeqs: [7] }), cached,
|
|
837
|
+
'第一层 replacement 必须沿 sourceEventSeqs 找回旧 seq 的判定')
|
|
838
|
+
}
|
|
839
|
+
|
|
840
|
+
// ---------------------------------------------------------------- 范围选择
|
|
841
|
+
{
|
|
842
|
+
// 会话:user → 3 组只读探查 → 1 组 Edit(承重)→ 1 条带 error 的探查 → 收尾文本
|
|
843
|
+
const evs = []
|
|
844
|
+
const push = (e) => evs.push(e)
|
|
845
|
+
let seq = 0
|
|
846
|
+
push(userEvent((seq += 1), '排查河牌底池 bug'))
|
|
847
|
+
const addStep = (tool, args, output) => {
|
|
848
|
+
const callId = `c${seq + 1}`
|
|
849
|
+
push(assistantWithCall((seq += 1), callId, tool, args))
|
|
850
|
+
push(toolResult((seq += 1), callId, output))
|
|
851
|
+
return seq
|
|
852
|
+
}
|
|
853
|
+
addStep('Read', { file_path: 'server/src/game/river.ts' }, 'x'.repeat(3000))
|
|
854
|
+
addStep('Grep', { pattern: 'basePot', path: 'server/src' }, 'y'.repeat(2500))
|
|
855
|
+
addStep('Bash', { command: 'wc -l server/src/*.ts' }, 'z'.repeat(2000))
|
|
856
|
+
addStep('Edit', { file_path: 'server/src/game/river.ts' }, 'Edit applied (+2 -1)')
|
|
857
|
+
addStep('Read', { file_path: 'server/src/game/pot.ts' }, `failed to load${'w'.repeat(2000)}`)
|
|
858
|
+
push({ seq: (seq += 1), type: 'assistant/message', data: { message: { content: [{ type: 'text', text: '继续' }] } } })
|
|
859
|
+
|
|
860
|
+
const at = (s) => evs.find((e) => e.seq === s)
|
|
861
|
+
const surf = evs.map((e) => e.seq)
|
|
862
|
+
const cache = new Map()
|
|
863
|
+
for (const e of evs) {
|
|
864
|
+
if (e.type !== 'tool/result') continue
|
|
865
|
+
cache.set(e.seq, { keep: false, prob: 0.1, effectProb: 0.05, chars: 3000, tool: 'Read' })
|
|
866
|
+
}
|
|
867
|
+
const baseCfg = {
|
|
868
|
+
preserveRecent: 0,
|
|
869
|
+
compactTools: DEFAULT_COMPACT_TOOLS,
|
|
870
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
871
|
+
evidenceGuard: true,
|
|
872
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
873
|
+
maxStepTextChars: 240,
|
|
874
|
+
maxStepReasoningChars: 240,
|
|
875
|
+
compactMinChars: 1000,
|
|
876
|
+
}
|
|
877
|
+
const run = (overrides = {}, cacheOverride) => selectReceiptRanges({
|
|
878
|
+
surface: surf,
|
|
879
|
+
eventAt: at,
|
|
880
|
+
cache: cacheOverride ?? cache,
|
|
881
|
+
dropVerdict: (s) => (cacheOverride ?? cache).get(s)?.droppable !== false,
|
|
882
|
+
cfg: { ...baseCfg, ...overrides },
|
|
883
|
+
})
|
|
884
|
+
|
|
885
|
+
// 1) 两条连续只读(Read/Grep)应合并成一段;Bash 不在默认只读白名单里,
|
|
886
|
+
// Edit 被黑名单拦、带 error 的那条被证据守卫拦 —— 各自把区间断开
|
|
887
|
+
const { ranges, stats } = run()
|
|
888
|
+
assert.equal(ranges.length, 1, '只应有一段合法范围')
|
|
889
|
+
const range = ranges[0]
|
|
890
|
+
assert.equal(range.steps.length, 2, '两条连续只读应被合并进同一段')
|
|
891
|
+
assert.equal(range.chars, 5500)
|
|
892
|
+
assert.equal(range.start, at(range.start).seq)
|
|
893
|
+
assert.equal(at(range.start).type, 'assistant/message')
|
|
894
|
+
assert.equal(at(range.end).type, 'tool/result')
|
|
895
|
+
assert.equal(stats.skippedTool, 2, 'Bash(不在只读白名单)与 Edit(黑名单)都应被工具规则排除')
|
|
896
|
+
assert.equal(stats.skippedGuard, 1, '含 error 的探查应被证据守卫排除')
|
|
897
|
+
assert.ok(stats.eligibleSteps >= 2)
|
|
898
|
+
|
|
899
|
+
// 2) 白名单只留 Read/Grep → 与默认等价(Bash 与 Edit 都被排除)
|
|
900
|
+
const narrowed = run({ compactTools: ['Read', 'Grep'] })
|
|
901
|
+
assert.equal(narrowed.ranges.length, 1)
|
|
902
|
+
assert.equal(narrowed.ranges[0].steps.length, 2)
|
|
903
|
+
assert.equal(narrowed.stats.skippedTool, 2, 'Bash 与 Edit 都应被工具规则排除')
|
|
904
|
+
|
|
905
|
+
// 3) 白名单显式设为 [] = 放宽到只受黑名单约束(不安全模式,须显式 opt-in);
|
|
906
|
+
// 此时 Bash 也可进候选,仍只有 Edit 被拦
|
|
907
|
+
const noAllowlist = run({ compactTools: [] })
|
|
908
|
+
assert.equal(noAllowlist.stats.skippedTool, 1, '只有 Edit 会被拦')
|
|
909
|
+
|
|
910
|
+
// 4) 关掉证据守卫后,带 error 的那条也能进(但它是孤立步骤,另成一段)
|
|
911
|
+
const noGuard = run({ evidenceGuard: false })
|
|
912
|
+
assert.equal(noGuard.stats.skippedGuard, 0)
|
|
913
|
+
assert.equal(noGuard.ranges.length, 2, 'Bash/Edit 断开只读区,error-Read 孤立成第二段')
|
|
914
|
+
|
|
915
|
+
// 5) 最近区保护
|
|
916
|
+
const tailSafe = run({ preserveRecent: 20 })
|
|
917
|
+
assert.equal(tailSafe.ranges.length, 0)
|
|
918
|
+
assert.ok(tailSafe.stats.skippedTail > 0)
|
|
919
|
+
|
|
920
|
+
// 6) assistant 可见文本过长 → 整步排除(text 轴)
|
|
921
|
+
const chatty = at(range.start)
|
|
922
|
+
chatty.data.message.content[0].text = '先想一下'.repeat(100)
|
|
923
|
+
const chattyRun = run()
|
|
924
|
+
assert.equal(chattyRun.stats.skippedText, 1, '过长的可见文本不应被整对移出')
|
|
925
|
+
assert.equal(chattyRun.stats.skippedReasoning, 0, 'text 轴拦截不得计入 reasoning 轴计数')
|
|
926
|
+
chatty.data.message.content[0].text = '继续'
|
|
927
|
+
|
|
928
|
+
// 7) 判定说"不可丢" → 排除
|
|
929
|
+
const guardedCache = new Map(cache)
|
|
930
|
+
guardedCache.set(3, { keep: true, prob: 0.9, effectProb: 0.9, chars: 3000, tool: 'Read', droppable: false })
|
|
931
|
+
const guardedRun = run({}, guardedCache)
|
|
932
|
+
assert.equal(guardedRun.stats.skippedVerdict, 1)
|
|
933
|
+
assert.equal(guardedRun.ranges[0].steps.length, 1, '被否掉的那步之后仍可继续成段(只剩 Grep;Bash 已被只读白名单拦下)')
|
|
934
|
+
|
|
935
|
+
// 8) 省得不够多 → 整个范围丢弃
|
|
936
|
+
const tiny = run({ compactMinChars: 100000 })
|
|
937
|
+
assert.equal(tiny.ranges.length, 0)
|
|
938
|
+
assert.equal(tiny.stats.skippedShort, 1)
|
|
939
|
+
}
|
|
940
|
+
|
|
941
|
+
// ---------------------------------------------------------------- 并行批量读的合格子集(issue #39)
|
|
942
|
+
{
|
|
943
|
+
const calls = Array.from({ length: 6 }, (_, i) => ({
|
|
944
|
+
type: 'tool-call', id: `batch-${i + 1}`, name: 'read',
|
|
945
|
+
arguments: JSON.stringify({ file_path: `src/file-${i + 1}.js` }),
|
|
946
|
+
}))
|
|
947
|
+
const evs = [{
|
|
948
|
+
seq: 1,
|
|
949
|
+
type: 'assistant/message',
|
|
950
|
+
data: { message: { content: [{ type: 'text', text: '批量读取。' }, ...calls] } },
|
|
951
|
+
}]
|
|
952
|
+
for (let i = 0; i < calls.length; i += 1) {
|
|
953
|
+
evs.push(toolResult(i + 2, calls[i].id, String(i + 1).repeat(1200)))
|
|
954
|
+
}
|
|
955
|
+
// 模拟并行完成顺序与 call 声明顺序不同;selector 必须按 callId 配,而非按 offset 猜。
|
|
956
|
+
;[evs[1], evs[2]] = [evs[2], evs[1]]
|
|
957
|
+
const at = (seq) => evs.find((event) => event.seq === seq)
|
|
958
|
+
const cache = new Map(evs.slice(1).map((event) => [event.seq, {
|
|
959
|
+
keep: false, prob: 0.05, effectProb: 0.04, chars: 1200, tool: 'read',
|
|
960
|
+
}]))
|
|
961
|
+
// 第 3 个结果不合格;旧实现要求 6/6 全部合格,因此整批 ranges=0。
|
|
962
|
+
cache.set(4, { keep: true, prob: 0.9, effectProb: 0.9, chars: 1200, tool: 'read' })
|
|
963
|
+
const selected = selectReceiptRanges({
|
|
964
|
+
surface: evs.map((event) => event.seq),
|
|
965
|
+
eventAt: at,
|
|
966
|
+
cache,
|
|
967
|
+
dropVerdict: (seq) => seq !== 4,
|
|
968
|
+
cfg: {
|
|
969
|
+
preserveRecent: 0,
|
|
970
|
+
compactTools: DEFAULT_COMPACT_TOOLS,
|
|
971
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
972
|
+
evidenceGuard: false,
|
|
973
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
974
|
+
maxStepTextChars: 240,
|
|
975
|
+
maxStepReasoningChars: 4000,
|
|
976
|
+
compactMinChars: 1000,
|
|
977
|
+
},
|
|
978
|
+
})
|
|
979
|
+
assert.equal(selected.ranges.length, 1, '6 个并行 read 中 5 个合格时必须产生一个部分回执动作')
|
|
980
|
+
assert.equal(selected.ranges[0].kind, 'partial')
|
|
981
|
+
assert.deepEqual(selected.ranges[0].steps[0].resultSeqs, [3, 2, 5, 6, 7],
|
|
982
|
+
'只选择 5 个合格结果,不得把不合格的 s4 混入')
|
|
983
|
+
assert.equal(selected.ranges[0].steps[0].calls.length, 5)
|
|
984
|
+
assert.equal(selected.stats.partialSteps, 1)
|
|
985
|
+
assert.equal(selected.stats.partialResults, 5)
|
|
986
|
+
assert.equal(selected.stats.skippedVerdict, 1)
|
|
987
|
+
|
|
988
|
+
// 最近区同样按 result 粒度保护:最后一个结果留原文,前 5 个仍可处理。
|
|
989
|
+
const tail = selectReceiptRanges({
|
|
990
|
+
surface: evs.map((event) => event.seq), eventAt: at, cache,
|
|
991
|
+
dropVerdict: () => true,
|
|
992
|
+
cfg: {
|
|
993
|
+
preserveRecent: 1,
|
|
994
|
+
compactTools: DEFAULT_COMPACT_TOOLS,
|
|
995
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
996
|
+
evidenceGuard: false,
|
|
997
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
998
|
+
maxStepTextChars: 240,
|
|
999
|
+
maxStepReasoningChars: 4000,
|
|
1000
|
+
compactMinChars: 1000,
|
|
1001
|
+
},
|
|
1002
|
+
})
|
|
1003
|
+
assert.equal(tail.ranges[0].kind, 'partial')
|
|
1004
|
+
assert.deepEqual(tail.ranges[0].steps[0].resultSeqs, [3, 2, 4, 5, 6])
|
|
1005
|
+
assert.equal(tail.stats.skippedTail, 1)
|
|
1006
|
+
}
|
|
1007
|
+
|
|
1008
|
+
// ---------------------------------------------------------------- replacement 来源链上的证据也必须守住
|
|
1009
|
+
// 第一层可能把位于正文中间的 error 截掉;第二层若只扫描 surface 上的 replacement,
|
|
1010
|
+
// 会误以为没有证据并把整个调用/结果对移出。
|
|
1011
|
+
{
|
|
1012
|
+
const original = toolResult(2, 'c1', `${'a'.repeat(1200)}fatal error: hidden in middle${'b'.repeat(1200)}`)
|
|
1013
|
+
const replacement = {
|
|
1014
|
+
...toolResult(3, 'c1', `${'a'.repeat(100)}${JEV_PRUNE_MARKER}${'b'.repeat(100)}`),
|
|
1015
|
+
sourceEventSeqs: [2],
|
|
1016
|
+
}
|
|
1017
|
+
const evs = [
|
|
1018
|
+
assistantWithCall(1, 'c1', 'Read', { file_path: 'hidden-error.txt' }),
|
|
1019
|
+
original,
|
|
1020
|
+
replacement,
|
|
1021
|
+
]
|
|
1022
|
+
const at = (seq) => evs.find((event) => event.seq === seq)
|
|
1023
|
+
const cache = new Map([[2, { keep: false, prob: 0.05, effectProb: 0.05, chars: 2500, tool: 'Read' }]])
|
|
1024
|
+
const { ranges, stats } = selectReceiptRanges({
|
|
1025
|
+
surface: [1, 3],
|
|
1026
|
+
eventAt: at,
|
|
1027
|
+
cache,
|
|
1028
|
+
dropVerdict: () => true,
|
|
1029
|
+
cfg: {
|
|
1030
|
+
preserveRecent: 0,
|
|
1031
|
+
compactTools: DEFAULT_COMPACT_TOOLS,
|
|
1032
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
1033
|
+
evidenceGuard: true,
|
|
1034
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
1035
|
+
maxStepTextChars: 240,
|
|
1036
|
+
maxStepReasoningChars: 240,
|
|
1037
|
+
compactMinChars: 10,
|
|
1038
|
+
},
|
|
1039
|
+
})
|
|
1040
|
+
assert.equal(ranges.length, 0, '原始结果中被第一层截掉的 error 仍应阻止第二层整对移出')
|
|
1041
|
+
assert.equal(stats.skippedGuard, 1, '来源链证据应计入 guard 排除,而不是 verdict/short')
|
|
1042
|
+
assert.deepEqual(stats.guardHits[0]?.matches, ['error'])
|
|
1043
|
+
}
|
|
1044
|
+
|
|
1045
|
+
// ---------------------------------------------------------------- blockedToolNames 诊断口径
|
|
1046
|
+
// issue #7:多调用步骤此前把**所有**调用名都记进 blockedToolNames,通过白名单的也中招——
|
|
1047
|
+
// 用户会按 jev_probe_shapes 的提示去"补配"一个本来就在白名单里的名字。
|
|
1048
|
+
{
|
|
1049
|
+
const evs = []
|
|
1050
|
+
let seq = 0
|
|
1051
|
+
// 一个 assistant 消息同时带两个 tool-call:read(在白名单)+ pwsh(不在)
|
|
1052
|
+
evs.push({
|
|
1053
|
+
seq: (seq += 1),
|
|
1054
|
+
type: 'assistant/message',
|
|
1055
|
+
data: { message: { content: [
|
|
1056
|
+
{ type: 'tool-call', id: 'ca', name: 'read', arguments: '{}' },
|
|
1057
|
+
{ type: 'tool-call', id: 'cb', name: 'pwsh', arguments: '{}' },
|
|
1058
|
+
] } },
|
|
1059
|
+
})
|
|
1060
|
+
evs.push({ seq: (seq += 1), type: 'tool/result', data: { message: { source: { callId: 'ca' }, content: [{ type: 'tool-result', content: [{ type: 'text', text: 'x'.repeat(2000) }] }] } } })
|
|
1061
|
+
evs.push({ seq: (seq += 1), type: 'tool/result', data: { message: { source: { callId: 'cb' }, content: [{ type: 'tool-result', content: [{ type: 'text', text: 'y'.repeat(2000) }] }] } } })
|
|
1062
|
+
|
|
1063
|
+
const at = (s) => evs.find((e) => e.seq === s)
|
|
1064
|
+
const cache = new Map()
|
|
1065
|
+
for (const e of evs) {
|
|
1066
|
+
if (e.type !== 'tool/result') continue
|
|
1067
|
+
cache.set(e.seq, { keep: false, prob: 0.1, effectProb: 0.05, chars: 2000, tool: 'read' })
|
|
1068
|
+
}
|
|
1069
|
+
const { stats } = selectReceiptRanges({
|
|
1070
|
+
surface: evs.map((e) => e.seq),
|
|
1071
|
+
eventAt: at,
|
|
1072
|
+
cache,
|
|
1073
|
+
dropVerdict: () => true,
|
|
1074
|
+
cfg: {
|
|
1075
|
+
preserveRecent: 0,
|
|
1076
|
+
compactTools: ['read', 'glob'],
|
|
1077
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
1078
|
+
evidenceGuard: false,
|
|
1079
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
1080
|
+
maxStepTextChars: 240,
|
|
1081
|
+
maxStepReasoningChars: 240,
|
|
1082
|
+
compactMinChars: 100,
|
|
1083
|
+
},
|
|
1084
|
+
})
|
|
1085
|
+
assert.equal(stats.skippedTool, 1)
|
|
1086
|
+
assert.deepEqual(Object.keys(stats.blockedToolNames).sort(), ['pwsh'],
|
|
1087
|
+
`blockedToolNames 只应含真正违规的名字,实际 ${JSON.stringify(stats.blockedToolNames)}`)
|
|
1088
|
+
assert.equal(stats.blockedToolNames.read, undefined, '通过白名单的名字不应被记为 blocked')
|
|
1089
|
+
}
|
|
1090
|
+
|
|
1091
|
+
// ---------------------------------------------------------------- 回执渲染
|
|
1092
|
+
{
|
|
1093
|
+
const evs = [
|
|
1094
|
+
assistantWithCall(1, 'c1', 'Glob', { pattern: '*.ts', path: 'server/src' }),
|
|
1095
|
+
toolResult(2, 'c1', 'z'.repeat(2000)),
|
|
1096
|
+
assistantWithCall(3, 'c2', 'Read', { file_path: 'server/src/game/pot.ts' }),
|
|
1097
|
+
toolResult(4, 'c2', 'y'.repeat(500)),
|
|
1098
|
+
]
|
|
1099
|
+
const at = (s) => evs.find((e) => e.seq === s)
|
|
1100
|
+
const surf = evs.map((e) => e.seq)
|
|
1101
|
+
const cache = new Map([
|
|
1102
|
+
[2, { keep: false, prob: 0.1, effectProb: 0.05, chars: 2000, tool: 'Glob' }],
|
|
1103
|
+
[4, { keep: false, prob: 0.1, effectProb: 0.05, chars: 500, tool: 'Read' }],
|
|
1104
|
+
])
|
|
1105
|
+
const { ranges } = selectReceiptRanges({
|
|
1106
|
+
surface: surf,
|
|
1107
|
+
eventAt: at,
|
|
1108
|
+
cache,
|
|
1109
|
+
dropVerdict: () => true,
|
|
1110
|
+
cfg: {
|
|
1111
|
+
preserveRecent: 0,
|
|
1112
|
+
compactTools: DEFAULT_COMPACT_TOOLS,
|
|
1113
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
1114
|
+
evidenceGuard: true,
|
|
1115
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
1116
|
+
maxStepTextChars: 240,
|
|
1117
|
+
maxStepReasoningChars: 240,
|
|
1118
|
+
compactMinChars: 100,
|
|
1119
|
+
},
|
|
1120
|
+
})
|
|
1121
|
+
assert.equal(ranges.length, 1)
|
|
1122
|
+
const text = renderReceipt(ranges[0], { eventAt: at })
|
|
1123
|
+
assert.ok(text.startsWith(RECEIPT_MARKER), '回执必须以识别前缀开头')
|
|
1124
|
+
assert.match(text, /s1–s4/, '必须写明被移出的 seq 范围')
|
|
1125
|
+
assert.match(text, /2 次工具调用/)
|
|
1126
|
+
// 事实必须逐字记录:模式/路径与文件路径都不能丢
|
|
1127
|
+
assert.match(text, /Glob:\*\.ts server\/src/, 'glob 的 pattern 与 path 都要逐字在回执里')
|
|
1128
|
+
assert.match(text, /server\/src\/game\/pot\.ts/)
|
|
1129
|
+
assert.match(text, /2000 字符输出/)
|
|
1130
|
+
assert.match(text, /原始事件仍完整保存在会话日志中/)
|
|
1131
|
+
// 确定性:同一输入两次渲染必须完全相同(这是"无幻觉"的可测形式)
|
|
1132
|
+
assert.equal(text, renderReceipt(ranges[0], { eventAt: at }))
|
|
1133
|
+
// 不得出现任何推断性表述
|
|
1134
|
+
for (const word of ['我们发现', '因此', '根因是', '结论', '我认为']) {
|
|
1135
|
+
assert.equal(text.includes(word), false, `回执不得包含推断性表述「${word}」`)
|
|
1136
|
+
}
|
|
1137
|
+
}
|
|
1138
|
+
|
|
1139
|
+
// ------------------------------------------------ 回执带上 assistant 可见文本原文摘录
|
|
1140
|
+
// 背景:第二层把「调用 + 结果」**整段**移出 surface,其中包含 assistant 消息本身。
|
|
1141
|
+
// 若回执只留调用事实,模型自己写下的**结论与进度**会一并消失。`maxStepTextChars`
|
|
1142
|
+
// 只拦「长文本」,而「每步一句短结论」的工作流文本很短,拦不住。
|
|
1143
|
+
// 实测影响:37 步逐文件读取任务中,模型的 12/13 条中间输出被逐步抹掉,
|
|
1144
|
+
// 随后在第 14 步只输出一句概括、不再发起工具调用,任务中途终止。
|
|
1145
|
+
{
|
|
1146
|
+
const head = {
|
|
1147
|
+
seq: 1,
|
|
1148
|
+
type: 'assistant/message',
|
|
1149
|
+
data: { message: { content: [
|
|
1150
|
+
{ type: 'reasoning', text: '这是思考草稿,不应进入回执' },
|
|
1151
|
+
{ type: 'text', text: 'identifiers:把版本字符串拆成标识符。' },
|
|
1152
|
+
{ type: 'tool-call', id: 'c1', name: 'read', arguments: JSON.stringify({ file_path: 'internal/identifiers.js' }) },
|
|
1153
|
+
] } },
|
|
1154
|
+
}
|
|
1155
|
+
const evs = [head, toolResult(2, 'c1', 'x'.repeat(900))]
|
|
1156
|
+
const at = (s) => evs.find((e) => e.seq === s)
|
|
1157
|
+
const range = {
|
|
1158
|
+
start: 1,
|
|
1159
|
+
end: 2,
|
|
1160
|
+
chars: 900,
|
|
1161
|
+
steps: [{
|
|
1162
|
+
headSeq: 1,
|
|
1163
|
+
head,
|
|
1164
|
+
calls: [{ name: 'read', arguments: JSON.stringify({ file_path: 'internal/identifiers.js' }) }],
|
|
1165
|
+
resultSeqs: [2],
|
|
1166
|
+
}],
|
|
1167
|
+
}
|
|
1168
|
+
const text = renderReceipt(range, { eventAt: at })
|
|
1169
|
+
assert.match(text, /模型原话(原文摘录):identifiers:把版本字符串拆成标识符。/,
|
|
1170
|
+
'回执必须保留该步 assistant 的可见文本原文摘录(否则模型丢失自己的进度/结论)')
|
|
1171
|
+
assert.equal(text.includes('这是思考草稿'), false, 'reasoning 草稿不得进入回执')
|
|
1172
|
+
assert.match(text, /internal\/identifiers\.js/, '调用事实仍必须在回执里')
|
|
1173
|
+
// 确定性仍然成立:同一输入两次渲染完全相同
|
|
1174
|
+
assert.equal(text, renderReceipt(range, { eventAt: at }))
|
|
1175
|
+
assert.equal(/· s\d+ 模型原话/.test(renderReceipt(range, { eventAt: at, textChars: 0 })), false,
|
|
1176
|
+
'receiptTextChars=0 必须真正关闭原文摘录,不能留下一个省略号占位')
|
|
1177
|
+
}
|
|
1178
|
+
|
|
1179
|
+
// ---------------------------------------------------------------- 证据守卫与入参渲染
|
|
1180
|
+
{
|
|
1181
|
+
assert.equal(scanEvidence('TypeError: x is undefined', ['error']).hit, true)
|
|
1182
|
+
assert.equal(scanEvidence('TypeError: x is undefined', ['ERROR']).hit, true, '大小写不敏感')
|
|
1183
|
+
assert.equal(scanEvidence('all good', ['error', 'fail']).hit, false)
|
|
1184
|
+
assert.deepEqual(scanEvidence('failed and error', ['error', 'fail']).matches, ['error', 'fail'])
|
|
1185
|
+
assert.equal(scanEvidence('', []).hit, false)
|
|
1186
|
+
|
|
1187
|
+
// 段首匹配(issue #1):命中必须落在标识符段开头 —— 压掉子串误报、保住真证据
|
|
1188
|
+
assert.equal(scanEvidence('debugging the parser', ['bug']).hit, false, 'bug 不应被 debug 触发')
|
|
1189
|
+
assert.equal(scanEvidence('__debug__', ['bug']).hit, false, '下划线包裹的复合标识符也不该触发')
|
|
1190
|
+
assert.equal(scanEvidence('this.debug = 1', ['bug']).hit, false)
|
|
1191
|
+
assert.equal(scanEvidence('bugs found', ['bug']).hit, true, '复数形式仍是证据')
|
|
1192
|
+
assert.equal(scanEvidence('bugfix applied', ['bug']).hit, true)
|
|
1193
|
+
assert.equal(scanEvidence('errors: 3', ['error']).hit, true, '复数形式仍是证据')
|
|
1194
|
+
assert.equal(scanEvidence('getError()', ['error']).hit, true, '驼峰分界算段首')
|
|
1195
|
+
assert.equal(scanEvidence('myerror', ['error']).hit, false, '无分隔符的复合标识符不算段首')
|
|
1196
|
+
assert.equal(scanEvidence('TODOs left', ['todo']).hit, true)
|
|
1197
|
+
assert.equal(scanEvidence('pseudotodo', ['todo']).hit, false)
|
|
1198
|
+
assert.equal(scanEvidence('default: x', ['fail:']).hit, false, 'fail: 不应被 default: 触发')
|
|
1199
|
+
assert.equal(scanEvidence('stack trace follows', ['stack trace']).hit, true)
|
|
1200
|
+
assert.equal(scanEvidence('mystack trace', ['stack trace']).hit, false)
|
|
1201
|
+
|
|
1202
|
+
// 入参渲染只接 tool-call **块**(不是裸 args)—— 契约写在测试里
|
|
1203
|
+
assert.equal(renderCallArgs({ arguments: { file_path: 'a/b.ts' } }), 'a/b.ts')
|
|
1204
|
+
assert.equal(renderCallArgs({ arguments: { command: 'ls -la', cwd: '/x' } }), 'ls -la cwd=/x',
|
|
1205
|
+
'次要字段要带 key 前缀,否则光看值不知道是哪个参数')
|
|
1206
|
+
assert.equal(renderCallArgs({ arguments: { pattern: 'p', path: 'src' } }), 'p src',
|
|
1207
|
+
'Grep 的 pattern 应排在 path 前面')
|
|
1208
|
+
assert.equal(renderCallArgs({ arguments: '{"file_path":"a.ts"}' }), 'a.ts', '字符串形式的 JSON 入参也要能解析')
|
|
1209
|
+
assert.equal(renderCallArgs({ arguments: 'not json' }), 'not json')
|
|
1210
|
+
assert.equal(renderCallArgs({ arguments: { unknown_key: 'v' } }), 'unknown_key=v')
|
|
1211
|
+
const payloadOnly = renderCallArgs({ arguments: { content: 'x'.repeat(500) } })
|
|
1212
|
+
assert.ok(payloadOnly.startsWith('{"content"'), '全是载荷型字段时整体序列化')
|
|
1213
|
+
assert.ok(payloadOnly.endsWith('…'), '并被截断')
|
|
1214
|
+
assert.equal(payloadOnly.length, 121)
|
|
1215
|
+
assert.equal(renderCallArgs({ arguments: null }), '')
|
|
1216
|
+
assert.equal(renderCallArgs({ arguments: { command: 'x'.repeat(300) } }, 20).length, 21, '超长入参要截断并带省略号')
|
|
1217
|
+
}
|
|
1218
|
+
|
|
1219
|
+
// ---------------------------------------------------------------- 工具名归一化与白名单陷阱
|
|
1220
|
+
// 这一组测试的由来(真实事故):真实 DSH 的工具名是 `pwsh` / `read` / `glob`(全小写、
|
|
1221
|
+
// shell 叫 pwsh),而我最初猜的白名单是 Claude Code 风格的 PascalCase —— 在真实会话里
|
|
1222
|
+
// **命中 0/11**,第二层因此静默地永不触发。所以默认不用白名单,且比较必须归一化。
|
|
1223
|
+
{
|
|
1224
|
+
assert.equal(normalizeToolName('Read'), 'read')
|
|
1225
|
+
assert.equal(normalizeToolName(' Pwsh '), 'pwsh')
|
|
1226
|
+
assert.equal(normalizeToolName('multi_edit'), 'multiedit')
|
|
1227
|
+
assert.equal(normalizeToolName('Get-ChildItem'), 'getchilditem')
|
|
1228
|
+
assert.equal(normalizeToolName(null), '')
|
|
1229
|
+
|
|
1230
|
+
assert.equal(isToolIn(['Edit'], 'edit'), true, '黑名单必须能拦住小写变体(否则保护静默失效)')
|
|
1231
|
+
assert.equal(isToolIn(['edit'], 'Edit'), true)
|
|
1232
|
+
assert.equal(isToolIn(['MultiEdit'], 'multi_edit'), true)
|
|
1233
|
+
assert.equal(isToolIn(['ApplyPatch'], 'apply_patch'), true)
|
|
1234
|
+
assert.equal(isToolIn(['read'], 'read'), true)
|
|
1235
|
+
assert.equal(isToolIn(['read'], 'write'), false)
|
|
1236
|
+
assert.equal(isToolIn([], 'read'), false)
|
|
1237
|
+
assert.equal(isToolIn(['read'], ''), false)
|
|
1238
|
+
|
|
1239
|
+
// 默认值就是设计决策本身:默认 = **只读白名单**(外部审查的结论:
|
|
1240
|
+
// "空白名单 = 允许一切 shell 调用"不是只读安全默认;而白名单失效的风险
|
|
1241
|
+
// 由 normalizeToolName + blockedToolNames 上报兜住,可观测)
|
|
1242
|
+
assert.ok(DEFAULT_COMPACT_TOOLS.length > 0, 'compactTools 默认必须是只读白名单,不能为空')
|
|
1243
|
+
assert.deepEqual(DEFAULT_COMPACT_TOOLS, DSH_READONLY_TOOLS)
|
|
1244
|
+
for (const shell of ['pwsh', 'bash', 'sh', 'shell', 'command']) {
|
|
1245
|
+
assert.equal(isToolIn(DEFAULT_COMPACT_TOOLS, shell), false,
|
|
1246
|
+
`shell 类工具「${shell}」绝不能进默认白名单(pwsh Remove-Item 复现过)`)
|
|
1247
|
+
}
|
|
1248
|
+
assert.ok(DSH_READONLY_TOOLS.includes('read'))
|
|
1249
|
+
assert.ok(DSH_READONLY_TOOLS.includes('glob'))
|
|
1250
|
+
assert.ok(DSH_READONLY_TOOLS.includes('grep'))
|
|
1251
|
+
assert.equal(DSH_READONLY_TOOLS.includes('Bash'), false, '预设里不该有猜出来的 PascalCase 名字')
|
|
1252
|
+
assert.ok(DEFAULT_NEVER_COMPACT_TOOLS.some((n) => normalizeToolName(n) === 'edit'))
|
|
1253
|
+
assert.ok(DEFAULT_NEVER_COMPACT_TOOLS.some((n) => normalizeToolName(n) === 'multiedit'))
|
|
1254
|
+
}
|
|
1255
|
+
|
|
1256
|
+
// ---------------------------------------------------------------- 工具名索引(含真实事件类型)
|
|
1257
|
+
{
|
|
1258
|
+
// 真实 DSH 同时存在两种来源:扁平的 tool/call 事件,以及 assistant 消息里的 tool-call 块
|
|
1259
|
+
const evs = [
|
|
1260
|
+
{ seq: 1, type: 'tool/call', data: { callId: 'c1', name: 'pwsh', arguments: '{}' } },
|
|
1261
|
+
{ seq: 2, type: 'tool/result', data: { message: { source: { kind: 'tool', callId: 'c1' }, content: [{ type: 'tool-result', content: [{ type: 'text', text: 'ok' }] }] } } },
|
|
1262
|
+
{ seq: 3, type: 'assistant/message', data: { message: { content: [{ type: 'tool-call', id: 'c2', name: 'read', arguments: '{"file_path":"a.ts"}' }] } } },
|
|
1263
|
+
{ seq: 4, type: 'tool/result', data: { message: { source: { kind: 'tool', callId: 'c2' }, content: [{ type: 'tool-result', content: [{ type: 'text', text: 'ok' }] }] } } },
|
|
1264
|
+
]
|
|
1265
|
+
const at = (s) => evs.find((e) => e.seq === s)
|
|
1266
|
+
const index = buildToolNameIndex(evs)
|
|
1267
|
+
assert.equal(index.get('c1'), 'pwsh', 'tool/call 事件必须被索引(最直接的来源)')
|
|
1268
|
+
assert.equal(index.get('c2'), 'read', 'assistant 消息里的 tool-call 块也要索引')
|
|
1269
|
+
assert.equal(toolNameOf(at(2), index), 'pwsh')
|
|
1270
|
+
assert.equal(toolNameOf(at(4), index), 'read')
|
|
1271
|
+
assert.equal(toolNameOf({ seq: 9, type: 'tool/result', data: { message: { source: { callId: 'nope' } } } }, index), 'unknown')
|
|
1272
|
+
|
|
1273
|
+
// probeToolNames 是"白名单为什么不命中"的取证块
|
|
1274
|
+
const probe = probeToolNames({ surface: [1, 2, 3, 4], eventAt: at, events: evs, limit: 10 })
|
|
1275
|
+
assert.equal(probe.indexSize, 2)
|
|
1276
|
+
assert.deepEqual(probe.names, ['pwsh', 'read'])
|
|
1277
|
+
assert.equal(probe.resolved, 2)
|
|
1278
|
+
assert.equal(probe.unresolved, 0)
|
|
1279
|
+
}
|
|
1280
|
+
|
|
1281
|
+
// ---------------------------------------------------------------- 工具名统计与推理文本门控
|
|
1282
|
+
{
|
|
1283
|
+
const evs = []
|
|
1284
|
+
let seq = 0
|
|
1285
|
+
const push = (e) => evs.push(e)
|
|
1286
|
+
const addStep = (tool, text, reasoning) => {
|
|
1287
|
+
const callId = `c${seq + 1}`
|
|
1288
|
+
const content = [{ type: 'tool-call', id: callId, name: tool, arguments: {} }]
|
|
1289
|
+
if (text != null) content.unshift({ type: 'text', text })
|
|
1290
|
+
if (reasoning != null) content.unshift({ type: 'reasoning', text: reasoning })
|
|
1291
|
+
push({ seq: (seq += 1), type: 'assistant/message', data: { message: { content } } })
|
|
1292
|
+
push({
|
|
1293
|
+
seq: (seq += 1),
|
|
1294
|
+
type: 'tool/result',
|
|
1295
|
+
data: { message: { source: { callId }, content: [{ type: 'tool-result', content: [{ type: 'text', text: 'z'.repeat(3000) }] }] } },
|
|
1296
|
+
})
|
|
1297
|
+
return seq
|
|
1298
|
+
}
|
|
1299
|
+
push({ seq: (seq += 1), type: 'user/message', data: { content: [{ type: 'text', text: 'go' }] } })
|
|
1300
|
+
addStep('pwsh', '看一下。', null)
|
|
1301
|
+
addStep('read', '看一下。', null)
|
|
1302
|
+
addStep('edit', '看一下。', null) // 黑名单
|
|
1303
|
+
addStep('read', '看一下。', '想'.repeat(400)) // 长 reasoning —— 必须被看见
|
|
1304
|
+
const at = (s) => evs.find((e) => e.seq === s)
|
|
1305
|
+
const surf = evs.map((e) => e.seq)
|
|
1306
|
+
const cache = new Map()
|
|
1307
|
+
for (const e of evs) {
|
|
1308
|
+
if (e.type !== 'tool/result') continue
|
|
1309
|
+
cache.set(e.seq, { keep: false, prob: 0.1, effectProb: 0.05, chars: 3000 })
|
|
1310
|
+
}
|
|
1311
|
+
const { ranges, stats } = selectReceiptRanges({
|
|
1312
|
+
surface: surf,
|
|
1313
|
+
eventAt: at,
|
|
1314
|
+
cache,
|
|
1315
|
+
dropVerdict: () => true,
|
|
1316
|
+
cfg: {
|
|
1317
|
+
preserveRecent: 0,
|
|
1318
|
+
compactTools: [], // 默认:只用黑名单
|
|
1319
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
1320
|
+
evidenceGuard: true,
|
|
1321
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
1322
|
+
maxStepTextChars: 240,
|
|
1323
|
+
maxStepReasoningChars: 240,
|
|
1324
|
+
compactMinChars: 100,
|
|
1325
|
+
},
|
|
1326
|
+
})
|
|
1327
|
+
// 前两步合格(pwsh 不在黑名单里、read 合格)→ 合成一段
|
|
1328
|
+
assert.equal(ranges.length, 1)
|
|
1329
|
+
assert.equal(ranges[0].steps.length, 2)
|
|
1330
|
+
assert.equal(stats.skippedTool, 1, 'edit 应被黑名单拦下(归一化后比较)')
|
|
1331
|
+
assert.deepEqual(stats.blockedToolNames, { edit: 1 }, '必须如实记下被拦下的名字')
|
|
1332
|
+
assert.deepEqual(stats.allowedToolNames, { pwsh: 1, read: 1 }, '通过的名字也要记,便于自查')
|
|
1333
|
+
assert.equal(stats.skippedReasoning, 1, '长 reasoning 必须计入 reasoning 轴门控(只数 text 会让门控形同虚设)')
|
|
1334
|
+
assert.equal(stats.skippedText, 0, 'reasoning 轴拦截不得计入 text 轴计数')
|
|
1335
|
+
|
|
1336
|
+
// 同一份数据,配上白名单 → 只允许 read(pwsh 被拦)
|
|
1337
|
+
const allow = selectReceiptRanges({
|
|
1338
|
+
surface: surf,
|
|
1339
|
+
eventAt: at,
|
|
1340
|
+
cache,
|
|
1341
|
+
dropVerdict: () => true,
|
|
1342
|
+
cfg: {
|
|
1343
|
+
preserveRecent: 0,
|
|
1344
|
+
compactTools: ['read'],
|
|
1345
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
1346
|
+
evidenceGuard: true,
|
|
1347
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
1348
|
+
maxStepTextChars: 240,
|
|
1349
|
+
maxStepReasoningChars: 240,
|
|
1350
|
+
compactMinChars: 100,
|
|
1351
|
+
},
|
|
1352
|
+
})
|
|
1353
|
+
assert.equal(allow.ranges.length, 1, '只允许 read → 剩短文本那一读'
|
|
1354
|
+
+ '(另一条 read 带长 reasoning,被 reasoning 轴门控拦下)')
|
|
1355
|
+
assert.equal(allow.ranges[0].steps.length, 1)
|
|
1356
|
+
assert.equal(allow.ranges[0].steps[0].calls[0].name, 'read')
|
|
1357
|
+
assert.equal(allow.stats.skippedReasoning, 1, '长 reasoning 的那条 read 必须被拦下')
|
|
1358
|
+
assert.equal(allow.stats.skippedTool, 2, 'pwsh 与 edit 都被工具门拦下')
|
|
1359
|
+
assert.deepEqual(allow.stats.blockedToolNames, { pwsh: 1, edit: 1 })
|
|
1360
|
+
}
|
|
1361
|
+
|
|
1362
|
+
// ---------------------------------------------------------------- 外部审查回归:shell 破坏性命令
|
|
1363
|
+
// 审查案例:`pwsh: Remove-Item important.txt` 在默认配置下会被选中整对压缩。
|
|
1364
|
+
// 修复后(默认 = 只读白名单)必须被工具门拦下,且名字要出现在 blockedToolNames 里。
|
|
1365
|
+
{
|
|
1366
|
+
const evs = []
|
|
1367
|
+
let seq = 0
|
|
1368
|
+
const push = (e) => evs.push(e)
|
|
1369
|
+
const addStep = (tool, args, output) => {
|
|
1370
|
+
const callId = `c${seq + 1}`
|
|
1371
|
+
push({ seq: (seq += 1), type: 'assistant/message', data: { message: { content: [{ type: 'tool-call', id: callId, name: tool, arguments: args }] } } })
|
|
1372
|
+
push({ seq: (seq += 1), type: 'tool/result', data: { message: { source: { callId }, content: [{ type: 'tool-result', content: [{ type: 'text', text: output }] }] } } })
|
|
1373
|
+
return seq
|
|
1374
|
+
}
|
|
1375
|
+
push({ seq: (seq += 1), type: 'user/message', data: { content: [{ type: 'text', text: 'clean up' }] } })
|
|
1376
|
+
addStep('pwsh', { command: 'Remove-Item important.txt' }, 'done')
|
|
1377
|
+
addStep('read', { file_path: 'notes.md' }, 'n'.repeat(3000))
|
|
1378
|
+
const at = (s) => evs.find((e) => e.seq === s)
|
|
1379
|
+
const surf = evs.map((e) => e.seq)
|
|
1380
|
+
// cache 按工具**结果**的 seq 索引:user(1) pwsh头(2) pwsh结果(3) read头(4) read结果(5)
|
|
1381
|
+
const cache = new Map([[3, { keep: false, prob: 0.05, effectProb: 0.02, chars: 10 }]])
|
|
1382
|
+
cache.set(5, { keep: false, prob: 0.05, effectProb: 0.02, chars: 3000 })
|
|
1383
|
+
|
|
1384
|
+
const buildCfg = (compactTools) => ({
|
|
1385
|
+
preserveRecent: 0,
|
|
1386
|
+
compactTools,
|
|
1387
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
1388
|
+
evidenceGuard: true,
|
|
1389
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
1390
|
+
maxStepTextChars: 240,
|
|
1391
|
+
maxStepReasoningChars: 240,
|
|
1392
|
+
compactMinChars: 100,
|
|
1393
|
+
})
|
|
1394
|
+
const runCase = (compactTools) => selectReceiptRanges({
|
|
1395
|
+
surface: surf,
|
|
1396
|
+
eventAt: at,
|
|
1397
|
+
cache,
|
|
1398
|
+
dropVerdict: () => true,
|
|
1399
|
+
cfg: buildCfg(compactTools),
|
|
1400
|
+
})
|
|
1401
|
+
|
|
1402
|
+
// 默认配置:pwsh 被拦,即使其结果"可丢 + 无副作用"(两轴全过也没用 —— 工具门在前)
|
|
1403
|
+
const safe = runCase(DEFAULT_COMPACT_TOOLS)
|
|
1404
|
+
assert.equal(safe.ranges.length, 1)
|
|
1405
|
+
assert.deepEqual(safe.ranges[0].steps.map((s) => s.calls[0].name), ['read'],
|
|
1406
|
+
'默认配置下 pwsh 步绝不能进压缩范围')
|
|
1407
|
+
assert.equal(safe.stats.blockedToolNames.pwsh, 1, '被拦的名字必须可观测')
|
|
1408
|
+
|
|
1409
|
+
// 反事实:显式放宽到 [] 才会让 pwsh 进候选 —— 证明拦截来自白名单这一道门
|
|
1410
|
+
const relaxed = runCase([])
|
|
1411
|
+
assert.equal(relaxed.ranges[0].steps.length, 2, '放宽模式下 pwsh 可进候选(显式 opt-in)')
|
|
1412
|
+
}
|
|
1413
|
+
|
|
1414
|
+
// ---------------------------------------------------------------- sessionEvents 回退链
|
|
1415
|
+
// 实测活的 DSH 会话对象上 session.events 是 undefined —— 只有 eventAt 与 surface 可用。
|
|
1416
|
+
// 之前所有 `session.events ?? []` 都拿到空数组,导致工具名索引 0 条、任务目标丢失。
|
|
1417
|
+
{
|
|
1418
|
+
const evs = [
|
|
1419
|
+
{ seq: 1, type: 'tool/call', data: { callId: 'c1', name: 'read', arguments: '{}' } },
|
|
1420
|
+
{ seq: 2, type: 'assistant/message', data: { message: { content: [{ type: 'tool-call', id: 'c2', name: 'pwsh', arguments: '{}' }] } } },
|
|
1421
|
+
{ seq: 3, type: 'tool/result', data: { message: { source: { callId: 'c1' }, content: [{ type: 'tool-result', content: [{ type: 'text', text: 'ok' }] }] } } },
|
|
1422
|
+
]
|
|
1423
|
+
const bySeq = new Map(evs.map((e) => [e.seq, e]))
|
|
1424
|
+
const mkSession = (overrides) => ({
|
|
1425
|
+
surface: { nodes: [1, 2, 3] },
|
|
1426
|
+
eventAt: (seq) => bySeq.get(seq) ?? null,
|
|
1427
|
+
...overrides,
|
|
1428
|
+
})
|
|
1429
|
+
|
|
1430
|
+
// ① 有 .events 数组 → 直接用
|
|
1431
|
+
assert.deepEqual(sessionEvents(mkSession({ events: evs })), evs)
|
|
1432
|
+
// ② .events 缺失/undefined(真实 DSH 的情况)→ 回退到遍历 surface + eventAt
|
|
1433
|
+
const fallback = sessionEvents(mkSession({ events: undefined }))
|
|
1434
|
+
assert.equal(fallback.length, 3, 'session.events 缺失时必须回退到 surface+eventAt')
|
|
1435
|
+
assert.deepEqual(fallback.map((e) => e.seq), [1, 2, 3])
|
|
1436
|
+
// ③ snapshotEvents 也可用
|
|
1437
|
+
const snap = sessionEvents(mkSession({ snapshotEvents: () => evs }))
|
|
1438
|
+
assert.equal(snap.length, 3)
|
|
1439
|
+
// ④ 什么都没有 → 空数组,不抛错
|
|
1440
|
+
assert.deepEqual(sessionEvents({ surface: { nodes: [] } }), [])
|
|
1441
|
+
assert.deepEqual(sessionEvents(null), [])
|
|
1442
|
+
}
|
|
1443
|
+
|
|
1444
|
+
// ---------------------------------------------------------------- 打包完整性
|
|
1445
|
+
// 这一类错误只有"装进宿主后启动"才会暴露:源码里加了一个模块、忘了同步
|
|
1446
|
+
// package.json 的 files 或复制清单 → 宿主里 ERR_MODULE_NOT_FOUND → 整个 profile 起不来。
|
|
1447
|
+
// 在这里提前守住。
|
|
1448
|
+
{
|
|
1449
|
+
const pkg = JSON.parse(readFileSync(join(here, 'package.json'), 'utf8'))
|
|
1450
|
+
const declared = new Set([...(pkg.files ?? []), 'package.json'])
|
|
1451
|
+
const walk = (file) => {
|
|
1452
|
+
assert.ok(existsSync(join(here, file)), 'Missing packaged path: ' + file)
|
|
1453
|
+
return statSync(join(here, file)).isDirectory()
|
|
1454
|
+
? readdirSync(join(here, file)).flatMap(name => walk(file + '/' + name))
|
|
1455
|
+
: [file]
|
|
1456
|
+
}
|
|
1457
|
+
const packaged = [...declared].flatMap(walk)
|
|
1458
|
+
const srcFiles = packaged.filter(file => /\.(?:js|mjs)$/.test(file))
|
|
1459
|
+
assert.ok(srcFiles.length >= 8)
|
|
1460
|
+
assert.ok(declared.has('index.js'))
|
|
1461
|
+
for (const file of srcFiles) {
|
|
1462
|
+
const text = readFileSync(join(here, file), 'utf8')
|
|
1463
|
+
for (const m of text.matchAll(/(?:from\s+|import\()['"](\.[^'"]+)['"]/g)) {
|
|
1464
|
+
const target = resolve(here, dirname(file), m[1])
|
|
1465
|
+
assert.ok(existsSync(target), file + ' references missing ' + m[1])
|
|
1466
|
+
assert.ok(packaged.some(p => resolve(here, p) === target), file + ' references an unpackaged module')
|
|
1467
|
+
}
|
|
1468
|
+
}
|
|
1469
|
+
}
|
|
1470
|
+
|
|
1471
|
+
// ---------------------------------------------------------------- 分位总体的可压缩性(外部审查)
|
|
1472
|
+
|
|
1473
|
+
{
|
|
1474
|
+
// 背景:第一层的判定缓存复用 selectCandidates 的候选,那份候选只排除黑名单
|
|
1475
|
+
// (第一层没有白名单),所以 shell 之类不可整对移出的调用也会进缓存。
|
|
1476
|
+
// 第二层若直接拿整份缓存当分位总体,尾部名额会被这些节点占掉、随后又被工具门
|
|
1477
|
+
// 全部拒绝 → 静默少压缩。下面三组断言把口径钉住。
|
|
1478
|
+
|
|
1479
|
+
const cfg = { compactTools: ['read', 'glob'], neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS }
|
|
1480
|
+
|
|
1481
|
+
assert.equal(isCompactableTool('read', cfg), true, '白名单内的只读工具可移出')
|
|
1482
|
+
assert.equal(isCompactableTool('pwsh', cfg), false, '白名单外的 shell 不可移出')
|
|
1483
|
+
assert.equal(isCompactableTool('edit', cfg), false, '黑名单优先于白名单')
|
|
1484
|
+
assert.equal(isCompactableTool('Edit', cfg), false, '黑名单比较要归一化(Edit ≡ edit)')
|
|
1485
|
+
|
|
1486
|
+
// compactTools=[] 是显式的不安全模式:只受黑名单约束
|
|
1487
|
+
const relaxed = { compactTools: [], neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS }
|
|
1488
|
+
assert.equal(isCompactableTool('pwsh', relaxed), true, '放宽模式下 shell 变成可移出')
|
|
1489
|
+
assert.equal(isCompactableTool('edit', relaxed), false, '放宽模式下黑名单仍然生效')
|
|
1490
|
+
|
|
1491
|
+
// 尾部名额不应被不可移出的节点占满:6 条 shell(更“过期”)+ 4 条 read
|
|
1492
|
+
const mk = (seq, tool, p, e) => ({ seq, tool, prob: p, effectProb: e })
|
|
1493
|
+
const verdicts = [
|
|
1494
|
+
mk(1, 'pwsh', 0.10, 0.05), mk(2, 'pwsh', 0.11, 0.06), mk(3, 'pwsh', 0.12, 0.07),
|
|
1495
|
+
mk(4, 'pwsh', 0.13, 0.08), mk(5, 'pwsh', 0.14, 0.09), mk(6, 'pwsh', 0.15, 0.10),
|
|
1496
|
+
mk(7, 'read', 0.40, 0.30), mk(8, 'read', 0.45, 0.35), mk(9, 'read', 0.50, 0.40), mk(10, 'read', 0.55, 0.45),
|
|
1497
|
+
]
|
|
1498
|
+
const filtered = verdicts.filter((v) => isCompactableTool(v.tool, cfg))
|
|
1499
|
+
const eligible = computeEligibleSeqs(filtered, { quantile: 0.34, minCandidates: 4 })
|
|
1500
|
+
assert.deepEqual([...eligible], [7], '过滤后尾部应落在可移出的 read 上,而不是被 shell 占满')
|
|
1501
|
+
assert.equal(computeEligibleSeqs(verdicts, { quantile: 0.34, minCandidates: 4 }).size > 0, true,
|
|
1502
|
+
'不过滤时仍会给出(无用的)尾部——这正是需要过滤的原因')
|
|
1503
|
+
}
|
|
1504
|
+
|
|
1505
|
+
// ---------------------------------------------------------------- 小总体降级(issue #27 回归)
|
|
1506
|
+
|
|
1507
|
+
{
|
|
1508
|
+
// 修复前:`usable.length < minCandidates` 一律返回空集。后果是**只读工具在写/执行
|
|
1509
|
+
// 密集会话里占少数时,第二层在绝大多数真实会话中静默不工作**,而报错只说
|
|
1510
|
+
// "需要 ≥4 个",读起来像"样本确实不够"而不像 bug。这个块把降级行为钉成断言。
|
|
1511
|
+
const near = [
|
|
1512
|
+
{ seq: 1, prob: 0.10, effectProb: 0.05 }, // 两轴都远低于 0.2 → 该选中
|
|
1513
|
+
{ seq: 2, prob: 0.30, effectProb: 0.10 }, // effect 低但 prob 不够低 → 单轴尾部,不该选
|
|
1514
|
+
{ seq: 3, prob: 0.50, effectProb: 0.45 }, // 两轴都高 → 不该选
|
|
1515
|
+
]
|
|
1516
|
+
const notes = []
|
|
1517
|
+
const small = computeEligibleSeqs(near, {
|
|
1518
|
+
quantile: 0.34, minCandidates: 4, minCandidatesForAbsolute: 3, floorThreshold: 0.2,
|
|
1519
|
+
onNote: (n) => notes.push(n),
|
|
1520
|
+
})
|
|
1521
|
+
assert.deepEqual([...small], [1],
|
|
1522
|
+
'总体=3 (<4) 时不得直接放弃,应降级为绝对下限:仅两轴同时 <0.2 的入选')
|
|
1523
|
+
assert.equal(notes.length, 1, '降级必须发出说明(否则用户又分不清"样本不够"和"功能坏了")')
|
|
1524
|
+
assert.ok(/降级为绝对下限/.test(notes[0]), `说明文案应点明降级:${notes[0]}`)
|
|
1525
|
+
|
|
1526
|
+
// 绝对下限比 compactThreshold(0.5) 严:0.3/0.3 在 absolute 模式下会被选中,
|
|
1527
|
+
// 但在小总体的降级模式下必须被拒(没有相对信息时只能用"更保守"来补偿)
|
|
1528
|
+
const mid = [
|
|
1529
|
+
{ seq: 4, prob: 0.30, effectProb: 0.30 },
|
|
1530
|
+
{ seq: 5, prob: 0.31, effectProb: 0.32 },
|
|
1531
|
+
{ seq: 6, prob: 0.33, effectProb: 0.33 },
|
|
1532
|
+
]
|
|
1533
|
+
assert.equal(computeEligibleSeqs(mid, {
|
|
1534
|
+
quantile: 0.34, minCandidates: 4, minCandidatesForAbsolute: 3, floorThreshold: 0.2,
|
|
1535
|
+
}).size, 0, '降级模式的阈值必须明显严于 compactThreshold,两轴 0.3 不该被放行')
|
|
1536
|
+
|
|
1537
|
+
// 样本连最低线都不到 → 仍然不做,且说明里点出原因
|
|
1538
|
+
const tinyNotes = []
|
|
1539
|
+
assert.equal(computeEligibleSeqs(near.slice(0, 2), {
|
|
1540
|
+
quantile: 0.34, minCandidates: 4, minCandidatesForAbsolute: 3,
|
|
1541
|
+
onNote: (n) => tinyNotes.push(n),
|
|
1542
|
+
}).size, 0)
|
|
1543
|
+
assert.ok(/低于绝对下限模式的最低样本/.test(tinyNotes[0] ?? ''), `应说明样本过少:${tinyNotes[0]}`)
|
|
1544
|
+
|
|
1545
|
+
// 总体达标时**不得**触发降级分支(口径不能被悄悄改掉)
|
|
1546
|
+
const bigNotes = []
|
|
1547
|
+
const big = computeEligibleSeqs(verdicts6(), {
|
|
1548
|
+
quantile: 0.5, minCandidates: 4, minCandidatesForAbsolute: 3,
|
|
1549
|
+
onNote: (n) => bigNotes.push(n),
|
|
1550
|
+
})
|
|
1551
|
+
assert.equal(bigNotes.length, 0, '总体达标时必须走正常的相对分位路径,不得降级')
|
|
1552
|
+
assert.ok(big.size > 0, '正常路径仍应给出结果')
|
|
1553
|
+
|
|
1554
|
+
function verdicts6() {
|
|
1555
|
+
return [
|
|
1556
|
+
{ seq: 10, prob: 0.10, effectProb: 0.05 }, { seq: 12, prob: 0.12, effectProb: 0.06 },
|
|
1557
|
+
{ seq: 14, prob: 0.11, effectProb: 0.07 }, { seq: 20, prob: 0.13, effectProb: 0.30 },
|
|
1558
|
+
{ seq: 22, prob: 0.14, effectProb: 0.33 }, { seq: 24, prob: 0.15, effectProb: 0.28 },
|
|
1559
|
+
]
|
|
1560
|
+
}
|
|
1561
|
+
}
|
|
1562
|
+
|
|
1563
|
+
// ---------------------------------------------------------------- 判定值的有效性(外部审查)
|
|
1564
|
+
|
|
1565
|
+
{
|
|
1566
|
+
// typeof NaN === 'number',所以用 typeof 过滤会让 NaN 混进总体;而排序比较
|
|
1567
|
+
// (a-b) 返回 NaN 被 V8 当作“相等”不换位,NaN 项便按数组位置混进尾部。
|
|
1568
|
+
const v = [
|
|
1569
|
+
{ seq: 1, prob: NaN, effectProb: 0.10 },
|
|
1570
|
+
{ seq: 2, prob: 0.40, effectProb: 0.20 },
|
|
1571
|
+
{ seq: 3, prob: 0.50, effectProb: 0.30 },
|
|
1572
|
+
{ seq: 4, prob: 0.60, effectProb: 0.40 },
|
|
1573
|
+
{ seq: 5, prob: 0.70, effectProb: 0.50 },
|
|
1574
|
+
]
|
|
1575
|
+
const r = computeEligibleSeqs(v, { quantile: 0.34, minCandidates: 4 })
|
|
1576
|
+
assert.equal(r.has(1), false, 'NaN 判定值不得被选为可整对移出')
|
|
1577
|
+
|
|
1578
|
+
const withUndefined = v.map((x) => (x.seq === 1 ? { seq: 1, prob: undefined, effectProb: null } : x))
|
|
1579
|
+
assert.equal(computeEligibleSeqs(withUndefined, { quantile: 0.34, minCandidates: 4 }).has(1), false,
|
|
1580
|
+
'undefined/null 判定值同样不得入选')
|
|
1581
|
+
}
|
|
1582
|
+
|
|
1583
|
+
// ---------------------------------------------------------------- 文本门控两轴分离(issue #26 回归)
|
|
1584
|
+
|
|
1585
|
+
{
|
|
1586
|
+
// 修复前:assistantTextChars = text + reasoning 累加,与单一阈值 maxStepTextChars 比较。
|
|
1587
|
+
// 后果:reasoning 的分布(实测 0~1207,且会溢出到数千)完全主导阈值,只要模型多写几句
|
|
1588
|
+
// 草稿,第二层就在没有任何日志的情况下整层失效。这个块把"两轴独立"钉成断言。
|
|
1589
|
+
const evs = []
|
|
1590
|
+
let seq = 0
|
|
1591
|
+
const addStep = (tool, text, reasoning) => {
|
|
1592
|
+
const callId = `t${seq + 1}`
|
|
1593
|
+
const content = [{ type: 'tool-call', id: callId, name: tool, arguments: {} }]
|
|
1594
|
+
if (text != null) content.unshift({ type: 'text', text })
|
|
1595
|
+
if (reasoning != null) content.unshift({ type: 'reasoning', text: reasoning })
|
|
1596
|
+
evs.push({ seq: (seq += 1), type: 'assistant/message', data: { message: { content } } })
|
|
1597
|
+
evs.push({
|
|
1598
|
+
seq: (seq += 1),
|
|
1599
|
+
type: 'tool/result',
|
|
1600
|
+
data: { message: { source: { callId }, content: [{ type: 'tool-result', content: [{ type: 'text', text: 'z'.repeat(3000) }] }] } },
|
|
1601
|
+
})
|
|
1602
|
+
}
|
|
1603
|
+
evs.push({ seq: (seq += 1), type: 'user/message', data: { content: [{ type: 'text', text: 'go' }] } })
|
|
1604
|
+
addStep('read', '看一下。', '想'.repeat(1300)) // 长 reasoning、短 text ← 修复前的"整层关闭"元凶
|
|
1605
|
+
const at2 = (s) => evs.find((e) => e.seq === s)
|
|
1606
|
+
const cache2 = new Map()
|
|
1607
|
+
for (const e of evs) if (e.type === 'tool/result') cache2.set(e.seq, { keep: false, prob: 0.1, effectProb: 0.05, chars: 3000 })
|
|
1608
|
+
const run2 = (over) => selectReceiptRanges({
|
|
1609
|
+
surface: evs.map((e) => e.seq),
|
|
1610
|
+
eventAt: at2,
|
|
1611
|
+
cache: cache2,
|
|
1612
|
+
dropVerdict: () => true,
|
|
1613
|
+
cfg: {
|
|
1614
|
+
preserveRecent: 0,
|
|
1615
|
+
compactTools: [],
|
|
1616
|
+
neverCompactTools: DEFAULT_NEVER_COMPACT_TOOLS,
|
|
1617
|
+
evidenceGuard: false,
|
|
1618
|
+
evidencePatterns: DEFAULT_EVIDENCE_PATTERNS,
|
|
1619
|
+
maxStepTextChars: 1200,
|
|
1620
|
+
maxStepReasoningChars: 4000,
|
|
1621
|
+
compactMinChars: 100,
|
|
1622
|
+
...over,
|
|
1623
|
+
},
|
|
1624
|
+
})
|
|
1625
|
+
|
|
1626
|
+
// ① 生产默认下,reasoning=1300 的步骤**必须合格**(修复前这里会是 0 段)
|
|
1627
|
+
const defaults = run2()
|
|
1628
|
+
assert.equal(defaults.ranges.length, 1,
|
|
1629
|
+
'reasoning=1300 在默认 maxStepReasoningChars=4000 下必须仍可整对移出'
|
|
1630
|
+
+ '(修复前与 1200 的 text 阈值累加 → 0 段,第二层静默失效)')
|
|
1631
|
+
assert.equal(defaults.stats.skippedReasoning, 0)
|
|
1632
|
+
assert.equal(defaults.stats.skippedText, 0)
|
|
1633
|
+
|
|
1634
|
+
// ② reasoning 真的超限时才拦,且只记到 reasoning 轴
|
|
1635
|
+
const tight = run2({ maxStepReasoningChars: 500 })
|
|
1636
|
+
assert.equal(tight.ranges.length, 0, 'reasoning 超过自己的阈值时必须拦下')
|
|
1637
|
+
assert.equal(tight.stats.skippedReasoning, 1)
|
|
1638
|
+
assert.equal(tight.stats.skippedText, 0, 'reasoning 轴拦截不得污染 text 轴计数')
|
|
1639
|
+
|
|
1640
|
+
// ③ 反过来:reasoning 很长、但 text 超限 → 必须记到 text 轴
|
|
1641
|
+
// (这条在修复前是"分不出来的"——两个原因共用一个计数器)
|
|
1642
|
+
const textOver = run2({ maxStepTextChars: 2 })
|
|
1643
|
+
assert.equal(textOver.stats.skippedText, 1, 'text 超限必须记到 text 轴')
|
|
1644
|
+
assert.equal(textOver.stats.skippedReasoning, 0, 'text 轴拦截不得污染 reasoning 轴计数')
|
|
1645
|
+
|
|
1646
|
+
// ④ 两轴阈值互不影响:只调 text 阈值不得改变 reasoning 的判定结果
|
|
1647
|
+
const textLoose = run2({ maxStepTextChars: 100000 })
|
|
1648
|
+
assert.equal(textLoose.ranges.length, 1, '放宽 text 阈值不应影响 reasoning 轴的通过性')
|
|
1649
|
+
}
|
|
1650
|
+
|
|
1651
|
+
// ---------------------------------------------------------------- 配置兜底完整性(外部审查)
|
|
1652
|
+
|
|
1653
|
+
{
|
|
1654
|
+
// 维护约定(见 index.js resolveConfig 注释):Config schema 的每个 default 都必须
|
|
1655
|
+
// 在 resolveConfig 里有对应兜底。这条约定此前只写在注释里、没有测试固化,
|
|
1656
|
+
// 结果 baseUrl 悄悄漏掉。这里把它变成断言。
|
|
1657
|
+
const schemaKeys = Object.keys(Config?.dict ?? {})
|
|
1658
|
+
assert.ok(schemaKeys.length > 0, '应能枚举出 Config schema 的键')
|
|
1659
|
+
|
|
1660
|
+
const resolved = resolveConfig({})
|
|
1661
|
+
const missing = schemaKeys.filter((key) => !(key in resolved))
|
|
1662
|
+
assert.deepEqual(missing, [], `resolveConfig 缺少这些键的兜底:${missing.join(', ')}`)
|
|
1663
|
+
|
|
1664
|
+
// 未经 schemastery 归一化时,所有布尔/数组/数值键都必须有确定值(不能是 undefined)
|
|
1665
|
+
const empty = resolveConfig({})
|
|
1666
|
+
for (const key of schemaKeys) {
|
|
1667
|
+
assert.notEqual(empty[key], undefined, `${key} 在未归一化配置下不得为 undefined`)
|
|
1668
|
+
}
|
|
1669
|
+
}
|
|
1670
|
+
|
|
1671
|
+
// ---------------------------------------------------------------- 越界配置钳制(issue #28 回归)
|
|
1672
|
+
|
|
1673
|
+
{
|
|
1674
|
+
// 修复前:schema 只有 .default()、resolveConfig 只有 `?? 兜底`(只挡 undefined/null),
|
|
1675
|
+
// 所以任何越界值原样穿透到运行时,且后果是**静默失效**而非报错。这个块把每条后果钉死。
|
|
1676
|
+
//
|
|
1677
|
+
// ⚠️ 这里有**两层独立防线**,测试必须分别打,否则会误判"修好了":
|
|
1678
|
+
// ① `Config(raw)`(schemastery)→ 越界**抛 ValidationError**,是"响亮的拒绝"
|
|
1679
|
+
// ② `resolveConfig(raw)`(我们自己的钳制)→ 越界**回落到默认值 + 告警**,是"静默的修正"
|
|
1680
|
+
// 真实的"未归一化"路径(cordis.patch.yml 直接注入配置对象、冒烟测试的 PLUGIN_CFG)
|
|
1681
|
+
// 只经过 ②,所以 ② 必须自己站得住,不能依赖 ①。
|
|
1682
|
+
|
|
1683
|
+
// ① schemastery 层必须响亮地拒绝(不是静默 clamp)
|
|
1684
|
+
assert.throws(() => Config({ preserveRecent: -5 }), /expected number >= 0/,
|
|
1685
|
+
'Config 应对越界值抛错,而不是悄悄改掉')
|
|
1686
|
+
assert.throws(() => Config({ compactPreserveRecent: -1 }), /expected number >= 0/)
|
|
1687
|
+
assert.throws(() => Config({ receiptMaxRatio: 5 }), /expected number <= 1/)
|
|
1688
|
+
|
|
1689
|
+
// ② 我们的钳制层:越界 → 回落到默认值(而不是钳到边界)
|
|
1690
|
+
const neg = resolveConfig({ preserveRecent: -5 })
|
|
1691
|
+
assert.equal(neg.preserveRecent, 4, 'preserveRecent=-5 必须回落到默认 4(负数会让最近区保护反向放大)')
|
|
1692
|
+
assert.ok(neg[CONFIG_WARNINGS].some((w) => /preserveRecent/.test(w)), '钳制必须留下告警')
|
|
1693
|
+
assert.ok(/低于下限/.test(neg[CONFIG_WARNINGS][0]), `告警应说明原因:${neg[CONFIG_WARNINGS][0]}`)
|
|
1694
|
+
assert.ok(/已改为 4/.test(neg[CONFIG_WARNINGS][0]), '告警应同时给出改后的值')
|
|
1695
|
+
assert.equal(resolveConfig({ compactPreserveRecent: -1 }).compactPreserveRecent, 1,
|
|
1696
|
+
'第二层独立最近区的非法值应回落默认 1')
|
|
1697
|
+
|
|
1698
|
+
// ②b 第二层永久静默失效:maxStepTextChars=-1 会让每一步都 text > -1
|
|
1699
|
+
assert.equal(resolveConfig({ maxStepTextChars: -1 }).maxStepTextChars, 1200)
|
|
1700
|
+
assert.equal(resolveConfig({ receiptTextChars: -1 }).receiptTextChars, 400,
|
|
1701
|
+
'receiptTextChars 越界时应回落默认 400')
|
|
1702
|
+
|
|
1703
|
+
// ②c 经济性门失效
|
|
1704
|
+
assert.equal(resolveConfig({ compactMinChars: -100 }).compactMinChars, 2000)
|
|
1705
|
+
|
|
1706
|
+
// ②d receiptMaxRatio 同时是"回执不得比原文大"的安全门 → 上界收在 1
|
|
1707
|
+
const high = resolveConfig({ receiptMaxRatio: 5 })
|
|
1708
|
+
assert.equal(high.receiptMaxRatio, 0.5, 'receiptMaxRatio > 1 等于允许"压缩后更占地方"')
|
|
1709
|
+
assert.ok(high[CONFIG_WARNINGS].some((w) => /高于上限/.test(w)), '超上限也要告警')
|
|
1710
|
+
|
|
1711
|
+
// ②e 概率类越界
|
|
1712
|
+
assert.equal(resolveConfig({ keepThreshold: 2 }).keepThreshold, 0.5)
|
|
1713
|
+
assert.equal(resolveConfig({ floorThreshold: 7 }).floorThreshold, DEFAULT_FLOOR_THRESHOLD)
|
|
1714
|
+
|
|
1715
|
+
// ②f 计数类下界不能是 0(配 0 等于把功能关掉,那是布尔开关的职责)
|
|
1716
|
+
assert.equal(resolveConfig({ minHistoryLines: 0 }).minHistoryLines, 8)
|
|
1717
|
+
assert.equal(resolveConfig({ maxCompactionsPerPass: 0 }).maxCompactionsPerPass, 3,
|
|
1718
|
+
'越界的配额回落默认值 3(issue #35 把默认从 1 提到 3)')
|
|
1719
|
+
assert.equal(resolveConfig({ minCandidatesForRelative: 1 }).minCandidatesForRelative,
|
|
1720
|
+
DEFAULT_MIN_CANDIDATES_FOR_RELATIVE, '相对分位至少要 2 条才谈得上"排序"')
|
|
1721
|
+
|
|
1722
|
+
// ① 非数值一律回落到默认值("改了多少"不可解释,"打到默认"才可解释)
|
|
1723
|
+
for (const bad of [NaN, Infinity, -Infinity, '600', {}, []]) {
|
|
1724
|
+
const got = clampConfigNumber('headChars', bad, 600, null)[0]
|
|
1725
|
+
assert.equal(got, 600, `headChars=${String(bad)} 应回落到默认值 600`)
|
|
1726
|
+
}
|
|
1727
|
+
|
|
1728
|
+
// ① 未配置(undefined/null/'')**不是**非法,必须静默用默认值、不产生告警
|
|
1729
|
+
for (const unset of [undefined, null, '']) {
|
|
1730
|
+
const [v, clamped] = clampConfigNumber('headChars', unset, 600, null)
|
|
1731
|
+
assert.equal(v, 600)
|
|
1732
|
+
assert.equal(clamped, false, `${String(unset)} 是"没配"而不是"配错",不该告警`)
|
|
1733
|
+
}
|
|
1734
|
+
assert.equal(resolveConfig({})[CONFIG_WARNINGS].length, 0, '全部合法的配置不得产生告警')
|
|
1735
|
+
|
|
1736
|
+
// keepMode 白名单(review 修复):拼错的模式必须回落 budget 并留痕,不得静默穿过
|
|
1737
|
+
{
|
|
1738
|
+
const bad = resolveConfig({ keepMode: 'budgt' })
|
|
1739
|
+
assert.equal(bad.keepMode, 'budget', '拼错的 keepMode 应回落 budget')
|
|
1740
|
+
assert.ok(bad[CONFIG_WARNINGS].some((w) => w.includes('keepMode')), 'keepMode 回落必须留下 configWarnings')
|
|
1741
|
+
const good = resolveConfig({ keepMode: 'absolute' })
|
|
1742
|
+
assert.equal(good.keepMode, 'absolute')
|
|
1743
|
+
assert.equal(good[CONFIG_WARNINGS].length, 0, '合法 keepMode 不告警')
|
|
1744
|
+
}
|
|
1745
|
+
|
|
1746
|
+
// 废弃键的运行时信号(review 反馈):显式配置 volumeBudgetThresholdChars / budgetMinChars
|
|
1747
|
+
// 必须推 configWarnings,否则用户配了却静默空转(与 keepMode 的留痕约定一致)。
|
|
1748
|
+
//
|
|
1749
|
+
// ⚠️ 必须走**宿主的真实取配置路径**:cordis 会先用 Config schema 校验用户配置、把默认值
|
|
1750
|
+
// 填进去,再把结果交给 apply()(`resolveConfig(runtime, config).value`)。直接调
|
|
1751
|
+
// `resolveConfig({...})` 传的是**没有默认值**的裸对象,测不到那个差异 ——
|
|
1752
|
+
// 第一版断言就是这么写的,于是"空配置也被报废弃"这个回归它完全看不见。
|
|
1753
|
+
const viaHost = (userCfg) => {
|
|
1754
|
+
const r = Config['~standard'].validate(userCfg)
|
|
1755
|
+
assert.ok(!r.issues, `schema 不应拒绝 ${JSON.stringify(userCfg)}`)
|
|
1756
|
+
return resolveConfig(r.value)
|
|
1757
|
+
}
|
|
1758
|
+
{
|
|
1759
|
+
// ① 关键回归:用户什么都没配 → 不得告警(schema 填的默认值不是"用户配置")
|
|
1760
|
+
const none = viaHost({})
|
|
1761
|
+
assert.equal(none[CONFIG_WARNINGS].length, 0,
|
|
1762
|
+
`空配置不得报废弃键(宿主已填默认值,实测会误报):${JSON.stringify(none[CONFIG_WARNINGS])}`)
|
|
1763
|
+
// ② 配了无关的键 → 同样不得告警
|
|
1764
|
+
const unrelated = viaHost({ dryRun: true })
|
|
1765
|
+
assert.equal(unrelated[CONFIG_WARNINGS].length, 0, '只配无关键不得报废弃键')
|
|
1766
|
+
// ③ 真的改了废弃键的值 → 必须留痕
|
|
1767
|
+
const changed1 = viaHost({ volumeBudgetThresholdChars: 5000 })
|
|
1768
|
+
assert.ok(changed1[CONFIG_WARNINGS].some((w) => w.includes('volumeBudgetThresholdChars')),
|
|
1769
|
+
'改了 volumeBudgetThresholdChars 必须留痕')
|
|
1770
|
+
const changed2 = viaHost({ budgetMinChars: 100 })
|
|
1771
|
+
assert.ok(changed2[CONFIG_WARNINGS].some((w) => w.includes('budgetMinChars')),
|
|
1772
|
+
'改了 budgetMinChars 必须留痕')
|
|
1773
|
+
// ④ 显式写成默认值 = 空操作,不告警(代价可接受,注释里写明)
|
|
1774
|
+
const asDefault = viaHost({ volumeBudgetThresholdChars: 8192, budgetMinChars: 0 })
|
|
1775
|
+
assert.equal(asDefault[CONFIG_WARNINGS].length, 0, '显式写成默认值属空操作,不告警')
|
|
1776
|
+
}
|
|
1777
|
+
|
|
1778
|
+
|
|
1779
|
+
// 合法值必须原样保留(钳制不能顺手改掉正常配置)
|
|
1780
|
+
const ok = resolveConfig({ preserveRecent: 0, compactPreserveRecent: 0, headChars: 0, maxStepTextChars: 5000, receiptTextChars: 0, receiptMaxRatio: 1 })
|
|
1781
|
+
assert.equal(ok.preserveRecent, 0, '0 是合法值(不保护最近区),不得被当成缺省')
|
|
1782
|
+
assert.equal(ok.compactPreserveRecent, 0, '第二层最近区也允许显式设为 0')
|
|
1783
|
+
assert.equal(ok.headChars, 0)
|
|
1784
|
+
assert.equal(ok.maxStepTextChars, 5000)
|
|
1785
|
+
assert.equal(ok.receiptTextChars, 0, '0 表示关闭回执中的 assistant 原文摘录,不得被当成缺省')
|
|
1786
|
+
assert.equal(ok.receiptMaxRatio, 1, '1 是上界本身,闭区间内')
|
|
1787
|
+
assert.equal(ok[CONFIG_WARNINGS].length, 0)
|
|
1788
|
+
|
|
1789
|
+
// 区间表必须覆盖**每一个**数值型 schema 键,否则新加的键会悄悄不受保护
|
|
1790
|
+
const numericKeys = Object.keys(Config?.dict ?? {}).filter((k) => Config.dict[k]?.type === 'number')
|
|
1791
|
+
assert.ok(numericKeys.length >= 20, `应能枚举出数值键(实际 ${numericKeys.length} 个)`)
|
|
1792
|
+
const uncovered = numericKeys.filter((k) => clampConfigNumber(k, 1e12, 1, null)[0] === 1e12)
|
|
1793
|
+
assert.deepEqual(uncovered, [], `这些数值键尚未纳入钳制区间表:${uncovered.join(', ')}`)
|
|
1794
|
+
|
|
1795
|
+
// schema 的 .min() 与钳制表必须同向(防止两处各写一个数字后漂移):
|
|
1796
|
+
// 比 schema 下界更小的值,在钳制层也必须被判为越界
|
|
1797
|
+
for (const key of numericKeys) {
|
|
1798
|
+
const min = Config.dict[key]?.meta?.min
|
|
1799
|
+
if (typeof min !== 'number') continue
|
|
1800
|
+
assert.equal(clampConfigNumber(key, min - 1, 0, null)[1], true,
|
|
1801
|
+
`${key}:低于 schema 下界 ${min} 的值在钳制层也必须被判为越界`)
|
|
1802
|
+
}
|
|
1803
|
+
}
|
|
1804
|
+
|
|
1805
|
+
// ------------------------------------------- Jev 客户端:可重试失败与退避(issue #34)
|
|
1806
|
+
// 旧实现单次失败就丢掉整轮判定:一次网络抖动 → 本次 pass 全部候选没有概率 →
|
|
1807
|
+
// 两层静默不动。这里把"该重试的重试、不该重试的立刻放弃"两件事都钉住。
|
|
1808
|
+
{
|
|
1809
|
+
const okBody = { answers: { q1: { noul: 0.12 } }, usage: { input_tokens: 10, output_tokens: 2 } }
|
|
1810
|
+
const okResponse = () => ({ ok: true, status: 200, text: async () => JSON.stringify(okBody) })
|
|
1811
|
+
|
|
1812
|
+
// 1) 网络层异常(fetch 抛错)→ 重试;第二次成功
|
|
1813
|
+
{
|
|
1814
|
+
let calls = 0
|
|
1815
|
+
const client = new JevClient({
|
|
1816
|
+
apiKey: 'k',
|
|
1817
|
+
maxRetries: 2,
|
|
1818
|
+
retryBaseMs: 0,
|
|
1819
|
+
fetchImpl: async () => {
|
|
1820
|
+
calls += 1
|
|
1821
|
+
if (calls === 1) throw new Error('ECONNRESET')
|
|
1822
|
+
return okResponse()
|
|
1823
|
+
},
|
|
1824
|
+
})
|
|
1825
|
+
const out = await client.ask('state', { q1: 'x' })
|
|
1826
|
+
assert.equal(out.q1, 0.12, '重试后应拿到概率')
|
|
1827
|
+
assert.equal(calls, 2, '网络异常应重试一次')
|
|
1828
|
+
assert.equal(client.retries, 1)
|
|
1829
|
+
assert.equal(client.lastRetries, 1, '本次 ask 重试了 1 次')
|
|
1830
|
+
assert.equal(client.requests, 2, 'requests 计入失败尝试:2 次尝试都真的发出去了')
|
|
1831
|
+
// 口径一致性(PR #28 review):requests = 成功 ask 数 + 重试数
|
|
1832
|
+
assert.equal(client.requests, 1 + client.retries, 'requests 必须等于成功次数 + 重试次数')
|
|
1833
|
+
|
|
1834
|
+
// 关键回归(PR #28 review):累计量不得被当成"本次"用。
|
|
1835
|
+
// 上一次 ask 重试过,这一次完全顺利 → lastRetries 必须归零,
|
|
1836
|
+
// 否则状态报告会从此永久显示"重试 N 次"。
|
|
1837
|
+
client.fetchImpl = okResponse
|
|
1838
|
+
await client.ask('state', { q1: 'x' })
|
|
1839
|
+
assert.equal(client.lastRetries, 0, '新的 ask 顺利时应报 lastRetries=0')
|
|
1840
|
+
assert.equal(client.retries, 1, '累计量保留历史(供"这个 client 重试过没有"判断)')
|
|
1841
|
+
}
|
|
1842
|
+
|
|
1843
|
+
// 2) 5xx → 重试;429 → 重试
|
|
1844
|
+
for (const status of [500, 503, 429]) {
|
|
1845
|
+
let calls = 0
|
|
1846
|
+
const client = new JevClient({
|
|
1847
|
+
apiKey: 'k',
|
|
1848
|
+
maxRetries: 2,
|
|
1849
|
+
retryBaseMs: 0,
|
|
1850
|
+
fetchImpl: async () => {
|
|
1851
|
+
calls += 1
|
|
1852
|
+
if (calls === 1) return { ok: false, status, text: async () => 'boom' }
|
|
1853
|
+
return okResponse()
|
|
1854
|
+
},
|
|
1855
|
+
})
|
|
1856
|
+
const out = await client.ask('state', { q1: 'x' })
|
|
1857
|
+
assert.equal(out.q1, 0.12, `${status} 之后应重试成功`)
|
|
1858
|
+
assert.equal(calls, 2)
|
|
1859
|
+
}
|
|
1860
|
+
|
|
1861
|
+
// 3) 4xx(密钥错 / 请求本身有问题)→ **不重试**,立刻抛
|
|
1862
|
+
for (const status of [400, 401, 403, 404]) {
|
|
1863
|
+
let calls = 0
|
|
1864
|
+
const client = new JevClient({
|
|
1865
|
+
apiKey: 'k',
|
|
1866
|
+
maxRetries: 3,
|
|
1867
|
+
retryBaseMs: 0,
|
|
1868
|
+
fetchImpl: async () => {
|
|
1869
|
+
calls += 1
|
|
1870
|
+
return { ok: false, status, text: async () => 'nope' }
|
|
1871
|
+
},
|
|
1872
|
+
})
|
|
1873
|
+
await assert.rejects(() => client.ask('state', { q1: 'x' }), JevError)
|
|
1874
|
+
assert.equal(calls, 1, `${status} 不该重试(实际发了 ${calls} 次)`)
|
|
1875
|
+
assert.equal(client.retries, 0)
|
|
1876
|
+
assert.equal(client.lastRetries, 0)
|
|
1877
|
+
assert.equal(client.requests, 1, '不可重试的失败也真的发过一次请求')
|
|
1878
|
+
}
|
|
1879
|
+
|
|
1880
|
+
// 4) 响应缺 answers → 不重试(换一次大概率还是坏响应)
|
|
1881
|
+
{
|
|
1882
|
+
let calls = 0
|
|
1883
|
+
const client = new JevClient({
|
|
1884
|
+
apiKey: 'k',
|
|
1885
|
+
maxRetries: 3,
|
|
1886
|
+
retryBaseMs: 0,
|
|
1887
|
+
fetchImpl: async () => {
|
|
1888
|
+
calls += 1
|
|
1889
|
+
return { ok: true, status: 200, text: async () => JSON.stringify({ usage: {} }) }
|
|
1890
|
+
},
|
|
1891
|
+
})
|
|
1892
|
+
await assert.rejects(() => client.ask('state', { q1: 'x' }), /缺少 answers/)
|
|
1893
|
+
assert.equal(calls, 1, '形状错误不该重试')
|
|
1894
|
+
}
|
|
1895
|
+
|
|
1896
|
+
// 5) 一直失败 → 尝试次数 = maxRetries + 1(不多不少),并记录 lastError
|
|
1897
|
+
{
|
|
1898
|
+
let calls = 0
|
|
1899
|
+
const client = new JevClient({
|
|
1900
|
+
apiKey: 'k',
|
|
1901
|
+
maxRetries: 2,
|
|
1902
|
+
retryBaseMs: 0,
|
|
1903
|
+
fetchImpl: async () => {
|
|
1904
|
+
calls += 1
|
|
1905
|
+
return { ok: false, status: 503, text: async () => 'down' }
|
|
1906
|
+
},
|
|
1907
|
+
})
|
|
1908
|
+
await assert.rejects(() => client.ask('state', { q1: 'x' }), JevError)
|
|
1909
|
+
assert.equal(calls, 3, 'maxRetries=2 表示最多 3 次尝试')
|
|
1910
|
+
assert.equal(client.retries, 2)
|
|
1911
|
+
assert.equal(client.lastRetries, 2)
|
|
1912
|
+
assert.equal(client.requests, 3, '全部失败的 3 次尝试都要计入 requests')
|
|
1913
|
+
assert.equal(client.requests, client.retries + 1, '口径:尝试数 = 重试数 + 1')
|
|
1914
|
+
assert.ok(client.lastError.length > 0, 'lastError 要留下原因')
|
|
1915
|
+
}
|
|
1916
|
+
|
|
1917
|
+
// 6) 外部 signal 已 abort → 一次都不发(用户中断不该触发重试)
|
|
1918
|
+
{
|
|
1919
|
+
let calls = 0
|
|
1920
|
+
const client = new JevClient({
|
|
1921
|
+
apiKey: 'k',
|
|
1922
|
+
maxRetries: 3,
|
|
1923
|
+
retryBaseMs: 0,
|
|
1924
|
+
fetchImpl: async () => {
|
|
1925
|
+
calls += 1
|
|
1926
|
+
throw new Error('should not be called')
|
|
1927
|
+
},
|
|
1928
|
+
})
|
|
1929
|
+
const ac = new AbortController()
|
|
1930
|
+
ac.abort()
|
|
1931
|
+
await assert.rejects(() => client.ask('state', { q1: 'x' }, { signal: ac.signal }), /中断/)
|
|
1932
|
+
assert.equal(calls, 0, 'abort 后不应发起请求')
|
|
1933
|
+
}
|
|
1934
|
+
|
|
1935
|
+
// 7) 请求中途被中断 → 不重试
|
|
1936
|
+
{
|
|
1937
|
+
let calls = 0
|
|
1938
|
+
const client = new JevClient({
|
|
1939
|
+
apiKey: 'k',
|
|
1940
|
+
maxRetries: 3,
|
|
1941
|
+
retryBaseMs: 0,
|
|
1942
|
+
fetchImpl: async () => {
|
|
1943
|
+
calls += 1
|
|
1944
|
+
throw new Error('aborted by signal')
|
|
1945
|
+
},
|
|
1946
|
+
})
|
|
1947
|
+
const ac = new AbortController()
|
|
1948
|
+
const promise = client.ask('state', { q1: 'x' }, { signal: ac.signal })
|
|
1949
|
+
ac.abort()
|
|
1950
|
+
await assert.rejects(() => promise, /中断/)
|
|
1951
|
+
assert.equal(calls, 1, '被中断时只发一次,不继续重试')
|
|
1952
|
+
}
|
|
1953
|
+
|
|
1954
|
+
// 8) maxRetries=0 时退化为旧行为(不重试),确保配置可关
|
|
1955
|
+
{
|
|
1956
|
+
let calls = 0
|
|
1957
|
+
const client = new JevClient({
|
|
1958
|
+
apiKey: 'k',
|
|
1959
|
+
maxRetries: 0,
|
|
1960
|
+
retryBaseMs: 0,
|
|
1961
|
+
fetchImpl: async () => {
|
|
1962
|
+
calls += 1
|
|
1963
|
+
throw new Error('boom')
|
|
1964
|
+
},
|
|
1965
|
+
})
|
|
1966
|
+
await assert.rejects(() => client.ask('state', { q1: 'x' }), JevError)
|
|
1967
|
+
assert.equal(calls, 1, 'maxRetries=0 应退化为单次尝试')
|
|
1968
|
+
}
|
|
1969
|
+
}
|
|
1970
|
+
|
|
1971
|
+
console.log('ok —— 纯函数自检全部通过')
|