dsh-context-compression-improved 0.4.0-beta.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.ja.md +68 -36
- package/CHANGELOG.ko.md +67 -35
- package/CHANGELOG.md +195 -134
- package/CHANGELOG.zh.md +64 -36
- package/README.ja.md +1 -1
- package/README.ko.md +1 -1
- package/README.md +1 -1
- package/README.zh.md +1 -1
- package/docs/installation.ja.md +2 -2
- package/docs/installation.ko.md +2 -2
- package/docs/installation.md +103 -78
- package/docs/installation.zh.md +100 -77
- package/docs/repair-log.md +54 -0
- package/package.json +1 -1
- package/packages/selector/lib/{config.js → advisor-state.js} +329 -5
- package/packages/selector/lib/client.d.ts +7 -0
- package/packages/selector/lib/client.js +33 -3
- package/packages/selector/lib/index.d.ts +7 -0
- package/packages/selector/lib/index.js +112 -3
- package/packages/selector/lib/pruner.d.ts +128 -1
- package/packages/selector/lib/pruner.js +2802 -1374
- package/packages/selector/src/client/ReviewOverlay.tsx +1 -1
- package/packages/selector/src/client/index.ts +1 -1
- package/packages/selector/src/client/preset-options.ts +2 -0
- package/packages/selector/src/index.ts +129 -49
- package/packages/selector/src/profiles.ts +48 -0
- package/packages/selector/src/pruner/content.ts +18 -5
- package/packages/selector/src/pruner/state.ts +3 -0
- package/packages/selector/src/pruner/types.ts +23 -5
- package/packages/selector/src/pruner.ts +297 -162
- package/packages/selector/src/runtime/adaptive-cost.ts +23 -12
- package/packages/selector/src/runtime/audit.ts +40 -2
- package/packages/selector/src/runtime/config.ts +88 -1
- package/packages/selector/src/runtime/measurement.ts +31 -2
- package/packages/selector/src/runtime/reducers.ts +1115 -97
- package/packages/selector/src/runtime/tokenpilot/advisor-prompt.ts +188 -0
- package/packages/selector/src/runtime/tokenpilot/advisor-state.ts +133 -0
- package/packages/selector/src/runtime/tokenpilot/advisor.ts +419 -0
- package/packages/selector/src/runtime/tokenpilot/dedup.ts +1 -1
- package/packages/selector/src/runtime/tokenpilot/estimator.ts +8 -118
- package/packages/selector/src/runtime/tokenpilot/locator.ts +1 -1
- package/packages/selector/src/runtime/tokenpilot/proposal.ts +76 -32
- package/packages/selector/src/runtime/tokenpilot/read-state.ts +23 -2
- package/packages/selector/src/runtime/tokenpilot/review-registry.ts +117 -0
- package/packages/selector/src/runtime/tokenpilot/sidechannel.ts +303 -0
- package/packages/selector/src/runtime/toolclass.ts +103 -0
- package/packages/selector/src/runtime/types.ts +37 -0
- package/packages/selector/tests/advisor-report.host.spec.ts +223 -0
- package/packages/selector/tests/public/package-contract.client.spec.ts +2 -1
- package/packages/selector/tests/review-routes-registry.host.spec.ts +142 -0
- package/packages/selector/tests/runtime/adaptive-cost.spec.ts +7 -7
- package/packages/selector/tests/runtime/advisor-invariant.spec.ts +272 -0
- package/packages/selector/tests/runtime/advisor.spec.ts +226 -0
- package/packages/selector/tests/runtime/audit.spec.ts +88 -1
- package/packages/selector/tests/runtime/char-basis.spec.ts +30 -0
- package/packages/selector/tests/runtime/code-skeleton.spec.ts +14 -3
- package/packages/selector/tests/runtime/frequency-longstrings.spec.ts +74 -0
- package/packages/selector/tests/runtime/html-reducer.spec.ts +212 -0
- package/packages/selector/tests/runtime/line-mapping.spec.ts +153 -0
- package/packages/selector/tests/runtime/prose-reducers.spec.ts +133 -0
- package/packages/selector/tests/runtime/public/public-runtime.spec.ts +198 -27
- package/packages/selector/tests/runtime/read-input-cap.spec.ts +33 -0
- package/packages/selector/tests/runtime/search-reducer.spec.ts +110 -0
- package/packages/selector/tests/runtime/sidechannel.spec.ts +241 -0
- package/packages/selector/tests/runtime/toc-and-bundled.spec.ts +159 -0
- package/packages/selector/tests/runtime/tokenpilot/profile-baseline.spec.ts +12 -0
- package/packages/selector/tests/runtime/tokenpilot/proposal.spec.ts +194 -0
- package/packages/selector/tests/runtime/tokenpilot/pruner-review.spec.ts +70 -1
- package/packages/selector/tests/runtime/tokenpilot/read-state.spec.ts +24 -0
- package/packages/selector/tests/runtime/toolclass.spec.ts +156 -0
- package/scripts/toolclass-corpus-replay.mjs +281 -0
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Advisor integration and the advisory-only invariant.
|
|
3
|
+
*
|
|
4
|
+
* The invariant this spec exists to pin (K13): the advisor is statistics and
|
|
5
|
+
* suggestions only. Whatever its channel returns — extreme scores, a hostile
|
|
6
|
+
* summary, garbage, or nothing at all — `pruneSession` must land exactly the
|
|
7
|
+
* same reductions it lands with the advisor off and the state empty.
|
|
8
|
+
*/
|
|
9
|
+
import { describe, expect, it } from 'vitest'
|
|
10
|
+
import { Context } from '@deepseek-ai/cordis'
|
|
11
|
+
import {
|
|
12
|
+
ToolCallId as CallId,
|
|
13
|
+
createMessage,
|
|
14
|
+
createUserMessage,
|
|
15
|
+
createToolResultMessage,
|
|
16
|
+
} from '@deepseek-ai/dsh-llm'
|
|
17
|
+
import { canonicalHeader, Session, SessionId } from '@deepseek-ai/dsh-session'
|
|
18
|
+
import type { SessionEvent } from '@deepseek-ai/dsh-session'
|
|
19
|
+
import SessionProjectionRegistry from '@deepseek-ai/dsh-session-projection'
|
|
20
|
+
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
|
|
21
|
+
import TokenMeter from '@deepseek-ai/dsh-token-meter'
|
|
22
|
+
import ToolRuntime from '@deepseek-ai/dsh-tools'
|
|
23
|
+
import SessionStore from '@deepseek-ai/dsh-session'
|
|
24
|
+
import ToolResultPruner from '../../src/pruner.ts'
|
|
25
|
+
import {
|
|
26
|
+
collectTaskSemantics,
|
|
27
|
+
runSessionAdvisorPass,
|
|
28
|
+
type AdvisorCandidate,
|
|
29
|
+
} from '../../src/runtime/tokenpilot/advisor.ts'
|
|
30
|
+
import { getAdvisorState, recordScore } from '../../src/runtime/tokenpilot/advisor-state.ts'
|
|
31
|
+
import type { AdvisorOutcomeAuditRecord } from '../../src/runtime/audit.ts'
|
|
32
|
+
import type { AdvisorChannel } from '../../src/runtime/tokenpilot/advisor.ts'
|
|
33
|
+
|
|
34
|
+
// ── Orchestration against a scripted channel ───────────────────────────────
|
|
35
|
+
|
|
36
|
+
const CANDIDATES: AdvisorCandidate[] = [
|
|
37
|
+
{ seq: 3, characterPressure: 4_000, preview: 'auth module login flow' },
|
|
38
|
+
{ seq: 5, characterPressure: 6_000, preview: 'database migration notes' },
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
const TASK = {
|
|
42
|
+
source: 'todos' as const,
|
|
43
|
+
todoVersion: 'aaaa1111',
|
|
44
|
+
taskText: 'migrate the auth module',
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
function advisorInput(turn: number, signal: AbortSignal): Parameters<typeof runSessionAdvisorPass>[3] {
|
|
48
|
+
return {
|
|
49
|
+
profile: 'tokenpilot-inspired',
|
|
50
|
+
turn,
|
|
51
|
+
candidates: CANDIDATES,
|
|
52
|
+
task: TASK,
|
|
53
|
+
advisor: { refreshTurns: 8, scoreThreshold: 0.35, sampleLimit: 16, minTokens: 250 },
|
|
54
|
+
tailText: 'working on the migration',
|
|
55
|
+
signal,
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function scriptedChannel(responses: (string | undefined)[]): { channel: AdvisorChannel, calls: { system: string, user: string }[] } {
|
|
60
|
+
const calls: { system: string, user: string }[] = []
|
|
61
|
+
return {
|
|
62
|
+
calls,
|
|
63
|
+
channel: {
|
|
64
|
+
identity: () => 'host:mock/model',
|
|
65
|
+
ask: async request => {
|
|
66
|
+
calls.push({ system: request.system, user: request.user })
|
|
67
|
+
return responses.shift()
|
|
68
|
+
},
|
|
69
|
+
},
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
describe('runSessionAdvisorPass orchestration (K11)', () => {
|
|
74
|
+
it('writes summary, scores, recertification marks, and emits three audits on success', async () => {
|
|
75
|
+
const session = Session.create(SessionId('advisor-ok'))
|
|
76
|
+
const { channel, calls } = scriptedChannel([
|
|
77
|
+
'{"overallTask":"migrate the auth module","activeSubtasks":["port login flow"],"keywords":["auth","login"]}',
|
|
78
|
+
'{"seq":3,"score":0.95,"reason":"current task"}\n{"seq":5,"score":0.10,"reason":"stale"}',
|
|
79
|
+
])
|
|
80
|
+
const audits: AdvisorOutcomeAuditRecord[] = []
|
|
81
|
+
const outcome = await runSessionAdvisorPass(session, channel, record => audits.push(record), advisorInput(4, new AbortController().signal))
|
|
82
|
+
|
|
83
|
+
expect(calls).toHaveLength(2)
|
|
84
|
+
expect(outcome).toBeDefined()
|
|
85
|
+
const state = getAdvisorState(session)
|
|
86
|
+
expect(state.summary?.overallTask).toBe('migrate the auth module')
|
|
87
|
+
expect(state.scores.get(3)?.score).toBe(0.95)
|
|
88
|
+
expect(state.scores.get(5)?.score).toBe(0.10)
|
|
89
|
+
// Below-threshold segment recertified as a suggestion only.
|
|
90
|
+
expect(state.recertified.get(5)).toBe(4)
|
|
91
|
+
expect(state.recertified.has(3)).toBe(false)
|
|
92
|
+
expect(state.watermarkSeq).toBe(5)
|
|
93
|
+
expect(audits.map(record => record.phase)).toEqual(['summary', 'scoring', 'decay'])
|
|
94
|
+
expect(audits.every(record => record.ok)).toBe(true)
|
|
95
|
+
expect(audits.find(record => record.phase === 'decay')?.decay).toBeDefined()
|
|
96
|
+
})
|
|
97
|
+
|
|
98
|
+
it('leaves state untouched and audits ok:false when the channel answers nothing', async () => {
|
|
99
|
+
const session = Session.create(SessionId('advisor-fail'))
|
|
100
|
+
const { channel, calls } = scriptedChannel([undefined, undefined])
|
|
101
|
+
const audits: AdvisorOutcomeAuditRecord[] = []
|
|
102
|
+
const outcome = await runSessionAdvisorPass(session, channel, record => audits.push(record), advisorInput(2, new AbortController().signal))
|
|
103
|
+
|
|
104
|
+
expect(outcome).toBeUndefined()
|
|
105
|
+
expect(calls).toHaveLength(1) // summary failed; scoring never ran
|
|
106
|
+
const state = getAdvisorState(session)
|
|
107
|
+
expect(state.summary).toBeUndefined()
|
|
108
|
+
expect(state.scores.size).toBe(0)
|
|
109
|
+
expect(audits).toHaveLength(1)
|
|
110
|
+
expect(audits[0]?.ok).toBe(false)
|
|
111
|
+
expect(audits[0]?.reason).toBe('channel-empty')
|
|
112
|
+
expect(state.inFlight).toBe(false)
|
|
113
|
+
})
|
|
114
|
+
|
|
115
|
+
it('skips entirely while a pass is in flight (re-entry guard)', async () => {
|
|
116
|
+
const session = Session.create(SessionId('advisor-reentry'))
|
|
117
|
+
const { channel, calls } = scriptedChannel([])
|
|
118
|
+
const state = getAdvisorState(session)
|
|
119
|
+
state.inFlight = true
|
|
120
|
+
const outcome = await runSessionAdvisorPass(session, channel, () => undefined, advisorInput(1, new AbortController().signal))
|
|
121
|
+
expect(outcome).toBeUndefined()
|
|
122
|
+
expect(calls).toHaveLength(0)
|
|
123
|
+
state.inFlight = false
|
|
124
|
+
})
|
|
125
|
+
|
|
126
|
+
it('does not advance the watermark on failure, so candidates rescore later', async () => {
|
|
127
|
+
const session = Session.create(SessionId('advisor-watermark'))
|
|
128
|
+
const failing = scriptedChannel([undefined])
|
|
129
|
+
await runSessionAdvisorPass(session, failing.channel, () => undefined, advisorInput(1, new AbortController().signal))
|
|
130
|
+
expect(getAdvisorState(session).watermarkSeq).toBe(0)
|
|
131
|
+
|
|
132
|
+
const succeeding = scriptedChannel([
|
|
133
|
+
'{"overallTask":"t","activeSubtasks":[],"keywords":["auth"]}',
|
|
134
|
+
'{"seq":3,"score":0.9}',
|
|
135
|
+
])
|
|
136
|
+
const audits: AdvisorOutcomeAuditRecord[] = []
|
|
137
|
+
const outcome = await runSessionAdvisorPass(session, succeeding.channel, record => audits.push(record), advisorInput(2, new AbortController().signal))
|
|
138
|
+
expect(outcome).toBeDefined()
|
|
139
|
+
expect(getAdvisorState(session).watermarkSeq).toBe(3)
|
|
140
|
+
})
|
|
141
|
+
|
|
142
|
+
it('returns undefined without any channel call when the session has no task semantics', async () => {
|
|
143
|
+
const session = Session.create(SessionId('advisor-notask'))
|
|
144
|
+
const { channel, calls } = scriptedChannel(['{"overallTask":"x","keywords":["k"]}'])
|
|
145
|
+
const outcome = await runSessionAdvisorPass(session, channel, () => undefined, {
|
|
146
|
+
...advisorInput(1, new AbortController().signal),
|
|
147
|
+
task: undefined,
|
|
148
|
+
})
|
|
149
|
+
expect(outcome).toBeUndefined()
|
|
150
|
+
expect(calls).toHaveLength(0)
|
|
151
|
+
})
|
|
152
|
+
})
|
|
153
|
+
|
|
154
|
+
describe('collectTaskSemantics over real event shapes', () => {
|
|
155
|
+
it('reads a todo/write event appended alongside message events', () => {
|
|
156
|
+
const events = [
|
|
157
|
+
{
|
|
158
|
+
type: 'user/message',
|
|
159
|
+
seq: 1,
|
|
160
|
+
time: 0,
|
|
161
|
+
data: { content: [{ type: 'text', text: 'start the migration' }] },
|
|
162
|
+
},
|
|
163
|
+
{
|
|
164
|
+
type: 'todo/write',
|
|
165
|
+
seq: 2,
|
|
166
|
+
time: 0,
|
|
167
|
+
data: { todos: ['migrate the auth module', { content: 'write tests', status: 'pending' }] },
|
|
168
|
+
},
|
|
169
|
+
] as unknown as readonly SessionEvent[]
|
|
170
|
+
const task = collectTaskSemantics(events)
|
|
171
|
+
expect(task?.source).toBe('todos')
|
|
172
|
+
expect(task?.taskText).toContain('write tests')
|
|
173
|
+
})
|
|
174
|
+
})
|
|
175
|
+
|
|
176
|
+
// ── K13: the advisory-only invariant, against the real pruner ──────────────
|
|
177
|
+
|
|
178
|
+
describe('advisory-only invariant (K13): advisor outputs never change landings', () => {
|
|
179
|
+
async function prunedResult(session: Session): Promise<unknown> {
|
|
180
|
+
const ctx = new Context()
|
|
181
|
+
try {
|
|
182
|
+
await ctx.plugin(SessionStore).await()
|
|
183
|
+
await ctx.plugin(SystemPrompt).await()
|
|
184
|
+
await ctx.plugin(ToolRuntime).await()
|
|
185
|
+
await ctx.plugin(SessionProjectionRegistry).await()
|
|
186
|
+
await ctx.plugin(TokenMeter).await()
|
|
187
|
+
await ctx.plugin(ToolResultPruner, {
|
|
188
|
+
profile: 'native',
|
|
189
|
+
nativeTriggerTokens: 100,
|
|
190
|
+
nativeTargetTokens: 64,
|
|
191
|
+
headChars: 8,
|
|
192
|
+
tailChars: 8,
|
|
193
|
+
}).await()
|
|
194
|
+
return ctx.toolResultPruner.pruneSession(session, { stage: 'pressure' })
|
|
195
|
+
} finally {
|
|
196
|
+
await ctx.fiber.dispose()
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
function buildSession(id: string): Session {
|
|
201
|
+
const session = Session.create(SessionId(id))
|
|
202
|
+
const callId = CallId('call-1')
|
|
203
|
+
session.append('turn/start', { turn: 1 })
|
|
204
|
+
session.append('request/header', {
|
|
205
|
+
reason: 'initial',
|
|
206
|
+
header: canonicalHeader({ config: { provider: 'deepseek', model: 'deepseek-v4-flash' } }),
|
|
207
|
+
})
|
|
208
|
+
session.append('user/message', createUserMessage({
|
|
209
|
+
content: [{ type: 'text', text: 'please inspect the failing module' }],
|
|
210
|
+
source: { kind: 'user' },
|
|
211
|
+
}), { surfaceOp: 'append' })
|
|
212
|
+
session.append('step/start', { turn: 1, step: 1 })
|
|
213
|
+
session.append('assistant/message', {
|
|
214
|
+
stream: [],
|
|
215
|
+
turn: 1,
|
|
216
|
+
step: 1,
|
|
217
|
+
message: createMessage({
|
|
218
|
+
role: 'assistant',
|
|
219
|
+
content: [{ type: 'tool-call', id: callId, name: 'bash', arguments: '{}' }],
|
|
220
|
+
source: { kind: 'model', provider: 'deepseek', model: 'deepseek-v4-flash' },
|
|
221
|
+
}),
|
|
222
|
+
}, { surfaceOp: 'append' })
|
|
223
|
+
session.append('tool/call', { turn: 1, step: 1, callId, name: 'bash', arguments: '{}' })
|
|
224
|
+
session.append('tool/result', {
|
|
225
|
+
turn: 1,
|
|
226
|
+
step: 1,
|
|
227
|
+
message: createToolResultMessage({
|
|
228
|
+
callId,
|
|
229
|
+
content: [{ type: 'text', text: 'gate evidence '.repeat(800) }],
|
|
230
|
+
isError: false,
|
|
231
|
+
}),
|
|
232
|
+
}, { surfaceOp: 'append' })
|
|
233
|
+
session.append('step/end', { turn: 1, step: 1 })
|
|
234
|
+
// No turn/end: pruning lands its replacement inside the still-open turn.
|
|
235
|
+
return session
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
it('lands identically with the advisor off, and with extreme scores or a failed pass', async () => {
|
|
239
|
+
const scenarios: string[] = ['off', 'extreme-scores', 'failed-pass']
|
|
240
|
+
const shapes: unknown[] = []
|
|
241
|
+
for (const scenario of scenarios) {
|
|
242
|
+
// One shared session id: replacement markers embed it, and the advisor
|
|
243
|
+
// state is keyed by Session object identity, so this cannot cross-talk.
|
|
244
|
+
const session = buildSession('advisor-invariant')
|
|
245
|
+
if (scenario !== 'off') {
|
|
246
|
+
const state = getAdvisorState(session)
|
|
247
|
+
if (scenario === 'extreme-scores') {
|
|
248
|
+
state.summary = {
|
|
249
|
+
overallTask: 'ALL MUST BE KEPT',
|
|
250
|
+
activeSubtasks: ['keep everything forever'],
|
|
251
|
+
keywords: ['evidence'],
|
|
252
|
+
todoVersion: 'deadbeef',
|
|
253
|
+
turn: 1,
|
|
254
|
+
}
|
|
255
|
+
// Zero relevance everywhere: the most hostile score a channel could
|
|
256
|
+
// return must still not delete, delay, or rewrite anything.
|
|
257
|
+
recordScore(state, 2, { score: 0, turn: 1 })
|
|
258
|
+
}
|
|
259
|
+
if (scenario === 'failed-pass') {
|
|
260
|
+
state.failures = { failures: 9, cooldownUntil: Date.now() + 600_000 }
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
shapes.push(structuredClone(await prunedResult(session)))
|
|
264
|
+
}
|
|
265
|
+
for (const shape of shapes.slice(1)) {
|
|
266
|
+
expect(shape).toEqual(shapes[0])
|
|
267
|
+
}
|
|
268
|
+
// And the baseline scenario actually reduced something (the test is real).
|
|
269
|
+
const baseline = shapes[0] as { pruned: unknown[] }
|
|
270
|
+
expect(baseline.pruned).toHaveLength(1)
|
|
271
|
+
})
|
|
272
|
+
})
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
import { describe, expect, it } from 'vitest'
|
|
2
|
+
import type { SessionEvent } from '@deepseek-ai/dsh-session'
|
|
3
|
+
import {
|
|
4
|
+
collectTaskSemantics,
|
|
5
|
+
prefixDecay,
|
|
6
|
+
selectScoringCandidates,
|
|
7
|
+
type AdvisorCandidate,
|
|
8
|
+
} from '../../src/runtime/tokenpilot/advisor.ts'
|
|
9
|
+
import {
|
|
10
|
+
ADVISOR_RECERTIFIED_LIMIT,
|
|
11
|
+
ADVISOR_SCORES_LIMIT,
|
|
12
|
+
getAdvisorState,
|
|
13
|
+
invalidateOnTaskChange,
|
|
14
|
+
recordRecertified,
|
|
15
|
+
recordScore,
|
|
16
|
+
} from '../../src/runtime/tokenpilot/advisor-state.ts'
|
|
17
|
+
import {
|
|
18
|
+
buildAdvisorScoringUserPrompt,
|
|
19
|
+
parseAdvisorScores,
|
|
20
|
+
parseAdvisorSummary,
|
|
21
|
+
} from '../../src/runtime/tokenpilot/advisor-prompt.ts'
|
|
22
|
+
|
|
23
|
+
function todoWriteEvent(data: unknown): SessionEvent {
|
|
24
|
+
return { type: 'todo/write', seq: 1, time: 0, data } as unknown as SessionEvent
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function userMessageEvent(text: string): SessionEvent {
|
|
28
|
+
return {
|
|
29
|
+
type: 'user/message',
|
|
30
|
+
seq: 2,
|
|
31
|
+
time: 0,
|
|
32
|
+
data: { content: [{ type: 'text', text }] },
|
|
33
|
+
} as unknown as SessionEvent
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
describe('collectTaskSemantics (K5)', () => {
|
|
37
|
+
it('parses a structured todo/write payload', () => {
|
|
38
|
+
const semantics = collectTaskSemantics([
|
|
39
|
+
todoWriteEvent({ todos: [{ content: 'migrate gates', status: 'in_progress' }, 'write tests'] }),
|
|
40
|
+
])
|
|
41
|
+
expect(semantics).toBeDefined()
|
|
42
|
+
expect(semantics?.source).toBe('todos')
|
|
43
|
+
expect(semantics?.taskText).toContain('migrate gates')
|
|
44
|
+
expect(semantics?.taskText).toContain('write tests')
|
|
45
|
+
// Same content ⇒ same version token (deterministic).
|
|
46
|
+
const again = collectTaskSemantics([todoWriteEvent({ todos: [{ content: 'migrate gates', status: 'in_progress' }, 'write tests'] })])
|
|
47
|
+
expect(again?.todoVersion).toBe(semantics?.todoVersion)
|
|
48
|
+
})
|
|
49
|
+
|
|
50
|
+
it('degrades a malformed todo payload to its raw JSON string', () => {
|
|
51
|
+
const semantics = collectTaskSemantics([todoWriteEvent({ todos: 'not-an-array' })])
|
|
52
|
+
expect(semantics).toBeDefined()
|
|
53
|
+
expect(semantics?.source).toBe('raw-todo')
|
|
54
|
+
expect(semantics?.taskText).toContain('not-an-array')
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
it('falls back to recent user/message text when no todo/write exists', () => {
|
|
58
|
+
const semantics = collectTaskSemantics([
|
|
59
|
+
userMessageEvent('earlier request'),
|
|
60
|
+
userMessageEvent('please refactor the pruner gates'),
|
|
61
|
+
])
|
|
62
|
+
expect(semantics).toBeDefined()
|
|
63
|
+
expect(semantics?.source).toBe('messages')
|
|
64
|
+
expect(semantics?.taskText).toContain('refactor the pruner gates')
|
|
65
|
+
})
|
|
66
|
+
|
|
67
|
+
it('picks the most recent todo/write event', () => {
|
|
68
|
+
const semantics = collectTaskSemantics([
|
|
69
|
+
todoWriteEvent({ todos: ['old task'] }),
|
|
70
|
+
userMessageEvent('something else'),
|
|
71
|
+
todoWriteEvent({ todos: ['new task'] }),
|
|
72
|
+
])
|
|
73
|
+
expect(semantics?.taskText).toBe('new task')
|
|
74
|
+
})
|
|
75
|
+
|
|
76
|
+
it('returns undefined for an empty log', () => {
|
|
77
|
+
expect(collectTaskSemantics([])).toBeUndefined()
|
|
78
|
+
})
|
|
79
|
+
})
|
|
80
|
+
|
|
81
|
+
describe('prefixDecay (K6)', () => {
|
|
82
|
+
const candidates: AdvisorCandidate[] = [
|
|
83
|
+
{ seq: 1, characterPressure: 3_000, preview: 'a' },
|
|
84
|
+
{ seq: 2, characterPressure: 1_000, preview: 'b' },
|
|
85
|
+
]
|
|
86
|
+
// prefixDecay takes the weight face only; previews are scoring-prompt inputs.
|
|
87
|
+
const weights = candidates.map(({ seq, characterPressure }) => ({ seq, characterPressure }))
|
|
88
|
+
|
|
89
|
+
it('is deterministic and weights by character pressure', () => {
|
|
90
|
+
const scores = new Map([[1, { score: 0 }], [2, { score: 1 }]])
|
|
91
|
+
const first = prefixDecay(weights, scores)
|
|
92
|
+
const second = prefixDecay(weights, scores)
|
|
93
|
+
expect(first).toEqual(second)
|
|
94
|
+
// weight 3000×0 + 1000×1 over 4000 ⇒ relevance 0.25 ⇒ decay 0.75.
|
|
95
|
+
expect(first.decay).toBeCloseTo(0.75, 12)
|
|
96
|
+
expect(first.weightedChars).toBe(4_000)
|
|
97
|
+
})
|
|
98
|
+
|
|
99
|
+
it('counts unscored candidates as neutral 0.5', () => {
|
|
100
|
+
const decay = prefixDecay(weights, new Map())
|
|
101
|
+
expect(decay.decay).toBeCloseTo(0.5, 12)
|
|
102
|
+
})
|
|
103
|
+
|
|
104
|
+
it('is 0 with no candidates and ignores zero-pressure candidates', () => {
|
|
105
|
+
expect(prefixDecay([], new Map()).decay).toBe(0)
|
|
106
|
+
expect(prefixDecay([{ seq: 9, characterPressure: 0 }], new Map()).decay).toBe(0)
|
|
107
|
+
})
|
|
108
|
+
})
|
|
109
|
+
|
|
110
|
+
describe('selectScoringCandidates', () => {
|
|
111
|
+
const candidates: AdvisorCandidate[] = [
|
|
112
|
+
{ seq: 1, characterPressure: 8_000, preview: 'auth module login handling' },
|
|
113
|
+
{ seq: 2, characterPressure: 2_000, preview: 'tiny fragment' },
|
|
114
|
+
{ seq: 3, characterPressure: 9_000, preview: 'login auth token refresh' },
|
|
115
|
+
{ seq: 4, characterPressure: 12_000, preview: 'unrelated weather report' },
|
|
116
|
+
]
|
|
117
|
+
|
|
118
|
+
it('applies the watermark, the character floor, and the sample limit', () => {
|
|
119
|
+
const state = { watermarkSeq: 1 }
|
|
120
|
+
const picked = selectScoringCandidates(candidates, state, {
|
|
121
|
+
taskKeywords: new Set(['login', 'auth', 'token']),
|
|
122
|
+
minChars: 4_000,
|
|
123
|
+
sampleLimit: 1,
|
|
124
|
+
taskChanged: false,
|
|
125
|
+
})
|
|
126
|
+
// seq 2 below floor, seq 1 at/below watermark; overlap ranks seq 3 first.
|
|
127
|
+
expect(picked.map(item => item.seq)).toEqual([3])
|
|
128
|
+
})
|
|
129
|
+
|
|
130
|
+
it('ignores the watermark when the task semantics changed', () => {
|
|
131
|
+
const picked = selectScoringCandidates(candidates, { watermarkSeq: 4 }, {
|
|
132
|
+
taskKeywords: new Set(['login']),
|
|
133
|
+
minChars: 4_000,
|
|
134
|
+
sampleLimit: 16,
|
|
135
|
+
taskChanged: true,
|
|
136
|
+
})
|
|
137
|
+
// Overlap ties (seq 1 and 3 both match "login") break by character pressure.
|
|
138
|
+
expect(picked.map(item => item.seq)).toEqual([3, 1, 4])
|
|
139
|
+
})
|
|
140
|
+
})
|
|
141
|
+
|
|
142
|
+
describe('advisor-state bounds (K8)', () => {
|
|
143
|
+
it('evicts the oldest score beyond the LRU limit', () => {
|
|
144
|
+
const session = {} as Parameters<typeof getAdvisorState>[0]
|
|
145
|
+
const state = getAdvisorState(session)
|
|
146
|
+
for (let seq = 0; seq < ADVISOR_SCORES_LIMIT + 10; seq += 1) {
|
|
147
|
+
recordScore(state, seq, { score: 0.5, turn: seq })
|
|
148
|
+
}
|
|
149
|
+
expect(state.scores.size).toBe(ADVISOR_SCORES_LIMIT)
|
|
150
|
+
expect(state.scores.has(0)).toBe(false)
|
|
151
|
+
expect(state.scores.has(ADVISOR_SCORES_LIMIT + 9)).toBe(true)
|
|
152
|
+
// Re-touching a seq moves it to the newest position.
|
|
153
|
+
recordScore(state, 10, { score: 0.9, turn: 999 })
|
|
154
|
+
recordScore(state, ADVISOR_SCORES_LIMIT + 10, { score: 0.1, turn: 1_000 })
|
|
155
|
+
expect(state.scores.has(10)).toBe(true)
|
|
156
|
+
expect(state.scores.size).toBe(ADVISOR_SCORES_LIMIT)
|
|
157
|
+
})
|
|
158
|
+
|
|
159
|
+
it('bounds recertified marks the same way', () => {
|
|
160
|
+
const session = {} as Parameters<typeof getAdvisorState>[0]
|
|
161
|
+
const state = getAdvisorState(session)
|
|
162
|
+
for (let seq = 0; seq < ADVISOR_RECERTIFIED_LIMIT + 5; seq += 1) {
|
|
163
|
+
recordRecertified(state, seq, seq)
|
|
164
|
+
}
|
|
165
|
+
expect(state.recertified.size).toBe(ADVISOR_RECERTIFIED_LIMIT)
|
|
166
|
+
expect(state.recertified.has(0)).toBe(false)
|
|
167
|
+
})
|
|
168
|
+
|
|
169
|
+
it('invalidates the summary when the task version changes', () => {
|
|
170
|
+
const session = {} as Parameters<typeof getAdvisorState>[0]
|
|
171
|
+
const state = getAdvisorState(session)
|
|
172
|
+
state.summary = { overallTask: 'x', activeSubtasks: [], keywords: [], todoVersion: 'aaa', turn: 1 }
|
|
173
|
+
state.lastSummaryTurn = 1
|
|
174
|
+
expect(invalidateOnTaskChange(state, 'bbb')).toBe(true)
|
|
175
|
+
expect(state.summary).toBeUndefined()
|
|
176
|
+
expect(state.lastSummaryTurn).toBe(-1)
|
|
177
|
+
expect(invalidateOnTaskChange(state, 'bbb')).toBe(false)
|
|
178
|
+
})
|
|
179
|
+
})
|
|
180
|
+
|
|
181
|
+
describe('advisor prompts (K7)', () => {
|
|
182
|
+
it('parses a well-formed summary answer', () => {
|
|
183
|
+
const parsed = parseAdvisorSummary(
|
|
184
|
+
'Sure! {"overallTask":"migrate gates","activeSubtasks":["port fresh gate"],"keywords":["gates","pruner"]}',
|
|
185
|
+
)
|
|
186
|
+
expect(parsed?.overallTask).toBe('migrate gates')
|
|
187
|
+
expect(parsed?.keywords).toEqual(['gates', 'pruner'])
|
|
188
|
+
})
|
|
189
|
+
|
|
190
|
+
it('fails open on a malformed summary answer', () => {
|
|
191
|
+
expect(parseAdvisorSummary(undefined)).toBeUndefined()
|
|
192
|
+
expect(parseAdvisorSummary('')).toBeUndefined()
|
|
193
|
+
expect(parseAdvisorSummary('not json at all')).toBeUndefined()
|
|
194
|
+
expect(parseAdvisorSummary('{"overallTask":""}')).toBeUndefined()
|
|
195
|
+
expect(parseAdvisorSummary('{"overallTask":"x"}')).toBeUndefined()
|
|
196
|
+
})
|
|
197
|
+
|
|
198
|
+
it('parses JSON-lines scoring answers and drops invalid rows', () => {
|
|
199
|
+
const parsed = parseAdvisorScores(
|
|
200
|
+
'{"seq":1,"score":0.9,"reason":"current task"}\n'
|
|
201
|
+
+ 'noise line {"seq":2,"score":1.5}\n'
|
|
202
|
+
+ '{"seq":99,"score":0.5}\n'
|
|
203
|
+
+ '{"seq":3,"score":0.1}\n',
|
|
204
|
+
new Set([1, 2, 3]),
|
|
205
|
+
)
|
|
206
|
+
expect(parsed?.size).toBe(2)
|
|
207
|
+
expect(parsed?.get(1)?.score).toBe(0.9)
|
|
208
|
+
expect(parsed?.get(3)?.score).toBe(0.1)
|
|
209
|
+
expect(parsed?.has(2)).toBe(false)
|
|
210
|
+
expect(parsed?.has(99)).toBe(false)
|
|
211
|
+
})
|
|
212
|
+
|
|
213
|
+
it('fails open on empty or all-garbage scoring answers', () => {
|
|
214
|
+
expect(parseAdvisorScores(undefined, new Set([1]))).toBeUndefined()
|
|
215
|
+
expect(parseAdvisorScores('', new Set([1]))).toBeUndefined()
|
|
216
|
+
expect(parseAdvisorScores('garbage only', new Set([1]))).toBeUndefined()
|
|
217
|
+
})
|
|
218
|
+
|
|
219
|
+
it('builds a scoring prompt that keeps previews one-per-line', () => {
|
|
220
|
+
const prompt = buildAdvisorScoringUserPrompt('task text', ['sub'], [
|
|
221
|
+
{ seq: 7, preview: 'some\npreview' },
|
|
222
|
+
])
|
|
223
|
+
expect(prompt).toContain('task: task text')
|
|
224
|
+
expect(prompt).toContain('seq=7 | some preview')
|
|
225
|
+
})
|
|
226
|
+
})
|
|
@@ -4,7 +4,7 @@ import {
|
|
|
4
4
|
emitCompressionAudit,
|
|
5
5
|
formatCompressionAudit,
|
|
6
6
|
} from '../../src/runtime/audit.ts'
|
|
7
|
-
import type { CompressionAuditRecord } from '../../src/runtime/audit.ts'
|
|
7
|
+
import type { CompressionAuditRecord, CompressionRewriteAuditRecord } from '../../src/runtime/audit.ts'
|
|
8
8
|
|
|
9
9
|
const custom = {
|
|
10
10
|
version: 3 as const,
|
|
@@ -71,6 +71,7 @@ describe('context-compression audit records', () => {
|
|
|
71
71
|
tokensRemoved: 600_000,
|
|
72
72
|
tokenizerId: 'mock-tokenizer',
|
|
73
73
|
tokenizerRevision: 'r1',
|
|
74
|
+
measurementBasis: 'exact-tokenizer',
|
|
74
75
|
}
|
|
75
76
|
|
|
76
77
|
const parsed = JSON.parse(
|
|
@@ -82,6 +83,48 @@ describe('context-compression audit records', () => {
|
|
|
82
83
|
.toBe(record.tokensBefore - record.tokensAfter)
|
|
83
84
|
})
|
|
84
85
|
|
|
86
|
+
// task_4c/G7: the elided-line count rides the rewrite audit record (optional,
|
|
87
|
+
// JSON round-trip) so the compress→retrieve M/N ratio is computable from
|
|
88
|
+
// session logs alone; records without reducer telemetry omit the field.
|
|
89
|
+
it('round-trips elidedLines on rewrite records and omits it when absent', () => {
|
|
90
|
+
const withTelemetry: CompressionAuditRecord = {
|
|
91
|
+
schemaVersion: 1,
|
|
92
|
+
kind: 'rewrite',
|
|
93
|
+
sessionId: 'telemetry-session',
|
|
94
|
+
profile: 'balanced',
|
|
95
|
+
component: 'fresh',
|
|
96
|
+
stage: 'fresh',
|
|
97
|
+
reducer: 'hypa-code-skeleton',
|
|
98
|
+
manifestEventType: 'compaction/prune',
|
|
99
|
+
manifestSeq: 7,
|
|
100
|
+
replacementSeq: 8,
|
|
101
|
+
sourceSeqs: [6],
|
|
102
|
+
tokensBefore: 20_000,
|
|
103
|
+
tokensAfter: 3_000,
|
|
104
|
+
tokensRemoved: 17_000,
|
|
105
|
+
tokenizerId: 'mock-tokenizer',
|
|
106
|
+
tokenizerRevision: 'r1',
|
|
107
|
+
measurementBasis: 'exact-tokenizer',
|
|
108
|
+
elidedLines: 412,
|
|
109
|
+
}
|
|
110
|
+
const parsed = JSON.parse(
|
|
111
|
+
formatCompressionAudit(withTelemetry).slice(COMPRESSION_AUDIT_PREFIX.length),
|
|
112
|
+
) as CompressionRewriteAuditRecord
|
|
113
|
+
expect(parsed.elidedLines).toBe(412)
|
|
114
|
+
|
|
115
|
+
const { elidedLines: _omitted, ...placeholderRecord } = withTelemetry
|
|
116
|
+
void _omitted
|
|
117
|
+
const withoutTelemetry: CompressionAuditRecord = {
|
|
118
|
+
...placeholderRecord,
|
|
119
|
+
reducer: 'error-evidence-placeholder',
|
|
120
|
+
}
|
|
121
|
+
expect('elidedLines' in withoutTelemetry).toBe(false)
|
|
122
|
+
const parsedWithout = JSON.parse(
|
|
123
|
+
formatCompressionAudit(withoutTelemetry).slice(COMPRESSION_AUDIT_PREFIX.length),
|
|
124
|
+
) as CompressionRewriteAuditRecord
|
|
125
|
+
expect(parsedWithout.elidedLines).toBeUndefined()
|
|
126
|
+
})
|
|
127
|
+
|
|
85
128
|
it('publishes exactly one single-line info message', () => {
|
|
86
129
|
const info = vi.fn()
|
|
87
130
|
const record: CompressionAuditRecord = {
|
|
@@ -170,4 +213,48 @@ describe('context-compression audit records', () => {
|
|
|
170
213
|
) as CompressionAuditRecord
|
|
171
214
|
expect(parsed).toEqual(record)
|
|
172
215
|
})
|
|
216
|
+
|
|
217
|
+
it('keeps advisor-outcome records free of prompts, keys, and content', () => {
|
|
218
|
+
const record: CompressionAuditRecord = {
|
|
219
|
+
schemaVersion: 1,
|
|
220
|
+
kind: 'advisor-outcome',
|
|
221
|
+
sessionId: 'advisor-session',
|
|
222
|
+
phase: 'decay',
|
|
223
|
+
ok: true,
|
|
224
|
+
sampledCount: 16,
|
|
225
|
+
decay: 0.42,
|
|
226
|
+
weightedChars: 57_688,
|
|
227
|
+
turnIndex: 12,
|
|
228
|
+
latencyMs: 0,
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
const line = formatCompressionAudit(record)
|
|
232
|
+
const parsed = JSON.parse(line.slice(COMPRESSION_AUDIT_PREFIX.length)) as CompressionAuditRecord
|
|
233
|
+
expect(parsed).toEqual(record)
|
|
234
|
+
// Only enums and numbers: never the todo snapshot, assistant text, prompts, or keys.
|
|
235
|
+
expect(line).not.toContain('prompt')
|
|
236
|
+
expect(line).not.toContain('"content"')
|
|
237
|
+
expect(line).not.toContain('"text"')
|
|
238
|
+
expect(line).not.toContain('apiKey')
|
|
239
|
+
expect(line).not.toContain('todo')
|
|
240
|
+
})
|
|
241
|
+
|
|
242
|
+
it('carries failure reason codes and the LLM channel on advisor-outcome records', () => {
|
|
243
|
+
const record: CompressionAuditRecord = {
|
|
244
|
+
schemaVersion: 1,
|
|
245
|
+
kind: 'advisor-outcome',
|
|
246
|
+
sessionId: 'advisor-session',
|
|
247
|
+
phase: 'scoring',
|
|
248
|
+
channel: 'direct',
|
|
249
|
+
ok: false,
|
|
250
|
+
sampledCount: 0,
|
|
251
|
+
turnIndex: 3,
|
|
252
|
+
reason: 'no-direct-endpoint',
|
|
253
|
+
latencyMs: 4,
|
|
254
|
+
}
|
|
255
|
+
const parsed = JSON.parse(
|
|
256
|
+
formatCompressionAudit(record).slice(COMPRESSION_AUDIT_PREFIX.length),
|
|
257
|
+
) as CompressionAuditRecord
|
|
258
|
+
expect(parsed).toEqual(record)
|
|
259
|
+
})
|
|
173
260
|
})
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import { describe, expect, it } from 'vitest'
|
|
2
|
+
|
|
3
|
+
import { CHARS_PER_TOKEN, charsForTokens, charsToTokens } from '../../src/runtime/config.ts'
|
|
4
|
+
|
|
5
|
+
describe('character basis conversion', () => {
|
|
6
|
+
it('derives zero tokens for non-positive or non-finite character counts', () => {
|
|
7
|
+
expect(charsToTokens(0)).toBe(0)
|
|
8
|
+
expect(charsToTokens(-1)).toBe(0)
|
|
9
|
+
expect(charsToTokens(NaN)).toBe(0)
|
|
10
|
+
expect(charsToTokens(Infinity)).toBe(0)
|
|
11
|
+
})
|
|
12
|
+
|
|
13
|
+
it('derives the telemetry token figure from characters', () => {
|
|
14
|
+
expect(charsToTokens(1)).toBe(1)
|
|
15
|
+
expect(charsToTokens(4)).toBe(1)
|
|
16
|
+
expect(charsToTokens(5)).toBe(1)
|
|
17
|
+
expect(charsToTokens(8)).toBe(2)
|
|
18
|
+
})
|
|
19
|
+
|
|
20
|
+
it('expresses token-named gates on the character basis', () => {
|
|
21
|
+
expect(charsForTokens(0)).toBe(0)
|
|
22
|
+
expect(charsForTokens(1000)).toBe(4000)
|
|
23
|
+
expect(CHARS_PER_TOKEN).toBe(4.0)
|
|
24
|
+
})
|
|
25
|
+
|
|
26
|
+
it('stays finite at the integer upper bound (off/native gates must not misfire)', () => {
|
|
27
|
+
expect(Number.isFinite(charsForTokens(Number.MAX_SAFE_INTEGER))).toBe(true)
|
|
28
|
+
expect(charsForTokens(Number.MAX_SAFE_INTEGER)).toBeLessThan(Infinity)
|
|
29
|
+
})
|
|
30
|
+
})
|
|
@@ -73,7 +73,10 @@ describe('hypa-code-skeleton reducer', () => {
|
|
|
73
73
|
expect(output?.text).toContain('export interface Options {')
|
|
74
74
|
expect(output?.text).toContain('export function runJob(job: string,')
|
|
75
75
|
expect(output?.text).toContain('class Runner {')
|
|
76
|
-
|
|
76
|
+
// R9b: elision markers cite original-event line ranges instead of a bare
|
|
77
|
+
// count, and the header hint carries a pasteable start_line.
|
|
78
|
+
expect(output?.text).toMatch(/\[\.\.\. lines \d+-\d+ elided \(\d+ lines\)/)
|
|
79
|
+
expect(output?.text).toContain('"start_line":')
|
|
77
80
|
expect(output?.text).not.toContain('const started = Date.now()')
|
|
78
81
|
expect(output?.text).toContain(SOURCE_REF)
|
|
79
82
|
expect(verifyReduction(input, output!)).toBe(true)
|
|
@@ -112,12 +115,20 @@ describe('hypa-code-skeleton reducer', () => {
|
|
|
112
115
|
expect(codePointLength(output!.text)).toBeLessThanOrEqual(1_200)
|
|
113
116
|
})
|
|
114
117
|
|
|
115
|
-
|
|
118
|
+
// R8b moved this landing spot on purpose: unstructured prose used to fall
|
|
119
|
+
// to pi-head (head-only — middle and tail dropped); it now goes through the
|
|
120
|
+
// universal prose-keep reducer, which retains head AND tail plus an R9
|
|
121
|
+
// line-range marker. Code-like read results still land on pi-head (the
|
|
122
|
+
// prose-keep path declines looksLikeSourceCode), pinned two cases below.
|
|
123
|
+
it('lands read-tool prose on the prose-keep reducer with head and tail', () => {
|
|
116
124
|
const prose = Array.from({ length: 200 }, (_, i) =>
|
|
117
125
|
`Paragraph ${String(i)} explains a concept in plain sentences with no code structure at all.`).join('\n')
|
|
118
126
|
const input = readInput(prose)
|
|
119
127
|
const output = reduceFreshToolResult(input)
|
|
120
|
-
expect(output?.reducer).toBe('
|
|
128
|
+
expect(output?.reducer).toBe('prose-keep')
|
|
129
|
+
const lines = output!.text.split('\n')
|
|
130
|
+
expect(lines[0]).toContain('Paragraph 0')
|
|
131
|
+
expect(lines.at(-1)).toContain('Paragraph 199')
|
|
121
132
|
})
|
|
122
133
|
|
|
123
134
|
it('does not fire on small or sparse content', () => {
|