dsh-tap 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +99 -0
- package/CHANGELOG.md +751 -0
- package/LICENSE +21 -0
- package/README.en.md +59 -0
- package/README.md +256 -0
- package/cordis.patch.yml +385 -0
- package/core/bridge.js +698 -0
- package/core/json-store.js +91 -0
- package/core/rotation.js +108 -0
- package/core/usage-meter.js +176 -0
- package/docs/diagnosis-cache-quota.md +181 -0
- package/docs/diagnosis-qoder-flash.md +67 -0
- package/docs/diagnosis-trae-3003.md +302 -0
- package/docs/goals/bridge-port-host-split.md +91 -0
- package/docs/goals/desktop-adaptation.md +106 -0
- package/docs/goals/qoder-cn-provider-design.md +308 -0
- package/docs/goals/settings-card-ux-redesign-plan.md +1680 -0
- package/docs/goals/settings-card-ux-redesign.md +162 -0
- package/docs/goals/trae-agent-v3.md +41 -0
- package/docs/goals/trae-work-cn-repair.md +44 -0
- package/docs/goals/v0.8-/351/242/235/345/272/246/345/217/257/350/247/201-/346/250/241/345/236/213/345/212/250/346/200/201/345/214/226-/345/244/232/346/234/215/345/212/241/345/225/206.md +85 -0
- package/docs/pitfalls.md +105 -0
- package/docs/reverse/trae-cloud-api.md +218 -0
- package/docs/reverse/trae-model-catalog.md +129 -0
- package/docs/reverse/traework-cn.md +520 -0
- package/docs/rules/STATE.md +198 -0
- package/docs/rules/content-moderation.md +54 -0
- package/docs/rules/dev-role-boundary.md +94 -0
- package/docs/rules/extra-providers.md +42 -0
- package/docs/rules/gateway-facts.md +91 -0
- package/docs/rules/oauth-handshake.md +76 -0
- package/docs/rules/prompt-cache.md +93 -0
- package/docs/rules/quota-signals.md +125 -0
- package/docs/rules/routing.md +84 -0
- package/docs/rules/templates/oauth-reverse-checklist.md +42 -0
- package/docs/rules/trae-surface.md +242 -0
- package/docs/rules/ua-validation.md +81 -0
- package/host-config.js +282 -0
- package/index.js +2306 -0
- package/lib/client.js +2893 -0
- package/local-scan.js +104 -0
- package/package.json +82 -0
- package/providers/ark/index.js +11 -0
- package/providers/bailian/index.js +10 -0
- package/providers/bigmodel/index.js +11 -0
- package/providers/codebuddy/agenttool.js +122 -0
- package/providers/codebuddy/catalog.js +227 -0
- package/providers/codebuddy/errors.js +44 -0
- package/providers/codebuddy/headers.js +36 -0
- package/providers/codebuddy/images.js +125 -0
- package/providers/codebuddy/index.js +123 -0
- package/providers/codebuddy/oauth.js +279 -0
- package/providers/deepseek/index.js +11 -0
- package/providers/moonshot/index.js +11 -0
- package/providers/openai-compat.js +177 -0
- package/providers/openrouter/index.js +27 -0
- package/providers/qoder/catalog.js +145 -0
- package/providers/qoder/cosy.js +419 -0
- package/providers/qoder/gateway.js +563 -0
- package/providers/qoder/index.js +116 -0
- package/providers/qoder/oauth.js +364 -0
- package/providers/qoder/qoder_auth.wasm +0 -0
- package/providers/qoder/quota.js +56 -0
- package/providers/qwen/index.js +15 -0
- package/providers/tool-pairing.js +129 -0
- package/providers/trae/catalog.js +103 -0
- package/providers/trae/errors.js +85 -0
- package/providers/trae/gateway.js +853 -0
- package/providers/trae/index.js +126 -0
- package/providers/trae/oauth.js +443 -0
- package/providers/trae/quota.js +75 -0
- package/providers/trae/remote.js +365 -0
- package/scripts/capture-cache.mjs +65 -0
- package/scripts/capture-traffic.mjs +83 -0
- package/scripts/hermes-probe-dev-role.mjs +263 -0
- package/scripts/measure-latency.mjs +253 -0
- package/scripts/probe-ark-thinking.mjs +298 -0
- package/scripts/probe-cache-decline.mjs +292 -0
- package/scripts/probe-cache-ttl.mjs +221 -0
- package/scripts/probe-cache.mjs +156 -0
- package/scripts/probe-codebuddy-efforts.mjs +355 -0
- package/scripts/probe-codebuddy-tier-wiring.mjs +244 -0
- package/scripts/probe-effort-gaps.mjs +133 -0
- package/scripts/probe-media.mjs +115 -0
- package/scripts/probe-moderation.mjs +159 -0
- package/scripts/probe-oauth.mjs +617 -0
- package/scripts/probe-qoder-attribution-arm8.mjs +92 -0
- package/scripts/probe-qoder-attribution-arm9.mjs +102 -0
- package/scripts/probe-qoder-attribution.mjs +266 -0
- package/scripts/probe-qoder-flash-confirm.mjs +94 -0
- package/scripts/probe-qoder-live.mjs +344 -0
- package/scripts/probe-qoder-matrix.mjs +274 -0
- package/scripts/probe-qoder-null-content.mjs +153 -0
- package/scripts/probe-qoder-pairing.mjs +308 -0
- package/scripts/probe-qoder-quota.mjs +162 -0
- package/scripts/probe-qoder-thinking-config.mjs +52 -0
- package/scripts/probe-qoder-thinking-efforts.mjs +269 -0
- package/scripts/probe-quota.mjs +136 -0
- package/scripts/probe-quota2.mjs +144 -0
- package/scripts/probe-routing.mjs +226 -0
- package/scripts/probe-trae-3003-diagnosis.mjs +139 -0
- package/scripts/probe-trae-agent-v3.mjs +399 -0
- package/scripts/probe-trae-efforts.mjs +149 -0
- package/scripts/probe-trae-live.mjs +147 -0
- package/scripts/probe-trae-max-effort.mjs +179 -0
- package/scripts/probe-trae-model-routing.mjs +381 -0
- package/scripts/probe-trae-thinking-scene.mjs +238 -0
- package/scripts/probe-trae-transport-outage.mjs +176 -0
- package/scripts/probe-ua.mjs +508 -0
- package/scripts/trae-model-catalog.mjs +632 -0
- package/scripts/verify-agents-md.mjs +45 -0
- package/scripts/verify-bridge.mjs +1041 -0
- package/scripts/verify-core-generic.mjs +302 -0
- package/scripts/verify-desktop-acceptance.mjs +184 -0
- package/scripts/verify-host-config.mjs +265 -0
- package/scripts/verify-models.mjs +390 -0
- package/scripts/verify-providers.mjs +255 -0
- package/scripts/verify-qoder-provider.mjs +1020 -0
- package/scripts/verify-rotation.mjs +308 -0
- package/scripts/verify-trae-model-catalog.mjs +284 -0
- package/scripts/verify-trae-provider.mjs +1451 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Cache invalidation-boundary probe (topic 2 of docs/rules/).
|
|
4
|
+
*
|
|
5
|
+
* Spell under test (AGENTS.md:30-31): "提示缓存按内容寻址、自动生效;命中
|
|
6
|
+
* 粒度 128 token;按模型分策略;glm-5.1/5.2 缓存条目秒-分钟级失效;
|
|
7
|
+
* v4-flash 阈值 412 tok;v4-flash TTL ≥60s 无影响"。
|
|
8
|
+
*
|
|
9
|
+
* What is NOT yet pinned (the boundary this probe maps):
|
|
10
|
+
* dim 1 前缀长度 — minimum cacheable prefix per model (only v4-flash has
|
|
11
|
+
* 412; v4-pro only has ≤2684)
|
|
12
|
+
* dim 2 时间窗 — TTL per model (only v4-flash has ≥60s)
|
|
13
|
+
* dim 3 按模型 — policy table for the 10 uncharacterized models
|
|
14
|
+
*
|
|
15
|
+
* Pre-registered hypotheses (2026-08-19):
|
|
16
|
+
* H-POLICY vendor-cluster: kimi-k2.5/k2.6/k3/k3-1 cache like kimi-k2.7;
|
|
17
|
+
* glm-5v-turbo flaps like glm-5.1/5.2; hy3-preview caches like
|
|
18
|
+
* hy3; deepseek-v3.2/r1 do NOT cache (like v3); minimax-m2.7/m3
|
|
19
|
+
* DO cache (non-DeepSeek vendors on this gateway all do)
|
|
20
|
+
* H-TTL glm-5.2 entry dies somewhere in 5s..120s ("秒-分钟级");
|
|
21
|
+
* v4-pro entry survives 120s (real-session evidence: hits across
|
|
22
|
+
* multi-minute turns)
|
|
23
|
+
* H-THRESH v4-pro minimum cacheable prefix is a 128-token multiple in
|
|
24
|
+
* (412, 2684] — most likely 1024 or 2048
|
|
25
|
+
* H-GRAN hit values are multiples of 128 on every caching model
|
|
26
|
+
*
|
|
27
|
+
* Modes:
|
|
28
|
+
* --mode ttl seed once, re-send at cumulative gaps (default
|
|
29
|
+
* 5,15,30,60,120 s), record hit on each probe
|
|
30
|
+
* --mode sweep policy sweep: N models × 2 immediate re-sends each
|
|
31
|
+
* --mode thresh threshold ladder: prefix sizes × 2 immediate re-sends
|
|
32
|
+
* --mode predict pre-registered confirmation round P1–P4 (2026-08-19):
|
|
33
|
+
* P1 v3.2 entry survives a 60s gap (DeepSeek caching
|
|
34
|
+
* family shares v4-pro's TTL behaviour) → hit>0
|
|
35
|
+
* P2 hy3-preview stays 0-hit at ~5k tokens (policy none,
|
|
36
|
+
* not threshold) → hit=0 on both calls
|
|
37
|
+
* P3 glm-5.2 retention is probabilistic: among 4 rapid
|
|
38
|
+
* (2s) resends after a seeded write, ≥1 misses →
|
|
39
|
+
* not all hit
|
|
40
|
+
* P4 v3.2 hit granularity: at a fresh ~3.3k prompt the
|
|
41
|
+
* second call's hit == floor(prompt/128)*128 exactly
|
|
42
|
+
*
|
|
43
|
+
* Every arm embeds a unique nonce so entries never collide across runs.
|
|
44
|
+
* Low rate: ≤1 call in flight, 2s between calls, max_tokens=16.
|
|
45
|
+
*
|
|
46
|
+
* Usage:
|
|
47
|
+
* node scripts/probe-cache-ttl.mjs --mode ttl --model glm-5.2 --gaps 5,15,30,60,120
|
|
48
|
+
* node scripts/probe-cache-ttl.mjs --mode sweep --models a,b,c [--repeat 72]
|
|
49
|
+
* node scripts/probe-cache-ttl.mjs --mode thresh --model deepseek-v4-pro --sizes 500,1000,1500,2000,2500
|
|
50
|
+
*/
|
|
51
|
+
|
|
52
|
+
import { readFileSync, existsSync, appendFileSync } from 'node:fs'
|
|
53
|
+
import { homedir } from 'node:os'
|
|
54
|
+
import { join } from 'node:path'
|
|
55
|
+
|
|
56
|
+
const DSH_HOME = process.env.DSH_HOME ?? join(homedir(), '.dsh')
|
|
57
|
+
const GATEWAY = 'https://copilot.tencent.com'
|
|
58
|
+
const argValue = (flag) => {
|
|
59
|
+
const i = process.argv.indexOf(flag)
|
|
60
|
+
return i > 0 ? process.argv[i + 1] : null
|
|
61
|
+
}
|
|
62
|
+
const MODE = argValue('--mode') ?? 'ttl'
|
|
63
|
+
const MODEL = argValue('--model') ?? 'glm-5.2'
|
|
64
|
+
const MODELS = argValue('--models')?.split(',') ?? null
|
|
65
|
+
const GAPS = (argValue('--gaps') ?? '5,15,30,60,120').split(',').map(Number)
|
|
66
|
+
const SIZES = (argValue('--sizes') ?? '500,1000,1500,2000,2500').split(',').map(Number)
|
|
67
|
+
const REPEAT = Number(argValue('--repeat') ?? 72) // 72 ≈ 2.6k tokens
|
|
68
|
+
const OUT = argValue('--out')
|
|
69
|
+
?? join('docs', 'probes', `cache-boundary-${new Date().toISOString().slice(0, 10)}.jsonl`)
|
|
70
|
+
const GAP_CALL_MS = 2000
|
|
71
|
+
|
|
72
|
+
const runId = Math.random().toString(36).slice(2, 8)
|
|
73
|
+
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
|
|
74
|
+
|
|
75
|
+
function resolveKey() {
|
|
76
|
+
try {
|
|
77
|
+
const cfg = JSON.parse(readFileSync(join(DSH_HOME, 'codebuddy-plugin.json'), 'utf8'))
|
|
78
|
+
const active = (cfg.apiKeys ?? []).find((k) => k.name === cfg.activeApiKey)
|
|
79
|
+
if (active?.key) return active.key
|
|
80
|
+
} catch { /* fall through */ }
|
|
81
|
+
if (process.env.CODEBUDDY_API_KEY) return process.env.CODEBUDDY_API_KEY
|
|
82
|
+
const credFile = join(DSH_HOME, '.credentials.yaml')
|
|
83
|
+
if (existsSync(credFile)) {
|
|
84
|
+
const m = readFileSync(credFile, 'utf8').match(/^\s*CODEBUDDY_API_KEY:\s*["']?([^"'\s]+)["']?\s*$/m)
|
|
85
|
+
if (m) return m[1]
|
|
86
|
+
}
|
|
87
|
+
throw new Error('no CodeBuddy credential found')
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function buildBody(model, nonce, repeat) {
|
|
91
|
+
const paragraph = `Cache boundary probe paragraph ${nonce}. Engineers measure prefix-cache invalidation across time gaps and prefix sizes while the quick brown fox jumps over the lazy dog. `
|
|
92
|
+
return JSON.stringify({
|
|
93
|
+
model,
|
|
94
|
+
stream: true,
|
|
95
|
+
stream_options: { include_usage: true },
|
|
96
|
+
max_tokens: 16,
|
|
97
|
+
messages: [
|
|
98
|
+
{ role: 'system', content: paragraph.repeat(repeat) },
|
|
99
|
+
{ role: 'user', content: `Probe ${nonce}: reply with exactly: OK` },
|
|
100
|
+
],
|
|
101
|
+
})
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
async function call(key, model, nonce, repeat) {
|
|
105
|
+
const t0 = Date.now()
|
|
106
|
+
let ttfbMs = null
|
|
107
|
+
try {
|
|
108
|
+
const res = await fetch(`${GATEWAY}/v2/chat/completions`, {
|
|
109
|
+
method: 'POST',
|
|
110
|
+
headers: {
|
|
111
|
+
'Content-Type': 'application/json',
|
|
112
|
+
Authorization: `Bearer ${key}`,
|
|
113
|
+
'User-Agent': 'CLI/unknown CodeBuddy/2.136.0',
|
|
114
|
+
},
|
|
115
|
+
body: buildBody(model, nonce, repeat),
|
|
116
|
+
})
|
|
117
|
+
const reader = res.body.getReader()
|
|
118
|
+
const decoder = new TextDecoder()
|
|
119
|
+
let buf = ''
|
|
120
|
+
let usage = null
|
|
121
|
+
for (;;) {
|
|
122
|
+
const { done, value } = await reader.read()
|
|
123
|
+
if (done) break
|
|
124
|
+
if (ttfbMs === null) ttfbMs = Date.now() - t0
|
|
125
|
+
buf += decoder.decode(value, { stream: true })
|
|
126
|
+
let nl
|
|
127
|
+
while ((nl = buf.indexOf('\n')) >= 0) {
|
|
128
|
+
const line = buf.slice(0, nl).trim()
|
|
129
|
+
buf = buf.slice(nl + 1)
|
|
130
|
+
if (!line.startsWith('data:') || !line.includes('"usage"')) continue
|
|
131
|
+
try {
|
|
132
|
+
const chunk = JSON.parse(line.slice(5).trim())
|
|
133
|
+
if (chunk.usage) usage = chunk.usage
|
|
134
|
+
} catch { /* skip */ }
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
return { status: res.status, ttfbMs, ms: Date.now() - t0, usage }
|
|
138
|
+
} catch (err) {
|
|
139
|
+
return { status: 0, ttfbMs, ms: Date.now() - t0, error: String(err?.message ?? err) }
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
function report(rec) {
|
|
144
|
+
const u = rec.usage ?? {}
|
|
145
|
+
appendFileSync(OUT, JSON.stringify(rec) + '\n')
|
|
146
|
+
console.log(
|
|
147
|
+
`${rec.mode.padEnd(6)} ${rec.model.padEnd(16)} ${rec.tag.padEnd(14)} http=${rec.status}`
|
|
148
|
+
+ ` prompt=${u.prompt_tokens ?? '-'} hit=${u.prompt_cache_hit_tokens ?? '-'} miss=${u.prompt_cache_miss_tokens ?? '-'}`
|
|
149
|
+
+ ` credit=${u.credit ?? '-'}${rec.error ? ' err=' + rec.error : ''}`,
|
|
150
|
+
)
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
const key = resolveKey()
|
|
154
|
+
appendFileSync(OUT, JSON.stringify({ type: 'run', at: new Date().toISOString(), mode: MODE, runId, args: process.argv.slice(2) }) + '\n')
|
|
155
|
+
console.log(`evidence → ${OUT} (run ${runId})`)
|
|
156
|
+
|
|
157
|
+
if (MODE === 'ttl') {
|
|
158
|
+
const nonce = `${runId}-ttl`
|
|
159
|
+
report({ type: 'probe', mode: 'ttl', model: MODEL, tag: 'seed', gapS: 0, ...await call(key, MODEL, nonce, REPEAT) })
|
|
160
|
+
let prev = 0
|
|
161
|
+
for (const gap of GAPS) {
|
|
162
|
+
await sleep((gap - prev) * 1000)
|
|
163
|
+
prev = gap
|
|
164
|
+
report({ type: 'probe', mode: 'ttl', model: MODEL, tag: `gap${gap}s`, gapS: gap, ...await call(key, MODEL, nonce, REPEAT) })
|
|
165
|
+
await sleep(GAP_CALL_MS)
|
|
166
|
+
}
|
|
167
|
+
} else if (MODE === 'sweep') {
|
|
168
|
+
for (const model of MODELS ?? []) {
|
|
169
|
+
const nonce = `${runId}-${model.replace(/[^a-z0-9]/gi, '')}`
|
|
170
|
+
for (let i = 1; i <= 2; i++) {
|
|
171
|
+
report({ type: 'probe', mode: 'sweep', model, tag: `call${i}`, ...await call(key, model, nonce, REPEAT) })
|
|
172
|
+
await sleep(GAP_CALL_MS)
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
} else if (MODE === 'thresh') {
|
|
176
|
+
// map approximate token sizes to repeat counts using the calibration from
|
|
177
|
+
// the first size, then probe each size twice (seed + immediate resend)
|
|
178
|
+
for (const size of SIZES) {
|
|
179
|
+
const repeat = Math.max(1, Math.round((size / 2600) * REPEAT))
|
|
180
|
+
const nonce = `${runId}-t${size}`
|
|
181
|
+
for (let i = 1; i <= 2; i++) {
|
|
182
|
+
report({ type: 'probe', mode: 'thresh', model: MODEL, tag: `~${size}tok#${i}`, targetTokens: size, repeat, ...await call(key, MODEL, nonce, repeat) })
|
|
183
|
+
await sleep(GAP_CALL_MS)
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
} else if (MODE === 'predict') {
|
|
187
|
+
// P1 — v3.2 TTL 60s
|
|
188
|
+
{
|
|
189
|
+
const nonce = `${runId}-p1`
|
|
190
|
+
report({ type: 'probe', mode: 'predict', pred: 'P1', model: 'deepseek-v3.2', tag: 'seed', ...await call(key, 'deepseek-v3.2', nonce, REPEAT) })
|
|
191
|
+
await sleep(60000)
|
|
192
|
+
report({ type: 'probe', mode: 'predict', pred: 'P1', model: 'deepseek-v3.2', tag: 'gap60s', gapS: 60, ...await call(key, 'deepseek-v3.2', nonce, REPEAT) })
|
|
193
|
+
await sleep(GAP_CALL_MS)
|
|
194
|
+
}
|
|
195
|
+
// P2 — hy3-preview at ~5k tokens
|
|
196
|
+
{
|
|
197
|
+
const nonce = `${runId}-p2`
|
|
198
|
+
for (let i = 1; i <= 2; i++) {
|
|
199
|
+
report({ type: 'probe', mode: 'predict', pred: 'P2', model: 'hy3-preview', tag: `call${i}`, ...await call(key, 'hy3-preview', nonce, 140) })
|
|
200
|
+
await sleep(GAP_CALL_MS)
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
// P3 — glm-5.2 probabilistic retention: 4 rapid resends
|
|
204
|
+
{
|
|
205
|
+
const nonce = `${runId}-p3`
|
|
206
|
+
report({ type: 'probe', mode: 'predict', pred: 'P3', model: 'glm-5.2', tag: 'seed', ...await call(key, 'glm-5.2', nonce, REPEAT) })
|
|
207
|
+
await sleep(GAP_CALL_MS)
|
|
208
|
+
for (let i = 1; i <= 4; i++) {
|
|
209
|
+
report({ type: 'probe', mode: 'predict', pred: 'P3', model: 'glm-5.2', tag: `rapid${i}`, ...await call(key, 'glm-5.2', nonce, REPEAT) })
|
|
210
|
+
await sleep(GAP_CALL_MS)
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
// P4 — v3.2 granularity at ~3.3k
|
|
214
|
+
{
|
|
215
|
+
const nonce = `${runId}-p4`
|
|
216
|
+
report({ type: 'probe', mode: 'predict', pred: 'P4', model: 'deepseek-v3.2', tag: '~3300tok#1', ...await call(key, 'deepseek-v3.2', nonce, 92) })
|
|
217
|
+
await sleep(GAP_CALL_MS)
|
|
218
|
+
report({ type: 'probe', mode: 'predict', pred: 'P4', model: 'deepseek-v3.2', tag: '~3300tok#2', ...await call(key, 'deepseek-v3.2', nonce, 92) })
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
console.log('done.')
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Gateway cache-routing probe (small, bounded: 4 arms × 3 calls = 12 chat
|
|
4
|
+
* completions, max_tokens 16 each).
|
|
5
|
+
*
|
|
6
|
+
* Question: does the CodeBuddy gateway's prompt/KV cache care about
|
|
7
|
+
* (a) session-affinity headers (openai set: session_id / x-client-request-id /
|
|
8
|
+
* x-session-affinity), or (b) a body `prompt_cache_key`? The dsh main chat
|
|
9
|
+
* sends neither (the pi-ai adapter strips the affinity switch), so every
|
|
10
|
+
* request is anonymous — live captures show hit rates swinging 0 → 15.7k
|
|
11
|
+
* tokens between consecutive turns of one session.
|
|
12
|
+
*
|
|
13
|
+
* Design: each arm sends the SAME payload 3 times back-to-back; every arm
|
|
14
|
+
* embeds a unique nonce so arms can never share cache entries. We record the
|
|
15
|
+
* full usage object (prompt_cache_hit_tokens / prompt_cache_miss_tokens /
|
|
16
|
+
* credit) plus TTFB and total latency.
|
|
17
|
+
*
|
|
18
|
+
* Usage: node scripts/probe-cache.mjs [--out /path/probe.jsonl]
|
|
19
|
+
* Credentials resolve exactly like the plugin's: active key from
|
|
20
|
+
* ~/.dsh/codebuddy-plugin.json (fallback: CODEBUDDY_API_KEY / credentials file).
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import { readFileSync, existsSync, appendFileSync } from 'node:fs'
|
|
24
|
+
import { homedir } from 'node:os'
|
|
25
|
+
import { join } from 'node:path'
|
|
26
|
+
|
|
27
|
+
const DSH_HOME = process.env.DSH_HOME ?? join(homedir(), '.dsh')
|
|
28
|
+
const GATEWAY = 'https://copilot.tencent.com'
|
|
29
|
+
const argValue = (flag) => {
|
|
30
|
+
const i = process.argv.indexOf(flag)
|
|
31
|
+
return i > 0 ? process.argv[i + 1] : null
|
|
32
|
+
}
|
|
33
|
+
// --base overrides the URL prefix, e.g. http://127.0.0.1:3901/v2 to send the
|
|
34
|
+
// identical payload through the plugin bridge instead of direct to gateway.
|
|
35
|
+
const BASE = argValue('--base') ?? `${GATEWAY}/v2`
|
|
36
|
+
const MODEL = argValue('--model') ?? 'deepseek-v3'
|
|
37
|
+
const EFFORT = argValue('--effort') ?? null
|
|
38
|
+
const ONLY_ARMS = argValue('--arms')?.split(',') ?? null
|
|
39
|
+
const REPEAT = Number(argValue('--repeat') ?? 72)
|
|
40
|
+
const CALLS = Number(argValue('--calls') ?? 3)
|
|
41
|
+
const MAX_TOKENS = 16
|
|
42
|
+
|
|
43
|
+
// ------------------------------------------------------------- credential
|
|
44
|
+
function resolveKey() {
|
|
45
|
+
try {
|
|
46
|
+
const cfg = JSON.parse(readFileSync(join(DSH_HOME, 'codebuddy-plugin.json'), 'utf8'))
|
|
47
|
+
const active = (cfg.apiKeys ?? []).find((k) => k.name === cfg.activeApiKey)
|
|
48
|
+
if (active?.key) return active.key
|
|
49
|
+
} catch { /* fall through */ }
|
|
50
|
+
if (process.env.CODEBUDDY_API_KEY) return process.env.CODEBUDDY_API_KEY
|
|
51
|
+
const credFile = join(DSH_HOME, '.credentials.yaml')
|
|
52
|
+
if (existsSync(credFile)) {
|
|
53
|
+
const m = readFileSync(credFile, 'utf8').match(/^\s*CODEBUDDY_API_KEY:\s*["']?([^"'\s]+)["']?\s*$/m)
|
|
54
|
+
if (m) return m[1]
|
|
55
|
+
}
|
|
56
|
+
throw new Error('no CodeBuddy credential found')
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// ------------------------------------------------------------- payload
|
|
60
|
+
/** ~1300-token system prompt with an arm-unique nonce baked in. */
|
|
61
|
+
function buildPayload(nonce, cacheKey) {
|
|
62
|
+
const paragraph = `Cache-routing probe paragraph ${nonce}. The quick brown fox jumps over the lazy dog near the riverbank while engineers measure prefix-cache behaviour across identical requests. `
|
|
63
|
+
const system = paragraph.repeat(REPEAT) // 72 ≈ 2.6k tokens; scale via --repeat
|
|
64
|
+
const body = {
|
|
65
|
+
model: MODEL,
|
|
66
|
+
stream: true,
|
|
67
|
+
stream_options: { include_usage: true },
|
|
68
|
+
max_tokens: MAX_TOKENS,
|
|
69
|
+
messages: [
|
|
70
|
+
{ role: 'system', content: system },
|
|
71
|
+
{ role: 'user', content: `Probe ${nonce}: reply with exactly: OK` },
|
|
72
|
+
],
|
|
73
|
+
}
|
|
74
|
+
if (cacheKey) body.prompt_cache_key = cacheKey
|
|
75
|
+
if (EFFORT) body.reasoning_effort = EFFORT
|
|
76
|
+
return JSON.stringify(body)
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// ------------------------------------------------------------- one call
|
|
80
|
+
async function call(key, { nonce, sessionHeaders, cacheKey }) {
|
|
81
|
+
const headers = {
|
|
82
|
+
'Content-Type': 'application/json',
|
|
83
|
+
Authorization: `Bearer ${key}`,
|
|
84
|
+
// the CLI-shaped UA the bridge sends (v2 does not enforce it, keep it faithful)
|
|
85
|
+
'User-Agent': 'CLI/unknown CodeBuddy/2.136.0',
|
|
86
|
+
'X-IDE-Type': 'CLI',
|
|
87
|
+
'X-IDE-Name': 'CLI',
|
|
88
|
+
'X-IDE-Version': '2.133.1',
|
|
89
|
+
'X-Product-Version': '2.133.1',
|
|
90
|
+
'X-Requested-With': 'XMLHttpRequest',
|
|
91
|
+
'X-Private-Data': 'false',
|
|
92
|
+
}
|
|
93
|
+
if (sessionHeaders) {
|
|
94
|
+
headers.session_id = sessionHeaders
|
|
95
|
+
headers['x-client-request-id'] = sessionHeaders
|
|
96
|
+
headers['x-session-affinity'] = sessionHeaders
|
|
97
|
+
}
|
|
98
|
+
const t0 = Date.now()
|
|
99
|
+
let ttfbMs = null
|
|
100
|
+
const res = await fetch(`${BASE}/chat/completions`, {
|
|
101
|
+
method: 'POST',
|
|
102
|
+
headers,
|
|
103
|
+
body: buildPayload(nonce, cacheKey),
|
|
104
|
+
})
|
|
105
|
+
const reader = res.body.getReader()
|
|
106
|
+
const decoder = new TextDecoder()
|
|
107
|
+
let buf = ''
|
|
108
|
+
let usage = null
|
|
109
|
+
for (;;) {
|
|
110
|
+
const { done, value } = await reader.read()
|
|
111
|
+
if (done) break
|
|
112
|
+
if (ttfbMs === null) ttfbMs = Date.now() - t0
|
|
113
|
+
buf += decoder.decode(value, { stream: true })
|
|
114
|
+
let nl
|
|
115
|
+
while ((nl = buf.indexOf('\n')) >= 0) {
|
|
116
|
+
const line = buf.slice(0, nl).trim()
|
|
117
|
+
buf = buf.slice(nl + 1)
|
|
118
|
+
if (!line.startsWith('data:') || !line.includes('"usage"')) continue
|
|
119
|
+
try {
|
|
120
|
+
const chunk = JSON.parse(line.slice(5).trim())
|
|
121
|
+
if (chunk.usage) usage = chunk.usage
|
|
122
|
+
} catch { /* skip */ }
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
return { status: res.status, ttfbMs, ms: Date.now() - t0, usage }
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// ------------------------------------------------------------- arms
|
|
129
|
+
const runId = Math.random().toString(36).slice(2, 8)
|
|
130
|
+
const ARMS = [
|
|
131
|
+
{ name: 'anon', sessionHeaders: null, cacheKey: null },
|
|
132
|
+
{ name: 'session', sessionHeaders: `probe-${runId}-s`, cacheKey: null },
|
|
133
|
+
{ name: 'cachekey', sessionHeaders: null, cacheKey: `probe-${runId}-k` },
|
|
134
|
+
{ name: 'session+cachekey', sessionHeaders: `probe-${runId}-sk`, cacheKey: `probe-${runId}-sk` },
|
|
135
|
+
]
|
|
136
|
+
|
|
137
|
+
const outPath = argValue('--out')
|
|
138
|
+
const key = resolveKey()
|
|
139
|
+
const arms = ONLY_ARMS ? ARMS.filter((a) => ONLY_ARMS.includes(a.name)) : ARMS
|
|
140
|
+
|
|
141
|
+
console.log(`probe run ${runId}: ${arms.length} arms × ${CALLS} calls, model=${MODEL}, base=${BASE}, max_tokens=${MAX_TOKENS}`)
|
|
142
|
+
for (const arm of arms) {
|
|
143
|
+
const nonce = `${runId}-${arm.name}`
|
|
144
|
+
for (let i = 1; i <= CALLS; i++) {
|
|
145
|
+
const r = await call(key, { nonce, sessionHeaders: arm.sessionHeaders, cacheKey: arm.cacheKey })
|
|
146
|
+
const u = r.usage ?? {}
|
|
147
|
+
const record = { runId, arm: arm.name, call: i, status: r.status, ttfbMs: r.ttfbMs, ms: r.ms, usage: r.usage }
|
|
148
|
+
if (outPath) appendFileSync(outPath, JSON.stringify(record) + '\n')
|
|
149
|
+
console.log(
|
|
150
|
+
`${arm.name.padEnd(17)} #${i} http=${r.status} ttfb=${String(r.ttfbMs).padStart(5)}ms total=${String(r.ms).padStart(5)}ms`
|
|
151
|
+
+ ` prompt=${u.prompt_tokens ?? '-'} hit=${u.prompt_cache_hit_tokens ?? '-'} miss=${u.prompt_cache_miss_tokens ?? '-'}`
|
|
152
|
+
+ ` cached_details=${u.prompt_tokens_details?.cached_tokens ?? '-'} credit=${u.credit ?? '-'}`,
|
|
153
|
+
)
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
console.log('probe done.')
|