dsh-tap 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. package/AGENTS.md +99 -0
  2. package/CHANGELOG.md +751 -0
  3. package/LICENSE +21 -0
  4. package/README.en.md +59 -0
  5. package/README.md +256 -0
  6. package/cordis.patch.yml +385 -0
  7. package/core/bridge.js +698 -0
  8. package/core/json-store.js +91 -0
  9. package/core/rotation.js +108 -0
  10. package/core/usage-meter.js +176 -0
  11. package/docs/diagnosis-cache-quota.md +181 -0
  12. package/docs/diagnosis-qoder-flash.md +67 -0
  13. package/docs/diagnosis-trae-3003.md +302 -0
  14. package/docs/goals/bridge-port-host-split.md +91 -0
  15. package/docs/goals/desktop-adaptation.md +106 -0
  16. package/docs/goals/qoder-cn-provider-design.md +308 -0
  17. package/docs/goals/settings-card-ux-redesign-plan.md +1680 -0
  18. package/docs/goals/settings-card-ux-redesign.md +162 -0
  19. package/docs/goals/trae-agent-v3.md +41 -0
  20. package/docs/goals/trae-work-cn-repair.md +44 -0
  21. package/docs/goals/v0.8-/351/242/235/345/272/246/345/217/257/350/247/201-/346/250/241/345/236/213/345/212/250/346/200/201/345/214/226-/345/244/232/346/234/215/345/212/241/345/225/206.md +85 -0
  22. package/docs/pitfalls.md +105 -0
  23. package/docs/reverse/trae-cloud-api.md +218 -0
  24. package/docs/reverse/trae-model-catalog.md +129 -0
  25. package/docs/reverse/traework-cn.md +520 -0
  26. package/docs/rules/STATE.md +198 -0
  27. package/docs/rules/content-moderation.md +54 -0
  28. package/docs/rules/dev-role-boundary.md +94 -0
  29. package/docs/rules/extra-providers.md +42 -0
  30. package/docs/rules/gateway-facts.md +91 -0
  31. package/docs/rules/oauth-handshake.md +76 -0
  32. package/docs/rules/prompt-cache.md +93 -0
  33. package/docs/rules/quota-signals.md +125 -0
  34. package/docs/rules/routing.md +84 -0
  35. package/docs/rules/templates/oauth-reverse-checklist.md +42 -0
  36. package/docs/rules/trae-surface.md +242 -0
  37. package/docs/rules/ua-validation.md +81 -0
  38. package/host-config.js +282 -0
  39. package/index.js +2306 -0
  40. package/lib/client.js +2893 -0
  41. package/local-scan.js +104 -0
  42. package/package.json +82 -0
  43. package/providers/ark/index.js +11 -0
  44. package/providers/bailian/index.js +10 -0
  45. package/providers/bigmodel/index.js +11 -0
  46. package/providers/codebuddy/agenttool.js +122 -0
  47. package/providers/codebuddy/catalog.js +227 -0
  48. package/providers/codebuddy/errors.js +44 -0
  49. package/providers/codebuddy/headers.js +36 -0
  50. package/providers/codebuddy/images.js +125 -0
  51. package/providers/codebuddy/index.js +123 -0
  52. package/providers/codebuddy/oauth.js +279 -0
  53. package/providers/deepseek/index.js +11 -0
  54. package/providers/moonshot/index.js +11 -0
  55. package/providers/openai-compat.js +177 -0
  56. package/providers/openrouter/index.js +27 -0
  57. package/providers/qoder/catalog.js +145 -0
  58. package/providers/qoder/cosy.js +419 -0
  59. package/providers/qoder/gateway.js +563 -0
  60. package/providers/qoder/index.js +116 -0
  61. package/providers/qoder/oauth.js +364 -0
  62. package/providers/qoder/qoder_auth.wasm +0 -0
  63. package/providers/qoder/quota.js +56 -0
  64. package/providers/qwen/index.js +15 -0
  65. package/providers/tool-pairing.js +129 -0
  66. package/providers/trae/catalog.js +103 -0
  67. package/providers/trae/errors.js +85 -0
  68. package/providers/trae/gateway.js +853 -0
  69. package/providers/trae/index.js +126 -0
  70. package/providers/trae/oauth.js +443 -0
  71. package/providers/trae/quota.js +75 -0
  72. package/providers/trae/remote.js +365 -0
  73. package/scripts/capture-cache.mjs +65 -0
  74. package/scripts/capture-traffic.mjs +83 -0
  75. package/scripts/hermes-probe-dev-role.mjs +263 -0
  76. package/scripts/measure-latency.mjs +253 -0
  77. package/scripts/probe-ark-thinking.mjs +298 -0
  78. package/scripts/probe-cache-decline.mjs +292 -0
  79. package/scripts/probe-cache-ttl.mjs +221 -0
  80. package/scripts/probe-cache.mjs +156 -0
  81. package/scripts/probe-codebuddy-efforts.mjs +355 -0
  82. package/scripts/probe-codebuddy-tier-wiring.mjs +244 -0
  83. package/scripts/probe-effort-gaps.mjs +133 -0
  84. package/scripts/probe-media.mjs +115 -0
  85. package/scripts/probe-moderation.mjs +159 -0
  86. package/scripts/probe-oauth.mjs +617 -0
  87. package/scripts/probe-qoder-attribution-arm8.mjs +92 -0
  88. package/scripts/probe-qoder-attribution-arm9.mjs +102 -0
  89. package/scripts/probe-qoder-attribution.mjs +266 -0
  90. package/scripts/probe-qoder-flash-confirm.mjs +94 -0
  91. package/scripts/probe-qoder-live.mjs +344 -0
  92. package/scripts/probe-qoder-matrix.mjs +274 -0
  93. package/scripts/probe-qoder-null-content.mjs +153 -0
  94. package/scripts/probe-qoder-pairing.mjs +308 -0
  95. package/scripts/probe-qoder-quota.mjs +162 -0
  96. package/scripts/probe-qoder-thinking-config.mjs +52 -0
  97. package/scripts/probe-qoder-thinking-efforts.mjs +269 -0
  98. package/scripts/probe-quota.mjs +136 -0
  99. package/scripts/probe-quota2.mjs +144 -0
  100. package/scripts/probe-routing.mjs +226 -0
  101. package/scripts/probe-trae-3003-diagnosis.mjs +139 -0
  102. package/scripts/probe-trae-agent-v3.mjs +399 -0
  103. package/scripts/probe-trae-efforts.mjs +149 -0
  104. package/scripts/probe-trae-live.mjs +147 -0
  105. package/scripts/probe-trae-max-effort.mjs +179 -0
  106. package/scripts/probe-trae-model-routing.mjs +381 -0
  107. package/scripts/probe-trae-thinking-scene.mjs +238 -0
  108. package/scripts/probe-trae-transport-outage.mjs +176 -0
  109. package/scripts/probe-ua.mjs +508 -0
  110. package/scripts/trae-model-catalog.mjs +632 -0
  111. package/scripts/verify-agents-md.mjs +45 -0
  112. package/scripts/verify-bridge.mjs +1041 -0
  113. package/scripts/verify-core-generic.mjs +302 -0
  114. package/scripts/verify-desktop-acceptance.mjs +184 -0
  115. package/scripts/verify-host-config.mjs +265 -0
  116. package/scripts/verify-models.mjs +390 -0
  117. package/scripts/verify-providers.mjs +255 -0
  118. package/scripts/verify-qoder-provider.mjs +1020 -0
  119. package/scripts/verify-rotation.mjs +308 -0
  120. package/scripts/verify-trae-model-catalog.mjs +284 -0
  121. package/scripts/verify-trae-provider.mjs +1451 -0
@@ -0,0 +1,221 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Cache invalidation-boundary probe (topic 2 of docs/rules/).
4
+ *
5
+ * Spell under test (AGENTS.md:30-31): "提示缓存按内容寻址、自动生效;命中
6
+ * 粒度 128 token;按模型分策略;glm-5.1/5.2 缓存条目秒-分钟级失效;
7
+ * v4-flash 阈值 412 tok;v4-flash TTL ≥60s 无影响"。
8
+ *
9
+ * What is NOT yet pinned (the boundary this probe maps):
10
+ * dim 1 前缀长度 — minimum cacheable prefix per model (only v4-flash has
11
+ * 412; v4-pro only has ≤2684)
12
+ * dim 2 时间窗 — TTL per model (only v4-flash has ≥60s)
13
+ * dim 3 按模型 — policy table for the 10 uncharacterized models
14
+ *
15
+ * Pre-registered hypotheses (2026-08-19):
16
+ * H-POLICY vendor-cluster: kimi-k2.5/k2.6/k3/k3-1 cache like kimi-k2.7;
17
+ * glm-5v-turbo flaps like glm-5.1/5.2; hy3-preview caches like
18
+ * hy3; deepseek-v3.2/r1 do NOT cache (like v3); minimax-m2.7/m3
19
+ * DO cache (non-DeepSeek vendors on this gateway all do)
20
+ * H-TTL glm-5.2 entry dies somewhere in 5s..120s ("秒-分钟级");
21
+ * v4-pro entry survives 120s (real-session evidence: hits across
22
+ * multi-minute turns)
23
+ * H-THRESH v4-pro minimum cacheable prefix is a 128-token multiple in
24
+ * (412, 2684] — most likely 1024 or 2048
25
+ * H-GRAN hit values are multiples of 128 on every caching model
26
+ *
27
+ * Modes:
28
+ * --mode ttl seed once, re-send at cumulative gaps (default
29
+ * 5,15,30,60,120 s), record hit on each probe
30
+ * --mode sweep policy sweep: N models × 2 immediate re-sends each
31
+ * --mode thresh threshold ladder: prefix sizes × 2 immediate re-sends
32
+ * --mode predict pre-registered confirmation round P1–P4 (2026-08-19):
33
+ * P1 v3.2 entry survives a 60s gap (DeepSeek caching
34
+ * family shares v4-pro's TTL behaviour) → hit>0
35
+ * P2 hy3-preview stays 0-hit at ~5k tokens (policy none,
36
+ * not threshold) → hit=0 on both calls
37
+ * P3 glm-5.2 retention is probabilistic: among 4 rapid
38
+ * (2s) resends after a seeded write, ≥1 misses →
39
+ * not all hit
40
+ * P4 v3.2 hit granularity: at a fresh ~3.3k prompt the
41
+ * second call's hit == floor(prompt/128)*128 exactly
42
+ *
43
+ * Every arm embeds a unique nonce so entries never collide across runs.
44
+ * Low rate: ≤1 call in flight, 2s between calls, max_tokens=16.
45
+ *
46
+ * Usage:
47
+ * node scripts/probe-cache-ttl.mjs --mode ttl --model glm-5.2 --gaps 5,15,30,60,120
48
+ * node scripts/probe-cache-ttl.mjs --mode sweep --models a,b,c [--repeat 72]
49
+ * node scripts/probe-cache-ttl.mjs --mode thresh --model deepseek-v4-pro --sizes 500,1000,1500,2000,2500
50
+ */
51
+
52
+ import { readFileSync, existsSync, appendFileSync } from 'node:fs'
53
+ import { homedir } from 'node:os'
54
+ import { join } from 'node:path'
55
+
56
+ const DSH_HOME = process.env.DSH_HOME ?? join(homedir(), '.dsh')
57
+ const GATEWAY = 'https://copilot.tencent.com'
58
+ const argValue = (flag) => {
59
+ const i = process.argv.indexOf(flag)
60
+ return i > 0 ? process.argv[i + 1] : null
61
+ }
62
+ const MODE = argValue('--mode') ?? 'ttl'
63
+ const MODEL = argValue('--model') ?? 'glm-5.2'
64
+ const MODELS = argValue('--models')?.split(',') ?? null
65
+ const GAPS = (argValue('--gaps') ?? '5,15,30,60,120').split(',').map(Number)
66
+ const SIZES = (argValue('--sizes') ?? '500,1000,1500,2000,2500').split(',').map(Number)
67
+ const REPEAT = Number(argValue('--repeat') ?? 72) // 72 ≈ 2.6k tokens
68
+ const OUT = argValue('--out')
69
+ ?? join('docs', 'probes', `cache-boundary-${new Date().toISOString().slice(0, 10)}.jsonl`)
70
+ const GAP_CALL_MS = 2000
71
+
72
+ const runId = Math.random().toString(36).slice(2, 8)
73
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
74
+
75
+ function resolveKey() {
76
+ try {
77
+ const cfg = JSON.parse(readFileSync(join(DSH_HOME, 'codebuddy-plugin.json'), 'utf8'))
78
+ const active = (cfg.apiKeys ?? []).find((k) => k.name === cfg.activeApiKey)
79
+ if (active?.key) return active.key
80
+ } catch { /* fall through */ }
81
+ if (process.env.CODEBUDDY_API_KEY) return process.env.CODEBUDDY_API_KEY
82
+ const credFile = join(DSH_HOME, '.credentials.yaml')
83
+ if (existsSync(credFile)) {
84
+ const m = readFileSync(credFile, 'utf8').match(/^\s*CODEBUDDY_API_KEY:\s*["']?([^"'\s]+)["']?\s*$/m)
85
+ if (m) return m[1]
86
+ }
87
+ throw new Error('no CodeBuddy credential found')
88
+ }
89
+
90
+ function buildBody(model, nonce, repeat) {
91
+ const paragraph = `Cache boundary probe paragraph ${nonce}. Engineers measure prefix-cache invalidation across time gaps and prefix sizes while the quick brown fox jumps over the lazy dog. `
92
+ return JSON.stringify({
93
+ model,
94
+ stream: true,
95
+ stream_options: { include_usage: true },
96
+ max_tokens: 16,
97
+ messages: [
98
+ { role: 'system', content: paragraph.repeat(repeat) },
99
+ { role: 'user', content: `Probe ${nonce}: reply with exactly: OK` },
100
+ ],
101
+ })
102
+ }
103
+
104
+ async function call(key, model, nonce, repeat) {
105
+ const t0 = Date.now()
106
+ let ttfbMs = null
107
+ try {
108
+ const res = await fetch(`${GATEWAY}/v2/chat/completions`, {
109
+ method: 'POST',
110
+ headers: {
111
+ 'Content-Type': 'application/json',
112
+ Authorization: `Bearer ${key}`,
113
+ 'User-Agent': 'CLI/unknown CodeBuddy/2.136.0',
114
+ },
115
+ body: buildBody(model, nonce, repeat),
116
+ })
117
+ const reader = res.body.getReader()
118
+ const decoder = new TextDecoder()
119
+ let buf = ''
120
+ let usage = null
121
+ for (;;) {
122
+ const { done, value } = await reader.read()
123
+ if (done) break
124
+ if (ttfbMs === null) ttfbMs = Date.now() - t0
125
+ buf += decoder.decode(value, { stream: true })
126
+ let nl
127
+ while ((nl = buf.indexOf('\n')) >= 0) {
128
+ const line = buf.slice(0, nl).trim()
129
+ buf = buf.slice(nl + 1)
130
+ if (!line.startsWith('data:') || !line.includes('"usage"')) continue
131
+ try {
132
+ const chunk = JSON.parse(line.slice(5).trim())
133
+ if (chunk.usage) usage = chunk.usage
134
+ } catch { /* skip */ }
135
+ }
136
+ }
137
+ return { status: res.status, ttfbMs, ms: Date.now() - t0, usage }
138
+ } catch (err) {
139
+ return { status: 0, ttfbMs, ms: Date.now() - t0, error: String(err?.message ?? err) }
140
+ }
141
+ }
142
+
143
+ function report(rec) {
144
+ const u = rec.usage ?? {}
145
+ appendFileSync(OUT, JSON.stringify(rec) + '\n')
146
+ console.log(
147
+ `${rec.mode.padEnd(6)} ${rec.model.padEnd(16)} ${rec.tag.padEnd(14)} http=${rec.status}`
148
+ + ` prompt=${u.prompt_tokens ?? '-'} hit=${u.prompt_cache_hit_tokens ?? '-'} miss=${u.prompt_cache_miss_tokens ?? '-'}`
149
+ + ` credit=${u.credit ?? '-'}${rec.error ? ' err=' + rec.error : ''}`,
150
+ )
151
+ }
152
+
153
+ const key = resolveKey()
154
+ appendFileSync(OUT, JSON.stringify({ type: 'run', at: new Date().toISOString(), mode: MODE, runId, args: process.argv.slice(2) }) + '\n')
155
+ console.log(`evidence → ${OUT} (run ${runId})`)
156
+
157
+ if (MODE === 'ttl') {
158
+ const nonce = `${runId}-ttl`
159
+ report({ type: 'probe', mode: 'ttl', model: MODEL, tag: 'seed', gapS: 0, ...await call(key, MODEL, nonce, REPEAT) })
160
+ let prev = 0
161
+ for (const gap of GAPS) {
162
+ await sleep((gap - prev) * 1000)
163
+ prev = gap
164
+ report({ type: 'probe', mode: 'ttl', model: MODEL, tag: `gap${gap}s`, gapS: gap, ...await call(key, MODEL, nonce, REPEAT) })
165
+ await sleep(GAP_CALL_MS)
166
+ }
167
+ } else if (MODE === 'sweep') {
168
+ for (const model of MODELS ?? []) {
169
+ const nonce = `${runId}-${model.replace(/[^a-z0-9]/gi, '')}`
170
+ for (let i = 1; i <= 2; i++) {
171
+ report({ type: 'probe', mode: 'sweep', model, tag: `call${i}`, ...await call(key, model, nonce, REPEAT) })
172
+ await sleep(GAP_CALL_MS)
173
+ }
174
+ }
175
+ } else if (MODE === 'thresh') {
176
+ // map approximate token sizes to repeat counts using the calibration from
177
+ // the first size, then probe each size twice (seed + immediate resend)
178
+ for (const size of SIZES) {
179
+ const repeat = Math.max(1, Math.round((size / 2600) * REPEAT))
180
+ const nonce = `${runId}-t${size}`
181
+ for (let i = 1; i <= 2; i++) {
182
+ report({ type: 'probe', mode: 'thresh', model: MODEL, tag: `~${size}tok#${i}`, targetTokens: size, repeat, ...await call(key, MODEL, nonce, repeat) })
183
+ await sleep(GAP_CALL_MS)
184
+ }
185
+ }
186
+ } else if (MODE === 'predict') {
187
+ // P1 — v3.2 TTL 60s
188
+ {
189
+ const nonce = `${runId}-p1`
190
+ report({ type: 'probe', mode: 'predict', pred: 'P1', model: 'deepseek-v3.2', tag: 'seed', ...await call(key, 'deepseek-v3.2', nonce, REPEAT) })
191
+ await sleep(60000)
192
+ report({ type: 'probe', mode: 'predict', pred: 'P1', model: 'deepseek-v3.2', tag: 'gap60s', gapS: 60, ...await call(key, 'deepseek-v3.2', nonce, REPEAT) })
193
+ await sleep(GAP_CALL_MS)
194
+ }
195
+ // P2 — hy3-preview at ~5k tokens
196
+ {
197
+ const nonce = `${runId}-p2`
198
+ for (let i = 1; i <= 2; i++) {
199
+ report({ type: 'probe', mode: 'predict', pred: 'P2', model: 'hy3-preview', tag: `call${i}`, ...await call(key, 'hy3-preview', nonce, 140) })
200
+ await sleep(GAP_CALL_MS)
201
+ }
202
+ }
203
+ // P3 — glm-5.2 probabilistic retention: 4 rapid resends
204
+ {
205
+ const nonce = `${runId}-p3`
206
+ report({ type: 'probe', mode: 'predict', pred: 'P3', model: 'glm-5.2', tag: 'seed', ...await call(key, 'glm-5.2', nonce, REPEAT) })
207
+ await sleep(GAP_CALL_MS)
208
+ for (let i = 1; i <= 4; i++) {
209
+ report({ type: 'probe', mode: 'predict', pred: 'P3', model: 'glm-5.2', tag: `rapid${i}`, ...await call(key, 'glm-5.2', nonce, REPEAT) })
210
+ await sleep(GAP_CALL_MS)
211
+ }
212
+ }
213
+ // P4 — v3.2 granularity at ~3.3k
214
+ {
215
+ const nonce = `${runId}-p4`
216
+ report({ type: 'probe', mode: 'predict', pred: 'P4', model: 'deepseek-v3.2', tag: '~3300tok#1', ...await call(key, 'deepseek-v3.2', nonce, 92) })
217
+ await sleep(GAP_CALL_MS)
218
+ report({ type: 'probe', mode: 'predict', pred: 'P4', model: 'deepseek-v3.2', tag: '~3300tok#2', ...await call(key, 'deepseek-v3.2', nonce, 92) })
219
+ }
220
+ }
221
+ console.log('done.')
@@ -0,0 +1,156 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Gateway cache-routing probe (small, bounded: 4 arms × 3 calls = 12 chat
4
+ * completions, max_tokens 16 each).
5
+ *
6
+ * Question: does the CodeBuddy gateway's prompt/KV cache care about
7
+ * (a) session-affinity headers (openai set: session_id / x-client-request-id /
8
+ * x-session-affinity), or (b) a body `prompt_cache_key`? The dsh main chat
9
+ * sends neither (the pi-ai adapter strips the affinity switch), so every
10
+ * request is anonymous — live captures show hit rates swinging 0 → 15.7k
11
+ * tokens between consecutive turns of one session.
12
+ *
13
+ * Design: each arm sends the SAME payload 3 times back-to-back; every arm
14
+ * embeds a unique nonce so arms can never share cache entries. We record the
15
+ * full usage object (prompt_cache_hit_tokens / prompt_cache_miss_tokens /
16
+ * credit) plus TTFB and total latency.
17
+ *
18
+ * Usage: node scripts/probe-cache.mjs [--out /path/probe.jsonl]
19
+ * Credentials resolve exactly like the plugin's: active key from
20
+ * ~/.dsh/codebuddy-plugin.json (fallback: CODEBUDDY_API_KEY / credentials file).
21
+ */
22
+
23
+ import { readFileSync, existsSync, appendFileSync } from 'node:fs'
24
+ import { homedir } from 'node:os'
25
+ import { join } from 'node:path'
26
+
27
+ const DSH_HOME = process.env.DSH_HOME ?? join(homedir(), '.dsh')
28
+ const GATEWAY = 'https://copilot.tencent.com'
29
+ const argValue = (flag) => {
30
+ const i = process.argv.indexOf(flag)
31
+ return i > 0 ? process.argv[i + 1] : null
32
+ }
33
+ // --base overrides the URL prefix, e.g. http://127.0.0.1:3901/v2 to send the
34
+ // identical payload through the plugin bridge instead of direct to gateway.
35
+ const BASE = argValue('--base') ?? `${GATEWAY}/v2`
36
+ const MODEL = argValue('--model') ?? 'deepseek-v3'
37
+ const EFFORT = argValue('--effort') ?? null
38
+ const ONLY_ARMS = argValue('--arms')?.split(',') ?? null
39
+ const REPEAT = Number(argValue('--repeat') ?? 72)
40
+ const CALLS = Number(argValue('--calls') ?? 3)
41
+ const MAX_TOKENS = 16
42
+
43
+ // ------------------------------------------------------------- credential
44
+ function resolveKey() {
45
+ try {
46
+ const cfg = JSON.parse(readFileSync(join(DSH_HOME, 'codebuddy-plugin.json'), 'utf8'))
47
+ const active = (cfg.apiKeys ?? []).find((k) => k.name === cfg.activeApiKey)
48
+ if (active?.key) return active.key
49
+ } catch { /* fall through */ }
50
+ if (process.env.CODEBUDDY_API_KEY) return process.env.CODEBUDDY_API_KEY
51
+ const credFile = join(DSH_HOME, '.credentials.yaml')
52
+ if (existsSync(credFile)) {
53
+ const m = readFileSync(credFile, 'utf8').match(/^\s*CODEBUDDY_API_KEY:\s*["']?([^"'\s]+)["']?\s*$/m)
54
+ if (m) return m[1]
55
+ }
56
+ throw new Error('no CodeBuddy credential found')
57
+ }
58
+
59
+ // ------------------------------------------------------------- payload
60
+ /** ~1300-token system prompt with an arm-unique nonce baked in. */
61
+ function buildPayload(nonce, cacheKey) {
62
+ const paragraph = `Cache-routing probe paragraph ${nonce}. The quick brown fox jumps over the lazy dog near the riverbank while engineers measure prefix-cache behaviour across identical requests. `
63
+ const system = paragraph.repeat(REPEAT) // 72 ≈ 2.6k tokens; scale via --repeat
64
+ const body = {
65
+ model: MODEL,
66
+ stream: true,
67
+ stream_options: { include_usage: true },
68
+ max_tokens: MAX_TOKENS,
69
+ messages: [
70
+ { role: 'system', content: system },
71
+ { role: 'user', content: `Probe ${nonce}: reply with exactly: OK` },
72
+ ],
73
+ }
74
+ if (cacheKey) body.prompt_cache_key = cacheKey
75
+ if (EFFORT) body.reasoning_effort = EFFORT
76
+ return JSON.stringify(body)
77
+ }
78
+
79
+ // ------------------------------------------------------------- one call
80
+ async function call(key, { nonce, sessionHeaders, cacheKey }) {
81
+ const headers = {
82
+ 'Content-Type': 'application/json',
83
+ Authorization: `Bearer ${key}`,
84
+ // the CLI-shaped UA the bridge sends (v2 does not enforce it, keep it faithful)
85
+ 'User-Agent': 'CLI/unknown CodeBuddy/2.136.0',
86
+ 'X-IDE-Type': 'CLI',
87
+ 'X-IDE-Name': 'CLI',
88
+ 'X-IDE-Version': '2.133.1',
89
+ 'X-Product-Version': '2.133.1',
90
+ 'X-Requested-With': 'XMLHttpRequest',
91
+ 'X-Private-Data': 'false',
92
+ }
93
+ if (sessionHeaders) {
94
+ headers.session_id = sessionHeaders
95
+ headers['x-client-request-id'] = sessionHeaders
96
+ headers['x-session-affinity'] = sessionHeaders
97
+ }
98
+ const t0 = Date.now()
99
+ let ttfbMs = null
100
+ const res = await fetch(`${BASE}/chat/completions`, {
101
+ method: 'POST',
102
+ headers,
103
+ body: buildPayload(nonce, cacheKey),
104
+ })
105
+ const reader = res.body.getReader()
106
+ const decoder = new TextDecoder()
107
+ let buf = ''
108
+ let usage = null
109
+ for (;;) {
110
+ const { done, value } = await reader.read()
111
+ if (done) break
112
+ if (ttfbMs === null) ttfbMs = Date.now() - t0
113
+ buf += decoder.decode(value, { stream: true })
114
+ let nl
115
+ while ((nl = buf.indexOf('\n')) >= 0) {
116
+ const line = buf.slice(0, nl).trim()
117
+ buf = buf.slice(nl + 1)
118
+ if (!line.startsWith('data:') || !line.includes('"usage"')) continue
119
+ try {
120
+ const chunk = JSON.parse(line.slice(5).trim())
121
+ if (chunk.usage) usage = chunk.usage
122
+ } catch { /* skip */ }
123
+ }
124
+ }
125
+ return { status: res.status, ttfbMs, ms: Date.now() - t0, usage }
126
+ }
127
+
128
+ // ------------------------------------------------------------- arms
129
+ const runId = Math.random().toString(36).slice(2, 8)
130
+ const ARMS = [
131
+ { name: 'anon', sessionHeaders: null, cacheKey: null },
132
+ { name: 'session', sessionHeaders: `probe-${runId}-s`, cacheKey: null },
133
+ { name: 'cachekey', sessionHeaders: null, cacheKey: `probe-${runId}-k` },
134
+ { name: 'session+cachekey', sessionHeaders: `probe-${runId}-sk`, cacheKey: `probe-${runId}-sk` },
135
+ ]
136
+
137
+ const outPath = argValue('--out')
138
+ const key = resolveKey()
139
+ const arms = ONLY_ARMS ? ARMS.filter((a) => ONLY_ARMS.includes(a.name)) : ARMS
140
+
141
+ console.log(`probe run ${runId}: ${arms.length} arms × ${CALLS} calls, model=${MODEL}, base=${BASE}, max_tokens=${MAX_TOKENS}`)
142
+ for (const arm of arms) {
143
+ const nonce = `${runId}-${arm.name}`
144
+ for (let i = 1; i <= CALLS; i++) {
145
+ const r = await call(key, { nonce, sessionHeaders: arm.sessionHeaders, cacheKey: arm.cacheKey })
146
+ const u = r.usage ?? {}
147
+ const record = { runId, arm: arm.name, call: i, status: r.status, ttfbMs: r.ttfbMs, ms: r.ms, usage: r.usage }
148
+ if (outPath) appendFileSync(outPath, JSON.stringify(record) + '\n')
149
+ console.log(
150
+ `${arm.name.padEnd(17)} #${i} http=${r.status} ttfb=${String(r.ttfbMs).padStart(5)}ms total=${String(r.ms).padStart(5)}ms`
151
+ + ` prompt=${u.prompt_tokens ?? '-'} hit=${u.prompt_cache_hit_tokens ?? '-'} miss=${u.prompt_cache_miss_tokens ?? '-'}`
152
+ + ` cached_details=${u.prompt_tokens_details?.cached_tokens ?? '-'} credit=${u.credit ?? '-'}`,
153
+ )
154
+ }
155
+ }
156
+ console.log('probe done.')