dsh-tap 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. package/AGENTS.md +99 -0
  2. package/CHANGELOG.md +751 -0
  3. package/LICENSE +21 -0
  4. package/README.en.md +59 -0
  5. package/README.md +256 -0
  6. package/cordis.patch.yml +385 -0
  7. package/core/bridge.js +698 -0
  8. package/core/json-store.js +91 -0
  9. package/core/rotation.js +108 -0
  10. package/core/usage-meter.js +176 -0
  11. package/docs/diagnosis-cache-quota.md +181 -0
  12. package/docs/diagnosis-qoder-flash.md +67 -0
  13. package/docs/diagnosis-trae-3003.md +302 -0
  14. package/docs/goals/bridge-port-host-split.md +91 -0
  15. package/docs/goals/desktop-adaptation.md +106 -0
  16. package/docs/goals/qoder-cn-provider-design.md +308 -0
  17. package/docs/goals/settings-card-ux-redesign-plan.md +1680 -0
  18. package/docs/goals/settings-card-ux-redesign.md +162 -0
  19. package/docs/goals/trae-agent-v3.md +41 -0
  20. package/docs/goals/trae-work-cn-repair.md +44 -0
  21. package/docs/goals/v0.8-/351/242/235/345/272/246/345/217/257/350/247/201-/346/250/241/345/236/213/345/212/250/346/200/201/345/214/226-/345/244/232/346/234/215/345/212/241/345/225/206.md +85 -0
  22. package/docs/pitfalls.md +105 -0
  23. package/docs/reverse/trae-cloud-api.md +218 -0
  24. package/docs/reverse/trae-model-catalog.md +129 -0
  25. package/docs/reverse/traework-cn.md +520 -0
  26. package/docs/rules/STATE.md +198 -0
  27. package/docs/rules/content-moderation.md +54 -0
  28. package/docs/rules/dev-role-boundary.md +94 -0
  29. package/docs/rules/extra-providers.md +42 -0
  30. package/docs/rules/gateway-facts.md +91 -0
  31. package/docs/rules/oauth-handshake.md +76 -0
  32. package/docs/rules/prompt-cache.md +93 -0
  33. package/docs/rules/quota-signals.md +125 -0
  34. package/docs/rules/routing.md +84 -0
  35. package/docs/rules/templates/oauth-reverse-checklist.md +42 -0
  36. package/docs/rules/trae-surface.md +242 -0
  37. package/docs/rules/ua-validation.md +81 -0
  38. package/host-config.js +282 -0
  39. package/index.js +2306 -0
  40. package/lib/client.js +2893 -0
  41. package/local-scan.js +104 -0
  42. package/package.json +82 -0
  43. package/providers/ark/index.js +11 -0
  44. package/providers/bailian/index.js +10 -0
  45. package/providers/bigmodel/index.js +11 -0
  46. package/providers/codebuddy/agenttool.js +122 -0
  47. package/providers/codebuddy/catalog.js +227 -0
  48. package/providers/codebuddy/errors.js +44 -0
  49. package/providers/codebuddy/headers.js +36 -0
  50. package/providers/codebuddy/images.js +125 -0
  51. package/providers/codebuddy/index.js +123 -0
  52. package/providers/codebuddy/oauth.js +279 -0
  53. package/providers/deepseek/index.js +11 -0
  54. package/providers/moonshot/index.js +11 -0
  55. package/providers/openai-compat.js +177 -0
  56. package/providers/openrouter/index.js +27 -0
  57. package/providers/qoder/catalog.js +145 -0
  58. package/providers/qoder/cosy.js +419 -0
  59. package/providers/qoder/gateway.js +563 -0
  60. package/providers/qoder/index.js +116 -0
  61. package/providers/qoder/oauth.js +364 -0
  62. package/providers/qoder/qoder_auth.wasm +0 -0
  63. package/providers/qoder/quota.js +56 -0
  64. package/providers/qwen/index.js +15 -0
  65. package/providers/tool-pairing.js +129 -0
  66. package/providers/trae/catalog.js +103 -0
  67. package/providers/trae/errors.js +85 -0
  68. package/providers/trae/gateway.js +853 -0
  69. package/providers/trae/index.js +126 -0
  70. package/providers/trae/oauth.js +443 -0
  71. package/providers/trae/quota.js +75 -0
  72. package/providers/trae/remote.js +365 -0
  73. package/scripts/capture-cache.mjs +65 -0
  74. package/scripts/capture-traffic.mjs +83 -0
  75. package/scripts/hermes-probe-dev-role.mjs +263 -0
  76. package/scripts/measure-latency.mjs +253 -0
  77. package/scripts/probe-ark-thinking.mjs +298 -0
  78. package/scripts/probe-cache-decline.mjs +292 -0
  79. package/scripts/probe-cache-ttl.mjs +221 -0
  80. package/scripts/probe-cache.mjs +156 -0
  81. package/scripts/probe-codebuddy-efforts.mjs +355 -0
  82. package/scripts/probe-codebuddy-tier-wiring.mjs +244 -0
  83. package/scripts/probe-effort-gaps.mjs +133 -0
  84. package/scripts/probe-media.mjs +115 -0
  85. package/scripts/probe-moderation.mjs +159 -0
  86. package/scripts/probe-oauth.mjs +617 -0
  87. package/scripts/probe-qoder-attribution-arm8.mjs +92 -0
  88. package/scripts/probe-qoder-attribution-arm9.mjs +102 -0
  89. package/scripts/probe-qoder-attribution.mjs +266 -0
  90. package/scripts/probe-qoder-flash-confirm.mjs +94 -0
  91. package/scripts/probe-qoder-live.mjs +344 -0
  92. package/scripts/probe-qoder-matrix.mjs +274 -0
  93. package/scripts/probe-qoder-null-content.mjs +153 -0
  94. package/scripts/probe-qoder-pairing.mjs +308 -0
  95. package/scripts/probe-qoder-quota.mjs +162 -0
  96. package/scripts/probe-qoder-thinking-config.mjs +52 -0
  97. package/scripts/probe-qoder-thinking-efforts.mjs +269 -0
  98. package/scripts/probe-quota.mjs +136 -0
  99. package/scripts/probe-quota2.mjs +144 -0
  100. package/scripts/probe-routing.mjs +226 -0
  101. package/scripts/probe-trae-3003-diagnosis.mjs +139 -0
  102. package/scripts/probe-trae-agent-v3.mjs +399 -0
  103. package/scripts/probe-trae-efforts.mjs +149 -0
  104. package/scripts/probe-trae-live.mjs +147 -0
  105. package/scripts/probe-trae-max-effort.mjs +179 -0
  106. package/scripts/probe-trae-model-routing.mjs +381 -0
  107. package/scripts/probe-trae-thinking-scene.mjs +238 -0
  108. package/scripts/probe-trae-transport-outage.mjs +176 -0
  109. package/scripts/probe-ua.mjs +508 -0
  110. package/scripts/trae-model-catalog.mjs +632 -0
  111. package/scripts/verify-agents-md.mjs +45 -0
  112. package/scripts/verify-bridge.mjs +1041 -0
  113. package/scripts/verify-core-generic.mjs +302 -0
  114. package/scripts/verify-desktop-acceptance.mjs +184 -0
  115. package/scripts/verify-host-config.mjs +265 -0
  116. package/scripts/verify-models.mjs +390 -0
  117. package/scripts/verify-providers.mjs +255 -0
  118. package/scripts/verify-qoder-provider.mjs +1020 -0
  119. package/scripts/verify-rotation.mjs +308 -0
  120. package/scripts/verify-trae-model-catalog.mjs +284 -0
  121. package/scripts/verify-trae-provider.mjs +1451 -0
@@ -0,0 +1,298 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * probe-ark-thinking.mjs — 火山 Ark `/api/plan`(anthropic 方言)逐模型思考面实测,
4
+ * 并**产出可直接用的 `reasoningEfforts` 声明**(新模型进 Ark 后不必再手工分析)。
5
+ *
6
+ * 为什么需要:宿主输入框的「推理等级」只由模型条目的 `reasoningEfforts` 决定
7
+ * (踩坑 #64),而 Ark 的 plan 端点**不发布目录/能力声明**(无 GET /models),
8
+ * 所以这张表只能靠实测。手工做过一次(2026-10-04,见 docs/goals/desktop-adaptation.md),
9
+ * 本脚本把那次分析固化成一条命令。
10
+ *
11
+ * 判据(默认**省额度两臂**;`--full-matrix` 才是旧三臂全矩阵):
12
+ * baseline = 不带 thinking → 有 thinking 块 = 该模型默认思考
13
+ * enabled = thinking{type:enabled, budget_tokens} → 200 = 接受显式档位
14
+ * disabled = thinking{type:disabled} → 200 且无 thinking 块 = **支持真关思考**(仅 --full-matrix)
15
+ * → 声明里给 `off: null`(= 省略参数… 注意
16
+ * 见下方 off 语义说明);400 = 不给 off 档
17
+ *
18
+ * 省额度默认(2026-10-04 起;此前一次 7 模型全量把 volces 额度打光):每模型只跑
19
+ * baseline+enabled 两臂,disabled 不跑 → off 视为「未测」,档位表不含 off(不臆造档位,
20
+ * 踩坑 #42);`--write` 遇 patch 里已有 off 档(此前 --full-matrix 实测写入)会保留不覆盖。
21
+ * off 的新证/否证只能靠 `--full-matrix`。
22
+ *
23
+ * off 语义(踩坑 #64):pi-ai 对 anthropic 方言的 off 线值就是发
24
+ * `thinking:{type:disabled}`,所以**只有 disabled 被接受的模型才声明 off**;
25
+ * 不声明 off 的模型"默认档" = 不带 thinking 参数 = 上游默认(照常思考),安全。
26
+ *
27
+ * 用法:
28
+ * node scripts/probe-ark-thinking.mjs # 省额度默认:baseline+enabled 两臂
29
+ * node scripts/probe-ark-thinking.mjs --full-matrix # 旧行为:三臂全矩阵(可证 off)
30
+ * node scripts/probe-ark-thinking.mjs --models a,b,c # 指定模型
31
+ * node scripts/probe-ark-thinking.mjs --emit yaml # 打印可粘贴的 reasoningEfforts 块
32
+ * node scripts/probe-ark-thinking.mjs --write <patch.yml> # 直接写进 profile patch(留 .bak)
33
+ * node scripts/probe-ark-thinking.mjs --levels low,medium,high,max # 档位拼写(--levels-as-arms 下每档位各加一臂)
34
+ * node scripts/probe-ark-thinking.mjs --levels-as-arms --repeat 3 # 档位臂模式:每声明档位一个 enabled 臂(budget=pi-ai 映射 low 2048/medium 8192/high 16384/max 16384;max_tokens=32768=真实宿主 defaultMaxTokens),baseline 保留 → 定论「档位是否单调(真旋钮 vs 只是 budget 上限)」
35
+ * node scripts/probe-ark-thinking.mjs --repeat 3 # 每臂 3 次并聚合(min/mean/max),默认 1
36
+ * node scripts/probe-ark-thinking.mjs --from <证据.json> # 复用证据,不打上游只 emit/write
37
+ *
38
+ * 额度账:每模型上游调用 = 两臂(默认)或三臂(--full-matrix)或 1+档位数 臂(--levels-as-arms)× --repeat;7 模型默认 14 次。
39
+ *
40
+ * 凭据:`~/.dsh/.credentials.yaml` 的 `refs.VOLCES_API_KEY`(或环境变量 VOLCES_API_KEY)。
41
+ * 证据 → docs/probes/ark-thinking-<ts>.json
42
+ */
43
+ import { readFileSync, writeFileSync, copyFileSync, mkdirSync, existsSync } from 'node:fs'
44
+ import { homedir } from 'node:os'
45
+ import { join } from 'node:path'
46
+ import YAML, { isScalar } from 'yaml'
47
+
48
+ const DSH_HOME = process.env.DSH_HOME ?? join(homedir(), '.dsh')
49
+ const ENDPOINT = process.env.ARK_PLAN_ENDPOINT ?? 'https://ark.cn-beijing.volces.com/api/plan/v1/messages'
50
+ const DEFAULT_MODELS = ['deepseek-v4.1-flash', 'ark-code-latest', 'glm-5.3-flash', 'glm-5.3', 'kimi-k3', 'doubao-seed-evolving', 'doubao-seed-2.1-lite']
51
+ const PROMPT = '一个农夫有 17 只羊,除了 9 只以外都跑了,还剩几只?只给最终数字。'
52
+
53
+ const args = process.argv.slice(2)
54
+ const argOf = (name, dflt) => {
55
+ const i = args.indexOf(name)
56
+ return i >= 0 && args[i + 1] !== undefined && !args[i + 1].startsWith('--') ? args[i + 1] : dflt
57
+ }
58
+ const MODELS = String(argOf('--models', DEFAULT_MODELS.join(','))).split(',').map((s) => s.trim()).filter(Boolean)
59
+ const LEVELS = String(argOf('--levels', 'low,medium,high,max')).split(',').map((s) => s.trim()).filter(Boolean)
60
+ const EMIT = args.includes('--emit')
61
+ const WRITE = argOf('--write', null)
62
+ const REPEAT = Number(argOf('--repeat', '1')) || 1
63
+ const FULL_MATRIX = args.includes('--full-matrix')
64
+ const LEVELS_AS_ARMS = args.includes('--levels-as-arms')
65
+ // pi-ai 档位→budget_tokens 映射(0.18.0 诚实边界①,docs/goals/desktop-adaptation.md:max 夹到 high);
66
+ // 未知档位回落 2048 = enabled 臂默认 budget。
67
+ const LEVEL_BUDGET = { low: 2048, medium: 8192, high: 16384, max: 16384 }
68
+ // 档位臂的 max_tokens 用真实宿主线值(desktop patch volces defaultMaxTokens):
69
+ // budget 8192/16384 若仍配 probe 旧的 max_tokens 2048,思考会被 max_tokens 钳死(甚至被上游拒),
70
+ // budget 就不再是臂间唯一变量。仅 levels-as-arms 模式生效,其余模式保持 2048 不动。
71
+ const LEVELS_AS_ARMS_MAX_TOKENS = 32768
72
+ const MODE = LEVELS_AS_ARMS ? (FULL_MATRIX ? 'levels-as-arms+full-matrix' : 'levels-as-arms') : FULL_MATRIX ? 'full-matrix' : 'eco'
73
+ const MODE_LABEL = LEVELS_AS_ARMS ? '档位臂(levels-as-arms)' : FULL_MATRIX ? '全矩阵' : '省额度(默认)'
74
+
75
+ const USAGE = `probe-ark-thinking.mjs — Ark /api/plan 逐模型思考面实测 → reasoningEfforts 声明
76
+
77
+ 用法:node scripts/probe-ark-thinking.mjs [选项]
78
+
79
+ 默认省额度模式:每模型只跑 baseline(不带 thinking)+ enabled(thinking enabled 默认 budget)
80
+ 两臂;disabled 臂不跑 → off 档视为「未测」,档位表不含 off(不臆造),--write 会保留
81
+ patch 里已有的 off 档。只有 --full-matrix 才跑 disabled 臂、才能下 off 结论。
82
+
83
+ --full-matrix 恢复全矩阵:baseline/enabled/disabled 三臂 × --repeat
84
+ --models a,b,c 指定模型(默认静态清单 7 个)
85
+ --levels a,b,c 档位拼写,只影响产出的声明拼写,不影响上游调用次数(默认 low,medium,high,max;
86
+ 但 --levels-as-arms 下每个档位各加一臂)
87
+ --levels-as-arms 档位臂模式:每个声明档位各开一个 enabled 臂,budget_tokens=pi-ai 映射
88
+ (low 2048/medium 8192/high 16384/max 16384),baseline 保留;max_tokens 用
89
+ 32768=真实宿主 defaultMaxTokens(否则 budget>2048 会被钳/拒)。定论「档位是否
90
+ 单调(真旋钮还是只是 budget 上限)」用;与 --repeat 正常组合
91
+ --repeat N 每臂重复 N 次并聚合(thinkingChars 给 min/mean/max,状态/错误归并);默认 1
92
+ --emit 打印可粘贴的 reasoningEfforts 块
93
+ --write <patch.yml> 直接写进 profile patch 的 volces.models(留 .bak)
94
+ --from <证据.json> 复用既有证据文件,不打上游,只做 emit/write
95
+ --help 本帮助
96
+
97
+ 凭据:~/.dsh/.credentials.yaml 的 refs.VOLCES_API_KEY(或环境变量 VOLCES_API_KEY)。
98
+ 证据 → docs/probes/ark-thinking-<ts>.json。
99
+ 额度账:每模型上游调用 = 两臂(默认)或三臂(--full-matrix)或 1+档位数 臂(--levels-as-arms)× --repeat;7 模型默认 14 次。`
100
+
101
+ if (args.includes('--help') || args.includes('-h')) {
102
+ console.log(USAGE)
103
+ process.exit(0)
104
+ }
105
+
106
+ function apiKey() {
107
+ if (process.env.VOLCES_API_KEY) return process.env.VOLCES_API_KEY
108
+ const p = join(DSH_HOME, '.credentials.yaml')
109
+ if (!existsSync(p)) return null
110
+ const doc = YAML.parse(readFileSync(p, 'utf8'))
111
+ return doc?.refs?.VOLCES_API_KEY ?? null
112
+ }
113
+
114
+ const key = apiKey()
115
+ if (!key) {
116
+ console.error(`未找到凭据:环境变量 VOLCES_API_KEY 或 ${join(DSH_HOME, '.credentials.yaml')} 的 refs.VOLCES_API_KEY`)
117
+ process.exit(2)
118
+ }
119
+
120
+ async function call(model, thinking, maxTokens = 2048) {
121
+ const res = await fetch(ENDPOINT, {
122
+ method: 'POST',
123
+ headers: { 'x-api-key': key, 'anthropic-version': '2023-06-01', 'content-type': 'application/json' },
124
+ body: JSON.stringify({
125
+ model,
126
+ max_tokens: maxTokens,
127
+ messages: [{ role: 'user', content: PROMPT }],
128
+ ...(thinking === undefined ? {} : { thinking }),
129
+ }),
130
+ })
131
+ const text = await res.text()
132
+ let json = null
133
+ try { json = JSON.parse(text) } catch { /* keep text */ }
134
+ const blocks = Array.isArray(json?.content) ? json.content : []
135
+ return {
136
+ status: res.status,
137
+ blockTypes: blocks.map((b) => b?.type).filter(Boolean),
138
+ thinkingChars: blocks.filter((b) => b?.type === 'thinking').reduce((n, b) => n + String(b.thinking ?? '').length, 0),
139
+ textChars: blocks.filter((b) => b?.type === 'text').reduce((n, b) => n + String(b.text ?? '').length, 0),
140
+ usage: json?.usage ?? null,
141
+ error: json?.error ? { code: json.error.code, message: String(json.error.message ?? '').slice(0, 160) } : (res.ok ? null : text.slice(0, 160)),
142
+ }
143
+ }
144
+
145
+ // --repeat>1 时每臂聚合:保留全部样本,标量字段归并(thinkingChars/textChars 给 min/mean/max,
146
+ // status/error 去重)。此前循环每轮直接覆盖 arms 只留最后一次,多打的样本纯烧配额——本次修复。
147
+ // runs===1 时保持旧标量形状,证据文件向后兼容。
148
+ function aggregateArm(samples) {
149
+ if (samples.length === 1) return { runs: 1, ...samples[0], samples }
150
+ const stat = (f) => {
151
+ const vs = samples.map(f)
152
+ return { min: Math.min(...vs), mean: Math.round((vs.reduce((a, b) => a + b, 0) / vs.length) * 10) / 10, max: Math.max(...vs) }
153
+ }
154
+ const statuses = [...new Set(samples.map((s) => s.status))]
155
+ const errors = [...new Set(samples.filter((s) => s.error).map((s) => JSON.stringify(s.error)))].map((s) => JSON.parse(s))
156
+ return {
157
+ runs: samples.length,
158
+ status: statuses.length === 1 ? statuses[0] : statuses,
159
+ thinkingChars: stat((s) => s.thinkingChars),
160
+ textChars: stat((s) => s.textChars),
161
+ error: errors.length ? errors : null,
162
+ samples,
163
+ }
164
+ }
165
+
166
+ function fmtArm(a) {
167
+ if (a.runs === 1) {
168
+ return `http=${a.status} blocks=${JSON.stringify(a.blockTypes)} thinking=${a.thinkingChars}${a.error ? ` ${JSON.stringify(a.error)}` : ''}`
169
+ }
170
+ return `http=${JSON.stringify(a.status)} ×${a.runs} thinking=min ${a.thinkingChars.min}/mean ${a.thinkingChars.mean}/max ${a.thinkingChars.max}${a.error ? ` errors=${JSON.stringify(a.error)}` : ''}`
171
+ }
172
+
173
+ const report = []
174
+ const FROM = argOf('--from', null)
175
+ if (FROM) {
176
+ // 复用既有证据(探一次、写多次):不再打上游,只做 emit/write。
177
+ const ev = JSON.parse(readFileSync(FROM, 'utf8'))
178
+ report.push(...(ev.models ?? []))
179
+ console.log(`复用证据 ${FROM}(${report.length} 个模型,本次不打上游)`)
180
+ } else {
181
+ const ARM_DEFS = LEVELS_AS_ARMS
182
+ ? [
183
+ ['baseline', undefined],
184
+ ...LEVELS.map((l) => {
185
+ const budget = LEVEL_BUDGET[l]
186
+ if (budget === undefined) console.error(`警告:档位 ${l} 不在 pi-ai 映射表(low/medium/high/max)内,budget 回落 2048`)
187
+ return [l, { type: 'enabled', budget_tokens: budget ?? 2048 }]
188
+ }),
189
+ ...(FULL_MATRIX ? [['disabled', { type: 'disabled' }]] : []),
190
+ ]
191
+ : FULL_MATRIX
192
+ ? [
193
+ ['baseline', undefined],
194
+ ['enabled', { type: 'enabled', budget_tokens: 2048 }],
195
+ ['disabled', { type: 'disabled' }],
196
+ ]
197
+ : [
198
+ ['baseline', undefined],
199
+ ['enabled', { type: 'enabled', budget_tokens: 2048 }],
200
+ ]
201
+ const armMaxTokens = LEVELS_AS_ARMS ? LEVELS_AS_ARMS_MAX_TOKENS : 2048
202
+ console.log(
203
+ `${MODE_LABEL}:${MODELS.length} 模型 × ${ARM_DEFS.length} 臂 × ${REPEAT} repeat = ${MODELS.length * ARM_DEFS.length * REPEAT} 次上游调用` +
204
+ (LEVELS_AS_ARMS
205
+ ? `(档位臂 budget=pi-ai 映射 low ${LEVEL_BUDGET.low}/medium ${LEVEL_BUDGET.medium}/high ${LEVEL_BUDGET.high}/max ${LEVEL_BUDGET.max},max_tokens=${armMaxTokens}=真实宿主 defaultMaxTokens)`
206
+ : FULL_MATRIX
207
+ ? ''
208
+ : '(disabled 臂不跑;需要 off 结论请加 --full-matrix)'),
209
+ )
210
+ for (const model of MODELS) {
211
+ const arms = {}
212
+ for (const [name, thinking] of ARM_DEFS) {
213
+ const samples = []
214
+ for (let i = 0; i < REPEAT; i++) samples.push(await call(model, thinking, armMaxTokens))
215
+ arms[name] = aggregateArm(samples)
216
+ }
217
+ const { baseline, disabled } = arms
218
+ // levels-as-arms 下没有单一 enabled 臂:每个档位臂都要 200 才出档。
219
+ const levelArms = LEVELS_AS_ARMS ? LEVELS.map((l) => [l, arms[l]]) : null
220
+ const enabledArms = levelArms ? levelArms.map(([, a]) => a) : [arms.enabled]
221
+ const thinksByDefault = baseline.samples.some((s) => s.thinkingChars > 0 || s.blockTypes.includes('thinking'))
222
+ const acceptsEnabled = enabledArms.every((a) => a.samples.every((s) => s.status === 200))
223
+ // off 只在 full-matrix 实测 disabled 后才能下结论;省额度模式记 null(未测),不臆造(踩坑 #42)。
224
+ const offAccepted = disabled
225
+ ? disabled.samples.every((s) => s.status === 200) && Math.max(...disabled.samples.map((s) => s.thinkingChars)) === 0
226
+ : null
227
+ // 档位表:接受 enabled 才谈档位;off 只在 disabled 实测被接受时给(未测不给)。
228
+ const table = acceptsEnabled
229
+ ? { ...(offAccepted ? { off: null } : {}), ...Object.fromEntries(LEVELS.map((l) => [l, l])) }
230
+ : null
231
+ report.push({ model, mode: MODE, thinksByDefault, acceptsEnabled, offAccepted, table, arms })
232
+ const verdict = !acceptsEnabled
233
+ ? LEVELS_AS_ARMS
234
+ ? 'ARM-REJECTED(某档位臂非 200,不出档)'
235
+ : 'ENABLED-REJECTED(不出档)'
236
+ : offAccepted === null
237
+ ? 'OK(off 未测:未跑 disabled 臂,--full-matrix 可证 off)'
238
+ : offAccepted ? 'OK(含 off)' : 'OK(无 off:上游拒 disabled)'
239
+ console.log(`\n### ${model} → ${verdict}`)
240
+ console.log(` baseline : ${fmtArm(baseline)}`)
241
+ if (levelArms) for (const [l, a] of levelArms) console.log(` ${l.padEnd(8)} : ${fmtArm(a)}`)
242
+ else console.log(` enabled : ${fmtArm(arms.enabled)}`)
243
+ if (disabled) console.log(` disabled : ${fmtArm(disabled)}`)
244
+ console.log(` → reasoningEfforts: ${table ? JSON.stringify(table) : '(不出)'}`)
245
+ }
246
+ }
247
+
248
+ if (EMIT) {
249
+ console.log('\n--- 可粘贴的声明(按模型填进 profile patch 的对应条目)---')
250
+ for (const r of report) {
251
+ if (!r.table) { console.log(`# ${r.model}: 不出档(上游拒 enabled)`); continue }
252
+ console.log(`# ${r.model}${r.offAccepted === null ? '(off 未测:省额度默认不测 disabled;--full-matrix 可证 off)' : ''}`)
253
+ console.log('reasoningEfforts:')
254
+ for (const [k, v] of Object.entries(r.table)) console.log(v === null ? ` ${k}: null` : ` ${k}: ${v}`)
255
+ }
256
+ }
257
+
258
+ if (WRITE) {
259
+ const doc = YAML.parseDocument(readFileSync(WRITE, 'utf8'))
260
+ const entry = (doc.contents?.items ?? []).find((it) => it.get?.('id') === 'llm-pi-ai')
261
+ const providers = entry?.get?.('config')?.get?.('providers')
262
+ const volces = providers?.get?.('volces')
263
+ const models = volces?.get?.('models')
264
+ if (!models?.items) {
265
+ console.error(`写入失败:${WRITE} 里找不到 llm-pi-ai.config.providers.volces.models`)
266
+ process.exit(1)
267
+ }
268
+ let changed = 0
269
+ for (const r of report) {
270
+ if (!r.table) continue
271
+ const item = models.items.find((m) => m.get?.('id') === r.model)
272
+ if (!item) { console.log(` 跳过 ${r.model}(patch 里没有这个模型条目)`); continue }
273
+ let table = r.table
274
+ if (r.offAccepted === null) {
275
+ // off 未测(省额度默认):patch 里已有 off 档(此前 --full-matrix 实测写入)则保留,不臆造也不抹掉。
276
+ // 注意 yaml 的 map.get() 对 `off: null` 返回 undefined(`?? undefined` 把 null 归并了),
277
+ // 不能用 get 读;要从 Pair 原始节点读值。
278
+ const prev = item.get?.('reasoningEfforts')
279
+ const offPair = prev?.items?.find?.((it) => (isScalar(it.key) ? it.key.value : it.key) === 'off')
280
+ if (offPair) {
281
+ table = { off: isScalar(offPair.value) ? offPair.value.value : offPair.value, ...table }
282
+ console.log(` ${r.model}: 保留既有 off 档(本次未测 disabled 臂)`)
283
+ }
284
+ }
285
+ item.set('reasoningEfforts', table)
286
+ changed += 1
287
+ }
288
+ if (!changed) { console.error('没有任何条目被更新(模型 id 对不上?)'); process.exit(1) }
289
+ copyFileSync(WRITE, `${WRITE}.pre-ark-probe.bak`)
290
+ writeFileSync(WRITE, String(doc))
291
+ console.log(`\n已写入 ${changed} 个模型的 reasoningEfforts → ${WRITE}(备份 ${WRITE}.pre-ark-probe.bak)`)
292
+ console.log('生效:重启 dsh / 桌面应用(profile patch 属启动期加载)。')
293
+ }
294
+
295
+ mkdirSync('docs/probes', { recursive: true })
296
+ const out = `docs/probes/ark-thinking-${Date.now()}.json`
297
+ writeFileSync(out, JSON.stringify({ at: new Date().toISOString(), endpoint: ENDPOINT, mode: MODE, repeat: REPEAT, levels: LEVELS, ...(LEVELS_AS_ARMS ? { levelBudget: LEVEL_BUDGET, maxTokens: LEVELS_AS_ARMS_MAX_TOKENS } : {}), prompt: PROMPT, models: report }, null, 2))
298
+ console.log('\n证据 →', out)
@@ -0,0 +1,292 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Cache-decline probe (docs/diagnosis-cache-decline.md, 2026-09-03).
4
+ *
5
+ * Question: why does deepseek-v4-flash prefix-cache hit rate decline much
6
+ * faster through the dsh-tap plugin (CodeBuddy gateway) than through direct
7
+ * dsh (DeepSeek official API)? This probe isolates the GATEWAY-side variables
8
+ * under controlled payloads (max_tokens=16, ≤1 in flight):
9
+ *
10
+ * --mode whoami account identity behind the ck_ key vs the OAuth token
11
+ * (are the two credentials even the same account?)
12
+ * --mode burst same-prompt re-fires: read stability of one entry.
13
+ * --auth ck|oauth|ab interleaves the two credentials (ABAB)
14
+ * — credential-class routing is the confound this settles.
15
+ * --mode ttl seed once, re-send at cumulative --gaps (entry unrefreshed
16
+ * between fires) → time dimension of the decay curve.
17
+ * --mode grow append-only growing prefix (mimics dsh tool-loop steps),
18
+ * short gaps → write+read stability within a "turn";
19
+ * --idle-after/--idle-grow adds one grown send after an idle
20
+ * gap (mimics a turn start after idle).
21
+ * --mode replay re-fire one REAL request body (--payload-file, a bridge
22
+ * dump; the bridge's developer→system + stream transform is
23
+ * applied) — same-prompt stability for the real dsh shape.
24
+ *
25
+ * Every arm embeds a unique nonce so entries never collide across runs.
26
+ * Evidence → docs/probes/cache-decline-2026-09-03.jsonl (append).
27
+ *
28
+ * Usage:
29
+ * node scripts/probe-cache-decline.mjs --mode whoami
30
+ * node scripts/probe-cache-decline.mjs --mode burst --auth ab --size 16000 --calls 12
31
+ * node scripts/probe-cache-decline.mjs --mode ttl --auth oauth --size 16000 --gaps 30,120,600
32
+ * node scripts/probe-cache-decline.mjs --mode ttl --auth oauth --size 32000 --gaps 30,120,600
33
+ * node scripts/probe-cache-decline.mjs --mode grow --auth oauth --start 16000 --step 1500 --steps 8 --idle-after 60 --idle-grow 1500
34
+ */
35
+
36
+ import { readFileSync, existsSync, writeFileSync, appendFileSync } from 'node:fs'
37
+ import { homedir } from 'node:os'
38
+ import { join } from 'node:path'
39
+
40
+ const DSH_HOME = process.env.DSH_HOME ?? join(homedir(), '.dsh')
41
+ const GATEWAY = 'https://copilot.tencent.com'
42
+ const argValue = (flag) => {
43
+ const i = process.argv.indexOf(flag)
44
+ return i > 0 ? process.argv[i + 1] : null
45
+ }
46
+ const MODE = argValue('--mode') ?? 'burst'
47
+ const AUTH = argValue('--auth') ?? 'ck'
48
+ const MODEL = argValue('--model') ?? 'deepseek-v4-flash'
49
+ const SIZE = Number(argValue('--size') ?? 16000)
50
+ const CALLS = Number(argValue('--calls') ?? 12)
51
+ const GAPS = (argValue('--gaps') ?? '30,120,600').split(',').map(Number)
52
+ const START = Number(argValue('--start') ?? 16000)
53
+ const STEP = Number(argValue('--step') ?? 1500)
54
+ const STEPS = Number(argValue('--steps') ?? 8)
55
+ const GAP_MS = Number(argValue('--gap-ms') ?? 2000)
56
+ const IDLE_AFTER = Number(argValue('--idle-after') ?? 0)
57
+ const IDLE_GROW = Number(argValue('--idle-grow') ?? 0)
58
+ const OUT = join('docs', 'probes', 'cache-decline-2026-09-03.jsonl')
59
+
60
+ const runId = Math.random().toString(36).slice(2, 8)
61
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
62
+ // calibration from probe-cache-ttl.mjs: 72 repeats ≈ 2.6k tokens
63
+ const TOK_PER_REPEAT = 36.1
64
+
65
+ // ---------------------------------------------------------------- credentials
66
+ function resolveCk() {
67
+ try {
68
+ const cfg = JSON.parse(readFileSync(join(DSH_HOME, 'codebuddy-plugin.json'), 'utf8'))
69
+ const active = (cfg.apiKeys ?? []).find((k) => k.name === cfg.activeApiKey)
70
+ if (active?.key) return active.key
71
+ } catch { /* fall through */ }
72
+ if (process.env.CODEBUDDY_API_KEY) return process.env.CODEBUDDY_API_KEY
73
+ const credFile = join(DSH_HOME, '.credentials.yaml')
74
+ if (existsSync(credFile)) {
75
+ const m = readFileSync(credFile, 'utf8').match(/^\s*CODEBUDDY_API_KEY:\s*["']?([^"'\s]+)["']?\s*$/m)
76
+ if (m) return m[1]
77
+ }
78
+ throw new Error('no CodeBuddy ck_ credential found')
79
+ }
80
+
81
+ const AUTH_PATH = join(DSH_HOME, 'codebuddy-plugin-auth.json')
82
+
83
+ /** OAuth access token, refreshing via the documented endpoint when near expiry. */
84
+ async function resolveOauth() {
85
+ const file = JSON.parse(readFileSync(AUTH_PATH, 'utf8'))
86
+ const auth = file.auth ?? file // live layout nests tokens under `auth`
87
+ if (!auth?.accessToken) throw new Error('no OAuth token in codebuddy-plugin-auth.json')
88
+ if (!auth.expiresAt || auth.expiresAt - Date.now() > 120_000) return auth.accessToken
89
+ const res = await fetch(`${GATEWAY}/v2/plugin/auth/token/refresh`, {
90
+ method: 'POST',
91
+ headers: {
92
+ 'Content-Type': 'application/json',
93
+ Authorization: `Bearer ${auth.accessToken}`,
94
+ 'X-Refresh-Token': auth.refreshToken ?? '',
95
+ },
96
+ })
97
+ const body = await res.json().catch(() => null)
98
+ if (!body || body.code !== 0 || !body.data?.accessToken) {
99
+ throw new Error(`oauth refresh failed: http=${res.status}`)
100
+ }
101
+ writeFileSync(AUTH_PATH, JSON.stringify({
102
+ ...file,
103
+ auth: {
104
+ ...auth,
105
+ accessToken: body.data.accessToken,
106
+ expiresAt: Date.now() + (body.data.expiresIn ?? 3600) * 1000,
107
+ refreshToken: body.data.refreshToken ?? auth.refreshToken,
108
+ refreshExpiresAt: body.data.refreshExpiresAt != null
109
+ ? Date.now() + body.data.refreshExpiresAt * 1000
110
+ : auth.refreshExpiresAt,
111
+ },
112
+ }))
113
+ return body.data.accessToken
114
+ }
115
+
116
+ // ---------------------------------------------------------------- payload
117
+ const PARAGRAPH = 'Cache decline probe paragraph. The quick brown fox jumps over the lazy dog near the riverbank while engineers measure prefix-cache retention across identical and growing requests under controlled budgets. '
118
+ const REPEAT = Math.max(1, Math.round(SIZE / TOK_PER_REPEAT))
119
+ const START_REPEAT = Math.max(1, Math.round(START / TOK_PER_REPEAT))
120
+ const STEP_REPEAT = Math.max(1, Math.round(STEP / TOK_PER_REPEAT))
121
+
122
+ /** Fixed-shape payload: big system + probe user (burst/ttl). */
123
+ function fixedPayload(nonce) {
124
+ return JSON.stringify({
125
+ model: MODEL,
126
+ stream: true,
127
+ stream_options: { include_usage: true },
128
+ max_tokens: 16,
129
+ messages: [
130
+ { role: 'system', content: `${nonce}\n` + PARAGRAPH.repeat(REPEAT) },
131
+ { role: 'user', content: `Probe ${nonce}: reply with exactly: OK` },
132
+ ],
133
+ })
134
+ }
135
+
136
+ /** Growing payload: fixed system head + k synthetic user blocks + probe tail. */
137
+ function growPayload(nonce, k) {
138
+ const blocks = []
139
+ for (let i = 1; i <= k; i++) {
140
+ blocks.push({ role: 'user', content: `Block ${nonce}-${i}: ` + PARAGRAPH.repeat(STEP_REPEAT) })
141
+ }
142
+ return JSON.stringify({
143
+ model: MODEL,
144
+ stream: true,
145
+ stream_options: { include_usage: true },
146
+ max_tokens: 16,
147
+ messages: [
148
+ { role: 'system', content: `${nonce} fixed system head.\n` + PARAGRAPH.repeat(START_REPEAT) },
149
+ ...blocks,
150
+ { role: 'user', content: `Probe ${nonce} step ${k}: reply with exactly: OK` },
151
+ ],
152
+ })
153
+ }
154
+
155
+ // ---------------------------------------------------------------- one call
156
+ async function call(token, body) {
157
+ const t0 = Date.now()
158
+ let ttfbMs = null
159
+ try {
160
+ const res = await fetch(`${GATEWAY}/v2/chat/completions`, {
161
+ method: 'POST',
162
+ headers: {
163
+ 'Content-Type': 'application/json',
164
+ Authorization: `Bearer ${token}`,
165
+ 'User-Agent': 'CLI/unknown CodeBuddy/2.136.0',
166
+ },
167
+ body,
168
+ })
169
+ const reader = res.body.getReader()
170
+ const decoder = new TextDecoder()
171
+ let buf = ''
172
+ let usage = null
173
+ let sawDone = false
174
+ for (;;) {
175
+ const { done, value } = await reader.read()
176
+ if (done) break
177
+ if (ttfbMs === null) ttfbMs = Date.now() - t0
178
+ buf += decoder.decode(value, { stream: true })
179
+ let nl
180
+ while ((nl = buf.indexOf('\n')) >= 0) {
181
+ const line = buf.slice(0, nl).trim()
182
+ buf = buf.slice(nl + 1)
183
+ if (line === 'data: [DONE]') { sawDone = true; continue }
184
+ if (!line.startsWith('data:') || !line.includes('"usage"')) continue
185
+ try {
186
+ const chunk = JSON.parse(line.slice(5).trim())
187
+ if (chunk.usage) usage = chunk.usage
188
+ } catch { /* skip */ }
189
+ }
190
+ }
191
+ return { status: res.status, ttfbMs, ms: Date.now() - t0, usage, complete: sawDone || usage != null }
192
+ } catch (err) {
193
+ return { status: 0, ttfbMs, ms: Date.now() - t0, error: String(err?.message ?? err), complete: false }
194
+ }
195
+ }
196
+
197
+ function report(rec) {
198
+ const u = rec.usage ?? {}
199
+ const prompt = u.prompt_tokens ?? 0
200
+ const hit = u.prompt_cache_hit_tokens ?? 0
201
+ const rate = prompt > 0 ? (100 * hit / prompt).toFixed(1) + '%' : '-'
202
+ appendFileSync(OUT, JSON.stringify({ type: 'probe', runId, ...rec }) + '\n')
203
+ console.log(
204
+ `${(rec.mode || '').padEnd(6)} ${(rec.auth ?? '').padEnd(5)} ${String(rec.tag ?? '').padEnd(12)} http=${rec.status}`
205
+ + ` prompt=${u.prompt_tokens ?? '-'} hit=${hit} miss=${u.prompt_cache_miss_tokens ?? '-'} rate=${rate}`
206
+ + ` credit=${u.credit ?? '-'} ttfb=${rec.ttfbMs ?? '-'}ms${rec.complete === false ? ' INCOMPLETE' : ''}`,
207
+ )
208
+ }
209
+
210
+ const ck = resolveCk()
211
+ const oauth = MODE === 'whoami' || AUTH === 'oauth' || AUTH === 'ab' ? await resolveOauth() : null
212
+ appendFileSync(OUT, JSON.stringify({ type: 'run', at: new Date().toISOString(), mode: MODE, runId, model: MODEL, args: process.argv.slice(2) }) + '\n')
213
+ console.log(`evidence → ${OUT} (run ${runId}, mode=${MODE})`)
214
+
215
+ if (MODE === 'whoami') {
216
+ for (const [name, token] of [['ck', ck], ['oauth', oauth]]) {
217
+ try {
218
+ const res = await fetch(`${GATEWAY}/v2/accounts`, {
219
+ headers: { Authorization: `Bearer ${token}`, 'User-Agent': 'CLI/unknown CodeBuddy/2.136.0' },
220
+ })
221
+ const body = await res.json().catch(() => null)
222
+ const accounts = body?.data?.accounts ?? body?.data ?? []
223
+ const summary = JSON.stringify(accounts).replace(/[a-f0-9-]{16,}/g, (m) => m.slice(0, 4) + '…')
224
+ appendFileSync(OUT, JSON.stringify({ type: 'whoami', runId, auth: name, status: res.status, body: JSON.parse(summary) }) + '\n')
225
+ console.log(name.padEnd(5), 'http=' + res.status, summary.slice(0, 400))
226
+ } catch (err) {
227
+ console.log(name, 'ERR', err.message)
228
+ }
229
+ }
230
+ } else if (MODE === 'burst') {
231
+ const arms = AUTH === 'ab' ? [['ck', ck], ['oauth', oauth]] : [[AUTH, AUTH === 'oauth' ? oauth : ck]]
232
+ // arm-unique nonces → independent entries; ABAB interleave controls for time
233
+ const nonceOf = Object.fromEntries(arms.map(([name]) => [name, `${runId}-${name}`]))
234
+ const bodyOf = (name) => fixedPayload(nonceOf[name])
235
+ // warm seed: one call per arm first so every measured call reads a live entry
236
+ for (const [name, token] of arms) {
237
+ report({ mode: 'burst', auth: name, tag: 'seed', ...await call(token, bodyOf(name)) })
238
+ await sleep(GAP_MS)
239
+ }
240
+ for (let i = 1; i <= CALLS; i++) {
241
+ for (const [name, token] of arms) {
242
+ report({ mode: 'burst', auth: name, tag: `fire${i}`, ...await call(token, bodyOf(name)) })
243
+ await sleep(GAP_MS)
244
+ }
245
+ }
246
+ } else if (MODE === 'ttl') {
247
+ const token = AUTH === 'oauth' ? oauth : ck
248
+ const nonce = `${runId}-ttl`
249
+ report({ mode: 'ttl', auth: AUTH, tag: 'seed', gapS: 0, ...await call(token, fixedPayload(nonce)) })
250
+ let prev = 0
251
+ for (const gap of GAPS) {
252
+ await sleep((gap - prev) * 1000)
253
+ prev = gap
254
+ report({ mode: 'ttl', auth: AUTH, tag: `age${gap}s`, gapS: gap, ...await call(token, fixedPayload(nonce)) })
255
+ await sleep(GAP_MS)
256
+ }
257
+ } else if (MODE === 'replay') {
258
+ // re-fire one REAL request body (a bridge dump, post developer→system
259
+ // transform as the bridge would send it) — tests same-prompt read
260
+ // stability for the real dsh shape (tools array, real content).
261
+ const payloadFile = argValue('--payload-file')
262
+ if (!payloadFile) throw new Error('--mode replay needs --payload-file')
263
+ const parsed = JSON.parse(readFileSync(payloadFile, 'utf8'))
264
+ parsed.stream = true
265
+ for (const m of parsed.messages ?? []) if (m?.role === 'developer') m.role = 'system'
266
+ const body = JSON.stringify(parsed)
267
+ const arms = AUTH === 'ab' ? [['ck', ck], ['oauth', oauth]] : [[AUTH, AUTH === 'oauth' ? oauth : ck]]
268
+ for (const [name, token] of arms) {
269
+ report({ mode: 'replay', auth: name, tag: 'seed', ...await call(token, body) })
270
+ await sleep(GAP_MS)
271
+ }
272
+ for (let i = 1; i <= CALLS; i++) {
273
+ for (const [name, token] of arms) {
274
+ report({ mode: 'replay', auth: name, tag: `fire${i}`, ...await call(token, body) })
275
+ await sleep(GAP_MS)
276
+ }
277
+ }
278
+ } else if (MODE === 'grow') {
279
+ const token = AUTH === 'oauth' ? oauth : ck
280
+ const nonce = `${runId}-grow`
281
+ for (let k = 1; k <= STEPS; k++) {
282
+ report({ mode: 'grow', auth: AUTH, tag: `step${k}`, ...await call(token, growPayload(nonce, k)) })
283
+ await sleep(GAP_MS)
284
+ }
285
+ if (IDLE_AFTER > 0 && IDLE_GROW > 0) {
286
+ await sleep(IDLE_AFTER * 1000)
287
+ const k = STEPS + 1
288
+ // one more block than the ladder covers → grown prefix after idle
289
+ report({ mode: 'grow', auth: AUTH, tag: `idle${IDLE_AFTER}s-grown`, ...await call(token, growPayload(nonce, k)) })
290
+ }
291
+ }
292
+ console.log('done.')