@ljwei-stak/dsh-model-router 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dsh-plugin/client.js +3092 -0
- package/.dsh-plugin/index.mjs +1651 -0
- package/.dsh-plugin/official-tools-remote-service.mjs +104 -0
- package/.dsh-plugin/shared/harness-plan.mjs +179 -0
- package/.dsh-plugin/shared/livebench.mjs +264 -0
- package/.dsh-plugin/shared/model-profiles.mjs +142 -0
- package/.dsh-plugin/shared/official-team-runtime.mjs +411 -0
- package/.dsh-plugin/shared/official-tool-executor.mjs +801 -0
- package/.dsh-plugin/shared/official-tool-registry.mjs +138 -0
- package/.dsh-plugin/shared/official-tools-remote.mjs +173 -0
- package/.dsh-plugin/shared/official-tools-runtime.mjs +642 -0
- package/.dsh-plugin/shared/router-state.mjs +207 -0
- package/.dsh-plugin/shared/router.mjs +1134 -0
- package/.dsh-plugin/shared/routing-presets.mjs +49 -0
- package/.dsh-plugin/shared/run-ledger.mjs +348 -0
- package/.dsh-plugin/shared/security-boundaries.mjs +54 -0
- package/.dsh-plugin/shared/subscription-billing.mjs +340 -0
- package/.dsh-plugin/shared/task-executors.mjs +1154 -0
- package/.dsh-plugin/shared/tool-health.mjs +311 -0
- package/.dsh-plugin/shared/vendor-mimo-grok-adapter.mjs +308 -0
- package/.dsh-plugin/shared/vendor-minimax-adapter.mjs +247 -0
- package/.dsh-plugin/shared/zcode-bundle.mjs +208 -0
- package/.dsh-plugin/shared/zcode-installer.mjs +247 -0
- package/CHANGELOG.md +36 -0
- package/INSTALLATION_GUIDE.zh.md +134 -0
- package/LICENSE +21 -0
- package/MIGRATION.md +53 -0
- package/README.i18n.yaml +3 -0
- package/README.md +424 -0
- package/README.zh.md +413 -0
- package/cordis.patch.yml +12 -0
- package/docs/assets/candidate-pruning.svg +80 -0
- package/docs/assets/desktop-official-tools-0.9.0.png +0 -0
- package/docs/assets/router-only-0.12.0.png +0 -0
- package/docs/assets/routing-workflow.svg +96 -0
- package/docs/assets/workbench-usage.svg +119 -0
- package/package.json +161 -0
|
@@ -0,0 +1,1134 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared, deterministic routing model used by the Host and GAL client.
|
|
3
|
+
* It intentionally exposes an auditable decision record, not private model
|
|
4
|
+
* reasoning. The optimizer is a constrained assignment heuristic over a small
|
|
5
|
+
* task DAG; this keeps plan generation bounded and reproducible in a desktop
|
|
6
|
+
* process while retaining the same objective used in the thesis model.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { DEFAULT_ROUTING_PRESET, normalizeRoutingPreset, presetFloor, presetWeights } from './routing-presets.mjs'
|
|
10
|
+
import { liveBenchRow } from './livebench.mjs'
|
|
11
|
+
|
|
12
|
+
export const OBJECTIVE_WEIGHTS = Object.freeze({
|
|
13
|
+
simple: Object.freeze({ quality: 0.28, cost: 0.45, latency: 0.14, specialty: 0.04, reasoning: 0.07, risk: 0.02 }),
|
|
14
|
+
balanced: Object.freeze({ quality: 0.40, cost: 0.26, latency: 0.10, specialty: 0.09, reasoning: 0.10, risk: 0.05 }),
|
|
15
|
+
complex: Object.freeze({ quality: 0.48, cost: 0.14, latency: 0.06, specialty: 0.14, reasoning: 0.11, risk: 0.07 }),
|
|
16
|
+
})
|
|
17
|
+
|
|
18
|
+
/** Canonical order used only to compare adapter-owned opaque effort ids. */
|
|
19
|
+
export const REASONING_EFFORT_ORDER = Object.freeze(['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max'])
|
|
20
|
+
|
|
21
|
+
const REASONING_EFFORT_MULTIPLIERS = Object.freeze({
|
|
22
|
+
off: Object.freeze({ output: 0.78, latency: 0.75 }),
|
|
23
|
+
minimal: Object.freeze({ output: 0.86, latency: 0.82 }),
|
|
24
|
+
low: Object.freeze({ output: 0.93, latency: 0.90 }),
|
|
25
|
+
medium: Object.freeze({ output: 1, latency: 1 }),
|
|
26
|
+
high: Object.freeze({ output: 1.16, latency: 1.15 }),
|
|
27
|
+
xhigh: Object.freeze({ output: 1.34, latency: 1.30 }),
|
|
28
|
+
max: Object.freeze({ output: 1.58, latency: 1.50 }),
|
|
29
|
+
})
|
|
30
|
+
|
|
31
|
+
/** Quality floor for a task node before a cost-saving substitution is allowed. */
|
|
32
|
+
export const QUALITY_FLOORS = Object.freeze({ simple: 0.75, balanced: 0.78, complex: 0.82 })
|
|
33
|
+
|
|
34
|
+
/** Host settings namespace used by the manual pricing editor. */
|
|
35
|
+
export const MODEL_ROUTER_SETTINGS_NAMESPACE = 'model-router'
|
|
36
|
+
|
|
37
|
+
/** Empty user layer means "use the experimental baseline". */
|
|
38
|
+
export const DEFAULT_ROUTER_SETTINGS = Object.freeze({
|
|
39
|
+
pricing: Object.freeze({}),
|
|
40
|
+
// The official site publishes versioned table/categories assets and the
|
|
41
|
+
// adapter discovers the newest release from this root URL.
|
|
42
|
+
liveBenchEndpoint: 'https://livebench.ai',
|
|
43
|
+
liveBenchTtlMs: 900000,
|
|
44
|
+
budgetUsd: 0,
|
|
45
|
+
cacheReadRatio: 0,
|
|
46
|
+
cacheWriteRatio: 0,
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
// USD per million tokens. Values are an initial catalog and can be replaced by
|
|
50
|
+
// a provider's live pricing without changing the scoring code.
|
|
51
|
+
export const MODEL_CATALOG = Object.freeze([
|
|
52
|
+
{ id: 'claude-fable-5', aliases: ['claude-fable-5', 'claude fable 5'], quality: 0.99, latency: 0.34, costIn: 10, costOut: 50, specialties: ['reasoning', 'writing', 'research'], risk: 0.06 },
|
|
53
|
+
{ id: 'claude-opus-4-8', aliases: ['claude-opus-4-8', 'claude opus 4.8'], quality: 0.97, latency: 0.39, costIn: 5, costOut: 25, specialties: ['reasoning', 'writing', 'code'], risk: 0.07 },
|
|
54
|
+
{ id: 'gpt-5.6-sol', aliases: ['gpt-5.6-sol', 'gpt 5.6 sol'], quality: 0.98, latency: 0.40, costIn: 5, costOut: 30, specialties: ['reasoning', 'code', 'math', 'vision'], risk: 0.06 },
|
|
55
|
+
{ id: 'gpt-5.5', aliases: ['gpt-5.5', 'gpt 5.5'], quality: 0.95, latency: 0.44, costIn: 5, costOut: 30, specialties: ['reasoning', 'code', 'math'], risk: 0.08 },
|
|
56
|
+
{ id: 'deepseek-v4-pro', aliases: ['deepseek-v4-pro', 'deepseek v4 pro'], quality: 0.93, latency: 0.52, costIn: 1.74, costOut: 3.48, specialties: ['code', 'math', 'reasoning'], risk: 0.10 },
|
|
57
|
+
{ id: 'deepseek-v4-flash', aliases: ['deepseek-v4-flash', 'deepseek v4 flash'], quality: 0.82, latency: 0.82, costIn: 0.14, costOut: 0.28, specialties: ['code', 'summarization', 'classification'], risk: 0.14 },
|
|
58
|
+
{ id: 'kimi-k3', aliases: ['kimi-k3', 'kimi k3'], quality: 0.91, latency: 0.56, costIn: 3, costOut: 15, specialties: ['reasoning', 'long-context', 'code'], risk: 0.10 },
|
|
59
|
+
{ id: 'qwen3.7-max', aliases: ['qwen3.7-max', 'qwen 3.7 max'], quality: 0.94, latency: 0.50, costIn: 2.5, costOut: 7.5, specialties: ['reasoning', 'math', 'code'], risk: 0.08 },
|
|
60
|
+
{ id: 'qwen3.7-plus', aliases: ['qwen3.7-plus', 'qwen 3.7 plus'], quality: 0.87, latency: 0.72, costIn: 0.4, costOut: 1.6, specialties: ['code', 'math', 'writing'], risk: 0.12 },
|
|
61
|
+
{ id: 'glm-5.2', aliases: ['glm-5.2', 'glm 5.2'], quality: 0.89, latency: 0.64, costIn: 1.4, costOut: 4.4, specialties: ['reasoning', 'writing', 'math'], risk: 0.11 },
|
|
62
|
+
{ id: 'gpt-5.6-luna', aliases: ['gpt-5.6-luna', 'gpt 5.6 luna'], quality: 0.84, latency: 0.86, costIn: 0.2, costOut: 1.2, specialties: ['classification', 'summarization', 'code'], risk: 0.14 },
|
|
63
|
+
{ id: 'gpt-5.6-terra', aliases: ['gpt-5.6-terra', 'gpt 5.6 terra'], quality: 0.91, latency: 0.66, costIn: 2, costOut: 12, specialties: ['code', 'writing', 'reasoning'], risk: 0.10 },
|
|
64
|
+
{ id: 'minimax-m3', aliases: ['minimax-m3', 'minimax m3'], quality: 0.86, latency: 0.69, costIn: 0.3, costOut: 1.2, specialties: ['writing', 'code', 'summarization'], risk: 0.13 },
|
|
65
|
+
{ id: 'gemini-3-flash', aliases: ['gemini 3 flash', 'gemini-3-flash'], quality: 0.88, latency: 0.73, costIn: 0.5, costOut: 3, specialties: ['vision', 'research', 'summarization'], risk: 0.12 },
|
|
66
|
+
{ id: 'big-pickle', aliases: ['big pickle'], quality: 0.70, latency: 0.88, costIn: 0, costOut: 0, specialties: ['classification', 'summarization'], risk: 0.24 },
|
|
67
|
+
])
|
|
68
|
+
|
|
69
|
+
const clamp = (value, min = 0, max = 1) => Math.max(min, Math.min(max, value))
|
|
70
|
+
const normalize = value => String(value ?? '').toLowerCase().replace(/[^a-z0-9]+/g, '')
|
|
71
|
+
const routeKey = (provider, model) => `${String(provider ?? '')}/${String(model ?? '')}`
|
|
72
|
+
|
|
73
|
+
/** OpenCode routes whose catalog entries carry their own protocol endpoint. */
|
|
74
|
+
export const OPENCODE_CATALOG_PROVIDERS = Object.freeze([
|
|
75
|
+
'opencode',
|
|
76
|
+
'opencode-go',
|
|
77
|
+
// Some OpenCode-compatible configuration examples use the product name as
|
|
78
|
+
// the route id. Treat those aliases as catalog routes too; the pi-ai catalog
|
|
79
|
+
// still owns the actual model endpoints.
|
|
80
|
+
'opencode-zen',
|
|
81
|
+
'opencode-go-zen',
|
|
82
|
+
])
|
|
83
|
+
|
|
84
|
+
/** Normalize the route ids used by OpenCode-compatible settings. */
|
|
85
|
+
function normalizeOpenCodeProvider(provider) {
|
|
86
|
+
const route = String(provider ?? '').trim().toLowerCase()
|
|
87
|
+
return route.replace(/-zen$/, '')
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* The pi-ai catalog stores different endpoints for OpenCode's wire families:
|
|
92
|
+
* Anthropic models use /zen while OpenAI-compatible models use /zen/v1 (and
|
|
93
|
+
* the Go route has the corresponding /zen/go variants). A provider-level URL
|
|
94
|
+
* for the public website would overwrite those model endpoints and produce a
|
|
95
|
+
* 404 HTML page. Only the official host is repaired; custom gateways remain
|
|
96
|
+
* fully user-controlled.
|
|
97
|
+
*/
|
|
98
|
+
export function isOfficialOpenCodeEndpoint(provider, baseURL) {
|
|
99
|
+
const route = String(provider ?? '').trim().toLowerCase()
|
|
100
|
+
if (!['opencode', 'opencode-go'].includes(normalizeOpenCodeProvider(route))) return false
|
|
101
|
+
if (typeof baseURL !== 'string' || baseURL.trim().length === 0) return false
|
|
102
|
+
try {
|
|
103
|
+
const parsed = new URL(baseURL.trim())
|
|
104
|
+
if (parsed.protocol !== 'https:') return false
|
|
105
|
+
const host = parsed.hostname.toLowerCase()
|
|
106
|
+
return host === 'opencode.ai' || host === 'www.opencode.ai'
|
|
107
|
+
} catch {
|
|
108
|
+
return false
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** Return settings mutations that restore the official catalog endpoints. */
|
|
113
|
+
export function collectOpenCodeEndpointRepairs(user) {
|
|
114
|
+
if (user === null || typeof user !== 'object' || Array.isArray(user)) return []
|
|
115
|
+
const providers = user.providers
|
|
116
|
+
if (providers === null || typeof providers !== 'object' || Array.isArray(providers)) return []
|
|
117
|
+
const ops = []
|
|
118
|
+
for (const [provider, profile] of Object.entries(providers)) {
|
|
119
|
+
if (profile === null || typeof profile !== 'object' || Array.isArray(profile)) continue
|
|
120
|
+
if (isOfficialOpenCodeEndpoint(provider, profile.baseURL)) {
|
|
121
|
+
ops.push({ op: 'unset', path: ['providers', provider, 'baseURL'] })
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
return ops
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
export function textFromMessages(messages) {
|
|
128
|
+
if (!Array.isArray(messages)) return ''
|
|
129
|
+
return messages.map(message => {
|
|
130
|
+
if (!message || !Array.isArray(message.content)) return ''
|
|
131
|
+
return message.content.map(block => typeof block?.text === 'string' ? block.text : '').join('\n')
|
|
132
|
+
}).join('\n').trim()
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
export function classifyTask(text) {
|
|
136
|
+
const value = String(text ?? '')
|
|
137
|
+
if (value.length < 80 && /翻译|解释|translate|explain/i.test(value)) return 'general'
|
|
138
|
+
return detectTaskTypes(value)[0] ?? 'general'
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const TASK_TYPE_RULES = Object.freeze([
|
|
142
|
+
['vision', /图片|图像|照片|视觉|image|vision|截图|识图/i],
|
|
143
|
+
['math', /数学|证明|定理|公式|方程|math|proof|theorem/i],
|
|
144
|
+
['code', /代码|编程|工程|项目|架构|接口|api|debug|实现|部署|测试|code/i],
|
|
145
|
+
['research', /研究|论文|文献|联网|检索|research|source|引用/i],
|
|
146
|
+
['summarization', /总结|摘要|提炼|提取|关键词|分类|翻译|summar|classif|extract/i],
|
|
147
|
+
['writing', /写作|润色|小说|文案|报告|writing|draft/i],
|
|
148
|
+
])
|
|
149
|
+
|
|
150
|
+
const TASK_TYPE_LABELS = Object.freeze({
|
|
151
|
+
vision: '视觉处理',
|
|
152
|
+
math: '数学推导',
|
|
153
|
+
code: '工程与代码',
|
|
154
|
+
research: '研究与检索',
|
|
155
|
+
summarization: '摘要与整理',
|
|
156
|
+
writing: '写作与表达',
|
|
157
|
+
})
|
|
158
|
+
|
|
159
|
+
/** Return all explicit business directions, ranked by signal count. */
|
|
160
|
+
export function detectTaskTypes(text) {
|
|
161
|
+
const value = String(text ?? '')
|
|
162
|
+
const ranked = TASK_TYPE_RULES.map(([type, pattern]) => ({
|
|
163
|
+
type,
|
|
164
|
+
signals: value.match(new RegExp(pattern.source, pattern.flags.includes('g') ? pattern.flags : `${pattern.flags}g`))?.length ?? 0,
|
|
165
|
+
})).filter(item => item.signals > 0)
|
|
166
|
+
ranked.sort((left, right) => right.signals - left.signals || left.type.localeCompare(right.type))
|
|
167
|
+
return ranked.map(item => item.type)
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
export function assessComplexity(text) {
|
|
171
|
+
const value = String(text ?? '')
|
|
172
|
+
const lengthScore = clamp(value.length / 2200)
|
|
173
|
+
const requirementScore = clamp((value.match(/(?:^|\n)\s*(?:[-*]|\d+[.)]|[一二三四五六七八九十]+[、.])/g) ?? []).length / 8)
|
|
174
|
+
const codeScore = /(代码|工程|架构|接口|实现|部署|测试|code|api|debug)/i.test(value) ? 0.22 : 0
|
|
175
|
+
const highReasoningScore = /(数学|证明|定理|研究|论文|复杂|多步骤|约束|比较|评估|架构|模块|部署|math|proof|research)/i.test(value) ? 0.20 : 0
|
|
176
|
+
const visionScore = /(图片|图像|照片|截图|视觉|image|vision)/i.test(value) ? 0.12 : 0
|
|
177
|
+
const domainMarkers = (value.match(/代码|工程|架构|接口|实现|部署|测试|模块|拆分|约束|评估|证明|定理|研究|论文|图片|图像|照片|视觉|code|api|debug|proof|research|vision/gi) ?? []).length
|
|
178
|
+
const domainComplexity = clamp(domainMarkers / 5) * 0.28
|
|
179
|
+
const raw = clamp(0.10 + lengthScore * 0.30 + requirementScore * 0.18 + domainComplexity + codeScore + highReasoningScore + visionScore)
|
|
180
|
+
const band = isHardRequirement(value) ? 'complex'
|
|
181
|
+
: isSimpleRequirement(value) && value.length <= 180 ? 'simple'
|
|
182
|
+
: raw < 0.34 ? 'simple' : raw < 0.66 ? 'balanced' : 'complex'
|
|
183
|
+
return { value: raw, band }
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
function specialtyMatch(model, taskType, liveScores = {}) {
|
|
187
|
+
const benchmark = asScore(liveScores?.[taskType])
|
|
188
|
+
if (benchmark !== undefined) return benchmark
|
|
189
|
+
if (model.specialties.includes(taskType)) return 1
|
|
190
|
+
if (taskType === 'general') return 0.58
|
|
191
|
+
if (taskType === 'research' && model.specialties.includes('writing')) return 0.68
|
|
192
|
+
if (taskType === 'writing' && model.specialties.includes('reasoning')) return 0.62
|
|
193
|
+
return 0.38
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
function qualityForTask(row, taskType) {
|
|
197
|
+
const base = asScore(row?.liveScores?.[taskType]) ?? asScore(row?.liveOverall) ?? row?.metadata?.quality ?? row?.quality ?? 0
|
|
198
|
+
// User ratings and reviews nudge a route by at most a few points.
|
|
199
|
+
return row?.qualityBias ? clamp(base + row.qualityBias) : base
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
function specialtyForTask(row, taskType) {
|
|
203
|
+
return specialtyMatch(row.metadata, taskType, row.liveScores)
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
function asScore(value) {
|
|
207
|
+
if (value === null || value === undefined || value === '') return undefined
|
|
208
|
+
const number = Number(value)
|
|
209
|
+
if (!Number.isFinite(number)) return undefined
|
|
210
|
+
return clamp(number > 1 ? number / 100 : number)
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
function normalizePricing(pricing) {
|
|
214
|
+
if (pricing === null || typeof pricing !== 'object' || Array.isArray(pricing)) return {}
|
|
215
|
+
const normalized = {}
|
|
216
|
+
for (const [id, raw] of Object.entries(pricing)) {
|
|
217
|
+
if (raw === null || typeof raw !== 'object' || Array.isArray(raw)) continue
|
|
218
|
+
if ((raw.input ?? raw.costIn) === null || (raw.input ?? raw.costIn) === undefined) continue
|
|
219
|
+
if ((raw.output ?? raw.costOut) === null || (raw.output ?? raw.costOut) === undefined) continue
|
|
220
|
+
const input = Number(raw.input ?? raw.costIn)
|
|
221
|
+
const output = Number(raw.output ?? raw.costOut)
|
|
222
|
+
const cacheRead = Number(raw.cacheRead ?? input)
|
|
223
|
+
const cacheWrite = Number(raw.cacheWrite ?? input)
|
|
224
|
+
if (![input, output, cacheRead, cacheWrite].every(value => Number.isFinite(value) && value >= 0)) continue
|
|
225
|
+
// Currency conversion must be supplied by the user or provider. Treating
|
|
226
|
+
// CNY or another currency as USD would make the budget calculation false.
|
|
227
|
+
if (String(raw.currency ?? 'USD').toUpperCase() !== 'USD') continue
|
|
228
|
+
normalized[normalize(id)] = {
|
|
229
|
+
input: input,
|
|
230
|
+
output,
|
|
231
|
+
cacheRead,
|
|
232
|
+
cacheWrite,
|
|
233
|
+
currency: String(raw.currency ?? 'USD').toUpperCase(),
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
return normalized
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
function pricingFor(model, pricing, provider = '') {
|
|
240
|
+
const normalizedPricing = normalizePricing(pricing)
|
|
241
|
+
const providerOverride = provider === '' ? undefined : normalizedPricing[normalize(`${provider}/${model.id}`)]
|
|
242
|
+
const override = providerOverride ?? normalizedPricing[normalize(model.id)]
|
|
243
|
+
if (override) return override
|
|
244
|
+
if (!Number.isFinite(Number(model.costIn)) || !Number.isFinite(Number(model.costOut))
|
|
245
|
+
|| model.costIn === null || model.costIn === undefined || model.costOut === null || model.costOut === undefined) return null
|
|
246
|
+
return {
|
|
247
|
+
input: Number(model.costIn),
|
|
248
|
+
output: Number(model.costOut),
|
|
249
|
+
cacheRead: Number(model.cacheRead ?? model.costIn),
|
|
250
|
+
cacheWrite: Number(model.cacheWrite ?? model.costIn),
|
|
251
|
+
currency: 'USD',
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
function normalizedCacheRatios(cacheReadRatio = 0, cacheWriteRatio = 0) {
|
|
256
|
+
const read = Number.isFinite(Number(cacheReadRatio)) ? clamp(Number(cacheReadRatio)) : 0
|
|
257
|
+
const write = Number.isFinite(Number(cacheWriteRatio)) ? Math.min(clamp(Number(cacheWriteRatio)), 1 - read) : 0
|
|
258
|
+
return { read, write }
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
function effectivePricing(pricing, cacheReadRatio = 0, cacheWriteRatio = 0) {
|
|
262
|
+
const { read, write } = normalizedCacheRatios(cacheReadRatio, cacheWriteRatio)
|
|
263
|
+
return {
|
|
264
|
+
input: (1 - read - write) * Number(pricing.input)
|
|
265
|
+
+ read * Number(pricing.cacheRead)
|
|
266
|
+
+ write * Number(pricing.cacheWrite),
|
|
267
|
+
output: Number(pricing.output),
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
function costScore(pricing, maxCost, cacheReadRatio = 0, cacheWriteRatio = 0) {
|
|
272
|
+
if (pricing === null) return 0
|
|
273
|
+
const effective = effectivePricing(pricing, cacheReadRatio, cacheWriteRatio)
|
|
274
|
+
const mean = (effective.input + effective.output) / 2
|
|
275
|
+
if (maxCost <= 0) return mean === 0 ? 1 : 0
|
|
276
|
+
return clamp(1 - mean / maxCost)
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
export function estimateCost(model, text, outputTokens = 900, pricingOverrides = {}, cacheReadRatio = 0, cacheWriteRatio = 0) {
|
|
280
|
+
const pricing = pricingFor(model, pricingOverrides)
|
|
281
|
+
if (pricing === null) return null
|
|
282
|
+
const inputTokens = Math.max(80, Math.ceil(String(text ?? '').length / 3.7))
|
|
283
|
+
const ratios = normalizedCacheRatios(cacheReadRatio, cacheWriteRatio)
|
|
284
|
+
const cacheReadTokens = Math.min(inputTokens, Math.max(0, Math.round(inputTokens * ratios.read)))
|
|
285
|
+
const cacheWriteTokens = Math.min(inputTokens - cacheReadTokens, Math.max(0, Math.round(inputTokens * ratios.write)))
|
|
286
|
+
const billableInputTokens = inputTokens - cacheReadTokens - cacheWriteTokens
|
|
287
|
+
return ((billableInputTokens * pricing.input)
|
|
288
|
+
+ (cacheReadTokens * pricing.cacheRead)
|
|
289
|
+
+ (cacheWriteTokens * pricing.cacheWrite)
|
|
290
|
+
+ (outputTokens * pricing.output)) / 1_000_000
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
export function modelMetadata(name) {
|
|
294
|
+
const key = normalize(name)
|
|
295
|
+
if (!key) return null
|
|
296
|
+
return MODEL_CATALOG.find(model => model.aliases.some(alias => key === normalize(alias))) ?? null
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
/**
|
|
300
|
+
* Return the staged collaboration task for one agent-loop step.
|
|
301
|
+
* Complex turns deliberately use the loop's real steps: each work report is
|
|
302
|
+
* logged as an assistant message, then the synthesis step receives the full
|
|
303
|
+
* durable history. This keeps the plan auditable without exposing private
|
|
304
|
+
* chain-of-thought.
|
|
305
|
+
*/
|
|
306
|
+
export function collaborationStage(plan, step) {
|
|
307
|
+
if (plan?.complexity?.band !== 'complex' || !Array.isArray(plan.subtasks)) return null
|
|
308
|
+
const index = Math.max(1, Number(step) || 1) - 1
|
|
309
|
+
const task = plan.subtasks[index]
|
|
310
|
+
return task === undefined ? null : { ...task, index: index + 1, total: plan.subtasks.length }
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/** Return the next staged task after a completed step, or null at synthesis end. */
|
|
314
|
+
export function nextCollaborationStage(plan, step) {
|
|
315
|
+
return collaborationStage(plan, (Number(step) || 0) + 1)
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Build a visible, model-facing stage instruction. It asks for a concise work
|
|
320
|
+
* report rather than hidden reasoning; the report is persisted and passed to
|
|
321
|
+
* later stages by the normal session history.
|
|
322
|
+
*/
|
|
323
|
+
export function collaborationInstruction(plan, step) {
|
|
324
|
+
const stage = collaborationStage(plan, step)
|
|
325
|
+
if (stage === null) return ''
|
|
326
|
+
const taskType = String(plan.taskType ?? 'general')
|
|
327
|
+
if (stage.purpose === 'synthesis') {
|
|
328
|
+
return [
|
|
329
|
+
`[Model Router 协作阶段 ${stage.index}/${stage.total}:结果校验与整合]`,
|
|
330
|
+
`你是最终汇总模型。请阅读主人原问题以及前序协作阶段的工作报告,完成${taskType}任务的交叉校验、冲突处理和最终回答。`,
|
|
331
|
+
'只输出面向主人的最终答案,不要复述内部调度指令,不要编造不存在的证据。',
|
|
332
|
+
'回答必须使用 Markdown;数学公式使用 KaTeX 兼容的 $...$ 或 $$...$$。',
|
|
333
|
+
].join('\n')
|
|
334
|
+
}
|
|
335
|
+
if (stage.purpose === 'analysis') {
|
|
336
|
+
return [
|
|
337
|
+
`[Model Router 协作阶段 ${stage.index}/${stage.total}:问题建模与约束提取]`,
|
|
338
|
+
`请针对主人的 ${taskType} 问题完成问题建模:提取目标、约束、输入输出、验收标准和关键风险。`,
|
|
339
|
+
'只提交结构化工作报告,供后续模型使用;不要直接替主人给最终答案,也不要输出隐私化的逐步思维链。',
|
|
340
|
+
].join('\n')
|
|
341
|
+
}
|
|
342
|
+
return [
|
|
343
|
+
`[Model Router 协作阶段 ${stage.index}/${stage.total}:${stage.name}]`,
|
|
344
|
+
`请阅读主人原问题和上一阶段报告,完成 ${taskType} 任务中负责的资料、代码、证据或方案处理。`,
|
|
345
|
+
...(stage.objective ? [`本工作包的具体目标:${stage.objective}`] : []),
|
|
346
|
+
'只提交可核验的结构化工作报告,列出结论、依据、待确认项和可直接复用的产物;不要直接替主人输出最终答案。',
|
|
347
|
+
].join('\n')
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
function taskTokenBudget(text, task, complexity, cacheReadRatio = 0, cacheWriteRatio = 0) {
|
|
351
|
+
const inputTokens = Math.max(80, Math.ceil(String(text ?? '').length / 3.7))
|
|
352
|
+
const multipliers = {
|
|
353
|
+
analysis: { input: 0.90, output: 0.55 },
|
|
354
|
+
execution: { input: 1.20, output: (task.difficulty ?? complexity) === 'complex' ? 1.45 : 1.00 },
|
|
355
|
+
verification: { input: 1.15, output: 0.70 },
|
|
356
|
+
synthesis: { input: 1.65, output: 1.30 },
|
|
357
|
+
}
|
|
358
|
+
const multiplier = multipliers[task.purpose] ?? { input: 1, output: 1 }
|
|
359
|
+
const effortMultiplier = reasoningEffortMultiplier(task.reasoningEffort)
|
|
360
|
+
const totalInputTokens = Math.max(80, Math.round(inputTokens * multiplier.input))
|
|
361
|
+
const ratios = normalizedCacheRatios(cacheReadRatio, cacheWriteRatio)
|
|
362
|
+
const cacheReadTokens = Math.min(totalInputTokens, Math.max(0, Math.round(totalInputTokens * ratios.read)))
|
|
363
|
+
const cacheWriteTokens = Math.min(totalInputTokens - cacheReadTokens, Math.max(0, Math.round(totalInputTokens * ratios.write)))
|
|
364
|
+
return {
|
|
365
|
+
inputTokens: totalInputTokens,
|
|
366
|
+
cacheReadTokens,
|
|
367
|
+
cacheWriteTokens,
|
|
368
|
+
outputTokens: Math.max(220, Math.round(900 * multiplier.output * effortMultiplier.output)),
|
|
369
|
+
}
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
function taskQualityFloor(band, task) {
|
|
373
|
+
const difficulty = task.difficulty ?? band
|
|
374
|
+
const base = QUALITY_FLOORS[difficulty] ?? QUALITY_FLOORS.balanced
|
|
375
|
+
if (task.purpose === 'synthesis') return Math.max(base, band === 'complex' ? 0.84 : base)
|
|
376
|
+
if (difficulty !== 'complex') return base
|
|
377
|
+
return clamp(base + Math.max(0, Number(task.criticality ?? 0.75) - 0.65) * 0.12)
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
const MAX_EXPLICIT_EXECUTION_PACKAGES = 6
|
|
381
|
+
const REQUIREMENT_ACTION = /(?:实现|新增|添加|修复|更新|构建|设计|完成|编写|优化|接入|支持|配置|部署|测试|验证|检查|整理|创建|移除|替换|迁移|适配|开发|调研|安装|下载|上传|发布|生成|集成|改造|封装|显示|提供|允许|确保|处理|解决|分析|探测|检测|选择|分配|拆分|对接|加入|保留|记录|输出|审查|核对|对比|提取|摘要|总结|分类|翻译|格式化|润色|可以|能够|需要|进行|implement|add|fix|update|build|design|write|improve|support|configure|deploy|test|verify|check|create|remove|replace|migrate|install|publish|generate|integrate|review|extract|summarize|classify|translate|format|should|must|need)/i
|
|
382
|
+
const HIGH_STAKES_WORK = /架构|安全|隐私|权限|并发|事务|迁移|生产|部署|发布|证明|定理|科研|论文|复杂|多步骤|跨系统|系统设计|architecture|security|migration|production|proof|theorem|research/i
|
|
383
|
+
const SIMPLE_WORK = /翻译|摘要|总结|提取|分类|格式化|列出|改写|润色|拼写|校对|translate|summarize|extract|classify|format|proofread/i
|
|
384
|
+
const HIGH_STAKES_ACTION = /设计|实现|修复|审计|测试|验证|证明|推导|部署|迁移|规划|研究|解决|推理|design|implement|fix|audit|verify|prove|derive|deploy|migrate|research|solve/i
|
|
385
|
+
const SIMPLE_REQUEST_START = /^(?:请|帮我)?(?:翻译|摘要|总结|提取|分类|格式化|列出|改写|润色|拼写|校对|translate|summarize|extract|classify|format|proofread)/i
|
|
386
|
+
|
|
387
|
+
function isSimpleRequirement(value) {
|
|
388
|
+
const item = String(value ?? '').trim()
|
|
389
|
+
return SIMPLE_REQUEST_START.test(item)
|
|
390
|
+
&& !/(?:并|然后|最后|同时|以及|and then).{0,20}(?:设计|实现|验证|部署|证明|审计|测试|design|implement|verify|deploy|prove|audit|test)/i.test(item)
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
function isHardRequirement(value) {
|
|
394
|
+
const item = String(value ?? '').trim()
|
|
395
|
+
if (isSimpleRequirement(item)) return false
|
|
396
|
+
return HIGH_STAKES_WORK.test(item) && HIGH_STAKES_ACTION.test(item)
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
function requirementDifficulty(objective, type, fallback = 'balanced') {
|
|
400
|
+
const value = String(objective ?? '').trim()
|
|
401
|
+
if (isHardRequirement(value)) return 'complex'
|
|
402
|
+
if (value.length <= 180 && SIMPLE_WORK.test(value)) return 'simple'
|
|
403
|
+
const assessed = assessComplexity(value)
|
|
404
|
+
if (type === 'math' && /证明|推导|proof|derive/i.test(value)) return 'complex'
|
|
405
|
+
if (assessed.band === 'simple' && (type === 'code' || type === 'research')) return 'balanced'
|
|
406
|
+
return assessed.band === 'simple' ? 'simple' : assessed.band === 'complex' ? 'complex' : fallback
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
function shouldSplitRequirements(requirements) {
|
|
410
|
+
if (requirements.length >= 3) return true
|
|
411
|
+
if (requirements.length !== 2) return false
|
|
412
|
+
const types = requirements.map(item => detectTaskTypes(item)[0] ?? 'general')
|
|
413
|
+
return types[0] !== types[1] || /^(?:最后|然后|接着|随后|再|基于|根据|测试|验证|部署|发布)|(?:完成|结束|实现)后/u.test(requirements[1])
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
function requirementText(value) {
|
|
417
|
+
return String(value ?? '').trim().replace(/[。;;\s]+$/u, '').trim()
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
function looksLikeRequirement(value) {
|
|
421
|
+
const item = requirementText(value)
|
|
422
|
+
return item.length >= 3 && !/[::]$/u.test(item)
|
|
423
|
+
&& !/(?:以下|下列|如下)(?:的)?(?:任务|需求|工作|事项|要求)/u.test(item)
|
|
424
|
+
&& REQUIREMENT_ACTION.test(item.slice(0, 32))
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
function explicitRequirements(text) {
|
|
428
|
+
const visibleLines = []
|
|
429
|
+
let fence = null
|
|
430
|
+
for (const line of String(text ?? '').split(/\r?\n/u)) {
|
|
431
|
+
const marker = /^\s*(`{3,}|~{3,})/u.exec(line)
|
|
432
|
+
if (fence) {
|
|
433
|
+
if (marker && marker[1][0] === fence.char && marker[1].length >= fence.length
|
|
434
|
+
&& !line.slice(marker[0].length).trim()) fence = null
|
|
435
|
+
continue
|
|
436
|
+
}
|
|
437
|
+
if (marker) {
|
|
438
|
+
fence = { char: marker[1][0], length: marker[1].length }
|
|
439
|
+
continue
|
|
440
|
+
}
|
|
441
|
+
visibleLines.push(line)
|
|
442
|
+
}
|
|
443
|
+
const value = visibleLines.join('\n')
|
|
444
|
+
const lines = visibleLines.map(line => line.trim()).filter(Boolean)
|
|
445
|
+
if (/^(?:请|帮我)?(?:总结|概括|翻译|摘要|解释)(?:以下|下列|下面|这份|这些)/u.test(lines[0] ?? '')
|
|
446
|
+
&& !/(?:执行|完成|实施|分配)/u.test(lines[0])) return []
|
|
447
|
+
const marker = /^(?:[-*•]\s+|\d{1,2}[.)、](?!\d)\s*|[一二三四五六七八九十]{1,3}[、.)]\s*)(.+)$/u
|
|
448
|
+
const marked = lines.map(line => marker.exec(line)?.[1]).filter(Boolean).map(requirementText)
|
|
449
|
+
.filter(looksLikeRequirement)
|
|
450
|
+
if (marked.length >= 2) return marked
|
|
451
|
+
|
|
452
|
+
// A newline is an item boundary only when each line looks like a request.
|
|
453
|
+
// This avoids decomposing wrapped prose, logs, examples, or a single paragraph.
|
|
454
|
+
const plain = lines.filter(line => !/^#{1,6}\s|^```|[::]$/u.test(line))
|
|
455
|
+
if (plain.length >= 2 && plain.every(looksLikeRequirement)) {
|
|
456
|
+
return plain.map(requirementText).filter(item => item.length >= 3)
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
// Commas and semicolons form work packages only when every clause is an
|
|
460
|
+
// action request. This keeps ordinary comma-separated context together.
|
|
461
|
+
const clauses = value.split(/[;;,,]/u).map(requirementText).filter(Boolean)
|
|
462
|
+
if (clauses.length >= 2 && clauses.every(looksLikeRequirement)) {
|
|
463
|
+
return clauses
|
|
464
|
+
}
|
|
465
|
+
// Complete Chinese action sentences can also be separate requirements.
|
|
466
|
+
// Do not split on ASCII periods, which commonly occur in paths and URLs.
|
|
467
|
+
const sentences = value.split(/[。!?]/u).map(requirementText).filter(Boolean)
|
|
468
|
+
if (sentences.length >= 2 && sentences.every(looksLikeRequirement)) {
|
|
469
|
+
return sentences
|
|
470
|
+
}
|
|
471
|
+
return []
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
const CN_DIGITS = Object.freeze({ 一: 1, 二: 2, 两: 2, 三: 3, 四: 4, 五: 5, 六: 6, 七: 7, 八: 8, 九: 9, 十: 10 })
|
|
475
|
+
const stepNumber = token => {
|
|
476
|
+
const value = String(token ?? '').trim()
|
|
477
|
+
if (/^\d{1,2}$/u.test(value)) return Number(value)
|
|
478
|
+
if (/^十[一二三四五六七八九]?$/u.test(value)) return 10 + (CN_DIGITS[value[1]] ?? 0)
|
|
479
|
+
if (/^[一二两三四五六七八九]十?$/u.test(value)) return CN_DIGITS[value[0]] * (value.length === 2 ? 10 : 1)
|
|
480
|
+
return null
|
|
481
|
+
}
|
|
482
|
+
const STEP_TOKEN = '(?:\\d{1,2}|[一二两三四五六七八九十]{1,2})'
|
|
483
|
+
const STEP_LIST = `${STEP_TOKEN}(?:\\s*(?:[、,,/]|和|与|及|以及|and|&|-|~|到|至)\\s*(?:第\\s*)?${STEP_TOKEN})*`
|
|
484
|
+
const STEP_REFERENCE_PATTERNS = Object.freeze([
|
|
485
|
+
// 依赖第 1 步 / 基于第 2、3 步 / 根据步骤 1 / 在第 2 步完成后 / 第 1 步之后
|
|
486
|
+
new RegExp(`(?:依赖|依靠|取决于|基于|根据|承接|使用|利用|需要|等待|待)(?:于)?\\s*(?:第\\s*(${STEP_LIST})\\s*(?:步|项|个?步骤|条)|步骤\\s*(${STEP_LIST}))`, 'giu'),
|
|
487
|
+
new RegExp(`(?:在|等)?\\s*第\\s*(${STEP_LIST})\\s*(?:步|项|个?步骤|条)(?:完成|结束|做完)?(?:之后|以后|后)`, 'giu'),
|
|
488
|
+
new RegExp(`步骤\\s*(${STEP_LIST})\\s*(?:完成|结束)?(?:之后|以后|后)`, 'giu'),
|
|
489
|
+
// depends on step 1 / after steps 1 and 2 / based on step #2 / requires step 3
|
|
490
|
+
new RegExp(`(?:depends?\\s+on|depending\\s+on|after|based\\s+on|builds?\\s+on|requires?|using(?:\\s+the\\s+output\\s+of)?)\\s+(?:the\\s+(?:result|output)s?\\s+of\\s+)?(?:steps?|items?|#)\\s*#?(${STEP_LIST})`, 'giu'),
|
|
491
|
+
])
|
|
492
|
+
|
|
493
|
+
/**
|
|
494
|
+
* Step numbers (1-based, in list order) that a requirement explicitly names
|
|
495
|
+
* as its prerequisites: “依赖第 1 步”, “基于第 2、3 步”, “第 1 步完成后”,
|
|
496
|
+
* “depends on step 1”, “after steps 1 and 2”. Ranges (“第 1-3 步”) expand.
|
|
497
|
+
*/
|
|
498
|
+
export function explicitStepReferences(objective) {
|
|
499
|
+
const found = new Set()
|
|
500
|
+
const value = String(objective ?? '')
|
|
501
|
+
for (const pattern of STEP_REFERENCE_PATTERNS) {
|
|
502
|
+
pattern.lastIndex = 0
|
|
503
|
+
for (const match of value.matchAll(pattern)) {
|
|
504
|
+
const list = match.slice(1).find(Boolean) ?? ''
|
|
505
|
+
const parts = list.split(/\s*(?:[、,,/]|和|与|及|以及|and|&)\s*/iu)
|
|
506
|
+
for (const part of parts) {
|
|
507
|
+
const range = /^(.+?)\s*(?:-|~|到|至)\s*(?:第\s*)?(.+)$/u.exec(part)
|
|
508
|
+
if (range) {
|
|
509
|
+
const from = stepNumber(range[1])
|
|
510
|
+
const to = stepNumber(range[2])
|
|
511
|
+
if (from && to && to >= from && to - from < 20) for (let step = from; step <= to; step += 1) found.add(step)
|
|
512
|
+
continue
|
|
513
|
+
}
|
|
514
|
+
const step = stepNumber(part.replace(/^第\s*/u, ''))
|
|
515
|
+
if (step) found.add(step)
|
|
516
|
+
}
|
|
517
|
+
}
|
|
518
|
+
}
|
|
519
|
+
return [...found].sort((a, b) => a - b)
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
function taskPackages(taskType, text, band) {
|
|
523
|
+
if (band !== 'complex') {
|
|
524
|
+
const task = { id: 'execution', name: '直接回答与必要校验', type: taskType, purpose: 'execution', difficulty: band, criticality: 0.65, dependsOn: [], preferredReasoningEffort: band === 'simple' ? 'low' : 'medium' }
|
|
525
|
+
return [{ ...task, qualityFloor: taskQualityFloor(band, task) }]
|
|
526
|
+
}
|
|
527
|
+
const value = String(text ?? '')
|
|
528
|
+
const packages = [
|
|
529
|
+
{ id: 'analysis', name: '问题建模与约束提取', type: 'reasoning', purpose: 'analysis', difficulty: 'balanced', criticality: 0.80, dependsOn: [], preferredReasoningEffort: 'medium' },
|
|
530
|
+
]
|
|
531
|
+
const requirements = explicitRequirements(text)
|
|
532
|
+
if (requirements.length >= 2) {
|
|
533
|
+
// Preserve every stated requirement when the list exceeds the package cap.
|
|
534
|
+
const groups = requirements.length > MAX_EXPLICIT_EXECUTION_PACKAGES
|
|
535
|
+
? [...requirements.slice(0, MAX_EXPLICIT_EXECUTION_PACKAGES - 1).map(item => [item]),
|
|
536
|
+
requirements.slice(MAX_EXPLICIT_EXECUTION_PACKAGES - 1)]
|
|
537
|
+
: requirements.map(item => [item])
|
|
538
|
+
groups.forEach((group, index) => {
|
|
539
|
+
const objective = group.length === 1 ? group[0]
|
|
540
|
+
: group.map((item, offset) => `${MAX_EXPLICIT_EXECUTION_PACKAGES + offset}. ${item}`).join('\n')
|
|
541
|
+
const type = detectTaskTypes(objective)[0] ?? 'general'
|
|
542
|
+
const difficulty = group.length > 1 ? 'complex' : requirementDifficulty(objective, type)
|
|
543
|
+
const previous = packages.at(-1)
|
|
544
|
+
// Explicit references (“依赖第 1 步”) become DAG edges; only earlier steps count, so the plan stays acyclic.
|
|
545
|
+
const groupOf = step => requirements.length > MAX_EXPLICIT_EXECUTION_PACKAGES && step >= MAX_EXPLICIT_EXECUTION_PACKAGES
|
|
546
|
+
? MAX_EXPLICIT_EXECUTION_PACKAGES - 1 : step - 1
|
|
547
|
+
const explicit = [...new Set(group.flatMap(item => explicitStepReferences(item)).map(groupOf))]
|
|
548
|
+
.filter(target => target >= 0 && target < index).map(target => `execution-${target + 1}`)
|
|
549
|
+
const sequential = explicit.length === 0 && /^(?:最后|然后|接着|随后|再|基于|根据|测试|验证|部署|发布)|(?:完成|结束|实现)后/u.test(objective)
|
|
550
|
+
packages.push({
|
|
551
|
+
id: `execution-${index + 1}`,
|
|
552
|
+
name: group.length === 1
|
|
553
|
+
? `需求 ${index + 1}:${group[0].slice(0, 28)}`
|
|
554
|
+
: `需求 ${index + 1}:其余 ${group.length} 项`,
|
|
555
|
+
objective,
|
|
556
|
+
type,
|
|
557
|
+
purpose: 'execution',
|
|
558
|
+
difficulty,
|
|
559
|
+
criticality: difficulty === 'simple' ? 0.55 : difficulty === 'balanced' ? 0.72 : 0.86,
|
|
560
|
+
dependsOn: explicit.length ? ['analysis', ...explicit]
|
|
561
|
+
: sequential && previous?.purpose === 'execution' ? ['analysis', previous.id] : ['analysis'],
|
|
562
|
+
preferredReasoningEffort: difficulty === 'simple' ? 'low' : difficulty === 'balanced' ? 'medium' : 'high',
|
|
563
|
+
})
|
|
564
|
+
})
|
|
565
|
+
} else {
|
|
566
|
+
const domains = [...new Set([...(detectTaskTypes(text)), taskType].filter(type => type !== 'general'))]
|
|
567
|
+
for (const type of domains.length > 0 ? domains : [taskType]) {
|
|
568
|
+
const objective = value.split(/[,,。;;]|最后|然后|接着|并且/u)
|
|
569
|
+
.map(item => item.trim()).filter(item => detectTaskTypes(item).includes(type)).join(';') || value
|
|
570
|
+
const difficulty = requirementDifficulty(objective, type)
|
|
571
|
+
packages.push({
|
|
572
|
+
id: `execution-${type}`,
|
|
573
|
+
name: `${TASK_TYPE_LABELS[type] ?? type}方向处理`,
|
|
574
|
+
objective,
|
|
575
|
+
type,
|
|
576
|
+
purpose: 'execution',
|
|
577
|
+
difficulty,
|
|
578
|
+
criticality: difficulty === 'simple' ? 0.55 : difficulty === 'balanced' ? 0.72 : 0.86,
|
|
579
|
+
dependsOn: ['analysis'],
|
|
580
|
+
preferredReasoningEffort: difficulty === 'simple' ? 'low' : difficulty === 'balanced' ? 'medium' : 'high',
|
|
581
|
+
})
|
|
582
|
+
}
|
|
583
|
+
}
|
|
584
|
+
if (/(测试|验证|评估|对比|benchmark|test|verify|audit)/i.test(value)) {
|
|
585
|
+
packages.push({
|
|
586
|
+
id: 'verification',
|
|
587
|
+
name: '验证、反例与风险审查',
|
|
588
|
+
type: 'reasoning',
|
|
589
|
+
purpose: 'verification',
|
|
590
|
+
difficulty: 'complex',
|
|
591
|
+
criticality: 0.88,
|
|
592
|
+
dependsOn: packages.filter(task => task.purpose === 'execution').map(task => task.id),
|
|
593
|
+
preferredReasoningEffort: 'high',
|
|
594
|
+
})
|
|
595
|
+
}
|
|
596
|
+
packages.push({
|
|
597
|
+
id: 'synthesis',
|
|
598
|
+
name: '结果校验与整合',
|
|
599
|
+
type: 'reasoning',
|
|
600
|
+
purpose: 'synthesis',
|
|
601
|
+
difficulty: 'complex',
|
|
602
|
+
criticality: 1,
|
|
603
|
+
dependsOn: packages.filter(task => task.purpose !== 'analysis').map(task => task.id),
|
|
604
|
+
preferredReasoningEffort: 'xhigh',
|
|
605
|
+
})
|
|
606
|
+
return packages.map(task => ({ ...task, qualityFloor: taskQualityFloor(band, task) }))
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
const SYNTHESIS_WEIGHTS = Object.freeze({ quality: 0.58, cost: 0.08, latency: 0.04, specialty: 0.08, reasoning: 0.16, risk: 0.06 })
|
|
610
|
+
const ROUTING_BEAM_WIDTH = 256
|
|
611
|
+
const ROUTING_CANDIDATE_LIMIT = 12
|
|
612
|
+
|
|
613
|
+
function weightsForTask(weights, task) {
|
|
614
|
+
if (task.weights) return task.weights
|
|
615
|
+
return task.purpose === 'synthesis' ? SYNTHESIS_WEIGHTS : (OBJECTIVE_WEIGHTS[task.difficulty] ?? weights)
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
function compareText(left, right) {
|
|
619
|
+
const a = String(left)
|
|
620
|
+
const b = String(right)
|
|
621
|
+
return a < b ? -1 : a > b ? 1 : 0
|
|
622
|
+
}
|
|
623
|
+
|
|
624
|
+
function compareRowsStable(left, right) {
|
|
625
|
+
return compareText(routeKey(left.provider, left.model), routeKey(right.provider, right.model))
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
function reasoningEffortRank(effort) {
|
|
629
|
+
const normalized = String(effort ?? '').trim().toLowerCase().replace(/[\s_-]+/g, '')
|
|
630
|
+
const aliases = { none: 'off', disabled: 'off', extra: 'xhigh', extrahigh: 'xhigh', maximum: 'max' }
|
|
631
|
+
const canonical = aliases[normalized] ?? normalized
|
|
632
|
+
const index = REASONING_EFFORT_ORDER.indexOf(canonical)
|
|
633
|
+
return index < 0 ? REASONING_EFFORT_ORDER.indexOf('medium') : index
|
|
634
|
+
}
|
|
635
|
+
|
|
636
|
+
function reasoningEffortMultiplier(effort) {
|
|
637
|
+
const normalized = String(effort ?? '').trim().toLowerCase().replace(/[\s_-]+/g, '')
|
|
638
|
+
const aliases = { none: 'off', disabled: 'off', extra: 'xhigh', extrahigh: 'xhigh', maximum: 'max' }
|
|
639
|
+
return REASONING_EFFORT_MULTIPLIERS[aliases[normalized] ?? normalized] ?? REASONING_EFFORT_MULTIPLIERS.medium
|
|
640
|
+
}
|
|
641
|
+
|
|
642
|
+
/** Pick the closest exact effort id exposed by an adapter. */
|
|
643
|
+
export function selectReasoningEffort(efforts, preferred = 'medium') {
|
|
644
|
+
const exact = Array.isArray(efforts)
|
|
645
|
+
? [...new Set(efforts.map(effort => String(effort?.id ?? effort ?? '')).filter(Boolean))]
|
|
646
|
+
: []
|
|
647
|
+
if (exact.length === 0) return undefined
|
|
648
|
+
const preferredRank = reasoningEffortRank(preferred)
|
|
649
|
+
return exact.slice().sort((left, right) => {
|
|
650
|
+
const distance = Math.abs(reasoningEffortRank(left) - preferredRank) - Math.abs(reasoningEffortRank(right) - preferredRank)
|
|
651
|
+
if (distance !== 0) return distance
|
|
652
|
+
return reasoningEffortRank(left) - reasoningEffortRank(right) || compareText(left, right)
|
|
653
|
+
})[0]
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
function reasoningDecision(row, task) {
|
|
657
|
+
const preferred = String(task.preferredReasoningEffort ?? 'medium')
|
|
658
|
+
const efforts = Array.isArray(row.reasoningEfforts) ? row.reasoningEfforts : []
|
|
659
|
+
if (efforts.length === 0) {
|
|
660
|
+
const knownUnsupported = row.reasoningKnown === true
|
|
661
|
+
const preferredRank = reasoningEffortRank(preferred)
|
|
662
|
+
return {
|
|
663
|
+
reasoningEffort: undefined,
|
|
664
|
+
reasoningFit: knownUnsupported ? clamp(0.78 - preferredRank * 0.08) : 0.55,
|
|
665
|
+
preferredReasoningEffort: preferred,
|
|
666
|
+
multiplier: REASONING_EFFORT_MULTIPLIERS.medium,
|
|
667
|
+
}
|
|
668
|
+
}
|
|
669
|
+
const preferredRank = reasoningEffortRank(preferred)
|
|
670
|
+
// On an equal distance the shared selector prefers the lower effort to
|
|
671
|
+
// avoid unnecessary latency and output-token cost.
|
|
672
|
+
const chosen = selectReasoningEffort(efforts, preferred)
|
|
673
|
+
const distance = Math.abs(reasoningEffortRank(chosen) - preferredRank)
|
|
674
|
+
return {
|
|
675
|
+
reasoningEffort: chosen,
|
|
676
|
+
reasoningFit: clamp(1 - distance / (REASONING_EFFORT_ORDER.length - 1)),
|
|
677
|
+
preferredReasoningEffort: preferred,
|
|
678
|
+
multiplier: reasoningEffortMultiplier(chosen),
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
function candidateUtility(row, task, weights, maxCost, usedRoutes, cacheReadRatio = 0, cacheWriteRatio = 0) {
|
|
683
|
+
const quality = qualityForTask(row, task.type)
|
|
684
|
+
const floor = Number(task.qualityFloor ?? taskQualityFloor('complex', task))
|
|
685
|
+
const qualityGap = Math.max(0, floor - quality)
|
|
686
|
+
// Reuse avoids handoff overhead and is preferable when the same affordable
|
|
687
|
+
// route is suitable for independent, low-risk work packages.
|
|
688
|
+
const duplicatePenalty = 0
|
|
689
|
+
const synthesisPreference = task.purpose === 'synthesis' && /deepseek[- ]?v4[- ]?pro/i.test(row.model) ? 0.025 : 0
|
|
690
|
+
const reasoning = reasoningDecision(row, task)
|
|
691
|
+
const cost = clamp(costScore(row.pricing, maxCost, cacheReadRatio, cacheWriteRatio) / Math.sqrt(reasoning.multiplier.output))
|
|
692
|
+
const score = weights.quality * quality
|
|
693
|
+
+ weights.cost * cost
|
|
694
|
+
+ weights.latency * (1 - clamp(row.latency * reasoning.multiplier.latency))
|
|
695
|
+
+ weights.specialty * specialtyForTask(row, task.type)
|
|
696
|
+
+ (weights.reasoning ?? 0) * reasoning.reasoningFit
|
|
697
|
+
- weights.risk * row.risk
|
|
698
|
+
- duplicatePenalty
|
|
699
|
+
- qualityGap * (task.criticality ?? 0.75)
|
|
700
|
+
+ synthesisPreference
|
|
701
|
+
return { score, floor, qualityGap, ...reasoning }
|
|
702
|
+
}
|
|
703
|
+
|
|
704
|
+
function chooseAssignment(rows, task, weights, maxCost, usedRoutes, preferred, cacheReadRatio = 0, cacheWriteRatio = 0) {
|
|
705
|
+
const ordered = rows
|
|
706
|
+
.map(row => ({ row, decision: candidateUtility(row, task, weights, maxCost, usedRoutes, cacheReadRatio, cacheWriteRatio) }))
|
|
707
|
+
.sort((left, right) => right.decision.score - left.decision.score)
|
|
708
|
+
const feasible = ordered.filter(item => qualityForTask(item.row, task.type) >= item.decision.floor)
|
|
709
|
+
const chosen = (preferred === true ? feasible : feasible.filter(item => !usedRoutes.has(routeKey(item.row.provider, item.row.model))))[0]
|
|
710
|
+
?? feasible[0]
|
|
711
|
+
?? ordered[0]
|
|
712
|
+
if (chosen === undefined) return { row: null, relaxed: true, floor: 0, qualityGap: 1 }
|
|
713
|
+
return { row: chosen.row, relaxed: qualityForTask(chosen.row, task.type) < chosen.decision.floor, floor: chosen.decision.floor, qualityGap: chosen.decision.qualityGap }
|
|
714
|
+
}
|
|
715
|
+
|
|
716
|
+
function taskCost(row, task, text, complexity, cacheReadRatio = 0, cacheWriteRatio = 0) {
|
|
717
|
+
if (row?.pricing === null) return 0 // Search sentinel; the public estimate remains null.
|
|
718
|
+
const decision = row === null ? null : reasoningDecision(row, task)
|
|
719
|
+
const tokens = taskTokenBudget(text, { ...task, reasoningEffort: decision?.reasoningEffort }, complexity, cacheReadRatio, cacheWriteRatio)
|
|
720
|
+
return row === null
|
|
721
|
+
? 0
|
|
722
|
+
: (((tokens.inputTokens - tokens.cacheReadTokens - tokens.cacheWriteTokens) * row.pricing.input)
|
|
723
|
+
+ (tokens.cacheReadTokens * row.pricing.cacheRead)
|
|
724
|
+
+ (tokens.cacheWriteTokens * row.pricing.cacheWrite)
|
|
725
|
+
+ (tokens.outputTokens * row.pricing.output)) / 1_000_000
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
function dominates(left, right, task, text, complexity, cacheReadRatio, cacheWriteRatio) {
|
|
729
|
+
if (left.pricing === null || right.pricing === null) return false
|
|
730
|
+
const leftReasoning = reasoningDecision(left, task)
|
|
731
|
+
const rightReasoning = reasoningDecision(right, task)
|
|
732
|
+
const leftValues = {
|
|
733
|
+
quality: qualityForTask(left, task.type),
|
|
734
|
+
cost: taskCost(left, task, text, complexity, cacheReadRatio, cacheWriteRatio),
|
|
735
|
+
latency: clamp(left.latency * leftReasoning.multiplier.latency),
|
|
736
|
+
specialty: specialtyForTask(left, task.type),
|
|
737
|
+
reasoning: leftReasoning.reasoningFit,
|
|
738
|
+
risk: clamp(left.risk),
|
|
739
|
+
}
|
|
740
|
+
const rightValues = {
|
|
741
|
+
quality: qualityForTask(right, task.type),
|
|
742
|
+
cost: taskCost(right, task, text, complexity, cacheReadRatio, cacheWriteRatio),
|
|
743
|
+
latency: clamp(right.latency * rightReasoning.multiplier.latency),
|
|
744
|
+
specialty: specialtyForTask(right, task.type),
|
|
745
|
+
reasoning: rightReasoning.reasoningFit,
|
|
746
|
+
risk: clamp(right.risk),
|
|
747
|
+
}
|
|
748
|
+
const noWorse = leftValues.quality >= rightValues.quality
|
|
749
|
+
&& leftValues.cost <= rightValues.cost
|
|
750
|
+
&& leftValues.latency <= rightValues.latency
|
|
751
|
+
&& leftValues.specialty >= rightValues.specialty
|
|
752
|
+
&& leftValues.reasoning >= rightValues.reasoning
|
|
753
|
+
&& leftValues.risk <= rightValues.risk
|
|
754
|
+
const strictlyBetter = leftValues.quality > rightValues.quality
|
|
755
|
+
|| leftValues.cost < rightValues.cost
|
|
756
|
+
|| leftValues.latency < rightValues.latency
|
|
757
|
+
|| leftValues.specialty > rightValues.specialty
|
|
758
|
+
|| leftValues.reasoning > rightValues.reasoning
|
|
759
|
+
|| leftValues.risk < rightValues.risk
|
|
760
|
+
return noWorse && strictlyBetter
|
|
761
|
+
}
|
|
762
|
+
|
|
763
|
+
function candidatePool(rows, task, weights, maxCost, text, complexity, cacheReadRatio, cacheWriteRatio) {
|
|
764
|
+
const eligibleRows = task.type === 'vision'
|
|
765
|
+
? rows.filter(row => row.inputModalities.length === 0 || row.inputModalities.includes('image'))
|
|
766
|
+
: rows
|
|
767
|
+
const floor = Number(task.qualityFloor ?? 0)
|
|
768
|
+
const feasible = eligibleRows.filter(row => qualityForTask(row, task.type) >= floor)
|
|
769
|
+
const source = feasible.length > 0
|
|
770
|
+
? feasible
|
|
771
|
+
: eligibleRows.slice().sort((left, right) => qualityForTask(right, task.type) - qualityForTask(left, task.type) || compareRowsStable(left, right)).slice(0, 3)
|
|
772
|
+
const taskWeights = weightsForTask(weights, task)
|
|
773
|
+
const scored = source.map(row => ({
|
|
774
|
+
row,
|
|
775
|
+
decision: candidateUtility(row, task, taskWeights, maxCost, new Set(), cacheReadRatio, cacheWriteRatio),
|
|
776
|
+
cost: taskCost(row, task, text, complexity, cacheReadRatio, cacheWriteRatio),
|
|
777
|
+
}))
|
|
778
|
+
const frontier = scored.filter(item => !source.some(other => other !== item.row && dominates(other, item.row, task, text, complexity, cacheReadRatio, cacheWriteRatio)))
|
|
779
|
+
const essential = [
|
|
780
|
+
scored.slice().sort((left, right) => left.cost - right.cost || compareRowsStable(left.row, right.row))[0],
|
|
781
|
+
scored.slice().sort((left, right) => right.decision.score - left.decision.score || compareRowsStable(left.row, right.row))[0],
|
|
782
|
+
scored.slice().sort((left, right) => qualityForTask(right.row, task.type) - qualityForTask(left.row, task.type) || compareRowsStable(left.row, right.row))[0],
|
|
783
|
+
].filter(Boolean)
|
|
784
|
+
const ordered = [...frontier, ...essential]
|
|
785
|
+
.filter((item, index, all) => all.findIndex(candidate => candidate.row === item.row) === index)
|
|
786
|
+
.sort((left, right) => right.decision.score - left.decision.score || left.cost - right.cost || compareRowsStable(left.row, right.row))
|
|
787
|
+
.slice(0, ROUTING_CANDIDATE_LIMIT)
|
|
788
|
+
return {
|
|
789
|
+
options: ordered,
|
|
790
|
+
relaxed: feasible.length === 0,
|
|
791
|
+
pruned: Math.max(0, eligibleRows.length - ordered.length),
|
|
792
|
+
}
|
|
793
|
+
}
|
|
794
|
+
|
|
795
|
+
function stateSignature(state) {
|
|
796
|
+
return state.assignments.map(assignment => `${routeKey(assignment.row?.provider, assignment.row?.model)}@${assignment.decision?.reasoningEffort ?? 'provider-default'}`).join('|')
|
|
797
|
+
}
|
|
798
|
+
|
|
799
|
+
function compareUtilityStates(left, right) {
|
|
800
|
+
return left.relaxedCount - right.relaxedCount
|
|
801
|
+
|| left.qualityShortfall - right.qualityShortfall
|
|
802
|
+
|| right.score - left.score
|
|
803
|
+
|| left.cost - right.cost
|
|
804
|
+
|| left.switches - right.switches
|
|
805
|
+
|| compareText(stateSignature(left), stateSignature(right))
|
|
806
|
+
}
|
|
807
|
+
|
|
808
|
+
function compareCostStates(left, right) {
|
|
809
|
+
return left.relaxedCount - right.relaxedCount
|
|
810
|
+
|| left.qualityShortfall - right.qualityShortfall
|
|
811
|
+
|| left.cost - right.cost
|
|
812
|
+
|| right.score - left.score
|
|
813
|
+
|| left.switches - right.switches
|
|
814
|
+
|| compareText(stateSignature(left), stateSignature(right))
|
|
815
|
+
}
|
|
816
|
+
|
|
817
|
+
function solveAssignments({ rows, tasks, weights, maxCost, text, complexity, budget, cacheReadRatio, cacheWriteRatio, minimizeCost = false }) {
|
|
818
|
+
const pools = tasks.map(task => candidatePool(rows, task, weights, maxCost, text, complexity, cacheReadRatio, cacheWriteRatio))
|
|
819
|
+
if (pools.some(pool => pool.options.length === 0)) return null
|
|
820
|
+
const suffixMinimum = Array(tasks.length + 1).fill(0)
|
|
821
|
+
for (let index = tasks.length - 1; index >= 0; index -= 1) {
|
|
822
|
+
const costOptions = Number.isFinite(budget)
|
|
823
|
+
? pools[index].options.filter(option => option.row.pricing !== null)
|
|
824
|
+
: pools[index].options
|
|
825
|
+
if (costOptions.length === 0) return null
|
|
826
|
+
suffixMinimum[index] = suffixMinimum[index + 1] + Math.min(...costOptions.map(option => option.cost))
|
|
827
|
+
}
|
|
828
|
+
if (Number.isFinite(budget) && suffixMinimum[0] > budget + 1e-12) return null
|
|
829
|
+
|
|
830
|
+
let states = [{ assignments: [], routesByTask: new Map(), usedRoutes: new Set(), score: 0, cost: 0, switches: 0, relaxedCount: 0, qualityShortfall: 0 }]
|
|
831
|
+
for (let index = 0; index < tasks.length; index += 1) {
|
|
832
|
+
const task = tasks[index]
|
|
833
|
+
const pool = pools[index]
|
|
834
|
+
const expanded = []
|
|
835
|
+
for (const state of states) {
|
|
836
|
+
for (const option of pool.options) {
|
|
837
|
+
if (Number.isFinite(budget) && option.row.pricing === null) continue
|
|
838
|
+
const nextCost = state.cost + option.cost
|
|
839
|
+
if (Number.isFinite(budget) && nextCost + suffixMinimum[index + 1] > budget + 1e-12) continue
|
|
840
|
+
const taskWeights = weightsForTask(weights, task)
|
|
841
|
+
const decision = candidateUtility(option.row, task, taskWeights, maxCost, state.usedRoutes, cacheReadRatio, cacheWriteRatio)
|
|
842
|
+
const route = routeKey(option.row.provider, option.row.model)
|
|
843
|
+
const dependencySwitches = (task.dependsOn ?? []).reduce((count, dependency) => {
|
|
844
|
+
const dependencyRoute = state.routesByTask.get(dependency)
|
|
845
|
+
return count + (dependencyRoute !== undefined && dependencyRoute !== route ? 1 : 0)
|
|
846
|
+
}, 0)
|
|
847
|
+
const handoffPenalty = dependencySwitches * 0.015
|
|
848
|
+
const qualityShortfall = Math.max(0, decision.floor - qualityForTask(option.row, task.type))
|
|
849
|
+
const usedRoutes = new Set(state.usedRoutes)
|
|
850
|
+
usedRoutes.add(route)
|
|
851
|
+
const routesByTask = new Map(state.routesByTask)
|
|
852
|
+
routesByTask.set(task.id, route)
|
|
853
|
+
expanded.push({
|
|
854
|
+
assignments: [...state.assignments, { task, row: option.row, decision: { ...decision, relaxed: qualityShortfall > 0 }, estimatedCost: option.cost, handoffPenalty }],
|
|
855
|
+
routesByTask,
|
|
856
|
+
usedRoutes,
|
|
857
|
+
score: state.score + decision.score - handoffPenalty,
|
|
858
|
+
cost: nextCost,
|
|
859
|
+
switches: state.switches + dependencySwitches,
|
|
860
|
+
relaxedCount: state.relaxedCount + (qualityShortfall > 0 ? 1 : 0),
|
|
861
|
+
qualityShortfall: state.qualityShortfall + qualityShortfall,
|
|
862
|
+
})
|
|
863
|
+
}
|
|
864
|
+
}
|
|
865
|
+
if (expanded.length === 0) return null
|
|
866
|
+
expanded.sort(minimizeCost ? compareCostStates : compareUtilityStates)
|
|
867
|
+
states = expanded.slice(0, ROUTING_BEAM_WIDTH)
|
|
868
|
+
}
|
|
869
|
+
states.sort(minimizeCost ? compareCostStates : compareUtilityStates)
|
|
870
|
+
return {
|
|
871
|
+
...states[0],
|
|
872
|
+
candidatePools: pools,
|
|
873
|
+
minimumFeasibleCost: suffixMinimum[0],
|
|
874
|
+
}
|
|
875
|
+
}
|
|
876
|
+
|
|
877
|
+
export function buildPlan({ text = '', available = [], mode = 'collective', pricing = {}, liveBench = null, liveBenchError = '', budgetUsd = 0, cacheReadRatio = 0, cacheWriteRatio = 0, preset = DEFAULT_ROUTING_PRESET } = {}) {
|
|
878
|
+
const presetId = normalizeRoutingPreset(preset)
|
|
879
|
+
const firstLine = String(text ?? '').split(/\r?\n/u)[0].trim()
|
|
880
|
+
const transformOnly = /^(?:请|帮我)?(?:总结|概括|翻译|摘要|解释)(?:以下|下列|下面|这份|这些)/u.test(firstLine)
|
|
881
|
+
&& !/(?:执行|完成|实施|分配)/u.test(firstLine)
|
|
882
|
+
const assessed = assessComplexity(transformOnly ? firstLine : text)
|
|
883
|
+
const requirements = explicitRequirements(text)
|
|
884
|
+
const compound = shouldSplitRequirements(requirements)
|
|
885
|
+
const complexity = compound && assessed.band !== 'complex'
|
|
886
|
+
? { value: Math.max(0.66, assessed.value), band: 'complex' }
|
|
887
|
+
: assessed
|
|
888
|
+
const taskType = classifyTask(transformOnly ? firstLine : text)
|
|
889
|
+
const weights = presetWeights(OBJECTIVE_WEIGHTS[complexity.band], presetId)
|
|
890
|
+
const discovered = Array.isArray(available)
|
|
891
|
+
? available.map(entry => {
|
|
892
|
+
const rawEfforts = Array.isArray(entry.reasoningEfforts) ? entry.reasoningEfforts : []
|
|
893
|
+
const reasoningEfforts = rawEfforts.map(effort => String(effort?.id ?? effort ?? '')).filter(Boolean)
|
|
894
|
+
return {
|
|
895
|
+
provider: String(entry.provider ?? ''),
|
|
896
|
+
model: String(entry.model ?? ''),
|
|
897
|
+
reasoningEfforts: [...new Set(reasoningEfforts)],
|
|
898
|
+
defaultReasoningEffort: entry.defaultReasoningEffort === undefined ? undefined : String(entry.defaultReasoningEffort),
|
|
899
|
+
reasoningKnown: entry.reasoningKnown === true
|
|
900
|
+
|| (entry.reasoningKnown === undefined && Array.isArray(entry.reasoningEfforts)),
|
|
901
|
+
quality: asScore(entry.quality),
|
|
902
|
+
qualityBias: Number.isFinite(entry.qualityBias) ? clamp(entry.qualityBias, -0.05, 0.05) : 0,
|
|
903
|
+
qualitySource: entry.qualitySource === 'user' ? 'user' : 'route',
|
|
904
|
+
latency: asScore(entry.latency),
|
|
905
|
+
risk: asScore(entry.risk),
|
|
906
|
+
specialties: Array.isArray(entry.specialties) ? entry.specialties.filter(item => typeof item === 'string') : null,
|
|
907
|
+
pricing: normalizePricing({ route: entry.pricing ?? entry.price }).route ?? null,
|
|
908
|
+
pricingSource: entry.pricingSource === 'user' ? 'user' : 'route',
|
|
909
|
+
inputModalities: Array.isArray(entry.inputModalities) ? entry.inputModalities.map(item => String(item).toLowerCase()) : [],
|
|
910
|
+
}
|
|
911
|
+
})
|
|
912
|
+
: []
|
|
913
|
+
const rows = []
|
|
914
|
+
const normalizedPrices = normalizePricing(pricing)
|
|
915
|
+
for (const route of discovered) {
|
|
916
|
+
if (!route.provider || !route.model) continue
|
|
917
|
+
const catalog = modelMetadata(route.model)
|
|
918
|
+
const metadata = {
|
|
919
|
+
...(catalog ?? { id: route.model, aliases: [route.model], specialties: [] }),
|
|
920
|
+
...(route.quality === undefined ? {} : { quality: route.quality }),
|
|
921
|
+
...(route.latency === undefined ? {} : { latency: route.latency }),
|
|
922
|
+
...(route.risk === undefined ? {} : { risk: route.risk }),
|
|
923
|
+
...(route.specialties === null ? {} : { specialties: route.specialties }),
|
|
924
|
+
}
|
|
925
|
+
const live = liveBenchRow(liveBench, route.model)
|
|
926
|
+
const liveScores = live?.scores ?? {}
|
|
927
|
+
const liveOverall = asScore(live?.overall)
|
|
928
|
+
const quality = clamp((asScore(liveScores?.[taskType]) ?? liveOverall ?? asScore(metadata.quality) ?? 0) + route.qualityBias)
|
|
929
|
+
const qualitySource = liveOverall !== undefined || asScore(liveScores?.[taskType]) !== undefined
|
|
930
|
+
? 'livebench' : route.quality !== undefined ? route.qualitySource : catalog ? 'catalog-heuristic' : 'unknown'
|
|
931
|
+
const userPrice = normalizedPrices[normalize(`${route.provider}/${route.model}`)] ?? normalizedPrices[normalize(route.model)]
|
|
932
|
+
// Catalog price numbers are historical hints, not a verified billable
|
|
933
|
+
// price for a user's provider account. A cost claim needs supplied USD
|
|
934
|
+
// prices for the exact configured route.
|
|
935
|
+
const pricingRow = userPrice ?? route.pricing ?? null
|
|
936
|
+
const pricingSource = userPrice ? 'user' : route.pricing ? route.pricingSource : 'unknown'
|
|
937
|
+
const specialty = specialtyMatch(metadata, taskType, liveScores)
|
|
938
|
+
rows.push({
|
|
939
|
+
provider: route.provider,
|
|
940
|
+
model: route.model,
|
|
941
|
+
metadata,
|
|
942
|
+
quality,
|
|
943
|
+
qualityBias: route.qualityBias,
|
|
944
|
+
qualitySource,
|
|
945
|
+
pricingSource,
|
|
946
|
+
liveScores,
|
|
947
|
+
liveOverall,
|
|
948
|
+
latency: metadata.latency ?? 0.5,
|
|
949
|
+
risk: metadata.risk ?? 0.2,
|
|
950
|
+
specialty,
|
|
951
|
+
reasoningEfforts: route.reasoningEfforts,
|
|
952
|
+
defaultReasoningEffort: route.defaultReasoningEffort,
|
|
953
|
+
reasoningKnown: route.reasoningKnown,
|
|
954
|
+
inputModalities: route.inputModalities,
|
|
955
|
+
pricing: pricingRow,
|
|
956
|
+
score: 0,
|
|
957
|
+
estimatedCost: pricingRow === null ? null : estimateCost({
|
|
958
|
+
id: route.model, costIn: pricingRow.input, costOut: pricingRow.output,
|
|
959
|
+
cacheRead: pricingRow.cacheRead, cacheWrite: pricingRow.cacheWrite,
|
|
960
|
+
}, text, 900, {}, cacheReadRatio, cacheWriteRatio),
|
|
961
|
+
})
|
|
962
|
+
}
|
|
963
|
+
const maxCost = Math.max(1, ...rows.filter(row => row.pricing !== null).map(row => {
|
|
964
|
+
const effective = effectivePricing(row.pricing, cacheReadRatio, cacheWriteRatio)
|
|
965
|
+
return effective.input + effective.output
|
|
966
|
+
}))
|
|
967
|
+
// Presets tilt each package's weights and quality floor; balanced is a no-op.
|
|
968
|
+
const taskNodes = taskPackages(taskType, text, complexity.band).map(task => presetId === DEFAULT_ROUTING_PRESET ? task : {
|
|
969
|
+
...task,
|
|
970
|
+
qualityFloor: presetFloor(task.qualityFloor, presetId),
|
|
971
|
+
weights: presetWeights(weightsForTask(weights, task), presetId),
|
|
972
|
+
})
|
|
973
|
+
const unassignableTasks = taskNodes.filter(task => task.type === 'vision'
|
|
974
|
+
&& !rows.some(row => row.inputModalities.length === 0 || row.inputModalities.includes('image')))
|
|
975
|
+
.map(task => task.id)
|
|
976
|
+
const budget = Number(budgetUsd)
|
|
977
|
+
const utilityPlan = solveAssignments({
|
|
978
|
+
rows,
|
|
979
|
+
tasks: taskNodes,
|
|
980
|
+
weights,
|
|
981
|
+
maxCost,
|
|
982
|
+
text,
|
|
983
|
+
complexity: complexity.band,
|
|
984
|
+
budget: Number.POSITIVE_INFINITY,
|
|
985
|
+
cacheReadRatio,
|
|
986
|
+
cacheWriteRatio,
|
|
987
|
+
})
|
|
988
|
+
const budgetPlan = budget > 0
|
|
989
|
+
? solveAssignments({
|
|
990
|
+
rows,
|
|
991
|
+
tasks: taskNodes,
|
|
992
|
+
weights,
|
|
993
|
+
maxCost,
|
|
994
|
+
text,
|
|
995
|
+
complexity: complexity.band,
|
|
996
|
+
budget,
|
|
997
|
+
cacheReadRatio,
|
|
998
|
+
cacheWriteRatio,
|
|
999
|
+
})
|
|
1000
|
+
: null
|
|
1001
|
+
const minimumCostPlan = budget > 0 && budgetPlan === null
|
|
1002
|
+
? solveAssignments({
|
|
1003
|
+
rows,
|
|
1004
|
+
tasks: taskNodes,
|
|
1005
|
+
weights,
|
|
1006
|
+
maxCost,
|
|
1007
|
+
text,
|
|
1008
|
+
complexity: complexity.band,
|
|
1009
|
+
budget: Number.POSITIVE_INFINITY,
|
|
1010
|
+
cacheReadRatio,
|
|
1011
|
+
cacheWriteRatio,
|
|
1012
|
+
minimizeCost: true,
|
|
1013
|
+
})
|
|
1014
|
+
: null
|
|
1015
|
+
const optimized = budget > 0 ? (budgetPlan ?? minimumCostPlan ?? utilityPlan) : utilityPlan
|
|
1016
|
+
const assignments = optimized?.assignments ?? []
|
|
1017
|
+
const usedRoutes = optimized?.usedRoutes ?? new Set()
|
|
1018
|
+
const constraintRelaxed = (optimized?.relaxedCount ?? 0) > 0
|
|
1019
|
+
for (const row of rows) {
|
|
1020
|
+
row.score = candidateUtility(row, taskNodes[0] ?? { type: taskType, qualityFloor: QUALITY_FLOORS[complexity.band] }, weights, maxCost, new Set(), cacheReadRatio, cacheWriteRatio).score
|
|
1021
|
+
}
|
|
1022
|
+
rows.sort((left, right) => right.score - left.score || compareRowsStable(left, right))
|
|
1023
|
+
const selectedAssignment = assignments[0]
|
|
1024
|
+
const selected = selectedAssignment?.row ?? (unassignableTasks.length > 0 ? null : rows[0] ?? null)
|
|
1025
|
+
const synthesizerAssignment = assignments.at(-1)
|
|
1026
|
+
const synthesizer = synthesizerAssignment?.row ?? (unassignableTasks.length > 0 ? null : rows.find(row => /deepseek/i.test(row.model)) ?? rows[0])
|
|
1027
|
+
const subtasks = assignments.map(({ task, row, decision }) => ({
|
|
1028
|
+
id: task.id,
|
|
1029
|
+
name: task.name,
|
|
1030
|
+
...(task.objective ? { objective: task.objective } : {}),
|
|
1031
|
+
type: task.type,
|
|
1032
|
+
difficulty: task.difficulty,
|
|
1033
|
+
recommended: row?.model ?? '待发现模型',
|
|
1034
|
+
recommendedProvider: row?.provider ?? '',
|
|
1035
|
+
qualitySource: row?.qualitySource ?? 'unknown',
|
|
1036
|
+
pricingSource: row?.pricingSource ?? 'unknown',
|
|
1037
|
+
recommendedReasoningEffort: decision?.reasoningEffort,
|
|
1038
|
+
preferredReasoningEffort: decision?.preferredReasoningEffort ?? task.preferredReasoningEffort,
|
|
1039
|
+
reasoningFit: Number(Number(decision?.reasoningFit ?? 0).toFixed(3)),
|
|
1040
|
+
purpose: task.purpose,
|
|
1041
|
+
criticality: task.criticality,
|
|
1042
|
+
qualityFloor: Number(task.qualityFloor.toFixed(3)),
|
|
1043
|
+
dependsOn: [...(task.dependsOn ?? [])],
|
|
1044
|
+
}))
|
|
1045
|
+
const costBreakdown = assignments.map(({ task, row, decision, estimatedCost, handoffPenalty }, index) => {
|
|
1046
|
+
const tokens = taskTokenBudget(text, { ...task, reasoningEffort: decision?.reasoningEffort }, complexity.band, cacheReadRatio, cacheWriteRatio)
|
|
1047
|
+
const taskEstimate = estimatedCost ?? taskCost(row, task, text, complexity.band, cacheReadRatio, cacheWriteRatio)
|
|
1048
|
+
return {
|
|
1049
|
+
stage: index + 1,
|
|
1050
|
+
purpose: task.purpose,
|
|
1051
|
+
difficulty: task.difficulty,
|
|
1052
|
+
model: row?.model ?? '待发现模型',
|
|
1053
|
+
provider: row?.provider ?? '',
|
|
1054
|
+
reasoningEffort: decision?.reasoningEffort,
|
|
1055
|
+
preferredReasoningEffort: decision?.preferredReasoningEffort,
|
|
1056
|
+
reasoningFit: Number(Number(decision?.reasoningFit ?? 0).toFixed(3)),
|
|
1057
|
+
reasoningOutputMultiplier: Number(Number(decision?.multiplier?.output ?? 1).toFixed(2)),
|
|
1058
|
+
inputTokens: tokens.inputTokens,
|
|
1059
|
+
cacheReadTokens: tokens.cacheReadTokens,
|
|
1060
|
+
cacheWriteTokens: tokens.cacheWriteTokens,
|
|
1061
|
+
outputTokens: tokens.outputTokens,
|
|
1062
|
+
estimatedCost: row?.pricing === null || row === null ? null : Number(taskEstimate.toFixed(6)),
|
|
1063
|
+
quality: row?.qualitySource === 'unknown' || row === null ? null : Number(qualityForTask(row, task.type).toFixed(3)),
|
|
1064
|
+
qualitySource: row?.qualitySource ?? 'unknown',
|
|
1065
|
+
pricingSource: row?.pricingSource ?? 'unknown',
|
|
1066
|
+
handoffPenalty: Number(Number(handoffPenalty ?? 0).toFixed(3)),
|
|
1067
|
+
}
|
|
1068
|
+
})
|
|
1069
|
+
const pricingComplete = assignments.length > 0 && assignments.every(({ row }) => row?.pricing !== null)
|
|
1070
|
+
const totalEstimate = pricingComplete ? costBreakdown.reduce((sum, row) => sum + row.estimatedCost, 0) : null
|
|
1071
|
+
const baselineRows = assignments.map(({ task }) => {
|
|
1072
|
+
const strongest = rows.reduce((best, row) => qualityForTask(row, task.type) > (best === null ? -1 : qualityForTask(best, task.type)) ? row : best, null)
|
|
1073
|
+
return { task, strongest }
|
|
1074
|
+
})
|
|
1075
|
+
const baselineCost = baselineRows.every(item => item.strongest?.pricing !== null && item.strongest !== null)
|
|
1076
|
+
? baselineRows.reduce((sum, { task, strongest }) => sum + taskCost(strongest, task, text, complexity.band, cacheReadRatio, cacheWriteRatio), 0)
|
|
1077
|
+
: null
|
|
1078
|
+
const qualityEvidenceComplete = assignments.length > 0 && assignments.every(({ row }) => ['livebench', 'route', 'user'].includes(row?.qualitySource))
|
|
1079
|
+
&& baselineRows.every(({ strongest }) => ['livebench', 'route', 'user'].includes(strongest?.qualitySource))
|
|
1080
|
+
const budgetExceeded = Number(budgetUsd) > 0 && totalEstimate !== null ? totalEstimate > Number(budgetUsd) : null
|
|
1081
|
+
const savings = baselineCost === null || totalEstimate === null || !qualityEvidenceComplete
|
|
1082
|
+
? null : baselineCost <= 0 ? 0 : clamp((baselineCost - totalEstimate) / baselineCost)
|
|
1083
|
+
const paretoPruned = (optimized?.candidatePools ?? []).reduce((sum, pool) => sum + pool.pruned, 0)
|
|
1084
|
+
const minimumFeasibleCost = rows.every(row => row.pricing !== null)
|
|
1085
|
+
? (optimized?.minimumFeasibleCost ?? minimumCostPlan?.cost ?? 0) : null
|
|
1086
|
+
const reason = selected === null
|
|
1087
|
+
? unassignableTasks.length > 0
|
|
1088
|
+
? `图像工作包 ${unassignableTasks.join('、')} 没有可用的图像模型,无法形成完整分配计划。`
|
|
1089
|
+
: '尚未发现可用模型,保留 Harness 原始模型选择。'
|
|
1090
|
+
: `${complexity.band === 'simple' ? '低复杂度优先成本、响应速度与较低推理开销' : complexity.band === 'balanced' ? '在质量、成本、推理等级、延迟与风险之间平衡' : '高复杂度执行包含推理等级的依赖感知全局约束分配'};任务类型为 ${taskType},已对 ${String(subtasks.length)} 个工作包进行 Pareto 剪枝和有界组合搜索。`
|
|
1091
|
+
return {
|
|
1092
|
+
mode,
|
|
1093
|
+
preset: presetId,
|
|
1094
|
+
complexity: { value: Number(complexity.value.toFixed(3)), band: complexity.band },
|
|
1095
|
+
compound,
|
|
1096
|
+
unassignableTasks,
|
|
1097
|
+
taskType,
|
|
1098
|
+
taskTypes: [...new Set(taskNodes.map(task => task.type).filter(type => type !== 'reasoning'))],
|
|
1099
|
+
objectiveWeights: weights,
|
|
1100
|
+
candidates: rows.slice(0, 8).map(row => {
|
|
1101
|
+
const decision = candidateUtility(row, taskNodes[0] ?? { type: taskType, qualityFloor: QUALITY_FLOORS[complexity.band], preferredReasoningEffort: complexity.band === 'simple' ? 'low' : 'medium' }, weights, maxCost, new Set(), cacheReadRatio, cacheWriteRatio)
|
|
1102
|
+
return { provider: row.provider, model: row.model, score: Number(row.score.toFixed(3)), quality: row.qualitySource === 'unknown' ? null : Number(row.quality.toFixed(3)), qualitySource: row.qualitySource, specialty: Number(row.specialty.toFixed(3)), reasoningEffort: decision.reasoningEffort, preferredReasoningEffort: decision.preferredReasoningEffort, reasoningFit: Number(decision.reasoningFit.toFixed(3)), reasoningKnown: row.reasoningKnown, reasoningEfforts: row.reasoningEfforts, estimatedCost: row.estimatedCost === null ? null : Number(row.estimatedCost.toFixed(6)), inputPrice: row.pricing?.input ?? null, outputPrice: row.pricing?.output ?? null, pricingSource: row.pricingSource }
|
|
1103
|
+
}),
|
|
1104
|
+
selected: selected === null ? null : { provider: selected.provider, model: selected.model, reasoningEffort: selectedAssignment?.decision?.reasoningEffort, estimatedCost: selected.estimatedCost === null ? null : Number(selected.estimatedCost.toFixed(6)), qualitySource: selected.qualitySource, pricingSource: selected.pricingSource },
|
|
1105
|
+
subtasks,
|
|
1106
|
+
synthesizer: synthesizer == null ? null : { provider: synthesizer.provider, model: synthesizer.model, reasoningEffort: synthesizerAssignment?.decision?.reasoningEffort },
|
|
1107
|
+
estimatedCost: totalEstimate === null ? null : Number(totalEstimate.toFixed(6)),
|
|
1108
|
+
costBreakdown,
|
|
1109
|
+
optimization: {
|
|
1110
|
+
solver: 'pareto-pruned quality-constrained beam assignment',
|
|
1111
|
+
qualityFloor: QUALITY_FLOORS[complexity.band],
|
|
1112
|
+
budgetUsd: Number(Number(budgetUsd) > 0 ? Number(budgetUsd) : 0),
|
|
1113
|
+
cacheReadRatio: normalizedCacheRatios(cacheReadRatio, cacheWriteRatio).read,
|
|
1114
|
+
cacheWriteRatio: normalizedCacheRatios(cacheReadRatio, cacheWriteRatio).write,
|
|
1115
|
+
budgetExceeded,
|
|
1116
|
+
constraintRelaxed,
|
|
1117
|
+
pricingComplete,
|
|
1118
|
+
qualityEvidenceComplete,
|
|
1119
|
+
baselineAllStrongCost: baselineCost === null ? null : Number(baselineCost.toFixed(6)),
|
|
1120
|
+
estimatedSavings: savings === null ? null : Number(savings.toFixed(4)),
|
|
1121
|
+
distinctRoutes: usedRoutes.size,
|
|
1122
|
+
handoffCount: optimized?.switches ?? 0,
|
|
1123
|
+
paretoPruned,
|
|
1124
|
+
beamWidth: ROUTING_BEAM_WIDTH,
|
|
1125
|
+
budgetFeasible: budget <= 0 ? (pricingComplete ? true : null) : budgetPlan !== null,
|
|
1126
|
+
minimumFeasibleCost: minimumFeasibleCost === null ? null : Number(Number(minimumFeasibleCost).toFixed(6)),
|
|
1127
|
+
liveBench: liveBench?.fetchedAt
|
|
1128
|
+
? { source: liveBench.source ?? 'livebench', fetchedAt: liveBench.fetchedAt, models: Object.keys(liveBench.models ?? {}).length, stale: String(liveBenchError).length > 0, error: String(liveBenchError || '') }
|
|
1129
|
+
: { source: 'experimental-baseline', fetchedAt: null, models: 0, stale: false, error: String(liveBenchError || '') },
|
|
1130
|
+
},
|
|
1131
|
+
reason,
|
|
1132
|
+
generatedAt: new Date().toISOString(),
|
|
1133
|
+
}
|
|
1134
|
+
}
|