@ljwei-stak/dsh-model-router 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dsh-plugin/client.js +3092 -0
- package/.dsh-plugin/index.mjs +1651 -0
- package/.dsh-plugin/official-tools-remote-service.mjs +104 -0
- package/.dsh-plugin/shared/harness-plan.mjs +179 -0
- package/.dsh-plugin/shared/livebench.mjs +264 -0
- package/.dsh-plugin/shared/model-profiles.mjs +142 -0
- package/.dsh-plugin/shared/official-team-runtime.mjs +411 -0
- package/.dsh-plugin/shared/official-tool-executor.mjs +801 -0
- package/.dsh-plugin/shared/official-tool-registry.mjs +138 -0
- package/.dsh-plugin/shared/official-tools-remote.mjs +173 -0
- package/.dsh-plugin/shared/official-tools-runtime.mjs +642 -0
- package/.dsh-plugin/shared/router-state.mjs +207 -0
- package/.dsh-plugin/shared/router.mjs +1134 -0
- package/.dsh-plugin/shared/routing-presets.mjs +49 -0
- package/.dsh-plugin/shared/run-ledger.mjs +348 -0
- package/.dsh-plugin/shared/security-boundaries.mjs +54 -0
- package/.dsh-plugin/shared/subscription-billing.mjs +340 -0
- package/.dsh-plugin/shared/task-executors.mjs +1154 -0
- package/.dsh-plugin/shared/tool-health.mjs +311 -0
- package/.dsh-plugin/shared/vendor-mimo-grok-adapter.mjs +308 -0
- package/.dsh-plugin/shared/vendor-minimax-adapter.mjs +247 -0
- package/.dsh-plugin/shared/zcode-bundle.mjs +208 -0
- package/.dsh-plugin/shared/zcode-installer.mjs +247 -0
- package/CHANGELOG.md +36 -0
- package/INSTALLATION_GUIDE.zh.md +134 -0
- package/LICENSE +21 -0
- package/MIGRATION.md +53 -0
- package/README.i18n.yaml +3 -0
- package/README.md +424 -0
- package/README.zh.md +413 -0
- package/cordis.patch.yml +12 -0
- package/docs/assets/candidate-pruning.svg +80 -0
- package/docs/assets/desktop-official-tools-0.9.0.png +0 -0
- package/docs/assets/router-only-0.12.0.png +0 -0
- package/docs/assets/routing-workflow.svg +96 -0
- package/docs/assets/workbench-usage.svg +119 -0
- package/package.json +161 -0
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
/** Host receiver for the official desktop's typed one-click installer RPC. */
|
|
2
|
+
import { TypertRemoteService } from '@deepseek-ai/dsh-typert-protocol'
|
|
3
|
+
import { getOfficialTool } from './shared/official-tool-registry.mjs'
|
|
4
|
+
import { officialToolExecutionCapabilities, officialToolReadiness } from './shared/official-tool-executor.mjs'
|
|
5
|
+
import {
|
|
6
|
+
defaultRunner,
|
|
7
|
+
cancelInstall,
|
|
8
|
+
ensureNpmPrefixOnPath,
|
|
9
|
+
installStatus,
|
|
10
|
+
probeAllTools,
|
|
11
|
+
probeToolWith,
|
|
12
|
+
startInstall,
|
|
13
|
+
} from './shared/official-tools-runtime.mjs'
|
|
14
|
+
import {
|
|
15
|
+
OFFICIAL_TOOLS_HOST_TYPERT,
|
|
16
|
+
OFFICIAL_TOOLS_REMOTE_NAMESPACE,
|
|
17
|
+
} from './shared/official-tools-remote.mjs'
|
|
18
|
+
|
|
19
|
+
function errorText(error) {
|
|
20
|
+
return error instanceof Error ? error.message : String(error)
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
const unavailable = () => { throw new Error('模型路由工作台服务尚未加载。') }
|
|
24
|
+
|
|
25
|
+
/** Wrap a Host operation so the client always receives a plain object. */
|
|
26
|
+
async function settled(operation) {
|
|
27
|
+
try { return { ok: true, value: await operation() } }
|
|
28
|
+
catch (error) { return { ok: false, error: errorText(error) } }
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export class OfficialToolsRemoteService extends TypertRemoteService {
|
|
32
|
+
constructor(ctx, services = {}) {
|
|
33
|
+
super(ctx, OFFICIAL_TOOLS_REMOTE_NAMESPACE)
|
|
34
|
+
this.services = services
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** 开箱体检: installed, version and login state per registry tool. */
|
|
38
|
+
health(fresh) { return settled(() => (this.services.health ?? unavailable)(fresh === true)) }
|
|
39
|
+
|
|
40
|
+
completeOnboarding() { return settled(() => (this.services.completeOnboarding ?? unavailable)()) }
|
|
41
|
+
|
|
42
|
+
/** Recent runs, spending, budget status and learned route biases. */
|
|
43
|
+
ledger() { return settled(() => (this.services.ledger ?? unavailable)()) }
|
|
44
|
+
|
|
45
|
+
rateResult(request) { return settled(() => (this.services.rate ?? unavailable)(request)) }
|
|
46
|
+
|
|
47
|
+
/** Retry one recorded step (optionally reassigned); unfinished downstream steps follow. */
|
|
48
|
+
rerunStep(request) { return settled(() => (this.services.rerun ?? unavailable)(request)) }
|
|
49
|
+
|
|
50
|
+
boundaries() { return settled(() => (this.services.boundaries ?? unavailable)()) }
|
|
51
|
+
|
|
52
|
+
/** Plan, cost estimate and the reasons that need the user's confirmation; runs nothing. */
|
|
53
|
+
previewRun(request) { return settled(() => (this.services.previewRun ?? unavailable)(request)) }
|
|
54
|
+
|
|
55
|
+
/** Execute a previewed run once every listed reason was confirmed. */
|
|
56
|
+
startRun(request) { return settled(() => (this.services.startRun ?? unavailable)(request)) }
|
|
57
|
+
|
|
58
|
+
/** Re-probe the local fixed registry; the caller cannot supply a command. */
|
|
59
|
+
async list() {
|
|
60
|
+
const tools = await probeAllTools({ fresh: true })
|
|
61
|
+
const executionReadiness = await Promise.all(tools.map(tool => tool.installed
|
|
62
|
+
? officialToolReadiness(tool.id)
|
|
63
|
+
: Promise.resolve({ id: tool.id, ready: false, reason: 'CLI 尚未安装或版本检测失败。' })))
|
|
64
|
+
return { tools, executionCapabilities: officialToolExecutionCapabilities(), executionReadiness }
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** Start one serialized fixed-registry install; return immediately for UI polling. */
|
|
68
|
+
async installTool(toolId) {
|
|
69
|
+
try {
|
|
70
|
+
await ensureNpmPrefixOnPath()
|
|
71
|
+
return { accepted: true, job: startInstall(toolId) }
|
|
72
|
+
} catch (error) {
|
|
73
|
+
return { accepted: false, error: errorText(error) }
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Stop a queued or running npm install, retaining its bounded job log. */
|
|
78
|
+
cancel(toolId) {
|
|
79
|
+
try {
|
|
80
|
+
return { accepted: true, job: cancelInstall(toolId) }
|
|
81
|
+
} catch (error) {
|
|
82
|
+
return { accepted: false, error: errorText(error) }
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Read bounded progress; after command success, verify that the CLI resolves. */
|
|
87
|
+
async status(toolId) {
|
|
88
|
+
const job = installStatus(toolId)
|
|
89
|
+
if (job?.status !== 'succeeded') return { job }
|
|
90
|
+
await ensureNpmPrefixOnPath()
|
|
91
|
+
const tool = getOfficialTool(toolId)
|
|
92
|
+
const postInstallProbe = tool === null ? null
|
|
93
|
+
: await probeToolWith(tool, defaultRunner, { cacheMs: 0 })
|
|
94
|
+
return { job, postInstallProbe }
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** Registration follows the Host plugin fiber; unload withdraws all endpoints. */
|
|
101
|
+
export function registerOfficialToolsRemote(ctx, services = {}) {
|
|
102
|
+
new OfficialToolsRemoteService(ctx, services)
|
|
103
|
+
ctx.effect(() => ctx.typert.register(OFFICIAL_TOOLS_HOST_TYPERT), 'model-router-galgame: official tools remote descriptors')
|
|
104
|
+
}
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
import { buildPlan, detectTaskTypes } from './router.mjs'
|
|
2
|
+
import { OFFICIAL_TOOLS, toolForProvider } from './official-tool-registry.mjs'
|
|
3
|
+
import { normalizeExecutionPreference } from './model-profiles.mjs'
|
|
4
|
+
|
|
5
|
+
const clean = value => typeof value === 'string' ? value.trim() : ''
|
|
6
|
+
|
|
7
|
+
const HEADLESS_TOOL_IDS = new Set(['claude-code', 'codex', 'gemini'])
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Resolve the intended execution channel for one provider without touching
|
|
11
|
+
* the filesystem: `official-cli` when the provider maps to a registry tool
|
|
12
|
+
* that the caller reports as installed, `harness-llm` otherwise. The probe
|
|
13
|
+
* snapshot comes from the Host caller, which owns the real process boundary.
|
|
14
|
+
*/
|
|
15
|
+
export function channelForProvider(provider, installedToolIds = [], runnableToolIds = [], route = null, loggedOutToolIds = []) {
|
|
16
|
+
const preference = normalizeExecutionPreference(route?.execution)
|
|
17
|
+
const tool = toolForProvider(provider)
|
|
18
|
+
if (preference === 'api') {
|
|
19
|
+
return {
|
|
20
|
+
kind: 'harness-llm', preference,
|
|
21
|
+
...(tool ? { tool: tool.id, label: tool.label } : {}),
|
|
22
|
+
detail: '该模型配置为只使用模型目录 API。',
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
if (!tool || tool.unsupported) {
|
|
26
|
+
return { kind: 'harness-llm', preference, detail: '通过官方模型目录 API 调用。' }
|
|
27
|
+
}
|
|
28
|
+
const installed = Array.isArray(installedToolIds) && installedToolIds.includes(tool.id)
|
|
29
|
+
if (installed && Array.isArray(loggedOutToolIds) && loggedOutToolIds.includes(tool.id)) {
|
|
30
|
+
return {
|
|
31
|
+
kind: 'harness-llm', preference, tool: tool.id, label: tool.label, loginRequired: true,
|
|
32
|
+
detail: `${tool.label} 已安装但未登录(开箱体检结果);实际调用直接使用模型目录 API,不等待 CLI 失败。`,
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
const runnable = installed && Array.isArray(runnableToolIds) && runnableToolIds.includes(tool.id)
|
|
36
|
+
const headless = installed && HEADLESS_TOOL_IDS.has(tool.id)
|
|
37
|
+
if (runnable || headless) {
|
|
38
|
+
return {
|
|
39
|
+
kind: 'official-cli', preference, tool: tool.id, label: tool.label,
|
|
40
|
+
detail: tool.id === 'zcode'
|
|
41
|
+
? 'ZCode 已安装,插件可调用其官方编程代理;3.14.3 的 CLI 使用自身配置的默认模型,不能保证与 Harness 建议模型一致。'
|
|
42
|
+
: headless && !runnable
|
|
43
|
+
? `${tool.label} 已安装。分配到该模型的任务会先走官方无界面命令;命令缺失或失败时回退模型目录 API。`
|
|
44
|
+
: `${tool.label} 已安装,插件可托管调用其官方 CLI;Harness 模型目录与厂商 CLI 名称可能不同,团队无法确认映射时使用 CLI 默认模型,实际模型仍须核对运行记录。`,
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
return {
|
|
48
|
+
kind: 'harness-llm',
|
|
49
|
+
preference,
|
|
50
|
+
tool: tool.id,
|
|
51
|
+
label: tool.label,
|
|
52
|
+
detail: installed
|
|
53
|
+
? `${tool.label} 已安装,但当前平台缺少经核验的托管执行适配器;实际调用使用官方模型目录 API。`
|
|
54
|
+
: `${tool.label} 未安装;实际调用使用官方模型目录 API。可在工作台一键安装,或运行 /tools install ${tool.id}。`,
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function annotate(channel) {
|
|
59
|
+
return {
|
|
60
|
+
executionChannel: channel.kind,
|
|
61
|
+
...channel.tool ? { channelTool: channel.tool } : {},
|
|
62
|
+
...channel.label ? { channelLabel: channel.label } : {},
|
|
63
|
+
...channel.preference ? { executionPreference: channel.preference } : {},
|
|
64
|
+
...channel.loginRequired ? { loginRequired: true } : {},
|
|
65
|
+
channelDetail: channel.detail,
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* The same deterministic planning projection is used by the Host tool and
|
|
71
|
+
* by the Desktop panel. Model discovery stays on the official side of each
|
|
72
|
+
* runtime and only the public provider/model directory crosses this boundary.
|
|
73
|
+
*/
|
|
74
|
+
export function createPlanFromRoutes(task, availableRoutes, {
|
|
75
|
+
mode = 'single', budgetUsd = 0, installedToolIds = [], runnableToolIds = [],
|
|
76
|
+
pricing = {}, liveBench = null, cacheReadRatio = 0, cacheWriteRatio = 0,
|
|
77
|
+
directProvider = '', directModel = '', preset = 'balanced', loggedOutToolIds = [],
|
|
78
|
+
} = {}) {
|
|
79
|
+
const taskText = clean(task)
|
|
80
|
+
if (!taskText) throw new Error('task must contain text')
|
|
81
|
+
const requestedDirect = mode === 'direct' || clean(directProvider) !== '' || clean(directModel) !== ''
|
|
82
|
+
const selectedMode = mode === 'team' ? 'team' : 'single'
|
|
83
|
+
let routes = Array.isArray(availableRoutes) ? availableRoutes : []
|
|
84
|
+
let directRoute = null
|
|
85
|
+
if (requestedDirect) {
|
|
86
|
+
const provider = clean(directProvider)
|
|
87
|
+
const model = clean(directModel)
|
|
88
|
+
if (!provider || !model) throw new Error('指定单一模型需要同时提供 provider 和 model')
|
|
89
|
+
const match = routes.filter(route => route.provider === provider && route.model === model)
|
|
90
|
+
if (match.length !== 1) throw new Error(`指定模型 ${provider}/${model} 不在当前模型目录中`)
|
|
91
|
+
routes = match
|
|
92
|
+
directRoute = { provider, model }
|
|
93
|
+
}
|
|
94
|
+
const installedIds = Array.isArray(installedToolIds) ? installedToolIds.filter(Boolean) : []
|
|
95
|
+
const channelCache = new Map()
|
|
96
|
+
const channelOf = (provider, model) => {
|
|
97
|
+
const route = routes.find(item => item.provider === provider && item.model === model)
|
|
98
|
+
const key = `${String(provider ?? '')}\0${String(model ?? '')}\0${route?.execution ?? ''}`
|
|
99
|
+
if (!channelCache.has(key)) channelCache.set(key, channelForProvider(provider, installedIds, runnableToolIds, route, loggedOutToolIds))
|
|
100
|
+
return channelCache.get(key)
|
|
101
|
+
}
|
|
102
|
+
const needsImage = detectTaskTypes(taskText).includes('vision')
|
|
103
|
+
const plan = buildPlan({
|
|
104
|
+
text: taskText,
|
|
105
|
+
available: routes,
|
|
106
|
+
mode: directRoute ? 'single' : selectedMode,
|
|
107
|
+
budgetUsd: Math.max(0, Number.isFinite(budgetUsd) ? budgetUsd : 0),
|
|
108
|
+
pricing,
|
|
109
|
+
liveBench,
|
|
110
|
+
cacheReadRatio,
|
|
111
|
+
cacheWriteRatio,
|
|
112
|
+
preset,
|
|
113
|
+
})
|
|
114
|
+
const selectedChannel = plan.selected ? channelOf(plan.selected.provider, plan.selected.model) : null
|
|
115
|
+
return {
|
|
116
|
+
...plan,
|
|
117
|
+
mode: directRoute ? 'direct' : plan.mode,
|
|
118
|
+
routingBypassed: directRoute !== null,
|
|
119
|
+
directRoute,
|
|
120
|
+
...(directRoute ? { reason: `已指定 ${directRoute.provider}/${directRoute.model},不与其他已配置模型比较。复杂度仍按任务文本估计。` } : {}),
|
|
121
|
+
contractVersion: 2,
|
|
122
|
+
availableRoutes: routes,
|
|
123
|
+
...(selectedChannel ? annotate(selectedChannel) : {}),
|
|
124
|
+
availabilityNotice: '模型目录列出的路线尚未验证当前凭据和网络;实际可用性以官方适配器调用结果为准。',
|
|
125
|
+
pricingNotice: plan.estimatedCost === null
|
|
126
|
+
? '部分路线尚未配置该供应商的美元输入/输出单价,无法计算可靠的总费用与节省比例;请在模型价格设置中补齐。'
|
|
127
|
+
: '费用按已提供的美元单价和估计 token 数计算,不是供应商账单,也不是硬性支出上限。',
|
|
128
|
+
qualityNotice: plan.optimization.qualityEvidenceComplete
|
|
129
|
+
? '模型质量使用已提供评分或基准数据估计,仍需实际任务验证。'
|
|
130
|
+
: '部分模型质量缺少可核验评分;目录启发式只供选择参考,质量门槛和节省比例无法保证。',
|
|
131
|
+
modalityNotice: needsImage
|
|
132
|
+
? plan.unassignableTasks.length > 0
|
|
133
|
+
? '图像工作包没有可确认支持图像输入的路线,当前计划无法完整分配;请在官方模型目录配置支持图像的模型。'
|
|
134
|
+
: routes.some(route => !Array.isArray(route.inputModalities) || route.inputModalities.length === 0)
|
|
135
|
+
? '图像工作包只分给已声明图像能力或未声明输入能力的模型;未声明能力的模型仍需实际验证。其他文本工作包可继续使用经济型文本模型。'
|
|
136
|
+
: '图像工作包只分给明确支持图像输入的模型;其他文本工作包可继续使用经济型文本模型。'
|
|
137
|
+
: null,
|
|
138
|
+
toolNotice: '执行渠道按官方工具注册表和已核验适配器标注:official-cli 表示该厂商官方 CLI 已安装且可托管执行;harness-llm 表示通过官方模型目录调用。可在工作台查看安装与执行支持状态。',
|
|
139
|
+
team: {
|
|
140
|
+
requested: selectedMode === 'team',
|
|
141
|
+
recommended: selectedMode === 'team' && plan.complexity.band === 'complex' && plan.subtasks.length > 1,
|
|
142
|
+
handoff: '可在官方会话调用 model_router_team_execute 托管执行当前平台支持的官方 CLI 工作包;官方 Agent Teams 可协作管理任务,但成员模型由宿主配置,不能直接按本计划逐个切换。',
|
|
143
|
+
workPackages: plan.subtasks.map((item, index) => ({
|
|
144
|
+
id: item.id,
|
|
145
|
+
name: item.name,
|
|
146
|
+
...(item.objective ? { objective: item.objective } : {}),
|
|
147
|
+
type: item.type,
|
|
148
|
+
purpose: item.purpose,
|
|
149
|
+
difficulty: item.difficulty,
|
|
150
|
+
qualitySource: item.qualitySource,
|
|
151
|
+
pricingSource: item.pricingSource,
|
|
152
|
+
dependsOn: item.dependsOn,
|
|
153
|
+
recommendedProvider: item.recommendedProvider,
|
|
154
|
+
recommendedModel: item.recommended,
|
|
155
|
+
estimatedCost: plan.costBreakdown[index]?.estimatedCost ?? null,
|
|
156
|
+
...item.recommendedReasoningEffort ? { recommendedReasoningEffort: item.recommendedReasoningEffort } : {},
|
|
157
|
+
...annotate(channelOf(item.recommendedProvider, item.recommended)),
|
|
158
|
+
verificationChecklist: item.purpose === 'synthesis'
|
|
159
|
+
? ['核对各工作包交付物与依赖', '记录冲突、未解决事项和最终验收结果']
|
|
160
|
+
: item.type === 'code'
|
|
161
|
+
? ['说明改动文件与接口', '运行与改动相关的验证并记录结果', '列出尚未完成的边界情况']
|
|
162
|
+
: item.type === 'research'
|
|
163
|
+
? ['列出来源、日期和可核对的结论', '标出推断与不确定事项']
|
|
164
|
+
: ['列出交付内容和验收依据', '标出未完成事项'],
|
|
165
|
+
})),
|
|
166
|
+
},
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
/** Registry summary for tools that a plan or panel may want to display. */
|
|
171
|
+
export function officialToolSummaries() {
|
|
172
|
+
return OFFICIAL_TOOLS.map(tool => ({
|
|
173
|
+
id: tool.id,
|
|
174
|
+
label: tool.label,
|
|
175
|
+
vendor: tool.vendor,
|
|
176
|
+
purpose: tool.purpose,
|
|
177
|
+
unsupported: tool.unsupported === true,
|
|
178
|
+
}))
|
|
179
|
+
}
|
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* LiveBench snapshot adapter.
|
|
3
|
+
*
|
|
4
|
+
* The public LiveBench site currently publishes versioned CSV/JSON assets
|
|
5
|
+
* rather than a stable JSON API. This adapter understands both that official
|
|
6
|
+
* layout and a user-supplied JSON/CSV mirror. A refresh failure is non-fatal:
|
|
7
|
+
* the Host keeps the last successful snapshot and the router exposes the
|
|
8
|
+
* fallback state in its audit record.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
const DEFAULT_RELEASE = '2026-06-25'
|
|
12
|
+
const TASK_ALIASES = Object.freeze({
|
|
13
|
+
reasoning: ['reasoning', 'reasoning_score', 'hard_reasoning'],
|
|
14
|
+
code: ['code', 'coding', 'coding_score'],
|
|
15
|
+
math: ['math', 'mathematics', 'math_score'],
|
|
16
|
+
research: ['research', 'retrieval', 'knowledge', 'data_analysis'],
|
|
17
|
+
writing: ['writing', 'creative_writing', 'language'],
|
|
18
|
+
vision: ['vision', 'multimodal', 'visual'],
|
|
19
|
+
summarization: ['summarization', 'summary', 'if'],
|
|
20
|
+
classification: ['classification', 'instruction_following'],
|
|
21
|
+
})
|
|
22
|
+
|
|
23
|
+
const CATEGORY_TO_TASK = Object.freeze({
|
|
24
|
+
reasoning: 'reasoning',
|
|
25
|
+
coding: 'code',
|
|
26
|
+
'agentic coding': 'code',
|
|
27
|
+
mathematics: 'math',
|
|
28
|
+
'data analysis': 'research',
|
|
29
|
+
language: 'writing',
|
|
30
|
+
if: 'summarization',
|
|
31
|
+
vision: 'vision',
|
|
32
|
+
multimodal: 'vision',
|
|
33
|
+
})
|
|
34
|
+
|
|
35
|
+
const clamp = (value, min = 0, max = 1) => Math.max(min, Math.min(max, value))
|
|
36
|
+
|
|
37
|
+
function normalized(value) {
|
|
38
|
+
return String(value ?? '').toLowerCase().replace(/[^a-z0-9]+/g, '')
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
function asScore(value) {
|
|
42
|
+
const number = Number(value)
|
|
43
|
+
if (!Number.isFinite(number)) return undefined
|
|
44
|
+
return clamp(number > 1 ? number / 100 : number)
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
function modelRows(payload) {
|
|
48
|
+
if (Array.isArray(payload)) return payload
|
|
49
|
+
if (payload !== null && typeof payload === 'object') {
|
|
50
|
+
for (const key of ['models', 'data', 'leaderboard', 'results', 'entries', 'rows']) {
|
|
51
|
+
if (Array.isArray(payload[key])) return payload[key]
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
return []
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function rowName(row) {
|
|
58
|
+
if (row === null || typeof row !== 'object') return ''
|
|
59
|
+
return String(row.model ?? row.model_name ?? row.name ?? row.id ?? row.slug ?? '')
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function rowScores(row) {
|
|
63
|
+
const source = row?.scores ?? row?.categories ?? row?.benchmark ?? row
|
|
64
|
+
const scores = {}
|
|
65
|
+
if (source === null || typeof source !== 'object') return scores
|
|
66
|
+
for (const [task, aliases] of Object.entries(TASK_ALIASES)) {
|
|
67
|
+
for (const alias of aliases) {
|
|
68
|
+
const score = asScore(source[alias])
|
|
69
|
+
if (score !== undefined) {
|
|
70
|
+
scores[task] = score
|
|
71
|
+
break
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return scores
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** Normalize a provider response into `{ models: Record<normalizedId, row> }`. */
|
|
79
|
+
export function normalizeLiveBenchPayload(payload, fetchedAt = Date.now(), source = 'livebench') {
|
|
80
|
+
const models = {}
|
|
81
|
+
for (const row of modelRows(payload)) {
|
|
82
|
+
const id = normalized(rowName(row))
|
|
83
|
+
if (id === '') continue
|
|
84
|
+
const scores = rowScores(row)
|
|
85
|
+
const values = Object.values(scores)
|
|
86
|
+
const overall = asScore(row?.overall ?? row?.score ?? row?.livebench_score)
|
|
87
|
+
?? (values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : undefined)
|
|
88
|
+
if (overall === undefined) continue
|
|
89
|
+
models[id] = {
|
|
90
|
+
overall: Number(clamp(overall).toFixed(4)),
|
|
91
|
+
scores,
|
|
92
|
+
rank: Number.isFinite(Number(row?.rank)) ? Number(row.rank) : undefined,
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
return { source, fetchedAt, models }
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/** Parse a small RFC-4180-compatible CSV without adding a runtime dependency. */
|
|
99
|
+
export function parseCsv(text) {
|
|
100
|
+
const rows = []
|
|
101
|
+
let row = []
|
|
102
|
+
let field = ''
|
|
103
|
+
let quoted = false
|
|
104
|
+
const value = String(text ?? '').replace(/^\uFEFF/, '')
|
|
105
|
+
for (let index = 0; index < value.length; index += 1) {
|
|
106
|
+
const char = value[index]
|
|
107
|
+
if (quoted) {
|
|
108
|
+
if (char === '"' && value[index + 1] === '"') {
|
|
109
|
+
field += '"'
|
|
110
|
+
index += 1
|
|
111
|
+
} else if (char === '"') {
|
|
112
|
+
quoted = false
|
|
113
|
+
} else {
|
|
114
|
+
field += char
|
|
115
|
+
}
|
|
116
|
+
} else if (char === '"') {
|
|
117
|
+
quoted = true
|
|
118
|
+
} else if (char === ',') {
|
|
119
|
+
row.push(field)
|
|
120
|
+
field = ''
|
|
121
|
+
} else if (char === '\n') {
|
|
122
|
+
row.push(field.replace(/\r$/, ''))
|
|
123
|
+
if (row.some(cell => cell.trim() !== '')) rows.push(row)
|
|
124
|
+
row = []
|
|
125
|
+
field = ''
|
|
126
|
+
} else {
|
|
127
|
+
field += char
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
row.push(field.replace(/\r$/, ''))
|
|
131
|
+
if (row.some(cell => cell.trim() !== '')) rows.push(row)
|
|
132
|
+
if (rows.length === 0) return []
|
|
133
|
+
const headers = rows[0].map(header => header.trim())
|
|
134
|
+
return rows.slice(1).map(cells => Object.fromEntries(headers.map((header, index) => [header, cells[index] ?? ''])))
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
function mean(values) {
|
|
138
|
+
const numbers = values.map(asScore).filter(value => value !== undefined)
|
|
139
|
+
return numbers.length === 0 ? undefined : numbers.reduce((sum, value) => sum + value, 0) / numbers.length
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
function officialCsvPayload(csv, categories = {}) {
|
|
143
|
+
const rows = parseCsv(csv)
|
|
144
|
+
const categoryColumns = new Map()
|
|
145
|
+
for (const [label, columns] of Object.entries(categories ?? {})) {
|
|
146
|
+
const task = CATEGORY_TO_TASK[String(label).trim().toLowerCase()]
|
|
147
|
+
if (task && Array.isArray(columns)) categoryColumns.set(task, columns)
|
|
148
|
+
}
|
|
149
|
+
const models = rows.map(row => {
|
|
150
|
+
const scores = {}
|
|
151
|
+
for (const [task, columns] of categoryColumns.entries()) {
|
|
152
|
+
const score = mean(columns.map(column => row[column]))
|
|
153
|
+
if (score !== undefined) scores[task] = score
|
|
154
|
+
}
|
|
155
|
+
// A mirror may already provide normalized task columns, so preserve them.
|
|
156
|
+
for (const [task, aliases] of Object.entries(TASK_ALIASES)) {
|
|
157
|
+
if (scores[task] !== undefined) continue
|
|
158
|
+
const score = mean(aliases.map(alias => row[alias]))
|
|
159
|
+
if (score !== undefined) scores[task] = score
|
|
160
|
+
}
|
|
161
|
+
return { model: row.model, scores, overall: mean(Object.values(row).slice(1)) }
|
|
162
|
+
})
|
|
163
|
+
return { models }
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
function releaseCandidatesFromText(text) {
|
|
167
|
+
const source = String(text ?? '')
|
|
168
|
+
const matches = []
|
|
169
|
+
// The current bundle has `const pe=["2024-...", ...]`. Restrict parsing to
|
|
170
|
+
// an actual array assignment so unrelated build timestamps are not treated
|
|
171
|
+
// as releases.
|
|
172
|
+
const arrays = source.matchAll(/(?:const|let|var)\s+\w+\s*=\s*\[((?:\s*["']20\d{2}-\d{2}-\d{2}["']\s*,?)+)\]/g)
|
|
173
|
+
for (const match of arrays) matches.push(...(match[1].match(/20\d{2}-\d{2}-\d{2}/g) ?? []))
|
|
174
|
+
if (matches.length === 0) matches.push(...(source.match(/20\d{2}-\d{2}-\d{2}/g) ?? []))
|
|
175
|
+
return [...new Set([...matches, DEFAULT_RELEASE])].sort().reverse()
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
function absoluteUrl(base, path) {
|
|
179
|
+
return new URL(path, base.endsWith('/') ? base : `${base}/`).toString()
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
async function readResponse(response) {
|
|
183
|
+
const type = String(response.headers?.get?.('content-type') ?? '').toLowerCase()
|
|
184
|
+
const text = await response.text()
|
|
185
|
+
return { type, text }
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
async function fetchWithTimeout(fetchImpl, url, signal) {
|
|
189
|
+
const response = await fetchImpl(url, {
|
|
190
|
+
headers: { accept: 'application/json,text/csv,text/html' },
|
|
191
|
+
signal,
|
|
192
|
+
})
|
|
193
|
+
if (!response.ok) throw new Error(`LiveBench returned HTTP ${String(response.status)} for ${url}`)
|
|
194
|
+
return response
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
async function officialSnapshot({ endpoint, fetchImpl, signal, fetchedAt }) {
|
|
198
|
+
const base = new URL(endpoint).origin
|
|
199
|
+
let releases = [DEFAULT_RELEASE]
|
|
200
|
+
try {
|
|
201
|
+
const htmlResponse = await fetchWithTimeout(fetchImpl, base, signal)
|
|
202
|
+
const html = (await readResponse(htmlResponse)).text
|
|
203
|
+
const scriptPaths = [...html.matchAll(/<script[^>]+src=["']([^"']+\.js)["']/gi)].map(match => match[1])
|
|
204
|
+
const scriptPath = scriptPaths.at(-1)
|
|
205
|
+
if (scriptPath !== undefined) {
|
|
206
|
+
const scriptResponse = await fetchWithTimeout(fetchImpl, absoluteUrl(base, scriptPath), signal)
|
|
207
|
+
releases = releaseCandidatesFromText((await readResponse(scriptResponse)).text)
|
|
208
|
+
} else {
|
|
209
|
+
releases = releaseCandidatesFromText(html)
|
|
210
|
+
}
|
|
211
|
+
} catch {
|
|
212
|
+
// A CDN may deny HTML/JS while still serving a pinned release asset.
|
|
213
|
+
}
|
|
214
|
+
let lastError
|
|
215
|
+
for (const release of releases) {
|
|
216
|
+
const token = release.replaceAll('-', '_')
|
|
217
|
+
try {
|
|
218
|
+
const tableResponse = await fetchWithTimeout(fetchImpl, absoluteUrl(base, `table_${token}.csv`), signal)
|
|
219
|
+
const categoryResponse = await fetchWithTimeout(fetchImpl, absoluteUrl(base, `categories_${token}.json`), signal)
|
|
220
|
+
const table = (await readResponse(tableResponse)).text
|
|
221
|
+
const categories = JSON.parse((await readResponse(categoryResponse)).text)
|
|
222
|
+
return normalizeLiveBenchPayload(officialCsvPayload(table, categories), fetchedAt, `livebench:${release}`)
|
|
223
|
+
} catch (error) {
|
|
224
|
+
lastError = error
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
throw lastError ?? new Error('LiveBench release assets are unavailable')
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/**
|
|
231
|
+
* Fetch one LiveBench snapshot. The official root automatically discovers the
|
|
232
|
+
* newest release. A custom endpoint may be JSON, CSV, or a `{release}` URL.
|
|
233
|
+
*/
|
|
234
|
+
export async function fetchLiveBenchSnapshot({
|
|
235
|
+
endpoint = 'https://livebench.ai',
|
|
236
|
+
fetchImpl = globalThis.fetch,
|
|
237
|
+
timeoutMs = 8000,
|
|
238
|
+
} = {}) {
|
|
239
|
+
if (typeof fetchImpl !== 'function') throw new Error('fetch is unavailable')
|
|
240
|
+
const controller = new AbortController()
|
|
241
|
+
const timer = setTimeout(() => controller.abort(), timeoutMs)
|
|
242
|
+
const fetchedAt = Date.now()
|
|
243
|
+
try {
|
|
244
|
+
const rawEndpoint = String(endpoint).trim()
|
|
245
|
+
const parsed = new URL(rawEndpoint)
|
|
246
|
+
if ((parsed.hostname === 'livebench.ai' || parsed.hostname === 'www.livebench.ai')
|
|
247
|
+
&& (parsed.pathname === '' || parsed.pathname === '/')) {
|
|
248
|
+
return await officialSnapshot({ endpoint: parsed.toString(), fetchImpl, signal: controller.signal, fetchedAt })
|
|
249
|
+
}
|
|
250
|
+
const url = parsed.toString().replace('{release}', DEFAULT_RELEASE)
|
|
251
|
+
const response = await fetchWithTimeout(fetchImpl, url, controller.signal)
|
|
252
|
+
const { type, text } = await readResponse(response)
|
|
253
|
+
const isCsv = type.includes('csv') || /\.csv(?:$|\?)/i.test(parsed.pathname)
|
|
254
|
+
if (isCsv) return normalizeLiveBenchPayload(officialCsvPayload(text), fetchedAt, 'livebench-csv-mirror')
|
|
255
|
+
return normalizeLiveBenchPayload(JSON.parse(text), fetchedAt, 'livebench-json-mirror')
|
|
256
|
+
} finally {
|
|
257
|
+
clearTimeout(timer)
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/** Return a model's benchmark row using the same normalization as the adapter. */
|
|
262
|
+
export function liveBenchRow(snapshot, model) {
|
|
263
|
+
return snapshot?.models?.[normalized(model)] ?? null
|
|
264
|
+
}
|