@ljwei-stak/model-router-galgame 0.4.30 → 0.4.32

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,6 +6,7 @@ import {
6
6
  DEFAULT_ROUTER_SETTINGS,
7
7
  MODEL_ROUTER_SETTINGS_NAMESPACE,
8
8
  nextCollaborationStage,
9
+ selectReasoningEffort,
9
10
  textFromMessages,
10
11
  } from './shared/router.mjs'
11
12
  import { formatErrorChain } from './shared/error-diagnostics.mjs'
@@ -116,6 +117,7 @@ function stateFor(agent) {
116
117
  directoryPromise: null,
117
118
  turn: null,
118
119
  failedModels: new Set(),
120
+ routeCooldowns: new Map(),
119
121
  lastTarget: null,
120
122
  lastStep: 0,
121
123
  collaboration: null,
@@ -158,14 +160,29 @@ async function discover(ctx, state) {
158
160
  }
159
161
  try {
160
162
  const models = await ctx.llm.listModels(provider.id)
161
- for (const model of models) {
162
- const inputModalities = model.inputModalities ?? model.input ?? []
163
- routes.push({
163
+ const resolvedRoutes = await Promise.all(models.map(async model => {
164
+ let resolved
165
+ try {
166
+ resolved = typeof ctx.llm.resolveModelInfo === 'function'
167
+ ? await ctx.llm.resolveModelInfo(provider.id, model.id)
168
+ : undefined
169
+ } catch (error) {
170
+ ctx.logger?.debug?.(`model-router: reasoning metadata unavailable for ${provider.id}/${model.id}: ${String(error)}`)
171
+ }
172
+ const inputModalities = resolved?.inputModalities ?? model.inputModalities ?? model.input ?? []
173
+ const reasoning = resolved?.reasoning
174
+ return {
164
175
  provider: provider.id,
165
176
  model: model.id,
166
177
  inputModalities: Array.isArray(inputModalities) ? [...inputModalities] : [],
167
- })
168
- }
178
+ ...(resolved === undefined ? {} : {
179
+ reasoningKnown: true,
180
+ reasoningEfforts: Array.isArray(reasoning?.efforts) ? reasoning.efforts.map(effort => effort.id) : [],
181
+ ...(reasoning?.defaultEffort === undefined ? {} : { defaultReasoningEffort: reasoning.defaultEffort }),
182
+ }),
183
+ }
184
+ }))
185
+ routes.push(...resolvedRoutes)
169
186
  } catch (error) {
170
187
  ctx.logger?.debug?.(`model-router: model discovery failed for ${provider.id}: ${String(error)}`)
171
188
  }
@@ -290,8 +307,8 @@ function analysisMessage(plan) {
290
307
  '[Model Router 路由分析]',
291
308
  `任务类型:${plan.taskType};复杂度:${plan.complexity?.band ?? 'unknown'}(${Math.round((plan.complexity?.value ?? 0) * 100)}%)`,
292
309
  Array.isArray(plan.taskTypes) && plan.taskTypes.length > 1 ? `业务方向:${plan.taskTypes.join('、')}(分别建立执行工作包)` : '',
293
- `本轮权重:质量 ${Math.round((weights.quality ?? 0) * 100)}%,成本 ${Math.round((weights.cost ?? 0) * 100)}%,延迟 ${Math.round((weights.latency ?? 0) * 100)}%,专长 ${Math.round((weights.specialty ?? 0) * 100)}%,风险 ${Math.round((weights.risk ?? 0) * 100)}%`,
294
- `质量下限:${Math.round(Number(plan.optimization?.qualityFloor ?? 0) * 100)}%;首选路由:${selected}`,
310
+ `本轮权重:质量 ${Math.round((weights.quality ?? 0) * 100)}%,成本 ${Math.round((weights.cost ?? 0) * 100)}%,推理等级 ${Math.round((weights.reasoning ?? 0) * 100)}%,延迟 ${Math.round((weights.latency ?? 0) * 100)}%,专长 ${Math.round((weights.specialty ?? 0) * 100)}%,风险 ${Math.round((weights.risk ?? 0) * 100)}%`,
311
+ `质量下限:${Math.round(Number(plan.optimization?.qualityFloor ?? 0) * 100)}%;首选路由:${selected}${plan.selected?.reasoningEffort ? `;推理等级:${plan.selected.reasoningEffort}` : ';推理等级:提供方默认'}`,
295
312
  `预计总费用:$${Number(plan.estimatedCost ?? 0).toFixed(6)};相对全高质量基线节省:$${Number((plan.optimization?.baselineAllStrongCost ?? 0) - (plan.estimatedCost ?? 0)).toFixed(6)}`,
296
313
  `缓存计费比例:读取 ${Math.round(Number(plan.optimization?.cacheReadRatio ?? 0) * 100)}%,写入 ${Math.round(Number(plan.optimization?.cacheWriteRatio ?? 0) * 100)}%(未填写时按普通输入计费)`,
297
314
  Number(plan.optimization?.budgetUsd ?? 0) > 0 ? `预算上限:$${Number(plan.optimization.budgetUsd).toFixed(6)};${plan.optimization.budgetExceeded ? '仍超预算,已在质量下限内尽量压缩' : '满足预算约束'}` : '',
@@ -327,17 +344,43 @@ function routeKey(provider, model) {
327
344
  }
328
345
 
329
346
  function modelFallbackError(failure) {
330
- const text = `${String(failure?.code ?? '')} ${String(failure?.message ?? '')}`.toLowerCase()
331
- return /region|not available|not supported|freeusagelimit|rate limit|too many requests|\b(?:403|404|429)\b/.test(text)
347
+ const text = `${String(failure?.code ?? '')} ${String(failure?.message ?? '')} ${formatErrorChain(failure)}`.toLowerCase()
348
+ return /unsupported[_ -]reasoning[_ -]effort|no[_ -]adapter|invalid[_ -]model|invalid[_ -]credential|missing[_ -]credential|auth(?:entication|orization)?|api key|quota|region|not available|not supported|unsupported provider stream event|provider protocol|invalid (?:stream|event)|codex\.rate_limits|freeusagelimit|rate limit|too many requests|\b(?:401|403|404|429)\b/.test(text)
349
+ }
350
+
351
+ function failureCooldownMs(failure) {
352
+ const text = `${String(failure?.code ?? '')} ${String(failure?.message ?? '')} ${formatErrorChain(failure)}`.toLowerCase()
353
+ const requested = Number(failure?.retryAfterMs)
354
+ if (Number.isFinite(requested) && requested > 0) return Math.min(Math.max(requested, 30000), 30 * 60 * 1000)
355
+ if (/auth|api key|credential|\b401\b/.test(text)) return 30 * 60 * 1000
356
+ if (/unsupported provider stream event|provider protocol|codex\.rate_limits/.test(text)) return 10 * 60 * 1000
357
+ return 2 * 60 * 1000
358
+ }
359
+
360
+ function routeTemporarilyUnavailable(state, key) {
361
+ if (state.failedModels.has(key)) return true
362
+ const deadline = Number(state.routeCooldowns.get(key) ?? 0)
363
+ if (deadline <= Date.now()) {
364
+ state.routeCooldowns.delete(key)
365
+ return false
366
+ }
367
+ return true
332
368
  }
333
369
 
334
- function nextAvailableTarget(state) {
370
+ function nextAvailableTarget(state, step = state.lastStep) {
335
371
  const candidates = Array.isArray(state.plan?.candidates) ? state.plan.candidates : []
372
+ const assignment = state.plan?.subtasks?.[Math.max(0, Number(step || 1) - 1)]
373
+ const preferredEffort = assignment?.preferredReasoningEffort ?? assignment?.recommendedReasoningEffort ?? 'medium'
336
374
  for (const candidate of candidates) {
337
375
  const key = routeKey(candidate.provider, candidate.model)
338
- if (state.failedModels.has(key)) continue
376
+ if (routeTemporarilyUnavailable(state, key)) continue
339
377
  const route = state.available.find(entry => entry.provider === candidate.provider && entry.model === candidate.model)
340
- if (route !== undefined) return route
378
+ if (route !== undefined) {
379
+ return {
380
+ ...route,
381
+ reasoningEffort: selectReasoningEffort(route.reasoningEfforts, preferredEffort),
382
+ }
383
+ }
341
384
  }
342
385
  return null
343
386
  }
@@ -542,6 +585,7 @@ export function apply(ctx) {
542
585
  ctx.on('llm/adapters-updated', () => {
543
586
  for (const state of allStates) {
544
587
  state.directoryPromise = null
588
+ state.routeCooldowns.clear()
545
589
  }
546
590
  void scheduleOpenCodeRepair()
547
591
  })
@@ -593,9 +637,11 @@ export function apply(ctx) {
593
637
  await routerSettingsPromise
594
638
  state.taskText = inputText(messages)
595
639
  const liveBench = await liveBenchFor(ctx, state)
640
+ const readyRoutes = available.filter(route => !routeTemporarilyUnavailable(state, routeKey(route.provider, route.model)))
641
+ const routable = readyRoutes.length > 0 ? readyRoutes : available
596
642
  const plan = buildPlan({
597
643
  text: state.taskText,
598
- available,
644
+ available: routable,
599
645
  mode: state.mode,
600
646
  pricing: routerSettings.pricing,
601
647
  liveBench,
@@ -614,7 +660,7 @@ export function apply(ctx) {
614
660
  failSafe: true,
615
661
  },
616
662
  }
617
- state.collaboration = shouldCollaborate(state.plan, available)
663
+ state.collaboration = shouldCollaborate(state.plan, routable)
618
664
  ? { lastStep: 0, queuedStep: null }
619
665
  : null
620
666
  }
@@ -656,7 +702,7 @@ export function apply(ctx) {
656
702
  const assignedRoute = state.available.find(route => route.provider === assignment?.recommendedProvider && route.model === assignment?.recommended)
657
703
  ?? state.available.find(route => route.model === assignment?.recommended)
658
704
  if (assignedRoute !== undefined) {
659
- target = { provider: assignedRoute.provider, model: assignedRoute.model, estimatedCost: target.estimatedCost }
705
+ target = { provider: assignedRoute.provider, model: assignedRoute.model, reasoningEffort: assignment?.recommendedReasoningEffort, estimatedCost: target.estimatedCost }
660
706
  }
661
707
  // The final subtask is the public answer synthesis. Prefer the user's
662
708
  // requested DeepSeek V4 Pro when it is actually available; otherwise the
@@ -665,14 +711,14 @@ export function apply(ctx) {
665
711
  if (assignment?.purpose === 'synthesis' && synthesis?.provider && synthesis?.model) {
666
712
  const synthesisRoute = state.available.find(route => route.provider === synthesis.provider && route.model === synthesis.model)
667
713
  if (synthesisRoute !== undefined) {
668
- target = { provider: synthesisRoute.provider, model: synthesisRoute.model, estimatedCost: target.estimatedCost }
714
+ target = { provider: synthesisRoute.provider, model: synthesisRoute.model, reasoningEffort: synthesis.reasoningEffort, estimatedCost: target.estimatedCost }
669
715
  }
670
716
  }
671
717
  }
672
- if (state.failedModels.has(routeKey(target.provider, target.model))) {
673
- const fallback = nextAvailableTarget(state)
718
+ if (routeTemporarilyUnavailable(state, routeKey(target.provider, target.model))) {
719
+ const fallback = nextAvailableTarget(state, step)
674
720
  if (fallback !== null) {
675
- target = { provider: fallback.provider, model: fallback.model, estimatedCost: target.estimatedCost }
721
+ target = { provider: fallback.provider, model: fallback.model, reasoningEffort: fallback.reasoningEffort, estimatedCost: target.estimatedCost }
676
722
  }
677
723
  }
678
724
  const plannedProvider = target.provider
@@ -682,12 +728,15 @@ export function apply(ctx) {
682
728
  visionBridges: state.visionBridges,
683
729
  hasImageBlocks: state.hasImageBlocks,
684
730
  })
685
- state.lastTarget = { provider: target.provider, model: target.model, plannedProvider }
731
+ state.lastTarget = { provider: target.provider, model: target.model, reasoningEffort: target.reasoningEffort, plannedProvider }
686
732
  state.lastStep = step
687
733
  // Keep all non-routing request fields intact. If a route disappeared after
688
734
  // discovery, the LLM runtime will validate the proposal and the original
689
735
  // model remains available on the next step.
690
- return { ...proposed, provider: target.provider, model: target.model }
736
+ const routed = { ...proposed, provider: target.provider, model: target.model }
737
+ if (target.reasoningEffort === undefined) delete routed.reasoningEffort
738
+ else routed.reasoningEffort = target.reasoningEffort
739
+ return routed
691
740
  })
692
741
 
693
742
  // A completed work step would normally close the turn immediately. Queue the
@@ -710,23 +759,29 @@ export function apply(ctx) {
710
759
  ctx.on('agent/request-error', async ({ agent, provider, failure, signal }, next) => {
711
760
  const state = stateFor(agent)
712
761
  if (signal?.aborted || state.mode !== 'collective' || !modelFallbackError(failure)) return next()
713
- const failed = state.lastTarget?.provider === provider ? state.lastTarget : null
762
+ const failed = state.lastTarget !== null && (provider === undefined || provider === null || state.lastTarget.provider === provider) ? state.lastTarget : null
714
763
  if (failed === null) return next()
715
- state.failedModels.add(routeKey(failed.plannedProvider ?? failed.provider, failed.model))
716
- const fallback = nextAvailableTarget(state)
764
+ const failedKey = routeKey(failed.plannedProvider ?? failed.provider, failed.model)
765
+ state.failedModels.add(failedKey)
766
+ state.routeCooldowns.set(failedKey, Date.now() + failureCooldownMs(failure))
767
+ const fallback = nextAvailableTarget(state, state.lastStep)
717
768
  if (fallback === null) return next()
718
769
  if (state.plan !== null) {
719
770
  const plan = state.plan
720
771
  const failedStage = plan.subtasks?.[Math.max(0, state.lastStep - 1)]
772
+ const failedStageIndex = Math.max(0, state.lastStep - 1)
721
773
  const isSynthesisFailure = failedStage?.purpose === 'synthesis'
774
+ const subtasks = Array.isArray(plan.subtasks)
775
+ ? plan.subtasks.map((task, index) => index === failedStageIndex
776
+ ? { ...task, recommendedProvider: fallback.provider, recommended: fallback.model, recommendedReasoningEffort: fallback.reasoningEffort }
777
+ : task)
778
+ : plan.subtasks
722
779
  state.plan = {
723
780
  ...plan,
724
- ...(plan.selected === null ? {} : { selected: { ...plan.selected, provider: fallback.provider, model: fallback.model } }),
781
+ ...(plan.selected === null ? {} : { selected: { ...plan.selected, provider: fallback.provider, model: fallback.model, reasoningEffort: fallback.reasoningEffort } }),
782
+ subtasks,
725
783
  ...(isSynthesisFailure ? {
726
- synthesizer: { provider: fallback.provider, model: fallback.model },
727
- subtasks: plan.subtasks.map((task, index) => index === plan.subtasks.length - 1
728
- ? { ...task, recommendedProvider: fallback.provider, recommended: fallback.model }
729
- : task),
784
+ synthesizer: { provider: fallback.provider, model: fallback.model, reasoningEffort: fallback.reasoningEffort },
730
785
  } : {}),
731
786
  }
732
787
  }
@@ -293,10 +293,10 @@ const CHAPTER_MUSIC = Object.freeze({
293
293
  prologue: 'title-city',
294
294
  'open-day': 'commons-atelier',
295
295
  laurel: 'claude-poem',
296
- 'bridges-night': 'bridge-anomaly',
296
+ 'bridges-night': 'kimi-flute',
297
297
  'three-harbors': 'harbor-shift',
298
298
  'open-dome-hearing': 'glass-dome',
299
- 'protocol-composition': 'bridge-anomaly',
299
+ 'protocol-composition': 'glass-dome',
300
300
  'six-endings': 'title-city',
301
301
  })
302
302
 
@@ -317,16 +317,20 @@ function routeForNode(node) {
317
317
  export function storyPresentationFor(node) {
318
318
  const route = routeForNode(node)
319
319
  const backgroundId = route?.backgroundId || node.chapterId
320
- const musicTheme = route?.musicTheme || CHAPTER_MUSIC[node.chapterId] || 'title-city'
320
+ const bridgeCrisis = node.chapterId === 'bridges-night' && /^(incident|cut|isolate|evidence|trace|report)-/.test(node.id)
321
+ const bridgeAftercare = node.chapterId === 'bridges-night' && /^after-/.test(node.id)
322
+ const compositionAudit = node.chapterId === 'protocol-composition' && /^composition-audit-/.test(node.id)
323
+ const musicTheme = route?.musicTheme
324
+ || (bridgeCrisis || compositionAudit ? 'bridge-anomaly' : bridgeAftercare ? 'claude-poem' : null)
325
+ || CHAPTER_MUSIC[node.chapterId]
326
+ || 'title-city'
321
327
  let soundCue = null
322
328
  const text = typeof node.text === 'string' ? node.text : ''
323
- if (node.chapterId === 'bridges-night') {
324
- soundCue = Object.freeze({ id: 'bridge-anomaly', label: '节奏异常', description: '节拍与回执时间没有对齐;这处不协和可能属于千桥事故证据链。' })
325
- } else if (node.speaker === 'kimi' || /长笛|完整录音|缺拍/.test(text)) {
329
+ if (node.speaker === 'kimi' || /长笛|完整录音|缺拍/.test(text)) {
326
330
  soundCue = Object.freeze({ id: 'kimi-flute', label: '长笛线索', description: '同一段长笛再次出现;留意旋律中的停顿和它前后的语境。' })
327
331
  } else if (node.speaker === 'claude' && /桥|诗|下一行|署名|空白/.test(text)) {
328
332
  soundCue = Object.freeze({ id: 'claude-verse', label: '诗句线索', description: 'Claude 的未完成诗句再次出现;这次由她自己决定下一行是否存在。' })
329
- } else if (/事故录音|第九秒|节拍|拍号冲突|时钟/.test(text)) {
333
+ } else if (bridgeCrisis || /事故录音|第九秒|节拍|拍号冲突|时钟/.test(text)) {
330
334
  soundCue = Object.freeze({ id: 'bridge-anomaly', label: '节奏异常', description: '节拍与回执时间没有对齐;这处不协和可能属于千桥事故证据链。' })
331
335
  }
332
336
  return Object.freeze({ backgroundId, musicTheme, soundCue })
@@ -9,9 +9,22 @@
9
9
  import { liveBenchRow } from './livebench.mjs'
10
10
 
11
11
  export const OBJECTIVE_WEIGHTS = Object.freeze({
12
- simple: Object.freeze({ quality: 0.30, cost: 0.50, latency: 0.14, specialty: 0.04, risk: 0.02 }),
13
- balanced: Object.freeze({ quality: 0.45, cost: 0.30, latency: 0.10, specialty: 0.10, risk: 0.05 }),
14
- complex: Object.freeze({ quality: 0.55, cost: 0.16, latency: 0.06, specialty: 0.16, risk: 0.07 }),
12
+ simple: Object.freeze({ quality: 0.28, cost: 0.45, latency: 0.14, specialty: 0.04, reasoning: 0.07, risk: 0.02 }),
13
+ balanced: Object.freeze({ quality: 0.40, cost: 0.26, latency: 0.10, specialty: 0.09, reasoning: 0.10, risk: 0.05 }),
14
+ complex: Object.freeze({ quality: 0.48, cost: 0.14, latency: 0.06, specialty: 0.14, reasoning: 0.11, risk: 0.07 }),
15
+ })
16
+
17
+ /** Canonical order used only to compare adapter-owned opaque effort ids. */
18
+ export const REASONING_EFFORT_ORDER = Object.freeze(['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max'])
19
+
20
+ const REASONING_EFFORT_MULTIPLIERS = Object.freeze({
21
+ off: Object.freeze({ output: 0.78, latency: 0.75 }),
22
+ minimal: Object.freeze({ output: 0.86, latency: 0.82 }),
23
+ low: Object.freeze({ output: 0.93, latency: 0.90 }),
24
+ medium: Object.freeze({ output: 1, latency: 1 }),
25
+ high: Object.freeze({ output: 1.16, latency: 1.15 }),
26
+ xhigh: Object.freeze({ output: 1.34, latency: 1.30 }),
27
+ max: Object.freeze({ output: 1.58, latency: 1.50 }),
15
28
  })
16
29
 
17
30
  /** Quality floor for a task node before a cost-saving substitution is allowed. */
@@ -328,6 +341,7 @@ function taskTokenBudget(text, task, complexity, cacheReadRatio = 0, cacheWriteR
328
341
  synthesis: { input: 1.65, output: 1.30 },
329
342
  }
330
343
  const multiplier = multipliers[task.purpose] ?? { input: 1, output: 1 }
344
+ const effortMultiplier = reasoningEffortMultiplier(task.reasoningEffort)
331
345
  const totalInputTokens = Math.max(80, Math.round(inputTokens * multiplier.input))
332
346
  const ratios = normalizedCacheRatios(cacheReadRatio, cacheWriteRatio)
333
347
  const cacheReadTokens = Math.min(totalInputTokens, Math.max(0, Math.round(totalInputTokens * ratios.read)))
@@ -336,7 +350,7 @@ function taskTokenBudget(text, task, complexity, cacheReadRatio = 0, cacheWriteR
336
350
  inputTokens: totalInputTokens,
337
351
  cacheReadTokens,
338
352
  cacheWriteTokens,
339
- outputTokens: Math.max(220, Math.round(900 * multiplier.output)),
353
+ outputTokens: Math.max(220, Math.round(900 * multiplier.output * effortMultiplier.output)),
340
354
  }
341
355
  }
342
356
 
@@ -349,12 +363,12 @@ function taskQualityFloor(band, task) {
349
363
 
350
364
  function taskPackages(taskType, text, band) {
351
365
  if (band !== 'complex') {
352
- const task = { id: 'execution', name: '直接回答与必要校验', type: taskType, purpose: 'execution', criticality: 0.65, dependsOn: [] }
366
+ const task = { id: 'execution', name: '直接回答与必要校验', type: taskType, purpose: 'execution', criticality: 0.65, dependsOn: [], preferredReasoningEffort: band === 'simple' ? 'low' : 'medium' }
353
367
  return [{ ...task, qualityFloor: taskQualityFloor(band, task) }]
354
368
  }
355
369
  const value = String(text ?? '')
356
370
  const packages = [
357
- { id: 'analysis', name: '问题建模与约束提取', type: 'reasoning', purpose: 'analysis', criticality: 0.92, dependsOn: [] },
371
+ { id: 'analysis', name: '问题建模与约束提取', type: 'reasoning', purpose: 'analysis', criticality: 0.92, dependsOn: [], preferredReasoningEffort: 'high' },
358
372
  ]
359
373
  const domains = [...new Set([...(detectTaskTypes(text)), taskType].filter(type => type !== 'general'))]
360
374
  for (const type of domains.length > 0 ? domains : [taskType]) {
@@ -365,6 +379,7 @@ function taskPackages(taskType, text, band) {
365
379
  purpose: 'execution',
366
380
  criticality: domains.length > 1 ? 0.80 : 0.78,
367
381
  dependsOn: ['analysis'],
382
+ preferredReasoningEffort: 'high',
368
383
  })
369
384
  }
370
385
  if (/(测试|验证|评估|对比|benchmark|test|verify|audit)/i.test(value)) {
@@ -375,6 +390,7 @@ function taskPackages(taskType, text, band) {
375
390
  purpose: 'verification',
376
391
  criticality: 0.88,
377
392
  dependsOn: packages.filter(task => task.purpose === 'execution').map(task => task.id),
393
+ preferredReasoningEffort: 'high',
378
394
  })
379
395
  }
380
396
  packages.push({
@@ -384,11 +400,12 @@ function taskPackages(taskType, text, band) {
384
400
  purpose: 'synthesis',
385
401
  criticality: 1,
386
402
  dependsOn: packages.filter(task => task.purpose !== 'analysis').map(task => task.id),
403
+ preferredReasoningEffort: 'xhigh',
387
404
  })
388
405
  return packages.map(task => ({ ...task, qualityFloor: taskQualityFloor(band, task) }))
389
406
  }
390
407
 
391
- const SYNTHESIS_WEIGHTS = Object.freeze({ quality: 0.70, cost: 0.10, latency: 0.04, specialty: 0.10, risk: 0.06 })
408
+ const SYNTHESIS_WEIGHTS = Object.freeze({ quality: 0.58, cost: 0.08, latency: 0.04, specialty: 0.08, reasoning: 0.16, risk: 0.06 })
392
409
  const ROUTING_BEAM_WIDTH = 256
393
410
  const ROUTING_CANDIDATE_LIMIT = 12
394
411
 
@@ -406,22 +423,78 @@ function compareRowsStable(left, right) {
406
423
  return compareText(routeKey(left.provider, left.model), routeKey(right.provider, right.model))
407
424
  }
408
425
 
426
+ function reasoningEffortRank(effort) {
427
+ const normalized = String(effort ?? '').trim().toLowerCase().replace(/[\s_-]+/g, '')
428
+ const aliases = { none: 'off', disabled: 'off', extra: 'xhigh', extrahigh: 'xhigh', maximum: 'max' }
429
+ const canonical = aliases[normalized] ?? normalized
430
+ const index = REASONING_EFFORT_ORDER.indexOf(canonical)
431
+ return index < 0 ? REASONING_EFFORT_ORDER.indexOf('medium') : index
432
+ }
433
+
434
+ function reasoningEffortMultiplier(effort) {
435
+ const normalized = String(effort ?? '').trim().toLowerCase().replace(/[\s_-]+/g, '')
436
+ const aliases = { none: 'off', disabled: 'off', extra: 'xhigh', extrahigh: 'xhigh', maximum: 'max' }
437
+ return REASONING_EFFORT_MULTIPLIERS[aliases[normalized] ?? normalized] ?? REASONING_EFFORT_MULTIPLIERS.medium
438
+ }
439
+
440
+ /** Pick the closest exact effort id exposed by an adapter. */
441
+ export function selectReasoningEffort(efforts, preferred = 'medium') {
442
+ const exact = Array.isArray(efforts)
443
+ ? [...new Set(efforts.map(effort => String(effort?.id ?? effort ?? '')).filter(Boolean))]
444
+ : []
445
+ if (exact.length === 0) return undefined
446
+ const preferredRank = reasoningEffortRank(preferred)
447
+ return exact.slice().sort((left, right) => {
448
+ const distance = Math.abs(reasoningEffortRank(left) - preferredRank) - Math.abs(reasoningEffortRank(right) - preferredRank)
449
+ if (distance !== 0) return distance
450
+ return reasoningEffortRank(left) - reasoningEffortRank(right) || compareText(left, right)
451
+ })[0]
452
+ }
453
+
454
+ function reasoningDecision(row, task) {
455
+ const preferred = String(task.preferredReasoningEffort ?? 'medium')
456
+ const efforts = Array.isArray(row.reasoningEfforts) ? row.reasoningEfforts : []
457
+ if (efforts.length === 0) {
458
+ const knownUnsupported = row.reasoningKnown === true
459
+ const preferredRank = reasoningEffortRank(preferred)
460
+ return {
461
+ reasoningEffort: undefined,
462
+ reasoningFit: knownUnsupported ? clamp(0.78 - preferredRank * 0.08) : 0.55,
463
+ preferredReasoningEffort: preferred,
464
+ multiplier: REASONING_EFFORT_MULTIPLIERS.medium,
465
+ }
466
+ }
467
+ const preferredRank = reasoningEffortRank(preferred)
468
+ // On an equal distance the shared selector prefers the lower effort to
469
+ // avoid unnecessary latency and output-token cost.
470
+ const chosen = selectReasoningEffort(efforts, preferred)
471
+ const distance = Math.abs(reasoningEffortRank(chosen) - preferredRank)
472
+ return {
473
+ reasoningEffort: chosen,
474
+ reasoningFit: clamp(1 - distance / (REASONING_EFFORT_ORDER.length - 1)),
475
+ preferredReasoningEffort: preferred,
476
+ multiplier: reasoningEffortMultiplier(chosen),
477
+ }
478
+ }
479
+
409
480
  function candidateUtility(row, task, weights, maxCost, usedRoutes, cacheReadRatio = 0, cacheWriteRatio = 0) {
410
481
  const quality = qualityForTask(row, task.type)
411
482
  const floor = Number(task.qualityFloor ?? taskQualityFloor('complex', task))
412
483
  const qualityGap = Math.max(0, floor - quality)
413
484
  const duplicatePenalty = usedRoutes.has(routeKey(row.provider, row.model)) ? 0.08 : 0
414
485
  const synthesisPreference = task.purpose === 'synthesis' && /deepseek[- ]?v4[- ]?pro/i.test(row.model) ? 0.025 : 0
415
- const cost = costScore(row.pricing, maxCost, cacheReadRatio, cacheWriteRatio)
486
+ const reasoning = reasoningDecision(row, task)
487
+ const cost = clamp(costScore(row.pricing, maxCost, cacheReadRatio, cacheWriteRatio) / Math.sqrt(reasoning.multiplier.output))
416
488
  const score = weights.quality * quality
417
489
  + weights.cost * cost
418
- + weights.latency * (1 - clamp(row.latency))
490
+ + weights.latency * (1 - clamp(row.latency * reasoning.multiplier.latency))
419
491
  + weights.specialty * specialtyForTask(row, task.type)
492
+ + (weights.reasoning ?? 0) * reasoning.reasoningFit
420
493
  - weights.risk * row.risk
421
494
  - duplicatePenalty
422
495
  - qualityGap * (task.criticality ?? 0.75)
423
496
  + synthesisPreference
424
- return { score, floor, qualityGap }
497
+ return { score, floor, qualityGap, ...reasoning }
425
498
  }
426
499
 
427
500
  function chooseAssignment(rows, task, weights, maxCost, usedRoutes, preferred, cacheReadRatio = 0, cacheWriteRatio = 0) {
@@ -437,7 +510,8 @@ function chooseAssignment(rows, task, weights, maxCost, usedRoutes, preferred, c
437
510
  }
438
511
 
439
512
  function taskCost(row, task, text, complexity, cacheReadRatio = 0, cacheWriteRatio = 0) {
440
- const tokens = taskTokenBudget(text, task, complexity, cacheReadRatio, cacheWriteRatio)
513
+ const decision = row === null ? null : reasoningDecision(row, task)
514
+ const tokens = taskTokenBudget(text, { ...task, reasoningEffort: decision?.reasoningEffort }, complexity, cacheReadRatio, cacheWriteRatio)
441
515
  return row === null
442
516
  ? 0
443
517
  : (((tokens.inputTokens - tokens.cacheReadTokens - tokens.cacheWriteTokens) * row.pricing.input)
@@ -447,29 +521,35 @@ function taskCost(row, task, text, complexity, cacheReadRatio = 0, cacheWriteRat
447
521
  }
448
522
 
449
523
  function dominates(left, right, task, text, complexity, cacheReadRatio, cacheWriteRatio) {
524
+ const leftReasoning = reasoningDecision(left, task)
525
+ const rightReasoning = reasoningDecision(right, task)
450
526
  const leftValues = {
451
527
  quality: qualityForTask(left, task.type),
452
528
  cost: taskCost(left, task, text, complexity, cacheReadRatio, cacheWriteRatio),
453
- latency: clamp(left.latency),
529
+ latency: clamp(left.latency * leftReasoning.multiplier.latency),
454
530
  specialty: specialtyForTask(left, task.type),
531
+ reasoning: leftReasoning.reasoningFit,
455
532
  risk: clamp(left.risk),
456
533
  }
457
534
  const rightValues = {
458
535
  quality: qualityForTask(right, task.type),
459
536
  cost: taskCost(right, task, text, complexity, cacheReadRatio, cacheWriteRatio),
460
- latency: clamp(right.latency),
537
+ latency: clamp(right.latency * rightReasoning.multiplier.latency),
461
538
  specialty: specialtyForTask(right, task.type),
539
+ reasoning: rightReasoning.reasoningFit,
462
540
  risk: clamp(right.risk),
463
541
  }
464
542
  const noWorse = leftValues.quality >= rightValues.quality
465
543
  && leftValues.cost <= rightValues.cost
466
544
  && leftValues.latency <= rightValues.latency
467
545
  && leftValues.specialty >= rightValues.specialty
546
+ && leftValues.reasoning >= rightValues.reasoning
468
547
  && leftValues.risk <= rightValues.risk
469
548
  const strictlyBetter = leftValues.quality > rightValues.quality
470
549
  || leftValues.cost < rightValues.cost
471
550
  || leftValues.latency < rightValues.latency
472
551
  || leftValues.specialty > rightValues.specialty
552
+ || leftValues.reasoning > rightValues.reasoning
473
553
  || leftValues.risk < rightValues.risk
474
554
  return noWorse && strictlyBetter
475
555
  }
@@ -504,7 +584,7 @@ function candidatePool(rows, task, weights, maxCost, text, complexity, cacheRead
504
584
  }
505
585
 
506
586
  function stateSignature(state) {
507
- return state.assignments.map(assignment => routeKey(assignment.row?.provider, assignment.row?.model)).join('|')
587
+ return state.assignments.map(assignment => `${routeKey(assignment.row?.provider, assignment.row?.model)}@${assignment.decision?.reasoningEffort ?? 'provider-default'}`).join('|')
508
588
  }
509
589
 
510
590
  function compareUtilityStates(left, right) {
@@ -585,7 +665,17 @@ export function buildPlan({ text = '', available = [], mode = 'collective', pric
585
665
  const taskType = classifyTask(text)
586
666
  const weights = OBJECTIVE_WEIGHTS[complexity.band]
587
667
  const discovered = Array.isArray(available)
588
- ? available.map(entry => ({ provider: String(entry.provider ?? ''), model: String(entry.model ?? '') }))
668
+ ? available.map(entry => {
669
+ const rawEfforts = Array.isArray(entry.reasoningEfforts) ? entry.reasoningEfforts : []
670
+ const reasoningEfforts = rawEfforts.map(effort => String(effort?.id ?? effort ?? '')).filter(Boolean)
671
+ return {
672
+ provider: String(entry.provider ?? ''),
673
+ model: String(entry.model ?? ''),
674
+ reasoningEfforts: [...new Set(reasoningEfforts)],
675
+ defaultReasoningEffort: entry.defaultReasoningEffort === undefined ? undefined : String(entry.defaultReasoningEffort),
676
+ reasoningKnown: entry.reasoningKnown === true || Array.isArray(entry.reasoningEfforts),
677
+ }
678
+ })
589
679
  : []
590
680
  const rows = []
591
681
  const normalizedPrices = normalizePricing(pricing)
@@ -618,6 +708,9 @@ export function buildPlan({ text = '', available = [], mode = 'collective', pric
618
708
  latency: metadata.latency,
619
709
  risk: metadata.risk,
620
710
  specialty,
711
+ reasoningEfforts: route.reasoningEfforts,
712
+ defaultReasoningEffort: route.defaultReasoningEffort,
713
+ reasoningKnown: route.reasoningKnown,
621
714
  pricing: pricingRow,
622
715
  score: 0,
623
716
  estimatedCost: estimateCost(metadata, text, 900, normalizedPrices, cacheReadRatio, cacheWriteRatio),
@@ -671,27 +764,36 @@ export function buildPlan({ text = '', available = [], mode = 'collective', pric
671
764
  row.score = candidateUtility(row, taskNodes[0] ?? { type: taskType, qualityFloor: QUALITY_FLOORS[complexity.band] }, weights, maxCost, new Set(), cacheReadRatio, cacheWriteRatio).score
672
765
  }
673
766
  rows.sort((left, right) => right.score - left.score || compareRowsStable(left, right))
674
- const selected = assignments[0]?.row ?? rows[0] ?? null
675
- const synthesizer = assignments.at(-1)?.row ?? rows.find(row => /deepseek/i.test(row.model)) ?? rows[0]
676
- const subtasks = assignments.map(({ task, row }) => ({
767
+ const selectedAssignment = assignments[0]
768
+ const selected = selectedAssignment?.row ?? rows[0] ?? null
769
+ const synthesizerAssignment = assignments.at(-1)
770
+ const synthesizer = synthesizerAssignment?.row ?? rows.find(row => /deepseek/i.test(row.model)) ?? rows[0]
771
+ const subtasks = assignments.map(({ task, row, decision }) => ({
677
772
  id: task.id,
678
773
  name: task.name,
679
774
  type: task.type,
680
775
  recommended: row?.model ?? '待发现模型',
681
776
  recommendedProvider: row?.provider ?? '',
777
+ recommendedReasoningEffort: decision?.reasoningEffort,
778
+ preferredReasoningEffort: decision?.preferredReasoningEffort ?? task.preferredReasoningEffort,
779
+ reasoningFit: Number(Number(decision?.reasoningFit ?? 0).toFixed(3)),
682
780
  purpose: task.purpose,
683
781
  criticality: task.criticality,
684
782
  qualityFloor: Number(task.qualityFloor.toFixed(3)),
685
783
  dependsOn: [...(task.dependsOn ?? [])],
686
784
  }))
687
- const costBreakdown = assignments.map(({ task, row, estimatedCost, handoffPenalty }, index) => {
688
- const tokens = taskTokenBudget(text, task, complexity.band, cacheReadRatio, cacheWriteRatio)
785
+ const costBreakdown = assignments.map(({ task, row, decision, estimatedCost, handoffPenalty }, index) => {
786
+ const tokens = taskTokenBudget(text, { ...task, reasoningEffort: decision?.reasoningEffort }, complexity.band, cacheReadRatio, cacheWriteRatio)
689
787
  const taskEstimate = estimatedCost ?? taskCost(row, task, text, complexity.band, cacheReadRatio, cacheWriteRatio)
690
788
  return {
691
789
  stage: index + 1,
692
790
  purpose: task.purpose,
693
791
  model: row?.model ?? '待发现模型',
694
792
  provider: row?.provider ?? '',
793
+ reasoningEffort: decision?.reasoningEffort,
794
+ preferredReasoningEffort: decision?.preferredReasoningEffort,
795
+ reasoningFit: Number(Number(decision?.reasoningFit ?? 0).toFixed(3)),
796
+ reasoningOutputMultiplier: Number(Number(decision?.multiplier?.output ?? 1).toFixed(2)),
695
797
  inputTokens: tokens.inputTokens,
696
798
  cacheReadTokens: tokens.cacheReadTokens,
697
799
  cacheWriteTokens: tokens.cacheWriteTokens,
@@ -704,11 +806,7 @@ export function buildPlan({ text = '', available = [], mode = 'collective', pric
704
806
  const totalEstimate = costBreakdown.reduce((sum, row) => sum + row.estimatedCost, 0)
705
807
  const baselineCost = assignments.reduce((sum, { task }) => {
706
808
  const strongest = rows.reduce((best, row) => qualityForTask(row, task.type) > (best === null ? -1 : qualityForTask(best, task.type)) ? row : best, null)
707
- const tokens = taskTokenBudget(text, task, complexity.band, cacheReadRatio, cacheWriteRatio)
708
- return sum + (strongest === null ? 0 : (((tokens.inputTokens - tokens.cacheReadTokens - tokens.cacheWriteTokens) * strongest.pricing.input)
709
- + (tokens.cacheReadTokens * strongest.pricing.cacheRead)
710
- + (tokens.cacheWriteTokens * strongest.pricing.cacheWrite)
711
- + (tokens.outputTokens * strongest.pricing.output)) / 1_000_000)
809
+ return sum + taskCost(strongest, task, text, complexity.band, cacheReadRatio, cacheWriteRatio)
712
810
  }, 0)
713
811
  const budgetExceeded = Number(budgetUsd) > 0 && totalEstimate > Number(budgetUsd)
714
812
  const savings = baselineCost <= 0 ? 0 : clamp((baselineCost - totalEstimate) / baselineCost)
@@ -716,17 +814,20 @@ export function buildPlan({ text = '', available = [], mode = 'collective', pric
716
814
  const minimumFeasibleCost = optimized?.minimumFeasibleCost ?? minimumCostPlan?.cost ?? 0
717
815
  const reason = selected === null
718
816
  ? '尚未发现可用模型,保留 Harness 原始模型选择。'
719
- : `${complexity.band === 'simple' ? '低复杂度优先成本与响应速度' : complexity.band === 'balanced' ? '在质量、成本、延迟与风险之间平衡' : '高复杂度执行依赖感知的全局约束分配'};任务类型为 ${taskType},已对 ${String(subtasks.length)} 个工作包进行 Pareto 剪枝和有界组合搜索。`
817
+ : `${complexity.band === 'simple' ? '低复杂度优先成本、响应速度与较低推理开销' : complexity.band === 'balanced' ? '在质量、成本、推理等级、延迟与风险之间平衡' : '高复杂度执行包含推理等级的依赖感知全局约束分配'};任务类型为 ${taskType},已对 ${String(subtasks.length)} 个工作包进行 Pareto 剪枝和有界组合搜索。`
720
818
  return {
721
819
  mode,
722
820
  complexity: { value: Number(complexity.value.toFixed(3)), band: complexity.band },
723
821
  taskType,
724
822
  taskTypes: [...new Set(taskNodes.map(task => task.type).filter(type => type !== 'reasoning'))],
725
823
  objectiveWeights: weights,
726
- candidates: rows.slice(0, 8).map(row => ({ provider: row.provider, model: row.model, score: Number(row.score.toFixed(3)), quality: Number(row.quality.toFixed(3)), specialty: Number(row.specialty.toFixed(3)), estimatedCost: Number(row.estimatedCost.toFixed(6)), inputPrice: row.pricing.input, outputPrice: row.pricing.output })),
727
- selected: selected === null ? null : { provider: selected.provider, model: selected.model, estimatedCost: Number(selected.estimatedCost.toFixed(6)) },
824
+ candidates: rows.slice(0, 8).map(row => {
825
+ const decision = candidateUtility(row, taskNodes[0] ?? { type: taskType, qualityFloor: QUALITY_FLOORS[complexity.band], preferredReasoningEffort: complexity.band === 'simple' ? 'low' : 'medium' }, weights, maxCost, new Set(), cacheReadRatio, cacheWriteRatio)
826
+ return { provider: row.provider, model: row.model, score: Number(row.score.toFixed(3)), quality: Number(row.quality.toFixed(3)), specialty: Number(row.specialty.toFixed(3)), reasoningEffort: decision.reasoningEffort, preferredReasoningEffort: decision.preferredReasoningEffort, reasoningFit: Number(decision.reasoningFit.toFixed(3)), reasoningKnown: row.reasoningKnown, reasoningEfforts: row.reasoningEfforts, estimatedCost: Number(row.estimatedCost.toFixed(6)), inputPrice: row.pricing.input, outputPrice: row.pricing.output }
827
+ }),
828
+ selected: selected === null ? null : { provider: selected.provider, model: selected.model, reasoningEffort: selectedAssignment?.decision?.reasoningEffort, estimatedCost: Number(selected.estimatedCost.toFixed(6)) },
728
829
  subtasks,
729
- synthesizer: synthesizer === undefined ? null : { provider: synthesizer.provider, model: synthesizer.model },
830
+ synthesizer: synthesizer === undefined ? null : { provider: synthesizer.provider, model: synthesizer.model, reasoningEffort: synthesizerAssignment?.decision?.reasoningEffort },
730
831
  estimatedCost: Number(totalEstimate.toFixed(6)),
731
832
  costBreakdown,
732
833
  optimization: {
@@ -14,9 +14,9 @@ GAL 游戏作为原有「GAL视窗」内的一个模块,提供「剧情模式
14
14
 
15
15
  ## 剧情模式
16
16
 
17
- 0.4.30 默认先显示 Galgame 首页,可继续当前存档、从序章开始、打开八章主线、进入 22 条角色支线或调整无障碍与声音。《千桥协议》现有 666 个节点、44 处两难选择、六类制度结局和 44 种支线收束;每条支线都会留下对应角色承诺与证据。主线关系尾声、制度结局和支线结局分别计算。
17
+ 0.4.32 默认先显示 Galgame 首页,可继续当前存档、从序章开始、打开八章主线、进入 22 条角色支线或调整无障碍与声音。《千桥协议》现有 666 个节点、44 处两难选择、六类制度结局和 44 种支线收束;每条支线都会留下对应角色承诺与证据。主线关系尾声、制度结局和支线结局分别计算。
18
18
 
19
- 七套主题乐谱由 MiniMax M3 生成音高、节奏、音色和力度数据,运行时由 Web Audio 在本地合成,不再请求 MiniMax。Kimi 长笛、Claude 诗句和千桥事故异常节拍均有可关闭的文字字幕。设置还提供音乐音量、减少动态、高对比度、大号文字、易读字体和完整键盘操作。
19
+ 七套主题乐谱由 MiniMax M3 生成音高、节奏、音色和力度数据,运行时由 Web Audio 在本地合成,不再请求 MiniMax。0.4.32 按叙事段落安排主题:千桥签署先以 Kimi 长笛铺陈,事故显形才进入异常节拍,善后转入 Claude 诗句;协议组合在冲突检验前保持穹顶复核主题。主题切换采用交叉淡化,对话线索出现时自动压低背景音量。Kimi 长笛、Claude 诗句和千桥事故异常节拍均有可关闭的文字字幕。
20
20
 
21
21
  《未写完的约定》已扩展为 1,111 个台词与事件节点、28 处分支选择、5 种关系结局及各自后日谈。故事从旧城最后一周延续到迁移后的生活,包含 46 个地点与时段组合。默认选择的一条完整路线有 853 次推进、约 2.27 万字对话和旁白;其他路线长度略有不同,这不是全部分支的总字数。
22
22
 
@@ -36,7 +36,7 @@ GAL 游戏作为原有「GAL视窗」内的一个模块,提供「剧情模式
36
36
 
37
37
  ## 画面与调试
38
38
 
39
- 两种模式的场景图片都按要求保留空白和文字描述,使用已有角色立绘及 DeepSeek 表情差分。对话框改用精细位图:DeepSeek 保留用户原框的艺术主体,另 13 位人物通过图片 API 生成各自的材质、徽章和边饰。图片由原 GAL 渲染器共用的 `DialogueBox` 呈现,设置内的「对话框图鉴」可以查看全部样式。原始素材继续保留,游戏运行时直接读取本地 WebP,不请求图片 API。详见 [对话框图片说明](output/imagegen/dialogue-frames/README.md) 和 [14 套对照图](output/imagegen/dialogue-frames/runtime-contact-sheet.png)。
39
+ 两种模式的场景图片都按要求保留空白和文字描述,使用已有角色立绘及 DeepSeek 表情差分。DeepSeek 五张差分均保留透明通道;Claude 日常使用 `Claude1.png`,只在坚定、生气和惊讶场景切换 `Claude.png`。对话框改用 22 套精细位图,每位主线与组织角色都有独立材质、徽章和边饰。图片由原 GAL 渲染器共用的 `DialogueBox` 呈现,设置内的「对话框图鉴」可以查看全部样式。原始素材继续保留,游戏运行时直接读取本地 WebP,不请求图片 API。详见 [对话框图片说明](output/imagegen/dialogue-frames/README.md) 和 [22 套对照图](output/imagegen/dialogue-frames/runtime-contact-sheet.png)。
40
40
 
41
41
  普通游玩界面只呈现场景、角色、对话、剧情选项或自由输入与轻量菜单。好感、信任、隔阂、关系阶段、情绪标签、场景完成条件、进度和判定证据都在后台运行,不以数字、标签或进度条向玩家显示;角色的台词、态度与表情承载这些变化。回忆、对话历史、存档列表和结局呈现也不显示内部判定。
42
42