dsh-vision-router 2.2.1 → 2.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/client.js CHANGED
@@ -71,9 +71,14 @@ window.__ModuleLoader__.load({
71
71
  localOllamaModel: 'Ollama 模型名',
72
72
  localOllamaTemperature: '温度 temperature',
73
73
  localOllamaTopP: 'top_p',
74
+ localMaxTokens: '最大输出 token',
75
+ localReasoningEffort: '推理强度',
76
+ localReasoningDefault: '服务端默认',
77
+ localReasoningNone: '关闭推理',
74
78
  localRequestFormat: '请求协议',
75
79
  localFormatOpenAI: 'OpenAI(/chat/completions)',
76
80
  localFormatAnthropic: 'Anthropic(/messages)',
81
+ localFormatLmStudio: 'LM Studio 原生(/api/v1/chat)',
77
82
  localTemperaturePlaceholder: '留空=服务端默认;建议 0.5',
78
83
  localTopPPlaceholder: '留空=服务端默认;建议 0.8',
79
84
  localLmStudioModelPlaceholder: '填写 Developer 页或 /v1/models 中的模型标识',
@@ -219,7 +224,7 @@ window.__ModuleLoader__.load({
219
224
  '用于兼容旧版行为,一般无需开启;开启后只使用上方识图模型链中的后端。',
220
225
  hintReverseRouting: '仅在「整轮交给视觉模型」开启时生效;纯文字消息继续交给聊天模型处理。',
221
226
  hintTool: '允许聊天模型按需查看、定位、裁剪和比较图片。推荐保持开启。',
222
- hintStructuredVisionBootstrap: '默认关闭。开启后,每个图片任务会先做一次不读取具体任务目标的结构化预识别,再由聊天模型根据原问题至少追加一次验证或深挖识图调用(1+x,x≥1)。启用的 Ollama / LM Studio 会和其他视觉后端一样参与这条识图链,不需要另开「即时识图」。准确性更高,但会增加至少一次视觉调用;若后续需要 OCR,engine=auto 仍会先尝试本地 Tesseract,失败或空结果再回退视觉模型;结构化模式不会改变这一顺序。需保持「识图工具」开启。',
227
+ hintStructuredVisionBootstrap: '默认关闭。开启后,每个图片任务会先做一次不读取具体任务目标的结构化预识别,再由聊天模型根据原问题至少追加一次验证或深挖识图调用(1+x,x≥1)。启用的 Ollama / LM Studio 会和其他视觉后端一样参与这条识图链,不需要另开「即时识图」。准确性更高,但会增加至少一次视觉调用;若后续需要 OCR,未显式指定 engine 时会遵循「OCR 默认引擎」设置;单次显式 engine=tesseract/vision 始终优先。需保持「识图工具」开启。',
223
228
  hintAutoWrapProviders: '自动为已启用的聊天模型创建带「+ 自动识图」的版本。原模型不受影响,推荐保持开启。',
224
229
  hintRewriteImages: '避免把无法读取的原始图片直接发送给纯文字模型。推荐保持开启。',
225
230
  hintDownscale: '超过像素预算的图片先缩放再送视觉模型,降低延迟与成本;默认开启。',
@@ -243,6 +248,11 @@ window.__ModuleLoader__.load({
243
248
  numTimeoutMs: '单次识图请求超时(毫秒)',
244
249
  numVisionTaskTimeoutMs: '识图任务最长时间(毫秒)',
245
250
  numOcrTimeoutMs: 'OCR 最长时间(毫秒)',
251
+ selectOcrEngine: 'OCR 默认引擎',
252
+ hintOcrEngine: '决定 vision_ocr 未显式指定 engine 时的默认策略。自动:先本地 Tesseract,失败/空结果再用视觉模型;Tesseract:只用本地 OCR;视觉模型:直接跳过 Tesseract。单次工具调用显式 engine=tesseract/vision 始终优先。',
253
+ ocrEngineAuto: '自动(Tesseract → 视觉模型)',
254
+ ocrEngineTesseract: '仅 Tesseract',
255
+ ocrEngineVision: '仅视觉模型',
246
256
  numDownscaleMaxPixels: '图片像素上限',
247
257
  numCacheTtlSeconds: '缓存有效期(秒)',
248
258
  numCacheMaxEntries: '最大缓存数量',
@@ -347,9 +357,14 @@ window.__ModuleLoader__.load({
347
357
  localOllamaModel: 'Ollama model name',
348
358
  localOllamaTemperature: 'Temperature',
349
359
  localOllamaTopP: 'Top-p',
360
+ localMaxTokens: 'Max output tokens',
361
+ localReasoningEffort: 'Reasoning effort',
362
+ localReasoningDefault: 'Provider default',
363
+ localReasoningNone: 'Disable reasoning',
350
364
  localRequestFormat: 'Request protocol',
351
365
  localFormatOpenAI: 'OpenAI (/chat/completions)',
352
366
  localFormatAnthropic: 'Anthropic (/messages)',
367
+ localFormatLmStudio: 'LM Studio native (/api/v1/chat)',
353
368
  localTemperaturePlaceholder: 'Blank = server default; suggested 0.5',
354
369
  localTopPPlaceholder: 'Blank = server default; suggested 0.8',
355
370
  localLmStudioModelPlaceholder: 'Model identifier from Developer or /v1/models',
@@ -496,7 +511,7 @@ window.__ModuleLoader__.load({
496
511
  'only the vision models configured above participate.',
497
512
  hintReverseRouting: 'Only applies when whole-turn vision routing is enabled; plain text messages continue to use the chat model.',
498
513
  hintTool: 'Lets the chat model inspect, locate, crop, and compare images as needed. Recommended on.',
499
- hintStructuredVisionBootstrap: 'Off by default. Each image task first gets one task-independent structured visual baseline with no task goal passed into the pre-scan, then the chat model must make at least one evidence or deepening vision call for the original request (1+x, x>=1). Enabled Ollama / LM Studio backends participate in this same vision chain; there is no separate instant-recognition switch. This improves evidence quality but adds at least one visual call. If OCR is needed, engine=auto still tries local Tesseract first and falls back to the vision model only when local OCR fails or returns no text; structured mode does not change this order. Keep Vision tools enabled.',
514
+ hintStructuredVisionBootstrap: 'Off by default. Each image task first gets one task-independent structured visual baseline with no task goal passed into the pre-scan, then the chat model must make at least one evidence or deepening vision call for the original request (1+x, x>=1). Enabled Ollama / LM Studio backends participate in this same vision chain; there is no separate instant-recognition switch. This improves evidence quality but adds at least one visual call. If OCR is needed, calls without an explicit engine follow the “Default OCR engine” setting; an explicit engine=tesseract/vision always wins for that call. Keep Vision tools enabled.',
500
515
  hintAutoWrapProviders: 'Automatically creates a “+ Auto Vision” version of enabled chat models. Original model groups stay unchanged. Recommended on.',
501
516
  hintRewriteImages: 'Prevents raw image content from being sent to a text-only model that cannot read it. Recommended on.',
502
517
  hintDownscale: 'Images beyond the pixel budget are resized before the vision call, cutting latency and cost; on by default.',
@@ -521,6 +536,11 @@ window.__ModuleLoader__.load({
521
536
  numTimeoutMs: 'Single vision request timeout (ms)',
522
537
  numVisionTaskTimeoutMs: 'Vision task maximum time (ms)',
523
538
  numOcrTimeoutMs: 'OCR maximum time (ms)',
539
+ selectOcrEngine: 'Default OCR engine',
540
+ hintOcrEngine: 'Default policy when vision_ocr does not explicitly choose an engine. Auto tries local Tesseract first and falls back to the vision model; Tesseract uses local OCR only; Vision skips Tesseract. An explicit engine=tesseract/vision on a tool call always wins.',
541
+ ocrEngineAuto: 'Auto (Tesseract → vision)',
542
+ ocrEngineTesseract: 'Tesseract only',
543
+ ocrEngineVision: 'Vision model only',
524
544
  numDownscaleMaxPixels: 'Image pixel limit',
525
545
  numCacheTtlSeconds: 'Cache TTL (seconds)',
526
546
  numCacheMaxEntries: 'Maximum cached answers',
@@ -719,10 +739,14 @@ window.__ModuleLoader__.load({
719
739
  localOllama: {
720
740
  baseURL: 'http://127.0.0.1:11434/v1',
721
741
  model: 'qwen2.5vl',
742
+ maxTokens: 4096,
743
+ reasoningEffort: 'none',
722
744
  },
723
745
  localLmStudio: {
724
746
  baseURL: 'http://localhost:1234/v1',
725
747
  model: '',
748
+ maxTokens: 4096,
749
+ reasoningEffort: 'none',
726
750
  },
727
751
  }
728
752
 
@@ -739,7 +763,17 @@ window.__ModuleLoader__.load({
739
763
  typeof input.model === 'string' && input.model !== ''
740
764
  ? input.model
741
765
  : defaults.model,
742
- format: input.format === 'anthropic' ? 'anthropic' : 'openai',
766
+ format: input.format === 'anthropic' ? 'anthropic' : input.format === 'lmstudio' ? 'lmstudio' : 'openai',
767
+ ...((Number.isInteger(input.maxTokens) ? input.maxTokens : defaults.maxTokens) === undefined
768
+ ? {}
769
+ : { maxTokens: Number.isInteger(input.maxTokens) ? input.maxTokens : defaults.maxTokens }),
770
+ ...((['provider_default', 'none', 'low', 'medium', 'high', 'max'].includes(input.reasoningEffort)
771
+ ? input.reasoningEffort
772
+ : defaults.reasoningEffort) === undefined
773
+ ? {}
774
+ : { reasoningEffort: ['provider_default', 'none', 'low', 'medium', 'high', 'max'].includes(input.reasoningEffort)
775
+ ? input.reasoningEffort
776
+ : defaults.reasoningEffort }),
743
777
  temperature:
744
778
  typeof input.temperature === 'number' && Number.isFinite(input.temperature)
745
779
  ? input.temperature
@@ -761,6 +795,12 @@ window.__ModuleLoader__.load({
761
795
  typeof input.top_p === 'number' && Number.isFinite(input.top_p)
762
796
  ? Math.min(1, Math.max(0, input.top_p))
763
797
  : undefined
798
+ const maxTokens = Number.isInteger(input.maxTokens)
799
+ ? Math.min(32768, Math.max(256, input.maxTokens))
800
+ : defaults.maxTokens
801
+ const reasoningEffort = ['provider_default', 'none', 'low', 'medium', 'high', 'max'].includes(input.reasoningEffort)
802
+ ? input.reasoningEffort
803
+ : defaults.reasoningEffort
764
804
  return {
765
805
  enabled: input.enabled === true,
766
806
  baseURL:
@@ -771,7 +811,9 @@ window.__ModuleLoader__.load({
771
811
  typeof input.model === 'string' && input.model.trim() !== ''
772
812
  ? input.model.trim()
773
813
  : defaults.model,
774
- format: input.format === 'anthropic' ? 'anthropic' : 'openai',
814
+ format: input.format === 'anthropic' ? 'anthropic' : input.format === 'lmstudio' ? 'lmstudio' : 'openai',
815
+ ...(maxTokens === undefined ? {} : { maxTokens }),
816
+ ...(reasoningEffort === undefined ? {} : { reasoningEffort }),
775
817
  ...(temperature === undefined ? {} : { temperature }),
776
818
  ...(topP === undefined ? {} : { top_p: topP }),
777
819
  }
@@ -865,6 +907,57 @@ window.__ModuleLoader__.load({
865
907
  : stored && settingsValueEqual(item.key, user[item.key], item.run.value)
866
908
  return { ok, stored }
867
909
  }
910
+ const finalize = () => {
911
+ const landed = failures.length === 0
912
+ const nextDrafts = landedFields.length === 0 ? drafts : { ...drafts }
913
+ for (const field of landedFields) delete nextDrafts[field]
914
+ return {
915
+ landed,
916
+ failed: !landed,
917
+ landedFields,
918
+ nextDrafts,
919
+ failures,
920
+ }
921
+ }
922
+ const mismatchFailure = (item, check) => ({
923
+ field: item.key,
924
+ operation: item.run.clear ? 'unset' : 'set',
925
+ reason: 'readback-mismatch',
926
+ detail: item.run.clear
927
+ ? 'field remained present in the user layer'
928
+ : check.stored
929
+ ? 'stored user-layer value differs from the requested value'
930
+ : 'field is absent from the user layer',
931
+ })
932
+ const batchWrite = scope && typeof scope.__visionRouterWritePlan === 'function'
933
+ ? scope.__visionRouterWritePlan.bind(scope)
934
+ : undefined
935
+ // DSH 0.1.7's local compatibility transport is ConfigEditor-backed.
936
+ // ConfigEditor reloads the plugin after a successful edit, so serial POSTs
937
+ // from one Save can cross generations and strand the second mutation.
938
+ // Its private scope seam commits the entire UI plan in one atomic edit.
939
+ if (plan.length > 1 && batchWrite) {
940
+ try {
941
+ await batchWrite(plan)
942
+ } catch (error) {
943
+ for (const item of plan) {
944
+ failures.push({
945
+ field: item.key,
946
+ operation: item.run.clear ? 'unset' : 'set',
947
+ reason: error && error.code === 'settings-conflict' ? 'settings-conflict' : 'write-error',
948
+ detail: settingsSaveErrorMessage(error),
949
+ })
950
+ }
951
+ return finalize()
952
+ }
953
+ for (const item of plan) {
954
+ const check = inspectReadback(item)
955
+ if (check.error) failures.push(check.error)
956
+ else if (check.ok) landedFields.push(item.key)
957
+ else failures.push(mismatchFailure(item, check))
958
+ }
959
+ return finalize()
960
+ }
868
961
  const writeItem = async (item) => {
869
962
  if (item.run.clear) await scope.unset(item.key)
870
963
  else await scope.set(item.key, item.run.value)
@@ -913,16 +1006,7 @@ window.__ModuleLoader__.load({
913
1006
  }
914
1007
  }
915
1008
 
916
- terminalFailure = {
917
- field: item.key,
918
- operation,
919
- reason: 'readback-mismatch',
920
- detail: item.run.clear
921
- ? 'field remained present in the user layer'
922
- : check.stored
923
- ? 'stored user-layer value differs from the requested value'
924
- : 'field is absent from the user layer',
925
- }
1009
+ terminalFailure = mismatchFailure(item, check)
926
1010
  }
927
1011
  if (success) landedFields.push(item.key)
928
1012
  else failures.push(terminalFailure ?? {
@@ -932,16 +1016,7 @@ window.__ModuleLoader__.load({
932
1016
  detail: 'write did not become visible in the user layer',
933
1017
  })
934
1018
  }
935
- const landed = failures.length === 0
936
- const nextDrafts = landedFields.length === 0 ? drafts : { ...drafts }
937
- for (const field of landedFields) delete nextDrafts[field]
938
- return {
939
- landed,
940
- failed: !landed,
941
- landedFields,
942
- nextDrafts,
943
- failures,
944
- }
1019
+ return finalize()
945
1020
  }
946
1021
 
947
1022
  const REMOTE_SETTINGS_CHANNEL = '/vision-router-settings'
@@ -1124,16 +1199,23 @@ window.__ModuleLoader__.load({
1124
1199
  }
1125
1200
 
1126
1201
  function shouldUseRemoteSettings(getConnection, locationLike) {
1202
+ // Browser page authority is the security boundary. DSH 0.1.7 can report
1203
+ // Connection.isLoopback=false even for a real 127.0.0.1 page, so letting
1204
+ // that hint win would incorrectly force the local Settings page onto the
1205
+ // trusted-host RPC path. Conversely, a non-loopback browser URL must not
1206
+ // become local merely because a stale Connection says true.
1207
+ const location = locationLike ?? (typeof window !== 'undefined' ? window.location : undefined)
1208
+ const hostname = typeof location?.hostname === 'string' ? location.hostname.toLowerCase().replace(/^\[|\]$/g, '') : ''
1209
+ if (hostname !== '') {
1210
+ if (hostname === 'localhost' || hostname.endsWith('.localhost') || hostname === '::1') return false
1211
+ if (/^127(?:\.\d{1,3}){3}$/.test(hostname)) return false
1212
+ return true
1213
+ }
1127
1214
  try {
1128
1215
  const connection = typeof getConnection === 'function' ? getConnection() : undefined
1129
1216
  if (connection && typeof connection.isLoopback === 'boolean') return connection.isLoopback === false
1130
- } catch { /* fall through to page authority */ }
1131
- const location = locationLike ?? (typeof window !== 'undefined' ? window.location : undefined)
1132
- const hostname = typeof location?.hostname === 'string' ? location.hostname.toLowerCase().replace(/^\[|\]$/g, '') : ''
1133
- if (hostname === '') return false
1134
- if (hostname === 'localhost' || hostname.endsWith('.localhost') || hostname === '::1') return false
1135
- if (/^127(?:\.\d{1,3}){3}$/.test(hostname)) return false
1136
- return true
1217
+ } catch { /* no page authority and no usable Connection: stay local-safe */ }
1218
+ return false
1137
1219
  }
1138
1220
 
1139
1221
  function reportSettingsSaveFailures(failures) {
@@ -2331,6 +2413,7 @@ window.__ModuleLoader__.load({
2331
2413
  timeoutMs: 'numTimeoutMs',
2332
2414
  visionTaskTimeoutMs: 'numVisionTaskTimeoutMs',
2333
2415
  ocrTimeoutMs: 'numOcrTimeoutMs',
2416
+ ocrEngine: 'selectOcrEngine',
2334
2417
  downscaleMaxPixels: 'numDownscaleMaxPixels',
2335
2418
  cacheTtlSeconds: 'numCacheTtlSeconds',
2336
2419
  cacheMaxEntries: 'numCacheMaxEntries',
@@ -2359,6 +2442,7 @@ window.__ModuleLoader__.load({
2359
2442
  timeoutMs: 'numHintTimeoutMs',
2360
2443
  visionTaskTimeoutMs: 'numHintVisionTaskTimeoutMs',
2361
2444
  ocrTimeoutMs: 'numHintOcrTimeoutMs',
2445
+ ocrEngine: 'hintOcrEngine',
2362
2446
  downscaleMaxPixels: 'numHintDownscaleMaxPixels',
2363
2447
  cacheTtlSeconds: 'numHintCacheTtlSeconds',
2364
2448
  cacheMaxEntries: 'numHintCacheMaxEntries',
@@ -2756,6 +2840,7 @@ window.__ModuleLoader__.load({
2756
2840
  return normalizeLocalProviderDraft(value, LOCAL_PROVIDER_DEFAULTS.localLmStudio)
2757
2841
  }
2758
2842
  if (key === 'localDescribeStyle') return value === 'structured' ? 'structured' : 'plain'
2843
+ if (key === 'ocrEngine') return value === 'tesseract' || value === 'vision' ? value : 'auto'
2759
2844
  if (SELECT_KEYS.includes(key)) {
2760
2845
  if (value === 'custom') return 'standard'
2761
2846
  return value === 'fast' || value === 'standard' || value === 'deep' ? value : 'standard'
@@ -2816,6 +2901,9 @@ window.__ModuleLoader__.load({
2816
2901
  if (key === 'localDescribeStyle') {
2817
2902
  return text === 'structured' || text === 'plain' ? { value: text } : undefined
2818
2903
  }
2904
+ if (key === 'ocrEngine') {
2905
+ return text === 'auto' || text === 'tesseract' || text === 'vision' ? { value: text } : undefined
2906
+ }
2819
2907
  if (SELECT_KEYS.includes(key)) {
2820
2908
  return text === 'fast' || text === 'standard' || text === 'deep' ? { value: text } : undefined
2821
2909
  }
@@ -3314,6 +3402,26 @@ window.__ModuleLoader__.load({
3314
3402
  h('option', { value: 'anthropic' }, t('localFormatAnthropic')),
3315
3403
  ),
3316
3404
  ),
3405
+ h('div', { className: 'vr-local-row vr-local-row-pair' },
3406
+ h('label', { className: 'vr-label vr-local-label' }, t('localMaxTokens')),
3407
+ h('input', {
3408
+ className: 'vr-input', value: String(value.maxTokens ?? 4096), disabled: editBlocked,
3409
+ type: 'number', step: '256', min: '256', max: '32768',
3410
+ onChange: (event) => update({ maxTokens: Number(event.target.value) }),
3411
+ }),
3412
+ h('label', { className: 'vr-label vr-local-label' }, t('localReasoningEffort')),
3413
+ h('select', {
3414
+ className: 'vr-input vr-select', value: value.reasoningEffort, disabled: editBlocked || value.format === 'anthropic',
3415
+ onChange: (event) => update({ reasoningEffort: event.target.value }),
3416
+ },
3417
+ h('option', { value: 'provider_default' }, t('localReasoningDefault')),
3418
+ h('option', { value: 'none' }, t('localReasoningNone')),
3419
+ h('option', { value: 'low' }, 'low'),
3420
+ h('option', { value: 'medium' }, 'medium'),
3421
+ h('option', { value: 'high' }, 'high'),
3422
+ h('option', { value: 'max' }, 'max'),
3423
+ ),
3424
+ ),
3317
3425
  h('div', { className: 'vr-local-row vr-local-row-pair' },
3318
3426
  h('label', { className: 'vr-label vr-local-label' }, t('localOllamaTemperature')),
3319
3427
  h('input', {
@@ -3378,6 +3486,27 @@ window.__ModuleLoader__.load({
3378
3486
  },
3379
3487
  h('option', { value: 'openai' }, t('localFormatOpenAI')),
3380
3488
  h('option', { value: 'anthropic' }, t('localFormatAnthropic')),
3489
+ h('option', { value: 'lmstudio' }, t('localFormatLmStudio')),
3490
+ ),
3491
+ ),
3492
+ h('div', { className: 'vr-local-row vr-local-row-pair' },
3493
+ h('label', { className: 'vr-label vr-local-label' }, t('localMaxTokens')),
3494
+ h('input', {
3495
+ className: 'vr-input', value: String(value.maxTokens ?? 4096), disabled: editBlocked,
3496
+ type: 'number', step: '256', min: '256', max: '32768',
3497
+ onChange: (event) => update({ maxTokens: Number(event.target.value) }),
3498
+ }),
3499
+ h('label', { className: 'vr-label vr-local-label' }, t('localReasoningEffort')),
3500
+ h('select', {
3501
+ className: 'vr-input vr-select', value: value.reasoningEffort, disabled: editBlocked || value.format !== 'lmstudio',
3502
+ onChange: (event) => update({ reasoningEffort: event.target.value }),
3503
+ },
3504
+ h('option', { value: 'provider_default' }, t('localReasoningDefault')),
3505
+ h('option', { value: 'none' }, t('localReasoningNone')),
3506
+ h('option', { value: 'low' }, 'low'),
3507
+ h('option', { value: 'medium' }, 'medium'),
3508
+ h('option', { value: 'high' }, 'high'),
3509
+ h('option', { value: 'max' }, 'max'),
3381
3510
  ),
3382
3511
  ),
3383
3512
  h('div', { className: 'vr-local-row vr-local-row-pair' },
@@ -4083,6 +4212,11 @@ window.__ModuleLoader__.load({
4083
4212
  h('p', { className: 'vr-group-title' }, t('groupPerformance')),
4084
4213
  PERFORMANCE_TOGGLE_KEYS.map((key) => toggleField(key)),
4085
4214
  NUMBER_KEYS.map((key) => textField(key, t(LABEL_KEY[key]), t(HINT_KEY[key]), false)),
4215
+ selectField('ocrEngine', t(LABEL_KEY.ocrEngine), t(HINT_KEY.ocrEngine), [
4216
+ { value: 'auto', label: t('ocrEngineAuto') },
4217
+ { value: 'tesseract', label: t('ocrEngineTesseract') },
4218
+ { value: 'vision', label: t('ocrEngineVision') },
4219
+ ]),
4086
4220
  ),
4087
4221
  h('div', { className: 'vr-group' },
4088
4222
  h('p', { className: 'vr-group-title' }, t('groupCompatibility')),
@@ -1439,8 +1439,9 @@ export function posterizeSvgColor(data, info, palette, timeoutMs = 60000) {
1439
1439
  }
1440
1440
 
1441
1441
  /** Resolve the effective vision_ocr engine without hiding explicit user/model intent. */
1442
- export function resolveVisionOcrEngine(requestedEngine) {
1442
+ export function resolveVisionOcrEngine(requestedEngine, configuredEngine = 'auto') {
1443
1443
  if (requestedEngine === 'tesseract' || requestedEngine === 'vision') return requestedEngine
1444
+ if (configuredEngine === 'tesseract' || configuredEngine === 'vision') return configuredEngine
1444
1445
  return 'auto'
1445
1446
  }
1446
1447
 
@@ -1857,11 +1858,15 @@ export function localOllamaProvidersOf(config) {
1857
1858
  baseURL,
1858
1859
  model,
1859
1860
  apiKeyEnv: '',
1860
- maxTokens: 2048,
1861
+ maxTokens: Number.isInteger(local.maxTokens) ? local.maxTokens : 4096,
1861
1862
  // Ollama's OpenAI-compatible Chat Completions endpoint supports
1862
1863
  // reasoning_effort='none'. Vision extraction should spend the bounded
1863
1864
  // completion budget on observable answer text rather than hidden thought.
1864
- reasoningEffort: 'none',
1865
+ ...(local.reasoningEffort === 'provider_default'
1866
+ ? {}
1867
+ : { reasoningEffort: ['none', 'low', 'medium', 'high', 'max'].includes(local.reasoningEffort)
1868
+ ? local.reasoningEffort
1869
+ : 'none' }),
1865
1870
  // 仅显式选择 anthropic 格式时携带(默认 openai 路径保持字节不变)。
1866
1871
  ...(local.format === 'anthropic' ? { format: 'anthropic' } : {}),
1867
1872
  // 建议值透传:温度/top_p 只在显式配置时携带(callOpenAICompatible
@@ -1889,8 +1894,12 @@ export function localLmStudioProvidersOf(config) {
1889
1894
  baseURL,
1890
1895
  model,
1891
1896
  apiKeyEnv: '',
1892
- maxTokens: 2048,
1897
+ maxTokens: Number.isInteger(local.maxTokens) ? local.maxTokens : 4096,
1898
+ ...(local.format === 'lmstudio' && ['none', 'low', 'medium', 'high', 'max'].includes(local.reasoningEffort)
1899
+ ? { reasoningEffort: local.reasoningEffort }
1900
+ : {}),
1893
1901
  ...(local.format === 'anthropic' ? { format: 'anthropic' } : {}),
1902
+ ...(local.format === 'lmstudio' ? { format: 'lmstudio' } : {}),
1894
1903
  ...(typeof local.temperature === 'number' ? { temperature: local.temperature } : {}),
1895
1904
  ...(typeof local.top_p === 'number' ? { top_p: local.top_p } : {}),
1896
1905
  },
@@ -1917,12 +1926,108 @@ export function localProvidersOf(config) {
1917
1926
  * temperature/top_p 仅显式配置时透传(两个 transport 的显式可选参数,
1918
1927
  * 现有调用不传,wire 保持 main 原样)。
1919
1928
  */
1929
+ function lmStudioNativeContent(messages) {
1930
+ const system = []
1931
+ const input = []
1932
+ for (const message of messages ?? []) {
1933
+ if (!message) continue
1934
+ const blocks = Array.isArray(message.content)
1935
+ ? message.content
1936
+ : typeof message.content === 'string'
1937
+ ? [{ type: 'text', text: message.content }]
1938
+ : []
1939
+ if (message.role === 'system') {
1940
+ for (const block of blocks) {
1941
+ if (block?.type === 'text' && typeof block.text === 'string' && block.text.trim() !== '') {
1942
+ system.push(block.text)
1943
+ }
1944
+ }
1945
+ continue
1946
+ }
1947
+ for (const block of blocks) {
1948
+ if (block?.type === 'text' && typeof block.text === 'string') {
1949
+ input.push({ type: 'text', content: block.text })
1950
+ } else if (block?.type === 'image_url' && typeof block.image_url?.url === 'string') {
1951
+ input.push({ type: 'image', data_url: block.image_url.url })
1952
+ }
1953
+ }
1954
+ }
1955
+ return { systemPrompt: system.join('\n').trim(), input }
1956
+ }
1957
+
1958
+ async function callLmStudioNative(provider, messages, options = {}) {
1959
+ const normalizedBaseURL = stripTrailingSlashes(String(provider.baseURL ?? 'http://localhost:1234/v1'))
1960
+ const apiRoot = normalizedBaseURL.endsWith('/v1') ? normalizedBaseURL.slice(0, -3) : normalizedBaseURL
1961
+ const { systemPrompt, input } = lmStudioNativeContent(messages)
1962
+ const reasoningEffort = options.reasoningEffort ?? provider.reasoningEffort
1963
+ const reasoning = reasoningEffort === 'none'
1964
+ ? 'off'
1965
+ : ['low', 'medium', 'high'].includes(reasoningEffort)
1966
+ ? reasoningEffort
1967
+ : reasoningEffort === 'max'
1968
+ ? 'high'
1969
+ : undefined
1970
+ const body = {
1971
+ model: provider.model,
1972
+ input,
1973
+ store: false,
1974
+ max_output_tokens: options.maxTokens ?? provider.maxTokens ?? 4096,
1975
+ ...(systemPrompt === '' ? {} : { system_prompt: systemPrompt }),
1976
+ ...(reasoning === undefined ? {} : { reasoning }),
1977
+ ...(typeof options.temperature === 'number' ? { temperature: options.temperature } : {}),
1978
+ ...(typeof options.top_p === 'number' ? { top_p: options.top_p } : {}),
1979
+ }
1980
+ const response = await fetch(`${apiRoot}/api/v1/chat`, {
1981
+ method: 'POST',
1982
+ headers: { 'content-type': 'application/json' },
1983
+ body: JSON.stringify(body),
1984
+ ...(options.signal === undefined ? {} : { signal: options.signal }),
1985
+ })
1986
+ if (!response.ok) {
1987
+ const detail = (await readResponseTextBounded(
1988
+ response,
1989
+ ERROR_RESPONSE_MAX_BYTES,
1990
+ { label: 'LM Studio native error response' },
1991
+ ).catch(() => '')).slice(0, 300)
1992
+ const error = new Error(`http provider "${provider.name}": ${response.status} ${detail}`)
1993
+ error.status = response.status
1994
+ error.code = kindForHttpStatus(response.status) ?? 'HTTP_PROVIDER_FAILED'
1995
+ throw error
1996
+ }
1997
+ const data = await readResponseJsonBounded(
1998
+ response,
1999
+ MODEL_RESPONSE_MAX_BYTES,
2000
+ { label: 'LM Studio native response' },
2001
+ )
2002
+ const text = Array.isArray(data?.output)
2003
+ ? data.output
2004
+ .filter((item) => item?.type === 'message' && typeof item.content === 'string')
2005
+ .map((item) => item.content)
2006
+ .join('\n')
2007
+ .trim()
2008
+ : ''
2009
+ if (text === '' && Number(data?.stats?.total_output_tokens) >= body.max_output_tokens) {
2010
+ const error = new Error(`http provider "${provider.name}": empty answer after exhausting the completion budget`)
2011
+ error.code = 'VISION_EMPTY_RESPONSE'
2012
+ throw error
2013
+ }
2014
+ return text
2015
+ }
2016
+
1920
2017
  export async function callLocalBackend(provider, messages, options = {}) {
1921
2018
  const maxTokens = options.maxTokens ?? provider.maxTokens ?? 2048
1922
2019
  const sampling = {
1923
2020
  ...(typeof provider.temperature === 'number' ? { temperature: provider.temperature } : {}),
1924
2021
  ...(typeof provider.top_p === 'number' ? { top_p: provider.top_p } : {}),
1925
2022
  }
2023
+ if (provider.format === 'lmstudio') {
2024
+ return callLmStudioNative(provider, messages, {
2025
+ maxTokens,
2026
+ signal: options.signal,
2027
+ ...(typeof provider.reasoningEffort === 'string' ? { reasoningEffort: provider.reasoningEffort } : {}),
2028
+ ...sampling,
2029
+ })
2030
+ }
1926
2031
  if (provider.format === 'anthropic') {
1927
2032
  const system = []
1928
2033
  const wire = []
@@ -2182,11 +2287,16 @@ export async function callOpenAICompatible(provider, messages, options = {}) {
2182
2287
  MODEL_RESPONSE_MAX_BYTES,
2183
2288
  { label: `http provider \"${provider.name}\" response` },
2184
2289
  )
2185
- const content = data && data.choices && data.choices[0] && data.choices[0].message
2186
- ? data.choices[0].message.content
2187
- : undefined
2290
+ const choice = data && data.choices && data.choices[0] ? data.choices[0] : undefined
2291
+ const content = choice && choice.message ? choice.message.content : undefined
2188
2292
  if (typeof content !== 'string') throw new Error(`http provider "${provider.name}": unexpected response shape`)
2189
- return content.trim()
2293
+ const text = content.trim()
2294
+ if (text === '' && choice?.finish_reason === 'length' && (provider.name === 'local-ollama' || provider.name === 'local-lmstudio')) {
2295
+ const error = new Error(`http provider "${provider.name}": empty answer after exhausting the completion budget`)
2296
+ error.code = 'VISION_EMPTY_RESPONSE'
2297
+ throw error
2298
+ }
2299
+ return text
2190
2300
  }
2191
2301
 
2192
2302
  /**
@@ -43,6 +43,7 @@ export function inspectDshHostCapabilities(ctx) {
43
43
  const llm = serviceOf(ctx, 'llm')
44
44
  const jobs = serviceOf(ctx, 'jobs')
45
45
  const settings = serviceOf(ctx, 'settings')
46
+ const configEditor = serviceOf(ctx, 'configEditor')
46
47
  const tools = serviceOf(ctx, 'tools')
47
48
 
48
49
  let maxImageDimension = false
@@ -57,6 +58,10 @@ export function inspectDshHostCapabilities(ctx) {
57
58
  const prepareCall = hasFunction(llm, 'prepareCall')
58
59
  const toolRegistration = hasFunction(tools, 'register') || hasFunction(tools, 'registerTool')
59
60
  const toolExecution = hasFunction(tools, 'call') || hasFunction(tools, 'execute') || hasFunction(tools, 'invoke')
61
+ const settingsLiveNamespace = hasFunction(settings, 'register')
62
+ || (hasFunction(settings, 'describe')
63
+ && hasFunction(configEditor, 'configuration')
64
+ && hasFunction(configEditor, 'edit'))
60
65
 
61
66
  return Object.freeze({
62
67
  batchAttachments: hasFunction(attachments, 'saveImages'),
@@ -70,10 +75,11 @@ export function inspectDshHostCapabilities(ctx) {
70
75
  // No stable read-only client-surface capability flag exists across the
71
76
  // supported Host window yet.
72
77
  surfaceReplacement: UNKNOWN,
73
- // DSH SettingsProvider.register() is the public live-namespace seam; get()
74
- // and watch() belong to the returned SettingsScope rather than the service.
75
- // Detect the stable service method and do not manufacture a registration.
76
- settingsLiveNamespace: hasFunction(settings, 'register'),
78
+ // Older Hosts expose SettingsProvider.register() directly. DSH 0.1.7's
79
+ // reviewed replacement is SettingsForms.describe() plus ConfigEditor
80
+ // configuration()/edit(), which DVR adapts into the same live namespace.
81
+ // Probe only readable method presence; never manufacture a registration or write.
82
+ settingsLiveNamespace,
77
83
  // Browser exposure is a presentation capability and is not safely implied
78
84
  // by the Host settings service alone.
79
85
  settingsWebExposure: UNKNOWN,