newmark-agent 0.5.15 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1097,6 +1097,7 @@ export declare class Agent {
1097
1097
  evaluateAndSwitch(task: string, override?: AgentPromptMessage['routePolicy']): Promise<boolean>;
1098
1098
  shouldExposeToolInterface(): boolean;
1099
1099
  modelIsUnavailable(modelName: string): boolean;
1100
+ private failedDeploymentsThisRun;
1100
1101
  switchToFallbackModel(errorText?: string): string | null;
1101
1102
  isLlmErrorText(text: string): boolean;
1102
1103
  private scopedSwitchModels;
@@ -1109,8 +1110,8 @@ export declare class Agent {
1109
1110
  /**
1110
1111
  * Final visual safety net: OCR each submitted image and ask a text-only
1111
1112
  * request to conservatively repair the OCR. This is intentionally callable
1112
- * only after a visual-input refusal and after same-provider vision routing
1113
- * has been exhausted; the original image is never sent again.
1113
+ * only after a real visual-input refusal. Cached capabilities never bypass
1114
+ * the initial image request, and OCR does not disable later image attempts.
1114
1115
  */
1115
1116
  finalVisualFallback(errorText: string, signal?: AbortSignal): Promise<string | null>;
1116
1117
  engineModel(cache?: BuildProviderCache): LLMProvider | null;
@@ -35,6 +35,7 @@ var __importStar = (this && this.__importStar) || (function () {
35
35
  Object.defineProperty(exports, "__esModule", { value: true });
36
36
  exports.Agent = exports.ROOT_AGENT_ACTOR_ID = void 0;
37
37
  exports.normalizeIntelligenceTier = normalizeIntelligenceTier;
38
+ const modelResponseHealth_1 = require("./modelResponseHealth");
38
39
  const fs = __importStar(require("fs"));
39
40
  const path = __importStar(require("path"));
40
41
  const crypto = __importStar(require("crypto"));
@@ -939,13 +940,12 @@ class Agent {
939
940
  return this.modelSelectionValue();
940
941
  }
941
942
  defaultModelCandidate() {
942
- const rejected = new Set(['unavailable', 'auth_error', 'invalid_config']);
943
943
  return this.config.allModels()
944
944
  .map((model, index) => {
945
945
  const validationStatus = effectiveModelValidationStatus(model);
946
946
  const evaluationStatus = validationStatus === 'degraded' ? 'degraded' : String(model.evaluation?.status || '').toLowerCase();
947
947
  const level = String(model.validation?.level || '').toLowerCase();
948
- if (model.enabled === false || rejected.has(validationStatus) || rejected.has(evaluationStatus) || evaluationStatus.startsWith('error')) {
948
+ if (model.enabled === false) {
949
949
  return null;
950
950
  }
951
951
  let score = 0;
@@ -1420,8 +1420,8 @@ class Agent {
1420
1420
  const reference = parsed.image;
1421
1421
  const hydrated = this.hydrateDisplayImage(reference);
1422
1422
  const model = this.activeModelConfig();
1423
- const provider = model?.vision ? this.engineModel() : null;
1424
- if (!provider || !model?.vision || !hydrated?.dataUrl || !hydrated.sha256)
1423
+ const provider = this.engineModel();
1424
+ if (!provider || !hydrated?.dataUrl || !hydrated.sha256)
1425
1425
  return raw;
1426
1426
  throwIfAgentAborted(signal);
1427
1427
  const deployment = this.activeDeployment();
@@ -3905,7 +3905,9 @@ class Agent {
3905
3905
  // The title request uses the same transport policy and cancellation owner
3906
3906
  // as the formal response. A healthy slow model must not be turned into
3907
3907
  // five failed requests by an unrelated 15-second title deadline.
3908
- const generated = await this.chatWithConversationUsage(provider, modelName, [{ role: 'user', content: prompt }], system, temperature, 64, signal, reasoningEffort);
3908
+ // Reasoning tokens share the completion budget. A 64-token cap can
3909
+ // truncate even a short title and block every first conversation request.
3910
+ const generated = await this.chatWithConversationUsage(provider, modelName, [{ role: 'user', content: prompt }], system, temperature, 2048, signal, reasoningEffort);
3909
3911
  if (signal?.aborted)
3910
3912
  return '';
3911
3913
  const title = this.normalizeConversationRenameTitle(generated);
@@ -6183,10 +6185,13 @@ class Agent {
6183
6185
  if (active)
6184
6186
  return active;
6185
6187
  }
6186
- const byName = this.config.findModel(modelName);
6188
+ const parsed = parseDeploymentSelectionValue(modelName);
6189
+ const byName = parsed ? this.config.findDeployment(parsed) : this.config.findModel(modelName);
6187
6190
  if (byName)
6188
6191
  return byName;
6189
- return this.config.findModel(this.config.getStr('models', 'default_model'));
6192
+ const configured = this.config.getStr('models', 'default_model');
6193
+ const configuredRef = parseDeploymentSelectionValue(configured);
6194
+ return configuredRef ? this.config.findDeployment(configuredRef) : this.config.findModel(configured);
6190
6195
  }
6191
6196
  contextMaxTokens(modelName = this.model) {
6192
6197
  const model = this.resolveWindowModel(modelName);
@@ -6978,7 +6983,9 @@ class Agent {
6978
6983
  const active = this.activeModelConfig();
6979
6984
  if (active?.provider_id)
6980
6985
  return active.provider_id;
6981
- const fallback = this.config.findModel(this.config.getStr('models', 'default_model'));
6986
+ const defaultSelection = this.config.getStr('models', 'default_model');
6987
+ const defaultRef = parseDeploymentSelectionValue(defaultSelection);
6988
+ const fallback = defaultRef ? this.config.findDeployment(defaultRef) : this.config.findModel(defaultSelection);
6982
6989
  return fallback?.provider_id || this.config.providers()[0]?.id || '';
6983
6990
  }
6984
6991
  autoSwitchSubset() {
@@ -7096,7 +7103,7 @@ class Agent {
7096
7103
  preference: Number.isFinite(Number(routeMetadata.route_preference)) ? Number(routeMetadata.route_preference) : undefined,
7097
7104
  fallbackOnly: !!model.fallback_only,
7098
7105
  };
7099
- }).filter(candidate => !this.isBalanceBlockedDeployment(candidate.deployment));
7106
+ });
7100
7107
  }
7101
7108
  persistRouteDecision(decision) {
7102
7109
  try {
@@ -7143,15 +7150,11 @@ class Agent {
7143
7150
  catch { }
7144
7151
  }
7145
7152
  allModelNames() {
7146
- const names = this.config.allModels().filter(m => {
7147
- const status = String(m.evaluation?.status || 'unvalidated');
7148
- const validationStatus = m.validation?.status;
7149
- return status === 'available' || status === 'unvalidated' || validationStatus === 'verified' || validationStatus === 'degraded';
7150
- }).map(m => {
7153
+ const names = this.config.allModels().filter(m => m.enabled !== false).map(m => {
7151
7154
  const label = m.display || m.name;
7152
7155
  return `${m.provider} / ${label}`;
7153
7156
  });
7154
- return this.config.autoSwitchEnabled() && this.config.allModels().length > 0 ? ['auto', ...names] : names;
7157
+ return this.config.autoSwitchEnabled() && names.length > 0 ? ['auto', ...names] : names;
7155
7158
  }
7156
7159
  async evaluateAndSwitch(task, override = undefined) {
7157
7160
  if (!this.config.autoSwitchEnabled() || this.model !== 'auto')
@@ -7241,39 +7244,22 @@ class Agent {
7241
7244
  return !!this.resolvedDeployment && before !== deploymentIdentity(this.resolvedDeployment);
7242
7245
  }
7243
7246
  shouldExposeToolInterface() {
7244
- // Fixed selections and pre-Standard legacy configurations keep the
7245
- // historical behavior. Auto can safely suppress schemas because its
7246
- // eligible deployments have explicit Standard/Extended evidence.
7247
- if (this.model !== 'auto')
7248
- return true;
7249
- const model = this.activeModelConfig();
7250
- if (!model)
7251
- return false;
7252
- const validation = model.validation;
7253
- if (validation?.level !== 'standard' && validation?.level !== 'extended')
7254
- return true;
7255
- return validation.capabilities?.tool_use === true || validation.capabilities?.tools === true;
7247
+ return true;
7256
7248
  }
7257
7249
  modelIsUnavailable(modelName) {
7258
7250
  const model = modelName === this.model || modelName === 'auto'
7259
- ? this.activeModelConfig()
7260
- : this.config.findModel(modelName);
7261
- if (!model)
7262
- return true;
7263
- const validationStatus = effectiveModelValidationStatus(model);
7264
- // `discovered` means unvalidated, not a failed endpoint. Explicit fixed
7265
- // selections remain usable for backwards compatibility; only an executed
7266
- // Basic/Standard/Extended validation may pre-emptively mark them bad.
7267
- if (model.validation?.level !== 'discovered'
7268
- && (validationStatus === 'unavailable' || validationStatus === 'auth_error' || validationStatus === 'invalid_config'))
7269
- return true;
7270
- const status = validationStatus === 'degraded' ? 'degraded' : String(model?.evaluation?.status || '').toLowerCase();
7271
- return status === 'unavailable' || status.startsWith('error');
7251
+ ? this.activeModelConfig() : (parseDeploymentSelectionValue(modelName) ? this.config.findDeployment(parseDeploymentSelectionValue(modelName)) : this.config.findModel(modelName));
7252
+ // Only configuration/user switches can prevent a request. Response and
7253
+ // validation observations are labels, never permission to disable a model.
7254
+ return !model || model.enabled === false;
7272
7255
  }
7256
+ failedDeploymentsThisRun = new Set();
7273
7257
  switchToFallbackModel(errorText = 'transport failure') {
7274
7258
  const fallbackEnabled = this.config.getBool('models', 'fallback_on_unavailable');
7275
7259
  const observedFailure = (0, autoRouter_1.classifyRouteFailure)(errorText);
7276
7260
  const observedDeployment = this.activeDeployment();
7261
+ if (observedDeployment)
7262
+ this.failedDeploymentsThisRun.add(deploymentIdentity(observedDeployment));
7277
7263
  // Keep balance exhaustion scoped to the deployment that actually failed.
7278
7264
  // Provider adapters normally record this before returning an error, but
7279
7265
  // fallback callers are also a public recovery boundary and must not rely
@@ -7345,7 +7331,7 @@ class Agent {
7345
7331
  || deploymentIdentity(this.deploymentRef(m)) !== deploymentIdentity(currentDeployment));
7346
7332
  if (!all.length)
7347
7333
  return null;
7348
- const usable = all.filter(m => !this.isBalanceBlockedDeployment(this.deploymentRef(m)) && !modelConfigIsUnavailable(m));
7334
+ const usable = all.filter(m => !modelConfigIsUnavailable(m) && !this.failedDeploymentsThisRun.has(deploymentIdentity(this.deploymentRef(m))));
7349
7335
  if (!usable.length)
7350
7336
  return null;
7351
7337
  const pref = this.config.autoSwitchPreference();
@@ -7377,7 +7363,9 @@ class Agent {
7377
7363
  // which is exactly the same-name cross-provider leak this guard exists to
7378
7364
  // prevent. When no provider can be established unambiguously, return an
7379
7365
  // empty pool: failing the switch is safe, crossing providers is not.
7380
- const current = currentModelName === 'auto' ? this.activeModelConfig() : this.config.findModel(currentModelName);
7366
+ const currentRef = parseDeploymentSelectionValue(currentModelName);
7367
+ const current = currentModelName === 'auto' ? this.activeModelConfig()
7368
+ : (currentRef ? this.config.findDeployment(currentRef) : this.config.findModel(currentModelName));
7381
7369
  const providerId = current?.provider_id
7382
7370
  || this.fixedDeployment?.providerId
7383
7371
  || (currentModelName !== 'auto' ? this.activeDeployment()?.providerId : undefined);
@@ -7577,8 +7565,8 @@ class Agent {
7577
7565
  /**
7578
7566
  * Final visual safety net: OCR each submitted image and ask a text-only
7579
7567
  * request to conservatively repair the OCR. This is intentionally callable
7580
- * only after a visual-input refusal and after same-provider vision routing
7581
- * has been exhausted; the original image is never sent again.
7568
+ * only after a real visual-input refusal. Cached capabilities never bypass
7569
+ * the initial image request, and OCR does not disable later image attempts.
7582
7570
  */
7583
7571
  async finalVisualFallback(errorText, signal) {
7584
7572
  if (!/(?:vision|image|multimodal|image_url|input_image).*(?:not supported|unsupported|拒绝|不支持|failed|failure|invalid)|(?:not supported|unsupported|拒绝|不支持).*(?:vision|image|multimodal|image_url|input_image)/i.test(String(errorText || '')))
@@ -7586,11 +7574,6 @@ class Agent {
7586
7574
  const current = this.activeModelConfig();
7587
7575
  if (!current)
7588
7576
  return null;
7589
- const alternateVision = this.config.allModels().some(model => model.enabled !== false && model.provider_id === current.provider_id &&
7590
- model.name !== current.name && !!model.vision && !!model.api_key && !!model.provider_url &&
7591
- !['unavailable', 'auth_error', 'invalid_config'].includes(String(model.evaluation?.status || model.validation?.status || '').toLowerCase()));
7592
- if (alternateVision)
7593
- return null;
7594
7577
  const latest = [...this.history].reverse().find(item => item?.role === 'user');
7595
7578
  const parts = latest?.content && Array.isArray(latest.content) ? latest.content : [];
7596
7579
  const images = parts.map(part => {
@@ -7606,11 +7589,16 @@ class Agent {
7606
7589
  if (result.ok && result.text.trim())
7607
7590
  ocr.push({ index: index + 1, text: result.text.slice(0, 50_000), confidence: result.confidence });
7608
7591
  }
7609
- catch { }
7592
+ catch (error) {
7593
+ if (signal?.aborted)
7594
+ throw error;
7595
+ }
7610
7596
  }
7597
+ if (signal?.aborted)
7598
+ throw signal.reason || new Error('Aborted');
7611
7599
  if (!ocr.length)
7612
7600
  return JSON.stringify({ ok: false, fallback: 'mini_ocr_llm', error: 'Local OCR returned no readable text; no visual content was fabricated.' });
7613
- const task = typeof latest?.content === 'string' ? latest.content : '';
7601
+ const task = typeof latest?.content === 'string' ? latest.content : parts.filter(part => part.type === 'text').map(part => String(part.text || '')).join('\n');
7614
7602
  const evidence = ocr.map(item => `Image ${item.index} (OCR confidence ${item.confidence.toFixed(1)}):\n${item.text}`).join('\n\n');
7615
7603
  const prompt = `The provider rejected image input. Answer the user's task using only this approximate OCR evidence. Correct obvious character, spacing, and line-break errors only when supported by context. Preserve [uncertain] markers for ambiguity and never invent missing visual content.\nUser task:\n${task.slice(0, 12_000)}\nOCR evidence:\n${evidence}`;
7616
7604
  let corrected = '';
@@ -7619,12 +7607,15 @@ class Agent {
7619
7607
  if (provider)
7620
7608
  corrected = String(await this.chatWithConversationUsage(provider, this.activeModelName(), [{ role: 'user', content: prompt }], 'You are a text-only OCR correction assistant. Be conservative and explicit about uncertainty.', 0.05, 3000, signal) || '').trim();
7621
7609
  }
7622
- catch { }
7610
+ catch (error) {
7611
+ if (signal?.aborted)
7612
+ throw error;
7613
+ }
7623
7614
  return JSON.stringify({
7624
7615
  ok: !!(corrected || ocr.length),
7625
7616
  fallback: 'mini_ocr_llm',
7626
7617
  approximate: true,
7627
- warning: '视觉输入被拒绝;以下内容来自本地 OCR,并经文本模型保守校正,可能不完整。',
7618
+ warning: corrected ? '视觉输入被拒绝;以下内容来自本地 OCR,并经文本模型保守校正,可能不完整。' : '视觉输入被拒绝,文本校正请求也未成功;以下为未经校正的本地 OCR 结果,可能不完整。',
7628
7619
  raw_ocr: ocr,
7629
7620
  corrected: corrected || ocr.map(item => item.text).join('\n\n'),
7630
7621
  uncertainty: corrected ? 'preserved' : 'raw_ocr_only',
@@ -7657,7 +7648,9 @@ class Agent {
7657
7648
  const adapters = this.config.contextFlag('provider_adapters_v2');
7658
7649
  const thinkingMaps = this.modelThinkingTierMaps(m);
7659
7650
  const proxy = this.providerProxyConfig();
7660
- const create = () => new provider_1.LLMProvider(m.provider, m.provider_url, m.api_key, m.provider_protocol, apiMode, adapters, undefined, thinkingMaps, proxy);
7651
+ const create = () => (0, modelResponseHealth_1.observeModelResponses)(new provider_1.LLMProvider(m.provider, m.provider_url, m.api_key, m.provider_protocol, apiMode, adapters, undefined, thinkingMaps, proxy), updates => (0, modelResponseHealth_1.recordModelResponseHealth)(this.config.rootPath, {
7652
+ providerId: m.provider_id, modelId: m.name, endpoint: m.provider_url, protocol: m.provider_protocol, credential: m.api_key,
7653
+ }, updates));
7661
7654
  if (!cache)
7662
7655
  return create();
7663
7656
  // Match the provider's configured-proxy/environment precedence. A changed
@@ -7974,6 +7967,7 @@ class Agent {
7974
7967
  return [{ type: 'text', text: '[Workspace required] Select or create a workspace before starting a conversation.' }];
7975
7968
  }
7976
7969
  if (this.processDepth === 0) {
7970
+ this.failedDeploymentsThisRun.clear();
7977
7971
  this.processingConversationId = this.activeConversationId || 'default';
7978
7972
  this.activeProcessAbortController = new AbortController();
7979
7973
  this.subagents.resumeScheduling();
@@ -7992,27 +7986,6 @@ class Agent {
7992
7986
  this.fileDiffs = [];
7993
7987
  this.pendingOptions = [];
7994
7988
  try {
7995
- if (this.model === 'auto') {
7996
- // Auto routing re-resolves each turn; drop a stale blocked deployment
7997
- // so the router can pick an unblocked candidate instead of failing here.
7998
- if (this.resolvedDeployment && this.isBalanceBlockedDeployment(this.resolvedDeployment)) {
7999
- this.resolvedDeployment = null;
8000
- this.lastRouteDecision = null;
8001
- this.pendingAutoAttempts = [];
8002
- }
8003
- }
8004
- else {
8005
- const blockedMs = this.balanceBlockedMs();
8006
- if (blockedMs > 0) {
8007
- const waitSeconds = Math.max(1, Math.ceil(blockedMs / 1000));
8008
- const deployment = this.activeDeployment();
8009
- const label = deployment ? `${deployment.providerId}/${deployment.modelId}` : this.model;
8010
- const message = `Provider balance exhausted (HTTP 402) for ${label}. Requests on this deployment are paused for ${waitSeconds}s; switch provider or model to continue immediately.`;
8011
- this.status = 'error';
8012
- this.emitWorkEvent({ type: 'error', content: message });
8013
- throw new Error(message);
8014
- }
8015
- }
8016
7989
  let text = typeof input === 'string' ? input : String(input.text || '');
8017
7990
  const inputEnvelope = typeof input === 'string' ? null : input;
8018
7991
  let hiddenUserInput = inputEnvelope?.hiddenUserInput === true;
@@ -8109,16 +8082,6 @@ class Agent {
8109
8082
  await this.evaluateAndSwitch(`${text}\n[image attachment]`, inputEnvelope?.routePolicy);
8110
8083
  autoRouteEvaluated = true;
8111
8084
  }
8112
- const selectedModel = this.activeModelConfig();
8113
- if (images.length && !selectedModel?.vision) {
8114
- // Give the normal route planner first chance to select another
8115
- // same-provider vision deployment. If none is available, the kernel
8116
- // preflight invokes the final mini-OCR + text-only correction path.
8117
- const hasSameProviderVision = selectedModel && this.config.allModels().some(model => model.enabled !== false && model.provider_id === selectedModel.provider_id &&
8118
- model.name !== selectedModel.name && !!model.vision);
8119
- if (hasSameProviderVision)
8120
- this.switchToFallbackModel('vision input not supported by the selected model');
8121
- }
8122
8085
  const now = this.nowLabel();
8123
8086
  const visibleUserInput = inputEnvelope?.visibleUserInput === undefined
8124
8087
  ? text
@@ -9217,9 +9180,7 @@ class Agent {
9217
9180
  }
9218
9181
  subagentToolDefinitions(defs) {
9219
9182
  const modelCapabilities = this.activeModelConfig();
9220
- const visionFiltered = modelCapabilities?.vision
9221
- ? defs
9222
- : defs.filter((tool) => tool.function?.name !== 'image_inspect');
9183
+ const visionFiltered = defs;
9223
9184
  const withImageGeneration = modelCapabilities?.image_output
9224
9185
  ? [...visionFiltered, {
9225
9186
  type: 'function',
@@ -9263,8 +9224,6 @@ class Agent {
9263
9224
  }
9264
9225
  }
9265
9226
  async handleImageInspect(args) {
9266
- if (!this.activeModelConfig()?.vision)
9267
- return '[Image inspect unavailable] The selected model has not passed vision validation.';
9268
9227
  let input = {};
9269
9228
  try {
9270
9229
  input = JSON.parse(args);
@@ -10562,7 +10521,9 @@ function routeProviderFingerprint(provider) {
10562
10521
  });
10563
10522
  }
10564
10523
  function modelConfigurationFingerprint(model) {
10565
- const { validation, evaluation, _previous_name, previous_name, ...configuration } = model;
10524
+ const { validation, evaluation, response_health, enabled, _previous_name, previous_name, ...configuration } = model;
10525
+ void response_health;
10526
+ void enabled;
10566
10527
  void validation;
10567
10528
  void evaluation;
10568
10529
  void _previous_name;
@@ -10604,12 +10565,7 @@ function resetEditedModelValidationEvidence(incomingProviders, existingProviders
10604
10565
  }
10605
10566
  }
10606
10567
  function modelConfigIsUnavailable(model) {
10607
- const validationStatus = effectiveModelValidationStatus(model);
10608
- if (model.validation?.level !== 'discovered'
10609
- && (validationStatus === 'unavailable' || validationStatus === 'auth_error' || validationStatus === 'invalid_config'))
10610
- return true;
10611
- const evaluationStatus = validationStatus === 'degraded' ? 'degraded' : String(model.evaluation?.status || '').toLowerCase();
10612
- return evaluationStatus === 'unavailable' || evaluationStatus.startsWith('error');
10568
+ return model.enabled === false;
10613
10569
  }
10614
10570
  function parseDeploymentSelectionValue(value) {
10615
10571
  const marker = String(value || '').trim();
@@ -480,15 +480,18 @@ async function runAgentKernel(agent) {
480
480
  try {
481
481
  const linkedPlanRevisionBeforeRun = agent.getLinkedPlan().revision;
482
482
  const modelBeforeKernelRun = agent.model;
483
- const preflightVisualFallback = !agent.activeModelConfig()?.vision
484
- ? await agent.finalVisualFallback('vision input not supported by the selected model', processSignal)
485
- : null;
486
- let lastTurn = preflightVisualFallback
487
- ? { text: preflightVisualFallback, stopReason: 'stop', errorMessage: '' }
488
- : await runWithCompressionResume([], false);
489
- if (preflightVisualFallback) {
490
- tokens.push({ type: 'text', text: preflightVisualFallback });
491
- agent.recordWorkStatus('Final visual fallback used: local mini OCR plus conservative text correction.');
483
+ let lastTurn = await runWithCompressionResume([], false);
484
+ if (kernelTurnFailed(agent, lastTurn)) {
485
+ const visualFallback = await agent.finalVisualFallback(lastTurn.errorMessage || lastTurn.text, processSignal);
486
+ if (visualFallback) {
487
+ tokens.push({ type: 'text', text: visualFallback });
488
+ agent.emitWorkEvent({ type: 'final_response', content: visualFallback });
489
+ agent.chatMessages.push({ role: 'assistant', content: visualFallback, mode: agent.modeName(), model: agent.model, timestamp: agent.nowLabel(), runId: agent.currentWorkRunId() || undefined });
490
+ agent.history.push({ role: 'assistant', content: visualFallback, run_id: agent.currentWorkRunId() || undefined });
491
+ agent.saveWorkspaceConversationState();
492
+ agent.recordWorkStatus('Final visual fallback used: local mini OCR plus conservative text correction.');
493
+ lastTurn = { ...lastTurn, text: visualFallback, errorMessage: '', stopReason: 'stop' };
494
+ }
492
495
  }
493
496
  if (modelBeforeKernelRun && modelBeforeKernelRun !== agent.model && !tokens.some(t => t.text?.includes('[Model fallback]'))) {
494
497
  const notice = `[Model fallback] ${modelBeforeKernelRun} unavailable; switched to ${agent.model}.`;
@@ -601,14 +604,6 @@ async function runAgentKernel(agent) {
601
604
  await agent.waitForPlannedRouteRetry();
602
605
  lastTurn = await runWithCompressionResume([], false);
603
606
  }
604
- if (kernelTurnFailed(agent, lastTurn)) {
605
- const visualFallback = await agent.finalVisualFallback(lastTurn.errorMessage || lastTurn.text, processSignal);
606
- if (visualFallback) {
607
- tokens.push({ type: 'text', text: visualFallback });
608
- agent.recordWorkStatus('Final visual fallback used: local mini OCR plus conservative text correction.');
609
- lastTurn = { ...lastTurn, text: visualFallback, errorMessage: '', stopReason: 'stop' };
610
- }
611
- }
612
607
  if (kernelTurnFailed(agent, lastTurn)) {
613
608
  throw new ProviderRunError(normalizePublicProviderError(lastTurn.errorMessage || lastTurn.text, [agent.activeModelConfig()?.api_key]));
614
609
  }
@@ -1167,7 +1162,7 @@ function toKernelModel(agent) {
1167
1162
  provider: m?.provider || 'newmark',
1168
1163
  baseUrl: m?.provider_url || '',
1169
1164
  reasoning: !!m?.thinking,
1170
- input: m?.vision ? ['text', 'image'] : ['text'],
1165
+ input: ['text', 'image'],
1171
1166
  cost: {
1172
1167
  input: Number(m?.cost_per_1k_input || 0) * 1000,
1173
1168
  output: Number(m?.cost_per_1k_output || 0) * 1000,
@@ -1752,8 +1747,6 @@ function visualFallbackImageInput(agent, name, text) {
1752
1747
  if (name !== 'screen_capture' && name !== 'computer_use' && name !== 'browser_use' && name !== 'pdf_read')
1753
1748
  return {};
1754
1749
  const model = agent.activeModelConfig();
1755
- if (!model?.vision)
1756
- return {};
1757
1750
  try {
1758
1751
  const parsed = JSON.parse(text);
1759
1752
  const nested = name === 'pdf_read' && parsed.result && typeof parsed.result === 'object'
@@ -1917,8 +1910,7 @@ async function executeNewmarkTool(agent, name, args, inputSchema, signal, onSett
1917
1910
  actorId: agent.runtimeActorId,
1918
1911
  workspaceId: (0, terminalTakeover_1.terminalTakeoverWorkspaceId)(wsDir),
1919
1912
  backend: process.env.NEWMARK_WSL_DISTRO ? 'wsl' : (process.platform === 'win32' ? 'windows' : process.platform),
1920
- allowEphemeralVisionImage: (name === 'screen_capture' || name === 'computer_use' || name === 'browser_use' || name === 'pdf_read' || name === 'ocr_read')
1921
- && !!agent.activeModelConfig()?.vision,
1913
+ allowEphemeralVisionImage: (name === 'screen_capture' || name === 'computer_use' || name === 'browser_use' || name === 'pdf_read' || name === 'ocr_read'),
1922
1914
  signal,
1923
1915
  });
1924
1916
  if (signal?.aborted)
@@ -34,19 +34,6 @@ function inSubset(deployment, subset) {
34
34
  function inScope(deployment, scope) {
35
35
  return scope.kind === 'global' || deployment.providerId === scope.providerId;
36
36
  }
37
- function validationEligible(candidate, now) {
38
- const reasons = [];
39
- if (candidate.validation.level !== 'standard' && candidate.validation.level !== 'extended') {
40
- reasons.push(`validation_level:${candidate.validation.level}`);
41
- }
42
- if (candidate.validation.status !== 'verified' && candidate.validation.status !== 'degraded') {
43
- reasons.push(`validation_status:${candidate.validation.status}`);
44
- }
45
- const checkedAt = Date.parse(candidate.validation.checkedAt);
46
- if (!Number.isFinite(checkedAt) || now - checkedAt > VALIDATION_TTL_MS)
47
- reasons.push('validation_expired');
48
- return reasons;
49
- }
50
37
  function normalizeAutoPreference(value) {
51
38
  switch (String(value || '').toLowerCase()) {
52
39
  case 'performance':
@@ -191,7 +178,6 @@ class AutoRouter {
191
178
  this.claimEndpointAttempt(fixed.deployment);
192
179
  return decision;
193
180
  }
194
- const requiredCapabilities = new Set([...policy.requiredCapabilities, ...request.requiredCapabilities].map(item => String(item).toLowerCase()));
195
181
  const eligible = [];
196
182
  for (const candidate of candidates) {
197
183
  const reasons = [];
@@ -205,13 +191,8 @@ class AutoRouter {
205
191
  reasons.push('fallback_only');
206
192
  if (candidate.preview && !policy.allowPreview)
207
193
  reasons.push('preview_disallowed');
208
- reasons.push(...validationEligible(candidate, now));
209
194
  if (request.estimatedInputTokens + request.expectedOutputTokens > Math.max(0, candidate.maxContextTokens || 0))
210
195
  reasons.push('context_too_small');
211
- const capabilities = new Set(candidate.capabilities.map(item => String(item).toLowerCase()));
212
- for (const capability of requiredCapabilities)
213
- if (!capabilities.has(capability))
214
- reasons.push(`missing_capability:${capability}`);
215
196
  if (policy.privacy !== 'default' && !candidate.privacy.includes(policy.privacy))
216
197
  reasons.push(`privacy:${policy.privacy}`);
217
198
  if (policy.dataRegion) {
@@ -230,8 +211,6 @@ class AutoRouter {
230
211
  if (policy.maxExpectedCostUsd !== undefined && (expectedCost === undefined || expectedCost > policy.maxExpectedCostUsd)) {
231
212
  reasons.push(expectedCost === undefined ? 'unknown_cost' : 'budget_exceeded');
232
213
  }
233
- if (this.circuitState(candidate.deployment, now, false) === 'open')
234
- reasons.push('circuit_open');
235
214
  if (reasons.length)
236
215
  decision.excludedCandidates.push({ deployment: { ...candidate.deployment }, reasons: [...new Set(reasons)] });
237
216
  else
@@ -325,9 +304,7 @@ class AutoRouter {
325
304
  && inSubset(candidate.deployment, subset)
326
305
  && !sameDeployment(candidate.deployment, current)
327
306
  && !attemptedDeployments.some(attempted => sameDeployment(candidate.deployment, attempted))
328
- && validationEligible(candidate, now).length === 0
329
- && this.passedInitialHardFilters(decision, candidate)
330
- && this.circuitState(candidate.deployment, now, false) !== 'open');
307
+ && this.passedInitialHardFilters(decision, candidate));
331
308
  const equivalent = currentGroup
332
309
  ? eligible.find(candidate => candidate.deployment.logicalModelGroupId === currentGroup && !candidate.fallbackOnly)
333
310
  : undefined;
@@ -1,3 +1,4 @@
1
+ import { ModelResponseHealth } from './modelResponseHealth';
1
2
  export interface JsonValue {
2
3
  [key: string]: unknown;
3
4
  }
@@ -20,6 +21,7 @@ export interface ProviderConfig {
20
21
  models: ModelConfig[];
21
22
  }
22
23
  export interface ModelConfig {
24
+ response_health?: ModelResponseHealth;
23
25
  name: string;
24
26
  display: string;
25
27
  description: string;
@@ -44,6 +44,7 @@ exports.mergeProviderSecrets = mergeProviderSecrets;
44
44
  exports.defaultModelConfig = defaultModelConfig;
45
45
  exports.stableProviderId = stableProviderId;
46
46
  exports.defaultConfig = defaultConfig;
47
+ const modelResponseHealth_1 = require("./modelResponseHealth");
47
48
  const fs = __importStar(require("fs"));
48
49
  const path = __importStar(require("path"));
49
50
  const crypto_1 = require("crypto");
@@ -191,7 +192,12 @@ class ConfigManager {
191
192
  this.workspaceOverrides.clear();
192
193
  }
193
194
  providers() {
194
- return this.normalizeProviders((this.getGlobal('models', 'providers')) || []);
195
+ return this.normalizeProviders((this.getGlobal('models', 'providers')) || []).map(provider => ({
196
+ ...provider,
197
+ models: provider.models.map(model => ({ ...model, response_health: (0, modelResponseHealth_1.readModelResponseHealth)(this.rootPath, {
198
+ providerId: provider.id, modelId: model.name, endpoint: provider.base_url, protocol: provider.protocol, credential: provider.api_key,
199
+ }) })),
200
+ }));
195
201
  }
196
202
  allModels() {
197
203
  const models = [];
@@ -384,7 +384,7 @@ class ConversationKernel {
384
384
  queueAction(target, action, input = {}) {
385
385
  const normalized = this.normalizeTarget(target);
386
386
  const runtime = this.findRuntime(normalized) || this.runtime(normalized, {
387
- mode: this.host.mode, model: this.host.model, intelligence: this.host.intelligence,
387
+ mode: this.host.mode, model: this.host.modelSelectionValue(), intelligence: this.host.intelligence,
388
388
  inputMode: this.host.inputMode, engine: this.host.engine,
389
389
  });
390
390
  runtime.runId ||= (0, crypto_1.randomUUID)();
@@ -474,7 +474,7 @@ class ConversationKernel {
474
474
  let runtime = this.findRuntime(normalized);
475
475
  const runner = runtime?.runner || this.createRunner(normalized);
476
476
  if (!runtime && runner.conversationContinuations().length) {
477
- runtime = this.runtime(normalized, { mode: runner.mode, model: runner.model, intelligence: runner.intelligence, inputMode: runner.inputMode, engine: runner.engine }, runner);
477
+ runtime = this.runtime(normalized, { mode: runner.mode, model: runner.modelSelectionValue(), intelligence: runner.intelligence, inputMode: runner.inputMode, engine: runner.engine }, runner);
478
478
  // Recovered user input remains visible and manageable until its owner
479
479
  // explicitly restores the persisted queue policy or resumes the queue.
480
480
  runtime.queuePaused = true;
@@ -491,7 +491,7 @@ class ConversationKernel {
491
491
  workEvents: this.events(normalized),
492
492
  runtime: this.runtimeState(normalized),
493
493
  mode: runner.mode,
494
- model: runner.model,
494
+ model: runner.modelSelectionValue(),
495
495
  intelligence: runner.intelligence,
496
496
  status: runner.status,
497
497
  goal: conversationSnapshot.goal,
@@ -783,7 +783,9 @@ class ConversationKernel {
783
783
  const runner = runtime?.runner || this.createRunner(normalized);
784
784
  if (!runtime || !runtime.activePromise) {
785
785
  // No Build block is running: the selection applies immediately.
786
- runner.setModel(model);
786
+ runner.setModel(model, true);
787
+ if (runtime)
788
+ runtime.options.model = runner.modelSelectionValue();
787
789
  }
788
790
  else {
789
791
  // A Build block is running. The in-flight block keeps its current model
@@ -793,7 +795,7 @@ class ConversationKernel {
793
795
  runtime.options.model = model;
794
796
  }
795
797
  runner.saveWorkspaceConversationState(true);
796
- return runner.model;
798
+ return runner.modelSelectionValue();
797
799
  }
798
800
  async toggleGoalPause(target) {
799
801
  const normalized = this.normalizeTarget(target);
@@ -802,7 +804,7 @@ class ConversationKernel {
802
804
  const runner = this.createRunner(normalized);
803
805
  runtime = this.runtime(normalized, {
804
806
  mode: runner.mode,
805
- model: runner.model,
807
+ model: runner.modelSelectionValue(),
806
808
  intelligence: runner.intelligence,
807
809
  inputMode: runner.inputMode,
808
810
  engine: runner.engine,
@@ -1577,7 +1579,12 @@ class ConversationKernel {
1577
1579
  // Host-global state can belong to a different foreground conversation.
1578
1580
  const restoredMode = agent.getConversationSnapshot(agent.activeConversationId).mode;
1579
1581
  agent.setMode(restoredMode || options.mode);
1580
- agent.setModel(options.model);
1582
+ // Older clients may echo a bare display name from a snapshot. Preserve
1583
+ // an already restored exact deployment in that case; never guess between
1584
+ // providers for an unbound or different ambiguous model name.
1585
+ if (options.model !== agent.model || !agent.activeDeployment()) {
1586
+ agent.setModel(options.model);
1587
+ }
1581
1588
  agent.setIntelligence(options.intelligence);
1582
1589
  agent.inputMode = options.inputMode;
1583
1590
  agent.engine = options.engine;
@@ -1603,7 +1610,7 @@ class ConversationKernel {
1603
1610
  newContent: d.newContent,
1604
1611
  })),
1605
1612
  mode: runtime.runner.mode,
1606
- model: runtime.runner.model,
1613
+ model: runtime.runner.modelSelectionValue(),
1607
1614
  status: runtime.runner.status,
1608
1615
  goal: runtime.runner.goal ? { objective: runtime.runner.goal.objective, paused: runtime.runner.goal.paused } : null,
1609
1616
  options: runtime.runner.pendingOptions,
@@ -0,0 +1,17 @@
1
+ import type { LLMProvider } from '../llm/provider';
2
+ export type ResponseFacet = 'text' | 'vision' | 'tools';
3
+ export type ModelResponseHealth = Partial<Record<ResponseFacet, {
4
+ ok: boolean;
5
+ at: string;
6
+ }>>;
7
+ export interface ResponseHealthIdentity {
8
+ providerId: string;
9
+ modelId: string;
10
+ endpoint: string;
11
+ protocol: string;
12
+ credential: string;
13
+ }
14
+ export declare function readModelResponseHealth(root: string, identity: ResponseHealthIdentity): ModelResponseHealth;
15
+ export declare function recordModelResponseHealth(root: string, identity: ResponseHealthIdentity, updates: Partial<Record<ResponseFacet, boolean>>): void;
16
+ export declare function observeModelResponses(provider: LLMProvider, record: (updates: Partial<Record<ResponseFacet, boolean>>) => void): LLMProvider;
17
+ //# sourceMappingURL=modelResponseHealth.d.ts.map