agent-nuvira 2.7.3 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/agents/writer-tool-calling.d.ts +14 -0
- package/dist/agents/agents/writer-tool-calling.d.ts.map +1 -1
- package/dist/agents/agents/writer-tool-calling.js +20 -0
- package/dist/agents/agents/writer-tool-calling.js.map +1 -1
- package/dist/agents/agents/writer.d.ts.map +1 -1
- package/dist/agents/agents/writer.js +15 -4
- package/dist/agents/agents/writer.js.map +1 -1
- package/dist/agents/edit-module.d.ts +7 -0
- package/dist/agents/edit-module.d.ts.map +1 -1
- package/dist/agents/edit-module.js +21 -6
- package/dist/agents/edit-module.js.map +1 -1
- package/dist/agents/orchestrator.d.ts +40 -0
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +116 -27
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/agents/tool-calling-agent.d.ts +19 -0
- package/dist/agents/tool-calling-agent.d.ts.map +1 -1
- package/dist/agents/tool-calling-agent.js +42 -2
- package/dist/agents/tool-calling-agent.js.map +1 -1
- package/dist/cli/chat.d.ts +55 -1
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +278 -72
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/cli-program.d.ts +6 -0
- package/dist/cli/cli-program.d.ts.map +1 -0
- package/dist/cli/cli-program.js +242 -0
- package/dist/cli/cli-program.js.map +1 -0
- package/dist/cli/edit.d.ts.map +1 -1
- package/dist/cli/edit.js +18 -3
- package/dist/cli/edit.js.map +1 -1
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +29 -5
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/failover-runner.d.ts +13 -0
- package/dist/cli/failover-runner.d.ts.map +1 -1
- package/dist/cli/failover-runner.js +11 -0
- package/dist/cli/failover-runner.js.map +1 -1
- package/dist/cli/gateway.d.ts +16 -0
- package/dist/cli/gateway.d.ts.map +1 -1
- package/dist/cli/gateway.js +108 -2
- package/dist/cli/gateway.js.map +1 -1
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +245 -12
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/plan.d.ts.map +1 -1
- package/dist/cli/plan.js +20 -6
- package/dist/cli/plan.js.map +1 -1
- package/dist/cli/process-control.d.ts +10 -0
- package/dist/cli/process-control.d.ts.map +1 -1
- package/dist/cli/process-control.js +73 -7
- package/dist/cli/process-control.js.map +1 -1
- package/dist/cli/router.d.ts +16 -5
- package/dist/cli/router.d.ts.map +1 -1
- package/dist/cli/router.js +31 -206
- package/dist/cli/router.js.map +1 -1
- package/dist/config/paths.d.ts +23 -0
- package/dist/config/paths.d.ts.map +1 -1
- package/dist/config/paths.js +30 -0
- package/dist/config/paths.js.map +1 -1
- package/dist/config/provider-env.d.ts +37 -0
- package/dist/config/provider-env.d.ts.map +1 -0
- package/dist/config/provider-env.js +86 -0
- package/dist/config/provider-env.js.map +1 -0
- package/dist/context/history.d.ts.map +1 -1
- package/dist/context/history.js +24 -8
- package/dist/context/history.js.map +1 -1
- package/dist/context/session-recall.d.ts +30 -0
- package/dist/context/session-recall.d.ts.map +1 -1
- package/dist/context/session-recall.js +61 -0
- package/dist/context/session-recall.js.map +1 -1
- package/dist/file.js +0 -1
- package/dist/file.js.map +1 -1
- package/dist/gateway/adapters.d.ts +10 -0
- package/dist/gateway/adapters.d.ts.map +1 -1
- package/dist/gateway/adapters.js +4 -1
- package/dist/gateway/adapters.js.map +1 -1
- package/dist/gateway/dedup.d.ts +105 -0
- package/dist/gateway/dedup.d.ts.map +1 -0
- package/dist/gateway/dedup.js +160 -0
- package/dist/gateway/dedup.js.map +1 -0
- package/dist/gateway/heartbeat.d.ts +99 -0
- package/dist/gateway/heartbeat.d.ts.map +1 -0
- package/dist/gateway/heartbeat.js +122 -0
- package/dist/gateway/heartbeat.js.map +1 -0
- package/dist/gateway/inbox.d.ts +14 -1
- package/dist/gateway/inbox.d.ts.map +1 -1
- package/dist/gateway/inbox.js.map +1 -1
- package/dist/gateway/platform-config.d.ts +5 -0
- package/dist/gateway/platform-config.d.ts.map +1 -1
- package/dist/gateway/platform-config.js +8 -13
- package/dist/gateway/platform-config.js.map +1 -1
- package/dist/gateway/registry.d.ts +47 -0
- package/dist/gateway/registry.d.ts.map +1 -1
- package/dist/gateway/registry.js +227 -21
- package/dist/gateway/registry.js.map +1 -1
- package/dist/gateway/whatsapp/baileys-bridge.d.ts +11 -1
- package/dist/gateway/whatsapp/baileys-bridge.d.ts.map +1 -1
- package/dist/gateway/whatsapp/baileys-bridge.js +64 -1
- package/dist/gateway/whatsapp/baileys-bridge.js.map +1 -1
- package/dist/gateway/whatsapp/bridge.d.ts +4 -2
- package/dist/gateway/whatsapp/bridge.d.ts.map +1 -1
- package/dist/gateway/whatsapp/bridge.js.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/inference/factory.d.ts +16 -0
- package/dist/inference/factory.d.ts.map +1 -1
- package/dist/inference/factory.js +42 -0
- package/dist/inference/factory.js.map +1 -1
- package/dist/inference/gemini-adapter.d.ts.map +1 -1
- package/dist/inference/gemini-adapter.js +3 -0
- package/dist/inference/gemini-adapter.js.map +1 -1
- package/dist/inference/interface.d.ts +12 -0
- package/dist/inference/interface.d.ts.map +1 -1
- package/dist/inference/local-adapter.d.ts +2 -0
- package/dist/inference/local-adapter.d.ts.map +1 -1
- package/dist/inference/local-adapter.js +36 -5
- package/dist/inference/local-adapter.js.map +1 -1
- package/dist/inference/native-tools.d.ts +8 -0
- package/dist/inference/native-tools.d.ts.map +1 -1
- package/dist/inference/native-tools.js +14 -3
- package/dist/inference/native-tools.js.map +1 -1
- package/dist/inference/tool-call-utils.d.ts +92 -7
- package/dist/inference/tool-call-utils.d.ts.map +1 -1
- package/dist/inference/tool-call-utils.js +203 -13
- package/dist/inference/tool-call-utils.js.map +1 -1
- package/dist/learning/auto-router.d.ts.map +1 -1
- package/dist/learning/auto-router.js +90 -5
- package/dist/learning/auto-router.js.map +1 -1
- package/dist/learning/context-budget.d.ts +133 -0
- package/dist/learning/context-budget.d.ts.map +1 -0
- package/dist/learning/context-budget.js +196 -0
- package/dist/learning/context-budget.js.map +1 -0
- package/dist/learning/cost-tracker.d.ts.map +1 -1
- package/dist/learning/cost-tracker.js +20 -8
- package/dist/learning/cost-tracker.js.map +1 -1
- package/dist/learning/eval-framework.d.ts.map +1 -1
- package/dist/learning/eval-framework.js +6 -2
- package/dist/learning/eval-framework.js.map +1 -1
- package/dist/learning/failure-bookkeeping.d.ts +19 -0
- package/dist/learning/failure-bookkeeping.d.ts.map +1 -1
- package/dist/learning/failure-bookkeeping.js +67 -2
- package/dist/learning/failure-bookkeeping.js.map +1 -1
- package/dist/learning/feedback.d.ts.map +1 -1
- package/dist/learning/feedback.js +19 -8
- package/dist/learning/feedback.js.map +1 -1
- package/dist/learning/model-harness.d.ts +76 -0
- package/dist/learning/model-harness.d.ts.map +1 -0
- package/dist/learning/model-harness.js +119 -0
- package/dist/learning/model-harness.js.map +1 -0
- package/dist/learning/model-selection.d.ts +15 -0
- package/dist/learning/model-selection.d.ts.map +1 -1
- package/dist/learning/model-selection.js +27 -1
- package/dist/learning/model-selection.js.map +1 -1
- package/dist/learning/provider-fallback.d.ts +12 -0
- package/dist/learning/provider-fallback.d.ts.map +1 -1
- package/dist/learning/provider-fallback.js +74 -0
- package/dist/learning/provider-fallback.js.map +1 -1
- package/dist/learning/provider-revival.d.ts +106 -0
- package/dist/learning/provider-revival.d.ts.map +1 -0
- package/dist/learning/provider-revival.js +154 -0
- package/dist/learning/provider-revival.js.map +1 -0
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +48 -3
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/nlu/conversation-gate.d.ts +37 -0
- package/dist/nlu/conversation-gate.d.ts.map +1 -1
- package/dist/nlu/conversation-gate.js +49 -0
- package/dist/nlu/conversation-gate.js.map +1 -1
- package/dist/nlu/schema.d.ts +2 -2
- package/dist/observability/dag-bridge.d.ts +69 -0
- package/dist/observability/dag-bridge.d.ts.map +1 -0
- package/dist/observability/dag-bridge.js +56 -0
- package/dist/observability/dag-bridge.js.map +1 -0
- package/dist/observability/event-bus.d.ts +5 -10
- package/dist/observability/event-bus.d.ts.map +1 -1
- package/dist/observability/event-bus.js +10 -42
- package/dist/observability/event-bus.js.map +1 -1
- package/dist/skills/execution-audit.d.ts.map +1 -1
- package/dist/skills/execution-audit.js +17 -5
- package/dist/skills/execution-audit.js.map +1 -1
- package/dist/skills/sandbox-executor.d.ts +12 -1
- package/dist/skills/sandbox-executor.d.ts.map +1 -1
- package/dist/skills/sandbox-executor.js +28 -6
- package/dist/skills/sandbox-executor.js.map +1 -1
- package/dist/skills/secret-capture.d.ts +45 -1
- package/dist/skills/secret-capture.d.ts.map +1 -1
- package/dist/skills/secret-capture.js +82 -19
- package/dist/skills/secret-capture.js.map +1 -1
- package/dist/skills/skill-env-inventory.d.ts +79 -0
- package/dist/skills/skill-env-inventory.d.ts.map +1 -0
- package/dist/skills/skill-env-inventory.js +167 -0
- package/dist/skills/skill-env-inventory.js.map +1 -0
- package/dist/skills/skill-executor.d.ts +2 -0
- package/dist/skills/skill-executor.d.ts.map +1 -1
- package/dist/skills/skill-executor.js +19 -24
- package/dist/skills/skill-executor.js.map +1 -1
- package/dist/tools/ask-user.d.ts +14 -1
- package/dist/tools/ask-user.d.ts.map +1 -1
- package/dist/tools/ask-user.js +25 -1
- package/dist/tools/ask-user.js.map +1 -1
- package/dist/tools/followup-utils.d.ts +64 -0
- package/dist/tools/followup-utils.d.ts.map +1 -0
- package/dist/tools/followup-utils.js +158 -0
- package/dist/tools/followup-utils.js.map +1 -0
- package/dist/tools/loop-skill-hint.d.ts.map +1 -1
- package/dist/tools/loop-skill-hint.js +22 -1
- package/dist/tools/loop-skill-hint.js.map +1 -1
- package/dist/tools/memory-tools.d.ts +35 -0
- package/dist/tools/memory-tools.d.ts.map +1 -1
- package/dist/tools/memory-tools.js +76 -0
- package/dist/tools/memory-tools.js.map +1 -1
- package/dist/tools/pipeline-tool.d.ts.map +1 -1
- package/dist/tools/pipeline-tool.js +8 -3
- package/dist/tools/pipeline-tool.js.map +1 -1
- package/dist/tools/registry.d.ts +25 -12
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +31 -6
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/skill-tool.d.ts.map +1 -1
- package/dist/tools/skill-tool.js +52 -5
- package/dist/tools/skill-tool.js.map +1 -1
- package/dist/tools/tool-loop.d.ts +63 -1
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +305 -66
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/tools/toolsets.d.ts +19 -3
- package/dist/tools/toolsets.d.ts.map +1 -1
- package/dist/tools/toolsets.js +21 -5
- package/dist/tools/toolsets.js.map +1 -1
- package/dist/utils/env.d.ts.map +1 -1
- package/dist/utils/env.js +9 -13
- package/dist/utils/env.js.map +1 -1
- package/dist/web-dashboard/chat-console.d.ts +13 -1
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +16 -2
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/hub-data.d.ts +2 -0
- package/dist/web-dashboard/hub-data.d.ts.map +1 -1
- package/dist/web-dashboard/hub-data.js +4 -0
- package/dist/web-dashboard/hub-data.js.map +1 -1
- package/dist/web-dashboard/server.d.ts +6 -0
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +192 -3
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +31 -1
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/package.json +11 -1
- package/src/web-dashboard/public/assets/index-56yPnN_m.js +207 -0
- package/src/web-dashboard/public/assets/index-56yPnN_m.js.map +1 -0
- package/src/web-dashboard/public/assets/{index-C507EUWf.css → index-Cyd6tIew.css} +1 -1
- package/src/web-dashboard/public/index.html +2 -2
- package/dist/example.d.ts +0 -1
- package/dist/example.d.ts.map +0 -1
- package/dist/example.js +0 -3
- package/dist/example.js.map +0 -1
- package/dist/fresh.d.ts +0 -2
- package/dist/fresh.d.ts.map +0 -1
- package/dist/fresh.js +0 -2
- package/dist/fresh.js.map +0 -1
- package/dist/mcp/mcp-dashboard-oauth.d.ts +0 -94
- package/dist/mcp/mcp-dashboard-oauth.d.ts.map +0 -1
- package/dist/mcp/mcp-dashboard-oauth.js +0 -140
- package/dist/mcp/mcp-dashboard-oauth.js.map +0 -1
- package/dist/memory/background-sync.d.ts +0 -107
- package/dist/memory/background-sync.d.ts.map +0 -1
- package/dist/memory/background-sync.js +0 -242
- package/dist/memory/background-sync.js.map +0 -1
- package/dist/memory/cross-session.d.ts +0 -68
- package/dist/memory/cross-session.d.ts.map +0 -1
- package/dist/memory/cross-session.js +0 -153
- package/dist/memory/cross-session.js.map +0 -1
- package/dist/memory/drift-detector.d.ts +0 -76
- package/dist/memory/drift-detector.d.ts.map +0 -1
- package/dist/memory/drift-detector.js +0 -216
- package/dist/memory/drift-detector.js.map +0 -1
- package/dist/memory/enhanced-manager.d.ts +0 -105
- package/dist/memory/enhanced-manager.d.ts.map +0 -1
- package/dist/memory/enhanced-manager.js +0 -280
- package/dist/memory/enhanced-manager.js.map +0 -1
- package/dist/memory/session-extraction.d.ts +0 -93
- package/dist/memory/session-extraction.d.ts.map +0 -1
- package/dist/memory/session-extraction.js +0 -290
- package/dist/memory/session-extraction.js.map +0 -1
- package/dist/memory/sqlite-store.d.ts +0 -208
- package/dist/memory/sqlite-store.d.ts.map +0 -1
- package/dist/memory/sqlite-store.js +0 -556
- package/dist/memory/sqlite-store.js.map +0 -1
- package/dist/skills/daytona-executor.d.ts +0 -78
- package/dist/skills/daytona-executor.d.ts.map +0 -1
- package/dist/skills/daytona-executor.js +0 -256
- package/dist/skills/daytona-executor.js.map +0 -1
- package/dist/skills/modal-executor.d.ts +0 -81
- package/dist/skills/modal-executor.d.ts.map +0 -1
- package/dist/skills/modal-executor.js +0 -285
- package/dist/skills/modal-executor.js.map +0 -1
- package/dist/sync/inventory.d.ts +0 -1
- package/dist/sync/inventory.d.ts.map +0 -1
- package/dist/sync/inventory.js +0 -3
- package/dist/sync/inventory.js.map +0 -1
- package/dist/sync/validate.d.ts +0 -1
- package/dist/sync/validate.d.ts.map +0 -1
- package/dist/sync/validate.js +0 -3
- package/dist/sync/validate.js.map +0 -1
- package/dist/test.d.ts +0 -2
- package/dist/test.d.ts.map +0 -1
- package/dist/test.js +0 -3
- package/dist/test.js.map +0 -1
- package/dist/tools/browser-tool.d.ts +0 -206
- package/dist/tools/browser-tool.d.ts.map +0 -1
- package/dist/tools/browser-tool.js +0 -726
- package/dist/tools/browser-tool.js.map +0 -1
- package/dist/tools/delegate-tool.d.ts +0 -109
- package/dist/tools/delegate-tool.d.ts.map +0 -1
- package/dist/tools/delegate-tool.js +0 -185
- package/dist/tools/delegate-tool.js.map +0 -1
- package/dist/tools/delegation-state.d.ts +0 -87
- package/dist/tools/delegation-state.d.ts.map +0 -1
- package/dist/tools/delegation-state.js +0 -278
- package/dist/tools/delegation-state.js.map +0 -1
- package/dist/tools/desktop-ui.d.ts +0 -148
- package/dist/tools/desktop-ui.d.ts.map +0 -1
- package/dist/tools/desktop-ui.js +0 -486
- package/dist/tools/desktop-ui.js.map +0 -1
- package/dist/tools/image-video-tool.d.ts +0 -88
- package/dist/tools/image-video-tool.d.ts.map +0 -1
- package/dist/tools/image-video-tool.js +0 -276
- package/dist/tools/image-video-tool.js.map +0 -1
- package/dist/tools/kanban-cron-tools.d.ts +0 -92
- package/dist/tools/kanban-cron-tools.d.ts.map +0 -1
- package/dist/tools/kanban-cron-tools.js +0 -197
- package/dist/tools/kanban-cron-tools.js.map +0 -1
- package/dist/tools/messaging-tool.d.ts +0 -127
- package/dist/tools/messaging-tool.d.ts.map +0 -1
- package/dist/tools/messaging-tool.js +0 -314
- package/dist/tools/messaging-tool.js.map +0 -1
- package/dist/tools/modality/generic-caller.d.ts +0 -43
- package/dist/tools/modality/generic-caller.d.ts.map +0 -1
- package/dist/tools/modality/generic-caller.js +0 -254
- package/dist/tools/modality/generic-caller.js.map +0 -1
- package/dist/tools/modality/modality-catalog.d.ts +0 -134
- package/dist/tools/modality/modality-catalog.d.ts.map +0 -1
- package/dist/tools/modality/modality-catalog.js +0 -445
- package/dist/tools/modality/modality-catalog.js.map +0 -1
- package/dist/tools/modality/tool-router.d.ts +0 -81
- package/dist/tools/modality/tool-router.d.ts.map +0 -1
- package/dist/tools/modality/tool-router.js +0 -84
- package/dist/tools/modality/tool-router.js.map +0 -1
- package/dist/tools/security-tools.d.ts +0 -107
- package/dist/tools/security-tools.d.ts.map +0 -1
- package/dist/tools/security-tools.js +0 -264
- package/dist/tools/security-tools.js.map +0 -1
- package/dist/tools/ssh-tool.d.ts +0 -130
- package/dist/tools/ssh-tool.d.ts.map +0 -1
- package/dist/tools/ssh-tool.js +0 -328
- package/dist/tools/ssh-tool.js.map +0 -1
- package/dist/tools/tts-tool.d.ts +0 -91
- package/dist/tools/tts-tool.d.ts.map +0 -1
- package/dist/tools/tts-tool.js +0 -406
- package/dist/tools/tts-tool.js.map +0 -1
- package/dist/utils/shell-detection.d.ts +0 -124
- package/dist/utils/shell-detection.d.ts.map +0 -1
- package/dist/utils/shell-detection.js +0 -305
- package/dist/utils/shell-detection.js.map +0 -1
- package/dist/utils/windows-paths.d.ts +0 -185
- package/dist/utils/windows-paths.d.ts.map +0 -1
- package/dist/utils/windows-paths.js +0 -472
- package/dist/utils/windows-paths.js.map +0 -1
- package/src/web-dashboard/public/assets/index-C4frng1Q.js +0 -207
- package/src/web-dashboard/public/assets/index-C4frng1Q.js.map +0 -1
package/dist/cli/chat.js
CHANGED
|
@@ -16,12 +16,13 @@ import { getMemoryManager } from '../memory/manager.js';
|
|
|
16
16
|
import { logger } from '../utils/logger.js';
|
|
17
17
|
import { printOrchestrationResult } from './execute.js';
|
|
18
18
|
import { applyActiveModel } from './model.js';
|
|
19
|
-
import { getProviderFallback, classifyFallbackError, isRetryableError, recordRegistrySuccess } from '../learning/provider-fallback.js';
|
|
20
|
-
import { recordActionFailure
|
|
19
|
+
import { getProviderFallback, classifyFallbackError, isRetryableError, isTransientForRetry, recordRegistrySuccess } from '../learning/provider-fallback.js';
|
|
20
|
+
import { recordActionFailure } from '../learning/failure-bookkeeping.js';
|
|
21
|
+
import { resolveThreadBudgetChars } from '../learning/context-budget.js';
|
|
21
22
|
import { getAutoRouter, isAutoModel, isAutoProvider } from '../learning/auto-router.js';
|
|
22
23
|
import { estimateTokens } from '../learning/cost-tracker.js';
|
|
23
24
|
import { getModelRegistry } from '../learning/model-registry.js';
|
|
24
|
-
import { refreshModelRegistry
|
|
25
|
+
import { refreshModelRegistry } from '../inference/model-probe.js';
|
|
25
26
|
import { recordRoutingDecision } from '../learning/routing-history.js';
|
|
26
27
|
import { shouldConfirmFailover, promptFailoverChoice } from './failover-prompt.js';
|
|
27
28
|
import { buildAutoResolveOptions } from '../learning/resolve-options.js';
|
|
@@ -31,15 +32,19 @@ import { PlanStore } from '../tools/plan-store.js';
|
|
|
31
32
|
import { withLogCorrelation } from '../enterprise/log.js';
|
|
32
33
|
import { recordMetricTime, getMetrics } from '../enterprise/metrics.js';
|
|
33
34
|
import { resolveDispatch } from '../nlu/actions.js';
|
|
34
|
-
import {
|
|
35
|
+
import { hasCodingAction, resolveAskKind } from '../nlu/conversation-gate.js';
|
|
35
36
|
import { runToolLoop, extractFallbackToolCalls } from '../tools/tool-loop.js';
|
|
36
|
-
import { looksLikeConfusedScaffoldingReply } from '../inference/tool-call-utils.js';
|
|
37
|
+
import { looksLikeConfusedScaffoldingReply, toUserFacingGenerationError, isToolCallingUnsupported, stripToolCallArtifacts, } from '../inference/tool-call-utils.js';
|
|
37
38
|
import { beginTrace, endTrace, recordStep } from '../learning/reasoning-trace.js';
|
|
38
39
|
import { getLoopExposureMode } from '../tools/toolsets.js';
|
|
40
|
+
import { resolveModelHarnessProfile, shouldSkipNativeTools } from '../learning/model-harness.js';
|
|
41
|
+
import { resolveAdapterDefault } from '../learning/model-selection.js';
|
|
39
42
|
import { buildLoopProjectContext } from '../tools/loop-project-context.js';
|
|
43
|
+
import { sweepTransientFailures, collectionRevivalStore } from '../learning/provider-revival.js';
|
|
40
44
|
import { analyzeComplexity } from '../learning/hybrid-router.js';
|
|
41
45
|
import { routingCacheSignature, withRoutingCache } from '../learning/routing-cache.js';
|
|
42
46
|
import { getTool, TOOL_CONTRACT_JSON } from '../tools/registry.js';
|
|
47
|
+
import { buildFollowupContinuationPrompt, isSuggestedFollowup, } from '../tools/followup-utils.js';
|
|
43
48
|
// S2/S3 — the shared tool-call reliability helpers (salvage failed_generation,
|
|
44
49
|
// compact fallback schemas). One copy for every tool-calling surface, not
|
|
45
50
|
// chat-private (execute/plan/… inherit the fix).
|
|
@@ -168,7 +173,11 @@ export function resolvePipelineDispatch(parsed, opts) {
|
|
|
168
173
|
// NLU alone would misread it as chat ("how do I add JWT auth?" → explain
|
|
169
174
|
// → chat, but the user wants the auth added).
|
|
170
175
|
if (opts?.text) {
|
|
171
|
-
|
|
176
|
+
// ONE shared rule for every surface (see resolveAskKind): a genuine
|
|
177
|
+
// question never dispatches; a coding verb in command position always
|
|
178
|
+
// does. Keeping the gateway on this same function is what stops the two
|
|
179
|
+
// from disagreeing about the same ask.
|
|
180
|
+
if (resolveAskKind(opts.text, parsed) === 'chat') {
|
|
172
181
|
return { dispatch: false, needConfirm: false };
|
|
173
182
|
}
|
|
174
183
|
if (hasCodingAction(opts.text)) {
|
|
@@ -217,6 +226,41 @@ export async function runDeveloperMode(goal, configManager, options) {
|
|
|
217
226
|
* (rules act only as the no-model fallback in the
|
|
218
227
|
* caller, never to skip the loop).
|
|
219
228
|
*/
|
|
229
|
+
/**
|
|
230
|
+
* Backoff schedule for a SAME-provider retry on a transient failure. Two extra
|
|
231
|
+
* attempts, deliberately short: a capacity spike at a shared endpoint clears in
|
|
232
|
+
* seconds, and the user is waiting in the foreground. Long/looping retries belong
|
|
233
|
+
* to the background runners, not the interactive turn.
|
|
234
|
+
*/
|
|
235
|
+
export const TRANSIENT_RETRY_DELAYS_MS = [1_000, 3_000];
|
|
236
|
+
/**
|
|
237
|
+
* Run one provider attempt, retrying transient failures against the SAME
|
|
238
|
+
* provider before giving up.
|
|
239
|
+
*
|
|
240
|
+
* Why this exists next to the failover walk rather than inside it: the walk
|
|
241
|
+
* needs a DIFFERENT provider to exist, and it books the failure against the one
|
|
242
|
+
* that just failed. Verified live — a single configured provider plus a Gemini
|
|
243
|
+
* 503 meant no retry at all, the circuit breaker parked the provider for 120s,
|
|
244
|
+
* and the agent degraded to editing with zero gathered context. A transient
|
|
245
|
+
* spike must cost a few seconds, not the whole task.
|
|
246
|
+
*
|
|
247
|
+
* Never retries: non-transient classes (auth, rate-limit, model/harness faults),
|
|
248
|
+
* a cancelled turn, or once the schedule is exhausted.
|
|
249
|
+
*/
|
|
250
|
+
export async function generateWithTransientRetry(attempt, signal, onRetry) {
|
|
251
|
+
for (let i = 0;; i += 1) {
|
|
252
|
+
try {
|
|
253
|
+
return await attempt();
|
|
254
|
+
}
|
|
255
|
+
catch (err) {
|
|
256
|
+
const canRetry = i < TRANSIENT_RETRY_DELAYS_MS.length && isTransientForRetry(err);
|
|
257
|
+
if (!canRetry || signal?.aborted)
|
|
258
|
+
throw err;
|
|
259
|
+
onRetry?.(i + 1, err);
|
|
260
|
+
await new Promise((resolve) => setTimeout(resolve, TRANSIENT_RETRY_DELAYS_MS[i]));
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
}
|
|
220
264
|
function buildToolSystemPrompt(parsed) {
|
|
221
265
|
return [
|
|
222
266
|
"You are Nuvira, Agent-Nuvira's AI agent. You code, create, write, analyze, and automate — anything the user needs. You identify as Nuvira (never 'Buff').",
|
|
@@ -316,6 +360,34 @@ export class ChatCommand extends BaseCommand {
|
|
|
316
360
|
const requestedModel = opts.model && opts.model !== 'default' ? opts.model : undefined;
|
|
317
361
|
const activeOpts = applyActiveModel({ provider: opts.provider, model: requestedModel });
|
|
318
362
|
const mergedOpts = { ...opts, provider: activeOpts.provider, model: activeOpts.model };
|
|
363
|
+
// When the CALLER supplies neither a provider nor a model — the dashboard
|
|
364
|
+
// chat console and the gateway chat engine both call answerOnce with just a
|
|
365
|
+
// message — fall back to the CONFIGURED defaultProvider instead of letting
|
|
366
|
+
// resolveProvider() land on one fixed provider.
|
|
367
|
+
//
|
|
368
|
+
// Why this matters (live, 2026-09-20): the shipped config default is
|
|
369
|
+
// `defaultProvider: "auto"`, but auto mode was only ever enabled by an
|
|
370
|
+
// EXPLICIT 'auto' from the flags or the `nuvira model switch` state. The
|
|
371
|
+
// dashboard passes neither, so every dashboard turn silently ran on ONE
|
|
372
|
+
// concrete provider with NO auto-failover walk (the non-auto path only
|
|
373
|
+
// walks `fallback.providers`, which ships empty). One 400/429/timeout then
|
|
374
|
+
// ended the turn with the canned "the language model was unavailable"
|
|
375
|
+
// line — while the CLI answered the identical prompt, because the CLI
|
|
376
|
+
// resolves auto from the same config. Same engine, two modes: this closes
|
|
377
|
+
// that gap.
|
|
378
|
+
// Only the AUTO default changes behavior: a concrete `defaultProvider` is a
|
|
379
|
+
// deliberate pin and keeps the non-auto path exactly as it is.
|
|
380
|
+
if (!mergedOpts.provider && !mergedOpts.model) {
|
|
381
|
+
try {
|
|
382
|
+
const cfg = this.configManager.getAll();
|
|
383
|
+
if (isAutoProvider(cfg.defaultProvider)) {
|
|
384
|
+
mergedOpts.provider = cfg.defaultProvider;
|
|
385
|
+
}
|
|
386
|
+
}
|
|
387
|
+
catch {
|
|
388
|
+
// Best-effort — an unreadable config leaves the previous behavior.
|
|
389
|
+
}
|
|
390
|
+
}
|
|
319
391
|
let autoMode = isAutoModel(mergedOpts.model) || isAutoProvider(mergedOpts.provider);
|
|
320
392
|
let { type, provider } = autoMode
|
|
321
393
|
? await this.getProvider({})
|
|
@@ -351,7 +423,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
351
423
|
}
|
|
352
424
|
const parsed = parseRequestSync(message);
|
|
353
425
|
const dispatchDecision = resolvePipelineDispatch(parsed, { dev: opts.dev, text: message });
|
|
354
|
-
const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal });
|
|
426
|
+
const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal, continuation: opts.continuation });
|
|
355
427
|
// No-model fallback: the tool loop could not generate a single response
|
|
356
428
|
// AND the rules assessed a high-confidence pipeline intent — run the
|
|
357
429
|
// pipeline directly (rules decide only when the model is unavailable; the
|
|
@@ -364,10 +436,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
364
436
|
return { content: r.result?.summary ?? '', followups: [], provider: type, model };
|
|
365
437
|
}
|
|
366
438
|
// E3b: strip raw suggest_followups JSON embedded in content by the model
|
|
367
|
-
const cleanContent = (answer.content || '')
|
|
368
|
-
.replace(/\n?\*?\s*\{\s*"tool"\s*:\s*"suggest_followups"[\s\S]*$/, '')
|
|
369
|
-
.replace(/\n?\*?\s*<function=suggest_followups[\s\S]*<\/function>/g, '')
|
|
370
|
-
.trim();
|
|
439
|
+
const cleanContent = stripToolCallArtifacts(answer.content || '');
|
|
371
440
|
return {
|
|
372
441
|
content: cleanContent,
|
|
373
442
|
followups: answer.followups ?? [],
|
|
@@ -500,15 +569,19 @@ export class ChatCommand extends BaseCommand {
|
|
|
500
569
|
// Continue to interactive mode (don't return)
|
|
501
570
|
logger.info('');
|
|
502
571
|
}
|
|
503
|
-
else
|
|
504
572
|
// Ordering: the ANSWER is always printed first, then followups — the
|
|
505
573
|
// user asked for the content, not a menu. On a real terminal the
|
|
506
574
|
// followups are SELECTABLE: picking a number runs that followup as the
|
|
507
575
|
// next turn (conversation threaded), pressing Enter continues interactively.
|
|
508
576
|
// Non-TTY (scripts/CI/pipes) keeps the current print-and-exit behavior
|
|
509
577
|
// so automation is never blocked by a prompt.
|
|
510
|
-
|
|
511
|
-
|
|
578
|
+
//
|
|
579
|
+
// Print parity with the dashboard console + gateway: the answer must not
|
|
580
|
+
// carry the model's tool-call artifacts (a raw trailing suggest_followups
|
|
581
|
+
// JSON blob, or the empty ```json fence the fallback transport leaves
|
|
582
|
+
// behind) — see stripToolCallArtifacts.
|
|
583
|
+
else if (answer.content.trim()) {
|
|
584
|
+
console.log('\n' + stripToolCallArtifacts(answer.content) + '\n');
|
|
512
585
|
}
|
|
513
586
|
if (!process.stdin.isTTY) {
|
|
514
587
|
await this.renderFollowups(answer.followups ?? [], false);
|
|
@@ -522,10 +595,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
522
595
|
const picked = await this.renderFollowups(singleAnswer.followups ?? [], true);
|
|
523
596
|
if (!picked)
|
|
524
597
|
break;
|
|
525
|
-
const next = await this.runChatAnswer(picked, history, { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parseRequestSync(picked)
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
598
|
+
const next = await this.runChatAnswer(picked, history, { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parseRequestSync(picked),
|
|
599
|
+
// P5 — a picked followup continues the previous execution.
|
|
600
|
+
{ continuation: true });
|
|
601
|
+
const nextText = stripToolCallArtifacts(next.content);
|
|
602
|
+
if (nextText) {
|
|
603
|
+
console.log('\n' + nextText + '\n');
|
|
604
|
+
// NOTE: no history.push here — runChatAnswer already recorded the
|
|
605
|
+
// assistant turn. The old duplicate push gave every subsequent turn
|
|
606
|
+
// TWO copies of the previous answer (relevance noise).
|
|
529
607
|
}
|
|
530
608
|
singleAnswer = next;
|
|
531
609
|
}
|
|
@@ -547,8 +625,13 @@ export class ChatCommand extends BaseCommand {
|
|
|
547
625
|
await maybeRunBackgroundDuties(this.configManager).catch(() => { });
|
|
548
626
|
}
|
|
549
627
|
let pendingMessage;
|
|
628
|
+
// P5 — the followups the agent just suggested, so a message that MATCHES
|
|
629
|
+
// one of them (a clicked chip, or the user re-typing it on the gateway) is
|
|
630
|
+
// recognised as a continuation of the previous execution.
|
|
631
|
+
let lastFollowups = [];
|
|
550
632
|
while (true) {
|
|
551
633
|
// E3b: a chosen follow-up recommendation becomes the next message.
|
|
634
|
+
const pickedFollowup = pendingMessage !== undefined;
|
|
552
635
|
const message = pendingMessage ?? (await this.readMultiLineInput('You:'));
|
|
553
636
|
pendingMessage = undefined;
|
|
554
637
|
if (!message)
|
|
@@ -602,7 +685,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
602
685
|
const parsed = recordMetricTime('rule.parse.ms', () => parseRequestSync(message));
|
|
603
686
|
const dispatchDecision = recordMetricTime('rule.dispatch.ms', () => resolvePipelineDispatch(parsed, { dev: this.devModeAuto, text: message }));
|
|
604
687
|
const session = { type, provider, model: effectiveModel };
|
|
605
|
-
const answer = await withLogCorrelation({ sessionId: chatSessionId }, () => recordMetricTime('llm.answer.ms', () => this.runChatAnswer(message, history, session, options || {}, cacheEnabled, { auto: autoMode }, parsed
|
|
688
|
+
const answer = await withLogCorrelation({ sessionId: chatSessionId }, () => recordMetricTime('llm.answer.ms', () => this.runChatAnswer(message, history, session, options || {}, cacheEnabled, { auto: autoMode }, parsed,
|
|
689
|
+
// P5 — a picked followup (or a typed one that matches the last
|
|
690
|
+
// suggestions) is a continuation, not a fresh independent request.
|
|
691
|
+
{ continuation: pickedFollowup || isSuggestedFollowup(message, lastFollowups) })));
|
|
606
692
|
// No-model fallback: the tool loop could not generate a single response
|
|
607
693
|
// AND the rules assessed a high-confidence pipeline intent — run the
|
|
608
694
|
// pipeline directly (rules decide only when the model is unavailable).
|
|
@@ -623,7 +709,8 @@ export class ChatCommand extends BaseCommand {
|
|
|
623
709
|
if (answer.content.trim()) {
|
|
624
710
|
console.log('\n' + answer.content + '\n');
|
|
625
711
|
}
|
|
626
|
-
|
|
712
|
+
lastFollowups = answer.followups ?? [];
|
|
713
|
+
const followupPrompt = await this.renderFollowups(lastFollowups, true);
|
|
627
714
|
if (followupPrompt) {
|
|
628
715
|
pendingMessage = followupPrompt;
|
|
629
716
|
}
|
|
@@ -703,9 +790,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
703
790
|
async runChatAnswer(message, history, session, options, cacheEnabled, mode, parsed, ctxOverrides) {
|
|
704
791
|
// Cache check first (same as the legacy path).
|
|
705
792
|
const cache = getCache();
|
|
793
|
+
const cacheModel = this.cacheModelFor(session);
|
|
706
794
|
if (cacheEnabled) {
|
|
707
795
|
try {
|
|
708
|
-
const cachedResult = await cache.get(message,
|
|
796
|
+
const cachedResult = await cache.get(message, cacheModel, session.type);
|
|
709
797
|
if (cachedResult) {
|
|
710
798
|
// NOTE: the cached answer is NOT printed here — the caller prints
|
|
711
799
|
// content AFTER runChatAnswer returns (answer-first ordering). A
|
|
@@ -807,7 +895,11 @@ export class ChatCommand extends BaseCommand {
|
|
|
807
895
|
role: (h.role === 'assistant' ? 'assistant' : 'user'),
|
|
808
896
|
content: h.content,
|
|
809
897
|
})),
|
|
810
|
-
|
|
898
|
+
// P5 — a picked followup reaches the model WITH the continuation marker
|
|
899
|
+
// (the raw text stays in history), so "add a day in Hanoi" is resolved
|
|
900
|
+
// against the plan the previous turn just produced instead of being read
|
|
901
|
+
// as a brand-new request.
|
|
902
|
+
{ role: 'user', content: ctxOverrides?.continuation ? buildFollowupContinuationPrompt(message) : message },
|
|
811
903
|
];
|
|
812
904
|
// I3: one artifact session per TURN — every tool
|
|
813
905
|
// deliverable in this turn lands in the same store folder.
|
|
@@ -940,13 +1032,26 @@ export class ChatCommand extends BaseCommand {
|
|
|
940
1032
|
}
|
|
941
1033
|
};
|
|
942
1034
|
try {
|
|
1035
|
+
// R1 — the harness follows the MODEL, not only the config: config says
|
|
1036
|
+
// what this deployment prefers, the profile decides what this model can
|
|
1037
|
+
// actually use (a 0.5B local model must not get a 120B's surface).
|
|
1038
|
+
const harness = resolveModelHarnessProfile({
|
|
1039
|
+
model: session.model,
|
|
1040
|
+
configExposure: getLoopExposureMode(this.configManager),
|
|
1041
|
+
});
|
|
943
1042
|
result = await runToolLoop({
|
|
944
1043
|
messages: thread,
|
|
945
1044
|
context: toolContext,
|
|
946
1045
|
maxSteps: 16,
|
|
947
|
-
//
|
|
948
|
-
//
|
|
949
|
-
|
|
1046
|
+
// Model-window-aware thread budget: a 1M-token model keeps its window
|
|
1047
|
+
// instead of being trimmed to the fixed ~50K-token default. Undefined
|
|
1048
|
+
// (unknown window) leaves the tool-loop default untouched.
|
|
1049
|
+
threadBudgetChars: resolveThreadBudgetChars({ provider: session.type, model: session.model }),
|
|
1050
|
+
// Tiered tool exposure — the tiered (core) set starts at ~16 schemas
|
|
1051
|
+
// (~3.8K tokens/step); 'all' hands over the full set (~17K tokens/step)
|
|
1052
|
+
// and is now additionally gated on the model having the context for it.
|
|
1053
|
+
toolExposure: harness.exposure,
|
|
1054
|
+
maxParallelReads: harness.maxParallelReads,
|
|
950
1055
|
onToken: ctxOverrides?.onToken,
|
|
951
1056
|
signal: ctxOverrides?.signal,
|
|
952
1057
|
deps: {
|
|
@@ -977,7 +1082,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
977
1082
|
logger.error(String(err));
|
|
978
1083
|
endTrace(chatTraceId, false);
|
|
979
1084
|
result = {
|
|
980
|
-
|
|
1085
|
+
// Sanitized on purpose: this content is delivered verbatim by every
|
|
1086
|
+
// surface (CLI print, dashboard bubble, gateway send).
|
|
1087
|
+
content: toUserFacingGenerationError(err),
|
|
981
1088
|
followups: [],
|
|
982
1089
|
toolCalls: [],
|
|
983
1090
|
steps: 0,
|
|
@@ -994,7 +1101,11 @@ export class ChatCommand extends BaseCommand {
|
|
|
994
1101
|
if (result.content.trim() && !result.generationFailed && !result.cancelled) {
|
|
995
1102
|
if (cacheEnabled) {
|
|
996
1103
|
try {
|
|
997
|
-
|
|
1104
|
+
// Keyed by the model that ACTUALLY answered (tryGenerate records it
|
|
1105
|
+
// on success), so a weak model's reply is never replayed as a strong
|
|
1106
|
+
// model's. `cacheModel` is the pre-flight fallback for the paths that
|
|
1107
|
+
// never resolve one (e.g. a cached-hit turn).
|
|
1108
|
+
await cache.set(message, result.content, this.cacheModelFor(session) || cacheModel, session.type);
|
|
998
1109
|
}
|
|
999
1110
|
catch {
|
|
1000
1111
|
// Best-effort.
|
|
@@ -1026,13 +1137,83 @@ export class ChatCommand extends BaseCommand {
|
|
|
1026
1137
|
* broken provider never crashes the turn (it answers from the next working
|
|
1027
1138
|
* candidate, exactly like the legacy generation block).
|
|
1028
1139
|
*/
|
|
1140
|
+
/**
|
|
1141
|
+
* The model id used in the response-cache key.
|
|
1142
|
+
*
|
|
1143
|
+
* NEVER returns the `'default'` sentinel (or an empty string). Keying the
|
|
1144
|
+
* cache on `'default'` — which is what `session.model ?? 'default'` did —
|
|
1145
|
+
* collapsed EVERY model of a provider into a single entry (observed live:
|
|
1146
|
+
* `cache.json` held `provider: gemini, model: "default"`). Two consequences,
|
|
1147
|
+
* both real: an answer produced by a weak model was replayed as though a
|
|
1148
|
+
* strong one had written it, and switching `nuvira model switch` could never
|
|
1149
|
+
* take effect for a message already cached. Falls back to the provider's
|
|
1150
|
+
* effective model, then to a provider-qualified marker so distinct providers
|
|
1151
|
+
* still never collide.
|
|
1152
|
+
*/
|
|
1153
|
+
cacheModelFor(session) {
|
|
1154
|
+
if (session.model && session.model !== 'default')
|
|
1155
|
+
return session.model;
|
|
1156
|
+
try {
|
|
1157
|
+
const providers = this.configManager.getAll().providers;
|
|
1158
|
+
return resolveAdapterDefault(session.type, providers?.[session.type]?.model) ?? `${session.type}:unresolved`;
|
|
1159
|
+
}
|
|
1160
|
+
catch {
|
|
1161
|
+
return `${session.type}:unresolved`;
|
|
1162
|
+
}
|
|
1163
|
+
}
|
|
1029
1164
|
buildToolCallModel(message, session, options, mode, onToken, signal) {
|
|
1030
1165
|
return async (messages, schemas, stepOnToken, stepSignal) => {
|
|
1031
1166
|
// The effective token sink: the caller's stream wins; when a step-level
|
|
1032
1167
|
// sink is also given (loop passthrough) they are the same channel.
|
|
1033
1168
|
const sink = stepOnToken ?? onToken;
|
|
1034
1169
|
const abort = stepSignal ?? signal;
|
|
1170
|
+
/**
|
|
1171
|
+
* The CONCRETE model an attempt will use.
|
|
1172
|
+
*
|
|
1173
|
+
* `session.model` is undefined or the `'default'` sentinel whenever the
|
|
1174
|
+
* router picks a provider but no single model (which is the common auto
|
|
1175
|
+
* case). That value used to flow into four places at once — the provider
|
|
1176
|
+
* request, the reasoning trace, registry telemetry and the response
|
|
1177
|
+
* cache key — so traces read `model: unknown`, the literal `default`
|
|
1178
|
+
* reached provider APIs (`The model \`default\` does not exist`, observed
|
|
1179
|
+
* live), and EVERY model of a provider shared one cache entry (a bad
|
|
1180
|
+
* answer produced by a weak model was then replayed as if it came from a
|
|
1181
|
+
* good one). Resolving here keeps all four on the same real model id.
|
|
1182
|
+
* Falls back to undefined only when nothing can be resolved, which leaves
|
|
1183
|
+
* the adapter's own last-resort resolution in charge.
|
|
1184
|
+
*/
|
|
1185
|
+
/**
|
|
1186
|
+
* Per-turn memo of candidates that REJECTED native tool calling, keyed by
|
|
1187
|
+
* `provider|model`. Once a model has answered "tool calling is not
|
|
1188
|
+
* supported", it can never start supporting it within this turn, so
|
|
1189
|
+
* re-issuing the native call is pure waste — the live execute run paid 13
|
|
1190
|
+
* failing native requests (one per step) before each fell back to the
|
|
1191
|
+
* JSON transport, burning the provider's rate limit for nothing.
|
|
1192
|
+
* Keyed per candidate on purpose: a DIFFERENT model that does support
|
|
1193
|
+
* native tools must still get them, so a failover re-enables the fast
|
|
1194
|
+
* path automatically.
|
|
1195
|
+
*/
|
|
1196
|
+
const nativeToolsRejected = new Set();
|
|
1197
|
+
const resolveEffectiveModel = (providerType, requested) => {
|
|
1198
|
+
if (requested && requested !== 'default')
|
|
1199
|
+
return requested;
|
|
1200
|
+
try {
|
|
1201
|
+
const providers = this.configManager.getAll().providers;
|
|
1202
|
+
return resolveAdapterDefault(providerType, providers?.[providerType]?.model);
|
|
1203
|
+
}
|
|
1204
|
+
catch {
|
|
1205
|
+
return undefined;
|
|
1206
|
+
}
|
|
1207
|
+
};
|
|
1035
1208
|
const tryGenerate = async (prov, typ, mdl) => {
|
|
1209
|
+
// One resolution per attempt — the request, the trace, the telemetry
|
|
1210
|
+
// and the cache key all read the same value (see above).
|
|
1211
|
+
const effectiveModel = resolveEffectiveModel(typ, mdl);
|
|
1212
|
+
// Record the ATTEMPTED model immediately so a failed step's trace and
|
|
1213
|
+
// telemetry name the model that failed, instead of "unknown".
|
|
1214
|
+
if (effectiveModel)
|
|
1215
|
+
session.model = effectiveModel;
|
|
1216
|
+
const nativeKey = `${typ}|${effectiveModel ?? ''}`;
|
|
1036
1217
|
// Answer-quality resilience: a CONFUSED reply — the model talking about
|
|
1037
1218
|
// the tool contract (e.g. apologizing that "the provided example call
|
|
1038
1219
|
// to suggest_followups is incomplete") instead of executing it — never
|
|
@@ -1048,22 +1229,35 @@ export class ChatCommand extends BaseCommand {
|
|
|
1048
1229
|
throw err;
|
|
1049
1230
|
}
|
|
1050
1231
|
};
|
|
1051
|
-
|
|
1232
|
+
/** Mark the model that actually produced this response. */
|
|
1233
|
+
const answered = (resp) => {
|
|
1234
|
+
if (effectiveModel)
|
|
1235
|
+
session.model = effectiveModel;
|
|
1236
|
+
return resp;
|
|
1237
|
+
};
|
|
1238
|
+
// R1 — deterministic transport. A tiny model cannot use a native tool
|
|
1239
|
+
// API, and finding that out by trying cost a 400 on every turn (the
|
|
1240
|
+
// refusal memo is per-call). Unknown families are untouched: they still
|
|
1241
|
+
// try native and fall back, so nothing that works today stops working.
|
|
1242
|
+
if (shouldSkipNativeTools({ model: effectiveModel })) {
|
|
1243
|
+
nativeToolsRejected.add(nativeKey);
|
|
1244
|
+
}
|
|
1245
|
+
if (!nativeToolsRejected.has(nativeKey) && typeof prov.generateTools === 'function' && schemas.length > 0) {
|
|
1052
1246
|
try {
|
|
1053
1247
|
// P4 — stream when the provider supports it AND a sink is wired
|
|
1054
1248
|
// (the dashboard); otherwise the one-shot path with the whole
|
|
1055
1249
|
// content delivered as a single chunk so the typewriter channel
|
|
1056
1250
|
// still receives the answer (appears at once — today's behavior).
|
|
1057
1251
|
if (sink && typeof prov.generateToolsStream === 'function') {
|
|
1058
|
-
const result = await prov.generateToolsStream(messages, schemas, { ...options, model:
|
|
1252
|
+
const result = await prov.generateToolsStream(messages, schemas, { ...options, model: effectiveModel, signal: abort }, sink);
|
|
1059
1253
|
confuseCheck(result.content);
|
|
1060
|
-
return result;
|
|
1254
|
+
return answered(result);
|
|
1061
1255
|
}
|
|
1062
|
-
const result = await prov.generateTools(messages, schemas, { ...options, model:
|
|
1256
|
+
const result = await prov.generateTools(messages, schemas, { ...options, model: effectiveModel, signal: abort });
|
|
1063
1257
|
confuseCheck(result.content);
|
|
1064
1258
|
if (sink && result.content)
|
|
1065
1259
|
sink(result.content);
|
|
1066
|
-
return result;
|
|
1260
|
+
return answered(result);
|
|
1067
1261
|
}
|
|
1068
1262
|
catch (err) {
|
|
1069
1263
|
// S3: a tool-call 400 often carries the model's COMPLETE answer in
|
|
@@ -1082,9 +1276,23 @@ export class ChatCommand extends BaseCommand {
|
|
|
1082
1276
|
const toolCalls = salvaged.followups?.length
|
|
1083
1277
|
? [{ id: 'call_salvage_1', name: 'suggest_followups', arguments: { followups: salvaged.followups } }]
|
|
1084
1278
|
: [];
|
|
1085
|
-
return { content: salvaged.content, toolCalls };
|
|
1279
|
+
return answered({ content: salvaged.content, toolCalls });
|
|
1280
|
+
}
|
|
1281
|
+
// The MODEL itself cannot do native tool calling — Groq answers
|
|
1282
|
+
// 400 "`tool calling` is not supported with this model". That is
|
|
1283
|
+
// not a reason to lose the turn: the loop already ships a transport
|
|
1284
|
+
// that needs no provider tool support, and the system prompt
|
|
1285
|
+
// carries the tool contract for it. Fall THROUGH to it (no throw)
|
|
1286
|
+
// so an otherwise-good model still answers.
|
|
1287
|
+
if (isToolCallingUnsupported(err)) {
|
|
1288
|
+
// Remember it for the REST of this turn so later steps go straight
|
|
1289
|
+
// to the JSON transport instead of re-paying the failing call.
|
|
1290
|
+
nativeToolsRejected.add(nativeKey);
|
|
1291
|
+
logger.warn(' ⚠️ Model does not support native tool calling — retrying this step over the JSON tool transport.');
|
|
1292
|
+
}
|
|
1293
|
+
else {
|
|
1294
|
+
throw err;
|
|
1086
1295
|
}
|
|
1087
|
-
throw err;
|
|
1088
1296
|
}
|
|
1089
1297
|
}
|
|
1090
1298
|
// JSON fallback transport: flatten the thread into one prompt with
|
|
@@ -1094,18 +1302,22 @@ export class ChatCommand extends BaseCommand {
|
|
|
1094
1302
|
let raw;
|
|
1095
1303
|
if (typeof prov.generateStream === 'function') {
|
|
1096
1304
|
const chunks = [];
|
|
1097
|
-
await prov.generateStream(prompt, { ...options, model:
|
|
1305
|
+
await prov.generateStream(prompt, { ...options, model: effectiveModel, signal: abort }, (t) => chunks.push(t));
|
|
1098
1306
|
raw = chunks.join('');
|
|
1099
1307
|
}
|
|
1100
1308
|
else {
|
|
1101
|
-
raw = await prov.generate(prompt, { ...options, model:
|
|
1309
|
+
raw = await prov.generate(prompt, { ...options, model: effectiveModel, signal: abort });
|
|
1102
1310
|
}
|
|
1103
1311
|
const { text, calls } = extractFallbackToolCalls(raw);
|
|
1104
1312
|
confuseCheck(text);
|
|
1105
|
-
return { content: text, toolCalls: calls };
|
|
1313
|
+
return answered({ content: text, toolCalls: calls });
|
|
1106
1314
|
};
|
|
1107
1315
|
try {
|
|
1108
|
-
|
|
1316
|
+
// Same-provider transient retry FIRST (see the helper's contract): a
|
|
1317
|
+
// 503 spike at a shared endpoint must not become a dead run. Failover
|
|
1318
|
+
// only helps if a DIFFERENT provider exists — and it also hides the real
|
|
1319
|
+
// failure from the user while parking a healthy provider for 120s.
|
|
1320
|
+
return await generateWithTransientRetry(() => tryGenerate(session.provider, session.type, session.model), abort, (attempt, err) => logger.warn(` ⏳ ${session.provider.name} transient failure (attempt ${attempt}) — retrying shortly: ${err instanceof Error ? err.message.split('\n')[0] : String(err)}`));
|
|
1109
1321
|
}
|
|
1110
1322
|
catch (err) {
|
|
1111
1323
|
// Auto mode: fail over across the ranked candidates (never stuck).
|
|
@@ -1149,7 +1361,11 @@ export class ChatCommand extends BaseCommand {
|
|
|
1149
1361
|
const resp = await tryGenerate(next.provider, next.type, next.model);
|
|
1150
1362
|
session.type = next.type;
|
|
1151
1363
|
session.provider = next.provider;
|
|
1152
|
-
|
|
1364
|
+
// Keep the model that actually answered: `next.model` is often
|
|
1365
|
+
// undefined ("provider default"), and assigning it here used to
|
|
1366
|
+
// erase the resolved id that tryGenerate just recorded.
|
|
1367
|
+
if (next.model && next.model !== 'default')
|
|
1368
|
+
session.model = next.model;
|
|
1153
1369
|
logger.success(`✅ Auto failover: answered from ${next.provider.name} (${next.model}) after ${firstType} failed`);
|
|
1154
1370
|
return resp;
|
|
1155
1371
|
}
|
|
@@ -1184,10 +1400,22 @@ export class ChatCommand extends BaseCommand {
|
|
|
1184
1400
|
// deliver THAT instead of failing the whole turn. A confusing answer
|
|
1185
1401
|
// still beats an error banner in a messaging app; the confusion is
|
|
1186
1402
|
// now also visible in the chat trace for post-mortem.
|
|
1403
|
+
// The old behavior here — DELIVER the confused reply as if it were the
|
|
1404
|
+
// answer (`return { content: confusedReply }`) — was the worst of both
|
|
1405
|
+
// worlds. It shipped contract meta-talk to the sender ("Sure, I can
|
|
1406
|
+
// help you with suggestions and followups. Please provide me with more
|
|
1407
|
+
// details…") AND marked the turn a SUCCESS, so the loop cached it for
|
|
1408
|
+
// an hour and every retry inside that window replayed the same
|
|
1409
|
+
// deflection. Live evidence: that exact string sat in
|
|
1410
|
+
// ~/.nuvira/cache.json with `model: "default"`.
|
|
1411
|
+
//
|
|
1412
|
+
// Now it stays a FAILURE: rethrow so the tool loop surfaces the
|
|
1413
|
+
// sanitized, user-facing line with `generationFailed: true` (never
|
|
1414
|
+
// cached, never persisted), while the raw reply is preserved in the
|
|
1415
|
+
// log and the reasoning trace for post-mortem.
|
|
1187
1416
|
const confusedReply = err.confusedReply;
|
|
1188
1417
|
if (typeof confusedReply === 'string' && confusedReply.trim()) {
|
|
1189
|
-
logger.warn(
|
|
1190
|
-
return { content: confusedReply, toolCalls: [] };
|
|
1418
|
+
logger.warn(` ⚠️ No alternative model answered — contract-confusion reply suppressed (${confusedReply.length} chars, kept in the trace): ${confusedReply.slice(0, 160)}`);
|
|
1191
1419
|
}
|
|
1192
1420
|
throw err;
|
|
1193
1421
|
}
|
|
@@ -1398,37 +1626,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1398
1626
|
// registry may still mark it unavailable (learned from the failure), and
|
|
1399
1627
|
// blindly re-admitting would fail again on the very next message. Recovery
|
|
1400
1628
|
// is discovered in SECONDS (a 1-token spot-check), not by re-failing.
|
|
1401
|
-
//
|
|
1402
|
-
//
|
|
1403
|
-
//
|
|
1404
|
-
//
|
|
1405
|
-
for
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
this.sessionTransientFailedProviders.delete(providerType);
|
|
1411
|
-
this.sessionFailedProviders.delete(providerType);
|
|
1412
|
-
try {
|
|
1413
|
-
const registry = getModelRegistry();
|
|
1414
|
-
// Only re-verify when the registry still believes the provider is dead
|
|
1415
|
-
// (unavailable/parked) — a healthy entry means it recovered already.
|
|
1416
|
-
if (!registry.getBlockedProviders().includes(providerType))
|
|
1417
|
-
continue;
|
|
1418
|
-
const desired = getAutoRouter().resolveModel(providerType, 'chat', this.configManager);
|
|
1419
|
-
const outcome = await spotCheckModel(providerType, desired, this.configManager);
|
|
1420
|
-
// 'skipped' = the model was VERIFIED recently (within the spot-check
|
|
1421
|
-
// throttle) — that's healthy, so treat it as a pass too.
|
|
1422
|
-
if (outcome !== 'verified' && outcome !== 'skipped') {
|
|
1423
|
-
// Still down — keep it excluded for another transient window.
|
|
1424
|
-
this.sessionFailedProviders.set(providerType, Date.now() + TRANSIENT_FAILURE_EXCLUSION_MS);
|
|
1425
|
-
this.sessionTransientFailedProviders.add(providerType);
|
|
1426
|
-
}
|
|
1427
|
-
}
|
|
1428
|
-
catch {
|
|
1429
|
-
// Best-effort — re-verification must never break routing.
|
|
1430
|
-
}
|
|
1431
|
-
}
|
|
1629
|
+
// The sweep itself now lives in `learning/provider-revival.ts` so every
|
|
1630
|
+
// entry path shares ONE implementation (chat was previously the only path
|
|
1631
|
+
// that read the transient-failure marker at all — the orchestrator, edit,
|
|
1632
|
+
// execute, plan and resilient-call allocated it and never acted on it, so a
|
|
1633
|
+
// recovered provider stayed excluded for the rest of their runs).
|
|
1634
|
+
await sweepTransientFailures({
|
|
1635
|
+
...collectionRevivalStore(this.sessionFailedProviders, this.sessionTransientFailedProviders),
|
|
1636
|
+
resolveProbeModel: (provider) => getAutoRouter().resolveModel(provider, 'chat', this.configManager),
|
|
1637
|
+
}, this.configManager, { agentType: 'chat' });
|
|
1432
1638
|
const excluded = new Set([
|
|
1433
1639
|
...excludeProviders,
|
|
1434
1640
|
...[...this.sessionFailedProviders.keys()].filter((p) => isActiveExclusion(p)),
|