agent-nuvira 2.7.2 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/agents/writer-tool-calling.d.ts +14 -0
- package/dist/agents/agents/writer-tool-calling.d.ts.map +1 -1
- package/dist/agents/agents/writer-tool-calling.js +20 -0
- package/dist/agents/agents/writer-tool-calling.js.map +1 -1
- package/dist/agents/agents/writer.d.ts.map +1 -1
- package/dist/agents/agents/writer.js +15 -4
- package/dist/agents/agents/writer.js.map +1 -1
- package/dist/agents/edit-module.d.ts +7 -0
- package/dist/agents/edit-module.d.ts.map +1 -1
- package/dist/agents/edit-module.js +21 -6
- package/dist/agents/edit-module.js.map +1 -1
- package/dist/agents/orchestrator.d.ts +40 -0
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +116 -27
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/agents/tool-calling-agent.d.ts +19 -0
- package/dist/agents/tool-calling-agent.d.ts.map +1 -1
- package/dist/agents/tool-calling-agent.js +42 -2
- package/dist/agents/tool-calling-agent.js.map +1 -1
- package/dist/auth-api.d.ts +2 -0
- package/dist/auth-api.d.ts.map +1 -0
- package/dist/auth-api.js +61 -0
- package/dist/auth-api.js.map +1 -0
- package/dist/auth.test.d.ts +1 -0
- package/dist/auth.test.d.ts.map +1 -0
- package/dist/auth.test.js +3 -0
- package/dist/auth.test.js.map +1 -0
- package/dist/cli/chat.d.ts +66 -1
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +323 -77
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/cli-program.d.ts +6 -0
- package/dist/cli/cli-program.d.ts.map +1 -0
- package/dist/cli/cli-program.js +222 -0
- package/dist/cli/cli-program.js.map +1 -0
- package/dist/cli/edit.d.ts.map +1 -1
- package/dist/cli/edit.js +18 -3
- package/dist/cli/edit.js.map +1 -1
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +29 -5
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/failover-runner.d.ts +13 -0
- package/dist/cli/failover-runner.d.ts.map +1 -1
- package/dist/cli/failover-runner.js +11 -0
- package/dist/cli/failover-runner.js.map +1 -1
- package/dist/cli/gateway.d.ts +16 -0
- package/dist/cli/gateway.d.ts.map +1 -1
- package/dist/cli/gateway.js +108 -2
- package/dist/cli/gateway.js.map +1 -1
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +296 -20
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/plan.d.ts.map +1 -1
- package/dist/cli/plan.js +20 -6
- package/dist/cli/plan.js.map +1 -1
- package/dist/cli/process-control.d.ts +10 -0
- package/dist/cli/process-control.d.ts.map +1 -1
- package/dist/cli/process-control.js +73 -7
- package/dist/cli/process-control.js.map +1 -1
- package/dist/cli/router.d.ts +16 -5
- package/dist/cli/router.d.ts.map +1 -1
- package/dist/cli/router.js +31 -206
- package/dist/cli/router.js.map +1 -1
- package/dist/config/paths.d.ts +23 -0
- package/dist/config/paths.d.ts.map +1 -1
- package/dist/config/paths.js +30 -0
- package/dist/config/paths.js.map +1 -1
- package/dist/config/provider-env.d.ts +37 -0
- package/dist/config/provider-env.d.ts.map +1 -0
- package/dist/config/provider-env.js +86 -0
- package/dist/config/provider-env.js.map +1 -0
- package/dist/context/history.d.ts.map +1 -1
- package/dist/context/history.js +24 -8
- package/dist/context/history.js.map +1 -1
- package/dist/context/session-recall.d.ts +30 -0
- package/dist/context/session-recall.d.ts.map +1 -1
- package/dist/context/session-recall.js +61 -0
- package/dist/context/session-recall.js.map +1 -1
- package/dist/file.js +0 -1
- package/dist/file.js.map +1 -1
- package/dist/gateway/adapters.d.ts +10 -0
- package/dist/gateway/adapters.d.ts.map +1 -1
- package/dist/gateway/adapters.js +4 -1
- package/dist/gateway/adapters.js.map +1 -1
- package/dist/gateway/dedup.d.ts +105 -0
- package/dist/gateway/dedup.d.ts.map +1 -0
- package/dist/gateway/dedup.js +160 -0
- package/dist/gateway/dedup.js.map +1 -0
- package/dist/gateway/heartbeat.d.ts +99 -0
- package/dist/gateway/heartbeat.d.ts.map +1 -0
- package/dist/gateway/heartbeat.js +122 -0
- package/dist/gateway/heartbeat.js.map +1 -0
- package/dist/gateway/inbox.d.ts +14 -1
- package/dist/gateway/inbox.d.ts.map +1 -1
- package/dist/gateway/inbox.js.map +1 -1
- package/dist/gateway/platform-config.d.ts +5 -0
- package/dist/gateway/platform-config.d.ts.map +1 -1
- package/dist/gateway/platform-config.js +8 -13
- package/dist/gateway/platform-config.js.map +1 -1
- package/dist/gateway/registry.d.ts +65 -0
- package/dist/gateway/registry.d.ts.map +1 -1
- package/dist/gateway/registry.js +307 -32
- package/dist/gateway/registry.js.map +1 -1
- package/dist/gateway/whatsapp/baileys-bridge.d.ts +11 -1
- package/dist/gateway/whatsapp/baileys-bridge.d.ts.map +1 -1
- package/dist/gateway/whatsapp/baileys-bridge.js +64 -1
- package/dist/gateway/whatsapp/baileys-bridge.js.map +1 -1
- package/dist/gateway/whatsapp/bridge.d.ts +4 -2
- package/dist/gateway/whatsapp/bridge.d.ts.map +1 -1
- package/dist/gateway/whatsapp/bridge.js.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/inference/factory.d.ts +16 -0
- package/dist/inference/factory.d.ts.map +1 -1
- package/dist/inference/factory.js +42 -0
- package/dist/inference/factory.js.map +1 -1
- package/dist/inference/gemini-adapter.d.ts.map +1 -1
- package/dist/inference/gemini-adapter.js +3 -0
- package/dist/inference/gemini-adapter.js.map +1 -1
- package/dist/inference/interface.d.ts +12 -0
- package/dist/inference/interface.d.ts.map +1 -1
- package/dist/inference/local-adapter.d.ts +2 -0
- package/dist/inference/local-adapter.d.ts.map +1 -1
- package/dist/inference/local-adapter.js +36 -5
- package/dist/inference/local-adapter.js.map +1 -1
- package/dist/inference/model-catalog.d.ts +8 -0
- package/dist/inference/model-catalog.d.ts.map +1 -1
- package/dist/inference/model-catalog.js +40 -0
- package/dist/inference/model-catalog.js.map +1 -1
- package/dist/inference/model-probe.d.ts +7 -0
- package/dist/inference/model-probe.d.ts.map +1 -1
- package/dist/inference/model-probe.js +25 -7
- package/dist/inference/model-probe.js.map +1 -1
- package/dist/inference/native-tools.d.ts +8 -0
- package/dist/inference/native-tools.d.ts.map +1 -1
- package/dist/inference/native-tools.js +14 -3
- package/dist/inference/native-tools.js.map +1 -1
- package/dist/inference/tool-call-utils.d.ts +66 -7
- package/dist/inference/tool-call-utils.d.ts.map +1 -1
- package/dist/inference/tool-call-utils.js +134 -13
- package/dist/inference/tool-call-utils.js.map +1 -1
- package/dist/learning/auto-router.d.ts +21 -0
- package/dist/learning/auto-router.d.ts.map +1 -1
- package/dist/learning/auto-router.js +282 -21
- package/dist/learning/auto-router.js.map +1 -1
- package/dist/learning/context-budget.d.ts +133 -0
- package/dist/learning/context-budget.d.ts.map +1 -0
- package/dist/learning/context-budget.js +196 -0
- package/dist/learning/context-budget.js.map +1 -0
- package/dist/learning/cost-tracker.d.ts.map +1 -1
- package/dist/learning/cost-tracker.js +20 -8
- package/dist/learning/cost-tracker.js.map +1 -1
- package/dist/learning/eval-framework.d.ts.map +1 -1
- package/dist/learning/eval-framework.js +6 -2
- package/dist/learning/eval-framework.js.map +1 -1
- package/dist/learning/failure-bookkeeping.d.ts +41 -0
- package/dist/learning/failure-bookkeeping.d.ts.map +1 -1
- package/dist/learning/failure-bookkeeping.js +122 -4
- package/dist/learning/failure-bookkeeping.js.map +1 -1
- package/dist/learning/feedback.d.ts.map +1 -1
- package/dist/learning/feedback.js +19 -8
- package/dist/learning/feedback.js.map +1 -1
- package/dist/learning/model-first-router.d.ts.map +1 -1
- package/dist/learning/model-first-router.js +5 -11
- package/dist/learning/model-first-router.js.map +1 -1
- package/dist/learning/model-harness.d.ts +76 -0
- package/dist/learning/model-harness.d.ts.map +1 -0
- package/dist/learning/model-harness.js +119 -0
- package/dist/learning/model-harness.js.map +1 -0
- package/dist/learning/model-registry.d.ts +29 -0
- package/dist/learning/model-registry.d.ts.map +1 -1
- package/dist/learning/model-registry.js +129 -3
- package/dist/learning/model-registry.js.map +1 -1
- package/dist/learning/model-scoring.d.ts.map +1 -1
- package/dist/learning/model-scoring.js +6 -1
- package/dist/learning/model-scoring.js.map +1 -1
- package/dist/learning/model-selection.d.ts +15 -0
- package/dist/learning/model-selection.d.ts.map +1 -1
- package/dist/learning/model-selection.js +40 -1
- package/dist/learning/model-selection.js.map +1 -1
- package/dist/learning/provider-fallback.d.ts +12 -0
- package/dist/learning/provider-fallback.d.ts.map +1 -1
- package/dist/learning/provider-fallback.js +84 -0
- package/dist/learning/provider-fallback.js.map +1 -1
- package/dist/learning/provider-revival.d.ts +106 -0
- package/dist/learning/provider-revival.d.ts.map +1 -0
- package/dist/learning/provider-revival.js +154 -0
- package/dist/learning/provider-revival.js.map +1 -0
- package/dist/learning/quota-ledger.d.ts +64 -3
- package/dist/learning/quota-ledger.d.ts.map +1 -1
- package/dist/learning/quota-ledger.js +122 -9
- package/dist/learning/quota-ledger.js.map +1 -1
- package/dist/learning/resilient-call.d.ts +98 -0
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +252 -46
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/nlu/conversation-gate.d.ts +37 -0
- package/dist/nlu/conversation-gate.d.ts.map +1 -1
- package/dist/nlu/conversation-gate.js +49 -0
- package/dist/nlu/conversation-gate.js.map +1 -1
- package/dist/nlu/schema.d.ts +4 -4
- package/dist/observability/dag-bridge.d.ts +69 -0
- package/dist/observability/dag-bridge.d.ts.map +1 -0
- package/dist/observability/dag-bridge.js +56 -0
- package/dist/observability/dag-bridge.js.map +1 -0
- package/dist/observability/event-bus.d.ts +5 -10
- package/dist/observability/event-bus.d.ts.map +1 -1
- package/dist/observability/event-bus.js +10 -42
- package/dist/observability/event-bus.js.map +1 -1
- package/dist/routes/auth.d.ts +3 -0
- package/dist/routes/auth.d.ts.map +1 -0
- package/dist/routes/auth.js +41 -36
- package/dist/routes/auth.js.map +1 -0
- package/dist/skills/execution-audit.d.ts.map +1 -1
- package/dist/skills/execution-audit.js +17 -5
- package/dist/skills/execution-audit.js.map +1 -1
- package/dist/skills/sandbox-executor.d.ts +12 -1
- package/dist/skills/sandbox-executor.d.ts.map +1 -1
- package/dist/skills/sandbox-executor.js +28 -6
- package/dist/skills/sandbox-executor.js.map +1 -1
- package/dist/skills/secret-capture.d.ts +45 -1
- package/dist/skills/secret-capture.d.ts.map +1 -1
- package/dist/skills/secret-capture.js +82 -19
- package/dist/skills/secret-capture.js.map +1 -1
- package/dist/skills/skill-env-inventory.d.ts +79 -0
- package/dist/skills/skill-env-inventory.d.ts.map +1 -0
- package/dist/skills/skill-env-inventory.js +167 -0
- package/dist/skills/skill-env-inventory.js.map +1 -0
- package/dist/skills/skill-executor.d.ts +2 -0
- package/dist/skills/skill-executor.d.ts.map +1 -1
- package/dist/skills/skill-executor.js +19 -24
- package/dist/skills/skill-executor.js.map +1 -1
- package/dist/tools/ask-user.d.ts +14 -1
- package/dist/tools/ask-user.d.ts.map +1 -1
- package/dist/tools/ask-user.js +25 -1
- package/dist/tools/ask-user.js.map +1 -1
- package/dist/tools/followup-utils.d.ts +64 -0
- package/dist/tools/followup-utils.d.ts.map +1 -0
- package/dist/tools/followup-utils.js +158 -0
- package/dist/tools/followup-utils.js.map +1 -0
- package/dist/tools/memory-tools.d.ts +35 -0
- package/dist/tools/memory-tools.d.ts.map +1 -1
- package/dist/tools/memory-tools.js +76 -0
- package/dist/tools/memory-tools.js.map +1 -1
- package/dist/tools/pipeline-tool.d.ts.map +1 -1
- package/dist/tools/pipeline-tool.js +8 -3
- package/dist/tools/pipeline-tool.js.map +1 -1
- package/dist/tools/registry.d.ts +25 -12
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +31 -6
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/skill-tool.d.ts.map +1 -1
- package/dist/tools/skill-tool.js +52 -5
- package/dist/tools/skill-tool.js.map +1 -1
- package/dist/tools/tool-loop.d.ts +63 -1
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +305 -66
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/tools/toolsets.d.ts +19 -3
- package/dist/tools/toolsets.d.ts.map +1 -1
- package/dist/tools/toolsets.js +21 -5
- package/dist/tools/toolsets.js.map +1 -1
- package/dist/utils/env.d.ts.map +1 -1
- package/dist/utils/env.js +9 -13
- package/dist/utils/env.js.map +1 -1
- package/dist/web-dashboard/chat-console.d.ts +13 -1
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +16 -2
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/hub-data.d.ts +2 -0
- package/dist/web-dashboard/hub-data.d.ts.map +1 -1
- package/dist/web-dashboard/hub-data.js +4 -0
- package/dist/web-dashboard/hub-data.js.map +1 -1
- package/dist/web-dashboard/server.d.ts +6 -0
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +196 -3
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +31 -1
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/package.json +11 -1
- package/src/web-dashboard/public/assets/index-56yPnN_m.js +207 -0
- package/src/web-dashboard/public/assets/index-56yPnN_m.js.map +1 -0
- package/src/web-dashboard/public/assets/{index-C507EUWf.css → index-Cyd6tIew.css} +1 -1
- package/src/web-dashboard/public/index.html +2 -2
- package/dist/example.d.ts +0 -1
- package/dist/example.d.ts.map +0 -1
- package/dist/example.js +0 -3
- package/dist/example.js.map +0 -1
- package/dist/fresh.d.ts +0 -2
- package/dist/fresh.d.ts.map +0 -1
- package/dist/fresh.js +0 -2
- package/dist/fresh.js.map +0 -1
- package/dist/mcp/mcp-dashboard-oauth.d.ts +0 -94
- package/dist/mcp/mcp-dashboard-oauth.d.ts.map +0 -1
- package/dist/mcp/mcp-dashboard-oauth.js +0 -140
- package/dist/mcp/mcp-dashboard-oauth.js.map +0 -1
- package/dist/memory/background-sync.d.ts +0 -107
- package/dist/memory/background-sync.d.ts.map +0 -1
- package/dist/memory/background-sync.js +0 -242
- package/dist/memory/background-sync.js.map +0 -1
- package/dist/memory/cross-session.d.ts +0 -68
- package/dist/memory/cross-session.d.ts.map +0 -1
- package/dist/memory/cross-session.js +0 -153
- package/dist/memory/cross-session.js.map +0 -1
- package/dist/memory/drift-detector.d.ts +0 -76
- package/dist/memory/drift-detector.d.ts.map +0 -1
- package/dist/memory/drift-detector.js +0 -216
- package/dist/memory/drift-detector.js.map +0 -1
- package/dist/memory/enhanced-manager.d.ts +0 -105
- package/dist/memory/enhanced-manager.d.ts.map +0 -1
- package/dist/memory/enhanced-manager.js +0 -280
- package/dist/memory/enhanced-manager.js.map +0 -1
- package/dist/memory/session-extraction.d.ts +0 -93
- package/dist/memory/session-extraction.d.ts.map +0 -1
- package/dist/memory/session-extraction.js +0 -290
- package/dist/memory/session-extraction.js.map +0 -1
- package/dist/memory/sqlite-store.d.ts +0 -208
- package/dist/memory/sqlite-store.d.ts.map +0 -1
- package/dist/memory/sqlite-store.js +0 -556
- package/dist/memory/sqlite-store.js.map +0 -1
- package/dist/skills/daytona-executor.d.ts +0 -78
- package/dist/skills/daytona-executor.d.ts.map +0 -1
- package/dist/skills/daytona-executor.js +0 -256
- package/dist/skills/daytona-executor.js.map +0 -1
- package/dist/skills/modal-executor.d.ts +0 -81
- package/dist/skills/modal-executor.d.ts.map +0 -1
- package/dist/skills/modal-executor.js +0 -285
- package/dist/skills/modal-executor.js.map +0 -1
- package/dist/sync/inventory.d.ts +0 -1
- package/dist/sync/inventory.d.ts.map +0 -1
- package/dist/sync/inventory.js +0 -3
- package/dist/sync/inventory.js.map +0 -1
- package/dist/sync/validate.d.ts +0 -1
- package/dist/sync/validate.d.ts.map +0 -1
- package/dist/sync/validate.js +0 -3
- package/dist/sync/validate.js.map +0 -1
- package/dist/test.d.ts +0 -2
- package/dist/test.d.ts.map +0 -1
- package/dist/test.js +0 -3
- package/dist/test.js.map +0 -1
- package/dist/tools/browser-tool.d.ts +0 -206
- package/dist/tools/browser-tool.d.ts.map +0 -1
- package/dist/tools/browser-tool.js +0 -726
- package/dist/tools/browser-tool.js.map +0 -1
- package/dist/tools/delegate-tool.d.ts +0 -109
- package/dist/tools/delegate-tool.d.ts.map +0 -1
- package/dist/tools/delegate-tool.js +0 -185
- package/dist/tools/delegate-tool.js.map +0 -1
- package/dist/tools/delegation-state.d.ts +0 -87
- package/dist/tools/delegation-state.d.ts.map +0 -1
- package/dist/tools/delegation-state.js +0 -278
- package/dist/tools/delegation-state.js.map +0 -1
- package/dist/tools/desktop-ui.d.ts +0 -148
- package/dist/tools/desktop-ui.d.ts.map +0 -1
- package/dist/tools/desktop-ui.js +0 -486
- package/dist/tools/desktop-ui.js.map +0 -1
- package/dist/tools/image-video-tool.d.ts +0 -88
- package/dist/tools/image-video-tool.d.ts.map +0 -1
- package/dist/tools/image-video-tool.js +0 -276
- package/dist/tools/image-video-tool.js.map +0 -1
- package/dist/tools/kanban-cron-tools.d.ts +0 -92
- package/dist/tools/kanban-cron-tools.d.ts.map +0 -1
- package/dist/tools/kanban-cron-tools.js +0 -197
- package/dist/tools/kanban-cron-tools.js.map +0 -1
- package/dist/tools/messaging-tool.d.ts +0 -127
- package/dist/tools/messaging-tool.d.ts.map +0 -1
- package/dist/tools/messaging-tool.js +0 -314
- package/dist/tools/messaging-tool.js.map +0 -1
- package/dist/tools/modality/generic-caller.d.ts +0 -43
- package/dist/tools/modality/generic-caller.d.ts.map +0 -1
- package/dist/tools/modality/generic-caller.js +0 -254
- package/dist/tools/modality/generic-caller.js.map +0 -1
- package/dist/tools/modality/modality-catalog.d.ts +0 -134
- package/dist/tools/modality/modality-catalog.d.ts.map +0 -1
- package/dist/tools/modality/modality-catalog.js +0 -445
- package/dist/tools/modality/modality-catalog.js.map +0 -1
- package/dist/tools/modality/tool-router.d.ts +0 -81
- package/dist/tools/modality/tool-router.d.ts.map +0 -1
- package/dist/tools/modality/tool-router.js +0 -84
- package/dist/tools/modality/tool-router.js.map +0 -1
- package/dist/tools/security-tools.d.ts +0 -107
- package/dist/tools/security-tools.d.ts.map +0 -1
- package/dist/tools/security-tools.js +0 -264
- package/dist/tools/security-tools.js.map +0 -1
- package/dist/tools/ssh-tool.d.ts +0 -130
- package/dist/tools/ssh-tool.d.ts.map +0 -1
- package/dist/tools/ssh-tool.js +0 -328
- package/dist/tools/ssh-tool.js.map +0 -1
- package/dist/tools/tts-tool.d.ts +0 -91
- package/dist/tools/tts-tool.d.ts.map +0 -1
- package/dist/tools/tts-tool.js +0 -406
- package/dist/tools/tts-tool.js.map +0 -1
- package/dist/utils/shell-detection.d.ts +0 -124
- package/dist/utils/shell-detection.d.ts.map +0 -1
- package/dist/utils/shell-detection.js +0 -305
- package/dist/utils/shell-detection.js.map +0 -1
- package/dist/utils/windows-paths.d.ts +0 -185
- package/dist/utils/windows-paths.d.ts.map +0 -1
- package/dist/utils/windows-paths.js +0 -472
- package/dist/utils/windows-paths.js.map +0 -1
- package/src/web-dashboard/public/assets/index-C4frng1Q.js +0 -207
- package/src/web-dashboard/public/assets/index-C4frng1Q.js.map +0 -1
package/dist/cli/chat.js
CHANGED
|
@@ -16,29 +16,35 @@ import { getMemoryManager } from '../memory/manager.js';
|
|
|
16
16
|
import { logger } from '../utils/logger.js';
|
|
17
17
|
import { printOrchestrationResult } from './execute.js';
|
|
18
18
|
import { applyActiveModel } from './model.js';
|
|
19
|
-
import { getProviderFallback, classifyFallbackError, isRetryableError, recordRegistrySuccess } from '../learning/provider-fallback.js';
|
|
20
|
-
import { recordActionFailure
|
|
19
|
+
import { getProviderFallback, classifyFallbackError, isRetryableError, isTransientForRetry, recordRegistrySuccess } from '../learning/provider-fallback.js';
|
|
20
|
+
import { recordActionFailure } from '../learning/failure-bookkeeping.js';
|
|
21
|
+
import { resolveThreadBudgetChars } from '../learning/context-budget.js';
|
|
21
22
|
import { getAutoRouter, isAutoModel, isAutoProvider } from '../learning/auto-router.js';
|
|
22
23
|
import { estimateTokens } from '../learning/cost-tracker.js';
|
|
23
24
|
import { getModelRegistry } from '../learning/model-registry.js';
|
|
24
|
-
import { refreshModelRegistry
|
|
25
|
+
import { refreshModelRegistry } from '../inference/model-probe.js';
|
|
25
26
|
import { recordRoutingDecision } from '../learning/routing-history.js';
|
|
26
27
|
import { shouldConfirmFailover, promptFailoverChoice } from './failover-prompt.js';
|
|
27
28
|
import { buildAutoResolveOptions } from '../learning/resolve-options.js';
|
|
29
|
+
import { buildDeepFailoverPool, createFailoverExclusionFilter } from '../learning/resilient-call.js';
|
|
28
30
|
import { parseRequestSync } from '../nlu/parser.js';
|
|
29
31
|
import { PlanStore } from '../tools/plan-store.js';
|
|
30
32
|
import { withLogCorrelation } from '../enterprise/log.js';
|
|
31
33
|
import { recordMetricTime, getMetrics } from '../enterprise/metrics.js';
|
|
32
34
|
import { resolveDispatch } from '../nlu/actions.js';
|
|
33
|
-
import {
|
|
35
|
+
import { hasCodingAction, resolveAskKind } from '../nlu/conversation-gate.js';
|
|
34
36
|
import { runToolLoop, extractFallbackToolCalls } from '../tools/tool-loop.js';
|
|
35
|
-
import { looksLikeConfusedScaffoldingReply } from '../inference/tool-call-utils.js';
|
|
37
|
+
import { looksLikeConfusedScaffoldingReply, toUserFacingGenerationError, isToolCallingUnsupported, } from '../inference/tool-call-utils.js';
|
|
36
38
|
import { beginTrace, endTrace, recordStep } from '../learning/reasoning-trace.js';
|
|
37
39
|
import { getLoopExposureMode } from '../tools/toolsets.js';
|
|
40
|
+
import { resolveModelHarnessProfile, shouldSkipNativeTools } from '../learning/model-harness.js';
|
|
41
|
+
import { resolveAdapterDefault } from '../learning/model-selection.js';
|
|
38
42
|
import { buildLoopProjectContext } from '../tools/loop-project-context.js';
|
|
43
|
+
import { sweepTransientFailures, collectionRevivalStore } from '../learning/provider-revival.js';
|
|
39
44
|
import { analyzeComplexity } from '../learning/hybrid-router.js';
|
|
40
45
|
import { routingCacheSignature, withRoutingCache } from '../learning/routing-cache.js';
|
|
41
46
|
import { getTool, TOOL_CONTRACT_JSON } from '../tools/registry.js';
|
|
47
|
+
import { buildFollowupContinuationPrompt, isSuggestedFollowup, } from '../tools/followup-utils.js';
|
|
42
48
|
// S2/S3 — the shared tool-call reliability helpers (salvage failed_generation,
|
|
43
49
|
// compact fallback schemas). One copy for every tool-calling surface, not
|
|
44
50
|
// chat-private (execute/plan/… inherit the fix).
|
|
@@ -167,7 +173,11 @@ export function resolvePipelineDispatch(parsed, opts) {
|
|
|
167
173
|
// NLU alone would misread it as chat ("how do I add JWT auth?" → explain
|
|
168
174
|
// → chat, but the user wants the auth added).
|
|
169
175
|
if (opts?.text) {
|
|
170
|
-
|
|
176
|
+
// ONE shared rule for every surface (see resolveAskKind): a genuine
|
|
177
|
+
// question never dispatches; a coding verb in command position always
|
|
178
|
+
// does. Keeping the gateway on this same function is what stops the two
|
|
179
|
+
// from disagreeing about the same ask.
|
|
180
|
+
if (resolveAskKind(opts.text, parsed) === 'chat') {
|
|
171
181
|
return { dispatch: false, needConfirm: false };
|
|
172
182
|
}
|
|
173
183
|
if (hasCodingAction(opts.text)) {
|
|
@@ -216,6 +226,41 @@ export async function runDeveloperMode(goal, configManager, options) {
|
|
|
216
226
|
* (rules act only as the no-model fallback in the
|
|
217
227
|
* caller, never to skip the loop).
|
|
218
228
|
*/
|
|
229
|
+
/**
|
|
230
|
+
* Backoff schedule for a SAME-provider retry on a transient failure. Two extra
|
|
231
|
+
* attempts, deliberately short: a capacity spike at a shared endpoint clears in
|
|
232
|
+
* seconds, and the user is waiting in the foreground. Long/looping retries belong
|
|
233
|
+
* to the background runners, not the interactive turn.
|
|
234
|
+
*/
|
|
235
|
+
export const TRANSIENT_RETRY_DELAYS_MS = [1_000, 3_000];
|
|
236
|
+
/**
|
|
237
|
+
* Run one provider attempt, retrying transient failures against the SAME
|
|
238
|
+
* provider before giving up.
|
|
239
|
+
*
|
|
240
|
+
* Why this exists next to the failover walk rather than inside it: the walk
|
|
241
|
+
* needs a DIFFERENT provider to exist, and it books the failure against the one
|
|
242
|
+
* that just failed. Verified live — a single configured provider plus a Gemini
|
|
243
|
+
* 503 meant no retry at all, the circuit breaker parked the provider for 120s,
|
|
244
|
+
* and the agent degraded to editing with zero gathered context. A transient
|
|
245
|
+
* spike must cost a few seconds, not the whole task.
|
|
246
|
+
*
|
|
247
|
+
* Never retries: non-transient classes (auth, rate-limit, model/harness faults),
|
|
248
|
+
* a cancelled turn, or once the schedule is exhausted.
|
|
249
|
+
*/
|
|
250
|
+
export async function generateWithTransientRetry(attempt, signal, onRetry) {
|
|
251
|
+
for (let i = 0;; i += 1) {
|
|
252
|
+
try {
|
|
253
|
+
return await attempt();
|
|
254
|
+
}
|
|
255
|
+
catch (err) {
|
|
256
|
+
const canRetry = i < TRANSIENT_RETRY_DELAYS_MS.length && isTransientForRetry(err);
|
|
257
|
+
if (!canRetry || signal?.aborted)
|
|
258
|
+
throw err;
|
|
259
|
+
onRetry?.(i + 1, err);
|
|
260
|
+
await new Promise((resolve) => setTimeout(resolve, TRANSIENT_RETRY_DELAYS_MS[i]));
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
}
|
|
219
264
|
function buildToolSystemPrompt(parsed) {
|
|
220
265
|
return [
|
|
221
266
|
"You are Nuvira, Agent-Nuvira's AI agent. You code, create, write, analyze, and automate — anything the user needs. You identify as Nuvira (never 'Buff').",
|
|
@@ -261,6 +306,17 @@ export class ChatCommand extends BaseCommand {
|
|
|
261
306
|
* Cleared when the chat exits.
|
|
262
307
|
*/
|
|
263
308
|
sessionFailedProviders = new Map();
|
|
309
|
+
/**
|
|
310
|
+
* `provider|model` → expiry of a MODEL-scoped session exclusion.
|
|
311
|
+
*
|
|
312
|
+
* A 429 on ONE model now records HERE rather than in sessionFailedProviders:
|
|
313
|
+
* free tiers meter per-model (RPD/TPM), so excluding the whole provider is
|
|
314
|
+
* what stopped chat from ever reaching a provider's 2nd-best model. Siblings
|
|
315
|
+
* of the failed model stay routable; the failure only escalates to the
|
|
316
|
+
* provider-wide map when several distinct models of that provider are
|
|
317
|
+
* rate-limited (a genuinely shared limit).
|
|
318
|
+
*/
|
|
319
|
+
sessionFailedModels = new Map();
|
|
264
320
|
// RATE_LIMIT_EXCLUSION_MS + TRANSIENT_FAILURE_EXCLUSION_MS now live in
|
|
265
321
|
// src/learning/failure-bookkeeping.ts (shared with every action) — see
|
|
266
322
|
// recordActionFailure. Behavior is identical: same values, same semantics.
|
|
@@ -296,7 +352,13 @@ export class ChatCommand extends BaseCommand {
|
|
|
296
352
|
* carries prior turns so the dashboard threads a real conversation.
|
|
297
353
|
*/
|
|
298
354
|
async answerOnce(message, opts = {}) {
|
|
299
|
-
|
|
355
|
+
// `'default'` is the config SENTINEL for "use the provider's default
|
|
356
|
+
// model", never a real model id. Left in place it (a) disables auto routing
|
|
357
|
+
// (isAutoModel('default') is false) and (b) is truthy, so the adapter's
|
|
358
|
+
// `options?.model || requireAdapterModel(...)` fallback is skipped and the
|
|
359
|
+
// literal string 'default' reaches the provider → "model not found".
|
|
360
|
+
const requestedModel = opts.model && opts.model !== 'default' ? opts.model : undefined;
|
|
361
|
+
const activeOpts = applyActiveModel({ provider: opts.provider, model: requestedModel });
|
|
300
362
|
const mergedOpts = { ...opts, provider: activeOpts.provider, model: activeOpts.model };
|
|
301
363
|
let autoMode = isAutoModel(mergedOpts.model) || isAutoProvider(mergedOpts.provider);
|
|
302
364
|
let { type, provider } = autoMode
|
|
@@ -333,7 +395,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
333
395
|
}
|
|
334
396
|
const parsed = parseRequestSync(message);
|
|
335
397
|
const dispatchDecision = resolvePipelineDispatch(parsed, { dev: opts.dev, text: message });
|
|
336
|
-
const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal });
|
|
398
|
+
const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal, continuation: opts.continuation });
|
|
337
399
|
// No-model fallback: the tool loop could not generate a single response
|
|
338
400
|
// AND the rules assessed a high-confidence pipeline intent — run the
|
|
339
401
|
// pipeline directly (rules decide only when the model is unavailable; the
|
|
@@ -504,10 +566,14 @@ export class ChatCommand extends BaseCommand {
|
|
|
504
566
|
const picked = await this.renderFollowups(singleAnswer.followups ?? [], true);
|
|
505
567
|
if (!picked)
|
|
506
568
|
break;
|
|
507
|
-
const next = await this.runChatAnswer(picked, history, { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parseRequestSync(picked)
|
|
569
|
+
const next = await this.runChatAnswer(picked, history, { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parseRequestSync(picked),
|
|
570
|
+
// P5 — a picked followup continues the previous execution.
|
|
571
|
+
{ continuation: true });
|
|
508
572
|
if (next.content.trim()) {
|
|
509
573
|
console.log('\n' + next.content + '\n');
|
|
510
|
-
history.push
|
|
574
|
+
// NOTE: no history.push here — runChatAnswer already recorded the
|
|
575
|
+
// assistant turn. The old duplicate push gave every subsequent turn
|
|
576
|
+
// TWO copies of the previous answer (relevance noise).
|
|
511
577
|
}
|
|
512
578
|
singleAnswer = next;
|
|
513
579
|
}
|
|
@@ -529,8 +595,13 @@ export class ChatCommand extends BaseCommand {
|
|
|
529
595
|
await maybeRunBackgroundDuties(this.configManager).catch(() => { });
|
|
530
596
|
}
|
|
531
597
|
let pendingMessage;
|
|
598
|
+
// P5 — the followups the agent just suggested, so a message that MATCHES
|
|
599
|
+
// one of them (a clicked chip, or the user re-typing it on the gateway) is
|
|
600
|
+
// recognised as a continuation of the previous execution.
|
|
601
|
+
let lastFollowups = [];
|
|
532
602
|
while (true) {
|
|
533
603
|
// E3b: a chosen follow-up recommendation becomes the next message.
|
|
604
|
+
const pickedFollowup = pendingMessage !== undefined;
|
|
534
605
|
const message = pendingMessage ?? (await this.readMultiLineInput('You:'));
|
|
535
606
|
pendingMessage = undefined;
|
|
536
607
|
if (!message)
|
|
@@ -584,7 +655,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
584
655
|
const parsed = recordMetricTime('rule.parse.ms', () => parseRequestSync(message));
|
|
585
656
|
const dispatchDecision = recordMetricTime('rule.dispatch.ms', () => resolvePipelineDispatch(parsed, { dev: this.devModeAuto, text: message }));
|
|
586
657
|
const session = { type, provider, model: effectiveModel };
|
|
587
|
-
const answer = await withLogCorrelation({ sessionId: chatSessionId }, () => recordMetricTime('llm.answer.ms', () => this.runChatAnswer(message, history, session, options || {}, cacheEnabled, { auto: autoMode }, parsed
|
|
658
|
+
const answer = await withLogCorrelation({ sessionId: chatSessionId }, () => recordMetricTime('llm.answer.ms', () => this.runChatAnswer(message, history, session, options || {}, cacheEnabled, { auto: autoMode }, parsed,
|
|
659
|
+
// P5 — a picked followup (or a typed one that matches the last
|
|
660
|
+
// suggestions) is a continuation, not a fresh independent request.
|
|
661
|
+
{ continuation: pickedFollowup || isSuggestedFollowup(message, lastFollowups) })));
|
|
588
662
|
// No-model fallback: the tool loop could not generate a single response
|
|
589
663
|
// AND the rules assessed a high-confidence pipeline intent — run the
|
|
590
664
|
// pipeline directly (rules decide only when the model is unavailable).
|
|
@@ -605,7 +679,8 @@ export class ChatCommand extends BaseCommand {
|
|
|
605
679
|
if (answer.content.trim()) {
|
|
606
680
|
console.log('\n' + answer.content + '\n');
|
|
607
681
|
}
|
|
608
|
-
|
|
682
|
+
lastFollowups = answer.followups ?? [];
|
|
683
|
+
const followupPrompt = await this.renderFollowups(lastFollowups, true);
|
|
609
684
|
if (followupPrompt) {
|
|
610
685
|
pendingMessage = followupPrompt;
|
|
611
686
|
}
|
|
@@ -685,9 +760,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
685
760
|
async runChatAnswer(message, history, session, options, cacheEnabled, mode, parsed, ctxOverrides) {
|
|
686
761
|
// Cache check first (same as the legacy path).
|
|
687
762
|
const cache = getCache();
|
|
763
|
+
const cacheModel = this.cacheModelFor(session);
|
|
688
764
|
if (cacheEnabled) {
|
|
689
765
|
try {
|
|
690
|
-
const cachedResult = await cache.get(message,
|
|
766
|
+
const cachedResult = await cache.get(message, cacheModel, session.type);
|
|
691
767
|
if (cachedResult) {
|
|
692
768
|
// NOTE: the cached answer is NOT printed here — the caller prints
|
|
693
769
|
// content AFTER runChatAnswer returns (answer-first ordering). A
|
|
@@ -789,7 +865,11 @@ export class ChatCommand extends BaseCommand {
|
|
|
789
865
|
role: (h.role === 'assistant' ? 'assistant' : 'user'),
|
|
790
866
|
content: h.content,
|
|
791
867
|
})),
|
|
792
|
-
|
|
868
|
+
// P5 — a picked followup reaches the model WITH the continuation marker
|
|
869
|
+
// (the raw text stays in history), so "add a day in Hanoi" is resolved
|
|
870
|
+
// against the plan the previous turn just produced instead of being read
|
|
871
|
+
// as a brand-new request.
|
|
872
|
+
{ role: 'user', content: ctxOverrides?.continuation ? buildFollowupContinuationPrompt(message) : message },
|
|
793
873
|
];
|
|
794
874
|
// I3: one artifact session per TURN — every tool
|
|
795
875
|
// deliverable in this turn lands in the same store folder.
|
|
@@ -922,13 +1002,26 @@ export class ChatCommand extends BaseCommand {
|
|
|
922
1002
|
}
|
|
923
1003
|
};
|
|
924
1004
|
try {
|
|
1005
|
+
// R1 — the harness follows the MODEL, not only the config: config says
|
|
1006
|
+
// what this deployment prefers, the profile decides what this model can
|
|
1007
|
+
// actually use (a 0.5B local model must not get a 120B's surface).
|
|
1008
|
+
const harness = resolveModelHarnessProfile({
|
|
1009
|
+
model: session.model,
|
|
1010
|
+
configExposure: getLoopExposureMode(this.configManager),
|
|
1011
|
+
});
|
|
925
1012
|
result = await runToolLoop({
|
|
926
1013
|
messages: thread,
|
|
927
1014
|
context: toolContext,
|
|
928
1015
|
maxSteps: 16,
|
|
929
|
-
//
|
|
930
|
-
//
|
|
931
|
-
|
|
1016
|
+
// Model-window-aware thread budget: a 1M-token model keeps its window
|
|
1017
|
+
// instead of being trimmed to the fixed ~50K-token default. Undefined
|
|
1018
|
+
// (unknown window) leaves the tool-loop default untouched.
|
|
1019
|
+
threadBudgetChars: resolveThreadBudgetChars({ provider: session.type, model: session.model }),
|
|
1020
|
+
// Tiered tool exposure — the tiered (core) set starts at ~16 schemas
|
|
1021
|
+
// (~3.8K tokens/step); 'all' hands over the full set (~17K tokens/step)
|
|
1022
|
+
// and is now additionally gated on the model having the context for it.
|
|
1023
|
+
toolExposure: harness.exposure,
|
|
1024
|
+
maxParallelReads: harness.maxParallelReads,
|
|
932
1025
|
onToken: ctxOverrides?.onToken,
|
|
933
1026
|
signal: ctxOverrides?.signal,
|
|
934
1027
|
deps: {
|
|
@@ -959,7 +1052,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
959
1052
|
logger.error(String(err));
|
|
960
1053
|
endTrace(chatTraceId, false);
|
|
961
1054
|
result = {
|
|
962
|
-
|
|
1055
|
+
// Sanitized on purpose: this content is delivered verbatim by every
|
|
1056
|
+
// surface (CLI print, dashboard bubble, gateway send).
|
|
1057
|
+
content: toUserFacingGenerationError(err),
|
|
963
1058
|
followups: [],
|
|
964
1059
|
toolCalls: [],
|
|
965
1060
|
steps: 0,
|
|
@@ -976,7 +1071,11 @@ export class ChatCommand extends BaseCommand {
|
|
|
976
1071
|
if (result.content.trim() && !result.generationFailed && !result.cancelled) {
|
|
977
1072
|
if (cacheEnabled) {
|
|
978
1073
|
try {
|
|
979
|
-
|
|
1074
|
+
// Keyed by the model that ACTUALLY answered (tryGenerate records it
|
|
1075
|
+
// on success), so a weak model's reply is never replayed as a strong
|
|
1076
|
+
// model's. `cacheModel` is the pre-flight fallback for the paths that
|
|
1077
|
+
// never resolve one (e.g. a cached-hit turn).
|
|
1078
|
+
await cache.set(message, result.content, this.cacheModelFor(session) || cacheModel, session.type);
|
|
980
1079
|
}
|
|
981
1080
|
catch {
|
|
982
1081
|
// Best-effort.
|
|
@@ -1008,13 +1107,83 @@ export class ChatCommand extends BaseCommand {
|
|
|
1008
1107
|
* broken provider never crashes the turn (it answers from the next working
|
|
1009
1108
|
* candidate, exactly like the legacy generation block).
|
|
1010
1109
|
*/
|
|
1110
|
+
/**
|
|
1111
|
+
* The model id used in the response-cache key.
|
|
1112
|
+
*
|
|
1113
|
+
* NEVER returns the `'default'` sentinel (or an empty string). Keying the
|
|
1114
|
+
* cache on `'default'` — which is what `session.model ?? 'default'` did —
|
|
1115
|
+
* collapsed EVERY model of a provider into a single entry (observed live:
|
|
1116
|
+
* `cache.json` held `provider: gemini, model: "default"`). Two consequences,
|
|
1117
|
+
* both real: an answer produced by a weak model was replayed as though a
|
|
1118
|
+
* strong one had written it, and switching `nuvira model switch` could never
|
|
1119
|
+
* take effect for a message already cached. Falls back to the provider's
|
|
1120
|
+
* effective model, then to a provider-qualified marker so distinct providers
|
|
1121
|
+
* still never collide.
|
|
1122
|
+
*/
|
|
1123
|
+
cacheModelFor(session) {
|
|
1124
|
+
if (session.model && session.model !== 'default')
|
|
1125
|
+
return session.model;
|
|
1126
|
+
try {
|
|
1127
|
+
const providers = this.configManager.getAll().providers;
|
|
1128
|
+
return resolveAdapterDefault(session.type, providers?.[session.type]?.model) ?? `${session.type}:unresolved`;
|
|
1129
|
+
}
|
|
1130
|
+
catch {
|
|
1131
|
+
return `${session.type}:unresolved`;
|
|
1132
|
+
}
|
|
1133
|
+
}
|
|
1011
1134
|
buildToolCallModel(message, session, options, mode, onToken, signal) {
|
|
1012
1135
|
return async (messages, schemas, stepOnToken, stepSignal) => {
|
|
1013
1136
|
// The effective token sink: the caller's stream wins; when a step-level
|
|
1014
1137
|
// sink is also given (loop passthrough) they are the same channel.
|
|
1015
1138
|
const sink = stepOnToken ?? onToken;
|
|
1016
1139
|
const abort = stepSignal ?? signal;
|
|
1140
|
+
/**
|
|
1141
|
+
* The CONCRETE model an attempt will use.
|
|
1142
|
+
*
|
|
1143
|
+
* `session.model` is undefined or the `'default'` sentinel whenever the
|
|
1144
|
+
* router picks a provider but no single model (which is the common auto
|
|
1145
|
+
* case). That value used to flow into four places at once — the provider
|
|
1146
|
+
* request, the reasoning trace, registry telemetry and the response
|
|
1147
|
+
* cache key — so traces read `model: unknown`, the literal `default`
|
|
1148
|
+
* reached provider APIs (`The model \`default\` does not exist`, observed
|
|
1149
|
+
* live), and EVERY model of a provider shared one cache entry (a bad
|
|
1150
|
+
* answer produced by a weak model was then replayed as if it came from a
|
|
1151
|
+
* good one). Resolving here keeps all four on the same real model id.
|
|
1152
|
+
* Falls back to undefined only when nothing can be resolved, which leaves
|
|
1153
|
+
* the adapter's own last-resort resolution in charge.
|
|
1154
|
+
*/
|
|
1155
|
+
/**
|
|
1156
|
+
* Per-turn memo of candidates that REJECTED native tool calling, keyed by
|
|
1157
|
+
* `provider|model`. Once a model has answered "tool calling is not
|
|
1158
|
+
* supported", it can never start supporting it within this turn, so
|
|
1159
|
+
* re-issuing the native call is pure waste — the live execute run paid 13
|
|
1160
|
+
* failing native requests (one per step) before each fell back to the
|
|
1161
|
+
* JSON transport, burning the provider's rate limit for nothing.
|
|
1162
|
+
* Keyed per candidate on purpose: a DIFFERENT model that does support
|
|
1163
|
+
* native tools must still get them, so a failover re-enables the fast
|
|
1164
|
+
* path automatically.
|
|
1165
|
+
*/
|
|
1166
|
+
const nativeToolsRejected = new Set();
|
|
1167
|
+
const resolveEffectiveModel = (providerType, requested) => {
|
|
1168
|
+
if (requested && requested !== 'default')
|
|
1169
|
+
return requested;
|
|
1170
|
+
try {
|
|
1171
|
+
const providers = this.configManager.getAll().providers;
|
|
1172
|
+
return resolveAdapterDefault(providerType, providers?.[providerType]?.model);
|
|
1173
|
+
}
|
|
1174
|
+
catch {
|
|
1175
|
+
return undefined;
|
|
1176
|
+
}
|
|
1177
|
+
};
|
|
1017
1178
|
const tryGenerate = async (prov, typ, mdl) => {
|
|
1179
|
+
// One resolution per attempt — the request, the trace, the telemetry
|
|
1180
|
+
// and the cache key all read the same value (see above).
|
|
1181
|
+
const effectiveModel = resolveEffectiveModel(typ, mdl);
|
|
1182
|
+
// Record the ATTEMPTED model immediately so a failed step's trace and
|
|
1183
|
+
// telemetry name the model that failed, instead of "unknown".
|
|
1184
|
+
if (effectiveModel)
|
|
1185
|
+
session.model = effectiveModel;
|
|
1186
|
+
const nativeKey = `${typ}|${effectiveModel ?? ''}`;
|
|
1018
1187
|
// Answer-quality resilience: a CONFUSED reply — the model talking about
|
|
1019
1188
|
// the tool contract (e.g. apologizing that "the provided example call
|
|
1020
1189
|
// to suggest_followups is incomplete") instead of executing it — never
|
|
@@ -1030,22 +1199,35 @@ export class ChatCommand extends BaseCommand {
|
|
|
1030
1199
|
throw err;
|
|
1031
1200
|
}
|
|
1032
1201
|
};
|
|
1033
|
-
|
|
1202
|
+
/** Mark the model that actually produced this response. */
|
|
1203
|
+
const answered = (resp) => {
|
|
1204
|
+
if (effectiveModel)
|
|
1205
|
+
session.model = effectiveModel;
|
|
1206
|
+
return resp;
|
|
1207
|
+
};
|
|
1208
|
+
// R1 — deterministic transport. A tiny model cannot use a native tool
|
|
1209
|
+
// API, and finding that out by trying cost a 400 on every turn (the
|
|
1210
|
+
// refusal memo is per-call). Unknown families are untouched: they still
|
|
1211
|
+
// try native and fall back, so nothing that works today stops working.
|
|
1212
|
+
if (shouldSkipNativeTools({ model: effectiveModel })) {
|
|
1213
|
+
nativeToolsRejected.add(nativeKey);
|
|
1214
|
+
}
|
|
1215
|
+
if (!nativeToolsRejected.has(nativeKey) && typeof prov.generateTools === 'function' && schemas.length > 0) {
|
|
1034
1216
|
try {
|
|
1035
1217
|
// P4 — stream when the provider supports it AND a sink is wired
|
|
1036
1218
|
// (the dashboard); otherwise the one-shot path with the whole
|
|
1037
1219
|
// content delivered as a single chunk so the typewriter channel
|
|
1038
1220
|
// still receives the answer (appears at once — today's behavior).
|
|
1039
1221
|
if (sink && typeof prov.generateToolsStream === 'function') {
|
|
1040
|
-
const result = await prov.generateToolsStream(messages, schemas, { ...options, model:
|
|
1222
|
+
const result = await prov.generateToolsStream(messages, schemas, { ...options, model: effectiveModel, signal: abort }, sink);
|
|
1041
1223
|
confuseCheck(result.content);
|
|
1042
|
-
return result;
|
|
1224
|
+
return answered(result);
|
|
1043
1225
|
}
|
|
1044
|
-
const result = await prov.generateTools(messages, schemas, { ...options, model:
|
|
1226
|
+
const result = await prov.generateTools(messages, schemas, { ...options, model: effectiveModel, signal: abort });
|
|
1045
1227
|
confuseCheck(result.content);
|
|
1046
1228
|
if (sink && result.content)
|
|
1047
1229
|
sink(result.content);
|
|
1048
|
-
return result;
|
|
1230
|
+
return answered(result);
|
|
1049
1231
|
}
|
|
1050
1232
|
catch (err) {
|
|
1051
1233
|
// S3: a tool-call 400 often carries the model's COMPLETE answer in
|
|
@@ -1064,9 +1246,23 @@ export class ChatCommand extends BaseCommand {
|
|
|
1064
1246
|
const toolCalls = salvaged.followups?.length
|
|
1065
1247
|
? [{ id: 'call_salvage_1', name: 'suggest_followups', arguments: { followups: salvaged.followups } }]
|
|
1066
1248
|
: [];
|
|
1067
|
-
return { content: salvaged.content, toolCalls };
|
|
1249
|
+
return answered({ content: salvaged.content, toolCalls });
|
|
1250
|
+
}
|
|
1251
|
+
// The MODEL itself cannot do native tool calling — Groq answers
|
|
1252
|
+
// 400 "`tool calling` is not supported with this model". That is
|
|
1253
|
+
// not a reason to lose the turn: the loop already ships a transport
|
|
1254
|
+
// that needs no provider tool support, and the system prompt
|
|
1255
|
+
// carries the tool contract for it. Fall THROUGH to it (no throw)
|
|
1256
|
+
// so an otherwise-good model still answers.
|
|
1257
|
+
if (isToolCallingUnsupported(err)) {
|
|
1258
|
+
// Remember it for the REST of this turn so later steps go straight
|
|
1259
|
+
// to the JSON transport instead of re-paying the failing call.
|
|
1260
|
+
nativeToolsRejected.add(nativeKey);
|
|
1261
|
+
logger.warn(' ⚠️ Model does not support native tool calling — retrying this step over the JSON tool transport.');
|
|
1262
|
+
}
|
|
1263
|
+
else {
|
|
1264
|
+
throw err;
|
|
1068
1265
|
}
|
|
1069
|
-
throw err;
|
|
1070
1266
|
}
|
|
1071
1267
|
}
|
|
1072
1268
|
// JSON fallback transport: flatten the thread into one prompt with
|
|
@@ -1076,18 +1272,22 @@ export class ChatCommand extends BaseCommand {
|
|
|
1076
1272
|
let raw;
|
|
1077
1273
|
if (typeof prov.generateStream === 'function') {
|
|
1078
1274
|
const chunks = [];
|
|
1079
|
-
await prov.generateStream(prompt, { ...options, model:
|
|
1275
|
+
await prov.generateStream(prompt, { ...options, model: effectiveModel, signal: abort }, (t) => chunks.push(t));
|
|
1080
1276
|
raw = chunks.join('');
|
|
1081
1277
|
}
|
|
1082
1278
|
else {
|
|
1083
|
-
raw = await prov.generate(prompt, { ...options, model:
|
|
1279
|
+
raw = await prov.generate(prompt, { ...options, model: effectiveModel, signal: abort });
|
|
1084
1280
|
}
|
|
1085
1281
|
const { text, calls } = extractFallbackToolCalls(raw);
|
|
1086
1282
|
confuseCheck(text);
|
|
1087
|
-
return { content: text, toolCalls: calls };
|
|
1283
|
+
return answered({ content: text, toolCalls: calls });
|
|
1088
1284
|
};
|
|
1089
1285
|
try {
|
|
1090
|
-
|
|
1286
|
+
// Same-provider transient retry FIRST (see the helper's contract): a
|
|
1287
|
+
// 503 spike at a shared endpoint must not become a dead run. Failover
|
|
1288
|
+
// only helps if a DIFFERENT provider exists — and it also hides the real
|
|
1289
|
+
// failure from the user while parking a healthy provider for 120s.
|
|
1290
|
+
return await generateWithTransientRetry(() => tryGenerate(session.provider, session.type, session.model), abort, (attempt, err) => logger.warn(` ⏳ ${session.provider.name} transient failure (attempt ${attempt}) — retrying shortly: ${err instanceof Error ? err.message.split('\n')[0] : String(err)}`));
|
|
1091
1291
|
}
|
|
1092
1292
|
catch (err) {
|
|
1093
1293
|
// Auto mode: fail over across the ranked candidates (never stuck).
|
|
@@ -1131,7 +1331,11 @@ export class ChatCommand extends BaseCommand {
|
|
|
1131
1331
|
const resp = await tryGenerate(next.provider, next.type, next.model);
|
|
1132
1332
|
session.type = next.type;
|
|
1133
1333
|
session.provider = next.provider;
|
|
1134
|
-
|
|
1334
|
+
// Keep the model that actually answered: `next.model` is often
|
|
1335
|
+
// undefined ("provider default"), and assigning it here used to
|
|
1336
|
+
// erase the resolved id that tryGenerate just recorded.
|
|
1337
|
+
if (next.model && next.model !== 'default')
|
|
1338
|
+
session.model = next.model;
|
|
1135
1339
|
logger.success(`✅ Auto failover: answered from ${next.provider.name} (${next.model}) after ${firstType} failed`);
|
|
1136
1340
|
return resp;
|
|
1137
1341
|
}
|
|
@@ -1166,10 +1370,22 @@ export class ChatCommand extends BaseCommand {
|
|
|
1166
1370
|
// deliver THAT instead of failing the whole turn. A confusing answer
|
|
1167
1371
|
// still beats an error banner in a messaging app; the confusion is
|
|
1168
1372
|
// now also visible in the chat trace for post-mortem.
|
|
1373
|
+
// The old behavior here — DELIVER the confused reply as if it were the
|
|
1374
|
+
// answer (`return { content: confusedReply }`) — was the worst of both
|
|
1375
|
+
// worlds. It shipped contract meta-talk to the sender ("Sure, I can
|
|
1376
|
+
// help you with suggestions and followups. Please provide me with more
|
|
1377
|
+
// details…") AND marked the turn a SUCCESS, so the loop cached it for
|
|
1378
|
+
// an hour and every retry inside that window replayed the same
|
|
1379
|
+
// deflection. Live evidence: that exact string sat in
|
|
1380
|
+
// ~/.nuvira/cache.json with `model: "default"`.
|
|
1381
|
+
//
|
|
1382
|
+
// Now it stays a FAILURE: rethrow so the tool loop surfaces the
|
|
1383
|
+
// sanitized, user-facing line with `generationFailed: true` (never
|
|
1384
|
+
// cached, never persisted), while the raw reply is preserved in the
|
|
1385
|
+
// log and the reasoning trace for post-mortem.
|
|
1169
1386
|
const confusedReply = err.confusedReply;
|
|
1170
1387
|
if (typeof confusedReply === 'string' && confusedReply.trim()) {
|
|
1171
|
-
logger.warn(
|
|
1172
|
-
return { content: confusedReply, toolCalls: [] };
|
|
1388
|
+
logger.warn(` ⚠️ No alternative model answered — contract-confusion reply suppressed (${confusedReply.length} chars, kept in the trace): ${confusedReply.slice(0, 160)}`);
|
|
1173
1389
|
}
|
|
1174
1390
|
throw err;
|
|
1175
1391
|
}
|
|
@@ -1260,6 +1476,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
1260
1476
|
recordActionFailure({
|
|
1261
1477
|
sessionFailedProviders: this.sessionFailedProviders,
|
|
1262
1478
|
sessionTransientFailedProviders: this.sessionTransientFailedProviders,
|
|
1479
|
+
// Model tracking ON: a rate-limit on one model excludes THAT model and
|
|
1480
|
+
// leaves the provider's siblings routable (per-model RPD/TPM limits).
|
|
1481
|
+
sessionFailedModels: this.sessionFailedModels,
|
|
1263
1482
|
}, providerType, err, this.configManager, { model, action: 'chat', apiKey });
|
|
1264
1483
|
}
|
|
1265
1484
|
async showModelPicker() {
|
|
@@ -1377,37 +1596,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1377
1596
|
// registry may still mark it unavailable (learned from the failure), and
|
|
1378
1597
|
// blindly re-admitting would fail again on the very next message. Recovery
|
|
1379
1598
|
// is discovered in SECONDS (a 1-token spot-check), not by re-failing.
|
|
1380
|
-
//
|
|
1381
|
-
//
|
|
1382
|
-
//
|
|
1383
|
-
//
|
|
1384
|
-
for
|
|
1385
|
-
|
|
1386
|
-
|
|
1387
|
-
|
|
1388
|
-
|
|
1389
|
-
this.sessionTransientFailedProviders.delete(providerType);
|
|
1390
|
-
this.sessionFailedProviders.delete(providerType);
|
|
1391
|
-
try {
|
|
1392
|
-
const registry = getModelRegistry();
|
|
1393
|
-
// Only re-verify when the registry still believes the provider is dead
|
|
1394
|
-
// (unavailable/parked) — a healthy entry means it recovered already.
|
|
1395
|
-
if (!registry.getBlockedProviders().includes(providerType))
|
|
1396
|
-
continue;
|
|
1397
|
-
const desired = getAutoRouter().resolveModel(providerType, 'chat', this.configManager);
|
|
1398
|
-
const outcome = await spotCheckModel(providerType, desired, this.configManager);
|
|
1399
|
-
// 'skipped' = the model was VERIFIED recently (within the spot-check
|
|
1400
|
-
// throttle) — that's healthy, so treat it as a pass too.
|
|
1401
|
-
if (outcome !== 'verified' && outcome !== 'skipped') {
|
|
1402
|
-
// Still down — keep it excluded for another transient window.
|
|
1403
|
-
this.sessionFailedProviders.set(providerType, Date.now() + TRANSIENT_FAILURE_EXCLUSION_MS);
|
|
1404
|
-
this.sessionTransientFailedProviders.add(providerType);
|
|
1405
|
-
}
|
|
1406
|
-
}
|
|
1407
|
-
catch {
|
|
1408
|
-
// Best-effort — re-verification must never break routing.
|
|
1409
|
-
}
|
|
1410
|
-
}
|
|
1599
|
+
// The sweep itself now lives in `learning/provider-revival.ts` so every
|
|
1600
|
+
// entry path shares ONE implementation (chat was previously the only path
|
|
1601
|
+
// that read the transient-failure marker at all — the orchestrator, edit,
|
|
1602
|
+
// execute, plan and resilient-call allocated it and never acted on it, so a
|
|
1603
|
+
// recovered provider stayed excluded for the rest of their runs).
|
|
1604
|
+
await sweepTransientFailures({
|
|
1605
|
+
...collectionRevivalStore(this.sessionFailedProviders, this.sessionTransientFailedProviders),
|
|
1606
|
+
resolveProbeModel: (provider) => getAutoRouter().resolveModel(provider, 'chat', this.configManager),
|
|
1607
|
+
}, this.configManager, { agentType: 'chat' });
|
|
1411
1608
|
const excluded = new Set([
|
|
1412
1609
|
...excludeProviders,
|
|
1413
1610
|
...[...this.sessionFailedProviders.keys()].filter((p) => isActiveExclusion(p)),
|
|
@@ -1424,30 +1621,79 @@ export class ChatCommand extends BaseCommand {
|
|
|
1424
1621
|
catch {
|
|
1425
1622
|
// Best-effort — registry bookkeeping must never break routing
|
|
1426
1623
|
}
|
|
1427
|
-
|
|
1428
|
-
|
|
1429
|
-
|
|
1430
|
-
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
|
|
1624
|
+
// ── DEEP FAILOVER candidate list: {provider, model} PAIRS ──────────────
|
|
1625
|
+
// Several models PER PROVIDER, so a 429 on one model retries the SAME
|
|
1626
|
+
// provider's next-best model before abandoning it. The previous
|
|
1627
|
+
// provider-only list meant chat could only ever reach a single model per
|
|
1628
|
+
// provider no matter how many that provider actually served (free tiers
|
|
1629
|
+
// meter per-model, so the siblings were very often usable).
|
|
1630
|
+
// Model-scoped exclusions come from the SAME predicate the orchestrator's
|
|
1631
|
+
// resilient walk uses. Cross-pipeline persistence is OFF here (chat's own
|
|
1632
|
+
// session accounting is the authority for an interactive turn) and the
|
|
1633
|
+
// registry check is OFF because `resolveWorkingModel` below OWNS per-model
|
|
1634
|
+
// repair; the provider-wide `registryBlocked` pre-filter above already
|
|
1635
|
+
// removes dead providers.
|
|
1636
|
+
const isModelExcluded = createFailoverExclusionFilter({
|
|
1637
|
+
sessionFailedModels: this.sessionFailedModels,
|
|
1638
|
+
crossPipelineMemory: false,
|
|
1639
|
+
registryCheck: false,
|
|
1640
|
+
});
|
|
1641
|
+
const chatCandidates = [];
|
|
1642
|
+
const seenPairs = new Set();
|
|
1643
|
+
const pushCandidate = (prov, mdl) => {
|
|
1644
|
+
if (!prov || excluded.has(prov) || registryBlocked.has(prov))
|
|
1645
|
+
return;
|
|
1646
|
+
const model = mdl && mdl !== 'default' ? mdl : 'default';
|
|
1647
|
+
// Model-scoped session exclusion — only THIS model, never its siblings.
|
|
1648
|
+
if (isModelExcluded(prov, model))
|
|
1649
|
+
return;
|
|
1650
|
+
const key = `${prov}|${model}`;
|
|
1651
|
+
if (seenPairs.has(key))
|
|
1652
|
+
return;
|
|
1653
|
+
// NOTE: deliberately NO registry-usability skip here. `resolveWorkingModel`
|
|
1654
|
+
// below OWNS model health — it repairs a dead/parked model to a live one
|
|
1655
|
+
// on the SAME provider. Filtering the candidate out first would skip the
|
|
1656
|
+
// whole provider and bypass that repair (observed: a stale gemini pin made
|
|
1657
|
+
// chat jump straight to local without ever trying gemini). The chain
|
|
1658
|
+
// already ranks healthy models first, so the parked pick is only ever a
|
|
1659
|
+
// last resort that the repair then fixes.
|
|
1660
|
+
seenPairs.add(key);
|
|
1661
|
+
chatCandidates.push({ provider: prov, model });
|
|
1662
|
+
};
|
|
1663
|
+
// The pool itself is the SAME one the orchestrator/tool/sub-agent path
|
|
1664
|
+
// walks: primary → model-first TIERED pool (same model on other providers,
|
|
1665
|
+
// same tier, escalate/de-escalate, local) → router chain (deep pairs +
|
|
1666
|
+
// reserve) → ranked placeholders → config fallback. Chat used to build a
|
|
1667
|
+
// shallower list here, which is why it reached strictly fewer models than
|
|
1668
|
+
// the orchestrator could.
|
|
1669
|
+
const pool = buildDeepFailoverPool(decision, {
|
|
1670
|
+
taskDescription: message,
|
|
1671
|
+
complexity: decision.complexity,
|
|
1672
|
+
configManager: this.configManager,
|
|
1673
|
+
});
|
|
1674
|
+
for (const c of pool)
|
|
1675
|
+
pushCandidate(c.provider, c.model);
|
|
1676
|
+
// Unique provider list (what callers use for their own failover) — derived
|
|
1677
|
+
// from the pair list so it stays consistent with what is actually tried.
|
|
1678
|
+
const candidates = [...new Set(chatCandidates.map((c) => c.provider))];
|
|
1679
|
+
for (const candidate of chatCandidates) {
|
|
1434
1680
|
try {
|
|
1435
|
-
const resolved = resolveProvider(this.configManager, candidate);
|
|
1681
|
+
const resolved = resolveProvider(this.configManager, candidate.provider);
|
|
1436
1682
|
if (await resolved.provider.isAvailable()) {
|
|
1437
|
-
const desired = candidate
|
|
1438
|
-
?
|
|
1439
|
-
: getAutoRouter().resolveModel(candidate, 'chat', this.configManager);
|
|
1683
|
+
const desired = candidate.model !== 'default'
|
|
1684
|
+
? candidate.model
|
|
1685
|
+
: getAutoRouter().resolveModel(candidate.provider, 'chat', this.configManager);
|
|
1440
1686
|
// Model health: only use models that actually exist on the provider.
|
|
1441
1687
|
// A provider's pinned config.model can be deprecated or a placeholder
|
|
1442
1688
|
// (e.g. gemini-2.0-flash-exp → 404) — repair to a live model.
|
|
1443
|
-
const model = await resolveWorkingModel(resolved.provider, candidate, desired);
|
|
1689
|
+
const model = await resolveWorkingModel(resolved.provider, candidate.provider, desired);
|
|
1444
1690
|
// Record the actually-used route for the dashboard audit trail
|
|
1445
1691
|
recordRoutingDecision({
|
|
1446
1692
|
source: 'chat',
|
|
1447
1693
|
agentType: 'chat',
|
|
1448
1694
|
task: message,
|
|
1449
1695
|
complexity: decision.complexity,
|
|
1450
|
-
provider: candidate,
|
|
1696
|
+
provider: candidate.provider,
|
|
1451
1697
|
model,
|
|
1452
1698
|
score: decision.score,
|
|
1453
1699
|
});
|