agent-nuvira 2.7.3 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (369) hide show
  1. package/dist/agents/agents/writer-tool-calling.d.ts +14 -0
  2. package/dist/agents/agents/writer-tool-calling.d.ts.map +1 -1
  3. package/dist/agents/agents/writer-tool-calling.js +20 -0
  4. package/dist/agents/agents/writer-tool-calling.js.map +1 -1
  5. package/dist/agents/agents/writer.d.ts.map +1 -1
  6. package/dist/agents/agents/writer.js +15 -4
  7. package/dist/agents/agents/writer.js.map +1 -1
  8. package/dist/agents/edit-module.d.ts +7 -0
  9. package/dist/agents/edit-module.d.ts.map +1 -1
  10. package/dist/agents/edit-module.js +21 -6
  11. package/dist/agents/edit-module.js.map +1 -1
  12. package/dist/agents/orchestrator.d.ts +40 -0
  13. package/dist/agents/orchestrator.d.ts.map +1 -1
  14. package/dist/agents/orchestrator.js +116 -27
  15. package/dist/agents/orchestrator.js.map +1 -1
  16. package/dist/agents/tool-calling-agent.d.ts +19 -0
  17. package/dist/agents/tool-calling-agent.d.ts.map +1 -1
  18. package/dist/agents/tool-calling-agent.js +42 -2
  19. package/dist/agents/tool-calling-agent.js.map +1 -1
  20. package/dist/cli/chat.d.ts +55 -1
  21. package/dist/cli/chat.d.ts.map +1 -1
  22. package/dist/cli/chat.js +278 -72
  23. package/dist/cli/chat.js.map +1 -1
  24. package/dist/cli/cli-program.d.ts +6 -0
  25. package/dist/cli/cli-program.d.ts.map +1 -0
  26. package/dist/cli/cli-program.js +242 -0
  27. package/dist/cli/cli-program.js.map +1 -0
  28. package/dist/cli/edit.d.ts.map +1 -1
  29. package/dist/cli/edit.js +18 -3
  30. package/dist/cli/edit.js.map +1 -1
  31. package/dist/cli/execute.d.ts.map +1 -1
  32. package/dist/cli/execute.js +29 -5
  33. package/dist/cli/execute.js.map +1 -1
  34. package/dist/cli/failover-runner.d.ts +13 -0
  35. package/dist/cli/failover-runner.d.ts.map +1 -1
  36. package/dist/cli/failover-runner.js +11 -0
  37. package/dist/cli/failover-runner.js.map +1 -1
  38. package/dist/cli/gateway.d.ts +16 -0
  39. package/dist/cli/gateway.d.ts.map +1 -1
  40. package/dist/cli/gateway.js +108 -2
  41. package/dist/cli/gateway.js.map +1 -1
  42. package/dist/cli/loop-executor.d.ts.map +1 -1
  43. package/dist/cli/loop-executor.js +245 -12
  44. package/dist/cli/loop-executor.js.map +1 -1
  45. package/dist/cli/plan.d.ts.map +1 -1
  46. package/dist/cli/plan.js +20 -6
  47. package/dist/cli/plan.js.map +1 -1
  48. package/dist/cli/process-control.d.ts +10 -0
  49. package/dist/cli/process-control.d.ts.map +1 -1
  50. package/dist/cli/process-control.js +73 -7
  51. package/dist/cli/process-control.js.map +1 -1
  52. package/dist/cli/router.d.ts +16 -5
  53. package/dist/cli/router.d.ts.map +1 -1
  54. package/dist/cli/router.js +31 -206
  55. package/dist/cli/router.js.map +1 -1
  56. package/dist/config/paths.d.ts +23 -0
  57. package/dist/config/paths.d.ts.map +1 -1
  58. package/dist/config/paths.js +30 -0
  59. package/dist/config/paths.js.map +1 -1
  60. package/dist/config/provider-env.d.ts +37 -0
  61. package/dist/config/provider-env.d.ts.map +1 -0
  62. package/dist/config/provider-env.js +86 -0
  63. package/dist/config/provider-env.js.map +1 -0
  64. package/dist/context/history.d.ts.map +1 -1
  65. package/dist/context/history.js +24 -8
  66. package/dist/context/history.js.map +1 -1
  67. package/dist/context/session-recall.d.ts +30 -0
  68. package/dist/context/session-recall.d.ts.map +1 -1
  69. package/dist/context/session-recall.js +61 -0
  70. package/dist/context/session-recall.js.map +1 -1
  71. package/dist/file.js +0 -1
  72. package/dist/file.js.map +1 -1
  73. package/dist/gateway/adapters.d.ts +10 -0
  74. package/dist/gateway/adapters.d.ts.map +1 -1
  75. package/dist/gateway/adapters.js +4 -1
  76. package/dist/gateway/adapters.js.map +1 -1
  77. package/dist/gateway/dedup.d.ts +105 -0
  78. package/dist/gateway/dedup.d.ts.map +1 -0
  79. package/dist/gateway/dedup.js +160 -0
  80. package/dist/gateway/dedup.js.map +1 -0
  81. package/dist/gateway/heartbeat.d.ts +99 -0
  82. package/dist/gateway/heartbeat.d.ts.map +1 -0
  83. package/dist/gateway/heartbeat.js +122 -0
  84. package/dist/gateway/heartbeat.js.map +1 -0
  85. package/dist/gateway/inbox.d.ts +14 -1
  86. package/dist/gateway/inbox.d.ts.map +1 -1
  87. package/dist/gateway/inbox.js.map +1 -1
  88. package/dist/gateway/platform-config.d.ts +5 -0
  89. package/dist/gateway/platform-config.d.ts.map +1 -1
  90. package/dist/gateway/platform-config.js +8 -13
  91. package/dist/gateway/platform-config.js.map +1 -1
  92. package/dist/gateway/registry.d.ts +47 -0
  93. package/dist/gateway/registry.d.ts.map +1 -1
  94. package/dist/gateway/registry.js +227 -21
  95. package/dist/gateway/registry.js.map +1 -1
  96. package/dist/gateway/whatsapp/baileys-bridge.d.ts +11 -1
  97. package/dist/gateway/whatsapp/baileys-bridge.d.ts.map +1 -1
  98. package/dist/gateway/whatsapp/baileys-bridge.js +64 -1
  99. package/dist/gateway/whatsapp/baileys-bridge.js.map +1 -1
  100. package/dist/gateway/whatsapp/bridge.d.ts +4 -2
  101. package/dist/gateway/whatsapp/bridge.d.ts.map +1 -1
  102. package/dist/gateway/whatsapp/bridge.js.map +1 -1
  103. package/dist/index.js +1 -1
  104. package/dist/index.js.map +1 -1
  105. package/dist/inference/factory.d.ts +16 -0
  106. package/dist/inference/factory.d.ts.map +1 -1
  107. package/dist/inference/factory.js +42 -0
  108. package/dist/inference/factory.js.map +1 -1
  109. package/dist/inference/gemini-adapter.d.ts.map +1 -1
  110. package/dist/inference/gemini-adapter.js +3 -0
  111. package/dist/inference/gemini-adapter.js.map +1 -1
  112. package/dist/inference/interface.d.ts +12 -0
  113. package/dist/inference/interface.d.ts.map +1 -1
  114. package/dist/inference/local-adapter.d.ts +2 -0
  115. package/dist/inference/local-adapter.d.ts.map +1 -1
  116. package/dist/inference/local-adapter.js +36 -5
  117. package/dist/inference/local-adapter.js.map +1 -1
  118. package/dist/inference/native-tools.d.ts +8 -0
  119. package/dist/inference/native-tools.d.ts.map +1 -1
  120. package/dist/inference/native-tools.js +14 -3
  121. package/dist/inference/native-tools.js.map +1 -1
  122. package/dist/inference/tool-call-utils.d.ts +92 -7
  123. package/dist/inference/tool-call-utils.d.ts.map +1 -1
  124. package/dist/inference/tool-call-utils.js +203 -13
  125. package/dist/inference/tool-call-utils.js.map +1 -1
  126. package/dist/learning/auto-router.d.ts.map +1 -1
  127. package/dist/learning/auto-router.js +90 -5
  128. package/dist/learning/auto-router.js.map +1 -1
  129. package/dist/learning/context-budget.d.ts +133 -0
  130. package/dist/learning/context-budget.d.ts.map +1 -0
  131. package/dist/learning/context-budget.js +196 -0
  132. package/dist/learning/context-budget.js.map +1 -0
  133. package/dist/learning/cost-tracker.d.ts.map +1 -1
  134. package/dist/learning/cost-tracker.js +20 -8
  135. package/dist/learning/cost-tracker.js.map +1 -1
  136. package/dist/learning/eval-framework.d.ts.map +1 -1
  137. package/dist/learning/eval-framework.js +6 -2
  138. package/dist/learning/eval-framework.js.map +1 -1
  139. package/dist/learning/failure-bookkeeping.d.ts +19 -0
  140. package/dist/learning/failure-bookkeeping.d.ts.map +1 -1
  141. package/dist/learning/failure-bookkeeping.js +67 -2
  142. package/dist/learning/failure-bookkeeping.js.map +1 -1
  143. package/dist/learning/feedback.d.ts.map +1 -1
  144. package/dist/learning/feedback.js +19 -8
  145. package/dist/learning/feedback.js.map +1 -1
  146. package/dist/learning/model-harness.d.ts +76 -0
  147. package/dist/learning/model-harness.d.ts.map +1 -0
  148. package/dist/learning/model-harness.js +119 -0
  149. package/dist/learning/model-harness.js.map +1 -0
  150. package/dist/learning/model-selection.d.ts +15 -0
  151. package/dist/learning/model-selection.d.ts.map +1 -1
  152. package/dist/learning/model-selection.js +27 -1
  153. package/dist/learning/model-selection.js.map +1 -1
  154. package/dist/learning/provider-fallback.d.ts +12 -0
  155. package/dist/learning/provider-fallback.d.ts.map +1 -1
  156. package/dist/learning/provider-fallback.js +74 -0
  157. package/dist/learning/provider-fallback.js.map +1 -1
  158. package/dist/learning/provider-revival.d.ts +106 -0
  159. package/dist/learning/provider-revival.d.ts.map +1 -0
  160. package/dist/learning/provider-revival.js +154 -0
  161. package/dist/learning/provider-revival.js.map +1 -0
  162. package/dist/learning/resilient-call.d.ts.map +1 -1
  163. package/dist/learning/resilient-call.js +48 -3
  164. package/dist/learning/resilient-call.js.map +1 -1
  165. package/dist/nlu/conversation-gate.d.ts +37 -0
  166. package/dist/nlu/conversation-gate.d.ts.map +1 -1
  167. package/dist/nlu/conversation-gate.js +49 -0
  168. package/dist/nlu/conversation-gate.js.map +1 -1
  169. package/dist/nlu/schema.d.ts +2 -2
  170. package/dist/observability/dag-bridge.d.ts +69 -0
  171. package/dist/observability/dag-bridge.d.ts.map +1 -0
  172. package/dist/observability/dag-bridge.js +56 -0
  173. package/dist/observability/dag-bridge.js.map +1 -0
  174. package/dist/observability/event-bus.d.ts +5 -10
  175. package/dist/observability/event-bus.d.ts.map +1 -1
  176. package/dist/observability/event-bus.js +10 -42
  177. package/dist/observability/event-bus.js.map +1 -1
  178. package/dist/skills/execution-audit.d.ts.map +1 -1
  179. package/dist/skills/execution-audit.js +17 -5
  180. package/dist/skills/execution-audit.js.map +1 -1
  181. package/dist/skills/sandbox-executor.d.ts +12 -1
  182. package/dist/skills/sandbox-executor.d.ts.map +1 -1
  183. package/dist/skills/sandbox-executor.js +28 -6
  184. package/dist/skills/sandbox-executor.js.map +1 -1
  185. package/dist/skills/secret-capture.d.ts +45 -1
  186. package/dist/skills/secret-capture.d.ts.map +1 -1
  187. package/dist/skills/secret-capture.js +82 -19
  188. package/dist/skills/secret-capture.js.map +1 -1
  189. package/dist/skills/skill-env-inventory.d.ts +79 -0
  190. package/dist/skills/skill-env-inventory.d.ts.map +1 -0
  191. package/dist/skills/skill-env-inventory.js +167 -0
  192. package/dist/skills/skill-env-inventory.js.map +1 -0
  193. package/dist/skills/skill-executor.d.ts +2 -0
  194. package/dist/skills/skill-executor.d.ts.map +1 -1
  195. package/dist/skills/skill-executor.js +19 -24
  196. package/dist/skills/skill-executor.js.map +1 -1
  197. package/dist/tools/ask-user.d.ts +14 -1
  198. package/dist/tools/ask-user.d.ts.map +1 -1
  199. package/dist/tools/ask-user.js +25 -1
  200. package/dist/tools/ask-user.js.map +1 -1
  201. package/dist/tools/followup-utils.d.ts +64 -0
  202. package/dist/tools/followup-utils.d.ts.map +1 -0
  203. package/dist/tools/followup-utils.js +158 -0
  204. package/dist/tools/followup-utils.js.map +1 -0
  205. package/dist/tools/loop-skill-hint.d.ts.map +1 -1
  206. package/dist/tools/loop-skill-hint.js +22 -1
  207. package/dist/tools/loop-skill-hint.js.map +1 -1
  208. package/dist/tools/memory-tools.d.ts +35 -0
  209. package/dist/tools/memory-tools.d.ts.map +1 -1
  210. package/dist/tools/memory-tools.js +76 -0
  211. package/dist/tools/memory-tools.js.map +1 -1
  212. package/dist/tools/pipeline-tool.d.ts.map +1 -1
  213. package/dist/tools/pipeline-tool.js +8 -3
  214. package/dist/tools/pipeline-tool.js.map +1 -1
  215. package/dist/tools/registry.d.ts +25 -12
  216. package/dist/tools/registry.d.ts.map +1 -1
  217. package/dist/tools/registry.js +31 -6
  218. package/dist/tools/registry.js.map +1 -1
  219. package/dist/tools/skill-tool.d.ts.map +1 -1
  220. package/dist/tools/skill-tool.js +52 -5
  221. package/dist/tools/skill-tool.js.map +1 -1
  222. package/dist/tools/tool-loop.d.ts +63 -1
  223. package/dist/tools/tool-loop.d.ts.map +1 -1
  224. package/dist/tools/tool-loop.js +305 -66
  225. package/dist/tools/tool-loop.js.map +1 -1
  226. package/dist/tools/toolsets.d.ts +19 -3
  227. package/dist/tools/toolsets.d.ts.map +1 -1
  228. package/dist/tools/toolsets.js +21 -5
  229. package/dist/tools/toolsets.js.map +1 -1
  230. package/dist/utils/env.d.ts.map +1 -1
  231. package/dist/utils/env.js +9 -13
  232. package/dist/utils/env.js.map +1 -1
  233. package/dist/web-dashboard/chat-console.d.ts +13 -1
  234. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  235. package/dist/web-dashboard/chat-console.js +16 -2
  236. package/dist/web-dashboard/chat-console.js.map +1 -1
  237. package/dist/web-dashboard/hub-data.d.ts +2 -0
  238. package/dist/web-dashboard/hub-data.d.ts.map +1 -1
  239. package/dist/web-dashboard/hub-data.js +4 -0
  240. package/dist/web-dashboard/hub-data.js.map +1 -1
  241. package/dist/web-dashboard/server.d.ts +6 -0
  242. package/dist/web-dashboard/server.d.ts.map +1 -1
  243. package/dist/web-dashboard/server.js +192 -3
  244. package/dist/web-dashboard/server.js.map +1 -1
  245. package/dist/web-dashboard/src/types.d.ts +31 -1
  246. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  247. package/package.json +11 -1
  248. package/src/web-dashboard/public/assets/index-56yPnN_m.js +207 -0
  249. package/src/web-dashboard/public/assets/index-56yPnN_m.js.map +1 -0
  250. package/src/web-dashboard/public/assets/{index-C507EUWf.css → index-Cyd6tIew.css} +1 -1
  251. package/src/web-dashboard/public/index.html +2 -2
  252. package/dist/example.d.ts +0 -1
  253. package/dist/example.d.ts.map +0 -1
  254. package/dist/example.js +0 -3
  255. package/dist/example.js.map +0 -1
  256. package/dist/fresh.d.ts +0 -2
  257. package/dist/fresh.d.ts.map +0 -1
  258. package/dist/fresh.js +0 -2
  259. package/dist/fresh.js.map +0 -1
  260. package/dist/mcp/mcp-dashboard-oauth.d.ts +0 -94
  261. package/dist/mcp/mcp-dashboard-oauth.d.ts.map +0 -1
  262. package/dist/mcp/mcp-dashboard-oauth.js +0 -140
  263. package/dist/mcp/mcp-dashboard-oauth.js.map +0 -1
  264. package/dist/memory/background-sync.d.ts +0 -107
  265. package/dist/memory/background-sync.d.ts.map +0 -1
  266. package/dist/memory/background-sync.js +0 -242
  267. package/dist/memory/background-sync.js.map +0 -1
  268. package/dist/memory/cross-session.d.ts +0 -68
  269. package/dist/memory/cross-session.d.ts.map +0 -1
  270. package/dist/memory/cross-session.js +0 -153
  271. package/dist/memory/cross-session.js.map +0 -1
  272. package/dist/memory/drift-detector.d.ts +0 -76
  273. package/dist/memory/drift-detector.d.ts.map +0 -1
  274. package/dist/memory/drift-detector.js +0 -216
  275. package/dist/memory/drift-detector.js.map +0 -1
  276. package/dist/memory/enhanced-manager.d.ts +0 -105
  277. package/dist/memory/enhanced-manager.d.ts.map +0 -1
  278. package/dist/memory/enhanced-manager.js +0 -280
  279. package/dist/memory/enhanced-manager.js.map +0 -1
  280. package/dist/memory/session-extraction.d.ts +0 -93
  281. package/dist/memory/session-extraction.d.ts.map +0 -1
  282. package/dist/memory/session-extraction.js +0 -290
  283. package/dist/memory/session-extraction.js.map +0 -1
  284. package/dist/memory/sqlite-store.d.ts +0 -208
  285. package/dist/memory/sqlite-store.d.ts.map +0 -1
  286. package/dist/memory/sqlite-store.js +0 -556
  287. package/dist/memory/sqlite-store.js.map +0 -1
  288. package/dist/skills/daytona-executor.d.ts +0 -78
  289. package/dist/skills/daytona-executor.d.ts.map +0 -1
  290. package/dist/skills/daytona-executor.js +0 -256
  291. package/dist/skills/daytona-executor.js.map +0 -1
  292. package/dist/skills/modal-executor.d.ts +0 -81
  293. package/dist/skills/modal-executor.d.ts.map +0 -1
  294. package/dist/skills/modal-executor.js +0 -285
  295. package/dist/skills/modal-executor.js.map +0 -1
  296. package/dist/sync/inventory.d.ts +0 -1
  297. package/dist/sync/inventory.d.ts.map +0 -1
  298. package/dist/sync/inventory.js +0 -3
  299. package/dist/sync/inventory.js.map +0 -1
  300. package/dist/sync/validate.d.ts +0 -1
  301. package/dist/sync/validate.d.ts.map +0 -1
  302. package/dist/sync/validate.js +0 -3
  303. package/dist/sync/validate.js.map +0 -1
  304. package/dist/test.d.ts +0 -2
  305. package/dist/test.d.ts.map +0 -1
  306. package/dist/test.js +0 -3
  307. package/dist/test.js.map +0 -1
  308. package/dist/tools/browser-tool.d.ts +0 -206
  309. package/dist/tools/browser-tool.d.ts.map +0 -1
  310. package/dist/tools/browser-tool.js +0 -726
  311. package/dist/tools/browser-tool.js.map +0 -1
  312. package/dist/tools/delegate-tool.d.ts +0 -109
  313. package/dist/tools/delegate-tool.d.ts.map +0 -1
  314. package/dist/tools/delegate-tool.js +0 -185
  315. package/dist/tools/delegate-tool.js.map +0 -1
  316. package/dist/tools/delegation-state.d.ts +0 -87
  317. package/dist/tools/delegation-state.d.ts.map +0 -1
  318. package/dist/tools/delegation-state.js +0 -278
  319. package/dist/tools/delegation-state.js.map +0 -1
  320. package/dist/tools/desktop-ui.d.ts +0 -148
  321. package/dist/tools/desktop-ui.d.ts.map +0 -1
  322. package/dist/tools/desktop-ui.js +0 -486
  323. package/dist/tools/desktop-ui.js.map +0 -1
  324. package/dist/tools/image-video-tool.d.ts +0 -88
  325. package/dist/tools/image-video-tool.d.ts.map +0 -1
  326. package/dist/tools/image-video-tool.js +0 -276
  327. package/dist/tools/image-video-tool.js.map +0 -1
  328. package/dist/tools/kanban-cron-tools.d.ts +0 -92
  329. package/dist/tools/kanban-cron-tools.d.ts.map +0 -1
  330. package/dist/tools/kanban-cron-tools.js +0 -197
  331. package/dist/tools/kanban-cron-tools.js.map +0 -1
  332. package/dist/tools/messaging-tool.d.ts +0 -127
  333. package/dist/tools/messaging-tool.d.ts.map +0 -1
  334. package/dist/tools/messaging-tool.js +0 -314
  335. package/dist/tools/messaging-tool.js.map +0 -1
  336. package/dist/tools/modality/generic-caller.d.ts +0 -43
  337. package/dist/tools/modality/generic-caller.d.ts.map +0 -1
  338. package/dist/tools/modality/generic-caller.js +0 -254
  339. package/dist/tools/modality/generic-caller.js.map +0 -1
  340. package/dist/tools/modality/modality-catalog.d.ts +0 -134
  341. package/dist/tools/modality/modality-catalog.d.ts.map +0 -1
  342. package/dist/tools/modality/modality-catalog.js +0 -445
  343. package/dist/tools/modality/modality-catalog.js.map +0 -1
  344. package/dist/tools/modality/tool-router.d.ts +0 -81
  345. package/dist/tools/modality/tool-router.d.ts.map +0 -1
  346. package/dist/tools/modality/tool-router.js +0 -84
  347. package/dist/tools/modality/tool-router.js.map +0 -1
  348. package/dist/tools/security-tools.d.ts +0 -107
  349. package/dist/tools/security-tools.d.ts.map +0 -1
  350. package/dist/tools/security-tools.js +0 -264
  351. package/dist/tools/security-tools.js.map +0 -1
  352. package/dist/tools/ssh-tool.d.ts +0 -130
  353. package/dist/tools/ssh-tool.d.ts.map +0 -1
  354. package/dist/tools/ssh-tool.js +0 -328
  355. package/dist/tools/ssh-tool.js.map +0 -1
  356. package/dist/tools/tts-tool.d.ts +0 -91
  357. package/dist/tools/tts-tool.d.ts.map +0 -1
  358. package/dist/tools/tts-tool.js +0 -406
  359. package/dist/tools/tts-tool.js.map +0 -1
  360. package/dist/utils/shell-detection.d.ts +0 -124
  361. package/dist/utils/shell-detection.d.ts.map +0 -1
  362. package/dist/utils/shell-detection.js +0 -305
  363. package/dist/utils/shell-detection.js.map +0 -1
  364. package/dist/utils/windows-paths.d.ts +0 -185
  365. package/dist/utils/windows-paths.d.ts.map +0 -1
  366. package/dist/utils/windows-paths.js +0 -472
  367. package/dist/utils/windows-paths.js.map +0 -1
  368. package/src/web-dashboard/public/assets/index-C4frng1Q.js +0 -207
  369. package/src/web-dashboard/public/assets/index-C4frng1Q.js.map +0 -1
package/dist/cli/chat.js CHANGED
@@ -16,12 +16,13 @@ import { getMemoryManager } from '../memory/manager.js';
16
16
  import { logger } from '../utils/logger.js';
17
17
  import { printOrchestrationResult } from './execute.js';
18
18
  import { applyActiveModel } from './model.js';
19
- import { getProviderFallback, classifyFallbackError, isRetryableError, recordRegistrySuccess } from '../learning/provider-fallback.js';
20
- import { recordActionFailure, TRANSIENT_FAILURE_EXCLUSION_MS } from '../learning/failure-bookkeeping.js';
19
+ import { getProviderFallback, classifyFallbackError, isRetryableError, isTransientForRetry, recordRegistrySuccess } from '../learning/provider-fallback.js';
20
+ import { recordActionFailure } from '../learning/failure-bookkeeping.js';
21
+ import { resolveThreadBudgetChars } from '../learning/context-budget.js';
21
22
  import { getAutoRouter, isAutoModel, isAutoProvider } from '../learning/auto-router.js';
22
23
  import { estimateTokens } from '../learning/cost-tracker.js';
23
24
  import { getModelRegistry } from '../learning/model-registry.js';
24
- import { refreshModelRegistry, spotCheckModel } from '../inference/model-probe.js';
25
+ import { refreshModelRegistry } from '../inference/model-probe.js';
25
26
  import { recordRoutingDecision } from '../learning/routing-history.js';
26
27
  import { shouldConfirmFailover, promptFailoverChoice } from './failover-prompt.js';
27
28
  import { buildAutoResolveOptions } from '../learning/resolve-options.js';
@@ -31,15 +32,19 @@ import { PlanStore } from '../tools/plan-store.js';
31
32
  import { withLogCorrelation } from '../enterprise/log.js';
32
33
  import { recordMetricTime, getMetrics } from '../enterprise/metrics.js';
33
34
  import { resolveDispatch } from '../nlu/actions.js';
34
- import { isConversationalQuestion, hasCodingAction } from '../nlu/conversation-gate.js';
35
+ import { hasCodingAction, resolveAskKind } from '../nlu/conversation-gate.js';
35
36
  import { runToolLoop, extractFallbackToolCalls } from '../tools/tool-loop.js';
36
- import { looksLikeConfusedScaffoldingReply } from '../inference/tool-call-utils.js';
37
+ import { looksLikeConfusedScaffoldingReply, toUserFacingGenerationError, isToolCallingUnsupported, stripToolCallArtifacts, } from '../inference/tool-call-utils.js';
37
38
  import { beginTrace, endTrace, recordStep } from '../learning/reasoning-trace.js';
38
39
  import { getLoopExposureMode } from '../tools/toolsets.js';
40
+ import { resolveModelHarnessProfile, shouldSkipNativeTools } from '../learning/model-harness.js';
41
+ import { resolveAdapterDefault } from '../learning/model-selection.js';
39
42
  import { buildLoopProjectContext } from '../tools/loop-project-context.js';
43
+ import { sweepTransientFailures, collectionRevivalStore } from '../learning/provider-revival.js';
40
44
  import { analyzeComplexity } from '../learning/hybrid-router.js';
41
45
  import { routingCacheSignature, withRoutingCache } from '../learning/routing-cache.js';
42
46
  import { getTool, TOOL_CONTRACT_JSON } from '../tools/registry.js';
47
+ import { buildFollowupContinuationPrompt, isSuggestedFollowup, } from '../tools/followup-utils.js';
43
48
  // S2/S3 — the shared tool-call reliability helpers (salvage failed_generation,
44
49
  // compact fallback schemas). One copy for every tool-calling surface, not
45
50
  // chat-private (execute/plan/… inherit the fix).
@@ -168,7 +173,11 @@ export function resolvePipelineDispatch(parsed, opts) {
168
173
  // NLU alone would misread it as chat ("how do I add JWT auth?" → explain
169
174
  // → chat, but the user wants the auth added).
170
175
  if (opts?.text) {
171
- if (isConversationalQuestion(opts.text)) {
176
+ // ONE shared rule for every surface (see resolveAskKind): a genuine
177
+ // question never dispatches; a coding verb in command position always
178
+ // does. Keeping the gateway on this same function is what stops the two
179
+ // from disagreeing about the same ask.
180
+ if (resolveAskKind(opts.text, parsed) === 'chat') {
172
181
  return { dispatch: false, needConfirm: false };
173
182
  }
174
183
  if (hasCodingAction(opts.text)) {
@@ -217,6 +226,41 @@ export async function runDeveloperMode(goal, configManager, options) {
217
226
  * (rules act only as the no-model fallback in the
218
227
  * caller, never to skip the loop).
219
228
  */
229
+ /**
230
+ * Backoff schedule for a SAME-provider retry on a transient failure. Two extra
231
+ * attempts, deliberately short: a capacity spike at a shared endpoint clears in
232
+ * seconds, and the user is waiting in the foreground. Long/looping retries belong
233
+ * to the background runners, not the interactive turn.
234
+ */
235
+ export const TRANSIENT_RETRY_DELAYS_MS = [1_000, 3_000];
236
+ /**
237
+ * Run one provider attempt, retrying transient failures against the SAME
238
+ * provider before giving up.
239
+ *
240
+ * Why this exists next to the failover walk rather than inside it: the walk
241
+ * needs a DIFFERENT provider to exist, and it books the failure against the one
242
+ * that just failed. Verified live — a single configured provider plus a Gemini
243
+ * 503 meant no retry at all, the circuit breaker parked the provider for 120s,
244
+ * and the agent degraded to editing with zero gathered context. A transient
245
+ * spike must cost a few seconds, not the whole task.
246
+ *
247
+ * Never retries: non-transient classes (auth, rate-limit, model/harness faults),
248
+ * a cancelled turn, or once the schedule is exhausted.
249
+ */
250
+ export async function generateWithTransientRetry(attempt, signal, onRetry) {
251
+ for (let i = 0;; i += 1) {
252
+ try {
253
+ return await attempt();
254
+ }
255
+ catch (err) {
256
+ const canRetry = i < TRANSIENT_RETRY_DELAYS_MS.length && isTransientForRetry(err);
257
+ if (!canRetry || signal?.aborted)
258
+ throw err;
259
+ onRetry?.(i + 1, err);
260
+ await new Promise((resolve) => setTimeout(resolve, TRANSIENT_RETRY_DELAYS_MS[i]));
261
+ }
262
+ }
263
+ }
220
264
  function buildToolSystemPrompt(parsed) {
221
265
  return [
222
266
  "You are Nuvira, Agent-Nuvira's AI agent. You code, create, write, analyze, and automate — anything the user needs. You identify as Nuvira (never 'Buff').",
@@ -316,6 +360,34 @@ export class ChatCommand extends BaseCommand {
316
360
  const requestedModel = opts.model && opts.model !== 'default' ? opts.model : undefined;
317
361
  const activeOpts = applyActiveModel({ provider: opts.provider, model: requestedModel });
318
362
  const mergedOpts = { ...opts, provider: activeOpts.provider, model: activeOpts.model };
363
+ // When the CALLER supplies neither a provider nor a model — the dashboard
364
+ // chat console and the gateway chat engine both call answerOnce with just a
365
+ // message — fall back to the CONFIGURED defaultProvider instead of letting
366
+ // resolveProvider() land on one fixed provider.
367
+ //
368
+ // Why this matters (live, 2026-09-20): the shipped config default is
369
+ // `defaultProvider: "auto"`, but auto mode was only ever enabled by an
370
+ // EXPLICIT 'auto' from the flags or the `nuvira model switch` state. The
371
+ // dashboard passes neither, so every dashboard turn silently ran on ONE
372
+ // concrete provider with NO auto-failover walk (the non-auto path only
373
+ // walks `fallback.providers`, which ships empty). One 400/429/timeout then
374
+ // ended the turn with the canned "the language model was unavailable"
375
+ // line — while the CLI answered the identical prompt, because the CLI
376
+ // resolves auto from the same config. Same engine, two modes: this closes
377
+ // that gap.
378
+ // Only the AUTO default changes behavior: a concrete `defaultProvider` is a
379
+ // deliberate pin and keeps the non-auto path exactly as it is.
380
+ if (!mergedOpts.provider && !mergedOpts.model) {
381
+ try {
382
+ const cfg = this.configManager.getAll();
383
+ if (isAutoProvider(cfg.defaultProvider)) {
384
+ mergedOpts.provider = cfg.defaultProvider;
385
+ }
386
+ }
387
+ catch {
388
+ // Best-effort — an unreadable config leaves the previous behavior.
389
+ }
390
+ }
319
391
  let autoMode = isAutoModel(mergedOpts.model) || isAutoProvider(mergedOpts.provider);
320
392
  let { type, provider } = autoMode
321
393
  ? await this.getProvider({})
@@ -351,7 +423,7 @@ export class ChatCommand extends BaseCommand {
351
423
  }
352
424
  const parsed = parseRequestSync(message);
353
425
  const dispatchDecision = resolvePipelineDispatch(parsed, { dev: opts.dev, text: message });
354
- const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal });
426
+ const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal, continuation: opts.continuation });
355
427
  // No-model fallback: the tool loop could not generate a single response
356
428
  // AND the rules assessed a high-confidence pipeline intent — run the
357
429
  // pipeline directly (rules decide only when the model is unavailable; the
@@ -364,10 +436,7 @@ export class ChatCommand extends BaseCommand {
364
436
  return { content: r.result?.summary ?? '', followups: [], provider: type, model };
365
437
  }
366
438
  // E3b: strip raw suggest_followups JSON embedded in content by the model
367
- const cleanContent = (answer.content || '')
368
- .replace(/\n?\*?\s*\{\s*"tool"\s*:\s*"suggest_followups"[\s\S]*$/, '')
369
- .replace(/\n?\*?\s*<function=suggest_followups[\s\S]*<\/function>/g, '')
370
- .trim();
439
+ const cleanContent = stripToolCallArtifacts(answer.content || '');
371
440
  return {
372
441
  content: cleanContent,
373
442
  followups: answer.followups ?? [],
@@ -500,15 +569,19 @@ export class ChatCommand extends BaseCommand {
500
569
  // Continue to interactive mode (don't return)
501
570
  logger.info('');
502
571
  }
503
- else
504
572
  // Ordering: the ANSWER is always printed first, then followups — the
505
573
  // user asked for the content, not a menu. On a real terminal the
506
574
  // followups are SELECTABLE: picking a number runs that followup as the
507
575
  // next turn (conversation threaded), pressing Enter continues interactively.
508
576
  // Non-TTY (scripts/CI/pipes) keeps the current print-and-exit behavior
509
577
  // so automation is never blocked by a prompt.
510
- if (answer.content.trim()) {
511
- console.log('\n' + answer.content + '\n');
578
+ //
579
+ // Print parity with the dashboard console + gateway: the answer must not
580
+ // carry the model's tool-call artifacts (a raw trailing suggest_followups
581
+ // JSON blob, or the empty ```json fence the fallback transport leaves
582
+ // behind) — see stripToolCallArtifacts.
583
+ else if (answer.content.trim()) {
584
+ console.log('\n' + stripToolCallArtifacts(answer.content) + '\n');
512
585
  }
513
586
  if (!process.stdin.isTTY) {
514
587
  await this.renderFollowups(answer.followups ?? [], false);
@@ -522,10 +595,15 @@ export class ChatCommand extends BaseCommand {
522
595
  const picked = await this.renderFollowups(singleAnswer.followups ?? [], true);
523
596
  if (!picked)
524
597
  break;
525
- const next = await this.runChatAnswer(picked, history, { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parseRequestSync(picked));
526
- if (next.content.trim()) {
527
- console.log('\n' + next.content + '\n');
528
- history.push({ role: 'assistant', content: next.content });
598
+ const next = await this.runChatAnswer(picked, history, { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parseRequestSync(picked),
599
+ // P5 — a picked followup continues the previous execution.
600
+ { continuation: true });
601
+ const nextText = stripToolCallArtifacts(next.content);
602
+ if (nextText) {
603
+ console.log('\n' + nextText + '\n');
604
+ // NOTE: no history.push here — runChatAnswer already recorded the
605
+ // assistant turn. The old duplicate push gave every subsequent turn
606
+ // TWO copies of the previous answer (relevance noise).
529
607
  }
530
608
  singleAnswer = next;
531
609
  }
@@ -547,8 +625,13 @@ export class ChatCommand extends BaseCommand {
547
625
  await maybeRunBackgroundDuties(this.configManager).catch(() => { });
548
626
  }
549
627
  let pendingMessage;
628
+ // P5 — the followups the agent just suggested, so a message that MATCHES
629
+ // one of them (a clicked chip, or the user re-typing it on the gateway) is
630
+ // recognised as a continuation of the previous execution.
631
+ let lastFollowups = [];
550
632
  while (true) {
551
633
  // E3b: a chosen follow-up recommendation becomes the next message.
634
+ const pickedFollowup = pendingMessage !== undefined;
552
635
  const message = pendingMessage ?? (await this.readMultiLineInput('You:'));
553
636
  pendingMessage = undefined;
554
637
  if (!message)
@@ -602,7 +685,10 @@ export class ChatCommand extends BaseCommand {
602
685
  const parsed = recordMetricTime('rule.parse.ms', () => parseRequestSync(message));
603
686
  const dispatchDecision = recordMetricTime('rule.dispatch.ms', () => resolvePipelineDispatch(parsed, { dev: this.devModeAuto, text: message }));
604
687
  const session = { type, provider, model: effectiveModel };
605
- const answer = await withLogCorrelation({ sessionId: chatSessionId }, () => recordMetricTime('llm.answer.ms', () => this.runChatAnswer(message, history, session, options || {}, cacheEnabled, { auto: autoMode }, parsed)));
688
+ const answer = await withLogCorrelation({ sessionId: chatSessionId }, () => recordMetricTime('llm.answer.ms', () => this.runChatAnswer(message, history, session, options || {}, cacheEnabled, { auto: autoMode }, parsed,
689
+ // P5 — a picked followup (or a typed one that matches the last
690
+ // suggestions) is a continuation, not a fresh independent request.
691
+ { continuation: pickedFollowup || isSuggestedFollowup(message, lastFollowups) })));
606
692
  // No-model fallback: the tool loop could not generate a single response
607
693
  // AND the rules assessed a high-confidence pipeline intent — run the
608
694
  // pipeline directly (rules decide only when the model is unavailable).
@@ -623,7 +709,8 @@ export class ChatCommand extends BaseCommand {
623
709
  if (answer.content.trim()) {
624
710
  console.log('\n' + answer.content + '\n');
625
711
  }
626
- const followupPrompt = await this.renderFollowups(answer.followups ?? [], true);
712
+ lastFollowups = answer.followups ?? [];
713
+ const followupPrompt = await this.renderFollowups(lastFollowups, true);
627
714
  if (followupPrompt) {
628
715
  pendingMessage = followupPrompt;
629
716
  }
@@ -703,9 +790,10 @@ export class ChatCommand extends BaseCommand {
703
790
  async runChatAnswer(message, history, session, options, cacheEnabled, mode, parsed, ctxOverrides) {
704
791
  // Cache check first (same as the legacy path).
705
792
  const cache = getCache();
793
+ const cacheModel = this.cacheModelFor(session);
706
794
  if (cacheEnabled) {
707
795
  try {
708
- const cachedResult = await cache.get(message, session.model ?? 'default', session.type);
796
+ const cachedResult = await cache.get(message, cacheModel, session.type);
709
797
  if (cachedResult) {
710
798
  // NOTE: the cached answer is NOT printed here — the caller prints
711
799
  // content AFTER runChatAnswer returns (answer-first ordering). A
@@ -807,7 +895,11 @@ export class ChatCommand extends BaseCommand {
807
895
  role: (h.role === 'assistant' ? 'assistant' : 'user'),
808
896
  content: h.content,
809
897
  })),
810
- { role: 'user', content: message },
898
+ // P5 — a picked followup reaches the model WITH the continuation marker
899
+ // (the raw text stays in history), so "add a day in Hanoi" is resolved
900
+ // against the plan the previous turn just produced instead of being read
901
+ // as a brand-new request.
902
+ { role: 'user', content: ctxOverrides?.continuation ? buildFollowupContinuationPrompt(message) : message },
811
903
  ];
812
904
  // I3: one artifact session per TURN — every tool
813
905
  // deliverable in this turn lands in the same store folder.
@@ -940,13 +1032,26 @@ export class ChatCommand extends BaseCommand {
940
1032
  }
941
1033
  };
942
1034
  try {
1035
+ // R1 — the harness follows the MODEL, not only the config: config says
1036
+ // what this deployment prefers, the profile decides what this model can
1037
+ // actually use (a 0.5B local model must not get a 120B's surface).
1038
+ const harness = resolveModelHarnessProfile({
1039
+ model: session.model,
1040
+ configExposure: getLoopExposureMode(this.configManager),
1041
+ });
943
1042
  result = await runToolLoop({
944
1043
  messages: thread,
945
1044
  context: toolContext,
946
1045
  maxSteps: 16,
947
- // Tiered tool exposure — config-gated (tools.loopExposure), default
948
- // 'all' = unchanged behavior until Phase 0 evals justify the flip.
949
- toolExposure: getLoopExposureMode(this.configManager),
1046
+ // Model-window-aware thread budget: a 1M-token model keeps its window
1047
+ // instead of being trimmed to the fixed ~50K-token default. Undefined
1048
+ // (unknown window) leaves the tool-loop default untouched.
1049
+ threadBudgetChars: resolveThreadBudgetChars({ provider: session.type, model: session.model }),
1050
+ // Tiered tool exposure — the tiered (core) set starts at ~16 schemas
1051
+ // (~3.8K tokens/step); 'all' hands over the full set (~17K tokens/step)
1052
+ // and is now additionally gated on the model having the context for it.
1053
+ toolExposure: harness.exposure,
1054
+ maxParallelReads: harness.maxParallelReads,
950
1055
  onToken: ctxOverrides?.onToken,
951
1056
  signal: ctxOverrides?.signal,
952
1057
  deps: {
@@ -977,7 +1082,9 @@ export class ChatCommand extends BaseCommand {
977
1082
  logger.error(String(err));
978
1083
  endTrace(chatTraceId, false);
979
1084
  result = {
980
- content: `I ran into a problem: ${err instanceof Error ? err.message : String(err)}`,
1085
+ // Sanitized on purpose: this content is delivered verbatim by every
1086
+ // surface (CLI print, dashboard bubble, gateway send).
1087
+ content: toUserFacingGenerationError(err),
981
1088
  followups: [],
982
1089
  toolCalls: [],
983
1090
  steps: 0,
@@ -994,7 +1101,11 @@ export class ChatCommand extends BaseCommand {
994
1101
  if (result.content.trim() && !result.generationFailed && !result.cancelled) {
995
1102
  if (cacheEnabled) {
996
1103
  try {
997
- await cache.set(message, result.content, session.model ?? 'default', session.type);
1104
+ // Keyed by the model that ACTUALLY answered (tryGenerate records it
1105
+ // on success), so a weak model's reply is never replayed as a strong
1106
+ // model's. `cacheModel` is the pre-flight fallback for the paths that
1107
+ // never resolve one (e.g. a cached-hit turn).
1108
+ await cache.set(message, result.content, this.cacheModelFor(session) || cacheModel, session.type);
998
1109
  }
999
1110
  catch {
1000
1111
  // Best-effort.
@@ -1026,13 +1137,83 @@ export class ChatCommand extends BaseCommand {
1026
1137
  * broken provider never crashes the turn (it answers from the next working
1027
1138
  * candidate, exactly like the legacy generation block).
1028
1139
  */
1140
+ /**
1141
+ * The model id used in the response-cache key.
1142
+ *
1143
+ * NEVER returns the `'default'` sentinel (or an empty string). Keying the
1144
+ * cache on `'default'` — which is what `session.model ?? 'default'` did —
1145
+ * collapsed EVERY model of a provider into a single entry (observed live:
1146
+ * `cache.json` held `provider: gemini, model: "default"`). Two consequences,
1147
+ * both real: an answer produced by a weak model was replayed as though a
1148
+ * strong one had written it, and switching `nuvira model switch` could never
1149
+ * take effect for a message already cached. Falls back to the provider's
1150
+ * effective model, then to a provider-qualified marker so distinct providers
1151
+ * still never collide.
1152
+ */
1153
+ cacheModelFor(session) {
1154
+ if (session.model && session.model !== 'default')
1155
+ return session.model;
1156
+ try {
1157
+ const providers = this.configManager.getAll().providers;
1158
+ return resolveAdapterDefault(session.type, providers?.[session.type]?.model) ?? `${session.type}:unresolved`;
1159
+ }
1160
+ catch {
1161
+ return `${session.type}:unresolved`;
1162
+ }
1163
+ }
1029
1164
  buildToolCallModel(message, session, options, mode, onToken, signal) {
1030
1165
  return async (messages, schemas, stepOnToken, stepSignal) => {
1031
1166
  // The effective token sink: the caller's stream wins; when a step-level
1032
1167
  // sink is also given (loop passthrough) they are the same channel.
1033
1168
  const sink = stepOnToken ?? onToken;
1034
1169
  const abort = stepSignal ?? signal;
1170
+ /**
1171
+ * The CONCRETE model an attempt will use.
1172
+ *
1173
+ * `session.model` is undefined or the `'default'` sentinel whenever the
1174
+ * router picks a provider but no single model (which is the common auto
1175
+ * case). That value used to flow into four places at once — the provider
1176
+ * request, the reasoning trace, registry telemetry and the response
1177
+ * cache key — so traces read `model: unknown`, the literal `default`
1178
+ * reached provider APIs (`The model \`default\` does not exist`, observed
1179
+ * live), and EVERY model of a provider shared one cache entry (a bad
1180
+ * answer produced by a weak model was then replayed as if it came from a
1181
+ * good one). Resolving here keeps all four on the same real model id.
1182
+ * Falls back to undefined only when nothing can be resolved, which leaves
1183
+ * the adapter's own last-resort resolution in charge.
1184
+ */
1185
+ /**
1186
+ * Per-turn memo of candidates that REJECTED native tool calling, keyed by
1187
+ * `provider|model`. Once a model has answered "tool calling is not
1188
+ * supported", it can never start supporting it within this turn, so
1189
+ * re-issuing the native call is pure waste — the live execute run paid 13
1190
+ * failing native requests (one per step) before each fell back to the
1191
+ * JSON transport, burning the provider's rate limit for nothing.
1192
+ * Keyed per candidate on purpose: a DIFFERENT model that does support
1193
+ * native tools must still get them, so a failover re-enables the fast
1194
+ * path automatically.
1195
+ */
1196
+ const nativeToolsRejected = new Set();
1197
+ const resolveEffectiveModel = (providerType, requested) => {
1198
+ if (requested && requested !== 'default')
1199
+ return requested;
1200
+ try {
1201
+ const providers = this.configManager.getAll().providers;
1202
+ return resolveAdapterDefault(providerType, providers?.[providerType]?.model);
1203
+ }
1204
+ catch {
1205
+ return undefined;
1206
+ }
1207
+ };
1035
1208
  const tryGenerate = async (prov, typ, mdl) => {
1209
+ // One resolution per attempt — the request, the trace, the telemetry
1210
+ // and the cache key all read the same value (see above).
1211
+ const effectiveModel = resolveEffectiveModel(typ, mdl);
1212
+ // Record the ATTEMPTED model immediately so a failed step's trace and
1213
+ // telemetry name the model that failed, instead of "unknown".
1214
+ if (effectiveModel)
1215
+ session.model = effectiveModel;
1216
+ const nativeKey = `${typ}|${effectiveModel ?? ''}`;
1036
1217
  // Answer-quality resilience: a CONFUSED reply — the model talking about
1037
1218
  // the tool contract (e.g. apologizing that "the provided example call
1038
1219
  // to suggest_followups is incomplete") instead of executing it — never
@@ -1048,22 +1229,35 @@ export class ChatCommand extends BaseCommand {
1048
1229
  throw err;
1049
1230
  }
1050
1231
  };
1051
- if (typeof prov.generateTools === 'function' && schemas.length > 0) {
1232
+ /** Mark the model that actually produced this response. */
1233
+ const answered = (resp) => {
1234
+ if (effectiveModel)
1235
+ session.model = effectiveModel;
1236
+ return resp;
1237
+ };
1238
+ // R1 — deterministic transport. A tiny model cannot use a native tool
1239
+ // API, and finding that out by trying cost a 400 on every turn (the
1240
+ // refusal memo is per-call). Unknown families are untouched: they still
1241
+ // try native and fall back, so nothing that works today stops working.
1242
+ if (shouldSkipNativeTools({ model: effectiveModel })) {
1243
+ nativeToolsRejected.add(nativeKey);
1244
+ }
1245
+ if (!nativeToolsRejected.has(nativeKey) && typeof prov.generateTools === 'function' && schemas.length > 0) {
1052
1246
  try {
1053
1247
  // P4 — stream when the provider supports it AND a sink is wired
1054
1248
  // (the dashboard); otherwise the one-shot path with the whole
1055
1249
  // content delivered as a single chunk so the typewriter channel
1056
1250
  // still receives the answer (appears at once — today's behavior).
1057
1251
  if (sink && typeof prov.generateToolsStream === 'function') {
1058
- const result = await prov.generateToolsStream(messages, schemas, { ...options, model: mdl, signal: abort }, sink);
1252
+ const result = await prov.generateToolsStream(messages, schemas, { ...options, model: effectiveModel, signal: abort }, sink);
1059
1253
  confuseCheck(result.content);
1060
- return result;
1254
+ return answered(result);
1061
1255
  }
1062
- const result = await prov.generateTools(messages, schemas, { ...options, model: mdl, signal: abort });
1256
+ const result = await prov.generateTools(messages, schemas, { ...options, model: effectiveModel, signal: abort });
1063
1257
  confuseCheck(result.content);
1064
1258
  if (sink && result.content)
1065
1259
  sink(result.content);
1066
- return result;
1260
+ return answered(result);
1067
1261
  }
1068
1262
  catch (err) {
1069
1263
  // S3: a tool-call 400 often carries the model's COMPLETE answer in
@@ -1082,9 +1276,23 @@ export class ChatCommand extends BaseCommand {
1082
1276
  const toolCalls = salvaged.followups?.length
1083
1277
  ? [{ id: 'call_salvage_1', name: 'suggest_followups', arguments: { followups: salvaged.followups } }]
1084
1278
  : [];
1085
- return { content: salvaged.content, toolCalls };
1279
+ return answered({ content: salvaged.content, toolCalls });
1280
+ }
1281
+ // The MODEL itself cannot do native tool calling — Groq answers
1282
+ // 400 "`tool calling` is not supported with this model". That is
1283
+ // not a reason to lose the turn: the loop already ships a transport
1284
+ // that needs no provider tool support, and the system prompt
1285
+ // carries the tool contract for it. Fall THROUGH to it (no throw)
1286
+ // so an otherwise-good model still answers.
1287
+ if (isToolCallingUnsupported(err)) {
1288
+ // Remember it for the REST of this turn so later steps go straight
1289
+ // to the JSON transport instead of re-paying the failing call.
1290
+ nativeToolsRejected.add(nativeKey);
1291
+ logger.warn(' ⚠️ Model does not support native tool calling — retrying this step over the JSON tool transport.');
1292
+ }
1293
+ else {
1294
+ throw err;
1086
1295
  }
1087
- throw err;
1088
1296
  }
1089
1297
  }
1090
1298
  // JSON fallback transport: flatten the thread into one prompt with
@@ -1094,18 +1302,22 @@ export class ChatCommand extends BaseCommand {
1094
1302
  let raw;
1095
1303
  if (typeof prov.generateStream === 'function') {
1096
1304
  const chunks = [];
1097
- await prov.generateStream(prompt, { ...options, model: mdl, signal: abort }, (t) => chunks.push(t));
1305
+ await prov.generateStream(prompt, { ...options, model: effectiveModel, signal: abort }, (t) => chunks.push(t));
1098
1306
  raw = chunks.join('');
1099
1307
  }
1100
1308
  else {
1101
- raw = await prov.generate(prompt, { ...options, model: mdl, signal: abort });
1309
+ raw = await prov.generate(prompt, { ...options, model: effectiveModel, signal: abort });
1102
1310
  }
1103
1311
  const { text, calls } = extractFallbackToolCalls(raw);
1104
1312
  confuseCheck(text);
1105
- return { content: text, toolCalls: calls };
1313
+ return answered({ content: text, toolCalls: calls });
1106
1314
  };
1107
1315
  try {
1108
- return await tryGenerate(session.provider, session.type, session.model);
1316
+ // Same-provider transient retry FIRST (see the helper's contract): a
1317
+ // 503 spike at a shared endpoint must not become a dead run. Failover
1318
+ // only helps if a DIFFERENT provider exists — and it also hides the real
1319
+ // failure from the user while parking a healthy provider for 120s.
1320
+ return await generateWithTransientRetry(() => tryGenerate(session.provider, session.type, session.model), abort, (attempt, err) => logger.warn(` ⏳ ${session.provider.name} transient failure (attempt ${attempt}) — retrying shortly: ${err instanceof Error ? err.message.split('\n')[0] : String(err)}`));
1109
1321
  }
1110
1322
  catch (err) {
1111
1323
  // Auto mode: fail over across the ranked candidates (never stuck).
@@ -1149,7 +1361,11 @@ export class ChatCommand extends BaseCommand {
1149
1361
  const resp = await tryGenerate(next.provider, next.type, next.model);
1150
1362
  session.type = next.type;
1151
1363
  session.provider = next.provider;
1152
- session.model = next.model;
1364
+ // Keep the model that actually answered: `next.model` is often
1365
+ // undefined ("provider default"), and assigning it here used to
1366
+ // erase the resolved id that tryGenerate just recorded.
1367
+ if (next.model && next.model !== 'default')
1368
+ session.model = next.model;
1153
1369
  logger.success(`✅ Auto failover: answered from ${next.provider.name} (${next.model}) after ${firstType} failed`);
1154
1370
  return resp;
1155
1371
  }
@@ -1184,10 +1400,22 @@ export class ChatCommand extends BaseCommand {
1184
1400
  // deliver THAT instead of failing the whole turn. A confusing answer
1185
1401
  // still beats an error banner in a messaging app; the confusion is
1186
1402
  // now also visible in the chat trace for post-mortem.
1403
+ // The old behavior here — DELIVER the confused reply as if it were the
1404
+ // answer (`return { content: confusedReply }`) — was the worst of both
1405
+ // worlds. It shipped contract meta-talk to the sender ("Sure, I can
1406
+ // help you with suggestions and followups. Please provide me with more
1407
+ // details…") AND marked the turn a SUCCESS, so the loop cached it for
1408
+ // an hour and every retry inside that window replayed the same
1409
+ // deflection. Live evidence: that exact string sat in
1410
+ // ~/.nuvira/cache.json with `model: "default"`.
1411
+ //
1412
+ // Now it stays a FAILURE: rethrow so the tool loop surfaces the
1413
+ // sanitized, user-facing line with `generationFailed: true` (never
1414
+ // cached, never persisted), while the raw reply is preserved in the
1415
+ // log and the reasoning trace for post-mortem.
1187
1416
  const confusedReply = err.confusedReply;
1188
1417
  if (typeof confusedReply === 'string' && confusedReply.trim()) {
1189
- logger.warn(' ⚠️ No alternative model answered — delivering the raw reply (contract-confusion fallback).');
1190
- return { content: confusedReply, toolCalls: [] };
1418
+ logger.warn(` ⚠️ No alternative model answered — contract-confusion reply suppressed (${confusedReply.length} chars, kept in the trace): ${confusedReply.slice(0, 160)}`);
1191
1419
  }
1192
1420
  throw err;
1193
1421
  }
@@ -1398,37 +1626,15 @@ export class ChatCommand extends BaseCommand {
1398
1626
  // registry may still mark it unavailable (learned from the failure), and
1399
1627
  // blindly re-admitting would fail again on the very next message. Recovery
1400
1628
  // is discovered in SECONDS (a 1-token spot-check), not by re-failing.
1401
- // NOTE: iterate a SNAPSHOT — the loop mutates the set (delete + re-add),
1402
- // and Set iteration can revisit a re-added key, double-spot-checking.
1403
- // Bounded: at most one spot-check per provider per 60s (the exclusion is
1404
- // re-armed on failure), and only for registry-blocked providers.
1405
- for (const providerType of [...this.sessionTransientFailedProviders]) {
1406
- const expiresAt = this.sessionFailedProviders.get(providerType);
1407
- // Skip still-active exclusions and already-cleared providers.
1408
- if (expiresAt !== undefined && expiresAt > exclusionTime)
1409
- continue;
1410
- this.sessionTransientFailedProviders.delete(providerType);
1411
- this.sessionFailedProviders.delete(providerType);
1412
- try {
1413
- const registry = getModelRegistry();
1414
- // Only re-verify when the registry still believes the provider is dead
1415
- // (unavailable/parked) — a healthy entry means it recovered already.
1416
- if (!registry.getBlockedProviders().includes(providerType))
1417
- continue;
1418
- const desired = getAutoRouter().resolveModel(providerType, 'chat', this.configManager);
1419
- const outcome = await spotCheckModel(providerType, desired, this.configManager);
1420
- // 'skipped' = the model was VERIFIED recently (within the spot-check
1421
- // throttle) — that's healthy, so treat it as a pass too.
1422
- if (outcome !== 'verified' && outcome !== 'skipped') {
1423
- // Still down — keep it excluded for another transient window.
1424
- this.sessionFailedProviders.set(providerType, Date.now() + TRANSIENT_FAILURE_EXCLUSION_MS);
1425
- this.sessionTransientFailedProviders.add(providerType);
1426
- }
1427
- }
1428
- catch {
1429
- // Best-effort — re-verification must never break routing.
1430
- }
1431
- }
1629
+ // The sweep itself now lives in `learning/provider-revival.ts` so every
1630
+ // entry path shares ONE implementation (chat was previously the only path
1631
+ // that read the transient-failure marker at all — the orchestrator, edit,
1632
+ // execute, plan and resilient-call allocated it and never acted on it, so a
1633
+ // recovered provider stayed excluded for the rest of their runs).
1634
+ await sweepTransientFailures({
1635
+ ...collectionRevivalStore(this.sessionFailedProviders, this.sessionTransientFailedProviders),
1636
+ resolveProbeModel: (provider) => getAutoRouter().resolveModel(provider, 'chat', this.configManager),
1637
+ }, this.configManager, { agentType: 'chat' });
1432
1638
  const excluded = new Set([
1433
1639
  ...excludeProviders,
1434
1640
  ...[...this.sessionFailedProviders.keys()].filter((p) => isActiveExclusion(p)),