cct-cli 0.7.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (325) hide show
  1. calc_terminal/__init__.py +14 -0
  2. calc_terminal/__main__.py +14 -0
  3. calc_terminal/activity.py +1334 -0
  4. calc_terminal/agent.py +3387 -0
  5. calc_terminal/agent_runtime.py +519 -0
  6. calc_terminal/ai_context.py +447 -0
  7. calc_terminal/ai_modes.py +752 -0
  8. calc_terminal/ai_personalization.py +286 -0
  9. calc_terminal/ai_preview_feedback.py +213 -0
  10. calc_terminal/aicore.py +2572 -0
  11. calc_terminal/anim.py +367 -0
  12. calc_terminal/app.py +3685 -0
  13. calc_terminal/art.py +639 -0
  14. calc_terminal/atomsim.py +368 -0
  15. calc_terminal/attachments.py +743 -0
  16. calc_terminal/benchmark_system.py +414 -0
  17. calc_terminal/browser/__init__.py +36 -0
  18. calc_terminal/browser/browser_state.py +346 -0
  19. calc_terminal/browser/devserver.py +176 -0
  20. calc_terminal/browser/engine.py +494 -0
  21. calc_terminal/browser/navigation.py +84 -0
  22. calc_terminal/browser/preview.py +429 -0
  23. calc_terminal/browser/preview_entry.py +95 -0
  24. calc_terminal/browser/project_detector.py +144 -0
  25. calc_terminal/browser/server.py +449 -0
  26. calc_terminal/browser/state.py +75 -0
  27. calc_terminal/browser/watcher.py +99 -0
  28. calc_terminal/browser_gui/__init__.py +1 -0
  29. calc_terminal/browser_gui/__main__.py +3 -0
  30. calc_terminal/browser_gui/launcher.py +173 -0
  31. calc_terminal/browser_gui/playwright_browser.py +117 -0
  32. calc_terminal/browser_gui/qt_browser.py +1501 -0
  33. calc_terminal/browser_gui/webview_browser.py +57 -0
  34. calc_terminal/capabilities/__init__.py +35 -0
  35. calc_terminal/capabilities/adapters/__init__.py +33 -0
  36. calc_terminal/capabilities/adapters/bioinformatics.py +204 -0
  37. calc_terminal/capabilities/adapters/browser_adapter.py +205 -0
  38. calc_terminal/capabilities/adapters/filesystem.py +206 -0
  39. calc_terminal/capabilities/adapters/git_adapter.py +202 -0
  40. calc_terminal/capabilities/adapters/jupyter_adapter.py +138 -0
  41. calc_terminal/capabilities/adapters/ml_frameworks.py +158 -0
  42. calc_terminal/capabilities/adapters/platforms.py +200 -0
  43. calc_terminal/capabilities/adapters/python_exec.py +93 -0
  44. calc_terminal/capabilities/adapters/quantum_adapter.py +150 -0
  45. calc_terminal/capabilities/adapters/scientific_comp.py +123 -0
  46. calc_terminal/capabilities/adapters/structural_bio.py +161 -0
  47. calc_terminal/capabilities/adapters/terminal.py +99 -0
  48. calc_terminal/capabilities/bus.py +178 -0
  49. calc_terminal/capabilities/discovery.py +207 -0
  50. calc_terminal/capabilities/schema.py +221 -0
  51. calc_terminal/cat.ico +0 -0
  52. calc_terminal/cat_browser.py +2018 -0
  53. calc_terminal/chat_store.py +703 -0
  54. calc_terminal/cli.py +1178 -0
  55. calc_terminal/code_editor.py +640 -0
  56. calc_terminal/collaboration.py +723 -0
  57. calc_terminal/commands_data.py +139 -0
  58. calc_terminal/compatibility_engine.py +352 -0
  59. calc_terminal/compute/__init__.py +31 -0
  60. calc_terminal/compute/fabric.py +350 -0
  61. calc_terminal/config.py +227 -0
  62. calc_terminal/core/__init__.py +41 -0
  63. calc_terminal/core/checkpoint.py +156 -0
  64. calc_terminal/core/input/__init__.py +45 -0
  65. calc_terminal/core/mode_registry.py +300 -0
  66. calc_terminal/core/project_graph.py +172 -0
  67. calc_terminal/core/recovery.py +129 -0
  68. calc_terminal/core/security_layer.py +112 -0
  69. calc_terminal/core/task_graph.py +202 -0
  70. calc_terminal/core/unified_runtime.py +184 -0
  71. calc_terminal/core/verification.py +257 -0
  72. calc_terminal/customization.py +1566 -0
  73. calc_terminal/derivations.py +153 -0
  74. calc_terminal/device_control.py +263 -0
  75. calc_terminal/diagnostics/__init__.py +27 -0
  76. calc_terminal/diagnostics/doctor_engine.py +382 -0
  77. calc_terminal/diagnostics/self_test.py +247 -0
  78. calc_terminal/doctor.py +519 -0
  79. calc_terminal/easter_eggs.py +274 -0
  80. calc_terminal/editor/__init__.py +1 -0
  81. calc_terminal/editor/actions.py +263 -0
  82. calc_terminal/editor/commands.py +160 -0
  83. calc_terminal/editor/shortcuts.py +226 -0
  84. calc_terminal/engine.py +259 -0
  85. calc_terminal/errors.py +120 -0
  86. calc_terminal/event_stream.py +146 -0
  87. calc_terminal/eventbus.py +133 -0
  88. calc_terminal/extensions.py +733 -0
  89. calc_terminal/fallback_cli.py +1321 -0
  90. calc_terminal/first_run.py +265 -0
  91. calc_terminal/fomoji_auth.py +1043 -0
  92. calc_terminal/formulas.py +82 -0
  93. calc_terminal/fs_cache.py +121 -0
  94. calc_terminal/fs_watcher.py +277 -0
  95. calc_terminal/game.py +193 -0
  96. calc_terminal/gen1.py +5 -0
  97. calc_terminal/generators.py +245 -0
  98. calc_terminal/gestures/__init__.py +42 -0
  99. calc_terminal/gestures/bindings.py +175 -0
  100. calc_terminal/gestures/manager.py +477 -0
  101. calc_terminal/goodbye.py +363 -0
  102. calc_terminal/gpu3d.py +290 -0
  103. calc_terminal/graphs.py +358 -0
  104. calc_terminal/hardware_analyzer.py +440 -0
  105. calc_terminal/host/__init__.py +30 -0
  106. calc_terminal/host/browser_manager.py +187 -0
  107. calc_terminal/host/desktop.py +1386 -0
  108. calc_terminal/host/launcher.py +395 -0
  109. calc_terminal/host/terminal.py +279 -0
  110. calc_terminal/identity.py +216 -0
  111. calc_terminal/input/__init__.py +54 -0
  112. calc_terminal/input/capabilities.py +258 -0
  113. calc_terminal/input/focus.py +87 -0
  114. calc_terminal/input/gestures.py +64 -0
  115. calc_terminal/input/pointer.py +114 -0
  116. calc_terminal/input/touch.py +345 -0
  117. calc_terminal/keys.py +84 -0
  118. calc_terminal/live_automation.py +165 -0
  119. calc_terminal/mathtext.py +433 -0
  120. calc_terminal/mcp.py +386 -0
  121. calc_terminal/memory.py +337 -0
  122. calc_terminal/memory_v2.py +479 -0
  123. calc_terminal/metrics.py +333 -0
  124. calc_terminal/mode_detection.py +146 -0
  125. calc_terminal/model.py +2431 -0
  126. calc_terminal/model_router.py +665 -0
  127. calc_terminal/models/__init__.py +0 -0
  128. calc_terminal/models/active_state.py +187 -0
  129. calc_terminal/models/dynamic_registry.py +584 -0
  130. calc_terminal/models/manager.py +781 -0
  131. calc_terminal/models/model_metadata.json +3526 -0
  132. calc_terminal/models/profiles.py +194 -0
  133. calc_terminal/models/registry.py +265 -0
  134. calc_terminal/models/schema.py +197 -0
  135. calc_terminal/models/validator.py +287 -0
  136. calc_terminal/models/verification_engine.py +368 -0
  137. calc_terminal/native_picker.py +215 -0
  138. calc_terminal/ollama_catalog.py +279 -0
  139. calc_terminal/ollama_download.py +233 -0
  140. calc_terminal/orchestrator.py +304 -0
  141. calc_terminal/package_research.py +322 -0
  142. calc_terminal/packages.py +1024 -0
  143. calc_terminal/pc_specs.py +116 -0
  144. calc_terminal/permissions.py +334 -0
  145. calc_terminal/pet.py +106 -0
  146. calc_terminal/pipeline.py +505 -0
  147. calc_terminal/platform/__init__.py +491 -0
  148. calc_terminal/platform/desktop.py +491 -0
  149. calc_terminal/platform/web.py +781 -0
  150. calc_terminal/preview/__init__.py +1 -0
  151. calc_terminal/preview/dev_server.py +303 -0
  152. calc_terminal/preview/diagnostics.py +131 -0
  153. calc_terminal/preview/live_reload.py +66 -0
  154. calc_terminal/preview/manager.py +129 -0
  155. calc_terminal/project_stats.py +209 -0
  156. calc_terminal/projects.py +328 -0
  157. calc_terminal/providers/__init__.py +0 -0
  158. calc_terminal/providers/adapters/__init__.py +80 -0
  159. calc_terminal/providers/adapters/anthropic_adapter.py +127 -0
  160. calc_terminal/providers/adapters/base.py +106 -0
  161. calc_terminal/providers/adapters/chinese_adapters.py +420 -0
  162. calc_terminal/providers/adapters/gemini_adapter.py +101 -0
  163. calc_terminal/providers/adapters/ollama_adapter.py +83 -0
  164. calc_terminal/providers/adapters/openai_adapter.py +159 -0
  165. calc_terminal/providers/adapters/other_adapters.py +246 -0
  166. calc_terminal/providers/anthropic_provider.py +172 -0
  167. calc_terminal/providers/auto_update.py +416 -0
  168. calc_terminal/providers/base_provider.py +105 -0
  169. calc_terminal/providers/discovery_manager.py +207 -0
  170. calc_terminal/providers/gemini_provider.py +178 -0
  171. calc_terminal/providers/lifecycle.py +767 -0
  172. calc_terminal/providers/ollama_adapter.py +707 -0
  173. calc_terminal/providers/openai_provider.py +254 -0
  174. calc_terminal/providers/provider_manager.py +1827 -0
  175. calc_terminal/providers/providers.json +4075 -0
  176. calc_terminal/reactionsim.py +279 -0
  177. calc_terminal/registry.py +337 -0
  178. calc_terminal/report.py +162 -0
  179. calc_terminal/research/__init__.py +45 -0
  180. calc_terminal/research/artifact_intel.py +126 -0
  181. calc_terminal/research/data_lineage.py +123 -0
  182. calc_terminal/research/experiment_ledger.py +303 -0
  183. calc_terminal/research/reproducibility.py +131 -0
  184. calc_terminal/resilience/__init__.py +47 -0
  185. calc_terminal/resilience/agent_state.py +121 -0
  186. calc_terminal/resilience/capability_matcher.py +174 -0
  187. calc_terminal/resilience/circuit_breaker.py +158 -0
  188. calc_terminal/resilience/failover_engine.py +230 -0
  189. calc_terminal/resilience/health_monitor.py +192 -0
  190. calc_terminal/resilience/ollama_adapter.py +125 -0
  191. calc_terminal/resilience/orchestrator.py +312 -0
  192. calc_terminal/resilience/types.py +134 -0
  193. calc_terminal/sandbox.py +98 -0
  194. calc_terminal/scires.py +558 -0
  195. calc_terminal/security_scanner.py +126 -0
  196. calc_terminal/session.py +294 -0
  197. calc_terminal/sim3d.py +206 -0
  198. calc_terminal/solver.py +276 -0
  199. calc_terminal/sound.py +127 -0
  200. calc_terminal/task_reports.py +287 -0
  201. calc_terminal/terminal_host.py +201 -0
  202. calc_terminal/terminal_identity.py +411 -0
  203. calc_terminal/test_ai_mode_reliability.py +344 -0
  204. calc_terminal/test_browser.py +368 -0
  205. calc_terminal/test_code_editor_upgrade.py +485 -0
  206. calc_terminal/test_customization.py +1148 -0
  207. calc_terminal/test_customization_ui.py +612 -0
  208. calc_terminal/test_dynamic_registry.py +304 -0
  209. calc_terminal/test_extensions.py +436 -0
  210. calc_terminal/test_overhaul.py +557 -0
  211. calc_terminal/test_project_detect.py +255 -0
  212. calc_terminal/test_root_cause_fix.py +527 -0
  213. calc_terminal/test_stability.py +532 -0
  214. calc_terminal/test_terminal_identity.py +132 -0
  215. calc_terminal/test_v079_speed.py +460 -0
  216. calc_terminal/theme.py +1107 -0
  217. calc_terminal/timeline.py +139 -0
  218. calc_terminal/todos.py +246 -0
  219. calc_terminal/tool_call_normalizer.py +419 -0
  220. calc_terminal/tui.py +104 -0
  221. calc_terminal/ui/__init__.py +8 -0
  222. calc_terminal/ui/activity_panel.py +231 -0
  223. calc_terminal/ui/activity_stream_panel.py +238 -0
  224. calc_terminal/ui/animations.py +122 -0
  225. calc_terminal/ui/app.py +7271 -0
  226. calc_terminal/ui/attach_panel.py +597 -0
  227. calc_terminal/ui/attachments.py +424 -0
  228. calc_terminal/ui/backup_panel.py +810 -0
  229. calc_terminal/ui/browser_shell.py +887 -0
  230. calc_terminal/ui/cat_agent.py +357 -0
  231. calc_terminal/ui/chats_panel.py +899 -0
  232. calc_terminal/ui/command_palette.py +125 -0
  233. calc_terminal/ui/command_palette_modal.py +166 -0
  234. calc_terminal/ui/composer.py +1141 -0
  235. calc_terminal/ui/context_menu.py +197 -0
  236. calc_terminal/ui/conversation.py +1435 -0
  237. calc_terminal/ui/customization_panel.py +1229 -0
  238. calc_terminal/ui/dashboard.py +404 -0
  239. calc_terminal/ui/design_system.py +557 -0
  240. calc_terminal/ui/diff_panel.py +213 -0
  241. calc_terminal/ui/editor.py +2102 -0
  242. calc_terminal/ui/empty_state.py +302 -0
  243. calc_terminal/ui/events.py +487 -0
  244. calc_terminal/ui/extensions_panel.py +815 -0
  245. calc_terminal/ui/footer.py +166 -0
  246. calc_terminal/ui/gestures_panel.py +383 -0
  247. calc_terminal/ui/goodbye_screen.py +100 -0
  248. calc_terminal/ui/header.py +1034 -0
  249. calc_terminal/ui/help_panel.py +254 -0
  250. calc_terminal/ui/live_activities.py +914 -0
  251. calc_terminal/ui/mcp_panel.py +570 -0
  252. calc_terminal/ui/memory_center.py +524 -0
  253. calc_terminal/ui/mode_colors_panel.py +525 -0
  254. calc_terminal/ui/nav_screens.py +747 -0
  255. calc_terminal/ui/ollama_panel.py +536 -0
  256. calc_terminal/ui/palette.py +221 -0
  257. calc_terminal/ui/permission_panel.py +269 -0
  258. calc_terminal/ui/personalization_panel.py +517 -0
  259. calc_terminal/ui/personalize_center.py +1568 -0
  260. calc_terminal/ui/preview_panel.py +441 -0
  261. calc_terminal/ui/resizers.py +402 -0
  262. calc_terminal/ui/sidebar.py +1285 -0
  263. calc_terminal/ui/statusbar.py +168 -0
  264. calc_terminal/ui/theme_css.py +1396 -0
  265. calc_terminal/ui/thinking.py +226 -0
  266. calc_terminal/ui/timeline_panel.py +102 -0
  267. calc_terminal/ui/todo_panel.py +193 -0
  268. calc_terminal/ui/viewport.py +136 -0
  269. calc_terminal/ui/vision_panel.py +489 -0
  270. calc_terminal/ui/welcome_modal.py +343 -0
  271. calc_terminal/ui/widgets.py +160 -0
  272. calc_terminal/ui/workspace.py +831 -0
  273. calc_terminal/viewers/__init__.py +1 -0
  274. calc_terminal/viewers/document_viewer.py +252 -0
  275. calc_terminal/viewers/image_viewer.py +241 -0
  276. calc_terminal/viewers/pdf_viewer.py +203 -0
  277. calc_terminal/viewers/presentation_viewer.py +164 -0
  278. calc_terminal/viewers/registry.py +120 -0
  279. calc_terminal/viewers/spreadsheet_viewer.py +204 -0
  280. calc_terminal/vision/__init__.py +89 -0
  281. calc_terminal/vision/analysis.py +194 -0
  282. calc_terminal/vision/annotations.py +297 -0
  283. calc_terminal/vision/capture.py +171 -0
  284. calc_terminal/vision/context.py +231 -0
  285. calc_terminal/vision/correlation.py +169 -0
  286. calc_terminal/vision/cursor.py +258 -0
  287. calc_terminal/vision/events.py +66 -0
  288. calc_terminal/vision/frame_pipeline.py +259 -0
  289. calc_terminal/vision/priority.py +218 -0
  290. calc_terminal/vision/provider.py +180 -0
  291. calc_terminal/vision/safety.py +149 -0
  292. calc_terminal/vision/session.py +281 -0
  293. calc_terminal/vision/verify.py +162 -0
  294. calc_terminal/vision.py +514 -0
  295. calc_terminal/vscode_integration.py +113 -0
  296. calc_terminal/web/__init__.py +8 -0
  297. calc_terminal/web/cat_runtime.py +710 -0
  298. calc_terminal/web/server.py +2891 -0
  299. calc_terminal/web/static/css/app.css +3152 -0
  300. calc_terminal/web/static/icons/badge-72.png +0 -0
  301. calc_terminal/web/static/icons/cat.ico +0 -0
  302. calc_terminal/web/static/icons/icon-128.png +0 -0
  303. calc_terminal/web/static/icons/icon-144.png +0 -0
  304. calc_terminal/web/static/icons/icon-152.png +0 -0
  305. calc_terminal/web/static/icons/icon-192.png +0 -0
  306. calc_terminal/web/static/icons/icon-384.png +0 -0
  307. calc_terminal/web/static/icons/icon-512.png +0 -0
  308. calc_terminal/web/static/icons/icon-72.png +0 -0
  309. calc_terminal/web/static/icons/icon-96.png +0 -0
  310. calc_terminal/web/static/icons/icon.svg +34 -0
  311. calc_terminal/web/static/icons/new-project.png +0 -0
  312. calc_terminal/web/static/icons/open-project.png +0 -0
  313. calc_terminal/web/static/index.html +734 -0
  314. calc_terminal/web/static/js/app.js +2403 -0
  315. calc_terminal/web/static/manifest.json +88 -0
  316. calc_terminal/web/static/sw.js +230 -0
  317. calc_terminal/workflow_engine.py +769 -0
  318. calc_terminal/workspace.py +593 -0
  319. calc_terminal/workspace_index.py +385 -0
  320. cct_cli-0.7.9.0.dist-info/METADATA +210 -0
  321. cct_cli-0.7.9.0.dist-info/RECORD +325 -0
  322. cct_cli-0.7.9.0.dist-info/WHEEL +5 -0
  323. cct_cli-0.7.9.0.dist-info/entry_points.txt +4 -0
  324. cct_cli-0.7.9.0.dist-info/licenses/LICENSE +21 -0
  325. cct_cli-0.7.9.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,2572 @@
1
+ """
2
+ AI Integration Core for CCT [BETA].
3
+ Supports local Ollama or API-based models (OpenAI/Claude style) for chemistry.
4
+ """
5
+
6
+ if __name__ == '__main__':
7
+ print("This is a library file and is not meant to be run directly.")
8
+ print("Please run 'python main.py' or 'python model.py' from the project root directory.")
9
+ import sys
10
+ sys.exit(1)
11
+
12
+ import os
13
+ import re
14
+ import json
15
+ import time
16
+ import sys
17
+ import base64
18
+ import zlib
19
+ import logging
20
+ import threading
21
+ import uuid
22
+ from urllib.parse import unquote, parse_qs, urlparse
23
+ from html import unescape as _html_unescape
24
+ try:
25
+ import requests
26
+ _HAS_REQUESTS = True
27
+ except ImportError:
28
+ requests = None
29
+ _HAS_REQUESTS = False
30
+
31
+ from . import theme
32
+ from . import identity
33
+ from .providers.provider_manager import (get_provider, get_provider_class,
34
+ list_builtin_providers, resolve_config,
35
+ fetch_models_for, connect_provider, mask_key)
36
+ from .providers.provider_manager import load_config as _pm_load_config
37
+ from .providers.provider_manager import save_config as _pm_save_config
38
+
39
+ _LOG = logging.getLogger("cct.aicore")
40
+
41
+ # Config is now managed by providers/provider_manager.py.
42
+ # kept here only for backward-compatible imports
43
+ CONFIG_FILE = os.path.join(os.path.expanduser("~"), ".cct_ai_config.json")
44
+
45
+ # Default chat system prompt, shared by query_ai()/stream_ai() when no
46
+ # caller-specific prompt is passed in (agent.py passes its own, richer
47
+ # prompts, which also fold in identity.IDENTITY_BLOCK -- see agent.py).
48
+ # Folding the identity block in here too means even code paths that call
49
+ # aicore directly with the bare default still answer "who made you"
50
+ # style questions correctly and consistently.
51
+ DEFAULT_SYSTEM_PROMPT = (
52
+ "You are CAT AI, a coding and science assistant inside the Coding Agent "
53
+ "calculations.\n\n" + identity.IDENTITY_BLOCK
54
+ )
55
+
56
+ # ---------------------------------------------------------------- provider registry
57
+ # Built dynamically from the provider SDK so every registered provider shows up
58
+ # automatically. query_ai()/stream_ai() dispatch on `api_style` through the
59
+ # provider instance. To add a new provider, create a subclass of BaseProvider
60
+ # and call provider_manager.register_provider() — no other code needs to change.
61
+ PROVIDERS = {}
62
+ _BUILTIN_PROVIDER_DICT = None
63
+
64
+ def _build_provider_dict(force_reload=False):
65
+ global _BUILTIN_PROVIDER_DICT
66
+ if _BUILTIN_PROVIDER_DICT is not None and not force_reload:
67
+ return _BUILTIN_PROVIDER_DICT
68
+ d = {}
69
+ for info in list_builtin_providers():
70
+ pid = info["id"]
71
+ d[pid] = {
72
+ "base_url": info.get("url", ""),
73
+ "default_model": info.get("default_model", ""),
74
+ "api_style": info.get("api_style", "openai"),
75
+ "needs_key": info.get("needs_key", True),
76
+ "extra_headers": {},
77
+ }
78
+ # Add providers that aren't built-in but essential
79
+ extras = {
80
+ "openrouter": {"base_url": "https://openrouter.ai/api/v1",
81
+ "default_model": "openai/gpt-4o-mini", "api_style": "openai",
82
+ "needs_key": True,
83
+ "extra_headers": {"HTTP-Referer": "https://github.com/cct",
84
+ "X-Title": "Chemistry Calc Terminal"}},
85
+ "groq": {"base_url": "https://api.groq.com/openai/v1",
86
+ "default_model": "llama3.1-70b-versatile", "api_style": "openai",
87
+ "needs_key": True, "extra_headers": {}},
88
+ "ollama": {"base_url": "http://localhost:11434", "default_model": "llama3.3",
89
+ "api_style": "ollama", "needs_key": False, "extra_headers": {},
90
+ "auto_discover": True},
91
+ "moonshot": {"base_url": "https://api.moonshot.cn/v1", "default_model": "kimi-k3",
92
+ "api_style": "openai", "needs_key": True, "extra_headers": {}},
93
+ "xai": {"base_url": "https://api.x.ai/v1", "default_model": "grok-3",
94
+ "api_style": "openai", "needs_key": True, "extra_headers": {}},
95
+ }
96
+ d.update(extras)
97
+ # Load ALL providers from providers.json so any provider selected via
98
+ # /model or /provider has a valid base_url (nvidia, deepseek, etc.)
99
+ try:
100
+ from .models.manager import load_providers
101
+ for p in load_providers():
102
+ pid = p.get("id", "")
103
+ if pid and pid not in d:
104
+ ep = p.get("api_endpoint", "")
105
+ d[pid] = {
106
+ "base_url": ep,
107
+ "default_model": p.get("default_model", ""),
108
+ "api_style": p.get("api_style", "openai"),
109
+ "needs_key": bool(p.get("needs_key", True)),
110
+ "extra_headers": {},
111
+ }
112
+ elif pid and pid in d and not d[pid].get("base_url"):
113
+ d[pid]["base_url"] = p.get("api_endpoint", "")
114
+ except Exception:
115
+ pass
116
+ _BUILTIN_PROVIDER_DICT = d
117
+ return d
118
+
119
+
120
+ def _get_provider_info(provider_id):
121
+ """Get provider info, with cache refresh if base_url is missing."""
122
+ providers = _build_provider_dict()
123
+ info = providers.get(provider_id)
124
+ # v0.7.10: If provider exists but has no base_url, refresh cache
125
+ if info and not info.get("base_url") and provider_id:
126
+ providers = _build_provider_dict(force_reload=True)
127
+ info = providers.get(provider_id)
128
+ return info
129
+
130
+ # ---------------------------------------------------------------- token usage
131
+ # Best-effort session token accounting. Real usage is pulled straight out of
132
+ # the provider's own response body when it reports one (OpenAI/Anthropic/
133
+ # Gemini/native-Ollama all do); anything that doesn't is estimated at ~4
134
+ # chars/token, the same rule-of-thumb every provider's own docs use.
135
+ SESSION_USAGE = {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0, "requests": 0}
136
+
137
+ # Known context-window sizes are data-driven now: curated model metadata
138
+ # lives in models/model_metadata.json (context_length), with the provider's
139
+ # fallback list in providers/providers.json. context_window_for() consults
140
+ # those stores first and only falls back to this default estimate.
141
+ DEFAULT_CONTEXT_WINDOW = 32000
142
+
143
+
144
+ def estimate_tokens(text):
145
+ """Rough ~4-chars-per-token estimate, used whenever a provider doesn't
146
+ report real usage numbers back."""
147
+ if not text:
148
+ return 0
149
+ return max(1, len(str(text)) // 4)
150
+
151
+
152
+ def context_window_for(model):
153
+ if not model:
154
+ return DEFAULT_CONTEXT_WINDOW
155
+ model = str(model).lower()
156
+ # Data-driven: curated metadata first (exact id, any provider)...
157
+ try:
158
+ from .models import manager as _mgr
159
+ meta = _mgr.get_model_meta(model)
160
+ ctx = meta.get("context_length")
161
+ if ctx:
162
+ return int(ctx)
163
+ except Exception:
164
+ pass
165
+ # ...then heuristic family match across all curated metadata.
166
+ try:
167
+ from .models.registry import get_model_info
168
+ for _, m in _load_metadata_items():
169
+ if m.get("family") and m["family"].lower() in model:
170
+ ctx = m.get("context_length")
171
+ if ctx:
172
+ return int(ctx)
173
+ except Exception:
174
+ pass
175
+ return DEFAULT_CONTEXT_WINDOW
176
+
177
+
178
+ def _load_metadata_items():
179
+ """Yield (model_id, metadata_dict) for every curated model."""
180
+ from .models import manager as _mgr
181
+ try:
182
+ with open(_mgr.MODEL_METADATA_FILE, "r", encoding="utf-8") as f:
183
+ import json as _json
184
+ for m in _json.load(f).get("models", []):
185
+ yield m.get("id", ""), m
186
+ except Exception:
187
+ return iter(())
188
+
189
+
190
+ def record_usage(prompt_tokens, completion_tokens):
191
+ SESSION_USAGE["prompt_tokens"] += int(prompt_tokens or 0)
192
+ SESSION_USAGE["completion_tokens"] += int(completion_tokens or 0)
193
+ SESSION_USAGE["total_tokens"] += int(prompt_tokens or 0) + int(completion_tokens or 0)
194
+ SESSION_USAGE["requests"] += 1
195
+
196
+
197
+ def get_session_usage():
198
+ """Snapshot of this session's token use plus a best-effort estimate of
199
+ how much of the current model's context window is left."""
200
+ config = load_config()
201
+ model = config.get("model", "")
202
+ window = context_window_for(model)
203
+ used = SESSION_USAGE["total_tokens"]
204
+ return {
205
+ "prompt_tokens": SESSION_USAGE["prompt_tokens"],
206
+ "completion_tokens": SESSION_USAGE["completion_tokens"],
207
+ "total_tokens": used,
208
+ "requests": SESSION_USAGE["requests"],
209
+ "model": model or "(not configured)",
210
+ "context_window": window,
211
+ "remaining_estimate": max(0, window - used),
212
+ }
213
+
214
+
215
+ def reset_session_usage():
216
+ SESSION_USAGE.update({"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0, "requests": 0})
217
+
218
+
219
+ def _track_usage_from_data(api_style, data, prompt_text, completion_text):
220
+ """Pull real usage numbers out of the provider's response body when
221
+ present; fall back to the character-based estimate otherwise. Never
222
+ raises — token accounting is a nice-to-have, not a dependency."""
223
+ try:
224
+ if api_style == "openai" and isinstance(data, dict) and isinstance(data.get("usage"), dict):
225
+ u = data["usage"]
226
+ record_usage(u.get("prompt_tokens", estimate_tokens(prompt_text)),
227
+ u.get("completion_tokens", estimate_tokens(completion_text)))
228
+ return
229
+ if api_style == "anthropic" and isinstance(data, dict) and isinstance(data.get("usage"), dict):
230
+ u = data["usage"]
231
+ record_usage(u.get("input_tokens", estimate_tokens(prompt_text)),
232
+ u.get("output_tokens", estimate_tokens(completion_text)))
233
+ return
234
+ if api_style == "gemini" and isinstance(data, dict) and isinstance(data.get("usageMetadata"), dict):
235
+ u = data["usageMetadata"]
236
+ record_usage(u.get("promptTokenCount", estimate_tokens(prompt_text)),
237
+ u.get("candidatesTokenCount", estimate_tokens(completion_text)))
238
+ return
239
+ if api_style == "ollama" and isinstance(data, dict) and "eval_count" in data:
240
+ record_usage(data.get("prompt_eval_count", estimate_tokens(prompt_text)),
241
+ data.get("eval_count", estimate_tokens(completion_text)))
242
+ return
243
+ except Exception:
244
+ pass
245
+ record_usage(estimate_tokens(prompt_text), estimate_tokens(completion_text))
246
+
247
+
248
+ # ---------------------------------------------------------------- model switch
249
+ # Model suggestions are data-driven: the provider's fallback list in
250
+ # providers.json (seeded from the provider's own model catalog). Live
251
+ # fetching happens when /model opens — see models/manager.py.
252
+ def common_models(provider_id):
253
+ """Recommended/known model ids for a provider (providers.json
254
+ fallback list, plus any cached live list). Never raises."""
255
+ out = []
256
+ try:
257
+ from .models import manager as _mgr
258
+ provider = _mgr.get_provider(provider_id)
259
+ if provider:
260
+ out = [str(m) for m in provider.get("fallback_models", [])]
261
+ cached = _mgr.get_cached(provider_id)
262
+ if cached:
263
+ live = [str(m) for m in cached[0]]
264
+ for m in live:
265
+ if m not in out:
266
+ out.append(m)
267
+ except Exception:
268
+ pass
269
+ return out[:40]
270
+
271
+
272
+ def switch_model(model_name):
273
+ """Change just the model for the currently configured provider —
274
+ lighter-weight than the full /ai setup_ai() wizard. Returns (ok, msg)."""
275
+ model_name = (model_name or "").strip()
276
+ if not model_name:
277
+ return False, "No model name given."
278
+ config = load_config()
279
+ if not config.get("provider"):
280
+ return False, "No AI provider configured yet. Run /ai or /model to set one up first."
281
+ old = config.get("model")
282
+ config["model"] = model_name
283
+ save_config(config)
284
+ return True, f"Model switched from '{old}' to '{model_name}' ({config['provider']})."
285
+
286
+
287
+ # load_config/save_config delegate to provider_manager (so callers that
288
+ # do `from calc_terminal.aicore import load_config` still work).
289
+ def load_config():
290
+ return _pm_load_config()
291
+
292
+ def save_config(config):
293
+ return _pm_save_config(config)
294
+
295
+ def list_models(config):
296
+ """Fetch the live list of model IDs actually available to this
297
+ provider/key. Returns a (possibly empty) list of strings, never
298
+ raises — callers should treat an empty list as 'listing unsupported
299
+ or unreachable right now', not as an error.
300
+
301
+ Priority chain (models/manager.py): provider API -> disk cache
302
+ (cache/models/) -> built-in fallback list (providers.json).
303
+ Successful fetches are cached on disk with a timestamp.
304
+ """
305
+ if not _HAS_REQUESTS:
306
+ return []
307
+ provider = config.get("provider")
308
+ if not provider:
309
+ return []
310
+ # Data-driven resolution with disk caching
311
+ try:
312
+ from .models import manager as _mgr
313
+ models, source = _mgr.get_models(provider, config)
314
+ # "default" is the last-resort placeholder — treat as unsupported
315
+ if models and models != ["default"]:
316
+ return list(models)
317
+ except Exception:
318
+ pass
319
+ # Legacy fallback for providers not covered by the manager
320
+ info = _get_provider_info(provider)
321
+ api_style = info["api_style"] if info else "openai"
322
+ if provider == "ollama":
323
+ base_url = config.get("ollama_url") or (info["base_url"] if info else "http://localhost:11434")
324
+ else:
325
+ base_url = config.get("base_url") or config.get("api_url") or (info["base_url"] if info else "")
326
+ base_url = base_url.rstrip("/")
327
+ api_key = config.get("api_key", "")
328
+ extra_headers = info["extra_headers"] if info else {}
329
+ try:
330
+ if api_style == "ollama" and not base_url.endswith("/v1"):
331
+ resp = requests.get(f"{base_url}/api/tags", timeout=8)
332
+ if resp.status_code >= 400:
333
+ return []
334
+ return sorted(m.get("name", "") for m in resp.json().get("models", []) if m.get("name"))
335
+ elif api_style == "openai":
336
+ headers = {"Content-Type": "application/json"}
337
+ if api_key:
338
+ headers["Authorization"] = f"Bearer {api_key}"
339
+ headers.update(extra_headers)
340
+ resp = requests.get(f"{base_url}/models", headers=headers, timeout=10)
341
+ if resp.status_code >= 400:
342
+ return []
343
+ data = resp.json().get("data", [])
344
+ return sorted(m.get("id", "") for m in data if isinstance(m, dict) and m.get("id"))
345
+ elif api_style == "gemini":
346
+ resp = requests.get(f"{base_url}/models?key={api_key}", timeout=10)
347
+ if resp.status_code >= 400:
348
+ return []
349
+ out = []
350
+ for m in resp.json().get("models", []):
351
+ methods = m.get("supportedGenerationMethods", [])
352
+ if not methods or "generateContent" in methods:
353
+ out.append(m.get("name", "").split("/")[-1])
354
+ return sorted(set(n for n in out if n))
355
+ else:
356
+ return []
357
+ except Exception:
358
+ return []
359
+
360
+
361
+ def _prompt_model(provider, default_model, available):
362
+ """Prompt for a model name, offering a live-fetched picker when one is
363
+ available, but ALWAYS accepting free-form text too — so any model,
364
+ including brand-new ones not in the list yet, still works. Also
365
+ catches the classic mistake of typing the provider's name itself
366
+ (e.g. typing 'gemini' as the model) instead of a real model id.
367
+ """
368
+ shown = available[:20]
369
+ if shown:
370
+ print(theme.dim(f" Found {len(available)} model(s) available to this key/server."
371
+ + (f" Showing first {len(shown)}:" if len(available) > len(shown) else "")))
372
+ for i, m in enumerate(shown, 1):
373
+ print(theme.cyan(f" {i}.") + " " + theme.text(m))
374
+ print(theme.faint(f" Type a number to pick one, or type ANY model name directly."))
375
+
376
+ while True:
377
+ model_in = input(theme.dim(f" Model (default: {default_model}) \u25b8 ")).strip()
378
+ if shown and model_in.isdigit() and 1 <= int(model_in) <= len(shown):
379
+ return shown[int(model_in) - 1]
380
+ candidate = _clean_model_name(model_in, default_model)
381
+ if candidate.strip().lower() == str(provider).strip().lower():
382
+ print(theme.orange(f" '{candidate}' is the provider name, not a model id — "
383
+ f"e.g. try '{default_model}', or pick a number above."))
384
+ continue
385
+ return candidate
386
+
387
+
388
+ def setup_ai():
389
+ if not _HAS_REQUESTS:
390
+ print(theme.red("\n Error: The 'requests' library is not installed."))
391
+ print(theme.dim(" AI features require the 'requests' package to communicate with APIs."))
392
+ print(theme.dim(" Please run: ") + theme.text("pip install requests", bold=True))
393
+ time.sleep(3)
394
+ return
395
+
396
+ while True:
397
+ theme.clear_screen()
398
+ # Build the menu dynamically from PROVIDERS so every registered
399
+ # provider shows up without hand-editing this list.
400
+ menu_lines = [
401
+ theme.badge("BETA", theme.BG_WARN) + " " + theme.purple("AI CONFIGURATION", bold=True),
402
+ theme.dim("CAT AI is specialized for complex problem solving & scientific computing."),
403
+ theme.dim("Any provider works with ANY model it supports — pick from the live list"),
404
+ theme.dim("shown after your key, or type a model name yourself at any time."),
405
+ "",
406
+ theme.text("Choose your AI Provider:"),
407
+ ]
408
+ provider_dict = _build_provider_dict()
409
+ keys = list(provider_dict.keys())
410
+ for i, name in enumerate(keys, 1):
411
+ info = provider_dict[name]
412
+ tag = "Local, no key" if not info["needs_key"] else "API Key"
413
+ menu_lines.append(theme.cyan(f"{i}. {name.capitalize()} ({tag})"))
414
+ menu_lines.append(theme.cyan(f"{len(keys) + 1}. Custom API Endpoint"))
415
+ menu_lines.append(theme.cyan(f"{len(keys) + 2}. Back to Menu"))
416
+ menu_lines += ["", theme.faint("Press Enter to skip / use default settings.")]
417
+ print(theme.panel(menu_lines, title="AI SETUP", color=theme.PURPLE, width=70))
418
+
419
+ choice = input(theme.dim(" choice \u25b8 ")).strip()
420
+ back_index = len(keys) + 2
421
+ if choice == str(back_index) or choice == "":
422
+ return
423
+
424
+ config = load_config()
425
+
426
+ # ---- pick a provider from the registry ----
427
+ if choice.isdigit() and 1 <= int(choice) <= len(keys):
428
+ name = keys[int(choice) - 1]
429
+ info = provider_dict[name]
430
+ config["provider"] = name
431
+ # v0.7.10: Always save base_url when selecting a provider
432
+ if info.get("base_url"):
433
+ config["base_url"] = info["base_url"]
434
+ if info["needs_key"]:
435
+ config["api_key"] = input(theme.dim(f" Enter {name.capitalize()} API Key \u25b8 ")).strip()
436
+ available = []
437
+ if info["needs_key"] or name == "ollama":
438
+ print(theme.dim(" Checking which models are available..."))
439
+ available = list_models(config)
440
+ config["model"] = _prompt_model(name, info["default_model"], available)
441
+ break
442
+
443
+ # ---- custom endpoint (user supplies everything) ----
444
+ elif choice == str(len(keys) + 1):
445
+ config["provider"] = "custom"
446
+ config["api_url"] = input(theme.dim(" API Endpoint URL \u25b8 ")).strip()
447
+ config["base_url"] = config["api_url"]
448
+ config["api_key"] = input(theme.dim(" API Key \u25b8 ")).strip()
449
+ print(theme.dim(" Checking which models are available (best-effort, OpenAI-style /models)..."))
450
+ available = list_models(config)
451
+ config["model"] = _prompt_model("custom", "", available) if available else \
452
+ input(theme.dim(" Model Name \u25b8 ")).strip()
453
+ break
454
+
455
+ save_config(config)
456
+ print(theme.green("\n AI Configuration saved successfully!"))
457
+ print(theme.dim(" Verifying connection..."))
458
+ ok, msg, models = verify_connection(config)
459
+ print((theme.green if ok else theme.red)((" \u2713 " if ok else " \u2717 ") + msg))
460
+ if models:
461
+ preview = ", ".join(models[:8])
462
+ print(theme.faint(f" Models seen: {preview}{', ...' if len(models) > 8 else ''}"))
463
+ time.sleep(1.5)
464
+
465
+
466
+ def _clean_model_name(user_input, default):
467
+ """Normalise the model name typed at the prompt. Empty input, or words like
468
+ 'yes'/'no'/'ok' (commonly typed meaning 'use the default'), fall back to the
469
+ real default — never saved literally as a model name."""
470
+ if not user_input:
471
+ return default
472
+ if user_input.lower() in ("yes", "y", "no", "n", "ok", "true", "false", "default"):
473
+ return default
474
+ return user_input
475
+
476
+ def verify_connection(config=None):
477
+ """Actively test the configured AI provider/model — a real network
478
+ round trip, not just 'is a config file present'. Returns
479
+ (ok: bool, message: str, models: list[str]).
480
+
481
+ Uses the provider SDK for all built-in providers; falls back to
482
+ the legacy per-api_style logic for non-SDK providers (ollama, groq).
483
+ """
484
+ if not _HAS_REQUESTS:
485
+ return False, "The 'requests' library is not installed. Run: pip install requests", []
486
+
487
+ config = config or load_config()
488
+ provider = config.get("provider")
489
+ if not provider:
490
+ return False, "No AI provider configured yet. Run /ai (or /ai-verify after setup) to configure one.", []
491
+
492
+ # Try provider SDK first
493
+ try:
494
+ inst = get_provider(config)
495
+ if inst:
496
+ ok, msg, models = inst.connect()
497
+ return ok, msg, models
498
+ except Exception:
499
+ pass
500
+
501
+ # Fallback for non-SDK providers
502
+ info = _get_provider_info(provider)
503
+ api_style = info["api_style"] if info else "openai"
504
+
505
+ if provider == "ollama":
506
+ base_url = config.get("ollama_url") or (info["base_url"] if info else "http://localhost:11434")
507
+ else:
508
+ base_url = config.get("base_url") or config.get("api_url") or (info["base_url"] if info else "")
509
+ base_url = base_url.rstrip("/")
510
+
511
+ api_key = config.get("api_key", "")
512
+ model = config.get("model") or (info["default_model"] if info else "")
513
+ extra_headers = info["extra_headers"] if info else {}
514
+
515
+ try:
516
+ if api_style == "ollama" and not base_url.endswith("/v1"):
517
+ resp = requests.get(f"{base_url}/api/tags", timeout=8)
518
+ _raise_for_status(resp)
519
+ models = [m.get("name", "") for m in resp.json().get("models", [])]
520
+ if models and model and not any(model in m for m in models):
521
+ return (True,
522
+ f"Connected to Ollama at {base_url}, but model '{model}' isn't pulled locally yet. "
523
+ f"Run: ollama pull {model}",
524
+ models)
525
+ return True, f"Connected to Ollama at {base_url} \u2014 {len(models)} local model(s) available.", models
526
+
527
+ elif api_style == "openai":
528
+ headers = {"Content-Type": "application/json"}
529
+ if api_key:
530
+ headers["Authorization"] = f"Bearer {api_key}"
531
+ headers.update(extra_headers)
532
+ resp = requests.get(f"{base_url}/models", headers=headers, timeout=10)
533
+ _raise_for_status(resp)
534
+ data = resp.json().get("data", [])
535
+ models = [m.get("id", "") for m in data if isinstance(m, dict)]
536
+ note = ""
537
+ if models and model and not any(model == m or model in m for m in models):
538
+ note = f" Note: '{model}' wasn't in the list returned for this key — double-check the model name."
539
+ return True, f"Connected to {provider} at {base_url} \u2014 {len(models)} model(s) visible to this key.{note}", models
540
+
541
+ elif api_style == "gemini":
542
+ resp = requests.get(f"{base_url}/models?key={api_key}", timeout=10)
543
+ _raise_for_status(resp)
544
+ models = []
545
+ for m in resp.json().get("models", []):
546
+ methods = m.get("supportedGenerationMethods", [])
547
+ if not methods or "generateContent" in methods:
548
+ models.append(m.get("name", "").split("/")[-1])
549
+ note = ""
550
+ if models and model and model not in models:
551
+ note = f" Note: '{model}' wasn't in the list returned for this key — pick one of the models shown, or double-check the name."
552
+ return True, f"Connected to Gemini \u2014 {len(models)} model(s) visible to this key.{note}", models
553
+
554
+ else:
555
+ reply = query_ai("Reply with only the single word: OK",
556
+ system_prompt="You are a connectivity test. Reply with only: OK")
557
+ failure_markers = ("could not reach", "the ai request timed out",
558
+ "error connecting", "http 4", "http 5", "not configured")
559
+ if reply and not any(reply.lower().startswith(m) for m in failure_markers):
560
+ return True, f"Connected to {provider}, model '{model}' responded successfully.", []
561
+ return False, f"Connection test failed: {reply}", []
562
+
563
+ except requests.exceptions.ConnectionError:
564
+ return False, f"Could not reach {base_url}. Check the URL / your internet, or that Ollama is running (ollama serve).", []
565
+ except requests.exceptions.Timeout:
566
+ return False, "Connection timed out.", []
567
+ except RuntimeError as e:
568
+ return False, f"Verification failed: {e}", []
569
+ except Exception as e:
570
+ return False, f"Verification failed: {e}", []
571
+
572
+
573
+ def _sanitize_history(history):
574
+ """Normalizes whatever a caller hands us into a clean list of
575
+ (role, text) pairs with only 'user'/'assistant' roles, no empty
576
+ turns, and no in-flight streaming placeholder (empty text). This
577
+ is the ONE place that decides what "conversation history" means
578
+ for every provider below — every api_style builds its request
579
+ from this, so none of them can silently drop it again."""
580
+ if not history:
581
+ return []
582
+ out = []
583
+ for item in history:
584
+ if isinstance(item, (list, tuple)) and len(item) == 2:
585
+ role, text = item
586
+ elif isinstance(item, dict):
587
+ role, text = item.get("role"), item.get("text", item.get("content", ""))
588
+ else:
589
+ continue
590
+ role = "assistant" if role in ("assistant", "ai", "model") else "user"
591
+ text = (text or "").strip()
592
+ if text:
593
+ out.append((role, text))
594
+ return out
595
+
596
+
597
+ def _history_char_count(history):
598
+ return sum(len(t) for _, t in history)
599
+
600
+
601
+ def _openai_messages(system_prompt, history, prompt, attachments=None, vision=False, model=""):
602
+ """Shared by every OpenAI-compatible api_style (openai, groq,
603
+ openrouter, vLLM, Ollama's /v1 endpoint) — system prompt, then
604
+ every prior turn in order, then the new user prompt.
605
+
606
+ v0.7.8.1: when `vision` is True and attachments carry images, the
607
+ final user message becomes a content array (text + image_url data
608
+ URLs) so vision-capable models actually see the attached images
609
+ instead of only reading their metadata. v0.7.8.2: image parts use
610
+ the OpenAI shape only here; Anthropic/Gemini get their own shapes
611
+ (see _user_content)."""
612
+ messages = []
613
+ mod_lower = (model or "").lower()
614
+ is_o1_mini = "o1-mini" in mod_lower or "o1-preview" in mod_lower
615
+ sys_role = "developer" if ("o1" in mod_lower or "o3" in mod_lower) and not is_o1_mini else "system"
616
+ if system_prompt:
617
+ if is_o1_mini:
618
+ prompt = f"{system_prompt}\n\n{prompt}"
619
+ else:
620
+ messages.append({"role": sys_role, "content": system_prompt})
621
+ for role, text in history:
622
+ messages.append({"role": role, "content": text})
623
+ messages.append({"role": "user", "content": _user_content(
624
+ prompt, attachments, vision, api_style="openai")})
625
+ return messages
626
+
627
+
628
+ def _anthropic_messages(history, prompt, attachments=None, vision=False):
629
+ """Anthropic keeps `system` as its own top-level field, so this
630
+ only builds the `messages` array (user/assistant turns)."""
631
+ messages = [{"role": role, "content": text} for role, text in history]
632
+ messages.append({"role": "user", "content": _user_content(
633
+ prompt, attachments, vision, api_style="anthropic")})
634
+ return messages
635
+
636
+
637
+ def _gemini_contents(history, prompt, attachments=None, vision=False):
638
+ """Gemini calls the assistant role 'model', not 'assistant'."""
639
+ contents = []
640
+ for role, text in history:
641
+ contents.append({"role": ("model" if role == "assistant" else "user"),
642
+ "parts": [{"text": text}]})
643
+ contents.append({"role": "user", "parts": _user_content(
644
+ prompt, attachments, vision, api_style="gemini")})
645
+ return contents
646
+
647
+
648
+ def _user_content(prompt, attachments=None, vision=False, api_style="openai"):
649
+ """The final user message content. Plain string when there is
650
+ nothing to attach natively; a provider-correct content array with
651
+ text + image parts when vision is available AND readable images are
652
+ attached. Images that failed to read are skipped here — their
653
+ textual fallback block (built by the attachment manager) still
654
+ carries the metadata.
655
+
656
+ v0.7.8.2 (attachment pipeline fix): the image parts are shaped per
657
+ provider — Anthropic and Gemini reject OpenAI's `image_url` block
658
+ (they each have their own schema), so a vision-capable Anthropic or
659
+ Gemini model previously received a malformed content array and the
660
+ attached image never reached it. Each api_style now emits its own
661
+ native shape:
662
+ - openai: {"type": "image_url", "image_url": {url: data-url}}
663
+ - anthropic: {"type": "image", "source": {base64, media_type}}
664
+ - gemini: {"inline_data": {mime_type, data}}
665
+ """
666
+ if not vision or not attachments:
667
+ return prompt
668
+ images = []
669
+ try:
670
+ from . import attachments as _att
671
+ for att in attachments:
672
+ if _att.attachment_has_image_payload(att):
673
+ # v0.7.9.0: the vision pipeline may hand us an ALREADY-
674
+ # normalized base64 payload (metadata["inline_b64"]) — use
675
+ # it directly instead of re-reading/re-encoding the file.
676
+ inline = None
677
+ try:
678
+ inline = (att.metadata or {}).get("inline_b64")
679
+ except Exception:
680
+ inline = None
681
+ if inline:
682
+ images.append((getattr(att, "mime_type", "image/png") or "image/png",
683
+ inline))
684
+ continue
685
+ try:
686
+ mime, b64 = _att.encode_image_data_url(att.path)
687
+ except Exception:
688
+ continue
689
+ images.append((mime, b64))
690
+ except Exception:
691
+ return prompt
692
+ if not images:
693
+ return prompt
694
+ parts = []
695
+ if api_style == "anthropic":
696
+ parts.append({"type": "text", "text": prompt})
697
+ for mime, b64 in images:
698
+ parts.append({"type": "image",
699
+ "source": {"type": "base64", "media_type": mime,
700
+ "data": b64}})
701
+ return parts
702
+ if api_style == "gemini":
703
+ parts.append({"text": prompt})
704
+ for mime, b64 in images:
705
+ parts.append({"inline_data": {"mime_type": mime, "data": b64}})
706
+ return parts
707
+ parts.append({"type": "text", "text": prompt})
708
+ for mime, b64 in images:
709
+ parts.append({"type": "image_url",
710
+ "image_url": {"url": f"data:{mime};base64,{b64}"}})
711
+ return parts
712
+
713
+
714
+ def _ollama_native_prompt(system_prompt, history, prompt):
715
+ """The native /api/generate endpoint (no v1 alias) only accepts one
716
+ flat `prompt` string — no messages array. To keep it from forgetting
717
+ the conversation the same way the message-based providers would, we
718
+ fold prior turns into the prompt as a plain transcript. `system` is
719
+ still passed separately via the `system` field."""
720
+ if not history:
721
+ return prompt
722
+ lines = [f"{'Assistant' if role == 'assistant' else 'User'}: {text}" for role, text in history]
723
+ transcript = "\n".join(lines)
724
+ return f"Conversation so far:\n{transcript}\n\nUser: {prompt}\nAssistant:"
725
+
726
+
727
+ def _resolve_provider(config):
728
+ """Shared provider/URL/model resolution — used by both query_ai
729
+ (blocking) and stream_ai (generator) so there's one place that
730
+ decides which base_url/model/headers a request uses, not two that
731
+ can drift apart. Returns (api_style, base_url, api_key, model,
732
+ temperature, extra_headers)."""
733
+ provider = config.get("provider")
734
+ # Try provider SDK first
735
+ _inst = get_provider(config)
736
+ if _inst:
737
+ base_url = _inst.get_base_url()
738
+ # v0.7.10: Ensure base_url is never empty for non-SDK providers
739
+ if not base_url and provider:
740
+ info = _get_provider_info(provider)
741
+ if info and info.get("base_url"):
742
+ base_url = info["base_url"]
743
+ return (_inst.API_STYLE, base_url, _inst.get_api_key(),
744
+ _inst.get_model(), config.get("temperature"), _inst.get_extra_headers())
745
+ info = _get_provider_info(provider)
746
+ api_style = info["api_style"] if info else "openai"
747
+ if provider == "ollama":
748
+ base_url = config.get("ollama_url") or (info["base_url"] if info else "http://localhost:11434")
749
+ else:
750
+ base_url = config.get("base_url") or config.get("api_url") or (info["base_url"] if info else "")
751
+ # v0.7.10: If still no base_url, try to get it from providers.json
752
+ if not base_url and provider:
753
+ try:
754
+ from .models.manager import load_providers
755
+ for p in load_providers():
756
+ if p.get("id") == provider:
757
+ base_url = p.get("api_endpoint", "")
758
+ if base_url:
759
+ break
760
+ except Exception:
761
+ pass
762
+ base_url = base_url.rstrip("/") if base_url else ""
763
+ api_key = config.get("api_key", "")
764
+ if not (api_key or "").strip() and provider:
765
+ try:
766
+ from .providers.provider_manager import get_env_api_key
767
+ api_key = get_env_api_key(provider)
768
+ except Exception:
769
+ pass
770
+ model = config.get("model") or (info["default_model"] if info else "gpt-4o-mini")
771
+ temperature = config.get("temperature")
772
+ extra_headers = info["extra_headers"] if info else {}
773
+ return api_style, base_url, api_key, model, temperature, extra_headers
774
+
775
+
776
+ # The exact prefixes query_ai() (and query_ai_with_image()) return
777
+ # instead of raising, on every known failure path — copied verbatim
778
+ # from those functions' own `return` statements below so this can
779
+ # never drift out of sync silently. Used by callers (app.py's cmd_ai/
780
+ # cmd_agent, ui/app.py's _stream_worker) that want to show a real
781
+ # error-recovery card (spec section 19) instead of rendering the
782
+ # failure as if it were a normal chat answer.
783
+ _ERROR_SIGNATURES = (
784
+ "The 'requests' library is required for AI features.",
785
+ "AI not configured.",
786
+ "Could not reach the AI server.",
787
+ "The AI request timed out.",
788
+ "Error connecting to AI:",
789
+ "No model response received.",
790
+ "Invalid API key.",
791
+ "Rate limited.",
792
+ "Quota exceeded.",
793
+ "Model '",
794
+ "Error: ",
795
+ "*(Generation timed out)*",
796
+ "(Generation timed out)",
797
+ "Generation timed out",
798
+ "*(interrupted",
799
+ "Ollama Error:",
800
+ "Error: Ollama",
801
+ "Could not reach Ollama",
802
+ "The Ollama request timed out",
803
+ "Ollama streaming failure:",
804
+ )
805
+ # not a fixed prefix (provider name is interpolated) — matched separately
806
+ _ERROR_SUFFIX = "is not supported yet. Run /ai to reconfigure."
807
+
808
+
809
+ def is_error_response(text):
810
+ """True if `text` is one of query_ai's own failure messages rather
811
+ than an actual model answer."""
812
+ if not text or not isinstance(text, str):
813
+ return False
814
+ t = text.strip()
815
+ if t.startswith(_ERROR_SIGNATURES) or t.endswith(_ERROR_SUFFIX):
816
+ return True
817
+ low = t.lower()
818
+ return any(h in low for h in _FALLOVER_HINTS)
819
+
820
+
821
+ # ---------------------------------------------------------------------------
822
+ # Backup-provider failover (v0.7.8 BONUS 1). query_ai()/stream_ai() first
823
+ # try the saved primary provider; when it reports quota exhaustion, a
824
+ # timeout, rate limiting, or is simply offline, they walk the enabled
825
+ # backup chain (providers/provider_manager.backup_configs(), priority
826
+ # order) and seamlessly finish the request against the next healthy one.
827
+ # The conversation never notices: history/summaries live in the UI and
828
+ # are re-sent verbatim to whichever provider answers.
829
+ #
830
+ # v0.7.9.5 REQUEST LIFECYCLE OVERHAUL (the '...' bug): the walk now has
831
+ # real engineering around it instead of one blind attempt per provider:
832
+ #
833
+ # * TOTAL + IDLE timeouts, configurable per prompt class (config.py:
834
+ # timeout_simple / timeout_normal / timeout_large /
835
+ # model_idle_timeout). A provider trickling bytes forever can no
836
+ # longer hold a turn open indefinitely, while an actively streaming
837
+ # model is never misclassified as frozen just because it's slow.
838
+ # * BOUNDED retries with EXPONENTIAL BACKOFF per provider
839
+ # (model_max_retries_per_provider, model_retry_backoff_base) — only
840
+ # for transient failures (timeouts / connection errors); quota and
841
+ # config errors skip straight to the next provider.
842
+ # * PROVIDER COOLDOWN (model_provider_cooldown): a provider that just
843
+ # failed sits out future requests briefly, so a dead primary stops
844
+ # taxing every turn; backups are tried first until it recovers.
845
+ # * CANCELLATION: every in-flight HTTP response is registered;
846
+ # cancel_active_requests() closes the sockets so Ctrl+C can stop a
847
+ # request that's blocked inside a socket read.
848
+ # * STRUCTURED REQUEST LOGGING (~/.cct_requests.log): one JSON line
849
+ # per lifecycle event (start/first_token/finish/error/retry/
850
+ # fallback) with timings and exception class — never API keys —
851
+ # so a stuck response can always be diagnosed after the fact.
852
+
853
+ _FALLOVER_HINTS = ("quota", "rate limit", "rate_limit", "insufficient_quota",
854
+ "limit exceeded", "exhausted", "429 ", "could not reach",
855
+ "connection", "api key", "unauthorized", "401", "403",
856
+ "model not found", "not found", "no model", "not running",
857
+ "timed out", "timeout", "timedout", "time out",
858
+ "generation timed out", "cannot connect", "failed to connect",
859
+ "could not reach ollama", "ensure ollama is running")
860
+
861
+ # A hook the UI can install (set_failover_hook) to surface failover
862
+ # moments as chat notes instead of silence.
863
+ _FAILOVER_HOOK = None
864
+
865
+
866
+ def _should_failover(text):
867
+ """True when a returned text looks like a provider-level failure worth
868
+ switching providers for — CCT's own error strings, or the typical
869
+ HTTP-level rate-limit / quota wording providers embed in bodies."""
870
+ if not text or not isinstance(text, str):
871
+ return True
872
+ if not text.strip():
873
+ return True
874
+ if is_error_response(text):
875
+ return True
876
+ low = text.lower()
877
+ return any(h in low for h in _FALLOVER_HINTS)
878
+
879
+
880
+ def _retryable_failure(text):
881
+ """True for TRANSIENT failures worth an immediate retry against the
882
+ SAME provider (network blip, momentary read timeout). Quota /
883
+ rate-limit / configuration failures are not retryable — retrying
884
+ them just burns seconds before the inevitable fallback."""
885
+ if not text:
886
+ return False
887
+ t = text.strip()
888
+ if t.startswith(("Could not reach the AI server.", "The AI request timed out.")):
889
+ return True
890
+ low = t.lower()
891
+ # Transient network/timeout issues are worth retrying
892
+ transient_hints = ("timed out", "timeout", "connection reset", "connection refused",
893
+ "connection error", "connection aborted", "broken pipe",
894
+ "eof occurred", "incomplete read", "remote end closed",
895
+ "server disconnected", "503", "502", "500")
896
+ # Non-retryable: quota, auth, config issues
897
+ non_retryable_hints = ("quota", "rate limit", "rate_limit", "401", "403",
898
+ "api key", "unauthorized", "not found", "404",
899
+ "not supported", "end of life", "deprecated")
900
+ if any(h in low for h in non_retryable_hints):
901
+ return False
902
+ return any(h in low for h in transient_hints)
903
+
904
+
905
+ def _notify_failover(message):
906
+ try:
907
+ if _FAILOVER_HOOK is not None:
908
+ _FAILOVER_HOOK(message)
909
+ except Exception:
910
+ pass
911
+
912
+
913
+ def set_failover_hook(callback):
914
+ """Install a callable(message) hook invoked on every failover step
915
+ (exhausted primary, switching to backup N, connected). The Textual UI
916
+ uses this to post chat system notes; the classic REPL may leave it
917
+ None to stay silent. Pass None to clear."""
918
+ global _FAILOVER_HOOK
919
+ _FAILOVER_HOOK = callback
920
+
921
+
922
+ def _backup_chain():
923
+ """Get the backup provider chain for failover.
924
+
925
+ v0.7.9.5: Improved Ollama handling for backup failover. Ollama
926
+ providers are prioritized for local inference when available,
927
+ providing a reliable fallback that works offline and has no
928
+ rate limits or quota issues.
929
+
930
+ v0.7.10: Enhanced model selection — always prefers instruct/chat
931
+ models over base models; broader keyword matching for quality
932
+ models; GPU/VRAM-aware model selection.
933
+ """
934
+ try:
935
+ from .providers.provider_manager import backup_configs
936
+ chain = backup_configs()
937
+
938
+ # Base model identifiers (these models can't follow instructions)
939
+ _BASE_MODEL_HINTS = ("-base", "_base", "base-q", "base_q",
940
+ ":base", "-base-", "base_model")
941
+ # Good instruct/chat model identifiers
942
+ _INSTRUCT_HINTS = ("instruct", "chat", "r1", "gemma", "qwen",
943
+ "llama-3", "phi-3", "phi-4", "mistral",
944
+ "codellama", "coder", "deepseek", "yi-",
945
+ "command", "mixtral", "wizard", "nous",
946
+ "solar", "neural", "orca", "zephyr",
947
+ "hermes", "dolphin", "tinyllama",
948
+ "starcoder", "codestral", "granite")
949
+
950
+ def _is_base_model(name):
951
+ low = name.lower()
952
+ return any(b in low for b in _BASE_MODEL_HINTS)
953
+
954
+ def _is_good_instruct(name):
955
+ low = name.lower()
956
+ if _is_base_model(name):
957
+ return False
958
+ return any(k in low for k in _INSTRUCT_HINTS)
959
+
960
+ enhanced_chain = []
961
+ for cfg, entry in chain:
962
+ if cfg.get("provider") == "ollama":
963
+ # Ensure Ollama has correct api_style and no key requirement
964
+ cfg["api_style"] = "ollama"
965
+ cfg["needs_key"] = False
966
+ base_url = cfg.get("base_url") or "http://localhost:11434"
967
+ cfg["base_url"] = base_url
968
+ # Auto-verify that the configured model is installed locally;
969
+ # If not or if it's a raw base model, fallback to a healthy instruct model!
970
+ try:
971
+ import requests
972
+ tag_r = requests.get(f"{base_url}/api/tags", timeout=2.0)
973
+ if tag_r.status_code == 200:
974
+ raw_models = [m.get("name") for m in tag_r.json().get("models", []) if m.get("name")]
975
+ # Split into instruct vs base
976
+ instruct_models = [m for m in raw_models if not _is_base_model(m)]
977
+ good_models = [m for m in instruct_models if _is_good_instruct(m)]
978
+ # Prefer good instruct > any non-base > all
979
+ candidates = good_models if good_models else (instruct_models if instruct_models else raw_models)
980
+
981
+ cur_m = cfg.get("model", "")
982
+ is_cur_base = _is_base_model(cur_m)
983
+ has_m = any(
984
+ cur_m == im or (":" not in cur_m and im.startswith(f"{cur_m}:"))
985
+ for im in candidates
986
+ )
987
+ if (not has_m or is_cur_base) and candidates:
988
+ # Score candidates: prefer r1 > instruct/chat > gemma > other
989
+ def _score(m):
990
+ ml = m.lower()
991
+ s = 0
992
+ if "r1" in ml: s += 100
993
+ if "instruct" in ml: s += 80
994
+ if "chat" in ml: s += 70
995
+ if "gemma" in ml: s += 60
996
+ if "qwen" in ml: s += 55
997
+ if "llama" in ml: s += 50
998
+ if "phi" in ml: s += 45
999
+ if "deepseek" in ml: s += 40
1000
+ if "mistral" in ml: s += 35
1001
+ return s
1002
+ best = max(candidates, key=_score)
1003
+ cfg["model"] = best
1004
+ _LOG.info("Backup chain: replaced base/missing model '%s' → '%s'", cur_m, best)
1005
+ except Exception:
1006
+ pass
1007
+ enhanced_chain.append((cfg, entry))
1008
+
1009
+ return enhanced_chain
1010
+ except Exception:
1011
+ return []
1012
+
1013
+
1014
+ # ------------------------------------------------------- request tuning --
1015
+ def request_timeouts(config=None, size_class="normal"):
1016
+ """(connect_timeout, idle_read_timeout, total_timeout) in seconds for
1017
+ one model attempt. `size_class` is one of 'simple' | 'normal' |
1018
+ 'large' and maps to the configurable total deadlines in config.py.
1019
+ The idle cap never exceeds the total deadline."""
1020
+ try:
1021
+ if config is None:
1022
+ from . import config as _cfgmod
1023
+ cfg = _cfgmod.get_config()
1024
+ else:
1025
+ cfg = config
1026
+ # Normalize dict vs object access
1027
+ if isinstance(cfg, dict):
1028
+ get = lambda k, d: cfg.get(k, d)
1029
+ else:
1030
+ get = lambda k, d: getattr(cfg, k, d)
1031
+ totals = {
1032
+ "simple": get("timeout_simple", 30),
1033
+ "normal": get("timeout_normal", 60),
1034
+ "large": get("timeout_large", 300),
1035
+ }
1036
+ total = float(totals.get(str(size_class), get("timeout_normal", 60)))
1037
+ idle = float(get("model_idle_timeout", 45))
1038
+ # Ollama local needs longer for model load on cold start (especially 7B+ on CPU)
1039
+ prov = ""
1040
+ try:
1041
+ if isinstance(cfg, dict):
1042
+ prov = cfg.get("provider","") or cfg.get("default_ai_provider","")
1043
+ else:
1044
+ prov = getattr(cfg, "default_ai_provider", "") or getattr(cfg, "provider","")
1045
+ if not prov and isinstance(config, dict):
1046
+ prov = config.get("provider","")
1047
+ if str(prov).lower() == "ollama":
1048
+ total = max(total, 180.0)
1049
+ idle = max(idle, 120.0)
1050
+ except Exception:
1051
+ pass
1052
+ # Cloud providers connect timeout should be resilient against latency/proxies
1053
+ connect = max(5.0, min(15.0, idle))
1054
+ # Ollama connect needs longer on cold start (model load from disk into RAM/VRAM)
1055
+ try:
1056
+ if str(prov).lower() == "ollama":
1057
+ connect = max(connect, 35.0)
1058
+ except Exception:
1059
+ pass
1060
+ # Reasoning models (o1, o3, deepseek-r1, Claude thinking) take 30-90s before first token
1061
+ try:
1062
+ mod = ""
1063
+ if isinstance(cfg, dict):
1064
+ mod = cfg.get("model", "") or cfg.get("default_model", "")
1065
+ else:
1066
+ mod = getattr(cfg, "model", "") or getattr(cfg, "default_model", "")
1067
+ if not mod and isinstance(config, dict):
1068
+ mod = config.get("model", "")
1069
+ if any(k in str(mod).lower() for k in ("o1", "o3", "r1", "reason", "thinking", "3-7", "3.7")):
1070
+ idle = max(idle, 120.0)
1071
+ total = max(total, 180.0)
1072
+ except Exception:
1073
+ pass
1074
+ idle = max(5.0, min(idle, total))
1075
+ return connect, idle, max(5.0, total)
1076
+ except Exception:
1077
+ return 15.0, 60.0, 180.0
1078
+
1079
+
1080
+ _RETRY_STATE_LOCK = threading.Lock()
1081
+ _PROVIDER_LAST_FAILURE = {} # (provider, model) -> monotonic time
1082
+
1083
+
1084
+ def _provider_key(cfg):
1085
+ return (str((cfg or {}).get("provider") or "?"),
1086
+ str((cfg or {}).get("model") or "?"))
1087
+
1088
+
1089
+ def _mark_provider_failed(cfg):
1090
+ try:
1091
+ with _RETRY_STATE_LOCK:
1092
+ _PROVIDER_LAST_FAILURE[_provider_key(cfg)] = time.monotonic()
1093
+ except Exception:
1094
+ pass
1095
+
1096
+
1097
+ def _mark_provider_ok(cfg):
1098
+ try:
1099
+ with _RETRY_STATE_LOCK:
1100
+ _PROVIDER_LAST_FAILURE.pop(_provider_key(cfg), None)
1101
+ except Exception:
1102
+ pass
1103
+
1104
+
1105
+ def _provider_in_cooldown(cfg):
1106
+ try:
1107
+ from . import config as _cfgmod
1108
+ cooldown = float(getattr(_cfgmod.get_config(), "model_provider_cooldown", 20.0))
1109
+ except Exception:
1110
+ cooldown = 20.0
1111
+ with _RETRY_STATE_LOCK:
1112
+ last = _PROVIDER_LAST_FAILURE.get(_provider_key(cfg))
1113
+ if last is None:
1114
+ return False
1115
+ return (time.monotonic() - last) < cooldown
1116
+
1117
+
1118
+ def _backoff_sleep(attempt_index):
1119
+ """Exponential backoff between same-provider retries: base *
1120
+ 2**attempt, capped at 4 s. Short by design — instant hammering
1121
+ feels broken, multi-second stalls feel frozen."""
1122
+ try:
1123
+ from . import config as _cfgmod
1124
+ base = float(getattr(_cfgmod.get_config(), "model_retry_backoff_base", 0.6))
1125
+ except Exception:
1126
+ base = 0.6
1127
+ delay = min(4.0, base * (2 ** max(0, attempt_index)))
1128
+ try:
1129
+ time.sleep(delay)
1130
+ except Exception:
1131
+ pass
1132
+
1133
+
1134
+ def _failover_targets(primary_cfg=None, requirements=None):
1135
+ """The ordered list of (config, entry_or_None, is_backup) attempts for
1136
+ one logical request: the resolved primary first, then the enabled
1137
+ backup chain.
1138
+
1139
+ Integrated with CAT's Backup Provider & Resilience System:
1140
+ - Tier 1: Fallback models on the same provider.
1141
+ - Tier 2: Priority-ordered dynamic backup pool.
1142
+ - Intelligent capability matching (vision, tools, coding).
1143
+ - Health monitoring and circuit breaker cooldowns.
1144
+ """
1145
+ primary = dict(primary_cfg or load_config())
1146
+ prim_prov = str(primary.get("provider", "")).lower()
1147
+ if prim_prov and not (primary.get("api_key") or "").strip():
1148
+ try:
1149
+ from .providers.provider_manager import get_env_api_key
1150
+ env_k = get_env_api_key(prim_prov)
1151
+ if env_k:
1152
+ primary["api_key"] = env_k
1153
+ except Exception:
1154
+ pass
1155
+
1156
+ try:
1157
+ from .resilience.failover_engine import get_failover_engine
1158
+ from .resilience.types import TaskRequirements
1159
+ from .model_router import get_privacy_policy
1160
+
1161
+ engine = get_failover_engine()
1162
+ req = requirements if isinstance(requirements, TaskRequirements) else TaskRequirements()
1163
+ pol = get_privacy_policy()
1164
+
1165
+ candidates = engine.build_failover_targets(
1166
+ primary_config=primary,
1167
+ requirements=req,
1168
+ privacy_policy=pol,
1169
+ )
1170
+ if candidates:
1171
+ return [(dict(c.config), c.entry, c.is_backup) for c in candidates]
1172
+ except Exception:
1173
+ pass
1174
+
1175
+ # Fallback to direct walk if resilience engine unavailable
1176
+ chain = [(primary, None, False)]
1177
+ prim_url = str(primary.get("base_url", "")).rstrip("/")
1178
+ prim_model = str(primary.get("model", "")).lower()
1179
+ for bcfg, entry in _backup_chain():
1180
+ try:
1181
+ b_prov = str(bcfg.get("provider", "")).lower()
1182
+ b_url = str(bcfg.get("base_url", "")).rstrip("/")
1183
+ b_model = str(bcfg.get("model", "")).lower()
1184
+ if b_prov == prim_prov and b_url == prim_url and b_model == prim_model:
1185
+ continue
1186
+ needs_key = bcfg.get("needs_key", True)
1187
+ if b_prov == "ollama":
1188
+ needs_key = False
1189
+ if needs_key and not (bcfg.get("api_key") or "").strip():
1190
+ try:
1191
+ from .providers.provider_manager import get_env_api_key
1192
+ bk_key = get_env_api_key(b_prov)
1193
+ if bk_key:
1194
+ bcfg["api_key"] = bk_key
1195
+ except Exception:
1196
+ pass
1197
+ if needs_key and not (bcfg.get("api_key") or "").strip():
1198
+ continue
1199
+ except Exception:
1200
+ pass
1201
+ chain.append((dict(bcfg), entry, True))
1202
+
1203
+ fresh = [c for c in chain if not c[2] or not _provider_in_cooldown(c[0])]
1204
+ cooled = [c for c in chain if c not in fresh]
1205
+ return fresh + cooled
1206
+
1207
+
1208
+ def _attempts_per_provider():
1209
+ try:
1210
+ from . import config as _cfgmod
1211
+ n = int(getattr(_cfgmod.get_config(), "model_max_retries_per_provider", 1))
1212
+ except Exception:
1213
+ n = 1
1214
+ return max(1, n) + 1 # configured retries PLUS the initial attempt
1215
+
1216
+
1217
+ # ----------------------------------------------------- cancellation kit --
1218
+ _ACTIVE_LOCK = threading.Lock()
1219
+ _ACTIVE_RESPONSES = set()
1220
+ _ACTIVE_PROVIDERS = set()
1221
+ _CANCELLED_AT = 0.0 # wall-clock stamp of the last cancel_active_requests()
1222
+
1223
+
1224
+ def _register_response(resp):
1225
+ try:
1226
+ with _ACTIVE_LOCK:
1227
+ _ACTIVE_RESPONSES.add(resp)
1228
+ except Exception:
1229
+ pass
1230
+
1231
+
1232
+ def _unregister_response(resp):
1233
+ try:
1234
+ with _ACTIVE_LOCK:
1235
+ _ACTIVE_RESPONSES.discard(resp)
1236
+ except Exception:
1237
+ pass
1238
+
1239
+
1240
+ def _register_provider(prov):
1241
+ try:
1242
+ with _ACTIVE_LOCK:
1243
+ _ACTIVE_PROVIDERS.add(prov)
1244
+ except Exception:
1245
+ pass
1246
+
1247
+
1248
+ def _unregister_provider(prov):
1249
+ try:
1250
+ with _ACTIVE_LOCK:
1251
+ _ACTIVE_PROVIDERS.discard(prov)
1252
+ except Exception:
1253
+ pass
1254
+
1255
+
1256
+ def cancel_active_requests():
1257
+ """Close every in-flight provider socket so a blocked iter_lines()
1258
+ read raises immediately instead of waiting out its idle timeout.
1259
+ Called by the UI when the user hits Ctrl+C / Stop. Safe to call
1260
+ when nothing is running."""
1261
+ global _CANCELLED_AT
1262
+ _CANCELLED_AT = time.time()
1263
+ with _ACTIVE_LOCK:
1264
+ current = list(_ACTIVE_RESPONSES)
1265
+ providers = list(_ACTIVE_PROVIDERS)
1266
+ for resp in current:
1267
+ try:
1268
+ resp.close()
1269
+ except Exception:
1270
+ pass
1271
+ for prov in providers:
1272
+ try:
1273
+ if hasattr(prov, "cancel_active"):
1274
+ prov.cancel_active()
1275
+ except Exception:
1276
+ pass
1277
+ return len(current) + len(providers)
1278
+
1279
+
1280
+ def _just_cancelled():
1281
+ """True within a short window after cancel_active_requests() — used
1282
+ to convert the close-induced socket exception into a clean stop
1283
+ rather than a scary error message."""
1284
+ return (time.time() - _CANCELLED_AT) < 3.0
1285
+
1286
+
1287
+ # ------------------------------------------------- structured req log --
1288
+ _REQUEST_LOG = os.path.join(os.path.expanduser("~"), ".cct_requests.log")
1289
+ _REQUEST_LOG_MAX = 512 * 1024
1290
+ _REQLOG_LOCK = threading.Lock()
1291
+
1292
+
1293
+ def log_request_event(request_id, event, provider=None, model=None,
1294
+ size_class=None, elapsed_ms=None, detail=None,
1295
+ chars=None, retry_count=None, fallback=None):
1296
+ """One JSON line per lifecycle event into ~/.cct_requests.log
1297
+ (rotated at ~512 KB). Records WHERE a request stopped and why —
1298
+ the diagnosis tool the permanent-'...' bug always needed. Never
1299
+ logs API keys, headers or prompt contents."""
1300
+ record = {"ts": round(time.time(), 3), "request_id": request_id,
1301
+ "event": event}
1302
+ if provider:
1303
+ record["provider"] = provider
1304
+ if model:
1305
+ record["model"] = model
1306
+ if size_class:
1307
+ record["size_class"] = size_class
1308
+ if elapsed_ms is not None:
1309
+ record["elapsed_ms"] = int(elapsed_ms)
1310
+ if chars is not None:
1311
+ record["chars"] = chars
1312
+ if retry_count is not None:
1313
+ record["retry_count"] = retry_count
1314
+ if fallback is not None:
1315
+ record["fallback"] = bool(fallback)
1316
+ if detail:
1317
+ record["detail"] = str(detail)[:200]
1318
+ line = json.dumps(record, ensure_ascii=True, default=str)
1319
+ try:
1320
+ with _REQLOG_LOCK:
1321
+ try:
1322
+ if os.path.exists(_REQUEST_LOG) and \
1323
+ os.path.getsize(_REQUEST_LOG) > _REQUEST_LOG_MAX:
1324
+ os.replace(_REQUEST_LOG, _REQUEST_LOG + ".old")
1325
+ except Exception:
1326
+ pass
1327
+ with open(_REQUEST_LOG, "a", encoding="utf-8") as f:
1328
+ f.write(line + "\n")
1329
+ except Exception:
1330
+ pass
1331
+ try:
1332
+ _LOG.debug("aicore.request %s", line)
1333
+ except Exception:
1334
+ pass
1335
+
1336
+
1337
+ def _query_ai_once(prompt, system_prompt=DEFAULT_SYSTEM_PROMPT,
1338
+ history=None, config=None, attachments=None,
1339
+ size_class="normal"):
1340
+ """Single-attempt blocking chat completion against ONE provider
1341
+ config. Implemented as a consumer of _stream_ai_once so every
1342
+ api_style branch lives in exactly one place — the concatenation of
1343
+ a streamed reply is byte-identical to the blocking reply, and both
1344
+ surface provider failures as the same error strings, which keeps
1345
+ the failover detection in query_ai consistent."""
1346
+ pieces = []
1347
+ _metrics_note_model_start(config)
1348
+ for piece in _stream_ai_once(prompt, system_prompt=system_prompt,
1349
+ history=history, config=config,
1350
+ attachments=attachments,
1351
+ size_class=size_class):
1352
+ pieces.append(piece)
1353
+ return "".join(pieces)
1354
+
1355
+
1356
+ def query_ai(prompt, system_prompt=DEFAULT_SYSTEM_PROMPT,
1357
+ history=None, config=None, on_failover=None, attachments=None,
1358
+ size_class="normal", requirements=None):
1359
+ """`history` is an optional list of (role, text) pairs — or {"role","text"}
1360
+ dicts — for every prior turn that should stay in context, oldest first.
1361
+ Every branch below sends it to the provider in whatever shape that
1362
+ provider's API expects; omit it (or pass None/[]) for a genuinely
1363
+ one-shot call. See ChatSession.as_prompt_history() for the usual source.
1364
+
1365
+ v0.7.8: on primary-provider failure (quota/timeout/rate-limit/offline)
1366
+ transparently falls back to the enabled backup chain, notifying the
1367
+ installed failover hook at each step. `on_failover` overrides the
1368
+ global hook for this one call when given.
1369
+
1370
+ v0.7.8.1: `attachments` is an optional list of Attachment objects
1371
+ (calc_terminal/attachments.py). They are verified before the request
1372
+ and sent natively (images) or as a text context block, per provider
1373
+ capability.
1374
+
1375
+ v0.7.9.5 request lifecycle: `size_class` ('simple' | 'normal' |
1376
+ 'large') picks the configurable total deadline; each provider gets
1377
+ bounded retries with exponential backoff for TRANSIENT failures;
1378
+ quota/config failures move on immediately; recently-failed providers
1379
+ sit out via cooldown until the chain exhausts.
1380
+
1381
+ Returns the assistant text, or (on total failure) ONE of the
1382
+ _ERROR_SIGNATURES strings — loading states can always key off those,
1383
+ and callers can always tell a real answer from a failure."""
1384
+ hook_override = on_failover or _FAILOVER_HOOK
1385
+
1386
+ def notify(message):
1387
+ try:
1388
+ if hook_override is not None:
1389
+ hook_override(message)
1390
+ except Exception:
1391
+ pass
1392
+
1393
+ targets = _failover_targets(config, requirements=requirements)
1394
+ max_attempts = _attempts_per_provider()
1395
+ result = ""
1396
+ fallback_used = False
1397
+
1398
+ for t_idx, (cfg, entry, is_backup) in enumerate(targets):
1399
+ provider = cfg.get("provider", "?")
1400
+ model = cfg.get("model", "")
1401
+ try:
1402
+ from .resilience.health_monitor import get_health_monitor
1403
+ get_health_monitor().record_turn_start(provider, model)
1404
+ except Exception:
1405
+ pass
1406
+ t_turn_start = time.time()
1407
+ for attempt in range(max_attempts):
1408
+ if attempt:
1409
+ log_request_event(_short_id(), "retry_same_provider",
1410
+ provider=provider,
1411
+ model=cfg.get("model"),
1412
+ size_class=size_class,
1413
+ retry_count=attempt)
1414
+ result = _query_ai_once(prompt, system_prompt=system_prompt,
1415
+ history=history, config=cfg,
1416
+ attachments=attachments,
1417
+ size_class=size_class)
1418
+ if not _should_failover(result):
1419
+ if is_backup:
1420
+ _mark_backup_success(entry)
1421
+ notify("\u2713 Connected successfully.")
1422
+ _mark_provider_ok(cfg)
1423
+ try:
1424
+ from .resilience.health_monitor import get_health_monitor
1425
+ lat = (time.time() - t_turn_start) * 1000.0
1426
+ get_health_monitor().record_success(provider, model, latency_ms=lat)
1427
+ except Exception:
1428
+ pass
1429
+ log_request_event(_short_id(), "finish",
1430
+ provider=provider, model=cfg.get("model"),
1431
+ size_class=size_class, chars=len(result),
1432
+ fallback=fallback_used)
1433
+ return result
1434
+ # Failure. Transient? → brief backoff, retry same provider.
1435
+ if attempt + 1 < max_attempts and _retryable_failure(result):
1436
+ log_request_event(_short_id(), "attempt_failed_retryable",
1437
+ provider=provider, detail=result[:120],
1438
+ size_class=size_class)
1439
+ _backoff_sleep(attempt)
1440
+ continue
1441
+ break # non-retryable → next provider
1442
+ _mark_provider_failed(cfg)
1443
+ try:
1444
+ from .resilience.health_monitor import get_health_monitor
1445
+ from .resilience.failover_engine import get_failover_engine
1446
+ from .resilience.types import FailureType
1447
+ ft = get_failover_engine().classify_failure(result)
1448
+ get_health_monitor().record_failure(
1449
+ provider, model, error_message=result,
1450
+ is_rate_limit=(ft == FailureType.RATE_LIMIT)
1451
+ )
1452
+ except Exception:
1453
+ pass
1454
+ log_request_event(_short_id(), "provider_exhausted",
1455
+ provider=provider, detail=result[:160],
1456
+ size_class=size_class, fallback=True)
1457
+ if t_idx + 1 < len(targets):
1458
+ fallback_used = True
1459
+ nxt_cfg = targets[t_idx + 1][0]
1460
+ nxt = nxt_cfg.get("provider", "?")
1461
+ nxt_m = nxt_cfg.get("model", "")
1462
+ cur_m = cfg.get("model", "")
1463
+ cur_desc = f"{provider} ({cur_m})" if cur_m else provider
1464
+ nxt_desc = f"{nxt} ({nxt_m})" if nxt_m else nxt
1465
+ notify(f"\u26a0 Provider issue with {cur_desc}. Switching to Backup "
1466
+ f"Provider ({nxt_desc})...")
1467
+
1468
+ # Every provider failed — surface the last failure message so the UI
1469
+ # shows a real error card instead of waiting forever.
1470
+ log_request_event(_short_id(), "all_providers_failed",
1471
+ detail=result[:200], size_class=size_class)
1472
+ return result
1473
+
1474
+
1475
+ def _short_id():
1476
+ """Short unique id for log correlation."""
1477
+ return uuid.uuid4().hex[:12]
1478
+
1479
+
1480
+ def _mark_backup_success(entry):
1481
+ try:
1482
+ from .providers.provider_manager import mark_backup_used
1483
+ mark_backup_used(entry)
1484
+ except Exception:
1485
+ pass
1486
+
1487
+
1488
+ def _peek_first(generator):
1489
+ """Pull the first item from a generator, returning (item, generator)
1490
+ so callers can inspect it (failover decision) and still stream the
1491
+ rest exactly as the underlying generator produced it.
1492
+
1493
+ The returned generator is the ORIGINAL generator, already advanced
1494
+ past the first item — it must NOT include `first` again, because
1495
+ every caller here does `yield first; yield from rest` and re-including
1496
+ it would emit the first chunk twice ("hellohello"). The previous
1497
+ `chain([first], generator)` implementation caused exactly that: every
1498
+ streamed reply started with its first fragment duplicated."""
1499
+ try:
1500
+ first = next(generator)
1501
+ except StopIteration:
1502
+ return None, iter(())
1503
+ return first, generator
1504
+
1505
+
1506
+ def stream_ai(prompt, system_prompt=DEFAULT_SYSTEM_PROMPT,
1507
+ history=None, config=None, on_failover=None, attachments=None,
1508
+ size_class="normal", requirements=None):
1509
+ """Generator version of query_ai — yields text fragments as they
1510
+ arrive instead of returning one finished string, so the primary UI
1511
+ (calc_terminal/ui/) can grow a ConversationItem token-by-token
1512
+ instead of freezing until the whole reply lands.
1513
+
1514
+ v0.7.8 failover: if the first fragment from the primary provider is a
1515
+ provider-failure signature, the whole stream is transparently retried
1516
+ against the next healthy backup instead of showing the error.
1517
+
1518
+ v0.7.9.5 request lifecycle hardening:
1519
+ * NOTHING is yielded until real content arrives, so a failing
1520
+ provider can never leave a permanent '...' on screen — the walk
1521
+ moves to backups first.
1522
+ * A failure BEFORE any token → bounded same-provider retries with
1523
+ exponential backoff, then the next provider (cooldown-aware).
1524
+ * A failure MID-STREAM (tokens already delivered) does NOT
1525
+ silently switch providers and re-run the whole answer — the
1526
+ partial reply is kept and the loss is disclosed in one honest
1527
+ line.
1528
+ * Every lifecycle step is written to ~/.cct_requests.log.
1529
+
1530
+ v0.7.8.1: `attachments` — see query_ai; verified before each attempt,
1531
+ attached natively (vision providers) or as text context."""
1532
+ hook_override = on_failover or _FAILOVER_HOOK
1533
+
1534
+ def notify(message):
1535
+ try:
1536
+ if hook_override is not None:
1537
+ hook_override(message)
1538
+ except Exception:
1539
+ pass
1540
+
1541
+ targets = _failover_targets(config, requirements=requirements)
1542
+ max_attempts = _attempts_per_provider()
1543
+ fallback_used = False
1544
+ last_error_piece = None
1545
+
1546
+ for t_idx, (cfg, entry, is_backup) in enumerate(targets):
1547
+ provider = cfg.get("provider", "?")
1548
+ for attempt in range(max_attempts):
1549
+ _metrics_note_model_start(cfg)
1550
+ got_content = False
1551
+ error_piece = None
1552
+ for piece in _stream_ai_once(prompt, system_prompt=system_prompt,
1553
+ history=history, config=cfg,
1554
+ attachments=attachments,
1555
+ size_class=size_class):
1556
+ if not got_content:
1557
+ if is_error_response(piece):
1558
+ error_piece = piece
1559
+ break
1560
+ else:
1561
+ # Once streaming has begun, individual fragments (including whitespace/newlines)
1562
+ # are NOT provider failures. Only known error signatures break the stream.
1563
+ t = piece.strip() if isinstance(piece, str) else ""
1564
+ if t and (t.startswith(_ERROR_SIGNATURES) or t.endswith(_ERROR_SUFFIX)):
1565
+ error_piece = piece
1566
+ break
1567
+ # Real content: stream it out immediately (requirement:
1568
+ # first token replaces '...' right away).
1569
+ got_content = True
1570
+ yield piece
1571
+ if error_piece is None:
1572
+ if got_content:
1573
+ if is_backup:
1574
+ _mark_backup_success(entry)
1575
+ notify("\u2713 Connected successfully.")
1576
+ _mark_provider_ok(cfg)
1577
+ log_request_event(_short_id(), "finish",
1578
+ provider=provider, model=cfg.get("model"),
1579
+ size_class=size_class, fallback=fallback_used)
1580
+ return
1581
+ # Generator ended with zero tokens and no signature —
1582
+ # treat as a provider-level empty response.
1583
+ error_piece = "No model response received."
1584
+ if got_content:
1585
+ # Mid-stream loss: keep the partial answer, disclose the
1586
+ # drop, do NOT duplicate the reply from another provider.
1587
+ _mark_provider_failed(cfg)
1588
+ log_request_event(_short_id(), "midstream_drop",
1589
+ provider=provider, detail=error_piece[:120],
1590
+ size_class=size_class)
1591
+ yield ("\n\n\u26a0 Connection lost mid-response \u2014 partial "
1592
+ "answer kept. Send again to continue.")
1593
+ return
1594
+ last_error_piece = error_piece
1595
+ if attempt + 1 < max_attempts and _retryable_failure(error_piece):
1596
+ log_request_event(_short_id(), "attempt_failed_retryable",
1597
+ provider=provider, detail=error_piece[:120],
1598
+ size_class=size_class)
1599
+ _backoff_sleep(attempt)
1600
+ continue
1601
+ break # non-retryable → next provider
1602
+ _mark_provider_failed(cfg)
1603
+ log_request_event(_short_id(), "provider_exhausted",
1604
+ provider=provider, detail=(error_piece or "")[:160],
1605
+ size_class=size_class, fallback=True)
1606
+ if t_idx + 1 < len(targets):
1607
+ fallback_used = True
1608
+ nxt_cfg = targets[t_idx + 1][0]
1609
+ nxt = nxt_cfg.get("provider", "?")
1610
+ nxt_m = nxt_cfg.get("model", "")
1611
+ cur_m = cfg.get("model", "")
1612
+ cur_desc = f"{provider} ({cur_m})" if cur_m else provider
1613
+ nxt_desc = f"{nxt} ({nxt_m})" if nxt_m else nxt
1614
+ notify(f"\u26a0 Provider issue with {cur_desc}. Switching to Backup "
1615
+ f"Provider ({nxt_desc})...")
1616
+
1617
+ # Nothing received anywhere — surface the final failure message so
1618
+ # the UI renders an error card instead of an eternal spinner.
1619
+ yield last_error_piece or "No model response received."
1620
+
1621
+
1622
+ def _prepare_attachments(attachments, config):
1623
+ """v0.7.8.1: verify attachment objects (spec section 7) and decide
1624
+ whether this provider receives them natively (vision-capable) or as
1625
+ text context. Returns the `vision` flag consumed by the request
1626
+ builders. Emits ProviderAdapter verification lines to the debug log
1627
+ — never file contents."""
1628
+ if not attachments:
1629
+ return False
1630
+ try:
1631
+ from . import attachments as _att
1632
+ except Exception:
1633
+ return False
1634
+ _att.AttachmentManager.verify(
1635
+ attachments, log=lambda m: _LOG.debug("ProviderAdapter: %s", m))
1636
+ vision = False
1637
+ try:
1638
+ vision = bool(_att.provider_capabilities(config).get("vision"))
1639
+ except Exception:
1640
+ vision = False
1641
+ has_image = any(
1642
+ _att.attachment_has_image_payload(a) for a in attachments if a is not None)
1643
+ if vision and has_image:
1644
+ _LOG.debug("ProviderAdapter: native image payload attached")
1645
+ else:
1646
+ _LOG.debug("ProviderAdapter: attachment_context = present (textual block)")
1647
+ return vision
1648
+
1649
+
1650
+ def _metrics_note_model_start(config=None):
1651
+ """Best-effort metrics/event hooks for one model request (never raises,
1652
+ never slows the path down meaningfully)."""
1653
+ try:
1654
+ from . import metrics as _m
1655
+ m = _m.current()
1656
+ if m is not None:
1657
+ m.count_model_call()
1658
+ m.stage_start("model_first_token")
1659
+ if config and not m.model_used:
1660
+ m.model_used = f"{config.get('provider', '?')}/{config.get('model', '?')}"
1661
+ except Exception:
1662
+ pass
1663
+ try:
1664
+ from .event_stream import stream, MODEL_REQUEST_STARTED
1665
+ stream.emit(MODEL_REQUEST_STARTED, source="aicore",
1666
+ provider=(config or {}).get("provider"),
1667
+ model=(config or {}).get("model"))
1668
+ except Exception:
1669
+ pass
1670
+
1671
+
1672
+ def _metrics_note_first_token():
1673
+ """Called on the first REAL streamed token of a completion."""
1674
+ try:
1675
+ from . import metrics as _m
1676
+ m = _m.current()
1677
+ if m is not None:
1678
+ m.note_first_token()
1679
+ m.stage_end("model_first_token")
1680
+ except Exception:
1681
+ pass
1682
+ try:
1683
+ from .event_stream import stream, MODEL_FIRST_TOKEN
1684
+ stream.emit(MODEL_FIRST_TOKEN, source="aicore")
1685
+ except Exception:
1686
+ pass
1687
+
1688
+
1689
+ class _TotalTimeout(Exception):
1690
+ """Raised between stream chunks when the TOTAL request deadline
1691
+ expires (distinct from requests' idle read timeout)."""
1692
+
1693
+
1694
+ def _stream_ai_once(prompt, system_prompt=DEFAULT_SYSTEM_PROMPT,
1695
+ history=None, config=None, attachments=None,
1696
+ size_class="normal", request_id=None):
1697
+ """Single-attempt streaming generator against ONE provider config —
1698
+ the core of the public stream_ai; see its docstring for streaming
1699
+ behavior semantics. `config` defaults to the saved primary config;
1700
+ the v0.7.8 failover wrapper retries this against each backup.
1701
+
1702
+ v0.7.8.1: `attachments` (Attachment objects) are verified here
1703
+ (spec section 7), then folded in by the request builder — natively
1704
+ as image parts when the provider is vision-capable, otherwise as
1705
+ text via the attachment manager's context block.
1706
+
1707
+ v0.7.9.5 lifecycle:
1708
+ * `size_class` selects the configurable total timeout.
1709
+ * Every HTTP call uses a (connect, idle) timeout tuple; the total
1710
+ deadline is enforced BETWEEN chunks via _TotalTimeout, so an
1711
+ actively streaming model is never cut off mid-tokens but a
1712
+ silent one can't hold the UI forever either.
1713
+ * The response socket is registered for cancel_active_requests().
1714
+ * start / first_token / finish / error are logged structurally."""
1715
+ request_id = request_id or _short_id()
1716
+ t_start = time.monotonic()
1717
+
1718
+ def elapsed_ms():
1719
+ return int((time.monotonic() - t_start) * 1000)
1720
+
1721
+ if not _HAS_REQUESTS:
1722
+ yield "The 'requests' library is required for AI features. Please run: pip install requests"
1723
+ return
1724
+
1725
+ config = config or load_config()
1726
+ provider = config.get("provider")
1727
+ if not provider:
1728
+ yield "AI not configured. Run /ai or /agent to configure your provider."
1729
+ return
1730
+
1731
+ api_style, base_url, api_key, model, temperature, extra_headers = _resolve_provider(config)
1732
+ history = _sanitize_history(history)
1733
+ usage_prompt_text = prompt if not history else "\n".join(t for _, t in history) + "\n" + prompt
1734
+
1735
+ vision = _prepare_attachments(attachments, config)
1736
+
1737
+ # ---- lifecycle instrumentation (v0.7.9.5) --------------------------
1738
+ t_connect_idle = request_timeouts(config, size_class)[:2]
1739
+ deadline = time.monotonic() + request_timeouts(config, size_class)[2]
1740
+ log_request_event(request_id, "start", provider=provider, model=model,
1741
+ size_class=size_class)
1742
+
1743
+ def _check_deadline():
1744
+ if time.monotonic() > deadline:
1745
+ raise _TotalTimeout()
1746
+
1747
+ def _lines_with_deadline(iterator):
1748
+ """Wraps resp.iter_lines() so the TOTAL deadline is enforced
1749
+ between chunks. Bytes keep flowing → no timeout (an actively
1750
+ streaming model is never misclassified as frozen); true silence
1751
+ is already bounded by the socket idle timeout."""
1752
+ for line in iterator:
1753
+ _check_deadline()
1754
+ yield line
1755
+
1756
+ _tracked = [] # registered responses, unregistered in finally
1757
+
1758
+ def _track(resp):
1759
+ _register_response(resp)
1760
+ _tracked.append(resp)
1761
+ return resp
1762
+
1763
+ class _FirstToken:
1764
+ fired = False
1765
+
1766
+ @classmethod
1767
+ def hit(cls):
1768
+ if not cls.fired:
1769
+ cls.fired = True
1770
+ _metrics_note_first_token()
1771
+ log_request_event(request_id, "first_token",
1772
+ provider=provider, model=model,
1773
+ size_class=size_class,
1774
+ elapsed_ms=elapsed_ms())
1775
+
1776
+ try:
1777
+ # v0.7.10: Validate base_url before making request
1778
+ if not base_url:
1779
+ yield "AI not configured — no base URL set for this provider. Run /ai to reconfigure."
1780
+ return
1781
+ if not model:
1782
+ yield "AI not configured — no model selected. Run /model to choose one."
1783
+ return
1784
+ # Upfront API key check so unconfigured providers fail over immediately
1785
+ _prov_info = _get_provider_info(provider)
1786
+ if _prov_info and _prov_info.get("needs_key") and not (api_key or "").strip():
1787
+ yield f"API key missing for provider '{provider}'. Run /key or /provider to configure."
1788
+ return
1789
+
1790
+ if api_style == "openai":
1791
+ url = f"{base_url}/chat/completions"
1792
+ headers = {"Content-Type": "application/json"}
1793
+ if api_key:
1794
+ headers["Authorization"] = f"Bearer {api_key}"
1795
+ headers.update(extra_headers)
1796
+ payload = {
1797
+ "model": model,
1798
+ "messages": _openai_messages(system_prompt, history, prompt,
1799
+ attachments=attachments, vision=vision, model=model),
1800
+ "stream": True,
1801
+ }
1802
+ mod_lower = (model or "").lower()
1803
+ is_o_reasoning = ("o1" in mod_lower or "o3" in mod_lower) and "openrouter" not in str(base_url).lower()
1804
+ if is_o_reasoning:
1805
+ payload["max_completion_tokens"] = 4096
1806
+ elif temperature is not None:
1807
+ payload["temperature"] = temperature
1808
+ resp = _track(requests.post(url, headers=headers, json=payload,
1809
+ timeout=t_connect_idle, stream=True))
1810
+ _raise_for_status(resp)
1811
+ full = []
1812
+ usage = None
1813
+ for line in _lines_with_deadline(resp.iter_lines(decode_unicode=True)):
1814
+ if not line or not line.startswith("data:"):
1815
+ continue
1816
+ data_str = line[len("data:"):].strip()
1817
+ if data_str == "[DONE]":
1818
+ break
1819
+ try:
1820
+ obj = json.loads(data_str)
1821
+ except ValueError:
1822
+ continue
1823
+ choices = obj.get("choices") or []
1824
+ if choices:
1825
+ delta = choices[0].get("delta") or {}
1826
+ piece = delta.get("content")
1827
+ if piece:
1828
+ full.append(piece)
1829
+ _FirstToken.hit()
1830
+ yield piece
1831
+ if isinstance(obj.get("usage"), dict):
1832
+ usage = obj["usage"]
1833
+ completion_text = "".join(full)
1834
+ if usage:
1835
+ record_usage(usage.get("prompt_tokens", estimate_tokens(usage_prompt_text)),
1836
+ usage.get("completion_tokens", estimate_tokens(completion_text)))
1837
+ else:
1838
+ record_usage(estimate_tokens(usage_prompt_text), estimate_tokens(completion_text))
1839
+ return
1840
+
1841
+ elif api_style == "anthropic":
1842
+ headers = {
1843
+ "x-api-key": api_key,
1844
+ "anthropic-version": "2023-06-01",
1845
+ "content-type": "application/json",
1846
+ }
1847
+ headers.update(extra_headers)
1848
+ payload = {
1849
+ "model": model,
1850
+ "max_tokens": 4096,
1851
+ "system": system_prompt,
1852
+ "messages": _anthropic_messages(history, prompt,
1853
+ attachments=attachments, vision=vision),
1854
+ "stream": True,
1855
+ }
1856
+ if temperature is not None:
1857
+ payload["temperature"] = temperature
1858
+ resp = _track(requests.post(f"{base_url}/messages", headers=headers,
1859
+ json=payload, timeout=t_connect_idle,
1860
+ stream=True))
1861
+ _raise_for_status(resp)
1862
+ full = []
1863
+ in_tokens = out_tokens = None
1864
+ for line in _lines_with_deadline(resp.iter_lines(decode_unicode=True)):
1865
+ if not line or not line.startswith("data:"):
1866
+ continue
1867
+ try:
1868
+ obj = json.loads(line[len("data:"):].strip())
1869
+ except ValueError:
1870
+ continue
1871
+ etype = obj.get("type")
1872
+ if etype == "content_block_delta":
1873
+ piece = (obj.get("delta") or {}).get("text")
1874
+ if piece:
1875
+ full.append(piece)
1876
+ _FirstToken.hit()
1877
+ yield piece
1878
+ elif etype == "message_start":
1879
+ u = (obj.get("message") or {}).get("usage") or {}
1880
+ in_tokens = u.get("input_tokens", in_tokens)
1881
+ elif etype == "message_delta":
1882
+ u = obj.get("usage") or {}
1883
+ out_tokens = u.get("output_tokens", out_tokens)
1884
+ completion_text = "".join(full)
1885
+ record_usage(in_tokens if in_tokens is not None else estimate_tokens(usage_prompt_text),
1886
+ out_tokens if out_tokens is not None else estimate_tokens(completion_text))
1887
+ return
1888
+
1889
+ elif api_style == "gemini":
1890
+ clean_model = model.removeprefix("models/")
1891
+ url = f"{base_url}/models/{clean_model}:streamGenerateContent?alt=sse&key={api_key}"
1892
+ payload = {
1893
+ "contents": _gemini_contents(history, prompt,
1894
+ attachments=attachments, vision=vision),
1895
+ "systemInstruction": {"parts": [{"text": system_prompt}]},
1896
+ "generationConfig": {
1897
+ "temperature": temperature if temperature is not None else 0.7,
1898
+ "maxOutputTokens": 4096,
1899
+ },
1900
+ }
1901
+ resp = _track(requests.post(url, json=payload, timeout=t_connect_idle,
1902
+ stream=True))
1903
+ _raise_for_status(resp)
1904
+ full = []
1905
+ usage_meta = None
1906
+ for line in _lines_with_deadline(resp.iter_lines(decode_unicode=True)):
1907
+ if not line or not line.startswith("data:"):
1908
+ continue
1909
+ try:
1910
+ obj = json.loads(line[len("data:"):].strip())
1911
+ except ValueError:
1912
+ continue
1913
+ candidates = obj.get("candidates") or []
1914
+ if candidates:
1915
+ parts = (candidates[0].get("content") or {}).get("parts") or []
1916
+ for part in parts:
1917
+ piece = part.get("text")
1918
+ if piece:
1919
+ full.append(piece)
1920
+ _FirstToken.hit()
1921
+ yield piece
1922
+ if isinstance(obj.get("usageMetadata"), dict):
1923
+ usage_meta = obj["usageMetadata"]
1924
+ completion_text = "".join(full)
1925
+ if usage_meta:
1926
+ record_usage(usage_meta.get("promptTokenCount", estimate_tokens(usage_prompt_text)),
1927
+ usage_meta.get("candidatesTokenCount", estimate_tokens(completion_text)))
1928
+ else:
1929
+ record_usage(estimate_tokens(usage_prompt_text), estimate_tokens(completion_text))
1930
+ return
1931
+
1932
+ elif api_style == "ollama":
1933
+ # Ensure Ollama is running — auto-start `ollama serve` if needed (fixes "Could not reach" when not running)
1934
+ try:
1935
+ from .ollama_download import ensure_ollama_running, is_ollama_installed
1936
+ ok, msg = ensure_ollama_running(base_url, timeout=3, auto_start=True)
1937
+ if not ok:
1938
+ hint = "Ollama not installed — install from https://ollama.com/download and run `ollama serve`" if not is_ollama_installed() else msg
1939
+ raise RuntimeError(f"Could not reach the AI server. Could not reach Ollama at {base_url}. {hint}. Or install a model via ☰ → Ollama Models.")
1940
+ except RuntimeError:
1941
+ raise
1942
+ except Exception:
1943
+ pass
1944
+
1945
+ from .models.profiles import get_model_profile
1946
+ from .providers.ollama_adapter import OllamaProvider
1947
+ from .ai_context import get_context_manager, AIRequest, AIMessage
1948
+ import uuid
1949
+
1950
+ profile = get_model_profile("ollama", model, config)
1951
+ options = profile.get_effective_options({"temperature": temperature})
1952
+
1953
+ # Detect current AI mode
1954
+ active_mode = "chat"
1955
+ if config and isinstance(config, dict) and config.get("mode"):
1956
+ active_mode = config.get("mode")
1957
+ else:
1958
+ try:
1959
+ from . import ai_modes as _am
1960
+ active_mode = _am.current_mode()
1961
+ except Exception:
1962
+ active_mode = "chat"
1963
+
1964
+ # Filter and budget history using AIContextManager to prevent cross-mode context pollution
1965
+ ctx_mgr = get_context_manager()
1966
+ ai_history = []
1967
+ for r, t in (history or []):
1968
+ ai_history.append(AIMessage(id=uuid.uuid4().hex, role=r, content=t, mode=active_mode))
1969
+
1970
+ req = AIRequest(
1971
+ session_id=str(getattr(config, "get", lambda k, d="": d)("session_id", "cct")),
1972
+ request_id=request_id or uuid.uuid4().hex,
1973
+ mode=active_mode,
1974
+ user_message=prompt,
1975
+ history=ai_history,
1976
+ model=model,
1977
+ provider="ollama",
1978
+ )
1979
+ built_ctx = ctx_mgr.build_context(req, profile=profile)
1980
+ messages = built_ctx.messages
1981
+
1982
+ # If caller supplied a custom system prompt (and not default), apply it to system role
1983
+ if system_prompt and system_prompt != DEFAULT_SYSTEM_PROMPT and messages and messages[0]["role"] == "system":
1984
+ if profile.is_small_or_base() and len(system_prompt) > 800:
1985
+ messages[0]["content"] = system_prompt[:800] + "\nAnswer concisely."
1986
+ else:
1987
+ messages[0]["content"] = system_prompt
1988
+
1989
+ ollama_prov = OllamaProvider(base_url)
1990
+ _register_provider(ollama_prov)
1991
+ full = []
1992
+ try:
1993
+ for piece in ollama_prov.stream_chat(
1994
+ model=model,
1995
+ messages=messages,
1996
+ options=options,
1997
+ timeout=t_connect_idle,
1998
+ request_id=request_id,
1999
+ deadline=deadline,
2000
+ ):
2001
+ _FirstToken.hit()
2002
+ full.append(piece)
2003
+ yield piece
2004
+ finally:
2005
+ _unregister_provider(ollama_prov)
2006
+
2007
+ completion_text = "".join(full)
2008
+ record_usage(estimate_tokens(usage_prompt_text), estimate_tokens(completion_text))
2009
+ return
2010
+
2011
+ else:
2012
+ yield f"Provider '{provider}' is not supported yet. Run /ai to reconfigure."
2013
+ return
2014
+
2015
+ except GeneratorExit:
2016
+ # Consumer stopped iterating (cancel / screen teardown). Run the
2017
+ # cleanup in finally and close the socket promptly.
2018
+ raise
2019
+ except _TotalTimeout:
2020
+ log_request_event(request_id, "error", provider=provider, model=model,
2021
+ size_class=size_class, elapsed_ms=elapsed_ms(),
2022
+ detail="total request deadline exceeded")
2023
+ yield ("The AI request timed out before the provider finished "
2024
+ "responding. Try again, or run `/model` to switch to a faster model.")
2025
+ except requests.exceptions.ConnectionError as ce:
2026
+ if _just_cancelled():
2027
+ return # user cancelled — stop cleanly, no scary message
2028
+ ce_str = str(ce).lower()
2029
+ log_request_event(request_id, "error", provider=provider, model=model,
2030
+ size_class=size_class, elapsed_ms=elapsed_ms(),
2031
+ detail=f"connection error: {ce_str[:120]}")
2032
+ if "ollama" in (provider or "").lower() or "localhost" in (base_url or ""):
2033
+ yield ("Could not reach your local Ollama server. "
2034
+ "Make sure Ollama is running (`ollama serve`), or run `/ai` to switch to a cloud provider.")
2035
+ else:
2036
+ yield (f"Could not reach {provider or 'the AI server'}. "
2037
+ "Check your internet connection and API URL, or run `/model` to switch providers.")
2038
+ except requests.exceptions.Timeout:
2039
+ if _just_cancelled():
2040
+ return
2041
+ log_request_event(request_id, "error", provider=provider, model=model,
2042
+ size_class=size_class, elapsed_ms=elapsed_ms(),
2043
+ detail="idle read timeout")
2044
+ if "ollama" in (provider or "").lower():
2045
+ yield ("Ollama took too long to respond — the model may be loading. "
2046
+ "Try sending your message again, or run `/model` to pick a smaller model.")
2047
+ else:
2048
+ yield (f"Request to {provider or 'AI'} timed out. "
2049
+ "Try again, use a faster model (`/model`), or check your network.")
2050
+ except RuntimeError as e:
2051
+ # v0.7.10: Better error messages for common HTTP errors
2052
+ if _just_cancelled():
2053
+ return
2054
+ err_msg = str(e)
2055
+ log_request_event(request_id, "error", provider=provider, model=model,
2056
+ size_class=size_class, elapsed_ms=elapsed_ms(),
2057
+ detail=err_msg[:200])
2058
+ if "410" in err_msg or "end of life" in err_msg.lower():
2059
+ yield (f"Model '{model}' has been retired by {provider}. "
2060
+ "Run `/model` to choose a current model.")
2061
+ elif "401" in err_msg or "unauthorized" in err_msg.lower():
2062
+ yield (f"Invalid API key for {provider}. "
2063
+ "Run `/ai` to reconfigure with a valid key.")
2064
+ elif "429" in err_msg or "rate limit" in err_msg.lower():
2065
+ yield (f"{provider} rate limit hit. Wait a moment and retry, "
2066
+ "or run `/model` to switch providers.")
2067
+ elif "402" in err_msg or "quota" in err_msg.lower() or "insufficient" in err_msg.lower():
2068
+ yield (f"{provider} quota exceeded. Run `/model` to switch to a free provider "
2069
+ "or add credits to your account.")
2070
+ elif "404" in err_msg or "not found" in err_msg.lower():
2071
+ yield (f"Model '{model}' not found on {provider}. "
2072
+ "Run `/model` to pick an available model.")
2073
+ elif "500" in err_msg or "internal server error" in err_msg.lower():
2074
+ yield (f"{provider} server error. This is usually temporary — "
2075
+ "try again in a moment, or run `/model` to switch.")
2076
+ else:
2077
+ yield f"Error from {provider}: {err_msg}"
2078
+ except Exception as e:
2079
+ if _just_cancelled():
2080
+ return
2081
+ log_request_event(request_id, "error", provider=provider, model=model,
2082
+ size_class=size_class, elapsed_ms=elapsed_ms(),
2083
+ detail=f"{type(e).__name__}: {e}")
2084
+ yield f"Error connecting to {provider or 'AI'}: {type(e).__name__}: {e}"
2085
+ finally:
2086
+ # ALWAYS drop the socket registrations — a finished or failed
2087
+ # request must never keep cancel_active_requests() holding stale
2088
+ # response objects (requirement: loading state always clears).
2089
+ for r in _tracked:
2090
+ _unregister_response(r)
2091
+ _tracked.clear()
2092
+
2093
+
2094
+ def _raise_for_status(resp):
2095
+ """Raise a clear error with the API's own message instead of a vague one.
2096
+
2097
+ Most providers return a helpful JSON body explaining *why* a call failed
2098
+ (bad model name, invalid key, quota), but requests only surfaces the HTTP
2099
+ code. This pulls that detail out so the user can actually fix the issue.
2100
+ """
2101
+ if resp.status_code >= 400:
2102
+ try:
2103
+ body = resp.json()
2104
+ err = (body.get("error", {}).get("message")
2105
+ or body.get("error", {}).get("status")
2106
+ or str(body)[:300])
2107
+ except Exception:
2108
+ err = resp.text[:300] if resp.text else resp.reason
2109
+ raise RuntimeError(f"HTTP {resp.status_code}: {err}")
2110
+
2111
+
2112
+ # ---------------------------------------------------------------- web search
2113
+ # Key-free web search using DuckDuckGo's HTML endpoint (no API key needed,
2114
+ # unlike Google/Bing search APIs). Used directly by /websearch and /research,
2115
+ # and wired into the agent as a tool so it can look things up mid-answer.
2116
+ def _unwrap_ddg_url(href):
2117
+ """DuckDuckGo's HTML results wrap outbound links in a redirect
2118
+ (/l/?uddg=<encoded-url>); unwrap that back to the real destination."""
2119
+ if href.startswith("//"):
2120
+ href = "https:" + href
2121
+ try:
2122
+ parsed = urlparse(href)
2123
+ if "duckduckgo.com" in parsed.netloc and parsed.path.startswith("/l/"):
2124
+ qs = parse_qs(parsed.query)
2125
+ if "uddg" in qs:
2126
+ return unquote(qs["uddg"][0])
2127
+ return href
2128
+ except Exception:
2129
+ return href
2130
+
2131
+
2132
+ _TAG_RE = re.compile(r"<.*?>", re.S)
2133
+
2134
+
2135
+ def web_search(query, max_results=5):
2136
+ """Best-effort web search, no API key required. Returns a list of
2137
+ {\"title\", \"url\", \"snippet\"} dicts, or [] if unreachable/blocked —
2138
+ callers should treat an empty list as 'couldn't search right now', not
2139
+ an error."""
2140
+ query = (query or "").strip()
2141
+ if not query or not _HAS_REQUESTS:
2142
+ return []
2143
+ try:
2144
+ resp = requests.post(
2145
+ "https://html.duckduckgo.com/html/",
2146
+ data={"q": query}, timeout=12,
2147
+ headers={"User-Agent": "Mozilla/5.0 (compatible; CAT/0.7.9.0)"},
2148
+ )
2149
+ if resp.status_code >= 400:
2150
+ return []
2151
+ html = resp.text
2152
+ results = []
2153
+ pattern = re.compile(
2154
+ r'result__a"[^>]*href="([^"]+)"[^>]*>(.*?)</a>.*?'
2155
+ r'result__snippet[^>]*>(.*?)</a>', re.S)
2156
+ for m in pattern.finditer(html):
2157
+ href, title_html, snippet_html = m.groups()
2158
+ title = _html_unescape(_TAG_RE.sub("", title_html)).strip()
2159
+ snippet = _html_unescape(_TAG_RE.sub("", snippet_html)).strip()
2160
+ url = _unwrap_ddg_url(href)
2161
+ if title and url:
2162
+ results.append({"title": title, "url": url, "snippet": snippet})
2163
+ if len(results) >= max_results:
2164
+ break
2165
+ return results
2166
+ except Exception:
2167
+ return []
2168
+
2169
+
2170
+ def deep_research(topic, num_queries=3, results_per_query=4):
2171
+ """Runs several web searches around different angles of `topic`, then
2172
+ asks the configured AI to synthesize a structured, source-numbered
2173
+ research summary from the actual retrieved snippets (not from the
2174
+ model's own unchecked memory). Returns (summary_text, sources) where
2175
+ sources is a de-duplicated list of {\"title\", \"url\", \"snippet\"} dicts,
2176
+ numbered in the same order referenced as [1], [2]... in the summary."""
2177
+ topic = (topic or "").strip()
2178
+ if not topic:
2179
+ return "No research topic given.", []
2180
+
2181
+ sub_queries = [topic]
2182
+ config = load_config()
2183
+ if config.get("provider"):
2184
+ angles_raw = query_ai(
2185
+ f"Give exactly {num_queries} short, distinct web-search queries (one per "
2186
+ f"line, no numbering, no extra commentary) that together would build a "
2187
+ f"thorough, well-rounded understanding of: {topic}",
2188
+ system_prompt="You output ONLY the search queries, one per line, nothing else.")
2189
+ angles = [a.strip("-•*0123456789. ").strip() for a in (angles_raw or "").splitlines() if a.strip()]
2190
+ if angles:
2191
+ sub_queries = angles[:num_queries]
2192
+
2193
+ all_results = []
2194
+ seen_urls = set()
2195
+ for q in sub_queries:
2196
+ for r in web_search(q, max_results=results_per_query):
2197
+ if r["url"] not in seen_urls:
2198
+ seen_urls.add(r["url"])
2199
+ all_results.append(r)
2200
+
2201
+ if not all_results:
2202
+ return (f"No web results could be retrieved for '{topic}' — check the "
2203
+ f"internet connection this terminal has, DuckDuckGo may also be "
2204
+ f"rate-limiting/blocking this network."), []
2205
+
2206
+ evidence = "\n\n".join(
2207
+ f"[{i + 1}] {r['title']}\n{r['url']}\n{r['snippet']}"
2208
+ for i, r in enumerate(all_results[:12]))
2209
+
2210
+ if not config.get("provider"):
2211
+ return ("AI not configured, so here are the raw sources found "
2212
+ "(run /model or /ai to also get a synthesized summary):\n\n" + evidence,
2213
+ all_results[:12])
2214
+
2215
+ summary = query_ai(
2216
+ f"Research topic: {topic}\n\nSources:\n{evidence}\n\n"
2217
+ "Write a clear, well-organized research summary of the topic using ONLY "
2218
+ "these sources. Reference sources inline as [1], [2] etc. matching the "
2219
+ "numbers above. Point out any disagreement between sources. Do not invent "
2220
+ "facts the sources don't support.",
2221
+ system_prompt=("You are a careful research assistant. Be accurate, cite "
2222
+ "sources by their bracket number, and never invent facts "
2223
+ "not supported by the given sources."))
2224
+ return summary, all_results[:12]
2225
+
2226
+
2227
+ # ------------------------------------------------------------ file / image import
2228
+ # Backs /import (feature 6): local text files get their content read straight
2229
+ # into the AI/agent's context; local images get sent to a vision-capable
2230
+ # provider (OpenAI/Anthropic/Gemini) so the model can actually "see" them.
2231
+ IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"}
2232
+ TEXT_FILE_MAX_CHARS = 20000
2233
+
2234
+
2235
+ def is_image_file(path):
2236
+ return os.path.splitext(str(path))[1].lower() in IMAGE_EXTS
2237
+
2238
+
2239
+ def read_text_file_for_context(path):
2240
+ """Returns (content, truncated) or (None, False) on failure."""
2241
+ try:
2242
+ with open(path, "r", encoding="utf-8", errors="replace") as f:
2243
+ content = f.read(TEXT_FILE_MAX_CHARS + 1)
2244
+ truncated = len(content) > TEXT_FILE_MAX_CHARS
2245
+ if truncated:
2246
+ content = content[:TEXT_FILE_MAX_CHARS]
2247
+ return content, truncated
2248
+ except Exception:
2249
+ return None, False
2250
+
2251
+
2252
+ # ------------------------------------------------------- attachment context
2253
+ # v0.7.6 Patch 1, Fix 7: one attachment gets one honest, real context block
2254
+ # fed to the AI. Plain text/code/markdown gets its actual content; CSV gets
2255
+ # a genuine header/sample/count; zip/tar get a real listing; PDF gets real
2256
+ # metadata + a best-effort text-layer extraction (stdlib-only — no pypdf in
2257
+ # this environment); audio/video report what's known (size/ext) and say
2258
+ # plainly when a metadata library isn't installed. Never fabricated data.
2259
+ AUDIO_EXTS = {".mp3", ".wav", ".ogg", ".flac", ".m4a", ".aac", ".wma", ".opus"}
2260
+ VIDEO_EXTS = {".mp4", ".mkv", ".avi", ".mov", ".webm", ".wmv", ".m4v", ".mpg", ".mpeg"}
2261
+ _BINARY_EXTS = {".exe", ".dll", ".so", ".dylib", ".bin", ".dat", ".db", ".sqlite",
2262
+ ".sqlite3", ".pyc", ".pdb", ".iso", ".img", ".parquet", ".h5", ".hdf5"}
2263
+
2264
+
2265
+ def _human_size(n):
2266
+ if n < 1024:
2267
+ return f"{n} B"
2268
+ if n < 1024 ** 2:
2269
+ return f"{n / 1024:.1f} KB"
2270
+ if n < 1024 ** 3:
2271
+ return f"{n / 1024 ** 2:.1f} MB"
2272
+ return f"{n / 1024 ** 3:.1f} GB"
2273
+
2274
+
2275
+ def _pdf_context(path, name, size_txt, max_chars):
2276
+ try:
2277
+ with open(path, "rb") as f:
2278
+ raw = f.read()
2279
+ except Exception as e:
2280
+ return f"[attachment: {name} ({size_txt}) \u2014 unreadable PDF: {e}]"
2281
+ counts = [int(m) for m in re.findall(rb"/Count\s+(\d+)", raw)]
2282
+ page_txt = f"{max(counts)} pages" if counts else "page count unknown"
2283
+ meta_bits = []
2284
+ for key in (b"Title", b"Author", b"Subject", b"Creator", b"Producer"):
2285
+ m = re.search(key + rb"\s*\(([^()\\]*(?:\\.[^()\\]*)*)\)", raw[:200000])
2286
+ if m:
2287
+ val = m.group(1)[:120].decode("latin-1", "replace")
2288
+ meta_bits.append(f"{key.decode()}: {val}")
2289
+ meta_txt = ("; ".join(meta_bits) + ".") if meta_bits else "no document metadata."
2290
+ # Best-effort text-layer extraction: decompress every object stream
2291
+ # and pull out literal text shown via Tj/TJ operators. Stdlib-only,
2292
+ # so genuinely imperfect — but real text, never fabricated.
2293
+ texts = []
2294
+ for m in re.finditer(rb"stream\r?\n(.*?)endstream", raw, re.DOTALL):
2295
+ data = m.group(1).lstrip(b"\r\n")
2296
+ for payload in (data,):
2297
+ try:
2298
+ payload = zlib.decompress(data)
2299
+ except Exception:
2300
+ pass
2301
+ texts += re.findall(rb"\(((?:[^()\\]|\\.)*)\)\s*Tj", payload)
2302
+ plain = " ".join(
2303
+ t.replace(b"\\(", b"(").replace(b"\\)", b")").replace(b"\\\\", b"\\")
2304
+ .decode("latin-1", "replace") for t in texts)
2305
+ plain = re.sub(r"\s+", " ", plain).strip()
2306
+ if plain:
2307
+ if len(plain) > max_chars:
2308
+ plain = plain[:max_chars] + " \u2026[truncated]"
2309
+ body = f"[attachment: {name} ({size_txt}) \u2014 PDF, {page_txt}; {meta_txt}]\n{plain}"
2310
+ else:
2311
+ body = (f"[attachment: {name} ({size_txt}) \u2014 PDF, {page_txt}; {meta_txt} "
2312
+ f"No extractable text layer (scanned image or glyph-encoded PDF).]")
2313
+ return body
2314
+
2315
+
2316
+ def _zip_context(path, name, size_txt):
2317
+ import zipfile
2318
+ try:
2319
+ with zipfile.ZipFile(path) as zf:
2320
+ infos = zf.infolist()
2321
+ files = [i for i in infos if not i.is_dir()]
2322
+ dirs = [i for i in infos if i.is_dir()]
2323
+ total = sum(i.file_size for i in files)
2324
+ listing = "\n".join(i.filename for i in files[:40])
2325
+ more = f"\n\u2026 and {len(files) - 40} more files" if len(files) > 40 else ""
2326
+ return (f"[attachment: {name} ({size_txt}) \u2014 zip archive: {len(files)} files, "
2327
+ f"{len(dirs)} folders, {_human_size(total)} uncompressed]\n"
2328
+ f"{listing}{more}" if files else
2329
+ f"[attachment: {name} ({size_txt}) \u2014 zip archive, empty.]")
2330
+ except Exception as e:
2331
+ return f"[attachment: {name} ({size_txt}) \u2014 not a readable zip archive: {e}]"
2332
+
2333
+
2334
+ def _tar_context(path, name, size_txt):
2335
+ import tarfile
2336
+ try:
2337
+ with tarfile.open(path) as tf:
2338
+ members = tf.getmembers()
2339
+ files = [m for m in members if m.isfile()]
2340
+ listing = "\n".join(m.name for m in files[:40])
2341
+ more = f"\n\u2026 and {len(files) - 40} more files" if len(files) > 40 else ""
2342
+ return (f"[attachment: {name} ({size_txt}) \u2014 tar archive: {len(files)} files]\n"
2343
+ f"{listing}{more}" if files else
2344
+ f"[attachment: {name} ({size_txt}) \u2014 tar archive, empty.]")
2345
+ except Exception as e:
2346
+ return f"[attachment: {name} ({size_txt}) \u2014 not a readable tar archive: {e}]"
2347
+
2348
+
2349
+ def _csv_context(path, name, size_txt, max_chars):
2350
+ import csv
2351
+ header, sample, total = None, [], 0
2352
+ capped = False
2353
+ try:
2354
+ with open(path, "r", encoding="utf-8", errors="replace", newline="") as f:
2355
+ for i, row in enumerate(csv.reader(f)):
2356
+ if i == 0:
2357
+ header = row
2358
+ elif len(sample) < 5:
2359
+ sample.append(row)
2360
+ total += 1
2361
+ if total >= 5000:
2362
+ capped = True
2363
+ break
2364
+ except Exception as e:
2365
+ return f"[attachment: {name} ({size_txt}) \u2014 could not parse CSV: {e}]"
2366
+ cols = f"{len(header)} columns" if header else "0 columns"
2367
+ total_txt = f"{total} rows" + (" (capped at 5000)" if capped else "")
2368
+ lines = [f"[attachment: {name} ({size_txt}) \u2014 CSV, {cols}, {total_txt}]"]
2369
+ if header:
2370
+ lines.append("header: " + " | ".join(header))
2371
+ if sample:
2372
+ lines.append("first rows:")
2373
+ lines += [" " + " | ".join(row) for row in sample]
2374
+ body = "\n".join(lines)
2375
+ return body[:max_chars] + (" \u2026[truncated]" if len(body) > max_chars else "")
2376
+
2377
+
2378
+ def _image_context(path, name, size_txt):
2379
+ note = (f"[Attached image: {path} \u2014 not yet included in streamed "
2380
+ f"replies; vision-capable providers analyze it via the "
2381
+ f"query_ai_with_image path instead]")
2382
+ try:
2383
+ from PIL import Image
2384
+ with Image.open(path) as im:
2385
+ w, h = im.size
2386
+ fmt = (im.format or "").upper()
2387
+ return (f"[Attached image: {name} ({size_txt}, {w}x{h} {fmt}) \u2014 "
2388
+ f"not yet included in streamed replies; vision-capable "
2389
+ f"providers analyze it via the query_ai_with_image path instead]")
2390
+ except Exception:
2391
+ return note
2392
+
2393
+
2394
+ def attachment_context_for(path, max_chars=TEXT_FILE_MAX_CHARS):
2395
+ """v0.7.6 Patch 1, Fix 7: the complete AI-side context block for one
2396
+ attached file — real content or real metadata per file type, always
2397
+ honest about what couldn't be extracted (no fabricated analysis).
2398
+ Images keep the existing vision path; text-like files (md, code,
2399
+ data, config) get their actual content; CSV/zip/tar/PDF get genuine
2400
+ structure; audio/video report what's knowable in this environment.
2401
+ """
2402
+ if not os.path.isfile(path):
2403
+ return f"[attachment not found: {path}]"
2404
+ name = os.path.basename(path)
2405
+ ext = os.path.splitext(name)[1].lower()
2406
+ try:
2407
+ size = os.path.getsize(path)
2408
+ except Exception:
2409
+ size = 0
2410
+ size_txt = _human_size(size)
2411
+
2412
+ if is_image_file(path):
2413
+ return _image_context(path, name, size_txt)
2414
+ if ext == ".pdf":
2415
+ return _pdf_context(path, name, size_txt, max_chars)
2416
+ if ext == ".zip":
2417
+ return _zip_context(path, name, size_txt)
2418
+ if ext in {".tar", ".gz", ".bz2", ".tgz"}:
2419
+ return _tar_context(path, name, size_txt)
2420
+ if ext in {".rar", ".7z"}:
2421
+ return (f"[attachment: {name} ({size_txt}) \u2014 {ext} archive; "
2422
+ f"no stdlib reader exists here, extract it first and attach "
2423
+ f"the contents instead]")
2424
+ if ext == ".csv":
2425
+ return _csv_context(path, name, size_txt, max_chars)
2426
+ if ext in AUDIO_EXTS:
2427
+ return (f"[attachment: {name} ({size_txt}) \u2014 audio file; duration "
2428
+ f"and tags would need the 'mutagen' package, which isn't "
2429
+ f"installed in this environment]")
2430
+ if ext in VIDEO_EXTS:
2431
+ return (f"[attachment: {name} ({size_txt}) \u2014 video file; duration "
2432
+ f"and codec info would need the 'mutagen' package, which "
2433
+ f"isn't installed in this environment]")
2434
+ if ext in _BINARY_EXTS:
2435
+ return f"[attachment: {name} ({size_txt}) \u2014 binary {ext} file, content not readable as text]"
2436
+
2437
+ content, truncated = read_text_file_for_context(path)
2438
+ if content is None:
2439
+ return f"[attachment: {name} ({size_txt}) \u2014 could not be read as text]"
2440
+ head = (f"[attachment: {name} ({size_txt})"
2441
+ + (" \u2014 truncated to first chars]" if truncated else "]"))
2442
+ return head + "\n" + content
2443
+
2444
+
2445
+ def encode_image_b64(path):
2446
+ with open(path, "rb") as f:
2447
+ return base64.b64encode(f.read()).decode("ascii")
2448
+
2449
+
2450
+ def _guess_image_mime(path):
2451
+ ext = os.path.splitext(str(path))[1].lower().lstrip(".")
2452
+ return {"jpg": "image/jpeg", "jpeg": "image/jpeg", "png": "image/png",
2453
+ "gif": "image/gif", "webp": "image/webp", "bmp": "image/bmp"}.get(ext, "image/png")
2454
+
2455
+
2456
+ def query_ai_with_image(prompt, image_path,
2457
+ system_prompt=("You are CAT AI. Describe and analyze the attached image "
2458
+ "precisely.\n\n" + identity.IDENTITY_BLOCK)):
2459
+ """Same idea as query_ai but attaches one local image, for providers whose
2460
+ API actually supports vision here (OpenAI-style, Anthropic, Gemini).
2461
+
2462
+ v0.7.9.0 (requirement #19 — multimodal fallback): the image is
2463
+ normalized through the vision pipeline first; if the ACTIVE model
2464
+ can't receive native images but another CONFIGURED model can, the
2465
+ request is rerouted there instead of replying 'model can't see this'.
2466
+ Only when no vision-capable model exists at all is that reported,
2467
+ clearly and honestly."""
2468
+ if not _HAS_REQUESTS:
2469
+ return "The 'requests' library is required for AI features. Please run: pip install requests"
2470
+
2471
+ try:
2472
+ from . import vision as _vision
2473
+ mime, b64 = _vision.encode_for_model(image_path, prompt)[:2]
2474
+ except Exception as e:
2475
+ return f"Could not read image '{image_path}': {e}"
2476
+
2477
+ config = load_config()
2478
+ provider = config.get("provider")
2479
+ if not provider:
2480
+ return "AI not configured. Run /ai, /agent, or /model to configure your provider first."
2481
+
2482
+ info = PROVIDERS.get(provider)
2483
+ api_style = info["api_style"] if info else "openai"
2484
+ used_config = config
2485
+ if api_style not in ("openai", "anthropic", "gemini"):
2486
+ # Primary can't take native images — look for a configured one that can.
2487
+ try:
2488
+ from . import model_router as _mr
2489
+ vcfg, _caps = _mr.find_vision_capable(config)
2490
+ except Exception:
2491
+ vcfg = None
2492
+ if vcfg is None:
2493
+ return ("No vision-capable model is currently configured, so the "
2494
+ f"image couldn't be analyzed visually ({provider} has no "
2495
+ "native vision transport here). Configure one via /model.")
2496
+ used_config = vcfg
2497
+ info = PROVIDERS.get(used_config.get("provider"), {})
2498
+ api_style = info["api_style"] if info else "openai"
2499
+
2500
+ if provider == "ollama":
2501
+ base_url = config.get("ollama_url") or (info["base_url"] if info else "http://localhost:11434")
2502
+ else:
2503
+ base_url = used_config.get("base_url") or used_config.get("api_url") or (info["base_url"] if info else "")
2504
+ base_url = base_url.rstrip("/")
2505
+ api_key = used_config.get("api_key", "")
2506
+ model = used_config.get("model") or (info["default_model"] if info else "gpt-4o-mini")
2507
+ extra_headers = info["extra_headers"] if info else {}
2508
+
2509
+ try:
2510
+ if api_style == "openai":
2511
+ url = f"{base_url}/chat/completions"
2512
+ headers = {"Content-Type": "application/json"}
2513
+ if api_key:
2514
+ headers["Authorization"] = f"Bearer {api_key}"
2515
+ headers.update(extra_headers)
2516
+ payload = {
2517
+ "model": model,
2518
+ "messages": [
2519
+ {"role": "system", "content": system_prompt},
2520
+ {"role": "user", "content": [
2521
+ {"type": "text", "text": prompt},
2522
+ {"type": "image_url", "image_url": {"url": f"data:{mime};base64,{b64}"}},
2523
+ ]},
2524
+ ],
2525
+ }
2526
+ resp = requests.post(url, headers=headers, json=payload, timeout=60)
2527
+ _raise_for_status(resp)
2528
+ data = resp.json()
2529
+ content = data['choices'][0]['message']['content']
2530
+ _track_usage_from_data("openai", data, prompt, content)
2531
+ return content
2532
+
2533
+ elif api_style == "anthropic":
2534
+ headers = {"x-api-key": api_key, "anthropic-version": "2023-06-01",
2535
+ "content-type": "application/json"}
2536
+ headers.update(extra_headers)
2537
+ payload = {
2538
+ "model": model, "max_tokens": 1024, "system": system_prompt,
2539
+ "messages": [{"role": "user", "content": [
2540
+ {"type": "image", "source": {"type": "base64", "media_type": mime, "data": b64}},
2541
+ {"type": "text", "text": prompt},
2542
+ ]}],
2543
+ }
2544
+ resp = requests.post(f"{base_url}/messages", headers=headers, json=payload, timeout=60)
2545
+ _raise_for_status(resp)
2546
+ data = resp.json()
2547
+ content = data['content'][0]['text']
2548
+ _track_usage_from_data("anthropic", data, prompt, content)
2549
+ return content
2550
+
2551
+ elif api_style == "gemini":
2552
+ url = f"{base_url}/models/{model}:generateContent?key={api_key}"
2553
+ payload = {
2554
+ "contents": [{"role": "user", "parts": [
2555
+ {"text": prompt},
2556
+ {"inline_data": {"mime_type": mime, "data": b64}},
2557
+ ]}],
2558
+ "systemInstruction": {"parts": [{"text": system_prompt}]},
2559
+ }
2560
+ resp = requests.post(url, json=payload, timeout=60)
2561
+ _raise_for_status(resp)
2562
+ data = resp.json()
2563
+ content = data['candidates'][0]['content']['parts'][0]['text']
2564
+ _track_usage_from_data("gemini", data, prompt, content)
2565
+ return content
2566
+
2567
+ except requests.exceptions.ConnectionError:
2568
+ return "Could not reach the AI server for image analysis. Check your internet/API URL."
2569
+ except requests.exceptions.Timeout:
2570
+ return "The image analysis request timed out. Try again, or use a faster model."
2571
+ except Exception as e:
2572
+ return f"Error analyzing image: {e}"