cct-cli 0.7.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- calc_terminal/__init__.py +14 -0
- calc_terminal/__main__.py +14 -0
- calc_terminal/activity.py +1334 -0
- calc_terminal/agent.py +3387 -0
- calc_terminal/agent_runtime.py +519 -0
- calc_terminal/ai_context.py +447 -0
- calc_terminal/ai_modes.py +752 -0
- calc_terminal/ai_personalization.py +286 -0
- calc_terminal/ai_preview_feedback.py +213 -0
- calc_terminal/aicore.py +2572 -0
- calc_terminal/anim.py +367 -0
- calc_terminal/app.py +3685 -0
- calc_terminal/art.py +639 -0
- calc_terminal/atomsim.py +368 -0
- calc_terminal/attachments.py +743 -0
- calc_terminal/benchmark_system.py +414 -0
- calc_terminal/browser/__init__.py +36 -0
- calc_terminal/browser/browser_state.py +346 -0
- calc_terminal/browser/devserver.py +176 -0
- calc_terminal/browser/engine.py +494 -0
- calc_terminal/browser/navigation.py +84 -0
- calc_terminal/browser/preview.py +429 -0
- calc_terminal/browser/preview_entry.py +95 -0
- calc_terminal/browser/project_detector.py +144 -0
- calc_terminal/browser/server.py +449 -0
- calc_terminal/browser/state.py +75 -0
- calc_terminal/browser/watcher.py +99 -0
- calc_terminal/browser_gui/__init__.py +1 -0
- calc_terminal/browser_gui/__main__.py +3 -0
- calc_terminal/browser_gui/launcher.py +173 -0
- calc_terminal/browser_gui/playwright_browser.py +117 -0
- calc_terminal/browser_gui/qt_browser.py +1501 -0
- calc_terminal/browser_gui/webview_browser.py +57 -0
- calc_terminal/capabilities/__init__.py +35 -0
- calc_terminal/capabilities/adapters/__init__.py +33 -0
- calc_terminal/capabilities/adapters/bioinformatics.py +204 -0
- calc_terminal/capabilities/adapters/browser_adapter.py +205 -0
- calc_terminal/capabilities/adapters/filesystem.py +206 -0
- calc_terminal/capabilities/adapters/git_adapter.py +202 -0
- calc_terminal/capabilities/adapters/jupyter_adapter.py +138 -0
- calc_terminal/capabilities/adapters/ml_frameworks.py +158 -0
- calc_terminal/capabilities/adapters/platforms.py +200 -0
- calc_terminal/capabilities/adapters/python_exec.py +93 -0
- calc_terminal/capabilities/adapters/quantum_adapter.py +150 -0
- calc_terminal/capabilities/adapters/scientific_comp.py +123 -0
- calc_terminal/capabilities/adapters/structural_bio.py +161 -0
- calc_terminal/capabilities/adapters/terminal.py +99 -0
- calc_terminal/capabilities/bus.py +178 -0
- calc_terminal/capabilities/discovery.py +207 -0
- calc_terminal/capabilities/schema.py +221 -0
- calc_terminal/cat.ico +0 -0
- calc_terminal/cat_browser.py +2018 -0
- calc_terminal/chat_store.py +703 -0
- calc_terminal/cli.py +1178 -0
- calc_terminal/code_editor.py +640 -0
- calc_terminal/collaboration.py +723 -0
- calc_terminal/commands_data.py +139 -0
- calc_terminal/compatibility_engine.py +352 -0
- calc_terminal/compute/__init__.py +31 -0
- calc_terminal/compute/fabric.py +350 -0
- calc_terminal/config.py +227 -0
- calc_terminal/core/__init__.py +41 -0
- calc_terminal/core/checkpoint.py +156 -0
- calc_terminal/core/input/__init__.py +45 -0
- calc_terminal/core/mode_registry.py +300 -0
- calc_terminal/core/project_graph.py +172 -0
- calc_terminal/core/recovery.py +129 -0
- calc_terminal/core/security_layer.py +112 -0
- calc_terminal/core/task_graph.py +202 -0
- calc_terminal/core/unified_runtime.py +184 -0
- calc_terminal/core/verification.py +257 -0
- calc_terminal/customization.py +1566 -0
- calc_terminal/derivations.py +153 -0
- calc_terminal/device_control.py +263 -0
- calc_terminal/diagnostics/__init__.py +27 -0
- calc_terminal/diagnostics/doctor_engine.py +382 -0
- calc_terminal/diagnostics/self_test.py +247 -0
- calc_terminal/doctor.py +519 -0
- calc_terminal/easter_eggs.py +274 -0
- calc_terminal/editor/__init__.py +1 -0
- calc_terminal/editor/actions.py +263 -0
- calc_terminal/editor/commands.py +160 -0
- calc_terminal/editor/shortcuts.py +226 -0
- calc_terminal/engine.py +259 -0
- calc_terminal/errors.py +120 -0
- calc_terminal/event_stream.py +146 -0
- calc_terminal/eventbus.py +133 -0
- calc_terminal/extensions.py +733 -0
- calc_terminal/fallback_cli.py +1321 -0
- calc_terminal/first_run.py +265 -0
- calc_terminal/fomoji_auth.py +1043 -0
- calc_terminal/formulas.py +82 -0
- calc_terminal/fs_cache.py +121 -0
- calc_terminal/fs_watcher.py +277 -0
- calc_terminal/game.py +193 -0
- calc_terminal/gen1.py +5 -0
- calc_terminal/generators.py +245 -0
- calc_terminal/gestures/__init__.py +42 -0
- calc_terminal/gestures/bindings.py +175 -0
- calc_terminal/gestures/manager.py +477 -0
- calc_terminal/goodbye.py +363 -0
- calc_terminal/gpu3d.py +290 -0
- calc_terminal/graphs.py +358 -0
- calc_terminal/hardware_analyzer.py +440 -0
- calc_terminal/host/__init__.py +30 -0
- calc_terminal/host/browser_manager.py +187 -0
- calc_terminal/host/desktop.py +1386 -0
- calc_terminal/host/launcher.py +395 -0
- calc_terminal/host/terminal.py +279 -0
- calc_terminal/identity.py +216 -0
- calc_terminal/input/__init__.py +54 -0
- calc_terminal/input/capabilities.py +258 -0
- calc_terminal/input/focus.py +87 -0
- calc_terminal/input/gestures.py +64 -0
- calc_terminal/input/pointer.py +114 -0
- calc_terminal/input/touch.py +345 -0
- calc_terminal/keys.py +84 -0
- calc_terminal/live_automation.py +165 -0
- calc_terminal/mathtext.py +433 -0
- calc_terminal/mcp.py +386 -0
- calc_terminal/memory.py +337 -0
- calc_terminal/memory_v2.py +479 -0
- calc_terminal/metrics.py +333 -0
- calc_terminal/mode_detection.py +146 -0
- calc_terminal/model.py +2431 -0
- calc_terminal/model_router.py +665 -0
- calc_terminal/models/__init__.py +0 -0
- calc_terminal/models/active_state.py +187 -0
- calc_terminal/models/dynamic_registry.py +584 -0
- calc_terminal/models/manager.py +781 -0
- calc_terminal/models/model_metadata.json +3526 -0
- calc_terminal/models/profiles.py +194 -0
- calc_terminal/models/registry.py +265 -0
- calc_terminal/models/schema.py +197 -0
- calc_terminal/models/validator.py +287 -0
- calc_terminal/models/verification_engine.py +368 -0
- calc_terminal/native_picker.py +215 -0
- calc_terminal/ollama_catalog.py +279 -0
- calc_terminal/ollama_download.py +233 -0
- calc_terminal/orchestrator.py +304 -0
- calc_terminal/package_research.py +322 -0
- calc_terminal/packages.py +1024 -0
- calc_terminal/pc_specs.py +116 -0
- calc_terminal/permissions.py +334 -0
- calc_terminal/pet.py +106 -0
- calc_terminal/pipeline.py +505 -0
- calc_terminal/platform/__init__.py +491 -0
- calc_terminal/platform/desktop.py +491 -0
- calc_terminal/platform/web.py +781 -0
- calc_terminal/preview/__init__.py +1 -0
- calc_terminal/preview/dev_server.py +303 -0
- calc_terminal/preview/diagnostics.py +131 -0
- calc_terminal/preview/live_reload.py +66 -0
- calc_terminal/preview/manager.py +129 -0
- calc_terminal/project_stats.py +209 -0
- calc_terminal/projects.py +328 -0
- calc_terminal/providers/__init__.py +0 -0
- calc_terminal/providers/adapters/__init__.py +80 -0
- calc_terminal/providers/adapters/anthropic_adapter.py +127 -0
- calc_terminal/providers/adapters/base.py +106 -0
- calc_terminal/providers/adapters/chinese_adapters.py +420 -0
- calc_terminal/providers/adapters/gemini_adapter.py +101 -0
- calc_terminal/providers/adapters/ollama_adapter.py +83 -0
- calc_terminal/providers/adapters/openai_adapter.py +159 -0
- calc_terminal/providers/adapters/other_adapters.py +246 -0
- calc_terminal/providers/anthropic_provider.py +172 -0
- calc_terminal/providers/auto_update.py +416 -0
- calc_terminal/providers/base_provider.py +105 -0
- calc_terminal/providers/discovery_manager.py +207 -0
- calc_terminal/providers/gemini_provider.py +178 -0
- calc_terminal/providers/lifecycle.py +767 -0
- calc_terminal/providers/ollama_adapter.py +707 -0
- calc_terminal/providers/openai_provider.py +254 -0
- calc_terminal/providers/provider_manager.py +1827 -0
- calc_terminal/providers/providers.json +4075 -0
- calc_terminal/reactionsim.py +279 -0
- calc_terminal/registry.py +337 -0
- calc_terminal/report.py +162 -0
- calc_terminal/research/__init__.py +45 -0
- calc_terminal/research/artifact_intel.py +126 -0
- calc_terminal/research/data_lineage.py +123 -0
- calc_terminal/research/experiment_ledger.py +303 -0
- calc_terminal/research/reproducibility.py +131 -0
- calc_terminal/resilience/__init__.py +47 -0
- calc_terminal/resilience/agent_state.py +121 -0
- calc_terminal/resilience/capability_matcher.py +174 -0
- calc_terminal/resilience/circuit_breaker.py +158 -0
- calc_terminal/resilience/failover_engine.py +230 -0
- calc_terminal/resilience/health_monitor.py +192 -0
- calc_terminal/resilience/ollama_adapter.py +125 -0
- calc_terminal/resilience/orchestrator.py +312 -0
- calc_terminal/resilience/types.py +134 -0
- calc_terminal/sandbox.py +98 -0
- calc_terminal/scires.py +558 -0
- calc_terminal/security_scanner.py +126 -0
- calc_terminal/session.py +294 -0
- calc_terminal/sim3d.py +206 -0
- calc_terminal/solver.py +276 -0
- calc_terminal/sound.py +127 -0
- calc_terminal/task_reports.py +287 -0
- calc_terminal/terminal_host.py +201 -0
- calc_terminal/terminal_identity.py +411 -0
- calc_terminal/test_ai_mode_reliability.py +344 -0
- calc_terminal/test_browser.py +368 -0
- calc_terminal/test_code_editor_upgrade.py +485 -0
- calc_terminal/test_customization.py +1148 -0
- calc_terminal/test_customization_ui.py +612 -0
- calc_terminal/test_dynamic_registry.py +304 -0
- calc_terminal/test_extensions.py +436 -0
- calc_terminal/test_overhaul.py +557 -0
- calc_terminal/test_project_detect.py +255 -0
- calc_terminal/test_root_cause_fix.py +527 -0
- calc_terminal/test_stability.py +532 -0
- calc_terminal/test_terminal_identity.py +132 -0
- calc_terminal/test_v079_speed.py +460 -0
- calc_terminal/theme.py +1107 -0
- calc_terminal/timeline.py +139 -0
- calc_terminal/todos.py +246 -0
- calc_terminal/tool_call_normalizer.py +419 -0
- calc_terminal/tui.py +104 -0
- calc_terminal/ui/__init__.py +8 -0
- calc_terminal/ui/activity_panel.py +231 -0
- calc_terminal/ui/activity_stream_panel.py +238 -0
- calc_terminal/ui/animations.py +122 -0
- calc_terminal/ui/app.py +7271 -0
- calc_terminal/ui/attach_panel.py +597 -0
- calc_terminal/ui/attachments.py +424 -0
- calc_terminal/ui/backup_panel.py +810 -0
- calc_terminal/ui/browser_shell.py +887 -0
- calc_terminal/ui/cat_agent.py +357 -0
- calc_terminal/ui/chats_panel.py +899 -0
- calc_terminal/ui/command_palette.py +125 -0
- calc_terminal/ui/command_palette_modal.py +166 -0
- calc_terminal/ui/composer.py +1141 -0
- calc_terminal/ui/context_menu.py +197 -0
- calc_terminal/ui/conversation.py +1435 -0
- calc_terminal/ui/customization_panel.py +1229 -0
- calc_terminal/ui/dashboard.py +404 -0
- calc_terminal/ui/design_system.py +557 -0
- calc_terminal/ui/diff_panel.py +213 -0
- calc_terminal/ui/editor.py +2102 -0
- calc_terminal/ui/empty_state.py +302 -0
- calc_terminal/ui/events.py +487 -0
- calc_terminal/ui/extensions_panel.py +815 -0
- calc_terminal/ui/footer.py +166 -0
- calc_terminal/ui/gestures_panel.py +383 -0
- calc_terminal/ui/goodbye_screen.py +100 -0
- calc_terminal/ui/header.py +1034 -0
- calc_terminal/ui/help_panel.py +254 -0
- calc_terminal/ui/live_activities.py +914 -0
- calc_terminal/ui/mcp_panel.py +570 -0
- calc_terminal/ui/memory_center.py +524 -0
- calc_terminal/ui/mode_colors_panel.py +525 -0
- calc_terminal/ui/nav_screens.py +747 -0
- calc_terminal/ui/ollama_panel.py +536 -0
- calc_terminal/ui/palette.py +221 -0
- calc_terminal/ui/permission_panel.py +269 -0
- calc_terminal/ui/personalization_panel.py +517 -0
- calc_terminal/ui/personalize_center.py +1568 -0
- calc_terminal/ui/preview_panel.py +441 -0
- calc_terminal/ui/resizers.py +402 -0
- calc_terminal/ui/sidebar.py +1285 -0
- calc_terminal/ui/statusbar.py +168 -0
- calc_terminal/ui/theme_css.py +1396 -0
- calc_terminal/ui/thinking.py +226 -0
- calc_terminal/ui/timeline_panel.py +102 -0
- calc_terminal/ui/todo_panel.py +193 -0
- calc_terminal/ui/viewport.py +136 -0
- calc_terminal/ui/vision_panel.py +489 -0
- calc_terminal/ui/welcome_modal.py +343 -0
- calc_terminal/ui/widgets.py +160 -0
- calc_terminal/ui/workspace.py +831 -0
- calc_terminal/viewers/__init__.py +1 -0
- calc_terminal/viewers/document_viewer.py +252 -0
- calc_terminal/viewers/image_viewer.py +241 -0
- calc_terminal/viewers/pdf_viewer.py +203 -0
- calc_terminal/viewers/presentation_viewer.py +164 -0
- calc_terminal/viewers/registry.py +120 -0
- calc_terminal/viewers/spreadsheet_viewer.py +204 -0
- calc_terminal/vision/__init__.py +89 -0
- calc_terminal/vision/analysis.py +194 -0
- calc_terminal/vision/annotations.py +297 -0
- calc_terminal/vision/capture.py +171 -0
- calc_terminal/vision/context.py +231 -0
- calc_terminal/vision/correlation.py +169 -0
- calc_terminal/vision/cursor.py +258 -0
- calc_terminal/vision/events.py +66 -0
- calc_terminal/vision/frame_pipeline.py +259 -0
- calc_terminal/vision/priority.py +218 -0
- calc_terminal/vision/provider.py +180 -0
- calc_terminal/vision/safety.py +149 -0
- calc_terminal/vision/session.py +281 -0
- calc_terminal/vision/verify.py +162 -0
- calc_terminal/vision.py +514 -0
- calc_terminal/vscode_integration.py +113 -0
- calc_terminal/web/__init__.py +8 -0
- calc_terminal/web/cat_runtime.py +710 -0
- calc_terminal/web/server.py +2891 -0
- calc_terminal/web/static/css/app.css +3152 -0
- calc_terminal/web/static/icons/badge-72.png +0 -0
- calc_terminal/web/static/icons/cat.ico +0 -0
- calc_terminal/web/static/icons/icon-128.png +0 -0
- calc_terminal/web/static/icons/icon-144.png +0 -0
- calc_terminal/web/static/icons/icon-152.png +0 -0
- calc_terminal/web/static/icons/icon-192.png +0 -0
- calc_terminal/web/static/icons/icon-384.png +0 -0
- calc_terminal/web/static/icons/icon-512.png +0 -0
- calc_terminal/web/static/icons/icon-72.png +0 -0
- calc_terminal/web/static/icons/icon-96.png +0 -0
- calc_terminal/web/static/icons/icon.svg +34 -0
- calc_terminal/web/static/icons/new-project.png +0 -0
- calc_terminal/web/static/icons/open-project.png +0 -0
- calc_terminal/web/static/index.html +734 -0
- calc_terminal/web/static/js/app.js +2403 -0
- calc_terminal/web/static/manifest.json +88 -0
- calc_terminal/web/static/sw.js +230 -0
- calc_terminal/workflow_engine.py +769 -0
- calc_terminal/workspace.py +593 -0
- calc_terminal/workspace_index.py +385 -0
- cct_cli-0.7.9.0.dist-info/METADATA +210 -0
- cct_cli-0.7.9.0.dist-info/RECORD +325 -0
- cct_cli-0.7.9.0.dist-info/WHEEL +5 -0
- cct_cli-0.7.9.0.dist-info/entry_points.txt +4 -0
- cct_cli-0.7.9.0.dist-info/licenses/LICENSE +21 -0
- cct_cli-0.7.9.0.dist-info/top_level.txt +1 -0
calc_terminal/aicore.py
ADDED
|
@@ -0,0 +1,2572 @@
|
|
|
1
|
+
"""
|
|
2
|
+
AI Integration Core for CCT [BETA].
|
|
3
|
+
Supports local Ollama or API-based models (OpenAI/Claude style) for chemistry.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
if __name__ == '__main__':
|
|
7
|
+
print("This is a library file and is not meant to be run directly.")
|
|
8
|
+
print("Please run 'python main.py' or 'python model.py' from the project root directory.")
|
|
9
|
+
import sys
|
|
10
|
+
sys.exit(1)
|
|
11
|
+
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
import json
|
|
15
|
+
import time
|
|
16
|
+
import sys
|
|
17
|
+
import base64
|
|
18
|
+
import zlib
|
|
19
|
+
import logging
|
|
20
|
+
import threading
|
|
21
|
+
import uuid
|
|
22
|
+
from urllib.parse import unquote, parse_qs, urlparse
|
|
23
|
+
from html import unescape as _html_unescape
|
|
24
|
+
try:
|
|
25
|
+
import requests
|
|
26
|
+
_HAS_REQUESTS = True
|
|
27
|
+
except ImportError:
|
|
28
|
+
requests = None
|
|
29
|
+
_HAS_REQUESTS = False
|
|
30
|
+
|
|
31
|
+
from . import theme
|
|
32
|
+
from . import identity
|
|
33
|
+
from .providers.provider_manager import (get_provider, get_provider_class,
|
|
34
|
+
list_builtin_providers, resolve_config,
|
|
35
|
+
fetch_models_for, connect_provider, mask_key)
|
|
36
|
+
from .providers.provider_manager import load_config as _pm_load_config
|
|
37
|
+
from .providers.provider_manager import save_config as _pm_save_config
|
|
38
|
+
|
|
39
|
+
_LOG = logging.getLogger("cct.aicore")
|
|
40
|
+
|
|
41
|
+
# Config is now managed by providers/provider_manager.py.
|
|
42
|
+
# kept here only for backward-compatible imports
|
|
43
|
+
CONFIG_FILE = os.path.join(os.path.expanduser("~"), ".cct_ai_config.json")
|
|
44
|
+
|
|
45
|
+
# Default chat system prompt, shared by query_ai()/stream_ai() when no
|
|
46
|
+
# caller-specific prompt is passed in (agent.py passes its own, richer
|
|
47
|
+
# prompts, which also fold in identity.IDENTITY_BLOCK -- see agent.py).
|
|
48
|
+
# Folding the identity block in here too means even code paths that call
|
|
49
|
+
# aicore directly with the bare default still answer "who made you"
|
|
50
|
+
# style questions correctly and consistently.
|
|
51
|
+
DEFAULT_SYSTEM_PROMPT = (
|
|
52
|
+
"You are CAT AI, a coding and science assistant inside the Coding Agent "
|
|
53
|
+
"calculations.\n\n" + identity.IDENTITY_BLOCK
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# ---------------------------------------------------------------- provider registry
|
|
57
|
+
# Built dynamically from the provider SDK so every registered provider shows up
|
|
58
|
+
# automatically. query_ai()/stream_ai() dispatch on `api_style` through the
|
|
59
|
+
# provider instance. To add a new provider, create a subclass of BaseProvider
|
|
60
|
+
# and call provider_manager.register_provider() — no other code needs to change.
|
|
61
|
+
PROVIDERS = {}
|
|
62
|
+
_BUILTIN_PROVIDER_DICT = None
|
|
63
|
+
|
|
64
|
+
def _build_provider_dict(force_reload=False):
|
|
65
|
+
global _BUILTIN_PROVIDER_DICT
|
|
66
|
+
if _BUILTIN_PROVIDER_DICT is not None and not force_reload:
|
|
67
|
+
return _BUILTIN_PROVIDER_DICT
|
|
68
|
+
d = {}
|
|
69
|
+
for info in list_builtin_providers():
|
|
70
|
+
pid = info["id"]
|
|
71
|
+
d[pid] = {
|
|
72
|
+
"base_url": info.get("url", ""),
|
|
73
|
+
"default_model": info.get("default_model", ""),
|
|
74
|
+
"api_style": info.get("api_style", "openai"),
|
|
75
|
+
"needs_key": info.get("needs_key", True),
|
|
76
|
+
"extra_headers": {},
|
|
77
|
+
}
|
|
78
|
+
# Add providers that aren't built-in but essential
|
|
79
|
+
extras = {
|
|
80
|
+
"openrouter": {"base_url": "https://openrouter.ai/api/v1",
|
|
81
|
+
"default_model": "openai/gpt-4o-mini", "api_style": "openai",
|
|
82
|
+
"needs_key": True,
|
|
83
|
+
"extra_headers": {"HTTP-Referer": "https://github.com/cct",
|
|
84
|
+
"X-Title": "Chemistry Calc Terminal"}},
|
|
85
|
+
"groq": {"base_url": "https://api.groq.com/openai/v1",
|
|
86
|
+
"default_model": "llama3.1-70b-versatile", "api_style": "openai",
|
|
87
|
+
"needs_key": True, "extra_headers": {}},
|
|
88
|
+
"ollama": {"base_url": "http://localhost:11434", "default_model": "llama3.3",
|
|
89
|
+
"api_style": "ollama", "needs_key": False, "extra_headers": {},
|
|
90
|
+
"auto_discover": True},
|
|
91
|
+
"moonshot": {"base_url": "https://api.moonshot.cn/v1", "default_model": "kimi-k3",
|
|
92
|
+
"api_style": "openai", "needs_key": True, "extra_headers": {}},
|
|
93
|
+
"xai": {"base_url": "https://api.x.ai/v1", "default_model": "grok-3",
|
|
94
|
+
"api_style": "openai", "needs_key": True, "extra_headers": {}},
|
|
95
|
+
}
|
|
96
|
+
d.update(extras)
|
|
97
|
+
# Load ALL providers from providers.json so any provider selected via
|
|
98
|
+
# /model or /provider has a valid base_url (nvidia, deepseek, etc.)
|
|
99
|
+
try:
|
|
100
|
+
from .models.manager import load_providers
|
|
101
|
+
for p in load_providers():
|
|
102
|
+
pid = p.get("id", "")
|
|
103
|
+
if pid and pid not in d:
|
|
104
|
+
ep = p.get("api_endpoint", "")
|
|
105
|
+
d[pid] = {
|
|
106
|
+
"base_url": ep,
|
|
107
|
+
"default_model": p.get("default_model", ""),
|
|
108
|
+
"api_style": p.get("api_style", "openai"),
|
|
109
|
+
"needs_key": bool(p.get("needs_key", True)),
|
|
110
|
+
"extra_headers": {},
|
|
111
|
+
}
|
|
112
|
+
elif pid and pid in d and not d[pid].get("base_url"):
|
|
113
|
+
d[pid]["base_url"] = p.get("api_endpoint", "")
|
|
114
|
+
except Exception:
|
|
115
|
+
pass
|
|
116
|
+
_BUILTIN_PROVIDER_DICT = d
|
|
117
|
+
return d
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _get_provider_info(provider_id):
|
|
121
|
+
"""Get provider info, with cache refresh if base_url is missing."""
|
|
122
|
+
providers = _build_provider_dict()
|
|
123
|
+
info = providers.get(provider_id)
|
|
124
|
+
# v0.7.10: If provider exists but has no base_url, refresh cache
|
|
125
|
+
if info and not info.get("base_url") and provider_id:
|
|
126
|
+
providers = _build_provider_dict(force_reload=True)
|
|
127
|
+
info = providers.get(provider_id)
|
|
128
|
+
return info
|
|
129
|
+
|
|
130
|
+
# ---------------------------------------------------------------- token usage
|
|
131
|
+
# Best-effort session token accounting. Real usage is pulled straight out of
|
|
132
|
+
# the provider's own response body when it reports one (OpenAI/Anthropic/
|
|
133
|
+
# Gemini/native-Ollama all do); anything that doesn't is estimated at ~4
|
|
134
|
+
# chars/token, the same rule-of-thumb every provider's own docs use.
|
|
135
|
+
SESSION_USAGE = {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0, "requests": 0}
|
|
136
|
+
|
|
137
|
+
# Known context-window sizes are data-driven now: curated model metadata
|
|
138
|
+
# lives in models/model_metadata.json (context_length), with the provider's
|
|
139
|
+
# fallback list in providers/providers.json. context_window_for() consults
|
|
140
|
+
# those stores first and only falls back to this default estimate.
|
|
141
|
+
DEFAULT_CONTEXT_WINDOW = 32000
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def estimate_tokens(text):
|
|
145
|
+
"""Rough ~4-chars-per-token estimate, used whenever a provider doesn't
|
|
146
|
+
report real usage numbers back."""
|
|
147
|
+
if not text:
|
|
148
|
+
return 0
|
|
149
|
+
return max(1, len(str(text)) // 4)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def context_window_for(model):
|
|
153
|
+
if not model:
|
|
154
|
+
return DEFAULT_CONTEXT_WINDOW
|
|
155
|
+
model = str(model).lower()
|
|
156
|
+
# Data-driven: curated metadata first (exact id, any provider)...
|
|
157
|
+
try:
|
|
158
|
+
from .models import manager as _mgr
|
|
159
|
+
meta = _mgr.get_model_meta(model)
|
|
160
|
+
ctx = meta.get("context_length")
|
|
161
|
+
if ctx:
|
|
162
|
+
return int(ctx)
|
|
163
|
+
except Exception:
|
|
164
|
+
pass
|
|
165
|
+
# ...then heuristic family match across all curated metadata.
|
|
166
|
+
try:
|
|
167
|
+
from .models.registry import get_model_info
|
|
168
|
+
for _, m in _load_metadata_items():
|
|
169
|
+
if m.get("family") and m["family"].lower() in model:
|
|
170
|
+
ctx = m.get("context_length")
|
|
171
|
+
if ctx:
|
|
172
|
+
return int(ctx)
|
|
173
|
+
except Exception:
|
|
174
|
+
pass
|
|
175
|
+
return DEFAULT_CONTEXT_WINDOW
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _load_metadata_items():
|
|
179
|
+
"""Yield (model_id, metadata_dict) for every curated model."""
|
|
180
|
+
from .models import manager as _mgr
|
|
181
|
+
try:
|
|
182
|
+
with open(_mgr.MODEL_METADATA_FILE, "r", encoding="utf-8") as f:
|
|
183
|
+
import json as _json
|
|
184
|
+
for m in _json.load(f).get("models", []):
|
|
185
|
+
yield m.get("id", ""), m
|
|
186
|
+
except Exception:
|
|
187
|
+
return iter(())
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def record_usage(prompt_tokens, completion_tokens):
|
|
191
|
+
SESSION_USAGE["prompt_tokens"] += int(prompt_tokens or 0)
|
|
192
|
+
SESSION_USAGE["completion_tokens"] += int(completion_tokens or 0)
|
|
193
|
+
SESSION_USAGE["total_tokens"] += int(prompt_tokens or 0) + int(completion_tokens or 0)
|
|
194
|
+
SESSION_USAGE["requests"] += 1
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def get_session_usage():
|
|
198
|
+
"""Snapshot of this session's token use plus a best-effort estimate of
|
|
199
|
+
how much of the current model's context window is left."""
|
|
200
|
+
config = load_config()
|
|
201
|
+
model = config.get("model", "")
|
|
202
|
+
window = context_window_for(model)
|
|
203
|
+
used = SESSION_USAGE["total_tokens"]
|
|
204
|
+
return {
|
|
205
|
+
"prompt_tokens": SESSION_USAGE["prompt_tokens"],
|
|
206
|
+
"completion_tokens": SESSION_USAGE["completion_tokens"],
|
|
207
|
+
"total_tokens": used,
|
|
208
|
+
"requests": SESSION_USAGE["requests"],
|
|
209
|
+
"model": model or "(not configured)",
|
|
210
|
+
"context_window": window,
|
|
211
|
+
"remaining_estimate": max(0, window - used),
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def reset_session_usage():
|
|
216
|
+
SESSION_USAGE.update({"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0, "requests": 0})
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _track_usage_from_data(api_style, data, prompt_text, completion_text):
|
|
220
|
+
"""Pull real usage numbers out of the provider's response body when
|
|
221
|
+
present; fall back to the character-based estimate otherwise. Never
|
|
222
|
+
raises — token accounting is a nice-to-have, not a dependency."""
|
|
223
|
+
try:
|
|
224
|
+
if api_style == "openai" and isinstance(data, dict) and isinstance(data.get("usage"), dict):
|
|
225
|
+
u = data["usage"]
|
|
226
|
+
record_usage(u.get("prompt_tokens", estimate_tokens(prompt_text)),
|
|
227
|
+
u.get("completion_tokens", estimate_tokens(completion_text)))
|
|
228
|
+
return
|
|
229
|
+
if api_style == "anthropic" and isinstance(data, dict) and isinstance(data.get("usage"), dict):
|
|
230
|
+
u = data["usage"]
|
|
231
|
+
record_usage(u.get("input_tokens", estimate_tokens(prompt_text)),
|
|
232
|
+
u.get("output_tokens", estimate_tokens(completion_text)))
|
|
233
|
+
return
|
|
234
|
+
if api_style == "gemini" and isinstance(data, dict) and isinstance(data.get("usageMetadata"), dict):
|
|
235
|
+
u = data["usageMetadata"]
|
|
236
|
+
record_usage(u.get("promptTokenCount", estimate_tokens(prompt_text)),
|
|
237
|
+
u.get("candidatesTokenCount", estimate_tokens(completion_text)))
|
|
238
|
+
return
|
|
239
|
+
if api_style == "ollama" and isinstance(data, dict) and "eval_count" in data:
|
|
240
|
+
record_usage(data.get("prompt_eval_count", estimate_tokens(prompt_text)),
|
|
241
|
+
data.get("eval_count", estimate_tokens(completion_text)))
|
|
242
|
+
return
|
|
243
|
+
except Exception:
|
|
244
|
+
pass
|
|
245
|
+
record_usage(estimate_tokens(prompt_text), estimate_tokens(completion_text))
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
# ---------------------------------------------------------------- model switch
|
|
249
|
+
# Model suggestions are data-driven: the provider's fallback list in
|
|
250
|
+
# providers.json (seeded from the provider's own model catalog). Live
|
|
251
|
+
# fetching happens when /model opens — see models/manager.py.
|
|
252
|
+
def common_models(provider_id):
|
|
253
|
+
"""Recommended/known model ids for a provider (providers.json
|
|
254
|
+
fallback list, plus any cached live list). Never raises."""
|
|
255
|
+
out = []
|
|
256
|
+
try:
|
|
257
|
+
from .models import manager as _mgr
|
|
258
|
+
provider = _mgr.get_provider(provider_id)
|
|
259
|
+
if provider:
|
|
260
|
+
out = [str(m) for m in provider.get("fallback_models", [])]
|
|
261
|
+
cached = _mgr.get_cached(provider_id)
|
|
262
|
+
if cached:
|
|
263
|
+
live = [str(m) for m in cached[0]]
|
|
264
|
+
for m in live:
|
|
265
|
+
if m not in out:
|
|
266
|
+
out.append(m)
|
|
267
|
+
except Exception:
|
|
268
|
+
pass
|
|
269
|
+
return out[:40]
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def switch_model(model_name):
|
|
273
|
+
"""Change just the model for the currently configured provider —
|
|
274
|
+
lighter-weight than the full /ai setup_ai() wizard. Returns (ok, msg)."""
|
|
275
|
+
model_name = (model_name or "").strip()
|
|
276
|
+
if not model_name:
|
|
277
|
+
return False, "No model name given."
|
|
278
|
+
config = load_config()
|
|
279
|
+
if not config.get("provider"):
|
|
280
|
+
return False, "No AI provider configured yet. Run /ai or /model to set one up first."
|
|
281
|
+
old = config.get("model")
|
|
282
|
+
config["model"] = model_name
|
|
283
|
+
save_config(config)
|
|
284
|
+
return True, f"Model switched from '{old}' to '{model_name}' ({config['provider']})."
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
# load_config/save_config delegate to provider_manager (so callers that
|
|
288
|
+
# do `from calc_terminal.aicore import load_config` still work).
|
|
289
|
+
def load_config():
|
|
290
|
+
return _pm_load_config()
|
|
291
|
+
|
|
292
|
+
def save_config(config):
|
|
293
|
+
return _pm_save_config(config)
|
|
294
|
+
|
|
295
|
+
def list_models(config):
|
|
296
|
+
"""Fetch the live list of model IDs actually available to this
|
|
297
|
+
provider/key. Returns a (possibly empty) list of strings, never
|
|
298
|
+
raises — callers should treat an empty list as 'listing unsupported
|
|
299
|
+
or unreachable right now', not as an error.
|
|
300
|
+
|
|
301
|
+
Priority chain (models/manager.py): provider API -> disk cache
|
|
302
|
+
(cache/models/) -> built-in fallback list (providers.json).
|
|
303
|
+
Successful fetches are cached on disk with a timestamp.
|
|
304
|
+
"""
|
|
305
|
+
if not _HAS_REQUESTS:
|
|
306
|
+
return []
|
|
307
|
+
provider = config.get("provider")
|
|
308
|
+
if not provider:
|
|
309
|
+
return []
|
|
310
|
+
# Data-driven resolution with disk caching
|
|
311
|
+
try:
|
|
312
|
+
from .models import manager as _mgr
|
|
313
|
+
models, source = _mgr.get_models(provider, config)
|
|
314
|
+
# "default" is the last-resort placeholder — treat as unsupported
|
|
315
|
+
if models and models != ["default"]:
|
|
316
|
+
return list(models)
|
|
317
|
+
except Exception:
|
|
318
|
+
pass
|
|
319
|
+
# Legacy fallback for providers not covered by the manager
|
|
320
|
+
info = _get_provider_info(provider)
|
|
321
|
+
api_style = info["api_style"] if info else "openai"
|
|
322
|
+
if provider == "ollama":
|
|
323
|
+
base_url = config.get("ollama_url") or (info["base_url"] if info else "http://localhost:11434")
|
|
324
|
+
else:
|
|
325
|
+
base_url = config.get("base_url") or config.get("api_url") or (info["base_url"] if info else "")
|
|
326
|
+
base_url = base_url.rstrip("/")
|
|
327
|
+
api_key = config.get("api_key", "")
|
|
328
|
+
extra_headers = info["extra_headers"] if info else {}
|
|
329
|
+
try:
|
|
330
|
+
if api_style == "ollama" and not base_url.endswith("/v1"):
|
|
331
|
+
resp = requests.get(f"{base_url}/api/tags", timeout=8)
|
|
332
|
+
if resp.status_code >= 400:
|
|
333
|
+
return []
|
|
334
|
+
return sorted(m.get("name", "") for m in resp.json().get("models", []) if m.get("name"))
|
|
335
|
+
elif api_style == "openai":
|
|
336
|
+
headers = {"Content-Type": "application/json"}
|
|
337
|
+
if api_key:
|
|
338
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
339
|
+
headers.update(extra_headers)
|
|
340
|
+
resp = requests.get(f"{base_url}/models", headers=headers, timeout=10)
|
|
341
|
+
if resp.status_code >= 400:
|
|
342
|
+
return []
|
|
343
|
+
data = resp.json().get("data", [])
|
|
344
|
+
return sorted(m.get("id", "") for m in data if isinstance(m, dict) and m.get("id"))
|
|
345
|
+
elif api_style == "gemini":
|
|
346
|
+
resp = requests.get(f"{base_url}/models?key={api_key}", timeout=10)
|
|
347
|
+
if resp.status_code >= 400:
|
|
348
|
+
return []
|
|
349
|
+
out = []
|
|
350
|
+
for m in resp.json().get("models", []):
|
|
351
|
+
methods = m.get("supportedGenerationMethods", [])
|
|
352
|
+
if not methods or "generateContent" in methods:
|
|
353
|
+
out.append(m.get("name", "").split("/")[-1])
|
|
354
|
+
return sorted(set(n for n in out if n))
|
|
355
|
+
else:
|
|
356
|
+
return []
|
|
357
|
+
except Exception:
|
|
358
|
+
return []
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _prompt_model(provider, default_model, available):
|
|
362
|
+
"""Prompt for a model name, offering a live-fetched picker when one is
|
|
363
|
+
available, but ALWAYS accepting free-form text too — so any model,
|
|
364
|
+
including brand-new ones not in the list yet, still works. Also
|
|
365
|
+
catches the classic mistake of typing the provider's name itself
|
|
366
|
+
(e.g. typing 'gemini' as the model) instead of a real model id.
|
|
367
|
+
"""
|
|
368
|
+
shown = available[:20]
|
|
369
|
+
if shown:
|
|
370
|
+
print(theme.dim(f" Found {len(available)} model(s) available to this key/server."
|
|
371
|
+
+ (f" Showing first {len(shown)}:" if len(available) > len(shown) else "")))
|
|
372
|
+
for i, m in enumerate(shown, 1):
|
|
373
|
+
print(theme.cyan(f" {i}.") + " " + theme.text(m))
|
|
374
|
+
print(theme.faint(f" Type a number to pick one, or type ANY model name directly."))
|
|
375
|
+
|
|
376
|
+
while True:
|
|
377
|
+
model_in = input(theme.dim(f" Model (default: {default_model}) \u25b8 ")).strip()
|
|
378
|
+
if shown and model_in.isdigit() and 1 <= int(model_in) <= len(shown):
|
|
379
|
+
return shown[int(model_in) - 1]
|
|
380
|
+
candidate = _clean_model_name(model_in, default_model)
|
|
381
|
+
if candidate.strip().lower() == str(provider).strip().lower():
|
|
382
|
+
print(theme.orange(f" '{candidate}' is the provider name, not a model id — "
|
|
383
|
+
f"e.g. try '{default_model}', or pick a number above."))
|
|
384
|
+
continue
|
|
385
|
+
return candidate
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def setup_ai():
|
|
389
|
+
if not _HAS_REQUESTS:
|
|
390
|
+
print(theme.red("\n Error: The 'requests' library is not installed."))
|
|
391
|
+
print(theme.dim(" AI features require the 'requests' package to communicate with APIs."))
|
|
392
|
+
print(theme.dim(" Please run: ") + theme.text("pip install requests", bold=True))
|
|
393
|
+
time.sleep(3)
|
|
394
|
+
return
|
|
395
|
+
|
|
396
|
+
while True:
|
|
397
|
+
theme.clear_screen()
|
|
398
|
+
# Build the menu dynamically from PROVIDERS so every registered
|
|
399
|
+
# provider shows up without hand-editing this list.
|
|
400
|
+
menu_lines = [
|
|
401
|
+
theme.badge("BETA", theme.BG_WARN) + " " + theme.purple("AI CONFIGURATION", bold=True),
|
|
402
|
+
theme.dim("CAT AI is specialized for complex problem solving & scientific computing."),
|
|
403
|
+
theme.dim("Any provider works with ANY model it supports — pick from the live list"),
|
|
404
|
+
theme.dim("shown after your key, or type a model name yourself at any time."),
|
|
405
|
+
"",
|
|
406
|
+
theme.text("Choose your AI Provider:"),
|
|
407
|
+
]
|
|
408
|
+
provider_dict = _build_provider_dict()
|
|
409
|
+
keys = list(provider_dict.keys())
|
|
410
|
+
for i, name in enumerate(keys, 1):
|
|
411
|
+
info = provider_dict[name]
|
|
412
|
+
tag = "Local, no key" if not info["needs_key"] else "API Key"
|
|
413
|
+
menu_lines.append(theme.cyan(f"{i}. {name.capitalize()} ({tag})"))
|
|
414
|
+
menu_lines.append(theme.cyan(f"{len(keys) + 1}. Custom API Endpoint"))
|
|
415
|
+
menu_lines.append(theme.cyan(f"{len(keys) + 2}. Back to Menu"))
|
|
416
|
+
menu_lines += ["", theme.faint("Press Enter to skip / use default settings.")]
|
|
417
|
+
print(theme.panel(menu_lines, title="AI SETUP", color=theme.PURPLE, width=70))
|
|
418
|
+
|
|
419
|
+
choice = input(theme.dim(" choice \u25b8 ")).strip()
|
|
420
|
+
back_index = len(keys) + 2
|
|
421
|
+
if choice == str(back_index) or choice == "":
|
|
422
|
+
return
|
|
423
|
+
|
|
424
|
+
config = load_config()
|
|
425
|
+
|
|
426
|
+
# ---- pick a provider from the registry ----
|
|
427
|
+
if choice.isdigit() and 1 <= int(choice) <= len(keys):
|
|
428
|
+
name = keys[int(choice) - 1]
|
|
429
|
+
info = provider_dict[name]
|
|
430
|
+
config["provider"] = name
|
|
431
|
+
# v0.7.10: Always save base_url when selecting a provider
|
|
432
|
+
if info.get("base_url"):
|
|
433
|
+
config["base_url"] = info["base_url"]
|
|
434
|
+
if info["needs_key"]:
|
|
435
|
+
config["api_key"] = input(theme.dim(f" Enter {name.capitalize()} API Key \u25b8 ")).strip()
|
|
436
|
+
available = []
|
|
437
|
+
if info["needs_key"] or name == "ollama":
|
|
438
|
+
print(theme.dim(" Checking which models are available..."))
|
|
439
|
+
available = list_models(config)
|
|
440
|
+
config["model"] = _prompt_model(name, info["default_model"], available)
|
|
441
|
+
break
|
|
442
|
+
|
|
443
|
+
# ---- custom endpoint (user supplies everything) ----
|
|
444
|
+
elif choice == str(len(keys) + 1):
|
|
445
|
+
config["provider"] = "custom"
|
|
446
|
+
config["api_url"] = input(theme.dim(" API Endpoint URL \u25b8 ")).strip()
|
|
447
|
+
config["base_url"] = config["api_url"]
|
|
448
|
+
config["api_key"] = input(theme.dim(" API Key \u25b8 ")).strip()
|
|
449
|
+
print(theme.dim(" Checking which models are available (best-effort, OpenAI-style /models)..."))
|
|
450
|
+
available = list_models(config)
|
|
451
|
+
config["model"] = _prompt_model("custom", "", available) if available else \
|
|
452
|
+
input(theme.dim(" Model Name \u25b8 ")).strip()
|
|
453
|
+
break
|
|
454
|
+
|
|
455
|
+
save_config(config)
|
|
456
|
+
print(theme.green("\n AI Configuration saved successfully!"))
|
|
457
|
+
print(theme.dim(" Verifying connection..."))
|
|
458
|
+
ok, msg, models = verify_connection(config)
|
|
459
|
+
print((theme.green if ok else theme.red)((" \u2713 " if ok else " \u2717 ") + msg))
|
|
460
|
+
if models:
|
|
461
|
+
preview = ", ".join(models[:8])
|
|
462
|
+
print(theme.faint(f" Models seen: {preview}{', ...' if len(models) > 8 else ''}"))
|
|
463
|
+
time.sleep(1.5)
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def _clean_model_name(user_input, default):
|
|
467
|
+
"""Normalise the model name typed at the prompt. Empty input, or words like
|
|
468
|
+
'yes'/'no'/'ok' (commonly typed meaning 'use the default'), fall back to the
|
|
469
|
+
real default — never saved literally as a model name."""
|
|
470
|
+
if not user_input:
|
|
471
|
+
return default
|
|
472
|
+
if user_input.lower() in ("yes", "y", "no", "n", "ok", "true", "false", "default"):
|
|
473
|
+
return default
|
|
474
|
+
return user_input
|
|
475
|
+
|
|
476
|
+
def verify_connection(config=None):
|
|
477
|
+
"""Actively test the configured AI provider/model — a real network
|
|
478
|
+
round trip, not just 'is a config file present'. Returns
|
|
479
|
+
(ok: bool, message: str, models: list[str]).
|
|
480
|
+
|
|
481
|
+
Uses the provider SDK for all built-in providers; falls back to
|
|
482
|
+
the legacy per-api_style logic for non-SDK providers (ollama, groq).
|
|
483
|
+
"""
|
|
484
|
+
if not _HAS_REQUESTS:
|
|
485
|
+
return False, "The 'requests' library is not installed. Run: pip install requests", []
|
|
486
|
+
|
|
487
|
+
config = config or load_config()
|
|
488
|
+
provider = config.get("provider")
|
|
489
|
+
if not provider:
|
|
490
|
+
return False, "No AI provider configured yet. Run /ai (or /ai-verify after setup) to configure one.", []
|
|
491
|
+
|
|
492
|
+
# Try provider SDK first
|
|
493
|
+
try:
|
|
494
|
+
inst = get_provider(config)
|
|
495
|
+
if inst:
|
|
496
|
+
ok, msg, models = inst.connect()
|
|
497
|
+
return ok, msg, models
|
|
498
|
+
except Exception:
|
|
499
|
+
pass
|
|
500
|
+
|
|
501
|
+
# Fallback for non-SDK providers
|
|
502
|
+
info = _get_provider_info(provider)
|
|
503
|
+
api_style = info["api_style"] if info else "openai"
|
|
504
|
+
|
|
505
|
+
if provider == "ollama":
|
|
506
|
+
base_url = config.get("ollama_url") or (info["base_url"] if info else "http://localhost:11434")
|
|
507
|
+
else:
|
|
508
|
+
base_url = config.get("base_url") or config.get("api_url") or (info["base_url"] if info else "")
|
|
509
|
+
base_url = base_url.rstrip("/")
|
|
510
|
+
|
|
511
|
+
api_key = config.get("api_key", "")
|
|
512
|
+
model = config.get("model") or (info["default_model"] if info else "")
|
|
513
|
+
extra_headers = info["extra_headers"] if info else {}
|
|
514
|
+
|
|
515
|
+
try:
|
|
516
|
+
if api_style == "ollama" and not base_url.endswith("/v1"):
|
|
517
|
+
resp = requests.get(f"{base_url}/api/tags", timeout=8)
|
|
518
|
+
_raise_for_status(resp)
|
|
519
|
+
models = [m.get("name", "") for m in resp.json().get("models", [])]
|
|
520
|
+
if models and model and not any(model in m for m in models):
|
|
521
|
+
return (True,
|
|
522
|
+
f"Connected to Ollama at {base_url}, but model '{model}' isn't pulled locally yet. "
|
|
523
|
+
f"Run: ollama pull {model}",
|
|
524
|
+
models)
|
|
525
|
+
return True, f"Connected to Ollama at {base_url} \u2014 {len(models)} local model(s) available.", models
|
|
526
|
+
|
|
527
|
+
elif api_style == "openai":
|
|
528
|
+
headers = {"Content-Type": "application/json"}
|
|
529
|
+
if api_key:
|
|
530
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
531
|
+
headers.update(extra_headers)
|
|
532
|
+
resp = requests.get(f"{base_url}/models", headers=headers, timeout=10)
|
|
533
|
+
_raise_for_status(resp)
|
|
534
|
+
data = resp.json().get("data", [])
|
|
535
|
+
models = [m.get("id", "") for m in data if isinstance(m, dict)]
|
|
536
|
+
note = ""
|
|
537
|
+
if models and model and not any(model == m or model in m for m in models):
|
|
538
|
+
note = f" Note: '{model}' wasn't in the list returned for this key — double-check the model name."
|
|
539
|
+
return True, f"Connected to {provider} at {base_url} \u2014 {len(models)} model(s) visible to this key.{note}", models
|
|
540
|
+
|
|
541
|
+
elif api_style == "gemini":
|
|
542
|
+
resp = requests.get(f"{base_url}/models?key={api_key}", timeout=10)
|
|
543
|
+
_raise_for_status(resp)
|
|
544
|
+
models = []
|
|
545
|
+
for m in resp.json().get("models", []):
|
|
546
|
+
methods = m.get("supportedGenerationMethods", [])
|
|
547
|
+
if not methods or "generateContent" in methods:
|
|
548
|
+
models.append(m.get("name", "").split("/")[-1])
|
|
549
|
+
note = ""
|
|
550
|
+
if models and model and model not in models:
|
|
551
|
+
note = f" Note: '{model}' wasn't in the list returned for this key — pick one of the models shown, or double-check the name."
|
|
552
|
+
return True, f"Connected to Gemini \u2014 {len(models)} model(s) visible to this key.{note}", models
|
|
553
|
+
|
|
554
|
+
else:
|
|
555
|
+
reply = query_ai("Reply with only the single word: OK",
|
|
556
|
+
system_prompt="You are a connectivity test. Reply with only: OK")
|
|
557
|
+
failure_markers = ("could not reach", "the ai request timed out",
|
|
558
|
+
"error connecting", "http 4", "http 5", "not configured")
|
|
559
|
+
if reply and not any(reply.lower().startswith(m) for m in failure_markers):
|
|
560
|
+
return True, f"Connected to {provider}, model '{model}' responded successfully.", []
|
|
561
|
+
return False, f"Connection test failed: {reply}", []
|
|
562
|
+
|
|
563
|
+
except requests.exceptions.ConnectionError:
|
|
564
|
+
return False, f"Could not reach {base_url}. Check the URL / your internet, or that Ollama is running (ollama serve).", []
|
|
565
|
+
except requests.exceptions.Timeout:
|
|
566
|
+
return False, "Connection timed out.", []
|
|
567
|
+
except RuntimeError as e:
|
|
568
|
+
return False, f"Verification failed: {e}", []
|
|
569
|
+
except Exception as e:
|
|
570
|
+
return False, f"Verification failed: {e}", []
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
def _sanitize_history(history):
|
|
574
|
+
"""Normalizes whatever a caller hands us into a clean list of
|
|
575
|
+
(role, text) pairs with only 'user'/'assistant' roles, no empty
|
|
576
|
+
turns, and no in-flight streaming placeholder (empty text). This
|
|
577
|
+
is the ONE place that decides what "conversation history" means
|
|
578
|
+
for every provider below — every api_style builds its request
|
|
579
|
+
from this, so none of them can silently drop it again."""
|
|
580
|
+
if not history:
|
|
581
|
+
return []
|
|
582
|
+
out = []
|
|
583
|
+
for item in history:
|
|
584
|
+
if isinstance(item, (list, tuple)) and len(item) == 2:
|
|
585
|
+
role, text = item
|
|
586
|
+
elif isinstance(item, dict):
|
|
587
|
+
role, text = item.get("role"), item.get("text", item.get("content", ""))
|
|
588
|
+
else:
|
|
589
|
+
continue
|
|
590
|
+
role = "assistant" if role in ("assistant", "ai", "model") else "user"
|
|
591
|
+
text = (text or "").strip()
|
|
592
|
+
if text:
|
|
593
|
+
out.append((role, text))
|
|
594
|
+
return out
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
def _history_char_count(history):
|
|
598
|
+
return sum(len(t) for _, t in history)
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
def _openai_messages(system_prompt, history, prompt, attachments=None, vision=False, model=""):
|
|
602
|
+
"""Shared by every OpenAI-compatible api_style (openai, groq,
|
|
603
|
+
openrouter, vLLM, Ollama's /v1 endpoint) — system prompt, then
|
|
604
|
+
every prior turn in order, then the new user prompt.
|
|
605
|
+
|
|
606
|
+
v0.7.8.1: when `vision` is True and attachments carry images, the
|
|
607
|
+
final user message becomes a content array (text + image_url data
|
|
608
|
+
URLs) so vision-capable models actually see the attached images
|
|
609
|
+
instead of only reading their metadata. v0.7.8.2: image parts use
|
|
610
|
+
the OpenAI shape only here; Anthropic/Gemini get their own shapes
|
|
611
|
+
(see _user_content)."""
|
|
612
|
+
messages = []
|
|
613
|
+
mod_lower = (model or "").lower()
|
|
614
|
+
is_o1_mini = "o1-mini" in mod_lower or "o1-preview" in mod_lower
|
|
615
|
+
sys_role = "developer" if ("o1" in mod_lower or "o3" in mod_lower) and not is_o1_mini else "system"
|
|
616
|
+
if system_prompt:
|
|
617
|
+
if is_o1_mini:
|
|
618
|
+
prompt = f"{system_prompt}\n\n{prompt}"
|
|
619
|
+
else:
|
|
620
|
+
messages.append({"role": sys_role, "content": system_prompt})
|
|
621
|
+
for role, text in history:
|
|
622
|
+
messages.append({"role": role, "content": text})
|
|
623
|
+
messages.append({"role": "user", "content": _user_content(
|
|
624
|
+
prompt, attachments, vision, api_style="openai")})
|
|
625
|
+
return messages
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
def _anthropic_messages(history, prompt, attachments=None, vision=False):
|
|
629
|
+
"""Anthropic keeps `system` as its own top-level field, so this
|
|
630
|
+
only builds the `messages` array (user/assistant turns)."""
|
|
631
|
+
messages = [{"role": role, "content": text} for role, text in history]
|
|
632
|
+
messages.append({"role": "user", "content": _user_content(
|
|
633
|
+
prompt, attachments, vision, api_style="anthropic")})
|
|
634
|
+
return messages
|
|
635
|
+
|
|
636
|
+
|
|
637
|
+
def _gemini_contents(history, prompt, attachments=None, vision=False):
|
|
638
|
+
"""Gemini calls the assistant role 'model', not 'assistant'."""
|
|
639
|
+
contents = []
|
|
640
|
+
for role, text in history:
|
|
641
|
+
contents.append({"role": ("model" if role == "assistant" else "user"),
|
|
642
|
+
"parts": [{"text": text}]})
|
|
643
|
+
contents.append({"role": "user", "parts": _user_content(
|
|
644
|
+
prompt, attachments, vision, api_style="gemini")})
|
|
645
|
+
return contents
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def _user_content(prompt, attachments=None, vision=False, api_style="openai"):
|
|
649
|
+
"""The final user message content. Plain string when there is
|
|
650
|
+
nothing to attach natively; a provider-correct content array with
|
|
651
|
+
text + image parts when vision is available AND readable images are
|
|
652
|
+
attached. Images that failed to read are skipped here — their
|
|
653
|
+
textual fallback block (built by the attachment manager) still
|
|
654
|
+
carries the metadata.
|
|
655
|
+
|
|
656
|
+
v0.7.8.2 (attachment pipeline fix): the image parts are shaped per
|
|
657
|
+
provider — Anthropic and Gemini reject OpenAI's `image_url` block
|
|
658
|
+
(they each have their own schema), so a vision-capable Anthropic or
|
|
659
|
+
Gemini model previously received a malformed content array and the
|
|
660
|
+
attached image never reached it. Each api_style now emits its own
|
|
661
|
+
native shape:
|
|
662
|
+
- openai: {"type": "image_url", "image_url": {url: data-url}}
|
|
663
|
+
- anthropic: {"type": "image", "source": {base64, media_type}}
|
|
664
|
+
- gemini: {"inline_data": {mime_type, data}}
|
|
665
|
+
"""
|
|
666
|
+
if not vision or not attachments:
|
|
667
|
+
return prompt
|
|
668
|
+
images = []
|
|
669
|
+
try:
|
|
670
|
+
from . import attachments as _att
|
|
671
|
+
for att in attachments:
|
|
672
|
+
if _att.attachment_has_image_payload(att):
|
|
673
|
+
# v0.7.9.0: the vision pipeline may hand us an ALREADY-
|
|
674
|
+
# normalized base64 payload (metadata["inline_b64"]) — use
|
|
675
|
+
# it directly instead of re-reading/re-encoding the file.
|
|
676
|
+
inline = None
|
|
677
|
+
try:
|
|
678
|
+
inline = (att.metadata or {}).get("inline_b64")
|
|
679
|
+
except Exception:
|
|
680
|
+
inline = None
|
|
681
|
+
if inline:
|
|
682
|
+
images.append((getattr(att, "mime_type", "image/png") or "image/png",
|
|
683
|
+
inline))
|
|
684
|
+
continue
|
|
685
|
+
try:
|
|
686
|
+
mime, b64 = _att.encode_image_data_url(att.path)
|
|
687
|
+
except Exception:
|
|
688
|
+
continue
|
|
689
|
+
images.append((mime, b64))
|
|
690
|
+
except Exception:
|
|
691
|
+
return prompt
|
|
692
|
+
if not images:
|
|
693
|
+
return prompt
|
|
694
|
+
parts = []
|
|
695
|
+
if api_style == "anthropic":
|
|
696
|
+
parts.append({"type": "text", "text": prompt})
|
|
697
|
+
for mime, b64 in images:
|
|
698
|
+
parts.append({"type": "image",
|
|
699
|
+
"source": {"type": "base64", "media_type": mime,
|
|
700
|
+
"data": b64}})
|
|
701
|
+
return parts
|
|
702
|
+
if api_style == "gemini":
|
|
703
|
+
parts.append({"text": prompt})
|
|
704
|
+
for mime, b64 in images:
|
|
705
|
+
parts.append({"inline_data": {"mime_type": mime, "data": b64}})
|
|
706
|
+
return parts
|
|
707
|
+
parts.append({"type": "text", "text": prompt})
|
|
708
|
+
for mime, b64 in images:
|
|
709
|
+
parts.append({"type": "image_url",
|
|
710
|
+
"image_url": {"url": f"data:{mime};base64,{b64}"}})
|
|
711
|
+
return parts
|
|
712
|
+
|
|
713
|
+
|
|
714
|
+
def _ollama_native_prompt(system_prompt, history, prompt):
|
|
715
|
+
"""The native /api/generate endpoint (no v1 alias) only accepts one
|
|
716
|
+
flat `prompt` string — no messages array. To keep it from forgetting
|
|
717
|
+
the conversation the same way the message-based providers would, we
|
|
718
|
+
fold prior turns into the prompt as a plain transcript. `system` is
|
|
719
|
+
still passed separately via the `system` field."""
|
|
720
|
+
if not history:
|
|
721
|
+
return prompt
|
|
722
|
+
lines = [f"{'Assistant' if role == 'assistant' else 'User'}: {text}" for role, text in history]
|
|
723
|
+
transcript = "\n".join(lines)
|
|
724
|
+
return f"Conversation so far:\n{transcript}\n\nUser: {prompt}\nAssistant:"
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
def _resolve_provider(config):
|
|
728
|
+
"""Shared provider/URL/model resolution — used by both query_ai
|
|
729
|
+
(blocking) and stream_ai (generator) so there's one place that
|
|
730
|
+
decides which base_url/model/headers a request uses, not two that
|
|
731
|
+
can drift apart. Returns (api_style, base_url, api_key, model,
|
|
732
|
+
temperature, extra_headers)."""
|
|
733
|
+
provider = config.get("provider")
|
|
734
|
+
# Try provider SDK first
|
|
735
|
+
_inst = get_provider(config)
|
|
736
|
+
if _inst:
|
|
737
|
+
base_url = _inst.get_base_url()
|
|
738
|
+
# v0.7.10: Ensure base_url is never empty for non-SDK providers
|
|
739
|
+
if not base_url and provider:
|
|
740
|
+
info = _get_provider_info(provider)
|
|
741
|
+
if info and info.get("base_url"):
|
|
742
|
+
base_url = info["base_url"]
|
|
743
|
+
return (_inst.API_STYLE, base_url, _inst.get_api_key(),
|
|
744
|
+
_inst.get_model(), config.get("temperature"), _inst.get_extra_headers())
|
|
745
|
+
info = _get_provider_info(provider)
|
|
746
|
+
api_style = info["api_style"] if info else "openai"
|
|
747
|
+
if provider == "ollama":
|
|
748
|
+
base_url = config.get("ollama_url") or (info["base_url"] if info else "http://localhost:11434")
|
|
749
|
+
else:
|
|
750
|
+
base_url = config.get("base_url") or config.get("api_url") or (info["base_url"] if info else "")
|
|
751
|
+
# v0.7.10: If still no base_url, try to get it from providers.json
|
|
752
|
+
if not base_url and provider:
|
|
753
|
+
try:
|
|
754
|
+
from .models.manager import load_providers
|
|
755
|
+
for p in load_providers():
|
|
756
|
+
if p.get("id") == provider:
|
|
757
|
+
base_url = p.get("api_endpoint", "")
|
|
758
|
+
if base_url:
|
|
759
|
+
break
|
|
760
|
+
except Exception:
|
|
761
|
+
pass
|
|
762
|
+
base_url = base_url.rstrip("/") if base_url else ""
|
|
763
|
+
api_key = config.get("api_key", "")
|
|
764
|
+
if not (api_key or "").strip() and provider:
|
|
765
|
+
try:
|
|
766
|
+
from .providers.provider_manager import get_env_api_key
|
|
767
|
+
api_key = get_env_api_key(provider)
|
|
768
|
+
except Exception:
|
|
769
|
+
pass
|
|
770
|
+
model = config.get("model") or (info["default_model"] if info else "gpt-4o-mini")
|
|
771
|
+
temperature = config.get("temperature")
|
|
772
|
+
extra_headers = info["extra_headers"] if info else {}
|
|
773
|
+
return api_style, base_url, api_key, model, temperature, extra_headers
|
|
774
|
+
|
|
775
|
+
|
|
776
|
+
# The exact prefixes query_ai() (and query_ai_with_image()) return
|
|
777
|
+
# instead of raising, on every known failure path — copied verbatim
|
|
778
|
+
# from those functions' own `return` statements below so this can
|
|
779
|
+
# never drift out of sync silently. Used by callers (app.py's cmd_ai/
|
|
780
|
+
# cmd_agent, ui/app.py's _stream_worker) that want to show a real
|
|
781
|
+
# error-recovery card (spec section 19) instead of rendering the
|
|
782
|
+
# failure as if it were a normal chat answer.
|
|
783
|
+
_ERROR_SIGNATURES = (
|
|
784
|
+
"The 'requests' library is required for AI features.",
|
|
785
|
+
"AI not configured.",
|
|
786
|
+
"Could not reach the AI server.",
|
|
787
|
+
"The AI request timed out.",
|
|
788
|
+
"Error connecting to AI:",
|
|
789
|
+
"No model response received.",
|
|
790
|
+
"Invalid API key.",
|
|
791
|
+
"Rate limited.",
|
|
792
|
+
"Quota exceeded.",
|
|
793
|
+
"Model '",
|
|
794
|
+
"Error: ",
|
|
795
|
+
"*(Generation timed out)*",
|
|
796
|
+
"(Generation timed out)",
|
|
797
|
+
"Generation timed out",
|
|
798
|
+
"*(interrupted",
|
|
799
|
+
"Ollama Error:",
|
|
800
|
+
"Error: Ollama",
|
|
801
|
+
"Could not reach Ollama",
|
|
802
|
+
"The Ollama request timed out",
|
|
803
|
+
"Ollama streaming failure:",
|
|
804
|
+
)
|
|
805
|
+
# not a fixed prefix (provider name is interpolated) — matched separately
|
|
806
|
+
_ERROR_SUFFIX = "is not supported yet. Run /ai to reconfigure."
|
|
807
|
+
|
|
808
|
+
|
|
809
|
+
def is_error_response(text):
|
|
810
|
+
"""True if `text` is one of query_ai's own failure messages rather
|
|
811
|
+
than an actual model answer."""
|
|
812
|
+
if not text or not isinstance(text, str):
|
|
813
|
+
return False
|
|
814
|
+
t = text.strip()
|
|
815
|
+
if t.startswith(_ERROR_SIGNATURES) or t.endswith(_ERROR_SUFFIX):
|
|
816
|
+
return True
|
|
817
|
+
low = t.lower()
|
|
818
|
+
return any(h in low for h in _FALLOVER_HINTS)
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
# ---------------------------------------------------------------------------
|
|
822
|
+
# Backup-provider failover (v0.7.8 BONUS 1). query_ai()/stream_ai() first
|
|
823
|
+
# try the saved primary provider; when it reports quota exhaustion, a
|
|
824
|
+
# timeout, rate limiting, or is simply offline, they walk the enabled
|
|
825
|
+
# backup chain (providers/provider_manager.backup_configs(), priority
|
|
826
|
+
# order) and seamlessly finish the request against the next healthy one.
|
|
827
|
+
# The conversation never notices: history/summaries live in the UI and
|
|
828
|
+
# are re-sent verbatim to whichever provider answers.
|
|
829
|
+
#
|
|
830
|
+
# v0.7.9.5 REQUEST LIFECYCLE OVERHAUL (the '...' bug): the walk now has
|
|
831
|
+
# real engineering around it instead of one blind attempt per provider:
|
|
832
|
+
#
|
|
833
|
+
# * TOTAL + IDLE timeouts, configurable per prompt class (config.py:
|
|
834
|
+
# timeout_simple / timeout_normal / timeout_large /
|
|
835
|
+
# model_idle_timeout). A provider trickling bytes forever can no
|
|
836
|
+
# longer hold a turn open indefinitely, while an actively streaming
|
|
837
|
+
# model is never misclassified as frozen just because it's slow.
|
|
838
|
+
# * BOUNDED retries with EXPONENTIAL BACKOFF per provider
|
|
839
|
+
# (model_max_retries_per_provider, model_retry_backoff_base) — only
|
|
840
|
+
# for transient failures (timeouts / connection errors); quota and
|
|
841
|
+
# config errors skip straight to the next provider.
|
|
842
|
+
# * PROVIDER COOLDOWN (model_provider_cooldown): a provider that just
|
|
843
|
+
# failed sits out future requests briefly, so a dead primary stops
|
|
844
|
+
# taxing every turn; backups are tried first until it recovers.
|
|
845
|
+
# * CANCELLATION: every in-flight HTTP response is registered;
|
|
846
|
+
# cancel_active_requests() closes the sockets so Ctrl+C can stop a
|
|
847
|
+
# request that's blocked inside a socket read.
|
|
848
|
+
# * STRUCTURED REQUEST LOGGING (~/.cct_requests.log): one JSON line
|
|
849
|
+
# per lifecycle event (start/first_token/finish/error/retry/
|
|
850
|
+
# fallback) with timings and exception class — never API keys —
|
|
851
|
+
# so a stuck response can always be diagnosed after the fact.
|
|
852
|
+
|
|
853
|
+
_FALLOVER_HINTS = ("quota", "rate limit", "rate_limit", "insufficient_quota",
|
|
854
|
+
"limit exceeded", "exhausted", "429 ", "could not reach",
|
|
855
|
+
"connection", "api key", "unauthorized", "401", "403",
|
|
856
|
+
"model not found", "not found", "no model", "not running",
|
|
857
|
+
"timed out", "timeout", "timedout", "time out",
|
|
858
|
+
"generation timed out", "cannot connect", "failed to connect",
|
|
859
|
+
"could not reach ollama", "ensure ollama is running")
|
|
860
|
+
|
|
861
|
+
# A hook the UI can install (set_failover_hook) to surface failover
|
|
862
|
+
# moments as chat notes instead of silence.
|
|
863
|
+
_FAILOVER_HOOK = None
|
|
864
|
+
|
|
865
|
+
|
|
866
|
+
def _should_failover(text):
|
|
867
|
+
"""True when a returned text looks like a provider-level failure worth
|
|
868
|
+
switching providers for — CCT's own error strings, or the typical
|
|
869
|
+
HTTP-level rate-limit / quota wording providers embed in bodies."""
|
|
870
|
+
if not text or not isinstance(text, str):
|
|
871
|
+
return True
|
|
872
|
+
if not text.strip():
|
|
873
|
+
return True
|
|
874
|
+
if is_error_response(text):
|
|
875
|
+
return True
|
|
876
|
+
low = text.lower()
|
|
877
|
+
return any(h in low for h in _FALLOVER_HINTS)
|
|
878
|
+
|
|
879
|
+
|
|
880
|
+
def _retryable_failure(text):
|
|
881
|
+
"""True for TRANSIENT failures worth an immediate retry against the
|
|
882
|
+
SAME provider (network blip, momentary read timeout). Quota /
|
|
883
|
+
rate-limit / configuration failures are not retryable — retrying
|
|
884
|
+
them just burns seconds before the inevitable fallback."""
|
|
885
|
+
if not text:
|
|
886
|
+
return False
|
|
887
|
+
t = text.strip()
|
|
888
|
+
if t.startswith(("Could not reach the AI server.", "The AI request timed out.")):
|
|
889
|
+
return True
|
|
890
|
+
low = t.lower()
|
|
891
|
+
# Transient network/timeout issues are worth retrying
|
|
892
|
+
transient_hints = ("timed out", "timeout", "connection reset", "connection refused",
|
|
893
|
+
"connection error", "connection aborted", "broken pipe",
|
|
894
|
+
"eof occurred", "incomplete read", "remote end closed",
|
|
895
|
+
"server disconnected", "503", "502", "500")
|
|
896
|
+
# Non-retryable: quota, auth, config issues
|
|
897
|
+
non_retryable_hints = ("quota", "rate limit", "rate_limit", "401", "403",
|
|
898
|
+
"api key", "unauthorized", "not found", "404",
|
|
899
|
+
"not supported", "end of life", "deprecated")
|
|
900
|
+
if any(h in low for h in non_retryable_hints):
|
|
901
|
+
return False
|
|
902
|
+
return any(h in low for h in transient_hints)
|
|
903
|
+
|
|
904
|
+
|
|
905
|
+
def _notify_failover(message):
|
|
906
|
+
try:
|
|
907
|
+
if _FAILOVER_HOOK is not None:
|
|
908
|
+
_FAILOVER_HOOK(message)
|
|
909
|
+
except Exception:
|
|
910
|
+
pass
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
def set_failover_hook(callback):
|
|
914
|
+
"""Install a callable(message) hook invoked on every failover step
|
|
915
|
+
(exhausted primary, switching to backup N, connected). The Textual UI
|
|
916
|
+
uses this to post chat system notes; the classic REPL may leave it
|
|
917
|
+
None to stay silent. Pass None to clear."""
|
|
918
|
+
global _FAILOVER_HOOK
|
|
919
|
+
_FAILOVER_HOOK = callback
|
|
920
|
+
|
|
921
|
+
|
|
922
|
+
def _backup_chain():
|
|
923
|
+
"""Get the backup provider chain for failover.
|
|
924
|
+
|
|
925
|
+
v0.7.9.5: Improved Ollama handling for backup failover. Ollama
|
|
926
|
+
providers are prioritized for local inference when available,
|
|
927
|
+
providing a reliable fallback that works offline and has no
|
|
928
|
+
rate limits or quota issues.
|
|
929
|
+
|
|
930
|
+
v0.7.10: Enhanced model selection — always prefers instruct/chat
|
|
931
|
+
models over base models; broader keyword matching for quality
|
|
932
|
+
models; GPU/VRAM-aware model selection.
|
|
933
|
+
"""
|
|
934
|
+
try:
|
|
935
|
+
from .providers.provider_manager import backup_configs
|
|
936
|
+
chain = backup_configs()
|
|
937
|
+
|
|
938
|
+
# Base model identifiers (these models can't follow instructions)
|
|
939
|
+
_BASE_MODEL_HINTS = ("-base", "_base", "base-q", "base_q",
|
|
940
|
+
":base", "-base-", "base_model")
|
|
941
|
+
# Good instruct/chat model identifiers
|
|
942
|
+
_INSTRUCT_HINTS = ("instruct", "chat", "r1", "gemma", "qwen",
|
|
943
|
+
"llama-3", "phi-3", "phi-4", "mistral",
|
|
944
|
+
"codellama", "coder", "deepseek", "yi-",
|
|
945
|
+
"command", "mixtral", "wizard", "nous",
|
|
946
|
+
"solar", "neural", "orca", "zephyr",
|
|
947
|
+
"hermes", "dolphin", "tinyllama",
|
|
948
|
+
"starcoder", "codestral", "granite")
|
|
949
|
+
|
|
950
|
+
def _is_base_model(name):
|
|
951
|
+
low = name.lower()
|
|
952
|
+
return any(b in low for b in _BASE_MODEL_HINTS)
|
|
953
|
+
|
|
954
|
+
def _is_good_instruct(name):
|
|
955
|
+
low = name.lower()
|
|
956
|
+
if _is_base_model(name):
|
|
957
|
+
return False
|
|
958
|
+
return any(k in low for k in _INSTRUCT_HINTS)
|
|
959
|
+
|
|
960
|
+
enhanced_chain = []
|
|
961
|
+
for cfg, entry in chain:
|
|
962
|
+
if cfg.get("provider") == "ollama":
|
|
963
|
+
# Ensure Ollama has correct api_style and no key requirement
|
|
964
|
+
cfg["api_style"] = "ollama"
|
|
965
|
+
cfg["needs_key"] = False
|
|
966
|
+
base_url = cfg.get("base_url") or "http://localhost:11434"
|
|
967
|
+
cfg["base_url"] = base_url
|
|
968
|
+
# Auto-verify that the configured model is installed locally;
|
|
969
|
+
# If not or if it's a raw base model, fallback to a healthy instruct model!
|
|
970
|
+
try:
|
|
971
|
+
import requests
|
|
972
|
+
tag_r = requests.get(f"{base_url}/api/tags", timeout=2.0)
|
|
973
|
+
if tag_r.status_code == 200:
|
|
974
|
+
raw_models = [m.get("name") for m in tag_r.json().get("models", []) if m.get("name")]
|
|
975
|
+
# Split into instruct vs base
|
|
976
|
+
instruct_models = [m for m in raw_models if not _is_base_model(m)]
|
|
977
|
+
good_models = [m for m in instruct_models if _is_good_instruct(m)]
|
|
978
|
+
# Prefer good instruct > any non-base > all
|
|
979
|
+
candidates = good_models if good_models else (instruct_models if instruct_models else raw_models)
|
|
980
|
+
|
|
981
|
+
cur_m = cfg.get("model", "")
|
|
982
|
+
is_cur_base = _is_base_model(cur_m)
|
|
983
|
+
has_m = any(
|
|
984
|
+
cur_m == im or (":" not in cur_m and im.startswith(f"{cur_m}:"))
|
|
985
|
+
for im in candidates
|
|
986
|
+
)
|
|
987
|
+
if (not has_m or is_cur_base) and candidates:
|
|
988
|
+
# Score candidates: prefer r1 > instruct/chat > gemma > other
|
|
989
|
+
def _score(m):
|
|
990
|
+
ml = m.lower()
|
|
991
|
+
s = 0
|
|
992
|
+
if "r1" in ml: s += 100
|
|
993
|
+
if "instruct" in ml: s += 80
|
|
994
|
+
if "chat" in ml: s += 70
|
|
995
|
+
if "gemma" in ml: s += 60
|
|
996
|
+
if "qwen" in ml: s += 55
|
|
997
|
+
if "llama" in ml: s += 50
|
|
998
|
+
if "phi" in ml: s += 45
|
|
999
|
+
if "deepseek" in ml: s += 40
|
|
1000
|
+
if "mistral" in ml: s += 35
|
|
1001
|
+
return s
|
|
1002
|
+
best = max(candidates, key=_score)
|
|
1003
|
+
cfg["model"] = best
|
|
1004
|
+
_LOG.info("Backup chain: replaced base/missing model '%s' → '%s'", cur_m, best)
|
|
1005
|
+
except Exception:
|
|
1006
|
+
pass
|
|
1007
|
+
enhanced_chain.append((cfg, entry))
|
|
1008
|
+
|
|
1009
|
+
return enhanced_chain
|
|
1010
|
+
except Exception:
|
|
1011
|
+
return []
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
# ------------------------------------------------------- request tuning --
|
|
1015
|
+
def request_timeouts(config=None, size_class="normal"):
|
|
1016
|
+
"""(connect_timeout, idle_read_timeout, total_timeout) in seconds for
|
|
1017
|
+
one model attempt. `size_class` is one of 'simple' | 'normal' |
|
|
1018
|
+
'large' and maps to the configurable total deadlines in config.py.
|
|
1019
|
+
The idle cap never exceeds the total deadline."""
|
|
1020
|
+
try:
|
|
1021
|
+
if config is None:
|
|
1022
|
+
from . import config as _cfgmod
|
|
1023
|
+
cfg = _cfgmod.get_config()
|
|
1024
|
+
else:
|
|
1025
|
+
cfg = config
|
|
1026
|
+
# Normalize dict vs object access
|
|
1027
|
+
if isinstance(cfg, dict):
|
|
1028
|
+
get = lambda k, d: cfg.get(k, d)
|
|
1029
|
+
else:
|
|
1030
|
+
get = lambda k, d: getattr(cfg, k, d)
|
|
1031
|
+
totals = {
|
|
1032
|
+
"simple": get("timeout_simple", 30),
|
|
1033
|
+
"normal": get("timeout_normal", 60),
|
|
1034
|
+
"large": get("timeout_large", 300),
|
|
1035
|
+
}
|
|
1036
|
+
total = float(totals.get(str(size_class), get("timeout_normal", 60)))
|
|
1037
|
+
idle = float(get("model_idle_timeout", 45))
|
|
1038
|
+
# Ollama local needs longer for model load on cold start (especially 7B+ on CPU)
|
|
1039
|
+
prov = ""
|
|
1040
|
+
try:
|
|
1041
|
+
if isinstance(cfg, dict):
|
|
1042
|
+
prov = cfg.get("provider","") or cfg.get("default_ai_provider","")
|
|
1043
|
+
else:
|
|
1044
|
+
prov = getattr(cfg, "default_ai_provider", "") or getattr(cfg, "provider","")
|
|
1045
|
+
if not prov and isinstance(config, dict):
|
|
1046
|
+
prov = config.get("provider","")
|
|
1047
|
+
if str(prov).lower() == "ollama":
|
|
1048
|
+
total = max(total, 180.0)
|
|
1049
|
+
idle = max(idle, 120.0)
|
|
1050
|
+
except Exception:
|
|
1051
|
+
pass
|
|
1052
|
+
# Cloud providers connect timeout should be resilient against latency/proxies
|
|
1053
|
+
connect = max(5.0, min(15.0, idle))
|
|
1054
|
+
# Ollama connect needs longer on cold start (model load from disk into RAM/VRAM)
|
|
1055
|
+
try:
|
|
1056
|
+
if str(prov).lower() == "ollama":
|
|
1057
|
+
connect = max(connect, 35.0)
|
|
1058
|
+
except Exception:
|
|
1059
|
+
pass
|
|
1060
|
+
# Reasoning models (o1, o3, deepseek-r1, Claude thinking) take 30-90s before first token
|
|
1061
|
+
try:
|
|
1062
|
+
mod = ""
|
|
1063
|
+
if isinstance(cfg, dict):
|
|
1064
|
+
mod = cfg.get("model", "") or cfg.get("default_model", "")
|
|
1065
|
+
else:
|
|
1066
|
+
mod = getattr(cfg, "model", "") or getattr(cfg, "default_model", "")
|
|
1067
|
+
if not mod and isinstance(config, dict):
|
|
1068
|
+
mod = config.get("model", "")
|
|
1069
|
+
if any(k in str(mod).lower() for k in ("o1", "o3", "r1", "reason", "thinking", "3-7", "3.7")):
|
|
1070
|
+
idle = max(idle, 120.0)
|
|
1071
|
+
total = max(total, 180.0)
|
|
1072
|
+
except Exception:
|
|
1073
|
+
pass
|
|
1074
|
+
idle = max(5.0, min(idle, total))
|
|
1075
|
+
return connect, idle, max(5.0, total)
|
|
1076
|
+
except Exception:
|
|
1077
|
+
return 15.0, 60.0, 180.0
|
|
1078
|
+
|
|
1079
|
+
|
|
1080
|
+
_RETRY_STATE_LOCK = threading.Lock()
|
|
1081
|
+
_PROVIDER_LAST_FAILURE = {} # (provider, model) -> monotonic time
|
|
1082
|
+
|
|
1083
|
+
|
|
1084
|
+
def _provider_key(cfg):
|
|
1085
|
+
return (str((cfg or {}).get("provider") or "?"),
|
|
1086
|
+
str((cfg or {}).get("model") or "?"))
|
|
1087
|
+
|
|
1088
|
+
|
|
1089
|
+
def _mark_provider_failed(cfg):
|
|
1090
|
+
try:
|
|
1091
|
+
with _RETRY_STATE_LOCK:
|
|
1092
|
+
_PROVIDER_LAST_FAILURE[_provider_key(cfg)] = time.monotonic()
|
|
1093
|
+
except Exception:
|
|
1094
|
+
pass
|
|
1095
|
+
|
|
1096
|
+
|
|
1097
|
+
def _mark_provider_ok(cfg):
|
|
1098
|
+
try:
|
|
1099
|
+
with _RETRY_STATE_LOCK:
|
|
1100
|
+
_PROVIDER_LAST_FAILURE.pop(_provider_key(cfg), None)
|
|
1101
|
+
except Exception:
|
|
1102
|
+
pass
|
|
1103
|
+
|
|
1104
|
+
|
|
1105
|
+
def _provider_in_cooldown(cfg):
|
|
1106
|
+
try:
|
|
1107
|
+
from . import config as _cfgmod
|
|
1108
|
+
cooldown = float(getattr(_cfgmod.get_config(), "model_provider_cooldown", 20.0))
|
|
1109
|
+
except Exception:
|
|
1110
|
+
cooldown = 20.0
|
|
1111
|
+
with _RETRY_STATE_LOCK:
|
|
1112
|
+
last = _PROVIDER_LAST_FAILURE.get(_provider_key(cfg))
|
|
1113
|
+
if last is None:
|
|
1114
|
+
return False
|
|
1115
|
+
return (time.monotonic() - last) < cooldown
|
|
1116
|
+
|
|
1117
|
+
|
|
1118
|
+
def _backoff_sleep(attempt_index):
|
|
1119
|
+
"""Exponential backoff between same-provider retries: base *
|
|
1120
|
+
2**attempt, capped at 4 s. Short by design — instant hammering
|
|
1121
|
+
feels broken, multi-second stalls feel frozen."""
|
|
1122
|
+
try:
|
|
1123
|
+
from . import config as _cfgmod
|
|
1124
|
+
base = float(getattr(_cfgmod.get_config(), "model_retry_backoff_base", 0.6))
|
|
1125
|
+
except Exception:
|
|
1126
|
+
base = 0.6
|
|
1127
|
+
delay = min(4.0, base * (2 ** max(0, attempt_index)))
|
|
1128
|
+
try:
|
|
1129
|
+
time.sleep(delay)
|
|
1130
|
+
except Exception:
|
|
1131
|
+
pass
|
|
1132
|
+
|
|
1133
|
+
|
|
1134
|
+
def _failover_targets(primary_cfg=None, requirements=None):
|
|
1135
|
+
"""The ordered list of (config, entry_or_None, is_backup) attempts for
|
|
1136
|
+
one logical request: the resolved primary first, then the enabled
|
|
1137
|
+
backup chain.
|
|
1138
|
+
|
|
1139
|
+
Integrated with CAT's Backup Provider & Resilience System:
|
|
1140
|
+
- Tier 1: Fallback models on the same provider.
|
|
1141
|
+
- Tier 2: Priority-ordered dynamic backup pool.
|
|
1142
|
+
- Intelligent capability matching (vision, tools, coding).
|
|
1143
|
+
- Health monitoring and circuit breaker cooldowns.
|
|
1144
|
+
"""
|
|
1145
|
+
primary = dict(primary_cfg or load_config())
|
|
1146
|
+
prim_prov = str(primary.get("provider", "")).lower()
|
|
1147
|
+
if prim_prov and not (primary.get("api_key") or "").strip():
|
|
1148
|
+
try:
|
|
1149
|
+
from .providers.provider_manager import get_env_api_key
|
|
1150
|
+
env_k = get_env_api_key(prim_prov)
|
|
1151
|
+
if env_k:
|
|
1152
|
+
primary["api_key"] = env_k
|
|
1153
|
+
except Exception:
|
|
1154
|
+
pass
|
|
1155
|
+
|
|
1156
|
+
try:
|
|
1157
|
+
from .resilience.failover_engine import get_failover_engine
|
|
1158
|
+
from .resilience.types import TaskRequirements
|
|
1159
|
+
from .model_router import get_privacy_policy
|
|
1160
|
+
|
|
1161
|
+
engine = get_failover_engine()
|
|
1162
|
+
req = requirements if isinstance(requirements, TaskRequirements) else TaskRequirements()
|
|
1163
|
+
pol = get_privacy_policy()
|
|
1164
|
+
|
|
1165
|
+
candidates = engine.build_failover_targets(
|
|
1166
|
+
primary_config=primary,
|
|
1167
|
+
requirements=req,
|
|
1168
|
+
privacy_policy=pol,
|
|
1169
|
+
)
|
|
1170
|
+
if candidates:
|
|
1171
|
+
return [(dict(c.config), c.entry, c.is_backup) for c in candidates]
|
|
1172
|
+
except Exception:
|
|
1173
|
+
pass
|
|
1174
|
+
|
|
1175
|
+
# Fallback to direct walk if resilience engine unavailable
|
|
1176
|
+
chain = [(primary, None, False)]
|
|
1177
|
+
prim_url = str(primary.get("base_url", "")).rstrip("/")
|
|
1178
|
+
prim_model = str(primary.get("model", "")).lower()
|
|
1179
|
+
for bcfg, entry in _backup_chain():
|
|
1180
|
+
try:
|
|
1181
|
+
b_prov = str(bcfg.get("provider", "")).lower()
|
|
1182
|
+
b_url = str(bcfg.get("base_url", "")).rstrip("/")
|
|
1183
|
+
b_model = str(bcfg.get("model", "")).lower()
|
|
1184
|
+
if b_prov == prim_prov and b_url == prim_url and b_model == prim_model:
|
|
1185
|
+
continue
|
|
1186
|
+
needs_key = bcfg.get("needs_key", True)
|
|
1187
|
+
if b_prov == "ollama":
|
|
1188
|
+
needs_key = False
|
|
1189
|
+
if needs_key and not (bcfg.get("api_key") or "").strip():
|
|
1190
|
+
try:
|
|
1191
|
+
from .providers.provider_manager import get_env_api_key
|
|
1192
|
+
bk_key = get_env_api_key(b_prov)
|
|
1193
|
+
if bk_key:
|
|
1194
|
+
bcfg["api_key"] = bk_key
|
|
1195
|
+
except Exception:
|
|
1196
|
+
pass
|
|
1197
|
+
if needs_key and not (bcfg.get("api_key") or "").strip():
|
|
1198
|
+
continue
|
|
1199
|
+
except Exception:
|
|
1200
|
+
pass
|
|
1201
|
+
chain.append((dict(bcfg), entry, True))
|
|
1202
|
+
|
|
1203
|
+
fresh = [c for c in chain if not c[2] or not _provider_in_cooldown(c[0])]
|
|
1204
|
+
cooled = [c for c in chain if c not in fresh]
|
|
1205
|
+
return fresh + cooled
|
|
1206
|
+
|
|
1207
|
+
|
|
1208
|
+
def _attempts_per_provider():
|
|
1209
|
+
try:
|
|
1210
|
+
from . import config as _cfgmod
|
|
1211
|
+
n = int(getattr(_cfgmod.get_config(), "model_max_retries_per_provider", 1))
|
|
1212
|
+
except Exception:
|
|
1213
|
+
n = 1
|
|
1214
|
+
return max(1, n) + 1 # configured retries PLUS the initial attempt
|
|
1215
|
+
|
|
1216
|
+
|
|
1217
|
+
# ----------------------------------------------------- cancellation kit --
|
|
1218
|
+
_ACTIVE_LOCK = threading.Lock()
|
|
1219
|
+
_ACTIVE_RESPONSES = set()
|
|
1220
|
+
_ACTIVE_PROVIDERS = set()
|
|
1221
|
+
_CANCELLED_AT = 0.0 # wall-clock stamp of the last cancel_active_requests()
|
|
1222
|
+
|
|
1223
|
+
|
|
1224
|
+
def _register_response(resp):
|
|
1225
|
+
try:
|
|
1226
|
+
with _ACTIVE_LOCK:
|
|
1227
|
+
_ACTIVE_RESPONSES.add(resp)
|
|
1228
|
+
except Exception:
|
|
1229
|
+
pass
|
|
1230
|
+
|
|
1231
|
+
|
|
1232
|
+
def _unregister_response(resp):
|
|
1233
|
+
try:
|
|
1234
|
+
with _ACTIVE_LOCK:
|
|
1235
|
+
_ACTIVE_RESPONSES.discard(resp)
|
|
1236
|
+
except Exception:
|
|
1237
|
+
pass
|
|
1238
|
+
|
|
1239
|
+
|
|
1240
|
+
def _register_provider(prov):
|
|
1241
|
+
try:
|
|
1242
|
+
with _ACTIVE_LOCK:
|
|
1243
|
+
_ACTIVE_PROVIDERS.add(prov)
|
|
1244
|
+
except Exception:
|
|
1245
|
+
pass
|
|
1246
|
+
|
|
1247
|
+
|
|
1248
|
+
def _unregister_provider(prov):
|
|
1249
|
+
try:
|
|
1250
|
+
with _ACTIVE_LOCK:
|
|
1251
|
+
_ACTIVE_PROVIDERS.discard(prov)
|
|
1252
|
+
except Exception:
|
|
1253
|
+
pass
|
|
1254
|
+
|
|
1255
|
+
|
|
1256
|
+
def cancel_active_requests():
|
|
1257
|
+
"""Close every in-flight provider socket so a blocked iter_lines()
|
|
1258
|
+
read raises immediately instead of waiting out its idle timeout.
|
|
1259
|
+
Called by the UI when the user hits Ctrl+C / Stop. Safe to call
|
|
1260
|
+
when nothing is running."""
|
|
1261
|
+
global _CANCELLED_AT
|
|
1262
|
+
_CANCELLED_AT = time.time()
|
|
1263
|
+
with _ACTIVE_LOCK:
|
|
1264
|
+
current = list(_ACTIVE_RESPONSES)
|
|
1265
|
+
providers = list(_ACTIVE_PROVIDERS)
|
|
1266
|
+
for resp in current:
|
|
1267
|
+
try:
|
|
1268
|
+
resp.close()
|
|
1269
|
+
except Exception:
|
|
1270
|
+
pass
|
|
1271
|
+
for prov in providers:
|
|
1272
|
+
try:
|
|
1273
|
+
if hasattr(prov, "cancel_active"):
|
|
1274
|
+
prov.cancel_active()
|
|
1275
|
+
except Exception:
|
|
1276
|
+
pass
|
|
1277
|
+
return len(current) + len(providers)
|
|
1278
|
+
|
|
1279
|
+
|
|
1280
|
+
def _just_cancelled():
|
|
1281
|
+
"""True within a short window after cancel_active_requests() — used
|
|
1282
|
+
to convert the close-induced socket exception into a clean stop
|
|
1283
|
+
rather than a scary error message."""
|
|
1284
|
+
return (time.time() - _CANCELLED_AT) < 3.0
|
|
1285
|
+
|
|
1286
|
+
|
|
1287
|
+
# ------------------------------------------------- structured req log --
|
|
1288
|
+
_REQUEST_LOG = os.path.join(os.path.expanduser("~"), ".cct_requests.log")
|
|
1289
|
+
_REQUEST_LOG_MAX = 512 * 1024
|
|
1290
|
+
_REQLOG_LOCK = threading.Lock()
|
|
1291
|
+
|
|
1292
|
+
|
|
1293
|
+
def log_request_event(request_id, event, provider=None, model=None,
|
|
1294
|
+
size_class=None, elapsed_ms=None, detail=None,
|
|
1295
|
+
chars=None, retry_count=None, fallback=None):
|
|
1296
|
+
"""One JSON line per lifecycle event into ~/.cct_requests.log
|
|
1297
|
+
(rotated at ~512 KB). Records WHERE a request stopped and why —
|
|
1298
|
+
the diagnosis tool the permanent-'...' bug always needed. Never
|
|
1299
|
+
logs API keys, headers or prompt contents."""
|
|
1300
|
+
record = {"ts": round(time.time(), 3), "request_id": request_id,
|
|
1301
|
+
"event": event}
|
|
1302
|
+
if provider:
|
|
1303
|
+
record["provider"] = provider
|
|
1304
|
+
if model:
|
|
1305
|
+
record["model"] = model
|
|
1306
|
+
if size_class:
|
|
1307
|
+
record["size_class"] = size_class
|
|
1308
|
+
if elapsed_ms is not None:
|
|
1309
|
+
record["elapsed_ms"] = int(elapsed_ms)
|
|
1310
|
+
if chars is not None:
|
|
1311
|
+
record["chars"] = chars
|
|
1312
|
+
if retry_count is not None:
|
|
1313
|
+
record["retry_count"] = retry_count
|
|
1314
|
+
if fallback is not None:
|
|
1315
|
+
record["fallback"] = bool(fallback)
|
|
1316
|
+
if detail:
|
|
1317
|
+
record["detail"] = str(detail)[:200]
|
|
1318
|
+
line = json.dumps(record, ensure_ascii=True, default=str)
|
|
1319
|
+
try:
|
|
1320
|
+
with _REQLOG_LOCK:
|
|
1321
|
+
try:
|
|
1322
|
+
if os.path.exists(_REQUEST_LOG) and \
|
|
1323
|
+
os.path.getsize(_REQUEST_LOG) > _REQUEST_LOG_MAX:
|
|
1324
|
+
os.replace(_REQUEST_LOG, _REQUEST_LOG + ".old")
|
|
1325
|
+
except Exception:
|
|
1326
|
+
pass
|
|
1327
|
+
with open(_REQUEST_LOG, "a", encoding="utf-8") as f:
|
|
1328
|
+
f.write(line + "\n")
|
|
1329
|
+
except Exception:
|
|
1330
|
+
pass
|
|
1331
|
+
try:
|
|
1332
|
+
_LOG.debug("aicore.request %s", line)
|
|
1333
|
+
except Exception:
|
|
1334
|
+
pass
|
|
1335
|
+
|
|
1336
|
+
|
|
1337
|
+
def _query_ai_once(prompt, system_prompt=DEFAULT_SYSTEM_PROMPT,
|
|
1338
|
+
history=None, config=None, attachments=None,
|
|
1339
|
+
size_class="normal"):
|
|
1340
|
+
"""Single-attempt blocking chat completion against ONE provider
|
|
1341
|
+
config. Implemented as a consumer of _stream_ai_once so every
|
|
1342
|
+
api_style branch lives in exactly one place — the concatenation of
|
|
1343
|
+
a streamed reply is byte-identical to the blocking reply, and both
|
|
1344
|
+
surface provider failures as the same error strings, which keeps
|
|
1345
|
+
the failover detection in query_ai consistent."""
|
|
1346
|
+
pieces = []
|
|
1347
|
+
_metrics_note_model_start(config)
|
|
1348
|
+
for piece in _stream_ai_once(prompt, system_prompt=system_prompt,
|
|
1349
|
+
history=history, config=config,
|
|
1350
|
+
attachments=attachments,
|
|
1351
|
+
size_class=size_class):
|
|
1352
|
+
pieces.append(piece)
|
|
1353
|
+
return "".join(pieces)
|
|
1354
|
+
|
|
1355
|
+
|
|
1356
|
+
def query_ai(prompt, system_prompt=DEFAULT_SYSTEM_PROMPT,
|
|
1357
|
+
history=None, config=None, on_failover=None, attachments=None,
|
|
1358
|
+
size_class="normal", requirements=None):
|
|
1359
|
+
"""`history` is an optional list of (role, text) pairs — or {"role","text"}
|
|
1360
|
+
dicts — for every prior turn that should stay in context, oldest first.
|
|
1361
|
+
Every branch below sends it to the provider in whatever shape that
|
|
1362
|
+
provider's API expects; omit it (or pass None/[]) for a genuinely
|
|
1363
|
+
one-shot call. See ChatSession.as_prompt_history() for the usual source.
|
|
1364
|
+
|
|
1365
|
+
v0.7.8: on primary-provider failure (quota/timeout/rate-limit/offline)
|
|
1366
|
+
transparently falls back to the enabled backup chain, notifying the
|
|
1367
|
+
installed failover hook at each step. `on_failover` overrides the
|
|
1368
|
+
global hook for this one call when given.
|
|
1369
|
+
|
|
1370
|
+
v0.7.8.1: `attachments` is an optional list of Attachment objects
|
|
1371
|
+
(calc_terminal/attachments.py). They are verified before the request
|
|
1372
|
+
and sent natively (images) or as a text context block, per provider
|
|
1373
|
+
capability.
|
|
1374
|
+
|
|
1375
|
+
v0.7.9.5 request lifecycle: `size_class` ('simple' | 'normal' |
|
|
1376
|
+
'large') picks the configurable total deadline; each provider gets
|
|
1377
|
+
bounded retries with exponential backoff for TRANSIENT failures;
|
|
1378
|
+
quota/config failures move on immediately; recently-failed providers
|
|
1379
|
+
sit out via cooldown until the chain exhausts.
|
|
1380
|
+
|
|
1381
|
+
Returns the assistant text, or (on total failure) ONE of the
|
|
1382
|
+
_ERROR_SIGNATURES strings — loading states can always key off those,
|
|
1383
|
+
and callers can always tell a real answer from a failure."""
|
|
1384
|
+
hook_override = on_failover or _FAILOVER_HOOK
|
|
1385
|
+
|
|
1386
|
+
def notify(message):
|
|
1387
|
+
try:
|
|
1388
|
+
if hook_override is not None:
|
|
1389
|
+
hook_override(message)
|
|
1390
|
+
except Exception:
|
|
1391
|
+
pass
|
|
1392
|
+
|
|
1393
|
+
targets = _failover_targets(config, requirements=requirements)
|
|
1394
|
+
max_attempts = _attempts_per_provider()
|
|
1395
|
+
result = ""
|
|
1396
|
+
fallback_used = False
|
|
1397
|
+
|
|
1398
|
+
for t_idx, (cfg, entry, is_backup) in enumerate(targets):
|
|
1399
|
+
provider = cfg.get("provider", "?")
|
|
1400
|
+
model = cfg.get("model", "")
|
|
1401
|
+
try:
|
|
1402
|
+
from .resilience.health_monitor import get_health_monitor
|
|
1403
|
+
get_health_monitor().record_turn_start(provider, model)
|
|
1404
|
+
except Exception:
|
|
1405
|
+
pass
|
|
1406
|
+
t_turn_start = time.time()
|
|
1407
|
+
for attempt in range(max_attempts):
|
|
1408
|
+
if attempt:
|
|
1409
|
+
log_request_event(_short_id(), "retry_same_provider",
|
|
1410
|
+
provider=provider,
|
|
1411
|
+
model=cfg.get("model"),
|
|
1412
|
+
size_class=size_class,
|
|
1413
|
+
retry_count=attempt)
|
|
1414
|
+
result = _query_ai_once(prompt, system_prompt=system_prompt,
|
|
1415
|
+
history=history, config=cfg,
|
|
1416
|
+
attachments=attachments,
|
|
1417
|
+
size_class=size_class)
|
|
1418
|
+
if not _should_failover(result):
|
|
1419
|
+
if is_backup:
|
|
1420
|
+
_mark_backup_success(entry)
|
|
1421
|
+
notify("\u2713 Connected successfully.")
|
|
1422
|
+
_mark_provider_ok(cfg)
|
|
1423
|
+
try:
|
|
1424
|
+
from .resilience.health_monitor import get_health_monitor
|
|
1425
|
+
lat = (time.time() - t_turn_start) * 1000.0
|
|
1426
|
+
get_health_monitor().record_success(provider, model, latency_ms=lat)
|
|
1427
|
+
except Exception:
|
|
1428
|
+
pass
|
|
1429
|
+
log_request_event(_short_id(), "finish",
|
|
1430
|
+
provider=provider, model=cfg.get("model"),
|
|
1431
|
+
size_class=size_class, chars=len(result),
|
|
1432
|
+
fallback=fallback_used)
|
|
1433
|
+
return result
|
|
1434
|
+
# Failure. Transient? → brief backoff, retry same provider.
|
|
1435
|
+
if attempt + 1 < max_attempts and _retryable_failure(result):
|
|
1436
|
+
log_request_event(_short_id(), "attempt_failed_retryable",
|
|
1437
|
+
provider=provider, detail=result[:120],
|
|
1438
|
+
size_class=size_class)
|
|
1439
|
+
_backoff_sleep(attempt)
|
|
1440
|
+
continue
|
|
1441
|
+
break # non-retryable → next provider
|
|
1442
|
+
_mark_provider_failed(cfg)
|
|
1443
|
+
try:
|
|
1444
|
+
from .resilience.health_monitor import get_health_monitor
|
|
1445
|
+
from .resilience.failover_engine import get_failover_engine
|
|
1446
|
+
from .resilience.types import FailureType
|
|
1447
|
+
ft = get_failover_engine().classify_failure(result)
|
|
1448
|
+
get_health_monitor().record_failure(
|
|
1449
|
+
provider, model, error_message=result,
|
|
1450
|
+
is_rate_limit=(ft == FailureType.RATE_LIMIT)
|
|
1451
|
+
)
|
|
1452
|
+
except Exception:
|
|
1453
|
+
pass
|
|
1454
|
+
log_request_event(_short_id(), "provider_exhausted",
|
|
1455
|
+
provider=provider, detail=result[:160],
|
|
1456
|
+
size_class=size_class, fallback=True)
|
|
1457
|
+
if t_idx + 1 < len(targets):
|
|
1458
|
+
fallback_used = True
|
|
1459
|
+
nxt_cfg = targets[t_idx + 1][0]
|
|
1460
|
+
nxt = nxt_cfg.get("provider", "?")
|
|
1461
|
+
nxt_m = nxt_cfg.get("model", "")
|
|
1462
|
+
cur_m = cfg.get("model", "")
|
|
1463
|
+
cur_desc = f"{provider} ({cur_m})" if cur_m else provider
|
|
1464
|
+
nxt_desc = f"{nxt} ({nxt_m})" if nxt_m else nxt
|
|
1465
|
+
notify(f"\u26a0 Provider issue with {cur_desc}. Switching to Backup "
|
|
1466
|
+
f"Provider ({nxt_desc})...")
|
|
1467
|
+
|
|
1468
|
+
# Every provider failed — surface the last failure message so the UI
|
|
1469
|
+
# shows a real error card instead of waiting forever.
|
|
1470
|
+
log_request_event(_short_id(), "all_providers_failed",
|
|
1471
|
+
detail=result[:200], size_class=size_class)
|
|
1472
|
+
return result
|
|
1473
|
+
|
|
1474
|
+
|
|
1475
|
+
def _short_id():
|
|
1476
|
+
"""Short unique id for log correlation."""
|
|
1477
|
+
return uuid.uuid4().hex[:12]
|
|
1478
|
+
|
|
1479
|
+
|
|
1480
|
+
def _mark_backup_success(entry):
|
|
1481
|
+
try:
|
|
1482
|
+
from .providers.provider_manager import mark_backup_used
|
|
1483
|
+
mark_backup_used(entry)
|
|
1484
|
+
except Exception:
|
|
1485
|
+
pass
|
|
1486
|
+
|
|
1487
|
+
|
|
1488
|
+
def _peek_first(generator):
|
|
1489
|
+
"""Pull the first item from a generator, returning (item, generator)
|
|
1490
|
+
so callers can inspect it (failover decision) and still stream the
|
|
1491
|
+
rest exactly as the underlying generator produced it.
|
|
1492
|
+
|
|
1493
|
+
The returned generator is the ORIGINAL generator, already advanced
|
|
1494
|
+
past the first item — it must NOT include `first` again, because
|
|
1495
|
+
every caller here does `yield first; yield from rest` and re-including
|
|
1496
|
+
it would emit the first chunk twice ("hellohello"). The previous
|
|
1497
|
+
`chain([first], generator)` implementation caused exactly that: every
|
|
1498
|
+
streamed reply started with its first fragment duplicated."""
|
|
1499
|
+
try:
|
|
1500
|
+
first = next(generator)
|
|
1501
|
+
except StopIteration:
|
|
1502
|
+
return None, iter(())
|
|
1503
|
+
return first, generator
|
|
1504
|
+
|
|
1505
|
+
|
|
1506
|
+
def stream_ai(prompt, system_prompt=DEFAULT_SYSTEM_PROMPT,
|
|
1507
|
+
history=None, config=None, on_failover=None, attachments=None,
|
|
1508
|
+
size_class="normal", requirements=None):
|
|
1509
|
+
"""Generator version of query_ai — yields text fragments as they
|
|
1510
|
+
arrive instead of returning one finished string, so the primary UI
|
|
1511
|
+
(calc_terminal/ui/) can grow a ConversationItem token-by-token
|
|
1512
|
+
instead of freezing until the whole reply lands.
|
|
1513
|
+
|
|
1514
|
+
v0.7.8 failover: if the first fragment from the primary provider is a
|
|
1515
|
+
provider-failure signature, the whole stream is transparently retried
|
|
1516
|
+
against the next healthy backup instead of showing the error.
|
|
1517
|
+
|
|
1518
|
+
v0.7.9.5 request lifecycle hardening:
|
|
1519
|
+
* NOTHING is yielded until real content arrives, so a failing
|
|
1520
|
+
provider can never leave a permanent '...' on screen — the walk
|
|
1521
|
+
moves to backups first.
|
|
1522
|
+
* A failure BEFORE any token → bounded same-provider retries with
|
|
1523
|
+
exponential backoff, then the next provider (cooldown-aware).
|
|
1524
|
+
* A failure MID-STREAM (tokens already delivered) does NOT
|
|
1525
|
+
silently switch providers and re-run the whole answer — the
|
|
1526
|
+
partial reply is kept and the loss is disclosed in one honest
|
|
1527
|
+
line.
|
|
1528
|
+
* Every lifecycle step is written to ~/.cct_requests.log.
|
|
1529
|
+
|
|
1530
|
+
v0.7.8.1: `attachments` — see query_ai; verified before each attempt,
|
|
1531
|
+
attached natively (vision providers) or as text context."""
|
|
1532
|
+
hook_override = on_failover or _FAILOVER_HOOK
|
|
1533
|
+
|
|
1534
|
+
def notify(message):
|
|
1535
|
+
try:
|
|
1536
|
+
if hook_override is not None:
|
|
1537
|
+
hook_override(message)
|
|
1538
|
+
except Exception:
|
|
1539
|
+
pass
|
|
1540
|
+
|
|
1541
|
+
targets = _failover_targets(config, requirements=requirements)
|
|
1542
|
+
max_attempts = _attempts_per_provider()
|
|
1543
|
+
fallback_used = False
|
|
1544
|
+
last_error_piece = None
|
|
1545
|
+
|
|
1546
|
+
for t_idx, (cfg, entry, is_backup) in enumerate(targets):
|
|
1547
|
+
provider = cfg.get("provider", "?")
|
|
1548
|
+
for attempt in range(max_attempts):
|
|
1549
|
+
_metrics_note_model_start(cfg)
|
|
1550
|
+
got_content = False
|
|
1551
|
+
error_piece = None
|
|
1552
|
+
for piece in _stream_ai_once(prompt, system_prompt=system_prompt,
|
|
1553
|
+
history=history, config=cfg,
|
|
1554
|
+
attachments=attachments,
|
|
1555
|
+
size_class=size_class):
|
|
1556
|
+
if not got_content:
|
|
1557
|
+
if is_error_response(piece):
|
|
1558
|
+
error_piece = piece
|
|
1559
|
+
break
|
|
1560
|
+
else:
|
|
1561
|
+
# Once streaming has begun, individual fragments (including whitespace/newlines)
|
|
1562
|
+
# are NOT provider failures. Only known error signatures break the stream.
|
|
1563
|
+
t = piece.strip() if isinstance(piece, str) else ""
|
|
1564
|
+
if t and (t.startswith(_ERROR_SIGNATURES) or t.endswith(_ERROR_SUFFIX)):
|
|
1565
|
+
error_piece = piece
|
|
1566
|
+
break
|
|
1567
|
+
# Real content: stream it out immediately (requirement:
|
|
1568
|
+
# first token replaces '...' right away).
|
|
1569
|
+
got_content = True
|
|
1570
|
+
yield piece
|
|
1571
|
+
if error_piece is None:
|
|
1572
|
+
if got_content:
|
|
1573
|
+
if is_backup:
|
|
1574
|
+
_mark_backup_success(entry)
|
|
1575
|
+
notify("\u2713 Connected successfully.")
|
|
1576
|
+
_mark_provider_ok(cfg)
|
|
1577
|
+
log_request_event(_short_id(), "finish",
|
|
1578
|
+
provider=provider, model=cfg.get("model"),
|
|
1579
|
+
size_class=size_class, fallback=fallback_used)
|
|
1580
|
+
return
|
|
1581
|
+
# Generator ended with zero tokens and no signature —
|
|
1582
|
+
# treat as a provider-level empty response.
|
|
1583
|
+
error_piece = "No model response received."
|
|
1584
|
+
if got_content:
|
|
1585
|
+
# Mid-stream loss: keep the partial answer, disclose the
|
|
1586
|
+
# drop, do NOT duplicate the reply from another provider.
|
|
1587
|
+
_mark_provider_failed(cfg)
|
|
1588
|
+
log_request_event(_short_id(), "midstream_drop",
|
|
1589
|
+
provider=provider, detail=error_piece[:120],
|
|
1590
|
+
size_class=size_class)
|
|
1591
|
+
yield ("\n\n\u26a0 Connection lost mid-response \u2014 partial "
|
|
1592
|
+
"answer kept. Send again to continue.")
|
|
1593
|
+
return
|
|
1594
|
+
last_error_piece = error_piece
|
|
1595
|
+
if attempt + 1 < max_attempts and _retryable_failure(error_piece):
|
|
1596
|
+
log_request_event(_short_id(), "attempt_failed_retryable",
|
|
1597
|
+
provider=provider, detail=error_piece[:120],
|
|
1598
|
+
size_class=size_class)
|
|
1599
|
+
_backoff_sleep(attempt)
|
|
1600
|
+
continue
|
|
1601
|
+
break # non-retryable → next provider
|
|
1602
|
+
_mark_provider_failed(cfg)
|
|
1603
|
+
log_request_event(_short_id(), "provider_exhausted",
|
|
1604
|
+
provider=provider, detail=(error_piece or "")[:160],
|
|
1605
|
+
size_class=size_class, fallback=True)
|
|
1606
|
+
if t_idx + 1 < len(targets):
|
|
1607
|
+
fallback_used = True
|
|
1608
|
+
nxt_cfg = targets[t_idx + 1][0]
|
|
1609
|
+
nxt = nxt_cfg.get("provider", "?")
|
|
1610
|
+
nxt_m = nxt_cfg.get("model", "")
|
|
1611
|
+
cur_m = cfg.get("model", "")
|
|
1612
|
+
cur_desc = f"{provider} ({cur_m})" if cur_m else provider
|
|
1613
|
+
nxt_desc = f"{nxt} ({nxt_m})" if nxt_m else nxt
|
|
1614
|
+
notify(f"\u26a0 Provider issue with {cur_desc}. Switching to Backup "
|
|
1615
|
+
f"Provider ({nxt_desc})...")
|
|
1616
|
+
|
|
1617
|
+
# Nothing received anywhere — surface the final failure message so
|
|
1618
|
+
# the UI renders an error card instead of an eternal spinner.
|
|
1619
|
+
yield last_error_piece or "No model response received."
|
|
1620
|
+
|
|
1621
|
+
|
|
1622
|
+
def _prepare_attachments(attachments, config):
|
|
1623
|
+
"""v0.7.8.1: verify attachment objects (spec section 7) and decide
|
|
1624
|
+
whether this provider receives them natively (vision-capable) or as
|
|
1625
|
+
text context. Returns the `vision` flag consumed by the request
|
|
1626
|
+
builders. Emits ProviderAdapter verification lines to the debug log
|
|
1627
|
+
— never file contents."""
|
|
1628
|
+
if not attachments:
|
|
1629
|
+
return False
|
|
1630
|
+
try:
|
|
1631
|
+
from . import attachments as _att
|
|
1632
|
+
except Exception:
|
|
1633
|
+
return False
|
|
1634
|
+
_att.AttachmentManager.verify(
|
|
1635
|
+
attachments, log=lambda m: _LOG.debug("ProviderAdapter: %s", m))
|
|
1636
|
+
vision = False
|
|
1637
|
+
try:
|
|
1638
|
+
vision = bool(_att.provider_capabilities(config).get("vision"))
|
|
1639
|
+
except Exception:
|
|
1640
|
+
vision = False
|
|
1641
|
+
has_image = any(
|
|
1642
|
+
_att.attachment_has_image_payload(a) for a in attachments if a is not None)
|
|
1643
|
+
if vision and has_image:
|
|
1644
|
+
_LOG.debug("ProviderAdapter: native image payload attached")
|
|
1645
|
+
else:
|
|
1646
|
+
_LOG.debug("ProviderAdapter: attachment_context = present (textual block)")
|
|
1647
|
+
return vision
|
|
1648
|
+
|
|
1649
|
+
|
|
1650
|
+
def _metrics_note_model_start(config=None):
|
|
1651
|
+
"""Best-effort metrics/event hooks for one model request (never raises,
|
|
1652
|
+
never slows the path down meaningfully)."""
|
|
1653
|
+
try:
|
|
1654
|
+
from . import metrics as _m
|
|
1655
|
+
m = _m.current()
|
|
1656
|
+
if m is not None:
|
|
1657
|
+
m.count_model_call()
|
|
1658
|
+
m.stage_start("model_first_token")
|
|
1659
|
+
if config and not m.model_used:
|
|
1660
|
+
m.model_used = f"{config.get('provider', '?')}/{config.get('model', '?')}"
|
|
1661
|
+
except Exception:
|
|
1662
|
+
pass
|
|
1663
|
+
try:
|
|
1664
|
+
from .event_stream import stream, MODEL_REQUEST_STARTED
|
|
1665
|
+
stream.emit(MODEL_REQUEST_STARTED, source="aicore",
|
|
1666
|
+
provider=(config or {}).get("provider"),
|
|
1667
|
+
model=(config or {}).get("model"))
|
|
1668
|
+
except Exception:
|
|
1669
|
+
pass
|
|
1670
|
+
|
|
1671
|
+
|
|
1672
|
+
def _metrics_note_first_token():
|
|
1673
|
+
"""Called on the first REAL streamed token of a completion."""
|
|
1674
|
+
try:
|
|
1675
|
+
from . import metrics as _m
|
|
1676
|
+
m = _m.current()
|
|
1677
|
+
if m is not None:
|
|
1678
|
+
m.note_first_token()
|
|
1679
|
+
m.stage_end("model_first_token")
|
|
1680
|
+
except Exception:
|
|
1681
|
+
pass
|
|
1682
|
+
try:
|
|
1683
|
+
from .event_stream import stream, MODEL_FIRST_TOKEN
|
|
1684
|
+
stream.emit(MODEL_FIRST_TOKEN, source="aicore")
|
|
1685
|
+
except Exception:
|
|
1686
|
+
pass
|
|
1687
|
+
|
|
1688
|
+
|
|
1689
|
+
class _TotalTimeout(Exception):
|
|
1690
|
+
"""Raised between stream chunks when the TOTAL request deadline
|
|
1691
|
+
expires (distinct from requests' idle read timeout)."""
|
|
1692
|
+
|
|
1693
|
+
|
|
1694
|
+
def _stream_ai_once(prompt, system_prompt=DEFAULT_SYSTEM_PROMPT,
|
|
1695
|
+
history=None, config=None, attachments=None,
|
|
1696
|
+
size_class="normal", request_id=None):
|
|
1697
|
+
"""Single-attempt streaming generator against ONE provider config —
|
|
1698
|
+
the core of the public stream_ai; see its docstring for streaming
|
|
1699
|
+
behavior semantics. `config` defaults to the saved primary config;
|
|
1700
|
+
the v0.7.8 failover wrapper retries this against each backup.
|
|
1701
|
+
|
|
1702
|
+
v0.7.8.1: `attachments` (Attachment objects) are verified here
|
|
1703
|
+
(spec section 7), then folded in by the request builder — natively
|
|
1704
|
+
as image parts when the provider is vision-capable, otherwise as
|
|
1705
|
+
text via the attachment manager's context block.
|
|
1706
|
+
|
|
1707
|
+
v0.7.9.5 lifecycle:
|
|
1708
|
+
* `size_class` selects the configurable total timeout.
|
|
1709
|
+
* Every HTTP call uses a (connect, idle) timeout tuple; the total
|
|
1710
|
+
deadline is enforced BETWEEN chunks via _TotalTimeout, so an
|
|
1711
|
+
actively streaming model is never cut off mid-tokens but a
|
|
1712
|
+
silent one can't hold the UI forever either.
|
|
1713
|
+
* The response socket is registered for cancel_active_requests().
|
|
1714
|
+
* start / first_token / finish / error are logged structurally."""
|
|
1715
|
+
request_id = request_id or _short_id()
|
|
1716
|
+
t_start = time.monotonic()
|
|
1717
|
+
|
|
1718
|
+
def elapsed_ms():
|
|
1719
|
+
return int((time.monotonic() - t_start) * 1000)
|
|
1720
|
+
|
|
1721
|
+
if not _HAS_REQUESTS:
|
|
1722
|
+
yield "The 'requests' library is required for AI features. Please run: pip install requests"
|
|
1723
|
+
return
|
|
1724
|
+
|
|
1725
|
+
config = config or load_config()
|
|
1726
|
+
provider = config.get("provider")
|
|
1727
|
+
if not provider:
|
|
1728
|
+
yield "AI not configured. Run /ai or /agent to configure your provider."
|
|
1729
|
+
return
|
|
1730
|
+
|
|
1731
|
+
api_style, base_url, api_key, model, temperature, extra_headers = _resolve_provider(config)
|
|
1732
|
+
history = _sanitize_history(history)
|
|
1733
|
+
usage_prompt_text = prompt if not history else "\n".join(t for _, t in history) + "\n" + prompt
|
|
1734
|
+
|
|
1735
|
+
vision = _prepare_attachments(attachments, config)
|
|
1736
|
+
|
|
1737
|
+
# ---- lifecycle instrumentation (v0.7.9.5) --------------------------
|
|
1738
|
+
t_connect_idle = request_timeouts(config, size_class)[:2]
|
|
1739
|
+
deadline = time.monotonic() + request_timeouts(config, size_class)[2]
|
|
1740
|
+
log_request_event(request_id, "start", provider=provider, model=model,
|
|
1741
|
+
size_class=size_class)
|
|
1742
|
+
|
|
1743
|
+
def _check_deadline():
|
|
1744
|
+
if time.monotonic() > deadline:
|
|
1745
|
+
raise _TotalTimeout()
|
|
1746
|
+
|
|
1747
|
+
def _lines_with_deadline(iterator):
|
|
1748
|
+
"""Wraps resp.iter_lines() so the TOTAL deadline is enforced
|
|
1749
|
+
between chunks. Bytes keep flowing → no timeout (an actively
|
|
1750
|
+
streaming model is never misclassified as frozen); true silence
|
|
1751
|
+
is already bounded by the socket idle timeout."""
|
|
1752
|
+
for line in iterator:
|
|
1753
|
+
_check_deadline()
|
|
1754
|
+
yield line
|
|
1755
|
+
|
|
1756
|
+
_tracked = [] # registered responses, unregistered in finally
|
|
1757
|
+
|
|
1758
|
+
def _track(resp):
|
|
1759
|
+
_register_response(resp)
|
|
1760
|
+
_tracked.append(resp)
|
|
1761
|
+
return resp
|
|
1762
|
+
|
|
1763
|
+
class _FirstToken:
|
|
1764
|
+
fired = False
|
|
1765
|
+
|
|
1766
|
+
@classmethod
|
|
1767
|
+
def hit(cls):
|
|
1768
|
+
if not cls.fired:
|
|
1769
|
+
cls.fired = True
|
|
1770
|
+
_metrics_note_first_token()
|
|
1771
|
+
log_request_event(request_id, "first_token",
|
|
1772
|
+
provider=provider, model=model,
|
|
1773
|
+
size_class=size_class,
|
|
1774
|
+
elapsed_ms=elapsed_ms())
|
|
1775
|
+
|
|
1776
|
+
try:
|
|
1777
|
+
# v0.7.10: Validate base_url before making request
|
|
1778
|
+
if not base_url:
|
|
1779
|
+
yield "AI not configured — no base URL set for this provider. Run /ai to reconfigure."
|
|
1780
|
+
return
|
|
1781
|
+
if not model:
|
|
1782
|
+
yield "AI not configured — no model selected. Run /model to choose one."
|
|
1783
|
+
return
|
|
1784
|
+
# Upfront API key check so unconfigured providers fail over immediately
|
|
1785
|
+
_prov_info = _get_provider_info(provider)
|
|
1786
|
+
if _prov_info and _prov_info.get("needs_key") and not (api_key or "").strip():
|
|
1787
|
+
yield f"API key missing for provider '{provider}'. Run /key or /provider to configure."
|
|
1788
|
+
return
|
|
1789
|
+
|
|
1790
|
+
if api_style == "openai":
|
|
1791
|
+
url = f"{base_url}/chat/completions"
|
|
1792
|
+
headers = {"Content-Type": "application/json"}
|
|
1793
|
+
if api_key:
|
|
1794
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
1795
|
+
headers.update(extra_headers)
|
|
1796
|
+
payload = {
|
|
1797
|
+
"model": model,
|
|
1798
|
+
"messages": _openai_messages(system_prompt, history, prompt,
|
|
1799
|
+
attachments=attachments, vision=vision, model=model),
|
|
1800
|
+
"stream": True,
|
|
1801
|
+
}
|
|
1802
|
+
mod_lower = (model or "").lower()
|
|
1803
|
+
is_o_reasoning = ("o1" in mod_lower or "o3" in mod_lower) and "openrouter" not in str(base_url).lower()
|
|
1804
|
+
if is_o_reasoning:
|
|
1805
|
+
payload["max_completion_tokens"] = 4096
|
|
1806
|
+
elif temperature is not None:
|
|
1807
|
+
payload["temperature"] = temperature
|
|
1808
|
+
resp = _track(requests.post(url, headers=headers, json=payload,
|
|
1809
|
+
timeout=t_connect_idle, stream=True))
|
|
1810
|
+
_raise_for_status(resp)
|
|
1811
|
+
full = []
|
|
1812
|
+
usage = None
|
|
1813
|
+
for line in _lines_with_deadline(resp.iter_lines(decode_unicode=True)):
|
|
1814
|
+
if not line or not line.startswith("data:"):
|
|
1815
|
+
continue
|
|
1816
|
+
data_str = line[len("data:"):].strip()
|
|
1817
|
+
if data_str == "[DONE]":
|
|
1818
|
+
break
|
|
1819
|
+
try:
|
|
1820
|
+
obj = json.loads(data_str)
|
|
1821
|
+
except ValueError:
|
|
1822
|
+
continue
|
|
1823
|
+
choices = obj.get("choices") or []
|
|
1824
|
+
if choices:
|
|
1825
|
+
delta = choices[0].get("delta") or {}
|
|
1826
|
+
piece = delta.get("content")
|
|
1827
|
+
if piece:
|
|
1828
|
+
full.append(piece)
|
|
1829
|
+
_FirstToken.hit()
|
|
1830
|
+
yield piece
|
|
1831
|
+
if isinstance(obj.get("usage"), dict):
|
|
1832
|
+
usage = obj["usage"]
|
|
1833
|
+
completion_text = "".join(full)
|
|
1834
|
+
if usage:
|
|
1835
|
+
record_usage(usage.get("prompt_tokens", estimate_tokens(usage_prompt_text)),
|
|
1836
|
+
usage.get("completion_tokens", estimate_tokens(completion_text)))
|
|
1837
|
+
else:
|
|
1838
|
+
record_usage(estimate_tokens(usage_prompt_text), estimate_tokens(completion_text))
|
|
1839
|
+
return
|
|
1840
|
+
|
|
1841
|
+
elif api_style == "anthropic":
|
|
1842
|
+
headers = {
|
|
1843
|
+
"x-api-key": api_key,
|
|
1844
|
+
"anthropic-version": "2023-06-01",
|
|
1845
|
+
"content-type": "application/json",
|
|
1846
|
+
}
|
|
1847
|
+
headers.update(extra_headers)
|
|
1848
|
+
payload = {
|
|
1849
|
+
"model": model,
|
|
1850
|
+
"max_tokens": 4096,
|
|
1851
|
+
"system": system_prompt,
|
|
1852
|
+
"messages": _anthropic_messages(history, prompt,
|
|
1853
|
+
attachments=attachments, vision=vision),
|
|
1854
|
+
"stream": True,
|
|
1855
|
+
}
|
|
1856
|
+
if temperature is not None:
|
|
1857
|
+
payload["temperature"] = temperature
|
|
1858
|
+
resp = _track(requests.post(f"{base_url}/messages", headers=headers,
|
|
1859
|
+
json=payload, timeout=t_connect_idle,
|
|
1860
|
+
stream=True))
|
|
1861
|
+
_raise_for_status(resp)
|
|
1862
|
+
full = []
|
|
1863
|
+
in_tokens = out_tokens = None
|
|
1864
|
+
for line in _lines_with_deadline(resp.iter_lines(decode_unicode=True)):
|
|
1865
|
+
if not line or not line.startswith("data:"):
|
|
1866
|
+
continue
|
|
1867
|
+
try:
|
|
1868
|
+
obj = json.loads(line[len("data:"):].strip())
|
|
1869
|
+
except ValueError:
|
|
1870
|
+
continue
|
|
1871
|
+
etype = obj.get("type")
|
|
1872
|
+
if etype == "content_block_delta":
|
|
1873
|
+
piece = (obj.get("delta") or {}).get("text")
|
|
1874
|
+
if piece:
|
|
1875
|
+
full.append(piece)
|
|
1876
|
+
_FirstToken.hit()
|
|
1877
|
+
yield piece
|
|
1878
|
+
elif etype == "message_start":
|
|
1879
|
+
u = (obj.get("message") or {}).get("usage") or {}
|
|
1880
|
+
in_tokens = u.get("input_tokens", in_tokens)
|
|
1881
|
+
elif etype == "message_delta":
|
|
1882
|
+
u = obj.get("usage") or {}
|
|
1883
|
+
out_tokens = u.get("output_tokens", out_tokens)
|
|
1884
|
+
completion_text = "".join(full)
|
|
1885
|
+
record_usage(in_tokens if in_tokens is not None else estimate_tokens(usage_prompt_text),
|
|
1886
|
+
out_tokens if out_tokens is not None else estimate_tokens(completion_text))
|
|
1887
|
+
return
|
|
1888
|
+
|
|
1889
|
+
elif api_style == "gemini":
|
|
1890
|
+
clean_model = model.removeprefix("models/")
|
|
1891
|
+
url = f"{base_url}/models/{clean_model}:streamGenerateContent?alt=sse&key={api_key}"
|
|
1892
|
+
payload = {
|
|
1893
|
+
"contents": _gemini_contents(history, prompt,
|
|
1894
|
+
attachments=attachments, vision=vision),
|
|
1895
|
+
"systemInstruction": {"parts": [{"text": system_prompt}]},
|
|
1896
|
+
"generationConfig": {
|
|
1897
|
+
"temperature": temperature if temperature is not None else 0.7,
|
|
1898
|
+
"maxOutputTokens": 4096,
|
|
1899
|
+
},
|
|
1900
|
+
}
|
|
1901
|
+
resp = _track(requests.post(url, json=payload, timeout=t_connect_idle,
|
|
1902
|
+
stream=True))
|
|
1903
|
+
_raise_for_status(resp)
|
|
1904
|
+
full = []
|
|
1905
|
+
usage_meta = None
|
|
1906
|
+
for line in _lines_with_deadline(resp.iter_lines(decode_unicode=True)):
|
|
1907
|
+
if not line or not line.startswith("data:"):
|
|
1908
|
+
continue
|
|
1909
|
+
try:
|
|
1910
|
+
obj = json.loads(line[len("data:"):].strip())
|
|
1911
|
+
except ValueError:
|
|
1912
|
+
continue
|
|
1913
|
+
candidates = obj.get("candidates") or []
|
|
1914
|
+
if candidates:
|
|
1915
|
+
parts = (candidates[0].get("content") or {}).get("parts") or []
|
|
1916
|
+
for part in parts:
|
|
1917
|
+
piece = part.get("text")
|
|
1918
|
+
if piece:
|
|
1919
|
+
full.append(piece)
|
|
1920
|
+
_FirstToken.hit()
|
|
1921
|
+
yield piece
|
|
1922
|
+
if isinstance(obj.get("usageMetadata"), dict):
|
|
1923
|
+
usage_meta = obj["usageMetadata"]
|
|
1924
|
+
completion_text = "".join(full)
|
|
1925
|
+
if usage_meta:
|
|
1926
|
+
record_usage(usage_meta.get("promptTokenCount", estimate_tokens(usage_prompt_text)),
|
|
1927
|
+
usage_meta.get("candidatesTokenCount", estimate_tokens(completion_text)))
|
|
1928
|
+
else:
|
|
1929
|
+
record_usage(estimate_tokens(usage_prompt_text), estimate_tokens(completion_text))
|
|
1930
|
+
return
|
|
1931
|
+
|
|
1932
|
+
elif api_style == "ollama":
|
|
1933
|
+
# Ensure Ollama is running — auto-start `ollama serve` if needed (fixes "Could not reach" when not running)
|
|
1934
|
+
try:
|
|
1935
|
+
from .ollama_download import ensure_ollama_running, is_ollama_installed
|
|
1936
|
+
ok, msg = ensure_ollama_running(base_url, timeout=3, auto_start=True)
|
|
1937
|
+
if not ok:
|
|
1938
|
+
hint = "Ollama not installed — install from https://ollama.com/download and run `ollama serve`" if not is_ollama_installed() else msg
|
|
1939
|
+
raise RuntimeError(f"Could not reach the AI server. Could not reach Ollama at {base_url}. {hint}. Or install a model via ☰ → Ollama Models.")
|
|
1940
|
+
except RuntimeError:
|
|
1941
|
+
raise
|
|
1942
|
+
except Exception:
|
|
1943
|
+
pass
|
|
1944
|
+
|
|
1945
|
+
from .models.profiles import get_model_profile
|
|
1946
|
+
from .providers.ollama_adapter import OllamaProvider
|
|
1947
|
+
from .ai_context import get_context_manager, AIRequest, AIMessage
|
|
1948
|
+
import uuid
|
|
1949
|
+
|
|
1950
|
+
profile = get_model_profile("ollama", model, config)
|
|
1951
|
+
options = profile.get_effective_options({"temperature": temperature})
|
|
1952
|
+
|
|
1953
|
+
# Detect current AI mode
|
|
1954
|
+
active_mode = "chat"
|
|
1955
|
+
if config and isinstance(config, dict) and config.get("mode"):
|
|
1956
|
+
active_mode = config.get("mode")
|
|
1957
|
+
else:
|
|
1958
|
+
try:
|
|
1959
|
+
from . import ai_modes as _am
|
|
1960
|
+
active_mode = _am.current_mode()
|
|
1961
|
+
except Exception:
|
|
1962
|
+
active_mode = "chat"
|
|
1963
|
+
|
|
1964
|
+
# Filter and budget history using AIContextManager to prevent cross-mode context pollution
|
|
1965
|
+
ctx_mgr = get_context_manager()
|
|
1966
|
+
ai_history = []
|
|
1967
|
+
for r, t in (history or []):
|
|
1968
|
+
ai_history.append(AIMessage(id=uuid.uuid4().hex, role=r, content=t, mode=active_mode))
|
|
1969
|
+
|
|
1970
|
+
req = AIRequest(
|
|
1971
|
+
session_id=str(getattr(config, "get", lambda k, d="": d)("session_id", "cct")),
|
|
1972
|
+
request_id=request_id or uuid.uuid4().hex,
|
|
1973
|
+
mode=active_mode,
|
|
1974
|
+
user_message=prompt,
|
|
1975
|
+
history=ai_history,
|
|
1976
|
+
model=model,
|
|
1977
|
+
provider="ollama",
|
|
1978
|
+
)
|
|
1979
|
+
built_ctx = ctx_mgr.build_context(req, profile=profile)
|
|
1980
|
+
messages = built_ctx.messages
|
|
1981
|
+
|
|
1982
|
+
# If caller supplied a custom system prompt (and not default), apply it to system role
|
|
1983
|
+
if system_prompt and system_prompt != DEFAULT_SYSTEM_PROMPT and messages and messages[0]["role"] == "system":
|
|
1984
|
+
if profile.is_small_or_base() and len(system_prompt) > 800:
|
|
1985
|
+
messages[0]["content"] = system_prompt[:800] + "\nAnswer concisely."
|
|
1986
|
+
else:
|
|
1987
|
+
messages[0]["content"] = system_prompt
|
|
1988
|
+
|
|
1989
|
+
ollama_prov = OllamaProvider(base_url)
|
|
1990
|
+
_register_provider(ollama_prov)
|
|
1991
|
+
full = []
|
|
1992
|
+
try:
|
|
1993
|
+
for piece in ollama_prov.stream_chat(
|
|
1994
|
+
model=model,
|
|
1995
|
+
messages=messages,
|
|
1996
|
+
options=options,
|
|
1997
|
+
timeout=t_connect_idle,
|
|
1998
|
+
request_id=request_id,
|
|
1999
|
+
deadline=deadline,
|
|
2000
|
+
):
|
|
2001
|
+
_FirstToken.hit()
|
|
2002
|
+
full.append(piece)
|
|
2003
|
+
yield piece
|
|
2004
|
+
finally:
|
|
2005
|
+
_unregister_provider(ollama_prov)
|
|
2006
|
+
|
|
2007
|
+
completion_text = "".join(full)
|
|
2008
|
+
record_usage(estimate_tokens(usage_prompt_text), estimate_tokens(completion_text))
|
|
2009
|
+
return
|
|
2010
|
+
|
|
2011
|
+
else:
|
|
2012
|
+
yield f"Provider '{provider}' is not supported yet. Run /ai to reconfigure."
|
|
2013
|
+
return
|
|
2014
|
+
|
|
2015
|
+
except GeneratorExit:
|
|
2016
|
+
# Consumer stopped iterating (cancel / screen teardown). Run the
|
|
2017
|
+
# cleanup in finally and close the socket promptly.
|
|
2018
|
+
raise
|
|
2019
|
+
except _TotalTimeout:
|
|
2020
|
+
log_request_event(request_id, "error", provider=provider, model=model,
|
|
2021
|
+
size_class=size_class, elapsed_ms=elapsed_ms(),
|
|
2022
|
+
detail="total request deadline exceeded")
|
|
2023
|
+
yield ("The AI request timed out before the provider finished "
|
|
2024
|
+
"responding. Try again, or run `/model` to switch to a faster model.")
|
|
2025
|
+
except requests.exceptions.ConnectionError as ce:
|
|
2026
|
+
if _just_cancelled():
|
|
2027
|
+
return # user cancelled — stop cleanly, no scary message
|
|
2028
|
+
ce_str = str(ce).lower()
|
|
2029
|
+
log_request_event(request_id, "error", provider=provider, model=model,
|
|
2030
|
+
size_class=size_class, elapsed_ms=elapsed_ms(),
|
|
2031
|
+
detail=f"connection error: {ce_str[:120]}")
|
|
2032
|
+
if "ollama" in (provider or "").lower() or "localhost" in (base_url or ""):
|
|
2033
|
+
yield ("Could not reach your local Ollama server. "
|
|
2034
|
+
"Make sure Ollama is running (`ollama serve`), or run `/ai` to switch to a cloud provider.")
|
|
2035
|
+
else:
|
|
2036
|
+
yield (f"Could not reach {provider or 'the AI server'}. "
|
|
2037
|
+
"Check your internet connection and API URL, or run `/model` to switch providers.")
|
|
2038
|
+
except requests.exceptions.Timeout:
|
|
2039
|
+
if _just_cancelled():
|
|
2040
|
+
return
|
|
2041
|
+
log_request_event(request_id, "error", provider=provider, model=model,
|
|
2042
|
+
size_class=size_class, elapsed_ms=elapsed_ms(),
|
|
2043
|
+
detail="idle read timeout")
|
|
2044
|
+
if "ollama" in (provider or "").lower():
|
|
2045
|
+
yield ("Ollama took too long to respond — the model may be loading. "
|
|
2046
|
+
"Try sending your message again, or run `/model` to pick a smaller model.")
|
|
2047
|
+
else:
|
|
2048
|
+
yield (f"Request to {provider or 'AI'} timed out. "
|
|
2049
|
+
"Try again, use a faster model (`/model`), or check your network.")
|
|
2050
|
+
except RuntimeError as e:
|
|
2051
|
+
# v0.7.10: Better error messages for common HTTP errors
|
|
2052
|
+
if _just_cancelled():
|
|
2053
|
+
return
|
|
2054
|
+
err_msg = str(e)
|
|
2055
|
+
log_request_event(request_id, "error", provider=provider, model=model,
|
|
2056
|
+
size_class=size_class, elapsed_ms=elapsed_ms(),
|
|
2057
|
+
detail=err_msg[:200])
|
|
2058
|
+
if "410" in err_msg or "end of life" in err_msg.lower():
|
|
2059
|
+
yield (f"Model '{model}' has been retired by {provider}. "
|
|
2060
|
+
"Run `/model` to choose a current model.")
|
|
2061
|
+
elif "401" in err_msg or "unauthorized" in err_msg.lower():
|
|
2062
|
+
yield (f"Invalid API key for {provider}. "
|
|
2063
|
+
"Run `/ai` to reconfigure with a valid key.")
|
|
2064
|
+
elif "429" in err_msg or "rate limit" in err_msg.lower():
|
|
2065
|
+
yield (f"{provider} rate limit hit. Wait a moment and retry, "
|
|
2066
|
+
"or run `/model` to switch providers.")
|
|
2067
|
+
elif "402" in err_msg or "quota" in err_msg.lower() or "insufficient" in err_msg.lower():
|
|
2068
|
+
yield (f"{provider} quota exceeded. Run `/model` to switch to a free provider "
|
|
2069
|
+
"or add credits to your account.")
|
|
2070
|
+
elif "404" in err_msg or "not found" in err_msg.lower():
|
|
2071
|
+
yield (f"Model '{model}' not found on {provider}. "
|
|
2072
|
+
"Run `/model` to pick an available model.")
|
|
2073
|
+
elif "500" in err_msg or "internal server error" in err_msg.lower():
|
|
2074
|
+
yield (f"{provider} server error. This is usually temporary — "
|
|
2075
|
+
"try again in a moment, or run `/model` to switch.")
|
|
2076
|
+
else:
|
|
2077
|
+
yield f"Error from {provider}: {err_msg}"
|
|
2078
|
+
except Exception as e:
|
|
2079
|
+
if _just_cancelled():
|
|
2080
|
+
return
|
|
2081
|
+
log_request_event(request_id, "error", provider=provider, model=model,
|
|
2082
|
+
size_class=size_class, elapsed_ms=elapsed_ms(),
|
|
2083
|
+
detail=f"{type(e).__name__}: {e}")
|
|
2084
|
+
yield f"Error connecting to {provider or 'AI'}: {type(e).__name__}: {e}"
|
|
2085
|
+
finally:
|
|
2086
|
+
# ALWAYS drop the socket registrations — a finished or failed
|
|
2087
|
+
# request must never keep cancel_active_requests() holding stale
|
|
2088
|
+
# response objects (requirement: loading state always clears).
|
|
2089
|
+
for r in _tracked:
|
|
2090
|
+
_unregister_response(r)
|
|
2091
|
+
_tracked.clear()
|
|
2092
|
+
|
|
2093
|
+
|
|
2094
|
+
def _raise_for_status(resp):
|
|
2095
|
+
"""Raise a clear error with the API's own message instead of a vague one.
|
|
2096
|
+
|
|
2097
|
+
Most providers return a helpful JSON body explaining *why* a call failed
|
|
2098
|
+
(bad model name, invalid key, quota), but requests only surfaces the HTTP
|
|
2099
|
+
code. This pulls that detail out so the user can actually fix the issue.
|
|
2100
|
+
"""
|
|
2101
|
+
if resp.status_code >= 400:
|
|
2102
|
+
try:
|
|
2103
|
+
body = resp.json()
|
|
2104
|
+
err = (body.get("error", {}).get("message")
|
|
2105
|
+
or body.get("error", {}).get("status")
|
|
2106
|
+
or str(body)[:300])
|
|
2107
|
+
except Exception:
|
|
2108
|
+
err = resp.text[:300] if resp.text else resp.reason
|
|
2109
|
+
raise RuntimeError(f"HTTP {resp.status_code}: {err}")
|
|
2110
|
+
|
|
2111
|
+
|
|
2112
|
+
# ---------------------------------------------------------------- web search
|
|
2113
|
+
# Key-free web search using DuckDuckGo's HTML endpoint (no API key needed,
|
|
2114
|
+
# unlike Google/Bing search APIs). Used directly by /websearch and /research,
|
|
2115
|
+
# and wired into the agent as a tool so it can look things up mid-answer.
|
|
2116
|
+
def _unwrap_ddg_url(href):
|
|
2117
|
+
"""DuckDuckGo's HTML results wrap outbound links in a redirect
|
|
2118
|
+
(/l/?uddg=<encoded-url>); unwrap that back to the real destination."""
|
|
2119
|
+
if href.startswith("//"):
|
|
2120
|
+
href = "https:" + href
|
|
2121
|
+
try:
|
|
2122
|
+
parsed = urlparse(href)
|
|
2123
|
+
if "duckduckgo.com" in parsed.netloc and parsed.path.startswith("/l/"):
|
|
2124
|
+
qs = parse_qs(parsed.query)
|
|
2125
|
+
if "uddg" in qs:
|
|
2126
|
+
return unquote(qs["uddg"][0])
|
|
2127
|
+
return href
|
|
2128
|
+
except Exception:
|
|
2129
|
+
return href
|
|
2130
|
+
|
|
2131
|
+
|
|
2132
|
+
_TAG_RE = re.compile(r"<.*?>", re.S)
|
|
2133
|
+
|
|
2134
|
+
|
|
2135
|
+
def web_search(query, max_results=5):
|
|
2136
|
+
"""Best-effort web search, no API key required. Returns a list of
|
|
2137
|
+
{\"title\", \"url\", \"snippet\"} dicts, or [] if unreachable/blocked —
|
|
2138
|
+
callers should treat an empty list as 'couldn't search right now', not
|
|
2139
|
+
an error."""
|
|
2140
|
+
query = (query or "").strip()
|
|
2141
|
+
if not query or not _HAS_REQUESTS:
|
|
2142
|
+
return []
|
|
2143
|
+
try:
|
|
2144
|
+
resp = requests.post(
|
|
2145
|
+
"https://html.duckduckgo.com/html/",
|
|
2146
|
+
data={"q": query}, timeout=12,
|
|
2147
|
+
headers={"User-Agent": "Mozilla/5.0 (compatible; CAT/0.7.9.0)"},
|
|
2148
|
+
)
|
|
2149
|
+
if resp.status_code >= 400:
|
|
2150
|
+
return []
|
|
2151
|
+
html = resp.text
|
|
2152
|
+
results = []
|
|
2153
|
+
pattern = re.compile(
|
|
2154
|
+
r'result__a"[^>]*href="([^"]+)"[^>]*>(.*?)</a>.*?'
|
|
2155
|
+
r'result__snippet[^>]*>(.*?)</a>', re.S)
|
|
2156
|
+
for m in pattern.finditer(html):
|
|
2157
|
+
href, title_html, snippet_html = m.groups()
|
|
2158
|
+
title = _html_unescape(_TAG_RE.sub("", title_html)).strip()
|
|
2159
|
+
snippet = _html_unescape(_TAG_RE.sub("", snippet_html)).strip()
|
|
2160
|
+
url = _unwrap_ddg_url(href)
|
|
2161
|
+
if title and url:
|
|
2162
|
+
results.append({"title": title, "url": url, "snippet": snippet})
|
|
2163
|
+
if len(results) >= max_results:
|
|
2164
|
+
break
|
|
2165
|
+
return results
|
|
2166
|
+
except Exception:
|
|
2167
|
+
return []
|
|
2168
|
+
|
|
2169
|
+
|
|
2170
|
+
def deep_research(topic, num_queries=3, results_per_query=4):
|
|
2171
|
+
"""Runs several web searches around different angles of `topic`, then
|
|
2172
|
+
asks the configured AI to synthesize a structured, source-numbered
|
|
2173
|
+
research summary from the actual retrieved snippets (not from the
|
|
2174
|
+
model's own unchecked memory). Returns (summary_text, sources) where
|
|
2175
|
+
sources is a de-duplicated list of {\"title\", \"url\", \"snippet\"} dicts,
|
|
2176
|
+
numbered in the same order referenced as [1], [2]... in the summary."""
|
|
2177
|
+
topic = (topic or "").strip()
|
|
2178
|
+
if not topic:
|
|
2179
|
+
return "No research topic given.", []
|
|
2180
|
+
|
|
2181
|
+
sub_queries = [topic]
|
|
2182
|
+
config = load_config()
|
|
2183
|
+
if config.get("provider"):
|
|
2184
|
+
angles_raw = query_ai(
|
|
2185
|
+
f"Give exactly {num_queries} short, distinct web-search queries (one per "
|
|
2186
|
+
f"line, no numbering, no extra commentary) that together would build a "
|
|
2187
|
+
f"thorough, well-rounded understanding of: {topic}",
|
|
2188
|
+
system_prompt="You output ONLY the search queries, one per line, nothing else.")
|
|
2189
|
+
angles = [a.strip("-•*0123456789. ").strip() for a in (angles_raw or "").splitlines() if a.strip()]
|
|
2190
|
+
if angles:
|
|
2191
|
+
sub_queries = angles[:num_queries]
|
|
2192
|
+
|
|
2193
|
+
all_results = []
|
|
2194
|
+
seen_urls = set()
|
|
2195
|
+
for q in sub_queries:
|
|
2196
|
+
for r in web_search(q, max_results=results_per_query):
|
|
2197
|
+
if r["url"] not in seen_urls:
|
|
2198
|
+
seen_urls.add(r["url"])
|
|
2199
|
+
all_results.append(r)
|
|
2200
|
+
|
|
2201
|
+
if not all_results:
|
|
2202
|
+
return (f"No web results could be retrieved for '{topic}' — check the "
|
|
2203
|
+
f"internet connection this terminal has, DuckDuckGo may also be "
|
|
2204
|
+
f"rate-limiting/blocking this network."), []
|
|
2205
|
+
|
|
2206
|
+
evidence = "\n\n".join(
|
|
2207
|
+
f"[{i + 1}] {r['title']}\n{r['url']}\n{r['snippet']}"
|
|
2208
|
+
for i, r in enumerate(all_results[:12]))
|
|
2209
|
+
|
|
2210
|
+
if not config.get("provider"):
|
|
2211
|
+
return ("AI not configured, so here are the raw sources found "
|
|
2212
|
+
"(run /model or /ai to also get a synthesized summary):\n\n" + evidence,
|
|
2213
|
+
all_results[:12])
|
|
2214
|
+
|
|
2215
|
+
summary = query_ai(
|
|
2216
|
+
f"Research topic: {topic}\n\nSources:\n{evidence}\n\n"
|
|
2217
|
+
"Write a clear, well-organized research summary of the topic using ONLY "
|
|
2218
|
+
"these sources. Reference sources inline as [1], [2] etc. matching the "
|
|
2219
|
+
"numbers above. Point out any disagreement between sources. Do not invent "
|
|
2220
|
+
"facts the sources don't support.",
|
|
2221
|
+
system_prompt=("You are a careful research assistant. Be accurate, cite "
|
|
2222
|
+
"sources by their bracket number, and never invent facts "
|
|
2223
|
+
"not supported by the given sources."))
|
|
2224
|
+
return summary, all_results[:12]
|
|
2225
|
+
|
|
2226
|
+
|
|
2227
|
+
# ------------------------------------------------------------ file / image import
|
|
2228
|
+
# Backs /import (feature 6): local text files get their content read straight
|
|
2229
|
+
# into the AI/agent's context; local images get sent to a vision-capable
|
|
2230
|
+
# provider (OpenAI/Anthropic/Gemini) so the model can actually "see" them.
|
|
2231
|
+
IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"}
|
|
2232
|
+
TEXT_FILE_MAX_CHARS = 20000
|
|
2233
|
+
|
|
2234
|
+
|
|
2235
|
+
def is_image_file(path):
|
|
2236
|
+
return os.path.splitext(str(path))[1].lower() in IMAGE_EXTS
|
|
2237
|
+
|
|
2238
|
+
|
|
2239
|
+
def read_text_file_for_context(path):
|
|
2240
|
+
"""Returns (content, truncated) or (None, False) on failure."""
|
|
2241
|
+
try:
|
|
2242
|
+
with open(path, "r", encoding="utf-8", errors="replace") as f:
|
|
2243
|
+
content = f.read(TEXT_FILE_MAX_CHARS + 1)
|
|
2244
|
+
truncated = len(content) > TEXT_FILE_MAX_CHARS
|
|
2245
|
+
if truncated:
|
|
2246
|
+
content = content[:TEXT_FILE_MAX_CHARS]
|
|
2247
|
+
return content, truncated
|
|
2248
|
+
except Exception:
|
|
2249
|
+
return None, False
|
|
2250
|
+
|
|
2251
|
+
|
|
2252
|
+
# ------------------------------------------------------- attachment context
|
|
2253
|
+
# v0.7.6 Patch 1, Fix 7: one attachment gets one honest, real context block
|
|
2254
|
+
# fed to the AI. Plain text/code/markdown gets its actual content; CSV gets
|
|
2255
|
+
# a genuine header/sample/count; zip/tar get a real listing; PDF gets real
|
|
2256
|
+
# metadata + a best-effort text-layer extraction (stdlib-only — no pypdf in
|
|
2257
|
+
# this environment); audio/video report what's known (size/ext) and say
|
|
2258
|
+
# plainly when a metadata library isn't installed. Never fabricated data.
|
|
2259
|
+
AUDIO_EXTS = {".mp3", ".wav", ".ogg", ".flac", ".m4a", ".aac", ".wma", ".opus"}
|
|
2260
|
+
VIDEO_EXTS = {".mp4", ".mkv", ".avi", ".mov", ".webm", ".wmv", ".m4v", ".mpg", ".mpeg"}
|
|
2261
|
+
_BINARY_EXTS = {".exe", ".dll", ".so", ".dylib", ".bin", ".dat", ".db", ".sqlite",
|
|
2262
|
+
".sqlite3", ".pyc", ".pdb", ".iso", ".img", ".parquet", ".h5", ".hdf5"}
|
|
2263
|
+
|
|
2264
|
+
|
|
2265
|
+
def _human_size(n):
|
|
2266
|
+
if n < 1024:
|
|
2267
|
+
return f"{n} B"
|
|
2268
|
+
if n < 1024 ** 2:
|
|
2269
|
+
return f"{n / 1024:.1f} KB"
|
|
2270
|
+
if n < 1024 ** 3:
|
|
2271
|
+
return f"{n / 1024 ** 2:.1f} MB"
|
|
2272
|
+
return f"{n / 1024 ** 3:.1f} GB"
|
|
2273
|
+
|
|
2274
|
+
|
|
2275
|
+
def _pdf_context(path, name, size_txt, max_chars):
|
|
2276
|
+
try:
|
|
2277
|
+
with open(path, "rb") as f:
|
|
2278
|
+
raw = f.read()
|
|
2279
|
+
except Exception as e:
|
|
2280
|
+
return f"[attachment: {name} ({size_txt}) \u2014 unreadable PDF: {e}]"
|
|
2281
|
+
counts = [int(m) for m in re.findall(rb"/Count\s+(\d+)", raw)]
|
|
2282
|
+
page_txt = f"{max(counts)} pages" if counts else "page count unknown"
|
|
2283
|
+
meta_bits = []
|
|
2284
|
+
for key in (b"Title", b"Author", b"Subject", b"Creator", b"Producer"):
|
|
2285
|
+
m = re.search(key + rb"\s*\(([^()\\]*(?:\\.[^()\\]*)*)\)", raw[:200000])
|
|
2286
|
+
if m:
|
|
2287
|
+
val = m.group(1)[:120].decode("latin-1", "replace")
|
|
2288
|
+
meta_bits.append(f"{key.decode()}: {val}")
|
|
2289
|
+
meta_txt = ("; ".join(meta_bits) + ".") if meta_bits else "no document metadata."
|
|
2290
|
+
# Best-effort text-layer extraction: decompress every object stream
|
|
2291
|
+
# and pull out literal text shown via Tj/TJ operators. Stdlib-only,
|
|
2292
|
+
# so genuinely imperfect — but real text, never fabricated.
|
|
2293
|
+
texts = []
|
|
2294
|
+
for m in re.finditer(rb"stream\r?\n(.*?)endstream", raw, re.DOTALL):
|
|
2295
|
+
data = m.group(1).lstrip(b"\r\n")
|
|
2296
|
+
for payload in (data,):
|
|
2297
|
+
try:
|
|
2298
|
+
payload = zlib.decompress(data)
|
|
2299
|
+
except Exception:
|
|
2300
|
+
pass
|
|
2301
|
+
texts += re.findall(rb"\(((?:[^()\\]|\\.)*)\)\s*Tj", payload)
|
|
2302
|
+
plain = " ".join(
|
|
2303
|
+
t.replace(b"\\(", b"(").replace(b"\\)", b")").replace(b"\\\\", b"\\")
|
|
2304
|
+
.decode("latin-1", "replace") for t in texts)
|
|
2305
|
+
plain = re.sub(r"\s+", " ", plain).strip()
|
|
2306
|
+
if plain:
|
|
2307
|
+
if len(plain) > max_chars:
|
|
2308
|
+
plain = plain[:max_chars] + " \u2026[truncated]"
|
|
2309
|
+
body = f"[attachment: {name} ({size_txt}) \u2014 PDF, {page_txt}; {meta_txt}]\n{plain}"
|
|
2310
|
+
else:
|
|
2311
|
+
body = (f"[attachment: {name} ({size_txt}) \u2014 PDF, {page_txt}; {meta_txt} "
|
|
2312
|
+
f"No extractable text layer (scanned image or glyph-encoded PDF).]")
|
|
2313
|
+
return body
|
|
2314
|
+
|
|
2315
|
+
|
|
2316
|
+
def _zip_context(path, name, size_txt):
|
|
2317
|
+
import zipfile
|
|
2318
|
+
try:
|
|
2319
|
+
with zipfile.ZipFile(path) as zf:
|
|
2320
|
+
infos = zf.infolist()
|
|
2321
|
+
files = [i for i in infos if not i.is_dir()]
|
|
2322
|
+
dirs = [i for i in infos if i.is_dir()]
|
|
2323
|
+
total = sum(i.file_size for i in files)
|
|
2324
|
+
listing = "\n".join(i.filename for i in files[:40])
|
|
2325
|
+
more = f"\n\u2026 and {len(files) - 40} more files" if len(files) > 40 else ""
|
|
2326
|
+
return (f"[attachment: {name} ({size_txt}) \u2014 zip archive: {len(files)} files, "
|
|
2327
|
+
f"{len(dirs)} folders, {_human_size(total)} uncompressed]\n"
|
|
2328
|
+
f"{listing}{more}" if files else
|
|
2329
|
+
f"[attachment: {name} ({size_txt}) \u2014 zip archive, empty.]")
|
|
2330
|
+
except Exception as e:
|
|
2331
|
+
return f"[attachment: {name} ({size_txt}) \u2014 not a readable zip archive: {e}]"
|
|
2332
|
+
|
|
2333
|
+
|
|
2334
|
+
def _tar_context(path, name, size_txt):
|
|
2335
|
+
import tarfile
|
|
2336
|
+
try:
|
|
2337
|
+
with tarfile.open(path) as tf:
|
|
2338
|
+
members = tf.getmembers()
|
|
2339
|
+
files = [m for m in members if m.isfile()]
|
|
2340
|
+
listing = "\n".join(m.name for m in files[:40])
|
|
2341
|
+
more = f"\n\u2026 and {len(files) - 40} more files" if len(files) > 40 else ""
|
|
2342
|
+
return (f"[attachment: {name} ({size_txt}) \u2014 tar archive: {len(files)} files]\n"
|
|
2343
|
+
f"{listing}{more}" if files else
|
|
2344
|
+
f"[attachment: {name} ({size_txt}) \u2014 tar archive, empty.]")
|
|
2345
|
+
except Exception as e:
|
|
2346
|
+
return f"[attachment: {name} ({size_txt}) \u2014 not a readable tar archive: {e}]"
|
|
2347
|
+
|
|
2348
|
+
|
|
2349
|
+
def _csv_context(path, name, size_txt, max_chars):
|
|
2350
|
+
import csv
|
|
2351
|
+
header, sample, total = None, [], 0
|
|
2352
|
+
capped = False
|
|
2353
|
+
try:
|
|
2354
|
+
with open(path, "r", encoding="utf-8", errors="replace", newline="") as f:
|
|
2355
|
+
for i, row in enumerate(csv.reader(f)):
|
|
2356
|
+
if i == 0:
|
|
2357
|
+
header = row
|
|
2358
|
+
elif len(sample) < 5:
|
|
2359
|
+
sample.append(row)
|
|
2360
|
+
total += 1
|
|
2361
|
+
if total >= 5000:
|
|
2362
|
+
capped = True
|
|
2363
|
+
break
|
|
2364
|
+
except Exception as e:
|
|
2365
|
+
return f"[attachment: {name} ({size_txt}) \u2014 could not parse CSV: {e}]"
|
|
2366
|
+
cols = f"{len(header)} columns" if header else "0 columns"
|
|
2367
|
+
total_txt = f"{total} rows" + (" (capped at 5000)" if capped else "")
|
|
2368
|
+
lines = [f"[attachment: {name} ({size_txt}) \u2014 CSV, {cols}, {total_txt}]"]
|
|
2369
|
+
if header:
|
|
2370
|
+
lines.append("header: " + " | ".join(header))
|
|
2371
|
+
if sample:
|
|
2372
|
+
lines.append("first rows:")
|
|
2373
|
+
lines += [" " + " | ".join(row) for row in sample]
|
|
2374
|
+
body = "\n".join(lines)
|
|
2375
|
+
return body[:max_chars] + (" \u2026[truncated]" if len(body) > max_chars else "")
|
|
2376
|
+
|
|
2377
|
+
|
|
2378
|
+
def _image_context(path, name, size_txt):
|
|
2379
|
+
note = (f"[Attached image: {path} \u2014 not yet included in streamed "
|
|
2380
|
+
f"replies; vision-capable providers analyze it via the "
|
|
2381
|
+
f"query_ai_with_image path instead]")
|
|
2382
|
+
try:
|
|
2383
|
+
from PIL import Image
|
|
2384
|
+
with Image.open(path) as im:
|
|
2385
|
+
w, h = im.size
|
|
2386
|
+
fmt = (im.format or "").upper()
|
|
2387
|
+
return (f"[Attached image: {name} ({size_txt}, {w}x{h} {fmt}) \u2014 "
|
|
2388
|
+
f"not yet included in streamed replies; vision-capable "
|
|
2389
|
+
f"providers analyze it via the query_ai_with_image path instead]")
|
|
2390
|
+
except Exception:
|
|
2391
|
+
return note
|
|
2392
|
+
|
|
2393
|
+
|
|
2394
|
+
def attachment_context_for(path, max_chars=TEXT_FILE_MAX_CHARS):
|
|
2395
|
+
"""v0.7.6 Patch 1, Fix 7: the complete AI-side context block for one
|
|
2396
|
+
attached file — real content or real metadata per file type, always
|
|
2397
|
+
honest about what couldn't be extracted (no fabricated analysis).
|
|
2398
|
+
Images keep the existing vision path; text-like files (md, code,
|
|
2399
|
+
data, config) get their actual content; CSV/zip/tar/PDF get genuine
|
|
2400
|
+
structure; audio/video report what's knowable in this environment.
|
|
2401
|
+
"""
|
|
2402
|
+
if not os.path.isfile(path):
|
|
2403
|
+
return f"[attachment not found: {path}]"
|
|
2404
|
+
name = os.path.basename(path)
|
|
2405
|
+
ext = os.path.splitext(name)[1].lower()
|
|
2406
|
+
try:
|
|
2407
|
+
size = os.path.getsize(path)
|
|
2408
|
+
except Exception:
|
|
2409
|
+
size = 0
|
|
2410
|
+
size_txt = _human_size(size)
|
|
2411
|
+
|
|
2412
|
+
if is_image_file(path):
|
|
2413
|
+
return _image_context(path, name, size_txt)
|
|
2414
|
+
if ext == ".pdf":
|
|
2415
|
+
return _pdf_context(path, name, size_txt, max_chars)
|
|
2416
|
+
if ext == ".zip":
|
|
2417
|
+
return _zip_context(path, name, size_txt)
|
|
2418
|
+
if ext in {".tar", ".gz", ".bz2", ".tgz"}:
|
|
2419
|
+
return _tar_context(path, name, size_txt)
|
|
2420
|
+
if ext in {".rar", ".7z"}:
|
|
2421
|
+
return (f"[attachment: {name} ({size_txt}) \u2014 {ext} archive; "
|
|
2422
|
+
f"no stdlib reader exists here, extract it first and attach "
|
|
2423
|
+
f"the contents instead]")
|
|
2424
|
+
if ext == ".csv":
|
|
2425
|
+
return _csv_context(path, name, size_txt, max_chars)
|
|
2426
|
+
if ext in AUDIO_EXTS:
|
|
2427
|
+
return (f"[attachment: {name} ({size_txt}) \u2014 audio file; duration "
|
|
2428
|
+
f"and tags would need the 'mutagen' package, which isn't "
|
|
2429
|
+
f"installed in this environment]")
|
|
2430
|
+
if ext in VIDEO_EXTS:
|
|
2431
|
+
return (f"[attachment: {name} ({size_txt}) \u2014 video file; duration "
|
|
2432
|
+
f"and codec info would need the 'mutagen' package, which "
|
|
2433
|
+
f"isn't installed in this environment]")
|
|
2434
|
+
if ext in _BINARY_EXTS:
|
|
2435
|
+
return f"[attachment: {name} ({size_txt}) \u2014 binary {ext} file, content not readable as text]"
|
|
2436
|
+
|
|
2437
|
+
content, truncated = read_text_file_for_context(path)
|
|
2438
|
+
if content is None:
|
|
2439
|
+
return f"[attachment: {name} ({size_txt}) \u2014 could not be read as text]"
|
|
2440
|
+
head = (f"[attachment: {name} ({size_txt})"
|
|
2441
|
+
+ (" \u2014 truncated to first chars]" if truncated else "]"))
|
|
2442
|
+
return head + "\n" + content
|
|
2443
|
+
|
|
2444
|
+
|
|
2445
|
+
def encode_image_b64(path):
|
|
2446
|
+
with open(path, "rb") as f:
|
|
2447
|
+
return base64.b64encode(f.read()).decode("ascii")
|
|
2448
|
+
|
|
2449
|
+
|
|
2450
|
+
def _guess_image_mime(path):
|
|
2451
|
+
ext = os.path.splitext(str(path))[1].lower().lstrip(".")
|
|
2452
|
+
return {"jpg": "image/jpeg", "jpeg": "image/jpeg", "png": "image/png",
|
|
2453
|
+
"gif": "image/gif", "webp": "image/webp", "bmp": "image/bmp"}.get(ext, "image/png")
|
|
2454
|
+
|
|
2455
|
+
|
|
2456
|
+
def query_ai_with_image(prompt, image_path,
|
|
2457
|
+
system_prompt=("You are CAT AI. Describe and analyze the attached image "
|
|
2458
|
+
"precisely.\n\n" + identity.IDENTITY_BLOCK)):
|
|
2459
|
+
"""Same idea as query_ai but attaches one local image, for providers whose
|
|
2460
|
+
API actually supports vision here (OpenAI-style, Anthropic, Gemini).
|
|
2461
|
+
|
|
2462
|
+
v0.7.9.0 (requirement #19 — multimodal fallback): the image is
|
|
2463
|
+
normalized through the vision pipeline first; if the ACTIVE model
|
|
2464
|
+
can't receive native images but another CONFIGURED model can, the
|
|
2465
|
+
request is rerouted there instead of replying 'model can't see this'.
|
|
2466
|
+
Only when no vision-capable model exists at all is that reported,
|
|
2467
|
+
clearly and honestly."""
|
|
2468
|
+
if not _HAS_REQUESTS:
|
|
2469
|
+
return "The 'requests' library is required for AI features. Please run: pip install requests"
|
|
2470
|
+
|
|
2471
|
+
try:
|
|
2472
|
+
from . import vision as _vision
|
|
2473
|
+
mime, b64 = _vision.encode_for_model(image_path, prompt)[:2]
|
|
2474
|
+
except Exception as e:
|
|
2475
|
+
return f"Could not read image '{image_path}': {e}"
|
|
2476
|
+
|
|
2477
|
+
config = load_config()
|
|
2478
|
+
provider = config.get("provider")
|
|
2479
|
+
if not provider:
|
|
2480
|
+
return "AI not configured. Run /ai, /agent, or /model to configure your provider first."
|
|
2481
|
+
|
|
2482
|
+
info = PROVIDERS.get(provider)
|
|
2483
|
+
api_style = info["api_style"] if info else "openai"
|
|
2484
|
+
used_config = config
|
|
2485
|
+
if api_style not in ("openai", "anthropic", "gemini"):
|
|
2486
|
+
# Primary can't take native images — look for a configured one that can.
|
|
2487
|
+
try:
|
|
2488
|
+
from . import model_router as _mr
|
|
2489
|
+
vcfg, _caps = _mr.find_vision_capable(config)
|
|
2490
|
+
except Exception:
|
|
2491
|
+
vcfg = None
|
|
2492
|
+
if vcfg is None:
|
|
2493
|
+
return ("No vision-capable model is currently configured, so the "
|
|
2494
|
+
f"image couldn't be analyzed visually ({provider} has no "
|
|
2495
|
+
"native vision transport here). Configure one via /model.")
|
|
2496
|
+
used_config = vcfg
|
|
2497
|
+
info = PROVIDERS.get(used_config.get("provider"), {})
|
|
2498
|
+
api_style = info["api_style"] if info else "openai"
|
|
2499
|
+
|
|
2500
|
+
if provider == "ollama":
|
|
2501
|
+
base_url = config.get("ollama_url") or (info["base_url"] if info else "http://localhost:11434")
|
|
2502
|
+
else:
|
|
2503
|
+
base_url = used_config.get("base_url") or used_config.get("api_url") or (info["base_url"] if info else "")
|
|
2504
|
+
base_url = base_url.rstrip("/")
|
|
2505
|
+
api_key = used_config.get("api_key", "")
|
|
2506
|
+
model = used_config.get("model") or (info["default_model"] if info else "gpt-4o-mini")
|
|
2507
|
+
extra_headers = info["extra_headers"] if info else {}
|
|
2508
|
+
|
|
2509
|
+
try:
|
|
2510
|
+
if api_style == "openai":
|
|
2511
|
+
url = f"{base_url}/chat/completions"
|
|
2512
|
+
headers = {"Content-Type": "application/json"}
|
|
2513
|
+
if api_key:
|
|
2514
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
2515
|
+
headers.update(extra_headers)
|
|
2516
|
+
payload = {
|
|
2517
|
+
"model": model,
|
|
2518
|
+
"messages": [
|
|
2519
|
+
{"role": "system", "content": system_prompt},
|
|
2520
|
+
{"role": "user", "content": [
|
|
2521
|
+
{"type": "text", "text": prompt},
|
|
2522
|
+
{"type": "image_url", "image_url": {"url": f"data:{mime};base64,{b64}"}},
|
|
2523
|
+
]},
|
|
2524
|
+
],
|
|
2525
|
+
}
|
|
2526
|
+
resp = requests.post(url, headers=headers, json=payload, timeout=60)
|
|
2527
|
+
_raise_for_status(resp)
|
|
2528
|
+
data = resp.json()
|
|
2529
|
+
content = data['choices'][0]['message']['content']
|
|
2530
|
+
_track_usage_from_data("openai", data, prompt, content)
|
|
2531
|
+
return content
|
|
2532
|
+
|
|
2533
|
+
elif api_style == "anthropic":
|
|
2534
|
+
headers = {"x-api-key": api_key, "anthropic-version": "2023-06-01",
|
|
2535
|
+
"content-type": "application/json"}
|
|
2536
|
+
headers.update(extra_headers)
|
|
2537
|
+
payload = {
|
|
2538
|
+
"model": model, "max_tokens": 1024, "system": system_prompt,
|
|
2539
|
+
"messages": [{"role": "user", "content": [
|
|
2540
|
+
{"type": "image", "source": {"type": "base64", "media_type": mime, "data": b64}},
|
|
2541
|
+
{"type": "text", "text": prompt},
|
|
2542
|
+
]}],
|
|
2543
|
+
}
|
|
2544
|
+
resp = requests.post(f"{base_url}/messages", headers=headers, json=payload, timeout=60)
|
|
2545
|
+
_raise_for_status(resp)
|
|
2546
|
+
data = resp.json()
|
|
2547
|
+
content = data['content'][0]['text']
|
|
2548
|
+
_track_usage_from_data("anthropic", data, prompt, content)
|
|
2549
|
+
return content
|
|
2550
|
+
|
|
2551
|
+
elif api_style == "gemini":
|
|
2552
|
+
url = f"{base_url}/models/{model}:generateContent?key={api_key}"
|
|
2553
|
+
payload = {
|
|
2554
|
+
"contents": [{"role": "user", "parts": [
|
|
2555
|
+
{"text": prompt},
|
|
2556
|
+
{"inline_data": {"mime_type": mime, "data": b64}},
|
|
2557
|
+
]}],
|
|
2558
|
+
"systemInstruction": {"parts": [{"text": system_prompt}]},
|
|
2559
|
+
}
|
|
2560
|
+
resp = requests.post(url, json=payload, timeout=60)
|
|
2561
|
+
_raise_for_status(resp)
|
|
2562
|
+
data = resp.json()
|
|
2563
|
+
content = data['candidates'][0]['content']['parts'][0]['text']
|
|
2564
|
+
_track_usage_from_data("gemini", data, prompt, content)
|
|
2565
|
+
return content
|
|
2566
|
+
|
|
2567
|
+
except requests.exceptions.ConnectionError:
|
|
2568
|
+
return "Could not reach the AI server for image analysis. Check your internet/API URL."
|
|
2569
|
+
except requests.exceptions.Timeout:
|
|
2570
|
+
return "The image analysis request timed out. Try again, or use a faster model."
|
|
2571
|
+
except Exception as e:
|
|
2572
|
+
return f"Error analyzing image: {e}"
|