cct-cli 0.7.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- calc_terminal/__init__.py +14 -0
- calc_terminal/__main__.py +14 -0
- calc_terminal/activity.py +1334 -0
- calc_terminal/agent.py +3387 -0
- calc_terminal/agent_runtime.py +519 -0
- calc_terminal/ai_context.py +447 -0
- calc_terminal/ai_modes.py +752 -0
- calc_terminal/ai_personalization.py +286 -0
- calc_terminal/ai_preview_feedback.py +213 -0
- calc_terminal/aicore.py +2572 -0
- calc_terminal/anim.py +367 -0
- calc_terminal/app.py +3685 -0
- calc_terminal/art.py +639 -0
- calc_terminal/atomsim.py +368 -0
- calc_terminal/attachments.py +743 -0
- calc_terminal/benchmark_system.py +414 -0
- calc_terminal/browser/__init__.py +36 -0
- calc_terminal/browser/browser_state.py +346 -0
- calc_terminal/browser/devserver.py +176 -0
- calc_terminal/browser/engine.py +494 -0
- calc_terminal/browser/navigation.py +84 -0
- calc_terminal/browser/preview.py +429 -0
- calc_terminal/browser/preview_entry.py +95 -0
- calc_terminal/browser/project_detector.py +144 -0
- calc_terminal/browser/server.py +449 -0
- calc_terminal/browser/state.py +75 -0
- calc_terminal/browser/watcher.py +99 -0
- calc_terminal/browser_gui/__init__.py +1 -0
- calc_terminal/browser_gui/__main__.py +3 -0
- calc_terminal/browser_gui/launcher.py +173 -0
- calc_terminal/browser_gui/playwright_browser.py +117 -0
- calc_terminal/browser_gui/qt_browser.py +1501 -0
- calc_terminal/browser_gui/webview_browser.py +57 -0
- calc_terminal/capabilities/__init__.py +35 -0
- calc_terminal/capabilities/adapters/__init__.py +33 -0
- calc_terminal/capabilities/adapters/bioinformatics.py +204 -0
- calc_terminal/capabilities/adapters/browser_adapter.py +205 -0
- calc_terminal/capabilities/adapters/filesystem.py +206 -0
- calc_terminal/capabilities/adapters/git_adapter.py +202 -0
- calc_terminal/capabilities/adapters/jupyter_adapter.py +138 -0
- calc_terminal/capabilities/adapters/ml_frameworks.py +158 -0
- calc_terminal/capabilities/adapters/platforms.py +200 -0
- calc_terminal/capabilities/adapters/python_exec.py +93 -0
- calc_terminal/capabilities/adapters/quantum_adapter.py +150 -0
- calc_terminal/capabilities/adapters/scientific_comp.py +123 -0
- calc_terminal/capabilities/adapters/structural_bio.py +161 -0
- calc_terminal/capabilities/adapters/terminal.py +99 -0
- calc_terminal/capabilities/bus.py +178 -0
- calc_terminal/capabilities/discovery.py +207 -0
- calc_terminal/capabilities/schema.py +221 -0
- calc_terminal/cat.ico +0 -0
- calc_terminal/cat_browser.py +2018 -0
- calc_terminal/chat_store.py +703 -0
- calc_terminal/cli.py +1178 -0
- calc_terminal/code_editor.py +640 -0
- calc_terminal/collaboration.py +723 -0
- calc_terminal/commands_data.py +139 -0
- calc_terminal/compatibility_engine.py +352 -0
- calc_terminal/compute/__init__.py +31 -0
- calc_terminal/compute/fabric.py +350 -0
- calc_terminal/config.py +227 -0
- calc_terminal/core/__init__.py +41 -0
- calc_terminal/core/checkpoint.py +156 -0
- calc_terminal/core/input/__init__.py +45 -0
- calc_terminal/core/mode_registry.py +300 -0
- calc_terminal/core/project_graph.py +172 -0
- calc_terminal/core/recovery.py +129 -0
- calc_terminal/core/security_layer.py +112 -0
- calc_terminal/core/task_graph.py +202 -0
- calc_terminal/core/unified_runtime.py +184 -0
- calc_terminal/core/verification.py +257 -0
- calc_terminal/customization.py +1566 -0
- calc_terminal/derivations.py +153 -0
- calc_terminal/device_control.py +263 -0
- calc_terminal/diagnostics/__init__.py +27 -0
- calc_terminal/diagnostics/doctor_engine.py +382 -0
- calc_terminal/diagnostics/self_test.py +247 -0
- calc_terminal/doctor.py +519 -0
- calc_terminal/easter_eggs.py +274 -0
- calc_terminal/editor/__init__.py +1 -0
- calc_terminal/editor/actions.py +263 -0
- calc_terminal/editor/commands.py +160 -0
- calc_terminal/editor/shortcuts.py +226 -0
- calc_terminal/engine.py +259 -0
- calc_terminal/errors.py +120 -0
- calc_terminal/event_stream.py +146 -0
- calc_terminal/eventbus.py +133 -0
- calc_terminal/extensions.py +733 -0
- calc_terminal/fallback_cli.py +1321 -0
- calc_terminal/first_run.py +265 -0
- calc_terminal/fomoji_auth.py +1043 -0
- calc_terminal/formulas.py +82 -0
- calc_terminal/fs_cache.py +121 -0
- calc_terminal/fs_watcher.py +277 -0
- calc_terminal/game.py +193 -0
- calc_terminal/gen1.py +5 -0
- calc_terminal/generators.py +245 -0
- calc_terminal/gestures/__init__.py +42 -0
- calc_terminal/gestures/bindings.py +175 -0
- calc_terminal/gestures/manager.py +477 -0
- calc_terminal/goodbye.py +363 -0
- calc_terminal/gpu3d.py +290 -0
- calc_terminal/graphs.py +358 -0
- calc_terminal/hardware_analyzer.py +440 -0
- calc_terminal/host/__init__.py +30 -0
- calc_terminal/host/browser_manager.py +187 -0
- calc_terminal/host/desktop.py +1386 -0
- calc_terminal/host/launcher.py +395 -0
- calc_terminal/host/terminal.py +279 -0
- calc_terminal/identity.py +216 -0
- calc_terminal/input/__init__.py +54 -0
- calc_terminal/input/capabilities.py +258 -0
- calc_terminal/input/focus.py +87 -0
- calc_terminal/input/gestures.py +64 -0
- calc_terminal/input/pointer.py +114 -0
- calc_terminal/input/touch.py +345 -0
- calc_terminal/keys.py +84 -0
- calc_terminal/live_automation.py +165 -0
- calc_terminal/mathtext.py +433 -0
- calc_terminal/mcp.py +386 -0
- calc_terminal/memory.py +337 -0
- calc_terminal/memory_v2.py +479 -0
- calc_terminal/metrics.py +333 -0
- calc_terminal/mode_detection.py +146 -0
- calc_terminal/model.py +2431 -0
- calc_terminal/model_router.py +665 -0
- calc_terminal/models/__init__.py +0 -0
- calc_terminal/models/active_state.py +187 -0
- calc_terminal/models/dynamic_registry.py +584 -0
- calc_terminal/models/manager.py +781 -0
- calc_terminal/models/model_metadata.json +3526 -0
- calc_terminal/models/profiles.py +194 -0
- calc_terminal/models/registry.py +265 -0
- calc_terminal/models/schema.py +197 -0
- calc_terminal/models/validator.py +287 -0
- calc_terminal/models/verification_engine.py +368 -0
- calc_terminal/native_picker.py +215 -0
- calc_terminal/ollama_catalog.py +279 -0
- calc_terminal/ollama_download.py +233 -0
- calc_terminal/orchestrator.py +304 -0
- calc_terminal/package_research.py +322 -0
- calc_terminal/packages.py +1024 -0
- calc_terminal/pc_specs.py +116 -0
- calc_terminal/permissions.py +334 -0
- calc_terminal/pet.py +106 -0
- calc_terminal/pipeline.py +505 -0
- calc_terminal/platform/__init__.py +491 -0
- calc_terminal/platform/desktop.py +491 -0
- calc_terminal/platform/web.py +781 -0
- calc_terminal/preview/__init__.py +1 -0
- calc_terminal/preview/dev_server.py +303 -0
- calc_terminal/preview/diagnostics.py +131 -0
- calc_terminal/preview/live_reload.py +66 -0
- calc_terminal/preview/manager.py +129 -0
- calc_terminal/project_stats.py +209 -0
- calc_terminal/projects.py +328 -0
- calc_terminal/providers/__init__.py +0 -0
- calc_terminal/providers/adapters/__init__.py +80 -0
- calc_terminal/providers/adapters/anthropic_adapter.py +127 -0
- calc_terminal/providers/adapters/base.py +106 -0
- calc_terminal/providers/adapters/chinese_adapters.py +420 -0
- calc_terminal/providers/adapters/gemini_adapter.py +101 -0
- calc_terminal/providers/adapters/ollama_adapter.py +83 -0
- calc_terminal/providers/adapters/openai_adapter.py +159 -0
- calc_terminal/providers/adapters/other_adapters.py +246 -0
- calc_terminal/providers/anthropic_provider.py +172 -0
- calc_terminal/providers/auto_update.py +416 -0
- calc_terminal/providers/base_provider.py +105 -0
- calc_terminal/providers/discovery_manager.py +207 -0
- calc_terminal/providers/gemini_provider.py +178 -0
- calc_terminal/providers/lifecycle.py +767 -0
- calc_terminal/providers/ollama_adapter.py +707 -0
- calc_terminal/providers/openai_provider.py +254 -0
- calc_terminal/providers/provider_manager.py +1827 -0
- calc_terminal/providers/providers.json +4075 -0
- calc_terminal/reactionsim.py +279 -0
- calc_terminal/registry.py +337 -0
- calc_terminal/report.py +162 -0
- calc_terminal/research/__init__.py +45 -0
- calc_terminal/research/artifact_intel.py +126 -0
- calc_terminal/research/data_lineage.py +123 -0
- calc_terminal/research/experiment_ledger.py +303 -0
- calc_terminal/research/reproducibility.py +131 -0
- calc_terminal/resilience/__init__.py +47 -0
- calc_terminal/resilience/agent_state.py +121 -0
- calc_terminal/resilience/capability_matcher.py +174 -0
- calc_terminal/resilience/circuit_breaker.py +158 -0
- calc_terminal/resilience/failover_engine.py +230 -0
- calc_terminal/resilience/health_monitor.py +192 -0
- calc_terminal/resilience/ollama_adapter.py +125 -0
- calc_terminal/resilience/orchestrator.py +312 -0
- calc_terminal/resilience/types.py +134 -0
- calc_terminal/sandbox.py +98 -0
- calc_terminal/scires.py +558 -0
- calc_terminal/security_scanner.py +126 -0
- calc_terminal/session.py +294 -0
- calc_terminal/sim3d.py +206 -0
- calc_terminal/solver.py +276 -0
- calc_terminal/sound.py +127 -0
- calc_terminal/task_reports.py +287 -0
- calc_terminal/terminal_host.py +201 -0
- calc_terminal/terminal_identity.py +411 -0
- calc_terminal/test_ai_mode_reliability.py +344 -0
- calc_terminal/test_browser.py +368 -0
- calc_terminal/test_code_editor_upgrade.py +485 -0
- calc_terminal/test_customization.py +1148 -0
- calc_terminal/test_customization_ui.py +612 -0
- calc_terminal/test_dynamic_registry.py +304 -0
- calc_terminal/test_extensions.py +436 -0
- calc_terminal/test_overhaul.py +557 -0
- calc_terminal/test_project_detect.py +255 -0
- calc_terminal/test_root_cause_fix.py +527 -0
- calc_terminal/test_stability.py +532 -0
- calc_terminal/test_terminal_identity.py +132 -0
- calc_terminal/test_v079_speed.py +460 -0
- calc_terminal/theme.py +1107 -0
- calc_terminal/timeline.py +139 -0
- calc_terminal/todos.py +246 -0
- calc_terminal/tool_call_normalizer.py +419 -0
- calc_terminal/tui.py +104 -0
- calc_terminal/ui/__init__.py +8 -0
- calc_terminal/ui/activity_panel.py +231 -0
- calc_terminal/ui/activity_stream_panel.py +238 -0
- calc_terminal/ui/animations.py +122 -0
- calc_terminal/ui/app.py +7271 -0
- calc_terminal/ui/attach_panel.py +597 -0
- calc_terminal/ui/attachments.py +424 -0
- calc_terminal/ui/backup_panel.py +810 -0
- calc_terminal/ui/browser_shell.py +887 -0
- calc_terminal/ui/cat_agent.py +357 -0
- calc_terminal/ui/chats_panel.py +899 -0
- calc_terminal/ui/command_palette.py +125 -0
- calc_terminal/ui/command_palette_modal.py +166 -0
- calc_terminal/ui/composer.py +1141 -0
- calc_terminal/ui/context_menu.py +197 -0
- calc_terminal/ui/conversation.py +1435 -0
- calc_terminal/ui/customization_panel.py +1229 -0
- calc_terminal/ui/dashboard.py +404 -0
- calc_terminal/ui/design_system.py +557 -0
- calc_terminal/ui/diff_panel.py +213 -0
- calc_terminal/ui/editor.py +2102 -0
- calc_terminal/ui/empty_state.py +302 -0
- calc_terminal/ui/events.py +487 -0
- calc_terminal/ui/extensions_panel.py +815 -0
- calc_terminal/ui/footer.py +166 -0
- calc_terminal/ui/gestures_panel.py +383 -0
- calc_terminal/ui/goodbye_screen.py +100 -0
- calc_terminal/ui/header.py +1034 -0
- calc_terminal/ui/help_panel.py +254 -0
- calc_terminal/ui/live_activities.py +914 -0
- calc_terminal/ui/mcp_panel.py +570 -0
- calc_terminal/ui/memory_center.py +524 -0
- calc_terminal/ui/mode_colors_panel.py +525 -0
- calc_terminal/ui/nav_screens.py +747 -0
- calc_terminal/ui/ollama_panel.py +536 -0
- calc_terminal/ui/palette.py +221 -0
- calc_terminal/ui/permission_panel.py +269 -0
- calc_terminal/ui/personalization_panel.py +517 -0
- calc_terminal/ui/personalize_center.py +1568 -0
- calc_terminal/ui/preview_panel.py +441 -0
- calc_terminal/ui/resizers.py +402 -0
- calc_terminal/ui/sidebar.py +1285 -0
- calc_terminal/ui/statusbar.py +168 -0
- calc_terminal/ui/theme_css.py +1396 -0
- calc_terminal/ui/thinking.py +226 -0
- calc_terminal/ui/timeline_panel.py +102 -0
- calc_terminal/ui/todo_panel.py +193 -0
- calc_terminal/ui/viewport.py +136 -0
- calc_terminal/ui/vision_panel.py +489 -0
- calc_terminal/ui/welcome_modal.py +343 -0
- calc_terminal/ui/widgets.py +160 -0
- calc_terminal/ui/workspace.py +831 -0
- calc_terminal/viewers/__init__.py +1 -0
- calc_terminal/viewers/document_viewer.py +252 -0
- calc_terminal/viewers/image_viewer.py +241 -0
- calc_terminal/viewers/pdf_viewer.py +203 -0
- calc_terminal/viewers/presentation_viewer.py +164 -0
- calc_terminal/viewers/registry.py +120 -0
- calc_terminal/viewers/spreadsheet_viewer.py +204 -0
- calc_terminal/vision/__init__.py +89 -0
- calc_terminal/vision/analysis.py +194 -0
- calc_terminal/vision/annotations.py +297 -0
- calc_terminal/vision/capture.py +171 -0
- calc_terminal/vision/context.py +231 -0
- calc_terminal/vision/correlation.py +169 -0
- calc_terminal/vision/cursor.py +258 -0
- calc_terminal/vision/events.py +66 -0
- calc_terminal/vision/frame_pipeline.py +259 -0
- calc_terminal/vision/priority.py +218 -0
- calc_terminal/vision/provider.py +180 -0
- calc_terminal/vision/safety.py +149 -0
- calc_terminal/vision/session.py +281 -0
- calc_terminal/vision/verify.py +162 -0
- calc_terminal/vision.py +514 -0
- calc_terminal/vscode_integration.py +113 -0
- calc_terminal/web/__init__.py +8 -0
- calc_terminal/web/cat_runtime.py +710 -0
- calc_terminal/web/server.py +2891 -0
- calc_terminal/web/static/css/app.css +3152 -0
- calc_terminal/web/static/icons/badge-72.png +0 -0
- calc_terminal/web/static/icons/cat.ico +0 -0
- calc_terminal/web/static/icons/icon-128.png +0 -0
- calc_terminal/web/static/icons/icon-144.png +0 -0
- calc_terminal/web/static/icons/icon-152.png +0 -0
- calc_terminal/web/static/icons/icon-192.png +0 -0
- calc_terminal/web/static/icons/icon-384.png +0 -0
- calc_terminal/web/static/icons/icon-512.png +0 -0
- calc_terminal/web/static/icons/icon-72.png +0 -0
- calc_terminal/web/static/icons/icon-96.png +0 -0
- calc_terminal/web/static/icons/icon.svg +34 -0
- calc_terminal/web/static/icons/new-project.png +0 -0
- calc_terminal/web/static/icons/open-project.png +0 -0
- calc_terminal/web/static/index.html +734 -0
- calc_terminal/web/static/js/app.js +2403 -0
- calc_terminal/web/static/manifest.json +88 -0
- calc_terminal/web/static/sw.js +230 -0
- calc_terminal/workflow_engine.py +769 -0
- calc_terminal/workspace.py +593 -0
- calc_terminal/workspace_index.py +385 -0
- cct_cli-0.7.9.0.dist-info/METADATA +210 -0
- cct_cli-0.7.9.0.dist-info/RECORD +325 -0
- cct_cli-0.7.9.0.dist-info/WHEEL +5 -0
- cct_cli-0.7.9.0.dist-info/entry_points.txt +4 -0
- cct_cli-0.7.9.0.dist-info/licenses/LICENSE +21 -0
- cct_cli-0.7.9.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,707 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Ollama Provider Adapter, Streaming Protocol & Repetition Guard for CAT.
|
|
3
|
+
|
|
4
|
+
Implements Sections 6-10, 12, 24, 25, 28, 30, 31, and 34 of the
|
|
5
|
+
CAT AI Mode Reliability Contract.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import logging
|
|
10
|
+
import re
|
|
11
|
+
import threading
|
|
12
|
+
import time
|
|
13
|
+
import uuid
|
|
14
|
+
from typing import Generator, Dict, Any, Optional, List, Tuple
|
|
15
|
+
|
|
16
|
+
_LOG = logging.getLogger("cct.ollama_adapter")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class RequestState:
|
|
20
|
+
IDLE = "idle"
|
|
21
|
+
QUEUED = "queued"
|
|
22
|
+
STARTING = "starting"
|
|
23
|
+
STREAMING = "streaming"
|
|
24
|
+
COMPLETED = "completed"
|
|
25
|
+
CANCELLED = "cancelled"
|
|
26
|
+
FAILED = "failed"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class GenerationGuard:
|
|
30
|
+
"""Detects abnormal generation loops and repetitions in small/base models.
|
|
31
|
+
|
|
32
|
+
Prevents runaway output such as:
|
|
33
|
+
- 'hihihihihihihihihi'
|
|
34
|
+
- 'ahahahaha...'
|
|
35
|
+
- Repeated identical sentences or phrases.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
def __init__(self, max_repeated_chars: int = 24, max_sentence_repeats: int = 3):
|
|
39
|
+
self.max_repeated_chars = max_repeated_chars
|
|
40
|
+
self.max_sentence_repeats = max_sentence_repeats
|
|
41
|
+
self.buffer = ""
|
|
42
|
+
self._recent_sentences: List[str] = []
|
|
43
|
+
self.repetition_detected = False
|
|
44
|
+
self.repetition_reason = ""
|
|
45
|
+
|
|
46
|
+
def feed(self, chunk: str) -> Tuple[bool, str]:
|
|
47
|
+
"""Feed a streamed text chunk into the guard.
|
|
48
|
+
|
|
49
|
+
Returns:
|
|
50
|
+
(should_continue, clean_chunk)
|
|
51
|
+
If repetition is detected, should_continue is False and
|
|
52
|
+
a stopping notice is returned.
|
|
53
|
+
"""
|
|
54
|
+
if not chunk or self.repetition_detected:
|
|
55
|
+
return False, ""
|
|
56
|
+
|
|
57
|
+
self.buffer += chunk
|
|
58
|
+
|
|
59
|
+
# 1. Check for character-level runaway loops (e.g. 'hihihihihihi' or 'ahahahaha')
|
|
60
|
+
# Matches any 1-6 char alphanumeric pattern repeated more than 8 times
|
|
61
|
+
loop_match = re.search(r"(.{1,6}?)\1{8,}", self.buffer[-100:])
|
|
62
|
+
if loop_match:
|
|
63
|
+
repeated_sub = loop_match.group(1)
|
|
64
|
+
# Only trigger on repeating letters or digits (e.g. 'hi', 'ha', 'abc').
|
|
65
|
+
# Markdown dividers ('---', '==='), code banners ('/* **** */'), and table rules
|
|
66
|
+
# are legitimate syntax and must NOT be treated as runaway loops.
|
|
67
|
+
if re.search(r"[a-zA-Z0-9]", repeated_sub) and not repeated_sub.isspace():
|
|
68
|
+
match_span = loop_match.group(0)
|
|
69
|
+
if len(match_span) >= self.max_repeated_chars:
|
|
70
|
+
self.repetition_detected = True
|
|
71
|
+
self.repetition_reason = f"Repeating pattern detected: '{repeated_sub}'"
|
|
72
|
+
_LOG.warning("GenerationGuard halted stream: %s", self.repetition_reason)
|
|
73
|
+
return False, "\n\n*(Generation stopped: repetitive output pattern detected)*"
|
|
74
|
+
|
|
75
|
+
# 2. Check for consecutive sentence-level repetitions (e.g. repeating the same prose sentence 3 times in a row)
|
|
76
|
+
sentences = [s.strip() for s in re.split(r"[.!?\n]+", self.buffer) if len(s.strip()) > 15]
|
|
77
|
+
if len(sentences) >= 3:
|
|
78
|
+
s1, s2, s3 = sentences[-1].lower(), sentences[-2].lower(), sentences[-3].lower()
|
|
79
|
+
# Ignore markdown list items, code comments, and code keywords which naturally repeat
|
|
80
|
+
is_code_or_list = any(s1.startswith(p) for p in ("- ", "* ", "#", "//", "/*", "return", "import", "def ", "class ", "print"))
|
|
81
|
+
if not is_code_or_list and s1 == s2 == s3:
|
|
82
|
+
self.repetition_detected = True
|
|
83
|
+
self.repetition_reason = f"Repeating sentence: '{s1[:40]}...'"
|
|
84
|
+
_LOG.warning("GenerationGuard halted stream: %s", self.repetition_reason)
|
|
85
|
+
return False, "\n\n*(Generation stopped: sentence repetition loop detected)*"
|
|
86
|
+
|
|
87
|
+
return True, chunk
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class OutputSanitizer:
|
|
91
|
+
"""Sanitizes generated tokens to remove stray protocol leakage without destroying code or markdown."""
|
|
92
|
+
|
|
93
|
+
# Leading prefixes accidentally generated by base models
|
|
94
|
+
_PREFIX_PATTERNS = [
|
|
95
|
+
re.compile(r"^\s*Assistant\s*:\s*", re.IGNORECASE),
|
|
96
|
+
re.compile(r"^\s*CAT\s*AI\s*:\s*", re.IGNORECASE),
|
|
97
|
+
re.compile(r"^\s*Response\s*:\s*", re.IGNORECASE),
|
|
98
|
+
]
|
|
99
|
+
|
|
100
|
+
@classmethod
|
|
101
|
+
def sanitize_first_chunk(cls, text: str) -> str:
|
|
102
|
+
"""Remove accidental assistant prefix from the beginning of a generation."""
|
|
103
|
+
s = text
|
|
104
|
+
for p in cls._PREFIX_PATTERNS:
|
|
105
|
+
s = p.sub("", s)
|
|
106
|
+
return s
|
|
107
|
+
|
|
108
|
+
@classmethod
|
|
109
|
+
def sanitize_final(cls, full_text: str) -> str:
|
|
110
|
+
"""Clean any remaining transport wrappers or trailing raw prompts."""
|
|
111
|
+
cleaned = full_text
|
|
112
|
+
for p in cls._PREFIX_PATTERNS:
|
|
113
|
+
cleaned = p.sub("", cleaned)
|
|
114
|
+
for stop_p in ("<|imend|>", "<|im_end|>", "<|endoftext|>", "<|eot_id|>", "<|end_of_text|>"):
|
|
115
|
+
cleaned = cleaned.replace(stop_p, "")
|
|
116
|
+
# Remove trailing unclosed conversation prompts if model hallucinated another turn
|
|
117
|
+
cleaned = re.sub(r"\n\s*User:\s*$", "", cleaned, flags=re.IGNORECASE)
|
|
118
|
+
cleaned = re.sub(r"\n\s*Human:\s*$", "", cleaned, flags=re.IGNORECASE)
|
|
119
|
+
return cleaned.strip()
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class ThinkingStreamFilter:
|
|
123
|
+
"""Separates reasoning/thinking tokens (<think>...</think> or message.thinking)
|
|
124
|
+
from final response text, streaming thoughts to live activities in real time
|
|
125
|
+
while ensuring only post-thinking answer text reaches the chat response bubble."""
|
|
126
|
+
|
|
127
|
+
def __init__(self, on_thinking_chunk, on_thinking_end):
|
|
128
|
+
self.on_thinking_chunk = on_thinking_chunk
|
|
129
|
+
self.on_thinking_end = on_thinking_end
|
|
130
|
+
self.in_thinking = False
|
|
131
|
+
self.buffer = ""
|
|
132
|
+
|
|
133
|
+
def process(self, thinking_part: str, content_part: str) -> str:
|
|
134
|
+
# If model explicitly provides thinking_part (e.g. Ollama 0.5.8+ with DeepSeek-R1)
|
|
135
|
+
if thinking_part:
|
|
136
|
+
if not self.in_thinking:
|
|
137
|
+
self.in_thinking = True
|
|
138
|
+
self.on_thinking_chunk(thinking_part)
|
|
139
|
+
if not content_part:
|
|
140
|
+
return ""
|
|
141
|
+
|
|
142
|
+
# Check if we were in thinking_part and now content arrived without thinking_part
|
|
143
|
+
if self.in_thinking and not thinking_part and content_part and "<think>" not in content_part and "</think>" not in content_part and not self.buffer:
|
|
144
|
+
self.in_thinking = False
|
|
145
|
+
self.on_thinking_end()
|
|
146
|
+
|
|
147
|
+
# Process content_part for embedded <think>...</think> tags
|
|
148
|
+
if not content_part:
|
|
149
|
+
return ""
|
|
150
|
+
|
|
151
|
+
text = self.buffer + content_part
|
|
152
|
+
self.buffer = ""
|
|
153
|
+
output_tokens = []
|
|
154
|
+
|
|
155
|
+
while text:
|
|
156
|
+
if not self.in_thinking:
|
|
157
|
+
idx = text.find("<think>")
|
|
158
|
+
if idx != -1:
|
|
159
|
+
if idx > 0:
|
|
160
|
+
output_tokens.append(text[:idx])
|
|
161
|
+
self.in_thinking = True
|
|
162
|
+
text = text[idx + 7:]
|
|
163
|
+
else:
|
|
164
|
+
partial_match = False
|
|
165
|
+
for i in range(1, 7):
|
|
166
|
+
if text.endswith("<think>"[:i]):
|
|
167
|
+
self.buffer = text[-i:]
|
|
168
|
+
output_tokens.append(text[:-i])
|
|
169
|
+
partial_match = True
|
|
170
|
+
text = ""
|
|
171
|
+
break
|
|
172
|
+
if not partial_match:
|
|
173
|
+
output_tokens.append(text)
|
|
174
|
+
text = ""
|
|
175
|
+
else:
|
|
176
|
+
idx = text.find("</think>")
|
|
177
|
+
if idx != -1:
|
|
178
|
+
thought = text[:idx]
|
|
179
|
+
if thought:
|
|
180
|
+
self.on_thinking_chunk(thought)
|
|
181
|
+
self.in_thinking = False
|
|
182
|
+
self.on_thinking_end()
|
|
183
|
+
text = text[idx + 8:]
|
|
184
|
+
else:
|
|
185
|
+
partial_match = False
|
|
186
|
+
for i in range(1, 8):
|
|
187
|
+
if text.endswith("</think>"[:i]):
|
|
188
|
+
self.buffer = text[-i:]
|
|
189
|
+
thought = text[:-i]
|
|
190
|
+
if thought:
|
|
191
|
+
self.on_thinking_chunk(thought)
|
|
192
|
+
partial_match = True
|
|
193
|
+
text = ""
|
|
194
|
+
break
|
|
195
|
+
if not partial_match:
|
|
196
|
+
self.on_thinking_chunk(text)
|
|
197
|
+
text = ""
|
|
198
|
+
|
|
199
|
+
return "".join(output_tokens)
|
|
200
|
+
|
|
201
|
+
def finalize(self) -> str:
|
|
202
|
+
res = ""
|
|
203
|
+
if self.in_thinking:
|
|
204
|
+
self.in_thinking = False
|
|
205
|
+
if self.buffer:
|
|
206
|
+
self.on_thinking_chunk(self.buffer)
|
|
207
|
+
self.on_thinking_end()
|
|
208
|
+
elif self.buffer:
|
|
209
|
+
res = self.buffer
|
|
210
|
+
self.buffer = ""
|
|
211
|
+
return res
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
class OllamaProvider:
|
|
215
|
+
"""Dedicated provider adapter for local Ollama instances.
|
|
216
|
+
|
|
217
|
+
Guarantees:
|
|
218
|
+
- Request isolation via unique request IDs.
|
|
219
|
+
- Stream reset before each generation.
|
|
220
|
+
- Safe cancellation without memory leaks or cross-request pollution.
|
|
221
|
+
- Structured /api/chat invocation with error handling.
|
|
222
|
+
- Repetition detection via GenerationGuard.
|
|
223
|
+
"""
|
|
224
|
+
|
|
225
|
+
def __init__(self, base_url: str = "http://localhost:11434"):
|
|
226
|
+
self.base_url = base_url.rstrip("/")
|
|
227
|
+
self._lock = threading.Lock()
|
|
228
|
+
self._active_request_id: Optional[str] = None
|
|
229
|
+
self._active_response = None
|
|
230
|
+
self._state: str = RequestState.IDLE
|
|
231
|
+
self._active_thinking_act_id: Optional[str] = None
|
|
232
|
+
|
|
233
|
+
@property
|
|
234
|
+
def state(self) -> str:
|
|
235
|
+
return self._state
|
|
236
|
+
|
|
237
|
+
def reset_stream(self, new_request_id: Optional[str] = None) -> str:
|
|
238
|
+
"""Reset output buffer and allocate new request ID."""
|
|
239
|
+
with self._lock:
|
|
240
|
+
# Cancel any existing active socket
|
|
241
|
+
if self._active_response is not None:
|
|
242
|
+
try:
|
|
243
|
+
self._active_response.close()
|
|
244
|
+
except Exception:
|
|
245
|
+
pass
|
|
246
|
+
self._active_response = None
|
|
247
|
+
|
|
248
|
+
self._active_request_id = new_request_id or uuid.uuid4().hex
|
|
249
|
+
self._state = RequestState.STARTING
|
|
250
|
+
return self._active_request_id
|
|
251
|
+
|
|
252
|
+
def cancel_active(self) -> bool:
|
|
253
|
+
"""Explicitly cancel current in-flight generation."""
|
|
254
|
+
with self._lock:
|
|
255
|
+
self._state = RequestState.CANCELLED
|
|
256
|
+
if getattr(self, "_active_thinking_act_id", None):
|
|
257
|
+
try:
|
|
258
|
+
from .. import activity as act_mod
|
|
259
|
+
act_mod.manager.update(self._active_thinking_act_id, status=act_mod.STATUS_CANCELLED, result="Cancelled")
|
|
260
|
+
except Exception:
|
|
261
|
+
pass
|
|
262
|
+
self._active_thinking_act_id = None
|
|
263
|
+
if self._active_response is not None:
|
|
264
|
+
try:
|
|
265
|
+
from ..aicore import _unregister_response
|
|
266
|
+
_unregister_response(self._active_response)
|
|
267
|
+
except Exception:
|
|
268
|
+
pass
|
|
269
|
+
try:
|
|
270
|
+
self._active_response.close()
|
|
271
|
+
except Exception:
|
|
272
|
+
pass
|
|
273
|
+
self._active_response = None
|
|
274
|
+
# Asynchronously ping Ollama to abort slot generation immediately
|
|
275
|
+
def _abort_slot():
|
|
276
|
+
try:
|
|
277
|
+
import requests
|
|
278
|
+
requests.post(f"{self.base_url}/api/generate", json={"model": "", "keep_alive": 0}, timeout=0.6)
|
|
279
|
+
except Exception:
|
|
280
|
+
pass
|
|
281
|
+
threading.Thread(target=_abort_slot, daemon=True).start()
|
|
282
|
+
return True
|
|
283
|
+
return False
|
|
284
|
+
|
|
285
|
+
def stream_chat(
|
|
286
|
+
self,
|
|
287
|
+
model: str,
|
|
288
|
+
messages: List[Dict[str, str]],
|
|
289
|
+
options: Optional[Dict[str, Any]] = None,
|
|
290
|
+
timeout: Tuple[float, float] = (90.0, 180.0),
|
|
291
|
+
request_id: Optional[str] = None,
|
|
292
|
+
deadline: Optional[float] = None,
|
|
293
|
+
) -> Generator[str, None, None]:
|
|
294
|
+
"""Stream chat tokens from Ollama's /api/chat endpoint."""
|
|
295
|
+
import requests
|
|
296
|
+
import os
|
|
297
|
+
|
|
298
|
+
req_id = self.reset_stream(request_id)
|
|
299
|
+
guard = GenerationGuard()
|
|
300
|
+
sanitizer = OutputSanitizer()
|
|
301
|
+
is_first_chunk = True
|
|
302
|
+
in_thinking = False
|
|
303
|
+
|
|
304
|
+
# Build payload according to Ollama /api/chat schema
|
|
305
|
+
eff_options = dict(options or {})
|
|
306
|
+
# Dynamically scale num_ctx based on input messages:
|
|
307
|
+
# For standard short/normal queries (<1500 tokens), default to lean 2048 context
|
|
308
|
+
# to maximize GPU offloading and ensure 4x faster CPU inference.
|
|
309
|
+
total_chars = sum(len(m.get("content", "")) for m in messages if isinstance(m, dict))
|
|
310
|
+
est_tokens = int(total_chars / 3.5)
|
|
311
|
+
if est_tokens >= 1500:
|
|
312
|
+
recommended_ctx = min(32768, int(est_tokens * 1.25) + 1024)
|
|
313
|
+
if "num_ctx" not in eff_options or eff_options["num_ctx"] < recommended_ctx:
|
|
314
|
+
eff_options["num_ctx"] = recommended_ctx
|
|
315
|
+
else:
|
|
316
|
+
eff_options.setdefault("num_ctx", 2048)
|
|
317
|
+
|
|
318
|
+
# Allocate optimal physical CPU cores for multi-threaded inference without thread thrashing
|
|
319
|
+
optimal_threads = min(8, max(2, (os.cpu_count() or 4) // 2 if (os.cpu_count() or 4) > 4 else (os.cpu_count() or 4)))
|
|
320
|
+
eff_options.setdefault("num_thread", optimal_threads)
|
|
321
|
+
# DeepSeek-R1 / reasoning models: default temperature 0.6 prevents runaway thinking loops
|
|
322
|
+
if any(r in model.lower() for r in ("r1", "reason", "qwq")):
|
|
323
|
+
eff_options.setdefault("temperature", 0.6)
|
|
324
|
+
# Ensure generation stops cleanly without spinning indefinitely
|
|
325
|
+
eff_options.setdefault("num_predict", 1024)
|
|
326
|
+
# Ensure stop tokens are passed so models like DeepSeek-R1 / Qwen don't leak EOS tokens
|
|
327
|
+
default_stops = ["<|imend|>", "<|im_end|>", "<|endoftext|>", "<|eot_id|>", "<|end_of_text|>"]
|
|
328
|
+
cur_stops = eff_options.get("stop", [])
|
|
329
|
+
if isinstance(cur_stops, str):
|
|
330
|
+
cur_stops = [cur_stops]
|
|
331
|
+
elif not isinstance(cur_stops, list):
|
|
332
|
+
cur_stops = []
|
|
333
|
+
for s in default_stops:
|
|
334
|
+
if s not in cur_stops:
|
|
335
|
+
cur_stops.append(s)
|
|
336
|
+
eff_options["stop"] = cur_stops
|
|
337
|
+
|
|
338
|
+
payload = {
|
|
339
|
+
"model": model,
|
|
340
|
+
"messages": messages,
|
|
341
|
+
"stream": True,
|
|
342
|
+
"options": eff_options,
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
url = f"{self.base_url}/api/chat"
|
|
346
|
+
resp = None
|
|
347
|
+
|
|
348
|
+
thinking_act = None
|
|
349
|
+
thinking_buffer: List[str] = []
|
|
350
|
+
last_thinking_update = [0.0]
|
|
351
|
+
|
|
352
|
+
def _on_think_chunk(txt: str):
|
|
353
|
+
nonlocal thinking_act
|
|
354
|
+
if not txt:
|
|
355
|
+
return
|
|
356
|
+
thinking_buffer.append(txt)
|
|
357
|
+
try:
|
|
358
|
+
from .. import activity as act_mod
|
|
359
|
+
if thinking_act is None:
|
|
360
|
+
current_turn = act_mod.get_current_turn() or request_id or ""
|
|
361
|
+
thinking_act = act_mod.manager.create(
|
|
362
|
+
type=act_mod.TYPE_STATUS,
|
|
363
|
+
action="thinking",
|
|
364
|
+
title="Thinking…",
|
|
365
|
+
status=act_mod.STATUS_RUNNING,
|
|
366
|
+
phase=act_mod.PHASE_ANALYSIS,
|
|
367
|
+
category=act_mod.CAT_MODEL,
|
|
368
|
+
turn_id=current_turn,
|
|
369
|
+
model=model,
|
|
370
|
+
provider="ollama",
|
|
371
|
+
details=txt,
|
|
372
|
+
stdout=txt,
|
|
373
|
+
)
|
|
374
|
+
self._active_thinking_act_id = getattr(thinking_act, "id", None)
|
|
375
|
+
last_thinking_update[0] = time.time()
|
|
376
|
+
else:
|
|
377
|
+
now = time.time()
|
|
378
|
+
if now - last_thinking_update[0] >= 0.1 or len(txt) > 25:
|
|
379
|
+
last_thinking_update[0] = now
|
|
380
|
+
full_thought = "".join(thinking_buffer).strip()
|
|
381
|
+
lines = [l.strip() for l in full_thought.splitlines() if l.strip()]
|
|
382
|
+
snippet = lines[-1] if lines else ""
|
|
383
|
+
if len(snippet) > 60:
|
|
384
|
+
snippet = "…" + snippet[-57:]
|
|
385
|
+
title = f"Thinking… ({snippet})" if snippet else "Thinking…"
|
|
386
|
+
act_mod.manager.update(
|
|
387
|
+
thinking_act.id,
|
|
388
|
+
title=title,
|
|
389
|
+
details=full_thought,
|
|
390
|
+
stdout=full_thought,
|
|
391
|
+
)
|
|
392
|
+
except Exception:
|
|
393
|
+
pass
|
|
394
|
+
|
|
395
|
+
def _on_think_end():
|
|
396
|
+
nonlocal thinking_act
|
|
397
|
+
if thinking_act is not None:
|
|
398
|
+
try:
|
|
399
|
+
from .. import activity as act_mod
|
|
400
|
+
full_thought = "".join(thinking_buffer).strip()
|
|
401
|
+
word_count = len(full_thought.split())
|
|
402
|
+
act_mod.manager.update(
|
|
403
|
+
thinking_act.id,
|
|
404
|
+
status=act_mod.STATUS_COMPLETED,
|
|
405
|
+
title="Reasoning & Thinking",
|
|
406
|
+
result=f"Thought for {word_count} words",
|
|
407
|
+
details=full_thought,
|
|
408
|
+
stdout=full_thought,
|
|
409
|
+
)
|
|
410
|
+
except Exception:
|
|
411
|
+
pass
|
|
412
|
+
self._active_thinking_act_id = None
|
|
413
|
+
thinking_act = None
|
|
414
|
+
|
|
415
|
+
filter_stream = ThinkingStreamFilter(
|
|
416
|
+
on_thinking_chunk=_on_think_chunk,
|
|
417
|
+
on_thinking_end=_on_think_end,
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
try:
|
|
421
|
+
with self._lock:
|
|
422
|
+
if self._state == RequestState.CANCELLED:
|
|
423
|
+
return
|
|
424
|
+
|
|
425
|
+
try:
|
|
426
|
+
resp = requests.post(
|
|
427
|
+
url,
|
|
428
|
+
json=payload,
|
|
429
|
+
timeout=timeout,
|
|
430
|
+
stream=True,
|
|
431
|
+
)
|
|
432
|
+
except (requests.exceptions.ConnectionError, requests.exceptions.RequestException) as req_err:
|
|
433
|
+
# If connection dropped due to GPU crash / llama-server restart, retry once with CPU
|
|
434
|
+
_LOG.warning("Ollama connection exception (%s); retrying with num_gpu=0 (CPU mode)", req_err)
|
|
435
|
+
time.sleep(1.0)
|
|
436
|
+
payload.setdefault("options", {})["num_gpu"] = 0
|
|
437
|
+
resp = requests.post(
|
|
438
|
+
url,
|
|
439
|
+
json=payload,
|
|
440
|
+
timeout=timeout,
|
|
441
|
+
stream=True,
|
|
442
|
+
)
|
|
443
|
+
|
|
444
|
+
with self._lock:
|
|
445
|
+
if self._state == RequestState.CANCELLED:
|
|
446
|
+
if resp is not None:
|
|
447
|
+
try:
|
|
448
|
+
resp.close()
|
|
449
|
+
except Exception:
|
|
450
|
+
pass
|
|
451
|
+
return
|
|
452
|
+
self._active_response = resp
|
|
453
|
+
self._state = RequestState.STREAMING
|
|
454
|
+
try:
|
|
455
|
+
from ..aicore import _register_response
|
|
456
|
+
_register_response(resp)
|
|
457
|
+
except Exception:
|
|
458
|
+
pass
|
|
459
|
+
|
|
460
|
+
if resp.status_code != 200:
|
|
461
|
+
err_body = resp.text[:400]
|
|
462
|
+
# Auto-fallback to CPU (num_gpu: 0) if GPU offloading crashed or had buffer overrun
|
|
463
|
+
is_gpu_crash = (
|
|
464
|
+
resp.status_code == 500
|
|
465
|
+
or any(k in err_body.lower() for k in (
|
|
466
|
+
"llama-server", "terminated", "0xc0000409", "ggml_assert",
|
|
467
|
+
"buffer in this appli", "overrun", "out of memory", "cuda", "split"
|
|
468
|
+
))
|
|
469
|
+
)
|
|
470
|
+
if is_gpu_crash and payload.get("options", {}).get("num_gpu") != 0:
|
|
471
|
+
_LOG.warning("Ollama GPU crash / 500 on model '%s'; auto-retrying with CPU mode (num_gpu=0)", model)
|
|
472
|
+
time.sleep(0.5)
|
|
473
|
+
payload.setdefault("options", {})["num_gpu"] = 0
|
|
474
|
+
try:
|
|
475
|
+
resp = requests.post(url, json=payload, timeout=timeout, stream=True)
|
|
476
|
+
if resp.status_code == 200:
|
|
477
|
+
with self._lock:
|
|
478
|
+
self._active_response = resp
|
|
479
|
+
self._state = RequestState.STREAMING
|
|
480
|
+
try:
|
|
481
|
+
from ..aicore import _register_response
|
|
482
|
+
_register_response(resp)
|
|
483
|
+
except Exception:
|
|
484
|
+
pass
|
|
485
|
+
except Exception:
|
|
486
|
+
pass
|
|
487
|
+
|
|
488
|
+
# Auto-recovery for context overflow (HTTP 400: request exceeds available context size)
|
|
489
|
+
is_ctx_overflow = (
|
|
490
|
+
resp.status_code == 400
|
|
491
|
+
and any(k in err_body.lower() for k in (
|
|
492
|
+
"exceeds the available context size", "context size",
|
|
493
|
+
"exceedcontextsizeerror", "num_ctx", "context window"
|
|
494
|
+
))
|
|
495
|
+
)
|
|
496
|
+
if is_ctx_overflow:
|
|
497
|
+
m = re.search(r"request\s*\((\d+)\s*tokens\)", err_body, re.IGNORECASE) or re.search(r"(\d+)\s*tokens", err_body, re.IGNORECASE)
|
|
498
|
+
if m:
|
|
499
|
+
req_tok = int(m.group(1))
|
|
500
|
+
new_ctx = max(8192, min(65536, int(req_tok * 1.3) + 2048))
|
|
501
|
+
else:
|
|
502
|
+
new_ctx = max(16384, eff_options.get("num_ctx", 8192) * 2)
|
|
503
|
+
_LOG.warning("Ollama context size exceeded (%s); auto-retrying with num_ctx=%d", err_body, new_ctx)
|
|
504
|
+
payload.setdefault("options", {})["num_ctx"] = new_ctx
|
|
505
|
+
time.sleep(0.3)
|
|
506
|
+
try:
|
|
507
|
+
resp = requests.post(url, json=payload, timeout=timeout, stream=True)
|
|
508
|
+
if resp.status_code == 200:
|
|
509
|
+
with self._lock:
|
|
510
|
+
self._active_response = resp
|
|
511
|
+
self._state = RequestState.STREAMING
|
|
512
|
+
try:
|
|
513
|
+
from ..aicore import _register_response
|
|
514
|
+
_register_response(resp)
|
|
515
|
+
except Exception:
|
|
516
|
+
pass
|
|
517
|
+
except Exception as retry_err:
|
|
518
|
+
_LOG.error("Failed to retry with larger num_ctx: %s", retry_err)
|
|
519
|
+
|
|
520
|
+
# Auto-fallback to an installed model if requested model is not found (404)
|
|
521
|
+
if resp.status_code == 404 and "not found" in err_body.lower():
|
|
522
|
+
try:
|
|
523
|
+
tag_resp = requests.get(f"{self.base_url}/api/tags", timeout=1.5)
|
|
524
|
+
if tag_resp.status_code == 200:
|
|
525
|
+
raw_models = [m.get("name") for m in tag_resp.json().get("models", []) if m.get("name")]
|
|
526
|
+
# Filter out raw base models so fallback is always an instruct/chat model
|
|
527
|
+
models = [m for m in raw_models if not any(b in m.lower() for b in ("-base", "_base", "base-q"))]
|
|
528
|
+
if not models:
|
|
529
|
+
models = raw_models
|
|
530
|
+
if models:
|
|
531
|
+
# Find best match: prefix match first (e.g. "deepseek-r1" in "deepseek-r1:latest")
|
|
532
|
+
alt_model = None
|
|
533
|
+
mod_base = model.split(":")[0].lower()
|
|
534
|
+
for m in models:
|
|
535
|
+
if mod_base in m.lower():
|
|
536
|
+
alt_model = m
|
|
537
|
+
break
|
|
538
|
+
if not alt_model:
|
|
539
|
+
# Prefer lightweight coding/chat models
|
|
540
|
+
for m in models:
|
|
541
|
+
if any(k in m.lower() for k in ("instruct", "chat", "coder", "qwen", "deepseek", "gemma")):
|
|
542
|
+
alt_model = m
|
|
543
|
+
break
|
|
544
|
+
if not alt_model:
|
|
545
|
+
alt_model = models[0]
|
|
546
|
+
_LOG.warning("Ollama model '%s' not found; auto-falling back to '%s'", model, alt_model)
|
|
547
|
+
payload["model"] = alt_model
|
|
548
|
+
resp = requests.post(url, json=payload, timeout=timeout, stream=True)
|
|
549
|
+
if resp.status_code == 200:
|
|
550
|
+
with self._lock:
|
|
551
|
+
self._active_response = resp
|
|
552
|
+
self._state = RequestState.STREAMING
|
|
553
|
+
try:
|
|
554
|
+
from ..aicore import _register_response
|
|
555
|
+
_register_response(resp)
|
|
556
|
+
except Exception:
|
|
557
|
+
pass
|
|
558
|
+
except Exception:
|
|
559
|
+
pass
|
|
560
|
+
|
|
561
|
+
if resp.status_code != 200:
|
|
562
|
+
err_body = resp.text[:300]
|
|
563
|
+
self._state = RequestState.FAILED
|
|
564
|
+
yield f"Error: Ollama ({resp.status_code}): {err_body}"
|
|
565
|
+
return
|
|
566
|
+
|
|
567
|
+
for raw_line in resp.iter_lines(decode_unicode=True):
|
|
568
|
+
# Check deadline if specified
|
|
569
|
+
if deadline and time.monotonic() > deadline:
|
|
570
|
+
yield "\n\n*(Generation timed out)*"
|
|
571
|
+
break
|
|
572
|
+
|
|
573
|
+
# Check if cancelled mid-stream
|
|
574
|
+
with self._lock:
|
|
575
|
+
if self._state == RequestState.CANCELLED:
|
|
576
|
+
yield "\n\n⚠️ **Generation interrupted by user.**"
|
|
577
|
+
return
|
|
578
|
+
|
|
579
|
+
if not raw_line or not raw_line.strip():
|
|
580
|
+
continue
|
|
581
|
+
|
|
582
|
+
try:
|
|
583
|
+
chunk_obj = json.loads(raw_line)
|
|
584
|
+
except (ValueError, json.JSONDecodeError):
|
|
585
|
+
continue
|
|
586
|
+
|
|
587
|
+
# Handle Ollama error in JSON body
|
|
588
|
+
if "error" in chunk_obj:
|
|
589
|
+
self._state = RequestState.FAILED
|
|
590
|
+
yield f"Ollama Error: {chunk_obj['error']}"
|
|
591
|
+
return
|
|
592
|
+
|
|
593
|
+
# Extract token from message.content or message.thinking (for DeepSeek-R1 / Qwen reasoning)
|
|
594
|
+
msg = chunk_obj.get("message") or {}
|
|
595
|
+
content_part = msg.get("content", "")
|
|
596
|
+
thinking_part = msg.get("thinking") or msg.get("reasoning_content") or chunk_obj.get("thinking") or ""
|
|
597
|
+
|
|
598
|
+
# Route thinking in real time to Live Activities; token holds only answer text
|
|
599
|
+
token = filter_stream.process(thinking_part, content_part)
|
|
600
|
+
|
|
601
|
+
if token:
|
|
602
|
+
# Check for ChatML/EOS tokens
|
|
603
|
+
stop_hit = False
|
|
604
|
+
for stop_pattern in ("<|imend|>", "<|im_end|>", "<|endoftext|>", "<|eot_id|>", "<|end_of_text|>"):
|
|
605
|
+
if stop_pattern in token:
|
|
606
|
+
token = token.split(stop_pattern)[0]
|
|
607
|
+
stop_hit = True
|
|
608
|
+
break
|
|
609
|
+
|
|
610
|
+
if is_first_chunk and token:
|
|
611
|
+
token = sanitizer.sanitize_first_chunk(token)
|
|
612
|
+
is_first_chunk = False
|
|
613
|
+
|
|
614
|
+
if token:
|
|
615
|
+
ok, safe_token = guard.feed(token)
|
|
616
|
+
if safe_token:
|
|
617
|
+
yield safe_token
|
|
618
|
+
if not ok or stop_hit:
|
|
619
|
+
break
|
|
620
|
+
elif stop_hit:
|
|
621
|
+
break
|
|
622
|
+
|
|
623
|
+
if chunk_obj.get("done", False):
|
|
624
|
+
tail = filter_stream.finalize()
|
|
625
|
+
if tail:
|
|
626
|
+
ok, safe_token = guard.feed(tail)
|
|
627
|
+
if safe_token:
|
|
628
|
+
yield safe_token
|
|
629
|
+
break
|
|
630
|
+
|
|
631
|
+
with self._lock:
|
|
632
|
+
if self._state != RequestState.CANCELLED:
|
|
633
|
+
self._state = RequestState.COMPLETED
|
|
634
|
+
|
|
635
|
+
except requests.exceptions.ConnectionError:
|
|
636
|
+
with self._lock:
|
|
637
|
+
if self._state == RequestState.CANCELLED:
|
|
638
|
+
yield "\n\n⚠️ **Generation interrupted by user.**"
|
|
639
|
+
return
|
|
640
|
+
self._state = RequestState.FAILED
|
|
641
|
+
yield (
|
|
642
|
+
f"Could not reach Ollama server at {self.base_url}. "
|
|
643
|
+
"Ensure Ollama is running (`ollama serve`)."
|
|
644
|
+
)
|
|
645
|
+
except requests.exceptions.Timeout:
|
|
646
|
+
with self._lock:
|
|
647
|
+
if self._state == RequestState.CANCELLED:
|
|
648
|
+
yield "\n\n⚠️ **Generation interrupted by user.**"
|
|
649
|
+
return
|
|
650
|
+
self._state = RequestState.FAILED
|
|
651
|
+
yield "The Ollama request timed out waiting for generation to start."
|
|
652
|
+
except Exception as e:
|
|
653
|
+
with self._lock:
|
|
654
|
+
if self._state == RequestState.CANCELLED:
|
|
655
|
+
yield "\n\n⚠️ **Generation interrupted by user.**"
|
|
656
|
+
return
|
|
657
|
+
self._state = RequestState.FAILED
|
|
658
|
+
yield f"Ollama streaming failure: {e}"
|
|
659
|
+
finally:
|
|
660
|
+
try:
|
|
661
|
+
filter_stream.finalize()
|
|
662
|
+
except Exception:
|
|
663
|
+
pass
|
|
664
|
+
if thinking_act is not None:
|
|
665
|
+
try:
|
|
666
|
+
from .. import activity as act_mod
|
|
667
|
+
st = act_mod.STATUS_CANCELLED if self._state == RequestState.CANCELLED else act_mod.STATUS_COMPLETED
|
|
668
|
+
act_mod.manager.update(thinking_act.id, status=st)
|
|
669
|
+
except Exception:
|
|
670
|
+
pass
|
|
671
|
+
self._active_thinking_act_id = None
|
|
672
|
+
thinking_act = None
|
|
673
|
+
with self._lock:
|
|
674
|
+
if self._active_response is not None:
|
|
675
|
+
try:
|
|
676
|
+
from ..aicore import _unregister_response
|
|
677
|
+
_unregister_response(self._active_response)
|
|
678
|
+
except Exception:
|
|
679
|
+
pass
|
|
680
|
+
try:
|
|
681
|
+
self._active_response.close()
|
|
682
|
+
except Exception:
|
|
683
|
+
pass
|
|
684
|
+
self._active_response = None
|
|
685
|
+
|
|
686
|
+
def run_hello_test(self, model: str) -> Tuple[bool, str]:
|
|
687
|
+
"""Diagnostic health check (Section 34): tests Ollama generation directly."""
|
|
688
|
+
import requests
|
|
689
|
+
|
|
690
|
+
try:
|
|
691
|
+
url = f"{self.base_url}/api/chat"
|
|
692
|
+
payload = {
|
|
693
|
+
"model": model,
|
|
694
|
+
"messages": [{"role": "user", "content": "Reply with exactly: HELLO_TEST"}],
|
|
695
|
+
"stream": False,
|
|
696
|
+
"options": {"temperature": 0.0},
|
|
697
|
+
}
|
|
698
|
+
resp = requests.post(url, json=payload, timeout=10.0)
|
|
699
|
+
if resp.status_code == 200:
|
|
700
|
+
data = resp.json()
|
|
701
|
+
reply = (data.get("message", {}).get("content", "") or "").strip()
|
|
702
|
+
if "HELLO_TEST" in reply:
|
|
703
|
+
return True, reply
|
|
704
|
+
return False, f"Model responded with unexpected output: {reply}"
|
|
705
|
+
return False, f"Ollama HTTP {resp.status_code}: {resp.text[:100]}"
|
|
706
|
+
except Exception as e:
|
|
707
|
+
return False, f"Connection test failed: {e}"
|