quantum-framework 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantum/__init__.py +1 -0
- quantum/cli/__init__.py +2 -0
- quantum/cli/commands/__init__.py +15 -0
- quantum/cli/commands/build.py +417 -0
- quantum/cli/commands/dev.py +185 -0
- quantum/cli/commands/docs.py +329 -0
- quantum/cli/commands/lint.py +523 -0
- quantum/cli/commands/migrate.py +527 -0
- quantum/cli/commands/new.py +622 -0
- quantum/cli/commands/serve.py +190 -0
- quantum/cli/commands/test.py +193 -0
- quantum/cli/deploy.py +810 -0
- quantum/cli/hot_reload.py +951 -0
- quantum/cli/jobs.py +356 -0
- quantum/cli/mq.py +582 -0
- quantum/cli/pkg.py +390 -0
- quantum/cli/runner.py +547 -0
- quantum/cli/server_process.py +159 -0
- quantum/cli/utils.py +334 -0
- quantum/compiler/__init__.py +30 -0
- quantum/compiler/base_generator.py +367 -0
- quantum/compiler/cli.py +295 -0
- quantum/compiler/expression_transformer.py +444 -0
- quantum/compiler/javascript/__init__.py +10 -0
- quantum/compiler/javascript/generator.py +659 -0
- quantum/compiler/optimizer.py +270 -0
- quantum/compiler/python/__init__.py +10 -0
- quantum/compiler/python/generator.py +883 -0
- quantum/compiler/python/runtime.py +863 -0
- quantum/compiler/transpiler.py +330 -0
- quantum/core/__init__.py +3 -0
- quantum/core/ast_nodes.py +2611 -0
- quantum/core/expression_diagnostics.py +87 -0
- quantum/core/expression_stdlib.py +148 -0
- quantum/core/expressions.py +591 -0
- quantum/core/features/agents/src/__init__.py +30 -0
- quantum/core/features/agents/src/ast_node.py +540 -0
- quantum/core/features/conditionals/src/__init__.py +8 -0
- quantum/core/features/conditionals/src/ast_node.py +68 -0
- quantum/core/features/data_fetching/src/__init__.py +21 -0
- quantum/core/features/data_fetching/src/ast_node.py +312 -0
- quantum/core/features/data_fetching/src/desktop_adapter.py +351 -0
- quantum/core/features/data_fetching/src/html_adapter.py +474 -0
- quantum/core/features/data_fetching/src/parser.py +225 -0
- quantum/core/features/data_import/src/__init__.py +0 -0
- quantum/core/features/data_import/src/ast_node.py +291 -0
- quantum/core/features/data_import/src/runtime.py +538 -0
- quantum/core/features/dump/src/__init__.py +12 -0
- quantum/core/features/dump/src/ast_node.py +106 -0
- quantum/core/features/dump/src/parser.py +61 -0
- quantum/core/features/dump/src/runtime.py +246 -0
- quantum/core/features/functions/src/__init__.py +8 -0
- quantum/core/features/functions/src/ast_node.py +149 -0
- quantum/core/features/game_engine_2d/src/__init__.py +19 -0
- quantum/core/features/game_engine_2d/src/ast_nodes.py +1719 -0
- quantum/core/features/game_engine_2d/src/parser.py +983 -0
- quantum/core/features/invocation/src/__init__.py +0 -0
- quantum/core/features/invocation/src/ast_node.py +146 -0
- quantum/core/features/invocation/src/runtime.py +327 -0
- quantum/core/features/knowledge_base/src/__init__.py +6 -0
- quantum/core/features/knowledge_base/src/ast_node.py +113 -0
- quantum/core/features/knowledge_base/src/parser.py +82 -0
- quantum/core/features/logging/src/__init__.py +12 -0
- quantum/core/features/logging/src/ast_node.py +111 -0
- quantum/core/features/logging/src/parser.py +50 -0
- quantum/core/features/logging/src/runtime.py +190 -0
- quantum/core/features/loops/src/__init__.py +8 -0
- quantum/core/features/loops/src/ast_node.py +60 -0
- quantum/core/features/query/src/__init__.py +0 -0
- quantum/core/features/query/src/database_service.py +322 -0
- quantum/core/features/query/src/query_validators.py +20 -0
- quantum/core/features/state_management/src/__init__.py +11 -0
- quantum/core/features/state_management/src/ast_node.py +228 -0
- quantum/core/features/terminal_engine/src/__init__.py +21 -0
- quantum/core/features/terminal_engine/src/ast_nodes.py +560 -0
- quantum/core/features/terminal_engine/src/parser.py +361 -0
- quantum/core/features/testing_engine/src/__init__.py +41 -0
- quantum/core/features/testing_engine/src/ast_nodes.py +1212 -0
- quantum/core/features/testing_engine/src/parser.py +604 -0
- quantum/core/features/theming/src/__init__.py +48 -0
- quantum/core/features/theming/src/ast_node.py +137 -0
- quantum/core/features/theming/src/presets.py +405 -0
- quantum/core/features/ui_engine/src/__init__.py +20 -0
- quantum/core/features/ui_engine/src/ast_nodes.py +1854 -0
- quantum/core/features/ui_engine/src/parser.py +1106 -0
- quantum/core/features/websocket/src/__init__.py +24 -0
- quantum/core/features/websocket/src/ast_node.py +247 -0
- quantum/core/html_compat.py +299 -0
- quantum/core/parser.py +1235 -0
- quantum/core/parser_registry.py +213 -0
- quantum/core/parsers/__init__.py +76 -0
- quantum/core/parsers/ai/__init__.py +12 -0
- quantum/core/parsers/ai/agent_parser.py +106 -0
- quantum/core/parsers/ai/knowledge_parser.py +86 -0
- quantum/core/parsers/ai/llm_parser.py +78 -0
- quantum/core/parsers/ai/team_parser.py +88 -0
- quantum/core/parsers/base.py +322 -0
- quantum/core/parsers/composition/__init__.py +10 -0
- quantum/core/parsers/composition/import_parser.py +50 -0
- quantum/core/parsers/composition/slot_parser.py +48 -0
- quantum/core/parsers/control_flow/__init__.py +11 -0
- quantum/core/parsers/control_flow/if_parser.py +68 -0
- quantum/core/parsers/control_flow/loop_parser.py +105 -0
- quantum/core/parsers/control_flow/set_parser.py +89 -0
- quantum/core/parsers/data/__init__.py +12 -0
- quantum/core/parsers/data/data_parser.py +217 -0
- quantum/core/parsers/data/invoke_parser.py +116 -0
- quantum/core/parsers/data/query_parser.py +188 -0
- quantum/core/parsers/data/transaction_parser.py +77 -0
- quantum/core/parsers/events/__init__.py +9 -0
- quantum/core/parsers/events/dispatch_event_parser.py +44 -0
- quantum/core/parsers/forms/__init__.py +11 -0
- quantum/core/parsers/forms/action_parser.py +54 -0
- quantum/core/parsers/forms/flash_parser.py +34 -0
- quantum/core/parsers/forms/redirect_parser.py +33 -0
- quantum/core/parsers/functions/__init__.py +11 -0
- quantum/core/parsers/functions/function_parser.py +137 -0
- quantum/core/parsers/functions/param_parser.py +66 -0
- quantum/core/parsers/functions/return_parser.py +31 -0
- quantum/core/parsers/html/__init__.py +10 -0
- quantum/core/parsers/html/component_call_parser.py +130 -0
- quantum/core/parsers/html/html_parser.py +115 -0
- quantum/core/parsers/jobs/__init__.py +11 -0
- quantum/core/parsers/jobs/job_parser.py +71 -0
- quantum/core/parsers/jobs/schedule_parser.py +62 -0
- quantum/core/parsers/jobs/thread_parser.py +57 -0
- quantum/core/parsers/messaging/__init__.py +17 -0
- quantum/core/parsers/messaging/message_ack_parser.py +30 -0
- quantum/core/parsers/messaging/message_nack_parser.py +30 -0
- quantum/core/parsers/messaging/message_parser.py +114 -0
- quantum/core/parsers/messaging/queue_parser.py +61 -0
- quantum/core/parsers/messaging/websocket_parser.py +121 -0
- quantum/core/parsers/persistence/__init__.py +9 -0
- quantum/core/parsers/persistence/persist_parser.py +64 -0
- quantum/core/parsers/routing/__init__.py +9 -0
- quantum/core/parsers/routing/route_parser.py +41 -0
- quantum/core/parsers/scripting/__init__.py +12 -0
- quantum/core/parsers/scripting/pyclass_parser.py +62 -0
- quantum/core/parsers/scripting/pydecorator_parser.py +68 -0
- quantum/core/parsers/scripting/pyimport_parser.py +52 -0
- quantum/core/parsers/scripting/python_parser.py +49 -0
- quantum/core/parsers/services/__init__.py +12 -0
- quantum/core/parsers/services/dump_parser.py +52 -0
- quantum/core/parsers/services/file_parser.py +48 -0
- quantum/core/parsers/services/log_parser.py +46 -0
- quantum/core/parsers/services/mail_parser.py +65 -0
- quantum/core/tiers.py +82 -0
- quantum/packages/__init__.py +28 -0
- quantum/packages/manager.py +413 -0
- quantum/packages/manifest.py +351 -0
- quantum/packages/registry.py +399 -0
- quantum/packages/resolver.py +336 -0
- quantum/plugins/__init__.py +33 -0
- quantum/plugins/hooks.py +329 -0
- quantum/plugins/loader.py +479 -0
- quantum/plugins/manifest.py +336 -0
- quantum/plugins/registry.py +371 -0
- quantum/runtime/__init__.py +28 -0
- quantum/runtime/action_handler.py +443 -0
- quantum/runtime/adapters/__init__.py +88 -0
- quantum/runtime/adapters/memory_adapter.py +690 -0
- quantum/runtime/adapters/rabbitmq_adapter.py +715 -0
- quantum/runtime/adapters/redis_adapter.py +582 -0
- quantum/runtime/adapters/sqlite_adapter.py +414 -0
- quantum/runtime/agent_service.py +1133 -0
- quantum/runtime/api_server.py +86 -0
- quantum/runtime/ast_cache.py +506 -0
- quantum/runtime/auth_service.py +267 -0
- quantum/runtime/component.py +990 -0
- quantum/runtime/component_composer.py +319 -0
- quantum/runtime/component_resolver.py +174 -0
- quantum/runtime/database_service.py +598 -0
- quantum/runtime/email_service.py +162 -0
- quantum/runtime/error_handler.py +295 -0
- quantum/runtime/execution_context.py +286 -0
- quantum/runtime/executor_registry.py +171 -0
- quantum/runtime/executors/__init__.py +71 -0
- quantum/runtime/executors/ai/__init__.py +12 -0
- quantum/runtime/executors/ai/agent_executor.py +217 -0
- quantum/runtime/executors/ai/knowledge_executor.py +114 -0
- quantum/runtime/executors/ai/llm_executor.py +153 -0
- quantum/runtime/executors/ai/team_executor.py +171 -0
- quantum/runtime/executors/base.py +262 -0
- quantum/runtime/executors/control_flow/__init__.py +11 -0
- quantum/runtime/executors/control_flow/if_executor.py +93 -0
- quantum/runtime/executors/control_flow/loop_executor.py +307 -0
- quantum/runtime/executors/control_flow/set_executor.py +412 -0
- quantum/runtime/executors/data/__init__.py +12 -0
- quantum/runtime/executors/data/data_executor.py +145 -0
- quantum/runtime/executors/data/invoke_executor.py +176 -0
- quantum/runtime/executors/data/query_executor.py +256 -0
- quantum/runtime/executors/data/transaction_executor.py +91 -0
- quantum/runtime/executors/jobs/__init__.py +11 -0
- quantum/runtime/executors/jobs/job_executor.py +190 -0
- quantum/runtime/executors/jobs/schedule_executor.py +132 -0
- quantum/runtime/executors/jobs/thread_executor.py +127 -0
- quantum/runtime/executors/messaging/__init__.py +17 -0
- quantum/runtime/executors/messaging/message_ack_executor.py +51 -0
- quantum/runtime/executors/messaging/message_executor.py +174 -0
- quantum/runtime/executors/messaging/queue_executor.py +103 -0
- quantum/runtime/executors/messaging/websocket_executor.py +197 -0
- quantum/runtime/executors/scripting/__init__.py +11 -0
- quantum/runtime/executors/scripting/pyclass_executor.py +90 -0
- quantum/runtime/executors/scripting/pyimport_executor.py +81 -0
- quantum/runtime/executors/scripting/python_executor.py +249 -0
- quantum/runtime/executors/services/__init__.py +12 -0
- quantum/runtime/executors/services/dump_executor.py +72 -0
- quantum/runtime/executors/services/file_executor.py +89 -0
- quantum/runtime/executors/services/log_executor.py +77 -0
- quantum/runtime/executors/services/mail_executor.py +81 -0
- quantum/runtime/expression_cache.py +498 -0
- quantum/runtime/file_upload_service.py +326 -0
- quantum/runtime/function_registry.py +118 -0
- quantum/runtime/game_builder.py +166 -0
- quantum/runtime/game_code_generator.py +2371 -0
- quantum/runtime/game_templates.py +2006 -0
- quantum/runtime/godot_code_generator.py +4681 -0
- quantum/runtime/godot_templates.py +1449 -0
- quantum/runtime/job_executor.py +1599 -0
- quantum/runtime/knowledge_service.py +500 -0
- quantum/runtime/llm_cache.py +100 -0
- quantum/runtime/llm_providers.py +704 -0
- quantum/runtime/llm_service.py +287 -0
- quantum/runtime/logging_setup.py +140 -0
- quantum/runtime/message_broker.py +364 -0
- quantum/runtime/message_queue_service.py +571 -0
- quantum/runtime/param_validation.py +184 -0
- quantum/runtime/pypy_compat.py +315 -0
- quantum/runtime/python_bridge.py +698 -0
- quantum/runtime/query_validators.py +304 -0
- quantum/runtime/renderer.py +733 -0
- quantum/runtime/service_container.py +444 -0
- quantum/runtime/terminal_builder.py +76 -0
- quantum/runtime/terminal_code_generator.py +607 -0
- quantum/runtime/terminal_templates.py +243 -0
- quantum/runtime/testing_builder.py +77 -0
- quantum/runtime/testing_code_generator.py +833 -0
- quantum/runtime/testing_templates.py +85 -0
- quantum/runtime/ui_builder.py +188 -0
- quantum/runtime/ui_desktop_adapter.py +1730 -0
- quantum/runtime/ui_desktop_templates.py +307 -0
- quantum/runtime/ui_html_adapter.py +2691 -0
- quantum/runtime/ui_html_templates.py +2297 -0
- quantum/runtime/ui_mobile_adapter.py +1832 -0
- quantum/runtime/ui_mobile_templates.py +1003 -0
- quantum/runtime/ui_textual_adapter.py +1866 -0
- quantum/runtime/ui_textual_templates.py +45 -0
- quantum/runtime/ui_tokens.py +465 -0
- quantum/runtime/ui_validator.py +365 -0
- quantum/runtime/validators.py +256 -0
- quantum/runtime/web_server.py +1766 -0
- quantum/runtime/websocket_adapter.py +501 -0
- quantum/runtime/websocket_service.py +585 -0
- quantum/runtime/websocket_transport.py +289 -0
- quantum/runtime/wsgi.py +101 -0
- quantum/utils/__init__.py +1 -0
- quantum_framework-0.9.0.dist-info/METADATA +244 -0
- quantum_framework-0.9.0.dist-info/RECORD +262 -0
- quantum_framework-0.9.0.dist-info/WHEEL +5 -0
- quantum_framework-0.9.0.dist-info/entry_points.txt +2 -0
- quantum_framework-0.9.0.dist-info/licenses/LICENSE +21 -0
- quantum_framework-0.9.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,500 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Quantum Knowledge Service - RAG with ChromaDB + Ollama embeddings.
|
|
3
|
+
|
|
4
|
+
Provides indexing, vector search, and RAG query capabilities for
|
|
5
|
+
the q:knowledge / q:query knowledge: datasource integration.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
import re
|
|
10
|
+
import glob
|
|
11
|
+
import hashlib
|
|
12
|
+
import logging
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import List, Dict, Any, Optional
|
|
15
|
+
|
|
16
|
+
import requests
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class KnowledgeError(Exception):
|
|
22
|
+
"""Raised when knowledge base operations fail."""
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class KnowledgeService:
|
|
27
|
+
"""ChromaDB + Ollama embeddings service for RAG."""
|
|
28
|
+
|
|
29
|
+
def __init__(self, llm_service=None):
|
|
30
|
+
self.llm_service = llm_service
|
|
31
|
+
self._collections: Dict[str, Any] = {} # name -> ChromaDB collection
|
|
32
|
+
self._client = None
|
|
33
|
+
self._ollama_base_url = os.getenv(
|
|
34
|
+
'QUANTUM_LLM_BASE_URL', 'http://localhost:11434'
|
|
35
|
+
).rstrip('/')
|
|
36
|
+
|
|
37
|
+
def _get_client(self, persist: bool = False, persist_path: Optional[str] = None):
|
|
38
|
+
"""Get or create ChromaDB client."""
|
|
39
|
+
if self._client is not None:
|
|
40
|
+
return self._client
|
|
41
|
+
|
|
42
|
+
try:
|
|
43
|
+
import chromadb
|
|
44
|
+
except ImportError:
|
|
45
|
+
raise KnowledgeError(
|
|
46
|
+
"q:knowledge needs the optional RAG dependencies, which are not "
|
|
47
|
+
"installed.\n"
|
|
48
|
+
" pip install 'quantum-framework[rag]' (or: pip install chromadb)\n"
|
|
49
|
+
"You also need an embedding model available on your LLM server, "
|
|
50
|
+
"e.g. `ollama pull nomic-embed-text`."
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
if persist and persist_path:
|
|
54
|
+
self._client = chromadb.PersistentClient(path=persist_path)
|
|
55
|
+
else:
|
|
56
|
+
self._client = chromadb.Client()
|
|
57
|
+
|
|
58
|
+
return self._client
|
|
59
|
+
|
|
60
|
+
def index_knowledge(
|
|
61
|
+
self,
|
|
62
|
+
name: str,
|
|
63
|
+
sources: list,
|
|
64
|
+
embed_model: str = "nomic-embed-text",
|
|
65
|
+
chunk_size: int = 500,
|
|
66
|
+
chunk_overlap: int = 50,
|
|
67
|
+
persist: bool = False,
|
|
68
|
+
persist_path: Optional[str] = None,
|
|
69
|
+
rebuild: bool = False,
|
|
70
|
+
database_service=None,
|
|
71
|
+
exec_context=None,
|
|
72
|
+
):
|
|
73
|
+
"""
|
|
74
|
+
Index knowledge base sources into ChromaDB.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
name: Knowledge base name (used as collection name)
|
|
78
|
+
sources: List of KnowledgeSourceNode objects
|
|
79
|
+
embed_model: Ollama embedding model name
|
|
80
|
+
chunk_size: Characters per chunk
|
|
81
|
+
chunk_overlap: Overlap between chunks
|
|
82
|
+
persist: Whether to persist ChromaDB to disk
|
|
83
|
+
persist_path: Path for ChromaDB persistence
|
|
84
|
+
rebuild: Force reindex even if collection exists
|
|
85
|
+
database_service: DatabaseService for query-type sources
|
|
86
|
+
exec_context: ExecutionContext for variable resolution
|
|
87
|
+
"""
|
|
88
|
+
client = self._get_client(persist, persist_path)
|
|
89
|
+
|
|
90
|
+
# Get or create collection
|
|
91
|
+
if rebuild:
|
|
92
|
+
try:
|
|
93
|
+
client.delete_collection(name)
|
|
94
|
+
except Exception:
|
|
95
|
+
pass
|
|
96
|
+
|
|
97
|
+
collection = client.get_or_create_collection(
|
|
98
|
+
name=name,
|
|
99
|
+
metadata={"hnsw:space": "cosine"}
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
# If collection already has documents and not rebuilding, skip
|
|
103
|
+
if collection.count() > 0 and not rebuild:
|
|
104
|
+
self._collections[name] = collection
|
|
105
|
+
logger.info(f"Knowledge base '{name}' already indexed ({collection.count()} chunks)")
|
|
106
|
+
return
|
|
107
|
+
|
|
108
|
+
# Extract text from all sources
|
|
109
|
+
all_texts = []
|
|
110
|
+
for source in sources:
|
|
111
|
+
texts = self._extract_source_text(
|
|
112
|
+
source, database_service, exec_context
|
|
113
|
+
)
|
|
114
|
+
all_texts.extend(texts)
|
|
115
|
+
|
|
116
|
+
if not all_texts:
|
|
117
|
+
logger.warning(f"Knowledge base '{name}': no text extracted from sources")
|
|
118
|
+
self._collections[name] = collection
|
|
119
|
+
return
|
|
120
|
+
|
|
121
|
+
# Chunk all extracted texts
|
|
122
|
+
all_chunks = []
|
|
123
|
+
all_metadata = []
|
|
124
|
+
for text_item in all_texts:
|
|
125
|
+
text = text_item['text']
|
|
126
|
+
source_label = text_item.get('source', 'inline')
|
|
127
|
+
cs = text_item.get('chunk_size', chunk_size)
|
|
128
|
+
co = text_item.get('chunk_overlap', chunk_overlap)
|
|
129
|
+
|
|
130
|
+
chunks = self._chunk_text(text, cs, co)
|
|
131
|
+
for i, chunk in enumerate(chunks):
|
|
132
|
+
all_chunks.append(chunk)
|
|
133
|
+
all_metadata.append({
|
|
134
|
+
"source": source_label,
|
|
135
|
+
"chunk_index": i,
|
|
136
|
+
"total_chunks": len(chunks),
|
|
137
|
+
})
|
|
138
|
+
|
|
139
|
+
if not all_chunks:
|
|
140
|
+
self._collections[name] = collection
|
|
141
|
+
return
|
|
142
|
+
|
|
143
|
+
# Generate embeddings via Ollama
|
|
144
|
+
embeddings = self._generate_embeddings(all_chunks, embed_model)
|
|
145
|
+
|
|
146
|
+
# Generate IDs
|
|
147
|
+
ids = [
|
|
148
|
+
f"{name}_{hashlib.md5(chunk.encode()).hexdigest()[:12]}_{i}"
|
|
149
|
+
for i, chunk in enumerate(all_chunks)
|
|
150
|
+
]
|
|
151
|
+
|
|
152
|
+
# Upsert into ChromaDB (batch to avoid limits)
|
|
153
|
+
batch_size = 100
|
|
154
|
+
for start in range(0, len(all_chunks), batch_size):
|
|
155
|
+
end = min(start + batch_size, len(all_chunks))
|
|
156
|
+
collection.upsert(
|
|
157
|
+
ids=ids[start:end],
|
|
158
|
+
documents=all_chunks[start:end],
|
|
159
|
+
embeddings=embeddings[start:end],
|
|
160
|
+
metadatas=all_metadata[start:end],
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
self._collections[name] = collection
|
|
164
|
+
logger.info(f"Knowledge base '{name}' indexed: {len(all_chunks)} chunks from {len(all_texts)} text segments")
|
|
165
|
+
|
|
166
|
+
def _extract_source_text(
|
|
167
|
+
self, source, database_service=None, exec_context=None
|
|
168
|
+
) -> List[Dict[str, Any]]:
|
|
169
|
+
"""
|
|
170
|
+
Extract text from a KnowledgeSourceNode.
|
|
171
|
+
|
|
172
|
+
Returns list of dicts: [{text, source, chunk_size?, chunk_overlap?}]
|
|
173
|
+
"""
|
|
174
|
+
results = []
|
|
175
|
+
st = source.source_type
|
|
176
|
+
extra = {}
|
|
177
|
+
if source.chunk_size is not None:
|
|
178
|
+
extra['chunk_size'] = source.chunk_size
|
|
179
|
+
if source.chunk_overlap is not None:
|
|
180
|
+
extra['chunk_overlap'] = source.chunk_overlap
|
|
181
|
+
|
|
182
|
+
if st == 'text':
|
|
183
|
+
if source.content:
|
|
184
|
+
results.append({'text': source.content, 'source': 'inline', **extra})
|
|
185
|
+
|
|
186
|
+
elif st == 'file':
|
|
187
|
+
path = Path(source.path)
|
|
188
|
+
if path.exists() and path.is_file():
|
|
189
|
+
try:
|
|
190
|
+
text = path.read_text(encoding='utf-8')
|
|
191
|
+
results.append({'text': text, 'source': str(path), **extra})
|
|
192
|
+
except Exception as e:
|
|
193
|
+
logger.warning(f"Failed to read file {path}: {e}")
|
|
194
|
+
else:
|
|
195
|
+
logger.warning(f"Knowledge source file not found: {source.path}")
|
|
196
|
+
|
|
197
|
+
elif st == 'directory':
|
|
198
|
+
dir_path = Path(source.path)
|
|
199
|
+
pattern = source.pattern or '*.md'
|
|
200
|
+
if dir_path.exists() and dir_path.is_dir():
|
|
201
|
+
for file_path in sorted(dir_path.glob(pattern)):
|
|
202
|
+
if file_path.is_file():
|
|
203
|
+
try:
|
|
204
|
+
text = file_path.read_text(encoding='utf-8')
|
|
205
|
+
results.append({'text': text, 'source': str(file_path), **extra})
|
|
206
|
+
except Exception as e:
|
|
207
|
+
logger.warning(f"Failed to read {file_path}: {e}")
|
|
208
|
+
else:
|
|
209
|
+
logger.warning(f"Knowledge source directory not found: {source.path}")
|
|
210
|
+
|
|
211
|
+
elif st == 'url':
|
|
212
|
+
logger.warning("URL source type is not yet implemented for q:knowledge")
|
|
213
|
+
|
|
214
|
+
elif st == 'query':
|
|
215
|
+
if database_service and source.datasource and source.sql:
|
|
216
|
+
try:
|
|
217
|
+
result = database_service.execute_query(
|
|
218
|
+
source.datasource, source.sql, {}
|
|
219
|
+
)
|
|
220
|
+
for row in result.data:
|
|
221
|
+
# Concatenate all string values in the row
|
|
222
|
+
parts = [str(v) for v in row.values() if v is not None]
|
|
223
|
+
text = ' '.join(parts)
|
|
224
|
+
if text.strip():
|
|
225
|
+
results.append({'text': text, 'source': f"query:{source.datasource}", **extra})
|
|
226
|
+
except Exception as e:
|
|
227
|
+
logger.warning(f"Failed to execute knowledge source query: {e}")
|
|
228
|
+
else:
|
|
229
|
+
logger.warning("Query source requires datasource, sql, and database_service")
|
|
230
|
+
|
|
231
|
+
return results
|
|
232
|
+
|
|
233
|
+
def _chunk_text(self, text: str, chunk_size: int = 500, chunk_overlap: int = 50) -> List[str]:
|
|
234
|
+
"""
|
|
235
|
+
Split text into chunks using a sliding window with paragraph/sentence awareness.
|
|
236
|
+
|
|
237
|
+
Args:
|
|
238
|
+
text: Input text
|
|
239
|
+
chunk_size: Target characters per chunk
|
|
240
|
+
chunk_overlap: Overlap between chunks
|
|
241
|
+
|
|
242
|
+
Returns:
|
|
243
|
+
List of text chunks
|
|
244
|
+
"""
|
|
245
|
+
if not text or not text.strip():
|
|
246
|
+
return []
|
|
247
|
+
|
|
248
|
+
text = text.strip()
|
|
249
|
+
|
|
250
|
+
# If text is smaller than chunk_size, return as single chunk
|
|
251
|
+
if len(text) <= chunk_size:
|
|
252
|
+
return [text]
|
|
253
|
+
|
|
254
|
+
# Split into paragraphs first
|
|
255
|
+
paragraphs = re.split(r'\n\s*\n', text)
|
|
256
|
+
|
|
257
|
+
chunks = []
|
|
258
|
+
current_chunk = ""
|
|
259
|
+
|
|
260
|
+
for paragraph in paragraphs:
|
|
261
|
+
paragraph = paragraph.strip()
|
|
262
|
+
if not paragraph:
|
|
263
|
+
continue
|
|
264
|
+
|
|
265
|
+
# If adding this paragraph exceeds chunk_size
|
|
266
|
+
if current_chunk and len(current_chunk) + len(paragraph) + 1 > chunk_size:
|
|
267
|
+
chunks.append(current_chunk.strip())
|
|
268
|
+
# Start new chunk with overlap from end of previous
|
|
269
|
+
if chunk_overlap > 0 and len(current_chunk) > chunk_overlap:
|
|
270
|
+
current_chunk = current_chunk[-chunk_overlap:] + "\n" + paragraph
|
|
271
|
+
else:
|
|
272
|
+
current_chunk = paragraph
|
|
273
|
+
else:
|
|
274
|
+
if current_chunk:
|
|
275
|
+
current_chunk += "\n" + paragraph
|
|
276
|
+
else:
|
|
277
|
+
current_chunk = paragraph
|
|
278
|
+
|
|
279
|
+
# If single paragraph exceeds chunk_size, split by sentences
|
|
280
|
+
while len(current_chunk) > chunk_size:
|
|
281
|
+
# Find a good split point (end of sentence)
|
|
282
|
+
split_at = chunk_size
|
|
283
|
+
for sep in ['. ', '! ', '? ', '\n', '; ', ', ']:
|
|
284
|
+
idx = current_chunk.rfind(sep, 0, chunk_size)
|
|
285
|
+
if idx > chunk_size // 3:
|
|
286
|
+
split_at = idx + len(sep)
|
|
287
|
+
break
|
|
288
|
+
|
|
289
|
+
chunks.append(current_chunk[:split_at].strip())
|
|
290
|
+
if chunk_overlap > 0:
|
|
291
|
+
overlap_start = max(0, split_at - chunk_overlap)
|
|
292
|
+
current_chunk = current_chunk[overlap_start:]
|
|
293
|
+
else:
|
|
294
|
+
current_chunk = current_chunk[split_at:]
|
|
295
|
+
|
|
296
|
+
# Don't forget the last chunk
|
|
297
|
+
if current_chunk.strip():
|
|
298
|
+
chunks.append(current_chunk.strip())
|
|
299
|
+
|
|
300
|
+
return chunks
|
|
301
|
+
|
|
302
|
+
def _generate_embeddings(self, texts: List[str], model: str = "nomic-embed-text") -> List[List[float]]:
|
|
303
|
+
"""
|
|
304
|
+
Generate embeddings via Ollama /api/embed endpoint.
|
|
305
|
+
|
|
306
|
+
Args:
|
|
307
|
+
texts: List of text strings to embed
|
|
308
|
+
model: Ollama embedding model name
|
|
309
|
+
|
|
310
|
+
Returns:
|
|
311
|
+
List of embedding vectors
|
|
312
|
+
"""
|
|
313
|
+
if not texts:
|
|
314
|
+
return []
|
|
315
|
+
|
|
316
|
+
url = f"{self._ollama_base_url}/api/embed"
|
|
317
|
+
|
|
318
|
+
# Batch to avoid very large payloads
|
|
319
|
+
all_embeddings = []
|
|
320
|
+
batch_size = 50
|
|
321
|
+
|
|
322
|
+
for start in range(0, len(texts), batch_size):
|
|
323
|
+
batch = texts[start:start + batch_size]
|
|
324
|
+
payload = {
|
|
325
|
+
"model": model,
|
|
326
|
+
"input": batch,
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
try:
|
|
330
|
+
embed_timeout = int(os.getenv('QUANTUM_EMBED_TIMEOUT', '50'))
|
|
331
|
+
resp = requests.post(url, json=payload, timeout=embed_timeout)
|
|
332
|
+
resp.raise_for_status()
|
|
333
|
+
data = resp.json()
|
|
334
|
+
|
|
335
|
+
embeddings = data.get("embeddings", [])
|
|
336
|
+
if len(embeddings) != len(batch):
|
|
337
|
+
raise KnowledgeError(
|
|
338
|
+
f"Embedding count mismatch: expected {len(batch)}, got {len(embeddings)}"
|
|
339
|
+
)
|
|
340
|
+
all_embeddings.extend(embeddings)
|
|
341
|
+
|
|
342
|
+
except requests.ConnectionError:
|
|
343
|
+
raise KnowledgeError(
|
|
344
|
+
f"Cannot connect to Ollama at {self._ollama_base_url}. "
|
|
345
|
+
"Ensure Ollama is running (ollama serve) and the embedding model is pulled "
|
|
346
|
+
f"(ollama pull {model})"
|
|
347
|
+
)
|
|
348
|
+
except requests.Timeout:
|
|
349
|
+
raise KnowledgeError(f"Embedding request timed out for model {model}")
|
|
350
|
+
except requests.HTTPError as e:
|
|
351
|
+
raise KnowledgeError(f"Ollama embedding API error: {e.response.status_code} - {e.response.text}")
|
|
352
|
+
except KnowledgeError:
|
|
353
|
+
raise
|
|
354
|
+
except Exception as e:
|
|
355
|
+
raise KnowledgeError(f"Embedding generation error: {e}")
|
|
356
|
+
|
|
357
|
+
return all_embeddings
|
|
358
|
+
|
|
359
|
+
def search(
|
|
360
|
+
self,
|
|
361
|
+
name: str,
|
|
362
|
+
query_text: str,
|
|
363
|
+
n_results: int = 5,
|
|
364
|
+
embed_model: str = "nomic-embed-text",
|
|
365
|
+
) -> List[Dict[str, Any]]:
|
|
366
|
+
"""
|
|
367
|
+
Vector similarity search on a knowledge base.
|
|
368
|
+
|
|
369
|
+
Args:
|
|
370
|
+
name: Knowledge base name
|
|
371
|
+
query_text: Search query text
|
|
372
|
+
n_results: Number of results to return
|
|
373
|
+
embed_model: Embedding model to use for query
|
|
374
|
+
|
|
375
|
+
Returns:
|
|
376
|
+
List of dicts: [{content, relevance, source, chunk_index}]
|
|
377
|
+
"""
|
|
378
|
+
collection = self._collections.get(name)
|
|
379
|
+
if collection is None:
|
|
380
|
+
raise KnowledgeError(f"Knowledge base '{name}' not found. Define it with <q:knowledge> first.")
|
|
381
|
+
|
|
382
|
+
if collection.count() == 0:
|
|
383
|
+
return []
|
|
384
|
+
|
|
385
|
+
# Generate query embedding
|
|
386
|
+
query_embedding = self._generate_embeddings([query_text], embed_model)
|
|
387
|
+
if not query_embedding:
|
|
388
|
+
return []
|
|
389
|
+
|
|
390
|
+
# Search ChromaDB
|
|
391
|
+
results = collection.query(
|
|
392
|
+
query_embeddings=query_embedding,
|
|
393
|
+
n_results=min(n_results, collection.count()),
|
|
394
|
+
include=["documents", "metadatas", "distances"],
|
|
395
|
+
)
|
|
396
|
+
|
|
397
|
+
# Format results
|
|
398
|
+
formatted = []
|
|
399
|
+
documents = results.get("documents", [[]])[0]
|
|
400
|
+
metadatas = results.get("metadatas", [[]])[0]
|
|
401
|
+
distances = results.get("distances", [[]])[0]
|
|
402
|
+
|
|
403
|
+
for doc, meta, dist in zip(documents, metadatas, distances):
|
|
404
|
+
# ChromaDB cosine distance: 0 = identical, 2 = opposite
|
|
405
|
+
# Convert to relevance score: 1.0 = perfect match, 0.0 = no match
|
|
406
|
+
relevance = max(0.0, 1.0 - dist / 2.0)
|
|
407
|
+
formatted.append({
|
|
408
|
+
"content": doc,
|
|
409
|
+
"relevance": round(relevance, 4),
|
|
410
|
+
"source": meta.get("source", "unknown"),
|
|
411
|
+
"chunk_index": meta.get("chunk_index", 0),
|
|
412
|
+
})
|
|
413
|
+
|
|
414
|
+
return formatted
|
|
415
|
+
|
|
416
|
+
def rag_query(
|
|
417
|
+
self,
|
|
418
|
+
name: str,
|
|
419
|
+
question: str,
|
|
420
|
+
model: Optional[str] = None,
|
|
421
|
+
n_results: int = 5,
|
|
422
|
+
embed_model: str = "nomic-embed-text",
|
|
423
|
+
) -> Dict[str, Any]:
|
|
424
|
+
"""
|
|
425
|
+
Full RAG pipeline: search + LLM answer.
|
|
426
|
+
|
|
427
|
+
Args:
|
|
428
|
+
name: Knowledge base name
|
|
429
|
+
question: User question
|
|
430
|
+
model: LLM model for answer generation
|
|
431
|
+
n_results: Number of chunks to retrieve for context
|
|
432
|
+
embed_model: Embedding model for search
|
|
433
|
+
|
|
434
|
+
Returns:
|
|
435
|
+
Dict: {answer, sources, confidence}
|
|
436
|
+
"""
|
|
437
|
+
if self.llm_service is None:
|
|
438
|
+
raise KnowledgeError("LLM service is required for RAG queries. Ensure Ollama is available.")
|
|
439
|
+
|
|
440
|
+
# Step 1: Vector search
|
|
441
|
+
search_results = self.search(name, question, n_results, embed_model)
|
|
442
|
+
|
|
443
|
+
if not search_results:
|
|
444
|
+
return {
|
|
445
|
+
"answer": "No relevant information found in the knowledge base.",
|
|
446
|
+
"sources": [],
|
|
447
|
+
"confidence": 0.0,
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
# Step 2: Build context from search results
|
|
451
|
+
context_parts = []
|
|
452
|
+
sources = []
|
|
453
|
+
for i, result in enumerate(search_results):
|
|
454
|
+
context_parts.append(f"[{i+1}] {result['content']}")
|
|
455
|
+
src = result.get('source', 'unknown')
|
|
456
|
+
if src not in sources:
|
|
457
|
+
sources.append(src)
|
|
458
|
+
|
|
459
|
+
context = "\n\n".join(context_parts)
|
|
460
|
+
|
|
461
|
+
# Step 3: Generate answer via LLM
|
|
462
|
+
system_prompt = (
|
|
463
|
+
"You are a helpful assistant that answers questions based on the provided context. "
|
|
464
|
+
"Only use information from the context below. If the context doesn't contain "
|
|
465
|
+
"enough information to answer, say so. Be concise and accurate."
|
|
466
|
+
)
|
|
467
|
+
|
|
468
|
+
prompt = (
|
|
469
|
+
f"Context:\n{context}\n\n"
|
|
470
|
+
f"Question: {question}\n\n"
|
|
471
|
+
f"Answer:"
|
|
472
|
+
)
|
|
473
|
+
|
|
474
|
+
try:
|
|
475
|
+
llm_result = self.llm_service.generate(
|
|
476
|
+
prompt=prompt,
|
|
477
|
+
model=model,
|
|
478
|
+
system=system_prompt,
|
|
479
|
+
temperature=0.3,
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
answer = llm_result.get("data", "")
|
|
483
|
+
success = llm_result.get("success", False)
|
|
484
|
+
|
|
485
|
+
# Calculate confidence based on search relevance
|
|
486
|
+
avg_relevance = sum(r['relevance'] for r in search_results) / len(search_results) if search_results else 0.0
|
|
487
|
+
|
|
488
|
+
return {
|
|
489
|
+
"answer": answer.strip() if answer else "Failed to generate answer.",
|
|
490
|
+
"sources": sources,
|
|
491
|
+
"confidence": round(avg_relevance, 4),
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
except Exception as e:
|
|
495
|
+
logger.error(f"RAG query failed for '{name}': {e}")
|
|
496
|
+
return {
|
|
497
|
+
"answer": f"Error generating answer: {e}",
|
|
498
|
+
"sources": sources,
|
|
499
|
+
"confidence": 0.0,
|
|
500
|
+
}
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Quantum LLM Cache - TTL cache for q:llm responses.
|
|
3
|
+
|
|
4
|
+
q:llm has carried a `cache` attribute since it was written, wired to nothing.
|
|
5
|
+
LLM calls are the slowest and most expensive thing a Quantum component can do,
|
|
6
|
+
so a cache is not a nicety — a repeated prompt during development costs seconds
|
|
7
|
+
and, on a paid provider, money.
|
|
8
|
+
|
|
9
|
+
Keyed on everything that can change the answer: provider, model, endpoint, the
|
|
10
|
+
full message list (or prompt+system), temperature, max_tokens and the requested
|
|
11
|
+
response format. Two calls that differ in any of those are different calls.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import hashlib
|
|
15
|
+
import json
|
|
16
|
+
import threading
|
|
17
|
+
import time
|
|
18
|
+
from typing import Any, Dict, Optional
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class LLMCache:
|
|
22
|
+
"""In-process TTL cache for LLM responses."""
|
|
23
|
+
|
|
24
|
+
def __init__(self, max_entries: int = 500):
|
|
25
|
+
self._entries: Dict[str, Dict[str, Any]] = {}
|
|
26
|
+
self._lock = threading.Lock()
|
|
27
|
+
self.max_entries = max_entries
|
|
28
|
+
self.hits = 0
|
|
29
|
+
self.misses = 0
|
|
30
|
+
|
|
31
|
+
@staticmethod
|
|
32
|
+
def build_key(**parts: Any) -> str:
|
|
33
|
+
"""Build a stable key from the call's distinguishing parameters."""
|
|
34
|
+
payload = json.dumps(parts, sort_keys=True, default=str)
|
|
35
|
+
return hashlib.sha256(payload.encode('utf-8')).hexdigest()
|
|
36
|
+
|
|
37
|
+
def get(self, key: str) -> Optional[Dict[str, Any]]:
|
|
38
|
+
"""Return a cached response, or None if absent or expired."""
|
|
39
|
+
with self._lock:
|
|
40
|
+
entry = self._entries.get(key)
|
|
41
|
+
if entry is None:
|
|
42
|
+
self.misses += 1
|
|
43
|
+
return None
|
|
44
|
+
|
|
45
|
+
expires_at = entry['expires_at']
|
|
46
|
+
if expires_at is not None and time.time() > expires_at:
|
|
47
|
+
del self._entries[key]
|
|
48
|
+
self.misses += 1
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
self.hits += 1
|
|
52
|
+
# Copy so a caller mutating the result cannot poison the cache
|
|
53
|
+
return dict(entry['value'])
|
|
54
|
+
|
|
55
|
+
def set(self, key: str, value: Dict[str, Any], ttl: Optional[int] = None):
|
|
56
|
+
"""Store a response. ttl in seconds; None means no expiry."""
|
|
57
|
+
with self._lock:
|
|
58
|
+
if len(self._entries) >= self.max_entries:
|
|
59
|
+
# Evict the entry closest to expiry (None sorts last)
|
|
60
|
+
oldest = min(
|
|
61
|
+
self._entries.items(),
|
|
62
|
+
key=lambda kv: (kv[1]['expires_at'] is None, kv[1]['expires_at'] or 0),
|
|
63
|
+
)
|
|
64
|
+
del self._entries[oldest[0]]
|
|
65
|
+
|
|
66
|
+
self._entries[key] = {
|
|
67
|
+
'value': dict(value),
|
|
68
|
+
'expires_at': (time.time() + ttl) if ttl else None,
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
def clear(self):
|
|
72
|
+
with self._lock:
|
|
73
|
+
self._entries.clear()
|
|
74
|
+
|
|
75
|
+
def stats(self) -> Dict[str, Any]:
|
|
76
|
+
with self._lock:
|
|
77
|
+
total = self.hits + self.misses
|
|
78
|
+
return {
|
|
79
|
+
'entries': len(self._entries),
|
|
80
|
+
'hits': self.hits,
|
|
81
|
+
'misses': self.misses,
|
|
82
|
+
'hitRate': round(self.hits / total, 4) if total else 0.0,
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
_llm_cache: Optional[LLMCache] = None
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def get_llm_cache() -> LLMCache:
|
|
90
|
+
"""Get the process-wide LLM cache."""
|
|
91
|
+
global _llm_cache
|
|
92
|
+
if _llm_cache is None:
|
|
93
|
+
_llm_cache = LLMCache()
|
|
94
|
+
return _llm_cache
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def reset_llm_cache():
|
|
98
|
+
"""Reset the global cache (for tests)."""
|
|
99
|
+
global _llm_cache
|
|
100
|
+
_llm_cache = None
|