vectorwave 0.2.6__cp312-cp312-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vectorwave/__init__.py +29 -0
- vectorwave/batch/__init__.py +0 -0
- vectorwave/batch/batch.py +176 -0
- vectorwave/core/__init__.py +0 -0
- vectorwave/core/auto_injector.py +100 -0
- vectorwave/core/core.py +0 -0
- vectorwave/core/decorator.py +191 -0
- vectorwave/core/generator.py +140 -0
- vectorwave/core/llm/__init__.py +0 -0
- vectorwave/core/llm/base.py +47 -0
- vectorwave/core/llm/factory.py +13 -0
- vectorwave/core/llm/openai_client.py +79 -0
- vectorwave/database/__init__.py +0 -0
- vectorwave/database/archiver.py +100 -0
- vectorwave/database/dataset.py +150 -0
- vectorwave/database/db.py +413 -0
- vectorwave/database/db_search.py +496 -0
- vectorwave/exception/__init__.py +0 -0
- vectorwave/exception/exceptions.py +22 -0
- vectorwave/models/__init__.py +0 -0
- vectorwave/models/db_config.py +143 -0
- vectorwave/monitoring/__init__.py +0 -0
- vectorwave/monitoring/alert/__init__.py +0 -0
- vectorwave/monitoring/alert/base.py +8 -0
- vectorwave/monitoring/alert/factory.py +21 -0
- vectorwave/monitoring/alert/null_alerter.py +7 -0
- vectorwave/monitoring/alert/webhook_alerter.py +69 -0
- vectorwave/monitoring/monitoring.py +0 -0
- vectorwave/monitoring/tracer.py +515 -0
- vectorwave/prediction/__init__.py +0 -0
- vectorwave/prediction/predictor.py +0 -0
- vectorwave/search/__init__.py +0 -0
- vectorwave/search/execution_search.py +154 -0
- vectorwave/search/extended_search.py +0 -0
- vectorwave/search/rag_search.py +154 -0
- vectorwave/utils/__init__.py +0 -0
- vectorwave/utils/context.py +4 -0
- vectorwave/utils/function_cache.py +96 -0
- vectorwave/utils/healer.py +155 -0
- vectorwave/utils/replayer.py +265 -0
- vectorwave/utils/replayer_semantic.py +234 -0
- vectorwave/utils/return_caching_utils.py +152 -0
- vectorwave/utils/status.py +55 -0
- vectorwave/vectorizer/__init__.py +0 -0
- vectorwave/vectorizer/base.py +12 -0
- vectorwave/vectorizer/factory.py +51 -0
- vectorwave/vectorizer/huggingface_vectorizer.py +34 -0
- vectorwave/vectorizer/openai_vectorizer.py +46 -0
- vectorwave/vectorwave_core.cp312-win_amd64.pyd +0 -0
- vectorwave-0.2.6.dist-info/METADATA +60 -0
- vectorwave-0.2.6.dist-info/RECORD +54 -0
- vectorwave-0.2.6.dist-info/WHEEL +4 -0
- vectorwave-0.2.6.dist-info/licenses/LICENSE +21 -0
- vectorwave-0.2.6.dist-info/licenses/NOTICE +31 -0
vectorwave/__init__.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
from .core.decorator import vectorize
|
|
2
|
+
from .database.db import initialize_database
|
|
3
|
+
from .database.db_search import search_functions, search_executions, search_errors_by_message, search_functions_hybrid
|
|
4
|
+
from .monitoring.tracer import trace_span
|
|
5
|
+
from .search.rag_search import search_and_answer, analyze_trace_log
|
|
6
|
+
from .core.generator import generate_and_register_metadata
|
|
7
|
+
from .utils.healer import VectorWaveHealer
|
|
8
|
+
from .utils.replayer import VectorWaveReplayer
|
|
9
|
+
from .utils.replayer_semantic import SemanticReplayer
|
|
10
|
+
from .database.dataset import VectorWaveDatasetManager
|
|
11
|
+
from .core.auto_injector import VectorWaveAutoInjector
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
'vectorize',
|
|
15
|
+
'initialize_database',
|
|
16
|
+
'search_functions',
|
|
17
|
+
'search_functions_hybrid',
|
|
18
|
+
'search_executions',
|
|
19
|
+
'search_errors_by_message',
|
|
20
|
+
'trace_span',
|
|
21
|
+
'search_and_answer',
|
|
22
|
+
'analyze_trace_log',
|
|
23
|
+
'generate_and_register_metadata',
|
|
24
|
+
'VectorWaveHealer',
|
|
25
|
+
'VectorWaveReplayer',
|
|
26
|
+
'SemanticReplayer',
|
|
27
|
+
'VectorWaveDatasetManager',
|
|
28
|
+
'VectorWaveAutoInjector'
|
|
29
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
import weaviate
|
|
2
|
+
import atexit
|
|
3
|
+
import logging
|
|
4
|
+
import threading
|
|
5
|
+
import queue
|
|
6
|
+
import time
|
|
7
|
+
from functools import lru_cache
|
|
8
|
+
from typing import Optional, List, Dict, Any
|
|
9
|
+
|
|
10
|
+
from ..models.db_config import get_weaviate_settings, WeaviateSettings
|
|
11
|
+
from ..database.db import get_weaviate_client
|
|
12
|
+
|
|
13
|
+
# Rust Core 모듈 Import 시도
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
try:
|
|
19
|
+
from vectorwave.vectorwave_core import RustBatchManager
|
|
20
|
+
USE_RUST_CORE = True
|
|
21
|
+
except ImportError:
|
|
22
|
+
USE_RUST_CORE = False
|
|
23
|
+
|
|
24
|
+
class WeaviateBatchManager:
|
|
25
|
+
"""
|
|
26
|
+
Manages Weaviate batch imports.
|
|
27
|
+
Uses High-Performance Rust Core if available, otherwise falls back to Python.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self):
|
|
31
|
+
self._initialized = False
|
|
32
|
+
self.settings: WeaviateSettings = get_weaviate_settings()
|
|
33
|
+
self.client: Optional[weaviate.WeaviateClient] = None
|
|
34
|
+
|
|
35
|
+
# Batch Configuration
|
|
36
|
+
self.batch_threshold = self.settings.BATCH_THRESHOLD
|
|
37
|
+
self.flush_interval = self.settings.FLUSH_INTERVAL_SECONDS
|
|
38
|
+
|
|
39
|
+
# Connect to DB
|
|
40
|
+
self._connect_client()
|
|
41
|
+
|
|
42
|
+
if USE_RUST_CORE:
|
|
43
|
+
logger.info(f"🚀 [VectorWave] Rust Core Activated! (Threshold: {self.batch_threshold}, Interval: {self.flush_interval}s)")
|
|
44
|
+
self._rust_manager = RustBatchManager(
|
|
45
|
+
self._flush_batch_core,
|
|
46
|
+
self.batch_threshold,
|
|
47
|
+
int(self.flush_interval * 1000) # ms 단위 변환
|
|
48
|
+
)
|
|
49
|
+
self._worker_thread = None
|
|
50
|
+
else:
|
|
51
|
+
logger.warning("⚠️ [VectorWave] Rust Core not found. Using slower Python implementation.")
|
|
52
|
+
# --- Legacy Python Implementation ---
|
|
53
|
+
self.queue = queue.Queue(maxsize=10000)
|
|
54
|
+
self._stop_event = threading.Event()
|
|
55
|
+
self._start_python_worker()
|
|
56
|
+
|
|
57
|
+
# Register shutdown handler
|
|
58
|
+
atexit.register(self.shutdown)
|
|
59
|
+
|
|
60
|
+
def _connect_client(self):
|
|
61
|
+
"""Attempts to connect to Weaviate."""
|
|
62
|
+
try:
|
|
63
|
+
self.client = get_weaviate_client(self.settings)
|
|
64
|
+
if self.client:
|
|
65
|
+
self._initialized = True
|
|
66
|
+
except Exception as e:
|
|
67
|
+
logger.warning(f"Initial DB connection failed: {e}")
|
|
68
|
+
self._initialized = False
|
|
69
|
+
|
|
70
|
+
def _start_python_worker(self):
|
|
71
|
+
"""Starts the legacy Python background thread."""
|
|
72
|
+
self._worker_thread = threading.Thread(target=self._python_worker_loop, daemon=True)
|
|
73
|
+
self._worker_thread.start()
|
|
74
|
+
|
|
75
|
+
def add_object(self, collection: str, properties: dict, uuid: str = None, vector: Optional[List[float]] = None):
|
|
76
|
+
"""
|
|
77
|
+
[Public API] Adds an object to the batch queue.
|
|
78
|
+
"""
|
|
79
|
+
if USE_RUST_CORE:
|
|
80
|
+
|
|
81
|
+
self._rust_manager.add_object(collection, properties, uuid, vector)
|
|
82
|
+
else:
|
|
83
|
+
# Python Legacy Queue
|
|
84
|
+
item = {
|
|
85
|
+
"collection": collection,
|
|
86
|
+
"properties": properties,
|
|
87
|
+
"uuid": uuid,
|
|
88
|
+
"vector": vector
|
|
89
|
+
}
|
|
90
|
+
try:
|
|
91
|
+
self.queue.put_nowait(item)
|
|
92
|
+
except queue.Full:
|
|
93
|
+
logger.warning("🚨 VectorWave Log Queue is FULL. Dropping log.")
|
|
94
|
+
|
|
95
|
+
def _flush_batch_core(self, items: List[Dict[str, Any]]):
|
|
96
|
+
"""
|
|
97
|
+
The actual flush logic called by either Rust or Python worker.
|
|
98
|
+
"""
|
|
99
|
+
if not items:
|
|
100
|
+
return
|
|
101
|
+
|
|
102
|
+
# 1. Check/Retry Connection
|
|
103
|
+
if not self._initialized or not self.client:
|
|
104
|
+
self._connect_client()
|
|
105
|
+
if not self._initialized:
|
|
106
|
+
return
|
|
107
|
+
|
|
108
|
+
# 2. Send Batch via Weaviate Client
|
|
109
|
+
try:
|
|
110
|
+
# Weaviate v4 batch context
|
|
111
|
+
with self.client.batch.dynamic() as batch:
|
|
112
|
+
for item in items:
|
|
113
|
+
batch.add_object(
|
|
114
|
+
collection=item['collection'],
|
|
115
|
+
properties=item['properties'],
|
|
116
|
+
uuid=item.get('uuid'),
|
|
117
|
+
vector=item.get('vector')
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
if len(self.client.batch.failed_objects) > 0:
|
|
121
|
+
for failed in self.client.batch.failed_objects:
|
|
122
|
+
logger.error(f"⚠️ Batch Item Failed: {failed.message}")
|
|
123
|
+
|
|
124
|
+
except RuntimeError:
|
|
125
|
+
return
|
|
126
|
+
except Exception as e:
|
|
127
|
+
msg = str(e).lower()
|
|
128
|
+
if "shutdown" in msg or "closed" in msg:
|
|
129
|
+
return
|
|
130
|
+
logger.error(f"❌ Batch Flush Error: {e}")
|
|
131
|
+
|
|
132
|
+
# --- Legacy Python Worker Methods (Only used if Rust is missing) ---
|
|
133
|
+
def _python_worker_loop(self):
|
|
134
|
+
pending_items = []
|
|
135
|
+
last_flush_time = time.time()
|
|
136
|
+
|
|
137
|
+
while not self._stop_event.is_set():
|
|
138
|
+
try:
|
|
139
|
+
item = self.queue.get(timeout=0.5)
|
|
140
|
+
pending_items.append(item)
|
|
141
|
+
except queue.Empty:
|
|
142
|
+
pass
|
|
143
|
+
|
|
144
|
+
current_time = time.time()
|
|
145
|
+
if len(pending_items) >= self.batch_threshold or (pending_items and current_time - last_flush_time >= self.flush_interval):
|
|
146
|
+
self._flush_batch_core(pending_items)
|
|
147
|
+
pending_items = []
|
|
148
|
+
last_flush_time = current_time
|
|
149
|
+
|
|
150
|
+
def shutdown(self):
|
|
151
|
+
"""Gracefully shuts down."""
|
|
152
|
+
if USE_RUST_CORE:
|
|
153
|
+
self._rust_manager.shutdown()
|
|
154
|
+
else:
|
|
155
|
+
if not self._stop_event.is_set():
|
|
156
|
+
self._stop_event.set()
|
|
157
|
+
if self._worker_thread and self._worker_thread.is_alive():
|
|
158
|
+
self._worker_thread.join(timeout=1.0)
|
|
159
|
+
|
|
160
|
+
# Flush remaining items
|
|
161
|
+
remaining = []
|
|
162
|
+
while not self.queue.empty():
|
|
163
|
+
remaining.append(self.queue.get_nowait())
|
|
164
|
+
if remaining:
|
|
165
|
+
self._flush_batch_core(remaining)
|
|
166
|
+
|
|
167
|
+
# Close client
|
|
168
|
+
if self.client:
|
|
169
|
+
try:
|
|
170
|
+
self.client.close()
|
|
171
|
+
except:
|
|
172
|
+
pass
|
|
173
|
+
|
|
174
|
+
@lru_cache(None)
|
|
175
|
+
def get_batch_manager() -> WeaviateBatchManager:
|
|
176
|
+
return WeaviateBatchManager()
|
|
File without changes
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
import inspect
|
|
2
|
+
import importlib
|
|
3
|
+
import logging
|
|
4
|
+
from functools import wraps
|
|
5
|
+
from typing import Any, Optional
|
|
6
|
+
|
|
7
|
+
# VectorWave Core Modules
|
|
8
|
+
from .decorator import vectorize
|
|
9
|
+
from ..monitoring.tracer import trace_span, current_tracer_var
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
def create_smart_wrapper(original_func, root_wrapper):
|
|
14
|
+
"""
|
|
15
|
+
Creates a wrapper that switches behavior based on the Tracing Context.
|
|
16
|
+
|
|
17
|
+
Args:
|
|
18
|
+
original_func: The original function (used when acting as a Child Span).
|
|
19
|
+
root_wrapper: The wrapper generated by vectorize (used when acting as a Root Span, includes metadata).
|
|
20
|
+
"""
|
|
21
|
+
@wraps(original_func)
|
|
22
|
+
def wrapper(*args, **kwargs):
|
|
23
|
+
# Check for existing tracer at runtime
|
|
24
|
+
tracer = current_tracer_var.get()
|
|
25
|
+
|
|
26
|
+
if tracer:
|
|
27
|
+
# [Case A] Parent trace exists -> Act as Child Span
|
|
28
|
+
# (Wrap original function with trace_span and execute)
|
|
29
|
+
return trace_span(original_func, capture_return_value=True)(*args, **kwargs)
|
|
30
|
+
else:
|
|
31
|
+
# [Case B] No parent trace -> Act as Root Span
|
|
32
|
+
# (Delegate to the already configured root_wrapper)
|
|
33
|
+
return root_wrapper(*args, **kwargs)
|
|
34
|
+
|
|
35
|
+
# Mark to prevent double injection
|
|
36
|
+
wrapper._is_vectorized = True
|
|
37
|
+
return wrapper
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class VectorWaveAutoInjector:
|
|
41
|
+
"""
|
|
42
|
+
Scans specified modules or packages to automatically inject (Monkey Patch) VectorWave functionality.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
# Global default configuration
|
|
46
|
+
_default_config = {}
|
|
47
|
+
|
|
48
|
+
@classmethod
|
|
49
|
+
def configure(cls, **kwargs):
|
|
50
|
+
"""
|
|
51
|
+
Sets the default configuration to be applied to all inject calls.
|
|
52
|
+
"""
|
|
53
|
+
cls._default_config.update(kwargs)
|
|
54
|
+
logger.info(f"⚙️ VectorWave AutoInjector Configured: {cls._default_config}")
|
|
55
|
+
|
|
56
|
+
@classmethod
|
|
57
|
+
def inject(cls, target_module_path: str, recursive: bool = False, **config):
|
|
58
|
+
"""
|
|
59
|
+
Finds functions within a module and injects VectorWave functionality.
|
|
60
|
+
**Metadata registration (DB save or Pending) is performed immediately at this point.**
|
|
61
|
+
"""
|
|
62
|
+
try:
|
|
63
|
+
module = importlib.import_module(target_module_path)
|
|
64
|
+
except ImportError as e:
|
|
65
|
+
logger.error(f"Failed to import module '{target_module_path}': {e}")
|
|
66
|
+
return None
|
|
67
|
+
|
|
68
|
+
# Merge configurations (Defaults < Individual Config)
|
|
69
|
+
final_config = cls._default_config.copy()
|
|
70
|
+
final_config.update(config)
|
|
71
|
+
|
|
72
|
+
logger.info(f"🌊 [AutoInjector] Injecting VectorWave into: {module.__name__}")
|
|
73
|
+
|
|
74
|
+
patched_count = 0
|
|
75
|
+
|
|
76
|
+
for name, obj in inspect.getmembers(module):
|
|
77
|
+
# Target only functions defined within the module
|
|
78
|
+
if inspect.isfunction(obj) and obj.__module__ == module.__name__:
|
|
79
|
+
|
|
80
|
+
# Skip if already processed
|
|
81
|
+
if getattr(obj, "_is_vectorized", False):
|
|
82
|
+
continue
|
|
83
|
+
|
|
84
|
+
logger.info(f" └─ Auto-wiring: {name}()")
|
|
85
|
+
|
|
86
|
+
# [Key Change] Call vectorize immediately at injection time to perform metadata registration!
|
|
87
|
+
# - If auto=False: Immediate DB save
|
|
88
|
+
# - If auto=True: Add to PENDING list
|
|
89
|
+
# The returned root_wrapper contains the execution logic (Root Span)
|
|
90
|
+
root_wrapper = vectorize(**final_config)(obj)
|
|
91
|
+
|
|
92
|
+
# Create smart wrapper to judge Context at runtime
|
|
93
|
+
smart_wrapper = create_smart_wrapper(obj, root_wrapper)
|
|
94
|
+
|
|
95
|
+
# Replace (Monkey Patching)
|
|
96
|
+
setattr(module, name, smart_wrapper)
|
|
97
|
+
patched_count += 1
|
|
98
|
+
|
|
99
|
+
logger.info(f"✨ Injection Complete. {patched_count} functions registered & auto-wired.")
|
|
100
|
+
return module
|
vectorwave/core/core.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
import inspect
|
|
2
|
+
import logging
|
|
3
|
+
from functools import wraps
|
|
4
|
+
from typing import List, Optional, Dict, Any
|
|
5
|
+
|
|
6
|
+
from weaviate.util import generate_uuid5
|
|
7
|
+
|
|
8
|
+
from ..batch.batch import get_batch_manager
|
|
9
|
+
from ..models.db_config import get_weaviate_settings
|
|
10
|
+
from ..monitoring.tracer import trace_root, trace_span
|
|
11
|
+
from ..utils.function_cache import function_cache_manager
|
|
12
|
+
from ..utils.return_caching_utils import _check_and_return_cached_result
|
|
13
|
+
from ..vectorizer.factory import get_vectorizer
|
|
14
|
+
from ..utils.context import execution_source_context
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
PENDING_FUNCTIONS: List[Dict[str, Any]] = []
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def vectorize(search_description: Optional[str] = None,
|
|
22
|
+
sequence_narrative: Optional[str] = None,
|
|
23
|
+
auto: bool = False,
|
|
24
|
+
capture_return_value: bool = False,
|
|
25
|
+
semantic_cache: bool = False,
|
|
26
|
+
cache_threshold: float = 0.9,
|
|
27
|
+
replay: bool = False,
|
|
28
|
+
attributes_to_capture: Optional[List[str]] = None,
|
|
29
|
+
**execution_tags):
|
|
30
|
+
"""
|
|
31
|
+
VectorWave Decorator with Auto-Generation support.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
if semantic_cache:
|
|
35
|
+
if get_vectorizer() is None:
|
|
36
|
+
logger.warning(
|
|
37
|
+
f"Semantic caching requested for '{search_description}' but no Python vectorizer is configured. "
|
|
38
|
+
f"Disabling semantic_cache."
|
|
39
|
+
)
|
|
40
|
+
semantic_cache = False
|
|
41
|
+
|
|
42
|
+
if semantic_cache and not capture_return_value:
|
|
43
|
+
capture_return_value = True
|
|
44
|
+
|
|
45
|
+
if replay and not capture_return_value:
|
|
46
|
+
capture_return_value = True
|
|
47
|
+
|
|
48
|
+
def decorator(func):
|
|
49
|
+
is_async_func = inspect.iscoroutinefunction(func)
|
|
50
|
+
|
|
51
|
+
module_name = func.__module__
|
|
52
|
+
function_name = func.__name__
|
|
53
|
+
func_identifier = f"{module_name}.{function_name}"
|
|
54
|
+
func_uuid = generate_uuid5(func_identifier)
|
|
55
|
+
|
|
56
|
+
# Prepare attributes to capture
|
|
57
|
+
final_attributes = ['function_uuid', 'team', 'priority', 'run_id', 'exec_source']
|
|
58
|
+
if attributes_to_capture:
|
|
59
|
+
for attr in attributes_to_capture:
|
|
60
|
+
if attr not in final_attributes:
|
|
61
|
+
final_attributes.append(attr)
|
|
62
|
+
|
|
63
|
+
if replay:
|
|
64
|
+
try:
|
|
65
|
+
sig = inspect.signature(func)
|
|
66
|
+
for param_name in sig.parameters:
|
|
67
|
+
if param_name not in ('self', 'cls') and param_name not in final_attributes:
|
|
68
|
+
final_attributes.append(param_name)
|
|
69
|
+
except Exception as e:
|
|
70
|
+
logger.warning(f"Failed to inspect signature for replay auto-capture in '{function_name}': {e}")
|
|
71
|
+
|
|
72
|
+
# Extract Execution Tags
|
|
73
|
+
valid_execution_tags = {}
|
|
74
|
+
settings = get_weaviate_settings()
|
|
75
|
+
if execution_tags and settings.custom_properties:
|
|
76
|
+
allowed_keys = set(settings.custom_properties.keys())
|
|
77
|
+
for key, value in execution_tags.items():
|
|
78
|
+
if key in allowed_keys:
|
|
79
|
+
valid_execution_tags[key] = value
|
|
80
|
+
else:
|
|
81
|
+
logger.warning(
|
|
82
|
+
"Function '%s' has undefined execution_tag: '%s'. "
|
|
83
|
+
"This tag will be IGNORED. Please add it to your .weaviate_properties file.",
|
|
84
|
+
function_name,
|
|
85
|
+
key
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
try:
|
|
89
|
+
# define static properties
|
|
90
|
+
docstring = inspect.getdoc(func) or ""
|
|
91
|
+
source_code = inspect.getsource(func)
|
|
92
|
+
|
|
93
|
+
static_properties = {
|
|
94
|
+
"function_name": function_name,
|
|
95
|
+
"module_name": module_name,
|
|
96
|
+
"docstring": docstring,
|
|
97
|
+
"source_code": source_code,
|
|
98
|
+
"search_description": search_description,
|
|
99
|
+
"sequence_narrative": sequence_narrative
|
|
100
|
+
}
|
|
101
|
+
static_properties.update(valid_execution_tags)
|
|
102
|
+
|
|
103
|
+
if auto:
|
|
104
|
+
logger.info(f"Function '{function_name}' registered for auto-metadata generation.")
|
|
105
|
+
PENDING_FUNCTIONS.append({
|
|
106
|
+
"func_name": function_name,
|
|
107
|
+
"func_uuid": func_uuid,
|
|
108
|
+
"func_identifier": func_identifier,
|
|
109
|
+
"static_properties": static_properties
|
|
110
|
+
})
|
|
111
|
+
else:
|
|
112
|
+
# [Existing] Immediate Registration Mode
|
|
113
|
+
current_content_hash = function_cache_manager.calculate_content_hash(func_identifier, static_properties)
|
|
114
|
+
|
|
115
|
+
if function_cache_manager.is_cached_and_unchanged(func_uuid, current_content_hash):
|
|
116
|
+
logger.info(f"Function '{function_name}' is UNCHANGED. Skipping DB write.")
|
|
117
|
+
else:
|
|
118
|
+
logger.info(f"Function '{function_name}' is NEW or CHANGED. Writing to DB.")
|
|
119
|
+
batch = get_batch_manager()
|
|
120
|
+
vectorizer = get_vectorizer()
|
|
121
|
+
vector_to_add = None
|
|
122
|
+
|
|
123
|
+
if vectorizer and search_description:
|
|
124
|
+
try:
|
|
125
|
+
vector_to_add = vectorizer.embed(search_description)
|
|
126
|
+
except Exception as e:
|
|
127
|
+
logger.warning(f"Failed to vectorize '{function_name}': {e}")
|
|
128
|
+
|
|
129
|
+
batch.add_object(
|
|
130
|
+
collection=settings.COLLECTION_NAME,
|
|
131
|
+
properties=static_properties,
|
|
132
|
+
uuid=func_uuid,
|
|
133
|
+
vector=vector_to_add
|
|
134
|
+
)
|
|
135
|
+
function_cache_manager.update_cache(func_uuid, current_content_hash)
|
|
136
|
+
|
|
137
|
+
except Exception as e:
|
|
138
|
+
logger.error("Error in @vectorize setup for '%s': %s", func.__name__, e)
|
|
139
|
+
|
|
140
|
+
# --- Wrapper Logic ---
|
|
141
|
+
|
|
142
|
+
if is_async_func:
|
|
143
|
+
@trace_root()
|
|
144
|
+
@trace_span(attributes_to_capture=final_attributes, capture_return_value=capture_return_value)
|
|
145
|
+
@wraps(func)
|
|
146
|
+
async def inner_wrapper(*args, **kwargs):
|
|
147
|
+
# Remove injected tags from kwargs before calling original func
|
|
148
|
+
clean_kwargs = {k: v for k, v in kwargs.items() if
|
|
149
|
+
k not in valid_execution_tags and k != 'function_uuid' and k != 'exec_source'}
|
|
150
|
+
return await func(*args, **clean_kwargs)
|
|
151
|
+
|
|
152
|
+
@wraps(func)
|
|
153
|
+
async def outer_wrapper(*args, **kwargs):
|
|
154
|
+
if semantic_cache:
|
|
155
|
+
cached = _check_and_return_cached_result(func, args, kwargs, function_name, cache_threshold, True)
|
|
156
|
+
if cached is not None: return cached
|
|
157
|
+
|
|
158
|
+
full_kwargs = kwargs.copy()
|
|
159
|
+
full_kwargs.update(valid_execution_tags)
|
|
160
|
+
full_kwargs['function_uuid'] = func_uuid
|
|
161
|
+
full_kwargs['exec_source'] = execution_source_context.get()
|
|
162
|
+
return await inner_wrapper(*args, **full_kwargs)
|
|
163
|
+
|
|
164
|
+
outer_wrapper._is_vectorized = True
|
|
165
|
+
return outer_wrapper
|
|
166
|
+
|
|
167
|
+
else: # Sync wrapper
|
|
168
|
+
@trace_root()
|
|
169
|
+
@trace_span(attributes_to_capture=final_attributes, capture_return_value=capture_return_value)
|
|
170
|
+
@wraps(func)
|
|
171
|
+
def inner_wrapper(*args, **kwargs):
|
|
172
|
+
clean_kwargs = {k: v for k, v in kwargs.items() if
|
|
173
|
+
k not in valid_execution_tags and k != 'function_uuid' and k != 'exec_source'}
|
|
174
|
+
return func(*args, **clean_kwargs)
|
|
175
|
+
|
|
176
|
+
@wraps(func)
|
|
177
|
+
def outer_wrapper(*args, **kwargs):
|
|
178
|
+
if semantic_cache:
|
|
179
|
+
cached = _check_and_return_cached_result(func, args, kwargs, function_name, cache_threshold, False)
|
|
180
|
+
if cached is not None: return cached
|
|
181
|
+
|
|
182
|
+
full_kwargs = kwargs.copy()
|
|
183
|
+
full_kwargs.update(valid_execution_tags)
|
|
184
|
+
full_kwargs['function_uuid'] = func_uuid
|
|
185
|
+
full_kwargs['exec_source'] = execution_source_context.get()
|
|
186
|
+
return inner_wrapper(*args, **full_kwargs)
|
|
187
|
+
|
|
188
|
+
outer_wrapper._is_vectorized = True
|
|
189
|
+
return outer_wrapper
|
|
190
|
+
|
|
191
|
+
return decorator
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import json
|
|
3
|
+
from typing import Optional, Dict, Any
|
|
4
|
+
|
|
5
|
+
from ..utils.function_cache import function_cache_manager
|
|
6
|
+
from ..models.db_config import get_weaviate_settings
|
|
7
|
+
from ..batch.batch import get_batch_manager
|
|
8
|
+
from ..vectorizer.factory import get_vectorizer
|
|
9
|
+
from .decorator import PENDING_FUNCTIONS
|
|
10
|
+
from .llm.factory import get_llm_client
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
try:
|
|
15
|
+
from openai import OpenAI
|
|
16
|
+
except ImportError:
|
|
17
|
+
OpenAI = None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def generate_metadata_via_llm(source_code: str, func_name: str) -> Optional[Dict[str, str]]:
|
|
21
|
+
"""Call LLM to generate description and narrative from source code."""
|
|
22
|
+
settings = get_weaviate_settings()
|
|
23
|
+
client = get_llm_client()
|
|
24
|
+
if not client:
|
|
25
|
+
return None
|
|
26
|
+
|
|
27
|
+
prompt = f"""
|
|
28
|
+
Analyze the Python function below and generate a JSON object with two keys.
|
|
29
|
+
Ensure the values are **single strings**, not nested objects.
|
|
30
|
+
|
|
31
|
+
1. "search_description": A concise summary of what this function does (for vector search).
|
|
32
|
+
2. "sequence_narrative": A brief explanation of the context, inputs, and outputs as a narrative text.
|
|
33
|
+
|
|
34
|
+
Function Name: {func_name}
|
|
35
|
+
Code:
|
|
36
|
+
```python
|
|
37
|
+
{source_code}
|
|
38
|
+
```
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
try:
|
|
42
|
+
# Refactored to use BaseLLMClient interface
|
|
43
|
+
response_text = client.create_chat_completion(
|
|
44
|
+
model="gpt-4o-mini",
|
|
45
|
+
messages=[
|
|
46
|
+
{"role": "system", "content": "You are a technical documentation assistant. Output only JSON."},
|
|
47
|
+
{"role": "user", "content": prompt}
|
|
48
|
+
],
|
|
49
|
+
temperature=0.0,
|
|
50
|
+
response_format={"type": "json_object"},
|
|
51
|
+
category="auto_doc"
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
if response_text:
|
|
55
|
+
return json.loads(response_text)
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
except Exception as e:
|
|
59
|
+
logger.error(f"LLM generation failed for '{func_name}': {e}")
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def generate_and_register_metadata():
|
|
64
|
+
"""
|
|
65
|
+
[Entry Point] Processes all functions in PENDING_FUNCTIONS.
|
|
66
|
+
"""
|
|
67
|
+
if not PENDING_FUNCTIONS:
|
|
68
|
+
logger.info("No pending functions for auto-generation.")
|
|
69
|
+
return
|
|
70
|
+
|
|
71
|
+
logger.info(f"🚀 Processing {len(PENDING_FUNCTIONS)} functions for auto-documentation...")
|
|
72
|
+
|
|
73
|
+
settings = get_weaviate_settings()
|
|
74
|
+
batch = get_batch_manager()
|
|
75
|
+
vectorizer = get_vectorizer()
|
|
76
|
+
|
|
77
|
+
processed_count = 0
|
|
78
|
+
|
|
79
|
+
for item in PENDING_FUNCTIONS:
|
|
80
|
+
func_name = item["func_name"]
|
|
81
|
+
func_uuid = item["func_uuid"]
|
|
82
|
+
func_identifier = item["func_identifier"]
|
|
83
|
+
static_props = item["static_properties"]
|
|
84
|
+
|
|
85
|
+
current_hash = function_cache_manager.calculate_content_hash(func_identifier, static_props)
|
|
86
|
+
|
|
87
|
+
cached_meta = function_cache_manager.get_cached_metadata(func_uuid, current_hash)
|
|
88
|
+
|
|
89
|
+
final_desc = ""
|
|
90
|
+
final_narr = ""
|
|
91
|
+
|
|
92
|
+
if cached_meta:
|
|
93
|
+
logger.info(f"✅ [Cache Hit] Loaded metadata for '{func_name}'.")
|
|
94
|
+
final_desc = cached_meta.get("search_description")
|
|
95
|
+
final_narr = cached_meta.get("sequence_narrative")
|
|
96
|
+
else:
|
|
97
|
+
logger.info(f"🤖 [Auto-Gen] Generating metadata for '{func_name}' via LLM...")
|
|
98
|
+
generated = generate_metadata_via_llm(static_props["source_code"], func_name)
|
|
99
|
+
|
|
100
|
+
if generated:
|
|
101
|
+
final_desc = generated.get("search_description", "")
|
|
102
|
+
final_narr = generated.get("sequence_narrative", "")
|
|
103
|
+
|
|
104
|
+
if not isinstance(final_desc, str):
|
|
105
|
+
final_desc = json.dumps(final_desc, ensure_ascii=False)
|
|
106
|
+
if not isinstance(final_narr, str):
|
|
107
|
+
final_narr = json.dumps(final_narr, ensure_ascii=False)
|
|
108
|
+
|
|
109
|
+
# Update Cache with new metadata
|
|
110
|
+
function_cache_manager.update_cache_with_metadata(
|
|
111
|
+
func_uuid, current_hash,
|
|
112
|
+
{"search_description": final_desc, "sequence_narrative": final_narr}
|
|
113
|
+
)
|
|
114
|
+
else:
|
|
115
|
+
logger.warning(f"⚠️ Skipping registration for '{func_name}' due to generation failure.")
|
|
116
|
+
continue
|
|
117
|
+
|
|
118
|
+
# 3. Update Properties
|
|
119
|
+
static_props["search_description"] = final_desc
|
|
120
|
+
static_props["sequence_narrative"] = final_narr
|
|
121
|
+
|
|
122
|
+
# 4. Vectorize the Description
|
|
123
|
+
vector_to_add = None
|
|
124
|
+
if vectorizer and final_desc:
|
|
125
|
+
try:
|
|
126
|
+
vector_to_add = vectorizer.embed(final_desc)
|
|
127
|
+
except Exception as e:
|
|
128
|
+
logger.warning(f"Vectorization failed for '{func_name}': {e}")
|
|
129
|
+
|
|
130
|
+
# 5. Register to DB
|
|
131
|
+
batch.add_object(
|
|
132
|
+
collection=settings.COLLECTION_NAME,
|
|
133
|
+
properties=static_props,
|
|
134
|
+
uuid=func_uuid,
|
|
135
|
+
vector=vector_to_add
|
|
136
|
+
)
|
|
137
|
+
processed_count += 1
|
|
138
|
+
|
|
139
|
+
PENDING_FUNCTIONS.clear()
|
|
140
|
+
logger.info(f"✨ Auto-generation complete. Registered {processed_count} functions.")
|
|
File without changes
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import List, Dict, Optional
|
|
3
|
+
|
|
4
|
+
class BaseLLMClient(ABC):
|
|
5
|
+
"""
|
|
6
|
+
Abstract interface that all LLM Providers (OpenAI, Anthropic, etc.) must implement.
|
|
7
|
+
Implementations must handle internal token usage logging.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
@abstractmethod
|
|
11
|
+
def create_embedding(self, text: str, model: str, category: str = "default") -> Optional[List[float]]:
|
|
12
|
+
"""
|
|
13
|
+
Generates text embeddings.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
text: The text to embed.
|
|
17
|
+
model: The name of the model to use.
|
|
18
|
+
category: Category for aggregating token usage (e.g., 'execution_log', 'auto_doc').
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
The generated list of embedding vectors (None on failure).
|
|
22
|
+
"""
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
@abstractmethod
|
|
26
|
+
def create_chat_completion(
|
|
27
|
+
self,
|
|
28
|
+
messages: List[Dict],
|
|
29
|
+
model: str,
|
|
30
|
+
temperature: float = 0.1,
|
|
31
|
+
response_format: Optional[Dict] = None,
|
|
32
|
+
category: str = "default"
|
|
33
|
+
) -> Optional[str]:
|
|
34
|
+
"""
|
|
35
|
+
Generates a chat completion (response).
|
|
36
|
+
|
|
37
|
+
Args:
|
|
38
|
+
messages: List of conversation messages [{"role": "user", "content": "..."}].
|
|
39
|
+
model: The name of the model to use.
|
|
40
|
+
temperature: Parameter for controlling generation diversity.
|
|
41
|
+
response_format: Response format (e.g., {"type": "json_object"}).
|
|
42
|
+
category: Category for aggregating token usage (e.g., 'execution_log', 'auto_doc').
|
|
43
|
+
|
|
44
|
+
Returns:
|
|
45
|
+
The generated text response (None on failure).
|
|
46
|
+
"""
|
|
47
|
+
pass
|