vectorwave 0.2.6__cp312-cp312-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. vectorwave/__init__.py +29 -0
  2. vectorwave/batch/__init__.py +0 -0
  3. vectorwave/batch/batch.py +176 -0
  4. vectorwave/core/__init__.py +0 -0
  5. vectorwave/core/auto_injector.py +100 -0
  6. vectorwave/core/core.py +0 -0
  7. vectorwave/core/decorator.py +191 -0
  8. vectorwave/core/generator.py +140 -0
  9. vectorwave/core/llm/__init__.py +0 -0
  10. vectorwave/core/llm/base.py +47 -0
  11. vectorwave/core/llm/factory.py +13 -0
  12. vectorwave/core/llm/openai_client.py +79 -0
  13. vectorwave/database/__init__.py +0 -0
  14. vectorwave/database/archiver.py +100 -0
  15. vectorwave/database/dataset.py +150 -0
  16. vectorwave/database/db.py +413 -0
  17. vectorwave/database/db_search.py +496 -0
  18. vectorwave/exception/__init__.py +0 -0
  19. vectorwave/exception/exceptions.py +22 -0
  20. vectorwave/models/__init__.py +0 -0
  21. vectorwave/models/db_config.py +143 -0
  22. vectorwave/monitoring/__init__.py +0 -0
  23. vectorwave/monitoring/alert/__init__.py +0 -0
  24. vectorwave/monitoring/alert/base.py +8 -0
  25. vectorwave/monitoring/alert/factory.py +21 -0
  26. vectorwave/monitoring/alert/null_alerter.py +7 -0
  27. vectorwave/monitoring/alert/webhook_alerter.py +69 -0
  28. vectorwave/monitoring/monitoring.py +0 -0
  29. vectorwave/monitoring/tracer.py +515 -0
  30. vectorwave/prediction/__init__.py +0 -0
  31. vectorwave/prediction/predictor.py +0 -0
  32. vectorwave/search/__init__.py +0 -0
  33. vectorwave/search/execution_search.py +154 -0
  34. vectorwave/search/extended_search.py +0 -0
  35. vectorwave/search/rag_search.py +154 -0
  36. vectorwave/utils/__init__.py +0 -0
  37. vectorwave/utils/context.py +4 -0
  38. vectorwave/utils/function_cache.py +96 -0
  39. vectorwave/utils/healer.py +155 -0
  40. vectorwave/utils/replayer.py +265 -0
  41. vectorwave/utils/replayer_semantic.py +234 -0
  42. vectorwave/utils/return_caching_utils.py +152 -0
  43. vectorwave/utils/status.py +55 -0
  44. vectorwave/vectorizer/__init__.py +0 -0
  45. vectorwave/vectorizer/base.py +12 -0
  46. vectorwave/vectorizer/factory.py +51 -0
  47. vectorwave/vectorizer/huggingface_vectorizer.py +34 -0
  48. vectorwave/vectorizer/openai_vectorizer.py +46 -0
  49. vectorwave/vectorwave_core.cp312-win_amd64.pyd +0 -0
  50. vectorwave-0.2.6.dist-info/METADATA +60 -0
  51. vectorwave-0.2.6.dist-info/RECORD +54 -0
  52. vectorwave-0.2.6.dist-info/WHEEL +4 -0
  53. vectorwave-0.2.6.dist-info/licenses/LICENSE +21 -0
  54. vectorwave-0.2.6.dist-info/licenses/NOTICE +31 -0
vectorwave/__init__.py ADDED
@@ -0,0 +1,29 @@
1
+ from .core.decorator import vectorize
2
+ from .database.db import initialize_database
3
+ from .database.db_search import search_functions, search_executions, search_errors_by_message, search_functions_hybrid
4
+ from .monitoring.tracer import trace_span
5
+ from .search.rag_search import search_and_answer, analyze_trace_log
6
+ from .core.generator import generate_and_register_metadata
7
+ from .utils.healer import VectorWaveHealer
8
+ from .utils.replayer import VectorWaveReplayer
9
+ from .utils.replayer_semantic import SemanticReplayer
10
+ from .database.dataset import VectorWaveDatasetManager
11
+ from .core.auto_injector import VectorWaveAutoInjector
12
+
13
+ __all__ = [
14
+ 'vectorize',
15
+ 'initialize_database',
16
+ 'search_functions',
17
+ 'search_functions_hybrid',
18
+ 'search_executions',
19
+ 'search_errors_by_message',
20
+ 'trace_span',
21
+ 'search_and_answer',
22
+ 'analyze_trace_log',
23
+ 'generate_and_register_metadata',
24
+ 'VectorWaveHealer',
25
+ 'VectorWaveReplayer',
26
+ 'SemanticReplayer',
27
+ 'VectorWaveDatasetManager',
28
+ 'VectorWaveAutoInjector'
29
+ ]
File without changes
@@ -0,0 +1,176 @@
1
+ import weaviate
2
+ import atexit
3
+ import logging
4
+ import threading
5
+ import queue
6
+ import time
7
+ from functools import lru_cache
8
+ from typing import Optional, List, Dict, Any
9
+
10
+ from ..models.db_config import get_weaviate_settings, WeaviateSettings
11
+ from ..database.db import get_weaviate_client
12
+
13
+ # Rust Core 모듈 Import 시도
14
+
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+ try:
19
+ from vectorwave.vectorwave_core import RustBatchManager
20
+ USE_RUST_CORE = True
21
+ except ImportError:
22
+ USE_RUST_CORE = False
23
+
24
+ class WeaviateBatchManager:
25
+ """
26
+ Manages Weaviate batch imports.
27
+ Uses High-Performance Rust Core if available, otherwise falls back to Python.
28
+ """
29
+
30
+ def __init__(self):
31
+ self._initialized = False
32
+ self.settings: WeaviateSettings = get_weaviate_settings()
33
+ self.client: Optional[weaviate.WeaviateClient] = None
34
+
35
+ # Batch Configuration
36
+ self.batch_threshold = self.settings.BATCH_THRESHOLD
37
+ self.flush_interval = self.settings.FLUSH_INTERVAL_SECONDS
38
+
39
+ # Connect to DB
40
+ self._connect_client()
41
+
42
+ if USE_RUST_CORE:
43
+ logger.info(f"🚀 [VectorWave] Rust Core Activated! (Threshold: {self.batch_threshold}, Interval: {self.flush_interval}s)")
44
+ self._rust_manager = RustBatchManager(
45
+ self._flush_batch_core,
46
+ self.batch_threshold,
47
+ int(self.flush_interval * 1000) # ms 단위 변환
48
+ )
49
+ self._worker_thread = None
50
+ else:
51
+ logger.warning("⚠️ [VectorWave] Rust Core not found. Using slower Python implementation.")
52
+ # --- Legacy Python Implementation ---
53
+ self.queue = queue.Queue(maxsize=10000)
54
+ self._stop_event = threading.Event()
55
+ self._start_python_worker()
56
+
57
+ # Register shutdown handler
58
+ atexit.register(self.shutdown)
59
+
60
+ def _connect_client(self):
61
+ """Attempts to connect to Weaviate."""
62
+ try:
63
+ self.client = get_weaviate_client(self.settings)
64
+ if self.client:
65
+ self._initialized = True
66
+ except Exception as e:
67
+ logger.warning(f"Initial DB connection failed: {e}")
68
+ self._initialized = False
69
+
70
+ def _start_python_worker(self):
71
+ """Starts the legacy Python background thread."""
72
+ self._worker_thread = threading.Thread(target=self._python_worker_loop, daemon=True)
73
+ self._worker_thread.start()
74
+
75
+ def add_object(self, collection: str, properties: dict, uuid: str = None, vector: Optional[List[float]] = None):
76
+ """
77
+ [Public API] Adds an object to the batch queue.
78
+ """
79
+ if USE_RUST_CORE:
80
+
81
+ self._rust_manager.add_object(collection, properties, uuid, vector)
82
+ else:
83
+ # Python Legacy Queue
84
+ item = {
85
+ "collection": collection,
86
+ "properties": properties,
87
+ "uuid": uuid,
88
+ "vector": vector
89
+ }
90
+ try:
91
+ self.queue.put_nowait(item)
92
+ except queue.Full:
93
+ logger.warning("🚨 VectorWave Log Queue is FULL. Dropping log.")
94
+
95
+ def _flush_batch_core(self, items: List[Dict[str, Any]]):
96
+ """
97
+ The actual flush logic called by either Rust or Python worker.
98
+ """
99
+ if not items:
100
+ return
101
+
102
+ # 1. Check/Retry Connection
103
+ if not self._initialized or not self.client:
104
+ self._connect_client()
105
+ if not self._initialized:
106
+ return
107
+
108
+ # 2. Send Batch via Weaviate Client
109
+ try:
110
+ # Weaviate v4 batch context
111
+ with self.client.batch.dynamic() as batch:
112
+ for item in items:
113
+ batch.add_object(
114
+ collection=item['collection'],
115
+ properties=item['properties'],
116
+ uuid=item.get('uuid'),
117
+ vector=item.get('vector')
118
+ )
119
+
120
+ if len(self.client.batch.failed_objects) > 0:
121
+ for failed in self.client.batch.failed_objects:
122
+ logger.error(f"⚠️ Batch Item Failed: {failed.message}")
123
+
124
+ except RuntimeError:
125
+ return
126
+ except Exception as e:
127
+ msg = str(e).lower()
128
+ if "shutdown" in msg or "closed" in msg:
129
+ return
130
+ logger.error(f"❌ Batch Flush Error: {e}")
131
+
132
+ # --- Legacy Python Worker Methods (Only used if Rust is missing) ---
133
+ def _python_worker_loop(self):
134
+ pending_items = []
135
+ last_flush_time = time.time()
136
+
137
+ while not self._stop_event.is_set():
138
+ try:
139
+ item = self.queue.get(timeout=0.5)
140
+ pending_items.append(item)
141
+ except queue.Empty:
142
+ pass
143
+
144
+ current_time = time.time()
145
+ if len(pending_items) >= self.batch_threshold or (pending_items and current_time - last_flush_time >= self.flush_interval):
146
+ self._flush_batch_core(pending_items)
147
+ pending_items = []
148
+ last_flush_time = current_time
149
+
150
+ def shutdown(self):
151
+ """Gracefully shuts down."""
152
+ if USE_RUST_CORE:
153
+ self._rust_manager.shutdown()
154
+ else:
155
+ if not self._stop_event.is_set():
156
+ self._stop_event.set()
157
+ if self._worker_thread and self._worker_thread.is_alive():
158
+ self._worker_thread.join(timeout=1.0)
159
+
160
+ # Flush remaining items
161
+ remaining = []
162
+ while not self.queue.empty():
163
+ remaining.append(self.queue.get_nowait())
164
+ if remaining:
165
+ self._flush_batch_core(remaining)
166
+
167
+ # Close client
168
+ if self.client:
169
+ try:
170
+ self.client.close()
171
+ except:
172
+ pass
173
+
174
+ @lru_cache(None)
175
+ def get_batch_manager() -> WeaviateBatchManager:
176
+ return WeaviateBatchManager()
File without changes
@@ -0,0 +1,100 @@
1
+ import inspect
2
+ import importlib
3
+ import logging
4
+ from functools import wraps
5
+ from typing import Any, Optional
6
+
7
+ # VectorWave Core Modules
8
+ from .decorator import vectorize
9
+ from ..monitoring.tracer import trace_span, current_tracer_var
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+ def create_smart_wrapper(original_func, root_wrapper):
14
+ """
15
+ Creates a wrapper that switches behavior based on the Tracing Context.
16
+
17
+ Args:
18
+ original_func: The original function (used when acting as a Child Span).
19
+ root_wrapper: The wrapper generated by vectorize (used when acting as a Root Span, includes metadata).
20
+ """
21
+ @wraps(original_func)
22
+ def wrapper(*args, **kwargs):
23
+ # Check for existing tracer at runtime
24
+ tracer = current_tracer_var.get()
25
+
26
+ if tracer:
27
+ # [Case A] Parent trace exists -> Act as Child Span
28
+ # (Wrap original function with trace_span and execute)
29
+ return trace_span(original_func, capture_return_value=True)(*args, **kwargs)
30
+ else:
31
+ # [Case B] No parent trace -> Act as Root Span
32
+ # (Delegate to the already configured root_wrapper)
33
+ return root_wrapper(*args, **kwargs)
34
+
35
+ # Mark to prevent double injection
36
+ wrapper._is_vectorized = True
37
+ return wrapper
38
+
39
+
40
+ class VectorWaveAutoInjector:
41
+ """
42
+ Scans specified modules or packages to automatically inject (Monkey Patch) VectorWave functionality.
43
+ """
44
+
45
+ # Global default configuration
46
+ _default_config = {}
47
+
48
+ @classmethod
49
+ def configure(cls, **kwargs):
50
+ """
51
+ Sets the default configuration to be applied to all inject calls.
52
+ """
53
+ cls._default_config.update(kwargs)
54
+ logger.info(f"⚙️ VectorWave AutoInjector Configured: {cls._default_config}")
55
+
56
+ @classmethod
57
+ def inject(cls, target_module_path: str, recursive: bool = False, **config):
58
+ """
59
+ Finds functions within a module and injects VectorWave functionality.
60
+ **Metadata registration (DB save or Pending) is performed immediately at this point.**
61
+ """
62
+ try:
63
+ module = importlib.import_module(target_module_path)
64
+ except ImportError as e:
65
+ logger.error(f"Failed to import module '{target_module_path}': {e}")
66
+ return None
67
+
68
+ # Merge configurations (Defaults < Individual Config)
69
+ final_config = cls._default_config.copy()
70
+ final_config.update(config)
71
+
72
+ logger.info(f"🌊 [AutoInjector] Injecting VectorWave into: {module.__name__}")
73
+
74
+ patched_count = 0
75
+
76
+ for name, obj in inspect.getmembers(module):
77
+ # Target only functions defined within the module
78
+ if inspect.isfunction(obj) and obj.__module__ == module.__name__:
79
+
80
+ # Skip if already processed
81
+ if getattr(obj, "_is_vectorized", False):
82
+ continue
83
+
84
+ logger.info(f" └─ Auto-wiring: {name}()")
85
+
86
+ # [Key Change] Call vectorize immediately at injection time to perform metadata registration!
87
+ # - If auto=False: Immediate DB save
88
+ # - If auto=True: Add to PENDING list
89
+ # The returned root_wrapper contains the execution logic (Root Span)
90
+ root_wrapper = vectorize(**final_config)(obj)
91
+
92
+ # Create smart wrapper to judge Context at runtime
93
+ smart_wrapper = create_smart_wrapper(obj, root_wrapper)
94
+
95
+ # Replace (Monkey Patching)
96
+ setattr(module, name, smart_wrapper)
97
+ patched_count += 1
98
+
99
+ logger.info(f"✨ Injection Complete. {patched_count} functions registered & auto-wired.")
100
+ return module
File without changes
@@ -0,0 +1,191 @@
1
+ import inspect
2
+ import logging
3
+ from functools import wraps
4
+ from typing import List, Optional, Dict, Any
5
+
6
+ from weaviate.util import generate_uuid5
7
+
8
+ from ..batch.batch import get_batch_manager
9
+ from ..models.db_config import get_weaviate_settings
10
+ from ..monitoring.tracer import trace_root, trace_span
11
+ from ..utils.function_cache import function_cache_manager
12
+ from ..utils.return_caching_utils import _check_and_return_cached_result
13
+ from ..vectorizer.factory import get_vectorizer
14
+ from ..utils.context import execution_source_context
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+ PENDING_FUNCTIONS: List[Dict[str, Any]] = []
19
+
20
+
21
+ def vectorize(search_description: Optional[str] = None,
22
+ sequence_narrative: Optional[str] = None,
23
+ auto: bool = False,
24
+ capture_return_value: bool = False,
25
+ semantic_cache: bool = False,
26
+ cache_threshold: float = 0.9,
27
+ replay: bool = False,
28
+ attributes_to_capture: Optional[List[str]] = None,
29
+ **execution_tags):
30
+ """
31
+ VectorWave Decorator with Auto-Generation support.
32
+ """
33
+
34
+ if semantic_cache:
35
+ if get_vectorizer() is None:
36
+ logger.warning(
37
+ f"Semantic caching requested for '{search_description}' but no Python vectorizer is configured. "
38
+ f"Disabling semantic_cache."
39
+ )
40
+ semantic_cache = False
41
+
42
+ if semantic_cache and not capture_return_value:
43
+ capture_return_value = True
44
+
45
+ if replay and not capture_return_value:
46
+ capture_return_value = True
47
+
48
+ def decorator(func):
49
+ is_async_func = inspect.iscoroutinefunction(func)
50
+
51
+ module_name = func.__module__
52
+ function_name = func.__name__
53
+ func_identifier = f"{module_name}.{function_name}"
54
+ func_uuid = generate_uuid5(func_identifier)
55
+
56
+ # Prepare attributes to capture
57
+ final_attributes = ['function_uuid', 'team', 'priority', 'run_id', 'exec_source']
58
+ if attributes_to_capture:
59
+ for attr in attributes_to_capture:
60
+ if attr not in final_attributes:
61
+ final_attributes.append(attr)
62
+
63
+ if replay:
64
+ try:
65
+ sig = inspect.signature(func)
66
+ for param_name in sig.parameters:
67
+ if param_name not in ('self', 'cls') and param_name not in final_attributes:
68
+ final_attributes.append(param_name)
69
+ except Exception as e:
70
+ logger.warning(f"Failed to inspect signature for replay auto-capture in '{function_name}': {e}")
71
+
72
+ # Extract Execution Tags
73
+ valid_execution_tags = {}
74
+ settings = get_weaviate_settings()
75
+ if execution_tags and settings.custom_properties:
76
+ allowed_keys = set(settings.custom_properties.keys())
77
+ for key, value in execution_tags.items():
78
+ if key in allowed_keys:
79
+ valid_execution_tags[key] = value
80
+ else:
81
+ logger.warning(
82
+ "Function '%s' has undefined execution_tag: '%s'. "
83
+ "This tag will be IGNORED. Please add it to your .weaviate_properties file.",
84
+ function_name,
85
+ key
86
+ )
87
+
88
+ try:
89
+ # define static properties
90
+ docstring = inspect.getdoc(func) or ""
91
+ source_code = inspect.getsource(func)
92
+
93
+ static_properties = {
94
+ "function_name": function_name,
95
+ "module_name": module_name,
96
+ "docstring": docstring,
97
+ "source_code": source_code,
98
+ "search_description": search_description,
99
+ "sequence_narrative": sequence_narrative
100
+ }
101
+ static_properties.update(valid_execution_tags)
102
+
103
+ if auto:
104
+ logger.info(f"Function '{function_name}' registered for auto-metadata generation.")
105
+ PENDING_FUNCTIONS.append({
106
+ "func_name": function_name,
107
+ "func_uuid": func_uuid,
108
+ "func_identifier": func_identifier,
109
+ "static_properties": static_properties
110
+ })
111
+ else:
112
+ # [Existing] Immediate Registration Mode
113
+ current_content_hash = function_cache_manager.calculate_content_hash(func_identifier, static_properties)
114
+
115
+ if function_cache_manager.is_cached_and_unchanged(func_uuid, current_content_hash):
116
+ logger.info(f"Function '{function_name}' is UNCHANGED. Skipping DB write.")
117
+ else:
118
+ logger.info(f"Function '{function_name}' is NEW or CHANGED. Writing to DB.")
119
+ batch = get_batch_manager()
120
+ vectorizer = get_vectorizer()
121
+ vector_to_add = None
122
+
123
+ if vectorizer and search_description:
124
+ try:
125
+ vector_to_add = vectorizer.embed(search_description)
126
+ except Exception as e:
127
+ logger.warning(f"Failed to vectorize '{function_name}': {e}")
128
+
129
+ batch.add_object(
130
+ collection=settings.COLLECTION_NAME,
131
+ properties=static_properties,
132
+ uuid=func_uuid,
133
+ vector=vector_to_add
134
+ )
135
+ function_cache_manager.update_cache(func_uuid, current_content_hash)
136
+
137
+ except Exception as e:
138
+ logger.error("Error in @vectorize setup for '%s': %s", func.__name__, e)
139
+
140
+ # --- Wrapper Logic ---
141
+
142
+ if is_async_func:
143
+ @trace_root()
144
+ @trace_span(attributes_to_capture=final_attributes, capture_return_value=capture_return_value)
145
+ @wraps(func)
146
+ async def inner_wrapper(*args, **kwargs):
147
+ # Remove injected tags from kwargs before calling original func
148
+ clean_kwargs = {k: v for k, v in kwargs.items() if
149
+ k not in valid_execution_tags and k != 'function_uuid' and k != 'exec_source'}
150
+ return await func(*args, **clean_kwargs)
151
+
152
+ @wraps(func)
153
+ async def outer_wrapper(*args, **kwargs):
154
+ if semantic_cache:
155
+ cached = _check_and_return_cached_result(func, args, kwargs, function_name, cache_threshold, True)
156
+ if cached is not None: return cached
157
+
158
+ full_kwargs = kwargs.copy()
159
+ full_kwargs.update(valid_execution_tags)
160
+ full_kwargs['function_uuid'] = func_uuid
161
+ full_kwargs['exec_source'] = execution_source_context.get()
162
+ return await inner_wrapper(*args, **full_kwargs)
163
+
164
+ outer_wrapper._is_vectorized = True
165
+ return outer_wrapper
166
+
167
+ else: # Sync wrapper
168
+ @trace_root()
169
+ @trace_span(attributes_to_capture=final_attributes, capture_return_value=capture_return_value)
170
+ @wraps(func)
171
+ def inner_wrapper(*args, **kwargs):
172
+ clean_kwargs = {k: v for k, v in kwargs.items() if
173
+ k not in valid_execution_tags and k != 'function_uuid' and k != 'exec_source'}
174
+ return func(*args, **clean_kwargs)
175
+
176
+ @wraps(func)
177
+ def outer_wrapper(*args, **kwargs):
178
+ if semantic_cache:
179
+ cached = _check_and_return_cached_result(func, args, kwargs, function_name, cache_threshold, False)
180
+ if cached is not None: return cached
181
+
182
+ full_kwargs = kwargs.copy()
183
+ full_kwargs.update(valid_execution_tags)
184
+ full_kwargs['function_uuid'] = func_uuid
185
+ full_kwargs['exec_source'] = execution_source_context.get()
186
+ return inner_wrapper(*args, **full_kwargs)
187
+
188
+ outer_wrapper._is_vectorized = True
189
+ return outer_wrapper
190
+
191
+ return decorator
@@ -0,0 +1,140 @@
1
+ import logging
2
+ import json
3
+ from typing import Optional, Dict, Any
4
+
5
+ from ..utils.function_cache import function_cache_manager
6
+ from ..models.db_config import get_weaviate_settings
7
+ from ..batch.batch import get_batch_manager
8
+ from ..vectorizer.factory import get_vectorizer
9
+ from .decorator import PENDING_FUNCTIONS
10
+ from .llm.factory import get_llm_client
11
+
12
+ logger = logging.getLogger(__name__)
13
+
14
+ try:
15
+ from openai import OpenAI
16
+ except ImportError:
17
+ OpenAI = None
18
+
19
+
20
+ def generate_metadata_via_llm(source_code: str, func_name: str) -> Optional[Dict[str, str]]:
21
+ """Call LLM to generate description and narrative from source code."""
22
+ settings = get_weaviate_settings()
23
+ client = get_llm_client()
24
+ if not client:
25
+ return None
26
+
27
+ prompt = f"""
28
+ Analyze the Python function below and generate a JSON object with two keys.
29
+ Ensure the values are **single strings**, not nested objects.
30
+
31
+ 1. "search_description": A concise summary of what this function does (for vector search).
32
+ 2. "sequence_narrative": A brief explanation of the context, inputs, and outputs as a narrative text.
33
+
34
+ Function Name: {func_name}
35
+ Code:
36
+ ```python
37
+ {source_code}
38
+ ```
39
+ """
40
+
41
+ try:
42
+ # Refactored to use BaseLLMClient interface
43
+ response_text = client.create_chat_completion(
44
+ model="gpt-4o-mini",
45
+ messages=[
46
+ {"role": "system", "content": "You are a technical documentation assistant. Output only JSON."},
47
+ {"role": "user", "content": prompt}
48
+ ],
49
+ temperature=0.0,
50
+ response_format={"type": "json_object"},
51
+ category="auto_doc"
52
+ )
53
+
54
+ if response_text:
55
+ return json.loads(response_text)
56
+ return None
57
+
58
+ except Exception as e:
59
+ logger.error(f"LLM generation failed for '{func_name}': {e}")
60
+ return None
61
+
62
+
63
+ def generate_and_register_metadata():
64
+ """
65
+ [Entry Point] Processes all functions in PENDING_FUNCTIONS.
66
+ """
67
+ if not PENDING_FUNCTIONS:
68
+ logger.info("No pending functions for auto-generation.")
69
+ return
70
+
71
+ logger.info(f"🚀 Processing {len(PENDING_FUNCTIONS)} functions for auto-documentation...")
72
+
73
+ settings = get_weaviate_settings()
74
+ batch = get_batch_manager()
75
+ vectorizer = get_vectorizer()
76
+
77
+ processed_count = 0
78
+
79
+ for item in PENDING_FUNCTIONS:
80
+ func_name = item["func_name"]
81
+ func_uuid = item["func_uuid"]
82
+ func_identifier = item["func_identifier"]
83
+ static_props = item["static_properties"]
84
+
85
+ current_hash = function_cache_manager.calculate_content_hash(func_identifier, static_props)
86
+
87
+ cached_meta = function_cache_manager.get_cached_metadata(func_uuid, current_hash)
88
+
89
+ final_desc = ""
90
+ final_narr = ""
91
+
92
+ if cached_meta:
93
+ logger.info(f"✅ [Cache Hit] Loaded metadata for '{func_name}'.")
94
+ final_desc = cached_meta.get("search_description")
95
+ final_narr = cached_meta.get("sequence_narrative")
96
+ else:
97
+ logger.info(f"🤖 [Auto-Gen] Generating metadata for '{func_name}' via LLM...")
98
+ generated = generate_metadata_via_llm(static_props["source_code"], func_name)
99
+
100
+ if generated:
101
+ final_desc = generated.get("search_description", "")
102
+ final_narr = generated.get("sequence_narrative", "")
103
+
104
+ if not isinstance(final_desc, str):
105
+ final_desc = json.dumps(final_desc, ensure_ascii=False)
106
+ if not isinstance(final_narr, str):
107
+ final_narr = json.dumps(final_narr, ensure_ascii=False)
108
+
109
+ # Update Cache with new metadata
110
+ function_cache_manager.update_cache_with_metadata(
111
+ func_uuid, current_hash,
112
+ {"search_description": final_desc, "sequence_narrative": final_narr}
113
+ )
114
+ else:
115
+ logger.warning(f"⚠️ Skipping registration for '{func_name}' due to generation failure.")
116
+ continue
117
+
118
+ # 3. Update Properties
119
+ static_props["search_description"] = final_desc
120
+ static_props["sequence_narrative"] = final_narr
121
+
122
+ # 4. Vectorize the Description
123
+ vector_to_add = None
124
+ if vectorizer and final_desc:
125
+ try:
126
+ vector_to_add = vectorizer.embed(final_desc)
127
+ except Exception as e:
128
+ logger.warning(f"Vectorization failed for '{func_name}': {e}")
129
+
130
+ # 5. Register to DB
131
+ batch.add_object(
132
+ collection=settings.COLLECTION_NAME,
133
+ properties=static_props,
134
+ uuid=func_uuid,
135
+ vector=vector_to_add
136
+ )
137
+ processed_count += 1
138
+
139
+ PENDING_FUNCTIONS.clear()
140
+ logger.info(f"✨ Auto-generation complete. Registered {processed_count} functions.")
File without changes
@@ -0,0 +1,47 @@
1
+ from abc import ABC, abstractmethod
2
+ from typing import List, Dict, Optional
3
+
4
+ class BaseLLMClient(ABC):
5
+ """
6
+ Abstract interface that all LLM Providers (OpenAI, Anthropic, etc.) must implement.
7
+ Implementations must handle internal token usage logging.
8
+ """
9
+
10
+ @abstractmethod
11
+ def create_embedding(self, text: str, model: str, category: str = "default") -> Optional[List[float]]:
12
+ """
13
+ Generates text embeddings.
14
+
15
+ Args:
16
+ text: The text to embed.
17
+ model: The name of the model to use.
18
+ category: Category for aggregating token usage (e.g., 'execution_log', 'auto_doc').
19
+
20
+ Returns:
21
+ The generated list of embedding vectors (None on failure).
22
+ """
23
+ pass
24
+
25
+ @abstractmethod
26
+ def create_chat_completion(
27
+ self,
28
+ messages: List[Dict],
29
+ model: str,
30
+ temperature: float = 0.1,
31
+ response_format: Optional[Dict] = None,
32
+ category: str = "default"
33
+ ) -> Optional[str]:
34
+ """
35
+ Generates a chat completion (response).
36
+
37
+ Args:
38
+ messages: List of conversation messages [{"role": "user", "content": "..."}].
39
+ model: The name of the model to use.
40
+ temperature: Parameter for controlling generation diversity.
41
+ response_format: Response format (e.g., {"type": "json_object"}).
42
+ category: Category for aggregating token usage (e.g., 'execution_log', 'auto_doc').
43
+
44
+ Returns:
45
+ The generated text response (None on failure).
46
+ """
47
+ pass