vectorwave 0.1.5__tar.gz → 0.1.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {vectorwave-0.1.5/src/vectorwave.egg-info → vectorwave-0.1.7}/PKG-INFO +43 -9
- {vectorwave-0.1.5 → vectorwave-0.1.7}/Readme.md +42 -8
- {vectorwave-0.1.5 → vectorwave-0.1.7}/pyproject.toml +1 -1
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/batch/test_batch.py +74 -11
- vectorwave-0.1.7/src/tests/conftest.py +27 -0
- vectorwave-0.1.7/src/tests/core/test_semantic_caching.py +353 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/database/test_db.py +3 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/models/test_db_config.py +36 -1
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/monitoring/test_tracer.py +144 -68
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/search/test_execution_search.py +12 -7
- vectorwave-0.1.7/src/tests/utils/test_function_cahe.py +126 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/__init__.py +7 -2
- vectorwave-0.1.7/src/vectorwave/batch/batch.py +176 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/core/decorator.py +90 -22
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/database/db.py +15 -6
- vectorwave-0.1.7/src/vectorwave/database/db_search.py +358 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/models/db_config.py +24 -4
- vectorwave-0.1.7/src/vectorwave/monitoring/tracer.py +477 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/search/execution_search.py +19 -38
- vectorwave-0.1.7/src/vectorwave/search/rag_search.py +161 -0
- vectorwave-0.1.7/src/vectorwave/utils/function_cache.py +73 -0
- vectorwave-0.1.7/src/vectorwave/utils/return_caching_utils.py +76 -0
- vectorwave-0.1.7/src/vectorwave/vectorizer/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7/src/vectorwave.egg-info}/PKG-INFO +43 -9
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave.egg-info/SOURCES.txt +7 -0
- vectorwave-0.1.5/src/vectorwave/batch/batch.py +0 -68
- vectorwave-0.1.5/src/vectorwave/database/db_search.py +0 -122
- vectorwave-0.1.5/src/vectorwave/monitoring/tracer.py +0 -273
- {vectorwave-0.1.5 → vectorwave-0.1.7}/LICENSE +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/MANIFEST.in +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/NOTICE +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/setup.cfg +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/batch/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/core/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/core/test_decorator.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/database/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/database/test_db_search.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/exception/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/models/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/monitoring/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/monitoring/alert/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/monitoring/alert/test_alerter.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/monitoring/test_async_trace.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/prediction/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/tests/search/__init__.py +0 -0
- {vectorwave-0.1.5/src/tests/vectorizer → vectorwave-0.1.7/src/tests/utils}/__init__.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/batch → vectorwave-0.1.7/src/tests/vectorizer}/__init__.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/core → vectorwave-0.1.7/src/vectorwave/batch}/__init__.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/database → vectorwave-0.1.7/src/vectorwave/core}/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/core/core.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/exception → vectorwave-0.1.7/src/vectorwave/database}/__init__.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/models → vectorwave-0.1.7/src/vectorwave/exception}/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/exception/exceptions.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/monitoring → vectorwave-0.1.7/src/vectorwave/models}/__init__.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/monitoring/alert → vectorwave-0.1.7/src/vectorwave/monitoring}/__init__.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/prediction → vectorwave-0.1.7/src/vectorwave/monitoring/alert}/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/base.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/factory.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/null_alerter.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/webhook_alerter.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/monitoring/monitoring.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/search → vectorwave-0.1.7/src/vectorwave/prediction}/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/prediction/predictor.py +0 -0
- {vectorwave-0.1.5/src/vectorwave/vectorizer → vectorwave-0.1.7/src/vectorwave/search}/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/search/extended_search.py +0 -0
- /vectorwave-0.1.5/src/vectorwave/search/rag_search.py → /vectorwave-0.1.7/src/vectorwave/utils/__init__.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/vectorizer/base.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/vectorizer/factory.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/vectorizer/huggingface_vectorizer.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave/vectorizer/openai_vectorizer.py +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave.egg-info/dependency_links.txt +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave.egg-info/requires.txt +0 -0
- {vectorwave-0.1.5 → vectorwave-0.1.7}/src/vectorwave.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: vectorwave
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.7
|
|
4
4
|
Summary: VectorWave: Seamless Auto-Vectorization Framework
|
|
5
5
|
Author-email: junyeonggim <junyeonggim5@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -24,7 +24,6 @@ Requires-Dist: requests
|
|
|
24
24
|
Dynamic: license-file
|
|
25
25
|
|
|
26
26
|
|
|
27
|
-
|
|
28
27
|
# VectorWave: Seamless Auto-Vectorization Framework
|
|
29
28
|
|
|
30
29
|
[](https://opensource.org/licenses/MIT)
|
|
@@ -40,6 +39,9 @@ Dynamic: license-file
|
|
|
40
39
|
* **`@vectorize` Decorator:**
|
|
41
40
|
1. **Static Data Collection:** Upon script load, the function's source code, docstring, and metadata are saved once to the `VectorWaveFunctions` collection.
|
|
42
41
|
2. **Dynamic Data Logging:** Each time the function is called, its execution time, success/failure status, error logs, and "dynamic tags" are recorded in the `VectorWaveExecutions` collection.
|
|
42
|
+
* **Semantic Caching and Performance Optimization:**
|
|
43
|
+
* Determines cache hits based on the **semantic similarity** of function inputs, bypassing actual execution for identical or highly similar inputs and returning stored results immediately.
|
|
44
|
+
* This significantly **reduces latency** and costs, especially for high-cost computation functions (e.g., LLM calls, complex data processing).
|
|
43
45
|
* **Distributed Tracing:** Combines `@vectorize` and `@trace_span` decorators to bundle the execution of complex, multi-step workflows under a single **`trace_id`** for analysis.
|
|
44
46
|
* **Search Interface:** Provides `search_functions` and `search_executions` to query the stored vector data (function definitions) and logs (execution history), facilitating the construction of RAG and monitoring systems.
|
|
45
47
|
|
|
@@ -69,7 +71,7 @@ try:
|
|
|
69
71
|
except Exception as e:
|
|
70
72
|
print(f"DB initialization failed: {e}")
|
|
71
73
|
exit()
|
|
72
|
-
|
|
74
|
+
````
|
|
73
75
|
|
|
74
76
|
### 2\. [Storage] Using `@vectorize` and Distributed Tracing
|
|
75
77
|
|
|
@@ -118,6 +120,37 @@ print("Now calling 'process_payment'...")
|
|
|
118
120
|
process_payment("user_789", 5000)
|
|
119
121
|
```
|
|
120
122
|
|
|
123
|
+
#### Semantic Caching Example
|
|
124
|
+
|
|
125
|
+
Configure a function to return a cached result if the input is semantically similar to a previous execution.
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
from vectorwave import vectorize
|
|
129
|
+
import time
|
|
130
|
+
|
|
131
|
+
@vectorize(
|
|
132
|
+
search_description="High-cost LLM summarization task",
|
|
133
|
+
sequence_narrative="LLM Summarization Step",
|
|
134
|
+
semantic_cache=True, # Enable caching
|
|
135
|
+
cache_threshold=0.95, # Cache hit if similarity >= 0.95
|
|
136
|
+
capture_return_value=True # Required to save the result
|
|
137
|
+
)
|
|
138
|
+
def summarize_document(document_text: str):
|
|
139
|
+
# Simulate an LLM call or heavy computation (e.g., 0.5 sec delay)
|
|
140
|
+
time.sleep(0.5)
|
|
141
|
+
print("--- [Cache Miss] Document is being summarized by LLM...")
|
|
142
|
+
return f"Summary of: {document_text[:20]}..."
|
|
143
|
+
|
|
144
|
+
# First call (Cache Miss) - takes ~0.5s, saves result to DB
|
|
145
|
+
result_1 = summarize_document("The first quarter results showed strong growth in Europe and Asia...")
|
|
146
|
+
|
|
147
|
+
# Second call (Cache Hit) - takes ~0.0s, returns cached value
|
|
148
|
+
# "Q1 results" is semantically similar to "first quarter results"
|
|
149
|
+
result_2 = summarize_document("The Q1 results demonstrated strong growth in Europe and Asia...")
|
|
150
|
+
|
|
151
|
+
# result_2 returns the stored value without executing the function's body.
|
|
152
|
+
```
|
|
153
|
+
|
|
121
154
|
### 3\. [Retrieval ①] Search Function Definitions (for RAG)
|
|
122
155
|
|
|
123
156
|
```python
|
|
@@ -185,7 +218,12 @@ You can select the text vectorization method via the `VECTORIZER` environment va
|
|
|
185
218
|
| **`weaviate_module`** | (Docker Delegate) Delegates vectorization to Weaviate's built-in module (e.g., `text2vec-openai`). | `WEAVIATE_VECTORIZER_MODULE`, `OPENAI_API_KEY` |
|
|
186
219
|
| **`none`** | Disables vectorization. Data is stored without vectors. | None |
|
|
187
220
|
|
|
188
|
-
|
|
221
|
+
#### ⚠️ Semantic Caching Prerequisites and Configuration
|
|
222
|
+
|
|
223
|
+
To use `semantic_cache=True`, the following conditions must be met:
|
|
224
|
+
|
|
225
|
+
* **Vectorizer Required:** A **Python-based vectorizer** (`huggingface` or `openai_client`) must be configured in your environment (`VECTORIZER` environment variable). Caching is automatically disabled if set to `weaviate_module` or `none`.
|
|
226
|
+
* **Return Value Capture:** The `capture_return_value` parameter is automatically set to `True` when `semantic_cache=True` is enabled.
|
|
189
227
|
|
|
190
228
|
### .env File Examples
|
|
191
229
|
|
|
@@ -266,8 +304,6 @@ FAILURE_MAPPING_FILE_PATH=.vectorwave_errors.json
|
|
|
266
304
|
RUN_ID=test-run-001
|
|
267
305
|
```
|
|
268
306
|
|
|
269
|
-
-----
|
|
270
|
-
|
|
271
307
|
### 🚀 Advanced Failure Tracing (Error Code)
|
|
272
308
|
|
|
273
309
|
This enhances `VectorWaveExecutions` logs beyond a simple `status: "ERROR"`. An `error_code` property is added to the schema for granular failure analysis.
|
|
@@ -396,7 +432,6 @@ def other_function():
|
|
|
396
432
|
pass
|
|
397
433
|
```
|
|
398
434
|
|
|
399
|
-
|
|
400
435
|
1. **Validation (Important):** Tags (global or function-specific) will **only** be saved to Weaviate if their key (e.g., `run_id`, `team`, `priority`) was first defined in the `.weaviate_properties` file (Step 1). Tags not defined in the schema are **ignored**, and a warning is logged at startup.
|
|
401
436
|
|
|
402
437
|
2. **Priority (Override):** If a tag key is defined in both places (e.g., global `RUN_ID` in `.env` and `run_id="override-xyz"` in the decorator), the **function-specific tag from the decorator always wins**.
|
|
@@ -413,6 +448,7 @@ def other_function():
|
|
|
413
448
|
Beyond just logging, `VectorWave` can send **real-time notifications via webhook** the instant an error occurs. This functionality is built directly into the tracer and can be activated simply by updating your `.env` file.
|
|
414
449
|
|
|
415
450
|
**How it Works:**
|
|
451
|
+
|
|
416
452
|
1. An exception is raised within a function decorated by `@trace_span` or `@vectorize`.
|
|
417
453
|
2. The tracer catches the exception in its `except` block and immediately calls the `alerter` object.
|
|
418
454
|
3. The alerter reads the `.env` configuration and uses the `WebhookAlerter` to dispatch the error details to your specified URL.
|
|
@@ -432,8 +468,6 @@ ALERTER_WEBHOOK_URL="[https://discord.com/api/webhooks/YOUR_HOOK_ID/](https://di
|
|
|
432
468
|
With just these two lines, running test_ex/example.py will now instantly send a Discord alert when the CustomValueError is raised.
|
|
433
469
|
|
|
434
470
|
Extensibility (Strategy Pattern): The alerting system is built on a Strategy Pattern. You can easily extend it by implementing the BaseAlerter interface to support other channels like email, PagerDuty, or more.
|
|
435
|
-
|
|
436
|
-
**Tag Merging and Validation Rules**
|
|
437
471
|
```
|
|
438
472
|
|
|
439
473
|
## 🤝 Contributing
|
|
@@ -1,5 +1,4 @@
|
|
|
1
1
|
|
|
2
|
-
|
|
3
2
|
# VectorWave: Seamless Auto-Vectorization Framework
|
|
4
3
|
|
|
5
4
|
[](https://opensource.org/licenses/MIT)
|
|
@@ -15,6 +14,9 @@
|
|
|
15
14
|
* **`@vectorize` Decorator:**
|
|
16
15
|
1. **Static Data Collection:** Upon script load, the function's source code, docstring, and metadata are saved once to the `VectorWaveFunctions` collection.
|
|
17
16
|
2. **Dynamic Data Logging:** Each time the function is called, its execution time, success/failure status, error logs, and "dynamic tags" are recorded in the `VectorWaveExecutions` collection.
|
|
17
|
+
* **Semantic Caching and Performance Optimization:**
|
|
18
|
+
* Determines cache hits based on the **semantic similarity** of function inputs, bypassing actual execution for identical or highly similar inputs and returning stored results immediately.
|
|
19
|
+
* This significantly **reduces latency** and costs, especially for high-cost computation functions (e.g., LLM calls, complex data processing).
|
|
18
20
|
* **Distributed Tracing:** Combines `@vectorize` and `@trace_span` decorators to bundle the execution of complex, multi-step workflows under a single **`trace_id`** for analysis.
|
|
19
21
|
* **Search Interface:** Provides `search_functions` and `search_executions` to query the stored vector data (function definitions) and logs (execution history), facilitating the construction of RAG and monitoring systems.
|
|
20
22
|
|
|
@@ -44,7 +46,7 @@ try:
|
|
|
44
46
|
except Exception as e:
|
|
45
47
|
print(f"DB initialization failed: {e}")
|
|
46
48
|
exit()
|
|
47
|
-
|
|
49
|
+
````
|
|
48
50
|
|
|
49
51
|
### 2\. [Storage] Using `@vectorize` and Distributed Tracing
|
|
50
52
|
|
|
@@ -93,6 +95,37 @@ print("Now calling 'process_payment'...")
|
|
|
93
95
|
process_payment("user_789", 5000)
|
|
94
96
|
```
|
|
95
97
|
|
|
98
|
+
#### Semantic Caching Example
|
|
99
|
+
|
|
100
|
+
Configure a function to return a cached result if the input is semantically similar to a previous execution.
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from vectorwave import vectorize
|
|
104
|
+
import time
|
|
105
|
+
|
|
106
|
+
@vectorize(
|
|
107
|
+
search_description="High-cost LLM summarization task",
|
|
108
|
+
sequence_narrative="LLM Summarization Step",
|
|
109
|
+
semantic_cache=True, # Enable caching
|
|
110
|
+
cache_threshold=0.95, # Cache hit if similarity >= 0.95
|
|
111
|
+
capture_return_value=True # Required to save the result
|
|
112
|
+
)
|
|
113
|
+
def summarize_document(document_text: str):
|
|
114
|
+
# Simulate an LLM call or heavy computation (e.g., 0.5 sec delay)
|
|
115
|
+
time.sleep(0.5)
|
|
116
|
+
print("--- [Cache Miss] Document is being summarized by LLM...")
|
|
117
|
+
return f"Summary of: {document_text[:20]}..."
|
|
118
|
+
|
|
119
|
+
# First call (Cache Miss) - takes ~0.5s, saves result to DB
|
|
120
|
+
result_1 = summarize_document("The first quarter results showed strong growth in Europe and Asia...")
|
|
121
|
+
|
|
122
|
+
# Second call (Cache Hit) - takes ~0.0s, returns cached value
|
|
123
|
+
# "Q1 results" is semantically similar to "first quarter results"
|
|
124
|
+
result_2 = summarize_document("The Q1 results demonstrated strong growth in Europe and Asia...")
|
|
125
|
+
|
|
126
|
+
# result_2 returns the stored value without executing the function's body.
|
|
127
|
+
```
|
|
128
|
+
|
|
96
129
|
### 3\. [Retrieval ①] Search Function Definitions (for RAG)
|
|
97
130
|
|
|
98
131
|
```python
|
|
@@ -160,7 +193,12 @@ You can select the text vectorization method via the `VECTORIZER` environment va
|
|
|
160
193
|
| **`weaviate_module`** | (Docker Delegate) Delegates vectorization to Weaviate's built-in module (e.g., `text2vec-openai`). | `WEAVIATE_VECTORIZER_MODULE`, `OPENAI_API_KEY` |
|
|
161
194
|
| **`none`** | Disables vectorization. Data is stored without vectors. | None |
|
|
162
195
|
|
|
163
|
-
|
|
196
|
+
#### ⚠️ Semantic Caching Prerequisites and Configuration
|
|
197
|
+
|
|
198
|
+
To use `semantic_cache=True`, the following conditions must be met:
|
|
199
|
+
|
|
200
|
+
* **Vectorizer Required:** A **Python-based vectorizer** (`huggingface` or `openai_client`) must be configured in your environment (`VECTORIZER` environment variable). Caching is automatically disabled if set to `weaviate_module` or `none`.
|
|
201
|
+
* **Return Value Capture:** The `capture_return_value` parameter is automatically set to `True` when `semantic_cache=True` is enabled.
|
|
164
202
|
|
|
165
203
|
### .env File Examples
|
|
166
204
|
|
|
@@ -241,8 +279,6 @@ FAILURE_MAPPING_FILE_PATH=.vectorwave_errors.json
|
|
|
241
279
|
RUN_ID=test-run-001
|
|
242
280
|
```
|
|
243
281
|
|
|
244
|
-
-----
|
|
245
|
-
|
|
246
282
|
### 🚀 Advanced Failure Tracing (Error Code)
|
|
247
283
|
|
|
248
284
|
This enhances `VectorWaveExecutions` logs beyond a simple `status: "ERROR"`. An `error_code` property is added to the schema for granular failure analysis.
|
|
@@ -371,7 +407,6 @@ def other_function():
|
|
|
371
407
|
pass
|
|
372
408
|
```
|
|
373
409
|
|
|
374
|
-
|
|
375
410
|
1. **Validation (Important):** Tags (global or function-specific) will **only** be saved to Weaviate if their key (e.g., `run_id`, `team`, `priority`) was first defined in the `.weaviate_properties` file (Step 1). Tags not defined in the schema are **ignored**, and a warning is logged at startup.
|
|
376
411
|
|
|
377
412
|
2. **Priority (Override):** If a tag key is defined in both places (e.g., global `RUN_ID` in `.env` and `run_id="override-xyz"` in the decorator), the **function-specific tag from the decorator always wins**.
|
|
@@ -388,6 +423,7 @@ def other_function():
|
|
|
388
423
|
Beyond just logging, `VectorWave` can send **real-time notifications via webhook** the instant an error occurs. This functionality is built directly into the tracer and can be activated simply by updating your `.env` file.
|
|
389
424
|
|
|
390
425
|
**How it Works:**
|
|
426
|
+
|
|
391
427
|
1. An exception is raised within a function decorated by `@trace_span` or `@vectorize`.
|
|
392
428
|
2. The tracer catches the exception in its `except` block and immediately calls the `alerter` object.
|
|
393
429
|
3. The alerter reads the `.env` configuration and uses the `WebhookAlerter` to dispatch the error details to your specified URL.
|
|
@@ -407,8 +443,6 @@ ALERTER_WEBHOOK_URL="[https://discord.com/api/webhooks/YOUR_HOOK_ID/](https://di
|
|
|
407
443
|
With just these two lines, running test_ex/example.py will now instantly send a Discord alert when the CustomValueError is raised.
|
|
408
444
|
|
|
409
445
|
Extensibility (Strategy Pattern): The alerting system is built on a Strategy Pattern. You can easily extend it by implementing the BaseAlerter interface to support other channels like email, PagerDuty, or more.
|
|
410
|
-
|
|
411
|
-
**Tag Merging and Validation Rules**
|
|
412
446
|
```
|
|
413
447
|
|
|
414
448
|
## 🤝 Contributing
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
from unittest.mock import MagicMock
|
|
1
|
+
from unittest.mock import MagicMock, call, ANY
|
|
2
2
|
|
|
3
3
|
import pytest
|
|
4
4
|
from vectorwave.batch.batch import get_batch_manager
|
|
@@ -23,6 +23,9 @@ def mock_deps(monkeypatch):
|
|
|
23
23
|
mock_collection.data = mock_collection_data
|
|
24
24
|
mock_client.collections.get = MagicMock(return_value=mock_collection)
|
|
25
25
|
|
|
26
|
+
mock_batch_context = MagicMock()
|
|
27
|
+
mock_client.batch.dynamic.return_value.__enter__.return_value = mock_batch_context
|
|
28
|
+
|
|
26
29
|
# Mock get_weaviate_client
|
|
27
30
|
mock_get_client = MagicMock(return_value=mock_client)
|
|
28
31
|
monkeypatch.setattr("vectorwave.batch.batch.get_weaviate_client", mock_get_client)
|
|
@@ -36,6 +39,9 @@ def mock_deps(monkeypatch):
|
|
|
36
39
|
mock_atexit_register = MagicMock()
|
|
37
40
|
monkeypatch.setattr("atexit.register", mock_atexit_register)
|
|
38
41
|
|
|
42
|
+
mock_thread = MagicMock()
|
|
43
|
+
monkeypatch.setattr("threading.Thread", mock_thread)
|
|
44
|
+
|
|
39
45
|
# Clear lru_cache
|
|
40
46
|
get_batch_manager.cache_clear()
|
|
41
47
|
|
|
@@ -44,7 +50,8 @@ def mock_deps(monkeypatch):
|
|
|
44
50
|
"get_settings": mock_get_settings,
|
|
45
51
|
"client": mock_client,
|
|
46
52
|
"settings": mock_settings,
|
|
47
|
-
"atexit": mock_atexit_register
|
|
53
|
+
"atexit": mock_atexit_register,
|
|
54
|
+
"batch_context": mock_batch_context
|
|
48
55
|
}
|
|
49
56
|
|
|
50
57
|
def test_get_batch_manager_is_singleton(mock_deps):
|
|
@@ -80,19 +87,75 @@ def test_batch_manager_init_failure(monkeypatch):
|
|
|
80
87
|
# The _initialized flag should be False if initialization fails
|
|
81
88
|
assert manager._initialized is False
|
|
82
89
|
|
|
83
|
-
def
|
|
90
|
+
def test_add_object_enqueues_item(mock_deps):
|
|
84
91
|
"""
|
|
85
|
-
Case 4: Test if add_object()
|
|
92
|
+
[Updated] Case 4: Test if add_object() puts the item into the local queue (Non-blocking)
|
|
93
|
+
Instead of calling client directly.
|
|
86
94
|
"""
|
|
87
95
|
manager = get_batch_manager()
|
|
88
96
|
props = {"key": "value"}
|
|
89
97
|
|
|
90
|
-
manager.
|
|
98
|
+
assert manager.queue.empty()
|
|
99
|
+
|
|
100
|
+
manager.add_object(collection="TestCollection", properties=props, uuid="test-uuid", vector=[0.1])
|
|
101
|
+
|
|
102
|
+
# 1. Should NOT call DB directly
|
|
103
|
+
mock_deps["client"].collections.get.assert_not_called()
|
|
104
|
+
|
|
105
|
+
# 2. Should be in the Queue
|
|
106
|
+
assert manager.queue.qsize() == 1
|
|
107
|
+
item = manager.queue.get()
|
|
108
|
+
|
|
109
|
+
assert item["collection"] == "TestCollection"
|
|
110
|
+
assert item["properties"] == props
|
|
111
|
+
assert item["uuid"] == "test-uuid"
|
|
112
|
+
assert item["vector"] == [0.1]
|
|
113
|
+
|
|
114
|
+
def test_flush_batch_sends_to_weaviate(mock_deps):
|
|
115
|
+
"""
|
|
116
|
+
[New] Case 5: Test if _flush_batch sends items using client.batch.dynamic context
|
|
117
|
+
"""
|
|
118
|
+
manager = get_batch_manager()
|
|
119
|
+
|
|
120
|
+
items = [
|
|
121
|
+
{"collection": "C1", "properties": {"p": 1}, "uuid": "u1", "vector": None},
|
|
122
|
+
{"collection": "C2", "properties": {"p": 2}, "uuid": "u2", "vector": [1.0]}
|
|
123
|
+
]
|
|
124
|
+
|
|
125
|
+
# Manually trigger flush
|
|
126
|
+
manager._flush_batch(items)
|
|
127
|
+
|
|
128
|
+
# Check if dynamic batch context was entered
|
|
129
|
+
mock_deps["client"].batch.dynamic.assert_called_once()
|
|
130
|
+
|
|
131
|
+
# Check if batch.add_object was called for each item
|
|
132
|
+
mock_batch_ctx = mock_deps["batch_context"]
|
|
133
|
+
assert mock_batch_ctx.add_object.call_count == 2
|
|
134
|
+
|
|
135
|
+
mock_batch_ctx.add_object.assert_has_calls([
|
|
136
|
+
call(collection="C1", properties={"p": 1}, uuid="u1", vector=None),
|
|
137
|
+
call(collection="C2", properties={"p": 2}, uuid="u2", vector=[1.0])
|
|
138
|
+
])
|
|
139
|
+
|
|
140
|
+
def test_flush_batch_reconnects_if_disconnected(mock_deps):
|
|
141
|
+
"""
|
|
142
|
+
[New] Case 6: Test reconnection logic when client is not initialized
|
|
143
|
+
"""
|
|
144
|
+
manager = get_batch_manager()
|
|
145
|
+
|
|
146
|
+
# Simulate disconnection
|
|
147
|
+
manager._initialized = False
|
|
148
|
+
manager.client = None
|
|
149
|
+
|
|
150
|
+
# Reset mocks
|
|
151
|
+
mock_deps["get_client"].reset_mock()
|
|
152
|
+
|
|
153
|
+
items = [{"collection": "C1", "properties": {}, "uuid": "u1", "vector": None}]
|
|
91
154
|
|
|
92
|
-
|
|
155
|
+
# Trigger flush
|
|
156
|
+
manager._flush_batch(items)
|
|
93
157
|
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
)
|
|
158
|
+
# Should try to reconnect
|
|
159
|
+
mock_deps["get_client"].assert_called_once()
|
|
160
|
+
# And then send
|
|
161
|
+
mock_deps["client"].batch.dynamic.assert_called_once()
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import pytest
|
|
2
|
+
import os
|
|
3
|
+
import logging
|
|
4
|
+
|
|
5
|
+
CACHE_FILE_PATH = ".vectorwave_functions_cache.json"
|
|
6
|
+
logger = logging.getLogger(__name__)
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _delete_cache():
|
|
10
|
+
"""Helper function to remove the cache file if it exists."""
|
|
11
|
+
if os.path.exists(CACHE_FILE_PATH):
|
|
12
|
+
try:
|
|
13
|
+
os.remove(CACHE_FILE_PATH)
|
|
14
|
+
except OSError as e:
|
|
15
|
+
print(f"\n[CacheFixture Error] Failed to remove {CACHE_FILE_PATH}: {e}")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@pytest.fixture(autouse=True, scope="function")
|
|
19
|
+
def atomic_function_cache():
|
|
20
|
+
# --- SETUP (Before Test) ---
|
|
21
|
+
_delete_cache()
|
|
22
|
+
|
|
23
|
+
# --- Run the test ---
|
|
24
|
+
yield
|
|
25
|
+
|
|
26
|
+
# --- TEARDOWN (After Test) ---
|
|
27
|
+
_delete_cache()
|