vectorwave 0.1.6__tar.gz → 0.1.7__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. {vectorwave-0.1.6/src/vectorwave.egg-info → vectorwave-0.1.7}/PKG-INFO +43 -9
  2. {vectorwave-0.1.6 → vectorwave-0.1.7}/Readme.md +42 -8
  3. {vectorwave-0.1.6 → vectorwave-0.1.7}/pyproject.toml +1 -1
  4. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/batch/test_batch.py +74 -11
  5. vectorwave-0.1.7/src/tests/core/test_semantic_caching.py +353 -0
  6. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/__init__.py +6 -2
  7. vectorwave-0.1.7/src/vectorwave/batch/batch.py +176 -0
  8. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/core/decorator.py +51 -0
  9. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/database/db_search.py +142 -0
  10. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/models/db_config.py +7 -2
  11. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/monitoring/tracer.py +92 -10
  12. vectorwave-0.1.7/src/vectorwave/search/rag_search.py +161 -0
  13. vectorwave-0.1.7/src/vectorwave/utils/return_caching_utils.py +76 -0
  14. {vectorwave-0.1.6 → vectorwave-0.1.7/src/vectorwave.egg-info}/PKG-INFO +43 -9
  15. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave.egg-info/SOURCES.txt +2 -0
  16. vectorwave-0.1.6/src/vectorwave/batch/batch.py +0 -68
  17. vectorwave-0.1.6/src/vectorwave/search/rag_search.py +0 -0
  18. {vectorwave-0.1.6 → vectorwave-0.1.7}/LICENSE +0 -0
  19. {vectorwave-0.1.6 → vectorwave-0.1.7}/MANIFEST.in +0 -0
  20. {vectorwave-0.1.6 → vectorwave-0.1.7}/NOTICE +0 -0
  21. {vectorwave-0.1.6 → vectorwave-0.1.7}/setup.cfg +0 -0
  22. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/__init__.py +0 -0
  23. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/batch/__init__.py +0 -0
  24. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/conftest.py +0 -0
  25. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/core/__init__.py +0 -0
  26. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/core/test_decorator.py +0 -0
  27. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/database/__init__.py +0 -0
  28. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/database/test_db.py +0 -0
  29. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/database/test_db_search.py +0 -0
  30. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/exception/__init__.py +0 -0
  31. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/models/__init__.py +0 -0
  32. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/models/test_db_config.py +0 -0
  33. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/monitoring/__init__.py +0 -0
  34. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/monitoring/alert/__init__.py +0 -0
  35. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/monitoring/alert/test_alerter.py +0 -0
  36. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/monitoring/test_async_trace.py +0 -0
  37. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/monitoring/test_tracer.py +0 -0
  38. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/prediction/__init__.py +0 -0
  39. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/search/__init__.py +0 -0
  40. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/search/test_execution_search.py +0 -0
  41. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/utils/__init__.py +0 -0
  42. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/utils/test_function_cahe.py +0 -0
  43. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/tests/vectorizer/__init__.py +0 -0
  44. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/batch/__init__.py +0 -0
  45. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/core/__init__.py +0 -0
  46. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/core/core.py +0 -0
  47. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/database/__init__.py +0 -0
  48. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/database/db.py +0 -0
  49. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/exception/__init__.py +0 -0
  50. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/exception/exceptions.py +0 -0
  51. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/models/__init__.py +0 -0
  52. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/monitoring/__init__.py +0 -0
  53. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/__init__.py +0 -0
  54. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/base.py +0 -0
  55. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/factory.py +0 -0
  56. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/null_alerter.py +0 -0
  57. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/monitoring/alert/webhook_alerter.py +0 -0
  58. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/monitoring/monitoring.py +0 -0
  59. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/prediction/__init__.py +0 -0
  60. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/prediction/predictor.py +0 -0
  61. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/search/__init__.py +0 -0
  62. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/search/execution_search.py +0 -0
  63. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/search/extended_search.py +0 -0
  64. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/utils/__init__.py +0 -0
  65. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/utils/function_cache.py +0 -0
  66. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/vectorizer/__init__.py +0 -0
  67. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/vectorizer/base.py +0 -0
  68. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/vectorizer/factory.py +0 -0
  69. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/vectorizer/huggingface_vectorizer.py +0 -0
  70. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave/vectorizer/openai_vectorizer.py +0 -0
  71. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave.egg-info/dependency_links.txt +0 -0
  72. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave.egg-info/requires.txt +0 -0
  73. {vectorwave-0.1.6 → vectorwave-0.1.7}/src/vectorwave.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: vectorwave
3
- Version: 0.1.6
3
+ Version: 0.1.7
4
4
  Summary: VectorWave: Seamless Auto-Vectorization Framework
5
5
  Author-email: junyeonggim <junyeonggim5@gmail.com>
6
6
  License-Expression: MIT
@@ -24,7 +24,6 @@ Requires-Dist: requests
24
24
  Dynamic: license-file
25
25
 
26
26
 
27
-
28
27
  # VectorWave: Seamless Auto-Vectorization Framework
29
28
 
30
29
  [](https://opensource.org/licenses/MIT)
@@ -40,6 +39,9 @@ Dynamic: license-file
40
39
  * **`@vectorize` Decorator:**
41
40
  1. **Static Data Collection:** Upon script load, the function's source code, docstring, and metadata are saved once to the `VectorWaveFunctions` collection.
42
41
  2. **Dynamic Data Logging:** Each time the function is called, its execution time, success/failure status, error logs, and "dynamic tags" are recorded in the `VectorWaveExecutions` collection.
42
+ * **Semantic Caching and Performance Optimization:**
43
+ * Determines cache hits based on the **semantic similarity** of function inputs, bypassing actual execution for identical or highly similar inputs and returning stored results immediately.
44
+ * This significantly **reduces latency** and costs, especially for high-cost computation functions (e.g., LLM calls, complex data processing).
43
45
  * **Distributed Tracing:** Combines `@vectorize` and `@trace_span` decorators to bundle the execution of complex, multi-step workflows under a single **`trace_id`** for analysis.
44
46
  * **Search Interface:** Provides `search_functions` and `search_executions` to query the stored vector data (function definitions) and logs (execution history), facilitating the construction of RAG and monitoring systems.
45
47
 
@@ -69,7 +71,7 @@ try:
69
71
  except Exception as e:
70
72
  print(f"DB initialization failed: {e}")
71
73
  exit()
72
- ```
74
+ ````
73
75
 
74
76
  ### 2\. [Storage] Using `@vectorize` and Distributed Tracing
75
77
 
@@ -118,6 +120,37 @@ print("Now calling 'process_payment'...")
118
120
  process_payment("user_789", 5000)
119
121
  ```
120
122
 
123
+ #### Semantic Caching Example
124
+
125
+ Configure a function to return a cached result if the input is semantically similar to a previous execution.
126
+
127
+ ```python
128
+ from vectorwave import vectorize
129
+ import time
130
+
131
+ @vectorize(
132
+ search_description="High-cost LLM summarization task",
133
+ sequence_narrative="LLM Summarization Step",
134
+ semantic_cache=True, # Enable caching
135
+ cache_threshold=0.95, # Cache hit if similarity >= 0.95
136
+ capture_return_value=True # Required to save the result
137
+ )
138
+ def summarize_document(document_text: str):
139
+ # Simulate an LLM call or heavy computation (e.g., 0.5 sec delay)
140
+ time.sleep(0.5)
141
+ print("--- [Cache Miss] Document is being summarized by LLM...")
142
+ return f"Summary of: {document_text[:20]}..."
143
+
144
+ # First call (Cache Miss) - takes ~0.5s, saves result to DB
145
+ result_1 = summarize_document("The first quarter results showed strong growth in Europe and Asia...")
146
+
147
+ # Second call (Cache Hit) - takes ~0.0s, returns cached value
148
+ # "Q1 results" is semantically similar to "first quarter results"
149
+ result_2 = summarize_document("The Q1 results demonstrated strong growth in Europe and Asia...")
150
+
151
+ # result_2 returns the stored value without executing the function's body.
152
+ ```
153
+
121
154
  ### 3\. [Retrieval ①] Search Function Definitions (for RAG)
122
155
 
123
156
  ```python
@@ -185,7 +218,12 @@ You can select the text vectorization method via the `VECTORIZER` environment va
185
218
  | **`weaviate_module`** | (Docker Delegate) Delegates vectorization to Weaviate's built-in module (e.g., `text2vec-openai`). | `WEAVIATE_VECTORIZER_MODULE`, `OPENAI_API_KEY` |
186
219
  | **`none`** | Disables vectorization. Data is stored without vectors. | None |
187
220
 
188
- -----
221
+ #### ⚠️ Semantic Caching Prerequisites and Configuration
222
+
223
+ To use `semantic_cache=True`, the following conditions must be met:
224
+
225
+ * **Vectorizer Required:** A **Python-based vectorizer** (`huggingface` or `openai_client`) must be configured in your environment (`VECTORIZER` environment variable). Caching is automatically disabled if set to `weaviate_module` or `none`.
226
+ * **Return Value Capture:** The `capture_return_value` parameter is automatically set to `True` when `semantic_cache=True` is enabled.
189
227
 
190
228
  ### .env File Examples
191
229
 
@@ -266,8 +304,6 @@ FAILURE_MAPPING_FILE_PATH=.vectorwave_errors.json
266
304
  RUN_ID=test-run-001
267
305
  ```
268
306
 
269
- -----
270
-
271
307
  ### 🚀 Advanced Failure Tracing (Error Code)
272
308
 
273
309
  This enhances `VectorWaveExecutions` logs beyond a simple `status: "ERROR"`. An `error_code` property is added to the schema for granular failure analysis.
@@ -396,7 +432,6 @@ def other_function():
396
432
  pass
397
433
  ```
398
434
 
399
-
400
435
  1. **Validation (Important):** Tags (global or function-specific) will **only** be saved to Weaviate if their key (e.g., `run_id`, `team`, `priority`) was first defined in the `.weaviate_properties` file (Step 1). Tags not defined in the schema are **ignored**, and a warning is logged at startup.
401
436
 
402
437
  2. **Priority (Override):** If a tag key is defined in both places (e.g., global `RUN_ID` in `.env` and `run_id="override-xyz"` in the decorator), the **function-specific tag from the decorator always wins**.
@@ -413,6 +448,7 @@ def other_function():
413
448
  Beyond just logging, `VectorWave` can send **real-time notifications via webhook** the instant an error occurs. This functionality is built directly into the tracer and can be activated simply by updating your `.env` file.
414
449
 
415
450
  **How it Works:**
451
+
416
452
  1. An exception is raised within a function decorated by `@trace_span` or `@vectorize`.
417
453
  2. The tracer catches the exception in its `except` block and immediately calls the `alerter` object.
418
454
  3. The alerter reads the `.env` configuration and uses the `WebhookAlerter` to dispatch the error details to your specified URL.
@@ -432,8 +468,6 @@ ALERTER_WEBHOOK_URL="[https://discord.com/api/webhooks/YOUR_HOOK_ID/](https://di
432
468
  With just these two lines, running test_ex/example.py will now instantly send a Discord alert when the CustomValueError is raised.
433
469
 
434
470
  Extensibility (Strategy Pattern): The alerting system is built on a Strategy Pattern. You can easily extend it by implementing the BaseAlerter interface to support other channels like email, PagerDuty, or more.
435
-
436
- **Tag Merging and Validation Rules**
437
471
  ```
438
472
 
439
473
  ## 🤝 Contributing
@@ -1,5 +1,4 @@
1
1
 
2
-
3
2
  # VectorWave: Seamless Auto-Vectorization Framework
4
3
 
5
4
  [](https://opensource.org/licenses/MIT)
@@ -15,6 +14,9 @@
15
14
  * **`@vectorize` Decorator:**
16
15
  1. **Static Data Collection:** Upon script load, the function's source code, docstring, and metadata are saved once to the `VectorWaveFunctions` collection.
17
16
  2. **Dynamic Data Logging:** Each time the function is called, its execution time, success/failure status, error logs, and "dynamic tags" are recorded in the `VectorWaveExecutions` collection.
17
+ * **Semantic Caching and Performance Optimization:**
18
+ * Determines cache hits based on the **semantic similarity** of function inputs, bypassing actual execution for identical or highly similar inputs and returning stored results immediately.
19
+ * This significantly **reduces latency** and costs, especially for high-cost computation functions (e.g., LLM calls, complex data processing).
18
20
  * **Distributed Tracing:** Combines `@vectorize` and `@trace_span` decorators to bundle the execution of complex, multi-step workflows under a single **`trace_id`** for analysis.
19
21
  * **Search Interface:** Provides `search_functions` and `search_executions` to query the stored vector data (function definitions) and logs (execution history), facilitating the construction of RAG and monitoring systems.
20
22
 
@@ -44,7 +46,7 @@ try:
44
46
  except Exception as e:
45
47
  print(f"DB initialization failed: {e}")
46
48
  exit()
47
- ```
49
+ ````
48
50
 
49
51
  ### 2\. [Storage] Using `@vectorize` and Distributed Tracing
50
52
 
@@ -93,6 +95,37 @@ print("Now calling 'process_payment'...")
93
95
  process_payment("user_789", 5000)
94
96
  ```
95
97
 
98
+ #### Semantic Caching Example
99
+
100
+ Configure a function to return a cached result if the input is semantically similar to a previous execution.
101
+
102
+ ```python
103
+ from vectorwave import vectorize
104
+ import time
105
+
106
+ @vectorize(
107
+ search_description="High-cost LLM summarization task",
108
+ sequence_narrative="LLM Summarization Step",
109
+ semantic_cache=True, # Enable caching
110
+ cache_threshold=0.95, # Cache hit if similarity >= 0.95
111
+ capture_return_value=True # Required to save the result
112
+ )
113
+ def summarize_document(document_text: str):
114
+ # Simulate an LLM call or heavy computation (e.g., 0.5 sec delay)
115
+ time.sleep(0.5)
116
+ print("--- [Cache Miss] Document is being summarized by LLM...")
117
+ return f"Summary of: {document_text[:20]}..."
118
+
119
+ # First call (Cache Miss) - takes ~0.5s, saves result to DB
120
+ result_1 = summarize_document("The first quarter results showed strong growth in Europe and Asia...")
121
+
122
+ # Second call (Cache Hit) - takes ~0.0s, returns cached value
123
+ # "Q1 results" is semantically similar to "first quarter results"
124
+ result_2 = summarize_document("The Q1 results demonstrated strong growth in Europe and Asia...")
125
+
126
+ # result_2 returns the stored value without executing the function's body.
127
+ ```
128
+
96
129
  ### 3\. [Retrieval ①] Search Function Definitions (for RAG)
97
130
 
98
131
  ```python
@@ -160,7 +193,12 @@ You can select the text vectorization method via the `VECTORIZER` environment va
160
193
  | **`weaviate_module`** | (Docker Delegate) Delegates vectorization to Weaviate's built-in module (e.g., `text2vec-openai`). | `WEAVIATE_VECTORIZER_MODULE`, `OPENAI_API_KEY` |
161
194
  | **`none`** | Disables vectorization. Data is stored without vectors. | None |
162
195
 
163
- -----
196
+ #### ⚠️ Semantic Caching Prerequisites and Configuration
197
+
198
+ To use `semantic_cache=True`, the following conditions must be met:
199
+
200
+ * **Vectorizer Required:** A **Python-based vectorizer** (`huggingface` or `openai_client`) must be configured in your environment (`VECTORIZER` environment variable). Caching is automatically disabled if set to `weaviate_module` or `none`.
201
+ * **Return Value Capture:** The `capture_return_value` parameter is automatically set to `True` when `semantic_cache=True` is enabled.
164
202
 
165
203
  ### .env File Examples
166
204
 
@@ -241,8 +279,6 @@ FAILURE_MAPPING_FILE_PATH=.vectorwave_errors.json
241
279
  RUN_ID=test-run-001
242
280
  ```
243
281
 
244
- -----
245
-
246
282
  ### 🚀 Advanced Failure Tracing (Error Code)
247
283
 
248
284
  This enhances `VectorWaveExecutions` logs beyond a simple `status: "ERROR"`. An `error_code` property is added to the schema for granular failure analysis.
@@ -371,7 +407,6 @@ def other_function():
371
407
  pass
372
408
  ```
373
409
 
374
-
375
410
  1. **Validation (Important):** Tags (global or function-specific) will **only** be saved to Weaviate if their key (e.g., `run_id`, `team`, `priority`) was first defined in the `.weaviate_properties` file (Step 1). Tags not defined in the schema are **ignored**, and a warning is logged at startup.
376
411
 
377
412
  2. **Priority (Override):** If a tag key is defined in both places (e.g., global `RUN_ID` in `.env` and `run_id="override-xyz"` in the decorator), the **function-specific tag from the decorator always wins**.
@@ -388,6 +423,7 @@ def other_function():
388
423
  Beyond just logging, `VectorWave` can send **real-time notifications via webhook** the instant an error occurs. This functionality is built directly into the tracer and can be activated simply by updating your `.env` file.
389
424
 
390
425
  **How it Works:**
426
+
391
427
  1. An exception is raised within a function decorated by `@trace_span` or `@vectorize`.
392
428
  2. The tracer catches the exception in its `except` block and immediately calls the `alerter` object.
393
429
  3. The alerter reads the `.env` configuration and uses the `WebhookAlerter` to dispatch the error details to your specified URL.
@@ -407,8 +443,6 @@ ALERTER_WEBHOOK_URL="[https://discord.com/api/webhooks/YOUR_HOOK_ID/](https://di
407
443
  With just these two lines, running test_ex/example.py will now instantly send a Discord alert when the CustomValueError is raised.
408
444
 
409
445
  Extensibility (Strategy Pattern): The alerting system is built on a Strategy Pattern. You can easily extend it by implementing the BaseAlerter interface to support other channels like email, PagerDuty, or more.
410
-
411
- **Tag Merging and Validation Rules**
412
446
  ```
413
447
 
414
448
  ## 🤝 Contributing
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "vectorwave"
7
- version = "0.1.6"
7
+ version = "0.1.7"
8
8
  authors = [
9
9
  { name = "junyeonggim", email = "junyeonggim5@gmail.com" },
10
10
  ]
@@ -1,4 +1,4 @@
1
- from unittest.mock import MagicMock
1
+ from unittest.mock import MagicMock, call, ANY
2
2
 
3
3
  import pytest
4
4
  from vectorwave.batch.batch import get_batch_manager
@@ -23,6 +23,9 @@ def mock_deps(monkeypatch):
23
23
  mock_collection.data = mock_collection_data
24
24
  mock_client.collections.get = MagicMock(return_value=mock_collection)
25
25
 
26
+ mock_batch_context = MagicMock()
27
+ mock_client.batch.dynamic.return_value.__enter__.return_value = mock_batch_context
28
+
26
29
  # Mock get_weaviate_client
27
30
  mock_get_client = MagicMock(return_value=mock_client)
28
31
  monkeypatch.setattr("vectorwave.batch.batch.get_weaviate_client", mock_get_client)
@@ -36,6 +39,9 @@ def mock_deps(monkeypatch):
36
39
  mock_atexit_register = MagicMock()
37
40
  monkeypatch.setattr("atexit.register", mock_atexit_register)
38
41
 
42
+ mock_thread = MagicMock()
43
+ monkeypatch.setattr("threading.Thread", mock_thread)
44
+
39
45
  # Clear lru_cache
40
46
  get_batch_manager.cache_clear()
41
47
 
@@ -44,7 +50,8 @@ def mock_deps(monkeypatch):
44
50
  "get_settings": mock_get_settings,
45
51
  "client": mock_client,
46
52
  "settings": mock_settings,
47
- "atexit": mock_atexit_register
53
+ "atexit": mock_atexit_register,
54
+ "batch_context": mock_batch_context
48
55
  }
49
56
 
50
57
  def test_get_batch_manager_is_singleton(mock_deps):
@@ -80,19 +87,75 @@ def test_batch_manager_init_failure(monkeypatch):
80
87
  # The _initialized flag should be False if initialization fails
81
88
  assert manager._initialized is False
82
89
 
83
- def test_add_object_calls_client_batch(mock_deps):
90
+ def test_add_object_enqueues_item(mock_deps):
84
91
  """
85
- Case 4: Test if add_object() correctly calls client.batch.add_object
92
+ [Updated] Case 4: Test if add_object() puts the item into the local queue (Non-blocking)
93
+ Instead of calling client directly.
86
94
  """
87
95
  manager = get_batch_manager()
88
96
  props = {"key": "value"}
89
97
 
90
- manager.add_object(collection="TestCollection", properties=props, uuid="test-uuid")
98
+ assert manager.queue.empty()
99
+
100
+ manager.add_object(collection="TestCollection", properties=props, uuid="test-uuid", vector=[0.1])
101
+
102
+ # 1. Should NOT call DB directly
103
+ mock_deps["client"].collections.get.assert_not_called()
104
+
105
+ # 2. Should be in the Queue
106
+ assert manager.queue.qsize() == 1
107
+ item = manager.queue.get()
108
+
109
+ assert item["collection"] == "TestCollection"
110
+ assert item["properties"] == props
111
+ assert item["uuid"] == "test-uuid"
112
+ assert item["vector"] == [0.1]
113
+
114
+ def test_flush_batch_sends_to_weaviate(mock_deps):
115
+ """
116
+ [New] Case 5: Test if _flush_batch sends items using client.batch.dynamic context
117
+ """
118
+ manager = get_batch_manager()
119
+
120
+ items = [
121
+ {"collection": "C1", "properties": {"p": 1}, "uuid": "u1", "vector": None},
122
+ {"collection": "C2", "properties": {"p": 2}, "uuid": "u2", "vector": [1.0]}
123
+ ]
124
+
125
+ # Manually trigger flush
126
+ manager._flush_batch(items)
127
+
128
+ # Check if dynamic batch context was entered
129
+ mock_deps["client"].batch.dynamic.assert_called_once()
130
+
131
+ # Check if batch.add_object was called for each item
132
+ mock_batch_ctx = mock_deps["batch_context"]
133
+ assert mock_batch_ctx.add_object.call_count == 2
134
+
135
+ mock_batch_ctx.add_object.assert_has_calls([
136
+ call(collection="C1", properties={"p": 1}, uuid="u1", vector=None),
137
+ call(collection="C2", properties={"p": 2}, uuid="u2", vector=[1.0])
138
+ ])
139
+
140
+ def test_flush_batch_reconnects_if_disconnected(mock_deps):
141
+ """
142
+ [New] Case 6: Test reconnection logic when client is not initialized
143
+ """
144
+ manager = get_batch_manager()
145
+
146
+ # Simulate disconnection
147
+ manager._initialized = False
148
+ manager.client = None
149
+
150
+ # Reset mocks
151
+ mock_deps["get_client"].reset_mock()
152
+
153
+ items = [{"collection": "C1", "properties": {}, "uuid": "u1", "vector": None}]
91
154
 
92
- mock_deps["client"].collections.get.assert_called_once_with("TestCollection")
155
+ # Trigger flush
156
+ manager._flush_batch(items)
93
157
 
94
- mock_deps["client"].collections.get.return_value.data.insert.assert_called_once_with(
95
- properties=props,
96
- uuid="test-uuid",
97
- vector=None
98
- )
158
+ # Should try to reconnect
159
+ mock_deps["get_client"].assert_called_once()
160
+ # And then send
161
+ mock_deps["client"].batch.dynamic.assert_called_once()