vectorwave 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. vectorwave-0.1.0/PKG-INFO +280 -0
  2. vectorwave-0.1.0/Readme.md +268 -0
  3. vectorwave-0.1.0/pyproject.toml +26 -0
  4. vectorwave-0.1.0/setup.cfg +4 -0
  5. vectorwave-0.1.0/src/tests/__init__.py +0 -0
  6. vectorwave-0.1.0/src/tests/batch/__init__.py +0 -0
  7. vectorwave-0.1.0/src/tests/batch/test_batch.py +97 -0
  8. vectorwave-0.1.0/src/tests/core/__init__.py +0 -0
  9. vectorwave-0.1.0/src/tests/core/test_decorator.py +345 -0
  10. vectorwave-0.1.0/src/tests/database/__init__.py +0 -0
  11. vectorwave-0.1.0/src/tests/database/test_db.py +464 -0
  12. vectorwave-0.1.0/src/tests/database/test_db_search.py +163 -0
  13. vectorwave-0.1.0/src/tests/exception/__init__.py +0 -0
  14. vectorwave-0.1.0/src/tests/models/__init__.py +0 -0
  15. vectorwave-0.1.0/src/tests/models/test_db_config.py +143 -0
  16. vectorwave-0.1.0/src/tests/monitoring/__init__.py +0 -0
  17. vectorwave-0.1.0/src/tests/prediction/__init__.py +0 -0
  18. vectorwave-0.1.0/src/vectorwave/__init__.py +13 -0
  19. vectorwave-0.1.0/src/vectorwave/batch/__init__.py +0 -0
  20. vectorwave-0.1.0/src/vectorwave/batch/batch.py +65 -0
  21. vectorwave-0.1.0/src/vectorwave/core/__init__.py +0 -0
  22. vectorwave-0.1.0/src/vectorwave/core/core.py +0 -0
  23. vectorwave-0.1.0/src/vectorwave/core/decorator.py +108 -0
  24. vectorwave-0.1.0/src/vectorwave/database/__init__.py +0 -0
  25. vectorwave-0.1.0/src/vectorwave/database/db.py +302 -0
  26. vectorwave-0.1.0/src/vectorwave/database/db_search.py +100 -0
  27. vectorwave-0.1.0/src/vectorwave/exception/__init__.py +0 -0
  28. vectorwave-0.1.0/src/vectorwave/exception/exceptions.py +22 -0
  29. vectorwave-0.1.0/src/vectorwave/models/__init__.py +0 -0
  30. vectorwave-0.1.0/src/vectorwave/models/db_config.py +82 -0
  31. vectorwave-0.1.0/src/vectorwave/monitoring/__init__.py +0 -0
  32. vectorwave-0.1.0/src/vectorwave/monitoring/monitoring.py +0 -0
  33. vectorwave-0.1.0/src/vectorwave/monitoring/tracer.py +128 -0
  34. vectorwave-0.1.0/src/vectorwave/prediction/__init__.py +0 -0
  35. vectorwave-0.1.0/src/vectorwave/prediction/predictor.py +0 -0
  36. vectorwave-0.1.0/src/vectorwave.egg-info/PKG-INFO +280 -0
  37. vectorwave-0.1.0/src/vectorwave.egg-info/SOURCES.txt +38 -0
  38. vectorwave-0.1.0/src/vectorwave.egg-info/dependency_links.txt +1 -0
  39. vectorwave-0.1.0/src/vectorwave.egg-info/requires.txt +2 -0
  40. vectorwave-0.1.0/src/vectorwave.egg-info/top_level.txt +2 -0
@@ -0,0 +1,280 @@
1
+ Metadata-Version: 2.4
2
+ Name: vectorwave
3
+ Version: 0.1.0
4
+ Summary: VectorWave: Seamless Auto-Vectorization Framework
5
+ Author-email: junyeonggim <junyeonggim5@gmail.com>
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Operating System :: OS Independent
8
+ Requires-Python: >=3.8
9
+ Description-Content-Type: text/markdown
10
+ Requires-Dist: weaviate-client>=4.0.0
11
+ Requires-Dist: pydantic-settings>=2.0.0
12
+
13
+
14
+ # VectorWave: Seamless Auto-Vectorization Framework
15
+
16
+ [](https://www.google.com/search?q=LICENSE)
17
+
18
+ ## 🌟 Overview
19
+
20
+ **VectorWave** is an innovative framework that uses a **decorator** to automatically save and manage the output of Python functions/methods in a **Vector Database (Vector DB)**. Developers can convert function outputs into intelligent vector data with a single line of code (`@vectorize`), without worrying about the complex processes of data collection, embedding generation, or storage in a Vector DB.
21
+
22
+ ---
23
+
24
+ ## ✨ Features
25
+
26
+ * **`@vectorize` Decorator:**
27
+ 1. **Static Data Collection:** Saves the function's source code, docstring, and metadata to the `VectorWaveFunctions` collection once when the script is loaded.
28
+ 2. **Dynamic Data Logging:** Records the execution time, success/failure status, error logs, and 'dynamic tags' to the `VectorWaveExecutions` collection every time the function is called.
29
+ * **Distributed Tracing:** By combining the `@vectorize` and `@trace_span` decorators, you can analyze the execution of complex multi-step workflows, grouped under a single **`trace_id`**.
30
+ * **Search Interface:** Provides `search_functions` (for vector search) and `search_executions` (for log filtering) to facilitate the construction of RAG and monitoring systems.
31
+
32
+ ---
33
+
34
+ ## 🚀 Usage
35
+
36
+ VectorWave consists of 'storing' via decorators and 'searching' via functions, and now includes **execution flow tracing**.
37
+
38
+ ### 1. (Required) Initialize the Database and Configuration
39
+
40
+ ```python
41
+ import time
42
+ from vectorwave import (
43
+ vectorize,
44
+ initialize_database,
45
+ search_functions,
46
+ search_executions
47
+ )
48
+ # [ADDITION] Import trace_span separately for distributed tracing.
49
+ from vectorwave.monitoring.tracer import trace_span
50
+
51
+ # This only needs to be called once when the script starts.
52
+ try:
53
+ client = initialize_database()
54
+ print("VectorWave DB initialized successfully.")
55
+ except Exception as e:
56
+ print(f"DB initialization failed: {e}")
57
+ exit()
58
+ ````
59
+
60
+ ### 2\. [Store] Use `@vectorize` with Distributed Tracing
61
+
62
+ The `@vectorize` acts as the **Root** for tracing, and `@trace_span` is used on internal functions to group the execution flow under a single `trace_id`.
63
+
64
+ ```python
65
+ # --- Child Span Function: Captures arguments ---
66
+ @trace_span(attributes_to_capture=['user_id', 'amount'])
67
+ def step_1_validate_payment(user_id: str, amount: int):
68
+ """(Span) Payment validation. Records user_id and amount in the log."""
69
+ print(f" [SPAN 1] Validating payment for {user_id}...")
70
+ time.sleep(0.1)
71
+ return True
72
+
73
+ @trace_span(attributes_to_capture=['user_id', 'receipt_id'])
74
+ def step_2_send_receipt(user_id: str, receipt_id: str):
75
+ """(Span) Sends the receipt."""
76
+ print(f" [SPAN 2] Sending receipt {receipt_id}...")
77
+ time.sleep(0.2)
78
+
79
+
80
+ # --- Root Function (@trace_root role) ---
81
+ @vectorize(
82
+ search_description="Charges a user in the payment system.",
83
+ sequence_narrative="Returns a receipt ID upon successful payment.",
84
+ team="billing", # <-- Custom Tag (recorded in all execution logs)
85
+ priority=1 # <-- Custom Tag (execution priority)
86
+ )
87
+ def process_payment(user_id: str, amount: int):
88
+ """(Root Span) Executes the user payment workflow."""
89
+ print(f" [ROOT EXEC] process_payment: Starting workflow for {user_id}...")
90
+
91
+ # When calling child functions, the same trace_id is automatically inherited via ContextVar.
92
+ step_1_validate_payment(user_id=user_id, amount=amount)
93
+
94
+ receipt_id = f"receipt_{user_id}_{amount}"
95
+ step_2_send_receipt(user_id=user_id, receipt_id=receipt_id)
96
+
97
+ print(f" [ROOT DONE] process_payment")
98
+ return {"status": "success", "receipt_id": receipt_id}
99
+
100
+ # --- Execute the Function ---
101
+ print("Now calling 'process_payment'...")
102
+ # This single call records 3 execution logs (spans) in the DB,
103
+ # all grouped under one 'trace_id'.
104
+ process_payment("user_789", 5000)
105
+ ```
106
+
107
+ ### 3\. [Search ①] Function Definition Search (for RAG)
108
+
109
+ ```python
110
+ # Search for functions related to 'payment' using natural language (vector search).
111
+ print("\n--- Searching for 'payment' functions ---")
112
+ payment_funcs = search_functions(
113
+ query="user payment processing",
114
+ limit=3
115
+ )
116
+ for func in payment_funcs:
117
+ print(f" - Function: {func['properties']['function_name']}")
118
+ print(f" - Description: {func['properties']['search_description']}")
119
+ print(f" - Similarity (Distance): {func['metadata'].distance:.4f}")
120
+ ```
121
+
122
+ ### 4\. [Search ②] Execution Log Search (Monitoring and Tracing)
123
+
124
+ The `search_executions` function can now search for all related execution logs (spans) based on the `trace_id`.
125
+
126
+ ```python
127
+ # 1. Find the Trace ID of a specific workflow (process_payment).
128
+ latest_payment_span = search_executions(
129
+ limit=1,
130
+ filters={"function_name": "process_payment"},
131
+ sort_by="timestamp_utc",
132
+ sort_ascending=False
133
+ )
134
+ trace_id = latest_payment_span[0]["trace_id"]
135
+
136
+ # 2. Search all spans belonging to that Trace ID, sorted chronologically.
137
+ print(f"\n--- Full Trace for ID ({trace_id[:8]}...) ---")
138
+ trace_spans = search_executions(
139
+ limit=10,
140
+ filters={"trace_id": trace_id},
141
+ sort_by="timestamp_utc",
142
+ sort_ascending=True # Ascending sort for workflow flow analysis
143
+ )
144
+
145
+ for i, span in enumerate(trace_spans):
146
+ print(f" - [Span {i+1}] {span['function_name']} ({span['duration_ms']:.2f}ms)")
147
+ # Captured arguments (user_id, amount, etc.) are displayed for the child spans.
148
+
149
+ # Example Output:
150
+ # - [Span 1] step_1_validate_payment (100.81ms)
151
+ # - [Span 2] step_2_send_receipt (202.06ms)
152
+ # - [Span 3] process_payment (333.18ms)
153
+ ```
154
+
155
+ -----
156
+
157
+ ## ⚙️ Configuration
158
+
159
+ VectorWave automatically reads Weaviate database connection information from **environment variables** or a `.env` file.
160
+
161
+ Create a `.env` file in the root directory of your project (e.g., where `main.py` is located) and set the required values.
162
+
163
+ ### .env File Example
164
+
165
+ ```ini
166
+ # .env
167
+ # --- Basic Weaviate Connection Settings ---
168
+ WEAVIATE_HOST=localhost
169
+ WEAVIATE_PORT=8080
170
+ WEAVIATE_GRPC_PORT=50051
171
+
172
+ # --- Vectorizer , Generative Module Config ---
173
+ # (default: text2vec-openai) Set to 'none' to disable vectorization.
174
+ VECTORIZER_CONFIG=text2vec-openai
175
+ # (default: generative-openai)
176
+ GENERATIVE_CONFIG=generative-openai
177
+ # An OpenAI API key is required if using modules like text2vec-openai.
178
+ # OPENAI_API_KEY=sk-your-key-here
179
+
180
+ # --- [Advanced] Custom Property Settings ---
181
+ # 1. The path to the JSON file defining custom properties to add to the schema.
182
+ CUSTOM_PROPERTIES_FILE_PATH=.weaviate_properties
183
+
184
+ # 2. Environment variables to be used for 'Global Dynamic Tagging'.
185
+ # ("run_id" must be defined in the .weaviate_properties file)
186
+ RUN_ID=test-run-001
187
+ EXPERIMENT_ID=exp-abc
188
+ ```
189
+
190
+ -----
191
+
192
+ ### Custom Properties and Dynamic Execution Tagging
193
+
194
+ VectorWave can store user-defined metadata in both static definitions (`VectorWaveFunctions`) and dynamic logs (`VectorWaveExecutions`). This works in two steps.
195
+
196
+ #### Step 1: Define Custom Schema (The "Allow-List")
197
+
198
+ Create a JSON file at the path specified by `CUSTOM_PROPERTIES_FILE_PATH` (default: `.weaviate_properties`).
199
+
200
+ This file instructs VectorWave to add **new properties (columns)** to the Weaviate collections. This file acts as an **"allow-list"** for all custom tags.
201
+
202
+ **`.weaviate_properties` Example:**
203
+
204
+ ```json
205
+ {
206
+ "run_id": {
207
+ "data_type": "TEXT",
208
+ "description": "The ID of the specific test run"
209
+ },
210
+ "experiment_id": {
211
+ "data_type": "TEXT",
212
+ "description": "Identifier for the experiment"
213
+ },
214
+ "team": {
215
+ "data_type": "TEXT",
216
+ "description": "The team responsible for this function"
217
+ },
218
+ "priority": {
219
+ "data_type": "INT",
220
+ "description": "Execution priority level"
221
+ }
222
+ }
223
+ ```
224
+
225
+ * Defining these will add `run_id`, `experiment_id`, `team`, and `priority` properties to *both* collections.
226
+
227
+ #### Step 2: Dynamic Execution Tagging (Adding Values)
228
+
229
+ When a function executes, VectorWave adds tags to the `VectorWaveExecutions` log. It does this in two ways, which are then merged:
230
+
231
+ **1. Global Tags (from Environment Variables)**
232
+ VectorWave searches for environment variables whose names match the **uppercase** keys from Step 1 (e.g., `RUN_ID`, `EXPERIMENT_ID`) and uses these for run-wide metadata.
233
+
234
+ **2. Function-Specific Tags (from Decorator)**
235
+ You can pass tags directly to the `@vectorize` decorator as keyword arguments (`**execution_tags`). This is ideal for function-specific metadata.
236
+
237
+ ```python
238
+ # --- .env file ---
239
+ # RUN_ID=global-run-abc
240
+ # TEAM=default-team
241
+
242
+ @vectorize(
243
+ search_description="Process a payment",
244
+ sequence_narrative="...",
245
+ team="billing", # <-- Function-specific tag
246
+ priority=1 # <-- Function-specific tag
247
+ )
248
+ def process_payment():
249
+ pass
250
+
251
+ @vectorize(
252
+ search_description="Another function",
253
+ sequence_narrative="...",
254
+ run_id="override-run-xyz" # <-- Overrides the global tag
255
+ )
256
+ def other_function():
257
+ pass
258
+ ```
259
+
260
+ **Tag Merging and Validation Rules**
261
+
262
+ 1. **Validation (Most Important):** A tag (either global or specific) will **only** be saved to Weaviate if its key (e.g., `run_id`, `team`, `priority`) was first defined in your `.weaviate_properties` file (Step 1). Tags not defined in the schema will be **ignored**, and a warning will be printed on startup.
263
+
264
+ 2. **Priority (Override):** If a tag key is defined in both places (e.g., a global `RUN_ID` in `.env` and a specific `run_id="override-run-xyz"` on the decorator), the **function-specific tag from the decorator will always win**.
265
+
266
+ **Resulting Logs:**
267
+
268
+ * `process_payment()` log will have: `{"run_id": "global-run-abc", "team": "billing", "priority": 1}`
269
+ * `other_function()` log will have: `{"run_id": "override-run-xyz", "team": "default-team"}`
270
+
271
+ -----
272
+
273
+ ## 🤝 Contributing
274
+
275
+ All forms of contribution are welcome, including bug reports, feature requests, and code contributions. For details, please refer to [CONTRIBUTING.md](https://www.google.com/search?q=httpsS://www.google.com/search%3Fq%3DCONTRIBUTING.md).
276
+
277
+ ## 📜 License
278
+
279
+ This project is distributed under the MIT License. See the [LICENSE](https://www.google.com/search?q=httpsS://www.google.com/search%3Fq%3DLICENSE) file for details.
280
+
@@ -0,0 +1,268 @@
1
+
2
+ # VectorWave: Seamless Auto-Vectorization Framework
3
+
4
+ [](https://www.google.com/search?q=LICENSE)
5
+
6
+ ## 🌟 Overview
7
+
8
+ **VectorWave** is an innovative framework that uses a **decorator** to automatically save and manage the output of Python functions/methods in a **Vector Database (Vector DB)**. Developers can convert function outputs into intelligent vector data with a single line of code (`@vectorize`), without worrying about the complex processes of data collection, embedding generation, or storage in a Vector DB.
9
+
10
+ ---
11
+
12
+ ## ✨ Features
13
+
14
+ * **`@vectorize` Decorator:**
15
+ 1. **Static Data Collection:** Saves the function's source code, docstring, and metadata to the `VectorWaveFunctions` collection once when the script is loaded.
16
+ 2. **Dynamic Data Logging:** Records the execution time, success/failure status, error logs, and 'dynamic tags' to the `VectorWaveExecutions` collection every time the function is called.
17
+ * **Distributed Tracing:** By combining the `@vectorize` and `@trace_span` decorators, you can analyze the execution of complex multi-step workflows, grouped under a single **`trace_id`**.
18
+ * **Search Interface:** Provides `search_functions` (for vector search) and `search_executions` (for log filtering) to facilitate the construction of RAG and monitoring systems.
19
+
20
+ ---
21
+
22
+ ## 🚀 Usage
23
+
24
+ VectorWave consists of 'storing' via decorators and 'searching' via functions, and now includes **execution flow tracing**.
25
+
26
+ ### 1. (Required) Initialize the Database and Configuration
27
+
28
+ ```python
29
+ import time
30
+ from vectorwave import (
31
+ vectorize,
32
+ initialize_database,
33
+ search_functions,
34
+ search_executions
35
+ )
36
+ # [ADDITION] Import trace_span separately for distributed tracing.
37
+ from vectorwave.monitoring.tracer import trace_span
38
+
39
+ # This only needs to be called once when the script starts.
40
+ try:
41
+ client = initialize_database()
42
+ print("VectorWave DB initialized successfully.")
43
+ except Exception as e:
44
+ print(f"DB initialization failed: {e}")
45
+ exit()
46
+ ````
47
+
48
+ ### 2\. [Store] Use `@vectorize` with Distributed Tracing
49
+
50
+ The `@vectorize` acts as the **Root** for tracing, and `@trace_span` is used on internal functions to group the execution flow under a single `trace_id`.
51
+
52
+ ```python
53
+ # --- Child Span Function: Captures arguments ---
54
+ @trace_span(attributes_to_capture=['user_id', 'amount'])
55
+ def step_1_validate_payment(user_id: str, amount: int):
56
+ """(Span) Payment validation. Records user_id and amount in the log."""
57
+ print(f" [SPAN 1] Validating payment for {user_id}...")
58
+ time.sleep(0.1)
59
+ return True
60
+
61
+ @trace_span(attributes_to_capture=['user_id', 'receipt_id'])
62
+ def step_2_send_receipt(user_id: str, receipt_id: str):
63
+ """(Span) Sends the receipt."""
64
+ print(f" [SPAN 2] Sending receipt {receipt_id}...")
65
+ time.sleep(0.2)
66
+
67
+
68
+ # --- Root Function (@trace_root role) ---
69
+ @vectorize(
70
+ search_description="Charges a user in the payment system.",
71
+ sequence_narrative="Returns a receipt ID upon successful payment.",
72
+ team="billing", # <-- Custom Tag (recorded in all execution logs)
73
+ priority=1 # <-- Custom Tag (execution priority)
74
+ )
75
+ def process_payment(user_id: str, amount: int):
76
+ """(Root Span) Executes the user payment workflow."""
77
+ print(f" [ROOT EXEC] process_payment: Starting workflow for {user_id}...")
78
+
79
+ # When calling child functions, the same trace_id is automatically inherited via ContextVar.
80
+ step_1_validate_payment(user_id=user_id, amount=amount)
81
+
82
+ receipt_id = f"receipt_{user_id}_{amount}"
83
+ step_2_send_receipt(user_id=user_id, receipt_id=receipt_id)
84
+
85
+ print(f" [ROOT DONE] process_payment")
86
+ return {"status": "success", "receipt_id": receipt_id}
87
+
88
+ # --- Execute the Function ---
89
+ print("Now calling 'process_payment'...")
90
+ # This single call records 3 execution logs (spans) in the DB,
91
+ # all grouped under one 'trace_id'.
92
+ process_payment("user_789", 5000)
93
+ ```
94
+
95
+ ### 3\. [Search ①] Function Definition Search (for RAG)
96
+
97
+ ```python
98
+ # Search for functions related to 'payment' using natural language (vector search).
99
+ print("\n--- Searching for 'payment' functions ---")
100
+ payment_funcs = search_functions(
101
+ query="user payment processing",
102
+ limit=3
103
+ )
104
+ for func in payment_funcs:
105
+ print(f" - Function: {func['properties']['function_name']}")
106
+ print(f" - Description: {func['properties']['search_description']}")
107
+ print(f" - Similarity (Distance): {func['metadata'].distance:.4f}")
108
+ ```
109
+
110
+ ### 4\. [Search ②] Execution Log Search (Monitoring and Tracing)
111
+
112
+ The `search_executions` function can now search for all related execution logs (spans) based on the `trace_id`.
113
+
114
+ ```python
115
+ # 1. Find the Trace ID of a specific workflow (process_payment).
116
+ latest_payment_span = search_executions(
117
+ limit=1,
118
+ filters={"function_name": "process_payment"},
119
+ sort_by="timestamp_utc",
120
+ sort_ascending=False
121
+ )
122
+ trace_id = latest_payment_span[0]["trace_id"]
123
+
124
+ # 2. Search all spans belonging to that Trace ID, sorted chronologically.
125
+ print(f"\n--- Full Trace for ID ({trace_id[:8]}...) ---")
126
+ trace_spans = search_executions(
127
+ limit=10,
128
+ filters={"trace_id": trace_id},
129
+ sort_by="timestamp_utc",
130
+ sort_ascending=True # Ascending sort for workflow flow analysis
131
+ )
132
+
133
+ for i, span in enumerate(trace_spans):
134
+ print(f" - [Span {i+1}] {span['function_name']} ({span['duration_ms']:.2f}ms)")
135
+ # Captured arguments (user_id, amount, etc.) are displayed for the child spans.
136
+
137
+ # Example Output:
138
+ # - [Span 1] step_1_validate_payment (100.81ms)
139
+ # - [Span 2] step_2_send_receipt (202.06ms)
140
+ # - [Span 3] process_payment (333.18ms)
141
+ ```
142
+
143
+ -----
144
+
145
+ ## ⚙️ Configuration
146
+
147
+ VectorWave automatically reads Weaviate database connection information from **environment variables** or a `.env` file.
148
+
149
+ Create a `.env` file in the root directory of your project (e.g., where `main.py` is located) and set the required values.
150
+
151
+ ### .env File Example
152
+
153
+ ```ini
154
+ # .env
155
+ # --- Basic Weaviate Connection Settings ---
156
+ WEAVIATE_HOST=localhost
157
+ WEAVIATE_PORT=8080
158
+ WEAVIATE_GRPC_PORT=50051
159
+
160
+ # --- Vectorizer , Generative Module Config ---
161
+ # (default: text2vec-openai) Set to 'none' to disable vectorization.
162
+ VECTORIZER_CONFIG=text2vec-openai
163
+ # (default: generative-openai)
164
+ GENERATIVE_CONFIG=generative-openai
165
+ # An OpenAI API key is required if using modules like text2vec-openai.
166
+ # OPENAI_API_KEY=sk-your-key-here
167
+
168
+ # --- [Advanced] Custom Property Settings ---
169
+ # 1. The path to the JSON file defining custom properties to add to the schema.
170
+ CUSTOM_PROPERTIES_FILE_PATH=.weaviate_properties
171
+
172
+ # 2. Environment variables to be used for 'Global Dynamic Tagging'.
173
+ # ("run_id" must be defined in the .weaviate_properties file)
174
+ RUN_ID=test-run-001
175
+ EXPERIMENT_ID=exp-abc
176
+ ```
177
+
178
+ -----
179
+
180
+ ### Custom Properties and Dynamic Execution Tagging
181
+
182
+ VectorWave can store user-defined metadata in both static definitions (`VectorWaveFunctions`) and dynamic logs (`VectorWaveExecutions`). This works in two steps.
183
+
184
+ #### Step 1: Define Custom Schema (The "Allow-List")
185
+
186
+ Create a JSON file at the path specified by `CUSTOM_PROPERTIES_FILE_PATH` (default: `.weaviate_properties`).
187
+
188
+ This file instructs VectorWave to add **new properties (columns)** to the Weaviate collections. This file acts as an **"allow-list"** for all custom tags.
189
+
190
+ **`.weaviate_properties` Example:**
191
+
192
+ ```json
193
+ {
194
+ "run_id": {
195
+ "data_type": "TEXT",
196
+ "description": "The ID of the specific test run"
197
+ },
198
+ "experiment_id": {
199
+ "data_type": "TEXT",
200
+ "description": "Identifier for the experiment"
201
+ },
202
+ "team": {
203
+ "data_type": "TEXT",
204
+ "description": "The team responsible for this function"
205
+ },
206
+ "priority": {
207
+ "data_type": "INT",
208
+ "description": "Execution priority level"
209
+ }
210
+ }
211
+ ```
212
+
213
+ * Defining these will add `run_id`, `experiment_id`, `team`, and `priority` properties to *both* collections.
214
+
215
+ #### Step 2: Dynamic Execution Tagging (Adding Values)
216
+
217
+ When a function executes, VectorWave adds tags to the `VectorWaveExecutions` log. It does this in two ways, which are then merged:
218
+
219
+ **1. Global Tags (from Environment Variables)**
220
+ VectorWave searches for environment variables whose names match the **uppercase** keys from Step 1 (e.g., `RUN_ID`, `EXPERIMENT_ID`) and uses these for run-wide metadata.
221
+
222
+ **2. Function-Specific Tags (from Decorator)**
223
+ You can pass tags directly to the `@vectorize` decorator as keyword arguments (`**execution_tags`). This is ideal for function-specific metadata.
224
+
225
+ ```python
226
+ # --- .env file ---
227
+ # RUN_ID=global-run-abc
228
+ # TEAM=default-team
229
+
230
+ @vectorize(
231
+ search_description="Process a payment",
232
+ sequence_narrative="...",
233
+ team="billing", # <-- Function-specific tag
234
+ priority=1 # <-- Function-specific tag
235
+ )
236
+ def process_payment():
237
+ pass
238
+
239
+ @vectorize(
240
+ search_description="Another function",
241
+ sequence_narrative="...",
242
+ run_id="override-run-xyz" # <-- Overrides the global tag
243
+ )
244
+ def other_function():
245
+ pass
246
+ ```
247
+
248
+ **Tag Merging and Validation Rules**
249
+
250
+ 1. **Validation (Most Important):** A tag (either global or specific) will **only** be saved to Weaviate if its key (e.g., `run_id`, `team`, `priority`) was first defined in your `.weaviate_properties` file (Step 1). Tags not defined in the schema will be **ignored**, and a warning will be printed on startup.
251
+
252
+ 2. **Priority (Override):** If a tag key is defined in both places (e.g., a global `RUN_ID` in `.env` and a specific `run_id="override-run-xyz"` on the decorator), the **function-specific tag from the decorator will always win**.
253
+
254
+ **Resulting Logs:**
255
+
256
+ * `process_payment()` log will have: `{"run_id": "global-run-abc", "team": "billing", "priority": 1}`
257
+ * `other_function()` log will have: `{"run_id": "override-run-xyz", "team": "default-team"}`
258
+
259
+ -----
260
+
261
+ ## 🤝 Contributing
262
+
263
+ All forms of contribution are welcome, including bug reports, feature requests, and code contributions. For details, please refer to [CONTRIBUTING.md](https://www.google.com/search?q=httpsS://www.google.com/search%3Fq%3DCONTRIBUTING.md).
264
+
265
+ ## 📜 License
266
+
267
+ This project is distributed under the MIT License. See the [LICENSE](https://www.google.com/search?q=httpsS://www.google.com/search%3Fq%3DLICENSE) file for details.
268
+
@@ -0,0 +1,26 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "vectorwave"
7
+ version = "0.1.0"
8
+ authors = [
9
+ { name = "junyeonggim", email = "junyeonggim5@gmail.com" },
10
+ ]
11
+ description = "VectorWave: Seamless Auto-Vectorization Framework"
12
+ readme = "Readme.md"
13
+ requires-python = ">=3.8"
14
+ classifiers = [
15
+ "Programming Language :: Python :: 3",
16
+ "Operating System :: OS Independent",
17
+ ]
18
+
19
+ dependencies = [
20
+ "weaviate-client>=4.0.0",
21
+ "pydantic-settings>=2.0.0"
22
+ ]
23
+ # ⬆️⬆️⬆️ ⬆️⬆️⬆️
24
+
25
+ [tool.setuptools.packages.find]
26
+ where = ["src"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
File without changes
File without changes
@@ -0,0 +1,97 @@
1
+ from unittest.mock import MagicMock
2
+
3
+ import pytest
4
+ from vectorwave.batch.batch import get_batch_manager
5
+ from vectorwave.exception.exceptions import WeaviateConnectionError
6
+ from vectorwave.models.db_config import WeaviateSettings
7
+
8
+
9
+ @pytest.fixture
10
+ def mock_deps(monkeypatch):
11
+ """
12
+ Fixture to mock dependencies for batch.py (db, config, atexit)
13
+ """
14
+ # Mock WeaviateClient
15
+ mock_client = MagicMock()
16
+ mock_client.batch = MagicMock()
17
+ mock_client.batch.configure = MagicMock()
18
+ mock_client.batch.add_object = MagicMock()
19
+ mock_client.batch.flush = MagicMock()
20
+
21
+ mock_collection_data = MagicMock()
22
+ mock_collection = MagicMock()
23
+ mock_collection.data = mock_collection_data
24
+ mock_client.collections.get = MagicMock(return_value=mock_collection)
25
+
26
+ # Mock get_weaviate_client
27
+ mock_get_client = MagicMock(return_value=mock_client)
28
+ monkeypatch.setattr("vectorwave.batch.batch.get_weaviate_client", mock_get_client)
29
+
30
+ # Mock get_weaviate_settings
31
+ mock_settings = WeaviateSettings()
32
+ mock_get_settings = MagicMock(return_value=mock_settings)
33
+ monkeypatch.setattr("vectorwave.batch.batch.get_weaviate_settings", mock_get_settings)
34
+
35
+ # Mock atexit.register
36
+ mock_atexit_register = MagicMock()
37
+ monkeypatch.setattr("atexit.register", mock_atexit_register)
38
+
39
+ # Clear lru_cache
40
+ get_batch_manager.cache_clear()
41
+
42
+ return {
43
+ "get_client": mock_get_client,
44
+ "get_settings": mock_get_settings,
45
+ "client": mock_client,
46
+ "settings": mock_settings,
47
+ "atexit": mock_atexit_register
48
+ }
49
+
50
+ def test_get_batch_manager_is_singleton(mock_deps):
51
+ """
52
+ Case 1: Test if get_batch_manager() always returns the same instance (singleton)
53
+ """
54
+ manager1 = get_batch_manager()
55
+ manager2 = get_batch_manager()
56
+ assert manager1 is manager2
57
+
58
+ def test_batch_manager_initialization(mock_deps):
59
+ """
60
+ Case 2: Test if BatchManager correctly calls dependencies (configure, atexit) upon initialization
61
+ """
62
+ manager = get_batch_manager()
63
+
64
+ mock_deps["get_settings"].assert_called_once()
65
+ mock_deps["get_client"].assert_called_once_with(mock_deps["settings"])
66
+
67
+ assert manager._initialized is True
68
+
69
+ def test_batch_manager_init_failure(monkeypatch):
70
+ """
71
+ Case 3: Test if _initialized remains False when DB connection (get_weaviate_client) fails
72
+ """
73
+ # Mock get_weaviate_client to raise an exception
74
+ mock_get_client_fail = MagicMock(side_effect=WeaviateConnectionError("Test connection error"))
75
+ monkeypatch.setattr("vectorwave.batch.batch.get_weaviate_client", mock_get_client_fail)
76
+
77
+ get_batch_manager.cache_clear()
78
+ manager = get_batch_manager()
79
+
80
+ # The _initialized flag should be False if initialization fails
81
+ assert manager._initialized is False
82
+
83
+ def test_add_object_calls_client_batch(mock_deps):
84
+ """
85
+ Case 4: Test if add_object() correctly calls client.batch.add_object
86
+ """
87
+ manager = get_batch_manager()
88
+ props = {"key": "value"}
89
+
90
+ manager.add_object(collection="TestCollection", properties=props, uuid="test-uuid")
91
+
92
+ mock_deps["client"].collections.get.assert_called_once_with("TestCollection")
93
+
94
+ mock_deps["client"].collections.get.return_value.data.insert.assert_called_once_with(
95
+ properties=props,
96
+ uuid="test-uuid"
97
+ )
File without changes