vectorwave 0.3.0__tar.gz → 1.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. vectorwave-1.0.1/PKG-INFO +360 -0
  2. vectorwave-1.0.1/Readme.md +312 -0
  3. {vectorwave-0.3.0 → vectorwave-1.0.1}/pyproject.toml +35 -4
  4. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/__init__.py +5 -0
  5. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/batch/batch.py +69 -6
  6. vectorwave-1.0.1/src/vectorwave/check/__init__.py +17 -0
  7. vectorwave-1.0.1/src/vectorwave/check/calibrate.py +358 -0
  8. vectorwave-1.0.1/src/vectorwave/check/cli.py +74 -0
  9. vectorwave-1.0.1/src/vectorwave/check/config.py +75 -0
  10. vectorwave-1.0.1/src/vectorwave/check/plugin.py +175 -0
  11. vectorwave-1.0.1/src/vectorwave/check/report.py +48 -0
  12. vectorwave-1.0.1/src/vectorwave/cli/__init__.py +316 -0
  13. vectorwave-1.0.1/src/vectorwave/cli/__main__.py +6 -0
  14. vectorwave-1.0.1/src/vectorwave/cli/dev/compose.yml +33 -0
  15. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/core/auto_injector.py +11 -0
  16. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/core/decorator.py +16 -5
  17. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/core/generator.py +15 -1
  18. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/database/archiver.py +35 -34
  19. vectorwave-1.0.1/src/vectorwave/database/dataset.py +154 -0
  20. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/database/db.py +13 -8
  21. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/database/db_search.py +127 -146
  22. vectorwave-1.0.1/src/vectorwave/monitoring/otel.py +167 -0
  23. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/monitoring/tracer.py +87 -28
  24. vectorwave-1.0.1/src/vectorwave/runtime.py +210 -0
  25. vectorwave-1.0.1/src/vectorwave/store/__init__.py +19 -0
  26. vectorwave-1.0.1/src/vectorwave/store/base.py +144 -0
  27. vectorwave-1.0.1/src/vectorwave/store/factory.py +43 -0
  28. vectorwave-1.0.1/src/vectorwave/store/lance_store.py +364 -0
  29. vectorwave-1.0.1/src/vectorwave/store/weaviate_store.py +306 -0
  30. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/function_cache.py +16 -8
  31. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/healer.py +18 -2
  32. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/replayer.py +52 -36
  33. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/return_caching_utils.py +36 -34
  34. vectorwave-1.0.1/src/vectorwave/vectorizer/__init__.py +0 -0
  35. vectorwave-0.3.0/PKG-INFO +0 -258
  36. vectorwave-0.3.0/Readme.md +0 -229
  37. vectorwave-0.3.0/src/vectorwave/database/dataset.py +0 -158
  38. {vectorwave-0.3.0 → vectorwave-1.0.1}/crates/Cargo.lock +0 -0
  39. {vectorwave-0.3.0 → vectorwave-1.0.1}/crates/Cargo.toml +0 -0
  40. {vectorwave-0.3.0 → vectorwave-1.0.1}/crates/src/lib.rs +0 -0
  41. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/batch/__init__.py +0 -0
  42. {vectorwave-0.3.0/src/vectorwave/core → vectorwave-1.0.1/src/vectorwave/cli/dev}/__init__.py +0 -0
  43. {vectorwave-0.3.0/src/vectorwave/core/llm → vectorwave-1.0.1/src/vectorwave/core}/__init__.py +0 -0
  44. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/core/core.py +0 -0
  45. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/core/initializer.py +0 -0
  46. {vectorwave-0.3.0/src/vectorwave/database → vectorwave-1.0.1/src/vectorwave/core/llm}/__init__.py +0 -0
  47. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/core/llm/base.py +0 -0
  48. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/core/llm/factory.py +0 -0
  49. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/core/llm/openai_client.py +0 -0
  50. {vectorwave-0.3.0/src/vectorwave/exception → vectorwave-1.0.1/src/vectorwave/database}/__init__.py +0 -0
  51. {vectorwave-0.3.0/src/vectorwave/models → vectorwave-1.0.1/src/vectorwave/exception}/__init__.py +0 -0
  52. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/exception/exceptions.py +0 -0
  53. {vectorwave-0.3.0/src/vectorwave/monitoring → vectorwave-1.0.1/src/vectorwave/models}/__init__.py +0 -0
  54. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/models/db_config.py +0 -0
  55. {vectorwave-0.3.0/src/vectorwave/monitoring/alert → vectorwave-1.0.1/src/vectorwave/monitoring}/__init__.py +0 -0
  56. {vectorwave-0.3.0/src/vectorwave/prediction → vectorwave-1.0.1/src/vectorwave/monitoring/alert}/__init__.py +0 -0
  57. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/monitoring/alert/base.py +0 -0
  58. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/monitoring/alert/factory.py +0 -0
  59. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/monitoring/alert/null_alerter.py +0 -0
  60. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/monitoring/alert/webhook_alerter.py +0 -0
  61. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/monitoring/monitoring.py +0 -0
  62. {vectorwave-0.3.0/src/vectorwave/search → vectorwave-1.0.1/src/vectorwave/prediction}/__init__.py +0 -0
  63. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/prediction/predictor.py +0 -0
  64. {vectorwave-0.3.0/src/vectorwave/utils → vectorwave-1.0.1/src/vectorwave/search}/__init__.py +0 -0
  65. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/search/execution_search.py +0 -0
  66. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/search/extended_search.py +0 -0
  67. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/search/rag_search.py +0 -0
  68. {vectorwave-0.3.0/src/vectorwave/vectorizer → vectorwave-1.0.1/src/vectorwave/utils}/__init__.py +0 -0
  69. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/context.py +0 -0
  70. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/github_pr.py +0 -0
  71. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/path_utils.py +0 -0
  72. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/replayer_semantic.py +0 -0
  73. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/scheduler.py +0 -0
  74. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/serialization.py +0 -0
  75. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/utils/status.py +0 -0
  76. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/vectorizer/base.py +0 -0
  77. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/vectorizer/factory.py +0 -0
  78. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/vectorizer/huggingface_vectorizer.py +0 -0
  79. {vectorwave-0.3.0 → vectorwave-1.0.1}/src/vectorwave/vectorizer/openai_vectorizer.py +0 -0
@@ -0,0 +1,360 @@
1
+ Metadata-Version: 2.4
2
+ Name: vectorwave
3
+ Version: 1.0.1
4
+ Classifier: Programming Language :: Python :: 3
5
+ Classifier: Programming Language :: Python :: 3.10
6
+ Classifier: Programming Language :: Python :: 3.11
7
+ Classifier: Programming Language :: Python :: 3.12
8
+ Classifier: Programming Language :: Python :: 3.13
9
+ Classifier: Programming Language :: Rust
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Development Status :: 5 - Production/Stable
12
+ Classifier: Intended Audience :: Developers
13
+ Requires-Dist: weaviate-client>=4.0.0
14
+ Requires-Dist: pydantic-settings>=2.0.0
15
+ Requires-Dist: sentence-transformers
16
+ Requires-Dist: requests
17
+ Requires-Dist: openai
18
+ Requires-Dist: pygithub
19
+ Requires-Dist: schedule
20
+ Requires-Dist: tomli>=2.0 ; python_full_version < '3.11'
21
+ Requires-Dist: pytest ; extra == 'dev'
22
+ Requires-Dist: pytest-asyncio ; extra == 'dev'
23
+ Requires-Dist: pytest-benchmark ; extra == 'dev'
24
+ Requires-Dist: pytest-recording ; extra == 'dev'
25
+ Requires-Dist: flake8 ; extra == 'dev'
26
+ Requires-Dist: lancedb>=0.30.0 ; extra == 'dev'
27
+ Requires-Dist: opentelemetry-api>=1.20.0 ; extra == 'dev'
28
+ Requires-Dist: opentelemetry-sdk>=1.20.0 ; extra == 'dev'
29
+ Requires-Dist: opentelemetry-exporter-otlp>=1.20.0 ; extra == 'dev'
30
+ Requires-Dist: testcontainers>=4.0.0 ; extra == 'dev'
31
+ Requires-Dist: vcrpy>=6.0.0 ; extra == 'dev'
32
+ Requires-Dist: lancedb>=0.30.0 ; extra == 'lite'
33
+ Requires-Dist: opentelemetry-api>=1.20.0 ; extra == 'otel'
34
+ Requires-Dist: opentelemetry-sdk>=1.20.0 ; extra == 'otel'
35
+ Requires-Dist: opentelemetry-exporter-otlp>=1.20.0 ; extra == 'otel'
36
+ Provides-Extra: dev
37
+ Provides-Extra: lite
38
+ Provides-Extra: otel
39
+ License-File: LICENSE
40
+ License-File: NOTICE
41
+ Summary: VectorWave: Seamless Auto-Vectorization Framework
42
+ Author-email: junyeonggim <junyeonggim5@gmail.com>
43
+ License-Expression: MIT
44
+ Requires-Python: >=3.10
45
+ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
46
+ Project-URL: Repository, https://github.com/cozymori/vectorwave
47
+
48
+
49
+ Need more information? Visit [here](https://www.cozymori.net/vectorwave)
50
+
51
+ # VectorWave
52
+ ![PyPI](https://badgen.net/pypi/v/vectorwave)
53
+ ![Python](https://badgen.net/pypi/python/vectorwave)
54
+ ![License](https://badgen.net/pypi/license/vectorwave)
55
+ ![CI](https://badgen.net/github/checks/cozymori/vectorwave)
56
+
57
+ **Semantic caching and golden-data regression testing for your LLM functions — from one decorator.**
58
+
59
+ `@vectorize` captures every call your function makes — inputs, outputs, vectors, timing. That single execution history powers two things teams usually build and wire up separately: a **semantic cache** that skips re-running work you've already done, and a **pytest regression oracle** that catches the drift `assert a == b` can't. Drift detection and an experimental auto-diagnosis step fall out of the same history — but caching and testing are the core.
60
+
61
+ ```bash
62
+ pip install vectorwave # Pro mode (Weaviate)
63
+ pip install "vectorwave[lite]" # Lite mode (LanceDB, no Docker)
64
+ pip install "vectorwave[otel]" # + OpenTelemetry mirror
65
+ ```
66
+
67
+ ### Requirements
68
+
69
+ - **Python**: 3.10 – 3.13
70
+ - **Docker** (Pro mode only): runs the Weaviate database. Skip with Lite mode.
71
+ - **OpenAI API Key** (optional): for AI auto-documentation and high-performance embeddings.
72
+
73
+ ### How to reach us
74
+
75
+ - **GitHub Issues**: [https://github.com/cozymori/vectorwave/issues](https://github.com/cozymori/vectorwave/issues)
76
+ - **Contributors**: [github.com/Cozymori/VectorWave/graphs/contributors](https://github.com/Cozymori/VectorWave/graphs/contributors)
77
+ - **Contributing guide**: [Contributing.md](./Contributing.md)
78
+
79
+ ---
80
+
81
+ ## 🚀 What is VectorWave?
82
+
83
+ VectorWave solves two problems that LLM-backed Python code keeps running into — with **one mechanism**:
84
+
85
+ 1. **The same prompt is processed twice.** Direct LLM calls are expensive; the same semantic input usually has the same useful output. → **Semantic caching** with cosine-similarity lookup.
86
+ 2. **You can't `assert a == b` on an LLM.** Output drifts a little every run. A regression that drops a clause looks identical to one that swaps a word. → **Pytest plugin** that compares to golden data by similarity, exact match, or LLM judge.
87
+
88
+ Both come from the same source: `@vectorize` records every call, and that **golden execution history** is reused as the cache *and* as the test oracle. Two more capabilities fall out of the same history — **semantic drift detection** and an **experimental automatic error diagnosis** step (opt-in; see below) — but caching and testing are the load-bearing core.
89
+
90
+ ---
91
+
92
+ ## 😊 Quick Start
93
+
94
+ ### 1. Install
95
+
96
+ ```bash
97
+ pip install vectorwave # Pro mode (default; requires Weaviate)
98
+ # or
99
+ pip install "vectorwave[lite]" # Lite mode — embedded LanceDB, no Docker
100
+ ```
101
+
102
+ For Pro mode (Weaviate), bring your own instance or use the bundled dev stack:
103
+
104
+ ```bash
105
+ vectorwave dev start # starts Weaviate + console on localhost
106
+ ```
107
+
108
+ For Lite mode, no setup beyond `pip install`:
109
+
110
+ ```bash
111
+ export VECTORWAVE_MODE=lite
112
+ ```
113
+
114
+ ### 2. Two demos in one decorator
115
+
116
+ #### A. Semantic caching
117
+
118
+ ```python
119
+ import time
120
+ from vectorwave import vectorize, initialize_database
121
+
122
+ initialize_database() # Pro mode only; in Lite mode (VECTORWAVE_MODE=lite) skip this
123
+
124
+ @vectorize(semantic_cache=True, cache_threshold=0.95, capture_return_value=True)
125
+ def expensive_llm_task(query: str):
126
+ time.sleep(2)
127
+ return f"Processed result for: {query}"
128
+
129
+ # First call: cache miss → ~2.0s
130
+ print(expensive_llm_task("How do I fix a Python bug?"))
131
+
132
+ # Second call — different words, same meaning: cache hit → ~0.02s
133
+ print(expensive_llm_task("What's the best way to fix a bug in Python?"))
134
+ ```
135
+
136
+ #### B. Pytest regression test
137
+
138
+ After your function has built up some golden executions, drop this in any test file:
139
+
140
+ ```python
141
+ import pytest
142
+
143
+ @pytest.mark.vectorwave(
144
+ target="myapp.expensive_llm_task",
145
+ strategy="similarity",
146
+ threshold=0.85,
147
+ limit=10,
148
+ )
149
+ def test_no_regression():
150
+ pass
151
+ ```
152
+
153
+ `pytest` re-runs the function against captured inputs and fails the test if the new output drifts below the threshold. There is also a `vw_replay` fixture for programmatic inspection. Configuration layers from marker kwargs → `[tool.vectorwave.check."<target>"]` in `pyproject.toml` → global defaults.
154
+
155
+ To pick a threshold instead of guessing:
156
+
157
+ ```bash
158
+ vectorwave check calibrate myapp.expensive_llm_task
159
+ # Reports p5/p10/p25/p50/p75/p95 of pairwise similarity and a recommended threshold.
160
+ # Use --rerun to measure the honest noise floor by re-executing the function.
161
+ ```
162
+
163
+ > **Experimental — Automatic error diagnosis:** VectorWave can also read a function's runtime errors from its execution history and open a GitHub PR with a suggested patch (opt-in, Pro-only, needs an LLM + GitHub token). It's deliberately not a headline guarantee — see **Automatic Error Diagnosis & Patch PRs** under Key Features below.
164
+
165
+ ---
166
+
167
+ ## ⭐ Key Features
168
+
169
+ ### ⚡ Semantic Caching
170
+
171
+ Cache LLM calls by meaning, not by exact string. Powered by HNSW vector indexes in Weaviate (Pro) or LanceDB (Lite). Threshold-based hit decisions per function.
172
+
173
+ - **Latency**: seconds → milliseconds.
174
+ - **Cost**: up to 90% fewer LLM tokens.
175
+
176
+ ![VectorWave Semantic Caching Architecture](./docs_kr/images/detail_arch.png)
177
+
178
+ ### 🧪 Pytest Regression Testing — *new in 1.0*
179
+
180
+ One marker, one threshold. Reuses your golden execution history as the test oracle.
181
+
182
+ ```python
183
+ @pytest.mark.vectorwave(target="myapp.fn", strategy="similarity", threshold=0.85)
184
+ def test_fn_regression():
185
+ pass
186
+ ```
187
+
188
+ Three strategies: `exact`, `similarity`, `llm` (LLM-as-a-judge). See [ADR-0002](./docs/adr/0002-pytest-plugin-design.md) for the design rationale.
189
+
190
+ ### 🎯 Threshold Calibration — *new in 1.0*
191
+
192
+ ```bash
193
+ vectorwave check calibrate myapp.summarize # cheap: output diversity
194
+ vectorwave check calibrate myapp.summarize --rerun # honest: noise floor
195
+ ```
196
+
197
+ Outputs a percentile distribution + a ready-to-paste `[tool.vectorwave.check."<target>"]` snippet. Recommends `strategy="exact"` for deterministic targets and `strategy="llm"` for highly variable ones.
198
+
199
+ ### 💾 Lite Mode — *new in 1.0*
200
+
201
+ Embedded LanceDB backend. No Docker, no ports, single on-disk directory.
202
+
203
+ ```bash
204
+ pip install "vectorwave[lite]"
205
+ export VECTORWAVE_MODE=lite
206
+ ```
207
+
208
+ Trade-off documented in [ADR-0001](./docs/adr/0001-vectorstore-abstraction.md): Lite mode gives up server-side vectorization and distributed batching in exchange for zero setup.
209
+
210
+ ### 📡 OpenTelemetry Mirror — *new in 1.0*
211
+
212
+ VW spans appear in your existing OTel stack (Datadog, Honeycomb, Jaeger, Tempo) alongside the rest of your service traces.
213
+
214
+ ```bash
215
+ pip install "vectorwave[otel]"
216
+ export OTEL_SERVICE_NAME=myapp
217
+ ```
218
+
219
+ Mirror, not replacement — VW's own pipeline keeps full semantic context. See [ADR-0003](./docs/adr/0003-opentelemetry-mirror.md).
220
+
221
+ ### 📊 Semantic Drift Radar
222
+
223
+ Detect when your users start asking things your model wasn't trained for.
224
+
225
+ - **Anomaly Detection**: distance between new queries and the Golden Dataset.
226
+ - **Alerting**: Discord / webhook notifications past the drift threshold (default 0.25).
227
+
228
+ ![VectorWave Drift Architecture](./docs_kr/images/semantic_drift.png)
229
+
230
+ ### 🧪 Automatic Error Diagnosis & Patch PRs — *experimental*
231
+
232
+ Opt-in and Pro-only. When a `@vectorize`d function raises, VectorWave can read the error from its execution history, ask an LLM for a fix, and open a **GitHub Pull Request** with the suggested patch — for you to review, never auto-merged.
233
+
234
+ - **RAG over your execution history** for root-cause context.
235
+ - **You stay in the loop**: it opens a PR, you approve. A cooldown prevents PR spam for the same error.
236
+ - **Requirements**: `OPENAI_API_KEY`, `GITHUB_TOKEN` + `GITHUB_REPO_NAME`, and Pro mode (Weaviate).
237
+
238
+ This is the most experimental part of VectorWave and is intentionally *not* one of its headline guarantees. Rationale and internals in [healer.py](./src/vectorwave/utils/healer.py).
239
+
240
+ ![VectorWave Healer Architecture](./docs_kr/images/self_healing.png)
241
+
242
+ ---
243
+
244
+ ## 🏗 Architecture
245
+
246
+ VectorWave sits as a transparent layer between your application and the LLM / infrastructure. Storage is pluggable behind a `VectorStore` protocol — Pro (Weaviate) and Lite (LanceDB) backends share every other component.
247
+
248
+ ![VectorWave Architecture](./docs_kr/images/module_arch.png)
249
+
250
+ ### Core Components
251
+
252
+ - **Optimization Engine**: intercepts function calls, checks semantic cache, returns a hit if one exists within threshold.
253
+ - **Trace Context Manager**: collects execution logs, inputs, outputs, vectors — without modifying your code structure.
254
+ - **VectorStore Layer**: backend-neutral storage protocol. New backends are a single-file addition.
255
+ - **Auto-Diagnosis Pipeline** *(experimental)*: on errors, diagnoses from history and opens a patch PR for review.
256
+ - **Check Plugin** (`vectorwave.check`): pytest entry-point + calibration CLI.
257
+
258
+ ---
259
+
260
+ ## ⏱ Performance
261
+
262
+ ### Caching gains (Pro mode, real LLM call)
263
+
264
+ | Metric | Direct execution | With VectorWave | Improvement |
265
+ |---|---|---|---|
266
+ | **Latency (cache hit)** | ~2.5 s (LLM API) | **~0.02 s** | **125× faster** |
267
+ | **Cost (cache hit)** | $0.03 / call | **$0.00** | **100% savings** |
268
+
269
+ ### Wrapper overhead (`@vectorize` on a bare Python function)
270
+
271
+ Median time per call, measured via `pytest src/tests/benchmarks/ --benchmark-only` on Darwin / CPython 3.12 / Apple Silicon:
272
+
273
+ | Variant | Median | Overhead vs bare |
274
+ |---|---|---|
275
+ | bare Python function (anchor) | 33.8 ns | 1× |
276
+ | `@vectorize`, no capture | **11.4 µs** | ~338× |
277
+ | `@vectorize` + `capture_return_value` | 13.9 µs | ~411× |
278
+ | `@vectorize` + `capture_inputs` | 14.8 µs | ~439× |
279
+
280
+ For any function doing >1 ms of real work, the wrapper tax is in the noise. The 33.8 ns baseline is an empty function — the "× vs bare" numbers look dramatic until you remember that.
281
+
282
+ ### Embeddings: how many, and how to keep them cheap
283
+
284
+ VectorWave embeds **input text**, and does so sparingly — one small vector per stored call:
285
+
286
+ | Per `@vectorize` call | Embeddings |
287
+ |---|---|
288
+ | cache hit (`semantic_cache=True`) | **1** — the lookup vector is reused for the hit log |
289
+ | cache miss (`semantic_cache=True`) | **2** — one to look up, one to store the new execution |
290
+ | logging only (no `semantic_cache`) | **1** — the stored execution vector |
291
+ | a call that errors | **1** — the error message (used for error search) |
292
+
293
+ Function **registration** embeds a one-line description once per function and caches it (`.vectorwave_functions_cache.json`) — it does *not* re-embed on every call. Drift detection reuses the stored vector, so it adds **zero** extra embeddings.
294
+
295
+ **Keeping it cheap:**
296
+
297
+ - `VECTORIZER=huggingface` (default) — embeddings run locally on CPU with `all-MiniLM-L6-v2` (384-dim). No API calls, no per-call cost, nothing leaves the process.
298
+ - `VECTORIZER=weaviate_module` — Weaviate embeds server-side (one round trip, no Python-side model).
299
+ - `VECTORIZER=openai_client` — each embedding is an OpenAI API call (highest quality, but real cost + latency per call). Reserve it for when you need it.
300
+ - `VECTORIZER=none` — no embeddings at all (disables semantic caching and drift).
301
+
302
+ Only calibration batches embeddings (`embed_batch`); the per-call path embeds one text at a time, and the batch manager batches **DB writes**, not embeddings.
303
+
304
+ ---
305
+
306
+ ## 🆚 How does VectorWave compare?
307
+
308
+ | Feature | GPTCache | ragas / deepeval / promptfoo | **VectorWave** |
309
+ | :--- | :---: | :---: | :---: |
310
+ | Semantic caching (execution-level) | O | X | **O** |
311
+ | Golden-data regression testing | X | O | **O** |
312
+ | Pytest integration | X | △ | **O (marker + fixture)** |
313
+ | Threshold calibration (CLI) | X | △ | **O (percentile-backed)** |
314
+ | Auto error diagnosis & patch PR *(experimental)* | X | X | **O** |
315
+ | OpenTelemetry mirror | X | X | **O** |
316
+ | Drift detection | X | △ | **O** |
317
+ | Zero-config local mode | △ | △ | **O (Lite mode)** |
318
+
319
+ Most teams pick a caching tool *and* a regression-testing tool *and* an observability integration and wire them up themselves. VectorWave gives you one decorator and one config file.
320
+
321
+ ---
322
+
323
+ ## 🧠 How it works
324
+
325
+ Unlike traditional Key-Value caching (e.g., Redis), VectorWave understands **Context**.
326
+
327
+ 1. **Vectorization**: function arguments → high-dimensional vectors via OpenAI / HuggingFace.
328
+ 2. **Search**: Approximate Nearest Neighbor (ANN) lookup in the vector store.
329
+ 3. **Decision**:
330
+ - Neighbor within `threshold` → **return cached result**.
331
+ - Otherwise → **execute function** → **async-log to DB** (the new entry becomes future golden data).
332
+
333
+ Golden executions feed the regression-test layer. Drift over time feeds the radar. Errors feed the experimental auto-diagnosis step. **Same substrate — caching and testing at the core, the rest derived.**
334
+
335
+ ---
336
+
337
+ ## 🔒 Data Handling
338
+
339
+ VectorWave stores what your functions do, so it's worth knowing exactly what lands where:
340
+
341
+ - **What's captured**: per call, the function's inputs (with `capture_inputs`/`replay`), return value (with `capture_return_value`), a vector embedding of the input text, and timing/status metadata. The function's **source code** is also stored (used by regression replay and the experimental auto-diagnosis).
342
+ - **Redaction**: before anything is stored, values whose keys match `sensitive_keys` are replaced with `[MASKED]` (defaults: `password`, `api_key`, `token`, `secret`, `auth_token`). Add your own PII/secret keys via settings — masking runs in the `mask_and_serialize` path on the way in.
343
+ - **Where it lives**: **Lite mode** keeps everything in a local on-disk directory (`.vectorwave/`) — nothing leaves your machine. **Pro mode** writes to the Weaviate instance *you* run.
344
+ - **Embeddings & third parties**: with `VECTORIZER=huggingface`, embeddings are computed **locally**; with `openai_client` or Weaviate's `text2vec-openai`, the input text is sent to OpenAI to be embedded. The experimental auto-diagnosis also sends source + error context to an LLM.
345
+ - **Retention**: 1.0 has no automatic TTL — you own the store's lifecycle. Drop collections to delete history; back up the Weaviate volume (Pro) or the `.vectorwave/` directory (Lite) as you would any datastore.
346
+
347
+ Treat the execution store like application logs that may contain user data: scope `sensitive_keys` to your fields, and prefer Lite mode with local embeddings when inputs are sensitive.
348
+
349
+ ---
350
+
351
+ ## 📚 Further reading
352
+
353
+ - [CHANGELOG.md](./CHANGELOG.md) — release history.
354
+ - [docs/adr/](./docs/adr/) — architectural decision records.
355
+ - [Contributing.md](./Contributing.md) — how to set up the dev environment and submit a PR.
356
+
357
+ ## 😍 Contributing
358
+
359
+ We are extremely open to contributions — new vectorizers, better diagnosis prompts, additional backends, doc improvements, typo fixes. Please read the [Contributing guide](./Contributing.md) before opening a PR.
360
+
@@ -0,0 +1,312 @@
1
+
2
+ Need more information? Visit [here](https://www.cozymori.net/vectorwave)
3
+
4
+ # VectorWave
5
+ ![PyPI](https://badgen.net/pypi/v/vectorwave)
6
+ ![Python](https://badgen.net/pypi/python/vectorwave)
7
+ ![License](https://badgen.net/pypi/license/vectorwave)
8
+ ![CI](https://badgen.net/github/checks/cozymori/vectorwave)
9
+
10
+ **Semantic caching and golden-data regression testing for your LLM functions — from one decorator.**
11
+
12
+ `@vectorize` captures every call your function makes — inputs, outputs, vectors, timing. That single execution history powers two things teams usually build and wire up separately: a **semantic cache** that skips re-running work you've already done, and a **pytest regression oracle** that catches the drift `assert a == b` can't. Drift detection and an experimental auto-diagnosis step fall out of the same history — but caching and testing are the core.
13
+
14
+ ```bash
15
+ pip install vectorwave # Pro mode (Weaviate)
16
+ pip install "vectorwave[lite]" # Lite mode (LanceDB, no Docker)
17
+ pip install "vectorwave[otel]" # + OpenTelemetry mirror
18
+ ```
19
+
20
+ ### Requirements
21
+
22
+ - **Python**: 3.10 – 3.13
23
+ - **Docker** (Pro mode only): runs the Weaviate database. Skip with Lite mode.
24
+ - **OpenAI API Key** (optional): for AI auto-documentation and high-performance embeddings.
25
+
26
+ ### How to reach us
27
+
28
+ - **GitHub Issues**: [https://github.com/cozymori/vectorwave/issues](https://github.com/cozymori/vectorwave/issues)
29
+ - **Contributors**: [github.com/Cozymori/VectorWave/graphs/contributors](https://github.com/Cozymori/VectorWave/graphs/contributors)
30
+ - **Contributing guide**: [Contributing.md](./Contributing.md)
31
+
32
+ ---
33
+
34
+ ## 🚀 What is VectorWave?
35
+
36
+ VectorWave solves two problems that LLM-backed Python code keeps running into — with **one mechanism**:
37
+
38
+ 1. **The same prompt is processed twice.** Direct LLM calls are expensive; the same semantic input usually has the same useful output. → **Semantic caching** with cosine-similarity lookup.
39
+ 2. **You can't `assert a == b` on an LLM.** Output drifts a little every run. A regression that drops a clause looks identical to one that swaps a word. → **Pytest plugin** that compares to golden data by similarity, exact match, or LLM judge.
40
+
41
+ Both come from the same source: `@vectorize` records every call, and that **golden execution history** is reused as the cache *and* as the test oracle. Two more capabilities fall out of the same history — **semantic drift detection** and an **experimental automatic error diagnosis** step (opt-in; see below) — but caching and testing are the load-bearing core.
42
+
43
+ ---
44
+
45
+ ## 😊 Quick Start
46
+
47
+ ### 1. Install
48
+
49
+ ```bash
50
+ pip install vectorwave # Pro mode (default; requires Weaviate)
51
+ # or
52
+ pip install "vectorwave[lite]" # Lite mode — embedded LanceDB, no Docker
53
+ ```
54
+
55
+ For Pro mode (Weaviate), bring your own instance or use the bundled dev stack:
56
+
57
+ ```bash
58
+ vectorwave dev start # starts Weaviate + console on localhost
59
+ ```
60
+
61
+ For Lite mode, no setup beyond `pip install`:
62
+
63
+ ```bash
64
+ export VECTORWAVE_MODE=lite
65
+ ```
66
+
67
+ ### 2. Two demos in one decorator
68
+
69
+ #### A. Semantic caching
70
+
71
+ ```python
72
+ import time
73
+ from vectorwave import vectorize, initialize_database
74
+
75
+ initialize_database() # Pro mode only; in Lite mode (VECTORWAVE_MODE=lite) skip this
76
+
77
+ @vectorize(semantic_cache=True, cache_threshold=0.95, capture_return_value=True)
78
+ def expensive_llm_task(query: str):
79
+ time.sleep(2)
80
+ return f"Processed result for: {query}"
81
+
82
+ # First call: cache miss → ~2.0s
83
+ print(expensive_llm_task("How do I fix a Python bug?"))
84
+
85
+ # Second call — different words, same meaning: cache hit → ~0.02s
86
+ print(expensive_llm_task("What's the best way to fix a bug in Python?"))
87
+ ```
88
+
89
+ #### B. Pytest regression test
90
+
91
+ After your function has built up some golden executions, drop this in any test file:
92
+
93
+ ```python
94
+ import pytest
95
+
96
+ @pytest.mark.vectorwave(
97
+ target="myapp.expensive_llm_task",
98
+ strategy="similarity",
99
+ threshold=0.85,
100
+ limit=10,
101
+ )
102
+ def test_no_regression():
103
+ pass
104
+ ```
105
+
106
+ `pytest` re-runs the function against captured inputs and fails the test if the new output drifts below the threshold. There is also a `vw_replay` fixture for programmatic inspection. Configuration layers from marker kwargs → `[tool.vectorwave.check."<target>"]` in `pyproject.toml` → global defaults.
107
+
108
+ To pick a threshold instead of guessing:
109
+
110
+ ```bash
111
+ vectorwave check calibrate myapp.expensive_llm_task
112
+ # Reports p5/p10/p25/p50/p75/p95 of pairwise similarity and a recommended threshold.
113
+ # Use --rerun to measure the honest noise floor by re-executing the function.
114
+ ```
115
+
116
+ > **Experimental — Automatic error diagnosis:** VectorWave can also read a function's runtime errors from its execution history and open a GitHub PR with a suggested patch (opt-in, Pro-only, needs an LLM + GitHub token). It's deliberately not a headline guarantee — see **Automatic Error Diagnosis & Patch PRs** under Key Features below.
117
+
118
+ ---
119
+
120
+ ## ⭐ Key Features
121
+
122
+ ### ⚡ Semantic Caching
123
+
124
+ Cache LLM calls by meaning, not by exact string. Powered by HNSW vector indexes in Weaviate (Pro) or LanceDB (Lite). Threshold-based hit decisions per function.
125
+
126
+ - **Latency**: seconds → milliseconds.
127
+ - **Cost**: up to 90% fewer LLM tokens.
128
+
129
+ ![VectorWave Semantic Caching Architecture](./docs_kr/images/detail_arch.png)
130
+
131
+ ### 🧪 Pytest Regression Testing — *new in 1.0*
132
+
133
+ One marker, one threshold. Reuses your golden execution history as the test oracle.
134
+
135
+ ```python
136
+ @pytest.mark.vectorwave(target="myapp.fn", strategy="similarity", threshold=0.85)
137
+ def test_fn_regression():
138
+ pass
139
+ ```
140
+
141
+ Three strategies: `exact`, `similarity`, `llm` (LLM-as-a-judge). See [ADR-0002](./docs/adr/0002-pytest-plugin-design.md) for the design rationale.
142
+
143
+ ### 🎯 Threshold Calibration — *new in 1.0*
144
+
145
+ ```bash
146
+ vectorwave check calibrate myapp.summarize # cheap: output diversity
147
+ vectorwave check calibrate myapp.summarize --rerun # honest: noise floor
148
+ ```
149
+
150
+ Outputs a percentile distribution + a ready-to-paste `[tool.vectorwave.check."<target>"]` snippet. Recommends `strategy="exact"` for deterministic targets and `strategy="llm"` for highly variable ones.
151
+
152
+ ### 💾 Lite Mode — *new in 1.0*
153
+
154
+ Embedded LanceDB backend. No Docker, no ports, single on-disk directory.
155
+
156
+ ```bash
157
+ pip install "vectorwave[lite]"
158
+ export VECTORWAVE_MODE=lite
159
+ ```
160
+
161
+ Trade-off documented in [ADR-0001](./docs/adr/0001-vectorstore-abstraction.md): Lite mode gives up server-side vectorization and distributed batching in exchange for zero setup.
162
+
163
+ ### 📡 OpenTelemetry Mirror — *new in 1.0*
164
+
165
+ VW spans appear in your existing OTel stack (Datadog, Honeycomb, Jaeger, Tempo) alongside the rest of your service traces.
166
+
167
+ ```bash
168
+ pip install "vectorwave[otel]"
169
+ export OTEL_SERVICE_NAME=myapp
170
+ ```
171
+
172
+ Mirror, not replacement — VW's own pipeline keeps full semantic context. See [ADR-0003](./docs/adr/0003-opentelemetry-mirror.md).
173
+
174
+ ### 📊 Semantic Drift Radar
175
+
176
+ Detect when your users start asking things your model wasn't trained for.
177
+
178
+ - **Anomaly Detection**: distance between new queries and the Golden Dataset.
179
+ - **Alerting**: Discord / webhook notifications past the drift threshold (default 0.25).
180
+
181
+ ![VectorWave Drift Architecture](./docs_kr/images/semantic_drift.png)
182
+
183
+ ### 🧪 Automatic Error Diagnosis & Patch PRs — *experimental*
184
+
185
+ Opt-in and Pro-only. When a `@vectorize`d function raises, VectorWave can read the error from its execution history, ask an LLM for a fix, and open a **GitHub Pull Request** with the suggested patch — for you to review, never auto-merged.
186
+
187
+ - **RAG over your execution history** for root-cause context.
188
+ - **You stay in the loop**: it opens a PR, you approve. A cooldown prevents PR spam for the same error.
189
+ - **Requirements**: `OPENAI_API_KEY`, `GITHUB_TOKEN` + `GITHUB_REPO_NAME`, and Pro mode (Weaviate).
190
+
191
+ This is the most experimental part of VectorWave and is intentionally *not* one of its headline guarantees. Rationale and internals in [healer.py](./src/vectorwave/utils/healer.py).
192
+
193
+ ![VectorWave Healer Architecture](./docs_kr/images/self_healing.png)
194
+
195
+ ---
196
+
197
+ ## 🏗 Architecture
198
+
199
+ VectorWave sits as a transparent layer between your application and the LLM / infrastructure. Storage is pluggable behind a `VectorStore` protocol — Pro (Weaviate) and Lite (LanceDB) backends share every other component.
200
+
201
+ ![VectorWave Architecture](./docs_kr/images/module_arch.png)
202
+
203
+ ### Core Components
204
+
205
+ - **Optimization Engine**: intercepts function calls, checks semantic cache, returns a hit if one exists within threshold.
206
+ - **Trace Context Manager**: collects execution logs, inputs, outputs, vectors — without modifying your code structure.
207
+ - **VectorStore Layer**: backend-neutral storage protocol. New backends are a single-file addition.
208
+ - **Auto-Diagnosis Pipeline** *(experimental)*: on errors, diagnoses from history and opens a patch PR for review.
209
+ - **Check Plugin** (`vectorwave.check`): pytest entry-point + calibration CLI.
210
+
211
+ ---
212
+
213
+ ## ⏱ Performance
214
+
215
+ ### Caching gains (Pro mode, real LLM call)
216
+
217
+ | Metric | Direct execution | With VectorWave | Improvement |
218
+ |---|---|---|---|
219
+ | **Latency (cache hit)** | ~2.5 s (LLM API) | **~0.02 s** | **125× faster** |
220
+ | **Cost (cache hit)** | $0.03 / call | **$0.00** | **100% savings** |
221
+
222
+ ### Wrapper overhead (`@vectorize` on a bare Python function)
223
+
224
+ Median time per call, measured via `pytest src/tests/benchmarks/ --benchmark-only` on Darwin / CPython 3.12 / Apple Silicon:
225
+
226
+ | Variant | Median | Overhead vs bare |
227
+ |---|---|---|
228
+ | bare Python function (anchor) | 33.8 ns | 1× |
229
+ | `@vectorize`, no capture | **11.4 µs** | ~338× |
230
+ | `@vectorize` + `capture_return_value` | 13.9 µs | ~411× |
231
+ | `@vectorize` + `capture_inputs` | 14.8 µs | ~439× |
232
+
233
+ For any function doing >1 ms of real work, the wrapper tax is in the noise. The 33.8 ns baseline is an empty function — the "× vs bare" numbers look dramatic until you remember that.
234
+
235
+ ### Embeddings: how many, and how to keep them cheap
236
+
237
+ VectorWave embeds **input text**, and does so sparingly — one small vector per stored call:
238
+
239
+ | Per `@vectorize` call | Embeddings |
240
+ |---|---|
241
+ | cache hit (`semantic_cache=True`) | **1** — the lookup vector is reused for the hit log |
242
+ | cache miss (`semantic_cache=True`) | **2** — one to look up, one to store the new execution |
243
+ | logging only (no `semantic_cache`) | **1** — the stored execution vector |
244
+ | a call that errors | **1** — the error message (used for error search) |
245
+
246
+ Function **registration** embeds a one-line description once per function and caches it (`.vectorwave_functions_cache.json`) — it does *not* re-embed on every call. Drift detection reuses the stored vector, so it adds **zero** extra embeddings.
247
+
248
+ **Keeping it cheap:**
249
+
250
+ - `VECTORIZER=huggingface` (default) — embeddings run locally on CPU with `all-MiniLM-L6-v2` (384-dim). No API calls, no per-call cost, nothing leaves the process.
251
+ - `VECTORIZER=weaviate_module` — Weaviate embeds server-side (one round trip, no Python-side model).
252
+ - `VECTORIZER=openai_client` — each embedding is an OpenAI API call (highest quality, but real cost + latency per call). Reserve it for when you need it.
253
+ - `VECTORIZER=none` — no embeddings at all (disables semantic caching and drift).
254
+
255
+ Only calibration batches embeddings (`embed_batch`); the per-call path embeds one text at a time, and the batch manager batches **DB writes**, not embeddings.
256
+
257
+ ---
258
+
259
+ ## 🆚 How does VectorWave compare?
260
+
261
+ | Feature | GPTCache | ragas / deepeval / promptfoo | **VectorWave** |
262
+ | :--- | :---: | :---: | :---: |
263
+ | Semantic caching (execution-level) | O | X | **O** |
264
+ | Golden-data regression testing | X | O | **O** |
265
+ | Pytest integration | X | △ | **O (marker + fixture)** |
266
+ | Threshold calibration (CLI) | X | △ | **O (percentile-backed)** |
267
+ | Auto error diagnosis & patch PR *(experimental)* | X | X | **O** |
268
+ | OpenTelemetry mirror | X | X | **O** |
269
+ | Drift detection | X | △ | **O** |
270
+ | Zero-config local mode | △ | △ | **O (Lite mode)** |
271
+
272
+ Most teams pick a caching tool *and* a regression-testing tool *and* an observability integration and wire them up themselves. VectorWave gives you one decorator and one config file.
273
+
274
+ ---
275
+
276
+ ## 🧠 How it works
277
+
278
+ Unlike traditional Key-Value caching (e.g., Redis), VectorWave understands **Context**.
279
+
280
+ 1. **Vectorization**: function arguments → high-dimensional vectors via OpenAI / HuggingFace.
281
+ 2. **Search**: Approximate Nearest Neighbor (ANN) lookup in the vector store.
282
+ 3. **Decision**:
283
+ - Neighbor within `threshold` → **return cached result**.
284
+ - Otherwise → **execute function** → **async-log to DB** (the new entry becomes future golden data).
285
+
286
+ Golden executions feed the regression-test layer. Drift over time feeds the radar. Errors feed the experimental auto-diagnosis step. **Same substrate — caching and testing at the core, the rest derived.**
287
+
288
+ ---
289
+
290
+ ## 🔒 Data Handling
291
+
292
+ VectorWave stores what your functions do, so it's worth knowing exactly what lands where:
293
+
294
+ - **What's captured**: per call, the function's inputs (with `capture_inputs`/`replay`), return value (with `capture_return_value`), a vector embedding of the input text, and timing/status metadata. The function's **source code** is also stored (used by regression replay and the experimental auto-diagnosis).
295
+ - **Redaction**: before anything is stored, values whose keys match `sensitive_keys` are replaced with `[MASKED]` (defaults: `password`, `api_key`, `token`, `secret`, `auth_token`). Add your own PII/secret keys via settings — masking runs in the `mask_and_serialize` path on the way in.
296
+ - **Where it lives**: **Lite mode** keeps everything in a local on-disk directory (`.vectorwave/`) — nothing leaves your machine. **Pro mode** writes to the Weaviate instance *you* run.
297
+ - **Embeddings & third parties**: with `VECTORIZER=huggingface`, embeddings are computed **locally**; with `openai_client` or Weaviate's `text2vec-openai`, the input text is sent to OpenAI to be embedded. The experimental auto-diagnosis also sends source + error context to an LLM.
298
+ - **Retention**: 1.0 has no automatic TTL — you own the store's lifecycle. Drop collections to delete history; back up the Weaviate volume (Pro) or the `.vectorwave/` directory (Lite) as you would any datastore.
299
+
300
+ Treat the execution store like application logs that may contain user data: scope `sensitive_keys` to your fields, and prefer Lite mode with local embeddings when inputs are sensitive.
301
+
302
+ ---
303
+
304
+ ## 📚 Further reading
305
+
306
+ - [CHANGELOG.md](./CHANGELOG.md) — release history.
307
+ - [docs/adr/](./docs/adr/) — architectural decision records.
308
+ - [Contributing.md](./Contributing.md) — how to set up the dev environment and submit a PR.
309
+
310
+ ## 😍 Contributing
311
+
312
+ We are extremely open to contributions — new vectorizers, better diagnosis prompts, additional backends, doc improvements, typo fixes. Please read the [Contributing guide](./Contributing.md) before opening a PR.