vicinity 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {vicinity-0.2.0 → vicinity-0.2.1}/Makefile +1 -1
  2. {vicinity-0.2.0 → vicinity-0.2.1}/PKG-INFO +33 -22
  3. {vicinity-0.2.0 → vicinity-0.2.1}/README.md +30 -21
  4. {vicinity-0.2.0 → vicinity-0.2.1}/pyproject.toml +1 -0
  5. {vicinity-0.2.0 → vicinity-0.2.1}/tests/conftest.py +2 -0
  6. {vicinity-0.2.0 → vicinity-0.2.1}/tests/test_vicinity.py +36 -9
  7. {vicinity-0.2.0 → vicinity-0.2.1}/uv.lock +234 -42
  8. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/backends/__init__.py +8 -1
  9. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/backends/annoy.py +2 -2
  10. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/backends/faiss.py +2 -15
  11. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/backends/hnsw.py +1 -2
  12. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/backends/pynndescent.py +4 -4
  13. vicinity-0.2.1/vicinity/backends/usearch.py +132 -0
  14. vicinity-0.2.1/vicinity/datatypes.py +24 -0
  15. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/utils.py +3 -1
  16. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/version.py +1 -1
  17. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/vicinity.py +5 -4
  18. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity.egg-info/PKG-INFO +33 -22
  19. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity.egg-info/SOURCES.txt +2 -1
  20. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity.egg-info/requires.txt +3 -0
  21. vicinity-0.2.0/vicinity/datatypes.py +0 -23
  22. {vicinity-0.2.0 → vicinity-0.2.1}/.github/workflows/ci.yaml +0 -0
  23. {vicinity-0.2.0 → vicinity-0.2.1}/.gitignore +0 -0
  24. {vicinity-0.2.0 → vicinity-0.2.1}/.pre-commit-config.yaml +0 -0
  25. {vicinity-0.2.0 → vicinity-0.2.1}/LICENSE +0 -0
  26. {vicinity-0.2.0 → vicinity-0.2.1}/setup.cfg +0 -0
  27. {vicinity-0.2.0 → vicinity-0.2.1}/tests/test_utils.py +0 -0
  28. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/__init__.py +0 -0
  29. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/backends/base.py +0 -0
  30. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/backends/basic.py +0 -0
  31. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity/py.typed +0 -0
  32. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity.egg-info/dependency_links.txt +0 -0
  33. {vicinity-0.2.0 → vicinity-0.2.1}/vicinity.egg-info/top_level.txt +0 -0
@@ -9,7 +9,7 @@ install: venv
9
9
  uv run pre-commit install
10
10
 
11
11
  install-no-pre-commit:
12
- uv pip install ".[dev,hnsw,pynndescent,annoy,faiss]"
12
+ uv pip install ".[dev,hnsw,pynndescent,annoy,faiss,usearch]"
13
13
 
14
14
  install-base:
15
15
  uv sync --extra dev
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: vicinity
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Lightweight Nearest Neighbors with Flexible Backends
5
5
  Author-email: Stéphan Tulkens <stephantul@gmail.com>, Thomas van Dongen <thomas123@live.nl>
6
6
  License: MIT License
@@ -65,6 +65,8 @@ Provides-Extra: annoy
65
65
  Requires-Dist: annoy; extra == "annoy"
66
66
  Provides-Extra: faiss
67
67
  Requires-Dist: faiss-cpu; extra == "faiss"
68
+ Provides-Extra: usearch
69
+ Requires-Dist: usearch; extra == "usearch"
68
70
 
69
71
  <div align="center">
70
72
 
@@ -105,6 +107,7 @@ Install the package with:
105
107
  pip install vicinity
106
108
  ```
107
109
 
110
+
108
111
  The following code snippet demonstrates how to use Vicinity for nearest neighbor search:
109
112
  ```python
110
113
  import numpy as np
@@ -136,36 +139,44 @@ vicinity = Vicinity.load('my_vector_store')
136
139
  Vicinity provides the following features:
137
140
  - Lightweight: Minimal dependencies and fast performance.
138
141
  - Flexible Backend Support: Use different backends for vector storage and search.
139
- - Dynamic Updates: Insert and delete items in the vector store.
140
142
  - Serialization: Save and load vector stores for persistence.
141
143
  - Easy to Use: Simple and intuitive API.
142
144
 
143
145
  ## Supported Backends
144
146
  The following backends are supported:
145
147
  - `BASIC`: A simple flat index for vector storage and search.
146
- - `HNSW`: Hierarchical Navigable Small World Graph for approximate nearest neighbor search.
147
- - `FAISS`: All FAISS indexes for approximate nearest neighbor search are supported.
148
- - `ANNOY`: "Approximate Nearest Neighbors Oh Yeah" for approximate nearest neighbor search.
149
- - `PYNNDescent`: Approximate nearest neighbor search using PyNNDescent.
148
+ - [HNSW](https://github.com/nmslib/hnswlib): Hierarchical Navigable Small World Graph (HNSW) for ANN search using hnswlib.
149
+ - [FAISS](https://github.com/facebookresearch/faiss): ANN search using FAISS. All FAISS indexes are supported.
150
+ - [ANNOY](https://github.com/spotify/annoy): "Approximate Nearest Neighbors Oh Yeah" for approximate nearest neighbor search.
151
+ - [PYNNDescent](https://github.com/lmcinnes/pynndescent): ANN search using PyNNDescent.
152
+ - [USEARCH](https://github.com/unum-cloud/usearch): ANN search using Usearch. This uses a highly optimized version of the HNSW algorithm.
153
+
154
+ NOTE: the ANN backends do not support dynamic deletion. To delete items, you need to recreate the index. Insertion is supported in the following backends: `FAISS`, `HNSW`, and `Usearch`. The `BASIC` backend supports both insertion and deletion.
150
155
 
151
156
  ### Backend Parameters
152
157
 
153
- | Backend | Parameter | Description | Default Value |
154
- |----------------|--------------------|-----------------------------------------------------------------------------------------------|---------------------|
155
- | **Annoy** | `metric` | Similarity metric to use (`dot`, `euclidean`, `cosine`). | `"cosine"` |
156
- | | `trees` | Number of trees to use for indexing. | `100` |
157
- | | `length` | Optional length of the dataset. | `None` |
158
- | **FAISS** | `index_type` | Type of FAISS index (`flat`, `ivf`, `hnsw`, `lsh`, `scalar`, `pq`, `ivf_scalar`, `ivfpq`, `ivfpqr`). | `"hnsw"` |
159
- | | `metric` | Similarity metric to use (`cosine`, `l2`). | `"cosine"` |
160
- | | `nlist` | Number of cells for IVF indexes. | `100` |
161
- | | `m` | Number of subquantizers for PQ and HNSW indexes. | `8` |
162
- | | `nbits` | Number of bits for LSH and PQ indexes. | `8` |
163
- | | `refine_nbits` | Number of bits for the refinement stage in IVFPQR indexes. | `8` |
164
- | **HNSW** | `space` | Similarity space to use (`cosine`, `l2`). | `"cosine"` |
165
- | | `ef_construction` | Size of the dynamic list during index construction. | `200` |
166
- | | `m` | Number of connections per layer. | `16` |
167
- | **PyNNDescent**| `n_neighbors` | Number of neighbors to use for search. | `15` |
168
- | | `metric` | Similarity metric to use (`cosine`, `euclidean`, `manhattan`). | `"cosine"` |
158
+
159
+ | Backend | Parameter | Description | Default Value |
160
+ |-----------------|---------------------|-----------------------------------------------------------------------------------------------|---------------------|
161
+ | **Annoy** | `metric` | Similarity metric to use (`dot`, `euclidean`, `cosine`). | `"cosine"` |
162
+ | | `trees` | Number of trees to use for indexing. | `100` |
163
+ | | `length` | Optional length of the dataset. | `None` |
164
+ | **FAISS** | `metric` | Similarity metric to use (`cosine`, `l2`). | `"cosine"` |
165
+ | | `index_type` | Type of FAISS index (`flat`, `ivf`, `hnsw`, `lsh`, `scalar`, `pq`, `ivf_scalar`, `ivfpq`, `ivfpqr`). | `"hnsw"` |
166
+ | | `nlist` | Number of cells for IVF indexes. | `100` |
167
+ | | `m` | Number of subquantizers for PQ and HNSW indexes. | `8` |
168
+ | | `nbits` | Number of bits for LSH and PQ indexes. | `8` |
169
+ | | `refine_nbits` | Number of bits for the refinement stage in IVFPQR indexes. | `8` |
170
+ | **HNSW** | `metric` | Similarity space to use (`cosine`, `l2`). | `"cosine"` |
171
+ | | `ef_construction` | Size of the dynamic list during index construction. | `200` |
172
+ | | `m` | Number of connections per layer. | `16` |
173
+ | **PyNNDescent** | `metric` | Similarity metric to use (`cosine`, `euclidean`, `manhattan`). | `"cosine"` |
174
+ | | `n_neighbors` | Number of neighbors to use for search. | `15` |
175
+ | **Usearch** | `metric` | Similarity metric to use (`cos`, `ip`, `l2sq`, `hamming`, `tanimoto`). | `"cos"` |
176
+ | | `connectivity` | Number of connections per node in the graph. | `16` |
177
+ | | `expansion_add` | Number of candidates considered during graph construction. | `128` |
178
+ | | `expansion_search` | Number of candidates considered during search. | `64` |
179
+
169
180
 
170
181
 
171
182
  ## Usage
@@ -37,6 +37,7 @@ Install the package with:
37
37
  pip install vicinity
38
38
  ```
39
39
 
40
+
40
41
  The following code snippet demonstrates how to use Vicinity for nearest neighbor search:
41
42
  ```python
42
43
  import numpy as np
@@ -68,36 +69,44 @@ vicinity = Vicinity.load('my_vector_store')
68
69
  Vicinity provides the following features:
69
70
  - Lightweight: Minimal dependencies and fast performance.
70
71
  - Flexible Backend Support: Use different backends for vector storage and search.
71
- - Dynamic Updates: Insert and delete items in the vector store.
72
72
  - Serialization: Save and load vector stores for persistence.
73
73
  - Easy to Use: Simple and intuitive API.
74
74
 
75
75
  ## Supported Backends
76
76
  The following backends are supported:
77
77
  - `BASIC`: A simple flat index for vector storage and search.
78
- - `HNSW`: Hierarchical Navigable Small World Graph for approximate nearest neighbor search.
79
- - `FAISS`: All FAISS indexes for approximate nearest neighbor search are supported.
80
- - `ANNOY`: "Approximate Nearest Neighbors Oh Yeah" for approximate nearest neighbor search.
81
- - `PYNNDescent`: Approximate nearest neighbor search using PyNNDescent.
78
+ - [HNSW](https://github.com/nmslib/hnswlib): Hierarchical Navigable Small World Graph (HNSW) for ANN search using hnswlib.
79
+ - [FAISS](https://github.com/facebookresearch/faiss): ANN search using FAISS. All FAISS indexes are supported.
80
+ - [ANNOY](https://github.com/spotify/annoy): "Approximate Nearest Neighbors Oh Yeah" for approximate nearest neighbor search.
81
+ - [PYNNDescent](https://github.com/lmcinnes/pynndescent): ANN search using PyNNDescent.
82
+ - [USEARCH](https://github.com/unum-cloud/usearch): ANN search using Usearch. This uses a highly optimized version of the HNSW algorithm.
83
+
84
+ NOTE: the ANN backends do not support dynamic deletion. To delete items, you need to recreate the index. Insertion is supported in the following backends: `FAISS`, `HNSW`, and `Usearch`. The `BASIC` backend supports both insertion and deletion.
82
85
 
83
86
  ### Backend Parameters
84
87
 
85
- | Backend | Parameter | Description | Default Value |
86
- |----------------|--------------------|-----------------------------------------------------------------------------------------------|---------------------|
87
- | **Annoy** | `metric` | Similarity metric to use (`dot`, `euclidean`, `cosine`). | `"cosine"` |
88
- | | `trees` | Number of trees to use for indexing. | `100` |
89
- | | `length` | Optional length of the dataset. | `None` |
90
- | **FAISS** | `index_type` | Type of FAISS index (`flat`, `ivf`, `hnsw`, `lsh`, `scalar`, `pq`, `ivf_scalar`, `ivfpq`, `ivfpqr`). | `"hnsw"` |
91
- | | `metric` | Similarity metric to use (`cosine`, `l2`). | `"cosine"` |
92
- | | `nlist` | Number of cells for IVF indexes. | `100` |
93
- | | `m` | Number of subquantizers for PQ and HNSW indexes. | `8` |
94
- | | `nbits` | Number of bits for LSH and PQ indexes. | `8` |
95
- | | `refine_nbits` | Number of bits for the refinement stage in IVFPQR indexes. | `8` |
96
- | **HNSW** | `space` | Similarity space to use (`cosine`, `l2`). | `"cosine"` |
97
- | | `ef_construction` | Size of the dynamic list during index construction. | `200` |
98
- | | `m` | Number of connections per layer. | `16` |
99
- | **PyNNDescent**| `n_neighbors` | Number of neighbors to use for search. | `15` |
100
- | | `metric` | Similarity metric to use (`cosine`, `euclidean`, `manhattan`). | `"cosine"` |
88
+
89
+ | Backend | Parameter | Description | Default Value |
90
+ |-----------------|---------------------|-----------------------------------------------------------------------------------------------|---------------------|
91
+ | **Annoy** | `metric` | Similarity metric to use (`dot`, `euclidean`, `cosine`). | `"cosine"` |
92
+ | | `trees` | Number of trees to use for indexing. | `100` |
93
+ | | `length` | Optional length of the dataset. | `None` |
94
+ | **FAISS** | `metric` | Similarity metric to use (`cosine`, `l2`). | `"cosine"` |
95
+ | | `index_type` | Type of FAISS index (`flat`, `ivf`, `hnsw`, `lsh`, `scalar`, `pq`, `ivf_scalar`, `ivfpq`, `ivfpqr`). | `"hnsw"` |
96
+ | | `nlist` | Number of cells for IVF indexes. | `100` |
97
+ | | `m` | Number of subquantizers for PQ and HNSW indexes. | `8` |
98
+ | | `nbits` | Number of bits for LSH and PQ indexes. | `8` |
99
+ | | `refine_nbits` | Number of bits for the refinement stage in IVFPQR indexes. | `8` |
100
+ | **HNSW** | `metric` | Similarity space to use (`cosine`, `l2`). | `"cosine"` |
101
+ | | `ef_construction` | Size of the dynamic list during index construction. | `200` |
102
+ | | `m` | Number of connections per layer. | `16` |
103
+ | **PyNNDescent** | `metric` | Similarity metric to use (`cosine`, `euclidean`, `manhattan`). | `"cosine"` |
104
+ | | `n_neighbors` | Number of neighbors to use for search. | `15` |
105
+ | **Usearch** | `metric` | Similarity metric to use (`cos`, `ip`, `l2sq`, `hamming`, `tanimoto`). | `"cos"` |
106
+ | | `connectivity` | Number of connections per node in the graph. | `16` |
107
+ | | `expansion_add` | Number of candidates considered during graph construction. | `128` |
108
+ | | `expansion_search` | Number of candidates considered during search. | `64` |
109
+
101
110
 
102
111
 
103
112
  ## Usage
@@ -51,6 +51,7 @@ pynndescent = [
51
51
  ]
52
52
  annoy = ["annoy"]
53
53
  faiss = ["faiss-cpu"]
54
+ usearch = ["usearch"]
54
55
 
55
56
  [project.urls]
56
57
  "Homepage" = "https://github.com/MinishLab"
@@ -34,8 +34,10 @@ BACKEND_PARAMS = [(Backend.FAISS, index_type) for index_type in _faiss_index_typ
34
34
  (Backend.HNSW, None),
35
35
  (Backend.ANNOY, None),
36
36
  (Backend.PYNNDESCENT, None),
37
+ (Backend.USEARCH, None),
37
38
  ]
38
39
 
40
+
39
41
  # Create human-readable ids for each backend type
40
42
  BACKEND_IDS = [f"{backend.name}-{index_type}" if index_type else backend.name for backend, index_type in BACKEND_PARAMS]
41
43
 
@@ -100,15 +100,8 @@ def test_vicinity_delete(vicinity_instance: Vicinity, items: list[str], vectors:
100
100
  :param items: List of item names.
101
101
  :param vectors: Array of vectors corresponding to items.
102
102
  """
103
- if vicinity_instance.backend.backend_type in {Backend.ANNOY, Backend.PYNNDESCENT}:
104
- # Skip delete for Annoy and Pynndescent backend
105
- return
106
-
107
- elif vicinity_instance.backend.backend_type == Backend.FAISS and vicinity_instance.backend.arguments.index_type in {
108
- "hnsw",
109
- "ivfpqr",
110
- }:
111
- # Skip delete test for FAISS index types that do not support deletion
103
+ if vicinity_instance.backend.backend_type != Backend.BASIC:
104
+ # Skip delete for non-basic backends
112
105
  return
113
106
 
114
107
  # Get the vector corresponding to "item2"
@@ -163,6 +156,9 @@ def test_vicinity_delete_nonexistent(vicinity_instance: Vicinity) -> None:
163
156
  :param vicinity_instance: A Vicinity instance.
164
157
  :raises ValueError: If deleting items that do not exist.
165
158
  """
159
+ if vicinity_instance.backend.backend_type != Backend.BASIC:
160
+ # Skip delete for non-basic backends
161
+ return
166
162
  with pytest.raises(ValueError):
167
163
  vicinity_instance.delete(["item10002"])
168
164
 
@@ -193,3 +189,34 @@ def test_vicinity_insert_wrong_dimension(vicinity_instance: Vicinity) -> None:
193
189
 
194
190
  with pytest.raises(ValueError):
195
191
  vicinity_instance.insert(new_item, new_vector)
192
+
193
+
194
+ def test_vicinity_delete_and_query(vicinity_instance: Vicinity, items: list[str], vectors: np.ndarray) -> None:
195
+ """
196
+ Test Vicinity's delete and query methods together to ensure that indices are correctly handled after deletions.
197
+
198
+ :param vicinity_instance: A Vicinity instance.
199
+ :param items: List of item names.
200
+ :param vectors: Array of vectors corresponding to items.
201
+ """
202
+ if vicinity_instance.backend.backend_type != Backend.BASIC:
203
+ # Skip delete for non-basic backends
204
+ return
205
+
206
+ # Delete some items from the Vicinity instance
207
+ items_to_delete = ["item2", "item4", "item6"]
208
+ vicinity_instance.delete(items_to_delete)
209
+
210
+ # Ensure the items are no longer in the items list
211
+ for item in items_to_delete:
212
+ assert item not in vicinity_instance.items
213
+
214
+ # Query using a vector of an item that wasn't deleted
215
+ item3_index = items.index("item3")
216
+ item3_vector = vectors[item3_index]
217
+
218
+ results = vicinity_instance.query(item3_vector, k=10)
219
+ returned_items = [item for item, _ in results[0]]
220
+
221
+ # Check that the queried item is in the results
222
+ assert "item3" in returned_items