zeusdb-vector-database 0.0.3__tar.gz → 0.0.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of zeusdb-vector-database might be problematic. Click here for more details.

@@ -6,6 +6,9 @@ This product includes software developed by:
6
6
  - The Python Software Foundation (https://www.python.org/psf/)
7
7
  - The Rust Project Developers (https://www.rust-lang.org/)
8
8
  - The PyO3 Project (https://pyo3.rs/)
9
+ - Jean-pierreBoth (https://crates.io/crates/hnsw_rs/)
10
+
11
+ Third-party components may be licensed under their respective terms. See LICENSES/ for full license texts.
9
12
 
10
13
  Licensed under the Apache License, Version 2.0 (the "License");
11
14
  you may not use this file except in compliance with the License.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: zeusdb-vector-database
3
- Version: 0.0.3
3
+ Version: 0.0.4
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Programming Language :: Python :: Implementation :: CPython
6
6
  Requires-Dist: numpy>=2.2.6,<3.0.0
@@ -75,13 +75,17 @@ ZeusDB leverages the HNSW (Hierarchical Navigable Small World) algorithm for spe
75
75
 
76
76
  | Metric | Description |
77
77
  |--------|--------------------------------------|
78
- | cosine | Cosine distance (1 - Cosine Similiarity) |
78
+ | cosine | Cosine Distance (1 - Cosine Similiarity) |
79
79
  <!--
80
80
  | l2 | Euclidean distance |
81
81
  | dot | Dot product |
82
82
 
83
83
  -->
84
84
 
85
+ Scores vs Distances:
86
+ - Similarity Scores (higher = more similar)
87
+ - Distances (lower = more similar)
88
+
85
89
  <br/>
86
90
 
87
91
  ## 📦 Installation
@@ -144,10 +148,23 @@ result = index.add(records)
144
148
  query_vec = [0.1, 0.2, 0.3, 0.1, 0.4, 0.2, 0.6, 0.7]
145
149
 
146
150
  # Query with no filter (all documents)
147
- print("\n--- Querying without filter (all documents) ---")
148
151
  results = index.query(vector=query_vec, filter=None, top_k=2)
149
- for doc_id, score in results:
150
- print(f"{doc_id} (score={score:.4f})")
152
+ print("\n--- Raw Results Format ---")
153
+ print(results)
154
+
155
+ print("\n--- Formatted Results ---")
156
+ for i, res in enumerate(results, 1):
157
+ print(f"{i}. ID: {res['id']}, Score: {res['score']:.4f}, Metadata: {res['metadata']}")
158
+ ```
159
+
160
+ *Results Output:*
161
+ ```
162
+ --- Raw Results Format ---
163
+ [{'id': 'doc_001', 'score': 0.0, 'metadata': {'author': 'Alice'}}, {'id': 'doc_003', 'score': 0.0009883458260446787, 'metadata': {'author': 'Alice'}}]
164
+
165
+ --- Formatted Results ---
166
+ 1. ID: doc_001, Score: 0.0000, Metadata: {'author': 'Alice'}
167
+ 2. ID: doc_003, Score: 0.0010, Metadata: {'author': 'Alice'}
151
168
  ```
152
169
 
153
170
  <br/>
@@ -280,14 +297,15 @@ This pre-filters on the given metadata prior to conducting the similarity search
280
297
  ```python
281
298
  print("\n--- Querying with filter: author = 'Alice' ---")
282
299
  results = index.query(vector=query_vec, filter={"author": "Alice"}, top_k=5)
283
- for doc_id, score in results:
284
- print(f"{doc_id} (score={score:.4f})")
300
+ print(results)
285
301
  ```
286
302
  *Output*
287
303
  ```
288
- doc_001 (score=0.0000)
289
- doc_003 (score=0.0010)
290
- doc_005 (score=0.0011)
304
+ [
305
+ {'id': 'doc_001', 'score': 0.0, 'metadata': {'author': 'Alice'}},
306
+ {'id': 'doc_003', 'score': 0.0009883458260446787, 'metadata': {'author': 'Alice'}},
307
+ {'id': 'doc_005', 'score': 0.0011433829786255956, 'metadata': {'author': 'Alice'}}
308
+ ]
291
309
  ```
292
310
 
293
311
 
@@ -56,13 +56,17 @@ ZeusDB leverages the HNSW (Hierarchical Navigable Small World) algorithm for spe
56
56
 
57
57
  | Metric | Description |
58
58
  |--------|--------------------------------------|
59
- | cosine | Cosine distance (1 - Cosine Similiarity) |
59
+ | cosine | Cosine Distance (1 - Cosine Similiarity) |
60
60
  <!--
61
61
  | l2 | Euclidean distance |
62
62
  | dot | Dot product |
63
63
 
64
64
  -->
65
65
 
66
+ Scores vs Distances:
67
+ - Similarity Scores (higher = more similar)
68
+ - Distances (lower = more similar)
69
+
66
70
  <br/>
67
71
 
68
72
  ## 📦 Installation
@@ -125,10 +129,23 @@ result = index.add(records)
125
129
  query_vec = [0.1, 0.2, 0.3, 0.1, 0.4, 0.2, 0.6, 0.7]
126
130
 
127
131
  # Query with no filter (all documents)
128
- print("\n--- Querying without filter (all documents) ---")
129
132
  results = index.query(vector=query_vec, filter=None, top_k=2)
130
- for doc_id, score in results:
131
- print(f"{doc_id} (score={score:.4f})")
133
+ print("\n--- Raw Results Format ---")
134
+ print(results)
135
+
136
+ print("\n--- Formatted Results ---")
137
+ for i, res in enumerate(results, 1):
138
+ print(f"{i}. ID: {res['id']}, Score: {res['score']:.4f}, Metadata: {res['metadata']}")
139
+ ```
140
+
141
+ *Results Output:*
142
+ ```
143
+ --- Raw Results Format ---
144
+ [{'id': 'doc_001', 'score': 0.0, 'metadata': {'author': 'Alice'}}, {'id': 'doc_003', 'score': 0.0009883458260446787, 'metadata': {'author': 'Alice'}}]
145
+
146
+ --- Formatted Results ---
147
+ 1. ID: doc_001, Score: 0.0000, Metadata: {'author': 'Alice'}
148
+ 2. ID: doc_003, Score: 0.0010, Metadata: {'author': 'Alice'}
132
149
  ```
133
150
 
134
151
  <br/>
@@ -261,14 +278,15 @@ This pre-filters on the given metadata prior to conducting the similarity search
261
278
  ```python
262
279
  print("\n--- Querying with filter: author = 'Alice' ---")
263
280
  results = index.query(vector=query_vec, filter={"author": "Alice"}, top_k=5)
264
- for doc_id, score in results:
265
- print(f"{doc_id} (score={score:.4f})")
281
+ print(results)
266
282
  ```
267
283
  *Output*
268
284
  ```
269
- doc_001 (score=0.0000)
270
- doc_003 (score=0.0010)
271
- doc_005 (score=0.0011)
285
+ [
286
+ {'id': 'doc_001', 'score': 0.0, 'metadata': {'author': 'Alice'}},
287
+ {'id': 'doc_003', 'score': 0.0009883458260446787, 'metadata': {'author': 'Alice'}},
288
+ {'id': 'doc_005', 'score': 0.0011433829786255956, 'metadata': {'author': 'Alice'}}
289
+ ]
272
290
  ```
273
291
 
274
292
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "zeusdb-vector-database"
3
- version = "0.0.3"
3
+ version = "0.0.4"
4
4
  description = "Blazing-fast vector DB with real-time similarity search and metadata filtering."
5
5
  readme = "README.md"
6
6
  authors = [
@@ -1,7 +1,7 @@
1
1
  """
2
2
  ZeusDB Vector Database Module
3
3
  """
4
- __version__ = "0.0.3"
4
+ __version__ = "0.0.4"
5
5
 
6
6
  from .vector_database import VectorDatabase # imports the VectorDatabase class from the vector_database.py file
7
7
 
@@ -1100,7 +1100,7 @@ dependencies = [
1100
1100
 
1101
1101
  [[package]]
1102
1102
  name = "zeusdb-vector-database"
1103
- version = "0.0.3"
1103
+ version = "0.0.4"
1104
1104
  dependencies = [
1105
1105
  "hnsw_rs",
1106
1106
  "numpy",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "zeusdb-vector-database"
3
- version = "0.0.3"
3
+ version = "0.0.4"
4
4
  edition = "2021"
5
5
  resolver = "2" # <-- Avoid compiling unnecessary features from dependencies.
6
6
 
@@ -526,15 +526,72 @@ impl HNSWIndex {
526
526
  Ok(result)
527
527
  }
528
528
 
529
- /// Query the index for the k-nearest neighbors of a vector
530
- #[pyo3(signature = (vector, filter=None, top_k=10, ef_search=None))]
529
+ // /// DEPRECIATED: This one returns a list of tuples (id, score)
530
+ // /// Query the index for the k-nearest neighbors of a vector
531
+ // #[pyo3(signature = (vector, filter=None, top_k=10, ef_search=None))]
532
+ // pub fn query(
533
+ // &self,
534
+ // vector: Vec<f32>,
535
+ // filter: Option<HashMap<String, String>>,
536
+ // top_k: usize,
537
+ // ef_search: Option<usize>,
538
+ // ) -> PyResult<Vec<(String, f32)>> {
539
+ // if vector.len() != self.dim {
540
+ // return Err(PyErr::new::<pyo3::exceptions::PyValueError, _>(format!(
541
+ // "Query vector dimension mismatch: expected {}, got {}",
542
+ // self.dim, vector.len()
543
+ // )));
544
+ // }
545
+
546
+ // // Get results from HNSW graph
547
+ // let ef = ef_search.unwrap_or_else(|| std::cmp::max(2 * top_k, 100));
548
+ // let results = self.hnsw.search(&vector, top_k, ef);
549
+
550
+ // let mut filtered_results = Vec::new();
551
+
552
+ // for neighbor in results {
553
+ // let score = neighbor.distance;
554
+ // let internal_id = neighbor.get_origin_id();
555
+
556
+ // if let Some(ext_id) = self.rev_map.get(&internal_id) {
557
+ // // Apply filtering if provided
558
+ // if let Some(ref filter_map) = filter {
559
+ // if let Some(meta) = self.vector_metadata.get(ext_id) {
560
+ // let mut matches = true;
561
+ // for (k, v) in filter_map {
562
+ // if meta.get(k) != Some(v) {
563
+ // matches = false;
564
+ // break;
565
+ // }
566
+ // }
567
+ // if !matches {
568
+ // continue;
569
+ // }
570
+ // } else {
571
+ // // No metadata, but filter required - skip
572
+ // continue;
573
+ // }
574
+ // }
575
+ // filtered_results.push((ext_id.clone(), score));
576
+ // }
577
+ // }
578
+
579
+ // Ok(filtered_results)
580
+ // }
581
+
582
+
583
+ /// Search for the k-nearest neighbors of a vector
584
+ /// Returns actual Python dictionaries which most common for ML workflows
585
+ #[pyo3(signature = (vector, filter=None, top_k=10, ef_search=None, return_vector=false))]
531
586
  pub fn query(
532
587
  &self,
588
+ py: Python<'_>,
533
589
  vector: Vec<f32>,
534
590
  filter: Option<HashMap<String, String>>,
535
591
  top_k: usize,
536
592
  ef_search: Option<usize>,
537
- ) -> PyResult<Vec<(String, f32)>> {
593
+ return_vector: bool,
594
+ ) -> PyResult<Vec<Py<PyDict>>> {
538
595
  if vector.len() != self.dim {
539
596
  return Err(PyErr::new::<pyo3::exceptions::PyValueError, _>(format!(
540
597
  "Query vector dimension mismatch: expected {}, got {}",
@@ -542,99 +599,183 @@ impl HNSWIndex {
542
599
  )));
543
600
  }
544
601
 
545
- // Get results from HNSW graph
546
602
  let ef = ef_search.unwrap_or_else(|| std::cmp::max(2 * top_k, 100));
547
603
  let results = self.hnsw.search(&vector, top_k, ef);
548
604
 
549
- let mut filtered_results = Vec::new();
605
+ let mut output = Vec::with_capacity(results.len());
550
606
 
551
607
  for neighbor in results {
552
608
  let score = neighbor.distance;
553
609
  let internal_id = neighbor.get_origin_id();
554
-
610
+
555
611
  if let Some(ext_id) = self.rev_map.get(&internal_id) {
556
- // Apply filtering if provided
612
+ // Apply optional metadata filter
557
613
  if let Some(ref filter_map) = filter {
558
614
  if let Some(meta) = self.vector_metadata.get(ext_id) {
559
- let mut matches = true;
560
- for (k, v) in filter_map {
561
- if meta.get(k) != Some(v) {
562
- matches = false;
563
- break;
564
- }
565
- }
615
+ let matches = filter_map.iter().all(|(k, v)| meta.get(k) == Some(v));
566
616
  if !matches {
567
617
  continue;
568
618
  }
569
619
  } else {
570
- // No metadata, but filter required - skip
571
- continue;
620
+ continue; // no metadata to match against
621
+ }
622
+ }
623
+
624
+ let dict = PyDict::new(py);
625
+ dict.set_item("id", ext_id)?;
626
+ dict.set_item("score", score)?;
627
+
628
+ let metadata = self.vector_metadata.get(ext_id).cloned().unwrap_or_default();
629
+ dict.set_item("metadata", metadata)?;
630
+
631
+ if return_vector {
632
+ if let Some(vec) = self.vectors.get(ext_id) {
633
+ dict.set_item("vector", vec.clone())?;
572
634
  }
573
635
  }
574
- filtered_results.push((ext_id.clone(), score));
636
+
637
+ output.push(dict.into());
575
638
  }
576
639
  }
577
640
 
578
- Ok(filtered_results)
641
+ Ok(output)
579
642
  }
580
643
 
581
- /// Search with metadata included in results
582
- #[pyo3(signature = (vector, filter=None, top_k=10, ef_search=None, include_metadata=false))]
583
- pub fn search_with_metadata(
584
- &self,
585
- vector: Vec<f32>,
586
- filter: Option<HashMap<String, String>>,
587
- top_k: usize,
588
- ef_search: Option<usize>,
589
- include_metadata: bool,
590
- ) -> PyResult<Vec<(String, f32, Option<HashMap<String, String>>)>> {
591
- if vector.len() != self.dim {
592
- return Err(PyErr::new::<pyo3::exceptions::PyValueError, _>(format!(
593
- "Query vector dimension mismatch: expected {}, got {}",
594
- self.dim, vector.len()
595
- )));
596
- }
597
644
 
598
- let ef = ef_search.unwrap_or_else(|| std::cmp::max(2 * top_k, 100));
599
- let results = self.hnsw.search(&vector, top_k, ef);
600
645
 
601
- let mut filtered_results = Vec::new();
602
646
 
603
- for neighbor in results {
604
- let score = neighbor.distance;
605
- let internal_id = neighbor.get_origin_id();
647
+
648
+
649
+
650
+
651
+
652
+ // /// DEPRECIATED: This one returns a list of tuples (id, score)
653
+ // /// Search with metadata included in results
654
+ // #[pyo3(signature = (vector, filter=None, top_k=10, ef_search=None, include_metadata=false))]
655
+ // pub fn search_with_metadata(
656
+ // &self,
657
+ // vector: Vec<f32>,
658
+ // filter: Option<HashMap<String, String>>,
659
+ // top_k: usize,
660
+ // ef_search: Option<usize>,
661
+ // include_metadata: bool,
662
+ // ) -> PyResult<Vec<(String, f32, Option<HashMap<String, String>>)>> {
663
+ // if vector.len() != self.dim {
664
+ // return Err(PyErr::new::<pyo3::exceptions::PyValueError, _>(format!(
665
+ // "Query vector dimension mismatch: expected {}, got {}",
666
+ // self.dim, vector.len()
667
+ // )));
668
+ // }
669
+
670
+ // let ef = ef_search.unwrap_or_else(|| std::cmp::max(2 * top_k, 100));
671
+ // let results = self.hnsw.search(&vector, top_k, ef);
672
+
673
+ // let mut filtered_results = Vec::new();
674
+
675
+ // for neighbor in results {
676
+ // let score = neighbor.distance;
677
+ // let internal_id = neighbor.get_origin_id();
606
678
 
607
- if let Some(ext_id) = self.rev_map.get(&internal_id) {
608
- // Apply filtering if provided
609
- if let Some(ref filter_map) = filter {
610
- if let Some(meta) = self.vector_metadata.get(ext_id) {
611
- let mut matches = true;
612
- for (k, v) in filter_map {
613
- if meta.get(k) != Some(v) {
614
- matches = false;
615
- break;
616
- }
617
- }
618
- if !matches {
619
- continue;
620
- }
621
- } else {
622
- continue;
623
- }
624
- }
679
+ // if let Some(ext_id) = self.rev_map.get(&internal_id) {
680
+ // // Apply filtering if provided
681
+ // if let Some(ref filter_map) = filter {
682
+ // if let Some(meta) = self.vector_metadata.get(ext_id) {
683
+ // let mut matches = true;
684
+ // for (k, v) in filter_map {
685
+ // if meta.get(k) != Some(v) {
686
+ // matches = false;
687
+ // break;
688
+ // }
689
+ // }
690
+ // if !matches {
691
+ // continue;
692
+ // }
693
+ // } else {
694
+ // continue;
695
+ // }
696
+ // }
697
+
698
+ // let metadata = if include_metadata {
699
+ // self.vector_metadata.get(ext_id).cloned()
700
+ // } else {
701
+ // None
702
+ // };
703
+
704
+ // filtered_results.push((ext_id.clone(), score, metadata));
705
+ // }
706
+ // }
707
+
708
+ // Ok(filtered_results)
709
+ // }
710
+
711
+
712
+ // /// DEPRECIATED: Use `query()` instead. Have made query the single do all function.
713
+ // /// Search with metadata included in results
714
+ // #[pyo3(signature = (vector, filter=None, top_k=10, ef_search=None, include_metadata=true))]
715
+ // pub fn search_with_metadata(
716
+ // &self,
717
+ // py: Python<'_>,
718
+ // vector: Vec<f32>,
719
+ // filter: Option<HashMap<String, String>>,
720
+ // top_k: usize,
721
+ // ef_search: Option<usize>,
722
+ // include_metadata: bool,
723
+ // ) -> PyResult<Vec<Py<PyDict>>> {
724
+ // if vector.len() != self.dim {
725
+ // return Err(PyErr::new::<pyo3::exceptions::PyValueError, _>(format!(
726
+ // "Query vector dimension mismatch: expected {}, got {}",
727
+ // self.dim, vector.len()
728
+ // )));
729
+ // }
730
+
731
+ // let ef = ef_search.unwrap_or_else(|| std::cmp::max(2 * top_k, 100));
732
+ // let results = self.hnsw.search(&vector, top_k, ef);
733
+
734
+ // let mut output = Vec::with_capacity(results.len());
735
+
736
+ // for neighbor in results {
737
+ // let score = neighbor.distance;
738
+ // let internal_id = neighbor.get_origin_id();
739
+
740
+ // if let Some(ext_id) = self.rev_map.get(&internal_id) {
741
+ // // Apply optional filtering
742
+ // if let Some(ref filter_map) = filter {
743
+ // if let Some(meta) = self.vector_metadata.get(ext_id) {
744
+ // let matches = filter_map.iter().all(|(k, v)| meta.get(k) == Some(v));
745
+ // if !matches {
746
+ // continue;
747
+ // }
748
+ // } else {
749
+ // continue;
750
+ // }
751
+ // }
752
+
753
+ // let dict = PyDict::new(py);
754
+ // dict.set_item("id", ext_id)?;
755
+ // dict.set_item("score", score)?;
756
+
757
+ // if include_metadata {
758
+ // let metadata = self.vector_metadata.get(ext_id).cloned().unwrap_or_default();
759
+ // dict.set_item("metadata", metadata)?;
760
+ // } else {
761
+ // dict.set_item("metadata", PyDict::new(py))?;
762
+ // }
763
+
764
+ // output.push(dict.into());
765
+ // }
766
+ // }
767
+
768
+ // Ok(output)
769
+ // }
770
+
771
+
772
+
773
+
774
+
775
+
625
776
 
626
- let metadata = if include_metadata {
627
- self.vector_metadata.get(ext_id).cloned()
628
- } else {
629
- None
630
- };
631
777
 
632
- filtered_results.push((ext_id.clone(), score, metadata));
633
- }
634
- }
635
778
 
636
- Ok(filtered_results)
637
- }
638
779
 
639
780
  /// Get vector by ID
640
781
  pub fn get_vector(&self, id: String) -> Option<Vec<f32>> {