sutradb-core 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sutradb/__init__.py +23 -0
- sutradb/bm25.py +143 -0
- sutradb/core.py +553 -0
- sutradb/distance.py +92 -0
- sutradb/filters.py +121 -0
- sutradb/fusion.py +112 -0
- sutradb/server.py +211 -0
- sutradb/storage.py +180 -0
- sutradb_core-2.0.0.dist-info/METADATA +204 -0
- sutradb_core-2.0.0.dist-info/RECORD +14 -0
- sutradb_core-2.0.0.dist-info/WHEEL +5 -0
- sutradb_core-2.0.0.dist-info/entry_points.txt +2 -0
- sutradb_core-2.0.0.dist-info/licenses/LICENSE +21 -0
- sutradb_core-2.0.0.dist-info/top_level.txt +1 -0
sutradb/__init__.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""
|
|
2
|
+
SutraDB (सूत्र DB) - High-performance, zero-dependency hybrid vector search
|
|
3
|
+
and BM25 lexical engine in pure Python.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from sutradb.core import SutraDB, Collection, Document, SearchResult
|
|
7
|
+
from sutradb.distance import Metric
|
|
8
|
+
from sutradb.filters import FilterEngine
|
|
9
|
+
from sutradb.bm25 import BM25Index
|
|
10
|
+
from sutradb.fusion import reciprocal_rank_fusion, linear_score_fusion
|
|
11
|
+
|
|
12
|
+
__version__ = "2.0.0"
|
|
13
|
+
__all__ = [
|
|
14
|
+
"SutraDB",
|
|
15
|
+
"Collection",
|
|
16
|
+
"Document",
|
|
17
|
+
"SearchResult",
|
|
18
|
+
"Metric",
|
|
19
|
+
"FilterEngine",
|
|
20
|
+
"BM25Index",
|
|
21
|
+
"reciprocal_rank_fusion",
|
|
22
|
+
"linear_score_fusion",
|
|
23
|
+
]
|
sutradb/bm25.py
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Lexical keyword search engine implementing BM25Okapi with an in-memory inverted index.
|
|
3
|
+
Used in SutraDB's hybrid search pipeline alongside dense vector similarity.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from collections import Counter, defaultdict
|
|
7
|
+
import math
|
|
8
|
+
import re
|
|
9
|
+
from typing import Dict, List, Optional, Set, Tuple
|
|
10
|
+
import numpy as np
|
|
11
|
+
|
|
12
|
+
# Standard lightweight English stopwords
|
|
13
|
+
STOPWORDS: Set[str] = {
|
|
14
|
+
"a", "about", "above", "after", "again", "against", "all", "am", "an", "and",
|
|
15
|
+
"any", "are", "aren't", "as", "at", "be", "because", "been", "before", "being",
|
|
16
|
+
"below", "between", "both", "but", "by", "can't", "cannot", "could", "couldn't",
|
|
17
|
+
"did", "didn't", "do", "does", "doesn't", "doing", "don't", "down", "during",
|
|
18
|
+
"each", "few", "for", "from", "further", "had", "hadn't", "has", "hasn't",
|
|
19
|
+
"have", "haven't", "having", "he", "he'd", "he'll", "he's", "her", "here",
|
|
20
|
+
"here's", "hers", "herself", "him", "himself", "his", "how", "how's", "i",
|
|
21
|
+
"i'd", "i'll", "i'm", "i've", "if", "in", "into", "is", "isn't", "it", "it's",
|
|
22
|
+
"its", "itself", "let's", "me", "more", "most", "mustn't", "my", "myself",
|
|
23
|
+
"no", "nor", "not", "of", "off", "on", "once", "only", "or", "other", "ought",
|
|
24
|
+
"our", "ours", "ourselves", "out", "over", "own", "same", "shan't", "she",
|
|
25
|
+
"she'd", "she'll", "she's", "should", "shouldn't", "so", "some", "such",
|
|
26
|
+
"than", "that", "that's", "the", "their", "theirs", "them", "themselves",
|
|
27
|
+
"then", "there", "there's", "these", "they", "they'd", "they'll", "they're",
|
|
28
|
+
"they've", "this", "those", "through", "to", "too", "under", "until", "up",
|
|
29
|
+
"very", "was", "wasn't", "we", "we'd", "we'll", "we're", "we've", "were",
|
|
30
|
+
"weren't", "what", "what's", "when", "when's", "where", "where's", "which",
|
|
31
|
+
"while", "who", "who's", "whom", "why", "why's", "with", "won't", "would",
|
|
32
|
+
"wouldn't", "you", "you'd", "you'll", "you're", "you've", "your", "yours",
|
|
33
|
+
"yourself", "yourselves"
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
_WORD_PATTERN = re.compile(r"\b[a-zA-Z0-9_\-\.\$]+\b")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def tokenize(text: str, filter_stopwords: bool = False) -> List[str]:
|
|
40
|
+
"""Tokenizes string into lowercase alphanumeric and symbol terms."""
|
|
41
|
+
if not text:
|
|
42
|
+
return []
|
|
43
|
+
tokens = [t.lower() for t in _WORD_PATTERN.findall(text)]
|
|
44
|
+
if filter_stopwords:
|
|
45
|
+
return [t for t in tokens if t not in STOPWORDS]
|
|
46
|
+
return tokens
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class BM25Index:
|
|
50
|
+
"""In-memory BM25Okapi index with dynamic document updates."""
|
|
51
|
+
|
|
52
|
+
def __init__(self, k1: float = 1.5, b: float = 0.75):
|
|
53
|
+
self.k1 = k1
|
|
54
|
+
self.b = b
|
|
55
|
+
self.doc_count: int = 0
|
|
56
|
+
self.doc_lengths: List[int] = []
|
|
57
|
+
self.avg_doc_length: float = 0.0
|
|
58
|
+
|
|
59
|
+
# Inverted index: term -> list of (doc_index, term_frequency)
|
|
60
|
+
self.inverted_index: Dict[str, List[Tuple[int, int]]] = defaultdict(list)
|
|
61
|
+
# Document frequencies: term -> number of docs containing term
|
|
62
|
+
self.doc_frequencies: Dict[str, int] = defaultdict(int)
|
|
63
|
+
# Cached IDF values
|
|
64
|
+
self.idf_cache: Dict[str, float] = {}
|
|
65
|
+
|
|
66
|
+
def add_documents(self, corpus: List[str]) -> None:
|
|
67
|
+
"""Indexes a batch of raw text strings."""
|
|
68
|
+
start_idx = self.doc_count
|
|
69
|
+
total_len = sum(self.doc_lengths)
|
|
70
|
+
|
|
71
|
+
for i, text in enumerate(corpus):
|
|
72
|
+
doc_idx = start_idx + i
|
|
73
|
+
tokens = tokenize(text)
|
|
74
|
+
doc_len = len(tokens)
|
|
75
|
+
self.doc_lengths.append(doc_len)
|
|
76
|
+
total_len += doc_len
|
|
77
|
+
|
|
78
|
+
counts = Counter(tokens)
|
|
79
|
+
for term, freq in counts.items():
|
|
80
|
+
self.inverted_index[term].append((doc_idx, freq))
|
|
81
|
+
self.doc_frequencies[term] += 1
|
|
82
|
+
|
|
83
|
+
self.doc_count += len(corpus)
|
|
84
|
+
if self.doc_count > 0:
|
|
85
|
+
self.avg_doc_length = total_len / self.doc_count
|
|
86
|
+
|
|
87
|
+
# Invalidate IDF cache
|
|
88
|
+
self.idf_cache.clear()
|
|
89
|
+
|
|
90
|
+
def _get_idf(self, term: str) -> float:
|
|
91
|
+
"""Calculates Robertson-Spärck Jones IDF with non-negative smoothing."""
|
|
92
|
+
if term in self.idf_cache:
|
|
93
|
+
return self.idf_cache[term]
|
|
94
|
+
|
|
95
|
+
n_q = self.doc_frequencies.get(term, 0)
|
|
96
|
+
if n_q == 0:
|
|
97
|
+
idf = 0.0
|
|
98
|
+
else:
|
|
99
|
+
# Robertson-Spärck Jones formulation guaranteeing positive scores
|
|
100
|
+
idf = math.log(1.0 + (self.doc_count - n_q + 0.5) / (n_q + 0.5))
|
|
101
|
+
|
|
102
|
+
self.idf_cache[term] = idf
|
|
103
|
+
return idf
|
|
104
|
+
|
|
105
|
+
def score_query(
|
|
106
|
+
self,
|
|
107
|
+
query: str,
|
|
108
|
+
mask: Optional[np.ndarray] = None
|
|
109
|
+
) -> np.ndarray:
|
|
110
|
+
"""
|
|
111
|
+
Calculates BM25 scores for all documents given a query string.
|
|
112
|
+
Optionally zeroes out any document indices masked as False.
|
|
113
|
+
"""
|
|
114
|
+
if self.doc_count == 0:
|
|
115
|
+
return np.empty(0, dtype=np.float32)
|
|
116
|
+
|
|
117
|
+
scores = np.zeros(self.doc_count, dtype=np.float32)
|
|
118
|
+
query_terms = tokenize(query)
|
|
119
|
+
if not query_terms:
|
|
120
|
+
return scores
|
|
121
|
+
|
|
122
|
+
# Calculate term frequency in query
|
|
123
|
+
q_counts = Counter(query_terms)
|
|
124
|
+
|
|
125
|
+
for term, _ in q_counts.items():
|
|
126
|
+
idf = self._get_idf(term)
|
|
127
|
+
if idf <= 0:
|
|
128
|
+
continue
|
|
129
|
+
|
|
130
|
+
postings = self.inverted_index.get(term)
|
|
131
|
+
if not postings:
|
|
132
|
+
continue
|
|
133
|
+
|
|
134
|
+
for doc_idx, tf in postings:
|
|
135
|
+
if mask is not None and not mask[doc_idx]:
|
|
136
|
+
continue
|
|
137
|
+
|
|
138
|
+
doc_len = self.doc_lengths[doc_idx]
|
|
139
|
+
denom = tf + self.k1 * (1.0 - self.b + self.b * (doc_len / (self.avg_doc_length or 1.0)))
|
|
140
|
+
numerator = tf * (self.k1 + 1.0)
|
|
141
|
+
scores[doc_idx] += idf * (numerator / denom)
|
|
142
|
+
|
|
143
|
+
return scores
|