sutradb-core 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sutradb/__init__.py ADDED
@@ -0,0 +1,23 @@
1
+ """
2
+ SutraDB (सूत्र DB) - High-performance, zero-dependency hybrid vector search
3
+ and BM25 lexical engine in pure Python.
4
+ """
5
+
6
+ from sutradb.core import SutraDB, Collection, Document, SearchResult
7
+ from sutradb.distance import Metric
8
+ from sutradb.filters import FilterEngine
9
+ from sutradb.bm25 import BM25Index
10
+ from sutradb.fusion import reciprocal_rank_fusion, linear_score_fusion
11
+
12
+ __version__ = "2.0.0"
13
+ __all__ = [
14
+ "SutraDB",
15
+ "Collection",
16
+ "Document",
17
+ "SearchResult",
18
+ "Metric",
19
+ "FilterEngine",
20
+ "BM25Index",
21
+ "reciprocal_rank_fusion",
22
+ "linear_score_fusion",
23
+ ]
sutradb/bm25.py ADDED
@@ -0,0 +1,143 @@
1
+ """
2
+ Lexical keyword search engine implementing BM25Okapi with an in-memory inverted index.
3
+ Used in SutraDB's hybrid search pipeline alongside dense vector similarity.
4
+ """
5
+
6
+ from collections import Counter, defaultdict
7
+ import math
8
+ import re
9
+ from typing import Dict, List, Optional, Set, Tuple
10
+ import numpy as np
11
+
12
+ # Standard lightweight English stopwords
13
+ STOPWORDS: Set[str] = {
14
+ "a", "about", "above", "after", "again", "against", "all", "am", "an", "and",
15
+ "any", "are", "aren't", "as", "at", "be", "because", "been", "before", "being",
16
+ "below", "between", "both", "but", "by", "can't", "cannot", "could", "couldn't",
17
+ "did", "didn't", "do", "does", "doesn't", "doing", "don't", "down", "during",
18
+ "each", "few", "for", "from", "further", "had", "hadn't", "has", "hasn't",
19
+ "have", "haven't", "having", "he", "he'd", "he'll", "he's", "her", "here",
20
+ "here's", "hers", "herself", "him", "himself", "his", "how", "how's", "i",
21
+ "i'd", "i'll", "i'm", "i've", "if", "in", "into", "is", "isn't", "it", "it's",
22
+ "its", "itself", "let's", "me", "more", "most", "mustn't", "my", "myself",
23
+ "no", "nor", "not", "of", "off", "on", "once", "only", "or", "other", "ought",
24
+ "our", "ours", "ourselves", "out", "over", "own", "same", "shan't", "she",
25
+ "she'd", "she'll", "she's", "should", "shouldn't", "so", "some", "such",
26
+ "than", "that", "that's", "the", "their", "theirs", "them", "themselves",
27
+ "then", "there", "there's", "these", "they", "they'd", "they'll", "they're",
28
+ "they've", "this", "those", "through", "to", "too", "under", "until", "up",
29
+ "very", "was", "wasn't", "we", "we'd", "we'll", "we're", "we've", "were",
30
+ "weren't", "what", "what's", "when", "when's", "where", "where's", "which",
31
+ "while", "who", "who's", "whom", "why", "why's", "with", "won't", "would",
32
+ "wouldn't", "you", "you'd", "you'll", "you're", "you've", "your", "yours",
33
+ "yourself", "yourselves"
34
+ }
35
+
36
+ _WORD_PATTERN = re.compile(r"\b[a-zA-Z0-9_\-\.\$]+\b")
37
+
38
+
39
+ def tokenize(text: str, filter_stopwords: bool = False) -> List[str]:
40
+ """Tokenizes string into lowercase alphanumeric and symbol terms."""
41
+ if not text:
42
+ return []
43
+ tokens = [t.lower() for t in _WORD_PATTERN.findall(text)]
44
+ if filter_stopwords:
45
+ return [t for t in tokens if t not in STOPWORDS]
46
+ return tokens
47
+
48
+
49
+ class BM25Index:
50
+ """In-memory BM25Okapi index with dynamic document updates."""
51
+
52
+ def __init__(self, k1: float = 1.5, b: float = 0.75):
53
+ self.k1 = k1
54
+ self.b = b
55
+ self.doc_count: int = 0
56
+ self.doc_lengths: List[int] = []
57
+ self.avg_doc_length: float = 0.0
58
+
59
+ # Inverted index: term -> list of (doc_index, term_frequency)
60
+ self.inverted_index: Dict[str, List[Tuple[int, int]]] = defaultdict(list)
61
+ # Document frequencies: term -> number of docs containing term
62
+ self.doc_frequencies: Dict[str, int] = defaultdict(int)
63
+ # Cached IDF values
64
+ self.idf_cache: Dict[str, float] = {}
65
+
66
+ def add_documents(self, corpus: List[str]) -> None:
67
+ """Indexes a batch of raw text strings."""
68
+ start_idx = self.doc_count
69
+ total_len = sum(self.doc_lengths)
70
+
71
+ for i, text in enumerate(corpus):
72
+ doc_idx = start_idx + i
73
+ tokens = tokenize(text)
74
+ doc_len = len(tokens)
75
+ self.doc_lengths.append(doc_len)
76
+ total_len += doc_len
77
+
78
+ counts = Counter(tokens)
79
+ for term, freq in counts.items():
80
+ self.inverted_index[term].append((doc_idx, freq))
81
+ self.doc_frequencies[term] += 1
82
+
83
+ self.doc_count += len(corpus)
84
+ if self.doc_count > 0:
85
+ self.avg_doc_length = total_len / self.doc_count
86
+
87
+ # Invalidate IDF cache
88
+ self.idf_cache.clear()
89
+
90
+ def _get_idf(self, term: str) -> float:
91
+ """Calculates Robertson-Spärck Jones IDF with non-negative smoothing."""
92
+ if term in self.idf_cache:
93
+ return self.idf_cache[term]
94
+
95
+ n_q = self.doc_frequencies.get(term, 0)
96
+ if n_q == 0:
97
+ idf = 0.0
98
+ else:
99
+ # Robertson-Spärck Jones formulation guaranteeing positive scores
100
+ idf = math.log(1.0 + (self.doc_count - n_q + 0.5) / (n_q + 0.5))
101
+
102
+ self.idf_cache[term] = idf
103
+ return idf
104
+
105
+ def score_query(
106
+ self,
107
+ query: str,
108
+ mask: Optional[np.ndarray] = None
109
+ ) -> np.ndarray:
110
+ """
111
+ Calculates BM25 scores for all documents given a query string.
112
+ Optionally zeroes out any document indices masked as False.
113
+ """
114
+ if self.doc_count == 0:
115
+ return np.empty(0, dtype=np.float32)
116
+
117
+ scores = np.zeros(self.doc_count, dtype=np.float32)
118
+ query_terms = tokenize(query)
119
+ if not query_terms:
120
+ return scores
121
+
122
+ # Calculate term frequency in query
123
+ q_counts = Counter(query_terms)
124
+
125
+ for term, _ in q_counts.items():
126
+ idf = self._get_idf(term)
127
+ if idf <= 0:
128
+ continue
129
+
130
+ postings = self.inverted_index.get(term)
131
+ if not postings:
132
+ continue
133
+
134
+ for doc_idx, tf in postings:
135
+ if mask is not None and not mask[doc_idx]:
136
+ continue
137
+
138
+ doc_len = self.doc_lengths[doc_idx]
139
+ denom = tf + self.k1 * (1.0 - self.b + self.b * (doc_len / (self.avg_doc_length or 1.0)))
140
+ numerator = tf * (self.k1 + 1.0)
141
+ scores[doc_idx] += idf * (numerator / denom)
142
+
143
+ return scores