semsift 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. semsift-0.0.1/LICENSE +21 -0
  2. semsift-0.0.1/PKG-INFO +86 -0
  3. semsift-0.0.1/README.md +59 -0
  4. semsift-0.0.1/pyproject.toml +36 -0
  5. semsift-0.0.1/setup.cfg +4 -0
  6. semsift-0.0.1/src/semsift/__init__.py +3 -0
  7. semsift-0.0.1/src/semsift/chunk/__init__.py +5 -0
  8. semsift-0.0.1/src/semsift/chunk/chunkers.py +206 -0
  9. semsift-0.0.1/src/semsift/embed/__init__.py +12 -0
  10. semsift-0.0.1/src/semsift/embed/backends.py +433 -0
  11. semsift-0.0.1/src/semsift/embed/policy.py +89 -0
  12. semsift-0.0.1/src/semsift/embed/space.py +43 -0
  13. semsift-0.0.1/src/semsift/evals/__init__.py +7 -0
  14. semsift-0.0.1/src/semsift/evals/harness.py +233 -0
  15. semsift-0.0.1/src/semsift/fuse/__init__.py +7 -0
  16. semsift-0.0.1/src/semsift/fuse/lists.py +177 -0
  17. semsift-0.0.1/src/semsift/rerank/__init__.py +5 -0
  18. semsift-0.0.1/src/semsift/rerank/rerankers.py +241 -0
  19. semsift-0.0.1/src/semsift/search/__init__.py +5 -0
  20. semsift-0.0.1/src/semsift/search/composer.py +205 -0
  21. semsift-0.0.1/src/semsift/store/__init__.py +8 -0
  22. semsift-0.0.1/src/semsift/store/codec.py +48 -0
  23. semsift-0.0.1/src/semsift/store/filters.py +231 -0
  24. semsift-0.0.1/src/semsift/store/index.py +82 -0
  25. semsift-0.0.1/src/semsift/store/store.py +621 -0
  26. semsift-0.0.1/src/semsift.egg-info/PKG-INFO +86 -0
  27. semsift-0.0.1/src/semsift.egg-info/SOURCES.txt +29 -0
  28. semsift-0.0.1/src/semsift.egg-info/dependency_links.txt +1 -0
  29. semsift-0.0.1/src/semsift.egg-info/requires.txt +20 -0
  30. semsift-0.0.1/src/semsift.egg-info/top_level.txt +1 -0
  31. semsift-0.0.1/tests/test_boundary.py +69 -0
semsift-0.0.1/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Nitin Kumar
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALING IN THE
21
+ SOFTWARE.
semsift-0.0.1/PKG-INFO ADDED
@@ -0,0 +1,86 @@
1
+ Metadata-Version: 2.4
2
+ Name: semsift
3
+ Version: 0.0.1
4
+ Summary: Building blocks for hybrid keyword and vector retrieval over SQLite
5
+ License-Expression: MIT
6
+ Project-URL: Source, https://github.com/nitkrar/semsift
7
+ Requires-Python: >=3.11
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: numpy>=1.26
11
+ Requires-Dist: vicinity>=0.4.6
12
+ Provides-Extra: static
13
+ Requires-Dist: model2vec>=0.9; extra == "static"
14
+ Requires-Dist: huggingface-hub>=0.23; extra == "static"
15
+ Provides-Extra: tree-sitter
16
+ Requires-Dist: tree-sitter-language-pack>=1.20; extra == "tree-sitter"
17
+ Provides-Extra: onnx
18
+ Requires-Dist: onnxruntime>=1.17; extra == "onnx"
19
+ Requires-Dist: huggingface-hub>=0.23; extra == "onnx"
20
+ Requires-Dist: tokenizers>=0.15; extra == "onnx"
21
+ Provides-Extra: webgpu
22
+ Requires-Dist: onnxruntime>=1.24.4; extra == "webgpu"
23
+ Requires-Dist: onnxruntime-ep-webgpu; extra == "webgpu"
24
+ Requires-Dist: huggingface-hub>=0.23; extra == "webgpu"
25
+ Requires-Dist: tokenizers>=0.15; extra == "webgpu"
26
+ Dynamic: license-file
27
+
28
+ # semsift
29
+
30
+ Building blocks for hybrid retrieval: chunk text, embed it, store vectors
31
+ and a keyword index together in SQLite or memory, search both, fuse the
32
+ ranked results and rerank them. It retrieves and ranks; it never
33
+ generates text.
34
+
35
+ **Status:** Implemented, not yet released. See
36
+ [docs/design.md](docs/design.md) for the blocks and their contracts, and
37
+ [docs/features.md](docs/features.md) for what was taken from other
38
+ projects.
39
+
40
+ ```python
41
+ import sqlite3
42
+
43
+ from semsift.embed import StaticEncoder
44
+ from semsift.search import KeywordSource, Search, VectorSource
45
+ from semsift.store import Field, Item, Store
46
+
47
+ conn = sqlite3.connect(":memory:", isolation_level=None)
48
+ store = Store(conn, "notes", [Field("source", "text")],
49
+ encoder=StaticEncoder("minishlab/potion-base-8M"))
50
+
51
+ items = [Item(1, "Rent is due on the first of the month.", {"source": "lease.md"}),
52
+ Item(2, "The boiler service is booked for Tuesday.", {"source": "home.md"})]
53
+ vectors = store.embed(items) # encode before opening the transaction
54
+ conn.execute("BEGIN")
55
+ store.upsert(items, vectors) # the store never commits; you do
56
+ conn.execute("COMMIT")
57
+
58
+ search = Search(store, sources=[VectorSource(store), KeywordSource(store)],
59
+ citation=("source",))
60
+ for hit in search.run("when is rent due", k=1).hits:
61
+ print(hit.citation["source"], hit.text)
62
+ ```
63
+
64
+ Install extras for what you use: `static` (model2vec encoders), `onnx`,
65
+ `webgpu`, `tree-sitter` (syntax-aware chunking).
66
+
67
+ ## Development
68
+
69
+ ```
70
+ uv venv .venv && uv pip install --python .venv/bin/python -e ".[static,onnx,tree-sitter]"
71
+ HF_HUB_OFFLINE=1 .venv/bin/python -m unittest discover -s tests -t .
72
+ .venv/bin/python benchmarks/run.py
73
+ ```
74
+
75
+ To use an unreleased semsift from another project, depend on the
76
+ checkout (`uv add --editable ../semsift`, or `pip install -e ../semsift`),
77
+ or build it (`uv build`) and install the wheel from `dist/`.
78
+
79
+ ## Releasing
80
+
81
+ The version lives only in `semsift.__version__`. Bump it, commit, and push
82
+ a matching tag (`v0.0.1`). The release workflow runs the tests, builds,
83
+ refuses to publish unless the tag, `__version__`, the sdist and the wheel
84
+ agree, publishes to PyPI through trusted publishing, and creates the
85
+ GitHub release. PyPI must list this repository's `release.yml` as a
86
+ trusted publisher for the `semsift` project.
@@ -0,0 +1,59 @@
1
+ # semsift
2
+
3
+ Building blocks for hybrid retrieval: chunk text, embed it, store vectors
4
+ and a keyword index together in SQLite or memory, search both, fuse the
5
+ ranked results and rerank them. It retrieves and ranks; it never
6
+ generates text.
7
+
8
+ **Status:** Implemented, not yet released. See
9
+ [docs/design.md](docs/design.md) for the blocks and their contracts, and
10
+ [docs/features.md](docs/features.md) for what was taken from other
11
+ projects.
12
+
13
+ ```python
14
+ import sqlite3
15
+
16
+ from semsift.embed import StaticEncoder
17
+ from semsift.search import KeywordSource, Search, VectorSource
18
+ from semsift.store import Field, Item, Store
19
+
20
+ conn = sqlite3.connect(":memory:", isolation_level=None)
21
+ store = Store(conn, "notes", [Field("source", "text")],
22
+ encoder=StaticEncoder("minishlab/potion-base-8M"))
23
+
24
+ items = [Item(1, "Rent is due on the first of the month.", {"source": "lease.md"}),
25
+ Item(2, "The boiler service is booked for Tuesday.", {"source": "home.md"})]
26
+ vectors = store.embed(items) # encode before opening the transaction
27
+ conn.execute("BEGIN")
28
+ store.upsert(items, vectors) # the store never commits; you do
29
+ conn.execute("COMMIT")
30
+
31
+ search = Search(store, sources=[VectorSource(store), KeywordSource(store)],
32
+ citation=("source",))
33
+ for hit in search.run("when is rent due", k=1).hits:
34
+ print(hit.citation["source"], hit.text)
35
+ ```
36
+
37
+ Install extras for what you use: `static` (model2vec encoders), `onnx`,
38
+ `webgpu`, `tree-sitter` (syntax-aware chunking).
39
+
40
+ ## Development
41
+
42
+ ```
43
+ uv venv .venv && uv pip install --python .venv/bin/python -e ".[static,onnx,tree-sitter]"
44
+ HF_HUB_OFFLINE=1 .venv/bin/python -m unittest discover -s tests -t .
45
+ .venv/bin/python benchmarks/run.py
46
+ ```
47
+
48
+ To use an unreleased semsift from another project, depend on the
49
+ checkout (`uv add --editable ../semsift`, or `pip install -e ../semsift`),
50
+ or build it (`uv build`) and install the wheel from `dist/`.
51
+
52
+ ## Releasing
53
+
54
+ The version lives only in `semsift.__version__`. Bump it, commit, and push
55
+ a matching tag (`v0.0.1`). The release workflow runs the tests, builds,
56
+ refuses to publish unless the tag, `__version__`, the sdist and the wheel
57
+ agree, publishes to PyPI through trusted publishing, and creates the
58
+ GitHub release. PyPI must list this repository's `release.yml` as a
59
+ trusted publisher for the `semsift` project.
@@ -0,0 +1,36 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "semsift"
7
+ dynamic = ["version"]
8
+ description = "Building blocks for hybrid keyword and vector retrieval over SQLite"
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ dependencies = [
14
+ "numpy>=1.26",
15
+ "vicinity>=0.4.6",
16
+ ]
17
+
18
+ [project.optional-dependencies]
19
+ static = ["model2vec>=0.9", "huggingface-hub>=0.23"]
20
+ tree-sitter = ["tree-sitter-language-pack>=1.20"]
21
+ onnx = ["onnxruntime>=1.17", "huggingface-hub>=0.23", "tokenizers>=0.15"]
22
+ webgpu = [
23
+ "onnxruntime>=1.24.4",
24
+ "onnxruntime-ep-webgpu",
25
+ "huggingface-hub>=0.23",
26
+ "tokenizers>=0.15",
27
+ ]
28
+
29
+ [project.urls]
30
+ Source = "https://github.com/nitkrar/semsift"
31
+
32
+ [tool.setuptools.dynamic]
33
+ version = { attr = "semsift.__version__" }
34
+
35
+ [tool.setuptools.packages.find]
36
+ where = ["src"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,3 @@
1
+ """Building blocks for hybrid keyword and vector retrieval."""
2
+
3
+ __version__ = "0.0.1"
@@ -0,0 +1,5 @@
1
+ """Splitting text into chunks for the store."""
2
+
3
+ from .chunkers import Chunk, TextChunker, TreeSitterChunker
4
+
5
+ __all__ = ["Chunk", "TextChunker", "TreeSitterChunker"]
@@ -0,0 +1,206 @@
1
+ """Splitting text into chunks whose spans double as citation spans."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from bisect import bisect_right
6
+ from dataclasses import dataclass
7
+
8
+
9
+ @dataclass(frozen=True)
10
+ class Chunk:
11
+ """`text` is exactly `source[start:end]`; lines are 1-based and inclusive.
12
+
13
+ `context` is the enclosing definitions (e.g. a class name) and
14
+ `symbols` the names defined inside, when the chunker knows them.
15
+ """
16
+
17
+ text: str
18
+ start: int
19
+ end: int
20
+ start_line: int
21
+ end_line: int
22
+ context: tuple[str, ...] = ()
23
+ symbols: tuple[str, ...] = ()
24
+
25
+
26
+ def _build(text: str, spans: list[tuple[int, int]], min_chars: int,
27
+ extras: dict[int, tuple[tuple[str, ...], tuple[str, ...]]] | None = None) -> list[Chunk]:
28
+ newlines = [i for i, ch in enumerate(text)
29
+ if ch == "\n" or (ch == "\r" and (i + 1 == len(text) or text[i + 1] != "\n"))]
30
+ out = []
31
+ for start, end in spans:
32
+ body = text[start:end]
33
+ if len(body.strip()) < min_chars:
34
+ continue
35
+ context, symbols = (extras or {}).get(start, ((), ()))
36
+ out.append(Chunk(body, start, end, bisect_right(newlines, start - 1) + 1,
37
+ bisect_right(newlines, max(start, end - 1) - 1) + 1,
38
+ context, symbols))
39
+ return out
40
+
41
+
42
+ def _check(max_chars: int, min_chars: int) -> None:
43
+ if max_chars <= 0:
44
+ raise ValueError("max_chars must be positive")
45
+ if not 0 <= min_chars <= max_chars:
46
+ raise ValueError("min_chars must be between 0 and max_chars")
47
+
48
+
49
+ class TextChunker:
50
+ """Line-based windows of at most `max_chars`, split at natural breaks.
51
+
52
+ With `markdown`, an ATX heading outside a fenced block always starts a
53
+ chunk; turn it off for text where `#` begins a comment. After a blank
54
+ line, a chunk at least half full ends before the next paragraph. A
55
+ line longer than `max_chars` is cut into pieces. Chunks whose stripped
56
+ text is shorter than `min_chars` are dropped, so the spans tile the
57
+ text only when `min_chars` is 0.
58
+ """
59
+
60
+ def __init__(self, max_chars: int = 750, min_chars: int = 1, *,
61
+ markdown: bool = True) -> None:
62
+ _check(max_chars, min_chars)
63
+ self.max_chars, self.min_chars, self.markdown = max_chars, min_chars, markdown
64
+
65
+ def spans(self, text: str) -> list[tuple[int, int]]:
66
+ spans: list[tuple[int, int]] = []
67
+ start = pos = 0
68
+ after_blank = False
69
+ fence: tuple[str, int] | None = None
70
+ for line in text.splitlines(keepends=True):
71
+ size = pos - start
72
+ blank = not line.strip()
73
+ marker = _markdown_fence(line)
74
+ heading = self.markdown and fence is None and _markdown_heading(line)
75
+ if size and (heading
76
+ or (after_blank and not blank and size >= self.max_chars // 2)
77
+ or size + len(line) > self.max_chars):
78
+ spans.append((start, pos))
79
+ start = pos
80
+ if len(line) > self.max_chars:
81
+ for cut in range(pos, pos + len(line), self.max_chars):
82
+ spans.append((cut, min(cut + self.max_chars, pos + len(line))))
83
+ start = pos + len(line)
84
+ pos += len(line)
85
+ after_blank = blank
86
+ if marker is not None:
87
+ char, width, rest = marker
88
+ if fence is None:
89
+ fence = (char, width)
90
+ elif char == fence[0] and width >= fence[1] and not rest.strip():
91
+ fence = None
92
+ if pos > start:
93
+ spans.append((start, pos))
94
+ return spans
95
+
96
+ def chunk(self, text: str) -> list[Chunk]:
97
+ return _build(text, self.spans(text), self.min_chars)
98
+
99
+
100
+ class TreeSitterChunker:
101
+ """Syntax-aligned chunks from tree-sitter-language-pack's chunker.
102
+
103
+ Needs the `tree-sitter` extra. The language pack downloads a
104
+ grammar the first time a language is parsed. A language it does not
105
+ know, a grammar it cannot download, or a failure to parse falls back
106
+ to `fallback`: a TextChunker of the same bounds that does not read
107
+ markdown, by default. So does a source over `max_source_bytes`, or a
108
+ parse that runs past `parse_timeout_ms`, so one generated or minified
109
+ file cannot stall indexing.
110
+ """
111
+
112
+ def __init__(self, max_chars: int = 750, min_chars: int = 1,
113
+ fallback: TextChunker | None = None, *,
114
+ max_source_bytes: int = 5_000_000, parse_timeout_ms: int = 5_000) -> None:
115
+ _check(max_chars, min_chars)
116
+ if (isinstance(max_source_bytes, bool)
117
+ or not isinstance(max_source_bytes, int) or max_source_bytes <= 0
118
+ or isinstance(parse_timeout_ms, bool)
119
+ or not isinstance(parse_timeout_ms, int) or parse_timeout_ms <= 0):
120
+ raise ValueError("max_source_bytes and parse_timeout_ms must be positive integers")
121
+ self.max_chars, self.min_chars = max_chars, min_chars
122
+ self.max_source_bytes, self.parse_timeout_ms = max_source_bytes, parse_timeout_ms
123
+ self.fallback = fallback or TextChunker(max_chars, min_chars, markdown=False)
124
+
125
+ def supports(self, language: str) -> bool:
126
+ """Whether the pack knows `language`; its grammar may still need a download."""
127
+ import tree_sitter_language_pack as pack
128
+
129
+ return language in pack.manifest_languages()
130
+
131
+ def chunk(self, text: str, language: str) -> list[Chunk]:
132
+ import tree_sitter_language_pack as pack
133
+
134
+ if not text.strip():
135
+ return self.fallback.chunk(text)
136
+ if not self.supports(language) or len(text.encode()) > self.max_source_bytes:
137
+ return self.fallback.chunk(text)
138
+ try:
139
+ result = pack.process(text, pack.ProcessConfig(
140
+ language=language, structure=False, imports=False, exports=False,
141
+ chunk_max_size=self.max_chars, max_source_bytes=self.max_source_bytes,
142
+ parse_timeout_ms=self.parse_timeout_ms))
143
+ except pack.Error:
144
+ return self.fallback.chunk(text)
145
+ pieces = sorted(result.chunks, key=lambda c: c.start_byte)
146
+ if not pieces:
147
+ return self.fallback.chunk(text)
148
+ data = text.encode()
149
+ # Chunks are cut at their start bytes, so they tile the text even
150
+ # if the pack leaves a gap; each start is moved back to a
151
+ # character boundary before converting to a character offset.
152
+ starts = sorted({0} | {_char_start(data, c.start_byte) for c in pieces})
153
+ chars = _char_offsets(data, starts)
154
+ extras = {}
155
+ for c in pieces:
156
+ meta = c.metadata
157
+ if meta is not None:
158
+ key = chars[_char_start(data, c.start_byte)]
159
+ extras.setdefault(key, (tuple(meta.context_path), tuple(meta.symbols_defined)))
160
+ bounds = [chars[b] for b in starts] + [len(text)]
161
+ spans = [(a, b) for a, b in zip(bounds, bounds[1:]) if b > a]
162
+ if any(end - start > self.max_chars for start, end in spans):
163
+ return self.fallback.chunk(text)
164
+ return _build(text, spans, self.min_chars, extras)
165
+
166
+
167
+ def _char_start(data: bytes, offset: int) -> int:
168
+ """`offset` moved back to the first byte of the UTF-8 character it is in."""
169
+ offset = min(max(offset, 0), len(data))
170
+ while 0 < offset < len(data) and data[offset] & 0xC0 == 0x80:
171
+ offset -= 1
172
+ return offset
173
+
174
+
175
+ def _char_offsets(data: bytes, byte_offsets: list[int]) -> dict[int, int]:
176
+ """Character offset of each sorted byte offset, which must be on a boundary."""
177
+ out, chars, prev = {}, 0, 0
178
+ for b in byte_offsets:
179
+ chars += len(data[prev:b].decode())
180
+ out[b] = chars
181
+ prev = b
182
+ return out
183
+
184
+
185
+ def _markdown_heading(line: str) -> bool:
186
+ body = line.rstrip("\r\n")
187
+ stripped = body.lstrip(" ")
188
+ if len(body) - len(stripped) > 3:
189
+ return False
190
+ width = len(stripped) - len(stripped.lstrip("#"))
191
+ return 1 <= width <= 6 and (width == len(stripped) or stripped[width] in " \t")
192
+
193
+
194
+ def _markdown_fence(line: str) -> tuple[str, int, str] | None:
195
+ body = line.rstrip("\r\n")
196
+ stripped = body.lstrip(" ")
197
+ if len(body) - len(stripped) > 3 or not stripped or stripped[0] not in "`~":
198
+ return None
199
+ char = stripped[0]
200
+ width = len(stripped) - len(stripped.lstrip(char))
201
+ if width < 3:
202
+ return None
203
+ rest = stripped[width:]
204
+ if char == "`" and "`" in rest:
205
+ return None
206
+ return char, width, rest
@@ -0,0 +1,12 @@
1
+ """Text to vectors.
2
+
3
+ Imports nothing else from semsift, so it can become its own package.
4
+ """
5
+
6
+ from .backends import (Encoder, FakeEncoder, HttpEncoder, OnnxEncoder,
7
+ StaticEncoder)
8
+ from .policy import resolve_prefix
9
+ from .space import VectorSpace
10
+
11
+ __all__ = ["Encoder", "FakeEncoder", "HttpEncoder", "OnnxEncoder",
12
+ "StaticEncoder", "VectorSpace", "resolve_prefix"]