semsift 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- semsift-0.0.1/LICENSE +21 -0
- semsift-0.0.1/PKG-INFO +86 -0
- semsift-0.0.1/README.md +59 -0
- semsift-0.0.1/pyproject.toml +36 -0
- semsift-0.0.1/setup.cfg +4 -0
- semsift-0.0.1/src/semsift/__init__.py +3 -0
- semsift-0.0.1/src/semsift/chunk/__init__.py +5 -0
- semsift-0.0.1/src/semsift/chunk/chunkers.py +206 -0
- semsift-0.0.1/src/semsift/embed/__init__.py +12 -0
- semsift-0.0.1/src/semsift/embed/backends.py +433 -0
- semsift-0.0.1/src/semsift/embed/policy.py +89 -0
- semsift-0.0.1/src/semsift/embed/space.py +43 -0
- semsift-0.0.1/src/semsift/evals/__init__.py +7 -0
- semsift-0.0.1/src/semsift/evals/harness.py +233 -0
- semsift-0.0.1/src/semsift/fuse/__init__.py +7 -0
- semsift-0.0.1/src/semsift/fuse/lists.py +177 -0
- semsift-0.0.1/src/semsift/rerank/__init__.py +5 -0
- semsift-0.0.1/src/semsift/rerank/rerankers.py +241 -0
- semsift-0.0.1/src/semsift/search/__init__.py +5 -0
- semsift-0.0.1/src/semsift/search/composer.py +205 -0
- semsift-0.0.1/src/semsift/store/__init__.py +8 -0
- semsift-0.0.1/src/semsift/store/codec.py +48 -0
- semsift-0.0.1/src/semsift/store/filters.py +231 -0
- semsift-0.0.1/src/semsift/store/index.py +82 -0
- semsift-0.0.1/src/semsift/store/store.py +621 -0
- semsift-0.0.1/src/semsift.egg-info/PKG-INFO +86 -0
- semsift-0.0.1/src/semsift.egg-info/SOURCES.txt +29 -0
- semsift-0.0.1/src/semsift.egg-info/dependency_links.txt +1 -0
- semsift-0.0.1/src/semsift.egg-info/requires.txt +20 -0
- semsift-0.0.1/src/semsift.egg-info/top_level.txt +1 -0
- semsift-0.0.1/tests/test_boundary.py +69 -0
semsift-0.0.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Nitin Kumar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALING IN THE
|
|
21
|
+
SOFTWARE.
|
semsift-0.0.1/PKG-INFO
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: semsift
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Building blocks for hybrid keyword and vector retrieval over SQLite
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Project-URL: Source, https://github.com/nitkrar/semsift
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: numpy>=1.26
|
|
11
|
+
Requires-Dist: vicinity>=0.4.6
|
|
12
|
+
Provides-Extra: static
|
|
13
|
+
Requires-Dist: model2vec>=0.9; extra == "static"
|
|
14
|
+
Requires-Dist: huggingface-hub>=0.23; extra == "static"
|
|
15
|
+
Provides-Extra: tree-sitter
|
|
16
|
+
Requires-Dist: tree-sitter-language-pack>=1.20; extra == "tree-sitter"
|
|
17
|
+
Provides-Extra: onnx
|
|
18
|
+
Requires-Dist: onnxruntime>=1.17; extra == "onnx"
|
|
19
|
+
Requires-Dist: huggingface-hub>=0.23; extra == "onnx"
|
|
20
|
+
Requires-Dist: tokenizers>=0.15; extra == "onnx"
|
|
21
|
+
Provides-Extra: webgpu
|
|
22
|
+
Requires-Dist: onnxruntime>=1.24.4; extra == "webgpu"
|
|
23
|
+
Requires-Dist: onnxruntime-ep-webgpu; extra == "webgpu"
|
|
24
|
+
Requires-Dist: huggingface-hub>=0.23; extra == "webgpu"
|
|
25
|
+
Requires-Dist: tokenizers>=0.15; extra == "webgpu"
|
|
26
|
+
Dynamic: license-file
|
|
27
|
+
|
|
28
|
+
# semsift
|
|
29
|
+
|
|
30
|
+
Building blocks for hybrid retrieval: chunk text, embed it, store vectors
|
|
31
|
+
and a keyword index together in SQLite or memory, search both, fuse the
|
|
32
|
+
ranked results and rerank them. It retrieves and ranks; it never
|
|
33
|
+
generates text.
|
|
34
|
+
|
|
35
|
+
**Status:** Implemented, not yet released. See
|
|
36
|
+
[docs/design.md](docs/design.md) for the blocks and their contracts, and
|
|
37
|
+
[docs/features.md](docs/features.md) for what was taken from other
|
|
38
|
+
projects.
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
import sqlite3
|
|
42
|
+
|
|
43
|
+
from semsift.embed import StaticEncoder
|
|
44
|
+
from semsift.search import KeywordSource, Search, VectorSource
|
|
45
|
+
from semsift.store import Field, Item, Store
|
|
46
|
+
|
|
47
|
+
conn = sqlite3.connect(":memory:", isolation_level=None)
|
|
48
|
+
store = Store(conn, "notes", [Field("source", "text")],
|
|
49
|
+
encoder=StaticEncoder("minishlab/potion-base-8M"))
|
|
50
|
+
|
|
51
|
+
items = [Item(1, "Rent is due on the first of the month.", {"source": "lease.md"}),
|
|
52
|
+
Item(2, "The boiler service is booked for Tuesday.", {"source": "home.md"})]
|
|
53
|
+
vectors = store.embed(items) # encode before opening the transaction
|
|
54
|
+
conn.execute("BEGIN")
|
|
55
|
+
store.upsert(items, vectors) # the store never commits; you do
|
|
56
|
+
conn.execute("COMMIT")
|
|
57
|
+
|
|
58
|
+
search = Search(store, sources=[VectorSource(store), KeywordSource(store)],
|
|
59
|
+
citation=("source",))
|
|
60
|
+
for hit in search.run("when is rent due", k=1).hits:
|
|
61
|
+
print(hit.citation["source"], hit.text)
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Install extras for what you use: `static` (model2vec encoders), `onnx`,
|
|
65
|
+
`webgpu`, `tree-sitter` (syntax-aware chunking).
|
|
66
|
+
|
|
67
|
+
## Development
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
uv venv .venv && uv pip install --python .venv/bin/python -e ".[static,onnx,tree-sitter]"
|
|
71
|
+
HF_HUB_OFFLINE=1 .venv/bin/python -m unittest discover -s tests -t .
|
|
72
|
+
.venv/bin/python benchmarks/run.py
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
To use an unreleased semsift from another project, depend on the
|
|
76
|
+
checkout (`uv add --editable ../semsift`, or `pip install -e ../semsift`),
|
|
77
|
+
or build it (`uv build`) and install the wheel from `dist/`.
|
|
78
|
+
|
|
79
|
+
## Releasing
|
|
80
|
+
|
|
81
|
+
The version lives only in `semsift.__version__`. Bump it, commit, and push
|
|
82
|
+
a matching tag (`v0.0.1`). The release workflow runs the tests, builds,
|
|
83
|
+
refuses to publish unless the tag, `__version__`, the sdist and the wheel
|
|
84
|
+
agree, publishes to PyPI through trusted publishing, and creates the
|
|
85
|
+
GitHub release. PyPI must list this repository's `release.yml` as a
|
|
86
|
+
trusted publisher for the `semsift` project.
|
semsift-0.0.1/README.md
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# semsift
|
|
2
|
+
|
|
3
|
+
Building blocks for hybrid retrieval: chunk text, embed it, store vectors
|
|
4
|
+
and a keyword index together in SQLite or memory, search both, fuse the
|
|
5
|
+
ranked results and rerank them. It retrieves and ranks; it never
|
|
6
|
+
generates text.
|
|
7
|
+
|
|
8
|
+
**Status:** Implemented, not yet released. See
|
|
9
|
+
[docs/design.md](docs/design.md) for the blocks and their contracts, and
|
|
10
|
+
[docs/features.md](docs/features.md) for what was taken from other
|
|
11
|
+
projects.
|
|
12
|
+
|
|
13
|
+
```python
|
|
14
|
+
import sqlite3
|
|
15
|
+
|
|
16
|
+
from semsift.embed import StaticEncoder
|
|
17
|
+
from semsift.search import KeywordSource, Search, VectorSource
|
|
18
|
+
from semsift.store import Field, Item, Store
|
|
19
|
+
|
|
20
|
+
conn = sqlite3.connect(":memory:", isolation_level=None)
|
|
21
|
+
store = Store(conn, "notes", [Field("source", "text")],
|
|
22
|
+
encoder=StaticEncoder("minishlab/potion-base-8M"))
|
|
23
|
+
|
|
24
|
+
items = [Item(1, "Rent is due on the first of the month.", {"source": "lease.md"}),
|
|
25
|
+
Item(2, "The boiler service is booked for Tuesday.", {"source": "home.md"})]
|
|
26
|
+
vectors = store.embed(items) # encode before opening the transaction
|
|
27
|
+
conn.execute("BEGIN")
|
|
28
|
+
store.upsert(items, vectors) # the store never commits; you do
|
|
29
|
+
conn.execute("COMMIT")
|
|
30
|
+
|
|
31
|
+
search = Search(store, sources=[VectorSource(store), KeywordSource(store)],
|
|
32
|
+
citation=("source",))
|
|
33
|
+
for hit in search.run("when is rent due", k=1).hits:
|
|
34
|
+
print(hit.citation["source"], hit.text)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Install extras for what you use: `static` (model2vec encoders), `onnx`,
|
|
38
|
+
`webgpu`, `tree-sitter` (syntax-aware chunking).
|
|
39
|
+
|
|
40
|
+
## Development
|
|
41
|
+
|
|
42
|
+
```
|
|
43
|
+
uv venv .venv && uv pip install --python .venv/bin/python -e ".[static,onnx,tree-sitter]"
|
|
44
|
+
HF_HUB_OFFLINE=1 .venv/bin/python -m unittest discover -s tests -t .
|
|
45
|
+
.venv/bin/python benchmarks/run.py
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
To use an unreleased semsift from another project, depend on the
|
|
49
|
+
checkout (`uv add --editable ../semsift`, or `pip install -e ../semsift`),
|
|
50
|
+
or build it (`uv build`) and install the wheel from `dist/`.
|
|
51
|
+
|
|
52
|
+
## Releasing
|
|
53
|
+
|
|
54
|
+
The version lives only in `semsift.__version__`. Bump it, commit, and push
|
|
55
|
+
a matching tag (`v0.0.1`). The release workflow runs the tests, builds,
|
|
56
|
+
refuses to publish unless the tag, `__version__`, the sdist and the wheel
|
|
57
|
+
agree, publishes to PyPI through trusted publishing, and creates the
|
|
58
|
+
GitHub release. PyPI must list this repository's `release.yml` as a
|
|
59
|
+
trusted publisher for the `semsift` project.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "semsift"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Building blocks for hybrid keyword and vector retrieval over SQLite"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
dependencies = [
|
|
14
|
+
"numpy>=1.26",
|
|
15
|
+
"vicinity>=0.4.6",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.optional-dependencies]
|
|
19
|
+
static = ["model2vec>=0.9", "huggingface-hub>=0.23"]
|
|
20
|
+
tree-sitter = ["tree-sitter-language-pack>=1.20"]
|
|
21
|
+
onnx = ["onnxruntime>=1.17", "huggingface-hub>=0.23", "tokenizers>=0.15"]
|
|
22
|
+
webgpu = [
|
|
23
|
+
"onnxruntime>=1.24.4",
|
|
24
|
+
"onnxruntime-ep-webgpu",
|
|
25
|
+
"huggingface-hub>=0.23",
|
|
26
|
+
"tokenizers>=0.15",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Source = "https://github.com/nitkrar/semsift"
|
|
31
|
+
|
|
32
|
+
[tool.setuptools.dynamic]
|
|
33
|
+
version = { attr = "semsift.__version__" }
|
|
34
|
+
|
|
35
|
+
[tool.setuptools.packages.find]
|
|
36
|
+
where = ["src"]
|
semsift-0.0.1/setup.cfg
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Splitting text into chunks whose spans double as citation spans."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from bisect import bisect_right
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True)
|
|
10
|
+
class Chunk:
|
|
11
|
+
"""`text` is exactly `source[start:end]`; lines are 1-based and inclusive.
|
|
12
|
+
|
|
13
|
+
`context` is the enclosing definitions (e.g. a class name) and
|
|
14
|
+
`symbols` the names defined inside, when the chunker knows them.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
text: str
|
|
18
|
+
start: int
|
|
19
|
+
end: int
|
|
20
|
+
start_line: int
|
|
21
|
+
end_line: int
|
|
22
|
+
context: tuple[str, ...] = ()
|
|
23
|
+
symbols: tuple[str, ...] = ()
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _build(text: str, spans: list[tuple[int, int]], min_chars: int,
|
|
27
|
+
extras: dict[int, tuple[tuple[str, ...], tuple[str, ...]]] | None = None) -> list[Chunk]:
|
|
28
|
+
newlines = [i for i, ch in enumerate(text)
|
|
29
|
+
if ch == "\n" or (ch == "\r" and (i + 1 == len(text) or text[i + 1] != "\n"))]
|
|
30
|
+
out = []
|
|
31
|
+
for start, end in spans:
|
|
32
|
+
body = text[start:end]
|
|
33
|
+
if len(body.strip()) < min_chars:
|
|
34
|
+
continue
|
|
35
|
+
context, symbols = (extras or {}).get(start, ((), ()))
|
|
36
|
+
out.append(Chunk(body, start, end, bisect_right(newlines, start - 1) + 1,
|
|
37
|
+
bisect_right(newlines, max(start, end - 1) - 1) + 1,
|
|
38
|
+
context, symbols))
|
|
39
|
+
return out
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _check(max_chars: int, min_chars: int) -> None:
|
|
43
|
+
if max_chars <= 0:
|
|
44
|
+
raise ValueError("max_chars must be positive")
|
|
45
|
+
if not 0 <= min_chars <= max_chars:
|
|
46
|
+
raise ValueError("min_chars must be between 0 and max_chars")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class TextChunker:
|
|
50
|
+
"""Line-based windows of at most `max_chars`, split at natural breaks.
|
|
51
|
+
|
|
52
|
+
With `markdown`, an ATX heading outside a fenced block always starts a
|
|
53
|
+
chunk; turn it off for text where `#` begins a comment. After a blank
|
|
54
|
+
line, a chunk at least half full ends before the next paragraph. A
|
|
55
|
+
line longer than `max_chars` is cut into pieces. Chunks whose stripped
|
|
56
|
+
text is shorter than `min_chars` are dropped, so the spans tile the
|
|
57
|
+
text only when `min_chars` is 0.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
def __init__(self, max_chars: int = 750, min_chars: int = 1, *,
|
|
61
|
+
markdown: bool = True) -> None:
|
|
62
|
+
_check(max_chars, min_chars)
|
|
63
|
+
self.max_chars, self.min_chars, self.markdown = max_chars, min_chars, markdown
|
|
64
|
+
|
|
65
|
+
def spans(self, text: str) -> list[tuple[int, int]]:
|
|
66
|
+
spans: list[tuple[int, int]] = []
|
|
67
|
+
start = pos = 0
|
|
68
|
+
after_blank = False
|
|
69
|
+
fence: tuple[str, int] | None = None
|
|
70
|
+
for line in text.splitlines(keepends=True):
|
|
71
|
+
size = pos - start
|
|
72
|
+
blank = not line.strip()
|
|
73
|
+
marker = _markdown_fence(line)
|
|
74
|
+
heading = self.markdown and fence is None and _markdown_heading(line)
|
|
75
|
+
if size and (heading
|
|
76
|
+
or (after_blank and not blank and size >= self.max_chars // 2)
|
|
77
|
+
or size + len(line) > self.max_chars):
|
|
78
|
+
spans.append((start, pos))
|
|
79
|
+
start = pos
|
|
80
|
+
if len(line) > self.max_chars:
|
|
81
|
+
for cut in range(pos, pos + len(line), self.max_chars):
|
|
82
|
+
spans.append((cut, min(cut + self.max_chars, pos + len(line))))
|
|
83
|
+
start = pos + len(line)
|
|
84
|
+
pos += len(line)
|
|
85
|
+
after_blank = blank
|
|
86
|
+
if marker is not None:
|
|
87
|
+
char, width, rest = marker
|
|
88
|
+
if fence is None:
|
|
89
|
+
fence = (char, width)
|
|
90
|
+
elif char == fence[0] and width >= fence[1] and not rest.strip():
|
|
91
|
+
fence = None
|
|
92
|
+
if pos > start:
|
|
93
|
+
spans.append((start, pos))
|
|
94
|
+
return spans
|
|
95
|
+
|
|
96
|
+
def chunk(self, text: str) -> list[Chunk]:
|
|
97
|
+
return _build(text, self.spans(text), self.min_chars)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class TreeSitterChunker:
|
|
101
|
+
"""Syntax-aligned chunks from tree-sitter-language-pack's chunker.
|
|
102
|
+
|
|
103
|
+
Needs the `tree-sitter` extra. The language pack downloads a
|
|
104
|
+
grammar the first time a language is parsed. A language it does not
|
|
105
|
+
know, a grammar it cannot download, or a failure to parse falls back
|
|
106
|
+
to `fallback`: a TextChunker of the same bounds that does not read
|
|
107
|
+
markdown, by default. So does a source over `max_source_bytes`, or a
|
|
108
|
+
parse that runs past `parse_timeout_ms`, so one generated or minified
|
|
109
|
+
file cannot stall indexing.
|
|
110
|
+
"""
|
|
111
|
+
|
|
112
|
+
def __init__(self, max_chars: int = 750, min_chars: int = 1,
|
|
113
|
+
fallback: TextChunker | None = None, *,
|
|
114
|
+
max_source_bytes: int = 5_000_000, parse_timeout_ms: int = 5_000) -> None:
|
|
115
|
+
_check(max_chars, min_chars)
|
|
116
|
+
if (isinstance(max_source_bytes, bool)
|
|
117
|
+
or not isinstance(max_source_bytes, int) or max_source_bytes <= 0
|
|
118
|
+
or isinstance(parse_timeout_ms, bool)
|
|
119
|
+
or not isinstance(parse_timeout_ms, int) or parse_timeout_ms <= 0):
|
|
120
|
+
raise ValueError("max_source_bytes and parse_timeout_ms must be positive integers")
|
|
121
|
+
self.max_chars, self.min_chars = max_chars, min_chars
|
|
122
|
+
self.max_source_bytes, self.parse_timeout_ms = max_source_bytes, parse_timeout_ms
|
|
123
|
+
self.fallback = fallback or TextChunker(max_chars, min_chars, markdown=False)
|
|
124
|
+
|
|
125
|
+
def supports(self, language: str) -> bool:
|
|
126
|
+
"""Whether the pack knows `language`; its grammar may still need a download."""
|
|
127
|
+
import tree_sitter_language_pack as pack
|
|
128
|
+
|
|
129
|
+
return language in pack.manifest_languages()
|
|
130
|
+
|
|
131
|
+
def chunk(self, text: str, language: str) -> list[Chunk]:
|
|
132
|
+
import tree_sitter_language_pack as pack
|
|
133
|
+
|
|
134
|
+
if not text.strip():
|
|
135
|
+
return self.fallback.chunk(text)
|
|
136
|
+
if not self.supports(language) or len(text.encode()) > self.max_source_bytes:
|
|
137
|
+
return self.fallback.chunk(text)
|
|
138
|
+
try:
|
|
139
|
+
result = pack.process(text, pack.ProcessConfig(
|
|
140
|
+
language=language, structure=False, imports=False, exports=False,
|
|
141
|
+
chunk_max_size=self.max_chars, max_source_bytes=self.max_source_bytes,
|
|
142
|
+
parse_timeout_ms=self.parse_timeout_ms))
|
|
143
|
+
except pack.Error:
|
|
144
|
+
return self.fallback.chunk(text)
|
|
145
|
+
pieces = sorted(result.chunks, key=lambda c: c.start_byte)
|
|
146
|
+
if not pieces:
|
|
147
|
+
return self.fallback.chunk(text)
|
|
148
|
+
data = text.encode()
|
|
149
|
+
# Chunks are cut at their start bytes, so they tile the text even
|
|
150
|
+
# if the pack leaves a gap; each start is moved back to a
|
|
151
|
+
# character boundary before converting to a character offset.
|
|
152
|
+
starts = sorted({0} | {_char_start(data, c.start_byte) for c in pieces})
|
|
153
|
+
chars = _char_offsets(data, starts)
|
|
154
|
+
extras = {}
|
|
155
|
+
for c in pieces:
|
|
156
|
+
meta = c.metadata
|
|
157
|
+
if meta is not None:
|
|
158
|
+
key = chars[_char_start(data, c.start_byte)]
|
|
159
|
+
extras.setdefault(key, (tuple(meta.context_path), tuple(meta.symbols_defined)))
|
|
160
|
+
bounds = [chars[b] for b in starts] + [len(text)]
|
|
161
|
+
spans = [(a, b) for a, b in zip(bounds, bounds[1:]) if b > a]
|
|
162
|
+
if any(end - start > self.max_chars for start, end in spans):
|
|
163
|
+
return self.fallback.chunk(text)
|
|
164
|
+
return _build(text, spans, self.min_chars, extras)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _char_start(data: bytes, offset: int) -> int:
|
|
168
|
+
"""`offset` moved back to the first byte of the UTF-8 character it is in."""
|
|
169
|
+
offset = min(max(offset, 0), len(data))
|
|
170
|
+
while 0 < offset < len(data) and data[offset] & 0xC0 == 0x80:
|
|
171
|
+
offset -= 1
|
|
172
|
+
return offset
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _char_offsets(data: bytes, byte_offsets: list[int]) -> dict[int, int]:
|
|
176
|
+
"""Character offset of each sorted byte offset, which must be on a boundary."""
|
|
177
|
+
out, chars, prev = {}, 0, 0
|
|
178
|
+
for b in byte_offsets:
|
|
179
|
+
chars += len(data[prev:b].decode())
|
|
180
|
+
out[b] = chars
|
|
181
|
+
prev = b
|
|
182
|
+
return out
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _markdown_heading(line: str) -> bool:
|
|
186
|
+
body = line.rstrip("\r\n")
|
|
187
|
+
stripped = body.lstrip(" ")
|
|
188
|
+
if len(body) - len(stripped) > 3:
|
|
189
|
+
return False
|
|
190
|
+
width = len(stripped) - len(stripped.lstrip("#"))
|
|
191
|
+
return 1 <= width <= 6 and (width == len(stripped) or stripped[width] in " \t")
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _markdown_fence(line: str) -> tuple[str, int, str] | None:
|
|
195
|
+
body = line.rstrip("\r\n")
|
|
196
|
+
stripped = body.lstrip(" ")
|
|
197
|
+
if len(body) - len(stripped) > 3 or not stripped or stripped[0] not in "`~":
|
|
198
|
+
return None
|
|
199
|
+
char = stripped[0]
|
|
200
|
+
width = len(stripped) - len(stripped.lstrip(char))
|
|
201
|
+
if width < 3:
|
|
202
|
+
return None
|
|
203
|
+
rest = stripped[width:]
|
|
204
|
+
if char == "`" and "`" in rest:
|
|
205
|
+
return None
|
|
206
|
+
return char, width, rest
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Text to vectors.
|
|
2
|
+
|
|
3
|
+
Imports nothing else from semsift, so it can become its own package.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from .backends import (Encoder, FakeEncoder, HttpEncoder, OnnxEncoder,
|
|
7
|
+
StaticEncoder)
|
|
8
|
+
from .policy import resolve_prefix
|
|
9
|
+
from .space import VectorSpace
|
|
10
|
+
|
|
11
|
+
__all__ = ["Encoder", "FakeEncoder", "HttpEncoder", "OnnxEncoder",
|
|
12
|
+
"StaticEncoder", "VectorSpace", "resolve_prefix"]
|