statembed-py 0.1.0__cp313-cp313-win32.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ from .statembed_py import *
2
+
3
+ __doc__ = statembed_py.__doc__
4
+ if hasattr(statembed_py, "__all__"):
5
+ __all__ = statembed_py.__all__
@@ -0,0 +1,31 @@
1
+ # This file is automatically generated by pyo3_stub_gen
2
+ # ruff: noqa: E501, F401, F403, F405
3
+
4
+ import builtins
5
+ import typing
6
+ __all__ = [
7
+ "StaticEmbedding",
8
+ ]
9
+
10
+ @typing.final
11
+ class StaticEmbedding:
12
+ def __new__(cls, model_dir: builtins.str, normalize: builtins.bool = True) -> StaticEmbedding:
13
+ r"""
14
+ Creates a `StaticEmbedding` from a local directory.
15
+
16
+ The directory must contain at least `model.safetensors`. When the
17
+ `tokenizers` feature is enabled, `tokenizer.json` is also required.
18
+
19
+ # Arguments
20
+ * `model_dir` - Path to the model directory.
21
+ * `normalize` - If `True`, output embeddings will be L2-normalized.
22
+ """
23
+ def embed_tokens(self, tokens: typing.Sequence[builtins.int]) -> builtins.list[builtins.float]:
24
+ r"""
25
+ Generates an embedding for a pre-tokenized sequence of token IDs (use whatever tokenization
26
+ libary you prefer to generate tokens).
27
+
28
+ The tensor is loaded lazily on first call. Embeddings are mean-pooled
29
+ and optionally normalized.
30
+ """
31
+
statembed_py/py.typed ADDED
File without changes
@@ -0,0 +1,82 @@
1
+ Metadata-Version: 2.4
2
+ Name: statembed-py
3
+ Version: 0.1.0
4
+ Classifier: Programming Language :: Rust
5
+ Classifier: Programming Language :: Python :: Implementation :: CPython
6
+ Classifier: Programming Language :: Python :: Implementation :: PyPy
7
+ Summary: Fast, lightweight static text embeddings for Python
8
+ License: MIT
9
+ Requires-Python: >=3.10
10
+ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
11
+
12
+ # statembed-py
13
+
14
+ Fast, lightweight static text embeddings.
15
+
16
+ > _Python bindings for the [`statembed`](../statembed/README.md) Rust crate, built with [PyO3](https://pyo3.rs)._
17
+
18
+ `statembed-py` loads a static embedding model (a `model.safetensors` file) and produces mean-pooled, optionally L2-normalized embeddings from token IDs. It does not tokenize text itself — pair it with a tokenizer library such as [`tokenizers`](https://pypi.org/project/tokenizers/).
19
+
20
+ ## Installation
21
+
22
+ ```bash
23
+ # with uv
24
+ uv add statembed-py
25
+ # with pip
26
+ pip install statembed-py
27
+ ```
28
+
29
+ ### Building from source
30
+
31
+ Requires [Rust](https://rustup.rs) and [maturin](https://www.maturin.rs):
32
+
33
+ ```bash
34
+ pip install maturin
35
+ maturin develop --release
36
+ ```
37
+
38
+ ## Usage
39
+
40
+ ```python
41
+ from functools import lru_cache
42
+ from tokenizers import Tokenizer
43
+ from statembed_py import StaticEmbedding
44
+
45
+ @lru_cache(maxsize=1)
46
+ def get_embedding_model() -> StaticEmbedding:
47
+ return StaticEmbedding(model_dir="./my-model") # must contain model.safetensors
48
+
49
+ @lru_cache(maxsize=1)
50
+ def get_tokenizer() -> Tokenizer:
51
+ return Tokenizer.from_file("./my-model/tokenizer.json")
52
+
53
+ def embed(text: str) -> list[float]:
54
+ tokens = get_tokenizer().encode(text).ids
55
+ embedding = get_embedding_model().embed_tokens(tokens)
56
+ return embedding
57
+ ```
58
+
59
+ ## API
60
+
61
+ ### `StaticEmbedding(model_dir, normalize=True)`
62
+
63
+ Loads a model from a local directory containing `model.safetensors`. The tensor is loaded lazily on the first `embed_tokens` call.
64
+
65
+ - `model_dir: str` — path to the model directory.
66
+ - `normalize: bool` — if `True` (default), output embeddings are L2-normalized.
67
+
68
+ ### `embed_tokens(tokens: Sequence[int]) -> list[float]`
69
+
70
+ Mean-pools the embedding rows for the given token IDs into a single fixed-length vector, applying normalization if enabled.
71
+
72
+ Type stubs (`statembed_py.pyi`) are bundled for editor and type-checker support.
73
+
74
+ ## Development
75
+
76
+ - `maturin develop` — build and install the extension into the active virtualenv.
77
+ - `cargo run --bin stub_gen` — regenerate `statembed_py.pyi` after changing the PyO3 bindings.
78
+
79
+ ## License
80
+
81
+ MIT
82
+
@@ -0,0 +1,8 @@
1
+ statembed_py/__init__.py,sha256=xUf6OEOLEKMBK0icUjzP-8H2rfQNDBSBu1Ykwpnfp90,131
2
+ statembed_py/__init__.pyi,sha256=1u8nEnH-cgxK_LNZfS-HKsdcolXw3KXwB6gQmSUktmU,1095
3
+ statembed_py/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ statembed_py/statembed_py.cp313-win32.pyd,sha256=Suvwfonj_mdWn2fJPrK5g5UQU-zLgPIwYcH-Eao6teE,384000
5
+ statembed_py-0.1.0.dist-info/METADATA,sha256=Wux0bJR-D_x6jltW2Tsxh16Qi-uR3sIbJtKX2oNuMnI,2562
6
+ statembed_py-0.1.0.dist-info/WHEEL,sha256=f7XMoWLYOwpeKech8XjAiAS9ZZeczBWGv9Mu3dlODxs,93
7
+ statembed_py-0.1.0.dist-info/sboms/statembed-py.cyclonedx.json,sha256=I8pAGFqjZSlnnwL844xb14W8MvIJfCScFnY154B7sZc,154403
8
+ statembed_py-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: maturin (1.14.1)
3
+ Root-Is-Purelib: false
4
+ Tag: cp313-cp313-win32