fastembed-gpu 0.4.0__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/PKG-INFO +2 -3
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/sparse/bm25.py +3 -4
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/sparse/bm42.py +4 -4
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/pyproject.toml +2 -3
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/LICENSE +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/README.md +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/__init__.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/common/__init__.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/common/model_management.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/common/onnx_model.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/common/preprocessor_utils.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/common/types.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/common/utils.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/image/__init__.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/image/image_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/image/image_embedding_base.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/image/onnx_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/image/onnx_image_model.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/image/transform/functional.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/image/transform/operators.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/late_interaction/__init__.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/late_interaction/colbert.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/late_interaction/late_interaction_embedding_base.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/late_interaction/late_interaction_text_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/parallel_processor.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/rerank/cross_encoder/__init__.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/rerank/cross_encoder/onnx_text_cross_encoder.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/rerank/cross_encoder/onnx_text_model.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/rerank/cross_encoder/text_cross_encoder.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/rerank/cross_encoder/text_cross_encoder_base.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/sparse/__init__.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/sparse/sparse_embedding_base.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/sparse/sparse_text_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/sparse/splade_pp.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/sparse/utils/tokenizer.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/__init__.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/clip_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/e5_onnx_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/onnx_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/onnx_text_model.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/pooled_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/pooled_normalized_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/text_embedding.py +0 -0
- {fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/text/text_embedding_base.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastembed-gpu
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: Fast, light, accurate library built for retrieval embedding generation
|
|
5
5
|
Home-page: https://github.com/qdrant/fastembed
|
|
6
6
|
License: Apache License
|
|
@@ -15,7 +15,6 @@ Classifier: Programming Language :: Python :: 3.9
|
|
|
15
15
|
Classifier: Programming Language :: Python :: 3.10
|
|
16
16
|
Classifier: Programming Language :: Python :: 3.11
|
|
17
17
|
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
-
Requires-Dist: PyStemmer (>=2.2.0,<3.0.0)
|
|
19
18
|
Requires-Dist: huggingface-hub (>=0.20,<1.0)
|
|
20
19
|
Requires-Dist: loguru (>=0.7.2,<0.8.0)
|
|
21
20
|
Requires-Dist: mmh3 (>=4.1.0,<5.0.0)
|
|
@@ -23,8 +22,8 @@ Requires-Dist: numpy (>=1.21,<2) ; python_version < "3.12"
|
|
|
23
22
|
Requires-Dist: numpy (>=1.26,<2) ; python_version >= "3.12"
|
|
24
23
|
Requires-Dist: onnxruntime-gpu (>=1.17.0,<2.0.0)
|
|
25
24
|
Requires-Dist: pillow (>=10.3.0,<11.0.0)
|
|
25
|
+
Requires-Dist: py-rust-stemmers (>=0.1.0,<0.2.0)
|
|
26
26
|
Requires-Dist: requests (>=2.31,<3.0)
|
|
27
|
-
Requires-Dist: snowballstemmer (>=2.2.0,<3.0.0)
|
|
28
27
|
Requires-Dist: tokenizers (>=0.15,<1.0)
|
|
29
28
|
Requires-Dist: tqdm (>=4.66,<5.0)
|
|
30
29
|
Project-URL: Repository, https://github.com/qdrant/fastembed
|
|
@@ -6,8 +6,7 @@ from typing import Any, Dict, Iterable, List, Optional, Tuple, Type, Union
|
|
|
6
6
|
|
|
7
7
|
import mmh3
|
|
8
8
|
import numpy as np
|
|
9
|
-
from
|
|
10
|
-
|
|
9
|
+
from py_rust_stemmers import SnowballStemmer
|
|
11
10
|
from fastembed.common.utils import (
|
|
12
11
|
define_cache_dir,
|
|
13
12
|
iter_batch,
|
|
@@ -130,7 +129,7 @@ class Bm25(SparseTextEmbeddingBase):
|
|
|
130
129
|
self.punctuation = set(get_all_punctuation())
|
|
131
130
|
self.stopwords = set(self._load_stopwords(self._model_dir, self.language))
|
|
132
131
|
|
|
133
|
-
self.stemmer =
|
|
132
|
+
self.stemmer = SnowballStemmer(language)
|
|
134
133
|
self.tokenizer = SimpleTokenizer
|
|
135
134
|
|
|
136
135
|
@classmethod
|
|
@@ -235,7 +234,7 @@ class Bm25(SparseTextEmbeddingBase):
|
|
|
235
234
|
if len(token) > self.token_max_length:
|
|
236
235
|
continue
|
|
237
236
|
|
|
238
|
-
stemmed_token = self.stemmer.
|
|
237
|
+
stemmed_token = self.stemmer.stem_word(token.lower())
|
|
239
238
|
|
|
240
239
|
if stemmed_token:
|
|
241
240
|
stemmed_tokens.append(stemmed_token)
|
|
@@ -5,7 +5,7 @@ from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple, Type, U
|
|
|
5
5
|
|
|
6
6
|
import mmh3
|
|
7
7
|
import numpy as np
|
|
8
|
-
from
|
|
8
|
+
from py_rust_stemmers import SnowballStemmer
|
|
9
9
|
|
|
10
10
|
from fastembed.common import OnnxProvider
|
|
11
11
|
from fastembed.common.onnx_model import OnnxOutputContext
|
|
@@ -119,8 +119,8 @@ class Bm42(SparseTextEmbeddingBase, OnnxTextModel[SparseEmbedding]):
|
|
|
119
119
|
self.special_tokens = set()
|
|
120
120
|
self.special_tokens_ids = set()
|
|
121
121
|
self.punctuation = set(string.punctuation)
|
|
122
|
-
self.stopwords = set()
|
|
123
|
-
self.stemmer =
|
|
122
|
+
self.stopwords = set(self._load_stopwords(self._model_dir))
|
|
123
|
+
self.stemmer = SnowballStemmer(MODEL_TO_LANGUAGE[model_name])
|
|
124
124
|
self.alpha = alpha
|
|
125
125
|
|
|
126
126
|
if not self.lazy_load:
|
|
@@ -152,7 +152,7 @@ class Bm42(SparseTextEmbeddingBase, OnnxTextModel[SparseEmbedding]):
|
|
|
152
152
|
def _stem_pair_tokens(self, tokens: List[Tuple[str, Any]]) -> List[Tuple[str, Any]]:
|
|
153
153
|
result = []
|
|
154
154
|
for token, value in tokens:
|
|
155
|
-
processed_token = self.stemmer.
|
|
155
|
+
processed_token = self.stemmer.stem_word(token)
|
|
156
156
|
result.append((processed_token, value))
|
|
157
157
|
return result
|
|
158
158
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "fastembed-gpu"
|
|
3
|
-
version = "0.4.
|
|
3
|
+
version = "0.4.1"
|
|
4
4
|
description = "Fast, light, accurate library built for retrieval embedding generation"
|
|
5
5
|
authors = ["Qdrant Team <info@qdrant.tech>", "NirantK <nirant.bits@gmail.com>"]
|
|
6
6
|
license = "Apache License"
|
|
@@ -23,9 +23,8 @@ numpy = [
|
|
|
23
23
|
{ version = ">=1.26, <2", python = ">=3.12" }
|
|
24
24
|
]
|
|
25
25
|
pillow = "^10.3.0"
|
|
26
|
-
snowballstemmer = "^2.2.0"
|
|
27
|
-
PyStemmer = "^2.2.0"
|
|
28
26
|
mmh3 = "^4.1.0"
|
|
27
|
+
py-rust-stemmers = "^0.1.0"
|
|
29
28
|
|
|
30
29
|
[tool.poetry.group.dev.dependencies]
|
|
31
30
|
pytest = "^7.4.2"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/rerank/cross_encoder/onnx_text_model.py
RENAMED
|
File without changes
|
{fastembed_gpu-0.4.0 → fastembed_gpu-0.4.1}/fastembed/rerank/cross_encoder/text_cross_encoder.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|