e5-large-v2-local 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,25 @@
1
+ Metadata-Version: 2.4
2
+ Name: e5-large-v2-local
3
+ Version: 1.0.0
4
+ Summary: Local E5-large-v2 embedding model with automatic Internet Archive download
5
+ Author: Your Name
6
+ Project-URL: Homepage, https://archive.org/details/e5-large-v2
7
+ Project-URL: Model, https://archive.org/details/e5-large-v2
8
+ Project-URL: Download, https://archive.org/download/e5-large-v2/e5-large-v2.zip
9
+ Requires-Python: >=3.10
10
+ Description-Content-Type: text/markdown
11
+ Requires-Dist: sentence-transformers>=3.0
12
+ Requires-Dist: requests>=2.31
13
+
14
+ # E5-large-v2 Local
15
+
16
+ A lightweight Python package for using the E5-large-v2 embedding model locally.
17
+
18
+ The package itself does not contain the 751 MB model.
19
+
20
+ On first use, the model is downloaded from Internet Archive and cached locally.
21
+
22
+ ## Installation
23
+
24
+ ```bash
25
+ pip install e5-large-v2-local
@@ -0,0 +1,12 @@
1
+ # E5-large-v2 Local
2
+
3
+ A lightweight Python package for using the E5-large-v2 embedding model locally.
4
+
5
+ The package itself does not contain the 751 MB model.
6
+
7
+ On first use, the model is downloaded from Internet Archive and cached locally.
8
+
9
+ ## Installation
10
+
11
+ ```bash
12
+ pip install e5-large-v2-local
@@ -0,0 +1,27 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "e5-large-v2-local"
7
+ version = "1.0.0"
8
+ description = "Local E5-large-v2 embedding model with automatic Internet Archive download"
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+
12
+ dependencies = [
13
+ "sentence-transformers>=3.0",
14
+ "requests>=2.31"
15
+ ]
16
+
17
+ authors = [
18
+ {name = "Your Name"}
19
+ ]
20
+
21
+ [project.urls]
22
+ Homepage = "https://archive.org/details/e5-large-v2"
23
+ Model = "https://archive.org/details/e5-large-v2"
24
+ Download = "https://archive.org/download/e5-large-v2/e5-large-v2.zip"
25
+
26
+ [tool.setuptools.packages.find]
27
+ where = ["src"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,3 @@
1
+ from .model import E5Model
2
+
3
+ __all__ = ["E5Model"]
@@ -0,0 +1,501 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import shutil
5
+ import zipfile
6
+ from pathlib import Path
7
+ from typing import Iterable
8
+
9
+ import requests
10
+
11
+
12
+ MODEL_URL = (
13
+ "https://archive.org/download/"
14
+ "e5-large-v2/e5-large-v2.zip"
15
+ )
16
+
17
+ DEFAULT_CACHE_DIR = (
18
+ Path.home()
19
+ / ".cache"
20
+ / "e5-large-v2"
21
+ )
22
+
23
+ CHUNK_SIZE = 8 * 1024 * 1024
24
+
25
+
26
+ class E5Model:
27
+ """
28
+ Local E5-large-v2 embedding model.
29
+
30
+ The model is downloaded from Internet Archive on first use
31
+ and cached locally.
32
+
33
+ Parameters
34
+ ----------
35
+ model_dir:
36
+ Optional custom directory for the model.
37
+
38
+ device:
39
+ SentenceTransformer device.
40
+ Examples: "cpu", "mps", "cuda".
41
+
42
+ local_files_only:
43
+ If True, never download the model.
44
+
45
+ sha256:
46
+ Optional SHA256 checksum of the ZIP file.
47
+ """
48
+
49
+ def __init__(
50
+ self,
51
+ model_dir: str | Path | None = None,
52
+ device: str | None = None,
53
+ local_files_only: bool = False,
54
+ sha256: str | None = None,
55
+ ):
56
+
57
+ self.model_dir = (
58
+ Path(model_dir).expanduser()
59
+ if model_dir
60
+ else DEFAULT_CACHE_DIR
61
+ )
62
+
63
+ self.device = device
64
+ self.local_files_only = local_files_only
65
+ self.sha256 = sha256
66
+
67
+ self.model_dir.parent.mkdir(
68
+ parents=True,
69
+ exist_ok=True
70
+ )
71
+
72
+ self._ensure_model()
73
+
74
+ from sentence_transformers import SentenceTransformer
75
+
76
+ kwargs = {}
77
+
78
+ if self.device:
79
+ kwargs["device"] = self.device
80
+
81
+ self.model = SentenceTransformer(
82
+ str(self.model_dir),
83
+ **kwargs
84
+ )
85
+
86
+ # ---------------------------------------------------------
87
+ # Model preparation
88
+ # ---------------------------------------------------------
89
+
90
+ def _ensure_model(self) -> None:
91
+
92
+ if self._is_model_directory(
93
+ self.model_dir
94
+ ):
95
+ return
96
+
97
+ if self.local_files_only:
98
+ raise FileNotFoundError(
99
+ "E5-large-v2 model was not found at:\n"
100
+ f"{self.model_dir}\n\n"
101
+ "local_files_only=True prevents downloading."
102
+ )
103
+
104
+ zip_path = (
105
+ self.model_dir.parent
106
+ / "e5-large-v2.zip"
107
+ )
108
+
109
+ temp_zip = (
110
+ self.model_dir.parent
111
+ / "e5-large-v2.zip.part"
112
+ )
113
+
114
+ extract_dir = (
115
+ self.model_dir.parent
116
+ / "e5-large-v2-extracted"
117
+ )
118
+
119
+ print()
120
+ print("E5-large-v2 model not found.")
121
+ print("Downloading model from Internet Archive.")
122
+ print()
123
+
124
+ try:
125
+
126
+ self._download(
127
+ MODEL_URL,
128
+ temp_zip
129
+ )
130
+
131
+ print("Download completed.")
132
+
133
+ if self.sha256:
134
+ print("Checking SHA256...")
135
+
136
+ actual = self._sha256(
137
+ temp_zip
138
+ )
139
+
140
+ if actual.lower() != self.sha256.lower():
141
+ raise RuntimeError(
142
+ "SHA256 checksum mismatch.\n"
143
+ f"Expected: {self.sha256}\n"
144
+ f"Actual: {actual}"
145
+ )
146
+
147
+ print("SHA256 verified.")
148
+
149
+ print("Validating ZIP...")
150
+
151
+ if not zipfile.is_zipfile(
152
+ temp_zip
153
+ ):
154
+ raise RuntimeError(
155
+ "Downloaded file is not a valid ZIP."
156
+ )
157
+
158
+ if extract_dir.exists():
159
+ shutil.rmtree(
160
+ extract_dir
161
+ )
162
+
163
+ extract_dir.mkdir(
164
+ parents=True,
165
+ exist_ok=True
166
+ )
167
+
168
+ print("Extracting model...")
169
+
170
+ with zipfile.ZipFile(
171
+ temp_zip,
172
+ "r"
173
+ ) as archive:
174
+
175
+ self._safe_extract(
176
+ archive,
177
+ extract_dir
178
+ )
179
+
180
+ model_root = (
181
+ self._find_model_root(
182
+ extract_dir
183
+ )
184
+ )
185
+
186
+ if model_root is None:
187
+ raise RuntimeError(
188
+ "Could not locate a valid "
189
+ "SentenceTransformer model "
190
+ "inside the ZIP archive."
191
+ )
192
+
193
+ if self.model_dir.exists():
194
+ shutil.rmtree(
195
+ self.model_dir
196
+ )
197
+
198
+ print("Installing model into cache...")
199
+
200
+ shutil.copytree(
201
+ model_root,
202
+ self.model_dir
203
+ )
204
+
205
+ print()
206
+ print(
207
+ "E5-large-v2 installed successfully."
208
+ )
209
+ print(
210
+ f"Model location: {self.model_dir}"
211
+ )
212
+ print()
213
+
214
+ finally:
215
+
216
+ temp_zip.unlink(
217
+ missing_ok=True
218
+ )
219
+
220
+ zip_path.unlink(
221
+ missing_ok=True
222
+ )
223
+
224
+ if extract_dir.exists():
225
+ shutil.rmtree(
226
+ extract_dir,
227
+ ignore_errors=True
228
+ )
229
+
230
+ # ---------------------------------------------------------
231
+ # Download
232
+ # ---------------------------------------------------------
233
+
234
+ @staticmethod
235
+ def _download(
236
+ url: str,
237
+ destination: Path
238
+ ) -> None:
239
+
240
+ destination.parent.mkdir(
241
+ parents=True,
242
+ exist_ok=True
243
+ )
244
+
245
+ with requests.get(
246
+ url,
247
+ stream=True,
248
+ timeout=(30, 120)
249
+ ) as response:
250
+
251
+ response.raise_for_status()
252
+
253
+ total = int(
254
+ response.headers.get(
255
+ "content-length",
256
+ 0
257
+ )
258
+ )
259
+
260
+ downloaded = 0
261
+
262
+ with open(
263
+ destination,
264
+ "wb"
265
+ ) as file:
266
+
267
+ for chunk in response.iter_content(
268
+ chunk_size=CHUNK_SIZE
269
+ ):
270
+
271
+ if not chunk:
272
+ continue
273
+
274
+ file.write(chunk)
275
+
276
+ downloaded += len(chunk)
277
+
278
+ if total:
279
+
280
+ percent = (
281
+ downloaded
282
+ / total
283
+ * 100
284
+ )
285
+
286
+ downloaded_mb = (
287
+ downloaded
288
+ / 1024
289
+ / 1024
290
+ )
291
+
292
+ total_mb = (
293
+ total
294
+ / 1024
295
+ / 1024
296
+ )
297
+
298
+ print(
299
+ f"\r"
300
+ f"Downloading: "
301
+ f"{percent:6.2f}% "
302
+ f"({downloaded_mb:.1f}/"
303
+ f"{total_mb:.1f} MB)",
304
+ end="",
305
+ flush=True
306
+ )
307
+
308
+ print()
309
+
310
+ # ---------------------------------------------------------
311
+ # SHA256
312
+ # ---------------------------------------------------------
313
+
314
+ @staticmethod
315
+ def _sha256(
316
+ path: Path
317
+ ) -> str:
318
+
319
+ digest = hashlib.sha256()
320
+
321
+ with open(
322
+ path,
323
+ "rb"
324
+ ) as file:
325
+
326
+ while chunk := file.read(
327
+ CHUNK_SIZE
328
+ ):
329
+
330
+ digest.update(chunk)
331
+
332
+ return digest.hexdigest()
333
+
334
+ # ---------------------------------------------------------
335
+ # ZIP security
336
+ # ---------------------------------------------------------
337
+
338
+ @staticmethod
339
+ def _safe_extract(
340
+ archive: zipfile.ZipFile,
341
+ destination: Path
342
+ ) -> None:
343
+
344
+ destination = destination.resolve()
345
+
346
+ for member in archive.infolist():
347
+
348
+ member_path = (
349
+ destination
350
+ / member.filename
351
+ ).resolve()
352
+
353
+ if not str(member_path).startswith(
354
+ str(destination)
355
+ ):
356
+ raise RuntimeError(
357
+ "Unsafe ZIP archive detected."
358
+ )
359
+
360
+ archive.extractall(
361
+ destination
362
+ )
363
+
364
+ # ---------------------------------------------------------
365
+ # Locate model
366
+ # ---------------------------------------------------------
367
+
368
+ @staticmethod
369
+ def _is_model_directory(
370
+ path: Path
371
+ ) -> bool:
372
+
373
+ if not path.is_dir():
374
+ return False
375
+
376
+ config = (
377
+ path / "config.json"
378
+ )
379
+
380
+ modules = (
381
+ path / "modules.json"
382
+ )
383
+
384
+ return (
385
+ config.exists()
386
+ and modules.exists()
387
+ )
388
+
389
+ @classmethod
390
+ def _find_model_root(
391
+ cls,
392
+ root: Path
393
+ ) -> Path | None:
394
+
395
+ if cls._is_model_directory(
396
+ root
397
+ ):
398
+ return root
399
+
400
+ for path in root.rglob("*"):
401
+
402
+ if path.is_dir():
403
+
404
+ if cls._is_model_directory(
405
+ path
406
+ ):
407
+ return path
408
+
409
+ return None
410
+
411
+ # ---------------------------------------------------------
412
+ # Encoding
413
+ # ---------------------------------------------------------
414
+
415
+ def encode(
416
+ self,
417
+ texts: str | Iterable[str],
418
+ *,
419
+ batch_size: int = 32,
420
+ normalize_embeddings: bool = True,
421
+ show_progress_bar: bool = False,
422
+ **kwargs
423
+ ):
424
+
425
+ if isinstance(
426
+ texts,
427
+ str
428
+ ):
429
+ texts = [texts]
430
+
431
+ return self.model.encode(
432
+ list(texts),
433
+ batch_size=batch_size,
434
+ normalize_embeddings=normalize_embeddings,
435
+ show_progress_bar=show_progress_bar,
436
+ **kwargs
437
+ )
438
+
439
+ # ---------------------------------------------------------
440
+ # E5 convenience methods
441
+ # ---------------------------------------------------------
442
+
443
+ def encode_queries(
444
+ self,
445
+ queries: str | Iterable[str],
446
+ **kwargs
447
+ ):
448
+
449
+ if isinstance(
450
+ queries,
451
+ str
452
+ ):
453
+ queries = [queries]
454
+
455
+ queries = [
456
+ q if q.startswith(
457
+ "query:"
458
+ )
459
+ else f"query: {q}"
460
+ for q in queries
461
+ ]
462
+
463
+ return self.encode(
464
+ queries,
465
+ **kwargs
466
+ )
467
+
468
+ def encode_passages(
469
+ self,
470
+ passages: str | Iterable[str],
471
+ **kwargs
472
+ ):
473
+
474
+ if isinstance(
475
+ passages,
476
+ str
477
+ ):
478
+ passages = [passages]
479
+
480
+ passages = [
481
+ p if p.startswith(
482
+ "passage:"
483
+ )
484
+ else f"passage: {p}"
485
+ for p in passages
486
+ ]
487
+
488
+ return self.encode(
489
+ passages,
490
+ **kwargs
491
+ )
492
+
493
+ @property
494
+ def dimension(self) -> int:
495
+
496
+ return self.model.get_sentence_embedding_dimension()
497
+
498
+ @property
499
+ def path(self) -> Path:
500
+
501
+ return self.model_dir
@@ -0,0 +1,25 @@
1
+ Metadata-Version: 2.4
2
+ Name: e5-large-v2-local
3
+ Version: 1.0.0
4
+ Summary: Local E5-large-v2 embedding model with automatic Internet Archive download
5
+ Author: Your Name
6
+ Project-URL: Homepage, https://archive.org/details/e5-large-v2
7
+ Project-URL: Model, https://archive.org/details/e5-large-v2
8
+ Project-URL: Download, https://archive.org/download/e5-large-v2/e5-large-v2.zip
9
+ Requires-Python: >=3.10
10
+ Description-Content-Type: text/markdown
11
+ Requires-Dist: sentence-transformers>=3.0
12
+ Requires-Dist: requests>=2.31
13
+
14
+ # E5-large-v2 Local
15
+
16
+ A lightweight Python package for using the E5-large-v2 embedding model locally.
17
+
18
+ The package itself does not contain the 751 MB model.
19
+
20
+ On first use, the model is downloaded from Internet Archive and cached locally.
21
+
22
+ ## Installation
23
+
24
+ ```bash
25
+ pip install e5-large-v2-local
@@ -0,0 +1,9 @@
1
+ README.md
2
+ pyproject.toml
3
+ src/e5_large_v2/__init__.py
4
+ src/e5_large_v2/model.py
5
+ src/e5_large_v2_local.egg-info/PKG-INFO
6
+ src/e5_large_v2_local.egg-info/SOURCES.txt
7
+ src/e5_large_v2_local.egg-info/dependency_links.txt
8
+ src/e5_large_v2_local.egg-info/requires.txt
9
+ src/e5_large_v2_local.egg-info/top_level.txt
@@ -0,0 +1,2 @@
1
+ sentence-transformers>=3.0
2
+ requests>=2.31