pgvector 0.3.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. pgvector/asyncpg/__init__.py +9 -0
  2. pgvector/asyncpg/register.py +31 -0
  3. pgvector/django/__init__.py +26 -0
  4. pgvector/django/bit.py +32 -0
  5. pgvector/django/extensions.py +6 -0
  6. pgvector/django/functions.py +55 -0
  7. pgvector/django/halfvec.py +60 -0
  8. pgvector/django/indexes.py +46 -0
  9. pgvector/django/sparsevec.py +55 -0
  10. pgvector/django/vector.py +73 -0
  11. pgvector/peewee/__init__.py +14 -0
  12. pgvector/peewee/bit.py +21 -0
  13. pgvector/peewee/halfvec.py +34 -0
  14. pgvector/peewee/sparsevec.py +34 -0
  15. pgvector/peewee/vector.py +34 -0
  16. pgvector/psycopg/__init__.py +11 -0
  17. pgvector/psycopg/bit.py +31 -0
  18. pgvector/psycopg/halfvec.py +53 -0
  19. pgvector/psycopg/register.py +37 -0
  20. pgvector/psycopg/sparsevec.py +53 -0
  21. pgvector/psycopg/vector.py +58 -0
  22. pgvector/psycopg2/__init__.py +8 -0
  23. pgvector/psycopg2/halfvec.py +20 -0
  24. pgvector/psycopg2/register.py +28 -0
  25. pgvector/psycopg2/sparsevec.py +20 -0
  26. pgvector/psycopg2/vector.py +21 -0
  27. pgvector/sqlalchemy/__init__.py +19 -0
  28. pgvector/sqlalchemy/bit.py +26 -0
  29. pgvector/sqlalchemy/functions.py +14 -0
  30. pgvector/sqlalchemy/halfvec.py +51 -0
  31. pgvector/sqlalchemy/sparsevec.py +51 -0
  32. pgvector/sqlalchemy/vector.py +51 -0
  33. pgvector/utils/__init__.py +11 -0
  34. pgvector/utils/bit.py +61 -0
  35. pgvector/utils/halfvec.py +78 -0
  36. pgvector/utils/sparsevec.py +156 -0
  37. pgvector/utils/vector.py +78 -0
  38. pgvector-0.3.5.dist-info/LICENSE.txt +21 -0
  39. pgvector-0.3.5.dist-info/METADATA +554 -0
  40. pgvector-0.3.5.dist-info/RECORD +42 -0
  41. pgvector-0.3.5.dist-info/WHEEL +5 -0
  42. pgvector-0.3.5.dist-info/top_level.txt +1 -0
@@ -0,0 +1,156 @@
1
+ import numpy as np
2
+ from struct import pack, unpack_from
3
+
4
+ NO_DEFAULT = object()
5
+
6
+
7
+ class SparseVector:
8
+ def __init__(self, value, dimensions=NO_DEFAULT, /):
9
+ if value.__class__.__module__.startswith('scipy.sparse.'):
10
+ if dimensions is not NO_DEFAULT:
11
+ raise ValueError('extra argument')
12
+
13
+ self._from_sparse(value)
14
+ elif isinstance(value, dict):
15
+ if dimensions is NO_DEFAULT:
16
+ raise ValueError('missing dimensions')
17
+
18
+ self._from_dict(value, dimensions)
19
+ else:
20
+ if dimensions is not NO_DEFAULT:
21
+ raise ValueError('extra argument')
22
+
23
+ self._from_dense(value)
24
+
25
+ def __repr__(self):
26
+ elements = dict(zip(self._indices, self._values))
27
+ return f'SparseVector({elements}, {self._dim})'
28
+
29
+ def dimensions(self):
30
+ return self._dim
31
+
32
+ def indices(self):
33
+ return self._indices
34
+
35
+ def values(self):
36
+ return self._values
37
+
38
+ def to_coo(self):
39
+ from scipy.sparse import coo_array
40
+
41
+ coords = ([0] * len(self._indices), self._indices)
42
+ return coo_array((self._values, coords), shape=(1, self._dim))
43
+
44
+ def to_list(self):
45
+ vec = [0.0] * self._dim
46
+ for i, v in zip(self._indices, self._values):
47
+ vec[i] = v
48
+ return vec
49
+
50
+ def to_numpy(self):
51
+ vec = np.repeat(0.0, self._dim).astype(np.float32)
52
+ for i, v in zip(self._indices, self._values):
53
+ vec[i] = v
54
+ return vec
55
+
56
+ def to_text(self):
57
+ return '{' + ','.join([f'{int(i) + 1}:{float(v)}' for i, v in zip(self._indices, self._values)]) + '}/' + str(int(self._dim))
58
+
59
+ def to_binary(self):
60
+ nnz = len(self._indices)
61
+ return pack(f'>iii{nnz}i{nnz}f', self._dim, nnz, 0, *self._indices, *self._values)
62
+
63
+ def _from_dict(self, d, dim):
64
+ elements = [(i, v) for i, v in d.items() if v != 0]
65
+ elements.sort()
66
+
67
+ self._dim = int(dim)
68
+ self._indices = [int(v[0]) for v in elements]
69
+ self._values = [float(v[1]) for v in elements]
70
+
71
+ def _from_sparse(self, value):
72
+ value = value.tocoo()
73
+
74
+ if value.ndim == 1:
75
+ self._dim = value.shape[0]
76
+ elif value.ndim == 2 and value.shape[0] == 1:
77
+ self._dim = value.shape[1]
78
+ else:
79
+ raise ValueError('expected ndim to be 1')
80
+
81
+ if hasattr(value, 'coords'):
82
+ # scipy 1.13+
83
+ self._indices = value.coords[0].tolist()
84
+ else:
85
+ self._indices = value.col.tolist()
86
+ self._values = value.data.tolist()
87
+
88
+ def _from_dense(self, value):
89
+ self._dim = len(value)
90
+ self._indices = [i for i, v in enumerate(value) if v != 0]
91
+ self._values = [float(value[i]) for i in self._indices]
92
+
93
+ @classmethod
94
+ def from_text(cls, value):
95
+ elements, dim = value.split('/', 2)
96
+ indices = []
97
+ values = []
98
+ # split on empty string returns single element list
99
+ if len(elements) > 2:
100
+ for e in elements[1:-1].split(','):
101
+ i, v = e.split(':', 2)
102
+ indices.append(int(i) - 1)
103
+ values.append(float(v))
104
+ return cls._from_parts(int(dim), indices, values)
105
+
106
+ @classmethod
107
+ def from_binary(cls, value):
108
+ dim, nnz, unused = unpack_from('>iii', value)
109
+ indices = unpack_from(f'>{nnz}i', value, 12)
110
+ values = unpack_from(f'>{nnz}f', value, 12 + nnz * 4)
111
+ return cls._from_parts(int(dim), indices, values)
112
+
113
+ @classmethod
114
+ def _from_parts(cls, dim, indices, values):
115
+ vec = cls.__new__(cls)
116
+ vec._dim = dim
117
+ vec._indices = indices
118
+ vec._values = values
119
+ return vec
120
+
121
+ @classmethod
122
+ def _to_db(cls, value, dim=None):
123
+ if value is None:
124
+ return value
125
+
126
+ if not isinstance(value, cls):
127
+ value = cls(value)
128
+
129
+ if dim is not None and value.dimensions() != dim:
130
+ raise ValueError('expected %d dimensions, not %d' % (dim, value.dimensions()))
131
+
132
+ return value.to_text()
133
+
134
+ @classmethod
135
+ def _to_db_binary(cls, value):
136
+ if value is None:
137
+ return value
138
+
139
+ if not isinstance(value, cls):
140
+ value = cls(value)
141
+
142
+ return value.to_binary()
143
+
144
+ @classmethod
145
+ def _from_db(cls, value):
146
+ if value is None or isinstance(value, cls):
147
+ return value
148
+
149
+ return cls.from_text(value)
150
+
151
+ @classmethod
152
+ def _from_db_binary(cls, value):
153
+ if value is None or isinstance(value, cls):
154
+ return value
155
+
156
+ return cls.from_binary(value)
@@ -0,0 +1,78 @@
1
+ import numpy as np
2
+ from struct import pack, unpack_from
3
+
4
+
5
+ class Vector:
6
+ def __init__(self, value):
7
+ # asarray still copies if same dtype
8
+ if not isinstance(value, np.ndarray) or value.dtype != '>f4':
9
+ value = np.asarray(value, dtype='>f4')
10
+
11
+ if value.ndim != 1:
12
+ raise ValueError('expected ndim to be 1')
13
+
14
+ self._value = value
15
+
16
+ def __repr__(self):
17
+ return f'Vector({self.to_list()})'
18
+
19
+ def dimensions(self):
20
+ return len(self._value)
21
+
22
+ def to_list(self):
23
+ return self._value.tolist()
24
+
25
+ def to_numpy(self):
26
+ return self._value
27
+
28
+ def to_text(self):
29
+ return '[' + ','.join([str(float(v)) for v in self._value]) + ']'
30
+
31
+ def to_binary(self):
32
+ return pack('>HH', self.dimensions(), 0) + self._value.tobytes()
33
+
34
+ @classmethod
35
+ def from_text(cls, value):
36
+ return cls([float(v) for v in value[1:-1].split(',')])
37
+
38
+ @classmethod
39
+ def from_binary(cls, value):
40
+ dim, unused = unpack_from('>HH', value)
41
+ return cls(np.frombuffer(value, dtype='>f4', count=dim, offset=4))
42
+
43
+ @classmethod
44
+ def _to_db(cls, value, dim=None):
45
+ if value is None:
46
+ return value
47
+
48
+ if not isinstance(value, cls):
49
+ value = cls(value)
50
+
51
+ if dim is not None and value.dimensions() != dim:
52
+ raise ValueError('expected %d dimensions, not %d' % (dim, value.dimensions()))
53
+
54
+ return value.to_text()
55
+
56
+ @classmethod
57
+ def _to_db_binary(cls, value):
58
+ if value is None:
59
+ return value
60
+
61
+ if not isinstance(value, cls):
62
+ value = cls(value)
63
+
64
+ return value.to_binary()
65
+
66
+ @classmethod
67
+ def _from_db(cls, value):
68
+ if value is None or isinstance(value, np.ndarray):
69
+ return value
70
+
71
+ return cls.from_text(value).to_numpy().astype(np.float32)
72
+
73
+ @classmethod
74
+ def _from_db_binary(cls, value):
75
+ if value is None or isinstance(value, np.ndarray):
76
+ return value
77
+
78
+ return cls.from_binary(value).to_numpy().astype(np.float32)
@@ -0,0 +1,21 @@
1
+ The MIT License (MIT)
2
+
3
+ Copyright (c) 2021-2024 Andrew Kane
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in
13
+ all copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
21
+ THE SOFTWARE.