pgvector 0.4.1__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (94) hide show
  1. {pgvector-0.4.1 → pgvector-0.5.0}/LICENSE.txt +1 -1
  2. {pgvector-0.4.1/pgvector.egg-info → pgvector-0.5.0}/PKG-INFO +86 -23
  3. pgvector-0.4.1/PKG-INFO → pgvector-0.5.0/README.md +83 -32
  4. pgvector-0.5.0/pgvector/_utils.py +23 -0
  5. pgvector-0.5.0/pgvector/asyncpg/register.py +32 -0
  6. pgvector-0.5.0/pgvector/bit.py +95 -0
  7. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/django/__init__.py +1 -6
  8. pgvector-0.5.0/pgvector/django/bit.py +37 -0
  9. pgvector-0.5.0/pgvector/django/extensions.py +11 -0
  10. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/django/functions.py +9 -10
  11. pgvector-0.5.0/pgvector/django/halfvec.py +54 -0
  12. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/django/indexes.py +7 -6
  13. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/django/sparsevec.py +19 -13
  14. pgvector-0.5.0/pgvector/django/vector.py +54 -0
  15. pgvector-0.5.0/pgvector/halfvec.py +114 -0
  16. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/peewee/__init__.py +1 -6
  17. pgvector-0.5.0/pgvector/peewee/bit.py +22 -0
  18. pgvector-0.5.0/pgvector/peewee/halfvec.py +35 -0
  19. pgvector-0.5.0/pgvector/peewee/sparsevec.py +35 -0
  20. pgvector-0.5.0/pgvector/peewee/vector.py +35 -0
  21. pgvector-0.5.0/pgvector/pg8000/__init__.py +5 -0
  22. pgvector-0.5.0/pgvector/pg8000/register.py +29 -0
  23. pgvector-0.5.0/pgvector/psycopg/__init__.py +6 -0
  24. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/psycopg/bit.py +10 -7
  25. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/psycopg/halfvec.py +14 -16
  26. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/psycopg/register.py +4 -2
  27. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/psycopg/sparsevec.py +14 -16
  28. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/psycopg/vector.py +22 -18
  29. pgvector-0.5.0/pgvector/psycopg2/__init__.py +5 -0
  30. pgvector-0.5.0/pgvector/psycopg2/halfvec.py +30 -0
  31. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/psycopg2/register.py +3 -3
  32. pgvector-0.5.0/pgvector/psycopg2/sparsevec.py +30 -0
  33. pgvector-0.5.0/pgvector/psycopg2/vector.py +40 -0
  34. pgvector-0.5.0/pgvector/py.typed +0 -0
  35. pgvector-0.5.0/pgvector/sparsevec.py +178 -0
  36. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/sqlalchemy/__init__.py +0 -5
  37. pgvector-0.5.0/pgvector/sqlalchemy/bit.py +24 -0
  38. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/sqlalchemy/functions.py +2 -1
  39. pgvector-0.5.0/pgvector/sqlalchemy/halfvec.py +55 -0
  40. pgvector-0.5.0/pgvector/sqlalchemy/sparsevec.py +55 -0
  41. pgvector-0.5.0/pgvector/sqlalchemy/vector.py +55 -0
  42. pgvector-0.5.0/pgvector/vector.py +112 -0
  43. pgvector-0.4.1/README.md → pgvector-0.5.0/pgvector.egg-info/PKG-INFO +95 -19
  44. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector.egg-info/SOURCES.txt +2 -2
  45. pgvector-0.5.0/pyproject.toml +87 -0
  46. pgvector-0.5.0/tests/test_asyncpg.py +147 -0
  47. pgvector-0.5.0/tests/test_bit.py +103 -0
  48. {pgvector-0.4.1 → pgvector-0.5.0}/tests/test_django.py +88 -77
  49. pgvector-0.5.0/tests/test_half_vector.py +75 -0
  50. {pgvector-0.4.1 → pgvector-0.5.0}/tests/test_peewee.py +44 -44
  51. {pgvector-0.4.1 → pgvector-0.5.0}/tests/test_pg8000.py +21 -30
  52. pgvector-0.5.0/tests/test_psycopg.py +236 -0
  53. {pgvector-0.4.1 → pgvector-0.5.0}/tests/test_psycopg2.py +31 -38
  54. {pgvector-0.4.1 → pgvector-0.5.0}/tests/test_sparse_vector.py +77 -33
  55. {pgvector-0.4.1 → pgvector-0.5.0}/tests/test_sqlalchemy.py +230 -181
  56. {pgvector-0.4.1 → pgvector-0.5.0}/tests/test_sqlmodel.py +74 -70
  57. pgvector-0.5.0/tests/test_vector.py +75 -0
  58. pgvector-0.4.1/pgvector/asyncpg/__init__.py +0 -11
  59. pgvector-0.4.1/pgvector/asyncpg/register.py +0 -31
  60. pgvector-0.4.1/pgvector/bit.py +0 -75
  61. pgvector-0.4.1/pgvector/django/bit.py +0 -32
  62. pgvector-0.4.1/pgvector/django/extensions.py +0 -6
  63. pgvector-0.4.1/pgvector/django/halfvec.py +0 -60
  64. pgvector-0.4.1/pgvector/django/vector.py +0 -73
  65. pgvector-0.4.1/pgvector/halfvec.py +0 -83
  66. pgvector-0.4.1/pgvector/peewee/bit.py +0 -21
  67. pgvector-0.4.1/pgvector/peewee/halfvec.py +0 -34
  68. pgvector-0.4.1/pgvector/peewee/sparsevec.py +0 -34
  69. pgvector-0.4.1/pgvector/peewee/vector.py +0 -34
  70. pgvector-0.4.1/pgvector/pg8000/register.py +0 -23
  71. pgvector-0.4.1/pgvector/psycopg/__init__.py +0 -13
  72. pgvector-0.4.1/pgvector/psycopg2/__init__.py +0 -10
  73. pgvector-0.4.1/pgvector/psycopg2/halfvec.py +0 -25
  74. pgvector-0.4.1/pgvector/psycopg2/sparsevec.py +0 -25
  75. pgvector-0.4.1/pgvector/psycopg2/vector.py +0 -27
  76. pgvector-0.4.1/pgvector/sparsevec.py +0 -161
  77. pgvector-0.4.1/pgvector/sqlalchemy/bit.py +0 -26
  78. pgvector-0.4.1/pgvector/sqlalchemy/halfvec.py +0 -51
  79. pgvector-0.4.1/pgvector/sqlalchemy/sparsevec.py +0 -51
  80. pgvector-0.4.1/pgvector/sqlalchemy/vector.py +0 -51
  81. pgvector-0.4.1/pgvector/utils/__init__.py +0 -9
  82. pgvector-0.4.1/pgvector/vector.py +0 -83
  83. pgvector-0.4.1/pgvector.egg-info/requires.txt +0 -1
  84. pgvector-0.4.1/pyproject.toml +0 -24
  85. pgvector-0.4.1/tests/test_asyncpg.py +0 -146
  86. pgvector-0.4.1/tests/test_bit.py +0 -63
  87. pgvector-0.4.1/tests/test_half_vector.py +0 -59
  88. pgvector-0.4.1/tests/test_psycopg.py +0 -223
  89. pgvector-0.4.1/tests/test_vector.py +0 -59
  90. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector/__init__.py +0 -0
  91. {pgvector-0.4.1/pgvector/pg8000 → pgvector-0.5.0/pgvector/asyncpg}/__init__.py +0 -0
  92. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector.egg-info/dependency_links.txt +0 -0
  93. {pgvector-0.4.1 → pgvector-0.5.0}/pgvector.egg-info/top_level.txt +0 -0
  94. {pgvector-0.4.1 → pgvector-0.5.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  The MIT License (MIT)
2
2
 
3
- Copyright (c) 2021-2025 Andrew Kane
3
+ Copyright (c) 2021-2026 Andrew Kane
4
4
 
5
5
  Permission is hereby granted, free of charge, to any person obtaining a copy
6
6
  of this software and associated documentation files (the "Software"), to deal
@@ -1,21 +1,20 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pgvector
3
- Version: 0.4.1
3
+ Version: 0.5.0
4
4
  Summary: pgvector support for Python
5
5
  Author-email: Andrew Kane <andrew@ankane.org>
6
- License: MIT
6
+ License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/pgvector/pgvector-python
8
- Requires-Python: >=3.9
8
+ Requires-Python: >=3.10
9
9
  Description-Content-Type: text/markdown
10
10
  License-File: LICENSE.txt
11
- Requires-Dist: numpy
12
11
  Dynamic: license-file
13
12
 
14
13
  # pgvector-python
15
14
 
16
15
  [pgvector](https://github.com/pgvector/pgvector) support for Python
17
16
 
18
- Supports [Django](https://github.com/django/django), [SQLAlchemy](https://github.com/sqlalchemy/sqlalchemy), [SQLModel](https://github.com/tiangolo/sqlmodel), [Psycopg 3](https://github.com/psycopg/psycopg), [Psycopg 2](https://github.com/psycopg/psycopg2), [asyncpg](https://github.com/MagicStack/asyncpg), [pg8000](https://github.com/tlocke/pg8000), and [Peewee](https://github.com/coleifer/peewee)
17
+ Supports [Django](https://github.com/django/django), [SQLAlchemy](https://github.com/sqlalchemy/sqlalchemy), [SQLModel](https://github.com/tiangolo/sqlmodel), [Psycopg 3](https://github.com/psycopg/psycopg), [Psycopg 2](https://github.com/psycopg/psycopg2), [asyncpg](https://github.com/MagicStack/asyncpg), [pg8000](https://codeberg.org/tlocke/pg8000), and [Peewee](https://github.com/coleifer/peewee)
19
18
 
20
19
  [![Build Status](https://github.com/pgvector/pgvector-python/actions/workflows/build.yml/badge.svg)](https://github.com/pgvector/pgvector-python/actions)
21
20
 
@@ -190,10 +189,10 @@ session.execute(text('CREATE EXTENSION IF NOT EXISTS vector'))
190
189
  Add a vector column
191
190
 
192
191
  ```python
193
- from pgvector.sqlalchemy import Vector
192
+ from pgvector.sqlalchemy import VECTOR
194
193
 
195
194
  class Item(Base):
196
- embedding = mapped_column(Vector(3))
195
+ embedding: Mapped[list[float]] = mapped_column(VECTOR(3))
197
196
  ```
198
197
 
199
198
  Also supports `HALFVEC`, `BIT`, and `SPARSEVEC`
@@ -272,7 +271,6 @@ index = Index(
272
271
  'my_index',
273
272
  func.cast(Item.embedding, HALFVEC(3)).label('embedding'),
274
273
  postgresql_using='hnsw',
275
- postgresql_with={'m': 16, 'ef_construction': 64},
276
274
  postgresql_ops={'embedding': 'halfvec_l2_ops'}
277
275
  )
278
276
  ```
@@ -284,16 +282,47 @@ order = func.cast(Item.embedding, HALFVEC(3)).l2_distance([3, 1, 2])
284
282
  session.scalars(select(Item).order_by(order).limit(5))
285
283
  ```
286
284
 
285
+ #### Binary Quantization
286
+
287
+ Use expression indexing for binary quantization
288
+
289
+ ```python
290
+ from pgvector.sqlalchemy import BIT
291
+ from sqlalchemy.sql import func
292
+
293
+ index = Index(
294
+ 'my_index',
295
+ func.cast(func.binary_quantize(Item.embedding), BIT(3)).label('embedding'),
296
+ postgresql_using='hnsw',
297
+ postgresql_ops={'embedding': 'bit_hamming_ops'}
298
+ )
299
+ ```
300
+
301
+ Get the nearest neighbors by Hamming distance
302
+
303
+ ```python
304
+ order = func.cast(func.binary_quantize(Item.embedding), BIT(3)).hamming_distance(func.binary_quantize(func.cast([3, -1, 2], VECTOR(3))))
305
+ session.scalars(select(Item).order_by(order).limit(5))
306
+ ```
307
+
308
+ Re-rank by the original vectors for better recall
309
+
310
+ ```python
311
+ order = func.cast(func.binary_quantize(Item.embedding), BIT(3)).hamming_distance(func.binary_quantize(func.cast([3, -1, 2], VECTOR(3))))
312
+ subquery = session.query(Item).order_by(order).limit(20).subquery()
313
+ session.scalars(select(subquery).order_by(subquery.c.embedding.cosine_distance([3, -1, 2])).limit(5))
314
+ ```
315
+
287
316
  #### Arrays
288
317
 
289
318
  Add an array column
290
319
 
291
320
  ```python
292
- from pgvector.sqlalchemy import Vector
321
+ from pgvector.sqlalchemy import VECTOR
293
322
  from sqlalchemy import ARRAY
294
323
 
295
324
  class Item(Base):
296
- embeddings = mapped_column(ARRAY(Vector(3)))
325
+ embeddings: Mapped[list[list[float]]] = mapped_column(ARRAY(VECTOR(3), dimensions=1))
297
326
  ```
298
327
 
299
328
  And register the types with the underlying driver
@@ -328,7 +357,7 @@ from sqlalchemy import event
328
357
 
329
358
  @event.listens_for(engine, "connect")
330
359
  def connect(dbapi_connection, connection_record):
331
- register_vector(dbapi_connection, arrays=True)
360
+ register_vector(dbapi_connection)
332
361
  ```
333
362
 
334
363
  ## SQLModel
@@ -342,10 +371,10 @@ session.exec(text('CREATE EXTENSION IF NOT EXISTS vector'))
342
371
  Add a vector column
343
372
 
344
373
  ```python
345
- from pgvector.sqlalchemy import Vector
374
+ from pgvector.sqlalchemy import VECTOR
346
375
 
347
376
  class Item(SQLModel, table=True):
348
- embedding: Any = Field(sa_type=Vector(3))
377
+ embedding: list[float] = Field(sa_type=VECTOR(3))
349
378
  ```
350
379
 
351
380
  Also supports `HALFVEC`, `BIT`, and `SPARSEVEC`
@@ -422,7 +451,7 @@ Enable the extension
422
451
  conn.execute('CREATE EXTENSION IF NOT EXISTS vector')
423
452
  ```
424
453
 
425
- Register the vector type with your connection
454
+ Register the types with your connection
426
455
 
427
456
  ```python
428
457
  from pgvector.psycopg import register_vector
@@ -456,7 +485,9 @@ conn.execute('CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3))
456
485
  Insert a vector
457
486
 
458
487
  ```python
459
- embedding = np.array([1, 2, 3])
488
+ from pgvector import Vector
489
+
490
+ embedding = Vector([1, 2, 3])
460
491
  conn.execute('INSERT INTO items (embedding) VALUES (%s)', (embedding,))
461
492
  ```
462
493
 
@@ -485,7 +516,7 @@ cur = conn.cursor()
485
516
  cur.execute('CREATE EXTENSION IF NOT EXISTS vector')
486
517
  ```
487
518
 
488
- Register the vector type with your connection or cursor
519
+ Register the types with your connection or cursor
489
520
 
490
521
  ```python
491
522
  from pgvector.psycopg2 import register_vector
@@ -502,7 +533,9 @@ cur.execute('CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3))'
502
533
  Insert a vector
503
534
 
504
535
  ```python
505
- embedding = np.array([1, 2, 3])
536
+ from pgvector import Vector
537
+
538
+ embedding = Vector([1, 2, 3])
506
539
  cur.execute('INSERT INTO items (embedding) VALUES (%s)', (embedding,))
507
540
  ```
508
541
 
@@ -531,7 +564,7 @@ Enable the extension
531
564
  await conn.execute('CREATE EXTENSION IF NOT EXISTS vector')
532
565
  ```
533
566
 
534
- Register the vector type with your connection
567
+ Register the types with your connection
535
568
 
536
569
  ```python
537
570
  from pgvector.asyncpg import register_vector
@@ -557,7 +590,9 @@ await conn.execute('CREATE TABLE items (id bigserial PRIMARY KEY, embedding vect
557
590
  Insert a vector
558
591
 
559
592
  ```python
560
- embedding = np.array([1, 2, 3])
593
+ from pgvector import Vector
594
+
595
+ embedding = Vector([1, 2, 3])
561
596
  await conn.execute('INSERT INTO items (embedding) VALUES ($1)', embedding)
562
597
  ```
563
598
 
@@ -585,7 +620,7 @@ Enable the extension
585
620
  conn.run('CREATE EXTENSION IF NOT EXISTS vector')
586
621
  ```
587
622
 
588
- Register the vector type with your connection
623
+ Register the types with your connection
589
624
 
590
625
  ```python
591
626
  from pgvector.pg8000 import register_vector
@@ -602,7 +637,9 @@ conn.run('CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3))')
602
637
  Insert a vector
603
638
 
604
639
  ```python
605
- embedding = np.array([1, 2, 3])
640
+ from pgvector import Vector
641
+
642
+ embedding = Vector([1, 2, 3])
606
643
  conn.run('INSERT INTO items (embedding) VALUES (:embedding)', embedding=embedding)
607
644
  ```
608
645
 
@@ -681,6 +718,32 @@ Use `vector_ip_ops` for inner product and `vector_cosine_ops` for cosine distanc
681
718
 
682
719
  ## Reference
683
720
 
721
+ ### Vectors
722
+
723
+ Create a vector from a list
724
+
725
+ ```python
726
+ vec = Vector([1, 2, 3])
727
+ ```
728
+
729
+ Or a NumPy array
730
+
731
+ ```python
732
+ vec = Vector(np.array([1, 2, 3]))
733
+ ```
734
+
735
+ Get a list
736
+
737
+ ```python
738
+ lst = vec.to_list()
739
+ ```
740
+
741
+ Get a NumPy array
742
+
743
+ ```python
744
+ arr = vec.to_numpy()
745
+ ```
746
+
684
747
  ### Half Vectors
685
748
 
686
749
  Create a half vector from a list
@@ -790,7 +853,7 @@ To get started with development:
790
853
  ```sh
791
854
  git clone https://github.com/pgvector/pgvector-python.git
792
855
  cd pgvector-python
793
- pip install -r requirements.txt
856
+ pip install --group dev
794
857
  createdb pgvector_python_test
795
858
  pytest
796
859
  ```
@@ -799,7 +862,7 @@ To run an example:
799
862
 
800
863
  ```sh
801
864
  cd examples/loading
802
- pip install -r requirements.txt
865
+ pip install --group dev
803
866
  createdb pgvector_example
804
867
  python3 example.py
805
868
  ```
@@ -1,21 +1,8 @@
1
- Metadata-Version: 2.4
2
- Name: pgvector
3
- Version: 0.4.1
4
- Summary: pgvector support for Python
5
- Author-email: Andrew Kane <andrew@ankane.org>
6
- License: MIT
7
- Project-URL: Homepage, https://github.com/pgvector/pgvector-python
8
- Requires-Python: >=3.9
9
- Description-Content-Type: text/markdown
10
- License-File: LICENSE.txt
11
- Requires-Dist: numpy
12
- Dynamic: license-file
13
-
14
1
  # pgvector-python
15
2
 
16
3
  [pgvector](https://github.com/pgvector/pgvector) support for Python
17
4
 
18
- Supports [Django](https://github.com/django/django), [SQLAlchemy](https://github.com/sqlalchemy/sqlalchemy), [SQLModel](https://github.com/tiangolo/sqlmodel), [Psycopg 3](https://github.com/psycopg/psycopg), [Psycopg 2](https://github.com/psycopg/psycopg2), [asyncpg](https://github.com/MagicStack/asyncpg), [pg8000](https://github.com/tlocke/pg8000), and [Peewee](https://github.com/coleifer/peewee)
5
+ Supports [Django](https://github.com/django/django), [SQLAlchemy](https://github.com/sqlalchemy/sqlalchemy), [SQLModel](https://github.com/tiangolo/sqlmodel), [Psycopg 3](https://github.com/psycopg/psycopg), [Psycopg 2](https://github.com/psycopg/psycopg2), [asyncpg](https://github.com/MagicStack/asyncpg), [pg8000](https://codeberg.org/tlocke/pg8000), and [Peewee](https://github.com/coleifer/peewee)
19
6
 
20
7
  [![Build Status](https://github.com/pgvector/pgvector-python/actions/workflows/build.yml/badge.svg)](https://github.com/pgvector/pgvector-python/actions)
21
8
 
@@ -190,10 +177,10 @@ session.execute(text('CREATE EXTENSION IF NOT EXISTS vector'))
190
177
  Add a vector column
191
178
 
192
179
  ```python
193
- from pgvector.sqlalchemy import Vector
180
+ from pgvector.sqlalchemy import VECTOR
194
181
 
195
182
  class Item(Base):
196
- embedding = mapped_column(Vector(3))
183
+ embedding: Mapped[list[float]] = mapped_column(VECTOR(3))
197
184
  ```
198
185
 
199
186
  Also supports `HALFVEC`, `BIT`, and `SPARSEVEC`
@@ -272,7 +259,6 @@ index = Index(
272
259
  'my_index',
273
260
  func.cast(Item.embedding, HALFVEC(3)).label('embedding'),
274
261
  postgresql_using='hnsw',
275
- postgresql_with={'m': 16, 'ef_construction': 64},
276
262
  postgresql_ops={'embedding': 'halfvec_l2_ops'}
277
263
  )
278
264
  ```
@@ -284,16 +270,47 @@ order = func.cast(Item.embedding, HALFVEC(3)).l2_distance([3, 1, 2])
284
270
  session.scalars(select(Item).order_by(order).limit(5))
285
271
  ```
286
272
 
273
+ #### Binary Quantization
274
+
275
+ Use expression indexing for binary quantization
276
+
277
+ ```python
278
+ from pgvector.sqlalchemy import BIT
279
+ from sqlalchemy.sql import func
280
+
281
+ index = Index(
282
+ 'my_index',
283
+ func.cast(func.binary_quantize(Item.embedding), BIT(3)).label('embedding'),
284
+ postgresql_using='hnsw',
285
+ postgresql_ops={'embedding': 'bit_hamming_ops'}
286
+ )
287
+ ```
288
+
289
+ Get the nearest neighbors by Hamming distance
290
+
291
+ ```python
292
+ order = func.cast(func.binary_quantize(Item.embedding), BIT(3)).hamming_distance(func.binary_quantize(func.cast([3, -1, 2], VECTOR(3))))
293
+ session.scalars(select(Item).order_by(order).limit(5))
294
+ ```
295
+
296
+ Re-rank by the original vectors for better recall
297
+
298
+ ```python
299
+ order = func.cast(func.binary_quantize(Item.embedding), BIT(3)).hamming_distance(func.binary_quantize(func.cast([3, -1, 2], VECTOR(3))))
300
+ subquery = session.query(Item).order_by(order).limit(20).subquery()
301
+ session.scalars(select(subquery).order_by(subquery.c.embedding.cosine_distance([3, -1, 2])).limit(5))
302
+ ```
303
+
287
304
  #### Arrays
288
305
 
289
306
  Add an array column
290
307
 
291
308
  ```python
292
- from pgvector.sqlalchemy import Vector
309
+ from pgvector.sqlalchemy import VECTOR
293
310
  from sqlalchemy import ARRAY
294
311
 
295
312
  class Item(Base):
296
- embeddings = mapped_column(ARRAY(Vector(3)))
313
+ embeddings: Mapped[list[list[float]]] = mapped_column(ARRAY(VECTOR(3), dimensions=1))
297
314
  ```
298
315
 
299
316
  And register the types with the underlying driver
@@ -328,7 +345,7 @@ from sqlalchemy import event
328
345
 
329
346
  @event.listens_for(engine, "connect")
330
347
  def connect(dbapi_connection, connection_record):
331
- register_vector(dbapi_connection, arrays=True)
348
+ register_vector(dbapi_connection)
332
349
  ```
333
350
 
334
351
  ## SQLModel
@@ -342,10 +359,10 @@ session.exec(text('CREATE EXTENSION IF NOT EXISTS vector'))
342
359
  Add a vector column
343
360
 
344
361
  ```python
345
- from pgvector.sqlalchemy import Vector
362
+ from pgvector.sqlalchemy import VECTOR
346
363
 
347
364
  class Item(SQLModel, table=True):
348
- embedding: Any = Field(sa_type=Vector(3))
365
+ embedding: list[float] = Field(sa_type=VECTOR(3))
349
366
  ```
350
367
 
351
368
  Also supports `HALFVEC`, `BIT`, and `SPARSEVEC`
@@ -422,7 +439,7 @@ Enable the extension
422
439
  conn.execute('CREATE EXTENSION IF NOT EXISTS vector')
423
440
  ```
424
441
 
425
- Register the vector type with your connection
442
+ Register the types with your connection
426
443
 
427
444
  ```python
428
445
  from pgvector.psycopg import register_vector
@@ -456,7 +473,9 @@ conn.execute('CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3))
456
473
  Insert a vector
457
474
 
458
475
  ```python
459
- embedding = np.array([1, 2, 3])
476
+ from pgvector import Vector
477
+
478
+ embedding = Vector([1, 2, 3])
460
479
  conn.execute('INSERT INTO items (embedding) VALUES (%s)', (embedding,))
461
480
  ```
462
481
 
@@ -485,7 +504,7 @@ cur = conn.cursor()
485
504
  cur.execute('CREATE EXTENSION IF NOT EXISTS vector')
486
505
  ```
487
506
 
488
- Register the vector type with your connection or cursor
507
+ Register the types with your connection or cursor
489
508
 
490
509
  ```python
491
510
  from pgvector.psycopg2 import register_vector
@@ -502,7 +521,9 @@ cur.execute('CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3))'
502
521
  Insert a vector
503
522
 
504
523
  ```python
505
- embedding = np.array([1, 2, 3])
524
+ from pgvector import Vector
525
+
526
+ embedding = Vector([1, 2, 3])
506
527
  cur.execute('INSERT INTO items (embedding) VALUES (%s)', (embedding,))
507
528
  ```
508
529
 
@@ -531,7 +552,7 @@ Enable the extension
531
552
  await conn.execute('CREATE EXTENSION IF NOT EXISTS vector')
532
553
  ```
533
554
 
534
- Register the vector type with your connection
555
+ Register the types with your connection
535
556
 
536
557
  ```python
537
558
  from pgvector.asyncpg import register_vector
@@ -557,7 +578,9 @@ await conn.execute('CREATE TABLE items (id bigserial PRIMARY KEY, embedding vect
557
578
  Insert a vector
558
579
 
559
580
  ```python
560
- embedding = np.array([1, 2, 3])
581
+ from pgvector import Vector
582
+
583
+ embedding = Vector([1, 2, 3])
561
584
  await conn.execute('INSERT INTO items (embedding) VALUES ($1)', embedding)
562
585
  ```
563
586
 
@@ -585,7 +608,7 @@ Enable the extension
585
608
  conn.run('CREATE EXTENSION IF NOT EXISTS vector')
586
609
  ```
587
610
 
588
- Register the vector type with your connection
611
+ Register the types with your connection
589
612
 
590
613
  ```python
591
614
  from pgvector.pg8000 import register_vector
@@ -602,7 +625,9 @@ conn.run('CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3))')
602
625
  Insert a vector
603
626
 
604
627
  ```python
605
- embedding = np.array([1, 2, 3])
628
+ from pgvector import Vector
629
+
630
+ embedding = Vector([1, 2, 3])
606
631
  conn.run('INSERT INTO items (embedding) VALUES (:embedding)', embedding=embedding)
607
632
  ```
608
633
 
@@ -681,6 +706,32 @@ Use `vector_ip_ops` for inner product and `vector_cosine_ops` for cosine distanc
681
706
 
682
707
  ## Reference
683
708
 
709
+ ### Vectors
710
+
711
+ Create a vector from a list
712
+
713
+ ```python
714
+ vec = Vector([1, 2, 3])
715
+ ```
716
+
717
+ Or a NumPy array
718
+
719
+ ```python
720
+ vec = Vector(np.array([1, 2, 3]))
721
+ ```
722
+
723
+ Get a list
724
+
725
+ ```python
726
+ lst = vec.to_list()
727
+ ```
728
+
729
+ Get a NumPy array
730
+
731
+ ```python
732
+ arr = vec.to_numpy()
733
+ ```
734
+
684
735
  ### Half Vectors
685
736
 
686
737
  Create a half vector from a list
@@ -790,7 +841,7 @@ To get started with development:
790
841
  ```sh
791
842
  git clone https://github.com/pgvector/pgvector-python.git
792
843
  cd pgvector-python
793
- pip install -r requirements.txt
844
+ pip install --group dev
794
845
  createdb pgvector_python_test
795
846
  pytest
796
847
  ```
@@ -799,7 +850,7 @@ To run an example:
799
850
 
800
851
  ```sh
801
852
  cd examples/loading
802
- pip install -r requirements.txt
853
+ pip install --group dev
803
854
  createdb pgvector_example
804
855
  python3 example.py
805
856
  ```
@@ -0,0 +1,23 @@
1
+ import sys
2
+ from typing import TYPE_CHECKING, TypeAlias
3
+
4
+ if TYPE_CHECKING:
5
+ import numpy as np
6
+
7
+ ndarray: TypeAlias = np.ndarray[tuple[int, ...], np.dtype[np.floating]]
8
+ else:
9
+ # any value works since not type checking
10
+ # TODO use Never when Python 3.10 no longer supported
11
+ ndarray = None
12
+
13
+
14
+ def is_ndarray(value: object, /) -> bool:
15
+ if (numpy := sys.modules.get('numpy')):
16
+ return isinstance(value, numpy.ndarray)
17
+ return False
18
+
19
+
20
+ def is_sparse_array(value: object, /) -> bool:
21
+ if (sparse := sys.modules.get('scipy.sparse')):
22
+ return isinstance(value, (sparse.sparray, sparse.spmatrix))
23
+ return False
@@ -0,0 +1,32 @@
1
+ from asyncpg import Connection
2
+ from .. import Vector, HalfVector, SparseVector
3
+
4
+
5
+ async def register_vector(conn: Connection, /, *, schema: str = 'public') -> None:
6
+ await conn.set_type_codec(
7
+ 'vector',
8
+ schema=schema,
9
+ encoder=lambda v: (v if isinstance(v, Vector) else Vector(v)).to_binary(),
10
+ decoder=Vector.from_binary,
11
+ format='binary'
12
+ )
13
+
14
+ try:
15
+ await conn.set_type_codec(
16
+ 'halfvec',
17
+ schema=schema,
18
+ encoder=lambda v: (v if isinstance(v, HalfVector) else HalfVector(v)).to_binary(),
19
+ decoder=HalfVector.from_binary,
20
+ format='binary'
21
+ )
22
+
23
+ await conn.set_type_codec(
24
+ 'sparsevec',
25
+ schema=schema,
26
+ encoder=lambda v: (v if isinstance(v, SparseVector) else SparseVector(v)).to_binary(),
27
+ decoder=SparseVector.from_binary,
28
+ format='binary'
29
+ )
30
+ except ValueError as e:
31
+ if not str(e).startswith('unknown type:'):
32
+ raise e
@@ -0,0 +1,95 @@
1
+ from __future__ import annotations
2
+ from struct import pack, unpack_from
3
+ from typing import TYPE_CHECKING
4
+ from ._utils import is_ndarray
5
+
6
+ if TYPE_CHECKING:
7
+ import numpy as np
8
+
9
+
10
+ class Bit:
11
+ _length: int
12
+ _data: bytes
13
+
14
+ def __init__(
15
+ self,
16
+ value: bytes | str | list[bool] | np.ndarray[tuple[int, ...], np.dtype[np.bool | np.uint8]],
17
+ /
18
+ ) -> None:
19
+ if isinstance(value, bytes):
20
+ self._length = 8 * len(value)
21
+ self._data = value
22
+ elif isinstance(value, (list, str)):
23
+ if isinstance(value, list):
24
+ bits = {True: '1', False: '0'}
25
+ try:
26
+ value = ''.join([bits[v] for v in value])
27
+ except (KeyError, TypeError):
28
+ raise ValueError('expected list[bool]')
29
+
30
+ length = len(value)
31
+ if length % 8 != 0:
32
+ value += '0' * (8 - (length % 8))
33
+
34
+ self._length = length
35
+ try:
36
+ self._data = int(value, 2).to_bytes(len(value) // 8, byteorder='big')
37
+ except ValueError:
38
+ raise ValueError('expected bit string')
39
+ elif is_ndarray(value):
40
+ import numpy as np
41
+
42
+ if value.dtype != np.bool:
43
+ # skip error for result of np.unpackbits
44
+ if value.dtype != np.uint8 or np.any(value > 1):
45
+ raise ValueError('expected elements to be boolean')
46
+ value = value.astype(bool)
47
+
48
+ if value.ndim != 1:
49
+ raise ValueError('expected ndim to be 1')
50
+
51
+ self._length = len(value)
52
+ self._data = np.packbits(value).tobytes() # type: ignore
53
+ else:
54
+ raise ValueError('expected bytes, str, list, or ndarray')
55
+
56
+ def __repr__(self) -> str:
57
+ return f'Bit({self.to_text()})'
58
+
59
+ def __eq__(self, other: object, /) -> bool:
60
+ if not isinstance(other, self.__class__):
61
+ return NotImplemented
62
+ return self._length == other._length and self._data == other._data
63
+
64
+ def to_list(self) -> list[bool]:
65
+ # TODO improve
66
+ return [v != '0' for v in self.to_text()]
67
+
68
+ def to_numpy(self) -> np.ndarray[tuple[int, ...], np.dtype[np.bool]]:
69
+ import numpy as np
70
+
71
+ return np.unpackbits(np.frombuffer(self._data, dtype=np.uint8), count=self._length).astype(bool)
72
+
73
+ def to_text(self) -> str:
74
+ return ''.join(format(v, '08b') for v in self._data)[:self._length]
75
+
76
+ def to_binary(self) -> bytes:
77
+ return pack('>i', self._length) + self._data
78
+
79
+ @classmethod
80
+ def from_text(cls, value: str, /) -> Bit:
81
+ # cast to ensure always uses str constructor
82
+ return cls(str(value))
83
+
84
+ @classmethod
85
+ def from_binary(cls, value: bytes | bytearray | memoryview, /) -> Bit:
86
+ length, = unpack_from('>i', value)
87
+ data = memoryview(value)[4:].tobytes()
88
+
89
+ if len(data) != (length + 7) // 8:
90
+ raise ValueError('invalid length')
91
+
92
+ bit = cls.__new__(cls)
93
+ bit._length = length
94
+ bit._data = data
95
+ return bit
@@ -6,9 +6,6 @@ from .indexes import IvfflatIndex, HnswIndex
6
6
  from .sparsevec import SparseVectorField
7
7
  from .vector import VectorField
8
8
 
9
- # TODO remove
10
- from .. import HalfVector, SparseVector
11
-
12
9
  __all__ = [
13
10
  'VectorExtension',
14
11
  'VectorField',
@@ -22,7 +19,5 @@ __all__ = [
22
19
  'CosineDistance',
23
20
  'L1Distance',
24
21
  'HammingDistance',
25
- 'JaccardDistance',
26
- 'HalfVector',
27
- 'SparseVector'
22
+ 'JaccardDistance'
28
23
  ]