pgvector-template 0.5.1__tar.gz → 0.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pgvector_template-0.5.1 → pgvector_template-0.6}/PKG-INFO +70 -92
- {pgvector_template-0.5.1 → pgvector_template-0.6}/README.md +69 -91
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/core/document.py +18 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/core/manager.py +152 -17
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/db/connection.py +33 -1
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template.egg-info/PKG-INFO +70 -92
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pyproject.toml +1 -1
- {pgvector_template-0.5.1 → pgvector_template-0.6}/LICENSE +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/__init__.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/core/__init__.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/core/embedder.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/core/search.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/db/__init__.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/db/document_db.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/models/__init__.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/models/search.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/service/__init__.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/service/document_service.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/types.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/utils/__init__.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/utils/metadata_filter.py +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template.egg-info/SOURCES.txt +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template.egg-info/dependency_links.txt +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template.egg-info/requires.txt +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template.egg-info/top_level.txt +0 -0
- {pgvector_template-0.5.1 → pgvector_template-0.6}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pgvector-template
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6
|
|
4
4
|
Summary: Template library for flexible PGVector RAG implementations
|
|
5
5
|
Author-email: DL <v49t9zpqd@mozmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -39,10 +39,10 @@ PGVector-Template provides a robust foundation for implementing vector-based doc
|
|
|
39
39
|
## Quick Start
|
|
40
40
|
|
|
41
41
|
```python
|
|
42
|
-
from pgvector_template import
|
|
42
|
+
from pgvector_template import DocumentDatabaseManager
|
|
43
43
|
from pgvector_template.core.document import BaseDocument
|
|
44
|
-
from
|
|
45
|
-
from
|
|
44
|
+
from pgvector_template.models.search import SearchQuery
|
|
45
|
+
from pgvector_template.service.document_service import DocumentService, DocumentServiceConfig
|
|
46
46
|
|
|
47
47
|
|
|
48
48
|
# 1. Define your document model
|
|
@@ -50,29 +50,25 @@ class MyDocument(BaseDocument):
|
|
|
50
50
|
__tablename__ = "my_documents"
|
|
51
51
|
|
|
52
52
|
|
|
53
|
-
# 2. Set up database
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
53
|
+
# 2. Set up the database: connects, enables pgvector, creates the schema and tables
|
|
54
|
+
db = DocumentDatabaseManager(
|
|
55
|
+
database_url="postgresql://user:pass@localhost/mydb",
|
|
56
|
+
schema_suffix="my_kb",
|
|
57
|
+
document_classes=[MyDocument],
|
|
58
|
+
)
|
|
59
|
+
db.setup()
|
|
57
60
|
|
|
58
|
-
# 3. Create
|
|
59
|
-
|
|
61
|
+
# 3. Create a service. `my_embedding_provider` is your `BaseEmbeddingProvider` subclass
|
|
62
|
+
config = DocumentServiceConfig(document_cls=MyDocument, embedding_provider=my_embedding_provider)
|
|
60
63
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
doc = MyDocument.from_props(
|
|
64
|
-
corpus_id=uuid4(),
|
|
65
|
-
chunk_index=0,
|
|
66
|
-
content="Your document content here",
|
|
67
|
-
embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
|
|
68
|
-
)
|
|
69
|
-
doc_manager.insert_document(session, doc)
|
|
64
|
+
with db.get_session() as session:
|
|
65
|
+
service = DocumentService(session, config)
|
|
70
66
|
|
|
71
|
-
#
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
)
|
|
67
|
+
# 4. Insert a corpus: it is split into chunks, embedded and stored
|
|
68
|
+
service.corpus_manager.insert_corpus("Your document content here", {}, corpus_id="doc-1")
|
|
69
|
+
|
|
70
|
+
# 5. Search
|
|
71
|
+
results = service.search_client.search(SearchQuery(text="your query", limit=5))
|
|
76
72
|
```
|
|
77
73
|
|
|
78
74
|
## Key Concepts
|
|
@@ -91,6 +87,7 @@ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all
|
|
|
91
87
|
- **Metadata Management**: JSON-based flexible metadata with GIN indexing
|
|
92
88
|
- **Collection Support**: Organize documents into logical collections
|
|
93
89
|
- **Chunk Management**: Handle long content by chunking into retrievable documents
|
|
90
|
+
- **Smart Upserts**: Re-inserting a corpus only re-embeds chunks whose content changed, and skips the write entirely if nothing did
|
|
94
91
|
- **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
|
|
95
92
|
- **Type Safety**: Full Pydantic validation and type hints
|
|
96
93
|
- **Production Ready**: Comprehensive testing and error handling
|
|
@@ -144,34 +141,14 @@ CREATE EXTENSION IF NOT EXISTS vector;
|
|
|
144
141
|
3. **Set up your connection:**
|
|
145
142
|
|
|
146
143
|
```python
|
|
147
|
-
from sqlalchemy import create_engine
|
|
148
144
|
from pgvector_template import DatabaseManager
|
|
149
145
|
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
db_manager = DatabaseManager(engine)
|
|
153
|
-
|
|
154
|
-
# Option 2: From environment variable
|
|
155
|
-
import os
|
|
156
|
-
|
|
157
|
-
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
158
|
-
db_manager = DatabaseManager(engine)
|
|
146
|
+
db_manager = DatabaseManager("postgresql://user:password@localhost:5432/mydb")
|
|
147
|
+
db_manager.initialize() # connects and enables the vector extension
|
|
159
148
|
```
|
|
160
149
|
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
```python
|
|
164
|
-
from sqlalchemy import create_engine
|
|
165
|
-
from sqlalchemy.pool import QueuePool
|
|
166
|
-
|
|
167
|
-
engine = create_engine(
|
|
168
|
-
"postgresql://user:password@localhost:5432/mydb",
|
|
169
|
-
poolclass=QueuePool,
|
|
170
|
-
pool_size=10,
|
|
171
|
-
max_overflow=20,
|
|
172
|
-
pool_pre_ping=True,
|
|
173
|
-
)
|
|
174
|
-
```
|
|
150
|
+
`create_tables` also adds nullable model columns that are missing from existing tables (for example `content_hash`).
|
|
151
|
+
Only the column name and type are applied: indexes, foreign keys, unique constraints and server defaults are not. This is a stopgap until proper migrations are added.
|
|
175
152
|
|
|
176
153
|
### Environment Variables
|
|
177
154
|
|
|
@@ -198,70 +175,67 @@ class MyDocument(BaseDocument):
|
|
|
198
175
|
__tablename__ = "my_documents"
|
|
199
176
|
```
|
|
200
177
|
|
|
201
|
-
### 2. Insert
|
|
178
|
+
### 2. Insert a Corpus
|
|
202
179
|
|
|
203
180
|
```python
|
|
204
181
|
from pgvector_template.core.document import BaseDocumentOptionalProps
|
|
205
|
-
from uuid import uuid4
|
|
206
182
|
|
|
207
|
-
# Create document with metadata
|
|
208
|
-
metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
|
|
209
183
|
optional_props = BaseDocumentOptionalProps(
|
|
210
184
|
title="Chapter 1: Introduction",
|
|
211
185
|
collection="textbooks",
|
|
212
186
|
language="en",
|
|
213
|
-
tags=["education", "intro"],
|
|
214
187
|
)
|
|
215
188
|
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
)
|
|
189
|
+
with db.get_session() as session:
|
|
190
|
+
service = DocumentService(session, config)
|
|
191
|
+
result = service.corpus_manager.insert_corpus(
|
|
192
|
+
"This is the document content...",
|
|
193
|
+
{"source_type": "pdf", "author": "John Doe"}, # corpus metadata
|
|
194
|
+
optional_props,
|
|
195
|
+
corpus_id="intro-chapter",
|
|
196
|
+
)
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
To insert chunks you have already split and embedded, use `insert_documents()`.
|
|
200
|
+
|
|
201
|
+
Calling `insert_corpus()` again with the same `corpus_id` replaces the corpus. Unchanged chunks reuse their stored vector, and the write is skipped if nothing changed:
|
|
224
202
|
|
|
225
|
-
|
|
226
|
-
|
|
203
|
+
```python
|
|
204
|
+
result = corpus_manager.insert_corpus(text, {"source": "pdf"}, corpus_id="handbook")
|
|
205
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=12, reused=0, total=12)
|
|
206
|
+
|
|
207
|
+
result = corpus_manager.insert_corpus(edited_text, {"source": "pdf"}, corpus_id="handbook")
|
|
208
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=1, reused=11, total=12)
|
|
227
209
|
```
|
|
228
210
|
|
|
211
|
+
Rows written before `content_hash` existed have no hash, so their first upsert re-embeds them.
|
|
212
|
+
|
|
229
213
|
### 3. Search Documents
|
|
230
214
|
|
|
231
215
|
```python
|
|
232
|
-
|
|
233
|
-
with db_manager.get_session() as session:
|
|
234
|
-
results = doc_manager.search_similar(
|
|
235
|
-
session=session, document_cls=MyDocument, query_embedding=query_vector, limit=10
|
|
236
|
-
)
|
|
237
|
-
|
|
238
|
-
# Search with filters
|
|
239
|
-
from pgvector_template.models.search import MetadataFilter
|
|
216
|
+
from pgvector_template.models.search import MetadataFilter, SearchQuery
|
|
240
217
|
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
MetadataFilter(key="author", value="John Doe"),
|
|
244
|
-
]
|
|
218
|
+
# Semantic search
|
|
219
|
+
results = service.search_client.search(SearchQuery(text="how do I reset my password", limit=10))
|
|
245
220
|
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
221
|
+
# Keyword search combined with a metadata filter
|
|
222
|
+
results = service.search_client.search(
|
|
223
|
+
SearchQuery(
|
|
224
|
+
keywords=["password", "reset"],
|
|
225
|
+
metadata_filters=[MetadataFilter(field_name="source_type", condition="eq", value="pdf")],
|
|
226
|
+
limit=10,
|
|
227
|
+
)
|
|
253
228
|
)
|
|
229
|
+
|
|
230
|
+
for r in results:
|
|
231
|
+
print(r.score, r.document.content)
|
|
254
232
|
```
|
|
255
233
|
|
|
256
|
-
### 4.
|
|
234
|
+
### 4. Read a Corpus Back
|
|
257
235
|
|
|
258
236
|
```python
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
docs = doc_manager.get_documents_by_collection(session, MyDocument, "textbooks")
|
|
262
|
-
|
|
263
|
-
# Get all chunks from a corpus
|
|
264
|
-
corpus_docs = doc_manager.get_documents_by_corpus_id(session, MyDocument, corpus_id)
|
|
237
|
+
corpus = service.corpus_manager.get_full_corpus("intro-chapter")
|
|
238
|
+
print(corpus.content, corpus.metadata, len(corpus.documents))
|
|
265
239
|
```
|
|
266
240
|
|
|
267
241
|
## Concept Reference
|
|
@@ -271,8 +245,11 @@ with db_manager.get_session() as session:
|
|
|
271
245
|
- **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
|
|
272
246
|
- **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
|
|
273
247
|
- **`BaseDocumentMetadata`**: Base schema for structured document metadata
|
|
274
|
-
- **`
|
|
275
|
-
- **`
|
|
248
|
+
- **`BaseEmbeddingProvider`**: Interface to implement for your embedding model
|
|
249
|
+
- **`BaseCorpusManager`**: Splits, embeds and upserts corpora (`insert_corpus`, `get_full_corpus`)
|
|
250
|
+
- **`BaseSearchClient`**: Semantic, keyword and metadata search (`search`)
|
|
251
|
+
- **`DocumentService`**: Bundles a corpus manager and a search client over one session
|
|
252
|
+
- **`DatabaseManager`** / **`DocumentDatabaseManager`**: Connection, schema and table setup
|
|
276
253
|
|
|
277
254
|
## Testing
|
|
278
255
|
|
|
@@ -306,7 +283,8 @@ python -m unittest discover -s integ-tests
|
|
|
306
283
|
|
|
307
284
|
```bash
|
|
308
285
|
pip install -e .[dev,test]
|
|
309
|
-
|
|
286
|
+
ruff check . && ruff format .
|
|
287
|
+
ty check .
|
|
310
288
|
```
|
|
311
289
|
|
|
312
290
|
## License
|
|
@@ -9,10 +9,10 @@ PGVector-Template provides a robust foundation for implementing vector-based doc
|
|
|
9
9
|
## Quick Start
|
|
10
10
|
|
|
11
11
|
```python
|
|
12
|
-
from pgvector_template import
|
|
12
|
+
from pgvector_template import DocumentDatabaseManager
|
|
13
13
|
from pgvector_template.core.document import BaseDocument
|
|
14
|
-
from
|
|
15
|
-
from
|
|
14
|
+
from pgvector_template.models.search import SearchQuery
|
|
15
|
+
from pgvector_template.service.document_service import DocumentService, DocumentServiceConfig
|
|
16
16
|
|
|
17
17
|
|
|
18
18
|
# 1. Define your document model
|
|
@@ -20,29 +20,25 @@ class MyDocument(BaseDocument):
|
|
|
20
20
|
__tablename__ = "my_documents"
|
|
21
21
|
|
|
22
22
|
|
|
23
|
-
# 2. Set up database
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
23
|
+
# 2. Set up the database: connects, enables pgvector, creates the schema and tables
|
|
24
|
+
db = DocumentDatabaseManager(
|
|
25
|
+
database_url="postgresql://user:pass@localhost/mydb",
|
|
26
|
+
schema_suffix="my_kb",
|
|
27
|
+
document_classes=[MyDocument],
|
|
28
|
+
)
|
|
29
|
+
db.setup()
|
|
27
30
|
|
|
28
|
-
# 3. Create
|
|
29
|
-
|
|
31
|
+
# 3. Create a service. `my_embedding_provider` is your `BaseEmbeddingProvider` subclass
|
|
32
|
+
config = DocumentServiceConfig(document_cls=MyDocument, embedding_provider=my_embedding_provider)
|
|
30
33
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
doc = MyDocument.from_props(
|
|
34
|
-
corpus_id=uuid4(),
|
|
35
|
-
chunk_index=0,
|
|
36
|
-
content="Your document content here",
|
|
37
|
-
embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
|
|
38
|
-
)
|
|
39
|
-
doc_manager.insert_document(session, doc)
|
|
34
|
+
with db.get_session() as session:
|
|
35
|
+
service = DocumentService(session, config)
|
|
40
36
|
|
|
41
|
-
#
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
)
|
|
37
|
+
# 4. Insert a corpus: it is split into chunks, embedded and stored
|
|
38
|
+
service.corpus_manager.insert_corpus("Your document content here", {}, corpus_id="doc-1")
|
|
39
|
+
|
|
40
|
+
# 5. Search
|
|
41
|
+
results = service.search_client.search(SearchQuery(text="your query", limit=5))
|
|
46
42
|
```
|
|
47
43
|
|
|
48
44
|
## Key Concepts
|
|
@@ -61,6 +57,7 @@ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all
|
|
|
61
57
|
- **Metadata Management**: JSON-based flexible metadata with GIN indexing
|
|
62
58
|
- **Collection Support**: Organize documents into logical collections
|
|
63
59
|
- **Chunk Management**: Handle long content by chunking into retrievable documents
|
|
60
|
+
- **Smart Upserts**: Re-inserting a corpus only re-embeds chunks whose content changed, and skips the write entirely if nothing did
|
|
64
61
|
- **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
|
|
65
62
|
- **Type Safety**: Full Pydantic validation and type hints
|
|
66
63
|
- **Production Ready**: Comprehensive testing and error handling
|
|
@@ -114,34 +111,14 @@ CREATE EXTENSION IF NOT EXISTS vector;
|
|
|
114
111
|
3. **Set up your connection:**
|
|
115
112
|
|
|
116
113
|
```python
|
|
117
|
-
from sqlalchemy import create_engine
|
|
118
114
|
from pgvector_template import DatabaseManager
|
|
119
115
|
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
db_manager = DatabaseManager(engine)
|
|
123
|
-
|
|
124
|
-
# Option 2: From environment variable
|
|
125
|
-
import os
|
|
126
|
-
|
|
127
|
-
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
128
|
-
db_manager = DatabaseManager(engine)
|
|
116
|
+
db_manager = DatabaseManager("postgresql://user:password@localhost:5432/mydb")
|
|
117
|
+
db_manager.initialize() # connects and enables the vector extension
|
|
129
118
|
```
|
|
130
119
|
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
```python
|
|
134
|
-
from sqlalchemy import create_engine
|
|
135
|
-
from sqlalchemy.pool import QueuePool
|
|
136
|
-
|
|
137
|
-
engine = create_engine(
|
|
138
|
-
"postgresql://user:password@localhost:5432/mydb",
|
|
139
|
-
poolclass=QueuePool,
|
|
140
|
-
pool_size=10,
|
|
141
|
-
max_overflow=20,
|
|
142
|
-
pool_pre_ping=True,
|
|
143
|
-
)
|
|
144
|
-
```
|
|
120
|
+
`create_tables` also adds nullable model columns that are missing from existing tables (for example `content_hash`).
|
|
121
|
+
Only the column name and type are applied: indexes, foreign keys, unique constraints and server defaults are not. This is a stopgap until proper migrations are added.
|
|
145
122
|
|
|
146
123
|
### Environment Variables
|
|
147
124
|
|
|
@@ -168,70 +145,67 @@ class MyDocument(BaseDocument):
|
|
|
168
145
|
__tablename__ = "my_documents"
|
|
169
146
|
```
|
|
170
147
|
|
|
171
|
-
### 2. Insert
|
|
148
|
+
### 2. Insert a Corpus
|
|
172
149
|
|
|
173
150
|
```python
|
|
174
151
|
from pgvector_template.core.document import BaseDocumentOptionalProps
|
|
175
|
-
from uuid import uuid4
|
|
176
152
|
|
|
177
|
-
# Create document with metadata
|
|
178
|
-
metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
|
|
179
153
|
optional_props = BaseDocumentOptionalProps(
|
|
180
154
|
title="Chapter 1: Introduction",
|
|
181
155
|
collection="textbooks",
|
|
182
156
|
language="en",
|
|
183
|
-
tags=["education", "intro"],
|
|
184
157
|
)
|
|
185
158
|
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
)
|
|
159
|
+
with db.get_session() as session:
|
|
160
|
+
service = DocumentService(session, config)
|
|
161
|
+
result = service.corpus_manager.insert_corpus(
|
|
162
|
+
"This is the document content...",
|
|
163
|
+
{"source_type": "pdf", "author": "John Doe"}, # corpus metadata
|
|
164
|
+
optional_props,
|
|
165
|
+
corpus_id="intro-chapter",
|
|
166
|
+
)
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
To insert chunks you have already split and embedded, use `insert_documents()`.
|
|
170
|
+
|
|
171
|
+
Calling `insert_corpus()` again with the same `corpus_id` replaces the corpus. Unchanged chunks reuse their stored vector, and the write is skipped if nothing changed:
|
|
194
172
|
|
|
195
|
-
|
|
196
|
-
|
|
173
|
+
```python
|
|
174
|
+
result = corpus_manager.insert_corpus(text, {"source": "pdf"}, corpus_id="handbook")
|
|
175
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=12, reused=0, total=12)
|
|
176
|
+
|
|
177
|
+
result = corpus_manager.insert_corpus(edited_text, {"source": "pdf"}, corpus_id="handbook")
|
|
178
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=1, reused=11, total=12)
|
|
197
179
|
```
|
|
198
180
|
|
|
181
|
+
Rows written before `content_hash` existed have no hash, so their first upsert re-embeds them.
|
|
182
|
+
|
|
199
183
|
### 3. Search Documents
|
|
200
184
|
|
|
201
185
|
```python
|
|
202
|
-
|
|
203
|
-
with db_manager.get_session() as session:
|
|
204
|
-
results = doc_manager.search_similar(
|
|
205
|
-
session=session, document_cls=MyDocument, query_embedding=query_vector, limit=10
|
|
206
|
-
)
|
|
207
|
-
|
|
208
|
-
# Search with filters
|
|
209
|
-
from pgvector_template.models.search import MetadataFilter
|
|
186
|
+
from pgvector_template.models.search import MetadataFilter, SearchQuery
|
|
210
187
|
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
MetadataFilter(key="author", value="John Doe"),
|
|
214
|
-
]
|
|
188
|
+
# Semantic search
|
|
189
|
+
results = service.search_client.search(SearchQuery(text="how do I reset my password", limit=10))
|
|
215
190
|
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
191
|
+
# Keyword search combined with a metadata filter
|
|
192
|
+
results = service.search_client.search(
|
|
193
|
+
SearchQuery(
|
|
194
|
+
keywords=["password", "reset"],
|
|
195
|
+
metadata_filters=[MetadataFilter(field_name="source_type", condition="eq", value="pdf")],
|
|
196
|
+
limit=10,
|
|
197
|
+
)
|
|
223
198
|
)
|
|
199
|
+
|
|
200
|
+
for r in results:
|
|
201
|
+
print(r.score, r.document.content)
|
|
224
202
|
```
|
|
225
203
|
|
|
226
|
-
### 4.
|
|
204
|
+
### 4. Read a Corpus Back
|
|
227
205
|
|
|
228
206
|
```python
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
docs = doc_manager.get_documents_by_collection(session, MyDocument, "textbooks")
|
|
232
|
-
|
|
233
|
-
# Get all chunks from a corpus
|
|
234
|
-
corpus_docs = doc_manager.get_documents_by_corpus_id(session, MyDocument, corpus_id)
|
|
207
|
+
corpus = service.corpus_manager.get_full_corpus("intro-chapter")
|
|
208
|
+
print(corpus.content, corpus.metadata, len(corpus.documents))
|
|
235
209
|
```
|
|
236
210
|
|
|
237
211
|
## Concept Reference
|
|
@@ -241,8 +215,11 @@ with db_manager.get_session() as session:
|
|
|
241
215
|
- **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
|
|
242
216
|
- **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
|
|
243
217
|
- **`BaseDocumentMetadata`**: Base schema for structured document metadata
|
|
244
|
-
- **`
|
|
245
|
-
- **`
|
|
218
|
+
- **`BaseEmbeddingProvider`**: Interface to implement for your embedding model
|
|
219
|
+
- **`BaseCorpusManager`**: Splits, embeds and upserts corpora (`insert_corpus`, `get_full_corpus`)
|
|
220
|
+
- **`BaseSearchClient`**: Semantic, keyword and metadata search (`search`)
|
|
221
|
+
- **`DocumentService`**: Bundles a corpus manager and a search client over one session
|
|
222
|
+
- **`DatabaseManager`** / **`DocumentDatabaseManager`**: Connection, schema and table setup
|
|
246
223
|
|
|
247
224
|
## Testing
|
|
248
225
|
|
|
@@ -276,7 +253,8 @@ python -m unittest discover -s integ-tests
|
|
|
276
253
|
|
|
277
254
|
```bash
|
|
278
255
|
pip install -e .[dev,test]
|
|
279
|
-
|
|
256
|
+
ruff check . && ruff format .
|
|
257
|
+
ty check .
|
|
280
258
|
```
|
|
281
259
|
|
|
282
260
|
## License
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import hashlib
|
|
2
|
+
import json
|
|
1
3
|
from datetime import UTC, datetime
|
|
2
4
|
from typing import Any, Self
|
|
3
5
|
from uuid import UUID as UuidLiteral
|
|
@@ -86,6 +88,12 @@ class BaseDocument(Base):
|
|
|
86
88
|
Use this field for auditing and diagnosing mixed-model corpora, not for runtime dispatch.
|
|
87
89
|
"""
|
|
88
90
|
|
|
91
|
+
content_hash = Column(String(64), nullable=True)
|
|
92
|
+
"""
|
|
93
|
+
Hex sha256 of `content` + `embedding_config`, see `compute_content_hash()`.
|
|
94
|
+
Lets upserts reuse a stored embedding when a chunk is unchanged. NULL for rows that pre-date this column.
|
|
95
|
+
"""
|
|
96
|
+
|
|
89
97
|
# Audit fields
|
|
90
98
|
created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(UTC))
|
|
91
99
|
updated_at = Column(
|
|
@@ -165,6 +173,7 @@ class BaseDocument(Base):
|
|
|
165
173
|
content=content,
|
|
166
174
|
embedding=embedding,
|
|
167
175
|
embedding_config=embedding_config,
|
|
176
|
+
content_hash=cls.compute_content_hash(content, embedding_config),
|
|
168
177
|
title=optional_props.title,
|
|
169
178
|
document_metadata=metadata or {},
|
|
170
179
|
collection=optional_props.collection,
|
|
@@ -172,6 +181,15 @@ class BaseDocument(Base):
|
|
|
172
181
|
language=optional_props.language,
|
|
173
182
|
)
|
|
174
183
|
|
|
184
|
+
@classmethod
|
|
185
|
+
def compute_content_hash(cls, content: str, embedding_config: dict[str, Any] | None) -> str:
|
|
186
|
+
"""
|
|
187
|
+
Hash of everything that determines a chunk's embedding vector. Metadata and other props are excluded.
|
|
188
|
+
Override to change what counts as "the same chunk".
|
|
189
|
+
"""
|
|
190
|
+
payload = json.dumps([content, embedding_config], sort_keys=True, ensure_ascii=False)
|
|
191
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
192
|
+
|
|
175
193
|
@classmethod
|
|
176
194
|
def get_embedding_index(cls, table_name: str) -> Index:
|
|
177
195
|
"""Override this method to customize the embedding index."""
|
|
@@ -1,11 +1,13 @@
|
|
|
1
|
+
import hashlib
|
|
2
|
+
import json
|
|
1
3
|
from abc import ABC
|
|
2
4
|
from dataclasses import dataclass
|
|
3
5
|
from logging import getLogger
|
|
4
|
-
from typing import Any
|
|
6
|
+
from typing import Any, cast
|
|
5
7
|
from uuid import UUID, uuid4
|
|
6
8
|
|
|
7
9
|
from pydantic import BaseModel, Field
|
|
8
|
-
from sqlalchemy.orm import Session
|
|
10
|
+
from sqlalchemy.orm import Session, load_only
|
|
9
11
|
|
|
10
12
|
from pgvector_template.core.document import (
|
|
11
13
|
BaseDocument,
|
|
@@ -45,6 +47,22 @@ class Corpus:
|
|
|
45
47
|
documents: list[BaseDocument]
|
|
46
48
|
|
|
47
49
|
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class UpsertResult:
|
|
52
|
+
"""Outcome of `BaseCorpusManager.insert_corpus()`."""
|
|
53
|
+
|
|
54
|
+
corpus_hash: str
|
|
55
|
+
"""Hex sha256 over the ordered chunk hashes. Computed in memory, not stored."""
|
|
56
|
+
skipped: bool
|
|
57
|
+
"""True when the stored corpus already matched, so nothing was embedded or written."""
|
|
58
|
+
embedded: int
|
|
59
|
+
"""Chunks sent to the embedding provider."""
|
|
60
|
+
reused: int
|
|
61
|
+
"""Chunks whose stored vector was reused."""
|
|
62
|
+
total: int
|
|
63
|
+
"""Number of chunks in the corpus."""
|
|
64
|
+
|
|
65
|
+
|
|
48
66
|
class BaseCorpusManager(ABC):
|
|
49
67
|
"""
|
|
50
68
|
Template class for `Corpus` & `Document` management operations.
|
|
@@ -121,31 +139,152 @@ class BaseCorpusManager(ABC):
|
|
|
121
139
|
corpus_id: UUID | str | None = None,
|
|
122
140
|
update_if_exists: bool = True,
|
|
123
141
|
**kwargs,
|
|
124
|
-
) ->
|
|
142
|
+
) -> UpsertResult:
|
|
125
143
|
"""
|
|
126
144
|
Insert a new `Corpus`, which will be split into 1-or-more `Document`s, depending on its length.
|
|
127
145
|
Each `Document` chunk shall have its own embedding vector, but reference the parent corpus_id.
|
|
128
146
|
|
|
147
|
+
When `update_if_exists` is set and the corpus already exists, chunks whose content hash is
|
|
148
|
+
unchanged reuse their stored vector, and only the rest are embedded. If the stored rows already
|
|
149
|
+
match the new ones entirely, nothing is written.
|
|
150
|
+
|
|
129
151
|
Args:
|
|
130
152
|
content: The text content to be inserted as a corpus
|
|
131
153
|
corpus_metadata: Dictionary of metadata associated with the corpus
|
|
132
154
|
optional_props: Optional properties for the documents (title, collection, etc.)
|
|
133
155
|
|
|
134
156
|
Returns:
|
|
135
|
-
|
|
157
|
+
UpsertResult: Whether the write was skipped, and how many chunks were embedded vs reused
|
|
136
158
|
"""
|
|
137
159
|
corpus_id = self._generate_corpus_id(corpus_id)
|
|
138
160
|
document_contents = self._split_corpus(content)
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
optional_props
|
|
146
|
-
|
|
161
|
+
if not document_contents:
|
|
162
|
+
return UpsertResult(self._corpus_hash([]), False, 0, 0, 0)
|
|
163
|
+
|
|
164
|
+
total = len(document_contents)
|
|
165
|
+
# vectors are filled in by `_attach_embeddings`
|
|
166
|
+
documents = self._create_documents(
|
|
167
|
+
corpus_id, document_contents, [[]] * total, corpus_metadata, optional_props
|
|
168
|
+
)
|
|
169
|
+
chunk_hashes = [cast(str, d.content_hash) for d in documents]
|
|
170
|
+
corpus_hash = self._corpus_hash(chunk_hashes)
|
|
171
|
+
|
|
172
|
+
reusable: dict[str, Any] = {}
|
|
173
|
+
if update_if_exists and (existing := self._load_existing_chunks(corpus_id)):
|
|
174
|
+
if self._is_unchanged(corpus_id, existing, documents):
|
|
175
|
+
logger.info(f"Corpus {corpus_id} unchanged, skipping ({total} chunks)")
|
|
176
|
+
self.session.commit() # release the read transaction
|
|
177
|
+
return UpsertResult(corpus_hash, True, 0, total, total)
|
|
178
|
+
reusable = self._load_reusable_vectors(corpus_id, set(chunk_hashes))
|
|
179
|
+
|
|
180
|
+
embedded = self._attach_embeddings(documents, document_contents, chunk_hashes, reusable)
|
|
181
|
+
self._replace_corpus(corpus_id, documents, delete_existing=update_if_exists)
|
|
182
|
+
return UpsertResult(corpus_hash, False, embedded, total - embedded, total)
|
|
183
|
+
|
|
184
|
+
def _attach_embeddings(
|
|
185
|
+
self,
|
|
186
|
+
documents: list[BaseDocument],
|
|
187
|
+
document_contents: list[str],
|
|
188
|
+
chunk_hashes: list[str],
|
|
189
|
+
reusable: dict[str, Any],
|
|
190
|
+
) -> int:
|
|
191
|
+
"""Set each document's vector from `reusable` or a fresh embedding. Returns the number embedded."""
|
|
192
|
+
misses = [i for i, h in enumerate(chunk_hashes) if h not in reusable]
|
|
193
|
+
miss_contents = [document_contents[i] for i in misses]
|
|
194
|
+
miss_embeddings: list[list[float]] = []
|
|
195
|
+
if misses:
|
|
196
|
+
# end the read transaction so the connection is not held idle during the remote call
|
|
197
|
+
self.session.commit()
|
|
198
|
+
miss_embeddings = self.embedding_provider.embed_batch(miss_contents)
|
|
199
|
+
self._validate_inputs(miss_contents, miss_embeddings)
|
|
200
|
+
fresh = dict(zip(misses, miss_embeddings))
|
|
201
|
+
for i, doc in enumerate(documents):
|
|
202
|
+
doc.embedding = fresh[i] if i in fresh else reusable[chunk_hashes[i]] # ty: ignore[invalid-assignment]
|
|
203
|
+
return len(misses)
|
|
204
|
+
|
|
205
|
+
@staticmethod
|
|
206
|
+
def _corpus_hash(chunk_hashes: list[str]) -> str:
|
|
207
|
+
return hashlib.sha256(json.dumps(chunk_hashes).encode("utf-8")).hexdigest()
|
|
208
|
+
|
|
209
|
+
@staticmethod
|
|
210
|
+
def _fingerprint(doc: BaseDocument) -> tuple:
|
|
211
|
+
"""Everything besides the vector that an upsert must keep in sync for a chunk."""
|
|
212
|
+
return (
|
|
213
|
+
doc.chunk_index,
|
|
214
|
+
doc.content_hash,
|
|
215
|
+
doc.document_metadata,
|
|
216
|
+
doc.title,
|
|
217
|
+
doc.collection,
|
|
218
|
+
doc.origin_url,
|
|
219
|
+
doc.language,
|
|
147
220
|
)
|
|
148
221
|
|
|
222
|
+
def _load_existing_chunks(self, corpus_id: UUID | str) -> list[BaseDocument]:
|
|
223
|
+
"""Stored chunks of a corpus ordered by `chunk_index`, without content or vectors."""
|
|
224
|
+
cls = self.config.document_cls
|
|
225
|
+
return (
|
|
226
|
+
self.session.query(cls)
|
|
227
|
+
.options(
|
|
228
|
+
load_only(
|
|
229
|
+
cls.chunk_index,
|
|
230
|
+
cls.content_hash,
|
|
231
|
+
cls.document_metadata,
|
|
232
|
+
cls.title,
|
|
233
|
+
cls.collection,
|
|
234
|
+
cls.origin_url,
|
|
235
|
+
cls.language,
|
|
236
|
+
cls.is_deleted,
|
|
237
|
+
)
|
|
238
|
+
)
|
|
239
|
+
.filter(cls.corpus_id == corpus_id)
|
|
240
|
+
.order_by(cls.chunk_index)
|
|
241
|
+
.all()
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
def _is_unchanged(
|
|
245
|
+
self, corpus_id: UUID | str, existing: list[BaseDocument], new: list[BaseDocument]
|
|
246
|
+
) -> bool:
|
|
247
|
+
"""True if the stored rows match the new ones and every stored row has a vector."""
|
|
248
|
+
if any(d.is_deleted for d in existing):
|
|
249
|
+
return False
|
|
250
|
+
if [self._fingerprint(d) for d in existing] != [self._fingerprint(d) for d in new]:
|
|
251
|
+
return False
|
|
252
|
+
return not self._has_missing_vectors(corpus_id)
|
|
253
|
+
|
|
254
|
+
def _has_missing_vectors(self, corpus_id: UUID | str) -> bool:
|
|
255
|
+
cls = self.config.document_cls
|
|
256
|
+
query = self.session.query(cls.id).filter(
|
|
257
|
+
cls.corpus_id == corpus_id, cls.embedding.is_(None)
|
|
258
|
+
)
|
|
259
|
+
return query.first() is not None
|
|
260
|
+
|
|
261
|
+
def _load_reusable_vectors(self, corpus_id: UUID | str, hashes: set[str]) -> dict[str, Any]:
|
|
262
|
+
"""Map `content_hash -> embedding` for stored chunks of this corpus whose hash is in `hashes`."""
|
|
263
|
+
cls = self.config.document_cls
|
|
264
|
+
rows = (
|
|
265
|
+
self.session.query(cls.content_hash, cls.embedding)
|
|
266
|
+
.filter(
|
|
267
|
+
cls.corpus_id == corpus_id,
|
|
268
|
+
cls.content_hash.in_(hashes),
|
|
269
|
+
cls.embedding.is_not(None),
|
|
270
|
+
)
|
|
271
|
+
.all()
|
|
272
|
+
)
|
|
273
|
+
return dict(rows)
|
|
274
|
+
|
|
275
|
+
def _replace_corpus(
|
|
276
|
+
self, corpus_id: UUID | str, documents: list[BaseDocument], delete_existing: bool
|
|
277
|
+
) -> None:
|
|
278
|
+
"""Delete (optionally) and insert in a single transaction."""
|
|
279
|
+
try:
|
|
280
|
+
if delete_existing:
|
|
281
|
+
self._delete_existing_corpus(corpus_id)
|
|
282
|
+
self.session.add_all(documents)
|
|
283
|
+
self.session.commit()
|
|
284
|
+
except Exception:
|
|
285
|
+
self.session.rollback()
|
|
286
|
+
raise
|
|
287
|
+
|
|
149
288
|
def insert_documents(
|
|
150
289
|
self,
|
|
151
290
|
corpus_id: UUID | str,
|
|
@@ -180,11 +319,7 @@ class BaseCorpusManager(ABC):
|
|
|
180
319
|
corpus_id, document_contents, document_embeddings, corpus_metadata, optional_props
|
|
181
320
|
)
|
|
182
321
|
|
|
183
|
-
|
|
184
|
-
self._delete_existing_corpus(corpus_id)
|
|
185
|
-
|
|
186
|
-
self.session.add_all(documents_to_insert)
|
|
187
|
-
self.session.commit()
|
|
322
|
+
self._replace_corpus(corpus_id, documents_to_insert, delete_existing=update_if_exists)
|
|
188
323
|
return len(documents_to_insert)
|
|
189
324
|
|
|
190
325
|
def _validate_inputs(
|
|
@@ -2,7 +2,7 @@ from collections.abc import Generator
|
|
|
2
2
|
from contextlib import contextmanager
|
|
3
3
|
from logging import getLogger
|
|
4
4
|
|
|
5
|
-
from sqlalchemy import Engine, create_engine, text
|
|
5
|
+
from sqlalchemy import Engine, create_engine, inspect, text
|
|
6
6
|
from sqlalchemy.ext.declarative import DeclarativeMeta
|
|
7
7
|
from sqlalchemy.orm import Session, sessionmaker
|
|
8
8
|
|
|
@@ -46,8 +46,40 @@ class DatabaseManager:
|
|
|
46
46
|
for table in base_class.metadata.tables.values():
|
|
47
47
|
table.schema = schema_name
|
|
48
48
|
base_class.metadata.create_all(self._require_engine(), checkfirst=True)
|
|
49
|
+
self._add_missing_columns(base_class, schema_name)
|
|
49
50
|
self.logger.info(f"Created tables for schema: {schema_name}")
|
|
50
51
|
|
|
52
|
+
def _add_missing_columns(self, base_class: type[DeclarativeMeta], schema_name: str) -> None:
|
|
53
|
+
"""
|
|
54
|
+
`create_all` never alters existing tables. Add nullable columns that are defined on the model but
|
|
55
|
+
missing from the database. Non-nullable columns can't be added safely to populated tables; they are skipped.
|
|
56
|
+
DDL is only issued for columns that are actually missing, so no table ownership is needed otherwise.
|
|
57
|
+
"""
|
|
58
|
+
tables = list(base_class.metadata.tables.values())
|
|
59
|
+
if not tables:
|
|
60
|
+
return
|
|
61
|
+
engine = self._require_engine()
|
|
62
|
+
inspector = inspect(engine)
|
|
63
|
+
preparer = engine.dialect.identifier_preparer
|
|
64
|
+
for table in tables:
|
|
65
|
+
existing = {c["name"] for c in inspector.get_columns(table.name, schema=schema_name)}
|
|
66
|
+
for column in table.columns:
|
|
67
|
+
if column.name in existing:
|
|
68
|
+
continue
|
|
69
|
+
if not column.nullable:
|
|
70
|
+
self.logger.warning(
|
|
71
|
+
f"Column {table.name}.{column.name} is missing and non-nullable; add it manually"
|
|
72
|
+
)
|
|
73
|
+
continue
|
|
74
|
+
ddl = (
|
|
75
|
+
f"ALTER TABLE {preparer.quote_schema(schema_name)}.{preparer.quote(table.name)} "
|
|
76
|
+
f"ADD COLUMN IF NOT EXISTS {preparer.quote(column.name)} "
|
|
77
|
+
f"{column.type.compile(engine.dialect)}"
|
|
78
|
+
)
|
|
79
|
+
with engine.begin() as conn:
|
|
80
|
+
conn.execute(text(ddl))
|
|
81
|
+
self.logger.info(f"Added column {schema_name}.{table.name}.{column.name}")
|
|
82
|
+
|
|
51
83
|
def _ensure_pgvector_extension(self):
|
|
52
84
|
"""Ensure pgvector extension is available"""
|
|
53
85
|
with self._require_engine().connect() as conn:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pgvector-template
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6
|
|
4
4
|
Summary: Template library for flexible PGVector RAG implementations
|
|
5
5
|
Author-email: DL <v49t9zpqd@mozmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -39,10 +39,10 @@ PGVector-Template provides a robust foundation for implementing vector-based doc
|
|
|
39
39
|
## Quick Start
|
|
40
40
|
|
|
41
41
|
```python
|
|
42
|
-
from pgvector_template import
|
|
42
|
+
from pgvector_template import DocumentDatabaseManager
|
|
43
43
|
from pgvector_template.core.document import BaseDocument
|
|
44
|
-
from
|
|
45
|
-
from
|
|
44
|
+
from pgvector_template.models.search import SearchQuery
|
|
45
|
+
from pgvector_template.service.document_service import DocumentService, DocumentServiceConfig
|
|
46
46
|
|
|
47
47
|
|
|
48
48
|
# 1. Define your document model
|
|
@@ -50,29 +50,25 @@ class MyDocument(BaseDocument):
|
|
|
50
50
|
__tablename__ = "my_documents"
|
|
51
51
|
|
|
52
52
|
|
|
53
|
-
# 2. Set up database
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
53
|
+
# 2. Set up the database: connects, enables pgvector, creates the schema and tables
|
|
54
|
+
db = DocumentDatabaseManager(
|
|
55
|
+
database_url="postgresql://user:pass@localhost/mydb",
|
|
56
|
+
schema_suffix="my_kb",
|
|
57
|
+
document_classes=[MyDocument],
|
|
58
|
+
)
|
|
59
|
+
db.setup()
|
|
57
60
|
|
|
58
|
-
# 3. Create
|
|
59
|
-
|
|
61
|
+
# 3. Create a service. `my_embedding_provider` is your `BaseEmbeddingProvider` subclass
|
|
62
|
+
config = DocumentServiceConfig(document_cls=MyDocument, embedding_provider=my_embedding_provider)
|
|
60
63
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
doc = MyDocument.from_props(
|
|
64
|
-
corpus_id=uuid4(),
|
|
65
|
-
chunk_index=0,
|
|
66
|
-
content="Your document content here",
|
|
67
|
-
embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
|
|
68
|
-
)
|
|
69
|
-
doc_manager.insert_document(session, doc)
|
|
64
|
+
with db.get_session() as session:
|
|
65
|
+
service = DocumentService(session, config)
|
|
70
66
|
|
|
71
|
-
#
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
)
|
|
67
|
+
# 4. Insert a corpus: it is split into chunks, embedded and stored
|
|
68
|
+
service.corpus_manager.insert_corpus("Your document content here", {}, corpus_id="doc-1")
|
|
69
|
+
|
|
70
|
+
# 5. Search
|
|
71
|
+
results = service.search_client.search(SearchQuery(text="your query", limit=5))
|
|
76
72
|
```
|
|
77
73
|
|
|
78
74
|
## Key Concepts
|
|
@@ -91,6 +87,7 @@ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all
|
|
|
91
87
|
- **Metadata Management**: JSON-based flexible metadata with GIN indexing
|
|
92
88
|
- **Collection Support**: Organize documents into logical collections
|
|
93
89
|
- **Chunk Management**: Handle long content by chunking into retrievable documents
|
|
90
|
+
- **Smart Upserts**: Re-inserting a corpus only re-embeds chunks whose content changed, and skips the write entirely if nothing did
|
|
94
91
|
- **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
|
|
95
92
|
- **Type Safety**: Full Pydantic validation and type hints
|
|
96
93
|
- **Production Ready**: Comprehensive testing and error handling
|
|
@@ -144,34 +141,14 @@ CREATE EXTENSION IF NOT EXISTS vector;
|
|
|
144
141
|
3. **Set up your connection:**
|
|
145
142
|
|
|
146
143
|
```python
|
|
147
|
-
from sqlalchemy import create_engine
|
|
148
144
|
from pgvector_template import DatabaseManager
|
|
149
145
|
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
db_manager = DatabaseManager(engine)
|
|
153
|
-
|
|
154
|
-
# Option 2: From environment variable
|
|
155
|
-
import os
|
|
156
|
-
|
|
157
|
-
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
158
|
-
db_manager = DatabaseManager(engine)
|
|
146
|
+
db_manager = DatabaseManager("postgresql://user:password@localhost:5432/mydb")
|
|
147
|
+
db_manager.initialize() # connects and enables the vector extension
|
|
159
148
|
```
|
|
160
149
|
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
```python
|
|
164
|
-
from sqlalchemy import create_engine
|
|
165
|
-
from sqlalchemy.pool import QueuePool
|
|
166
|
-
|
|
167
|
-
engine = create_engine(
|
|
168
|
-
"postgresql://user:password@localhost:5432/mydb",
|
|
169
|
-
poolclass=QueuePool,
|
|
170
|
-
pool_size=10,
|
|
171
|
-
max_overflow=20,
|
|
172
|
-
pool_pre_ping=True,
|
|
173
|
-
)
|
|
174
|
-
```
|
|
150
|
+
`create_tables` also adds nullable model columns that are missing from existing tables (for example `content_hash`).
|
|
151
|
+
Only the column name and type are applied: indexes, foreign keys, unique constraints and server defaults are not. This is a stopgap until proper migrations are added.
|
|
175
152
|
|
|
176
153
|
### Environment Variables
|
|
177
154
|
|
|
@@ -198,70 +175,67 @@ class MyDocument(BaseDocument):
|
|
|
198
175
|
__tablename__ = "my_documents"
|
|
199
176
|
```
|
|
200
177
|
|
|
201
|
-
### 2. Insert
|
|
178
|
+
### 2. Insert a Corpus
|
|
202
179
|
|
|
203
180
|
```python
|
|
204
181
|
from pgvector_template.core.document import BaseDocumentOptionalProps
|
|
205
|
-
from uuid import uuid4
|
|
206
182
|
|
|
207
|
-
# Create document with metadata
|
|
208
|
-
metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
|
|
209
183
|
optional_props = BaseDocumentOptionalProps(
|
|
210
184
|
title="Chapter 1: Introduction",
|
|
211
185
|
collection="textbooks",
|
|
212
186
|
language="en",
|
|
213
|
-
tags=["education", "intro"],
|
|
214
187
|
)
|
|
215
188
|
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
)
|
|
189
|
+
with db.get_session() as session:
|
|
190
|
+
service = DocumentService(session, config)
|
|
191
|
+
result = service.corpus_manager.insert_corpus(
|
|
192
|
+
"This is the document content...",
|
|
193
|
+
{"source_type": "pdf", "author": "John Doe"}, # corpus metadata
|
|
194
|
+
optional_props,
|
|
195
|
+
corpus_id="intro-chapter",
|
|
196
|
+
)
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
To insert chunks you have already split and embedded, use `insert_documents()`.
|
|
200
|
+
|
|
201
|
+
Calling `insert_corpus()` again with the same `corpus_id` replaces the corpus. Unchanged chunks reuse their stored vector, and the write is skipped if nothing changed:
|
|
224
202
|
|
|
225
|
-
|
|
226
|
-
|
|
203
|
+
```python
|
|
204
|
+
result = corpus_manager.insert_corpus(text, {"source": "pdf"}, corpus_id="handbook")
|
|
205
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=12, reused=0, total=12)
|
|
206
|
+
|
|
207
|
+
result = corpus_manager.insert_corpus(edited_text, {"source": "pdf"}, corpus_id="handbook")
|
|
208
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=1, reused=11, total=12)
|
|
227
209
|
```
|
|
228
210
|
|
|
211
|
+
Rows written before `content_hash` existed have no hash, so their first upsert re-embeds them.
|
|
212
|
+
|
|
229
213
|
### 3. Search Documents
|
|
230
214
|
|
|
231
215
|
```python
|
|
232
|
-
|
|
233
|
-
with db_manager.get_session() as session:
|
|
234
|
-
results = doc_manager.search_similar(
|
|
235
|
-
session=session, document_cls=MyDocument, query_embedding=query_vector, limit=10
|
|
236
|
-
)
|
|
237
|
-
|
|
238
|
-
# Search with filters
|
|
239
|
-
from pgvector_template.models.search import MetadataFilter
|
|
216
|
+
from pgvector_template.models.search import MetadataFilter, SearchQuery
|
|
240
217
|
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
MetadataFilter(key="author", value="John Doe"),
|
|
244
|
-
]
|
|
218
|
+
# Semantic search
|
|
219
|
+
results = service.search_client.search(SearchQuery(text="how do I reset my password", limit=10))
|
|
245
220
|
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
221
|
+
# Keyword search combined with a metadata filter
|
|
222
|
+
results = service.search_client.search(
|
|
223
|
+
SearchQuery(
|
|
224
|
+
keywords=["password", "reset"],
|
|
225
|
+
metadata_filters=[MetadataFilter(field_name="source_type", condition="eq", value="pdf")],
|
|
226
|
+
limit=10,
|
|
227
|
+
)
|
|
253
228
|
)
|
|
229
|
+
|
|
230
|
+
for r in results:
|
|
231
|
+
print(r.score, r.document.content)
|
|
254
232
|
```
|
|
255
233
|
|
|
256
|
-
### 4.
|
|
234
|
+
### 4. Read a Corpus Back
|
|
257
235
|
|
|
258
236
|
```python
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
docs = doc_manager.get_documents_by_collection(session, MyDocument, "textbooks")
|
|
262
|
-
|
|
263
|
-
# Get all chunks from a corpus
|
|
264
|
-
corpus_docs = doc_manager.get_documents_by_corpus_id(session, MyDocument, corpus_id)
|
|
237
|
+
corpus = service.corpus_manager.get_full_corpus("intro-chapter")
|
|
238
|
+
print(corpus.content, corpus.metadata, len(corpus.documents))
|
|
265
239
|
```
|
|
266
240
|
|
|
267
241
|
## Concept Reference
|
|
@@ -271,8 +245,11 @@ with db_manager.get_session() as session:
|
|
|
271
245
|
- **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
|
|
272
246
|
- **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
|
|
273
247
|
- **`BaseDocumentMetadata`**: Base schema for structured document metadata
|
|
274
|
-
- **`
|
|
275
|
-
- **`
|
|
248
|
+
- **`BaseEmbeddingProvider`**: Interface to implement for your embedding model
|
|
249
|
+
- **`BaseCorpusManager`**: Splits, embeds and upserts corpora (`insert_corpus`, `get_full_corpus`)
|
|
250
|
+
- **`BaseSearchClient`**: Semantic, keyword and metadata search (`search`)
|
|
251
|
+
- **`DocumentService`**: Bundles a corpus manager and a search client over one session
|
|
252
|
+
- **`DatabaseManager`** / **`DocumentDatabaseManager`**: Connection, schema and table setup
|
|
276
253
|
|
|
277
254
|
## Testing
|
|
278
255
|
|
|
@@ -306,7 +283,8 @@ python -m unittest discover -s integ-tests
|
|
|
306
283
|
|
|
307
284
|
```bash
|
|
308
285
|
pip install -e .[dev,test]
|
|
309
|
-
|
|
286
|
+
ruff check . && ruff format .
|
|
287
|
+
ty check .
|
|
310
288
|
```
|
|
311
289
|
|
|
312
290
|
## License
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pgvector-template"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.6"
|
|
8
8
|
description = "Template library for flexible PGVector RAG implementations"
|
|
9
9
|
authors = [{ name="DL", email="v49t9zpqd@mozmail.com" }]
|
|
10
10
|
license = { text = "MIT" }
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/service/document_service.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template/utils/metadata_filter.py
RENAMED
|
File without changes
|
|
File without changes
|
{pgvector_template-0.5.1 → pgvector_template-0.6}/pgvector_template.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|