pgvector-template 0.5__tar.gz → 0.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pgvector_template-0.5 → pgvector_template-0.6}/PKG-INFO +77 -101
- {pgvector_template-0.5 → pgvector_template-0.6}/README.md +76 -100
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/__init__.py +6 -11
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/document.py +33 -17
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/manager.py +156 -22
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/search.py +9 -9
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/db/connection.py +36 -4
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/db/document_db.py +1 -3
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/models/__init__.py +4 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/models/search.py +1 -1
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/service/document_service.py +5 -5
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/utils/metadata_filter.py +4 -6
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/PKG-INFO +77 -101
- {pgvector_template-0.5 → pgvector_template-0.6}/pyproject.toml +1 -1
- {pgvector_template-0.5 → pgvector_template-0.6}/LICENSE +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/__init__.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/embedder.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/db/__init__.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/service/__init__.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/types.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/utils/__init__.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/SOURCES.txt +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/dependency_links.txt +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/requires.txt +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/top_level.txt +0 -0
- {pgvector_template-0.5 → pgvector_template-0.6}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pgvector-template
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6
|
|
4
4
|
Summary: Template library for flexible PGVector RAG implementations
|
|
5
5
|
Author-email: DL <v49t9zpqd@mozmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -39,38 +39,36 @@ PGVector-Template provides a robust foundation for implementing vector-based doc
|
|
|
39
39
|
## Quick Start
|
|
40
40
|
|
|
41
41
|
```python
|
|
42
|
-
from pgvector_template import
|
|
42
|
+
from pgvector_template import DocumentDatabaseManager
|
|
43
43
|
from pgvector_template.core.document import BaseDocument
|
|
44
|
-
from
|
|
45
|
-
from
|
|
44
|
+
from pgvector_template.models.search import SearchQuery
|
|
45
|
+
from pgvector_template.service.document_service import DocumentService, DocumentServiceConfig
|
|
46
|
+
|
|
46
47
|
|
|
47
48
|
# 1. Define your document model
|
|
48
49
|
class MyDocument(BaseDocument):
|
|
49
50
|
__tablename__ = "my_documents"
|
|
50
51
|
|
|
51
|
-
# 2. Set up database connection
|
|
52
|
-
engine = create_engine("postgresql://user:pass@localhost/mydb")
|
|
53
|
-
db_manager = DatabaseManager(engine)
|
|
54
|
-
db_manager.create_tables([MyDocument])
|
|
55
|
-
|
|
56
|
-
# 3. Create document manager
|
|
57
|
-
doc_manager = DocumentDatabaseManager(db_manager)
|
|
58
|
-
|
|
59
|
-
# 4. Insert a document
|
|
60
|
-
with db_manager.get_session() as session:
|
|
61
|
-
doc = MyDocument.from_props(
|
|
62
|
-
corpus_id=uuid4(),
|
|
63
|
-
chunk_index=0,
|
|
64
|
-
content="Your document content here",
|
|
65
|
-
embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
|
|
66
|
-
)
|
|
67
|
-
doc_manager.insert_document(session, doc)
|
|
68
52
|
|
|
69
|
-
#
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
53
|
+
# 2. Set up the database: connects, enables pgvector, creates the schema and tables
|
|
54
|
+
db = DocumentDatabaseManager(
|
|
55
|
+
database_url="postgresql://user:pass@localhost/mydb",
|
|
56
|
+
schema_suffix="my_kb",
|
|
57
|
+
document_classes=[MyDocument],
|
|
58
|
+
)
|
|
59
|
+
db.setup()
|
|
60
|
+
|
|
61
|
+
# 3. Create a service. `my_embedding_provider` is your `BaseEmbeddingProvider` subclass
|
|
62
|
+
config = DocumentServiceConfig(document_cls=MyDocument, embedding_provider=my_embedding_provider)
|
|
63
|
+
|
|
64
|
+
with db.get_session() as session:
|
|
65
|
+
service = DocumentService(session, config)
|
|
66
|
+
|
|
67
|
+
# 4. Insert a corpus: it is split into chunks, embedded and stored
|
|
68
|
+
service.corpus_manager.insert_corpus("Your document content here", {}, corpus_id="doc-1")
|
|
69
|
+
|
|
70
|
+
# 5. Search
|
|
71
|
+
results = service.search_client.search(SearchQuery(text="your query", limit=5))
|
|
74
72
|
```
|
|
75
73
|
|
|
76
74
|
## Key Concepts
|
|
@@ -89,6 +87,7 @@ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all
|
|
|
89
87
|
- **Metadata Management**: JSON-based flexible metadata with GIN indexing
|
|
90
88
|
- **Collection Support**: Organize documents into logical collections
|
|
91
89
|
- **Chunk Management**: Handle long content by chunking into retrievable documents
|
|
90
|
+
- **Smart Upserts**: Re-inserting a corpus only re-embeds chunks whose content changed, and skips the write entirely if nothing did
|
|
92
91
|
- **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
|
|
93
92
|
- **Type Safety**: Full Pydantic validation and type hints
|
|
94
93
|
- **Production Ready**: Comprehensive testing and error handling
|
|
@@ -142,33 +141,14 @@ CREATE EXTENSION IF NOT EXISTS vector;
|
|
|
142
141
|
3. **Set up your connection:**
|
|
143
142
|
|
|
144
143
|
```python
|
|
145
|
-
from sqlalchemy import create_engine
|
|
146
144
|
from pgvector_template import DatabaseManager
|
|
147
145
|
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
db_manager = DatabaseManager(engine)
|
|
151
|
-
|
|
152
|
-
# Option 2: From environment variable
|
|
153
|
-
import os
|
|
154
|
-
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
155
|
-
db_manager = DatabaseManager(engine)
|
|
146
|
+
db_manager = DatabaseManager("postgresql://user:password@localhost:5432/mydb")
|
|
147
|
+
db_manager.initialize() # connects and enables the vector extension
|
|
156
148
|
```
|
|
157
149
|
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
```python
|
|
161
|
-
from sqlalchemy import create_engine
|
|
162
|
-
from sqlalchemy.pool import QueuePool
|
|
163
|
-
|
|
164
|
-
engine = create_engine(
|
|
165
|
-
"postgresql://user:password@localhost:5432/mydb",
|
|
166
|
-
poolclass=QueuePool,
|
|
167
|
-
pool_size=10,
|
|
168
|
-
max_overflow=20,
|
|
169
|
-
pool_pre_ping=True,
|
|
170
|
-
)
|
|
171
|
-
```
|
|
150
|
+
`create_tables` also adds nullable model columns that are missing from existing tables (for example `content_hash`).
|
|
151
|
+
Only the column name and type are applied: indexes, foreign keys, unique constraints and server defaults are not. This is a stopgap until proper migrations are added.
|
|
172
152
|
|
|
173
153
|
### Environment Variables
|
|
174
154
|
|
|
@@ -185,85 +165,77 @@ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
|
|
|
185
165
|
from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
|
|
186
166
|
from pydantic import Field
|
|
187
167
|
|
|
168
|
+
|
|
188
169
|
class MyDocumentMetadata(BaseDocumentMetadata):
|
|
189
170
|
source_type: str = Field(..., description="Type of source document")
|
|
190
171
|
author: str = Field(default="unknown", description="Document author")
|
|
191
172
|
|
|
173
|
+
|
|
192
174
|
class MyDocument(BaseDocument):
|
|
193
175
|
__tablename__ = "my_documents"
|
|
194
176
|
```
|
|
195
177
|
|
|
196
|
-
### 2. Insert
|
|
178
|
+
### 2. Insert a Corpus
|
|
197
179
|
|
|
198
180
|
```python
|
|
199
181
|
from pgvector_template.core.document import BaseDocumentOptionalProps
|
|
200
|
-
from uuid import uuid4
|
|
201
182
|
|
|
202
|
-
# Create document with metadata
|
|
203
|
-
metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
|
|
204
183
|
optional_props = BaseDocumentOptionalProps(
|
|
205
184
|
title="Chapter 1: Introduction",
|
|
206
185
|
collection="textbooks",
|
|
207
186
|
language="en",
|
|
208
|
-
tags=["education", "intro"]
|
|
209
187
|
)
|
|
210
188
|
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
)
|
|
189
|
+
with db.get_session() as session:
|
|
190
|
+
service = DocumentService(session, config)
|
|
191
|
+
result = service.corpus_manager.insert_corpus(
|
|
192
|
+
"This is the document content...",
|
|
193
|
+
{"source_type": "pdf", "author": "John Doe"}, # corpus metadata
|
|
194
|
+
optional_props,
|
|
195
|
+
corpus_id="intro-chapter",
|
|
196
|
+
)
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
To insert chunks you have already split and embedded, use `insert_documents()`.
|
|
219
200
|
|
|
220
|
-
|
|
221
|
-
|
|
201
|
+
Calling `insert_corpus()` again with the same `corpus_id` replaces the corpus. Unchanged chunks reuse their stored vector, and the write is skipped if nothing changed:
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
result = corpus_manager.insert_corpus(text, {"source": "pdf"}, corpus_id="handbook")
|
|
205
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=12, reused=0, total=12)
|
|
206
|
+
|
|
207
|
+
result = corpus_manager.insert_corpus(edited_text, {"source": "pdf"}, corpus_id="handbook")
|
|
208
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=1, reused=11, total=12)
|
|
222
209
|
```
|
|
223
210
|
|
|
211
|
+
Rows written before `content_hash` existed have no hash, so their first upsert re-embeds them.
|
|
212
|
+
|
|
224
213
|
### 3. Search Documents
|
|
225
214
|
|
|
226
215
|
```python
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
document_cls=MyDocument,
|
|
232
|
-
query_embedding=query_vector,
|
|
233
|
-
limit=10
|
|
234
|
-
)
|
|
216
|
+
from pgvector_template.models.search import MetadataFilter, SearchQuery
|
|
217
|
+
|
|
218
|
+
# Semantic search
|
|
219
|
+
results = service.search_client.search(SearchQuery(text="how do I reset my password", limit=10))
|
|
235
220
|
|
|
236
|
-
#
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
results = doc_manager.search_similar(
|
|
245
|
-
session=session,
|
|
246
|
-
document_cls=MyDocument,
|
|
247
|
-
query_embedding=query_vector,
|
|
248
|
-
limit=10,
|
|
249
|
-
metadata_filters=filters,
|
|
250
|
-
collection="textbooks"
|
|
221
|
+
# Keyword search combined with a metadata filter
|
|
222
|
+
results = service.search_client.search(
|
|
223
|
+
SearchQuery(
|
|
224
|
+
keywords=["password", "reset"],
|
|
225
|
+
metadata_filters=[MetadataFilter(field_name="source_type", condition="eq", value="pdf")],
|
|
226
|
+
limit=10,
|
|
227
|
+
)
|
|
251
228
|
)
|
|
229
|
+
|
|
230
|
+
for r in results:
|
|
231
|
+
print(r.score, r.document.content)
|
|
252
232
|
```
|
|
253
233
|
|
|
254
|
-
### 4.
|
|
234
|
+
### 4. Read a Corpus Back
|
|
255
235
|
|
|
256
236
|
```python
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
docs = doc_manager.get_documents_by_collection(
|
|
260
|
-
session, MyDocument, "textbooks"
|
|
261
|
-
)
|
|
262
|
-
|
|
263
|
-
# Get all chunks from a corpus
|
|
264
|
-
corpus_docs = doc_manager.get_documents_by_corpus_id(
|
|
265
|
-
session, MyDocument, corpus_id
|
|
266
|
-
)
|
|
237
|
+
corpus = service.corpus_manager.get_full_corpus("intro-chapter")
|
|
238
|
+
print(corpus.content, corpus.metadata, len(corpus.documents))
|
|
267
239
|
```
|
|
268
240
|
|
|
269
241
|
## Concept Reference
|
|
@@ -273,8 +245,11 @@ with db_manager.get_session() as session:
|
|
|
273
245
|
- **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
|
|
274
246
|
- **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
|
|
275
247
|
- **`BaseDocumentMetadata`**: Base schema for structured document metadata
|
|
276
|
-
- **`
|
|
277
|
-
- **`
|
|
248
|
+
- **`BaseEmbeddingProvider`**: Interface to implement for your embedding model
|
|
249
|
+
- **`BaseCorpusManager`**: Splits, embeds and upserts corpora (`insert_corpus`, `get_full_corpus`)
|
|
250
|
+
- **`BaseSearchClient`**: Semantic, keyword and metadata search (`search`)
|
|
251
|
+
- **`DocumentService`**: Bundles a corpus manager and a search client over one session
|
|
252
|
+
- **`DatabaseManager`** / **`DocumentDatabaseManager`**: Connection, schema and table setup
|
|
278
253
|
|
|
279
254
|
## Testing
|
|
280
255
|
|
|
@@ -308,7 +283,8 @@ python -m unittest discover -s integ-tests
|
|
|
308
283
|
|
|
309
284
|
```bash
|
|
310
285
|
pip install -e .[dev,test]
|
|
311
|
-
|
|
286
|
+
ruff check . && ruff format .
|
|
287
|
+
ty check .
|
|
312
288
|
```
|
|
313
289
|
|
|
314
290
|
## License
|
|
@@ -9,38 +9,36 @@ PGVector-Template provides a robust foundation for implementing vector-based doc
|
|
|
9
9
|
## Quick Start
|
|
10
10
|
|
|
11
11
|
```python
|
|
12
|
-
from pgvector_template import
|
|
12
|
+
from pgvector_template import DocumentDatabaseManager
|
|
13
13
|
from pgvector_template.core.document import BaseDocument
|
|
14
|
-
from
|
|
15
|
-
from
|
|
14
|
+
from pgvector_template.models.search import SearchQuery
|
|
15
|
+
from pgvector_template.service.document_service import DocumentService, DocumentServiceConfig
|
|
16
|
+
|
|
16
17
|
|
|
17
18
|
# 1. Define your document model
|
|
18
19
|
class MyDocument(BaseDocument):
|
|
19
20
|
__tablename__ = "my_documents"
|
|
20
21
|
|
|
21
|
-
# 2. Set up database connection
|
|
22
|
-
engine = create_engine("postgresql://user:pass@localhost/mydb")
|
|
23
|
-
db_manager = DatabaseManager(engine)
|
|
24
|
-
db_manager.create_tables([MyDocument])
|
|
25
|
-
|
|
26
|
-
# 3. Create document manager
|
|
27
|
-
doc_manager = DocumentDatabaseManager(db_manager)
|
|
28
|
-
|
|
29
|
-
# 4. Insert a document
|
|
30
|
-
with db_manager.get_session() as session:
|
|
31
|
-
doc = MyDocument.from_props(
|
|
32
|
-
corpus_id=uuid4(),
|
|
33
|
-
chunk_index=0,
|
|
34
|
-
content="Your document content here",
|
|
35
|
-
embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
|
|
36
|
-
)
|
|
37
|
-
doc_manager.insert_document(session, doc)
|
|
38
22
|
|
|
39
|
-
#
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
23
|
+
# 2. Set up the database: connects, enables pgvector, creates the schema and tables
|
|
24
|
+
db = DocumentDatabaseManager(
|
|
25
|
+
database_url="postgresql://user:pass@localhost/mydb",
|
|
26
|
+
schema_suffix="my_kb",
|
|
27
|
+
document_classes=[MyDocument],
|
|
28
|
+
)
|
|
29
|
+
db.setup()
|
|
30
|
+
|
|
31
|
+
# 3. Create a service. `my_embedding_provider` is your `BaseEmbeddingProvider` subclass
|
|
32
|
+
config = DocumentServiceConfig(document_cls=MyDocument, embedding_provider=my_embedding_provider)
|
|
33
|
+
|
|
34
|
+
with db.get_session() as session:
|
|
35
|
+
service = DocumentService(session, config)
|
|
36
|
+
|
|
37
|
+
# 4. Insert a corpus: it is split into chunks, embedded and stored
|
|
38
|
+
service.corpus_manager.insert_corpus("Your document content here", {}, corpus_id="doc-1")
|
|
39
|
+
|
|
40
|
+
# 5. Search
|
|
41
|
+
results = service.search_client.search(SearchQuery(text="your query", limit=5))
|
|
44
42
|
```
|
|
45
43
|
|
|
46
44
|
## Key Concepts
|
|
@@ -59,6 +57,7 @@ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all
|
|
|
59
57
|
- **Metadata Management**: JSON-based flexible metadata with GIN indexing
|
|
60
58
|
- **Collection Support**: Organize documents into logical collections
|
|
61
59
|
- **Chunk Management**: Handle long content by chunking into retrievable documents
|
|
60
|
+
- **Smart Upserts**: Re-inserting a corpus only re-embeds chunks whose content changed, and skips the write entirely if nothing did
|
|
62
61
|
- **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
|
|
63
62
|
- **Type Safety**: Full Pydantic validation and type hints
|
|
64
63
|
- **Production Ready**: Comprehensive testing and error handling
|
|
@@ -112,33 +111,14 @@ CREATE EXTENSION IF NOT EXISTS vector;
|
|
|
112
111
|
3. **Set up your connection:**
|
|
113
112
|
|
|
114
113
|
```python
|
|
115
|
-
from sqlalchemy import create_engine
|
|
116
114
|
from pgvector_template import DatabaseManager
|
|
117
115
|
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
db_manager = DatabaseManager(engine)
|
|
121
|
-
|
|
122
|
-
# Option 2: From environment variable
|
|
123
|
-
import os
|
|
124
|
-
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
125
|
-
db_manager = DatabaseManager(engine)
|
|
116
|
+
db_manager = DatabaseManager("postgresql://user:password@localhost:5432/mydb")
|
|
117
|
+
db_manager.initialize() # connects and enables the vector extension
|
|
126
118
|
```
|
|
127
119
|
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
```python
|
|
131
|
-
from sqlalchemy import create_engine
|
|
132
|
-
from sqlalchemy.pool import QueuePool
|
|
133
|
-
|
|
134
|
-
engine = create_engine(
|
|
135
|
-
"postgresql://user:password@localhost:5432/mydb",
|
|
136
|
-
poolclass=QueuePool,
|
|
137
|
-
pool_size=10,
|
|
138
|
-
max_overflow=20,
|
|
139
|
-
pool_pre_ping=True,
|
|
140
|
-
)
|
|
141
|
-
```
|
|
120
|
+
`create_tables` also adds nullable model columns that are missing from existing tables (for example `content_hash`).
|
|
121
|
+
Only the column name and type are applied: indexes, foreign keys, unique constraints and server defaults are not. This is a stopgap until proper migrations are added.
|
|
142
122
|
|
|
143
123
|
### Environment Variables
|
|
144
124
|
|
|
@@ -155,85 +135,77 @@ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
|
|
|
155
135
|
from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
|
|
156
136
|
from pydantic import Field
|
|
157
137
|
|
|
138
|
+
|
|
158
139
|
class MyDocumentMetadata(BaseDocumentMetadata):
|
|
159
140
|
source_type: str = Field(..., description="Type of source document")
|
|
160
141
|
author: str = Field(default="unknown", description="Document author")
|
|
161
142
|
|
|
143
|
+
|
|
162
144
|
class MyDocument(BaseDocument):
|
|
163
145
|
__tablename__ = "my_documents"
|
|
164
146
|
```
|
|
165
147
|
|
|
166
|
-
### 2. Insert
|
|
148
|
+
### 2. Insert a Corpus
|
|
167
149
|
|
|
168
150
|
```python
|
|
169
151
|
from pgvector_template.core.document import BaseDocumentOptionalProps
|
|
170
|
-
from uuid import uuid4
|
|
171
152
|
|
|
172
|
-
# Create document with metadata
|
|
173
|
-
metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
|
|
174
153
|
optional_props = BaseDocumentOptionalProps(
|
|
175
154
|
title="Chapter 1: Introduction",
|
|
176
155
|
collection="textbooks",
|
|
177
156
|
language="en",
|
|
178
|
-
tags=["education", "intro"]
|
|
179
157
|
)
|
|
180
158
|
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
)
|
|
159
|
+
with db.get_session() as session:
|
|
160
|
+
service = DocumentService(session, config)
|
|
161
|
+
result = service.corpus_manager.insert_corpus(
|
|
162
|
+
"This is the document content...",
|
|
163
|
+
{"source_type": "pdf", "author": "John Doe"}, # corpus metadata
|
|
164
|
+
optional_props,
|
|
165
|
+
corpus_id="intro-chapter",
|
|
166
|
+
)
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
To insert chunks you have already split and embedded, use `insert_documents()`.
|
|
189
170
|
|
|
190
|
-
|
|
191
|
-
|
|
171
|
+
Calling `insert_corpus()` again with the same `corpus_id` replaces the corpus. Unchanged chunks reuse their stored vector, and the write is skipped if nothing changed:
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
result = corpus_manager.insert_corpus(text, {"source": "pdf"}, corpus_id="handbook")
|
|
175
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=12, reused=0, total=12)
|
|
176
|
+
|
|
177
|
+
result = corpus_manager.insert_corpus(edited_text, {"source": "pdf"}, corpus_id="handbook")
|
|
178
|
+
# UpsertResult(corpus_hash=..., skipped=False, embedded=1, reused=11, total=12)
|
|
192
179
|
```
|
|
193
180
|
|
|
181
|
+
Rows written before `content_hash` existed have no hash, so their first upsert re-embeds them.
|
|
182
|
+
|
|
194
183
|
### 3. Search Documents
|
|
195
184
|
|
|
196
185
|
```python
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
document_cls=MyDocument,
|
|
202
|
-
query_embedding=query_vector,
|
|
203
|
-
limit=10
|
|
204
|
-
)
|
|
186
|
+
from pgvector_template.models.search import MetadataFilter, SearchQuery
|
|
187
|
+
|
|
188
|
+
# Semantic search
|
|
189
|
+
results = service.search_client.search(SearchQuery(text="how do I reset my password", limit=10))
|
|
205
190
|
|
|
206
|
-
#
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
results = doc_manager.search_similar(
|
|
215
|
-
session=session,
|
|
216
|
-
document_cls=MyDocument,
|
|
217
|
-
query_embedding=query_vector,
|
|
218
|
-
limit=10,
|
|
219
|
-
metadata_filters=filters,
|
|
220
|
-
collection="textbooks"
|
|
191
|
+
# Keyword search combined with a metadata filter
|
|
192
|
+
results = service.search_client.search(
|
|
193
|
+
SearchQuery(
|
|
194
|
+
keywords=["password", "reset"],
|
|
195
|
+
metadata_filters=[MetadataFilter(field_name="source_type", condition="eq", value="pdf")],
|
|
196
|
+
limit=10,
|
|
197
|
+
)
|
|
221
198
|
)
|
|
199
|
+
|
|
200
|
+
for r in results:
|
|
201
|
+
print(r.score, r.document.content)
|
|
222
202
|
```
|
|
223
203
|
|
|
224
|
-
### 4.
|
|
204
|
+
### 4. Read a Corpus Back
|
|
225
205
|
|
|
226
206
|
```python
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
docs = doc_manager.get_documents_by_collection(
|
|
230
|
-
session, MyDocument, "textbooks"
|
|
231
|
-
)
|
|
232
|
-
|
|
233
|
-
# Get all chunks from a corpus
|
|
234
|
-
corpus_docs = doc_manager.get_documents_by_corpus_id(
|
|
235
|
-
session, MyDocument, corpus_id
|
|
236
|
-
)
|
|
207
|
+
corpus = service.corpus_manager.get_full_corpus("intro-chapter")
|
|
208
|
+
print(corpus.content, corpus.metadata, len(corpus.documents))
|
|
237
209
|
```
|
|
238
210
|
|
|
239
211
|
## Concept Reference
|
|
@@ -243,8 +215,11 @@ with db_manager.get_session() as session:
|
|
|
243
215
|
- **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
|
|
244
216
|
- **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
|
|
245
217
|
- **`BaseDocumentMetadata`**: Base schema for structured document metadata
|
|
246
|
-
- **`
|
|
247
|
-
- **`
|
|
218
|
+
- **`BaseEmbeddingProvider`**: Interface to implement for your embedding model
|
|
219
|
+
- **`BaseCorpusManager`**: Splits, embeds and upserts corpora (`insert_corpus`, `get_full_corpus`)
|
|
220
|
+
- **`BaseSearchClient`**: Semantic, keyword and metadata search (`search`)
|
|
221
|
+
- **`DocumentService`**: Bundles a corpus manager and a search client over one session
|
|
222
|
+
- **`DatabaseManager`** / **`DocumentDatabaseManager`**: Connection, schema and table setup
|
|
248
223
|
|
|
249
224
|
## Testing
|
|
250
225
|
|
|
@@ -278,7 +253,8 @@ python -m unittest discover -s integ-tests
|
|
|
278
253
|
|
|
279
254
|
```bash
|
|
280
255
|
pip install -e .[dev,test]
|
|
281
|
-
|
|
256
|
+
ruff check . && ruff format .
|
|
257
|
+
ty check .
|
|
282
258
|
```
|
|
283
259
|
|
|
284
260
|
## License
|
|
@@ -5,21 +5,16 @@ from pgvector_template.core.document import (
|
|
|
5
5
|
)
|
|
6
6
|
from pgvector_template.core.embedder import BaseEmbeddingProvider
|
|
7
7
|
from pgvector_template.core.manager import BaseCorpusManager, BaseCorpusManagerConfig, Corpus
|
|
8
|
-
from pgvector_template.core.search import
|
|
9
|
-
|
|
8
|
+
from pgvector_template.core.search import BaseSearchClient, BaseSearchClientConfig
|
|
10
9
|
|
|
11
10
|
__all__ = [
|
|
12
|
-
|
|
13
|
-
"
|
|
11
|
+
"BaseCorpusManager",
|
|
12
|
+
"BaseCorpusManagerConfig",
|
|
14
13
|
"BaseDocument",
|
|
15
14
|
"BaseDocumentMetadata",
|
|
16
|
-
"
|
|
17
|
-
### embedder
|
|
15
|
+
"BaseDocumentOptionalProps",
|
|
18
16
|
"BaseEmbeddingProvider",
|
|
19
|
-
### manager
|
|
20
|
-
"BaseCorpusManager",
|
|
21
|
-
"BaseCorpusManagerConfig",
|
|
22
|
-
### search
|
|
23
|
-
"BaseSearchClientConfig",
|
|
24
17
|
"BaseSearchClient",
|
|
18
|
+
"BaseSearchClientConfig",
|
|
19
|
+
"Corpus",
|
|
25
20
|
]
|
|
@@ -1,21 +1,24 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
from
|
|
1
|
+
import hashlib
|
|
2
|
+
import json
|
|
3
|
+
from datetime import UTC, datetime
|
|
4
|
+
from typing import Any, Self
|
|
5
|
+
from uuid import UUID as UuidLiteral
|
|
6
|
+
from uuid import uuid4
|
|
4
7
|
|
|
8
|
+
from pgvector.sqlalchemy import Vector
|
|
5
9
|
from pydantic import BaseModel, Field
|
|
6
10
|
from sqlalchemy import (
|
|
11
|
+
Boolean,
|
|
7
12
|
Column,
|
|
8
|
-
String,
|
|
9
|
-
Text,
|
|
10
13
|
DateTime,
|
|
11
|
-
Boolean,
|
|
12
|
-
Integer,
|
|
13
14
|
Index,
|
|
15
|
+
Integer,
|
|
16
|
+
String,
|
|
17
|
+
Text,
|
|
14
18
|
UniqueConstraint,
|
|
15
19
|
)
|
|
20
|
+
from sqlalchemy.dialects.postgresql import JSONB, UUID
|
|
16
21
|
from sqlalchemy.orm import declarative_base
|
|
17
|
-
from sqlalchemy.dialects.postgresql import UUID, JSONB
|
|
18
|
-
from pgvector.sqlalchemy import Vector
|
|
19
22
|
|
|
20
23
|
Base = declarative_base()
|
|
21
24
|
|
|
@@ -33,9 +36,6 @@ class BaseDocumentOptionalProps(BaseModel):
|
|
|
33
36
|
"""Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'"""
|
|
34
37
|
|
|
35
38
|
|
|
36
|
-
T = TypeVar("T", bound="BaseDocument")
|
|
37
|
-
|
|
38
|
-
|
|
39
39
|
class BaseDocument(Base):
|
|
40
40
|
"""
|
|
41
41
|
Template table for Documents, that works for all collection types.
|
|
@@ -88,12 +88,18 @@ class BaseDocument(Base):
|
|
|
88
88
|
Use this field for auditing and diagnosing mixed-model corpora, not for runtime dispatch.
|
|
89
89
|
"""
|
|
90
90
|
|
|
91
|
+
content_hash = Column(String(64), nullable=True)
|
|
92
|
+
"""
|
|
93
|
+
Hex sha256 of `content` + `embedding_config`, see `compute_content_hash()`.
|
|
94
|
+
Lets upserts reuse a stored embedding when a chunk is unchanged. NULL for rows that pre-date this column.
|
|
95
|
+
"""
|
|
96
|
+
|
|
91
97
|
# Audit fields
|
|
92
|
-
created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(
|
|
98
|
+
created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(UTC))
|
|
93
99
|
updated_at = Column(
|
|
94
100
|
DateTime(timezone=True),
|
|
95
|
-
default=lambda: datetime.now(
|
|
96
|
-
onupdate=lambda: datetime.now(
|
|
101
|
+
default=lambda: datetime.now(UTC),
|
|
102
|
+
onupdate=lambda: datetime.now(UTC),
|
|
97
103
|
)
|
|
98
104
|
is_deleted = Column(Boolean, default=False, index=True)
|
|
99
105
|
"""Entries can be logically marked for deletion before they are permanently deleted."""
|
|
@@ -131,7 +137,7 @@ class BaseDocument(Base):
|
|
|
131
137
|
|
|
132
138
|
@classmethod
|
|
133
139
|
def from_props(
|
|
134
|
-
cls
|
|
140
|
+
cls,
|
|
135
141
|
corpus_id: UuidLiteral | str,
|
|
136
142
|
chunk_index: int,
|
|
137
143
|
content: str,
|
|
@@ -139,7 +145,7 @@ class BaseDocument(Base):
|
|
|
139
145
|
embedding_config: dict[str, Any] | None = None,
|
|
140
146
|
metadata: dict[str, Any] | None = None,
|
|
141
147
|
optional_props: BaseDocumentOptionalProps | None = None,
|
|
142
|
-
) ->
|
|
148
|
+
) -> Self:
|
|
143
149
|
"""
|
|
144
150
|
Create a BaseDocument instance from mandatory and optional properties.
|
|
145
151
|
|
|
@@ -167,6 +173,7 @@ class BaseDocument(Base):
|
|
|
167
173
|
content=content,
|
|
168
174
|
embedding=embedding,
|
|
169
175
|
embedding_config=embedding_config,
|
|
176
|
+
content_hash=cls.compute_content_hash(content, embedding_config),
|
|
170
177
|
title=optional_props.title,
|
|
171
178
|
document_metadata=metadata or {},
|
|
172
179
|
collection=optional_props.collection,
|
|
@@ -174,6 +181,15 @@ class BaseDocument(Base):
|
|
|
174
181
|
language=optional_props.language,
|
|
175
182
|
)
|
|
176
183
|
|
|
184
|
+
@classmethod
|
|
185
|
+
def compute_content_hash(cls, content: str, embedding_config: dict[str, Any] | None) -> str:
|
|
186
|
+
"""
|
|
187
|
+
Hash of everything that determines a chunk's embedding vector. Metadata and other props are excluded.
|
|
188
|
+
Override to change what counts as "the same chunk".
|
|
189
|
+
"""
|
|
190
|
+
payload = json.dumps([content, embedding_config], sort_keys=True, ensure_ascii=False)
|
|
191
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
192
|
+
|
|
177
193
|
@classmethod
|
|
178
194
|
def get_embedding_index(cls, table_name: str) -> Index:
|
|
179
195
|
"""Override this method to customize the embedding index."""
|