pgvector-template 0.5__tar.gz → 0.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {pgvector_template-0.5 → pgvector_template-0.6}/PKG-INFO +77 -101
  2. {pgvector_template-0.5 → pgvector_template-0.6}/README.md +76 -100
  3. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/__init__.py +6 -11
  4. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/document.py +33 -17
  5. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/manager.py +156 -22
  6. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/search.py +9 -9
  7. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/db/connection.py +36 -4
  8. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/db/document_db.py +1 -3
  9. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/models/__init__.py +4 -0
  10. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/models/search.py +1 -1
  11. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/service/document_service.py +5 -5
  12. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/utils/metadata_filter.py +4 -6
  13. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/PKG-INFO +77 -101
  14. {pgvector_template-0.5 → pgvector_template-0.6}/pyproject.toml +1 -1
  15. {pgvector_template-0.5 → pgvector_template-0.6}/LICENSE +0 -0
  16. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/__init__.py +0 -0
  17. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/core/embedder.py +0 -0
  18. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/db/__init__.py +0 -0
  19. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/service/__init__.py +0 -0
  20. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/types.py +0 -0
  21. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template/utils/__init__.py +0 -0
  22. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/SOURCES.txt +0 -0
  23. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/dependency_links.txt +0 -0
  24. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/requires.txt +0 -0
  25. {pgvector_template-0.5 → pgvector_template-0.6}/pgvector_template.egg-info/top_level.txt +0 -0
  26. {pgvector_template-0.5 → pgvector_template-0.6}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pgvector-template
3
- Version: 0.5
3
+ Version: 0.6
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -39,38 +39,36 @@ PGVector-Template provides a robust foundation for implementing vector-based doc
39
39
  ## Quick Start
40
40
 
41
41
  ```python
42
- from pgvector_template import DatabaseManager, DocumentDatabaseManager
42
+ from pgvector_template import DocumentDatabaseManager
43
43
  from pgvector_template.core.document import BaseDocument
44
- from sqlalchemy import create_engine
45
- from uuid import uuid4
44
+ from pgvector_template.models.search import SearchQuery
45
+ from pgvector_template.service.document_service import DocumentService, DocumentServiceConfig
46
+
46
47
 
47
48
  # 1. Define your document model
48
49
  class MyDocument(BaseDocument):
49
50
  __tablename__ = "my_documents"
50
51
 
51
- # 2. Set up database connection
52
- engine = create_engine("postgresql://user:pass@localhost/mydb")
53
- db_manager = DatabaseManager(engine)
54
- db_manager.create_tables([MyDocument])
55
-
56
- # 3. Create document manager
57
- doc_manager = DocumentDatabaseManager(db_manager)
58
-
59
- # 4. Insert a document
60
- with db_manager.get_session() as session:
61
- doc = MyDocument.from_props(
62
- corpus_id=uuid4(),
63
- chunk_index=0,
64
- content="Your document content here",
65
- embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
66
- )
67
- doc_manager.insert_document(session, doc)
68
52
 
69
- # 5. Search similar documents
70
- with db_manager.get_session() as session:
71
- results = doc_manager.search_similar(
72
- session, MyDocument, query_embedding=[0.1, 0.2, 0.3, ...], limit=5
73
- )
53
+ # 2. Set up the database: connects, enables pgvector, creates the schema and tables
54
+ db = DocumentDatabaseManager(
55
+ database_url="postgresql://user:pass@localhost/mydb",
56
+ schema_suffix="my_kb",
57
+ document_classes=[MyDocument],
58
+ )
59
+ db.setup()
60
+
61
+ # 3. Create a service. `my_embedding_provider` is your `BaseEmbeddingProvider` subclass
62
+ config = DocumentServiceConfig(document_cls=MyDocument, embedding_provider=my_embedding_provider)
63
+
64
+ with db.get_session() as session:
65
+ service = DocumentService(session, config)
66
+
67
+ # 4. Insert a corpus: it is split into chunks, embedded and stored
68
+ service.corpus_manager.insert_corpus("Your document content here", {}, corpus_id="doc-1")
69
+
70
+ # 5. Search
71
+ results = service.search_client.search(SearchQuery(text="your query", limit=5))
74
72
  ```
75
73
 
76
74
  ## Key Concepts
@@ -89,6 +87,7 @@ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all
89
87
  - **Metadata Management**: JSON-based flexible metadata with GIN indexing
90
88
  - **Collection Support**: Organize documents into logical collections
91
89
  - **Chunk Management**: Handle long content by chunking into retrievable documents
90
+ - **Smart Upserts**: Re-inserting a corpus only re-embeds chunks whose content changed, and skips the write entirely if nothing did
92
91
  - **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
93
92
  - **Type Safety**: Full Pydantic validation and type hints
94
93
  - **Production Ready**: Comprehensive testing and error handling
@@ -142,33 +141,14 @@ CREATE EXTENSION IF NOT EXISTS vector;
142
141
  3. **Set up your connection:**
143
142
 
144
143
  ```python
145
- from sqlalchemy import create_engine
146
144
  from pgvector_template import DatabaseManager
147
145
 
148
- # Option 1: Direct connection string
149
- engine = create_engine("postgresql://user:password@localhost:5432/mydb")
150
- db_manager = DatabaseManager(engine)
151
-
152
- # Option 2: From environment variable
153
- import os
154
- engine = create_engine(os.getenv("DATABASE_URL"))
155
- db_manager = DatabaseManager(engine)
146
+ db_manager = DatabaseManager("postgresql://user:password@localhost:5432/mydb")
147
+ db_manager.initialize() # connects and enables the vector extension
156
148
  ```
157
149
 
158
- ### Production Configuration
159
-
160
- ```python
161
- from sqlalchemy import create_engine
162
- from sqlalchemy.pool import QueuePool
163
-
164
- engine = create_engine(
165
- "postgresql://user:password@localhost:5432/mydb",
166
- poolclass=QueuePool,
167
- pool_size=10,
168
- max_overflow=20,
169
- pool_pre_ping=True,
170
- )
171
- ```
150
+ `create_tables` also adds nullable model columns that are missing from existing tables (for example `content_hash`).
151
+ Only the column name and type are applied: indexes, foreign keys, unique constraints and server defaults are not. This is a stopgap until proper migrations are added.
172
152
 
173
153
  ### Environment Variables
174
154
 
@@ -185,85 +165,77 @@ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
185
165
  from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
186
166
  from pydantic import Field
187
167
 
168
+
188
169
  class MyDocumentMetadata(BaseDocumentMetadata):
189
170
  source_type: str = Field(..., description="Type of source document")
190
171
  author: str = Field(default="unknown", description="Document author")
191
172
 
173
+
192
174
  class MyDocument(BaseDocument):
193
175
  __tablename__ = "my_documents"
194
176
  ```
195
177
 
196
- ### 2. Insert Documents
178
+ ### 2. Insert a Corpus
197
179
 
198
180
  ```python
199
181
  from pgvector_template.core.document import BaseDocumentOptionalProps
200
- from uuid import uuid4
201
182
 
202
- # Create document with metadata
203
- metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
204
183
  optional_props = BaseDocumentOptionalProps(
205
184
  title="Chapter 1: Introduction",
206
185
  collection="textbooks",
207
186
  language="en",
208
- tags=["education", "intro"]
209
187
  )
210
188
 
211
- doc = MyDocument.from_props(
212
- corpus_id=uuid4(),
213
- chunk_index=0,
214
- content="This is the document content...",
215
- embedding=your_embedding_vector, # list[float] with 1024 dimensions
216
- metadata=metadata.to_dict(),
217
- optional_props=optional_props
218
- )
189
+ with db.get_session() as session:
190
+ service = DocumentService(session, config)
191
+ result = service.corpus_manager.insert_corpus(
192
+ "This is the document content...",
193
+ {"source_type": "pdf", "author": "John Doe"}, # corpus metadata
194
+ optional_props,
195
+ corpus_id="intro-chapter",
196
+ )
197
+ ```
198
+
199
+ To insert chunks you have already split and embedded, use `insert_documents()`.
219
200
 
220
- with db_manager.get_session() as session:
221
- doc_manager.insert_document(session, doc)
201
+ Calling `insert_corpus()` again with the same `corpus_id` replaces the corpus. Unchanged chunks reuse their stored vector, and the write is skipped if nothing changed:
202
+
203
+ ```python
204
+ result = corpus_manager.insert_corpus(text, {"source": "pdf"}, corpus_id="handbook")
205
+ # UpsertResult(corpus_hash=..., skipped=False, embedded=12, reused=0, total=12)
206
+
207
+ result = corpus_manager.insert_corpus(edited_text, {"source": "pdf"}, corpus_id="handbook")
208
+ # UpsertResult(corpus_hash=..., skipped=False, embedded=1, reused=11, total=12)
222
209
  ```
223
210
 
211
+ Rows written before `content_hash` existed have no hash, so their first upsert re-embeds them.
212
+
224
213
  ### 3. Search Documents
225
214
 
226
215
  ```python
227
- # Basic similarity search
228
- with db_manager.get_session() as session:
229
- results = doc_manager.search_similar(
230
- session=session,
231
- document_cls=MyDocument,
232
- query_embedding=query_vector,
233
- limit=10
234
- )
216
+ from pgvector_template.models.search import MetadataFilter, SearchQuery
217
+
218
+ # Semantic search
219
+ results = service.search_client.search(SearchQuery(text="how do I reset my password", limit=10))
235
220
 
236
- # Search with filters
237
- from pgvector_template.models.search import MetadataFilter
238
-
239
- filters = [
240
- MetadataFilter(key="source_type", value="pdf"),
241
- MetadataFilter(key="author", value="John Doe")
242
- ]
243
-
244
- results = doc_manager.search_similar(
245
- session=session,
246
- document_cls=MyDocument,
247
- query_embedding=query_vector,
248
- limit=10,
249
- metadata_filters=filters,
250
- collection="textbooks"
221
+ # Keyword search combined with a metadata filter
222
+ results = service.search_client.search(
223
+ SearchQuery(
224
+ keywords=["password", "reset"],
225
+ metadata_filters=[MetadataFilter(field_name="source_type", condition="eq", value="pdf")],
226
+ limit=10,
227
+ )
251
228
  )
229
+
230
+ for r in results:
231
+ print(r.score, r.document.content)
252
232
  ```
253
233
 
254
- ### 4. Manage Collections
234
+ ### 4. Read a Corpus Back
255
235
 
256
236
  ```python
257
- # Get all documents in a collection
258
- with db_manager.get_session() as session:
259
- docs = doc_manager.get_documents_by_collection(
260
- session, MyDocument, "textbooks"
261
- )
262
-
263
- # Get all chunks from a corpus
264
- corpus_docs = doc_manager.get_documents_by_corpus_id(
265
- session, MyDocument, corpus_id
266
- )
237
+ corpus = service.corpus_manager.get_full_corpus("intro-chapter")
238
+ print(corpus.content, corpus.metadata, len(corpus.documents))
267
239
  ```
268
240
 
269
241
  ## Concept Reference
@@ -273,8 +245,11 @@ with db_manager.get_session() as session:
273
245
  - **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
274
246
  - **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
275
247
  - **`BaseDocumentMetadata`**: Base schema for structured document metadata
276
- - **`DatabaseManager`**: Database connection and session management
277
- - **`DocumentDatabaseManager`**: High-level document CRUD operations
248
+ - **`BaseEmbeddingProvider`**: Interface to implement for your embedding model
249
+ - **`BaseCorpusManager`**: Splits, embeds and upserts corpora (`insert_corpus`, `get_full_corpus`)
250
+ - **`BaseSearchClient`**: Semantic, keyword and metadata search (`search`)
251
+ - **`DocumentService`**: Bundles a corpus manager and a search client over one session
252
+ - **`DatabaseManager`** / **`DocumentDatabaseManager`**: Connection, schema and table setup
278
253
 
279
254
  ## Testing
280
255
 
@@ -308,7 +283,8 @@ python -m unittest discover -s integ-tests
308
283
 
309
284
  ```bash
310
285
  pip install -e .[dev,test]
311
- black . # Format code
286
+ ruff check . && ruff format .
287
+ ty check .
312
288
  ```
313
289
 
314
290
  ## License
@@ -9,38 +9,36 @@ PGVector-Template provides a robust foundation for implementing vector-based doc
9
9
  ## Quick Start
10
10
 
11
11
  ```python
12
- from pgvector_template import DatabaseManager, DocumentDatabaseManager
12
+ from pgvector_template import DocumentDatabaseManager
13
13
  from pgvector_template.core.document import BaseDocument
14
- from sqlalchemy import create_engine
15
- from uuid import uuid4
14
+ from pgvector_template.models.search import SearchQuery
15
+ from pgvector_template.service.document_service import DocumentService, DocumentServiceConfig
16
+
16
17
 
17
18
  # 1. Define your document model
18
19
  class MyDocument(BaseDocument):
19
20
  __tablename__ = "my_documents"
20
21
 
21
- # 2. Set up database connection
22
- engine = create_engine("postgresql://user:pass@localhost/mydb")
23
- db_manager = DatabaseManager(engine)
24
- db_manager.create_tables([MyDocument])
25
-
26
- # 3. Create document manager
27
- doc_manager = DocumentDatabaseManager(db_manager)
28
-
29
- # 4. Insert a document
30
- with db_manager.get_session() as session:
31
- doc = MyDocument.from_props(
32
- corpus_id=uuid4(),
33
- chunk_index=0,
34
- content="Your document content here",
35
- embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
36
- )
37
- doc_manager.insert_document(session, doc)
38
22
 
39
- # 5. Search similar documents
40
- with db_manager.get_session() as session:
41
- results = doc_manager.search_similar(
42
- session, MyDocument, query_embedding=[0.1, 0.2, 0.3, ...], limit=5
43
- )
23
+ # 2. Set up the database: connects, enables pgvector, creates the schema and tables
24
+ db = DocumentDatabaseManager(
25
+ database_url="postgresql://user:pass@localhost/mydb",
26
+ schema_suffix="my_kb",
27
+ document_classes=[MyDocument],
28
+ )
29
+ db.setup()
30
+
31
+ # 3. Create a service. `my_embedding_provider` is your `BaseEmbeddingProvider` subclass
32
+ config = DocumentServiceConfig(document_cls=MyDocument, embedding_provider=my_embedding_provider)
33
+
34
+ with db.get_session() as session:
35
+ service = DocumentService(session, config)
36
+
37
+ # 4. Insert a corpus: it is split into chunks, embedded and stored
38
+ service.corpus_manager.insert_corpus("Your document content here", {}, corpus_id="doc-1")
39
+
40
+ # 5. Search
41
+ results = service.search_client.search(SearchQuery(text="your query", limit=5))
44
42
  ```
45
43
 
46
44
  ## Key Concepts
@@ -59,6 +57,7 @@ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all
59
57
  - **Metadata Management**: JSON-based flexible metadata with GIN indexing
60
58
  - **Collection Support**: Organize documents into logical collections
61
59
  - **Chunk Management**: Handle long content by chunking into retrievable documents
60
+ - **Smart Upserts**: Re-inserting a corpus only re-embeds chunks whose content changed, and skips the write entirely if nothing did
62
61
  - **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
63
62
  - **Type Safety**: Full Pydantic validation and type hints
64
63
  - **Production Ready**: Comprehensive testing and error handling
@@ -112,33 +111,14 @@ CREATE EXTENSION IF NOT EXISTS vector;
112
111
  3. **Set up your connection:**
113
112
 
114
113
  ```python
115
- from sqlalchemy import create_engine
116
114
  from pgvector_template import DatabaseManager
117
115
 
118
- # Option 1: Direct connection string
119
- engine = create_engine("postgresql://user:password@localhost:5432/mydb")
120
- db_manager = DatabaseManager(engine)
121
-
122
- # Option 2: From environment variable
123
- import os
124
- engine = create_engine(os.getenv("DATABASE_URL"))
125
- db_manager = DatabaseManager(engine)
116
+ db_manager = DatabaseManager("postgresql://user:password@localhost:5432/mydb")
117
+ db_manager.initialize() # connects and enables the vector extension
126
118
  ```
127
119
 
128
- ### Production Configuration
129
-
130
- ```python
131
- from sqlalchemy import create_engine
132
- from sqlalchemy.pool import QueuePool
133
-
134
- engine = create_engine(
135
- "postgresql://user:password@localhost:5432/mydb",
136
- poolclass=QueuePool,
137
- pool_size=10,
138
- max_overflow=20,
139
- pool_pre_ping=True,
140
- )
141
- ```
120
+ `create_tables` also adds nullable model columns that are missing from existing tables (for example `content_hash`).
121
+ Only the column name and type are applied: indexes, foreign keys, unique constraints and server defaults are not. This is a stopgap until proper migrations are added.
142
122
 
143
123
  ### Environment Variables
144
124
 
@@ -155,85 +135,77 @@ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
155
135
  from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
156
136
  from pydantic import Field
157
137
 
138
+
158
139
  class MyDocumentMetadata(BaseDocumentMetadata):
159
140
  source_type: str = Field(..., description="Type of source document")
160
141
  author: str = Field(default="unknown", description="Document author")
161
142
 
143
+
162
144
  class MyDocument(BaseDocument):
163
145
  __tablename__ = "my_documents"
164
146
  ```
165
147
 
166
- ### 2. Insert Documents
148
+ ### 2. Insert a Corpus
167
149
 
168
150
  ```python
169
151
  from pgvector_template.core.document import BaseDocumentOptionalProps
170
- from uuid import uuid4
171
152
 
172
- # Create document with metadata
173
- metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
174
153
  optional_props = BaseDocumentOptionalProps(
175
154
  title="Chapter 1: Introduction",
176
155
  collection="textbooks",
177
156
  language="en",
178
- tags=["education", "intro"]
179
157
  )
180
158
 
181
- doc = MyDocument.from_props(
182
- corpus_id=uuid4(),
183
- chunk_index=0,
184
- content="This is the document content...",
185
- embedding=your_embedding_vector, # list[float] with 1024 dimensions
186
- metadata=metadata.to_dict(),
187
- optional_props=optional_props
188
- )
159
+ with db.get_session() as session:
160
+ service = DocumentService(session, config)
161
+ result = service.corpus_manager.insert_corpus(
162
+ "This is the document content...",
163
+ {"source_type": "pdf", "author": "John Doe"}, # corpus metadata
164
+ optional_props,
165
+ corpus_id="intro-chapter",
166
+ )
167
+ ```
168
+
169
+ To insert chunks you have already split and embedded, use `insert_documents()`.
189
170
 
190
- with db_manager.get_session() as session:
191
- doc_manager.insert_document(session, doc)
171
+ Calling `insert_corpus()` again with the same `corpus_id` replaces the corpus. Unchanged chunks reuse their stored vector, and the write is skipped if nothing changed:
172
+
173
+ ```python
174
+ result = corpus_manager.insert_corpus(text, {"source": "pdf"}, corpus_id="handbook")
175
+ # UpsertResult(corpus_hash=..., skipped=False, embedded=12, reused=0, total=12)
176
+
177
+ result = corpus_manager.insert_corpus(edited_text, {"source": "pdf"}, corpus_id="handbook")
178
+ # UpsertResult(corpus_hash=..., skipped=False, embedded=1, reused=11, total=12)
192
179
  ```
193
180
 
181
+ Rows written before `content_hash` existed have no hash, so their first upsert re-embeds them.
182
+
194
183
  ### 3. Search Documents
195
184
 
196
185
  ```python
197
- # Basic similarity search
198
- with db_manager.get_session() as session:
199
- results = doc_manager.search_similar(
200
- session=session,
201
- document_cls=MyDocument,
202
- query_embedding=query_vector,
203
- limit=10
204
- )
186
+ from pgvector_template.models.search import MetadataFilter, SearchQuery
187
+
188
+ # Semantic search
189
+ results = service.search_client.search(SearchQuery(text="how do I reset my password", limit=10))
205
190
 
206
- # Search with filters
207
- from pgvector_template.models.search import MetadataFilter
208
-
209
- filters = [
210
- MetadataFilter(key="source_type", value="pdf"),
211
- MetadataFilter(key="author", value="John Doe")
212
- ]
213
-
214
- results = doc_manager.search_similar(
215
- session=session,
216
- document_cls=MyDocument,
217
- query_embedding=query_vector,
218
- limit=10,
219
- metadata_filters=filters,
220
- collection="textbooks"
191
+ # Keyword search combined with a metadata filter
192
+ results = service.search_client.search(
193
+ SearchQuery(
194
+ keywords=["password", "reset"],
195
+ metadata_filters=[MetadataFilter(field_name="source_type", condition="eq", value="pdf")],
196
+ limit=10,
197
+ )
221
198
  )
199
+
200
+ for r in results:
201
+ print(r.score, r.document.content)
222
202
  ```
223
203
 
224
- ### 4. Manage Collections
204
+ ### 4. Read a Corpus Back
225
205
 
226
206
  ```python
227
- # Get all documents in a collection
228
- with db_manager.get_session() as session:
229
- docs = doc_manager.get_documents_by_collection(
230
- session, MyDocument, "textbooks"
231
- )
232
-
233
- # Get all chunks from a corpus
234
- corpus_docs = doc_manager.get_documents_by_corpus_id(
235
- session, MyDocument, corpus_id
236
- )
207
+ corpus = service.corpus_manager.get_full_corpus("intro-chapter")
208
+ print(corpus.content, corpus.metadata, len(corpus.documents))
237
209
  ```
238
210
 
239
211
  ## Concept Reference
@@ -243,8 +215,11 @@ with db_manager.get_session() as session:
243
215
  - **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
244
216
  - **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
245
217
  - **`BaseDocumentMetadata`**: Base schema for structured document metadata
246
- - **`DatabaseManager`**: Database connection and session management
247
- - **`DocumentDatabaseManager`**: High-level document CRUD operations
218
+ - **`BaseEmbeddingProvider`**: Interface to implement for your embedding model
219
+ - **`BaseCorpusManager`**: Splits, embeds and upserts corpora (`insert_corpus`, `get_full_corpus`)
220
+ - **`BaseSearchClient`**: Semantic, keyword and metadata search (`search`)
221
+ - **`DocumentService`**: Bundles a corpus manager and a search client over one session
222
+ - **`DatabaseManager`** / **`DocumentDatabaseManager`**: Connection, schema and table setup
248
223
 
249
224
  ## Testing
250
225
 
@@ -278,7 +253,8 @@ python -m unittest discover -s integ-tests
278
253
 
279
254
  ```bash
280
255
  pip install -e .[dev,test]
281
- black . # Format code
256
+ ruff check . && ruff format .
257
+ ty check .
282
258
  ```
283
259
 
284
260
  ## License
@@ -5,21 +5,16 @@ from pgvector_template.core.document import (
5
5
  )
6
6
  from pgvector_template.core.embedder import BaseEmbeddingProvider
7
7
  from pgvector_template.core.manager import BaseCorpusManager, BaseCorpusManagerConfig, Corpus
8
- from pgvector_template.core.search import BaseSearchClientConfig, BaseSearchClient
9
-
8
+ from pgvector_template.core.search import BaseSearchClient, BaseSearchClientConfig
10
9
 
11
10
  __all__ = [
12
- ### document
13
- "BaseDocumentOptionalProps",
11
+ "BaseCorpusManager",
12
+ "BaseCorpusManagerConfig",
14
13
  "BaseDocument",
15
14
  "BaseDocumentMetadata",
16
- "Corpus",
17
- ### embedder
15
+ "BaseDocumentOptionalProps",
18
16
  "BaseEmbeddingProvider",
19
- ### manager
20
- "BaseCorpusManager",
21
- "BaseCorpusManagerConfig",
22
- ### search
23
- "BaseSearchClientConfig",
24
17
  "BaseSearchClient",
18
+ "BaseSearchClientConfig",
19
+ "Corpus",
25
20
  ]
@@ -1,21 +1,24 @@
1
- from datetime import datetime, timezone
2
- from typing import Any, Type, TypeVar
3
- from uuid import uuid4, UUID as UuidLiteral
1
+ import hashlib
2
+ import json
3
+ from datetime import UTC, datetime
4
+ from typing import Any, Self
5
+ from uuid import UUID as UuidLiteral
6
+ from uuid import uuid4
4
7
 
8
+ from pgvector.sqlalchemy import Vector
5
9
  from pydantic import BaseModel, Field
6
10
  from sqlalchemy import (
11
+ Boolean,
7
12
  Column,
8
- String,
9
- Text,
10
13
  DateTime,
11
- Boolean,
12
- Integer,
13
14
  Index,
15
+ Integer,
16
+ String,
17
+ Text,
14
18
  UniqueConstraint,
15
19
  )
20
+ from sqlalchemy.dialects.postgresql import JSONB, UUID
16
21
  from sqlalchemy.orm import declarative_base
17
- from sqlalchemy.dialects.postgresql import UUID, JSONB
18
- from pgvector.sqlalchemy import Vector
19
22
 
20
23
  Base = declarative_base()
21
24
 
@@ -33,9 +36,6 @@ class BaseDocumentOptionalProps(BaseModel):
33
36
  """Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'"""
34
37
 
35
38
 
36
- T = TypeVar("T", bound="BaseDocument")
37
-
38
-
39
39
  class BaseDocument(Base):
40
40
  """
41
41
  Template table for Documents, that works for all collection types.
@@ -88,12 +88,18 @@ class BaseDocument(Base):
88
88
  Use this field for auditing and diagnosing mixed-model corpora, not for runtime dispatch.
89
89
  """
90
90
 
91
+ content_hash = Column(String(64), nullable=True)
92
+ """
93
+ Hex sha256 of `content` + `embedding_config`, see `compute_content_hash()`.
94
+ Lets upserts reuse a stored embedding when a chunk is unchanged. NULL for rows that pre-date this column.
95
+ """
96
+
91
97
  # Audit fields
92
- created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(timezone.utc))
98
+ created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(UTC))
93
99
  updated_at = Column(
94
100
  DateTime(timezone=True),
95
- default=lambda: datetime.now(timezone.utc),
96
- onupdate=lambda: datetime.now(timezone.utc),
101
+ default=lambda: datetime.now(UTC),
102
+ onupdate=lambda: datetime.now(UTC),
97
103
  )
98
104
  is_deleted = Column(Boolean, default=False, index=True)
99
105
  """Entries can be logically marked for deletion before they are permanently deleted."""
@@ -131,7 +137,7 @@ class BaseDocument(Base):
131
137
 
132
138
  @classmethod
133
139
  def from_props(
134
- cls: Type[T],
140
+ cls,
135
141
  corpus_id: UuidLiteral | str,
136
142
  chunk_index: int,
137
143
  content: str,
@@ -139,7 +145,7 @@ class BaseDocument(Base):
139
145
  embedding_config: dict[str, Any] | None = None,
140
146
  metadata: dict[str, Any] | None = None,
141
147
  optional_props: BaseDocumentOptionalProps | None = None,
142
- ) -> T:
148
+ ) -> Self:
143
149
  """
144
150
  Create a BaseDocument instance from mandatory and optional properties.
145
151
 
@@ -167,6 +173,7 @@ class BaseDocument(Base):
167
173
  content=content,
168
174
  embedding=embedding,
169
175
  embedding_config=embedding_config,
176
+ content_hash=cls.compute_content_hash(content, embedding_config),
170
177
  title=optional_props.title,
171
178
  document_metadata=metadata or {},
172
179
  collection=optional_props.collection,
@@ -174,6 +181,15 @@ class BaseDocument(Base):
174
181
  language=optional_props.language,
175
182
  )
176
183
 
184
+ @classmethod
185
+ def compute_content_hash(cls, content: str, embedding_config: dict[str, Any] | None) -> str:
186
+ """
187
+ Hash of everything that determines a chunk's embedding vector. Metadata and other props are excluded.
188
+ Override to change what counts as "the same chunk".
189
+ """
190
+ payload = json.dumps([content, embedding_config], sort_keys=True, ensure_ascii=False)
191
+ return hashlib.sha256(payload.encode("utf-8")).hexdigest()
192
+
177
193
  @classmethod
178
194
  def get_embedding_index(cls, table_name: str) -> Index:
179
195
  """Override this method to customize the embedding index."""