pgvector-template 0.3.1__tar.gz → 0.3.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pgvector_template-0.3.3/PKG-INFO +320 -0
- pgvector_template-0.3.3/README.md +291 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/core/__init__.py +1 -3
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/core/document.py +5 -2
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/core/manager.py +52 -14
- pgvector_template-0.3.3/pgvector_template/models/__init__.py +1 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/models/search.py +3 -1
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/service/__init__.py +1 -1
- pgvector_template-0.3.3/pgvector_template.egg-info/PKG-INFO +320 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template.egg-info/requires.txt +4 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pyproject.toml +5 -1
- pgvector_template-0.3.1/PKG-INFO +0 -152
- pgvector_template-0.3.1/README.md +0 -127
- pgvector_template-0.3.1/pgvector_template/utils/__init__.py +0 -0
- pgvector_template-0.3.1/pgvector_template.egg-info/PKG-INFO +0 -152
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/LICENSE +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/__init__.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/core/embedder.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/core/search.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/db/__init__.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/db/connection.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/db/document_db.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/service/document_service.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/types.py +0 -0
- {pgvector_template-0.3.1/pgvector_template/models → pgvector_template-0.3.3/pgvector_template/utils}/__init__.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template/utils/metadata_filter.py +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template.egg-info/SOURCES.txt +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template.egg-info/dependency_links.txt +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/pgvector_template.egg-info/top_level.txt +0 -0
- {pgvector_template-0.3.1 → pgvector_template-0.3.3}/setup.cfg +0 -0
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pgvector-template
|
|
3
|
+
Version: 0.3.3
|
|
4
|
+
Summary: Template library for flexible PGVector RAG implementations
|
|
5
|
+
Author-email: DL <v49t9zpqd@mozmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/DavidLiuGit/PGVector-Template
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Requires-Python: >=3.11
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE
|
|
14
|
+
Requires-Dist: pgvector>=0.2.0
|
|
15
|
+
Requires-Dist: pydantic<3.0,>=2.11
|
|
16
|
+
Requires-Dist: sqlalchemy>=2.0.0
|
|
17
|
+
Requires-Dist: typing-extensions>=4.0.0
|
|
18
|
+
Provides-Extra: test
|
|
19
|
+
Requires-Dist: psycopg[binary]>=3.1.0; extra == "test"
|
|
20
|
+
Requires-Dist: pytest>=7.0.0; extra == "test"
|
|
21
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "test"
|
|
22
|
+
Requires-Dist: python-dotenv>=1.0.0; extra == "test"
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
25
|
+
Provides-Extra: dist
|
|
26
|
+
Requires-Dist: build>=1.2.2; extra == "dist"
|
|
27
|
+
Requires-Dist: twine>=6.1.0; extra == "dist"
|
|
28
|
+
Dynamic: license-file
|
|
29
|
+
|
|
30
|
+
# PGVector-Template
|
|
31
|
+
|
|
32
|
+
A flexible, production-ready template library for building Retrieval-Augmented Generation (RAG) applications using PostgreSQL with PGVector extensions.
|
|
33
|
+
|
|
34
|
+
## Overview
|
|
35
|
+
|
|
36
|
+
PGVector-Template provides a robust foundation for implementing vector-based document storage and retrieval systems. It offers a clean abstraction layer over PostgreSQL's PGVector extension, making it easy to build scalable RAG applications with proper document management, metadata handling, and efficient vector search capabilities.
|
|
37
|
+
|
|
38
|
+
## Quick Start
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
from pgvector_template import DatabaseManager, DocumentDatabaseManager
|
|
42
|
+
from pgvector_template.core.document import BaseDocument
|
|
43
|
+
from sqlalchemy import create_engine
|
|
44
|
+
from uuid import uuid4
|
|
45
|
+
|
|
46
|
+
# 1. Define your document model
|
|
47
|
+
class MyDocument(BaseDocument):
|
|
48
|
+
__tablename__ = "my_documents"
|
|
49
|
+
|
|
50
|
+
# 2. Set up database connection
|
|
51
|
+
engine = create_engine("postgresql://user:pass@localhost/mydb")
|
|
52
|
+
db_manager = DatabaseManager(engine)
|
|
53
|
+
db_manager.create_tables([MyDocument])
|
|
54
|
+
|
|
55
|
+
# 3. Create document manager
|
|
56
|
+
doc_manager = DocumentDatabaseManager(db_manager)
|
|
57
|
+
|
|
58
|
+
# 4. Insert a document
|
|
59
|
+
with db_manager.get_session() as session:
|
|
60
|
+
doc = MyDocument.from_props(
|
|
61
|
+
corpus_id=uuid4(),
|
|
62
|
+
chunk_index=0,
|
|
63
|
+
content="Your document content here",
|
|
64
|
+
embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
|
|
65
|
+
)
|
|
66
|
+
doc_manager.insert_document(session, doc)
|
|
67
|
+
|
|
68
|
+
# 5. Search similar documents
|
|
69
|
+
with db_manager.get_session() as session:
|
|
70
|
+
results = doc_manager.search_similar(
|
|
71
|
+
session, MyDocument, query_embedding=[0.1, 0.2, 0.3, ...], limit=5
|
|
72
|
+
)
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## Key Concepts
|
|
76
|
+
|
|
77
|
+
**Corpus vs Document**: Understanding the hierarchy is essential:
|
|
78
|
+
- **Corpus**: A complete source document (e.g., a full PDF, article, or book)
|
|
79
|
+
- **Document**: A chunk or segment of a corpus that fits within embedding limits
|
|
80
|
+
- **Collection**: A logical grouping of related corpora (e.g., "legal_docs", "user_manuals")
|
|
81
|
+
|
|
82
|
+
Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all sharing the same `corpus_id` but with different `chunk_index` values.
|
|
83
|
+
|
|
84
|
+
## Key Features
|
|
85
|
+
|
|
86
|
+
- **Flexible Document Model**: Abstract base classes for customizable document schemas
|
|
87
|
+
- **Vector Search**: Optimized HNSW indexing for fast similarity search
|
|
88
|
+
- **Metadata Management**: JSON-based flexible metadata with GIN indexing
|
|
89
|
+
- **Collection Support**: Organize documents into logical collections
|
|
90
|
+
- **Chunk Management**: Handle long content by chunking into retrievable documents
|
|
91
|
+
- **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
|
|
92
|
+
- **Type Safety**: Full Pydantic validation and type hints
|
|
93
|
+
- **Production Ready**: Comprehensive testing and error handling
|
|
94
|
+
|
|
95
|
+
## Architecture
|
|
96
|
+
|
|
97
|
+
The library is organized into several key components:
|
|
98
|
+
|
|
99
|
+
- **Core**: Document models, embedders, search functionality
|
|
100
|
+
- **Database**: Connection management and document database operations
|
|
101
|
+
- **Service**: High-level document service layer
|
|
102
|
+
- **Types**: Shared type definitions and schemas
|
|
103
|
+
|
|
104
|
+
## Installation
|
|
105
|
+
|
|
106
|
+
### Basic Installation
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
pip install pgvector-template
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
### With Database Driver
|
|
113
|
+
|
|
114
|
+
For production use, you'll also need a PostgreSQL driver:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
# For binary driver (recommended)
|
|
118
|
+
pip install pgvector-template psycopg[binary]
|
|
119
|
+
|
|
120
|
+
# Or for source driver
|
|
121
|
+
pip install pgvector-template psycopg
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### Prerequisites
|
|
125
|
+
|
|
126
|
+
- Python 3.11+
|
|
127
|
+
- PostgreSQL 12+ with PGVector extension
|
|
128
|
+
- For development: Additional test dependencies
|
|
129
|
+
|
|
130
|
+
## Configuration
|
|
131
|
+
|
|
132
|
+
### Database Setup
|
|
133
|
+
|
|
134
|
+
1. **Install PostgreSQL with PGVector extension**
|
|
135
|
+
2. **Create your database and enable the vector extension:**
|
|
136
|
+
|
|
137
|
+
```sql
|
|
138
|
+
CREATE EXTENSION IF NOT EXISTS vector;
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
3. **Set up your connection:**
|
|
142
|
+
|
|
143
|
+
```python
|
|
144
|
+
from sqlalchemy import create_engine
|
|
145
|
+
from pgvector_template import DatabaseManager
|
|
146
|
+
|
|
147
|
+
# Option 1: Direct connection string
|
|
148
|
+
engine = create_engine("postgresql://user:password@localhost:5432/mydb")
|
|
149
|
+
db_manager = DatabaseManager(engine)
|
|
150
|
+
|
|
151
|
+
# Option 2: From environment variable
|
|
152
|
+
import os
|
|
153
|
+
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
154
|
+
db_manager = DatabaseManager(engine)
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
### Production Configuration
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
from sqlalchemy import create_engine
|
|
161
|
+
from sqlalchemy.pool import QueuePool
|
|
162
|
+
|
|
163
|
+
engine = create_engine(
|
|
164
|
+
"postgresql://user:password@localhost:5432/mydb",
|
|
165
|
+
poolclass=QueuePool,
|
|
166
|
+
pool_size=10,
|
|
167
|
+
max_overflow=20,
|
|
168
|
+
pool_pre_ping=True,
|
|
169
|
+
)
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
### Environment Variables
|
|
173
|
+
|
|
174
|
+
```bash
|
|
175
|
+
# Development
|
|
176
|
+
DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
## Usage Examples
|
|
180
|
+
|
|
181
|
+
### 1. Define Your Document Model
|
|
182
|
+
|
|
183
|
+
```python
|
|
184
|
+
from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
|
|
185
|
+
from pydantic import Field
|
|
186
|
+
|
|
187
|
+
class MyDocumentMetadata(BaseDocumentMetadata):
|
|
188
|
+
source_type: str = Field(..., description="Type of source document")
|
|
189
|
+
author: str = Field(default="unknown", description="Document author")
|
|
190
|
+
|
|
191
|
+
class MyDocument(BaseDocument):
|
|
192
|
+
__tablename__ = "my_documents"
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
### 2. Insert Documents
|
|
196
|
+
|
|
197
|
+
```python
|
|
198
|
+
from pgvector_template.core.document import BaseDocumentOptionalProps
|
|
199
|
+
from uuid import uuid4
|
|
200
|
+
|
|
201
|
+
# Create document with metadata
|
|
202
|
+
metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
|
|
203
|
+
optional_props = BaseDocumentOptionalProps(
|
|
204
|
+
title="Chapter 1: Introduction",
|
|
205
|
+
collection="textbooks",
|
|
206
|
+
language="en",
|
|
207
|
+
tags=["education", "intro"]
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
doc = MyDocument.from_props(
|
|
211
|
+
corpus_id=uuid4(),
|
|
212
|
+
chunk_index=0,
|
|
213
|
+
content="This is the document content...",
|
|
214
|
+
embedding=your_embedding_vector, # list[float] with 1024 dimensions
|
|
215
|
+
metadata=metadata.to_dict(),
|
|
216
|
+
optional_props=optional_props
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
with db_manager.get_session() as session:
|
|
220
|
+
doc_manager.insert_document(session, doc)
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
### 3. Search Documents
|
|
224
|
+
|
|
225
|
+
```python
|
|
226
|
+
# Basic similarity search
|
|
227
|
+
with db_manager.get_session() as session:
|
|
228
|
+
results = doc_manager.search_similar(
|
|
229
|
+
session=session,
|
|
230
|
+
document_cls=MyDocument,
|
|
231
|
+
query_embedding=query_vector,
|
|
232
|
+
limit=10
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
# Search with filters
|
|
236
|
+
from pgvector_template.models.search import MetadataFilter
|
|
237
|
+
|
|
238
|
+
filters = [
|
|
239
|
+
MetadataFilter(key="source_type", value="pdf"),
|
|
240
|
+
MetadataFilter(key="author", value="John Doe")
|
|
241
|
+
]
|
|
242
|
+
|
|
243
|
+
results = doc_manager.search_similar(
|
|
244
|
+
session=session,
|
|
245
|
+
document_cls=MyDocument,
|
|
246
|
+
query_embedding=query_vector,
|
|
247
|
+
limit=10,
|
|
248
|
+
metadata_filters=filters,
|
|
249
|
+
collection="textbooks"
|
|
250
|
+
)
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
### 4. Manage Collections
|
|
254
|
+
|
|
255
|
+
```python
|
|
256
|
+
# Get all documents in a collection
|
|
257
|
+
with db_manager.get_session() as session:
|
|
258
|
+
docs = doc_manager.get_documents_by_collection(
|
|
259
|
+
session, MyDocument, "textbooks"
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
# Get all chunks from a corpus
|
|
263
|
+
corpus_docs = doc_manager.get_documents_by_corpus_id(
|
|
264
|
+
session, MyDocument, corpus_id
|
|
265
|
+
)
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
## Concept Reference
|
|
269
|
+
|
|
270
|
+
### Core Classes
|
|
271
|
+
|
|
272
|
+
- **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
|
|
273
|
+
- **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
|
|
274
|
+
- **`BaseDocumentMetadata`**: Base schema for structured document metadata
|
|
275
|
+
- **`DatabaseManager`**: Database connection and session management
|
|
276
|
+
- **`DocumentDatabaseManager`**: High-level document CRUD operations
|
|
277
|
+
|
|
278
|
+
## Testing
|
|
279
|
+
|
|
280
|
+
Install dependencies (preferably in a virtualenv) before running tests:
|
|
281
|
+
```bash
|
|
282
|
+
pip install -e .[test]
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
### Unit Tests
|
|
286
|
+
```bash
|
|
287
|
+
python -m unittest
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
### Integration Tests
|
|
291
|
+
|
|
292
|
+
Integration tests require a PostgreSQL database with PGVector extension. Set up your test database and configure the connection in `integ-tests/.env`:
|
|
293
|
+
|
|
294
|
+
```bash
|
|
295
|
+
python -m unittest discover -s integ-tests
|
|
296
|
+
```
|
|
297
|
+
|
|
298
|
+
## Contributing
|
|
299
|
+
|
|
300
|
+
1. Fork the repository
|
|
301
|
+
2. Create a feature branch
|
|
302
|
+
3. Make your changes with tests
|
|
303
|
+
4. Run the test suite
|
|
304
|
+
5. Submit a pull request
|
|
305
|
+
|
|
306
|
+
### Development Setup
|
|
307
|
+
|
|
308
|
+
```bash
|
|
309
|
+
pip install -e .[dev,test]
|
|
310
|
+
black . # Format code
|
|
311
|
+
```
|
|
312
|
+
|
|
313
|
+
## License
|
|
314
|
+
|
|
315
|
+
MIT License - see [LICENSE](LICENSE) file for details.
|
|
316
|
+
|
|
317
|
+
## Links
|
|
318
|
+
|
|
319
|
+
- [GitHub Repository](https://github.com/DavidLiuGit/PGVector-Template)
|
|
320
|
+
- [PGVector Documentation](https://github.com/pgvector/pgvector)
|
|
@@ -0,0 +1,291 @@
|
|
|
1
|
+
# PGVector-Template
|
|
2
|
+
|
|
3
|
+
A flexible, production-ready template library for building Retrieval-Augmented Generation (RAG) applications using PostgreSQL with PGVector extensions.
|
|
4
|
+
|
|
5
|
+
## Overview
|
|
6
|
+
|
|
7
|
+
PGVector-Template provides a robust foundation for implementing vector-based document storage and retrieval systems. It offers a clean abstraction layer over PostgreSQL's PGVector extension, making it easy to build scalable RAG applications with proper document management, metadata handling, and efficient vector search capabilities.
|
|
8
|
+
|
|
9
|
+
## Quick Start
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
from pgvector_template import DatabaseManager, DocumentDatabaseManager
|
|
13
|
+
from pgvector_template.core.document import BaseDocument
|
|
14
|
+
from sqlalchemy import create_engine
|
|
15
|
+
from uuid import uuid4
|
|
16
|
+
|
|
17
|
+
# 1. Define your document model
|
|
18
|
+
class MyDocument(BaseDocument):
|
|
19
|
+
__tablename__ = "my_documents"
|
|
20
|
+
|
|
21
|
+
# 2. Set up database connection
|
|
22
|
+
engine = create_engine("postgresql://user:pass@localhost/mydb")
|
|
23
|
+
db_manager = DatabaseManager(engine)
|
|
24
|
+
db_manager.create_tables([MyDocument])
|
|
25
|
+
|
|
26
|
+
# 3. Create document manager
|
|
27
|
+
doc_manager = DocumentDatabaseManager(db_manager)
|
|
28
|
+
|
|
29
|
+
# 4. Insert a document
|
|
30
|
+
with db_manager.get_session() as session:
|
|
31
|
+
doc = MyDocument.from_props(
|
|
32
|
+
corpus_id=uuid4(),
|
|
33
|
+
chunk_index=0,
|
|
34
|
+
content="Your document content here",
|
|
35
|
+
embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
|
|
36
|
+
)
|
|
37
|
+
doc_manager.insert_document(session, doc)
|
|
38
|
+
|
|
39
|
+
# 5. Search similar documents
|
|
40
|
+
with db_manager.get_session() as session:
|
|
41
|
+
results = doc_manager.search_similar(
|
|
42
|
+
session, MyDocument, query_embedding=[0.1, 0.2, 0.3, ...], limit=5
|
|
43
|
+
)
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## Key Concepts
|
|
47
|
+
|
|
48
|
+
**Corpus vs Document**: Understanding the hierarchy is essential:
|
|
49
|
+
- **Corpus**: A complete source document (e.g., a full PDF, article, or book)
|
|
50
|
+
- **Document**: A chunk or segment of a corpus that fits within embedding limits
|
|
51
|
+
- **Collection**: A logical grouping of related corpora (e.g., "legal_docs", "user_manuals")
|
|
52
|
+
|
|
53
|
+
Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all sharing the same `corpus_id` but with different `chunk_index` values.
|
|
54
|
+
|
|
55
|
+
## Key Features
|
|
56
|
+
|
|
57
|
+
- **Flexible Document Model**: Abstract base classes for customizable document schemas
|
|
58
|
+
- **Vector Search**: Optimized HNSW indexing for fast similarity search
|
|
59
|
+
- **Metadata Management**: JSON-based flexible metadata with GIN indexing
|
|
60
|
+
- **Collection Support**: Organize documents into logical collections
|
|
61
|
+
- **Chunk Management**: Handle long content by chunking into retrievable documents
|
|
62
|
+
- **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
|
|
63
|
+
- **Type Safety**: Full Pydantic validation and type hints
|
|
64
|
+
- **Production Ready**: Comprehensive testing and error handling
|
|
65
|
+
|
|
66
|
+
## Architecture
|
|
67
|
+
|
|
68
|
+
The library is organized into several key components:
|
|
69
|
+
|
|
70
|
+
- **Core**: Document models, embedders, search functionality
|
|
71
|
+
- **Database**: Connection management and document database operations
|
|
72
|
+
- **Service**: High-level document service layer
|
|
73
|
+
- **Types**: Shared type definitions and schemas
|
|
74
|
+
|
|
75
|
+
## Installation
|
|
76
|
+
|
|
77
|
+
### Basic Installation
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pip install pgvector-template
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
### With Database Driver
|
|
84
|
+
|
|
85
|
+
For production use, you'll also need a PostgreSQL driver:
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
# For binary driver (recommended)
|
|
89
|
+
pip install pgvector-template psycopg[binary]
|
|
90
|
+
|
|
91
|
+
# Or for source driver
|
|
92
|
+
pip install pgvector-template psycopg
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
### Prerequisites
|
|
96
|
+
|
|
97
|
+
- Python 3.11+
|
|
98
|
+
- PostgreSQL 12+ with PGVector extension
|
|
99
|
+
- For development: Additional test dependencies
|
|
100
|
+
|
|
101
|
+
## Configuration
|
|
102
|
+
|
|
103
|
+
### Database Setup
|
|
104
|
+
|
|
105
|
+
1. **Install PostgreSQL with PGVector extension**
|
|
106
|
+
2. **Create your database and enable the vector extension:**
|
|
107
|
+
|
|
108
|
+
```sql
|
|
109
|
+
CREATE EXTENSION IF NOT EXISTS vector;
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
3. **Set up your connection:**
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
from sqlalchemy import create_engine
|
|
116
|
+
from pgvector_template import DatabaseManager
|
|
117
|
+
|
|
118
|
+
# Option 1: Direct connection string
|
|
119
|
+
engine = create_engine("postgresql://user:password@localhost:5432/mydb")
|
|
120
|
+
db_manager = DatabaseManager(engine)
|
|
121
|
+
|
|
122
|
+
# Option 2: From environment variable
|
|
123
|
+
import os
|
|
124
|
+
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
125
|
+
db_manager = DatabaseManager(engine)
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
### Production Configuration
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
from sqlalchemy import create_engine
|
|
132
|
+
from sqlalchemy.pool import QueuePool
|
|
133
|
+
|
|
134
|
+
engine = create_engine(
|
|
135
|
+
"postgresql://user:password@localhost:5432/mydb",
|
|
136
|
+
poolclass=QueuePool,
|
|
137
|
+
pool_size=10,
|
|
138
|
+
max_overflow=20,
|
|
139
|
+
pool_pre_ping=True,
|
|
140
|
+
)
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
### Environment Variables
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
# Development
|
|
147
|
+
DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
## Usage Examples
|
|
151
|
+
|
|
152
|
+
### 1. Define Your Document Model
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
|
|
156
|
+
from pydantic import Field
|
|
157
|
+
|
|
158
|
+
class MyDocumentMetadata(BaseDocumentMetadata):
|
|
159
|
+
source_type: str = Field(..., description="Type of source document")
|
|
160
|
+
author: str = Field(default="unknown", description="Document author")
|
|
161
|
+
|
|
162
|
+
class MyDocument(BaseDocument):
|
|
163
|
+
__tablename__ = "my_documents"
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
### 2. Insert Documents
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
from pgvector_template.core.document import BaseDocumentOptionalProps
|
|
170
|
+
from uuid import uuid4
|
|
171
|
+
|
|
172
|
+
# Create document with metadata
|
|
173
|
+
metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
|
|
174
|
+
optional_props = BaseDocumentOptionalProps(
|
|
175
|
+
title="Chapter 1: Introduction",
|
|
176
|
+
collection="textbooks",
|
|
177
|
+
language="en",
|
|
178
|
+
tags=["education", "intro"]
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
doc = MyDocument.from_props(
|
|
182
|
+
corpus_id=uuid4(),
|
|
183
|
+
chunk_index=0,
|
|
184
|
+
content="This is the document content...",
|
|
185
|
+
embedding=your_embedding_vector, # list[float] with 1024 dimensions
|
|
186
|
+
metadata=metadata.to_dict(),
|
|
187
|
+
optional_props=optional_props
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
with db_manager.get_session() as session:
|
|
191
|
+
doc_manager.insert_document(session, doc)
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
### 3. Search Documents
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
# Basic similarity search
|
|
198
|
+
with db_manager.get_session() as session:
|
|
199
|
+
results = doc_manager.search_similar(
|
|
200
|
+
session=session,
|
|
201
|
+
document_cls=MyDocument,
|
|
202
|
+
query_embedding=query_vector,
|
|
203
|
+
limit=10
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
# Search with filters
|
|
207
|
+
from pgvector_template.models.search import MetadataFilter
|
|
208
|
+
|
|
209
|
+
filters = [
|
|
210
|
+
MetadataFilter(key="source_type", value="pdf"),
|
|
211
|
+
MetadataFilter(key="author", value="John Doe")
|
|
212
|
+
]
|
|
213
|
+
|
|
214
|
+
results = doc_manager.search_similar(
|
|
215
|
+
session=session,
|
|
216
|
+
document_cls=MyDocument,
|
|
217
|
+
query_embedding=query_vector,
|
|
218
|
+
limit=10,
|
|
219
|
+
metadata_filters=filters,
|
|
220
|
+
collection="textbooks"
|
|
221
|
+
)
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
### 4. Manage Collections
|
|
225
|
+
|
|
226
|
+
```python
|
|
227
|
+
# Get all documents in a collection
|
|
228
|
+
with db_manager.get_session() as session:
|
|
229
|
+
docs = doc_manager.get_documents_by_collection(
|
|
230
|
+
session, MyDocument, "textbooks"
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
# Get all chunks from a corpus
|
|
234
|
+
corpus_docs = doc_manager.get_documents_by_corpus_id(
|
|
235
|
+
session, MyDocument, corpus_id
|
|
236
|
+
)
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
## Concept Reference
|
|
240
|
+
|
|
241
|
+
### Core Classes
|
|
242
|
+
|
|
243
|
+
- **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
|
|
244
|
+
- **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
|
|
245
|
+
- **`BaseDocumentMetadata`**: Base schema for structured document metadata
|
|
246
|
+
- **`DatabaseManager`**: Database connection and session management
|
|
247
|
+
- **`DocumentDatabaseManager`**: High-level document CRUD operations
|
|
248
|
+
|
|
249
|
+
## Testing
|
|
250
|
+
|
|
251
|
+
Install dependencies (preferably in a virtualenv) before running tests:
|
|
252
|
+
```bash
|
|
253
|
+
pip install -e .[test]
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
### Unit Tests
|
|
257
|
+
```bash
|
|
258
|
+
python -m unittest
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
### Integration Tests
|
|
262
|
+
|
|
263
|
+
Integration tests require a PostgreSQL database with PGVector extension. Set up your test database and configure the connection in `integ-tests/.env`:
|
|
264
|
+
|
|
265
|
+
```bash
|
|
266
|
+
python -m unittest discover -s integ-tests
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
## Contributing
|
|
270
|
+
|
|
271
|
+
1. Fork the repository
|
|
272
|
+
2. Create a feature branch
|
|
273
|
+
3. Make your changes with tests
|
|
274
|
+
4. Run the test suite
|
|
275
|
+
5. Submit a pull request
|
|
276
|
+
|
|
277
|
+
### Development Setup
|
|
278
|
+
|
|
279
|
+
```bash
|
|
280
|
+
pip install -e .[dev,test]
|
|
281
|
+
black . # Format code
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
## License
|
|
285
|
+
|
|
286
|
+
MIT License - see [LICENSE](LICENSE) file for details.
|
|
287
|
+
|
|
288
|
+
## Links
|
|
289
|
+
|
|
290
|
+
- [GitHub Repository](https://github.com/DavidLiuGit/PGVector-Template)
|
|
291
|
+
- [PGVector Documentation](https://github.com/pgvector/pgvector)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata, BaseDocumentOptionalProps
|
|
2
2
|
from pgvector_template.core.embedder import BaseEmbeddingProvider
|
|
3
3
|
from pgvector_template.core.manager import BaseCorpusManager, BaseCorpusManagerConfig, Corpus
|
|
4
|
-
from pgvector_template.core.search import
|
|
4
|
+
from pgvector_template.core.search import BaseSearchClientConfig, BaseSearchClient
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
__all__ = [
|
|
@@ -16,8 +16,6 @@ __all__ = [
|
|
|
16
16
|
"BaseCorpusManager",
|
|
17
17
|
"BaseCorpusManagerConfig",
|
|
18
18
|
### search
|
|
19
|
-
"SearchQuery",
|
|
20
|
-
"RetrievalResult",
|
|
21
19
|
"BaseSearchClientConfig",
|
|
22
20
|
"BaseSearchClient",
|
|
23
21
|
]
|
|
@@ -195,9 +195,12 @@ class BaseDocumentMetadata(BaseModel):
|
|
|
195
195
|
"""
|
|
196
196
|
|
|
197
197
|
document_type: str = Field(
|
|
198
|
-
..., description="
|
|
198
|
+
..., description="Original document type/format/extension, e.g. md, pdf, html, json, etc"
|
|
199
|
+
)
|
|
200
|
+
schema_version: str = Field(
|
|
201
|
+
default="1.0",
|
|
202
|
+
description="Schema version for the metadata. Intended for housekeeping only",
|
|
199
203
|
)
|
|
200
|
-
schema_version: str = Field(default="1.0", description="Schema version for the metadata")
|
|
201
204
|
|
|
202
205
|
def to_dict(self) -> dict[str, Any]:
|
|
203
206
|
return self.model_dump()
|