pgvector-template 0.3.2__tar.gz → 0.3.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. pgvector_template-0.3.3/PKG-INFO +320 -0
  2. pgvector_template-0.3.3/README.md +291 -0
  3. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/core/manager.py +41 -15
  4. pgvector_template-0.3.3/pgvector_template.egg-info/PKG-INFO +320 -0
  5. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template.egg-info/requires.txt +4 -0
  6. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pyproject.toml +5 -1
  7. pgvector_template-0.3.2/PKG-INFO +0 -152
  8. pgvector_template-0.3.2/README.md +0 -127
  9. pgvector_template-0.3.2/pgvector_template.egg-info/PKG-INFO +0 -152
  10. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/LICENSE +0 -0
  11. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/__init__.py +0 -0
  12. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/core/__init__.py +0 -0
  13. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/core/document.py +0 -0
  14. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/core/embedder.py +0 -0
  15. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/core/search.py +0 -0
  16. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/db/__init__.py +0 -0
  17. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/db/connection.py +0 -0
  18. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/db/document_db.py +0 -0
  19. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/models/__init__.py +0 -0
  20. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/models/search.py +0 -0
  21. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/service/__init__.py +0 -0
  22. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/service/document_service.py +0 -0
  23. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/types.py +0 -0
  24. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/utils/__init__.py +0 -0
  25. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template/utils/metadata_filter.py +0 -0
  26. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template.egg-info/SOURCES.txt +0 -0
  27. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template.egg-info/dependency_links.txt +0 -0
  28. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/pgvector_template.egg-info/top_level.txt +0 -0
  29. {pgvector_template-0.3.2 → pgvector_template-0.3.3}/setup.cfg +0 -0
@@ -0,0 +1,320 @@
1
+ Metadata-Version: 2.4
2
+ Name: pgvector-template
3
+ Version: 0.3.3
4
+ Summary: Template library for flexible PGVector RAG implementations
5
+ Author-email: DL <v49t9zpqd@mozmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/DavidLiuGit/PGVector-Template
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: OS Independent
11
+ Requires-Python: >=3.11
12
+ Description-Content-Type: text/markdown
13
+ License-File: LICENSE
14
+ Requires-Dist: pgvector>=0.2.0
15
+ Requires-Dist: pydantic<3.0,>=2.11
16
+ Requires-Dist: sqlalchemy>=2.0.0
17
+ Requires-Dist: typing-extensions>=4.0.0
18
+ Provides-Extra: test
19
+ Requires-Dist: psycopg[binary]>=3.1.0; extra == "test"
20
+ Requires-Dist: pytest>=7.0.0; extra == "test"
21
+ Requires-Dist: pytest-cov>=4.0.0; extra == "test"
22
+ Requires-Dist: python-dotenv>=1.0.0; extra == "test"
23
+ Provides-Extra: dev
24
+ Requires-Dist: black>=23.0.0; extra == "dev"
25
+ Provides-Extra: dist
26
+ Requires-Dist: build>=1.2.2; extra == "dist"
27
+ Requires-Dist: twine>=6.1.0; extra == "dist"
28
+ Dynamic: license-file
29
+
30
+ # PGVector-Template
31
+
32
+ A flexible, production-ready template library for building Retrieval-Augmented Generation (RAG) applications using PostgreSQL with PGVector extensions.
33
+
34
+ ## Overview
35
+
36
+ PGVector-Template provides a robust foundation for implementing vector-based document storage and retrieval systems. It offers a clean abstraction layer over PostgreSQL's PGVector extension, making it easy to build scalable RAG applications with proper document management, metadata handling, and efficient vector search capabilities.
37
+
38
+ ## Quick Start
39
+
40
+ ```python
41
+ from pgvector_template import DatabaseManager, DocumentDatabaseManager
42
+ from pgvector_template.core.document import BaseDocument
43
+ from sqlalchemy import create_engine
44
+ from uuid import uuid4
45
+
46
+ # 1. Define your document model
47
+ class MyDocument(BaseDocument):
48
+ __tablename__ = "my_documents"
49
+
50
+ # 2. Set up database connection
51
+ engine = create_engine("postgresql://user:pass@localhost/mydb")
52
+ db_manager = DatabaseManager(engine)
53
+ db_manager.create_tables([MyDocument])
54
+
55
+ # 3. Create document manager
56
+ doc_manager = DocumentDatabaseManager(db_manager)
57
+
58
+ # 4. Insert a document
59
+ with db_manager.get_session() as session:
60
+ doc = MyDocument.from_props(
61
+ corpus_id=uuid4(),
62
+ chunk_index=0,
63
+ content="Your document content here",
64
+ embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
65
+ )
66
+ doc_manager.insert_document(session, doc)
67
+
68
+ # 5. Search similar documents
69
+ with db_manager.get_session() as session:
70
+ results = doc_manager.search_similar(
71
+ session, MyDocument, query_embedding=[0.1, 0.2, 0.3, ...], limit=5
72
+ )
73
+ ```
74
+
75
+ ## Key Concepts
76
+
77
+ **Corpus vs Document**: Understanding the hierarchy is essential:
78
+ - **Corpus**: A complete source document (e.g., a full PDF, article, or book)
79
+ - **Document**: A chunk or segment of a corpus that fits within embedding limits
80
+ - **Collection**: A logical grouping of related corpora (e.g., "legal_docs", "user_manuals")
81
+
82
+ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all sharing the same `corpus_id` but with different `chunk_index` values.
83
+
84
+ ## Key Features
85
+
86
+ - **Flexible Document Model**: Abstract base classes for customizable document schemas
87
+ - **Vector Search**: Optimized HNSW indexing for fast similarity search
88
+ - **Metadata Management**: JSON-based flexible metadata with GIN indexing
89
+ - **Collection Support**: Organize documents into logical collections
90
+ - **Chunk Management**: Handle long content by chunking into retrievable documents
91
+ - **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
92
+ - **Type Safety**: Full Pydantic validation and type hints
93
+ - **Production Ready**: Comprehensive testing and error handling
94
+
95
+ ## Architecture
96
+
97
+ The library is organized into several key components:
98
+
99
+ - **Core**: Document models, embedders, search functionality
100
+ - **Database**: Connection management and document database operations
101
+ - **Service**: High-level document service layer
102
+ - **Types**: Shared type definitions and schemas
103
+
104
+ ## Installation
105
+
106
+ ### Basic Installation
107
+
108
+ ```bash
109
+ pip install pgvector-template
110
+ ```
111
+
112
+ ### With Database Driver
113
+
114
+ For production use, you'll also need a PostgreSQL driver:
115
+
116
+ ```bash
117
+ # For binary driver (recommended)
118
+ pip install pgvector-template psycopg[binary]
119
+
120
+ # Or for source driver
121
+ pip install pgvector-template psycopg
122
+ ```
123
+
124
+ ### Prerequisites
125
+
126
+ - Python 3.11+
127
+ - PostgreSQL 12+ with PGVector extension
128
+ - For development: Additional test dependencies
129
+
130
+ ## Configuration
131
+
132
+ ### Database Setup
133
+
134
+ 1. **Install PostgreSQL with PGVector extension**
135
+ 2. **Create your database and enable the vector extension:**
136
+
137
+ ```sql
138
+ CREATE EXTENSION IF NOT EXISTS vector;
139
+ ```
140
+
141
+ 3. **Set up your connection:**
142
+
143
+ ```python
144
+ from sqlalchemy import create_engine
145
+ from pgvector_template import DatabaseManager
146
+
147
+ # Option 1: Direct connection string
148
+ engine = create_engine("postgresql://user:password@localhost:5432/mydb")
149
+ db_manager = DatabaseManager(engine)
150
+
151
+ # Option 2: From environment variable
152
+ import os
153
+ engine = create_engine(os.getenv("DATABASE_URL"))
154
+ db_manager = DatabaseManager(engine)
155
+ ```
156
+
157
+ ### Production Configuration
158
+
159
+ ```python
160
+ from sqlalchemy import create_engine
161
+ from sqlalchemy.pool import QueuePool
162
+
163
+ engine = create_engine(
164
+ "postgresql://user:password@localhost:5432/mydb",
165
+ poolclass=QueuePool,
166
+ pool_size=10,
167
+ max_overflow=20,
168
+ pool_pre_ping=True,
169
+ )
170
+ ```
171
+
172
+ ### Environment Variables
173
+
174
+ ```bash
175
+ # Development
176
+ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
177
+ ```
178
+
179
+ ## Usage Examples
180
+
181
+ ### 1. Define Your Document Model
182
+
183
+ ```python
184
+ from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
185
+ from pydantic import Field
186
+
187
+ class MyDocumentMetadata(BaseDocumentMetadata):
188
+ source_type: str = Field(..., description="Type of source document")
189
+ author: str = Field(default="unknown", description="Document author")
190
+
191
+ class MyDocument(BaseDocument):
192
+ __tablename__ = "my_documents"
193
+ ```
194
+
195
+ ### 2. Insert Documents
196
+
197
+ ```python
198
+ from pgvector_template.core.document import BaseDocumentOptionalProps
199
+ from uuid import uuid4
200
+
201
+ # Create document with metadata
202
+ metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
203
+ optional_props = BaseDocumentOptionalProps(
204
+ title="Chapter 1: Introduction",
205
+ collection="textbooks",
206
+ language="en",
207
+ tags=["education", "intro"]
208
+ )
209
+
210
+ doc = MyDocument.from_props(
211
+ corpus_id=uuid4(),
212
+ chunk_index=0,
213
+ content="This is the document content...",
214
+ embedding=your_embedding_vector, # list[float] with 1024 dimensions
215
+ metadata=metadata.to_dict(),
216
+ optional_props=optional_props
217
+ )
218
+
219
+ with db_manager.get_session() as session:
220
+ doc_manager.insert_document(session, doc)
221
+ ```
222
+
223
+ ### 3. Search Documents
224
+
225
+ ```python
226
+ # Basic similarity search
227
+ with db_manager.get_session() as session:
228
+ results = doc_manager.search_similar(
229
+ session=session,
230
+ document_cls=MyDocument,
231
+ query_embedding=query_vector,
232
+ limit=10
233
+ )
234
+
235
+ # Search with filters
236
+ from pgvector_template.models.search import MetadataFilter
237
+
238
+ filters = [
239
+ MetadataFilter(key="source_type", value="pdf"),
240
+ MetadataFilter(key="author", value="John Doe")
241
+ ]
242
+
243
+ results = doc_manager.search_similar(
244
+ session=session,
245
+ document_cls=MyDocument,
246
+ query_embedding=query_vector,
247
+ limit=10,
248
+ metadata_filters=filters,
249
+ collection="textbooks"
250
+ )
251
+ ```
252
+
253
+ ### 4. Manage Collections
254
+
255
+ ```python
256
+ # Get all documents in a collection
257
+ with db_manager.get_session() as session:
258
+ docs = doc_manager.get_documents_by_collection(
259
+ session, MyDocument, "textbooks"
260
+ )
261
+
262
+ # Get all chunks from a corpus
263
+ corpus_docs = doc_manager.get_documents_by_corpus_id(
264
+ session, MyDocument, corpus_id
265
+ )
266
+ ```
267
+
268
+ ## Concept Reference
269
+
270
+ ### Core Classes
271
+
272
+ - **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
273
+ - **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
274
+ - **`BaseDocumentMetadata`**: Base schema for structured document metadata
275
+ - **`DatabaseManager`**: Database connection and session management
276
+ - **`DocumentDatabaseManager`**: High-level document CRUD operations
277
+
278
+ ## Testing
279
+
280
+ Install dependencies (preferably in a virtualenv) before running tests:
281
+ ```bash
282
+ pip install -e .[test]
283
+ ```
284
+
285
+ ### Unit Tests
286
+ ```bash
287
+ python -m unittest
288
+ ```
289
+
290
+ ### Integration Tests
291
+
292
+ Integration tests require a PostgreSQL database with PGVector extension. Set up your test database and configure the connection in `integ-tests/.env`:
293
+
294
+ ```bash
295
+ python -m unittest discover -s integ-tests
296
+ ```
297
+
298
+ ## Contributing
299
+
300
+ 1. Fork the repository
301
+ 2. Create a feature branch
302
+ 3. Make your changes with tests
303
+ 4. Run the test suite
304
+ 5. Submit a pull request
305
+
306
+ ### Development Setup
307
+
308
+ ```bash
309
+ pip install -e .[dev,test]
310
+ black . # Format code
311
+ ```
312
+
313
+ ## License
314
+
315
+ MIT License - see [LICENSE](LICENSE) file for details.
316
+
317
+ ## Links
318
+
319
+ - [GitHub Repository](https://github.com/DavidLiuGit/PGVector-Template)
320
+ - [PGVector Documentation](https://github.com/pgvector/pgvector)
@@ -0,0 +1,291 @@
1
+ # PGVector-Template
2
+
3
+ A flexible, production-ready template library for building Retrieval-Augmented Generation (RAG) applications using PostgreSQL with PGVector extensions.
4
+
5
+ ## Overview
6
+
7
+ PGVector-Template provides a robust foundation for implementing vector-based document storage and retrieval systems. It offers a clean abstraction layer over PostgreSQL's PGVector extension, making it easy to build scalable RAG applications with proper document management, metadata handling, and efficient vector search capabilities.
8
+
9
+ ## Quick Start
10
+
11
+ ```python
12
+ from pgvector_template import DatabaseManager, DocumentDatabaseManager
13
+ from pgvector_template.core.document import BaseDocument
14
+ from sqlalchemy import create_engine
15
+ from uuid import uuid4
16
+
17
+ # 1. Define your document model
18
+ class MyDocument(BaseDocument):
19
+ __tablename__ = "my_documents"
20
+
21
+ # 2. Set up database connection
22
+ engine = create_engine("postgresql://user:pass@localhost/mydb")
23
+ db_manager = DatabaseManager(engine)
24
+ db_manager.create_tables([MyDocument])
25
+
26
+ # 3. Create document manager
27
+ doc_manager = DocumentDatabaseManager(db_manager)
28
+
29
+ # 4. Insert a document
30
+ with db_manager.get_session() as session:
31
+ doc = MyDocument.from_props(
32
+ corpus_id=uuid4(),
33
+ chunk_index=0,
34
+ content="Your document content here",
35
+ embedding=[0.1, 0.2, 0.3, ...], # Your embedding vector
36
+ )
37
+ doc_manager.insert_document(session, doc)
38
+
39
+ # 5. Search similar documents
40
+ with db_manager.get_session() as session:
41
+ results = doc_manager.search_similar(
42
+ session, MyDocument, query_embedding=[0.1, 0.2, 0.3, ...], limit=5
43
+ )
44
+ ```
45
+
46
+ ## Key Concepts
47
+
48
+ **Corpus vs Document**: Understanding the hierarchy is essential:
49
+ - **Corpus**: A complete source document (e.g., a full PDF, article, or book)
50
+ - **Document**: A chunk or segment of a corpus that fits within embedding limits
51
+ - **Collection**: A logical grouping of related corpora (e.g., "legal_docs", "user_manuals")
52
+
53
+ Example: A 50-page PDF (corpus) might be split into 200 documents (chunks), all sharing the same `corpus_id` but with different `chunk_index` values.
54
+
55
+ ## Key Features
56
+
57
+ - **Flexible Document Model**: Abstract base classes for customizable document schemas
58
+ - **Vector Search**: Optimized HNSW indexing for fast similarity search
59
+ - **Metadata Management**: JSON-based flexible metadata with GIN indexing
60
+ - **Collection Support**: Organize documents into logical collections
61
+ - **Chunk Management**: Handle long content by chunking into retrievable documents
62
+ - **Database Abstraction**: Clean SQLAlchemy-based database layer with schema creation API
63
+ - **Type Safety**: Full Pydantic validation and type hints
64
+ - **Production Ready**: Comprehensive testing and error handling
65
+
66
+ ## Architecture
67
+
68
+ The library is organized into several key components:
69
+
70
+ - **Core**: Document models, embedders, search functionality
71
+ - **Database**: Connection management and document database operations
72
+ - **Service**: High-level document service layer
73
+ - **Types**: Shared type definitions and schemas
74
+
75
+ ## Installation
76
+
77
+ ### Basic Installation
78
+
79
+ ```bash
80
+ pip install pgvector-template
81
+ ```
82
+
83
+ ### With Database Driver
84
+
85
+ For production use, you'll also need a PostgreSQL driver:
86
+
87
+ ```bash
88
+ # For binary driver (recommended)
89
+ pip install pgvector-template psycopg[binary]
90
+
91
+ # Or for source driver
92
+ pip install pgvector-template psycopg
93
+ ```
94
+
95
+ ### Prerequisites
96
+
97
+ - Python 3.11+
98
+ - PostgreSQL 12+ with PGVector extension
99
+ - For development: Additional test dependencies
100
+
101
+ ## Configuration
102
+
103
+ ### Database Setup
104
+
105
+ 1. **Install PostgreSQL with PGVector extension**
106
+ 2. **Create your database and enable the vector extension:**
107
+
108
+ ```sql
109
+ CREATE EXTENSION IF NOT EXISTS vector;
110
+ ```
111
+
112
+ 3. **Set up your connection:**
113
+
114
+ ```python
115
+ from sqlalchemy import create_engine
116
+ from pgvector_template import DatabaseManager
117
+
118
+ # Option 1: Direct connection string
119
+ engine = create_engine("postgresql://user:password@localhost:5432/mydb")
120
+ db_manager = DatabaseManager(engine)
121
+
122
+ # Option 2: From environment variable
123
+ import os
124
+ engine = create_engine(os.getenv("DATABASE_URL"))
125
+ db_manager = DatabaseManager(engine)
126
+ ```
127
+
128
+ ### Production Configuration
129
+
130
+ ```python
131
+ from sqlalchemy import create_engine
132
+ from sqlalchemy.pool import QueuePool
133
+
134
+ engine = create_engine(
135
+ "postgresql://user:password@localhost:5432/mydb",
136
+ poolclass=QueuePool,
137
+ pool_size=10,
138
+ max_overflow=20,
139
+ pool_pre_ping=True,
140
+ )
141
+ ```
142
+
143
+ ### Environment Variables
144
+
145
+ ```bash
146
+ # Development
147
+ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
148
+ ```
149
+
150
+ ## Usage Examples
151
+
152
+ ### 1. Define Your Document Model
153
+
154
+ ```python
155
+ from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
156
+ from pydantic import Field
157
+
158
+ class MyDocumentMetadata(BaseDocumentMetadata):
159
+ source_type: str = Field(..., description="Type of source document")
160
+ author: str = Field(default="unknown", description="Document author")
161
+
162
+ class MyDocument(BaseDocument):
163
+ __tablename__ = "my_documents"
164
+ ```
165
+
166
+ ### 2. Insert Documents
167
+
168
+ ```python
169
+ from pgvector_template.core.document import BaseDocumentOptionalProps
170
+ from uuid import uuid4
171
+
172
+ # Create document with metadata
173
+ metadata = MyDocumentMetadata(source_type="pdf", author="John Doe")
174
+ optional_props = BaseDocumentOptionalProps(
175
+ title="Chapter 1: Introduction",
176
+ collection="textbooks",
177
+ language="en",
178
+ tags=["education", "intro"]
179
+ )
180
+
181
+ doc = MyDocument.from_props(
182
+ corpus_id=uuid4(),
183
+ chunk_index=0,
184
+ content="This is the document content...",
185
+ embedding=your_embedding_vector, # list[float] with 1024 dimensions
186
+ metadata=metadata.to_dict(),
187
+ optional_props=optional_props
188
+ )
189
+
190
+ with db_manager.get_session() as session:
191
+ doc_manager.insert_document(session, doc)
192
+ ```
193
+
194
+ ### 3. Search Documents
195
+
196
+ ```python
197
+ # Basic similarity search
198
+ with db_manager.get_session() as session:
199
+ results = doc_manager.search_similar(
200
+ session=session,
201
+ document_cls=MyDocument,
202
+ query_embedding=query_vector,
203
+ limit=10
204
+ )
205
+
206
+ # Search with filters
207
+ from pgvector_template.models.search import MetadataFilter
208
+
209
+ filters = [
210
+ MetadataFilter(key="source_type", value="pdf"),
211
+ MetadataFilter(key="author", value="John Doe")
212
+ ]
213
+
214
+ results = doc_manager.search_similar(
215
+ session=session,
216
+ document_cls=MyDocument,
217
+ query_embedding=query_vector,
218
+ limit=10,
219
+ metadata_filters=filters,
220
+ collection="textbooks"
221
+ )
222
+ ```
223
+
224
+ ### 4. Manage Collections
225
+
226
+ ```python
227
+ # Get all documents in a collection
228
+ with db_manager.get_session() as session:
229
+ docs = doc_manager.get_documents_by_collection(
230
+ session, MyDocument, "textbooks"
231
+ )
232
+
233
+ # Get all chunks from a corpus
234
+ corpus_docs = doc_manager.get_documents_by_corpus_id(
235
+ session, MyDocument, corpus_id
236
+ )
237
+ ```
238
+
239
+ ## Concept Reference
240
+
241
+ ### Core Classes
242
+
243
+ - **`BaseDocument`**: Abstract SQLAlchemy model for documents with vector embeddings
244
+ - **`BaseDocumentOptionalProps`**: Pydantic model for optional document properties
245
+ - **`BaseDocumentMetadata`**: Base schema for structured document metadata
246
+ - **`DatabaseManager`**: Database connection and session management
247
+ - **`DocumentDatabaseManager`**: High-level document CRUD operations
248
+
249
+ ## Testing
250
+
251
+ Install dependencies (preferably in a virtualenv) before running tests:
252
+ ```bash
253
+ pip install -e .[test]
254
+ ```
255
+
256
+ ### Unit Tests
257
+ ```bash
258
+ python -m unittest
259
+ ```
260
+
261
+ ### Integration Tests
262
+
263
+ Integration tests require a PostgreSQL database with PGVector extension. Set up your test database and configure the connection in `integ-tests/.env`:
264
+
265
+ ```bash
266
+ python -m unittest discover -s integ-tests
267
+ ```
268
+
269
+ ## Contributing
270
+
271
+ 1. Fork the repository
272
+ 2. Create a feature branch
273
+ 3. Make your changes with tests
274
+ 4. Run the test suite
275
+ 5. Submit a pull request
276
+
277
+ ### Development Setup
278
+
279
+ ```bash
280
+ pip install -e .[dev,test]
281
+ black . # Format code
282
+ ```
283
+
284
+ ## License
285
+
286
+ MIT License - see [LICENSE](LICENSE) file for details.
287
+
288
+ ## Links
289
+
290
+ - [GitHub Repository](https://github.com/DavidLiuGit/PGVector-Template)
291
+ - [PGVector Documentation](https://github.com/pgvector/pgvector)
@@ -165,32 +165,58 @@ class BaseCorpusManager(ABC):
165
165
  Raises:
166
166
  ValueError: If the length of document_contents doesn't match document_embeddings
167
167
  """
168
- if len(document_contents) != len(document_embeddings):
169
- raise ValueError("Number of embeddings does not match number of documents")
168
+ self._validate_inputs(document_contents, document_embeddings)
170
169
  if len(document_contents) == 0:
171
170
  return 0
172
- documents_to_insert = []
173
- for i in range(len(document_contents)):
174
- chunk_md = self._extract_chunk_metadata(document_contents[i])
171
+
172
+ documents_to_insert = self._create_documents(
173
+ corpus_id, document_contents, document_embeddings, corpus_metadata, optional_props
174
+ )
175
+
176
+ if update_if_exists:
177
+ self._delete_existing_corpus(corpus_id)
178
+
179
+ self.session.add_all(documents_to_insert)
180
+ self.session.commit()
181
+ return len(documents_to_insert)
182
+
183
+ def _validate_inputs(
184
+ self, document_contents: list[str], document_embeddings: list[list[float]]
185
+ ) -> None:
186
+ """Validate that document contents and embeddings match in length"""
187
+ if len(document_contents) != len(document_embeddings):
188
+ raise ValueError("Number of embeddings does not match number of documents")
189
+
190
+ def _create_documents(
191
+ self,
192
+ corpus_id: UUID | str,
193
+ document_contents: list[str],
194
+ document_embeddings: list[list[float]],
195
+ corpus_metadata: dict[str, Any],
196
+ optional_props: BaseDocumentOptionalProps | None,
197
+ ) -> list[BaseDocument]:
198
+ """Create document instances from the provided data"""
199
+ documents = []
200
+ for i, (content, embedding) in enumerate(zip(document_contents, document_embeddings)):
201
+ chunk_md = self._extract_chunk_metadata(content)
175
202
  base_metadata = self.document_metadata_class(**(corpus_metadata | chunk_md))
176
- documents_to_insert.append(
203
+ documents.append(
177
204
  self.config.document_cls.from_props(
178
205
  corpus_id=corpus_id,
179
206
  chunk_index=i,
180
- content=document_contents[i],
181
- embedding=document_embeddings[i],
207
+ content=content,
208
+ embedding=embedding,
182
209
  metadata=base_metadata.model_dump(),
183
210
  optional_props=optional_props,
184
211
  )
185
212
  )
213
+ return documents
186
214
 
187
- if update_if_exists:
188
- for doc in documents_to_insert:
189
- self.session.merge(doc)
190
- else:
191
- self.session.add_all(documents_to_insert)
192
- self.session.commit()
193
- return len(documents_to_insert)
215
+ def _delete_existing_corpus(self, corpus_id: UUID | str) -> None:
216
+ """Delete all existing documents for the given corpus_id"""
217
+ self.session.query(self.config.document_cls).filter(
218
+ self.config.document_cls.corpus_id == corpus_id
219
+ ).delete()
194
220
 
195
221
  def _split_corpus(self, content: str, **kwargs) -> list[str]:
196
222
  """