pgvector-template 0.5__tar.gz → 0.5.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pgvector_template-0.5 → pgvector_template-0.5.1}/PKG-INFO +14 -16
- {pgvector_template-0.5 → pgvector_template-0.5.1}/README.md +13 -15
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/core/__init__.py +6 -11
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/core/document.py +15 -17
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/core/manager.py +5 -6
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/core/search.py +9 -9
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/db/connection.py +3 -3
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/db/document_db.py +1 -3
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/models/__init__.py +4 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/models/search.py +1 -1
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/service/document_service.py +5 -5
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/utils/metadata_filter.py +4 -6
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template.egg-info/PKG-INFO +14 -16
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pyproject.toml +1 -1
- {pgvector_template-0.5 → pgvector_template-0.5.1}/LICENSE +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/__init__.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/core/embedder.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/db/__init__.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/service/__init__.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/types.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/utils/__init__.py +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template.egg-info/SOURCES.txt +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template.egg-info/dependency_links.txt +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template.egg-info/requires.txt +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template.egg-info/top_level.txt +0 -0
- {pgvector_template-0.5 → pgvector_template-0.5.1}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pgvector-template
|
|
3
|
-
Version: 0.5
|
|
3
|
+
Version: 0.5.1
|
|
4
4
|
Summary: Template library for flexible PGVector RAG implementations
|
|
5
5
|
Author-email: DL <v49t9zpqd@mozmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -44,10 +44,12 @@ from pgvector_template.core.document import BaseDocument
|
|
|
44
44
|
from sqlalchemy import create_engine
|
|
45
45
|
from uuid import uuid4
|
|
46
46
|
|
|
47
|
+
|
|
47
48
|
# 1. Define your document model
|
|
48
49
|
class MyDocument(BaseDocument):
|
|
49
50
|
__tablename__ = "my_documents"
|
|
50
51
|
|
|
52
|
+
|
|
51
53
|
# 2. Set up database connection
|
|
52
54
|
engine = create_engine("postgresql://user:pass@localhost/mydb")
|
|
53
55
|
db_manager = DatabaseManager(engine)
|
|
@@ -151,6 +153,7 @@ db_manager = DatabaseManager(engine)
|
|
|
151
153
|
|
|
152
154
|
# Option 2: From environment variable
|
|
153
155
|
import os
|
|
156
|
+
|
|
154
157
|
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
155
158
|
db_manager = DatabaseManager(engine)
|
|
156
159
|
```
|
|
@@ -185,10 +188,12 @@ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
|
|
|
185
188
|
from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
|
|
186
189
|
from pydantic import Field
|
|
187
190
|
|
|
191
|
+
|
|
188
192
|
class MyDocumentMetadata(BaseDocumentMetadata):
|
|
189
193
|
source_type: str = Field(..., description="Type of source document")
|
|
190
194
|
author: str = Field(default="unknown", description="Document author")
|
|
191
195
|
|
|
196
|
+
|
|
192
197
|
class MyDocument(BaseDocument):
|
|
193
198
|
__tablename__ = "my_documents"
|
|
194
199
|
```
|
|
@@ -205,7 +210,7 @@ optional_props = BaseDocumentOptionalProps(
|
|
|
205
210
|
title="Chapter 1: Introduction",
|
|
206
211
|
collection="textbooks",
|
|
207
212
|
language="en",
|
|
208
|
-
tags=["education", "intro"]
|
|
213
|
+
tags=["education", "intro"],
|
|
209
214
|
)
|
|
210
215
|
|
|
211
216
|
doc = MyDocument.from_props(
|
|
@@ -214,7 +219,7 @@ doc = MyDocument.from_props(
|
|
|
214
219
|
content="This is the document content...",
|
|
215
220
|
embedding=your_embedding_vector, # list[float] with 1024 dimensions
|
|
216
221
|
metadata=metadata.to_dict(),
|
|
217
|
-
optional_props=optional_props
|
|
222
|
+
optional_props=optional_props,
|
|
218
223
|
)
|
|
219
224
|
|
|
220
225
|
with db_manager.get_session() as session:
|
|
@@ -227,10 +232,7 @@ with db_manager.get_session() as session:
|
|
|
227
232
|
# Basic similarity search
|
|
228
233
|
with db_manager.get_session() as session:
|
|
229
234
|
results = doc_manager.search_similar(
|
|
230
|
-
session=session,
|
|
231
|
-
document_cls=MyDocument,
|
|
232
|
-
query_embedding=query_vector,
|
|
233
|
-
limit=10
|
|
235
|
+
session=session, document_cls=MyDocument, query_embedding=query_vector, limit=10
|
|
234
236
|
)
|
|
235
237
|
|
|
236
238
|
# Search with filters
|
|
@@ -238,7 +240,7 @@ from pgvector_template.models.search import MetadataFilter
|
|
|
238
240
|
|
|
239
241
|
filters = [
|
|
240
242
|
MetadataFilter(key="source_type", value="pdf"),
|
|
241
|
-
MetadataFilter(key="author", value="John Doe")
|
|
243
|
+
MetadataFilter(key="author", value="John Doe"),
|
|
242
244
|
]
|
|
243
245
|
|
|
244
246
|
results = doc_manager.search_similar(
|
|
@@ -247,7 +249,7 @@ results = doc_manager.search_similar(
|
|
|
247
249
|
query_embedding=query_vector,
|
|
248
250
|
limit=10,
|
|
249
251
|
metadata_filters=filters,
|
|
250
|
-
collection="textbooks"
|
|
252
|
+
collection="textbooks",
|
|
251
253
|
)
|
|
252
254
|
```
|
|
253
255
|
|
|
@@ -256,14 +258,10 @@ results = doc_manager.search_similar(
|
|
|
256
258
|
```python
|
|
257
259
|
# Get all documents in a collection
|
|
258
260
|
with db_manager.get_session() as session:
|
|
259
|
-
docs = doc_manager.get_documents_by_collection(
|
|
260
|
-
session, MyDocument, "textbooks"
|
|
261
|
-
)
|
|
261
|
+
docs = doc_manager.get_documents_by_collection(session, MyDocument, "textbooks")
|
|
262
262
|
|
|
263
|
-
# Get all chunks from a corpus
|
|
264
|
-
corpus_docs = doc_manager.get_documents_by_corpus_id(
|
|
265
|
-
session, MyDocument, corpus_id
|
|
266
|
-
)
|
|
263
|
+
# Get all chunks from a corpus
|
|
264
|
+
corpus_docs = doc_manager.get_documents_by_corpus_id(session, MyDocument, corpus_id)
|
|
267
265
|
```
|
|
268
266
|
|
|
269
267
|
## Concept Reference
|
|
@@ -14,10 +14,12 @@ from pgvector_template.core.document import BaseDocument
|
|
|
14
14
|
from sqlalchemy import create_engine
|
|
15
15
|
from uuid import uuid4
|
|
16
16
|
|
|
17
|
+
|
|
17
18
|
# 1. Define your document model
|
|
18
19
|
class MyDocument(BaseDocument):
|
|
19
20
|
__tablename__ = "my_documents"
|
|
20
21
|
|
|
22
|
+
|
|
21
23
|
# 2. Set up database connection
|
|
22
24
|
engine = create_engine("postgresql://user:pass@localhost/mydb")
|
|
23
25
|
db_manager = DatabaseManager(engine)
|
|
@@ -121,6 +123,7 @@ db_manager = DatabaseManager(engine)
|
|
|
121
123
|
|
|
122
124
|
# Option 2: From environment variable
|
|
123
125
|
import os
|
|
126
|
+
|
|
124
127
|
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
125
128
|
db_manager = DatabaseManager(engine)
|
|
126
129
|
```
|
|
@@ -155,10 +158,12 @@ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
|
|
|
155
158
|
from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
|
|
156
159
|
from pydantic import Field
|
|
157
160
|
|
|
161
|
+
|
|
158
162
|
class MyDocumentMetadata(BaseDocumentMetadata):
|
|
159
163
|
source_type: str = Field(..., description="Type of source document")
|
|
160
164
|
author: str = Field(default="unknown", description="Document author")
|
|
161
165
|
|
|
166
|
+
|
|
162
167
|
class MyDocument(BaseDocument):
|
|
163
168
|
__tablename__ = "my_documents"
|
|
164
169
|
```
|
|
@@ -175,7 +180,7 @@ optional_props = BaseDocumentOptionalProps(
|
|
|
175
180
|
title="Chapter 1: Introduction",
|
|
176
181
|
collection="textbooks",
|
|
177
182
|
language="en",
|
|
178
|
-
tags=["education", "intro"]
|
|
183
|
+
tags=["education", "intro"],
|
|
179
184
|
)
|
|
180
185
|
|
|
181
186
|
doc = MyDocument.from_props(
|
|
@@ -184,7 +189,7 @@ doc = MyDocument.from_props(
|
|
|
184
189
|
content="This is the document content...",
|
|
185
190
|
embedding=your_embedding_vector, # list[float] with 1024 dimensions
|
|
186
191
|
metadata=metadata.to_dict(),
|
|
187
|
-
optional_props=optional_props
|
|
192
|
+
optional_props=optional_props,
|
|
188
193
|
)
|
|
189
194
|
|
|
190
195
|
with db_manager.get_session() as session:
|
|
@@ -197,10 +202,7 @@ with db_manager.get_session() as session:
|
|
|
197
202
|
# Basic similarity search
|
|
198
203
|
with db_manager.get_session() as session:
|
|
199
204
|
results = doc_manager.search_similar(
|
|
200
|
-
session=session,
|
|
201
|
-
document_cls=MyDocument,
|
|
202
|
-
query_embedding=query_vector,
|
|
203
|
-
limit=10
|
|
205
|
+
session=session, document_cls=MyDocument, query_embedding=query_vector, limit=10
|
|
204
206
|
)
|
|
205
207
|
|
|
206
208
|
# Search with filters
|
|
@@ -208,7 +210,7 @@ from pgvector_template.models.search import MetadataFilter
|
|
|
208
210
|
|
|
209
211
|
filters = [
|
|
210
212
|
MetadataFilter(key="source_type", value="pdf"),
|
|
211
|
-
MetadataFilter(key="author", value="John Doe")
|
|
213
|
+
MetadataFilter(key="author", value="John Doe"),
|
|
212
214
|
]
|
|
213
215
|
|
|
214
216
|
results = doc_manager.search_similar(
|
|
@@ -217,7 +219,7 @@ results = doc_manager.search_similar(
|
|
|
217
219
|
query_embedding=query_vector,
|
|
218
220
|
limit=10,
|
|
219
221
|
metadata_filters=filters,
|
|
220
|
-
collection="textbooks"
|
|
222
|
+
collection="textbooks",
|
|
221
223
|
)
|
|
222
224
|
```
|
|
223
225
|
|
|
@@ -226,14 +228,10 @@ results = doc_manager.search_similar(
|
|
|
226
228
|
```python
|
|
227
229
|
# Get all documents in a collection
|
|
228
230
|
with db_manager.get_session() as session:
|
|
229
|
-
docs = doc_manager.get_documents_by_collection(
|
|
230
|
-
session, MyDocument, "textbooks"
|
|
231
|
-
)
|
|
231
|
+
docs = doc_manager.get_documents_by_collection(session, MyDocument, "textbooks")
|
|
232
232
|
|
|
233
|
-
# Get all chunks from a corpus
|
|
234
|
-
corpus_docs = doc_manager.get_documents_by_corpus_id(
|
|
235
|
-
session, MyDocument, corpus_id
|
|
236
|
-
)
|
|
233
|
+
# Get all chunks from a corpus
|
|
234
|
+
corpus_docs = doc_manager.get_documents_by_corpus_id(session, MyDocument, corpus_id)
|
|
237
235
|
```
|
|
238
236
|
|
|
239
237
|
## Concept Reference
|
|
@@ -5,21 +5,16 @@ from pgvector_template.core.document import (
|
|
|
5
5
|
)
|
|
6
6
|
from pgvector_template.core.embedder import BaseEmbeddingProvider
|
|
7
7
|
from pgvector_template.core.manager import BaseCorpusManager, BaseCorpusManagerConfig, Corpus
|
|
8
|
-
from pgvector_template.core.search import
|
|
9
|
-
|
|
8
|
+
from pgvector_template.core.search import BaseSearchClient, BaseSearchClientConfig
|
|
10
9
|
|
|
11
10
|
__all__ = [
|
|
12
|
-
|
|
13
|
-
"
|
|
11
|
+
"BaseCorpusManager",
|
|
12
|
+
"BaseCorpusManagerConfig",
|
|
14
13
|
"BaseDocument",
|
|
15
14
|
"BaseDocumentMetadata",
|
|
16
|
-
"
|
|
17
|
-
### embedder
|
|
15
|
+
"BaseDocumentOptionalProps",
|
|
18
16
|
"BaseEmbeddingProvider",
|
|
19
|
-
### manager
|
|
20
|
-
"BaseCorpusManager",
|
|
21
|
-
"BaseCorpusManagerConfig",
|
|
22
|
-
### search
|
|
23
|
-
"BaseSearchClientConfig",
|
|
24
17
|
"BaseSearchClient",
|
|
18
|
+
"BaseSearchClientConfig",
|
|
19
|
+
"Corpus",
|
|
25
20
|
]
|
|
@@ -1,21 +1,22 @@
|
|
|
1
|
-
from datetime import
|
|
2
|
-
from typing import Any,
|
|
3
|
-
from uuid import
|
|
1
|
+
from datetime import UTC, datetime
|
|
2
|
+
from typing import Any, Self
|
|
3
|
+
from uuid import UUID as UuidLiteral
|
|
4
|
+
from uuid import uuid4
|
|
4
5
|
|
|
6
|
+
from pgvector.sqlalchemy import Vector
|
|
5
7
|
from pydantic import BaseModel, Field
|
|
6
8
|
from sqlalchemy import (
|
|
9
|
+
Boolean,
|
|
7
10
|
Column,
|
|
8
|
-
String,
|
|
9
|
-
Text,
|
|
10
11
|
DateTime,
|
|
11
|
-
Boolean,
|
|
12
|
-
Integer,
|
|
13
12
|
Index,
|
|
13
|
+
Integer,
|
|
14
|
+
String,
|
|
15
|
+
Text,
|
|
14
16
|
UniqueConstraint,
|
|
15
17
|
)
|
|
18
|
+
from sqlalchemy.dialects.postgresql import JSONB, UUID
|
|
16
19
|
from sqlalchemy.orm import declarative_base
|
|
17
|
-
from sqlalchemy.dialects.postgresql import UUID, JSONB
|
|
18
|
-
from pgvector.sqlalchemy import Vector
|
|
19
20
|
|
|
20
21
|
Base = declarative_base()
|
|
21
22
|
|
|
@@ -33,9 +34,6 @@ class BaseDocumentOptionalProps(BaseModel):
|
|
|
33
34
|
"""Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'"""
|
|
34
35
|
|
|
35
36
|
|
|
36
|
-
T = TypeVar("T", bound="BaseDocument")
|
|
37
|
-
|
|
38
|
-
|
|
39
37
|
class BaseDocument(Base):
|
|
40
38
|
"""
|
|
41
39
|
Template table for Documents, that works for all collection types.
|
|
@@ -89,11 +87,11 @@ class BaseDocument(Base):
|
|
|
89
87
|
"""
|
|
90
88
|
|
|
91
89
|
# Audit fields
|
|
92
|
-
created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(
|
|
90
|
+
created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(UTC))
|
|
93
91
|
updated_at = Column(
|
|
94
92
|
DateTime(timezone=True),
|
|
95
|
-
default=lambda: datetime.now(
|
|
96
|
-
onupdate=lambda: datetime.now(
|
|
93
|
+
default=lambda: datetime.now(UTC),
|
|
94
|
+
onupdate=lambda: datetime.now(UTC),
|
|
97
95
|
)
|
|
98
96
|
is_deleted = Column(Boolean, default=False, index=True)
|
|
99
97
|
"""Entries can be logically marked for deletion before they are permanently deleted."""
|
|
@@ -131,7 +129,7 @@ class BaseDocument(Base):
|
|
|
131
129
|
|
|
132
130
|
@classmethod
|
|
133
131
|
def from_props(
|
|
134
|
-
cls
|
|
132
|
+
cls,
|
|
135
133
|
corpus_id: UuidLiteral | str,
|
|
136
134
|
chunk_index: int,
|
|
137
135
|
content: str,
|
|
@@ -139,7 +137,7 @@ class BaseDocument(Base):
|
|
|
139
137
|
embedding_config: dict[str, Any] | None = None,
|
|
140
138
|
metadata: dict[str, Any] | None = None,
|
|
141
139
|
optional_props: BaseDocumentOptionalProps | None = None,
|
|
142
|
-
) ->
|
|
140
|
+
) -> Self:
|
|
143
141
|
"""
|
|
144
142
|
Create a BaseDocument instance from mandatory and optional properties.
|
|
145
143
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
from abc import ABC
|
|
2
2
|
from dataclasses import dataclass
|
|
3
3
|
from logging import getLogger
|
|
4
|
-
from typing import Any
|
|
4
|
+
from typing import Any
|
|
5
5
|
from uuid import UUID, uuid4
|
|
6
6
|
|
|
7
7
|
from pydantic import BaseModel, Field
|
|
@@ -14,18 +14,17 @@ from pgvector_template.core.document import (
|
|
|
14
14
|
)
|
|
15
15
|
from pgvector_template.core.embedder import BaseEmbeddingProvider
|
|
16
16
|
|
|
17
|
-
|
|
18
17
|
logger = getLogger(__name__)
|
|
19
18
|
|
|
20
19
|
|
|
21
20
|
class BaseCorpusManagerConfig(BaseModel):
|
|
22
21
|
"""Base configuration for `Corpus` & `Document` management operations"""
|
|
23
22
|
|
|
24
|
-
document_cls:
|
|
23
|
+
document_cls: type[BaseDocument] = Field(...)
|
|
25
24
|
"""Document class **type** (not an instance). Must be subclass of `BaseDocument`."""
|
|
26
25
|
embedding_provider: BaseEmbeddingProvider | None = Field(default=None)
|
|
27
26
|
"""Instance of `BaseEmbeddingProvider` child class. Acts as embedding provider for insert operations."""
|
|
28
|
-
document_metadata_cls:
|
|
27
|
+
document_metadata_cls: type[BaseDocumentMetadata] = Field(default=BaseDocumentMetadata)
|
|
29
28
|
"""Document metadata class **type** (not an instance). Must be subclass of BaseDocumentMetadata."""
|
|
30
29
|
|
|
31
30
|
model_config = {"arbitrary_types_allowed": True}
|
|
@@ -59,7 +58,7 @@ class BaseCorpusManager(ABC):
|
|
|
59
58
|
return self._cfg
|
|
60
59
|
|
|
61
60
|
@property
|
|
62
|
-
def document_metadata_class(self) ->
|
|
61
|
+
def document_metadata_class(self) -> type[BaseDocumentMetadata]:
|
|
63
62
|
"""Returns the document metadata class, raising an error if it's not set."""
|
|
64
63
|
return self.config.document_metadata_cls
|
|
65
64
|
|
|
@@ -88,7 +87,7 @@ class BaseCorpusManager(ABC):
|
|
|
88
87
|
self.session.query(self.config.document_cls)
|
|
89
88
|
.filter(
|
|
90
89
|
self.config.document_cls.corpus_id == corpus_id,
|
|
91
|
-
self.config.document_cls.is_deleted == False,
|
|
90
|
+
self.config.document_cls.is_deleted == False,
|
|
92
91
|
)
|
|
93
92
|
.order_by(self.config.document_cls.chunk_index)
|
|
94
93
|
.all()
|
|
@@ -1,31 +1,31 @@
|
|
|
1
|
+
from collections.abc import Sequence
|
|
1
2
|
from logging import getLogger
|
|
2
|
-
from typing import Any
|
|
3
|
+
from typing import Any
|
|
3
4
|
|
|
4
5
|
from pydantic import BaseModel, Field
|
|
5
|
-
from sqlalchemy import
|
|
6
|
+
from sqlalchemy import Float, Integer, or_, select
|
|
6
7
|
from sqlalchemy.orm import Session
|
|
7
|
-
from sqlalchemy.sql import
|
|
8
|
+
from sqlalchemy.sql import ColumnElement, Select
|
|
8
9
|
|
|
9
10
|
from pgvector_template.core import (
|
|
10
|
-
BaseEmbeddingProvider,
|
|
11
11
|
BaseDocument,
|
|
12
12
|
BaseDocumentMetadata,
|
|
13
|
+
BaseEmbeddingProvider,
|
|
13
14
|
)
|
|
14
|
-
from pgvector_template.models.search import
|
|
15
|
+
from pgvector_template.models.search import MetadataFilter, RetrievalResult, SearchQuery
|
|
15
16
|
from pgvector_template.utils.metadata_filter import validate_metadata_filters
|
|
16
17
|
|
|
17
|
-
|
|
18
18
|
logger = getLogger(__name__)
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
class BaseSearchClientConfig(BaseModel):
|
|
22
22
|
"""Config obj for `BaseSearchClient`."""
|
|
23
23
|
|
|
24
|
-
document_cls:
|
|
24
|
+
document_cls: type[BaseDocument] = Field(default=BaseDocument)
|
|
25
25
|
"""Document class **type** (not an instance). Must be subclass of `BaseDocument`."""
|
|
26
26
|
embedding_provider: BaseEmbeddingProvider | None = Field(default=None)
|
|
27
27
|
"""Instance of `BaseEmbeddingProvider` child class. Acts as embedding provider for semantic search."""
|
|
28
|
-
document_metadata_cls:
|
|
28
|
+
document_metadata_cls: type[BaseDocumentMetadata] = Field(default=BaseDocumentMetadata)
|
|
29
29
|
"""Document metadata class type. Used for metadata search operations."""
|
|
30
30
|
|
|
31
31
|
model_config = {"arbitrary_types_allowed": True}
|
|
@@ -39,7 +39,7 @@ class BaseSearchClient:
|
|
|
39
39
|
return self._cfg
|
|
40
40
|
|
|
41
41
|
@property
|
|
42
|
-
def document_metadata_class(self) ->
|
|
42
|
+
def document_metadata_class(self) -> type[BaseDocumentMetadata]:
|
|
43
43
|
"""Returns the document metadata class, raising an error if it's not set."""
|
|
44
44
|
return self.config.document_metadata_cls
|
|
45
45
|
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
+
from collections.abc import Generator
|
|
1
2
|
from contextlib import contextmanager
|
|
2
3
|
from logging import getLogger
|
|
3
|
-
from typing import Generator, Type
|
|
4
4
|
|
|
5
5
|
from sqlalchemy import Engine, create_engine, text
|
|
6
|
-
from sqlalchemy.orm import sessionmaker, Session
|
|
7
6
|
from sqlalchemy.ext.declarative import DeclarativeMeta
|
|
7
|
+
from sqlalchemy.orm import Session, sessionmaker
|
|
8
8
|
|
|
9
9
|
|
|
10
10
|
class DatabaseManager:
|
|
@@ -40,7 +40,7 @@ class DatabaseManager:
|
|
|
40
40
|
|
|
41
41
|
self.logger.info(f"Created schema: {schema_name}")
|
|
42
42
|
|
|
43
|
-
def create_tables(self, base_class:
|
|
43
|
+
def create_tables(self, base_class: type[DeclarativeMeta], schema_name: str) -> None:
|
|
44
44
|
"""Create tables for a specific schema"""
|
|
45
45
|
# Ensure all tables in this base class use the specified schema
|
|
46
46
|
for table in base_class.metadata.tables.values():
|
|
@@ -1,5 +1,3 @@
|
|
|
1
|
-
from typing import List, Type
|
|
2
|
-
|
|
3
1
|
from sqlalchemy import text
|
|
4
2
|
|
|
5
3
|
from pgvector_template.core.document import BaseDocument
|
|
@@ -24,7 +22,7 @@ class DocumentDatabaseManager(DatabaseManager):
|
|
|
24
22
|
SCHEMA_PREFIX = "knowledge_base_"
|
|
25
23
|
|
|
26
24
|
def __init__(
|
|
27
|
-
self, database_url: str, schema_suffix: str, document_classes:
|
|
25
|
+
self, database_url: str, schema_suffix: str, document_classes: list[type[BaseDocument]]
|
|
28
26
|
):
|
|
29
27
|
"""
|
|
30
28
|
Initialize a document-oriented database manager
|
{pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/service/document_service.py
RENAMED
|
@@ -3,7 +3,7 @@ Document service layer combining corpus management and search capabilities.
|
|
|
3
3
|
"""
|
|
4
4
|
|
|
5
5
|
from logging import getLogger
|
|
6
|
-
from typing import Generic,
|
|
6
|
+
from typing import Generic, TypeVar
|
|
7
7
|
|
|
8
8
|
from pydantic import BaseModel, Field
|
|
9
9
|
from sqlalchemy.orm import Session
|
|
@@ -23,17 +23,17 @@ class DocumentServiceConfig(BaseModel):
|
|
|
23
23
|
"""Configuration for DocumentService"""
|
|
24
24
|
|
|
25
25
|
# Required fields
|
|
26
|
-
document_cls:
|
|
26
|
+
document_cls: type[BaseDocument] = Field(...)
|
|
27
27
|
"""Document class **type** (not an instance). Must be subclass of `BaseDocument`."""
|
|
28
|
-
corpus_manager_cls:
|
|
28
|
+
corpus_manager_cls: type[BaseCorpusManager] = Field(default=BaseCorpusManager)
|
|
29
29
|
"""CorpusManager class **type** (not an instance). Must be child of `BaseCorpusManager`."""
|
|
30
|
-
search_client_cls:
|
|
30
|
+
search_client_cls: type[BaseSearchClient] = Field(default=BaseSearchClient)
|
|
31
31
|
"""SearchClient class **type** (not an instance). Must be child of `BaseSearchClient`."""
|
|
32
32
|
|
|
33
33
|
# Optional fields with defaults
|
|
34
34
|
embedding_provider: BaseEmbeddingProvider | None = Field(default=None)
|
|
35
35
|
"""Embedding provider for insert & vector-search operations."""
|
|
36
|
-
document_metadata_cls:
|
|
36
|
+
document_metadata_cls: type[BaseDocumentMetadata] = Field(default=BaseDocumentMetadata)
|
|
37
37
|
"""Document metadata schema. Must be child of `BaseDocumentMetadata`."""
|
|
38
38
|
corpus_manager_cfg: BaseCorpusManagerConfig = Field(
|
|
39
39
|
default=BaseCorpusManagerConfig(document_cls=BaseDocument)
|
{pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template/utils/metadata_filter.py
RENAMED
|
@@ -1,11 +1,9 @@
|
|
|
1
|
-
from typing import Type
|
|
2
|
-
|
|
3
1
|
from pgvector_template.core.document import BaseDocumentMetadata
|
|
4
2
|
from pgvector_template.models.search import MetadataFilter
|
|
5
3
|
|
|
6
4
|
|
|
7
5
|
def validate_metadata_filters(
|
|
8
|
-
filter_obj_list: list[MetadataFilter], metadata_cls:
|
|
6
|
+
filter_obj_list: list[MetadataFilter], metadata_cls: type[BaseDocumentMetadata]
|
|
9
7
|
) -> None:
|
|
10
8
|
"""Validate a `list[MetadataFilter]` against schema and condition compatibility.
|
|
11
9
|
|
|
@@ -18,7 +16,7 @@ def validate_metadata_filters(
|
|
|
18
16
|
|
|
19
17
|
|
|
20
18
|
def validate_metadata_filter(
|
|
21
|
-
filter_obj: MetadataFilter, metadata_cls:
|
|
19
|
+
filter_obj: MetadataFilter, metadata_cls: type[BaseDocumentMetadata]
|
|
22
20
|
) -> None:
|
|
23
21
|
"""Validate metadata filter against schema and condition compatibility.
|
|
24
22
|
|
|
@@ -57,7 +55,7 @@ def validate_metadata_filter(
|
|
|
57
55
|
validate_condition_compatibility(field_type, filter_obj.condition)
|
|
58
56
|
|
|
59
57
|
|
|
60
|
-
def validate_condition_compatibility(field_type:
|
|
58
|
+
def validate_condition_compatibility(field_type: type, condition: str) -> None:
|
|
61
59
|
"""Validate that condition is compatible with field type."""
|
|
62
60
|
# Extract base type from Optional/Union types
|
|
63
61
|
origin = getattr(field_type, "__origin__", None)
|
|
@@ -66,7 +64,7 @@ def validate_condition_compatibility(field_type: Type, condition: str) -> None:
|
|
|
66
64
|
if origin is list:
|
|
67
65
|
field_type = list
|
|
68
66
|
elif len(args) > 0:
|
|
69
|
-
field_type = args[0]
|
|
67
|
+
field_type = args[0]
|
|
70
68
|
|
|
71
69
|
valid_conditions: dict[type, set[str]] = {
|
|
72
70
|
str: {"eq", "gt", "gte", "lt", "lte", "in", "exists"},
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pgvector-template
|
|
3
|
-
Version: 0.5
|
|
3
|
+
Version: 0.5.1
|
|
4
4
|
Summary: Template library for flexible PGVector RAG implementations
|
|
5
5
|
Author-email: DL <v49t9zpqd@mozmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -44,10 +44,12 @@ from pgvector_template.core.document import BaseDocument
|
|
|
44
44
|
from sqlalchemy import create_engine
|
|
45
45
|
from uuid import uuid4
|
|
46
46
|
|
|
47
|
+
|
|
47
48
|
# 1. Define your document model
|
|
48
49
|
class MyDocument(BaseDocument):
|
|
49
50
|
__tablename__ = "my_documents"
|
|
50
51
|
|
|
52
|
+
|
|
51
53
|
# 2. Set up database connection
|
|
52
54
|
engine = create_engine("postgresql://user:pass@localhost/mydb")
|
|
53
55
|
db_manager = DatabaseManager(engine)
|
|
@@ -151,6 +153,7 @@ db_manager = DatabaseManager(engine)
|
|
|
151
153
|
|
|
152
154
|
# Option 2: From environment variable
|
|
153
155
|
import os
|
|
156
|
+
|
|
154
157
|
engine = create_engine(os.getenv("DATABASE_URL"))
|
|
155
158
|
db_manager = DatabaseManager(engine)
|
|
156
159
|
```
|
|
@@ -185,10 +188,12 @@ DATABASE_URL=postgresql://user:password@localhost:5432/dev_db
|
|
|
185
188
|
from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata
|
|
186
189
|
from pydantic import Field
|
|
187
190
|
|
|
191
|
+
|
|
188
192
|
class MyDocumentMetadata(BaseDocumentMetadata):
|
|
189
193
|
source_type: str = Field(..., description="Type of source document")
|
|
190
194
|
author: str = Field(default="unknown", description="Document author")
|
|
191
195
|
|
|
196
|
+
|
|
192
197
|
class MyDocument(BaseDocument):
|
|
193
198
|
__tablename__ = "my_documents"
|
|
194
199
|
```
|
|
@@ -205,7 +210,7 @@ optional_props = BaseDocumentOptionalProps(
|
|
|
205
210
|
title="Chapter 1: Introduction",
|
|
206
211
|
collection="textbooks",
|
|
207
212
|
language="en",
|
|
208
|
-
tags=["education", "intro"]
|
|
213
|
+
tags=["education", "intro"],
|
|
209
214
|
)
|
|
210
215
|
|
|
211
216
|
doc = MyDocument.from_props(
|
|
@@ -214,7 +219,7 @@ doc = MyDocument.from_props(
|
|
|
214
219
|
content="This is the document content...",
|
|
215
220
|
embedding=your_embedding_vector, # list[float] with 1024 dimensions
|
|
216
221
|
metadata=metadata.to_dict(),
|
|
217
|
-
optional_props=optional_props
|
|
222
|
+
optional_props=optional_props,
|
|
218
223
|
)
|
|
219
224
|
|
|
220
225
|
with db_manager.get_session() as session:
|
|
@@ -227,10 +232,7 @@ with db_manager.get_session() as session:
|
|
|
227
232
|
# Basic similarity search
|
|
228
233
|
with db_manager.get_session() as session:
|
|
229
234
|
results = doc_manager.search_similar(
|
|
230
|
-
session=session,
|
|
231
|
-
document_cls=MyDocument,
|
|
232
|
-
query_embedding=query_vector,
|
|
233
|
-
limit=10
|
|
235
|
+
session=session, document_cls=MyDocument, query_embedding=query_vector, limit=10
|
|
234
236
|
)
|
|
235
237
|
|
|
236
238
|
# Search with filters
|
|
@@ -238,7 +240,7 @@ from pgvector_template.models.search import MetadataFilter
|
|
|
238
240
|
|
|
239
241
|
filters = [
|
|
240
242
|
MetadataFilter(key="source_type", value="pdf"),
|
|
241
|
-
MetadataFilter(key="author", value="John Doe")
|
|
243
|
+
MetadataFilter(key="author", value="John Doe"),
|
|
242
244
|
]
|
|
243
245
|
|
|
244
246
|
results = doc_manager.search_similar(
|
|
@@ -247,7 +249,7 @@ results = doc_manager.search_similar(
|
|
|
247
249
|
query_embedding=query_vector,
|
|
248
250
|
limit=10,
|
|
249
251
|
metadata_filters=filters,
|
|
250
|
-
collection="textbooks"
|
|
252
|
+
collection="textbooks",
|
|
251
253
|
)
|
|
252
254
|
```
|
|
253
255
|
|
|
@@ -256,14 +258,10 @@ results = doc_manager.search_similar(
|
|
|
256
258
|
```python
|
|
257
259
|
# Get all documents in a collection
|
|
258
260
|
with db_manager.get_session() as session:
|
|
259
|
-
docs = doc_manager.get_documents_by_collection(
|
|
260
|
-
session, MyDocument, "textbooks"
|
|
261
|
-
)
|
|
261
|
+
docs = doc_manager.get_documents_by_collection(session, MyDocument, "textbooks")
|
|
262
262
|
|
|
263
|
-
# Get all chunks from a corpus
|
|
264
|
-
corpus_docs = doc_manager.get_documents_by_corpus_id(
|
|
265
|
-
session, MyDocument, corpus_id
|
|
266
|
-
)
|
|
263
|
+
# Get all chunks from a corpus
|
|
264
|
+
corpus_docs = doc_manager.get_documents_by_corpus_id(session, MyDocument, corpus_id)
|
|
267
265
|
```
|
|
268
266
|
|
|
269
267
|
## Concept Reference
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pgvector-template"
|
|
7
|
-
version = "0.5"
|
|
7
|
+
version = "0.5.1"
|
|
8
8
|
description = "Template library for flexible PGVector RAG implementations"
|
|
9
9
|
authors = [{ name="DL", email="v49t9zpqd@mozmail.com" }]
|
|
10
10
|
license = { text = "MIT" }
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pgvector_template-0.5 → pgvector_template-0.5.1}/pgvector_template.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|