pgvector-template 0.1.0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/PKG-INFO +2 -2
  2. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/core/__init__.py +2 -2
  3. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/core/document.py +11 -21
  4. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/core/manager.py +87 -35
  5. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template.egg-info/PKG-INFO +2 -2
  6. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pyproject.toml +2 -2
  7. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/LICENSE +0 -0
  8. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/README.md +0 -0
  9. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/__init__.py +0 -0
  10. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/core/embedder.py +0 -0
  11. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/core/retriever.py +0 -0
  12. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/core/search.py +0 -0
  13. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/db/__init__.py +0 -0
  14. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/db/connection.py +0 -0
  15. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/db/document_db.py +0 -0
  16. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/search/__init__.py +0 -0
  17. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/types.py +0 -0
  18. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template/utils/__init__.py +0 -0
  19. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template.egg-info/SOURCES.txt +0 -0
  20. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template.egg-info/dependency_links.txt +0 -0
  21. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template.egg-info/requires.txt +1 -1
  22. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/pgvector_template.egg-info/top_level.txt +0 -0
  23. {pgvector_template-0.1.0 → pgvector_template-0.1.2}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: pgvector-template
3
- Version: 0.1.0
3
+ Version: 0.1.2
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -11,12 +11,12 @@ Classifier: Operating System :: OS Independent
11
11
  Requires-Python: >=3.11
12
12
  Description-Content-Type: text/markdown
13
13
  License-File: LICENSE
14
- Requires-Dist: psycopg[binary]>=3.1.0
15
14
  Requires-Dist: pgvector>=0.2.0
16
15
  Requires-Dist: pydantic<3.0,>=2.11
17
16
  Requires-Dist: sqlalchemy>=2.0.0
18
17
  Requires-Dist: typing-extensions>=4.0.0
19
18
  Provides-Extra: test
19
+ Requires-Dist: psycopg[binary]>=3.1.0; extra == "test"
20
20
  Requires-Dist: pytest>=7.0.0; extra == "test"
21
21
  Requires-Dist: pytest-cov>=4.0.0; extra == "test"
22
22
  Requires-Dist: python-dotenv>=1.0.0; extra == "test"
@@ -1,6 +1,6 @@
1
- from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata, Corpus, BaseDocumentOptionalProps
1
+ from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata, BaseDocumentOptionalProps
2
2
  from pgvector_template.core.embedder import BaseEmbeddingProvider
3
- from pgvector_template.core.manager import BaseCorpusManager, BaseCorpusManagerConfig
3
+ from pgvector_template.core.manager import BaseCorpusManager, BaseCorpusManagerConfig, Corpus
4
4
  from pgvector_template.core.retriever import RetrievalResult, SearchQuery
5
5
  from pgvector_template.core.search import BaseSearchClient
6
6
 
@@ -1,7 +1,6 @@
1
- from dataclasses import dataclass, field
2
1
  from datetime import datetime
3
- from typing import Any, ClassVar, Type, TypeVar, Annotated
4
- from uuid import uuid4, UUID
2
+ from typing import Any, Type, TypeVar
3
+ from uuid import uuid4, UUID as UuidLiteral
5
4
 
6
5
  from pydantic import BaseModel, Field, field_validator, model_validator
7
6
  from sqlalchemy import (
@@ -13,7 +12,6 @@ from sqlalchemy import (
13
12
  Integer,
14
13
  Float,
15
14
  Index,
16
- text,
17
15
  )
18
16
  from sqlalchemy.orm import declarative_base
19
17
  from sqlalchemy.dialects.postgresql import UUID, JSONB
@@ -106,7 +104,7 @@ class BaseDocument(Base):
106
104
  @classmethod
107
105
  def from_props(
108
106
  cls: Type[T],
109
- corpus_id: UUID,
107
+ corpus_id: UuidLiteral | str,
110
108
  chunk_index: int,
111
109
  content: str,
112
110
  embedding: list[float],
@@ -117,7 +115,7 @@ class BaseDocument(Base):
117
115
  Create a BaseDocument instance from mandatory and optional properties.
118
116
 
119
117
  Args:
120
- corpus_id: UUID of the corpus this document belongs to
118
+ corpus_id: UUID or string (max 64 chars) of the corpus this document belongs to
121
119
  chunk_index: Index of this chunk within the corpus
122
120
  content: Text content of the document
123
121
  embedding: Vector embedding of the content
@@ -127,8 +125,9 @@ class BaseDocument(Base):
127
125
  A new BaseDocument instance of the calling class type
128
126
  """
129
127
  if optional_props is None:
130
- optional_props = BaseDocumentOptionalProps()
128
+ optional_props = BaseDocumentOptionalProps() # type: ignore
131
129
 
130
+ # SQLAlchemy handles string-to-UUID conversion automatically with as_uuid=True
132
131
  return cls(
133
132
  corpus_id=corpus_id,
134
133
  chunk_index=chunk_index,
@@ -149,23 +148,14 @@ class BaseDocument(Base):
149
148
 
150
149
 
151
150
  class BaseDocumentMetadata(BaseModel):
152
- """Base metadata structure"""
151
+ """
152
+ Base metadata structure.
153
+ It is generally expected that every `BaseDocument`'s metadata follows this exact schema,
154
+ without any extraneous properties, or any missing properties, to avoid ambiguity.
155
+ """
153
156
 
154
157
  document_type: str = Field(..., description="Description for type of document, e.g. markdown, pdf, etc")
155
158
  schema_version: str = Field("1.0", description="Schema version for the metadata")
156
159
 
157
160
  def to_dict(self) -> dict[str, Any]:
158
161
  return self.model_dump()
159
-
160
-
161
- @dataclass
162
- class Corpus:
163
- """
164
- Logical grouping of one or more documents (chunks) belonging to the same original source.
165
-
166
- Typically all documents in a corpus share the same `corpus_id` and are ordered by `chunk_index`.
167
- """
168
-
169
- corpus_id: UUID
170
- documents: list[BaseDocument]
171
- metadata: dict[str, Any] # e.g. source, tags, etc.
@@ -1,9 +1,10 @@
1
1
  from abc import ABC, abstractmethod
2
+ from dataclasses import dataclass
2
3
  from logging import getLogger
3
- from typing import Any, Optional, Type
4
+ from typing import Any, Type
4
5
  from uuid import UUID, uuid4
5
6
 
6
- from pydantic import BaseModel
7
+ from pydantic import BaseModel, Field
7
8
  from sqlalchemy.orm import Session
8
9
 
9
10
  from pgvector_template.core.document import BaseDocument, BaseDocumentMetadata, BaseDocumentOptionalProps
@@ -16,14 +17,32 @@ logger = getLogger(__name__)
16
17
  class BaseCorpusManagerConfig(BaseModel):
17
18
  """Base configuration for `Corpus` & `Document` management operations"""
18
19
 
19
- schema_name: str
20
- document_cls: Type[BaseDocument]
21
- embedding_provider: BaseEmbeddingProvider
22
- document_metadata: Type[BaseDocumentMetadata]
20
+ document_cls: Type[BaseDocument] = Field(..., description="Document class, must be subclass of BaseDocument")
21
+ embedding_provider: BaseEmbeddingProvider | None = Field(
22
+ None, description="Embedding provider for insert operations"
23
+ )
24
+ document_metadata_cls: Type[BaseDocumentMetadata] | None = Field(
25
+ None, description="Document metadata class, must be subclass of BaseDocumentMetadata"
26
+ )
23
27
 
24
28
  model_config = {"arbitrary_types_allowed": True}
25
29
 
26
30
 
31
+ @dataclass
32
+ class Corpus:
33
+ """
34
+ Serves as the return type for `BaseCorpusManager.get_full_corpus()`.
35
+
36
+ Logical grouping of one or more documents (chunks) belonging to the same original source (corpus).
37
+ Typically all documents in a corpus share the same `corpus_id` and are ordered by `chunk_index`.
38
+ """
39
+
40
+ corpus_id: UUID | str
41
+ content: str
42
+ metadata: dict[str, Any] # e.g. source, tags, etc.
43
+ documents: list[BaseDocument]
44
+
45
+
27
46
  class BaseCorpusManager(ABC):
28
47
  """
29
48
  Template class for `Corpus` & `Document` management operations.
@@ -43,44 +62,33 @@ class BaseCorpusManager(ABC):
43
62
  ) -> None:
44
63
  self.session = session
45
64
  self._cfg = config
46
- self.schema_name = config.schema_name
47
65
 
48
- def get_full_corpus(self, corpus_id: str, chunk_delimiter: str = "\n") -> Optional[dict[str, Any]]:
66
+ def get_full_corpus(self, corpus_id: str) -> Corpus | None:
49
67
  """Reconstruct full corpus from its individual documents/chunks"""
50
68
  chunks = (
51
- self.session.query(BaseDocument)
52
- .filter(BaseDocument.corpus_id == corpus_id, BaseDocument.is_deleted == False)
53
- .order_by(BaseDocument.chunk_index)
69
+ self.session.query(self.config.document_cls)
70
+ .filter(self.config.document_cls.corpus_id == corpus_id, self.config.document_cls.is_deleted == False)
71
+ .order_by(self.config.document_cls.chunk_index)
54
72
  .all()
55
73
  )
56
74
 
57
75
  if not chunks:
58
76
  return None
59
77
 
60
- # Full document is chunk_index = 0, or reconstruct from chunks
61
- full_doc = next((c for c in chunks if c.chunk_index == 0), None)
62
- if full_doc:
63
- return {
64
- "id": full_doc.original_id,
65
- "content": full_doc.content,
66
- "metadata": full_doc.document_metadata,
67
- "chunks": [{"id": c.id, "index": c.chunk_index, "title": c.title} for c in chunks if c.chunk_index > 0],
68
- }
69
-
70
- # Reconstruct from chunks
71
- reconstructed_content = chunk_delimiter.join([c.content for c in chunks])
72
- return {
73
- "id": corpus_id,
74
- "content": reconstructed_content,
75
- "metadata": chunks[0].document_metadata, # Use first chunk's metadata
76
- "chunks": [{"id": c.id, "index": c.chunk_index, "title": c.title} for c in chunks],
77
- }
78
+ content, metadata = self._join_documents(chunks)
79
+ return Corpus(
80
+ corpus_id=corpus_id,
81
+ content=content,
82
+ metadata=metadata,
83
+ documents=chunks,
84
+ )
78
85
 
79
86
  def insert_corpus(
80
87
  self,
81
88
  content: str,
82
89
  corpus_metadata: dict[str, Any],
83
- optional_props: BaseDocumentOptionalProps,
90
+ optional_props: BaseDocumentOptionalProps | None = None,
91
+ corpus_id: UUID | str | None = None,
84
92
  ) -> int:
85
93
  """
86
94
  Insert a new `Corpus`, which will be split into 1-or-more `Document`s, depending on its length.
@@ -94,14 +102,17 @@ class BaseCorpusManager(ABC):
94
102
  Returns:
95
103
  int: The number of **documents** inserted for the provided corpus
96
104
  """
97
- corpus_id = uuid4()
105
+ self._check_insert_dependencies()
106
+
107
+ if not corpus_id:
108
+ corpus_id = uuid4()
98
109
  document_contents = self._split_corpus(content)
99
110
  document_embeddings = self.config.embedding_provider.embed_batch(document_contents)
100
111
  return self.insert_documents(corpus_id, document_contents, document_embeddings, corpus_metadata, optional_props)
101
112
 
102
113
  def insert_documents(
103
114
  self,
104
- corpus_id: UUID,
115
+ corpus_id: UUID | str,
105
116
  document_contents: list[str],
106
117
  document_embeddings: list[list[float]],
107
118
  corpus_metadata: dict[str, Any],
@@ -123,6 +134,8 @@ class BaseCorpusManager(ABC):
123
134
  Raises:
124
135
  ValueError: If the length of document_contents doesn't match document_embeddings
125
136
  """
137
+ self._check_insert_dependencies()
138
+
126
139
  if len(document_contents) != len(document_embeddings):
127
140
  raise ValueError("Number of embeddings does not match number of documents")
128
141
  if len(document_contents) == 0:
@@ -130,7 +143,7 @@ class BaseCorpusManager(ABC):
130
143
  documents_to_insert = []
131
144
  for i in range(len(document_contents)):
132
145
  chunk_md = self._extract_chunk_metadata(document_contents[i])
133
- base_metadata = self.config.document_metadata(**(corpus_metadata | chunk_md))
146
+ base_metadata = self.config.document_metadata_cls(**(corpus_metadata | chunk_md))
134
147
  documents_to_insert.append(
135
148
  self.config.document_cls.from_props(
136
149
  corpus_id=corpus_id,
@@ -148,20 +161,59 @@ class BaseCorpusManager(ABC):
148
161
  def _split_corpus(self, content: str, **kwargs) -> list[str]:
149
162
  """
150
163
  **It is highly recommended to override this method.**
151
- Split a corpus into chunks.
164
+ Split a corpus' string content into smaller chunks.
152
165
  """
153
166
  if self.__class__ is not BaseCorpusManager:
154
167
  logger.warning("Using default _split_corpus. Override this method to improve performance.")
155
168
  split_content = [content[i : i + 1000] for i in range(0, len(content), 1000)]
156
169
  return [c for c in split_content if len(c.strip()) > 0]
157
170
 
171
+ def _join_documents(self, documents: list[BaseDocument]) -> tuple[str, dict[str, Any]]:
172
+ """
173
+ **It is highly recommended to override this method.**
174
+ **This method should effectively reverse the `_split_corpus` method.**
175
+ Join a list of documents back into a single corpus string.
176
+ Return an instance of corpus metadata. This is a best effort, since not all properties in
177
+ `BaseDocumentMetadata`/`document_metadata_cls` are relevant to the corpus.
178
+ """
179
+ if self.__class__ is not BaseCorpusManager:
180
+ logger.warning("Using default _join_documents. Override this method to improve functionality.")
181
+ documents.sort(key=lambda d: d.chunk_index) # type: ignore
182
+ # since _split_corpus performs a simple split on every 1000 chars, we can simply call `join`
183
+ corpus_content = "".join(d.content for d in documents) # type: ignore
184
+ corpus_metadata = self._infer_corpus_metadata(documents)
185
+ return corpus_content, corpus_metadata
186
+
158
187
  def _extract_chunk_metadata(self, content: str) -> dict[str, Any]:
159
188
  """
160
189
  **It is highly recommended to override this method.**
161
- Extract metadata from a chunk of content, to be appended to corpus metadata
190
+ Extract metadata from a chunk of content, to be appended to corpus metadata.
191
+ Note: returning a key-value pair here does NOT guarantee its inclusion when it's added to the database.
162
192
  """
163
193
  if self.__class__ is not BaseCorpusManager:
164
194
  logger.warning("Using default _extract_chunk_metadata. It is highly recommended to override this method.")
195
+ # this is simply an example. Since the document metadata that gets saved
165
196
  return {
166
197
  "chunk_length": len(content),
167
198
  }
199
+
200
+ def _infer_corpus_metadata(self, documents: list[BaseDocument]) -> dict[str, Any]:
201
+ """
202
+ **It is highly recommended to override this method.**
203
+ **This method should be a best-effort reversal of `extract_chunk_metadata()`**
204
+ Infer metadata for the corpus from the constituent `BaseDocument`s.
205
+ """
206
+ if self.__class__ is not BaseCorpusManager:
207
+ logger.warning("Using default _infer_corpus_metadata. It is highly recommended to override this method.")
208
+ # merge all document.document_metadata together, and return
209
+ merged = {}
210
+ for d in documents:
211
+ merged.update(d.document_metadata) # type: ignore
212
+ return merged
213
+
214
+ def _check_insert_dependencies(self) -> None:
215
+ """Check that required dependencies for insert operations are available"""
216
+ if not self.config.embedding_provider:
217
+ raise ValueError("embedding_provider must be provided in config for insert operations")
218
+ if not self.config.document_metadata_cls:
219
+ raise ValueError("document_metadata_cls must be provided in config for insert operations")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: pgvector-template
3
- Version: 0.1.0
3
+ Version: 0.1.2
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -11,12 +11,12 @@ Classifier: Operating System :: OS Independent
11
11
  Requires-Python: >=3.11
12
12
  Description-Content-Type: text/markdown
13
13
  License-File: LICENSE
14
- Requires-Dist: psycopg[binary]>=3.1.0
15
14
  Requires-Dist: pgvector>=0.2.0
16
15
  Requires-Dist: pydantic<3.0,>=2.11
17
16
  Requires-Dist: sqlalchemy>=2.0.0
18
17
  Requires-Dist: typing-extensions>=4.0.0
19
18
  Provides-Extra: test
19
+ Requires-Dist: psycopg[binary]>=3.1.0; extra == "test"
20
20
  Requires-Dist: pytest>=7.0.0; extra == "test"
21
21
  Requires-Dist: pytest-cov>=4.0.0; extra == "test"
22
22
  Requires-Dist: python-dotenv>=1.0.0; extra == "test"
@@ -4,12 +4,11 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "pgvector-template"
7
- version = "0.1.0"
7
+ version = "0.1.2"
8
8
  description = "Template library for flexible PGVector RAG implementations"
9
9
  authors = [{ name="DL", email="v49t9zpqd@mozmail.com" }]
10
10
  license = { text = "MIT" }
11
11
  dependencies = [
12
- "psycopg[binary]>=3.1.0",
13
12
  "pgvector>=0.2.0",
14
13
  "pydantic>=2.11,<3.0",
15
14
  "sqlalchemy>=2.0.0",
@@ -25,6 +24,7 @@ classifiers = [
25
24
 
26
25
  [project.optional-dependencies]
27
26
  test = [
27
+ "psycopg[binary]>=3.1.0",
28
28
  "pytest>=7.0.0",
29
29
  "pytest-cov>=4.0.0",
30
30
  "python-dotenv>=1.0.0",
@@ -1,10 +1,10 @@
1
- psycopg[binary]>=3.1.0
2
1
  pgvector>=0.2.0
3
2
  pydantic<3.0,>=2.11
4
3
  sqlalchemy>=2.0.0
5
4
  typing-extensions>=4.0.0
6
5
 
7
6
  [test]
7
+ psycopg[binary]>=3.1.0
8
8
  pytest>=7.0.0
9
9
  pytest-cov>=4.0.0
10
10
  python-dotenv>=1.0.0