pgvector-template 0.2.1__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/PKG-INFO +1 -1
  2. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/core/document.py +46 -6
  3. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/core/search.py +5 -4
  4. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/db/document_db.py +2 -1
  5. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/service/document_service.py +2 -0
  6. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template.egg-info/PKG-INFO +1 -1
  7. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pyproject.toml +1 -1
  8. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/LICENSE +0 -0
  9. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/README.md +0 -0
  10. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/__init__.py +0 -0
  11. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/core/__init__.py +0 -0
  12. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/core/embedder.py +0 -0
  13. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/core/manager.py +0 -0
  14. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/db/__init__.py +0 -0
  15. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/db/connection.py +0 -0
  16. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/service/__init__.py +0 -0
  17. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/types.py +0 -0
  18. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template/utils/__init__.py +0 -0
  19. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template.egg-info/SOURCES.txt +0 -0
  20. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template.egg-info/dependency_links.txt +0 -0
  21. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template.egg-info/requires.txt +0 -0
  22. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/pgvector_template.egg-info/top_level.txt +0 -0
  23. {pgvector_template-0.2.1 → pgvector_template-0.2.3}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: pgvector-template
3
- Version: 0.2.1
3
+ Version: 0.2.3
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -12,6 +12,7 @@ from sqlalchemy import (
12
12
  Integer,
13
13
  Float,
14
14
  Index,
15
+ UniqueConstraint,
15
16
  )
16
17
  from sqlalchemy.orm import declarative_base
17
18
  from sqlalchemy.dialects.postgresql import UUID, JSONB
@@ -63,6 +64,7 @@ class BaseDocument(Base):
63
64
  """
64
65
 
65
66
  __abstract__ = True
67
+ __tablename__ = "base_document"
66
68
 
67
69
  id = Column(UUID(as_uuid=True), primary_key=True, default=uuid4)
68
70
  """Primary key of the Document table. Represents unique ID of a Document"""
@@ -84,7 +86,7 @@ class BaseDocument(Base):
84
86
  """Flexible metadata as JSON"""
85
87
  origin_url = Column(String(2048), nullable=True)
86
88
  """Optional source URL"""
87
- language = Column(String(10), default="en")
89
+ language = Column(String(2), default="en")
88
90
  """Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'."""
89
91
  score = Column(Float, nullable=True)
90
92
  """Optional score assigned during ingestion (e.g., relevance, confidence)."""
@@ -98,9 +100,39 @@ class BaseDocument(Base):
98
100
  # Audit fields
99
101
  created_at = Column(DateTime, default=datetime.utcnow)
100
102
  updated_at = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
101
- is_deleted = Column(Boolean, default=False)
103
+ is_deleted = Column(Boolean, default=False, index=True)
102
104
  """Entries can be logically marked for deletion before they are permanently deleted."""
103
105
 
106
+ def __init_subclass__(cls, **kwargs):
107
+ """Simply declare `__table_args__` to add more args in child classes."""
108
+ super().__init_subclass__(**kwargs)
109
+
110
+ if hasattr(cls, "__tablename__"):
111
+ table_name = cls.__tablename__
112
+
113
+ base_table_args = (
114
+ Index(
115
+ f"{table_name}_metadata_gin_idx", "document_metadata", postgresql_using="gin"
116
+ ),
117
+ cls.get_embedding_index(table_name),
118
+ Index(f"{table_name}_tags_gin_idx", "tags", postgresql_using="gin"),
119
+ Index(f"{table_name}_collection_corpus_idx", "collection", "corpus_id"),
120
+ UniqueConstraint(
121
+ "collection",
122
+ "corpus_id",
123
+ "chunk_index",
124
+ name=f"{table_name}_collection_corpus_chunk_unique",
125
+ ),
126
+ )
127
+
128
+ subclass_args = getattr(cls, "__table_args__", ())
129
+ if isinstance(subclass_args, dict):
130
+ cls.__table_args__ = base_table_args + (subclass_args,)
131
+ else:
132
+ if not isinstance(subclass_args, tuple):
133
+ subclass_args = (subclass_args,) if subclass_args else ()
134
+ cls.__table_args__ = base_table_args + subclass_args
135
+
104
136
  @classmethod
105
137
  def from_props(
106
138
  cls: Type[T],
@@ -142,9 +174,15 @@ class BaseDocument(Base):
142
174
  tags=optional_props.tags,
143
175
  )
144
176
 
145
- # Index("ix_corpus_chunk", "corpus_id", "chunk_index")
146
- # Index("ix_content_trgm", text("content gin_trgm_ops"), postgresql_using="gin") # For fuzzy text search
147
- # Index("ix_metadata_gin", "metadata", postgresql_using="gin")
177
+ @classmethod
178
+ def get_embedding_index(cls, table_name: str) -> Index:
179
+ """Override this method to customize the embedding index."""
180
+ return Index(
181
+ f"{table_name}_embedding_hnsw_idx",
182
+ "embedding",
183
+ postgresql_using="hnsw",
184
+ postgresql_ops={"embedding": "vector_cosine_ops"},
185
+ )
148
186
 
149
187
 
150
188
  class BaseDocumentMetadata(BaseModel):
@@ -154,7 +192,9 @@ class BaseDocumentMetadata(BaseModel):
154
192
  without any extraneous properties, or any missing properties, to avoid ambiguity.
155
193
  """
156
194
 
157
- document_type: str = Field(..., description="Description for type of document, e.g. markdown, pdf, etc")
195
+ document_type: str = Field(
196
+ ..., description="Description for type of document, e.g. markdown, pdf, etc"
197
+ )
158
198
  schema_version: str = Field(default="1.0", description="Schema version for the metadata")
159
199
 
160
200
  def to_dict(self) -> dict[str, Any]:
@@ -3,10 +3,9 @@ from datetime import datetime
3
3
  from logging import getLogger
4
4
  from typing import Any, Type, Sequence
5
5
 
6
- from pydantic import BaseModel, Field, model_validator
7
- from sqlalchemy import text, select, or_, and_
6
+ from pydantic import BaseModel, ConfigDict, Field, model_validator
7
+ from sqlalchemy import select, or_
8
8
  from sqlalchemy.sql import Select
9
- from pgvector.sqlalchemy import Vector
10
9
 
11
10
  from pgvector_template.core import (
12
11
  BaseEmbeddingProvider,
@@ -23,7 +22,7 @@ class SearchQuery(BaseModel):
23
22
  """Standardized search query structure. At least 1 search criterion is required."""
24
23
 
25
24
  text: str | None = None
26
- """String to approximate-search (using vector distance) in a semantic search."""
25
+ """String to match against using in a semantic search, i.e. using vector distance."""
27
26
  keywords: list[str] | None = None
28
27
  """List of keywords to exact-match in a keyword search."""
29
28
  metadata_filters: dict[str, Any] | None = None
@@ -36,6 +35,8 @@ class SearchQuery(BaseModel):
36
35
  )
37
36
  """Maximum number of results to return."""
38
37
 
38
+ model_config = ConfigDict(use_attribute_docstrings=True)
39
+
39
40
  @model_validator(mode="after")
40
41
  def ensure_criterion(self):
41
42
  if not any([self.text, self.keywords, self.metadata_filters, self.date_range]):
@@ -37,7 +37,7 @@ class DocumentDatabaseManager(DatabaseManager):
37
37
  self.schema_name = f"{self.SCHEMA_PREFIX}{schema_suffix}"
38
38
  self.document_classes = document_classes
39
39
 
40
- def setup(self) -> None:
40
+ def setup(self) -> str:
41
41
  """One-step setup: initialize connection, create schema and tables for all document classes"""
42
42
  self.initialize()
43
43
  self.create_schema(self.schema_name)
@@ -50,6 +50,7 @@ class DocumentDatabaseManager(DatabaseManager):
50
50
  self.logger.info(
51
51
  f"Document database setup complete for {self.schema_name} with {len(self.document_classes)} tables"
52
52
  )
53
+ return self.schema_name
53
54
 
54
55
 
55
56
  class TempDocumentDatabaseManager(DocumentDatabaseManager):
@@ -50,8 +50,10 @@ class DocumentServiceConfig(BaseModel):
50
50
  # iff either config is an instance of their respective base config classes (and not a subclass)
51
51
  if type(self.corpus_manager_cfg) is BaseCorpusManagerConfig:
52
52
  self.corpus_manager_cfg.document_cls = self.document_cls
53
+ self.corpus_manager_cfg.document_metadata_cls = self.document_metadata_cls
53
54
  if type(self.search_client_cfg) is BaseSearchClientConfig:
54
55
  self.search_client_cfg.document_cls = self.document_cls
56
+ self.search_client_cfg.document_metadata_cls = self.document_metadata_cls
55
57
 
56
58
  # assign embedding_provider to CorpusManager & SearchClient configs
57
59
  if not self.corpus_manager_cfg.embedding_provider:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: pgvector-template
3
- Version: 0.2.1
3
+ Version: 0.2.3
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "pgvector-template"
7
- version = "0.2.1"
7
+ version = "0.2.3"
8
8
  description = "Template library for flexible PGVector RAG implementations"
9
9
  authors = [{ name="DL", email="v49t9zpqd@mozmail.com" }]
10
10
  license = { text = "MIT" }