pgvector-template 0.3.5__tar.gz → 0.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {pgvector_template-0.3.5 → pgvector_template-0.4}/PKG-INFO +1 -1
  2. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/document.py +30 -30
  3. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/embedder.py +13 -0
  4. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/manager.py +2 -0
  5. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/PKG-INFO +1 -1
  6. {pgvector_template-0.3.5 → pgvector_template-0.4}/pyproject.toml +2 -2
  7. {pgvector_template-0.3.5 → pgvector_template-0.4}/LICENSE +0 -0
  8. {pgvector_template-0.3.5 → pgvector_template-0.4}/README.md +0 -0
  9. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/__init__.py +0 -0
  10. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/__init__.py +0 -0
  11. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/search.py +0 -0
  12. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/db/__init__.py +0 -0
  13. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/db/connection.py +0 -0
  14. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/db/document_db.py +0 -0
  15. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/models/__init__.py +0 -0
  16. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/models/search.py +0 -0
  17. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/service/__init__.py +0 -0
  18. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/service/document_service.py +0 -0
  19. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/types.py +0 -0
  20. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/utils/__init__.py +0 -0
  21. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/utils/metadata_filter.py +0 -0
  22. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/SOURCES.txt +0 -0
  23. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/dependency_links.txt +0 -0
  24. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/requires.txt +0 -0
  25. {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/top_level.txt +0 -0
  26. {pgvector_template-0.3.5 → pgvector_template-0.4}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pgvector-template
3
- Version: 0.3.5
3
+ Version: 0.4
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -1,8 +1,8 @@
1
- from datetime import datetime
1
+ from datetime import datetime, timezone
2
2
  from typing import Any, Type, TypeVar
3
3
  from uuid import uuid4, UUID as UuidLiteral
4
4
 
5
- from pydantic import BaseModel, Field, field_validator
5
+ from pydantic import BaseModel, Field
6
6
  from sqlalchemy import (
7
7
  Column,
8
8
  String,
@@ -10,7 +10,6 @@ from sqlalchemy import (
10
10
  DateTime,
11
11
  Boolean,
12
12
  Integer,
13
- Float,
14
13
  Index,
15
14
  UniqueConstraint,
16
15
  )
@@ -32,21 +31,6 @@ class BaseDocumentOptionalProps(BaseModel):
32
31
  """Optional source URL for the document"""
33
32
  language: str | None = Field(default="en", pattern=r"^[a-z]{2}(-[A-Z]{2})?$")
34
33
  """Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'"""
35
- score: float | None = Field(default=None, ge=0.0, le=1.0)
36
- """Optional score assigned during ingestion (e.g., relevance, confidence)"""
37
- tags: list[str] | None = None
38
- """List of tags or keywords for filtering, categorization, or faceted search"""
39
-
40
- @field_validator("tags")
41
- @classmethod
42
- def validate_tags(cls, v):
43
- if v is not None:
44
- # Ensure all tags are strings and not empty
45
- if not all(isinstance(tag, str) and tag.strip() for tag in v):
46
- raise ValueError("All tags must be non-empty strings")
47
- # Remove duplicates while preserving order
48
- return list(dict.fromkeys(v))
49
- return v
50
34
 
51
35
 
52
36
  T = TypeVar("T", bound="BaseDocument")
@@ -88,18 +72,29 @@ class BaseDocument(Base):
88
72
  """Optional source URL"""
89
73
  language = Column(String(2), default="en")
90
74
  """Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'."""
91
- score = Column(Float, nullable=True)
92
- """Optional score assigned during ingestion (e.g., relevance, confidence)."""
93
- tags = Column(JSONB, nullable=True, default=list)
94
- """List of tags or keywords for filtering, categorization, or faceted search."""
95
75
 
96
76
  # Vector embedding
97
77
  embedding = Column(Vector(1024))
98
78
  """Embedding vector. 1024 dimensions by default. Adjust as-needed."""
79
+ embedding_config = Column(JSONB, nullable=True)
80
+ """
81
+ Configuration used to generate the embedding vector. Stored as a JSON object so it
82
+ can capture provider-specific parameters beyond just the model ID — e.g. input_type,
83
+ dimensions, truncation strategy, or any future provider-specific options.
84
+
85
+ Recommended shape: {"model": "cohere.embed-v4:0", "input_type": "search_document", ...}
86
+
87
+ Intentionally nullable: rows embedded before this column was added will have NULL here.
88
+ Use this field for auditing and diagnosing mixed-model corpora, not for runtime dispatch.
89
+ """
99
90
 
100
91
  # Audit fields
101
- created_at = Column(DateTime, default=datetime.utcnow)
102
- updated_at = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
92
+ created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(timezone.utc))
93
+ updated_at = Column(
94
+ DateTime(timezone=True),
95
+ default=lambda: datetime.now(timezone.utc),
96
+ onupdate=lambda: datetime.now(timezone.utc),
97
+ )
103
98
  is_deleted = Column(Boolean, default=False, index=True)
104
99
  """Entries can be logically marked for deletion before they are permanently deleted."""
105
100
 
@@ -112,10 +107,11 @@ class BaseDocument(Base):
112
107
 
113
108
  base_table_args = (
114
109
  Index(
115
- f"{table_name}_metadata_gin_idx", "document_metadata", postgresql_using="gin"
110
+ f"{table_name}_metadata_gin_idx",
111
+ "document_metadata",
112
+ postgresql_using="gin",
116
113
  ),
117
114
  cls.get_embedding_index(table_name),
118
- Index(f"{table_name}_tags_gin_idx", "tags", postgresql_using="gin"),
119
115
  Index(f"{table_name}_collection_corpus_idx", "collection", "corpus_id"),
120
116
  UniqueConstraint(
121
117
  "collection",
@@ -140,7 +136,8 @@ class BaseDocument(Base):
140
136
  chunk_index: int,
141
137
  content: str,
142
138
  embedding: list[float],
143
- metadata: dict[str, Any] = {},
139
+ embedding_config: dict[str, Any] | None = None,
140
+ metadata: dict[str, Any] | None = None,
144
141
  optional_props: BaseDocumentOptionalProps | None = None,
145
142
  ) -> T:
146
143
  """
@@ -151,6 +148,10 @@ class BaseDocument(Base):
151
148
  chunk_index: Index of this chunk within the corpus
152
149
  content: Text content of the document
153
150
  embedding: Vector embedding of the content
151
+ embedding_config: Provider config used to generate the embedding. Should include
152
+ at minimum {"model": "<model_id>"} plus any provider-specific params
153
+ (e.g. {"model": "cohere.embed-v4:0", "input_type": "search_document"}).
154
+ Nullable — rows pre-dating this column will have NULL.
154
155
  optional_props: Optional properties for the document
155
156
 
156
157
  Returns:
@@ -165,13 +166,12 @@ class BaseDocument(Base):
165
166
  chunk_index=chunk_index,
166
167
  content=content,
167
168
  embedding=embedding,
169
+ embedding_config=embedding_config,
168
170
  title=optional_props.title,
169
- document_metadata=metadata,
171
+ document_metadata=metadata or {},
170
172
  collection=optional_props.collection,
171
173
  origin_url=optional_props.original_url,
172
174
  language=optional_props.language,
173
- score=optional_props.score,
174
- tags=optional_props.tags,
175
175
  )
176
176
 
177
177
  @classmethod
@@ -1,9 +1,22 @@
1
1
  from abc import ABC, abstractmethod
2
+ from typing import Any
2
3
 
3
4
 
4
5
  class BaseEmbeddingProvider(ABC):
5
6
  """Abstract base for embedding generation"""
6
7
 
8
+ def __init__(self, model_id: str, **kwargs):
9
+ self.model_id = model_id
10
+
11
+ def get_embedding_config(self) -> dict[str, Any]:
12
+ """
13
+ Return a JSON-serializable dict describing the full embedding configuration.
14
+ Stored in the `embedding_config` column on every document row.
15
+ Subclasses should override to include provider-specific params (e.g. input_type).
16
+ Default implementation returns just the model ID.
17
+ """
18
+ return {"model": self.model_id}
19
+
7
20
  @abstractmethod
8
21
  def embed_text(self, text: str) -> list[float]:
9
22
  """Generate embedding vector for text"""
@@ -204,6 +204,7 @@ class BaseCorpusManager(ABC):
204
204
  optional_props: BaseDocumentOptionalProps | None,
205
205
  ) -> list[BaseDocument]:
206
206
  """Create document instances from the provided data"""
207
+ embedding_config = self.embedding_provider.get_embedding_config()
207
208
  documents = []
208
209
  for i, (content, embedding) in enumerate(zip(document_contents, document_embeddings)):
209
210
  chunk_md = self._extract_chunk_metadata(content)
@@ -214,6 +215,7 @@ class BaseCorpusManager(ABC):
214
215
  chunk_index=i,
215
216
  content=content,
216
217
  embedding=embedding,
218
+ embedding_config=embedding_config,
217
219
  metadata=base_metadata.model_dump(),
218
220
  optional_props=optional_props,
219
221
  )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pgvector-template
3
- Version: 0.3.5
3
+ Version: 0.4
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "pgvector-template"
7
- version = "0.3.5"
7
+ version = "0.4"
8
8
  description = "Template library for flexible PGVector RAG implementations"
9
9
  authors = [{ name="DL", email="v49t9zpqd@mozmail.com" }]
10
10
  license = { text = "MIT" }
@@ -52,4 +52,4 @@ line-length = 100
52
52
  target-version = "py311"
53
53
 
54
54
  [tool.ty.src]
55
- exclude = []
55
+ exclude = ["integ-tests"]