pgvector-template 0.3.5__tar.gz → 0.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {pgvector_template-0.3.5 → pgvector_template-0.5}/PKG-INFO +1 -1
  2. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/document.py +30 -30
  3. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/embedder.py +13 -0
  4. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/manager.py +2 -0
  5. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/search.py +31 -24
  6. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/models/search.py +14 -6
  7. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/PKG-INFO +1 -1
  8. {pgvector_template-0.3.5 → pgvector_template-0.5}/pyproject.toml +2 -2
  9. {pgvector_template-0.3.5 → pgvector_template-0.5}/LICENSE +0 -0
  10. {pgvector_template-0.3.5 → pgvector_template-0.5}/README.md +0 -0
  11. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/__init__.py +0 -0
  12. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/__init__.py +0 -0
  13. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/db/__init__.py +0 -0
  14. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/db/connection.py +0 -0
  15. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/db/document_db.py +0 -0
  16. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/models/__init__.py +0 -0
  17. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/service/__init__.py +0 -0
  18. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/service/document_service.py +0 -0
  19. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/types.py +0 -0
  20. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/utils/__init__.py +0 -0
  21. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/utils/metadata_filter.py +0 -0
  22. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/SOURCES.txt +0 -0
  23. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/dependency_links.txt +0 -0
  24. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/requires.txt +0 -0
  25. {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/top_level.txt +0 -0
  26. {pgvector_template-0.3.5 → pgvector_template-0.5}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pgvector-template
3
- Version: 0.3.5
3
+ Version: 0.5
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -1,8 +1,8 @@
1
- from datetime import datetime
1
+ from datetime import datetime, timezone
2
2
  from typing import Any, Type, TypeVar
3
3
  from uuid import uuid4, UUID as UuidLiteral
4
4
 
5
- from pydantic import BaseModel, Field, field_validator
5
+ from pydantic import BaseModel, Field
6
6
  from sqlalchemy import (
7
7
  Column,
8
8
  String,
@@ -10,7 +10,6 @@ from sqlalchemy import (
10
10
  DateTime,
11
11
  Boolean,
12
12
  Integer,
13
- Float,
14
13
  Index,
15
14
  UniqueConstraint,
16
15
  )
@@ -32,21 +31,6 @@ class BaseDocumentOptionalProps(BaseModel):
32
31
  """Optional source URL for the document"""
33
32
  language: str | None = Field(default="en", pattern=r"^[a-z]{2}(-[A-Z]{2})?$")
34
33
  """Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'"""
35
- score: float | None = Field(default=None, ge=0.0, le=1.0)
36
- """Optional score assigned during ingestion (e.g., relevance, confidence)"""
37
- tags: list[str] | None = None
38
- """List of tags or keywords for filtering, categorization, or faceted search"""
39
-
40
- @field_validator("tags")
41
- @classmethod
42
- def validate_tags(cls, v):
43
- if v is not None:
44
- # Ensure all tags are strings and not empty
45
- if not all(isinstance(tag, str) and tag.strip() for tag in v):
46
- raise ValueError("All tags must be non-empty strings")
47
- # Remove duplicates while preserving order
48
- return list(dict.fromkeys(v))
49
- return v
50
34
 
51
35
 
52
36
  T = TypeVar("T", bound="BaseDocument")
@@ -88,18 +72,29 @@ class BaseDocument(Base):
88
72
  """Optional source URL"""
89
73
  language = Column(String(2), default="en")
90
74
  """Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'."""
91
- score = Column(Float, nullable=True)
92
- """Optional score assigned during ingestion (e.g., relevance, confidence)."""
93
- tags = Column(JSONB, nullable=True, default=list)
94
- """List of tags or keywords for filtering, categorization, or faceted search."""
95
75
 
96
76
  # Vector embedding
97
77
  embedding = Column(Vector(1024))
98
78
  """Embedding vector. 1024 dimensions by default. Adjust as-needed."""
79
+ embedding_config = Column(JSONB, nullable=True)
80
+ """
81
+ Configuration used to generate the embedding vector. Stored as a JSON object so it
82
+ can capture provider-specific parameters beyond just the model ID — e.g. input_type,
83
+ dimensions, truncation strategy, or any future provider-specific options.
84
+
85
+ Recommended shape: {"model": "cohere.embed-v4:0", "input_type": "search_document", ...}
86
+
87
+ Intentionally nullable: rows embedded before this column was added will have NULL here.
88
+ Use this field for auditing and diagnosing mixed-model corpora, not for runtime dispatch.
89
+ """
99
90
 
100
91
  # Audit fields
101
- created_at = Column(DateTime, default=datetime.utcnow)
102
- updated_at = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
92
+ created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(timezone.utc))
93
+ updated_at = Column(
94
+ DateTime(timezone=True),
95
+ default=lambda: datetime.now(timezone.utc),
96
+ onupdate=lambda: datetime.now(timezone.utc),
97
+ )
103
98
  is_deleted = Column(Boolean, default=False, index=True)
104
99
  """Entries can be logically marked for deletion before they are permanently deleted."""
105
100
 
@@ -112,10 +107,11 @@ class BaseDocument(Base):
112
107
 
113
108
  base_table_args = (
114
109
  Index(
115
- f"{table_name}_metadata_gin_idx", "document_metadata", postgresql_using="gin"
110
+ f"{table_name}_metadata_gin_idx",
111
+ "document_metadata",
112
+ postgresql_using="gin",
116
113
  ),
117
114
  cls.get_embedding_index(table_name),
118
- Index(f"{table_name}_tags_gin_idx", "tags", postgresql_using="gin"),
119
115
  Index(f"{table_name}_collection_corpus_idx", "collection", "corpus_id"),
120
116
  UniqueConstraint(
121
117
  "collection",
@@ -140,7 +136,8 @@ class BaseDocument(Base):
140
136
  chunk_index: int,
141
137
  content: str,
142
138
  embedding: list[float],
143
- metadata: dict[str, Any] = {},
139
+ embedding_config: dict[str, Any] | None = None,
140
+ metadata: dict[str, Any] | None = None,
144
141
  optional_props: BaseDocumentOptionalProps | None = None,
145
142
  ) -> T:
146
143
  """
@@ -151,6 +148,10 @@ class BaseDocument(Base):
151
148
  chunk_index: Index of this chunk within the corpus
152
149
  content: Text content of the document
153
150
  embedding: Vector embedding of the content
151
+ embedding_config: Provider config used to generate the embedding. Should include
152
+ at minimum {"model": "<model_id>"} plus any provider-specific params
153
+ (e.g. {"model": "cohere.embed-v4:0", "input_type": "search_document"}).
154
+ Nullable — rows pre-dating this column will have NULL.
154
155
  optional_props: Optional properties for the document
155
156
 
156
157
  Returns:
@@ -165,13 +166,12 @@ class BaseDocument(Base):
165
166
  chunk_index=chunk_index,
166
167
  content=content,
167
168
  embedding=embedding,
169
+ embedding_config=embedding_config,
168
170
  title=optional_props.title,
169
- document_metadata=metadata,
171
+ document_metadata=metadata or {},
170
172
  collection=optional_props.collection,
171
173
  origin_url=optional_props.original_url,
172
174
  language=optional_props.language,
173
- score=optional_props.score,
174
- tags=optional_props.tags,
175
175
  )
176
176
 
177
177
  @classmethod
@@ -1,9 +1,22 @@
1
1
  from abc import ABC, abstractmethod
2
+ from typing import Any
2
3
 
3
4
 
4
5
  class BaseEmbeddingProvider(ABC):
5
6
  """Abstract base for embedding generation"""
6
7
 
8
+ def __init__(self, model_id: str, **kwargs):
9
+ self.model_id = model_id
10
+
11
+ def get_embedding_config(self) -> dict[str, Any]:
12
+ """
13
+ Return a JSON-serializable dict describing the full embedding configuration.
14
+ Stored in the `embedding_config` column on every document row.
15
+ Subclasses should override to include provider-specific params (e.g. input_type).
16
+ Default implementation returns just the model ID.
17
+ """
18
+ return {"model": self.model_id}
19
+
7
20
  @abstractmethod
8
21
  def embed_text(self, text: str) -> list[float]:
9
22
  """Generate embedding vector for text"""
@@ -204,6 +204,7 @@ class BaseCorpusManager(ABC):
204
204
  optional_props: BaseDocumentOptionalProps | None,
205
205
  ) -> list[BaseDocument]:
206
206
  """Create document instances from the provided data"""
207
+ embedding_config = self.embedding_provider.get_embedding_config()
207
208
  documents = []
208
209
  for i, (content, embedding) in enumerate(zip(document_contents, document_embeddings)):
209
210
  chunk_md = self._extract_chunk_metadata(content)
@@ -214,6 +215,7 @@ class BaseCorpusManager(ABC):
214
215
  chunk_index=i,
215
216
  content=content,
216
217
  embedding=embedding,
218
+ embedding_config=embedding_config,
217
219
  metadata=base_metadata.model_dump(),
218
220
  optional_props=optional_props,
219
221
  )
@@ -69,38 +69,43 @@ class BaseSearchClient:
69
69
  query: Search query containing text, metadata filters, and pagination.
70
70
 
71
71
  Returns:
72
- List of retrieval results matching the search criteria.
72
+ List of retrieval results matching the search criteria. Each result's `score`
73
+ is a float in [0, 1] when `query.text` is set (1.0 = most similar), or `None`
74
+ for keyword/metadata-only queries.
73
75
  """
74
- db_query = select(self.config.document_cls)
76
+ distance_expr: ColumnElement | None = None
75
77
 
76
78
  if query.text:
77
- db_query = self._apply_semantic_search(db_query, query)
79
+ distance_expr = self._build_distance_expr(query)
80
+ db_query = select(self.config.document_cls, distance_expr).order_by(distance_expr)
81
+ else:
82
+ db_query = select(self.config.document_cls)
83
+
78
84
  db_query = self._apply_keyword_search(db_query, query)
79
85
  if query.metadata_filters:
80
86
  db_query = self._apply_metadata_filters(db_query, query)
81
87
  db_query = db_query.limit(query.limit)
82
88
 
83
- # execute query and return results
84
- results = self.session.scalars(db_query).all()
85
- return self._convert_to_retrieval_results(results)
89
+ if distance_expr is not None:
90
+ rows = self.session.execute(db_query).all()
91
+ return self._convert_to_retrieval_results(rows)
92
+ else:
93
+ results = self.session.scalars(db_query).all()
94
+ return self._convert_to_retrieval_results([(doc, None) for doc in results])
86
95
 
87
- def _apply_semantic_search(self, db_query: Select, search_query: SearchQuery) -> Select:
88
- """Apply semantic (vector) search criteria to the query.
89
- `embedding_provider` must be provided at instantiation, or an `ValueError` will be raised.
90
- In PGVector, `<=>` operator is used to compare cosine distance. Lower = more similar.
96
+ def _build_distance_expr(self, search_query: SearchQuery) -> ColumnElement:
97
+ """Build a labeled cosine-distance expression for the given query text.
91
98
 
92
- Args:
93
- query: The base SQLAlchemy query.
94
- search_query: The search query containing the text to search for.
99
+ `embedding_provider` must be provided at instantiation, or a `ValueError` is raised.
100
+ In PGVector, `<=>` computes cosine distance (0 = identical, up to 2 = opposite).
95
101
 
96
102
  Returns:
97
- Updated SQLAlchemy query with semantic search applied.
103
+ A SQLAlchemy labeled column expression suitable for SELECT and ORDER BY.
98
104
  """
99
- if not search_query.text:
100
- return db_query
105
+ assert search_query.text is not None
101
106
  query_embedding = self.embedding_provider.embed_text(search_query.text)
102
- return db_query.order_by(
103
- self.config.document_cls.embedding.cosine_distance(query_embedding)
107
+ return self.config.document_cls.embedding.cosine_distance(query_embedding).label(
108
+ "_score_distance"
104
109
  )
105
110
 
106
111
  def _apply_keyword_search(self, db_query: Select, search_query: SearchQuery) -> Select:
@@ -199,16 +204,18 @@ class BaseSearchClient:
199
204
  raise ValueError(f"Unsupported condition: {filter_obj.condition}")
200
205
 
201
206
  def _convert_to_retrieval_results(self, results: Sequence[Any]) -> list[RetrievalResult]:
202
- """Convert database results to RetrievalResult objects.
207
+ """Convert (document, distance | None) pairs to RetrievalResult objects.
203
208
 
204
209
  Args:
205
- results: Raw database results.
206
- search_query: The original search query.
210
+ results: Sequence of (document, cosine_distance) tuples. Pass distance=None
211
+ for keyword/metadata-only queries that have no semantic ranking.
207
212
 
208
213
  Returns:
209
- List of RetrievalResult objects.
214
+ List of RetrievalResult objects with score = max(0.0, 1.0 - distance),
215
+ or score=None when distance is None.
210
216
  """
211
217
  retrieval_results = []
212
- for result in results:
213
- retrieval_results.append(RetrievalResult(document=result, score=1.0))
218
+ for document, distance in results:
219
+ score = max(0.0, 1.0 - distance / 2.0) if distance is not None else None
220
+ retrieval_results.append(RetrievalResult(document=document, score=score))
214
221
  return retrieval_results
@@ -1,5 +1,4 @@
1
1
  from dataclasses import dataclass, asdict
2
- from datetime import datetime
3
2
  from typing import Any, Literal
4
3
 
5
4
  from pydantic import BaseModel, ConfigDict, Field, model_validator
@@ -38,7 +37,13 @@ class SearchQuery(BaseModel):
38
37
  text: str | None = None
39
38
  """String to match against using in a semantic search, i.e. using vector distance."""
40
39
  keywords: list[str] = []
41
- """List of keywords to **exact-match** in a keyword search."""
40
+ """
41
+ List of keywords for a case-insensitive **substring** search (SQL `ILIKE '%keyword%'`).
42
+ An entry matches if **any** keyword appears anywhere in its content (OR-combined).
43
+ Pass multiple variants, synonyms, or spellings to increase recall — e.g.
44
+ ["run", "running", "ran"] will match entries containing any of those forms.
45
+ Note: substring matching means "cat" also matches "category"; prefer distinctive stems.
46
+ """
42
47
  metadata_filters: list[MetadataFilter] = Field(
43
48
  default=[],
44
49
  json_schema_extra={"metadata_schema": BaseDocumentMetadata.model_json_schema()},
@@ -47,8 +52,6 @@ class SearchQuery(BaseModel):
47
52
  List of metadata conditions that must be matched.
48
53
  Refer to `metadata_schema` for the expected schema, as it exists in the database.
49
54
  """
50
- date_range: tuple[datetime, datetime] | None = None
51
- """Retrieve/limit results based on created_at & updated_at timestamps (i.e. database operations)"""
52
55
  limit: int = Field(
53
56
  ...,
54
57
  ge=1,
@@ -62,7 +65,7 @@ class SearchQuery(BaseModel):
62
65
 
63
66
  @model_validator(mode="after")
64
67
  def ensure_criterion(self):
65
- if not any([self.text, self.keywords, self.metadata_filters, self.date_range]):
68
+ if not any([self.text, self.keywords, self.metadata_filters]):
66
69
  raise ValueError("At least one search criterion is required")
67
70
  return self
68
71
 
@@ -72,7 +75,12 @@ class RetrievalResult:
72
75
  """Standardized result structure for all retrieval operations"""
73
76
 
74
77
  document: BaseDocument
75
- score: float
78
+ score: float | None
79
+ """
80
+ Semantic similarity score in [0, 1] when a `text` query was provided (1.0 = identical,
81
+ 0.0 = maximally distant). `None` when the search had no semantic component (keywords /
82
+ metadata filters only) — results are still filtered correctly but have no meaningful ranking.
83
+ """
76
84
 
77
85
  def to_dict(self) -> dict[str, Any]:
78
86
  return asdict(self)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pgvector-template
3
- Version: 0.3.5
3
+ Version: 0.5
4
4
  Summary: Template library for flexible PGVector RAG implementations
5
5
  Author-email: DL <v49t9zpqd@mozmail.com>
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "pgvector-template"
7
- version = "0.3.5"
7
+ version = "0.5"
8
8
  description = "Template library for flexible PGVector RAG implementations"
9
9
  authors = [{ name="DL", email="v49t9zpqd@mozmail.com" }]
10
10
  license = { text = "MIT" }
@@ -52,4 +52,4 @@ line-length = 100
52
52
  target-version = "py311"
53
53
 
54
54
  [tool.ty.src]
55
- exclude = []
55
+ exclude = ["integ-tests"]