pgvector-template 0.3.5__tar.gz → 0.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pgvector_template-0.3.5 → pgvector_template-0.5}/PKG-INFO +1 -1
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/document.py +30 -30
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/embedder.py +13 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/manager.py +2 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/search.py +31 -24
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/models/search.py +14 -6
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/PKG-INFO +1 -1
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pyproject.toml +2 -2
- {pgvector_template-0.3.5 → pgvector_template-0.5}/LICENSE +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/README.md +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/core/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/db/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/db/connection.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/db/document_db.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/models/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/service/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/service/document_service.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/types.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/utils/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/utils/metadata_filter.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/SOURCES.txt +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/dependency_links.txt +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/requires.txt +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/top_level.txt +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.5}/setup.cfg +0 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
from datetime import datetime
|
|
1
|
+
from datetime import datetime, timezone
|
|
2
2
|
from typing import Any, Type, TypeVar
|
|
3
3
|
from uuid import uuid4, UUID as UuidLiteral
|
|
4
4
|
|
|
5
|
-
from pydantic import BaseModel, Field
|
|
5
|
+
from pydantic import BaseModel, Field
|
|
6
6
|
from sqlalchemy import (
|
|
7
7
|
Column,
|
|
8
8
|
String,
|
|
@@ -10,7 +10,6 @@ from sqlalchemy import (
|
|
|
10
10
|
DateTime,
|
|
11
11
|
Boolean,
|
|
12
12
|
Integer,
|
|
13
|
-
Float,
|
|
14
13
|
Index,
|
|
15
14
|
UniqueConstraint,
|
|
16
15
|
)
|
|
@@ -32,21 +31,6 @@ class BaseDocumentOptionalProps(BaseModel):
|
|
|
32
31
|
"""Optional source URL for the document"""
|
|
33
32
|
language: str | None = Field(default="en", pattern=r"^[a-z]{2}(-[A-Z]{2})?$")
|
|
34
33
|
"""Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'"""
|
|
35
|
-
score: float | None = Field(default=None, ge=0.0, le=1.0)
|
|
36
|
-
"""Optional score assigned during ingestion (e.g., relevance, confidence)"""
|
|
37
|
-
tags: list[str] | None = None
|
|
38
|
-
"""List of tags or keywords for filtering, categorization, or faceted search"""
|
|
39
|
-
|
|
40
|
-
@field_validator("tags")
|
|
41
|
-
@classmethod
|
|
42
|
-
def validate_tags(cls, v):
|
|
43
|
-
if v is not None:
|
|
44
|
-
# Ensure all tags are strings and not empty
|
|
45
|
-
if not all(isinstance(tag, str) and tag.strip() for tag in v):
|
|
46
|
-
raise ValueError("All tags must be non-empty strings")
|
|
47
|
-
# Remove duplicates while preserving order
|
|
48
|
-
return list(dict.fromkeys(v))
|
|
49
|
-
return v
|
|
50
34
|
|
|
51
35
|
|
|
52
36
|
T = TypeVar("T", bound="BaseDocument")
|
|
@@ -88,18 +72,29 @@ class BaseDocument(Base):
|
|
|
88
72
|
"""Optional source URL"""
|
|
89
73
|
language = Column(String(2), default="en")
|
|
90
74
|
"""Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'."""
|
|
91
|
-
score = Column(Float, nullable=True)
|
|
92
|
-
"""Optional score assigned during ingestion (e.g., relevance, confidence)."""
|
|
93
|
-
tags = Column(JSONB, nullable=True, default=list)
|
|
94
|
-
"""List of tags or keywords for filtering, categorization, or faceted search."""
|
|
95
75
|
|
|
96
76
|
# Vector embedding
|
|
97
77
|
embedding = Column(Vector(1024))
|
|
98
78
|
"""Embedding vector. 1024 dimensions by default. Adjust as-needed."""
|
|
79
|
+
embedding_config = Column(JSONB, nullable=True)
|
|
80
|
+
"""
|
|
81
|
+
Configuration used to generate the embedding vector. Stored as a JSON object so it
|
|
82
|
+
can capture provider-specific parameters beyond just the model ID — e.g. input_type,
|
|
83
|
+
dimensions, truncation strategy, or any future provider-specific options.
|
|
84
|
+
|
|
85
|
+
Recommended shape: {"model": "cohere.embed-v4:0", "input_type": "search_document", ...}
|
|
86
|
+
|
|
87
|
+
Intentionally nullable: rows embedded before this column was added will have NULL here.
|
|
88
|
+
Use this field for auditing and diagnosing mixed-model corpora, not for runtime dispatch.
|
|
89
|
+
"""
|
|
99
90
|
|
|
100
91
|
# Audit fields
|
|
101
|
-
created_at = Column(DateTime, default=datetime.
|
|
102
|
-
updated_at = Column(
|
|
92
|
+
created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(timezone.utc))
|
|
93
|
+
updated_at = Column(
|
|
94
|
+
DateTime(timezone=True),
|
|
95
|
+
default=lambda: datetime.now(timezone.utc),
|
|
96
|
+
onupdate=lambda: datetime.now(timezone.utc),
|
|
97
|
+
)
|
|
103
98
|
is_deleted = Column(Boolean, default=False, index=True)
|
|
104
99
|
"""Entries can be logically marked for deletion before they are permanently deleted."""
|
|
105
100
|
|
|
@@ -112,10 +107,11 @@ class BaseDocument(Base):
|
|
|
112
107
|
|
|
113
108
|
base_table_args = (
|
|
114
109
|
Index(
|
|
115
|
-
f"{table_name}_metadata_gin_idx",
|
|
110
|
+
f"{table_name}_metadata_gin_idx",
|
|
111
|
+
"document_metadata",
|
|
112
|
+
postgresql_using="gin",
|
|
116
113
|
),
|
|
117
114
|
cls.get_embedding_index(table_name),
|
|
118
|
-
Index(f"{table_name}_tags_gin_idx", "tags", postgresql_using="gin"),
|
|
119
115
|
Index(f"{table_name}_collection_corpus_idx", "collection", "corpus_id"),
|
|
120
116
|
UniqueConstraint(
|
|
121
117
|
"collection",
|
|
@@ -140,7 +136,8 @@ class BaseDocument(Base):
|
|
|
140
136
|
chunk_index: int,
|
|
141
137
|
content: str,
|
|
142
138
|
embedding: list[float],
|
|
143
|
-
|
|
139
|
+
embedding_config: dict[str, Any] | None = None,
|
|
140
|
+
metadata: dict[str, Any] | None = None,
|
|
144
141
|
optional_props: BaseDocumentOptionalProps | None = None,
|
|
145
142
|
) -> T:
|
|
146
143
|
"""
|
|
@@ -151,6 +148,10 @@ class BaseDocument(Base):
|
|
|
151
148
|
chunk_index: Index of this chunk within the corpus
|
|
152
149
|
content: Text content of the document
|
|
153
150
|
embedding: Vector embedding of the content
|
|
151
|
+
embedding_config: Provider config used to generate the embedding. Should include
|
|
152
|
+
at minimum {"model": "<model_id>"} plus any provider-specific params
|
|
153
|
+
(e.g. {"model": "cohere.embed-v4:0", "input_type": "search_document"}).
|
|
154
|
+
Nullable — rows pre-dating this column will have NULL.
|
|
154
155
|
optional_props: Optional properties for the document
|
|
155
156
|
|
|
156
157
|
Returns:
|
|
@@ -165,13 +166,12 @@ class BaseDocument(Base):
|
|
|
165
166
|
chunk_index=chunk_index,
|
|
166
167
|
content=content,
|
|
167
168
|
embedding=embedding,
|
|
169
|
+
embedding_config=embedding_config,
|
|
168
170
|
title=optional_props.title,
|
|
169
|
-
document_metadata=metadata,
|
|
171
|
+
document_metadata=metadata or {},
|
|
170
172
|
collection=optional_props.collection,
|
|
171
173
|
origin_url=optional_props.original_url,
|
|
172
174
|
language=optional_props.language,
|
|
173
|
-
score=optional_props.score,
|
|
174
|
-
tags=optional_props.tags,
|
|
175
175
|
)
|
|
176
176
|
|
|
177
177
|
@classmethod
|
|
@@ -1,9 +1,22 @@
|
|
|
1
1
|
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import Any
|
|
2
3
|
|
|
3
4
|
|
|
4
5
|
class BaseEmbeddingProvider(ABC):
|
|
5
6
|
"""Abstract base for embedding generation"""
|
|
6
7
|
|
|
8
|
+
def __init__(self, model_id: str, **kwargs):
|
|
9
|
+
self.model_id = model_id
|
|
10
|
+
|
|
11
|
+
def get_embedding_config(self) -> dict[str, Any]:
|
|
12
|
+
"""
|
|
13
|
+
Return a JSON-serializable dict describing the full embedding configuration.
|
|
14
|
+
Stored in the `embedding_config` column on every document row.
|
|
15
|
+
Subclasses should override to include provider-specific params (e.g. input_type).
|
|
16
|
+
Default implementation returns just the model ID.
|
|
17
|
+
"""
|
|
18
|
+
return {"model": self.model_id}
|
|
19
|
+
|
|
7
20
|
@abstractmethod
|
|
8
21
|
def embed_text(self, text: str) -> list[float]:
|
|
9
22
|
"""Generate embedding vector for text"""
|
|
@@ -204,6 +204,7 @@ class BaseCorpusManager(ABC):
|
|
|
204
204
|
optional_props: BaseDocumentOptionalProps | None,
|
|
205
205
|
) -> list[BaseDocument]:
|
|
206
206
|
"""Create document instances from the provided data"""
|
|
207
|
+
embedding_config = self.embedding_provider.get_embedding_config()
|
|
207
208
|
documents = []
|
|
208
209
|
for i, (content, embedding) in enumerate(zip(document_contents, document_embeddings)):
|
|
209
210
|
chunk_md = self._extract_chunk_metadata(content)
|
|
@@ -214,6 +215,7 @@ class BaseCorpusManager(ABC):
|
|
|
214
215
|
chunk_index=i,
|
|
215
216
|
content=content,
|
|
216
217
|
embedding=embedding,
|
|
218
|
+
embedding_config=embedding_config,
|
|
217
219
|
metadata=base_metadata.model_dump(),
|
|
218
220
|
optional_props=optional_props,
|
|
219
221
|
)
|
|
@@ -69,38 +69,43 @@ class BaseSearchClient:
|
|
|
69
69
|
query: Search query containing text, metadata filters, and pagination.
|
|
70
70
|
|
|
71
71
|
Returns:
|
|
72
|
-
List of retrieval results matching the search criteria.
|
|
72
|
+
List of retrieval results matching the search criteria. Each result's `score`
|
|
73
|
+
is a float in [0, 1] when `query.text` is set (1.0 = most similar), or `None`
|
|
74
|
+
for keyword/metadata-only queries.
|
|
73
75
|
"""
|
|
74
|
-
|
|
76
|
+
distance_expr: ColumnElement | None = None
|
|
75
77
|
|
|
76
78
|
if query.text:
|
|
77
|
-
|
|
79
|
+
distance_expr = self._build_distance_expr(query)
|
|
80
|
+
db_query = select(self.config.document_cls, distance_expr).order_by(distance_expr)
|
|
81
|
+
else:
|
|
82
|
+
db_query = select(self.config.document_cls)
|
|
83
|
+
|
|
78
84
|
db_query = self._apply_keyword_search(db_query, query)
|
|
79
85
|
if query.metadata_filters:
|
|
80
86
|
db_query = self._apply_metadata_filters(db_query, query)
|
|
81
87
|
db_query = db_query.limit(query.limit)
|
|
82
88
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
89
|
+
if distance_expr is not None:
|
|
90
|
+
rows = self.session.execute(db_query).all()
|
|
91
|
+
return self._convert_to_retrieval_results(rows)
|
|
92
|
+
else:
|
|
93
|
+
results = self.session.scalars(db_query).all()
|
|
94
|
+
return self._convert_to_retrieval_results([(doc, None) for doc in results])
|
|
86
95
|
|
|
87
|
-
def
|
|
88
|
-
"""
|
|
89
|
-
`embedding_provider` must be provided at instantiation, or an `ValueError` will be raised.
|
|
90
|
-
In PGVector, `<=>` operator is used to compare cosine distance. Lower = more similar.
|
|
96
|
+
def _build_distance_expr(self, search_query: SearchQuery) -> ColumnElement:
|
|
97
|
+
"""Build a labeled cosine-distance expression for the given query text.
|
|
91
98
|
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
search_query: The search query containing the text to search for.
|
|
99
|
+
`embedding_provider` must be provided at instantiation, or a `ValueError` is raised.
|
|
100
|
+
In PGVector, `<=>` computes cosine distance (0 = identical, up to 2 = opposite).
|
|
95
101
|
|
|
96
102
|
Returns:
|
|
97
|
-
|
|
103
|
+
A SQLAlchemy labeled column expression suitable for SELECT and ORDER BY.
|
|
98
104
|
"""
|
|
99
|
-
|
|
100
|
-
return db_query
|
|
105
|
+
assert search_query.text is not None
|
|
101
106
|
query_embedding = self.embedding_provider.embed_text(search_query.text)
|
|
102
|
-
return
|
|
103
|
-
|
|
107
|
+
return self.config.document_cls.embedding.cosine_distance(query_embedding).label(
|
|
108
|
+
"_score_distance"
|
|
104
109
|
)
|
|
105
110
|
|
|
106
111
|
def _apply_keyword_search(self, db_query: Select, search_query: SearchQuery) -> Select:
|
|
@@ -199,16 +204,18 @@ class BaseSearchClient:
|
|
|
199
204
|
raise ValueError(f"Unsupported condition: {filter_obj.condition}")
|
|
200
205
|
|
|
201
206
|
def _convert_to_retrieval_results(self, results: Sequence[Any]) -> list[RetrievalResult]:
|
|
202
|
-
"""Convert
|
|
207
|
+
"""Convert (document, distance | None) pairs to RetrievalResult objects.
|
|
203
208
|
|
|
204
209
|
Args:
|
|
205
|
-
results:
|
|
206
|
-
|
|
210
|
+
results: Sequence of (document, cosine_distance) tuples. Pass distance=None
|
|
211
|
+
for keyword/metadata-only queries that have no semantic ranking.
|
|
207
212
|
|
|
208
213
|
Returns:
|
|
209
|
-
List of RetrievalResult objects.
|
|
214
|
+
List of RetrievalResult objects with score = max(0.0, 1.0 - distance),
|
|
215
|
+
or score=None when distance is None.
|
|
210
216
|
"""
|
|
211
217
|
retrieval_results = []
|
|
212
|
-
for
|
|
213
|
-
|
|
218
|
+
for document, distance in results:
|
|
219
|
+
score = max(0.0, 1.0 - distance / 2.0) if distance is not None else None
|
|
220
|
+
retrieval_results.append(RetrievalResult(document=document, score=score))
|
|
214
221
|
return retrieval_results
|
|
@@ -1,5 +1,4 @@
|
|
|
1
1
|
from dataclasses import dataclass, asdict
|
|
2
|
-
from datetime import datetime
|
|
3
2
|
from typing import Any, Literal
|
|
4
3
|
|
|
5
4
|
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
@@ -38,7 +37,13 @@ class SearchQuery(BaseModel):
|
|
|
38
37
|
text: str | None = None
|
|
39
38
|
"""String to match against using in a semantic search, i.e. using vector distance."""
|
|
40
39
|
keywords: list[str] = []
|
|
41
|
-
"""
|
|
40
|
+
"""
|
|
41
|
+
List of keywords for a case-insensitive **substring** search (SQL `ILIKE '%keyword%'`).
|
|
42
|
+
An entry matches if **any** keyword appears anywhere in its content (OR-combined).
|
|
43
|
+
Pass multiple variants, synonyms, or spellings to increase recall — e.g.
|
|
44
|
+
["run", "running", "ran"] will match entries containing any of those forms.
|
|
45
|
+
Note: substring matching means "cat" also matches "category"; prefer distinctive stems.
|
|
46
|
+
"""
|
|
42
47
|
metadata_filters: list[MetadataFilter] = Field(
|
|
43
48
|
default=[],
|
|
44
49
|
json_schema_extra={"metadata_schema": BaseDocumentMetadata.model_json_schema()},
|
|
@@ -47,8 +52,6 @@ class SearchQuery(BaseModel):
|
|
|
47
52
|
List of metadata conditions that must be matched.
|
|
48
53
|
Refer to `metadata_schema` for the expected schema, as it exists in the database.
|
|
49
54
|
"""
|
|
50
|
-
date_range: tuple[datetime, datetime] | None = None
|
|
51
|
-
"""Retrieve/limit results based on created_at & updated_at timestamps (i.e. database operations)"""
|
|
52
55
|
limit: int = Field(
|
|
53
56
|
...,
|
|
54
57
|
ge=1,
|
|
@@ -62,7 +65,7 @@ class SearchQuery(BaseModel):
|
|
|
62
65
|
|
|
63
66
|
@model_validator(mode="after")
|
|
64
67
|
def ensure_criterion(self):
|
|
65
|
-
if not any([self.text, self.keywords, self.metadata_filters
|
|
68
|
+
if not any([self.text, self.keywords, self.metadata_filters]):
|
|
66
69
|
raise ValueError("At least one search criterion is required")
|
|
67
70
|
return self
|
|
68
71
|
|
|
@@ -72,7 +75,12 @@ class RetrievalResult:
|
|
|
72
75
|
"""Standardized result structure for all retrieval operations"""
|
|
73
76
|
|
|
74
77
|
document: BaseDocument
|
|
75
|
-
score: float
|
|
78
|
+
score: float | None
|
|
79
|
+
"""
|
|
80
|
+
Semantic similarity score in [0, 1] when a `text` query was provided (1.0 = identical,
|
|
81
|
+
0.0 = maximally distant). `None` when the search had no semantic component (keywords /
|
|
82
|
+
metadata filters only) — results are still filtered correctly but have no meaningful ranking.
|
|
83
|
+
"""
|
|
76
84
|
|
|
77
85
|
def to_dict(self) -> dict[str, Any]:
|
|
78
86
|
return asdict(self)
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pgvector-template"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.5"
|
|
8
8
|
description = "Template library for flexible PGVector RAG implementations"
|
|
9
9
|
authors = [{ name="DL", email="v49t9zpqd@mozmail.com" }]
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -52,4 +52,4 @@ line-length = 100
|
|
|
52
52
|
target-version = "py311"
|
|
53
53
|
|
|
54
54
|
[tool.ty.src]
|
|
55
|
-
exclude = []
|
|
55
|
+
exclude = ["integ-tests"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/service/document_service.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template/utils/metadata_filter.py
RENAMED
|
File without changes
|
|
File without changes
|
{pgvector_template-0.3.5 → pgvector_template-0.5}/pgvector_template.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|