pgvector-template 0.3.5__tar.gz → 0.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pgvector_template-0.3.5 → pgvector_template-0.4}/PKG-INFO +1 -1
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/document.py +30 -30
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/embedder.py +13 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/manager.py +2 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/PKG-INFO +1 -1
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pyproject.toml +2 -2
- {pgvector_template-0.3.5 → pgvector_template-0.4}/LICENSE +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/README.md +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/core/search.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/db/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/db/connection.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/db/document_db.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/models/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/models/search.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/service/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/service/document_service.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/types.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/utils/__init__.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/utils/metadata_filter.py +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/SOURCES.txt +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/dependency_links.txt +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/requires.txt +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/top_level.txt +0 -0
- {pgvector_template-0.3.5 → pgvector_template-0.4}/setup.cfg +0 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
from datetime import datetime
|
|
1
|
+
from datetime import datetime, timezone
|
|
2
2
|
from typing import Any, Type, TypeVar
|
|
3
3
|
from uuid import uuid4, UUID as UuidLiteral
|
|
4
4
|
|
|
5
|
-
from pydantic import BaseModel, Field
|
|
5
|
+
from pydantic import BaseModel, Field
|
|
6
6
|
from sqlalchemy import (
|
|
7
7
|
Column,
|
|
8
8
|
String,
|
|
@@ -10,7 +10,6 @@ from sqlalchemy import (
|
|
|
10
10
|
DateTime,
|
|
11
11
|
Boolean,
|
|
12
12
|
Integer,
|
|
13
|
-
Float,
|
|
14
13
|
Index,
|
|
15
14
|
UniqueConstraint,
|
|
16
15
|
)
|
|
@@ -32,21 +31,6 @@ class BaseDocumentOptionalProps(BaseModel):
|
|
|
32
31
|
"""Optional source URL for the document"""
|
|
33
32
|
language: str | None = Field(default="en", pattern=r"^[a-z]{2}(-[A-Z]{2})?$")
|
|
34
33
|
"""Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'"""
|
|
35
|
-
score: float | None = Field(default=None, ge=0.0, le=1.0)
|
|
36
|
-
"""Optional score assigned during ingestion (e.g., relevance, confidence)"""
|
|
37
|
-
tags: list[str] | None = None
|
|
38
|
-
"""List of tags or keywords for filtering, categorization, or faceted search"""
|
|
39
|
-
|
|
40
|
-
@field_validator("tags")
|
|
41
|
-
@classmethod
|
|
42
|
-
def validate_tags(cls, v):
|
|
43
|
-
if v is not None:
|
|
44
|
-
# Ensure all tags are strings and not empty
|
|
45
|
-
if not all(isinstance(tag, str) and tag.strip() for tag in v):
|
|
46
|
-
raise ValueError("All tags must be non-empty strings")
|
|
47
|
-
# Remove duplicates while preserving order
|
|
48
|
-
return list(dict.fromkeys(v))
|
|
49
|
-
return v
|
|
50
34
|
|
|
51
35
|
|
|
52
36
|
T = TypeVar("T", bound="BaseDocument")
|
|
@@ -88,18 +72,29 @@ class BaseDocument(Base):
|
|
|
88
72
|
"""Optional source URL"""
|
|
89
73
|
language = Column(String(2), default="en")
|
|
90
74
|
"""Language of the content (ISO 639-1 code), e.g., 'en', 'es', 'zh'."""
|
|
91
|
-
score = Column(Float, nullable=True)
|
|
92
|
-
"""Optional score assigned during ingestion (e.g., relevance, confidence)."""
|
|
93
|
-
tags = Column(JSONB, nullable=True, default=list)
|
|
94
|
-
"""List of tags or keywords for filtering, categorization, or faceted search."""
|
|
95
75
|
|
|
96
76
|
# Vector embedding
|
|
97
77
|
embedding = Column(Vector(1024))
|
|
98
78
|
"""Embedding vector. 1024 dimensions by default. Adjust as-needed."""
|
|
79
|
+
embedding_config = Column(JSONB, nullable=True)
|
|
80
|
+
"""
|
|
81
|
+
Configuration used to generate the embedding vector. Stored as a JSON object so it
|
|
82
|
+
can capture provider-specific parameters beyond just the model ID — e.g. input_type,
|
|
83
|
+
dimensions, truncation strategy, or any future provider-specific options.
|
|
84
|
+
|
|
85
|
+
Recommended shape: {"model": "cohere.embed-v4:0", "input_type": "search_document", ...}
|
|
86
|
+
|
|
87
|
+
Intentionally nullable: rows embedded before this column was added will have NULL here.
|
|
88
|
+
Use this field for auditing and diagnosing mixed-model corpora, not for runtime dispatch.
|
|
89
|
+
"""
|
|
99
90
|
|
|
100
91
|
# Audit fields
|
|
101
|
-
created_at = Column(DateTime, default=datetime.
|
|
102
|
-
updated_at = Column(
|
|
92
|
+
created_at = Column(DateTime(timezone=True), default=lambda: datetime.now(timezone.utc))
|
|
93
|
+
updated_at = Column(
|
|
94
|
+
DateTime(timezone=True),
|
|
95
|
+
default=lambda: datetime.now(timezone.utc),
|
|
96
|
+
onupdate=lambda: datetime.now(timezone.utc),
|
|
97
|
+
)
|
|
103
98
|
is_deleted = Column(Boolean, default=False, index=True)
|
|
104
99
|
"""Entries can be logically marked for deletion before they are permanently deleted."""
|
|
105
100
|
|
|
@@ -112,10 +107,11 @@ class BaseDocument(Base):
|
|
|
112
107
|
|
|
113
108
|
base_table_args = (
|
|
114
109
|
Index(
|
|
115
|
-
f"{table_name}_metadata_gin_idx",
|
|
110
|
+
f"{table_name}_metadata_gin_idx",
|
|
111
|
+
"document_metadata",
|
|
112
|
+
postgresql_using="gin",
|
|
116
113
|
),
|
|
117
114
|
cls.get_embedding_index(table_name),
|
|
118
|
-
Index(f"{table_name}_tags_gin_idx", "tags", postgresql_using="gin"),
|
|
119
115
|
Index(f"{table_name}_collection_corpus_idx", "collection", "corpus_id"),
|
|
120
116
|
UniqueConstraint(
|
|
121
117
|
"collection",
|
|
@@ -140,7 +136,8 @@ class BaseDocument(Base):
|
|
|
140
136
|
chunk_index: int,
|
|
141
137
|
content: str,
|
|
142
138
|
embedding: list[float],
|
|
143
|
-
|
|
139
|
+
embedding_config: dict[str, Any] | None = None,
|
|
140
|
+
metadata: dict[str, Any] | None = None,
|
|
144
141
|
optional_props: BaseDocumentOptionalProps | None = None,
|
|
145
142
|
) -> T:
|
|
146
143
|
"""
|
|
@@ -151,6 +148,10 @@ class BaseDocument(Base):
|
|
|
151
148
|
chunk_index: Index of this chunk within the corpus
|
|
152
149
|
content: Text content of the document
|
|
153
150
|
embedding: Vector embedding of the content
|
|
151
|
+
embedding_config: Provider config used to generate the embedding. Should include
|
|
152
|
+
at minimum {"model": "<model_id>"} plus any provider-specific params
|
|
153
|
+
(e.g. {"model": "cohere.embed-v4:0", "input_type": "search_document"}).
|
|
154
|
+
Nullable — rows pre-dating this column will have NULL.
|
|
154
155
|
optional_props: Optional properties for the document
|
|
155
156
|
|
|
156
157
|
Returns:
|
|
@@ -165,13 +166,12 @@ class BaseDocument(Base):
|
|
|
165
166
|
chunk_index=chunk_index,
|
|
166
167
|
content=content,
|
|
167
168
|
embedding=embedding,
|
|
169
|
+
embedding_config=embedding_config,
|
|
168
170
|
title=optional_props.title,
|
|
169
|
-
document_metadata=metadata,
|
|
171
|
+
document_metadata=metadata or {},
|
|
170
172
|
collection=optional_props.collection,
|
|
171
173
|
origin_url=optional_props.original_url,
|
|
172
174
|
language=optional_props.language,
|
|
173
|
-
score=optional_props.score,
|
|
174
|
-
tags=optional_props.tags,
|
|
175
175
|
)
|
|
176
176
|
|
|
177
177
|
@classmethod
|
|
@@ -1,9 +1,22 @@
|
|
|
1
1
|
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import Any
|
|
2
3
|
|
|
3
4
|
|
|
4
5
|
class BaseEmbeddingProvider(ABC):
|
|
5
6
|
"""Abstract base for embedding generation"""
|
|
6
7
|
|
|
8
|
+
def __init__(self, model_id: str, **kwargs):
|
|
9
|
+
self.model_id = model_id
|
|
10
|
+
|
|
11
|
+
def get_embedding_config(self) -> dict[str, Any]:
|
|
12
|
+
"""
|
|
13
|
+
Return a JSON-serializable dict describing the full embedding configuration.
|
|
14
|
+
Stored in the `embedding_config` column on every document row.
|
|
15
|
+
Subclasses should override to include provider-specific params (e.g. input_type).
|
|
16
|
+
Default implementation returns just the model ID.
|
|
17
|
+
"""
|
|
18
|
+
return {"model": self.model_id}
|
|
19
|
+
|
|
7
20
|
@abstractmethod
|
|
8
21
|
def embed_text(self, text: str) -> list[float]:
|
|
9
22
|
"""Generate embedding vector for text"""
|
|
@@ -204,6 +204,7 @@ class BaseCorpusManager(ABC):
|
|
|
204
204
|
optional_props: BaseDocumentOptionalProps | None,
|
|
205
205
|
) -> list[BaseDocument]:
|
|
206
206
|
"""Create document instances from the provided data"""
|
|
207
|
+
embedding_config = self.embedding_provider.get_embedding_config()
|
|
207
208
|
documents = []
|
|
208
209
|
for i, (content, embedding) in enumerate(zip(document_contents, document_embeddings)):
|
|
209
210
|
chunk_md = self._extract_chunk_metadata(content)
|
|
@@ -214,6 +215,7 @@ class BaseCorpusManager(ABC):
|
|
|
214
215
|
chunk_index=i,
|
|
215
216
|
content=content,
|
|
216
217
|
embedding=embedding,
|
|
218
|
+
embedding_config=embedding_config,
|
|
217
219
|
metadata=base_metadata.model_dump(),
|
|
218
220
|
optional_props=optional_props,
|
|
219
221
|
)
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pgvector-template"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.4"
|
|
8
8
|
description = "Template library for flexible PGVector RAG implementations"
|
|
9
9
|
authors = [{ name="DL", email="v49t9zpqd@mozmail.com" }]
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -52,4 +52,4 @@ line-length = 100
|
|
|
52
52
|
target-version = "py311"
|
|
53
53
|
|
|
54
54
|
[tool.ty.src]
|
|
55
|
-
exclude = []
|
|
55
|
+
exclude = ["integ-tests"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/service/document_service.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template/utils/metadata_filter.py
RENAMED
|
File without changes
|
|
File without changes
|
{pgvector_template-0.3.5 → pgvector_template-0.4}/pgvector_template.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|