raglite-toolkit 1.2.0__tar.gz → 1.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/.github/workflows/pr-verify.yml +9 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/.gitignore +1 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/ARCHITECTURE.md +15 -10
- raglite_toolkit-1.2.2/CHANGELOG.md +76 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/PKG-INFO +7 -6
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/README.md +5 -4
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/pyproject.toml +1 -1
- raglite_toolkit-1.2.2/scripts/sync_shared_fixtures.py +54 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/chunking/recursive.py +9 -2
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/config.py +5 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/constants.py +10 -1
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/core/collection.py +21 -3
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/core/document.py +45 -23
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/web.py +2 -1
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/types.py +3 -1
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/pinecone.py +5 -1
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/qdrant.py +32 -9
- raglite_toolkit-1.2.2/tests/fixtures/shared/chunker.json +100 -0
- raglite_toolkit-1.2.2/tests/fixtures/shared/hash.json +34 -0
- raglite_toolkit-1.2.2/tests/unit/test_regressions.py +254 -0
- raglite_toolkit-1.2.2/tests/unit/test_shared_fixtures.py +42 -0
- raglite_toolkit-1.2.0/CHANGELOG.md +0 -46
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/.github/workflows/publish.yml +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/LICENSE +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/basic.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/custom_store_example.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/multi_provider.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/ollama_test.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/qdrant_example.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/sample.txt +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/serve.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/scripts/pre-commit.sh +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/scripts/pre-release.sh +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/api/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/api/schemas.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/api/server.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/chunking/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/chunking/base.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/cli.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/core/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/base.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/factory.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/local.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/models.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/remote.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/errors.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/answer.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/factory.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/models.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/prompt.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/base.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/directory.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/docx.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/json.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/markdown.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/pdf.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/txt.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/retrieval/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/retrieval/retriever.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/utils/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/utils/hash.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/utils/logger.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/base.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/factory.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/memory.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/conftest.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_api.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_ask.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_collection.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_document.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_ollama.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/__init__.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_chunking.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_cli.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_config.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_directory_loader.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_errors.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_hash.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_loaders.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_prompt.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_retriever.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_vectordb.py +0 -0
- {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_web_loader.py +0 -0
|
@@ -34,3 +34,12 @@ jobs:
|
|
|
34
34
|
|
|
35
35
|
- name: Run tests
|
|
36
36
|
run: python -m pytest tests/ -q
|
|
37
|
+
|
|
38
|
+
- name: Check out TypeScript SDK for shared fixtures
|
|
39
|
+
uses: actions/checkout@v4
|
|
40
|
+
with:
|
|
41
|
+
repository: creatorpiyush/raglite
|
|
42
|
+
path: .ts-sdk
|
|
43
|
+
|
|
44
|
+
- name: Check shared fixtures match the TypeScript SDK
|
|
45
|
+
run: python scripts/sync_shared_fixtures.py --check --source .ts-sdk
|
|
@@ -126,9 +126,9 @@ graph TD
|
|
|
126
126
|
### 4.1 Ingestion & Indexing
|
|
127
127
|
1. `doc.build()` invokes the appropriate `BaseLoader` based on file extension (`.pdf`, `.txt`, `.json`, `.md`, `.docx`).
|
|
128
128
|
2. Calculates SHA-256 hash of raw document content.
|
|
129
|
-
3. Checks existing `IndexMetadata` in `VectorStore`.
|
|
129
|
+
3. Checks existing `IndexMetadata` in `VectorStore`. The cached index is reused when the index format version, content hash, chunk size, overlap and embedding provider/model all match (URL sources are hashed by their fetched text).
|
|
130
130
|
4. If hash differs or force rebuild requested:
|
|
131
|
-
- `
|
|
131
|
+
- `RecursiveChunker` splits text into word-based chunks (default 500 words, 50-word overlap).
|
|
132
132
|
- `EmbeddingFactory` generates normalized vectors for each chunk.
|
|
133
133
|
- `VectorStore.add()` saves chunks and `VectorStore.save_index_metadata()` persists index metadata.
|
|
134
134
|
|
|
@@ -162,27 +162,32 @@ Built using **FastAPI** framework for async capabilities, automatic OpenAPI docs
|
|
|
162
162
|
```python
|
|
163
163
|
from abc import ABC, abstractmethod
|
|
164
164
|
from typing import List, Optional
|
|
165
|
-
from raglite.types import IndexMetadata,
|
|
165
|
+
from raglite.types import IndexMetadata, StoredChunk
|
|
166
|
+
from raglite.vectordb.base import VectorSearchHit
|
|
166
167
|
|
|
167
168
|
class VectorStore(ABC):
|
|
169
|
+
@property
|
|
168
170
|
@abstractmethod
|
|
169
|
-
def
|
|
171
|
+
def namespace(self) -> str: ...
|
|
170
172
|
|
|
171
173
|
@abstractmethod
|
|
172
|
-
def
|
|
174
|
+
def load(self) -> None: ...
|
|
173
175
|
|
|
174
176
|
@abstractmethod
|
|
175
|
-
def
|
|
177
|
+
def reset(self) -> None: ...
|
|
176
178
|
|
|
177
179
|
@abstractmethod
|
|
178
|
-
def
|
|
180
|
+
def add(self, chunks: List[StoredChunk]) -> None: ...
|
|
179
181
|
|
|
180
182
|
@abstractmethod
|
|
181
|
-
def
|
|
183
|
+
def search(self, embedding: List[float], top_k: int) -> List[VectorSearchHit]: ...
|
|
182
184
|
|
|
183
185
|
@abstractmethod
|
|
184
|
-
def
|
|
186
|
+
def count(self) -> int: ...
|
|
185
187
|
|
|
186
188
|
@abstractmethod
|
|
187
|
-
def
|
|
189
|
+
def save_index_metadata(self, metadata: IndexMetadata) -> None: ...
|
|
190
|
+
|
|
191
|
+
@abstractmethod
|
|
192
|
+
def read_index_metadata(self) -> Optional[IndexMetadata]: ...
|
|
188
193
|
```
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project will be documented in this file.
|
|
4
|
+
|
|
5
|
+
## [1.2.2] - Unreleased
|
|
6
|
+
|
|
7
|
+
### Changed
|
|
8
|
+
- **Upgrades keep cached indexes:** The build cache is now keyed on an index format version (`formatVersion` in `IndexMetadata`) instead of the package version, so upgrading RAGLite no longer re-embeds every index. Indexes built by 1.2.1 are reused as-is; indexes from older releases are rebuilt once.
|
|
9
|
+
|
|
10
|
+
### Fixed
|
|
11
|
+
- **Chunker parity with the TypeScript SDK:** Words are now split on exactly the whitespace characters JavaScript treats as whitespace. Previously text containing a byte order mark (U+FEFF), U+0085 or U+001C–U+001F was chunked differently from the TypeScript SDK. Existing indexes are not rebuilt for this; pass `rebuild=True` if your sources contain those characters.
|
|
12
|
+
- **Docs:** ARCHITECTURE.md now describes the word-based `RecursiveChunker` (500 words, 50 overlap) and the actual `VectorStore` interface. The README no longer lists LanceDB, which is TypeScript-only.
|
|
13
|
+
|
|
14
|
+
### Internal
|
|
15
|
+
- Added cross-SDK fixtures (`tests/fixtures/shared/`, copied from the TypeScript SDK with `scripts/sync_shared_fixtures.py`) that pin chunking and hashing output. CI fails if the copy is out of date.
|
|
16
|
+
|
|
17
|
+
## [1.2.1] - 2026-10-02
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
- **URL sources:** `Document.build()` no longer fails with `LoaderError: File does not exist` for web URLs, so URLs work on their own and inside a `DocumentCollection`.
|
|
21
|
+
- **Custom `VectorStore` instances:** Passing a `VectorStore` subclass instance as `{"vectorStore": store}` no longer fails config validation, so the documented custom store usage works.
|
|
22
|
+
- **Qdrant shared collections:** When `indexName` is set, documents sharing one Qdrant collection no longer wipe each other. `reset()`, search and index metadata are now scoped to each document's namespace instead of the whole collection. Without `indexName`, behaviour is unchanged.
|
|
23
|
+
- **Shared `VectorStore` instances in collections:** `DocumentCollection.build()` now raises `ConfigError` when a single `VectorStore` instance would be shared by more than one document, instead of each document silently resetting the previous one's index. Pass a vector store provider config instead.
|
|
24
|
+
- **Web URL re-indexing:** URL sources are now fingerprinted by their fetched content rather than the URL string, so a changed page is re-indexed on the next `build()`.
|
|
25
|
+
- **Query embedder after reload:** Searching an existing index in a new process now embeds queries with the provider and model the index was built with, rather than the constructor default. Configured credentials are reused when the provider matches.
|
|
26
|
+
- **Collection search errors:** `DocumentCollection.search()` (and `ask`/`ask_stream`) now logs per-document search failures instead of silently dropping them, and raises if every document fails.
|
|
27
|
+
- **Web loader User-Agent:** Now reports the actual package version.
|
|
28
|
+
|
|
29
|
+
### Upgrade notes
|
|
30
|
+
- As with every release, cached indexes are rebuilt once on first `build()` because the package version is part of the cache key.
|
|
31
|
+
- Existing Qdrant indexes created with `indexName` are rebuilt once. Their old untagged points stay in the collection but are no longer returned by search; drop and recreate the collection to remove them.
|
|
32
|
+
- Each `build()` on a URL source now fetches the page to check for changes, even when the cached index is reused.
|
|
33
|
+
- When every document in a collection fails to search, the REST server now returns an error response (400 for RAGLite errors, 500 otherwise) instead of empty results.
|
|
34
|
+
|
|
35
|
+
## [1.2.0] - 2026-08-02
|
|
36
|
+
|
|
37
|
+
### Added
|
|
38
|
+
- **Multi-Document & Directory Ingestion (`DocumentCollection`):**
|
|
39
|
+
- Added `DocumentCollection` class to manage semantic indexing, multi-document retrieval, and Q&A across folders, glob patterns, web URLs, and mixed file lists.
|
|
40
|
+
- Parallel semantic search over collection vector stores with score-based top-$K$ merging and ranking.
|
|
41
|
+
- Contextual Q&A synthesis (`ask` and `ask_stream`) across multi-document collections.
|
|
42
|
+
- **Directory Loader (`DirectoryLoader`):**
|
|
43
|
+
- Recursive directory scanner (`recursive=True`) with glob pattern matching (e.g. `./docs/**/*.md`).
|
|
44
|
+
- Auto-detection of supported extensions (`.pdf`, `.txt`, `.md`, `.json`, `.docx`).
|
|
45
|
+
- Detailed error reporting and warning logs for unsupported/empty files.
|
|
46
|
+
- **Web Loader (`WebLoader`):**
|
|
47
|
+
- Native loader for fetching HTTP/HTTPS web URLs directly.
|
|
48
|
+
- Automatic HTML cleaning into formatted text/markdown with script, style, and SVG tag stripping.
|
|
49
|
+
- JSON and plain text content-type parsing.
|
|
50
|
+
- **CLI & FastAPI REST Server Support:**
|
|
51
|
+
- Upgraded `raglite index`, `search`, `ask`, and `serve` CLI commands to process directories, glob patterns, and URLs.
|
|
52
|
+
- Updated FastAPI REST server to support `DocumentCollection` and single `Document` targets.
|
|
53
|
+
- **Unit & Integration Tests:**
|
|
54
|
+
- Added unit and integration tests for `DirectoryLoader`, `WebLoader`, and `DocumentCollection`.
|
|
55
|
+
|
|
56
|
+
## [1.1.0] - 2026-07-19
|
|
57
|
+
|
|
58
|
+
### Added
|
|
59
|
+
- **Pluggable Vector Databases:** Added support for custom local and cloud vector database backends via a new `vectorStore` config option.
|
|
60
|
+
- **Memory Store (`"memory"`):** Default in-memory store persisting indexes locally to JSON (unchanged behaviour).
|
|
61
|
+
- **Qdrant Store (`"qdrant"`):** Wrapper for Qdrant local/cloud using stdlib `urllib` REST requests. Supports auto-collection creation, API key auth, and custom collection names.
|
|
62
|
+
- **Pinecone Store (`"pinecone"`):** Cloud database support using Pinecone Namespaces and stdlib `urllib` REST requests. Stores index metadata as a reserved `__metadata__` vector.
|
|
63
|
+
- **Custom Adapters:** Pass any class instance implementing the `VectorStore` ABC directly as `vectorStore` in `DocumentOptions`.
|
|
64
|
+
- **`VectorStoreProviderConfig` type:** New Pydantic model in `types.py` describing provider, URL, API key, index name, and store directory.
|
|
65
|
+
- **Factory:** `create_vector_store(config, namespace)` utility in `vectordb/factory.py` resolving the correct store from config.
|
|
66
|
+
- **Examples:**
|
|
67
|
+
- `examples/qdrant_example.py` — full index + search demo with Qdrant.
|
|
68
|
+
- `examples/custom_store_example.py` — implementing and using a custom VectorStore ABC subclass.
|
|
69
|
+
- **Automation Scripts:**
|
|
70
|
+
- `scripts/pre-commit.sh` — runs ruff lint, mypy type-check, and pytest before committing.
|
|
71
|
+
- `scripts/pre-release.sh` — cleans builds, runs full verification, and builds distribution packages.
|
|
72
|
+
- **GitHub Actions Workflow:** `pr-verify.yml` — automatically runs code style checks and the full test suite on every pull request and push to `main`/`master`.
|
|
73
|
+
|
|
74
|
+
### Changed
|
|
75
|
+
- **`DocumentOptions` / `ResolvedConfig`:** Added optional `vectorStore` field supporting `VectorStoreProviderConfig` or a custom `VectorStore` instance.
|
|
76
|
+
- **`Document.__init__`:** Constructor now resolves the appropriate vector store from config, accepting provider configs or custom instances.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: raglite-toolkit
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python.
|
|
5
5
|
Project-URL: Homepage, https://github.com/creatorpiyush/raglite-py
|
|
6
6
|
Project-URL: Repository, https://github.com/creatorpiyush/raglite-py
|
|
@@ -172,7 +172,7 @@ print()
|
|
|
172
172
|
|
|
173
173
|
## Pluggable Vector Databases
|
|
174
174
|
|
|
175
|
-
`raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone,
|
|
175
|
+
`raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, or custom subclasses; LanceDB is currently TypeScript-only):
|
|
176
176
|
|
|
177
177
|
### Memory Store (Default)
|
|
178
178
|
```python
|
|
@@ -182,6 +182,7 @@ doc = Document("./policy.pdf", {
|
|
|
182
182
|
```
|
|
183
183
|
|
|
184
184
|
### Qdrant Store
|
|
185
|
+
By default each document gets its own collection (`raglite_<namespace>`). Set `indexName` to keep several documents in one shared collection; each document's points are tagged with its namespace, so rebuilding one document never affects the others.
|
|
185
186
|
```python
|
|
186
187
|
doc = Document("./policy.pdf", {
|
|
187
188
|
"vectorStore": {
|
|
@@ -335,7 +336,7 @@ Document("./policy.pdf", {
|
|
|
335
336
|
|
|
336
337
|
## How Caching Works
|
|
337
338
|
|
|
338
|
-
Every `build()` call fingerprints the source file with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
|
|
339
|
+
Every `build()` call fingerprints the source (file bytes, or the fetched text for URLs) with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
|
|
339
340
|
|
|
340
341
|
| Factor | Triggers rebuild if changed |
|
|
341
342
|
|--------|-----------------------------|
|
|
@@ -343,7 +344,7 @@ Every `build()` call fingerprints the source file with a **SHA-256 content hash*
|
|
|
343
344
|
| Chunk size | `chunkSize` changed |
|
|
344
345
|
| Overlap | `overlap` changed |
|
|
345
346
|
| Embedding provider/model | Provider or model string changed |
|
|
346
|
-
|
|
|
347
|
+
| Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
|
|
347
348
|
|
|
348
349
|
Pass `rebuild=True` to `build()` to force a fresh index regardless.
|
|
349
350
|
|
|
@@ -359,7 +360,7 @@ Each document is stored under `.raglite/<sha256-prefix>/`, so multiple documents
|
|
|
359
360
|
from raglite.vectordb.base import VectorStore
|
|
360
361
|
|
|
361
362
|
class MyVectorStore(VectorStore):
|
|
362
|
-
# Implement: load, reset, add, search, count,
|
|
363
|
+
# Implement: namespace (property), load, reset, add, search, count,
|
|
363
364
|
# save_index_metadata, read_index_metadata
|
|
364
365
|
...
|
|
365
366
|
```
|
|
@@ -126,7 +126,7 @@ print()
|
|
|
126
126
|
|
|
127
127
|
## Pluggable Vector Databases
|
|
128
128
|
|
|
129
|
-
`raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone,
|
|
129
|
+
`raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, or custom subclasses; LanceDB is currently TypeScript-only):
|
|
130
130
|
|
|
131
131
|
### Memory Store (Default)
|
|
132
132
|
```python
|
|
@@ -136,6 +136,7 @@ doc = Document("./policy.pdf", {
|
|
|
136
136
|
```
|
|
137
137
|
|
|
138
138
|
### Qdrant Store
|
|
139
|
+
By default each document gets its own collection (`raglite_<namespace>`). Set `indexName` to keep several documents in one shared collection; each document's points are tagged with its namespace, so rebuilding one document never affects the others.
|
|
139
140
|
```python
|
|
140
141
|
doc = Document("./policy.pdf", {
|
|
141
142
|
"vectorStore": {
|
|
@@ -289,7 +290,7 @@ Document("./policy.pdf", {
|
|
|
289
290
|
|
|
290
291
|
## How Caching Works
|
|
291
292
|
|
|
292
|
-
Every `build()` call fingerprints the source file with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
|
|
293
|
+
Every `build()` call fingerprints the source (file bytes, or the fetched text for URLs) with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
|
|
293
294
|
|
|
294
295
|
| Factor | Triggers rebuild if changed |
|
|
295
296
|
|--------|-----------------------------|
|
|
@@ -297,7 +298,7 @@ Every `build()` call fingerprints the source file with a **SHA-256 content hash*
|
|
|
297
298
|
| Chunk size | `chunkSize` changed |
|
|
298
299
|
| Overlap | `overlap` changed |
|
|
299
300
|
| Embedding provider/model | Provider or model string changed |
|
|
300
|
-
|
|
|
301
|
+
| Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
|
|
301
302
|
|
|
302
303
|
Pass `rebuild=True` to `build()` to force a fresh index regardless.
|
|
303
304
|
|
|
@@ -313,7 +314,7 @@ Each document is stored under `.raglite/<sha256-prefix>/`, so multiple documents
|
|
|
313
314
|
from raglite.vectordb.base import VectorStore
|
|
314
315
|
|
|
315
316
|
class MyVectorStore(VectorStore):
|
|
316
|
-
# Implement: load, reset, add, search, count,
|
|
317
|
+
# Implement: namespace (property), load, reset, add, search, count,
|
|
317
318
|
# save_index_metadata, read_index_metadata
|
|
318
319
|
...
|
|
319
320
|
```
|
|
@@ -8,7 +8,7 @@ packages = ["src/raglite"]
|
|
|
8
8
|
|
|
9
9
|
[project]
|
|
10
10
|
name = "raglite-toolkit"
|
|
11
|
-
version = "1.2.
|
|
11
|
+
version = "1.2.2"
|
|
12
12
|
description = "Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python."
|
|
13
13
|
readme = "README.md"
|
|
14
14
|
requires-python = ">=3.10"
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Copy or check the cross-SDK fixtures shared with the TypeScript SDK.
|
|
2
|
+
|
|
3
|
+
The TypeScript repo (raglite) is the source of truth: it generates
|
|
4
|
+
tests/fixtures/shared/*.json with scripts/generate-shared-fixtures.ts. This
|
|
5
|
+
repo keeps an identical copy so both SDKs are tested against the same
|
|
6
|
+
expected outputs.
|
|
7
|
+
|
|
8
|
+
python scripts/sync_shared_fixtures.py # copy from ../raglite
|
|
9
|
+
python scripts/sync_shared_fixtures.py --check # exit 1 if out of date
|
|
10
|
+
python scripts/sync_shared_fixtures.py --source /path/to/raglite
|
|
11
|
+
"""
|
|
12
|
+
import argparse
|
|
13
|
+
import filecmp
|
|
14
|
+
import shutil
|
|
15
|
+
import sys
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
REPO_ROOT = Path(__file__).resolve().parent.parent
|
|
19
|
+
TARGET = REPO_ROOT / "tests" / "fixtures" / "shared"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def main() -> int:
|
|
23
|
+
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
24
|
+
parser.add_argument("--source", type=Path, default=REPO_ROOT.parent / "raglite")
|
|
25
|
+
parser.add_argument("--check", action="store_true")
|
|
26
|
+
args = parser.parse_args()
|
|
27
|
+
|
|
28
|
+
source = args.source / "tests" / "fixtures" / "shared"
|
|
29
|
+
files = sorted(source.glob("*.json"))
|
|
30
|
+
if not files:
|
|
31
|
+
print(f"No shared fixtures found in {source}", file=sys.stderr)
|
|
32
|
+
return 1
|
|
33
|
+
|
|
34
|
+
stale = [f.name for f in files if not (TARGET / f.name).exists() or not filecmp.cmp(f, TARGET / f.name, shallow=False)]
|
|
35
|
+
extra = sorted({p.name for p in TARGET.glob("*.json")} - {f.name for f in files})
|
|
36
|
+
|
|
37
|
+
if args.check:
|
|
38
|
+
for name in stale:
|
|
39
|
+
print(f"out of date: {name}", file=sys.stderr)
|
|
40
|
+
for name in extra:
|
|
41
|
+
print(f"not in source: {name}", file=sys.stderr)
|
|
42
|
+
return 1 if stale or extra else 0
|
|
43
|
+
|
|
44
|
+
TARGET.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
for f in files:
|
|
46
|
+
shutil.copyfile(f, TARGET / f.name)
|
|
47
|
+
for name in extra:
|
|
48
|
+
(TARGET / name).unlink()
|
|
49
|
+
print(f"Synced {len(files)} fixture(s) from {source}")
|
|
50
|
+
return 0
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
if __name__ == "__main__":
|
|
54
|
+
sys.exit(main())
|
|
@@ -4,17 +4,24 @@ from typing import List
|
|
|
4
4
|
from ..errors import ChunkingError
|
|
5
5
|
from .base import BaseChunker
|
|
6
6
|
|
|
7
|
+
# The exact set JavaScript's \s matches. Python's \s differs (it includes
|
|
8
|
+
# \x1c-\x1f and \x85 but not \ufeff), which would make the two SDKs chunk the
|
|
9
|
+
# same text differently.
|
|
10
|
+
_JS_WHITESPACE = re.compile(
|
|
11
|
+
"[\t\n\v\f\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff]+"
|
|
12
|
+
)
|
|
13
|
+
|
|
7
14
|
|
|
8
15
|
class RecursiveChunker(BaseChunker):
|
|
9
16
|
def split(self, text: str) -> List[str]:
|
|
10
|
-
if not
|
|
17
|
+
if not _JS_WHITESPACE.sub("", text):
|
|
11
18
|
return []
|
|
12
19
|
if self.overlap >= self.chunk_size:
|
|
13
20
|
raise ChunkingError(
|
|
14
21
|
f"overlap ({self.overlap}) must be smaller than chunkSize ({self.chunk_size})"
|
|
15
22
|
)
|
|
16
23
|
|
|
17
|
-
words = [w for w in
|
|
24
|
+
words = [w for w in _JS_WHITESPACE.split(text) if w]
|
|
18
25
|
if len(words) <= self.chunk_size:
|
|
19
26
|
return [" ".join(words)]
|
|
20
27
|
|
|
@@ -10,6 +10,7 @@ from .constants import (
|
|
|
10
10
|
DEFAULT_TOP_K,
|
|
11
11
|
)
|
|
12
12
|
from .types import EmbeddingProviderConfig, LLMProviderConfig, VectorStoreProviderConfig
|
|
13
|
+
from .vectordb.base import VectorStore
|
|
13
14
|
|
|
14
15
|
|
|
15
16
|
class DocumentOptions(BaseModel):
|
|
@@ -44,6 +45,10 @@ def resolve_config(options: Optional[Union[DocumentOptions, Dict[str, Any]]] = N
|
|
|
44
45
|
if options is None:
|
|
45
46
|
opts = DocumentOptions()
|
|
46
47
|
elif isinstance(options, dict):
|
|
48
|
+
# A VectorStore instance is read straight from the raw options by
|
|
49
|
+
# Document, so keep it out of the provider-config validation.
|
|
50
|
+
if isinstance(options.get("vectorStore"), VectorStore):
|
|
51
|
+
options = {k: v for k, v in options.items() if k != "vectorStore"}
|
|
47
52
|
opts = DocumentOptions.model_validate(options)
|
|
48
53
|
else:
|
|
49
54
|
opts = options
|
|
@@ -5,7 +5,16 @@ try:
|
|
|
5
5
|
PACKAGE_VERSION = importlib.metadata.version("raglite-toolkit")
|
|
6
6
|
except importlib.metadata.PackageNotFoundError:
|
|
7
7
|
PACKAGE_NAME = "raglite-toolkit"
|
|
8
|
-
PACKAGE_VERSION = "1.2.
|
|
8
|
+
PACKAGE_VERSION = "1.2.2" # local development fallback
|
|
9
|
+
|
|
10
|
+
# Version of the stored index layout (chunking, ids, payloads). The build cache
|
|
11
|
+
# is keyed on this instead of PACKAGE_VERSION so upgrading the package does not
|
|
12
|
+
# re-embed every index. Bump it only when a change makes existing indexes
|
|
13
|
+
# incompatible.
|
|
14
|
+
INDEX_FORMAT_VERSION = 1
|
|
15
|
+
|
|
16
|
+
# Releases that wrote format-1 indexes before ``formatVersion`` was recorded.
|
|
17
|
+
LEGACY_FORMAT_1_VERSIONS = frozenset({"1.2.1"})
|
|
9
18
|
|
|
10
19
|
|
|
11
20
|
SUPPORTED_EXTENSIONS = {".pdf", ".txt", ".json", ".md", ".markdown", ".docx"}
|
|
@@ -4,11 +4,12 @@ from pathlib import Path
|
|
|
4
4
|
from typing import Any, Dict, Generator, List, Optional, Union
|
|
5
5
|
|
|
6
6
|
from ..config import DocumentOptions, ResolvedConfig, resolve_config
|
|
7
|
-
from ..errors import RagLiteError
|
|
7
|
+
from ..errors import ConfigError, RagLiteError
|
|
8
8
|
from ..llm import generate_answer, stream_answer
|
|
9
9
|
from ..loaders import DirectoryLoader, is_supported_file, is_url
|
|
10
10
|
from ..types import AnswerResult, SearchResult
|
|
11
11
|
from ..utils.logger import Logger, create_logger
|
|
12
|
+
from ..vectordb import VectorStore
|
|
12
13
|
from .document import Document
|
|
13
14
|
|
|
14
15
|
|
|
@@ -91,6 +92,16 @@ class DocumentCollection:
|
|
|
91
92
|
self.logger.warning(msg)
|
|
92
93
|
errors.append({"source": src, "error": msg})
|
|
93
94
|
|
|
95
|
+
# A VectorStore instance has a single namespace, so every document would
|
|
96
|
+
# share it and each build() would reset the previous document's index.
|
|
97
|
+
raw_store = self.options.get("vectorStore") if isinstance(self.options, dict) else None
|
|
98
|
+
if isinstance(raw_store, VectorStore) and len(set(file_list)) > 1:
|
|
99
|
+
raise ConfigError(
|
|
100
|
+
"A VectorStore instance cannot be shared by multiple documents in a "
|
|
101
|
+
"DocumentCollection. Pass a vector store provider config "
|
|
102
|
+
'(e.g. {"provider": "qdrant", ...}) instead.'
|
|
103
|
+
)
|
|
104
|
+
|
|
94
105
|
total_chunks = 0
|
|
95
106
|
cached_docs = 0
|
|
96
107
|
new_docs = 0
|
|
@@ -146,12 +157,19 @@ class DocumentCollection:
|
|
|
146
157
|
st = score_threshold if score_threshold is not None else opts.get("scoreThreshold", opts.get("score_threshold", self.config.scoreThreshold))
|
|
147
158
|
|
|
148
159
|
all_hits: List[SearchResult] = []
|
|
160
|
+
failures: List[Exception] = []
|
|
149
161
|
for doc in self.documents.values():
|
|
150
162
|
try:
|
|
151
163
|
hits = doc.search(query, top_k=tk * 2, score_threshold=st)
|
|
152
164
|
all_hits.extend(hits)
|
|
153
|
-
except Exception:
|
|
154
|
-
|
|
165
|
+
except Exception as err:
|
|
166
|
+
failures.append(err)
|
|
167
|
+
self.logger.warning(f'Search failed for document "{doc.file_path}": {err}')
|
|
168
|
+
|
|
169
|
+
# One broken document should not hide results from the others, but if
|
|
170
|
+
# every document failed the caller needs the error, not an empty list.
|
|
171
|
+
if len(failures) == len(self.documents):
|
|
172
|
+
raise failures[0]
|
|
155
173
|
|
|
156
174
|
all_hits.sort(key=lambda h: h.score, reverse=True)
|
|
157
175
|
return all_hits[:tk]
|
|
@@ -4,13 +4,19 @@ from typing import Any, Dict, Generator, Optional, Union
|
|
|
4
4
|
|
|
5
5
|
from ..chunking import RecursiveChunker
|
|
6
6
|
from ..config import DocumentOptions, ResolvedConfig, resolve_config
|
|
7
|
-
from ..constants import PACKAGE_VERSION
|
|
7
|
+
from ..constants import INDEX_FORMAT_VERSION, LEGACY_FORMAT_1_VERSIONS, PACKAGE_VERSION
|
|
8
8
|
from ..embeddings import Embedder, create_embedder
|
|
9
9
|
from ..errors import FileNotIndexedError, LoaderError, RagLiteError
|
|
10
10
|
from ..llm import generate_answer, stream_answer
|
|
11
11
|
from ..loaders import get_loader, is_url
|
|
12
12
|
from ..retrieval import Retriever
|
|
13
|
-
from ..types import
|
|
13
|
+
from ..types import (
|
|
14
|
+
AnswerResult,
|
|
15
|
+
ChunkMetadata,
|
|
16
|
+
EmbeddingProviderConfig,
|
|
17
|
+
IndexMetadata,
|
|
18
|
+
StoredChunk,
|
|
19
|
+
)
|
|
14
20
|
from ..utils.hash import hash_file, hash_string, namespace_from_path
|
|
15
21
|
from ..utils.logger import create_logger
|
|
16
22
|
from ..vectordb import MemoryVectorStore, VectorStore, create_vector_store
|
|
@@ -74,16 +80,17 @@ class Document:
|
|
|
74
80
|
if should_rebuild is None:
|
|
75
81
|
should_rebuild = opts.get("rebuild", False)
|
|
76
82
|
|
|
77
|
-
if not os.path.exists(self.file_path):
|
|
83
|
+
if not is_url(self.file_path) and not os.path.exists(self.file_path):
|
|
78
84
|
raise LoaderError(f"File does not exist: {self.file_path}")
|
|
79
85
|
|
|
80
86
|
self.store.load()
|
|
81
87
|
existing = self.store.read_index_metadata()
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
88
|
+
# Web pages change without their URL changing, so fingerprint the
|
|
89
|
+
# fetched content rather than the URL.
|
|
90
|
+
text: Optional[str] = None
|
|
91
|
+
if is_url(self.file_path):
|
|
92
|
+
text = get_loader(self.file_path).load()
|
|
93
|
+
source_hash = hash_string(text) if text is not None else hash_file(self.file_path)
|
|
87
94
|
|
|
88
95
|
if (
|
|
89
96
|
not should_rebuild
|
|
@@ -110,8 +117,8 @@ class Document:
|
|
|
110
117
|
self.store.reset()
|
|
111
118
|
self.store.load()
|
|
112
119
|
|
|
113
|
-
|
|
114
|
-
|
|
120
|
+
if text is None:
|
|
121
|
+
text = get_loader(self.file_path).load()
|
|
115
122
|
if not text:
|
|
116
123
|
raise LoaderError(
|
|
117
124
|
f"Loader returned empty text for {self.file_path}"
|
|
@@ -161,6 +168,7 @@ class Document:
|
|
|
161
168
|
from datetime import timezone
|
|
162
169
|
metadata = IndexMetadata(
|
|
163
170
|
version=PACKAGE_VERSION,
|
|
171
|
+
formatVersion=INDEX_FORMAT_VERSION,
|
|
164
172
|
source=self.file_path,
|
|
165
173
|
sourceHash=source_hash,
|
|
166
174
|
chunkSize=c_size,
|
|
@@ -340,19 +348,10 @@ class Document:
|
|
|
340
348
|
f'No RagLite index found for "{self.file_path}". Call build() first.'
|
|
341
349
|
)
|
|
342
350
|
|
|
343
|
-
embed_config = self.config.embeddings
|
|
344
|
-
if embed_config.model is None:
|
|
345
|
-
from ..types import EmbeddingProviderConfig
|
|
346
|
-
|
|
347
|
-
embed_config = EmbeddingProviderConfig(
|
|
348
|
-
provider=embed_config.provider,
|
|
349
|
-
model=existing.embeddingModel,
|
|
350
|
-
apiKey=embed_config.apiKey,
|
|
351
|
-
baseURL=embed_config.baseURL,
|
|
352
|
-
)
|
|
353
|
-
|
|
354
351
|
if self.embedder is None:
|
|
355
|
-
self.embedder = create_embedder(
|
|
352
|
+
self.embedder = create_embedder(
|
|
353
|
+
_query_embeddings_config(existing, self.config.embeddings)
|
|
354
|
+
)
|
|
356
355
|
self.ready = True
|
|
357
356
|
|
|
358
357
|
def _cache_still_valid(
|
|
@@ -363,7 +362,7 @@ class Document:
|
|
|
363
362
|
overlap: int,
|
|
364
363
|
embeddings_config: Any,
|
|
365
364
|
) -> bool:
|
|
366
|
-
if existing
|
|
365
|
+
if index_format_version(existing) != INDEX_FORMAT_VERSION:
|
|
367
366
|
return False
|
|
368
367
|
if existing.sourceHash != source_hash:
|
|
369
368
|
return False
|
|
@@ -378,3 +377,26 @@ class Document:
|
|
|
378
377
|
if req_model is not None and req_model != existing.embeddingModel:
|
|
379
378
|
return False
|
|
380
379
|
return True
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def index_format_version(metadata: IndexMetadata) -> Optional[int]:
|
|
383
|
+
"""Indexes written before ``formatVersion`` existed are identified by package version."""
|
|
384
|
+
if metadata.formatVersion is not None:
|
|
385
|
+
return metadata.formatVersion
|
|
386
|
+
return 1 if metadata.version in LEGACY_FORMAT_1_VERSIONS else None
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _query_embeddings_config(
|
|
390
|
+
existing: IndexMetadata, configured: EmbeddingProviderConfig
|
|
391
|
+
) -> EmbeddingProviderConfig:
|
|
392
|
+
"""Queries must be embedded with the same provider and model as the index,
|
|
393
|
+
which may differ from the constructor default when ``build(embeddings=...)``
|
|
394
|
+
overrode it. Credentials from the configured provider are reused when it
|
|
395
|
+
matches; otherwise the provider falls back to its environment variables."""
|
|
396
|
+
same_provider = configured.provider == existing.embeddingProvider
|
|
397
|
+
return EmbeddingProviderConfig(
|
|
398
|
+
provider=existing.embeddingProvider,
|
|
399
|
+
model=existing.embeddingModel,
|
|
400
|
+
apiKey=configured.apiKey if same_provider else None,
|
|
401
|
+
baseURL=configured.baseURL if same_provider else None,
|
|
402
|
+
)
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import html
|
|
2
2
|
import re
|
|
3
3
|
|
|
4
|
+
from ..constants import PACKAGE_VERSION
|
|
4
5
|
from ..errors import LoaderError
|
|
5
6
|
from .base import BaseLoader
|
|
6
7
|
|
|
@@ -18,7 +19,7 @@ class WebLoader(BaseLoader):
|
|
|
18
19
|
import httpx
|
|
19
20
|
|
|
20
21
|
headers = {
|
|
21
|
-
"User-Agent": "RAGLite/
|
|
22
|
+
"User-Agent": f"RAGLite/{PACKAGE_VERSION} (Python/3.10+)",
|
|
22
23
|
"Accept": "text/html,text/plain,application/xhtml+xml;q=0.9,*/*;q=0.8",
|
|
23
24
|
}
|
|
24
25
|
response = httpx.get(self.url, headers=headers, follow_redirects=True, timeout=15.0)
|
|
@@ -96,7 +96,9 @@ class AnswerResult(BaseModel):
|
|
|
96
96
|
class IndexMetadata(BaseModel):
|
|
97
97
|
model_config = ConfigDict(populate_by_name=True, extra="allow")
|
|
98
98
|
|
|
99
|
-
version: str
|
|
99
|
+
version: str # package version that built the index (informational)
|
|
100
|
+
# Index layout version; see INDEX_FORMAT_VERSION. None on indexes built before 1.2.2.
|
|
101
|
+
formatVersion: Optional[int] = Field(default=None, alias="formatVersion")
|
|
100
102
|
source: str
|
|
101
103
|
sourceHash: str = Field(..., alias="sourceHash")
|
|
102
104
|
chunkSize: int = Field(..., alias="chunkSize")
|
|
@@ -126,7 +126,8 @@ class PineconeVectorStore(VectorStore):
|
|
|
126
126
|
def save_index_metadata(self, metadata: IndexMetadata) -> None:
|
|
127
127
|
dim = metadata.embeddingDimensions
|
|
128
128
|
zero_vec = [0.0] * dim
|
|
129
|
-
|
|
129
|
+
# Pinecone rejects null metadata values.
|
|
130
|
+
payload = metadata.model_dump(by_alias=True, exclude_none=True)
|
|
130
131
|
payload["isMetadata"] = True
|
|
131
132
|
self._request(
|
|
132
133
|
"POST",
|
|
@@ -154,6 +155,9 @@ class PineconeVectorStore(VectorStore):
|
|
|
154
155
|
try:
|
|
155
156
|
return IndexMetadata(
|
|
156
157
|
version=m.get("version", ""),
|
|
158
|
+
formatVersion=(
|
|
159
|
+
int(m["formatVersion"]) if m.get("formatVersion") is not None else None
|
|
160
|
+
),
|
|
157
161
|
source=m.get("source", ""),
|
|
158
162
|
sourceHash=m.get("sourceHash", ""),
|
|
159
163
|
chunkSize=int(m.get("chunkSize", 0)),
|