raglite-toolkit 1.2.1__tar.gz → 1.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/.github/workflows/pr-verify.yml +9 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/.gitignore +1 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/ARCHITECTURE.md +15 -10
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/CHANGELOG.md +12 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/PKG-INFO +5 -5
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/README.md +4 -4
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/pyproject.toml +1 -1
- raglite_toolkit-1.2.2/scripts/sync_shared_fixtures.py +54 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/chunking/recursive.py +9 -2
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/constants.py +10 -1
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/core/document.py +10 -2
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/types.py +3 -1
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/pinecone.py +5 -1
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/qdrant.py +1 -0
- raglite_toolkit-1.2.2/tests/fixtures/shared/chunker.json +100 -0
- raglite_toolkit-1.2.2/tests/fixtures/shared/hash.json +34 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_regressions.py +44 -0
- raglite_toolkit-1.2.2/tests/unit/test_shared_fixtures.py +42 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/.github/workflows/publish.yml +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/LICENSE +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/basic.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/custom_store_example.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/multi_provider.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/ollama_test.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/qdrant_example.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/sample.txt +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/serve.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/scripts/pre-commit.sh +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/scripts/pre-release.sh +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/api/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/api/schemas.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/api/server.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/chunking/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/chunking/base.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/cli.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/config.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/core/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/core/collection.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/base.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/factory.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/local.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/models.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/remote.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/errors.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/answer.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/factory.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/models.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/prompt.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/base.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/directory.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/docx.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/json.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/markdown.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/pdf.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/txt.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/web.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/retrieval/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/retrieval/retriever.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/utils/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/utils/hash.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/utils/logger.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/base.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/factory.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/memory.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/conftest.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_api.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_ask.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_collection.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_document.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_ollama.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/__init__.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_chunking.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_cli.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_config.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_directory_loader.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_errors.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_hash.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_loaders.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_prompt.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_retriever.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_vectordb.py +0 -0
- {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_web_loader.py +0 -0
|
@@ -34,3 +34,12 @@ jobs:
|
|
|
34
34
|
|
|
35
35
|
- name: Run tests
|
|
36
36
|
run: python -m pytest tests/ -q
|
|
37
|
+
|
|
38
|
+
- name: Check out TypeScript SDK for shared fixtures
|
|
39
|
+
uses: actions/checkout@v4
|
|
40
|
+
with:
|
|
41
|
+
repository: creatorpiyush/raglite
|
|
42
|
+
path: .ts-sdk
|
|
43
|
+
|
|
44
|
+
- name: Check shared fixtures match the TypeScript SDK
|
|
45
|
+
run: python scripts/sync_shared_fixtures.py --check --source .ts-sdk
|
|
@@ -126,9 +126,9 @@ graph TD
|
|
|
126
126
|
### 4.1 Ingestion & Indexing
|
|
127
127
|
1. `doc.build()` invokes the appropriate `BaseLoader` based on file extension (`.pdf`, `.txt`, `.json`, `.md`, `.docx`).
|
|
128
128
|
2. Calculates SHA-256 hash of raw document content.
|
|
129
|
-
3. Checks existing `IndexMetadata` in `VectorStore`.
|
|
129
|
+
3. Checks existing `IndexMetadata` in `VectorStore`. The cached index is reused when the index format version, content hash, chunk size, overlap and embedding provider/model all match (URL sources are hashed by their fetched text).
|
|
130
130
|
4. If hash differs or force rebuild requested:
|
|
131
|
-
- `
|
|
131
|
+
- `RecursiveChunker` splits text into word-based chunks (default 500 words, 50-word overlap).
|
|
132
132
|
- `EmbeddingFactory` generates normalized vectors for each chunk.
|
|
133
133
|
- `VectorStore.add()` saves chunks and `VectorStore.save_index_metadata()` persists index metadata.
|
|
134
134
|
|
|
@@ -162,27 +162,32 @@ Built using **FastAPI** framework for async capabilities, automatic OpenAPI docs
|
|
|
162
162
|
```python
|
|
163
163
|
from abc import ABC, abstractmethod
|
|
164
164
|
from typing import List, Optional
|
|
165
|
-
from raglite.types import IndexMetadata,
|
|
165
|
+
from raglite.types import IndexMetadata, StoredChunk
|
|
166
|
+
from raglite.vectordb.base import VectorSearchHit
|
|
166
167
|
|
|
167
168
|
class VectorStore(ABC):
|
|
169
|
+
@property
|
|
168
170
|
@abstractmethod
|
|
169
|
-
def
|
|
171
|
+
def namespace(self) -> str: ...
|
|
170
172
|
|
|
171
173
|
@abstractmethod
|
|
172
|
-
def
|
|
174
|
+
def load(self) -> None: ...
|
|
173
175
|
|
|
174
176
|
@abstractmethod
|
|
175
|
-
def
|
|
177
|
+
def reset(self) -> None: ...
|
|
176
178
|
|
|
177
179
|
@abstractmethod
|
|
178
|
-
def
|
|
180
|
+
def add(self, chunks: List[StoredChunk]) -> None: ...
|
|
179
181
|
|
|
180
182
|
@abstractmethod
|
|
181
|
-
def
|
|
183
|
+
def search(self, embedding: List[float], top_k: int) -> List[VectorSearchHit]: ...
|
|
182
184
|
|
|
183
185
|
@abstractmethod
|
|
184
|
-
def
|
|
186
|
+
def count(self) -> int: ...
|
|
185
187
|
|
|
186
188
|
@abstractmethod
|
|
187
|
-
def
|
|
189
|
+
def save_index_metadata(self, metadata: IndexMetadata) -> None: ...
|
|
190
|
+
|
|
191
|
+
@abstractmethod
|
|
192
|
+
def read_index_metadata(self) -> Optional[IndexMetadata]: ...
|
|
188
193
|
```
|
|
@@ -2,6 +2,18 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
+
## [1.2.2] - Unreleased
|
|
6
|
+
|
|
7
|
+
### Changed
|
|
8
|
+
- **Upgrades keep cached indexes:** The build cache is now keyed on an index format version (`formatVersion` in `IndexMetadata`) instead of the package version, so upgrading RAGLite no longer re-embeds every index. Indexes built by 1.2.1 are reused as-is; indexes from older releases are rebuilt once.
|
|
9
|
+
|
|
10
|
+
### Fixed
|
|
11
|
+
- **Chunker parity with the TypeScript SDK:** Words are now split on exactly the whitespace characters JavaScript treats as whitespace. Previously text containing a byte order mark (U+FEFF), U+0085 or U+001C–U+001F was chunked differently from the TypeScript SDK. Existing indexes are not rebuilt for this; pass `rebuild=True` if your sources contain those characters.
|
|
12
|
+
- **Docs:** ARCHITECTURE.md now describes the word-based `RecursiveChunker` (500 words, 50 overlap) and the actual `VectorStore` interface. The README no longer lists LanceDB, which is TypeScript-only.
|
|
13
|
+
|
|
14
|
+
### Internal
|
|
15
|
+
- Added cross-SDK fixtures (`tests/fixtures/shared/`, copied from the TypeScript SDK with `scripts/sync_shared_fixtures.py`) that pin chunking and hashing output. CI fails if the copy is out of date.
|
|
16
|
+
|
|
5
17
|
## [1.2.1] - 2026-10-02
|
|
6
18
|
|
|
7
19
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: raglite-toolkit
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python.
|
|
5
5
|
Project-URL: Homepage, https://github.com/creatorpiyush/raglite-py
|
|
6
6
|
Project-URL: Repository, https://github.com/creatorpiyush/raglite-py
|
|
@@ -172,7 +172,7 @@ print()
|
|
|
172
172
|
|
|
173
173
|
## Pluggable Vector Databases
|
|
174
174
|
|
|
175
|
-
`raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone,
|
|
175
|
+
`raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, or custom subclasses; LanceDB is currently TypeScript-only):
|
|
176
176
|
|
|
177
177
|
### Memory Store (Default)
|
|
178
178
|
```python
|
|
@@ -336,7 +336,7 @@ Document("./policy.pdf", {
|
|
|
336
336
|
|
|
337
337
|
## How Caching Works
|
|
338
338
|
|
|
339
|
-
Every `build()` call fingerprints the source file with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
|
|
339
|
+
Every `build()` call fingerprints the source (file bytes, or the fetched text for URLs) with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
|
|
340
340
|
|
|
341
341
|
| Factor | Triggers rebuild if changed |
|
|
342
342
|
|--------|-----------------------------|
|
|
@@ -344,7 +344,7 @@ Every `build()` call fingerprints the source file with a **SHA-256 content hash*
|
|
|
344
344
|
| Chunk size | `chunkSize` changed |
|
|
345
345
|
| Overlap | `overlap` changed |
|
|
346
346
|
| Embedding provider/model | Provider or model string changed |
|
|
347
|
-
|
|
|
347
|
+
| Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
|
|
348
348
|
|
|
349
349
|
Pass `rebuild=True` to `build()` to force a fresh index regardless.
|
|
350
350
|
|
|
@@ -360,7 +360,7 @@ Each document is stored under `.raglite/<sha256-prefix>/`, so multiple documents
|
|
|
360
360
|
from raglite.vectordb.base import VectorStore
|
|
361
361
|
|
|
362
362
|
class MyVectorStore(VectorStore):
|
|
363
|
-
# Implement: load, reset, add, search, count,
|
|
363
|
+
# Implement: namespace (property), load, reset, add, search, count,
|
|
364
364
|
# save_index_metadata, read_index_metadata
|
|
365
365
|
...
|
|
366
366
|
```
|
|
@@ -126,7 +126,7 @@ print()
|
|
|
126
126
|
|
|
127
127
|
## Pluggable Vector Databases
|
|
128
128
|
|
|
129
|
-
`raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone,
|
|
129
|
+
`raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, or custom subclasses; LanceDB is currently TypeScript-only):
|
|
130
130
|
|
|
131
131
|
### Memory Store (Default)
|
|
132
132
|
```python
|
|
@@ -290,7 +290,7 @@ Document("./policy.pdf", {
|
|
|
290
290
|
|
|
291
291
|
## How Caching Works
|
|
292
292
|
|
|
293
|
-
Every `build()` call fingerprints the source file with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
|
|
293
|
+
Every `build()` call fingerprints the source (file bytes, or the fetched text for URLs) with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
|
|
294
294
|
|
|
295
295
|
| Factor | Triggers rebuild if changed |
|
|
296
296
|
|--------|-----------------------------|
|
|
@@ -298,7 +298,7 @@ Every `build()` call fingerprints the source file with a **SHA-256 content hash*
|
|
|
298
298
|
| Chunk size | `chunkSize` changed |
|
|
299
299
|
| Overlap | `overlap` changed |
|
|
300
300
|
| Embedding provider/model | Provider or model string changed |
|
|
301
|
-
|
|
|
301
|
+
| Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
|
|
302
302
|
|
|
303
303
|
Pass `rebuild=True` to `build()` to force a fresh index regardless.
|
|
304
304
|
|
|
@@ -314,7 +314,7 @@ Each document is stored under `.raglite/<sha256-prefix>/`, so multiple documents
|
|
|
314
314
|
from raglite.vectordb.base import VectorStore
|
|
315
315
|
|
|
316
316
|
class MyVectorStore(VectorStore):
|
|
317
|
-
# Implement: load, reset, add, search, count,
|
|
317
|
+
# Implement: namespace (property), load, reset, add, search, count,
|
|
318
318
|
# save_index_metadata, read_index_metadata
|
|
319
319
|
...
|
|
320
320
|
```
|
|
@@ -8,7 +8,7 @@ packages = ["src/raglite"]
|
|
|
8
8
|
|
|
9
9
|
[project]
|
|
10
10
|
name = "raglite-toolkit"
|
|
11
|
-
version = "1.2.
|
|
11
|
+
version = "1.2.2"
|
|
12
12
|
description = "Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python."
|
|
13
13
|
readme = "README.md"
|
|
14
14
|
requires-python = ">=3.10"
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Copy or check the cross-SDK fixtures shared with the TypeScript SDK.
|
|
2
|
+
|
|
3
|
+
The TypeScript repo (raglite) is the source of truth: it generates
|
|
4
|
+
tests/fixtures/shared/*.json with scripts/generate-shared-fixtures.ts. This
|
|
5
|
+
repo keeps an identical copy so both SDKs are tested against the same
|
|
6
|
+
expected outputs.
|
|
7
|
+
|
|
8
|
+
python scripts/sync_shared_fixtures.py # copy from ../raglite
|
|
9
|
+
python scripts/sync_shared_fixtures.py --check # exit 1 if out of date
|
|
10
|
+
python scripts/sync_shared_fixtures.py --source /path/to/raglite
|
|
11
|
+
"""
|
|
12
|
+
import argparse
|
|
13
|
+
import filecmp
|
|
14
|
+
import shutil
|
|
15
|
+
import sys
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
REPO_ROOT = Path(__file__).resolve().parent.parent
|
|
19
|
+
TARGET = REPO_ROOT / "tests" / "fixtures" / "shared"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def main() -> int:
|
|
23
|
+
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
24
|
+
parser.add_argument("--source", type=Path, default=REPO_ROOT.parent / "raglite")
|
|
25
|
+
parser.add_argument("--check", action="store_true")
|
|
26
|
+
args = parser.parse_args()
|
|
27
|
+
|
|
28
|
+
source = args.source / "tests" / "fixtures" / "shared"
|
|
29
|
+
files = sorted(source.glob("*.json"))
|
|
30
|
+
if not files:
|
|
31
|
+
print(f"No shared fixtures found in {source}", file=sys.stderr)
|
|
32
|
+
return 1
|
|
33
|
+
|
|
34
|
+
stale = [f.name for f in files if not (TARGET / f.name).exists() or not filecmp.cmp(f, TARGET / f.name, shallow=False)]
|
|
35
|
+
extra = sorted({p.name for p in TARGET.glob("*.json")} - {f.name for f in files})
|
|
36
|
+
|
|
37
|
+
if args.check:
|
|
38
|
+
for name in stale:
|
|
39
|
+
print(f"out of date: {name}", file=sys.stderr)
|
|
40
|
+
for name in extra:
|
|
41
|
+
print(f"not in source: {name}", file=sys.stderr)
|
|
42
|
+
return 1 if stale or extra else 0
|
|
43
|
+
|
|
44
|
+
TARGET.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
for f in files:
|
|
46
|
+
shutil.copyfile(f, TARGET / f.name)
|
|
47
|
+
for name in extra:
|
|
48
|
+
(TARGET / name).unlink()
|
|
49
|
+
print(f"Synced {len(files)} fixture(s) from {source}")
|
|
50
|
+
return 0
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
if __name__ == "__main__":
|
|
54
|
+
sys.exit(main())
|
|
@@ -4,17 +4,24 @@ from typing import List
|
|
|
4
4
|
from ..errors import ChunkingError
|
|
5
5
|
from .base import BaseChunker
|
|
6
6
|
|
|
7
|
+
# The exact set JavaScript's \s matches. Python's \s differs (it includes
|
|
8
|
+
# \x1c-\x1f and \x85 but not \ufeff), which would make the two SDKs chunk the
|
|
9
|
+
# same text differently.
|
|
10
|
+
_JS_WHITESPACE = re.compile(
|
|
11
|
+
"[\t\n\v\f\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff]+"
|
|
12
|
+
)
|
|
13
|
+
|
|
7
14
|
|
|
8
15
|
class RecursiveChunker(BaseChunker):
|
|
9
16
|
def split(self, text: str) -> List[str]:
|
|
10
|
-
if not
|
|
17
|
+
if not _JS_WHITESPACE.sub("", text):
|
|
11
18
|
return []
|
|
12
19
|
if self.overlap >= self.chunk_size:
|
|
13
20
|
raise ChunkingError(
|
|
14
21
|
f"overlap ({self.overlap}) must be smaller than chunkSize ({self.chunk_size})"
|
|
15
22
|
)
|
|
16
23
|
|
|
17
|
-
words = [w for w in
|
|
24
|
+
words = [w for w in _JS_WHITESPACE.split(text) if w]
|
|
18
25
|
if len(words) <= self.chunk_size:
|
|
19
26
|
return [" ".join(words)]
|
|
20
27
|
|
|
@@ -5,7 +5,16 @@ try:
|
|
|
5
5
|
PACKAGE_VERSION = importlib.metadata.version("raglite-toolkit")
|
|
6
6
|
except importlib.metadata.PackageNotFoundError:
|
|
7
7
|
PACKAGE_NAME = "raglite-toolkit"
|
|
8
|
-
PACKAGE_VERSION = "1.2.
|
|
8
|
+
PACKAGE_VERSION = "1.2.2" # local development fallback
|
|
9
|
+
|
|
10
|
+
# Version of the stored index layout (chunking, ids, payloads). The build cache
|
|
11
|
+
# is keyed on this instead of PACKAGE_VERSION so upgrading the package does not
|
|
12
|
+
# re-embed every index. Bump it only when a change makes existing indexes
|
|
13
|
+
# incompatible.
|
|
14
|
+
INDEX_FORMAT_VERSION = 1
|
|
15
|
+
|
|
16
|
+
# Releases that wrote format-1 indexes before ``formatVersion`` was recorded.
|
|
17
|
+
LEGACY_FORMAT_1_VERSIONS = frozenset({"1.2.1"})
|
|
9
18
|
|
|
10
19
|
|
|
11
20
|
SUPPORTED_EXTENSIONS = {".pdf", ".txt", ".json", ".md", ".markdown", ".docx"}
|
|
@@ -4,7 +4,7 @@ from typing import Any, Dict, Generator, Optional, Union
|
|
|
4
4
|
|
|
5
5
|
from ..chunking import RecursiveChunker
|
|
6
6
|
from ..config import DocumentOptions, ResolvedConfig, resolve_config
|
|
7
|
-
from ..constants import PACKAGE_VERSION
|
|
7
|
+
from ..constants import INDEX_FORMAT_VERSION, LEGACY_FORMAT_1_VERSIONS, PACKAGE_VERSION
|
|
8
8
|
from ..embeddings import Embedder, create_embedder
|
|
9
9
|
from ..errors import FileNotIndexedError, LoaderError, RagLiteError
|
|
10
10
|
from ..llm import generate_answer, stream_answer
|
|
@@ -168,6 +168,7 @@ class Document:
|
|
|
168
168
|
from datetime import timezone
|
|
169
169
|
metadata = IndexMetadata(
|
|
170
170
|
version=PACKAGE_VERSION,
|
|
171
|
+
formatVersion=INDEX_FORMAT_VERSION,
|
|
171
172
|
source=self.file_path,
|
|
172
173
|
sourceHash=source_hash,
|
|
173
174
|
chunkSize=c_size,
|
|
@@ -361,7 +362,7 @@ class Document:
|
|
|
361
362
|
overlap: int,
|
|
362
363
|
embeddings_config: Any,
|
|
363
364
|
) -> bool:
|
|
364
|
-
if existing
|
|
365
|
+
if index_format_version(existing) != INDEX_FORMAT_VERSION:
|
|
365
366
|
return False
|
|
366
367
|
if existing.sourceHash != source_hash:
|
|
367
368
|
return False
|
|
@@ -378,6 +379,13 @@ class Document:
|
|
|
378
379
|
return True
|
|
379
380
|
|
|
380
381
|
|
|
382
|
+
def index_format_version(metadata: IndexMetadata) -> Optional[int]:
|
|
383
|
+
"""Indexes written before ``formatVersion`` existed are identified by package version."""
|
|
384
|
+
if metadata.formatVersion is not None:
|
|
385
|
+
return metadata.formatVersion
|
|
386
|
+
return 1 if metadata.version in LEGACY_FORMAT_1_VERSIONS else None
|
|
387
|
+
|
|
388
|
+
|
|
381
389
|
def _query_embeddings_config(
|
|
382
390
|
existing: IndexMetadata, configured: EmbeddingProviderConfig
|
|
383
391
|
) -> EmbeddingProviderConfig:
|
|
@@ -96,7 +96,9 @@ class AnswerResult(BaseModel):
|
|
|
96
96
|
class IndexMetadata(BaseModel):
|
|
97
97
|
model_config = ConfigDict(populate_by_name=True, extra="allow")
|
|
98
98
|
|
|
99
|
-
version: str
|
|
99
|
+
version: str # package version that built the index (informational)
|
|
100
|
+
# Index layout version; see INDEX_FORMAT_VERSION. None on indexes built before 1.2.2.
|
|
101
|
+
formatVersion: Optional[int] = Field(default=None, alias="formatVersion")
|
|
100
102
|
source: str
|
|
101
103
|
sourceHash: str = Field(..., alias="sourceHash")
|
|
102
104
|
chunkSize: int = Field(..., alias="chunkSize")
|
|
@@ -126,7 +126,8 @@ class PineconeVectorStore(VectorStore):
|
|
|
126
126
|
def save_index_metadata(self, metadata: IndexMetadata) -> None:
|
|
127
127
|
dim = metadata.embeddingDimensions
|
|
128
128
|
zero_vec = [0.0] * dim
|
|
129
|
-
|
|
129
|
+
# Pinecone rejects null metadata values.
|
|
130
|
+
payload = metadata.model_dump(by_alias=True, exclude_none=True)
|
|
130
131
|
payload["isMetadata"] = True
|
|
131
132
|
self._request(
|
|
132
133
|
"POST",
|
|
@@ -154,6 +155,9 @@ class PineconeVectorStore(VectorStore):
|
|
|
154
155
|
try:
|
|
155
156
|
return IndexMetadata(
|
|
156
157
|
version=m.get("version", ""),
|
|
158
|
+
formatVersion=(
|
|
159
|
+
int(m["formatVersion"]) if m.get("formatVersion") is not None else None
|
|
160
|
+
),
|
|
157
161
|
source=m.get("source", ""),
|
|
158
162
|
sourceHash=m.get("sourceHash", ""),
|
|
159
163
|
chunkSize=int(m.get("chunkSize", 0)),
|
|
@@ -197,6 +197,7 @@ class QdrantVectorStore(VectorStore):
|
|
|
197
197
|
try:
|
|
198
198
|
return IndexMetadata(
|
|
199
199
|
version=payload.get("version", ""),
|
|
200
|
+
formatVersion=payload.get("formatVersion"),
|
|
200
201
|
source=payload.get("source", ""),
|
|
201
202
|
sourceHash=payload.get("sourceHash", ""),
|
|
202
203
|
chunkSize=payload.get("chunkSize", 0),
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"name": "empty",
|
|
4
|
+
"text": "",
|
|
5
|
+
"chunkSize": 500,
|
|
6
|
+
"overlap": 50,
|
|
7
|
+
"chunks": []
|
|
8
|
+
},
|
|
9
|
+
{
|
|
10
|
+
"name": "whitespace only",
|
|
11
|
+
"text": " \n\t ",
|
|
12
|
+
"chunkSize": 500,
|
|
13
|
+
"overlap": 50,
|
|
14
|
+
"chunks": []
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
"name": "short text",
|
|
18
|
+
"text": "Refunds are issued within 30 days.",
|
|
19
|
+
"chunkSize": 500,
|
|
20
|
+
"overlap": 50,
|
|
21
|
+
"chunks": ["Refunds are issued within 30 days."]
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"name": "sliding window",
|
|
25
|
+
"text": "w1 w2 w3 w4 w5 w6 w7 w8 w9 w10 w11 w12",
|
|
26
|
+
"chunkSize": 5,
|
|
27
|
+
"overlap": 2,
|
|
28
|
+
"chunks": ["w1 w2 w3 w4 w5", "w4 w5 w6 w7 w8", "w7 w8 w9 w10 w11", "w10 w11 w12"]
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
"name": "last window ends exactly",
|
|
32
|
+
"text": "w1 w2 w3 w4 w5 w6 w7 w8",
|
|
33
|
+
"chunkSize": 5,
|
|
34
|
+
"overlap": 2,
|
|
35
|
+
"chunks": ["w1 w2 w3 w4 w5", "w4 w5 w6 w7 w8"]
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"name": "no overlap",
|
|
39
|
+
"text": "w1 w2 w3 w4 w5 w6 w7 w8 w9 w10",
|
|
40
|
+
"chunkSize": 4,
|
|
41
|
+
"overlap": 0,
|
|
42
|
+
"chunks": ["w1 w2 w3 w4", "w5 w6 w7 w8", "w9 w10"]
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"name": "mixed whitespace collapses",
|
|
46
|
+
"text": "a\n\nb\t c d\r\ne",
|
|
47
|
+
"chunkSize": 500,
|
|
48
|
+
"overlap": 50,
|
|
49
|
+
"chunks": ["a b c d e"]
|
|
50
|
+
},
|
|
51
|
+
{
|
|
52
|
+
"name": "no-break and ideographic spaces",
|
|
53
|
+
"text": "a b c d",
|
|
54
|
+
"chunkSize": 2,
|
|
55
|
+
"overlap": 0,
|
|
56
|
+
"chunks": ["a b", "c d"]
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
"name": "byte order mark",
|
|
60
|
+
"text": "alpha betagamma",
|
|
61
|
+
"chunkSize": 500,
|
|
62
|
+
"overlap": 50,
|
|
63
|
+
"chunks": ["alpha beta gamma"]
|
|
64
|
+
},
|
|
65
|
+
{
|
|
66
|
+
"name": "next line control",
|
|
67
|
+
"text": "alpha
beta",
|
|
68
|
+
"chunkSize": 500,
|
|
69
|
+
"overlap": 50,
|
|
70
|
+
"chunks": ["alpha
beta"]
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"name": "file separator control",
|
|
74
|
+
"text": "alpha\u001cbeta",
|
|
75
|
+
"chunkSize": 500,
|
|
76
|
+
"overlap": 50,
|
|
77
|
+
"chunks": ["alpha\u001cbeta"]
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
"name": "accents and emoji",
|
|
81
|
+
"text": "café naïve 返金 🚀 ok",
|
|
82
|
+
"chunkSize": 2,
|
|
83
|
+
"overlap": 1,
|
|
84
|
+
"chunks": ["café naïve", "naïve 返金", "返金 🚀", "🚀 ok"]
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
"name": "unspaced CJK stays one word",
|
|
88
|
+
"text": "返金は三十日以内に行われます。配送には二営業日かかります。",
|
|
89
|
+
"chunkSize": 5,
|
|
90
|
+
"overlap": 1,
|
|
91
|
+
"chunks": ["返金は三十日以内に行われます。配送には二営業日かかります。"]
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
"name": "overlap not smaller than chunk size",
|
|
95
|
+
"text": "w1 w2 w3",
|
|
96
|
+
"chunkSize": 2,
|
|
97
|
+
"overlap": 2,
|
|
98
|
+
"error": true
|
|
99
|
+
}
|
|
100
|
+
]
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
{
|
|
2
|
+
"hashString": [
|
|
3
|
+
{
|
|
4
|
+
"input": "",
|
|
5
|
+
"sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
|
|
6
|
+
},
|
|
7
|
+
{
|
|
8
|
+
"input": "hello",
|
|
9
|
+
"sha256": "2cf24dba5fb0a30e26e83b2ac5b9e29e1b161e5c1fa7425e73043362938b9824"
|
|
10
|
+
},
|
|
11
|
+
{
|
|
12
|
+
"input": "Refunds 30 días — 返金 🚀",
|
|
13
|
+
"sha256": "8c0b6fdd3be678888cf2ab1a1f688435999e249e317f809cd444616b147dba9f"
|
|
14
|
+
},
|
|
15
|
+
{
|
|
16
|
+
"input": "line1\r\nline2",
|
|
17
|
+
"sha256": "d14a91a6d1c6ee83bf0c774ebecbee6d8b393b395dae29eea839c354d6fba9c0"
|
|
18
|
+
}
|
|
19
|
+
],
|
|
20
|
+
"namespaceFromPath": [
|
|
21
|
+
{
|
|
22
|
+
"input": "/tmp/docs/policy.pdf",
|
|
23
|
+
"namespace": "1b0f23231e019b65"
|
|
24
|
+
},
|
|
25
|
+
{
|
|
26
|
+
"input": "C:\\docs\\policy.pdf",
|
|
27
|
+
"namespace": "b0ea2bdcc7ddf050"
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"input": "https://example.com/policy",
|
|
31
|
+
"namespace": "fd4f660e25200691"
|
|
32
|
+
}
|
|
33
|
+
]
|
|
34
|
+
}
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"""
|
|
2
2
|
Regression tests for URL indexing and shared vector stores across documents.
|
|
3
3
|
"""
|
|
4
|
+
import json
|
|
4
5
|
import shutil
|
|
5
6
|
import tempfile
|
|
6
7
|
from pathlib import Path
|
|
@@ -8,6 +9,7 @@ from unittest.mock import patch
|
|
|
8
9
|
|
|
9
10
|
import pytest
|
|
10
11
|
|
|
12
|
+
from raglite.constants import INDEX_FORMAT_VERSION
|
|
11
13
|
from raglite.core.collection import DocumentCollection
|
|
12
14
|
from raglite.core.document import Document
|
|
13
15
|
from raglite.embeddings.local import LocalEmbedder
|
|
@@ -208,3 +210,45 @@ def test_collection_search_raises_when_every_document_fails(tmp_dir, embedder_co
|
|
|
208
210
|
):
|
|
209
211
|
with pytest.raises(RuntimeError, match="bad api key"):
|
|
210
212
|
collection.search("refund")
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _build_then_edit(tmp_dir, edit):
|
|
216
|
+
path = Path(tmp_dir) / "policy.txt"
|
|
217
|
+
path.write_text("Refunds are issued within 30 days.", encoding="utf-8")
|
|
218
|
+
opts = {"storeDir": str(Path(tmp_dir) / ".raglite"), "logLevel": "silent"}
|
|
219
|
+
doc = Document(str(path), opts)
|
|
220
|
+
doc.build()
|
|
221
|
+
meta_path = Path(tmp_dir) / ".raglite" / doc.store_namespace / "metadata.json"
|
|
222
|
+
meta = json.loads(meta_path.read_text(encoding="utf-8"))
|
|
223
|
+
assert meta["formatVersion"] == INDEX_FORMAT_VERSION
|
|
224
|
+
edit(meta)
|
|
225
|
+
meta_path.write_text(json.dumps(meta), encoding="utf-8")
|
|
226
|
+
return Document(str(path), opts).build()
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def test_index_reused_across_package_versions(tmp_dir, embedder_configs):
|
|
230
|
+
result = _build_then_edit(tmp_dir, lambda m: m.update(version="9.9.9"))
|
|
231
|
+
assert result["cached"] is True
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def test_index_from_1_2_1_without_format_version_is_reused(tmp_dir, embedder_configs):
|
|
235
|
+
def edit(m):
|
|
236
|
+
m["version"] = "1.2.1"
|
|
237
|
+
m.pop("formatVersion")
|
|
238
|
+
|
|
239
|
+
assert _build_then_edit(tmp_dir, edit)["cached"] is True
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def test_older_index_without_format_version_is_rebuilt(tmp_dir, embedder_configs):
|
|
243
|
+
def edit(m):
|
|
244
|
+
m["version"] = "1.2.0"
|
|
245
|
+
m.pop("formatVersion")
|
|
246
|
+
|
|
247
|
+
assert _build_then_edit(tmp_dir, edit)["cached"] is False
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def test_index_with_other_format_version_is_rebuilt(tmp_dir, embedder_configs):
|
|
251
|
+
result = _build_then_edit(
|
|
252
|
+
tmp_dir, lambda m: m.update(formatVersion=INDEX_FORMAT_VERSION + 1)
|
|
253
|
+
)
|
|
254
|
+
assert result["cached"] is False
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Cross-SDK fixtures: the TypeScript SDK generates these expected outputs and
|
|
3
|
+
both SDKs must reproduce them. Refresh with scripts/sync_shared_fixtures.py.
|
|
4
|
+
"""
|
|
5
|
+
import json
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
from raglite.chunking import RecursiveChunker
|
|
11
|
+
from raglite.errors import ChunkingError
|
|
12
|
+
from raglite.utils.hash import hash_string, namespace_from_path
|
|
13
|
+
|
|
14
|
+
FIXTURES = Path(__file__).resolve().parent.parent / "fixtures" / "shared"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _load(name):
|
|
18
|
+
return json.loads((FIXTURES / name).read_text(encoding="utf-8"))
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
CHUNKER = _load("chunker.json")
|
|
22
|
+
HASH = _load("hash.json")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@pytest.mark.parametrize("case", CHUNKER, ids=[c["name"] for c in CHUNKER])
|
|
26
|
+
def test_chunker_matches_typescript(case):
|
|
27
|
+
chunker = RecursiveChunker(case["chunkSize"], case["overlap"])
|
|
28
|
+
if case.get("error"):
|
|
29
|
+
with pytest.raises(ChunkingError):
|
|
30
|
+
chunker.split(case["text"])
|
|
31
|
+
else:
|
|
32
|
+
assert chunker.split(case["text"]) == case["chunks"]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@pytest.mark.parametrize("case", HASH["hashString"], ids=lambda c: repr(c["input"]))
|
|
36
|
+
def test_hash_string_matches_typescript(case):
|
|
37
|
+
assert hash_string(case["input"]) == case["sha256"]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@pytest.mark.parametrize("case", HASH["namespaceFromPath"], ids=lambda c: c["input"])
|
|
41
|
+
def test_namespace_matches_typescript(case):
|
|
42
|
+
assert namespace_from_path(case["input"]) == case["namespace"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|