raglite-toolkit 1.2.0__tar.gz → 1.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/.github/workflows/pr-verify.yml +9 -0
  2. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/.gitignore +1 -0
  3. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/ARCHITECTURE.md +15 -10
  4. raglite_toolkit-1.2.2/CHANGELOG.md +76 -0
  5. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/PKG-INFO +7 -6
  6. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/README.md +5 -4
  7. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/pyproject.toml +1 -1
  8. raglite_toolkit-1.2.2/scripts/sync_shared_fixtures.py +54 -0
  9. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/chunking/recursive.py +9 -2
  10. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/config.py +5 -0
  11. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/constants.py +10 -1
  12. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/core/collection.py +21 -3
  13. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/core/document.py +45 -23
  14. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/web.py +2 -1
  15. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/types.py +3 -1
  16. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/pinecone.py +5 -1
  17. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/qdrant.py +32 -9
  18. raglite_toolkit-1.2.2/tests/fixtures/shared/chunker.json +100 -0
  19. raglite_toolkit-1.2.2/tests/fixtures/shared/hash.json +34 -0
  20. raglite_toolkit-1.2.2/tests/unit/test_regressions.py +254 -0
  21. raglite_toolkit-1.2.2/tests/unit/test_shared_fixtures.py +42 -0
  22. raglite_toolkit-1.2.0/CHANGELOG.md +0 -46
  23. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/.github/workflows/publish.yml +0 -0
  24. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/LICENSE +0 -0
  25. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/basic.py +0 -0
  26. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/custom_store_example.py +0 -0
  27. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/multi_provider.py +0 -0
  28. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/ollama_test.py +0 -0
  29. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/qdrant_example.py +0 -0
  30. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/sample.txt +0 -0
  31. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/examples/serve.py +0 -0
  32. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/scripts/pre-commit.sh +0 -0
  33. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/scripts/pre-release.sh +0 -0
  34. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/__init__.py +0 -0
  35. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/api/__init__.py +0 -0
  36. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/api/schemas.py +0 -0
  37. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/api/server.py +0 -0
  38. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/chunking/__init__.py +0 -0
  39. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/chunking/base.py +0 -0
  40. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/cli.py +0 -0
  41. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/core/__init__.py +0 -0
  42. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/__init__.py +0 -0
  43. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/base.py +0 -0
  44. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/factory.py +0 -0
  45. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/local.py +0 -0
  46. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/models.py +0 -0
  47. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/embeddings/remote.py +0 -0
  48. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/errors.py +0 -0
  49. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/__init__.py +0 -0
  50. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/answer.py +0 -0
  51. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/factory.py +0 -0
  52. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/models.py +0 -0
  53. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/llm/prompt.py +0 -0
  54. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/__init__.py +0 -0
  55. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/base.py +0 -0
  56. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/directory.py +0 -0
  57. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/docx.py +0 -0
  58. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/json.py +0 -0
  59. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/markdown.py +0 -0
  60. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/pdf.py +0 -0
  61. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/loaders/txt.py +0 -0
  62. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/retrieval/__init__.py +0 -0
  63. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/retrieval/retriever.py +0 -0
  64. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/utils/__init__.py +0 -0
  65. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/utils/hash.py +0 -0
  66. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/utils/logger.py +0 -0
  67. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/__init__.py +0 -0
  68. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/base.py +0 -0
  69. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/factory.py +0 -0
  70. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/src/raglite/vectordb/memory.py +0 -0
  71. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/__init__.py +0 -0
  72. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/conftest.py +0 -0
  73. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/__init__.py +0 -0
  74. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_api.py +0 -0
  75. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_ask.py +0 -0
  76. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_collection.py +0 -0
  77. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_document.py +0 -0
  78. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/integration/test_ollama.py +0 -0
  79. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/__init__.py +0 -0
  80. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_chunking.py +0 -0
  81. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_cli.py +0 -0
  82. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_config.py +0 -0
  83. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_directory_loader.py +0 -0
  84. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_errors.py +0 -0
  85. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_hash.py +0 -0
  86. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_loaders.py +0 -0
  87. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_prompt.py +0 -0
  88. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_retriever.py +0 -0
  89. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_vectordb.py +0 -0
  90. {raglite_toolkit-1.2.0 → raglite_toolkit-1.2.2}/tests/unit/test_web_loader.py +0 -0
@@ -34,3 +34,12 @@ jobs:
34
34
 
35
35
  - name: Run tests
36
36
  run: python -m pytest tests/ -q
37
+
38
+ - name: Check out TypeScript SDK for shared fixtures
39
+ uses: actions/checkout@v4
40
+ with:
41
+ repository: creatorpiyush/raglite
42
+ path: .ts-sdk
43
+
44
+ - name: Check shared fixtures match the TypeScript SDK
45
+ run: python scripts/sync_shared_fixtures.py --check --source .ts-sdk
@@ -42,3 +42,4 @@ coverage.xml
42
42
 
43
43
  # Logs
44
44
  *.log
45
+ .ts-sdk/
@@ -126,9 +126,9 @@ graph TD
126
126
  ### 4.1 Ingestion & Indexing
127
127
  1. `doc.build()` invokes the appropriate `BaseLoader` based on file extension (`.pdf`, `.txt`, `.json`, `.md`, `.docx`).
128
128
  2. Calculates SHA-256 hash of raw document content.
129
- 3. Checks existing `IndexMetadata` in `VectorStore`. If hash matches, skips re-indexing.
129
+ 3. Checks existing `IndexMetadata` in `VectorStore`. The cached index is reused when the index format version, content hash, chunk size, overlap and embedding provider/model all match (URL sources are hashed by their fetched text).
130
130
  4. If hash differs or force rebuild requested:
131
- - `RecursiveCharacterTextSplitter` chunks text (default size 1000, overlap 200).
131
+ - `RecursiveChunker` splits text into word-based chunks (default 500 words, 50-word overlap).
132
132
  - `EmbeddingFactory` generates normalized vectors for each chunk.
133
133
  - `VectorStore.add()` saves chunks and `VectorStore.save_index_metadata()` persists index metadata.
134
134
 
@@ -162,27 +162,32 @@ Built using **FastAPI** framework for async capabilities, automatic OpenAPI docs
162
162
  ```python
163
163
  from abc import ABC, abstractmethod
164
164
  from typing import List, Optional
165
- from raglite.types import IndexMetadata, SearchResult, StoredChunk
165
+ from raglite.types import IndexMetadata, StoredChunk
166
+ from raglite.vectordb.base import VectorSearchHit
166
167
 
167
168
  class VectorStore(ABC):
169
+ @property
168
170
  @abstractmethod
169
- def load(self, doc_id: str) -> None: ...
171
+ def namespace(self) -> str: ...
170
172
 
171
173
  @abstractmethod
172
- def reset(self, doc_id: str) -> None: ...
174
+ def load(self) -> None: ...
173
175
 
174
176
  @abstractmethod
175
- def add(self, doc_id: str, chunks: List[StoredChunk]) -> None: ...
177
+ def reset(self) -> None: ...
176
178
 
177
179
  @abstractmethod
178
- def search(self, doc_id: str, query_vector: List[float], top_k: int, min_score: Optional[float] = None) -> List[SearchResult]: ...
180
+ def add(self, chunks: List[StoredChunk]) -> None: ...
179
181
 
180
182
  @abstractmethod
181
- def count(self, doc_id: str) -> int: ...
183
+ def search(self, embedding: List[float], top_k: int) -> List[VectorSearchHit]: ...
182
184
 
183
185
  @abstractmethod
184
- def save_index_metadata(self, doc_id: str, meta: IndexMetadata) -> None: ...
186
+ def count(self) -> int: ...
185
187
 
186
188
  @abstractmethod
187
- def read_index_metadata(self, doc_id: str) -> Optional[IndexMetadata]: ...
189
+ def save_index_metadata(self, metadata: IndexMetadata) -> None: ...
190
+
191
+ @abstractmethod
192
+ def read_index_metadata(self) -> Optional[IndexMetadata]: ...
188
193
  ```
@@ -0,0 +1,76 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project will be documented in this file.
4
+
5
+ ## [1.2.2] - Unreleased
6
+
7
+ ### Changed
8
+ - **Upgrades keep cached indexes:** The build cache is now keyed on an index format version (`formatVersion` in `IndexMetadata`) instead of the package version, so upgrading RAGLite no longer re-embeds every index. Indexes built by 1.2.1 are reused as-is; indexes from older releases are rebuilt once.
9
+
10
+ ### Fixed
11
+ - **Chunker parity with the TypeScript SDK:** Words are now split on exactly the whitespace characters JavaScript treats as whitespace. Previously text containing a byte order mark (U+FEFF), U+0085 or U+001C–U+001F was chunked differently from the TypeScript SDK. Existing indexes are not rebuilt for this; pass `rebuild=True` if your sources contain those characters.
12
+ - **Docs:** ARCHITECTURE.md now describes the word-based `RecursiveChunker` (500 words, 50 overlap) and the actual `VectorStore` interface. The README no longer lists LanceDB, which is TypeScript-only.
13
+
14
+ ### Internal
15
+ - Added cross-SDK fixtures (`tests/fixtures/shared/`, copied from the TypeScript SDK with `scripts/sync_shared_fixtures.py`) that pin chunking and hashing output. CI fails if the copy is out of date.
16
+
17
+ ## [1.2.1] - 2026-10-02
18
+
19
+ ### Fixed
20
+ - **URL sources:** `Document.build()` no longer fails with `LoaderError: File does not exist` for web URLs, so URLs work on their own and inside a `DocumentCollection`.
21
+ - **Custom `VectorStore` instances:** Passing a `VectorStore` subclass instance as `{"vectorStore": store}` no longer fails config validation, so the documented custom store usage works.
22
+ - **Qdrant shared collections:** When `indexName` is set, documents sharing one Qdrant collection no longer wipe each other. `reset()`, search and index metadata are now scoped to each document's namespace instead of the whole collection. Without `indexName`, behaviour is unchanged.
23
+ - **Shared `VectorStore` instances in collections:** `DocumentCollection.build()` now raises `ConfigError` when a single `VectorStore` instance would be shared by more than one document, instead of each document silently resetting the previous one's index. Pass a vector store provider config instead.
24
+ - **Web URL re-indexing:** URL sources are now fingerprinted by their fetched content rather than the URL string, so a changed page is re-indexed on the next `build()`.
25
+ - **Query embedder after reload:** Searching an existing index in a new process now embeds queries with the provider and model the index was built with, rather than the constructor default. Configured credentials are reused when the provider matches.
26
+ - **Collection search errors:** `DocumentCollection.search()` (and `ask`/`ask_stream`) now logs per-document search failures instead of silently dropping them, and raises if every document fails.
27
+ - **Web loader User-Agent:** Now reports the actual package version.
28
+
29
+ ### Upgrade notes
30
+ - As with every release, cached indexes are rebuilt once on first `build()` because the package version is part of the cache key.
31
+ - Existing Qdrant indexes created with `indexName` are rebuilt once. Their old untagged points stay in the collection but are no longer returned by search; drop and recreate the collection to remove them.
32
+ - Each `build()` on a URL source now fetches the page to check for changes, even when the cached index is reused.
33
+ - When every document in a collection fails to search, the REST server now returns an error response (400 for RAGLite errors, 500 otherwise) instead of empty results.
34
+
35
+ ## [1.2.0] - 2026-08-02
36
+
37
+ ### Added
38
+ - **Multi-Document & Directory Ingestion (`DocumentCollection`):**
39
+ - Added `DocumentCollection` class to manage semantic indexing, multi-document retrieval, and Q&A across folders, glob patterns, web URLs, and mixed file lists.
40
+ - Parallel semantic search over collection vector stores with score-based top-$K$ merging and ranking.
41
+ - Contextual Q&A synthesis (`ask` and `ask_stream`) across multi-document collections.
42
+ - **Directory Loader (`DirectoryLoader`):**
43
+ - Recursive directory scanner (`recursive=True`) with glob pattern matching (e.g. `./docs/**/*.md`).
44
+ - Auto-detection of supported extensions (`.pdf`, `.txt`, `.md`, `.json`, `.docx`).
45
+ - Detailed error reporting and warning logs for unsupported/empty files.
46
+ - **Web Loader (`WebLoader`):**
47
+ - Native loader for fetching HTTP/HTTPS web URLs directly.
48
+ - Automatic HTML cleaning into formatted text/markdown with script, style, and SVG tag stripping.
49
+ - JSON and plain text content-type parsing.
50
+ - **CLI & FastAPI REST Server Support:**
51
+ - Upgraded `raglite index`, `search`, `ask`, and `serve` CLI commands to process directories, glob patterns, and URLs.
52
+ - Updated FastAPI REST server to support `DocumentCollection` and single `Document` targets.
53
+ - **Unit & Integration Tests:**
54
+ - Added unit and integration tests for `DirectoryLoader`, `WebLoader`, and `DocumentCollection`.
55
+
56
+ ## [1.1.0] - 2026-07-19
57
+
58
+ ### Added
59
+ - **Pluggable Vector Databases:** Added support for custom local and cloud vector database backends via a new `vectorStore` config option.
60
+ - **Memory Store (`"memory"`):** Default in-memory store persisting indexes locally to JSON (unchanged behaviour).
61
+ - **Qdrant Store (`"qdrant"`):** Wrapper for Qdrant local/cloud using stdlib `urllib` REST requests. Supports auto-collection creation, API key auth, and custom collection names.
62
+ - **Pinecone Store (`"pinecone"`):** Cloud database support using Pinecone Namespaces and stdlib `urllib` REST requests. Stores index metadata as a reserved `__metadata__` vector.
63
+ - **Custom Adapters:** Pass any class instance implementing the `VectorStore` ABC directly as `vectorStore` in `DocumentOptions`.
64
+ - **`VectorStoreProviderConfig` type:** New Pydantic model in `types.py` describing provider, URL, API key, index name, and store directory.
65
+ - **Factory:** `create_vector_store(config, namespace)` utility in `vectordb/factory.py` resolving the correct store from config.
66
+ - **Examples:**
67
+ - `examples/qdrant_example.py` — full index + search demo with Qdrant.
68
+ - `examples/custom_store_example.py` — implementing and using a custom VectorStore ABC subclass.
69
+ - **Automation Scripts:**
70
+ - `scripts/pre-commit.sh` — runs ruff lint, mypy type-check, and pytest before committing.
71
+ - `scripts/pre-release.sh` — cleans builds, runs full verification, and builds distribution packages.
72
+ - **GitHub Actions Workflow:** `pr-verify.yml` — automatically runs code style checks and the full test suite on every pull request and push to `main`/`master`.
73
+
74
+ ### Changed
75
+ - **`DocumentOptions` / `ResolvedConfig`:** Added optional `vectorStore` field supporting `VectorStoreProviderConfig` or a custom `VectorStore` instance.
76
+ - **`Document.__init__`:** Constructor now resolves the appropriate vector store from config, accepting provider configs or custom instances.
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: raglite-toolkit
3
- Version: 1.2.0
3
+ Version: 1.2.2
4
4
  Summary: Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python.
5
5
  Project-URL: Homepage, https://github.com/creatorpiyush/raglite-py
6
6
  Project-URL: Repository, https://github.com/creatorpiyush/raglite-py
@@ -172,7 +172,7 @@ print()
172
172
 
173
173
  ## Pluggable Vector Databases
174
174
 
175
- `raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, LanceDB, or custom subclasses):
175
+ `raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, or custom subclasses; LanceDB is currently TypeScript-only):
176
176
 
177
177
  ### Memory Store (Default)
178
178
  ```python
@@ -182,6 +182,7 @@ doc = Document("./policy.pdf", {
182
182
  ```
183
183
 
184
184
  ### Qdrant Store
185
+ By default each document gets its own collection (`raglite_<namespace>`). Set `indexName` to keep several documents in one shared collection; each document's points are tagged with its namespace, so rebuilding one document never affects the others.
185
186
  ```python
186
187
  doc = Document("./policy.pdf", {
187
188
  "vectorStore": {
@@ -335,7 +336,7 @@ Document("./policy.pdf", {
335
336
 
336
337
  ## How Caching Works
337
338
 
338
- Every `build()` call fingerprints the source file with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
339
+ Every `build()` call fingerprints the source (file bytes, or the fetched text for URLs) with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
339
340
 
340
341
  | Factor | Triggers rebuild if changed |
341
342
  |--------|-----------------------------|
@@ -343,7 +344,7 @@ Every `build()` call fingerprints the source file with a **SHA-256 content hash*
343
344
  | Chunk size | `chunkSize` changed |
344
345
  | Overlap | `overlap` changed |
345
346
  | Embedding provider/model | Provider or model string changed |
346
- | Library version | Package version bumped |
347
+ | Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
347
348
 
348
349
  Pass `rebuild=True` to `build()` to force a fresh index regardless.
349
350
 
@@ -359,7 +360,7 @@ Each document is stored under `.raglite/<sha256-prefix>/`, so multiple documents
359
360
  from raglite.vectordb.base import VectorStore
360
361
 
361
362
  class MyVectorStore(VectorStore):
362
- # Implement: load, reset, add, search, count,
363
+ # Implement: namespace (property), load, reset, add, search, count,
363
364
  # save_index_metadata, read_index_metadata
364
365
  ...
365
366
  ```
@@ -126,7 +126,7 @@ print()
126
126
 
127
127
  ## Pluggable Vector Databases
128
128
 
129
- `raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, LanceDB, or custom subclasses):
129
+ `raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, or custom subclasses; LanceDB is currently TypeScript-only):
130
130
 
131
131
  ### Memory Store (Default)
132
132
  ```python
@@ -136,6 +136,7 @@ doc = Document("./policy.pdf", {
136
136
  ```
137
137
 
138
138
  ### Qdrant Store
139
+ By default each document gets its own collection (`raglite_<namespace>`). Set `indexName` to keep several documents in one shared collection; each document's points are tagged with its namespace, so rebuilding one document never affects the others.
139
140
  ```python
140
141
  doc = Document("./policy.pdf", {
141
142
  "vectorStore": {
@@ -289,7 +290,7 @@ Document("./policy.pdf", {
289
290
 
290
291
  ## How Caching Works
291
292
 
292
- Every `build()` call fingerprints the source file with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
293
+ Every `build()` call fingerprints the source (file bytes, or the fetched text for URLs) with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
293
294
 
294
295
  | Factor | Triggers rebuild if changed |
295
296
  |--------|-----------------------------|
@@ -297,7 +298,7 @@ Every `build()` call fingerprints the source file with a **SHA-256 content hash*
297
298
  | Chunk size | `chunkSize` changed |
298
299
  | Overlap | `overlap` changed |
299
300
  | Embedding provider/model | Provider or model string changed |
300
- | Library version | Package version bumped |
301
+ | Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
301
302
 
302
303
  Pass `rebuild=True` to `build()` to force a fresh index regardless.
303
304
 
@@ -313,7 +314,7 @@ Each document is stored under `.raglite/<sha256-prefix>/`, so multiple documents
313
314
  from raglite.vectordb.base import VectorStore
314
315
 
315
316
  class MyVectorStore(VectorStore):
316
- # Implement: load, reset, add, search, count,
317
+ # Implement: namespace (property), load, reset, add, search, count,
317
318
  # save_index_metadata, read_index_metadata
318
319
  ...
319
320
  ```
@@ -8,7 +8,7 @@ packages = ["src/raglite"]
8
8
 
9
9
  [project]
10
10
  name = "raglite-toolkit"
11
- version = "1.2.0"
11
+ version = "1.2.2"
12
12
  description = "Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python."
13
13
  readme = "README.md"
14
14
  requires-python = ">=3.10"
@@ -0,0 +1,54 @@
1
+ """Copy or check the cross-SDK fixtures shared with the TypeScript SDK.
2
+
3
+ The TypeScript repo (raglite) is the source of truth: it generates
4
+ tests/fixtures/shared/*.json with scripts/generate-shared-fixtures.ts. This
5
+ repo keeps an identical copy so both SDKs are tested against the same
6
+ expected outputs.
7
+
8
+ python scripts/sync_shared_fixtures.py # copy from ../raglite
9
+ python scripts/sync_shared_fixtures.py --check # exit 1 if out of date
10
+ python scripts/sync_shared_fixtures.py --source /path/to/raglite
11
+ """
12
+ import argparse
13
+ import filecmp
14
+ import shutil
15
+ import sys
16
+ from pathlib import Path
17
+
18
+ REPO_ROOT = Path(__file__).resolve().parent.parent
19
+ TARGET = REPO_ROOT / "tests" / "fixtures" / "shared"
20
+
21
+
22
+ def main() -> int:
23
+ parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
24
+ parser.add_argument("--source", type=Path, default=REPO_ROOT.parent / "raglite")
25
+ parser.add_argument("--check", action="store_true")
26
+ args = parser.parse_args()
27
+
28
+ source = args.source / "tests" / "fixtures" / "shared"
29
+ files = sorted(source.glob("*.json"))
30
+ if not files:
31
+ print(f"No shared fixtures found in {source}", file=sys.stderr)
32
+ return 1
33
+
34
+ stale = [f.name for f in files if not (TARGET / f.name).exists() or not filecmp.cmp(f, TARGET / f.name, shallow=False)]
35
+ extra = sorted({p.name for p in TARGET.glob("*.json")} - {f.name for f in files})
36
+
37
+ if args.check:
38
+ for name in stale:
39
+ print(f"out of date: {name}", file=sys.stderr)
40
+ for name in extra:
41
+ print(f"not in source: {name}", file=sys.stderr)
42
+ return 1 if stale or extra else 0
43
+
44
+ TARGET.mkdir(parents=True, exist_ok=True)
45
+ for f in files:
46
+ shutil.copyfile(f, TARGET / f.name)
47
+ for name in extra:
48
+ (TARGET / name).unlink()
49
+ print(f"Synced {len(files)} fixture(s) from {source}")
50
+ return 0
51
+
52
+
53
+ if __name__ == "__main__":
54
+ sys.exit(main())
@@ -4,17 +4,24 @@ from typing import List
4
4
  from ..errors import ChunkingError
5
5
  from .base import BaseChunker
6
6
 
7
+ # The exact set JavaScript's \s matches. Python's \s differs (it includes
8
+ # \x1c-\x1f and \x85 but not \ufeff), which would make the two SDKs chunk the
9
+ # same text differently.
10
+ _JS_WHITESPACE = re.compile(
11
+ "[\t\n\v\f\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff]+"
12
+ )
13
+
7
14
 
8
15
  class RecursiveChunker(BaseChunker):
9
16
  def split(self, text: str) -> List[str]:
10
- if not text.strip():
17
+ if not _JS_WHITESPACE.sub("", text):
11
18
  return []
12
19
  if self.overlap >= self.chunk_size:
13
20
  raise ChunkingError(
14
21
  f"overlap ({self.overlap}) must be smaller than chunkSize ({self.chunk_size})"
15
22
  )
16
23
 
17
- words = [w for w in re.split(r"\s+", text) if w]
24
+ words = [w for w in _JS_WHITESPACE.split(text) if w]
18
25
  if len(words) <= self.chunk_size:
19
26
  return [" ".join(words)]
20
27
 
@@ -10,6 +10,7 @@ from .constants import (
10
10
  DEFAULT_TOP_K,
11
11
  )
12
12
  from .types import EmbeddingProviderConfig, LLMProviderConfig, VectorStoreProviderConfig
13
+ from .vectordb.base import VectorStore
13
14
 
14
15
 
15
16
  class DocumentOptions(BaseModel):
@@ -44,6 +45,10 @@ def resolve_config(options: Optional[Union[DocumentOptions, Dict[str, Any]]] = N
44
45
  if options is None:
45
46
  opts = DocumentOptions()
46
47
  elif isinstance(options, dict):
48
+ # A VectorStore instance is read straight from the raw options by
49
+ # Document, so keep it out of the provider-config validation.
50
+ if isinstance(options.get("vectorStore"), VectorStore):
51
+ options = {k: v for k, v in options.items() if k != "vectorStore"}
47
52
  opts = DocumentOptions.model_validate(options)
48
53
  else:
49
54
  opts = options
@@ -5,7 +5,16 @@ try:
5
5
  PACKAGE_VERSION = importlib.metadata.version("raglite-toolkit")
6
6
  except importlib.metadata.PackageNotFoundError:
7
7
  PACKAGE_NAME = "raglite-toolkit"
8
- PACKAGE_VERSION = "1.2.0" # local development fallback
8
+ PACKAGE_VERSION = "1.2.2" # local development fallback
9
+
10
+ # Version of the stored index layout (chunking, ids, payloads). The build cache
11
+ # is keyed on this instead of PACKAGE_VERSION so upgrading the package does not
12
+ # re-embed every index. Bump it only when a change makes existing indexes
13
+ # incompatible.
14
+ INDEX_FORMAT_VERSION = 1
15
+
16
+ # Releases that wrote format-1 indexes before ``formatVersion`` was recorded.
17
+ LEGACY_FORMAT_1_VERSIONS = frozenset({"1.2.1"})
9
18
 
10
19
 
11
20
  SUPPORTED_EXTENSIONS = {".pdf", ".txt", ".json", ".md", ".markdown", ".docx"}
@@ -4,11 +4,12 @@ from pathlib import Path
4
4
  from typing import Any, Dict, Generator, List, Optional, Union
5
5
 
6
6
  from ..config import DocumentOptions, ResolvedConfig, resolve_config
7
- from ..errors import RagLiteError
7
+ from ..errors import ConfigError, RagLiteError
8
8
  from ..llm import generate_answer, stream_answer
9
9
  from ..loaders import DirectoryLoader, is_supported_file, is_url
10
10
  from ..types import AnswerResult, SearchResult
11
11
  from ..utils.logger import Logger, create_logger
12
+ from ..vectordb import VectorStore
12
13
  from .document import Document
13
14
 
14
15
 
@@ -91,6 +92,16 @@ class DocumentCollection:
91
92
  self.logger.warning(msg)
92
93
  errors.append({"source": src, "error": msg})
93
94
 
95
+ # A VectorStore instance has a single namespace, so every document would
96
+ # share it and each build() would reset the previous document's index.
97
+ raw_store = self.options.get("vectorStore") if isinstance(self.options, dict) else None
98
+ if isinstance(raw_store, VectorStore) and len(set(file_list)) > 1:
99
+ raise ConfigError(
100
+ "A VectorStore instance cannot be shared by multiple documents in a "
101
+ "DocumentCollection. Pass a vector store provider config "
102
+ '(e.g. {"provider": "qdrant", ...}) instead.'
103
+ )
104
+
94
105
  total_chunks = 0
95
106
  cached_docs = 0
96
107
  new_docs = 0
@@ -146,12 +157,19 @@ class DocumentCollection:
146
157
  st = score_threshold if score_threshold is not None else opts.get("scoreThreshold", opts.get("score_threshold", self.config.scoreThreshold))
147
158
 
148
159
  all_hits: List[SearchResult] = []
160
+ failures: List[Exception] = []
149
161
  for doc in self.documents.values():
150
162
  try:
151
163
  hits = doc.search(query, top_k=tk * 2, score_threshold=st)
152
164
  all_hits.extend(hits)
153
- except Exception:
154
- continue
165
+ except Exception as err:
166
+ failures.append(err)
167
+ self.logger.warning(f'Search failed for document "{doc.file_path}": {err}')
168
+
169
+ # One broken document should not hide results from the others, but if
170
+ # every document failed the caller needs the error, not an empty list.
171
+ if len(failures) == len(self.documents):
172
+ raise failures[0]
155
173
 
156
174
  all_hits.sort(key=lambda h: h.score, reverse=True)
157
175
  return all_hits[:tk]
@@ -4,13 +4,19 @@ from typing import Any, Dict, Generator, Optional, Union
4
4
 
5
5
  from ..chunking import RecursiveChunker
6
6
  from ..config import DocumentOptions, ResolvedConfig, resolve_config
7
- from ..constants import PACKAGE_VERSION
7
+ from ..constants import INDEX_FORMAT_VERSION, LEGACY_FORMAT_1_VERSIONS, PACKAGE_VERSION
8
8
  from ..embeddings import Embedder, create_embedder
9
9
  from ..errors import FileNotIndexedError, LoaderError, RagLiteError
10
10
  from ..llm import generate_answer, stream_answer
11
11
  from ..loaders import get_loader, is_url
12
12
  from ..retrieval import Retriever
13
- from ..types import AnswerResult, ChunkMetadata, IndexMetadata, StoredChunk
13
+ from ..types import (
14
+ AnswerResult,
15
+ ChunkMetadata,
16
+ EmbeddingProviderConfig,
17
+ IndexMetadata,
18
+ StoredChunk,
19
+ )
14
20
  from ..utils.hash import hash_file, hash_string, namespace_from_path
15
21
  from ..utils.logger import create_logger
16
22
  from ..vectordb import MemoryVectorStore, VectorStore, create_vector_store
@@ -74,16 +80,17 @@ class Document:
74
80
  if should_rebuild is None:
75
81
  should_rebuild = opts.get("rebuild", False)
76
82
 
77
- if not os.path.exists(self.file_path):
83
+ if not is_url(self.file_path) and not os.path.exists(self.file_path):
78
84
  raise LoaderError(f"File does not exist: {self.file_path}")
79
85
 
80
86
  self.store.load()
81
87
  existing = self.store.read_index_metadata()
82
- source_hash = (
83
- hash_string(self.file_path)
84
- if is_url(self.file_path)
85
- else hash_file(self.file_path)
86
- )
88
+ # Web pages change without their URL changing, so fingerprint the
89
+ # fetched content rather than the URL.
90
+ text: Optional[str] = None
91
+ if is_url(self.file_path):
92
+ text = get_loader(self.file_path).load()
93
+ source_hash = hash_string(text) if text is not None else hash_file(self.file_path)
87
94
 
88
95
  if (
89
96
  not should_rebuild
@@ -110,8 +117,8 @@ class Document:
110
117
  self.store.reset()
111
118
  self.store.load()
112
119
 
113
- loader = get_loader(self.file_path)
114
- text = loader.load()
120
+ if text is None:
121
+ text = get_loader(self.file_path).load()
115
122
  if not text:
116
123
  raise LoaderError(
117
124
  f"Loader returned empty text for {self.file_path}"
@@ -161,6 +168,7 @@ class Document:
161
168
  from datetime import timezone
162
169
  metadata = IndexMetadata(
163
170
  version=PACKAGE_VERSION,
171
+ formatVersion=INDEX_FORMAT_VERSION,
164
172
  source=self.file_path,
165
173
  sourceHash=source_hash,
166
174
  chunkSize=c_size,
@@ -340,19 +348,10 @@ class Document:
340
348
  f'No RagLite index found for "{self.file_path}". Call build() first.'
341
349
  )
342
350
 
343
- embed_config = self.config.embeddings
344
- if embed_config.model is None:
345
- from ..types import EmbeddingProviderConfig
346
-
347
- embed_config = EmbeddingProviderConfig(
348
- provider=embed_config.provider,
349
- model=existing.embeddingModel,
350
- apiKey=embed_config.apiKey,
351
- baseURL=embed_config.baseURL,
352
- )
353
-
354
351
  if self.embedder is None:
355
- self.embedder = create_embedder(embed_config)
352
+ self.embedder = create_embedder(
353
+ _query_embeddings_config(existing, self.config.embeddings)
354
+ )
356
355
  self.ready = True
357
356
 
358
357
  def _cache_still_valid(
@@ -363,7 +362,7 @@ class Document:
363
362
  overlap: int,
364
363
  embeddings_config: Any,
365
364
  ) -> bool:
366
- if existing.version != PACKAGE_VERSION:
365
+ if index_format_version(existing) != INDEX_FORMAT_VERSION:
367
366
  return False
368
367
  if existing.sourceHash != source_hash:
369
368
  return False
@@ -378,3 +377,26 @@ class Document:
378
377
  if req_model is not None and req_model != existing.embeddingModel:
379
378
  return False
380
379
  return True
380
+
381
+
382
+ def index_format_version(metadata: IndexMetadata) -> Optional[int]:
383
+ """Indexes written before ``formatVersion`` existed are identified by package version."""
384
+ if metadata.formatVersion is not None:
385
+ return metadata.formatVersion
386
+ return 1 if metadata.version in LEGACY_FORMAT_1_VERSIONS else None
387
+
388
+
389
+ def _query_embeddings_config(
390
+ existing: IndexMetadata, configured: EmbeddingProviderConfig
391
+ ) -> EmbeddingProviderConfig:
392
+ """Queries must be embedded with the same provider and model as the index,
393
+ which may differ from the constructor default when ``build(embeddings=...)``
394
+ overrode it. Credentials from the configured provider are reused when it
395
+ matches; otherwise the provider falls back to its environment variables."""
396
+ same_provider = configured.provider == existing.embeddingProvider
397
+ return EmbeddingProviderConfig(
398
+ provider=existing.embeddingProvider,
399
+ model=existing.embeddingModel,
400
+ apiKey=configured.apiKey if same_provider else None,
401
+ baseURL=configured.baseURL if same_provider else None,
402
+ )
@@ -1,6 +1,7 @@
1
1
  import html
2
2
  import re
3
3
 
4
+ from ..constants import PACKAGE_VERSION
4
5
  from ..errors import LoaderError
5
6
  from .base import BaseLoader
6
7
 
@@ -18,7 +19,7 @@ class WebLoader(BaseLoader):
18
19
  import httpx
19
20
 
20
21
  headers = {
21
- "User-Agent": "RAGLite/1.2.0 (Python/3.10+)",
22
+ "User-Agent": f"RAGLite/{PACKAGE_VERSION} (Python/3.10+)",
22
23
  "Accept": "text/html,text/plain,application/xhtml+xml;q=0.9,*/*;q=0.8",
23
24
  }
24
25
  response = httpx.get(self.url, headers=headers, follow_redirects=True, timeout=15.0)
@@ -96,7 +96,9 @@ class AnswerResult(BaseModel):
96
96
  class IndexMetadata(BaseModel):
97
97
  model_config = ConfigDict(populate_by_name=True, extra="allow")
98
98
 
99
- version: str
99
+ version: str # package version that built the index (informational)
100
+ # Index layout version; see INDEX_FORMAT_VERSION. None on indexes built before 1.2.2.
101
+ formatVersion: Optional[int] = Field(default=None, alias="formatVersion")
100
102
  source: str
101
103
  sourceHash: str = Field(..., alias="sourceHash")
102
104
  chunkSize: int = Field(..., alias="chunkSize")
@@ -126,7 +126,8 @@ class PineconeVectorStore(VectorStore):
126
126
  def save_index_metadata(self, metadata: IndexMetadata) -> None:
127
127
  dim = metadata.embeddingDimensions
128
128
  zero_vec = [0.0] * dim
129
- payload = metadata.model_dump(by_alias=True)
129
+ # Pinecone rejects null metadata values.
130
+ payload = metadata.model_dump(by_alias=True, exclude_none=True)
130
131
  payload["isMetadata"] = True
131
132
  self._request(
132
133
  "POST",
@@ -154,6 +155,9 @@ class PineconeVectorStore(VectorStore):
154
155
  try:
155
156
  return IndexMetadata(
156
157
  version=m.get("version", ""),
158
+ formatVersion=(
159
+ int(m["formatVersion"]) if m.get("formatVersion") is not None else None
160
+ ),
157
161
  source=m.get("source", ""),
158
162
  sourceHash=m.get("sourceHash", ""),
159
163
  chunkSize=int(m.get("chunkSize", 0)),