raglite-toolkit 1.2.1__tar.gz → 1.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/.github/workflows/pr-verify.yml +9 -0
  2. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/.gitignore +1 -0
  3. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/ARCHITECTURE.md +15 -10
  4. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/CHANGELOG.md +12 -0
  5. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/PKG-INFO +5 -5
  6. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/README.md +4 -4
  7. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/pyproject.toml +1 -1
  8. raglite_toolkit-1.2.2/scripts/sync_shared_fixtures.py +54 -0
  9. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/chunking/recursive.py +9 -2
  10. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/constants.py +10 -1
  11. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/core/document.py +10 -2
  12. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/types.py +3 -1
  13. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/pinecone.py +5 -1
  14. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/qdrant.py +1 -0
  15. raglite_toolkit-1.2.2/tests/fixtures/shared/chunker.json +100 -0
  16. raglite_toolkit-1.2.2/tests/fixtures/shared/hash.json +34 -0
  17. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_regressions.py +44 -0
  18. raglite_toolkit-1.2.2/tests/unit/test_shared_fixtures.py +42 -0
  19. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/.github/workflows/publish.yml +0 -0
  20. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/LICENSE +0 -0
  21. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/basic.py +0 -0
  22. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/custom_store_example.py +0 -0
  23. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/multi_provider.py +0 -0
  24. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/ollama_test.py +0 -0
  25. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/qdrant_example.py +0 -0
  26. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/sample.txt +0 -0
  27. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/examples/serve.py +0 -0
  28. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/scripts/pre-commit.sh +0 -0
  29. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/scripts/pre-release.sh +0 -0
  30. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/__init__.py +0 -0
  31. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/api/__init__.py +0 -0
  32. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/api/schemas.py +0 -0
  33. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/api/server.py +0 -0
  34. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/chunking/__init__.py +0 -0
  35. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/chunking/base.py +0 -0
  36. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/cli.py +0 -0
  37. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/config.py +0 -0
  38. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/core/__init__.py +0 -0
  39. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/core/collection.py +0 -0
  40. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/__init__.py +0 -0
  41. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/base.py +0 -0
  42. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/factory.py +0 -0
  43. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/local.py +0 -0
  44. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/models.py +0 -0
  45. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/embeddings/remote.py +0 -0
  46. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/errors.py +0 -0
  47. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/__init__.py +0 -0
  48. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/answer.py +0 -0
  49. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/factory.py +0 -0
  50. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/models.py +0 -0
  51. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/llm/prompt.py +0 -0
  52. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/__init__.py +0 -0
  53. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/base.py +0 -0
  54. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/directory.py +0 -0
  55. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/docx.py +0 -0
  56. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/json.py +0 -0
  57. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/markdown.py +0 -0
  58. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/pdf.py +0 -0
  59. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/txt.py +0 -0
  60. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/loaders/web.py +0 -0
  61. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/retrieval/__init__.py +0 -0
  62. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/retrieval/retriever.py +0 -0
  63. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/utils/__init__.py +0 -0
  64. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/utils/hash.py +0 -0
  65. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/utils/logger.py +0 -0
  66. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/__init__.py +0 -0
  67. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/base.py +0 -0
  68. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/factory.py +0 -0
  69. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/src/raglite/vectordb/memory.py +0 -0
  70. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/__init__.py +0 -0
  71. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/conftest.py +0 -0
  72. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/__init__.py +0 -0
  73. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_api.py +0 -0
  74. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_ask.py +0 -0
  75. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_collection.py +0 -0
  76. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_document.py +0 -0
  77. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/integration/test_ollama.py +0 -0
  78. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/__init__.py +0 -0
  79. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_chunking.py +0 -0
  80. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_cli.py +0 -0
  81. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_config.py +0 -0
  82. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_directory_loader.py +0 -0
  83. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_errors.py +0 -0
  84. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_hash.py +0 -0
  85. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_loaders.py +0 -0
  86. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_prompt.py +0 -0
  87. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_retriever.py +0 -0
  88. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_vectordb.py +0 -0
  89. {raglite_toolkit-1.2.1 → raglite_toolkit-1.2.2}/tests/unit/test_web_loader.py +0 -0
@@ -34,3 +34,12 @@ jobs:
34
34
 
35
35
  - name: Run tests
36
36
  run: python -m pytest tests/ -q
37
+
38
+ - name: Check out TypeScript SDK for shared fixtures
39
+ uses: actions/checkout@v4
40
+ with:
41
+ repository: creatorpiyush/raglite
42
+ path: .ts-sdk
43
+
44
+ - name: Check shared fixtures match the TypeScript SDK
45
+ run: python scripts/sync_shared_fixtures.py --check --source .ts-sdk
@@ -42,3 +42,4 @@ coverage.xml
42
42
 
43
43
  # Logs
44
44
  *.log
45
+ .ts-sdk/
@@ -126,9 +126,9 @@ graph TD
126
126
  ### 4.1 Ingestion & Indexing
127
127
  1. `doc.build()` invokes the appropriate `BaseLoader` based on file extension (`.pdf`, `.txt`, `.json`, `.md`, `.docx`).
128
128
  2. Calculates SHA-256 hash of raw document content.
129
- 3. Checks existing `IndexMetadata` in `VectorStore`. If hash matches, skips re-indexing.
129
+ 3. Checks existing `IndexMetadata` in `VectorStore`. The cached index is reused when the index format version, content hash, chunk size, overlap and embedding provider/model all match (URL sources are hashed by their fetched text).
130
130
  4. If hash differs or force rebuild requested:
131
- - `RecursiveCharacterTextSplitter` chunks text (default size 1000, overlap 200).
131
+ - `RecursiveChunker` splits text into word-based chunks (default 500 words, 50-word overlap).
132
132
  - `EmbeddingFactory` generates normalized vectors for each chunk.
133
133
  - `VectorStore.add()` saves chunks and `VectorStore.save_index_metadata()` persists index metadata.
134
134
 
@@ -162,27 +162,32 @@ Built using **FastAPI** framework for async capabilities, automatic OpenAPI docs
162
162
  ```python
163
163
  from abc import ABC, abstractmethod
164
164
  from typing import List, Optional
165
- from raglite.types import IndexMetadata, SearchResult, StoredChunk
165
+ from raglite.types import IndexMetadata, StoredChunk
166
+ from raglite.vectordb.base import VectorSearchHit
166
167
 
167
168
  class VectorStore(ABC):
169
+ @property
168
170
  @abstractmethod
169
- def load(self, doc_id: str) -> None: ...
171
+ def namespace(self) -> str: ...
170
172
 
171
173
  @abstractmethod
172
- def reset(self, doc_id: str) -> None: ...
174
+ def load(self) -> None: ...
173
175
 
174
176
  @abstractmethod
175
- def add(self, doc_id: str, chunks: List[StoredChunk]) -> None: ...
177
+ def reset(self) -> None: ...
176
178
 
177
179
  @abstractmethod
178
- def search(self, doc_id: str, query_vector: List[float], top_k: int, min_score: Optional[float] = None) -> List[SearchResult]: ...
180
+ def add(self, chunks: List[StoredChunk]) -> None: ...
179
181
 
180
182
  @abstractmethod
181
- def count(self, doc_id: str) -> int: ...
183
+ def search(self, embedding: List[float], top_k: int) -> List[VectorSearchHit]: ...
182
184
 
183
185
  @abstractmethod
184
- def save_index_metadata(self, doc_id: str, meta: IndexMetadata) -> None: ...
186
+ def count(self) -> int: ...
185
187
 
186
188
  @abstractmethod
187
- def read_index_metadata(self, doc_id: str) -> Optional[IndexMetadata]: ...
189
+ def save_index_metadata(self, metadata: IndexMetadata) -> None: ...
190
+
191
+ @abstractmethod
192
+ def read_index_metadata(self) -> Optional[IndexMetadata]: ...
188
193
  ```
@@ -2,6 +2,18 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file.
4
4
 
5
+ ## [1.2.2] - Unreleased
6
+
7
+ ### Changed
8
+ - **Upgrades keep cached indexes:** The build cache is now keyed on an index format version (`formatVersion` in `IndexMetadata`) instead of the package version, so upgrading RAGLite no longer re-embeds every index. Indexes built by 1.2.1 are reused as-is; indexes from older releases are rebuilt once.
9
+
10
+ ### Fixed
11
+ - **Chunker parity with the TypeScript SDK:** Words are now split on exactly the whitespace characters JavaScript treats as whitespace. Previously text containing a byte order mark (U+FEFF), U+0085 or U+001C–U+001F was chunked differently from the TypeScript SDK. Existing indexes are not rebuilt for this; pass `rebuild=True` if your sources contain those characters.
12
+ - **Docs:** ARCHITECTURE.md now describes the word-based `RecursiveChunker` (500 words, 50 overlap) and the actual `VectorStore` interface. The README no longer lists LanceDB, which is TypeScript-only.
13
+
14
+ ### Internal
15
+ - Added cross-SDK fixtures (`tests/fixtures/shared/`, copied from the TypeScript SDK with `scripts/sync_shared_fixtures.py`) that pin chunking and hashing output. CI fails if the copy is out of date.
16
+
5
17
  ## [1.2.1] - 2026-10-02
6
18
 
7
19
  ### Fixed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: raglite-toolkit
3
- Version: 1.2.1
3
+ Version: 1.2.2
4
4
  Summary: Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python.
5
5
  Project-URL: Homepage, https://github.com/creatorpiyush/raglite-py
6
6
  Project-URL: Repository, https://github.com/creatorpiyush/raglite-py
@@ -172,7 +172,7 @@ print()
172
172
 
173
173
  ## Pluggable Vector Databases
174
174
 
175
- `raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, LanceDB, or custom subclasses):
175
+ `raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, or custom subclasses; LanceDB is currently TypeScript-only):
176
176
 
177
177
  ### Memory Store (Default)
178
178
  ```python
@@ -336,7 +336,7 @@ Document("./policy.pdf", {
336
336
 
337
337
  ## How Caching Works
338
338
 
339
- Every `build()` call fingerprints the source file with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
339
+ Every `build()` call fingerprints the source (file bytes, or the fetched text for URLs) with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
340
340
 
341
341
  | Factor | Triggers rebuild if changed |
342
342
  |--------|-----------------------------|
@@ -344,7 +344,7 @@ Every `build()` call fingerprints the source file with a **SHA-256 content hash*
344
344
  | Chunk size | `chunkSize` changed |
345
345
  | Overlap | `overlap` changed |
346
346
  | Embedding provider/model | Provider or model string changed |
347
- | Library version | Package version bumped |
347
+ | Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
348
348
 
349
349
  Pass `rebuild=True` to `build()` to force a fresh index regardless.
350
350
 
@@ -360,7 +360,7 @@ Each document is stored under `.raglite/<sha256-prefix>/`, so multiple documents
360
360
  from raglite.vectordb.base import VectorStore
361
361
 
362
362
  class MyVectorStore(VectorStore):
363
- # Implement: load, reset, add, search, count,
363
+ # Implement: namespace (property), load, reset, add, search, count,
364
364
  # save_index_metadata, read_index_metadata
365
365
  ...
366
366
  ```
@@ -126,7 +126,7 @@ print()
126
126
 
127
127
  ## Pluggable Vector Databases
128
128
 
129
- `raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, LanceDB, or custom subclasses):
129
+ `raglite` supports pluggable vector stores (Memory, Qdrant, Pinecone, or custom subclasses; LanceDB is currently TypeScript-only):
130
130
 
131
131
  ### Memory Store (Default)
132
132
  ```python
@@ -290,7 +290,7 @@ Document("./policy.pdf", {
290
290
 
291
291
  ## How Caching Works
292
292
 
293
- Every `build()` call fingerprints the source file with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
293
+ Every `build()` call fingerprints the source (file bytes, or the fetched text for URLs) with a **SHA-256 content hash** and persists it alongside the vectors. The cached index is reused only if **all** of the following match the stored index:
294
294
 
295
295
  | Factor | Triggers rebuild if changed |
296
296
  |--------|-----------------------------|
@@ -298,7 +298,7 @@ Every `build()` call fingerprints the source file with a **SHA-256 content hash*
298
298
  | Chunk size | `chunkSize` changed |
299
299
  | Overlap | `overlap` changed |
300
300
  | Embedding provider/model | Provider or model string changed |
301
- | Library version | Package version bumped |
301
+ | Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
302
302
 
303
303
  Pass `rebuild=True` to `build()` to force a fresh index regardless.
304
304
 
@@ -314,7 +314,7 @@ Each document is stored under `.raglite/<sha256-prefix>/`, so multiple documents
314
314
  from raglite.vectordb.base import VectorStore
315
315
 
316
316
  class MyVectorStore(VectorStore):
317
- # Implement: load, reset, add, search, count,
317
+ # Implement: namespace (property), load, reset, add, search, count,
318
318
  # save_index_metadata, read_index_metadata
319
319
  ...
320
320
  ```
@@ -8,7 +8,7 @@ packages = ["src/raglite"]
8
8
 
9
9
  [project]
10
10
  name = "raglite-toolkit"
11
- version = "1.2.1"
11
+ version = "1.2.2"
12
12
  description = "Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python."
13
13
  readme = "README.md"
14
14
  requires-python = ">=3.10"
@@ -0,0 +1,54 @@
1
+ """Copy or check the cross-SDK fixtures shared with the TypeScript SDK.
2
+
3
+ The TypeScript repo (raglite) is the source of truth: it generates
4
+ tests/fixtures/shared/*.json with scripts/generate-shared-fixtures.ts. This
5
+ repo keeps an identical copy so both SDKs are tested against the same
6
+ expected outputs.
7
+
8
+ python scripts/sync_shared_fixtures.py # copy from ../raglite
9
+ python scripts/sync_shared_fixtures.py --check # exit 1 if out of date
10
+ python scripts/sync_shared_fixtures.py --source /path/to/raglite
11
+ """
12
+ import argparse
13
+ import filecmp
14
+ import shutil
15
+ import sys
16
+ from pathlib import Path
17
+
18
+ REPO_ROOT = Path(__file__).resolve().parent.parent
19
+ TARGET = REPO_ROOT / "tests" / "fixtures" / "shared"
20
+
21
+
22
+ def main() -> int:
23
+ parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
24
+ parser.add_argument("--source", type=Path, default=REPO_ROOT.parent / "raglite")
25
+ parser.add_argument("--check", action="store_true")
26
+ args = parser.parse_args()
27
+
28
+ source = args.source / "tests" / "fixtures" / "shared"
29
+ files = sorted(source.glob("*.json"))
30
+ if not files:
31
+ print(f"No shared fixtures found in {source}", file=sys.stderr)
32
+ return 1
33
+
34
+ stale = [f.name for f in files if not (TARGET / f.name).exists() or not filecmp.cmp(f, TARGET / f.name, shallow=False)]
35
+ extra = sorted({p.name for p in TARGET.glob("*.json")} - {f.name for f in files})
36
+
37
+ if args.check:
38
+ for name in stale:
39
+ print(f"out of date: {name}", file=sys.stderr)
40
+ for name in extra:
41
+ print(f"not in source: {name}", file=sys.stderr)
42
+ return 1 if stale or extra else 0
43
+
44
+ TARGET.mkdir(parents=True, exist_ok=True)
45
+ for f in files:
46
+ shutil.copyfile(f, TARGET / f.name)
47
+ for name in extra:
48
+ (TARGET / name).unlink()
49
+ print(f"Synced {len(files)} fixture(s) from {source}")
50
+ return 0
51
+
52
+
53
+ if __name__ == "__main__":
54
+ sys.exit(main())
@@ -4,17 +4,24 @@ from typing import List
4
4
  from ..errors import ChunkingError
5
5
  from .base import BaseChunker
6
6
 
7
+ # The exact set JavaScript's \s matches. Python's \s differs (it includes
8
+ # \x1c-\x1f and \x85 but not \ufeff), which would make the two SDKs chunk the
9
+ # same text differently.
10
+ _JS_WHITESPACE = re.compile(
11
+ "[\t\n\v\f\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff]+"
12
+ )
13
+
7
14
 
8
15
  class RecursiveChunker(BaseChunker):
9
16
  def split(self, text: str) -> List[str]:
10
- if not text.strip():
17
+ if not _JS_WHITESPACE.sub("", text):
11
18
  return []
12
19
  if self.overlap >= self.chunk_size:
13
20
  raise ChunkingError(
14
21
  f"overlap ({self.overlap}) must be smaller than chunkSize ({self.chunk_size})"
15
22
  )
16
23
 
17
- words = [w for w in re.split(r"\s+", text) if w]
24
+ words = [w for w in _JS_WHITESPACE.split(text) if w]
18
25
  if len(words) <= self.chunk_size:
19
26
  return [" ".join(words)]
20
27
 
@@ -5,7 +5,16 @@ try:
5
5
  PACKAGE_VERSION = importlib.metadata.version("raglite-toolkit")
6
6
  except importlib.metadata.PackageNotFoundError:
7
7
  PACKAGE_NAME = "raglite-toolkit"
8
- PACKAGE_VERSION = "1.2.1" # local development fallback
8
+ PACKAGE_VERSION = "1.2.2" # local development fallback
9
+
10
+ # Version of the stored index layout (chunking, ids, payloads). The build cache
11
+ # is keyed on this instead of PACKAGE_VERSION so upgrading the package does not
12
+ # re-embed every index. Bump it only when a change makes existing indexes
13
+ # incompatible.
14
+ INDEX_FORMAT_VERSION = 1
15
+
16
+ # Releases that wrote format-1 indexes before ``formatVersion`` was recorded.
17
+ LEGACY_FORMAT_1_VERSIONS = frozenset({"1.2.1"})
9
18
 
10
19
 
11
20
  SUPPORTED_EXTENSIONS = {".pdf", ".txt", ".json", ".md", ".markdown", ".docx"}
@@ -4,7 +4,7 @@ from typing import Any, Dict, Generator, Optional, Union
4
4
 
5
5
  from ..chunking import RecursiveChunker
6
6
  from ..config import DocumentOptions, ResolvedConfig, resolve_config
7
- from ..constants import PACKAGE_VERSION
7
+ from ..constants import INDEX_FORMAT_VERSION, LEGACY_FORMAT_1_VERSIONS, PACKAGE_VERSION
8
8
  from ..embeddings import Embedder, create_embedder
9
9
  from ..errors import FileNotIndexedError, LoaderError, RagLiteError
10
10
  from ..llm import generate_answer, stream_answer
@@ -168,6 +168,7 @@ class Document:
168
168
  from datetime import timezone
169
169
  metadata = IndexMetadata(
170
170
  version=PACKAGE_VERSION,
171
+ formatVersion=INDEX_FORMAT_VERSION,
171
172
  source=self.file_path,
172
173
  sourceHash=source_hash,
173
174
  chunkSize=c_size,
@@ -361,7 +362,7 @@ class Document:
361
362
  overlap: int,
362
363
  embeddings_config: Any,
363
364
  ) -> bool:
364
- if existing.version != PACKAGE_VERSION:
365
+ if index_format_version(existing) != INDEX_FORMAT_VERSION:
365
366
  return False
366
367
  if existing.sourceHash != source_hash:
367
368
  return False
@@ -378,6 +379,13 @@ class Document:
378
379
  return True
379
380
 
380
381
 
382
+ def index_format_version(metadata: IndexMetadata) -> Optional[int]:
383
+ """Indexes written before ``formatVersion`` existed are identified by package version."""
384
+ if metadata.formatVersion is not None:
385
+ return metadata.formatVersion
386
+ return 1 if metadata.version in LEGACY_FORMAT_1_VERSIONS else None
387
+
388
+
381
389
  def _query_embeddings_config(
382
390
  existing: IndexMetadata, configured: EmbeddingProviderConfig
383
391
  ) -> EmbeddingProviderConfig:
@@ -96,7 +96,9 @@ class AnswerResult(BaseModel):
96
96
  class IndexMetadata(BaseModel):
97
97
  model_config = ConfigDict(populate_by_name=True, extra="allow")
98
98
 
99
- version: str
99
+ version: str # package version that built the index (informational)
100
+ # Index layout version; see INDEX_FORMAT_VERSION. None on indexes built before 1.2.2.
101
+ formatVersion: Optional[int] = Field(default=None, alias="formatVersion")
100
102
  source: str
101
103
  sourceHash: str = Field(..., alias="sourceHash")
102
104
  chunkSize: int = Field(..., alias="chunkSize")
@@ -126,7 +126,8 @@ class PineconeVectorStore(VectorStore):
126
126
  def save_index_metadata(self, metadata: IndexMetadata) -> None:
127
127
  dim = metadata.embeddingDimensions
128
128
  zero_vec = [0.0] * dim
129
- payload = metadata.model_dump(by_alias=True)
129
+ # Pinecone rejects null metadata values.
130
+ payload = metadata.model_dump(by_alias=True, exclude_none=True)
130
131
  payload["isMetadata"] = True
131
132
  self._request(
132
133
  "POST",
@@ -154,6 +155,9 @@ class PineconeVectorStore(VectorStore):
154
155
  try:
155
156
  return IndexMetadata(
156
157
  version=m.get("version", ""),
158
+ formatVersion=(
159
+ int(m["formatVersion"]) if m.get("formatVersion") is not None else None
160
+ ),
157
161
  source=m.get("source", ""),
158
162
  sourceHash=m.get("sourceHash", ""),
159
163
  chunkSize=int(m.get("chunkSize", 0)),
@@ -197,6 +197,7 @@ class QdrantVectorStore(VectorStore):
197
197
  try:
198
198
  return IndexMetadata(
199
199
  version=payload.get("version", ""),
200
+ formatVersion=payload.get("formatVersion"),
200
201
  source=payload.get("source", ""),
201
202
  sourceHash=payload.get("sourceHash", ""),
202
203
  chunkSize=payload.get("chunkSize", 0),
@@ -0,0 +1,100 @@
1
+ [
2
+ {
3
+ "name": "empty",
4
+ "text": "",
5
+ "chunkSize": 500,
6
+ "overlap": 50,
7
+ "chunks": []
8
+ },
9
+ {
10
+ "name": "whitespace only",
11
+ "text": " \n\t ",
12
+ "chunkSize": 500,
13
+ "overlap": 50,
14
+ "chunks": []
15
+ },
16
+ {
17
+ "name": "short text",
18
+ "text": "Refunds are issued within 30 days.",
19
+ "chunkSize": 500,
20
+ "overlap": 50,
21
+ "chunks": ["Refunds are issued within 30 days."]
22
+ },
23
+ {
24
+ "name": "sliding window",
25
+ "text": "w1 w2 w3 w4 w5 w6 w7 w8 w9 w10 w11 w12",
26
+ "chunkSize": 5,
27
+ "overlap": 2,
28
+ "chunks": ["w1 w2 w3 w4 w5", "w4 w5 w6 w7 w8", "w7 w8 w9 w10 w11", "w10 w11 w12"]
29
+ },
30
+ {
31
+ "name": "last window ends exactly",
32
+ "text": "w1 w2 w3 w4 w5 w6 w7 w8",
33
+ "chunkSize": 5,
34
+ "overlap": 2,
35
+ "chunks": ["w1 w2 w3 w4 w5", "w4 w5 w6 w7 w8"]
36
+ },
37
+ {
38
+ "name": "no overlap",
39
+ "text": "w1 w2 w3 w4 w5 w6 w7 w8 w9 w10",
40
+ "chunkSize": 4,
41
+ "overlap": 0,
42
+ "chunks": ["w1 w2 w3 w4", "w5 w6 w7 w8", "w9 w10"]
43
+ },
44
+ {
45
+ "name": "mixed whitespace collapses",
46
+ "text": "a\n\nb\t c d\r\ne",
47
+ "chunkSize": 500,
48
+ "overlap": 50,
49
+ "chunks": ["a b c d e"]
50
+ },
51
+ {
52
+ "name": "no-break and ideographic spaces",
53
+ "text": "a b c d",
54
+ "chunkSize": 2,
55
+ "overlap": 0,
56
+ "chunks": ["a b", "c d"]
57
+ },
58
+ {
59
+ "name": "byte order mark",
60
+ "text": "alpha betagamma",
61
+ "chunkSize": 500,
62
+ "overlap": 50,
63
+ "chunks": ["alpha beta gamma"]
64
+ },
65
+ {
66
+ "name": "next line control",
67
+ "text": "alpha…beta",
68
+ "chunkSize": 500,
69
+ "overlap": 50,
70
+ "chunks": ["alpha…beta"]
71
+ },
72
+ {
73
+ "name": "file separator control",
74
+ "text": "alpha\u001cbeta",
75
+ "chunkSize": 500,
76
+ "overlap": 50,
77
+ "chunks": ["alpha\u001cbeta"]
78
+ },
79
+ {
80
+ "name": "accents and emoji",
81
+ "text": "café naïve 返金 🚀 ok",
82
+ "chunkSize": 2,
83
+ "overlap": 1,
84
+ "chunks": ["café naïve", "naïve 返金", "返金 🚀", "🚀 ok"]
85
+ },
86
+ {
87
+ "name": "unspaced CJK stays one word",
88
+ "text": "返金は三十日以内に行われます。配送には二営業日かかります。",
89
+ "chunkSize": 5,
90
+ "overlap": 1,
91
+ "chunks": ["返金は三十日以内に行われます。配送には二営業日かかります。"]
92
+ },
93
+ {
94
+ "name": "overlap not smaller than chunk size",
95
+ "text": "w1 w2 w3",
96
+ "chunkSize": 2,
97
+ "overlap": 2,
98
+ "error": true
99
+ }
100
+ ]
@@ -0,0 +1,34 @@
1
+ {
2
+ "hashString": [
3
+ {
4
+ "input": "",
5
+ "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
6
+ },
7
+ {
8
+ "input": "hello",
9
+ "sha256": "2cf24dba5fb0a30e26e83b2ac5b9e29e1b161e5c1fa7425e73043362938b9824"
10
+ },
11
+ {
12
+ "input": "Refunds 30 días — 返金 🚀",
13
+ "sha256": "8c0b6fdd3be678888cf2ab1a1f688435999e249e317f809cd444616b147dba9f"
14
+ },
15
+ {
16
+ "input": "line1\r\nline2",
17
+ "sha256": "d14a91a6d1c6ee83bf0c774ebecbee6d8b393b395dae29eea839c354d6fba9c0"
18
+ }
19
+ ],
20
+ "namespaceFromPath": [
21
+ {
22
+ "input": "/tmp/docs/policy.pdf",
23
+ "namespace": "1b0f23231e019b65"
24
+ },
25
+ {
26
+ "input": "C:\\docs\\policy.pdf",
27
+ "namespace": "b0ea2bdcc7ddf050"
28
+ },
29
+ {
30
+ "input": "https://example.com/policy",
31
+ "namespace": "fd4f660e25200691"
32
+ }
33
+ ]
34
+ }
@@ -1,6 +1,7 @@
1
1
  """
2
2
  Regression tests for URL indexing and shared vector stores across documents.
3
3
  """
4
+ import json
4
5
  import shutil
5
6
  import tempfile
6
7
  from pathlib import Path
@@ -8,6 +9,7 @@ from unittest.mock import patch
8
9
 
9
10
  import pytest
10
11
 
12
+ from raglite.constants import INDEX_FORMAT_VERSION
11
13
  from raglite.core.collection import DocumentCollection
12
14
  from raglite.core.document import Document
13
15
  from raglite.embeddings.local import LocalEmbedder
@@ -208,3 +210,45 @@ def test_collection_search_raises_when_every_document_fails(tmp_dir, embedder_co
208
210
  ):
209
211
  with pytest.raises(RuntimeError, match="bad api key"):
210
212
  collection.search("refund")
213
+
214
+
215
+ def _build_then_edit(tmp_dir, edit):
216
+ path = Path(tmp_dir) / "policy.txt"
217
+ path.write_text("Refunds are issued within 30 days.", encoding="utf-8")
218
+ opts = {"storeDir": str(Path(tmp_dir) / ".raglite"), "logLevel": "silent"}
219
+ doc = Document(str(path), opts)
220
+ doc.build()
221
+ meta_path = Path(tmp_dir) / ".raglite" / doc.store_namespace / "metadata.json"
222
+ meta = json.loads(meta_path.read_text(encoding="utf-8"))
223
+ assert meta["formatVersion"] == INDEX_FORMAT_VERSION
224
+ edit(meta)
225
+ meta_path.write_text(json.dumps(meta), encoding="utf-8")
226
+ return Document(str(path), opts).build()
227
+
228
+
229
+ def test_index_reused_across_package_versions(tmp_dir, embedder_configs):
230
+ result = _build_then_edit(tmp_dir, lambda m: m.update(version="9.9.9"))
231
+ assert result["cached"] is True
232
+
233
+
234
+ def test_index_from_1_2_1_without_format_version_is_reused(tmp_dir, embedder_configs):
235
+ def edit(m):
236
+ m["version"] = "1.2.1"
237
+ m.pop("formatVersion")
238
+
239
+ assert _build_then_edit(tmp_dir, edit)["cached"] is True
240
+
241
+
242
+ def test_older_index_without_format_version_is_rebuilt(tmp_dir, embedder_configs):
243
+ def edit(m):
244
+ m["version"] = "1.2.0"
245
+ m.pop("formatVersion")
246
+
247
+ assert _build_then_edit(tmp_dir, edit)["cached"] is False
248
+
249
+
250
+ def test_index_with_other_format_version_is_rebuilt(tmp_dir, embedder_configs):
251
+ result = _build_then_edit(
252
+ tmp_dir, lambda m: m.update(formatVersion=INDEX_FORMAT_VERSION + 1)
253
+ )
254
+ assert result["cached"] is False
@@ -0,0 +1,42 @@
1
+ """
2
+ Cross-SDK fixtures: the TypeScript SDK generates these expected outputs and
3
+ both SDKs must reproduce them. Refresh with scripts/sync_shared_fixtures.py.
4
+ """
5
+ import json
6
+ from pathlib import Path
7
+
8
+ import pytest
9
+
10
+ from raglite.chunking import RecursiveChunker
11
+ from raglite.errors import ChunkingError
12
+ from raglite.utils.hash import hash_string, namespace_from_path
13
+
14
+ FIXTURES = Path(__file__).resolve().parent.parent / "fixtures" / "shared"
15
+
16
+
17
+ def _load(name):
18
+ return json.loads((FIXTURES / name).read_text(encoding="utf-8"))
19
+
20
+
21
+ CHUNKER = _load("chunker.json")
22
+ HASH = _load("hash.json")
23
+
24
+
25
+ @pytest.mark.parametrize("case", CHUNKER, ids=[c["name"] for c in CHUNKER])
26
+ def test_chunker_matches_typescript(case):
27
+ chunker = RecursiveChunker(case["chunkSize"], case["overlap"])
28
+ if case.get("error"):
29
+ with pytest.raises(ChunkingError):
30
+ chunker.split(case["text"])
31
+ else:
32
+ assert chunker.split(case["text"]) == case["chunks"]
33
+
34
+
35
+ @pytest.mark.parametrize("case", HASH["hashString"], ids=lambda c: repr(c["input"]))
36
+ def test_hash_string_matches_typescript(case):
37
+ assert hash_string(case["input"]) == case["sha256"]
38
+
39
+
40
+ @pytest.mark.parametrize("case", HASH["namespaceFromPath"], ids=lambda c: c["input"])
41
+ def test_namespace_matches_typescript(case):
42
+ assert namespace_from_path(case["input"]) == case["namespace"]
File without changes