raglite-toolkit 1.2.2__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/ARCHITECTURE.md +14 -7
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/CHANGELOG.md +26 -1
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/PKG-INFO +47 -5
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/README.md +46 -4
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/pyproject.toml +1 -1
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/__init__.py +30 -1
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/api/schemas.py +3 -1
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/api/server.py +3 -0
- raglite_toolkit-1.3.0/src/raglite/chunking/recursive.py +90 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/cli.py +51 -35
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/config.py +11 -2
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/constants.py +10 -2
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/core/collection.py +64 -30
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/core/document.py +244 -34
- raglite_toolkit-1.3.0/src/raglite/retrieval/__init__.py +14 -0
- raglite_toolkit-1.3.0/src/raglite/retrieval/fusion.py +51 -0
- raglite_toolkit-1.3.0/src/raglite/retrieval/keyword_index.py +144 -0
- raglite_toolkit-1.3.0/src/raglite/retrieval/plan.py +75 -0
- raglite_toolkit-1.3.0/src/raglite/text/__init__.py +11 -0
- raglite_toolkit-1.3.0/src/raglite/text/scripts.py +64 -0
- raglite_toolkit-1.3.0/src/raglite/text/tokenizer.py +106 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/types.py +60 -2
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/base.py +21 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/memory.py +4 -1
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/qdrant.py +34 -1
- raglite_toolkit-1.3.0/tests/fixtures/shared/bm25.json +102 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/fixtures/shared/chunker.json +32 -3
- raglite_toolkit-1.3.0/tests/fixtures/shared/rrf.json +321 -0
- raglite_toolkit-1.3.0/tests/fixtures/shared/tokenizer.json +141 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_api.py +3 -1
- raglite_toolkit-1.3.0/tests/integration/test_hybrid.py +328 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_cli.py +2 -2
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_regressions.py +130 -0
- raglite_toolkit-1.3.0/tests/unit/test_shared_fixtures.py +99 -0
- raglite_toolkit-1.2.2/src/raglite/chunking/recursive.py +0 -42
- raglite_toolkit-1.2.2/src/raglite/retrieval/__init__.py +0 -3
- raglite_toolkit-1.2.2/tests/unit/test_shared_fixtures.py +0 -42
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/.github/workflows/pr-verify.yml +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/.github/workflows/publish.yml +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/.gitignore +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/LICENSE +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/basic.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/custom_store_example.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/multi_provider.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/ollama_test.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/qdrant_example.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/sample.txt +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/serve.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/scripts/pre-commit.sh +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/scripts/pre-release.sh +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/scripts/sync_shared_fixtures.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/api/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/chunking/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/chunking/base.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/core/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/base.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/factory.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/local.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/models.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/remote.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/errors.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/answer.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/factory.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/models.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/prompt.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/base.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/directory.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/docx.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/json.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/markdown.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/pdf.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/txt.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/web.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/retrieval/retriever.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/utils/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/utils/hash.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/utils/logger.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/factory.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/pinecone.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/conftest.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/fixtures/shared/hash.json +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_ask.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_collection.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_document.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_ollama.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/__init__.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_chunking.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_config.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_directory_loader.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_errors.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_hash.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_loaders.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_prompt.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_retriever.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_vectordb.py +0 -0
- {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_web_loader.py +0 -0
|
@@ -128,17 +128,20 @@ graph TD
|
|
|
128
128
|
2. Calculates SHA-256 hash of raw document content.
|
|
129
129
|
3. Checks existing `IndexMetadata` in `VectorStore`. The cached index is reused when the index format version, content hash, chunk size, overlap and embedding provider/model all match (URL sources are hashed by their fetched text).
|
|
130
130
|
4. If hash differs or force rebuild requested:
|
|
131
|
-
- `RecursiveChunker` splits text into word-based chunks (default 500 words, 50-word overlap).
|
|
131
|
+
- `RecursiveChunker` splits text into word-based chunks (default 500 words, 50-word overlap). In scripts written without spaces (Chinese, Japanese, Thai, ...) each character counts as one word.
|
|
132
132
|
- `EmbeddingFactory` generates normalized vectors for each chunk.
|
|
133
133
|
- `VectorStore.add()` saves chunks and `VectorStore.save_index_metadata()` persists index metadata.
|
|
134
|
+
- `KeywordIndex.build()` builds the BM25 keyword index from the chunk texts and saves it as `<storeDir>/<namespace>/keyword.json`.
|
|
134
135
|
|
|
135
136
|
### 4.2 Retrieval & Answer Synthesis
|
|
136
|
-
1. `doc.search(query, top_k)`:
|
|
137
|
-
-
|
|
138
|
-
-
|
|
139
|
-
-
|
|
137
|
+
1. `doc.search(query, top_k=..., mode=...)`:
|
|
138
|
+
- `vector` (default): embeds the query and runs `VectorStore.search()` (cosine similarity over L2-normalized vectors).
|
|
139
|
+
- `keyword`: BM25 over the keyword index (`retrieval/keyword_index.py`), or the store's own `keyword_search()` if it has one.
|
|
140
|
+
- `hybrid`: both lists (each `candidates` long, the vector list filtered by `scoreThreshold`), merged with Reciprocal Rank Fusion (`retrieval/fusion.py`).
|
|
141
|
+
- Terms come from the `raglite-v1` tokenizer (`text/tokenizer.py`): NFKC + lowercase words, code identifiers kept whole, character bigrams for CJK and Thai. The TypeScript SDK must produce identical terms and scores; `tests/fixtures/shared/` checks this.
|
|
142
|
+
- `DocumentCollection` merges every document's vector list and keyword list first and fuses once, because BM25 statistics are per document.
|
|
140
143
|
2. `doc.ask(question, options)`:
|
|
141
|
-
- Runs `search(question)
|
|
144
|
+
- Runs `search(question)` with the same retrieval options.
|
|
142
145
|
- Constructs context-augmented system/user prompt via `build_prompt()`.
|
|
143
146
|
- Calls `generate_answer()` or `stream_answer()` with selected LLM adapter (OpenAI, Anthropic, Google, Groq, Ollama, etc.).
|
|
144
147
|
|
|
@@ -152,7 +155,7 @@ Built using **FastAPI** framework for async capabilities, automatic OpenAPI docs
|
|
|
152
155
|
| :--- | :--- | :--- | :--- |
|
|
153
156
|
| `GET` | `/health` | No | Liveness status & total chunk count |
|
|
154
157
|
| `GET` | `/info` | Yes (Bearer token) | Configuration details & index state |
|
|
155
|
-
| `POST` | `/search` | Yes (Bearer token) |
|
|
158
|
+
| `POST` | `/search` | Yes (Bearer token) | Vector, keyword or hybrid search (`mode`) |
|
|
156
159
|
| `POST` | `/ask` | Yes (Bearer token) | Context Q&A generation (supports streaming via StreamingResponse) |
|
|
157
160
|
|
|
158
161
|
---
|
|
@@ -190,4 +193,8 @@ class VectorStore(ABC):
|
|
|
190
193
|
|
|
191
194
|
@abstractmethod
|
|
192
195
|
def read_index_metadata(self) -> Optional[IndexMetadata]: ...
|
|
196
|
+
|
|
197
|
+
# Optional (return None when unsupported):
|
|
198
|
+
def list_chunks(self) -> Optional[List[IndexedChunk]]: ... # rebuild a missing keyword index
|
|
199
|
+
def keyword_search(self, query: str, top_k: int) -> Optional[List[VectorSearchHit]]: ... # native keyword search
|
|
193
200
|
```
|
|
@@ -2,7 +2,32 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
-
## [1.
|
|
5
|
+
## [1.3.0] - 2026-10-03
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
- **Hybrid search:** `search()`, `ask()` and `ask_stream()` accept `mode="vector" | "keyword" | "hybrid"` (or `{"mode": ...}` in the options dict; default `"vector"`), on `Document` and `DocumentCollection`. Keyword mode uses BM25; hybrid merges vector and keyword results with Reciprocal Rank Fusion, so exact terms such as `ERR_4021`, SKUs or function names are found even when embeddings miss them. Tune it with `hybrid={"rrfK", "candidates", "weights"}`, or set defaults with the new `retrieval` option. Results in these modes carry `scores` (`vector`, `keyword`, `fused`); vector-mode results are unchanged.
|
|
9
|
+
- **Keyword index:** `build()` now also writes a BM25 index (`<storeDir>/<namespace>/keyword.json`) from the chunk texts. It needs no extra embedding calls, works with every vector store, and uses the same file layout as the TypeScript SDK.
|
|
10
|
+
- **Multilingual keyword tokenizer (`raglite-v1`):** NFKC and lowercase normalisation, code identifiers kept whole (`gpt-4.1`, `snake_case`), and character bigrams for Chinese, Japanese, Korean, Thai, Lao, Khmer and Myanmar. Exported as `tokenize()`. It produces exactly the same terms as the TypeScript SDK.
|
|
11
|
+
- **HTTP and CLI:** optional `mode` on `/search` and `/ask`; `--mode` on `raglite search`, `ask` and `serve`. `/info` reports `retrievalMode`.
|
|
12
|
+
- **`VectorStore` extension points (optional):** `list_chunks()` lets keyword and hybrid search rebuild a missing keyword index from the store, and `keyword_search(query, top_k)` replaces the built-in BM25 index. The memory and Qdrant stores implement `list_chunks()`. Existing custom stores need no changes.
|
|
13
|
+
|
|
14
|
+
### Changed
|
|
15
|
+
- **Chunking of text without spaces:** Chinese, Japanese, Thai, Lao, Khmer and Myanmar text is now chunked by character instead of becoming one giant "word". Previously a document in these scripts became a single chunk of any length, which could exceed embedding model limits. `chunkSize` and `overlap` count characters for these scripts and words for everything else.
|
|
16
|
+
|
|
17
|
+
### Fixed
|
|
18
|
+
- **Custom chunk sizes are kept:** `build()` without `chunk_size` or `overlap` (in the call or the constructor) now reuses the existing index's values instead of the defaults. Previously `raglite search` or `raglite ask` after `raglite index --chunk-size N` silently re-embedded the whole index at the default 500 words. Defaults still apply to a new index, and explicit values still trigger a rebuild when they differ.
|
|
19
|
+
- **Embedding provider is kept:** with no `embeddings` configured, `build()` now keeps the existing index's provider and model (reusing configured credentials when the provider matches) instead of switching to the local default. The CLI no longer assumes `--embed-provider local` when no `--embed-*` flag is given, so `raglite search` and `raglite ask` reuse an index built with `--embed-provider openai` instead of re-embedding it locally. New indexes still default to local embeddings.
|
|
20
|
+
- **`build(options)` embeddings dict:** `build({"embeddings": {...}})` now accepts a plain dict, as the constructor does.
|
|
21
|
+
- **`raglite serve`:** no longer crashes with `TypeError` when given `--llm-provider` or `--token`.
|
|
22
|
+
- **`DocumentCollection.serve()`:** no longer fails with `ImportError`, so serving a directory works. It now also falls back to the collection's configured `llm`.
|
|
23
|
+
- **`ask()` options:** an explicit `scoreThreshold` of `0` is no longer replaced by the configured default.
|
|
24
|
+
|
|
25
|
+
### Upgrade notes
|
|
26
|
+
- The index format version is now 2. Indexes whose source contains no Chinese, Japanese, Thai, Lao, Khmer or Myanmar text are upgraded in place on the next `build()`, without re-embedding (the source file is read once to check). Indexes of sources that do contain such text are rebuilt once.
|
|
27
|
+
- Indexes built before 1.3.0 have no keyword index. The memory and Qdrant stores build it from the stored chunks on the first keyword or hybrid search. Pinecone and custom stores without `list_chunks()` log a warning and use vector search until you run `build(rebuild=True)`.
|
|
28
|
+
- With Qdrant or Pinecone, the keyword index lives on local disk under `storeDir`. Keep `storeDir` on persistent storage when you use keyword or hybrid search.
|
|
29
|
+
|
|
30
|
+
## [1.2.2] - 2026-10-02
|
|
6
31
|
|
|
7
32
|
### Changed
|
|
8
33
|
- **Upgrades keep cached indexes:** The build cache is now keyed on an index format version (`formatVersion` in `IndexMetadata`) instead of the package version, so upgrading RAGLite no longer re-embeds every index. Indexes built by 1.2.1 are reused as-is; indexes from older releases are rebuilt once.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: raglite-toolkit
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python.
|
|
5
5
|
Project-URL: Homepage, https://github.com/creatorpiyush/raglite-py
|
|
6
6
|
Project-URL: Repository, https://github.com/creatorpiyush/raglite-py
|
|
@@ -59,6 +59,8 @@ Description-Content-Type: text/markdown
|
|
|
59
59
|
- 🤖 **Multi-provider LLMs** — OpenAI, Anthropic (Claude), Google (Gemini), Mistral, Cohere, Groq, xAI, Ollama
|
|
60
60
|
- 🔢 **Multi-provider embeddings** — OpenAI, Google, Mistral, Cohere, Voyage, Ollama, or a **local** offline sentence-transformer (no API key needed)
|
|
61
61
|
- 📐 **Cosine similarity** scoring with L2-normalized vectors
|
|
62
|
+
- 🔎 **Hybrid search** — BM25 keyword search fused with vector search, for exact terms like error codes and SKUs
|
|
63
|
+
- 🌏 **Any language** — Chinese, Japanese, Korean and Thai text is chunked and keyword-indexed correctly
|
|
62
64
|
- ♻️ **Content-hash cache** — reindexes only when the file actually changes
|
|
63
65
|
- 🗂 **Per-document namespacing** — indexes are isolated, two documents never collide
|
|
64
66
|
- 🌐 **REST API** via FastAPI with optional **bearer-token auth**
|
|
@@ -129,6 +131,41 @@ print(answer.text)
|
|
|
129
131
|
|
|
130
132
|
---
|
|
131
133
|
|
|
134
|
+
## Hybrid Search (Keyword + Vector)
|
|
135
|
+
|
|
136
|
+
Vector search matches meaning, but it can miss exact terms such as error codes, SKUs, function names or rare product names. Keyword search (BM25) finds those exactly. `hybrid` runs both and merges the results with Reciprocal Rank Fusion.
|
|
137
|
+
|
|
138
|
+
```python
|
|
139
|
+
doc.search("ERR_4021", mode="hybrid") # "vector" (default) | "keyword" | "hybrid"
|
|
140
|
+
doc.ask("What does ERR_4021 mean?", {"mode": "hybrid"})
|
|
141
|
+
|
|
142
|
+
# Or set a default once; it applies to search(), ask() and ask_stream().
|
|
143
|
+
Document("./runbook.md", {
|
|
144
|
+
"retrieval": {
|
|
145
|
+
"mode": "hybrid",
|
|
146
|
+
"hybrid": {"rrfK": 60, "candidates": 50, "weights": {"vector": 1, "keyword": 1}},
|
|
147
|
+
},
|
|
148
|
+
})
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
- In `keyword` and `hybrid` modes, `score` is the fused rank score scaled to 0..1 (1 means ranked first by every retriever). Each result also has `scores` with `vector`, `keyword` and `fused`. Vector mode results are unchanged.
|
|
152
|
+
- `scoreThreshold` is still a cosine similarity. In hybrid mode it filters the vector results before fusion; keyword matches are not filtered by it.
|
|
153
|
+
- The keyword index is built by `build()` from the chunk texts, with no extra embedding calls, and saved as `<storeDir>/<namespace>/keyword.json`. It works with every vector store. With Qdrant or Pinecone, keep `storeDir` on persistent disk.
|
|
154
|
+
- Indexes built before 1.3: the memory and Qdrant stores build the keyword index from the stored chunks on the first keyword or hybrid search. Other stores log a warning and use vector search until you run `build(rebuild=True)`.
|
|
155
|
+
- The same options work over HTTP (`"mode"` on `/search` and `/ask`) and in the CLI (`--mode hybrid`).
|
|
156
|
+
|
|
157
|
+
The tokenizer is the same in the Python and TypeScript SDKs:
|
|
158
|
+
|
|
159
|
+
| Text | How it is indexed |
|
|
160
|
+
|------|-------------------|
|
|
161
|
+
| Latin, Cyrillic, Greek, Arabic, Devanagari and other spaced scripts | Words, lowercased and NFKC-normalised (`fi` → `fi`, `ABC` → `abc`) |
|
|
162
|
+
| Code identifiers | `gpt-4.1`, `snake_case` and `ERR_42` stay whole, and their parts are indexed too |
|
|
163
|
+
| Chinese, Japanese, Korean, Thai, Lao, Khmer, Myanmar | Overlapping character pairs (`退款处理` → `退款`, `款处`, `处理`), so no dictionary is needed |
|
|
164
|
+
|
|
165
|
+
There is no stemming or stopword list, because both are language-specific: in keyword mode `refund` does not match `refunds`. Hybrid mode's vector side covers those cases.
|
|
166
|
+
|
|
167
|
+
---
|
|
168
|
+
|
|
132
169
|
## Fully Offline — No API Key Needed
|
|
133
170
|
|
|
134
171
|
```python
|
|
@@ -266,6 +303,7 @@ raglite index ./policy.pdf --embed-provider local
|
|
|
266
303
|
|
|
267
304
|
# Semantic search
|
|
268
305
|
raglite search ./docs "refund policy" --top-k 5
|
|
306
|
+
raglite search ./docs "ERR_4021" --mode hybrid
|
|
269
307
|
|
|
270
308
|
# Ask a question (streaming)
|
|
271
309
|
raglite ask ./docs "What is the refund policy?" \
|
|
@@ -313,12 +351,13 @@ raglite serve https://example.com \
|
|
|
313
351
|
```python
|
|
314
352
|
Document("./policy.pdf", {
|
|
315
353
|
# Chunking
|
|
316
|
-
"chunkSize": 500, # words per chunk (default: 500)
|
|
354
|
+
"chunkSize": 500, # words per chunk; characters for Chinese, Japanese, Thai, ... (default: 500)
|
|
317
355
|
"overlap": 50, # overlapping words between chunks (default: 50)
|
|
318
356
|
|
|
319
357
|
# Retrieval
|
|
320
358
|
"topK": 5, # default results returned (default: 5)
|
|
321
359
|
"scoreThreshold": 0.0, # minimum cosine similarity (0..1, default: 0)
|
|
360
|
+
"retrieval": {"mode": "vector"}, # "vector" | "keyword" | "hybrid" (default: vector)
|
|
322
361
|
|
|
323
362
|
# Storage
|
|
324
363
|
"storeDir": ".raglite", # where indexes are persisted (default: .raglite)
|
|
@@ -341,9 +380,9 @@ Every `build()` call fingerprints the source (file bytes, or the fetched text fo
|
|
|
341
380
|
| Factor | Triggers rebuild if changed |
|
|
342
381
|
|--------|-----------------------------|
|
|
343
382
|
| File content | SHA-256 hash differs |
|
|
344
|
-
| Chunk size | `chunkSize` changed |
|
|
345
|
-
| Overlap | `overlap` changed |
|
|
346
|
-
| Embedding provider/model | Provider or model string changed |
|
|
383
|
+
| Chunk size | `chunkSize` changed (when not set, the existing index's value is kept) |
|
|
384
|
+
| Overlap | `overlap` changed (when not set, the existing index's value is kept) |
|
|
385
|
+
| Embedding provider/model | Provider or model string changed (when no `embeddings` are configured, the existing index's are kept) |
|
|
347
386
|
| Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
|
|
348
387
|
|
|
349
388
|
Pass `rebuild=True` to `build()` to force a fresh index regardless.
|
|
@@ -362,6 +401,9 @@ from raglite.vectordb.base import VectorStore
|
|
|
362
401
|
class MyVectorStore(VectorStore):
|
|
363
402
|
# Implement: namespace (property), load, reset, add, search, count,
|
|
364
403
|
# save_index_metadata, read_index_metadata
|
|
404
|
+
# Optional: list_chunks() lets hybrid search rebuild a missing keyword index
|
|
405
|
+
# from your store; keyword_search(query, top_k) replaces the
|
|
406
|
+
# built-in BM25 index with your own.
|
|
365
407
|
...
|
|
366
408
|
```
|
|
367
409
|
|
|
@@ -13,6 +13,8 @@
|
|
|
13
13
|
- 🤖 **Multi-provider LLMs** — OpenAI, Anthropic (Claude), Google (Gemini), Mistral, Cohere, Groq, xAI, Ollama
|
|
14
14
|
- 🔢 **Multi-provider embeddings** — OpenAI, Google, Mistral, Cohere, Voyage, Ollama, or a **local** offline sentence-transformer (no API key needed)
|
|
15
15
|
- 📐 **Cosine similarity** scoring with L2-normalized vectors
|
|
16
|
+
- 🔎 **Hybrid search** — BM25 keyword search fused with vector search, for exact terms like error codes and SKUs
|
|
17
|
+
- 🌏 **Any language** — Chinese, Japanese, Korean and Thai text is chunked and keyword-indexed correctly
|
|
16
18
|
- ♻️ **Content-hash cache** — reindexes only when the file actually changes
|
|
17
19
|
- 🗂 **Per-document namespacing** — indexes are isolated, two documents never collide
|
|
18
20
|
- 🌐 **REST API** via FastAPI with optional **bearer-token auth**
|
|
@@ -83,6 +85,41 @@ print(answer.text)
|
|
|
83
85
|
|
|
84
86
|
---
|
|
85
87
|
|
|
88
|
+
## Hybrid Search (Keyword + Vector)
|
|
89
|
+
|
|
90
|
+
Vector search matches meaning, but it can miss exact terms such as error codes, SKUs, function names or rare product names. Keyword search (BM25) finds those exactly. `hybrid` runs both and merges the results with Reciprocal Rank Fusion.
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
doc.search("ERR_4021", mode="hybrid") # "vector" (default) | "keyword" | "hybrid"
|
|
94
|
+
doc.ask("What does ERR_4021 mean?", {"mode": "hybrid"})
|
|
95
|
+
|
|
96
|
+
# Or set a default once; it applies to search(), ask() and ask_stream().
|
|
97
|
+
Document("./runbook.md", {
|
|
98
|
+
"retrieval": {
|
|
99
|
+
"mode": "hybrid",
|
|
100
|
+
"hybrid": {"rrfK": 60, "candidates": 50, "weights": {"vector": 1, "keyword": 1}},
|
|
101
|
+
},
|
|
102
|
+
})
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
- In `keyword` and `hybrid` modes, `score` is the fused rank score scaled to 0..1 (1 means ranked first by every retriever). Each result also has `scores` with `vector`, `keyword` and `fused`. Vector mode results are unchanged.
|
|
106
|
+
- `scoreThreshold` is still a cosine similarity. In hybrid mode it filters the vector results before fusion; keyword matches are not filtered by it.
|
|
107
|
+
- The keyword index is built by `build()` from the chunk texts, with no extra embedding calls, and saved as `<storeDir>/<namespace>/keyword.json`. It works with every vector store. With Qdrant or Pinecone, keep `storeDir` on persistent disk.
|
|
108
|
+
- Indexes built before 1.3: the memory and Qdrant stores build the keyword index from the stored chunks on the first keyword or hybrid search. Other stores log a warning and use vector search until you run `build(rebuild=True)`.
|
|
109
|
+
- The same options work over HTTP (`"mode"` on `/search` and `/ask`) and in the CLI (`--mode hybrid`).
|
|
110
|
+
|
|
111
|
+
The tokenizer is the same in the Python and TypeScript SDKs:
|
|
112
|
+
|
|
113
|
+
| Text | How it is indexed |
|
|
114
|
+
|------|-------------------|
|
|
115
|
+
| Latin, Cyrillic, Greek, Arabic, Devanagari and other spaced scripts | Words, lowercased and NFKC-normalised (`fi` → `fi`, `ABC` → `abc`) |
|
|
116
|
+
| Code identifiers | `gpt-4.1`, `snake_case` and `ERR_42` stay whole, and their parts are indexed too |
|
|
117
|
+
| Chinese, Japanese, Korean, Thai, Lao, Khmer, Myanmar | Overlapping character pairs (`退款处理` → `退款`, `款处`, `处理`), so no dictionary is needed |
|
|
118
|
+
|
|
119
|
+
There is no stemming or stopword list, because both are language-specific: in keyword mode `refund` does not match `refunds`. Hybrid mode's vector side covers those cases.
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
86
123
|
## Fully Offline — No API Key Needed
|
|
87
124
|
|
|
88
125
|
```python
|
|
@@ -220,6 +257,7 @@ raglite index ./policy.pdf --embed-provider local
|
|
|
220
257
|
|
|
221
258
|
# Semantic search
|
|
222
259
|
raglite search ./docs "refund policy" --top-k 5
|
|
260
|
+
raglite search ./docs "ERR_4021" --mode hybrid
|
|
223
261
|
|
|
224
262
|
# Ask a question (streaming)
|
|
225
263
|
raglite ask ./docs "What is the refund policy?" \
|
|
@@ -267,12 +305,13 @@ raglite serve https://example.com \
|
|
|
267
305
|
```python
|
|
268
306
|
Document("./policy.pdf", {
|
|
269
307
|
# Chunking
|
|
270
|
-
"chunkSize": 500, # words per chunk (default: 500)
|
|
308
|
+
"chunkSize": 500, # words per chunk; characters for Chinese, Japanese, Thai, ... (default: 500)
|
|
271
309
|
"overlap": 50, # overlapping words between chunks (default: 50)
|
|
272
310
|
|
|
273
311
|
# Retrieval
|
|
274
312
|
"topK": 5, # default results returned (default: 5)
|
|
275
313
|
"scoreThreshold": 0.0, # minimum cosine similarity (0..1, default: 0)
|
|
314
|
+
"retrieval": {"mode": "vector"}, # "vector" | "keyword" | "hybrid" (default: vector)
|
|
276
315
|
|
|
277
316
|
# Storage
|
|
278
317
|
"storeDir": ".raglite", # where indexes are persisted (default: .raglite)
|
|
@@ -295,9 +334,9 @@ Every `build()` call fingerprints the source (file bytes, or the fetched text fo
|
|
|
295
334
|
| Factor | Triggers rebuild if changed |
|
|
296
335
|
|--------|-----------------------------|
|
|
297
336
|
| File content | SHA-256 hash differs |
|
|
298
|
-
| Chunk size | `chunkSize` changed |
|
|
299
|
-
| Overlap | `overlap` changed |
|
|
300
|
-
| Embedding provider/model | Provider or model string changed |
|
|
337
|
+
| Chunk size | `chunkSize` changed (when not set, the existing index's value is kept) |
|
|
338
|
+
| Overlap | `overlap` changed (when not set, the existing index's value is kept) |
|
|
339
|
+
| Embedding provider/model | Provider or model string changed (when no `embeddings` are configured, the existing index's are kept) |
|
|
301
340
|
| Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
|
|
302
341
|
|
|
303
342
|
Pass `rebuild=True` to `build()` to force a fresh index regardless.
|
|
@@ -316,6 +355,9 @@ from raglite.vectordb.base import VectorStore
|
|
|
316
355
|
class MyVectorStore(VectorStore):
|
|
317
356
|
# Implement: namespace (property), load, reset, add, search, count,
|
|
318
357
|
# save_index_metadata, read_index_metadata
|
|
358
|
+
# Optional: list_chunks() lets hybrid search rebuild a missing keyword index
|
|
359
|
+
# from your store; keyword_search(query, top_k) replaces the
|
|
360
|
+
# built-in BM25 index with your own.
|
|
319
361
|
...
|
|
320
362
|
```
|
|
321
363
|
|
|
@@ -8,7 +8,7 @@ packages = ["src/raglite"]
|
|
|
8
8
|
|
|
9
9
|
[project]
|
|
10
10
|
name = "raglite-toolkit"
|
|
11
|
-
version = "1.
|
|
11
|
+
version = "1.3.0"
|
|
12
12
|
description = "Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python."
|
|
13
13
|
readme = "README.md"
|
|
14
14
|
requires-python = ">=3.10"
|
|
@@ -41,20 +41,34 @@ from .loaders import (
|
|
|
41
41
|
is_supported_file,
|
|
42
42
|
is_url,
|
|
43
43
|
)
|
|
44
|
-
from .retrieval import
|
|
44
|
+
from .retrieval import (
|
|
45
|
+
KeywordHit,
|
|
46
|
+
KeywordIndex,
|
|
47
|
+
RankedList,
|
|
48
|
+
RetrievalPlan,
|
|
49
|
+
Retriever,
|
|
50
|
+
reciprocal_rank_fusion,
|
|
51
|
+
resolve_retrieval_plan,
|
|
52
|
+
)
|
|
53
|
+
from .text import TOKENIZER_NAME, tokenize
|
|
45
54
|
from .types import (
|
|
46
55
|
AnswerResult,
|
|
47
56
|
ChunkMetadata,
|
|
48
57
|
EmbeddingProviderConfig,
|
|
49
58
|
EmbeddingProviderName,
|
|
59
|
+
HybridOptions,
|
|
50
60
|
IndexMetadata,
|
|
51
61
|
LLMProviderConfig,
|
|
52
62
|
LLMProviderName,
|
|
63
|
+
RetrievalMode,
|
|
64
|
+
RetrievalOptions,
|
|
53
65
|
SearchResult,
|
|
66
|
+
SearchScores,
|
|
54
67
|
StoredChunk,
|
|
55
68
|
)
|
|
56
69
|
from .utils.logger import create_logger
|
|
57
70
|
from .vectordb import MemoryVectorStore
|
|
71
|
+
from .vectordb.base import IndexedChunk, VectorSearchHit, VectorStore
|
|
58
72
|
|
|
59
73
|
VERSION = PACKAGE_VERSION
|
|
60
74
|
|
|
@@ -78,6 +92,10 @@ __all__ = [
|
|
|
78
92
|
"ChunkMetadata",
|
|
79
93
|
"StoredChunk",
|
|
80
94
|
"SearchResult",
|
|
95
|
+
"SearchScores",
|
|
96
|
+
"RetrievalMode",
|
|
97
|
+
"RetrievalOptions",
|
|
98
|
+
"HybridOptions",
|
|
81
99
|
"AnswerResult",
|
|
82
100
|
"IndexMetadata",
|
|
83
101
|
"DocumentOptions",
|
|
@@ -97,7 +115,18 @@ __all__ = [
|
|
|
97
115
|
"BaseChunker",
|
|
98
116
|
"RecursiveChunker",
|
|
99
117
|
"MemoryVectorStore",
|
|
118
|
+
"VectorStore",
|
|
119
|
+
"VectorSearchHit",
|
|
120
|
+
"IndexedChunk",
|
|
100
121
|
"Retriever",
|
|
122
|
+
"KeywordIndex",
|
|
123
|
+
"KeywordHit",
|
|
124
|
+
"RankedList",
|
|
125
|
+
"RetrievalPlan",
|
|
126
|
+
"reciprocal_rank_fusion",
|
|
127
|
+
"resolve_retrieval_plan",
|
|
128
|
+
"tokenize",
|
|
129
|
+
"TOKENIZER_NAME",
|
|
101
130
|
"DEFAULT_EMBEDDING_MODELS",
|
|
102
131
|
"LocalEmbedder",
|
|
103
132
|
"RemoteEmbedder",
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
from typing import Optional
|
|
1
|
+
from typing import Literal, Optional
|
|
2
2
|
|
|
3
3
|
from pydantic import BaseModel, ConfigDict, Field
|
|
4
4
|
|
|
@@ -11,6 +11,7 @@ class SearchRequest(BaseModel):
|
|
|
11
11
|
scoreThreshold: Optional[float] = Field(
|
|
12
12
|
default=None, alias="scoreThreshold", ge=-1.0, le=1.0
|
|
13
13
|
)
|
|
14
|
+
mode: Optional[Literal["vector", "keyword", "hybrid"]] = None
|
|
14
15
|
|
|
15
16
|
|
|
16
17
|
class AskRequest(BaseModel):
|
|
@@ -21,6 +22,7 @@ class AskRequest(BaseModel):
|
|
|
21
22
|
scoreThreshold: Optional[float] = Field(
|
|
22
23
|
default=None, alias="scoreThreshold", ge=-1.0, le=1.0
|
|
23
24
|
)
|
|
25
|
+
mode: Optional[Literal["vector", "keyword", "hybrid"]] = None
|
|
24
26
|
includeCitations: Optional[bool] = Field(
|
|
25
27
|
default=None, alias="includeCitations"
|
|
26
28
|
)
|
|
@@ -103,6 +103,7 @@ def build_app(document: Any, options: Dict[str, Any]) -> FastAPI:
|
|
|
103
103
|
"chunkSize": cfg.chunkSize,
|
|
104
104
|
"overlap": cfg.overlap,
|
|
105
105
|
"topK": cfg.topK,
|
|
106
|
+
"retrievalMode": cfg.retrieval.mode or "vector",
|
|
106
107
|
"embeddings": {
|
|
107
108
|
"provider": cfg.embeddings.provider,
|
|
108
109
|
"model": cfg.embeddings.model
|
|
@@ -118,6 +119,7 @@ def build_app(document: Any, options: Dict[str, Any]) -> FastAPI:
|
|
|
118
119
|
req.query,
|
|
119
120
|
top_k=req.topK,
|
|
120
121
|
score_threshold=req.scoreThreshold,
|
|
122
|
+
mode=req.mode,
|
|
121
123
|
)
|
|
122
124
|
return {
|
|
123
125
|
"results": [r.model_dump(by_alias=True) for r in results]
|
|
@@ -133,6 +135,7 @@ def build_app(document: Any, options: Dict[str, Any]) -> FastAPI:
|
|
|
133
135
|
"llm": ask_provider,
|
|
134
136
|
"topK": req.topK,
|
|
135
137
|
"scoreThreshold": req.scoreThreshold,
|
|
138
|
+
"mode": req.mode,
|
|
136
139
|
"includeCitations": req.includeCitations,
|
|
137
140
|
}
|
|
138
141
|
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from typing import List, Tuple
|
|
3
|
+
|
|
4
|
+
from ..errors import ChunkingError
|
|
5
|
+
from ..text.scripts import has_unspaced_text, is_mark, is_unspaced_char
|
|
6
|
+
from .base import BaseChunker
|
|
7
|
+
|
|
8
|
+
# The exact set JavaScript's \s matches. Python's \s differs (it includes
|
|
9
|
+
# \x1c-\x1f and \x85 but not ), which would make the two SDKs chunk the
|
|
10
|
+
# same text differently.
|
|
11
|
+
_JS_WHITESPACE = re.compile(
|
|
12
|
+
"[\t\n\v\f\r -
]+"
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
# One chunking unit: (text, spaced). ``spaced`` is False when the unit
|
|
16
|
+
# continues the previous unit's word, so no space goes between them.
|
|
17
|
+
_Unit = Tuple[str, bool]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class RecursiveChunker(BaseChunker):
|
|
21
|
+
"""Word-based recursive chunker with overlap.
|
|
22
|
+
|
|
23
|
+
Scripts written without spaces (Chinese, Japanese, Thai, ...) have no word
|
|
24
|
+
boundaries to split on, so each of their characters counts as one unit.
|
|
25
|
+
Without this a whole unspaced document would be a single "word" and
|
|
26
|
+
therefore a single chunk of any length.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
def split(self, text: str) -> List[str]:
|
|
30
|
+
if not _JS_WHITESPACE.sub("", text):
|
|
31
|
+
return []
|
|
32
|
+
if self.overlap >= self.chunk_size:
|
|
33
|
+
raise ChunkingError(
|
|
34
|
+
f"overlap ({self.overlap}) must be smaller than chunkSize ({self.chunk_size})"
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
words = [w for w in _JS_WHITESPACE.split(text) if w]
|
|
38
|
+
if has_unspaced_text(text):
|
|
39
|
+
units = [u for w in words for u in _word_units(w)]
|
|
40
|
+
else:
|
|
41
|
+
units = [(w, True) for w in words]
|
|
42
|
+
if len(units) <= self.chunk_size:
|
|
43
|
+
return [_render(units)]
|
|
44
|
+
|
|
45
|
+
step = self.chunk_size - self.overlap
|
|
46
|
+
chunks: List[str] = []
|
|
47
|
+
|
|
48
|
+
start = 0
|
|
49
|
+
while start < len(units):
|
|
50
|
+
end = start + self.chunk_size
|
|
51
|
+
slice_units = units[start:end]
|
|
52
|
+
if not slice_units:
|
|
53
|
+
break
|
|
54
|
+
chunks.append(_render(slice_units))
|
|
55
|
+
if end >= len(units):
|
|
56
|
+
break
|
|
57
|
+
start += step
|
|
58
|
+
|
|
59
|
+
return chunks
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _word_units(word: str) -> List[_Unit]:
|
|
63
|
+
"""Split a word into units: each unspaced-script character (with its
|
|
64
|
+
combining marks) is a unit, and every run of other characters is a unit."""
|
|
65
|
+
units: List[List] = []
|
|
66
|
+
run = ""
|
|
67
|
+
for ch in word:
|
|
68
|
+
if is_mark(ch) and run == "" and units:
|
|
69
|
+
units[-1][0] += ch
|
|
70
|
+
elif is_mark(ch):
|
|
71
|
+
run += ch
|
|
72
|
+
elif is_unspaced_char(ord(ch)):
|
|
73
|
+
if run:
|
|
74
|
+
units.append([run, not units])
|
|
75
|
+
run = ""
|
|
76
|
+
units.append([ch, not units])
|
|
77
|
+
else:
|
|
78
|
+
run += ch
|
|
79
|
+
if run:
|
|
80
|
+
units.append([run, not units])
|
|
81
|
+
return [(text, spaced) for text, spaced in units]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _render(units: List[_Unit]) -> str:
|
|
85
|
+
out = ""
|
|
86
|
+
for i, (text, spaced) in enumerate(units):
|
|
87
|
+
if i > 0 and spaced:
|
|
88
|
+
out += " "
|
|
89
|
+
out += text
|
|
90
|
+
return out
|
|
@@ -5,7 +5,7 @@ import sys
|
|
|
5
5
|
import time
|
|
6
6
|
from typing import Optional, Union
|
|
7
7
|
|
|
8
|
-
from .constants import PACKAGE_VERSION
|
|
8
|
+
from .constants import DEFAULT_HOST, DEFAULT_PORT, PACKAGE_VERSION
|
|
9
9
|
from .core.collection import DocumentCollection
|
|
10
10
|
from .core.document import Document
|
|
11
11
|
from .loaders import is_url
|
|
@@ -14,9 +14,9 @@ HELP = f"""raglite v{PACKAGE_VERSION}
|
|
|
14
14
|
|
|
15
15
|
Usage:
|
|
16
16
|
raglite index <path|url> [--chunk-size N] [--overlap N] [--embed-provider P] [--embed-model M] [--embed-key K] [--rebuild]
|
|
17
|
-
raglite search <path|url> "query" [--top-k N]
|
|
18
|
-
raglite ask <path|url> "question" --llm-provider P [--llm-model M] [--llm-key K] [--stream]
|
|
19
|
-
raglite serve <path|url> --llm-provider P [--llm-key K] [--host H] [--port N] [--token T]
|
|
17
|
+
raglite search <path|url> "query" [--top-k N] [--mode vector|keyword|hybrid]
|
|
18
|
+
raglite ask <path|url> "question" --llm-provider P [--llm-model M] [--llm-key K] [--stream] [--mode M]
|
|
19
|
+
raglite serve <path|url> --llm-provider P [--llm-key K] [--host H] [--port N] [--token T] [--mode M]
|
|
20
20
|
raglite --help
|
|
21
21
|
raglite --version
|
|
22
22
|
|
|
@@ -26,7 +26,13 @@ Providers:
|
|
|
26
26
|
"""
|
|
27
27
|
|
|
28
28
|
|
|
29
|
-
|
|
29
|
+
_MODES = ("vector", "keyword", "hybrid")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def parse_common_embedding(args_dict: dict) -> Optional[dict]:
|
|
33
|
+
"""None when no --embed-* flag is given, so an existing index keeps its provider."""
|
|
34
|
+
if not any(args_dict.get(k) for k in ("embed_provider", "embed_model", "embed_key")):
|
|
35
|
+
return None
|
|
30
36
|
provider = args_dict.get("embed_provider") or "local"
|
|
31
37
|
config = {"provider": provider}
|
|
32
38
|
if args_dict.get("embed_model"):
|
|
@@ -36,6 +42,12 @@ def parse_common_embedding(args_dict: dict) -> dict:
|
|
|
36
42
|
return config
|
|
37
43
|
|
|
38
44
|
|
|
45
|
+
def _with_embeddings(options: dict, embeddings: Optional[dict]) -> dict:
|
|
46
|
+
if embeddings is not None:
|
|
47
|
+
options["embeddings"] = embeddings
|
|
48
|
+
return options
|
|
49
|
+
|
|
50
|
+
|
|
39
51
|
def parse_llm(args_dict: dict) -> Optional[dict]:
|
|
40
52
|
provider = args_dict.get("llm_provider")
|
|
41
53
|
if not provider:
|
|
@@ -70,7 +82,7 @@ def run_index(args):
|
|
|
70
82
|
parsed = parser.parse_args(args)
|
|
71
83
|
|
|
72
84
|
embeddings = parse_common_embedding(vars(parsed))
|
|
73
|
-
target = resolve_target(parsed.file, {
|
|
85
|
+
target = resolve_target(parsed.file, _with_embeddings({}, embeddings))
|
|
74
86
|
|
|
75
87
|
build_opts = {}
|
|
76
88
|
if parsed.chunk_size is not None:
|
|
@@ -90,6 +102,7 @@ def run_search(args):
|
|
|
90
102
|
parser.add_argument("file")
|
|
91
103
|
parser.add_argument("query")
|
|
92
104
|
parser.add_argument("--top-k", type=int)
|
|
105
|
+
parser.add_argument("--mode", choices=_MODES)
|
|
93
106
|
parser.add_argument("--embed-provider")
|
|
94
107
|
parser.add_argument("--embed-model")
|
|
95
108
|
parser.add_argument("--embed-key")
|
|
@@ -97,11 +110,13 @@ def run_search(args):
|
|
|
97
110
|
parsed = parser.parse_args(args)
|
|
98
111
|
|
|
99
112
|
embeddings = parse_common_embedding(vars(parsed))
|
|
100
|
-
target = resolve_target(parsed.file, {
|
|
113
|
+
target = resolve_target(parsed.file, _with_embeddings({}, embeddings))
|
|
101
114
|
|
|
102
115
|
search_opts = {}
|
|
103
116
|
if parsed.top_k is not None:
|
|
104
117
|
search_opts["topK"] = parsed.top_k
|
|
118
|
+
if parsed.mode is not None:
|
|
119
|
+
search_opts["mode"] = parsed.mode
|
|
105
120
|
|
|
106
121
|
results = target.search(parsed.query, search_opts)
|
|
107
122
|
serialized = [r.model_dump(by_alias=True) for r in results]
|
|
@@ -120,17 +135,20 @@ def run_ask(args):
|
|
|
120
135
|
parser.add_argument("--llm-model")
|
|
121
136
|
parser.add_argument("--llm-key")
|
|
122
137
|
parser.add_argument("--stream", action="store_true")
|
|
138
|
+
parser.add_argument("--mode", choices=_MODES)
|
|
123
139
|
|
|
124
140
|
parsed = parser.parse_args(args)
|
|
125
141
|
|
|
126
142
|
embeddings = parse_common_embedding(vars(parsed))
|
|
127
143
|
llm = parse_llm(vars(parsed))
|
|
128
144
|
|
|
129
|
-
target = resolve_target(parsed.file, {"
|
|
145
|
+
target = resolve_target(parsed.file, _with_embeddings({"llm": llm}, embeddings))
|
|
130
146
|
|
|
131
147
|
opts = {}
|
|
132
148
|
if parsed.top_k is not None:
|
|
133
149
|
opts["topK"] = parsed.top_k
|
|
150
|
+
if parsed.mode is not None:
|
|
151
|
+
opts["mode"] = parsed.mode
|
|
134
152
|
|
|
135
153
|
if parsed.stream:
|
|
136
154
|
for chunk in target.ask_stream(parsed.question, opts):
|
|
@@ -154,42 +172,40 @@ def run_serve(args):
|
|
|
154
172
|
parser.add_argument("--host")
|
|
155
173
|
parser.add_argument("--port", type=int)
|
|
156
174
|
parser.add_argument("--token")
|
|
175
|
+
parser.add_argument("--mode", choices=_MODES)
|
|
157
176
|
|
|
158
177
|
parsed = parser.parse_args(args)
|
|
159
178
|
|
|
160
179
|
embeddings = parse_common_embedding(vars(parsed))
|
|
161
180
|
llm = parse_llm(vars(parsed))
|
|
162
181
|
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
182
|
+
options: dict = _with_embeddings({}, embeddings)
|
|
183
|
+
if llm:
|
|
184
|
+
options["llm"] = llm
|
|
185
|
+
if parsed.mode:
|
|
186
|
+
options["retrieval"] = {"mode": parsed.mode}
|
|
187
|
+
target = resolve_target(parsed.file, options)
|
|
167
188
|
target.build()
|
|
168
189
|
|
|
169
|
-
|
|
170
|
-
if
|
|
171
|
-
|
|
172
|
-
if
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
if parsed.token:
|
|
177
|
-
serve_opts["bearerToken"] = parsed.token
|
|
178
|
-
|
|
179
|
-
if hasattr(target, "serve"):
|
|
180
|
-
target.serve(**serve_opts)
|
|
181
|
-
else:
|
|
182
|
-
from .api.server import create_server
|
|
183
|
-
handle = create_server(target, serve_opts)
|
|
184
|
-
sys.stdout.write(f"RagLite listening on {handle.url}\n")
|
|
185
|
-
sys.stdout.flush()
|
|
190
|
+
host = parsed.host or DEFAULT_HOST
|
|
191
|
+
port = parsed.port if parsed.port is not None else DEFAULT_PORT
|
|
192
|
+
|
|
193
|
+
if isinstance(target, DocumentCollection):
|
|
194
|
+
# Blocks until the server stops.
|
|
195
|
+
target.serve(host=host, port=port, bearer_token=parsed.token, llm=llm)
|
|
196
|
+
return
|
|
186
197
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
198
|
+
handle = target.serve(
|
|
199
|
+
{"llm": llm} if llm else None, host=host, port=port, bearer_token=parsed.token
|
|
200
|
+
)
|
|
201
|
+
sys.stdout.write(f"RagLite listening on {handle.url}\n")
|
|
202
|
+
sys.stdout.flush()
|
|
203
|
+
try:
|
|
204
|
+
while True:
|
|
205
|
+
time.sleep(1)
|
|
206
|
+
except KeyboardInterrupt:
|
|
207
|
+
handle.close()
|
|
208
|
+
sys.exit(0)
|
|
193
209
|
|
|
194
210
|
|
|
195
211
|
def main():
|