raglite-toolkit 1.2.2__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/ARCHITECTURE.md +14 -7
  2. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/CHANGELOG.md +26 -1
  3. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/PKG-INFO +47 -5
  4. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/README.md +46 -4
  5. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/pyproject.toml +1 -1
  6. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/__init__.py +30 -1
  7. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/api/schemas.py +3 -1
  8. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/api/server.py +3 -0
  9. raglite_toolkit-1.3.0/src/raglite/chunking/recursive.py +90 -0
  10. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/cli.py +51 -35
  11. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/config.py +11 -2
  12. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/constants.py +10 -2
  13. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/core/collection.py +64 -30
  14. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/core/document.py +244 -34
  15. raglite_toolkit-1.3.0/src/raglite/retrieval/__init__.py +14 -0
  16. raglite_toolkit-1.3.0/src/raglite/retrieval/fusion.py +51 -0
  17. raglite_toolkit-1.3.0/src/raglite/retrieval/keyword_index.py +144 -0
  18. raglite_toolkit-1.3.0/src/raglite/retrieval/plan.py +75 -0
  19. raglite_toolkit-1.3.0/src/raglite/text/__init__.py +11 -0
  20. raglite_toolkit-1.3.0/src/raglite/text/scripts.py +64 -0
  21. raglite_toolkit-1.3.0/src/raglite/text/tokenizer.py +106 -0
  22. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/types.py +60 -2
  23. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/base.py +21 -0
  24. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/memory.py +4 -1
  25. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/qdrant.py +34 -1
  26. raglite_toolkit-1.3.0/tests/fixtures/shared/bm25.json +102 -0
  27. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/fixtures/shared/chunker.json +32 -3
  28. raglite_toolkit-1.3.0/tests/fixtures/shared/rrf.json +321 -0
  29. raglite_toolkit-1.3.0/tests/fixtures/shared/tokenizer.json +141 -0
  30. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_api.py +3 -1
  31. raglite_toolkit-1.3.0/tests/integration/test_hybrid.py +328 -0
  32. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_cli.py +2 -2
  33. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_regressions.py +130 -0
  34. raglite_toolkit-1.3.0/tests/unit/test_shared_fixtures.py +99 -0
  35. raglite_toolkit-1.2.2/src/raglite/chunking/recursive.py +0 -42
  36. raglite_toolkit-1.2.2/src/raglite/retrieval/__init__.py +0 -3
  37. raglite_toolkit-1.2.2/tests/unit/test_shared_fixtures.py +0 -42
  38. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/.github/workflows/pr-verify.yml +0 -0
  39. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/.github/workflows/publish.yml +0 -0
  40. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/.gitignore +0 -0
  41. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/LICENSE +0 -0
  42. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/basic.py +0 -0
  43. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/custom_store_example.py +0 -0
  44. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/multi_provider.py +0 -0
  45. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/ollama_test.py +0 -0
  46. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/qdrant_example.py +0 -0
  47. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/sample.txt +0 -0
  48. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/examples/serve.py +0 -0
  49. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/scripts/pre-commit.sh +0 -0
  50. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/scripts/pre-release.sh +0 -0
  51. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/scripts/sync_shared_fixtures.py +0 -0
  52. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/api/__init__.py +0 -0
  53. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/chunking/__init__.py +0 -0
  54. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/chunking/base.py +0 -0
  55. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/core/__init__.py +0 -0
  56. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/__init__.py +0 -0
  57. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/base.py +0 -0
  58. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/factory.py +0 -0
  59. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/local.py +0 -0
  60. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/models.py +0 -0
  61. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/embeddings/remote.py +0 -0
  62. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/errors.py +0 -0
  63. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/__init__.py +0 -0
  64. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/answer.py +0 -0
  65. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/factory.py +0 -0
  66. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/models.py +0 -0
  67. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/llm/prompt.py +0 -0
  68. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/__init__.py +0 -0
  69. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/base.py +0 -0
  70. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/directory.py +0 -0
  71. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/docx.py +0 -0
  72. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/json.py +0 -0
  73. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/markdown.py +0 -0
  74. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/pdf.py +0 -0
  75. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/txt.py +0 -0
  76. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/loaders/web.py +0 -0
  77. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/retrieval/retriever.py +0 -0
  78. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/utils/__init__.py +0 -0
  79. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/utils/hash.py +0 -0
  80. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/utils/logger.py +0 -0
  81. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/__init__.py +0 -0
  82. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/factory.py +0 -0
  83. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/src/raglite/vectordb/pinecone.py +0 -0
  84. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/__init__.py +0 -0
  85. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/conftest.py +0 -0
  86. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/fixtures/shared/hash.json +0 -0
  87. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/__init__.py +0 -0
  88. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_ask.py +0 -0
  89. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_collection.py +0 -0
  90. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_document.py +0 -0
  91. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/integration/test_ollama.py +0 -0
  92. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/__init__.py +0 -0
  93. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_chunking.py +0 -0
  94. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_config.py +0 -0
  95. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_directory_loader.py +0 -0
  96. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_errors.py +0 -0
  97. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_hash.py +0 -0
  98. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_loaders.py +0 -0
  99. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_prompt.py +0 -0
  100. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_retriever.py +0 -0
  101. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_vectordb.py +0 -0
  102. {raglite_toolkit-1.2.2 → raglite_toolkit-1.3.0}/tests/unit/test_web_loader.py +0 -0
@@ -128,17 +128,20 @@ graph TD
128
128
  2. Calculates SHA-256 hash of raw document content.
129
129
  3. Checks existing `IndexMetadata` in `VectorStore`. The cached index is reused when the index format version, content hash, chunk size, overlap and embedding provider/model all match (URL sources are hashed by their fetched text).
130
130
  4. If hash differs or force rebuild requested:
131
- - `RecursiveChunker` splits text into word-based chunks (default 500 words, 50-word overlap).
131
+ - `RecursiveChunker` splits text into word-based chunks (default 500 words, 50-word overlap). In scripts written without spaces (Chinese, Japanese, Thai, ...) each character counts as one word.
132
132
  - `EmbeddingFactory` generates normalized vectors for each chunk.
133
133
  - `VectorStore.add()` saves chunks and `VectorStore.save_index_metadata()` persists index metadata.
134
+ - `KeywordIndex.build()` builds the BM25 keyword index from the chunk texts and saves it as `<storeDir>/<namespace>/keyword.json`.
134
135
 
135
136
  ### 4.2 Retrieval & Answer Synthesis
136
- 1. `doc.search(query, top_k)`:
137
- - Embeds query text.
138
- - Executes vector search via `VectorStore.search()`, calculating cosine similarity over L2-normalized vectors.
139
- - Returns top $K$ `SearchResult` objects.
137
+ 1. `doc.search(query, top_k=..., mode=...)`:
138
+ - `vector` (default): embeds the query and runs `VectorStore.search()` (cosine similarity over L2-normalized vectors).
139
+ - `keyword`: BM25 over the keyword index (`retrieval/keyword_index.py`), or the store's own `keyword_search()` if it has one.
140
+ - `hybrid`: both lists (each `candidates` long, the vector list filtered by `scoreThreshold`), merged with Reciprocal Rank Fusion (`retrieval/fusion.py`).
141
+ - Terms come from the `raglite-v1` tokenizer (`text/tokenizer.py`): NFKC + lowercase words, code identifiers kept whole, character bigrams for CJK and Thai. The TypeScript SDK must produce identical terms and scores; `tests/fixtures/shared/` checks this.
142
+ - `DocumentCollection` merges every document's vector list and keyword list first and fuses once, because BM25 statistics are per document.
140
143
  2. `doc.ask(question, options)`:
141
- - Runs `search(question)`.
144
+ - Runs `search(question)` with the same retrieval options.
142
145
  - Constructs context-augmented system/user prompt via `build_prompt()`.
143
146
  - Calls `generate_answer()` or `stream_answer()` with selected LLM adapter (OpenAI, Anthropic, Google, Groq, Ollama, etc.).
144
147
 
@@ -152,7 +155,7 @@ Built using **FastAPI** framework for async capabilities, automatic OpenAPI docs
152
155
  | :--- | :--- | :--- | :--- |
153
156
  | `GET` | `/health` | No | Liveness status & total chunk count |
154
157
  | `GET` | `/info` | Yes (Bearer token) | Configuration details & index state |
155
- | `POST` | `/search` | Yes (Bearer token) | Semantic vector search |
158
+ | `POST` | `/search` | Yes (Bearer token) | Vector, keyword or hybrid search (`mode`) |
156
159
  | `POST` | `/ask` | Yes (Bearer token) | Context Q&A generation (supports streaming via StreamingResponse) |
157
160
 
158
161
  ---
@@ -190,4 +193,8 @@ class VectorStore(ABC):
190
193
 
191
194
  @abstractmethod
192
195
  def read_index_metadata(self) -> Optional[IndexMetadata]: ...
196
+
197
+ # Optional (return None when unsupported):
198
+ def list_chunks(self) -> Optional[List[IndexedChunk]]: ... # rebuild a missing keyword index
199
+ def keyword_search(self, query: str, top_k: int) -> Optional[List[VectorSearchHit]]: ... # native keyword search
193
200
  ```
@@ -2,7 +2,32 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file.
4
4
 
5
- ## [1.2.2] - Unreleased
5
+ ## [1.3.0] - 2026-10-03
6
+
7
+ ### Added
8
+ - **Hybrid search:** `search()`, `ask()` and `ask_stream()` accept `mode="vector" | "keyword" | "hybrid"` (or `{"mode": ...}` in the options dict; default `"vector"`), on `Document` and `DocumentCollection`. Keyword mode uses BM25; hybrid merges vector and keyword results with Reciprocal Rank Fusion, so exact terms such as `ERR_4021`, SKUs or function names are found even when embeddings miss them. Tune it with `hybrid={"rrfK", "candidates", "weights"}`, or set defaults with the new `retrieval` option. Results in these modes carry `scores` (`vector`, `keyword`, `fused`); vector-mode results are unchanged.
9
+ - **Keyword index:** `build()` now also writes a BM25 index (`<storeDir>/<namespace>/keyword.json`) from the chunk texts. It needs no extra embedding calls, works with every vector store, and uses the same file layout as the TypeScript SDK.
10
+ - **Multilingual keyword tokenizer (`raglite-v1`):** NFKC and lowercase normalisation, code identifiers kept whole (`gpt-4.1`, `snake_case`), and character bigrams for Chinese, Japanese, Korean, Thai, Lao, Khmer and Myanmar. Exported as `tokenize()`. It produces exactly the same terms as the TypeScript SDK.
11
+ - **HTTP and CLI:** optional `mode` on `/search` and `/ask`; `--mode` on `raglite search`, `ask` and `serve`. `/info` reports `retrievalMode`.
12
+ - **`VectorStore` extension points (optional):** `list_chunks()` lets keyword and hybrid search rebuild a missing keyword index from the store, and `keyword_search(query, top_k)` replaces the built-in BM25 index. The memory and Qdrant stores implement `list_chunks()`. Existing custom stores need no changes.
13
+
14
+ ### Changed
15
+ - **Chunking of text without spaces:** Chinese, Japanese, Thai, Lao, Khmer and Myanmar text is now chunked by character instead of becoming one giant "word". Previously a document in these scripts became a single chunk of any length, which could exceed embedding model limits. `chunkSize` and `overlap` count characters for these scripts and words for everything else.
16
+
17
+ ### Fixed
18
+ - **Custom chunk sizes are kept:** `build()` without `chunk_size` or `overlap` (in the call or the constructor) now reuses the existing index's values instead of the defaults. Previously `raglite search` or `raglite ask` after `raglite index --chunk-size N` silently re-embedded the whole index at the default 500 words. Defaults still apply to a new index, and explicit values still trigger a rebuild when they differ.
19
+ - **Embedding provider is kept:** with no `embeddings` configured, `build()` now keeps the existing index's provider and model (reusing configured credentials when the provider matches) instead of switching to the local default. The CLI no longer assumes `--embed-provider local` when no `--embed-*` flag is given, so `raglite search` and `raglite ask` reuse an index built with `--embed-provider openai` instead of re-embedding it locally. New indexes still default to local embeddings.
20
+ - **`build(options)` embeddings dict:** `build({"embeddings": {...}})` now accepts a plain dict, as the constructor does.
21
+ - **`raglite serve`:** no longer crashes with `TypeError` when given `--llm-provider` or `--token`.
22
+ - **`DocumentCollection.serve()`:** no longer fails with `ImportError`, so serving a directory works. It now also falls back to the collection's configured `llm`.
23
+ - **`ask()` options:** an explicit `scoreThreshold` of `0` is no longer replaced by the configured default.
24
+
25
+ ### Upgrade notes
26
+ - The index format version is now 2. Indexes whose source contains no Chinese, Japanese, Thai, Lao, Khmer or Myanmar text are upgraded in place on the next `build()`, without re-embedding (the source file is read once to check). Indexes of sources that do contain such text are rebuilt once.
27
+ - Indexes built before 1.3.0 have no keyword index. The memory and Qdrant stores build it from the stored chunks on the first keyword or hybrid search. Pinecone and custom stores without `list_chunks()` log a warning and use vector search until you run `build(rebuild=True)`.
28
+ - With Qdrant or Pinecone, the keyword index lives on local disk under `storeDir`. Keep `storeDir` on persistent storage when you use keyword or hybrid search.
29
+
30
+ ## [1.2.2] - 2026-10-02
6
31
 
7
32
  ### Changed
8
33
  - **Upgrades keep cached indexes:** The build cache is now keyed on an index format version (`formatVersion` in `IndexMetadata`) instead of the package version, so upgrading RAGLite no longer re-embeds every index. Indexes built by 1.2.1 are reused as-is; indexes from older releases are rebuilt once.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: raglite-toolkit
3
- Version: 1.2.2
3
+ Version: 1.3.0
4
4
  Summary: Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python.
5
5
  Project-URL: Homepage, https://github.com/creatorpiyush/raglite-py
6
6
  Project-URL: Repository, https://github.com/creatorpiyush/raglite-py
@@ -59,6 +59,8 @@ Description-Content-Type: text/markdown
59
59
  - 🤖 **Multi-provider LLMs** — OpenAI, Anthropic (Claude), Google (Gemini), Mistral, Cohere, Groq, xAI, Ollama
60
60
  - 🔢 **Multi-provider embeddings** — OpenAI, Google, Mistral, Cohere, Voyage, Ollama, or a **local** offline sentence-transformer (no API key needed)
61
61
  - 📐 **Cosine similarity** scoring with L2-normalized vectors
62
+ - 🔎 **Hybrid search** — BM25 keyword search fused with vector search, for exact terms like error codes and SKUs
63
+ - 🌏 **Any language** — Chinese, Japanese, Korean and Thai text is chunked and keyword-indexed correctly
62
64
  - ♻️ **Content-hash cache** — reindexes only when the file actually changes
63
65
  - 🗂 **Per-document namespacing** — indexes are isolated, two documents never collide
64
66
  - 🌐 **REST API** via FastAPI with optional **bearer-token auth**
@@ -129,6 +131,41 @@ print(answer.text)
129
131
 
130
132
  ---
131
133
 
134
+ ## Hybrid Search (Keyword + Vector)
135
+
136
+ Vector search matches meaning, but it can miss exact terms such as error codes, SKUs, function names or rare product names. Keyword search (BM25) finds those exactly. `hybrid` runs both and merges the results with Reciprocal Rank Fusion.
137
+
138
+ ```python
139
+ doc.search("ERR_4021", mode="hybrid") # "vector" (default) | "keyword" | "hybrid"
140
+ doc.ask("What does ERR_4021 mean?", {"mode": "hybrid"})
141
+
142
+ # Or set a default once; it applies to search(), ask() and ask_stream().
143
+ Document("./runbook.md", {
144
+ "retrieval": {
145
+ "mode": "hybrid",
146
+ "hybrid": {"rrfK": 60, "candidates": 50, "weights": {"vector": 1, "keyword": 1}},
147
+ },
148
+ })
149
+ ```
150
+
151
+ - In `keyword` and `hybrid` modes, `score` is the fused rank score scaled to 0..1 (1 means ranked first by every retriever). Each result also has `scores` with `vector`, `keyword` and `fused`. Vector mode results are unchanged.
152
+ - `scoreThreshold` is still a cosine similarity. In hybrid mode it filters the vector results before fusion; keyword matches are not filtered by it.
153
+ - The keyword index is built by `build()` from the chunk texts, with no extra embedding calls, and saved as `<storeDir>/<namespace>/keyword.json`. It works with every vector store. With Qdrant or Pinecone, keep `storeDir` on persistent disk.
154
+ - Indexes built before 1.3: the memory and Qdrant stores build the keyword index from the stored chunks on the first keyword or hybrid search. Other stores log a warning and use vector search until you run `build(rebuild=True)`.
155
+ - The same options work over HTTP (`"mode"` on `/search` and `/ask`) and in the CLI (`--mode hybrid`).
156
+
157
+ The tokenizer is the same in the Python and TypeScript SDKs:
158
+
159
+ | Text | How it is indexed |
160
+ |------|-------------------|
161
+ | Latin, Cyrillic, Greek, Arabic, Devanagari and other spaced scripts | Words, lowercased and NFKC-normalised (`fi` → `fi`, `ABC` → `abc`) |
162
+ | Code identifiers | `gpt-4.1`, `snake_case` and `ERR_42` stay whole, and their parts are indexed too |
163
+ | Chinese, Japanese, Korean, Thai, Lao, Khmer, Myanmar | Overlapping character pairs (`退款处理` → `退款`, `款处`, `处理`), so no dictionary is needed |
164
+
165
+ There is no stemming or stopword list, because both are language-specific: in keyword mode `refund` does not match `refunds`. Hybrid mode's vector side covers those cases.
166
+
167
+ ---
168
+
132
169
  ## Fully Offline — No API Key Needed
133
170
 
134
171
  ```python
@@ -266,6 +303,7 @@ raglite index ./policy.pdf --embed-provider local
266
303
 
267
304
  # Semantic search
268
305
  raglite search ./docs "refund policy" --top-k 5
306
+ raglite search ./docs "ERR_4021" --mode hybrid
269
307
 
270
308
  # Ask a question (streaming)
271
309
  raglite ask ./docs "What is the refund policy?" \
@@ -313,12 +351,13 @@ raglite serve https://example.com \
313
351
  ```python
314
352
  Document("./policy.pdf", {
315
353
  # Chunking
316
- "chunkSize": 500, # words per chunk (default: 500)
354
+ "chunkSize": 500, # words per chunk; characters for Chinese, Japanese, Thai, ... (default: 500)
317
355
  "overlap": 50, # overlapping words between chunks (default: 50)
318
356
 
319
357
  # Retrieval
320
358
  "topK": 5, # default results returned (default: 5)
321
359
  "scoreThreshold": 0.0, # minimum cosine similarity (0..1, default: 0)
360
+ "retrieval": {"mode": "vector"}, # "vector" | "keyword" | "hybrid" (default: vector)
322
361
 
323
362
  # Storage
324
363
  "storeDir": ".raglite", # where indexes are persisted (default: .raglite)
@@ -341,9 +380,9 @@ Every `build()` call fingerprints the source (file bytes, or the fetched text fo
341
380
  | Factor | Triggers rebuild if changed |
342
381
  |--------|-----------------------------|
343
382
  | File content | SHA-256 hash differs |
344
- | Chunk size | `chunkSize` changed |
345
- | Overlap | `overlap` changed |
346
- | Embedding provider/model | Provider or model string changed |
383
+ | Chunk size | `chunkSize` changed (when not set, the existing index's value is kept) |
384
+ | Overlap | `overlap` changed (when not set, the existing index's value is kept) |
385
+ | Embedding provider/model | Provider or model string changed (when no `embeddings` are configured, the existing index's are kept) |
347
386
  | Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
348
387
 
349
388
  Pass `rebuild=True` to `build()` to force a fresh index regardless.
@@ -362,6 +401,9 @@ from raglite.vectordb.base import VectorStore
362
401
  class MyVectorStore(VectorStore):
363
402
  # Implement: namespace (property), load, reset, add, search, count,
364
403
  # save_index_metadata, read_index_metadata
404
+ # Optional: list_chunks() lets hybrid search rebuild a missing keyword index
405
+ # from your store; keyword_search(query, top_k) replaces the
406
+ # built-in BM25 index with your own.
365
407
  ...
366
408
  ```
367
409
 
@@ -13,6 +13,8 @@
13
13
  - 🤖 **Multi-provider LLMs** — OpenAI, Anthropic (Claude), Google (Gemini), Mistral, Cohere, Groq, xAI, Ollama
14
14
  - 🔢 **Multi-provider embeddings** — OpenAI, Google, Mistral, Cohere, Voyage, Ollama, or a **local** offline sentence-transformer (no API key needed)
15
15
  - 📐 **Cosine similarity** scoring with L2-normalized vectors
16
+ - 🔎 **Hybrid search** — BM25 keyword search fused with vector search, for exact terms like error codes and SKUs
17
+ - 🌏 **Any language** — Chinese, Japanese, Korean and Thai text is chunked and keyword-indexed correctly
16
18
  - ♻️ **Content-hash cache** — reindexes only when the file actually changes
17
19
  - 🗂 **Per-document namespacing** — indexes are isolated, two documents never collide
18
20
  - 🌐 **REST API** via FastAPI with optional **bearer-token auth**
@@ -83,6 +85,41 @@ print(answer.text)
83
85
 
84
86
  ---
85
87
 
88
+ ## Hybrid Search (Keyword + Vector)
89
+
90
+ Vector search matches meaning, but it can miss exact terms such as error codes, SKUs, function names or rare product names. Keyword search (BM25) finds those exactly. `hybrid` runs both and merges the results with Reciprocal Rank Fusion.
91
+
92
+ ```python
93
+ doc.search("ERR_4021", mode="hybrid") # "vector" (default) | "keyword" | "hybrid"
94
+ doc.ask("What does ERR_4021 mean?", {"mode": "hybrid"})
95
+
96
+ # Or set a default once; it applies to search(), ask() and ask_stream().
97
+ Document("./runbook.md", {
98
+ "retrieval": {
99
+ "mode": "hybrid",
100
+ "hybrid": {"rrfK": 60, "candidates": 50, "weights": {"vector": 1, "keyword": 1}},
101
+ },
102
+ })
103
+ ```
104
+
105
+ - In `keyword` and `hybrid` modes, `score` is the fused rank score scaled to 0..1 (1 means ranked first by every retriever). Each result also has `scores` with `vector`, `keyword` and `fused`. Vector mode results are unchanged.
106
+ - `scoreThreshold` is still a cosine similarity. In hybrid mode it filters the vector results before fusion; keyword matches are not filtered by it.
107
+ - The keyword index is built by `build()` from the chunk texts, with no extra embedding calls, and saved as `<storeDir>/<namespace>/keyword.json`. It works with every vector store. With Qdrant or Pinecone, keep `storeDir` on persistent disk.
108
+ - Indexes built before 1.3: the memory and Qdrant stores build the keyword index from the stored chunks on the first keyword or hybrid search. Other stores log a warning and use vector search until you run `build(rebuild=True)`.
109
+ - The same options work over HTTP (`"mode"` on `/search` and `/ask`) and in the CLI (`--mode hybrid`).
110
+
111
+ The tokenizer is the same in the Python and TypeScript SDKs:
112
+
113
+ | Text | How it is indexed |
114
+ |------|-------------------|
115
+ | Latin, Cyrillic, Greek, Arabic, Devanagari and other spaced scripts | Words, lowercased and NFKC-normalised (`fi` → `fi`, `ABC` → `abc`) |
116
+ | Code identifiers | `gpt-4.1`, `snake_case` and `ERR_42` stay whole, and their parts are indexed too |
117
+ | Chinese, Japanese, Korean, Thai, Lao, Khmer, Myanmar | Overlapping character pairs (`退款处理` → `退款`, `款处`, `处理`), so no dictionary is needed |
118
+
119
+ There is no stemming or stopword list, because both are language-specific: in keyword mode `refund` does not match `refunds`. Hybrid mode's vector side covers those cases.
120
+
121
+ ---
122
+
86
123
  ## Fully Offline — No API Key Needed
87
124
 
88
125
  ```python
@@ -220,6 +257,7 @@ raglite index ./policy.pdf --embed-provider local
220
257
 
221
258
  # Semantic search
222
259
  raglite search ./docs "refund policy" --top-k 5
260
+ raglite search ./docs "ERR_4021" --mode hybrid
223
261
 
224
262
  # Ask a question (streaming)
225
263
  raglite ask ./docs "What is the refund policy?" \
@@ -267,12 +305,13 @@ raglite serve https://example.com \
267
305
  ```python
268
306
  Document("./policy.pdf", {
269
307
  # Chunking
270
- "chunkSize": 500, # words per chunk (default: 500)
308
+ "chunkSize": 500, # words per chunk; characters for Chinese, Japanese, Thai, ... (default: 500)
271
309
  "overlap": 50, # overlapping words between chunks (default: 50)
272
310
 
273
311
  # Retrieval
274
312
  "topK": 5, # default results returned (default: 5)
275
313
  "scoreThreshold": 0.0, # minimum cosine similarity (0..1, default: 0)
314
+ "retrieval": {"mode": "vector"}, # "vector" | "keyword" | "hybrid" (default: vector)
276
315
 
277
316
  # Storage
278
317
  "storeDir": ".raglite", # where indexes are persisted (default: .raglite)
@@ -295,9 +334,9 @@ Every `build()` call fingerprints the source (file bytes, or the fetched text fo
295
334
  | Factor | Triggers rebuild if changed |
296
335
  |--------|-----------------------------|
297
336
  | File content | SHA-256 hash differs |
298
- | Chunk size | `chunkSize` changed |
299
- | Overlap | `overlap` changed |
300
- | Embedding provider/model | Provider or model string changed |
337
+ | Chunk size | `chunkSize` changed (when not set, the existing index's value is kept) |
338
+ | Overlap | `overlap` changed (when not set, the existing index's value is kept) |
339
+ | Embedding provider/model | Provider or model string changed (when no `embeddings` are configured, the existing index's are kept) |
301
340
  | Index format | Stored index layout changed by a release (rare; ordinary upgrades reuse the index) |
302
341
 
303
342
  Pass `rebuild=True` to `build()` to force a fresh index regardless.
@@ -316,6 +355,9 @@ from raglite.vectordb.base import VectorStore
316
355
  class MyVectorStore(VectorStore):
317
356
  # Implement: namespace (property), load, reset, add, search, count,
318
357
  # save_index_metadata, read_index_metadata
358
+ # Optional: list_chunks() lets hybrid search rebuild a missing keyword index
359
+ # from your store; keyword_search(query, top_k) replaces the
360
+ # built-in BM25 index with your own.
319
361
  ...
320
362
  ```
321
363
 
@@ -8,7 +8,7 @@ packages = ["src/raglite"]
8
8
 
9
9
  [project]
10
10
  name = "raglite-toolkit"
11
- version = "1.2.2"
11
+ version = "1.3.0"
12
12
  description = "Build semantic search, multi-provider question answering, and REST APIs over your documents in a few lines of Python."
13
13
  readme = "README.md"
14
14
  requires-python = ">=3.10"
@@ -41,20 +41,34 @@ from .loaders import (
41
41
  is_supported_file,
42
42
  is_url,
43
43
  )
44
- from .retrieval import Retriever
44
+ from .retrieval import (
45
+ KeywordHit,
46
+ KeywordIndex,
47
+ RankedList,
48
+ RetrievalPlan,
49
+ Retriever,
50
+ reciprocal_rank_fusion,
51
+ resolve_retrieval_plan,
52
+ )
53
+ from .text import TOKENIZER_NAME, tokenize
45
54
  from .types import (
46
55
  AnswerResult,
47
56
  ChunkMetadata,
48
57
  EmbeddingProviderConfig,
49
58
  EmbeddingProviderName,
59
+ HybridOptions,
50
60
  IndexMetadata,
51
61
  LLMProviderConfig,
52
62
  LLMProviderName,
63
+ RetrievalMode,
64
+ RetrievalOptions,
53
65
  SearchResult,
66
+ SearchScores,
54
67
  StoredChunk,
55
68
  )
56
69
  from .utils.logger import create_logger
57
70
  from .vectordb import MemoryVectorStore
71
+ from .vectordb.base import IndexedChunk, VectorSearchHit, VectorStore
58
72
 
59
73
  VERSION = PACKAGE_VERSION
60
74
 
@@ -78,6 +92,10 @@ __all__ = [
78
92
  "ChunkMetadata",
79
93
  "StoredChunk",
80
94
  "SearchResult",
95
+ "SearchScores",
96
+ "RetrievalMode",
97
+ "RetrievalOptions",
98
+ "HybridOptions",
81
99
  "AnswerResult",
82
100
  "IndexMetadata",
83
101
  "DocumentOptions",
@@ -97,7 +115,18 @@ __all__ = [
97
115
  "BaseChunker",
98
116
  "RecursiveChunker",
99
117
  "MemoryVectorStore",
118
+ "VectorStore",
119
+ "VectorSearchHit",
120
+ "IndexedChunk",
100
121
  "Retriever",
122
+ "KeywordIndex",
123
+ "KeywordHit",
124
+ "RankedList",
125
+ "RetrievalPlan",
126
+ "reciprocal_rank_fusion",
127
+ "resolve_retrieval_plan",
128
+ "tokenize",
129
+ "TOKENIZER_NAME",
101
130
  "DEFAULT_EMBEDDING_MODELS",
102
131
  "LocalEmbedder",
103
132
  "RemoteEmbedder",
@@ -1,4 +1,4 @@
1
- from typing import Optional
1
+ from typing import Literal, Optional
2
2
 
3
3
  from pydantic import BaseModel, ConfigDict, Field
4
4
 
@@ -11,6 +11,7 @@ class SearchRequest(BaseModel):
11
11
  scoreThreshold: Optional[float] = Field(
12
12
  default=None, alias="scoreThreshold", ge=-1.0, le=1.0
13
13
  )
14
+ mode: Optional[Literal["vector", "keyword", "hybrid"]] = None
14
15
 
15
16
 
16
17
  class AskRequest(BaseModel):
@@ -21,6 +22,7 @@ class AskRequest(BaseModel):
21
22
  scoreThreshold: Optional[float] = Field(
22
23
  default=None, alias="scoreThreshold", ge=-1.0, le=1.0
23
24
  )
25
+ mode: Optional[Literal["vector", "keyword", "hybrid"]] = None
24
26
  includeCitations: Optional[bool] = Field(
25
27
  default=None, alias="includeCitations"
26
28
  )
@@ -103,6 +103,7 @@ def build_app(document: Any, options: Dict[str, Any]) -> FastAPI:
103
103
  "chunkSize": cfg.chunkSize,
104
104
  "overlap": cfg.overlap,
105
105
  "topK": cfg.topK,
106
+ "retrievalMode": cfg.retrieval.mode or "vector",
106
107
  "embeddings": {
107
108
  "provider": cfg.embeddings.provider,
108
109
  "model": cfg.embeddings.model
@@ -118,6 +119,7 @@ def build_app(document: Any, options: Dict[str, Any]) -> FastAPI:
118
119
  req.query,
119
120
  top_k=req.topK,
120
121
  score_threshold=req.scoreThreshold,
122
+ mode=req.mode,
121
123
  )
122
124
  return {
123
125
  "results": [r.model_dump(by_alias=True) for r in results]
@@ -133,6 +135,7 @@ def build_app(document: Any, options: Dict[str, Any]) -> FastAPI:
133
135
  "llm": ask_provider,
134
136
  "topK": req.topK,
135
137
  "scoreThreshold": req.scoreThreshold,
138
+ "mode": req.mode,
136
139
  "includeCitations": req.includeCitations,
137
140
  }
138
141
 
@@ -0,0 +1,90 @@
1
+ import re
2
+ from typing import List, Tuple
3
+
4
+ from ..errors import ChunkingError
5
+ from ..text.scripts import has_unspaced_text, is_mark, is_unspaced_char
6
+ from .base import BaseChunker
7
+
8
+ # The exact set JavaScript's \s matches. Python's \s differs (it includes
9
+ # \x1c-\x1f and \x85 but not ), which would make the two SDKs chunk the
10
+ # same text differently.
11
+ _JS_WHITESPACE = re.compile(
12
+ "[\t\n\v\f\r    - 

   ]+"
13
+ )
14
+
15
+ # One chunking unit: (text, spaced). ``spaced`` is False when the unit
16
+ # continues the previous unit's word, so no space goes between them.
17
+ _Unit = Tuple[str, bool]
18
+
19
+
20
+ class RecursiveChunker(BaseChunker):
21
+ """Word-based recursive chunker with overlap.
22
+
23
+ Scripts written without spaces (Chinese, Japanese, Thai, ...) have no word
24
+ boundaries to split on, so each of their characters counts as one unit.
25
+ Without this a whole unspaced document would be a single "word" and
26
+ therefore a single chunk of any length.
27
+ """
28
+
29
+ def split(self, text: str) -> List[str]:
30
+ if not _JS_WHITESPACE.sub("", text):
31
+ return []
32
+ if self.overlap >= self.chunk_size:
33
+ raise ChunkingError(
34
+ f"overlap ({self.overlap}) must be smaller than chunkSize ({self.chunk_size})"
35
+ )
36
+
37
+ words = [w for w in _JS_WHITESPACE.split(text) if w]
38
+ if has_unspaced_text(text):
39
+ units = [u for w in words for u in _word_units(w)]
40
+ else:
41
+ units = [(w, True) for w in words]
42
+ if len(units) <= self.chunk_size:
43
+ return [_render(units)]
44
+
45
+ step = self.chunk_size - self.overlap
46
+ chunks: List[str] = []
47
+
48
+ start = 0
49
+ while start < len(units):
50
+ end = start + self.chunk_size
51
+ slice_units = units[start:end]
52
+ if not slice_units:
53
+ break
54
+ chunks.append(_render(slice_units))
55
+ if end >= len(units):
56
+ break
57
+ start += step
58
+
59
+ return chunks
60
+
61
+
62
+ def _word_units(word: str) -> List[_Unit]:
63
+ """Split a word into units: each unspaced-script character (with its
64
+ combining marks) is a unit, and every run of other characters is a unit."""
65
+ units: List[List] = []
66
+ run = ""
67
+ for ch in word:
68
+ if is_mark(ch) and run == "" and units:
69
+ units[-1][0] += ch
70
+ elif is_mark(ch):
71
+ run += ch
72
+ elif is_unspaced_char(ord(ch)):
73
+ if run:
74
+ units.append([run, not units])
75
+ run = ""
76
+ units.append([ch, not units])
77
+ else:
78
+ run += ch
79
+ if run:
80
+ units.append([run, not units])
81
+ return [(text, spaced) for text, spaced in units]
82
+
83
+
84
+ def _render(units: List[_Unit]) -> str:
85
+ out = ""
86
+ for i, (text, spaced) in enumerate(units):
87
+ if i > 0 and spaced:
88
+ out += " "
89
+ out += text
90
+ return out
@@ -5,7 +5,7 @@ import sys
5
5
  import time
6
6
  from typing import Optional, Union
7
7
 
8
- from .constants import PACKAGE_VERSION
8
+ from .constants import DEFAULT_HOST, DEFAULT_PORT, PACKAGE_VERSION
9
9
  from .core.collection import DocumentCollection
10
10
  from .core.document import Document
11
11
  from .loaders import is_url
@@ -14,9 +14,9 @@ HELP = f"""raglite v{PACKAGE_VERSION}
14
14
 
15
15
  Usage:
16
16
  raglite index <path|url> [--chunk-size N] [--overlap N] [--embed-provider P] [--embed-model M] [--embed-key K] [--rebuild]
17
- raglite search <path|url> "query" [--top-k N]
18
- raglite ask <path|url> "question" --llm-provider P [--llm-model M] [--llm-key K] [--stream]
19
- raglite serve <path|url> --llm-provider P [--llm-key K] [--host H] [--port N] [--token T]
17
+ raglite search <path|url> "query" [--top-k N] [--mode vector|keyword|hybrid]
18
+ raglite ask <path|url> "question" --llm-provider P [--llm-model M] [--llm-key K] [--stream] [--mode M]
19
+ raglite serve <path|url> --llm-provider P [--llm-key K] [--host H] [--port N] [--token T] [--mode M]
20
20
  raglite --help
21
21
  raglite --version
22
22
 
@@ -26,7 +26,13 @@ Providers:
26
26
  """
27
27
 
28
28
 
29
- def parse_common_embedding(args_dict: dict) -> dict:
29
+ _MODES = ("vector", "keyword", "hybrid")
30
+
31
+
32
+ def parse_common_embedding(args_dict: dict) -> Optional[dict]:
33
+ """None when no --embed-* flag is given, so an existing index keeps its provider."""
34
+ if not any(args_dict.get(k) for k in ("embed_provider", "embed_model", "embed_key")):
35
+ return None
30
36
  provider = args_dict.get("embed_provider") or "local"
31
37
  config = {"provider": provider}
32
38
  if args_dict.get("embed_model"):
@@ -36,6 +42,12 @@ def parse_common_embedding(args_dict: dict) -> dict:
36
42
  return config
37
43
 
38
44
 
45
+ def _with_embeddings(options: dict, embeddings: Optional[dict]) -> dict:
46
+ if embeddings is not None:
47
+ options["embeddings"] = embeddings
48
+ return options
49
+
50
+
39
51
  def parse_llm(args_dict: dict) -> Optional[dict]:
40
52
  provider = args_dict.get("llm_provider")
41
53
  if not provider:
@@ -70,7 +82,7 @@ def run_index(args):
70
82
  parsed = parser.parse_args(args)
71
83
 
72
84
  embeddings = parse_common_embedding(vars(parsed))
73
- target = resolve_target(parsed.file, {"embeddings": embeddings})
85
+ target = resolve_target(parsed.file, _with_embeddings({}, embeddings))
74
86
 
75
87
  build_opts = {}
76
88
  if parsed.chunk_size is not None:
@@ -90,6 +102,7 @@ def run_search(args):
90
102
  parser.add_argument("file")
91
103
  parser.add_argument("query")
92
104
  parser.add_argument("--top-k", type=int)
105
+ parser.add_argument("--mode", choices=_MODES)
93
106
  parser.add_argument("--embed-provider")
94
107
  parser.add_argument("--embed-model")
95
108
  parser.add_argument("--embed-key")
@@ -97,11 +110,13 @@ def run_search(args):
97
110
  parsed = parser.parse_args(args)
98
111
 
99
112
  embeddings = parse_common_embedding(vars(parsed))
100
- target = resolve_target(parsed.file, {"embeddings": embeddings})
113
+ target = resolve_target(parsed.file, _with_embeddings({}, embeddings))
101
114
 
102
115
  search_opts = {}
103
116
  if parsed.top_k is not None:
104
117
  search_opts["topK"] = parsed.top_k
118
+ if parsed.mode is not None:
119
+ search_opts["mode"] = parsed.mode
105
120
 
106
121
  results = target.search(parsed.query, search_opts)
107
122
  serialized = [r.model_dump(by_alias=True) for r in results]
@@ -120,17 +135,20 @@ def run_ask(args):
120
135
  parser.add_argument("--llm-model")
121
136
  parser.add_argument("--llm-key")
122
137
  parser.add_argument("--stream", action="store_true")
138
+ parser.add_argument("--mode", choices=_MODES)
123
139
 
124
140
  parsed = parser.parse_args(args)
125
141
 
126
142
  embeddings = parse_common_embedding(vars(parsed))
127
143
  llm = parse_llm(vars(parsed))
128
144
 
129
- target = resolve_target(parsed.file, {"embeddings": embeddings, "llm": llm})
145
+ target = resolve_target(parsed.file, _with_embeddings({"llm": llm}, embeddings))
130
146
 
131
147
  opts = {}
132
148
  if parsed.top_k is not None:
133
149
  opts["topK"] = parsed.top_k
150
+ if parsed.mode is not None:
151
+ opts["mode"] = parsed.mode
134
152
 
135
153
  if parsed.stream:
136
154
  for chunk in target.ask_stream(parsed.question, opts):
@@ -154,42 +172,40 @@ def run_serve(args):
154
172
  parser.add_argument("--host")
155
173
  parser.add_argument("--port", type=int)
156
174
  parser.add_argument("--token")
175
+ parser.add_argument("--mode", choices=_MODES)
157
176
 
158
177
  parsed = parser.parse_args(args)
159
178
 
160
179
  embeddings = parse_common_embedding(vars(parsed))
161
180
  llm = parse_llm(vars(parsed))
162
181
 
163
- target = resolve_target(
164
- parsed.file,
165
- {"embeddings": embeddings, **({"llm": llm} if llm else {})},
166
- )
182
+ options: dict = _with_embeddings({}, embeddings)
183
+ if llm:
184
+ options["llm"] = llm
185
+ if parsed.mode:
186
+ options["retrieval"] = {"mode": parsed.mode}
187
+ target = resolve_target(parsed.file, options)
167
188
  target.build()
168
189
 
169
- serve_opts = {}
170
- if llm:
171
- serve_opts["llm"] = llm
172
- if parsed.host:
173
- serve_opts["host"] = parsed.host
174
- if parsed.port is not None:
175
- serve_opts["port"] = parsed.port
176
- if parsed.token:
177
- serve_opts["bearerToken"] = parsed.token
178
-
179
- if hasattr(target, "serve"):
180
- target.serve(**serve_opts)
181
- else:
182
- from .api.server import create_server
183
- handle = create_server(target, serve_opts)
184
- sys.stdout.write(f"RagLite listening on {handle.url}\n")
185
- sys.stdout.flush()
190
+ host = parsed.host or DEFAULT_HOST
191
+ port = parsed.port if parsed.port is not None else DEFAULT_PORT
192
+
193
+ if isinstance(target, DocumentCollection):
194
+ # Blocks until the server stops.
195
+ target.serve(host=host, port=port, bearer_token=parsed.token, llm=llm)
196
+ return
186
197
 
187
- try:
188
- while True:
189
- time.sleep(1)
190
- except KeyboardInterrupt:
191
- handle.close()
192
- sys.exit(0)
198
+ handle = target.serve(
199
+ {"llm": llm} if llm else None, host=host, port=port, bearer_token=parsed.token
200
+ )
201
+ sys.stdout.write(f"RagLite listening on {handle.url}\n")
202
+ sys.stdout.flush()
203
+ try:
204
+ while True:
205
+ time.sleep(1)
206
+ except KeyboardInterrupt:
207
+ handle.close()
208
+ sys.exit(0)
193
209
 
194
210
 
195
211
  def main():