knowledge-rag 4.2.0__tar.gz → 4.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (21) hide show
  1. knowledge_rag-4.2.0/README.md → knowledge_rag-4.3.1/PKG-INFO +193 -51
  2. knowledge_rag-4.2.0/PKG-INFO → knowledge_rag-4.3.1/README.md +145 -91
  3. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/__init__.py +1 -1
  4. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/server.py +159 -16
  5. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/pyproject.toml +18 -4
  6. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/.gitignore +0 -0
  7. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/LICENSE +0 -0
  8. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/config.example.yaml +0 -0
  9. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/config.py +0 -0
  10. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/guarded.py +0 -0
  11. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/ingestion.py +0 -0
  12. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/instance_lock.py +0 -0
  13. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/metrics.py +0 -0
  14. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/preflight.py +0 -0
  15. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/mcp_server/ratelimit.py +0 -0
  16. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/npm/README.md +0 -0
  17. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/presets/cybersecurity.yaml +0 -0
  18. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/presets/developer.yaml +0 -0
  19. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/presets/general.yaml +0 -0
  20. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/presets/research.yaml +0 -0
  21. {knowledge_rag-4.2.0 → knowledge_rag-4.3.1}/requirements.txt +0 -0
@@ -1,3 +1,51 @@
1
+ Metadata-Version: 2.4
2
+ Name: knowledge-rag
3
+ Version: 4.3.1
4
+ Summary: Local RAG System for Claude Code — Hybrid search + Cross-encoder Reranking + 13 MCP Tools + 20 Format Parsers. Zero external servers.
5
+ Project-URL: Homepage, https://github.com/lyonzin/knowledge-rag
6
+ Project-URL: Repository, https://github.com/lyonzin/knowledge-rag
7
+ Project-URL: Issues, https://github.com/lyonzin/knowledge-rag/issues
8
+ Project-URL: Changelog, https://github.com/lyonzin/knowledge-rag/releases
9
+ Author-email: "Lyon." <lyonzin@users.noreply.github.com>
10
+ License: MIT
11
+ License-File: LICENSE
12
+ Keywords: bm25,chromadb,claude-code,embeddings,fastembed,hybrid-search,knowledge-base,local-ai,mcp,rag,reranking,retrieval-augmented-generation,semantic-search
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Topic :: Text Processing :: Indexing
22
+ Requires-Python: >=3.11
23
+ Requires-Dist: beautifulsoup4>=4.12.0
24
+ Requires-Dist: chromadb>=1.4.0
25
+ Requires-Dist: fastembed[reranking]>=0.4.0
26
+ Requires-Dist: mcp>=1.6.0
27
+ Requires-Dist: numpy>=1.24.0
28
+ Requires-Dist: openpyxl>=3.1.0
29
+ Requires-Dist: pymupdf>=1.23.0
30
+ Requires-Dist: python-docx>=1.0.0
31
+ Requires-Dist: python-pptx>=1.0.0
32
+ Requires-Dist: pyyaml>=6.0
33
+ Requires-Dist: requests>=2.33.0
34
+ Requires-Dist: watchdog>=4.0.0
35
+ Provides-Extra: gpu
36
+ Requires-Dist: nvidia-cublas-cu12; extra == 'gpu'
37
+ Requires-Dist: nvidia-cuda-runtime-cu12; extra == 'gpu'
38
+ Requires-Dist: nvidia-cudnn-cu12; extra == 'gpu'
39
+ Requires-Dist: nvidia-cufft-cu12; extra == 'gpu'
40
+ Requires-Dist: nvidia-curand-cu12; extra == 'gpu'
41
+ Requires-Dist: nvidia-cusolver-cu12; extra == 'gpu'
42
+ Requires-Dist: nvidia-cusparse-cu12; extra == 'gpu'
43
+ Requires-Dist: nvidia-nvjitlink-cu12; extra == 'gpu'
44
+ Requires-Dist: onnxruntime-gpu>=1.14.0; extra == 'gpu'
45
+ Provides-Extra: server
46
+ Requires-Dist: uvicorn>=0.20.0; extra == 'server'
47
+ Description-Content-Type: text/markdown
48
+
1
49
  # Knowledge RAG
2
50
 
3
51
  <div align="center">
@@ -17,7 +65,7 @@
17
65
  ### Your docs, your machine, zero cloud. Claude Code searches them natively.
18
66
 
19
67
  Drop your PDFs, markdown, code, notebooks — **1800+ files, 39K chunks, indexed in under 3 minutes.**<br/>
20
- Hybrid search (BM25 + semantic vectors + cross-encoder reranking) through 12 MCP tools.<br/>
68
+ Hybrid search (BM25 + semantic vectors + cross-encoder reranking) through 13 MCP tools.<br/>
21
69
  Everything runs locally via ONNX. No Docker, no Ollama, no API keys, no data leaves your machine.
22
70
 
23
71
  ```
@@ -26,7 +74,7 @@ pip install knowledge-rag → restart Claude Code → search_knowledge("your que
26
74
 
27
75
  ---
28
76
 
29
- **12 MCP Tools** | **Hybrid Search + Reranking** | **20 File Formats** | **Optional NVIDIA GPU** | **100% Local**
77
+ **13 MCP Tools** | **Hybrid Search + Reranking** | **20 File Formats** | **Optional NVIDIA GPU** | **100% Local**
30
78
 
31
79
  [What's New](#whats-new-in-v420) | [Supported Formats](#supported-formats) | [Installation](#installation) | [Configuration](#configuration) | [API Reference](#api-reference) | [Architecture](#architecture)
32
80
 
@@ -81,7 +129,7 @@ Or via CLI: `knowledge-rag --transport sse`
81
129
  - **Prometheus metrics**: `/metrics` endpoint on separate port
82
130
  - **Bearer auth**: Token validation for SSE/HTTP connections
83
131
 
84
- All 12 MCP tools are instrumented with `@rate_limited` and `@instrument` decorators — zero overhead when features are disabled. Default transport remains **stdio** for full backwards compatibility.
132
+ All 13 MCP tools are instrumented with `@rate_limited` and `@instrument` decorators — zero overhead when features are disabled. Default transport remains **stdio** for full backwards compatibility.
85
133
 
86
134
  > **Migration**: Existing users need zero changes. SSE mode is opt-in via `server.transport: "sse"` in config.yaml. See [Configuration](#configuration) for details.
87
135
 
@@ -205,7 +253,7 @@ See [Changelog](#changelog) for full history.
205
253
  | **MMR Diversification** | Maximal Marginal Relevance reduces redundant results |
206
254
  | **Persistent Model Cache** | Embedding models cached in `models_cache/` — survives reboots |
207
255
  | **Auto-Migration** | Detects embedding dimension mismatch and rebuilds automatically |
208
- | **12 MCP Tools** | Full CRUD + search + evaluation via Claude Code |
256
+ | **13 MCP Tools** | Full CRUD + search + evaluation via Claude Code |
209
257
 
210
258
  ---
211
259
 
@@ -217,7 +265,7 @@ See [Changelog](#changelog) for full history.
217
265
  flowchart TB
218
266
  subgraph MCP["MCP SERVER (FastMCP)"]
219
267
  direction TB
220
- TOOLS["12 MCP Tools<br/>search | get | add | update | remove<br/>reindex | list | stats | url | similar | evaluate"]
268
+ TOOLS["13 MCP Tools<br/>search | get | add | update | remove<br/>reindex | reindex_status | list | stats | url | similar | evaluate"]
221
269
  end
222
270
 
223
271
  subgraph SEARCH["HYBRID SEARCH ENGINE"]
@@ -392,7 +440,53 @@ flowchart LR
392
440
  - Claude Code CLI
393
441
  - *…or any other MCP client (Claude Desktop, Cursor, VS Code, Antigravity, opencode, Windsurf) — see [Use with other MCP clients](#use-with-other-mcp-clients)*
394
442
  - ~200MB disk for model cache (auto-downloaded on first run)
395
- - *Optional:* NVIDIA GPU + CUDA for accelerated embeddings (`pip install knowledge-rag[gpu]` + `models.embedding.gpu: true` in config)
443
+ - *Optional:* NVIDIA GPU + CUDA 12 for accelerated embeddings (see [GPU Acceleration](#gpu-acceleration) below)
444
+
445
+ ### GPU Acceleration
446
+
447
+ GPU mode accelerates embedding generation during indexing and search. It requires an NVIDIA GPU with CUDA 12 support. No GPU? No problem — the server runs on CPU by default and GPU is entirely optional.
448
+
449
+ **Requirements:**
450
+
451
+ | Component | Minimum | How to check / get it |
452
+ |-----------|---------|----------------------|
453
+ | NVIDIA GPU (Turing+) | RTX 20xx / 30xx / 40xx / 50xx, or Tesla T4+ | `nvidia-smi` |
454
+ | NVIDIA Driver | ≥ 525 | `nvidia-smi` — [nvidia.com/drivers](https://www.nvidia.com/drivers) |
455
+ | CUDA 12 runtime | Provided by pip packages below | Automatic |
456
+
457
+ **Setup (2 steps):**
458
+
459
+ ```bash
460
+ # 1. Install GPU dependencies (onnxruntime-gpu + all CUDA 12 runtime DLLs)
461
+ pip install knowledge-rag[gpu]
462
+
463
+ # 2. Enable in config.yaml
464
+ # models:
465
+ # embedding:
466
+ # gpu: true
467
+ ```
468
+
469
+ The `[gpu]` extra installs `onnxruntime-gpu` plus 7 NVIDIA CUDA 12 packages (`cublas`, `cudnn`, `cuda-runtime`, `cufft`, `cusparse`, `cusolver`, `curand`, `nvjitlink`) so you don't need a full CUDA Toolkit install.
470
+
471
+ **Verify GPU is active:**
472
+
473
+ On server startup, look for the GPU status banner:
474
+ ```
475
+ ============================================================
476
+ GPU STATUS: ACTIVE
477
+ Provider: CUDAExecutionProvider
478
+ Device: NVIDIA GeForce RTX 3080 Ti
479
+ VRAM: 12.0 GB
480
+ ============================================================
481
+ ```
482
+
483
+ Or programmatically:
484
+ ```bash
485
+ python -c "import onnxruntime; print(onnxruntime.get_available_providers())"
486
+ # Should include: 'CUDAExecutionProvider'
487
+ ```
488
+
489
+ > **Fallback**: If CUDA is unavailable at runtime (wrong driver, missing DLLs, no GPU), the server falls back to CPU automatically with a `[WARN]` log — it never crashes. The `gpu: true` config is a preference, not a requirement.
396
490
 
397
491
  ### Install Methods
398
492
 
@@ -663,7 +757,7 @@ search_knowledge("lateral movement strategies", hybrid_alpha=1.0)
663
757
 
664
758
  ### Indexing
665
759
 
666
- Documents are automatically indexed on first startup. To manage the index:
760
+ Documents are automatically indexed on first startup. All reindex operations run **in background** — they return immediately and you poll progress via `get_reindex_status()`:
667
761
 
668
762
  ```python
669
763
  # Incremental: only re-index changed files (fast)
@@ -674,6 +768,10 @@ reindex_documents(force=True)
674
768
 
675
769
  # Nuclear rebuild: delete everything, re-embed all (use after model change)
676
770
  reindex_documents(full_rebuild=True)
771
+
772
+ # Poll progress (lightweight, no full stats computation)
773
+ get_reindex_status()
774
+ # → {"reindex": {"active": true, "percent": 56, "progress": "2090/3734", ...}}
677
775
  ```
678
776
 
679
777
  ### Evaluating Retrieval Quality
@@ -755,14 +853,39 @@ Retrieve the full content of a specific document.
755
853
 
756
854
  #### `reindex_documents`
757
855
 
758
- Index or reindex all documents in the knowledge base.
856
+ Index or reindex all documents in the knowledge base. **Runs in background** — returns immediately. Poll progress via `get_reindex_status()`.
759
857
 
760
858
  | Parameter | Type | Default | Description |
761
859
  |-----------|------|---------|-------------|
762
860
  | `force` | bool | false | Smart reindex: detects changes, rebuilds BM25. Fast. |
763
861
  | `full_rebuild` | bool | false | Nuclear rebuild: deletes everything, re-embeds all documents. Use after model change. |
764
862
 
765
- **Returns:** JSON with indexing statistics (indexed, updated, skipped, deleted, chunks_added, chunks_removed, dedup_skipped, elapsed_seconds).
863
+ **Returns:** `{"status": "started", "operation": "..."}` immediately. If already running, returns `{"status": "already_running", "progress": "1200/3734"}`.
864
+
865
+ ---
866
+
867
+ #### `get_reindex_status`
868
+
869
+ Get the current status of a background reindex operation. Lightweight — does not compute full index statistics.
870
+
871
+ **Returns (active):**
872
+ ```json
873
+ {
874
+ "status": "success",
875
+ "reindex": {
876
+ "active": true,
877
+ "operation": "nuclear_rebuild",
878
+ "progress": "1200/3734",
879
+ "percent": 32,
880
+ "indexed": 1200,
881
+ "skipped": 0,
882
+ "errors": 0,
883
+ "started_at": "2026-06-17T18:29:49"
884
+ }
885
+ }
886
+ ```
887
+
888
+ **Returns (idle):** `{"status": "success", "reindex": {"active": false}}`
766
889
 
767
890
  ---
768
891
 
@@ -1071,7 +1194,7 @@ For `.md` files, chunking splits at `##` and `###` header boundaries first. Sect
1071
1194
  |-------|---------|-------------|
1072
1195
  | `models.embedding.model` | `BAAI/bge-small-en-v1.5` | Embedding model (ONNX, runs locally) |
1073
1196
  | `models.embedding.dimensions` | 384 | Vector dimensions (must match model) |
1074
- | `models.embedding.gpu` | false | Enable CUDA GPU acceleration. Requires `pip install knowledge-rag[gpu]` |
1197
+ | `models.embedding.gpu` | false | Enable CUDA GPU acceleration. See [GPU Acceleration](#gpu-acceleration) for full setup |
1075
1198
  | `models.reranker.enabled` | true | Enable cross-encoder reranking |
1076
1199
  | `models.reranker.model` | `Xenova/ms-marco-MiniLM-L-6-v2` | Reranker model |
1077
1200
  | `models.reranker.top_k_multiplier` | 3 | Fetch N*multiplier candidates for reranking |
@@ -1324,6 +1447,64 @@ Common issues:
1324
1447
 
1325
1448
  ## Changelog
1326
1449
 
1450
+ ### Unreleased
1451
+
1452
+ ### v4.3.1 (2026-06-22) — Hybrid Search Fixes
1453
+
1454
+ - **FIX**: Accept `"general"` as a valid category in `search_knowledge`. The parser hardcodes `"general"` as the fallback in `_detect_category` (`ingestion.py`), but the validator only built `valid_categories` from `config.keyword_routes` + `config.category_mappings.values()` — so users who customized `config.yaml` and dropped the default `"general": "general"` mapping hit `Invalid category` even though the index contained `general` documents. Validator now always tolerates `"general"`. (#98, thanks @Hohlas)
1455
+ - **FIX**: Skip BM25-only search results when Chroma can no longer resolve the chunk ID. Stale BM25 indices (typically right after `remove_document` or in the window between async reindex and BM25 rebuild) returned hits whose `collection.get()` came back empty; the previous fallback inserted entries with `document=""` / `metadata={}` into the reranker, polluting results with empty matches. The pipeline now `continue`s past those, dropping the stale hit cleanly. (#98, thanks @Hohlas)
1456
+ - **TEST**: Added `tests/test_pr98_regression.py` (4 tests) pinning both contracts so future refactors cannot silently revert either fix. Test count baseline: 227 → 231. (#99)
1457
+ - **CI**: Bumped `[tool.mypy] python_version` from 3.11 to 3.12 to accept PEP 695 `type` statements in the numpy stub (`numpy/__init__.pyi`) which were breaking the Pillar 7 strict gate. Only affects static analysis; `requires-python = ">=3.11"` unchanged. (#100)
1458
+
1459
+ ### v4.3.0 (2026-06-17) — Async Reindex, GPU CUDA 12, 13th MCP Tool
1460
+
1461
+ - **NEW**: `get_reindex_status` MCP tool — lightweight reindex progress polling without computing full index stats. Returns active/idle status, percent, processed/total, errors, and last result.
1462
+ - **NEW**: `reindex_documents` now runs in background via daemon thread — returns immediately with `{"status": "started"}`. Eliminates MCP timeout on large document sets (5K+ files). Concurrent calls return `already_running` with current progress.
1463
+ - **NEW**: GPU acceleration with full CUDA 12 support — `onnxruntime-gpu` + 7 NVIDIA pip packages (`cublas`, `cudnn`, `cuda-runtime`, `cufft`, `cusparse`, `cusolver`, `curand`, `nvjitlink`). Server auto-detects GPU on startup with 4-step verification (providers, DLLs, nvidia-smi, session creation). Falls back to CPU gracefully.
1464
+ - **NEW**: `_setup_cuda_dll_paths()` adds NVIDIA pip package DLL directories to `PATH` automatically on Windows — onnxruntime finds CUDA 12 DLLs without a full CUDA Toolkit install.
1465
+ - **DEPS**: `[gpu]` extra expanded from 3 to 8 packages (added `cufft`, `cusparse`, `cusolver`, `curand`, `nvjitlink`).
1466
+ - **FIX**: GPU status reporting now uses actual ONNX session creation test instead of just checking `get_available_providers()` — prevents false "GPU ACTIVE" when CUDA DLLs are missing.
1467
+ - **DOCS**: GPU Acceleration section rewritten with complete requirements table, setup steps, verification instructions, and fallback behavior.
1468
+ - **DOCS**: Tool reference updated — `reindex_documents` async behavior documented, `get_reindex_status` reference added.
1469
+ - **TEST**: Backwards-compat baseline updated for 13 MCP tools.
1470
+
1471
+ ### v4.2.0 (2026-06-17) — Search Performance & Output Quality
1472
+
1473
+ - **PERF**: Custom inverted-index BM25 replaces `rank-bm25` full-corpus scan — 128× faster keyword search on 50K+ chunk corpora. Only documents containing query terms are scored via posting lists.
1474
+ - **PERF**: `numpy.argpartition` for O(n) top-k selection instead of O(n log n) sort.
1475
+ - **PERF**: Batched adjacent chunk fetch — single ChromaDB `collection.get()` call replaces N round-trips per result.
1476
+ - **PERF**: O(1) reverse lookup via `_source_to_docid` dict eliminates linear scans of `_indexed_docs` in `search_similar`, `update_document`, `remove_document`, and `_expand_with_adjacent_chunks`.
1477
+ - **NEW**: `snippet_mode` parameter on `search_knowledge` (default: `true`) — truncates content to ~500 chars at natural break points with `content_length` field. Reduces token consumption by ~72%.
1478
+ - **NEW**: `min_score` parameter on `search_knowledge` (default: `0.0`) — filters results below a normalized relevance threshold. Response includes `filtered_by_score` count.
1479
+ - **NEW**: `filtered_by_score` field in search response JSON for transparency.
1480
+ - **DEPS**: `numpy` added as direct dependency (was transitive via fastembed); `rank-bm25` import removed from server.py.
1481
+ - **TEST**: 6 new tests for `min_score` filtering and `snippet_mode` truncation.
1482
+ - **TEST**: Updated backwards-compat baseline to include new `search_knowledge` parameters.
1483
+
1484
+ ### v4.1.2 (2026-06-17)
1485
+
1486
+ - **FIX**: `_save_metadata` dict snapshot prevents concurrent modification crash during file watcher events.
1487
+ - **STYLE**: ruff format applied to server.py.
1488
+
1489
+ ### v4.1.1 (2026-06-17)
1490
+
1491
+ - **FIX**: All `_indexed_docs` iterations now use `list()` snapshot, preventing `dictionary changed size during iteration` crash when FileWatcher modifies the index concurrently with MCP tool calls (affects `search_knowledge`, `search_similar`, `update_document`, `remove_document`, `evaluate_retrieval`, `list_categories`, `list_documents`)
1492
+
1493
+ ### v4.1.0 (2026-06-17)
1494
+
1495
+ - **Added:** `query_expansion_groups` config for symmetric synonym expansion (#92)
1496
+ - **Improved:** `expand_query()` now returns deterministic expansion order (set → ordered list with dedup)
1497
+
1498
+ ### v4.0.1 (2026-06-16)
1499
+
1500
+ - **FIX**: Orphan cleanup now runs before indexing loop, preventing chunk loss when files are moved (#90).
1501
+ - **FIX**: Chunk deduplication is now per-document instead of global, preventing cross-document chunk deletion (#91).
1502
+ - **FIX**: Added `on_moved` handler to `DocumentWatcher` for proper file move detection.
1503
+ - **FIX**: Startup preflight probes ChromaDB in a child process and moves crashing persistent indexes to `data/backups/auto-repair-*` before MCP initialization.
1504
+ - **FIX**: Reranker load failures now fall back to RRF ordering instead of failing `search_knowledge` on offline machines.
1505
+ - **FIX**: Virtualenv project-root detection now handles Python symlinks that resolve to the system interpreter.
1506
+ - **NEW**: `knowledge-rag-guarded` console script kept as an explicit guarded startup alias.
1507
+
1327
1508
  ### v4.0.0 (2026-06-09) — Enterprise Concurrent Access
1328
1509
 
1329
1510
  - **NEW**: SSE and streamable-http transport modes — 1 server serves N clients (`server.transport: "sse"` in config.yaml or `--transport sse` CLI).
@@ -1331,7 +1512,7 @@ Common issues:
1331
1512
  - **NEW**: ChromaDB WAL mode enabled automatically in SSE/HTTP mode for concurrent read performance.
1332
1513
  - **NEW**: Optional rate limiting — sliding-window counter, configurable RPM and burst, disabled by default.
1333
1514
  - **NEW**: Optional Prometheus metrics endpoint — tool call counts, latency histograms, separate port, disabled by default.
1334
- - **NEW**: All 12 MCP tools instrumented with `@rate_limited` and `@instrument` decorators (zero-cost when disabled).
1515
+ - **NEW**: All 13 MCP tools instrumented with `@rate_limited` and `@instrument` decorators (zero-cost when disabled).
1335
1516
  - **NEW**: `--transport` CLI override for Docker/systemd deployments.
1336
1517
  - **NEW**: `pip install knowledge-rag[server]` optional dependency for SSE/HTTP (uvicorn).
1337
1518
  - **CHANGED**: SSE/HTTP mode auto-enables single-instance lock (port collision prevention).
@@ -1359,7 +1540,7 @@ Common issues:
1359
1540
  - **NEW** Property-based fuzzing of all parsers via Hypothesis (`tests/test_ingestion_property.py`) — 200 random examples per CI run.
1360
1541
  - **NEW** Memory baseline regression tests (`tests/test_memory_baseline.py`, cross-platform via psutil) — RSS bounded under 1000 queries; nightly soak amplifies to 50K iterations.
1361
1542
  - **NEW** Property/locale/format/preset matrices (`tests/test_presets.py`, `tests/test_locale.py`, `tests/test_format_smoke.py`).
1362
- - **NEW** Backwards-compatibility regression tests (`tests/test_backwards_compat.py`) — legacy YAML configs from v3.6.0 / v3.7.0 still parse; all 12 MCP tool parameter names frozen.
1543
+ - **NEW** Backwards-compatibility regression tests (`tests/test_backwards_compat.py`) — legacy YAML configs from v3.6.0 / v3.7.0 still parse; all 13 MCP tool parameter names frozen.
1363
1544
  - **NEW** AST-based public API surface diff (`scripts/check_api_surface.py`) — any breaking change blocks merge, baseline at `.github/api-surface-baseline.json`.
1364
1545
  - **NEW** CHANGELOG enforcement (`scripts/check_changelog.py`) — user-facing PRs must add a bullet under `## Unreleased`; bypass via `skip-changelog` label.
1365
1546
  - **NEW** Test count anti-regression (`scripts/check_test_count.py`) — guards against silent test deletion.
@@ -1389,45 +1570,6 @@ Common issues:
1389
1570
  - **CHORE**: pytest `tmp_path_retention_count=1` to avoid Windows atexit cleanup race in CI.
1390
1571
  - **ROADMAP**: Tracked v4.0 shared-service architecture (one daemon, many thin MCP clients) as the long-term fix for multi-process resource duplication. (#34)
1391
1572
 
1392
- ### Unreleased
1393
-
1394
- ### v4.2.0 (2026-06-17) — Search Performance & Output Quality
1395
-
1396
- - **PERF**: Custom inverted-index BM25 replaces `rank-bm25` full-corpus scan — 128× faster keyword search on 50K+ chunk corpora. Only documents containing query terms are scored via posting lists.
1397
- - **PERF**: `numpy.argpartition` for O(n) top-k selection instead of O(n log n) sort.
1398
- - **PERF**: Batched adjacent chunk fetch — single ChromaDB `collection.get()` call replaces N round-trips per result.
1399
- - **PERF**: O(1) reverse lookup via `_source_to_docid` dict eliminates linear scans of `_indexed_docs` in `search_similar`, `update_document`, `remove_document`, and `_expand_with_adjacent_chunks`.
1400
- - **NEW**: `snippet_mode` parameter on `search_knowledge` (default: `true`) — truncates content to ~500 chars at natural break points with `content_length` field. Reduces token consumption by ~72%.
1401
- - **NEW**: `min_score` parameter on `search_knowledge` (default: `0.0`) — filters results below a normalized relevance threshold. Response includes `filtered_by_score` count.
1402
- - **NEW**: `filtered_by_score` field in search response JSON for transparency.
1403
- - **DEPS**: `numpy` added as direct dependency (was transitive via fastembed); `rank-bm25` import removed from server.py.
1404
- - **TEST**: 6 new tests for `min_score` filtering and `snippet_mode` truncation.
1405
- - **TEST**: Updated backwards-compat baseline to include new `search_knowledge` parameters.
1406
-
1407
- ### v4.1.2 (2026-06-17)
1408
-
1409
- - **FIX**: `_save_metadata` dict snapshot prevents concurrent modification crash during file watcher events.
1410
- - **STYLE**: ruff format applied to server.py.
1411
-
1412
- ### v4.1.1 (2026-06-17)
1413
-
1414
- - **FIX**: All `_indexed_docs` iterations now use `list()` snapshot, preventing `dictionary changed size during iteration` crash when FileWatcher modifies the index concurrently with MCP tool calls (affects `search_knowledge`, `search_similar`, `update_document`, `remove_document`, `evaluate_retrieval`, `list_categories`, `list_documents`)
1415
-
1416
- ### v4.1.0 (2026-06-17)
1417
-
1418
- - **Added:** `query_expansion_groups` config for symmetric synonym expansion (#92)
1419
- - **Improved:** `expand_query()` now returns deterministic expansion order (set → ordered list with dedup)
1420
-
1421
- ### v4.0.1 (2026-06-16)
1422
-
1423
- - **FIX**: Orphan cleanup now runs before indexing loop, preventing chunk loss when files are moved (#90).
1424
- - **FIX**: Chunk deduplication is now per-document instead of global, preventing cross-document chunk deletion (#91).
1425
- - **FIX**: Added `on_moved` handler to `DocumentWatcher` for proper file move detection.
1426
- - **FIX**: Startup preflight probes ChromaDB in a child process and moves crashing persistent indexes to `data/backups/auto-repair-*` before MCP initialization.
1427
- - **FIX**: Reranker load failures now fall back to RRF ordering instead of failing `search_knowledge` on offline machines.
1428
- - **FIX**: Virtualenv project-root detection now handles Python symlinks that resolve to the system interpreter.
1429
- - **NEW**: `knowledge-rag-guarded` console script kept as an explicit guarded startup alias.
1430
-
1431
1573
  ### v3.6.2 (2026-04-23)
1432
1574
 
1433
1575
  - **INFRA**: NPM provenance attestation (SLSA supply chain security), full README on npm page
@@ -1,43 +1,3 @@
1
- Metadata-Version: 2.4
2
- Name: knowledge-rag
3
- Version: 4.2.0
4
- Summary: Local RAG System for Claude Code — Hybrid search + Cross-encoder Reranking + 12 MCP Tools + 20 Format Parsers. Zero external servers.
5
- Project-URL: Homepage, https://github.com/lyonzin/knowledge-rag
6
- Project-URL: Repository, https://github.com/lyonzin/knowledge-rag
7
- Project-URL: Issues, https://github.com/lyonzin/knowledge-rag/issues
8
- Project-URL: Changelog, https://github.com/lyonzin/knowledge-rag/releases
9
- Author-email: "Lyon." <lyonzin@users.noreply.github.com>
10
- License: MIT
11
- License-File: LICENSE
12
- Keywords: bm25,chromadb,claude-code,embeddings,fastembed,hybrid-search,knowledge-base,local-ai,mcp,rag,reranking,retrieval-augmented-generation,semantic-search
13
- Classifier: Development Status :: 4 - Beta
14
- Classifier: Intended Audience :: Developers
15
- Classifier: Intended Audience :: Science/Research
16
- Classifier: License :: OSI Approved :: MIT License
17
- Classifier: Operating System :: OS Independent
18
- Classifier: Programming Language :: Python :: 3.11
19
- Classifier: Programming Language :: Python :: 3.12
20
- Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
- Classifier: Topic :: Text Processing :: Indexing
22
- Requires-Python: >=3.11
23
- Requires-Dist: beautifulsoup4>=4.12.0
24
- Requires-Dist: chromadb>=1.4.0
25
- Requires-Dist: fastembed[reranking]>=0.4.0
26
- Requires-Dist: mcp>=1.6.0
27
- Requires-Dist: numpy>=1.24.0
28
- Requires-Dist: openpyxl>=3.1.0
29
- Requires-Dist: pymupdf>=1.23.0
30
- Requires-Dist: python-docx>=1.0.0
31
- Requires-Dist: python-pptx>=1.0.0
32
- Requires-Dist: pyyaml>=6.0
33
- Requires-Dist: requests>=2.33.0
34
- Requires-Dist: watchdog>=4.0.0
35
- Provides-Extra: gpu
36
- Requires-Dist: onnxruntime-gpu>=1.14.0; extra == 'gpu'
37
- Provides-Extra: server
38
- Requires-Dist: uvicorn>=0.20.0; extra == 'server'
39
- Description-Content-Type: text/markdown
40
-
41
1
  # Knowledge RAG
42
2
 
43
3
  <div align="center">
@@ -57,7 +17,7 @@ Description-Content-Type: text/markdown
57
17
  ### Your docs, your machine, zero cloud. Claude Code searches them natively.
58
18
 
59
19
  Drop your PDFs, markdown, code, notebooks — **1800+ files, 39K chunks, indexed in under 3 minutes.**<br/>
60
- Hybrid search (BM25 + semantic vectors + cross-encoder reranking) through 12 MCP tools.<br/>
20
+ Hybrid search (BM25 + semantic vectors + cross-encoder reranking) through 13 MCP tools.<br/>
61
21
  Everything runs locally via ONNX. No Docker, no Ollama, no API keys, no data leaves your machine.
62
22
 
63
23
  ```
@@ -66,7 +26,7 @@ pip install knowledge-rag → restart Claude Code → search_knowledge("your que
66
26
 
67
27
  ---
68
28
 
69
- **12 MCP Tools** | **Hybrid Search + Reranking** | **20 File Formats** | **Optional NVIDIA GPU** | **100% Local**
29
+ **13 MCP Tools** | **Hybrid Search + Reranking** | **20 File Formats** | **Optional NVIDIA GPU** | **100% Local**
70
30
 
71
31
  [What's New](#whats-new-in-v420) | [Supported Formats](#supported-formats) | [Installation](#installation) | [Configuration](#configuration) | [API Reference](#api-reference) | [Architecture](#architecture)
72
32
 
@@ -121,7 +81,7 @@ Or via CLI: `knowledge-rag --transport sse`
121
81
  - **Prometheus metrics**: `/metrics` endpoint on separate port
122
82
  - **Bearer auth**: Token validation for SSE/HTTP connections
123
83
 
124
- All 12 MCP tools are instrumented with `@rate_limited` and `@instrument` decorators — zero overhead when features are disabled. Default transport remains **stdio** for full backwards compatibility.
84
+ All 13 MCP tools are instrumented with `@rate_limited` and `@instrument` decorators — zero overhead when features are disabled. Default transport remains **stdio** for full backwards compatibility.
125
85
 
126
86
  > **Migration**: Existing users need zero changes. SSE mode is opt-in via `server.transport: "sse"` in config.yaml. See [Configuration](#configuration) for details.
127
87
 
@@ -245,7 +205,7 @@ See [Changelog](#changelog) for full history.
245
205
  | **MMR Diversification** | Maximal Marginal Relevance reduces redundant results |
246
206
  | **Persistent Model Cache** | Embedding models cached in `models_cache/` — survives reboots |
247
207
  | **Auto-Migration** | Detects embedding dimension mismatch and rebuilds automatically |
248
- | **12 MCP Tools** | Full CRUD + search + evaluation via Claude Code |
208
+ | **13 MCP Tools** | Full CRUD + search + evaluation via Claude Code |
249
209
 
250
210
  ---
251
211
 
@@ -257,7 +217,7 @@ See [Changelog](#changelog) for full history.
257
217
  flowchart TB
258
218
  subgraph MCP["MCP SERVER (FastMCP)"]
259
219
  direction TB
260
- TOOLS["12 MCP Tools<br/>search | get | add | update | remove<br/>reindex | list | stats | url | similar | evaluate"]
220
+ TOOLS["13 MCP Tools<br/>search | get | add | update | remove<br/>reindex | reindex_status | list | stats | url | similar | evaluate"]
261
221
  end
262
222
 
263
223
  subgraph SEARCH["HYBRID SEARCH ENGINE"]
@@ -432,7 +392,53 @@ flowchart LR
432
392
  - Claude Code CLI
433
393
  - *…or any other MCP client (Claude Desktop, Cursor, VS Code, Antigravity, opencode, Windsurf) — see [Use with other MCP clients](#use-with-other-mcp-clients)*
434
394
  - ~200MB disk for model cache (auto-downloaded on first run)
435
- - *Optional:* NVIDIA GPU + CUDA for accelerated embeddings (`pip install knowledge-rag[gpu]` + `models.embedding.gpu: true` in config)
395
+ - *Optional:* NVIDIA GPU + CUDA 12 for accelerated embeddings (see [GPU Acceleration](#gpu-acceleration) below)
396
+
397
+ ### GPU Acceleration
398
+
399
+ GPU mode accelerates embedding generation during indexing and search. It requires an NVIDIA GPU with CUDA 12 support. No GPU? No problem — the server runs on CPU by default and GPU is entirely optional.
400
+
401
+ **Requirements:**
402
+
403
+ | Component | Minimum | How to check / get it |
404
+ |-----------|---------|----------------------|
405
+ | NVIDIA GPU (Turing+) | RTX 20xx / 30xx / 40xx / 50xx, or Tesla T4+ | `nvidia-smi` |
406
+ | NVIDIA Driver | ≥ 525 | `nvidia-smi` — [nvidia.com/drivers](https://www.nvidia.com/drivers) |
407
+ | CUDA 12 runtime | Provided by pip packages below | Automatic |
408
+
409
+ **Setup (2 steps):**
410
+
411
+ ```bash
412
+ # 1. Install GPU dependencies (onnxruntime-gpu + all CUDA 12 runtime DLLs)
413
+ pip install knowledge-rag[gpu]
414
+
415
+ # 2. Enable in config.yaml
416
+ # models:
417
+ # embedding:
418
+ # gpu: true
419
+ ```
420
+
421
+ The `[gpu]` extra installs `onnxruntime-gpu` plus 7 NVIDIA CUDA 12 packages (`cublas`, `cudnn`, `cuda-runtime`, `cufft`, `cusparse`, `cusolver`, `curand`, `nvjitlink`) so you don't need a full CUDA Toolkit install.
422
+
423
+ **Verify GPU is active:**
424
+
425
+ On server startup, look for the GPU status banner:
426
+ ```
427
+ ============================================================
428
+ GPU STATUS: ACTIVE
429
+ Provider: CUDAExecutionProvider
430
+ Device: NVIDIA GeForce RTX 3080 Ti
431
+ VRAM: 12.0 GB
432
+ ============================================================
433
+ ```
434
+
435
+ Or programmatically:
436
+ ```bash
437
+ python -c "import onnxruntime; print(onnxruntime.get_available_providers())"
438
+ # Should include: 'CUDAExecutionProvider'
439
+ ```
440
+
441
+ > **Fallback**: If CUDA is unavailable at runtime (wrong driver, missing DLLs, no GPU), the server falls back to CPU automatically with a `[WARN]` log — it never crashes. The `gpu: true` config is a preference, not a requirement.
436
442
 
437
443
  ### Install Methods
438
444
 
@@ -703,7 +709,7 @@ search_knowledge("lateral movement strategies", hybrid_alpha=1.0)
703
709
 
704
710
  ### Indexing
705
711
 
706
- Documents are automatically indexed on first startup. To manage the index:
712
+ Documents are automatically indexed on first startup. All reindex operations run **in background** — they return immediately and you poll progress via `get_reindex_status()`:
707
713
 
708
714
  ```python
709
715
  # Incremental: only re-index changed files (fast)
@@ -714,6 +720,10 @@ reindex_documents(force=True)
714
720
 
715
721
  # Nuclear rebuild: delete everything, re-embed all (use after model change)
716
722
  reindex_documents(full_rebuild=True)
723
+
724
+ # Poll progress (lightweight, no full stats computation)
725
+ get_reindex_status()
726
+ # → {"reindex": {"active": true, "percent": 56, "progress": "2090/3734", ...}}
717
727
  ```
718
728
 
719
729
  ### Evaluating Retrieval Quality
@@ -795,14 +805,39 @@ Retrieve the full content of a specific document.
795
805
 
796
806
  #### `reindex_documents`
797
807
 
798
- Index or reindex all documents in the knowledge base.
808
+ Index or reindex all documents in the knowledge base. **Runs in background** — returns immediately. Poll progress via `get_reindex_status()`.
799
809
 
800
810
  | Parameter | Type | Default | Description |
801
811
  |-----------|------|---------|-------------|
802
812
  | `force` | bool | false | Smart reindex: detects changes, rebuilds BM25. Fast. |
803
813
  | `full_rebuild` | bool | false | Nuclear rebuild: deletes everything, re-embeds all documents. Use after model change. |
804
814
 
805
- **Returns:** JSON with indexing statistics (indexed, updated, skipped, deleted, chunks_added, chunks_removed, dedup_skipped, elapsed_seconds).
815
+ **Returns:** `{"status": "started", "operation": "..."}` immediately. If already running, returns `{"status": "already_running", "progress": "1200/3734"}`.
816
+
817
+ ---
818
+
819
+ #### `get_reindex_status`
820
+
821
+ Get the current status of a background reindex operation. Lightweight — does not compute full index statistics.
822
+
823
+ **Returns (active):**
824
+ ```json
825
+ {
826
+ "status": "success",
827
+ "reindex": {
828
+ "active": true,
829
+ "operation": "nuclear_rebuild",
830
+ "progress": "1200/3734",
831
+ "percent": 32,
832
+ "indexed": 1200,
833
+ "skipped": 0,
834
+ "errors": 0,
835
+ "started_at": "2026-06-17T18:29:49"
836
+ }
837
+ }
838
+ ```
839
+
840
+ **Returns (idle):** `{"status": "success", "reindex": {"active": false}}`
806
841
 
807
842
  ---
808
843
 
@@ -1111,7 +1146,7 @@ For `.md` files, chunking splits at `##` and `###` header boundaries first. Sect
1111
1146
  |-------|---------|-------------|
1112
1147
  | `models.embedding.model` | `BAAI/bge-small-en-v1.5` | Embedding model (ONNX, runs locally) |
1113
1148
  | `models.embedding.dimensions` | 384 | Vector dimensions (must match model) |
1114
- | `models.embedding.gpu` | false | Enable CUDA GPU acceleration. Requires `pip install knowledge-rag[gpu]` |
1149
+ | `models.embedding.gpu` | false | Enable CUDA GPU acceleration. See [GPU Acceleration](#gpu-acceleration) for full setup |
1115
1150
  | `models.reranker.enabled` | true | Enable cross-encoder reranking |
1116
1151
  | `models.reranker.model` | `Xenova/ms-marco-MiniLM-L-6-v2` | Reranker model |
1117
1152
  | `models.reranker.top_k_multiplier` | 3 | Fetch N*multiplier candidates for reranking |
@@ -1364,6 +1399,64 @@ Common issues:
1364
1399
 
1365
1400
  ## Changelog
1366
1401
 
1402
+ ### Unreleased
1403
+
1404
+ ### v4.3.1 (2026-06-22) — Hybrid Search Fixes
1405
+
1406
+ - **FIX**: Accept `"general"` as a valid category in `search_knowledge`. The parser hardcodes `"general"` as the fallback in `_detect_category` (`ingestion.py`), but the validator only built `valid_categories` from `config.keyword_routes` + `config.category_mappings.values()` — so users who customized `config.yaml` and dropped the default `"general": "general"` mapping hit `Invalid category` even though the index contained `general` documents. Validator now always tolerates `"general"`. (#98, thanks @Hohlas)
1407
+ - **FIX**: Skip BM25-only search results when Chroma can no longer resolve the chunk ID. Stale BM25 indices (typically right after `remove_document` or in the window between async reindex and BM25 rebuild) returned hits whose `collection.get()` came back empty; the previous fallback inserted entries with `document=""` / `metadata={}` into the reranker, polluting results with empty matches. The pipeline now `continue`s past those, dropping the stale hit cleanly. (#98, thanks @Hohlas)
1408
+ - **TEST**: Added `tests/test_pr98_regression.py` (4 tests) pinning both contracts so future refactors cannot silently revert either fix. Test count baseline: 227 → 231. (#99)
1409
+ - **CI**: Bumped `[tool.mypy] python_version` from 3.11 to 3.12 to accept PEP 695 `type` statements in the numpy stub (`numpy/__init__.pyi`) which were breaking the Pillar 7 strict gate. Only affects static analysis; `requires-python = ">=3.11"` unchanged. (#100)
1410
+
1411
+ ### v4.3.0 (2026-06-17) — Async Reindex, GPU CUDA 12, 13th MCP Tool
1412
+
1413
+ - **NEW**: `get_reindex_status` MCP tool — lightweight reindex progress polling without computing full index stats. Returns active/idle status, percent, processed/total, errors, and last result.
1414
+ - **NEW**: `reindex_documents` now runs in background via daemon thread — returns immediately with `{"status": "started"}`. Eliminates MCP timeout on large document sets (5K+ files). Concurrent calls return `already_running` with current progress.
1415
+ - **NEW**: GPU acceleration with full CUDA 12 support — `onnxruntime-gpu` + 7 NVIDIA pip packages (`cublas`, `cudnn`, `cuda-runtime`, `cufft`, `cusparse`, `cusolver`, `curand`, `nvjitlink`). Server auto-detects GPU on startup with 4-step verification (providers, DLLs, nvidia-smi, session creation). Falls back to CPU gracefully.
1416
+ - **NEW**: `_setup_cuda_dll_paths()` adds NVIDIA pip package DLL directories to `PATH` automatically on Windows — onnxruntime finds CUDA 12 DLLs without a full CUDA Toolkit install.
1417
+ - **DEPS**: `[gpu]` extra expanded from 3 to 8 packages (added `cufft`, `cusparse`, `cusolver`, `curand`, `nvjitlink`).
1418
+ - **FIX**: GPU status reporting now uses actual ONNX session creation test instead of just checking `get_available_providers()` — prevents false "GPU ACTIVE" when CUDA DLLs are missing.
1419
+ - **DOCS**: GPU Acceleration section rewritten with complete requirements table, setup steps, verification instructions, and fallback behavior.
1420
+ - **DOCS**: Tool reference updated — `reindex_documents` async behavior documented, `get_reindex_status` reference added.
1421
+ - **TEST**: Backwards-compat baseline updated for 13 MCP tools.
1422
+
1423
+ ### v4.2.0 (2026-06-17) — Search Performance & Output Quality
1424
+
1425
+ - **PERF**: Custom inverted-index BM25 replaces `rank-bm25` full-corpus scan — 128× faster keyword search on 50K+ chunk corpora. Only documents containing query terms are scored via posting lists.
1426
+ - **PERF**: `numpy.argpartition` for O(n) top-k selection instead of O(n log n) sort.
1427
+ - **PERF**: Batched adjacent chunk fetch — single ChromaDB `collection.get()` call replaces N round-trips per result.
1428
+ - **PERF**: O(1) reverse lookup via `_source_to_docid` dict eliminates linear scans of `_indexed_docs` in `search_similar`, `update_document`, `remove_document`, and `_expand_with_adjacent_chunks`.
1429
+ - **NEW**: `snippet_mode` parameter on `search_knowledge` (default: `true`) — truncates content to ~500 chars at natural break points with `content_length` field. Reduces token consumption by ~72%.
1430
+ - **NEW**: `min_score` parameter on `search_knowledge` (default: `0.0`) — filters results below a normalized relevance threshold. Response includes `filtered_by_score` count.
1431
+ - **NEW**: `filtered_by_score` field in search response JSON for transparency.
1432
+ - **DEPS**: `numpy` added as direct dependency (was transitive via fastembed); `rank-bm25` import removed from server.py.
1433
+ - **TEST**: 6 new tests for `min_score` filtering and `snippet_mode` truncation.
1434
+ - **TEST**: Updated backwards-compat baseline to include new `search_knowledge` parameters.
1435
+
1436
+ ### v4.1.2 (2026-06-17)
1437
+
1438
+ - **FIX**: `_save_metadata` dict snapshot prevents concurrent modification crash during file watcher events.
1439
+ - **STYLE**: ruff format applied to server.py.
1440
+
1441
+ ### v4.1.1 (2026-06-17)
1442
+
1443
+ - **FIX**: All `_indexed_docs` iterations now use `list()` snapshot, preventing `dictionary changed size during iteration` crash when FileWatcher modifies the index concurrently with MCP tool calls (affects `search_knowledge`, `search_similar`, `update_document`, `remove_document`, `evaluate_retrieval`, `list_categories`, `list_documents`)
1444
+
1445
+ ### v4.1.0 (2026-06-17)
1446
+
1447
+ - **Added:** `query_expansion_groups` config for symmetric synonym expansion (#92)
1448
+ - **Improved:** `expand_query()` now returns deterministic expansion order (set → ordered list with dedup)
1449
+
1450
+ ### v4.0.1 (2026-06-16)
1451
+
1452
+ - **FIX**: Orphan cleanup now runs before indexing loop, preventing chunk loss when files are moved (#90).
1453
+ - **FIX**: Chunk deduplication is now per-document instead of global, preventing cross-document chunk deletion (#91).
1454
+ - **FIX**: Added `on_moved` handler to `DocumentWatcher` for proper file move detection.
1455
+ - **FIX**: Startup preflight probes ChromaDB in a child process and moves crashing persistent indexes to `data/backups/auto-repair-*` before MCP initialization.
1456
+ - **FIX**: Reranker load failures now fall back to RRF ordering instead of failing `search_knowledge` on offline machines.
1457
+ - **FIX**: Virtualenv project-root detection now handles Python symlinks that resolve to the system interpreter.
1458
+ - **NEW**: `knowledge-rag-guarded` console script kept as an explicit guarded startup alias.
1459
+
1367
1460
  ### v4.0.0 (2026-06-09) — Enterprise Concurrent Access
1368
1461
 
1369
1462
  - **NEW**: SSE and streamable-http transport modes — 1 server serves N clients (`server.transport: "sse"` in config.yaml or `--transport sse` CLI).
@@ -1371,7 +1464,7 @@ Common issues:
1371
1464
  - **NEW**: ChromaDB WAL mode enabled automatically in SSE/HTTP mode for concurrent read performance.
1372
1465
  - **NEW**: Optional rate limiting — sliding-window counter, configurable RPM and burst, disabled by default.
1373
1466
  - **NEW**: Optional Prometheus metrics endpoint — tool call counts, latency histograms, separate port, disabled by default.
1374
- - **NEW**: All 12 MCP tools instrumented with `@rate_limited` and `@instrument` decorators (zero-cost when disabled).
1467
+ - **NEW**: All 13 MCP tools instrumented with `@rate_limited` and `@instrument` decorators (zero-cost when disabled).
1375
1468
  - **NEW**: `--transport` CLI override for Docker/systemd deployments.
1376
1469
  - **NEW**: `pip install knowledge-rag[server]` optional dependency for SSE/HTTP (uvicorn).
1377
1470
  - **CHANGED**: SSE/HTTP mode auto-enables single-instance lock (port collision prevention).
@@ -1399,7 +1492,7 @@ Common issues:
1399
1492
  - **NEW** Property-based fuzzing of all parsers via Hypothesis (`tests/test_ingestion_property.py`) — 200 random examples per CI run.
1400
1493
  - **NEW** Memory baseline regression tests (`tests/test_memory_baseline.py`, cross-platform via psutil) — RSS bounded under 1000 queries; nightly soak amplifies to 50K iterations.
1401
1494
  - **NEW** Property/locale/format/preset matrices (`tests/test_presets.py`, `tests/test_locale.py`, `tests/test_format_smoke.py`).
1402
- - **NEW** Backwards-compatibility regression tests (`tests/test_backwards_compat.py`) — legacy YAML configs from v3.6.0 / v3.7.0 still parse; all 12 MCP tool parameter names frozen.
1495
+ - **NEW** Backwards-compatibility regression tests (`tests/test_backwards_compat.py`) — legacy YAML configs from v3.6.0 / v3.7.0 still parse; all 13 MCP tool parameter names frozen.
1403
1496
  - **NEW** AST-based public API surface diff (`scripts/check_api_surface.py`) — any breaking change blocks merge, baseline at `.github/api-surface-baseline.json`.
1404
1497
  - **NEW** CHANGELOG enforcement (`scripts/check_changelog.py`) — user-facing PRs must add a bullet under `## Unreleased`; bypass via `skip-changelog` label.
1405
1498
  - **NEW** Test count anti-regression (`scripts/check_test_count.py`) — guards against silent test deletion.
@@ -1429,45 +1522,6 @@ Common issues:
1429
1522
  - **CHORE**: pytest `tmp_path_retention_count=1` to avoid Windows atexit cleanup race in CI.
1430
1523
  - **ROADMAP**: Tracked v4.0 shared-service architecture (one daemon, many thin MCP clients) as the long-term fix for multi-process resource duplication. (#34)
1431
1524
 
1432
- ### Unreleased
1433
-
1434
- ### v4.2.0 (2026-06-17) — Search Performance & Output Quality
1435
-
1436
- - **PERF**: Custom inverted-index BM25 replaces `rank-bm25` full-corpus scan — 128× faster keyword search on 50K+ chunk corpora. Only documents containing query terms are scored via posting lists.
1437
- - **PERF**: `numpy.argpartition` for O(n) top-k selection instead of O(n log n) sort.
1438
- - **PERF**: Batched adjacent chunk fetch — single ChromaDB `collection.get()` call replaces N round-trips per result.
1439
- - **PERF**: O(1) reverse lookup via `_source_to_docid` dict eliminates linear scans of `_indexed_docs` in `search_similar`, `update_document`, `remove_document`, and `_expand_with_adjacent_chunks`.
1440
- - **NEW**: `snippet_mode` parameter on `search_knowledge` (default: `true`) — truncates content to ~500 chars at natural break points with `content_length` field. Reduces token consumption by ~72%.
1441
- - **NEW**: `min_score` parameter on `search_knowledge` (default: `0.0`) — filters results below a normalized relevance threshold. Response includes `filtered_by_score` count.
1442
- - **NEW**: `filtered_by_score` field in search response JSON for transparency.
1443
- - **DEPS**: `numpy` added as direct dependency (was transitive via fastembed); `rank-bm25` import removed from server.py.
1444
- - **TEST**: 6 new tests for `min_score` filtering and `snippet_mode` truncation.
1445
- - **TEST**: Updated backwards-compat baseline to include new `search_knowledge` parameters.
1446
-
1447
- ### v4.1.2 (2026-06-17)
1448
-
1449
- - **FIX**: `_save_metadata` dict snapshot prevents concurrent modification crash during file watcher events.
1450
- - **STYLE**: ruff format applied to server.py.
1451
-
1452
- ### v4.1.1 (2026-06-17)
1453
-
1454
- - **FIX**: All `_indexed_docs` iterations now use `list()` snapshot, preventing `dictionary changed size during iteration` crash when FileWatcher modifies the index concurrently with MCP tool calls (affects `search_knowledge`, `search_similar`, `update_document`, `remove_document`, `evaluate_retrieval`, `list_categories`, `list_documents`)
1455
-
1456
- ### v4.1.0 (2026-06-17)
1457
-
1458
- - **Added:** `query_expansion_groups` config for symmetric synonym expansion (#92)
1459
- - **Improved:** `expand_query()` now returns deterministic expansion order (set → ordered list with dedup)
1460
-
1461
- ### v4.0.1 (2026-06-16)
1462
-
1463
- - **FIX**: Orphan cleanup now runs before indexing loop, preventing chunk loss when files are moved (#90).
1464
- - **FIX**: Chunk deduplication is now per-document instead of global, preventing cross-document chunk deletion (#91).
1465
- - **FIX**: Added `on_moved` handler to `DocumentWatcher` for proper file move detection.
1466
- - **FIX**: Startup preflight probes ChromaDB in a child process and moves crashing persistent indexes to `data/backups/auto-repair-*` before MCP initialization.
1467
- - **FIX**: Reranker load failures now fall back to RRF ordering instead of failing `search_knowledge` on offline machines.
1468
- - **FIX**: Virtualenv project-root detection now handles Python symlinks that resolve to the system interpreter.
1469
- - **NEW**: `knowledge-rag-guarded` console script kept as an explicit guarded startup alias.
1470
-
1471
1525
  ### v3.6.2 (2026-04-23)
1472
1526
 
1473
1527
  - **INFRA**: NPM provenance attestation (SLSA supply chain security), full README on npm page
@@ -8,7 +8,7 @@ import sys # noqa: I001
8
8
  _original_stdout = sys.stdout
9
9
  sys.stdout = sys.stderr
10
10
 
11
- __version__ = "4.2.0"
11
+ __version__ = "4.3.1"
12
12
  __author__ = "Ailton Rocha (Lyon.)"
13
13
 
14
14
  from .config import Config # noqa: E402
@@ -986,6 +986,9 @@ class KnowledgeOrchestrator:
986
986
  # Migration: deferred — checked in main() after full init
987
987
  self._needs_rebuild = False
988
988
 
989
+ # Background reindex progress (polled via get_index_stats)
990
+ self._reindex_progress: Dict[str, Any] = {"active": False}
991
+
989
992
  def _safe_get_collection(self):
990
993
  """
991
994
  Get or create ChromaDB collection with auto-recovery.
@@ -1134,6 +1137,7 @@ class KnowledgeOrchestrator:
1134
1137
 
1135
1138
  documents = self.parser.parse_directory()
1136
1139
  stats["total_files"] = len(documents)
1140
+ self._reindex_progress["total_files"] = stats["total_files"]
1137
1141
  if stats["total_files"] > 100:
1138
1142
  print(f"[INDEX] Scanning {stats['total_files']} documents...")
1139
1143
 
@@ -1226,6 +1230,15 @@ class KnowledgeOrchestrator:
1226
1230
  stats["errors"] += 1
1227
1231
  print(f"[ERROR] Failed to index {doc.source}: {e}")
1228
1232
 
1233
+ self._reindex_progress.update(
1234
+ {
1235
+ "processed": idx + 1,
1236
+ "indexed": stats["indexed"],
1237
+ "skipped": stats["skipped"],
1238
+ "errors": stats["errors"],
1239
+ }
1240
+ )
1241
+
1229
1242
  if stats["total_files"] > 100 and (idx + 1) % _progress_interval == 0:
1230
1243
  pct = int((idx + 1) / stats["total_files"] * 100)
1231
1244
  print(
@@ -1308,6 +1321,43 @@ class KnowledgeOrchestrator:
1308
1321
 
1309
1322
  return 0
1310
1323
 
1324
+ def start_reindex_background(self, mode: str) -> Dict[str, Any]:
1325
+ """Start reindex in a background thread. Returns immediately."""
1326
+ if self._reindex_progress.get("active"):
1327
+ return {"status": "already_running", "progress": dict(self._reindex_progress)}
1328
+
1329
+ self._reindex_progress = {
1330
+ "active": True,
1331
+ "operation": mode,
1332
+ "total_files": 0,
1333
+ "processed": 0,
1334
+ "indexed": 0,
1335
+ "skipped": 0,
1336
+ "errors": 0,
1337
+ "started_at": datetime.now().isoformat(),
1338
+ }
1339
+
1340
+ target = {
1341
+ "incremental": lambda: self.index_all(force=False),
1342
+ "smart_reindex": self.reindex_all,
1343
+ "nuclear_rebuild": self.nuclear_rebuild,
1344
+ }[mode]
1345
+
1346
+ thread = threading.Thread(target=self._run_reindex, args=(target,), daemon=True)
1347
+ thread.start()
1348
+ return {"status": "started", "operation": mode}
1349
+
1350
+ def _run_reindex(self, target: Any) -> None:
1351
+ """Background thread runner for reindex operations."""
1352
+ try:
1353
+ result = target()
1354
+ self._reindex_progress["result"] = result
1355
+ except Exception as e:
1356
+ self._reindex_progress["error"] = str(e)
1357
+ print(f"[ERROR] Background reindex failed: {e}")
1358
+ finally:
1359
+ self._reindex_progress["active"] = False
1360
+
1311
1361
  def reindex_all(self) -> Dict[str, Any]:
1312
1362
  """Smart reindex: incremental detection + BM25 rebuild + orphan cleanup."""
1313
1363
  import shutil
@@ -1493,9 +1543,16 @@ class KnowledgeOrchestrator:
1493
1543
  else:
1494
1544
  try:
1495
1545
  fetched = self.collection.get(ids=[chunk_id], include=["documents", "metadatas"])
1546
+ if (
1547
+ not fetched["documents"]
1548
+ or not fetched["metadatas"]
1549
+ or not fetched["documents"][0]
1550
+ or not fetched["metadatas"][0]
1551
+ ):
1552
+ continue
1496
1553
  data = {
1497
- "document": fetched["documents"][0] if fetched["documents"] else "",
1498
- "metadata": fetched["metadatas"][0] if fetched["metadatas"] else {},
1554
+ "document": fetched["documents"][0],
1555
+ "metadata": fetched["metadatas"][0],
1499
1556
  "distance": 0,
1500
1557
  }
1501
1558
  except Exception:
@@ -2043,8 +2100,8 @@ class KnowledgeOrchestrator:
2043
2100
  return docs
2044
2101
 
2045
2102
  def get_stats(self) -> Dict[str, Any]:
2046
- """Get index statistics"""
2047
- return {
2103
+ """Get index statistics including background reindex progress."""
2104
+ stats = {
2048
2105
  "total_documents": len(self._indexed_docs),
2049
2106
  "total_chunks": self.collection.count(),
2050
2107
  "categories": self.list_categories(),
@@ -2057,6 +2114,48 @@ class KnowledgeOrchestrator:
2057
2114
  "query_cache": self.query_cache.stats(),
2058
2115
  }
2059
2116
 
2117
+ progress = self._reindex_progress
2118
+ if progress.get("active"):
2119
+ total = max(1, progress.get("total_files", 1))
2120
+ processed = progress.get("processed", 0)
2121
+ stats["reindex"] = {
2122
+ "active": True,
2123
+ "operation": progress.get("operation"),
2124
+ "progress": f"{processed}/{progress.get('total_files', 0)}",
2125
+ "percent": round(processed / total * 100),
2126
+ "indexed": progress.get("indexed", 0),
2127
+ "errors": progress.get("errors", 0),
2128
+ "started_at": progress.get("started_at"),
2129
+ }
2130
+ else:
2131
+ stats["reindex"] = {"active": False}
2132
+
2133
+ return stats
2134
+
2135
+ def get_reindex_status(self) -> Dict[str, Any]:
2136
+ """Get background reindex progress without computing full index stats."""
2137
+ progress = self._reindex_progress
2138
+ if progress.get("active"):
2139
+ total = max(1, progress.get("total_files", 1))
2140
+ processed = progress.get("processed", 0)
2141
+ return {
2142
+ "active": True,
2143
+ "operation": progress.get("operation"),
2144
+ "progress": f"{processed}/{progress.get('total_files', 0)}",
2145
+ "percent": round(processed / total * 100),
2146
+ "indexed": progress.get("indexed", 0),
2147
+ "skipped": progress.get("skipped", 0),
2148
+ "errors": progress.get("errors", 0),
2149
+ "started_at": progress.get("started_at"),
2150
+ }
2151
+
2152
+ result: Dict[str, Any] = {"active": False}
2153
+ if "result" in progress:
2154
+ result["last_result"] = progress["result"]
2155
+ if "error" in progress:
2156
+ result["last_error"] = progress["error"]
2157
+ return result
2158
+
2060
2159
  def _load_metadata(self) -> Dict[str, Dict]:
2061
2160
  """Load index metadata from disk"""
2062
2161
  if self._metadata_file.exists():
@@ -2181,7 +2280,7 @@ def search_knowledge(
2181
2280
  hybrid_alpha = max(0.0, min(hybrid_alpha if hybrid_alpha is not None else 0.3, 1.0))
2182
2281
  min_score = max(0.0, min(min_score if min_score is not None else 0.0, 1.0))
2183
2282
 
2184
- valid_categories = list(config.keyword_routes.keys()) + list(set(config.category_mappings.values()))
2283
+ valid_categories = list(config.keyword_routes.keys()) + list(set(config.category_mappings.values())) + ["general"]
2185
2284
  if category and category not in valid_categories:
2186
2285
  return json.dumps(
2187
2286
  {"status": "error", "message": f"Invalid category '{category}'. Valid: {', '.join(valid_categories)}"}
@@ -2258,16 +2357,17 @@ def reindex_documents(force: bool = False, full_rebuild: bool = False) -> str:
2258
2357
  """
2259
2358
  Index or reindex all documents in the knowledge base.
2260
2359
 
2261
- Mutating — modifies the vector index. CPU/IO intensive for full_rebuild (~6 min for 200 docs).
2360
+ Runs in background — returns immediately. Use get_reindex_status() to monitor progress.
2262
2361
 
2263
2362
  Args:
2264
- force: If True, smart reindex (detects changed files + rebuilds BM25 index). Fast (~5s
2265
- for 200 docs). Use after manually editing files on disk outside of add_document().
2363
+ force: If True, smart reindex (detects changed files + rebuilds BM25 index).
2364
+ Use after manually editing files on disk outside of add_document().
2266
2365
  full_rebuild: If True, nuclear rebuild — deletes all vectors and re-embeds everything
2267
2366
  from scratch. Use only if the embedding model changed or the index is corrupted.
2268
2367
 
2269
2368
  Returns:
2270
- JSON string with indexing statistics (docs processed, added, skipped, errors).
2369
+ JSON string with operation status. Poll get_reindex_status() for reindex.active,
2370
+ reindex.progress, and reindex.percent until reindex.active becomes false.
2271
2371
 
2272
2372
  Usage: Normal workflow does not require this — add_document(), update_document(), and
2273
2373
  add_from_url() all auto-index on call. Use force=True only after direct filesystem edits.
@@ -2277,16 +2377,59 @@ def reindex_documents(force: bool = False, full_rebuild: bool = False) -> str:
2277
2377
  orchestrator = get_orchestrator()
2278
2378
 
2279
2379
  if full_rebuild:
2280
- stats = orchestrator.nuclear_rebuild()
2281
- operation = "nuclear_rebuild"
2380
+ mode = "nuclear_rebuild"
2282
2381
  elif force:
2283
- stats = orchestrator.reindex_all()
2284
- operation = "smart_reindex"
2382
+ mode = "smart_reindex"
2285
2383
  else:
2286
- stats = orchestrator.index_all()
2287
- operation = "incremental_index"
2384
+ mode = "incremental"
2385
+
2386
+ result = orchestrator.start_reindex_background(mode)
2387
+
2388
+ if result["status"] == "already_running":
2389
+ progress = result["progress"]
2390
+ return json.dumps(
2391
+ {
2392
+ "status": "already_running",
2393
+ "progress": f"{progress.get('processed', 0)}/{progress.get('total_files', 0)}",
2394
+ "operation": progress.get("operation"),
2395
+ "hint": "Use get_reindex_status() to check progress",
2396
+ },
2397
+ indent=2,
2398
+ ensure_ascii=False,
2399
+ )
2400
+
2401
+ return json.dumps(
2402
+ {
2403
+ "status": "started",
2404
+ "operation": mode,
2405
+ "message": "Reindex running in background. Use get_reindex_status() to monitor progress.",
2406
+ },
2407
+ indent=2,
2408
+ ensure_ascii=False,
2409
+ )
2410
+
2288
2411
 
2289
- return json.dumps({"status": "success", "operation": operation, "stats": stats}, indent=2, ensure_ascii=False)
2412
+ @mcp.tool()
2413
+ @rate_limited
2414
+ @instrument("get_reindex_status")
2415
+ def get_reindex_status() -> str:
2416
+ """
2417
+ Get the current status of a background reindex operation.
2418
+
2419
+ Lightweight — does not compute full index statistics. Use this to poll progress
2420
+ after calling reindex_documents().
2421
+
2422
+ Returns:
2423
+ JSON string with reindex status. When active: operation name, progress (processed/total),
2424
+ percent complete, indexed/skipped/errors counts, and start time. When inactive: active=false,
2425
+ plus last_result or last_error from the most recent completed reindex.
2426
+
2427
+ Usage: Call repeatedly after reindex_documents() to monitor progress. When reindex.active
2428
+ becomes false, the operation is complete. Use get_index_stats() for full index health metrics.
2429
+ """
2430
+ orchestrator = get_orchestrator()
2431
+ status = orchestrator.get_reindex_status()
2432
+ return json.dumps({"status": "success", "reindex": status}, indent=2)
2290
2433
 
2291
2434
 
2292
2435
  @mcp.tool()
@@ -4,8 +4,8 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "knowledge-rag"
7
- version = "4.2.0"
8
- description = "Local RAG System for Claude Code — Hybrid search + Cross-encoder Reranking + 12 MCP Tools + 20 Format Parsers. Zero external servers."
7
+ version = "4.3.1"
8
+ description = "Local RAG System for Claude Code — Hybrid search + Cross-encoder Reranking + 13 MCP Tools + 20 Format Parsers. Zero external servers."
9
9
  readme = "README.md"
10
10
  license = {text = "MIT"}
11
11
  requires-python = ">=3.11"
@@ -44,7 +44,17 @@ dependencies = [
44
44
  ]
45
45
 
46
46
  [project.optional-dependencies]
47
- gpu = ["onnxruntime-gpu>=1.14.0"]
47
+ gpu = [
48
+ "onnxruntime-gpu>=1.14.0",
49
+ "nvidia-cublas-cu12",
50
+ "nvidia-cudnn-cu12",
51
+ "nvidia-cuda-runtime-cu12",
52
+ "nvidia-cufft-cu12",
53
+ "nvidia-cusparse-cu12",
54
+ "nvidia-cusolver-cu12",
55
+ "nvidia-curand-cu12",
56
+ "nvidia-nvjitlink-cu12",
57
+ ]
48
58
  server = ["uvicorn>=0.20.0"]
49
59
 
50
60
  [project.urls]
@@ -131,7 +141,11 @@ fail_under = 35
131
141
  # we incrementally annotate the legacy modules. The CI job runs strict on the
132
142
  # allowlist below; new modules are added as they earn full annotations.
133
143
  [tool.mypy]
134
- python_version = "3.11"
144
+ # Pinned to 3.12 to match the CI runtime (Python 3.12) and to accept PEP 695
145
+ # ``type`` statements that recent third-party stubs (e.g. numpy/__init__.pyi)
146
+ # emit. The package itself still supports Python 3.11+ at runtime — this
147
+ # setting only governs the static-analysis target.
148
+ python_version = "3.12"
135
149
  strict = true
136
150
  show_error_codes = true
137
151
  warn_unused_configs = true
File without changes
File without changes