multilingual-rag-mcp 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,20 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ test:
10
+ runs-on: ubuntu-latest
11
+ strategy:
12
+ matrix:
13
+ python-version: ["3.11", "3.12", "3.13"]
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: actions/setup-python@v5
17
+ with:
18
+ python-version: ${{ matrix.python-version }}
19
+ - run: pip install -e ".[test]"
20
+ - run: pytest
@@ -0,0 +1,21 @@
1
+ name: Publish to PyPI
2
+
3
+ on:
4
+ push:
5
+ tags: ["v*"]
6
+
7
+ permissions:
8
+ id-token: write
9
+
10
+ jobs:
11
+ publish:
12
+ runs-on: ubuntu-latest
13
+ environment: pypi
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: actions/setup-python@v5
17
+ with:
18
+ python-version: "3.12"
19
+ - run: pip install build
20
+ - run: python -m build
21
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,8 @@
1
+ __pycache__/
2
+ *.pyc
3
+ *.egg-info/
4
+ dist/
5
+ build/
6
+ .venv/
7
+ *.egg
8
+ .DS_Store
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Aliaksandr Kazarez
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,132 @@
1
+ Metadata-Version: 2.5
2
+ Name: multilingual-rag-mcp
3
+ Version: 0.1.0
4
+ Summary: Multilingual RAG MCP server — cross-lingual search over local documents
5
+ Project-URL: Repository, https://github.com/aliaksandr-kazarez/multilingual-rag-mcp
6
+ Project-URL: Issues, https://github.com/aliaksandr-kazarez/multilingual-rag-mcp/issues
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Keywords: cross-lingual,mcp,multilingual,rag,semantic-search
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: License :: OSI Approved :: MIT License
12
+ Classifier: Programming Language :: Python :: 3.11
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Topic :: Text Processing :: Indexing
16
+ Requires-Python: >=3.11
17
+ Requires-Dist: chromadb>=1.0
18
+ Requires-Dist: fastembed>=0.4
19
+ Requires-Dist: mcp<3.0,>=2.0
20
+ Requires-Dist: pymupdf>=1.23
21
+ Provides-Extra: test
22
+ Requires-Dist: pytest>=8.0; extra == 'test'
23
+ Description-Content-Type: text/markdown
24
+
25
+ # multilingual-rag-mcp
26
+
27
+ Multilingual RAG MCP server for local document search. Query in one language, find content in another.
28
+
29
+ Built for the common case where you talk to AI agents in English but your documents are in Russian (or any other language). The multilingual embedding model maps semantically similar concepts across 50+ languages to the same vector space — no translation step needed.
30
+
31
+ ## Install
32
+
33
+ ### pip from PyPI
34
+
35
+ ```bash
36
+ pip install multilingual-rag-mcp
37
+ ```
38
+
39
+ ### One-liner with uvx (no install needed)
40
+
41
+ ```bash
42
+ uvx --from multilingual-rag-mcp rag-mcp index ./docs/
43
+ ```
44
+
45
+ ### pip from GitHub (latest)
46
+
47
+ ```bash
48
+ pip install git+https://github.com/aliaksandr-kazarez/multilingual-rag-mcp.git
49
+ ```
50
+
51
+ ## Quick start
52
+
53
+ ### 1. Index your documents
54
+
55
+ ```bash
56
+ rag-mcp index ~/documents/
57
+ ```
58
+
59
+ ### 2. Add to Claude Code
60
+
61
+ ```bash
62
+ claude mcp add rag -- uvx --from multilingual-rag-mcp rag-mcp
63
+ ```
64
+
65
+ Set the document paths via env vars:
66
+
67
+ ```bash
68
+ claude mcp add rag \
69
+ -e RAG_DOCS=$HOME/documents \
70
+ -- uvx --from multilingual-rag-mcp rag-mcp
71
+ ```
72
+
73
+ If installed locally (pip install), use the simpler form:
74
+
75
+ ```bash
76
+ claude mcp add rag -e RAG_DOCS=$HOME/documents -- rag-mcp
77
+ ```
78
+
79
+ ### 3. Search
80
+
81
+ From Claude Code, the `search` tool handles cross-lingual queries automatically:
82
+
83
+ - "search for protein recommendations" finds Russian articles about белок
84
+ - "найди рецепты" finds recipe content regardless of language
85
+
86
+ ## MCP tools
87
+
88
+ | Tool | Description |
89
+ |------|-------------|
90
+ | `search(query, n=5)` | Semantic search across all indexed documents |
91
+ | `get_document(path)` | Retrieve full document content |
92
+ | `list_documents(filter?)` | List indexed documents, optionally filtered |
93
+ | `reindex()` | Re-index all configured document directories |
94
+ | `stats()` | Index statistics (documents, chunks, categories) |
95
+
96
+ ## Configuration
97
+
98
+ | Env var | Default | Description |
99
+ |---------|---------|-------------|
100
+ | `RAG_DOCS` | — | Comma-separated paths to document directories |
101
+ | `RAG_DATA` | `~/.local/share/rag-mcp/` | Index storage location |
102
+ | `RAG_MODEL` | `sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2` | Embedding model (any fastembed-compatible model) |
103
+ | `RAG_CHUNK_SIZE` | `1000` | Maximum chunk size in characters |
104
+ | `RAG_CHUNK_OVERLAP` | `200` | Overlap between chunks |
105
+
106
+ ## Supported formats
107
+
108
+ Markdown (`.md`), plain text (`.txt`), PDF (`.pdf`), JSON (`.json`), CSV (`.csv`).
109
+
110
+ Markdown files with YAML frontmatter have their metadata (title, date, etc.) extracted automatically.
111
+
112
+ ## CLI
113
+
114
+ ```
115
+ rag-mcp index <dir> [<dir> ...] Index documents
116
+ rag-mcp stats Print index statistics
117
+ rag-mcp Start MCP server (stdio)
118
+ rag-mcp --help Show help
119
+ rag-mcp --version Show version
120
+ ```
121
+
122
+ ## How it works
123
+
124
+ 1. Documents are parsed and split into chunks (markdown-aware: respects headers, code blocks)
125
+ 2. Each chunk is embedded with a multilingual model via [fastembed](https://github.com/qdrant/fastembed) (ONNX, no PyTorch needed)
126
+ 3. Embeddings are stored in [ChromaDB](https://www.trychroma.com/) (local, file-based)
127
+ 4. Queries are embedded with the same model and matched by cosine similarity
128
+ 5. Cross-lingual retrieval works because the model places semantically similar text from different languages near each other in vector space
129
+
130
+ ## License
131
+
132
+ MIT
@@ -0,0 +1,108 @@
1
+ # multilingual-rag-mcp
2
+
3
+ Multilingual RAG MCP server for local document search. Query in one language, find content in another.
4
+
5
+ Built for the common case where you talk to AI agents in English but your documents are in Russian (or any other language). The multilingual embedding model maps semantically similar concepts across 50+ languages to the same vector space — no translation step needed.
6
+
7
+ ## Install
8
+
9
+ ### pip from PyPI
10
+
11
+ ```bash
12
+ pip install multilingual-rag-mcp
13
+ ```
14
+
15
+ ### One-liner with uvx (no install needed)
16
+
17
+ ```bash
18
+ uvx --from multilingual-rag-mcp rag-mcp index ./docs/
19
+ ```
20
+
21
+ ### pip from GitHub (latest)
22
+
23
+ ```bash
24
+ pip install git+https://github.com/aliaksandr-kazarez/multilingual-rag-mcp.git
25
+ ```
26
+
27
+ ## Quick start
28
+
29
+ ### 1. Index your documents
30
+
31
+ ```bash
32
+ rag-mcp index ~/documents/
33
+ ```
34
+
35
+ ### 2. Add to Claude Code
36
+
37
+ ```bash
38
+ claude mcp add rag -- uvx --from multilingual-rag-mcp rag-mcp
39
+ ```
40
+
41
+ Set the document paths via env vars:
42
+
43
+ ```bash
44
+ claude mcp add rag \
45
+ -e RAG_DOCS=$HOME/documents \
46
+ -- uvx --from multilingual-rag-mcp rag-mcp
47
+ ```
48
+
49
+ If installed locally (pip install), use the simpler form:
50
+
51
+ ```bash
52
+ claude mcp add rag -e RAG_DOCS=$HOME/documents -- rag-mcp
53
+ ```
54
+
55
+ ### 3. Search
56
+
57
+ From Claude Code, the `search` tool handles cross-lingual queries automatically:
58
+
59
+ - "search for protein recommendations" finds Russian articles about белок
60
+ - "найди рецепты" finds recipe content regardless of language
61
+
62
+ ## MCP tools
63
+
64
+ | Tool | Description |
65
+ |------|-------------|
66
+ | `search(query, n=5)` | Semantic search across all indexed documents |
67
+ | `get_document(path)` | Retrieve full document content |
68
+ | `list_documents(filter?)` | List indexed documents, optionally filtered |
69
+ | `reindex()` | Re-index all configured document directories |
70
+ | `stats()` | Index statistics (documents, chunks, categories) |
71
+
72
+ ## Configuration
73
+
74
+ | Env var | Default | Description |
75
+ |---------|---------|-------------|
76
+ | `RAG_DOCS` | — | Comma-separated paths to document directories |
77
+ | `RAG_DATA` | `~/.local/share/rag-mcp/` | Index storage location |
78
+ | `RAG_MODEL` | `sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2` | Embedding model (any fastembed-compatible model) |
79
+ | `RAG_CHUNK_SIZE` | `1000` | Maximum chunk size in characters |
80
+ | `RAG_CHUNK_OVERLAP` | `200` | Overlap between chunks |
81
+
82
+ ## Supported formats
83
+
84
+ Markdown (`.md`), plain text (`.txt`), PDF (`.pdf`), JSON (`.json`), CSV (`.csv`).
85
+
86
+ Markdown files with YAML frontmatter have their metadata (title, date, etc.) extracted automatically.
87
+
88
+ ## CLI
89
+
90
+ ```
91
+ rag-mcp index <dir> [<dir> ...] Index documents
92
+ rag-mcp stats Print index statistics
93
+ rag-mcp Start MCP server (stdio)
94
+ rag-mcp --help Show help
95
+ rag-mcp --version Show version
96
+ ```
97
+
98
+ ## How it works
99
+
100
+ 1. Documents are parsed and split into chunks (markdown-aware: respects headers, code blocks)
101
+ 2. Each chunk is embedded with a multilingual model via [fastembed](https://github.com/qdrant/fastembed) (ONNX, no PyTorch needed)
102
+ 3. Embeddings are stored in [ChromaDB](https://www.trychroma.com/) (local, file-based)
103
+ 4. Queries are embedded with the same model and matched by cosine similarity
104
+ 5. Cross-lingual retrieval works because the model places semantically similar text from different languages near each other in vector space
105
+
106
+ ## License
107
+
108
+ MIT
@@ -0,0 +1,3 @@
1
+ """Multilingual RAG MCP server — cross-lingual search over local documents."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,36 @@
1
+ import sys
2
+
3
+
4
+ def main():
5
+ if len(sys.argv) > 1 and sys.argv[1] in ("-h", "--help", "help"):
6
+ print(
7
+ "Usage: rag-mcp [command]\n"
8
+ "\n"
9
+ "Commands:\n"
10
+ " index <dir> [<dir> ...] Index documents from directories\n"
11
+ " stats Print index statistics\n"
12
+ " (none) Start MCP server (stdio)\n"
13
+ "\n"
14
+ "Environment:\n"
15
+ " RAG_DOCS Comma-separated document directories\n"
16
+ " RAG_DATA Index storage (default: ~/.local/share/rag-mcp/)\n"
17
+ " RAG_MODEL Embedding model (default: paraphrase-multilingual-MiniLM-L12-v2)\n"
18
+ )
19
+ elif len(sys.argv) > 1 and sys.argv[1] in ("-V", "--version"):
20
+ from multilingual_rag_mcp import __version__
21
+ print(f"rag-mcp {__version__}")
22
+ elif len(sys.argv) > 1 and sys.argv[1] == "index":
23
+ from multilingual_rag_mcp.ingest import index_cli
24
+ index_cli(sys.argv[2:])
25
+ elif len(sys.argv) > 1 and sys.argv[1] == "stats":
26
+ from multilingual_rag_mcp.store import VectorStore
27
+ import json
28
+ store = VectorStore()
29
+ print(json.dumps(store.stats(), indent=2, ensure_ascii=False))
30
+ else:
31
+ from multilingual_rag_mcp.server import mcp
32
+ mcp.run()
33
+
34
+
35
+ if __name__ == "__main__":
36
+ main()
@@ -0,0 +1,201 @@
1
+ """Document parsing, chunking, and indexing."""
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ import re
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ SUPPORTED = {".md", ".txt", ".pdf", ".json", ".csv"}
10
+
11
+
12
+ def parse_file(path: Path) -> tuple[str, dict]:
13
+ """Parse a file → (text, metadata). Strips markdown frontmatter into metadata."""
14
+ meta: dict = {"source": str(path), "filename": path.name}
15
+ suffix = path.suffix.lower()
16
+
17
+ if suffix == ".pdf":
18
+ import pymupdf
19
+
20
+ doc = pymupdf.open(str(path))
21
+ text = "\n\n".join(page.get_text() for page in doc)
22
+ meta["title"] = doc.metadata.get("title") or path.stem
23
+ doc.close()
24
+ return text, meta
25
+
26
+ raw = path.read_text(encoding="utf-8", errors="replace")
27
+
28
+ if suffix == ".json":
29
+ try:
30
+ data = json.loads(raw)
31
+ text = json.dumps(data, indent=2, ensure_ascii=False)
32
+ except json.JSONDecodeError:
33
+ text = raw
34
+ meta["title"] = path.stem
35
+ return text, meta
36
+
37
+ if suffix == ".md":
38
+ text, fm = _strip_frontmatter(raw)
39
+ meta.update(fm)
40
+ if "title" not in meta:
41
+ meta["title"] = _first_heading(text) or path.stem
42
+ else:
43
+ text = raw
44
+ meta["title"] = path.stem
45
+
46
+ return text, meta
47
+
48
+
49
+ def _strip_frontmatter(text: str) -> tuple[str, dict]:
50
+ """Remove YAML frontmatter and return (body, parsed fields)."""
51
+ if not text.startswith("---"):
52
+ return text, {}
53
+ end = text.find("\n---", 3)
54
+ if end < 0:
55
+ return text, {}
56
+ fm_block = text[3:end]
57
+ body = text[end + 4 :].strip()
58
+ fields: dict = {}
59
+ for line in fm_block.splitlines():
60
+ if ":" in line:
61
+ key, val = line.split(":", 1)
62
+ fields[key.strip()] = val.strip().strip("\"'")
63
+ return body, fields
64
+
65
+
66
+ def _first_heading(text: str) -> str | None:
67
+ for line in text.splitlines():
68
+ if line.startswith("# "):
69
+ return line[2:].strip()
70
+ return None
71
+
72
+
73
+ def chunk_markdown(text: str, max_size: int = 1000, overlap: int = 200) -> list[str]:
74
+ """Split markdown into chunks, respecting headers and code blocks."""
75
+ sections = re.split(r"(?=^#{1,3} )", text, flags=re.MULTILINE)
76
+ chunks: list[str] = []
77
+ for section in sections:
78
+ section = section.strip()
79
+ if not section:
80
+ continue
81
+ if len(section) <= max_size:
82
+ if len(section) > 30:
83
+ chunks.append(section)
84
+ else:
85
+ chunks.extend(_split_paragraphs(section, max_size, overlap))
86
+ return chunks
87
+
88
+
89
+ def _split_paragraphs(
90
+ text: str, max_size: int = 1000, overlap: int = 200
91
+ ) -> list[str]:
92
+ paragraphs = re.split(r"\n\s*\n", text)
93
+ chunks: list[str] = []
94
+ current = ""
95
+ for para in paragraphs:
96
+ para = para.strip()
97
+ if not para:
98
+ continue
99
+ if len(current) + len(para) + 2 > max_size and current:
100
+ chunks.append(current.strip())
101
+ tail = current[-overlap:].strip() if overlap else ""
102
+ current = (tail + "\n\n" + para) if tail else para
103
+ else:
104
+ current = (current + "\n\n" + para) if current else para
105
+ if current.strip() and len(current.strip()) > 30:
106
+ chunks.append(current.strip())
107
+ return chunks
108
+
109
+
110
+ def doc_id_from_path(path: Path, base_dir: Path) -> str:
111
+ try:
112
+ return str(path.relative_to(base_dir))
113
+ except ValueError:
114
+ return str(path)
115
+
116
+
117
+ def ingest_directory(
118
+ store,
119
+ docs_dir: Path,
120
+ *,
121
+ category: str = "",
122
+ chunk_size: int = 1000,
123
+ chunk_overlap: int = 200,
124
+ ) -> dict:
125
+ """Index all supported files under docs_dir."""
126
+ stats = {"indexed": 0, "chunks": 0, "skipped": 0, "errors": []}
127
+
128
+ for path in sorted(docs_dir.rglob("*")):
129
+ if not path.is_file():
130
+ continue
131
+ if path.suffix.lower() not in SUPPORTED:
132
+ continue
133
+ if any(part.startswith(".") for part in path.relative_to(docs_dir).parts):
134
+ continue
135
+
136
+ try:
137
+ text, meta = parse_file(path)
138
+ if not text or len(text.strip()) < 50:
139
+ stats["skipped"] += 1
140
+ continue
141
+
142
+ did = doc_id_from_path(path, docs_dir)
143
+ meta["doc_id"] = did
144
+ meta["category"] = category or docs_dir.name
145
+
146
+ chunks = chunk_markdown(text, chunk_size, chunk_overlap)
147
+ if not chunks:
148
+ stats["skipped"] += 1
149
+ continue
150
+
151
+ metadatas = [{**meta, "chunk_index": i} for i in range(len(chunks))]
152
+ store.add_chunks(did, chunks, metadatas)
153
+ stats["indexed"] += 1
154
+ stats["chunks"] += len(chunks)
155
+ except Exception as e:
156
+ stats["errors"].append(f"{path.name}: {e}")
157
+
158
+ return stats
159
+
160
+
161
+ def index_cli(args: list[str]) -> None:
162
+ """CLI entry: `rag-mcp index /path/to/docs [/more/docs ...]`"""
163
+ import os
164
+
165
+ dirs = args or [
166
+ p.strip()
167
+ for p in os.environ.get("RAG_DOCS", "").split(",")
168
+ if p.strip()
169
+ ]
170
+ if not dirs:
171
+ print("Usage: rag-mcp index <docs_dir> [<docs_dir> ...]", file=sys.stderr)
172
+ print(" or: RAG_DOCS=/path/to/docs rag-mcp index", file=sys.stderr)
173
+ sys.exit(1)
174
+
175
+ from multilingual_rag_mcp.store import VectorStore
176
+
177
+ store = VectorStore()
178
+ chunk_size = int(os.environ.get("RAG_CHUNK_SIZE", "1000"))
179
+ chunk_overlap = int(os.environ.get("RAG_CHUNK_OVERLAP", "200"))
180
+
181
+ for d in dirs:
182
+ p = Path(d).expanduser().resolve()
183
+ if not p.is_dir():
184
+ print(f"SKIP {p}: not a directory", file=sys.stderr)
185
+ continue
186
+ print(f"Indexing {p} ...", file=sys.stderr)
187
+ s = ingest_directory(store, p, chunk_size=chunk_size, chunk_overlap=chunk_overlap)
188
+ print(
189
+ f" {s['indexed']} docs, {s['chunks']} chunks, "
190
+ f"{s['skipped']} skipped, {len(s['errors'])} errors",
191
+ file=sys.stderr,
192
+ )
193
+ for err in s["errors"][:10]:
194
+ print(f" ERROR: {err}", file=sys.stderr)
195
+
196
+ total = store.stats()
197
+ print(
198
+ f"\nDone. {total['total_documents']} documents, "
199
+ f"{total['total_chunks']} chunks in index.",
200
+ file=sys.stderr,
201
+ )
@@ -0,0 +1,135 @@
1
+ """MCP server exposing multilingual RAG search tools."""
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ import os
6
+ from pathlib import Path
7
+
8
+ from mcp.server.fastmcp import FastMCP
9
+
10
+ from multilingual_rag_mcp.ingest import ingest_directory, parse_file
11
+ from multilingual_rag_mcp.store import VectorStore
12
+
13
+ mcp = FastMCP("rag-mcp")
14
+
15
+ _store: VectorStore | None = None
16
+
17
+
18
+ def _get_store() -> VectorStore:
19
+ global _store
20
+ if _store is None:
21
+ _store = VectorStore()
22
+ return _store
23
+
24
+
25
+ def _docs_dirs() -> list[Path]:
26
+ raw = os.environ.get("RAG_DOCS", "")
27
+ return [Path(p.strip()).expanduser().resolve() for p in raw.split(",") if p.strip()]
28
+
29
+
30
+ @mcp.tool()
31
+ def search(query: str, n: int = 5) -> str:
32
+ """Search the knowledge base with a natural-language query.
33
+
34
+ Works across languages: query in English to find Russian content,
35
+ or query in Russian to find English content. The multilingual
36
+ embedding model maps semantically similar concepts across languages.
37
+ """
38
+ store = _get_store()
39
+ results = store.search(query, n)
40
+ if not results:
41
+ return "No results found. If the index is empty, call the reindex tool first."
42
+
43
+ parts = []
44
+ for r in results:
45
+ meta = r["metadata"]
46
+ score = max(0, 1 - r["distance"])
47
+ parts.append(
48
+ f"### {meta.get('title', '?')} (score: {score:.2f})\n"
49
+ f"**Source:** {meta.get('source', '?')} \n"
50
+ f"**Category:** {meta.get('category', '?')}\n\n"
51
+ f"{r['text']}\n"
52
+ )
53
+ return "\n---\n".join(parts)
54
+
55
+
56
+ @mcp.tool()
57
+ def get_document(path: str) -> str:
58
+ """Retrieve the full content of a source document by path.
59
+
60
+ Accepts an absolute path or a path relative to any configured RAG_DOCS directory.
61
+ """
62
+ p = Path(path)
63
+ if not p.exists():
64
+ for d in _docs_dirs():
65
+ candidate = d / path
66
+ if candidate.exists():
67
+ p = candidate
68
+ break
69
+ else:
70
+ return f"Not found: {path}"
71
+
72
+ text, meta = parse_file(p)
73
+ return f"# {meta.get('title', p.name)}\n\n{text}"
74
+
75
+
76
+ @mcp.tool()
77
+ def list_documents(filter: str | None = None) -> str:
78
+ """List indexed documents. Optionally filter by keyword in title or path."""
79
+ docs = _get_store().list_documents()
80
+ if filter:
81
+ fl = filter.lower()
82
+ docs = [
83
+ d
84
+ for d in docs
85
+ if fl in d["title"].lower() or fl in d["source"].lower()
86
+ ]
87
+
88
+ if not docs:
89
+ return "No documents indexed. Call reindex to populate the knowledge base."
90
+
91
+ lines = ["| Title | Category | Chunks |", "|---|---|---|"]
92
+ for d in docs:
93
+ lines.append(f"| {d['title'][:60]} | {d['category'][:20]} | {d['chunks']} |")
94
+ lines.append(f"\n**Total: {len(docs)} documents**")
95
+ return "\n".join(lines)
96
+
97
+
98
+ @mcp.tool()
99
+ def reindex() -> str:
100
+ """Re-index all documents from the configured RAG_DOCS directories."""
101
+ dirs = _docs_dirs()
102
+ if not dirs:
103
+ return (
104
+ "No RAG_DOCS configured. Set the RAG_DOCS environment variable to a "
105
+ "comma-separated list of directories containing your documents."
106
+ )
107
+
108
+ store = _get_store()
109
+ chunk_size = int(os.environ.get("RAG_CHUNK_SIZE", "1000"))
110
+ chunk_overlap = int(os.environ.get("RAG_CHUNK_OVERLAP", "200"))
111
+
112
+ results = []
113
+ for d in dirs:
114
+ if not d.is_dir():
115
+ results.append(f"{d}: not a directory, skipped")
116
+ continue
117
+ s = ingest_directory(store, d, chunk_size=chunk_size, chunk_overlap=chunk_overlap)
118
+ results.append(
119
+ f"{d.name}: {s['indexed']} docs, {s['chunks']} chunks, "
120
+ f"{s['skipped']} skipped, {len(s['errors'])} errors"
121
+ )
122
+ for err in s["errors"][:5]:
123
+ results.append(f" ERROR: {err}")
124
+
125
+ total = store.stats()
126
+ results.append(
127
+ f"\nTotal: {total['total_documents']} documents, {total['total_chunks']} chunks"
128
+ )
129
+ return "\n".join(results)
130
+
131
+
132
+ @mcp.tool()
133
+ def stats() -> str:
134
+ """Get knowledge base statistics: document count, chunk count, categories."""
135
+ return json.dumps(_get_store().stats(), indent=2, ensure_ascii=False)
@@ -0,0 +1,106 @@
1
+ """Vector store backed by ChromaDB with multilingual embeddings."""
2
+ from __future__ import annotations
3
+
4
+ import os
5
+ from pathlib import Path
6
+
7
+ import chromadb
8
+ from chromadb.api.types import Documents, EmbeddingFunction, Embeddings
9
+
10
+ DEFAULT_MODEL = "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2"
11
+ DEFAULT_DATA_DIR = Path.home() / ".local" / "share" / "rag-mcp"
12
+
13
+
14
+ class _FastEmbedEF(EmbeddingFunction[Documents]):
15
+ def __init__(self, model_name: str):
16
+ from fastembed import TextEmbedding
17
+
18
+ self._model = TextEmbedding(model_name=model_name)
19
+
20
+ def __call__(self, input: Documents) -> Embeddings:
21
+ return [e.tolist() for e in self._model.embed(list(input))]
22
+
23
+
24
+ class VectorStore:
25
+ def __init__(
26
+ self,
27
+ data_dir: str | Path | None = None,
28
+ model: str | None = None,
29
+ ):
30
+ data_dir = Path(data_dir or os.environ.get("RAG_DATA", str(DEFAULT_DATA_DIR)))
31
+ data_dir.mkdir(parents=True, exist_ok=True)
32
+ self._model_name = model or os.environ.get("RAG_MODEL", DEFAULT_MODEL)
33
+
34
+ self._ef = _FastEmbedEF(self._model_name)
35
+ self._client = chromadb.PersistentClient(path=str(data_dir / "chroma"))
36
+ self._col = self._client.get_or_create_collection(
37
+ name="documents",
38
+ embedding_function=self._ef,
39
+ metadata={"hnsw:space": "cosine"},
40
+ )
41
+
42
+ def add_chunks(
43
+ self, doc_id: str, chunks: list[str], metadatas: list[dict]
44
+ ) -> None:
45
+ ids = [f"{doc_id}::{i}" for i in range(len(chunks))]
46
+ self._col.upsert(ids=ids, documents=chunks, metadatas=metadatas)
47
+
48
+ def remove_document(self, doc_id: str) -> int:
49
+ existing = self._col.get(where={"doc_id": doc_id})
50
+ if existing["ids"]:
51
+ self._col.delete(ids=existing["ids"])
52
+ return len(existing["ids"])
53
+
54
+ def search(self, query: str, n: int = 5) -> list[dict]:
55
+ if self._col.count() == 0:
56
+ return []
57
+ # Fetch extra candidates to deduplicate by document
58
+ fetch_n = min(n * 3, self._col.count())
59
+ results = self._col.query(query_texts=[query], n_results=fetch_n)
60
+ out = []
61
+ seen_docs: set[str] = set()
62
+ for i in range(len(results["ids"][0])):
63
+ doc_id = results["metadatas"][0][i].get("doc_id", "")
64
+ if doc_id in seen_docs:
65
+ continue
66
+ seen_docs.add(doc_id)
67
+ out.append(
68
+ {
69
+ "id": results["ids"][0][i],
70
+ "text": results["documents"][0][i],
71
+ "metadata": results["metadatas"][0][i],
72
+ "distance": results["distances"][0][i],
73
+ }
74
+ )
75
+ if len(out) >= n:
76
+ break
77
+ return out
78
+
79
+ def list_documents(self) -> list[dict]:
80
+ all_data = self._col.get(include=["metadatas"])
81
+ docs: dict[str, dict] = {}
82
+ for m in all_data["metadatas"]:
83
+ doc_id = m.get("doc_id", "?")
84
+ if doc_id not in docs:
85
+ docs[doc_id] = {
86
+ "doc_id": doc_id,
87
+ "title": m.get("title", ""),
88
+ "source": m.get("source", ""),
89
+ "category": m.get("category", ""),
90
+ "chunks": 0,
91
+ }
92
+ docs[doc_id]["chunks"] += 1
93
+ return sorted(docs.values(), key=lambda d: d["source"])
94
+
95
+ def stats(self) -> dict:
96
+ docs = self.list_documents()
97
+ categories = {}
98
+ for d in docs:
99
+ cat = d.get("category", "uncategorized") or "uncategorized"
100
+ categories[cat] = categories.get(cat, 0) + 1
101
+ return {
102
+ "total_chunks": self._col.count(),
103
+ "total_documents": len(docs),
104
+ "categories": categories,
105
+ "model": self._model_name,
106
+ }
@@ -0,0 +1,36 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "multilingual-rag-mcp"
7
+ version = "0.1.0"
8
+ description = "Multilingual RAG MCP server — cross-lingual search over local documents"
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ requires-python = ">=3.11"
12
+ keywords = ["mcp", "rag", "multilingual", "semantic-search", "cross-lingual"]
13
+ classifiers = [
14
+ "Development Status :: 4 - Beta",
15
+ "License :: OSI Approved :: MIT License",
16
+ "Programming Language :: Python :: 3.11",
17
+ "Programming Language :: Python :: 3.12",
18
+ "Programming Language :: Python :: 3.13",
19
+ "Topic :: Text Processing :: Indexing",
20
+ ]
21
+ dependencies = [
22
+ "mcp>=2.0,<3.0",
23
+ "chromadb>=1.0",
24
+ "fastembed>=0.4",
25
+ "pymupdf>=1.23",
26
+ ]
27
+
28
+ [project.optional-dependencies]
29
+ test = ["pytest>=8.0"]
30
+
31
+ [project.urls]
32
+ Repository = "https://github.com/aliaksandr-kazarez/multilingual-rag-mcp"
33
+ Issues = "https://github.com/aliaksandr-kazarez/multilingual-rag-mcp/issues"
34
+
35
+ [project.scripts]
36
+ rag-mcp = "multilingual_rag_mcp.__main__:main"
File without changes
@@ -0,0 +1,69 @@
1
+ import subprocess
2
+ import sys
3
+
4
+ import pytest
5
+
6
+
7
+ def test_version_import():
8
+ from multilingual_rag_mcp import __version__
9
+ assert __version__ == "0.1.0"
10
+
11
+
12
+ def test_cli_version():
13
+ result = subprocess.run(
14
+ [sys.executable, "-m", "multilingual_rag_mcp", "--version"],
15
+ capture_output=True, text=True,
16
+ )
17
+ assert result.returncode == 0
18
+ assert "rag-mcp 0.1.0" in result.stdout
19
+
20
+
21
+ def test_cli_help():
22
+ result = subprocess.run(
23
+ [sys.executable, "-m", "multilingual_rag_mcp", "--help"],
24
+ capture_output=True, text=True,
25
+ )
26
+ assert result.returncode == 0
27
+ assert "index" in result.stdout
28
+ assert "stats" in result.stdout
29
+
30
+
31
+ def test_store_defaults():
32
+ from multilingual_rag_mcp.store import DEFAULT_MODEL, DEFAULT_DATA_DIR
33
+ assert "multilingual" in DEFAULT_MODEL
34
+ assert "rag-mcp" in str(DEFAULT_DATA_DIR)
35
+
36
+
37
+ def test_ingest_supported_formats():
38
+ from multilingual_rag_mcp.ingest import SUPPORTED
39
+ assert ".md" in SUPPORTED
40
+ assert ".pdf" in SUPPORTED
41
+ assert ".txt" in SUPPORTED
42
+
43
+
44
+ def test_chunk_markdown():
45
+ from multilingual_rag_mcp.ingest import chunk_markdown
46
+ text = (
47
+ "# Title\n\n"
48
+ "This is the first paragraph with enough content to pass the length filter.\n\n"
49
+ "## Section\n\n"
50
+ "This is the second paragraph with enough content to pass the length filter."
51
+ )
52
+ chunks = chunk_markdown(text, max_size=1000, overlap=0)
53
+ assert len(chunks) >= 1
54
+ assert any("first paragraph" in c for c in chunks)
55
+
56
+
57
+ def test_parse_markdown_frontmatter():
58
+ import tempfile
59
+ from pathlib import Path
60
+ from multilingual_rag_mcp.ingest import parse_file
61
+
62
+ content = "---\ntitle: Test\ndate: 2025-01-01\n---\n\n# Hello\n\nBody text."
63
+ with tempfile.NamedTemporaryFile(suffix=".md", mode="w", delete=False) as f:
64
+ f.write(content)
65
+ f.flush()
66
+ text, meta = parse_file(Path(f.name))
67
+
68
+ assert "Body text" in text
69
+ assert meta.get("title") == "Test"