multilingual-rag-mcp 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- multilingual_rag_mcp-0.1.0/.github/workflows/ci.yml +20 -0
- multilingual_rag_mcp-0.1.0/.github/workflows/publish.yml +21 -0
- multilingual_rag_mcp-0.1.0/.gitignore +8 -0
- multilingual_rag_mcp-0.1.0/LICENSE +21 -0
- multilingual_rag_mcp-0.1.0/PKG-INFO +132 -0
- multilingual_rag_mcp-0.1.0/README.md +108 -0
- multilingual_rag_mcp-0.1.0/multilingual_rag_mcp/__init__.py +3 -0
- multilingual_rag_mcp-0.1.0/multilingual_rag_mcp/__main__.py +36 -0
- multilingual_rag_mcp-0.1.0/multilingual_rag_mcp/ingest.py +201 -0
- multilingual_rag_mcp-0.1.0/multilingual_rag_mcp/server.py +135 -0
- multilingual_rag_mcp-0.1.0/multilingual_rag_mcp/store.py +106 -0
- multilingual_rag_mcp-0.1.0/pyproject.toml +36 -0
- multilingual_rag_mcp-0.1.0/tests/__init__.py +0 -0
- multilingual_rag_mcp-0.1.0/tests/test_smoke.py +69 -0
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
matrix:
|
|
13
|
+
python-version: ["3.11", "3.12", "3.13"]
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: actions/setup-python@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: ${{ matrix.python-version }}
|
|
19
|
+
- run: pip install -e ".[test]"
|
|
20
|
+
- run: pytest
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags: ["v*"]
|
|
6
|
+
|
|
7
|
+
permissions:
|
|
8
|
+
id-token: write
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
publish:
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
environment: pypi
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: actions/setup-python@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: "3.12"
|
|
19
|
+
- run: pip install build
|
|
20
|
+
- run: python -m build
|
|
21
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Aliaksandr Kazarez
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: multilingual-rag-mcp
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Multilingual RAG MCP server — cross-lingual search over local documents
|
|
5
|
+
Project-URL: Repository, https://github.com/aliaksandr-kazarez/multilingual-rag-mcp
|
|
6
|
+
Project-URL: Issues, https://github.com/aliaksandr-kazarez/multilingual-rag-mcp/issues
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: cross-lingual,mcp,multilingual,rag,semantic-search
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
16
|
+
Requires-Python: >=3.11
|
|
17
|
+
Requires-Dist: chromadb>=1.0
|
|
18
|
+
Requires-Dist: fastembed>=0.4
|
|
19
|
+
Requires-Dist: mcp<3.0,>=2.0
|
|
20
|
+
Requires-Dist: pymupdf>=1.23
|
|
21
|
+
Provides-Extra: test
|
|
22
|
+
Requires-Dist: pytest>=8.0; extra == 'test'
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# multilingual-rag-mcp
|
|
26
|
+
|
|
27
|
+
Multilingual RAG MCP server for local document search. Query in one language, find content in another.
|
|
28
|
+
|
|
29
|
+
Built for the common case where you talk to AI agents in English but your documents are in Russian (or any other language). The multilingual embedding model maps semantically similar concepts across 50+ languages to the same vector space — no translation step needed.
|
|
30
|
+
|
|
31
|
+
## Install
|
|
32
|
+
|
|
33
|
+
### pip from PyPI
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
pip install multilingual-rag-mcp
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
### One-liner with uvx (no install needed)
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
uvx --from multilingual-rag-mcp rag-mcp index ./docs/
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
### pip from GitHub (latest)
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install git+https://github.com/aliaksandr-kazarez/multilingual-rag-mcp.git
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Quick start
|
|
52
|
+
|
|
53
|
+
### 1. Index your documents
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
rag-mcp index ~/documents/
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
### 2. Add to Claude Code
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
claude mcp add rag -- uvx --from multilingual-rag-mcp rag-mcp
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Set the document paths via env vars:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
claude mcp add rag \
|
|
69
|
+
-e RAG_DOCS=$HOME/documents \
|
|
70
|
+
-- uvx --from multilingual-rag-mcp rag-mcp
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
If installed locally (pip install), use the simpler form:
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
claude mcp add rag -e RAG_DOCS=$HOME/documents -- rag-mcp
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
### 3. Search
|
|
80
|
+
|
|
81
|
+
From Claude Code, the `search` tool handles cross-lingual queries automatically:
|
|
82
|
+
|
|
83
|
+
- "search for protein recommendations" finds Russian articles about белок
|
|
84
|
+
- "найди рецепты" finds recipe content regardless of language
|
|
85
|
+
|
|
86
|
+
## MCP tools
|
|
87
|
+
|
|
88
|
+
| Tool | Description |
|
|
89
|
+
|------|-------------|
|
|
90
|
+
| `search(query, n=5)` | Semantic search across all indexed documents |
|
|
91
|
+
| `get_document(path)` | Retrieve full document content |
|
|
92
|
+
| `list_documents(filter?)` | List indexed documents, optionally filtered |
|
|
93
|
+
| `reindex()` | Re-index all configured document directories |
|
|
94
|
+
| `stats()` | Index statistics (documents, chunks, categories) |
|
|
95
|
+
|
|
96
|
+
## Configuration
|
|
97
|
+
|
|
98
|
+
| Env var | Default | Description |
|
|
99
|
+
|---------|---------|-------------|
|
|
100
|
+
| `RAG_DOCS` | — | Comma-separated paths to document directories |
|
|
101
|
+
| `RAG_DATA` | `~/.local/share/rag-mcp/` | Index storage location |
|
|
102
|
+
| `RAG_MODEL` | `sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2` | Embedding model (any fastembed-compatible model) |
|
|
103
|
+
| `RAG_CHUNK_SIZE` | `1000` | Maximum chunk size in characters |
|
|
104
|
+
| `RAG_CHUNK_OVERLAP` | `200` | Overlap between chunks |
|
|
105
|
+
|
|
106
|
+
## Supported formats
|
|
107
|
+
|
|
108
|
+
Markdown (`.md`), plain text (`.txt`), PDF (`.pdf`), JSON (`.json`), CSV (`.csv`).
|
|
109
|
+
|
|
110
|
+
Markdown files with YAML frontmatter have their metadata (title, date, etc.) extracted automatically.
|
|
111
|
+
|
|
112
|
+
## CLI
|
|
113
|
+
|
|
114
|
+
```
|
|
115
|
+
rag-mcp index <dir> [<dir> ...] Index documents
|
|
116
|
+
rag-mcp stats Print index statistics
|
|
117
|
+
rag-mcp Start MCP server (stdio)
|
|
118
|
+
rag-mcp --help Show help
|
|
119
|
+
rag-mcp --version Show version
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## How it works
|
|
123
|
+
|
|
124
|
+
1. Documents are parsed and split into chunks (markdown-aware: respects headers, code blocks)
|
|
125
|
+
2. Each chunk is embedded with a multilingual model via [fastembed](https://github.com/qdrant/fastembed) (ONNX, no PyTorch needed)
|
|
126
|
+
3. Embeddings are stored in [ChromaDB](https://www.trychroma.com/) (local, file-based)
|
|
127
|
+
4. Queries are embedded with the same model and matched by cosine similarity
|
|
128
|
+
5. Cross-lingual retrieval works because the model places semantically similar text from different languages near each other in vector space
|
|
129
|
+
|
|
130
|
+
## License
|
|
131
|
+
|
|
132
|
+
MIT
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
# multilingual-rag-mcp
|
|
2
|
+
|
|
3
|
+
Multilingual RAG MCP server for local document search. Query in one language, find content in another.
|
|
4
|
+
|
|
5
|
+
Built for the common case where you talk to AI agents in English but your documents are in Russian (or any other language). The multilingual embedding model maps semantically similar concepts across 50+ languages to the same vector space — no translation step needed.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
### pip from PyPI
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install multilingual-rag-mcp
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
### One-liner with uvx (no install needed)
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
uvx --from multilingual-rag-mcp rag-mcp index ./docs/
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
### pip from GitHub (latest)
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install git+https://github.com/aliaksandr-kazarez/multilingual-rag-mcp.git
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## Quick start
|
|
28
|
+
|
|
29
|
+
### 1. Index your documents
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
rag-mcp index ~/documents/
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
### 2. Add to Claude Code
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
claude mcp add rag -- uvx --from multilingual-rag-mcp rag-mcp
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Set the document paths via env vars:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
claude mcp add rag \
|
|
45
|
+
-e RAG_DOCS=$HOME/documents \
|
|
46
|
+
-- uvx --from multilingual-rag-mcp rag-mcp
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
If installed locally (pip install), use the simpler form:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
claude mcp add rag -e RAG_DOCS=$HOME/documents -- rag-mcp
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
### 3. Search
|
|
56
|
+
|
|
57
|
+
From Claude Code, the `search` tool handles cross-lingual queries automatically:
|
|
58
|
+
|
|
59
|
+
- "search for protein recommendations" finds Russian articles about белок
|
|
60
|
+
- "найди рецепты" finds recipe content regardless of language
|
|
61
|
+
|
|
62
|
+
## MCP tools
|
|
63
|
+
|
|
64
|
+
| Tool | Description |
|
|
65
|
+
|------|-------------|
|
|
66
|
+
| `search(query, n=5)` | Semantic search across all indexed documents |
|
|
67
|
+
| `get_document(path)` | Retrieve full document content |
|
|
68
|
+
| `list_documents(filter?)` | List indexed documents, optionally filtered |
|
|
69
|
+
| `reindex()` | Re-index all configured document directories |
|
|
70
|
+
| `stats()` | Index statistics (documents, chunks, categories) |
|
|
71
|
+
|
|
72
|
+
## Configuration
|
|
73
|
+
|
|
74
|
+
| Env var | Default | Description |
|
|
75
|
+
|---------|---------|-------------|
|
|
76
|
+
| `RAG_DOCS` | — | Comma-separated paths to document directories |
|
|
77
|
+
| `RAG_DATA` | `~/.local/share/rag-mcp/` | Index storage location |
|
|
78
|
+
| `RAG_MODEL` | `sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2` | Embedding model (any fastembed-compatible model) |
|
|
79
|
+
| `RAG_CHUNK_SIZE` | `1000` | Maximum chunk size in characters |
|
|
80
|
+
| `RAG_CHUNK_OVERLAP` | `200` | Overlap between chunks |
|
|
81
|
+
|
|
82
|
+
## Supported formats
|
|
83
|
+
|
|
84
|
+
Markdown (`.md`), plain text (`.txt`), PDF (`.pdf`), JSON (`.json`), CSV (`.csv`).
|
|
85
|
+
|
|
86
|
+
Markdown files with YAML frontmatter have their metadata (title, date, etc.) extracted automatically.
|
|
87
|
+
|
|
88
|
+
## CLI
|
|
89
|
+
|
|
90
|
+
```
|
|
91
|
+
rag-mcp index <dir> [<dir> ...] Index documents
|
|
92
|
+
rag-mcp stats Print index statistics
|
|
93
|
+
rag-mcp Start MCP server (stdio)
|
|
94
|
+
rag-mcp --help Show help
|
|
95
|
+
rag-mcp --version Show version
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
## How it works
|
|
99
|
+
|
|
100
|
+
1. Documents are parsed and split into chunks (markdown-aware: respects headers, code blocks)
|
|
101
|
+
2. Each chunk is embedded with a multilingual model via [fastembed](https://github.com/qdrant/fastembed) (ONNX, no PyTorch needed)
|
|
102
|
+
3. Embeddings are stored in [ChromaDB](https://www.trychroma.com/) (local, file-based)
|
|
103
|
+
4. Queries are embedded with the same model and matched by cosine similarity
|
|
104
|
+
5. Cross-lingual retrieval works because the model places semantically similar text from different languages near each other in vector space
|
|
105
|
+
|
|
106
|
+
## License
|
|
107
|
+
|
|
108
|
+
MIT
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def main():
|
|
5
|
+
if len(sys.argv) > 1 and sys.argv[1] in ("-h", "--help", "help"):
|
|
6
|
+
print(
|
|
7
|
+
"Usage: rag-mcp [command]\n"
|
|
8
|
+
"\n"
|
|
9
|
+
"Commands:\n"
|
|
10
|
+
" index <dir> [<dir> ...] Index documents from directories\n"
|
|
11
|
+
" stats Print index statistics\n"
|
|
12
|
+
" (none) Start MCP server (stdio)\n"
|
|
13
|
+
"\n"
|
|
14
|
+
"Environment:\n"
|
|
15
|
+
" RAG_DOCS Comma-separated document directories\n"
|
|
16
|
+
" RAG_DATA Index storage (default: ~/.local/share/rag-mcp/)\n"
|
|
17
|
+
" RAG_MODEL Embedding model (default: paraphrase-multilingual-MiniLM-L12-v2)\n"
|
|
18
|
+
)
|
|
19
|
+
elif len(sys.argv) > 1 and sys.argv[1] in ("-V", "--version"):
|
|
20
|
+
from multilingual_rag_mcp import __version__
|
|
21
|
+
print(f"rag-mcp {__version__}")
|
|
22
|
+
elif len(sys.argv) > 1 and sys.argv[1] == "index":
|
|
23
|
+
from multilingual_rag_mcp.ingest import index_cli
|
|
24
|
+
index_cli(sys.argv[2:])
|
|
25
|
+
elif len(sys.argv) > 1 and sys.argv[1] == "stats":
|
|
26
|
+
from multilingual_rag_mcp.store import VectorStore
|
|
27
|
+
import json
|
|
28
|
+
store = VectorStore()
|
|
29
|
+
print(json.dumps(store.stats(), indent=2, ensure_ascii=False))
|
|
30
|
+
else:
|
|
31
|
+
from multilingual_rag_mcp.server import mcp
|
|
32
|
+
mcp.run()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
if __name__ == "__main__":
|
|
36
|
+
main()
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Document parsing, chunking, and indexing."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
SUPPORTED = {".md", ".txt", ".pdf", ".json", ".csv"}
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def parse_file(path: Path) -> tuple[str, dict]:
|
|
13
|
+
"""Parse a file → (text, metadata). Strips markdown frontmatter into metadata."""
|
|
14
|
+
meta: dict = {"source": str(path), "filename": path.name}
|
|
15
|
+
suffix = path.suffix.lower()
|
|
16
|
+
|
|
17
|
+
if suffix == ".pdf":
|
|
18
|
+
import pymupdf
|
|
19
|
+
|
|
20
|
+
doc = pymupdf.open(str(path))
|
|
21
|
+
text = "\n\n".join(page.get_text() for page in doc)
|
|
22
|
+
meta["title"] = doc.metadata.get("title") or path.stem
|
|
23
|
+
doc.close()
|
|
24
|
+
return text, meta
|
|
25
|
+
|
|
26
|
+
raw = path.read_text(encoding="utf-8", errors="replace")
|
|
27
|
+
|
|
28
|
+
if suffix == ".json":
|
|
29
|
+
try:
|
|
30
|
+
data = json.loads(raw)
|
|
31
|
+
text = json.dumps(data, indent=2, ensure_ascii=False)
|
|
32
|
+
except json.JSONDecodeError:
|
|
33
|
+
text = raw
|
|
34
|
+
meta["title"] = path.stem
|
|
35
|
+
return text, meta
|
|
36
|
+
|
|
37
|
+
if suffix == ".md":
|
|
38
|
+
text, fm = _strip_frontmatter(raw)
|
|
39
|
+
meta.update(fm)
|
|
40
|
+
if "title" not in meta:
|
|
41
|
+
meta["title"] = _first_heading(text) or path.stem
|
|
42
|
+
else:
|
|
43
|
+
text = raw
|
|
44
|
+
meta["title"] = path.stem
|
|
45
|
+
|
|
46
|
+
return text, meta
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _strip_frontmatter(text: str) -> tuple[str, dict]:
|
|
50
|
+
"""Remove YAML frontmatter and return (body, parsed fields)."""
|
|
51
|
+
if not text.startswith("---"):
|
|
52
|
+
return text, {}
|
|
53
|
+
end = text.find("\n---", 3)
|
|
54
|
+
if end < 0:
|
|
55
|
+
return text, {}
|
|
56
|
+
fm_block = text[3:end]
|
|
57
|
+
body = text[end + 4 :].strip()
|
|
58
|
+
fields: dict = {}
|
|
59
|
+
for line in fm_block.splitlines():
|
|
60
|
+
if ":" in line:
|
|
61
|
+
key, val = line.split(":", 1)
|
|
62
|
+
fields[key.strip()] = val.strip().strip("\"'")
|
|
63
|
+
return body, fields
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _first_heading(text: str) -> str | None:
|
|
67
|
+
for line in text.splitlines():
|
|
68
|
+
if line.startswith("# "):
|
|
69
|
+
return line[2:].strip()
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def chunk_markdown(text: str, max_size: int = 1000, overlap: int = 200) -> list[str]:
|
|
74
|
+
"""Split markdown into chunks, respecting headers and code blocks."""
|
|
75
|
+
sections = re.split(r"(?=^#{1,3} )", text, flags=re.MULTILINE)
|
|
76
|
+
chunks: list[str] = []
|
|
77
|
+
for section in sections:
|
|
78
|
+
section = section.strip()
|
|
79
|
+
if not section:
|
|
80
|
+
continue
|
|
81
|
+
if len(section) <= max_size:
|
|
82
|
+
if len(section) > 30:
|
|
83
|
+
chunks.append(section)
|
|
84
|
+
else:
|
|
85
|
+
chunks.extend(_split_paragraphs(section, max_size, overlap))
|
|
86
|
+
return chunks
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _split_paragraphs(
|
|
90
|
+
text: str, max_size: int = 1000, overlap: int = 200
|
|
91
|
+
) -> list[str]:
|
|
92
|
+
paragraphs = re.split(r"\n\s*\n", text)
|
|
93
|
+
chunks: list[str] = []
|
|
94
|
+
current = ""
|
|
95
|
+
for para in paragraphs:
|
|
96
|
+
para = para.strip()
|
|
97
|
+
if not para:
|
|
98
|
+
continue
|
|
99
|
+
if len(current) + len(para) + 2 > max_size and current:
|
|
100
|
+
chunks.append(current.strip())
|
|
101
|
+
tail = current[-overlap:].strip() if overlap else ""
|
|
102
|
+
current = (tail + "\n\n" + para) if tail else para
|
|
103
|
+
else:
|
|
104
|
+
current = (current + "\n\n" + para) if current else para
|
|
105
|
+
if current.strip() and len(current.strip()) > 30:
|
|
106
|
+
chunks.append(current.strip())
|
|
107
|
+
return chunks
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def doc_id_from_path(path: Path, base_dir: Path) -> str:
|
|
111
|
+
try:
|
|
112
|
+
return str(path.relative_to(base_dir))
|
|
113
|
+
except ValueError:
|
|
114
|
+
return str(path)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def ingest_directory(
|
|
118
|
+
store,
|
|
119
|
+
docs_dir: Path,
|
|
120
|
+
*,
|
|
121
|
+
category: str = "",
|
|
122
|
+
chunk_size: int = 1000,
|
|
123
|
+
chunk_overlap: int = 200,
|
|
124
|
+
) -> dict:
|
|
125
|
+
"""Index all supported files under docs_dir."""
|
|
126
|
+
stats = {"indexed": 0, "chunks": 0, "skipped": 0, "errors": []}
|
|
127
|
+
|
|
128
|
+
for path in sorted(docs_dir.rglob("*")):
|
|
129
|
+
if not path.is_file():
|
|
130
|
+
continue
|
|
131
|
+
if path.suffix.lower() not in SUPPORTED:
|
|
132
|
+
continue
|
|
133
|
+
if any(part.startswith(".") for part in path.relative_to(docs_dir).parts):
|
|
134
|
+
continue
|
|
135
|
+
|
|
136
|
+
try:
|
|
137
|
+
text, meta = parse_file(path)
|
|
138
|
+
if not text or len(text.strip()) < 50:
|
|
139
|
+
stats["skipped"] += 1
|
|
140
|
+
continue
|
|
141
|
+
|
|
142
|
+
did = doc_id_from_path(path, docs_dir)
|
|
143
|
+
meta["doc_id"] = did
|
|
144
|
+
meta["category"] = category or docs_dir.name
|
|
145
|
+
|
|
146
|
+
chunks = chunk_markdown(text, chunk_size, chunk_overlap)
|
|
147
|
+
if not chunks:
|
|
148
|
+
stats["skipped"] += 1
|
|
149
|
+
continue
|
|
150
|
+
|
|
151
|
+
metadatas = [{**meta, "chunk_index": i} for i in range(len(chunks))]
|
|
152
|
+
store.add_chunks(did, chunks, metadatas)
|
|
153
|
+
stats["indexed"] += 1
|
|
154
|
+
stats["chunks"] += len(chunks)
|
|
155
|
+
except Exception as e:
|
|
156
|
+
stats["errors"].append(f"{path.name}: {e}")
|
|
157
|
+
|
|
158
|
+
return stats
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def index_cli(args: list[str]) -> None:
|
|
162
|
+
"""CLI entry: `rag-mcp index /path/to/docs [/more/docs ...]`"""
|
|
163
|
+
import os
|
|
164
|
+
|
|
165
|
+
dirs = args or [
|
|
166
|
+
p.strip()
|
|
167
|
+
for p in os.environ.get("RAG_DOCS", "").split(",")
|
|
168
|
+
if p.strip()
|
|
169
|
+
]
|
|
170
|
+
if not dirs:
|
|
171
|
+
print("Usage: rag-mcp index <docs_dir> [<docs_dir> ...]", file=sys.stderr)
|
|
172
|
+
print(" or: RAG_DOCS=/path/to/docs rag-mcp index", file=sys.stderr)
|
|
173
|
+
sys.exit(1)
|
|
174
|
+
|
|
175
|
+
from multilingual_rag_mcp.store import VectorStore
|
|
176
|
+
|
|
177
|
+
store = VectorStore()
|
|
178
|
+
chunk_size = int(os.environ.get("RAG_CHUNK_SIZE", "1000"))
|
|
179
|
+
chunk_overlap = int(os.environ.get("RAG_CHUNK_OVERLAP", "200"))
|
|
180
|
+
|
|
181
|
+
for d in dirs:
|
|
182
|
+
p = Path(d).expanduser().resolve()
|
|
183
|
+
if not p.is_dir():
|
|
184
|
+
print(f"SKIP {p}: not a directory", file=sys.stderr)
|
|
185
|
+
continue
|
|
186
|
+
print(f"Indexing {p} ...", file=sys.stderr)
|
|
187
|
+
s = ingest_directory(store, p, chunk_size=chunk_size, chunk_overlap=chunk_overlap)
|
|
188
|
+
print(
|
|
189
|
+
f" {s['indexed']} docs, {s['chunks']} chunks, "
|
|
190
|
+
f"{s['skipped']} skipped, {len(s['errors'])} errors",
|
|
191
|
+
file=sys.stderr,
|
|
192
|
+
)
|
|
193
|
+
for err in s["errors"][:10]:
|
|
194
|
+
print(f" ERROR: {err}", file=sys.stderr)
|
|
195
|
+
|
|
196
|
+
total = store.stats()
|
|
197
|
+
print(
|
|
198
|
+
f"\nDone. {total['total_documents']} documents, "
|
|
199
|
+
f"{total['total_chunks']} chunks in index.",
|
|
200
|
+
file=sys.stderr,
|
|
201
|
+
)
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""MCP server exposing multilingual RAG search tools."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from mcp.server.fastmcp import FastMCP
|
|
9
|
+
|
|
10
|
+
from multilingual_rag_mcp.ingest import ingest_directory, parse_file
|
|
11
|
+
from multilingual_rag_mcp.store import VectorStore
|
|
12
|
+
|
|
13
|
+
mcp = FastMCP("rag-mcp")
|
|
14
|
+
|
|
15
|
+
_store: VectorStore | None = None
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _get_store() -> VectorStore:
|
|
19
|
+
global _store
|
|
20
|
+
if _store is None:
|
|
21
|
+
_store = VectorStore()
|
|
22
|
+
return _store
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _docs_dirs() -> list[Path]:
|
|
26
|
+
raw = os.environ.get("RAG_DOCS", "")
|
|
27
|
+
return [Path(p.strip()).expanduser().resolve() for p in raw.split(",") if p.strip()]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@mcp.tool()
|
|
31
|
+
def search(query: str, n: int = 5) -> str:
|
|
32
|
+
"""Search the knowledge base with a natural-language query.
|
|
33
|
+
|
|
34
|
+
Works across languages: query in English to find Russian content,
|
|
35
|
+
or query in Russian to find English content. The multilingual
|
|
36
|
+
embedding model maps semantically similar concepts across languages.
|
|
37
|
+
"""
|
|
38
|
+
store = _get_store()
|
|
39
|
+
results = store.search(query, n)
|
|
40
|
+
if not results:
|
|
41
|
+
return "No results found. If the index is empty, call the reindex tool first."
|
|
42
|
+
|
|
43
|
+
parts = []
|
|
44
|
+
for r in results:
|
|
45
|
+
meta = r["metadata"]
|
|
46
|
+
score = max(0, 1 - r["distance"])
|
|
47
|
+
parts.append(
|
|
48
|
+
f"### {meta.get('title', '?')} (score: {score:.2f})\n"
|
|
49
|
+
f"**Source:** {meta.get('source', '?')} \n"
|
|
50
|
+
f"**Category:** {meta.get('category', '?')}\n\n"
|
|
51
|
+
f"{r['text']}\n"
|
|
52
|
+
)
|
|
53
|
+
return "\n---\n".join(parts)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@mcp.tool()
|
|
57
|
+
def get_document(path: str) -> str:
|
|
58
|
+
"""Retrieve the full content of a source document by path.
|
|
59
|
+
|
|
60
|
+
Accepts an absolute path or a path relative to any configured RAG_DOCS directory.
|
|
61
|
+
"""
|
|
62
|
+
p = Path(path)
|
|
63
|
+
if not p.exists():
|
|
64
|
+
for d in _docs_dirs():
|
|
65
|
+
candidate = d / path
|
|
66
|
+
if candidate.exists():
|
|
67
|
+
p = candidate
|
|
68
|
+
break
|
|
69
|
+
else:
|
|
70
|
+
return f"Not found: {path}"
|
|
71
|
+
|
|
72
|
+
text, meta = parse_file(p)
|
|
73
|
+
return f"# {meta.get('title', p.name)}\n\n{text}"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@mcp.tool()
|
|
77
|
+
def list_documents(filter: str | None = None) -> str:
|
|
78
|
+
"""List indexed documents. Optionally filter by keyword in title or path."""
|
|
79
|
+
docs = _get_store().list_documents()
|
|
80
|
+
if filter:
|
|
81
|
+
fl = filter.lower()
|
|
82
|
+
docs = [
|
|
83
|
+
d
|
|
84
|
+
for d in docs
|
|
85
|
+
if fl in d["title"].lower() or fl in d["source"].lower()
|
|
86
|
+
]
|
|
87
|
+
|
|
88
|
+
if not docs:
|
|
89
|
+
return "No documents indexed. Call reindex to populate the knowledge base."
|
|
90
|
+
|
|
91
|
+
lines = ["| Title | Category | Chunks |", "|---|---|---|"]
|
|
92
|
+
for d in docs:
|
|
93
|
+
lines.append(f"| {d['title'][:60]} | {d['category'][:20]} | {d['chunks']} |")
|
|
94
|
+
lines.append(f"\n**Total: {len(docs)} documents**")
|
|
95
|
+
return "\n".join(lines)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@mcp.tool()
|
|
99
|
+
def reindex() -> str:
|
|
100
|
+
"""Re-index all documents from the configured RAG_DOCS directories."""
|
|
101
|
+
dirs = _docs_dirs()
|
|
102
|
+
if not dirs:
|
|
103
|
+
return (
|
|
104
|
+
"No RAG_DOCS configured. Set the RAG_DOCS environment variable to a "
|
|
105
|
+
"comma-separated list of directories containing your documents."
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
store = _get_store()
|
|
109
|
+
chunk_size = int(os.environ.get("RAG_CHUNK_SIZE", "1000"))
|
|
110
|
+
chunk_overlap = int(os.environ.get("RAG_CHUNK_OVERLAP", "200"))
|
|
111
|
+
|
|
112
|
+
results = []
|
|
113
|
+
for d in dirs:
|
|
114
|
+
if not d.is_dir():
|
|
115
|
+
results.append(f"{d}: not a directory, skipped")
|
|
116
|
+
continue
|
|
117
|
+
s = ingest_directory(store, d, chunk_size=chunk_size, chunk_overlap=chunk_overlap)
|
|
118
|
+
results.append(
|
|
119
|
+
f"{d.name}: {s['indexed']} docs, {s['chunks']} chunks, "
|
|
120
|
+
f"{s['skipped']} skipped, {len(s['errors'])} errors"
|
|
121
|
+
)
|
|
122
|
+
for err in s["errors"][:5]:
|
|
123
|
+
results.append(f" ERROR: {err}")
|
|
124
|
+
|
|
125
|
+
total = store.stats()
|
|
126
|
+
results.append(
|
|
127
|
+
f"\nTotal: {total['total_documents']} documents, {total['total_chunks']} chunks"
|
|
128
|
+
)
|
|
129
|
+
return "\n".join(results)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
@mcp.tool()
|
|
133
|
+
def stats() -> str:
|
|
134
|
+
"""Get knowledge base statistics: document count, chunk count, categories."""
|
|
135
|
+
return json.dumps(_get_store().stats(), indent=2, ensure_ascii=False)
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""Vector store backed by ChromaDB with multilingual embeddings."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import os
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import chromadb
|
|
8
|
+
from chromadb.api.types import Documents, EmbeddingFunction, Embeddings
|
|
9
|
+
|
|
10
|
+
DEFAULT_MODEL = "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2"
|
|
11
|
+
DEFAULT_DATA_DIR = Path.home() / ".local" / "share" / "rag-mcp"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class _FastEmbedEF(EmbeddingFunction[Documents]):
|
|
15
|
+
def __init__(self, model_name: str):
|
|
16
|
+
from fastembed import TextEmbedding
|
|
17
|
+
|
|
18
|
+
self._model = TextEmbedding(model_name=model_name)
|
|
19
|
+
|
|
20
|
+
def __call__(self, input: Documents) -> Embeddings:
|
|
21
|
+
return [e.tolist() for e in self._model.embed(list(input))]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class VectorStore:
|
|
25
|
+
def __init__(
|
|
26
|
+
self,
|
|
27
|
+
data_dir: str | Path | None = None,
|
|
28
|
+
model: str | None = None,
|
|
29
|
+
):
|
|
30
|
+
data_dir = Path(data_dir or os.environ.get("RAG_DATA", str(DEFAULT_DATA_DIR)))
|
|
31
|
+
data_dir.mkdir(parents=True, exist_ok=True)
|
|
32
|
+
self._model_name = model or os.environ.get("RAG_MODEL", DEFAULT_MODEL)
|
|
33
|
+
|
|
34
|
+
self._ef = _FastEmbedEF(self._model_name)
|
|
35
|
+
self._client = chromadb.PersistentClient(path=str(data_dir / "chroma"))
|
|
36
|
+
self._col = self._client.get_or_create_collection(
|
|
37
|
+
name="documents",
|
|
38
|
+
embedding_function=self._ef,
|
|
39
|
+
metadata={"hnsw:space": "cosine"},
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
def add_chunks(
|
|
43
|
+
self, doc_id: str, chunks: list[str], metadatas: list[dict]
|
|
44
|
+
) -> None:
|
|
45
|
+
ids = [f"{doc_id}::{i}" for i in range(len(chunks))]
|
|
46
|
+
self._col.upsert(ids=ids, documents=chunks, metadatas=metadatas)
|
|
47
|
+
|
|
48
|
+
def remove_document(self, doc_id: str) -> int:
|
|
49
|
+
existing = self._col.get(where={"doc_id": doc_id})
|
|
50
|
+
if existing["ids"]:
|
|
51
|
+
self._col.delete(ids=existing["ids"])
|
|
52
|
+
return len(existing["ids"])
|
|
53
|
+
|
|
54
|
+
def search(self, query: str, n: int = 5) -> list[dict]:
|
|
55
|
+
if self._col.count() == 0:
|
|
56
|
+
return []
|
|
57
|
+
# Fetch extra candidates to deduplicate by document
|
|
58
|
+
fetch_n = min(n * 3, self._col.count())
|
|
59
|
+
results = self._col.query(query_texts=[query], n_results=fetch_n)
|
|
60
|
+
out = []
|
|
61
|
+
seen_docs: set[str] = set()
|
|
62
|
+
for i in range(len(results["ids"][0])):
|
|
63
|
+
doc_id = results["metadatas"][0][i].get("doc_id", "")
|
|
64
|
+
if doc_id in seen_docs:
|
|
65
|
+
continue
|
|
66
|
+
seen_docs.add(doc_id)
|
|
67
|
+
out.append(
|
|
68
|
+
{
|
|
69
|
+
"id": results["ids"][0][i],
|
|
70
|
+
"text": results["documents"][0][i],
|
|
71
|
+
"metadata": results["metadatas"][0][i],
|
|
72
|
+
"distance": results["distances"][0][i],
|
|
73
|
+
}
|
|
74
|
+
)
|
|
75
|
+
if len(out) >= n:
|
|
76
|
+
break
|
|
77
|
+
return out
|
|
78
|
+
|
|
79
|
+
def list_documents(self) -> list[dict]:
|
|
80
|
+
all_data = self._col.get(include=["metadatas"])
|
|
81
|
+
docs: dict[str, dict] = {}
|
|
82
|
+
for m in all_data["metadatas"]:
|
|
83
|
+
doc_id = m.get("doc_id", "?")
|
|
84
|
+
if doc_id not in docs:
|
|
85
|
+
docs[doc_id] = {
|
|
86
|
+
"doc_id": doc_id,
|
|
87
|
+
"title": m.get("title", ""),
|
|
88
|
+
"source": m.get("source", ""),
|
|
89
|
+
"category": m.get("category", ""),
|
|
90
|
+
"chunks": 0,
|
|
91
|
+
}
|
|
92
|
+
docs[doc_id]["chunks"] += 1
|
|
93
|
+
return sorted(docs.values(), key=lambda d: d["source"])
|
|
94
|
+
|
|
95
|
+
def stats(self) -> dict:
|
|
96
|
+
docs = self.list_documents()
|
|
97
|
+
categories = {}
|
|
98
|
+
for d in docs:
|
|
99
|
+
cat = d.get("category", "uncategorized") or "uncategorized"
|
|
100
|
+
categories[cat] = categories.get(cat, 0) + 1
|
|
101
|
+
return {
|
|
102
|
+
"total_chunks": self._col.count(),
|
|
103
|
+
"total_documents": len(docs),
|
|
104
|
+
"categories": categories,
|
|
105
|
+
"model": self._model_name,
|
|
106
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "multilingual-rag-mcp"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Multilingual RAG MCP server — cross-lingual search over local documents"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.11"
|
|
12
|
+
keywords = ["mcp", "rag", "multilingual", "semantic-search", "cross-lingual"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 4 - Beta",
|
|
15
|
+
"License :: OSI Approved :: MIT License",
|
|
16
|
+
"Programming Language :: Python :: 3.11",
|
|
17
|
+
"Programming Language :: Python :: 3.12",
|
|
18
|
+
"Programming Language :: Python :: 3.13",
|
|
19
|
+
"Topic :: Text Processing :: Indexing",
|
|
20
|
+
]
|
|
21
|
+
dependencies = [
|
|
22
|
+
"mcp>=2.0,<3.0",
|
|
23
|
+
"chromadb>=1.0",
|
|
24
|
+
"fastembed>=0.4",
|
|
25
|
+
"pymupdf>=1.23",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
test = ["pytest>=8.0"]
|
|
30
|
+
|
|
31
|
+
[project.urls]
|
|
32
|
+
Repository = "https://github.com/aliaksandr-kazarez/multilingual-rag-mcp"
|
|
33
|
+
Issues = "https://github.com/aliaksandr-kazarez/multilingual-rag-mcp/issues"
|
|
34
|
+
|
|
35
|
+
[project.scripts]
|
|
36
|
+
rag-mcp = "multilingual_rag_mcp.__main__:main"
|
|
File without changes
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import subprocess
|
|
2
|
+
import sys
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def test_version_import():
|
|
8
|
+
from multilingual_rag_mcp import __version__
|
|
9
|
+
assert __version__ == "0.1.0"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def test_cli_version():
|
|
13
|
+
result = subprocess.run(
|
|
14
|
+
[sys.executable, "-m", "multilingual_rag_mcp", "--version"],
|
|
15
|
+
capture_output=True, text=True,
|
|
16
|
+
)
|
|
17
|
+
assert result.returncode == 0
|
|
18
|
+
assert "rag-mcp 0.1.0" in result.stdout
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_cli_help():
|
|
22
|
+
result = subprocess.run(
|
|
23
|
+
[sys.executable, "-m", "multilingual_rag_mcp", "--help"],
|
|
24
|
+
capture_output=True, text=True,
|
|
25
|
+
)
|
|
26
|
+
assert result.returncode == 0
|
|
27
|
+
assert "index" in result.stdout
|
|
28
|
+
assert "stats" in result.stdout
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_store_defaults():
|
|
32
|
+
from multilingual_rag_mcp.store import DEFAULT_MODEL, DEFAULT_DATA_DIR
|
|
33
|
+
assert "multilingual" in DEFAULT_MODEL
|
|
34
|
+
assert "rag-mcp" in str(DEFAULT_DATA_DIR)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_ingest_supported_formats():
|
|
38
|
+
from multilingual_rag_mcp.ingest import SUPPORTED
|
|
39
|
+
assert ".md" in SUPPORTED
|
|
40
|
+
assert ".pdf" in SUPPORTED
|
|
41
|
+
assert ".txt" in SUPPORTED
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_chunk_markdown():
|
|
45
|
+
from multilingual_rag_mcp.ingest import chunk_markdown
|
|
46
|
+
text = (
|
|
47
|
+
"# Title\n\n"
|
|
48
|
+
"This is the first paragraph with enough content to pass the length filter.\n\n"
|
|
49
|
+
"## Section\n\n"
|
|
50
|
+
"This is the second paragraph with enough content to pass the length filter."
|
|
51
|
+
)
|
|
52
|
+
chunks = chunk_markdown(text, max_size=1000, overlap=0)
|
|
53
|
+
assert len(chunks) >= 1
|
|
54
|
+
assert any("first paragraph" in c for c in chunks)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def test_parse_markdown_frontmatter():
|
|
58
|
+
import tempfile
|
|
59
|
+
from pathlib import Path
|
|
60
|
+
from multilingual_rag_mcp.ingest import parse_file
|
|
61
|
+
|
|
62
|
+
content = "---\ntitle: Test\ndate: 2025-01-01\n---\n\n# Hello\n\nBody text."
|
|
63
|
+
with tempfile.NamedTemporaryFile(suffix=".md", mode="w", delete=False) as f:
|
|
64
|
+
f.write(content)
|
|
65
|
+
f.flush()
|
|
66
|
+
text, meta = parse_file(Path(f.name))
|
|
67
|
+
|
|
68
|
+
assert "Body text" in text
|
|
69
|
+
assert meta.get("title") == "Test"
|