mcp-win-stdio-rag 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_win_stdio_rag-0.2.5/.gitignore +63 -0
- mcp_win_stdio_rag-0.2.5/PKG-INFO +25 -0
- mcp_win_stdio_rag-0.2.5/README.md +9 -0
- mcp_win_stdio_rag-0.2.5/pyproject.toml +32 -0
- mcp_win_stdio_rag-0.2.5/src/mcp_win_stdio/rag/__init__.py +6 -0
- mcp_win_stdio_rag-0.2.5/src/mcp_win_stdio/rag/__main__.py +5 -0
- mcp_win_stdio_rag-0.2.5/src/mcp_win_stdio/rag/cli.py +24 -0
- mcp_win_stdio_rag-0.2.5/src/mcp_win_stdio/rag/crawler.py +179 -0
- mcp_win_stdio_rag-0.2.5/src/mcp_win_stdio/rag/guide.py +47 -0
- mcp_win_stdio_rag-0.2.5/src/mcp_win_stdio/rag/ingest.py +592 -0
- mcp_win_stdio_rag-0.2.5/src/mcp_win_stdio/rag/server.py +762 -0
- mcp_win_stdio_rag-0.2.5/src/mcp_win_stdio/rag/store.py +1105 -0
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
*$py.class
|
|
4
|
+
*.so
|
|
5
|
+
.Python
|
|
6
|
+
build/
|
|
7
|
+
develop-eggs/
|
|
8
|
+
dist/
|
|
9
|
+
downloads/
|
|
10
|
+
eggs/
|
|
11
|
+
.eggs/
|
|
12
|
+
lib/
|
|
13
|
+
lib64/
|
|
14
|
+
parts/
|
|
15
|
+
sdist/
|
|
16
|
+
var/
|
|
17
|
+
wheels/
|
|
18
|
+
share/python-wheels/
|
|
19
|
+
*.egg-info/
|
|
20
|
+
.installed.cfg
|
|
21
|
+
*.egg
|
|
22
|
+
MANIFEST
|
|
23
|
+
|
|
24
|
+
*.manifest
|
|
25
|
+
*.spec
|
|
26
|
+
api.txt
|
|
27
|
+
pypi-api.txt
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
pip-log.txt
|
|
31
|
+
pip-delete-this-directory.txt
|
|
32
|
+
pypi api.txt
|
|
33
|
+
htmlcov/
|
|
34
|
+
.tox/
|
|
35
|
+
.nox/
|
|
36
|
+
.coverage
|
|
37
|
+
.coverage.*
|
|
38
|
+
.cache
|
|
39
|
+
nosetests.xml
|
|
40
|
+
coverage.xml
|
|
41
|
+
*.cover
|
|
42
|
+
*.py,cover
|
|
43
|
+
.hypothesis/
|
|
44
|
+
.pytest_cache/
|
|
45
|
+
cover/
|
|
46
|
+
|
|
47
|
+
*.mo
|
|
48
|
+
*.pot
|
|
49
|
+
|
|
50
|
+
.env
|
|
51
|
+
.venv
|
|
52
|
+
env/
|
|
53
|
+
venv/
|
|
54
|
+
ENV/
|
|
55
|
+
env.bak/
|
|
56
|
+
venv.bak/
|
|
57
|
+
|
|
58
|
+
.idea/
|
|
59
|
+
.vscode/
|
|
60
|
+
*.swp
|
|
61
|
+
*.swo
|
|
62
|
+
|
|
63
|
+
*.log
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: mcp-win-stdio-rag
|
|
3
|
+
Version: 0.2.5
|
|
4
|
+
Summary: Windows-optimized RAG MCP Server with Playwright crawling, Link Graph trees, and local hybrid search.
|
|
5
|
+
Author: Mohan Kumar Indala
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: crawler,knowledge-graph,mcp,playwright,rag,stdio,vector-search
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Requires-Dist: beautifulsoup4>=4.12.0
|
|
10
|
+
Requires-Dist: mcp-win-stdio>=0.2.5
|
|
11
|
+
Requires-Dist: mcp>=1.2.0
|
|
12
|
+
Requires-Dist: networkx>=3.0
|
|
13
|
+
Requires-Dist: numpy>=1.24.0
|
|
14
|
+
Requires-Dist: playwright>=1.40.0
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
|
|
17
|
+
# mcp-win-stdio-rag
|
|
18
|
+
|
|
19
|
+
Windows-optimized RAG MCP Server with Playwright crawling, Link Graph trees, and local hybrid search.
|
|
20
|
+
|
|
21
|
+
## Features
|
|
22
|
+
* Automated Async Playwright Crawler
|
|
23
|
+
* NetworkX Link Tree Graph generation
|
|
24
|
+
* SQLite-backed Hybrid Vector & BM25 search
|
|
25
|
+
* Zero-token context cost cross-collection queries
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# mcp-win-stdio-rag
|
|
2
|
+
|
|
3
|
+
Windows-optimized RAG MCP Server with Playwright crawling, Link Graph trees, and local hybrid search.
|
|
4
|
+
|
|
5
|
+
## Features
|
|
6
|
+
* Automated Async Playwright Crawler
|
|
7
|
+
* NetworkX Link Tree Graph generation
|
|
8
|
+
* SQLite-backed Hybrid Vector & BM25 search
|
|
9
|
+
* Zero-token context cost cross-collection queries
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "mcp-win-stdio-rag"
|
|
7
|
+
version = "0.2.5"
|
|
8
|
+
description = "Windows-optimized RAG MCP Server with Playwright crawling, Link Graph trees, and local hybrid search."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Mohan Kumar Indala" }
|
|
14
|
+
]
|
|
15
|
+
keywords = ["mcp", "rag", "vector-search", "playwright", "crawler", "knowledge-graph", "stdio"]
|
|
16
|
+
|
|
17
|
+
dependencies = [
|
|
18
|
+
"mcp-win-stdio>=0.2.5",
|
|
19
|
+
"mcp>=1.2.0",
|
|
20
|
+
"networkx>=3.0",
|
|
21
|
+
"numpy>=1.24.0",
|
|
22
|
+
"playwright>=1.40.0",
|
|
23
|
+
"beautifulsoup4>=4.12.0",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
[project.scripts]
|
|
27
|
+
mcp-win-stdio-rag = "mcp_win_stdio.rag.cli:main"
|
|
28
|
+
mws-rag = "mcp_win_stdio.rag.cli:main"
|
|
29
|
+
|
|
30
|
+
[tool.hatch.build.targets.wheel]
|
|
31
|
+
packages = ["src/mcp_win_stdio"]
|
|
32
|
+
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
CLI entry point for mcp-win-stdio-rag.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
|
|
8
|
+
from mcp_win_stdio.rag.server import mcp
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main():
|
|
12
|
+
parser = argparse.ArgumentParser(description="mcp-win-stdio-rag MCP Server")
|
|
13
|
+
parser.add_argument("--transport", default="stdio", choices=["stdio", "sse"], help="MCP transport mechanism")
|
|
14
|
+
parser.add_argument("--port", type=int, default=8000, help="Port for SSE transport")
|
|
15
|
+
args = parser.parse_args()
|
|
16
|
+
|
|
17
|
+
if args.transport == "sse":
|
|
18
|
+
mcp.run(transport="sse", port=args.port)
|
|
19
|
+
else:
|
|
20
|
+
mcp.run(transport="stdio")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
if __name__ == "__main__":
|
|
24
|
+
main()
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Playwright-powered Automated Section-Aware Documentation Crawler and Link Graph Builder.
|
|
4
|
+
Supports Single-Page API docs with Anchor ID deep-links (e.g. Zenodo, Slate, Redoc).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import hashlib
|
|
8
|
+
from typing import Any, Dict, List, Optional, Set, Tuple
|
|
9
|
+
from urllib.parse import urljoin, urlparse
|
|
10
|
+
|
|
11
|
+
import networkx as nx
|
|
12
|
+
from playwright.async_api import async_playwright
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def get_domain_hash(url: str) -> str:
|
|
16
|
+
"""Generate a stable folder hash for a target URL domain."""
|
|
17
|
+
parsed = urlparse(url)
|
|
18
|
+
domain = parsed.netloc or parsed.path
|
|
19
|
+
return hashlib.md5(domain.encode("utf-8")).hexdigest()[:12]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def normalize_canonical_url(url: str, base_url: str) -> str:
|
|
23
|
+
"""Normalize URL, enforce HTTPS, strip tracking params and URL fragments."""
|
|
24
|
+
joined = urljoin(base_url, url)
|
|
25
|
+
parsed = urlparse(joined)
|
|
26
|
+
scheme = "https" if parsed.scheme in ("http", "https") else parsed.scheme
|
|
27
|
+
netloc = parsed.netloc.lower()
|
|
28
|
+
path = parsed.path.rstrip("/")
|
|
29
|
+
if not path:
|
|
30
|
+
path = "/"
|
|
31
|
+
return f"{scheme}://{netloc}{path}"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class AsyncPlaywrightCrawler:
|
|
35
|
+
"""Async Web Crawler with Section & Anchor-Aware Knowledge Graph Generation."""
|
|
36
|
+
|
|
37
|
+
def __init__(self, target_url: str, max_depth: int = 2, max_pages: int = 40):
|
|
38
|
+
self.target_url = target_url
|
|
39
|
+
self.max_depth = max_depth
|
|
40
|
+
self.max_pages = max_pages
|
|
41
|
+
self.domain = urlparse(target_url).netloc.lower()
|
|
42
|
+
self.visited_urls: Set[str] = set()
|
|
43
|
+
self.graph = nx.DiGraph()
|
|
44
|
+
self.pages_data: List[Dict[str, Any]] = []
|
|
45
|
+
|
|
46
|
+
def _is_same_domain(self, url: str) -> bool:
|
|
47
|
+
parsed = urlparse(url)
|
|
48
|
+
return parsed.netloc.lower() == self.domain or not parsed.netloc
|
|
49
|
+
|
|
50
|
+
async def crawl(self) -> Dict[str, Any]:
|
|
51
|
+
"""Crawl target site and extract section-aware chunks and deep-link anchors."""
|
|
52
|
+
queue: List[Tuple[str, int, Optional[str]]] = [(self.target_url, 0, None)]
|
|
53
|
+
total_sections_count = 0
|
|
54
|
+
|
|
55
|
+
async with async_playwright() as p:
|
|
56
|
+
try:
|
|
57
|
+
browser = await p.chromium.launch(headless=True)
|
|
58
|
+
except Exception as e:
|
|
59
|
+
err_text = str(e)
|
|
60
|
+
if "Executable doesn't exist" in err_text or "playwright install" in err_text:
|
|
61
|
+
import subprocess
|
|
62
|
+
import sys
|
|
63
|
+
|
|
64
|
+
sys.stderr.write("Playwright Chromium browser missing. Installing automatically...\n")
|
|
65
|
+
sys.stderr.flush()
|
|
66
|
+
subprocess.run([sys.executable, "-m", "playwright", "install", "chromium"], check=True)
|
|
67
|
+
browser = await p.chromium.launch(headless=True)
|
|
68
|
+
else:
|
|
69
|
+
raise
|
|
70
|
+
context = await browser.new_context(
|
|
71
|
+
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) Antigravity-RAG-Crawler/1.0"
|
|
72
|
+
)
|
|
73
|
+
page = await context.new_page()
|
|
74
|
+
|
|
75
|
+
while queue and len(self.visited_urls) < self.max_pages:
|
|
76
|
+
current_url, depth, parent_url = queue.pop(0)
|
|
77
|
+
norm_url = normalize_canonical_url(current_url, self.target_url)
|
|
78
|
+
|
|
79
|
+
if norm_url in self.visited_urls or depth > self.max_depth:
|
|
80
|
+
continue
|
|
81
|
+
|
|
82
|
+
self.visited_urls.add(norm_url)
|
|
83
|
+
self.graph.add_node(norm_url, depth=depth)
|
|
84
|
+
|
|
85
|
+
if parent_url:
|
|
86
|
+
self.graph.add_edge(parent_url, norm_url)
|
|
87
|
+
|
|
88
|
+
try:
|
|
89
|
+
await page.goto(norm_url, timeout=18000, wait_until="domcontentloaded")
|
|
90
|
+
title = await page.title()
|
|
91
|
+
|
|
92
|
+
# Extract all sections with their anchor IDs via JavaScript DOM traversal
|
|
93
|
+
sections_data = await page.evaluate("""
|
|
94
|
+
() => {
|
|
95
|
+
const headings = Array.from(document.querySelectorAll('h1, h2, h3, h4, section[id], article[id]'));
|
|
96
|
+
const results = [];
|
|
97
|
+
|
|
98
|
+
if (headings.length === 0) {
|
|
99
|
+
// Fallback for pages without standard headings
|
|
100
|
+
results.push({
|
|
101
|
+
anchor_id: '',
|
|
102
|
+
section_title: document.title || 'Main',
|
|
103
|
+
text: document.body ? document.body.innerText.trim() : ''
|
|
104
|
+
});
|
|
105
|
+
return results;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
for (let i = 0; i < headings.length; i++) {
|
|
109
|
+
const current = headings[i];
|
|
110
|
+
const anchor_id = current.id || current.getAttribute('name') || '';
|
|
111
|
+
const section_title = current.innerText.trim();
|
|
112
|
+
|
|
113
|
+
// Collect text content between this heading and the next heading
|
|
114
|
+
let textContent = [];
|
|
115
|
+
let nextNode = current.nextElementSibling;
|
|
116
|
+
|
|
117
|
+
while (nextNode && !['H1','H2','H3','H4','SECTION','ARTICLE'].includes(nextNode.tagName)) {
|
|
118
|
+
if (nextNode.innerText && nextNode.innerText.trim()) {
|
|
119
|
+
textContent.push(nextNode.innerText.trim());
|
|
120
|
+
}
|
|
121
|
+
nextNode = nextNode.nextElementSibling;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
const sectionBody = textContent.join('\\n\\n').trim();
|
|
125
|
+
if (section_title || sectionBody) {
|
|
126
|
+
results.push({
|
|
127
|
+
anchor_id: anchor_id,
|
|
128
|
+
section_title: section_title,
|
|
129
|
+
text: sectionBody ? `${section_title}\\n\\n${sectionBody}` : section_title
|
|
130
|
+
});
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
return results;
|
|
134
|
+
}
|
|
135
|
+
""")
|
|
136
|
+
|
|
137
|
+
# Extract internal links for continued crawling
|
|
138
|
+
hrefs = await page.eval_on_selector_all(
|
|
139
|
+
"a[href]", "elements => elements.map(el => el.getAttribute('href'))"
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
total_sections_count += len(sections_data)
|
|
143
|
+
page_info = {
|
|
144
|
+
"url": norm_url,
|
|
145
|
+
"title": title,
|
|
146
|
+
"depth": depth,
|
|
147
|
+
"parent": parent_url,
|
|
148
|
+
"sections": sections_data,
|
|
149
|
+
"sections_count": len(sections_data),
|
|
150
|
+
}
|
|
151
|
+
self.pages_data.append(page_info)
|
|
152
|
+
|
|
153
|
+
# Enqueue child links
|
|
154
|
+
if depth < self.max_depth:
|
|
155
|
+
for href in hrefs:
|
|
156
|
+
if not href or href.startswith(("javascript:", "mailto:", "tel:")) or href.startswith("#"):
|
|
157
|
+
continue
|
|
158
|
+
child_norm = normalize_canonical_url(href, norm_url)
|
|
159
|
+
if self._is_same_domain(child_norm) and child_norm not in self.visited_urls:
|
|
160
|
+
queue.append((child_norm, depth + 1, norm_url))
|
|
161
|
+
|
|
162
|
+
except Exception as e:
|
|
163
|
+
self.pages_data.append(
|
|
164
|
+
{"url": norm_url, "title": f"Error loading {norm_url}", "error": str(e), "sections": []}
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
await browser.close()
|
|
168
|
+
|
|
169
|
+
tree_structure = nx.node_link_data(self.graph)
|
|
170
|
+
is_single_page = len(self.visited_urls) == 1 and total_sections_count > 5
|
|
171
|
+
|
|
172
|
+
return {
|
|
173
|
+
"target_url": self.target_url,
|
|
174
|
+
"total_pages_crawled": len(self.visited_urls),
|
|
175
|
+
"total_sections_extracted": total_sections_count,
|
|
176
|
+
"is_single_page_doc": is_single_page,
|
|
177
|
+
"pages": self.pages_data,
|
|
178
|
+
"site_graph": tree_structure,
|
|
179
|
+
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Interactive and printable guide for the RAG MCP server.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
RAG_GUIDE = """
|
|
6
|
+
# ================================================================
|
|
7
|
+
# RAG MCP (MULTI-COLLECTION EMBEDDINGS & SEARCH) - USER GUIDE
|
|
8
|
+
# ================================================================
|
|
9
|
+
|
|
10
|
+
The RAG MCP server provides 9 specialized tools for multi-collection web crawling,
|
|
11
|
+
incremental local codebase indexing, zero-clone remote GitHub repository streaming,
|
|
12
|
+
token-compact hybrid search, and on-demand chunk expansion on Windows.
|
|
13
|
+
|
|
14
|
+
----------------------------------------------------------------
|
|
15
|
+
1. TOOL SUMMARY (9 TOOLS)
|
|
16
|
+
----------------------------------------------------------------
|
|
17
|
+
* Crawling & Ingestion:
|
|
18
|
+
- crawl_and_index_url(target_url, collection_name, max_depth, max_pages):
|
|
19
|
+
Asynchronous headless Playwright web crawler extracting clean markdown into vector collections.
|
|
20
|
+
- index_local_codebase(project_path, collection_name, include_code, extensions, force_reindex):
|
|
21
|
+
Fast AST & chunked indexing of local codebases with SHA-256 incremental cache.
|
|
22
|
+
- index_remote_repo(repo_url, collection_name, subpath, is_temp, auth_token):
|
|
23
|
+
Streams public or private GitHub repository archives directly in-memory without cloning.
|
|
24
|
+
|
|
25
|
+
* Querying & Knowledge Retrieval (Agent-Friendly & Token-Efficient):
|
|
26
|
+
- query_knowledge_base(query, target_url_or_collection, top_k, max_chars_per_snippet, compact):
|
|
27
|
+
Performs hybrid BM25 + dense semantic vector search with relevance ranking.
|
|
28
|
+
Features compact snippet mode and character limits (default 500 chars) to prevent context window bloat.
|
|
29
|
+
- get_chunk_context(chunk_id, target_url_or_collection, window):
|
|
30
|
+
Retrieves full untruncated content for any chunk, optionally expanding surrounding context window.
|
|
31
|
+
- get_knowledge_tree(target_url_or_collection, filter_path, max_items):
|
|
32
|
+
Returns hierarchical outlines and indexed document paths in a collection with path filtering.
|
|
33
|
+
|
|
34
|
+
* Collection Lifecycle & Cache Management:
|
|
35
|
+
- list_rag_collections():
|
|
36
|
+
Lists all indexed knowledge collections with document counts, chunk statistics, and sizes.
|
|
37
|
+
- delete_rag_collection(collection_name):
|
|
38
|
+
Permanently deletes a collection and its vector index from disk.
|
|
39
|
+
- prune_rag_cache(max_total_mb, purge_temp_only):
|
|
40
|
+
Frees disk space by deleting outdated crawl artifacts and orphan chunks.
|
|
41
|
+
# ================================================================
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def print_rag_guide() -> None:
|
|
46
|
+
"""Print the formatted RAG MCP guide to stdout."""
|
|
47
|
+
print(RAG_GUIDE.strip())
|