mcp-win-stdio-rag 0.2.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ """
2
+ RAG MCP Server package for mcp-win-stdio.
3
+ Automated Playwright Web Crawling, Link Graph Trees, and Local Hybrid RAG Search.
4
+ """
5
+
6
+ __version__ = "0.2.5"
@@ -0,0 +1,5 @@
1
+ #!/usr/bin/env python3
2
+ from mcp_win_stdio.rag.cli import main
3
+
4
+ if __name__ == "__main__":
5
+ main()
@@ -0,0 +1,24 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ CLI entry point for mcp-win-stdio-rag.
4
+ """
5
+
6
+ import argparse
7
+
8
+ from mcp_win_stdio.rag.server import mcp
9
+
10
+
11
+ def main():
12
+ parser = argparse.ArgumentParser(description="mcp-win-stdio-rag MCP Server")
13
+ parser.add_argument("--transport", default="stdio", choices=["stdio", "sse"], help="MCP transport mechanism")
14
+ parser.add_argument("--port", type=int, default=8000, help="Port for SSE transport")
15
+ args = parser.parse_args()
16
+
17
+ if args.transport == "sse":
18
+ mcp.run(transport="sse", port=args.port)
19
+ else:
20
+ mcp.run(transport="stdio")
21
+
22
+
23
+ if __name__ == "__main__":
24
+ main()
@@ -0,0 +1,179 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Playwright-powered Automated Section-Aware Documentation Crawler and Link Graph Builder.
4
+ Supports Single-Page API docs with Anchor ID deep-links (e.g. Zenodo, Slate, Redoc).
5
+ """
6
+
7
+ import hashlib
8
+ from typing import Any, Dict, List, Optional, Set, Tuple
9
+ from urllib.parse import urljoin, urlparse
10
+
11
+ import networkx as nx
12
+ from playwright.async_api import async_playwright
13
+
14
+
15
+ def get_domain_hash(url: str) -> str:
16
+ """Generate a stable folder hash for a target URL domain."""
17
+ parsed = urlparse(url)
18
+ domain = parsed.netloc or parsed.path
19
+ return hashlib.md5(domain.encode("utf-8")).hexdigest()[:12]
20
+
21
+
22
+ def normalize_canonical_url(url: str, base_url: str) -> str:
23
+ """Normalize URL, enforce HTTPS, strip tracking params and URL fragments."""
24
+ joined = urljoin(base_url, url)
25
+ parsed = urlparse(joined)
26
+ scheme = "https" if parsed.scheme in ("http", "https") else parsed.scheme
27
+ netloc = parsed.netloc.lower()
28
+ path = parsed.path.rstrip("/")
29
+ if not path:
30
+ path = "/"
31
+ return f"{scheme}://{netloc}{path}"
32
+
33
+
34
+ class AsyncPlaywrightCrawler:
35
+ """Async Web Crawler with Section & Anchor-Aware Knowledge Graph Generation."""
36
+
37
+ def __init__(self, target_url: str, max_depth: int = 2, max_pages: int = 40):
38
+ self.target_url = target_url
39
+ self.max_depth = max_depth
40
+ self.max_pages = max_pages
41
+ self.domain = urlparse(target_url).netloc.lower()
42
+ self.visited_urls: Set[str] = set()
43
+ self.graph = nx.DiGraph()
44
+ self.pages_data: List[Dict[str, Any]] = []
45
+
46
+ def _is_same_domain(self, url: str) -> bool:
47
+ parsed = urlparse(url)
48
+ return parsed.netloc.lower() == self.domain or not parsed.netloc
49
+
50
+ async def crawl(self) -> Dict[str, Any]:
51
+ """Crawl target site and extract section-aware chunks and deep-link anchors."""
52
+ queue: List[Tuple[str, int, Optional[str]]] = [(self.target_url, 0, None)]
53
+ total_sections_count = 0
54
+
55
+ async with async_playwright() as p:
56
+ try:
57
+ browser = await p.chromium.launch(headless=True)
58
+ except Exception as e:
59
+ err_text = str(e)
60
+ if "Executable doesn't exist" in err_text or "playwright install" in err_text:
61
+ import subprocess
62
+ import sys
63
+
64
+ sys.stderr.write("Playwright Chromium browser missing. Installing automatically...\n")
65
+ sys.stderr.flush()
66
+ subprocess.run([sys.executable, "-m", "playwright", "install", "chromium"], check=True)
67
+ browser = await p.chromium.launch(headless=True)
68
+ else:
69
+ raise
70
+ context = await browser.new_context(
71
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) Antigravity-RAG-Crawler/1.0"
72
+ )
73
+ page = await context.new_page()
74
+
75
+ while queue and len(self.visited_urls) < self.max_pages:
76
+ current_url, depth, parent_url = queue.pop(0)
77
+ norm_url = normalize_canonical_url(current_url, self.target_url)
78
+
79
+ if norm_url in self.visited_urls or depth > self.max_depth:
80
+ continue
81
+
82
+ self.visited_urls.add(norm_url)
83
+ self.graph.add_node(norm_url, depth=depth)
84
+
85
+ if parent_url:
86
+ self.graph.add_edge(parent_url, norm_url)
87
+
88
+ try:
89
+ await page.goto(norm_url, timeout=18000, wait_until="domcontentloaded")
90
+ title = await page.title()
91
+
92
+ # Extract all sections with their anchor IDs via JavaScript DOM traversal
93
+ sections_data = await page.evaluate("""
94
+ () => {
95
+ const headings = Array.from(document.querySelectorAll('h1, h2, h3, h4, section[id], article[id]'));
96
+ const results = [];
97
+
98
+ if (headings.length === 0) {
99
+ // Fallback for pages without standard headings
100
+ results.push({
101
+ anchor_id: '',
102
+ section_title: document.title || 'Main',
103
+ text: document.body ? document.body.innerText.trim() : ''
104
+ });
105
+ return results;
106
+ }
107
+
108
+ for (let i = 0; i < headings.length; i++) {
109
+ const current = headings[i];
110
+ const anchor_id = current.id || current.getAttribute('name') || '';
111
+ const section_title = current.innerText.trim();
112
+
113
+ // Collect text content between this heading and the next heading
114
+ let textContent = [];
115
+ let nextNode = current.nextElementSibling;
116
+
117
+ while (nextNode && !['H1','H2','H3','H4','SECTION','ARTICLE'].includes(nextNode.tagName)) {
118
+ if (nextNode.innerText && nextNode.innerText.trim()) {
119
+ textContent.push(nextNode.innerText.trim());
120
+ }
121
+ nextNode = nextNode.nextElementSibling;
122
+ }
123
+
124
+ const sectionBody = textContent.join('\\n\\n').trim();
125
+ if (section_title || sectionBody) {
126
+ results.push({
127
+ anchor_id: anchor_id,
128
+ section_title: section_title,
129
+ text: sectionBody ? `${section_title}\\n\\n${sectionBody}` : section_title
130
+ });
131
+ }
132
+ }
133
+ return results;
134
+ }
135
+ """)
136
+
137
+ # Extract internal links for continued crawling
138
+ hrefs = await page.eval_on_selector_all(
139
+ "a[href]", "elements => elements.map(el => el.getAttribute('href'))"
140
+ )
141
+
142
+ total_sections_count += len(sections_data)
143
+ page_info = {
144
+ "url": norm_url,
145
+ "title": title,
146
+ "depth": depth,
147
+ "parent": parent_url,
148
+ "sections": sections_data,
149
+ "sections_count": len(sections_data),
150
+ }
151
+ self.pages_data.append(page_info)
152
+
153
+ # Enqueue child links
154
+ if depth < self.max_depth:
155
+ for href in hrefs:
156
+ if not href or href.startswith(("javascript:", "mailto:", "tel:")) or href.startswith("#"):
157
+ continue
158
+ child_norm = normalize_canonical_url(href, norm_url)
159
+ if self._is_same_domain(child_norm) and child_norm not in self.visited_urls:
160
+ queue.append((child_norm, depth + 1, norm_url))
161
+
162
+ except Exception as e:
163
+ self.pages_data.append(
164
+ {"url": norm_url, "title": f"Error loading {norm_url}", "error": str(e), "sections": []}
165
+ )
166
+
167
+ await browser.close()
168
+
169
+ tree_structure = nx.node_link_data(self.graph)
170
+ is_single_page = len(self.visited_urls) == 1 and total_sections_count > 5
171
+
172
+ return {
173
+ "target_url": self.target_url,
174
+ "total_pages_crawled": len(self.visited_urls),
175
+ "total_sections_extracted": total_sections_count,
176
+ "is_single_page_doc": is_single_page,
177
+ "pages": self.pages_data,
178
+ "site_graph": tree_structure,
179
+ }
@@ -0,0 +1,47 @@
1
+ """
2
+ Interactive and printable guide for the RAG MCP server.
3
+ """
4
+
5
+ RAG_GUIDE = """
6
+ # ================================================================
7
+ # RAG MCP (MULTI-COLLECTION EMBEDDINGS & SEARCH) - USER GUIDE
8
+ # ================================================================
9
+
10
+ The RAG MCP server provides 9 specialized tools for multi-collection web crawling,
11
+ incremental local codebase indexing, zero-clone remote GitHub repository streaming,
12
+ token-compact hybrid search, and on-demand chunk expansion on Windows.
13
+
14
+ ----------------------------------------------------------------
15
+ 1. TOOL SUMMARY (9 TOOLS)
16
+ ----------------------------------------------------------------
17
+ * Crawling & Ingestion:
18
+ - crawl_and_index_url(target_url, collection_name, max_depth, max_pages):
19
+ Asynchronous headless Playwright web crawler extracting clean markdown into vector collections.
20
+ - index_local_codebase(project_path, collection_name, include_code, extensions, force_reindex):
21
+ Fast AST & chunked indexing of local codebases with SHA-256 incremental cache.
22
+ - index_remote_repo(repo_url, collection_name, subpath, is_temp, auth_token):
23
+ Streams public or private GitHub repository archives directly in-memory without cloning.
24
+
25
+ * Querying & Knowledge Retrieval (Agent-Friendly & Token-Efficient):
26
+ - query_knowledge_base(query, target_url_or_collection, top_k, max_chars_per_snippet, compact):
27
+ Performs hybrid BM25 + dense semantic vector search with relevance ranking.
28
+ Features compact snippet mode and character limits (default 500 chars) to prevent context window bloat.
29
+ - get_chunk_context(chunk_id, target_url_or_collection, window):
30
+ Retrieves full untruncated content for any chunk, optionally expanding surrounding context window.
31
+ - get_knowledge_tree(target_url_or_collection, filter_path, max_items):
32
+ Returns hierarchical outlines and indexed document paths in a collection with path filtering.
33
+
34
+ * Collection Lifecycle & Cache Management:
35
+ - list_rag_collections():
36
+ Lists all indexed knowledge collections with document counts, chunk statistics, and sizes.
37
+ - delete_rag_collection(collection_name):
38
+ Permanently deletes a collection and its vector index from disk.
39
+ - prune_rag_cache(max_total_mb, purge_temp_only):
40
+ Frees disk space by deleting outdated crawl artifacts and orphan chunks.
41
+ # ================================================================
42
+ """
43
+
44
+
45
+ def print_rag_guide() -> None:
46
+ """Print the formatted RAG MCP guide to stdout."""
47
+ print(RAG_GUIDE.strip())