deepcode-hku 1.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. cli/__init__.py +18 -0
  2. cli/cli_app.py +296 -0
  3. cli/cli_interface.py +744 -0
  4. cli/cli_launcher.py +155 -0
  5. cli/main_cli.py +243 -0
  6. cli/workflows/__init__.py +11 -0
  7. cli/workflows/cli_workflow_adapter.py +336 -0
  8. deepcode.py +219 -0
  9. deepcode_hku-1.0.1.dist-info/METADATA +695 -0
  10. deepcode_hku-1.0.1.dist-info/RECORD +44 -0
  11. deepcode_hku-1.0.1.dist-info/WHEEL +5 -0
  12. deepcode_hku-1.0.1.dist-info/entry_points.txt +2 -0
  13. deepcode_hku-1.0.1.dist-info/licenses/LICENSE +21 -0
  14. deepcode_hku-1.0.1.dist-info/top_level.txt +6 -0
  15. tools/__init__.py +0 -0
  16. tools/code_implementation_server.py +1045 -0
  17. tools/code_indexer.py +1657 -0
  18. tools/code_reference_indexer.py +486 -0
  19. tools/command_executor.py +324 -0
  20. tools/git_command.py +356 -0
  21. tools/pdf_converter.py +640 -0
  22. tools/pdf_downloader.py +1370 -0
  23. tools/pdf_utils.py +52 -0
  24. ui/__init__.py +43 -0
  25. ui/app.py +13 -0
  26. ui/components.py +1450 -0
  27. ui/handlers.py +773 -0
  28. ui/layout.py +106 -0
  29. ui/streamlit_app.py +38 -0
  30. ui/styles.py +2116 -0
  31. utils/__init__.py +17 -0
  32. utils/cli_interface.py +459 -0
  33. utils/dialogue_logger.py +671 -0
  34. utils/file_processor.py +426 -0
  35. utils/simple_llm_logger.py +198 -0
  36. workflows/__init__.py +31 -0
  37. workflows/agent_orchestration_engine.py +1371 -0
  38. workflows/agents/__init__.py +13 -0
  39. workflows/agents/code_implementation_agent.py +1093 -0
  40. workflows/agents/memory_agent_concise.py +923 -0
  41. workflows/agents/memory_agent_concise_index.py +935 -0
  42. workflows/code_implementation_workflow.py +924 -0
  43. workflows/code_implementation_workflow_index.py +931 -0
  44. workflows/codebase_index_workflow.py +726 -0
@@ -0,0 +1,1370 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Smart PDF Downloader MCP Tool
4
+
5
+ A standardized MCP tool using FastMCP for intelligent file downloading and document conversion.
6
+ Supports natural language instructions for downloading files from URLs, moving local files,
7
+ and automatic conversion to Markdown format with image extraction.
8
+
9
+ Features:
10
+ - Natural language instruction parsing
11
+ - URL and local path extraction
12
+ - Automatic document conversion (PDF, DOCX, PPTX, HTML, etc.)
13
+ - Image extraction and preservation
14
+ - Multi-format support with fallback options
15
+ """
16
+
17
+ import os
18
+ import re
19
+ import aiohttp
20
+ import aiofiles
21
+ import shutil
22
+ import sys
23
+ import io
24
+ from typing import List, Dict, Optional, Any
25
+ from urllib.parse import urlparse, unquote
26
+ from datetime import datetime
27
+
28
+ from mcp.server import FastMCP
29
+
30
+ # Docling imports for document conversion
31
+ try:
32
+ from docling.document_converter import DocumentConverter
33
+ from docling.datamodel.base_models import InputFormat
34
+ from docling.datamodel.pipeline_options import PdfPipelineOptions
35
+ from docling.document_converter import PdfFormatOption
36
+
37
+ DOCLING_AVAILABLE = True
38
+ except ImportError:
39
+ DOCLING_AVAILABLE = False
40
+ print(
41
+ "Warning: docling package not available. Document conversion will be disabled."
42
+ )
43
+
44
+ # Fallback PDF text extraction
45
+ try:
46
+ import PyPDF2
47
+
48
+ PYPDF2_AVAILABLE = True
49
+ except ImportError:
50
+ PYPDF2_AVAILABLE = False
51
+ print(
52
+ "Warning: PyPDF2 package not available. Fallback PDF extraction will be disabled."
53
+ )
54
+
55
+ # 设置标准输出编码为UTF-8
56
+ if sys.stdout.encoding != "utf-8":
57
+ try:
58
+ if hasattr(sys.stdout, "reconfigure"):
59
+ sys.stdout.reconfigure(encoding="utf-8")
60
+ sys.stderr.reconfigure(encoding="utf-8")
61
+ else:
62
+ sys.stdout = io.TextIOWrapper(sys.stdout.detach(), encoding="utf-8")
63
+ sys.stderr = io.TextIOWrapper(sys.stderr.detach(), encoding="utf-8")
64
+ except Exception as e:
65
+ print(f"Warning: Could not set UTF-8 encoding: {e}")
66
+
67
+ # 创建 FastMCP 实例
68
+ mcp = FastMCP("smart-pdf-downloader")
69
+
70
+
71
+ # 辅助函数
72
+ def format_success_message(action: str, details: Dict[str, Any]) -> str:
73
+ """格式化成功消息"""
74
+ return f"✅ {action}\n" + "\n".join(f" {k}: {v}" for k, v in details.items())
75
+
76
+
77
+ def format_error_message(action: str, error: str) -> str:
78
+ """格式化错误消息"""
79
+ return f"❌ {action}\n Error: {error}"
80
+
81
+
82
+ def format_warning_message(action: str, warning: str) -> str:
83
+ """格式化警告消息"""
84
+ return f"⚠️ {action}\n Warning: {warning}"
85
+
86
+
87
+ async def perform_document_conversion(
88
+ file_path: str, extract_images: bool = True
89
+ ) -> Optional[str]:
90
+ """
91
+ 执行文档转换的共用逻辑
92
+
93
+ Args:
94
+ file_path: 文件路径
95
+ extract_images: 是否提取图片
96
+
97
+ Returns:
98
+ 转换信息字符串,如果没有转换则返回None
99
+ """
100
+ if not file_path:
101
+ return None
102
+
103
+ conversion_success = False
104
+ conversion_msg = ""
105
+
106
+ # 首先尝试使用简单的PDF转换器(对于PDF文件)
107
+ if file_path.lower().endswith(".pdf") and PYPDF2_AVAILABLE and not extract_images:
108
+ try:
109
+ simple_converter = SimplePdfConverter()
110
+ conversion_result = simple_converter.convert_pdf_to_markdown(file_path)
111
+ if conversion_result["success"]:
112
+ conversion_msg = "\n [INFO] PDF converted to Markdown (PyPDF2)"
113
+ conversion_msg += (
114
+ f"\n Markdown file: {conversion_result['output_file']}"
115
+ )
116
+ conversion_msg += (
117
+ f"\n Conversion time: {conversion_result['duration']:.2f} seconds"
118
+ )
119
+ conversion_msg += (
120
+ f"\n Pages extracted: {conversion_result['pages_extracted']}"
121
+ )
122
+ conversion_success = True
123
+ else:
124
+ conversion_msg = f"\n [WARNING] PDF conversion failed: {conversion_result['error']}"
125
+ except Exception as conv_error:
126
+ conversion_msg = f"\n [WARNING] PDF conversion error: {str(conv_error)}"
127
+
128
+ # 如果简单转换失败,尝试使用docling(支持图片提取)
129
+ if not conversion_success and DOCLING_AVAILABLE:
130
+ try:
131
+ converter = DoclingConverter()
132
+ if converter.is_supported_format(file_path):
133
+ conversion_result = converter.convert_to_markdown(
134
+ file_path, extract_images=extract_images
135
+ )
136
+ if conversion_result["success"]:
137
+ conversion_msg = (
138
+ "\n [INFO] Document converted to Markdown (docling)"
139
+ )
140
+ conversion_msg += (
141
+ f"\n Markdown file: {conversion_result['output_file']}"
142
+ )
143
+ conversion_msg += f"\n Conversion time: {conversion_result['duration']:.2f} seconds"
144
+ if conversion_result.get("images_extracted", 0) > 0:
145
+ conversion_msg += f"\n Images extracted: {conversion_result['images_extracted']}"
146
+ images_dir = os.path.join(
147
+ os.path.dirname(conversion_result["output_file"]), "images"
148
+ )
149
+ conversion_msg += f"\n Images saved to: {images_dir}"
150
+ else:
151
+ conversion_msg = f"\n [WARNING] Docling conversion failed: {conversion_result['error']}"
152
+ except Exception as conv_error:
153
+ conversion_msg = (
154
+ f"\n [WARNING] Docling conversion error: {str(conv_error)}"
155
+ )
156
+
157
+ return conversion_msg if conversion_msg else None
158
+
159
+
160
+ def format_file_operation_result(
161
+ operation: str,
162
+ source: str,
163
+ destination: str,
164
+ result: Dict[str, Any],
165
+ conversion_msg: Optional[str] = None,
166
+ ) -> str:
167
+ """
168
+ 格式化文件操作结果的共用逻辑
169
+
170
+ Args:
171
+ operation: 操作类型 ("download" 或 "move")
172
+ source: 源文件/URL
173
+ destination: 目标路径
174
+ result: 操作结果字典
175
+ conversion_msg: 转换消息
176
+
177
+ Returns:
178
+ 格式化的结果消息
179
+ """
180
+ if result["success"]:
181
+ size_mb = result["size"] / (1024 * 1024)
182
+ msg = f"[SUCCESS] Successfully {operation}d: {source}\n"
183
+
184
+ if operation == "download":
185
+ msg += f" File: {destination}\n"
186
+ msg += f" Size: {size_mb:.2f} MB\n"
187
+ msg += f" Time: {result['duration']:.2f} seconds\n"
188
+ speed_mb = result.get("speed", 0) / (1024 * 1024)
189
+ msg += f" Speed: {speed_mb:.2f} MB/s"
190
+ else: # move
191
+ msg += f" To: {destination}\n"
192
+ msg += f" Size: {size_mb:.2f} MB\n"
193
+ msg += f" Time: {result['duration']:.2f} seconds"
194
+
195
+ if conversion_msg:
196
+ msg += conversion_msg
197
+
198
+ return msg
199
+ else:
200
+ return f"[ERROR] Failed to {operation}: {source}\n Error: {result.get('error', 'Unknown error')}"
201
+
202
+
203
+ class LocalPathExtractor:
204
+ """本地路径提取器"""
205
+
206
+ @staticmethod
207
+ def is_local_path(path: str) -> bool:
208
+ """判断是否为本地路径"""
209
+ path = path.strip("\"'")
210
+
211
+ # 检查是否为URL
212
+ if re.match(r"^https?://", path, re.IGNORECASE) or re.match(
213
+ r"^ftp://", path, re.IGNORECASE
214
+ ):
215
+ return False
216
+
217
+ # 路径指示符
218
+ path_indicators = [os.path.sep, "/", "\\", "~", ".", ".."]
219
+ has_extension = bool(os.path.splitext(path)[1])
220
+
221
+ if any(indicator in path for indicator in path_indicators) or has_extension:
222
+ expanded_path = os.path.expanduser(path)
223
+ return os.path.exists(expanded_path) or any(
224
+ indicator in path for indicator in path_indicators
225
+ )
226
+
227
+ return False
228
+
229
+ @staticmethod
230
+ def extract_local_paths(text: str) -> List[str]:
231
+ """从文本中提取本地文件路径"""
232
+ patterns = [
233
+ r'"([^"]+)"',
234
+ r"'([^']+)'",
235
+ r"(?:^|\s)((?:[~./\\]|[A-Za-z]:)?(?:[^/\\\s]+[/\\])*[^/\\\s]+\.[A-Za-z0-9]+)(?:\s|$)",
236
+ r"(?:^|\s)((?:~|\.{1,2})?/[^\s]+)(?:\s|$)",
237
+ r"(?:^|\s)([A-Za-z]:[/\\][^\s]+)(?:\s|$)",
238
+ r"(?:^|\s)(\.{1,2}[/\\][^\s]+)(?:\s|$)",
239
+ ]
240
+
241
+ local_paths = []
242
+ potential_paths = []
243
+
244
+ for pattern in patterns:
245
+ matches = re.findall(pattern, text, re.MULTILINE)
246
+ potential_paths.extend(matches)
247
+
248
+ for path in potential_paths:
249
+ path = path.strip()
250
+ if path and LocalPathExtractor.is_local_path(path):
251
+ expanded_path = os.path.expanduser(path)
252
+ if expanded_path not in local_paths:
253
+ local_paths.append(expanded_path)
254
+
255
+ return local_paths
256
+
257
+
258
+ class URLExtractor:
259
+ """URL提取器"""
260
+
261
+ URL_PATTERNS = [
262
+ r"https?://(?:[-\w.]|(?:%[\da-fA-F]{2}))+(?:/(?:[-\w._~!$&\'()*+,;=:@]|%[\da-fA-F]{2})*)*(?:\?(?:[-\w._~!$&\'()*+,;=:@/?]|%[\da-fA-F]{2})*)?(?:#(?:[-\w._~!$&\'()*+,;=:@/?]|%[\da-fA-F]{2})*)?",
263
+ r"ftp://(?:[-\w.]|(?:%[\da-fA-F]{2}))+(?:/(?:[-\w._~!$&\'()*+,;=:@]|%[\da-fA-F]{2})*)*",
264
+ r"(?<!\S)(?:www\.)?[-\w]+(?:\.[-\w]+)+/(?:[-\w._~!$&\'()*+,;=:@/]|%[\da-fA-F]{2})+",
265
+ ]
266
+
267
+ @staticmethod
268
+ def convert_arxiv_url(url: str) -> str:
269
+ """将arXiv网页链接转换为PDF下载链接"""
270
+ # 匹配arXiv论文ID的正则表达式
271
+ arxiv_pattern = r"arxiv\.org/abs/(\d+\.\d+)(?:v\d+)?"
272
+ match = re.search(arxiv_pattern, url, re.IGNORECASE)
273
+ if match:
274
+ paper_id = match.group(1)
275
+ return f"https://arxiv.org/pdf/{paper_id}.pdf"
276
+ return url
277
+
278
+ @classmethod
279
+ def extract_urls(cls, text: str) -> List[str]:
280
+ """从文本中提取URL"""
281
+ urls = []
282
+
283
+ # 首先处理特殊情况:@开头的URL
284
+ at_url_pattern = r"@(https?://[^\s]+)"
285
+ at_matches = re.findall(at_url_pattern, text, re.IGNORECASE)
286
+ for match in at_matches:
287
+ # 处理arXiv链接
288
+ url = cls.convert_arxiv_url(match.rstrip("/"))
289
+ urls.append(url)
290
+
291
+ # 然后使用原有的正则模式
292
+ for pattern in cls.URL_PATTERNS:
293
+ matches = re.findall(pattern, text, re.IGNORECASE)
294
+ for match in matches:
295
+ # 处理可能缺少协议的URL
296
+ if not match.startswith(("http://", "https://", "ftp://")):
297
+ # 检查是否是 www 开头
298
+ if match.startswith("www."):
299
+ match = "https://" + match
300
+ else:
301
+ # 其他情况也添加 https
302
+ match = "https://" + match
303
+
304
+ # 处理arXiv链接
305
+ url = cls.convert_arxiv_url(match.rstrip("/"))
306
+ urls.append(url)
307
+
308
+ # 去重并保持顺序
309
+ seen = set()
310
+ unique_urls = []
311
+ for url in urls:
312
+ if url not in seen:
313
+ seen.add(url)
314
+ unique_urls.append(url)
315
+
316
+ return unique_urls
317
+
318
+ @staticmethod
319
+ def infer_filename_from_url(url: str) -> str:
320
+ """从URL推断文件名"""
321
+ parsed = urlparse(url)
322
+ path = unquote(parsed.path)
323
+
324
+ # 从路径中提取文件名
325
+ filename = os.path.basename(path)
326
+
327
+ # 特殊处理:arxiv PDF链接
328
+ if "arxiv.org" in parsed.netloc and "/pdf/" in path:
329
+ if filename:
330
+ # 检查是否已经有合适的文件扩展名
331
+ if not filename.lower().endswith((".pdf", ".doc", ".docx", ".txt")):
332
+ filename = f"{filename}.pdf"
333
+ else:
334
+ path_parts = [p for p in path.split("/") if p]
335
+ if path_parts and path_parts[-1]:
336
+ filename = f"{path_parts[-1]}.pdf"
337
+ else:
338
+ timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
339
+ filename = f"arxiv_paper_{timestamp}.pdf"
340
+
341
+ # 如果没有文件名或没有扩展名,生成一个
342
+ elif not filename or "." not in filename:
343
+ # 尝试从URL生成有意义的文件名
344
+ domain = parsed.netloc.replace("www.", "").replace(".", "_")
345
+ timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
346
+
347
+ # 尝试根据路径推断文件类型
348
+ if not path or path == "/":
349
+ filename = f"{domain}_{timestamp}.html"
350
+ else:
351
+ # 使用路径的最后一部分
352
+ path_parts = [p for p in path.split("/") if p]
353
+ if path_parts:
354
+ filename = f"{path_parts[-1]}_{timestamp}"
355
+ else:
356
+ filename = f"{domain}_{timestamp}"
357
+
358
+ # 如果还是没有扩展名,根据路径推断
359
+ if "." not in filename:
360
+ # 根据路径中的关键词推断文件类型
361
+ if "/pdf/" in path.lower() or path.lower().endswith("pdf"):
362
+ filename += ".pdf"
363
+ elif any(
364
+ ext in path.lower() for ext in ["/doc/", "/word/", ".docx"]
365
+ ):
366
+ filename += ".docx"
367
+ elif any(
368
+ ext in path.lower()
369
+ for ext in ["/ppt/", "/powerpoint/", ".pptx"]
370
+ ):
371
+ filename += ".pptx"
372
+ elif any(ext in path.lower() for ext in ["/csv/", ".csv"]):
373
+ filename += ".csv"
374
+ elif any(ext in path.lower() for ext in ["/zip/", ".zip"]):
375
+ filename += ".zip"
376
+ else:
377
+ filename += ".html"
378
+
379
+ return filename
380
+
381
+
382
+ class PathExtractor:
383
+ """路径提取器"""
384
+
385
+ @staticmethod
386
+ def extract_target_path(text: str) -> Optional[str]:
387
+ """从文本中提取目标路径"""
388
+ patterns = [
389
+ r'(?:save|download|store|put|place|write|copy|move)\s+(?:to|into|in|at)\s+["\']?([^\s"\']+)["\']?',
390
+ r'(?:to|into|in|at)\s+(?:folder|directory|dir|path|location)\s*["\']?([^\s"\']+)["\']?',
391
+ r'(?:destination|target|output)\s*(?:is|:)?\s*["\']?([^\s"\']+)["\']?',
392
+ r'(?:保存|下载|存储|放到|写入|复制|移动)(?:到|至|去)\s*["\']?([^\s"\']+)["\']?',
393
+ r'(?:到|在|至)\s*["\']?([^\s"\']+)["\']?\s*(?:文件夹|目录|路径|位置)',
394
+ ]
395
+
396
+ filter_words = {
397
+ "here",
398
+ "there",
399
+ "current",
400
+ "local",
401
+ "this",
402
+ "that",
403
+ "这里",
404
+ "那里",
405
+ "当前",
406
+ "本地",
407
+ "这个",
408
+ "那个",
409
+ }
410
+
411
+ for pattern in patterns:
412
+ match = re.search(pattern, text, re.IGNORECASE)
413
+ if match:
414
+ path = match.group(1).strip("。,,.、")
415
+ if path and path.lower() not in filter_words:
416
+ return path
417
+
418
+ return None
419
+
420
+
421
+ class SimplePdfConverter:
422
+ """简单的PDF转换器,使用PyPDF2提取文本"""
423
+
424
+ def convert_pdf_to_markdown(
425
+ self, input_file: str, output_file: Optional[str] = None
426
+ ) -> Dict[str, Any]:
427
+ """
428
+ 使用PyPDF2将PDF转换为Markdown格式
429
+
430
+ Args:
431
+ input_file: 输入PDF文件路径
432
+ output_file: 输出Markdown文件路径(可选)
433
+
434
+ Returns:
435
+ 转换结果字典
436
+ """
437
+ if not PYPDF2_AVAILABLE:
438
+ return {"success": False, "error": "PyPDF2 package is not available"}
439
+
440
+ try:
441
+ # 检查输入文件是否存在
442
+ if not os.path.exists(input_file):
443
+ return {
444
+ "success": False,
445
+ "error": f"Input file not found: {input_file}",
446
+ }
447
+
448
+ # 如果没有指定输出文件,自动生成
449
+ if not output_file:
450
+ base_name = os.path.splitext(input_file)[0]
451
+ output_file = f"{base_name}.md"
452
+
453
+ # 确保输出目录存在
454
+ output_dir = os.path.dirname(output_file)
455
+ if output_dir:
456
+ os.makedirs(output_dir, exist_ok=True)
457
+
458
+ # 执行转换
459
+ start_time = datetime.now()
460
+
461
+ # 读取PDF文件
462
+ with open(input_file, "rb") as file:
463
+ pdf_reader = PyPDF2.PdfReader(file)
464
+ text_content = []
465
+
466
+ # 提取每页文本
467
+ for page_num, page in enumerate(pdf_reader.pages, 1):
468
+ text = page.extract_text()
469
+ if text.strip():
470
+ text_content.append(f"## Page {page_num}\n\n{text.strip()}\n\n")
471
+
472
+ # 生成Markdown内容
473
+ markdown_content = f"# Extracted from {os.path.basename(input_file)}\n\n"
474
+ markdown_content += f"*Total pages: {len(pdf_reader.pages)}*\n\n"
475
+ markdown_content += "---\n\n"
476
+ markdown_content += "".join(text_content)
477
+
478
+ # 保存到文件
479
+ with open(output_file, "w", encoding="utf-8") as f:
480
+ f.write(markdown_content)
481
+
482
+ # 计算转换时间
483
+ duration = (datetime.now() - start_time).total_seconds()
484
+
485
+ # 获取文件大小
486
+ input_size = os.path.getsize(input_file)
487
+ output_size = os.path.getsize(output_file)
488
+
489
+ return {
490
+ "success": True,
491
+ "input_file": input_file,
492
+ "output_file": output_file,
493
+ "input_size": input_size,
494
+ "output_size": output_size,
495
+ "duration": duration,
496
+ "markdown_content": markdown_content,
497
+ "pages_extracted": len(pdf_reader.pages),
498
+ }
499
+
500
+ except Exception as e:
501
+ return {
502
+ "success": False,
503
+ "input_file": input_file,
504
+ "error": f"Conversion failed: {str(e)}",
505
+ }
506
+
507
+
508
+ class DoclingConverter:
509
+ """文档转换器,使用docling将文档转换为Markdown格式,支持图片提取"""
510
+
511
+ def __init__(self):
512
+ if not DOCLING_AVAILABLE:
513
+ raise ImportError(
514
+ "docling package is not available. Please install it first."
515
+ )
516
+
517
+ # 配置PDF处理选项
518
+ pdf_pipeline_options = PdfPipelineOptions()
519
+ pdf_pipeline_options.do_ocr = False # 暂时禁用OCR以避免认证问题
520
+ pdf_pipeline_options.do_table_structure = False # 暂时禁用表格结构识别
521
+
522
+ # 创建文档转换器(使用基础模式)
523
+ try:
524
+ self.converter = DocumentConverter(
525
+ format_options={
526
+ InputFormat.PDF: PdfFormatOption(
527
+ pipeline_options=pdf_pipeline_options
528
+ )
529
+ }
530
+ )
531
+ except Exception:
532
+ # 如果失败,尝试更简单的配置
533
+ self.converter = DocumentConverter()
534
+
535
+ def is_supported_format(self, file_path: str) -> bool:
536
+ """检查文件格式是否支持转换"""
537
+ if not DOCLING_AVAILABLE:
538
+ return False
539
+
540
+ supported_extensions = {".pdf", ".docx", ".pptx", ".html", ".md", ".txt"}
541
+ file_extension = os.path.splitext(file_path)[1].lower()
542
+ return file_extension in supported_extensions
543
+
544
+ def is_url(self, path: str) -> bool:
545
+ """检查路径是否为URL"""
546
+ try:
547
+ result = urlparse(path)
548
+ return result.scheme in ("http", "https")
549
+ except Exception:
550
+ return False
551
+
552
+ def extract_images(self, doc, output_dir: str) -> Dict[str, str]:
553
+ """
554
+ 提取文档中的图片并保存到本地
555
+
556
+ Args:
557
+ doc: docling文档对象
558
+ output_dir: 输出目录
559
+
560
+ Returns:
561
+ 图片ID到本地文件路径的映射
562
+ """
563
+ images_dir = os.path.join(output_dir, "images")
564
+ os.makedirs(images_dir, exist_ok=True)
565
+ image_map = {} # docling图片id -> 本地文件名
566
+
567
+ try:
568
+ # 获取文档中的图片
569
+ images = getattr(doc, "images", [])
570
+
571
+ for idx, img in enumerate(images):
572
+ try:
573
+ # 获取图片格式,默认为png
574
+ ext = getattr(img, "format", None) or "png"
575
+ if ext.lower() not in ["png", "jpg", "jpeg", "gif", "bmp", "webp"]:
576
+ ext = "png"
577
+
578
+ # 生成文件名
579
+ filename = f"image_{idx+1}.{ext}"
580
+ filepath = os.path.join(images_dir, filename)
581
+
582
+ # 保存图片数据
583
+ img_data = getattr(img, "data", None)
584
+ if img_data:
585
+ with open(filepath, "wb") as f:
586
+ f.write(img_data)
587
+
588
+ # 计算相对路径
589
+ rel_path = os.path.relpath(filepath, output_dir)
590
+ img_id = getattr(img, "id", str(idx + 1))
591
+ image_map[img_id] = rel_path
592
+
593
+ except Exception as img_error:
594
+ print(f"Warning: Failed to extract image {idx+1}: {img_error}")
595
+ continue
596
+
597
+ except Exception as e:
598
+ print(f"Warning: Failed to extract images: {e}")
599
+
600
+ return image_map
601
+
602
+ def process_markdown_with_images(
603
+ self, markdown_content: str, image_map: Dict[str, str]
604
+ ) -> str:
605
+ """
606
+ 处理Markdown内容,替换图片占位符为实际的图片路径
607
+
608
+ Args:
609
+ markdown_content: 原始Markdown内容
610
+ image_map: 图片ID到本地路径的映射
611
+
612
+ Returns:
613
+ 处理后的Markdown内容
614
+ """
615
+
616
+ def replace_img(match):
617
+ img_id = match.group(1)
618
+ if img_id in image_map:
619
+ return f"![Image]({image_map[img_id]})"
620
+ else:
621
+ return match.group(0)
622
+
623
+ # 替换docling的图片占位符
624
+ processed_content = re.sub(
625
+ r"!\[Image\]\(docling://image/([^)]+)\)", replace_img, markdown_content
626
+ )
627
+
628
+ return processed_content
629
+
630
+ def convert_to_markdown(
631
+ self,
632
+ input_file: str,
633
+ output_file: Optional[str] = None,
634
+ extract_images: bool = True,
635
+ ) -> Dict[str, Any]:
636
+ """
637
+ 将文档转换为Markdown格式,支持图片提取
638
+
639
+ Args:
640
+ input_file: 输入文件路径或URL
641
+ output_file: 输出Markdown文件路径(可选)
642
+ extract_images: 是否提取图片(默认True)
643
+
644
+ Returns:
645
+ 转换结果字典
646
+ """
647
+ if not DOCLING_AVAILABLE:
648
+ return {"success": False, "error": "docling package is not available"}
649
+
650
+ try:
651
+ # 检查输入文件(如果不是URL)
652
+ if not self.is_url(input_file):
653
+ if not os.path.exists(input_file):
654
+ return {
655
+ "success": False,
656
+ "error": f"Input file not found: {input_file}",
657
+ }
658
+
659
+ # 检查文件格式是否支持
660
+ if not self.is_supported_format(input_file):
661
+ return {
662
+ "success": False,
663
+ "error": f"Unsupported file format: {os.path.splitext(input_file)[1]}",
664
+ }
665
+ else:
666
+ # 对于URL,检查是否为支持的格式
667
+ if not input_file.lower().endswith(
668
+ (".pdf", ".docx", ".pptx", ".html", ".md", ".txt")
669
+ ):
670
+ return {
671
+ "success": False,
672
+ "error": f"Unsupported URL format: {input_file}",
673
+ }
674
+
675
+ # 如果没有指定输出文件,自动生成
676
+ if not output_file:
677
+ if self.is_url(input_file):
678
+ # 从URL生成文件名
679
+ filename = URLExtractor.infer_filename_from_url(input_file)
680
+ base_name = os.path.splitext(filename)[0]
681
+ else:
682
+ base_name = os.path.splitext(input_file)[0]
683
+ output_file = f"{base_name}.md"
684
+
685
+ # 确保输出目录存在
686
+ output_dir = os.path.dirname(output_file) or "."
687
+ os.makedirs(output_dir, exist_ok=True)
688
+
689
+ # 执行转换
690
+ start_time = datetime.now()
691
+ result = self.converter.convert(input_file)
692
+ doc = result.document
693
+
694
+ # 提取图片(如果启用)
695
+ image_map = {}
696
+ images_extracted = 0
697
+ if extract_images:
698
+ image_map = self.extract_images(doc, output_dir)
699
+ images_extracted = len(image_map)
700
+
701
+ # 获取Markdown内容
702
+ markdown_content = doc.export_to_markdown()
703
+
704
+ # 处理图片占位符
705
+ if extract_images and image_map:
706
+ markdown_content = self.process_markdown_with_images(
707
+ markdown_content, image_map
708
+ )
709
+
710
+ # 保存到文件
711
+ with open(output_file, "w", encoding="utf-8") as f:
712
+ f.write(markdown_content)
713
+
714
+ # 计算转换时间
715
+ duration = (datetime.now() - start_time).total_seconds()
716
+
717
+ # 获取文件大小
718
+ if self.is_url(input_file):
719
+ input_size = 0 # URL无法直接获取大小
720
+ else:
721
+ input_size = os.path.getsize(input_file)
722
+ output_size = os.path.getsize(output_file)
723
+
724
+ return {
725
+ "success": True,
726
+ "input_file": input_file,
727
+ "output_file": output_file,
728
+ "input_size": input_size,
729
+ "output_size": output_size,
730
+ "duration": duration,
731
+ "markdown_content": markdown_content,
732
+ "images_extracted": images_extracted,
733
+ "image_map": image_map,
734
+ }
735
+
736
+ except Exception as e:
737
+ return {
738
+ "success": False,
739
+ "input_file": input_file,
740
+ "error": f"Conversion failed: {str(e)}",
741
+ }
742
+
743
+
744
+ async def check_url_accessible(url: str) -> Dict[str, Any]:
745
+ """检查URL是否可访问"""
746
+ try:
747
+ timeout = aiohttp.ClientTimeout(total=10)
748
+ async with aiohttp.ClientSession(timeout=timeout) as session:
749
+ async with session.head(url, allow_redirects=True) as response:
750
+ return {
751
+ "accessible": response.status < 400,
752
+ "status": response.status,
753
+ "content_type": response.headers.get("Content-Type", ""),
754
+ "content_length": response.headers.get("Content-Length", 0),
755
+ }
756
+ except Exception:
757
+ return {
758
+ "accessible": False,
759
+ "status": 0,
760
+ "content_type": "",
761
+ "content_length": 0,
762
+ }
763
+
764
+
765
+ async def download_file(url: str, destination: str) -> Dict[str, Any]:
766
+ """下载单个文件"""
767
+ start_time = datetime.now()
768
+ chunk_size = 8192
769
+
770
+ try:
771
+ timeout = aiohttp.ClientTimeout(total=300) # 5分钟超时
772
+ async with aiohttp.ClientSession(timeout=timeout) as session:
773
+ async with session.get(url) as response:
774
+ # 检查响应状态
775
+ response.raise_for_status()
776
+
777
+ # 获取文件信息
778
+ content_type = response.headers.get(
779
+ "Content-Type", "application/octet-stream"
780
+ )
781
+
782
+ # 确保目标目录存在
783
+ parent_dir = os.path.dirname(destination)
784
+ if parent_dir:
785
+ os.makedirs(parent_dir, exist_ok=True)
786
+
787
+ # 下载文件
788
+ downloaded = 0
789
+ async with aiofiles.open(destination, "wb") as file:
790
+ async for chunk in response.content.iter_chunked(chunk_size):
791
+ await file.write(chunk)
792
+ downloaded += len(chunk)
793
+
794
+ # 计算下载时间
795
+ duration = (datetime.now() - start_time).total_seconds()
796
+
797
+ return {
798
+ "success": True,
799
+ "url": url,
800
+ "destination": destination,
801
+ "size": downloaded,
802
+ "content_type": content_type,
803
+ "duration": duration,
804
+ "speed": downloaded / duration if duration > 0 else 0,
805
+ }
806
+
807
+ except aiohttp.ClientError as e:
808
+ return {
809
+ "success": False,
810
+ "url": url,
811
+ "destination": destination,
812
+ "error": f"Network error: {str(e)}",
813
+ }
814
+ except Exception as e:
815
+ return {
816
+ "success": False,
817
+ "url": url,
818
+ "destination": destination,
819
+ "error": f"Download error: {str(e)}",
820
+ }
821
+
822
+
823
+ async def move_local_file(source_path: str, destination: str) -> Dict[str, Any]:
824
+ """移动本地文件到目标位置"""
825
+ start_time = datetime.now()
826
+
827
+ try:
828
+ # 检查源文件是否存在
829
+ if not os.path.exists(source_path):
830
+ return {
831
+ "success": False,
832
+ "source": source_path,
833
+ "destination": destination,
834
+ "error": f"Source file not found: {source_path}",
835
+ }
836
+
837
+ # 获取源文件信息
838
+ source_size = os.path.getsize(source_path)
839
+
840
+ # 确保目标目录存在
841
+ parent_dir = os.path.dirname(destination)
842
+ if parent_dir:
843
+ os.makedirs(parent_dir, exist_ok=True)
844
+
845
+ # 执行移动操作
846
+ shutil.move(source_path, destination)
847
+
848
+ # 计算操作时间
849
+ duration = (datetime.now() - start_time).total_seconds()
850
+
851
+ return {
852
+ "success": True,
853
+ "source": source_path,
854
+ "destination": destination,
855
+ "size": source_size,
856
+ "duration": duration,
857
+ "operation": "move",
858
+ }
859
+
860
+ except Exception as e:
861
+ return {
862
+ "success": False,
863
+ "source": source_path,
864
+ "destination": destination,
865
+ "error": f"Move error: {str(e)}",
866
+ }
867
+
868
+
869
+ @mcp.tool()
870
+ async def download_files(instruction: str) -> str:
871
+ """
872
+ Download files from URLs or move local files mentioned in natural language instructions.
873
+
874
+ Args:
875
+ instruction: Natural language instruction containing URLs/local paths and optional destination paths
876
+
877
+ Returns:
878
+ Status message about the download/move operations
879
+
880
+ Examples:
881
+ - "Download https://example.com/file.pdf to documents folder"
882
+ - "Move /home/user/file.pdf to documents folder"
883
+ - "Please get https://raw.githubusercontent.com/user/repo/main/data.csv and save it to ~/downloads"
884
+ - "移动 ~/Desktop/report.docx 到 /tmp/documents/"
885
+ - "Download www.example.com/report.xlsx"
886
+ """
887
+ urls = URLExtractor.extract_urls(instruction)
888
+ local_paths = LocalPathExtractor.extract_local_paths(instruction)
889
+
890
+ if not urls and not local_paths:
891
+ return format_error_message(
892
+ "Failed to parse instruction",
893
+ "No downloadable URLs or movable local files found",
894
+ )
895
+
896
+ target_path = PathExtractor.extract_target_path(instruction)
897
+
898
+ # 处理文件
899
+ results = []
900
+
901
+ # 处理URL下载
902
+ for url in urls:
903
+ try:
904
+ # 推断文件名
905
+ filename = URLExtractor.infer_filename_from_url(url)
906
+
907
+ # 构建完整的目标路径
908
+ if target_path:
909
+ # 处理路径
910
+ if target_path.startswith("~"):
911
+ target_path = os.path.expanduser(target_path)
912
+
913
+ # 确保使用相对路径(如果不是绝对路径)
914
+ if not os.path.isabs(target_path):
915
+ target_path = os.path.normpath(target_path)
916
+
917
+ # 判断是文件路径还是目录路径
918
+ if os.path.splitext(target_path)[1]: # 有扩展名,是文件
919
+ destination = target_path
920
+ else: # 是目录
921
+ destination = os.path.join(target_path, filename)
922
+ else:
923
+ # 默认下载到当前目录
924
+ destination = filename
925
+
926
+ # 检查文件是否已存在
927
+ if os.path.exists(destination):
928
+ results.append(
929
+ f"[WARNING] Skipped {url}: File already exists at {destination}"
930
+ )
931
+ continue
932
+
933
+ # 先检查URL是否可访问
934
+ check_result = await check_url_accessible(url)
935
+ if not check_result["accessible"]:
936
+ results.append(
937
+ f"[ERROR] Failed to access {url}: HTTP {check_result['status'] or 'Connection failed'}"
938
+ )
939
+ continue
940
+
941
+ # 执行下载
942
+ result = await download_file(url, destination)
943
+
944
+ # 执行转换(如果成功下载)
945
+ conversion_msg = None
946
+ if result["success"]:
947
+ conversion_msg = await perform_document_conversion(
948
+ destination, extract_images=True
949
+ )
950
+
951
+ # 格式化结果
952
+ msg = format_file_operation_result(
953
+ "download", url, destination, result, conversion_msg
954
+ )
955
+
956
+ except Exception as e:
957
+ msg = f"[ERROR] Failed to download: {url}\n"
958
+ msg += f" Error: {str(e)}"
959
+
960
+ results.append(msg)
961
+
962
+ # 处理本地文件移动
963
+ for local_path in local_paths:
964
+ try:
965
+ # 获取文件名
966
+ filename = os.path.basename(local_path)
967
+
968
+ # 构建完整的目标路径
969
+ if target_path:
970
+ # 处理路径
971
+ if target_path.startswith("~"):
972
+ target_path = os.path.expanduser(target_path)
973
+
974
+ # 确保使用相对路径(如果不是绝对路径)
975
+ if not os.path.isabs(target_path):
976
+ target_path = os.path.normpath(target_path)
977
+
978
+ # 判断是文件路径还是目录路径
979
+ if os.path.splitext(target_path)[1]: # 有扩展名,是文件
980
+ destination = target_path
981
+ else: # 是目录
982
+ destination = os.path.join(target_path, filename)
983
+ else:
984
+ # 默认移动到当前目录
985
+ destination = filename
986
+
987
+ # 检查目标文件是否已存在
988
+ if os.path.exists(destination):
989
+ results.append(
990
+ f"[WARNING] Skipped {local_path}: File already exists at {destination}"
991
+ )
992
+ continue
993
+
994
+ # 执行移动
995
+ result = await move_local_file(local_path, destination)
996
+
997
+ # 执行转换(如果成功移动)
998
+ conversion_msg = None
999
+ if result["success"]:
1000
+ conversion_msg = await perform_document_conversion(
1001
+ destination, extract_images=True
1002
+ )
1003
+
1004
+ # 格式化结果
1005
+ msg = format_file_operation_result(
1006
+ "move", local_path, destination, result, conversion_msg
1007
+ )
1008
+
1009
+ except Exception as e:
1010
+ msg = f"[ERROR] Failed to move: {local_path}\n"
1011
+ msg += f" Error: {str(e)}"
1012
+
1013
+ results.append(msg)
1014
+
1015
+ return "\n\n".join(results)
1016
+
1017
+
1018
+ @mcp.tool()
1019
+ async def parse_download_urls(text: str) -> str:
1020
+ """
1021
+ Extract URLs, local paths and target paths from text without downloading or moving.
1022
+
1023
+ Args:
1024
+ text: Text containing URLs, local paths and optional destination paths
1025
+
1026
+ Returns:
1027
+ Parsed URLs, local paths and target path information
1028
+ """
1029
+ urls = URLExtractor.extract_urls(text)
1030
+ local_paths = LocalPathExtractor.extract_local_paths(text)
1031
+ target_path = PathExtractor.extract_target_path(text)
1032
+
1033
+ content = "📋 Parsed file operation information:\n\n"
1034
+
1035
+ if urls:
1036
+ content += f"🔗 URLs found ({len(urls)}):\n"
1037
+ for i, url in enumerate(urls, 1):
1038
+ filename = URLExtractor.infer_filename_from_url(url)
1039
+ content += f" {i}. {url}\n 📄 Filename: {filename}\n"
1040
+ else:
1041
+ content += "🔗 No URLs found\n"
1042
+
1043
+ if local_paths:
1044
+ content += f"\n📁 Local files found ({len(local_paths)}):\n"
1045
+ for i, path in enumerate(local_paths, 1):
1046
+ exists = os.path.exists(path)
1047
+ content += f" {i}. {path}\n"
1048
+ content += f" ✅ Exists: {'Yes' if exists else 'No'}\n"
1049
+ if exists:
1050
+ size_mb = os.path.getsize(path) / (1024 * 1024)
1051
+ content += f" 📊 Size: {size_mb:.2f} MB\n"
1052
+ else:
1053
+ content += "\n📁 No local files found\n"
1054
+
1055
+ if target_path:
1056
+ content += f"\n🎯 Target path: {target_path}"
1057
+ if target_path.startswith("~"):
1058
+ content += f"\n (Expanded: {os.path.expanduser(target_path)})"
1059
+ else:
1060
+ content += "\n🎯 Target path: Not specified (will use current directory)"
1061
+
1062
+ return content
1063
+
1064
+
1065
+ @mcp.tool()
1066
+ async def download_file_to(
1067
+ url: str, destination: Optional[str] = None, filename: Optional[str] = None
1068
+ ) -> str:
1069
+ """
1070
+ Download a specific file with detailed options.
1071
+
1072
+ Args:
1073
+ url: URL to download from
1074
+ destination: Target directory or full file path (optional)
1075
+ filename: Specific filename to use (optional, ignored if destination is a full file path)
1076
+
1077
+ Returns:
1078
+ Status message about the download operation
1079
+ """
1080
+ # 确定文件名
1081
+ if not filename:
1082
+ filename = URLExtractor.infer_filename_from_url(url)
1083
+
1084
+ # 确定完整路径
1085
+ if destination:
1086
+ # 展开用户目录
1087
+ if destination.startswith("~"):
1088
+ destination = os.path.expanduser(destination)
1089
+
1090
+ # 检查是否是完整文件路径
1091
+ if os.path.splitext(destination)[1]: # 有扩展名
1092
+ target_path = destination
1093
+ else: # 是目录
1094
+ target_path = os.path.join(destination, filename)
1095
+ else:
1096
+ target_path = filename
1097
+
1098
+ # 确保使用相对路径(如果不是绝对路径)
1099
+ if not os.path.isabs(target_path):
1100
+ target_path = os.path.normpath(target_path)
1101
+
1102
+ # 检查文件是否已存在
1103
+ if os.path.exists(target_path):
1104
+ return format_error_message(
1105
+ "Download aborted", f"File already exists at {target_path}"
1106
+ )
1107
+
1108
+ # 先检查URL
1109
+ check_result = await check_url_accessible(url)
1110
+ if not check_result["accessible"]:
1111
+ return format_error_message(
1112
+ "Cannot access URL",
1113
+ f"{url} (HTTP {check_result['status'] or 'Connection failed'})",
1114
+ )
1115
+
1116
+ # 显示下载信息
1117
+ size_mb = (
1118
+ int(check_result["content_length"]) / (1024 * 1024)
1119
+ if check_result["content_length"]
1120
+ else 0
1121
+ )
1122
+ msg = "[INFO] Downloading file:\n"
1123
+ msg += f" URL: {url}\n"
1124
+ msg += f" Target: {target_path}\n"
1125
+ if size_mb > 0:
1126
+ msg += f" Expected size: {size_mb:.2f} MB\n"
1127
+ msg += "\n"
1128
+
1129
+ # 执行下载
1130
+ result = await download_file(url, target_path)
1131
+
1132
+ # 执行转换(如果成功下载)
1133
+ conversion_msg = None
1134
+ if result["success"]:
1135
+ conversion_msg = await perform_document_conversion(
1136
+ target_path, extract_images=True
1137
+ )
1138
+
1139
+ # 添加下载信息前缀
1140
+ actual_size_mb = result["size"] / (1024 * 1024)
1141
+ speed_mb = result["speed"] / (1024 * 1024)
1142
+ info_msg = "[SUCCESS] Download completed!\n"
1143
+ info_msg += f" Saved to: {target_path}\n"
1144
+ info_msg += f" Size: {actual_size_mb:.2f} MB\n"
1145
+ info_msg += f" Duration: {result['duration']:.2f} seconds\n"
1146
+ info_msg += f" Speed: {speed_mb:.2f} MB/s\n"
1147
+ info_msg += f" Type: {result['content_type']}"
1148
+
1149
+ if conversion_msg:
1150
+ info_msg += conversion_msg
1151
+
1152
+ return msg + info_msg
1153
+ else:
1154
+ return msg + f"[ERROR] Download failed!\n Error: {result['error']}"
1155
+
1156
+
1157
+ @mcp.tool()
1158
+ async def move_file_to(
1159
+ source: str, destination: Optional[str] = None, filename: Optional[str] = None
1160
+ ) -> str:
1161
+ """
1162
+ Move a local file to a new location with detailed options.
1163
+
1164
+ Args:
1165
+ source: Source file path to move
1166
+ destination: Target directory or full file path (optional)
1167
+ filename: Specific filename to use (optional, ignored if destination is a full file path)
1168
+
1169
+ Returns:
1170
+ Status message about the move operation
1171
+ """
1172
+ # 展开源路径
1173
+ if source.startswith("~"):
1174
+ source = os.path.expanduser(source)
1175
+
1176
+ # 检查源文件是否存在
1177
+ if not os.path.exists(source):
1178
+ return format_error_message("Move aborted", f"Source file not found: {source}")
1179
+
1180
+ # 确定文件名
1181
+ if not filename:
1182
+ filename = os.path.basename(source)
1183
+
1184
+ # 确定完整路径
1185
+ if destination:
1186
+ # 展开用户目录
1187
+ if destination.startswith("~"):
1188
+ destination = os.path.expanduser(destination)
1189
+
1190
+ # 检查是否是完整文件路径
1191
+ if os.path.splitext(destination)[1]: # 有扩展名
1192
+ target_path = destination
1193
+ else: # 是目录
1194
+ target_path = os.path.join(destination, filename)
1195
+ else:
1196
+ target_path = filename
1197
+
1198
+ # 确保使用相对路径(如果不是绝对路径)
1199
+ if not os.path.isabs(target_path):
1200
+ target_path = os.path.normpath(target_path)
1201
+
1202
+ # 检查目标文件是否已存在
1203
+ if os.path.exists(target_path):
1204
+ return f"[ERROR] Target file already exists: {target_path}"
1205
+
1206
+ # 显示移动信息
1207
+ source_size_mb = os.path.getsize(source) / (1024 * 1024)
1208
+ msg = "[INFO] Moving file:\n"
1209
+ msg += f" Source: {source}\n"
1210
+ msg += f" Target: {target_path}\n"
1211
+ msg += f" Size: {source_size_mb:.2f} MB\n"
1212
+ msg += "\n"
1213
+
1214
+ # 执行移动
1215
+ result = await move_local_file(source, target_path)
1216
+
1217
+ # 执行转换(如果成功移动)
1218
+ conversion_msg = None
1219
+ if result["success"]:
1220
+ conversion_msg = await perform_document_conversion(
1221
+ target_path, extract_images=True
1222
+ )
1223
+
1224
+ # 添加移动信息前缀
1225
+ info_msg = "[SUCCESS] File moved successfully!\n"
1226
+ info_msg += f" From: {source}\n"
1227
+ info_msg += f" To: {target_path}\n"
1228
+ info_msg += f" Duration: {result['duration']:.2f} seconds"
1229
+
1230
+ if conversion_msg:
1231
+ info_msg += conversion_msg
1232
+
1233
+ return msg + info_msg
1234
+ else:
1235
+ return msg + f"[ERROR] Move failed!\n Error: {result['error']}"
1236
+
1237
+
1238
+ @mcp.tool()
1239
+ async def convert_document_to_markdown(
1240
+ file_path: str, output_path: Optional[str] = None, extract_images: bool = True
1241
+ ) -> str:
1242
+ """
1243
+ Convert a document to Markdown format with image extraction support.
1244
+
1245
+ Supports both local files and URLs. Uses docling for advanced conversion with image extraction,
1246
+ or falls back to PyPDF2 for simple PDF text extraction.
1247
+
1248
+ Args:
1249
+ file_path: Path to the input document file or URL (supports PDF, DOCX, PPTX, HTML, TXT, MD)
1250
+ output_path: Path for the output Markdown file (optional, auto-generated if not provided)
1251
+ extract_images: Whether to extract images from the document (default: True)
1252
+
1253
+ Returns:
1254
+ Status message about the conversion operation with preview of converted content
1255
+
1256
+ Examples:
1257
+ - "convert_document_to_markdown('paper.pdf')"
1258
+ - "convert_document_to_markdown('https://example.com/doc.pdf', 'output.md')"
1259
+ - "convert_document_to_markdown('presentation.pptx', extract_images=False)"
1260
+ """
1261
+ # 检查是否为URL
1262
+ is_url_input = False
1263
+ try:
1264
+ parsed = urlparse(file_path)
1265
+ is_url_input = parsed.scheme in ("http", "https")
1266
+ except Exception:
1267
+ is_url_input = False
1268
+
1269
+ # 检查文件是否存在(如果不是URL)
1270
+ if not is_url_input and not os.path.exists(file_path):
1271
+ return f"[ERROR] Input file not found: {file_path}"
1272
+
1273
+ # 检查是否是PDF文件,优先使用简单转换器(仅对本地文件)
1274
+ if (
1275
+ not is_url_input
1276
+ and file_path.lower().endswith(".pdf")
1277
+ and PYPDF2_AVAILABLE
1278
+ and not extract_images
1279
+ ):
1280
+ try:
1281
+ simple_converter = SimplePdfConverter()
1282
+ result = simple_converter.convert_pdf_to_markdown(file_path, output_path)
1283
+ except Exception as e:
1284
+ return f"[ERROR] PDF conversion error: {str(e)}"
1285
+ elif DOCLING_AVAILABLE:
1286
+ try:
1287
+ converter = DoclingConverter()
1288
+
1289
+ # 检查文件格式是否支持
1290
+ if not is_url_input and not converter.is_supported_format(file_path):
1291
+ supported_formats = [".pdf", ".docx", ".pptx", ".html", ".md", ".txt"]
1292
+ return f"[ERROR] Unsupported file format. Supported formats: {', '.join(supported_formats)}"
1293
+ elif is_url_input and not file_path.lower().endswith(
1294
+ (".pdf", ".docx", ".pptx", ".html", ".md", ".txt")
1295
+ ):
1296
+ return f"[ERROR] Unsupported URL format: {file_path}"
1297
+
1298
+ # 执行转换(支持图片提取)
1299
+ result = converter.convert_to_markdown(
1300
+ file_path, output_path, extract_images
1301
+ )
1302
+ except Exception as e:
1303
+ return f"[ERROR] Docling conversion error: {str(e)}"
1304
+ else:
1305
+ return (
1306
+ "[ERROR] No conversion tools available. Please install docling or PyPDF2."
1307
+ )
1308
+
1309
+ if result["success"]:
1310
+ msg = "[SUCCESS] Document converted successfully!\n"
1311
+ msg += f" Input: {result['input_file']}\n"
1312
+ msg += f" Output file: {result['output_file']}\n"
1313
+ msg += f" Conversion time: {result['duration']:.2f} seconds\n"
1314
+
1315
+ if result["input_size"] > 0:
1316
+ msg += f" Original size: {result['input_size'] / 1024:.1f} KB\n"
1317
+ msg += f" Markdown size: {result['output_size'] / 1024:.1f} KB\n"
1318
+
1319
+ # 显示图片提取信息
1320
+ if extract_images and "images_extracted" in result:
1321
+ images_count = result["images_extracted"]
1322
+ if images_count > 0:
1323
+ msg += f" Images extracted: {images_count}\n"
1324
+ msg += f" Images saved to: {os.path.join(os.path.dirname(result['output_file']), 'images')}\n"
1325
+ else:
1326
+ msg += " No images found in document\n"
1327
+
1328
+ # 显示Markdown内容的前几行作为预览
1329
+ content_lines = result["markdown_content"].split("\n")
1330
+ preview_lines = content_lines[:5]
1331
+ if len(content_lines) > 5:
1332
+ preview_lines.append("...")
1333
+
1334
+ msg += "\n[PREVIEW] First few lines of converted Markdown:\n"
1335
+ for line in preview_lines:
1336
+ msg += f" {line}\n"
1337
+ else:
1338
+ msg = "[ERROR] Conversion failed!\n"
1339
+ msg += f" Error: {result['error']}"
1340
+
1341
+ return msg
1342
+
1343
+
1344
+ if __name__ == "__main__":
1345
+ print("📄 Smart PDF Downloader MCP Tool")
1346
+ print("📝 Starting server with FastMCP...")
1347
+
1348
+ if DOCLING_AVAILABLE:
1349
+ print("✅ Document conversion to Markdown is ENABLED (docling available)")
1350
+ else:
1351
+ print("❌ Document conversion to Markdown is DISABLED (docling not available)")
1352
+ print(" Install docling to enable: pip install docling")
1353
+
1354
+ print("\nAvailable tools:")
1355
+ print(
1356
+ " • download_files - Download files or move local files from natural language"
1357
+ )
1358
+ print(" • parse_download_urls - Extract URLs, local paths and destination paths")
1359
+ print(" • download_file_to - Download a specific file with options")
1360
+ print(" • move_file_to - Move a specific local file with options")
1361
+ print(" • convert_document_to_markdown - Convert documents to Markdown format")
1362
+
1363
+ if DOCLING_AVAILABLE:
1364
+ print("\nSupported formats: PDF, DOCX, PPTX, HTML, TXT, MD")
1365
+ print("Features: Image extraction, Layout preservation, Automatic conversion")
1366
+
1367
+ print("")
1368
+
1369
+ # 运行服务器
1370
+ mcp.run()