deepcode-hku 1.0.4__tar.gz → 1.0.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. {deepcode_hku-1.0.4/deepcode_hku.egg-info → deepcode_hku-1.0.6}/PKG-INFO +84 -2
  2. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/README.md +82 -1
  3. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/__init__.py +1 -1
  4. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/cli_app.py +3 -55
  5. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/cli_interface.py +65 -3
  6. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/main_cli.py +13 -11
  7. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/workflows/cli_workflow_adapter.py +101 -58
  8. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode.py +74 -1
  9. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6/deepcode_hku.egg-info}/PKG-INFO +84 -2
  10. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/SOURCES.txt +4 -1
  11. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/requires.txt +1 -0
  12. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/mcp_agent.config.yaml +30 -12
  13. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/mcp_agent.secrets.yaml +4 -2
  14. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/prompts/code_prompts.py +122 -40
  15. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/requirements.txt +1 -0
  16. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/schema/mcp-agent.config.schema.json +0 -34
  17. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/code_implementation_server.py +432 -0
  18. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/pdf_downloader.py +48 -24
  19. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/components.py +548 -377
  20. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/handlers.py +234 -5
  21. deepcode_hku-1.0.6/ui/layout.py +161 -0
  22. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/styles.py +1390 -71
  23. deepcode_hku-1.0.6/utils/cross_platform_file_handler.py +475 -0
  24. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/file_processor.py +12 -0
  25. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/llm_utils.py +6 -0
  26. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agent_orchestration_engine.py +190 -13
  27. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/memory_agent_concise.py +676 -151
  28. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/memory_agent_concise_index.py +724 -180
  29. deepcode_hku-1.0.6/workflows/agents/memory_agent_concise_multi.py +1659 -0
  30. deepcode_hku-1.0.6/workflows/agents/requirement_analysis_agent.py +410 -0
  31. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/code_implementation_workflow.py +384 -51
  32. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/code_implementation_workflow_index.py +371 -41
  33. deepcode_hku-1.0.4/ui/layout.py +0 -106
  34. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/.pre-commit-config.yaml +0 -0
  35. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/LICENSE +0 -0
  36. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/MANIFEST.in +0 -0
  37. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/__init__.py +0 -0
  38. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/cli_launcher.py +0 -0
  39. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/workflows/__init__.py +0 -0
  40. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/dependency_links.txt +0 -0
  41. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/entry_points.txt +0 -0
  42. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/top_level.txt +0 -0
  43. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/setup.cfg +0 -0
  44. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/setup.py +0 -0
  45. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/__init__.py +0 -0
  46. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/bocha_search_server.py +0 -0
  47. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/code_indexer.py +0 -0
  48. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/code_reference_indexer.py +0 -0
  49. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/command_executor.py +0 -0
  50. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/document_segmentation_server.py +0 -0
  51. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/git_command.py +0 -0
  52. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/pdf_converter.py +0 -0
  53. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/pdf_utils.py +0 -0
  54. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/__init__.py +0 -0
  55. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/app.py +0 -0
  56. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/streamlit_app.py +0 -0
  57. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/__init__.py +0 -0
  58. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/cli_interface.py +0 -0
  59. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/dialogue_logger.py +0 -0
  60. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/simple_llm_logger.py +0 -0
  61. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/__init__.py +0 -0
  62. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/__init__.py +0 -0
  63. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/code_implementation_agent.py +0 -0
  64. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/document_segmentation_agent.py +0 -0
  65. {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/codebase_index_workflow.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: deepcode-hku
3
- Version: 1.0.4
3
+ Version: 1.0.6
4
4
  Summary: AI Research Engine - Transform research papers into working code automatically
5
5
  Home-page: https://github.com/HKUDS/DeepCode
6
6
  Author: DeepCodeTeam
@@ -27,6 +27,7 @@ Requires-Dist: docling
27
27
  Requires-Dist: mcp-agent
28
28
  Requires-Dist: mcp-server-git
29
29
  Requires-Dist: nest_asyncio
30
+ Requires-Dist: openai
30
31
  Requires-Dist: pathlib2
31
32
  Requires-Dist: PyPDF2>=2.0.0
32
33
  Requires-Dist: reportlab>=3.5.0
@@ -60,6 +61,10 @@ Dynamic: summary
60
61
  </tr>
61
62
  </table>
62
63
 
64
+ <div align="center">
65
+ <a href="https://trendshift.io/repositories/14665" target="_blank"><img src="https://trendshift.io/api/badge/repositories/14665" alt="HKUDS%2FDeepCode | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
66
+ </div>
67
+
63
68
  <!-- <img src="https://readme-typing-svg.herokuapp.com?font=Russo+One&size=28&duration=2000&pause=800&color=06B6D4&background=00000000&center=true&vCenter=true&width=800&height=50&lines=%E2%9A%A1+OPEN+AGENTIC+CODING+%E2%9A%A1" alt="DeepCode Tech Subtitle" style="margin-top: 5px; filter: drop-shadow(0 0 12px #06B6D4) drop-shadow(0 0 24px rgba(6,182,212,0.4));"/> -->
64
69
 
65
70
  # <img src="https://github.com/Zongwei9888/Experiment_Images/raw/43c585dca3d21b8e4b6390d835cdd34dc4b4b23d/DeepCode_images/title_logo.svg" alt="DeepCode Logo" width="32" height="32" style="vertical-align: middle; margin-right: 8px;"/> DeepCode: Open Agentic Coding
@@ -92,6 +97,15 @@ Dynamic: summary
92
97
  </a>
93
98
  </div>
94
99
 
100
+ <div align="center" style="margin-top: 10px;">
101
+ <a href="README.md">
102
+ <img src="https://img.shields.io/badge/English-00d4ff?style=for-the-badge&logo=readme&logoColor=white&labelColor=1a1a2e" alt="English">
103
+ </a>
104
+ <a href="README_ZH.md">
105
+ <img src="https://img.shields.io/badge/中文-00d4ff?style=for-the-badge&logo=readme&logoColor=white&labelColor=1a1a2e" alt="中文">
106
+ </a>
107
+ </div>
108
+
95
109
  ### 🖥️ **Interface Showcase**
96
110
 
97
111
  <table align="center" width="100%" style="border: none; border-collapse: collapse; margin: 30px 0;">
@@ -173,14 +187,30 @@ Dynamic: summary
173
187
 
174
188
  ## 📑 Table of Contents
175
189
 
190
+ - [📰 News](#-news)
176
191
  - [🚀 Key Features](#-key-features)
177
192
  - [🏗️ Architecture](#️-architecture)
193
+ - [📊 Experimental Results](#-experimental-results)
178
194
  - [🚀 Quick Start](#-quick-start)
179
195
  - [💡 Examples](#-examples)
180
196
  - [🎬 Live Demonstrations](#-live-demonstrations)
181
197
  - [⭐ Star History](#-star-history)
182
198
  - [📄 License](#-license)
183
199
 
200
+
201
+ ---
202
+
203
+ ## 📰 News
204
+
205
+ 🎉 **[2025-10] 🎉 [2025-10-28] DeepCode Achieves SOTA on PaperBench!**
206
+
207
+ DeepCode sets new benchmarks on OpenAI's PaperBench Code-Dev across all categories:
208
+
209
+ - 🏆 **Surpasses Human Experts**: **75.9%** (DeepCode) vs Top Machine Learning PhDs 72.4% (+3.5%).
210
+ - 🥇 **Outperforms SOTA Commercial Code Agents**: **84.8%** (DeepCode) vs Leading Commercial Code Agents (+26.1%) (Cursor, Claude Code, and Codex).
211
+ - 🔬 **Advances Scientific Coding**: **73.5%** (DeepCode) vs PaperCoder 51.1% (+22.4%).
212
+ - 🚀 **Beats LLM Agents**: **73.5%** (DeepCode) vs best LLM frameworks 43.3% (+30.2%).
213
+
184
214
  ---
185
215
 
186
216
  ## 🚀 Key Features
@@ -257,7 +287,58 @@ Dynamic: summary
257
287
 
258
288
  <br/>
259
289
 
260
- ### 🎯 **Autonomous Multi-Agent Workflow**
290
+ ---
291
+
292
+ ## 📊 Experimental Results
293
+
294
+ <div align="center">
295
+ <img src='./assets/result_main02.jpg' /><br>
296
+ </div>
297
+ <br/>
298
+
299
+ We evaluate **DeepCode** on the [*PaperBench*](https://openai.com/index/paperbench/) benchmark (released by OpenAI), a rigorous testbed requiring AI agents to independently reproduce 20 ICML 2024 papers from scratch. The benchmark comprises 8,316 gradable components assessed using SimpleJudge with hierarchical weighting.
300
+
301
+ Our experiments compare DeepCode against four baseline categories: **(1) Human Experts**, **(2) State-of-the-Art Commercial Code Agents**, **(3) Scientific Code Agents**, and **(4) LLM-Based Agents**.
302
+
303
+ ### ① 🧠 Human Expert Performance (Top Machine Learning PhD)
304
+
305
+ **DeepCode: 75.9% vs. Top Machine Learning PhD: 72.4% (+3.5%)**
306
+
307
+ DeepCode achieves **75.9%** on the 3-paper human evaluation subset, **surpassing the best-of-3 human expert baseline (72.4%) by +3.5 percentage points**. This demonstrates that our framework not only matches but exceeds expert-level code reproduction capabilities, representing a significant milestone in autonomous scientific software engineering.
308
+
309
+ ### ② 💼 State-of-the-Art Commercial Code Agents
310
+
311
+ **DeepCode: 84.8% vs. Best Commercial Agent: 58.7% (+26.1%)**
312
+
313
+ On the 5-paper subset, DeepCode substantially outperforms leading commercial coding tools:
314
+ - Cursor: 58.4%
315
+ - Claude Code: 58.7%
316
+ - Codex: 40.0%
317
+ - **DeepCode: 84.8%**
318
+
319
+ This represents a **+26.1% improvement** over the leading commercial code agent. All commercial agents utilize Claude Sonnet 4.5 or GPT-5 Codex-high, highlighting that **DeepCode's superior architecture**—rather than base model capability—drives this performance gap.
320
+
321
+ ### ③ 🔬 Scientific Code Agents
322
+
323
+ **DeepCode: 73.5% vs. PaperCoder: 51.1% (+22.4%)**
324
+
325
+ Compared to PaperCoder (**51.1%**), the state-of-the-art scientific code reproduction framework, DeepCode achieves **73.5%**, demonstrating a **+22.4% relative improvement**. This substantial margin validates our multi-module architecture combining planning, hierarchical task decomposition, code generation, and iterative debugging over simpler pipeline-based approaches.
326
+
327
+ ### ④ 🤖 LLM-Based Agents
328
+
329
+ **DeepCode: 73.5% vs. Best LLM Agent: 43.3% (+30.2%)**
330
+
331
+ DeepCode significantly outperforms all tested LLM agents:
332
+ - Claude 3.5 Sonnet + IterativeAgent: 27.5%
333
+ - o1 + IterativeAgent (36 hours): 42.4%
334
+ - o1 BasicAgent: 43.3%
335
+ - **DeepCode: 73.5%**
336
+
337
+ The **+30.2% improvement** over the best-performing LLM agent demonstrates that sophisticated agent scaffolding, rather than extended inference time or larger models, is critical for complex code reproduction tasks.
338
+
339
+ ---
340
+
341
+ ### 🎯 **Autonomous Self-Orchestrating Multi-Agent Architecture**
261
342
 
262
343
  **The Challenges**:
263
344
 
@@ -485,6 +566,7 @@ Implementation Generation • Testing • Documentation
485
566
 
486
567
  ---
487
568
 
569
+
488
570
  ## 🚀 Quick Start
489
571
 
490
572
 
@@ -16,6 +16,10 @@
16
16
  </tr>
17
17
  </table>
18
18
 
19
+ <div align="center">
20
+ <a href="https://trendshift.io/repositories/14665" target="_blank"><img src="https://trendshift.io/api/badge/repositories/14665" alt="HKUDS%2FDeepCode | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
21
+ </div>
22
+
19
23
  <!-- <img src="https://readme-typing-svg.herokuapp.com?font=Russo+One&size=28&duration=2000&pause=800&color=06B6D4&background=00000000&center=true&vCenter=true&width=800&height=50&lines=%E2%9A%A1+OPEN+AGENTIC+CODING+%E2%9A%A1" alt="DeepCode Tech Subtitle" style="margin-top: 5px; filter: drop-shadow(0 0 12px #06B6D4) drop-shadow(0 0 24px rgba(6,182,212,0.4));"/> -->
20
24
 
21
25
  # <img src="https://github.com/Zongwei9888/Experiment_Images/raw/43c585dca3d21b8e4b6390d835cdd34dc4b4b23d/DeepCode_images/title_logo.svg" alt="DeepCode Logo" width="32" height="32" style="vertical-align: middle; margin-right: 8px;"/> DeepCode: Open Agentic Coding
@@ -48,6 +52,15 @@
48
52
  </a>
49
53
  </div>
50
54
 
55
+ <div align="center" style="margin-top: 10px;">
56
+ <a href="README.md">
57
+ <img src="https://img.shields.io/badge/English-00d4ff?style=for-the-badge&logo=readme&logoColor=white&labelColor=1a1a2e" alt="English">
58
+ </a>
59
+ <a href="README_ZH.md">
60
+ <img src="https://img.shields.io/badge/中文-00d4ff?style=for-the-badge&logo=readme&logoColor=white&labelColor=1a1a2e" alt="中文">
61
+ </a>
62
+ </div>
63
+
51
64
  ### 🖥️ **Interface Showcase**
52
65
 
53
66
  <table align="center" width="100%" style="border: none; border-collapse: collapse; margin: 30px 0;">
@@ -129,14 +142,30 @@
129
142
 
130
143
  ## 📑 Table of Contents
131
144
 
145
+ - [📰 News](#-news)
132
146
  - [🚀 Key Features](#-key-features)
133
147
  - [🏗️ Architecture](#️-architecture)
148
+ - [📊 Experimental Results](#-experimental-results)
134
149
  - [🚀 Quick Start](#-quick-start)
135
150
  - [💡 Examples](#-examples)
136
151
  - [🎬 Live Demonstrations](#-live-demonstrations)
137
152
  - [⭐ Star History](#-star-history)
138
153
  - [📄 License](#-license)
139
154
 
155
+
156
+ ---
157
+
158
+ ## 📰 News
159
+
160
+ 🎉 **[2025-10] 🎉 [2025-10-28] DeepCode Achieves SOTA on PaperBench!**
161
+
162
+ DeepCode sets new benchmarks on OpenAI's PaperBench Code-Dev across all categories:
163
+
164
+ - 🏆 **Surpasses Human Experts**: **75.9%** (DeepCode) vs Top Machine Learning PhDs 72.4% (+3.5%).
165
+ - 🥇 **Outperforms SOTA Commercial Code Agents**: **84.8%** (DeepCode) vs Leading Commercial Code Agents (+26.1%) (Cursor, Claude Code, and Codex).
166
+ - 🔬 **Advances Scientific Coding**: **73.5%** (DeepCode) vs PaperCoder 51.1% (+22.4%).
167
+ - 🚀 **Beats LLM Agents**: **73.5%** (DeepCode) vs best LLM frameworks 43.3% (+30.2%).
168
+
140
169
  ---
141
170
 
142
171
  ## 🚀 Key Features
@@ -213,7 +242,58 @@
213
242
 
214
243
  <br/>
215
244
 
216
- ### 🎯 **Autonomous Multi-Agent Workflow**
245
+ ---
246
+
247
+ ## 📊 Experimental Results
248
+
249
+ <div align="center">
250
+ <img src='./assets/result_main02.jpg' /><br>
251
+ </div>
252
+ <br/>
253
+
254
+ We evaluate **DeepCode** on the [*PaperBench*](https://openai.com/index/paperbench/) benchmark (released by OpenAI), a rigorous testbed requiring AI agents to independently reproduce 20 ICML 2024 papers from scratch. The benchmark comprises 8,316 gradable components assessed using SimpleJudge with hierarchical weighting.
255
+
256
+ Our experiments compare DeepCode against four baseline categories: **(1) Human Experts**, **(2) State-of-the-Art Commercial Code Agents**, **(3) Scientific Code Agents**, and **(4) LLM-Based Agents**.
257
+
258
+ ### ① 🧠 Human Expert Performance (Top Machine Learning PhD)
259
+
260
+ **DeepCode: 75.9% vs. Top Machine Learning PhD: 72.4% (+3.5%)**
261
+
262
+ DeepCode achieves **75.9%** on the 3-paper human evaluation subset, **surpassing the best-of-3 human expert baseline (72.4%) by +3.5 percentage points**. This demonstrates that our framework not only matches but exceeds expert-level code reproduction capabilities, representing a significant milestone in autonomous scientific software engineering.
263
+
264
+ ### ② 💼 State-of-the-Art Commercial Code Agents
265
+
266
+ **DeepCode: 84.8% vs. Best Commercial Agent: 58.7% (+26.1%)**
267
+
268
+ On the 5-paper subset, DeepCode substantially outperforms leading commercial coding tools:
269
+ - Cursor: 58.4%
270
+ - Claude Code: 58.7%
271
+ - Codex: 40.0%
272
+ - **DeepCode: 84.8%**
273
+
274
+ This represents a **+26.1% improvement** over the leading commercial code agent. All commercial agents utilize Claude Sonnet 4.5 or GPT-5 Codex-high, highlighting that **DeepCode's superior architecture**—rather than base model capability—drives this performance gap.
275
+
276
+ ### ③ 🔬 Scientific Code Agents
277
+
278
+ **DeepCode: 73.5% vs. PaperCoder: 51.1% (+22.4%)**
279
+
280
+ Compared to PaperCoder (**51.1%**), the state-of-the-art scientific code reproduction framework, DeepCode achieves **73.5%**, demonstrating a **+22.4% relative improvement**. This substantial margin validates our multi-module architecture combining planning, hierarchical task decomposition, code generation, and iterative debugging over simpler pipeline-based approaches.
281
+
282
+ ### ④ 🤖 LLM-Based Agents
283
+
284
+ **DeepCode: 73.5% vs. Best LLM Agent: 43.3% (+30.2%)**
285
+
286
+ DeepCode significantly outperforms all tested LLM agents:
287
+ - Claude 3.5 Sonnet + IterativeAgent: 27.5%
288
+ - o1 + IterativeAgent (36 hours): 42.4%
289
+ - o1 BasicAgent: 43.3%
290
+ - **DeepCode: 73.5%**
291
+
292
+ The **+30.2% improvement** over the best-performing LLM agent demonstrates that sophisticated agent scaffolding, rather than extended inference time or larger models, is critical for complex code reproduction tasks.
293
+
294
+ ---
295
+
296
+ ### 🎯 **Autonomous Self-Orchestrating Multi-Agent Architecture**
217
297
 
218
298
  **The Challenges**:
219
299
 
@@ -441,6 +521,7 @@ Implementation Generation • Testing • Documentation
441
521
 
442
522
  ---
443
523
 
524
+
444
525
  ## 🚀 Quick Start
445
526
 
446
527
 
@@ -5,7 +5,7 @@ DeepCode - AI Research Engine
5
5
  ⚡ Transform research papers into working code automatically
6
6
  """
7
7
 
8
- __version__ = "1.0.4"
8
+ __version__ = "1.0.6"
9
9
  __author__ = "DeepCode Team"
10
10
  __url__ = "https://github.com/HKUDS/DeepCode"
11
11
 
@@ -37,8 +37,7 @@ class CLIApp:
37
37
  self.app = None # Will be initialized by workflow adapter
38
38
  self.logger = None
39
39
  self.context = None
40
- # Document segmentation configuration
41
- self.segmentation_config = {"enabled": True, "size_threshold_chars": 50000}
40
+ # Document segmentation will be managed by CLI interface
42
41
 
43
42
  async def initialize_mcp_app(self):
44
43
  """初始化MCP应用 - 使用工作流适配器"""
@@ -49,50 +48,10 @@ class CLIApp:
49
48
  """清理MCP应用 - 使用工作流适配器"""
50
49
  await self.workflow_adapter.cleanup_mcp_app()
51
50
 
52
- def update_segmentation_config(self):
53
- """Update document segmentation configuration in mcp_agent.config.yaml"""
54
- import yaml
55
- import os
56
-
57
- config_path = os.path.join(
58
- os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
59
- "mcp_agent.config.yaml",
60
- )
61
-
62
- try:
63
- # Read current config
64
- with open(config_path, "r", encoding="utf-8") as f:
65
- config = yaml.safe_load(f)
66
-
67
- # Update document segmentation settings
68
- if "document_segmentation" not in config:
69
- config["document_segmentation"] = {}
70
-
71
- config["document_segmentation"]["enabled"] = self.segmentation_config[
72
- "enabled"
73
- ]
74
- config["document_segmentation"]["size_threshold_chars"] = (
75
- self.segmentation_config["size_threshold_chars"]
76
- )
77
-
78
- # Write updated config
79
- with open(config_path, "w", encoding="utf-8") as f:
80
- yaml.dump(config, f, default_flow_style=False, allow_unicode=True)
81
-
82
- self.cli.print_status(
83
- "📄 Document segmentation configuration updated", "success"
84
- )
85
-
86
- except Exception as e:
87
- self.cli.print_status(
88
- f"⚠️ Failed to update segmentation config: {str(e)}", "warning"
89
- )
90
-
91
51
  async def process_input(self, input_source: str, input_type: str):
92
52
  """处理输入源(URL或文件)- 使用升级版智能体编排引擎"""
93
53
  try:
94
- # Update segmentation configuration before processing
95
- self.update_segmentation_config()
54
+ # Document segmentation configuration is managed by CLI interface
96
55
 
97
56
  self.cli.print_separator()
98
57
  self.cli.print_status(
@@ -281,20 +240,9 @@ class CLIApp:
281
240
  self.cli.show_history()
282
241
 
283
242
  elif choice in ["c", "config", "configure"]:
284
- # Sync current segmentation config from CLI interface
285
- self.segmentation_config["enabled"] = self.cli.segmentation_enabled
286
- self.segmentation_config["size_threshold_chars"] = (
287
- self.cli.segmentation_threshold
288
- )
289
-
243
+ # Show configuration menu - all settings managed by CLI interface
290
244
  self.cli.show_configuration_menu()
291
245
 
292
- # Sync back from CLI interface after configuration changes
293
- self.segmentation_config["enabled"] = self.cli.segmentation_enabled
294
- self.segmentation_config["size_threshold_chars"] = (
295
- self.cli.segmentation_threshold
296
- )
297
-
298
246
  else:
299
247
  self.cli.print_status(
300
248
  "Invalid choice. Please select U, F, T, C, H, or Q.", "warning"
@@ -39,10 +39,70 @@ class CLIInterface:
39
39
  self.uploaded_file = None
40
40
  self.is_running = True
41
41
  self.processing_history = []
42
- self.enable_indexing = True # Default configuration
43
- self.segmentation_enabled = True # Default to smart segmentation
44
- self.segmentation_threshold = 50000 # Default threshold
42
+ self.enable_indexing = (
43
+ False # Default configuration (matching UI: fast mode by default)
44
+ )
45
+
46
+ # Load segmentation config from the same source as UI
47
+ self._load_segmentation_config()
48
+
49
+ # Initialize tkinter availability
50
+ self._init_tkinter()
51
+
52
+ def _load_segmentation_config(self):
53
+ """Load segmentation configuration from mcp_agent.config.yaml"""
54
+ try:
55
+ from utils.llm_utils import get_document_segmentation_config
56
+
57
+ seg_config = get_document_segmentation_config()
58
+ self.segmentation_enabled = seg_config.get("enabled", True)
59
+ self.segmentation_threshold = seg_config.get("size_threshold_chars", 50000)
60
+ except Exception as e:
61
+ print(f"⚠️ Warning: Failed to load segmentation config: {e}")
62
+ # Fall back to defaults
63
+ self.segmentation_enabled = True
64
+ self.segmentation_threshold = 50000
65
+
66
+ def _save_segmentation_config(self):
67
+ """Save segmentation configuration to mcp_agent.config.yaml"""
68
+ import yaml
69
+ import os
70
+
71
+ # Get the project root directory (where mcp_agent.config.yaml is located)
72
+ current_file = os.path.abspath(__file__)
73
+ cli_dir = os.path.dirname(current_file) # cli directory
74
+ project_root = os.path.dirname(cli_dir) # project root
75
+ config_path = os.path.join(project_root, "mcp_agent.config.yaml")
76
+
77
+ try:
78
+ # Read current config
79
+ with open(config_path, "r", encoding="utf-8") as f:
80
+ config = yaml.safe_load(f)
81
+
82
+ # Update document segmentation settings
83
+ if "document_segmentation" not in config:
84
+ config["document_segmentation"] = {}
85
+
86
+ config["document_segmentation"]["enabled"] = self.segmentation_enabled
87
+ config["document_segmentation"]["size_threshold_chars"] = (
88
+ self.segmentation_threshold
89
+ )
90
+
91
+ # Write updated config
92
+ with open(config_path, "w", encoding="utf-8") as f:
93
+ yaml.dump(config, f, default_flow_style=False, allow_unicode=True)
94
+
95
+ print(
96
+ f"{Colors.OKGREEN}✅ Document segmentation configuration updated{Colors.ENDC}"
97
+ )
98
+
99
+ except Exception as e:
100
+ print(
101
+ f"{Colors.WARNING}⚠️ Failed to update segmentation config: {str(e)}{Colors.ENDC}"
102
+ )
45
103
 
104
+ def _init_tkinter(self):
105
+ """Initialize tkinter availability check"""
46
106
  # Check tkinter availability for file dialogs
47
107
  self.tkinter_available = True
48
108
  try:
@@ -765,6 +825,8 @@ class CLIInterface:
765
825
  elif choice in ["s", "segmentation"]:
766
826
  current_state = getattr(self, "segmentation_enabled", True)
767
827
  self.segmentation_enabled = not current_state
828
+ # Save the configuration to file
829
+ self._save_segmentation_config()
768
830
  seg_mode = (
769
831
  "📄 Smart Segmentation"
770
832
  if self.segmentation_enabled
@@ -214,15 +214,17 @@ async def main():
214
214
  # 创建CLI应用
215
215
  app = CLIApp()
216
216
 
217
- # 设置配置
217
+ # 设置配置 - 默认禁用索引功能以加快处理速度
218
218
  if args.optimized:
219
219
  app.cli.enable_indexing = False
220
220
  print(
221
221
  f"\n{Colors.YELLOW}⚡ Optimized mode enabled - indexing disabled{Colors.ENDC}"
222
222
  )
223
223
  else:
224
+ # 默认也禁用索引功能
225
+ app.cli.enable_indexing = False
224
226
  print(
225
- f"\n{Colors.GREEN}🧠 Comprehensive mode enabled - full intelligence analysis{Colors.ENDC}"
227
+ f"\n{Colors.YELLOW}⚡ Fast mode enabled - indexing disabled by default{Colors.ENDC}"
226
228
  )
227
229
 
228
230
  # Configure document segmentation settings
@@ -230,18 +232,16 @@ async def main():
230
232
  print(
231
233
  f"\n{Colors.MAGENTA}📄 Document segmentation disabled - using traditional processing{Colors.ENDC}"
232
234
  )
233
- app.segmentation_config = {
234
- "enabled": False,
235
- "size_threshold_chars": args.segmentation_threshold,
236
- }
235
+ app.cli.segmentation_enabled = False
236
+ app.cli.segmentation_threshold = args.segmentation_threshold
237
+ app.cli._save_segmentation_config()
237
238
  else:
238
239
  print(
239
240
  f"\n{Colors.BLUE}📄 Smart document segmentation enabled (threshold: {args.segmentation_threshold} chars){Colors.ENDC}"
240
241
  )
241
- app.segmentation_config = {
242
- "enabled": True,
243
- "size_threshold_chars": args.segmentation_threshold,
244
- }
242
+ app.cli.segmentation_enabled = True
243
+ app.cli.segmentation_threshold = args.segmentation_threshold
244
+ app.cli._save_segmentation_config()
245
245
 
246
246
  # 检查是否为直接处理模式
247
247
  if args.file or args.url or args.chat:
@@ -250,7 +250,9 @@ async def main():
250
250
  if not os.path.exists(args.file):
251
251
  print(f"{Colors.FAIL}❌ File not found: {args.file}{Colors.ENDC}")
252
252
  sys.exit(1)
253
- success = await run_direct_processing(app, args.file, "file")
253
+ # 使用 file:// 前缀保持与交互模式一致,确保文件被复制而非移动
254
+ file_url = f"file://{os.path.abspath(args.file)}"
255
+ success = await run_direct_processing(app, file_url, "file")
254
256
  elif args.url:
255
257
  success = await run_direct_processing(app, args.url, "url")
256
258
  elif args.chat: