deepcode-hku 1.0.4__tar.gz → 1.0.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deepcode_hku-1.0.4/deepcode_hku.egg-info → deepcode_hku-1.0.6}/PKG-INFO +84 -2
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/README.md +82 -1
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/__init__.py +1 -1
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/cli_app.py +3 -55
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/cli_interface.py +65 -3
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/main_cli.py +13 -11
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/workflows/cli_workflow_adapter.py +101 -58
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode.py +74 -1
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6/deepcode_hku.egg-info}/PKG-INFO +84 -2
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/SOURCES.txt +4 -1
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/requires.txt +1 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/mcp_agent.config.yaml +30 -12
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/mcp_agent.secrets.yaml +4 -2
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/prompts/code_prompts.py +122 -40
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/requirements.txt +1 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/schema/mcp-agent.config.schema.json +0 -34
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/code_implementation_server.py +432 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/pdf_downloader.py +48 -24
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/components.py +548 -377
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/handlers.py +234 -5
- deepcode_hku-1.0.6/ui/layout.py +161 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/styles.py +1390 -71
- deepcode_hku-1.0.6/utils/cross_platform_file_handler.py +475 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/file_processor.py +12 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/llm_utils.py +6 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agent_orchestration_engine.py +190 -13
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/memory_agent_concise.py +676 -151
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/memory_agent_concise_index.py +724 -180
- deepcode_hku-1.0.6/workflows/agents/memory_agent_concise_multi.py +1659 -0
- deepcode_hku-1.0.6/workflows/agents/requirement_analysis_agent.py +410 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/code_implementation_workflow.py +384 -51
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/code_implementation_workflow_index.py +371 -41
- deepcode_hku-1.0.4/ui/layout.py +0 -106
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/.pre-commit-config.yaml +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/LICENSE +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/MANIFEST.in +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/__init__.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/cli_launcher.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/cli/workflows/__init__.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/dependency_links.txt +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/entry_points.txt +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/deepcode_hku.egg-info/top_level.txt +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/setup.cfg +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/setup.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/__init__.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/bocha_search_server.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/code_indexer.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/code_reference_indexer.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/command_executor.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/document_segmentation_server.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/git_command.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/pdf_converter.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/tools/pdf_utils.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/__init__.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/app.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/ui/streamlit_app.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/__init__.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/cli_interface.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/dialogue_logger.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/utils/simple_llm_logger.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/__init__.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/__init__.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/code_implementation_agent.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/agents/document_segmentation_agent.py +0 -0
- {deepcode_hku-1.0.4 → deepcode_hku-1.0.6}/workflows/codebase_index_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deepcode-hku
|
|
3
|
-
Version: 1.0.
|
|
3
|
+
Version: 1.0.6
|
|
4
4
|
Summary: AI Research Engine - Transform research papers into working code automatically
|
|
5
5
|
Home-page: https://github.com/HKUDS/DeepCode
|
|
6
6
|
Author: DeepCodeTeam
|
|
@@ -27,6 +27,7 @@ Requires-Dist: docling
|
|
|
27
27
|
Requires-Dist: mcp-agent
|
|
28
28
|
Requires-Dist: mcp-server-git
|
|
29
29
|
Requires-Dist: nest_asyncio
|
|
30
|
+
Requires-Dist: openai
|
|
30
31
|
Requires-Dist: pathlib2
|
|
31
32
|
Requires-Dist: PyPDF2>=2.0.0
|
|
32
33
|
Requires-Dist: reportlab>=3.5.0
|
|
@@ -60,6 +61,10 @@ Dynamic: summary
|
|
|
60
61
|
</tr>
|
|
61
62
|
</table>
|
|
62
63
|
|
|
64
|
+
<div align="center">
|
|
65
|
+
<a href="https://trendshift.io/repositories/14665" target="_blank"><img src="https://trendshift.io/api/badge/repositories/14665" alt="HKUDS%2FDeepCode | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
|
66
|
+
</div>
|
|
67
|
+
|
|
63
68
|
<!-- <img src="https://readme-typing-svg.herokuapp.com?font=Russo+One&size=28&duration=2000&pause=800&color=06B6D4&background=00000000¢er=true&vCenter=true&width=800&height=50&lines=%E2%9A%A1+OPEN+AGENTIC+CODING+%E2%9A%A1" alt="DeepCode Tech Subtitle" style="margin-top: 5px; filter: drop-shadow(0 0 12px #06B6D4) drop-shadow(0 0 24px rgba(6,182,212,0.4));"/> -->
|
|
64
69
|
|
|
65
70
|
# <img src="https://github.com/Zongwei9888/Experiment_Images/raw/43c585dca3d21b8e4b6390d835cdd34dc4b4b23d/DeepCode_images/title_logo.svg" alt="DeepCode Logo" width="32" height="32" style="vertical-align: middle; margin-right: 8px;"/> DeepCode: Open Agentic Coding
|
|
@@ -92,6 +97,15 @@ Dynamic: summary
|
|
|
92
97
|
</a>
|
|
93
98
|
</div>
|
|
94
99
|
|
|
100
|
+
<div align="center" style="margin-top: 10px;">
|
|
101
|
+
<a href="README.md">
|
|
102
|
+
<img src="https://img.shields.io/badge/English-00d4ff?style=for-the-badge&logo=readme&logoColor=white&labelColor=1a1a2e" alt="English">
|
|
103
|
+
</a>
|
|
104
|
+
<a href="README_ZH.md">
|
|
105
|
+
<img src="https://img.shields.io/badge/中文-00d4ff?style=for-the-badge&logo=readme&logoColor=white&labelColor=1a1a2e" alt="中文">
|
|
106
|
+
</a>
|
|
107
|
+
</div>
|
|
108
|
+
|
|
95
109
|
### 🖥️ **Interface Showcase**
|
|
96
110
|
|
|
97
111
|
<table align="center" width="100%" style="border: none; border-collapse: collapse; margin: 30px 0;">
|
|
@@ -173,14 +187,30 @@ Dynamic: summary
|
|
|
173
187
|
|
|
174
188
|
## 📑 Table of Contents
|
|
175
189
|
|
|
190
|
+
- [📰 News](#-news)
|
|
176
191
|
- [🚀 Key Features](#-key-features)
|
|
177
192
|
- [🏗️ Architecture](#️-architecture)
|
|
193
|
+
- [📊 Experimental Results](#-experimental-results)
|
|
178
194
|
- [🚀 Quick Start](#-quick-start)
|
|
179
195
|
- [💡 Examples](#-examples)
|
|
180
196
|
- [🎬 Live Demonstrations](#-live-demonstrations)
|
|
181
197
|
- [⭐ Star History](#-star-history)
|
|
182
198
|
- [📄 License](#-license)
|
|
183
199
|
|
|
200
|
+
|
|
201
|
+
---
|
|
202
|
+
|
|
203
|
+
## 📰 News
|
|
204
|
+
|
|
205
|
+
🎉 **[2025-10] 🎉 [2025-10-28] DeepCode Achieves SOTA on PaperBench!**
|
|
206
|
+
|
|
207
|
+
DeepCode sets new benchmarks on OpenAI's PaperBench Code-Dev across all categories:
|
|
208
|
+
|
|
209
|
+
- 🏆 **Surpasses Human Experts**: **75.9%** (DeepCode) vs Top Machine Learning PhDs 72.4% (+3.5%).
|
|
210
|
+
- 🥇 **Outperforms SOTA Commercial Code Agents**: **84.8%** (DeepCode) vs Leading Commercial Code Agents (+26.1%) (Cursor, Claude Code, and Codex).
|
|
211
|
+
- 🔬 **Advances Scientific Coding**: **73.5%** (DeepCode) vs PaperCoder 51.1% (+22.4%).
|
|
212
|
+
- 🚀 **Beats LLM Agents**: **73.5%** (DeepCode) vs best LLM frameworks 43.3% (+30.2%).
|
|
213
|
+
|
|
184
214
|
---
|
|
185
215
|
|
|
186
216
|
## 🚀 Key Features
|
|
@@ -257,7 +287,58 @@ Dynamic: summary
|
|
|
257
287
|
|
|
258
288
|
<br/>
|
|
259
289
|
|
|
260
|
-
|
|
290
|
+
---
|
|
291
|
+
|
|
292
|
+
## 📊 Experimental Results
|
|
293
|
+
|
|
294
|
+
<div align="center">
|
|
295
|
+
<img src='./assets/result_main02.jpg' /><br>
|
|
296
|
+
</div>
|
|
297
|
+
<br/>
|
|
298
|
+
|
|
299
|
+
We evaluate **DeepCode** on the [*PaperBench*](https://openai.com/index/paperbench/) benchmark (released by OpenAI), a rigorous testbed requiring AI agents to independently reproduce 20 ICML 2024 papers from scratch. The benchmark comprises 8,316 gradable components assessed using SimpleJudge with hierarchical weighting.
|
|
300
|
+
|
|
301
|
+
Our experiments compare DeepCode against four baseline categories: **(1) Human Experts**, **(2) State-of-the-Art Commercial Code Agents**, **(3) Scientific Code Agents**, and **(4) LLM-Based Agents**.
|
|
302
|
+
|
|
303
|
+
### ① 🧠 Human Expert Performance (Top Machine Learning PhD)
|
|
304
|
+
|
|
305
|
+
**DeepCode: 75.9% vs. Top Machine Learning PhD: 72.4% (+3.5%)**
|
|
306
|
+
|
|
307
|
+
DeepCode achieves **75.9%** on the 3-paper human evaluation subset, **surpassing the best-of-3 human expert baseline (72.4%) by +3.5 percentage points**. This demonstrates that our framework not only matches but exceeds expert-level code reproduction capabilities, representing a significant milestone in autonomous scientific software engineering.
|
|
308
|
+
|
|
309
|
+
### ② 💼 State-of-the-Art Commercial Code Agents
|
|
310
|
+
|
|
311
|
+
**DeepCode: 84.8% vs. Best Commercial Agent: 58.7% (+26.1%)**
|
|
312
|
+
|
|
313
|
+
On the 5-paper subset, DeepCode substantially outperforms leading commercial coding tools:
|
|
314
|
+
- Cursor: 58.4%
|
|
315
|
+
- Claude Code: 58.7%
|
|
316
|
+
- Codex: 40.0%
|
|
317
|
+
- **DeepCode: 84.8%**
|
|
318
|
+
|
|
319
|
+
This represents a **+26.1% improvement** over the leading commercial code agent. All commercial agents utilize Claude Sonnet 4.5 or GPT-5 Codex-high, highlighting that **DeepCode's superior architecture**—rather than base model capability—drives this performance gap.
|
|
320
|
+
|
|
321
|
+
### ③ 🔬 Scientific Code Agents
|
|
322
|
+
|
|
323
|
+
**DeepCode: 73.5% vs. PaperCoder: 51.1% (+22.4%)**
|
|
324
|
+
|
|
325
|
+
Compared to PaperCoder (**51.1%**), the state-of-the-art scientific code reproduction framework, DeepCode achieves **73.5%**, demonstrating a **+22.4% relative improvement**. This substantial margin validates our multi-module architecture combining planning, hierarchical task decomposition, code generation, and iterative debugging over simpler pipeline-based approaches.
|
|
326
|
+
|
|
327
|
+
### ④ 🤖 LLM-Based Agents
|
|
328
|
+
|
|
329
|
+
**DeepCode: 73.5% vs. Best LLM Agent: 43.3% (+30.2%)**
|
|
330
|
+
|
|
331
|
+
DeepCode significantly outperforms all tested LLM agents:
|
|
332
|
+
- Claude 3.5 Sonnet + IterativeAgent: 27.5%
|
|
333
|
+
- o1 + IterativeAgent (36 hours): 42.4%
|
|
334
|
+
- o1 BasicAgent: 43.3%
|
|
335
|
+
- **DeepCode: 73.5%**
|
|
336
|
+
|
|
337
|
+
The **+30.2% improvement** over the best-performing LLM agent demonstrates that sophisticated agent scaffolding, rather than extended inference time or larger models, is critical for complex code reproduction tasks.
|
|
338
|
+
|
|
339
|
+
---
|
|
340
|
+
|
|
341
|
+
### 🎯 **Autonomous Self-Orchestrating Multi-Agent Architecture**
|
|
261
342
|
|
|
262
343
|
**The Challenges**:
|
|
263
344
|
|
|
@@ -485,6 +566,7 @@ Implementation Generation • Testing • Documentation
|
|
|
485
566
|
|
|
486
567
|
---
|
|
487
568
|
|
|
569
|
+
|
|
488
570
|
## 🚀 Quick Start
|
|
489
571
|
|
|
490
572
|
|
|
@@ -16,6 +16,10 @@
|
|
|
16
16
|
</tr>
|
|
17
17
|
</table>
|
|
18
18
|
|
|
19
|
+
<div align="center">
|
|
20
|
+
<a href="https://trendshift.io/repositories/14665" target="_blank"><img src="https://trendshift.io/api/badge/repositories/14665" alt="HKUDS%2FDeepCode | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
|
21
|
+
</div>
|
|
22
|
+
|
|
19
23
|
<!-- <img src="https://readme-typing-svg.herokuapp.com?font=Russo+One&size=28&duration=2000&pause=800&color=06B6D4&background=00000000¢er=true&vCenter=true&width=800&height=50&lines=%E2%9A%A1+OPEN+AGENTIC+CODING+%E2%9A%A1" alt="DeepCode Tech Subtitle" style="margin-top: 5px; filter: drop-shadow(0 0 12px #06B6D4) drop-shadow(0 0 24px rgba(6,182,212,0.4));"/> -->
|
|
20
24
|
|
|
21
25
|
# <img src="https://github.com/Zongwei9888/Experiment_Images/raw/43c585dca3d21b8e4b6390d835cdd34dc4b4b23d/DeepCode_images/title_logo.svg" alt="DeepCode Logo" width="32" height="32" style="vertical-align: middle; margin-right: 8px;"/> DeepCode: Open Agentic Coding
|
|
@@ -48,6 +52,15 @@
|
|
|
48
52
|
</a>
|
|
49
53
|
</div>
|
|
50
54
|
|
|
55
|
+
<div align="center" style="margin-top: 10px;">
|
|
56
|
+
<a href="README.md">
|
|
57
|
+
<img src="https://img.shields.io/badge/English-00d4ff?style=for-the-badge&logo=readme&logoColor=white&labelColor=1a1a2e" alt="English">
|
|
58
|
+
</a>
|
|
59
|
+
<a href="README_ZH.md">
|
|
60
|
+
<img src="https://img.shields.io/badge/中文-00d4ff?style=for-the-badge&logo=readme&logoColor=white&labelColor=1a1a2e" alt="中文">
|
|
61
|
+
</a>
|
|
62
|
+
</div>
|
|
63
|
+
|
|
51
64
|
### 🖥️ **Interface Showcase**
|
|
52
65
|
|
|
53
66
|
<table align="center" width="100%" style="border: none; border-collapse: collapse; margin: 30px 0;">
|
|
@@ -129,14 +142,30 @@
|
|
|
129
142
|
|
|
130
143
|
## 📑 Table of Contents
|
|
131
144
|
|
|
145
|
+
- [📰 News](#-news)
|
|
132
146
|
- [🚀 Key Features](#-key-features)
|
|
133
147
|
- [🏗️ Architecture](#️-architecture)
|
|
148
|
+
- [📊 Experimental Results](#-experimental-results)
|
|
134
149
|
- [🚀 Quick Start](#-quick-start)
|
|
135
150
|
- [💡 Examples](#-examples)
|
|
136
151
|
- [🎬 Live Demonstrations](#-live-demonstrations)
|
|
137
152
|
- [⭐ Star History](#-star-history)
|
|
138
153
|
- [📄 License](#-license)
|
|
139
154
|
|
|
155
|
+
|
|
156
|
+
---
|
|
157
|
+
|
|
158
|
+
## 📰 News
|
|
159
|
+
|
|
160
|
+
🎉 **[2025-10] 🎉 [2025-10-28] DeepCode Achieves SOTA on PaperBench!**
|
|
161
|
+
|
|
162
|
+
DeepCode sets new benchmarks on OpenAI's PaperBench Code-Dev across all categories:
|
|
163
|
+
|
|
164
|
+
- 🏆 **Surpasses Human Experts**: **75.9%** (DeepCode) vs Top Machine Learning PhDs 72.4% (+3.5%).
|
|
165
|
+
- 🥇 **Outperforms SOTA Commercial Code Agents**: **84.8%** (DeepCode) vs Leading Commercial Code Agents (+26.1%) (Cursor, Claude Code, and Codex).
|
|
166
|
+
- 🔬 **Advances Scientific Coding**: **73.5%** (DeepCode) vs PaperCoder 51.1% (+22.4%).
|
|
167
|
+
- 🚀 **Beats LLM Agents**: **73.5%** (DeepCode) vs best LLM frameworks 43.3% (+30.2%).
|
|
168
|
+
|
|
140
169
|
---
|
|
141
170
|
|
|
142
171
|
## 🚀 Key Features
|
|
@@ -213,7 +242,58 @@
|
|
|
213
242
|
|
|
214
243
|
<br/>
|
|
215
244
|
|
|
216
|
-
|
|
245
|
+
---
|
|
246
|
+
|
|
247
|
+
## 📊 Experimental Results
|
|
248
|
+
|
|
249
|
+
<div align="center">
|
|
250
|
+
<img src='./assets/result_main02.jpg' /><br>
|
|
251
|
+
</div>
|
|
252
|
+
<br/>
|
|
253
|
+
|
|
254
|
+
We evaluate **DeepCode** on the [*PaperBench*](https://openai.com/index/paperbench/) benchmark (released by OpenAI), a rigorous testbed requiring AI agents to independently reproduce 20 ICML 2024 papers from scratch. The benchmark comprises 8,316 gradable components assessed using SimpleJudge with hierarchical weighting.
|
|
255
|
+
|
|
256
|
+
Our experiments compare DeepCode against four baseline categories: **(1) Human Experts**, **(2) State-of-the-Art Commercial Code Agents**, **(3) Scientific Code Agents**, and **(4) LLM-Based Agents**.
|
|
257
|
+
|
|
258
|
+
### ① 🧠 Human Expert Performance (Top Machine Learning PhD)
|
|
259
|
+
|
|
260
|
+
**DeepCode: 75.9% vs. Top Machine Learning PhD: 72.4% (+3.5%)**
|
|
261
|
+
|
|
262
|
+
DeepCode achieves **75.9%** on the 3-paper human evaluation subset, **surpassing the best-of-3 human expert baseline (72.4%) by +3.5 percentage points**. This demonstrates that our framework not only matches but exceeds expert-level code reproduction capabilities, representing a significant milestone in autonomous scientific software engineering.
|
|
263
|
+
|
|
264
|
+
### ② 💼 State-of-the-Art Commercial Code Agents
|
|
265
|
+
|
|
266
|
+
**DeepCode: 84.8% vs. Best Commercial Agent: 58.7% (+26.1%)**
|
|
267
|
+
|
|
268
|
+
On the 5-paper subset, DeepCode substantially outperforms leading commercial coding tools:
|
|
269
|
+
- Cursor: 58.4%
|
|
270
|
+
- Claude Code: 58.7%
|
|
271
|
+
- Codex: 40.0%
|
|
272
|
+
- **DeepCode: 84.8%**
|
|
273
|
+
|
|
274
|
+
This represents a **+26.1% improvement** over the leading commercial code agent. All commercial agents utilize Claude Sonnet 4.5 or GPT-5 Codex-high, highlighting that **DeepCode's superior architecture**—rather than base model capability—drives this performance gap.
|
|
275
|
+
|
|
276
|
+
### ③ 🔬 Scientific Code Agents
|
|
277
|
+
|
|
278
|
+
**DeepCode: 73.5% vs. PaperCoder: 51.1% (+22.4%)**
|
|
279
|
+
|
|
280
|
+
Compared to PaperCoder (**51.1%**), the state-of-the-art scientific code reproduction framework, DeepCode achieves **73.5%**, demonstrating a **+22.4% relative improvement**. This substantial margin validates our multi-module architecture combining planning, hierarchical task decomposition, code generation, and iterative debugging over simpler pipeline-based approaches.
|
|
281
|
+
|
|
282
|
+
### ④ 🤖 LLM-Based Agents
|
|
283
|
+
|
|
284
|
+
**DeepCode: 73.5% vs. Best LLM Agent: 43.3% (+30.2%)**
|
|
285
|
+
|
|
286
|
+
DeepCode significantly outperforms all tested LLM agents:
|
|
287
|
+
- Claude 3.5 Sonnet + IterativeAgent: 27.5%
|
|
288
|
+
- o1 + IterativeAgent (36 hours): 42.4%
|
|
289
|
+
- o1 BasicAgent: 43.3%
|
|
290
|
+
- **DeepCode: 73.5%**
|
|
291
|
+
|
|
292
|
+
The **+30.2% improvement** over the best-performing LLM agent demonstrates that sophisticated agent scaffolding, rather than extended inference time or larger models, is critical for complex code reproduction tasks.
|
|
293
|
+
|
|
294
|
+
---
|
|
295
|
+
|
|
296
|
+
### 🎯 **Autonomous Self-Orchestrating Multi-Agent Architecture**
|
|
217
297
|
|
|
218
298
|
**The Challenges**:
|
|
219
299
|
|
|
@@ -441,6 +521,7 @@ Implementation Generation • Testing • Documentation
|
|
|
441
521
|
|
|
442
522
|
---
|
|
443
523
|
|
|
524
|
+
|
|
444
525
|
## 🚀 Quick Start
|
|
445
526
|
|
|
446
527
|
|
|
@@ -37,8 +37,7 @@ class CLIApp:
|
|
|
37
37
|
self.app = None # Will be initialized by workflow adapter
|
|
38
38
|
self.logger = None
|
|
39
39
|
self.context = None
|
|
40
|
-
# Document segmentation
|
|
41
|
-
self.segmentation_config = {"enabled": True, "size_threshold_chars": 50000}
|
|
40
|
+
# Document segmentation will be managed by CLI interface
|
|
42
41
|
|
|
43
42
|
async def initialize_mcp_app(self):
|
|
44
43
|
"""初始化MCP应用 - 使用工作流适配器"""
|
|
@@ -49,50 +48,10 @@ class CLIApp:
|
|
|
49
48
|
"""清理MCP应用 - 使用工作流适配器"""
|
|
50
49
|
await self.workflow_adapter.cleanup_mcp_app()
|
|
51
50
|
|
|
52
|
-
def update_segmentation_config(self):
|
|
53
|
-
"""Update document segmentation configuration in mcp_agent.config.yaml"""
|
|
54
|
-
import yaml
|
|
55
|
-
import os
|
|
56
|
-
|
|
57
|
-
config_path = os.path.join(
|
|
58
|
-
os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
|
|
59
|
-
"mcp_agent.config.yaml",
|
|
60
|
-
)
|
|
61
|
-
|
|
62
|
-
try:
|
|
63
|
-
# Read current config
|
|
64
|
-
with open(config_path, "r", encoding="utf-8") as f:
|
|
65
|
-
config = yaml.safe_load(f)
|
|
66
|
-
|
|
67
|
-
# Update document segmentation settings
|
|
68
|
-
if "document_segmentation" not in config:
|
|
69
|
-
config["document_segmentation"] = {}
|
|
70
|
-
|
|
71
|
-
config["document_segmentation"]["enabled"] = self.segmentation_config[
|
|
72
|
-
"enabled"
|
|
73
|
-
]
|
|
74
|
-
config["document_segmentation"]["size_threshold_chars"] = (
|
|
75
|
-
self.segmentation_config["size_threshold_chars"]
|
|
76
|
-
)
|
|
77
|
-
|
|
78
|
-
# Write updated config
|
|
79
|
-
with open(config_path, "w", encoding="utf-8") as f:
|
|
80
|
-
yaml.dump(config, f, default_flow_style=False, allow_unicode=True)
|
|
81
|
-
|
|
82
|
-
self.cli.print_status(
|
|
83
|
-
"📄 Document segmentation configuration updated", "success"
|
|
84
|
-
)
|
|
85
|
-
|
|
86
|
-
except Exception as e:
|
|
87
|
-
self.cli.print_status(
|
|
88
|
-
f"⚠️ Failed to update segmentation config: {str(e)}", "warning"
|
|
89
|
-
)
|
|
90
|
-
|
|
91
51
|
async def process_input(self, input_source: str, input_type: str):
|
|
92
52
|
"""处理输入源(URL或文件)- 使用升级版智能体编排引擎"""
|
|
93
53
|
try:
|
|
94
|
-
#
|
|
95
|
-
self.update_segmentation_config()
|
|
54
|
+
# Document segmentation configuration is managed by CLI interface
|
|
96
55
|
|
|
97
56
|
self.cli.print_separator()
|
|
98
57
|
self.cli.print_status(
|
|
@@ -281,20 +240,9 @@ class CLIApp:
|
|
|
281
240
|
self.cli.show_history()
|
|
282
241
|
|
|
283
242
|
elif choice in ["c", "config", "configure"]:
|
|
284
|
-
#
|
|
285
|
-
self.segmentation_config["enabled"] = self.cli.segmentation_enabled
|
|
286
|
-
self.segmentation_config["size_threshold_chars"] = (
|
|
287
|
-
self.cli.segmentation_threshold
|
|
288
|
-
)
|
|
289
|
-
|
|
243
|
+
# Show configuration menu - all settings managed by CLI interface
|
|
290
244
|
self.cli.show_configuration_menu()
|
|
291
245
|
|
|
292
|
-
# Sync back from CLI interface after configuration changes
|
|
293
|
-
self.segmentation_config["enabled"] = self.cli.segmentation_enabled
|
|
294
|
-
self.segmentation_config["size_threshold_chars"] = (
|
|
295
|
-
self.cli.segmentation_threshold
|
|
296
|
-
)
|
|
297
|
-
|
|
298
246
|
else:
|
|
299
247
|
self.cli.print_status(
|
|
300
248
|
"Invalid choice. Please select U, F, T, C, H, or Q.", "warning"
|
|
@@ -39,10 +39,70 @@ class CLIInterface:
|
|
|
39
39
|
self.uploaded_file = None
|
|
40
40
|
self.is_running = True
|
|
41
41
|
self.processing_history = []
|
|
42
|
-
self.enable_indexing =
|
|
43
|
-
|
|
44
|
-
|
|
42
|
+
self.enable_indexing = (
|
|
43
|
+
False # Default configuration (matching UI: fast mode by default)
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
# Load segmentation config from the same source as UI
|
|
47
|
+
self._load_segmentation_config()
|
|
48
|
+
|
|
49
|
+
# Initialize tkinter availability
|
|
50
|
+
self._init_tkinter()
|
|
51
|
+
|
|
52
|
+
def _load_segmentation_config(self):
|
|
53
|
+
"""Load segmentation configuration from mcp_agent.config.yaml"""
|
|
54
|
+
try:
|
|
55
|
+
from utils.llm_utils import get_document_segmentation_config
|
|
56
|
+
|
|
57
|
+
seg_config = get_document_segmentation_config()
|
|
58
|
+
self.segmentation_enabled = seg_config.get("enabled", True)
|
|
59
|
+
self.segmentation_threshold = seg_config.get("size_threshold_chars", 50000)
|
|
60
|
+
except Exception as e:
|
|
61
|
+
print(f"⚠️ Warning: Failed to load segmentation config: {e}")
|
|
62
|
+
# Fall back to defaults
|
|
63
|
+
self.segmentation_enabled = True
|
|
64
|
+
self.segmentation_threshold = 50000
|
|
65
|
+
|
|
66
|
+
def _save_segmentation_config(self):
|
|
67
|
+
"""Save segmentation configuration to mcp_agent.config.yaml"""
|
|
68
|
+
import yaml
|
|
69
|
+
import os
|
|
70
|
+
|
|
71
|
+
# Get the project root directory (where mcp_agent.config.yaml is located)
|
|
72
|
+
current_file = os.path.abspath(__file__)
|
|
73
|
+
cli_dir = os.path.dirname(current_file) # cli directory
|
|
74
|
+
project_root = os.path.dirname(cli_dir) # project root
|
|
75
|
+
config_path = os.path.join(project_root, "mcp_agent.config.yaml")
|
|
76
|
+
|
|
77
|
+
try:
|
|
78
|
+
# Read current config
|
|
79
|
+
with open(config_path, "r", encoding="utf-8") as f:
|
|
80
|
+
config = yaml.safe_load(f)
|
|
81
|
+
|
|
82
|
+
# Update document segmentation settings
|
|
83
|
+
if "document_segmentation" not in config:
|
|
84
|
+
config["document_segmentation"] = {}
|
|
85
|
+
|
|
86
|
+
config["document_segmentation"]["enabled"] = self.segmentation_enabled
|
|
87
|
+
config["document_segmentation"]["size_threshold_chars"] = (
|
|
88
|
+
self.segmentation_threshold
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
# Write updated config
|
|
92
|
+
with open(config_path, "w", encoding="utf-8") as f:
|
|
93
|
+
yaml.dump(config, f, default_flow_style=False, allow_unicode=True)
|
|
94
|
+
|
|
95
|
+
print(
|
|
96
|
+
f"{Colors.OKGREEN}✅ Document segmentation configuration updated{Colors.ENDC}"
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
except Exception as e:
|
|
100
|
+
print(
|
|
101
|
+
f"{Colors.WARNING}⚠️ Failed to update segmentation config: {str(e)}{Colors.ENDC}"
|
|
102
|
+
)
|
|
45
103
|
|
|
104
|
+
def _init_tkinter(self):
|
|
105
|
+
"""Initialize tkinter availability check"""
|
|
46
106
|
# Check tkinter availability for file dialogs
|
|
47
107
|
self.tkinter_available = True
|
|
48
108
|
try:
|
|
@@ -765,6 +825,8 @@ class CLIInterface:
|
|
|
765
825
|
elif choice in ["s", "segmentation"]:
|
|
766
826
|
current_state = getattr(self, "segmentation_enabled", True)
|
|
767
827
|
self.segmentation_enabled = not current_state
|
|
828
|
+
# Save the configuration to file
|
|
829
|
+
self._save_segmentation_config()
|
|
768
830
|
seg_mode = (
|
|
769
831
|
"📄 Smart Segmentation"
|
|
770
832
|
if self.segmentation_enabled
|
|
@@ -214,15 +214,17 @@ async def main():
|
|
|
214
214
|
# 创建CLI应用
|
|
215
215
|
app = CLIApp()
|
|
216
216
|
|
|
217
|
-
# 设置配置
|
|
217
|
+
# 设置配置 - 默认禁用索引功能以加快处理速度
|
|
218
218
|
if args.optimized:
|
|
219
219
|
app.cli.enable_indexing = False
|
|
220
220
|
print(
|
|
221
221
|
f"\n{Colors.YELLOW}⚡ Optimized mode enabled - indexing disabled{Colors.ENDC}"
|
|
222
222
|
)
|
|
223
223
|
else:
|
|
224
|
+
# 默认也禁用索引功能
|
|
225
|
+
app.cli.enable_indexing = False
|
|
224
226
|
print(
|
|
225
|
-
f"\n{Colors.
|
|
227
|
+
f"\n{Colors.YELLOW}⚡ Fast mode enabled - indexing disabled by default{Colors.ENDC}"
|
|
226
228
|
)
|
|
227
229
|
|
|
228
230
|
# Configure document segmentation settings
|
|
@@ -230,18 +232,16 @@ async def main():
|
|
|
230
232
|
print(
|
|
231
233
|
f"\n{Colors.MAGENTA}📄 Document segmentation disabled - using traditional processing{Colors.ENDC}"
|
|
232
234
|
)
|
|
233
|
-
app.
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
}
|
|
235
|
+
app.cli.segmentation_enabled = False
|
|
236
|
+
app.cli.segmentation_threshold = args.segmentation_threshold
|
|
237
|
+
app.cli._save_segmentation_config()
|
|
237
238
|
else:
|
|
238
239
|
print(
|
|
239
240
|
f"\n{Colors.BLUE}📄 Smart document segmentation enabled (threshold: {args.segmentation_threshold} chars){Colors.ENDC}"
|
|
240
241
|
)
|
|
241
|
-
app.
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
}
|
|
242
|
+
app.cli.segmentation_enabled = True
|
|
243
|
+
app.cli.segmentation_threshold = args.segmentation_threshold
|
|
244
|
+
app.cli._save_segmentation_config()
|
|
245
245
|
|
|
246
246
|
# 检查是否为直接处理模式
|
|
247
247
|
if args.file or args.url or args.chat:
|
|
@@ -250,7 +250,9 @@ async def main():
|
|
|
250
250
|
if not os.path.exists(args.file):
|
|
251
251
|
print(f"{Colors.FAIL}❌ File not found: {args.file}{Colors.ENDC}")
|
|
252
252
|
sys.exit(1)
|
|
253
|
-
|
|
253
|
+
# 使用 file:// 前缀保持与交互模式一致,确保文件被复制而非移动
|
|
254
|
+
file_url = f"file://{os.path.abspath(args.file)}"
|
|
255
|
+
success = await run_direct_processing(app, file_url, "file")
|
|
254
256
|
elif args.url:
|
|
255
257
|
success = await run_direct_processing(app, args.url, "url")
|
|
256
258
|
elif args.chat:
|