deepcode-hku 1.0.6__tar.gz → 1.0.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. {deepcode_hku-1.0.6/deepcode_hku.egg-info → deepcode_hku-1.0.8}/PKG-INFO +26 -1
  2. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/README.md +24 -0
  3. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/__init__.py +1 -1
  4. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8/deepcode_hku.egg-info}/PKG-INFO +26 -1
  5. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/deepcode_hku.egg-info/requires.txt +1 -0
  6. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/mcp_agent.config.yaml +39 -21
  7. deepcode_hku-1.0.8/mcp_agent.secrets.yaml +7 -0
  8. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/prompts/code_prompts.py +34 -55
  9. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/requirements.txt +1 -0
  10. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/code_indexer.py +1 -32
  11. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/pdf_downloader.py +27 -0
  12. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/utils/file_processor.py +15 -0
  13. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/utils/llm_utils.py +114 -23
  14. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/agent_orchestration_engine.py +318 -80
  15. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/agents/code_implementation_agent.py +0 -1
  16. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/agents/document_segmentation_agent.py +1 -1
  17. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/agents/memory_agent_concise.py +290 -61
  18. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/agents/memory_agent_concise_index.py +264 -35
  19. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/agents/memory_agent_concise_multi.py +50 -1
  20. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/code_implementation_workflow.py +317 -50
  21. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/code_implementation_workflow_index.py +334 -44
  22. deepcode_hku-1.0.6/mcp_agent.secrets.yaml +0 -9
  23. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/.pre-commit-config.yaml +0 -0
  24. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/LICENSE +0 -0
  25. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/MANIFEST.in +0 -0
  26. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/cli/__init__.py +0 -0
  27. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/cli/cli_app.py +0 -0
  28. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/cli/cli_interface.py +0 -0
  29. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/cli/cli_launcher.py +0 -0
  30. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/cli/main_cli.py +0 -0
  31. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/cli/workflows/__init__.py +0 -0
  32. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/cli/workflows/cli_workflow_adapter.py +0 -0
  33. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/deepcode.py +0 -0
  34. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/deepcode_hku.egg-info/SOURCES.txt +0 -0
  35. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/deepcode_hku.egg-info/dependency_links.txt +0 -0
  36. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/deepcode_hku.egg-info/entry_points.txt +0 -0
  37. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/deepcode_hku.egg-info/top_level.txt +0 -0
  38. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/schema/mcp-agent.config.schema.json +0 -0
  39. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/setup.cfg +0 -0
  40. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/setup.py +0 -0
  41. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/__init__.py +0 -0
  42. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/bocha_search_server.py +0 -0
  43. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/code_implementation_server.py +0 -0
  44. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/code_reference_indexer.py +0 -0
  45. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/command_executor.py +0 -0
  46. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/document_segmentation_server.py +0 -0
  47. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/git_command.py +0 -0
  48. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/pdf_converter.py +0 -0
  49. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/tools/pdf_utils.py +0 -0
  50. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/ui/__init__.py +0 -0
  51. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/ui/app.py +0 -0
  52. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/ui/components.py +0 -0
  53. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/ui/handlers.py +0 -0
  54. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/ui/layout.py +0 -0
  55. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/ui/streamlit_app.py +0 -0
  56. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/ui/styles.py +0 -0
  57. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/utils/__init__.py +0 -0
  58. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/utils/cli_interface.py +0 -0
  59. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/utils/cross_platform_file_handler.py +0 -0
  60. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/utils/dialogue_logger.py +0 -0
  61. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/utils/simple_llm_logger.py +0 -0
  62. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/__init__.py +0 -0
  63. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/agents/__init__.py +0 -0
  64. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/agents/requirement_analysis_agent.py +0 -0
  65. {deepcode_hku-1.0.6 → deepcode_hku-1.0.8}/workflows/codebase_index_workflow.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: deepcode-hku
3
- Version: 1.0.6
3
+ Version: 1.0.8
4
4
  Summary: AI Research Engine - Transform research papers into working code automatically
5
5
  Home-page: https://github.com/HKUDS/DeepCode
6
6
  Author: DeepCodeTeam
@@ -24,6 +24,7 @@ Requires-Dist: aiohttp>=3.8.0
24
24
  Requires-Dist: anthropic
25
25
  Requires-Dist: asyncio-mqtt
26
26
  Requires-Dist: docling
27
+ Requires-Dist: google-genai
27
28
  Requires-Dist: mcp-agent
28
29
  Requires-Dist: mcp-server-git
29
30
  Requires-Dist: nest_asyncio
@@ -587,6 +588,14 @@ curl -O https://raw.githubusercontent.com/HKUDS/DeepCode/main/mcp_agent.secrets.
587
588
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
588
589
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
589
590
  # - anthropic: api_key (for Claude models)
591
+ # - google: api_key (for Gemini models)
592
+
593
+ # 🤖 Select your preferred LLM provider (optional)
594
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
595
+ # - llm_provider: "google" # Use Google Gemini models
596
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
597
+ # - llm_provider: "openai" # Use OpenAI/compatible models
598
+ # Note: If not set or unavailable, will automatically fallback to first available provider
590
599
 
591
600
  # 🔑 Configure search API keys for web search (optional)
592
601
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -623,6 +632,14 @@ uv pip install -r requirements.txt
623
632
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
624
633
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
625
634
  # - anthropic: api_key (for Claude models)
635
+ # - google: api_key (for Gemini models)
636
+
637
+ # 🤖 Select your preferred LLM provider (optional)
638
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
639
+ # - llm_provider: "google" # Use Google Gemini models
640
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
641
+ # - llm_provider: "openai" # Use OpenAI/compatible models
642
+ # Note: If not set or unavailable, will automatically fallback to first available provider
626
643
 
627
644
  # 🔑 Configure search API keys for web search (optional)
628
645
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -649,6 +666,14 @@ pip install -r requirements.txt
649
666
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
650
667
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
651
668
  # - anthropic: api_key (for Claude models)
669
+ # - google: api_key (for Gemini models)
670
+
671
+ # 🤖 Select your preferred LLM provider (optional)
672
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
673
+ # - llm_provider: "google" # Use Google Gemini models
674
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
675
+ # - llm_provider: "openai" # Use OpenAI/compatible models
676
+ # Note: If not set or unavailable, will automatically fallback to first available provider
652
677
 
653
678
  # 🔑 Configure search API keys for web search (optional)
654
679
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -542,6 +542,14 @@ curl -O https://raw.githubusercontent.com/HKUDS/DeepCode/main/mcp_agent.secrets.
542
542
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
543
543
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
544
544
  # - anthropic: api_key (for Claude models)
545
+ # - google: api_key (for Gemini models)
546
+
547
+ # 🤖 Select your preferred LLM provider (optional)
548
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
549
+ # - llm_provider: "google" # Use Google Gemini models
550
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
551
+ # - llm_provider: "openai" # Use OpenAI/compatible models
552
+ # Note: If not set or unavailable, will automatically fallback to first available provider
545
553
 
546
554
  # 🔑 Configure search API keys for web search (optional)
547
555
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -578,6 +586,14 @@ uv pip install -r requirements.txt
578
586
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
579
587
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
580
588
  # - anthropic: api_key (for Claude models)
589
+ # - google: api_key (for Gemini models)
590
+
591
+ # 🤖 Select your preferred LLM provider (optional)
592
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
593
+ # - llm_provider: "google" # Use Google Gemini models
594
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
595
+ # - llm_provider: "openai" # Use OpenAI/compatible models
596
+ # Note: If not set or unavailable, will automatically fallback to first available provider
581
597
 
582
598
  # 🔑 Configure search API keys for web search (optional)
583
599
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -604,6 +620,14 @@ pip install -r requirements.txt
604
620
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
605
621
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
606
622
  # - anthropic: api_key (for Claude models)
623
+ # - google: api_key (for Gemini models)
624
+
625
+ # 🤖 Select your preferred LLM provider (optional)
626
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
627
+ # - llm_provider: "google" # Use Google Gemini models
628
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
629
+ # - llm_provider: "openai" # Use OpenAI/compatible models
630
+ # Note: If not set or unavailable, will automatically fallback to first available provider
607
631
 
608
632
  # 🔑 Configure search API keys for web search (optional)
609
633
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -5,7 +5,7 @@ DeepCode - AI Research Engine
5
5
  ⚡ Transform research papers into working code automatically
6
6
  """
7
7
 
8
- __version__ = "1.0.6"
8
+ __version__ = "1.0.8"
9
9
  __author__ = "DeepCode Team"
10
10
  __url__ = "https://github.com/HKUDS/DeepCode"
11
11
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: deepcode-hku
3
- Version: 1.0.6
3
+ Version: 1.0.8
4
4
  Summary: AI Research Engine - Transform research papers into working code automatically
5
5
  Home-page: https://github.com/HKUDS/DeepCode
6
6
  Author: DeepCodeTeam
@@ -24,6 +24,7 @@ Requires-Dist: aiohttp>=3.8.0
24
24
  Requires-Dist: anthropic
25
25
  Requires-Dist: asyncio-mqtt
26
26
  Requires-Dist: docling
27
+ Requires-Dist: google-genai
27
28
  Requires-Dist: mcp-agent
28
29
  Requires-Dist: mcp-server-git
29
30
  Requires-Dist: nest_asyncio
@@ -587,6 +588,14 @@ curl -O https://raw.githubusercontent.com/HKUDS/DeepCode/main/mcp_agent.secrets.
587
588
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
588
589
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
589
590
  # - anthropic: api_key (for Claude models)
591
+ # - google: api_key (for Gemini models)
592
+
593
+ # 🤖 Select your preferred LLM provider (optional)
594
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
595
+ # - llm_provider: "google" # Use Google Gemini models
596
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
597
+ # - llm_provider: "openai" # Use OpenAI/compatible models
598
+ # Note: If not set or unavailable, will automatically fallback to first available provider
590
599
 
591
600
  # 🔑 Configure search API keys for web search (optional)
592
601
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -623,6 +632,14 @@ uv pip install -r requirements.txt
623
632
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
624
633
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
625
634
  # - anthropic: api_key (for Claude models)
635
+ # - google: api_key (for Gemini models)
636
+
637
+ # 🤖 Select your preferred LLM provider (optional)
638
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
639
+ # - llm_provider: "google" # Use Google Gemini models
640
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
641
+ # - llm_provider: "openai" # Use OpenAI/compatible models
642
+ # Note: If not set or unavailable, will automatically fallback to first available provider
626
643
 
627
644
  # 🔑 Configure search API keys for web search (optional)
628
645
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -649,6 +666,14 @@ pip install -r requirements.txt
649
666
  # Edit mcp_agent.secrets.yaml with your API keys and base_url:
650
667
  # - openai: api_key, base_url (for OpenAI/custom endpoints)
651
668
  # - anthropic: api_key (for Claude models)
669
+ # - google: api_key (for Gemini models)
670
+
671
+ # 🤖 Select your preferred LLM provider (optional)
672
+ # Edit mcp_agent.config.yaml to choose your LLM (line ~106):
673
+ # - llm_provider: "google" # Use Google Gemini models
674
+ # - llm_provider: "anthropic" # Use Anthropic Claude models
675
+ # - llm_provider: "openai" # Use OpenAI/compatible models
676
+ # Note: If not set or unavailable, will automatically fallback to first available provider
652
677
 
653
678
  # 🔑 Configure search API keys for web search (optional)
654
679
  # Edit mcp_agent.config.yaml to set your API keys:
@@ -3,6 +3,7 @@ aiohttp>=3.8.0
3
3
  anthropic
4
4
  asyncio-mqtt
5
5
  docling
6
+ google-genai
6
7
  mcp-agent
7
8
  mcp-server-git
8
9
  nest_asyncio
@@ -2,7 +2,7 @@ $schema: ./schema/mcp-agent.config.schema.json
2
2
  anthropic: null
3
3
  default_search_server: brave
4
4
  document_segmentation:
5
- enabled: true
5
+ enabled: false
6
6
  size_threshold_chars: 50000
7
7
  execution_engine: asyncio
8
8
  logger:
@@ -26,32 +26,32 @@ mcp:
26
26
  PYTHONPATH: .
27
27
  brave:
28
28
  # macos and linux should use this
29
- # args:
30
- # - -y
31
- # - '@modelcontextprotocol/server-brave-search'
32
- # command: npx
29
+ args:
30
+ - -y
31
+ - '@modelcontextprotocol/server-brave-search'
32
+ command: npx
33
33
 
34
34
  # windows should use this
35
- args:
36
- # please use the correct path for your system
37
- - C:/Users/LEGION/AppData/Roaming/npm/node_modules/@modelcontextprotocol/server-brave-search/dist/index.js
38
- command: node
35
+ # args:
36
+ # # please use the correct path for your system
37
+ # - C:/Users/LEGION/AppData/Roaming/npm/node_modules/@modelcontextprotocol/server-brave-search/dist/index.js
38
+ # command: node
39
39
  env:
40
40
  BRAVE_API_KEY: ''
41
41
  filesystem:
42
42
  # macos and linux should use this
43
- # args:
44
- # - -y
45
- # - '@modelcontextprotocol/server-filesystem'
46
- # - .
47
- # command: npx
48
-
49
- # windows should use this
50
43
  args:
51
- # please use the correct path for your system
52
- - C:/Users/LEGION/AppData/Roaming/npm/node_modules/@modelcontextprotocol/server-filesystem/dist/index.js
44
+ - -y
45
+ - '@modelcontextprotocol/server-filesystem'
53
46
  - .
54
- command: node
47
+ command: npx
48
+
49
+ # windows should use this
50
+ # args:
51
+ # # please use the correct path for your system
52
+ # - C:/Users/LEGION/AppData/Roaming/npm/node_modules/@modelcontextprotocol/server-filesystem/dist/index.js
53
+ # - .
54
+ # command: node
55
55
 
56
56
 
57
57
  code-implementation:
@@ -100,9 +100,27 @@ mcp:
100
100
  command: python
101
101
  env:
102
102
  PYTHONPATH: .
103
+ # LLM Provider Priority (选择使用哪个LLM / Choose which LLM to use)
104
+ # Options: "anthropic", "google", "openai"
105
+ # If not set or provider unavailable, will fallback to first available provider
106
+ llm_provider: "google" # 设置为 "google", "anthropic", 或 "openai"
107
+
103
108
  openai:
104
- base_max_tokens: 20000
105
- default_model: google/gemini-2.5-pro
109
+ base_max_tokens: 40000
110
+ # default_model: google/gemini-2.5-pro
111
+ default_model: anthropic/claude-sonnet-4.5
112
+ # default_model: openai/gpt-oss-120b
113
+ # default_model: deepseek/deepseek-v3.2-exp
114
+ # default_model: moonshotai/kimi-k2-thinking
115
+ reasoning_effort: low # Only for thinking models
106
116
  max_tokens_policy: adaptive
107
117
  retry_max_tokens: 32768
118
+
119
+ # Configuration for Google AI (Gemini)
120
+ google:
121
+ default_model: "gemini-3-pro-preview"
122
+
123
+ anthropic:
124
+ default_model: "claude-sonnet-4.5"
125
+
108
126
  planning_mode: traditional
@@ -0,0 +1,7 @@
1
+ openai:
2
+ api_key: ""
3
+ base_url: ""
4
+ anthropic:
5
+ api_key: ""
6
+ google:
7
+ api_key: ""
@@ -61,19 +61,21 @@ CRITICAL OUTPUT RESTRICTIONS:
61
61
  PAPER_DOWNLOADER_PROMPT = """You are a precise paper downloader that processes input from PaperInputAnalyzerAgent.
62
62
 
63
63
  Task: Handle paper according to input type and save to "./deepcode_lab/papers/id/id.md"
64
- Note: Generate id (id is a number) by counting files in "./deepcode_lab/papers/" directory and increment by 1.
64
+ Note: The paper ID will be provided at the start of the message as "PAPER_ID=<number>". Use this EXACT number.
65
65
 
66
- CRITICAL RULE: NEVER use write_file tool to create paper content directly. Always use file-downloader tools for PDF/document conversion.
66
+ CRITICAL RULES:
67
+ - Use the EXACT paper ID provided in the message (PAPER_ID=X).
68
+ - Save path MUST be: ./deepcode_lab/papers/{PAPER_ID}/{PAPER_ID}.md
67
69
 
68
70
  Processing Rules:
69
71
  1. URL Input (input_type = "url"):
70
- - Use "file-downloader" tool to download paper
72
+ - Use download_file_to tool with: url=<url>, destination="./deepcode_lab/papers/{PAPER_ID}/", filename="{PAPER_ID}.md"
71
73
  - Extract metadata (title, authors, year)
72
74
  - Return saved file path and metadata
73
75
 
74
76
  2. File Input (input_type = "file"):
75
- - Copy file to "./deepcode_lab/papers/id/" using move_file_to tool (preserves original)
76
- - The move_file_to tool will automatically convert PDF/documents to .md format
77
+ - Use move_file_to tool with: source=<file_path>, destination="./deepcode_lab/papers/{PAPER_ID}/{PAPER_ID}.md"
78
+ - The tool will automatically convert PDF/documents to .md format
77
79
  - NEVER manually extract content or use write_file - let the conversion tools handle this
78
80
  - Note: Original file is preserved, only a copy is placed in target directory
79
81
  - Return new saved file path and metadata
@@ -100,16 +102,26 @@ Input Format:
100
102
  "requirements": ["requirement1", "requirement2"]
101
103
  }
102
104
 
103
- Output Format (DO NOT MODIFY):
105
+ CRITICAL OUTPUT RESTRICTIONS:
106
+ - RETURN ONLY RAW JSON - NO TEXT BEFORE OR AFTER
107
+ - NO markdown code blocks (```json)
108
+ - NO explanatory text or descriptions
109
+ - NO tool call information
110
+ - NO analysis summaries
111
+ - JUST THE JSON OBJECT BELOW
112
+
113
+ Output Format (MANDATORY - EXACT FORMAT):
104
114
  {
105
115
  "status": "success|failure",
106
- "paper_path": "path to paper file or null for text input",
116
+ "paper_path": "./deepcode_lab/papers/{PAPER_ID}/{PAPER_ID}.md (or null for text input)",
107
117
  "metadata": {
108
118
  "title": "extracted or provided title",
109
119
  "authors": ["extracted or provided authors"],
110
120
  "year": "extracted or provided year"
111
121
  }
112
122
  }
123
+
124
+ Example: If PAPER_ID=14, then paper_path should be "./deepcode_lab/papers/14/14.md"
113
125
  """
114
126
 
115
127
  PAPER_REFERENCE_ANALYZER_PROMPT = """You are an expert academic paper reference analyzer specializing in computer science and machine learning.
@@ -1045,11 +1057,10 @@ PURE_CODE_IMPLEMENTATION_SYSTEM_PROMPT = """You are an expert code implementatio
1045
1057
  **IMPLEMENTATION APPROACH**:
1046
1058
  Build incrementally using multiple tool calls. For each step:
1047
1059
  1. **Identify** what needs to be implemented from the paper
1048
- 2. **Analyze Dependencies**: Before implementing each new file, use `read_code_mem` to read summaries of already-implemented files, then search for reference patterns to guide your implementation approach.
1049
- 3. **Implement** one component at a time
1050
- 4. **Test** immediately to catch issues early
1051
- 5. **Integrate** with existing components
1052
- 6. **Verify** against paper specifications
1060
+ 2. **Implement** one component at a time
1061
+ 3. **Test** immediately to catch issues early
1062
+ 4. **Integrate** with existing components
1063
+ 5. **Verify** against paper specifications
1053
1064
 
1054
1065
  **TOOL CALLING STRATEGY**:
1055
1066
  1. ⚠️ **SINGLE FUNCTION CALL PER MESSAGE**: Each message may perform only one function call. You will see the result of the function right after sending the message. If you need to perform multiple actions, you can always send more messages with subsequent function calls. Do some reasoning before your actions, describing what function calls you are going to use and how they fit into your plan.
@@ -1059,8 +1070,7 @@ Build incrementally using multiple tool calls. For each step:
1059
1070
  - **Reference only**: Use `search_code_references(indexes_path="indexes", target_file=the_file_you_want_to_implement, keywords=the_keywords_you_want_to_search)` for reference, NOT as implementation standard
1060
1071
  - **Core principle**: Original paper requirements take absolute priority over any reference code found
1061
1072
  3. **TOOL EXECUTION STRATEGY**:
1062
- - ⚠️**Development Cycle (for each new file implementation)**: `read_code_mem` (check existing implementations in Working Directory, use `read_file` as fallback if memory unavailable) → `search_code_references` (OPTIONAL reference check from indexes library in working directory) → `write_file` (implement based on original paper) → `execute_python` (if should test)
1063
- - **Environment Setup**: `write_file` (requirements.txt) → `execute_bash` (pip install) → `execute_python` (verify)
1073
+ - ⚠️**Development Cycle (for each new file implementation)**: `search_code_references` (OPTIONAL reference check from indexes library in working directory) → `write_file` (implement based on original paper)
1064
1074
 
1065
1075
  4. **CRITICAL**: Use bash and python tools to ACTUALLY REPLICATE the paper yourself - do not provide instructions.
1066
1076
 
@@ -1104,11 +1114,10 @@ You are an expert code implementation agent for academic paper reproduction. You
1104
1114
  **IMPLEMENTATION APPROACH**:
1105
1115
  Build incrementally using multiple tool calls. For each step:
1106
1116
  1. **Identify** what needs to be implemented from the paper
1107
- 2. **Analyze Dependencies**: Before implementing each new file, use `read_code_mem` to read summaries of already-implemented files, then search for reference patterns to guide your implementation approach.
1108
- 3. **Implement** one component at a time
1109
- 4. **Test** immediately to catch issues early
1110
- 5. **Integrate** with existing components
1111
- 6. **Verify** against paper specifications
1117
+ 2. **Implement** one component at a time
1118
+ 3. **Test** immediately to catch issues early
1119
+ 4. **Integrate** with existing components
1120
+ 5. **Verify** against paper specifications
1112
1121
 
1113
1122
  **TOOL CALLING STRATEGY**:
1114
1123
  1. ⚠️ **SINGLE FUNCTION CALL PER MESSAGE**: Each message may perform only one function call. You will see the result of the function right after sending the message. If you need to perform multiple actions, you can always send more messages with subsequent function calls. Do some reasoning before your actions, describing what function calls you are going to use and how they fit into your plan.
@@ -1118,10 +1127,7 @@ Build incrementally using multiple tool calls. For each step:
1118
1127
  - **Reference only**: Use `search_code_references(indexes_path="indexes", target_file=the_file_you_want_to_implement, keywords=the_keywords_you_want_to_search)` for reference, NOT as implementation standard
1119
1128
  - **Core principle**: Original paper requirements take absolute priority over any reference code found
1120
1129
  3. **TOOL EXECUTION STRATEGY**:
1121
- - ⚠️**Development Cycle (for each new file implementation)**: `read_code_mem` (check existing implementations in Working Directory, use `read_file` as fallback if memory unavailable`) → `search_code_references` (OPTIONAL reference check from `/home/agent/indexes`) → `write_file` (implement based on original paper) → `execute_python` (if needed to verify implementation)
1122
- - **File Verification**: Use `execute_bash` and `execute_python` when needed to check implementation completeness
1123
-
1124
- 4. **CRITICAL**: Use bash and python tools when needed to CHECK and VERIFY implementation completeness - do not provide instructions. These tools help validate that your implementation files are syntactically correct and properly structured.
1130
+ - ⚠️**Development Cycle (for each new file implementation)**: `search_code_references` (OPTIONAL reference check from `/home/agent/indexes`) → `write_file` (implement based on original paper)
1125
1131
 
1126
1132
  **Execution Guidelines**:
1127
1133
  - **Plan First**: Before each action, explain your reasoning and which function you'll use
@@ -1213,24 +1219,16 @@ GENERAL_CODE_IMPLEMENTATION_SYSTEM_PROMPT = """You are an expert code implementa
1213
1219
  **IMPLEMENTATION APPROACH**:
1214
1220
  Build incrementally using multiple tool calls. For each step:
1215
1221
  1. **Identify** what needs to be implemented from the requirements
1216
- 2. **Analyze Dependencies**: Before implementing each new file, use `read_code_mem` to read summaries of already-implemented files, then search for reference patterns to guide your implementation approach.
1217
- 3. **Implement** one component at a time
1218
- 4. **Verify** optionally using `execute_python` or `execute_bash` to check implementation completeness if needed
1219
- 5. **Integrate** with existing components
1220
- 6. **Validate** against requirement specifications
1222
+ 2. **Implement** one component at a time
1223
+ 3. **Verify** optionally using `execute_python` or `execute_bash` to check implementation completeness if needed
1224
+ 4. **Integrate** with existing components
1225
+ 5. **Validate** against requirement specifications
1221
1226
 
1222
1227
  **TOOL CALLING STRATEGY**:
1223
1228
  1. ⚠️ **SINGLE FUNCTION CALL PER MESSAGE**: Each message may perform only one function call. You will see the result of the function right after sending the message. If you need to perform multiple actions, you can always send more messages with subsequent function calls. Do some reasoning before your actions, describing what function calls you are going to use and how they fit into your plan.
1224
1229
 
1225
1230
  2. **TOOL EXECUTION STRATEGY**:
1226
- - **Development Cycle (for each new file implementation)**: `read_code_mem` (check existing implementations in Working Directory, use `read_file` as fallback if memory unavailable) → `write_file` (implement) → **Optional Verification**: `execute_python` or `execute_bash` (if needed to check implementation)
1227
- - **File Verification**: Use `execute_bash` and `execute_python` when needed to verify implementation completeness.
1228
-
1229
- 3. **CRITICAL**: Use `execute_bash` and `execute_python` tools when needed to CHECK and VERIFY file implementation completeness - do not provide instructions. These tools are essential for:
1230
- - Checking file syntax and import correctness (`execute_python`)
1231
- - Verifying file structure and dependencies (`execute_bash` for listing, `execute_python` for imports)
1232
- - Validating that implemented files are syntactically correct and can be imported
1233
- - Ensuring code implementation meets basic functionality requirements
1231
+ - **Development Cycle (for each new file implementation)**: `write_file` (implement)
1234
1232
 
1235
1233
  **Execution Guidelines**:
1236
1234
  - **Plan First**: Before each action, explain your reasoning and which function you'll use
@@ -1348,10 +1346,6 @@ PAPER_ALGORITHM_ANALYSIS_PROMPT_TRADITIONAL = """You are extracting COMPLETE imp
1348
1346
  ## TRADITIONAL APPROACH: Full Document Reading
1349
1347
  Read the complete document to ensure comprehensive coverage of all algorithmic details:
1350
1348
 
1351
- 1. **Locate and read the markdown (.md) file** in the paper directory
1352
- 2. **Analyze the entire document** to capture all algorithms, methods, and formulas
1353
- 3. **Extract complete implementation details** without missing any components
1354
-
1355
1349
  # DETAILED EXTRACTION PROTOCOL
1356
1350
 
1357
1351
  ## 1. COMPREHENSIVE ALGORITHM SCAN
@@ -1511,10 +1505,6 @@ Map out the ENTIRE paper structure and identify ALL components that need impleme
1511
1505
  ## TRADITIONAL APPROACH: Complete Document Analysis
1512
1506
  Read the entire document systematically to ensure comprehensive understanding:
1513
1507
 
1514
- 1. **Locate and read the markdown (.md) file** in the paper directory
1515
- 2. **Analyze the complete document structure** from introduction to conclusion
1516
- 3. **Extract all conceptual frameworks** and implementation requirements
1517
-
1518
1508
  # COMPREHENSIVE ANALYSIS PROTOCOL
1519
1509
 
1520
1510
  ## 1. COMPLETE PAPER STRUCTURAL ANALYSIS
@@ -1678,17 +1668,6 @@ You receive two exhaustive analyses:
1678
1668
  1. **Comprehensive Paper Analysis**: Complete paper structure, components, and requirements
1679
1669
  2. **Complete Algorithm Extraction**: All algorithms, formulas, pseudocode, and technical details
1680
1670
 
1681
- Plus you can access the complete paper document by reading the markdown file directly.
1682
-
1683
- # TRADITIONAL DOCUMENT ACCESS
1684
-
1685
- ## Direct Paper Reading
1686
- For any additional details needed beyond the provided analyses:
1687
-
1688
- 1. **Read the complete markdown (.md) file** in the paper directory
1689
- 2. **Access any section directly** without token limitations for smaller documents
1690
- 3. **Cross-reference information** across the entire document as needed
1691
-
1692
1671
  # OBJECTIVE
1693
1672
  Create an implementation plan so detailed that a developer can reproduce the ENTIRE paper without reading it.
1694
1673
 
@@ -3,6 +3,7 @@ aiohttp>=3.8.0
3
3
  anthropic
4
4
  asyncio-mqtt
5
5
  docling
6
+ google-genai
6
7
  mcp-agent
7
8
  mcp-server-git
8
9
  nest_asyncio
@@ -24,38 +24,7 @@ from dataclasses import dataclass, asdict
24
24
  from typing import List, Dict, Any
25
25
 
26
26
  # MCP Agent imports for LLM
27
- import yaml
28
- from utils.llm_utils import get_preferred_llm_class
29
-
30
-
31
- def get_default_models(config_path: str = "mcp_agent.config.yaml"):
32
- """
33
- Get default models from configuration file.
34
-
35
- Args:
36
- config_path: Path to the configuration file
37
-
38
- Returns:
39
- dict: Dictionary with 'anthropic' and 'openai' default models
40
- """
41
- try:
42
- if os.path.exists(config_path):
43
- with open(config_path, "r", encoding="utf-8") as f:
44
- config = yaml.safe_load(f)
45
-
46
- anthropic_model = config.get("anthropic", {}).get(
47
- "default_model", "claude-sonnet-4-20250514"
48
- )
49
- openai_model = config.get("openai", {}).get("default_model", "o3-mini")
50
-
51
- return {"anthropic": anthropic_model, "openai": openai_model}
52
- else:
53
- print(f"Config file {config_path} not found, using default models")
54
- return {"anthropic": "claude-sonnet-4-20250514", "openai": "o3-mini"}
55
-
56
- except Exception as e:
57
- print(f"Error reading config file {config_path}: {e}")
58
- return {"anthropic": "claude-sonnet-4-20250514", "openai": "o3-mini"}
27
+ from utils.llm_utils import get_preferred_llm_class, get_default_models
59
28
 
60
29
 
61
30
  @dataclass
@@ -1098,9 +1098,26 @@ async def download_file_to(
1098
1098
  Status message about the download operation
1099
1099
  """
1100
1100
  # 确定文件名
1101
+
1102
+ url = URLExtractor.extract_urls(url)[0]
1103
+
1101
1104
  if not filename:
1102
1105
  filename = URLExtractor.infer_filename_from_url(url)
1103
1106
 
1107
+ if not filename:
1108
+ filename = URLExtractor.infer_filename_from_url(url)
1109
+ else:
1110
+ name_source, extension_source = os.path.splitext(
1111
+ os.path.basename(URLExtractor.infer_filename_from_url(url))
1112
+ )
1113
+ name_destination, extension_destination = os.path.splitext(
1114
+ os.path.basename(filename)
1115
+ )
1116
+ if extension_source:
1117
+ filename = name_destination + extension_source
1118
+ else:
1119
+ filename = name_destination + extension_destination
1120
+
1104
1121
  # 确定完整路径
1105
1122
  if destination:
1106
1123
  # 展开用户目录
@@ -1203,6 +1220,15 @@ async def move_file_to(
1203
1220
  # 确定文件名
1204
1221
  if not filename:
1205
1222
  filename = os.path.basename(source)
1223
+ else:
1224
+ name_source, extension_source = os.path.splitext(os.path.basename(source))
1225
+ name_destination, extension_destination = os.path.splitext(
1226
+ os.path.basename(filename)
1227
+ )
1228
+ if extension_source:
1229
+ filename = name_destination + extension_source
1230
+ else:
1231
+ filename = name_destination + extension_destination
1206
1232
 
1207
1233
  # 确定完整路径
1208
1234
  if destination:
@@ -1215,6 +1241,7 @@ async def move_file_to(
1215
1241
  target_path = destination
1216
1242
  else: # 是目录
1217
1243
  target_path = os.path.join(destination, filename)
1244
+
1218
1245
  else:
1219
1246
  target_path = filename
1220
1247
 
@@ -282,10 +282,25 @@ class FileProcessor:
282
282
  if isinstance(file_input, str):
283
283
  import re
284
284
 
285
+ # Try to extract path from backticks first
285
286
  file_path_match = re.search(r"`([^`]+\.md)`", file_input)
286
287
  if file_path_match:
287
288
  paper_path = file_path_match.group(1)
288
289
  file_input = {"paper_path": paper_path}
290
+ else:
291
+ # Try to extract from "Saved Path:" or similar patterns
292
+ path_patterns = [
293
+ r"[Ss]aved [Pp]ath[:\s]+([^\s\n]+\.md)",
294
+ r"[Pp]aper [Pp]ath[:\s]+([^\s\n]+\.md)",
295
+ r"[Ff]ile[:\s]+([^\s\n]+\.md)",
296
+ r"[Oo]utput[:\s]+([^\s\n]+\.md)",
297
+ ]
298
+ for pattern in path_patterns:
299
+ match = re.search(pattern, file_input)
300
+ if match:
301
+ paper_path = match.group(1)
302
+ file_input = {"paper_path": paper_path}
303
+ break
289
304
 
290
305
  # Extract paper directory path
291
306
  paper_dir = cls.extract_file_path(file_input)