deepcode-hku 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli/__init__.py +18 -0
- cli/cli_app.py +296 -0
- cli/cli_interface.py +744 -0
- cli/cli_launcher.py +155 -0
- cli/main_cli.py +243 -0
- cli/workflows/__init__.py +11 -0
- cli/workflows/cli_workflow_adapter.py +336 -0
- deepcode.py +219 -0
- deepcode_hku-1.0.1.dist-info/METADATA +695 -0
- deepcode_hku-1.0.1.dist-info/RECORD +44 -0
- deepcode_hku-1.0.1.dist-info/WHEEL +5 -0
- deepcode_hku-1.0.1.dist-info/entry_points.txt +2 -0
- deepcode_hku-1.0.1.dist-info/licenses/LICENSE +21 -0
- deepcode_hku-1.0.1.dist-info/top_level.txt +6 -0
- tools/__init__.py +0 -0
- tools/code_implementation_server.py +1045 -0
- tools/code_indexer.py +1657 -0
- tools/code_reference_indexer.py +486 -0
- tools/command_executor.py +324 -0
- tools/git_command.py +356 -0
- tools/pdf_converter.py +640 -0
- tools/pdf_downloader.py +1370 -0
- tools/pdf_utils.py +52 -0
- ui/__init__.py +43 -0
- ui/app.py +13 -0
- ui/components.py +1450 -0
- ui/handlers.py +773 -0
- ui/layout.py +106 -0
- ui/streamlit_app.py +38 -0
- ui/styles.py +2116 -0
- utils/__init__.py +17 -0
- utils/cli_interface.py +459 -0
- utils/dialogue_logger.py +671 -0
- utils/file_processor.py +426 -0
- utils/simple_llm_logger.py +198 -0
- workflows/__init__.py +31 -0
- workflows/agent_orchestration_engine.py +1371 -0
- workflows/agents/__init__.py +13 -0
- workflows/agents/code_implementation_agent.py +1093 -0
- workflows/agents/memory_agent_concise.py +923 -0
- workflows/agents/memory_agent_concise_index.py +935 -0
- workflows/code_implementation_workflow.py +924 -0
- workflows/code_implementation_workflow_index.py +931 -0
- workflows/codebase_index_workflow.py +726 -0
|
@@ -0,0 +1,726 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Codebase Index Workflow
|
|
3
|
+
代码库索引工作流
|
|
4
|
+
|
|
5
|
+
This workflow integrates the functionality of run_indexer.py and code_indexer.py
|
|
6
|
+
to build intelligent relationships between existing codebase and target structure.
|
|
7
|
+
|
|
8
|
+
该工作流集成了run_indexer.py和code_indexer.py的功能,
|
|
9
|
+
用于在现有代码库和目标结构之间建立智能关系。
|
|
10
|
+
|
|
11
|
+
Features:
|
|
12
|
+
- 从initial_plan.txt提取目标文件结构 / Extract target file structure from initial_plan.txt
|
|
13
|
+
- 分析代码库并建立索引 / Analyze codebase and build indexes
|
|
14
|
+
- 生成关系映射和统计报告 / Generate relationship mappings and statistical reports
|
|
15
|
+
- 为代码复现提供参考依据 / Provide reference basis for code reproduction
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import json
|
|
20
|
+
import logging
|
|
21
|
+
import os
|
|
22
|
+
import re
|
|
23
|
+
import sys
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Dict, Any, Optional
|
|
26
|
+
import yaml
|
|
27
|
+
|
|
28
|
+
# 添加tools目录到路径中 / Add tools directory to path
|
|
29
|
+
sys.path.append(str(Path(__file__).parent.parent / "tools"))
|
|
30
|
+
|
|
31
|
+
from tools.code_indexer import CodeIndexer
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class CodebaseIndexWorkflow:
|
|
35
|
+
"""代码库索引工作流类 / Codebase Index Workflow Class"""
|
|
36
|
+
|
|
37
|
+
def __init__(self, logger=None):
|
|
38
|
+
"""
|
|
39
|
+
初始化工作流
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
logger: 日志记录器实例
|
|
43
|
+
"""
|
|
44
|
+
self.logger = logger or self._setup_default_logger()
|
|
45
|
+
self.indexer = None
|
|
46
|
+
|
|
47
|
+
def _setup_default_logger(self) -> logging.Logger:
|
|
48
|
+
"""设置默认日志记录器"""
|
|
49
|
+
logger = logging.getLogger("CodebaseIndexWorkflow")
|
|
50
|
+
logger.setLevel(logging.INFO)
|
|
51
|
+
|
|
52
|
+
if not logger.handlers:
|
|
53
|
+
handler = logging.StreamHandler()
|
|
54
|
+
formatter = logging.Formatter(
|
|
55
|
+
"%(asctime)s - %(name)s - %(levelname)s - %(message)s"
|
|
56
|
+
)
|
|
57
|
+
handler.setFormatter(formatter)
|
|
58
|
+
logger.addHandler(handler)
|
|
59
|
+
|
|
60
|
+
return logger
|
|
61
|
+
|
|
62
|
+
def extract_file_tree_from_plan(self, plan_content: str) -> Optional[str]:
|
|
63
|
+
"""
|
|
64
|
+
从initial_plan.txt内容中提取文件树结构
|
|
65
|
+
Extract file tree structure from initial_plan.txt content
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
plan_content: Content of the initial_plan.txt file
|
|
69
|
+
|
|
70
|
+
Returns:
|
|
71
|
+
Extracted file tree structure as string
|
|
72
|
+
"""
|
|
73
|
+
# 查找文件结构部分,特别是"## File Structure"格式
|
|
74
|
+
file_structure_pattern = r"## File Structure[^\n]*\n```[^\n]*\n(.*?)\n```"
|
|
75
|
+
|
|
76
|
+
match = re.search(file_structure_pattern, plan_content, re.DOTALL)
|
|
77
|
+
if match:
|
|
78
|
+
file_tree = match.group(1).strip()
|
|
79
|
+
lines = file_tree.split("\n")
|
|
80
|
+
|
|
81
|
+
# 清理树结构 - 移除空行和不属于结构的注释
|
|
82
|
+
cleaned_lines = []
|
|
83
|
+
for line in lines:
|
|
84
|
+
# 保留树结构的行
|
|
85
|
+
if line.strip() and (
|
|
86
|
+
any(char in line for char in ["├──", "└──", "│"])
|
|
87
|
+
or line.strip().endswith("/")
|
|
88
|
+
or "." in line.split("/")[-1] # 有文件扩展名
|
|
89
|
+
or line.strip().endswith(".py")
|
|
90
|
+
or line.strip().endswith(".txt")
|
|
91
|
+
or line.strip().endswith(".md")
|
|
92
|
+
or line.strip().endswith(".yaml")
|
|
93
|
+
):
|
|
94
|
+
cleaned_lines.append(line)
|
|
95
|
+
|
|
96
|
+
if len(cleaned_lines) >= 5:
|
|
97
|
+
file_tree = "\n".join(cleaned_lines)
|
|
98
|
+
self.logger.info(
|
|
99
|
+
f"📊 从## File Structure部分提取文件树结构 ({len(cleaned_lines)} lines)"
|
|
100
|
+
)
|
|
101
|
+
return file_tree
|
|
102
|
+
|
|
103
|
+
# 备用方案:查找包含项目结构的任何代码块
|
|
104
|
+
code_block_patterns = [
|
|
105
|
+
r"```[^\n]*\n(rice_framework/.*?(?:├──|└──).*?)\n```",
|
|
106
|
+
r"```[^\n]*\n(project/.*?(?:├──|└──).*?)\n```",
|
|
107
|
+
r"```[^\n]*\n(src/.*?(?:├──|└──).*?)\n```",
|
|
108
|
+
r"```[^\n]*\n(.*?(?:├──|└──).*?(?:\.py|\.txt|\.md|\.yaml).*?)\n```",
|
|
109
|
+
]
|
|
110
|
+
|
|
111
|
+
for pattern in code_block_patterns:
|
|
112
|
+
match = re.search(pattern, plan_content, re.DOTALL)
|
|
113
|
+
if match:
|
|
114
|
+
file_tree = match.group(1).strip()
|
|
115
|
+
lines = [line for line in file_tree.split("\n") if line.strip()]
|
|
116
|
+
if len(lines) >= 5:
|
|
117
|
+
self.logger.info(
|
|
118
|
+
f"📊 从代码块中提取文件树结构 ({len(lines)} lines)"
|
|
119
|
+
)
|
|
120
|
+
return file_tree
|
|
121
|
+
|
|
122
|
+
# 最终备用方案:从文件提及中提取文件路径并创建基本结构
|
|
123
|
+
self.logger.warning("⚠️ 未找到标准文件树,尝试从文件提及中提取...")
|
|
124
|
+
|
|
125
|
+
# 在整个文档中查找反引号中的文件路径
|
|
126
|
+
file_mentions = re.findall(
|
|
127
|
+
r"`([^`]*(?:\.py|\.txt|\.md|\.yaml|\.yml)[^`]*)`", plan_content
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
if file_mentions:
|
|
131
|
+
# 将文件组织成目录结构
|
|
132
|
+
dirs = set()
|
|
133
|
+
files_by_dir = {}
|
|
134
|
+
|
|
135
|
+
for file_path in file_mentions:
|
|
136
|
+
file_path = file_path.strip()
|
|
137
|
+
if "/" in file_path:
|
|
138
|
+
dir_path = "/".join(file_path.split("/")[:-1])
|
|
139
|
+
filename = file_path.split("/")[-1]
|
|
140
|
+
dirs.add(dir_path)
|
|
141
|
+
if dir_path not in files_by_dir:
|
|
142
|
+
files_by_dir[dir_path] = []
|
|
143
|
+
files_by_dir[dir_path].append(filename)
|
|
144
|
+
else:
|
|
145
|
+
if "root" not in files_by_dir:
|
|
146
|
+
files_by_dir["root"] = []
|
|
147
|
+
files_by_dir["root"].append(file_path)
|
|
148
|
+
|
|
149
|
+
# 创建树结构
|
|
150
|
+
structure_lines = []
|
|
151
|
+
|
|
152
|
+
# 确定根目录名称
|
|
153
|
+
root_name = (
|
|
154
|
+
"rice_framework"
|
|
155
|
+
if any("rice" in f for f in file_mentions)
|
|
156
|
+
else "project"
|
|
157
|
+
)
|
|
158
|
+
structure_lines.append(f"{root_name}/")
|
|
159
|
+
|
|
160
|
+
# 添加目录和文件
|
|
161
|
+
sorted_dirs = sorted(dirs) if dirs else []
|
|
162
|
+
for i, dir_path in enumerate(sorted_dirs):
|
|
163
|
+
is_last_dir = i == len(sorted_dirs) - 1
|
|
164
|
+
prefix = "└──" if is_last_dir else "├──"
|
|
165
|
+
structure_lines.append(f"{prefix} {dir_path}/")
|
|
166
|
+
|
|
167
|
+
if dir_path in files_by_dir:
|
|
168
|
+
files = sorted(files_by_dir[dir_path])
|
|
169
|
+
for j, filename in enumerate(files):
|
|
170
|
+
is_last_file = j == len(files) - 1
|
|
171
|
+
if is_last_dir:
|
|
172
|
+
file_prefix = " └──" if is_last_file else " ├──"
|
|
173
|
+
else:
|
|
174
|
+
file_prefix = "│ └──" if is_last_file else "│ ├──"
|
|
175
|
+
structure_lines.append(f"{file_prefix} {filename}")
|
|
176
|
+
|
|
177
|
+
# 添加根文件(如果有)
|
|
178
|
+
if "root" in files_by_dir:
|
|
179
|
+
root_files = sorted(files_by_dir["root"])
|
|
180
|
+
for i, filename in enumerate(root_files):
|
|
181
|
+
is_last = (i == len(root_files) - 1) and not sorted_dirs
|
|
182
|
+
prefix = "└──" if is_last else "├──"
|
|
183
|
+
structure_lines.append(f"{prefix} {filename}")
|
|
184
|
+
|
|
185
|
+
if len(structure_lines) >= 3:
|
|
186
|
+
file_tree = "\n".join(structure_lines)
|
|
187
|
+
self.logger.info(
|
|
188
|
+
f"📊 从文件提及生成文件树 ({len(structure_lines)} lines)"
|
|
189
|
+
)
|
|
190
|
+
return file_tree
|
|
191
|
+
|
|
192
|
+
# 如果未找到文件树,返回None
|
|
193
|
+
self.logger.warning("⚠️ 在初始计划中未找到文件树结构")
|
|
194
|
+
return None
|
|
195
|
+
|
|
196
|
+
def load_target_structure_from_plan(self, plan_path: str) -> str:
|
|
197
|
+
"""
|
|
198
|
+
从initial_plan.txt加载目标结构并提取文件树
|
|
199
|
+
Load target structure from initial_plan.txt and extract file tree
|
|
200
|
+
|
|
201
|
+
Args:
|
|
202
|
+
plan_path: Path to initial_plan.txt file
|
|
203
|
+
|
|
204
|
+
Returns:
|
|
205
|
+
Extracted file tree structure
|
|
206
|
+
"""
|
|
207
|
+
try:
|
|
208
|
+
# 加载完整的计划内容
|
|
209
|
+
with open(plan_path, "r", encoding="utf-8") as f:
|
|
210
|
+
plan_content = f.read()
|
|
211
|
+
|
|
212
|
+
self.logger.info(f"📄 已加载初始计划 ({len(plan_content)} characters)")
|
|
213
|
+
|
|
214
|
+
# 提取文件树结构
|
|
215
|
+
file_tree = self.extract_file_tree_from_plan(plan_content)
|
|
216
|
+
|
|
217
|
+
if file_tree:
|
|
218
|
+
self.logger.info("✅ 成功从初始计划中提取文件树")
|
|
219
|
+
self.logger.info("📋 提取结构预览:")
|
|
220
|
+
# 显示提取树的前几行
|
|
221
|
+
preview_lines = file_tree.split("\n")[:8]
|
|
222
|
+
for line in preview_lines:
|
|
223
|
+
self.logger.info(f" {line}")
|
|
224
|
+
if len(file_tree.split("\n")) > 8:
|
|
225
|
+
self.logger.info(f" ... 还有 {len(file_tree.split('\n')) - 8} 行")
|
|
226
|
+
return file_tree
|
|
227
|
+
else:
|
|
228
|
+
self.logger.warning("⚠️ 无法从初始计划中提取文件树")
|
|
229
|
+
self.logger.info("🔄 回退到默认目标结构")
|
|
230
|
+
return self.get_default_target_structure()
|
|
231
|
+
|
|
232
|
+
except Exception as e:
|
|
233
|
+
self.logger.error(f"❌ 加载初始计划文件失败 {plan_path}: {e}")
|
|
234
|
+
self.logger.info("🔄 回退到默认目标结构")
|
|
235
|
+
return self.get_default_target_structure()
|
|
236
|
+
|
|
237
|
+
def get_default_target_structure(self) -> str:
|
|
238
|
+
"""获取默认目标结构"""
|
|
239
|
+
return """
|
|
240
|
+
project/
|
|
241
|
+
├── src/
|
|
242
|
+
│ ├── core/
|
|
243
|
+
│ │ ├── gcn.py # GCN encoder
|
|
244
|
+
│ │ ├── diffusion.py # forward/reverse processes
|
|
245
|
+
│ │ ├── denoiser.py # denoising MLP
|
|
246
|
+
│ │ └── fusion.py # fusion combiner
|
|
247
|
+
│ ├── models/ # model wrapper classes
|
|
248
|
+
│ │ └── recdiff.py
|
|
249
|
+
│ ├── utils/
|
|
250
|
+
│ │ ├── data.py # loading & preprocessing
|
|
251
|
+
│ │ ├── predictor.py # scoring functions
|
|
252
|
+
│ │ ├── loss.py # loss functions
|
|
253
|
+
│ │ ├── metrics.py # NDCG, Recall etc.
|
|
254
|
+
│ │ └── sched.py # beta/alpha schedule utils
|
|
255
|
+
│ └── configs/
|
|
256
|
+
│ └── default.yaml # hyperparameters, paths
|
|
257
|
+
├── tests/
|
|
258
|
+
│ ├── test_gcn.py
|
|
259
|
+
│ ├── test_diffusion.py
|
|
260
|
+
│ ├── test_denoiser.py
|
|
261
|
+
│ ├── test_loss.py
|
|
262
|
+
│ └── test_pipeline.py
|
|
263
|
+
├── docs/
|
|
264
|
+
│ ├── architecture.md
|
|
265
|
+
│ ├── api_reference.md
|
|
266
|
+
│ └── README.md
|
|
267
|
+
├── experiments/
|
|
268
|
+
│ ├── run_experiment.py
|
|
269
|
+
│ └── notebooks/
|
|
270
|
+
│ └── analysis.ipynb
|
|
271
|
+
├── requirements.txt
|
|
272
|
+
└── setup.py
|
|
273
|
+
"""
|
|
274
|
+
|
|
275
|
+
def load_or_create_indexer_config(self, paper_dir: str) -> Dict[str, Any]:
|
|
276
|
+
"""
|
|
277
|
+
加载或创建索引器配置
|
|
278
|
+
Load or create indexer configuration
|
|
279
|
+
|
|
280
|
+
Args:
|
|
281
|
+
paper_dir: 论文目录路径
|
|
282
|
+
|
|
283
|
+
Returns:
|
|
284
|
+
配置字典
|
|
285
|
+
"""
|
|
286
|
+
# 尝试加载现有的配置文件
|
|
287
|
+
config_path = Path(__file__).parent.parent / "tools" / "indexer_config.yaml"
|
|
288
|
+
|
|
289
|
+
try:
|
|
290
|
+
if config_path.exists():
|
|
291
|
+
with open(config_path, "r", encoding="utf-8") as f:
|
|
292
|
+
config = yaml.safe_load(f)
|
|
293
|
+
|
|
294
|
+
# 更新路径配置为当前论文目录
|
|
295
|
+
if "paths" not in config:
|
|
296
|
+
config["paths"] = {}
|
|
297
|
+
config["paths"]["code_base_path"] = os.path.join(paper_dir, "code_base")
|
|
298
|
+
config["paths"]["output_dir"] = os.path.join(paper_dir, "indexes")
|
|
299
|
+
|
|
300
|
+
# 调整性能设置以适应工作流
|
|
301
|
+
if "performance" in config:
|
|
302
|
+
config["performance"]["enable_concurrent_analysis"] = (
|
|
303
|
+
False # 禁用并发以避免API限制
|
|
304
|
+
)
|
|
305
|
+
if "debug" in config:
|
|
306
|
+
config["debug"]["verbose_output"] = True # 启用详细输出
|
|
307
|
+
if "llm" in config:
|
|
308
|
+
config["llm"]["request_delay"] = 0.5 # 增加请求间隔
|
|
309
|
+
|
|
310
|
+
self.logger.info(f"已加载配置文件: {config_path}")
|
|
311
|
+
return config
|
|
312
|
+
|
|
313
|
+
except Exception as e:
|
|
314
|
+
self.logger.warning(f"加载配置文件失败: {e}")
|
|
315
|
+
|
|
316
|
+
# 如果加载失败,使用默认配置
|
|
317
|
+
self.logger.info("使用默认配置")
|
|
318
|
+
default_config = {
|
|
319
|
+
"paths": {
|
|
320
|
+
"code_base_path": os.path.join(paper_dir, "code_base"),
|
|
321
|
+
"output_dir": os.path.join(paper_dir, "indexes"),
|
|
322
|
+
},
|
|
323
|
+
"llm": {
|
|
324
|
+
"model_provider": "anthropic",
|
|
325
|
+
"max_tokens": 4000,
|
|
326
|
+
"temperature": 0.3,
|
|
327
|
+
"request_delay": 0.5, # 增加请求间隔
|
|
328
|
+
"max_retries": 3,
|
|
329
|
+
"retry_delay": 1.0,
|
|
330
|
+
},
|
|
331
|
+
"file_analysis": {
|
|
332
|
+
"max_file_size": 1048576, # 1MB
|
|
333
|
+
"max_content_length": 3000,
|
|
334
|
+
"supported_extensions": [
|
|
335
|
+
".py",
|
|
336
|
+
".js",
|
|
337
|
+
".ts",
|
|
338
|
+
".java",
|
|
339
|
+
".cpp",
|
|
340
|
+
".c",
|
|
341
|
+
".h",
|
|
342
|
+
".hpp",
|
|
343
|
+
".cs",
|
|
344
|
+
".php",
|
|
345
|
+
".rb",
|
|
346
|
+
".go",
|
|
347
|
+
".rs",
|
|
348
|
+
".scala",
|
|
349
|
+
".kt",
|
|
350
|
+
".yaml",
|
|
351
|
+
".yml",
|
|
352
|
+
".json",
|
|
353
|
+
".xml",
|
|
354
|
+
".toml",
|
|
355
|
+
".md",
|
|
356
|
+
".txt",
|
|
357
|
+
],
|
|
358
|
+
"skip_directories": [
|
|
359
|
+
"__pycache__",
|
|
360
|
+
"node_modules",
|
|
361
|
+
"target",
|
|
362
|
+
"build",
|
|
363
|
+
"dist",
|
|
364
|
+
"venv",
|
|
365
|
+
"env",
|
|
366
|
+
".git",
|
|
367
|
+
".svn",
|
|
368
|
+
"data",
|
|
369
|
+
"datasets",
|
|
370
|
+
],
|
|
371
|
+
},
|
|
372
|
+
"relationships": {
|
|
373
|
+
"min_confidence_score": 0.3,
|
|
374
|
+
"high_confidence_threshold": 0.7,
|
|
375
|
+
"relationship_types": {
|
|
376
|
+
"direct_match": 1.0,
|
|
377
|
+
"partial_match": 0.8,
|
|
378
|
+
"reference": 0.6,
|
|
379
|
+
"utility": 0.4,
|
|
380
|
+
},
|
|
381
|
+
},
|
|
382
|
+
"performance": {
|
|
383
|
+
"enable_concurrent_analysis": False, # 禁用并发以避免API限制
|
|
384
|
+
"max_concurrent_files": 3,
|
|
385
|
+
"enable_content_caching": True,
|
|
386
|
+
"max_cache_size": 100,
|
|
387
|
+
},
|
|
388
|
+
"debug": {
|
|
389
|
+
"verbose_output": True,
|
|
390
|
+
"save_raw_responses": False,
|
|
391
|
+
"mock_llm_responses": False,
|
|
392
|
+
},
|
|
393
|
+
"output": {
|
|
394
|
+
"generate_summary": True,
|
|
395
|
+
"generate_statistics": True,
|
|
396
|
+
"include_metadata": True,
|
|
397
|
+
"json_indent": 2,
|
|
398
|
+
},
|
|
399
|
+
"logging": {"level": "INFO", "log_to_file": False},
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
return default_config
|
|
403
|
+
|
|
404
|
+
async def run_indexing_workflow(
|
|
405
|
+
self,
|
|
406
|
+
paper_dir: str,
|
|
407
|
+
initial_plan_path: Optional[str] = None,
|
|
408
|
+
config_path: str = "mcp_agent.secrets.yaml",
|
|
409
|
+
) -> Dict[str, Any]:
|
|
410
|
+
"""
|
|
411
|
+
运行完整的代码索引工作流
|
|
412
|
+
Run the complete code indexing workflow
|
|
413
|
+
|
|
414
|
+
Args:
|
|
415
|
+
paper_dir: 论文目录路径
|
|
416
|
+
initial_plan_path: 初始计划文件路径(可选)
|
|
417
|
+
config_path: API配置文件路径
|
|
418
|
+
|
|
419
|
+
Returns:
|
|
420
|
+
索引结果字典
|
|
421
|
+
"""
|
|
422
|
+
try:
|
|
423
|
+
self.logger.info("🚀 开始代码库索引工作流...")
|
|
424
|
+
|
|
425
|
+
# 步骤1:确定初始计划文件路径
|
|
426
|
+
if not initial_plan_path:
|
|
427
|
+
initial_plan_path = os.path.join(paper_dir, "initial_plan.txt")
|
|
428
|
+
|
|
429
|
+
# 步骤2:加载目标结构
|
|
430
|
+
if os.path.exists(initial_plan_path):
|
|
431
|
+
self.logger.info(f"📐 从 {initial_plan_path} 加载目标结构")
|
|
432
|
+
target_structure = self.load_target_structure_from_plan(
|
|
433
|
+
initial_plan_path
|
|
434
|
+
)
|
|
435
|
+
else:
|
|
436
|
+
self.logger.warning(f"⚠️ 初始计划文件不存在: {initial_plan_path}")
|
|
437
|
+
self.logger.info("📐 使用默认目标结构")
|
|
438
|
+
target_structure = self.get_default_target_structure()
|
|
439
|
+
|
|
440
|
+
# 步骤3:检查代码库路径
|
|
441
|
+
code_base_path = os.path.join(paper_dir, "code_base")
|
|
442
|
+
if not os.path.exists(code_base_path):
|
|
443
|
+
self.logger.error(f"❌ 代码库路径不存在: {code_base_path}")
|
|
444
|
+
return {
|
|
445
|
+
"status": "error",
|
|
446
|
+
"message": f"Code base path does not exist: {code_base_path}",
|
|
447
|
+
"output_files": {},
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
# 步骤4:创建输出目录
|
|
451
|
+
output_dir = os.path.join(paper_dir, "indexes")
|
|
452
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
453
|
+
|
|
454
|
+
# 步骤5:加载配置
|
|
455
|
+
indexer_config = self.load_or_create_indexer_config(paper_dir)
|
|
456
|
+
|
|
457
|
+
self.logger.info(f"📁 代码库路径: {code_base_path}")
|
|
458
|
+
self.logger.info(f"📤 输出目录: {output_dir}")
|
|
459
|
+
|
|
460
|
+
# 步骤6:创建代码索引器
|
|
461
|
+
self.indexer = CodeIndexer(
|
|
462
|
+
code_base_path=code_base_path,
|
|
463
|
+
target_structure=target_structure,
|
|
464
|
+
output_dir=output_dir,
|
|
465
|
+
config_path=config_path,
|
|
466
|
+
enable_pre_filtering=True,
|
|
467
|
+
)
|
|
468
|
+
|
|
469
|
+
# 应用配置设置 / Apply configuration settings
|
|
470
|
+
self.indexer.indexer_config = indexer_config
|
|
471
|
+
|
|
472
|
+
# 直接设置配置属性到索引器 / Directly set configuration attributes to indexer
|
|
473
|
+
if "file_analysis" in indexer_config:
|
|
474
|
+
file_config = indexer_config["file_analysis"]
|
|
475
|
+
self.indexer.supported_extensions = set(
|
|
476
|
+
file_config.get(
|
|
477
|
+
"supported_extensions", self.indexer.supported_extensions
|
|
478
|
+
)
|
|
479
|
+
)
|
|
480
|
+
self.indexer.skip_directories = set(
|
|
481
|
+
file_config.get("skip_directories", self.indexer.skip_directories)
|
|
482
|
+
)
|
|
483
|
+
self.indexer.max_file_size = file_config.get(
|
|
484
|
+
"max_file_size", self.indexer.max_file_size
|
|
485
|
+
)
|
|
486
|
+
self.indexer.max_content_length = file_config.get(
|
|
487
|
+
"max_content_length", self.indexer.max_content_length
|
|
488
|
+
)
|
|
489
|
+
|
|
490
|
+
if "llm" in indexer_config:
|
|
491
|
+
llm_config = indexer_config["llm"]
|
|
492
|
+
self.indexer.model_provider = llm_config.get(
|
|
493
|
+
"model_provider", self.indexer.model_provider
|
|
494
|
+
)
|
|
495
|
+
self.indexer.llm_max_tokens = llm_config.get(
|
|
496
|
+
"max_tokens", self.indexer.llm_max_tokens
|
|
497
|
+
)
|
|
498
|
+
self.indexer.llm_temperature = llm_config.get(
|
|
499
|
+
"temperature", self.indexer.llm_temperature
|
|
500
|
+
)
|
|
501
|
+
self.indexer.request_delay = llm_config.get(
|
|
502
|
+
"request_delay", self.indexer.request_delay
|
|
503
|
+
)
|
|
504
|
+
self.indexer.max_retries = llm_config.get(
|
|
505
|
+
"max_retries", self.indexer.max_retries
|
|
506
|
+
)
|
|
507
|
+
self.indexer.retry_delay = llm_config.get(
|
|
508
|
+
"retry_delay", self.indexer.retry_delay
|
|
509
|
+
)
|
|
510
|
+
|
|
511
|
+
if "relationships" in indexer_config:
|
|
512
|
+
rel_config = indexer_config["relationships"]
|
|
513
|
+
self.indexer.min_confidence_score = rel_config.get(
|
|
514
|
+
"min_confidence_score", self.indexer.min_confidence_score
|
|
515
|
+
)
|
|
516
|
+
self.indexer.high_confidence_threshold = rel_config.get(
|
|
517
|
+
"high_confidence_threshold", self.indexer.high_confidence_threshold
|
|
518
|
+
)
|
|
519
|
+
self.indexer.relationship_types = rel_config.get(
|
|
520
|
+
"relationship_types", self.indexer.relationship_types
|
|
521
|
+
)
|
|
522
|
+
|
|
523
|
+
if "performance" in indexer_config:
|
|
524
|
+
perf_config = indexer_config["performance"]
|
|
525
|
+
self.indexer.enable_concurrent_analysis = perf_config.get(
|
|
526
|
+
"enable_concurrent_analysis",
|
|
527
|
+
self.indexer.enable_concurrent_analysis,
|
|
528
|
+
)
|
|
529
|
+
self.indexer.max_concurrent_files = perf_config.get(
|
|
530
|
+
"max_concurrent_files", self.indexer.max_concurrent_files
|
|
531
|
+
)
|
|
532
|
+
self.indexer.enable_content_caching = perf_config.get(
|
|
533
|
+
"enable_content_caching", self.indexer.enable_content_caching
|
|
534
|
+
)
|
|
535
|
+
self.indexer.max_cache_size = perf_config.get(
|
|
536
|
+
"max_cache_size", self.indexer.max_cache_size
|
|
537
|
+
)
|
|
538
|
+
|
|
539
|
+
if "debug" in indexer_config:
|
|
540
|
+
debug_config = indexer_config["debug"]
|
|
541
|
+
self.indexer.verbose_output = debug_config.get(
|
|
542
|
+
"verbose_output", self.indexer.verbose_output
|
|
543
|
+
)
|
|
544
|
+
self.indexer.save_raw_responses = debug_config.get(
|
|
545
|
+
"save_raw_responses", self.indexer.save_raw_responses
|
|
546
|
+
)
|
|
547
|
+
self.indexer.mock_llm_responses = debug_config.get(
|
|
548
|
+
"mock_llm_responses", self.indexer.mock_llm_responses
|
|
549
|
+
)
|
|
550
|
+
|
|
551
|
+
if "output" in indexer_config:
|
|
552
|
+
output_config = indexer_config["output"]
|
|
553
|
+
self.indexer.generate_summary = output_config.get(
|
|
554
|
+
"generate_summary", self.indexer.generate_summary
|
|
555
|
+
)
|
|
556
|
+
self.indexer.generate_statistics = output_config.get(
|
|
557
|
+
"generate_statistics", self.indexer.generate_statistics
|
|
558
|
+
)
|
|
559
|
+
self.indexer.include_metadata = output_config.get(
|
|
560
|
+
"include_metadata", self.indexer.include_metadata
|
|
561
|
+
)
|
|
562
|
+
|
|
563
|
+
self.logger.info("🔧 索引器配置完成")
|
|
564
|
+
self.logger.info(f"🤖 模型提供商: {self.indexer.model_provider}")
|
|
565
|
+
self.logger.info(
|
|
566
|
+
f"⚡ 并发分析: {'启用' if self.indexer.enable_concurrent_analysis else '禁用'}"
|
|
567
|
+
)
|
|
568
|
+
self.logger.info(
|
|
569
|
+
f"🗄️ 内容缓存: {'启用' if self.indexer.enable_content_caching else '禁用'}"
|
|
570
|
+
)
|
|
571
|
+
self.logger.info(
|
|
572
|
+
f"🔍 预过滤: {'启用' if self.indexer.enable_pre_filtering else '禁用'}"
|
|
573
|
+
)
|
|
574
|
+
|
|
575
|
+
self.logger.info("=" * 60)
|
|
576
|
+
self.logger.info("🚀 开始代码索引过程...")
|
|
577
|
+
|
|
578
|
+
# 步骤7:构建所有索引
|
|
579
|
+
output_files = await self.indexer.build_all_indexes()
|
|
580
|
+
|
|
581
|
+
# 步骤8:生成摘要报告
|
|
582
|
+
if output_files:
|
|
583
|
+
summary_report = self.indexer.generate_summary_report(output_files)
|
|
584
|
+
|
|
585
|
+
self.logger.info("=" * 60)
|
|
586
|
+
self.logger.info("✅ 索引完成成功!")
|
|
587
|
+
self.logger.info(f"📊 处理了 {len(output_files)} 个仓库")
|
|
588
|
+
self.logger.info("📁 生成的索引文件:")
|
|
589
|
+
for repo_name, file_path in output_files.items():
|
|
590
|
+
self.logger.info(f" 📄 {repo_name}: {file_path}")
|
|
591
|
+
self.logger.info(f"📋 摘要报告: {summary_report}")
|
|
592
|
+
|
|
593
|
+
# 统计信息(如果启用)
|
|
594
|
+
if self.indexer.generate_statistics:
|
|
595
|
+
self.logger.info("\n📈 处理统计:")
|
|
596
|
+
total_relationships = 0
|
|
597
|
+
high_confidence_relationships = 0
|
|
598
|
+
|
|
599
|
+
for file_path in output_files.values():
|
|
600
|
+
try:
|
|
601
|
+
with open(file_path, "r", encoding="utf-8") as f:
|
|
602
|
+
index_data = json.load(f)
|
|
603
|
+
relationships = index_data.get("relationships", [])
|
|
604
|
+
total_relationships += len(relationships)
|
|
605
|
+
high_confidence_relationships += len(
|
|
606
|
+
[
|
|
607
|
+
r
|
|
608
|
+
for r in relationships
|
|
609
|
+
if r.get("confidence_score", 0)
|
|
610
|
+
> self.indexer.high_confidence_threshold
|
|
611
|
+
]
|
|
612
|
+
)
|
|
613
|
+
except Exception as e:
|
|
614
|
+
self.logger.warning(
|
|
615
|
+
f" ⚠️ 无法从 {file_path} 加载统计: {e}"
|
|
616
|
+
)
|
|
617
|
+
|
|
618
|
+
self.logger.info(f" 🔗 找到的总关系数: {total_relationships}")
|
|
619
|
+
self.logger.info(
|
|
620
|
+
f" ⭐ 高置信度关系: {high_confidence_relationships}"
|
|
621
|
+
)
|
|
622
|
+
self.logger.info(
|
|
623
|
+
f" 📊 每个仓库的平均关系: {total_relationships / len(output_files) if output_files else 0:.1f}"
|
|
624
|
+
)
|
|
625
|
+
|
|
626
|
+
self.logger.info("\n🎉 代码索引过程成功完成!")
|
|
627
|
+
|
|
628
|
+
return {
|
|
629
|
+
"status": "success",
|
|
630
|
+
"message": f"Successfully indexed {len(output_files)} repositories",
|
|
631
|
+
"output_files": output_files,
|
|
632
|
+
"summary_report": summary_report,
|
|
633
|
+
"statistics": {
|
|
634
|
+
"total_repositories": len(output_files),
|
|
635
|
+
"total_relationships": total_relationships,
|
|
636
|
+
"high_confidence_relationships": high_confidence_relationships,
|
|
637
|
+
}
|
|
638
|
+
if self.indexer.generate_statistics
|
|
639
|
+
else None,
|
|
640
|
+
}
|
|
641
|
+
else:
|
|
642
|
+
self.logger.warning("⚠️ 未生成索引文件")
|
|
643
|
+
return {
|
|
644
|
+
"status": "warning",
|
|
645
|
+
"message": "No index files were generated",
|
|
646
|
+
"output_files": {},
|
|
647
|
+
}
|
|
648
|
+
|
|
649
|
+
except Exception as e:
|
|
650
|
+
self.logger.error(f"❌ 索引工作流失败: {e}")
|
|
651
|
+
# 如果有详细的错误信息,记录下来
|
|
652
|
+
import traceback
|
|
653
|
+
|
|
654
|
+
self.logger.error(f"详细错误信息: {traceback.format_exc()}")
|
|
655
|
+
return {"status": "error", "message": str(e), "output_files": {}}
|
|
656
|
+
|
|
657
|
+
def print_banner(self):
|
|
658
|
+
"""打印应用横幅"""
|
|
659
|
+
banner = """
|
|
660
|
+
╔═══════════════════════════════════════════════════════════════════════╗
|
|
661
|
+
║ 🔍 Codebase Index Workflow v1.0 ║
|
|
662
|
+
║ Intelligent Code Relationship Analysis Tool ║
|
|
663
|
+
╠═══════════════════════════════════════════════════════════════════════╣
|
|
664
|
+
║ 📁 分析现有代码库 / Analyzes existing codebases ║
|
|
665
|
+
║ 🔗 与目标结构建立智能关系 / Builds intelligent relationships ║
|
|
666
|
+
║ 🤖 由LLM分析驱动 / Powered by LLM analysis ║
|
|
667
|
+
║ 📊 生成详细的JSON索引 / Generates detailed JSON indexes ║
|
|
668
|
+
║ 🎯 为代码复现提供参考 / Provides reference for code reproduction ║
|
|
669
|
+
╚═══════════════════════════════════════════════════════════════════════╝
|
|
670
|
+
"""
|
|
671
|
+
print(banner)
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
# 便捷函数,用于直接调用工作流
|
|
675
|
+
async def run_codebase_indexing(
|
|
676
|
+
paper_dir: str,
|
|
677
|
+
initial_plan_path: Optional[str] = None,
|
|
678
|
+
config_path: str = "mcp_agent.secrets.yaml",
|
|
679
|
+
logger=None,
|
|
680
|
+
) -> Dict[str, Any]:
|
|
681
|
+
"""
|
|
682
|
+
运行代码库索引的便捷函数
|
|
683
|
+
Convenience function to run codebase indexing
|
|
684
|
+
|
|
685
|
+
Args:
|
|
686
|
+
paper_dir: 论文目录路径
|
|
687
|
+
initial_plan_path: 初始计划文件路径(可选)
|
|
688
|
+
config_path: API配置文件路径
|
|
689
|
+
logger: 日志记录器实例(可选)
|
|
690
|
+
|
|
691
|
+
Returns:
|
|
692
|
+
索引结果字典
|
|
693
|
+
"""
|
|
694
|
+
workflow = CodebaseIndexWorkflow(logger=logger)
|
|
695
|
+
workflow.print_banner()
|
|
696
|
+
|
|
697
|
+
return await workflow.run_indexing_workflow(
|
|
698
|
+
paper_dir=paper_dir,
|
|
699
|
+
initial_plan_path=initial_plan_path,
|
|
700
|
+
config_path=config_path,
|
|
701
|
+
)
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
# 用于测试的主函数
|
|
705
|
+
async def main():
|
|
706
|
+
"""主函数用于测试工作流"""
|
|
707
|
+
import logging
|
|
708
|
+
|
|
709
|
+
# 设置日志
|
|
710
|
+
logging.basicConfig(level=logging.INFO)
|
|
711
|
+
logger = logging.getLogger(__name__)
|
|
712
|
+
|
|
713
|
+
# 测试参数
|
|
714
|
+
paper_dir = "./deepcode_lab/papers/2"
|
|
715
|
+
initial_plan_path = os.path.join(paper_dir, "initial_plan.txt")
|
|
716
|
+
|
|
717
|
+
# 运行工作流
|
|
718
|
+
result = await run_codebase_indexing(
|
|
719
|
+
paper_dir=paper_dir, initial_plan_path=initial_plan_path, logger=logger
|
|
720
|
+
)
|
|
721
|
+
|
|
722
|
+
logger.info(f"索引结果: {result}")
|
|
723
|
+
|
|
724
|
+
|
|
725
|
+
if __name__ == "__main__":
|
|
726
|
+
asyncio.run(main())
|