deepcode-hku 1.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. cli/__init__.py +18 -0
  2. cli/cli_app.py +296 -0
  3. cli/cli_interface.py +744 -0
  4. cli/cli_launcher.py +155 -0
  5. cli/main_cli.py +243 -0
  6. cli/workflows/__init__.py +11 -0
  7. cli/workflows/cli_workflow_adapter.py +336 -0
  8. deepcode.py +219 -0
  9. deepcode_hku-1.0.1.dist-info/METADATA +695 -0
  10. deepcode_hku-1.0.1.dist-info/RECORD +44 -0
  11. deepcode_hku-1.0.1.dist-info/WHEEL +5 -0
  12. deepcode_hku-1.0.1.dist-info/entry_points.txt +2 -0
  13. deepcode_hku-1.0.1.dist-info/licenses/LICENSE +21 -0
  14. deepcode_hku-1.0.1.dist-info/top_level.txt +6 -0
  15. tools/__init__.py +0 -0
  16. tools/code_implementation_server.py +1045 -0
  17. tools/code_indexer.py +1657 -0
  18. tools/code_reference_indexer.py +486 -0
  19. tools/command_executor.py +324 -0
  20. tools/git_command.py +356 -0
  21. tools/pdf_converter.py +640 -0
  22. tools/pdf_downloader.py +1370 -0
  23. tools/pdf_utils.py +52 -0
  24. ui/__init__.py +43 -0
  25. ui/app.py +13 -0
  26. ui/components.py +1450 -0
  27. ui/handlers.py +773 -0
  28. ui/layout.py +106 -0
  29. ui/streamlit_app.py +38 -0
  30. ui/styles.py +2116 -0
  31. utils/__init__.py +17 -0
  32. utils/cli_interface.py +459 -0
  33. utils/dialogue_logger.py +671 -0
  34. utils/file_processor.py +426 -0
  35. utils/simple_llm_logger.py +198 -0
  36. workflows/__init__.py +31 -0
  37. workflows/agent_orchestration_engine.py +1371 -0
  38. workflows/agents/__init__.py +13 -0
  39. workflows/agents/code_implementation_agent.py +1093 -0
  40. workflows/agents/memory_agent_concise.py +923 -0
  41. workflows/agents/memory_agent_concise_index.py +935 -0
  42. workflows/code_implementation_workflow.py +924 -0
  43. workflows/code_implementation_workflow_index.py +931 -0
  44. workflows/codebase_index_workflow.py +726 -0
@@ -0,0 +1,726 @@
1
+ """
2
+ Codebase Index Workflow
3
+ 代码库索引工作流
4
+
5
+ This workflow integrates the functionality of run_indexer.py and code_indexer.py
6
+ to build intelligent relationships between existing codebase and target structure.
7
+
8
+ 该工作流集成了run_indexer.py和code_indexer.py的功能,
9
+ 用于在现有代码库和目标结构之间建立智能关系。
10
+
11
+ Features:
12
+ - 从initial_plan.txt提取目标文件结构 / Extract target file structure from initial_plan.txt
13
+ - 分析代码库并建立索引 / Analyze codebase and build indexes
14
+ - 生成关系映射和统计报告 / Generate relationship mappings and statistical reports
15
+ - 为代码复现提供参考依据 / Provide reference basis for code reproduction
16
+ """
17
+
18
+ import asyncio
19
+ import json
20
+ import logging
21
+ import os
22
+ import re
23
+ import sys
24
+ from pathlib import Path
25
+ from typing import Dict, Any, Optional
26
+ import yaml
27
+
28
+ # 添加tools目录到路径中 / Add tools directory to path
29
+ sys.path.append(str(Path(__file__).parent.parent / "tools"))
30
+
31
+ from tools.code_indexer import CodeIndexer
32
+
33
+
34
+ class CodebaseIndexWorkflow:
35
+ """代码库索引工作流类 / Codebase Index Workflow Class"""
36
+
37
+ def __init__(self, logger=None):
38
+ """
39
+ 初始化工作流
40
+
41
+ Args:
42
+ logger: 日志记录器实例
43
+ """
44
+ self.logger = logger or self._setup_default_logger()
45
+ self.indexer = None
46
+
47
+ def _setup_default_logger(self) -> logging.Logger:
48
+ """设置默认日志记录器"""
49
+ logger = logging.getLogger("CodebaseIndexWorkflow")
50
+ logger.setLevel(logging.INFO)
51
+
52
+ if not logger.handlers:
53
+ handler = logging.StreamHandler()
54
+ formatter = logging.Formatter(
55
+ "%(asctime)s - %(name)s - %(levelname)s - %(message)s"
56
+ )
57
+ handler.setFormatter(formatter)
58
+ logger.addHandler(handler)
59
+
60
+ return logger
61
+
62
+ def extract_file_tree_from_plan(self, plan_content: str) -> Optional[str]:
63
+ """
64
+ 从initial_plan.txt内容中提取文件树结构
65
+ Extract file tree structure from initial_plan.txt content
66
+
67
+ Args:
68
+ plan_content: Content of the initial_plan.txt file
69
+
70
+ Returns:
71
+ Extracted file tree structure as string
72
+ """
73
+ # 查找文件结构部分,特别是"## File Structure"格式
74
+ file_structure_pattern = r"## File Structure[^\n]*\n```[^\n]*\n(.*?)\n```"
75
+
76
+ match = re.search(file_structure_pattern, plan_content, re.DOTALL)
77
+ if match:
78
+ file_tree = match.group(1).strip()
79
+ lines = file_tree.split("\n")
80
+
81
+ # 清理树结构 - 移除空行和不属于结构的注释
82
+ cleaned_lines = []
83
+ for line in lines:
84
+ # 保留树结构的行
85
+ if line.strip() and (
86
+ any(char in line for char in ["├──", "└──", "│"])
87
+ or line.strip().endswith("/")
88
+ or "." in line.split("/")[-1] # 有文件扩展名
89
+ or line.strip().endswith(".py")
90
+ or line.strip().endswith(".txt")
91
+ or line.strip().endswith(".md")
92
+ or line.strip().endswith(".yaml")
93
+ ):
94
+ cleaned_lines.append(line)
95
+
96
+ if len(cleaned_lines) >= 5:
97
+ file_tree = "\n".join(cleaned_lines)
98
+ self.logger.info(
99
+ f"📊 从## File Structure部分提取文件树结构 ({len(cleaned_lines)} lines)"
100
+ )
101
+ return file_tree
102
+
103
+ # 备用方案:查找包含项目结构的任何代码块
104
+ code_block_patterns = [
105
+ r"```[^\n]*\n(rice_framework/.*?(?:├──|└──).*?)\n```",
106
+ r"```[^\n]*\n(project/.*?(?:├──|└──).*?)\n```",
107
+ r"```[^\n]*\n(src/.*?(?:├──|└──).*?)\n```",
108
+ r"```[^\n]*\n(.*?(?:├──|└──).*?(?:\.py|\.txt|\.md|\.yaml).*?)\n```",
109
+ ]
110
+
111
+ for pattern in code_block_patterns:
112
+ match = re.search(pattern, plan_content, re.DOTALL)
113
+ if match:
114
+ file_tree = match.group(1).strip()
115
+ lines = [line for line in file_tree.split("\n") if line.strip()]
116
+ if len(lines) >= 5:
117
+ self.logger.info(
118
+ f"📊 从代码块中提取文件树结构 ({len(lines)} lines)"
119
+ )
120
+ return file_tree
121
+
122
+ # 最终备用方案:从文件提及中提取文件路径并创建基本结构
123
+ self.logger.warning("⚠️ 未找到标准文件树,尝试从文件提及中提取...")
124
+
125
+ # 在整个文档中查找反引号中的文件路径
126
+ file_mentions = re.findall(
127
+ r"`([^`]*(?:\.py|\.txt|\.md|\.yaml|\.yml)[^`]*)`", plan_content
128
+ )
129
+
130
+ if file_mentions:
131
+ # 将文件组织成目录结构
132
+ dirs = set()
133
+ files_by_dir = {}
134
+
135
+ for file_path in file_mentions:
136
+ file_path = file_path.strip()
137
+ if "/" in file_path:
138
+ dir_path = "/".join(file_path.split("/")[:-1])
139
+ filename = file_path.split("/")[-1]
140
+ dirs.add(dir_path)
141
+ if dir_path not in files_by_dir:
142
+ files_by_dir[dir_path] = []
143
+ files_by_dir[dir_path].append(filename)
144
+ else:
145
+ if "root" not in files_by_dir:
146
+ files_by_dir["root"] = []
147
+ files_by_dir["root"].append(file_path)
148
+
149
+ # 创建树结构
150
+ structure_lines = []
151
+
152
+ # 确定根目录名称
153
+ root_name = (
154
+ "rice_framework"
155
+ if any("rice" in f for f in file_mentions)
156
+ else "project"
157
+ )
158
+ structure_lines.append(f"{root_name}/")
159
+
160
+ # 添加目录和文件
161
+ sorted_dirs = sorted(dirs) if dirs else []
162
+ for i, dir_path in enumerate(sorted_dirs):
163
+ is_last_dir = i == len(sorted_dirs) - 1
164
+ prefix = "└──" if is_last_dir else "├──"
165
+ structure_lines.append(f"{prefix} {dir_path}/")
166
+
167
+ if dir_path in files_by_dir:
168
+ files = sorted(files_by_dir[dir_path])
169
+ for j, filename in enumerate(files):
170
+ is_last_file = j == len(files) - 1
171
+ if is_last_dir:
172
+ file_prefix = " └──" if is_last_file else " ├──"
173
+ else:
174
+ file_prefix = "│ └──" if is_last_file else "│ ├──"
175
+ structure_lines.append(f"{file_prefix} {filename}")
176
+
177
+ # 添加根文件(如果有)
178
+ if "root" in files_by_dir:
179
+ root_files = sorted(files_by_dir["root"])
180
+ for i, filename in enumerate(root_files):
181
+ is_last = (i == len(root_files) - 1) and not sorted_dirs
182
+ prefix = "└──" if is_last else "├──"
183
+ structure_lines.append(f"{prefix} {filename}")
184
+
185
+ if len(structure_lines) >= 3:
186
+ file_tree = "\n".join(structure_lines)
187
+ self.logger.info(
188
+ f"📊 从文件提及生成文件树 ({len(structure_lines)} lines)"
189
+ )
190
+ return file_tree
191
+
192
+ # 如果未找到文件树,返回None
193
+ self.logger.warning("⚠️ 在初始计划中未找到文件树结构")
194
+ return None
195
+
196
+ def load_target_structure_from_plan(self, plan_path: str) -> str:
197
+ """
198
+ 从initial_plan.txt加载目标结构并提取文件树
199
+ Load target structure from initial_plan.txt and extract file tree
200
+
201
+ Args:
202
+ plan_path: Path to initial_plan.txt file
203
+
204
+ Returns:
205
+ Extracted file tree structure
206
+ """
207
+ try:
208
+ # 加载完整的计划内容
209
+ with open(plan_path, "r", encoding="utf-8") as f:
210
+ plan_content = f.read()
211
+
212
+ self.logger.info(f"📄 已加载初始计划 ({len(plan_content)} characters)")
213
+
214
+ # 提取文件树结构
215
+ file_tree = self.extract_file_tree_from_plan(plan_content)
216
+
217
+ if file_tree:
218
+ self.logger.info("✅ 成功从初始计划中提取文件树")
219
+ self.logger.info("📋 提取结构预览:")
220
+ # 显示提取树的前几行
221
+ preview_lines = file_tree.split("\n")[:8]
222
+ for line in preview_lines:
223
+ self.logger.info(f" {line}")
224
+ if len(file_tree.split("\n")) > 8:
225
+ self.logger.info(f" ... 还有 {len(file_tree.split('\n')) - 8} 行")
226
+ return file_tree
227
+ else:
228
+ self.logger.warning("⚠️ 无法从初始计划中提取文件树")
229
+ self.logger.info("🔄 回退到默认目标结构")
230
+ return self.get_default_target_structure()
231
+
232
+ except Exception as e:
233
+ self.logger.error(f"❌ 加载初始计划文件失败 {plan_path}: {e}")
234
+ self.logger.info("🔄 回退到默认目标结构")
235
+ return self.get_default_target_structure()
236
+
237
+ def get_default_target_structure(self) -> str:
238
+ """获取默认目标结构"""
239
+ return """
240
+ project/
241
+ ├── src/
242
+ │ ├── core/
243
+ │ │ ├── gcn.py # GCN encoder
244
+ │ │ ├── diffusion.py # forward/reverse processes
245
+ │ │ ├── denoiser.py # denoising MLP
246
+ │ │ └── fusion.py # fusion combiner
247
+ │ ├── models/ # model wrapper classes
248
+ │ │ └── recdiff.py
249
+ │ ├── utils/
250
+ │ │ ├── data.py # loading & preprocessing
251
+ │ │ ├── predictor.py # scoring functions
252
+ │ │ ├── loss.py # loss functions
253
+ │ │ ├── metrics.py # NDCG, Recall etc.
254
+ │ │ └── sched.py # beta/alpha schedule utils
255
+ │ └── configs/
256
+ │ └── default.yaml # hyperparameters, paths
257
+ ├── tests/
258
+ │ ├── test_gcn.py
259
+ │ ├── test_diffusion.py
260
+ │ ├── test_denoiser.py
261
+ │ ├── test_loss.py
262
+ │ └── test_pipeline.py
263
+ ├── docs/
264
+ │ ├── architecture.md
265
+ │ ├── api_reference.md
266
+ │ └── README.md
267
+ ├── experiments/
268
+ │ ├── run_experiment.py
269
+ │ └── notebooks/
270
+ │ └── analysis.ipynb
271
+ ├── requirements.txt
272
+ └── setup.py
273
+ """
274
+
275
+ def load_or_create_indexer_config(self, paper_dir: str) -> Dict[str, Any]:
276
+ """
277
+ 加载或创建索引器配置
278
+ Load or create indexer configuration
279
+
280
+ Args:
281
+ paper_dir: 论文目录路径
282
+
283
+ Returns:
284
+ 配置字典
285
+ """
286
+ # 尝试加载现有的配置文件
287
+ config_path = Path(__file__).parent.parent / "tools" / "indexer_config.yaml"
288
+
289
+ try:
290
+ if config_path.exists():
291
+ with open(config_path, "r", encoding="utf-8") as f:
292
+ config = yaml.safe_load(f)
293
+
294
+ # 更新路径配置为当前论文目录
295
+ if "paths" not in config:
296
+ config["paths"] = {}
297
+ config["paths"]["code_base_path"] = os.path.join(paper_dir, "code_base")
298
+ config["paths"]["output_dir"] = os.path.join(paper_dir, "indexes")
299
+
300
+ # 调整性能设置以适应工作流
301
+ if "performance" in config:
302
+ config["performance"]["enable_concurrent_analysis"] = (
303
+ False # 禁用并发以避免API限制
304
+ )
305
+ if "debug" in config:
306
+ config["debug"]["verbose_output"] = True # 启用详细输出
307
+ if "llm" in config:
308
+ config["llm"]["request_delay"] = 0.5 # 增加请求间隔
309
+
310
+ self.logger.info(f"已加载配置文件: {config_path}")
311
+ return config
312
+
313
+ except Exception as e:
314
+ self.logger.warning(f"加载配置文件失败: {e}")
315
+
316
+ # 如果加载失败,使用默认配置
317
+ self.logger.info("使用默认配置")
318
+ default_config = {
319
+ "paths": {
320
+ "code_base_path": os.path.join(paper_dir, "code_base"),
321
+ "output_dir": os.path.join(paper_dir, "indexes"),
322
+ },
323
+ "llm": {
324
+ "model_provider": "anthropic",
325
+ "max_tokens": 4000,
326
+ "temperature": 0.3,
327
+ "request_delay": 0.5, # 增加请求间隔
328
+ "max_retries": 3,
329
+ "retry_delay": 1.0,
330
+ },
331
+ "file_analysis": {
332
+ "max_file_size": 1048576, # 1MB
333
+ "max_content_length": 3000,
334
+ "supported_extensions": [
335
+ ".py",
336
+ ".js",
337
+ ".ts",
338
+ ".java",
339
+ ".cpp",
340
+ ".c",
341
+ ".h",
342
+ ".hpp",
343
+ ".cs",
344
+ ".php",
345
+ ".rb",
346
+ ".go",
347
+ ".rs",
348
+ ".scala",
349
+ ".kt",
350
+ ".yaml",
351
+ ".yml",
352
+ ".json",
353
+ ".xml",
354
+ ".toml",
355
+ ".md",
356
+ ".txt",
357
+ ],
358
+ "skip_directories": [
359
+ "__pycache__",
360
+ "node_modules",
361
+ "target",
362
+ "build",
363
+ "dist",
364
+ "venv",
365
+ "env",
366
+ ".git",
367
+ ".svn",
368
+ "data",
369
+ "datasets",
370
+ ],
371
+ },
372
+ "relationships": {
373
+ "min_confidence_score": 0.3,
374
+ "high_confidence_threshold": 0.7,
375
+ "relationship_types": {
376
+ "direct_match": 1.0,
377
+ "partial_match": 0.8,
378
+ "reference": 0.6,
379
+ "utility": 0.4,
380
+ },
381
+ },
382
+ "performance": {
383
+ "enable_concurrent_analysis": False, # 禁用并发以避免API限制
384
+ "max_concurrent_files": 3,
385
+ "enable_content_caching": True,
386
+ "max_cache_size": 100,
387
+ },
388
+ "debug": {
389
+ "verbose_output": True,
390
+ "save_raw_responses": False,
391
+ "mock_llm_responses": False,
392
+ },
393
+ "output": {
394
+ "generate_summary": True,
395
+ "generate_statistics": True,
396
+ "include_metadata": True,
397
+ "json_indent": 2,
398
+ },
399
+ "logging": {"level": "INFO", "log_to_file": False},
400
+ }
401
+
402
+ return default_config
403
+
404
+ async def run_indexing_workflow(
405
+ self,
406
+ paper_dir: str,
407
+ initial_plan_path: Optional[str] = None,
408
+ config_path: str = "mcp_agent.secrets.yaml",
409
+ ) -> Dict[str, Any]:
410
+ """
411
+ 运行完整的代码索引工作流
412
+ Run the complete code indexing workflow
413
+
414
+ Args:
415
+ paper_dir: 论文目录路径
416
+ initial_plan_path: 初始计划文件路径(可选)
417
+ config_path: API配置文件路径
418
+
419
+ Returns:
420
+ 索引结果字典
421
+ """
422
+ try:
423
+ self.logger.info("🚀 开始代码库索引工作流...")
424
+
425
+ # 步骤1:确定初始计划文件路径
426
+ if not initial_plan_path:
427
+ initial_plan_path = os.path.join(paper_dir, "initial_plan.txt")
428
+
429
+ # 步骤2:加载目标结构
430
+ if os.path.exists(initial_plan_path):
431
+ self.logger.info(f"📐 从 {initial_plan_path} 加载目标结构")
432
+ target_structure = self.load_target_structure_from_plan(
433
+ initial_plan_path
434
+ )
435
+ else:
436
+ self.logger.warning(f"⚠️ 初始计划文件不存在: {initial_plan_path}")
437
+ self.logger.info("📐 使用默认目标结构")
438
+ target_structure = self.get_default_target_structure()
439
+
440
+ # 步骤3:检查代码库路径
441
+ code_base_path = os.path.join(paper_dir, "code_base")
442
+ if not os.path.exists(code_base_path):
443
+ self.logger.error(f"❌ 代码库路径不存在: {code_base_path}")
444
+ return {
445
+ "status": "error",
446
+ "message": f"Code base path does not exist: {code_base_path}",
447
+ "output_files": {},
448
+ }
449
+
450
+ # 步骤4:创建输出目录
451
+ output_dir = os.path.join(paper_dir, "indexes")
452
+ os.makedirs(output_dir, exist_ok=True)
453
+
454
+ # 步骤5:加载配置
455
+ indexer_config = self.load_or_create_indexer_config(paper_dir)
456
+
457
+ self.logger.info(f"📁 代码库路径: {code_base_path}")
458
+ self.logger.info(f"📤 输出目录: {output_dir}")
459
+
460
+ # 步骤6:创建代码索引器
461
+ self.indexer = CodeIndexer(
462
+ code_base_path=code_base_path,
463
+ target_structure=target_structure,
464
+ output_dir=output_dir,
465
+ config_path=config_path,
466
+ enable_pre_filtering=True,
467
+ )
468
+
469
+ # 应用配置设置 / Apply configuration settings
470
+ self.indexer.indexer_config = indexer_config
471
+
472
+ # 直接设置配置属性到索引器 / Directly set configuration attributes to indexer
473
+ if "file_analysis" in indexer_config:
474
+ file_config = indexer_config["file_analysis"]
475
+ self.indexer.supported_extensions = set(
476
+ file_config.get(
477
+ "supported_extensions", self.indexer.supported_extensions
478
+ )
479
+ )
480
+ self.indexer.skip_directories = set(
481
+ file_config.get("skip_directories", self.indexer.skip_directories)
482
+ )
483
+ self.indexer.max_file_size = file_config.get(
484
+ "max_file_size", self.indexer.max_file_size
485
+ )
486
+ self.indexer.max_content_length = file_config.get(
487
+ "max_content_length", self.indexer.max_content_length
488
+ )
489
+
490
+ if "llm" in indexer_config:
491
+ llm_config = indexer_config["llm"]
492
+ self.indexer.model_provider = llm_config.get(
493
+ "model_provider", self.indexer.model_provider
494
+ )
495
+ self.indexer.llm_max_tokens = llm_config.get(
496
+ "max_tokens", self.indexer.llm_max_tokens
497
+ )
498
+ self.indexer.llm_temperature = llm_config.get(
499
+ "temperature", self.indexer.llm_temperature
500
+ )
501
+ self.indexer.request_delay = llm_config.get(
502
+ "request_delay", self.indexer.request_delay
503
+ )
504
+ self.indexer.max_retries = llm_config.get(
505
+ "max_retries", self.indexer.max_retries
506
+ )
507
+ self.indexer.retry_delay = llm_config.get(
508
+ "retry_delay", self.indexer.retry_delay
509
+ )
510
+
511
+ if "relationships" in indexer_config:
512
+ rel_config = indexer_config["relationships"]
513
+ self.indexer.min_confidence_score = rel_config.get(
514
+ "min_confidence_score", self.indexer.min_confidence_score
515
+ )
516
+ self.indexer.high_confidence_threshold = rel_config.get(
517
+ "high_confidence_threshold", self.indexer.high_confidence_threshold
518
+ )
519
+ self.indexer.relationship_types = rel_config.get(
520
+ "relationship_types", self.indexer.relationship_types
521
+ )
522
+
523
+ if "performance" in indexer_config:
524
+ perf_config = indexer_config["performance"]
525
+ self.indexer.enable_concurrent_analysis = perf_config.get(
526
+ "enable_concurrent_analysis",
527
+ self.indexer.enable_concurrent_analysis,
528
+ )
529
+ self.indexer.max_concurrent_files = perf_config.get(
530
+ "max_concurrent_files", self.indexer.max_concurrent_files
531
+ )
532
+ self.indexer.enable_content_caching = perf_config.get(
533
+ "enable_content_caching", self.indexer.enable_content_caching
534
+ )
535
+ self.indexer.max_cache_size = perf_config.get(
536
+ "max_cache_size", self.indexer.max_cache_size
537
+ )
538
+
539
+ if "debug" in indexer_config:
540
+ debug_config = indexer_config["debug"]
541
+ self.indexer.verbose_output = debug_config.get(
542
+ "verbose_output", self.indexer.verbose_output
543
+ )
544
+ self.indexer.save_raw_responses = debug_config.get(
545
+ "save_raw_responses", self.indexer.save_raw_responses
546
+ )
547
+ self.indexer.mock_llm_responses = debug_config.get(
548
+ "mock_llm_responses", self.indexer.mock_llm_responses
549
+ )
550
+
551
+ if "output" in indexer_config:
552
+ output_config = indexer_config["output"]
553
+ self.indexer.generate_summary = output_config.get(
554
+ "generate_summary", self.indexer.generate_summary
555
+ )
556
+ self.indexer.generate_statistics = output_config.get(
557
+ "generate_statistics", self.indexer.generate_statistics
558
+ )
559
+ self.indexer.include_metadata = output_config.get(
560
+ "include_metadata", self.indexer.include_metadata
561
+ )
562
+
563
+ self.logger.info("🔧 索引器配置完成")
564
+ self.logger.info(f"🤖 模型提供商: {self.indexer.model_provider}")
565
+ self.logger.info(
566
+ f"⚡ 并发分析: {'启用' if self.indexer.enable_concurrent_analysis else '禁用'}"
567
+ )
568
+ self.logger.info(
569
+ f"🗄️ 内容缓存: {'启用' if self.indexer.enable_content_caching else '禁用'}"
570
+ )
571
+ self.logger.info(
572
+ f"🔍 预过滤: {'启用' if self.indexer.enable_pre_filtering else '禁用'}"
573
+ )
574
+
575
+ self.logger.info("=" * 60)
576
+ self.logger.info("🚀 开始代码索引过程...")
577
+
578
+ # 步骤7:构建所有索引
579
+ output_files = await self.indexer.build_all_indexes()
580
+
581
+ # 步骤8:生成摘要报告
582
+ if output_files:
583
+ summary_report = self.indexer.generate_summary_report(output_files)
584
+
585
+ self.logger.info("=" * 60)
586
+ self.logger.info("✅ 索引完成成功!")
587
+ self.logger.info(f"📊 处理了 {len(output_files)} 个仓库")
588
+ self.logger.info("📁 生成的索引文件:")
589
+ for repo_name, file_path in output_files.items():
590
+ self.logger.info(f" 📄 {repo_name}: {file_path}")
591
+ self.logger.info(f"📋 摘要报告: {summary_report}")
592
+
593
+ # 统计信息(如果启用)
594
+ if self.indexer.generate_statistics:
595
+ self.logger.info("\n📈 处理统计:")
596
+ total_relationships = 0
597
+ high_confidence_relationships = 0
598
+
599
+ for file_path in output_files.values():
600
+ try:
601
+ with open(file_path, "r", encoding="utf-8") as f:
602
+ index_data = json.load(f)
603
+ relationships = index_data.get("relationships", [])
604
+ total_relationships += len(relationships)
605
+ high_confidence_relationships += len(
606
+ [
607
+ r
608
+ for r in relationships
609
+ if r.get("confidence_score", 0)
610
+ > self.indexer.high_confidence_threshold
611
+ ]
612
+ )
613
+ except Exception as e:
614
+ self.logger.warning(
615
+ f" ⚠️ 无法从 {file_path} 加载统计: {e}"
616
+ )
617
+
618
+ self.logger.info(f" 🔗 找到的总关系数: {total_relationships}")
619
+ self.logger.info(
620
+ f" ⭐ 高置信度关系: {high_confidence_relationships}"
621
+ )
622
+ self.logger.info(
623
+ f" 📊 每个仓库的平均关系: {total_relationships / len(output_files) if output_files else 0:.1f}"
624
+ )
625
+
626
+ self.logger.info("\n🎉 代码索引过程成功完成!")
627
+
628
+ return {
629
+ "status": "success",
630
+ "message": f"Successfully indexed {len(output_files)} repositories",
631
+ "output_files": output_files,
632
+ "summary_report": summary_report,
633
+ "statistics": {
634
+ "total_repositories": len(output_files),
635
+ "total_relationships": total_relationships,
636
+ "high_confidence_relationships": high_confidence_relationships,
637
+ }
638
+ if self.indexer.generate_statistics
639
+ else None,
640
+ }
641
+ else:
642
+ self.logger.warning("⚠️ 未生成索引文件")
643
+ return {
644
+ "status": "warning",
645
+ "message": "No index files were generated",
646
+ "output_files": {},
647
+ }
648
+
649
+ except Exception as e:
650
+ self.logger.error(f"❌ 索引工作流失败: {e}")
651
+ # 如果有详细的错误信息,记录下来
652
+ import traceback
653
+
654
+ self.logger.error(f"详细错误信息: {traceback.format_exc()}")
655
+ return {"status": "error", "message": str(e), "output_files": {}}
656
+
657
+ def print_banner(self):
658
+ """打印应用横幅"""
659
+ banner = """
660
+ ╔═══════════════════════════════════════════════════════════════════════╗
661
+ ║ 🔍 Codebase Index Workflow v1.0 ║
662
+ ║ Intelligent Code Relationship Analysis Tool ║
663
+ ╠═══════════════════════════════════════════════════════════════════════╣
664
+ ║ 📁 分析现有代码库 / Analyzes existing codebases ║
665
+ ║ 🔗 与目标结构建立智能关系 / Builds intelligent relationships ║
666
+ ║ 🤖 由LLM分析驱动 / Powered by LLM analysis ║
667
+ ║ 📊 生成详细的JSON索引 / Generates detailed JSON indexes ║
668
+ ║ 🎯 为代码复现提供参考 / Provides reference for code reproduction ║
669
+ ╚═══════════════════════════════════════════════════════════════════════╝
670
+ """
671
+ print(banner)
672
+
673
+
674
+ # 便捷函数,用于直接调用工作流
675
+ async def run_codebase_indexing(
676
+ paper_dir: str,
677
+ initial_plan_path: Optional[str] = None,
678
+ config_path: str = "mcp_agent.secrets.yaml",
679
+ logger=None,
680
+ ) -> Dict[str, Any]:
681
+ """
682
+ 运行代码库索引的便捷函数
683
+ Convenience function to run codebase indexing
684
+
685
+ Args:
686
+ paper_dir: 论文目录路径
687
+ initial_plan_path: 初始计划文件路径(可选)
688
+ config_path: API配置文件路径
689
+ logger: 日志记录器实例(可选)
690
+
691
+ Returns:
692
+ 索引结果字典
693
+ """
694
+ workflow = CodebaseIndexWorkflow(logger=logger)
695
+ workflow.print_banner()
696
+
697
+ return await workflow.run_indexing_workflow(
698
+ paper_dir=paper_dir,
699
+ initial_plan_path=initial_plan_path,
700
+ config_path=config_path,
701
+ )
702
+
703
+
704
+ # 用于测试的主函数
705
+ async def main():
706
+ """主函数用于测试工作流"""
707
+ import logging
708
+
709
+ # 设置日志
710
+ logging.basicConfig(level=logging.INFO)
711
+ logger = logging.getLogger(__name__)
712
+
713
+ # 测试参数
714
+ paper_dir = "./deepcode_lab/papers/2"
715
+ initial_plan_path = os.path.join(paper_dir, "initial_plan.txt")
716
+
717
+ # 运行工作流
718
+ result = await run_codebase_indexing(
719
+ paper_dir=paper_dir, initial_plan_path=initial_plan_path, logger=logger
720
+ )
721
+
722
+ logger.info(f"索引结果: {result}")
723
+
724
+
725
+ if __name__ == "__main__":
726
+ asyncio.run(main())