deepcode-hku 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli/__init__.py +18 -0
- cli/cli_app.py +296 -0
- cli/cli_interface.py +744 -0
- cli/cli_launcher.py +155 -0
- cli/main_cli.py +243 -0
- cli/workflows/__init__.py +11 -0
- cli/workflows/cli_workflow_adapter.py +336 -0
- deepcode.py +219 -0
- deepcode_hku-1.0.1.dist-info/METADATA +695 -0
- deepcode_hku-1.0.1.dist-info/RECORD +44 -0
- deepcode_hku-1.0.1.dist-info/WHEEL +5 -0
- deepcode_hku-1.0.1.dist-info/entry_points.txt +2 -0
- deepcode_hku-1.0.1.dist-info/licenses/LICENSE +21 -0
- deepcode_hku-1.0.1.dist-info/top_level.txt +6 -0
- tools/__init__.py +0 -0
- tools/code_implementation_server.py +1045 -0
- tools/code_indexer.py +1657 -0
- tools/code_reference_indexer.py +486 -0
- tools/command_executor.py +324 -0
- tools/git_command.py +356 -0
- tools/pdf_converter.py +640 -0
- tools/pdf_downloader.py +1370 -0
- tools/pdf_utils.py +52 -0
- ui/__init__.py +43 -0
- ui/app.py +13 -0
- ui/components.py +1450 -0
- ui/handlers.py +773 -0
- ui/layout.py +106 -0
- ui/streamlit_app.py +38 -0
- ui/styles.py +2116 -0
- utils/__init__.py +17 -0
- utils/cli_interface.py +459 -0
- utils/dialogue_logger.py +671 -0
- utils/file_processor.py +426 -0
- utils/simple_llm_logger.py +198 -0
- workflows/__init__.py +31 -0
- workflows/agent_orchestration_engine.py +1371 -0
- workflows/agents/__init__.py +13 -0
- workflows/agents/code_implementation_agent.py +1093 -0
- workflows/agents/memory_agent_concise.py +923 -0
- workflows/agents/memory_agent_concise_index.py +935 -0
- workflows/code_implementation_workflow.py +924 -0
- workflows/code_implementation_workflow_index.py +931 -0
- workflows/codebase_index_workflow.py +726 -0
tools/code_indexer.py
ADDED
|
@@ -0,0 +1,1657 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Code Indexer for Repository Analysis
|
|
3
|
+
|
|
4
|
+
Analyzes code repositories to build comprehensive indexes for each subdirectory,
|
|
5
|
+
identifying file relationships and reusable components for implementation.
|
|
6
|
+
|
|
7
|
+
Features:
|
|
8
|
+
- Recursive file traversal
|
|
9
|
+
- LLM-powered code similarity analysis
|
|
10
|
+
- JSON-based relationship storage
|
|
11
|
+
- Configurable matching strategies
|
|
12
|
+
- Progress tracking and error handling
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import asyncio
|
|
16
|
+
import json
|
|
17
|
+
import logging
|
|
18
|
+
import os
|
|
19
|
+
import re
|
|
20
|
+
from datetime import datetime
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from dataclasses import dataclass, asdict
|
|
23
|
+
from typing import List, Dict, Any
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class FileRelationship:
|
|
28
|
+
"""Represents a relationship between a repo file and target structure file"""
|
|
29
|
+
|
|
30
|
+
repo_file_path: str
|
|
31
|
+
target_file_path: str
|
|
32
|
+
relationship_type: str # 'direct_match', 'partial_match', 'reference', 'utility'
|
|
33
|
+
confidence_score: float # 0.0 to 1.0
|
|
34
|
+
helpful_aspects: List[str]
|
|
35
|
+
potential_contributions: List[str]
|
|
36
|
+
usage_suggestions: str
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class FileSummary:
|
|
41
|
+
"""Summary information for a repository file"""
|
|
42
|
+
|
|
43
|
+
file_path: str
|
|
44
|
+
file_type: str
|
|
45
|
+
main_functions: List[str]
|
|
46
|
+
key_concepts: List[str]
|
|
47
|
+
dependencies: List[str]
|
|
48
|
+
summary: str
|
|
49
|
+
lines_of_code: int
|
|
50
|
+
last_modified: str
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass
|
|
54
|
+
class RepoIndex:
|
|
55
|
+
"""Complete index for a repository"""
|
|
56
|
+
|
|
57
|
+
repo_name: str
|
|
58
|
+
total_files: int
|
|
59
|
+
file_summaries: List[FileSummary]
|
|
60
|
+
relationships: List[FileRelationship]
|
|
61
|
+
analysis_metadata: Dict[str, Any]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class CodeIndexer:
|
|
65
|
+
"""Main class for building code repository indexes"""
|
|
66
|
+
|
|
67
|
+
def __init__(
|
|
68
|
+
self,
|
|
69
|
+
code_base_path: str = None,
|
|
70
|
+
target_structure: str = None,
|
|
71
|
+
output_dir: str = None,
|
|
72
|
+
config_path: str = "mcp_agent.secrets.yaml",
|
|
73
|
+
indexer_config_path: str = None,
|
|
74
|
+
enable_pre_filtering: bool = True,
|
|
75
|
+
):
|
|
76
|
+
# Load configurations first
|
|
77
|
+
self.config_path = config_path
|
|
78
|
+
self.indexer_config_path = indexer_config_path
|
|
79
|
+
self.api_config = self._load_api_config()
|
|
80
|
+
self.indexer_config = self._load_indexer_config()
|
|
81
|
+
|
|
82
|
+
# Use config paths if not provided as parameters
|
|
83
|
+
paths_config = self.indexer_config.get("paths", {})
|
|
84
|
+
self.code_base_path = Path(
|
|
85
|
+
code_base_path or paths_config.get("code_base_path", "code_base")
|
|
86
|
+
)
|
|
87
|
+
self.output_dir = Path(output_dir or paths_config.get("output_dir", "indexes"))
|
|
88
|
+
self.target_structure = (
|
|
89
|
+
target_structure # This must be provided as it's project-specific
|
|
90
|
+
)
|
|
91
|
+
self.enable_pre_filtering = enable_pre_filtering
|
|
92
|
+
|
|
93
|
+
# LLM clients
|
|
94
|
+
self.llm_client = None
|
|
95
|
+
self.llm_client_type = None
|
|
96
|
+
|
|
97
|
+
# Initialize logger early
|
|
98
|
+
self.logger = self._setup_logger()
|
|
99
|
+
|
|
100
|
+
# Create output directory if it doesn't exist
|
|
101
|
+
self.output_dir.mkdir(parents=True, exist_ok=True)
|
|
102
|
+
|
|
103
|
+
# Load file analysis configuration
|
|
104
|
+
file_analysis_config = self.indexer_config.get("file_analysis", {})
|
|
105
|
+
self.supported_extensions = set(
|
|
106
|
+
file_analysis_config.get(
|
|
107
|
+
"supported_extensions",
|
|
108
|
+
[
|
|
109
|
+
".py",
|
|
110
|
+
".js",
|
|
111
|
+
".ts",
|
|
112
|
+
".java",
|
|
113
|
+
".cpp",
|
|
114
|
+
".c",
|
|
115
|
+
".h",
|
|
116
|
+
".hpp",
|
|
117
|
+
".cs",
|
|
118
|
+
".php",
|
|
119
|
+
".rb",
|
|
120
|
+
".go",
|
|
121
|
+
".rs",
|
|
122
|
+
".scala",
|
|
123
|
+
".kt",
|
|
124
|
+
".swift",
|
|
125
|
+
".m",
|
|
126
|
+
".mm",
|
|
127
|
+
".r",
|
|
128
|
+
".matlab",
|
|
129
|
+
".sql",
|
|
130
|
+
".sh",
|
|
131
|
+
".bat",
|
|
132
|
+
".ps1",
|
|
133
|
+
".yaml",
|
|
134
|
+
".yml",
|
|
135
|
+
".json",
|
|
136
|
+
".xml",
|
|
137
|
+
".toml",
|
|
138
|
+
],
|
|
139
|
+
)
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
self.skip_directories = set(
|
|
143
|
+
file_analysis_config.get(
|
|
144
|
+
"skip_directories",
|
|
145
|
+
[
|
|
146
|
+
"__pycache__",
|
|
147
|
+
"node_modules",
|
|
148
|
+
"target",
|
|
149
|
+
"build",
|
|
150
|
+
"dist",
|
|
151
|
+
"venv",
|
|
152
|
+
"env",
|
|
153
|
+
],
|
|
154
|
+
)
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
self.max_file_size = file_analysis_config.get("max_file_size", 1048576) # 1MB
|
|
158
|
+
self.max_content_length = file_analysis_config.get("max_content_length", 3000)
|
|
159
|
+
|
|
160
|
+
# Load LLM configuration
|
|
161
|
+
llm_config = self.indexer_config.get("llm", {})
|
|
162
|
+
self.model_provider = llm_config.get("model_provider", "anthropic")
|
|
163
|
+
self.llm_max_tokens = llm_config.get("max_tokens", 4000)
|
|
164
|
+
self.llm_temperature = llm_config.get("temperature", 0.3)
|
|
165
|
+
self.llm_system_prompt = llm_config.get(
|
|
166
|
+
"system_prompt",
|
|
167
|
+
"You are a code analysis expert. Provide precise, structured analysis of code relationships and similarities.",
|
|
168
|
+
)
|
|
169
|
+
self.request_delay = llm_config.get("request_delay", 0.1)
|
|
170
|
+
self.max_retries = llm_config.get("max_retries", 3)
|
|
171
|
+
self.retry_delay = llm_config.get("retry_delay", 1.0)
|
|
172
|
+
|
|
173
|
+
# Load relationship configuration
|
|
174
|
+
relationship_config = self.indexer_config.get("relationships", {})
|
|
175
|
+
self.min_confidence_score = relationship_config.get("min_confidence_score", 0.3)
|
|
176
|
+
self.high_confidence_threshold = relationship_config.get(
|
|
177
|
+
"high_confidence_threshold", 0.7
|
|
178
|
+
)
|
|
179
|
+
self.relationship_types = relationship_config.get(
|
|
180
|
+
"relationship_types",
|
|
181
|
+
{
|
|
182
|
+
"direct_match": 1.0,
|
|
183
|
+
"partial_match": 0.8,
|
|
184
|
+
"reference": 0.6,
|
|
185
|
+
"utility": 0.4,
|
|
186
|
+
},
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
# Load performance configuration
|
|
190
|
+
performance_config = self.indexer_config.get("performance", {})
|
|
191
|
+
self.enable_concurrent_analysis = performance_config.get(
|
|
192
|
+
"enable_concurrent_analysis", False
|
|
193
|
+
)
|
|
194
|
+
self.max_concurrent_files = performance_config.get("max_concurrent_files", 5)
|
|
195
|
+
self.enable_content_caching = performance_config.get(
|
|
196
|
+
"enable_content_caching", False
|
|
197
|
+
)
|
|
198
|
+
self.max_cache_size = performance_config.get("max_cache_size", 100)
|
|
199
|
+
|
|
200
|
+
# Load debug configuration
|
|
201
|
+
debug_config = self.indexer_config.get("debug", {})
|
|
202
|
+
self.save_raw_responses = debug_config.get("save_raw_responses", False)
|
|
203
|
+
self.raw_responses_dir = debug_config.get(
|
|
204
|
+
"raw_responses_dir", "debug_responses"
|
|
205
|
+
)
|
|
206
|
+
self.verbose_output = debug_config.get("verbose_output", False)
|
|
207
|
+
self.mock_llm_responses = debug_config.get("mock_llm_responses", False)
|
|
208
|
+
|
|
209
|
+
# Load output configuration
|
|
210
|
+
output_config = self.indexer_config.get("output", {})
|
|
211
|
+
self.generate_summary = output_config.get("generate_summary", True)
|
|
212
|
+
self.generate_statistics = output_config.get("generate_statistics", True)
|
|
213
|
+
self.include_metadata = output_config.get("include_metadata", True)
|
|
214
|
+
self.index_filename_pattern = output_config.get(
|
|
215
|
+
"index_filename_pattern", "{repo_name}_index.json"
|
|
216
|
+
)
|
|
217
|
+
self.summary_filename = output_config.get(
|
|
218
|
+
"summary_filename", "indexing_summary.json"
|
|
219
|
+
)
|
|
220
|
+
self.stats_filename = output_config.get(
|
|
221
|
+
"stats_filename", "indexing_statistics.json"
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
# Initialize caching if enabled
|
|
225
|
+
self.content_cache = {} if self.enable_content_caching else None
|
|
226
|
+
|
|
227
|
+
# Create debug directory if needed
|
|
228
|
+
if self.save_raw_responses:
|
|
229
|
+
Path(self.raw_responses_dir).mkdir(parents=True, exist_ok=True)
|
|
230
|
+
|
|
231
|
+
# Debug logging
|
|
232
|
+
if self.verbose_output:
|
|
233
|
+
self.logger.info(
|
|
234
|
+
f"Initialized CodeIndexer with config: {self.indexer_config_path}"
|
|
235
|
+
)
|
|
236
|
+
self.logger.info(f"Code base path: {self.code_base_path}")
|
|
237
|
+
self.logger.info(f"Output directory: {self.output_dir}")
|
|
238
|
+
self.logger.info(f"Model provider: {self.model_provider}")
|
|
239
|
+
self.logger.info(f"Concurrent analysis: {self.enable_concurrent_analysis}")
|
|
240
|
+
self.logger.info(f"Content caching: {self.enable_content_caching}")
|
|
241
|
+
self.logger.info(f"Mock LLM responses: {self.mock_llm_responses}")
|
|
242
|
+
|
|
243
|
+
def _setup_logger(self) -> logging.Logger:
|
|
244
|
+
"""Setup logging configuration from config file"""
|
|
245
|
+
logger = logging.getLogger("CodeIndexer")
|
|
246
|
+
|
|
247
|
+
# Get logging config
|
|
248
|
+
logging_config = self.indexer_config.get("logging", {})
|
|
249
|
+
log_level = logging_config.get("level", "INFO")
|
|
250
|
+
log_format = logging_config.get(
|
|
251
|
+
"log_format", "%(asctime)s - %(name)s - %(levelname)s - %(message)s"
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
logger.setLevel(getattr(logging, log_level.upper(), logging.INFO))
|
|
255
|
+
|
|
256
|
+
# Clear existing handlers
|
|
257
|
+
logger.handlers.clear()
|
|
258
|
+
|
|
259
|
+
# Console handler
|
|
260
|
+
handler = logging.StreamHandler()
|
|
261
|
+
formatter = logging.Formatter(log_format)
|
|
262
|
+
handler.setFormatter(formatter)
|
|
263
|
+
logger.addHandler(handler)
|
|
264
|
+
|
|
265
|
+
# File handler if enabled
|
|
266
|
+
if logging_config.get("log_to_file", False):
|
|
267
|
+
log_file = logging_config.get("log_file", "indexer.log")
|
|
268
|
+
file_handler = logging.FileHandler(log_file, encoding="utf-8")
|
|
269
|
+
file_handler.setFormatter(formatter)
|
|
270
|
+
logger.addHandler(file_handler)
|
|
271
|
+
|
|
272
|
+
return logger
|
|
273
|
+
|
|
274
|
+
def _load_api_config(self) -> Dict[str, Any]:
|
|
275
|
+
"""Load API configuration from YAML file"""
|
|
276
|
+
try:
|
|
277
|
+
import yaml
|
|
278
|
+
|
|
279
|
+
with open(self.config_path, "r", encoding="utf-8") as f:
|
|
280
|
+
return yaml.safe_load(f)
|
|
281
|
+
except Exception as e:
|
|
282
|
+
# Create a basic logger for this error since self.logger doesn't exist yet
|
|
283
|
+
print(f"Warning: Failed to load API config from {self.config_path}: {e}")
|
|
284
|
+
return {}
|
|
285
|
+
|
|
286
|
+
def _load_indexer_config(self) -> Dict[str, Any]:
|
|
287
|
+
"""Load indexer configuration from YAML file"""
|
|
288
|
+
try:
|
|
289
|
+
import yaml
|
|
290
|
+
|
|
291
|
+
with open(self.indexer_config_path, "r", encoding="utf-8") as f:
|
|
292
|
+
config = yaml.safe_load(f)
|
|
293
|
+
if config is None:
|
|
294
|
+
config = {}
|
|
295
|
+
return config
|
|
296
|
+
except Exception as e:
|
|
297
|
+
print(
|
|
298
|
+
f"Warning: Failed to load indexer config from {self.indexer_config_path}: {e}"
|
|
299
|
+
)
|
|
300
|
+
print("Using default configuration values")
|
|
301
|
+
return {}
|
|
302
|
+
|
|
303
|
+
async def _initialize_llm_client(self):
|
|
304
|
+
"""Initialize LLM client based on configured provider"""
|
|
305
|
+
if self.llm_client is not None:
|
|
306
|
+
return self.llm_client, self.llm_client_type
|
|
307
|
+
|
|
308
|
+
# Check if mock responses are enabled
|
|
309
|
+
if self.mock_llm_responses:
|
|
310
|
+
self.logger.info("Using mock LLM responses for testing")
|
|
311
|
+
self.llm_client = "mock"
|
|
312
|
+
self.llm_client_type = "mock"
|
|
313
|
+
return "mock", "mock"
|
|
314
|
+
|
|
315
|
+
# Try configured provider first
|
|
316
|
+
if self.model_provider.lower() == "anthropic":
|
|
317
|
+
try:
|
|
318
|
+
anthropic_key = self.api_config.get("anthropic", {}).get("api_key")
|
|
319
|
+
if anthropic_key:
|
|
320
|
+
from anthropic import AsyncAnthropic
|
|
321
|
+
|
|
322
|
+
client = AsyncAnthropic(api_key=anthropic_key)
|
|
323
|
+
# Test connection
|
|
324
|
+
await client.messages.create(
|
|
325
|
+
model="claude-sonnet-4-20250514",
|
|
326
|
+
max_tokens=10,
|
|
327
|
+
messages=[{"role": "user", "content": "test"}],
|
|
328
|
+
)
|
|
329
|
+
self.logger.info("Using Anthropic API for code analysis")
|
|
330
|
+
self.llm_client = client
|
|
331
|
+
self.llm_client_type = "anthropic"
|
|
332
|
+
return client, "anthropic"
|
|
333
|
+
except Exception as e:
|
|
334
|
+
self.logger.warning(f"Configured Anthropic API unavailable: {e}")
|
|
335
|
+
|
|
336
|
+
elif self.model_provider.lower() == "openai":
|
|
337
|
+
try:
|
|
338
|
+
openai_key = self.api_config.get("openai", {}).get("api_key")
|
|
339
|
+
if openai_key:
|
|
340
|
+
from openai import AsyncOpenAI
|
|
341
|
+
|
|
342
|
+
client = AsyncOpenAI(api_key=openai_key)
|
|
343
|
+
# Test connection
|
|
344
|
+
await client.chat.completions.create(
|
|
345
|
+
model="gpt-3.5-turbo",
|
|
346
|
+
max_tokens=10,
|
|
347
|
+
messages=[{"role": "user", "content": "test"}],
|
|
348
|
+
)
|
|
349
|
+
self.logger.info("Using OpenAI API for code analysis")
|
|
350
|
+
self.llm_client = client
|
|
351
|
+
self.llm_client_type = "openai"
|
|
352
|
+
return client, "openai"
|
|
353
|
+
except Exception as e:
|
|
354
|
+
self.logger.warning(f"Configured OpenAI API unavailable: {e}")
|
|
355
|
+
|
|
356
|
+
# Fallback: try other provider
|
|
357
|
+
self.logger.info("Trying fallback provider...")
|
|
358
|
+
|
|
359
|
+
# Try Anthropic as fallback
|
|
360
|
+
try:
|
|
361
|
+
anthropic_key = self.api_config.get("anthropic", {}).get("api_key")
|
|
362
|
+
if anthropic_key:
|
|
363
|
+
from anthropic import AsyncAnthropic
|
|
364
|
+
|
|
365
|
+
client = AsyncAnthropic(api_key=anthropic_key)
|
|
366
|
+
await client.messages.create(
|
|
367
|
+
model="claude-sonnet-4-20250514",
|
|
368
|
+
max_tokens=10,
|
|
369
|
+
messages=[{"role": "user", "content": "test"}],
|
|
370
|
+
)
|
|
371
|
+
self.logger.info("Using Anthropic API as fallback")
|
|
372
|
+
self.llm_client = client
|
|
373
|
+
self.llm_client_type = "anthropic"
|
|
374
|
+
return client, "anthropic"
|
|
375
|
+
except Exception as e:
|
|
376
|
+
self.logger.warning(f"Anthropic fallback failed: {e}")
|
|
377
|
+
|
|
378
|
+
# Try OpenAI as fallback
|
|
379
|
+
try:
|
|
380
|
+
openai_key = self.api_config.get("openai", {}).get("api_key")
|
|
381
|
+
if openai_key:
|
|
382
|
+
from openai import AsyncOpenAI
|
|
383
|
+
|
|
384
|
+
client = AsyncOpenAI(api_key=openai_key)
|
|
385
|
+
await client.chat.completions.create(
|
|
386
|
+
model="gpt-3.5-turbo",
|
|
387
|
+
max_tokens=10,
|
|
388
|
+
messages=[{"role": "user", "content": "test"}],
|
|
389
|
+
)
|
|
390
|
+
self.logger.info("Using OpenAI API as fallback")
|
|
391
|
+
self.llm_client = client
|
|
392
|
+
self.llm_client_type = "openai"
|
|
393
|
+
return client, "openai"
|
|
394
|
+
except Exception as e:
|
|
395
|
+
self.logger.warning(f"OpenAI fallback failed: {e}")
|
|
396
|
+
|
|
397
|
+
raise ValueError("No available LLM API for code analysis")
|
|
398
|
+
|
|
399
|
+
async def _call_llm(
|
|
400
|
+
self, prompt: str, system_prompt: str = None, max_tokens: int = None
|
|
401
|
+
) -> str:
|
|
402
|
+
"""Call LLM for code analysis with retry mechanism and debugging support"""
|
|
403
|
+
if system_prompt is None:
|
|
404
|
+
system_prompt = self.llm_system_prompt
|
|
405
|
+
if max_tokens is None:
|
|
406
|
+
max_tokens = self.llm_max_tokens
|
|
407
|
+
|
|
408
|
+
# Mock response for testing
|
|
409
|
+
if self.mock_llm_responses:
|
|
410
|
+
mock_response = self._generate_mock_response(prompt)
|
|
411
|
+
if self.save_raw_responses:
|
|
412
|
+
self._save_debug_response("mock", prompt, mock_response)
|
|
413
|
+
return mock_response
|
|
414
|
+
|
|
415
|
+
last_error = None
|
|
416
|
+
|
|
417
|
+
# Retry mechanism
|
|
418
|
+
for attempt in range(self.max_retries):
|
|
419
|
+
try:
|
|
420
|
+
if self.verbose_output and attempt > 0:
|
|
421
|
+
self.logger.info(
|
|
422
|
+
f"LLM call attempt {attempt + 1}/{self.max_retries}"
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
client, client_type = await self._initialize_llm_client()
|
|
426
|
+
|
|
427
|
+
if client_type == "anthropic":
|
|
428
|
+
response = await client.messages.create(
|
|
429
|
+
model="claude-sonnet-4-20250514",
|
|
430
|
+
system=system_prompt,
|
|
431
|
+
messages=[{"role": "user", "content": prompt}],
|
|
432
|
+
max_tokens=max_tokens,
|
|
433
|
+
temperature=self.llm_temperature,
|
|
434
|
+
)
|
|
435
|
+
|
|
436
|
+
content = ""
|
|
437
|
+
for block in response.content:
|
|
438
|
+
if block.type == "text":
|
|
439
|
+
content += block.text
|
|
440
|
+
|
|
441
|
+
# Save debug response if enabled
|
|
442
|
+
if self.save_raw_responses:
|
|
443
|
+
self._save_debug_response("anthropic", prompt, content)
|
|
444
|
+
|
|
445
|
+
return content
|
|
446
|
+
|
|
447
|
+
elif client_type == "openai":
|
|
448
|
+
messages = [
|
|
449
|
+
{"role": "system", "content": system_prompt},
|
|
450
|
+
{"role": "user", "content": prompt},
|
|
451
|
+
]
|
|
452
|
+
|
|
453
|
+
response = await client.chat.completions.create(
|
|
454
|
+
model="gpt-4-1106-preview",
|
|
455
|
+
messages=messages,
|
|
456
|
+
max_tokens=max_tokens,
|
|
457
|
+
temperature=self.llm_temperature,
|
|
458
|
+
)
|
|
459
|
+
|
|
460
|
+
content = response.choices[0].message.content or ""
|
|
461
|
+
|
|
462
|
+
# Save debug response if enabled
|
|
463
|
+
if self.save_raw_responses:
|
|
464
|
+
self._save_debug_response("openai", prompt, content)
|
|
465
|
+
|
|
466
|
+
return content
|
|
467
|
+
else:
|
|
468
|
+
raise ValueError(f"Unsupported client type: {client_type}")
|
|
469
|
+
|
|
470
|
+
except Exception as e:
|
|
471
|
+
last_error = e
|
|
472
|
+
self.logger.warning(f"LLM call attempt {attempt + 1} failed: {e}")
|
|
473
|
+
|
|
474
|
+
if attempt < self.max_retries - 1:
|
|
475
|
+
await asyncio.sleep(
|
|
476
|
+
self.retry_delay * (attempt + 1)
|
|
477
|
+
) # Exponential backoff
|
|
478
|
+
|
|
479
|
+
# All retries failed
|
|
480
|
+
error_msg = f"LLM call failed after {self.max_retries} attempts. Last error: {str(last_error)}"
|
|
481
|
+
self.logger.error(error_msg)
|
|
482
|
+
return f"Error in LLM analysis: {error_msg}"
|
|
483
|
+
|
|
484
|
+
def _generate_mock_response(self, prompt: str) -> str:
|
|
485
|
+
"""Generate mock LLM response for testing"""
|
|
486
|
+
if "JSON format" in prompt and "file_type" in prompt:
|
|
487
|
+
# File analysis mock
|
|
488
|
+
return """
|
|
489
|
+
{
|
|
490
|
+
"file_type": "Python module",
|
|
491
|
+
"main_functions": ["main_function", "helper_function"],
|
|
492
|
+
"key_concepts": ["data_processing", "algorithm"],
|
|
493
|
+
"dependencies": ["numpy", "pandas"],
|
|
494
|
+
"summary": "Mock analysis of code file functionality."
|
|
495
|
+
}
|
|
496
|
+
"""
|
|
497
|
+
elif "relationships" in prompt:
|
|
498
|
+
# Relationship analysis mock
|
|
499
|
+
return """
|
|
500
|
+
{
|
|
501
|
+
"relationships": [
|
|
502
|
+
{
|
|
503
|
+
"target_file_path": "src/core/mock.py",
|
|
504
|
+
"relationship_type": "partial_match",
|
|
505
|
+
"confidence_score": 0.8,
|
|
506
|
+
"helpful_aspects": ["algorithm implementation", "data structures"],
|
|
507
|
+
"potential_contributions": ["core functionality", "utility methods"],
|
|
508
|
+
"usage_suggestions": "Mock relationship suggestion for testing."
|
|
509
|
+
}
|
|
510
|
+
]
|
|
511
|
+
}
|
|
512
|
+
"""
|
|
513
|
+
elif "relevant_files" in prompt:
|
|
514
|
+
# File filtering mock
|
|
515
|
+
return """
|
|
516
|
+
{
|
|
517
|
+
"relevant_files": [
|
|
518
|
+
{
|
|
519
|
+
"file_path": "mock_file.py",
|
|
520
|
+
"relevance_reason": "Mock relevance reason",
|
|
521
|
+
"confidence": 0.9,
|
|
522
|
+
"expected_contribution": "Mock contribution"
|
|
523
|
+
}
|
|
524
|
+
],
|
|
525
|
+
"summary": {
|
|
526
|
+
"total_files_analyzed": "10",
|
|
527
|
+
"relevant_files_count": "1",
|
|
528
|
+
"filtering_strategy": "Mock filtering strategy"
|
|
529
|
+
}
|
|
530
|
+
}
|
|
531
|
+
"""
|
|
532
|
+
else:
|
|
533
|
+
return "Mock LLM response for testing purposes."
|
|
534
|
+
|
|
535
|
+
def _save_debug_response(self, provider: str, prompt: str, response: str):
|
|
536
|
+
"""Save LLM response for debugging"""
|
|
537
|
+
try:
|
|
538
|
+
import hashlib
|
|
539
|
+
from datetime import datetime
|
|
540
|
+
|
|
541
|
+
# Create a hash of the prompt for filename
|
|
542
|
+
prompt_hash = hashlib.md5(prompt.encode()).hexdigest()[:8]
|
|
543
|
+
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
544
|
+
filename = f"{provider}_{timestamp}_{prompt_hash}.json"
|
|
545
|
+
|
|
546
|
+
debug_data = {
|
|
547
|
+
"timestamp": datetime.now().isoformat(),
|
|
548
|
+
"provider": provider,
|
|
549
|
+
"prompt": prompt[:500] + "..." if len(prompt) > 500 else prompt,
|
|
550
|
+
"response": response,
|
|
551
|
+
"full_prompt_length": len(prompt),
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
debug_file = Path(self.raw_responses_dir) / filename
|
|
555
|
+
with open(debug_file, "w", encoding="utf-8") as f:
|
|
556
|
+
json.dump(debug_data, f, indent=2, ensure_ascii=False)
|
|
557
|
+
|
|
558
|
+
except Exception as e:
|
|
559
|
+
self.logger.warning(f"Failed to save debug response: {e}")
|
|
560
|
+
|
|
561
|
+
def get_all_repo_files(self, repo_path: Path) -> List[Path]:
|
|
562
|
+
"""Recursively get all supported files in a repository"""
|
|
563
|
+
files = []
|
|
564
|
+
|
|
565
|
+
try:
|
|
566
|
+
for root, dirs, filenames in os.walk(repo_path):
|
|
567
|
+
# Skip common non-code directories
|
|
568
|
+
dirs[:] = [
|
|
569
|
+
d
|
|
570
|
+
for d in dirs
|
|
571
|
+
if not d.startswith(".") and d not in self.skip_directories
|
|
572
|
+
]
|
|
573
|
+
|
|
574
|
+
for filename in filenames:
|
|
575
|
+
file_path = Path(root) / filename
|
|
576
|
+
if file_path.suffix.lower() in self.supported_extensions:
|
|
577
|
+
files.append(file_path)
|
|
578
|
+
|
|
579
|
+
except Exception as e:
|
|
580
|
+
self.logger.error(f"Error traversing {repo_path}: {e}")
|
|
581
|
+
|
|
582
|
+
return files
|
|
583
|
+
|
|
584
|
+
def generate_file_tree(self, repo_path: Path, max_depth: int = 5) -> str:
|
|
585
|
+
"""Generate file tree structure string for the repository"""
|
|
586
|
+
tree_lines = []
|
|
587
|
+
|
|
588
|
+
def add_to_tree(current_path: Path, prefix: str = "", depth: int = 0):
|
|
589
|
+
if depth > max_depth:
|
|
590
|
+
return
|
|
591
|
+
|
|
592
|
+
try:
|
|
593
|
+
items = sorted(
|
|
594
|
+
current_path.iterdir(), key=lambda x: (x.is_file(), x.name.lower())
|
|
595
|
+
)
|
|
596
|
+
# Filter out irrelevant directories and files
|
|
597
|
+
items = [
|
|
598
|
+
item
|
|
599
|
+
for item in items
|
|
600
|
+
if not item.name.startswith(".")
|
|
601
|
+
and item.name not in self.skip_directories
|
|
602
|
+
]
|
|
603
|
+
|
|
604
|
+
for i, item in enumerate(items):
|
|
605
|
+
is_last = i == len(items) - 1
|
|
606
|
+
current_prefix = "└── " if is_last else "├── "
|
|
607
|
+
tree_lines.append(f"{prefix}{current_prefix}{item.name}")
|
|
608
|
+
|
|
609
|
+
if item.is_dir():
|
|
610
|
+
extension_prefix = " " if is_last else "│ "
|
|
611
|
+
add_to_tree(item, prefix + extension_prefix, depth + 1)
|
|
612
|
+
elif item.suffix.lower() in self.supported_extensions:
|
|
613
|
+
# Add file size information
|
|
614
|
+
try:
|
|
615
|
+
size = item.stat().st_size
|
|
616
|
+
if size > 1024:
|
|
617
|
+
size_str = f" ({size // 1024}KB)"
|
|
618
|
+
else:
|
|
619
|
+
size_str = f" ({size}B)"
|
|
620
|
+
tree_lines[-1] += size_str
|
|
621
|
+
except (OSError, PermissionError):
|
|
622
|
+
pass
|
|
623
|
+
|
|
624
|
+
except PermissionError:
|
|
625
|
+
tree_lines.append(f"{prefix}├── [Permission Denied]")
|
|
626
|
+
except Exception as e:
|
|
627
|
+
tree_lines.append(f"{prefix}├── [Error: {str(e)}]")
|
|
628
|
+
|
|
629
|
+
tree_lines.append(f"{repo_path.name}/")
|
|
630
|
+
add_to_tree(repo_path)
|
|
631
|
+
return "\n".join(tree_lines)
|
|
632
|
+
|
|
633
|
+
async def pre_filter_files(self, repo_path: Path, file_tree: str) -> List[str]:
|
|
634
|
+
"""Use LLM to pre-filter relevant files based on target structure"""
|
|
635
|
+
filter_prompt = f"""
|
|
636
|
+
You are a code analysis expert. Please analyze the following code repository file tree based on the target project structure and filter out files that may be relevant to the target project.
|
|
637
|
+
|
|
638
|
+
Target Project Structure:
|
|
639
|
+
{self.target_structure}
|
|
640
|
+
|
|
641
|
+
Code Repository File Tree:
|
|
642
|
+
{file_tree}
|
|
643
|
+
|
|
644
|
+
Please analyze which files might be helpful for implementing the target project structure, including:
|
|
645
|
+
- Core algorithm implementation files (such as GCN, recommendation systems, graph neural networks, etc.)
|
|
646
|
+
- Data processing and preprocessing files
|
|
647
|
+
- Loss functions and evaluation metric files
|
|
648
|
+
- Configuration and utility files
|
|
649
|
+
- Test files
|
|
650
|
+
- Documentation files
|
|
651
|
+
|
|
652
|
+
Please return the filtering results in JSON format:
|
|
653
|
+
{{
|
|
654
|
+
"relevant_files": [
|
|
655
|
+
{{
|
|
656
|
+
"file_path": "file path relative to repository root",
|
|
657
|
+
"relevance_reason": "why this file is relevant",
|
|
658
|
+
"confidence": 0.0-1.0,
|
|
659
|
+
"expected_contribution": "expected contribution to the target project"
|
|
660
|
+
}}
|
|
661
|
+
],
|
|
662
|
+
"summary": {{
|
|
663
|
+
"total_files_analyzed": "total number of files analyzed",
|
|
664
|
+
"relevant_files_count": "number of relevant files",
|
|
665
|
+
"filtering_strategy": "explanation of filtering strategy"
|
|
666
|
+
}}
|
|
667
|
+
}}
|
|
668
|
+
|
|
669
|
+
Only return files with confidence > {self.min_confidence_score}. Focus on files related to recommendation systems, graph neural networks, and diffusion models.
|
|
670
|
+
"""
|
|
671
|
+
|
|
672
|
+
try:
|
|
673
|
+
self.logger.info("Starting LLM pre-filtering of files...")
|
|
674
|
+
llm_response = await self._call_llm(
|
|
675
|
+
filter_prompt,
|
|
676
|
+
system_prompt="You are a professional code analysis and project architecture expert, skilled at identifying code file functionality and relevance.",
|
|
677
|
+
max_tokens=2000,
|
|
678
|
+
)
|
|
679
|
+
|
|
680
|
+
# Parse JSON response
|
|
681
|
+
match = re.search(r"\{.*\}", llm_response, re.DOTALL)
|
|
682
|
+
if not match:
|
|
683
|
+
self.logger.warning(
|
|
684
|
+
"Unable to parse LLM filtering response, will use all files"
|
|
685
|
+
)
|
|
686
|
+
return []
|
|
687
|
+
|
|
688
|
+
filter_data = json.loads(match.group(0))
|
|
689
|
+
relevant_files = filter_data.get("relevant_files", [])
|
|
690
|
+
|
|
691
|
+
# Extract file paths
|
|
692
|
+
selected_files = []
|
|
693
|
+
for file_info in relevant_files:
|
|
694
|
+
file_path = file_info.get("file_path", "")
|
|
695
|
+
confidence = file_info.get("confidence", 0.0)
|
|
696
|
+
# Use configured minimum confidence threshold
|
|
697
|
+
if file_path and confidence > self.min_confidence_score:
|
|
698
|
+
selected_files.append(file_path)
|
|
699
|
+
|
|
700
|
+
summary = filter_data.get("summary", {})
|
|
701
|
+
self.logger.info(
|
|
702
|
+
f"LLM filtering completed: {summary.get('relevant_files_count', len(selected_files))} relevant files selected"
|
|
703
|
+
)
|
|
704
|
+
self.logger.info(
|
|
705
|
+
f"Filtering strategy: {summary.get('filtering_strategy', 'Not provided')}"
|
|
706
|
+
)
|
|
707
|
+
|
|
708
|
+
return selected_files
|
|
709
|
+
|
|
710
|
+
except Exception as e:
|
|
711
|
+
self.logger.error(f"LLM pre-filtering failed: {e}")
|
|
712
|
+
self.logger.info("Will fallback to analyzing all files")
|
|
713
|
+
return []
|
|
714
|
+
|
|
715
|
+
def filter_files_by_paths(
|
|
716
|
+
self, all_files: List[Path], selected_paths: List[str], repo_path: Path
|
|
717
|
+
) -> List[Path]:
|
|
718
|
+
"""Filter file list based on LLM-selected paths"""
|
|
719
|
+
if not selected_paths:
|
|
720
|
+
return all_files
|
|
721
|
+
|
|
722
|
+
filtered_files = []
|
|
723
|
+
|
|
724
|
+
for file_path in all_files:
|
|
725
|
+
# Get path relative to repository root
|
|
726
|
+
relative_path = str(file_path.relative_to(repo_path))
|
|
727
|
+
|
|
728
|
+
# Check if it's in the selected list
|
|
729
|
+
for selected_path in selected_paths:
|
|
730
|
+
# Normalize path comparison
|
|
731
|
+
if (
|
|
732
|
+
relative_path == selected_path
|
|
733
|
+
or relative_path.replace("\\", "/")
|
|
734
|
+
== selected_path.replace("\\", "/")
|
|
735
|
+
or selected_path in relative_path
|
|
736
|
+
or relative_path in selected_path
|
|
737
|
+
):
|
|
738
|
+
filtered_files.append(file_path)
|
|
739
|
+
break
|
|
740
|
+
|
|
741
|
+
return filtered_files
|
|
742
|
+
|
|
743
|
+
def _get_cache_key(self, file_path: Path) -> str:
|
|
744
|
+
"""Generate cache key for file content"""
|
|
745
|
+
try:
|
|
746
|
+
stats = file_path.stat()
|
|
747
|
+
return f"{file_path}:{stats.st_mtime}:{stats.st_size}"
|
|
748
|
+
except (OSError, PermissionError):
|
|
749
|
+
return str(file_path)
|
|
750
|
+
|
|
751
|
+
def _manage_cache_size(self):
|
|
752
|
+
"""Manage cache size to stay within limits"""
|
|
753
|
+
if not self.enable_content_caching or not self.content_cache:
|
|
754
|
+
return
|
|
755
|
+
|
|
756
|
+
if len(self.content_cache) > self.max_cache_size:
|
|
757
|
+
# Remove oldest entries (simple FIFO strategy)
|
|
758
|
+
excess_count = len(self.content_cache) - self.max_cache_size + 10
|
|
759
|
+
keys_to_remove = list(self.content_cache.keys())[:excess_count]
|
|
760
|
+
|
|
761
|
+
for key in keys_to_remove:
|
|
762
|
+
del self.content_cache[key]
|
|
763
|
+
|
|
764
|
+
if self.verbose_output:
|
|
765
|
+
self.logger.info(
|
|
766
|
+
f"Cache cleaned: removed {excess_count} entries, {len(self.content_cache)} entries remaining"
|
|
767
|
+
)
|
|
768
|
+
|
|
769
|
+
async def analyze_file_content(self, file_path: Path) -> FileSummary:
|
|
770
|
+
"""Analyze a single file and create summary with caching support"""
|
|
771
|
+
try:
|
|
772
|
+
# Check file size before reading
|
|
773
|
+
file_size = file_path.stat().st_size
|
|
774
|
+
if file_size > self.max_file_size:
|
|
775
|
+
self.logger.warning(
|
|
776
|
+
f"Skipping file {file_path} - size {file_size} bytes exceeds limit {self.max_file_size}"
|
|
777
|
+
)
|
|
778
|
+
return FileSummary(
|
|
779
|
+
file_path=str(file_path.relative_to(self.code_base_path)),
|
|
780
|
+
file_type="skipped - too large",
|
|
781
|
+
main_functions=[],
|
|
782
|
+
key_concepts=[],
|
|
783
|
+
dependencies=[],
|
|
784
|
+
summary=f"File skipped - size {file_size} bytes exceeds {self.max_file_size} byte limit",
|
|
785
|
+
lines_of_code=0,
|
|
786
|
+
last_modified=datetime.fromtimestamp(
|
|
787
|
+
file_path.stat().st_mtime
|
|
788
|
+
).isoformat(),
|
|
789
|
+
)
|
|
790
|
+
|
|
791
|
+
# Check cache if enabled
|
|
792
|
+
cache_key = None
|
|
793
|
+
if self.enable_content_caching:
|
|
794
|
+
cache_key = self._get_cache_key(file_path)
|
|
795
|
+
if cache_key in self.content_cache:
|
|
796
|
+
if self.verbose_output:
|
|
797
|
+
self.logger.info(f"Using cached analysis for {file_path.name}")
|
|
798
|
+
return self.content_cache[cache_key]
|
|
799
|
+
|
|
800
|
+
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
|
|
801
|
+
content = f.read()
|
|
802
|
+
|
|
803
|
+
# Get file stats
|
|
804
|
+
stats = file_path.stat()
|
|
805
|
+
lines_of_code = len([line for line in content.split("\n") if line.strip()])
|
|
806
|
+
|
|
807
|
+
# Truncate content based on config
|
|
808
|
+
content_for_analysis = content[: self.max_content_length]
|
|
809
|
+
content_suffix = "..." if len(content) > self.max_content_length else ""
|
|
810
|
+
|
|
811
|
+
# Create analysis prompt
|
|
812
|
+
analysis_prompt = f"""
|
|
813
|
+
Analyze this code file and provide a structured summary:
|
|
814
|
+
|
|
815
|
+
File: {file_path.name}
|
|
816
|
+
Content:
|
|
817
|
+
```
|
|
818
|
+
{content_for_analysis}{content_suffix}
|
|
819
|
+
```
|
|
820
|
+
|
|
821
|
+
Please provide analysis in this JSON format:
|
|
822
|
+
{{
|
|
823
|
+
"file_type": "description of what type of file this is",
|
|
824
|
+
"main_functions": ["list", "of", "main", "functions", "or", "classes"],
|
|
825
|
+
"key_concepts": ["important", "concepts", "algorithms", "patterns"],
|
|
826
|
+
"dependencies": ["external", "libraries", "or", "imports"],
|
|
827
|
+
"summary": "2-3 sentence summary of what this file does"
|
|
828
|
+
}}
|
|
829
|
+
|
|
830
|
+
Focus on the core functionality and potential reusability.
|
|
831
|
+
"""
|
|
832
|
+
|
|
833
|
+
# Get LLM analysis with configured parameters
|
|
834
|
+
llm_response = await self._call_llm(analysis_prompt, max_tokens=1000)
|
|
835
|
+
|
|
836
|
+
try:
|
|
837
|
+
# Try to parse JSON response
|
|
838
|
+
match = re.search(r"\{.*\}", llm_response, re.DOTALL)
|
|
839
|
+
analysis_data = json.loads(match.group(0))
|
|
840
|
+
except json.JSONDecodeError:
|
|
841
|
+
# Fallback to basic analysis if JSON parsing fails
|
|
842
|
+
analysis_data = {
|
|
843
|
+
"file_type": f"{file_path.suffix} file",
|
|
844
|
+
"main_functions": [],
|
|
845
|
+
"key_concepts": [],
|
|
846
|
+
"dependencies": [],
|
|
847
|
+
"summary": "File analysis failed - JSON parsing error",
|
|
848
|
+
}
|
|
849
|
+
|
|
850
|
+
file_summary = FileSummary(
|
|
851
|
+
file_path=str(file_path.relative_to(self.code_base_path)),
|
|
852
|
+
file_type=analysis_data.get("file_type", "unknown"),
|
|
853
|
+
main_functions=analysis_data.get("main_functions", []),
|
|
854
|
+
key_concepts=analysis_data.get("key_concepts", []),
|
|
855
|
+
dependencies=analysis_data.get("dependencies", []),
|
|
856
|
+
summary=analysis_data.get("summary", "No summary available"),
|
|
857
|
+
lines_of_code=lines_of_code,
|
|
858
|
+
last_modified=datetime.fromtimestamp(stats.st_mtime).isoformat(),
|
|
859
|
+
)
|
|
860
|
+
|
|
861
|
+
# Cache the result if caching is enabled
|
|
862
|
+
if self.enable_content_caching and cache_key:
|
|
863
|
+
self.content_cache[cache_key] = file_summary
|
|
864
|
+
self._manage_cache_size()
|
|
865
|
+
|
|
866
|
+
return file_summary
|
|
867
|
+
|
|
868
|
+
except Exception as e:
|
|
869
|
+
self.logger.error(f"Error analyzing file {file_path}: {e}")
|
|
870
|
+
return FileSummary(
|
|
871
|
+
file_path=str(file_path.relative_to(self.code_base_path)),
|
|
872
|
+
file_type="error",
|
|
873
|
+
main_functions=[],
|
|
874
|
+
key_concepts=[],
|
|
875
|
+
dependencies=[],
|
|
876
|
+
summary=f"Analysis failed: {str(e)}",
|
|
877
|
+
lines_of_code=0,
|
|
878
|
+
last_modified="",
|
|
879
|
+
)
|
|
880
|
+
|
|
881
|
+
async def find_relationships(
|
|
882
|
+
self, file_summary: FileSummary
|
|
883
|
+
) -> List[FileRelationship]:
|
|
884
|
+
"""Find relationships between a repo file and target structure"""
|
|
885
|
+
|
|
886
|
+
# Build relationship type description from config
|
|
887
|
+
relationship_type_desc = []
|
|
888
|
+
for rel_type, weight in self.relationship_types.items():
|
|
889
|
+
relationship_type_desc.append(f"- {rel_type} (priority: {weight})")
|
|
890
|
+
|
|
891
|
+
relationship_prompt = f"""
|
|
892
|
+
Analyze the relationship between this existing code file and the target project structure.
|
|
893
|
+
|
|
894
|
+
Existing File Analysis:
|
|
895
|
+
- Path: {file_summary.file_path}
|
|
896
|
+
- Type: {file_summary.file_type}
|
|
897
|
+
- Functions: {', '.join(file_summary.main_functions)}
|
|
898
|
+
- Concepts: {', '.join(file_summary.key_concepts)}
|
|
899
|
+
- Summary: {file_summary.summary}
|
|
900
|
+
|
|
901
|
+
Target Project Structure:
|
|
902
|
+
{self.target_structure}
|
|
903
|
+
|
|
904
|
+
Available relationship types (with priority weights):
|
|
905
|
+
{chr(10).join(relationship_type_desc)}
|
|
906
|
+
|
|
907
|
+
Identify potential relationships and provide analysis in this JSON format:
|
|
908
|
+
{{
|
|
909
|
+
"relationships": [
|
|
910
|
+
{{
|
|
911
|
+
"target_file_path": "path/in/target/structure",
|
|
912
|
+
"relationship_type": "direct_match|partial_match|reference|utility",
|
|
913
|
+
"confidence_score": 0.0-1.0,
|
|
914
|
+
"helpful_aspects": ["specific", "aspects", "that", "could", "help"],
|
|
915
|
+
"potential_contributions": ["how", "this", "could", "contribute"],
|
|
916
|
+
"usage_suggestions": "detailed suggestion on how to use this file"
|
|
917
|
+
}}
|
|
918
|
+
]
|
|
919
|
+
}}
|
|
920
|
+
|
|
921
|
+
Consider the priority weights when determining relationship types. Higher weight types should be preferred when multiple types apply.
|
|
922
|
+
Only include relationships with confidence > {self.min_confidence_score}. Focus on concrete, actionable connections.
|
|
923
|
+
"""
|
|
924
|
+
|
|
925
|
+
try:
|
|
926
|
+
llm_response = await self._call_llm(relationship_prompt, max_tokens=1500)
|
|
927
|
+
|
|
928
|
+
match = re.search(r"\{.*\}", llm_response, re.DOTALL)
|
|
929
|
+
relationship_data = json.loads(match.group(0))
|
|
930
|
+
|
|
931
|
+
relationships = []
|
|
932
|
+
for rel_data in relationship_data.get("relationships", []):
|
|
933
|
+
confidence_score = float(rel_data.get("confidence_score", 0.0))
|
|
934
|
+
relationship_type = rel_data.get("relationship_type", "reference")
|
|
935
|
+
|
|
936
|
+
# Validate relationship type is in config
|
|
937
|
+
if relationship_type not in self.relationship_types:
|
|
938
|
+
if self.verbose_output:
|
|
939
|
+
self.logger.warning(
|
|
940
|
+
f"Unknown relationship type '{relationship_type}', using 'reference'"
|
|
941
|
+
)
|
|
942
|
+
relationship_type = "reference"
|
|
943
|
+
|
|
944
|
+
# Apply configured minimum confidence filter
|
|
945
|
+
if confidence_score > self.min_confidence_score:
|
|
946
|
+
relationship = FileRelationship(
|
|
947
|
+
repo_file_path=file_summary.file_path,
|
|
948
|
+
target_file_path=rel_data.get("target_file_path", ""),
|
|
949
|
+
relationship_type=relationship_type,
|
|
950
|
+
confidence_score=confidence_score,
|
|
951
|
+
helpful_aspects=rel_data.get("helpful_aspects", []),
|
|
952
|
+
potential_contributions=rel_data.get(
|
|
953
|
+
"potential_contributions", []
|
|
954
|
+
),
|
|
955
|
+
usage_suggestions=rel_data.get("usage_suggestions", ""),
|
|
956
|
+
)
|
|
957
|
+
relationships.append(relationship)
|
|
958
|
+
|
|
959
|
+
return relationships
|
|
960
|
+
|
|
961
|
+
except Exception as e:
|
|
962
|
+
self.logger.error(
|
|
963
|
+
f"Error finding relationships for {file_summary.file_path}: {e}"
|
|
964
|
+
)
|
|
965
|
+
return []
|
|
966
|
+
|
|
967
|
+
async def _analyze_single_file_with_relationships(
|
|
968
|
+
self, file_path: Path, index: int, total: int
|
|
969
|
+
) -> tuple:
|
|
970
|
+
"""Analyze a single file and its relationships (for concurrent processing)"""
|
|
971
|
+
if self.verbose_output:
|
|
972
|
+
self.logger.info(f"Analyzing file {index}/{total}: {file_path.name}")
|
|
973
|
+
|
|
974
|
+
# Get file summary
|
|
975
|
+
file_summary = await self.analyze_file_content(file_path)
|
|
976
|
+
|
|
977
|
+
# Find relationships
|
|
978
|
+
relationships = await self.find_relationships(file_summary)
|
|
979
|
+
|
|
980
|
+
return file_summary, relationships
|
|
981
|
+
|
|
982
|
+
async def process_repository(self, repo_path: Path) -> RepoIndex:
|
|
983
|
+
"""Process a single repository and create complete index with optional concurrent processing"""
|
|
984
|
+
repo_name = repo_path.name
|
|
985
|
+
self.logger.info(f"Processing repository: {repo_name}")
|
|
986
|
+
|
|
987
|
+
# Step 1: Generate file tree
|
|
988
|
+
self.logger.info("Generating file tree structure...")
|
|
989
|
+
file_tree = self.generate_file_tree(repo_path)
|
|
990
|
+
|
|
991
|
+
# Step 2: Get all files
|
|
992
|
+
all_files = self.get_all_repo_files(repo_path)
|
|
993
|
+
self.logger.info(f"Found {len(all_files)} files in {repo_name}")
|
|
994
|
+
|
|
995
|
+
# Step 3: LLM pre-filtering of relevant files
|
|
996
|
+
if self.enable_pre_filtering:
|
|
997
|
+
self.logger.info("Using LLM for file pre-filtering...")
|
|
998
|
+
selected_file_paths = await self.pre_filter_files(repo_path, file_tree)
|
|
999
|
+
else:
|
|
1000
|
+
self.logger.info("Pre-filtering is disabled, will analyze all files")
|
|
1001
|
+
selected_file_paths = []
|
|
1002
|
+
|
|
1003
|
+
# Step 4: Filter file list based on filtering results
|
|
1004
|
+
if selected_file_paths:
|
|
1005
|
+
files_to_analyze = self.filter_files_by_paths(
|
|
1006
|
+
all_files, selected_file_paths, repo_path
|
|
1007
|
+
)
|
|
1008
|
+
self.logger.info(
|
|
1009
|
+
f"After LLM filtering, will analyze {len(files_to_analyze)} relevant files (from {len(all_files)} total)"
|
|
1010
|
+
)
|
|
1011
|
+
else:
|
|
1012
|
+
files_to_analyze = all_files
|
|
1013
|
+
self.logger.info("LLM filtering failed, will analyze all files")
|
|
1014
|
+
|
|
1015
|
+
# Step 5: Analyze filtered files (concurrent or sequential)
|
|
1016
|
+
if self.enable_concurrent_analysis and len(files_to_analyze) > 1:
|
|
1017
|
+
self.logger.info(
|
|
1018
|
+
f"Using concurrent analysis with max {self.max_concurrent_files} parallel files"
|
|
1019
|
+
)
|
|
1020
|
+
file_summaries, all_relationships = await self._process_files_concurrently(
|
|
1021
|
+
files_to_analyze
|
|
1022
|
+
)
|
|
1023
|
+
else:
|
|
1024
|
+
self.logger.info("Using sequential file analysis")
|
|
1025
|
+
file_summaries, all_relationships = await self._process_files_sequentially(
|
|
1026
|
+
files_to_analyze
|
|
1027
|
+
)
|
|
1028
|
+
|
|
1029
|
+
# Step 6: Create repository index
|
|
1030
|
+
repo_index = RepoIndex(
|
|
1031
|
+
repo_name=repo_name,
|
|
1032
|
+
total_files=len(all_files), # Record original file count
|
|
1033
|
+
file_summaries=file_summaries,
|
|
1034
|
+
relationships=all_relationships,
|
|
1035
|
+
analysis_metadata={
|
|
1036
|
+
"analysis_date": datetime.now().isoformat(),
|
|
1037
|
+
"target_structure_analyzed": self.target_structure[:200] + "...",
|
|
1038
|
+
"total_relationships_found": len(all_relationships),
|
|
1039
|
+
"high_confidence_relationships": len(
|
|
1040
|
+
[
|
|
1041
|
+
r
|
|
1042
|
+
for r in all_relationships
|
|
1043
|
+
if r.confidence_score > self.high_confidence_threshold
|
|
1044
|
+
]
|
|
1045
|
+
),
|
|
1046
|
+
"analyzer_version": "1.3.0", # Updated version to reflect concurrent support
|
|
1047
|
+
"pre_filtering_enabled": self.enable_pre_filtering,
|
|
1048
|
+
"files_before_filtering": len(all_files),
|
|
1049
|
+
"files_after_filtering": len(files_to_analyze),
|
|
1050
|
+
"filtering_efficiency": round(
|
|
1051
|
+
(1 - len(files_to_analyze) / len(all_files)) * 100, 2
|
|
1052
|
+
)
|
|
1053
|
+
if all_files
|
|
1054
|
+
else 0,
|
|
1055
|
+
"config_file_used": self.indexer_config_path,
|
|
1056
|
+
"min_confidence_score": self.min_confidence_score,
|
|
1057
|
+
"high_confidence_threshold": self.high_confidence_threshold,
|
|
1058
|
+
"concurrent_analysis_used": self.enable_concurrent_analysis,
|
|
1059
|
+
"content_caching_enabled": self.enable_content_caching,
|
|
1060
|
+
"cache_hits": len(self.content_cache) if self.content_cache else 0,
|
|
1061
|
+
},
|
|
1062
|
+
)
|
|
1063
|
+
|
|
1064
|
+
return repo_index
|
|
1065
|
+
|
|
1066
|
+
async def _process_files_sequentially(self, files_to_analyze: list) -> tuple:
|
|
1067
|
+
"""Process files sequentially (original method)"""
|
|
1068
|
+
file_summaries = []
|
|
1069
|
+
all_relationships = []
|
|
1070
|
+
|
|
1071
|
+
for i, file_path in enumerate(files_to_analyze, 1):
|
|
1072
|
+
(
|
|
1073
|
+
file_summary,
|
|
1074
|
+
relationships,
|
|
1075
|
+
) = await self._analyze_single_file_with_relationships(
|
|
1076
|
+
file_path, i, len(files_to_analyze)
|
|
1077
|
+
)
|
|
1078
|
+
file_summaries.append(file_summary)
|
|
1079
|
+
all_relationships.extend(relationships)
|
|
1080
|
+
|
|
1081
|
+
# Add configured delay to avoid overwhelming the LLM API
|
|
1082
|
+
await asyncio.sleep(self.request_delay)
|
|
1083
|
+
|
|
1084
|
+
return file_summaries, all_relationships
|
|
1085
|
+
|
|
1086
|
+
async def _process_files_concurrently(self, files_to_analyze: list) -> tuple:
|
|
1087
|
+
"""Process files concurrently with semaphore limiting"""
|
|
1088
|
+
file_summaries = []
|
|
1089
|
+
all_relationships = []
|
|
1090
|
+
|
|
1091
|
+
# Create semaphore to limit concurrent tasks
|
|
1092
|
+
semaphore = asyncio.Semaphore(self.max_concurrent_files)
|
|
1093
|
+
tasks = []
|
|
1094
|
+
|
|
1095
|
+
async def _process_with_semaphore(file_path: Path, index: int, total: int):
|
|
1096
|
+
async with semaphore:
|
|
1097
|
+
# Add a small delay to space out concurrent requests
|
|
1098
|
+
if index > 1:
|
|
1099
|
+
await asyncio.sleep(
|
|
1100
|
+
self.request_delay * 0.5
|
|
1101
|
+
) # Reduced delay for concurrent processing
|
|
1102
|
+
return await self._analyze_single_file_with_relationships(
|
|
1103
|
+
file_path, index, total
|
|
1104
|
+
)
|
|
1105
|
+
|
|
1106
|
+
try:
|
|
1107
|
+
# Create tasks for all files
|
|
1108
|
+
tasks = [
|
|
1109
|
+
_process_with_semaphore(file_path, i, len(files_to_analyze))
|
|
1110
|
+
for i, file_path in enumerate(files_to_analyze, 1)
|
|
1111
|
+
]
|
|
1112
|
+
|
|
1113
|
+
# Process tasks and collect results
|
|
1114
|
+
if self.verbose_output:
|
|
1115
|
+
self.logger.info(
|
|
1116
|
+
f"Starting concurrent analysis of {len(tasks)} files..."
|
|
1117
|
+
)
|
|
1118
|
+
|
|
1119
|
+
try:
|
|
1120
|
+
results = await asyncio.gather(*tasks, return_exceptions=True)
|
|
1121
|
+
|
|
1122
|
+
for i, result in enumerate(results):
|
|
1123
|
+
if isinstance(result, Exception):
|
|
1124
|
+
self.logger.error(
|
|
1125
|
+
f"Failed to analyze file {files_to_analyze[i]}: {result}"
|
|
1126
|
+
)
|
|
1127
|
+
# Create error summary
|
|
1128
|
+
error_summary = FileSummary(
|
|
1129
|
+
file_path=str(
|
|
1130
|
+
files_to_analyze[i].relative_to(self.code_base_path)
|
|
1131
|
+
),
|
|
1132
|
+
file_type="error",
|
|
1133
|
+
main_functions=[],
|
|
1134
|
+
key_concepts=[],
|
|
1135
|
+
dependencies=[],
|
|
1136
|
+
summary=f"Concurrent analysis failed: {str(result)}",
|
|
1137
|
+
lines_of_code=0,
|
|
1138
|
+
last_modified="",
|
|
1139
|
+
)
|
|
1140
|
+
file_summaries.append(error_summary)
|
|
1141
|
+
else:
|
|
1142
|
+
file_summary, relationships = result
|
|
1143
|
+
file_summaries.append(file_summary)
|
|
1144
|
+
all_relationships.extend(relationships)
|
|
1145
|
+
|
|
1146
|
+
except Exception as e:
|
|
1147
|
+
self.logger.error(f"Concurrent processing failed: {e}")
|
|
1148
|
+
# Cancel any remaining tasks
|
|
1149
|
+
for task in tasks:
|
|
1150
|
+
if not task.done() and not task.cancelled():
|
|
1151
|
+
task.cancel()
|
|
1152
|
+
|
|
1153
|
+
# Wait for cancelled tasks to complete
|
|
1154
|
+
try:
|
|
1155
|
+
await asyncio.sleep(0.1) # Brief wait for cancellation
|
|
1156
|
+
except Exception:
|
|
1157
|
+
pass
|
|
1158
|
+
|
|
1159
|
+
# Fallback to sequential processing
|
|
1160
|
+
self.logger.info("Falling back to sequential processing...")
|
|
1161
|
+
return await self._process_files_sequentially(files_to_analyze)
|
|
1162
|
+
|
|
1163
|
+
if self.verbose_output:
|
|
1164
|
+
self.logger.info(
|
|
1165
|
+
f"Concurrent analysis completed: {len(file_summaries)} files processed"
|
|
1166
|
+
)
|
|
1167
|
+
|
|
1168
|
+
return file_summaries, all_relationships
|
|
1169
|
+
|
|
1170
|
+
except Exception as e:
|
|
1171
|
+
# Ensure all tasks are cancelled in case of unexpected errors
|
|
1172
|
+
if tasks:
|
|
1173
|
+
for task in tasks:
|
|
1174
|
+
if not task.done() and not task.cancelled():
|
|
1175
|
+
task.cancel()
|
|
1176
|
+
|
|
1177
|
+
# Wait briefly for cancellation to complete
|
|
1178
|
+
try:
|
|
1179
|
+
await asyncio.sleep(0.1)
|
|
1180
|
+
except Exception:
|
|
1181
|
+
pass
|
|
1182
|
+
|
|
1183
|
+
self.logger.error(f"Critical error in concurrent processing: {e}")
|
|
1184
|
+
# Fallback to sequential processing
|
|
1185
|
+
self.logger.info(
|
|
1186
|
+
"Falling back to sequential processing due to critical error..."
|
|
1187
|
+
)
|
|
1188
|
+
return await self._process_files_sequentially(files_to_analyze)
|
|
1189
|
+
|
|
1190
|
+
finally:
|
|
1191
|
+
# Final cleanup: ensure all tasks are properly finished
|
|
1192
|
+
if tasks:
|
|
1193
|
+
for task in tasks:
|
|
1194
|
+
if not task.done() and not task.cancelled():
|
|
1195
|
+
task.cancel()
|
|
1196
|
+
|
|
1197
|
+
# Clear task references to help with garbage collection
|
|
1198
|
+
tasks.clear()
|
|
1199
|
+
|
|
1200
|
+
# Force garbage collection to help clean up semaphore and related resources
|
|
1201
|
+
import gc
|
|
1202
|
+
|
|
1203
|
+
gc.collect()
|
|
1204
|
+
|
|
1205
|
+
async def build_all_indexes(self) -> Dict[str, str]:
|
|
1206
|
+
"""Build indexes for all repositories in code_base"""
|
|
1207
|
+
if not self.code_base_path.exists():
|
|
1208
|
+
raise FileNotFoundError(
|
|
1209
|
+
f"Code base path does not exist: {self.code_base_path}"
|
|
1210
|
+
)
|
|
1211
|
+
|
|
1212
|
+
# Get all repository directories
|
|
1213
|
+
repo_dirs = [
|
|
1214
|
+
d
|
|
1215
|
+
for d in self.code_base_path.iterdir()
|
|
1216
|
+
if d.is_dir() and not d.name.startswith(".")
|
|
1217
|
+
]
|
|
1218
|
+
|
|
1219
|
+
if not repo_dirs:
|
|
1220
|
+
raise ValueError(f"No repositories found in {self.code_base_path}")
|
|
1221
|
+
|
|
1222
|
+
self.logger.info(f"Found {len(repo_dirs)} repositories to process")
|
|
1223
|
+
|
|
1224
|
+
# Process each repository
|
|
1225
|
+
output_files = {}
|
|
1226
|
+
statistics_data = []
|
|
1227
|
+
|
|
1228
|
+
for repo_dir in repo_dirs:
|
|
1229
|
+
try:
|
|
1230
|
+
# Process repository
|
|
1231
|
+
repo_index = await self.process_repository(repo_dir)
|
|
1232
|
+
|
|
1233
|
+
# Generate output filename using configured pattern
|
|
1234
|
+
output_filename = self.index_filename_pattern.format(
|
|
1235
|
+
repo_name=repo_index.repo_name
|
|
1236
|
+
)
|
|
1237
|
+
output_file = self.output_dir / output_filename
|
|
1238
|
+
|
|
1239
|
+
# Get output configuration
|
|
1240
|
+
output_config = self.indexer_config.get("output", {})
|
|
1241
|
+
json_indent = output_config.get("json_indent", 2)
|
|
1242
|
+
ensure_ascii = not output_config.get("ensure_ascii", False)
|
|
1243
|
+
|
|
1244
|
+
# Save to JSON file
|
|
1245
|
+
with open(output_file, "w", encoding="utf-8") as f:
|
|
1246
|
+
if self.include_metadata:
|
|
1247
|
+
json.dump(
|
|
1248
|
+
asdict(repo_index),
|
|
1249
|
+
f,
|
|
1250
|
+
indent=json_indent,
|
|
1251
|
+
ensure_ascii=ensure_ascii,
|
|
1252
|
+
)
|
|
1253
|
+
else:
|
|
1254
|
+
# Save without metadata if disabled
|
|
1255
|
+
index_data = asdict(repo_index)
|
|
1256
|
+
index_data.pop("analysis_metadata", None)
|
|
1257
|
+
json.dump(
|
|
1258
|
+
index_data, f, indent=json_indent, ensure_ascii=ensure_ascii
|
|
1259
|
+
)
|
|
1260
|
+
|
|
1261
|
+
output_files[repo_index.repo_name] = str(output_file)
|
|
1262
|
+
self.logger.info(
|
|
1263
|
+
f"Saved index for {repo_index.repo_name} to {output_file}"
|
|
1264
|
+
)
|
|
1265
|
+
|
|
1266
|
+
# Collect statistics for report
|
|
1267
|
+
if self.generate_statistics:
|
|
1268
|
+
stats = self._extract_repository_statistics(repo_index)
|
|
1269
|
+
statistics_data.append(stats)
|
|
1270
|
+
|
|
1271
|
+
except Exception as e:
|
|
1272
|
+
self.logger.error(f"Failed to process repository {repo_dir.name}: {e}")
|
|
1273
|
+
continue
|
|
1274
|
+
|
|
1275
|
+
# Generate additional reports if configured
|
|
1276
|
+
if self.generate_summary:
|
|
1277
|
+
summary_path = self.generate_summary_report(output_files)
|
|
1278
|
+
self.logger.info(f"Generated summary report: {summary_path}")
|
|
1279
|
+
|
|
1280
|
+
if self.generate_statistics:
|
|
1281
|
+
stats_path = self.generate_statistics_report(statistics_data)
|
|
1282
|
+
self.logger.info(f"Generated statistics report: {stats_path}")
|
|
1283
|
+
|
|
1284
|
+
return output_files
|
|
1285
|
+
|
|
1286
|
+
def _extract_repository_statistics(self, repo_index: RepoIndex) -> Dict[str, Any]:
|
|
1287
|
+
"""Extract statistical information from a repository index"""
|
|
1288
|
+
metadata = repo_index.analysis_metadata
|
|
1289
|
+
|
|
1290
|
+
# Count relationship types
|
|
1291
|
+
relationship_type_counts = {}
|
|
1292
|
+
for rel in repo_index.relationships:
|
|
1293
|
+
rel_type = rel.relationship_type
|
|
1294
|
+
relationship_type_counts[rel_type] = (
|
|
1295
|
+
relationship_type_counts.get(rel_type, 0) + 1
|
|
1296
|
+
)
|
|
1297
|
+
|
|
1298
|
+
# Count file types
|
|
1299
|
+
file_type_counts = {}
|
|
1300
|
+
for file_summary in repo_index.file_summaries:
|
|
1301
|
+
file_type = file_summary.file_type
|
|
1302
|
+
file_type_counts[file_type] = file_type_counts.get(file_type, 0) + 1
|
|
1303
|
+
|
|
1304
|
+
# Calculate statistics
|
|
1305
|
+
total_lines = sum(fs.lines_of_code for fs in repo_index.file_summaries)
|
|
1306
|
+
avg_lines = (
|
|
1307
|
+
total_lines / len(repo_index.file_summaries)
|
|
1308
|
+
if repo_index.file_summaries
|
|
1309
|
+
else 0
|
|
1310
|
+
)
|
|
1311
|
+
|
|
1312
|
+
avg_confidence = (
|
|
1313
|
+
sum(r.confidence_score for r in repo_index.relationships)
|
|
1314
|
+
/ len(repo_index.relationships)
|
|
1315
|
+
if repo_index.relationships
|
|
1316
|
+
else 0
|
|
1317
|
+
)
|
|
1318
|
+
|
|
1319
|
+
return {
|
|
1320
|
+
"repo_name": repo_index.repo_name,
|
|
1321
|
+
"total_files": repo_index.total_files,
|
|
1322
|
+
"analyzed_files": len(repo_index.file_summaries),
|
|
1323
|
+
"total_relationships": len(repo_index.relationships),
|
|
1324
|
+
"high_confidence_relationships": metadata.get(
|
|
1325
|
+
"high_confidence_relationships", 0
|
|
1326
|
+
),
|
|
1327
|
+
"relationship_type_counts": relationship_type_counts,
|
|
1328
|
+
"file_type_counts": file_type_counts,
|
|
1329
|
+
"total_lines_of_code": total_lines,
|
|
1330
|
+
"average_lines_per_file": round(avg_lines, 2),
|
|
1331
|
+
"average_confidence_score": round(avg_confidence, 3),
|
|
1332
|
+
"filtering_efficiency": metadata.get("filtering_efficiency", 0),
|
|
1333
|
+
"concurrent_analysis_used": metadata.get("concurrent_analysis_used", False),
|
|
1334
|
+
"cache_hits": metadata.get("cache_hits", 0),
|
|
1335
|
+
"analysis_date": metadata.get("analysis_date", "unknown"),
|
|
1336
|
+
}
|
|
1337
|
+
|
|
1338
|
+
def generate_statistics_report(self, statistics_data: List[Dict[str, Any]]) -> str:
|
|
1339
|
+
"""Generate a detailed statistics report"""
|
|
1340
|
+
stats_path = self.output_dir / self.stats_filename
|
|
1341
|
+
|
|
1342
|
+
# Calculate aggregate statistics
|
|
1343
|
+
total_repos = len(statistics_data)
|
|
1344
|
+
total_files_analyzed = sum(stat["analyzed_files"] for stat in statistics_data)
|
|
1345
|
+
total_relationships = sum(
|
|
1346
|
+
stat["total_relationships"] for stat in statistics_data
|
|
1347
|
+
)
|
|
1348
|
+
total_lines = sum(stat["total_lines_of_code"] for stat in statistics_data)
|
|
1349
|
+
|
|
1350
|
+
# Aggregate relationship types
|
|
1351
|
+
aggregated_rel_types = {}
|
|
1352
|
+
for stat in statistics_data:
|
|
1353
|
+
for rel_type, count in stat["relationship_type_counts"].items():
|
|
1354
|
+
aggregated_rel_types[rel_type] = (
|
|
1355
|
+
aggregated_rel_types.get(rel_type, 0) + count
|
|
1356
|
+
)
|
|
1357
|
+
|
|
1358
|
+
# Aggregate file types
|
|
1359
|
+
aggregated_file_types = {}
|
|
1360
|
+
for stat in statistics_data:
|
|
1361
|
+
for file_type, count in stat["file_type_counts"].items():
|
|
1362
|
+
aggregated_file_types[file_type] = (
|
|
1363
|
+
aggregated_file_types.get(file_type, 0) + count
|
|
1364
|
+
)
|
|
1365
|
+
|
|
1366
|
+
# Calculate averages
|
|
1367
|
+
avg_files_per_repo = total_files_analyzed / total_repos if total_repos else 0
|
|
1368
|
+
avg_relationships_per_repo = (
|
|
1369
|
+
total_relationships / total_repos if total_repos else 0
|
|
1370
|
+
)
|
|
1371
|
+
avg_lines_per_repo = total_lines / total_repos if total_repos else 0
|
|
1372
|
+
|
|
1373
|
+
# Build statistics report
|
|
1374
|
+
statistics_report = {
|
|
1375
|
+
"report_generation_time": datetime.now().isoformat(),
|
|
1376
|
+
"analyzer_version": "1.3.0",
|
|
1377
|
+
"configuration_used": {
|
|
1378
|
+
"config_file": self.indexer_config_path,
|
|
1379
|
+
"concurrent_analysis_enabled": self.enable_concurrent_analysis,
|
|
1380
|
+
"content_caching_enabled": self.enable_content_caching,
|
|
1381
|
+
"pre_filtering_enabled": self.enable_pre_filtering,
|
|
1382
|
+
"min_confidence_score": self.min_confidence_score,
|
|
1383
|
+
"high_confidence_threshold": self.high_confidence_threshold,
|
|
1384
|
+
},
|
|
1385
|
+
"aggregate_statistics": {
|
|
1386
|
+
"total_repositories_processed": total_repos,
|
|
1387
|
+
"total_files_analyzed": total_files_analyzed,
|
|
1388
|
+
"total_relationships_found": total_relationships,
|
|
1389
|
+
"total_lines_of_code": total_lines,
|
|
1390
|
+
"average_files_per_repository": round(avg_files_per_repo, 2),
|
|
1391
|
+
"average_relationships_per_repository": round(
|
|
1392
|
+
avg_relationships_per_repo, 2
|
|
1393
|
+
),
|
|
1394
|
+
"average_lines_per_repository": round(avg_lines_per_repo, 2),
|
|
1395
|
+
},
|
|
1396
|
+
"relationship_type_distribution": aggregated_rel_types,
|
|
1397
|
+
"file_type_distribution": aggregated_file_types,
|
|
1398
|
+
"repository_details": statistics_data,
|
|
1399
|
+
"performance_metrics": {
|
|
1400
|
+
"concurrent_processing_repos": sum(
|
|
1401
|
+
1
|
|
1402
|
+
for s in statistics_data
|
|
1403
|
+
if s.get("concurrent_analysis_used", False)
|
|
1404
|
+
),
|
|
1405
|
+
"cache_efficiency": {
|
|
1406
|
+
"total_cache_hits": sum(
|
|
1407
|
+
s.get("cache_hits", 0) for s in statistics_data
|
|
1408
|
+
),
|
|
1409
|
+
"repositories_with_caching": sum(
|
|
1410
|
+
1 for s in statistics_data if s.get("cache_hits", 0) > 0
|
|
1411
|
+
),
|
|
1412
|
+
},
|
|
1413
|
+
"filtering_efficiency": {
|
|
1414
|
+
"average_filtering_efficiency": round(
|
|
1415
|
+
sum(s.get("filtering_efficiency", 0) for s in statistics_data)
|
|
1416
|
+
/ total_repos,
|
|
1417
|
+
2,
|
|
1418
|
+
)
|
|
1419
|
+
if total_repos
|
|
1420
|
+
else 0,
|
|
1421
|
+
"max_filtering_efficiency": max(
|
|
1422
|
+
(s.get("filtering_efficiency", 0) for s in statistics_data),
|
|
1423
|
+
default=0,
|
|
1424
|
+
),
|
|
1425
|
+
"min_filtering_efficiency": min(
|
|
1426
|
+
(s.get("filtering_efficiency", 0) for s in statistics_data),
|
|
1427
|
+
default=0,
|
|
1428
|
+
),
|
|
1429
|
+
},
|
|
1430
|
+
},
|
|
1431
|
+
}
|
|
1432
|
+
|
|
1433
|
+
# Get output configuration
|
|
1434
|
+
output_config = self.indexer_config.get("output", {})
|
|
1435
|
+
json_indent = output_config.get("json_indent", 2)
|
|
1436
|
+
ensure_ascii = not output_config.get("ensure_ascii", False)
|
|
1437
|
+
|
|
1438
|
+
with open(stats_path, "w", encoding="utf-8") as f:
|
|
1439
|
+
json.dump(
|
|
1440
|
+
statistics_report, f, indent=json_indent, ensure_ascii=ensure_ascii
|
|
1441
|
+
)
|
|
1442
|
+
|
|
1443
|
+
return str(stats_path)
|
|
1444
|
+
|
|
1445
|
+
def generate_summary_report(self, output_files: Dict[str, str]) -> str:
|
|
1446
|
+
"""Generate a summary report of all indexes created"""
|
|
1447
|
+
report_path = self.output_dir / "indexing_summary.json"
|
|
1448
|
+
|
|
1449
|
+
# Get output configuration from config file
|
|
1450
|
+
output_config = self.indexer_config.get("output", {})
|
|
1451
|
+
json_indent = output_config.get("json_indent", 2)
|
|
1452
|
+
ensure_ascii = not output_config.get("ensure_ascii", False)
|
|
1453
|
+
|
|
1454
|
+
summary_data = {
|
|
1455
|
+
"indexing_completion_time": datetime.now().isoformat(),
|
|
1456
|
+
"total_repositories_processed": len(output_files),
|
|
1457
|
+
"output_files": output_files,
|
|
1458
|
+
"target_structure": self.target_structure,
|
|
1459
|
+
"code_base_path": str(self.code_base_path),
|
|
1460
|
+
"configuration": {
|
|
1461
|
+
"config_file_used": self.indexer_config_path,
|
|
1462
|
+
"api_config_file": self.config_path,
|
|
1463
|
+
"pre_filtering_enabled": self.enable_pre_filtering,
|
|
1464
|
+
"min_confidence_score": self.min_confidence_score,
|
|
1465
|
+
"high_confidence_threshold": self.high_confidence_threshold,
|
|
1466
|
+
"max_file_size": self.max_file_size,
|
|
1467
|
+
"max_content_length": self.max_content_length,
|
|
1468
|
+
"request_delay": self.request_delay,
|
|
1469
|
+
"supported_extensions_count": len(self.supported_extensions),
|
|
1470
|
+
"skip_directories_count": len(self.skip_directories),
|
|
1471
|
+
},
|
|
1472
|
+
}
|
|
1473
|
+
|
|
1474
|
+
with open(report_path, "w", encoding="utf-8") as f:
|
|
1475
|
+
json.dump(summary_data, f, indent=json_indent, ensure_ascii=ensure_ascii)
|
|
1476
|
+
|
|
1477
|
+
return str(report_path)
|
|
1478
|
+
|
|
1479
|
+
|
|
1480
|
+
async def main():
|
|
1481
|
+
"""Main function to run the code indexer with full configuration support"""
|
|
1482
|
+
|
|
1483
|
+
# Configuration - can be overridden by config file
|
|
1484
|
+
config_file = "deepcode-mcp/tools/indexer_config.yaml"
|
|
1485
|
+
|
|
1486
|
+
# You can override these parameters or let them be read from config
|
|
1487
|
+
code_base_path = None # Will use config file value if None
|
|
1488
|
+
output_dir = None # Will use config file value if None
|
|
1489
|
+
|
|
1490
|
+
# Target structure - this should be customized for your specific project
|
|
1491
|
+
target_structure = """
|
|
1492
|
+
project/
|
|
1493
|
+
├── src/
|
|
1494
|
+
│ ├── core/
|
|
1495
|
+
│ │ ├── gcn.py # GCN encoder
|
|
1496
|
+
│ │ ├── diffusion.py # forward/reverse processes
|
|
1497
|
+
│ │ ├── denoiser.py # denoising MLP
|
|
1498
|
+
│ │ └── fusion.py # fusion combiner
|
|
1499
|
+
│ ├── models/ # model wrapper classes
|
|
1500
|
+
│ │ └── recdiff.py
|
|
1501
|
+
│ ├── utils/
|
|
1502
|
+
│ │ ├── data.py # loading & preprocessing
|
|
1503
|
+
│ │ ├── predictor.py # scoring functions
|
|
1504
|
+
│ │ ├── loss.py # loss functions
|
|
1505
|
+
│ │ ├── metrics.py # NDCG, Recall etc.
|
|
1506
|
+
│ │ └── sched.py # beta/alpha schedule utils
|
|
1507
|
+
│ └── configs/
|
|
1508
|
+
│ └── default.yaml # hyperparameters, paths
|
|
1509
|
+
├── tests/
|
|
1510
|
+
│ ├── test_gcn.py
|
|
1511
|
+
│ ├── test_diffusion.py
|
|
1512
|
+
│ ├── test_denoiser.py
|
|
1513
|
+
│ ├── test_loss.py
|
|
1514
|
+
│ └── test_pipeline.py
|
|
1515
|
+
├── docs/
|
|
1516
|
+
│ ├── architecture.md
|
|
1517
|
+
│ ├── api_reference.md
|
|
1518
|
+
│ └── README.md
|
|
1519
|
+
├── experiments/
|
|
1520
|
+
│ ├── run_experiment.py
|
|
1521
|
+
│ └── notebooks/
|
|
1522
|
+
│ └── analysis.ipynb
|
|
1523
|
+
├── requirements.txt
|
|
1524
|
+
└── setup.py
|
|
1525
|
+
"""
|
|
1526
|
+
|
|
1527
|
+
print("🚀 Starting Code Indexer with Enhanced Configuration Support")
|
|
1528
|
+
print(f"📋 Configuration file: {config_file}")
|
|
1529
|
+
|
|
1530
|
+
# Create indexer with full configuration support
|
|
1531
|
+
try:
|
|
1532
|
+
indexer = CodeIndexer(
|
|
1533
|
+
code_base_path=code_base_path, # None = read from config
|
|
1534
|
+
target_structure=target_structure, # Required - project specific
|
|
1535
|
+
output_dir=output_dir, # None = read from config
|
|
1536
|
+
indexer_config_path=config_file, # Configuration file
|
|
1537
|
+
enable_pre_filtering=True, # Can be overridden in config
|
|
1538
|
+
)
|
|
1539
|
+
|
|
1540
|
+
# Display configuration information
|
|
1541
|
+
print(f"📁 Code base path: {indexer.code_base_path}")
|
|
1542
|
+
print(f"📂 Output directory: {indexer.output_dir}")
|
|
1543
|
+
print(f"🤖 Model provider: {indexer.model_provider}")
|
|
1544
|
+
print(
|
|
1545
|
+
f"⚡ Concurrent analysis: {'enabled' if indexer.enable_concurrent_analysis else 'disabled'}"
|
|
1546
|
+
)
|
|
1547
|
+
print(
|
|
1548
|
+
f"🗄️ Content caching: {'enabled' if indexer.enable_content_caching else 'disabled'}"
|
|
1549
|
+
)
|
|
1550
|
+
print(
|
|
1551
|
+
f"🔍 Pre-filtering: {'enabled' if indexer.enable_pre_filtering else 'disabled'}"
|
|
1552
|
+
)
|
|
1553
|
+
print(f"🐛 Debug mode: {'enabled' if indexer.verbose_output else 'disabled'}")
|
|
1554
|
+
print(
|
|
1555
|
+
f"🎭 Mock responses: {'enabled' if indexer.mock_llm_responses else 'disabled'}"
|
|
1556
|
+
)
|
|
1557
|
+
|
|
1558
|
+
# Validate configuration
|
|
1559
|
+
if not indexer.code_base_path.exists():
|
|
1560
|
+
raise FileNotFoundError(
|
|
1561
|
+
f"Code base path does not exist: {indexer.code_base_path}"
|
|
1562
|
+
)
|
|
1563
|
+
|
|
1564
|
+
if not target_structure:
|
|
1565
|
+
raise ValueError("Target structure is required for analysis")
|
|
1566
|
+
|
|
1567
|
+
print("\n🔧 Starting indexing process...")
|
|
1568
|
+
|
|
1569
|
+
# Build all indexes
|
|
1570
|
+
output_files = await indexer.build_all_indexes()
|
|
1571
|
+
|
|
1572
|
+
# Display results
|
|
1573
|
+
print("\n✅ Indexing completed successfully!")
|
|
1574
|
+
print(f"📊 Processed {len(output_files)} repositories")
|
|
1575
|
+
print("📁 Output files:")
|
|
1576
|
+
for repo_name, file_path in output_files.items():
|
|
1577
|
+
print(f" - {repo_name}: {file_path}")
|
|
1578
|
+
|
|
1579
|
+
# Display additional reports generated
|
|
1580
|
+
if indexer.generate_summary:
|
|
1581
|
+
summary_file = indexer.output_dir / indexer.summary_filename
|
|
1582
|
+
if summary_file.exists():
|
|
1583
|
+
print(f"📋 Summary report: {summary_file}")
|
|
1584
|
+
|
|
1585
|
+
if indexer.generate_statistics:
|
|
1586
|
+
stats_file = indexer.output_dir / indexer.stats_filename
|
|
1587
|
+
if stats_file.exists():
|
|
1588
|
+
print(f"📈 Statistics report: {stats_file}")
|
|
1589
|
+
|
|
1590
|
+
# Performance information
|
|
1591
|
+
if indexer.enable_content_caching and indexer.content_cache:
|
|
1592
|
+
print(f"🗄️ Cache performance: {len(indexer.content_cache)} items cached")
|
|
1593
|
+
|
|
1594
|
+
print("\n🎉 Code indexing process completed successfully!")
|
|
1595
|
+
|
|
1596
|
+
except FileNotFoundError as e:
|
|
1597
|
+
print(f"❌ File not found error: {e}")
|
|
1598
|
+
print("💡 Please check your configuration file paths")
|
|
1599
|
+
except ValueError as e:
|
|
1600
|
+
print(f"❌ Configuration error: {e}")
|
|
1601
|
+
print("💡 Please check your configuration file settings")
|
|
1602
|
+
except Exception as e:
|
|
1603
|
+
print(f"❌ Indexing failed: {e}")
|
|
1604
|
+
print("💡 Check the logs for more details")
|
|
1605
|
+
|
|
1606
|
+
# Print debug information if available
|
|
1607
|
+
try:
|
|
1608
|
+
indexer
|
|
1609
|
+
if indexer.verbose_output:
|
|
1610
|
+
import traceback
|
|
1611
|
+
|
|
1612
|
+
print("\n🐛 Debug information:")
|
|
1613
|
+
traceback.print_exc()
|
|
1614
|
+
except NameError:
|
|
1615
|
+
pass
|
|
1616
|
+
|
|
1617
|
+
|
|
1618
|
+
def print_usage_example():
|
|
1619
|
+
"""Print usage examples for different scenarios"""
|
|
1620
|
+
print("""
|
|
1621
|
+
📖 Code Indexer Usage Examples:
|
|
1622
|
+
|
|
1623
|
+
1. Basic usage with config file:
|
|
1624
|
+
- Update paths in indexer_config.yaml
|
|
1625
|
+
- Run: python code_indexer.py
|
|
1626
|
+
|
|
1627
|
+
2. Enable debugging:
|
|
1628
|
+
- Set debug.verbose_output: true in config
|
|
1629
|
+
- Set debug.save_raw_responses: true to save LLM responses
|
|
1630
|
+
|
|
1631
|
+
3. Enable concurrent processing:
|
|
1632
|
+
- Set performance.enable_concurrent_analysis: true
|
|
1633
|
+
- Adjust performance.max_concurrent_files as needed
|
|
1634
|
+
|
|
1635
|
+
4. Enable caching:
|
|
1636
|
+
- Set performance.enable_content_caching: true
|
|
1637
|
+
- Adjust performance.max_cache_size as needed
|
|
1638
|
+
|
|
1639
|
+
5. Mock mode for testing:
|
|
1640
|
+
- Set debug.mock_llm_responses: true
|
|
1641
|
+
- No API calls will be made
|
|
1642
|
+
|
|
1643
|
+
6. Custom output:
|
|
1644
|
+
- Modify output.index_filename_pattern
|
|
1645
|
+
- Set output.generate_statistics: true for detailed reports
|
|
1646
|
+
|
|
1647
|
+
📋 Configuration file location: tools/indexer_config.yaml
|
|
1648
|
+
""")
|
|
1649
|
+
|
|
1650
|
+
|
|
1651
|
+
if __name__ == "__main__":
|
|
1652
|
+
import sys
|
|
1653
|
+
|
|
1654
|
+
if len(sys.argv) > 1 and sys.argv[1] in ["--help", "-h", "help"]:
|
|
1655
|
+
print_usage_example()
|
|
1656
|
+
else:
|
|
1657
|
+
asyncio.run(main())
|