deepcode-hku 1.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. cli/__init__.py +18 -0
  2. cli/cli_app.py +296 -0
  3. cli/cli_interface.py +744 -0
  4. cli/cli_launcher.py +155 -0
  5. cli/main_cli.py +243 -0
  6. cli/workflows/__init__.py +11 -0
  7. cli/workflows/cli_workflow_adapter.py +336 -0
  8. deepcode.py +219 -0
  9. deepcode_hku-1.0.1.dist-info/METADATA +695 -0
  10. deepcode_hku-1.0.1.dist-info/RECORD +44 -0
  11. deepcode_hku-1.0.1.dist-info/WHEEL +5 -0
  12. deepcode_hku-1.0.1.dist-info/entry_points.txt +2 -0
  13. deepcode_hku-1.0.1.dist-info/licenses/LICENSE +21 -0
  14. deepcode_hku-1.0.1.dist-info/top_level.txt +6 -0
  15. tools/__init__.py +0 -0
  16. tools/code_implementation_server.py +1045 -0
  17. tools/code_indexer.py +1657 -0
  18. tools/code_reference_indexer.py +486 -0
  19. tools/command_executor.py +324 -0
  20. tools/git_command.py +356 -0
  21. tools/pdf_converter.py +640 -0
  22. tools/pdf_downloader.py +1370 -0
  23. tools/pdf_utils.py +52 -0
  24. ui/__init__.py +43 -0
  25. ui/app.py +13 -0
  26. ui/components.py +1450 -0
  27. ui/handlers.py +773 -0
  28. ui/layout.py +106 -0
  29. ui/streamlit_app.py +38 -0
  30. ui/styles.py +2116 -0
  31. utils/__init__.py +17 -0
  32. utils/cli_interface.py +459 -0
  33. utils/dialogue_logger.py +671 -0
  34. utils/file_processor.py +426 -0
  35. utils/simple_llm_logger.py +198 -0
  36. workflows/__init__.py +31 -0
  37. workflows/agent_orchestration_engine.py +1371 -0
  38. workflows/agents/__init__.py +13 -0
  39. workflows/agents/code_implementation_agent.py +1093 -0
  40. workflows/agents/memory_agent_concise.py +923 -0
  41. workflows/agents/memory_agent_concise_index.py +935 -0
  42. workflows/code_implementation_workflow.py +924 -0
  43. workflows/code_implementation_workflow_index.py +931 -0
  44. workflows/codebase_index_workflow.py +726 -0
tools/code_indexer.py ADDED
@@ -0,0 +1,1657 @@
1
+ """
2
+ Code Indexer for Repository Analysis
3
+
4
+ Analyzes code repositories to build comprehensive indexes for each subdirectory,
5
+ identifying file relationships and reusable components for implementation.
6
+
7
+ Features:
8
+ - Recursive file traversal
9
+ - LLM-powered code similarity analysis
10
+ - JSON-based relationship storage
11
+ - Configurable matching strategies
12
+ - Progress tracking and error handling
13
+ """
14
+
15
+ import asyncio
16
+ import json
17
+ import logging
18
+ import os
19
+ import re
20
+ from datetime import datetime
21
+ from pathlib import Path
22
+ from dataclasses import dataclass, asdict
23
+ from typing import List, Dict, Any
24
+
25
+
26
+ @dataclass
27
+ class FileRelationship:
28
+ """Represents a relationship between a repo file and target structure file"""
29
+
30
+ repo_file_path: str
31
+ target_file_path: str
32
+ relationship_type: str # 'direct_match', 'partial_match', 'reference', 'utility'
33
+ confidence_score: float # 0.0 to 1.0
34
+ helpful_aspects: List[str]
35
+ potential_contributions: List[str]
36
+ usage_suggestions: str
37
+
38
+
39
+ @dataclass
40
+ class FileSummary:
41
+ """Summary information for a repository file"""
42
+
43
+ file_path: str
44
+ file_type: str
45
+ main_functions: List[str]
46
+ key_concepts: List[str]
47
+ dependencies: List[str]
48
+ summary: str
49
+ lines_of_code: int
50
+ last_modified: str
51
+
52
+
53
+ @dataclass
54
+ class RepoIndex:
55
+ """Complete index for a repository"""
56
+
57
+ repo_name: str
58
+ total_files: int
59
+ file_summaries: List[FileSummary]
60
+ relationships: List[FileRelationship]
61
+ analysis_metadata: Dict[str, Any]
62
+
63
+
64
+ class CodeIndexer:
65
+ """Main class for building code repository indexes"""
66
+
67
+ def __init__(
68
+ self,
69
+ code_base_path: str = None,
70
+ target_structure: str = None,
71
+ output_dir: str = None,
72
+ config_path: str = "mcp_agent.secrets.yaml",
73
+ indexer_config_path: str = None,
74
+ enable_pre_filtering: bool = True,
75
+ ):
76
+ # Load configurations first
77
+ self.config_path = config_path
78
+ self.indexer_config_path = indexer_config_path
79
+ self.api_config = self._load_api_config()
80
+ self.indexer_config = self._load_indexer_config()
81
+
82
+ # Use config paths if not provided as parameters
83
+ paths_config = self.indexer_config.get("paths", {})
84
+ self.code_base_path = Path(
85
+ code_base_path or paths_config.get("code_base_path", "code_base")
86
+ )
87
+ self.output_dir = Path(output_dir or paths_config.get("output_dir", "indexes"))
88
+ self.target_structure = (
89
+ target_structure # This must be provided as it's project-specific
90
+ )
91
+ self.enable_pre_filtering = enable_pre_filtering
92
+
93
+ # LLM clients
94
+ self.llm_client = None
95
+ self.llm_client_type = None
96
+
97
+ # Initialize logger early
98
+ self.logger = self._setup_logger()
99
+
100
+ # Create output directory if it doesn't exist
101
+ self.output_dir.mkdir(parents=True, exist_ok=True)
102
+
103
+ # Load file analysis configuration
104
+ file_analysis_config = self.indexer_config.get("file_analysis", {})
105
+ self.supported_extensions = set(
106
+ file_analysis_config.get(
107
+ "supported_extensions",
108
+ [
109
+ ".py",
110
+ ".js",
111
+ ".ts",
112
+ ".java",
113
+ ".cpp",
114
+ ".c",
115
+ ".h",
116
+ ".hpp",
117
+ ".cs",
118
+ ".php",
119
+ ".rb",
120
+ ".go",
121
+ ".rs",
122
+ ".scala",
123
+ ".kt",
124
+ ".swift",
125
+ ".m",
126
+ ".mm",
127
+ ".r",
128
+ ".matlab",
129
+ ".sql",
130
+ ".sh",
131
+ ".bat",
132
+ ".ps1",
133
+ ".yaml",
134
+ ".yml",
135
+ ".json",
136
+ ".xml",
137
+ ".toml",
138
+ ],
139
+ )
140
+ )
141
+
142
+ self.skip_directories = set(
143
+ file_analysis_config.get(
144
+ "skip_directories",
145
+ [
146
+ "__pycache__",
147
+ "node_modules",
148
+ "target",
149
+ "build",
150
+ "dist",
151
+ "venv",
152
+ "env",
153
+ ],
154
+ )
155
+ )
156
+
157
+ self.max_file_size = file_analysis_config.get("max_file_size", 1048576) # 1MB
158
+ self.max_content_length = file_analysis_config.get("max_content_length", 3000)
159
+
160
+ # Load LLM configuration
161
+ llm_config = self.indexer_config.get("llm", {})
162
+ self.model_provider = llm_config.get("model_provider", "anthropic")
163
+ self.llm_max_tokens = llm_config.get("max_tokens", 4000)
164
+ self.llm_temperature = llm_config.get("temperature", 0.3)
165
+ self.llm_system_prompt = llm_config.get(
166
+ "system_prompt",
167
+ "You are a code analysis expert. Provide precise, structured analysis of code relationships and similarities.",
168
+ )
169
+ self.request_delay = llm_config.get("request_delay", 0.1)
170
+ self.max_retries = llm_config.get("max_retries", 3)
171
+ self.retry_delay = llm_config.get("retry_delay", 1.0)
172
+
173
+ # Load relationship configuration
174
+ relationship_config = self.indexer_config.get("relationships", {})
175
+ self.min_confidence_score = relationship_config.get("min_confidence_score", 0.3)
176
+ self.high_confidence_threshold = relationship_config.get(
177
+ "high_confidence_threshold", 0.7
178
+ )
179
+ self.relationship_types = relationship_config.get(
180
+ "relationship_types",
181
+ {
182
+ "direct_match": 1.0,
183
+ "partial_match": 0.8,
184
+ "reference": 0.6,
185
+ "utility": 0.4,
186
+ },
187
+ )
188
+
189
+ # Load performance configuration
190
+ performance_config = self.indexer_config.get("performance", {})
191
+ self.enable_concurrent_analysis = performance_config.get(
192
+ "enable_concurrent_analysis", False
193
+ )
194
+ self.max_concurrent_files = performance_config.get("max_concurrent_files", 5)
195
+ self.enable_content_caching = performance_config.get(
196
+ "enable_content_caching", False
197
+ )
198
+ self.max_cache_size = performance_config.get("max_cache_size", 100)
199
+
200
+ # Load debug configuration
201
+ debug_config = self.indexer_config.get("debug", {})
202
+ self.save_raw_responses = debug_config.get("save_raw_responses", False)
203
+ self.raw_responses_dir = debug_config.get(
204
+ "raw_responses_dir", "debug_responses"
205
+ )
206
+ self.verbose_output = debug_config.get("verbose_output", False)
207
+ self.mock_llm_responses = debug_config.get("mock_llm_responses", False)
208
+
209
+ # Load output configuration
210
+ output_config = self.indexer_config.get("output", {})
211
+ self.generate_summary = output_config.get("generate_summary", True)
212
+ self.generate_statistics = output_config.get("generate_statistics", True)
213
+ self.include_metadata = output_config.get("include_metadata", True)
214
+ self.index_filename_pattern = output_config.get(
215
+ "index_filename_pattern", "{repo_name}_index.json"
216
+ )
217
+ self.summary_filename = output_config.get(
218
+ "summary_filename", "indexing_summary.json"
219
+ )
220
+ self.stats_filename = output_config.get(
221
+ "stats_filename", "indexing_statistics.json"
222
+ )
223
+
224
+ # Initialize caching if enabled
225
+ self.content_cache = {} if self.enable_content_caching else None
226
+
227
+ # Create debug directory if needed
228
+ if self.save_raw_responses:
229
+ Path(self.raw_responses_dir).mkdir(parents=True, exist_ok=True)
230
+
231
+ # Debug logging
232
+ if self.verbose_output:
233
+ self.logger.info(
234
+ f"Initialized CodeIndexer with config: {self.indexer_config_path}"
235
+ )
236
+ self.logger.info(f"Code base path: {self.code_base_path}")
237
+ self.logger.info(f"Output directory: {self.output_dir}")
238
+ self.logger.info(f"Model provider: {self.model_provider}")
239
+ self.logger.info(f"Concurrent analysis: {self.enable_concurrent_analysis}")
240
+ self.logger.info(f"Content caching: {self.enable_content_caching}")
241
+ self.logger.info(f"Mock LLM responses: {self.mock_llm_responses}")
242
+
243
+ def _setup_logger(self) -> logging.Logger:
244
+ """Setup logging configuration from config file"""
245
+ logger = logging.getLogger("CodeIndexer")
246
+
247
+ # Get logging config
248
+ logging_config = self.indexer_config.get("logging", {})
249
+ log_level = logging_config.get("level", "INFO")
250
+ log_format = logging_config.get(
251
+ "log_format", "%(asctime)s - %(name)s - %(levelname)s - %(message)s"
252
+ )
253
+
254
+ logger.setLevel(getattr(logging, log_level.upper(), logging.INFO))
255
+
256
+ # Clear existing handlers
257
+ logger.handlers.clear()
258
+
259
+ # Console handler
260
+ handler = logging.StreamHandler()
261
+ formatter = logging.Formatter(log_format)
262
+ handler.setFormatter(formatter)
263
+ logger.addHandler(handler)
264
+
265
+ # File handler if enabled
266
+ if logging_config.get("log_to_file", False):
267
+ log_file = logging_config.get("log_file", "indexer.log")
268
+ file_handler = logging.FileHandler(log_file, encoding="utf-8")
269
+ file_handler.setFormatter(formatter)
270
+ logger.addHandler(file_handler)
271
+
272
+ return logger
273
+
274
+ def _load_api_config(self) -> Dict[str, Any]:
275
+ """Load API configuration from YAML file"""
276
+ try:
277
+ import yaml
278
+
279
+ with open(self.config_path, "r", encoding="utf-8") as f:
280
+ return yaml.safe_load(f)
281
+ except Exception as e:
282
+ # Create a basic logger for this error since self.logger doesn't exist yet
283
+ print(f"Warning: Failed to load API config from {self.config_path}: {e}")
284
+ return {}
285
+
286
+ def _load_indexer_config(self) -> Dict[str, Any]:
287
+ """Load indexer configuration from YAML file"""
288
+ try:
289
+ import yaml
290
+
291
+ with open(self.indexer_config_path, "r", encoding="utf-8") as f:
292
+ config = yaml.safe_load(f)
293
+ if config is None:
294
+ config = {}
295
+ return config
296
+ except Exception as e:
297
+ print(
298
+ f"Warning: Failed to load indexer config from {self.indexer_config_path}: {e}"
299
+ )
300
+ print("Using default configuration values")
301
+ return {}
302
+
303
+ async def _initialize_llm_client(self):
304
+ """Initialize LLM client based on configured provider"""
305
+ if self.llm_client is not None:
306
+ return self.llm_client, self.llm_client_type
307
+
308
+ # Check if mock responses are enabled
309
+ if self.mock_llm_responses:
310
+ self.logger.info("Using mock LLM responses for testing")
311
+ self.llm_client = "mock"
312
+ self.llm_client_type = "mock"
313
+ return "mock", "mock"
314
+
315
+ # Try configured provider first
316
+ if self.model_provider.lower() == "anthropic":
317
+ try:
318
+ anthropic_key = self.api_config.get("anthropic", {}).get("api_key")
319
+ if anthropic_key:
320
+ from anthropic import AsyncAnthropic
321
+
322
+ client = AsyncAnthropic(api_key=anthropic_key)
323
+ # Test connection
324
+ await client.messages.create(
325
+ model="claude-sonnet-4-20250514",
326
+ max_tokens=10,
327
+ messages=[{"role": "user", "content": "test"}],
328
+ )
329
+ self.logger.info("Using Anthropic API for code analysis")
330
+ self.llm_client = client
331
+ self.llm_client_type = "anthropic"
332
+ return client, "anthropic"
333
+ except Exception as e:
334
+ self.logger.warning(f"Configured Anthropic API unavailable: {e}")
335
+
336
+ elif self.model_provider.lower() == "openai":
337
+ try:
338
+ openai_key = self.api_config.get("openai", {}).get("api_key")
339
+ if openai_key:
340
+ from openai import AsyncOpenAI
341
+
342
+ client = AsyncOpenAI(api_key=openai_key)
343
+ # Test connection
344
+ await client.chat.completions.create(
345
+ model="gpt-3.5-turbo",
346
+ max_tokens=10,
347
+ messages=[{"role": "user", "content": "test"}],
348
+ )
349
+ self.logger.info("Using OpenAI API for code analysis")
350
+ self.llm_client = client
351
+ self.llm_client_type = "openai"
352
+ return client, "openai"
353
+ except Exception as e:
354
+ self.logger.warning(f"Configured OpenAI API unavailable: {e}")
355
+
356
+ # Fallback: try other provider
357
+ self.logger.info("Trying fallback provider...")
358
+
359
+ # Try Anthropic as fallback
360
+ try:
361
+ anthropic_key = self.api_config.get("anthropic", {}).get("api_key")
362
+ if anthropic_key:
363
+ from anthropic import AsyncAnthropic
364
+
365
+ client = AsyncAnthropic(api_key=anthropic_key)
366
+ await client.messages.create(
367
+ model="claude-sonnet-4-20250514",
368
+ max_tokens=10,
369
+ messages=[{"role": "user", "content": "test"}],
370
+ )
371
+ self.logger.info("Using Anthropic API as fallback")
372
+ self.llm_client = client
373
+ self.llm_client_type = "anthropic"
374
+ return client, "anthropic"
375
+ except Exception as e:
376
+ self.logger.warning(f"Anthropic fallback failed: {e}")
377
+
378
+ # Try OpenAI as fallback
379
+ try:
380
+ openai_key = self.api_config.get("openai", {}).get("api_key")
381
+ if openai_key:
382
+ from openai import AsyncOpenAI
383
+
384
+ client = AsyncOpenAI(api_key=openai_key)
385
+ await client.chat.completions.create(
386
+ model="gpt-3.5-turbo",
387
+ max_tokens=10,
388
+ messages=[{"role": "user", "content": "test"}],
389
+ )
390
+ self.logger.info("Using OpenAI API as fallback")
391
+ self.llm_client = client
392
+ self.llm_client_type = "openai"
393
+ return client, "openai"
394
+ except Exception as e:
395
+ self.logger.warning(f"OpenAI fallback failed: {e}")
396
+
397
+ raise ValueError("No available LLM API for code analysis")
398
+
399
+ async def _call_llm(
400
+ self, prompt: str, system_prompt: str = None, max_tokens: int = None
401
+ ) -> str:
402
+ """Call LLM for code analysis with retry mechanism and debugging support"""
403
+ if system_prompt is None:
404
+ system_prompt = self.llm_system_prompt
405
+ if max_tokens is None:
406
+ max_tokens = self.llm_max_tokens
407
+
408
+ # Mock response for testing
409
+ if self.mock_llm_responses:
410
+ mock_response = self._generate_mock_response(prompt)
411
+ if self.save_raw_responses:
412
+ self._save_debug_response("mock", prompt, mock_response)
413
+ return mock_response
414
+
415
+ last_error = None
416
+
417
+ # Retry mechanism
418
+ for attempt in range(self.max_retries):
419
+ try:
420
+ if self.verbose_output and attempt > 0:
421
+ self.logger.info(
422
+ f"LLM call attempt {attempt + 1}/{self.max_retries}"
423
+ )
424
+
425
+ client, client_type = await self._initialize_llm_client()
426
+
427
+ if client_type == "anthropic":
428
+ response = await client.messages.create(
429
+ model="claude-sonnet-4-20250514",
430
+ system=system_prompt,
431
+ messages=[{"role": "user", "content": prompt}],
432
+ max_tokens=max_tokens,
433
+ temperature=self.llm_temperature,
434
+ )
435
+
436
+ content = ""
437
+ for block in response.content:
438
+ if block.type == "text":
439
+ content += block.text
440
+
441
+ # Save debug response if enabled
442
+ if self.save_raw_responses:
443
+ self._save_debug_response("anthropic", prompt, content)
444
+
445
+ return content
446
+
447
+ elif client_type == "openai":
448
+ messages = [
449
+ {"role": "system", "content": system_prompt},
450
+ {"role": "user", "content": prompt},
451
+ ]
452
+
453
+ response = await client.chat.completions.create(
454
+ model="gpt-4-1106-preview",
455
+ messages=messages,
456
+ max_tokens=max_tokens,
457
+ temperature=self.llm_temperature,
458
+ )
459
+
460
+ content = response.choices[0].message.content or ""
461
+
462
+ # Save debug response if enabled
463
+ if self.save_raw_responses:
464
+ self._save_debug_response("openai", prompt, content)
465
+
466
+ return content
467
+ else:
468
+ raise ValueError(f"Unsupported client type: {client_type}")
469
+
470
+ except Exception as e:
471
+ last_error = e
472
+ self.logger.warning(f"LLM call attempt {attempt + 1} failed: {e}")
473
+
474
+ if attempt < self.max_retries - 1:
475
+ await asyncio.sleep(
476
+ self.retry_delay * (attempt + 1)
477
+ ) # Exponential backoff
478
+
479
+ # All retries failed
480
+ error_msg = f"LLM call failed after {self.max_retries} attempts. Last error: {str(last_error)}"
481
+ self.logger.error(error_msg)
482
+ return f"Error in LLM analysis: {error_msg}"
483
+
484
+ def _generate_mock_response(self, prompt: str) -> str:
485
+ """Generate mock LLM response for testing"""
486
+ if "JSON format" in prompt and "file_type" in prompt:
487
+ # File analysis mock
488
+ return """
489
+ {
490
+ "file_type": "Python module",
491
+ "main_functions": ["main_function", "helper_function"],
492
+ "key_concepts": ["data_processing", "algorithm"],
493
+ "dependencies": ["numpy", "pandas"],
494
+ "summary": "Mock analysis of code file functionality."
495
+ }
496
+ """
497
+ elif "relationships" in prompt:
498
+ # Relationship analysis mock
499
+ return """
500
+ {
501
+ "relationships": [
502
+ {
503
+ "target_file_path": "src/core/mock.py",
504
+ "relationship_type": "partial_match",
505
+ "confidence_score": 0.8,
506
+ "helpful_aspects": ["algorithm implementation", "data structures"],
507
+ "potential_contributions": ["core functionality", "utility methods"],
508
+ "usage_suggestions": "Mock relationship suggestion for testing."
509
+ }
510
+ ]
511
+ }
512
+ """
513
+ elif "relevant_files" in prompt:
514
+ # File filtering mock
515
+ return """
516
+ {
517
+ "relevant_files": [
518
+ {
519
+ "file_path": "mock_file.py",
520
+ "relevance_reason": "Mock relevance reason",
521
+ "confidence": 0.9,
522
+ "expected_contribution": "Mock contribution"
523
+ }
524
+ ],
525
+ "summary": {
526
+ "total_files_analyzed": "10",
527
+ "relevant_files_count": "1",
528
+ "filtering_strategy": "Mock filtering strategy"
529
+ }
530
+ }
531
+ """
532
+ else:
533
+ return "Mock LLM response for testing purposes."
534
+
535
+ def _save_debug_response(self, provider: str, prompt: str, response: str):
536
+ """Save LLM response for debugging"""
537
+ try:
538
+ import hashlib
539
+ from datetime import datetime
540
+
541
+ # Create a hash of the prompt for filename
542
+ prompt_hash = hashlib.md5(prompt.encode()).hexdigest()[:8]
543
+ timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
544
+ filename = f"{provider}_{timestamp}_{prompt_hash}.json"
545
+
546
+ debug_data = {
547
+ "timestamp": datetime.now().isoformat(),
548
+ "provider": provider,
549
+ "prompt": prompt[:500] + "..." if len(prompt) > 500 else prompt,
550
+ "response": response,
551
+ "full_prompt_length": len(prompt),
552
+ }
553
+
554
+ debug_file = Path(self.raw_responses_dir) / filename
555
+ with open(debug_file, "w", encoding="utf-8") as f:
556
+ json.dump(debug_data, f, indent=2, ensure_ascii=False)
557
+
558
+ except Exception as e:
559
+ self.logger.warning(f"Failed to save debug response: {e}")
560
+
561
+ def get_all_repo_files(self, repo_path: Path) -> List[Path]:
562
+ """Recursively get all supported files in a repository"""
563
+ files = []
564
+
565
+ try:
566
+ for root, dirs, filenames in os.walk(repo_path):
567
+ # Skip common non-code directories
568
+ dirs[:] = [
569
+ d
570
+ for d in dirs
571
+ if not d.startswith(".") and d not in self.skip_directories
572
+ ]
573
+
574
+ for filename in filenames:
575
+ file_path = Path(root) / filename
576
+ if file_path.suffix.lower() in self.supported_extensions:
577
+ files.append(file_path)
578
+
579
+ except Exception as e:
580
+ self.logger.error(f"Error traversing {repo_path}: {e}")
581
+
582
+ return files
583
+
584
+ def generate_file_tree(self, repo_path: Path, max_depth: int = 5) -> str:
585
+ """Generate file tree structure string for the repository"""
586
+ tree_lines = []
587
+
588
+ def add_to_tree(current_path: Path, prefix: str = "", depth: int = 0):
589
+ if depth > max_depth:
590
+ return
591
+
592
+ try:
593
+ items = sorted(
594
+ current_path.iterdir(), key=lambda x: (x.is_file(), x.name.lower())
595
+ )
596
+ # Filter out irrelevant directories and files
597
+ items = [
598
+ item
599
+ for item in items
600
+ if not item.name.startswith(".")
601
+ and item.name not in self.skip_directories
602
+ ]
603
+
604
+ for i, item in enumerate(items):
605
+ is_last = i == len(items) - 1
606
+ current_prefix = "└── " if is_last else "├── "
607
+ tree_lines.append(f"{prefix}{current_prefix}{item.name}")
608
+
609
+ if item.is_dir():
610
+ extension_prefix = " " if is_last else "│ "
611
+ add_to_tree(item, prefix + extension_prefix, depth + 1)
612
+ elif item.suffix.lower() in self.supported_extensions:
613
+ # Add file size information
614
+ try:
615
+ size = item.stat().st_size
616
+ if size > 1024:
617
+ size_str = f" ({size // 1024}KB)"
618
+ else:
619
+ size_str = f" ({size}B)"
620
+ tree_lines[-1] += size_str
621
+ except (OSError, PermissionError):
622
+ pass
623
+
624
+ except PermissionError:
625
+ tree_lines.append(f"{prefix}├── [Permission Denied]")
626
+ except Exception as e:
627
+ tree_lines.append(f"{prefix}├── [Error: {str(e)}]")
628
+
629
+ tree_lines.append(f"{repo_path.name}/")
630
+ add_to_tree(repo_path)
631
+ return "\n".join(tree_lines)
632
+
633
+ async def pre_filter_files(self, repo_path: Path, file_tree: str) -> List[str]:
634
+ """Use LLM to pre-filter relevant files based on target structure"""
635
+ filter_prompt = f"""
636
+ You are a code analysis expert. Please analyze the following code repository file tree based on the target project structure and filter out files that may be relevant to the target project.
637
+
638
+ Target Project Structure:
639
+ {self.target_structure}
640
+
641
+ Code Repository File Tree:
642
+ {file_tree}
643
+
644
+ Please analyze which files might be helpful for implementing the target project structure, including:
645
+ - Core algorithm implementation files (such as GCN, recommendation systems, graph neural networks, etc.)
646
+ - Data processing and preprocessing files
647
+ - Loss functions and evaluation metric files
648
+ - Configuration and utility files
649
+ - Test files
650
+ - Documentation files
651
+
652
+ Please return the filtering results in JSON format:
653
+ {{
654
+ "relevant_files": [
655
+ {{
656
+ "file_path": "file path relative to repository root",
657
+ "relevance_reason": "why this file is relevant",
658
+ "confidence": 0.0-1.0,
659
+ "expected_contribution": "expected contribution to the target project"
660
+ }}
661
+ ],
662
+ "summary": {{
663
+ "total_files_analyzed": "total number of files analyzed",
664
+ "relevant_files_count": "number of relevant files",
665
+ "filtering_strategy": "explanation of filtering strategy"
666
+ }}
667
+ }}
668
+
669
+ Only return files with confidence > {self.min_confidence_score}. Focus on files related to recommendation systems, graph neural networks, and diffusion models.
670
+ """
671
+
672
+ try:
673
+ self.logger.info("Starting LLM pre-filtering of files...")
674
+ llm_response = await self._call_llm(
675
+ filter_prompt,
676
+ system_prompt="You are a professional code analysis and project architecture expert, skilled at identifying code file functionality and relevance.",
677
+ max_tokens=2000,
678
+ )
679
+
680
+ # Parse JSON response
681
+ match = re.search(r"\{.*\}", llm_response, re.DOTALL)
682
+ if not match:
683
+ self.logger.warning(
684
+ "Unable to parse LLM filtering response, will use all files"
685
+ )
686
+ return []
687
+
688
+ filter_data = json.loads(match.group(0))
689
+ relevant_files = filter_data.get("relevant_files", [])
690
+
691
+ # Extract file paths
692
+ selected_files = []
693
+ for file_info in relevant_files:
694
+ file_path = file_info.get("file_path", "")
695
+ confidence = file_info.get("confidence", 0.0)
696
+ # Use configured minimum confidence threshold
697
+ if file_path and confidence > self.min_confidence_score:
698
+ selected_files.append(file_path)
699
+
700
+ summary = filter_data.get("summary", {})
701
+ self.logger.info(
702
+ f"LLM filtering completed: {summary.get('relevant_files_count', len(selected_files))} relevant files selected"
703
+ )
704
+ self.logger.info(
705
+ f"Filtering strategy: {summary.get('filtering_strategy', 'Not provided')}"
706
+ )
707
+
708
+ return selected_files
709
+
710
+ except Exception as e:
711
+ self.logger.error(f"LLM pre-filtering failed: {e}")
712
+ self.logger.info("Will fallback to analyzing all files")
713
+ return []
714
+
715
+ def filter_files_by_paths(
716
+ self, all_files: List[Path], selected_paths: List[str], repo_path: Path
717
+ ) -> List[Path]:
718
+ """Filter file list based on LLM-selected paths"""
719
+ if not selected_paths:
720
+ return all_files
721
+
722
+ filtered_files = []
723
+
724
+ for file_path in all_files:
725
+ # Get path relative to repository root
726
+ relative_path = str(file_path.relative_to(repo_path))
727
+
728
+ # Check if it's in the selected list
729
+ for selected_path in selected_paths:
730
+ # Normalize path comparison
731
+ if (
732
+ relative_path == selected_path
733
+ or relative_path.replace("\\", "/")
734
+ == selected_path.replace("\\", "/")
735
+ or selected_path in relative_path
736
+ or relative_path in selected_path
737
+ ):
738
+ filtered_files.append(file_path)
739
+ break
740
+
741
+ return filtered_files
742
+
743
+ def _get_cache_key(self, file_path: Path) -> str:
744
+ """Generate cache key for file content"""
745
+ try:
746
+ stats = file_path.stat()
747
+ return f"{file_path}:{stats.st_mtime}:{stats.st_size}"
748
+ except (OSError, PermissionError):
749
+ return str(file_path)
750
+
751
+ def _manage_cache_size(self):
752
+ """Manage cache size to stay within limits"""
753
+ if not self.enable_content_caching or not self.content_cache:
754
+ return
755
+
756
+ if len(self.content_cache) > self.max_cache_size:
757
+ # Remove oldest entries (simple FIFO strategy)
758
+ excess_count = len(self.content_cache) - self.max_cache_size + 10
759
+ keys_to_remove = list(self.content_cache.keys())[:excess_count]
760
+
761
+ for key in keys_to_remove:
762
+ del self.content_cache[key]
763
+
764
+ if self.verbose_output:
765
+ self.logger.info(
766
+ f"Cache cleaned: removed {excess_count} entries, {len(self.content_cache)} entries remaining"
767
+ )
768
+
769
+ async def analyze_file_content(self, file_path: Path) -> FileSummary:
770
+ """Analyze a single file and create summary with caching support"""
771
+ try:
772
+ # Check file size before reading
773
+ file_size = file_path.stat().st_size
774
+ if file_size > self.max_file_size:
775
+ self.logger.warning(
776
+ f"Skipping file {file_path} - size {file_size} bytes exceeds limit {self.max_file_size}"
777
+ )
778
+ return FileSummary(
779
+ file_path=str(file_path.relative_to(self.code_base_path)),
780
+ file_type="skipped - too large",
781
+ main_functions=[],
782
+ key_concepts=[],
783
+ dependencies=[],
784
+ summary=f"File skipped - size {file_size} bytes exceeds {self.max_file_size} byte limit",
785
+ lines_of_code=0,
786
+ last_modified=datetime.fromtimestamp(
787
+ file_path.stat().st_mtime
788
+ ).isoformat(),
789
+ )
790
+
791
+ # Check cache if enabled
792
+ cache_key = None
793
+ if self.enable_content_caching:
794
+ cache_key = self._get_cache_key(file_path)
795
+ if cache_key in self.content_cache:
796
+ if self.verbose_output:
797
+ self.logger.info(f"Using cached analysis for {file_path.name}")
798
+ return self.content_cache[cache_key]
799
+
800
+ with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
801
+ content = f.read()
802
+
803
+ # Get file stats
804
+ stats = file_path.stat()
805
+ lines_of_code = len([line for line in content.split("\n") if line.strip()])
806
+
807
+ # Truncate content based on config
808
+ content_for_analysis = content[: self.max_content_length]
809
+ content_suffix = "..." if len(content) > self.max_content_length else ""
810
+
811
+ # Create analysis prompt
812
+ analysis_prompt = f"""
813
+ Analyze this code file and provide a structured summary:
814
+
815
+ File: {file_path.name}
816
+ Content:
817
+ ```
818
+ {content_for_analysis}{content_suffix}
819
+ ```
820
+
821
+ Please provide analysis in this JSON format:
822
+ {{
823
+ "file_type": "description of what type of file this is",
824
+ "main_functions": ["list", "of", "main", "functions", "or", "classes"],
825
+ "key_concepts": ["important", "concepts", "algorithms", "patterns"],
826
+ "dependencies": ["external", "libraries", "or", "imports"],
827
+ "summary": "2-3 sentence summary of what this file does"
828
+ }}
829
+
830
+ Focus on the core functionality and potential reusability.
831
+ """
832
+
833
+ # Get LLM analysis with configured parameters
834
+ llm_response = await self._call_llm(analysis_prompt, max_tokens=1000)
835
+
836
+ try:
837
+ # Try to parse JSON response
838
+ match = re.search(r"\{.*\}", llm_response, re.DOTALL)
839
+ analysis_data = json.loads(match.group(0))
840
+ except json.JSONDecodeError:
841
+ # Fallback to basic analysis if JSON parsing fails
842
+ analysis_data = {
843
+ "file_type": f"{file_path.suffix} file",
844
+ "main_functions": [],
845
+ "key_concepts": [],
846
+ "dependencies": [],
847
+ "summary": "File analysis failed - JSON parsing error",
848
+ }
849
+
850
+ file_summary = FileSummary(
851
+ file_path=str(file_path.relative_to(self.code_base_path)),
852
+ file_type=analysis_data.get("file_type", "unknown"),
853
+ main_functions=analysis_data.get("main_functions", []),
854
+ key_concepts=analysis_data.get("key_concepts", []),
855
+ dependencies=analysis_data.get("dependencies", []),
856
+ summary=analysis_data.get("summary", "No summary available"),
857
+ lines_of_code=lines_of_code,
858
+ last_modified=datetime.fromtimestamp(stats.st_mtime).isoformat(),
859
+ )
860
+
861
+ # Cache the result if caching is enabled
862
+ if self.enable_content_caching and cache_key:
863
+ self.content_cache[cache_key] = file_summary
864
+ self._manage_cache_size()
865
+
866
+ return file_summary
867
+
868
+ except Exception as e:
869
+ self.logger.error(f"Error analyzing file {file_path}: {e}")
870
+ return FileSummary(
871
+ file_path=str(file_path.relative_to(self.code_base_path)),
872
+ file_type="error",
873
+ main_functions=[],
874
+ key_concepts=[],
875
+ dependencies=[],
876
+ summary=f"Analysis failed: {str(e)}",
877
+ lines_of_code=0,
878
+ last_modified="",
879
+ )
880
+
881
+ async def find_relationships(
882
+ self, file_summary: FileSummary
883
+ ) -> List[FileRelationship]:
884
+ """Find relationships between a repo file and target structure"""
885
+
886
+ # Build relationship type description from config
887
+ relationship_type_desc = []
888
+ for rel_type, weight in self.relationship_types.items():
889
+ relationship_type_desc.append(f"- {rel_type} (priority: {weight})")
890
+
891
+ relationship_prompt = f"""
892
+ Analyze the relationship between this existing code file and the target project structure.
893
+
894
+ Existing File Analysis:
895
+ - Path: {file_summary.file_path}
896
+ - Type: {file_summary.file_type}
897
+ - Functions: {', '.join(file_summary.main_functions)}
898
+ - Concepts: {', '.join(file_summary.key_concepts)}
899
+ - Summary: {file_summary.summary}
900
+
901
+ Target Project Structure:
902
+ {self.target_structure}
903
+
904
+ Available relationship types (with priority weights):
905
+ {chr(10).join(relationship_type_desc)}
906
+
907
+ Identify potential relationships and provide analysis in this JSON format:
908
+ {{
909
+ "relationships": [
910
+ {{
911
+ "target_file_path": "path/in/target/structure",
912
+ "relationship_type": "direct_match|partial_match|reference|utility",
913
+ "confidence_score": 0.0-1.0,
914
+ "helpful_aspects": ["specific", "aspects", "that", "could", "help"],
915
+ "potential_contributions": ["how", "this", "could", "contribute"],
916
+ "usage_suggestions": "detailed suggestion on how to use this file"
917
+ }}
918
+ ]
919
+ }}
920
+
921
+ Consider the priority weights when determining relationship types. Higher weight types should be preferred when multiple types apply.
922
+ Only include relationships with confidence > {self.min_confidence_score}. Focus on concrete, actionable connections.
923
+ """
924
+
925
+ try:
926
+ llm_response = await self._call_llm(relationship_prompt, max_tokens=1500)
927
+
928
+ match = re.search(r"\{.*\}", llm_response, re.DOTALL)
929
+ relationship_data = json.loads(match.group(0))
930
+
931
+ relationships = []
932
+ for rel_data in relationship_data.get("relationships", []):
933
+ confidence_score = float(rel_data.get("confidence_score", 0.0))
934
+ relationship_type = rel_data.get("relationship_type", "reference")
935
+
936
+ # Validate relationship type is in config
937
+ if relationship_type not in self.relationship_types:
938
+ if self.verbose_output:
939
+ self.logger.warning(
940
+ f"Unknown relationship type '{relationship_type}', using 'reference'"
941
+ )
942
+ relationship_type = "reference"
943
+
944
+ # Apply configured minimum confidence filter
945
+ if confidence_score > self.min_confidence_score:
946
+ relationship = FileRelationship(
947
+ repo_file_path=file_summary.file_path,
948
+ target_file_path=rel_data.get("target_file_path", ""),
949
+ relationship_type=relationship_type,
950
+ confidence_score=confidence_score,
951
+ helpful_aspects=rel_data.get("helpful_aspects", []),
952
+ potential_contributions=rel_data.get(
953
+ "potential_contributions", []
954
+ ),
955
+ usage_suggestions=rel_data.get("usage_suggestions", ""),
956
+ )
957
+ relationships.append(relationship)
958
+
959
+ return relationships
960
+
961
+ except Exception as e:
962
+ self.logger.error(
963
+ f"Error finding relationships for {file_summary.file_path}: {e}"
964
+ )
965
+ return []
966
+
967
+ async def _analyze_single_file_with_relationships(
968
+ self, file_path: Path, index: int, total: int
969
+ ) -> tuple:
970
+ """Analyze a single file and its relationships (for concurrent processing)"""
971
+ if self.verbose_output:
972
+ self.logger.info(f"Analyzing file {index}/{total}: {file_path.name}")
973
+
974
+ # Get file summary
975
+ file_summary = await self.analyze_file_content(file_path)
976
+
977
+ # Find relationships
978
+ relationships = await self.find_relationships(file_summary)
979
+
980
+ return file_summary, relationships
981
+
982
+ async def process_repository(self, repo_path: Path) -> RepoIndex:
983
+ """Process a single repository and create complete index with optional concurrent processing"""
984
+ repo_name = repo_path.name
985
+ self.logger.info(f"Processing repository: {repo_name}")
986
+
987
+ # Step 1: Generate file tree
988
+ self.logger.info("Generating file tree structure...")
989
+ file_tree = self.generate_file_tree(repo_path)
990
+
991
+ # Step 2: Get all files
992
+ all_files = self.get_all_repo_files(repo_path)
993
+ self.logger.info(f"Found {len(all_files)} files in {repo_name}")
994
+
995
+ # Step 3: LLM pre-filtering of relevant files
996
+ if self.enable_pre_filtering:
997
+ self.logger.info("Using LLM for file pre-filtering...")
998
+ selected_file_paths = await self.pre_filter_files(repo_path, file_tree)
999
+ else:
1000
+ self.logger.info("Pre-filtering is disabled, will analyze all files")
1001
+ selected_file_paths = []
1002
+
1003
+ # Step 4: Filter file list based on filtering results
1004
+ if selected_file_paths:
1005
+ files_to_analyze = self.filter_files_by_paths(
1006
+ all_files, selected_file_paths, repo_path
1007
+ )
1008
+ self.logger.info(
1009
+ f"After LLM filtering, will analyze {len(files_to_analyze)} relevant files (from {len(all_files)} total)"
1010
+ )
1011
+ else:
1012
+ files_to_analyze = all_files
1013
+ self.logger.info("LLM filtering failed, will analyze all files")
1014
+
1015
+ # Step 5: Analyze filtered files (concurrent or sequential)
1016
+ if self.enable_concurrent_analysis and len(files_to_analyze) > 1:
1017
+ self.logger.info(
1018
+ f"Using concurrent analysis with max {self.max_concurrent_files} parallel files"
1019
+ )
1020
+ file_summaries, all_relationships = await self._process_files_concurrently(
1021
+ files_to_analyze
1022
+ )
1023
+ else:
1024
+ self.logger.info("Using sequential file analysis")
1025
+ file_summaries, all_relationships = await self._process_files_sequentially(
1026
+ files_to_analyze
1027
+ )
1028
+
1029
+ # Step 6: Create repository index
1030
+ repo_index = RepoIndex(
1031
+ repo_name=repo_name,
1032
+ total_files=len(all_files), # Record original file count
1033
+ file_summaries=file_summaries,
1034
+ relationships=all_relationships,
1035
+ analysis_metadata={
1036
+ "analysis_date": datetime.now().isoformat(),
1037
+ "target_structure_analyzed": self.target_structure[:200] + "...",
1038
+ "total_relationships_found": len(all_relationships),
1039
+ "high_confidence_relationships": len(
1040
+ [
1041
+ r
1042
+ for r in all_relationships
1043
+ if r.confidence_score > self.high_confidence_threshold
1044
+ ]
1045
+ ),
1046
+ "analyzer_version": "1.3.0", # Updated version to reflect concurrent support
1047
+ "pre_filtering_enabled": self.enable_pre_filtering,
1048
+ "files_before_filtering": len(all_files),
1049
+ "files_after_filtering": len(files_to_analyze),
1050
+ "filtering_efficiency": round(
1051
+ (1 - len(files_to_analyze) / len(all_files)) * 100, 2
1052
+ )
1053
+ if all_files
1054
+ else 0,
1055
+ "config_file_used": self.indexer_config_path,
1056
+ "min_confidence_score": self.min_confidence_score,
1057
+ "high_confidence_threshold": self.high_confidence_threshold,
1058
+ "concurrent_analysis_used": self.enable_concurrent_analysis,
1059
+ "content_caching_enabled": self.enable_content_caching,
1060
+ "cache_hits": len(self.content_cache) if self.content_cache else 0,
1061
+ },
1062
+ )
1063
+
1064
+ return repo_index
1065
+
1066
+ async def _process_files_sequentially(self, files_to_analyze: list) -> tuple:
1067
+ """Process files sequentially (original method)"""
1068
+ file_summaries = []
1069
+ all_relationships = []
1070
+
1071
+ for i, file_path in enumerate(files_to_analyze, 1):
1072
+ (
1073
+ file_summary,
1074
+ relationships,
1075
+ ) = await self._analyze_single_file_with_relationships(
1076
+ file_path, i, len(files_to_analyze)
1077
+ )
1078
+ file_summaries.append(file_summary)
1079
+ all_relationships.extend(relationships)
1080
+
1081
+ # Add configured delay to avoid overwhelming the LLM API
1082
+ await asyncio.sleep(self.request_delay)
1083
+
1084
+ return file_summaries, all_relationships
1085
+
1086
+ async def _process_files_concurrently(self, files_to_analyze: list) -> tuple:
1087
+ """Process files concurrently with semaphore limiting"""
1088
+ file_summaries = []
1089
+ all_relationships = []
1090
+
1091
+ # Create semaphore to limit concurrent tasks
1092
+ semaphore = asyncio.Semaphore(self.max_concurrent_files)
1093
+ tasks = []
1094
+
1095
+ async def _process_with_semaphore(file_path: Path, index: int, total: int):
1096
+ async with semaphore:
1097
+ # Add a small delay to space out concurrent requests
1098
+ if index > 1:
1099
+ await asyncio.sleep(
1100
+ self.request_delay * 0.5
1101
+ ) # Reduced delay for concurrent processing
1102
+ return await self._analyze_single_file_with_relationships(
1103
+ file_path, index, total
1104
+ )
1105
+
1106
+ try:
1107
+ # Create tasks for all files
1108
+ tasks = [
1109
+ _process_with_semaphore(file_path, i, len(files_to_analyze))
1110
+ for i, file_path in enumerate(files_to_analyze, 1)
1111
+ ]
1112
+
1113
+ # Process tasks and collect results
1114
+ if self.verbose_output:
1115
+ self.logger.info(
1116
+ f"Starting concurrent analysis of {len(tasks)} files..."
1117
+ )
1118
+
1119
+ try:
1120
+ results = await asyncio.gather(*tasks, return_exceptions=True)
1121
+
1122
+ for i, result in enumerate(results):
1123
+ if isinstance(result, Exception):
1124
+ self.logger.error(
1125
+ f"Failed to analyze file {files_to_analyze[i]}: {result}"
1126
+ )
1127
+ # Create error summary
1128
+ error_summary = FileSummary(
1129
+ file_path=str(
1130
+ files_to_analyze[i].relative_to(self.code_base_path)
1131
+ ),
1132
+ file_type="error",
1133
+ main_functions=[],
1134
+ key_concepts=[],
1135
+ dependencies=[],
1136
+ summary=f"Concurrent analysis failed: {str(result)}",
1137
+ lines_of_code=0,
1138
+ last_modified="",
1139
+ )
1140
+ file_summaries.append(error_summary)
1141
+ else:
1142
+ file_summary, relationships = result
1143
+ file_summaries.append(file_summary)
1144
+ all_relationships.extend(relationships)
1145
+
1146
+ except Exception as e:
1147
+ self.logger.error(f"Concurrent processing failed: {e}")
1148
+ # Cancel any remaining tasks
1149
+ for task in tasks:
1150
+ if not task.done() and not task.cancelled():
1151
+ task.cancel()
1152
+
1153
+ # Wait for cancelled tasks to complete
1154
+ try:
1155
+ await asyncio.sleep(0.1) # Brief wait for cancellation
1156
+ except Exception:
1157
+ pass
1158
+
1159
+ # Fallback to sequential processing
1160
+ self.logger.info("Falling back to sequential processing...")
1161
+ return await self._process_files_sequentially(files_to_analyze)
1162
+
1163
+ if self.verbose_output:
1164
+ self.logger.info(
1165
+ f"Concurrent analysis completed: {len(file_summaries)} files processed"
1166
+ )
1167
+
1168
+ return file_summaries, all_relationships
1169
+
1170
+ except Exception as e:
1171
+ # Ensure all tasks are cancelled in case of unexpected errors
1172
+ if tasks:
1173
+ for task in tasks:
1174
+ if not task.done() and not task.cancelled():
1175
+ task.cancel()
1176
+
1177
+ # Wait briefly for cancellation to complete
1178
+ try:
1179
+ await asyncio.sleep(0.1)
1180
+ except Exception:
1181
+ pass
1182
+
1183
+ self.logger.error(f"Critical error in concurrent processing: {e}")
1184
+ # Fallback to sequential processing
1185
+ self.logger.info(
1186
+ "Falling back to sequential processing due to critical error..."
1187
+ )
1188
+ return await self._process_files_sequentially(files_to_analyze)
1189
+
1190
+ finally:
1191
+ # Final cleanup: ensure all tasks are properly finished
1192
+ if tasks:
1193
+ for task in tasks:
1194
+ if not task.done() and not task.cancelled():
1195
+ task.cancel()
1196
+
1197
+ # Clear task references to help with garbage collection
1198
+ tasks.clear()
1199
+
1200
+ # Force garbage collection to help clean up semaphore and related resources
1201
+ import gc
1202
+
1203
+ gc.collect()
1204
+
1205
+ async def build_all_indexes(self) -> Dict[str, str]:
1206
+ """Build indexes for all repositories in code_base"""
1207
+ if not self.code_base_path.exists():
1208
+ raise FileNotFoundError(
1209
+ f"Code base path does not exist: {self.code_base_path}"
1210
+ )
1211
+
1212
+ # Get all repository directories
1213
+ repo_dirs = [
1214
+ d
1215
+ for d in self.code_base_path.iterdir()
1216
+ if d.is_dir() and not d.name.startswith(".")
1217
+ ]
1218
+
1219
+ if not repo_dirs:
1220
+ raise ValueError(f"No repositories found in {self.code_base_path}")
1221
+
1222
+ self.logger.info(f"Found {len(repo_dirs)} repositories to process")
1223
+
1224
+ # Process each repository
1225
+ output_files = {}
1226
+ statistics_data = []
1227
+
1228
+ for repo_dir in repo_dirs:
1229
+ try:
1230
+ # Process repository
1231
+ repo_index = await self.process_repository(repo_dir)
1232
+
1233
+ # Generate output filename using configured pattern
1234
+ output_filename = self.index_filename_pattern.format(
1235
+ repo_name=repo_index.repo_name
1236
+ )
1237
+ output_file = self.output_dir / output_filename
1238
+
1239
+ # Get output configuration
1240
+ output_config = self.indexer_config.get("output", {})
1241
+ json_indent = output_config.get("json_indent", 2)
1242
+ ensure_ascii = not output_config.get("ensure_ascii", False)
1243
+
1244
+ # Save to JSON file
1245
+ with open(output_file, "w", encoding="utf-8") as f:
1246
+ if self.include_metadata:
1247
+ json.dump(
1248
+ asdict(repo_index),
1249
+ f,
1250
+ indent=json_indent,
1251
+ ensure_ascii=ensure_ascii,
1252
+ )
1253
+ else:
1254
+ # Save without metadata if disabled
1255
+ index_data = asdict(repo_index)
1256
+ index_data.pop("analysis_metadata", None)
1257
+ json.dump(
1258
+ index_data, f, indent=json_indent, ensure_ascii=ensure_ascii
1259
+ )
1260
+
1261
+ output_files[repo_index.repo_name] = str(output_file)
1262
+ self.logger.info(
1263
+ f"Saved index for {repo_index.repo_name} to {output_file}"
1264
+ )
1265
+
1266
+ # Collect statistics for report
1267
+ if self.generate_statistics:
1268
+ stats = self._extract_repository_statistics(repo_index)
1269
+ statistics_data.append(stats)
1270
+
1271
+ except Exception as e:
1272
+ self.logger.error(f"Failed to process repository {repo_dir.name}: {e}")
1273
+ continue
1274
+
1275
+ # Generate additional reports if configured
1276
+ if self.generate_summary:
1277
+ summary_path = self.generate_summary_report(output_files)
1278
+ self.logger.info(f"Generated summary report: {summary_path}")
1279
+
1280
+ if self.generate_statistics:
1281
+ stats_path = self.generate_statistics_report(statistics_data)
1282
+ self.logger.info(f"Generated statistics report: {stats_path}")
1283
+
1284
+ return output_files
1285
+
1286
+ def _extract_repository_statistics(self, repo_index: RepoIndex) -> Dict[str, Any]:
1287
+ """Extract statistical information from a repository index"""
1288
+ metadata = repo_index.analysis_metadata
1289
+
1290
+ # Count relationship types
1291
+ relationship_type_counts = {}
1292
+ for rel in repo_index.relationships:
1293
+ rel_type = rel.relationship_type
1294
+ relationship_type_counts[rel_type] = (
1295
+ relationship_type_counts.get(rel_type, 0) + 1
1296
+ )
1297
+
1298
+ # Count file types
1299
+ file_type_counts = {}
1300
+ for file_summary in repo_index.file_summaries:
1301
+ file_type = file_summary.file_type
1302
+ file_type_counts[file_type] = file_type_counts.get(file_type, 0) + 1
1303
+
1304
+ # Calculate statistics
1305
+ total_lines = sum(fs.lines_of_code for fs in repo_index.file_summaries)
1306
+ avg_lines = (
1307
+ total_lines / len(repo_index.file_summaries)
1308
+ if repo_index.file_summaries
1309
+ else 0
1310
+ )
1311
+
1312
+ avg_confidence = (
1313
+ sum(r.confidence_score for r in repo_index.relationships)
1314
+ / len(repo_index.relationships)
1315
+ if repo_index.relationships
1316
+ else 0
1317
+ )
1318
+
1319
+ return {
1320
+ "repo_name": repo_index.repo_name,
1321
+ "total_files": repo_index.total_files,
1322
+ "analyzed_files": len(repo_index.file_summaries),
1323
+ "total_relationships": len(repo_index.relationships),
1324
+ "high_confidence_relationships": metadata.get(
1325
+ "high_confidence_relationships", 0
1326
+ ),
1327
+ "relationship_type_counts": relationship_type_counts,
1328
+ "file_type_counts": file_type_counts,
1329
+ "total_lines_of_code": total_lines,
1330
+ "average_lines_per_file": round(avg_lines, 2),
1331
+ "average_confidence_score": round(avg_confidence, 3),
1332
+ "filtering_efficiency": metadata.get("filtering_efficiency", 0),
1333
+ "concurrent_analysis_used": metadata.get("concurrent_analysis_used", False),
1334
+ "cache_hits": metadata.get("cache_hits", 0),
1335
+ "analysis_date": metadata.get("analysis_date", "unknown"),
1336
+ }
1337
+
1338
+ def generate_statistics_report(self, statistics_data: List[Dict[str, Any]]) -> str:
1339
+ """Generate a detailed statistics report"""
1340
+ stats_path = self.output_dir / self.stats_filename
1341
+
1342
+ # Calculate aggregate statistics
1343
+ total_repos = len(statistics_data)
1344
+ total_files_analyzed = sum(stat["analyzed_files"] for stat in statistics_data)
1345
+ total_relationships = sum(
1346
+ stat["total_relationships"] for stat in statistics_data
1347
+ )
1348
+ total_lines = sum(stat["total_lines_of_code"] for stat in statistics_data)
1349
+
1350
+ # Aggregate relationship types
1351
+ aggregated_rel_types = {}
1352
+ for stat in statistics_data:
1353
+ for rel_type, count in stat["relationship_type_counts"].items():
1354
+ aggregated_rel_types[rel_type] = (
1355
+ aggregated_rel_types.get(rel_type, 0) + count
1356
+ )
1357
+
1358
+ # Aggregate file types
1359
+ aggregated_file_types = {}
1360
+ for stat in statistics_data:
1361
+ for file_type, count in stat["file_type_counts"].items():
1362
+ aggregated_file_types[file_type] = (
1363
+ aggregated_file_types.get(file_type, 0) + count
1364
+ )
1365
+
1366
+ # Calculate averages
1367
+ avg_files_per_repo = total_files_analyzed / total_repos if total_repos else 0
1368
+ avg_relationships_per_repo = (
1369
+ total_relationships / total_repos if total_repos else 0
1370
+ )
1371
+ avg_lines_per_repo = total_lines / total_repos if total_repos else 0
1372
+
1373
+ # Build statistics report
1374
+ statistics_report = {
1375
+ "report_generation_time": datetime.now().isoformat(),
1376
+ "analyzer_version": "1.3.0",
1377
+ "configuration_used": {
1378
+ "config_file": self.indexer_config_path,
1379
+ "concurrent_analysis_enabled": self.enable_concurrent_analysis,
1380
+ "content_caching_enabled": self.enable_content_caching,
1381
+ "pre_filtering_enabled": self.enable_pre_filtering,
1382
+ "min_confidence_score": self.min_confidence_score,
1383
+ "high_confidence_threshold": self.high_confidence_threshold,
1384
+ },
1385
+ "aggregate_statistics": {
1386
+ "total_repositories_processed": total_repos,
1387
+ "total_files_analyzed": total_files_analyzed,
1388
+ "total_relationships_found": total_relationships,
1389
+ "total_lines_of_code": total_lines,
1390
+ "average_files_per_repository": round(avg_files_per_repo, 2),
1391
+ "average_relationships_per_repository": round(
1392
+ avg_relationships_per_repo, 2
1393
+ ),
1394
+ "average_lines_per_repository": round(avg_lines_per_repo, 2),
1395
+ },
1396
+ "relationship_type_distribution": aggregated_rel_types,
1397
+ "file_type_distribution": aggregated_file_types,
1398
+ "repository_details": statistics_data,
1399
+ "performance_metrics": {
1400
+ "concurrent_processing_repos": sum(
1401
+ 1
1402
+ for s in statistics_data
1403
+ if s.get("concurrent_analysis_used", False)
1404
+ ),
1405
+ "cache_efficiency": {
1406
+ "total_cache_hits": sum(
1407
+ s.get("cache_hits", 0) for s in statistics_data
1408
+ ),
1409
+ "repositories_with_caching": sum(
1410
+ 1 for s in statistics_data if s.get("cache_hits", 0) > 0
1411
+ ),
1412
+ },
1413
+ "filtering_efficiency": {
1414
+ "average_filtering_efficiency": round(
1415
+ sum(s.get("filtering_efficiency", 0) for s in statistics_data)
1416
+ / total_repos,
1417
+ 2,
1418
+ )
1419
+ if total_repos
1420
+ else 0,
1421
+ "max_filtering_efficiency": max(
1422
+ (s.get("filtering_efficiency", 0) for s in statistics_data),
1423
+ default=0,
1424
+ ),
1425
+ "min_filtering_efficiency": min(
1426
+ (s.get("filtering_efficiency", 0) for s in statistics_data),
1427
+ default=0,
1428
+ ),
1429
+ },
1430
+ },
1431
+ }
1432
+
1433
+ # Get output configuration
1434
+ output_config = self.indexer_config.get("output", {})
1435
+ json_indent = output_config.get("json_indent", 2)
1436
+ ensure_ascii = not output_config.get("ensure_ascii", False)
1437
+
1438
+ with open(stats_path, "w", encoding="utf-8") as f:
1439
+ json.dump(
1440
+ statistics_report, f, indent=json_indent, ensure_ascii=ensure_ascii
1441
+ )
1442
+
1443
+ return str(stats_path)
1444
+
1445
+ def generate_summary_report(self, output_files: Dict[str, str]) -> str:
1446
+ """Generate a summary report of all indexes created"""
1447
+ report_path = self.output_dir / "indexing_summary.json"
1448
+
1449
+ # Get output configuration from config file
1450
+ output_config = self.indexer_config.get("output", {})
1451
+ json_indent = output_config.get("json_indent", 2)
1452
+ ensure_ascii = not output_config.get("ensure_ascii", False)
1453
+
1454
+ summary_data = {
1455
+ "indexing_completion_time": datetime.now().isoformat(),
1456
+ "total_repositories_processed": len(output_files),
1457
+ "output_files": output_files,
1458
+ "target_structure": self.target_structure,
1459
+ "code_base_path": str(self.code_base_path),
1460
+ "configuration": {
1461
+ "config_file_used": self.indexer_config_path,
1462
+ "api_config_file": self.config_path,
1463
+ "pre_filtering_enabled": self.enable_pre_filtering,
1464
+ "min_confidence_score": self.min_confidence_score,
1465
+ "high_confidence_threshold": self.high_confidence_threshold,
1466
+ "max_file_size": self.max_file_size,
1467
+ "max_content_length": self.max_content_length,
1468
+ "request_delay": self.request_delay,
1469
+ "supported_extensions_count": len(self.supported_extensions),
1470
+ "skip_directories_count": len(self.skip_directories),
1471
+ },
1472
+ }
1473
+
1474
+ with open(report_path, "w", encoding="utf-8") as f:
1475
+ json.dump(summary_data, f, indent=json_indent, ensure_ascii=ensure_ascii)
1476
+
1477
+ return str(report_path)
1478
+
1479
+
1480
+ async def main():
1481
+ """Main function to run the code indexer with full configuration support"""
1482
+
1483
+ # Configuration - can be overridden by config file
1484
+ config_file = "deepcode-mcp/tools/indexer_config.yaml"
1485
+
1486
+ # You can override these parameters or let them be read from config
1487
+ code_base_path = None # Will use config file value if None
1488
+ output_dir = None # Will use config file value if None
1489
+
1490
+ # Target structure - this should be customized for your specific project
1491
+ target_structure = """
1492
+ project/
1493
+ ├── src/
1494
+ │ ├── core/
1495
+ │ │ ├── gcn.py # GCN encoder
1496
+ │ │ ├── diffusion.py # forward/reverse processes
1497
+ │ │ ├── denoiser.py # denoising MLP
1498
+ │ │ └── fusion.py # fusion combiner
1499
+ │ ├── models/ # model wrapper classes
1500
+ │ │ └── recdiff.py
1501
+ │ ├── utils/
1502
+ │ │ ├── data.py # loading & preprocessing
1503
+ │ │ ├── predictor.py # scoring functions
1504
+ │ │ ├── loss.py # loss functions
1505
+ │ │ ├── metrics.py # NDCG, Recall etc.
1506
+ │ │ └── sched.py # beta/alpha schedule utils
1507
+ │ └── configs/
1508
+ │ └── default.yaml # hyperparameters, paths
1509
+ ├── tests/
1510
+ │ ├── test_gcn.py
1511
+ │ ├── test_diffusion.py
1512
+ │ ├── test_denoiser.py
1513
+ │ ├── test_loss.py
1514
+ │ └── test_pipeline.py
1515
+ ├── docs/
1516
+ │ ├── architecture.md
1517
+ │ ├── api_reference.md
1518
+ │ └── README.md
1519
+ ├── experiments/
1520
+ │ ├── run_experiment.py
1521
+ │ └── notebooks/
1522
+ │ └── analysis.ipynb
1523
+ ├── requirements.txt
1524
+ └── setup.py
1525
+ """
1526
+
1527
+ print("🚀 Starting Code Indexer with Enhanced Configuration Support")
1528
+ print(f"📋 Configuration file: {config_file}")
1529
+
1530
+ # Create indexer with full configuration support
1531
+ try:
1532
+ indexer = CodeIndexer(
1533
+ code_base_path=code_base_path, # None = read from config
1534
+ target_structure=target_structure, # Required - project specific
1535
+ output_dir=output_dir, # None = read from config
1536
+ indexer_config_path=config_file, # Configuration file
1537
+ enable_pre_filtering=True, # Can be overridden in config
1538
+ )
1539
+
1540
+ # Display configuration information
1541
+ print(f"📁 Code base path: {indexer.code_base_path}")
1542
+ print(f"📂 Output directory: {indexer.output_dir}")
1543
+ print(f"🤖 Model provider: {indexer.model_provider}")
1544
+ print(
1545
+ f"⚡ Concurrent analysis: {'enabled' if indexer.enable_concurrent_analysis else 'disabled'}"
1546
+ )
1547
+ print(
1548
+ f"🗄️ Content caching: {'enabled' if indexer.enable_content_caching else 'disabled'}"
1549
+ )
1550
+ print(
1551
+ f"🔍 Pre-filtering: {'enabled' if indexer.enable_pre_filtering else 'disabled'}"
1552
+ )
1553
+ print(f"🐛 Debug mode: {'enabled' if indexer.verbose_output else 'disabled'}")
1554
+ print(
1555
+ f"🎭 Mock responses: {'enabled' if indexer.mock_llm_responses else 'disabled'}"
1556
+ )
1557
+
1558
+ # Validate configuration
1559
+ if not indexer.code_base_path.exists():
1560
+ raise FileNotFoundError(
1561
+ f"Code base path does not exist: {indexer.code_base_path}"
1562
+ )
1563
+
1564
+ if not target_structure:
1565
+ raise ValueError("Target structure is required for analysis")
1566
+
1567
+ print("\n🔧 Starting indexing process...")
1568
+
1569
+ # Build all indexes
1570
+ output_files = await indexer.build_all_indexes()
1571
+
1572
+ # Display results
1573
+ print("\n✅ Indexing completed successfully!")
1574
+ print(f"📊 Processed {len(output_files)} repositories")
1575
+ print("📁 Output files:")
1576
+ for repo_name, file_path in output_files.items():
1577
+ print(f" - {repo_name}: {file_path}")
1578
+
1579
+ # Display additional reports generated
1580
+ if indexer.generate_summary:
1581
+ summary_file = indexer.output_dir / indexer.summary_filename
1582
+ if summary_file.exists():
1583
+ print(f"📋 Summary report: {summary_file}")
1584
+
1585
+ if indexer.generate_statistics:
1586
+ stats_file = indexer.output_dir / indexer.stats_filename
1587
+ if stats_file.exists():
1588
+ print(f"📈 Statistics report: {stats_file}")
1589
+
1590
+ # Performance information
1591
+ if indexer.enable_content_caching and indexer.content_cache:
1592
+ print(f"🗄️ Cache performance: {len(indexer.content_cache)} items cached")
1593
+
1594
+ print("\n🎉 Code indexing process completed successfully!")
1595
+
1596
+ except FileNotFoundError as e:
1597
+ print(f"❌ File not found error: {e}")
1598
+ print("💡 Please check your configuration file paths")
1599
+ except ValueError as e:
1600
+ print(f"❌ Configuration error: {e}")
1601
+ print("💡 Please check your configuration file settings")
1602
+ except Exception as e:
1603
+ print(f"❌ Indexing failed: {e}")
1604
+ print("💡 Check the logs for more details")
1605
+
1606
+ # Print debug information if available
1607
+ try:
1608
+ indexer
1609
+ if indexer.verbose_output:
1610
+ import traceback
1611
+
1612
+ print("\n🐛 Debug information:")
1613
+ traceback.print_exc()
1614
+ except NameError:
1615
+ pass
1616
+
1617
+
1618
+ def print_usage_example():
1619
+ """Print usage examples for different scenarios"""
1620
+ print("""
1621
+ 📖 Code Indexer Usage Examples:
1622
+
1623
+ 1. Basic usage with config file:
1624
+ - Update paths in indexer_config.yaml
1625
+ - Run: python code_indexer.py
1626
+
1627
+ 2. Enable debugging:
1628
+ - Set debug.verbose_output: true in config
1629
+ - Set debug.save_raw_responses: true to save LLM responses
1630
+
1631
+ 3. Enable concurrent processing:
1632
+ - Set performance.enable_concurrent_analysis: true
1633
+ - Adjust performance.max_concurrent_files as needed
1634
+
1635
+ 4. Enable caching:
1636
+ - Set performance.enable_content_caching: true
1637
+ - Adjust performance.max_cache_size as needed
1638
+
1639
+ 5. Mock mode for testing:
1640
+ - Set debug.mock_llm_responses: true
1641
+ - No API calls will be made
1642
+
1643
+ 6. Custom output:
1644
+ - Modify output.index_filename_pattern
1645
+ - Set output.generate_statistics: true for detailed reports
1646
+
1647
+ 📋 Configuration file location: tools/indexer_config.yaml
1648
+ """)
1649
+
1650
+
1651
+ if __name__ == "__main__":
1652
+ import sys
1653
+
1654
+ if len(sys.argv) > 1 and sys.argv[1] in ["--help", "-h", "help"]:
1655
+ print_usage_example()
1656
+ else:
1657
+ asyncio.run(main())