deepcode-hku 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli/__init__.py +18 -0
- cli/cli_app.py +296 -0
- cli/cli_interface.py +744 -0
- cli/cli_launcher.py +155 -0
- cli/main_cli.py +243 -0
- cli/workflows/__init__.py +11 -0
- cli/workflows/cli_workflow_adapter.py +336 -0
- deepcode.py +219 -0
- deepcode_hku-1.0.1.dist-info/METADATA +695 -0
- deepcode_hku-1.0.1.dist-info/RECORD +44 -0
- deepcode_hku-1.0.1.dist-info/WHEEL +5 -0
- deepcode_hku-1.0.1.dist-info/entry_points.txt +2 -0
- deepcode_hku-1.0.1.dist-info/licenses/LICENSE +21 -0
- deepcode_hku-1.0.1.dist-info/top_level.txt +6 -0
- tools/__init__.py +0 -0
- tools/code_implementation_server.py +1045 -0
- tools/code_indexer.py +1657 -0
- tools/code_reference_indexer.py +486 -0
- tools/command_executor.py +324 -0
- tools/git_command.py +356 -0
- tools/pdf_converter.py +640 -0
- tools/pdf_downloader.py +1370 -0
- tools/pdf_utils.py +52 -0
- ui/__init__.py +43 -0
- ui/app.py +13 -0
- ui/components.py +1450 -0
- ui/handlers.py +773 -0
- ui/layout.py +106 -0
- ui/streamlit_app.py +38 -0
- ui/styles.py +2116 -0
- utils/__init__.py +17 -0
- utils/cli_interface.py +459 -0
- utils/dialogue_logger.py +671 -0
- utils/file_processor.py +426 -0
- utils/simple_llm_logger.py +198 -0
- workflows/__init__.py +31 -0
- workflows/agent_orchestration_engine.py +1371 -0
- workflows/agents/__init__.py +13 -0
- workflows/agents/code_implementation_agent.py +1093 -0
- workflows/agents/memory_agent_concise.py +923 -0
- workflows/agents/memory_agent_concise_index.py +935 -0
- workflows/code_implementation_workflow.py +924 -0
- workflows/code_implementation_workflow_index.py +931 -0
- workflows/codebase_index_workflow.py +726 -0
utils/file_processor.py
ADDED
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
"""
|
|
2
|
+
File processing utilities for handling paper files and related operations.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
from typing import Dict, List, Optional, Union
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class FileProcessor:
|
|
12
|
+
"""
|
|
13
|
+
A class to handle file processing operations including path extraction and file reading.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
@staticmethod
|
|
17
|
+
def extract_file_path(file_info: Union[str, Dict]) -> Optional[str]:
|
|
18
|
+
"""
|
|
19
|
+
Extract paper directory path from the input information.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
file_info: Either a JSON string or a dictionary containing file information
|
|
23
|
+
|
|
24
|
+
Returns:
|
|
25
|
+
Optional[str]: The extracted paper directory path or None if not found
|
|
26
|
+
"""
|
|
27
|
+
try:
|
|
28
|
+
# Handle direct file path input
|
|
29
|
+
if isinstance(file_info, str):
|
|
30
|
+
# Check if it's a file path (existing or not)
|
|
31
|
+
if file_info.endswith(
|
|
32
|
+
(".md", ".pdf", ".txt", ".docx", ".doc", ".html", ".htm")
|
|
33
|
+
):
|
|
34
|
+
# It's a file path, return the directory
|
|
35
|
+
return os.path.dirname(os.path.abspath(file_info))
|
|
36
|
+
elif os.path.exists(file_info):
|
|
37
|
+
if os.path.isfile(file_info):
|
|
38
|
+
return os.path.dirname(os.path.abspath(file_info))
|
|
39
|
+
elif os.path.isdir(file_info):
|
|
40
|
+
return os.path.abspath(file_info)
|
|
41
|
+
|
|
42
|
+
# Try to parse as JSON
|
|
43
|
+
try:
|
|
44
|
+
info_dict = json.loads(file_info)
|
|
45
|
+
except json.JSONDecodeError:
|
|
46
|
+
# 尝试从文本中提取JSON
|
|
47
|
+
info_dict = FileProcessor.extract_json_from_text(file_info)
|
|
48
|
+
if not info_dict:
|
|
49
|
+
# If not JSON and doesn't look like a file path, raise error
|
|
50
|
+
raise ValueError(
|
|
51
|
+
f"Input is neither a valid file path nor JSON: {file_info}"
|
|
52
|
+
)
|
|
53
|
+
else:
|
|
54
|
+
info_dict = file_info
|
|
55
|
+
|
|
56
|
+
# Extract paper path from dictionary
|
|
57
|
+
paper_path = info_dict.get("paper_path")
|
|
58
|
+
if not paper_path:
|
|
59
|
+
raise ValueError("No paper_path found in input dictionary")
|
|
60
|
+
|
|
61
|
+
# Get the directory path instead of the file path
|
|
62
|
+
paper_dir = os.path.dirname(paper_path)
|
|
63
|
+
|
|
64
|
+
# Convert to absolute path if relative
|
|
65
|
+
if not os.path.isabs(paper_dir):
|
|
66
|
+
paper_dir = os.path.abspath(paper_dir)
|
|
67
|
+
|
|
68
|
+
return paper_dir
|
|
69
|
+
|
|
70
|
+
except (AttributeError, TypeError) as e:
|
|
71
|
+
raise ValueError(f"Invalid input format: {str(e)}")
|
|
72
|
+
|
|
73
|
+
@staticmethod
|
|
74
|
+
def find_markdown_file(directory: str) -> Optional[str]:
|
|
75
|
+
"""
|
|
76
|
+
Find the first markdown file in the given directory.
|
|
77
|
+
|
|
78
|
+
Args:
|
|
79
|
+
directory: Directory path to search
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
Optional[str]: Path to the markdown file or None if not found
|
|
83
|
+
"""
|
|
84
|
+
if not os.path.isdir(directory):
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
for file in os.listdir(directory):
|
|
88
|
+
if file.endswith(".md"):
|
|
89
|
+
return os.path.join(directory, file)
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
@staticmethod
|
|
93
|
+
def parse_markdown_sections(content: str) -> List[Dict[str, Union[str, int, List]]]:
|
|
94
|
+
"""
|
|
95
|
+
Parse markdown content and organize it by sections based on headers.
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
content: The markdown content to parse
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
List[Dict]: A list of sections, each containing:
|
|
102
|
+
- level: The header level (1-6)
|
|
103
|
+
- title: The section title
|
|
104
|
+
- content: The section content
|
|
105
|
+
- subsections: List of subsections
|
|
106
|
+
"""
|
|
107
|
+
# Split content into lines
|
|
108
|
+
lines = content.split("\n")
|
|
109
|
+
sections = []
|
|
110
|
+
current_section = None
|
|
111
|
+
current_content = []
|
|
112
|
+
|
|
113
|
+
for line in lines:
|
|
114
|
+
# Check if line is a header
|
|
115
|
+
header_match = re.match(r"^(#{1,6})\s+(.+)$", line)
|
|
116
|
+
|
|
117
|
+
if header_match:
|
|
118
|
+
# If we were building a section, save its content
|
|
119
|
+
if current_section is not None:
|
|
120
|
+
current_section["content"] = "\n".join(current_content).strip()
|
|
121
|
+
sections.append(current_section)
|
|
122
|
+
|
|
123
|
+
# Start a new section
|
|
124
|
+
level = len(header_match.group(1))
|
|
125
|
+
title = header_match.group(2).strip()
|
|
126
|
+
current_section = {
|
|
127
|
+
"level": level,
|
|
128
|
+
"title": title,
|
|
129
|
+
"content": "",
|
|
130
|
+
"subsections": [],
|
|
131
|
+
}
|
|
132
|
+
current_content = []
|
|
133
|
+
elif current_section is not None:
|
|
134
|
+
current_content.append(line)
|
|
135
|
+
|
|
136
|
+
# Don't forget to save the last section
|
|
137
|
+
if current_section is not None:
|
|
138
|
+
current_section["content"] = "\n".join(current_content).strip()
|
|
139
|
+
sections.append(current_section)
|
|
140
|
+
|
|
141
|
+
return FileProcessor._organize_sections(sections)
|
|
142
|
+
|
|
143
|
+
@staticmethod
|
|
144
|
+
def _organize_sections(sections: List[Dict]) -> List[Dict]:
|
|
145
|
+
"""
|
|
146
|
+
Organize sections into a hierarchical structure based on their levels.
|
|
147
|
+
|
|
148
|
+
Args:
|
|
149
|
+
sections: List of sections with their levels
|
|
150
|
+
|
|
151
|
+
Returns:
|
|
152
|
+
List[Dict]: Organized hierarchical structure of sections
|
|
153
|
+
"""
|
|
154
|
+
result = []
|
|
155
|
+
section_stack = []
|
|
156
|
+
|
|
157
|
+
for section in sections:
|
|
158
|
+
while section_stack and section_stack[-1]["level"] >= section["level"]:
|
|
159
|
+
section_stack.pop()
|
|
160
|
+
|
|
161
|
+
if section_stack:
|
|
162
|
+
section_stack[-1]["subsections"].append(section)
|
|
163
|
+
else:
|
|
164
|
+
result.append(section)
|
|
165
|
+
|
|
166
|
+
section_stack.append(section)
|
|
167
|
+
|
|
168
|
+
return result
|
|
169
|
+
|
|
170
|
+
@staticmethod
|
|
171
|
+
async def read_file_content(file_path: str) -> str:
|
|
172
|
+
"""
|
|
173
|
+
Read the content of a file asynchronously.
|
|
174
|
+
|
|
175
|
+
Args:
|
|
176
|
+
file_path: Path to the file to read
|
|
177
|
+
|
|
178
|
+
Returns:
|
|
179
|
+
str: The content of the file
|
|
180
|
+
|
|
181
|
+
Raises:
|
|
182
|
+
FileNotFoundError: If the file doesn't exist
|
|
183
|
+
IOError: If there's an error reading the file
|
|
184
|
+
"""
|
|
185
|
+
try:
|
|
186
|
+
# Ensure the file exists
|
|
187
|
+
if not os.path.exists(file_path):
|
|
188
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
189
|
+
|
|
190
|
+
# Read file content
|
|
191
|
+
# Note: Using async with would be better for large files
|
|
192
|
+
# but for simplicity and compatibility, using regular file reading
|
|
193
|
+
with open(file_path, "r", encoding="utf-8") as f:
|
|
194
|
+
content = f.read()
|
|
195
|
+
|
|
196
|
+
return content
|
|
197
|
+
|
|
198
|
+
except Exception as e:
|
|
199
|
+
raise IOError(f"Error reading file {file_path}: {str(e)}")
|
|
200
|
+
|
|
201
|
+
@staticmethod
|
|
202
|
+
def format_section_content(section: Dict) -> str:
|
|
203
|
+
"""
|
|
204
|
+
Format a section's content with standardized spacing and structure.
|
|
205
|
+
|
|
206
|
+
Args:
|
|
207
|
+
section: Dictionary containing section information
|
|
208
|
+
|
|
209
|
+
Returns:
|
|
210
|
+
str: Formatted section content
|
|
211
|
+
"""
|
|
212
|
+
# Start with section title
|
|
213
|
+
formatted = f"\n{'#' * section['level']} {section['title']}\n"
|
|
214
|
+
|
|
215
|
+
# Add section content if it exists
|
|
216
|
+
if section["content"]:
|
|
217
|
+
formatted += f"\n{section['content'].strip()}\n"
|
|
218
|
+
|
|
219
|
+
# Process subsections
|
|
220
|
+
if section["subsections"]:
|
|
221
|
+
# Add a separator before subsections if there's content
|
|
222
|
+
if section["content"]:
|
|
223
|
+
formatted += "\n---\n"
|
|
224
|
+
|
|
225
|
+
# Process each subsection
|
|
226
|
+
for subsection in section["subsections"]:
|
|
227
|
+
formatted += FileProcessor.format_section_content(subsection)
|
|
228
|
+
|
|
229
|
+
# Add section separator
|
|
230
|
+
formatted += "\n" + "=" * 80 + "\n"
|
|
231
|
+
|
|
232
|
+
return formatted
|
|
233
|
+
|
|
234
|
+
@staticmethod
|
|
235
|
+
def standardize_output(sections: List[Dict]) -> str:
|
|
236
|
+
"""
|
|
237
|
+
Convert structured sections into a standardized string format.
|
|
238
|
+
|
|
239
|
+
Args:
|
|
240
|
+
sections: List of section dictionaries
|
|
241
|
+
|
|
242
|
+
Returns:
|
|
243
|
+
str: Standardized string output
|
|
244
|
+
"""
|
|
245
|
+
output = []
|
|
246
|
+
|
|
247
|
+
# Process each top-level section
|
|
248
|
+
for section in sections:
|
|
249
|
+
output.append(FileProcessor.format_section_content(section))
|
|
250
|
+
|
|
251
|
+
# Join all sections with clear separation
|
|
252
|
+
return "\n".join(output)
|
|
253
|
+
|
|
254
|
+
@classmethod
|
|
255
|
+
async def process_file_input(
|
|
256
|
+
cls, file_input: Union[str, Dict], base_dir: str = None
|
|
257
|
+
) -> Dict:
|
|
258
|
+
"""
|
|
259
|
+
Process file input information and return the structured content.
|
|
260
|
+
|
|
261
|
+
Args:
|
|
262
|
+
file_input: File input information (JSON string, dict, or direct file path)
|
|
263
|
+
base_dir: Optional base directory to use for creating paper directories (for sync support)
|
|
264
|
+
|
|
265
|
+
Returns:
|
|
266
|
+
Dict: The structured content with sections and standardized text
|
|
267
|
+
"""
|
|
268
|
+
try:
|
|
269
|
+
# 首先尝试从字符串中提取markdown文件路径
|
|
270
|
+
if isinstance(file_input, str):
|
|
271
|
+
import re
|
|
272
|
+
|
|
273
|
+
file_path_match = re.search(r"`([^`]+\.md)`", file_input)
|
|
274
|
+
if file_path_match:
|
|
275
|
+
paper_path = file_path_match.group(1)
|
|
276
|
+
file_input = {"paper_path": paper_path}
|
|
277
|
+
|
|
278
|
+
# Extract paper directory path
|
|
279
|
+
paper_dir = cls.extract_file_path(file_input)
|
|
280
|
+
|
|
281
|
+
# If base_dir is provided, adjust paper_dir to be relative to base_dir
|
|
282
|
+
if base_dir and paper_dir:
|
|
283
|
+
# If paper_dir is using default location, move it to base_dir
|
|
284
|
+
if paper_dir.endswith(("deepcode_lab", "agent_folders")):
|
|
285
|
+
paper_dir = base_dir
|
|
286
|
+
else:
|
|
287
|
+
# Extract the relative part and combine with base_dir
|
|
288
|
+
paper_name = os.path.basename(paper_dir)
|
|
289
|
+
# 保持原始目录名不变,不做任何替换
|
|
290
|
+
paper_dir = os.path.join(base_dir, "papers", paper_name)
|
|
291
|
+
|
|
292
|
+
# Ensure the directory exists
|
|
293
|
+
os.makedirs(paper_dir, exist_ok=True)
|
|
294
|
+
|
|
295
|
+
if not paper_dir:
|
|
296
|
+
raise ValueError("Could not determine paper directory path")
|
|
297
|
+
|
|
298
|
+
# Get the actual file path
|
|
299
|
+
file_path = None
|
|
300
|
+
if isinstance(file_input, str):
|
|
301
|
+
# 尝试解析为JSON(处理下载结果)
|
|
302
|
+
try:
|
|
303
|
+
parsed_json = json.loads(file_input)
|
|
304
|
+
if isinstance(parsed_json, dict) and "paper_path" in parsed_json:
|
|
305
|
+
file_path = parsed_json.get("paper_path")
|
|
306
|
+
# 如果文件不存在,尝试查找markdown文件
|
|
307
|
+
if file_path and not os.path.exists(file_path):
|
|
308
|
+
paper_dir = os.path.dirname(file_path)
|
|
309
|
+
if os.path.isdir(paper_dir):
|
|
310
|
+
file_path = cls.find_markdown_file(paper_dir)
|
|
311
|
+
if not file_path:
|
|
312
|
+
raise ValueError(
|
|
313
|
+
f"No markdown file found in directory: {paper_dir}"
|
|
314
|
+
)
|
|
315
|
+
else:
|
|
316
|
+
raise ValueError("Invalid JSON format: missing paper_path")
|
|
317
|
+
except json.JSONDecodeError:
|
|
318
|
+
# 尝试从文本中提取JSON(处理包含额外文本的下载结果)
|
|
319
|
+
extracted_json = cls.extract_json_from_text(file_input)
|
|
320
|
+
if extracted_json and "paper_path" in extracted_json:
|
|
321
|
+
file_path = extracted_json.get("paper_path")
|
|
322
|
+
# 如果文件不存在,尝试查找markdown文件
|
|
323
|
+
if file_path and not os.path.exists(file_path):
|
|
324
|
+
paper_dir = os.path.dirname(file_path)
|
|
325
|
+
if os.path.isdir(paper_dir):
|
|
326
|
+
file_path = cls.find_markdown_file(paper_dir)
|
|
327
|
+
if not file_path:
|
|
328
|
+
raise ValueError(
|
|
329
|
+
f"No markdown file found in directory: {paper_dir}"
|
|
330
|
+
)
|
|
331
|
+
else:
|
|
332
|
+
# 不是JSON,按文件路径处理
|
|
333
|
+
# Check if it's a file path (existing or not)
|
|
334
|
+
if file_input.endswith(
|
|
335
|
+
(".md", ".pdf", ".txt", ".docx", ".doc", ".html", ".htm")
|
|
336
|
+
):
|
|
337
|
+
if os.path.exists(file_input):
|
|
338
|
+
file_path = file_input
|
|
339
|
+
else:
|
|
340
|
+
# File doesn't exist, try to find markdown in the directory
|
|
341
|
+
file_path = cls.find_markdown_file(paper_dir)
|
|
342
|
+
if not file_path:
|
|
343
|
+
raise ValueError(
|
|
344
|
+
f"No markdown file found in directory: {paper_dir}"
|
|
345
|
+
)
|
|
346
|
+
elif os.path.exists(file_input):
|
|
347
|
+
if os.path.isfile(file_input):
|
|
348
|
+
file_path = file_input
|
|
349
|
+
elif os.path.isdir(file_input):
|
|
350
|
+
# If it's a directory, find the markdown file
|
|
351
|
+
file_path = cls.find_markdown_file(file_input)
|
|
352
|
+
if not file_path:
|
|
353
|
+
raise ValueError(
|
|
354
|
+
f"No markdown file found in directory: {file_input}"
|
|
355
|
+
)
|
|
356
|
+
else:
|
|
357
|
+
raise ValueError(f"Invalid input: {file_input}")
|
|
358
|
+
else:
|
|
359
|
+
# Dictionary input
|
|
360
|
+
file_path = file_input.get("paper_path")
|
|
361
|
+
# If the file doesn't exist, try to find markdown in the directory
|
|
362
|
+
if file_path and not os.path.exists(file_path):
|
|
363
|
+
paper_dir = os.path.dirname(file_path)
|
|
364
|
+
if os.path.isdir(paper_dir):
|
|
365
|
+
file_path = cls.find_markdown_file(paper_dir)
|
|
366
|
+
if not file_path:
|
|
367
|
+
raise ValueError(
|
|
368
|
+
f"No markdown file found in directory: {paper_dir}"
|
|
369
|
+
)
|
|
370
|
+
|
|
371
|
+
if not file_path:
|
|
372
|
+
raise ValueError("No valid file path found")
|
|
373
|
+
|
|
374
|
+
# Read file content
|
|
375
|
+
content = await cls.read_file_content(file_path)
|
|
376
|
+
|
|
377
|
+
# Parse and structure the content
|
|
378
|
+
structured_content = cls.parse_markdown_sections(content)
|
|
379
|
+
|
|
380
|
+
# Generate standardized text output
|
|
381
|
+
standardized_text = cls.standardize_output(structured_content)
|
|
382
|
+
|
|
383
|
+
return {
|
|
384
|
+
"paper_dir": paper_dir,
|
|
385
|
+
"file_path": file_path,
|
|
386
|
+
"sections": structured_content,
|
|
387
|
+
"standardized_text": standardized_text,
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
except Exception as e:
|
|
391
|
+
raise ValueError(f"Error processing file input: {str(e)}")
|
|
392
|
+
|
|
393
|
+
@staticmethod
|
|
394
|
+
def extract_json_from_text(text: str) -> Optional[Dict]:
|
|
395
|
+
"""
|
|
396
|
+
Extract JSON from text that may contain markdown code blocks or other content.
|
|
397
|
+
|
|
398
|
+
Args:
|
|
399
|
+
text: Text that may contain JSON
|
|
400
|
+
|
|
401
|
+
Returns:
|
|
402
|
+
Optional[Dict]: Extracted JSON as dictionary or None if not found
|
|
403
|
+
"""
|
|
404
|
+
import re
|
|
405
|
+
|
|
406
|
+
# Try to find JSON in markdown code blocks
|
|
407
|
+
json_pattern = r"```json\s*(\{.*?\})\s*```"
|
|
408
|
+
match = re.search(json_pattern, text, re.DOTALL)
|
|
409
|
+
if match:
|
|
410
|
+
try:
|
|
411
|
+
return json.loads(match.group(1))
|
|
412
|
+
except json.JSONDecodeError:
|
|
413
|
+
pass
|
|
414
|
+
|
|
415
|
+
# Try to find standalone JSON
|
|
416
|
+
json_pattern = r"(\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\})"
|
|
417
|
+
matches = re.findall(json_pattern, text, re.DOTALL)
|
|
418
|
+
for match in matches:
|
|
419
|
+
try:
|
|
420
|
+
parsed = json.loads(match)
|
|
421
|
+
if isinstance(parsed, dict) and "paper_path" in parsed:
|
|
422
|
+
return parsed
|
|
423
|
+
except json.JSONDecodeError:
|
|
424
|
+
continue
|
|
425
|
+
|
|
426
|
+
return None
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""
|
|
4
|
+
超简化LLM响应日志记录器
|
|
5
|
+
专注于记录LLM回复的核心内容,配置简单易用
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
import yaml
|
|
11
|
+
from datetime import datetime
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Dict, Any
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class SimpleLLMLogger:
|
|
17
|
+
"""超简化的LLM响应日志记录器"""
|
|
18
|
+
|
|
19
|
+
def __init__(self, config_path: str = "mcp_agent.config.yaml"):
|
|
20
|
+
"""
|
|
21
|
+
初始化日志记录器
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
config_path: 配置文件路径
|
|
25
|
+
"""
|
|
26
|
+
self.config = self._load_config(config_path)
|
|
27
|
+
self.llm_config = self.config.get("llm_logger", {})
|
|
28
|
+
|
|
29
|
+
# 如果禁用则直接返回
|
|
30
|
+
if not self.llm_config.get("enabled", True):
|
|
31
|
+
self.enabled = False
|
|
32
|
+
return
|
|
33
|
+
|
|
34
|
+
self.enabled = True
|
|
35
|
+
self._setup_logger()
|
|
36
|
+
|
|
37
|
+
def _load_config(self, config_path: str) -> Dict[str, Any]:
|
|
38
|
+
"""加载配置文件"""
|
|
39
|
+
try:
|
|
40
|
+
with open(config_path, "r", encoding="utf-8") as f:
|
|
41
|
+
return yaml.safe_load(f)
|
|
42
|
+
except Exception as e:
|
|
43
|
+
print(f"⚠️ 配置文件加载失败: {e},使用默认配置")
|
|
44
|
+
return self._get_default_config()
|
|
45
|
+
|
|
46
|
+
def _get_default_config(self) -> Dict[str, Any]:
|
|
47
|
+
"""获取默认配置"""
|
|
48
|
+
return {
|
|
49
|
+
"llm_logger": {
|
|
50
|
+
"enabled": True,
|
|
51
|
+
"output_format": "json",
|
|
52
|
+
"log_level": "basic",
|
|
53
|
+
"log_directory": "logs/llm_responses",
|
|
54
|
+
"filename_pattern": "llm_responses_{timestamp}.jsonl",
|
|
55
|
+
"include_models": ["claude-sonnet-4", "gpt-4", "o3-mini"],
|
|
56
|
+
"min_response_length": 50,
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
def _setup_logger(self):
|
|
61
|
+
"""设置日志记录器"""
|
|
62
|
+
log_dir = self.llm_config.get("log_directory", "logs/llm_responses")
|
|
63
|
+
|
|
64
|
+
# 创建日志目录
|
|
65
|
+
Path(log_dir).mkdir(parents=True, exist_ok=True)
|
|
66
|
+
|
|
67
|
+
# 生成日志文件名
|
|
68
|
+
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
69
|
+
filename_pattern = self.llm_config.get(
|
|
70
|
+
"filename_pattern", "llm_responses_{timestamp}.jsonl"
|
|
71
|
+
)
|
|
72
|
+
self.log_file = os.path.join(
|
|
73
|
+
log_dir, filename_pattern.format(timestamp=timestamp)
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
print(f"📝 LLM响应日志: {self.log_file}")
|
|
77
|
+
|
|
78
|
+
def log_response(self, content: str, model: str = "", agent: str = "", **kwargs):
|
|
79
|
+
"""
|
|
80
|
+
记录LLM响应 - 简化版本
|
|
81
|
+
|
|
82
|
+
Args:
|
|
83
|
+
content: LLM响应内容
|
|
84
|
+
model: 模型名称
|
|
85
|
+
agent: Agent名称
|
|
86
|
+
**kwargs: 其他可选信息
|
|
87
|
+
"""
|
|
88
|
+
if not self.enabled:
|
|
89
|
+
return
|
|
90
|
+
|
|
91
|
+
# 检查是否应该记录
|
|
92
|
+
if not self._should_log(content, model):
|
|
93
|
+
return
|
|
94
|
+
|
|
95
|
+
# 构建日志记录
|
|
96
|
+
log_entry = self._build_entry(content, model, agent, kwargs)
|
|
97
|
+
|
|
98
|
+
# 写入日志
|
|
99
|
+
self._write_log(log_entry)
|
|
100
|
+
|
|
101
|
+
# 控制台显示
|
|
102
|
+
self._console_log(content, model, agent)
|
|
103
|
+
|
|
104
|
+
def _should_log(self, content: str, model: str) -> bool:
|
|
105
|
+
"""检查是否应该记录"""
|
|
106
|
+
# 检查长度
|
|
107
|
+
min_length = self.llm_config.get("min_response_length", 50)
|
|
108
|
+
if len(content) < min_length:
|
|
109
|
+
return False
|
|
110
|
+
|
|
111
|
+
# 检查模型
|
|
112
|
+
include_models = self.llm_config.get("include_models", [])
|
|
113
|
+
if include_models and not any(m in model for m in include_models):
|
|
114
|
+
return False
|
|
115
|
+
|
|
116
|
+
return True
|
|
117
|
+
|
|
118
|
+
def _build_entry(self, content: str, model: str, agent: str, extra: Dict) -> Dict:
|
|
119
|
+
"""构建日志条目"""
|
|
120
|
+
log_level = self.llm_config.get("log_level", "basic")
|
|
121
|
+
|
|
122
|
+
if log_level == "basic":
|
|
123
|
+
# 基础级别:只记录核心内容
|
|
124
|
+
return {
|
|
125
|
+
"timestamp": datetime.now().isoformat(),
|
|
126
|
+
"content": content,
|
|
127
|
+
"model": model,
|
|
128
|
+
}
|
|
129
|
+
else:
|
|
130
|
+
# 详细级别:包含更多信息
|
|
131
|
+
entry = {
|
|
132
|
+
"timestamp": datetime.now().isoformat(),
|
|
133
|
+
"content": content,
|
|
134
|
+
"model": model,
|
|
135
|
+
"agent": agent,
|
|
136
|
+
}
|
|
137
|
+
# 添加额外信息
|
|
138
|
+
if "token_usage" in extra:
|
|
139
|
+
entry["tokens"] = extra["token_usage"]
|
|
140
|
+
if "session_id" in extra:
|
|
141
|
+
entry["session"] = extra["session_id"]
|
|
142
|
+
return entry
|
|
143
|
+
|
|
144
|
+
def _write_log(self, entry: Dict):
|
|
145
|
+
"""写入日志文件"""
|
|
146
|
+
output_format = self.llm_config.get("output_format", "json")
|
|
147
|
+
|
|
148
|
+
try:
|
|
149
|
+
with open(self.log_file, "a", encoding="utf-8") as f:
|
|
150
|
+
if output_format == "json":
|
|
151
|
+
f.write(json.dumps(entry, ensure_ascii=False) + "\n")
|
|
152
|
+
elif output_format == "text":
|
|
153
|
+
timestamp = entry.get("timestamp", "")
|
|
154
|
+
model = entry.get("model", "")
|
|
155
|
+
content = entry.get("content", "")
|
|
156
|
+
f.write(f"[{timestamp}] {model}: {content}\n\n")
|
|
157
|
+
elif output_format == "markdown":
|
|
158
|
+
timestamp = entry.get("timestamp", "")
|
|
159
|
+
model = entry.get("model", "")
|
|
160
|
+
content = entry.get("content", "")
|
|
161
|
+
f.write(f"**{timestamp}** | {model}\n\n{content}\n\n---\n\n")
|
|
162
|
+
except Exception as e:
|
|
163
|
+
print(f"⚠️ 写入日志失败: {e}")
|
|
164
|
+
|
|
165
|
+
def _console_log(self, content: str, model: str, agent: str):
|
|
166
|
+
"""控制台简要显示"""
|
|
167
|
+
preview = content[:80] + "..." if len(content) > 80 else content
|
|
168
|
+
print(f"🤖 {model} ({agent}): {preview}")
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# 全局实例
|
|
172
|
+
_global_logger = None
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def get_llm_logger() -> SimpleLLMLogger:
|
|
176
|
+
"""获取全局LLM日志记录器实例"""
|
|
177
|
+
global _global_logger
|
|
178
|
+
if _global_logger is None:
|
|
179
|
+
_global_logger = SimpleLLMLogger()
|
|
180
|
+
return _global_logger
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def log_llm_response(content: str, model: str = "", agent: str = "", **kwargs):
|
|
184
|
+
"""便捷函数:记录LLM响应"""
|
|
185
|
+
logger = get_llm_logger()
|
|
186
|
+
logger.log_response(content, model, agent, **kwargs)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
# 示例使用
|
|
190
|
+
if __name__ == "__main__":
|
|
191
|
+
# 测试日志记录
|
|
192
|
+
log_llm_response(
|
|
193
|
+
content="这是一个测试的LLM响应内容,用于验证简化日志记录器的功能是否正常工作。",
|
|
194
|
+
model="claude-sonnet-4-20250514",
|
|
195
|
+
agent="TestAgent",
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
print("✅ 简化LLM日志测试完成")
|
workflows/__init__.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Intelligent Agent Orchestration Workflows for Research-to-Code Automation.
|
|
3
|
+
|
|
4
|
+
This package provides advanced AI-driven workflow orchestration capabilities
|
|
5
|
+
for automated research analysis and code implementation synthesis.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from .agent_orchestration_engine import (
|
|
9
|
+
run_research_analyzer,
|
|
10
|
+
run_resource_processor,
|
|
11
|
+
run_code_analyzer,
|
|
12
|
+
github_repo_download,
|
|
13
|
+
paper_reference_analyzer,
|
|
14
|
+
execute_multi_agent_research_pipeline,
|
|
15
|
+
paper_code_preparation, # Deprecated, for backward compatibility
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
from .code_implementation_workflow import CodeImplementationWorkflow
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
# Initial workflows
|
|
22
|
+
"run_research_analyzer",
|
|
23
|
+
"run_resource_processor",
|
|
24
|
+
"run_code_analyzer",
|
|
25
|
+
"github_repo_download",
|
|
26
|
+
"paper_reference_analyzer",
|
|
27
|
+
"execute_multi_agent_research_pipeline", # Main multi-agent pipeline function
|
|
28
|
+
"paper_code_preparation", # Deprecated, for backward compatibility
|
|
29
|
+
# Code implementation workflows
|
|
30
|
+
"CodeImplementationWorkflow",
|
|
31
|
+
]
|