deepcode-hku 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli/__init__.py +18 -0
- cli/cli_app.py +296 -0
- cli/cli_interface.py +744 -0
- cli/cli_launcher.py +155 -0
- cli/main_cli.py +243 -0
- cli/workflows/__init__.py +11 -0
- cli/workflows/cli_workflow_adapter.py +336 -0
- deepcode.py +219 -0
- deepcode_hku-1.0.1.dist-info/METADATA +695 -0
- deepcode_hku-1.0.1.dist-info/RECORD +44 -0
- deepcode_hku-1.0.1.dist-info/WHEEL +5 -0
- deepcode_hku-1.0.1.dist-info/entry_points.txt +2 -0
- deepcode_hku-1.0.1.dist-info/licenses/LICENSE +21 -0
- deepcode_hku-1.0.1.dist-info/top_level.txt +6 -0
- tools/__init__.py +0 -0
- tools/code_implementation_server.py +1045 -0
- tools/code_indexer.py +1657 -0
- tools/code_reference_indexer.py +486 -0
- tools/command_executor.py +324 -0
- tools/git_command.py +356 -0
- tools/pdf_converter.py +640 -0
- tools/pdf_downloader.py +1370 -0
- tools/pdf_utils.py +52 -0
- ui/__init__.py +43 -0
- ui/app.py +13 -0
- ui/components.py +1450 -0
- ui/handlers.py +773 -0
- ui/layout.py +106 -0
- ui/streamlit_app.py +38 -0
- ui/styles.py +2116 -0
- utils/__init__.py +17 -0
- utils/cli_interface.py +459 -0
- utils/dialogue_logger.py +671 -0
- utils/file_processor.py +426 -0
- utils/simple_llm_logger.py +198 -0
- workflows/__init__.py +31 -0
- workflows/agent_orchestration_engine.py +1371 -0
- workflows/agents/__init__.py +13 -0
- workflows/agents/code_implementation_agent.py +1093 -0
- workflows/agents/memory_agent_concise.py +923 -0
- workflows/agents/memory_agent_concise_index.py +935 -0
- workflows/code_implementation_workflow.py +924 -0
- workflows/code_implementation_workflow_index.py +931 -0
- workflows/codebase_index_workflow.py +726 -0
utils/dialogue_logger.py
ADDED
|
@@ -0,0 +1,671 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""
|
|
4
|
+
Comprehensive Dialogue Logger for Code Implementation Workflow
|
|
5
|
+
Logs complete conversation rounds with detailed formatting and paper-specific organization
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
from datetime import datetime
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Dict, Any, List
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class DialogueLogger:
|
|
16
|
+
"""
|
|
17
|
+
Comprehensive dialogue logger for code implementation workflow
|
|
18
|
+
Captures complete conversation rounds with proper formatting and organization
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(self, paper_id: str, base_path: str = None):
|
|
22
|
+
"""
|
|
23
|
+
Initialize dialogue logger for a specific paper
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
paper_id: Paper identifier (e.g., "1", "2", etc.)
|
|
27
|
+
base_path: Base path for logs (defaults to agent_folders structure)
|
|
28
|
+
"""
|
|
29
|
+
self.paper_id = paper_id
|
|
30
|
+
self.base_path = (
|
|
31
|
+
base_path
|
|
32
|
+
or "/data2/bjdwhzzh/project-hku/Code-Agent2.0/Code-Agent/deepcode-mcp/agent_folders"
|
|
33
|
+
)
|
|
34
|
+
self.log_directory = os.path.join(
|
|
35
|
+
self.base_path, "papers", str(paper_id), "logs"
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
# Create log directory if it doesn't exist
|
|
39
|
+
Path(self.log_directory).mkdir(parents=True, exist_ok=True)
|
|
40
|
+
|
|
41
|
+
# Session tracking (initialize before log file creation)
|
|
42
|
+
self.round_counter = 0
|
|
43
|
+
self.session_start_time = datetime.now()
|
|
44
|
+
self.current_round_data = {}
|
|
45
|
+
|
|
46
|
+
# Generate log filename with timestamp
|
|
47
|
+
timestamp = self.session_start_time.strftime("%Y%m%d_%H%M%S")
|
|
48
|
+
self.log_filename = f"dialogue_log_{timestamp}.md"
|
|
49
|
+
self.log_filepath = os.path.join(self.log_directory, self.log_filename)
|
|
50
|
+
|
|
51
|
+
# Initialize log file with header
|
|
52
|
+
self._initialize_log_file()
|
|
53
|
+
|
|
54
|
+
print(f"📝 Dialogue Logger initialized for Paper {paper_id}")
|
|
55
|
+
print(f"📁 Log file: {self.log_filepath}")
|
|
56
|
+
|
|
57
|
+
def _initialize_log_file(self):
|
|
58
|
+
"""Initialize the log file with header information"""
|
|
59
|
+
header = f"""# Code Implementation Dialogue Log
|
|
60
|
+
|
|
61
|
+
**Paper ID:** {self.paper_id}
|
|
62
|
+
**Session Start:** {self.session_start_time.strftime('%Y-%m-%d %H:%M:%S')}
|
|
63
|
+
**Log File:** {self.log_filename}
|
|
64
|
+
|
|
65
|
+
---
|
|
66
|
+
|
|
67
|
+
## Session Overview
|
|
68
|
+
|
|
69
|
+
This log contains the complete conversation rounds between the user and assistant during the code implementation workflow. Each round includes:
|
|
70
|
+
|
|
71
|
+
- System prompts and user messages
|
|
72
|
+
- Assistant responses with tool calls
|
|
73
|
+
- Tool execution results
|
|
74
|
+
- Implementation progress markers
|
|
75
|
+
|
|
76
|
+
---
|
|
77
|
+
|
|
78
|
+
"""
|
|
79
|
+
try:
|
|
80
|
+
with open(self.log_filepath, "w", encoding="utf-8") as f:
|
|
81
|
+
f.write(header)
|
|
82
|
+
except Exception as e:
|
|
83
|
+
print(f"⚠️ Failed to initialize log file: {e}")
|
|
84
|
+
|
|
85
|
+
def start_new_round(
|
|
86
|
+
self, round_type: str = "implementation", context: Dict[str, Any] = None
|
|
87
|
+
):
|
|
88
|
+
"""
|
|
89
|
+
Start a new dialogue round
|
|
90
|
+
|
|
91
|
+
Args:
|
|
92
|
+
round_type: Type of round (implementation, summary, error_handling, etc.)
|
|
93
|
+
context: Additional context information (may include 'iteration' to sync with workflow)
|
|
94
|
+
"""
|
|
95
|
+
# Use iteration from context if provided, otherwise increment round_counter
|
|
96
|
+
if context and "iteration" in context:
|
|
97
|
+
self.round_counter = context["iteration"]
|
|
98
|
+
else:
|
|
99
|
+
self.round_counter += 1
|
|
100
|
+
|
|
101
|
+
self.current_round_data = {
|
|
102
|
+
"round_number": self.round_counter,
|
|
103
|
+
"round_type": round_type,
|
|
104
|
+
"start_time": datetime.now(),
|
|
105
|
+
"context": context or {},
|
|
106
|
+
"messages": [],
|
|
107
|
+
"tool_calls": [],
|
|
108
|
+
"results": [],
|
|
109
|
+
"metadata": {},
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
print(f"🔄 Starting Round {self.round_counter}: {round_type}")
|
|
113
|
+
|
|
114
|
+
def log_system_prompt(self, prompt: str, prompt_type: str = "system"):
|
|
115
|
+
"""
|
|
116
|
+
Log system prompt or instructions
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
prompt: System prompt content
|
|
120
|
+
prompt_type: Type of prompt (system, instruction, etc.)
|
|
121
|
+
"""
|
|
122
|
+
if not self.current_round_data:
|
|
123
|
+
self.start_new_round("system_setup")
|
|
124
|
+
|
|
125
|
+
self.current_round_data["messages"].append(
|
|
126
|
+
{
|
|
127
|
+
"role": "system",
|
|
128
|
+
"type": prompt_type,
|
|
129
|
+
"content": prompt,
|
|
130
|
+
"timestamp": datetime.now().isoformat(),
|
|
131
|
+
}
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
def log_user_message(self, message: str, message_type: str = "user_input"):
|
|
135
|
+
"""
|
|
136
|
+
Log user message
|
|
137
|
+
|
|
138
|
+
Args:
|
|
139
|
+
message: User message content
|
|
140
|
+
message_type: Type of message (user_input, feedback, guidance, etc.)
|
|
141
|
+
"""
|
|
142
|
+
if not self.current_round_data:
|
|
143
|
+
self.start_new_round("user_interaction")
|
|
144
|
+
|
|
145
|
+
self.current_round_data["messages"].append(
|
|
146
|
+
{
|
|
147
|
+
"role": "user",
|
|
148
|
+
"type": message_type,
|
|
149
|
+
"content": message,
|
|
150
|
+
"timestamp": datetime.now().isoformat(),
|
|
151
|
+
}
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
def log_assistant_response(
|
|
155
|
+
self, response: str, response_type: str = "assistant_response"
|
|
156
|
+
):
|
|
157
|
+
"""
|
|
158
|
+
Log assistant response
|
|
159
|
+
|
|
160
|
+
Args:
|
|
161
|
+
response: Assistant response content
|
|
162
|
+
response_type: Type of response (assistant_response, analysis, etc.)
|
|
163
|
+
"""
|
|
164
|
+
if not self.current_round_data:
|
|
165
|
+
self.start_new_round("assistant_interaction")
|
|
166
|
+
|
|
167
|
+
self.current_round_data["messages"].append(
|
|
168
|
+
{
|
|
169
|
+
"role": "assistant",
|
|
170
|
+
"type": response_type,
|
|
171
|
+
"content": response,
|
|
172
|
+
"timestamp": datetime.now().isoformat(),
|
|
173
|
+
}
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
def log_tool_calls(self, tool_calls: List[Dict[str, Any]]):
|
|
177
|
+
"""
|
|
178
|
+
Log tool calls made by the assistant
|
|
179
|
+
|
|
180
|
+
Args:
|
|
181
|
+
tool_calls: List of tool calls with id, name, and input
|
|
182
|
+
"""
|
|
183
|
+
if not self.current_round_data:
|
|
184
|
+
self.start_new_round("tool_execution")
|
|
185
|
+
|
|
186
|
+
for tool_call in tool_calls:
|
|
187
|
+
self.current_round_data["tool_calls"].append(
|
|
188
|
+
{
|
|
189
|
+
"id": tool_call.get("id", ""),
|
|
190
|
+
"name": tool_call.get("name", ""),
|
|
191
|
+
"input": tool_call.get("input", {}),
|
|
192
|
+
"timestamp": datetime.now().isoformat(),
|
|
193
|
+
}
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
def log_tool_results(self, tool_results: List[Dict[str, Any]]):
|
|
197
|
+
"""
|
|
198
|
+
Log tool execution results
|
|
199
|
+
|
|
200
|
+
Args:
|
|
201
|
+
tool_results: List of tool results with tool_name and result
|
|
202
|
+
"""
|
|
203
|
+
if not self.current_round_data:
|
|
204
|
+
self.start_new_round("tool_results")
|
|
205
|
+
|
|
206
|
+
for result in tool_results:
|
|
207
|
+
self.current_round_data["results"].append(
|
|
208
|
+
{
|
|
209
|
+
"tool_name": result.get("tool_name", ""),
|
|
210
|
+
"result": result.get("result", ""),
|
|
211
|
+
"timestamp": datetime.now().isoformat(),
|
|
212
|
+
}
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
def log_metadata(self, key: str, value: Any):
|
|
216
|
+
"""
|
|
217
|
+
Log metadata information
|
|
218
|
+
|
|
219
|
+
Args:
|
|
220
|
+
key: Metadata key
|
|
221
|
+
value: Metadata value
|
|
222
|
+
"""
|
|
223
|
+
if not self.current_round_data:
|
|
224
|
+
self.start_new_round("metadata")
|
|
225
|
+
|
|
226
|
+
self.current_round_data["metadata"][key] = value
|
|
227
|
+
|
|
228
|
+
def log_memory_optimization(
|
|
229
|
+
self,
|
|
230
|
+
messages_before: List[Dict],
|
|
231
|
+
messages_after: List[Dict],
|
|
232
|
+
optimization_stats: Dict[str, Any],
|
|
233
|
+
approach: str = "memory_optimization",
|
|
234
|
+
):
|
|
235
|
+
"""
|
|
236
|
+
Log memory optimization details including before/after message content
|
|
237
|
+
|
|
238
|
+
Args:
|
|
239
|
+
messages_before: Messages before optimization
|
|
240
|
+
messages_after: Messages after optimization
|
|
241
|
+
optimization_stats: Statistics about the optimization
|
|
242
|
+
approach: Optimization approach used
|
|
243
|
+
"""
|
|
244
|
+
if not self.current_round_data:
|
|
245
|
+
self.start_new_round("memory_optimization")
|
|
246
|
+
|
|
247
|
+
# Calculate what was removed/kept
|
|
248
|
+
removed_count = len(messages_before) - len(messages_after)
|
|
249
|
+
compression_ratio = (
|
|
250
|
+
(removed_count / len(messages_before) * 100) if messages_before else 0
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
# Log the optimization details
|
|
254
|
+
optimization_data = {
|
|
255
|
+
"approach": approach,
|
|
256
|
+
"messages_before_count": len(messages_before),
|
|
257
|
+
"messages_after_count": len(messages_after),
|
|
258
|
+
"messages_removed_count": removed_count,
|
|
259
|
+
"compression_ratio": f"{compression_ratio:.1f}%",
|
|
260
|
+
"optimization_stats": optimization_stats,
|
|
261
|
+
"timestamp": datetime.now().isoformat(),
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
# Store the optimization data
|
|
265
|
+
if "memory_optimizations" not in self.current_round_data:
|
|
266
|
+
self.current_round_data["memory_optimizations"] = []
|
|
267
|
+
|
|
268
|
+
self.current_round_data["memory_optimizations"].append(
|
|
269
|
+
{
|
|
270
|
+
"optimization_data": optimization_data,
|
|
271
|
+
"messages_before": messages_before,
|
|
272
|
+
"messages_after": messages_after,
|
|
273
|
+
}
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
# Log metadata
|
|
277
|
+
self.log_metadata("memory_optimization", optimization_data)
|
|
278
|
+
|
|
279
|
+
print(
|
|
280
|
+
f"🧹 Memory optimization logged: {len(messages_before)} → {len(messages_after)} messages ({compression_ratio:.1f}% compression)"
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
def complete_round(self, summary: str = "", status: str = "completed"):
|
|
284
|
+
"""
|
|
285
|
+
Complete the current round and write to log file
|
|
286
|
+
|
|
287
|
+
Args:
|
|
288
|
+
summary: Round summary
|
|
289
|
+
status: Round completion status
|
|
290
|
+
"""
|
|
291
|
+
if not self.current_round_data:
|
|
292
|
+
print("⚠️ No active round to complete")
|
|
293
|
+
return
|
|
294
|
+
|
|
295
|
+
self.current_round_data["end_time"] = datetime.now()
|
|
296
|
+
self.current_round_data["duration"] = (
|
|
297
|
+
self.current_round_data["end_time"] - self.current_round_data["start_time"]
|
|
298
|
+
).total_seconds()
|
|
299
|
+
self.current_round_data["summary"] = summary
|
|
300
|
+
self.current_round_data["status"] = status
|
|
301
|
+
|
|
302
|
+
# Write round to log file
|
|
303
|
+
self._write_round_to_log()
|
|
304
|
+
|
|
305
|
+
print(f"✅ Round {self.round_counter} completed: {status}")
|
|
306
|
+
|
|
307
|
+
# Clear current round data
|
|
308
|
+
self.current_round_data = {}
|
|
309
|
+
|
|
310
|
+
def _write_round_to_log(self):
|
|
311
|
+
"""Write the current round data to the log file in markdown format"""
|
|
312
|
+
try:
|
|
313
|
+
with open(self.log_filepath, "a", encoding="utf-8") as f:
|
|
314
|
+
round_data = self.current_round_data
|
|
315
|
+
|
|
316
|
+
# Round header
|
|
317
|
+
f.write(
|
|
318
|
+
f"\n## Round {round_data['round_number']}: {round_data['round_type'].title()}\n\n"
|
|
319
|
+
)
|
|
320
|
+
f.write(
|
|
321
|
+
f"**Start Time:** {round_data['start_time'].strftime('%Y-%m-%d %H:%M:%S')}\n"
|
|
322
|
+
)
|
|
323
|
+
f.write(
|
|
324
|
+
f"**End Time:** {round_data['end_time'].strftime('%Y-%m-%d %H:%M:%S')}\n"
|
|
325
|
+
)
|
|
326
|
+
f.write(f"**Duration:** {round_data['duration']:.2f} seconds\n")
|
|
327
|
+
f.write(f"**Status:** {round_data['status']}\n\n")
|
|
328
|
+
|
|
329
|
+
# Context information
|
|
330
|
+
if round_data.get("context"):
|
|
331
|
+
f.write("### Context\n\n")
|
|
332
|
+
for key, value in round_data["context"].items():
|
|
333
|
+
f.write(f"- **{key}:** {value}\n")
|
|
334
|
+
f.write("\n")
|
|
335
|
+
|
|
336
|
+
# Messages
|
|
337
|
+
if round_data.get("messages"):
|
|
338
|
+
f.write("### Messages\n\n")
|
|
339
|
+
for i, msg in enumerate(round_data["messages"], 1):
|
|
340
|
+
role_emoji = {
|
|
341
|
+
"system": "🔧",
|
|
342
|
+
"user": "👤",
|
|
343
|
+
"assistant": "🤖",
|
|
344
|
+
}.get(msg["role"], "📝")
|
|
345
|
+
f.write(
|
|
346
|
+
f"#### {role_emoji} {msg['role'].title()} Message {i}\n\n"
|
|
347
|
+
)
|
|
348
|
+
f.write(f"**Type:** {msg['type']}\n")
|
|
349
|
+
f.write(f"**Timestamp:** {msg['timestamp']}\n\n")
|
|
350
|
+
f.write("```\n")
|
|
351
|
+
f.write(msg["content"])
|
|
352
|
+
f.write("\n```\n\n")
|
|
353
|
+
|
|
354
|
+
# Tool calls
|
|
355
|
+
if round_data.get("tool_calls"):
|
|
356
|
+
f.write("### Tool Calls\n\n")
|
|
357
|
+
for i, tool_call in enumerate(round_data["tool_calls"], 1):
|
|
358
|
+
f.write(f"#### 🛠️ Tool Call {i}: {tool_call['name']}\n\n")
|
|
359
|
+
f.write(f"**ID:** {tool_call['id']}\n")
|
|
360
|
+
f.write(f"**Timestamp:** {tool_call['timestamp']}\n\n")
|
|
361
|
+
f.write("**Input:**\n")
|
|
362
|
+
f.write("```json\n")
|
|
363
|
+
f.write(
|
|
364
|
+
json.dumps(tool_call["input"], indent=2, ensure_ascii=False)
|
|
365
|
+
)
|
|
366
|
+
f.write("\n```\n\n")
|
|
367
|
+
|
|
368
|
+
# Tool results
|
|
369
|
+
if round_data.get("results"):
|
|
370
|
+
f.write("### Tool Results\n\n")
|
|
371
|
+
for i, result in enumerate(round_data["results"], 1):
|
|
372
|
+
f.write(f"#### 📊 Result {i}: {result['tool_name']}\n\n")
|
|
373
|
+
f.write(f"**Timestamp:** {result['timestamp']}\n\n")
|
|
374
|
+
f.write("**Result:**\n")
|
|
375
|
+
f.write("```\n")
|
|
376
|
+
f.write(str(result["result"]))
|
|
377
|
+
f.write("\n```\n\n")
|
|
378
|
+
|
|
379
|
+
# Memory Optimizations
|
|
380
|
+
if round_data.get("memory_optimizations"):
|
|
381
|
+
f.write("### Memory Optimizations\n\n")
|
|
382
|
+
for i, opt in enumerate(round_data["memory_optimizations"], 1):
|
|
383
|
+
opt_data = opt["optimization_data"]
|
|
384
|
+
messages_before = opt["messages_before"]
|
|
385
|
+
messages_after = opt["messages_after"]
|
|
386
|
+
|
|
387
|
+
f.write(f"#### 🧹 Memory Optimization {i}\n\n")
|
|
388
|
+
f.write(f"**Approach:** {opt_data['approach']}\n")
|
|
389
|
+
f.write(
|
|
390
|
+
f"**Messages Before:** {opt_data['messages_before_count']}\n"
|
|
391
|
+
)
|
|
392
|
+
f.write(
|
|
393
|
+
f"**Messages After:** {opt_data['messages_after_count']}\n"
|
|
394
|
+
)
|
|
395
|
+
f.write(
|
|
396
|
+
f"**Messages Removed:** {opt_data['messages_removed_count']}\n"
|
|
397
|
+
)
|
|
398
|
+
f.write(
|
|
399
|
+
f"**Compression Ratio:** {opt_data['compression_ratio']}\n"
|
|
400
|
+
)
|
|
401
|
+
f.write(f"**Timestamp:** {opt_data['timestamp']}\n\n")
|
|
402
|
+
|
|
403
|
+
# Show optimization stats
|
|
404
|
+
if opt_data.get("optimization_stats"):
|
|
405
|
+
f.write("**Optimization Statistics:**\n")
|
|
406
|
+
f.write("```json\n")
|
|
407
|
+
f.write(
|
|
408
|
+
json.dumps(
|
|
409
|
+
opt_data["optimization_stats"],
|
|
410
|
+
indent=2,
|
|
411
|
+
ensure_ascii=False,
|
|
412
|
+
)
|
|
413
|
+
)
|
|
414
|
+
f.write("\n```\n\n")
|
|
415
|
+
|
|
416
|
+
# Show messages before optimization (limited to last 5 for readability)
|
|
417
|
+
if messages_before:
|
|
418
|
+
f.write("**Messages Before Optimization (last 5):**\n\n")
|
|
419
|
+
for j, msg in enumerate(messages_before[-5:], 1):
|
|
420
|
+
role = msg.get("role", "unknown")
|
|
421
|
+
content = msg.get("content", "")
|
|
422
|
+
# Truncate very long messages
|
|
423
|
+
if len(content) > 3000:
|
|
424
|
+
content = content[:3000] + "...[truncated]"
|
|
425
|
+
f.write(
|
|
426
|
+
f"- **{role} {j}:** {content[:3000]}{'...' if len(content) > 100 else ''}\n"
|
|
427
|
+
)
|
|
428
|
+
f.write("\n")
|
|
429
|
+
|
|
430
|
+
# Show messages after optimization
|
|
431
|
+
if messages_after:
|
|
432
|
+
f.write("**Messages After Optimization:**\n\n")
|
|
433
|
+
for j, msg in enumerate(messages_after, 1):
|
|
434
|
+
role = msg.get("role", "unknown")
|
|
435
|
+
content = msg.get("content", "")
|
|
436
|
+
# Truncate very long messages
|
|
437
|
+
if len(content) > 3000:
|
|
438
|
+
content = content[:3000] + "...[truncated]"
|
|
439
|
+
f.write(
|
|
440
|
+
f"- **{role} {j}:** {content[:3000]}{'...' if len(content) > 100 else ''}\n"
|
|
441
|
+
)
|
|
442
|
+
f.write("\n")
|
|
443
|
+
|
|
444
|
+
# Show what was removed
|
|
445
|
+
if len(messages_before) > len(messages_after):
|
|
446
|
+
removed_messages = (
|
|
447
|
+
messages_before[: -len(messages_after)]
|
|
448
|
+
if messages_after
|
|
449
|
+
else messages_before
|
|
450
|
+
)
|
|
451
|
+
f.write(
|
|
452
|
+
f"**Messages Removed ({len(removed_messages)}):**\n\n"
|
|
453
|
+
)
|
|
454
|
+
for j, msg in enumerate(
|
|
455
|
+
removed_messages[-3:], 1
|
|
456
|
+
): # Show last 3 removed
|
|
457
|
+
role = msg.get("role", "unknown")
|
|
458
|
+
content = msg.get("content", "")
|
|
459
|
+
if len(content) > 3000:
|
|
460
|
+
content = content[:3000] + "...[truncated]"
|
|
461
|
+
f.write(f"- **{role} {j}:** {content}\n")
|
|
462
|
+
f.write("\n")
|
|
463
|
+
|
|
464
|
+
f.write("\n")
|
|
465
|
+
|
|
466
|
+
# Metadata
|
|
467
|
+
if round_data.get("metadata"):
|
|
468
|
+
f.write("### Metadata\n\n")
|
|
469
|
+
for key, value in round_data["metadata"].items():
|
|
470
|
+
if (
|
|
471
|
+
key != "memory_optimization"
|
|
472
|
+
): # Skip memory optimization metadata as it's shown above
|
|
473
|
+
f.write(f"- **{key}:** {value}\n")
|
|
474
|
+
f.write("\n")
|
|
475
|
+
|
|
476
|
+
# Summary
|
|
477
|
+
if round_data.get("summary"):
|
|
478
|
+
f.write("### Summary\n\n")
|
|
479
|
+
f.write(round_data["summary"])
|
|
480
|
+
f.write("\n\n")
|
|
481
|
+
|
|
482
|
+
# Separator
|
|
483
|
+
f.write("---\n\n")
|
|
484
|
+
|
|
485
|
+
except Exception as e:
|
|
486
|
+
print(f"⚠️ Failed to write round to log: {e}")
|
|
487
|
+
|
|
488
|
+
def log_complete_exchange(
|
|
489
|
+
self,
|
|
490
|
+
system_prompt: str = "",
|
|
491
|
+
user_message: str = "",
|
|
492
|
+
assistant_response: str = "",
|
|
493
|
+
tool_calls: List[Dict] = None,
|
|
494
|
+
tool_results: List[Dict] = None,
|
|
495
|
+
round_type: str = "exchange",
|
|
496
|
+
context: Dict = None,
|
|
497
|
+
summary: str = "",
|
|
498
|
+
):
|
|
499
|
+
"""
|
|
500
|
+
Log a complete exchange in a single call
|
|
501
|
+
|
|
502
|
+
Args:
|
|
503
|
+
system_prompt: System prompt (optional)
|
|
504
|
+
user_message: User message
|
|
505
|
+
assistant_response: Assistant response
|
|
506
|
+
tool_calls: Tool calls made
|
|
507
|
+
tool_results: Tool execution results
|
|
508
|
+
round_type: Type of round
|
|
509
|
+
context: Additional context
|
|
510
|
+
summary: Round summary
|
|
511
|
+
"""
|
|
512
|
+
self.start_new_round(round_type, context)
|
|
513
|
+
|
|
514
|
+
if system_prompt:
|
|
515
|
+
self.log_system_prompt(system_prompt)
|
|
516
|
+
|
|
517
|
+
if user_message:
|
|
518
|
+
self.log_user_message(user_message)
|
|
519
|
+
|
|
520
|
+
if assistant_response:
|
|
521
|
+
self.log_assistant_response(assistant_response)
|
|
522
|
+
|
|
523
|
+
if tool_calls:
|
|
524
|
+
self.log_tool_calls(tool_calls)
|
|
525
|
+
|
|
526
|
+
if tool_results:
|
|
527
|
+
self.log_tool_results(tool_results)
|
|
528
|
+
|
|
529
|
+
self.complete_round(summary)
|
|
530
|
+
|
|
531
|
+
def get_session_stats(self) -> Dict[str, Any]:
|
|
532
|
+
"""Get session statistics"""
|
|
533
|
+
return {
|
|
534
|
+
"paper_id": self.paper_id,
|
|
535
|
+
"session_start": self.session_start_time.isoformat(),
|
|
536
|
+
"total_rounds": self.round_counter,
|
|
537
|
+
"log_file": self.log_filepath,
|
|
538
|
+
"session_duration": (
|
|
539
|
+
datetime.now() - self.session_start_time
|
|
540
|
+
).total_seconds(),
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
def finalize_session(self, final_summary: str = ""):
|
|
544
|
+
"""
|
|
545
|
+
Finalize the logging session
|
|
546
|
+
|
|
547
|
+
Args:
|
|
548
|
+
final_summary: Final session summary
|
|
549
|
+
"""
|
|
550
|
+
try:
|
|
551
|
+
with open(self.log_filepath, "a", encoding="utf-8") as f:
|
|
552
|
+
f.write("\n## Session Summary\n\n")
|
|
553
|
+
f.write(f"**Total Rounds:** {self.round_counter}\n")
|
|
554
|
+
f.write(
|
|
555
|
+
f"**Session Duration:** {(datetime.now() - self.session_start_time).total_seconds():.2f} seconds\n"
|
|
556
|
+
)
|
|
557
|
+
f.write(
|
|
558
|
+
f"**End Time:** {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}\n\n"
|
|
559
|
+
)
|
|
560
|
+
|
|
561
|
+
if final_summary:
|
|
562
|
+
f.write("### Final Summary\n\n")
|
|
563
|
+
f.write(final_summary)
|
|
564
|
+
f.write("\n\n")
|
|
565
|
+
|
|
566
|
+
f.write("---\n\n")
|
|
567
|
+
f.write("*End of Session*\n")
|
|
568
|
+
|
|
569
|
+
except Exception as e:
|
|
570
|
+
print(f"⚠️ Failed to finalize session: {e}")
|
|
571
|
+
|
|
572
|
+
print(f"🎯 Session finalized: {self.round_counter} rounds logged")
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
# Utility functions for easy integration
|
|
576
|
+
def create_dialogue_logger(paper_id: str, base_path: str = None) -> DialogueLogger:
|
|
577
|
+
"""
|
|
578
|
+
Create a dialogue logger for a specific paper
|
|
579
|
+
|
|
580
|
+
Args:
|
|
581
|
+
paper_id: Paper identifier
|
|
582
|
+
base_path: Base path for logs
|
|
583
|
+
|
|
584
|
+
Returns:
|
|
585
|
+
DialogueLogger instance
|
|
586
|
+
"""
|
|
587
|
+
return DialogueLogger(paper_id, base_path)
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
def extract_paper_id_from_path(path: str) -> str:
|
|
591
|
+
"""
|
|
592
|
+
Extract paper ID from a file path
|
|
593
|
+
|
|
594
|
+
Args:
|
|
595
|
+
path: File path containing paper information
|
|
596
|
+
|
|
597
|
+
Returns:
|
|
598
|
+
Paper ID string
|
|
599
|
+
"""
|
|
600
|
+
# Extract paper ID from path like "/data2/.../papers/1/initial_plan.txt"
|
|
601
|
+
parts = path.split("/")
|
|
602
|
+
for i, part in enumerate(parts):
|
|
603
|
+
if part == "papers" and i + 1 < len(parts):
|
|
604
|
+
return parts[i + 1]
|
|
605
|
+
return "unknown"
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
# Example usage
|
|
609
|
+
if __name__ == "__main__":
|
|
610
|
+
# Test the dialogue logger
|
|
611
|
+
logger = DialogueLogger("1")
|
|
612
|
+
|
|
613
|
+
# Log a complete exchange
|
|
614
|
+
logger.log_complete_exchange(
|
|
615
|
+
system_prompt="You are a code implementation assistant.",
|
|
616
|
+
user_message="Implement the transformer model",
|
|
617
|
+
assistant_response="I'll implement the transformer model step by step.",
|
|
618
|
+
tool_calls=[
|
|
619
|
+
{"id": "1", "name": "write_file", "input": {"filename": "transformer.py"}}
|
|
620
|
+
],
|
|
621
|
+
tool_results=[
|
|
622
|
+
{"tool_name": "write_file", "result": "File created successfully"}
|
|
623
|
+
],
|
|
624
|
+
round_type="implementation",
|
|
625
|
+
context={"files_implemented": 1},
|
|
626
|
+
summary="Successfully implemented transformer model",
|
|
627
|
+
)
|
|
628
|
+
|
|
629
|
+
# Test memory optimization logging
|
|
630
|
+
logger.start_new_round(
|
|
631
|
+
"memory_optimization", {"trigger_reason": "write_file_detected"}
|
|
632
|
+
)
|
|
633
|
+
|
|
634
|
+
# Mock messages before and after optimization
|
|
635
|
+
messages_before = [
|
|
636
|
+
{"role": "user", "content": "Original message 1"},
|
|
637
|
+
{"role": "assistant", "content": "Original response 1"},
|
|
638
|
+
{"role": "user", "content": "Original message 2"},
|
|
639
|
+
{"role": "assistant", "content": "Original response 2"},
|
|
640
|
+
{"role": "user", "content": "Original message 3"},
|
|
641
|
+
]
|
|
642
|
+
|
|
643
|
+
messages_after = [
|
|
644
|
+
{"role": "user", "content": "Original message 1"},
|
|
645
|
+
{"role": "assistant", "content": "Original response 1"},
|
|
646
|
+
{"role": "user", "content": "Original message 3"},
|
|
647
|
+
]
|
|
648
|
+
|
|
649
|
+
# Mock optimization stats
|
|
650
|
+
optimization_stats = {
|
|
651
|
+
"implemented_files_tracked": 2,
|
|
652
|
+
"current_round": 5,
|
|
653
|
+
"concise_mode_active": True,
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
# Log memory optimization
|
|
657
|
+
logger.log_memory_optimization(
|
|
658
|
+
messages_before=messages_before,
|
|
659
|
+
messages_after=messages_after,
|
|
660
|
+
optimization_stats=optimization_stats,
|
|
661
|
+
approach="clear_after_write_file",
|
|
662
|
+
)
|
|
663
|
+
|
|
664
|
+
logger.complete_round("Memory optimization test completed")
|
|
665
|
+
|
|
666
|
+
# Finalize session
|
|
667
|
+
logger.finalize_session(
|
|
668
|
+
"Test session with memory optimization logging completed successfully"
|
|
669
|
+
)
|
|
670
|
+
|
|
671
|
+
print("✅ Dialogue logger test completed with memory optimization")
|