ras-commander-mcp 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ras_commander_mcp-0.3.0/.ai_tools/generate_llm_knowledge_bases.py +560 -0
- ras_commander_mcp-0.3.0/.ai_tools/llm_knowledge_bases/ras-commander-mcp_fullrepo.txt +5260 -0
- ras_commander_mcp-0.3.0/.claude/settings.local.json +18 -0
- ras_commander_mcp-0.3.0/.gitattributes +2 -0
- ras_commander_mcp-0.3.0/.gitignore +46 -0
- ras_commander_mcp-0.3.0/CLAUDE.md +54 -0
- ras_commander_mcp-0.3.0/DEVELOPMENT.md +469 -0
- ras_commander_mcp-0.3.0/LICENSE +21 -0
- ras_commander_mcp-0.3.0/MANIFEST.in +24 -0
- ras_commander_mcp-0.3.0/PKG-INFO +374 -0
- ras_commander_mcp-0.3.0/RAS Commander MCP brainstorming.txt +74 -0
- ras_commander_mcp-0.3.0/RAS Commander MCP.txt +417 -0
- ras_commander_mcp-0.3.0/README.md +341 -0
- ras_commander_mcp-0.3.0/TRADEMARKS.md +22 -0
- ras_commander_mcp-0.3.0/api.md +1823 -0
- ras_commander_mcp-0.3.0/claude_desktop_config.json +11 -0
- ras_commander_mcp-0.3.0/example_client.py +201 -0
- ras_commander_mcp-0.3.0/examples/claude_desktop_config.json +11 -0
- ras_commander_mcp-0.3.0/examples/example_client.py +138 -0
- ras_commander_mcp-0.3.0/instructions.txt +373 -0
- ras_commander_mcp-0.3.0/nonworking_tools.py +368 -0
- ras_commander_mcp-0.3.0/package.json +20 -0
- ras_commander_mcp-0.3.0/pyproject.toml +70 -0
- ras_commander_mcp-0.3.0/ras-commander API to use for developing tool calls.txt +126 -0
- ras_commander_mcp-0.3.0/ras_commander_mcp_logo.svg +88 -0
- ras_commander_mcp-0.3.0/requirements.txt +3 -0
- ras_commander_mcp-0.3.0/server.py +11 -0
- ras_commander_mcp-0.3.0/src/ras_commander_mcp/__init__.py +7 -0
- ras_commander_mcp-0.3.0/src/ras_commander_mcp/server.py +1045 -0
- ras_commander_mcp-0.3.0/test_mcp_server.ipynb +400 -0
- ras_commander_mcp-0.3.0/test_server.py +181 -0
|
@@ -0,0 +1,560 @@
|
|
|
1
|
+
"""
|
|
2
|
+
This script generates a comprehensive summary knowledge base for the ras-commander library.
|
|
3
|
+
It processes the project files and creates the following output file:
|
|
4
|
+
|
|
5
|
+
ras-commander_fullrepo.txt:
|
|
6
|
+
A comprehensive summary of all relevant project files, including their content
|
|
7
|
+
and structure. This file provides an overview of the entire codebase, including
|
|
8
|
+
all files and folders except those specified in OMIT_FOLDERS and OMIT_FILES.
|
|
9
|
+
|
|
10
|
+
The output file is generated in the 'llm_knowledge_bases' directory and serves
|
|
11
|
+
as a complete reference for AI assistants or developers who need a full overview
|
|
12
|
+
of the project structure and content.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
import re
|
|
18
|
+
import json
|
|
19
|
+
from typing import Dict, Any, List, Union
|
|
20
|
+
|
|
21
|
+
# Configuration
|
|
22
|
+
OMIT_FOLDERS = [
|
|
23
|
+
"testdata", ".ai_tools", ".git", ".gemini", ".claude", "ArcHydro Default Layers", "Images", "Bald Eagle Creek", "__pycache__", ".git", ".github", "tests", "docs", "library_assistant", "__pycache__", ".conda", "workspace"
|
|
24
|
+
"build", "dist", "ras_commander.egg-info", "venv", "ras_commander.egg-info", "log_folder", "logs",
|
|
25
|
+
"example_projects", "llm_knowledge_bases", "misc", "ai_tools", "FEMA_BLE_Models", "hdf_example_data", "ras_example_categories", "data", "apidocs", "build", "dist", "ras_commander.egg-info", "venv", "log_folder", "logs",
|
|
26
|
+
]
|
|
27
|
+
OMIT_FILES = [
|
|
28
|
+
".lyrx", ".png",".hdf", ".pyc", ".pyo", ".pyd", ".dll", ".so", ".dylib", ".exe",
|
|
29
|
+
".bat", ".sh", ".log", ".tmp", ".bak", ".swp",
|
|
30
|
+
".DS_Store", "Thumbs.db", "example_projects.zip",
|
|
31
|
+
"Example_Projects_6_6.zip", "example_projects.ipynb", "11_Using_RasExamples.ipynb",
|
|
32
|
+
"future_dev_roadmap.ipynb", "structures_attributes.csv", "example_projects.csv",
|
|
33
|
+
".ico", ".png", ".jpg", ".jpeg", ".gif", ".svg", ".webp", ".mp4", ".avi", ".mov", ".mp3", ".wav", ".m4a", ".m4v", ".ogg", ".webm"
|
|
34
|
+
]
|
|
35
|
+
SUMMARY_OUTPUT_DIR = "llm_knowledge_bases"
|
|
36
|
+
SCRIPT_NAME = Path(__file__).name
|
|
37
|
+
|
|
38
|
+
# Recursively delete all __pycache__ folders and their contents
|
|
39
|
+
for folder in Path(__file__).parent.parent.rglob("__pycache__"):
|
|
40
|
+
if folder.is_dir():
|
|
41
|
+
print(f"Deleting __pycache__ folder and contents: {folder}")
|
|
42
|
+
try:
|
|
43
|
+
# Recursively delete all subfolders and files
|
|
44
|
+
for item in folder.rglob("*"):
|
|
45
|
+
if item.is_file():
|
|
46
|
+
item.unlink()
|
|
47
|
+
elif item.is_dir():
|
|
48
|
+
item.rmdir()
|
|
49
|
+
# Delete the empty __pycache__ folder itself
|
|
50
|
+
folder.rmdir()
|
|
51
|
+
print(f"Successfully deleted {folder} and all contents")
|
|
52
|
+
except Exception as e:
|
|
53
|
+
print(f"Error deleting {folder}: {e}")
|
|
54
|
+
|
|
55
|
+
def ensure_output_dir(base_path: Path) -> Path:
|
|
56
|
+
output_dir = base_path / SUMMARY_OUTPUT_DIR
|
|
57
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
58
|
+
print(f"Output directory ensured to exist: {output_dir}")
|
|
59
|
+
return output_dir
|
|
60
|
+
|
|
61
|
+
def should_omit(filepath: Path) -> bool:
|
|
62
|
+
if filepath.name == SCRIPT_NAME:
|
|
63
|
+
return True
|
|
64
|
+
if any(omit_folder in filepath.parts for omit_folder in OMIT_FOLDERS):
|
|
65
|
+
return True
|
|
66
|
+
if any(filepath.suffix == ext or filepath.name == ext for ext in OMIT_FILES):
|
|
67
|
+
return True
|
|
68
|
+
return False
|
|
69
|
+
|
|
70
|
+
def process_notebook_content(filepath: Path) -> str:
|
|
71
|
+
"""
|
|
72
|
+
Process a Jupyter notebook to remove images and truncate dataframe outputs.
|
|
73
|
+
|
|
74
|
+
Args:
|
|
75
|
+
filepath: Path to the notebook file
|
|
76
|
+
|
|
77
|
+
Returns:
|
|
78
|
+
Processed notebook content as a string
|
|
79
|
+
"""
|
|
80
|
+
try:
|
|
81
|
+
with open(filepath, 'r', encoding='utf-8') as f:
|
|
82
|
+
notebook = json.load(f)
|
|
83
|
+
|
|
84
|
+
# Process each cell
|
|
85
|
+
for cell in notebook.get('cells', []):
|
|
86
|
+
if cell.get('cell_type') == 'code':
|
|
87
|
+
# Process outputs
|
|
88
|
+
if 'outputs' in cell:
|
|
89
|
+
cell['outputs'] = clean_notebook_outputs(cell['outputs'])
|
|
90
|
+
|
|
91
|
+
# Convert back to string with indentation for readability
|
|
92
|
+
return json.dumps(notebook, indent=2)
|
|
93
|
+
|
|
94
|
+
except Exception as e:
|
|
95
|
+
print(f"Error processing notebook {filepath}: {e}")
|
|
96
|
+
# Fall back to original content
|
|
97
|
+
with open(filepath, 'r', encoding='utf-8') as f:
|
|
98
|
+
return f.read()
|
|
99
|
+
|
|
100
|
+
def process_notebook_no_outputs(filepath: Path) -> str:
|
|
101
|
+
"""
|
|
102
|
+
Process a Jupyter notebook to completely remove all outputs.
|
|
103
|
+
|
|
104
|
+
Args:
|
|
105
|
+
filepath: Path to the notebook file
|
|
106
|
+
|
|
107
|
+
Returns:
|
|
108
|
+
Processed notebook content as a string with all outputs removed
|
|
109
|
+
"""
|
|
110
|
+
try:
|
|
111
|
+
with open(filepath, 'r', encoding='utf-8') as f:
|
|
112
|
+
notebook = json.load(f)
|
|
113
|
+
|
|
114
|
+
# Process each cell
|
|
115
|
+
for cell in notebook.get('cells', []):
|
|
116
|
+
if cell.get('cell_type') == 'code':
|
|
117
|
+
# Remove all outputs
|
|
118
|
+
cell['outputs'] = []
|
|
119
|
+
# Reset execution count
|
|
120
|
+
if 'execution_count' in cell:
|
|
121
|
+
cell['execution_count'] = None
|
|
122
|
+
|
|
123
|
+
# Convert back to string with indentation for readability
|
|
124
|
+
return json.dumps(notebook, indent=2)
|
|
125
|
+
|
|
126
|
+
except Exception as e:
|
|
127
|
+
print(f"Error processing notebook {filepath}: {e}")
|
|
128
|
+
# Fall back to original content
|
|
129
|
+
with open(filepath, 'r', encoding='utf-8') as f:
|
|
130
|
+
return f.read()
|
|
131
|
+
|
|
132
|
+
def save_cleaned_notebooks(summarize_subfolder: Path, output_dir: Path) -> None:
|
|
133
|
+
"""
|
|
134
|
+
Save cleaned versions of all notebooks to a separate subfolder.
|
|
135
|
+
All notebooks will be placed directly in the root of example_notebooks_cleaned.
|
|
136
|
+
If the cleaned notebook is >100KB, remove all outputs.
|
|
137
|
+
"""
|
|
138
|
+
cleaned_notebooks_dir = output_dir / "example_notebooks_cleaned"
|
|
139
|
+
cleaned_notebooks_dir.mkdir(parents=True, exist_ok=True)
|
|
140
|
+
print(f"Creating cleaned notebooks directory: {cleaned_notebooks_dir}")
|
|
141
|
+
|
|
142
|
+
# Find all notebooks
|
|
143
|
+
notebooks = list(summarize_subfolder.rglob('*.ipynb'))
|
|
144
|
+
|
|
145
|
+
for notebook_path in notebooks:
|
|
146
|
+
if should_omit(notebook_path):
|
|
147
|
+
continue
|
|
148
|
+
try:
|
|
149
|
+
# Process the notebook to clean outputs
|
|
150
|
+
with open(notebook_path, 'r', encoding='utf-8') as f:
|
|
151
|
+
notebook = json.load(f)
|
|
152
|
+
# Process each cell
|
|
153
|
+
for cell in notebook.get('cells', []):
|
|
154
|
+
if cell.get('cell_type') == 'code':
|
|
155
|
+
if 'outputs' in cell:
|
|
156
|
+
cell['outputs'] = clean_notebook_outputs(cell['outputs'])
|
|
157
|
+
# Use only the filename for the target path (no subdirectories)
|
|
158
|
+
target_path = cleaned_notebooks_dir / notebook_path.name
|
|
159
|
+
# Overwrite any existing file with the same name
|
|
160
|
+
if target_path.exists():
|
|
161
|
+
target_path.unlink()
|
|
162
|
+
# Save the cleaned notebook
|
|
163
|
+
with open(target_path, 'w', encoding='utf-8') as f:
|
|
164
|
+
json.dump(notebook, f, indent=2)
|
|
165
|
+
# Check file size; if >100KB, remove all outputs and save again
|
|
166
|
+
if target_path.stat().st_size > 100 * 1024:
|
|
167
|
+
print(f"Notebook {notebook_path} cleaned version >100KB, removing all outputs.")
|
|
168
|
+
no_output_content = process_notebook_no_outputs(notebook_path)
|
|
169
|
+
with open(target_path, 'w', encoding='utf-8') as f:
|
|
170
|
+
f.write(no_output_content)
|
|
171
|
+
print(f"Saved cleaned notebook: {target_path}")
|
|
172
|
+
except Exception as e:
|
|
173
|
+
print(f"Error processing notebook {notebook_path}: {e}")
|
|
174
|
+
|
|
175
|
+
def clean_notebook_outputs(outputs: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
|
176
|
+
"""
|
|
177
|
+
Clean cell outputs by removing images and truncating dataframes.
|
|
178
|
+
|
|
179
|
+
Args:
|
|
180
|
+
outputs: List of cell output dictionaries
|
|
181
|
+
|
|
182
|
+
Returns:
|
|
183
|
+
Cleaned outputs
|
|
184
|
+
"""
|
|
185
|
+
cleaned_outputs = []
|
|
186
|
+
|
|
187
|
+
for output in outputs:
|
|
188
|
+
output_type = output.get('output_type', '')
|
|
189
|
+
|
|
190
|
+
# Handle display_data outputs (images, HTML, etc.)
|
|
191
|
+
if output_type == 'display_data':
|
|
192
|
+
new_output = output.copy()
|
|
193
|
+
|
|
194
|
+
# Remove image data (png, jpeg, etc.)
|
|
195
|
+
if 'data' in new_output:
|
|
196
|
+
# Remove all image formats
|
|
197
|
+
for img_format in ['image/png', 'image/jpeg', 'image/svg+xml']:
|
|
198
|
+
if img_format in new_output['data']:
|
|
199
|
+
del new_output['data'][img_format]
|
|
200
|
+
|
|
201
|
+
# Check if this might be a DataFrame or other rich HTML display
|
|
202
|
+
if 'text/html' in new_output['data']:
|
|
203
|
+
html_content = new_output['data']['text/html']
|
|
204
|
+
new_output['data']['text/html'] = process_html_output(html_content)
|
|
205
|
+
|
|
206
|
+
# If data dict is now empty or only has empty values, add a placeholder
|
|
207
|
+
if not new_output['data'] or all(not v for v in new_output['data'].values()):
|
|
208
|
+
new_output['data']['text/plain'] = '[Image or rich display removed during preprocessing]'
|
|
209
|
+
|
|
210
|
+
cleaned_outputs.append(new_output)
|
|
211
|
+
continue
|
|
212
|
+
|
|
213
|
+
# Handle execute_result outputs (including dataframes, xarrays)
|
|
214
|
+
elif output_type == 'execute_result':
|
|
215
|
+
new_output = output.copy()
|
|
216
|
+
|
|
217
|
+
if 'data' in new_output:
|
|
218
|
+
# Process HTML content (DataFrames, xarray objects)
|
|
219
|
+
if 'text/html' in new_output['data']:
|
|
220
|
+
html_content = new_output['data']['text/html']
|
|
221
|
+
new_output['data']['text/html'] = process_html_output(html_content)
|
|
222
|
+
|
|
223
|
+
# Process plain text output
|
|
224
|
+
if 'text/plain' in new_output['data']:
|
|
225
|
+
text_content = new_output['data']['text/plain']
|
|
226
|
+
new_output['data']['text/plain'] = process_text_output(text_content)
|
|
227
|
+
|
|
228
|
+
cleaned_outputs.append(new_output)
|
|
229
|
+
|
|
230
|
+
# Handle stream outputs (stdout/stderr)
|
|
231
|
+
elif output_type == 'stream':
|
|
232
|
+
new_output = output.copy()
|
|
233
|
+
|
|
234
|
+
# Truncate very long text outputs
|
|
235
|
+
if 'text' in new_output and isinstance(new_output['text'], str):
|
|
236
|
+
text = new_output['text']
|
|
237
|
+
lines = text.splitlines()
|
|
238
|
+
|
|
239
|
+
# Truncate if more than 20 lines
|
|
240
|
+
if len(lines) > 20:
|
|
241
|
+
truncated_text = '\n'.join(lines[:10]) + '\n...\n' + '\n'.join(lines[-5:])
|
|
242
|
+
truncated_text += f"\n[Output truncated, {len(lines)} lines total]"
|
|
243
|
+
new_output['text'] = truncated_text
|
|
244
|
+
|
|
245
|
+
cleaned_outputs.append(new_output)
|
|
246
|
+
|
|
247
|
+
# Handle error outputs
|
|
248
|
+
elif output_type == 'error':
|
|
249
|
+
# Keep error outputs as they are (they're usually important)
|
|
250
|
+
cleaned_outputs.append(output)
|
|
251
|
+
|
|
252
|
+
else:
|
|
253
|
+
# For other output types, include them as is
|
|
254
|
+
cleaned_outputs.append(output)
|
|
255
|
+
|
|
256
|
+
return cleaned_outputs
|
|
257
|
+
|
|
258
|
+
def process_html_output(html_content: str) -> str:
|
|
259
|
+
"""
|
|
260
|
+
Process HTML output content to truncate and simplify it.
|
|
261
|
+
|
|
262
|
+
Args:
|
|
263
|
+
html_content: HTML content to process
|
|
264
|
+
|
|
265
|
+
Returns:
|
|
266
|
+
Processed HTML content
|
|
267
|
+
"""
|
|
268
|
+
# Handle case where html_content is not a string
|
|
269
|
+
if not isinstance(html_content, str):
|
|
270
|
+
try:
|
|
271
|
+
# Try to convert to string if possible
|
|
272
|
+
html_content = str(html_content)
|
|
273
|
+
except Exception:
|
|
274
|
+
# If conversion fails, return a placeholder
|
|
275
|
+
return "<div><pre>[Non-string HTML content removed during preprocessing]</pre></div>"
|
|
276
|
+
|
|
277
|
+
# Check for DataFrame HTML pattern
|
|
278
|
+
if '<table' in html_content and ('dataframe' in html_content or '<style' in html_content):
|
|
279
|
+
return truncate_dataframe_html(html_content)
|
|
280
|
+
|
|
281
|
+
# Check for xarray HTML pattern
|
|
282
|
+
elif 'xarray' in html_content.lower() and ('<table' in html_content or '<div' in html_content):
|
|
283
|
+
# Extract xarray type
|
|
284
|
+
xarray_type = "xarray.Dataset" if "xarray.Dataset" in html_content else "xarray.DataArray"
|
|
285
|
+
|
|
286
|
+
# Extract dimensions if possible
|
|
287
|
+
dims_match = re.search(r'Dimensions:(.+?)<', html_content, re.DOTALL)
|
|
288
|
+
dims_info = dims_match.group(1).strip() if dims_match else "Unknown dimensions"
|
|
289
|
+
|
|
290
|
+
# Return simplified version
|
|
291
|
+
return f"""<div><pre>{xarray_type} with {dims_info}
|
|
292
|
+
[Full xarray output truncated during preprocessing]</pre></div>"""
|
|
293
|
+
|
|
294
|
+
# Check for other kinds of rich HTML content (plots, widgets, etc.)
|
|
295
|
+
elif any(pattern in html_content.lower() for pattern in
|
|
296
|
+
['<svg', 'matplotlib', 'bokeh', 'plotly', 'widget', 'vis']):
|
|
297
|
+
return """<div><pre>[Visualization or interactive content removed during preprocessing]</pre></div>"""
|
|
298
|
+
|
|
299
|
+
# Other HTML content - truncate if very long
|
|
300
|
+
elif len(html_content) > 5000:
|
|
301
|
+
return f"""<div><pre>[Long HTML output truncated: {len(html_content)} characters]</pre></div>"""
|
|
302
|
+
|
|
303
|
+
# Otherwise, keep the HTML content as is
|
|
304
|
+
return html_content
|
|
305
|
+
|
|
306
|
+
def process_text_output(text_content: str) -> str:
|
|
307
|
+
"""
|
|
308
|
+
Process text output content to truncate and simplify it.
|
|
309
|
+
|
|
310
|
+
Args:
|
|
311
|
+
text_content: Text content to process
|
|
312
|
+
|
|
313
|
+
Returns:
|
|
314
|
+
Processed text content
|
|
315
|
+
"""
|
|
316
|
+
# Handle case where text_content is not a string
|
|
317
|
+
if not isinstance(text_content, str):
|
|
318
|
+
try:
|
|
319
|
+
# Try to convert to string if possible
|
|
320
|
+
text_content = str(text_content)
|
|
321
|
+
except Exception:
|
|
322
|
+
# If conversion fails, return a placeholder
|
|
323
|
+
return "[Non-string text content removed during preprocessing]"
|
|
324
|
+
|
|
325
|
+
# Check for DataFrame text representation
|
|
326
|
+
if ('DataFrame' in text_content and '\n' in text_content) or \
|
|
327
|
+
('[' in text_content and ']' in text_content and '\n' in text_content):
|
|
328
|
+
|
|
329
|
+
# Count the number of lines
|
|
330
|
+
lines = text_content.splitlines()
|
|
331
|
+
if len(lines) > 10:
|
|
332
|
+
# Simple truncation for DataFrames
|
|
333
|
+
return "[DataFrame output truncated, showing preview only]\n" + '\n'.join(lines[:7]) + '\n...'
|
|
334
|
+
|
|
335
|
+
# Check for xarray text representation
|
|
336
|
+
elif 'xarray.Dataset' in text_content or 'xarray.DataArray' in text_content:
|
|
337
|
+
# Extract xarray type
|
|
338
|
+
xarray_type = "xarray.Dataset" if "xarray.Dataset" in text_content else "xarray.DataArray"
|
|
339
|
+
|
|
340
|
+
# Extract dimensions if possible
|
|
341
|
+
dims_match = re.search(r'Dimensions:(.+?)\n', text_content)
|
|
342
|
+
dims_info = dims_match.group(1).strip() if dims_match else "Unknown dimensions"
|
|
343
|
+
|
|
344
|
+
# Abbreviated description
|
|
345
|
+
return f"{xarray_type} with {dims_info}\n[Full xarray output truncated during preprocessing]"
|
|
346
|
+
|
|
347
|
+
# Truncate general long text outputs
|
|
348
|
+
elif len(text_content) > 2000:
|
|
349
|
+
lines = text_content.splitlines()
|
|
350
|
+
if len(lines) > 20:
|
|
351
|
+
return '\n'.join(lines[:10]) + '\n...\n' + '\n'.join(lines[-5:]) + \
|
|
352
|
+
f"\n[Output truncated, {len(lines)} lines total]"
|
|
353
|
+
else:
|
|
354
|
+
return text_content[:1000] + f"\n...\n[Output truncated, {len(text_content)} characters total]"
|
|
355
|
+
|
|
356
|
+
# Otherwise, keep the text as is
|
|
357
|
+
return text_content
|
|
358
|
+
|
|
359
|
+
def truncate_dataframe_html(html_content: str) -> str:
|
|
360
|
+
"""
|
|
361
|
+
Truncate an HTML dataframe to show only the header and a few rows.
|
|
362
|
+
|
|
363
|
+
Args:
|
|
364
|
+
html_content: HTML content containing a dataframe
|
|
365
|
+
|
|
366
|
+
Returns:
|
|
367
|
+
Truncated HTML content
|
|
368
|
+
"""
|
|
369
|
+
# Handle case where html_content is not a string
|
|
370
|
+
if not isinstance(html_content, str):
|
|
371
|
+
try:
|
|
372
|
+
# Try to convert to string if possible
|
|
373
|
+
html_content = str(html_content)
|
|
374
|
+
except Exception:
|
|
375
|
+
# If conversion fails, return a placeholder
|
|
376
|
+
return "<div><pre>[Non-string HTML DataFrame content removed during preprocessing]</pre></div>"
|
|
377
|
+
|
|
378
|
+
# Keep the styling information
|
|
379
|
+
style_match = re.search(r'<style.*?</style>', html_content, re.DOTALL)
|
|
380
|
+
style_section = style_match.group(0) if style_match else ""
|
|
381
|
+
|
|
382
|
+
# Find the table
|
|
383
|
+
table_match = re.search(r'<table.*?</table>', html_content, re.DOTALL)
|
|
384
|
+
if not table_match:
|
|
385
|
+
return html_content # Not a table, return as is
|
|
386
|
+
|
|
387
|
+
table_content = table_match.group(0)
|
|
388
|
+
|
|
389
|
+
# Extract the header
|
|
390
|
+
header_match = re.search(r'<thead.*?</thead>', table_content, re.DOTALL)
|
|
391
|
+
header_section = header_match.group(0) if header_match else ""
|
|
392
|
+
|
|
393
|
+
# Extract the first few data rows (up to 5)
|
|
394
|
+
body_match = re.search(r'<tbody.*?</tbody>', table_content, re.DOTALL)
|
|
395
|
+
if body_match:
|
|
396
|
+
body_content = body_match.group(0)
|
|
397
|
+
row_matches = re.findall(r'<tr>.*?</tr>', body_content, re.DOTALL)
|
|
398
|
+
|
|
399
|
+
max_rows = min(5, len(row_matches))
|
|
400
|
+
first_rows = ''.join(row_matches[:max_rows])
|
|
401
|
+
|
|
402
|
+
# Construct a new tbody with limited rows plus truncation message
|
|
403
|
+
truncated_body = f"<tbody>\n {first_rows}\n <tr><td colspan=\"100%\" style=\"text-align:center\">[... additional rows truncated ...]</td></tr>\n </tbody>"
|
|
404
|
+
else:
|
|
405
|
+
truncated_body = "<tbody><tr><td>[No data rows]</td></tr></tbody>"
|
|
406
|
+
|
|
407
|
+
# Reconstruct the table
|
|
408
|
+
table_start_match = re.search(r'<table.*?>', table_content)
|
|
409
|
+
table_start = table_start_match.group(0) if table_start_match else "<table>"
|
|
410
|
+
|
|
411
|
+
truncated_table = f"{table_start}\n {header_section}\n {truncated_body}\n</table>"
|
|
412
|
+
|
|
413
|
+
# Put it all together
|
|
414
|
+
return f"<div>\n{style_section}\n{truncated_table}\n</div>"
|
|
415
|
+
|
|
416
|
+
def read_file_contents(filepath: Path) -> str:
|
|
417
|
+
try:
|
|
418
|
+
# Process Jupyter notebooks specially
|
|
419
|
+
if filepath.suffix.lower() == '.ipynb':
|
|
420
|
+
print(f"Processing notebook: {filepath}")
|
|
421
|
+
# For full repo, truncate outputs
|
|
422
|
+
return process_notebook_content(filepath)
|
|
423
|
+
|
|
424
|
+
# For XML files, remove binary image data and <Binary><Thumbnail><Data ...>...</Data></Thumbnail></Binary> blocks
|
|
425
|
+
if filepath.suffix.lower() == '.xml':
|
|
426
|
+
with open(filepath, 'r', encoding='utf-8') as infile:
|
|
427
|
+
content = infile.read()
|
|
428
|
+
# Remove binary image data between <Enclosure> tags
|
|
429
|
+
content = re.sub(r'<Enclosure[^>]*>.*?</Enclosure>', '', content, flags=re.DOTALL)
|
|
430
|
+
# Remove <Binary><Thumbnail><Data ...>...</Data></Thumbnail></Binary> blocks
|
|
431
|
+
content = re.sub(
|
|
432
|
+
r'<Binary>\s*<Thumbnail>\s*<Data[^>]*>.*?</Data>\s*</Thumbnail>\s*</Binary>',
|
|
433
|
+
'',
|
|
434
|
+
content,
|
|
435
|
+
flags=re.DOTALL | re.IGNORECASE
|
|
436
|
+
)
|
|
437
|
+
print(f"Reading and cleaning XML content of file: {filepath}")
|
|
438
|
+
return content
|
|
439
|
+
|
|
440
|
+
# Regular file reading for other files
|
|
441
|
+
with open(filepath, 'r', encoding='utf-8') as infile:
|
|
442
|
+
content = infile.read()
|
|
443
|
+
print(f"Reading content of file: {filepath}")
|
|
444
|
+
except UnicodeDecodeError:
|
|
445
|
+
with open(filepath, 'rb') as infile:
|
|
446
|
+
content = infile.read().decode('utf-8', errors='ignore')
|
|
447
|
+
print(f"Reading and converting content of file: {filepath}")
|
|
448
|
+
return content
|
|
449
|
+
|
|
450
|
+
def build_project_tree(filepaths: List[Path], base_path: Path) -> str:
|
|
451
|
+
"""
|
|
452
|
+
Build a project structure tree as a string, given a list of filepaths.
|
|
453
|
+
Only includes files actually included in the knowledge base.
|
|
454
|
+
"""
|
|
455
|
+
from collections import defaultdict
|
|
456
|
+
|
|
457
|
+
# Create a tree structure using nested dictionaries
|
|
458
|
+
tree: Dict[str, Any] = {}
|
|
459
|
+
|
|
460
|
+
for path in filepaths:
|
|
461
|
+
rel_parts = Path(path).relative_to(base_path).parts
|
|
462
|
+
current_level = tree
|
|
463
|
+
|
|
464
|
+
# Navigate through the directory structure
|
|
465
|
+
for part in rel_parts[:-1]:
|
|
466
|
+
if part not in current_level:
|
|
467
|
+
current_level[part] = {}
|
|
468
|
+
current_level = current_level[part]
|
|
469
|
+
|
|
470
|
+
# Add the file at the final level
|
|
471
|
+
current_level[rel_parts[-1]] = "FILE"
|
|
472
|
+
|
|
473
|
+
def render(subtree: Dict[str, Any], prefix: str = "") -> List[str]:
|
|
474
|
+
lines = []
|
|
475
|
+
items = sorted(subtree.items())
|
|
476
|
+
|
|
477
|
+
for i, (name, child) in enumerate(items):
|
|
478
|
+
connector = "└── " if i == len(items) - 1 else "├── "
|
|
479
|
+
lines.append(f"{prefix}{connector}{name}")
|
|
480
|
+
|
|
481
|
+
if isinstance(child, dict):
|
|
482
|
+
extension = " " if i == len(items) - 1 else "│ "
|
|
483
|
+
lines.extend(render(child, prefix + extension))
|
|
484
|
+
|
|
485
|
+
return lines
|
|
486
|
+
|
|
487
|
+
return '\n'.join(render(tree))
|
|
488
|
+
|
|
489
|
+
def write_project_tree_and_content(outfile, filepaths: List[Path], base_path: Path, cleaned_notebooks_dir: Path) -> None:
|
|
490
|
+
"""
|
|
491
|
+
Write the project structure tree and then the content for each file.
|
|
492
|
+
"""
|
|
493
|
+
tree_str = build_project_tree(filepaths, base_path)
|
|
494
|
+
outfile.write("Project Structure (files included):\n")
|
|
495
|
+
outfile.write(tree_str + "\n\n")
|
|
496
|
+
|
|
497
|
+
for filepath in filepaths:
|
|
498
|
+
outfile.write(f"File: {filepath}\n")
|
|
499
|
+
outfile.write("="*50 + "\n")
|
|
500
|
+
|
|
501
|
+
if filepath.suffix.lower() == '.ipynb':
|
|
502
|
+
# Use cleaned notebook if available
|
|
503
|
+
cleaned_notebook_path = cleaned_notebooks_dir / filepath.name
|
|
504
|
+
if cleaned_notebook_path.exists():
|
|
505
|
+
with open(cleaned_notebook_path, 'r', encoding='utf-8') as cleanfile:
|
|
506
|
+
content = cleanfile.read()
|
|
507
|
+
outfile.write(content)
|
|
508
|
+
outfile.write("\n" + "="*50 + "\n\n")
|
|
509
|
+
continue
|
|
510
|
+
|
|
511
|
+
content = read_file_contents(filepath)
|
|
512
|
+
outfile.write(content)
|
|
513
|
+
outfile.write("\n" + "="*50 + "\n\n")
|
|
514
|
+
|
|
515
|
+
def generate_full_summary(summarize_subfolder: Path, output_dir: Path) -> None:
|
|
516
|
+
output_file_name = f"{summarize_subfolder.name}_fullrepo.txt"
|
|
517
|
+
output_file_path = output_dir / output_file_name
|
|
518
|
+
print(f"Generating Full Summary: {output_file_path}")
|
|
519
|
+
|
|
520
|
+
cleaned_notebooks_dir = output_dir / "example_notebooks_cleaned"
|
|
521
|
+
|
|
522
|
+
filepaths = []
|
|
523
|
+
for filepath in summarize_subfolder.rglob('*'):
|
|
524
|
+
if should_omit(filepath):
|
|
525
|
+
continue
|
|
526
|
+
if filepath.is_file():
|
|
527
|
+
filepaths.append(filepath)
|
|
528
|
+
|
|
529
|
+
with open(output_file_path, 'w', encoding='utf-8') as outfile:
|
|
530
|
+
write_project_tree_and_content(outfile, filepaths, summarize_subfolder, cleaned_notebooks_dir)
|
|
531
|
+
|
|
532
|
+
print(f"Full summary created at '{output_file_path}'")
|
|
533
|
+
|
|
534
|
+
def main() -> None:
|
|
535
|
+
# Get the name of this script
|
|
536
|
+
this_script = SCRIPT_NAME
|
|
537
|
+
print(f"Script name: {this_script}")
|
|
538
|
+
|
|
539
|
+
# Define the subfolder to summarize (parent of the script's parent)
|
|
540
|
+
summarize_subfolder = Path(__file__).parent.parent
|
|
541
|
+
print(f"Subfolder to summarize: {summarize_subfolder}")
|
|
542
|
+
|
|
543
|
+
# Ensure the output directory exists
|
|
544
|
+
output_dir = ensure_output_dir(Path(__file__).parent)
|
|
545
|
+
|
|
546
|
+
# Delete all existing files in the output directory and its subfolders
|
|
547
|
+
for file in output_dir.rglob('*'):
|
|
548
|
+
if file.is_file():
|
|
549
|
+
file.unlink()
|
|
550
|
+
|
|
551
|
+
# Save cleaned notebooks to a separate subfolder
|
|
552
|
+
save_cleaned_notebooks(summarize_subfolder, output_dir)
|
|
553
|
+
|
|
554
|
+
# Generate the full repository summary
|
|
555
|
+
generate_full_summary(summarize_subfolder, output_dir)
|
|
556
|
+
|
|
557
|
+
print(f"Full repository summary has been generated in '{output_dir}'")
|
|
558
|
+
|
|
559
|
+
if __name__ == "__main__":
|
|
560
|
+
main()
|