ecrawler 0.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ecrawler-0.0.2/LICENSE +21 -0
- ecrawler-0.0.2/PKG-INFO +24 -0
- ecrawler-0.0.2/README.md +9 -0
- ecrawler-0.0.2/ecrawler.egg-info/PKG-INFO +24 -0
- ecrawler-0.0.2/ecrawler.egg-info/SOURCES.txt +12 -0
- ecrawler-0.0.2/ecrawler.egg-info/dependency_links.txt +1 -0
- ecrawler-0.0.2/ecrawler.egg-info/requires.txt +4 -0
- ecrawler-0.0.2/ecrawler.egg-info/top_level.txt +1 -0
- ecrawler-0.0.2/lusofona/pckg/common/base.py +8 -0
- ecrawler-0.0.2/lusofona/pckg/common/tracker.py +636 -0
- ecrawler-0.0.2/lusofona/pckg/statistics/extensions.py +11 -0
- ecrawler-0.0.2/lusofona/pckg/utils/visualizations.py +15 -0
- ecrawler-0.0.2/pyproject.toml +23 -0
- ecrawler-0.0.2/setup.cfg +4 -0
ecrawler-0.0.2/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Gonçalo Cabeleira & João Caldeira
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
ecrawler-0.0.2/PKG-INFO
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ecrawler
|
|
3
|
+
Version: 0.0.2
|
|
4
|
+
Summary: Python package developed at Lusófona University to track notebook execution times and energy consumption.
|
|
5
|
+
Author-email: Gonçalo Cabeleira <goncalocabeleira123@gmail.com>, João Caldeira <joao.caldeira@ulusofona.pt>
|
|
6
|
+
Project-URL: homepage, https://github.com/Streetcode5/ecrawler.git.
|
|
7
|
+
Requires-Python: >=3.12
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: rich
|
|
11
|
+
Requires-Dist: pandas
|
|
12
|
+
Requires-Dist: numpy
|
|
13
|
+
Requires-Dist: codecarbon
|
|
14
|
+
Dynamic: license-file
|
|
15
|
+
|
|
16
|
+
# ecrawler
|
|
17
|
+
|
|
18
|
+
Python package developed at Lusófona University to track Notebook execution times and energy consumption.
|
|
19
|
+
|
|
20
|
+
## Installation
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install ecrawler
|
|
24
|
+
```
|
ecrawler-0.0.2/README.md
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ecrawler
|
|
3
|
+
Version: 0.0.2
|
|
4
|
+
Summary: Python package developed at Lusófona University to track notebook execution times and energy consumption.
|
|
5
|
+
Author-email: Gonçalo Cabeleira <goncalocabeleira123@gmail.com>, João Caldeira <joao.caldeira@ulusofona.pt>
|
|
6
|
+
Project-URL: homepage, https://github.com/Streetcode5/ecrawler.git.
|
|
7
|
+
Requires-Python: >=3.12
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: rich
|
|
11
|
+
Requires-Dist: pandas
|
|
12
|
+
Requires-Dist: numpy
|
|
13
|
+
Requires-Dist: codecarbon
|
|
14
|
+
Dynamic: license-file
|
|
15
|
+
|
|
16
|
+
# ecrawler
|
|
17
|
+
|
|
18
|
+
Python package developed at Lusófona University to track Notebook execution times and energy consumption.
|
|
19
|
+
|
|
20
|
+
## Installation
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install ecrawler
|
|
24
|
+
```
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
ecrawler.egg-info/PKG-INFO
|
|
5
|
+
ecrawler.egg-info/SOURCES.txt
|
|
6
|
+
ecrawler.egg-info/dependency_links.txt
|
|
7
|
+
ecrawler.egg-info/requires.txt
|
|
8
|
+
ecrawler.egg-info/top_level.txt
|
|
9
|
+
lusofona/pckg/common/base.py
|
|
10
|
+
lusofona/pckg/common/tracker.py
|
|
11
|
+
lusofona/pckg/statistics/extensions.py
|
|
12
|
+
lusofona/pckg/utils/visualizations.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
lusofona
|
|
@@ -0,0 +1,636 @@
|
|
|
1
|
+
from codecarbon import OfflineEmissionsTracker, EmissionsTracker
|
|
2
|
+
import json
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import re
|
|
5
|
+
import os
|
|
6
|
+
import numpy as np
|
|
7
|
+
import time
|
|
8
|
+
import psutil
|
|
9
|
+
import cpuinfo
|
|
10
|
+
import ast
|
|
11
|
+
from IPython.display import display
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class Tracker:
|
|
16
|
+
|
|
17
|
+
def __init__(self, experimentid, filename, dataset_path, code_blocks_to_exclude):
|
|
18
|
+
self.tracker = EmissionsTracker()
|
|
19
|
+
self.experimentid = experimentid
|
|
20
|
+
self.final_data = {}
|
|
21
|
+
self.dataset_path = dataset_path
|
|
22
|
+
self.code_blocks_to_exclude = code_blocks_to_exclude
|
|
23
|
+
|
|
24
|
+
self.notebook_path = str(filename)
|
|
25
|
+
|
|
26
|
+
with open(self.notebook_path, 'r', encoding='utf-8') as f:
|
|
27
|
+
self.notebook = json.load(f)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def __estimate_cpu_flops(self):
|
|
31
|
+
# Estimate theoretical peak FLOPS for the CPU.
|
|
32
|
+
cpu_info = psutil.cpu_freq() #cpuinfo.get_cpu_info()
|
|
33
|
+
|
|
34
|
+
#print(cpu_info.keys())
|
|
35
|
+
|
|
36
|
+
# Get number of physical cores
|
|
37
|
+
num_cores = psutil.cpu_count(logical=False)
|
|
38
|
+
|
|
39
|
+
# Get clock speed in Hz
|
|
40
|
+
if cpu_info is not None:
|
|
41
|
+
clock_speed = cpu_info.current #cpu_info['hz_actual'][0] # In Hz
|
|
42
|
+
|
|
43
|
+
# Estimate FLOP per cycle (simplified: assuming 8 for modern CPUs)
|
|
44
|
+
flops_per_cycle = 8 # Conservative estimate (varies by architecture)
|
|
45
|
+
|
|
46
|
+
# Compute estimated FLOPS
|
|
47
|
+
estimated_flops = num_cores * clock_speed * flops_per_cycle
|
|
48
|
+
|
|
49
|
+
# Convert to GFLOPS (billion FLOPS)
|
|
50
|
+
estimated_gflops = estimated_flops / 1e9
|
|
51
|
+
|
|
52
|
+
return f"{estimated_gflops:.2f}"
|
|
53
|
+
|
|
54
|
+
def __measure_actual_flops(self):
|
|
55
|
+
# Measure actual FLOPS using matrix multiplication.
|
|
56
|
+
N = 1000 # Matrix size (increase for better measurement)
|
|
57
|
+
A = np.random.rand(N, N)
|
|
58
|
+
B = np.random.rand(N, N)
|
|
59
|
+
|
|
60
|
+
start_time = time.time()
|
|
61
|
+
C = np.dot(A, B) # Perform matrix multiplication
|
|
62
|
+
end_time = time.time()
|
|
63
|
+
|
|
64
|
+
# Number of floating point operations (approx: 2*N^3)
|
|
65
|
+
num_operations = 2 * (N ** 3)
|
|
66
|
+
|
|
67
|
+
# Time taken
|
|
68
|
+
elapsed_time = end_time - start_time
|
|
69
|
+
|
|
70
|
+
# Compute FLOPS
|
|
71
|
+
flops = num_operations / elapsed_time
|
|
72
|
+
gflops = flops / 1e9 # Convert to GFLOPS
|
|
73
|
+
|
|
74
|
+
return f"{gflops:.2f}"
|
|
75
|
+
|
|
76
|
+
def __ram_measurement(self):
|
|
77
|
+
ram_info = psutil.virtual_memory()
|
|
78
|
+
|
|
79
|
+
total_ram = round(ram_info.total / (1024 ** 3), 2)
|
|
80
|
+
used_ram = round(ram_info.used / (1024 ** 3), 2)
|
|
81
|
+
available_ram = round(ram_info.available / (1024 ** 3), 2)
|
|
82
|
+
ram_usage = ram_info.percent
|
|
83
|
+
|
|
84
|
+
return {"Total_ram":total_ram,
|
|
85
|
+
"Used_ram": used_ram,
|
|
86
|
+
"Avaialble_ram": available_ram,
|
|
87
|
+
"Ram_Usage": ram_usage}
|
|
88
|
+
|
|
89
|
+
def __measure_disk_speed(self):
|
|
90
|
+
test_file = "disk_speed_test.tmp"
|
|
91
|
+
data = os.urandom(100 * 1024 * 1024) # 100MB of random data
|
|
92
|
+
|
|
93
|
+
# Measure write speed
|
|
94
|
+
start_time = time.time()
|
|
95
|
+
with open(test_file, "wb") as f:
|
|
96
|
+
f.write(data)
|
|
97
|
+
write_time = time.time() - start_time
|
|
98
|
+
write_speed = (100 / write_time) if write_time > 0 else 0 # MB/s
|
|
99
|
+
|
|
100
|
+
# Measure read speed
|
|
101
|
+
start_time = time.time()
|
|
102
|
+
with open(test_file, "rb") as f:
|
|
103
|
+
f.read()
|
|
104
|
+
read_time = time.time() - start_time
|
|
105
|
+
read_speed = (100 / read_time) if read_time > 0 else 0 # MB/s
|
|
106
|
+
|
|
107
|
+
# Clean up
|
|
108
|
+
os.remove(test_file)
|
|
109
|
+
|
|
110
|
+
return {"Write_Speed": write_speed,
|
|
111
|
+
"Read_Speed": read_speed}
|
|
112
|
+
|
|
113
|
+
def __count_dataframes_in_notebook(self, notebook):
|
|
114
|
+
|
|
115
|
+
dataframe_patterns = [
|
|
116
|
+
r"(\w+)\s*=\s*pd\.DataFrame\(",
|
|
117
|
+
r"(\w+)\s*=\s*pd\.read_csv\(",
|
|
118
|
+
r"(\w+)\s*=\s*pd\.read_excel\(",
|
|
119
|
+
r"(\w+)\s*=\s*pd\.read_json\(",
|
|
120
|
+
r"(\w+)\s*=\s*pd\.read_parquet\(",
|
|
121
|
+
r"(\w+)\s*=\s*pd\.read_sql\(",
|
|
122
|
+
r"(\w+)\s*=\s*pl\.read_csv\(",
|
|
123
|
+
r"(\w+)\s*=\s*pl\.read_json\("
|
|
124
|
+
]
|
|
125
|
+
|
|
126
|
+
dataframe_count = 0
|
|
127
|
+
|
|
128
|
+
for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
129
|
+
if cell.get("cell_type") == "code":
|
|
130
|
+
code = "".join(cell.get("source", []))
|
|
131
|
+
|
|
132
|
+
for pattern in dataframe_patterns:
|
|
133
|
+
matches = re.findall(pattern, code)
|
|
134
|
+
dataframe_count += len(matches)
|
|
135
|
+
|
|
136
|
+
return dataframe_count
|
|
137
|
+
|
|
138
|
+
def __detect_file_extensions_in_notebook(self, notebook):
|
|
139
|
+
file_extensions = set()
|
|
140
|
+
|
|
141
|
+
for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
142
|
+
if cell.get("cell_type") == "code":
|
|
143
|
+
code = "".join(cell.get("source", []))
|
|
144
|
+
|
|
145
|
+
file_paths = re.findall(r'["\']([^"\']+\.(csv|json|xlsx|parquet|tsv|txt|xml|sql|gz))["\']', code)
|
|
146
|
+
|
|
147
|
+
for file_path_tuple in file_paths:
|
|
148
|
+
file_name = os.path.basename(file_path_tuple[0])
|
|
149
|
+
file_extensions.add(file_name)
|
|
150
|
+
|
|
151
|
+
return file_extensions
|
|
152
|
+
|
|
153
|
+
def __assess_dataframe_structure(self):
|
|
154
|
+
total_rows = 0
|
|
155
|
+
total_columns = 0
|
|
156
|
+
dataset_info = {}
|
|
157
|
+
|
|
158
|
+
count = 0
|
|
159
|
+
|
|
160
|
+
for file_path in self.dataset_path:
|
|
161
|
+
try:
|
|
162
|
+
df = pd.read_csv(file_path, sep=None, engine="python")
|
|
163
|
+
|
|
164
|
+
dataset_name = os.path.basename(file_path)
|
|
165
|
+
|
|
166
|
+
num_rows, num_columns = df.shape
|
|
167
|
+
total_rows += num_rows
|
|
168
|
+
total_columns += num_columns
|
|
169
|
+
|
|
170
|
+
dataset_info[dataset_name] = {
|
|
171
|
+
"rows": num_rows,
|
|
172
|
+
"columns": num_columns,
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
count+=1
|
|
176
|
+
|
|
177
|
+
except Exception as e:
|
|
178
|
+
print(f"Error processing {file_path}: {e}")
|
|
179
|
+
|
|
180
|
+
return {
|
|
181
|
+
"total_rows": total_rows,
|
|
182
|
+
"total_columns": total_columns,
|
|
183
|
+
"datafiles": count
|
|
184
|
+
} if dataset_info else "No datafiles detected or structures are empty."
|
|
185
|
+
|
|
186
|
+
def __detectlib(self, notebook):
|
|
187
|
+
|
|
188
|
+
set_of_lib = set()
|
|
189
|
+
|
|
190
|
+
for cell in notebook.get('cells', [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
191
|
+
if cell.get('cell_type') == 'code':
|
|
192
|
+
for line in cell.get('source', []):
|
|
193
|
+
matches = re.findall(r'^\s*(?:import|from) ([\w\.]+)', line)
|
|
194
|
+
if matches:
|
|
195
|
+
for match in matches:
|
|
196
|
+
lib_name = match.split('.')[0]
|
|
197
|
+
set_of_lib.add(lib_name)
|
|
198
|
+
|
|
199
|
+
return {
|
|
200
|
+
"num_libs": len(set_of_lib),
|
|
201
|
+
"libs": set_of_lib}
|
|
202
|
+
|
|
203
|
+
def __count_imports_in_notebook(self, notebook):
|
|
204
|
+
import_count = 0
|
|
205
|
+
notebook_content = None
|
|
206
|
+
|
|
207
|
+
try:
|
|
208
|
+
if isinstance(notebook, str) and os.path.exists(notebook):
|
|
209
|
+
with open(notebook, 'r', encoding='utf-8') as notebook_file:
|
|
210
|
+
notebook_content = json.load(notebook_file)
|
|
211
|
+
elif isinstance(notebook, dict):
|
|
212
|
+
notebook_content = notebook
|
|
213
|
+
else:
|
|
214
|
+
raise ValueError("Invalid input.")
|
|
215
|
+
|
|
216
|
+
for cell in notebook_content.get('cells', [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
217
|
+
if cell['cell_type'] == 'code':
|
|
218
|
+
for line in cell['source']:
|
|
219
|
+
if re.match(r'^\s*(import|from)\s+', line):
|
|
220
|
+
import_count += 1
|
|
221
|
+
|
|
222
|
+
except Exception as e:
|
|
223
|
+
print(f"Error reading the notebook: {e}")
|
|
224
|
+
|
|
225
|
+
return import_count
|
|
226
|
+
|
|
227
|
+
def __get_kernel_consumption(self):
|
|
228
|
+
try:
|
|
229
|
+
# Get the current process ID (PID)
|
|
230
|
+
current_pid = os.getpid()
|
|
231
|
+
|
|
232
|
+
# Find the kernel process
|
|
233
|
+
kernel_process = None
|
|
234
|
+
for process in psutil.process_iter(attrs=['pid', 'name']):
|
|
235
|
+
try:
|
|
236
|
+
if process.info['pid'] == current_pid or 'python' in process.info['name'].lower():
|
|
237
|
+
kernel_process = process
|
|
238
|
+
break # We assume the first matching process is the kernel
|
|
239
|
+
except (psutil.NoSuchProcess, psutil.AccessDenied, psutil.ZombieProcess):
|
|
240
|
+
continue
|
|
241
|
+
|
|
242
|
+
if kernel_process is None:
|
|
243
|
+
return "Kernel process not found."
|
|
244
|
+
|
|
245
|
+
# Get CPU and memory usage
|
|
246
|
+
cpu_usage = kernel_process.cpu_percent(interval=1) # CPU usage after 1-second measurement
|
|
247
|
+
memory_usage = kernel_process.memory_info().rss / (1024 ** 2) # Convert bytes to MB
|
|
248
|
+
|
|
249
|
+
return {"CPU Usage (%)": cpu_usage, "Memory Usage (MB)": round(memory_usage, 2)}
|
|
250
|
+
|
|
251
|
+
except Exception as e:
|
|
252
|
+
return f"Error retrieving kernel consumption: {e}"
|
|
253
|
+
|
|
254
|
+
def __count_algorithms_in_notebook(self, notebook):
|
|
255
|
+
try:
|
|
256
|
+
with open(notebook, 'r', encoding='utf-8') as notebook_file:
|
|
257
|
+
notebook_content = json.load(notebook_file)
|
|
258
|
+
|
|
259
|
+
# Patterns to detect algorithms
|
|
260
|
+
function_def_pattern = r"^\s*def\s+\w+\s*\("
|
|
261
|
+
loop_pattern = r"^\s*(for|while)\s+"
|
|
262
|
+
function_call_pattern = r"^\s*\w+\s*\(.*\)" # Simplified function call pattern
|
|
263
|
+
|
|
264
|
+
function_count = 0
|
|
265
|
+
loop_count = 0
|
|
266
|
+
function_call_count = 0
|
|
267
|
+
|
|
268
|
+
for cell in notebook_content.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
269
|
+
if cell.get("cell_type") == "code":
|
|
270
|
+
for line in cell.get("source", []):
|
|
271
|
+
if re.match(function_def_pattern, line):
|
|
272
|
+
function_count += 1
|
|
273
|
+
elif re.match(loop_pattern, line):
|
|
274
|
+
loop_count += 1
|
|
275
|
+
elif re.match(function_call_pattern, line) and not re.match(function_def_pattern, line):
|
|
276
|
+
function_call_count += 1
|
|
277
|
+
|
|
278
|
+
total_algorithms = function_count + loop_count + function_call_count
|
|
279
|
+
return {
|
|
280
|
+
"Function Definitions": function_count,
|
|
281
|
+
"Loops": loop_count,
|
|
282
|
+
"Function Calls": function_call_count,
|
|
283
|
+
"Total Algorithms": total_algorithms
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
except Exception as e:
|
|
287
|
+
return {
|
|
288
|
+
"Function Definitions": 0,
|
|
289
|
+
"Loops": 0,
|
|
290
|
+
"Function Calls": 0,
|
|
291
|
+
"Total Algorithms": 0,
|
|
292
|
+
"Error": str(e)
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
def __count_classification_algorithms(self, notebook):
|
|
296
|
+
# Scans a Jupyter notebook for classification algorithms and counts them.
|
|
297
|
+
|
|
298
|
+
# It detects classifiers from libraries such as:
|
|
299
|
+
# - Scikit-learn
|
|
300
|
+
# - XGBoost
|
|
301
|
+
# - LightGBM
|
|
302
|
+
# - TensorFlow/Keras
|
|
303
|
+
# - PyTorch
|
|
304
|
+
|
|
305
|
+
# List of common classification algorithms
|
|
306
|
+
classification_keywords = [
|
|
307
|
+
# Scikit-learn classifiers
|
|
308
|
+
r"\bLogisticRegression\(", r"\bRandomForestClassifier\(", r"\bSVC\(",
|
|
309
|
+
r"\bDecisionTreeClassifier\(", r"\bKNeighborsClassifier\(", r"\bGaussianNB\(",
|
|
310
|
+
r"\bGradientBoostingClassifier\(", r"\bAdaBoostClassifier\(", r"\bMLPClassifier\(",
|
|
311
|
+
r"\bHistGradientBoostingClassifier\(", r"\bExtraTreesClassifier\(",
|
|
312
|
+
|
|
313
|
+
# XGBoost
|
|
314
|
+
r"\bXGBClassifier\(",
|
|
315
|
+
|
|
316
|
+
# LightGBM
|
|
317
|
+
r"\bLGBMClassifier\(",
|
|
318
|
+
|
|
319
|
+
# TensorFlow/Keras models
|
|
320
|
+
r"\bSequential\(", r"\bDense\(", r"\bConv2D\(", r"\bLSTM\(",
|
|
321
|
+
|
|
322
|
+
# PyTorch classifiers
|
|
323
|
+
r"\btorch\.nn\.Linear\(", r"\btorch\.nn\.Conv2d\(", r"\btorch\.nn\.LSTM\("
|
|
324
|
+
]
|
|
325
|
+
|
|
326
|
+
# Load notebook content
|
|
327
|
+
try:
|
|
328
|
+
with open(notebook, 'r', encoding='utf-8') as notebook_file:
|
|
329
|
+
notebook_content = json.load(notebook_file)
|
|
330
|
+
|
|
331
|
+
classification_count = 0
|
|
332
|
+
|
|
333
|
+
# Scan each cell in the notebook
|
|
334
|
+
for cell in notebook_content.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
335
|
+
if cell.get("cell_type") == "code":
|
|
336
|
+
code = "".join(cell.get("source", [])) # Get full cell code
|
|
337
|
+
|
|
338
|
+
# Check for classifier patterns
|
|
339
|
+
for pattern in classification_keywords:
|
|
340
|
+
matches = re.findall(pattern, code)
|
|
341
|
+
classification_count += len(matches)
|
|
342
|
+
|
|
343
|
+
return classification_count
|
|
344
|
+
|
|
345
|
+
except Exception as e:
|
|
346
|
+
return {"Error": f"Error reading the notebook: {e}"}
|
|
347
|
+
|
|
348
|
+
def __count_regression_algorithms(self, notebook):
|
|
349
|
+
# Scans a Jupyter notebook for regression algorithms and counts them.
|
|
350
|
+
|
|
351
|
+
# It detects regressors from:
|
|
352
|
+
# - Scikit-learn
|
|
353
|
+
# - XGBoost
|
|
354
|
+
# - LightGBM
|
|
355
|
+
# - TensorFlow/Keras
|
|
356
|
+
# - PyTorch
|
|
357
|
+
|
|
358
|
+
# List of common regression algorithms
|
|
359
|
+
regression_keywords = [
|
|
360
|
+
# Scikit-learn regressors
|
|
361
|
+
r"\bLinearRegression\(", r"\bRidge\(", r"\bLasso\(",
|
|
362
|
+
r"\bElasticNet\(", r"\bSVR\(", r"\bDecisionTreeRegressor\(",
|
|
363
|
+
r"\bRandomForestRegressor\(", r"\bGradientBoostingRegressor\(",
|
|
364
|
+
r"\bAdaBoostRegressor\(", r"\bKNeighborsRegressor\(",
|
|
365
|
+
r"\bMLPRegressor\(", r"\bExtraTreesRegressor\(",
|
|
366
|
+
r"\bHistGradientBoostingRegressor\(",
|
|
367
|
+
|
|
368
|
+
# XGBoost
|
|
369
|
+
r"\bXGBRegressor\(",
|
|
370
|
+
|
|
371
|
+
# LightGBM
|
|
372
|
+
r"\bLGBMRegressor\(",
|
|
373
|
+
|
|
374
|
+
# TensorFlow/Keras models
|
|
375
|
+
r"\bSequential\(", r"\bDense\(",
|
|
376
|
+
|
|
377
|
+
# PyTorch models
|
|
378
|
+
r"\btorch\.nn\.Linear\(", r"\btorch\.nn\.Conv2d\("
|
|
379
|
+
]
|
|
380
|
+
|
|
381
|
+
# Load notebook content
|
|
382
|
+
try:
|
|
383
|
+
with open(notebook, 'r', encoding='utf-8') as notebook_file:
|
|
384
|
+
notebook_content = json.load(notebook_file)
|
|
385
|
+
|
|
386
|
+
regression_count = 0
|
|
387
|
+
|
|
388
|
+
# Scan each cell in the notebook
|
|
389
|
+
for cell in notebook_content.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
390
|
+
if cell.get("cell_type") == "code":
|
|
391
|
+
code = "".join(cell.get("source", [])) # Get full cell code
|
|
392
|
+
|
|
393
|
+
# Check for regression patterns
|
|
394
|
+
for pattern in regression_keywords:
|
|
395
|
+
matches = re.findall(pattern, code)
|
|
396
|
+
regression_count += len(matches)
|
|
397
|
+
|
|
398
|
+
return regression_count
|
|
399
|
+
|
|
400
|
+
except Exception as e:
|
|
401
|
+
return {"Error": f"Error reading the notebook: {e}"}
|
|
402
|
+
|
|
403
|
+
def __compute_henry_kafura_metrics(self):
|
|
404
|
+
function_calls = {}
|
|
405
|
+
global_calls = set()
|
|
406
|
+
|
|
407
|
+
for cell in self.notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
408
|
+
if cell.get("cell_type") == "code":
|
|
409
|
+
code = "".join(cell.get("source", []))
|
|
410
|
+
try:
|
|
411
|
+
tree = ast.parse(code)
|
|
412
|
+
except SyntaxError:
|
|
413
|
+
continue
|
|
414
|
+
|
|
415
|
+
for node in ast.iter_child_nodes(tree):
|
|
416
|
+
# Track function definitions and their internal calls
|
|
417
|
+
if isinstance(node, ast.FunctionDef):
|
|
418
|
+
func_name = node.name
|
|
419
|
+
function_calls[func_name] = set()
|
|
420
|
+
|
|
421
|
+
for child in ast.walk(node):
|
|
422
|
+
if isinstance(child, ast.Call):
|
|
423
|
+
if isinstance(child.func, ast.Name):
|
|
424
|
+
function_calls[func_name].add(child.func.id)
|
|
425
|
+
else:
|
|
426
|
+
# Also track top-level (global scope) function calls
|
|
427
|
+
for child in ast.walk(node):
|
|
428
|
+
if isinstance(child, ast.Call) and isinstance(child.func, ast.Name):
|
|
429
|
+
global_calls.add(child.func.id)
|
|
430
|
+
|
|
431
|
+
# If nothing was found
|
|
432
|
+
if not function_calls and not global_calls:
|
|
433
|
+
return "Error: Could not retrieve function call data."
|
|
434
|
+
|
|
435
|
+
# Add global scope calls as a synthetic "global" function
|
|
436
|
+
if global_calls:
|
|
437
|
+
function_calls["_global_scope_"] = global_calls
|
|
438
|
+
|
|
439
|
+
# Compute Henry-Kafura metrics
|
|
440
|
+
complexity_metrics = {}
|
|
441
|
+
for function, calls in function_calls.items():
|
|
442
|
+
F_in = sum(1 for funcs in function_calls.values() if function in funcs)
|
|
443
|
+
F_out = len(calls)
|
|
444
|
+
complexity_metrics[function] = {
|
|
445
|
+
"F_in": F_in,
|
|
446
|
+
"F_out": F_out,
|
|
447
|
+
"Complexity": F_in * (F_out ** 2)
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
return complexity_metrics
|
|
451
|
+
|
|
452
|
+
def __cyclomatic_complexity_ast(self, tree):
|
|
453
|
+
# Computes Cyclomatic Complexity (CC) from an AST tree.
|
|
454
|
+
nodes = 1 # Start with 1 (entry point)
|
|
455
|
+
edges = 0
|
|
456
|
+
|
|
457
|
+
for node in ast.walk(tree):
|
|
458
|
+
if isinstance(node, (ast.If, ast.While, ast.For, ast.Try, ast.With)):
|
|
459
|
+
nodes += 1
|
|
460
|
+
edges += 2 # Each control structure introduces a new path
|
|
461
|
+
elif isinstance(node, ast.IfExp): # Ternary operator (e.g., x if cond else y)
|
|
462
|
+
nodes += 1
|
|
463
|
+
edges += 2
|
|
464
|
+
elif isinstance(node, ast.BinOp) and isinstance(node.op, (ast.And, ast.Or)):
|
|
465
|
+
edges += 1 # Boolean short-circuiting increases complexity
|
|
466
|
+
|
|
467
|
+
complexity = edges - nodes + 2
|
|
468
|
+
return max(complexity, 1) # CC is at least 1
|
|
469
|
+
|
|
470
|
+
def __compute_cyclomatic_complexity(self, notebook_path):
|
|
471
|
+
# Computes total Cyclomatic Complexity across code cells in the notebook.
|
|
472
|
+
with open(notebook_path, 'r', encoding='utf-8') as f:
|
|
473
|
+
notebook = json.load(f)
|
|
474
|
+
|
|
475
|
+
total_complexity = 0
|
|
476
|
+
|
|
477
|
+
for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
478
|
+
if cell["cell_type"] == "code":
|
|
479
|
+
code = "".join(cell["source"])
|
|
480
|
+
try:
|
|
481
|
+
tree = ast.parse(code)
|
|
482
|
+
total_complexity += self.__cyclomatic_complexity_ast(tree)
|
|
483
|
+
except SyntaxError:
|
|
484
|
+
continue # skip cells with invalid syntax
|
|
485
|
+
|
|
486
|
+
return total_complexity
|
|
487
|
+
|
|
488
|
+
def __count_executed_lines_of_code(self, notebook_path):
|
|
489
|
+
with open(notebook_path, 'r', encoding='utf-8') as f:
|
|
490
|
+
notebook = json.load(f)
|
|
491
|
+
|
|
492
|
+
total_lines = 0
|
|
493
|
+
|
|
494
|
+
for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
495
|
+
if cell["cell_type"] == "code":
|
|
496
|
+
code_lines = cell["source"]
|
|
497
|
+
executable_lines = [
|
|
498
|
+
line for line in code_lines
|
|
499
|
+
if line.strip() and not line.strip().startswith("#") # Ignore empty lines and comments
|
|
500
|
+
]
|
|
501
|
+
total_lines += len(executable_lines)
|
|
502
|
+
|
|
503
|
+
return total_lines
|
|
504
|
+
|
|
505
|
+
|
|
506
|
+
def __count_functions_in_notebook(self, notebook_path):
|
|
507
|
+
with open(notebook_path, 'r', encoding='utf-8') as f:
|
|
508
|
+
notebook = json.load(f)
|
|
509
|
+
|
|
510
|
+
function_count = 0
|
|
511
|
+
|
|
512
|
+
for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
|
|
513
|
+
if cell["cell_type"] == "code": # Only analyze code cells
|
|
514
|
+
code = "".join(cell["source"]) # Get full code as a string
|
|
515
|
+
try:
|
|
516
|
+
tree = ast.parse(code) # Parse code into an AST
|
|
517
|
+
function_count += sum(isinstance(node, ast.FunctionDef) for node in ast.walk(tree))
|
|
518
|
+
except SyntaxError:
|
|
519
|
+
pass # Skip cells with invalid Python syntax
|
|
520
|
+
|
|
521
|
+
return function_count
|
|
522
|
+
|
|
523
|
+
def hello( self, name ):
|
|
524
|
+
print( "Hello, " + name + "!. Welcome from Tracker package." )
|
|
525
|
+
|
|
526
|
+
def version(self):
|
|
527
|
+
from importlib.metadata import version
|
|
528
|
+
return "lusofona-pckg v. " + version('lusofona-pckg')
|
|
529
|
+
|
|
530
|
+
def start(self):
|
|
531
|
+
self.tracker.start()
|
|
532
|
+
|
|
533
|
+
def collect(self):
|
|
534
|
+
|
|
535
|
+
ram_data = self.__ram_measurement()
|
|
536
|
+
|
|
537
|
+
disk_speed = self.__measure_disk_speed()
|
|
538
|
+
|
|
539
|
+
dataframe_structure = self.__assess_dataframe_structure()
|
|
540
|
+
|
|
541
|
+
libraries = self.__detectlib(self.notebook)
|
|
542
|
+
|
|
543
|
+
algorithms = self.__count_algorithms_in_notebook(self.notebook_path)
|
|
544
|
+
|
|
545
|
+
code_carbon_data = json.loads(self.tracker.final_emissions_data.toJSON())
|
|
546
|
+
|
|
547
|
+
hk_metrics = self.__compute_henry_kafura_metrics()
|
|
548
|
+
|
|
549
|
+
cc_plus_data = {'cpu_flops': float(self.__estimate_cpu_flops()),
|
|
550
|
+
'actual_flops': float(self.__measure_actual_flops()),
|
|
551
|
+
'total_ram': ram_data["Total_ram"],
|
|
552
|
+
'used_ram': ram_data["Used_ram"],
|
|
553
|
+
'available_ram': ram_data["Avaialble_ram"],
|
|
554
|
+
'ram_usage': ram_data["Ram_Usage"],
|
|
555
|
+
'write_speed': disk_speed["Write_Speed"],
|
|
556
|
+
'read_speed': disk_speed["Read_Speed"],
|
|
557
|
+
'dataframe_count': self.__count_dataframes_in_notebook(self.notebook),
|
|
558
|
+
'file_extensions': self.__detect_file_extensions_in_notebook(self.notebook),
|
|
559
|
+
'total_rows': dataframe_structure["total_rows"],
|
|
560
|
+
'total_columns': dataframe_structure["total_columns"],
|
|
561
|
+
'datafiles': dataframe_structure["datafiles"],
|
|
562
|
+
'number_of_libraries': libraries['num_libs'],
|
|
563
|
+
'library_names': libraries['libs'],
|
|
564
|
+
'import_count': self.__count_imports_in_notebook(self.notebook),
|
|
565
|
+
'kernel_consumption': self.__get_kernel_consumption(),
|
|
566
|
+
'function_definitions': algorithms['Function Definitions'],
|
|
567
|
+
'loops': algorithms['Loops'],
|
|
568
|
+
'function_calls': algorithms['Function Calls'],
|
|
569
|
+
'total_algorithms': algorithms['Total Algorithms'],
|
|
570
|
+
'classification_algorithms': self.__count_classification_algorithms(self.notebook_path),
|
|
571
|
+
'regression_algorithms': self.__count_regression_algorithms(self.notebook_path),
|
|
572
|
+
'F_in': hk_metrics['_global_scope_']['F_in'],
|
|
573
|
+
'F_out': hk_metrics['_global_scope_']['F_out'],
|
|
574
|
+
'Complexity': hk_metrics['_global_scope_']['Complexity'],
|
|
575
|
+
'cyclomatic_complexity': self.__compute_cyclomatic_complexity(self.notebook_path),
|
|
576
|
+
'number_of_lines': self.__count_executed_lines_of_code(self.notebook_path),
|
|
577
|
+
'number_of_functions': self.__count_functions_in_notebook(self.notebook_path),
|
|
578
|
+
'region': code_carbon_data["region"],
|
|
579
|
+
'carbon_emissions_kg': code_carbon_data["emissions"],
|
|
580
|
+
'energy_consumed_kwh': code_carbon_data["energy_consumed"],
|
|
581
|
+
'duration_s': code_carbon_data["duration"],
|
|
582
|
+
'cpu_power_watt': code_carbon_data["cpu_power"],
|
|
583
|
+
'ram_power_watt': code_carbon_data["ram_power"],
|
|
584
|
+
'timestamp': code_carbon_data.get("timestamp"),
|
|
585
|
+
'project_name': "etracker", #code_carbon_data.get("project_name"),
|
|
586
|
+
'run_id': code_carbon_data.get("run_id"),
|
|
587
|
+
'experiment_id': self.experimentid, #code_carbon_data.get("experiment_id"),
|
|
588
|
+
'duration_s': code_carbon_data.get("duration"),
|
|
589
|
+
'carbon_emissions_kg': code_carbon_data.get("emissions"),
|
|
590
|
+
'emissions_rate': code_carbon_data.get("emissions_rate"),
|
|
591
|
+
'cpu_power_watt': code_carbon_data.get("cpu_power"),
|
|
592
|
+
'gpu_power_watt': code_carbon_data.get("gpu_power"),
|
|
593
|
+
'ram_power_watt': code_carbon_data.get("ram_power"),
|
|
594
|
+
'cpu_energy': code_carbon_data.get("cpu_energy"),
|
|
595
|
+
'gpu_energy': code_carbon_data.get("gpu_energy"),
|
|
596
|
+
'ram_energy': code_carbon_data.get("ram_energy"),
|
|
597
|
+
'energy_consumed_kwh': code_carbon_data.get("energy_consumed"),
|
|
598
|
+
'country_name': code_carbon_data.get("country_name"),
|
|
599
|
+
'country_iso_code': code_carbon_data.get("country_iso_code"),
|
|
600
|
+
'region': code_carbon_data.get("region"),
|
|
601
|
+
'cloud_provider': code_carbon_data.get("cloud_provider"),
|
|
602
|
+
'cloud_region': code_carbon_data.get("cloud_region"),
|
|
603
|
+
'os': code_carbon_data.get("os"),
|
|
604
|
+
'python_version': code_carbon_data.get("python_version"),
|
|
605
|
+
'codecarbon_version': code_carbon_data.get("codecarbon_version"),
|
|
606
|
+
'cpu_count': code_carbon_data.get("cpu_count"),
|
|
607
|
+
'cpu_model': code_carbon_data.get("cpu_model"),
|
|
608
|
+
'gpu_count': code_carbon_data.get("gpu_count"),
|
|
609
|
+
'gpu_model': code_carbon_data.get("gpu_model"),
|
|
610
|
+
'longitude': code_carbon_data.get("longitude"),
|
|
611
|
+
'latitude': code_carbon_data.get("latitude"),
|
|
612
|
+
'ram_total_size': code_carbon_data.get("ram_total_size"),
|
|
613
|
+
'tracking_mode': code_carbon_data.get("tracking_mode"),
|
|
614
|
+
'on_cloud': code_carbon_data.get("on_cloud"),
|
|
615
|
+
'pue': code_carbon_data.get("pue"),
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
self.final_data = {**cc_plus_data}
|
|
619
|
+
|
|
620
|
+
# write data to csv file if it does not exist , otherwise append to it
|
|
621
|
+
|
|
622
|
+
df = pd.DataFrame([self.final_data])
|
|
623
|
+
output_file = f"tracker_output.csv"
|
|
624
|
+
if not os.path.exists(output_file):
|
|
625
|
+
df.to_csv(output_file, index=False)
|
|
626
|
+
else:
|
|
627
|
+
df.to_csv(output_file, mode='a', header=False, index=False)
|
|
628
|
+
|
|
629
|
+
|
|
630
|
+
def stop(self):
|
|
631
|
+
self.tracker.stop()
|
|
632
|
+
|
|
633
|
+
self.collect()
|
|
634
|
+
|
|
635
|
+
def view(self):
|
|
636
|
+
display(self.final_data)
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
|
|
2
|
+
#import pandas as pd
|
|
3
|
+
|
|
4
|
+
# Adds variance, skewness and kurtosis to the describe() method
|
|
5
|
+
def extended_describe( df ):
|
|
6
|
+
numerical = df.select_dtypes( exclude=[ 'object', 'category', 'datetime64[ns]' ] )
|
|
7
|
+
extended_df = df.describe( include='all' ).T
|
|
8
|
+
extended_df['var'] = numerical.var()
|
|
9
|
+
extended_df['skew'] = numerical.skew()
|
|
10
|
+
extended_df['kurt'] = numerical.kurt()
|
|
11
|
+
return extended_df
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
|
|
2
|
+
from IPython.display import display, HTML
|
|
3
|
+
|
|
4
|
+
# Plot dataframes side by side to avoid scrolling
|
|
5
|
+
def plot_dataframes_side_by_side( dfs:list, captions:list, tablespacing=5 ):
|
|
6
|
+
"""Display tables side by side to save vertical space
|
|
7
|
+
Input:
|
|
8
|
+
dfs: list of pandas.DataFrame
|
|
9
|
+
captions: list of table captions
|
|
10
|
+
"""
|
|
11
|
+
output = ""
|
|
12
|
+
for (caption, df) in zip(captions, dfs):
|
|
13
|
+
output += df.style.set_table_attributes("style='display:inline'").set_caption(caption)._repr_html_()
|
|
14
|
+
output += tablespacing * "\xa0"
|
|
15
|
+
display(HTML(output))
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77.0.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "ecrawler"
|
|
7
|
+
version = "0.0.2"
|
|
8
|
+
description = "Python package developed at Lusófona University to track notebook execution times and energy consumption."
|
|
9
|
+
|
|
10
|
+
readme = "README.md"
|
|
11
|
+
requires-python = ">=3.12"
|
|
12
|
+
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "Gonçalo Cabeleira", email = "goncalocabeleira123@gmail.com" },
|
|
15
|
+
{ name = "João Caldeira", email = "joao.caldeira@ulusofona.pt" },
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
license-files = ["LICENSE"]
|
|
19
|
+
|
|
20
|
+
dependencies = ["rich", "pandas", "numpy", "codecarbon"]
|
|
21
|
+
|
|
22
|
+
[project.urls]
|
|
23
|
+
homepage = "https://github.com/Streetcode5/ecrawler.git."
|
ecrawler-0.0.2/setup.cfg
ADDED