ecrawler 0.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ecrawler-0.0.2/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Gonçalo Cabeleira & João Caldeira
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,24 @@
1
+ Metadata-Version: 2.4
2
+ Name: ecrawler
3
+ Version: 0.0.2
4
+ Summary: Python package developed at Lusófona University to track notebook execution times and energy consumption.
5
+ Author-email: Gonçalo Cabeleira <goncalocabeleira123@gmail.com>, João Caldeira <joao.caldeira@ulusofona.pt>
6
+ Project-URL: homepage, https://github.com/Streetcode5/ecrawler.git.
7
+ Requires-Python: >=3.12
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: rich
11
+ Requires-Dist: pandas
12
+ Requires-Dist: numpy
13
+ Requires-Dist: codecarbon
14
+ Dynamic: license-file
15
+
16
+ # ecrawler
17
+
18
+ Python package developed at Lusófona University to track Notebook execution times and energy consumption.
19
+
20
+ ## Installation
21
+
22
+ ```bash
23
+ pip install ecrawler
24
+ ```
@@ -0,0 +1,9 @@
1
+ # ecrawler
2
+
3
+ Python package developed at Lusófona University to track Notebook execution times and energy consumption.
4
+
5
+ ## Installation
6
+
7
+ ```bash
8
+ pip install ecrawler
9
+ ```
@@ -0,0 +1,24 @@
1
+ Metadata-Version: 2.4
2
+ Name: ecrawler
3
+ Version: 0.0.2
4
+ Summary: Python package developed at Lusófona University to track notebook execution times and energy consumption.
5
+ Author-email: Gonçalo Cabeleira <goncalocabeleira123@gmail.com>, João Caldeira <joao.caldeira@ulusofona.pt>
6
+ Project-URL: homepage, https://github.com/Streetcode5/ecrawler.git.
7
+ Requires-Python: >=3.12
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: rich
11
+ Requires-Dist: pandas
12
+ Requires-Dist: numpy
13
+ Requires-Dist: codecarbon
14
+ Dynamic: license-file
15
+
16
+ # ecrawler
17
+
18
+ Python package developed at Lusófona University to track Notebook execution times and energy consumption.
19
+
20
+ ## Installation
21
+
22
+ ```bash
23
+ pip install ecrawler
24
+ ```
@@ -0,0 +1,12 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ ecrawler.egg-info/PKG-INFO
5
+ ecrawler.egg-info/SOURCES.txt
6
+ ecrawler.egg-info/dependency_links.txt
7
+ ecrawler.egg-info/requires.txt
8
+ ecrawler.egg-info/top_level.txt
9
+ lusofona/pckg/common/base.py
10
+ lusofona/pckg/common/tracker.py
11
+ lusofona/pckg/statistics/extensions.py
12
+ lusofona/pckg/utils/visualizations.py
@@ -0,0 +1,4 @@
1
+ rich
2
+ pandas
3
+ numpy
4
+ codecarbon
@@ -0,0 +1 @@
1
+ lusofona
@@ -0,0 +1,8 @@
1
+ # File with most basic functions
2
+
3
+ def hello( name ):
4
+ print( "Hello, " + name + "!. Welcome to a Python Package Creation Example." )
5
+
6
+ def version():
7
+ from importlib.metadata import version
8
+ return "lusofona-pckg v. " + version('lusofona-pckg')
@@ -0,0 +1,636 @@
1
+ from codecarbon import OfflineEmissionsTracker, EmissionsTracker
2
+ import json
3
+ import pandas as pd
4
+ import re
5
+ import os
6
+ import numpy as np
7
+ import time
8
+ import psutil
9
+ import cpuinfo
10
+ import ast
11
+ from IPython.display import display
12
+
13
+
14
+
15
+ class Tracker:
16
+
17
+ def __init__(self, experimentid, filename, dataset_path, code_blocks_to_exclude):
18
+ self.tracker = EmissionsTracker()
19
+ self.experimentid = experimentid
20
+ self.final_data = {}
21
+ self.dataset_path = dataset_path
22
+ self.code_blocks_to_exclude = code_blocks_to_exclude
23
+
24
+ self.notebook_path = str(filename)
25
+
26
+ with open(self.notebook_path, 'r', encoding='utf-8') as f:
27
+ self.notebook = json.load(f)
28
+
29
+
30
+ def __estimate_cpu_flops(self):
31
+ # Estimate theoretical peak FLOPS for the CPU.
32
+ cpu_info = psutil.cpu_freq() #cpuinfo.get_cpu_info()
33
+
34
+ #print(cpu_info.keys())
35
+
36
+ # Get number of physical cores
37
+ num_cores = psutil.cpu_count(logical=False)
38
+
39
+ # Get clock speed in Hz
40
+ if cpu_info is not None:
41
+ clock_speed = cpu_info.current #cpu_info['hz_actual'][0] # In Hz
42
+
43
+ # Estimate FLOP per cycle (simplified: assuming 8 for modern CPUs)
44
+ flops_per_cycle = 8 # Conservative estimate (varies by architecture)
45
+
46
+ # Compute estimated FLOPS
47
+ estimated_flops = num_cores * clock_speed * flops_per_cycle
48
+
49
+ # Convert to GFLOPS (billion FLOPS)
50
+ estimated_gflops = estimated_flops / 1e9
51
+
52
+ return f"{estimated_gflops:.2f}"
53
+
54
+ def __measure_actual_flops(self):
55
+ # Measure actual FLOPS using matrix multiplication.
56
+ N = 1000 # Matrix size (increase for better measurement)
57
+ A = np.random.rand(N, N)
58
+ B = np.random.rand(N, N)
59
+
60
+ start_time = time.time()
61
+ C = np.dot(A, B) # Perform matrix multiplication
62
+ end_time = time.time()
63
+
64
+ # Number of floating point operations (approx: 2*N^3)
65
+ num_operations = 2 * (N ** 3)
66
+
67
+ # Time taken
68
+ elapsed_time = end_time - start_time
69
+
70
+ # Compute FLOPS
71
+ flops = num_operations / elapsed_time
72
+ gflops = flops / 1e9 # Convert to GFLOPS
73
+
74
+ return f"{gflops:.2f}"
75
+
76
+ def __ram_measurement(self):
77
+ ram_info = psutil.virtual_memory()
78
+
79
+ total_ram = round(ram_info.total / (1024 ** 3), 2)
80
+ used_ram = round(ram_info.used / (1024 ** 3), 2)
81
+ available_ram = round(ram_info.available / (1024 ** 3), 2)
82
+ ram_usage = ram_info.percent
83
+
84
+ return {"Total_ram":total_ram,
85
+ "Used_ram": used_ram,
86
+ "Avaialble_ram": available_ram,
87
+ "Ram_Usage": ram_usage}
88
+
89
+ def __measure_disk_speed(self):
90
+ test_file = "disk_speed_test.tmp"
91
+ data = os.urandom(100 * 1024 * 1024) # 100MB of random data
92
+
93
+ # Measure write speed
94
+ start_time = time.time()
95
+ with open(test_file, "wb") as f:
96
+ f.write(data)
97
+ write_time = time.time() - start_time
98
+ write_speed = (100 / write_time) if write_time > 0 else 0 # MB/s
99
+
100
+ # Measure read speed
101
+ start_time = time.time()
102
+ with open(test_file, "rb") as f:
103
+ f.read()
104
+ read_time = time.time() - start_time
105
+ read_speed = (100 / read_time) if read_time > 0 else 0 # MB/s
106
+
107
+ # Clean up
108
+ os.remove(test_file)
109
+
110
+ return {"Write_Speed": write_speed,
111
+ "Read_Speed": read_speed}
112
+
113
+ def __count_dataframes_in_notebook(self, notebook):
114
+
115
+ dataframe_patterns = [
116
+ r"(\w+)\s*=\s*pd\.DataFrame\(",
117
+ r"(\w+)\s*=\s*pd\.read_csv\(",
118
+ r"(\w+)\s*=\s*pd\.read_excel\(",
119
+ r"(\w+)\s*=\s*pd\.read_json\(",
120
+ r"(\w+)\s*=\s*pd\.read_parquet\(",
121
+ r"(\w+)\s*=\s*pd\.read_sql\(",
122
+ r"(\w+)\s*=\s*pl\.read_csv\(",
123
+ r"(\w+)\s*=\s*pl\.read_json\("
124
+ ]
125
+
126
+ dataframe_count = 0
127
+
128
+ for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
129
+ if cell.get("cell_type") == "code":
130
+ code = "".join(cell.get("source", []))
131
+
132
+ for pattern in dataframe_patterns:
133
+ matches = re.findall(pattern, code)
134
+ dataframe_count += len(matches)
135
+
136
+ return dataframe_count
137
+
138
+ def __detect_file_extensions_in_notebook(self, notebook):
139
+ file_extensions = set()
140
+
141
+ for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
142
+ if cell.get("cell_type") == "code":
143
+ code = "".join(cell.get("source", []))
144
+
145
+ file_paths = re.findall(r'["\']([^"\']+\.(csv|json|xlsx|parquet|tsv|txt|xml|sql|gz))["\']', code)
146
+
147
+ for file_path_tuple in file_paths:
148
+ file_name = os.path.basename(file_path_tuple[0])
149
+ file_extensions.add(file_name)
150
+
151
+ return file_extensions
152
+
153
+ def __assess_dataframe_structure(self):
154
+ total_rows = 0
155
+ total_columns = 0
156
+ dataset_info = {}
157
+
158
+ count = 0
159
+
160
+ for file_path in self.dataset_path:
161
+ try:
162
+ df = pd.read_csv(file_path, sep=None, engine="python")
163
+
164
+ dataset_name = os.path.basename(file_path)
165
+
166
+ num_rows, num_columns = df.shape
167
+ total_rows += num_rows
168
+ total_columns += num_columns
169
+
170
+ dataset_info[dataset_name] = {
171
+ "rows": num_rows,
172
+ "columns": num_columns,
173
+ }
174
+
175
+ count+=1
176
+
177
+ except Exception as e:
178
+ print(f"Error processing {file_path}: {e}")
179
+
180
+ return {
181
+ "total_rows": total_rows,
182
+ "total_columns": total_columns,
183
+ "datafiles": count
184
+ } if dataset_info else "No datafiles detected or structures are empty."
185
+
186
+ def __detectlib(self, notebook):
187
+
188
+ set_of_lib = set()
189
+
190
+ for cell in notebook.get('cells', [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
191
+ if cell.get('cell_type') == 'code':
192
+ for line in cell.get('source', []):
193
+ matches = re.findall(r'^\s*(?:import|from) ([\w\.]+)', line)
194
+ if matches:
195
+ for match in matches:
196
+ lib_name = match.split('.')[0]
197
+ set_of_lib.add(lib_name)
198
+
199
+ return {
200
+ "num_libs": len(set_of_lib),
201
+ "libs": set_of_lib}
202
+
203
+ def __count_imports_in_notebook(self, notebook):
204
+ import_count = 0
205
+ notebook_content = None
206
+
207
+ try:
208
+ if isinstance(notebook, str) and os.path.exists(notebook):
209
+ with open(notebook, 'r', encoding='utf-8') as notebook_file:
210
+ notebook_content = json.load(notebook_file)
211
+ elif isinstance(notebook, dict):
212
+ notebook_content = notebook
213
+ else:
214
+ raise ValueError("Invalid input.")
215
+
216
+ for cell in notebook_content.get('cells', [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
217
+ if cell['cell_type'] == 'code':
218
+ for line in cell['source']:
219
+ if re.match(r'^\s*(import|from)\s+', line):
220
+ import_count += 1
221
+
222
+ except Exception as e:
223
+ print(f"Error reading the notebook: {e}")
224
+
225
+ return import_count
226
+
227
+ def __get_kernel_consumption(self):
228
+ try:
229
+ # Get the current process ID (PID)
230
+ current_pid = os.getpid()
231
+
232
+ # Find the kernel process
233
+ kernel_process = None
234
+ for process in psutil.process_iter(attrs=['pid', 'name']):
235
+ try:
236
+ if process.info['pid'] == current_pid or 'python' in process.info['name'].lower():
237
+ kernel_process = process
238
+ break # We assume the first matching process is the kernel
239
+ except (psutil.NoSuchProcess, psutil.AccessDenied, psutil.ZombieProcess):
240
+ continue
241
+
242
+ if kernel_process is None:
243
+ return "Kernel process not found."
244
+
245
+ # Get CPU and memory usage
246
+ cpu_usage = kernel_process.cpu_percent(interval=1) # CPU usage after 1-second measurement
247
+ memory_usage = kernel_process.memory_info().rss / (1024 ** 2) # Convert bytes to MB
248
+
249
+ return {"CPU Usage (%)": cpu_usage, "Memory Usage (MB)": round(memory_usage, 2)}
250
+
251
+ except Exception as e:
252
+ return f"Error retrieving kernel consumption: {e}"
253
+
254
+ def __count_algorithms_in_notebook(self, notebook):
255
+ try:
256
+ with open(notebook, 'r', encoding='utf-8') as notebook_file:
257
+ notebook_content = json.load(notebook_file)
258
+
259
+ # Patterns to detect algorithms
260
+ function_def_pattern = r"^\s*def\s+\w+\s*\("
261
+ loop_pattern = r"^\s*(for|while)\s+"
262
+ function_call_pattern = r"^\s*\w+\s*\(.*\)" # Simplified function call pattern
263
+
264
+ function_count = 0
265
+ loop_count = 0
266
+ function_call_count = 0
267
+
268
+ for cell in notebook_content.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
269
+ if cell.get("cell_type") == "code":
270
+ for line in cell.get("source", []):
271
+ if re.match(function_def_pattern, line):
272
+ function_count += 1
273
+ elif re.match(loop_pattern, line):
274
+ loop_count += 1
275
+ elif re.match(function_call_pattern, line) and not re.match(function_def_pattern, line):
276
+ function_call_count += 1
277
+
278
+ total_algorithms = function_count + loop_count + function_call_count
279
+ return {
280
+ "Function Definitions": function_count,
281
+ "Loops": loop_count,
282
+ "Function Calls": function_call_count,
283
+ "Total Algorithms": total_algorithms
284
+ }
285
+
286
+ except Exception as e:
287
+ return {
288
+ "Function Definitions": 0,
289
+ "Loops": 0,
290
+ "Function Calls": 0,
291
+ "Total Algorithms": 0,
292
+ "Error": str(e)
293
+ }
294
+
295
+ def __count_classification_algorithms(self, notebook):
296
+ # Scans a Jupyter notebook for classification algorithms and counts them.
297
+
298
+ # It detects classifiers from libraries such as:
299
+ # - Scikit-learn
300
+ # - XGBoost
301
+ # - LightGBM
302
+ # - TensorFlow/Keras
303
+ # - PyTorch
304
+
305
+ # List of common classification algorithms
306
+ classification_keywords = [
307
+ # Scikit-learn classifiers
308
+ r"\bLogisticRegression\(", r"\bRandomForestClassifier\(", r"\bSVC\(",
309
+ r"\bDecisionTreeClassifier\(", r"\bKNeighborsClassifier\(", r"\bGaussianNB\(",
310
+ r"\bGradientBoostingClassifier\(", r"\bAdaBoostClassifier\(", r"\bMLPClassifier\(",
311
+ r"\bHistGradientBoostingClassifier\(", r"\bExtraTreesClassifier\(",
312
+
313
+ # XGBoost
314
+ r"\bXGBClassifier\(",
315
+
316
+ # LightGBM
317
+ r"\bLGBMClassifier\(",
318
+
319
+ # TensorFlow/Keras models
320
+ r"\bSequential\(", r"\bDense\(", r"\bConv2D\(", r"\bLSTM\(",
321
+
322
+ # PyTorch classifiers
323
+ r"\btorch\.nn\.Linear\(", r"\btorch\.nn\.Conv2d\(", r"\btorch\.nn\.LSTM\("
324
+ ]
325
+
326
+ # Load notebook content
327
+ try:
328
+ with open(notebook, 'r', encoding='utf-8') as notebook_file:
329
+ notebook_content = json.load(notebook_file)
330
+
331
+ classification_count = 0
332
+
333
+ # Scan each cell in the notebook
334
+ for cell in notebook_content.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
335
+ if cell.get("cell_type") == "code":
336
+ code = "".join(cell.get("source", [])) # Get full cell code
337
+
338
+ # Check for classifier patterns
339
+ for pattern in classification_keywords:
340
+ matches = re.findall(pattern, code)
341
+ classification_count += len(matches)
342
+
343
+ return classification_count
344
+
345
+ except Exception as e:
346
+ return {"Error": f"Error reading the notebook: {e}"}
347
+
348
+ def __count_regression_algorithms(self, notebook):
349
+ # Scans a Jupyter notebook for regression algorithms and counts them.
350
+
351
+ # It detects regressors from:
352
+ # - Scikit-learn
353
+ # - XGBoost
354
+ # - LightGBM
355
+ # - TensorFlow/Keras
356
+ # - PyTorch
357
+
358
+ # List of common regression algorithms
359
+ regression_keywords = [
360
+ # Scikit-learn regressors
361
+ r"\bLinearRegression\(", r"\bRidge\(", r"\bLasso\(",
362
+ r"\bElasticNet\(", r"\bSVR\(", r"\bDecisionTreeRegressor\(",
363
+ r"\bRandomForestRegressor\(", r"\bGradientBoostingRegressor\(",
364
+ r"\bAdaBoostRegressor\(", r"\bKNeighborsRegressor\(",
365
+ r"\bMLPRegressor\(", r"\bExtraTreesRegressor\(",
366
+ r"\bHistGradientBoostingRegressor\(",
367
+
368
+ # XGBoost
369
+ r"\bXGBRegressor\(",
370
+
371
+ # LightGBM
372
+ r"\bLGBMRegressor\(",
373
+
374
+ # TensorFlow/Keras models
375
+ r"\bSequential\(", r"\bDense\(",
376
+
377
+ # PyTorch models
378
+ r"\btorch\.nn\.Linear\(", r"\btorch\.nn\.Conv2d\("
379
+ ]
380
+
381
+ # Load notebook content
382
+ try:
383
+ with open(notebook, 'r', encoding='utf-8') as notebook_file:
384
+ notebook_content = json.load(notebook_file)
385
+
386
+ regression_count = 0
387
+
388
+ # Scan each cell in the notebook
389
+ for cell in notebook_content.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
390
+ if cell.get("cell_type") == "code":
391
+ code = "".join(cell.get("source", [])) # Get full cell code
392
+
393
+ # Check for regression patterns
394
+ for pattern in regression_keywords:
395
+ matches = re.findall(pattern, code)
396
+ regression_count += len(matches)
397
+
398
+ return regression_count
399
+
400
+ except Exception as e:
401
+ return {"Error": f"Error reading the notebook: {e}"}
402
+
403
+ def __compute_henry_kafura_metrics(self):
404
+ function_calls = {}
405
+ global_calls = set()
406
+
407
+ for cell in self.notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
408
+ if cell.get("cell_type") == "code":
409
+ code = "".join(cell.get("source", []))
410
+ try:
411
+ tree = ast.parse(code)
412
+ except SyntaxError:
413
+ continue
414
+
415
+ for node in ast.iter_child_nodes(tree):
416
+ # Track function definitions and their internal calls
417
+ if isinstance(node, ast.FunctionDef):
418
+ func_name = node.name
419
+ function_calls[func_name] = set()
420
+
421
+ for child in ast.walk(node):
422
+ if isinstance(child, ast.Call):
423
+ if isinstance(child.func, ast.Name):
424
+ function_calls[func_name].add(child.func.id)
425
+ else:
426
+ # Also track top-level (global scope) function calls
427
+ for child in ast.walk(node):
428
+ if isinstance(child, ast.Call) and isinstance(child.func, ast.Name):
429
+ global_calls.add(child.func.id)
430
+
431
+ # If nothing was found
432
+ if not function_calls and not global_calls:
433
+ return "Error: Could not retrieve function call data."
434
+
435
+ # Add global scope calls as a synthetic "global" function
436
+ if global_calls:
437
+ function_calls["_global_scope_"] = global_calls
438
+
439
+ # Compute Henry-Kafura metrics
440
+ complexity_metrics = {}
441
+ for function, calls in function_calls.items():
442
+ F_in = sum(1 for funcs in function_calls.values() if function in funcs)
443
+ F_out = len(calls)
444
+ complexity_metrics[function] = {
445
+ "F_in": F_in,
446
+ "F_out": F_out,
447
+ "Complexity": F_in * (F_out ** 2)
448
+ }
449
+
450
+ return complexity_metrics
451
+
452
+ def __cyclomatic_complexity_ast(self, tree):
453
+ # Computes Cyclomatic Complexity (CC) from an AST tree.
454
+ nodes = 1 # Start with 1 (entry point)
455
+ edges = 0
456
+
457
+ for node in ast.walk(tree):
458
+ if isinstance(node, (ast.If, ast.While, ast.For, ast.Try, ast.With)):
459
+ nodes += 1
460
+ edges += 2 # Each control structure introduces a new path
461
+ elif isinstance(node, ast.IfExp): # Ternary operator (e.g., x if cond else y)
462
+ nodes += 1
463
+ edges += 2
464
+ elif isinstance(node, ast.BinOp) and isinstance(node.op, (ast.And, ast.Or)):
465
+ edges += 1 # Boolean short-circuiting increases complexity
466
+
467
+ complexity = edges - nodes + 2
468
+ return max(complexity, 1) # CC is at least 1
469
+
470
+ def __compute_cyclomatic_complexity(self, notebook_path):
471
+ # Computes total Cyclomatic Complexity across code cells in the notebook.
472
+ with open(notebook_path, 'r', encoding='utf-8') as f:
473
+ notebook = json.load(f)
474
+
475
+ total_complexity = 0
476
+
477
+ for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
478
+ if cell["cell_type"] == "code":
479
+ code = "".join(cell["source"])
480
+ try:
481
+ tree = ast.parse(code)
482
+ total_complexity += self.__cyclomatic_complexity_ast(tree)
483
+ except SyntaxError:
484
+ continue # skip cells with invalid syntax
485
+
486
+ return total_complexity
487
+
488
+ def __count_executed_lines_of_code(self, notebook_path):
489
+ with open(notebook_path, 'r', encoding='utf-8') as f:
490
+ notebook = json.load(f)
491
+
492
+ total_lines = 0
493
+
494
+ for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
495
+ if cell["cell_type"] == "code":
496
+ code_lines = cell["source"]
497
+ executable_lines = [
498
+ line for line in code_lines
499
+ if line.strip() and not line.strip().startswith("#") # Ignore empty lines and comments
500
+ ]
501
+ total_lines += len(executable_lines)
502
+
503
+ return total_lines
504
+
505
+
506
+ def __count_functions_in_notebook(self, notebook_path):
507
+ with open(notebook_path, 'r', encoding='utf-8') as f:
508
+ notebook = json.load(f)
509
+
510
+ function_count = 0
511
+
512
+ for cell in notebook.get("cells", [])[self.code_blocks_to_exclude:-self.code_blocks_to_exclude]:
513
+ if cell["cell_type"] == "code": # Only analyze code cells
514
+ code = "".join(cell["source"]) # Get full code as a string
515
+ try:
516
+ tree = ast.parse(code) # Parse code into an AST
517
+ function_count += sum(isinstance(node, ast.FunctionDef) for node in ast.walk(tree))
518
+ except SyntaxError:
519
+ pass # Skip cells with invalid Python syntax
520
+
521
+ return function_count
522
+
523
+ def hello( self, name ):
524
+ print( "Hello, " + name + "!. Welcome from Tracker package." )
525
+
526
+ def version(self):
527
+ from importlib.metadata import version
528
+ return "lusofona-pckg v. " + version('lusofona-pckg')
529
+
530
+ def start(self):
531
+ self.tracker.start()
532
+
533
+ def collect(self):
534
+
535
+ ram_data = self.__ram_measurement()
536
+
537
+ disk_speed = self.__measure_disk_speed()
538
+
539
+ dataframe_structure = self.__assess_dataframe_structure()
540
+
541
+ libraries = self.__detectlib(self.notebook)
542
+
543
+ algorithms = self.__count_algorithms_in_notebook(self.notebook_path)
544
+
545
+ code_carbon_data = json.loads(self.tracker.final_emissions_data.toJSON())
546
+
547
+ hk_metrics = self.__compute_henry_kafura_metrics()
548
+
549
+ cc_plus_data = {'cpu_flops': float(self.__estimate_cpu_flops()),
550
+ 'actual_flops': float(self.__measure_actual_flops()),
551
+ 'total_ram': ram_data["Total_ram"],
552
+ 'used_ram': ram_data["Used_ram"],
553
+ 'available_ram': ram_data["Avaialble_ram"],
554
+ 'ram_usage': ram_data["Ram_Usage"],
555
+ 'write_speed': disk_speed["Write_Speed"],
556
+ 'read_speed': disk_speed["Read_Speed"],
557
+ 'dataframe_count': self.__count_dataframes_in_notebook(self.notebook),
558
+ 'file_extensions': self.__detect_file_extensions_in_notebook(self.notebook),
559
+ 'total_rows': dataframe_structure["total_rows"],
560
+ 'total_columns': dataframe_structure["total_columns"],
561
+ 'datafiles': dataframe_structure["datafiles"],
562
+ 'number_of_libraries': libraries['num_libs'],
563
+ 'library_names': libraries['libs'],
564
+ 'import_count': self.__count_imports_in_notebook(self.notebook),
565
+ 'kernel_consumption': self.__get_kernel_consumption(),
566
+ 'function_definitions': algorithms['Function Definitions'],
567
+ 'loops': algorithms['Loops'],
568
+ 'function_calls': algorithms['Function Calls'],
569
+ 'total_algorithms': algorithms['Total Algorithms'],
570
+ 'classification_algorithms': self.__count_classification_algorithms(self.notebook_path),
571
+ 'regression_algorithms': self.__count_regression_algorithms(self.notebook_path),
572
+ 'F_in': hk_metrics['_global_scope_']['F_in'],
573
+ 'F_out': hk_metrics['_global_scope_']['F_out'],
574
+ 'Complexity': hk_metrics['_global_scope_']['Complexity'],
575
+ 'cyclomatic_complexity': self.__compute_cyclomatic_complexity(self.notebook_path),
576
+ 'number_of_lines': self.__count_executed_lines_of_code(self.notebook_path),
577
+ 'number_of_functions': self.__count_functions_in_notebook(self.notebook_path),
578
+ 'region': code_carbon_data["region"],
579
+ 'carbon_emissions_kg': code_carbon_data["emissions"],
580
+ 'energy_consumed_kwh': code_carbon_data["energy_consumed"],
581
+ 'duration_s': code_carbon_data["duration"],
582
+ 'cpu_power_watt': code_carbon_data["cpu_power"],
583
+ 'ram_power_watt': code_carbon_data["ram_power"],
584
+ 'timestamp': code_carbon_data.get("timestamp"),
585
+ 'project_name': "etracker", #code_carbon_data.get("project_name"),
586
+ 'run_id': code_carbon_data.get("run_id"),
587
+ 'experiment_id': self.experimentid, #code_carbon_data.get("experiment_id"),
588
+ 'duration_s': code_carbon_data.get("duration"),
589
+ 'carbon_emissions_kg': code_carbon_data.get("emissions"),
590
+ 'emissions_rate': code_carbon_data.get("emissions_rate"),
591
+ 'cpu_power_watt': code_carbon_data.get("cpu_power"),
592
+ 'gpu_power_watt': code_carbon_data.get("gpu_power"),
593
+ 'ram_power_watt': code_carbon_data.get("ram_power"),
594
+ 'cpu_energy': code_carbon_data.get("cpu_energy"),
595
+ 'gpu_energy': code_carbon_data.get("gpu_energy"),
596
+ 'ram_energy': code_carbon_data.get("ram_energy"),
597
+ 'energy_consumed_kwh': code_carbon_data.get("energy_consumed"),
598
+ 'country_name': code_carbon_data.get("country_name"),
599
+ 'country_iso_code': code_carbon_data.get("country_iso_code"),
600
+ 'region': code_carbon_data.get("region"),
601
+ 'cloud_provider': code_carbon_data.get("cloud_provider"),
602
+ 'cloud_region': code_carbon_data.get("cloud_region"),
603
+ 'os': code_carbon_data.get("os"),
604
+ 'python_version': code_carbon_data.get("python_version"),
605
+ 'codecarbon_version': code_carbon_data.get("codecarbon_version"),
606
+ 'cpu_count': code_carbon_data.get("cpu_count"),
607
+ 'cpu_model': code_carbon_data.get("cpu_model"),
608
+ 'gpu_count': code_carbon_data.get("gpu_count"),
609
+ 'gpu_model': code_carbon_data.get("gpu_model"),
610
+ 'longitude': code_carbon_data.get("longitude"),
611
+ 'latitude': code_carbon_data.get("latitude"),
612
+ 'ram_total_size': code_carbon_data.get("ram_total_size"),
613
+ 'tracking_mode': code_carbon_data.get("tracking_mode"),
614
+ 'on_cloud': code_carbon_data.get("on_cloud"),
615
+ 'pue': code_carbon_data.get("pue"),
616
+ }
617
+
618
+ self.final_data = {**cc_plus_data}
619
+
620
+ # write data to csv file if it does not exist , otherwise append to it
621
+
622
+ df = pd.DataFrame([self.final_data])
623
+ output_file = f"tracker_output.csv"
624
+ if not os.path.exists(output_file):
625
+ df.to_csv(output_file, index=False)
626
+ else:
627
+ df.to_csv(output_file, mode='a', header=False, index=False)
628
+
629
+
630
+ def stop(self):
631
+ self.tracker.stop()
632
+
633
+ self.collect()
634
+
635
+ def view(self):
636
+ display(self.final_data)
@@ -0,0 +1,11 @@
1
+
2
+ #import pandas as pd
3
+
4
+ # Adds variance, skewness and kurtosis to the describe() method
5
+ def extended_describe( df ):
6
+ numerical = df.select_dtypes( exclude=[ 'object', 'category', 'datetime64[ns]' ] )
7
+ extended_df = df.describe( include='all' ).T
8
+ extended_df['var'] = numerical.var()
9
+ extended_df['skew'] = numerical.skew()
10
+ extended_df['kurt'] = numerical.kurt()
11
+ return extended_df
@@ -0,0 +1,15 @@
1
+
2
+ from IPython.display import display, HTML
3
+
4
+ # Plot dataframes side by side to avoid scrolling
5
+ def plot_dataframes_side_by_side( dfs:list, captions:list, tablespacing=5 ):
6
+ """Display tables side by side to save vertical space
7
+ Input:
8
+ dfs: list of pandas.DataFrame
9
+ captions: list of table captions
10
+ """
11
+ output = ""
12
+ for (caption, df) in zip(captions, dfs):
13
+ output += df.style.set_table_attributes("style='display:inline'").set_caption(caption)._repr_html_()
14
+ output += tablespacing * "\xa0"
15
+ display(HTML(output))
@@ -0,0 +1,23 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77.0.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "ecrawler"
7
+ version = "0.0.2"
8
+ description = "Python package developed at Lusófona University to track notebook execution times and energy consumption."
9
+
10
+ readme = "README.md"
11
+ requires-python = ">=3.12"
12
+
13
+ authors = [
14
+ { name = "Gonçalo Cabeleira", email = "goncalocabeleira123@gmail.com" },
15
+ { name = "João Caldeira", email = "joao.caldeira@ulusofona.pt" },
16
+ ]
17
+
18
+ license-files = ["LICENSE"]
19
+
20
+ dependencies = ["rich", "pandas", "numpy", "codecarbon"]
21
+
22
+ [project.urls]
23
+ homepage = "https://github.com/Streetcode5/ecrawler.git."
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+