pythonlabtools 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pythonlabtools/__init__.py +39 -0
- pythonlabtools/association.py +98 -0
- pythonlabtools/classification.py +92 -0
- pythonlabtools/clustering.py +60 -0
- pythonlabtools/code_viewer.py +305 -0
- pythonlabtools/database.py +49 -0
- pythonlabtools/numpy_tools.py +51 -0
- pythonlabtools/pandas_tools.py +64 -0
- pythonlabtools/plotting.py +141 -0
- pythonlabtools/preprocessing.py +133 -0
- pythonlabtools/regression.py +53 -0
- pythonlabtools-2.0.0.dist-info/METADATA +109 -0
- pythonlabtools-2.0.0.dist-info/RECORD +15 -0
- pythonlabtools-2.0.0.dist-info/WHEEL +5 -0
- pythonlabtools-2.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
__version__ = "2.0.0"
|
|
2
|
+
|
|
3
|
+
# Core tools
|
|
4
|
+
from .numpy_tools import (
|
|
5
|
+
create_range_array,
|
|
6
|
+
reshape_array,
|
|
7
|
+
transpose_array,
|
|
8
|
+
array_sums,
|
|
9
|
+
even_numbers,
|
|
10
|
+
center_slice,
|
|
11
|
+
reverse_array,
|
|
12
|
+
flatten_array,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
from .pandas_tools import (
|
|
16
|
+
series_to_dataframe,
|
|
17
|
+
add_pass_fail,
|
|
18
|
+
passed_students,
|
|
19
|
+
average_by_group,
|
|
20
|
+
highest_salary_employee,
|
|
21
|
+
concat_rows,
|
|
22
|
+
concat_columns,
|
|
23
|
+
merge_dataframes,
|
|
24
|
+
value_counts_all,
|
|
25
|
+
group_age_statistics,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# Database tools
|
|
29
|
+
from .database import (
|
|
30
|
+
connect_sqlite,
|
|
31
|
+
create_table,
|
|
32
|
+
insert_student,
|
|
33
|
+
select_students,
|
|
34
|
+
update_student,
|
|
35
|
+
delete_student,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
# Code viewer
|
|
39
|
+
from .code_viewer import show_code
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Small dependency-light association-rule helpers."""
|
|
2
|
+
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from itertools import combinations
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def item_counts(transactions):
|
|
9
|
+
"""Count individual items across transactions."""
|
|
10
|
+
counts = Counter()
|
|
11
|
+
|
|
12
|
+
for transaction in transactions:
|
|
13
|
+
for item in transaction:
|
|
14
|
+
counts[item] += 1
|
|
15
|
+
|
|
16
|
+
return counts
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def frequent_items(transactions, min_count=2):
|
|
20
|
+
"""Return items meeting the minimum occurrence count."""
|
|
21
|
+
counts = item_counts(transactions)
|
|
22
|
+
|
|
23
|
+
return {
|
|
24
|
+
item: count
|
|
25
|
+
for item, count in counts.items()
|
|
26
|
+
if count >= min_count
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def support(transactions, itemset):
|
|
31
|
+
"""Return support of an itemset."""
|
|
32
|
+
itemset = set(itemset)
|
|
33
|
+
count = 0
|
|
34
|
+
|
|
35
|
+
for transaction in transactions:
|
|
36
|
+
if itemset.issubset(set(transaction)):
|
|
37
|
+
count += 1
|
|
38
|
+
|
|
39
|
+
return count / len(transactions)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def confidence(transactions, A, B):
|
|
43
|
+
"""Return confidence for A -> B."""
|
|
44
|
+
return support(transactions, list(A) + list(B)) / support(
|
|
45
|
+
transactions, A
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def lift(transactions, A, B):
|
|
50
|
+
"""Return lift for A -> B."""
|
|
51
|
+
return confidence(transactions, A, B) / support(
|
|
52
|
+
transactions, B
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def generate_pair_rules(transactions, min_support=0.1):
|
|
57
|
+
"""Generate two-item directional rules."""
|
|
58
|
+
items = list(item_counts(transactions).keys())
|
|
59
|
+
rules = []
|
|
60
|
+
|
|
61
|
+
for A, B in combinations(items, 2):
|
|
62
|
+
pair_support = support(
|
|
63
|
+
transactions,
|
|
64
|
+
[A, B]
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
if pair_support >= min_support:
|
|
68
|
+
rules.append({
|
|
69
|
+
"Rule": f"{A} -> {B}",
|
|
70
|
+
"Support": pair_support,
|
|
71
|
+
"Confidence": confidence(
|
|
72
|
+
transactions, [A], [B]
|
|
73
|
+
),
|
|
74
|
+
"Lift": lift(
|
|
75
|
+
transactions, [A], [B]
|
|
76
|
+
),
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
rules.append({
|
|
80
|
+
"Rule": f"{B} -> {A}",
|
|
81
|
+
"Support": pair_support,
|
|
82
|
+
"Confidence": confidence(
|
|
83
|
+
transactions, [B], [A]
|
|
84
|
+
),
|
|
85
|
+
"Lift": lift(
|
|
86
|
+
transactions, [B], [A]
|
|
87
|
+
),
|
|
88
|
+
})
|
|
89
|
+
|
|
90
|
+
return pd.DataFrame(rules)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def top_rules_by_lift(rules_df, n=5):
|
|
94
|
+
"""Return top n rules sorted by lift."""
|
|
95
|
+
return rules_df.sort_values(
|
|
96
|
+
"Lift",
|
|
97
|
+
ascending=False
|
|
98
|
+
).head(n)
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Classification helpers for Q50-style lab programs."""
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.tree import DecisionTreeClassifier
|
|
5
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
6
|
+
from sklearn.linear_model import LogisticRegression
|
|
7
|
+
from sklearn.metrics import (
|
|
8
|
+
accuracy_score,
|
|
9
|
+
recall_score,
|
|
10
|
+
confusion_matrix,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def train_classification_models(
|
|
15
|
+
X_train,
|
|
16
|
+
X_test,
|
|
17
|
+
y_train,
|
|
18
|
+
y_test,
|
|
19
|
+
scaled_X_train=None,
|
|
20
|
+
scaled_X_test=None
|
|
21
|
+
):
|
|
22
|
+
"""
|
|
23
|
+
Train Decision Tree, Random Forest and Logistic Regression.
|
|
24
|
+
|
|
25
|
+
Logistic Regression uses scaled data if supplied.
|
|
26
|
+
"""
|
|
27
|
+
dt = DecisionTreeClassifier(random_state=42)
|
|
28
|
+
rf = RandomForestClassifier(
|
|
29
|
+
n_estimators=100,
|
|
30
|
+
random_state=42
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
dt.fit(X_train, y_train)
|
|
34
|
+
rf.fit(X_train, y_train)
|
|
35
|
+
|
|
36
|
+
if scaled_X_train is None:
|
|
37
|
+
scaled_X_train = X_train
|
|
38
|
+
if scaled_X_test is None:
|
|
39
|
+
scaled_X_test = X_test
|
|
40
|
+
|
|
41
|
+
lr = LogisticRegression(max_iter=5000)
|
|
42
|
+
lr.fit(scaled_X_train, y_train)
|
|
43
|
+
|
|
44
|
+
models = {
|
|
45
|
+
"Decision Tree": dt,
|
|
46
|
+
"Random Forest": rf,
|
|
47
|
+
"Logistic Regression": lr,
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
predictions = {
|
|
51
|
+
"Decision Tree": dt.predict(X_test),
|
|
52
|
+
"Random Forest": rf.predict(X_test),
|
|
53
|
+
"Logistic Regression": lr.predict(scaled_X_test),
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
return models, predictions
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def evaluate_classification(y_test, predictions):
|
|
60
|
+
"""Return accuracy, recall and confusion matrix for each model."""
|
|
61
|
+
results = {}
|
|
62
|
+
|
|
63
|
+
for name, pred in predictions.items():
|
|
64
|
+
results[name] = {
|
|
65
|
+
"Accuracy": accuracy_score(y_test, pred),
|
|
66
|
+
"Recall": recall_score(y_test, pred),
|
|
67
|
+
"Confusion Matrix": confusion_matrix(y_test, pred),
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
return results
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def confusion_values(y_test, prediction):
|
|
74
|
+
"""Return TN, FP, FN, TP for binary classification."""
|
|
75
|
+
cm = confusion_matrix(y_test, prediction)
|
|
76
|
+
return tuple(cm.ravel())
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def top_random_forest_features(model, feature_names, n=10):
|
|
80
|
+
"""Return top n features by Random Forest importance."""
|
|
81
|
+
importance = pd.Series(
|
|
82
|
+
model.feature_importances_,
|
|
83
|
+
index=feature_names
|
|
84
|
+
)
|
|
85
|
+
return importance.sort_values(
|
|
86
|
+
ascending=False
|
|
87
|
+
).head(n)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def predict_new_sample(model, sample):
|
|
91
|
+
"""Predict one or more new samples."""
|
|
92
|
+
return model.predict(sample)
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""K-Means and PCA helpers for Q51-style lab programs."""
|
|
2
|
+
|
|
3
|
+
import matplotlib.pyplot as plt
|
|
4
|
+
from sklearn.cluster import KMeans
|
|
5
|
+
from sklearn.decomposition import PCA
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def kmeans_models(X, ks=(2, 3, 4), random_state=42):
|
|
9
|
+
"""Run K-Means for several k values."""
|
|
10
|
+
models = {}
|
|
11
|
+
labels = {}
|
|
12
|
+
|
|
13
|
+
for k in ks:
|
|
14
|
+
model = KMeans(
|
|
15
|
+
n_clusters=k,
|
|
16
|
+
random_state=random_state,
|
|
17
|
+
n_init=10
|
|
18
|
+
)
|
|
19
|
+
labels[k] = model.fit_predict(X)
|
|
20
|
+
models[k] = model
|
|
21
|
+
|
|
22
|
+
return models, labels
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def elbow_method(X, k_range=range(2, 11), random_state=42):
|
|
26
|
+
"""Return k values and K-Means inertia values."""
|
|
27
|
+
ks = list(k_range)
|
|
28
|
+
inertias = []
|
|
29
|
+
|
|
30
|
+
for k in ks:
|
|
31
|
+
model = KMeans(
|
|
32
|
+
n_clusters=k,
|
|
33
|
+
random_state=random_state,
|
|
34
|
+
n_init=10
|
|
35
|
+
)
|
|
36
|
+
model.fit(X)
|
|
37
|
+
inertias.append(model.inertia_)
|
|
38
|
+
|
|
39
|
+
return ks, inertias
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def pca_2d(X):
|
|
43
|
+
"""Reduce features to two PCA components."""
|
|
44
|
+
pca = PCA(n_components=2)
|
|
45
|
+
return pca.fit_transform(X), pca
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def plot_clusters(X_pca, labels, title="K-Means Clusters"):
|
|
49
|
+
"""Plot 2D PCA data colored by cluster label."""
|
|
50
|
+
plt.figure(figsize=(8, 5))
|
|
51
|
+
plt.scatter(
|
|
52
|
+
X_pca[:, 0],
|
|
53
|
+
X_pca[:, 1],
|
|
54
|
+
c=labels
|
|
55
|
+
)
|
|
56
|
+
plt.xlabel("Principal Component 1")
|
|
57
|
+
plt.ylabel("Principal Component 2")
|
|
58
|
+
plt.title(title)
|
|
59
|
+
plt.tight_layout()
|
|
60
|
+
plt.show()
|
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
import inspect
|
|
2
|
+
import importlib
|
|
3
|
+
import pythonlabtools
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
# Functions that can be displayed directly
|
|
7
|
+
FUNCTION_MODULES = {
|
|
8
|
+
# NumPy
|
|
9
|
+
"create_range_array": "numpy_tools",
|
|
10
|
+
"reshape_array": "numpy_tools",
|
|
11
|
+
"transpose_array": "numpy_tools",
|
|
12
|
+
"array_sums": "numpy_tools",
|
|
13
|
+
"even_numbers": "numpy_tools",
|
|
14
|
+
"center_slice": "numpy_tools",
|
|
15
|
+
"reverse_array": "numpy_tools",
|
|
16
|
+
"flatten_array": "numpy_tools",
|
|
17
|
+
|
|
18
|
+
# Pandas
|
|
19
|
+
"series_to_dataframe": "pandas_tools",
|
|
20
|
+
"add_pass_fail": "pandas_tools",
|
|
21
|
+
"passed_students": "pandas_tools",
|
|
22
|
+
"average_by_group": "pandas_tools",
|
|
23
|
+
"highest_salary_employee": "pandas_tools",
|
|
24
|
+
"concat_rows": "pandas_tools",
|
|
25
|
+
"concat_columns": "pandas_tools",
|
|
26
|
+
"merge_dataframes": "pandas_tools",
|
|
27
|
+
"value_counts_all": "pandas_tools",
|
|
28
|
+
"group_age_statistics": "pandas_tools",
|
|
29
|
+
|
|
30
|
+
# Preprocessing
|
|
31
|
+
"show_missing": "preprocessing",
|
|
32
|
+
"drop_missing": "preprocessing",
|
|
33
|
+
"fill_missing_mean": "preprocessing",
|
|
34
|
+
"fill_missing_median": "preprocessing",
|
|
35
|
+
"fill_missing_mode": "preprocessing",
|
|
36
|
+
"convert_numeric": "preprocessing",
|
|
37
|
+
"remove_duplicates": "preprocessing",
|
|
38
|
+
"replace_invalid": "preprocessing",
|
|
39
|
+
"encode_categorical": "preprocessing",
|
|
40
|
+
"scale_features": "preprocessing",
|
|
41
|
+
"train_test_scale": "preprocessing",
|
|
42
|
+
"preprocess_basic": "preprocessing",
|
|
43
|
+
|
|
44
|
+
# Plotting
|
|
45
|
+
"line_plot": "plotting",
|
|
46
|
+
"bar_plot": "plotting",
|
|
47
|
+
"scatter_plot": "plotting",
|
|
48
|
+
"histogram": "plotting",
|
|
49
|
+
"multiple_plots": "plotting",
|
|
50
|
+
"correlation_heatmap": "plotting",
|
|
51
|
+
"box_plot": "plotting",
|
|
52
|
+
|
|
53
|
+
# Regression
|
|
54
|
+
"train_regression_models": "regression",
|
|
55
|
+
"evaluate_regression": "regression",
|
|
56
|
+
|
|
57
|
+
# Classification
|
|
58
|
+
"train_classification_models": "classification",
|
|
59
|
+
"evaluate_classification": "classification",
|
|
60
|
+
"confusion_values": "classification",
|
|
61
|
+
"top_random_forest_features": "classification",
|
|
62
|
+
"predict_new_sample": "classification",
|
|
63
|
+
|
|
64
|
+
# Clustering
|
|
65
|
+
"kmeans_models": "clustering",
|
|
66
|
+
"elbow_method": "clustering",
|
|
67
|
+
"pca_2d": "clustering",
|
|
68
|
+
"plot_clusters": "clustering",
|
|
69
|
+
|
|
70
|
+
# Association
|
|
71
|
+
"item_counts": "association",
|
|
72
|
+
"frequent_items": "association",
|
|
73
|
+
"support": "association",
|
|
74
|
+
"confidence": "association",
|
|
75
|
+
"lift": "association",
|
|
76
|
+
"generate_pair_rules": "association",
|
|
77
|
+
"top_rules_by_lift": "association",
|
|
78
|
+
|
|
79
|
+
# Database
|
|
80
|
+
"connect_sqlite": "database",
|
|
81
|
+
"create_table": "database",
|
|
82
|
+
"insert_student": "database",
|
|
83
|
+
"select_students": "database",
|
|
84
|
+
"update_student": "database",
|
|
85
|
+
"delete_student": "database",
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
MODULES = {
|
|
90
|
+
"numpy": "numpy_tools",
|
|
91
|
+
"pandas": "pandas_tools",
|
|
92
|
+
"preprocessing": "preprocessing",
|
|
93
|
+
"plotting": "plotting",
|
|
94
|
+
"regression": "regression",
|
|
95
|
+
"classification": "classification",
|
|
96
|
+
"clustering": "clustering",
|
|
97
|
+
"association": "association",
|
|
98
|
+
"database": "database",
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
QUESTION_FILES = {
|
|
103
|
+
"q49": "examples.q49_regression_template",
|
|
104
|
+
"q50": "examples.q50_classification_template",
|
|
105
|
+
"q51": "examples.q51_clustering_template",
|
|
106
|
+
"q52": "examples.q52_association_template",
|
|
107
|
+
"q30": "examples.q30_sqlite_template",
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def show_code(name):
|
|
112
|
+
"""
|
|
113
|
+
Display the source code of a pythonlabtools function,
|
|
114
|
+
module, or exam question template.
|
|
115
|
+
|
|
116
|
+
Examples:
|
|
117
|
+
show_code("fill_missing_mean")
|
|
118
|
+
show_code("preprocessing")
|
|
119
|
+
show_code("q49")
|
|
120
|
+
"""
|
|
121
|
+
|
|
122
|
+
name = name.lower().strip()
|
|
123
|
+
|
|
124
|
+
# --------------------------------------------------
|
|
125
|
+
# 1. Show a specific function
|
|
126
|
+
# --------------------------------------------------
|
|
127
|
+
if name in FUNCTION_MODULES:
|
|
128
|
+
|
|
129
|
+
module_name = FUNCTION_MODULES[name]
|
|
130
|
+
|
|
131
|
+
module = importlib.import_module(
|
|
132
|
+
f"pythonlabtools.{module_name}"
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
function = getattr(module, name)
|
|
136
|
+
|
|
137
|
+
print("=" * 70)
|
|
138
|
+
print(f"CODE: {name}")
|
|
139
|
+
print("=" * 70)
|
|
140
|
+
|
|
141
|
+
print(inspect.getsource(function))
|
|
142
|
+
|
|
143
|
+
return
|
|
144
|
+
|
|
145
|
+
# --------------------------------------------------
|
|
146
|
+
# 2. Show an entire module
|
|
147
|
+
# --------------------------------------------------
|
|
148
|
+
if name in MODULES:
|
|
149
|
+
|
|
150
|
+
module_name = MODULES[name]
|
|
151
|
+
|
|
152
|
+
module = importlib.import_module(
|
|
153
|
+
f"pythonlabtools.{module_name}"
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
print("=" * 70)
|
|
157
|
+
print(f"MODULE: {module_name}")
|
|
158
|
+
print("=" * 70)
|
|
159
|
+
|
|
160
|
+
print(inspect.getsource(module))
|
|
161
|
+
|
|
162
|
+
return
|
|
163
|
+
|
|
164
|
+
# --------------------------------------------------
|
|
165
|
+
# 3. Show an exam question template
|
|
166
|
+
# --------------------------------------------------
|
|
167
|
+
if name in QUESTION_FILES:
|
|
168
|
+
|
|
169
|
+
module_name = QUESTION_FILES[name]
|
|
170
|
+
|
|
171
|
+
module = importlib.import_module(module_name)
|
|
172
|
+
|
|
173
|
+
print("=" * 70)
|
|
174
|
+
print(f"EXAM TEMPLATE: {name.upper()}")
|
|
175
|
+
print("=" * 70)
|
|
176
|
+
|
|
177
|
+
print(inspect.getsource(module))
|
|
178
|
+
|
|
179
|
+
return
|
|
180
|
+
|
|
181
|
+
# --------------------------------------------------
|
|
182
|
+
# 4. Help
|
|
183
|
+
# --------------------------------------------------
|
|
184
|
+
if name in ["help", "list", "all"]:
|
|
185
|
+
|
|
186
|
+
print("""
|
|
187
|
+
============================================================
|
|
188
|
+
PYTHONLABTOOLS CODE VIEWER
|
|
189
|
+
============================================================
|
|
190
|
+
|
|
191
|
+
NUMPY
|
|
192
|
+
-----
|
|
193
|
+
create_range_array
|
|
194
|
+
reshape_array
|
|
195
|
+
transpose_array
|
|
196
|
+
array_sums
|
|
197
|
+
even_numbers
|
|
198
|
+
center_slice
|
|
199
|
+
reverse_array
|
|
200
|
+
flatten_array
|
|
201
|
+
|
|
202
|
+
PANDAS
|
|
203
|
+
------
|
|
204
|
+
series_to_dataframe
|
|
205
|
+
add_pass_fail
|
|
206
|
+
passed_students
|
|
207
|
+
average_by_group
|
|
208
|
+
highest_salary_employee
|
|
209
|
+
concat_rows
|
|
210
|
+
concat_columns
|
|
211
|
+
merge_dataframes
|
|
212
|
+
value_counts_all
|
|
213
|
+
group_age_statistics
|
|
214
|
+
|
|
215
|
+
PREPROCESSING
|
|
216
|
+
-------------
|
|
217
|
+
show_missing
|
|
218
|
+
drop_missing
|
|
219
|
+
fill_missing_mean
|
|
220
|
+
fill_missing_median
|
|
221
|
+
fill_missing_mode
|
|
222
|
+
convert_numeric
|
|
223
|
+
remove_duplicates
|
|
224
|
+
replace_invalid
|
|
225
|
+
encode_categorical
|
|
226
|
+
scale_features
|
|
227
|
+
train_test_scale
|
|
228
|
+
preprocess_basic
|
|
229
|
+
|
|
230
|
+
PLOTTING
|
|
231
|
+
--------
|
|
232
|
+
line_plot
|
|
233
|
+
bar_plot
|
|
234
|
+
scatter_plot
|
|
235
|
+
histogram
|
|
236
|
+
multiple_plots
|
|
237
|
+
correlation_heatmap
|
|
238
|
+
box_plot
|
|
239
|
+
|
|
240
|
+
MACHINE LEARNING
|
|
241
|
+
----------------
|
|
242
|
+
train_regression_models
|
|
243
|
+
evaluate_regression
|
|
244
|
+
train_classification_models
|
|
245
|
+
evaluate_classification
|
|
246
|
+
confusion_values
|
|
247
|
+
top_random_forest_features
|
|
248
|
+
predict_new_sample
|
|
249
|
+
|
|
250
|
+
CLUSTERING
|
|
251
|
+
----------
|
|
252
|
+
kmeans_models
|
|
253
|
+
elbow_method
|
|
254
|
+
pca_2d
|
|
255
|
+
plot_clusters
|
|
256
|
+
|
|
257
|
+
ASSOCIATION RULES
|
|
258
|
+
-----------------
|
|
259
|
+
item_counts
|
|
260
|
+
frequent_items
|
|
261
|
+
support
|
|
262
|
+
confidence
|
|
263
|
+
lift
|
|
264
|
+
generate_pair_rules
|
|
265
|
+
top_rules_by_lift
|
|
266
|
+
|
|
267
|
+
DATABASE
|
|
268
|
+
--------
|
|
269
|
+
connect_sqlite
|
|
270
|
+
create_table
|
|
271
|
+
insert_student
|
|
272
|
+
select_students
|
|
273
|
+
update_student
|
|
274
|
+
delete_student
|
|
275
|
+
|
|
276
|
+
EXAM TEMPLATES
|
|
277
|
+
--------------
|
|
278
|
+
q30
|
|
279
|
+
q49
|
|
280
|
+
q50
|
|
281
|
+
q51
|
|
282
|
+
q52
|
|
283
|
+
|
|
284
|
+
EXAMPLES
|
|
285
|
+
--------
|
|
286
|
+
show_code("fill_missing_mean")
|
|
287
|
+
show_code("preprocessing")
|
|
288
|
+
show_code("q49")
|
|
289
|
+
show_code("help")
|
|
290
|
+
|
|
291
|
+
============================================================
|
|
292
|
+
""")
|
|
293
|
+
|
|
294
|
+
return
|
|
295
|
+
|
|
296
|
+
# --------------------------------------------------
|
|
297
|
+
# 5. Invalid name
|
|
298
|
+
# --------------------------------------------------
|
|
299
|
+
|
|
300
|
+
print(f"'{name}' was not found.")
|
|
301
|
+
|
|
302
|
+
print("\nType:")
|
|
303
|
+
print(' show_code("help")')
|
|
304
|
+
|
|
305
|
+
print("to see the available functions.")
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Simple SQLite helpers for the database portion of the lab."""
|
|
2
|
+
|
|
3
|
+
import sqlite3
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def connect_sqlite(path="students.db"):
|
|
7
|
+
return sqlite3.connect(path)
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def create_table(conn):
|
|
11
|
+
conn.execute("""
|
|
12
|
+
CREATE TABLE IF NOT EXISTS students (
|
|
13
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
14
|
+
name TEXT,
|
|
15
|
+
marks INTEGER
|
|
16
|
+
)
|
|
17
|
+
""")
|
|
18
|
+
conn.commit()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def insert_student(conn, name, marks):
|
|
22
|
+
conn.execute(
|
|
23
|
+
"INSERT INTO students (name, marks) VALUES (?, ?)",
|
|
24
|
+
(name, marks)
|
|
25
|
+
)
|
|
26
|
+
conn.commit()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def select_students(conn):
|
|
30
|
+
cursor = conn.execute(
|
|
31
|
+
"SELECT * FROM students"
|
|
32
|
+
)
|
|
33
|
+
return cursor.fetchall()
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def update_student(conn, name, new_marks):
|
|
37
|
+
conn.execute(
|
|
38
|
+
"UPDATE students SET marks=? WHERE name=?",
|
|
39
|
+
(new_marks, name)
|
|
40
|
+
)
|
|
41
|
+
conn.commit()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def delete_student(conn, name):
|
|
45
|
+
conn.execute(
|
|
46
|
+
"DELETE FROM students WHERE name=?",
|
|
47
|
+
(name,)
|
|
48
|
+
)
|
|
49
|
+
conn.commit()
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""NumPy helpers matching common lab questions."""
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def create_range_array(start=1, stop=20):
|
|
7
|
+
"""Create a 1-D array from start through stop, inclusive."""
|
|
8
|
+
return np.arange(start, stop + 1)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def reshape_array(arr, rows, columns):
|
|
12
|
+
"""Reshape an array into rows x columns."""
|
|
13
|
+
return np.asarray(arr).reshape(rows, columns)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def transpose_array(arr):
|
|
17
|
+
"""Return the transpose of an array."""
|
|
18
|
+
return np.asarray(arr).T
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def array_sums(arr):
|
|
22
|
+
"""Return total, row-wise and column-wise sums."""
|
|
23
|
+
a = np.asarray(arr)
|
|
24
|
+
return {
|
|
25
|
+
"total": a.sum(),
|
|
26
|
+
"row_sum": a.sum(axis=1),
|
|
27
|
+
"column_sum": a.sum(axis=0),
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def even_numbers(arr):
|
|
32
|
+
"""Return even elements using Boolean indexing."""
|
|
33
|
+
a = np.asarray(arr)
|
|
34
|
+
return a[a % 2 == 0]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def center_slice(arr):
|
|
38
|
+
"""Return the center 3x3 section of a 5x5-style matrix."""
|
|
39
|
+
a = np.asarray(arr)
|
|
40
|
+
return a[1:4, 1:4]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def reverse_array(arr):
|
|
44
|
+
"""Reverse an array along both dimensions."""
|
|
45
|
+
a = np.asarray(arr)
|
|
46
|
+
return a[::-1, ::-1]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def flatten_array(arr):
|
|
50
|
+
"""Return all elements in one dimension."""
|
|
51
|
+
return np.asarray(arr).flatten()
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Pandas helpers for common lab questions."""
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def series_to_dataframe(name, age, course, marks):
|
|
7
|
+
"""Create a DataFrame from four Series/lists."""
|
|
8
|
+
return pd.DataFrame({
|
|
9
|
+
"Name": name,
|
|
10
|
+
"Age": age,
|
|
11
|
+
"Course": course,
|
|
12
|
+
"Marks": marks,
|
|
13
|
+
})
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def add_pass_fail(df, marks_column="Marks", result_column="Result"):
|
|
17
|
+
"""Add Pass/Fail using marks >= 40."""
|
|
18
|
+
result = df.copy()
|
|
19
|
+
result[result_column] = result[marks_column].apply(
|
|
20
|
+
lambda x: "Pass" if x >= 40 else "Fail"
|
|
21
|
+
)
|
|
22
|
+
return result
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def passed_students(df, result_column="Result"):
|
|
26
|
+
"""Return only rows marked Pass."""
|
|
27
|
+
return df[df[result_column] == "Pass"].copy()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def average_by_group(df, group_column, value_column):
|
|
31
|
+
"""Average value for each group."""
|
|
32
|
+
return df.groupby(group_column)[value_column].mean()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def highest_salary_employee(df, salary_column="Salary"):
|
|
36
|
+
"""Return the row containing the highest salary."""
|
|
37
|
+
return df.loc[df[salary_column].idxmax()]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def concat_rows(df1, df2, ignore_index=True):
|
|
41
|
+
"""Join DataFrames vertically."""
|
|
42
|
+
return pd.concat([df1, df2], ignore_index=ignore_index)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def concat_columns(df1, df2):
|
|
46
|
+
"""Join DataFrames horizontally."""
|
|
47
|
+
return pd.concat([df1, df2], axis=1)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def merge_dataframes(df1, df2, on, how="inner"):
|
|
51
|
+
"""Merge DataFrames on a common column."""
|
|
52
|
+
return pd.merge(df1, df2, on=on, how=how)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def value_counts_all(df):
|
|
56
|
+
"""Count occurrences of complete rows."""
|
|
57
|
+
return df.value_counts()
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def group_age_statistics(df, group_column="school", age_column="age"):
|
|
61
|
+
"""Return mean, min and max age for each group."""
|
|
62
|
+
return df.groupby(group_column)[age_column].agg(
|
|
63
|
+
["mean", "min", "max"]
|
|
64
|
+
)
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"""Matplotlib plotting helpers."""
|
|
2
|
+
|
|
3
|
+
import matplotlib.pyplot as plt
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def line_plot(x, y, title="Line Plot", xlabel="X", ylabel="Y",
|
|
7
|
+
marker=None, rotation=0):
|
|
8
|
+
plt.figure(figsize=(8, 5))
|
|
9
|
+
plt.plot(x, y, marker=marker)
|
|
10
|
+
plt.title(title)
|
|
11
|
+
plt.xlabel(xlabel)
|
|
12
|
+
plt.ylabel(ylabel)
|
|
13
|
+
if rotation:
|
|
14
|
+
plt.xticks(rotation=rotation)
|
|
15
|
+
plt.tight_layout()
|
|
16
|
+
plt.show()
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def bar_plot(x, y, title="Bar Plot", xlabel="X", ylabel="Y",
|
|
20
|
+
rotation=0):
|
|
21
|
+
plt.figure(figsize=(8, 5))
|
|
22
|
+
plt.bar(x, y)
|
|
23
|
+
plt.title(title)
|
|
24
|
+
plt.xlabel(xlabel)
|
|
25
|
+
plt.ylabel(ylabel)
|
|
26
|
+
if rotation:
|
|
27
|
+
plt.xticks(rotation=rotation)
|
|
28
|
+
plt.tight_layout()
|
|
29
|
+
plt.show()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def scatter_plot(x, y, title="Scatter Plot", xlabel="X", ylabel="Y"):
|
|
33
|
+
plt.figure(figsize=(8, 5))
|
|
34
|
+
plt.scatter(x, y)
|
|
35
|
+
plt.title(title)
|
|
36
|
+
plt.xlabel(xlabel)
|
|
37
|
+
plt.ylabel(ylabel)
|
|
38
|
+
plt.tight_layout()
|
|
39
|
+
plt.show()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def histogram(data, bins=10, title="Histogram",
|
|
43
|
+
xlabel="Value", ylabel="Frequency", label=None):
|
|
44
|
+
plt.figure(figsize=(8, 5))
|
|
45
|
+
plt.hist(data, bins=bins, label=label)
|
|
46
|
+
plt.title(title)
|
|
47
|
+
plt.xlabel(xlabel)
|
|
48
|
+
plt.ylabel(ylabel)
|
|
49
|
+
if label:
|
|
50
|
+
plt.legend()
|
|
51
|
+
plt.tight_layout()
|
|
52
|
+
plt.show()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def multiple_plots(x, ys, titles=None, plot_types=None,
|
|
56
|
+
figsize=(10, 8)):
|
|
57
|
+
if titles is None:
|
|
58
|
+
titles = [f"Plot {i+1}" for i in range(len(ys))]
|
|
59
|
+
if plot_types is None:
|
|
60
|
+
plot_types = ["line"] * len(ys)
|
|
61
|
+
|
|
62
|
+
plt.figure(figsize=figsize)
|
|
63
|
+
|
|
64
|
+
for i, (y, title, kind) in enumerate(
|
|
65
|
+
zip(ys, titles, plot_types), start=1
|
|
66
|
+
):
|
|
67
|
+
plt.subplot(2, 2, i)
|
|
68
|
+
|
|
69
|
+
if kind == "line":
|
|
70
|
+
plt.plot(x, y)
|
|
71
|
+
elif kind == "bar":
|
|
72
|
+
plt.bar(x, y)
|
|
73
|
+
elif kind == "scatter":
|
|
74
|
+
plt.scatter(x, y)
|
|
75
|
+
elif kind == "hist":
|
|
76
|
+
plt.hist(y)
|
|
77
|
+
else:
|
|
78
|
+
raise ValueError(
|
|
79
|
+
"plot_types must be line, bar, scatter or hist"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
plt.title(title)
|
|
83
|
+
|
|
84
|
+
plt.tight_layout()
|
|
85
|
+
plt.show()
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def correlation_heatmap(df, figsize=(10, 8),
|
|
89
|
+
title="Correlation Heatmap"):
|
|
90
|
+
corr = df.corr(numeric_only=True)
|
|
91
|
+
|
|
92
|
+
plt.figure(figsize=figsize)
|
|
93
|
+
plt.imshow(corr, aspect="auto")
|
|
94
|
+
plt.colorbar()
|
|
95
|
+
|
|
96
|
+
plt.xticks(
|
|
97
|
+
range(len(corr.columns)),
|
|
98
|
+
corr.columns,
|
|
99
|
+
rotation=90
|
|
100
|
+
)
|
|
101
|
+
plt.yticks(
|
|
102
|
+
range(len(corr.columns)),
|
|
103
|
+
corr.columns
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
plt.title(title)
|
|
107
|
+
plt.tight_layout()
|
|
108
|
+
plt.show()
|
|
109
|
+
|
|
110
|
+
return corr
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def box_plot(df, column, by=None, title="Box Plot"):
|
|
114
|
+
plt.figure(figsize=(8, 5))
|
|
115
|
+
|
|
116
|
+
if by is None:
|
|
117
|
+
plt.boxplot(df[column].dropna())
|
|
118
|
+
plt.ylabel(column)
|
|
119
|
+
else:
|
|
120
|
+
groups = []
|
|
121
|
+
labels = []
|
|
122
|
+
|
|
123
|
+
for name, group in df.groupby(by):
|
|
124
|
+
groups.append(group[column].dropna())
|
|
125
|
+
labels.append(str(name))
|
|
126
|
+
|
|
127
|
+
plt.boxplot(groups)
|
|
128
|
+
plt.xticks(
|
|
129
|
+
range(1, len(labels) + 1),
|
|
130
|
+
labels
|
|
131
|
+
)
|
|
132
|
+
plt.ylabel(column)
|
|
133
|
+
|
|
134
|
+
plt.title(title)
|
|
135
|
+
plt.tight_layout()
|
|
136
|
+
plt.show()
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
if __name__ == "__main__":
|
|
140
|
+
print("pythonlabtools.plotting: All dependencies loaded successfully.")
|
|
141
|
+
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""Data cleaning and preprocessing helpers."""
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.preprocessing import StandardScaler
|
|
5
|
+
from sklearn.model_selection import train_test_split
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def show_missing(df):
|
|
9
|
+
result = df.isnull().sum()
|
|
10
|
+
print(result)
|
|
11
|
+
return result
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def drop_missing(df, axis=0):
|
|
15
|
+
return df.dropna(axis=axis).copy()
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def fill_missing_mean(df, column):
|
|
19
|
+
df = df.copy()
|
|
20
|
+
df[column] = df[column].fillna(df[column].mean())
|
|
21
|
+
return df
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def fill_missing_median(df, column):
|
|
25
|
+
df = df.copy()
|
|
26
|
+
df[column] = df[column].fillna(df[column].median())
|
|
27
|
+
return df
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def fill_missing_mode(df, column):
|
|
31
|
+
df = df.copy()
|
|
32
|
+
mode = df[column].mode()
|
|
33
|
+
if len(mode):
|
|
34
|
+
df[column] = df[column].fillna(mode.iloc[0])
|
|
35
|
+
return df
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def convert_numeric(df, column, errors="coerce"):
|
|
39
|
+
df = df.copy()
|
|
40
|
+
df[column] = pd.to_numeric(df[column], errors=errors)
|
|
41
|
+
return df
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def remove_duplicates(df):
|
|
45
|
+
return df.drop_duplicates().copy()
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def replace_invalid(df, column, condition, replacement=None):
|
|
49
|
+
"""
|
|
50
|
+
Replace values where condition(value) is True.
|
|
51
|
+
Example:
|
|
52
|
+
replace_invalid(df, "Age", lambda x: x < 0)
|
|
53
|
+
"""
|
|
54
|
+
df = df.copy()
|
|
55
|
+
mask = df[column].apply(condition)
|
|
56
|
+
df.loc[mask, column] = replacement
|
|
57
|
+
return df
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def encode_categorical(df, columns=None, drop_first=False):
|
|
61
|
+
df = df.copy()
|
|
62
|
+
if columns is None:
|
|
63
|
+
columns = df.select_dtypes(
|
|
64
|
+
include=["object", "category"]
|
|
65
|
+
).columns.tolist()
|
|
66
|
+
return pd.get_dummies(
|
|
67
|
+
df, columns=columns, drop_first=drop_first, dtype=int
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def scale_features(df, columns):
|
|
72
|
+
df = df.copy()
|
|
73
|
+
scaler = StandardScaler()
|
|
74
|
+
df[columns] = scaler.fit_transform(df[columns])
|
|
75
|
+
return df, scaler
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def train_test_scale(X, y, test_size=0.2, random_state=42, stratify=None):
|
|
79
|
+
X_train, X_test, y_train, y_test = train_test_split(
|
|
80
|
+
X, y,
|
|
81
|
+
test_size=test_size,
|
|
82
|
+
random_state=random_state,
|
|
83
|
+
stratify=stratify
|
|
84
|
+
)
|
|
85
|
+
scaler = StandardScaler()
|
|
86
|
+
X_train_scaled = scaler.fit_transform(X_train)
|
|
87
|
+
X_test_scaled = scaler.transform(X_test)
|
|
88
|
+
return X_train_scaled, X_test_scaled, y_train, y_test, scaler
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def preprocess_basic(
|
|
92
|
+
df,
|
|
93
|
+
fill_numeric="mean",
|
|
94
|
+
encode=True,
|
|
95
|
+
scale_columns=None
|
|
96
|
+
):
|
|
97
|
+
"""Basic cleaning pipeline for lab practice."""
|
|
98
|
+
df = df.copy()
|
|
99
|
+
|
|
100
|
+
numeric_cols = df.select_dtypes(
|
|
101
|
+
include="number"
|
|
102
|
+
).columns.tolist()
|
|
103
|
+
|
|
104
|
+
categorical_cols = df.select_dtypes(
|
|
105
|
+
include=["object", "category"]
|
|
106
|
+
).columns.tolist()
|
|
107
|
+
|
|
108
|
+
for col in numeric_cols:
|
|
109
|
+
if df[col].isnull().any():
|
|
110
|
+
if fill_numeric == "median":
|
|
111
|
+
df[col] = df[col].fillna(df[col].median())
|
|
112
|
+
else:
|
|
113
|
+
df[col] = df[col].fillna(df[col].mean())
|
|
114
|
+
|
|
115
|
+
for col in categorical_cols:
|
|
116
|
+
if df[col].isnull().any():
|
|
117
|
+
mode = df[col].mode()
|
|
118
|
+
if len(mode):
|
|
119
|
+
df[col] = df[col].fillna(mode.iloc[0])
|
|
120
|
+
|
|
121
|
+
if encode and categorical_cols:
|
|
122
|
+
df = pd.get_dummies(df, columns=categorical_cols, dtype=int)
|
|
123
|
+
|
|
124
|
+
if scale_columns:
|
|
125
|
+
existing = [
|
|
126
|
+
c for c in scale_columns
|
|
127
|
+
if c in df.columns
|
|
128
|
+
]
|
|
129
|
+
if existing:
|
|
130
|
+
scaler = StandardScaler()
|
|
131
|
+
df[existing] = scaler.fit_transform(df[existing])
|
|
132
|
+
|
|
133
|
+
return df
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Regression model helpers for Q49-style lab programs."""
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
from sklearn.linear_model import LinearRegression
|
|
5
|
+
from sklearn.tree import DecisionTreeRegressor
|
|
6
|
+
from sklearn.ensemble import RandomForestRegressor
|
|
7
|
+
from sklearn.metrics import (
|
|
8
|
+
r2_score,
|
|
9
|
+
mean_absolute_error,
|
|
10
|
+
mean_squared_error,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def train_regression_models(X_train, X_test, y_train, y_test):
|
|
15
|
+
"""
|
|
16
|
+
Train three common regression models.
|
|
17
|
+
Returns predictions and models.
|
|
18
|
+
"""
|
|
19
|
+
models = {
|
|
20
|
+
"Linear Regression": LinearRegression(),
|
|
21
|
+
"Decision Tree": DecisionTreeRegressor(random_state=42),
|
|
22
|
+
"Random Forest": RandomForestRegressor(
|
|
23
|
+
n_estimators=100,
|
|
24
|
+
random_state=42
|
|
25
|
+
),
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
predictions = {}
|
|
29
|
+
trained_models = {}
|
|
30
|
+
|
|
31
|
+
for name, model in models.items():
|
|
32
|
+
model.fit(X_train, y_train)
|
|
33
|
+
predictions[name] = model.predict(X_test)
|
|
34
|
+
trained_models[name] = model
|
|
35
|
+
|
|
36
|
+
return trained_models, predictions
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def evaluate_regression(y_test, predictions):
|
|
40
|
+
"""Calculate R2, MAE, MSE and RMSE."""
|
|
41
|
+
results = {}
|
|
42
|
+
|
|
43
|
+
for name, pred in predictions.items():
|
|
44
|
+
mse = mean_squared_error(y_test, pred)
|
|
45
|
+
|
|
46
|
+
results[name] = {
|
|
47
|
+
"R2": r2_score(y_test, pred),
|
|
48
|
+
"MAE": mean_absolute_error(y_test, pred),
|
|
49
|
+
"MSE": mse,
|
|
50
|
+
"RMSE": np.sqrt(mse),
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
return results
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pythonlabtools
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Reusable Python lab exam toolkit for NumPy, Pandas, preprocessing, plotting and machine learning.
|
|
5
|
+
Author: JP
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.9
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
Requires-Dist: numpy>=1.23
|
|
10
|
+
Requires-Dist: pandas>=1.5
|
|
11
|
+
Requires-Dist: matplotlib>=3.6
|
|
12
|
+
Requires-Dist: scikit-learn>=1.2
|
|
13
|
+
|
|
14
|
+
# PythonLabTools — Python Lab Exam Toolkit
|
|
15
|
+
|
|
16
|
+
Version 2.0.0
|
|
17
|
+
|
|
18
|
+
This package turns common Python lab-exam programs into reusable functions.
|
|
19
|
+
|
|
20
|
+
## Covered areas
|
|
21
|
+
|
|
22
|
+
- NumPy arrays and slicing
|
|
23
|
+
- Pandas DataFrames, filtering, grouping, joining and merging
|
|
24
|
+
- Missing values and dirty-data preprocessing
|
|
25
|
+
- Categorical encoding
|
|
26
|
+
- Feature scaling and train/test splitting
|
|
27
|
+
- Line, bar, scatter, histogram and multiple plots
|
|
28
|
+
- Correlation heatmaps
|
|
29
|
+
- Regression: Linear Regression, Decision Tree, Random Forest
|
|
30
|
+
- Classification: Logistic Regression, Decision Tree, Random Forest
|
|
31
|
+
- Classification metrics and confusion matrix
|
|
32
|
+
- K-Means clustering and Elbow method
|
|
33
|
+
- PCA
|
|
34
|
+
- Association-rule calculations: support, confidence and lift
|
|
35
|
+
- Simple SQLite CRUD helpers
|
|
36
|
+
- Exam templates
|
|
37
|
+
|
|
38
|
+
## Install locally
|
|
39
|
+
|
|
40
|
+
Extract the ZIP, open a terminal in the extracted folder:
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install .
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Then:
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
from pythonlabtools import *
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Google Colab
|
|
53
|
+
|
|
54
|
+
Upload the ZIP, extract it, and install:
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
!unzip -q /content/pythonlabtools-exam-toolkit-v2.zip -d /content/
|
|
58
|
+
!pip install /content/pythonlabtools_exam_toolkit
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Or after extracting:
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
%cd /content/pythonlabtools_exam_toolkit
|
|
65
|
+
!pip install .
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## Example
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
import pandas as pd
|
|
72
|
+
from pythonlabtools import show_missing, fill_missing_mean
|
|
73
|
+
from pythonlabtools import line_plot
|
|
74
|
+
|
|
75
|
+
df = pd.DataFrame({
|
|
76
|
+
"Day": ["Mon", "Tue", "Wed"],
|
|
77
|
+
"Temperature": [30, None, 32]
|
|
78
|
+
})
|
|
79
|
+
|
|
80
|
+
show_missing(df)
|
|
81
|
+
df = fill_missing_mean(df, "Temperature")
|
|
82
|
+
|
|
83
|
+
line_plot(
|
|
84
|
+
df["Day"],
|
|
85
|
+
df["Temperature"],
|
|
86
|
+
title="Temperature vs Day",
|
|
87
|
+
xlabel="Day",
|
|
88
|
+
ylabel="Temperature"
|
|
89
|
+
)
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## Important exam principle
|
|
93
|
+
|
|
94
|
+
Use the toolkit to save typing, but understand what each function does.
|
|
95
|
+
In a viva, you should be able to explain:
|
|
96
|
+
|
|
97
|
+
- `groupby`
|
|
98
|
+
- `concat`
|
|
99
|
+
- `merge`
|
|
100
|
+
- `fillna`
|
|
101
|
+
- `get_dummies`
|
|
102
|
+
- `StandardScaler`
|
|
103
|
+
- `train_test_split`
|
|
104
|
+
- `fit`
|
|
105
|
+
- `predict`
|
|
106
|
+
- R2 / MAE / MSE / RMSE
|
|
107
|
+
- accuracy / recall / confusion matrix
|
|
108
|
+
- K-Means / inertia / PCA
|
|
109
|
+
- support / confidence / lift
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
pythonlabtools/__init__.py,sha256=j9IVD_niyxORkDr-3glnjKh6JvbgkCfyxZRFY9x6bkM,685
|
|
2
|
+
pythonlabtools/association.py,sha256=iM1Fko-OlwSvbpsYqmBC8f1dEYDBZnR3EV3X43mvK9s,2400
|
|
3
|
+
pythonlabtools/classification.py,sha256=ecUJDd8uhuPWDjcO8Y87UzSunIsTaSsWwTRT_wS5MA0,2315
|
|
4
|
+
pythonlabtools/clustering.py,sha256=9U_RLKyhxRsmMy9EiP0Ca5bBz5ToTbEGcSzARnwAUpA,1410
|
|
5
|
+
pythonlabtools/code_viewer.py,sha256=w6okpilnvc5stvhTBw6QhUd1391YAgjDIWb--9agvbg,7141
|
|
6
|
+
pythonlabtools/database.py,sha256=y2Gpw-cFKH4pessCBWflpxoCbD8auOKx2QIBH1PA3ro,976
|
|
7
|
+
pythonlabtools/numpy_tools.py,sha256=YjW5RKGtSldB_VoKy3OnjIJgq2D_2o2Z2ATEDsPnLtk,1176
|
|
8
|
+
pythonlabtools/pandas_tools.py,sha256=kNS5iWVcyMQreG9l-N_t-63n34VTPEPylCX2qNARyJ0,1743
|
|
9
|
+
pythonlabtools/plotting.py,sha256=nBPAILvEsIjOFZ4PLweCeePn0nkXNcg5_TmBAfr99YY,3208
|
|
10
|
+
pythonlabtools/preprocessing.py,sha256=Nz4-vOT9PFphRFTpS7WbrisXXeffYG6o_l99nYQ2qpU,3321
|
|
11
|
+
pythonlabtools/regression.py,sha256=XCBhC89CocbbM5uF4Wfhj3TcukGQrZl51NC-0OtWCgQ,1392
|
|
12
|
+
pythonlabtools-2.0.0.dist-info/METADATA,sha256=5xiBfKix0f4MCAsFtlWY3wEme-reOEKaDBw4q4DflP4,2486
|
|
13
|
+
pythonlabtools-2.0.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
14
|
+
pythonlabtools-2.0.0.dist-info/top_level.txt,sha256=uxOSGRTQvMM4cnMtmUI0QKMe3VXO9glKbaIvNS8HE2A,15
|
|
15
|
+
pythonlabtools-2.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
pythonlabtools
|