pythonlabtools 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,39 @@
1
+ __version__ = "2.0.0"
2
+
3
+ # Core tools
4
+ from .numpy_tools import (
5
+ create_range_array,
6
+ reshape_array,
7
+ transpose_array,
8
+ array_sums,
9
+ even_numbers,
10
+ center_slice,
11
+ reverse_array,
12
+ flatten_array,
13
+ )
14
+
15
+ from .pandas_tools import (
16
+ series_to_dataframe,
17
+ add_pass_fail,
18
+ passed_students,
19
+ average_by_group,
20
+ highest_salary_employee,
21
+ concat_rows,
22
+ concat_columns,
23
+ merge_dataframes,
24
+ value_counts_all,
25
+ group_age_statistics,
26
+ )
27
+
28
+ # Database tools
29
+ from .database import (
30
+ connect_sqlite,
31
+ create_table,
32
+ insert_student,
33
+ select_students,
34
+ update_student,
35
+ delete_student,
36
+ )
37
+
38
+ # Code viewer
39
+ from .code_viewer import show_code
@@ -0,0 +1,98 @@
1
+ """Small dependency-light association-rule helpers."""
2
+
3
+ from collections import Counter
4
+ from itertools import combinations
5
+ import pandas as pd
6
+
7
+
8
+ def item_counts(transactions):
9
+ """Count individual items across transactions."""
10
+ counts = Counter()
11
+
12
+ for transaction in transactions:
13
+ for item in transaction:
14
+ counts[item] += 1
15
+
16
+ return counts
17
+
18
+
19
+ def frequent_items(transactions, min_count=2):
20
+ """Return items meeting the minimum occurrence count."""
21
+ counts = item_counts(transactions)
22
+
23
+ return {
24
+ item: count
25
+ for item, count in counts.items()
26
+ if count >= min_count
27
+ }
28
+
29
+
30
+ def support(transactions, itemset):
31
+ """Return support of an itemset."""
32
+ itemset = set(itemset)
33
+ count = 0
34
+
35
+ for transaction in transactions:
36
+ if itemset.issubset(set(transaction)):
37
+ count += 1
38
+
39
+ return count / len(transactions)
40
+
41
+
42
+ def confidence(transactions, A, B):
43
+ """Return confidence for A -> B."""
44
+ return support(transactions, list(A) + list(B)) / support(
45
+ transactions, A
46
+ )
47
+
48
+
49
+ def lift(transactions, A, B):
50
+ """Return lift for A -> B."""
51
+ return confidence(transactions, A, B) / support(
52
+ transactions, B
53
+ )
54
+
55
+
56
+ def generate_pair_rules(transactions, min_support=0.1):
57
+ """Generate two-item directional rules."""
58
+ items = list(item_counts(transactions).keys())
59
+ rules = []
60
+
61
+ for A, B in combinations(items, 2):
62
+ pair_support = support(
63
+ transactions,
64
+ [A, B]
65
+ )
66
+
67
+ if pair_support >= min_support:
68
+ rules.append({
69
+ "Rule": f"{A} -> {B}",
70
+ "Support": pair_support,
71
+ "Confidence": confidence(
72
+ transactions, [A], [B]
73
+ ),
74
+ "Lift": lift(
75
+ transactions, [A], [B]
76
+ ),
77
+ })
78
+
79
+ rules.append({
80
+ "Rule": f"{B} -> {A}",
81
+ "Support": pair_support,
82
+ "Confidence": confidence(
83
+ transactions, [B], [A]
84
+ ),
85
+ "Lift": lift(
86
+ transactions, [B], [A]
87
+ ),
88
+ })
89
+
90
+ return pd.DataFrame(rules)
91
+
92
+
93
+ def top_rules_by_lift(rules_df, n=5):
94
+ """Return top n rules sorted by lift."""
95
+ return rules_df.sort_values(
96
+ "Lift",
97
+ ascending=False
98
+ ).head(n)
@@ -0,0 +1,92 @@
1
+ """Classification helpers for Q50-style lab programs."""
2
+
3
+ import pandas as pd
4
+ from sklearn.tree import DecisionTreeClassifier
5
+ from sklearn.ensemble import RandomForestClassifier
6
+ from sklearn.linear_model import LogisticRegression
7
+ from sklearn.metrics import (
8
+ accuracy_score,
9
+ recall_score,
10
+ confusion_matrix,
11
+ )
12
+
13
+
14
+ def train_classification_models(
15
+ X_train,
16
+ X_test,
17
+ y_train,
18
+ y_test,
19
+ scaled_X_train=None,
20
+ scaled_X_test=None
21
+ ):
22
+ """
23
+ Train Decision Tree, Random Forest and Logistic Regression.
24
+
25
+ Logistic Regression uses scaled data if supplied.
26
+ """
27
+ dt = DecisionTreeClassifier(random_state=42)
28
+ rf = RandomForestClassifier(
29
+ n_estimators=100,
30
+ random_state=42
31
+ )
32
+
33
+ dt.fit(X_train, y_train)
34
+ rf.fit(X_train, y_train)
35
+
36
+ if scaled_X_train is None:
37
+ scaled_X_train = X_train
38
+ if scaled_X_test is None:
39
+ scaled_X_test = X_test
40
+
41
+ lr = LogisticRegression(max_iter=5000)
42
+ lr.fit(scaled_X_train, y_train)
43
+
44
+ models = {
45
+ "Decision Tree": dt,
46
+ "Random Forest": rf,
47
+ "Logistic Regression": lr,
48
+ }
49
+
50
+ predictions = {
51
+ "Decision Tree": dt.predict(X_test),
52
+ "Random Forest": rf.predict(X_test),
53
+ "Logistic Regression": lr.predict(scaled_X_test),
54
+ }
55
+
56
+ return models, predictions
57
+
58
+
59
+ def evaluate_classification(y_test, predictions):
60
+ """Return accuracy, recall and confusion matrix for each model."""
61
+ results = {}
62
+
63
+ for name, pred in predictions.items():
64
+ results[name] = {
65
+ "Accuracy": accuracy_score(y_test, pred),
66
+ "Recall": recall_score(y_test, pred),
67
+ "Confusion Matrix": confusion_matrix(y_test, pred),
68
+ }
69
+
70
+ return results
71
+
72
+
73
+ def confusion_values(y_test, prediction):
74
+ """Return TN, FP, FN, TP for binary classification."""
75
+ cm = confusion_matrix(y_test, prediction)
76
+ return tuple(cm.ravel())
77
+
78
+
79
+ def top_random_forest_features(model, feature_names, n=10):
80
+ """Return top n features by Random Forest importance."""
81
+ importance = pd.Series(
82
+ model.feature_importances_,
83
+ index=feature_names
84
+ )
85
+ return importance.sort_values(
86
+ ascending=False
87
+ ).head(n)
88
+
89
+
90
+ def predict_new_sample(model, sample):
91
+ """Predict one or more new samples."""
92
+ return model.predict(sample)
@@ -0,0 +1,60 @@
1
+ """K-Means and PCA helpers for Q51-style lab programs."""
2
+
3
+ import matplotlib.pyplot as plt
4
+ from sklearn.cluster import KMeans
5
+ from sklearn.decomposition import PCA
6
+
7
+
8
+ def kmeans_models(X, ks=(2, 3, 4), random_state=42):
9
+ """Run K-Means for several k values."""
10
+ models = {}
11
+ labels = {}
12
+
13
+ for k in ks:
14
+ model = KMeans(
15
+ n_clusters=k,
16
+ random_state=random_state,
17
+ n_init=10
18
+ )
19
+ labels[k] = model.fit_predict(X)
20
+ models[k] = model
21
+
22
+ return models, labels
23
+
24
+
25
+ def elbow_method(X, k_range=range(2, 11), random_state=42):
26
+ """Return k values and K-Means inertia values."""
27
+ ks = list(k_range)
28
+ inertias = []
29
+
30
+ for k in ks:
31
+ model = KMeans(
32
+ n_clusters=k,
33
+ random_state=random_state,
34
+ n_init=10
35
+ )
36
+ model.fit(X)
37
+ inertias.append(model.inertia_)
38
+
39
+ return ks, inertias
40
+
41
+
42
+ def pca_2d(X):
43
+ """Reduce features to two PCA components."""
44
+ pca = PCA(n_components=2)
45
+ return pca.fit_transform(X), pca
46
+
47
+
48
+ def plot_clusters(X_pca, labels, title="K-Means Clusters"):
49
+ """Plot 2D PCA data colored by cluster label."""
50
+ plt.figure(figsize=(8, 5))
51
+ plt.scatter(
52
+ X_pca[:, 0],
53
+ X_pca[:, 1],
54
+ c=labels
55
+ )
56
+ plt.xlabel("Principal Component 1")
57
+ plt.ylabel("Principal Component 2")
58
+ plt.title(title)
59
+ plt.tight_layout()
60
+ plt.show()
@@ -0,0 +1,305 @@
1
+ import inspect
2
+ import importlib
3
+ import pythonlabtools
4
+
5
+
6
+ # Functions that can be displayed directly
7
+ FUNCTION_MODULES = {
8
+ # NumPy
9
+ "create_range_array": "numpy_tools",
10
+ "reshape_array": "numpy_tools",
11
+ "transpose_array": "numpy_tools",
12
+ "array_sums": "numpy_tools",
13
+ "even_numbers": "numpy_tools",
14
+ "center_slice": "numpy_tools",
15
+ "reverse_array": "numpy_tools",
16
+ "flatten_array": "numpy_tools",
17
+
18
+ # Pandas
19
+ "series_to_dataframe": "pandas_tools",
20
+ "add_pass_fail": "pandas_tools",
21
+ "passed_students": "pandas_tools",
22
+ "average_by_group": "pandas_tools",
23
+ "highest_salary_employee": "pandas_tools",
24
+ "concat_rows": "pandas_tools",
25
+ "concat_columns": "pandas_tools",
26
+ "merge_dataframes": "pandas_tools",
27
+ "value_counts_all": "pandas_tools",
28
+ "group_age_statistics": "pandas_tools",
29
+
30
+ # Preprocessing
31
+ "show_missing": "preprocessing",
32
+ "drop_missing": "preprocessing",
33
+ "fill_missing_mean": "preprocessing",
34
+ "fill_missing_median": "preprocessing",
35
+ "fill_missing_mode": "preprocessing",
36
+ "convert_numeric": "preprocessing",
37
+ "remove_duplicates": "preprocessing",
38
+ "replace_invalid": "preprocessing",
39
+ "encode_categorical": "preprocessing",
40
+ "scale_features": "preprocessing",
41
+ "train_test_scale": "preprocessing",
42
+ "preprocess_basic": "preprocessing",
43
+
44
+ # Plotting
45
+ "line_plot": "plotting",
46
+ "bar_plot": "plotting",
47
+ "scatter_plot": "plotting",
48
+ "histogram": "plotting",
49
+ "multiple_plots": "plotting",
50
+ "correlation_heatmap": "plotting",
51
+ "box_plot": "plotting",
52
+
53
+ # Regression
54
+ "train_regression_models": "regression",
55
+ "evaluate_regression": "regression",
56
+
57
+ # Classification
58
+ "train_classification_models": "classification",
59
+ "evaluate_classification": "classification",
60
+ "confusion_values": "classification",
61
+ "top_random_forest_features": "classification",
62
+ "predict_new_sample": "classification",
63
+
64
+ # Clustering
65
+ "kmeans_models": "clustering",
66
+ "elbow_method": "clustering",
67
+ "pca_2d": "clustering",
68
+ "plot_clusters": "clustering",
69
+
70
+ # Association
71
+ "item_counts": "association",
72
+ "frequent_items": "association",
73
+ "support": "association",
74
+ "confidence": "association",
75
+ "lift": "association",
76
+ "generate_pair_rules": "association",
77
+ "top_rules_by_lift": "association",
78
+
79
+ # Database
80
+ "connect_sqlite": "database",
81
+ "create_table": "database",
82
+ "insert_student": "database",
83
+ "select_students": "database",
84
+ "update_student": "database",
85
+ "delete_student": "database",
86
+ }
87
+
88
+
89
+ MODULES = {
90
+ "numpy": "numpy_tools",
91
+ "pandas": "pandas_tools",
92
+ "preprocessing": "preprocessing",
93
+ "plotting": "plotting",
94
+ "regression": "regression",
95
+ "classification": "classification",
96
+ "clustering": "clustering",
97
+ "association": "association",
98
+ "database": "database",
99
+ }
100
+
101
+
102
+ QUESTION_FILES = {
103
+ "q49": "examples.q49_regression_template",
104
+ "q50": "examples.q50_classification_template",
105
+ "q51": "examples.q51_clustering_template",
106
+ "q52": "examples.q52_association_template",
107
+ "q30": "examples.q30_sqlite_template",
108
+ }
109
+
110
+
111
+ def show_code(name):
112
+ """
113
+ Display the source code of a pythonlabtools function,
114
+ module, or exam question template.
115
+
116
+ Examples:
117
+ show_code("fill_missing_mean")
118
+ show_code("preprocessing")
119
+ show_code("q49")
120
+ """
121
+
122
+ name = name.lower().strip()
123
+
124
+ # --------------------------------------------------
125
+ # 1. Show a specific function
126
+ # --------------------------------------------------
127
+ if name in FUNCTION_MODULES:
128
+
129
+ module_name = FUNCTION_MODULES[name]
130
+
131
+ module = importlib.import_module(
132
+ f"pythonlabtools.{module_name}"
133
+ )
134
+
135
+ function = getattr(module, name)
136
+
137
+ print("=" * 70)
138
+ print(f"CODE: {name}")
139
+ print("=" * 70)
140
+
141
+ print(inspect.getsource(function))
142
+
143
+ return
144
+
145
+ # --------------------------------------------------
146
+ # 2. Show an entire module
147
+ # --------------------------------------------------
148
+ if name in MODULES:
149
+
150
+ module_name = MODULES[name]
151
+
152
+ module = importlib.import_module(
153
+ f"pythonlabtools.{module_name}"
154
+ )
155
+
156
+ print("=" * 70)
157
+ print(f"MODULE: {module_name}")
158
+ print("=" * 70)
159
+
160
+ print(inspect.getsource(module))
161
+
162
+ return
163
+
164
+ # --------------------------------------------------
165
+ # 3. Show an exam question template
166
+ # --------------------------------------------------
167
+ if name in QUESTION_FILES:
168
+
169
+ module_name = QUESTION_FILES[name]
170
+
171
+ module = importlib.import_module(module_name)
172
+
173
+ print("=" * 70)
174
+ print(f"EXAM TEMPLATE: {name.upper()}")
175
+ print("=" * 70)
176
+
177
+ print(inspect.getsource(module))
178
+
179
+ return
180
+
181
+ # --------------------------------------------------
182
+ # 4. Help
183
+ # --------------------------------------------------
184
+ if name in ["help", "list", "all"]:
185
+
186
+ print("""
187
+ ============================================================
188
+ PYTHONLABTOOLS CODE VIEWER
189
+ ============================================================
190
+
191
+ NUMPY
192
+ -----
193
+ create_range_array
194
+ reshape_array
195
+ transpose_array
196
+ array_sums
197
+ even_numbers
198
+ center_slice
199
+ reverse_array
200
+ flatten_array
201
+
202
+ PANDAS
203
+ ------
204
+ series_to_dataframe
205
+ add_pass_fail
206
+ passed_students
207
+ average_by_group
208
+ highest_salary_employee
209
+ concat_rows
210
+ concat_columns
211
+ merge_dataframes
212
+ value_counts_all
213
+ group_age_statistics
214
+
215
+ PREPROCESSING
216
+ -------------
217
+ show_missing
218
+ drop_missing
219
+ fill_missing_mean
220
+ fill_missing_median
221
+ fill_missing_mode
222
+ convert_numeric
223
+ remove_duplicates
224
+ replace_invalid
225
+ encode_categorical
226
+ scale_features
227
+ train_test_scale
228
+ preprocess_basic
229
+
230
+ PLOTTING
231
+ --------
232
+ line_plot
233
+ bar_plot
234
+ scatter_plot
235
+ histogram
236
+ multiple_plots
237
+ correlation_heatmap
238
+ box_plot
239
+
240
+ MACHINE LEARNING
241
+ ----------------
242
+ train_regression_models
243
+ evaluate_regression
244
+ train_classification_models
245
+ evaluate_classification
246
+ confusion_values
247
+ top_random_forest_features
248
+ predict_new_sample
249
+
250
+ CLUSTERING
251
+ ----------
252
+ kmeans_models
253
+ elbow_method
254
+ pca_2d
255
+ plot_clusters
256
+
257
+ ASSOCIATION RULES
258
+ -----------------
259
+ item_counts
260
+ frequent_items
261
+ support
262
+ confidence
263
+ lift
264
+ generate_pair_rules
265
+ top_rules_by_lift
266
+
267
+ DATABASE
268
+ --------
269
+ connect_sqlite
270
+ create_table
271
+ insert_student
272
+ select_students
273
+ update_student
274
+ delete_student
275
+
276
+ EXAM TEMPLATES
277
+ --------------
278
+ q30
279
+ q49
280
+ q50
281
+ q51
282
+ q52
283
+
284
+ EXAMPLES
285
+ --------
286
+ show_code("fill_missing_mean")
287
+ show_code("preprocessing")
288
+ show_code("q49")
289
+ show_code("help")
290
+
291
+ ============================================================
292
+ """)
293
+
294
+ return
295
+
296
+ # --------------------------------------------------
297
+ # 5. Invalid name
298
+ # --------------------------------------------------
299
+
300
+ print(f"'{name}' was not found.")
301
+
302
+ print("\nType:")
303
+ print(' show_code("help")')
304
+
305
+ print("to see the available functions.")
@@ -0,0 +1,49 @@
1
+ """Simple SQLite helpers for the database portion of the lab."""
2
+
3
+ import sqlite3
4
+
5
+
6
+ def connect_sqlite(path="students.db"):
7
+ return sqlite3.connect(path)
8
+
9
+
10
+ def create_table(conn):
11
+ conn.execute("""
12
+ CREATE TABLE IF NOT EXISTS students (
13
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
14
+ name TEXT,
15
+ marks INTEGER
16
+ )
17
+ """)
18
+ conn.commit()
19
+
20
+
21
+ def insert_student(conn, name, marks):
22
+ conn.execute(
23
+ "INSERT INTO students (name, marks) VALUES (?, ?)",
24
+ (name, marks)
25
+ )
26
+ conn.commit()
27
+
28
+
29
+ def select_students(conn):
30
+ cursor = conn.execute(
31
+ "SELECT * FROM students"
32
+ )
33
+ return cursor.fetchall()
34
+
35
+
36
+ def update_student(conn, name, new_marks):
37
+ conn.execute(
38
+ "UPDATE students SET marks=? WHERE name=?",
39
+ (new_marks, name)
40
+ )
41
+ conn.commit()
42
+
43
+
44
+ def delete_student(conn, name):
45
+ conn.execute(
46
+ "DELETE FROM students WHERE name=?",
47
+ (name,)
48
+ )
49
+ conn.commit()
@@ -0,0 +1,51 @@
1
+ """NumPy helpers matching common lab questions."""
2
+
3
+ import numpy as np
4
+
5
+
6
+ def create_range_array(start=1, stop=20):
7
+ """Create a 1-D array from start through stop, inclusive."""
8
+ return np.arange(start, stop + 1)
9
+
10
+
11
+ def reshape_array(arr, rows, columns):
12
+ """Reshape an array into rows x columns."""
13
+ return np.asarray(arr).reshape(rows, columns)
14
+
15
+
16
+ def transpose_array(arr):
17
+ """Return the transpose of an array."""
18
+ return np.asarray(arr).T
19
+
20
+
21
+ def array_sums(arr):
22
+ """Return total, row-wise and column-wise sums."""
23
+ a = np.asarray(arr)
24
+ return {
25
+ "total": a.sum(),
26
+ "row_sum": a.sum(axis=1),
27
+ "column_sum": a.sum(axis=0),
28
+ }
29
+
30
+
31
+ def even_numbers(arr):
32
+ """Return even elements using Boolean indexing."""
33
+ a = np.asarray(arr)
34
+ return a[a % 2 == 0]
35
+
36
+
37
+ def center_slice(arr):
38
+ """Return the center 3x3 section of a 5x5-style matrix."""
39
+ a = np.asarray(arr)
40
+ return a[1:4, 1:4]
41
+
42
+
43
+ def reverse_array(arr):
44
+ """Reverse an array along both dimensions."""
45
+ a = np.asarray(arr)
46
+ return a[::-1, ::-1]
47
+
48
+
49
+ def flatten_array(arr):
50
+ """Return all elements in one dimension."""
51
+ return np.asarray(arr).flatten()
@@ -0,0 +1,64 @@
1
+ """Pandas helpers for common lab questions."""
2
+
3
+ import pandas as pd
4
+
5
+
6
+ def series_to_dataframe(name, age, course, marks):
7
+ """Create a DataFrame from four Series/lists."""
8
+ return pd.DataFrame({
9
+ "Name": name,
10
+ "Age": age,
11
+ "Course": course,
12
+ "Marks": marks,
13
+ })
14
+
15
+
16
+ def add_pass_fail(df, marks_column="Marks", result_column="Result"):
17
+ """Add Pass/Fail using marks >= 40."""
18
+ result = df.copy()
19
+ result[result_column] = result[marks_column].apply(
20
+ lambda x: "Pass" if x >= 40 else "Fail"
21
+ )
22
+ return result
23
+
24
+
25
+ def passed_students(df, result_column="Result"):
26
+ """Return only rows marked Pass."""
27
+ return df[df[result_column] == "Pass"].copy()
28
+
29
+
30
+ def average_by_group(df, group_column, value_column):
31
+ """Average value for each group."""
32
+ return df.groupby(group_column)[value_column].mean()
33
+
34
+
35
+ def highest_salary_employee(df, salary_column="Salary"):
36
+ """Return the row containing the highest salary."""
37
+ return df.loc[df[salary_column].idxmax()]
38
+
39
+
40
+ def concat_rows(df1, df2, ignore_index=True):
41
+ """Join DataFrames vertically."""
42
+ return pd.concat([df1, df2], ignore_index=ignore_index)
43
+
44
+
45
+ def concat_columns(df1, df2):
46
+ """Join DataFrames horizontally."""
47
+ return pd.concat([df1, df2], axis=1)
48
+
49
+
50
+ def merge_dataframes(df1, df2, on, how="inner"):
51
+ """Merge DataFrames on a common column."""
52
+ return pd.merge(df1, df2, on=on, how=how)
53
+
54
+
55
+ def value_counts_all(df):
56
+ """Count occurrences of complete rows."""
57
+ return df.value_counts()
58
+
59
+
60
+ def group_age_statistics(df, group_column="school", age_column="age"):
61
+ """Return mean, min and max age for each group."""
62
+ return df.groupby(group_column)[age_column].agg(
63
+ ["mean", "min", "max"]
64
+ )
@@ -0,0 +1,141 @@
1
+ """Matplotlib plotting helpers."""
2
+
3
+ import matplotlib.pyplot as plt
4
+
5
+
6
+ def line_plot(x, y, title="Line Plot", xlabel="X", ylabel="Y",
7
+ marker=None, rotation=0):
8
+ plt.figure(figsize=(8, 5))
9
+ plt.plot(x, y, marker=marker)
10
+ plt.title(title)
11
+ plt.xlabel(xlabel)
12
+ plt.ylabel(ylabel)
13
+ if rotation:
14
+ plt.xticks(rotation=rotation)
15
+ plt.tight_layout()
16
+ plt.show()
17
+
18
+
19
+ def bar_plot(x, y, title="Bar Plot", xlabel="X", ylabel="Y",
20
+ rotation=0):
21
+ plt.figure(figsize=(8, 5))
22
+ plt.bar(x, y)
23
+ plt.title(title)
24
+ plt.xlabel(xlabel)
25
+ plt.ylabel(ylabel)
26
+ if rotation:
27
+ plt.xticks(rotation=rotation)
28
+ plt.tight_layout()
29
+ plt.show()
30
+
31
+
32
+ def scatter_plot(x, y, title="Scatter Plot", xlabel="X", ylabel="Y"):
33
+ plt.figure(figsize=(8, 5))
34
+ plt.scatter(x, y)
35
+ plt.title(title)
36
+ plt.xlabel(xlabel)
37
+ plt.ylabel(ylabel)
38
+ plt.tight_layout()
39
+ plt.show()
40
+
41
+
42
+ def histogram(data, bins=10, title="Histogram",
43
+ xlabel="Value", ylabel="Frequency", label=None):
44
+ plt.figure(figsize=(8, 5))
45
+ plt.hist(data, bins=bins, label=label)
46
+ plt.title(title)
47
+ plt.xlabel(xlabel)
48
+ plt.ylabel(ylabel)
49
+ if label:
50
+ plt.legend()
51
+ plt.tight_layout()
52
+ plt.show()
53
+
54
+
55
+ def multiple_plots(x, ys, titles=None, plot_types=None,
56
+ figsize=(10, 8)):
57
+ if titles is None:
58
+ titles = [f"Plot {i+1}" for i in range(len(ys))]
59
+ if plot_types is None:
60
+ plot_types = ["line"] * len(ys)
61
+
62
+ plt.figure(figsize=figsize)
63
+
64
+ for i, (y, title, kind) in enumerate(
65
+ zip(ys, titles, plot_types), start=1
66
+ ):
67
+ plt.subplot(2, 2, i)
68
+
69
+ if kind == "line":
70
+ plt.plot(x, y)
71
+ elif kind == "bar":
72
+ plt.bar(x, y)
73
+ elif kind == "scatter":
74
+ plt.scatter(x, y)
75
+ elif kind == "hist":
76
+ plt.hist(y)
77
+ else:
78
+ raise ValueError(
79
+ "plot_types must be line, bar, scatter or hist"
80
+ )
81
+
82
+ plt.title(title)
83
+
84
+ plt.tight_layout()
85
+ plt.show()
86
+
87
+
88
+ def correlation_heatmap(df, figsize=(10, 8),
89
+ title="Correlation Heatmap"):
90
+ corr = df.corr(numeric_only=True)
91
+
92
+ plt.figure(figsize=figsize)
93
+ plt.imshow(corr, aspect="auto")
94
+ plt.colorbar()
95
+
96
+ plt.xticks(
97
+ range(len(corr.columns)),
98
+ corr.columns,
99
+ rotation=90
100
+ )
101
+ plt.yticks(
102
+ range(len(corr.columns)),
103
+ corr.columns
104
+ )
105
+
106
+ plt.title(title)
107
+ plt.tight_layout()
108
+ plt.show()
109
+
110
+ return corr
111
+
112
+
113
+ def box_plot(df, column, by=None, title="Box Plot"):
114
+ plt.figure(figsize=(8, 5))
115
+
116
+ if by is None:
117
+ plt.boxplot(df[column].dropna())
118
+ plt.ylabel(column)
119
+ else:
120
+ groups = []
121
+ labels = []
122
+
123
+ for name, group in df.groupby(by):
124
+ groups.append(group[column].dropna())
125
+ labels.append(str(name))
126
+
127
+ plt.boxplot(groups)
128
+ plt.xticks(
129
+ range(1, len(labels) + 1),
130
+ labels
131
+ )
132
+ plt.ylabel(column)
133
+
134
+ plt.title(title)
135
+ plt.tight_layout()
136
+ plt.show()
137
+
138
+
139
+ if __name__ == "__main__":
140
+ print("pythonlabtools.plotting: All dependencies loaded successfully.")
141
+
@@ -0,0 +1,133 @@
1
+ """Data cleaning and preprocessing helpers."""
2
+
3
+ import pandas as pd
4
+ from sklearn.preprocessing import StandardScaler
5
+ from sklearn.model_selection import train_test_split
6
+
7
+
8
+ def show_missing(df):
9
+ result = df.isnull().sum()
10
+ print(result)
11
+ return result
12
+
13
+
14
+ def drop_missing(df, axis=0):
15
+ return df.dropna(axis=axis).copy()
16
+
17
+
18
+ def fill_missing_mean(df, column):
19
+ df = df.copy()
20
+ df[column] = df[column].fillna(df[column].mean())
21
+ return df
22
+
23
+
24
+ def fill_missing_median(df, column):
25
+ df = df.copy()
26
+ df[column] = df[column].fillna(df[column].median())
27
+ return df
28
+
29
+
30
+ def fill_missing_mode(df, column):
31
+ df = df.copy()
32
+ mode = df[column].mode()
33
+ if len(mode):
34
+ df[column] = df[column].fillna(mode.iloc[0])
35
+ return df
36
+
37
+
38
+ def convert_numeric(df, column, errors="coerce"):
39
+ df = df.copy()
40
+ df[column] = pd.to_numeric(df[column], errors=errors)
41
+ return df
42
+
43
+
44
+ def remove_duplicates(df):
45
+ return df.drop_duplicates().copy()
46
+
47
+
48
+ def replace_invalid(df, column, condition, replacement=None):
49
+ """
50
+ Replace values where condition(value) is True.
51
+ Example:
52
+ replace_invalid(df, "Age", lambda x: x < 0)
53
+ """
54
+ df = df.copy()
55
+ mask = df[column].apply(condition)
56
+ df.loc[mask, column] = replacement
57
+ return df
58
+
59
+
60
+ def encode_categorical(df, columns=None, drop_first=False):
61
+ df = df.copy()
62
+ if columns is None:
63
+ columns = df.select_dtypes(
64
+ include=["object", "category"]
65
+ ).columns.tolist()
66
+ return pd.get_dummies(
67
+ df, columns=columns, drop_first=drop_first, dtype=int
68
+ )
69
+
70
+
71
+ def scale_features(df, columns):
72
+ df = df.copy()
73
+ scaler = StandardScaler()
74
+ df[columns] = scaler.fit_transform(df[columns])
75
+ return df, scaler
76
+
77
+
78
+ def train_test_scale(X, y, test_size=0.2, random_state=42, stratify=None):
79
+ X_train, X_test, y_train, y_test = train_test_split(
80
+ X, y,
81
+ test_size=test_size,
82
+ random_state=random_state,
83
+ stratify=stratify
84
+ )
85
+ scaler = StandardScaler()
86
+ X_train_scaled = scaler.fit_transform(X_train)
87
+ X_test_scaled = scaler.transform(X_test)
88
+ return X_train_scaled, X_test_scaled, y_train, y_test, scaler
89
+
90
+
91
+ def preprocess_basic(
92
+ df,
93
+ fill_numeric="mean",
94
+ encode=True,
95
+ scale_columns=None
96
+ ):
97
+ """Basic cleaning pipeline for lab practice."""
98
+ df = df.copy()
99
+
100
+ numeric_cols = df.select_dtypes(
101
+ include="number"
102
+ ).columns.tolist()
103
+
104
+ categorical_cols = df.select_dtypes(
105
+ include=["object", "category"]
106
+ ).columns.tolist()
107
+
108
+ for col in numeric_cols:
109
+ if df[col].isnull().any():
110
+ if fill_numeric == "median":
111
+ df[col] = df[col].fillna(df[col].median())
112
+ else:
113
+ df[col] = df[col].fillna(df[col].mean())
114
+
115
+ for col in categorical_cols:
116
+ if df[col].isnull().any():
117
+ mode = df[col].mode()
118
+ if len(mode):
119
+ df[col] = df[col].fillna(mode.iloc[0])
120
+
121
+ if encode and categorical_cols:
122
+ df = pd.get_dummies(df, columns=categorical_cols, dtype=int)
123
+
124
+ if scale_columns:
125
+ existing = [
126
+ c for c in scale_columns
127
+ if c in df.columns
128
+ ]
129
+ if existing:
130
+ scaler = StandardScaler()
131
+ df[existing] = scaler.fit_transform(df[existing])
132
+
133
+ return df
@@ -0,0 +1,53 @@
1
+ """Regression model helpers for Q49-style lab programs."""
2
+
3
+ import numpy as np
4
+ from sklearn.linear_model import LinearRegression
5
+ from sklearn.tree import DecisionTreeRegressor
6
+ from sklearn.ensemble import RandomForestRegressor
7
+ from sklearn.metrics import (
8
+ r2_score,
9
+ mean_absolute_error,
10
+ mean_squared_error,
11
+ )
12
+
13
+
14
+ def train_regression_models(X_train, X_test, y_train, y_test):
15
+ """
16
+ Train three common regression models.
17
+ Returns predictions and models.
18
+ """
19
+ models = {
20
+ "Linear Regression": LinearRegression(),
21
+ "Decision Tree": DecisionTreeRegressor(random_state=42),
22
+ "Random Forest": RandomForestRegressor(
23
+ n_estimators=100,
24
+ random_state=42
25
+ ),
26
+ }
27
+
28
+ predictions = {}
29
+ trained_models = {}
30
+
31
+ for name, model in models.items():
32
+ model.fit(X_train, y_train)
33
+ predictions[name] = model.predict(X_test)
34
+ trained_models[name] = model
35
+
36
+ return trained_models, predictions
37
+
38
+
39
+ def evaluate_regression(y_test, predictions):
40
+ """Calculate R2, MAE, MSE and RMSE."""
41
+ results = {}
42
+
43
+ for name, pred in predictions.items():
44
+ mse = mean_squared_error(y_test, pred)
45
+
46
+ results[name] = {
47
+ "R2": r2_score(y_test, pred),
48
+ "MAE": mean_absolute_error(y_test, pred),
49
+ "MSE": mse,
50
+ "RMSE": np.sqrt(mse),
51
+ }
52
+
53
+ return results
@@ -0,0 +1,109 @@
1
+ Metadata-Version: 2.4
2
+ Name: pythonlabtools
3
+ Version: 2.0.0
4
+ Summary: Reusable Python lab exam toolkit for NumPy, Pandas, preprocessing, plotting and machine learning.
5
+ Author: JP
6
+ License: MIT
7
+ Requires-Python: >=3.9
8
+ Description-Content-Type: text/markdown
9
+ Requires-Dist: numpy>=1.23
10
+ Requires-Dist: pandas>=1.5
11
+ Requires-Dist: matplotlib>=3.6
12
+ Requires-Dist: scikit-learn>=1.2
13
+
14
+ # PythonLabTools — Python Lab Exam Toolkit
15
+
16
+ Version 2.0.0
17
+
18
+ This package turns common Python lab-exam programs into reusable functions.
19
+
20
+ ## Covered areas
21
+
22
+ - NumPy arrays and slicing
23
+ - Pandas DataFrames, filtering, grouping, joining and merging
24
+ - Missing values and dirty-data preprocessing
25
+ - Categorical encoding
26
+ - Feature scaling and train/test splitting
27
+ - Line, bar, scatter, histogram and multiple plots
28
+ - Correlation heatmaps
29
+ - Regression: Linear Regression, Decision Tree, Random Forest
30
+ - Classification: Logistic Regression, Decision Tree, Random Forest
31
+ - Classification metrics and confusion matrix
32
+ - K-Means clustering and Elbow method
33
+ - PCA
34
+ - Association-rule calculations: support, confidence and lift
35
+ - Simple SQLite CRUD helpers
36
+ - Exam templates
37
+
38
+ ## Install locally
39
+
40
+ Extract the ZIP, open a terminal in the extracted folder:
41
+
42
+ ```bash
43
+ pip install .
44
+ ```
45
+
46
+ Then:
47
+
48
+ ```python
49
+ from pythonlabtools import *
50
+ ```
51
+
52
+ ## Google Colab
53
+
54
+ Upload the ZIP, extract it, and install:
55
+
56
+ ```python
57
+ !unzip -q /content/pythonlabtools-exam-toolkit-v2.zip -d /content/
58
+ !pip install /content/pythonlabtools_exam_toolkit
59
+ ```
60
+
61
+ Or after extracting:
62
+
63
+ ```python
64
+ %cd /content/pythonlabtools_exam_toolkit
65
+ !pip install .
66
+ ```
67
+
68
+ ## Example
69
+
70
+ ```python
71
+ import pandas as pd
72
+ from pythonlabtools import show_missing, fill_missing_mean
73
+ from pythonlabtools import line_plot
74
+
75
+ df = pd.DataFrame({
76
+ "Day": ["Mon", "Tue", "Wed"],
77
+ "Temperature": [30, None, 32]
78
+ })
79
+
80
+ show_missing(df)
81
+ df = fill_missing_mean(df, "Temperature")
82
+
83
+ line_plot(
84
+ df["Day"],
85
+ df["Temperature"],
86
+ title="Temperature vs Day",
87
+ xlabel="Day",
88
+ ylabel="Temperature"
89
+ )
90
+ ```
91
+
92
+ ## Important exam principle
93
+
94
+ Use the toolkit to save typing, but understand what each function does.
95
+ In a viva, you should be able to explain:
96
+
97
+ - `groupby`
98
+ - `concat`
99
+ - `merge`
100
+ - `fillna`
101
+ - `get_dummies`
102
+ - `StandardScaler`
103
+ - `train_test_split`
104
+ - `fit`
105
+ - `predict`
106
+ - R2 / MAE / MSE / RMSE
107
+ - accuracy / recall / confusion matrix
108
+ - K-Means / inertia / PCA
109
+ - support / confidence / lift
@@ -0,0 +1,15 @@
1
+ pythonlabtools/__init__.py,sha256=j9IVD_niyxORkDr-3glnjKh6JvbgkCfyxZRFY9x6bkM,685
2
+ pythonlabtools/association.py,sha256=iM1Fko-OlwSvbpsYqmBC8f1dEYDBZnR3EV3X43mvK9s,2400
3
+ pythonlabtools/classification.py,sha256=ecUJDd8uhuPWDjcO8Y87UzSunIsTaSsWwTRT_wS5MA0,2315
4
+ pythonlabtools/clustering.py,sha256=9U_RLKyhxRsmMy9EiP0Ca5bBz5ToTbEGcSzARnwAUpA,1410
5
+ pythonlabtools/code_viewer.py,sha256=w6okpilnvc5stvhTBw6QhUd1391YAgjDIWb--9agvbg,7141
6
+ pythonlabtools/database.py,sha256=y2Gpw-cFKH4pessCBWflpxoCbD8auOKx2QIBH1PA3ro,976
7
+ pythonlabtools/numpy_tools.py,sha256=YjW5RKGtSldB_VoKy3OnjIJgq2D_2o2Z2ATEDsPnLtk,1176
8
+ pythonlabtools/pandas_tools.py,sha256=kNS5iWVcyMQreG9l-N_t-63n34VTPEPylCX2qNARyJ0,1743
9
+ pythonlabtools/plotting.py,sha256=nBPAILvEsIjOFZ4PLweCeePn0nkXNcg5_TmBAfr99YY,3208
10
+ pythonlabtools/preprocessing.py,sha256=Nz4-vOT9PFphRFTpS7WbrisXXeffYG6o_l99nYQ2qpU,3321
11
+ pythonlabtools/regression.py,sha256=XCBhC89CocbbM5uF4Wfhj3TcukGQrZl51NC-0OtWCgQ,1392
12
+ pythonlabtools-2.0.0.dist-info/METADATA,sha256=5xiBfKix0f4MCAsFtlWY3wEme-reOEKaDBw4q4DflP4,2486
13
+ pythonlabtools-2.0.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
14
+ pythonlabtools-2.0.0.dist-info/top_level.txt,sha256=uxOSGRTQvMM4cnMtmUI0QKMe3VXO9glKbaIvNS8HE2A,15
15
+ pythonlabtools-2.0.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1 @@
1
+ pythonlabtools