pythonlabtools 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pythonlabtools-2.0.0/PKG-INFO +109 -0
- pythonlabtools-2.0.0/README.md +96 -0
- pythonlabtools-2.0.0/pyproject.toml +25 -0
- pythonlabtools-2.0.0/pythonlabtools/__init__.py +39 -0
- pythonlabtools-2.0.0/pythonlabtools/association.py +98 -0
- pythonlabtools-2.0.0/pythonlabtools/classification.py +92 -0
- pythonlabtools-2.0.0/pythonlabtools/clustering.py +60 -0
- pythonlabtools-2.0.0/pythonlabtools/code_viewer.py +305 -0
- pythonlabtools-2.0.0/pythonlabtools/database.py +49 -0
- pythonlabtools-2.0.0/pythonlabtools/numpy_tools.py +51 -0
- pythonlabtools-2.0.0/pythonlabtools/pandas_tools.py +64 -0
- pythonlabtools-2.0.0/pythonlabtools/plotting.py +141 -0
- pythonlabtools-2.0.0/pythonlabtools/preprocessing.py +133 -0
- pythonlabtools-2.0.0/pythonlabtools/regression.py +53 -0
- pythonlabtools-2.0.0/pythonlabtools.egg-info/PKG-INFO +109 -0
- pythonlabtools-2.0.0/pythonlabtools.egg-info/SOURCES.txt +18 -0
- pythonlabtools-2.0.0/pythonlabtools.egg-info/dependency_links.txt +1 -0
- pythonlabtools-2.0.0/pythonlabtools.egg-info/requires.txt +4 -0
- pythonlabtools-2.0.0/pythonlabtools.egg-info/top_level.txt +1 -0
- pythonlabtools-2.0.0/setup.cfg +4 -0
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pythonlabtools
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Reusable Python lab exam toolkit for NumPy, Pandas, preprocessing, plotting and machine learning.
|
|
5
|
+
Author: JP
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.9
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
Requires-Dist: numpy>=1.23
|
|
10
|
+
Requires-Dist: pandas>=1.5
|
|
11
|
+
Requires-Dist: matplotlib>=3.6
|
|
12
|
+
Requires-Dist: scikit-learn>=1.2
|
|
13
|
+
|
|
14
|
+
# PythonLabTools — Python Lab Exam Toolkit
|
|
15
|
+
|
|
16
|
+
Version 2.0.0
|
|
17
|
+
|
|
18
|
+
This package turns common Python lab-exam programs into reusable functions.
|
|
19
|
+
|
|
20
|
+
## Covered areas
|
|
21
|
+
|
|
22
|
+
- NumPy arrays and slicing
|
|
23
|
+
- Pandas DataFrames, filtering, grouping, joining and merging
|
|
24
|
+
- Missing values and dirty-data preprocessing
|
|
25
|
+
- Categorical encoding
|
|
26
|
+
- Feature scaling and train/test splitting
|
|
27
|
+
- Line, bar, scatter, histogram and multiple plots
|
|
28
|
+
- Correlation heatmaps
|
|
29
|
+
- Regression: Linear Regression, Decision Tree, Random Forest
|
|
30
|
+
- Classification: Logistic Regression, Decision Tree, Random Forest
|
|
31
|
+
- Classification metrics and confusion matrix
|
|
32
|
+
- K-Means clustering and Elbow method
|
|
33
|
+
- PCA
|
|
34
|
+
- Association-rule calculations: support, confidence and lift
|
|
35
|
+
- Simple SQLite CRUD helpers
|
|
36
|
+
- Exam templates
|
|
37
|
+
|
|
38
|
+
## Install locally
|
|
39
|
+
|
|
40
|
+
Extract the ZIP, open a terminal in the extracted folder:
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install .
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Then:
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
from pythonlabtools import *
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Google Colab
|
|
53
|
+
|
|
54
|
+
Upload the ZIP, extract it, and install:
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
!unzip -q /content/pythonlabtools-exam-toolkit-v2.zip -d /content/
|
|
58
|
+
!pip install /content/pythonlabtools_exam_toolkit
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Or after extracting:
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
%cd /content/pythonlabtools_exam_toolkit
|
|
65
|
+
!pip install .
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## Example
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
import pandas as pd
|
|
72
|
+
from pythonlabtools import show_missing, fill_missing_mean
|
|
73
|
+
from pythonlabtools import line_plot
|
|
74
|
+
|
|
75
|
+
df = pd.DataFrame({
|
|
76
|
+
"Day": ["Mon", "Tue", "Wed"],
|
|
77
|
+
"Temperature": [30, None, 32]
|
|
78
|
+
})
|
|
79
|
+
|
|
80
|
+
show_missing(df)
|
|
81
|
+
df = fill_missing_mean(df, "Temperature")
|
|
82
|
+
|
|
83
|
+
line_plot(
|
|
84
|
+
df["Day"],
|
|
85
|
+
df["Temperature"],
|
|
86
|
+
title="Temperature vs Day",
|
|
87
|
+
xlabel="Day",
|
|
88
|
+
ylabel="Temperature"
|
|
89
|
+
)
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## Important exam principle
|
|
93
|
+
|
|
94
|
+
Use the toolkit to save typing, but understand what each function does.
|
|
95
|
+
In a viva, you should be able to explain:
|
|
96
|
+
|
|
97
|
+
- `groupby`
|
|
98
|
+
- `concat`
|
|
99
|
+
- `merge`
|
|
100
|
+
- `fillna`
|
|
101
|
+
- `get_dummies`
|
|
102
|
+
- `StandardScaler`
|
|
103
|
+
- `train_test_split`
|
|
104
|
+
- `fit`
|
|
105
|
+
- `predict`
|
|
106
|
+
- R2 / MAE / MSE / RMSE
|
|
107
|
+
- accuracy / recall / confusion matrix
|
|
108
|
+
- K-Means / inertia / PCA
|
|
109
|
+
- support / confidence / lift
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# PythonLabTools — Python Lab Exam Toolkit
|
|
2
|
+
|
|
3
|
+
Version 2.0.0
|
|
4
|
+
|
|
5
|
+
This package turns common Python lab-exam programs into reusable functions.
|
|
6
|
+
|
|
7
|
+
## Covered areas
|
|
8
|
+
|
|
9
|
+
- NumPy arrays and slicing
|
|
10
|
+
- Pandas DataFrames, filtering, grouping, joining and merging
|
|
11
|
+
- Missing values and dirty-data preprocessing
|
|
12
|
+
- Categorical encoding
|
|
13
|
+
- Feature scaling and train/test splitting
|
|
14
|
+
- Line, bar, scatter, histogram and multiple plots
|
|
15
|
+
- Correlation heatmaps
|
|
16
|
+
- Regression: Linear Regression, Decision Tree, Random Forest
|
|
17
|
+
- Classification: Logistic Regression, Decision Tree, Random Forest
|
|
18
|
+
- Classification metrics and confusion matrix
|
|
19
|
+
- K-Means clustering and Elbow method
|
|
20
|
+
- PCA
|
|
21
|
+
- Association-rule calculations: support, confidence and lift
|
|
22
|
+
- Simple SQLite CRUD helpers
|
|
23
|
+
- Exam templates
|
|
24
|
+
|
|
25
|
+
## Install locally
|
|
26
|
+
|
|
27
|
+
Extract the ZIP, open a terminal in the extracted folder:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pip install .
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Then:
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
from pythonlabtools import *
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Google Colab
|
|
40
|
+
|
|
41
|
+
Upload the ZIP, extract it, and install:
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
!unzip -q /content/pythonlabtools-exam-toolkit-v2.zip -d /content/
|
|
45
|
+
!pip install /content/pythonlabtools_exam_toolkit
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Or after extracting:
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
%cd /content/pythonlabtools_exam_toolkit
|
|
52
|
+
!pip install .
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## Example
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
import pandas as pd
|
|
59
|
+
from pythonlabtools import show_missing, fill_missing_mean
|
|
60
|
+
from pythonlabtools import line_plot
|
|
61
|
+
|
|
62
|
+
df = pd.DataFrame({
|
|
63
|
+
"Day": ["Mon", "Tue", "Wed"],
|
|
64
|
+
"Temperature": [30, None, 32]
|
|
65
|
+
})
|
|
66
|
+
|
|
67
|
+
show_missing(df)
|
|
68
|
+
df = fill_missing_mean(df, "Temperature")
|
|
69
|
+
|
|
70
|
+
line_plot(
|
|
71
|
+
df["Day"],
|
|
72
|
+
df["Temperature"],
|
|
73
|
+
title="Temperature vs Day",
|
|
74
|
+
xlabel="Day",
|
|
75
|
+
ylabel="Temperature"
|
|
76
|
+
)
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
## Important exam principle
|
|
80
|
+
|
|
81
|
+
Use the toolkit to save typing, but understand what each function does.
|
|
82
|
+
In a viva, you should be able to explain:
|
|
83
|
+
|
|
84
|
+
- `groupby`
|
|
85
|
+
- `concat`
|
|
86
|
+
- `merge`
|
|
87
|
+
- `fillna`
|
|
88
|
+
- `get_dummies`
|
|
89
|
+
- `StandardScaler`
|
|
90
|
+
- `train_test_split`
|
|
91
|
+
- `fit`
|
|
92
|
+
- `predict`
|
|
93
|
+
- R2 / MAE / MSE / RMSE
|
|
94
|
+
- accuracy / recall / confusion matrix
|
|
95
|
+
- K-Means / inertia / PCA
|
|
96
|
+
- support / confidence / lift
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pythonlabtools"
|
|
7
|
+
version = "2.0.0"
|
|
8
|
+
description = "Reusable Python lab exam toolkit for NumPy, Pandas, preprocessing, plotting and machine learning."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = {text = "MIT"}
|
|
12
|
+
|
|
13
|
+
authors = [
|
|
14
|
+
{name = "JP"}
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
dependencies = [
|
|
18
|
+
"numpy>=1.23",
|
|
19
|
+
"pandas>=1.5",
|
|
20
|
+
"matplotlib>=3.6",
|
|
21
|
+
"scikit-learn>=1.2"
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
[tool.setuptools.packages.find]
|
|
25
|
+
include = ["pythonlabtools*"]
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
__version__ = "2.0.0"
|
|
2
|
+
|
|
3
|
+
# Core tools
|
|
4
|
+
from .numpy_tools import (
|
|
5
|
+
create_range_array,
|
|
6
|
+
reshape_array,
|
|
7
|
+
transpose_array,
|
|
8
|
+
array_sums,
|
|
9
|
+
even_numbers,
|
|
10
|
+
center_slice,
|
|
11
|
+
reverse_array,
|
|
12
|
+
flatten_array,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
from .pandas_tools import (
|
|
16
|
+
series_to_dataframe,
|
|
17
|
+
add_pass_fail,
|
|
18
|
+
passed_students,
|
|
19
|
+
average_by_group,
|
|
20
|
+
highest_salary_employee,
|
|
21
|
+
concat_rows,
|
|
22
|
+
concat_columns,
|
|
23
|
+
merge_dataframes,
|
|
24
|
+
value_counts_all,
|
|
25
|
+
group_age_statistics,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# Database tools
|
|
29
|
+
from .database import (
|
|
30
|
+
connect_sqlite,
|
|
31
|
+
create_table,
|
|
32
|
+
insert_student,
|
|
33
|
+
select_students,
|
|
34
|
+
update_student,
|
|
35
|
+
delete_student,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
# Code viewer
|
|
39
|
+
from .code_viewer import show_code
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Small dependency-light association-rule helpers."""
|
|
2
|
+
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from itertools import combinations
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def item_counts(transactions):
|
|
9
|
+
"""Count individual items across transactions."""
|
|
10
|
+
counts = Counter()
|
|
11
|
+
|
|
12
|
+
for transaction in transactions:
|
|
13
|
+
for item in transaction:
|
|
14
|
+
counts[item] += 1
|
|
15
|
+
|
|
16
|
+
return counts
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def frequent_items(transactions, min_count=2):
|
|
20
|
+
"""Return items meeting the minimum occurrence count."""
|
|
21
|
+
counts = item_counts(transactions)
|
|
22
|
+
|
|
23
|
+
return {
|
|
24
|
+
item: count
|
|
25
|
+
for item, count in counts.items()
|
|
26
|
+
if count >= min_count
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def support(transactions, itemset):
|
|
31
|
+
"""Return support of an itemset."""
|
|
32
|
+
itemset = set(itemset)
|
|
33
|
+
count = 0
|
|
34
|
+
|
|
35
|
+
for transaction in transactions:
|
|
36
|
+
if itemset.issubset(set(transaction)):
|
|
37
|
+
count += 1
|
|
38
|
+
|
|
39
|
+
return count / len(transactions)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def confidence(transactions, A, B):
|
|
43
|
+
"""Return confidence for A -> B."""
|
|
44
|
+
return support(transactions, list(A) + list(B)) / support(
|
|
45
|
+
transactions, A
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def lift(transactions, A, B):
|
|
50
|
+
"""Return lift for A -> B."""
|
|
51
|
+
return confidence(transactions, A, B) / support(
|
|
52
|
+
transactions, B
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def generate_pair_rules(transactions, min_support=0.1):
|
|
57
|
+
"""Generate two-item directional rules."""
|
|
58
|
+
items = list(item_counts(transactions).keys())
|
|
59
|
+
rules = []
|
|
60
|
+
|
|
61
|
+
for A, B in combinations(items, 2):
|
|
62
|
+
pair_support = support(
|
|
63
|
+
transactions,
|
|
64
|
+
[A, B]
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
if pair_support >= min_support:
|
|
68
|
+
rules.append({
|
|
69
|
+
"Rule": f"{A} -> {B}",
|
|
70
|
+
"Support": pair_support,
|
|
71
|
+
"Confidence": confidence(
|
|
72
|
+
transactions, [A], [B]
|
|
73
|
+
),
|
|
74
|
+
"Lift": lift(
|
|
75
|
+
transactions, [A], [B]
|
|
76
|
+
),
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
rules.append({
|
|
80
|
+
"Rule": f"{B} -> {A}",
|
|
81
|
+
"Support": pair_support,
|
|
82
|
+
"Confidence": confidence(
|
|
83
|
+
transactions, [B], [A]
|
|
84
|
+
),
|
|
85
|
+
"Lift": lift(
|
|
86
|
+
transactions, [B], [A]
|
|
87
|
+
),
|
|
88
|
+
})
|
|
89
|
+
|
|
90
|
+
return pd.DataFrame(rules)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def top_rules_by_lift(rules_df, n=5):
|
|
94
|
+
"""Return top n rules sorted by lift."""
|
|
95
|
+
return rules_df.sort_values(
|
|
96
|
+
"Lift",
|
|
97
|
+
ascending=False
|
|
98
|
+
).head(n)
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Classification helpers for Q50-style lab programs."""
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.tree import DecisionTreeClassifier
|
|
5
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
6
|
+
from sklearn.linear_model import LogisticRegression
|
|
7
|
+
from sklearn.metrics import (
|
|
8
|
+
accuracy_score,
|
|
9
|
+
recall_score,
|
|
10
|
+
confusion_matrix,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def train_classification_models(
|
|
15
|
+
X_train,
|
|
16
|
+
X_test,
|
|
17
|
+
y_train,
|
|
18
|
+
y_test,
|
|
19
|
+
scaled_X_train=None,
|
|
20
|
+
scaled_X_test=None
|
|
21
|
+
):
|
|
22
|
+
"""
|
|
23
|
+
Train Decision Tree, Random Forest and Logistic Regression.
|
|
24
|
+
|
|
25
|
+
Logistic Regression uses scaled data if supplied.
|
|
26
|
+
"""
|
|
27
|
+
dt = DecisionTreeClassifier(random_state=42)
|
|
28
|
+
rf = RandomForestClassifier(
|
|
29
|
+
n_estimators=100,
|
|
30
|
+
random_state=42
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
dt.fit(X_train, y_train)
|
|
34
|
+
rf.fit(X_train, y_train)
|
|
35
|
+
|
|
36
|
+
if scaled_X_train is None:
|
|
37
|
+
scaled_X_train = X_train
|
|
38
|
+
if scaled_X_test is None:
|
|
39
|
+
scaled_X_test = X_test
|
|
40
|
+
|
|
41
|
+
lr = LogisticRegression(max_iter=5000)
|
|
42
|
+
lr.fit(scaled_X_train, y_train)
|
|
43
|
+
|
|
44
|
+
models = {
|
|
45
|
+
"Decision Tree": dt,
|
|
46
|
+
"Random Forest": rf,
|
|
47
|
+
"Logistic Regression": lr,
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
predictions = {
|
|
51
|
+
"Decision Tree": dt.predict(X_test),
|
|
52
|
+
"Random Forest": rf.predict(X_test),
|
|
53
|
+
"Logistic Regression": lr.predict(scaled_X_test),
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
return models, predictions
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def evaluate_classification(y_test, predictions):
|
|
60
|
+
"""Return accuracy, recall and confusion matrix for each model."""
|
|
61
|
+
results = {}
|
|
62
|
+
|
|
63
|
+
for name, pred in predictions.items():
|
|
64
|
+
results[name] = {
|
|
65
|
+
"Accuracy": accuracy_score(y_test, pred),
|
|
66
|
+
"Recall": recall_score(y_test, pred),
|
|
67
|
+
"Confusion Matrix": confusion_matrix(y_test, pred),
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
return results
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def confusion_values(y_test, prediction):
|
|
74
|
+
"""Return TN, FP, FN, TP for binary classification."""
|
|
75
|
+
cm = confusion_matrix(y_test, prediction)
|
|
76
|
+
return tuple(cm.ravel())
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def top_random_forest_features(model, feature_names, n=10):
|
|
80
|
+
"""Return top n features by Random Forest importance."""
|
|
81
|
+
importance = pd.Series(
|
|
82
|
+
model.feature_importances_,
|
|
83
|
+
index=feature_names
|
|
84
|
+
)
|
|
85
|
+
return importance.sort_values(
|
|
86
|
+
ascending=False
|
|
87
|
+
).head(n)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def predict_new_sample(model, sample):
|
|
91
|
+
"""Predict one or more new samples."""
|
|
92
|
+
return model.predict(sample)
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""K-Means and PCA helpers for Q51-style lab programs."""
|
|
2
|
+
|
|
3
|
+
import matplotlib.pyplot as plt
|
|
4
|
+
from sklearn.cluster import KMeans
|
|
5
|
+
from sklearn.decomposition import PCA
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def kmeans_models(X, ks=(2, 3, 4), random_state=42):
|
|
9
|
+
"""Run K-Means for several k values."""
|
|
10
|
+
models = {}
|
|
11
|
+
labels = {}
|
|
12
|
+
|
|
13
|
+
for k in ks:
|
|
14
|
+
model = KMeans(
|
|
15
|
+
n_clusters=k,
|
|
16
|
+
random_state=random_state,
|
|
17
|
+
n_init=10
|
|
18
|
+
)
|
|
19
|
+
labels[k] = model.fit_predict(X)
|
|
20
|
+
models[k] = model
|
|
21
|
+
|
|
22
|
+
return models, labels
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def elbow_method(X, k_range=range(2, 11), random_state=42):
|
|
26
|
+
"""Return k values and K-Means inertia values."""
|
|
27
|
+
ks = list(k_range)
|
|
28
|
+
inertias = []
|
|
29
|
+
|
|
30
|
+
for k in ks:
|
|
31
|
+
model = KMeans(
|
|
32
|
+
n_clusters=k,
|
|
33
|
+
random_state=random_state,
|
|
34
|
+
n_init=10
|
|
35
|
+
)
|
|
36
|
+
model.fit(X)
|
|
37
|
+
inertias.append(model.inertia_)
|
|
38
|
+
|
|
39
|
+
return ks, inertias
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def pca_2d(X):
|
|
43
|
+
"""Reduce features to two PCA components."""
|
|
44
|
+
pca = PCA(n_components=2)
|
|
45
|
+
return pca.fit_transform(X), pca
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def plot_clusters(X_pca, labels, title="K-Means Clusters"):
|
|
49
|
+
"""Plot 2D PCA data colored by cluster label."""
|
|
50
|
+
plt.figure(figsize=(8, 5))
|
|
51
|
+
plt.scatter(
|
|
52
|
+
X_pca[:, 0],
|
|
53
|
+
X_pca[:, 1],
|
|
54
|
+
c=labels
|
|
55
|
+
)
|
|
56
|
+
plt.xlabel("Principal Component 1")
|
|
57
|
+
plt.ylabel("Principal Component 2")
|
|
58
|
+
plt.title(title)
|
|
59
|
+
plt.tight_layout()
|
|
60
|
+
plt.show()
|