mymllibrary 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mymllibrary-0.1.0/PKG-INFO +52 -0
- mymllibrary-0.1.0/README.md +41 -0
- mymllibrary-0.1.0/myml/Dbscan_clustering.py +213 -0
- mymllibrary-0.1.0/myml/K_mean_clustering.py +187 -0
- mymllibrary-0.1.0/myml/Knn.py +67 -0
- mymllibrary-0.1.0/myml/Linear_regression.py +48 -0
- mymllibrary-0.1.0/myml/Logistic_regression.py +71 -0
- mymllibrary-0.1.0/myml/Multilayer_perceptron.py +344 -0
- mymllibrary-0.1.0/myml/Naive_bayes.py +83 -0
- mymllibrary-0.1.0/myml/Performance_matrix.py +57 -0
- mymllibrary-0.1.0/myml/Pla.py +69 -0
- mymllibrary-0.1.0/myml/Slp.py +104 -0
- mymllibrary-0.1.0/myml/__init__.py +139 -0
- mymllibrary-0.1.0/myml/code_viewer.py +61 -0
- mymllibrary-0.1.0/mymllibrary.egg-info/PKG-INFO +52 -0
- mymllibrary-0.1.0/mymllibrary.egg-info/SOURCES.txt +19 -0
- mymllibrary-0.1.0/mymllibrary.egg-info/dependency_links.txt +1 -0
- mymllibrary-0.1.0/mymllibrary.egg-info/requires.txt +2 -0
- mymllibrary-0.1.0/mymllibrary.egg-info/top_level.txt +1 -0
- mymllibrary-0.1.0/pyproject.toml +25 -0
- mymllibrary-0.1.0/setup.cfg +4 -0
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mymllibrary
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A beginner-friendly Python machine learning library implemented from scratch.
|
|
5
|
+
Author: Ayush Kumar Gour
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.8
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
Requires-Dist: numpy
|
|
10
|
+
Requires-Dist: matplotlib
|
|
11
|
+
|
|
12
|
+
# MyMLLibrary
|
|
13
|
+
|
|
14
|
+
A simple Python Machine Learning library containing machine learning algorithms implemented from scratch for learning and educational purposes.
|
|
15
|
+
|
|
16
|
+
## Features
|
|
17
|
+
|
|
18
|
+
- Machine Learning algorithms implemented from scratch
|
|
19
|
+
- Simple and beginner-friendly implementations
|
|
20
|
+
- No automatic dataset loading
|
|
21
|
+
- No automatic model training or prediction
|
|
22
|
+
- Source code of each algorithm can be displayed directly
|
|
23
|
+
- Designed for learning how ML algorithms work internally
|
|
24
|
+
|
|
25
|
+
## Algorithms
|
|
26
|
+
|
|
27
|
+
Currently available:
|
|
28
|
+
|
|
29
|
+
1. Perceptron Learning Algorithm (PLA)=
|
|
30
|
+
2. Linear Regression
|
|
31
|
+
3. Logistic Regression
|
|
32
|
+
4. K-Nearest Neighbors (KNN)
|
|
33
|
+
5. Naive Bayes
|
|
34
|
+
6. K-Means Clustering
|
|
35
|
+
7. DBSCAN
|
|
36
|
+
8. Single Layer Perceptron (SLP)
|
|
37
|
+
9. Performance matrix
|
|
38
|
+
10. Multilayer Perceptron (MLP)
|
|
39
|
+
|
|
40
|
+
### Supporting Modules
|
|
41
|
+
|
|
42
|
+
- Performance Metrics
|
|
43
|
+
- Code Viewer
|
|
44
|
+
|
|
45
|
+
> More algorithms will be added in future versions.
|
|
46
|
+
|
|
47
|
+
## Installation
|
|
48
|
+
|
|
49
|
+
Clone the repository:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
git clone https://github.com/AYUSHKUMAR0401/MyMLLibrary.git
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# MyMLLibrary
|
|
2
|
+
|
|
3
|
+
A simple Python Machine Learning library containing machine learning algorithms implemented from scratch for learning and educational purposes.
|
|
4
|
+
|
|
5
|
+
## Features
|
|
6
|
+
|
|
7
|
+
- Machine Learning algorithms implemented from scratch
|
|
8
|
+
- Simple and beginner-friendly implementations
|
|
9
|
+
- No automatic dataset loading
|
|
10
|
+
- No automatic model training or prediction
|
|
11
|
+
- Source code of each algorithm can be displayed directly
|
|
12
|
+
- Designed for learning how ML algorithms work internally
|
|
13
|
+
|
|
14
|
+
## Algorithms
|
|
15
|
+
|
|
16
|
+
Currently available:
|
|
17
|
+
|
|
18
|
+
1. Perceptron Learning Algorithm (PLA)=
|
|
19
|
+
2. Linear Regression
|
|
20
|
+
3. Logistic Regression
|
|
21
|
+
4. K-Nearest Neighbors (KNN)
|
|
22
|
+
5. Naive Bayes
|
|
23
|
+
6. K-Means Clustering
|
|
24
|
+
7. DBSCAN
|
|
25
|
+
8. Single Layer Perceptron (SLP)
|
|
26
|
+
9. Performance matrix
|
|
27
|
+
10. Multilayer Perceptron (MLP)
|
|
28
|
+
|
|
29
|
+
### Supporting Modules
|
|
30
|
+
|
|
31
|
+
- Performance Metrics
|
|
32
|
+
- Code Viewer
|
|
33
|
+
|
|
34
|
+
> More algorithms will be added in future versions.
|
|
35
|
+
|
|
36
|
+
## Installation
|
|
37
|
+
|
|
38
|
+
Clone the repository:
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
git clone https://github.com/AYUSHKUMAR0401/MyMLLibrary.git
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
import math
|
|
2
|
+
import matplotlib.pyplot as plt
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
# --------------------------------------------------
|
|
6
|
+
# Function 1: Calculate Euclidean Distance
|
|
7
|
+
# --------------------------------------------------
|
|
8
|
+
|
|
9
|
+
def distance(p1, p2):
|
|
10
|
+
|
|
11
|
+
x1, y1 = p1
|
|
12
|
+
x2, y2 = p2
|
|
13
|
+
|
|
14
|
+
return math.sqrt(
|
|
15
|
+
(x1 - x2) ** 2 +
|
|
16
|
+
(y1 - y2) ** 2
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# --------------------------------------------------
|
|
21
|
+
# Function 2: Find Neighbors of a Point
|
|
22
|
+
# --------------------------------------------------
|
|
23
|
+
|
|
24
|
+
def get_neighbors(points, point, eps):
|
|
25
|
+
|
|
26
|
+
neighbors = []
|
|
27
|
+
|
|
28
|
+
for p in points:
|
|
29
|
+
|
|
30
|
+
if p != point:
|
|
31
|
+
|
|
32
|
+
if distance(
|
|
33
|
+
points[point],
|
|
34
|
+
points[p]
|
|
35
|
+
) <= eps:
|
|
36
|
+
|
|
37
|
+
neighbors.append(p)
|
|
38
|
+
|
|
39
|
+
return neighbors
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
# --------------------------------------------------
|
|
43
|
+
# Function 3: DBSCAN Algorithm
|
|
44
|
+
# --------------------------------------------------
|
|
45
|
+
|
|
46
|
+
def dbscan(points, eps, minPts):
|
|
47
|
+
|
|
48
|
+
visited = set()
|
|
49
|
+
|
|
50
|
+
clusters = []
|
|
51
|
+
|
|
52
|
+
noise = []
|
|
53
|
+
|
|
54
|
+
for point in points:
|
|
55
|
+
|
|
56
|
+
if point in visited:
|
|
57
|
+
continue
|
|
58
|
+
|
|
59
|
+
visited.add(point)
|
|
60
|
+
|
|
61
|
+
neighbors = get_neighbors(
|
|
62
|
+
points,
|
|
63
|
+
point,
|
|
64
|
+
eps
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
# Include the point itself
|
|
68
|
+
# when checking MinPts
|
|
69
|
+
if len(neighbors) + 1 < minPts:
|
|
70
|
+
|
|
71
|
+
noise.append(point)
|
|
72
|
+
|
|
73
|
+
continue
|
|
74
|
+
|
|
75
|
+
# Start a new cluster
|
|
76
|
+
cluster = [point]
|
|
77
|
+
|
|
78
|
+
# Expand cluster
|
|
79
|
+
i = 0
|
|
80
|
+
|
|
81
|
+
while i < len(neighbors):
|
|
82
|
+
|
|
83
|
+
current = neighbors[i]
|
|
84
|
+
|
|
85
|
+
if current not in visited:
|
|
86
|
+
|
|
87
|
+
visited.add(current)
|
|
88
|
+
|
|
89
|
+
current_neighbors = get_neighbors(
|
|
90
|
+
points,
|
|
91
|
+
current,
|
|
92
|
+
eps
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
if len(current_neighbors) + 1 >= minPts:
|
|
96
|
+
|
|
97
|
+
for n in current_neighbors:
|
|
98
|
+
|
|
99
|
+
if n not in neighbors:
|
|
100
|
+
neighbors.append(n)
|
|
101
|
+
|
|
102
|
+
# Add point to cluster
|
|
103
|
+
if current not in cluster:
|
|
104
|
+
|
|
105
|
+
cluster.append(current)
|
|
106
|
+
|
|
107
|
+
i += 1
|
|
108
|
+
|
|
109
|
+
clusters.append(cluster)
|
|
110
|
+
|
|
111
|
+
return clusters, noise
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
# --------------------------------------------------
|
|
115
|
+
# Function 4: Plot DBSCAN Clusters
|
|
116
|
+
# --------------------------------------------------
|
|
117
|
+
|
|
118
|
+
def plot_dbscan(points, clusters, noise):
|
|
119
|
+
|
|
120
|
+
# Plot clusters
|
|
121
|
+
for i, cluster in enumerate(clusters):
|
|
122
|
+
|
|
123
|
+
x = [
|
|
124
|
+
points[p][0]
|
|
125
|
+
for p in cluster
|
|
126
|
+
]
|
|
127
|
+
|
|
128
|
+
y = [
|
|
129
|
+
points[p][1]
|
|
130
|
+
for p in cluster
|
|
131
|
+
]
|
|
132
|
+
|
|
133
|
+
plt.scatter(
|
|
134
|
+
x,
|
|
135
|
+
y,
|
|
136
|
+
label="Cluster " + str(i + 1)
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
# Plot noise / outliers
|
|
140
|
+
if len(noise) > 0:
|
|
141
|
+
|
|
142
|
+
x = [
|
|
143
|
+
points[p][0]
|
|
144
|
+
for p in noise
|
|
145
|
+
]
|
|
146
|
+
|
|
147
|
+
y = [
|
|
148
|
+
points[p][1]
|
|
149
|
+
for p in noise
|
|
150
|
+
]
|
|
151
|
+
|
|
152
|
+
plt.scatter(
|
|
153
|
+
x,
|
|
154
|
+
y,
|
|
155
|
+
label="Noise / Outlier"
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
# Display point names
|
|
159
|
+
for p, (x, y) in points.items():
|
|
160
|
+
|
|
161
|
+
plt.text(
|
|
162
|
+
x + 0.3,
|
|
163
|
+
y + 0.3,
|
|
164
|
+
p
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
plt.xlabel("X")
|
|
168
|
+
plt.ylabel("Y")
|
|
169
|
+
|
|
170
|
+
plt.title("DBSCAN Clustering")
|
|
171
|
+
|
|
172
|
+
plt.legend()
|
|
173
|
+
|
|
174
|
+
plt.grid(True)
|
|
175
|
+
|
|
176
|
+
plt.show()
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
# --------------------------------------------------
|
|
180
|
+
# Function 5: Display DBSCAN Results
|
|
181
|
+
# --------------------------------------------------
|
|
182
|
+
|
|
183
|
+
def display_clusters(clusters, noise):
|
|
184
|
+
|
|
185
|
+
print("Clusters:")
|
|
186
|
+
|
|
187
|
+
for i in range(len(clusters)):
|
|
188
|
+
|
|
189
|
+
print(
|
|
190
|
+
"Cluster",
|
|
191
|
+
i + 1,
|
|
192
|
+
":",
|
|
193
|
+
clusters[i]
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
print(
|
|
197
|
+
"Noise / Outliers:",
|
|
198
|
+
noise
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
# --------------------------------------------------
|
|
203
|
+
# Function 6: Display DBSCAN Source Code
|
|
204
|
+
# --------------------------------------------------
|
|
205
|
+
|
|
206
|
+
def show_dbscan_code():
|
|
207
|
+
|
|
208
|
+
import sys
|
|
209
|
+
from .code_viewer import show_code7
|
|
210
|
+
|
|
211
|
+
module = sys.modules[__name__]
|
|
212
|
+
|
|
213
|
+
show_code7(module)
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import matplotlib.pyplot as plt
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
# --------------------------------------------------
|
|
6
|
+
# Function 1: Calculate Euclidean Distance
|
|
7
|
+
# --------------------------------------------------
|
|
8
|
+
|
|
9
|
+
def distance(point, centroid):
|
|
10
|
+
|
|
11
|
+
return np.sqrt(
|
|
12
|
+
np.sum((point - centroid) ** 2)
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# --------------------------------------------------
|
|
17
|
+
# Function 2: Assign each point to nearest centroid
|
|
18
|
+
# --------------------------------------------------
|
|
19
|
+
|
|
20
|
+
def assign_clusters(data, centroids):
|
|
21
|
+
|
|
22
|
+
clusters = []
|
|
23
|
+
|
|
24
|
+
for point in data:
|
|
25
|
+
|
|
26
|
+
distances = []
|
|
27
|
+
|
|
28
|
+
for centroid in centroids:
|
|
29
|
+
|
|
30
|
+
distances.append(
|
|
31
|
+
distance(point, centroid)
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
# Get index of minimum distance
|
|
35
|
+
cluster = np.argmin(distances)
|
|
36
|
+
|
|
37
|
+
clusters.append(cluster)
|
|
38
|
+
|
|
39
|
+
return np.array(clusters, dtype=int)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
# --------------------------------------------------
|
|
43
|
+
# Function 3: Calculate new centroids
|
|
44
|
+
# --------------------------------------------------
|
|
45
|
+
|
|
46
|
+
def calculate_centroids(data, clusters, k):
|
|
47
|
+
|
|
48
|
+
new_centroids = []
|
|
49
|
+
|
|
50
|
+
for i in range(k):
|
|
51
|
+
|
|
52
|
+
cluster_points = data[clusters == i]
|
|
53
|
+
|
|
54
|
+
centroid = np.mean(
|
|
55
|
+
cluster_points,
|
|
56
|
+
axis=0
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
new_centroids.append(centroid)
|
|
60
|
+
|
|
61
|
+
return np.array(new_centroids)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# --------------------------------------------------
|
|
65
|
+
# Function 4: K-Means Algorithm
|
|
66
|
+
# --------------------------------------------------
|
|
67
|
+
|
|
68
|
+
def kmeans(data, k):
|
|
69
|
+
|
|
70
|
+
# Initial centroids = first k points
|
|
71
|
+
centroids = data[:k].copy()
|
|
72
|
+
|
|
73
|
+
while True:
|
|
74
|
+
|
|
75
|
+
# Assign clusters
|
|
76
|
+
clusters = assign_clusters(
|
|
77
|
+
data,
|
|
78
|
+
centroids
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
# Calculate new centroids
|
|
82
|
+
new_centroids = calculate_centroids(
|
|
83
|
+
data,
|
|
84
|
+
clusters,
|
|
85
|
+
k
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
# Check convergence
|
|
89
|
+
if np.allclose(
|
|
90
|
+
centroids,
|
|
91
|
+
new_centroids
|
|
92
|
+
):
|
|
93
|
+
break
|
|
94
|
+
|
|
95
|
+
centroids = new_centroids
|
|
96
|
+
|
|
97
|
+
return clusters, centroids
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
# --------------------------------------------------
|
|
101
|
+
# Function 5: Display Final Clusters
|
|
102
|
+
# --------------------------------------------------
|
|
103
|
+
|
|
104
|
+
def display_clusters(
|
|
105
|
+
names,
|
|
106
|
+
data,
|
|
107
|
+
clusters,
|
|
108
|
+
centroids
|
|
109
|
+
):
|
|
110
|
+
|
|
111
|
+
print("\n----- FINAL CLUSTERS -----")
|
|
112
|
+
|
|
113
|
+
for i in range(len(centroids)):
|
|
114
|
+
|
|
115
|
+
print("\nCluster", i + 1)
|
|
116
|
+
|
|
117
|
+
for j in range(len(data)):
|
|
118
|
+
|
|
119
|
+
if clusters[j] == i:
|
|
120
|
+
|
|
121
|
+
print(
|
|
122
|
+
names[j],
|
|
123
|
+
"->",
|
|
124
|
+
data[j]
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
print(
|
|
128
|
+
"Centroid =",
|
|
129
|
+
centroids[i]
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
# --------------------------------------------------
|
|
134
|
+
# Function 6: Plot Graph
|
|
135
|
+
# --------------------------------------------------
|
|
136
|
+
|
|
137
|
+
def plot_graph(
|
|
138
|
+
data,
|
|
139
|
+
clusters,
|
|
140
|
+
centroids,
|
|
141
|
+
names
|
|
142
|
+
):
|
|
143
|
+
|
|
144
|
+
for i in range(len(centroids)):
|
|
145
|
+
|
|
146
|
+
points = data[clusters == i]
|
|
147
|
+
|
|
148
|
+
plt.scatter(
|
|
149
|
+
points[:, 0],
|
|
150
|
+
points[:, 1],
|
|
151
|
+
label="Cluster " + str(i + 1)
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
# Display point names
|
|
155
|
+
for i in range(len(data)):
|
|
156
|
+
|
|
157
|
+
plt.annotate(
|
|
158
|
+
names[i],
|
|
159
|
+
(data[i][0], data[i][1])
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
# Display centroids
|
|
163
|
+
plt.scatter(
|
|
164
|
+
centroids[:, 0],
|
|
165
|
+
centroids[:, 1],
|
|
166
|
+
marker="X",
|
|
167
|
+
s=200,
|
|
168
|
+
label="Centroids"
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
plt.xlabel("X")
|
|
172
|
+
plt.ylabel("Y")
|
|
173
|
+
plt.title("K-Means Clustering")
|
|
174
|
+
plt.legend()
|
|
175
|
+
plt.grid(True)
|
|
176
|
+
plt.show()
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
# --------------------------------------------------
|
|
180
|
+
# Function 7: Display K-Means Source Code
|
|
181
|
+
# --------------------------------------------------
|
|
182
|
+
|
|
183
|
+
def show_kmeans_code():
|
|
184
|
+
import sys
|
|
185
|
+
from .code_viewer import show_code6
|
|
186
|
+
module = sys.modules[__name__]
|
|
187
|
+
show_code6(module)
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import math
|
|
2
|
+
from collections import Counter
|
|
3
|
+
|
|
4
|
+
def calculate_distance(x1, y1, x2, y2):
|
|
5
|
+
distance = math.sqrt(
|
|
6
|
+
(x1 - x2) ** 2 +
|
|
7
|
+
(y1 - y2) ** 2
|
|
8
|
+
)
|
|
9
|
+
return distance
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
# --------------------------------------------------
|
|
13
|
+
# Function 2: Calculate distances from test point
|
|
14
|
+
# --------------------------------------------------
|
|
15
|
+
|
|
16
|
+
def calculate_distances(data, test_x, test_y):
|
|
17
|
+
|
|
18
|
+
distances = []
|
|
19
|
+
|
|
20
|
+
for index, row in data.iterrows():
|
|
21
|
+
|
|
22
|
+
distance = calculate_distance(
|
|
23
|
+
test_x,
|
|
24
|
+
test_y,
|
|
25
|
+
row["X"],
|
|
26
|
+
row["Y"]
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
distances.append(
|
|
30
|
+
(row["ID"], distance, row["Label"])
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
return distances
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# --------------------------------------------------
|
|
37
|
+
# Function 3: KNN Classification
|
|
38
|
+
# --------------------------------------------------
|
|
39
|
+
|
|
40
|
+
def knn_classification(data, test_x, test_y, k):
|
|
41
|
+
|
|
42
|
+
distances = calculate_distances(
|
|
43
|
+
data,
|
|
44
|
+
test_x,
|
|
45
|
+
test_y
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
# Sort according to distance
|
|
49
|
+
distances.sort(key=lambda x: x[1])
|
|
50
|
+
|
|
51
|
+
# Select k nearest neighbours
|
|
52
|
+
nearest = distances[:k]
|
|
53
|
+
|
|
54
|
+
# Get labels
|
|
55
|
+
labels = [item[2] for item in nearest]
|
|
56
|
+
|
|
57
|
+
count = Counter(labels)
|
|
58
|
+
prediction = count.most_common(1)[0][0]
|
|
59
|
+
return prediction
|
|
60
|
+
def show_knn_code():
|
|
61
|
+
|
|
62
|
+
import sys
|
|
63
|
+
from .code_viewer import show_code4
|
|
64
|
+
|
|
65
|
+
module = sys.modules[__name__]
|
|
66
|
+
|
|
67
|
+
show_code4(module)
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import matplotlib.pyplot as plt
|
|
3
|
+
from .code_viewer import show_code2
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def linear_regression(X, Y):
|
|
7
|
+
x_mean = np.mean(X)
|
|
8
|
+
y_mean = np.mean(Y)
|
|
9
|
+
|
|
10
|
+
numerator = np.sum((X - x_mean) * (Y - y_mean))
|
|
11
|
+
denominator = np.sum((X - x_mean) ** 2)
|
|
12
|
+
|
|
13
|
+
slope = numerator / denominator
|
|
14
|
+
intercept = y_mean - slope * x_mean
|
|
15
|
+
|
|
16
|
+
Y_pred = intercept + slope * X
|
|
17
|
+
|
|
18
|
+
return intercept, slope, Y_pred
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def error_metrics(Y, Y_pred):
|
|
22
|
+
MSE = np.mean((Y - Y_pred) ** 2)
|
|
23
|
+
RMSE = np.sqrt(MSE)
|
|
24
|
+
MAE = np.mean(np.abs(Y - Y_pred))
|
|
25
|
+
|
|
26
|
+
SS_res = np.sum((Y - Y_pred) ** 2)
|
|
27
|
+
SS_tot = np.sum((Y - np.mean(Y)) ** 2)
|
|
28
|
+
|
|
29
|
+
R2 = 1 - (SS_res / SS_tot)
|
|
30
|
+
|
|
31
|
+
return MSE, RMSE, MAE, R2
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def plot_regression(X, Y, Y_pred):
|
|
35
|
+
plt.scatter(X, Y, label="Actual Data")
|
|
36
|
+
plt.plot(X, Y_pred, label="Regression Line")
|
|
37
|
+
|
|
38
|
+
plt.xlabel("Class_Test_Marks")
|
|
39
|
+
plt.ylabel("Semester_Marks")
|
|
40
|
+
plt.title("Linear Regression")
|
|
41
|
+
|
|
42
|
+
plt.legend()
|
|
43
|
+
plt.grid(True)
|
|
44
|
+
plt.show()
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def show_linear_regression_code():
|
|
48
|
+
show_code2(__import__(__name__, fromlist=["*"]))
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
from .code_viewer import show_code3
|
|
3
|
+
|
|
4
|
+
def sigmoid(z):
|
|
5
|
+
return 1 / (1 + np.exp(-z))
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def hypothesis(X, w, b):
|
|
9
|
+
z = w * X + b
|
|
10
|
+
return sigmoid(z)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def cost_function(X, y, w, b):
|
|
14
|
+
m = len(X)
|
|
15
|
+
|
|
16
|
+
h = hypothesis(X, w, b)
|
|
17
|
+
|
|
18
|
+
h = np.clip(h, 1e-10, 1 - 1e-10)
|
|
19
|
+
|
|
20
|
+
cost = (-1 / m) * np.sum(
|
|
21
|
+
y * np.log(h) + (1 - y) * np.log(1 - h)
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
return cost
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def gradient_descent(X, y, w, b, learning_rate, iterations):
|
|
28
|
+
m = len(X)
|
|
29
|
+
|
|
30
|
+
for i in range(iterations):
|
|
31
|
+
|
|
32
|
+
h = hypothesis(X, w, b)
|
|
33
|
+
|
|
34
|
+
dw = (1 / m) * np.dot(X.T, (h - y))
|
|
35
|
+
|
|
36
|
+
db = (1 / m) * np.sum(h - y)
|
|
37
|
+
|
|
38
|
+
w -= learning_rate * dw
|
|
39
|
+
b -= learning_rate * db
|
|
40
|
+
|
|
41
|
+
return w, b
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def logistic_regression(
|
|
45
|
+
X,
|
|
46
|
+
y,
|
|
47
|
+
learning_rate=0.0001,
|
|
48
|
+
iterations=100000
|
|
49
|
+
):
|
|
50
|
+
|
|
51
|
+
w = 0.0
|
|
52
|
+
b = 0.0
|
|
53
|
+
|
|
54
|
+
w, b = gradient_descent(
|
|
55
|
+
X,
|
|
56
|
+
y,
|
|
57
|
+
w,
|
|
58
|
+
b,
|
|
59
|
+
learning_rate,
|
|
60
|
+
iterations
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
return w, b
|
|
64
|
+
|
|
65
|
+
def show_logistic_regression_code():
|
|
66
|
+
import sys
|
|
67
|
+
from .code_viewer import show_code2
|
|
68
|
+
|
|
69
|
+
module = sys.modules[__name__]
|
|
70
|
+
|
|
71
|
+
show_code3(module)
|