mymllibrary 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,52 @@
1
+ Metadata-Version: 2.4
2
+ Name: mymllibrary
3
+ Version: 0.1.0
4
+ Summary: A beginner-friendly Python machine learning library implemented from scratch.
5
+ Author: Ayush Kumar Gour
6
+ License: MIT
7
+ Requires-Python: >=3.8
8
+ Description-Content-Type: text/markdown
9
+ Requires-Dist: numpy
10
+ Requires-Dist: matplotlib
11
+
12
+ # MyMLLibrary
13
+
14
+ A simple Python Machine Learning library containing machine learning algorithms implemented from scratch for learning and educational purposes.
15
+
16
+ ## Features
17
+
18
+ - Machine Learning algorithms implemented from scratch
19
+ - Simple and beginner-friendly implementations
20
+ - No automatic dataset loading
21
+ - No automatic model training or prediction
22
+ - Source code of each algorithm can be displayed directly
23
+ - Designed for learning how ML algorithms work internally
24
+
25
+ ## Algorithms
26
+
27
+ Currently available:
28
+
29
+ 1. Perceptron Learning Algorithm (PLA)=
30
+ 2. Linear Regression
31
+ 3. Logistic Regression
32
+ 4. K-Nearest Neighbors (KNN)
33
+ 5. Naive Bayes
34
+ 6. K-Means Clustering
35
+ 7. DBSCAN
36
+ 8. Single Layer Perceptron (SLP)
37
+ 9. Performance matrix
38
+ 10. Multilayer Perceptron (MLP)
39
+
40
+ ### Supporting Modules
41
+
42
+ - Performance Metrics
43
+ - Code Viewer
44
+
45
+ > More algorithms will be added in future versions.
46
+
47
+ ## Installation
48
+
49
+ Clone the repository:
50
+
51
+ ```bash
52
+ git clone https://github.com/AYUSHKUMAR0401/MyMLLibrary.git
@@ -0,0 +1,41 @@
1
+ # MyMLLibrary
2
+
3
+ A simple Python Machine Learning library containing machine learning algorithms implemented from scratch for learning and educational purposes.
4
+
5
+ ## Features
6
+
7
+ - Machine Learning algorithms implemented from scratch
8
+ - Simple and beginner-friendly implementations
9
+ - No automatic dataset loading
10
+ - No automatic model training or prediction
11
+ - Source code of each algorithm can be displayed directly
12
+ - Designed for learning how ML algorithms work internally
13
+
14
+ ## Algorithms
15
+
16
+ Currently available:
17
+
18
+ 1. Perceptron Learning Algorithm (PLA)=
19
+ 2. Linear Regression
20
+ 3. Logistic Regression
21
+ 4. K-Nearest Neighbors (KNN)
22
+ 5. Naive Bayes
23
+ 6. K-Means Clustering
24
+ 7. DBSCAN
25
+ 8. Single Layer Perceptron (SLP)
26
+ 9. Performance matrix
27
+ 10. Multilayer Perceptron (MLP)
28
+
29
+ ### Supporting Modules
30
+
31
+ - Performance Metrics
32
+ - Code Viewer
33
+
34
+ > More algorithms will be added in future versions.
35
+
36
+ ## Installation
37
+
38
+ Clone the repository:
39
+
40
+ ```bash
41
+ git clone https://github.com/AYUSHKUMAR0401/MyMLLibrary.git
@@ -0,0 +1,213 @@
1
+ import math
2
+ import matplotlib.pyplot as plt
3
+
4
+
5
+ # --------------------------------------------------
6
+ # Function 1: Calculate Euclidean Distance
7
+ # --------------------------------------------------
8
+
9
+ def distance(p1, p2):
10
+
11
+ x1, y1 = p1
12
+ x2, y2 = p2
13
+
14
+ return math.sqrt(
15
+ (x1 - x2) ** 2 +
16
+ (y1 - y2) ** 2
17
+ )
18
+
19
+
20
+ # --------------------------------------------------
21
+ # Function 2: Find Neighbors of a Point
22
+ # --------------------------------------------------
23
+
24
+ def get_neighbors(points, point, eps):
25
+
26
+ neighbors = []
27
+
28
+ for p in points:
29
+
30
+ if p != point:
31
+
32
+ if distance(
33
+ points[point],
34
+ points[p]
35
+ ) <= eps:
36
+
37
+ neighbors.append(p)
38
+
39
+ return neighbors
40
+
41
+
42
+ # --------------------------------------------------
43
+ # Function 3: DBSCAN Algorithm
44
+ # --------------------------------------------------
45
+
46
+ def dbscan(points, eps, minPts):
47
+
48
+ visited = set()
49
+
50
+ clusters = []
51
+
52
+ noise = []
53
+
54
+ for point in points:
55
+
56
+ if point in visited:
57
+ continue
58
+
59
+ visited.add(point)
60
+
61
+ neighbors = get_neighbors(
62
+ points,
63
+ point,
64
+ eps
65
+ )
66
+
67
+ # Include the point itself
68
+ # when checking MinPts
69
+ if len(neighbors) + 1 < minPts:
70
+
71
+ noise.append(point)
72
+
73
+ continue
74
+
75
+ # Start a new cluster
76
+ cluster = [point]
77
+
78
+ # Expand cluster
79
+ i = 0
80
+
81
+ while i < len(neighbors):
82
+
83
+ current = neighbors[i]
84
+
85
+ if current not in visited:
86
+
87
+ visited.add(current)
88
+
89
+ current_neighbors = get_neighbors(
90
+ points,
91
+ current,
92
+ eps
93
+ )
94
+
95
+ if len(current_neighbors) + 1 >= minPts:
96
+
97
+ for n in current_neighbors:
98
+
99
+ if n not in neighbors:
100
+ neighbors.append(n)
101
+
102
+ # Add point to cluster
103
+ if current not in cluster:
104
+
105
+ cluster.append(current)
106
+
107
+ i += 1
108
+
109
+ clusters.append(cluster)
110
+
111
+ return clusters, noise
112
+
113
+
114
+ # --------------------------------------------------
115
+ # Function 4: Plot DBSCAN Clusters
116
+ # --------------------------------------------------
117
+
118
+ def plot_dbscan(points, clusters, noise):
119
+
120
+ # Plot clusters
121
+ for i, cluster in enumerate(clusters):
122
+
123
+ x = [
124
+ points[p][0]
125
+ for p in cluster
126
+ ]
127
+
128
+ y = [
129
+ points[p][1]
130
+ for p in cluster
131
+ ]
132
+
133
+ plt.scatter(
134
+ x,
135
+ y,
136
+ label="Cluster " + str(i + 1)
137
+ )
138
+
139
+ # Plot noise / outliers
140
+ if len(noise) > 0:
141
+
142
+ x = [
143
+ points[p][0]
144
+ for p in noise
145
+ ]
146
+
147
+ y = [
148
+ points[p][1]
149
+ for p in noise
150
+ ]
151
+
152
+ plt.scatter(
153
+ x,
154
+ y,
155
+ label="Noise / Outlier"
156
+ )
157
+
158
+ # Display point names
159
+ for p, (x, y) in points.items():
160
+
161
+ plt.text(
162
+ x + 0.3,
163
+ y + 0.3,
164
+ p
165
+ )
166
+
167
+ plt.xlabel("X")
168
+ plt.ylabel("Y")
169
+
170
+ plt.title("DBSCAN Clustering")
171
+
172
+ plt.legend()
173
+
174
+ plt.grid(True)
175
+
176
+ plt.show()
177
+
178
+
179
+ # --------------------------------------------------
180
+ # Function 5: Display DBSCAN Results
181
+ # --------------------------------------------------
182
+
183
+ def display_clusters(clusters, noise):
184
+
185
+ print("Clusters:")
186
+
187
+ for i in range(len(clusters)):
188
+
189
+ print(
190
+ "Cluster",
191
+ i + 1,
192
+ ":",
193
+ clusters[i]
194
+ )
195
+
196
+ print(
197
+ "Noise / Outliers:",
198
+ noise
199
+ )
200
+
201
+
202
+ # --------------------------------------------------
203
+ # Function 6: Display DBSCAN Source Code
204
+ # --------------------------------------------------
205
+
206
+ def show_dbscan_code():
207
+
208
+ import sys
209
+ from .code_viewer import show_code7
210
+
211
+ module = sys.modules[__name__]
212
+
213
+ show_code7(module)
@@ -0,0 +1,187 @@
1
+ import numpy as np
2
+ import matplotlib.pyplot as plt
3
+
4
+
5
+ # --------------------------------------------------
6
+ # Function 1: Calculate Euclidean Distance
7
+ # --------------------------------------------------
8
+
9
+ def distance(point, centroid):
10
+
11
+ return np.sqrt(
12
+ np.sum((point - centroid) ** 2)
13
+ )
14
+
15
+
16
+ # --------------------------------------------------
17
+ # Function 2: Assign each point to nearest centroid
18
+ # --------------------------------------------------
19
+
20
+ def assign_clusters(data, centroids):
21
+
22
+ clusters = []
23
+
24
+ for point in data:
25
+
26
+ distances = []
27
+
28
+ for centroid in centroids:
29
+
30
+ distances.append(
31
+ distance(point, centroid)
32
+ )
33
+
34
+ # Get index of minimum distance
35
+ cluster = np.argmin(distances)
36
+
37
+ clusters.append(cluster)
38
+
39
+ return np.array(clusters, dtype=int)
40
+
41
+
42
+ # --------------------------------------------------
43
+ # Function 3: Calculate new centroids
44
+ # --------------------------------------------------
45
+
46
+ def calculate_centroids(data, clusters, k):
47
+
48
+ new_centroids = []
49
+
50
+ for i in range(k):
51
+
52
+ cluster_points = data[clusters == i]
53
+
54
+ centroid = np.mean(
55
+ cluster_points,
56
+ axis=0
57
+ )
58
+
59
+ new_centroids.append(centroid)
60
+
61
+ return np.array(new_centroids)
62
+
63
+
64
+ # --------------------------------------------------
65
+ # Function 4: K-Means Algorithm
66
+ # --------------------------------------------------
67
+
68
+ def kmeans(data, k):
69
+
70
+ # Initial centroids = first k points
71
+ centroids = data[:k].copy()
72
+
73
+ while True:
74
+
75
+ # Assign clusters
76
+ clusters = assign_clusters(
77
+ data,
78
+ centroids
79
+ )
80
+
81
+ # Calculate new centroids
82
+ new_centroids = calculate_centroids(
83
+ data,
84
+ clusters,
85
+ k
86
+ )
87
+
88
+ # Check convergence
89
+ if np.allclose(
90
+ centroids,
91
+ new_centroids
92
+ ):
93
+ break
94
+
95
+ centroids = new_centroids
96
+
97
+ return clusters, centroids
98
+
99
+
100
+ # --------------------------------------------------
101
+ # Function 5: Display Final Clusters
102
+ # --------------------------------------------------
103
+
104
+ def display_clusters(
105
+ names,
106
+ data,
107
+ clusters,
108
+ centroids
109
+ ):
110
+
111
+ print("\n----- FINAL CLUSTERS -----")
112
+
113
+ for i in range(len(centroids)):
114
+
115
+ print("\nCluster", i + 1)
116
+
117
+ for j in range(len(data)):
118
+
119
+ if clusters[j] == i:
120
+
121
+ print(
122
+ names[j],
123
+ "->",
124
+ data[j]
125
+ )
126
+
127
+ print(
128
+ "Centroid =",
129
+ centroids[i]
130
+ )
131
+
132
+
133
+ # --------------------------------------------------
134
+ # Function 6: Plot Graph
135
+ # --------------------------------------------------
136
+
137
+ def plot_graph(
138
+ data,
139
+ clusters,
140
+ centroids,
141
+ names
142
+ ):
143
+
144
+ for i in range(len(centroids)):
145
+
146
+ points = data[clusters == i]
147
+
148
+ plt.scatter(
149
+ points[:, 0],
150
+ points[:, 1],
151
+ label="Cluster " + str(i + 1)
152
+ )
153
+
154
+ # Display point names
155
+ for i in range(len(data)):
156
+
157
+ plt.annotate(
158
+ names[i],
159
+ (data[i][0], data[i][1])
160
+ )
161
+
162
+ # Display centroids
163
+ plt.scatter(
164
+ centroids[:, 0],
165
+ centroids[:, 1],
166
+ marker="X",
167
+ s=200,
168
+ label="Centroids"
169
+ )
170
+
171
+ plt.xlabel("X")
172
+ plt.ylabel("Y")
173
+ plt.title("K-Means Clustering")
174
+ plt.legend()
175
+ plt.grid(True)
176
+ plt.show()
177
+
178
+
179
+ # --------------------------------------------------
180
+ # Function 7: Display K-Means Source Code
181
+ # --------------------------------------------------
182
+
183
+ def show_kmeans_code():
184
+ import sys
185
+ from .code_viewer import show_code6
186
+ module = sys.modules[__name__]
187
+ show_code6(module)
@@ -0,0 +1,67 @@
1
+ import math
2
+ from collections import Counter
3
+
4
+ def calculate_distance(x1, y1, x2, y2):
5
+ distance = math.sqrt(
6
+ (x1 - x2) ** 2 +
7
+ (y1 - y2) ** 2
8
+ )
9
+ return distance
10
+
11
+
12
+ # --------------------------------------------------
13
+ # Function 2: Calculate distances from test point
14
+ # --------------------------------------------------
15
+
16
+ def calculate_distances(data, test_x, test_y):
17
+
18
+ distances = []
19
+
20
+ for index, row in data.iterrows():
21
+
22
+ distance = calculate_distance(
23
+ test_x,
24
+ test_y,
25
+ row["X"],
26
+ row["Y"]
27
+ )
28
+
29
+ distances.append(
30
+ (row["ID"], distance, row["Label"])
31
+ )
32
+
33
+ return distances
34
+
35
+
36
+ # --------------------------------------------------
37
+ # Function 3: KNN Classification
38
+ # --------------------------------------------------
39
+
40
+ def knn_classification(data, test_x, test_y, k):
41
+
42
+ distances = calculate_distances(
43
+ data,
44
+ test_x,
45
+ test_y
46
+ )
47
+
48
+ # Sort according to distance
49
+ distances.sort(key=lambda x: x[1])
50
+
51
+ # Select k nearest neighbours
52
+ nearest = distances[:k]
53
+
54
+ # Get labels
55
+ labels = [item[2] for item in nearest]
56
+
57
+ count = Counter(labels)
58
+ prediction = count.most_common(1)[0][0]
59
+ return prediction
60
+ def show_knn_code():
61
+
62
+ import sys
63
+ from .code_viewer import show_code4
64
+
65
+ module = sys.modules[__name__]
66
+
67
+ show_code4(module)
@@ -0,0 +1,48 @@
1
+ import numpy as np
2
+ import matplotlib.pyplot as plt
3
+ from .code_viewer import show_code2
4
+
5
+
6
+ def linear_regression(X, Y):
7
+ x_mean = np.mean(X)
8
+ y_mean = np.mean(Y)
9
+
10
+ numerator = np.sum((X - x_mean) * (Y - y_mean))
11
+ denominator = np.sum((X - x_mean) ** 2)
12
+
13
+ slope = numerator / denominator
14
+ intercept = y_mean - slope * x_mean
15
+
16
+ Y_pred = intercept + slope * X
17
+
18
+ return intercept, slope, Y_pred
19
+
20
+
21
+ def error_metrics(Y, Y_pred):
22
+ MSE = np.mean((Y - Y_pred) ** 2)
23
+ RMSE = np.sqrt(MSE)
24
+ MAE = np.mean(np.abs(Y - Y_pred))
25
+
26
+ SS_res = np.sum((Y - Y_pred) ** 2)
27
+ SS_tot = np.sum((Y - np.mean(Y)) ** 2)
28
+
29
+ R2 = 1 - (SS_res / SS_tot)
30
+
31
+ return MSE, RMSE, MAE, R2
32
+
33
+
34
+ def plot_regression(X, Y, Y_pred):
35
+ plt.scatter(X, Y, label="Actual Data")
36
+ plt.plot(X, Y_pred, label="Regression Line")
37
+
38
+ plt.xlabel("Class_Test_Marks")
39
+ plt.ylabel("Semester_Marks")
40
+ plt.title("Linear Regression")
41
+
42
+ plt.legend()
43
+ plt.grid(True)
44
+ plt.show()
45
+
46
+
47
+ def show_linear_regression_code():
48
+ show_code2(__import__(__name__, fromlist=["*"]))
@@ -0,0 +1,71 @@
1
+ import numpy as np
2
+ from .code_viewer import show_code3
3
+
4
+ def sigmoid(z):
5
+ return 1 / (1 + np.exp(-z))
6
+
7
+
8
+ def hypothesis(X, w, b):
9
+ z = w * X + b
10
+ return sigmoid(z)
11
+
12
+
13
+ def cost_function(X, y, w, b):
14
+ m = len(X)
15
+
16
+ h = hypothesis(X, w, b)
17
+
18
+ h = np.clip(h, 1e-10, 1 - 1e-10)
19
+
20
+ cost = (-1 / m) * np.sum(
21
+ y * np.log(h) + (1 - y) * np.log(1 - h)
22
+ )
23
+
24
+ return cost
25
+
26
+
27
+ def gradient_descent(X, y, w, b, learning_rate, iterations):
28
+ m = len(X)
29
+
30
+ for i in range(iterations):
31
+
32
+ h = hypothesis(X, w, b)
33
+
34
+ dw = (1 / m) * np.dot(X.T, (h - y))
35
+
36
+ db = (1 / m) * np.sum(h - y)
37
+
38
+ w -= learning_rate * dw
39
+ b -= learning_rate * db
40
+
41
+ return w, b
42
+
43
+
44
+ def logistic_regression(
45
+ X,
46
+ y,
47
+ learning_rate=0.0001,
48
+ iterations=100000
49
+ ):
50
+
51
+ w = 0.0
52
+ b = 0.0
53
+
54
+ w, b = gradient_descent(
55
+ X,
56
+ y,
57
+ w,
58
+ b,
59
+ learning_rate,
60
+ iterations
61
+ )
62
+
63
+ return w, b
64
+
65
+ def show_logistic_regression_code():
66
+ import sys
67
+ from .code_viewer import show_code2
68
+
69
+ module = sys.modules[__name__]
70
+
71
+ show_code3(module)