mlhitk 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mlhitk-0.1.0/PKG-INFO +16 -0
- mlhitk-0.1.0/mlhitk/__init__.py +333 -0
- mlhitk-0.1.0/mlhitk.egg-info/PKG-INFO +16 -0
- mlhitk-0.1.0/mlhitk.egg-info/SOURCES.txt +6 -0
- mlhitk-0.1.0/mlhitk.egg-info/dependency_links.txt +1 -0
- mlhitk-0.1.0/mlhitk.egg-info/top_level.txt +1 -0
- mlhitk-0.1.0/setup.cfg +4 -0
- mlhitk-0.1.0/setup.py +17 -0
mlhitk-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mlhitk
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A library for printing ML code snippets
|
|
5
|
+
Home-page: https://github.com/ayush/mlhitk
|
|
6
|
+
Author: Ayush
|
|
7
|
+
Author-email: ayush@example.com
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Requires-Python: >=3.6
|
|
11
|
+
Dynamic: author
|
|
12
|
+
Dynamic: author-email
|
|
13
|
+
Dynamic: classifier
|
|
14
|
+
Dynamic: home-page
|
|
15
|
+
Dynamic: requires-python
|
|
16
|
+
Dynamic: summary
|
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
|
|
2
|
+
def show_pla():
|
|
3
|
+
print(r"""import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
df = pd.read_csv('demo.csv')
|
|
7
|
+
X = df.iloc[:, :-1].values
|
|
8
|
+
y = df.iloc[:, -1].values
|
|
9
|
+
|
|
10
|
+
w = np.zeros(X.shape[1])
|
|
11
|
+
b = 0
|
|
12
|
+
lr = 1
|
|
13
|
+
|
|
14
|
+
while True:
|
|
15
|
+
errors = 0
|
|
16
|
+
for i in range(len(X)):
|
|
17
|
+
if y[i] * (np.dot(X[i], w) + b) <= 0:
|
|
18
|
+
w += lr * y[i] * X[i]
|
|
19
|
+
b += lr * y[i]
|
|
20
|
+
errors += 1
|
|
21
|
+
if errors == 0:
|
|
22
|
+
break
|
|
23
|
+
|
|
24
|
+
print(f"Final weights: {w}, bias: {b}")
|
|
25
|
+
""")
|
|
26
|
+
|
|
27
|
+
def show_lr():
|
|
28
|
+
print(r"""import numpy as np
|
|
29
|
+
import pandas as pd
|
|
30
|
+
|
|
31
|
+
df = pd.read_csv('demo.csv')
|
|
32
|
+
# Assuming format from image: Sl No, Marks in Class test, Marks in Semester
|
|
33
|
+
X = df.iloc[:, 1].values
|
|
34
|
+
Y = df.iloc[:, 2].values
|
|
35
|
+
|
|
36
|
+
n = len(X)
|
|
37
|
+
m = (n * np.sum(X*Y) - np.sum(X)*np.sum(Y)) / (n * np.sum(X**2) - np.sum(X)**2)
|
|
38
|
+
c = (np.sum(Y) - m * np.sum(X)) / n
|
|
39
|
+
|
|
40
|
+
pred_20 = m * 20 + c
|
|
41
|
+
print(f"Estimated marks for 20 in class test: {pred_20}")
|
|
42
|
+
|
|
43
|
+
Y_pred = m * X + c
|
|
44
|
+
mae = np.mean(np.abs(Y - Y_pred))
|
|
45
|
+
mse = np.mean((Y - Y_pred)**2)
|
|
46
|
+
rmse = np.sqrt(mse)
|
|
47
|
+
ss_tot = np.sum((Y - np.mean(Y))**2)
|
|
48
|
+
ss_res = np.sum((Y - Y_pred)**2)
|
|
49
|
+
r2 = 1 - (ss_res / ss_tot)
|
|
50
|
+
|
|
51
|
+
print(f"MAE: {mae}, MSE: {mse}, RMSE: {rmse}, R2: {r2}")
|
|
52
|
+
""")
|
|
53
|
+
|
|
54
|
+
def show_logr():
|
|
55
|
+
print(r"""import numpy as np
|
|
56
|
+
import pandas as pd
|
|
57
|
+
|
|
58
|
+
df = pd.read_csv('demo.csv')
|
|
59
|
+
X = df.iloc[:, 1].values
|
|
60
|
+
y = df.iloc[:, 2].values
|
|
61
|
+
y = np.where(y == np.min(y), 0, 1)
|
|
62
|
+
|
|
63
|
+
mean_X = np.mean(X)
|
|
64
|
+
std_X = np.std(X)
|
|
65
|
+
X = (X - mean_X) / std_X
|
|
66
|
+
|
|
67
|
+
def sigmoid(z):
|
|
68
|
+
return 1 / (1 + np.exp(-z))
|
|
69
|
+
|
|
70
|
+
def cost_function(X, y, w, b):
|
|
71
|
+
m = len(y)
|
|
72
|
+
h = sigmoid(X * w + b)
|
|
73
|
+
return -1/m * np.sum(y * np.log(h + 1e-15) + (1-y) * np.log(1-h + 1e-15))
|
|
74
|
+
|
|
75
|
+
w, b = 0.0, 0.0
|
|
76
|
+
lr, epochs = 0.1, 10000
|
|
77
|
+
|
|
78
|
+
for _ in range(epochs):
|
|
79
|
+
h = sigmoid(X * w + b)
|
|
80
|
+
w -= lr * np.mean((h - y) * X)
|
|
81
|
+
b -= lr * np.mean(h - y)
|
|
82
|
+
|
|
83
|
+
x_test = (20 - mean_X) / std_X
|
|
84
|
+
prob = sigmoid(w * x_test + b)
|
|
85
|
+
print(f"Weights: w={w}, b={b}")
|
|
86
|
+
print(f"Probability of passing for 20 marks: {prob}")
|
|
87
|
+
print("Pass" if prob >= 0.5 else "Fail")
|
|
88
|
+
""")
|
|
89
|
+
|
|
90
|
+
def show_knn():
|
|
91
|
+
print(r"""import numpy as np
|
|
92
|
+
import pandas as pd
|
|
93
|
+
|
|
94
|
+
df = pd.read_csv('demo.csv')
|
|
95
|
+
# Assuming features and then label as last column
|
|
96
|
+
X = df.iloc[:, 1:-1].values
|
|
97
|
+
y = df.iloc[:, -1].values
|
|
98
|
+
|
|
99
|
+
def knn(X_train, y_train, x_query, k=3):
|
|
100
|
+
distances = np.sqrt(np.sum((X_train - x_query)**2, axis=1))
|
|
101
|
+
nearest_idx = np.argsort(distances)[:k]
|
|
102
|
+
nearest_labels = y_train[nearest_idx]
|
|
103
|
+
values, counts = np.unique(nearest_labels, return_counts=True)
|
|
104
|
+
return values[np.argmax(counts)]
|
|
105
|
+
|
|
106
|
+
print("KNN Predictions on the training set (k=3):")
|
|
107
|
+
for i, x_q in enumerate(X):
|
|
108
|
+
print(f"Sample {i} Prediction:", knn(X, y, x_q, k=3))
|
|
109
|
+
""")
|
|
110
|
+
|
|
111
|
+
def show_nbc():
|
|
112
|
+
print(r"""import pandas as pd
|
|
113
|
+
import numpy as np
|
|
114
|
+
|
|
115
|
+
df = pd.read_csv('demo.csv')
|
|
116
|
+
X = df.iloc[:, 1:-1].values
|
|
117
|
+
y = df.iloc[:, -1].values
|
|
118
|
+
|
|
119
|
+
class NaiveBayes:
|
|
120
|
+
def fit(self, X, y):
|
|
121
|
+
self.classes = np.unique(y)
|
|
122
|
+
self.mean = np.zeros((len(self.classes), X.shape[1]))
|
|
123
|
+
self.var = np.zeros((len(self.classes), X.shape[1]))
|
|
124
|
+
self.priors = np.zeros(len(self.classes))
|
|
125
|
+
for idx, c in enumerate(self.classes):
|
|
126
|
+
X_c = X[y == c]
|
|
127
|
+
self.mean[idx, :] = X_c.mean(axis=0)
|
|
128
|
+
self.var[idx, :] = X_c.var(axis=0) + 1e-6
|
|
129
|
+
self.priors[idx] = X_c.shape[0] / float(X.shape[0])
|
|
130
|
+
|
|
131
|
+
def predict(self, X):
|
|
132
|
+
return np.array([self._predict(x) for x in X])
|
|
133
|
+
|
|
134
|
+
def _predict(self, x):
|
|
135
|
+
posteriors = []
|
|
136
|
+
for idx, c in enumerate(self.classes):
|
|
137
|
+
prior = np.log(self.priors[idx])
|
|
138
|
+
num = np.exp(-((x - self.mean[idx])**2) / (2 * self.var[idx]))
|
|
139
|
+
den = np.sqrt(2 * np.pi * self.var[idx])
|
|
140
|
+
posterior = np.sum(np.log(num / den))
|
|
141
|
+
posteriors.append(prior + posterior)
|
|
142
|
+
return self.classes[np.argmax(posteriors)]
|
|
143
|
+
|
|
144
|
+
nb = NaiveBayes()
|
|
145
|
+
nb.fit(X, y)
|
|
146
|
+
print("Naive Bayes predictions:", nb.predict(X))
|
|
147
|
+
""")
|
|
148
|
+
|
|
149
|
+
def show_dbscan():
|
|
150
|
+
print(r"""import numpy as np
|
|
151
|
+
import pandas as pd
|
|
152
|
+
|
|
153
|
+
df = pd.read_csv('demo.csv')
|
|
154
|
+
X = df.iloc[:, 1:3].values
|
|
155
|
+
epsilon = 2
|
|
156
|
+
min_pts = 2
|
|
157
|
+
|
|
158
|
+
labels = np.full(X.shape[0], -1)
|
|
159
|
+
cluster_id = 0
|
|
160
|
+
|
|
161
|
+
def region_query(p_idx):
|
|
162
|
+
return np.where(np.linalg.norm(X - X[p_idx], axis=1) <= epsilon)[0]
|
|
163
|
+
|
|
164
|
+
for p in range(X.shape[0]):
|
|
165
|
+
if labels[p] != -1:
|
|
166
|
+
continue
|
|
167
|
+
neighbors = region_query(p)
|
|
168
|
+
if len(neighbors) < min_pts:
|
|
169
|
+
labels[p] = -1 # Noise
|
|
170
|
+
else:
|
|
171
|
+
labels[p] = cluster_id
|
|
172
|
+
i = 0
|
|
173
|
+
while i < len(neighbors):
|
|
174
|
+
pn = neighbors[i]
|
|
175
|
+
if labels[pn] == -1:
|
|
176
|
+
labels[pn] = cluster_id
|
|
177
|
+
elif labels[pn] < 0:
|
|
178
|
+
labels[pn] = cluster_id
|
|
179
|
+
pn_neighbors = region_query(pn)
|
|
180
|
+
if len(pn_neighbors) >= min_pts:
|
|
181
|
+
neighbors = np.append(neighbors, pn_neighbors)
|
|
182
|
+
i += 1
|
|
183
|
+
cluster_id += 1
|
|
184
|
+
|
|
185
|
+
print("DBSCAN Labels (-1 is noise):", labels)
|
|
186
|
+
""")
|
|
187
|
+
|
|
188
|
+
def show_kmeans():
|
|
189
|
+
print(r"""import numpy as np
|
|
190
|
+
import pandas as pd
|
|
191
|
+
|
|
192
|
+
df = pd.read_csv('demo.csv')
|
|
193
|
+
X = df.iloc[:, 1:3].values
|
|
194
|
+
K = 3
|
|
195
|
+
|
|
196
|
+
np.random.seed(42)
|
|
197
|
+
centroids = X[np.random.choice(X.shape[0], K, replace=False)]
|
|
198
|
+
|
|
199
|
+
while True:
|
|
200
|
+
distances = np.linalg.norm(X[:, np.newaxis] - centroids, axis=2)
|
|
201
|
+
labels = np.argmin(distances, axis=1)
|
|
202
|
+
new_centroids = np.array([X[labels == k].mean(axis=0) if np.any(labels == k) else centroids[k] for k in range(K)])
|
|
203
|
+
if np.all(centroids == new_centroids):
|
|
204
|
+
break
|
|
205
|
+
centroids = new_centroids
|
|
206
|
+
|
|
207
|
+
print("Final centroids:\n", centroids)
|
|
208
|
+
print("Labels:", labels)
|
|
209
|
+
""")
|
|
210
|
+
|
|
211
|
+
def show_pm():
|
|
212
|
+
print(r"""import numpy as np
|
|
213
|
+
|
|
214
|
+
y_actual = np.array([0]*900 + [1]*100)
|
|
215
|
+
np.random.shuffle(y_actual)
|
|
216
|
+
|
|
217
|
+
y_pred = np.array([0]*900 + [1]*100)
|
|
218
|
+
np.random.shuffle(y_pred)
|
|
219
|
+
|
|
220
|
+
TP = np.sum((y_actual == 1) & (y_pred == 1))
|
|
221
|
+
TN = np.sum((y_actual == 0) & (y_pred == 0))
|
|
222
|
+
FP = np.sum((y_actual == 0) & (y_pred == 1))
|
|
223
|
+
FN = np.sum((y_actual == 1) & (y_pred == 0))
|
|
224
|
+
|
|
225
|
+
print(f"Confusion Matrix:\n[[{TN}, {FP}]\n [{FN}, {TP}]]")
|
|
226
|
+
|
|
227
|
+
print("Decision: Precision is used when FP is costly, Recall is used when FN is costly.")
|
|
228
|
+
|
|
229
|
+
priority = np.random.choice(['FP', 'FN'])
|
|
230
|
+
beta = 0.5 if priority == 'FP' else 2
|
|
231
|
+
print(f"Randomly assigned priority to: {priority} => Beta = {beta}")
|
|
232
|
+
|
|
233
|
+
precision = TP / (TP + FP) if (TP + FP) else 0
|
|
234
|
+
recall = TP / (TP + FN) if (TP + FN) else 0
|
|
235
|
+
|
|
236
|
+
f_beta = (1 + beta**2) * (precision * recall) / ((beta**2 * precision) + recall) if (precision + recall) else 0
|
|
237
|
+
print(f"Precision: {precision:.4f}, Recall: {recall:.4f}")
|
|
238
|
+
print(f"F-beta Score: {f_beta:.4f}")
|
|
239
|
+
""")
|
|
240
|
+
|
|
241
|
+
def show_slp():
|
|
242
|
+
print(r"""import numpy as np
|
|
243
|
+
|
|
244
|
+
def train_slp():
|
|
245
|
+
# Logical AND gate
|
|
246
|
+
X = np.array([[0, 0], [0, 1], [1, 0], [1, 1]])
|
|
247
|
+
y = np.array([0, 0, 0, 1])
|
|
248
|
+
|
|
249
|
+
w = np.zeros(2)
|
|
250
|
+
b = 0
|
|
251
|
+
lr = 0.1
|
|
252
|
+
|
|
253
|
+
while True:
|
|
254
|
+
errors = 0
|
|
255
|
+
for i in range(4):
|
|
256
|
+
y_pred = 1 if (np.dot(X[i], w) + b) >= 0 else 0
|
|
257
|
+
if y[i] != y_pred:
|
|
258
|
+
w += lr * (y[i] - y_pred) * X[i]
|
|
259
|
+
b += lr * (y[i] - y_pred)
|
|
260
|
+
errors += 1
|
|
261
|
+
if errors == 0:
|
|
262
|
+
break
|
|
263
|
+
return w, b
|
|
264
|
+
|
|
265
|
+
w, b = train_slp()
|
|
266
|
+
print(f"Trained SLP for AND gate. Weights: {w}, Bias: {b}")
|
|
267
|
+
|
|
268
|
+
print("Testing the trained perceptron:")
|
|
269
|
+
for x in [[0,0], [0,1], [1,0], [1,1]]:
|
|
270
|
+
pred = 1 if (np.dot(x, w) + b) >= 0 else 0
|
|
271
|
+
print(f"Input {x} -> Prediction {pred}")
|
|
272
|
+
""")
|
|
273
|
+
|
|
274
|
+
def show_mlp():
|
|
275
|
+
print(r"""import numpy as np
|
|
276
|
+
import pandas as pd
|
|
277
|
+
|
|
278
|
+
df = pd.read_csv('demo.csv')
|
|
279
|
+
X = df.iloc[:, :-1].values
|
|
280
|
+
y = df.iloc[:, -1].values
|
|
281
|
+
|
|
282
|
+
labels, y_int = np.unique(y, return_inverse=True)
|
|
283
|
+
y_onehot = np.zeros((y_int.size, y_int.max() + 1))
|
|
284
|
+
y_onehot[np.arange(y_int.size), y_int] = 1
|
|
285
|
+
|
|
286
|
+
X = (X - X.mean(axis=0)) / X.std(axis=0)
|
|
287
|
+
|
|
288
|
+
def sigmoid(z): return 1 / (1 + np.exp(-z))
|
|
289
|
+
def sigmoid_deriv(z): return z * (1 - z)
|
|
290
|
+
|
|
291
|
+
np.random.seed(42)
|
|
292
|
+
input_dim = X.shape[1]
|
|
293
|
+
hidden_dim = 5
|
|
294
|
+
output_dim = y_onehot.shape[1]
|
|
295
|
+
|
|
296
|
+
W1 = np.random.randn(input_dim, hidden_dim)
|
|
297
|
+
b1 = np.zeros((1, hidden_dim))
|
|
298
|
+
W2 = np.random.randn(hidden_dim, output_dim)
|
|
299
|
+
b2 = np.zeros((1, output_dim))
|
|
300
|
+
|
|
301
|
+
lr = 0.1
|
|
302
|
+
epochs = 100
|
|
303
|
+
|
|
304
|
+
for epoch in range(1, epochs + 1):
|
|
305
|
+
# Forward
|
|
306
|
+
z1 = np.dot(X, W1) + b1
|
|
307
|
+
a1 = sigmoid(z1)
|
|
308
|
+
z2 = np.dot(a1, W2) + b2
|
|
309
|
+
a2 = sigmoid(z2)
|
|
310
|
+
|
|
311
|
+
loss = np.mean((y_onehot - a2)**2)
|
|
312
|
+
|
|
313
|
+
# Backprop
|
|
314
|
+
d2 = (a2 - y_onehot) * sigmoid_deriv(a2)
|
|
315
|
+
dW2 = np.dot(a1.T, d2)
|
|
316
|
+
db2 = np.sum(d2, axis=0, keepdims=True)
|
|
317
|
+
|
|
318
|
+
d1 = np.dot(d2, W2.T) * sigmoid_deriv(a1)
|
|
319
|
+
dW1 = np.dot(X.T, d1)
|
|
320
|
+
db1 = np.sum(d1, axis=0, keepdims=True)
|
|
321
|
+
|
|
322
|
+
W1 -= lr * dW1 / X.shape[0]
|
|
323
|
+
b1 -= lr * db1 / X.shape[0]
|
|
324
|
+
W2 -= lr * dW2 / X.shape[0]
|
|
325
|
+
b2 -= lr * db2 / X.shape[0]
|
|
326
|
+
|
|
327
|
+
if epoch % 10 == 0 or epoch == 1:
|
|
328
|
+
print(f"Epoch {epoch}, Loss: {loss:.4f}")
|
|
329
|
+
|
|
330
|
+
predictions = np.argmax(a2, axis=1)
|
|
331
|
+
accuracy = np.mean(predictions == y_int)
|
|
332
|
+
print(f"\nFinal Accuracy: {accuracy * 100:.2f}%")
|
|
333
|
+
""")
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mlhitk
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A library for printing ML code snippets
|
|
5
|
+
Home-page: https://github.com/ayush/mlhitk
|
|
6
|
+
Author: Ayush
|
|
7
|
+
Author-email: ayush@example.com
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Requires-Python: >=3.6
|
|
11
|
+
Dynamic: author
|
|
12
|
+
Dynamic: author-email
|
|
13
|
+
Dynamic: classifier
|
|
14
|
+
Dynamic: home-page
|
|
15
|
+
Dynamic: requires-python
|
|
16
|
+
Dynamic: summary
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
mlhitk
|
mlhitk-0.1.0/setup.cfg
ADDED
mlhitk-0.1.0/setup.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from setuptools import setup, find_packages
|
|
2
|
+
|
|
3
|
+
setup(
|
|
4
|
+
name='mlhitk',
|
|
5
|
+
version='0.1.0',
|
|
6
|
+
packages=find_packages(),
|
|
7
|
+
description='A library for printing ML code snippets',
|
|
8
|
+
author='Ayush',
|
|
9
|
+
author_email='ayush@example.com',
|
|
10
|
+
url='https://github.com/ayush/mlhitk',
|
|
11
|
+
classifiers=[
|
|
12
|
+
'Programming Language :: Python :: 3',
|
|
13
|
+
'Operating System :: OS Independent',
|
|
14
|
+
],
|
|
15
|
+
python_requires='>=3.6',
|
|
16
|
+
)
|
|
17
|
+
|