mlkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mlkit/__init__.py ADDED
@@ -0,0 +1,35 @@
1
+ """mlkit - Machine Learning Toolkit.
2
+
3
+ A small, readable NumPy toolkit with a scikit-learn style ``fit``/``predict``
4
+ interface, classic models, preprocessing, model selection and metrics.
5
+ """
6
+
7
+ from . import datasets, metrics
8
+ from .base import BaseEstimator, NotFittedError, clone
9
+ from .cluster import KMeans
10
+ from .linear_model import LinearRegression, LogisticRegression, Ridge
11
+ from .model_selection import KFold, cross_val_score, train_test_split
12
+ from .neighbors import KNeighborsClassifier, KNeighborsRegressor
13
+ from .preprocessing import MinMaxScaler, StandardScaler
14
+
15
+ __version__ = "0.1.0"
16
+
17
+ __all__ = [
18
+ "BaseEstimator",
19
+ "NotFittedError",
20
+ "clone",
21
+ "LinearRegression",
22
+ "Ridge",
23
+ "LogisticRegression",
24
+ "KNeighborsClassifier",
25
+ "KNeighborsRegressor",
26
+ "KMeans",
27
+ "StandardScaler",
28
+ "MinMaxScaler",
29
+ "train_test_split",
30
+ "KFold",
31
+ "cross_val_score",
32
+ "datasets",
33
+ "metrics",
34
+ "__version__",
35
+ ]
mlkit/base.py ADDED
@@ -0,0 +1,165 @@
1
+ """Core estimator interface, mixins and input validation helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import inspect
6
+ from typing import Any, Optional, Tuple
7
+
8
+ import numpy as np
9
+ from numpy.typing import ArrayLike
10
+
11
+ __all__ = [
12
+ "BaseEstimator",
13
+ "ClassifierMixin",
14
+ "RegressorMixin",
15
+ "TransformerMixin",
16
+ "ClusterMixin",
17
+ "NotFittedError",
18
+ "check_array",
19
+ "check_X_y",
20
+ "check_is_fitted",
21
+ "clone",
22
+ ]
23
+
24
+
25
+ class NotFittedError(RuntimeError):
26
+ """Raised when ``predict``/``transform`` is called before ``fit``."""
27
+
28
+
29
+ def check_array(X: ArrayLike, *, ensure_2d: bool = True) -> np.ndarray:
30
+ """Convert ``X`` to a finite float array.
31
+
32
+ Args:
33
+ X: Array-like input.
34
+ ensure_2d: If True, a 2-D array of shape ``(n_samples, n_features)``
35
+ is required.
36
+
37
+ Returns:
38
+ A ``float64`` NumPy array.
39
+
40
+ Raises:
41
+ ValueError: If ``X`` is empty, has the wrong number of dimensions or
42
+ contains NaN/inf values.
43
+ """
44
+ arr = np.asarray(X, dtype=float)
45
+ if ensure_2d and arr.ndim != 2:
46
+ raise ValueError(f"Expected a 2-D array, got an array with ndim={arr.ndim}.")
47
+ if arr.size == 0:
48
+ raise ValueError("Input array is empty.")
49
+ if not np.all(np.isfinite(arr)):
50
+ raise ValueError("Input contains NaN or infinity.")
51
+ return arr
52
+
53
+
54
+ def check_X_y(X: ArrayLike, y: ArrayLike, *, y_numeric: bool = False) -> Tuple[np.ndarray, np.ndarray]:
55
+ """Validate a feature matrix and a 1-D target vector of matching length.
56
+
57
+ Args:
58
+ X: Array-like of shape ``(n_samples, n_features)``.
59
+ y: Array-like of shape ``(n_samples,)``.
60
+ y_numeric: If True, ``y`` is converted to ``float64``.
61
+
62
+ Returns:
63
+ The validated ``(X, y)`` pair.
64
+ """
65
+ X_arr = check_array(X)
66
+ y_arr = np.asarray(y, dtype=float if y_numeric else None)
67
+ if y_arr.ndim != 1:
68
+ raise ValueError(f"y must be 1-D, got ndim={y_arr.ndim}.")
69
+ if y_arr.shape[0] != X_arr.shape[0]:
70
+ raise ValueError(
71
+ f"X and y have inconsistent lengths: {X_arr.shape[0]} != {y_arr.shape[0]}."
72
+ )
73
+ if y_numeric and not np.all(np.isfinite(y_arr)):
74
+ raise ValueError("y contains NaN or infinity.")
75
+ return X_arr, y_arr
76
+
77
+
78
+ def check_is_fitted(estimator: "BaseEstimator", attribute: str) -> None:
79
+ """Raise :class:`NotFittedError` if ``estimator`` lacks ``attribute``."""
80
+ if getattr(estimator, attribute, None) is None:
81
+ raise NotFittedError(
82
+ f"This {type(estimator).__name__} instance is not fitted yet. "
83
+ "Call 'fit' with appropriate arguments first."
84
+ )
85
+
86
+
87
+ class BaseEstimator:
88
+ """Base class for all estimators.
89
+
90
+ Hyper-parameters are the keyword arguments of ``__init__``; they are stored
91
+ unchanged as attributes. Learned state uses a trailing underscore
92
+ (e.g. ``coef_``) and only exists after ``fit``.
93
+ """
94
+
95
+ @classmethod
96
+ def _param_names(cls) -> list:
97
+ sig = inspect.signature(cls.__init__)
98
+ return [
99
+ name
100
+ for name, p in sig.parameters.items()
101
+ if name != "self" and p.kind not in (p.VAR_POSITIONAL, p.VAR_KEYWORD)
102
+ ]
103
+
104
+ def get_params(self) -> dict:
105
+ """Return the estimator's hyper-parameters as a dict."""
106
+ return {name: getattr(self, name) for name in self._param_names()}
107
+
108
+ def set_params(self, **params: Any) -> "BaseEstimator":
109
+ """Set hyper-parameters and return ``self``.
110
+
111
+ Raises:
112
+ ValueError: If an unknown parameter name is given.
113
+ """
114
+ valid = set(self._param_names())
115
+ for key, value in params.items():
116
+ if key not in valid:
117
+ raise ValueError(f"Invalid parameter {key!r} for {type(self).__name__}.")
118
+ setattr(self, key, value)
119
+ return self
120
+
121
+ def __repr__(self) -> str:
122
+ args = ", ".join(f"{k}={v!r}" for k, v in self.get_params().items())
123
+ return f"{type(self).__name__}({args})"
124
+
125
+
126
+ def clone(estimator: BaseEstimator) -> BaseEstimator:
127
+ """Return a new, unfitted estimator with the same hyper-parameters."""
128
+ return type(estimator)(**estimator.get_params())
129
+
130
+
131
+ class ClassifierMixin:
132
+ """Adds :meth:`score` returning mean accuracy."""
133
+
134
+ def score(self, X: ArrayLike, y: ArrayLike) -> float:
135
+ """Return the accuracy of ``self.predict(X)`` against ``y``."""
136
+ from .metrics import accuracy_score
137
+
138
+ return accuracy_score(y, self.predict(X)) # type: ignore[attr-defined]
139
+
140
+
141
+ class RegressorMixin:
142
+ """Adds :meth:`score` returning the coefficient of determination R^2."""
143
+
144
+ def score(self, X: ArrayLike, y: ArrayLike) -> float:
145
+ """Return the R^2 of ``self.predict(X)`` against ``y``."""
146
+ from .metrics import r2_score
147
+
148
+ return r2_score(y, self.predict(X)) # type: ignore[attr-defined]
149
+
150
+
151
+ class TransformerMixin:
152
+ """Adds :meth:`fit_transform`."""
153
+
154
+ def fit_transform(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> np.ndarray:
155
+ """Fit to ``X`` and return the transformed data."""
156
+ return self.fit(X, y).transform(X) # type: ignore[attr-defined]
157
+
158
+
159
+ class ClusterMixin:
160
+ """Adds :meth:`fit_predict`."""
161
+
162
+ def fit_predict(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> np.ndarray:
163
+ """Fit to ``X`` and return the cluster label of each sample."""
164
+ self.fit(X) # type: ignore[attr-defined]
165
+ return self.labels_ # type: ignore[attr-defined]
mlkit/cluster.py ADDED
@@ -0,0 +1,108 @@
1
+ """k-means clustering with k-means++ initialisation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Optional
6
+
7
+ import numpy as np
8
+ from numpy.typing import ArrayLike
9
+
10
+ from .base import BaseEstimator, ClusterMixin, check_array, check_is_fitted
11
+ from .neighbors import pairwise_distances
12
+
13
+ __all__ = ["KMeans"]
14
+
15
+
16
+ class KMeans(BaseEstimator, ClusterMixin):
17
+ """Lloyd's k-means with k-means++ seeding and multiple restarts.
18
+
19
+ Args:
20
+ n_clusters: Number of clusters ``k``.
21
+ n_init: Number of independent runs; the run with the lowest inertia wins.
22
+ max_iter: Maximum Lloyd iterations per run.
23
+ tol: Stop a run when no centroid moves more than ``tol``.
24
+ random_state: Seed for reproducible initialisation.
25
+
26
+ Attributes:
27
+ cluster_centers_: Centroids of shape ``(n_clusters, n_features)``.
28
+ labels_: Cluster index of each training sample.
29
+ inertia_: Sum of squared distances of samples to their closest centroid.
30
+ n_iter_: Lloyd iterations used by the best run.
31
+ """
32
+
33
+ def __init__(
34
+ self,
35
+ n_clusters: int = 8,
36
+ n_init: int = 10,
37
+ max_iter: int = 300,
38
+ tol: float = 1e-6,
39
+ random_state: Optional[int] = None,
40
+ ) -> None:
41
+ self.n_clusters = n_clusters
42
+ self.n_init = n_init
43
+ self.max_iter = max_iter
44
+ self.tol = tol
45
+ self.random_state = random_state
46
+ self.cluster_centers_: np.ndarray = None # type: ignore[assignment]
47
+ self.labels_: np.ndarray = None # type: ignore[assignment]
48
+ self.inertia_: float = float("nan")
49
+ self.n_iter_: int = 0
50
+
51
+ @staticmethod
52
+ def _kmeans_pp(X: np.ndarray, k: int, rng: np.random.Generator) -> np.ndarray:
53
+ n = X.shape[0]
54
+ centers = np.empty((k, X.shape[1]))
55
+ centers[0] = X[rng.integers(n)]
56
+ d2 = ((X - centers[0]) ** 2).sum(axis=1)
57
+ for i in range(1, k):
58
+ total = d2.sum()
59
+ j = rng.integers(n) if total == 0 else rng.choice(n, p=d2 / total)
60
+ centers[i] = X[j]
61
+ d2 = np.minimum(d2, ((X - centers[i]) ** 2).sum(axis=1))
62
+ return centers
63
+
64
+ def _single_run(self, X: np.ndarray, rng: np.random.Generator):
65
+ centers = self._kmeans_pp(X, self.n_clusters, rng)
66
+ n_iter = 0
67
+ for n_iter in range(1, self.max_iter + 1):
68
+ labels = np.argmin(pairwise_distances(X, centers), axis=1)
69
+ new_centers = centers.copy()
70
+ for c in range(self.n_clusters):
71
+ members = X[labels == c]
72
+ if members.shape[0] > 0: # keep an empty cluster's centroid in place
73
+ new_centers[c] = members.mean(axis=0)
74
+ shift = np.abs(new_centers - centers).max()
75
+ centers = new_centers
76
+ if shift <= self.tol:
77
+ break
78
+ labels = np.argmin(pairwise_distances(X, centers), axis=1)
79
+ inertia = float(((X - centers[labels]) ** 2).sum())
80
+ return centers, labels, inertia, n_iter
81
+
82
+ def fit(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> "KMeans":
83
+ """Cluster ``X`` of shape ``(n_samples, n_features)``. ``y`` is ignored."""
84
+ X = check_array(X)
85
+ if not 1 <= self.n_clusters <= X.shape[0]:
86
+ raise ValueError(
87
+ f"n_clusters must be in [1, n_samples={X.shape[0]}], got {self.n_clusters}."
88
+ )
89
+ if self.n_init < 1:
90
+ raise ValueError("n_init must be >= 1.")
91
+ rng = np.random.default_rng(self.random_state)
92
+ best = None
93
+ for _ in range(self.n_init):
94
+ run = self._single_run(X, rng)
95
+ if best is None or run[2] < best[2]:
96
+ best = run
97
+ self.cluster_centers_, self.labels_, self.inertia_, self.n_iter_ = best # type: ignore[misc]
98
+ return self
99
+
100
+ def predict(self, X: ArrayLike) -> np.ndarray:
101
+ """Assign each sample in ``X`` to its nearest centroid."""
102
+ check_is_fitted(self, "cluster_centers_")
103
+ return np.argmin(pairwise_distances(X, self.cluster_centers_), axis=1)
104
+
105
+ def transform(self, X: ArrayLike) -> np.ndarray:
106
+ """Return distances to each centroid, shape ``(n_samples, n_clusters)``."""
107
+ check_is_fitted(self, "cluster_centers_")
108
+ return pairwise_distances(X, self.cluster_centers_)
mlkit/datasets.py ADDED
@@ -0,0 +1,73 @@
1
+ """Synthetic dataset generators for examples and tests."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Optional, Tuple, Union
6
+
7
+ import numpy as np
8
+ from numpy.typing import ArrayLike
9
+
10
+ __all__ = ["make_blobs", "make_regression"]
11
+
12
+
13
+ def make_blobs(
14
+ n_samples: int = 100,
15
+ n_features: int = 2,
16
+ centers: Union[int, ArrayLike] = 3,
17
+ cluster_std: float = 1.0,
18
+ random_state: Optional[int] = None,
19
+ ) -> Tuple[np.ndarray, np.ndarray]:
20
+ """Generate isotropic Gaussian blobs for clustering/classification.
21
+
22
+ Args:
23
+ n_samples: Total number of points, split as evenly as possible.
24
+ n_features: Dimensionality (ignored if ``centers`` is an array).
25
+ centers: Number of centres (drawn uniformly in ``[-10, 10]``) or an
26
+ explicit array of shape ``(n_centers, n_features)``.
27
+ cluster_std: Standard deviation of each blob.
28
+ random_state: Seed for reproducibility.
29
+
30
+ Returns:
31
+ ``(X, y)`` with ``X`` of shape ``(n_samples, n_features)`` and integer
32
+ labels ``y`` of shape ``(n_samples,)``.
33
+ """
34
+ rng = np.random.default_rng(random_state)
35
+ if isinstance(centers, (int, np.integer)):
36
+ C = rng.uniform(-10.0, 10.0, size=(int(centers), n_features))
37
+ else:
38
+ C = np.asarray(centers, dtype=float)
39
+ if C.ndim != 2:
40
+ raise ValueError("centers must be an int or a 2-D array.")
41
+ k = C.shape[0]
42
+ counts = np.full(k, n_samples // k)
43
+ counts[: n_samples % k] += 1
44
+ y = np.repeat(np.arange(k), counts)
45
+ X = C[y] + rng.normal(scale=cluster_std, size=(n_samples, C.shape[1]))
46
+ perm = rng.permutation(n_samples)
47
+ return X[perm], y[perm]
48
+
49
+
50
+ def make_regression(
51
+ n_samples: int = 100,
52
+ n_features: int = 3,
53
+ noise: float = 0.0,
54
+ bias: float = 0.0,
55
+ random_state: Optional[int] = None,
56
+ ) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
57
+ """Generate a random linear regression problem ``y = X @ coef + bias + noise``.
58
+
59
+ Args:
60
+ n_samples: Number of samples.
61
+ n_features: Number of features.
62
+ noise: Standard deviation of Gaussian noise added to ``y``.
63
+ bias: Intercept of the underlying model.
64
+ random_state: Seed for reproducibility.
65
+
66
+ Returns:
67
+ ``(X, y, coef)`` where ``coef`` holds the true weights.
68
+ """
69
+ rng = np.random.default_rng(random_state)
70
+ X = rng.normal(size=(n_samples, n_features))
71
+ coef = rng.uniform(-5.0, 5.0, size=n_features)
72
+ y = X @ coef + bias + rng.normal(scale=noise, size=n_samples) if noise > 0 else X @ coef + bias
73
+ return X, y, coef
mlkit/linear_model.py ADDED
@@ -0,0 +1,191 @@
1
+ """Linear models: ordinary least squares, ridge and logistic regression."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Tuple
6
+
7
+ import numpy as np
8
+ from numpy.typing import ArrayLike
9
+
10
+ from .base import BaseEstimator, ClassifierMixin, RegressorMixin, check_array, check_is_fitted, check_X_y
11
+
12
+ __all__ = ["LinearRegression", "Ridge", "LogisticRegression"]
13
+
14
+
15
+ def _center(X: np.ndarray, y: np.ndarray, fit_intercept: bool) -> Tuple[np.ndarray, np.ndarray, np.ndarray, float]:
16
+ if fit_intercept:
17
+ X_mean = X.mean(axis=0)
18
+ y_mean = float(y.mean())
19
+ return X - X_mean, y - y_mean, X_mean, y_mean
20
+ return X, y, np.zeros(X.shape[1]), 0.0
21
+
22
+
23
+ class _LinearRegressorBase(BaseEstimator, RegressorMixin):
24
+ coef_: np.ndarray
25
+ intercept_: float
26
+
27
+ def predict(self, X: ArrayLike) -> np.ndarray:
28
+ """Predict targets for ``X`` of shape ``(n_samples, n_features)``."""
29
+ check_is_fitted(self, "coef_")
30
+ X = check_array(X)
31
+ if X.shape[1] != self.coef_.shape[0]:
32
+ raise ValueError(f"X has {X.shape[1]} features, expected {self.coef_.shape[0]}.")
33
+ return X @ self.coef_ + self.intercept_
34
+
35
+
36
+ class LinearRegression(_LinearRegressorBase):
37
+ """Ordinary least squares linear regression.
38
+
39
+ Solves ``min ||y - X w - b||^2`` with :func:`numpy.linalg.lstsq`, which is
40
+ robust to rank-deficient ``X``.
41
+
42
+ Args:
43
+ fit_intercept: Whether to learn an intercept ``b``.
44
+
45
+ Attributes:
46
+ coef_: Weights of shape ``(n_features,)``.
47
+ intercept_: Bias term (``0.0`` if ``fit_intercept=False``).
48
+ """
49
+
50
+ def __init__(self, fit_intercept: bool = True) -> None:
51
+ self.fit_intercept = fit_intercept
52
+ self.coef_ = None # type: ignore[assignment]
53
+ self.intercept_ = 0.0
54
+
55
+ def fit(self, X: ArrayLike, y: ArrayLike) -> "LinearRegression":
56
+ """Fit the model to ``X`` of shape ``(n_samples, n_features)`` and ``y``."""
57
+ X, y = check_X_y(X, y, y_numeric=True)
58
+ Xc, yc, X_mean, y_mean = _center(X, y, self.fit_intercept)
59
+ coef, *_ = np.linalg.lstsq(Xc, yc, rcond=None)
60
+ self.coef_ = coef
61
+ self.intercept_ = float(y_mean - X_mean @ coef)
62
+ return self
63
+
64
+
65
+ class Ridge(_LinearRegressorBase):
66
+ """Linear regression with L2 regularisation (closed form).
67
+
68
+ Minimises ``||y - X w - b||^2 + alpha * ||w||^2``. The intercept is not
69
+ penalised.
70
+
71
+ Args:
72
+ alpha: Non-negative regularisation strength.
73
+ fit_intercept: Whether to learn an intercept ``b``.
74
+
75
+ Attributes:
76
+ coef_: Weights of shape ``(n_features,)``.
77
+ intercept_: Bias term.
78
+ """
79
+
80
+ def __init__(self, alpha: float = 1.0, fit_intercept: bool = True) -> None:
81
+ self.alpha = alpha
82
+ self.fit_intercept = fit_intercept
83
+ self.coef_ = None # type: ignore[assignment]
84
+ self.intercept_ = 0.0
85
+
86
+ def fit(self, X: ArrayLike, y: ArrayLike) -> "Ridge":
87
+ """Fit the model to ``X`` of shape ``(n_samples, n_features)`` and ``y``."""
88
+ if self.alpha < 0:
89
+ raise ValueError("alpha must be non-negative.")
90
+ X, y = check_X_y(X, y, y_numeric=True)
91
+ Xc, yc, X_mean, y_mean = _center(X, y, self.fit_intercept)
92
+ n_features = X.shape[1]
93
+ A = Xc.T @ Xc + self.alpha * np.eye(n_features)
94
+ coef = np.linalg.lstsq(A, Xc.T @ yc, rcond=None)[0]
95
+ self.coef_ = coef
96
+ self.intercept_ = float(y_mean - X_mean @ coef)
97
+ return self
98
+
99
+
100
+ def _softmax(z: np.ndarray) -> np.ndarray:
101
+ z = z - z.max(axis=1, keepdims=True)
102
+ e = np.exp(z)
103
+ return e / e.sum(axis=1, keepdims=True)
104
+
105
+
106
+ class LogisticRegression(BaseEstimator, ClassifierMixin):
107
+ """Multinomial logistic regression trained by full-batch gradient descent.
108
+
109
+ Binary problems are handled as the two-class case of the softmax model.
110
+ Features should be on comparable scales (see
111
+ :class:`mlkit.preprocessing.StandardScaler`) for fast convergence.
112
+
113
+ Args:
114
+ learning_rate: Gradient descent step size.
115
+ max_iter: Maximum number of gradient steps.
116
+ tol: Stop when the largest absolute gradient entry falls below ``tol``.
117
+ alpha: L2 penalty on the weights (not the intercepts).
118
+ fit_intercept: Whether to learn per-class intercepts.
119
+
120
+ Attributes:
121
+ classes_: Sorted unique class labels, shape ``(n_classes,)``.
122
+ coef_: Weights of shape ``(n_classes, n_features)``.
123
+ intercept_: Intercepts of shape ``(n_classes,)``.
124
+ n_iter_: Number of gradient steps actually taken.
125
+ """
126
+
127
+ def __init__(
128
+ self,
129
+ learning_rate: float = 0.1,
130
+ max_iter: int = 1000,
131
+ tol: float = 1e-6,
132
+ alpha: float = 0.0,
133
+ fit_intercept: bool = True,
134
+ ) -> None:
135
+ self.learning_rate = learning_rate
136
+ self.max_iter = max_iter
137
+ self.tol = tol
138
+ self.alpha = alpha
139
+ self.fit_intercept = fit_intercept
140
+ self.classes_: np.ndarray = None # type: ignore[assignment]
141
+ self.coef_: np.ndarray = None # type: ignore[assignment]
142
+ self.intercept_: np.ndarray = None # type: ignore[assignment]
143
+ self.n_iter_: int = 0
144
+
145
+ def fit(self, X: ArrayLike, y: ArrayLike) -> "LogisticRegression":
146
+ """Fit the model to ``X`` of shape ``(n_samples, n_features)`` and labels ``y``."""
147
+ X, y = check_X_y(X, y)
148
+ classes, y_idx = np.unique(y, return_inverse=True)
149
+ if classes.shape[0] < 2:
150
+ raise ValueError("LogisticRegression needs at least two classes in y.")
151
+ n_samples, n_features = X.shape
152
+ n_classes = classes.shape[0]
153
+ Y = np.eye(n_classes)[y_idx]
154
+
155
+ W = np.zeros((n_classes, n_features))
156
+ b = np.zeros(n_classes)
157
+ n_iter = 0
158
+ for n_iter in range(1, self.max_iter + 1):
159
+ P = _softmax(X @ W.T + b)
160
+ diff = (P - Y) / n_samples
161
+ grad_W = diff.T @ X + self.alpha * W
162
+ grad_b = diff.sum(axis=0) if self.fit_intercept else np.zeros(n_classes)
163
+ W -= self.learning_rate * grad_W
164
+ b -= self.learning_rate * grad_b
165
+ if max(np.abs(grad_W).max(), np.abs(grad_b).max()) < self.tol:
166
+ break
167
+
168
+ self.classes_ = classes
169
+ self.coef_ = W
170
+ self.intercept_ = b
171
+ self.n_iter_ = n_iter
172
+ return self
173
+
174
+ def decision_function(self, X: ArrayLike) -> np.ndarray:
175
+ """Return per-class linear scores of shape ``(n_samples, n_classes)``."""
176
+ check_is_fitted(self, "coef_")
177
+ X = check_array(X)
178
+ if X.shape[1] != self.coef_.shape[1]:
179
+ raise ValueError(f"X has {X.shape[1]} features, expected {self.coef_.shape[1]}.")
180
+ return X @ self.coef_.T + self.intercept_
181
+
182
+ def predict_proba(self, X: ArrayLike) -> np.ndarray:
183
+ """Return class probabilities of shape ``(n_samples, n_classes)``.
184
+
185
+ Columns follow the order of ``classes_``.
186
+ """
187
+ return _softmax(self.decision_function(X))
188
+
189
+ def predict(self, X: ArrayLike) -> np.ndarray:
190
+ """Return the most probable class label for each sample."""
191
+ return self.classes_[np.argmax(self.decision_function(X), axis=1)]
mlkit/metrics.py ADDED
@@ -0,0 +1,139 @@
1
+ """Evaluation metrics for regression and classification."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any, Optional, Sequence, Tuple
6
+
7
+ import numpy as np
8
+ from numpy.typing import ArrayLike
9
+
10
+ __all__ = [
11
+ "accuracy_score",
12
+ "confusion_matrix",
13
+ "precision_score",
14
+ "recall_score",
15
+ "f1_score",
16
+ "mean_squared_error",
17
+ "mean_absolute_error",
18
+ "r2_score",
19
+ ]
20
+
21
+ _AVERAGES = ("binary", "macro")
22
+
23
+
24
+ def _check_pair(y_true: ArrayLike, y_pred: ArrayLike, numeric: bool = False) -> Tuple[np.ndarray, np.ndarray]:
25
+ dtype = float if numeric else None
26
+ t = np.asarray(y_true, dtype=dtype)
27
+ p = np.asarray(y_pred, dtype=dtype)
28
+ if t.ndim != 1 or p.ndim != 1:
29
+ raise ValueError("y_true and y_pred must be 1-D.")
30
+ if t.shape[0] != p.shape[0]:
31
+ raise ValueError(f"Inconsistent lengths: {t.shape[0]} != {p.shape[0]}.")
32
+ if t.shape[0] == 0:
33
+ raise ValueError("Empty input.")
34
+ return t, p
35
+
36
+
37
+ # --------------------------------------------------------------------------- #
38
+ # Regression
39
+ # --------------------------------------------------------------------------- #
40
+
41
+
42
+ def mean_squared_error(y_true: ArrayLike, y_pred: ArrayLike) -> float:
43
+ """Mean of squared residuals."""
44
+ t, p = _check_pair(y_true, y_pred, numeric=True)
45
+ return float(np.mean((t - p) ** 2))
46
+
47
+
48
+ def mean_absolute_error(y_true: ArrayLike, y_pred: ArrayLike) -> float:
49
+ """Mean of absolute residuals."""
50
+ t, p = _check_pair(y_true, y_pred, numeric=True)
51
+ return float(np.mean(np.abs(t - p)))
52
+
53
+
54
+ def r2_score(y_true: ArrayLike, y_pred: ArrayLike) -> float:
55
+ """Coefficient of determination ``1 - SS_res / SS_tot``.
56
+
57
+ Returns 1.0 for a perfect fit of constant ``y_true`` and 0.0 for an
58
+ imperfect one (where ``SS_tot`` is zero).
59
+ """
60
+ t, p = _check_pair(y_true, y_pred, numeric=True)
61
+ ss_res = float(np.sum((t - p) ** 2))
62
+ ss_tot = float(np.sum((t - t.mean()) ** 2))
63
+ if ss_tot == 0.0:
64
+ return 1.0 if ss_res == 0.0 else 0.0
65
+ return 1.0 - ss_res / ss_tot
66
+
67
+
68
+ # --------------------------------------------------------------------------- #
69
+ # Classification
70
+ # --------------------------------------------------------------------------- #
71
+
72
+
73
+ def accuracy_score(y_true: ArrayLike, y_pred: ArrayLike) -> float:
74
+ """Fraction of exactly matching labels."""
75
+ t, p = _check_pair(y_true, y_pred)
76
+ return float(np.mean(t == p))
77
+
78
+
79
+ def confusion_matrix(y_true: ArrayLike, y_pred: ArrayLike, labels: Optional[Sequence[Any]] = None) -> np.ndarray:
80
+ """Count matrix ``C`` where ``C[i, j]`` is the number of samples of true
81
+ class ``labels[i]`` predicted as ``labels[j]``.
82
+
83
+ Args:
84
+ y_true: True labels.
85
+ y_pred: Predicted labels.
86
+ labels: Label order; defaults to the sorted union of both inputs.
87
+ Samples whose labels are not listed are ignored.
88
+ """
89
+ t, p = _check_pair(y_true, y_pred)
90
+ lab = np.unique(np.concatenate([t, p])) if labels is None else np.asarray(labels)
91
+ index = {v: i for i, v in enumerate(lab.tolist())}
92
+ cm = np.zeros((len(lab), len(lab)), dtype=int)
93
+ for a, b in zip(t.tolist(), p.tolist()):
94
+ if a in index and b in index:
95
+ cm[index[a], index[b]] += 1
96
+ return cm
97
+
98
+
99
+ def _per_class_counts(t: np.ndarray, p: np.ndarray, average: str, pos_label: Any):
100
+ if average not in _AVERAGES:
101
+ raise ValueError(f"average must be one of {_AVERAGES}, got {average!r}.")
102
+ labels = [pos_label] if average == "binary" else np.unique(np.concatenate([t, p])).tolist()
103
+ tp = np.array([np.sum((p == c) & (t == c)) for c in labels], dtype=float)
104
+ fp = np.array([np.sum((p == c) & (t != c)) for c in labels], dtype=float)
105
+ fn = np.array([np.sum((p != c) & (t == c)) for c in labels], dtype=float)
106
+ return tp, fp, fn
107
+
108
+
109
+ def _safe_div(num: np.ndarray, den: np.ndarray) -> np.ndarray:
110
+ return np.divide(num, den, out=np.zeros_like(num), where=den != 0)
111
+
112
+
113
+ def precision_score(y_true: ArrayLike, y_pred: ArrayLike, *, average: str = "binary", pos_label: Any = 1) -> float:
114
+ """Precision ``tp / (tp + fp)``.
115
+
116
+ Args:
117
+ average: ``"binary"`` scores only ``pos_label``; ``"macro"`` averages
118
+ the per-class scores over all labels present.
119
+ pos_label: Positive class used when ``average="binary"``.
120
+
121
+ Undefined ratios (zero denominator) count as 0.0.
122
+ """
123
+ t, p = _check_pair(y_true, y_pred)
124
+ tp, fp, _ = _per_class_counts(t, p, average, pos_label)
125
+ return float(np.mean(_safe_div(tp, tp + fp)))
126
+
127
+
128
+ def recall_score(y_true: ArrayLike, y_pred: ArrayLike, *, average: str = "binary", pos_label: Any = 1) -> float:
129
+ """Recall ``tp / (tp + fn)``. See :func:`precision_score` for arguments."""
130
+ t, p = _check_pair(y_true, y_pred)
131
+ tp, _, fn = _per_class_counts(t, p, average, pos_label)
132
+ return float(np.mean(_safe_div(tp, tp + fn)))
133
+
134
+
135
+ def f1_score(y_true: ArrayLike, y_pred: ArrayLike, *, average: str = "binary", pos_label: Any = 1) -> float:
136
+ """F1 ``2 tp / (2 tp + fp + fn)``. See :func:`precision_score` for arguments."""
137
+ t, p = _check_pair(y_true, y_pred)
138
+ tp, fp, fn = _per_class_counts(t, p, average, pos_label)
139
+ return float(np.mean(_safe_div(2 * tp, 2 * tp + fp + fn)))
@@ -0,0 +1,117 @@
1
+ """Data splitting and cross-validation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Callable, Iterator, List, Optional, Tuple, Union
6
+
7
+ import numpy as np
8
+ from numpy.typing import ArrayLike
9
+
10
+ from .base import BaseEstimator, clone
11
+
12
+ __all__ = ["train_test_split", "KFold", "cross_val_score"]
13
+
14
+
15
+ def train_test_split(
16
+ *arrays: ArrayLike,
17
+ test_size: Union[float, int] = 0.25,
18
+ shuffle: bool = True,
19
+ random_state: Optional[int] = None,
20
+ ) -> List[np.ndarray]:
21
+ """Split arrays into random train and test subsets.
22
+
23
+ Args:
24
+ *arrays: One or more arrays with the same first dimension.
25
+ test_size: Fraction in ``(0, 1)`` or absolute number of test samples.
26
+ shuffle: Shuffle before splitting.
27
+ random_state: Seed for reproducible shuffling.
28
+
29
+ Returns:
30
+ ``[a_train, a_test, b_train, b_test, ...]`` for each input array.
31
+
32
+ Example:
33
+ >>> X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
34
+ """
35
+ if not arrays:
36
+ raise ValueError("At least one array is required.")
37
+ arrs = [np.asarray(a) for a in arrays]
38
+ n = arrs[0].shape[0]
39
+ if any(a.shape[0] != n for a in arrs):
40
+ raise ValueError("All arrays must have the same number of samples.")
41
+
42
+ if isinstance(test_size, float):
43
+ if not 0.0 < test_size < 1.0:
44
+ raise ValueError("A float test_size must be in (0, 1).")
45
+ n_test = int(np.ceil(test_size * n))
46
+ else:
47
+ n_test = int(test_size)
48
+ if not 0 < n_test < n:
49
+ raise ValueError(f"test_size={test_size} leaves an empty train or test set for n={n}.")
50
+
51
+ idx = np.random.default_rng(random_state).permutation(n) if shuffle else np.arange(n)
52
+ test_idx, train_idx = idx[:n_test], idx[n_test:]
53
+ out: List[np.ndarray] = []
54
+ for a in arrs:
55
+ out.extend([a[train_idx], a[test_idx]])
56
+ return out
57
+
58
+
59
+ class KFold:
60
+ """K-fold cross-validation splitter.
61
+
62
+ Args:
63
+ n_splits: Number of folds (>= 2).
64
+ shuffle: Shuffle sample indices before folding.
65
+ random_state: Seed used when ``shuffle=True``.
66
+ """
67
+
68
+ def __init__(self, n_splits: int = 5, shuffle: bool = False, random_state: Optional[int] = None) -> None:
69
+ if n_splits < 2:
70
+ raise ValueError("n_splits must be >= 2.")
71
+ self.n_splits = n_splits
72
+ self.shuffle = shuffle
73
+ self.random_state = random_state
74
+
75
+ def split(self, X: ArrayLike) -> Iterator[Tuple[np.ndarray, np.ndarray]]:
76
+ """Yield ``(train_indices, test_indices)`` for each fold."""
77
+ n = np.asarray(X).shape[0]
78
+ if self.n_splits > n:
79
+ raise ValueError(f"Cannot make {self.n_splits} folds from {n} samples.")
80
+ idx = np.random.default_rng(self.random_state).permutation(n) if self.shuffle else np.arange(n)
81
+ for test_idx in np.array_split(idx, self.n_splits):
82
+ yield np.setdiff1d(idx, test_idx, assume_unique=True), test_idx
83
+
84
+
85
+ def cross_val_score(
86
+ estimator: BaseEstimator,
87
+ X: ArrayLike,
88
+ y: ArrayLike,
89
+ *,
90
+ cv: Union[int, KFold] = 5,
91
+ scoring: Optional[Callable[[np.ndarray, np.ndarray], float]] = None,
92
+ ) -> np.ndarray:
93
+ """Evaluate ``estimator`` with K-fold cross-validation.
94
+
95
+ A fresh :func:`~mlkit.base.clone` of ``estimator`` is fitted on each fold.
96
+
97
+ Args:
98
+ estimator: An unfitted estimator with ``fit``/``predict``.
99
+ X: Features, shape ``(n_samples, n_features)``.
100
+ y: Targets, shape ``(n_samples,)``.
101
+ cv: Number of folds or a :class:`KFold` instance.
102
+ scoring: ``metric(y_true, y_pred) -> float``; defaults to the
103
+ estimator's own ``score`` method.
104
+
105
+ Returns:
106
+ Array of per-fold scores, shape ``(n_splits,)``.
107
+ """
108
+ X_arr, y_arr = np.asarray(X), np.asarray(y)
109
+ splitter = KFold(n_splits=cv) if isinstance(cv, int) else cv
110
+ scores = []
111
+ for train_idx, test_idx in splitter.split(X_arr):
112
+ model = clone(estimator).fit(X_arr[train_idx], y_arr[train_idx]) # type: ignore[attr-defined]
113
+ if scoring is None:
114
+ scores.append(model.score(X_arr[test_idx], y_arr[test_idx]))
115
+ else:
116
+ scores.append(scoring(y_arr[test_idx], model.predict(X_arr[test_idx])))
117
+ return np.asarray(scores, dtype=float)
mlkit/neighbors.py ADDED
@@ -0,0 +1,132 @@
1
+ """k-nearest-neighbour classification and regression (brute-force Euclidean)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Tuple
6
+
7
+ import numpy as np
8
+ from numpy.typing import ArrayLike
9
+
10
+ from .base import BaseEstimator, ClassifierMixin, RegressorMixin, check_array, check_is_fitted, check_X_y
11
+
12
+ __all__ = ["KNeighborsClassifier", "KNeighborsRegressor", "pairwise_distances"]
13
+
14
+ _WEIGHTS = ("uniform", "distance")
15
+
16
+
17
+ def pairwise_distances(A: ArrayLike, B: ArrayLike) -> np.ndarray:
18
+ """Euclidean distances between rows of ``A`` and rows of ``B``.
19
+
20
+ Args:
21
+ A: Array of shape ``(n_a, n_features)``.
22
+ B: Array of shape ``(n_b, n_features)``.
23
+
24
+ Returns:
25
+ Array of shape ``(n_a, n_b)``.
26
+ """
27
+ A = check_array(A)
28
+ B = check_array(B)
29
+ if A.shape[1] != B.shape[1]:
30
+ raise ValueError(f"Feature mismatch: {A.shape[1]} != {B.shape[1]}.")
31
+ sq = (A * A).sum(axis=1)[:, None] - 2.0 * A @ B.T + (B * B).sum(axis=1)[None, :]
32
+ return np.sqrt(np.maximum(sq, 0.0))
33
+
34
+
35
+ class _KNeighborsBase(BaseEstimator):
36
+ def __init__(self, n_neighbors: int = 5, weights: str = "uniform") -> None:
37
+ self.n_neighbors = n_neighbors
38
+ self.weights = weights
39
+ self._fit_X: np.ndarray = None # type: ignore[assignment]
40
+ self._fit_y: np.ndarray = None # type: ignore[assignment]
41
+
42
+ def _validate_params(self, n_samples: int) -> None:
43
+ if self.weights not in _WEIGHTS:
44
+ raise ValueError(f"weights must be one of {_WEIGHTS}, got {self.weights!r}.")
45
+ if not 1 <= self.n_neighbors <= n_samples:
46
+ raise ValueError(
47
+ f"n_neighbors must be in [1, n_samples={n_samples}], got {self.n_neighbors}."
48
+ )
49
+
50
+ def kneighbors(self, X: ArrayLike) -> Tuple[np.ndarray, np.ndarray]:
51
+ """Find the ``n_neighbors`` nearest training samples for each row of ``X``.
52
+
53
+ Returns:
54
+ ``(distances, indices)``, each of shape ``(n_samples, n_neighbors)``,
55
+ sorted by increasing distance.
56
+ """
57
+ check_is_fitted(self, "_fit_X")
58
+ D = pairwise_distances(X, self._fit_X)
59
+ idx = np.argsort(D, axis=1, kind="stable")[:, : self.n_neighbors]
60
+ return np.take_along_axis(D, idx, axis=1), idx
61
+
62
+ def _neighbor_weights(self, dist: np.ndarray) -> np.ndarray:
63
+ if self.weights == "uniform":
64
+ return np.ones_like(dist)
65
+ # Exact matches dominate: give them weight 1 and everything else 0.
66
+ with np.errstate(divide="ignore"):
67
+ w = 1.0 / dist
68
+ exact = np.isinf(w)
69
+ rows = exact.any(axis=1)
70
+ w[rows] = exact[rows].astype(float)
71
+ return w
72
+
73
+
74
+ class KNeighborsClassifier(_KNeighborsBase, ClassifierMixin):
75
+ """Classifier voting among the ``k`` nearest training samples.
76
+
77
+ Args:
78
+ n_neighbors: Number of neighbours ``k``.
79
+ weights: ``"uniform"`` (each neighbour counts equally) or
80
+ ``"distance"`` (votes weighted by inverse distance).
81
+
82
+ Attributes:
83
+ classes_: Sorted unique class labels.
84
+ """
85
+
86
+ def __init__(self, n_neighbors: int = 5, weights: str = "uniform") -> None:
87
+ super().__init__(n_neighbors=n_neighbors, weights=weights)
88
+ self.classes_: np.ndarray = None # type: ignore[assignment]
89
+
90
+ def fit(self, X: ArrayLike, y: ArrayLike) -> "KNeighborsClassifier":
91
+ """Store the training data ``X`` and labels ``y``."""
92
+ X, y = check_X_y(X, y)
93
+ self._validate_params(X.shape[0])
94
+ self.classes_, self._fit_y = np.unique(y, return_inverse=True)
95
+ self._fit_X = X
96
+ return self
97
+
98
+ def predict_proba(self, X: ArrayLike) -> np.ndarray:
99
+ """Return vote shares of shape ``(n_samples, n_classes)`` (columns follow ``classes_``)."""
100
+ dist, idx = self.kneighbors(X)
101
+ w = self._neighbor_weights(dist)
102
+ labels = self._fit_y[idx]
103
+ proba = np.zeros((idx.shape[0], self.classes_.shape[0]))
104
+ for c in range(self.classes_.shape[0]):
105
+ proba[:, c] = (w * (labels == c)).sum(axis=1)
106
+ return proba / proba.sum(axis=1, keepdims=True)
107
+
108
+ def predict(self, X: ArrayLike) -> np.ndarray:
109
+ """Return the majority-vote label for each sample (ties go to the smaller class)."""
110
+ return self.classes_[np.argmax(self.predict_proba(X), axis=1)]
111
+
112
+
113
+ class KNeighborsRegressor(_KNeighborsBase, RegressorMixin):
114
+ """Regressor averaging the targets of the ``k`` nearest training samples.
115
+
116
+ Args:
117
+ n_neighbors: Number of neighbours ``k``.
118
+ weights: ``"uniform"`` or ``"distance"`` (inverse-distance weighted mean).
119
+ """
120
+
121
+ def fit(self, X: ArrayLike, y: ArrayLike) -> "KNeighborsRegressor":
122
+ """Store the training data ``X`` and numeric targets ``y``."""
123
+ X, y = check_X_y(X, y, y_numeric=True)
124
+ self._validate_params(X.shape[0])
125
+ self._fit_X, self._fit_y = X, y
126
+ return self
127
+
128
+ def predict(self, X: ArrayLike) -> np.ndarray:
129
+ """Return the (weighted) mean neighbour target for each sample."""
130
+ dist, idx = self.kneighbors(X)
131
+ w = self._neighbor_weights(dist)
132
+ return (w * self._fit_y[idx]).sum(axis=1) / w.sum(axis=1)
mlkit/preprocessing.py ADDED
@@ -0,0 +1,111 @@
1
+ """Feature scaling transformers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Optional, Tuple
6
+
7
+ import numpy as np
8
+ from numpy.typing import ArrayLike
9
+
10
+ from .base import BaseEstimator, TransformerMixin, check_array, check_is_fitted
11
+
12
+ __all__ = ["StandardScaler", "MinMaxScaler"]
13
+
14
+
15
+ def _check_n_features(X: np.ndarray, expected: int) -> None:
16
+ if X.shape[1] != expected:
17
+ raise ValueError(f"X has {X.shape[1]} features, expected {expected}.")
18
+
19
+
20
+ class StandardScaler(BaseEstimator, TransformerMixin):
21
+ """Standardise features to zero mean and unit variance.
22
+
23
+ Constant features (zero variance) are left centred but not scaled.
24
+
25
+ Args:
26
+ with_mean: Subtract the per-feature mean.
27
+ with_std: Divide by the per-feature (population) standard deviation.
28
+
29
+ Attributes:
30
+ mean_: Per-feature means, shape ``(n_features,)``.
31
+ scale_: Per-feature divisors, shape ``(n_features,)``.
32
+ """
33
+
34
+ def __init__(self, with_mean: bool = True, with_std: bool = True) -> None:
35
+ self.with_mean = with_mean
36
+ self.with_std = with_std
37
+ self.mean_: np.ndarray = None # type: ignore[assignment]
38
+ self.scale_: np.ndarray = None # type: ignore[assignment]
39
+
40
+ def fit(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> "StandardScaler":
41
+ """Compute the mean and standard deviation of each column of ``X``."""
42
+ X = check_array(X)
43
+ self.mean_ = X.mean(axis=0) if self.with_mean else np.zeros(X.shape[1])
44
+ if self.with_std:
45
+ std = X.std(axis=0)
46
+ self.scale_ = np.where(std == 0.0, 1.0, std)
47
+ else:
48
+ self.scale_ = np.ones(X.shape[1])
49
+ return self
50
+
51
+ def transform(self, X: ArrayLike) -> np.ndarray:
52
+ """Return ``(X - mean_) / scale_``."""
53
+ check_is_fitted(self, "mean_")
54
+ X = check_array(X)
55
+ _check_n_features(X, self.mean_.shape[0])
56
+ return (X - self.mean_) / self.scale_
57
+
58
+ def inverse_transform(self, X: ArrayLike) -> np.ndarray:
59
+ """Undo :meth:`transform`."""
60
+ check_is_fitted(self, "mean_")
61
+ X = check_array(X)
62
+ _check_n_features(X, self.mean_.shape[0])
63
+ return X * self.scale_ + self.mean_
64
+
65
+
66
+ class MinMaxScaler(BaseEstimator, TransformerMixin):
67
+ """Rescale each feature linearly into ``feature_range``.
68
+
69
+ Constant features are mapped to the lower bound of the range.
70
+
71
+ Args:
72
+ feature_range: Target ``(min, max)`` interval.
73
+
74
+ Attributes:
75
+ data_min_: Per-feature minimum seen in ``fit``.
76
+ data_max_: Per-feature maximum seen in ``fit``.
77
+ """
78
+
79
+ def __init__(self, feature_range: Tuple[float, float] = (0.0, 1.0)) -> None:
80
+ self.feature_range = feature_range
81
+ self.data_min_: np.ndarray = None # type: ignore[assignment]
82
+ self.data_max_: np.ndarray = None # type: ignore[assignment]
83
+
84
+ def fit(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> "MinMaxScaler":
85
+ """Record the per-feature minimum and maximum of ``X``."""
86
+ lo, hi = self.feature_range
87
+ if not lo < hi:
88
+ raise ValueError(f"feature_range min must be < max, got {self.feature_range}.")
89
+ X = check_array(X)
90
+ self.data_min_ = X.min(axis=0)
91
+ self.data_max_ = X.max(axis=0)
92
+ return self
93
+
94
+ def _scale(self) -> np.ndarray:
95
+ rng = self.data_max_ - self.data_min_
96
+ lo, hi = self.feature_range
97
+ return (hi - lo) / np.where(rng == 0.0, 1.0, rng)
98
+
99
+ def transform(self, X: ArrayLike) -> np.ndarray:
100
+ """Map ``X`` into ``feature_range`` (values outside the fitted range are not clipped)."""
101
+ check_is_fitted(self, "data_min_")
102
+ X = check_array(X)
103
+ _check_n_features(X, self.data_min_.shape[0])
104
+ return (X - self.data_min_) * self._scale() + self.feature_range[0]
105
+
106
+ def inverse_transform(self, X: ArrayLike) -> np.ndarray:
107
+ """Undo :meth:`transform`."""
108
+ check_is_fitted(self, "data_min_")
109
+ X = check_array(X)
110
+ _check_n_features(X, self.data_min_.shape[0])
111
+ return (X - self.feature_range[0]) / self._scale() + self.data_min_
mlkit/py.typed ADDED
File without changes
@@ -0,0 +1,191 @@
1
+ Metadata-Version: 2.4
2
+ Name: mlkit
3
+ Version: 0.1.0
4
+ Summary: Machine Learning Toolkit: a small, readable NumPy toolkit with a fit/predict API, classic models, preprocessing and metrics.
5
+ Author: nehz
6
+ License-Expression: MIT
7
+ Keywords: machine learning,regression,classification,clustering,numpy
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: Intended Audience :: Education
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3 :: Only
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Typing :: Typed
22
+ Requires-Python: >=3.9
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: numpy>=1.22
26
+ Provides-Extra: test
27
+ Requires-Dist: pytest>=7; extra == "test"
28
+ Dynamic: license-file
29
+
30
+ # mlkit
31
+
32
+ **Machine Learning Toolkit**: a small, readable machine learning library built on NumPy.
33
+
34
+ mlkit has the familiar `fit` / `predict` / `transform` estimator interface and a
35
+ handful of classic algorithms, each implemented in a few dozen lines of plain
36
+ NumPy. Use it to learn how the algorithms work, to teach them, or for small
37
+ projects where a large ML stack is too much. NumPy is its only dependency.
38
+
39
+ ## Features
40
+
41
+ - **Estimator API**: `fit`, `predict`, `transform`, `score`, `get_params` / `set_params`, `clone`
42
+ - **Linear models**: `LinearRegression` (least squares), `Ridge` (L2, closed form), `LogisticRegression` (multinomial, gradient descent, optional L2)
43
+ - **Neighbours**: `KNeighborsClassifier`, `KNeighborsRegressor` (uniform or inverse-distance weights)
44
+ - **Clustering**: `KMeans` (k-means++ seeding, multiple restarts, reproducible via `random_state`)
45
+ - **Preprocessing**: `StandardScaler`, `MinMaxScaler` (both with `inverse_transform`)
46
+ - **Model selection**: `train_test_split`, `KFold`, `cross_val_score`
47
+ - **Metrics**: accuracy, confusion matrix, precision / recall / F1 (binary and macro), MSE, MAE, R²
48
+ - **Datasets**: `make_blobs`, `make_regression` synthetic generators
49
+ - Input validation with clear errors, and `NotFittedError` when an estimator is used before `fit`
50
+ - Type hints throughout (ships `py.typed`)
51
+
52
+ ## Installation
53
+
54
+ ```bash
55
+ pip install mlkit
56
+ ```
57
+
58
+ From a source checkout:
59
+
60
+ ```bash
61
+ python3 -m venv .venv
62
+ .venv/bin/pip install -e ".[test]"
63
+ .venv/bin/python -m pytest
64
+ ```
65
+
66
+ Requires Python 3.9+ and NumPy 1.22+.
67
+
68
+ ## Quickstart
69
+
70
+ ```python
71
+ from mlkit import KMeans, LinearRegression, LogisticRegression, StandardScaler, train_test_split
72
+ from mlkit.datasets import make_blobs, make_regression
73
+ from mlkit.metrics import accuracy_score, r2_score
74
+
75
+ # Classification: scale the features, then fit a logistic regression
76
+ X, y = make_blobs(n_samples=300, centers=3, random_state=0)
77
+ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=0)
78
+
79
+ scaler = StandardScaler().fit(X_train)
80
+ clf = LogisticRegression(learning_rate=0.5, max_iter=2000)
81
+ clf.fit(scaler.transform(X_train), y_train)
82
+ print("accuracy:", accuracy_score(y_test, clf.predict(scaler.transform(X_test))))
83
+
84
+ # Regression
85
+ X, y, true_coef = make_regression(n_samples=200, n_features=3, noise=0.5, random_state=1)
86
+ reg = LinearRegression().fit(X, y)
87
+ print("R^2:", r2_score(y, reg.predict(X)), "coef:", reg.coef_)
88
+
89
+ # Clustering
90
+ km = KMeans(n_clusters=3, random_state=0).fit(X_train)
91
+ print("inertia:", km.inertia_, "labels:", km.labels_[:10])
92
+ ```
93
+
94
+ Cross-validation:
95
+
96
+ ```python
97
+ from mlkit import KFold, KNeighborsClassifier, cross_val_score
98
+
99
+ scores = cross_val_score(KNeighborsClassifier(n_neighbors=5), X_train, y_train,
100
+ cv=KFold(n_splits=5, shuffle=True, random_state=0))
101
+ print(scores.mean())
102
+ ```
103
+
104
+ ## API overview
105
+
106
+ All estimators subclass `mlkit.BaseEstimator`. Hyper-parameters are the
107
+ constructor arguments. Learned attributes end in `_` and exist only after
108
+ `fit`. Every `fit` returns `self`. Calling `predict` or `transform` before `fit`
109
+ raises `mlkit.NotFittedError`.
110
+
111
+ ### `mlkit` (top level)
112
+
113
+ | Name | Description |
114
+ | --- | --- |
115
+ | `BaseEstimator` | Base class with `get_params()`, `set_params(**params)`, and a readable `repr` |
116
+ | `NotFittedError` | Raised when an estimator is used before `fit` |
117
+ | `clone(estimator)` | New unfitted estimator with the same hyper-parameters |
118
+ | `__version__` | `"0.1.0"` |
119
+
120
+ ### `mlkit.linear_model`
121
+
122
+ | Class | Parameters | Methods | Fitted attributes |
123
+ | --- | --- | --- | --- |
124
+ | `LinearRegression` | `fit_intercept=True` | `fit(X, y)`, `predict(X)`, `score(X, y)` (R²) | `coef_`, `intercept_` |
125
+ | `Ridge` | `alpha=1.0`, `fit_intercept=True` | `fit`, `predict`, `score` (R²) | `coef_`, `intercept_` |
126
+ | `LogisticRegression` | `learning_rate=0.1`, `max_iter=1000`, `tol=1e-6`, `alpha=0.0`, `fit_intercept=True` | `fit`, `predict`, `predict_proba`, `decision_function`, `score` (accuracy) | `classes_`, `coef_` (n_classes × n_features), `intercept_`, `n_iter_` |
127
+
128
+ ### `mlkit.neighbors`
129
+
130
+ | Name | Parameters | Methods / notes |
131
+ | --- | --- | --- |
132
+ | `KNeighborsClassifier` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `predict_proba`, `kneighbors(X)`, `score` (accuracy); `classes_` |
133
+ | `KNeighborsRegressor` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `kneighbors(X)`, `score` (R²) |
134
+ | `pairwise_distances(A, B)` | | Euclidean distance matrix of shape `(len(A), len(B))` |
135
+
136
+ ### `mlkit.cluster`
137
+
138
+ | Class | Parameters | Methods | Fitted attributes |
139
+ | --- | --- | --- | --- |
140
+ | `KMeans` | `n_clusters=8`, `n_init=10`, `max_iter=300`, `tol=1e-6`, `random_state=None` | `fit(X)`, `predict(X)`, `fit_predict(X)`, `transform(X)` (distances to centroids) | `cluster_centers_`, `labels_`, `inertia_`, `n_iter_` |
141
+
142
+ ### `mlkit.preprocessing`
143
+
144
+ | Class | Parameters | Methods | Fitted attributes |
145
+ | --- | --- | --- | --- |
146
+ | `StandardScaler` | `with_mean=True`, `with_std=True` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `mean_`, `scale_` |
147
+ | `MinMaxScaler` | `feature_range=(0.0, 1.0)` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `data_min_`, `data_max_` |
148
+
149
+ ### `mlkit.model_selection`
150
+
151
+ | Name | Signature |
152
+ | --- | --- |
153
+ | `train_test_split` | `train_test_split(*arrays, test_size=0.25, shuffle=True, random_state=None)` returns `[a_train, a_test, b_train, b_test, ...]` |
154
+ | `KFold` | `KFold(n_splits=5, shuffle=False, random_state=None)`; `.split(X)` yields `(train_idx, test_idx)` |
155
+ | `cross_val_score` | `cross_val_score(estimator, X, y, *, cv=5, scoring=None)` returns an array of per-fold scores (`scoring` is `metric(y_true, y_pred)`; it defaults to `estimator.score`) |
156
+
157
+ ### `mlkit.metrics`
158
+
159
+ | Function | Notes |
160
+ | --- | --- |
161
+ | `accuracy_score(y_true, y_pred)` | Fraction of exact matches |
162
+ | `confusion_matrix(y_true, y_pred, labels=None)` | `C[i, j]`: true `labels[i]` predicted as `labels[j]` |
163
+ | `precision_score(y_true, y_pred, *, average="binary", pos_label=1)` | `average` is `"binary"` or `"macro"`. A zero denominator gives 0.0 |
164
+ | `recall_score(...)` | Same arguments as `precision_score` |
165
+ | `f1_score(...)` | Same arguments as `precision_score` |
166
+ | `mean_squared_error(y_true, y_pred)` | |
167
+ | `mean_absolute_error(y_true, y_pred)` | |
168
+ | `r2_score(y_true, y_pred)` | Coefficient of determination |
169
+
170
+ ### `mlkit.datasets`
171
+
172
+ | Function | Returns |
173
+ | --- | --- |
174
+ | `make_blobs(n_samples=100, n_features=2, centers=3, cluster_std=1.0, random_state=None)` | `(X, y)` |
175
+ | `make_regression(n_samples=100, n_features=3, noise=0.0, bias=0.0, random_state=None)` | `(X, y, coef)` |
176
+
177
+ ### `mlkit.base`
178
+
179
+ Contains the `ClassifierMixin`, `RegressorMixin`, `TransformerMixin` and
180
+ `ClusterMixin` mixins for writing your own estimators, plus the validation
181
+ helpers `check_array`, `check_X_y` and `check_is_fitted`.
182
+
183
+ ## Scope and limitations
184
+
185
+ mlkit puts clarity ahead of speed. k-NN uses brute-force distances, and
186
+ logistic regression uses plain full-batch gradient descent, so scale your
187
+ features first. It is not a replacement for scikit-learn on large datasets.
188
+
189
+ ## License
190
+
191
+ MIT
@@ -0,0 +1,15 @@
1
+ mlkit/__init__.py,sha256=PIitq7wzOvuCtYaW5QRLeynAylx95OmtPvPYsSutM9k,946
2
+ mlkit/base.py,sha256=RBQfQ9eN1gF4i5ulqfUq5HroRaXliJasnAQFWBHo7oY,5440
3
+ mlkit/cluster.py,sha256=0a3fmSAL7BdxmDoDyk7SSdAnSS8g7n-rISbZAntrgCU,4311
4
+ mlkit/datasets.py,sha256=iUktHqbv5RxaSTzMxYd1m29QHJsGiZF5Vd08FsdOZ20,2560
5
+ mlkit/linear_model.py,sha256=ws3qNCKif56Qm3otu-WP0zcoDrqjw-kGll7RrHKtEII,7142
6
+ mlkit/metrics.py,sha256=Goj7d7PjDbmVLkZJB4E8zN-lFUTZphNfcdqmMVr4MBk,5191
7
+ mlkit/model_selection.py,sha256=0XQ_XctTHMfLH7z_Uy13WyvMZwbA4NXy7m7lYk5IOlI,4243
8
+ mlkit/neighbors.py,sha256=U9SOTHZSrT184mFOOS1ldPR6AUaHScbHCWhZ6El5fFk,5155
9
+ mlkit/preprocessing.py,sha256=J_K3VgSfIwhIj_4tHYxuFSBPa2LlZqODn1z3KFq3XWM,4114
10
+ mlkit/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
11
+ mlkit-0.1.0.dist-info/licenses/LICENSE,sha256=pAfYREEW9GAy7cnK20OXjDn7ofJahYny9GCuIZQTDAA,1061
12
+ mlkit-0.1.0.dist-info/METADATA,sha256=yAhR-NbYVhkl539eXSuCS4cNIQoBT86b-hzgEK_OjvI,8318
13
+ mlkit-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
14
+ mlkit-0.1.0.dist-info/top_level.txt,sha256=bMLuHaaaZpWUcZAZQPEDE7tlGs8ZYUJ9zAsfZNhuoAs,6
15
+ mlkit-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 nehz
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ mlkit