mlkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mlkit/__init__.py +35 -0
- mlkit/base.py +165 -0
- mlkit/cluster.py +108 -0
- mlkit/datasets.py +73 -0
- mlkit/linear_model.py +191 -0
- mlkit/metrics.py +139 -0
- mlkit/model_selection.py +117 -0
- mlkit/neighbors.py +132 -0
- mlkit/preprocessing.py +111 -0
- mlkit/py.typed +0 -0
- mlkit-0.1.0.dist-info/METADATA +191 -0
- mlkit-0.1.0.dist-info/RECORD +15 -0
- mlkit-0.1.0.dist-info/WHEEL +5 -0
- mlkit-0.1.0.dist-info/licenses/LICENSE +21 -0
- mlkit-0.1.0.dist-info/top_level.txt +1 -0
mlkit/__init__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""mlkit - Machine Learning Toolkit.
|
|
2
|
+
|
|
3
|
+
A small, readable NumPy toolkit with a scikit-learn style ``fit``/``predict``
|
|
4
|
+
interface, classic models, preprocessing, model selection and metrics.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from . import datasets, metrics
|
|
8
|
+
from .base import BaseEstimator, NotFittedError, clone
|
|
9
|
+
from .cluster import KMeans
|
|
10
|
+
from .linear_model import LinearRegression, LogisticRegression, Ridge
|
|
11
|
+
from .model_selection import KFold, cross_val_score, train_test_split
|
|
12
|
+
from .neighbors import KNeighborsClassifier, KNeighborsRegressor
|
|
13
|
+
from .preprocessing import MinMaxScaler, StandardScaler
|
|
14
|
+
|
|
15
|
+
__version__ = "0.1.0"
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"BaseEstimator",
|
|
19
|
+
"NotFittedError",
|
|
20
|
+
"clone",
|
|
21
|
+
"LinearRegression",
|
|
22
|
+
"Ridge",
|
|
23
|
+
"LogisticRegression",
|
|
24
|
+
"KNeighborsClassifier",
|
|
25
|
+
"KNeighborsRegressor",
|
|
26
|
+
"KMeans",
|
|
27
|
+
"StandardScaler",
|
|
28
|
+
"MinMaxScaler",
|
|
29
|
+
"train_test_split",
|
|
30
|
+
"KFold",
|
|
31
|
+
"cross_val_score",
|
|
32
|
+
"datasets",
|
|
33
|
+
"metrics",
|
|
34
|
+
"__version__",
|
|
35
|
+
]
|
mlkit/base.py
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""Core estimator interface, mixins and input validation helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import inspect
|
|
6
|
+
from typing import Any, Optional, Tuple
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
from numpy.typing import ArrayLike
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"BaseEstimator",
|
|
13
|
+
"ClassifierMixin",
|
|
14
|
+
"RegressorMixin",
|
|
15
|
+
"TransformerMixin",
|
|
16
|
+
"ClusterMixin",
|
|
17
|
+
"NotFittedError",
|
|
18
|
+
"check_array",
|
|
19
|
+
"check_X_y",
|
|
20
|
+
"check_is_fitted",
|
|
21
|
+
"clone",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class NotFittedError(RuntimeError):
|
|
26
|
+
"""Raised when ``predict``/``transform`` is called before ``fit``."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def check_array(X: ArrayLike, *, ensure_2d: bool = True) -> np.ndarray:
|
|
30
|
+
"""Convert ``X`` to a finite float array.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
X: Array-like input.
|
|
34
|
+
ensure_2d: If True, a 2-D array of shape ``(n_samples, n_features)``
|
|
35
|
+
is required.
|
|
36
|
+
|
|
37
|
+
Returns:
|
|
38
|
+
A ``float64`` NumPy array.
|
|
39
|
+
|
|
40
|
+
Raises:
|
|
41
|
+
ValueError: If ``X`` is empty, has the wrong number of dimensions or
|
|
42
|
+
contains NaN/inf values.
|
|
43
|
+
"""
|
|
44
|
+
arr = np.asarray(X, dtype=float)
|
|
45
|
+
if ensure_2d and arr.ndim != 2:
|
|
46
|
+
raise ValueError(f"Expected a 2-D array, got an array with ndim={arr.ndim}.")
|
|
47
|
+
if arr.size == 0:
|
|
48
|
+
raise ValueError("Input array is empty.")
|
|
49
|
+
if not np.all(np.isfinite(arr)):
|
|
50
|
+
raise ValueError("Input contains NaN or infinity.")
|
|
51
|
+
return arr
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def check_X_y(X: ArrayLike, y: ArrayLike, *, y_numeric: bool = False) -> Tuple[np.ndarray, np.ndarray]:
|
|
55
|
+
"""Validate a feature matrix and a 1-D target vector of matching length.
|
|
56
|
+
|
|
57
|
+
Args:
|
|
58
|
+
X: Array-like of shape ``(n_samples, n_features)``.
|
|
59
|
+
y: Array-like of shape ``(n_samples,)``.
|
|
60
|
+
y_numeric: If True, ``y`` is converted to ``float64``.
|
|
61
|
+
|
|
62
|
+
Returns:
|
|
63
|
+
The validated ``(X, y)`` pair.
|
|
64
|
+
"""
|
|
65
|
+
X_arr = check_array(X)
|
|
66
|
+
y_arr = np.asarray(y, dtype=float if y_numeric else None)
|
|
67
|
+
if y_arr.ndim != 1:
|
|
68
|
+
raise ValueError(f"y must be 1-D, got ndim={y_arr.ndim}.")
|
|
69
|
+
if y_arr.shape[0] != X_arr.shape[0]:
|
|
70
|
+
raise ValueError(
|
|
71
|
+
f"X and y have inconsistent lengths: {X_arr.shape[0]} != {y_arr.shape[0]}."
|
|
72
|
+
)
|
|
73
|
+
if y_numeric and not np.all(np.isfinite(y_arr)):
|
|
74
|
+
raise ValueError("y contains NaN or infinity.")
|
|
75
|
+
return X_arr, y_arr
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def check_is_fitted(estimator: "BaseEstimator", attribute: str) -> None:
|
|
79
|
+
"""Raise :class:`NotFittedError` if ``estimator`` lacks ``attribute``."""
|
|
80
|
+
if getattr(estimator, attribute, None) is None:
|
|
81
|
+
raise NotFittedError(
|
|
82
|
+
f"This {type(estimator).__name__} instance is not fitted yet. "
|
|
83
|
+
"Call 'fit' with appropriate arguments first."
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class BaseEstimator:
|
|
88
|
+
"""Base class for all estimators.
|
|
89
|
+
|
|
90
|
+
Hyper-parameters are the keyword arguments of ``__init__``; they are stored
|
|
91
|
+
unchanged as attributes. Learned state uses a trailing underscore
|
|
92
|
+
(e.g. ``coef_``) and only exists after ``fit``.
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
@classmethod
|
|
96
|
+
def _param_names(cls) -> list:
|
|
97
|
+
sig = inspect.signature(cls.__init__)
|
|
98
|
+
return [
|
|
99
|
+
name
|
|
100
|
+
for name, p in sig.parameters.items()
|
|
101
|
+
if name != "self" and p.kind not in (p.VAR_POSITIONAL, p.VAR_KEYWORD)
|
|
102
|
+
]
|
|
103
|
+
|
|
104
|
+
def get_params(self) -> dict:
|
|
105
|
+
"""Return the estimator's hyper-parameters as a dict."""
|
|
106
|
+
return {name: getattr(self, name) for name in self._param_names()}
|
|
107
|
+
|
|
108
|
+
def set_params(self, **params: Any) -> "BaseEstimator":
|
|
109
|
+
"""Set hyper-parameters and return ``self``.
|
|
110
|
+
|
|
111
|
+
Raises:
|
|
112
|
+
ValueError: If an unknown parameter name is given.
|
|
113
|
+
"""
|
|
114
|
+
valid = set(self._param_names())
|
|
115
|
+
for key, value in params.items():
|
|
116
|
+
if key not in valid:
|
|
117
|
+
raise ValueError(f"Invalid parameter {key!r} for {type(self).__name__}.")
|
|
118
|
+
setattr(self, key, value)
|
|
119
|
+
return self
|
|
120
|
+
|
|
121
|
+
def __repr__(self) -> str:
|
|
122
|
+
args = ", ".join(f"{k}={v!r}" for k, v in self.get_params().items())
|
|
123
|
+
return f"{type(self).__name__}({args})"
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def clone(estimator: BaseEstimator) -> BaseEstimator:
|
|
127
|
+
"""Return a new, unfitted estimator with the same hyper-parameters."""
|
|
128
|
+
return type(estimator)(**estimator.get_params())
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class ClassifierMixin:
|
|
132
|
+
"""Adds :meth:`score` returning mean accuracy."""
|
|
133
|
+
|
|
134
|
+
def score(self, X: ArrayLike, y: ArrayLike) -> float:
|
|
135
|
+
"""Return the accuracy of ``self.predict(X)`` against ``y``."""
|
|
136
|
+
from .metrics import accuracy_score
|
|
137
|
+
|
|
138
|
+
return accuracy_score(y, self.predict(X)) # type: ignore[attr-defined]
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class RegressorMixin:
|
|
142
|
+
"""Adds :meth:`score` returning the coefficient of determination R^2."""
|
|
143
|
+
|
|
144
|
+
def score(self, X: ArrayLike, y: ArrayLike) -> float:
|
|
145
|
+
"""Return the R^2 of ``self.predict(X)`` against ``y``."""
|
|
146
|
+
from .metrics import r2_score
|
|
147
|
+
|
|
148
|
+
return r2_score(y, self.predict(X)) # type: ignore[attr-defined]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class TransformerMixin:
|
|
152
|
+
"""Adds :meth:`fit_transform`."""
|
|
153
|
+
|
|
154
|
+
def fit_transform(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> np.ndarray:
|
|
155
|
+
"""Fit to ``X`` and return the transformed data."""
|
|
156
|
+
return self.fit(X, y).transform(X) # type: ignore[attr-defined]
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class ClusterMixin:
|
|
160
|
+
"""Adds :meth:`fit_predict`."""
|
|
161
|
+
|
|
162
|
+
def fit_predict(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> np.ndarray:
|
|
163
|
+
"""Fit to ``X`` and return the cluster label of each sample."""
|
|
164
|
+
self.fit(X) # type: ignore[attr-defined]
|
|
165
|
+
return self.labels_ # type: ignore[attr-defined]
|
mlkit/cluster.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""k-means clustering with k-means++ initialisation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
|
|
10
|
+
from .base import BaseEstimator, ClusterMixin, check_array, check_is_fitted
|
|
11
|
+
from .neighbors import pairwise_distances
|
|
12
|
+
|
|
13
|
+
__all__ = ["KMeans"]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class KMeans(BaseEstimator, ClusterMixin):
|
|
17
|
+
"""Lloyd's k-means with k-means++ seeding and multiple restarts.
|
|
18
|
+
|
|
19
|
+
Args:
|
|
20
|
+
n_clusters: Number of clusters ``k``.
|
|
21
|
+
n_init: Number of independent runs; the run with the lowest inertia wins.
|
|
22
|
+
max_iter: Maximum Lloyd iterations per run.
|
|
23
|
+
tol: Stop a run when no centroid moves more than ``tol``.
|
|
24
|
+
random_state: Seed for reproducible initialisation.
|
|
25
|
+
|
|
26
|
+
Attributes:
|
|
27
|
+
cluster_centers_: Centroids of shape ``(n_clusters, n_features)``.
|
|
28
|
+
labels_: Cluster index of each training sample.
|
|
29
|
+
inertia_: Sum of squared distances of samples to their closest centroid.
|
|
30
|
+
n_iter_: Lloyd iterations used by the best run.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
def __init__(
|
|
34
|
+
self,
|
|
35
|
+
n_clusters: int = 8,
|
|
36
|
+
n_init: int = 10,
|
|
37
|
+
max_iter: int = 300,
|
|
38
|
+
tol: float = 1e-6,
|
|
39
|
+
random_state: Optional[int] = None,
|
|
40
|
+
) -> None:
|
|
41
|
+
self.n_clusters = n_clusters
|
|
42
|
+
self.n_init = n_init
|
|
43
|
+
self.max_iter = max_iter
|
|
44
|
+
self.tol = tol
|
|
45
|
+
self.random_state = random_state
|
|
46
|
+
self.cluster_centers_: np.ndarray = None # type: ignore[assignment]
|
|
47
|
+
self.labels_: np.ndarray = None # type: ignore[assignment]
|
|
48
|
+
self.inertia_: float = float("nan")
|
|
49
|
+
self.n_iter_: int = 0
|
|
50
|
+
|
|
51
|
+
@staticmethod
|
|
52
|
+
def _kmeans_pp(X: np.ndarray, k: int, rng: np.random.Generator) -> np.ndarray:
|
|
53
|
+
n = X.shape[0]
|
|
54
|
+
centers = np.empty((k, X.shape[1]))
|
|
55
|
+
centers[0] = X[rng.integers(n)]
|
|
56
|
+
d2 = ((X - centers[0]) ** 2).sum(axis=1)
|
|
57
|
+
for i in range(1, k):
|
|
58
|
+
total = d2.sum()
|
|
59
|
+
j = rng.integers(n) if total == 0 else rng.choice(n, p=d2 / total)
|
|
60
|
+
centers[i] = X[j]
|
|
61
|
+
d2 = np.minimum(d2, ((X - centers[i]) ** 2).sum(axis=1))
|
|
62
|
+
return centers
|
|
63
|
+
|
|
64
|
+
def _single_run(self, X: np.ndarray, rng: np.random.Generator):
|
|
65
|
+
centers = self._kmeans_pp(X, self.n_clusters, rng)
|
|
66
|
+
n_iter = 0
|
|
67
|
+
for n_iter in range(1, self.max_iter + 1):
|
|
68
|
+
labels = np.argmin(pairwise_distances(X, centers), axis=1)
|
|
69
|
+
new_centers = centers.copy()
|
|
70
|
+
for c in range(self.n_clusters):
|
|
71
|
+
members = X[labels == c]
|
|
72
|
+
if members.shape[0] > 0: # keep an empty cluster's centroid in place
|
|
73
|
+
new_centers[c] = members.mean(axis=0)
|
|
74
|
+
shift = np.abs(new_centers - centers).max()
|
|
75
|
+
centers = new_centers
|
|
76
|
+
if shift <= self.tol:
|
|
77
|
+
break
|
|
78
|
+
labels = np.argmin(pairwise_distances(X, centers), axis=1)
|
|
79
|
+
inertia = float(((X - centers[labels]) ** 2).sum())
|
|
80
|
+
return centers, labels, inertia, n_iter
|
|
81
|
+
|
|
82
|
+
def fit(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> "KMeans":
|
|
83
|
+
"""Cluster ``X`` of shape ``(n_samples, n_features)``. ``y`` is ignored."""
|
|
84
|
+
X = check_array(X)
|
|
85
|
+
if not 1 <= self.n_clusters <= X.shape[0]:
|
|
86
|
+
raise ValueError(
|
|
87
|
+
f"n_clusters must be in [1, n_samples={X.shape[0]}], got {self.n_clusters}."
|
|
88
|
+
)
|
|
89
|
+
if self.n_init < 1:
|
|
90
|
+
raise ValueError("n_init must be >= 1.")
|
|
91
|
+
rng = np.random.default_rng(self.random_state)
|
|
92
|
+
best = None
|
|
93
|
+
for _ in range(self.n_init):
|
|
94
|
+
run = self._single_run(X, rng)
|
|
95
|
+
if best is None or run[2] < best[2]:
|
|
96
|
+
best = run
|
|
97
|
+
self.cluster_centers_, self.labels_, self.inertia_, self.n_iter_ = best # type: ignore[misc]
|
|
98
|
+
return self
|
|
99
|
+
|
|
100
|
+
def predict(self, X: ArrayLike) -> np.ndarray:
|
|
101
|
+
"""Assign each sample in ``X`` to its nearest centroid."""
|
|
102
|
+
check_is_fitted(self, "cluster_centers_")
|
|
103
|
+
return np.argmin(pairwise_distances(X, self.cluster_centers_), axis=1)
|
|
104
|
+
|
|
105
|
+
def transform(self, X: ArrayLike) -> np.ndarray:
|
|
106
|
+
"""Return distances to each centroid, shape ``(n_samples, n_clusters)``."""
|
|
107
|
+
check_is_fitted(self, "cluster_centers_")
|
|
108
|
+
return pairwise_distances(X, self.cluster_centers_)
|
mlkit/datasets.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Synthetic dataset generators for examples and tests."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Optional, Tuple, Union
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
|
|
10
|
+
__all__ = ["make_blobs", "make_regression"]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def make_blobs(
|
|
14
|
+
n_samples: int = 100,
|
|
15
|
+
n_features: int = 2,
|
|
16
|
+
centers: Union[int, ArrayLike] = 3,
|
|
17
|
+
cluster_std: float = 1.0,
|
|
18
|
+
random_state: Optional[int] = None,
|
|
19
|
+
) -> Tuple[np.ndarray, np.ndarray]:
|
|
20
|
+
"""Generate isotropic Gaussian blobs for clustering/classification.
|
|
21
|
+
|
|
22
|
+
Args:
|
|
23
|
+
n_samples: Total number of points, split as evenly as possible.
|
|
24
|
+
n_features: Dimensionality (ignored if ``centers`` is an array).
|
|
25
|
+
centers: Number of centres (drawn uniformly in ``[-10, 10]``) or an
|
|
26
|
+
explicit array of shape ``(n_centers, n_features)``.
|
|
27
|
+
cluster_std: Standard deviation of each blob.
|
|
28
|
+
random_state: Seed for reproducibility.
|
|
29
|
+
|
|
30
|
+
Returns:
|
|
31
|
+
``(X, y)`` with ``X`` of shape ``(n_samples, n_features)`` and integer
|
|
32
|
+
labels ``y`` of shape ``(n_samples,)``.
|
|
33
|
+
"""
|
|
34
|
+
rng = np.random.default_rng(random_state)
|
|
35
|
+
if isinstance(centers, (int, np.integer)):
|
|
36
|
+
C = rng.uniform(-10.0, 10.0, size=(int(centers), n_features))
|
|
37
|
+
else:
|
|
38
|
+
C = np.asarray(centers, dtype=float)
|
|
39
|
+
if C.ndim != 2:
|
|
40
|
+
raise ValueError("centers must be an int or a 2-D array.")
|
|
41
|
+
k = C.shape[0]
|
|
42
|
+
counts = np.full(k, n_samples // k)
|
|
43
|
+
counts[: n_samples % k] += 1
|
|
44
|
+
y = np.repeat(np.arange(k), counts)
|
|
45
|
+
X = C[y] + rng.normal(scale=cluster_std, size=(n_samples, C.shape[1]))
|
|
46
|
+
perm = rng.permutation(n_samples)
|
|
47
|
+
return X[perm], y[perm]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def make_regression(
|
|
51
|
+
n_samples: int = 100,
|
|
52
|
+
n_features: int = 3,
|
|
53
|
+
noise: float = 0.0,
|
|
54
|
+
bias: float = 0.0,
|
|
55
|
+
random_state: Optional[int] = None,
|
|
56
|
+
) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
|
|
57
|
+
"""Generate a random linear regression problem ``y = X @ coef + bias + noise``.
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
n_samples: Number of samples.
|
|
61
|
+
n_features: Number of features.
|
|
62
|
+
noise: Standard deviation of Gaussian noise added to ``y``.
|
|
63
|
+
bias: Intercept of the underlying model.
|
|
64
|
+
random_state: Seed for reproducibility.
|
|
65
|
+
|
|
66
|
+
Returns:
|
|
67
|
+
``(X, y, coef)`` where ``coef`` holds the true weights.
|
|
68
|
+
"""
|
|
69
|
+
rng = np.random.default_rng(random_state)
|
|
70
|
+
X = rng.normal(size=(n_samples, n_features))
|
|
71
|
+
coef = rng.uniform(-5.0, 5.0, size=n_features)
|
|
72
|
+
y = X @ coef + bias + rng.normal(scale=noise, size=n_samples) if noise > 0 else X @ coef + bias
|
|
73
|
+
return X, y, coef
|
mlkit/linear_model.py
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""Linear models: ordinary least squares, ridge and logistic regression."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Tuple
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
|
|
10
|
+
from .base import BaseEstimator, ClassifierMixin, RegressorMixin, check_array, check_is_fitted, check_X_y
|
|
11
|
+
|
|
12
|
+
__all__ = ["LinearRegression", "Ridge", "LogisticRegression"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _center(X: np.ndarray, y: np.ndarray, fit_intercept: bool) -> Tuple[np.ndarray, np.ndarray, np.ndarray, float]:
|
|
16
|
+
if fit_intercept:
|
|
17
|
+
X_mean = X.mean(axis=0)
|
|
18
|
+
y_mean = float(y.mean())
|
|
19
|
+
return X - X_mean, y - y_mean, X_mean, y_mean
|
|
20
|
+
return X, y, np.zeros(X.shape[1]), 0.0
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class _LinearRegressorBase(BaseEstimator, RegressorMixin):
|
|
24
|
+
coef_: np.ndarray
|
|
25
|
+
intercept_: float
|
|
26
|
+
|
|
27
|
+
def predict(self, X: ArrayLike) -> np.ndarray:
|
|
28
|
+
"""Predict targets for ``X`` of shape ``(n_samples, n_features)``."""
|
|
29
|
+
check_is_fitted(self, "coef_")
|
|
30
|
+
X = check_array(X)
|
|
31
|
+
if X.shape[1] != self.coef_.shape[0]:
|
|
32
|
+
raise ValueError(f"X has {X.shape[1]} features, expected {self.coef_.shape[0]}.")
|
|
33
|
+
return X @ self.coef_ + self.intercept_
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class LinearRegression(_LinearRegressorBase):
|
|
37
|
+
"""Ordinary least squares linear regression.
|
|
38
|
+
|
|
39
|
+
Solves ``min ||y - X w - b||^2`` with :func:`numpy.linalg.lstsq`, which is
|
|
40
|
+
robust to rank-deficient ``X``.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
fit_intercept: Whether to learn an intercept ``b``.
|
|
44
|
+
|
|
45
|
+
Attributes:
|
|
46
|
+
coef_: Weights of shape ``(n_features,)``.
|
|
47
|
+
intercept_: Bias term (``0.0`` if ``fit_intercept=False``).
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(self, fit_intercept: bool = True) -> None:
|
|
51
|
+
self.fit_intercept = fit_intercept
|
|
52
|
+
self.coef_ = None # type: ignore[assignment]
|
|
53
|
+
self.intercept_ = 0.0
|
|
54
|
+
|
|
55
|
+
def fit(self, X: ArrayLike, y: ArrayLike) -> "LinearRegression":
|
|
56
|
+
"""Fit the model to ``X`` of shape ``(n_samples, n_features)`` and ``y``."""
|
|
57
|
+
X, y = check_X_y(X, y, y_numeric=True)
|
|
58
|
+
Xc, yc, X_mean, y_mean = _center(X, y, self.fit_intercept)
|
|
59
|
+
coef, *_ = np.linalg.lstsq(Xc, yc, rcond=None)
|
|
60
|
+
self.coef_ = coef
|
|
61
|
+
self.intercept_ = float(y_mean - X_mean @ coef)
|
|
62
|
+
return self
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class Ridge(_LinearRegressorBase):
|
|
66
|
+
"""Linear regression with L2 regularisation (closed form).
|
|
67
|
+
|
|
68
|
+
Minimises ``||y - X w - b||^2 + alpha * ||w||^2``. The intercept is not
|
|
69
|
+
penalised.
|
|
70
|
+
|
|
71
|
+
Args:
|
|
72
|
+
alpha: Non-negative regularisation strength.
|
|
73
|
+
fit_intercept: Whether to learn an intercept ``b``.
|
|
74
|
+
|
|
75
|
+
Attributes:
|
|
76
|
+
coef_: Weights of shape ``(n_features,)``.
|
|
77
|
+
intercept_: Bias term.
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
def __init__(self, alpha: float = 1.0, fit_intercept: bool = True) -> None:
|
|
81
|
+
self.alpha = alpha
|
|
82
|
+
self.fit_intercept = fit_intercept
|
|
83
|
+
self.coef_ = None # type: ignore[assignment]
|
|
84
|
+
self.intercept_ = 0.0
|
|
85
|
+
|
|
86
|
+
def fit(self, X: ArrayLike, y: ArrayLike) -> "Ridge":
|
|
87
|
+
"""Fit the model to ``X`` of shape ``(n_samples, n_features)`` and ``y``."""
|
|
88
|
+
if self.alpha < 0:
|
|
89
|
+
raise ValueError("alpha must be non-negative.")
|
|
90
|
+
X, y = check_X_y(X, y, y_numeric=True)
|
|
91
|
+
Xc, yc, X_mean, y_mean = _center(X, y, self.fit_intercept)
|
|
92
|
+
n_features = X.shape[1]
|
|
93
|
+
A = Xc.T @ Xc + self.alpha * np.eye(n_features)
|
|
94
|
+
coef = np.linalg.lstsq(A, Xc.T @ yc, rcond=None)[0]
|
|
95
|
+
self.coef_ = coef
|
|
96
|
+
self.intercept_ = float(y_mean - X_mean @ coef)
|
|
97
|
+
return self
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _softmax(z: np.ndarray) -> np.ndarray:
|
|
101
|
+
z = z - z.max(axis=1, keepdims=True)
|
|
102
|
+
e = np.exp(z)
|
|
103
|
+
return e / e.sum(axis=1, keepdims=True)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class LogisticRegression(BaseEstimator, ClassifierMixin):
|
|
107
|
+
"""Multinomial logistic regression trained by full-batch gradient descent.
|
|
108
|
+
|
|
109
|
+
Binary problems are handled as the two-class case of the softmax model.
|
|
110
|
+
Features should be on comparable scales (see
|
|
111
|
+
:class:`mlkit.preprocessing.StandardScaler`) for fast convergence.
|
|
112
|
+
|
|
113
|
+
Args:
|
|
114
|
+
learning_rate: Gradient descent step size.
|
|
115
|
+
max_iter: Maximum number of gradient steps.
|
|
116
|
+
tol: Stop when the largest absolute gradient entry falls below ``tol``.
|
|
117
|
+
alpha: L2 penalty on the weights (not the intercepts).
|
|
118
|
+
fit_intercept: Whether to learn per-class intercepts.
|
|
119
|
+
|
|
120
|
+
Attributes:
|
|
121
|
+
classes_: Sorted unique class labels, shape ``(n_classes,)``.
|
|
122
|
+
coef_: Weights of shape ``(n_classes, n_features)``.
|
|
123
|
+
intercept_: Intercepts of shape ``(n_classes,)``.
|
|
124
|
+
n_iter_: Number of gradient steps actually taken.
|
|
125
|
+
"""
|
|
126
|
+
|
|
127
|
+
def __init__(
|
|
128
|
+
self,
|
|
129
|
+
learning_rate: float = 0.1,
|
|
130
|
+
max_iter: int = 1000,
|
|
131
|
+
tol: float = 1e-6,
|
|
132
|
+
alpha: float = 0.0,
|
|
133
|
+
fit_intercept: bool = True,
|
|
134
|
+
) -> None:
|
|
135
|
+
self.learning_rate = learning_rate
|
|
136
|
+
self.max_iter = max_iter
|
|
137
|
+
self.tol = tol
|
|
138
|
+
self.alpha = alpha
|
|
139
|
+
self.fit_intercept = fit_intercept
|
|
140
|
+
self.classes_: np.ndarray = None # type: ignore[assignment]
|
|
141
|
+
self.coef_: np.ndarray = None # type: ignore[assignment]
|
|
142
|
+
self.intercept_: np.ndarray = None # type: ignore[assignment]
|
|
143
|
+
self.n_iter_: int = 0
|
|
144
|
+
|
|
145
|
+
def fit(self, X: ArrayLike, y: ArrayLike) -> "LogisticRegression":
|
|
146
|
+
"""Fit the model to ``X`` of shape ``(n_samples, n_features)`` and labels ``y``."""
|
|
147
|
+
X, y = check_X_y(X, y)
|
|
148
|
+
classes, y_idx = np.unique(y, return_inverse=True)
|
|
149
|
+
if classes.shape[0] < 2:
|
|
150
|
+
raise ValueError("LogisticRegression needs at least two classes in y.")
|
|
151
|
+
n_samples, n_features = X.shape
|
|
152
|
+
n_classes = classes.shape[0]
|
|
153
|
+
Y = np.eye(n_classes)[y_idx]
|
|
154
|
+
|
|
155
|
+
W = np.zeros((n_classes, n_features))
|
|
156
|
+
b = np.zeros(n_classes)
|
|
157
|
+
n_iter = 0
|
|
158
|
+
for n_iter in range(1, self.max_iter + 1):
|
|
159
|
+
P = _softmax(X @ W.T + b)
|
|
160
|
+
diff = (P - Y) / n_samples
|
|
161
|
+
grad_W = diff.T @ X + self.alpha * W
|
|
162
|
+
grad_b = diff.sum(axis=0) if self.fit_intercept else np.zeros(n_classes)
|
|
163
|
+
W -= self.learning_rate * grad_W
|
|
164
|
+
b -= self.learning_rate * grad_b
|
|
165
|
+
if max(np.abs(grad_W).max(), np.abs(grad_b).max()) < self.tol:
|
|
166
|
+
break
|
|
167
|
+
|
|
168
|
+
self.classes_ = classes
|
|
169
|
+
self.coef_ = W
|
|
170
|
+
self.intercept_ = b
|
|
171
|
+
self.n_iter_ = n_iter
|
|
172
|
+
return self
|
|
173
|
+
|
|
174
|
+
def decision_function(self, X: ArrayLike) -> np.ndarray:
|
|
175
|
+
"""Return per-class linear scores of shape ``(n_samples, n_classes)``."""
|
|
176
|
+
check_is_fitted(self, "coef_")
|
|
177
|
+
X = check_array(X)
|
|
178
|
+
if X.shape[1] != self.coef_.shape[1]:
|
|
179
|
+
raise ValueError(f"X has {X.shape[1]} features, expected {self.coef_.shape[1]}.")
|
|
180
|
+
return X @ self.coef_.T + self.intercept_
|
|
181
|
+
|
|
182
|
+
def predict_proba(self, X: ArrayLike) -> np.ndarray:
|
|
183
|
+
"""Return class probabilities of shape ``(n_samples, n_classes)``.
|
|
184
|
+
|
|
185
|
+
Columns follow the order of ``classes_``.
|
|
186
|
+
"""
|
|
187
|
+
return _softmax(self.decision_function(X))
|
|
188
|
+
|
|
189
|
+
def predict(self, X: ArrayLike) -> np.ndarray:
|
|
190
|
+
"""Return the most probable class label for each sample."""
|
|
191
|
+
return self.classes_[np.argmax(self.decision_function(X), axis=1)]
|
mlkit/metrics.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""Evaluation metrics for regression and classification."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Optional, Sequence, Tuple
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"accuracy_score",
|
|
12
|
+
"confusion_matrix",
|
|
13
|
+
"precision_score",
|
|
14
|
+
"recall_score",
|
|
15
|
+
"f1_score",
|
|
16
|
+
"mean_squared_error",
|
|
17
|
+
"mean_absolute_error",
|
|
18
|
+
"r2_score",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
_AVERAGES = ("binary", "macro")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _check_pair(y_true: ArrayLike, y_pred: ArrayLike, numeric: bool = False) -> Tuple[np.ndarray, np.ndarray]:
|
|
25
|
+
dtype = float if numeric else None
|
|
26
|
+
t = np.asarray(y_true, dtype=dtype)
|
|
27
|
+
p = np.asarray(y_pred, dtype=dtype)
|
|
28
|
+
if t.ndim != 1 or p.ndim != 1:
|
|
29
|
+
raise ValueError("y_true and y_pred must be 1-D.")
|
|
30
|
+
if t.shape[0] != p.shape[0]:
|
|
31
|
+
raise ValueError(f"Inconsistent lengths: {t.shape[0]} != {p.shape[0]}.")
|
|
32
|
+
if t.shape[0] == 0:
|
|
33
|
+
raise ValueError("Empty input.")
|
|
34
|
+
return t, p
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
# --------------------------------------------------------------------------- #
|
|
38
|
+
# Regression
|
|
39
|
+
# --------------------------------------------------------------------------- #
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def mean_squared_error(y_true: ArrayLike, y_pred: ArrayLike) -> float:
|
|
43
|
+
"""Mean of squared residuals."""
|
|
44
|
+
t, p = _check_pair(y_true, y_pred, numeric=True)
|
|
45
|
+
return float(np.mean((t - p) ** 2))
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def mean_absolute_error(y_true: ArrayLike, y_pred: ArrayLike) -> float:
|
|
49
|
+
"""Mean of absolute residuals."""
|
|
50
|
+
t, p = _check_pair(y_true, y_pred, numeric=True)
|
|
51
|
+
return float(np.mean(np.abs(t - p)))
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def r2_score(y_true: ArrayLike, y_pred: ArrayLike) -> float:
|
|
55
|
+
"""Coefficient of determination ``1 - SS_res / SS_tot``.
|
|
56
|
+
|
|
57
|
+
Returns 1.0 for a perfect fit of constant ``y_true`` and 0.0 for an
|
|
58
|
+
imperfect one (where ``SS_tot`` is zero).
|
|
59
|
+
"""
|
|
60
|
+
t, p = _check_pair(y_true, y_pred, numeric=True)
|
|
61
|
+
ss_res = float(np.sum((t - p) ** 2))
|
|
62
|
+
ss_tot = float(np.sum((t - t.mean()) ** 2))
|
|
63
|
+
if ss_tot == 0.0:
|
|
64
|
+
return 1.0 if ss_res == 0.0 else 0.0
|
|
65
|
+
return 1.0 - ss_res / ss_tot
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# --------------------------------------------------------------------------- #
|
|
69
|
+
# Classification
|
|
70
|
+
# --------------------------------------------------------------------------- #
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def accuracy_score(y_true: ArrayLike, y_pred: ArrayLike) -> float:
|
|
74
|
+
"""Fraction of exactly matching labels."""
|
|
75
|
+
t, p = _check_pair(y_true, y_pred)
|
|
76
|
+
return float(np.mean(t == p))
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def confusion_matrix(y_true: ArrayLike, y_pred: ArrayLike, labels: Optional[Sequence[Any]] = None) -> np.ndarray:
|
|
80
|
+
"""Count matrix ``C`` where ``C[i, j]`` is the number of samples of true
|
|
81
|
+
class ``labels[i]`` predicted as ``labels[j]``.
|
|
82
|
+
|
|
83
|
+
Args:
|
|
84
|
+
y_true: True labels.
|
|
85
|
+
y_pred: Predicted labels.
|
|
86
|
+
labels: Label order; defaults to the sorted union of both inputs.
|
|
87
|
+
Samples whose labels are not listed are ignored.
|
|
88
|
+
"""
|
|
89
|
+
t, p = _check_pair(y_true, y_pred)
|
|
90
|
+
lab = np.unique(np.concatenate([t, p])) if labels is None else np.asarray(labels)
|
|
91
|
+
index = {v: i for i, v in enumerate(lab.tolist())}
|
|
92
|
+
cm = np.zeros((len(lab), len(lab)), dtype=int)
|
|
93
|
+
for a, b in zip(t.tolist(), p.tolist()):
|
|
94
|
+
if a in index and b in index:
|
|
95
|
+
cm[index[a], index[b]] += 1
|
|
96
|
+
return cm
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _per_class_counts(t: np.ndarray, p: np.ndarray, average: str, pos_label: Any):
|
|
100
|
+
if average not in _AVERAGES:
|
|
101
|
+
raise ValueError(f"average must be one of {_AVERAGES}, got {average!r}.")
|
|
102
|
+
labels = [pos_label] if average == "binary" else np.unique(np.concatenate([t, p])).tolist()
|
|
103
|
+
tp = np.array([np.sum((p == c) & (t == c)) for c in labels], dtype=float)
|
|
104
|
+
fp = np.array([np.sum((p == c) & (t != c)) for c in labels], dtype=float)
|
|
105
|
+
fn = np.array([np.sum((p != c) & (t == c)) for c in labels], dtype=float)
|
|
106
|
+
return tp, fp, fn
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _safe_div(num: np.ndarray, den: np.ndarray) -> np.ndarray:
|
|
110
|
+
return np.divide(num, den, out=np.zeros_like(num), where=den != 0)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def precision_score(y_true: ArrayLike, y_pred: ArrayLike, *, average: str = "binary", pos_label: Any = 1) -> float:
|
|
114
|
+
"""Precision ``tp / (tp + fp)``.
|
|
115
|
+
|
|
116
|
+
Args:
|
|
117
|
+
average: ``"binary"`` scores only ``pos_label``; ``"macro"`` averages
|
|
118
|
+
the per-class scores over all labels present.
|
|
119
|
+
pos_label: Positive class used when ``average="binary"``.
|
|
120
|
+
|
|
121
|
+
Undefined ratios (zero denominator) count as 0.0.
|
|
122
|
+
"""
|
|
123
|
+
t, p = _check_pair(y_true, y_pred)
|
|
124
|
+
tp, fp, _ = _per_class_counts(t, p, average, pos_label)
|
|
125
|
+
return float(np.mean(_safe_div(tp, tp + fp)))
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def recall_score(y_true: ArrayLike, y_pred: ArrayLike, *, average: str = "binary", pos_label: Any = 1) -> float:
|
|
129
|
+
"""Recall ``tp / (tp + fn)``. See :func:`precision_score` for arguments."""
|
|
130
|
+
t, p = _check_pair(y_true, y_pred)
|
|
131
|
+
tp, _, fn = _per_class_counts(t, p, average, pos_label)
|
|
132
|
+
return float(np.mean(_safe_div(tp, tp + fn)))
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def f1_score(y_true: ArrayLike, y_pred: ArrayLike, *, average: str = "binary", pos_label: Any = 1) -> float:
|
|
136
|
+
"""F1 ``2 tp / (2 tp + fp + fn)``. See :func:`precision_score` for arguments."""
|
|
137
|
+
t, p = _check_pair(y_true, y_pred)
|
|
138
|
+
tp, fp, fn = _per_class_counts(t, p, average, pos_label)
|
|
139
|
+
return float(np.mean(_safe_div(2 * tp, 2 * tp + fp + fn)))
|
mlkit/model_selection.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""Data splitting and cross-validation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Callable, Iterator, List, Optional, Tuple, Union
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
|
|
10
|
+
from .base import BaseEstimator, clone
|
|
11
|
+
|
|
12
|
+
__all__ = ["train_test_split", "KFold", "cross_val_score"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def train_test_split(
|
|
16
|
+
*arrays: ArrayLike,
|
|
17
|
+
test_size: Union[float, int] = 0.25,
|
|
18
|
+
shuffle: bool = True,
|
|
19
|
+
random_state: Optional[int] = None,
|
|
20
|
+
) -> List[np.ndarray]:
|
|
21
|
+
"""Split arrays into random train and test subsets.
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
*arrays: One or more arrays with the same first dimension.
|
|
25
|
+
test_size: Fraction in ``(0, 1)`` or absolute number of test samples.
|
|
26
|
+
shuffle: Shuffle before splitting.
|
|
27
|
+
random_state: Seed for reproducible shuffling.
|
|
28
|
+
|
|
29
|
+
Returns:
|
|
30
|
+
``[a_train, a_test, b_train, b_test, ...]`` for each input array.
|
|
31
|
+
|
|
32
|
+
Example:
|
|
33
|
+
>>> X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
|
34
|
+
"""
|
|
35
|
+
if not arrays:
|
|
36
|
+
raise ValueError("At least one array is required.")
|
|
37
|
+
arrs = [np.asarray(a) for a in arrays]
|
|
38
|
+
n = arrs[0].shape[0]
|
|
39
|
+
if any(a.shape[0] != n for a in arrs):
|
|
40
|
+
raise ValueError("All arrays must have the same number of samples.")
|
|
41
|
+
|
|
42
|
+
if isinstance(test_size, float):
|
|
43
|
+
if not 0.0 < test_size < 1.0:
|
|
44
|
+
raise ValueError("A float test_size must be in (0, 1).")
|
|
45
|
+
n_test = int(np.ceil(test_size * n))
|
|
46
|
+
else:
|
|
47
|
+
n_test = int(test_size)
|
|
48
|
+
if not 0 < n_test < n:
|
|
49
|
+
raise ValueError(f"test_size={test_size} leaves an empty train or test set for n={n}.")
|
|
50
|
+
|
|
51
|
+
idx = np.random.default_rng(random_state).permutation(n) if shuffle else np.arange(n)
|
|
52
|
+
test_idx, train_idx = idx[:n_test], idx[n_test:]
|
|
53
|
+
out: List[np.ndarray] = []
|
|
54
|
+
for a in arrs:
|
|
55
|
+
out.extend([a[train_idx], a[test_idx]])
|
|
56
|
+
return out
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class KFold:
|
|
60
|
+
"""K-fold cross-validation splitter.
|
|
61
|
+
|
|
62
|
+
Args:
|
|
63
|
+
n_splits: Number of folds (>= 2).
|
|
64
|
+
shuffle: Shuffle sample indices before folding.
|
|
65
|
+
random_state: Seed used when ``shuffle=True``.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
def __init__(self, n_splits: int = 5, shuffle: bool = False, random_state: Optional[int] = None) -> None:
|
|
69
|
+
if n_splits < 2:
|
|
70
|
+
raise ValueError("n_splits must be >= 2.")
|
|
71
|
+
self.n_splits = n_splits
|
|
72
|
+
self.shuffle = shuffle
|
|
73
|
+
self.random_state = random_state
|
|
74
|
+
|
|
75
|
+
def split(self, X: ArrayLike) -> Iterator[Tuple[np.ndarray, np.ndarray]]:
|
|
76
|
+
"""Yield ``(train_indices, test_indices)`` for each fold."""
|
|
77
|
+
n = np.asarray(X).shape[0]
|
|
78
|
+
if self.n_splits > n:
|
|
79
|
+
raise ValueError(f"Cannot make {self.n_splits} folds from {n} samples.")
|
|
80
|
+
idx = np.random.default_rng(self.random_state).permutation(n) if self.shuffle else np.arange(n)
|
|
81
|
+
for test_idx in np.array_split(idx, self.n_splits):
|
|
82
|
+
yield np.setdiff1d(idx, test_idx, assume_unique=True), test_idx
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def cross_val_score(
|
|
86
|
+
estimator: BaseEstimator,
|
|
87
|
+
X: ArrayLike,
|
|
88
|
+
y: ArrayLike,
|
|
89
|
+
*,
|
|
90
|
+
cv: Union[int, KFold] = 5,
|
|
91
|
+
scoring: Optional[Callable[[np.ndarray, np.ndarray], float]] = None,
|
|
92
|
+
) -> np.ndarray:
|
|
93
|
+
"""Evaluate ``estimator`` with K-fold cross-validation.
|
|
94
|
+
|
|
95
|
+
A fresh :func:`~mlkit.base.clone` of ``estimator`` is fitted on each fold.
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
estimator: An unfitted estimator with ``fit``/``predict``.
|
|
99
|
+
X: Features, shape ``(n_samples, n_features)``.
|
|
100
|
+
y: Targets, shape ``(n_samples,)``.
|
|
101
|
+
cv: Number of folds or a :class:`KFold` instance.
|
|
102
|
+
scoring: ``metric(y_true, y_pred) -> float``; defaults to the
|
|
103
|
+
estimator's own ``score`` method.
|
|
104
|
+
|
|
105
|
+
Returns:
|
|
106
|
+
Array of per-fold scores, shape ``(n_splits,)``.
|
|
107
|
+
"""
|
|
108
|
+
X_arr, y_arr = np.asarray(X), np.asarray(y)
|
|
109
|
+
splitter = KFold(n_splits=cv) if isinstance(cv, int) else cv
|
|
110
|
+
scores = []
|
|
111
|
+
for train_idx, test_idx in splitter.split(X_arr):
|
|
112
|
+
model = clone(estimator).fit(X_arr[train_idx], y_arr[train_idx]) # type: ignore[attr-defined]
|
|
113
|
+
if scoring is None:
|
|
114
|
+
scores.append(model.score(X_arr[test_idx], y_arr[test_idx]))
|
|
115
|
+
else:
|
|
116
|
+
scores.append(scoring(y_arr[test_idx], model.predict(X_arr[test_idx])))
|
|
117
|
+
return np.asarray(scores, dtype=float)
|
mlkit/neighbors.py
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""k-nearest-neighbour classification and regression (brute-force Euclidean)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Tuple
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
|
|
10
|
+
from .base import BaseEstimator, ClassifierMixin, RegressorMixin, check_array, check_is_fitted, check_X_y
|
|
11
|
+
|
|
12
|
+
__all__ = ["KNeighborsClassifier", "KNeighborsRegressor", "pairwise_distances"]
|
|
13
|
+
|
|
14
|
+
_WEIGHTS = ("uniform", "distance")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def pairwise_distances(A: ArrayLike, B: ArrayLike) -> np.ndarray:
|
|
18
|
+
"""Euclidean distances between rows of ``A`` and rows of ``B``.
|
|
19
|
+
|
|
20
|
+
Args:
|
|
21
|
+
A: Array of shape ``(n_a, n_features)``.
|
|
22
|
+
B: Array of shape ``(n_b, n_features)``.
|
|
23
|
+
|
|
24
|
+
Returns:
|
|
25
|
+
Array of shape ``(n_a, n_b)``.
|
|
26
|
+
"""
|
|
27
|
+
A = check_array(A)
|
|
28
|
+
B = check_array(B)
|
|
29
|
+
if A.shape[1] != B.shape[1]:
|
|
30
|
+
raise ValueError(f"Feature mismatch: {A.shape[1]} != {B.shape[1]}.")
|
|
31
|
+
sq = (A * A).sum(axis=1)[:, None] - 2.0 * A @ B.T + (B * B).sum(axis=1)[None, :]
|
|
32
|
+
return np.sqrt(np.maximum(sq, 0.0))
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class _KNeighborsBase(BaseEstimator):
|
|
36
|
+
def __init__(self, n_neighbors: int = 5, weights: str = "uniform") -> None:
|
|
37
|
+
self.n_neighbors = n_neighbors
|
|
38
|
+
self.weights = weights
|
|
39
|
+
self._fit_X: np.ndarray = None # type: ignore[assignment]
|
|
40
|
+
self._fit_y: np.ndarray = None # type: ignore[assignment]
|
|
41
|
+
|
|
42
|
+
def _validate_params(self, n_samples: int) -> None:
|
|
43
|
+
if self.weights not in _WEIGHTS:
|
|
44
|
+
raise ValueError(f"weights must be one of {_WEIGHTS}, got {self.weights!r}.")
|
|
45
|
+
if not 1 <= self.n_neighbors <= n_samples:
|
|
46
|
+
raise ValueError(
|
|
47
|
+
f"n_neighbors must be in [1, n_samples={n_samples}], got {self.n_neighbors}."
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
def kneighbors(self, X: ArrayLike) -> Tuple[np.ndarray, np.ndarray]:
|
|
51
|
+
"""Find the ``n_neighbors`` nearest training samples for each row of ``X``.
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
``(distances, indices)``, each of shape ``(n_samples, n_neighbors)``,
|
|
55
|
+
sorted by increasing distance.
|
|
56
|
+
"""
|
|
57
|
+
check_is_fitted(self, "_fit_X")
|
|
58
|
+
D = pairwise_distances(X, self._fit_X)
|
|
59
|
+
idx = np.argsort(D, axis=1, kind="stable")[:, : self.n_neighbors]
|
|
60
|
+
return np.take_along_axis(D, idx, axis=1), idx
|
|
61
|
+
|
|
62
|
+
def _neighbor_weights(self, dist: np.ndarray) -> np.ndarray:
|
|
63
|
+
if self.weights == "uniform":
|
|
64
|
+
return np.ones_like(dist)
|
|
65
|
+
# Exact matches dominate: give them weight 1 and everything else 0.
|
|
66
|
+
with np.errstate(divide="ignore"):
|
|
67
|
+
w = 1.0 / dist
|
|
68
|
+
exact = np.isinf(w)
|
|
69
|
+
rows = exact.any(axis=1)
|
|
70
|
+
w[rows] = exact[rows].astype(float)
|
|
71
|
+
return w
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class KNeighborsClassifier(_KNeighborsBase, ClassifierMixin):
|
|
75
|
+
"""Classifier voting among the ``k`` nearest training samples.
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
n_neighbors: Number of neighbours ``k``.
|
|
79
|
+
weights: ``"uniform"`` (each neighbour counts equally) or
|
|
80
|
+
``"distance"`` (votes weighted by inverse distance).
|
|
81
|
+
|
|
82
|
+
Attributes:
|
|
83
|
+
classes_: Sorted unique class labels.
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
def __init__(self, n_neighbors: int = 5, weights: str = "uniform") -> None:
|
|
87
|
+
super().__init__(n_neighbors=n_neighbors, weights=weights)
|
|
88
|
+
self.classes_: np.ndarray = None # type: ignore[assignment]
|
|
89
|
+
|
|
90
|
+
def fit(self, X: ArrayLike, y: ArrayLike) -> "KNeighborsClassifier":
|
|
91
|
+
"""Store the training data ``X`` and labels ``y``."""
|
|
92
|
+
X, y = check_X_y(X, y)
|
|
93
|
+
self._validate_params(X.shape[0])
|
|
94
|
+
self.classes_, self._fit_y = np.unique(y, return_inverse=True)
|
|
95
|
+
self._fit_X = X
|
|
96
|
+
return self
|
|
97
|
+
|
|
98
|
+
def predict_proba(self, X: ArrayLike) -> np.ndarray:
|
|
99
|
+
"""Return vote shares of shape ``(n_samples, n_classes)`` (columns follow ``classes_``)."""
|
|
100
|
+
dist, idx = self.kneighbors(X)
|
|
101
|
+
w = self._neighbor_weights(dist)
|
|
102
|
+
labels = self._fit_y[idx]
|
|
103
|
+
proba = np.zeros((idx.shape[0], self.classes_.shape[0]))
|
|
104
|
+
for c in range(self.classes_.shape[0]):
|
|
105
|
+
proba[:, c] = (w * (labels == c)).sum(axis=1)
|
|
106
|
+
return proba / proba.sum(axis=1, keepdims=True)
|
|
107
|
+
|
|
108
|
+
def predict(self, X: ArrayLike) -> np.ndarray:
|
|
109
|
+
"""Return the majority-vote label for each sample (ties go to the smaller class)."""
|
|
110
|
+
return self.classes_[np.argmax(self.predict_proba(X), axis=1)]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class KNeighborsRegressor(_KNeighborsBase, RegressorMixin):
|
|
114
|
+
"""Regressor averaging the targets of the ``k`` nearest training samples.
|
|
115
|
+
|
|
116
|
+
Args:
|
|
117
|
+
n_neighbors: Number of neighbours ``k``.
|
|
118
|
+
weights: ``"uniform"`` or ``"distance"`` (inverse-distance weighted mean).
|
|
119
|
+
"""
|
|
120
|
+
|
|
121
|
+
def fit(self, X: ArrayLike, y: ArrayLike) -> "KNeighborsRegressor":
|
|
122
|
+
"""Store the training data ``X`` and numeric targets ``y``."""
|
|
123
|
+
X, y = check_X_y(X, y, y_numeric=True)
|
|
124
|
+
self._validate_params(X.shape[0])
|
|
125
|
+
self._fit_X, self._fit_y = X, y
|
|
126
|
+
return self
|
|
127
|
+
|
|
128
|
+
def predict(self, X: ArrayLike) -> np.ndarray:
|
|
129
|
+
"""Return the (weighted) mean neighbour target for each sample."""
|
|
130
|
+
dist, idx = self.kneighbors(X)
|
|
131
|
+
w = self._neighbor_weights(dist)
|
|
132
|
+
return (w * self._fit_y[idx]).sum(axis=1) / w.sum(axis=1)
|
mlkit/preprocessing.py
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Feature scaling transformers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Optional, Tuple
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
from numpy.typing import ArrayLike
|
|
9
|
+
|
|
10
|
+
from .base import BaseEstimator, TransformerMixin, check_array, check_is_fitted
|
|
11
|
+
|
|
12
|
+
__all__ = ["StandardScaler", "MinMaxScaler"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _check_n_features(X: np.ndarray, expected: int) -> None:
|
|
16
|
+
if X.shape[1] != expected:
|
|
17
|
+
raise ValueError(f"X has {X.shape[1]} features, expected {expected}.")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class StandardScaler(BaseEstimator, TransformerMixin):
|
|
21
|
+
"""Standardise features to zero mean and unit variance.
|
|
22
|
+
|
|
23
|
+
Constant features (zero variance) are left centred but not scaled.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
with_mean: Subtract the per-feature mean.
|
|
27
|
+
with_std: Divide by the per-feature (population) standard deviation.
|
|
28
|
+
|
|
29
|
+
Attributes:
|
|
30
|
+
mean_: Per-feature means, shape ``(n_features,)``.
|
|
31
|
+
scale_: Per-feature divisors, shape ``(n_features,)``.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def __init__(self, with_mean: bool = True, with_std: bool = True) -> None:
|
|
35
|
+
self.with_mean = with_mean
|
|
36
|
+
self.with_std = with_std
|
|
37
|
+
self.mean_: np.ndarray = None # type: ignore[assignment]
|
|
38
|
+
self.scale_: np.ndarray = None # type: ignore[assignment]
|
|
39
|
+
|
|
40
|
+
def fit(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> "StandardScaler":
|
|
41
|
+
"""Compute the mean and standard deviation of each column of ``X``."""
|
|
42
|
+
X = check_array(X)
|
|
43
|
+
self.mean_ = X.mean(axis=0) if self.with_mean else np.zeros(X.shape[1])
|
|
44
|
+
if self.with_std:
|
|
45
|
+
std = X.std(axis=0)
|
|
46
|
+
self.scale_ = np.where(std == 0.0, 1.0, std)
|
|
47
|
+
else:
|
|
48
|
+
self.scale_ = np.ones(X.shape[1])
|
|
49
|
+
return self
|
|
50
|
+
|
|
51
|
+
def transform(self, X: ArrayLike) -> np.ndarray:
|
|
52
|
+
"""Return ``(X - mean_) / scale_``."""
|
|
53
|
+
check_is_fitted(self, "mean_")
|
|
54
|
+
X = check_array(X)
|
|
55
|
+
_check_n_features(X, self.mean_.shape[0])
|
|
56
|
+
return (X - self.mean_) / self.scale_
|
|
57
|
+
|
|
58
|
+
def inverse_transform(self, X: ArrayLike) -> np.ndarray:
|
|
59
|
+
"""Undo :meth:`transform`."""
|
|
60
|
+
check_is_fitted(self, "mean_")
|
|
61
|
+
X = check_array(X)
|
|
62
|
+
_check_n_features(X, self.mean_.shape[0])
|
|
63
|
+
return X * self.scale_ + self.mean_
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class MinMaxScaler(BaseEstimator, TransformerMixin):
|
|
67
|
+
"""Rescale each feature linearly into ``feature_range``.
|
|
68
|
+
|
|
69
|
+
Constant features are mapped to the lower bound of the range.
|
|
70
|
+
|
|
71
|
+
Args:
|
|
72
|
+
feature_range: Target ``(min, max)`` interval.
|
|
73
|
+
|
|
74
|
+
Attributes:
|
|
75
|
+
data_min_: Per-feature minimum seen in ``fit``.
|
|
76
|
+
data_max_: Per-feature maximum seen in ``fit``.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
def __init__(self, feature_range: Tuple[float, float] = (0.0, 1.0)) -> None:
|
|
80
|
+
self.feature_range = feature_range
|
|
81
|
+
self.data_min_: np.ndarray = None # type: ignore[assignment]
|
|
82
|
+
self.data_max_: np.ndarray = None # type: ignore[assignment]
|
|
83
|
+
|
|
84
|
+
def fit(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> "MinMaxScaler":
|
|
85
|
+
"""Record the per-feature minimum and maximum of ``X``."""
|
|
86
|
+
lo, hi = self.feature_range
|
|
87
|
+
if not lo < hi:
|
|
88
|
+
raise ValueError(f"feature_range min must be < max, got {self.feature_range}.")
|
|
89
|
+
X = check_array(X)
|
|
90
|
+
self.data_min_ = X.min(axis=0)
|
|
91
|
+
self.data_max_ = X.max(axis=0)
|
|
92
|
+
return self
|
|
93
|
+
|
|
94
|
+
def _scale(self) -> np.ndarray:
|
|
95
|
+
rng = self.data_max_ - self.data_min_
|
|
96
|
+
lo, hi = self.feature_range
|
|
97
|
+
return (hi - lo) / np.where(rng == 0.0, 1.0, rng)
|
|
98
|
+
|
|
99
|
+
def transform(self, X: ArrayLike) -> np.ndarray:
|
|
100
|
+
"""Map ``X`` into ``feature_range`` (values outside the fitted range are not clipped)."""
|
|
101
|
+
check_is_fitted(self, "data_min_")
|
|
102
|
+
X = check_array(X)
|
|
103
|
+
_check_n_features(X, self.data_min_.shape[0])
|
|
104
|
+
return (X - self.data_min_) * self._scale() + self.feature_range[0]
|
|
105
|
+
|
|
106
|
+
def inverse_transform(self, X: ArrayLike) -> np.ndarray:
|
|
107
|
+
"""Undo :meth:`transform`."""
|
|
108
|
+
check_is_fitted(self, "data_min_")
|
|
109
|
+
X = check_array(X)
|
|
110
|
+
_check_n_features(X, self.data_min_.shape[0])
|
|
111
|
+
return (X - self.feature_range[0]) / self._scale() + self.data_min_
|
mlkit/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mlkit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Machine Learning Toolkit: a small, readable NumPy toolkit with a fit/predict API, classic models, preprocessing and metrics.
|
|
5
|
+
Author: nehz
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: machine learning,regression,classification,clustering,numpy
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Intended Audience :: Education
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Python: >=3.9
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: numpy>=1.22
|
|
26
|
+
Provides-Extra: test
|
|
27
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
28
|
+
Dynamic: license-file
|
|
29
|
+
|
|
30
|
+
# mlkit
|
|
31
|
+
|
|
32
|
+
**Machine Learning Toolkit**: a small, readable machine learning library built on NumPy.
|
|
33
|
+
|
|
34
|
+
mlkit has the familiar `fit` / `predict` / `transform` estimator interface and a
|
|
35
|
+
handful of classic algorithms, each implemented in a few dozen lines of plain
|
|
36
|
+
NumPy. Use it to learn how the algorithms work, to teach them, or for small
|
|
37
|
+
projects where a large ML stack is too much. NumPy is its only dependency.
|
|
38
|
+
|
|
39
|
+
## Features
|
|
40
|
+
|
|
41
|
+
- **Estimator API**: `fit`, `predict`, `transform`, `score`, `get_params` / `set_params`, `clone`
|
|
42
|
+
- **Linear models**: `LinearRegression` (least squares), `Ridge` (L2, closed form), `LogisticRegression` (multinomial, gradient descent, optional L2)
|
|
43
|
+
- **Neighbours**: `KNeighborsClassifier`, `KNeighborsRegressor` (uniform or inverse-distance weights)
|
|
44
|
+
- **Clustering**: `KMeans` (k-means++ seeding, multiple restarts, reproducible via `random_state`)
|
|
45
|
+
- **Preprocessing**: `StandardScaler`, `MinMaxScaler` (both with `inverse_transform`)
|
|
46
|
+
- **Model selection**: `train_test_split`, `KFold`, `cross_val_score`
|
|
47
|
+
- **Metrics**: accuracy, confusion matrix, precision / recall / F1 (binary and macro), MSE, MAE, R²
|
|
48
|
+
- **Datasets**: `make_blobs`, `make_regression` synthetic generators
|
|
49
|
+
- Input validation with clear errors, and `NotFittedError` when an estimator is used before `fit`
|
|
50
|
+
- Type hints throughout (ships `py.typed`)
|
|
51
|
+
|
|
52
|
+
## Installation
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install mlkit
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
From a source checkout:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
python3 -m venv .venv
|
|
62
|
+
.venv/bin/pip install -e ".[test]"
|
|
63
|
+
.venv/bin/python -m pytest
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Requires Python 3.9+ and NumPy 1.22+.
|
|
67
|
+
|
|
68
|
+
## Quickstart
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from mlkit import KMeans, LinearRegression, LogisticRegression, StandardScaler, train_test_split
|
|
72
|
+
from mlkit.datasets import make_blobs, make_regression
|
|
73
|
+
from mlkit.metrics import accuracy_score, r2_score
|
|
74
|
+
|
|
75
|
+
# Classification: scale the features, then fit a logistic regression
|
|
76
|
+
X, y = make_blobs(n_samples=300, centers=3, random_state=0)
|
|
77
|
+
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=0)
|
|
78
|
+
|
|
79
|
+
scaler = StandardScaler().fit(X_train)
|
|
80
|
+
clf = LogisticRegression(learning_rate=0.5, max_iter=2000)
|
|
81
|
+
clf.fit(scaler.transform(X_train), y_train)
|
|
82
|
+
print("accuracy:", accuracy_score(y_test, clf.predict(scaler.transform(X_test))))
|
|
83
|
+
|
|
84
|
+
# Regression
|
|
85
|
+
X, y, true_coef = make_regression(n_samples=200, n_features=3, noise=0.5, random_state=1)
|
|
86
|
+
reg = LinearRegression().fit(X, y)
|
|
87
|
+
print("R^2:", r2_score(y, reg.predict(X)), "coef:", reg.coef_)
|
|
88
|
+
|
|
89
|
+
# Clustering
|
|
90
|
+
km = KMeans(n_clusters=3, random_state=0).fit(X_train)
|
|
91
|
+
print("inertia:", km.inertia_, "labels:", km.labels_[:10])
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Cross-validation:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from mlkit import KFold, KNeighborsClassifier, cross_val_score
|
|
98
|
+
|
|
99
|
+
scores = cross_val_score(KNeighborsClassifier(n_neighbors=5), X_train, y_train,
|
|
100
|
+
cv=KFold(n_splits=5, shuffle=True, random_state=0))
|
|
101
|
+
print(scores.mean())
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## API overview
|
|
105
|
+
|
|
106
|
+
All estimators subclass `mlkit.BaseEstimator`. Hyper-parameters are the
|
|
107
|
+
constructor arguments. Learned attributes end in `_` and exist only after
|
|
108
|
+
`fit`. Every `fit` returns `self`. Calling `predict` or `transform` before `fit`
|
|
109
|
+
raises `mlkit.NotFittedError`.
|
|
110
|
+
|
|
111
|
+
### `mlkit` (top level)
|
|
112
|
+
|
|
113
|
+
| Name | Description |
|
|
114
|
+
| --- | --- |
|
|
115
|
+
| `BaseEstimator` | Base class with `get_params()`, `set_params(**params)`, and a readable `repr` |
|
|
116
|
+
| `NotFittedError` | Raised when an estimator is used before `fit` |
|
|
117
|
+
| `clone(estimator)` | New unfitted estimator with the same hyper-parameters |
|
|
118
|
+
| `__version__` | `"0.1.0"` |
|
|
119
|
+
|
|
120
|
+
### `mlkit.linear_model`
|
|
121
|
+
|
|
122
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
123
|
+
| --- | --- | --- | --- |
|
|
124
|
+
| `LinearRegression` | `fit_intercept=True` | `fit(X, y)`, `predict(X)`, `score(X, y)` (R²) | `coef_`, `intercept_` |
|
|
125
|
+
| `Ridge` | `alpha=1.0`, `fit_intercept=True` | `fit`, `predict`, `score` (R²) | `coef_`, `intercept_` |
|
|
126
|
+
| `LogisticRegression` | `learning_rate=0.1`, `max_iter=1000`, `tol=1e-6`, `alpha=0.0`, `fit_intercept=True` | `fit`, `predict`, `predict_proba`, `decision_function`, `score` (accuracy) | `classes_`, `coef_` (n_classes × n_features), `intercept_`, `n_iter_` |
|
|
127
|
+
|
|
128
|
+
### `mlkit.neighbors`
|
|
129
|
+
|
|
130
|
+
| Name | Parameters | Methods / notes |
|
|
131
|
+
| --- | --- | --- |
|
|
132
|
+
| `KNeighborsClassifier` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `predict_proba`, `kneighbors(X)`, `score` (accuracy); `classes_` |
|
|
133
|
+
| `KNeighborsRegressor` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `kneighbors(X)`, `score` (R²) |
|
|
134
|
+
| `pairwise_distances(A, B)` | | Euclidean distance matrix of shape `(len(A), len(B))` |
|
|
135
|
+
|
|
136
|
+
### `mlkit.cluster`
|
|
137
|
+
|
|
138
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
139
|
+
| --- | --- | --- | --- |
|
|
140
|
+
| `KMeans` | `n_clusters=8`, `n_init=10`, `max_iter=300`, `tol=1e-6`, `random_state=None` | `fit(X)`, `predict(X)`, `fit_predict(X)`, `transform(X)` (distances to centroids) | `cluster_centers_`, `labels_`, `inertia_`, `n_iter_` |
|
|
141
|
+
|
|
142
|
+
### `mlkit.preprocessing`
|
|
143
|
+
|
|
144
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
145
|
+
| --- | --- | --- | --- |
|
|
146
|
+
| `StandardScaler` | `with_mean=True`, `with_std=True` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `mean_`, `scale_` |
|
|
147
|
+
| `MinMaxScaler` | `feature_range=(0.0, 1.0)` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `data_min_`, `data_max_` |
|
|
148
|
+
|
|
149
|
+
### `mlkit.model_selection`
|
|
150
|
+
|
|
151
|
+
| Name | Signature |
|
|
152
|
+
| --- | --- |
|
|
153
|
+
| `train_test_split` | `train_test_split(*arrays, test_size=0.25, shuffle=True, random_state=None)` returns `[a_train, a_test, b_train, b_test, ...]` |
|
|
154
|
+
| `KFold` | `KFold(n_splits=5, shuffle=False, random_state=None)`; `.split(X)` yields `(train_idx, test_idx)` |
|
|
155
|
+
| `cross_val_score` | `cross_val_score(estimator, X, y, *, cv=5, scoring=None)` returns an array of per-fold scores (`scoring` is `metric(y_true, y_pred)`; it defaults to `estimator.score`) |
|
|
156
|
+
|
|
157
|
+
### `mlkit.metrics`
|
|
158
|
+
|
|
159
|
+
| Function | Notes |
|
|
160
|
+
| --- | --- |
|
|
161
|
+
| `accuracy_score(y_true, y_pred)` | Fraction of exact matches |
|
|
162
|
+
| `confusion_matrix(y_true, y_pred, labels=None)` | `C[i, j]`: true `labels[i]` predicted as `labels[j]` |
|
|
163
|
+
| `precision_score(y_true, y_pred, *, average="binary", pos_label=1)` | `average` is `"binary"` or `"macro"`. A zero denominator gives 0.0 |
|
|
164
|
+
| `recall_score(...)` | Same arguments as `precision_score` |
|
|
165
|
+
| `f1_score(...)` | Same arguments as `precision_score` |
|
|
166
|
+
| `mean_squared_error(y_true, y_pred)` | |
|
|
167
|
+
| `mean_absolute_error(y_true, y_pred)` | |
|
|
168
|
+
| `r2_score(y_true, y_pred)` | Coefficient of determination |
|
|
169
|
+
|
|
170
|
+
### `mlkit.datasets`
|
|
171
|
+
|
|
172
|
+
| Function | Returns |
|
|
173
|
+
| --- | --- |
|
|
174
|
+
| `make_blobs(n_samples=100, n_features=2, centers=3, cluster_std=1.0, random_state=None)` | `(X, y)` |
|
|
175
|
+
| `make_regression(n_samples=100, n_features=3, noise=0.0, bias=0.0, random_state=None)` | `(X, y, coef)` |
|
|
176
|
+
|
|
177
|
+
### `mlkit.base`
|
|
178
|
+
|
|
179
|
+
Contains the `ClassifierMixin`, `RegressorMixin`, `TransformerMixin` and
|
|
180
|
+
`ClusterMixin` mixins for writing your own estimators, plus the validation
|
|
181
|
+
helpers `check_array`, `check_X_y` and `check_is_fitted`.
|
|
182
|
+
|
|
183
|
+
## Scope and limitations
|
|
184
|
+
|
|
185
|
+
mlkit puts clarity ahead of speed. k-NN uses brute-force distances, and
|
|
186
|
+
logistic regression uses plain full-batch gradient descent, so scale your
|
|
187
|
+
features first. It is not a replacement for scikit-learn on large datasets.
|
|
188
|
+
|
|
189
|
+
## License
|
|
190
|
+
|
|
191
|
+
MIT
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
mlkit/__init__.py,sha256=PIitq7wzOvuCtYaW5QRLeynAylx95OmtPvPYsSutM9k,946
|
|
2
|
+
mlkit/base.py,sha256=RBQfQ9eN1gF4i5ulqfUq5HroRaXliJasnAQFWBHo7oY,5440
|
|
3
|
+
mlkit/cluster.py,sha256=0a3fmSAL7BdxmDoDyk7SSdAnSS8g7n-rISbZAntrgCU,4311
|
|
4
|
+
mlkit/datasets.py,sha256=iUktHqbv5RxaSTzMxYd1m29QHJsGiZF5Vd08FsdOZ20,2560
|
|
5
|
+
mlkit/linear_model.py,sha256=ws3qNCKif56Qm3otu-WP0zcoDrqjw-kGll7RrHKtEII,7142
|
|
6
|
+
mlkit/metrics.py,sha256=Goj7d7PjDbmVLkZJB4E8zN-lFUTZphNfcdqmMVr4MBk,5191
|
|
7
|
+
mlkit/model_selection.py,sha256=0XQ_XctTHMfLH7z_Uy13WyvMZwbA4NXy7m7lYk5IOlI,4243
|
|
8
|
+
mlkit/neighbors.py,sha256=U9SOTHZSrT184mFOOS1ldPR6AUaHScbHCWhZ6El5fFk,5155
|
|
9
|
+
mlkit/preprocessing.py,sha256=J_K3VgSfIwhIj_4tHYxuFSBPa2LlZqODn1z3KFq3XWM,4114
|
|
10
|
+
mlkit/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
11
|
+
mlkit-0.1.0.dist-info/licenses/LICENSE,sha256=pAfYREEW9GAy7cnK20OXjDn7ofJahYny9GCuIZQTDAA,1061
|
|
12
|
+
mlkit-0.1.0.dist-info/METADATA,sha256=yAhR-NbYVhkl539eXSuCS4cNIQoBT86b-hzgEK_OjvI,8318
|
|
13
|
+
mlkit-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
14
|
+
mlkit-0.1.0.dist-info/top_level.txt,sha256=bMLuHaaaZpWUcZAZQPEDE7tlGs8ZYUJ9zAsfZNhuoAs,6
|
|
15
|
+
mlkit-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 nehz
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
mlkit
|