mlkit 0.0.1__tar.gz → 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mlkit-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 nehz
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
mlkit-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,191 @@
1
+ Metadata-Version: 2.4
2
+ Name: mlkit
3
+ Version: 0.1.0
4
+ Summary: Machine Learning Toolkit: a small, readable NumPy toolkit with a fit/predict API, classic models, preprocessing and metrics.
5
+ Author: nehz
6
+ License-Expression: MIT
7
+ Keywords: machine learning,regression,classification,clustering,numpy
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: Intended Audience :: Education
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3 :: Only
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Typing :: Typed
22
+ Requires-Python: >=3.9
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: numpy>=1.22
26
+ Provides-Extra: test
27
+ Requires-Dist: pytest>=7; extra == "test"
28
+ Dynamic: license-file
29
+
30
+ # mlkit
31
+
32
+ **Machine Learning Toolkit**: a small, readable machine learning library built on NumPy.
33
+
34
+ mlkit has the familiar `fit` / `predict` / `transform` estimator interface and a
35
+ handful of classic algorithms, each implemented in a few dozen lines of plain
36
+ NumPy. Use it to learn how the algorithms work, to teach them, or for small
37
+ projects where a large ML stack is too much. NumPy is its only dependency.
38
+
39
+ ## Features
40
+
41
+ - **Estimator API**: `fit`, `predict`, `transform`, `score`, `get_params` / `set_params`, `clone`
42
+ - **Linear models**: `LinearRegression` (least squares), `Ridge` (L2, closed form), `LogisticRegression` (multinomial, gradient descent, optional L2)
43
+ - **Neighbours**: `KNeighborsClassifier`, `KNeighborsRegressor` (uniform or inverse-distance weights)
44
+ - **Clustering**: `KMeans` (k-means++ seeding, multiple restarts, reproducible via `random_state`)
45
+ - **Preprocessing**: `StandardScaler`, `MinMaxScaler` (both with `inverse_transform`)
46
+ - **Model selection**: `train_test_split`, `KFold`, `cross_val_score`
47
+ - **Metrics**: accuracy, confusion matrix, precision / recall / F1 (binary and macro), MSE, MAE, R²
48
+ - **Datasets**: `make_blobs`, `make_regression` synthetic generators
49
+ - Input validation with clear errors, and `NotFittedError` when an estimator is used before `fit`
50
+ - Type hints throughout (ships `py.typed`)
51
+
52
+ ## Installation
53
+
54
+ ```bash
55
+ pip install mlkit
56
+ ```
57
+
58
+ From a source checkout:
59
+
60
+ ```bash
61
+ python3 -m venv .venv
62
+ .venv/bin/pip install -e ".[test]"
63
+ .venv/bin/python -m pytest
64
+ ```
65
+
66
+ Requires Python 3.9+ and NumPy 1.22+.
67
+
68
+ ## Quickstart
69
+
70
+ ```python
71
+ from mlkit import KMeans, LinearRegression, LogisticRegression, StandardScaler, train_test_split
72
+ from mlkit.datasets import make_blobs, make_regression
73
+ from mlkit.metrics import accuracy_score, r2_score
74
+
75
+ # Classification: scale the features, then fit a logistic regression
76
+ X, y = make_blobs(n_samples=300, centers=3, random_state=0)
77
+ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=0)
78
+
79
+ scaler = StandardScaler().fit(X_train)
80
+ clf = LogisticRegression(learning_rate=0.5, max_iter=2000)
81
+ clf.fit(scaler.transform(X_train), y_train)
82
+ print("accuracy:", accuracy_score(y_test, clf.predict(scaler.transform(X_test))))
83
+
84
+ # Regression
85
+ X, y, true_coef = make_regression(n_samples=200, n_features=3, noise=0.5, random_state=1)
86
+ reg = LinearRegression().fit(X, y)
87
+ print("R^2:", r2_score(y, reg.predict(X)), "coef:", reg.coef_)
88
+
89
+ # Clustering
90
+ km = KMeans(n_clusters=3, random_state=0).fit(X_train)
91
+ print("inertia:", km.inertia_, "labels:", km.labels_[:10])
92
+ ```
93
+
94
+ Cross-validation:
95
+
96
+ ```python
97
+ from mlkit import KFold, KNeighborsClassifier, cross_val_score
98
+
99
+ scores = cross_val_score(KNeighborsClassifier(n_neighbors=5), X_train, y_train,
100
+ cv=KFold(n_splits=5, shuffle=True, random_state=0))
101
+ print(scores.mean())
102
+ ```
103
+
104
+ ## API overview
105
+
106
+ All estimators subclass `mlkit.BaseEstimator`. Hyper-parameters are the
107
+ constructor arguments. Learned attributes end in `_` and exist only after
108
+ `fit`. Every `fit` returns `self`. Calling `predict` or `transform` before `fit`
109
+ raises `mlkit.NotFittedError`.
110
+
111
+ ### `mlkit` (top level)
112
+
113
+ | Name | Description |
114
+ | --- | --- |
115
+ | `BaseEstimator` | Base class with `get_params()`, `set_params(**params)`, and a readable `repr` |
116
+ | `NotFittedError` | Raised when an estimator is used before `fit` |
117
+ | `clone(estimator)` | New unfitted estimator with the same hyper-parameters |
118
+ | `__version__` | `"0.1.0"` |
119
+
120
+ ### `mlkit.linear_model`
121
+
122
+ | Class | Parameters | Methods | Fitted attributes |
123
+ | --- | --- | --- | --- |
124
+ | `LinearRegression` | `fit_intercept=True` | `fit(X, y)`, `predict(X)`, `score(X, y)` (R²) | `coef_`, `intercept_` |
125
+ | `Ridge` | `alpha=1.0`, `fit_intercept=True` | `fit`, `predict`, `score` (R²) | `coef_`, `intercept_` |
126
+ | `LogisticRegression` | `learning_rate=0.1`, `max_iter=1000`, `tol=1e-6`, `alpha=0.0`, `fit_intercept=True` | `fit`, `predict`, `predict_proba`, `decision_function`, `score` (accuracy) | `classes_`, `coef_` (n_classes × n_features), `intercept_`, `n_iter_` |
127
+
128
+ ### `mlkit.neighbors`
129
+
130
+ | Name | Parameters | Methods / notes |
131
+ | --- | --- | --- |
132
+ | `KNeighborsClassifier` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `predict_proba`, `kneighbors(X)`, `score` (accuracy); `classes_` |
133
+ | `KNeighborsRegressor` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `kneighbors(X)`, `score` (R²) |
134
+ | `pairwise_distances(A, B)` | | Euclidean distance matrix of shape `(len(A), len(B))` |
135
+
136
+ ### `mlkit.cluster`
137
+
138
+ | Class | Parameters | Methods | Fitted attributes |
139
+ | --- | --- | --- | --- |
140
+ | `KMeans` | `n_clusters=8`, `n_init=10`, `max_iter=300`, `tol=1e-6`, `random_state=None` | `fit(X)`, `predict(X)`, `fit_predict(X)`, `transform(X)` (distances to centroids) | `cluster_centers_`, `labels_`, `inertia_`, `n_iter_` |
141
+
142
+ ### `mlkit.preprocessing`
143
+
144
+ | Class | Parameters | Methods | Fitted attributes |
145
+ | --- | --- | --- | --- |
146
+ | `StandardScaler` | `with_mean=True`, `with_std=True` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `mean_`, `scale_` |
147
+ | `MinMaxScaler` | `feature_range=(0.0, 1.0)` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `data_min_`, `data_max_` |
148
+
149
+ ### `mlkit.model_selection`
150
+
151
+ | Name | Signature |
152
+ | --- | --- |
153
+ | `train_test_split` | `train_test_split(*arrays, test_size=0.25, shuffle=True, random_state=None)` returns `[a_train, a_test, b_train, b_test, ...]` |
154
+ | `KFold` | `KFold(n_splits=5, shuffle=False, random_state=None)`; `.split(X)` yields `(train_idx, test_idx)` |
155
+ | `cross_val_score` | `cross_val_score(estimator, X, y, *, cv=5, scoring=None)` returns an array of per-fold scores (`scoring` is `metric(y_true, y_pred)`; it defaults to `estimator.score`) |
156
+
157
+ ### `mlkit.metrics`
158
+
159
+ | Function | Notes |
160
+ | --- | --- |
161
+ | `accuracy_score(y_true, y_pred)` | Fraction of exact matches |
162
+ | `confusion_matrix(y_true, y_pred, labels=None)` | `C[i, j]`: true `labels[i]` predicted as `labels[j]` |
163
+ | `precision_score(y_true, y_pred, *, average="binary", pos_label=1)` | `average` is `"binary"` or `"macro"`. A zero denominator gives 0.0 |
164
+ | `recall_score(...)` | Same arguments as `precision_score` |
165
+ | `f1_score(...)` | Same arguments as `precision_score` |
166
+ | `mean_squared_error(y_true, y_pred)` | |
167
+ | `mean_absolute_error(y_true, y_pred)` | |
168
+ | `r2_score(y_true, y_pred)` | Coefficient of determination |
169
+
170
+ ### `mlkit.datasets`
171
+
172
+ | Function | Returns |
173
+ | --- | --- |
174
+ | `make_blobs(n_samples=100, n_features=2, centers=3, cluster_std=1.0, random_state=None)` | `(X, y)` |
175
+ | `make_regression(n_samples=100, n_features=3, noise=0.0, bias=0.0, random_state=None)` | `(X, y, coef)` |
176
+
177
+ ### `mlkit.base`
178
+
179
+ Contains the `ClassifierMixin`, `RegressorMixin`, `TransformerMixin` and
180
+ `ClusterMixin` mixins for writing your own estimators, plus the validation
181
+ helpers `check_array`, `check_X_y` and `check_is_fitted`.
182
+
183
+ ## Scope and limitations
184
+
185
+ mlkit puts clarity ahead of speed. k-NN uses brute-force distances, and
186
+ logistic regression uses plain full-batch gradient descent, so scale your
187
+ features first. It is not a replacement for scikit-learn on large datasets.
188
+
189
+ ## License
190
+
191
+ MIT
mlkit-0.1.0/README.md ADDED
@@ -0,0 +1,162 @@
1
+ # mlkit
2
+
3
+ **Machine Learning Toolkit**: a small, readable machine learning library built on NumPy.
4
+
5
+ mlkit has the familiar `fit` / `predict` / `transform` estimator interface and a
6
+ handful of classic algorithms, each implemented in a few dozen lines of plain
7
+ NumPy. Use it to learn how the algorithms work, to teach them, or for small
8
+ projects where a large ML stack is too much. NumPy is its only dependency.
9
+
10
+ ## Features
11
+
12
+ - **Estimator API**: `fit`, `predict`, `transform`, `score`, `get_params` / `set_params`, `clone`
13
+ - **Linear models**: `LinearRegression` (least squares), `Ridge` (L2, closed form), `LogisticRegression` (multinomial, gradient descent, optional L2)
14
+ - **Neighbours**: `KNeighborsClassifier`, `KNeighborsRegressor` (uniform or inverse-distance weights)
15
+ - **Clustering**: `KMeans` (k-means++ seeding, multiple restarts, reproducible via `random_state`)
16
+ - **Preprocessing**: `StandardScaler`, `MinMaxScaler` (both with `inverse_transform`)
17
+ - **Model selection**: `train_test_split`, `KFold`, `cross_val_score`
18
+ - **Metrics**: accuracy, confusion matrix, precision / recall / F1 (binary and macro), MSE, MAE, R²
19
+ - **Datasets**: `make_blobs`, `make_regression` synthetic generators
20
+ - Input validation with clear errors, and `NotFittedError` when an estimator is used before `fit`
21
+ - Type hints throughout (ships `py.typed`)
22
+
23
+ ## Installation
24
+
25
+ ```bash
26
+ pip install mlkit
27
+ ```
28
+
29
+ From a source checkout:
30
+
31
+ ```bash
32
+ python3 -m venv .venv
33
+ .venv/bin/pip install -e ".[test]"
34
+ .venv/bin/python -m pytest
35
+ ```
36
+
37
+ Requires Python 3.9+ and NumPy 1.22+.
38
+
39
+ ## Quickstart
40
+
41
+ ```python
42
+ from mlkit import KMeans, LinearRegression, LogisticRegression, StandardScaler, train_test_split
43
+ from mlkit.datasets import make_blobs, make_regression
44
+ from mlkit.metrics import accuracy_score, r2_score
45
+
46
+ # Classification: scale the features, then fit a logistic regression
47
+ X, y = make_blobs(n_samples=300, centers=3, random_state=0)
48
+ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=0)
49
+
50
+ scaler = StandardScaler().fit(X_train)
51
+ clf = LogisticRegression(learning_rate=0.5, max_iter=2000)
52
+ clf.fit(scaler.transform(X_train), y_train)
53
+ print("accuracy:", accuracy_score(y_test, clf.predict(scaler.transform(X_test))))
54
+
55
+ # Regression
56
+ X, y, true_coef = make_regression(n_samples=200, n_features=3, noise=0.5, random_state=1)
57
+ reg = LinearRegression().fit(X, y)
58
+ print("R^2:", r2_score(y, reg.predict(X)), "coef:", reg.coef_)
59
+
60
+ # Clustering
61
+ km = KMeans(n_clusters=3, random_state=0).fit(X_train)
62
+ print("inertia:", km.inertia_, "labels:", km.labels_[:10])
63
+ ```
64
+
65
+ Cross-validation:
66
+
67
+ ```python
68
+ from mlkit import KFold, KNeighborsClassifier, cross_val_score
69
+
70
+ scores = cross_val_score(KNeighborsClassifier(n_neighbors=5), X_train, y_train,
71
+ cv=KFold(n_splits=5, shuffle=True, random_state=0))
72
+ print(scores.mean())
73
+ ```
74
+
75
+ ## API overview
76
+
77
+ All estimators subclass `mlkit.BaseEstimator`. Hyper-parameters are the
78
+ constructor arguments. Learned attributes end in `_` and exist only after
79
+ `fit`. Every `fit` returns `self`. Calling `predict` or `transform` before `fit`
80
+ raises `mlkit.NotFittedError`.
81
+
82
+ ### `mlkit` (top level)
83
+
84
+ | Name | Description |
85
+ | --- | --- |
86
+ | `BaseEstimator` | Base class with `get_params()`, `set_params(**params)`, and a readable `repr` |
87
+ | `NotFittedError` | Raised when an estimator is used before `fit` |
88
+ | `clone(estimator)` | New unfitted estimator with the same hyper-parameters |
89
+ | `__version__` | `"0.1.0"` |
90
+
91
+ ### `mlkit.linear_model`
92
+
93
+ | Class | Parameters | Methods | Fitted attributes |
94
+ | --- | --- | --- | --- |
95
+ | `LinearRegression` | `fit_intercept=True` | `fit(X, y)`, `predict(X)`, `score(X, y)` (R²) | `coef_`, `intercept_` |
96
+ | `Ridge` | `alpha=1.0`, `fit_intercept=True` | `fit`, `predict`, `score` (R²) | `coef_`, `intercept_` |
97
+ | `LogisticRegression` | `learning_rate=0.1`, `max_iter=1000`, `tol=1e-6`, `alpha=0.0`, `fit_intercept=True` | `fit`, `predict`, `predict_proba`, `decision_function`, `score` (accuracy) | `classes_`, `coef_` (n_classes × n_features), `intercept_`, `n_iter_` |
98
+
99
+ ### `mlkit.neighbors`
100
+
101
+ | Name | Parameters | Methods / notes |
102
+ | --- | --- | --- |
103
+ | `KNeighborsClassifier` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `predict_proba`, `kneighbors(X)`, `score` (accuracy); `classes_` |
104
+ | `KNeighborsRegressor` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `kneighbors(X)`, `score` (R²) |
105
+ | `pairwise_distances(A, B)` | | Euclidean distance matrix of shape `(len(A), len(B))` |
106
+
107
+ ### `mlkit.cluster`
108
+
109
+ | Class | Parameters | Methods | Fitted attributes |
110
+ | --- | --- | --- | --- |
111
+ | `KMeans` | `n_clusters=8`, `n_init=10`, `max_iter=300`, `tol=1e-6`, `random_state=None` | `fit(X)`, `predict(X)`, `fit_predict(X)`, `transform(X)` (distances to centroids) | `cluster_centers_`, `labels_`, `inertia_`, `n_iter_` |
112
+
113
+ ### `mlkit.preprocessing`
114
+
115
+ | Class | Parameters | Methods | Fitted attributes |
116
+ | --- | --- | --- | --- |
117
+ | `StandardScaler` | `with_mean=True`, `with_std=True` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `mean_`, `scale_` |
118
+ | `MinMaxScaler` | `feature_range=(0.0, 1.0)` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `data_min_`, `data_max_` |
119
+
120
+ ### `mlkit.model_selection`
121
+
122
+ | Name | Signature |
123
+ | --- | --- |
124
+ | `train_test_split` | `train_test_split(*arrays, test_size=0.25, shuffle=True, random_state=None)` returns `[a_train, a_test, b_train, b_test, ...]` |
125
+ | `KFold` | `KFold(n_splits=5, shuffle=False, random_state=None)`; `.split(X)` yields `(train_idx, test_idx)` |
126
+ | `cross_val_score` | `cross_val_score(estimator, X, y, *, cv=5, scoring=None)` returns an array of per-fold scores (`scoring` is `metric(y_true, y_pred)`; it defaults to `estimator.score`) |
127
+
128
+ ### `mlkit.metrics`
129
+
130
+ | Function | Notes |
131
+ | --- | --- |
132
+ | `accuracy_score(y_true, y_pred)` | Fraction of exact matches |
133
+ | `confusion_matrix(y_true, y_pred, labels=None)` | `C[i, j]`: true `labels[i]` predicted as `labels[j]` |
134
+ | `precision_score(y_true, y_pred, *, average="binary", pos_label=1)` | `average` is `"binary"` or `"macro"`. A zero denominator gives 0.0 |
135
+ | `recall_score(...)` | Same arguments as `precision_score` |
136
+ | `f1_score(...)` | Same arguments as `precision_score` |
137
+ | `mean_squared_error(y_true, y_pred)` | |
138
+ | `mean_absolute_error(y_true, y_pred)` | |
139
+ | `r2_score(y_true, y_pred)` | Coefficient of determination |
140
+
141
+ ### `mlkit.datasets`
142
+
143
+ | Function | Returns |
144
+ | --- | --- |
145
+ | `make_blobs(n_samples=100, n_features=2, centers=3, cluster_std=1.0, random_state=None)` | `(X, y)` |
146
+ | `make_regression(n_samples=100, n_features=3, noise=0.0, bias=0.0, random_state=None)` | `(X, y, coef)` |
147
+
148
+ ### `mlkit.base`
149
+
150
+ Contains the `ClassifierMixin`, `RegressorMixin`, `TransformerMixin` and
151
+ `ClusterMixin` mixins for writing your own estimators, plus the validation
152
+ helpers `check_array`, `check_X_y` and `check_is_fitted`.
153
+
154
+ ## Scope and limitations
155
+
156
+ mlkit puts clarity ahead of speed. k-NN uses brute-force distances, and
157
+ logistic regression uses plain full-batch gradient descent, so scale your
158
+ features first. It is not a replacement for scikit-learn on large datasets.
159
+
160
+ ## License
161
+
162
+ MIT
@@ -0,0 +1,44 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "mlkit"
7
+ version = "0.1.0"
8
+ description = "Machine Learning Toolkit: a small, readable NumPy toolkit with a fit/predict API, classic models, preprocessing and metrics."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "nehz" }]
14
+ keywords = ["machine learning", "regression", "classification", "clustering", "numpy"]
15
+ classifiers = [
16
+ "Development Status :: 3 - Alpha",
17
+ "Intended Audience :: Developers",
18
+ "Intended Audience :: Education",
19
+ "Intended Audience :: Science/Research",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3 :: Only",
23
+ "Programming Language :: Python :: 3.9",
24
+ "Programming Language :: Python :: 3.10",
25
+ "Programming Language :: Python :: 3.11",
26
+ "Programming Language :: Python :: 3.12",
27
+ "Programming Language :: Python :: 3.13",
28
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
29
+ "Typing :: Typed",
30
+ ]
31
+ dependencies = ["numpy>=1.22"]
32
+
33
+ [project.optional-dependencies]
34
+ test = ["pytest>=7"]
35
+
36
+ [tool.setuptools.packages.find]
37
+ where = ["src"]
38
+
39
+ [tool.setuptools.package-data]
40
+ mlkit = ["py.typed"]
41
+
42
+ [tool.pytest.ini_options]
43
+ testpaths = ["tests"]
44
+ addopts = "-q"
@@ -1,4 +1,4 @@
1
- [egg_info]
2
- tag_build =
3
- tag_date = 0
4
-
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,35 @@
1
+ """mlkit - Machine Learning Toolkit.
2
+
3
+ A small, readable NumPy toolkit with a scikit-learn style ``fit``/``predict``
4
+ interface, classic models, preprocessing, model selection and metrics.
5
+ """
6
+
7
+ from . import datasets, metrics
8
+ from .base import BaseEstimator, NotFittedError, clone
9
+ from .cluster import KMeans
10
+ from .linear_model import LinearRegression, LogisticRegression, Ridge
11
+ from .model_selection import KFold, cross_val_score, train_test_split
12
+ from .neighbors import KNeighborsClassifier, KNeighborsRegressor
13
+ from .preprocessing import MinMaxScaler, StandardScaler
14
+
15
+ __version__ = "0.1.0"
16
+
17
+ __all__ = [
18
+ "BaseEstimator",
19
+ "NotFittedError",
20
+ "clone",
21
+ "LinearRegression",
22
+ "Ridge",
23
+ "LogisticRegression",
24
+ "KNeighborsClassifier",
25
+ "KNeighborsRegressor",
26
+ "KMeans",
27
+ "StandardScaler",
28
+ "MinMaxScaler",
29
+ "train_test_split",
30
+ "KFold",
31
+ "cross_val_score",
32
+ "datasets",
33
+ "metrics",
34
+ "__version__",
35
+ ]
@@ -0,0 +1,165 @@
1
+ """Core estimator interface, mixins and input validation helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import inspect
6
+ from typing import Any, Optional, Tuple
7
+
8
+ import numpy as np
9
+ from numpy.typing import ArrayLike
10
+
11
+ __all__ = [
12
+ "BaseEstimator",
13
+ "ClassifierMixin",
14
+ "RegressorMixin",
15
+ "TransformerMixin",
16
+ "ClusterMixin",
17
+ "NotFittedError",
18
+ "check_array",
19
+ "check_X_y",
20
+ "check_is_fitted",
21
+ "clone",
22
+ ]
23
+
24
+
25
+ class NotFittedError(RuntimeError):
26
+ """Raised when ``predict``/``transform`` is called before ``fit``."""
27
+
28
+
29
+ def check_array(X: ArrayLike, *, ensure_2d: bool = True) -> np.ndarray:
30
+ """Convert ``X`` to a finite float array.
31
+
32
+ Args:
33
+ X: Array-like input.
34
+ ensure_2d: If True, a 2-D array of shape ``(n_samples, n_features)``
35
+ is required.
36
+
37
+ Returns:
38
+ A ``float64`` NumPy array.
39
+
40
+ Raises:
41
+ ValueError: If ``X`` is empty, has the wrong number of dimensions or
42
+ contains NaN/inf values.
43
+ """
44
+ arr = np.asarray(X, dtype=float)
45
+ if ensure_2d and arr.ndim != 2:
46
+ raise ValueError(f"Expected a 2-D array, got an array with ndim={arr.ndim}.")
47
+ if arr.size == 0:
48
+ raise ValueError("Input array is empty.")
49
+ if not np.all(np.isfinite(arr)):
50
+ raise ValueError("Input contains NaN or infinity.")
51
+ return arr
52
+
53
+
54
+ def check_X_y(X: ArrayLike, y: ArrayLike, *, y_numeric: bool = False) -> Tuple[np.ndarray, np.ndarray]:
55
+ """Validate a feature matrix and a 1-D target vector of matching length.
56
+
57
+ Args:
58
+ X: Array-like of shape ``(n_samples, n_features)``.
59
+ y: Array-like of shape ``(n_samples,)``.
60
+ y_numeric: If True, ``y`` is converted to ``float64``.
61
+
62
+ Returns:
63
+ The validated ``(X, y)`` pair.
64
+ """
65
+ X_arr = check_array(X)
66
+ y_arr = np.asarray(y, dtype=float if y_numeric else None)
67
+ if y_arr.ndim != 1:
68
+ raise ValueError(f"y must be 1-D, got ndim={y_arr.ndim}.")
69
+ if y_arr.shape[0] != X_arr.shape[0]:
70
+ raise ValueError(
71
+ f"X and y have inconsistent lengths: {X_arr.shape[0]} != {y_arr.shape[0]}."
72
+ )
73
+ if y_numeric and not np.all(np.isfinite(y_arr)):
74
+ raise ValueError("y contains NaN or infinity.")
75
+ return X_arr, y_arr
76
+
77
+
78
+ def check_is_fitted(estimator: "BaseEstimator", attribute: str) -> None:
79
+ """Raise :class:`NotFittedError` if ``estimator`` lacks ``attribute``."""
80
+ if getattr(estimator, attribute, None) is None:
81
+ raise NotFittedError(
82
+ f"This {type(estimator).__name__} instance is not fitted yet. "
83
+ "Call 'fit' with appropriate arguments first."
84
+ )
85
+
86
+
87
+ class BaseEstimator:
88
+ """Base class for all estimators.
89
+
90
+ Hyper-parameters are the keyword arguments of ``__init__``; they are stored
91
+ unchanged as attributes. Learned state uses a trailing underscore
92
+ (e.g. ``coef_``) and only exists after ``fit``.
93
+ """
94
+
95
+ @classmethod
96
+ def _param_names(cls) -> list:
97
+ sig = inspect.signature(cls.__init__)
98
+ return [
99
+ name
100
+ for name, p in sig.parameters.items()
101
+ if name != "self" and p.kind not in (p.VAR_POSITIONAL, p.VAR_KEYWORD)
102
+ ]
103
+
104
+ def get_params(self) -> dict:
105
+ """Return the estimator's hyper-parameters as a dict."""
106
+ return {name: getattr(self, name) for name in self._param_names()}
107
+
108
+ def set_params(self, **params: Any) -> "BaseEstimator":
109
+ """Set hyper-parameters and return ``self``.
110
+
111
+ Raises:
112
+ ValueError: If an unknown parameter name is given.
113
+ """
114
+ valid = set(self._param_names())
115
+ for key, value in params.items():
116
+ if key not in valid:
117
+ raise ValueError(f"Invalid parameter {key!r} for {type(self).__name__}.")
118
+ setattr(self, key, value)
119
+ return self
120
+
121
+ def __repr__(self) -> str:
122
+ args = ", ".join(f"{k}={v!r}" for k, v in self.get_params().items())
123
+ return f"{type(self).__name__}({args})"
124
+
125
+
126
+ def clone(estimator: BaseEstimator) -> BaseEstimator:
127
+ """Return a new, unfitted estimator with the same hyper-parameters."""
128
+ return type(estimator)(**estimator.get_params())
129
+
130
+
131
+ class ClassifierMixin:
132
+ """Adds :meth:`score` returning mean accuracy."""
133
+
134
+ def score(self, X: ArrayLike, y: ArrayLike) -> float:
135
+ """Return the accuracy of ``self.predict(X)`` against ``y``."""
136
+ from .metrics import accuracy_score
137
+
138
+ return accuracy_score(y, self.predict(X)) # type: ignore[attr-defined]
139
+
140
+
141
+ class RegressorMixin:
142
+ """Adds :meth:`score` returning the coefficient of determination R^2."""
143
+
144
+ def score(self, X: ArrayLike, y: ArrayLike) -> float:
145
+ """Return the R^2 of ``self.predict(X)`` against ``y``."""
146
+ from .metrics import r2_score
147
+
148
+ return r2_score(y, self.predict(X)) # type: ignore[attr-defined]
149
+
150
+
151
+ class TransformerMixin:
152
+ """Adds :meth:`fit_transform`."""
153
+
154
+ def fit_transform(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> np.ndarray:
155
+ """Fit to ``X`` and return the transformed data."""
156
+ return self.fit(X, y).transform(X) # type: ignore[attr-defined]
157
+
158
+
159
+ class ClusterMixin:
160
+ """Adds :meth:`fit_predict`."""
161
+
162
+ def fit_predict(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> np.ndarray:
163
+ """Fit to ``X`` and return the cluster label of each sample."""
164
+ self.fit(X) # type: ignore[attr-defined]
165
+ return self.labels_ # type: ignore[attr-defined]