mlkit 0.0.1__tar.gz → 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mlkit-0.1.0/LICENSE +21 -0
- mlkit-0.1.0/PKG-INFO +191 -0
- mlkit-0.1.0/README.md +162 -0
- mlkit-0.1.0/pyproject.toml +44 -0
- {mlkit-0.0.1 → mlkit-0.1.0}/setup.cfg +4 -4
- mlkit-0.1.0/src/mlkit/__init__.py +35 -0
- mlkit-0.1.0/src/mlkit/base.py +165 -0
- mlkit-0.1.0/src/mlkit/cluster.py +108 -0
- mlkit-0.1.0/src/mlkit/datasets.py +73 -0
- mlkit-0.1.0/src/mlkit/linear_model.py +191 -0
- mlkit-0.1.0/src/mlkit/metrics.py +139 -0
- mlkit-0.1.0/src/mlkit/model_selection.py +117 -0
- mlkit-0.1.0/src/mlkit/neighbors.py +132 -0
- mlkit-0.1.0/src/mlkit/preprocessing.py +111 -0
- mlkit-0.1.0/src/mlkit.egg-info/PKG-INFO +191 -0
- mlkit-0.1.0/src/mlkit.egg-info/SOURCES.txt +24 -0
- mlkit-0.1.0/src/mlkit.egg-info/requires.txt +4 -0
- mlkit-0.1.0/tests/test_cluster.py +63 -0
- mlkit-0.1.0/tests/test_linear_model.py +87 -0
- mlkit-0.1.0/tests/test_metrics.py +66 -0
- mlkit-0.1.0/tests/test_model_selection_and_base.py +97 -0
- mlkit-0.1.0/tests/test_neighbors.py +62 -0
- mlkit-0.1.0/tests/test_preprocessing.py +51 -0
- mlkit-0.0.1/PKG-INFO +0 -10
- mlkit-0.0.1/mlkit.egg-info/PKG-INFO +0 -10
- mlkit-0.0.1/mlkit.egg-info/SOURCES.txt +0 -6
- mlkit-0.0.1/setup.py +0 -17
- /mlkit-0.0.1/mlkit/__init__.py → /mlkit-0.1.0/src/mlkit/py.typed +0 -0
- {mlkit-0.0.1 → mlkit-0.1.0/src}/mlkit.egg-info/dependency_links.txt +0 -0
- {mlkit-0.0.1 → mlkit-0.1.0/src}/mlkit.egg-info/top_level.txt +0 -0
mlkit-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 nehz
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
mlkit-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mlkit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Machine Learning Toolkit: a small, readable NumPy toolkit with a fit/predict API, classic models, preprocessing and metrics.
|
|
5
|
+
Author: nehz
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: machine learning,regression,classification,clustering,numpy
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Intended Audience :: Education
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Python: >=3.9
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: numpy>=1.22
|
|
26
|
+
Provides-Extra: test
|
|
27
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
28
|
+
Dynamic: license-file
|
|
29
|
+
|
|
30
|
+
# mlkit
|
|
31
|
+
|
|
32
|
+
**Machine Learning Toolkit**: a small, readable machine learning library built on NumPy.
|
|
33
|
+
|
|
34
|
+
mlkit has the familiar `fit` / `predict` / `transform` estimator interface and a
|
|
35
|
+
handful of classic algorithms, each implemented in a few dozen lines of plain
|
|
36
|
+
NumPy. Use it to learn how the algorithms work, to teach them, or for small
|
|
37
|
+
projects where a large ML stack is too much. NumPy is its only dependency.
|
|
38
|
+
|
|
39
|
+
## Features
|
|
40
|
+
|
|
41
|
+
- **Estimator API**: `fit`, `predict`, `transform`, `score`, `get_params` / `set_params`, `clone`
|
|
42
|
+
- **Linear models**: `LinearRegression` (least squares), `Ridge` (L2, closed form), `LogisticRegression` (multinomial, gradient descent, optional L2)
|
|
43
|
+
- **Neighbours**: `KNeighborsClassifier`, `KNeighborsRegressor` (uniform or inverse-distance weights)
|
|
44
|
+
- **Clustering**: `KMeans` (k-means++ seeding, multiple restarts, reproducible via `random_state`)
|
|
45
|
+
- **Preprocessing**: `StandardScaler`, `MinMaxScaler` (both with `inverse_transform`)
|
|
46
|
+
- **Model selection**: `train_test_split`, `KFold`, `cross_val_score`
|
|
47
|
+
- **Metrics**: accuracy, confusion matrix, precision / recall / F1 (binary and macro), MSE, MAE, R²
|
|
48
|
+
- **Datasets**: `make_blobs`, `make_regression` synthetic generators
|
|
49
|
+
- Input validation with clear errors, and `NotFittedError` when an estimator is used before `fit`
|
|
50
|
+
- Type hints throughout (ships `py.typed`)
|
|
51
|
+
|
|
52
|
+
## Installation
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install mlkit
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
From a source checkout:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
python3 -m venv .venv
|
|
62
|
+
.venv/bin/pip install -e ".[test]"
|
|
63
|
+
.venv/bin/python -m pytest
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Requires Python 3.9+ and NumPy 1.22+.
|
|
67
|
+
|
|
68
|
+
## Quickstart
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from mlkit import KMeans, LinearRegression, LogisticRegression, StandardScaler, train_test_split
|
|
72
|
+
from mlkit.datasets import make_blobs, make_regression
|
|
73
|
+
from mlkit.metrics import accuracy_score, r2_score
|
|
74
|
+
|
|
75
|
+
# Classification: scale the features, then fit a logistic regression
|
|
76
|
+
X, y = make_blobs(n_samples=300, centers=3, random_state=0)
|
|
77
|
+
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=0)
|
|
78
|
+
|
|
79
|
+
scaler = StandardScaler().fit(X_train)
|
|
80
|
+
clf = LogisticRegression(learning_rate=0.5, max_iter=2000)
|
|
81
|
+
clf.fit(scaler.transform(X_train), y_train)
|
|
82
|
+
print("accuracy:", accuracy_score(y_test, clf.predict(scaler.transform(X_test))))
|
|
83
|
+
|
|
84
|
+
# Regression
|
|
85
|
+
X, y, true_coef = make_regression(n_samples=200, n_features=3, noise=0.5, random_state=1)
|
|
86
|
+
reg = LinearRegression().fit(X, y)
|
|
87
|
+
print("R^2:", r2_score(y, reg.predict(X)), "coef:", reg.coef_)
|
|
88
|
+
|
|
89
|
+
# Clustering
|
|
90
|
+
km = KMeans(n_clusters=3, random_state=0).fit(X_train)
|
|
91
|
+
print("inertia:", km.inertia_, "labels:", km.labels_[:10])
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Cross-validation:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from mlkit import KFold, KNeighborsClassifier, cross_val_score
|
|
98
|
+
|
|
99
|
+
scores = cross_val_score(KNeighborsClassifier(n_neighbors=5), X_train, y_train,
|
|
100
|
+
cv=KFold(n_splits=5, shuffle=True, random_state=0))
|
|
101
|
+
print(scores.mean())
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## API overview
|
|
105
|
+
|
|
106
|
+
All estimators subclass `mlkit.BaseEstimator`. Hyper-parameters are the
|
|
107
|
+
constructor arguments. Learned attributes end in `_` and exist only after
|
|
108
|
+
`fit`. Every `fit` returns `self`. Calling `predict` or `transform` before `fit`
|
|
109
|
+
raises `mlkit.NotFittedError`.
|
|
110
|
+
|
|
111
|
+
### `mlkit` (top level)
|
|
112
|
+
|
|
113
|
+
| Name | Description |
|
|
114
|
+
| --- | --- |
|
|
115
|
+
| `BaseEstimator` | Base class with `get_params()`, `set_params(**params)`, and a readable `repr` |
|
|
116
|
+
| `NotFittedError` | Raised when an estimator is used before `fit` |
|
|
117
|
+
| `clone(estimator)` | New unfitted estimator with the same hyper-parameters |
|
|
118
|
+
| `__version__` | `"0.1.0"` |
|
|
119
|
+
|
|
120
|
+
### `mlkit.linear_model`
|
|
121
|
+
|
|
122
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
123
|
+
| --- | --- | --- | --- |
|
|
124
|
+
| `LinearRegression` | `fit_intercept=True` | `fit(X, y)`, `predict(X)`, `score(X, y)` (R²) | `coef_`, `intercept_` |
|
|
125
|
+
| `Ridge` | `alpha=1.0`, `fit_intercept=True` | `fit`, `predict`, `score` (R²) | `coef_`, `intercept_` |
|
|
126
|
+
| `LogisticRegression` | `learning_rate=0.1`, `max_iter=1000`, `tol=1e-6`, `alpha=0.0`, `fit_intercept=True` | `fit`, `predict`, `predict_proba`, `decision_function`, `score` (accuracy) | `classes_`, `coef_` (n_classes × n_features), `intercept_`, `n_iter_` |
|
|
127
|
+
|
|
128
|
+
### `mlkit.neighbors`
|
|
129
|
+
|
|
130
|
+
| Name | Parameters | Methods / notes |
|
|
131
|
+
| --- | --- | --- |
|
|
132
|
+
| `KNeighborsClassifier` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `predict_proba`, `kneighbors(X)`, `score` (accuracy); `classes_` |
|
|
133
|
+
| `KNeighborsRegressor` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `kneighbors(X)`, `score` (R²) |
|
|
134
|
+
| `pairwise_distances(A, B)` | | Euclidean distance matrix of shape `(len(A), len(B))` |
|
|
135
|
+
|
|
136
|
+
### `mlkit.cluster`
|
|
137
|
+
|
|
138
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
139
|
+
| --- | --- | --- | --- |
|
|
140
|
+
| `KMeans` | `n_clusters=8`, `n_init=10`, `max_iter=300`, `tol=1e-6`, `random_state=None` | `fit(X)`, `predict(X)`, `fit_predict(X)`, `transform(X)` (distances to centroids) | `cluster_centers_`, `labels_`, `inertia_`, `n_iter_` |
|
|
141
|
+
|
|
142
|
+
### `mlkit.preprocessing`
|
|
143
|
+
|
|
144
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
145
|
+
| --- | --- | --- | --- |
|
|
146
|
+
| `StandardScaler` | `with_mean=True`, `with_std=True` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `mean_`, `scale_` |
|
|
147
|
+
| `MinMaxScaler` | `feature_range=(0.0, 1.0)` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `data_min_`, `data_max_` |
|
|
148
|
+
|
|
149
|
+
### `mlkit.model_selection`
|
|
150
|
+
|
|
151
|
+
| Name | Signature |
|
|
152
|
+
| --- | --- |
|
|
153
|
+
| `train_test_split` | `train_test_split(*arrays, test_size=0.25, shuffle=True, random_state=None)` returns `[a_train, a_test, b_train, b_test, ...]` |
|
|
154
|
+
| `KFold` | `KFold(n_splits=5, shuffle=False, random_state=None)`; `.split(X)` yields `(train_idx, test_idx)` |
|
|
155
|
+
| `cross_val_score` | `cross_val_score(estimator, X, y, *, cv=5, scoring=None)` returns an array of per-fold scores (`scoring` is `metric(y_true, y_pred)`; it defaults to `estimator.score`) |
|
|
156
|
+
|
|
157
|
+
### `mlkit.metrics`
|
|
158
|
+
|
|
159
|
+
| Function | Notes |
|
|
160
|
+
| --- | --- |
|
|
161
|
+
| `accuracy_score(y_true, y_pred)` | Fraction of exact matches |
|
|
162
|
+
| `confusion_matrix(y_true, y_pred, labels=None)` | `C[i, j]`: true `labels[i]` predicted as `labels[j]` |
|
|
163
|
+
| `precision_score(y_true, y_pred, *, average="binary", pos_label=1)` | `average` is `"binary"` or `"macro"`. A zero denominator gives 0.0 |
|
|
164
|
+
| `recall_score(...)` | Same arguments as `precision_score` |
|
|
165
|
+
| `f1_score(...)` | Same arguments as `precision_score` |
|
|
166
|
+
| `mean_squared_error(y_true, y_pred)` | |
|
|
167
|
+
| `mean_absolute_error(y_true, y_pred)` | |
|
|
168
|
+
| `r2_score(y_true, y_pred)` | Coefficient of determination |
|
|
169
|
+
|
|
170
|
+
### `mlkit.datasets`
|
|
171
|
+
|
|
172
|
+
| Function | Returns |
|
|
173
|
+
| --- | --- |
|
|
174
|
+
| `make_blobs(n_samples=100, n_features=2, centers=3, cluster_std=1.0, random_state=None)` | `(X, y)` |
|
|
175
|
+
| `make_regression(n_samples=100, n_features=3, noise=0.0, bias=0.0, random_state=None)` | `(X, y, coef)` |
|
|
176
|
+
|
|
177
|
+
### `mlkit.base`
|
|
178
|
+
|
|
179
|
+
Contains the `ClassifierMixin`, `RegressorMixin`, `TransformerMixin` and
|
|
180
|
+
`ClusterMixin` mixins for writing your own estimators, plus the validation
|
|
181
|
+
helpers `check_array`, `check_X_y` and `check_is_fitted`.
|
|
182
|
+
|
|
183
|
+
## Scope and limitations
|
|
184
|
+
|
|
185
|
+
mlkit puts clarity ahead of speed. k-NN uses brute-force distances, and
|
|
186
|
+
logistic regression uses plain full-batch gradient descent, so scale your
|
|
187
|
+
features first. It is not a replacement for scikit-learn on large datasets.
|
|
188
|
+
|
|
189
|
+
## License
|
|
190
|
+
|
|
191
|
+
MIT
|
mlkit-0.1.0/README.md
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
# mlkit
|
|
2
|
+
|
|
3
|
+
**Machine Learning Toolkit**: a small, readable machine learning library built on NumPy.
|
|
4
|
+
|
|
5
|
+
mlkit has the familiar `fit` / `predict` / `transform` estimator interface and a
|
|
6
|
+
handful of classic algorithms, each implemented in a few dozen lines of plain
|
|
7
|
+
NumPy. Use it to learn how the algorithms work, to teach them, or for small
|
|
8
|
+
projects where a large ML stack is too much. NumPy is its only dependency.
|
|
9
|
+
|
|
10
|
+
## Features
|
|
11
|
+
|
|
12
|
+
- **Estimator API**: `fit`, `predict`, `transform`, `score`, `get_params` / `set_params`, `clone`
|
|
13
|
+
- **Linear models**: `LinearRegression` (least squares), `Ridge` (L2, closed form), `LogisticRegression` (multinomial, gradient descent, optional L2)
|
|
14
|
+
- **Neighbours**: `KNeighborsClassifier`, `KNeighborsRegressor` (uniform or inverse-distance weights)
|
|
15
|
+
- **Clustering**: `KMeans` (k-means++ seeding, multiple restarts, reproducible via `random_state`)
|
|
16
|
+
- **Preprocessing**: `StandardScaler`, `MinMaxScaler` (both with `inverse_transform`)
|
|
17
|
+
- **Model selection**: `train_test_split`, `KFold`, `cross_val_score`
|
|
18
|
+
- **Metrics**: accuracy, confusion matrix, precision / recall / F1 (binary and macro), MSE, MAE, R²
|
|
19
|
+
- **Datasets**: `make_blobs`, `make_regression` synthetic generators
|
|
20
|
+
- Input validation with clear errors, and `NotFittedError` when an estimator is used before `fit`
|
|
21
|
+
- Type hints throughout (ships `py.typed`)
|
|
22
|
+
|
|
23
|
+
## Installation
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install mlkit
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
From a source checkout:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
python3 -m venv .venv
|
|
33
|
+
.venv/bin/pip install -e ".[test]"
|
|
34
|
+
.venv/bin/python -m pytest
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Requires Python 3.9+ and NumPy 1.22+.
|
|
38
|
+
|
|
39
|
+
## Quickstart
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from mlkit import KMeans, LinearRegression, LogisticRegression, StandardScaler, train_test_split
|
|
43
|
+
from mlkit.datasets import make_blobs, make_regression
|
|
44
|
+
from mlkit.metrics import accuracy_score, r2_score
|
|
45
|
+
|
|
46
|
+
# Classification: scale the features, then fit a logistic regression
|
|
47
|
+
X, y = make_blobs(n_samples=300, centers=3, random_state=0)
|
|
48
|
+
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=0)
|
|
49
|
+
|
|
50
|
+
scaler = StandardScaler().fit(X_train)
|
|
51
|
+
clf = LogisticRegression(learning_rate=0.5, max_iter=2000)
|
|
52
|
+
clf.fit(scaler.transform(X_train), y_train)
|
|
53
|
+
print("accuracy:", accuracy_score(y_test, clf.predict(scaler.transform(X_test))))
|
|
54
|
+
|
|
55
|
+
# Regression
|
|
56
|
+
X, y, true_coef = make_regression(n_samples=200, n_features=3, noise=0.5, random_state=1)
|
|
57
|
+
reg = LinearRegression().fit(X, y)
|
|
58
|
+
print("R^2:", r2_score(y, reg.predict(X)), "coef:", reg.coef_)
|
|
59
|
+
|
|
60
|
+
# Clustering
|
|
61
|
+
km = KMeans(n_clusters=3, random_state=0).fit(X_train)
|
|
62
|
+
print("inertia:", km.inertia_, "labels:", km.labels_[:10])
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Cross-validation:
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
from mlkit import KFold, KNeighborsClassifier, cross_val_score
|
|
69
|
+
|
|
70
|
+
scores = cross_val_score(KNeighborsClassifier(n_neighbors=5), X_train, y_train,
|
|
71
|
+
cv=KFold(n_splits=5, shuffle=True, random_state=0))
|
|
72
|
+
print(scores.mean())
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## API overview
|
|
76
|
+
|
|
77
|
+
All estimators subclass `mlkit.BaseEstimator`. Hyper-parameters are the
|
|
78
|
+
constructor arguments. Learned attributes end in `_` and exist only after
|
|
79
|
+
`fit`. Every `fit` returns `self`. Calling `predict` or `transform` before `fit`
|
|
80
|
+
raises `mlkit.NotFittedError`.
|
|
81
|
+
|
|
82
|
+
### `mlkit` (top level)
|
|
83
|
+
|
|
84
|
+
| Name | Description |
|
|
85
|
+
| --- | --- |
|
|
86
|
+
| `BaseEstimator` | Base class with `get_params()`, `set_params(**params)`, and a readable `repr` |
|
|
87
|
+
| `NotFittedError` | Raised when an estimator is used before `fit` |
|
|
88
|
+
| `clone(estimator)` | New unfitted estimator with the same hyper-parameters |
|
|
89
|
+
| `__version__` | `"0.1.0"` |
|
|
90
|
+
|
|
91
|
+
### `mlkit.linear_model`
|
|
92
|
+
|
|
93
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
94
|
+
| --- | --- | --- | --- |
|
|
95
|
+
| `LinearRegression` | `fit_intercept=True` | `fit(X, y)`, `predict(X)`, `score(X, y)` (R²) | `coef_`, `intercept_` |
|
|
96
|
+
| `Ridge` | `alpha=1.0`, `fit_intercept=True` | `fit`, `predict`, `score` (R²) | `coef_`, `intercept_` |
|
|
97
|
+
| `LogisticRegression` | `learning_rate=0.1`, `max_iter=1000`, `tol=1e-6`, `alpha=0.0`, `fit_intercept=True` | `fit`, `predict`, `predict_proba`, `decision_function`, `score` (accuracy) | `classes_`, `coef_` (n_classes × n_features), `intercept_`, `n_iter_` |
|
|
98
|
+
|
|
99
|
+
### `mlkit.neighbors`
|
|
100
|
+
|
|
101
|
+
| Name | Parameters | Methods / notes |
|
|
102
|
+
| --- | --- | --- |
|
|
103
|
+
| `KNeighborsClassifier` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `predict_proba`, `kneighbors(X)`, `score` (accuracy); `classes_` |
|
|
104
|
+
| `KNeighborsRegressor` | `n_neighbors=5`, `weights="uniform"` \| `"distance"` | `fit`, `predict`, `kneighbors(X)`, `score` (R²) |
|
|
105
|
+
| `pairwise_distances(A, B)` | | Euclidean distance matrix of shape `(len(A), len(B))` |
|
|
106
|
+
|
|
107
|
+
### `mlkit.cluster`
|
|
108
|
+
|
|
109
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
110
|
+
| --- | --- | --- | --- |
|
|
111
|
+
| `KMeans` | `n_clusters=8`, `n_init=10`, `max_iter=300`, `tol=1e-6`, `random_state=None` | `fit(X)`, `predict(X)`, `fit_predict(X)`, `transform(X)` (distances to centroids) | `cluster_centers_`, `labels_`, `inertia_`, `n_iter_` |
|
|
112
|
+
|
|
113
|
+
### `mlkit.preprocessing`
|
|
114
|
+
|
|
115
|
+
| Class | Parameters | Methods | Fitted attributes |
|
|
116
|
+
| --- | --- | --- | --- |
|
|
117
|
+
| `StandardScaler` | `with_mean=True`, `with_std=True` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `mean_`, `scale_` |
|
|
118
|
+
| `MinMaxScaler` | `feature_range=(0.0, 1.0)` | `fit`, `transform`, `fit_transform`, `inverse_transform` | `data_min_`, `data_max_` |
|
|
119
|
+
|
|
120
|
+
### `mlkit.model_selection`
|
|
121
|
+
|
|
122
|
+
| Name | Signature |
|
|
123
|
+
| --- | --- |
|
|
124
|
+
| `train_test_split` | `train_test_split(*arrays, test_size=0.25, shuffle=True, random_state=None)` returns `[a_train, a_test, b_train, b_test, ...]` |
|
|
125
|
+
| `KFold` | `KFold(n_splits=5, shuffle=False, random_state=None)`; `.split(X)` yields `(train_idx, test_idx)` |
|
|
126
|
+
| `cross_val_score` | `cross_val_score(estimator, X, y, *, cv=5, scoring=None)` returns an array of per-fold scores (`scoring` is `metric(y_true, y_pred)`; it defaults to `estimator.score`) |
|
|
127
|
+
|
|
128
|
+
### `mlkit.metrics`
|
|
129
|
+
|
|
130
|
+
| Function | Notes |
|
|
131
|
+
| --- | --- |
|
|
132
|
+
| `accuracy_score(y_true, y_pred)` | Fraction of exact matches |
|
|
133
|
+
| `confusion_matrix(y_true, y_pred, labels=None)` | `C[i, j]`: true `labels[i]` predicted as `labels[j]` |
|
|
134
|
+
| `precision_score(y_true, y_pred, *, average="binary", pos_label=1)` | `average` is `"binary"` or `"macro"`. A zero denominator gives 0.0 |
|
|
135
|
+
| `recall_score(...)` | Same arguments as `precision_score` |
|
|
136
|
+
| `f1_score(...)` | Same arguments as `precision_score` |
|
|
137
|
+
| `mean_squared_error(y_true, y_pred)` | |
|
|
138
|
+
| `mean_absolute_error(y_true, y_pred)` | |
|
|
139
|
+
| `r2_score(y_true, y_pred)` | Coefficient of determination |
|
|
140
|
+
|
|
141
|
+
### `mlkit.datasets`
|
|
142
|
+
|
|
143
|
+
| Function | Returns |
|
|
144
|
+
| --- | --- |
|
|
145
|
+
| `make_blobs(n_samples=100, n_features=2, centers=3, cluster_std=1.0, random_state=None)` | `(X, y)` |
|
|
146
|
+
| `make_regression(n_samples=100, n_features=3, noise=0.0, bias=0.0, random_state=None)` | `(X, y, coef)` |
|
|
147
|
+
|
|
148
|
+
### `mlkit.base`
|
|
149
|
+
|
|
150
|
+
Contains the `ClassifierMixin`, `RegressorMixin`, `TransformerMixin` and
|
|
151
|
+
`ClusterMixin` mixins for writing your own estimators, plus the validation
|
|
152
|
+
helpers `check_array`, `check_X_y` and `check_is_fitted`.
|
|
153
|
+
|
|
154
|
+
## Scope and limitations
|
|
155
|
+
|
|
156
|
+
mlkit puts clarity ahead of speed. k-NN uses brute-force distances, and
|
|
157
|
+
logistic regression uses plain full-batch gradient descent, so scale your
|
|
158
|
+
features first. It is not a replacement for scikit-learn on large datasets.
|
|
159
|
+
|
|
160
|
+
## License
|
|
161
|
+
|
|
162
|
+
MIT
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "mlkit"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Machine Learning Toolkit: a small, readable NumPy toolkit with a fit/predict API, classic models, preprocessing and metrics."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "nehz" }]
|
|
14
|
+
keywords = ["machine learning", "regression", "classification", "clustering", "numpy"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 3 - Alpha",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Intended Audience :: Education",
|
|
19
|
+
"Intended Audience :: Science/Research",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
23
|
+
"Programming Language :: Python :: 3.9",
|
|
24
|
+
"Programming Language :: Python :: 3.10",
|
|
25
|
+
"Programming Language :: Python :: 3.11",
|
|
26
|
+
"Programming Language :: Python :: 3.12",
|
|
27
|
+
"Programming Language :: Python :: 3.13",
|
|
28
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
29
|
+
"Typing :: Typed",
|
|
30
|
+
]
|
|
31
|
+
dependencies = ["numpy>=1.22"]
|
|
32
|
+
|
|
33
|
+
[project.optional-dependencies]
|
|
34
|
+
test = ["pytest>=7"]
|
|
35
|
+
|
|
36
|
+
[tool.setuptools.packages.find]
|
|
37
|
+
where = ["src"]
|
|
38
|
+
|
|
39
|
+
[tool.setuptools.package-data]
|
|
40
|
+
mlkit = ["py.typed"]
|
|
41
|
+
|
|
42
|
+
[tool.pytest.ini_options]
|
|
43
|
+
testpaths = ["tests"]
|
|
44
|
+
addopts = "-q"
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
[egg_info]
|
|
2
|
-
tag_build =
|
|
3
|
-
tag_date = 0
|
|
4
|
-
|
|
1
|
+
[egg_info]
|
|
2
|
+
tag_build =
|
|
3
|
+
tag_date = 0
|
|
4
|
+
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""mlkit - Machine Learning Toolkit.
|
|
2
|
+
|
|
3
|
+
A small, readable NumPy toolkit with a scikit-learn style ``fit``/``predict``
|
|
4
|
+
interface, classic models, preprocessing, model selection and metrics.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from . import datasets, metrics
|
|
8
|
+
from .base import BaseEstimator, NotFittedError, clone
|
|
9
|
+
from .cluster import KMeans
|
|
10
|
+
from .linear_model import LinearRegression, LogisticRegression, Ridge
|
|
11
|
+
from .model_selection import KFold, cross_val_score, train_test_split
|
|
12
|
+
from .neighbors import KNeighborsClassifier, KNeighborsRegressor
|
|
13
|
+
from .preprocessing import MinMaxScaler, StandardScaler
|
|
14
|
+
|
|
15
|
+
__version__ = "0.1.0"
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"BaseEstimator",
|
|
19
|
+
"NotFittedError",
|
|
20
|
+
"clone",
|
|
21
|
+
"LinearRegression",
|
|
22
|
+
"Ridge",
|
|
23
|
+
"LogisticRegression",
|
|
24
|
+
"KNeighborsClassifier",
|
|
25
|
+
"KNeighborsRegressor",
|
|
26
|
+
"KMeans",
|
|
27
|
+
"StandardScaler",
|
|
28
|
+
"MinMaxScaler",
|
|
29
|
+
"train_test_split",
|
|
30
|
+
"KFold",
|
|
31
|
+
"cross_val_score",
|
|
32
|
+
"datasets",
|
|
33
|
+
"metrics",
|
|
34
|
+
"__version__",
|
|
35
|
+
]
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""Core estimator interface, mixins and input validation helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import inspect
|
|
6
|
+
from typing import Any, Optional, Tuple
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
from numpy.typing import ArrayLike
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"BaseEstimator",
|
|
13
|
+
"ClassifierMixin",
|
|
14
|
+
"RegressorMixin",
|
|
15
|
+
"TransformerMixin",
|
|
16
|
+
"ClusterMixin",
|
|
17
|
+
"NotFittedError",
|
|
18
|
+
"check_array",
|
|
19
|
+
"check_X_y",
|
|
20
|
+
"check_is_fitted",
|
|
21
|
+
"clone",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class NotFittedError(RuntimeError):
|
|
26
|
+
"""Raised when ``predict``/``transform`` is called before ``fit``."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def check_array(X: ArrayLike, *, ensure_2d: bool = True) -> np.ndarray:
|
|
30
|
+
"""Convert ``X`` to a finite float array.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
X: Array-like input.
|
|
34
|
+
ensure_2d: If True, a 2-D array of shape ``(n_samples, n_features)``
|
|
35
|
+
is required.
|
|
36
|
+
|
|
37
|
+
Returns:
|
|
38
|
+
A ``float64`` NumPy array.
|
|
39
|
+
|
|
40
|
+
Raises:
|
|
41
|
+
ValueError: If ``X`` is empty, has the wrong number of dimensions or
|
|
42
|
+
contains NaN/inf values.
|
|
43
|
+
"""
|
|
44
|
+
arr = np.asarray(X, dtype=float)
|
|
45
|
+
if ensure_2d and arr.ndim != 2:
|
|
46
|
+
raise ValueError(f"Expected a 2-D array, got an array with ndim={arr.ndim}.")
|
|
47
|
+
if arr.size == 0:
|
|
48
|
+
raise ValueError("Input array is empty.")
|
|
49
|
+
if not np.all(np.isfinite(arr)):
|
|
50
|
+
raise ValueError("Input contains NaN or infinity.")
|
|
51
|
+
return arr
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def check_X_y(X: ArrayLike, y: ArrayLike, *, y_numeric: bool = False) -> Tuple[np.ndarray, np.ndarray]:
|
|
55
|
+
"""Validate a feature matrix and a 1-D target vector of matching length.
|
|
56
|
+
|
|
57
|
+
Args:
|
|
58
|
+
X: Array-like of shape ``(n_samples, n_features)``.
|
|
59
|
+
y: Array-like of shape ``(n_samples,)``.
|
|
60
|
+
y_numeric: If True, ``y`` is converted to ``float64``.
|
|
61
|
+
|
|
62
|
+
Returns:
|
|
63
|
+
The validated ``(X, y)`` pair.
|
|
64
|
+
"""
|
|
65
|
+
X_arr = check_array(X)
|
|
66
|
+
y_arr = np.asarray(y, dtype=float if y_numeric else None)
|
|
67
|
+
if y_arr.ndim != 1:
|
|
68
|
+
raise ValueError(f"y must be 1-D, got ndim={y_arr.ndim}.")
|
|
69
|
+
if y_arr.shape[0] != X_arr.shape[0]:
|
|
70
|
+
raise ValueError(
|
|
71
|
+
f"X and y have inconsistent lengths: {X_arr.shape[0]} != {y_arr.shape[0]}."
|
|
72
|
+
)
|
|
73
|
+
if y_numeric and not np.all(np.isfinite(y_arr)):
|
|
74
|
+
raise ValueError("y contains NaN or infinity.")
|
|
75
|
+
return X_arr, y_arr
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def check_is_fitted(estimator: "BaseEstimator", attribute: str) -> None:
|
|
79
|
+
"""Raise :class:`NotFittedError` if ``estimator`` lacks ``attribute``."""
|
|
80
|
+
if getattr(estimator, attribute, None) is None:
|
|
81
|
+
raise NotFittedError(
|
|
82
|
+
f"This {type(estimator).__name__} instance is not fitted yet. "
|
|
83
|
+
"Call 'fit' with appropriate arguments first."
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class BaseEstimator:
|
|
88
|
+
"""Base class for all estimators.
|
|
89
|
+
|
|
90
|
+
Hyper-parameters are the keyword arguments of ``__init__``; they are stored
|
|
91
|
+
unchanged as attributes. Learned state uses a trailing underscore
|
|
92
|
+
(e.g. ``coef_``) and only exists after ``fit``.
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
@classmethod
|
|
96
|
+
def _param_names(cls) -> list:
|
|
97
|
+
sig = inspect.signature(cls.__init__)
|
|
98
|
+
return [
|
|
99
|
+
name
|
|
100
|
+
for name, p in sig.parameters.items()
|
|
101
|
+
if name != "self" and p.kind not in (p.VAR_POSITIONAL, p.VAR_KEYWORD)
|
|
102
|
+
]
|
|
103
|
+
|
|
104
|
+
def get_params(self) -> dict:
|
|
105
|
+
"""Return the estimator's hyper-parameters as a dict."""
|
|
106
|
+
return {name: getattr(self, name) for name in self._param_names()}
|
|
107
|
+
|
|
108
|
+
def set_params(self, **params: Any) -> "BaseEstimator":
|
|
109
|
+
"""Set hyper-parameters and return ``self``.
|
|
110
|
+
|
|
111
|
+
Raises:
|
|
112
|
+
ValueError: If an unknown parameter name is given.
|
|
113
|
+
"""
|
|
114
|
+
valid = set(self._param_names())
|
|
115
|
+
for key, value in params.items():
|
|
116
|
+
if key not in valid:
|
|
117
|
+
raise ValueError(f"Invalid parameter {key!r} for {type(self).__name__}.")
|
|
118
|
+
setattr(self, key, value)
|
|
119
|
+
return self
|
|
120
|
+
|
|
121
|
+
def __repr__(self) -> str:
|
|
122
|
+
args = ", ".join(f"{k}={v!r}" for k, v in self.get_params().items())
|
|
123
|
+
return f"{type(self).__name__}({args})"
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def clone(estimator: BaseEstimator) -> BaseEstimator:
|
|
127
|
+
"""Return a new, unfitted estimator with the same hyper-parameters."""
|
|
128
|
+
return type(estimator)(**estimator.get_params())
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class ClassifierMixin:
|
|
132
|
+
"""Adds :meth:`score` returning mean accuracy."""
|
|
133
|
+
|
|
134
|
+
def score(self, X: ArrayLike, y: ArrayLike) -> float:
|
|
135
|
+
"""Return the accuracy of ``self.predict(X)`` against ``y``."""
|
|
136
|
+
from .metrics import accuracy_score
|
|
137
|
+
|
|
138
|
+
return accuracy_score(y, self.predict(X)) # type: ignore[attr-defined]
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class RegressorMixin:
|
|
142
|
+
"""Adds :meth:`score` returning the coefficient of determination R^2."""
|
|
143
|
+
|
|
144
|
+
def score(self, X: ArrayLike, y: ArrayLike) -> float:
|
|
145
|
+
"""Return the R^2 of ``self.predict(X)`` against ``y``."""
|
|
146
|
+
from .metrics import r2_score
|
|
147
|
+
|
|
148
|
+
return r2_score(y, self.predict(X)) # type: ignore[attr-defined]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class TransformerMixin:
|
|
152
|
+
"""Adds :meth:`fit_transform`."""
|
|
153
|
+
|
|
154
|
+
def fit_transform(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> np.ndarray:
|
|
155
|
+
"""Fit to ``X`` and return the transformed data."""
|
|
156
|
+
return self.fit(X, y).transform(X) # type: ignore[attr-defined]
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class ClusterMixin:
|
|
160
|
+
"""Adds :meth:`fit_predict`."""
|
|
161
|
+
|
|
162
|
+
def fit_predict(self, X: ArrayLike, y: Optional[ArrayLike] = None) -> np.ndarray:
|
|
163
|
+
"""Fit to ``X`` and return the cluster label of each sample."""
|
|
164
|
+
self.fit(X) # type: ignore[attr-defined]
|
|
165
|
+
return self.labels_ # type: ignore[attr-defined]
|