millwright 0.1.1__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. millwright-0.2.0/CHANGELOG.md +118 -0
  2. millwright-0.2.0/CONTRIBUTING.md +55 -0
  3. {millwright-0.1.1 → millwright-0.2.0}/Cargo.lock +2 -1
  4. {millwright-0.1.1 → millwright-0.2.0}/Cargo.toml +8 -5
  5. millwright-0.2.0/GUIDE.md +41 -0
  6. {millwright-0.1.1 → millwright-0.2.0}/PKG-INFO +12 -3
  7. {millwright-0.1.1 → millwright-0.2.0}/README.md +11 -2
  8. {millwright-0.1.1 → millwright-0.2.0}/RELEASING.md +13 -1
  9. millwright-0.2.0/docs/data.html +114 -0
  10. millwright-0.2.0/docs/deploy.html +134 -0
  11. millwright-0.2.0/docs/index.html +156 -0
  12. millwright-0.2.0/docs/insight.html +102 -0
  13. millwright-0.2.0/docs/pipelines.html +187 -0
  14. millwright-0.2.0/docs/python.html +106 -0
  15. millwright-0.2.0/docs/site.css +152 -0
  16. millwright-0.2.0/guide.html +9 -0
  17. {millwright-0.1.1 → millwright-0.2.0}/index.html +30 -26
  18. {millwright-0.1.1 → millwright-0.2.0}/pyproject.toml +1 -1
  19. millwright-0.2.0/scripts/release.sh +54 -0
  20. {millwright-0.1.1 → millwright-0.2.0}/src/automl.rs +136 -14
  21. millwright-0.2.0/src/backends/smartcore.rs +624 -0
  22. {millwright-0.1.1 → millwright-0.2.0}/src/ensemble.rs +231 -9
  23. {millwright-0.1.1 → millwright-0.2.0}/src/lib.rs +2 -2
  24. {millwright-0.1.1 → millwright-0.2.0}/src/onnx.rs +56 -0
  25. {millwright-0.1.1 → millwright-0.2.0}/src/profile.rs +232 -15
  26. millwright-0.2.0/src/python.rs +1010 -0
  27. {millwright-0.1.1 → millwright-0.2.0}/src/selection/cv.rs +15 -10
  28. {millwright-0.1.1 → millwright-0.2.0}/src/selection/mod.rs +39 -0
  29. {millwright-0.1.1 → millwright-0.2.0}/src/selection/scoring.rs +35 -1
  30. {millwright-0.1.1 → millwright-0.2.0}/src/table.rs +24 -0
  31. {millwright-0.1.1 → millwright-0.2.0}/src/traits.rs +4 -4
  32. millwright-0.2.0/tests/ensemble_zoo.rs +42 -0
  33. millwright-0.1.1/CHANGELOG.md +0 -65
  34. millwright-0.1.1/GUIDE.md +0 -621
  35. millwright-0.1.1/guide.html +0 -574
  36. millwright-0.1.1/src/backends/smartcore.rs +0 -290
  37. millwright-0.1.1/src/python.rs +0 -179
  38. {millwright-0.1.1 → millwright-0.2.0}/.github/workflows/ci.yml +0 -0
  39. {millwright-0.1.1 → millwright-0.2.0}/.github/workflows/release-crate.yml +0 -0
  40. {millwright-0.1.1 → millwright-0.2.0}/.github/workflows/release-python.yml +0 -0
  41. {millwright-0.1.1 → millwright-0.2.0}/.gitignore +0 -0
  42. {millwright-0.1.1 → millwright-0.2.0}/CNAME +0 -0
  43. {millwright-0.1.1 → millwright-0.2.0}/LICENSE +0 -0
  44. {millwright-0.1.1 → millwright-0.2.0}/benches/throughput.rs +0 -0
  45. {millwright-0.1.1 → millwright-0.2.0}/examples/automl.rs +0 -0
  46. {millwright-0.1.1 → millwright-0.2.0}/examples/backends.rs +0 -0
  47. {millwright-0.1.1 → millwright-0.2.0}/examples/explore.rs +0 -0
  48. {millwright-0.1.1 → millwright-0.2.0}/examples/insight.rs +0 -0
  49. {millwright-0.1.1 → millwright-0.2.0}/examples/operations.rs +0 -0
  50. {millwright-0.1.1 → millwright-0.2.0}/examples/portability.rs +0 -0
  51. {millwright-0.1.1 → millwright-0.2.0}/examples/specialized.rs +0 -0
  52. {millwright-0.1.1 → millwright-0.2.0}/examples/spine.rs +0 -0
  53. {millwright-0.1.1 → millwright-0.2.0}/examples/trust.rs +0 -0
  54. {millwright-0.1.1 → millwright-0.2.0}/examples/workflow.rs +0 -0
  55. {millwright-0.1.1 → millwright-0.2.0}/src/anomaly.rs +0 -0
  56. {millwright-0.1.1 → millwright-0.2.0}/src/backends/chronos.rs +0 -0
  57. {millwright-0.1.1 → millwright-0.2.0}/src/backends/incremental.rs +0 -0
  58. {millwright-0.1.1 → millwright-0.2.0}/src/backends/linfa.rs +0 -0
  59. {millwright-0.1.1 → millwright-0.2.0}/src/backends/mod.rs +0 -0
  60. {millwright-0.1.1 → millwright-0.2.0}/src/balance.rs +0 -0
  61. {millwright-0.1.1 → millwright-0.2.0}/src/calibration.rs +0 -0
  62. {millwright-0.1.1 → millwright-0.2.0}/src/diagnostics.rs +0 -0
  63. {millwright-0.1.1 → millwright-0.2.0}/src/error.rs +0 -0
  64. {millwright-0.1.1 → millwright-0.2.0}/src/evaluate.rs +0 -0
  65. {millwright-0.1.1 → millwright-0.2.0}/src/explain.rs +0 -0
  66. {millwright-0.1.1 → millwright-0.2.0}/src/frame.rs +0 -0
  67. {millwright-0.1.1 → millwright-0.2.0}/src/logistic.rs +0 -0
  68. {millwright-0.1.1 → millwright-0.2.0}/src/monitor.rs +0 -0
  69. {millwright-0.1.1 → millwright-0.2.0}/src/pipeline.rs +0 -0
  70. {millwright-0.1.1 → millwright-0.2.0}/src/registry.rs +0 -0
  71. {millwright-0.1.1 → millwright-0.2.0}/src/rng.rs +0 -0
  72. {millwright-0.1.1 → millwright-0.2.0}/src/selection/search.rs +0 -0
  73. {millwright-0.1.1 → millwright-0.2.0}/src/serve.rs +0 -0
  74. {millwright-0.1.1 → millwright-0.2.0}/src/transform.rs +0 -0
  75. {millwright-0.1.1 → millwright-0.2.0}/src/viz.rs +0 -0
  76. {millwright-0.1.1 → millwright-0.2.0}/tests/golden.rs +0 -0
  77. {millwright-0.1.1 → millwright-0.2.0}/tests/real_data.rs +0 -0
@@ -0,0 +1,118 @@
1
+ # Changelog
2
+
3
+ All notable changes to Millwright are recorded here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/), and the project aims at
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [0.2.0] - 2026-08-23
10
+
11
+ ### Added
12
+ - **rayon parallelism.** Cross-validation folds and search candidates now
13
+ evaluate in parallel: `cross_val_score` is fold-parallel, `Bagging` fits its
14
+ base estimators in parallel, and `AutoML::parallel()` adds candidate-level
15
+ parallelism. The core contract traits gained `Send + Sync` bounds (every
16
+ concrete model already satisfied them). Results stay seed-reproducible.
17
+ - **`Boosting`** — SAMME adaptive boosting over any weak learner (an
18
+ `alpha`-weighted vote of models each reweighted toward the last round's
19
+ mistakes), joining `Voting`/`Bagging`/`Stacking`.
20
+ - **Three more models over the smartcore backend:** `Knn` (k-nearest-neighbours),
21
+ `Svc` (support vector classifier, linear or RBF, one-vs-one for multiclass),
22
+ and `NaiveBayes` (Gaussian). All implement the same `Estimator`/`Predictor`
23
+ contract, so they drop into pipelines, ensembles, and search unchanged — and
24
+ they are exposed to Python too (`mw.Knn`, `mw.Svc` / `mw.Svc.rbf()`,
25
+ `mw.NaiveBayes`, plus `pipe.knn()` / `pipe.svc()` / `pipe.naive_bayes()`).
26
+ - **Python: a scikit-learn-shaped object API.** `mw.Frame` with `from_pandas` /
27
+ `from_numpy` / `from_rows` ingest; composable transformer/estimator objects
28
+ (`StandardScaler`, `MinMaxScaler`, `SimpleImputer`, `OneHotEncoder`,
29
+ `RandomForest`, `LinearRegression`) added via `pipe.step(name, obj)` /
30
+ `pipe.estimator(name, obj)`; `fit`/`predict`/`evaluate` accept a `Frame`.
31
+ - **Python: `pipeline.explain(...)`** returns SHAP feature importance
32
+ (`Explainer.kernel()` configurable), and **`pipeline.export_onnx(path)`**
33
+ writes the fitted pipeline to a single ONNX file. The `python` wheel now
34
+ bundles the `model-selection`, `explain`, and `onnx` engines.
35
+ - **Python: `GridSearch` / `KFold` / `StratifiedKFold`** over a pipeline, with
36
+ a `SearchResult` (`best_score`, `best_params()`, `predict()`).
37
+ - **`InferenceModel` is now an `Estimator` + `Predictor`**, so a pre-trained
38
+ ONNX model (from scikit-learn, PyTorch, …) can be dropped into a `Pipeline`
39
+ as a frozen estimator behind Millwright's preprocessing — in Rust and, via
40
+ `mw.OnnxModel(path)`, from Python.
41
+ - **Python: `mw.Table` + `mw.Profile`** — dtype-aware CSV/Parquet ingest and
42
+ automated EDA (`Profile.of(table_or_frame).to_html(path)`). The wheel now
43
+ bundles the `eda` (polars) engine, so it is larger than the pure-model build.
44
+ - **`Table::from_frame`** (Rust): build a numeric `Table` from a `Frame`, so the
45
+ numeric world can round-trip back into the typed one (e.g. to profile it).
46
+ - **Richer EDA in `Profile`** — excess `kurtosis` and z-score outlier counts per
47
+ numeric column; a **Spearman** rank-correlation matrix beside the Pearson one;
48
+ a **co-missing** map (columns whose null patterns correlate); high-cardinality
49
+ categoricals now suggest `TargetEncoder`; and `suggest_pipeline` adds a
50
+ train-time **SMOTE** balancer on class imbalance (with `preprocessing`).
51
+ - **AutoML is seeded by EDA.** With the `eda` engine on, the search fixes its
52
+ preprocessing to `Profile::suggest_pipeline()` and varies only the model,
53
+ pruning the space (it falls back to a scaler sweep without `eda`).
54
+
55
+ ### Fixed
56
+ - **Cross-validated F1 no longer returns `NaN`.** A fold whose predictions
57
+ contain no true positives left smartcore's F1 evaluating `0/0`; a single NaN
58
+ fold poisoned the CV mean and a search's `best_score`. An undefined F1 is now
59
+ 0.0 (scikit-learn's `zero_division=0` convention), so `GridSearch(...).scoring(F1)`
60
+ yields finite, comparable scores.
61
+
62
+ ## [0.1.1]
63
+
64
+ ### Added
65
+
66
+ - **`LogisticRegression`** — a native, core binary classifier with genuine
67
+ `predict_proba`: the framework's first real `ProbaPredictor`.
68
+ - **`calibration` feature** — `PlattScaling`, `IsotonicRegression`,
69
+ `reliability_curve`, and `CalibratedClassifier`, which wraps any
70
+ `ProbaPredictor` and returns calibrated probabilities.
71
+ - **`anomaly` feature** — `Mahalanobis` and `KnnScore`, unified behind an
72
+ `OutlierDetector` trait.
73
+ - **`eda` feature** — a polars-backed, dtype-aware `Table` (CSV/Parquet ingest
74
+ that lowers to the numeric `Frame`) and a typed `Profile` with an HTML report,
75
+ actionable alerts, and `suggest_pipeline()`.
76
+ - **Transformers** — `Winsorize`, `PowerTransform` (Yeo-Johnson),
77
+ `ColumnTransformer`, and the supervised `TargetEncoder`.
78
+ - **Convenience** — `Frame::from_csv` (dependency-free numeric loader),
79
+ `Table::head`.
80
+ - **Python** — `min_max_scaler`, `simple_imputer`, `one_hot`,
81
+ `linear_regression`, and `evaluate()`.
82
+ - **Examples** — `explore` (ingest → profile → pipeline) and `trust`
83
+ (calibration → reliability → anomaly detection).
84
+ - **Benchmarks** — `benches/throughput.rs` (criterion): the boundary conversion
85
+ and core fit/predict, backing the "Rust speed" claim.
86
+ - **One-hot ingest** — `Table::to_frame_with` / `into_dataset_with` and a
87
+ `CategoryEncoding` enum: lower nominal categories to 0/1 indicator columns
88
+ instead of ordinal codes.
89
+ - **Schema-aware preprocessing** — `Frame` carries a per-column `Dtype`; `Table`
90
+ marks categoricals as it lowers; scalers / `Winsorize` / `PowerTransform` pass
91
+ categorical columns through untouched, and `OneHotEncoder` encodes by dtype
92
+ rather than a value heuristic when the schema is known.
93
+ - **Real-data validation** — an end-to-end integration test on Quinlan's
94
+ PlayTennis (`tests/real_data.rs`): CSV → profile → suggested pipeline → fit.
95
+
96
+ ### Changed
97
+
98
+ - Exact-version pins on every engine crate; `Cargo.lock` committed; a
99
+ feature-matrix CI (fmt, clippy `-D warnings`, docs, matrix, OS, examples,
100
+ benches, publish dry-run, wheel).
101
+ - `selection.rs` split into `selection/{scoring,cv,search}`.
102
+ - MSRV is **1.95** (dep-dictated — `sysinfo` via tract, and polars); enforced by
103
+ cargo via `rust-version` rather than a dedicated CI job (which would break on
104
+ every transitive bump). The default install needs 1.85.
105
+ - De-staled the crate and module docs (no more "Phase 0 · the spine").
106
+
107
+ ### Fixed
108
+
109
+ - Golden tests and the crate doctest build under every feature subset (they were
110
+ unconditionally referencing backend-gated types).
111
+ - Float sorts use `f64::total_cmp`, closing a NaN-driven panic class.
112
+
113
+ ## [0.1.0]
114
+
115
+ - Phases 0–8: the `Frame`/trait spine and smartcore backend, preprocessing and
116
+ model selection, a second backend (linfa) and HPO, evaluation/diagnostics/
117
+ explainability, ONNX export and inference, serving + drift monitoring + a model
118
+ registry, time-series and out-of-core estimators, AutoML, and 1.0 hardening.
@@ -0,0 +1,55 @@
1
+ # Contributing to Millwright
2
+
3
+ Thanks for your interest! Millwright assembles proven Rust crates into one
4
+ composable ML lifecycle. Bug reports, small fixes, and focused features are all
5
+ welcome.
6
+
7
+ ## Getting set up
8
+
9
+ ```bash
10
+ git clone https://github.com/mi7plus/millwright
11
+ cd millwright
12
+ cargo test # default features
13
+ ```
14
+
15
+ Rust **1.95+** is required (the floor is dictated by transitive engine deps).
16
+
17
+ ## Before you open a PR
18
+
19
+ The CI is a feature matrix, so run the same checks locally:
20
+
21
+ ```bash
22
+ cargo fmt --all --check
23
+ cargo clippy --features full --all-targets -- -D warnings
24
+ cargo test --features full
25
+ cargo test --no-default-features # the bare-core build must pass too
26
+ cargo doc --no-deps --features full # with RUSTDOCFLAGS="-D warnings"
27
+ ```
28
+
29
+ - **Every capability is a cargo feature.** New functionality behind a young
30
+ single-author crate goes behind its own feature; the core stays lean. See the
31
+ feature list in `Cargo.toml`.
32
+ - **Engines are pinned to exact versions** (`=x.y.z`) and `Cargo.lock` is
33
+ committed — bump them deliberately, one line, one commit.
34
+ - **Add tests.** Prefer a `#[test]` next to the code; lock numeric behaviour in
35
+ `tests/golden.rs` when it matters. Feature-gate tests that need a backend.
36
+ - **Run the examples** you touch: `cargo run --features full --example <name>`.
37
+
38
+ ## Try it out
39
+
40
+ - Examples: [`examples/`](examples) — one runnable program per feature group.
41
+ - Benchmarks: `cargo bench --features smartcore-backend`.
42
+ - Docs/tutorial: <https://millwright-rs.dev/docs/>.
43
+
44
+ ## Releasing
45
+
46
+ Maintainers: bump `Cargo.toml` + `pyproject.toml`, roll `CHANGELOG.md`, and tag
47
+ `vX.Y.Z` — the workflows publish to crates.io and PyPI via OIDC. The helper
48
+ `scripts/release.sh X.Y.Z` does the mechanical steps. See
49
+ [`RELEASING.md`](RELEASING.md).
50
+
51
+ ## Scope & conduct
52
+
53
+ Millwright is a thin facade over proven engines — it orchestrates, it doesn't
54
+ reimplement numerics. Please keep PRs focused, and be kind and constructive in
55
+ issues and reviews.
@@ -1935,7 +1935,7 @@ dependencies = [
1935
1935
 
1936
1936
  [[package]]
1937
1937
  name = "millwright"
1938
- version = "0.1.1"
1938
+ version = "0.2.0"
1939
1939
  dependencies = [
1940
1940
  "axum",
1941
1941
  "chronos-ts",
@@ -1957,6 +1957,7 @@ dependencies = [
1957
1957
  "plotters-statistical",
1958
1958
  "polars",
1959
1959
  "pyo3",
1960
+ "rayon",
1960
1961
  "regression-diagnostics",
1961
1962
  "serde",
1962
1963
  "serde_json",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "millwright"
3
- version = "0.1.1"
3
+ version = "0.2.0"
4
4
  edition = "2021"
5
5
  rust-version = "1.95"
6
6
  description = "A unified ML framework for Rust — proven Rust crates, assembled into one machine."
@@ -56,6 +56,7 @@ polars = { version = "=0.55.2", default-features = false, features = ["csv", "pa
56
56
  # --- general infrastructure: caret ranges, reproduced via Cargo.lock ---
57
57
  pyo3 = { version = "0.24", features = ["extension-module", "abi3-py39"], optional = true }
58
58
  ndarray = { version = "0.16", optional = true }
59
+ rayon = { version = "1", optional = true }
59
60
  plotters = { version = "0.3.7", default-features = false, features = ["svg_backend", "all_series"], optional = true }
60
61
  tract-onnx = { version = "0.23", optional = true }
61
62
  axum = { version = "0.8", optional = true }
@@ -90,11 +91,13 @@ smartcore-backend = ["dep:smartcore"]
90
91
  preprocessing = ["dep:imbalance-rs", "dep:ndarray"]
91
92
 
92
93
  # Cross-validation splitters, scoring, and grid/random search over a pipeline.
93
- model-selection = ["dep:model-selection-rs", "dep:ndarray"]
94
+ # CV folds and search candidates are evaluated in parallel via rayon.
95
+ model-selection = ["dep:model-selection-rs", "dep:ndarray", "dep:rayon"]
94
96
 
95
97
  # Voting / bagging / stacking meta-estimators, composed over the four traits.
96
- # Leak-free stacking uses the model-selection CV engine when it is enabled.
97
- ensemble = []
98
+ # Leak-free stacking uses the model-selection CV engine when it is enabled;
99
+ # bagging fits its base estimators in parallel via rayon.
100
+ ensemble = ["dep:rayon"]
98
101
 
99
102
  # Ingest & EDA — the front of the lifecycle. A polars-backed, dtype-aware
100
103
  # `Table` (CSV/Parquet in, real string/categorical/datetime/null columns) that
@@ -151,7 +154,7 @@ onnx = [
151
154
  # Deliberately NOT part of `full`: pyo3's `extension-module` defers libpython
152
155
  # symbols, so a plain `cargo test`/`clippy` cannot link a test binary with it.
153
156
  # Build & test this feature through maturin instead (see pyproject.toml).
154
- python = ["dep:pyo3", "smartcore-backend"]
157
+ python = ["dep:pyo3", "smartcore-backend", "model-selection", "explain", "onnx", "eda"]
155
158
 
156
159
  # Phase 5 · OPERATIONS — past where scikit-learn stops.
157
160
  # A versioned model registry: content-addressed ONNX artifact + metadata +
@@ -0,0 +1,41 @@
1
+ # The Millwright Guide
2
+
3
+ *A unified ML framework for Rust — ten crates, one lifecycle.*
4
+
5
+ The full hands-on tutorial now lives as a browsable, multi-page site:
6
+
7
+ ### → **[millwright-rs.dev/docs/](https://millwright-rs.dev/docs/)** (source in [`docs/`](docs/index.html))
8
+
9
+ It walks the whole lifecycle, one topic per page: **Data & EDA** · **Pipelines &
10
+ Models** · **Insight** (evaluate, explain, calibrate, detect) · **Deploy**
11
+ (ONNX, serving, registry, AutoML) · **Python**.
12
+
13
+ ## Quickstart
14
+
15
+ ```toml
16
+ [dependencies]
17
+ millwright = "0.1" # or features = ["full"] for the whole lifecycle
18
+ ```
19
+
20
+ ```rust
21
+ use millwright::prelude::*;
22
+
23
+ // features as rows + a target -> a Dataset
24
+ let train = Dataset::new(x, y)?;
25
+
26
+ // standardize, then a random forest — one composable object
27
+ let mut pipe = Pipeline::new()
28
+ .step("scale", StandardScaler::new())
29
+ .estimator("rf", RandomForest::new());
30
+
31
+ pipe.fit(&train)?;
32
+ let preds = pipe.predict(&test)?;
33
+ ```
34
+
35
+ ## Where to go
36
+
37
+ - **Tutorial:** <https://millwright-rs.dev/docs/> — the full, hands-on walk-through.
38
+ - **API reference:** [docs.rs/millwright](https://docs.rs/millwright).
39
+ - **Design brief (the *why*):** <https://millwright-rs.dev/>.
40
+ - **Examples:** [`examples/`](examples) — a runnable program for each feature group.
41
+ - **Python:** [`pip install millwright`](https://pypi.org/project/millwright/).
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: millwright
3
- Version: 0.1.1
3
+ Version: 0.2.0
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Programming Language :: Python :: 3
6
6
  Classifier: Intended Audience :: Science/Research
@@ -16,10 +16,19 @@ Project-URL: Repository, https://github.com/mi7plus/millwright
16
16
 
17
17
  # millwright
18
18
 
19
+ [![crates.io](https://img.shields.io/crates/v/millwright.svg)](https://crates.io/crates/millwright)
20
+ [![docs.rs](https://img.shields.io/docsrs/millwright)](https://docs.rs/millwright)
21
+ [![PyPI](https://img.shields.io/pypi/v/millwright.svg)](https://pypi.org/project/millwright/)
22
+ [![CI](https://github.com/mi7plus/millwright/actions/workflows/ci.yml/badge.svg)](https://github.com/mi7plus/millwright/actions/workflows/ci.yml)
23
+ [![downloads](https://img.shields.io/crates/d/millwright.svg)](https://crates.io/crates/millwright)
24
+ [![license](https://img.shields.io/crates/l/millwright.svg)](LICENSE)
25
+
19
26
  A unified ML framework for Rust — *ten crates, one lifecycle.*
20
27
 
21
- - **Tutorial:** [`GUIDE.md`](GUIDE.md) — the hands-on walk through the whole
22
- lifecycle (also as a page: [`guide.html`](guide.html)).
28
+ - **Tutorial:** the hands-on docs site at <https://millwright-rs.dev/docs/>
29
+ (source in [`docs/`](docs/index.html); a short [`GUIDE.md`](GUIDE.md) has the
30
+ quickstart).
31
+ - **Contributing:** see [`CONTRIBUTING.md`](CONTRIBUTING.md).
23
32
  - **Design brief:** the *why*, at **<https://millwright-rs.dev/>**.
24
33
 
25
34
  > *"Ten crates"* is the ecosystem this project assembles —
@@ -1,9 +1,18 @@
1
1
  # millwright
2
2
 
3
+ [![crates.io](https://img.shields.io/crates/v/millwright.svg)](https://crates.io/crates/millwright)
4
+ [![docs.rs](https://img.shields.io/docsrs/millwright)](https://docs.rs/millwright)
5
+ [![PyPI](https://img.shields.io/pypi/v/millwright.svg)](https://pypi.org/project/millwright/)
6
+ [![CI](https://github.com/mi7plus/millwright/actions/workflows/ci.yml/badge.svg)](https://github.com/mi7plus/millwright/actions/workflows/ci.yml)
7
+ [![downloads](https://img.shields.io/crates/d/millwright.svg)](https://crates.io/crates/millwright)
8
+ [![license](https://img.shields.io/crates/l/millwright.svg)](LICENSE)
9
+
3
10
  A unified ML framework for Rust — *ten crates, one lifecycle.*
4
11
 
5
- - **Tutorial:** [`GUIDE.md`](GUIDE.md) — the hands-on walk through the whole
6
- lifecycle (also as a page: [`guide.html`](guide.html)).
12
+ - **Tutorial:** the hands-on docs site at <https://millwright-rs.dev/docs/>
13
+ (source in [`docs/`](docs/index.html); a short [`GUIDE.md`](GUIDE.md) has the
14
+ quickstart).
15
+ - **Contributing:** see [`CONTRIBUTING.md`](CONTRIBUTING.md).
7
16
  - **Design brief:** the *why*, at **<https://millwright-rs.dev/>**.
8
17
 
9
18
  > *"Ten crates"* is the ecosystem this project assembles —
@@ -3,7 +3,19 @@
3
3
  Millwright ships as one Rust crate (crates.io) and one Python wheel (PyPI), from
4
4
  the same source.
5
5
 
6
- ## Preflight
6
+ ## The easy way
7
+
8
+ With `main` green and `CHANGELOG.md`'s `[Unreleased]` section filled in, run:
9
+
10
+ ```bash
11
+ bash scripts/release.sh 0.1.2
12
+ ```
13
+
14
+ It bumps both manifests, rolls the changelog, syncs `Cargo.lock`, verifies the
15
+ package, commits, tags `v0.1.2`, and (after a prompt) pushes — which triggers the
16
+ publish workflows below. The rest of this file is the manual equivalent.
17
+
18
+ ## Preflight (manual)
7
19
 
8
20
  1. `main` is green in CI and `CHANGELOG.md`'s `[Unreleased]` section is current.
9
21
  2. Bump the version in **both** `Cargo.toml` and `pyproject.toml`, move the
@@ -0,0 +1,114 @@
1
+ <html><head><meta http-equiv="Content-Type" content="text/html; charset=UTF-8"><title>Millwright · Data &amp; EDA</title>
2
+ <meta name="viewport" content="width=device-width, initial-scale=1">
3
+ <link rel="stylesheet" href="site.css">
4
+ </head><body><header class="top">
5
+ <div class="wrap">
6
+ <div class="brand"><a href="../index.html"><span class="mark">⚙</span>millwright</a><span class="ver">docs</span></div>
7
+ <nav>
8
+ <a href="index.html">home</a>
9
+ <a href="data.html" class="active">data &amp; EDA</a>
10
+ <a href="pipelines.html">pipelines</a>
11
+ <a href="insight.html">insight</a>
12
+ <a href="deploy.html">deploy</a>
13
+ <a href="python.html">python</a>
14
+ <a href="../index.html">design brief</a>
15
+ <a class="repo" href="https://github.com/mi7plus/millwright">GitHub ↗</a>
16
+ </nav>
17
+ </div>
18
+ </header>
19
+
20
+ <main>
21
+ <div class="wrap">
22
+ <div class="hero">
23
+ <div class="eyebrow">01 · data &amp; EDA</div>
24
+ <h1>The boundary type,<br>and the typed layer in front of it.</h1>
25
+ <p class="lede"><code class="inl">Frame</code> is the numeric boundary the whole API speaks. <code class="inl">Table</code> (feature <code class="inl">eda</code>) is the polars-backed, dtype-aware world that ingests real CSV/Parquet and lowers into it.</p>
26
+ </div>
27
+ </div>
28
+
29
+ <!-- FRAME -->
30
+ <section id="frame">
31
+ <div class="wrap">
32
+ <div class="head col">
33
+ <div class="eyebrow">Frame &amp; Dataset</div>
34
+ <h2>One contiguous <span class="mono">f64</span> buffer, plus a schema.</h2>
35
+ <p class="muted">Everything the <em>public</em> API speaks is a <code class="inl">Frame</code>: a contiguous, row-major <code class="inl">f64</code> buffer with named columns. It is what lets a linfa model and a smartcore <code class="inl">DenseMatrix</code> meet in one signature without your code ever naming their array versions — each backend converts <code class="inl">Frame</code> ⇄ its native type inside the adapter only. A <code class="inl">Dataset</code> pairs a frame with a target.</p>
36
+ </div>
37
+ <pre><span class="k">use</span> millwright::prelude::*;
38
+
39
+ <span class="k">let</span> x = <span class="f">Frame</span>::from_rows(
40
+ <span class="f">vec!</span>[<span class="f">vec!</span>[<span class="k">0.0</span>, <span class="k">0.1</span>], <span class="f">vec!</span>[<span class="k">0.4</span>, <span class="k">0.2</span>], <span class="f">vec!</span>[<span class="k">9.0</span>, <span class="k">9.1</span>], <span class="f">vec!</span>[<span class="k">9.4</span>, <span class="k">8.7</span>]],
41
+ <span class="f">vec!</span>[<span class="s">"a"</span>.into(), <span class="s">"b"</span>.into()],
42
+ )?;
43
+ <span class="k">assert_eq!</span>(x.shape(), (<span class="k">4</span>, <span class="k">2</span>)); <span class="c">// (rows, cols)</span>
44
+
45
+ <span class="k">let</span> train = <span class="f">Dataset</span>::new(x.clone(), <span class="f">vec!</span>[<span class="k">0.0</span>, <span class="k">0.0</span>, <span class="k">1.0</span>, <span class="k">1.0</span>])?;
46
+ <span class="k">let</span> _features = train.features(); <span class="c">// &Frame</span>
47
+ <span class="k">let</span> _target = train.target(); <span class="c">// &[f64]</span></pre>
48
+ <p class="tiny">The task — classification vs. regression — is inferred from the target: an all-integral target is class labels, anything else is regression. A pure-numeric CSV loads directly with <code class="inl">Frame::from_csv</code>; typed data uses <code class="inl">Table</code> below.</p>
49
+ </div>
50
+ </section>
51
+
52
+ <!-- INGEST -->
53
+ <section id="ingest">
54
+ <div class="wrap">
55
+ <div class="head col">
56
+ <div class="eyebrow">Ingest &amp; explore</div>
57
+ <h2><span class="mono">Table</span> reads it, <span class="mono">Profile</span> reports it.</h2>
58
+ <p class="muted">Behind the <code class="inl">eda</code> feature, a polars-backed <code class="inl">Table</code> reads real CSV/Parquet — strings, categories, dates, booleans, nulls — and a <code class="inl">Profile</code> returns a <em>typed</em> analysis (not just an HTML blob) and drafts the preprocessing.</p>
59
+ </div>
60
+ <pre><span class="k">let</span> table = <span class="f">Table</span>::from_csv(<span class="s">"customers.csv"</span>)?; <span class="c">// or ::from_parquet(…)</span>
61
+
62
+ <span class="c">// a typed profile — overview, per-column stats, missingness, correlations, alerts</span>
63
+ <span class="k">let</span> profile = <span class="f">Profile</span>::of_with_target(&amp;table, <span class="s">"churned"</span>)?;
64
+ <span class="f">println!</span>(<span class="s">"{}"</span>, profile.summary());
65
+ <span class="k">for</span> alert <span class="k">in</span> profile.alerts() {
66
+ <span class="f">println!</span>(<span class="s">"{alert}"</span>); <span class="c">// "[city] categorical (3 levels) → OneHotEncoder"</span>
67
+ }
68
+ profile.to_html(<span class="s">"eda_report.html"</span>)?; <span class="c">// a shareable report</span></pre>
69
+ <p class="muted">Because Millwright owns EDA <em>and</em> the pipeline, the profile drafts the starting preprocessing from its own findings — the loop scikit-learn can't close:</p>
70
+ <pre><span class="c">// lower the typed table to the numeric world</span>
71
+ <span class="k">let</span> train = table.into_dataset(<span class="s">"churned"</span>)?; <span class="c">// categoricals encoded, nulls → NaN</span>
72
+
73
+ <span class="c">// EDA drafts the pipeline; you just add the model</span>
74
+ <span class="k">let mut</span> pipe = profile.suggest_pipeline() <span class="c">// impute · encode · scale, from the alerts</span>
75
+ .estimator(<span class="s">"rf"</span>, <span class="f">RandomForest</span>::new());
76
+ pipe.fit(&amp;train)?;</pre>
77
+ <p class="run">cargo run --example explore --features "eda smartcore-backend"</p>
78
+ </div>
79
+ </section>
80
+
81
+ <!-- DTYPES -->
82
+ <section id="dtypes">
83
+ <div class="wrap">
84
+ <div class="head col">
85
+ <div class="eyebrow">Dtype-aware</div>
86
+ <h2>Types flow through the pipeline.</h2>
87
+ <p class="muted">A <code class="inl">Frame</code> carries a per-column <code class="inl">Dtype</code> (defaulting to <code class="inl">Numeric</code>). <code class="inl">Table</code> marks the columns it knows are <code class="inl">Categorical</code> as it lowers — so preprocessing doesn't have to <em>guess</em>: scalers, <code class="inl">Winsorize</code>, and <code class="inl">PowerTransform</code> pass categorical columns through untouched, and <code class="inl">OneHotEncoder</code> encodes by dtype rather than a value heuristic.</p>
88
+ </div>
89
+ <pre><span class="c">// a genuinely-integer feature is NOT wrongly one-hot'd; the categorical one is</span>
90
+ <span class="k">let</span> f = <span class="f">Frame</span>::from_rows(rows, cols)?
91
+ .with_dtypes(<span class="f">vec!</span>[<span class="f">Dtype</span>::Categorical, <span class="f">Dtype</span>::Numeric])?;
92
+ <span class="k">let</span> out = <span class="f">OneHotEncoder</span>::infer().fit_transform(&amp;f)?; <span class="c">// expands only column 0</span>
93
+
94
+ <span class="c">// or one-hot at the Table boundary, with real category names</span>
95
+ <span class="k">let</span> train = table.into_dataset_with(<span class="s">"churned"</span>, <span class="f">CategoryEncoding</span>::OneHot)?;</pre>
96
+ <p class="tiny">Nominal categories become <code class="inl">"{col}={value}"</code> indicator columns instead of ordinal codes — the correct representation for linear and tree models.</p>
97
+ </div>
98
+ </section>
99
+
100
+ <div class="wrap">
101
+ <div class="pager">
102
+ <a href="index.html"><span class="dir">← prev</span><b>Home</b></a>
103
+ <a class="next" href="pipelines.html"><span class="dir">next →</span><b>Pipelines &amp; Models</b></a>
104
+ </div>
105
+ </div>
106
+ </main>
107
+
108
+ <footer>
109
+ <div class="wrap">
110
+ <span class="mono">⚙ millwright docs</span>
111
+ <span class="mono"><a href="../index.html">design brief</a> · <a href="https://crates.io/crates/millwright">crates.io</a> · <a href="https://pypi.org/project/millwright/">PyPI</a> · <a href="https://docs.rs/millwright">docs.rs</a> · <a href="https://github.com/mi7plus/millwright">GitHub</a></span>
112
+ </div>
113
+ </footer>
114
+ </body></html>
@@ -0,0 +1,134 @@
1
+ <html><head><meta http-equiv="Content-Type" content="text/html; charset=UTF-8"><title>Millwright · Deploy</title>
2
+ <meta name="viewport" content="width=device-width, initial-scale=1">
3
+ <link rel="stylesheet" href="site.css">
4
+ </head><body><header class="top">
5
+ <div class="wrap">
6
+ <div class="brand"><a href="../index.html"><span class="mark">⚙</span>millwright</a><span class="ver">docs</span></div>
7
+ <nav>
8
+ <a href="index.html">home</a>
9
+ <a href="data.html">data &amp; EDA</a>
10
+ <a href="pipelines.html">pipelines</a>
11
+ <a href="insight.html">insight</a>
12
+ <a href="deploy.html" class="active">deploy</a>
13
+ <a href="python.html">python</a>
14
+ <a href="../index.html">design brief</a>
15
+ <a class="repo" href="https://github.com/mi7plus/millwright">GitHub ↗</a>
16
+ </nav>
17
+ </div>
18
+ </header>
19
+
20
+ <main>
21
+ <div class="wrap">
22
+ <div class="hero">
23
+ <div class="eyebrow">04 · deploy</div>
24
+ <h1>Past where<br>scikit-learn stops.</h1>
25
+ <p class="lede">Export to one portable ONNX artifact, serve a drift-monitored endpoint, version every model with its lineage, and — the framework pointed at itself — let AutoML search for the best <em>deployable</em> pipeline.</p>
26
+ </div>
27
+ </div>
28
+
29
+ <!-- ONNX -->
30
+ <section id="onnx">
31
+ <div class="wrap">
32
+ <div class="head col">
33
+ <div class="eyebrow">Portability</div>
34
+ <h2>ONNX in and out.</h2>
35
+ <p class="muted">With <code class="inl">onnx</code>, any model — or a whole pipeline — exports to one <code class="inl">.onnx</code> file. Whole-pipeline export folds leading affine scalers into the estimator's graph: raw features in, predictions out. <code class="inl">InferenceModel::load</code> runs any ONNX file back through tract.</p>
36
+ </div>
37
+ <pre><span class="k">let mut</span> pipe = <span class="f">Pipeline</span>::new()
38
+ .step(<span class="s">"scale"</span>, <span class="f">StandardScaler</span>::new())
39
+ .estimator(<span class="s">"lr"</span>, <span class="f">LinearRegression</span>::new());
40
+ pipe.fit(&amp;train)?;
41
+ <span class="k">let</span> native = pipe.predict(&amp;probe)?;
42
+
43
+ pipe.export_onnx(<span class="s">"pipeline.onnx"</span>)?; <span class="c">// scaler + model, one graph</span>
44
+ <span class="k">let</span> model = <span class="f">InferenceModel</span>::load(<span class="s">"pipeline.onnx"</span>)?;
45
+ <span class="k">let</span> via_onnx = model.predict(&amp;probe)?; <span class="c">// matches `native`</span></pre>
46
+ <p class="tiny">Linear/affine/pipeline graphs run inside tract for a full round-trip. A random forest exports to a valid ONNX-ML tree-ensemble artifact for external runtimes (onnxruntime); tract implements NN ops, not the ONNX-ML tree ops.</p>
47
+ <p class="run">cargo run --example portability --features "smartcore-backend onnx"</p>
48
+ </div>
49
+ </section>
50
+
51
+ <!-- OPERATE -->
52
+ <section id="operate">
53
+ <div class="wrap">
54
+ <div class="head col">
55
+ <div class="eyebrow">Operations</div>
56
+ <h2>Registry, drift, serving.</h2>
57
+ <p class="muted">The <code class="inl">registry</code> versions the ONNX artifact (content-addressed, with a reference distribution and movable tags); <code class="inl">monitor</code> watches the prediction stream for PSI drift; <code class="inl">serve</code> exposes a validated endpoint that feeds the monitor.</p>
58
+ </div>
59
+ <pre><span class="k">let</span> reg = <span class="f">Registry</span>::local(<span class="s">"./models"</span>);
60
+ <span class="k">let</span> v1 = reg.register(<span class="s">"demand"</span>, &amp;model, <span class="f">Metadata</span> {
61
+ metrics: <span class="f">vec!</span>[(<span class="s">"r2"</span>.into(), <span class="k">1.0</span>)],
62
+ reference: reference.clone(), <span class="c">// the distribution drift watches against</span>
63
+ note: <span class="s">"baseline"</span>.into(),
64
+ })?;
65
+ reg.tag(<span class="s">"demand"</span>, &amp;v1.id, <span class="s">"prod"</span>)?;
66
+ <span class="k">let</span> reverted = reg.rollback(<span class="s">"demand"</span>, <span class="s">"prod"</span>)?; <span class="c">// revert in one line</span>
67
+
68
+ <span class="c">// serve the prod artifact, watching for drift on every request</span>
69
+ <span class="f">Server</span>::from_onnx(reg.onnx_path(<span class="s">"demand"</span>, <span class="s">"prod"</span>)?)?
70
+ .route(<span class="s">"/predict"</span>)
71
+ .with_monitor(<span class="f">DriftMonitor</span>::psi(&amp;reference)?)
72
+ .serve(<span class="s">"0.0.0.0:8080"</span>).<span class="k">await</span>?; <span class="c">// POST /predict, GET /metrics</span></pre>
73
+ <p class="run">cargo run --example operations --features "onnx registry monitor serve"</p>
74
+ </div>
75
+ </section>
76
+
77
+ <!-- SPECIALIZED -->
78
+ <section id="specialized">
79
+ <div class="wrap">
80
+ <div class="head col">
81
+ <div class="eyebrow">Specialized shapes</div>
82
+ <h2>Time series &amp; out-of-core.</h2>
83
+ <p class="muted">Same contract, different data shapes — each gets its own trait. These two crates pin <code class="inl">ndarray 0.15</code> while the rest of the stack uses <code class="inl">0.16</code>; Cargo links both and converts only inside the adapters.</p>
84
+ </div>
85
+ <pre><span class="c">// time series (feature = "timeseries")</span>
86
+ <span class="k">let mut</span> arima = <span class="f">AutoArima</span>::new().max_p(<span class="k">3</span>).max_q(<span class="k">3</span>);
87
+ arima.fit(&amp;series)?; <span class="c">// &[f64]</span>
88
+ <span class="k">let</span> forecast = arima.forecast(<span class="k">6</span>)?; <span class="c">// six steps ahead</span>
89
+
90
+ <span class="c">// out-of-core (feature = "incremental") — never holds the whole set in memory</span>
91
+ <span class="k">let mut</span> model = <span class="f">IncrementalLinear</span>::with_rate(<span class="k">0.05</span>, <span class="k">0.0</span>);
92
+ <span class="k">for</span> batch <span class="k">in</span> batches {
93
+ model.partial_fit(&amp;batch)?; <span class="c">// one batch at a time</span>
94
+ }</pre>
95
+ <p class="run">cargo run --example specialized --features "timeseries incremental"</p>
96
+ </div>
97
+ </section>
98
+
99
+ <!-- AUTOML -->
100
+ <section id="automl">
101
+ <div class="wrap">
102
+ <div class="head col">
103
+ <div class="eyebrow">Synthesis</div>
104
+ <h2>AutoML — the framework, pointed at itself.</h2>
105
+ <p class="muted">Profiling, preprocessing, CV, search, and ensembling are exactly what an AutoML engine needs — so <code class="inl">AutoML</code> is not a bolt-on, it is the framework orchestrating its own parts. Point it at data and a budget; get a leaderboard and the best <em>deployable</em> model.</p>
106
+ </div>
107
+ <pre><span class="k">let</span> result = <span class="f">AutoML</span>::classifier() <span class="c">// or ::regressor()</span>
108
+ .budget(<span class="f">Budget</span>::trials(<span class="k">20</span>)) <span class="c">// or Budget::minutes(10)</span>
109
+ .metric(<span class="f">Metric</span>::F1)
110
+ .cv(<span class="f">StratifiedKFold</span>::new(<span class="k">5</span>))
111
+ .seed(<span class="k">0</span>)
112
+ .fit(&amp;train)?;
113
+
114
+ <span class="f">println!</span>(<span class="s">"{}"</span>, result.leaderboard());
115
+ result.export_onnx(<span class="s">"model.onnx"</span>)?; <span class="c">// deployable — unlike a TPOT object</span></pre>
116
+ <p class="run">cargo run --example automl --features "automl onnx"</p>
117
+ </div>
118
+ </section>
119
+
120
+ <div class="wrap">
121
+ <div class="pager">
122
+ <a href="insight.html"><span class="dir">← prev</span><b>Insight</b></a>
123
+ <a class="next" href="python.html"><span class="dir">next →</span><b>Python</b></a>
124
+ </div>
125
+ </div>
126
+ </main>
127
+
128
+ <footer>
129
+ <div class="wrap">
130
+ <span class="mono">⚙ millwright docs</span>
131
+ <span class="mono"><a href="../index.html">design brief</a> · <a href="https://crates.io/crates/millwright">crates.io</a> · <a href="https://pypi.org/project/millwright/">PyPI</a> · <a href="https://docs.rs/millwright">docs.rs</a> · <a href="https://github.com/mi7plus/millwright">GitHub</a></span>
132
+ </div>
133
+ </footer>
134
+ </body></html>