millwright 0.1.1__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. millwright-0.2.1/CHANGELOG.md +144 -0
  2. millwright-0.2.1/CONTRIBUTING.md +55 -0
  3. {millwright-0.1.1 → millwright-0.2.1}/Cargo.lock +3 -1
  4. {millwright-0.1.1 → millwright-0.2.1}/Cargo.toml +13 -5
  5. millwright-0.2.1/GUIDE.md +41 -0
  6. {millwright-0.1.1 → millwright-0.2.1}/PKG-INFO +12 -3
  7. {millwright-0.1.1 → millwright-0.2.1}/README.md +11 -2
  8. {millwright-0.1.1 → millwright-0.2.1}/RELEASING.md +13 -1
  9. millwright-0.2.1/docs/data.html +114 -0
  10. millwright-0.2.1/docs/deploy.html +135 -0
  11. millwright-0.2.1/docs/index.html +156 -0
  12. millwright-0.2.1/docs/insight.html +102 -0
  13. millwright-0.2.1/docs/pipelines.html +187 -0
  14. millwright-0.2.1/docs/python.html +106 -0
  15. millwright-0.2.1/docs/site.css +152 -0
  16. millwright-0.2.1/guide.html +9 -0
  17. {millwright-0.1.1 → millwright-0.2.1}/index.html +42 -38
  18. {millwright-0.1.1 → millwright-0.2.1}/pyproject.toml +1 -1
  19. millwright-0.2.1/scripts/release.sh +54 -0
  20. {millwright-0.1.1 → millwright-0.2.1}/src/automl.rs +136 -14
  21. millwright-0.2.1/src/backends/smartcore.rs +624 -0
  22. {millwright-0.1.1 → millwright-0.2.1}/src/ensemble.rs +231 -9
  23. {millwright-0.1.1 → millwright-0.2.1}/src/lib.rs +2 -2
  24. {millwright-0.1.1 → millwright-0.2.1}/src/monitor.rs +9 -0
  25. millwright-0.2.1/src/onnx.rs +1075 -0
  26. {millwright-0.1.1 → millwright-0.2.1}/src/pipeline.rs +29 -34
  27. {millwright-0.1.1 → millwright-0.2.1}/src/profile.rs +232 -15
  28. millwright-0.2.1/src/python.rs +1010 -0
  29. {millwright-0.1.1 → millwright-0.2.1}/src/registry.rs +37 -0
  30. {millwright-0.1.1 → millwright-0.2.1}/src/selection/cv.rs +15 -10
  31. {millwright-0.1.1 → millwright-0.2.1}/src/selection/mod.rs +66 -0
  32. {millwright-0.1.1 → millwright-0.2.1}/src/selection/scoring.rs +35 -1
  33. {millwright-0.1.1 → millwright-0.2.1}/src/selection/search.rs +9 -0
  34. {millwright-0.1.1 → millwright-0.2.1}/src/serve.rs +12 -0
  35. {millwright-0.1.1 → millwright-0.2.1}/src/table.rs +24 -0
  36. {millwright-0.1.1 → millwright-0.2.1}/src/traits.rs +15 -4
  37. {millwright-0.1.1 → millwright-0.2.1}/src/transform.rs +15 -0
  38. millwright-0.2.1/tests/ensemble_zoo.rs +42 -0
  39. millwright-0.2.1/tests/hero_snippet.rs +51 -0
  40. millwright-0.2.1/tests/onnx_export.rs +151 -0
  41. millwright-0.1.1/CHANGELOG.md +0 -65
  42. millwright-0.1.1/GUIDE.md +0 -621
  43. millwright-0.1.1/guide.html +0 -574
  44. millwright-0.1.1/src/backends/smartcore.rs +0 -290
  45. millwright-0.1.1/src/onnx.rs +0 -271
  46. millwright-0.1.1/src/python.rs +0 -179
  47. {millwright-0.1.1 → millwright-0.2.1}/.github/workflows/ci.yml +0 -0
  48. {millwright-0.1.1 → millwright-0.2.1}/.github/workflows/release-crate.yml +0 -0
  49. {millwright-0.1.1 → millwright-0.2.1}/.github/workflows/release-python.yml +0 -0
  50. {millwright-0.1.1 → millwright-0.2.1}/.gitignore +0 -0
  51. {millwright-0.1.1 → millwright-0.2.1}/CNAME +0 -0
  52. {millwright-0.1.1 → millwright-0.2.1}/LICENSE +0 -0
  53. {millwright-0.1.1 → millwright-0.2.1}/benches/throughput.rs +0 -0
  54. {millwright-0.1.1 → millwright-0.2.1}/examples/automl.rs +0 -0
  55. {millwright-0.1.1 → millwright-0.2.1}/examples/backends.rs +0 -0
  56. {millwright-0.1.1 → millwright-0.2.1}/examples/explore.rs +0 -0
  57. {millwright-0.1.1 → millwright-0.2.1}/examples/insight.rs +0 -0
  58. {millwright-0.1.1 → millwright-0.2.1}/examples/operations.rs +0 -0
  59. {millwright-0.1.1 → millwright-0.2.1}/examples/portability.rs +0 -0
  60. {millwright-0.1.1 → millwright-0.2.1}/examples/specialized.rs +0 -0
  61. {millwright-0.1.1 → millwright-0.2.1}/examples/spine.rs +0 -0
  62. {millwright-0.1.1 → millwright-0.2.1}/examples/trust.rs +0 -0
  63. {millwright-0.1.1 → millwright-0.2.1}/examples/workflow.rs +0 -0
  64. {millwright-0.1.1 → millwright-0.2.1}/src/anomaly.rs +0 -0
  65. {millwright-0.1.1 → millwright-0.2.1}/src/backends/chronos.rs +0 -0
  66. {millwright-0.1.1 → millwright-0.2.1}/src/backends/incremental.rs +0 -0
  67. {millwright-0.1.1 → millwright-0.2.1}/src/backends/linfa.rs +0 -0
  68. {millwright-0.1.1 → millwright-0.2.1}/src/backends/mod.rs +0 -0
  69. {millwright-0.1.1 → millwright-0.2.1}/src/balance.rs +0 -0
  70. {millwright-0.1.1 → millwright-0.2.1}/src/calibration.rs +0 -0
  71. {millwright-0.1.1 → millwright-0.2.1}/src/diagnostics.rs +0 -0
  72. {millwright-0.1.1 → millwright-0.2.1}/src/error.rs +0 -0
  73. {millwright-0.1.1 → millwright-0.2.1}/src/evaluate.rs +0 -0
  74. {millwright-0.1.1 → millwright-0.2.1}/src/explain.rs +0 -0
  75. {millwright-0.1.1 → millwright-0.2.1}/src/frame.rs +0 -0
  76. {millwright-0.1.1 → millwright-0.2.1}/src/logistic.rs +0 -0
  77. {millwright-0.1.1 → millwright-0.2.1}/src/rng.rs +0 -0
  78. {millwright-0.1.1 → millwright-0.2.1}/src/viz.rs +0 -0
  79. {millwright-0.1.1 → millwright-0.2.1}/tests/golden.rs +0 -0
  80. {millwright-0.1.1 → millwright-0.2.1}/tests/real_data.rs +0 -0
@@ -0,0 +1,144 @@
1
+ # Changelog
2
+
3
+ All notable changes to Millwright are recorded here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/), and the project aims at
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [0.2.1] - 2026-08-23
10
+
11
+ ### Added
12
+ - **Exported forests now serve in Millwright.** `InferenceModel` runs linear / NN
13
+ ONNX graphs through tract as before, and evaluates the ONNX-ML tree-ensemble
14
+ ops tract doesn't implement (from an exported `RandomForest`) with a small
15
+ native interpreter. A model exported by Millwright always round-trips back into
16
+ `InferenceModel`/`Server` — forests included — while the artifact stays portable
17
+ to any ONNX runtime.
18
+ - **Imputers and one-hot encoders are ONNX-exportable.** A `SimpleImputer` step
19
+ exports as `Where(IsNaN(x), fill, x)`; a `OneHotEncoder` step exports as a
20
+ per-column `Gather` → `Round` → `Equal` → `Cast` → `Concat` expansion (with the
21
+ graph input re-declared at the raw feature width). So the whole realistic
22
+ `impute → one-hot → scale → model` pipeline exports and serves as one graph —
23
+ through tract (linear) or the native interpreter (behind a forest), verified
24
+ identical to the in-memory pipeline. The pipeline export generalized from
25
+ folding one affine map to splicing an ordered chain of transformer "prefixes"
26
+ (new `Transformer::onnx_prefix`). Only steps with no ONNX form now error.
27
+ - **`SearchResult::export_onnx`** — a `GridSearch`/`RandomSearch` winner can now
28
+ be exported to ONNX (it could only `predict` before). Backed by a new
29
+ object-safe `Estimator::to_onnx_proto` on `Pipeline`.
30
+ - **`Server::from_registry(&reg, name, tag)`** and
31
+ **`DriftMonitor::from_registry(&version)`** — serve a tagged artifact straight
32
+ from a registry, and build a PSI monitor from the version's stored reference
33
+ distribution.
34
+
35
+ ## [0.2.0] - 2026-08-23
36
+
37
+ ### Added
38
+ - **rayon parallelism.** Cross-validation folds and search candidates now
39
+ evaluate in parallel: `cross_val_score` is fold-parallel, `Bagging` fits its
40
+ base estimators in parallel, and `AutoML::parallel()` adds candidate-level
41
+ parallelism. The core contract traits gained `Send + Sync` bounds (every
42
+ concrete model already satisfied them). Results stay seed-reproducible.
43
+ - **`Boosting`** — SAMME adaptive boosting over any weak learner (an
44
+ `alpha`-weighted vote of models each reweighted toward the last round's
45
+ mistakes), joining `Voting`/`Bagging`/`Stacking`.
46
+ - **Three more models over the smartcore backend:** `Knn` (k-nearest-neighbours),
47
+ `Svc` (support vector classifier, linear or RBF, one-vs-one for multiclass),
48
+ and `NaiveBayes` (Gaussian). All implement the same `Estimator`/`Predictor`
49
+ contract, so they drop into pipelines, ensembles, and search unchanged — and
50
+ they are exposed to Python too (`mw.Knn`, `mw.Svc` / `mw.Svc.rbf()`,
51
+ `mw.NaiveBayes`, plus `pipe.knn()` / `pipe.svc()` / `pipe.naive_bayes()`).
52
+ - **Python: a scikit-learn-shaped object API.** `mw.Frame` with `from_pandas` /
53
+ `from_numpy` / `from_rows` ingest; composable transformer/estimator objects
54
+ (`StandardScaler`, `MinMaxScaler`, `SimpleImputer`, `OneHotEncoder`,
55
+ `RandomForest`, `LinearRegression`) added via `pipe.step(name, obj)` /
56
+ `pipe.estimator(name, obj)`; `fit`/`predict`/`evaluate` accept a `Frame`.
57
+ - **Python: `pipeline.explain(...)`** returns SHAP feature importance
58
+ (`Explainer.kernel()` configurable), and **`pipeline.export_onnx(path)`**
59
+ writes the fitted pipeline to a single ONNX file. The `python` wheel now
60
+ bundles the `model-selection`, `explain`, and `onnx` engines.
61
+ - **Python: `GridSearch` / `KFold` / `StratifiedKFold`** over a pipeline, with
62
+ a `SearchResult` (`best_score`, `best_params()`, `predict()`).
63
+ - **`InferenceModel` is now an `Estimator` + `Predictor`**, so a pre-trained
64
+ ONNX model (from scikit-learn, PyTorch, …) can be dropped into a `Pipeline`
65
+ as a frozen estimator behind Millwright's preprocessing — in Rust and, via
66
+ `mw.OnnxModel(path)`, from Python.
67
+ - **Python: `mw.Table` + `mw.Profile`** — dtype-aware CSV/Parquet ingest and
68
+ automated EDA (`Profile.of(table_or_frame).to_html(path)`). The wheel now
69
+ bundles the `eda` (polars) engine, so it is larger than the pure-model build.
70
+ - **`Table::from_frame`** (Rust): build a numeric `Table` from a `Frame`, so the
71
+ numeric world can round-trip back into the typed one (e.g. to profile it).
72
+ - **Richer EDA in `Profile`** — excess `kurtosis` and z-score outlier counts per
73
+ numeric column; a **Spearman** rank-correlation matrix beside the Pearson one;
74
+ a **co-missing** map (columns whose null patterns correlate); high-cardinality
75
+ categoricals now suggest `TargetEncoder`; and `suggest_pipeline` adds a
76
+ train-time **SMOTE** balancer on class imbalance (with `preprocessing`).
77
+ - **AutoML is seeded by EDA.** With the `eda` engine on, the search fixes its
78
+ preprocessing to `Profile::suggest_pipeline()` and varies only the model,
79
+ pruning the space (it falls back to a scaler sweep without `eda`).
80
+
81
+ ### Fixed
82
+ - **Cross-validated F1 no longer returns `NaN`.** A fold whose predictions
83
+ contain no true positives left smartcore's F1 evaluating `0/0`; a single NaN
84
+ fold poisoned the CV mean and a search's `best_score`. An undefined F1 is now
85
+ 0.0 (scikit-learn's `zero_division=0` convention), so `GridSearch(...).scoring(F1)`
86
+ yields finite, comparable scores.
87
+
88
+ ## [0.1.1]
89
+
90
+ ### Added
91
+
92
+ - **`LogisticRegression`** — a native, core binary classifier with genuine
93
+ `predict_proba`: the framework's first real `ProbaPredictor`.
94
+ - **`calibration` feature** — `PlattScaling`, `IsotonicRegression`,
95
+ `reliability_curve`, and `CalibratedClassifier`, which wraps any
96
+ `ProbaPredictor` and returns calibrated probabilities.
97
+ - **`anomaly` feature** — `Mahalanobis` and `KnnScore`, unified behind an
98
+ `OutlierDetector` trait.
99
+ - **`eda` feature** — a polars-backed, dtype-aware `Table` (CSV/Parquet ingest
100
+ that lowers to the numeric `Frame`) and a typed `Profile` with an HTML report,
101
+ actionable alerts, and `suggest_pipeline()`.
102
+ - **Transformers** — `Winsorize`, `PowerTransform` (Yeo-Johnson),
103
+ `ColumnTransformer`, and the supervised `TargetEncoder`.
104
+ - **Convenience** — `Frame::from_csv` (dependency-free numeric loader),
105
+ `Table::head`.
106
+ - **Python** — `min_max_scaler`, `simple_imputer`, `one_hot`,
107
+ `linear_regression`, and `evaluate()`.
108
+ - **Examples** — `explore` (ingest → profile → pipeline) and `trust`
109
+ (calibration → reliability → anomaly detection).
110
+ - **Benchmarks** — `benches/throughput.rs` (criterion): the boundary conversion
111
+ and core fit/predict, backing the "Rust speed" claim.
112
+ - **One-hot ingest** — `Table::to_frame_with` / `into_dataset_with` and a
113
+ `CategoryEncoding` enum: lower nominal categories to 0/1 indicator columns
114
+ instead of ordinal codes.
115
+ - **Schema-aware preprocessing** — `Frame` carries a per-column `Dtype`; `Table`
116
+ marks categoricals as it lowers; scalers / `Winsorize` / `PowerTransform` pass
117
+ categorical columns through untouched, and `OneHotEncoder` encodes by dtype
118
+ rather than a value heuristic when the schema is known.
119
+ - **Real-data validation** — an end-to-end integration test on Quinlan's
120
+ PlayTennis (`tests/real_data.rs`): CSV → profile → suggested pipeline → fit.
121
+
122
+ ### Changed
123
+
124
+ - Exact-version pins on every engine crate; `Cargo.lock` committed; a
125
+ feature-matrix CI (fmt, clippy `-D warnings`, docs, matrix, OS, examples,
126
+ benches, publish dry-run, wheel).
127
+ - `selection.rs` split into `selection/{scoring,cv,search}`.
128
+ - MSRV is **1.95** (dep-dictated — `sysinfo` via tract, and polars); enforced by
129
+ cargo via `rust-version` rather than a dedicated CI job (which would break on
130
+ every transitive bump). The default install needs 1.85.
131
+ - De-staled the crate and module docs (no more "Phase 0 · the spine").
132
+
133
+ ### Fixed
134
+
135
+ - Golden tests and the crate doctest build under every feature subset (they were
136
+ unconditionally referencing backend-gated types).
137
+ - Float sorts use `f64::total_cmp`, closing a NaN-driven panic class.
138
+
139
+ ## [0.1.0]
140
+
141
+ - Phases 0–8: the `Frame`/trait spine and smartcore backend, preprocessing and
142
+ model selection, a second backend (linfa) and HPO, evaluation/diagnostics/
143
+ explainability, ONNX export and inference, serving + drift monitoring + a model
144
+ registry, time-series and out-of-core estimators, AutoML, and 1.0 hardening.
@@ -0,0 +1,55 @@
1
+ # Contributing to Millwright
2
+
3
+ Thanks for your interest! Millwright assembles proven Rust crates into one
4
+ composable ML lifecycle. Bug reports, small fixes, and focused features are all
5
+ welcome.
6
+
7
+ ## Getting set up
8
+
9
+ ```bash
10
+ git clone https://github.com/mi7plus/millwright
11
+ cd millwright
12
+ cargo test # default features
13
+ ```
14
+
15
+ Rust **1.95+** is required (the floor is dictated by transitive engine deps).
16
+
17
+ ## Before you open a PR
18
+
19
+ The CI is a feature matrix, so run the same checks locally:
20
+
21
+ ```bash
22
+ cargo fmt --all --check
23
+ cargo clippy --features full --all-targets -- -D warnings
24
+ cargo test --features full
25
+ cargo test --no-default-features # the bare-core build must pass too
26
+ cargo doc --no-deps --features full # with RUSTDOCFLAGS="-D warnings"
27
+ ```
28
+
29
+ - **Every capability is a cargo feature.** New functionality behind a young
30
+ single-author crate goes behind its own feature; the core stays lean. See the
31
+ feature list in `Cargo.toml`.
32
+ - **Engines are pinned to exact versions** (`=x.y.z`) and `Cargo.lock` is
33
+ committed — bump them deliberately, one line, one commit.
34
+ - **Add tests.** Prefer a `#[test]` next to the code; lock numeric behaviour in
35
+ `tests/golden.rs` when it matters. Feature-gate tests that need a backend.
36
+ - **Run the examples** you touch: `cargo run --features full --example <name>`.
37
+
38
+ ## Try it out
39
+
40
+ - Examples: [`examples/`](examples) — one runnable program per feature group.
41
+ - Benchmarks: `cargo bench --features smartcore-backend`.
42
+ - Docs/tutorial: <https://millwright-rs.dev/docs/>.
43
+
44
+ ## Releasing
45
+
46
+ Maintainers: bump `Cargo.toml` + `pyproject.toml`, roll `CHANGELOG.md`, and tag
47
+ `vX.Y.Z` — the workflows publish to crates.io and PyPI via OIDC. The helper
48
+ `scripts/release.sh X.Y.Z` does the mechanical steps. See
49
+ [`RELEASING.md`](RELEASING.md).
50
+
51
+ ## Scope & conduct
52
+
53
+ Millwright is a thin facade over proven engines — it orchestrates, it doesn't
54
+ reimplement numerics. Please keep PRs focused, and be kind and constructive in
55
+ issues and reviews.
@@ -1935,7 +1935,7 @@ dependencies = [
1935
1935
 
1936
1936
  [[package]]
1937
1937
  name = "millwright"
1938
- version = "0.1.1"
1938
+ version = "0.2.1"
1939
1939
  dependencies = [
1940
1940
  "axum",
1941
1941
  "chronos-ts",
@@ -1956,7 +1956,9 @@ dependencies = [
1956
1956
  "plotters",
1957
1957
  "plotters-statistical",
1958
1958
  "polars",
1959
+ "prost 0.13.5",
1959
1960
  "pyo3",
1961
+ "rayon",
1960
1962
  "regression-diagnostics",
1961
1963
  "serde",
1962
1964
  "serde_json",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "millwright"
3
- version = "0.1.1"
3
+ version = "0.2.1"
4
4
  edition = "2021"
5
5
  rust-version = "1.95"
6
6
  description = "A unified ML framework for Rust — proven Rust crates, assembled into one machine."
@@ -56,6 +56,11 @@ polars = { version = "=0.55.2", default-features = false, features = ["csv", "pa
56
56
  # --- general infrastructure: caret ranges, reproduced via Cargo.lock ---
57
57
  pyo3 = { version = "0.24", features = ["extension-module", "abi3-py39"], optional = true }
58
58
  ndarray = { version = "0.16", optional = true }
59
+ rayon = { version = "1", optional = true }
60
+ # Decode a saved ONNX file back into the proto, to natively evaluate ONNX-ML ops
61
+ # (tree ensembles) that tract does not implement. Same version onnx-export-rs
62
+ # uses, so the generated proto types decode cleanly.
63
+ prost = { version = "0.13", optional = true }
59
64
  plotters = { version = "0.3.7", default-features = false, features = ["svg_backend", "all_series"], optional = true }
60
65
  tract-onnx = { version = "0.23", optional = true }
61
66
  axum = { version = "0.8", optional = true }
@@ -90,11 +95,13 @@ smartcore-backend = ["dep:smartcore"]
90
95
  preprocessing = ["dep:imbalance-rs", "dep:ndarray"]
91
96
 
92
97
  # Cross-validation splitters, scoring, and grid/random search over a pipeline.
93
- model-selection = ["dep:model-selection-rs", "dep:ndarray"]
98
+ # CV folds and search candidates are evaluated in parallel via rayon.
99
+ model-selection = ["dep:model-selection-rs", "dep:ndarray", "dep:rayon"]
94
100
 
95
101
  # Voting / bagging / stacking meta-estimators, composed over the four traits.
96
- # Leak-free stacking uses the model-selection CV engine when it is enabled.
97
- ensemble = []
102
+ # Leak-free stacking uses the model-selection CV engine when it is enabled;
103
+ # bagging fits its base estimators in parallel via rayon.
104
+ ensemble = ["dep:rayon"]
98
105
 
99
106
  # Ingest & EDA — the front of the lifecycle. A polars-backed, dtype-aware
100
107
  # `Table` (CSV/Parquet in, real string/categorical/datetime/null columns) that
@@ -143,6 +150,7 @@ onnx = [
143
150
  "onnx-export-rs/smartcore-compat",
144
151
  "dep:tract-onnx",
145
152
  "dep:ndarray",
153
+ "dep:prost",
146
154
  "smartcore-backend",
147
155
  "smartcore/serde",
148
156
  ]
@@ -151,7 +159,7 @@ onnx = [
151
159
  # Deliberately NOT part of `full`: pyo3's `extension-module` defers libpython
152
160
  # symbols, so a plain `cargo test`/`clippy` cannot link a test binary with it.
153
161
  # Build & test this feature through maturin instead (see pyproject.toml).
154
- python = ["dep:pyo3", "smartcore-backend"]
162
+ python = ["dep:pyo3", "smartcore-backend", "model-selection", "explain", "onnx", "eda"]
155
163
 
156
164
  # Phase 5 · OPERATIONS — past where scikit-learn stops.
157
165
  # A versioned model registry: content-addressed ONNX artifact + metadata +
@@ -0,0 +1,41 @@
1
+ # The Millwright Guide
2
+
3
+ *A unified ML framework for Rust — ten crates, one lifecycle.*
4
+
5
+ The full hands-on tutorial now lives as a browsable, multi-page site:
6
+
7
+ ### → **[millwright-rs.dev/docs/](https://millwright-rs.dev/docs/)** (source in [`docs/`](docs/index.html))
8
+
9
+ It walks the whole lifecycle, one topic per page: **Data & EDA** · **Pipelines &
10
+ Models** · **Insight** (evaluate, explain, calibrate, detect) · **Deploy**
11
+ (ONNX, serving, registry, AutoML) · **Python**.
12
+
13
+ ## Quickstart
14
+
15
+ ```toml
16
+ [dependencies]
17
+ millwright = "0.1" # or features = ["full"] for the whole lifecycle
18
+ ```
19
+
20
+ ```rust
21
+ use millwright::prelude::*;
22
+
23
+ // features as rows + a target -> a Dataset
24
+ let train = Dataset::new(x, y)?;
25
+
26
+ // standardize, then a random forest — one composable object
27
+ let mut pipe = Pipeline::new()
28
+ .step("scale", StandardScaler::new())
29
+ .estimator("rf", RandomForest::new());
30
+
31
+ pipe.fit(&train)?;
32
+ let preds = pipe.predict(&test)?;
33
+ ```
34
+
35
+ ## Where to go
36
+
37
+ - **Tutorial:** <https://millwright-rs.dev/docs/> — the full, hands-on walk-through.
38
+ - **API reference:** [docs.rs/millwright](https://docs.rs/millwright).
39
+ - **Design brief (the *why*):** <https://millwright-rs.dev/>.
40
+ - **Examples:** [`examples/`](examples) — a runnable program for each feature group.
41
+ - **Python:** [`pip install millwright`](https://pypi.org/project/millwright/).
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: millwright
3
- Version: 0.1.1
3
+ Version: 0.2.1
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Programming Language :: Python :: 3
6
6
  Classifier: Intended Audience :: Science/Research
@@ -16,10 +16,19 @@ Project-URL: Repository, https://github.com/mi7plus/millwright
16
16
 
17
17
  # millwright
18
18
 
19
+ [![crates.io](https://img.shields.io/crates/v/millwright.svg)](https://crates.io/crates/millwright)
20
+ [![docs.rs](https://img.shields.io/docsrs/millwright)](https://docs.rs/millwright)
21
+ [![PyPI](https://img.shields.io/pypi/v/millwright.svg)](https://pypi.org/project/millwright/)
22
+ [![CI](https://github.com/mi7plus/millwright/actions/workflows/ci.yml/badge.svg)](https://github.com/mi7plus/millwright/actions/workflows/ci.yml)
23
+ [![downloads](https://img.shields.io/crates/d/millwright.svg)](https://crates.io/crates/millwright)
24
+ [![license](https://img.shields.io/crates/l/millwright.svg)](LICENSE)
25
+
19
26
  A unified ML framework for Rust — *ten crates, one lifecycle.*
20
27
 
21
- - **Tutorial:** [`GUIDE.md`](GUIDE.md) — the hands-on walk through the whole
22
- lifecycle (also as a page: [`guide.html`](guide.html)).
28
+ - **Tutorial:** the hands-on docs site at <https://millwright-rs.dev/docs/>
29
+ (source in [`docs/`](docs/index.html); a short [`GUIDE.md`](GUIDE.md) has the
30
+ quickstart).
31
+ - **Contributing:** see [`CONTRIBUTING.md`](CONTRIBUTING.md).
23
32
  - **Design brief:** the *why*, at **<https://millwright-rs.dev/>**.
24
33
 
25
34
  > *"Ten crates"* is the ecosystem this project assembles —
@@ -1,9 +1,18 @@
1
1
  # millwright
2
2
 
3
+ [![crates.io](https://img.shields.io/crates/v/millwright.svg)](https://crates.io/crates/millwright)
4
+ [![docs.rs](https://img.shields.io/docsrs/millwright)](https://docs.rs/millwright)
5
+ [![PyPI](https://img.shields.io/pypi/v/millwright.svg)](https://pypi.org/project/millwright/)
6
+ [![CI](https://github.com/mi7plus/millwright/actions/workflows/ci.yml/badge.svg)](https://github.com/mi7plus/millwright/actions/workflows/ci.yml)
7
+ [![downloads](https://img.shields.io/crates/d/millwright.svg)](https://crates.io/crates/millwright)
8
+ [![license](https://img.shields.io/crates/l/millwright.svg)](LICENSE)
9
+
3
10
  A unified ML framework for Rust — *ten crates, one lifecycle.*
4
11
 
5
- - **Tutorial:** [`GUIDE.md`](GUIDE.md) — the hands-on walk through the whole
6
- lifecycle (also as a page: [`guide.html`](guide.html)).
12
+ - **Tutorial:** the hands-on docs site at <https://millwright-rs.dev/docs/>
13
+ (source in [`docs/`](docs/index.html); a short [`GUIDE.md`](GUIDE.md) has the
14
+ quickstart).
15
+ - **Contributing:** see [`CONTRIBUTING.md`](CONTRIBUTING.md).
7
16
  - **Design brief:** the *why*, at **<https://millwright-rs.dev/>**.
8
17
 
9
18
  > *"Ten crates"* is the ecosystem this project assembles —
@@ -3,7 +3,19 @@
3
3
  Millwright ships as one Rust crate (crates.io) and one Python wheel (PyPI), from
4
4
  the same source.
5
5
 
6
- ## Preflight
6
+ ## The easy way
7
+
8
+ With `main` green and `CHANGELOG.md`'s `[Unreleased]` section filled in, run:
9
+
10
+ ```bash
11
+ bash scripts/release.sh 0.1.2
12
+ ```
13
+
14
+ It bumps both manifests, rolls the changelog, syncs `Cargo.lock`, verifies the
15
+ package, commits, tags `v0.1.2`, and (after a prompt) pushes — which triggers the
16
+ publish workflows below. The rest of this file is the manual equivalent.
17
+
18
+ ## Preflight (manual)
7
19
 
8
20
  1. `main` is green in CI and `CHANGELOG.md`'s `[Unreleased]` section is current.
9
21
  2. Bump the version in **both** `Cargo.toml` and `pyproject.toml`, move the
@@ -0,0 +1,114 @@
1
+ <html><head><meta http-equiv="Content-Type" content="text/html; charset=UTF-8"><title>Millwright · Data &amp; EDA</title>
2
+ <meta name="viewport" content="width=device-width, initial-scale=1">
3
+ <link rel="stylesheet" href="site.css">
4
+ </head><body><header class="top">
5
+ <div class="wrap">
6
+ <div class="brand"><a href="../index.html"><span class="mark">⚙</span>millwright</a><span class="ver">docs</span></div>
7
+ <nav>
8
+ <a href="index.html">home</a>
9
+ <a href="data.html" class="active">data &amp; EDA</a>
10
+ <a href="pipelines.html">pipelines</a>
11
+ <a href="insight.html">insight</a>
12
+ <a href="deploy.html">deploy</a>
13
+ <a href="python.html">python</a>
14
+ <a href="../index.html">design brief</a>
15
+ <a class="repo" href="https://github.com/mi7plus/millwright">GitHub ↗</a>
16
+ </nav>
17
+ </div>
18
+ </header>
19
+
20
+ <main>
21
+ <div class="wrap">
22
+ <div class="hero">
23
+ <div class="eyebrow">01 · data &amp; EDA</div>
24
+ <h1>The boundary type,<br>and the typed layer in front of it.</h1>
25
+ <p class="lede"><code class="inl">Frame</code> is the numeric boundary the whole API speaks. <code class="inl">Table</code> (feature <code class="inl">eda</code>) is the polars-backed, dtype-aware world that ingests real CSV/Parquet and lowers into it.</p>
26
+ </div>
27
+ </div>
28
+
29
+ <!-- FRAME -->
30
+ <section id="frame">
31
+ <div class="wrap">
32
+ <div class="head col">
33
+ <div class="eyebrow">Frame &amp; Dataset</div>
34
+ <h2>One contiguous <span class="mono">f64</span> buffer, plus a schema.</h2>
35
+ <p class="muted">Everything the <em>public</em> API speaks is a <code class="inl">Frame</code>: a contiguous, row-major <code class="inl">f64</code> buffer with named columns. It is what lets a linfa model and a smartcore <code class="inl">DenseMatrix</code> meet in one signature without your code ever naming their array versions — each backend converts <code class="inl">Frame</code> ⇄ its native type inside the adapter only. A <code class="inl">Dataset</code> pairs a frame with a target.</p>
36
+ </div>
37
+ <pre><span class="k">use</span> millwright::prelude::*;
38
+
39
+ <span class="k">let</span> x = <span class="f">Frame</span>::from_rows(
40
+ <span class="f">vec!</span>[<span class="f">vec!</span>[<span class="k">0.0</span>, <span class="k">0.1</span>], <span class="f">vec!</span>[<span class="k">0.4</span>, <span class="k">0.2</span>], <span class="f">vec!</span>[<span class="k">9.0</span>, <span class="k">9.1</span>], <span class="f">vec!</span>[<span class="k">9.4</span>, <span class="k">8.7</span>]],
41
+ <span class="f">vec!</span>[<span class="s">"a"</span>.into(), <span class="s">"b"</span>.into()],
42
+ )?;
43
+ <span class="k">assert_eq!</span>(x.shape(), (<span class="k">4</span>, <span class="k">2</span>)); <span class="c">// (rows, cols)</span>
44
+
45
+ <span class="k">let</span> train = <span class="f">Dataset</span>::new(x.clone(), <span class="f">vec!</span>[<span class="k">0.0</span>, <span class="k">0.0</span>, <span class="k">1.0</span>, <span class="k">1.0</span>])?;
46
+ <span class="k">let</span> _features = train.features(); <span class="c">// &Frame</span>
47
+ <span class="k">let</span> _target = train.target(); <span class="c">// &[f64]</span></pre>
48
+ <p class="tiny">The task — classification vs. regression — is inferred from the target: an all-integral target is class labels, anything else is regression. A pure-numeric CSV loads directly with <code class="inl">Frame::from_csv</code>; typed data uses <code class="inl">Table</code> below.</p>
49
+ </div>
50
+ </section>
51
+
52
+ <!-- INGEST -->
53
+ <section id="ingest">
54
+ <div class="wrap">
55
+ <div class="head col">
56
+ <div class="eyebrow">Ingest &amp; explore</div>
57
+ <h2><span class="mono">Table</span> reads it, <span class="mono">Profile</span> reports it.</h2>
58
+ <p class="muted">Behind the <code class="inl">eda</code> feature, a polars-backed <code class="inl">Table</code> reads real CSV/Parquet — strings, categories, dates, booleans, nulls — and a <code class="inl">Profile</code> returns a <em>typed</em> analysis (not just an HTML blob) and drafts the preprocessing.</p>
59
+ </div>
60
+ <pre><span class="k">let</span> table = <span class="f">Table</span>::from_csv(<span class="s">"customers.csv"</span>)?; <span class="c">// or ::from_parquet(…)</span>
61
+
62
+ <span class="c">// a typed profile — overview, per-column stats, missingness, correlations, alerts</span>
63
+ <span class="k">let</span> profile = <span class="f">Profile</span>::of_with_target(&amp;table, <span class="s">"churned"</span>)?;
64
+ <span class="f">println!</span>(<span class="s">"{}"</span>, profile.summary());
65
+ <span class="k">for</span> alert <span class="k">in</span> profile.alerts() {
66
+ <span class="f">println!</span>(<span class="s">"{alert}"</span>); <span class="c">// "[city] categorical (3 levels) → OneHotEncoder"</span>
67
+ }
68
+ profile.to_html(<span class="s">"eda_report.html"</span>)?; <span class="c">// a shareable report</span></pre>
69
+ <p class="muted">Because Millwright owns EDA <em>and</em> the pipeline, the profile drafts the starting preprocessing from its own findings — the loop scikit-learn can't close:</p>
70
+ <pre><span class="c">// lower the typed table to the numeric world</span>
71
+ <span class="k">let</span> train = table.into_dataset(<span class="s">"churned"</span>)?; <span class="c">// categoricals encoded, nulls → NaN</span>
72
+
73
+ <span class="c">// EDA drafts the pipeline; you just add the model</span>
74
+ <span class="k">let mut</span> pipe = profile.suggest_pipeline() <span class="c">// impute · encode · scale, from the alerts</span>
75
+ .estimator(<span class="s">"rf"</span>, <span class="f">RandomForest</span>::new());
76
+ pipe.fit(&amp;train)?;</pre>
77
+ <p class="run">cargo run --example explore --features "eda smartcore-backend"</p>
78
+ </div>
79
+ </section>
80
+
81
+ <!-- DTYPES -->
82
+ <section id="dtypes">
83
+ <div class="wrap">
84
+ <div class="head col">
85
+ <div class="eyebrow">Dtype-aware</div>
86
+ <h2>Types flow through the pipeline.</h2>
87
+ <p class="muted">A <code class="inl">Frame</code> carries a per-column <code class="inl">Dtype</code> (defaulting to <code class="inl">Numeric</code>). <code class="inl">Table</code> marks the columns it knows are <code class="inl">Categorical</code> as it lowers — so preprocessing doesn't have to <em>guess</em>: scalers, <code class="inl">Winsorize</code>, and <code class="inl">PowerTransform</code> pass categorical columns through untouched, and <code class="inl">OneHotEncoder</code> encodes by dtype rather than a value heuristic.</p>
88
+ </div>
89
+ <pre><span class="c">// a genuinely-integer feature is NOT wrongly one-hot'd; the categorical one is</span>
90
+ <span class="k">let</span> f = <span class="f">Frame</span>::from_rows(rows, cols)?
91
+ .with_dtypes(<span class="f">vec!</span>[<span class="f">Dtype</span>::Categorical, <span class="f">Dtype</span>::Numeric])?;
92
+ <span class="k">let</span> out = <span class="f">OneHotEncoder</span>::infer().fit_transform(&amp;f)?; <span class="c">// expands only column 0</span>
93
+
94
+ <span class="c">// or one-hot at the Table boundary, with real category names</span>
95
+ <span class="k">let</span> train = table.into_dataset_with(<span class="s">"churned"</span>, <span class="f">CategoryEncoding</span>::OneHot)?;</pre>
96
+ <p class="tiny">Nominal categories become <code class="inl">"{col}={value}"</code> indicator columns instead of ordinal codes — the correct representation for linear and tree models.</p>
97
+ </div>
98
+ </section>
99
+
100
+ <div class="wrap">
101
+ <div class="pager">
102
+ <a href="index.html"><span class="dir">← prev</span><b>Home</b></a>
103
+ <a class="next" href="pipelines.html"><span class="dir">next →</span><b>Pipelines &amp; Models</b></a>
104
+ </div>
105
+ </div>
106
+ </main>
107
+
108
+ <footer>
109
+ <div class="wrap">
110
+ <span class="mono">⚙ millwright docs</span>
111
+ <span class="mono"><a href="../index.html">design brief</a> · <a href="https://crates.io/crates/millwright">crates.io</a> · <a href="https://pypi.org/project/millwright/">PyPI</a> · <a href="https://docs.rs/millwright">docs.rs</a> · <a href="https://github.com/mi7plus/millwright">GitHub</a></span>
112
+ </div>
113
+ </footer>
114
+ </body></html>