geds-python 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- geds/__init__.py +25 -0
- geds/_backend.py +458 -0
- geds/_estimators.py +844 -0
- geds/_plotting.py +69 -0
- geds/_validation.py +148 -0
- geds/check.py +64 -0
- geds/py.typed +1 -0
- geds_python-0.1.0.dist-info/METADATA +406 -0
- geds_python-0.1.0.dist-info/RECORD +11 -0
- geds_python-0.1.0.dist-info/WHEEL +4 -0
- geds_python-0.1.0.dist-info/licenses/LICENSE +674 -0
geds/_plotting.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Optional Python-native plotting helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import pandas as pd
|
|
9
|
+
from sklearn.utils.validation import check_is_fitted
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def plot_fit(
|
|
13
|
+
estimator: Any,
|
|
14
|
+
X: Any,
|
|
15
|
+
y: Any | None = None,
|
|
16
|
+
*,
|
|
17
|
+
ax: Any | None = None,
|
|
18
|
+
grid_size: int = 500,
|
|
19
|
+
show_knots: bool = True,
|
|
20
|
+
) -> Any:
|
|
21
|
+
"""Plot a fitted univariate GeDS model and return its Matplotlib axes.
|
|
22
|
+
|
|
23
|
+
This helper only visualizes predictions already produced by GeDS; it does
|
|
24
|
+
not implement any statistical calculation in Python.
|
|
25
|
+
"""
|
|
26
|
+
try:
|
|
27
|
+
import matplotlib.pyplot as plt
|
|
28
|
+
except ImportError as exc: # pragma: no cover - depends on optional extra
|
|
29
|
+
raise ImportError(
|
|
30
|
+
'Plotting requires Matplotlib; install "geds-python[plot]".'
|
|
31
|
+
) from exc
|
|
32
|
+
|
|
33
|
+
check_is_fitted(estimator, "_r_model_")
|
|
34
|
+
frame, named_input = estimator._frame(X)
|
|
35
|
+
if frame.shape[1] != 1 or estimator.n_features_in_ != 1:
|
|
36
|
+
raise ValueError("plot_fit supports fitted models with one feature only.")
|
|
37
|
+
if grid_size < 2:
|
|
38
|
+
raise ValueError("grid_size must be at least 2.")
|
|
39
|
+
values = np.asarray(frame.iloc[:, 0], dtype=float)
|
|
40
|
+
if not np.isfinite(values).all():
|
|
41
|
+
raise ValueError("X must contain only finite values.")
|
|
42
|
+
grid_values = np.linspace(values.min(), values.max(), grid_size)
|
|
43
|
+
if named_input:
|
|
44
|
+
grid = pd.DataFrame({frame.columns[0]: grid_values})
|
|
45
|
+
else:
|
|
46
|
+
grid = grid_values.reshape(-1, 1)
|
|
47
|
+
fitted = estimator.predict(grid)
|
|
48
|
+
|
|
49
|
+
if ax is None:
|
|
50
|
+
_, ax = plt.subplots()
|
|
51
|
+
if y is not None:
|
|
52
|
+
response = np.asarray(y, dtype=float)
|
|
53
|
+
if response.ndim != 1 or len(response) != len(values):
|
|
54
|
+
raise ValueError("y must be one-dimensional and have the same length as X.")
|
|
55
|
+
ax.scatter(values, response, s=12, alpha=0.35, label="Data")
|
|
56
|
+
ax.plot(grid_values, fitted, linewidth=2, label="GeDS fit")
|
|
57
|
+
if show_knots and estimator.knots_ is not None:
|
|
58
|
+
knots = np.asarray(estimator.knots_, dtype=float).ravel()
|
|
59
|
+
for index, knot in enumerate(knots):
|
|
60
|
+
ax.axvline(
|
|
61
|
+
knot,
|
|
62
|
+
color="tab:red",
|
|
63
|
+
linestyle="--",
|
|
64
|
+
alpha=0.55,
|
|
65
|
+
label="Internal knots" if index == 0 else None,
|
|
66
|
+
)
|
|
67
|
+
ax.set(xlabel=str(frame.columns[0]), ylabel="Response", title="GeDS spline regression")
|
|
68
|
+
ax.legend()
|
|
69
|
+
return ax
|
geds/_validation.py
ADDED
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""Python inputs for R GeDS's specialized cross-validation routine."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any, Mapping, Sequence
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
import pandas as pd
|
|
10
|
+
from sklearn.base import clone
|
|
11
|
+
|
|
12
|
+
from ._backend import get_backend
|
|
13
|
+
from ._estimators import (
|
|
14
|
+
GeDSBoostRegressor, GeDSGAMRegressor, GeDSGeneralizedRegressor,
|
|
15
|
+
GeDSRegressor,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class GeDSCrossValidationResult:
|
|
21
|
+
"""R's best parameter combination and full cross-validation table."""
|
|
22
|
+
|
|
23
|
+
best_params: pd.DataFrame
|
|
24
|
+
results: pd.DataFrame
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def cross_validate_geds(
|
|
28
|
+
estimator: GeDSRegressor | GeDSGeneralizedRegressor | GeDSGAMRegressor | GeDSBoostRegressor,
|
|
29
|
+
X: Any,
|
|
30
|
+
y: Any,
|
|
31
|
+
parameter_grid: Mapping[str, Sequence[float]],
|
|
32
|
+
*,
|
|
33
|
+
n_folds: int = 5,
|
|
34
|
+
n_cores: int = 1,
|
|
35
|
+
random_state: int | None = None,
|
|
36
|
+
) -> GeDSCrossValidationResult:
|
|
37
|
+
"""Tune GeDS parameters with R ``crossv_GeDS()`` (Gaussian models only).
|
|
38
|
+
|
|
39
|
+
This is R's specialized grid search, not scikit-learn cross-validation.
|
|
40
|
+
It returns R's MSE and knot/iteration summary without fitting the input
|
|
41
|
+
estimator. R currently applies its own fitting defaults for non-grid
|
|
42
|
+
options in NGeDS, GGeDS, and NGeDSgam cross-validation.
|
|
43
|
+
"""
|
|
44
|
+
if not isinstance(estimator, (
|
|
45
|
+
GeDSRegressor, GeDSGeneralizedRegressor,
|
|
46
|
+
GeDSGAMRegressor, GeDSBoostRegressor,
|
|
47
|
+
)):
|
|
48
|
+
raise TypeError("estimator must be a GeDS Python estimator.")
|
|
49
|
+
if not isinstance(n_folds, (int, np.integer)) or n_folds < 2:
|
|
50
|
+
raise ValueError("n_folds must be an integer of at least 2.")
|
|
51
|
+
if not isinstance(n_cores, (int, np.integer)) or n_cores < 1:
|
|
52
|
+
raise ValueError("n_cores must be a positive integer.")
|
|
53
|
+
if random_state is not None and (
|
|
54
|
+
not isinstance(random_state, (int, np.integer)) or random_state < 0
|
|
55
|
+
):
|
|
56
|
+
raise ValueError("random_state must be a non-negative integer or None.")
|
|
57
|
+
if hasattr(estimator, "family") and estimator.family.lower() != "gaussian":
|
|
58
|
+
raise ValueError(
|
|
59
|
+
"R crossv_GeDS currently uses the Gaussian default for these "
|
|
60
|
+
"models; non-Gaussian estimators are not supported here."
|
|
61
|
+
)
|
|
62
|
+
if hasattr(estimator, "link") and estimator.link is not None:
|
|
63
|
+
raise ValueError("A custom link is not supported by R crossv_GeDS.")
|
|
64
|
+
|
|
65
|
+
# R's non-boost cross-validation routine only forwards beta, phi, q, and
|
|
66
|
+
# the order-derived higher_order flag. Reject other custom settings so
|
|
67
|
+
# they are never silently ignored.
|
|
68
|
+
if not isinstance(estimator, GeDSBoostRegressor):
|
|
69
|
+
defaults_estimator = type(estimator)()
|
|
70
|
+
forwarded = {
|
|
71
|
+
"spline_features", "spline_terms", "linear_features",
|
|
72
|
+
"order", "higher_order", "beta", "phi", "q", "family", "link",
|
|
73
|
+
}
|
|
74
|
+
unsupported = [
|
|
75
|
+
name for name, value in estimator.get_params(deep=False).items()
|
|
76
|
+
if name not in forwarded
|
|
77
|
+
and not np.array_equal(value, getattr(defaults_estimator, name))
|
|
78
|
+
]
|
|
79
|
+
if unsupported:
|
|
80
|
+
raise ValueError(
|
|
81
|
+
"R crossv_GeDS does not forward these custom settings: "
|
|
82
|
+
+ ", ".join(sorted(unsupported))
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
working = clone(estimator)
|
|
86
|
+
working._validate_configuration()
|
|
87
|
+
if isinstance(working, (GeDSGAMRegressor, GeDSBoostRegressor)):
|
|
88
|
+
frame, formula, _ = working._prepare_additive_data(X, y, None)
|
|
89
|
+
else:
|
|
90
|
+
frame, formula, _ = working._prepare_training_data(X, y, None, None)
|
|
91
|
+
if n_folds > len(frame):
|
|
92
|
+
raise ValueError("n_folds cannot exceed the number of observations.")
|
|
93
|
+
|
|
94
|
+
defaults: dict[str, float] = {
|
|
95
|
+
"beta": 0.5 if working.beta is None else float(working.beta),
|
|
96
|
+
"phi": float(working.phi),
|
|
97
|
+
"q": float(working.q),
|
|
98
|
+
}
|
|
99
|
+
if isinstance(working, GeDSBoostRegressor):
|
|
100
|
+
defaults.update({
|
|
101
|
+
"int_knots_init": float(working.int_knots_init),
|
|
102
|
+
"shrinkage": float(working.shrinkage),
|
|
103
|
+
})
|
|
104
|
+
extra = set(parameter_grid) - set(defaults)
|
|
105
|
+
if extra:
|
|
106
|
+
raise ValueError(f"Unsupported parameter grid names: {sorted(extra)}.")
|
|
107
|
+
parameters: dict[str, np.ndarray] = {}
|
|
108
|
+
for name, default in defaults.items():
|
|
109
|
+
values = np.asarray(parameter_grid.get(name, [default]), dtype=float)
|
|
110
|
+
if values.ndim != 1 or not len(values) or not np.isfinite(values).all():
|
|
111
|
+
raise ValueError(f"{name} grid must contain finite numeric values.")
|
|
112
|
+
if name in {"beta", "phi"} and np.any((values < 0) | (values > 1)):
|
|
113
|
+
raise ValueError(f"{name} grid values must lie in [0, 1].")
|
|
114
|
+
if name in {"q", "int_knots_init"} and (
|
|
115
|
+
np.any(values != np.floor(values))
|
|
116
|
+
or np.any(values < (1 if name == "q" else 0))
|
|
117
|
+
):
|
|
118
|
+
raise ValueError(f"{name} grid values must be valid non-negative integers.")
|
|
119
|
+
if name == "shrinkage" and np.any((values <= 0) | (values > 1)):
|
|
120
|
+
raise ValueError("shrinkage grid values must lie in (0, 1].")
|
|
121
|
+
r_name = "int.knots_init" if name == "int_knots_init" else name
|
|
122
|
+
parameters[f"{r_name}_grid"] = values
|
|
123
|
+
|
|
124
|
+
if isinstance(working, GeDSBoostRegressor):
|
|
125
|
+
model_name = "NGeDSboost"
|
|
126
|
+
fit_kwargs = {
|
|
127
|
+
"max_iterations": working.max_iterations,
|
|
128
|
+
"min_iterations": working.min_iterations,
|
|
129
|
+
"normalize_data": working.normalize_data,
|
|
130
|
+
"initial_learner": working.initial_learner,
|
|
131
|
+
"int.knots_boost": working.int_knots_boost,
|
|
132
|
+
"phi_boost_exit": working.phi_boost_exit,
|
|
133
|
+
"q_boost": working.q_boost,
|
|
134
|
+
"boosting_with_memory": working.boosting_with_memory,
|
|
135
|
+
}
|
|
136
|
+
fit_kwargs = {k: v for k, v in fit_kwargs.items() if v is not None}
|
|
137
|
+
elif isinstance(working, GeDSGAMRegressor):
|
|
138
|
+
model_name, fit_kwargs = "NGeDSgam", {}
|
|
139
|
+
elif isinstance(working, GeDSGeneralizedRegressor):
|
|
140
|
+
model_name, fit_kwargs = "GGeDS", {}
|
|
141
|
+
else:
|
|
142
|
+
model_name, fit_kwargs = "NGeDS", {}
|
|
143
|
+
|
|
144
|
+
best, results = get_backend().cross_validate(
|
|
145
|
+
model_name, formula, frame, parameters, working.order,
|
|
146
|
+
int(n_folds), int(n_cores), random_state, **fit_kwargs,
|
|
147
|
+
)
|
|
148
|
+
return GeDSCrossValidationResult(best_params=best, results=results)
|
geds/check.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Command-line environment check for the Python/R bridge."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
from typing import Sequence
|
|
9
|
+
|
|
10
|
+
from ._backend import BackendUnavailableError, diagnostics
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _parser() -> argparse.ArgumentParser:
|
|
14
|
+
parser = argparse.ArgumentParser(
|
|
15
|
+
prog="python -m geds.check",
|
|
16
|
+
description="Check that Python can initialize R and load the GeDS package.",
|
|
17
|
+
)
|
|
18
|
+
parser.add_argument(
|
|
19
|
+
"--json",
|
|
20
|
+
action="store_true",
|
|
21
|
+
help="print machine-readable JSON instead of the human-readable report",
|
|
22
|
+
)
|
|
23
|
+
return parser
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
27
|
+
"""Run the environment check and return a process exit code."""
|
|
28
|
+
args = _parser().parse_args(argv)
|
|
29
|
+
try:
|
|
30
|
+
information = diagnostics()
|
|
31
|
+
except BackendUnavailableError as exc:
|
|
32
|
+
if args.json:
|
|
33
|
+
print(json.dumps({"status": "error", "message": str(exc)}, indent=2))
|
|
34
|
+
else:
|
|
35
|
+
print("GeDS environment check: FAILED", file=sys.stderr)
|
|
36
|
+
print(str(exc), file=sys.stderr)
|
|
37
|
+
print(
|
|
38
|
+
"Set R_HOME if the intended R installation is not discovered, "
|
|
39
|
+
"and set GEDS_R_LIBRARY if GeDS is in a non-default R library.",
|
|
40
|
+
file=sys.stderr,
|
|
41
|
+
)
|
|
42
|
+
return 1
|
|
43
|
+
|
|
44
|
+
report = {"status": "ok", **information}
|
|
45
|
+
if args.json:
|
|
46
|
+
print(json.dumps(report, indent=2, sort_keys=True))
|
|
47
|
+
else:
|
|
48
|
+
labels = {
|
|
49
|
+
"python": "Python",
|
|
50
|
+
"r_version": "R",
|
|
51
|
+
"r_home": "R home",
|
|
52
|
+
"rpy2_version": "rpy2",
|
|
53
|
+
"geds_version": "GeDS",
|
|
54
|
+
"geds_library": "GeDS library",
|
|
55
|
+
"minimum_geds_version": "Minimum GeDS",
|
|
56
|
+
}
|
|
57
|
+
print("GeDS environment check: OK")
|
|
58
|
+
for key, label in labels.items():
|
|
59
|
+
print(f" {label}: {information[key]}")
|
|
60
|
+
return 0
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
if __name__ == "__main__": # pragma: no cover - exercised as a module
|
|
64
|
+
raise SystemExit(main())
|
geds/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,406 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: geds-python
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Python estimators backed by the GeDS R package
|
|
5
|
+
Project-URL: Homepage, https://github.com/emilioluissaenzguillen/GeDS-python
|
|
6
|
+
Project-URL: Repository, https://github.com/emilioluissaenzguillen/GeDS-python
|
|
7
|
+
Project-URL: Issues, https://github.com/emilioluissaenzguillen/GeDS-python/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/emilioluissaenzguillen/GeDS-python/blob/main/CHANGELOG.md
|
|
9
|
+
Project-URL: R package, https://github.com/emilioluissaenzguillen/GeDS
|
|
10
|
+
Author: Dimitrina S. Dimitrova, Vladimir K. Kaishev, Andrea Lattuada, Emilio L. Sáenz Guillén, Richard J. Verrall
|
|
11
|
+
Maintainer-email: "Emilio L. Sáenz Guillén" <emilioluissaenzguillen@gmail.com>
|
|
12
|
+
License-Expression: GPL-3.0-only
|
|
13
|
+
License-File: LICENSE
|
|
14
|
+
Keywords: R,regression,scikit-learn,splines,statistics
|
|
15
|
+
Classifier: Development Status :: 4 - Beta
|
|
16
|
+
Classifier: Intended Audience :: Science/Research
|
|
17
|
+
Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
|
|
18
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
19
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
20
|
+
Classifier: Programming Language :: Python :: 3
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering
|
|
25
|
+
Requires-Python: >=3.10
|
|
26
|
+
Requires-Dist: numpy>=1.24
|
|
27
|
+
Requires-Dist: pandas>=2.0
|
|
28
|
+
Requires-Dist: rpy2<3.7,>=3.6.7
|
|
29
|
+
Requires-Dist: scikit-learn>=1.4
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: build>=1.2; extra == 'dev'
|
|
32
|
+
Requires-Dist: matplotlib>=3.8; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
34
|
+
Provides-Extra: plot
|
|
35
|
+
Requires-Dist: matplotlib>=3.8; extra == 'plot'
|
|
36
|
+
Provides-Extra: test
|
|
37
|
+
Requires-Dist: pytest>=8; extra == 'test'
|
|
38
|
+
Description-Content-Type: text/markdown
|
|
39
|
+
|
|
40
|
+
# GeDS for Python
|
|
41
|
+
|
|
42
|
+
This package provides a Python interface to the
|
|
43
|
+
[GeDS R package](https://github.com/emilioluissaenzguillen/GeDS). The R package
|
|
44
|
+
is the sole implementation of the statistical methods. Python supplies a
|
|
45
|
+
scikit-learn-style API, pandas/NumPy conversion, environment diagnostics, and
|
|
46
|
+
model serialization.
|
|
47
|
+
|
|
48
|
+
## Requirements
|
|
49
|
+
|
|
50
|
+
- R 4.4 or newer (R 4.6.1 is used for development)
|
|
51
|
+
- GeDS 0.3.6 or newer with the Python bridge fixes
|
|
52
|
+
- Python 3.10 or newer
|
|
53
|
+
|
|
54
|
+
Install the Python package, including the optional plotting dependency used in
|
|
55
|
+
the example:
|
|
56
|
+
|
|
57
|
+
```console
|
|
58
|
+
python -m pip install "geds-python[plot]"
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Install the R package separately, using R 4.6.1 or another supported R
|
|
62
|
+
installation. Install the tested GeDS 0.3.6 build from GitHub:
|
|
63
|
+
|
|
64
|
+
```r
|
|
65
|
+
install.packages("remotes")
|
|
66
|
+
remotes::install_git(
|
|
67
|
+
"https://github.com/emilioluissaenzguillen/GeDS.git",
|
|
68
|
+
ref = "91b8ddd13aae8f39994c87fc356f05da4799f911",
|
|
69
|
+
dependencies = NA, upgrade = "never"
|
|
70
|
+
)
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
An older GeDS build, even one reporting version `0.3.6`, may lack the fixes
|
|
74
|
+
required by this wrapper. `install.packages("GeDS")` alone is not guaranteed
|
|
75
|
+
to provide them while the CRAN review follows its separate schedule.
|
|
76
|
+
Check the installed R version with `packageVersion("GeDS")`.
|
|
77
|
+
|
|
78
|
+
On Windows, building the GitHub source package requires Rtools compatible
|
|
79
|
+
with the selected R installation. GeDS remains version `0.3.6` on GitHub;
|
|
80
|
+
the wrapper checks an internal compatibility marker for the fit and prediction
|
|
81
|
+
fixes as well as the package version.
|
|
82
|
+
|
|
83
|
+
The wrapper discovers the newest R installation under `Program Files/R` on
|
|
84
|
+
Windows or uses `Rscript` from `PATH` on other platforms. Set `R_HOME` to select
|
|
85
|
+
a particular R installation. If GeDS is installed in a non-default R library,
|
|
86
|
+
set `GEDS_R_LIBRARY` to that library directory before importing `geds`.
|
|
87
|
+
|
|
88
|
+
The Python and R packages have independent release cycles. `geds-python`
|
|
89
|
+
checks the installed GeDS version when its backend first starts and reports the
|
|
90
|
+
selected R installation and package library through `geds.diagnostics()`.
|
|
91
|
+
|
|
92
|
+
Check the backend before fitting:
|
|
93
|
+
|
|
94
|
+
```console
|
|
95
|
+
python -m geds.check
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
For a machine-readable report, use `python -m geds.check --json`. The same
|
|
99
|
+
information is available inside Python:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
import geds
|
|
103
|
+
|
|
104
|
+
print(geds.diagnostics())
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
### Selecting R and its package library
|
|
108
|
+
|
|
109
|
+
Usually no configuration is necessary. If several R installations are
|
|
110
|
+
available, select one before starting Python:
|
|
111
|
+
|
|
112
|
+
```powershell
|
|
113
|
+
$env:R_HOME = "C:\Program Files\R\R-4.6.1"
|
|
114
|
+
python -m geds.check
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
export R_HOME="/Library/Frameworks/R.framework/Resources" # macOS
|
|
119
|
+
# export R_HOME="/usr/lib/R" # Linux
|
|
120
|
+
python -m geds.check
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
If GeDS is installed in a personal or otherwise non-default R library, set
|
|
124
|
+
`GEDS_R_LIBRARY` to the directory that contains the `GeDS` folder. You can
|
|
125
|
+
find that directory from R with `find.package("GeDS")`; use its parent
|
|
126
|
+
directory as `GEDS_R_LIBRARY`.
|
|
127
|
+
|
|
128
|
+
If the check reports that R is missing, install R or set `R_HOME`. If it finds
|
|
129
|
+
R but not GeDS, start that same R installation and run
|
|
130
|
+
one of the GeDS installation commands above, then rerun the check.
|
|
131
|
+
|
|
132
|
+
## Example
|
|
133
|
+
|
|
134
|
+
Install the optional plotting dependency with
|
|
135
|
+
`python -m pip install "geds-python[plot]"`, then fit and visualize a nonlinear
|
|
136
|
+
regression:
|
|
137
|
+
|
|
138
|
+
```python
|
|
139
|
+
import matplotlib.pyplot as plt
|
|
140
|
+
import numpy as np
|
|
141
|
+
import pandas as pd
|
|
142
|
+
|
|
143
|
+
from geds import GeDSRegressor, plot_fit
|
|
144
|
+
|
|
145
|
+
rng = np.random.RandomState(123)
|
|
146
|
+
n = 500
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def f_1(x):
|
|
150
|
+
return (10 * x / (1 + 100 * x**2)) * 4 + 4
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
x = np.sort(rng.uniform(-2.0, 2.0, size=n))
|
|
154
|
+
means = f_1(x)
|
|
155
|
+
y = rng.normal(means, scale=0.1)
|
|
156
|
+
X = pd.DataFrame({"x": x})
|
|
157
|
+
|
|
158
|
+
model = GeDSRegressor(order=3).fit(X, y)
|
|
159
|
+
knots = np.asarray(model.knots_, dtype=float)
|
|
160
|
+
|
|
161
|
+
print("Internal knots:", knots)
|
|
162
|
+
|
|
163
|
+
fig, ax = plt.subplots()
|
|
164
|
+
plot_fit(model, X, y, ax=ax)
|
|
165
|
+
grid_x = np.linspace(x.min(), x.max(), 500)
|
|
166
|
+
ax.plot(
|
|
167
|
+
grid_x,
|
|
168
|
+
f_1(grid_x),
|
|
169
|
+
color="0.25",
|
|
170
|
+
linestyle=":",
|
|
171
|
+
linewidth=2,
|
|
172
|
+
label="True mean",
|
|
173
|
+
)
|
|
174
|
+
ax.set(ylabel="y")
|
|
175
|
+
ax.legend()
|
|
176
|
+
fig.tight_layout()
|
|
177
|
+
plt.show()
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
With GeDS 0.3.6 and R 4.6.1, this seeded example fits 16 internal knots.
|
|
181
|
+
The dashed vertical lines show how GeDS places more knots around the sharp
|
|
182
|
+
variation near zero while retaining knots across the wider domain.
|
|
183
|
+
|
|
184
|
+
`GeDSRegressor` delegates to `GeDS::NGeDS()`. For exponential-family models,
|
|
185
|
+
use `GeDSGeneralizedRegressor`, which delegates to `GeDS::GGeDS()`.
|
|
186
|
+
|
|
187
|
+
For fitted models, `get_deviance(order=...)`, `get_log_likelihood(order=...)`,
|
|
188
|
+
and `get_confidence_intervals(order=..., level=...)` call the corresponding R
|
|
189
|
+
methods. Confidence intervals are returned as a pandas DataFrame with `lower`
|
|
190
|
+
and `upper` columns. As in R, these are coefficient intervals, not confidence
|
|
191
|
+
bands for the fitted curve.
|
|
192
|
+
|
|
193
|
+
The estimators also work with standard scikit-learn tools such as
|
|
194
|
+
`cross_val_score()` and `GridSearchCV`. Use sequential execution (`n_jobs=1`)
|
|
195
|
+
when cross-validating: the wrapper embeds R in the Python process, and
|
|
196
|
+
parallel-worker behavior is not part of the supported interface.
|
|
197
|
+
|
|
198
|
+
R also has a specialized `crossv_GeDS()` routine, which returns a parameter
|
|
199
|
+
grid with cross-validated mean squared error and knot/iteration summaries.
|
|
200
|
+
Use its Python interface when those R-specific results are needed:
|
|
201
|
+
|
|
202
|
+
```python
|
|
203
|
+
from geds import cross_validate_geds
|
|
204
|
+
|
|
205
|
+
cv = cross_validate_geds(
|
|
206
|
+
GeDSRegressor(order=3), X, y,
|
|
207
|
+
{"beta": [0.5, 0.7], "phi": [0.95], "q": [2]},
|
|
208
|
+
n_folds=5, n_cores=1, random_state=123,
|
|
209
|
+
)
|
|
210
|
+
print(cv.best_params)
|
|
211
|
+
print(cv.results)
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
This delegates the entire search to R and does not fit or change the input
|
|
215
|
+
estimator. It currently supports Gaussian models only, accepts the R tuning
|
|
216
|
+
parameters `beta`, `phi`, `q`, and (for boosting) `int_knots_init` and
|
|
217
|
+
`shrinkage`, and defaults to one R worker. R's current non-boost routine does
|
|
218
|
+
not forward other fitting settings; the Python interface rejects custom
|
|
219
|
+
settings it would otherwise silently ignore. Use scikit-learn's grid search
|
|
220
|
+
when you need those settings or a non-Gaussian family.
|
|
221
|
+
|
|
222
|
+
For a fitted univariate spline without extra linear features, R's calculus
|
|
223
|
+
and spline-conversion utilities are available as model methods:
|
|
224
|
+
|
|
225
|
+
```python
|
|
226
|
+
slopes = model.derive([-0.5, 0.0, 0.5], derivative_order=1)
|
|
227
|
+
areas = model.integrate(-1.0, [-0.5, 0.0, 0.5])
|
|
228
|
+
piece_knots, piece_coefficients = model.piecewise_polynomial()
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
`derive()` and `integrate()` operate on the predictor (link) scale, as in R.
|
|
232
|
+
`piecewise_polynomial()` returns the R `PPolyRep()` knot vector and coefficient
|
|
233
|
+
matrix; its last coefficient row is extraneous in R's representation. These
|
|
234
|
+
methods use the estimator's selected spline order unless `order=` is given.
|
|
235
|
+
|
|
236
|
+
For a normal univariate fit, impose a shape constraint without changing the
|
|
237
|
+
original fitted model:
|
|
238
|
+
|
|
239
|
+
```python
|
|
240
|
+
increasing_model = model.shape_constrain("increasing")
|
|
241
|
+
increasing_and_convex = model.shape_constrain(["increasing", "convex"])
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
This calls R's `shapeConstrain()` and returns a new Python estimator. R also
|
|
245
|
+
supports constraints on one selected univariate smoother in Gaussian GAM and
|
|
246
|
+
boosting fits, via `shape_constrain(..., base_learner="f(x)")`. Those additive
|
|
247
|
+
fits must use `normalize_data=False`. Constrained fits do not provide the usual
|
|
248
|
+
unconstrained coefficient confidence intervals.
|
|
249
|
+
|
|
250
|
+
For count data, the generalized estimator uses `GeDS::GGeDS()` and supports
|
|
251
|
+
both response-scale and link-scale prediction:
|
|
252
|
+
|
|
253
|
+
```python
|
|
254
|
+
import numpy as np
|
|
255
|
+
import pandas as pd
|
|
256
|
+
from geds import GeDSGeneralizedRegressor, plot_fit
|
|
257
|
+
|
|
258
|
+
rng = np.random.default_rng(123)
|
|
259
|
+
x = np.sort(rng.uniform(-2, 2, 120))
|
|
260
|
+
X = pd.DataFrame({"x": x})
|
|
261
|
+
counts = rng.poisson(np.exp(1 + np.sin(x)))
|
|
262
|
+
|
|
263
|
+
model = GeDSGeneralizedRegressor(
|
|
264
|
+
family="poisson", beta=0.2, phi=0.95, min_internal_knots=3
|
|
265
|
+
).fit(X, counts)
|
|
266
|
+
mean_counts = model.predict(X)
|
|
267
|
+
log_mean_counts = model.predict_link(X)
|
|
268
|
+
ax = plot_fit(model, X, counts)
|
|
269
|
+
```
|
|
270
|
+
|
|
271
|
+
`min_internal_knots` controls the minimum number of stage-A knots; it is used
|
|
272
|
+
here to make a small sample's fitted spline visible. GeDS determines the
|
|
273
|
+
final knot positions.
|
|
274
|
+
|
|
275
|
+
For a univariate spline with a known offset (for example log exposure in a
|
|
276
|
+
Poisson model), pass one offset value per observation to both fitting and
|
|
277
|
+
prediction. These values are on the link scale:
|
|
278
|
+
|
|
279
|
+
```python
|
|
280
|
+
import numpy as np
|
|
281
|
+
import pandas as pd
|
|
282
|
+
from geds import GeDSGeneralizedRegressor
|
|
283
|
+
|
|
284
|
+
rng = np.random.default_rng(321)
|
|
285
|
+
x = np.linspace(-1.5, 1.5, 90)
|
|
286
|
+
X = pd.DataFrame({"x": x})
|
|
287
|
+
exposure = np.linspace(1.1, 2.0, len(x))
|
|
288
|
+
counts = rng.poisson(exposure * np.exp(1.4 + np.sin(x))) + 1
|
|
289
|
+
log_exposure = np.log(exposure)
|
|
290
|
+
model = GeDSGeneralizedRegressor(
|
|
291
|
+
family="poisson", spline_features=["x"], order=2,
|
|
292
|
+
higher_order=False,
|
|
293
|
+
).fit(X, counts, offset=log_exposure)
|
|
294
|
+
expected_counts = model.predict(X, offset=log_exposure)
|
|
295
|
+
contributions = model.predict_terms(X, offset=log_exposure)
|
|
296
|
+
```
|
|
297
|
+
|
|
298
|
+
`predict_terms()` returns a DataFrame of the R spline and parametric term
|
|
299
|
+
contributions. Its rows sum to the link prediction after adding the offset;
|
|
300
|
+
the offset is not itself a term column. Offset prediction currently supports
|
|
301
|
+
one spline feature only because the R bivariate prediction method does not
|
|
302
|
+
apply new-data offsets consistently. A model fitted with an offset requires
|
|
303
|
+
an offset at prediction time.
|
|
304
|
+
|
|
305
|
+
This offset interface requires the GeDS GitHub fit and prediction fixes. A
|
|
306
|
+
earlier GeDS `0.3.6` installation without those fixes may mishandle
|
|
307
|
+
generalized-model offsets; the backend rejects that version.
|
|
308
|
+
|
|
309
|
+
Choose spline and parametric components explicitly for mixed data:
|
|
310
|
+
|
|
311
|
+
```python
|
|
312
|
+
model = GeDSRegressor(
|
|
313
|
+
spline_features=["x"],
|
|
314
|
+
linear_features=["group"],
|
|
315
|
+
).fit(X, y)
|
|
316
|
+
```
|
|
317
|
+
|
|
318
|
+
Spline features must be numeric. Parametric features may be numeric or
|
|
319
|
+
categorical; their encoding is performed by the R package so fitting and
|
|
320
|
+
prediction use R's native factor semantics.
|
|
321
|
+
If `spline_features` is omitted, all columns are used in a single joint spline
|
|
322
|
+
term. Select `spline_features=["x"]` and `linear_features=["group"]` to keep
|
|
323
|
+
`group` parametric instead. Two spline features create a joint bivariate
|
|
324
|
+
surface, not two separate additive smooths; R's support for more than two
|
|
325
|
+
spline features is experimental. With named pandas columns, prediction may
|
|
326
|
+
receive columns in a different order because the wrapper restores the fitted
|
|
327
|
+
column order before calling R.
|
|
328
|
+
|
|
329
|
+
### Additive GAM and boosting models
|
|
330
|
+
|
|
331
|
+
Use `GeDSGAMRegressor` for R's `NGeDSgam()` and `GeDSBoostRegressor` for
|
|
332
|
+
`NGeDSboost()`. Each entry in `spline_terms` is one additive smooth. Put two
|
|
333
|
+
features in the same entry for a joint surface. If omitted, each non-linear
|
|
334
|
+
feature gets its own smooth; `linear_features` selects parametric terms.
|
|
335
|
+
|
|
336
|
+
```python
|
|
337
|
+
import numpy as np
|
|
338
|
+
import pandas as pd
|
|
339
|
+
from geds import GeDSGAMRegressor, GeDSBoostRegressor
|
|
340
|
+
|
|
341
|
+
x = np.linspace(-2, 2, 100)
|
|
342
|
+
X = pd.DataFrame({"x": x, "z": x**2})
|
|
343
|
+
y = np.sin(x) + 0.3 * x**2
|
|
344
|
+
|
|
345
|
+
gam = GeDSGAMRegressor(
|
|
346
|
+
spline_terms=[("x",), ("z",)], max_iterations=10
|
|
347
|
+
).fit(X, y)
|
|
348
|
+
boost = GeDSBoostRegressor(
|
|
349
|
+
spline_terms=[("x",), ("z",)], max_iterations=20
|
|
350
|
+
).fit(X, y)
|
|
351
|
+
|
|
352
|
+
gam_predictions = gam.predict(X)
|
|
353
|
+
boost_predictions = boost.predict(X)
|
|
354
|
+
x_contribution = gam.predict_component(X, "f(x)")
|
|
355
|
+
importance = boost.get_base_learner_importance()
|
|
356
|
+
```
|
|
357
|
+
|
|
358
|
+
Both estimators expose `predict_link()`, order-specific coefficients, knots,
|
|
359
|
+
deviance, log likelihood, and coefficient confidence intervals through R.
|
|
360
|
+
`predict_component()` delegates a named learner prediction to R. The R
|
|
361
|
+
GAM/boost prediction method does not support `type="terms"`, so these
|
|
362
|
+
estimators do not offer `predict_terms()`. The GAM wrapper supports the
|
|
363
|
+
families accepted by `GeDSGeneralizedRegressor`; for binomial fits it accepts
|
|
364
|
+
0/1 responses and creates the factor required by R. The boosting
|
|
365
|
+
wrapper maps `gaussian`, `poisson`, `binomial`, and `gamma` to mboost families.
|
|
366
|
+
The fitted boosting estimator's `n_iter_` is R's total boosting iteration
|
|
367
|
+
count. `get_base_learner_importance()` returns R's `bl_imp()` in-bag risk
|
|
368
|
+
reductions as a pandas Series with the original Python feature names.
|
|
369
|
+
For a boosted fit with one univariate spline feature, R's iteration plots can
|
|
370
|
+
be saved to a multipage PDF without opening R directly:
|
|
371
|
+
|
|
372
|
+
```python
|
|
373
|
+
single_boost = GeDSBoostRegressor(max_iterations=10).fit(X[["x"]], y)
|
|
374
|
+
single_boost.save_boosting_diagnostics(
|
|
375
|
+
"boosting.pdf", iterations=[0, 1, 2], final_fits=True
|
|
376
|
+
)
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
The method refuses to replace an existing file unless `overwrite=True`.
|
|
380
|
+
For binomial boosting, R expects responses encoded as -1 and 1. Offset
|
|
381
|
+
prediction is not offered for these additive estimators yet.
|
|
382
|
+
|
|
383
|
+
Fitted estimators contain a serialized R model and can be saved with
|
|
384
|
+
`model.save(path)` and restored with `GeDSRegressor.load(path)`. As with any
|
|
385
|
+
pickle-based format, only load files from trusted sources.
|
|
386
|
+
|
|
387
|
+
## Development
|
|
388
|
+
|
|
389
|
+
Clone the repository, then install the development dependencies and run the
|
|
390
|
+
integration tests with:
|
|
391
|
+
|
|
392
|
+
```console
|
|
393
|
+
git clone https://github.com/emilioluissaenzguillen/GeDS-python.git
|
|
394
|
+
cd GeDS-python
|
|
395
|
+
python -m pip install -e ".[dev]"
|
|
396
|
+
python -m pytest
|
|
397
|
+
python -m build
|
|
398
|
+
```
|
|
399
|
+
|
|
400
|
+
The tests start an embedded R session and therefore require a working GeDS
|
|
401
|
+
installation; they do not substitute or reimplement any GeDS calculations.
|
|
402
|
+
|
|
403
|
+
## Contact
|
|
404
|
+
|
|
405
|
+
For questions about the Python interface, contact Emilio L. Sáenz Guillén at
|
|
406
|
+
[emilioluissaenzguillen@gmail.com](mailto:emilioluissaenzguillen@gmail.com).
|