geds-python 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
geds/_plotting.py ADDED
@@ -0,0 +1,69 @@
1
+ """Optional Python-native plotting helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ import numpy as np
8
+ import pandas as pd
9
+ from sklearn.utils.validation import check_is_fitted
10
+
11
+
12
+ def plot_fit(
13
+ estimator: Any,
14
+ X: Any,
15
+ y: Any | None = None,
16
+ *,
17
+ ax: Any | None = None,
18
+ grid_size: int = 500,
19
+ show_knots: bool = True,
20
+ ) -> Any:
21
+ """Plot a fitted univariate GeDS model and return its Matplotlib axes.
22
+
23
+ This helper only visualizes predictions already produced by GeDS; it does
24
+ not implement any statistical calculation in Python.
25
+ """
26
+ try:
27
+ import matplotlib.pyplot as plt
28
+ except ImportError as exc: # pragma: no cover - depends on optional extra
29
+ raise ImportError(
30
+ 'Plotting requires Matplotlib; install "geds-python[plot]".'
31
+ ) from exc
32
+
33
+ check_is_fitted(estimator, "_r_model_")
34
+ frame, named_input = estimator._frame(X)
35
+ if frame.shape[1] != 1 or estimator.n_features_in_ != 1:
36
+ raise ValueError("plot_fit supports fitted models with one feature only.")
37
+ if grid_size < 2:
38
+ raise ValueError("grid_size must be at least 2.")
39
+ values = np.asarray(frame.iloc[:, 0], dtype=float)
40
+ if not np.isfinite(values).all():
41
+ raise ValueError("X must contain only finite values.")
42
+ grid_values = np.linspace(values.min(), values.max(), grid_size)
43
+ if named_input:
44
+ grid = pd.DataFrame({frame.columns[0]: grid_values})
45
+ else:
46
+ grid = grid_values.reshape(-1, 1)
47
+ fitted = estimator.predict(grid)
48
+
49
+ if ax is None:
50
+ _, ax = plt.subplots()
51
+ if y is not None:
52
+ response = np.asarray(y, dtype=float)
53
+ if response.ndim != 1 or len(response) != len(values):
54
+ raise ValueError("y must be one-dimensional and have the same length as X.")
55
+ ax.scatter(values, response, s=12, alpha=0.35, label="Data")
56
+ ax.plot(grid_values, fitted, linewidth=2, label="GeDS fit")
57
+ if show_knots and estimator.knots_ is not None:
58
+ knots = np.asarray(estimator.knots_, dtype=float).ravel()
59
+ for index, knot in enumerate(knots):
60
+ ax.axvline(
61
+ knot,
62
+ color="tab:red",
63
+ linestyle="--",
64
+ alpha=0.55,
65
+ label="Internal knots" if index == 0 else None,
66
+ )
67
+ ax.set(xlabel=str(frame.columns[0]), ylabel="Response", title="GeDS spline regression")
68
+ ax.legend()
69
+ return ax
geds/_validation.py ADDED
@@ -0,0 +1,148 @@
1
+ """Python inputs for R GeDS's specialized cross-validation routine."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from typing import Any, Mapping, Sequence
7
+
8
+ import numpy as np
9
+ import pandas as pd
10
+ from sklearn.base import clone
11
+
12
+ from ._backend import get_backend
13
+ from ._estimators import (
14
+ GeDSBoostRegressor, GeDSGAMRegressor, GeDSGeneralizedRegressor,
15
+ GeDSRegressor,
16
+ )
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class GeDSCrossValidationResult:
21
+ """R's best parameter combination and full cross-validation table."""
22
+
23
+ best_params: pd.DataFrame
24
+ results: pd.DataFrame
25
+
26
+
27
+ def cross_validate_geds(
28
+ estimator: GeDSRegressor | GeDSGeneralizedRegressor | GeDSGAMRegressor | GeDSBoostRegressor,
29
+ X: Any,
30
+ y: Any,
31
+ parameter_grid: Mapping[str, Sequence[float]],
32
+ *,
33
+ n_folds: int = 5,
34
+ n_cores: int = 1,
35
+ random_state: int | None = None,
36
+ ) -> GeDSCrossValidationResult:
37
+ """Tune GeDS parameters with R ``crossv_GeDS()`` (Gaussian models only).
38
+
39
+ This is R's specialized grid search, not scikit-learn cross-validation.
40
+ It returns R's MSE and knot/iteration summary without fitting the input
41
+ estimator. R currently applies its own fitting defaults for non-grid
42
+ options in NGeDS, GGeDS, and NGeDSgam cross-validation.
43
+ """
44
+ if not isinstance(estimator, (
45
+ GeDSRegressor, GeDSGeneralizedRegressor,
46
+ GeDSGAMRegressor, GeDSBoostRegressor,
47
+ )):
48
+ raise TypeError("estimator must be a GeDS Python estimator.")
49
+ if not isinstance(n_folds, (int, np.integer)) or n_folds < 2:
50
+ raise ValueError("n_folds must be an integer of at least 2.")
51
+ if not isinstance(n_cores, (int, np.integer)) or n_cores < 1:
52
+ raise ValueError("n_cores must be a positive integer.")
53
+ if random_state is not None and (
54
+ not isinstance(random_state, (int, np.integer)) or random_state < 0
55
+ ):
56
+ raise ValueError("random_state must be a non-negative integer or None.")
57
+ if hasattr(estimator, "family") and estimator.family.lower() != "gaussian":
58
+ raise ValueError(
59
+ "R crossv_GeDS currently uses the Gaussian default for these "
60
+ "models; non-Gaussian estimators are not supported here."
61
+ )
62
+ if hasattr(estimator, "link") and estimator.link is not None:
63
+ raise ValueError("A custom link is not supported by R crossv_GeDS.")
64
+
65
+ # R's non-boost cross-validation routine only forwards beta, phi, q, and
66
+ # the order-derived higher_order flag. Reject other custom settings so
67
+ # they are never silently ignored.
68
+ if not isinstance(estimator, GeDSBoostRegressor):
69
+ defaults_estimator = type(estimator)()
70
+ forwarded = {
71
+ "spline_features", "spline_terms", "linear_features",
72
+ "order", "higher_order", "beta", "phi", "q", "family", "link",
73
+ }
74
+ unsupported = [
75
+ name for name, value in estimator.get_params(deep=False).items()
76
+ if name not in forwarded
77
+ and not np.array_equal(value, getattr(defaults_estimator, name))
78
+ ]
79
+ if unsupported:
80
+ raise ValueError(
81
+ "R crossv_GeDS does not forward these custom settings: "
82
+ + ", ".join(sorted(unsupported))
83
+ )
84
+
85
+ working = clone(estimator)
86
+ working._validate_configuration()
87
+ if isinstance(working, (GeDSGAMRegressor, GeDSBoostRegressor)):
88
+ frame, formula, _ = working._prepare_additive_data(X, y, None)
89
+ else:
90
+ frame, formula, _ = working._prepare_training_data(X, y, None, None)
91
+ if n_folds > len(frame):
92
+ raise ValueError("n_folds cannot exceed the number of observations.")
93
+
94
+ defaults: dict[str, float] = {
95
+ "beta": 0.5 if working.beta is None else float(working.beta),
96
+ "phi": float(working.phi),
97
+ "q": float(working.q),
98
+ }
99
+ if isinstance(working, GeDSBoostRegressor):
100
+ defaults.update({
101
+ "int_knots_init": float(working.int_knots_init),
102
+ "shrinkage": float(working.shrinkage),
103
+ })
104
+ extra = set(parameter_grid) - set(defaults)
105
+ if extra:
106
+ raise ValueError(f"Unsupported parameter grid names: {sorted(extra)}.")
107
+ parameters: dict[str, np.ndarray] = {}
108
+ for name, default in defaults.items():
109
+ values = np.asarray(parameter_grid.get(name, [default]), dtype=float)
110
+ if values.ndim != 1 or not len(values) or not np.isfinite(values).all():
111
+ raise ValueError(f"{name} grid must contain finite numeric values.")
112
+ if name in {"beta", "phi"} and np.any((values < 0) | (values > 1)):
113
+ raise ValueError(f"{name} grid values must lie in [0, 1].")
114
+ if name in {"q", "int_knots_init"} and (
115
+ np.any(values != np.floor(values))
116
+ or np.any(values < (1 if name == "q" else 0))
117
+ ):
118
+ raise ValueError(f"{name} grid values must be valid non-negative integers.")
119
+ if name == "shrinkage" and np.any((values <= 0) | (values > 1)):
120
+ raise ValueError("shrinkage grid values must lie in (0, 1].")
121
+ r_name = "int.knots_init" if name == "int_knots_init" else name
122
+ parameters[f"{r_name}_grid"] = values
123
+
124
+ if isinstance(working, GeDSBoostRegressor):
125
+ model_name = "NGeDSboost"
126
+ fit_kwargs = {
127
+ "max_iterations": working.max_iterations,
128
+ "min_iterations": working.min_iterations,
129
+ "normalize_data": working.normalize_data,
130
+ "initial_learner": working.initial_learner,
131
+ "int.knots_boost": working.int_knots_boost,
132
+ "phi_boost_exit": working.phi_boost_exit,
133
+ "q_boost": working.q_boost,
134
+ "boosting_with_memory": working.boosting_with_memory,
135
+ }
136
+ fit_kwargs = {k: v for k, v in fit_kwargs.items() if v is not None}
137
+ elif isinstance(working, GeDSGAMRegressor):
138
+ model_name, fit_kwargs = "NGeDSgam", {}
139
+ elif isinstance(working, GeDSGeneralizedRegressor):
140
+ model_name, fit_kwargs = "GGeDS", {}
141
+ else:
142
+ model_name, fit_kwargs = "NGeDS", {}
143
+
144
+ best, results = get_backend().cross_validate(
145
+ model_name, formula, frame, parameters, working.order,
146
+ int(n_folds), int(n_cores), random_state, **fit_kwargs,
147
+ )
148
+ return GeDSCrossValidationResult(best_params=best, results=results)
geds/check.py ADDED
@@ -0,0 +1,64 @@
1
+ """Command-line environment check for the Python/R bridge."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+ from typing import Sequence
9
+
10
+ from ._backend import BackendUnavailableError, diagnostics
11
+
12
+
13
+ def _parser() -> argparse.ArgumentParser:
14
+ parser = argparse.ArgumentParser(
15
+ prog="python -m geds.check",
16
+ description="Check that Python can initialize R and load the GeDS package.",
17
+ )
18
+ parser.add_argument(
19
+ "--json",
20
+ action="store_true",
21
+ help="print machine-readable JSON instead of the human-readable report",
22
+ )
23
+ return parser
24
+
25
+
26
+ def main(argv: Sequence[str] | None = None) -> int:
27
+ """Run the environment check and return a process exit code."""
28
+ args = _parser().parse_args(argv)
29
+ try:
30
+ information = diagnostics()
31
+ except BackendUnavailableError as exc:
32
+ if args.json:
33
+ print(json.dumps({"status": "error", "message": str(exc)}, indent=2))
34
+ else:
35
+ print("GeDS environment check: FAILED", file=sys.stderr)
36
+ print(str(exc), file=sys.stderr)
37
+ print(
38
+ "Set R_HOME if the intended R installation is not discovered, "
39
+ "and set GEDS_R_LIBRARY if GeDS is in a non-default R library.",
40
+ file=sys.stderr,
41
+ )
42
+ return 1
43
+
44
+ report = {"status": "ok", **information}
45
+ if args.json:
46
+ print(json.dumps(report, indent=2, sort_keys=True))
47
+ else:
48
+ labels = {
49
+ "python": "Python",
50
+ "r_version": "R",
51
+ "r_home": "R home",
52
+ "rpy2_version": "rpy2",
53
+ "geds_version": "GeDS",
54
+ "geds_library": "GeDS library",
55
+ "minimum_geds_version": "Minimum GeDS",
56
+ }
57
+ print("GeDS environment check: OK")
58
+ for key, label in labels.items():
59
+ print(f" {label}: {information[key]}")
60
+ return 0
61
+
62
+
63
+ if __name__ == "__main__": # pragma: no cover - exercised as a module
64
+ raise SystemExit(main())
geds/py.typed ADDED
@@ -0,0 +1 @@
1
+
@@ -0,0 +1,406 @@
1
+ Metadata-Version: 2.5
2
+ Name: geds-python
3
+ Version: 0.1.0
4
+ Summary: Python estimators backed by the GeDS R package
5
+ Project-URL: Homepage, https://github.com/emilioluissaenzguillen/GeDS-python
6
+ Project-URL: Repository, https://github.com/emilioluissaenzguillen/GeDS-python
7
+ Project-URL: Issues, https://github.com/emilioluissaenzguillen/GeDS-python/issues
8
+ Project-URL: Changelog, https://github.com/emilioluissaenzguillen/GeDS-python/blob/main/CHANGELOG.md
9
+ Project-URL: R package, https://github.com/emilioluissaenzguillen/GeDS
10
+ Author: Dimitrina S. Dimitrova, Vladimir K. Kaishev, Andrea Lattuada, Emilio L. Sáenz Guillén, Richard J. Verrall
11
+ Maintainer-email: "Emilio L. Sáenz Guillén" <emilioluissaenzguillen@gmail.com>
12
+ License-Expression: GPL-3.0-only
13
+ License-File: LICENSE
14
+ Keywords: R,regression,scikit-learn,splines,statistics
15
+ Classifier: Development Status :: 4 - Beta
16
+ Classifier: Intended Audience :: Science/Research
17
+ Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
18
+ Classifier: Operating System :: Microsoft :: Windows
19
+ Classifier: Operating System :: POSIX :: Linux
20
+ Classifier: Programming Language :: Python :: 3
21
+ Classifier: Programming Language :: Python :: 3.10
22
+ Classifier: Programming Language :: Python :: 3.11
23
+ Classifier: Programming Language :: Python :: 3.12
24
+ Classifier: Topic :: Scientific/Engineering
25
+ Requires-Python: >=3.10
26
+ Requires-Dist: numpy>=1.24
27
+ Requires-Dist: pandas>=2.0
28
+ Requires-Dist: rpy2<3.7,>=3.6.7
29
+ Requires-Dist: scikit-learn>=1.4
30
+ Provides-Extra: dev
31
+ Requires-Dist: build>=1.2; extra == 'dev'
32
+ Requires-Dist: matplotlib>=3.8; extra == 'dev'
33
+ Requires-Dist: pytest>=8; extra == 'dev'
34
+ Provides-Extra: plot
35
+ Requires-Dist: matplotlib>=3.8; extra == 'plot'
36
+ Provides-Extra: test
37
+ Requires-Dist: pytest>=8; extra == 'test'
38
+ Description-Content-Type: text/markdown
39
+
40
+ # GeDS for Python
41
+
42
+ This package provides a Python interface to the
43
+ [GeDS R package](https://github.com/emilioluissaenzguillen/GeDS). The R package
44
+ is the sole implementation of the statistical methods. Python supplies a
45
+ scikit-learn-style API, pandas/NumPy conversion, environment diagnostics, and
46
+ model serialization.
47
+
48
+ ## Requirements
49
+
50
+ - R 4.4 or newer (R 4.6.1 is used for development)
51
+ - GeDS 0.3.6 or newer with the Python bridge fixes
52
+ - Python 3.10 or newer
53
+
54
+ Install the Python package, including the optional plotting dependency used in
55
+ the example:
56
+
57
+ ```console
58
+ python -m pip install "geds-python[plot]"
59
+ ```
60
+
61
+ Install the R package separately, using R 4.6.1 or another supported R
62
+ installation. Install the tested GeDS 0.3.6 build from GitHub:
63
+
64
+ ```r
65
+ install.packages("remotes")
66
+ remotes::install_git(
67
+ "https://github.com/emilioluissaenzguillen/GeDS.git",
68
+ ref = "91b8ddd13aae8f39994c87fc356f05da4799f911",
69
+ dependencies = NA, upgrade = "never"
70
+ )
71
+ ```
72
+
73
+ An older GeDS build, even one reporting version `0.3.6`, may lack the fixes
74
+ required by this wrapper. `install.packages("GeDS")` alone is not guaranteed
75
+ to provide them while the CRAN review follows its separate schedule.
76
+ Check the installed R version with `packageVersion("GeDS")`.
77
+
78
+ On Windows, building the GitHub source package requires Rtools compatible
79
+ with the selected R installation. GeDS remains version `0.3.6` on GitHub;
80
+ the wrapper checks an internal compatibility marker for the fit and prediction
81
+ fixes as well as the package version.
82
+
83
+ The wrapper discovers the newest R installation under `Program Files/R` on
84
+ Windows or uses `Rscript` from `PATH` on other platforms. Set `R_HOME` to select
85
+ a particular R installation. If GeDS is installed in a non-default R library,
86
+ set `GEDS_R_LIBRARY` to that library directory before importing `geds`.
87
+
88
+ The Python and R packages have independent release cycles. `geds-python`
89
+ checks the installed GeDS version when its backend first starts and reports the
90
+ selected R installation and package library through `geds.diagnostics()`.
91
+
92
+ Check the backend before fitting:
93
+
94
+ ```console
95
+ python -m geds.check
96
+ ```
97
+
98
+ For a machine-readable report, use `python -m geds.check --json`. The same
99
+ information is available inside Python:
100
+
101
+ ```python
102
+ import geds
103
+
104
+ print(geds.diagnostics())
105
+ ```
106
+
107
+ ### Selecting R and its package library
108
+
109
+ Usually no configuration is necessary. If several R installations are
110
+ available, select one before starting Python:
111
+
112
+ ```powershell
113
+ $env:R_HOME = "C:\Program Files\R\R-4.6.1"
114
+ python -m geds.check
115
+ ```
116
+
117
+ ```bash
118
+ export R_HOME="/Library/Frameworks/R.framework/Resources" # macOS
119
+ # export R_HOME="/usr/lib/R" # Linux
120
+ python -m geds.check
121
+ ```
122
+
123
+ If GeDS is installed in a personal or otherwise non-default R library, set
124
+ `GEDS_R_LIBRARY` to the directory that contains the `GeDS` folder. You can
125
+ find that directory from R with `find.package("GeDS")`; use its parent
126
+ directory as `GEDS_R_LIBRARY`.
127
+
128
+ If the check reports that R is missing, install R or set `R_HOME`. If it finds
129
+ R but not GeDS, start that same R installation and run
130
+ one of the GeDS installation commands above, then rerun the check.
131
+
132
+ ## Example
133
+
134
+ Install the optional plotting dependency with
135
+ `python -m pip install "geds-python[plot]"`, then fit and visualize a nonlinear
136
+ regression:
137
+
138
+ ```python
139
+ import matplotlib.pyplot as plt
140
+ import numpy as np
141
+ import pandas as pd
142
+
143
+ from geds import GeDSRegressor, plot_fit
144
+
145
+ rng = np.random.RandomState(123)
146
+ n = 500
147
+
148
+
149
+ def f_1(x):
150
+ return (10 * x / (1 + 100 * x**2)) * 4 + 4
151
+
152
+
153
+ x = np.sort(rng.uniform(-2.0, 2.0, size=n))
154
+ means = f_1(x)
155
+ y = rng.normal(means, scale=0.1)
156
+ X = pd.DataFrame({"x": x})
157
+
158
+ model = GeDSRegressor(order=3).fit(X, y)
159
+ knots = np.asarray(model.knots_, dtype=float)
160
+
161
+ print("Internal knots:", knots)
162
+
163
+ fig, ax = plt.subplots()
164
+ plot_fit(model, X, y, ax=ax)
165
+ grid_x = np.linspace(x.min(), x.max(), 500)
166
+ ax.plot(
167
+ grid_x,
168
+ f_1(grid_x),
169
+ color="0.25",
170
+ linestyle=":",
171
+ linewidth=2,
172
+ label="True mean",
173
+ )
174
+ ax.set(ylabel="y")
175
+ ax.legend()
176
+ fig.tight_layout()
177
+ plt.show()
178
+ ```
179
+
180
+ With GeDS 0.3.6 and R 4.6.1, this seeded example fits 16 internal knots.
181
+ The dashed vertical lines show how GeDS places more knots around the sharp
182
+ variation near zero while retaining knots across the wider domain.
183
+
184
+ `GeDSRegressor` delegates to `GeDS::NGeDS()`. For exponential-family models,
185
+ use `GeDSGeneralizedRegressor`, which delegates to `GeDS::GGeDS()`.
186
+
187
+ For fitted models, `get_deviance(order=...)`, `get_log_likelihood(order=...)`,
188
+ and `get_confidence_intervals(order=..., level=...)` call the corresponding R
189
+ methods. Confidence intervals are returned as a pandas DataFrame with `lower`
190
+ and `upper` columns. As in R, these are coefficient intervals, not confidence
191
+ bands for the fitted curve.
192
+
193
+ The estimators also work with standard scikit-learn tools such as
194
+ `cross_val_score()` and `GridSearchCV`. Use sequential execution (`n_jobs=1`)
195
+ when cross-validating: the wrapper embeds R in the Python process, and
196
+ parallel-worker behavior is not part of the supported interface.
197
+
198
+ R also has a specialized `crossv_GeDS()` routine, which returns a parameter
199
+ grid with cross-validated mean squared error and knot/iteration summaries.
200
+ Use its Python interface when those R-specific results are needed:
201
+
202
+ ```python
203
+ from geds import cross_validate_geds
204
+
205
+ cv = cross_validate_geds(
206
+ GeDSRegressor(order=3), X, y,
207
+ {"beta": [0.5, 0.7], "phi": [0.95], "q": [2]},
208
+ n_folds=5, n_cores=1, random_state=123,
209
+ )
210
+ print(cv.best_params)
211
+ print(cv.results)
212
+ ```
213
+
214
+ This delegates the entire search to R and does not fit or change the input
215
+ estimator. It currently supports Gaussian models only, accepts the R tuning
216
+ parameters `beta`, `phi`, `q`, and (for boosting) `int_knots_init` and
217
+ `shrinkage`, and defaults to one R worker. R's current non-boost routine does
218
+ not forward other fitting settings; the Python interface rejects custom
219
+ settings it would otherwise silently ignore. Use scikit-learn's grid search
220
+ when you need those settings or a non-Gaussian family.
221
+
222
+ For a fitted univariate spline without extra linear features, R's calculus
223
+ and spline-conversion utilities are available as model methods:
224
+
225
+ ```python
226
+ slopes = model.derive([-0.5, 0.0, 0.5], derivative_order=1)
227
+ areas = model.integrate(-1.0, [-0.5, 0.0, 0.5])
228
+ piece_knots, piece_coefficients = model.piecewise_polynomial()
229
+ ```
230
+
231
+ `derive()` and `integrate()` operate on the predictor (link) scale, as in R.
232
+ `piecewise_polynomial()` returns the R `PPolyRep()` knot vector and coefficient
233
+ matrix; its last coefficient row is extraneous in R's representation. These
234
+ methods use the estimator's selected spline order unless `order=` is given.
235
+
236
+ For a normal univariate fit, impose a shape constraint without changing the
237
+ original fitted model:
238
+
239
+ ```python
240
+ increasing_model = model.shape_constrain("increasing")
241
+ increasing_and_convex = model.shape_constrain(["increasing", "convex"])
242
+ ```
243
+
244
+ This calls R's `shapeConstrain()` and returns a new Python estimator. R also
245
+ supports constraints on one selected univariate smoother in Gaussian GAM and
246
+ boosting fits, via `shape_constrain(..., base_learner="f(x)")`. Those additive
247
+ fits must use `normalize_data=False`. Constrained fits do not provide the usual
248
+ unconstrained coefficient confidence intervals.
249
+
250
+ For count data, the generalized estimator uses `GeDS::GGeDS()` and supports
251
+ both response-scale and link-scale prediction:
252
+
253
+ ```python
254
+ import numpy as np
255
+ import pandas as pd
256
+ from geds import GeDSGeneralizedRegressor, plot_fit
257
+
258
+ rng = np.random.default_rng(123)
259
+ x = np.sort(rng.uniform(-2, 2, 120))
260
+ X = pd.DataFrame({"x": x})
261
+ counts = rng.poisson(np.exp(1 + np.sin(x)))
262
+
263
+ model = GeDSGeneralizedRegressor(
264
+ family="poisson", beta=0.2, phi=0.95, min_internal_knots=3
265
+ ).fit(X, counts)
266
+ mean_counts = model.predict(X)
267
+ log_mean_counts = model.predict_link(X)
268
+ ax = plot_fit(model, X, counts)
269
+ ```
270
+
271
+ `min_internal_knots` controls the minimum number of stage-A knots; it is used
272
+ here to make a small sample's fitted spline visible. GeDS determines the
273
+ final knot positions.
274
+
275
+ For a univariate spline with a known offset (for example log exposure in a
276
+ Poisson model), pass one offset value per observation to both fitting and
277
+ prediction. These values are on the link scale:
278
+
279
+ ```python
280
+ import numpy as np
281
+ import pandas as pd
282
+ from geds import GeDSGeneralizedRegressor
283
+
284
+ rng = np.random.default_rng(321)
285
+ x = np.linspace(-1.5, 1.5, 90)
286
+ X = pd.DataFrame({"x": x})
287
+ exposure = np.linspace(1.1, 2.0, len(x))
288
+ counts = rng.poisson(exposure * np.exp(1.4 + np.sin(x))) + 1
289
+ log_exposure = np.log(exposure)
290
+ model = GeDSGeneralizedRegressor(
291
+ family="poisson", spline_features=["x"], order=2,
292
+ higher_order=False,
293
+ ).fit(X, counts, offset=log_exposure)
294
+ expected_counts = model.predict(X, offset=log_exposure)
295
+ contributions = model.predict_terms(X, offset=log_exposure)
296
+ ```
297
+
298
+ `predict_terms()` returns a DataFrame of the R spline and parametric term
299
+ contributions. Its rows sum to the link prediction after adding the offset;
300
+ the offset is not itself a term column. Offset prediction currently supports
301
+ one spline feature only because the R bivariate prediction method does not
302
+ apply new-data offsets consistently. A model fitted with an offset requires
303
+ an offset at prediction time.
304
+
305
+ This offset interface requires the GeDS GitHub fit and prediction fixes. A
306
+ earlier GeDS `0.3.6` installation without those fixes may mishandle
307
+ generalized-model offsets; the backend rejects that version.
308
+
309
+ Choose spline and parametric components explicitly for mixed data:
310
+
311
+ ```python
312
+ model = GeDSRegressor(
313
+ spline_features=["x"],
314
+ linear_features=["group"],
315
+ ).fit(X, y)
316
+ ```
317
+
318
+ Spline features must be numeric. Parametric features may be numeric or
319
+ categorical; their encoding is performed by the R package so fitting and
320
+ prediction use R's native factor semantics.
321
+ If `spline_features` is omitted, all columns are used in a single joint spline
322
+ term. Select `spline_features=["x"]` and `linear_features=["group"]` to keep
323
+ `group` parametric instead. Two spline features create a joint bivariate
324
+ surface, not two separate additive smooths; R's support for more than two
325
+ spline features is experimental. With named pandas columns, prediction may
326
+ receive columns in a different order because the wrapper restores the fitted
327
+ column order before calling R.
328
+
329
+ ### Additive GAM and boosting models
330
+
331
+ Use `GeDSGAMRegressor` for R's `NGeDSgam()` and `GeDSBoostRegressor` for
332
+ `NGeDSboost()`. Each entry in `spline_terms` is one additive smooth. Put two
333
+ features in the same entry for a joint surface. If omitted, each non-linear
334
+ feature gets its own smooth; `linear_features` selects parametric terms.
335
+
336
+ ```python
337
+ import numpy as np
338
+ import pandas as pd
339
+ from geds import GeDSGAMRegressor, GeDSBoostRegressor
340
+
341
+ x = np.linspace(-2, 2, 100)
342
+ X = pd.DataFrame({"x": x, "z": x**2})
343
+ y = np.sin(x) + 0.3 * x**2
344
+
345
+ gam = GeDSGAMRegressor(
346
+ spline_terms=[("x",), ("z",)], max_iterations=10
347
+ ).fit(X, y)
348
+ boost = GeDSBoostRegressor(
349
+ spline_terms=[("x",), ("z",)], max_iterations=20
350
+ ).fit(X, y)
351
+
352
+ gam_predictions = gam.predict(X)
353
+ boost_predictions = boost.predict(X)
354
+ x_contribution = gam.predict_component(X, "f(x)")
355
+ importance = boost.get_base_learner_importance()
356
+ ```
357
+
358
+ Both estimators expose `predict_link()`, order-specific coefficients, knots,
359
+ deviance, log likelihood, and coefficient confidence intervals through R.
360
+ `predict_component()` delegates a named learner prediction to R. The R
361
+ GAM/boost prediction method does not support `type="terms"`, so these
362
+ estimators do not offer `predict_terms()`. The GAM wrapper supports the
363
+ families accepted by `GeDSGeneralizedRegressor`; for binomial fits it accepts
364
+ 0/1 responses and creates the factor required by R. The boosting
365
+ wrapper maps `gaussian`, `poisson`, `binomial`, and `gamma` to mboost families.
366
+ The fitted boosting estimator's `n_iter_` is R's total boosting iteration
367
+ count. `get_base_learner_importance()` returns R's `bl_imp()` in-bag risk
368
+ reductions as a pandas Series with the original Python feature names.
369
+ For a boosted fit with one univariate spline feature, R's iteration plots can
370
+ be saved to a multipage PDF without opening R directly:
371
+
372
+ ```python
373
+ single_boost = GeDSBoostRegressor(max_iterations=10).fit(X[["x"]], y)
374
+ single_boost.save_boosting_diagnostics(
375
+ "boosting.pdf", iterations=[0, 1, 2], final_fits=True
376
+ )
377
+ ```
378
+
379
+ The method refuses to replace an existing file unless `overwrite=True`.
380
+ For binomial boosting, R expects responses encoded as -1 and 1. Offset
381
+ prediction is not offered for these additive estimators yet.
382
+
383
+ Fitted estimators contain a serialized R model and can be saved with
384
+ `model.save(path)` and restored with `GeDSRegressor.load(path)`. As with any
385
+ pickle-based format, only load files from trusted sources.
386
+
387
+ ## Development
388
+
389
+ Clone the repository, then install the development dependencies and run the
390
+ integration tests with:
391
+
392
+ ```console
393
+ git clone https://github.com/emilioluissaenzguillen/GeDS-python.git
394
+ cd GeDS-python
395
+ python -m pip install -e ".[dev]"
396
+ python -m pytest
397
+ python -m build
398
+ ```
399
+
400
+ The tests start an embedded R session and therefore require a working GeDS
401
+ installation; they do not substitute or reimplement any GeDS calculations.
402
+
403
+ ## Contact
404
+
405
+ For questions about the Python interface, contact Emilio L. Sáenz Guillén at
406
+ [emilioluissaenzguillen@gmail.com](mailto:emilioluissaenzguillen@gmail.com).