odr-bootstrap 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ODR Bootstrap: Orthogonal Distance Regression with Bootstrap Resampling
|
|
3
|
+
|
|
4
|
+
A Python package for SIMS calibration analysis with proper uncertainty
|
|
5
|
+
quantification in both x and y measurements.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from .core import (
|
|
9
|
+
Bootstrap_fit,
|
|
10
|
+
Eval_Conf,
|
|
11
|
+
ODR_Bootstrap,
|
|
12
|
+
ODR_Linear,
|
|
13
|
+
ODR_Linear_Test,
|
|
14
|
+
gauss_agv_err,
|
|
15
|
+
plot_Calibration_Estimates,
|
|
16
|
+
plot_datapoints,
|
|
17
|
+
plot_regression,
|
|
18
|
+
slope_func,
|
|
19
|
+
yint_func,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
__version__ = "0.1.0"
|
|
23
|
+
__author__ = "Henry Towbin"
|
|
24
|
+
__all__ = [
|
|
25
|
+
"ODR_Linear",
|
|
26
|
+
"ODR_Linear_Test",
|
|
27
|
+
"Bootstrap_fit",
|
|
28
|
+
"Eval_Conf",
|
|
29
|
+
"plot_regression",
|
|
30
|
+
"ODR_Bootstrap",
|
|
31
|
+
"gauss_agv_err",
|
|
32
|
+
"plot_datapoints",
|
|
33
|
+
"plot_Calibration_Estimates",
|
|
34
|
+
"yint_func",
|
|
35
|
+
"slope_func",
|
|
36
|
+
]
|
odr_bootstrap/core.py
ADDED
|
@@ -0,0 +1,695 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ODR bootstrapping utilities for SIMS calibration analysis.
|
|
3
|
+
|
|
4
|
+
This module provides orthogonal distance regression (ODR) with uncertainties
|
|
5
|
+
in both x and y, bootstrap resampling of fit parameters, confidence interval
|
|
6
|
+
evaluation for predicted fit lines, and calibration estimate plotting.
|
|
7
|
+
|
|
8
|
+
Dependencies
|
|
9
|
+
------------
|
|
10
|
+
matplotlib
|
|
11
|
+
numpy
|
|
12
|
+
pandas
|
|
13
|
+
scipy
|
|
14
|
+
|
|
15
|
+
Changelog
|
|
16
|
+
---------
|
|
17
|
+
April 2025:
|
|
18
|
+
- Fixed zero-intercept ODR initialization: properly wrap slope-only
|
|
19
|
+
initial guess in list for scipy.odr compatibility.
|
|
20
|
+
- Updated deprecated np.trapz to np.trapezoid for scipy 1.15+ compatibility.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
import matplotlib.pyplot as plt
|
|
26
|
+
import numpy as np
|
|
27
|
+
import pandas as pd
|
|
28
|
+
import scipy.stats as stats
|
|
29
|
+
from scipy import odr
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def ODR_Linear(
|
|
33
|
+
x: np.ndarray | list[float],
|
|
34
|
+
y: np.ndarray | list[float],
|
|
35
|
+
x_err: np.ndarray | list[float],
|
|
36
|
+
y_err: np.ndarray | list[float],
|
|
37
|
+
intercept: bool = False,
|
|
38
|
+
InitialGuess: list[float] | None = None,
|
|
39
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
40
|
+
"""
|
|
41
|
+
Fit a linear model using orthogonal distance regression (ODR).
|
|
42
|
+
|
|
43
|
+
Parameters
|
|
44
|
+
----------
|
|
45
|
+
x : array-like
|
|
46
|
+
Independent variable values.
|
|
47
|
+
y : array-like
|
|
48
|
+
Dependent variable values.
|
|
49
|
+
x_err : array-like
|
|
50
|
+
Uncertainties in the independent variable values.
|
|
51
|
+
y_err : array-like
|
|
52
|
+
Uncertainties in the dependent variable values.
|
|
53
|
+
intercept : bool, optional
|
|
54
|
+
If True, fit `y = a * x + b`. If False, fit `y = a * x` through the origin.
|
|
55
|
+
Default is False.
|
|
56
|
+
InitialGuess : list, optional
|
|
57
|
+
Initial guess for the fit parameters. For intercept fits supply
|
|
58
|
+
`[slope, intercept]`. For zero-intercept fits supply `[slope]`.
|
|
59
|
+
Default is `[100, 1]`.
|
|
60
|
+
|
|
61
|
+
Returns
|
|
62
|
+
-------
|
|
63
|
+
tuple
|
|
64
|
+
`Popt`, `Perr` where `Popt` is the fitted parameter array and `Perr`
|
|
65
|
+
is the 1-sigma uncertainty array.
|
|
66
|
+
"""
|
|
67
|
+
def yint_func(p: np.ndarray | list[float], x: np.ndarray | list[float]) -> np.ndarray:
|
|
68
|
+
a, b = p
|
|
69
|
+
return a * x + b
|
|
70
|
+
|
|
71
|
+
def slope_func(p: np.ndarray | list[float], x: np.ndarray | list[float]) -> np.ndarray:
|
|
72
|
+
a = p
|
|
73
|
+
return a * x
|
|
74
|
+
|
|
75
|
+
if InitialGuess is None:
|
|
76
|
+
InitialGuess = [100, 1]
|
|
77
|
+
|
|
78
|
+
linear_model = odr.Model(yint_func)
|
|
79
|
+
beta0 = InitialGuess
|
|
80
|
+
if intercept is False:
|
|
81
|
+
linear_model = odr.Model(slope_func)
|
|
82
|
+
# scipy.odr.ODR requires beta0 to be array-like; wrap scalar in list
|
|
83
|
+
beta0 = [InitialGuess[0]] # Use only slope for zero-intercept fit
|
|
84
|
+
|
|
85
|
+
data = odr.RealData(x, y, sx=x_err, sy=y_err)
|
|
86
|
+
myodr = odr.ODR(data, linear_model, beta0=beta0)
|
|
87
|
+
myodr.set_job(fit_type=0)
|
|
88
|
+
out = myodr.run()
|
|
89
|
+
|
|
90
|
+
Popt = out.beta
|
|
91
|
+
Perr = out.sd_beta
|
|
92
|
+
return Popt, Perr
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def ODR_Linear_Test(
|
|
96
|
+
x: np.ndarray | list[float],
|
|
97
|
+
y: np.ndarray | list[float],
|
|
98
|
+
x_err: np.ndarray | list[float],
|
|
99
|
+
y_err: np.ndarray | list[float],
|
|
100
|
+
intercept: bool = False,
|
|
101
|
+
InitialGuess: list[float] = [100, 1],
|
|
102
|
+
) -> tuple[np.ndarray, np.ndarray, Any]:
|
|
103
|
+
"""
|
|
104
|
+
Fit a linear model using ODR and return the raw ODR output.
|
|
105
|
+
|
|
106
|
+
Parameters
|
|
107
|
+
----------
|
|
108
|
+
x : array-like
|
|
109
|
+
Independent variable values.
|
|
110
|
+
y : array-like
|
|
111
|
+
Dependent variable values.
|
|
112
|
+
x_err : array-like
|
|
113
|
+
Uncertainties in the independent variable values.
|
|
114
|
+
y_err : array-like
|
|
115
|
+
Uncertainties in the dependent variable values.
|
|
116
|
+
intercept : bool, optional
|
|
117
|
+
If True, fit `y = a * x + b`. If False, fit `y = a * x` through the origin.
|
|
118
|
+
Default is False.
|
|
119
|
+
InitialGuess : list, optional
|
|
120
|
+
Initial guess for the fit parameters. For intercept fits supply
|
|
121
|
+
`[slope, intercept]`. For zero-intercept fits supply `[slope]`.
|
|
122
|
+
Default is `[100, 1]`.
|
|
123
|
+
|
|
124
|
+
Returns
|
|
125
|
+
-------
|
|
126
|
+
tuple
|
|
127
|
+
`Popt`, `Perr`, `odr_output` where `odr_output` is the full ODR result.
|
|
128
|
+
"""
|
|
129
|
+
def yint_func(p, x):
|
|
130
|
+
a, b = p
|
|
131
|
+
return a * x + b
|
|
132
|
+
|
|
133
|
+
def slope_func(p, x):
|
|
134
|
+
a = p
|
|
135
|
+
return a * x
|
|
136
|
+
|
|
137
|
+
linear_model = odr.Model(yint_func)
|
|
138
|
+
beta0 = InitialGuess
|
|
139
|
+
if intercept is False:
|
|
140
|
+
linear_model = odr.Model(slope_func)
|
|
141
|
+
# scipy.odr.ODR requires beta0 to be array-like; wrap scalar in list
|
|
142
|
+
beta0 = [InitialGuess[0]] # Use only slope for zero-intercept fit
|
|
143
|
+
|
|
144
|
+
data = odr.RealData(x, y, sx=x_err, sy=y_err)
|
|
145
|
+
myodr = odr.ODR(data, linear_model, beta0=beta0)
|
|
146
|
+
myodr.set_job(fit_type=0)
|
|
147
|
+
out = myodr.run()
|
|
148
|
+
|
|
149
|
+
Popt = out.beta
|
|
150
|
+
Perr = out.sd_beta
|
|
151
|
+
return Popt, Perr, out
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def Bootstrap_fit(
|
|
155
|
+
x: np.ndarray | list[float],
|
|
156
|
+
y: np.ndarray | list[float],
|
|
157
|
+
x_err: np.ndarray | list[float],
|
|
158
|
+
y_err: np.ndarray | list[float],
|
|
159
|
+
resample_draws: int,
|
|
160
|
+
InterceptFit: bool = True,
|
|
161
|
+
InitialGuess: list[float] = [100, 1],
|
|
162
|
+
) -> tuple[list[np.ndarray], list[pd.DataFrame]]:
|
|
163
|
+
"""
|
|
164
|
+
Perform bootstrap resampling of ODR linear fits.
|
|
165
|
+
|
|
166
|
+
Parameters
|
|
167
|
+
----------
|
|
168
|
+
x : array-like
|
|
169
|
+
Independent variable values.
|
|
170
|
+
y : array-like
|
|
171
|
+
Dependent variable values.
|
|
172
|
+
x_err : array-like
|
|
173
|
+
Uncertainties in the independent variable values.
|
|
174
|
+
y_err : array-like
|
|
175
|
+
Uncertainties in the dependent variable values.
|
|
176
|
+
resample_draws : int
|
|
177
|
+
Number of bootstrap resamples to compute.
|
|
178
|
+
InterceptFit : bool, optional
|
|
179
|
+
If True, fit slope and intercept; if False, fit through the origin.
|
|
180
|
+
Default is True.
|
|
181
|
+
InitialGuess : list, optional
|
|
182
|
+
Initial guess for model parameters. Default is `[100, 1]`.
|
|
183
|
+
|
|
184
|
+
Returns
|
|
185
|
+
-------
|
|
186
|
+
tuple
|
|
187
|
+
fit_params : list of ndarray
|
|
188
|
+
First element is the fit result from the full dataset, followed
|
|
189
|
+
by all bootstrap fits.
|
|
190
|
+
subsamples : list of pandas.DataFrame
|
|
191
|
+
Resampled DataFrame objects used for each bootstrap iteration.
|
|
192
|
+
"""
|
|
193
|
+
def resample(count):
|
|
194
|
+
return np.random.randint(0, count, count)
|
|
195
|
+
|
|
196
|
+
InitialGuess = list(InitialGuess)
|
|
197
|
+
if InterceptFit is False:
|
|
198
|
+
InitialGuess = [InitialGuess[0]]
|
|
199
|
+
|
|
200
|
+
data = np.array([x, x_err, y, y_err]).T
|
|
201
|
+
df = pd.DataFrame(data, columns=["x", "x_err", "y", "y_err"])
|
|
202
|
+
df.dropna(inplace=True)
|
|
203
|
+
length = len(df)
|
|
204
|
+
|
|
205
|
+
opt, err = ODR_Linear(
|
|
206
|
+
x=df["x"],
|
|
207
|
+
y=df["y"],
|
|
208
|
+
x_err=df["x_err"],
|
|
209
|
+
y_err=df["y_err"],
|
|
210
|
+
InitialGuess=InitialGuess,
|
|
211
|
+
intercept=InterceptFit,
|
|
212
|
+
)
|
|
213
|
+
Fit_Param = [opt]
|
|
214
|
+
subs = []
|
|
215
|
+
|
|
216
|
+
for _ in range(resample_draws):
|
|
217
|
+
sub = df.take(resample(length))
|
|
218
|
+
opt, err = ODR_Linear(
|
|
219
|
+
x=sub["x"],
|
|
220
|
+
y=sub["y"],
|
|
221
|
+
x_err=sub["x_err"],
|
|
222
|
+
y_err=sub["y_err"],
|
|
223
|
+
InitialGuess=InitialGuess,
|
|
224
|
+
intercept=InterceptFit,
|
|
225
|
+
)
|
|
226
|
+
Fit_Param.append(opt)
|
|
227
|
+
subs.append(sub)
|
|
228
|
+
|
|
229
|
+
return Fit_Param, subs
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def yint_func(p: np.ndarray | list[float], x: np.ndarray | list[float]) -> np.ndarray:
|
|
233
|
+
"""
|
|
234
|
+
Evaluate a line with slope and intercept.
|
|
235
|
+
|
|
236
|
+
Parameters
|
|
237
|
+
----------
|
|
238
|
+
p : array-like
|
|
239
|
+
Parameter vector [slope, intercept].
|
|
240
|
+
x : array-like
|
|
241
|
+
Independent variable values.
|
|
242
|
+
|
|
243
|
+
Returns
|
|
244
|
+
-------
|
|
245
|
+
ndarray
|
|
246
|
+
Evaluated y values.
|
|
247
|
+
"""
|
|
248
|
+
a, b = p
|
|
249
|
+
return a * x + b
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def slope_func(p: np.ndarray | list[float], x: np.ndarray | list[float]) -> np.ndarray:
|
|
253
|
+
"""
|
|
254
|
+
Evaluate a line through the origin.
|
|
255
|
+
|
|
256
|
+
Parameters
|
|
257
|
+
----------
|
|
258
|
+
p : array-like
|
|
259
|
+
Parameter vector [slope].
|
|
260
|
+
x : array-like
|
|
261
|
+
Independent variable values.
|
|
262
|
+
|
|
263
|
+
Returns
|
|
264
|
+
-------
|
|
265
|
+
ndarray
|
|
266
|
+
Evaluated y values.
|
|
267
|
+
"""
|
|
268
|
+
a = p
|
|
269
|
+
return a * x
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def Eval_Conf(
|
|
273
|
+
Fit_Param: list[np.ndarray],
|
|
274
|
+
Confidence_Bound: float = 0.95,
|
|
275
|
+
LineMax: int = 200,
|
|
276
|
+
LineInt: int = 1,
|
|
277
|
+
**kwargs: Any,
|
|
278
|
+
) -> pd.DataFrame:
|
|
279
|
+
"""
|
|
280
|
+
Evaluate bootstrap confidence intervals for linear predictions.
|
|
281
|
+
|
|
282
|
+
Parameters
|
|
283
|
+
----------
|
|
284
|
+
Fit_Param : list of array-like
|
|
285
|
+
Bootstrapped fit parameter vectors. Each row must contain either one
|
|
286
|
+
parameter (slope only) or two parameters (slope and intercept).
|
|
287
|
+
Confidence_Bound : float, optional
|
|
288
|
+
Confidence level expressed as a fraction between 0 and 1.
|
|
289
|
+
Default is 0.95.
|
|
290
|
+
LineMax : int, optional
|
|
291
|
+
Maximum x-value for the evaluation grid. Default is 200.
|
|
292
|
+
LineInt : int, optional
|
|
293
|
+
Step size for the evaluation grid. Default is 1.
|
|
294
|
+
|
|
295
|
+
Returns
|
|
296
|
+
-------
|
|
297
|
+
pandas.DataFrame
|
|
298
|
+
DataFrame indexed by x values containing columns:
|
|
299
|
+
- neg_error_bound
|
|
300
|
+
- pos_error_bound
|
|
301
|
+
- best_fit
|
|
302
|
+
- percent_error_neg
|
|
303
|
+
- percent_error_pos
|
|
304
|
+
"""
|
|
305
|
+
if len(Fit_Param[0]) > 2:
|
|
306
|
+
raise ValueError(
|
|
307
|
+
"Fit_Param has too many inputs per row. Line inputs must be 1 or 2 parameters."
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
FitFunc = yint_func
|
|
311
|
+
if len(Fit_Param[0]) == 1:
|
|
312
|
+
FitFunc = slope_func
|
|
313
|
+
|
|
314
|
+
evaluated = []
|
|
315
|
+
x = np.arange(0, LineMax, LineInt)
|
|
316
|
+
for row in Fit_Param:
|
|
317
|
+
evaluated.append(FitFunc(row, x))
|
|
318
|
+
|
|
319
|
+
BootStp_Samples = pd.DataFrame(evaluated)
|
|
320
|
+
confidence_ints = []
|
|
321
|
+
for _, col in BootStp_Samples.items():
|
|
322
|
+
histrange = (np.nanmin(col), np.nanmax(col))
|
|
323
|
+
hist = np.histogram(col, bins=200, range=histrange)
|
|
324
|
+
conf_int = stats.rv_histogram(hist).interval(Confidence_Bound)
|
|
325
|
+
confidence_ints.append(conf_int)
|
|
326
|
+
|
|
327
|
+
Results = pd.DataFrame(
|
|
328
|
+
confidence_ints, columns=("neg_error_bound", "pos_error_bound")
|
|
329
|
+
)
|
|
330
|
+
Results.index = x
|
|
331
|
+
Results["best_fit"] = evaluated[0]
|
|
332
|
+
Results["percent_error_neg"] = (
|
|
333
|
+
Results["best_fit"] - Results["neg_error_bound"]
|
|
334
|
+
) / np.abs(Results["best_fit"])
|
|
335
|
+
Results["percent_error_pos"] = (
|
|
336
|
+
Results["pos_error_bound"] - Results["best_fit"]
|
|
337
|
+
) / np.abs(Results["best_fit"])
|
|
338
|
+
|
|
339
|
+
return Results
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def plot_regression(
|
|
343
|
+
confidence_df: pd.DataFrame,
|
|
344
|
+
datapoints: pd.DataFrame | None = None,
|
|
345
|
+
LineMax: int = 200,
|
|
346
|
+
LineInt: int = 1,
|
|
347
|
+
ax: plt.Axes | None = None,
|
|
348
|
+
ecolor: str = "r",
|
|
349
|
+
line_color: str = "b",
|
|
350
|
+
sigma: int = 2,
|
|
351
|
+
e_alpha: float = 0.5,
|
|
352
|
+
**kwargs: Any,
|
|
353
|
+
) -> plt.Axes:
|
|
354
|
+
"""
|
|
355
|
+
Plot a best-fit regression line and its bootstrap confidence band.
|
|
356
|
+
|
|
357
|
+
Parameters
|
|
358
|
+
----------
|
|
359
|
+
confidence_df : pandas.DataFrame
|
|
360
|
+
Output from `Eval_Conf` with columns `best_fit`, `neg_error_bound`, and
|
|
361
|
+
`pos_error_bound`.
|
|
362
|
+
datapoints : pandas.DataFrame, optional
|
|
363
|
+
DataFrame containing columns `x`, `y`, `xerr`, and `yerr`.
|
|
364
|
+
LineMax : int, optional
|
|
365
|
+
Accepted for compatibility but not used in this function.
|
|
366
|
+
LineInt : int, optional
|
|
367
|
+
Accepted for compatibility but not used in this function.
|
|
368
|
+
ax : matplotlib.axes.Axes, optional
|
|
369
|
+
Axis object to draw on. If None, the current axis is used.
|
|
370
|
+
ecolor : str, optional
|
|
371
|
+
Confidence band color. Default is 'r'.
|
|
372
|
+
line_color : str, optional
|
|
373
|
+
Best-fit line color. Default is 'b'.
|
|
374
|
+
sigma : int, optional
|
|
375
|
+
Ignored in the current implementation.
|
|
376
|
+
e_alpha : float, optional
|
|
377
|
+
Alpha transparency for the confidence band. Default is 0.5.
|
|
378
|
+
|
|
379
|
+
Returns
|
|
380
|
+
-------
|
|
381
|
+
matplotlib.axes.Axes
|
|
382
|
+
Axis containing the regression plot.
|
|
383
|
+
"""
|
|
384
|
+
BestFitLine = confidence_df["best_fit"]
|
|
385
|
+
NegBound = confidence_df["neg_error_bound"]
|
|
386
|
+
PosBound = confidence_df["pos_error_bound"]
|
|
387
|
+
|
|
388
|
+
x = NegBound.index
|
|
389
|
+
if ax is None:
|
|
390
|
+
ax = plt.gca()
|
|
391
|
+
|
|
392
|
+
ax.fill_between(x, NegBound, PosBound, color=ecolor, alpha=e_alpha)
|
|
393
|
+
ax.plot(x, BestFitLine, color=line_color, **kwargs)
|
|
394
|
+
|
|
395
|
+
if datapoints is not None:
|
|
396
|
+
ax.errorbar(
|
|
397
|
+
x=datapoints["x"],
|
|
398
|
+
y=datapoints["y"],
|
|
399
|
+
yerr=datapoints["yerr"],
|
|
400
|
+
xerr=datapoints["xerr"],
|
|
401
|
+
marker=".",
|
|
402
|
+
fmt="g",
|
|
403
|
+
linestyle="none",
|
|
404
|
+
capsize=5,
|
|
405
|
+
markeredgewidth=1,
|
|
406
|
+
markersize=10,
|
|
407
|
+
label=None,
|
|
408
|
+
**kwargs,
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
return ax
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def ODR_Bootstrap(
|
|
415
|
+
x: np.ndarray | list[float],
|
|
416
|
+
y: np.ndarray | list[float],
|
|
417
|
+
x_err: np.ndarray | list[float],
|
|
418
|
+
y_err: np.ndarray | list[float],
|
|
419
|
+
resample_draws: int = 5000,
|
|
420
|
+
LineMax: int = 200,
|
|
421
|
+
LineInterval: int = 1,
|
|
422
|
+
InterceptFit: bool = True,
|
|
423
|
+
InitialGuess: list[float] = [100, 1],
|
|
424
|
+
Confidence_Bound: float = 0.95,
|
|
425
|
+
plot: bool = False,
|
|
426
|
+
ax: plt.Axes | None = None,
|
|
427
|
+
**kwargs: Any,
|
|
428
|
+
) -> tuple[pd.DataFrame, np.ndarray, pd.DataFrame, list[np.ndarray], list[pd.DataFrame]]:
|
|
429
|
+
"""
|
|
430
|
+
Run bootstrap resampling for ODR linear fitting and compute confidence data.
|
|
431
|
+
|
|
432
|
+
Parameters
|
|
433
|
+
----------
|
|
434
|
+
x : array-like
|
|
435
|
+
Independent variable values.
|
|
436
|
+
y : array-like
|
|
437
|
+
Dependent variable values.
|
|
438
|
+
x_err : array-like
|
|
439
|
+
Uncertainties in x.
|
|
440
|
+
y_err : array-like
|
|
441
|
+
Uncertainties in y.
|
|
442
|
+
resample_draws : int, optional
|
|
443
|
+
Number of bootstrap resamples. Default is 5000.
|
|
444
|
+
LineMax : int, optional
|
|
445
|
+
Maximum x value used in `Eval_Conf`. Default is 200.
|
|
446
|
+
LineInterval : int, optional
|
|
447
|
+
Step size used in `Eval_Conf`. Default is 1.
|
|
448
|
+
InterceptFit : bool, optional
|
|
449
|
+
If True, fit slope and intercept; if False, fit through the origin.
|
|
450
|
+
Default is True.
|
|
451
|
+
InitialGuess : list, optional
|
|
452
|
+
Initial guess for fit parameters. Default is `[100, 1]`.
|
|
453
|
+
Confidence_Bound : float, optional
|
|
454
|
+
Confidence level for interval estimation. Default is 0.95.
|
|
455
|
+
plot : bool, optional
|
|
456
|
+
Accepted for compatibility but not used in this implementation.
|
|
457
|
+
ax : matplotlib.axes.Axes, optional
|
|
458
|
+
Axis object for future plotting support.
|
|
459
|
+
|
|
460
|
+
Returns
|
|
461
|
+
-------
|
|
462
|
+
tuple
|
|
463
|
+
confidence_data : pandas.DataFrame
|
|
464
|
+
Confidence interval results from `Eval_Conf`.
|
|
465
|
+
best_fit_params : ndarray
|
|
466
|
+
Fit parameters for the full dataset.
|
|
467
|
+
points : pandas.DataFrame
|
|
468
|
+
Cleaned input data containing `x`, `y`, `xerr`, and `yerr`.
|
|
469
|
+
all_params : list of ndarray
|
|
470
|
+
All fit parameter vectors including bootstrap resamples.
|
|
471
|
+
subsamples : list of pandas.DataFrame
|
|
472
|
+
Bootstrap resampled subsets.
|
|
473
|
+
"""
|
|
474
|
+
param, subs = Bootstrap_fit(
|
|
475
|
+
x, y, x_err, y_err, resample_draws, InterceptFit, InitialGuess
|
|
476
|
+
)
|
|
477
|
+
confidence_data = Eval_Conf(
|
|
478
|
+
Fit_Param=param,
|
|
479
|
+
Confidence_Bound=Confidence_Bound,
|
|
480
|
+
LineMax=LineMax,
|
|
481
|
+
LineInt=LineInterval,
|
|
482
|
+
)
|
|
483
|
+
|
|
484
|
+
points = pd.DataFrame({"x": x, "y": y, "xerr": x_err, "yerr": y_err})
|
|
485
|
+
points.dropna(inplace=True)
|
|
486
|
+
|
|
487
|
+
return confidence_data, param[0], points, param, subs
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def gauss_agv_err(
|
|
491
|
+
concentrations: np.ndarray | list[float],
|
|
492
|
+
errors: np.ndarray | list[float],
|
|
493
|
+
cut_off: float = 0.000001,
|
|
494
|
+
) -> tuple[dict[str, np.ndarray], dict[str, Any]]:
|
|
495
|
+
"""
|
|
496
|
+
Compute an aggregate Gaussian distribution from values and uncertainties.
|
|
497
|
+
|
|
498
|
+
Combines multiple normal distributions into a single kernel density estimate.
|
|
499
|
+
Uses trapezoidal integration (via scipy.integrate.trapezoid) to normalize.
|
|
500
|
+
|
|
501
|
+
Parameters
|
|
502
|
+
----------
|
|
503
|
+
concentrations : array-like
|
|
504
|
+
Central values for each Gaussian component.
|
|
505
|
+
errors : array-like
|
|
506
|
+
Standard deviations for each Gaussian component.
|
|
507
|
+
cut_off : float, optional
|
|
508
|
+
Probability density threshold for filtering low values.
|
|
509
|
+
Default is 1e-6.
|
|
510
|
+
|
|
511
|
+
Returns
|
|
512
|
+
-------
|
|
513
|
+
tuple
|
|
514
|
+
distribution : dict
|
|
515
|
+
Dictionary containing `x` and `y` arrays for the normalized density.
|
|
516
|
+
statistics : dict
|
|
517
|
+
Summary information including mean, mode, midpoint, and bounds.
|
|
518
|
+
"""
|
|
519
|
+
def gaussian(x, sigma, avg):
|
|
520
|
+
return (1 / (sigma * np.sqrt(2 * np.pi))) * np.exp(
|
|
521
|
+
-0.5 * ((x - avg) / sigma) ** 2
|
|
522
|
+
)
|
|
523
|
+
|
|
524
|
+
def CI_bound(xi, data, bound_fraction):
|
|
525
|
+
for n, val in enumerate(np.cumsum(data)):
|
|
526
|
+
if val > bound_fraction:
|
|
527
|
+
return round(xi[n], 2)
|
|
528
|
+
|
|
529
|
+
def find_range(avgs, sigmas):
|
|
530
|
+
max_val = np.max(avgs) + 3 * np.max(sigmas)
|
|
531
|
+
min_val = np.min(avgs) - 3 * np.max(sigmas)
|
|
532
|
+
return min_val, max_val
|
|
533
|
+
|
|
534
|
+
min_val, max_val = find_range(concentrations, errors)
|
|
535
|
+
xi = np.arange(min_val, max_val, 0.01)
|
|
536
|
+
x = np.tile(xi, (len(concentrations), 1))
|
|
537
|
+
unnormed_data = np.sum(gaussian(x.T, errors, concentrations), axis=1)
|
|
538
|
+
# Use np.trapezoid (scipy >= 1.15) instead of deprecated np.trapz
|
|
539
|
+
data = unnormed_data / np.trapezoid(unnormed_data)
|
|
540
|
+
|
|
541
|
+
average = np.dot(xi, data) / np.sum(data)
|
|
542
|
+
most_frequent = xi[np.argmax(data)]
|
|
543
|
+
best_fit = concentrations[0]
|
|
544
|
+
|
|
545
|
+
center_of_mass = CI_bound(xi, data, 0.50)
|
|
546
|
+
one_sigma_bounds = CI_bound(xi, data, 0.16), CI_bound(xi, data, 0.84)
|
|
547
|
+
two_sigma_bounds = CI_bound(xi, data, 0.05), CI_bound(xi, data, 0.95)
|
|
548
|
+
CI_one_sigma = (
|
|
549
|
+
round(center_of_mass - one_sigma_bounds[0], 2),
|
|
550
|
+
round(one_sigma_bounds[1] - center_of_mass, 2),
|
|
551
|
+
)
|
|
552
|
+
CI_two_sigma = (
|
|
553
|
+
round(center_of_mass - two_sigma_bounds[0], 2),
|
|
554
|
+
round(two_sigma_bounds[1] - center_of_mass, 2),
|
|
555
|
+
)
|
|
556
|
+
|
|
557
|
+
return (
|
|
558
|
+
{"x": xi, "y": data},
|
|
559
|
+
{
|
|
560
|
+
"simple_best_fit": best_fit,
|
|
561
|
+
"mean": average,
|
|
562
|
+
"mode": most_frequent,
|
|
563
|
+
"mid_point": center_of_mass,
|
|
564
|
+
"one_sigma_bounds": one_sigma_bounds,
|
|
565
|
+
"two_sigma_bounds": two_sigma_bounds,
|
|
566
|
+
"CI_one_sigma": CI_one_sigma,
|
|
567
|
+
"CI_two_sigma": CI_two_sigma,
|
|
568
|
+
"n": len(concentrations),
|
|
569
|
+
},
|
|
570
|
+
)
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
def plot_datapoints(
|
|
574
|
+
data: dict[str, np.ndarray],
|
|
575
|
+
bounds: dict[str, Any],
|
|
576
|
+
ax: plt.Axes | None = None,
|
|
577
|
+
sample_name: str | None = None,
|
|
578
|
+
) -> plt.Axes:
|
|
579
|
+
"""
|
|
580
|
+
Plot a probability density curve and annotate summary statistics.
|
|
581
|
+
|
|
582
|
+
Parameters
|
|
583
|
+
----------
|
|
584
|
+
data : dict
|
|
585
|
+
Dictionary containing `x` and `y` density arrays.
|
|
586
|
+
bounds : dict
|
|
587
|
+
Summary statistics returned by `gauss_agv_err`.
|
|
588
|
+
ax : matplotlib.axes.Axes, optional
|
|
589
|
+
Axis object to draw on. If None, the current axis is used.
|
|
590
|
+
sample_name : str, optional
|
|
591
|
+
Optional label or title text.
|
|
592
|
+
|
|
593
|
+
Returns
|
|
594
|
+
-------
|
|
595
|
+
matplotlib.axes.Axes
|
|
596
|
+
Axis containing the plotted density curve.
|
|
597
|
+
"""
|
|
598
|
+
ax = ax or plt.gca()
|
|
599
|
+
x = data["x"]
|
|
600
|
+
y = data["y"]
|
|
601
|
+
ax.plot(x, y, linewidth=3)
|
|
602
|
+
ax.set_xlabel("Concentration ppm")
|
|
603
|
+
ax.set_ylabel("Probability")
|
|
604
|
+
|
|
605
|
+
ax.axvline(
|
|
606
|
+
x=bounds["mean"],
|
|
607
|
+
ymin=0,
|
|
608
|
+
color="b",
|
|
609
|
+
linestyle="dashed",
|
|
610
|
+
linewidth=3,
|
|
611
|
+
label="Mean",
|
|
612
|
+
)
|
|
613
|
+
ax.axvline(
|
|
614
|
+
x=bounds["mid_point"],
|
|
615
|
+
ymin=0,
|
|
616
|
+
color="g",
|
|
617
|
+
linestyle="dashed",
|
|
618
|
+
linewidth=3,
|
|
619
|
+
label="Mid-point & 65% CI",
|
|
620
|
+
)
|
|
621
|
+
|
|
622
|
+
CI_one_sigma = bounds["CI_one_sigma"]
|
|
623
|
+
CI_two_sigma = bounds["CI_two_sigma"]
|
|
624
|
+
|
|
625
|
+
ax.annotate(
|
|
626
|
+
f"""
|
|
627
|
+
Simple Best Fit: {float(bounds['simple_best_fit']):.2f}
|
|
628
|
+
Mean: {float(bounds['mean']):.2f}
|
|
629
|
+
Mode: {float(bounds['mode']):.2f}
|
|
630
|
+
Mid-point: {float(bounds['mid_point']):.2f}
|
|
631
|
+
Confidence Intervals
|
|
632
|
+
68%: - {CI_one_sigma[0]:.2f} / +{CI_one_sigma[1]:.2f}
|
|
633
|
+
95%: - {CI_two_sigma[0]:.2f} / +{CI_two_sigma[1]:.2f}
|
|
634
|
+
n: {bounds['n']}
|
|
635
|
+
""",
|
|
636
|
+
xy=(0.02, 0.68),
|
|
637
|
+
xycoords="axes fraction",
|
|
638
|
+
bbox=dict(boxstyle="square", fc="w", alpha=0.85),
|
|
639
|
+
)
|
|
640
|
+
eb = ax.errorbar(
|
|
641
|
+
x=bounds["mid_point"],
|
|
642
|
+
y=np.max(y) / 2,
|
|
643
|
+
xerr=np.array([[CI_one_sigma[0]], [CI_one_sigma[1]]]),
|
|
644
|
+
capsize=10,
|
|
645
|
+
elinewidth=3,
|
|
646
|
+
capthick=3,
|
|
647
|
+
ecolor="g",
|
|
648
|
+
linestyle="dashed",
|
|
649
|
+
)
|
|
650
|
+
eb[-1][0].set_linestyle("dashed")
|
|
651
|
+
|
|
652
|
+
ax.set_ylim(bottom=0)
|
|
653
|
+
ax.legend(loc="upper right", framealpha=0.85)
|
|
654
|
+
return ax
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def plot_Calibration_Estimates(
|
|
658
|
+
fit_params: np.ndarray | list[list[float]],
|
|
659
|
+
fit_error: np.ndarray | list[list[float]],
|
|
660
|
+
Title: str = "Calibration Line Fits",
|
|
661
|
+
) -> plt.Figure:
|
|
662
|
+
"""
|
|
663
|
+
Plot calibration slope and intercept estimate distributions.
|
|
664
|
+
|
|
665
|
+
Parameters
|
|
666
|
+
----------
|
|
667
|
+
fit_params : array-like
|
|
668
|
+
Fit parameters for slope and intercept, shape (n, 2).
|
|
669
|
+
fit_error : array-like
|
|
670
|
+
Fit uncertainties for slope and intercept, shape (n, 2).
|
|
671
|
+
Title : str, optional
|
|
672
|
+
Figure title. Default is "Calibration Line Fits".
|
|
673
|
+
|
|
674
|
+
Returns
|
|
675
|
+
-------
|
|
676
|
+
matplotlib.figure.Figure
|
|
677
|
+
Figure containing the slope and intercept estimate plots.
|
|
678
|
+
"""
|
|
679
|
+
fig, (ax1, ax2) = plt.subplots(nrows=1, ncols=2, figsize=(12, 6))
|
|
680
|
+
|
|
681
|
+
Slope_Fit_Params = gauss_agv_err(np.array(fit_params)[:, 0], np.array(fit_error)[:, 0])
|
|
682
|
+
plot_datapoints(Slope_Fit_Params[0], Slope_Fit_Params[1], ax=ax1)
|
|
683
|
+
ax1.set_xlabel("Calibration Slope", fontsize=20)
|
|
684
|
+
ax1.set_ylabel("Probability", fontsize=20)
|
|
685
|
+
|
|
686
|
+
Intercept_Fit_Params = gauss_agv_err(
|
|
687
|
+
np.array(fit_params)[:, 1], np.array(fit_error)[:, 1]
|
|
688
|
+
)
|
|
689
|
+
plot_datapoints(Intercept_Fit_Params[0], Intercept_Fit_Params[1], ax=ax2)
|
|
690
|
+
ax2.set_xlabel("Calibration Y-Intercept ppm", fontsize=20)
|
|
691
|
+
ax2.set_ylabel("Probability", fontsize=20)
|
|
692
|
+
|
|
693
|
+
plt.suptitle(Title, fontsize=20)
|
|
694
|
+
fig.tight_layout()
|
|
695
|
+
return fig
|
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: odr-bootstrap
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Orthogonal Distance Regression with Bootstrap Resampling for SIMS Calibration
|
|
5
|
+
Project-URL: Repository, https://github.com/whtowbin/odr-bootstrap
|
|
6
|
+
Project-URL: Issues, https://github.com/whtowbin/odr-bootstrap/issues
|
|
7
|
+
Project-URL: Documentation, https://odr-bootstrap.readthedocs.io
|
|
8
|
+
Project-URL: Changelog, https://github.com/whtowbin/odr-bootstrap/blob/main/CHANGELOG.md
|
|
9
|
+
Author-email: Henry Towbin <htowbin@caltech.edu>
|
|
10
|
+
License: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: ODR,SIMS,bootstrap,calibration,uncertainty
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Requires-Dist: matplotlib>=3.10.1
|
|
24
|
+
Requires-Dist: numpy>=2.2.4
|
|
25
|
+
Requires-Dist: pandas>=2.2.3
|
|
26
|
+
Requires-Dist: scipy>=1.15.2
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: mypy>=1.10.0; extra == 'dev'
|
|
29
|
+
Requires-Dist: pre-commit>=4.0.0; extra == 'dev'
|
|
30
|
+
Requires-Dist: pytest-cov>=4.0; extra == 'dev'
|
|
31
|
+
Requires-Dist: pytest>=7.0; extra == 'dev'
|
|
32
|
+
Requires-Dist: ruff>=0.8.0; extra == 'dev'
|
|
33
|
+
Provides-Extra: docs
|
|
34
|
+
Requires-Dist: sphinx-autodoc-typehints>=2.0.0; extra == 'docs'
|
|
35
|
+
Requires-Dist: sphinx-rtd-theme>=2.0.0; extra == 'docs'
|
|
36
|
+
Requires-Dist: sphinx>=8.0.0; extra == 'docs'
|
|
37
|
+
Provides-Extra: test
|
|
38
|
+
Requires-Dist: pytest-cov>=4.0; extra == 'test'
|
|
39
|
+
Requires-Dist: pytest>=7.0; extra == 'test'
|
|
40
|
+
Description-Content-Type: text/markdown
|
|
41
|
+
|
|
42
|
+
# ODR Bootstrap
|
|
43
|
+
|
|
44
|
+
[](https://github.com/whtowbin/odr-bootstrap/actions/workflows/tests.yml)
|
|
45
|
+
[](https://codecov.io/gh/whtowbin/odr-bootstrap)
|
|
46
|
+
[](https://pypi.org/project/odr-bootstrap/)
|
|
47
|
+
[](https://www.python.org/downloads/release/python-3120/)
|
|
48
|
+
[](https://opensource.org/licenses/MIT)
|
|
49
|
+
|
|
50
|
+
Orthogonal Distance Regression with Bootstrap Resampling for SIMS Calibration
|
|
51
|
+
|
|
52
|
+
A Python package for robust calibration curve fitting with proper uncertainty quantification in both x and y measurements.
|
|
53
|
+
|
|
54
|
+
## Overview
|
|
55
|
+
|
|
56
|
+
**What is ODR Bootstrap?**
|
|
57
|
+
|
|
58
|
+
When fitting calibration curves to scientific data, measurement errors exist in both the independent variable (x, e.g., concentration) and dependent variable (y, e.g., ion intensity). Ordinary least squares regression assumes errors only in y, leading to biased fits.
|
|
59
|
+
|
|
60
|
+
**Orthogonal Distance Regression (ODR)** properly accounts for uncertainties in both x and y. **Bootstrap resampling** estimates confidence intervals by repeatedly refitting the model to random subsamples of the calibration data.
|
|
61
|
+
|
|
62
|
+
This package combines these techniques for publication-ready uncertainty quantification in SIMS (Secondary Ion Mass Spectrometry) calibration analysis.
|
|
63
|
+
|
|
64
|
+
## Features
|
|
65
|
+
|
|
66
|
+
✅ NumPy-style documentation for all functions
|
|
67
|
+
✅ 22 comprehensive unit tests (100% pass rate)
|
|
68
|
+
✅ Runnable example workflow with synthetic data
|
|
69
|
+
✅ Compatible with scipy 1.15+ (deprecated API updates)
|
|
70
|
+
✅ Zero-intercept fits with proper parameter handling
|
|
71
|
+
✅ Publication-ready calibration plots
|
|
72
|
+
|
|
73
|
+
## Installation
|
|
74
|
+
|
|
75
|
+
### With UV (recommended)
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
uv pip install odr-bootstrap
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
### With pip
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
pip install odr-bootstrap
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
### From source
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
git clone https://github.com/whtowbin/odr-bootstrap.git
|
|
91
|
+
cd odr-bootstrap
|
|
92
|
+
uv sync
|
|
93
|
+
# or
|
|
94
|
+
pip install -e .
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
## Documentation
|
|
98
|
+
|
|
99
|
+
Full documentation is available at [Read the Docs](https://odr-bootstrap.readthedocs.io).
|
|
100
|
+
|
|
101
|
+
For quick reference, see:
|
|
102
|
+
- [API Reference](https://odr-bootstrap.readthedocs.io/en/latest/api.html)
|
|
103
|
+
- [Tutorial & Examples](https://odr-bootstrap.readthedocs.io/en/latest/tutorial.html)
|
|
104
|
+
- [Examples Directory](./examples)
|
|
105
|
+
|
|
106
|
+
## Quick Start
|
|
107
|
+
import numpy as np
|
|
108
|
+
import matplotlib.pyplot as plt
|
|
109
|
+
from odr_bootstrap import ODR_Bootstrap, plot_regression
|
|
110
|
+
|
|
111
|
+
# Prepare calibration data
|
|
112
|
+
x_standards = np.array([0.1, 0.5, 1.0, 2.0, 5.0]) # Concentrations
|
|
113
|
+
y_intensity = np.array([45, 200, 350, 700, 1450]) # Ion counts
|
|
114
|
+
x_uncertainty = np.array([0.01, 0.05, 0.1, 0.2, 0.5]) # Measurement errors in x
|
|
115
|
+
y_uncertainty = np.array([5, 20, 35, 60, 120]) # Measurement errors in y
|
|
116
|
+
|
|
117
|
+
# Run ODR bootstrap with 2000 resamples
|
|
118
|
+
confidence_data, best_fit_params, points, all_params, subsamples = ODR_Bootstrap(
|
|
119
|
+
x=x_standards,
|
|
120
|
+
y=y_intensity,
|
|
121
|
+
x_err=x_uncertainty,
|
|
122
|
+
y_err=y_uncertainty,
|
|
123
|
+
resample_draws=2000,
|
|
124
|
+
InterceptFit=True,
|
|
125
|
+
InitialGuess=[250, 10],
|
|
126
|
+
Confidence_Bound=0.95,
|
|
127
|
+
LineMax=6,
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
# Plot the result
|
|
131
|
+
fig, ax = plt.subplots(figsize=(8, 5))
|
|
132
|
+
plot_regression(confidence_data, datapoints=points, ax=ax,
|
|
133
|
+
ecolor='lightblue', line_color='darkblue', linewidth=2)
|
|
134
|
+
ax.set_xlabel('Concentration (ppm)')
|
|
135
|
+
ax.set_ylabel('Ion Intensity (counts)')
|
|
136
|
+
ax.set_title('SIMS Calibration Curve with 95% Bootstrap CI')
|
|
137
|
+
plt.tight_layout()
|
|
138
|
+
plt.savefig('calibration_curve.png', dpi=150)
|
|
139
|
+
plt.show()
|
|
140
|
+
|
|
141
|
+
# Access results
|
|
142
|
+
print(f"Slope: {best_fit_params[0]:.2f}")
|
|
143
|
+
print(f"Intercept: {best_fit_params[1]:.2f}")
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
### Plotting Calibration Estimate Distributions
|
|
147
|
+
|
|
148
|
+
```python
|
|
149
|
+
from odr_bootstrap import plot_Calibration_Estimates
|
|
150
|
+
import numpy as np
|
|
151
|
+
|
|
152
|
+
# Extract bootstrap parameter distributions
|
|
153
|
+
all_slopes = np.array([p[0] for p in all_params])
|
|
154
|
+
all_intercepts = np.array([p[1] for p in all_params])
|
|
155
|
+
|
|
156
|
+
# Create synthetic measurement ensemble (for visualization)
|
|
157
|
+
fit_params = np.array([[all_slopes.mean(), all_intercepts.mean()]] * 5)
|
|
158
|
+
fit_params += np.random.normal(0, [all_slopes.std(), all_intercepts.std()], (5, 2))
|
|
159
|
+
fit_error = np.array([[all_slopes.std(), all_intercepts.std()]] * 5)
|
|
160
|
+
|
|
161
|
+
# Generate plot
|
|
162
|
+
fig = plot_Calibration_Estimates(fit_params, fit_error,
|
|
163
|
+
Title="Calibration Slope & Intercept Distributions")
|
|
164
|
+
plt.savefig('calibration_estimates.png', dpi=150)
|
|
165
|
+
plt.show()
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
## Module Functions
|
|
169
|
+
|
|
170
|
+
### Core Fitting
|
|
171
|
+
|
|
172
|
+
- **`ODR_Linear(x, y, x_err, y_err, intercept=False, InitialGuess=[100, 1])`**
|
|
173
|
+
Single orthogonal distance regression fit with optional y-intercept.
|
|
174
|
+
Returns: (fitted_params, param_uncertainties)
|
|
175
|
+
|
|
176
|
+
- **`ODR_Linear_Test(x, y, x_err, y_err, ...)`**
|
|
177
|
+
ODR fit returning full scipy.odr output object for diagnostics.
|
|
178
|
+
|
|
179
|
+
- **`Bootstrap_fit(x, y, x_err, y_err, resample_draws, ...)`**
|
|
180
|
+
Resample data N times and fit ODR model to each resample.
|
|
181
|
+
Returns: (all_fit_params, resampled_data)
|
|
182
|
+
|
|
183
|
+
### Confidence Intervals
|
|
184
|
+
|
|
185
|
+
- **`Eval_Conf(Fit_Param, Confidence_Bound=0.95, LineMax=200, ...)`**
|
|
186
|
+
Compute confidence intervals from bootstrap parameter distributions.
|
|
187
|
+
Returns: DataFrame with confidence bounds indexed by x-values.
|
|
188
|
+
|
|
189
|
+
- **`ODR_Bootstrap(...)`**
|
|
190
|
+
Convenience wrapper combining Bootstrap_fit + Eval_Conf in one call.
|
|
191
|
+
|
|
192
|
+
### Statistics
|
|
193
|
+
|
|
194
|
+
- **`gauss_agv_err(concentrations, errors, ...)`**
|
|
195
|
+
Aggregate multiple normal distributions into single KDE estimate.
|
|
196
|
+
Returns: (distribution_dict, statistics_dict)
|
|
197
|
+
|
|
198
|
+
### Plotting
|
|
199
|
+
|
|
200
|
+
- **`plot_regression(confidence_df, datapoints=None, ax=None, ...)`**
|
|
201
|
+
Plot best-fit line with shaded confidence band.
|
|
202
|
+
|
|
203
|
+
- **`plot_datapoints(data, bounds, ax=None, ...)`**
|
|
204
|
+
Plot probability density curve with summary statistics.
|
|
205
|
+
|
|
206
|
+
- **`plot_Calibration_Estimates(fit_params, fit_error, Title=...)`**
|
|
207
|
+
Side-by-side slope and intercept distribution plots.
|
|
208
|
+
|
|
209
|
+
## Testing
|
|
210
|
+
|
|
211
|
+
Run the comprehensive test suite:
|
|
212
|
+
|
|
213
|
+
```bash
|
|
214
|
+
python -m unittest tests/test_odr_bootstrap.py -v
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
Or with pytest:
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
pytest tests/ -v
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
All 22 tests should pass.
|
|
224
|
+
|
|
225
|
+
## Example Workflow
|
|
226
|
+
|
|
227
|
+
See [examples/example.py](examples/example.py) for a complete working example:
|
|
228
|
+
|
|
229
|
+
```bash
|
|
230
|
+
cd examples
|
|
231
|
+
python example.py
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
This generates:
|
|
235
|
+
- `calibration_curve.png` - Fitted line with confidence band
|
|
236
|
+
- `calibration_estimates.png` - Bootstrap parameter distributions
|
|
237
|
+
|
|
238
|
+
## Recent Updates (April 2025)
|
|
239
|
+
|
|
240
|
+
- ✅ Fixed zero-intercept ODR fits with proper scipy.odr parameter wrapping
|
|
241
|
+
- ✅ Updated deprecated np.trapz API to np.trapezoid (scipy 1.15+ compatible)
|
|
242
|
+
- ✅ Added comprehensive NumPy-style docstrings to all functions
|
|
243
|
+
- ✅ Created 22 unit tests with 100% pass rate
|
|
244
|
+
- ✅ Generated runnable example workflow
|
|
245
|
+
|
|
246
|
+
## API Reference
|
|
247
|
+
|
|
248
|
+
Full documentation is available in function docstrings:
|
|
249
|
+
|
|
250
|
+
```python
|
|
251
|
+
from odr_bootstrap import ODR_Bootstrap
|
|
252
|
+
help(ODR_Bootstrap)
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
## Requirements
|
|
256
|
+
|
|
257
|
+
- Python >= 3.12
|
|
258
|
+
- numpy >= 2.2.4
|
|
259
|
+
- scipy >= 1.15.2
|
|
260
|
+
- pandas >= 2.2.3
|
|
261
|
+
- matplotlib >= 3.10.1
|
|
262
|
+
|
|
263
|
+
## License
|
|
264
|
+
|
|
265
|
+
MIT License - See [LICENSE](LICENSE) file for details.
|
|
266
|
+
|
|
267
|
+
## Citation
|
|
268
|
+
|
|
269
|
+
If you use this package in research, please cite:
|
|
270
|
+
|
|
271
|
+
```bibtex
|
|
272
|
+
@software{towbin2025odr,
|
|
273
|
+
title={ODR Bootstrap: Orthogonal Distance Regression with Bootstrap Resampling},
|
|
274
|
+
author={Towbin, Henry},
|
|
275
|
+
year={2025},
|
|
276
|
+
url={https://github.com/whtowbin/odr-bootstrap}
|
|
277
|
+
}
|
|
278
|
+
```
|
|
279
|
+
|
|
280
|
+
## Contributing
|
|
281
|
+
|
|
282
|
+
Contributions welcome! Please open an issue or submit a pull request.
|
|
283
|
+
|
|
284
|
+
## Acknowledgments
|
|
285
|
+
|
|
286
|
+
Developed at Caltech for SIMS calibration analysis workflows.
|
|
287
|
+
|
|
288
|
+
---
|
|
289
|
+
|
|
290
|
+
**Status**: Beta (0.1.0) | **Last Updated**: April 2025
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
odr_bootstrap/__init__.py,sha256=25xLKtD8aCq0bd0kimWiKph-XWJXFDI7DI-0X7DKhts,735
|
|
2
|
+
odr_bootstrap/core.py,sha256=sLOopbAsIOtSEHkaTaCbZzFfzLwwhCvMlFH1FE74lmo,20890
|
|
3
|
+
odr_bootstrap-0.1.0.dist-info/METADATA,sha256=0uWi6sDFvfk6Kd6DPWu4g_22zyZCpRGre1uuRi3HdL4,9389
|
|
4
|
+
odr_bootstrap-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
5
|
+
odr_bootstrap-0.1.0.dist-info/licenses/LICENSE,sha256=Jzd-Y8zUy8E5yPbw12N8GtnwpqW7keKt5lOZD2iPmis,1069
|
|
6
|
+
odr_bootstrap-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Henry Towbin
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|