odr-bootstrap 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,36 @@
1
+ """
2
+ ODR Bootstrap: Orthogonal Distance Regression with Bootstrap Resampling
3
+
4
+ A Python package for SIMS calibration analysis with proper uncertainty
5
+ quantification in both x and y measurements.
6
+ """
7
+
8
+ from .core import (
9
+ Bootstrap_fit,
10
+ Eval_Conf,
11
+ ODR_Bootstrap,
12
+ ODR_Linear,
13
+ ODR_Linear_Test,
14
+ gauss_agv_err,
15
+ plot_Calibration_Estimates,
16
+ plot_datapoints,
17
+ plot_regression,
18
+ slope_func,
19
+ yint_func,
20
+ )
21
+
22
+ __version__ = "0.1.0"
23
+ __author__ = "Henry Towbin"
24
+ __all__ = [
25
+ "ODR_Linear",
26
+ "ODR_Linear_Test",
27
+ "Bootstrap_fit",
28
+ "Eval_Conf",
29
+ "plot_regression",
30
+ "ODR_Bootstrap",
31
+ "gauss_agv_err",
32
+ "plot_datapoints",
33
+ "plot_Calibration_Estimates",
34
+ "yint_func",
35
+ "slope_func",
36
+ ]
odr_bootstrap/core.py ADDED
@@ -0,0 +1,695 @@
1
+ """
2
+ ODR bootstrapping utilities for SIMS calibration analysis.
3
+
4
+ This module provides orthogonal distance regression (ODR) with uncertainties
5
+ in both x and y, bootstrap resampling of fit parameters, confidence interval
6
+ evaluation for predicted fit lines, and calibration estimate plotting.
7
+
8
+ Dependencies
9
+ ------------
10
+ matplotlib
11
+ numpy
12
+ pandas
13
+ scipy
14
+
15
+ Changelog
16
+ ---------
17
+ April 2025:
18
+ - Fixed zero-intercept ODR initialization: properly wrap slope-only
19
+ initial guess in list for scipy.odr compatibility.
20
+ - Updated deprecated np.trapz to np.trapezoid for scipy 1.15+ compatibility.
21
+ """
22
+
23
+ from typing import Any
24
+
25
+ import matplotlib.pyplot as plt
26
+ import numpy as np
27
+ import pandas as pd
28
+ import scipy.stats as stats
29
+ from scipy import odr
30
+
31
+
32
+ def ODR_Linear(
33
+ x: np.ndarray | list[float],
34
+ y: np.ndarray | list[float],
35
+ x_err: np.ndarray | list[float],
36
+ y_err: np.ndarray | list[float],
37
+ intercept: bool = False,
38
+ InitialGuess: list[float] | None = None,
39
+ ) -> tuple[np.ndarray, np.ndarray]:
40
+ """
41
+ Fit a linear model using orthogonal distance regression (ODR).
42
+
43
+ Parameters
44
+ ----------
45
+ x : array-like
46
+ Independent variable values.
47
+ y : array-like
48
+ Dependent variable values.
49
+ x_err : array-like
50
+ Uncertainties in the independent variable values.
51
+ y_err : array-like
52
+ Uncertainties in the dependent variable values.
53
+ intercept : bool, optional
54
+ If True, fit `y = a * x + b`. If False, fit `y = a * x` through the origin.
55
+ Default is False.
56
+ InitialGuess : list, optional
57
+ Initial guess for the fit parameters. For intercept fits supply
58
+ `[slope, intercept]`. For zero-intercept fits supply `[slope]`.
59
+ Default is `[100, 1]`.
60
+
61
+ Returns
62
+ -------
63
+ tuple
64
+ `Popt`, `Perr` where `Popt` is the fitted parameter array and `Perr`
65
+ is the 1-sigma uncertainty array.
66
+ """
67
+ def yint_func(p: np.ndarray | list[float], x: np.ndarray | list[float]) -> np.ndarray:
68
+ a, b = p
69
+ return a * x + b
70
+
71
+ def slope_func(p: np.ndarray | list[float], x: np.ndarray | list[float]) -> np.ndarray:
72
+ a = p
73
+ return a * x
74
+
75
+ if InitialGuess is None:
76
+ InitialGuess = [100, 1]
77
+
78
+ linear_model = odr.Model(yint_func)
79
+ beta0 = InitialGuess
80
+ if intercept is False:
81
+ linear_model = odr.Model(slope_func)
82
+ # scipy.odr.ODR requires beta0 to be array-like; wrap scalar in list
83
+ beta0 = [InitialGuess[0]] # Use only slope for zero-intercept fit
84
+
85
+ data = odr.RealData(x, y, sx=x_err, sy=y_err)
86
+ myodr = odr.ODR(data, linear_model, beta0=beta0)
87
+ myodr.set_job(fit_type=0)
88
+ out = myodr.run()
89
+
90
+ Popt = out.beta
91
+ Perr = out.sd_beta
92
+ return Popt, Perr
93
+
94
+
95
+ def ODR_Linear_Test(
96
+ x: np.ndarray | list[float],
97
+ y: np.ndarray | list[float],
98
+ x_err: np.ndarray | list[float],
99
+ y_err: np.ndarray | list[float],
100
+ intercept: bool = False,
101
+ InitialGuess: list[float] = [100, 1],
102
+ ) -> tuple[np.ndarray, np.ndarray, Any]:
103
+ """
104
+ Fit a linear model using ODR and return the raw ODR output.
105
+
106
+ Parameters
107
+ ----------
108
+ x : array-like
109
+ Independent variable values.
110
+ y : array-like
111
+ Dependent variable values.
112
+ x_err : array-like
113
+ Uncertainties in the independent variable values.
114
+ y_err : array-like
115
+ Uncertainties in the dependent variable values.
116
+ intercept : bool, optional
117
+ If True, fit `y = a * x + b`. If False, fit `y = a * x` through the origin.
118
+ Default is False.
119
+ InitialGuess : list, optional
120
+ Initial guess for the fit parameters. For intercept fits supply
121
+ `[slope, intercept]`. For zero-intercept fits supply `[slope]`.
122
+ Default is `[100, 1]`.
123
+
124
+ Returns
125
+ -------
126
+ tuple
127
+ `Popt`, `Perr`, `odr_output` where `odr_output` is the full ODR result.
128
+ """
129
+ def yint_func(p, x):
130
+ a, b = p
131
+ return a * x + b
132
+
133
+ def slope_func(p, x):
134
+ a = p
135
+ return a * x
136
+
137
+ linear_model = odr.Model(yint_func)
138
+ beta0 = InitialGuess
139
+ if intercept is False:
140
+ linear_model = odr.Model(slope_func)
141
+ # scipy.odr.ODR requires beta0 to be array-like; wrap scalar in list
142
+ beta0 = [InitialGuess[0]] # Use only slope for zero-intercept fit
143
+
144
+ data = odr.RealData(x, y, sx=x_err, sy=y_err)
145
+ myodr = odr.ODR(data, linear_model, beta0=beta0)
146
+ myodr.set_job(fit_type=0)
147
+ out = myodr.run()
148
+
149
+ Popt = out.beta
150
+ Perr = out.sd_beta
151
+ return Popt, Perr, out
152
+
153
+
154
+ def Bootstrap_fit(
155
+ x: np.ndarray | list[float],
156
+ y: np.ndarray | list[float],
157
+ x_err: np.ndarray | list[float],
158
+ y_err: np.ndarray | list[float],
159
+ resample_draws: int,
160
+ InterceptFit: bool = True,
161
+ InitialGuess: list[float] = [100, 1],
162
+ ) -> tuple[list[np.ndarray], list[pd.DataFrame]]:
163
+ """
164
+ Perform bootstrap resampling of ODR linear fits.
165
+
166
+ Parameters
167
+ ----------
168
+ x : array-like
169
+ Independent variable values.
170
+ y : array-like
171
+ Dependent variable values.
172
+ x_err : array-like
173
+ Uncertainties in the independent variable values.
174
+ y_err : array-like
175
+ Uncertainties in the dependent variable values.
176
+ resample_draws : int
177
+ Number of bootstrap resamples to compute.
178
+ InterceptFit : bool, optional
179
+ If True, fit slope and intercept; if False, fit through the origin.
180
+ Default is True.
181
+ InitialGuess : list, optional
182
+ Initial guess for model parameters. Default is `[100, 1]`.
183
+
184
+ Returns
185
+ -------
186
+ tuple
187
+ fit_params : list of ndarray
188
+ First element is the fit result from the full dataset, followed
189
+ by all bootstrap fits.
190
+ subsamples : list of pandas.DataFrame
191
+ Resampled DataFrame objects used for each bootstrap iteration.
192
+ """
193
+ def resample(count):
194
+ return np.random.randint(0, count, count)
195
+
196
+ InitialGuess = list(InitialGuess)
197
+ if InterceptFit is False:
198
+ InitialGuess = [InitialGuess[0]]
199
+
200
+ data = np.array([x, x_err, y, y_err]).T
201
+ df = pd.DataFrame(data, columns=["x", "x_err", "y", "y_err"])
202
+ df.dropna(inplace=True)
203
+ length = len(df)
204
+
205
+ opt, err = ODR_Linear(
206
+ x=df["x"],
207
+ y=df["y"],
208
+ x_err=df["x_err"],
209
+ y_err=df["y_err"],
210
+ InitialGuess=InitialGuess,
211
+ intercept=InterceptFit,
212
+ )
213
+ Fit_Param = [opt]
214
+ subs = []
215
+
216
+ for _ in range(resample_draws):
217
+ sub = df.take(resample(length))
218
+ opt, err = ODR_Linear(
219
+ x=sub["x"],
220
+ y=sub["y"],
221
+ x_err=sub["x_err"],
222
+ y_err=sub["y_err"],
223
+ InitialGuess=InitialGuess,
224
+ intercept=InterceptFit,
225
+ )
226
+ Fit_Param.append(opt)
227
+ subs.append(sub)
228
+
229
+ return Fit_Param, subs
230
+
231
+
232
+ def yint_func(p: np.ndarray | list[float], x: np.ndarray | list[float]) -> np.ndarray:
233
+ """
234
+ Evaluate a line with slope and intercept.
235
+
236
+ Parameters
237
+ ----------
238
+ p : array-like
239
+ Parameter vector [slope, intercept].
240
+ x : array-like
241
+ Independent variable values.
242
+
243
+ Returns
244
+ -------
245
+ ndarray
246
+ Evaluated y values.
247
+ """
248
+ a, b = p
249
+ return a * x + b
250
+
251
+
252
+ def slope_func(p: np.ndarray | list[float], x: np.ndarray | list[float]) -> np.ndarray:
253
+ """
254
+ Evaluate a line through the origin.
255
+
256
+ Parameters
257
+ ----------
258
+ p : array-like
259
+ Parameter vector [slope].
260
+ x : array-like
261
+ Independent variable values.
262
+
263
+ Returns
264
+ -------
265
+ ndarray
266
+ Evaluated y values.
267
+ """
268
+ a = p
269
+ return a * x
270
+
271
+
272
+ def Eval_Conf(
273
+ Fit_Param: list[np.ndarray],
274
+ Confidence_Bound: float = 0.95,
275
+ LineMax: int = 200,
276
+ LineInt: int = 1,
277
+ **kwargs: Any,
278
+ ) -> pd.DataFrame:
279
+ """
280
+ Evaluate bootstrap confidence intervals for linear predictions.
281
+
282
+ Parameters
283
+ ----------
284
+ Fit_Param : list of array-like
285
+ Bootstrapped fit parameter vectors. Each row must contain either one
286
+ parameter (slope only) or two parameters (slope and intercept).
287
+ Confidence_Bound : float, optional
288
+ Confidence level expressed as a fraction between 0 and 1.
289
+ Default is 0.95.
290
+ LineMax : int, optional
291
+ Maximum x-value for the evaluation grid. Default is 200.
292
+ LineInt : int, optional
293
+ Step size for the evaluation grid. Default is 1.
294
+
295
+ Returns
296
+ -------
297
+ pandas.DataFrame
298
+ DataFrame indexed by x values containing columns:
299
+ - neg_error_bound
300
+ - pos_error_bound
301
+ - best_fit
302
+ - percent_error_neg
303
+ - percent_error_pos
304
+ """
305
+ if len(Fit_Param[0]) > 2:
306
+ raise ValueError(
307
+ "Fit_Param has too many inputs per row. Line inputs must be 1 or 2 parameters."
308
+ )
309
+
310
+ FitFunc = yint_func
311
+ if len(Fit_Param[0]) == 1:
312
+ FitFunc = slope_func
313
+
314
+ evaluated = []
315
+ x = np.arange(0, LineMax, LineInt)
316
+ for row in Fit_Param:
317
+ evaluated.append(FitFunc(row, x))
318
+
319
+ BootStp_Samples = pd.DataFrame(evaluated)
320
+ confidence_ints = []
321
+ for _, col in BootStp_Samples.items():
322
+ histrange = (np.nanmin(col), np.nanmax(col))
323
+ hist = np.histogram(col, bins=200, range=histrange)
324
+ conf_int = stats.rv_histogram(hist).interval(Confidence_Bound)
325
+ confidence_ints.append(conf_int)
326
+
327
+ Results = pd.DataFrame(
328
+ confidence_ints, columns=("neg_error_bound", "pos_error_bound")
329
+ )
330
+ Results.index = x
331
+ Results["best_fit"] = evaluated[0]
332
+ Results["percent_error_neg"] = (
333
+ Results["best_fit"] - Results["neg_error_bound"]
334
+ ) / np.abs(Results["best_fit"])
335
+ Results["percent_error_pos"] = (
336
+ Results["pos_error_bound"] - Results["best_fit"]
337
+ ) / np.abs(Results["best_fit"])
338
+
339
+ return Results
340
+
341
+
342
+ def plot_regression(
343
+ confidence_df: pd.DataFrame,
344
+ datapoints: pd.DataFrame | None = None,
345
+ LineMax: int = 200,
346
+ LineInt: int = 1,
347
+ ax: plt.Axes | None = None,
348
+ ecolor: str = "r",
349
+ line_color: str = "b",
350
+ sigma: int = 2,
351
+ e_alpha: float = 0.5,
352
+ **kwargs: Any,
353
+ ) -> plt.Axes:
354
+ """
355
+ Plot a best-fit regression line and its bootstrap confidence band.
356
+
357
+ Parameters
358
+ ----------
359
+ confidence_df : pandas.DataFrame
360
+ Output from `Eval_Conf` with columns `best_fit`, `neg_error_bound`, and
361
+ `pos_error_bound`.
362
+ datapoints : pandas.DataFrame, optional
363
+ DataFrame containing columns `x`, `y`, `xerr`, and `yerr`.
364
+ LineMax : int, optional
365
+ Accepted for compatibility but not used in this function.
366
+ LineInt : int, optional
367
+ Accepted for compatibility but not used in this function.
368
+ ax : matplotlib.axes.Axes, optional
369
+ Axis object to draw on. If None, the current axis is used.
370
+ ecolor : str, optional
371
+ Confidence band color. Default is 'r'.
372
+ line_color : str, optional
373
+ Best-fit line color. Default is 'b'.
374
+ sigma : int, optional
375
+ Ignored in the current implementation.
376
+ e_alpha : float, optional
377
+ Alpha transparency for the confidence band. Default is 0.5.
378
+
379
+ Returns
380
+ -------
381
+ matplotlib.axes.Axes
382
+ Axis containing the regression plot.
383
+ """
384
+ BestFitLine = confidence_df["best_fit"]
385
+ NegBound = confidence_df["neg_error_bound"]
386
+ PosBound = confidence_df["pos_error_bound"]
387
+
388
+ x = NegBound.index
389
+ if ax is None:
390
+ ax = plt.gca()
391
+
392
+ ax.fill_between(x, NegBound, PosBound, color=ecolor, alpha=e_alpha)
393
+ ax.plot(x, BestFitLine, color=line_color, **kwargs)
394
+
395
+ if datapoints is not None:
396
+ ax.errorbar(
397
+ x=datapoints["x"],
398
+ y=datapoints["y"],
399
+ yerr=datapoints["yerr"],
400
+ xerr=datapoints["xerr"],
401
+ marker=".",
402
+ fmt="g",
403
+ linestyle="none",
404
+ capsize=5,
405
+ markeredgewidth=1,
406
+ markersize=10,
407
+ label=None,
408
+ **kwargs,
409
+ )
410
+
411
+ return ax
412
+
413
+
414
+ def ODR_Bootstrap(
415
+ x: np.ndarray | list[float],
416
+ y: np.ndarray | list[float],
417
+ x_err: np.ndarray | list[float],
418
+ y_err: np.ndarray | list[float],
419
+ resample_draws: int = 5000,
420
+ LineMax: int = 200,
421
+ LineInterval: int = 1,
422
+ InterceptFit: bool = True,
423
+ InitialGuess: list[float] = [100, 1],
424
+ Confidence_Bound: float = 0.95,
425
+ plot: bool = False,
426
+ ax: plt.Axes | None = None,
427
+ **kwargs: Any,
428
+ ) -> tuple[pd.DataFrame, np.ndarray, pd.DataFrame, list[np.ndarray], list[pd.DataFrame]]:
429
+ """
430
+ Run bootstrap resampling for ODR linear fitting and compute confidence data.
431
+
432
+ Parameters
433
+ ----------
434
+ x : array-like
435
+ Independent variable values.
436
+ y : array-like
437
+ Dependent variable values.
438
+ x_err : array-like
439
+ Uncertainties in x.
440
+ y_err : array-like
441
+ Uncertainties in y.
442
+ resample_draws : int, optional
443
+ Number of bootstrap resamples. Default is 5000.
444
+ LineMax : int, optional
445
+ Maximum x value used in `Eval_Conf`. Default is 200.
446
+ LineInterval : int, optional
447
+ Step size used in `Eval_Conf`. Default is 1.
448
+ InterceptFit : bool, optional
449
+ If True, fit slope and intercept; if False, fit through the origin.
450
+ Default is True.
451
+ InitialGuess : list, optional
452
+ Initial guess for fit parameters. Default is `[100, 1]`.
453
+ Confidence_Bound : float, optional
454
+ Confidence level for interval estimation. Default is 0.95.
455
+ plot : bool, optional
456
+ Accepted for compatibility but not used in this implementation.
457
+ ax : matplotlib.axes.Axes, optional
458
+ Axis object for future plotting support.
459
+
460
+ Returns
461
+ -------
462
+ tuple
463
+ confidence_data : pandas.DataFrame
464
+ Confidence interval results from `Eval_Conf`.
465
+ best_fit_params : ndarray
466
+ Fit parameters for the full dataset.
467
+ points : pandas.DataFrame
468
+ Cleaned input data containing `x`, `y`, `xerr`, and `yerr`.
469
+ all_params : list of ndarray
470
+ All fit parameter vectors including bootstrap resamples.
471
+ subsamples : list of pandas.DataFrame
472
+ Bootstrap resampled subsets.
473
+ """
474
+ param, subs = Bootstrap_fit(
475
+ x, y, x_err, y_err, resample_draws, InterceptFit, InitialGuess
476
+ )
477
+ confidence_data = Eval_Conf(
478
+ Fit_Param=param,
479
+ Confidence_Bound=Confidence_Bound,
480
+ LineMax=LineMax,
481
+ LineInt=LineInterval,
482
+ )
483
+
484
+ points = pd.DataFrame({"x": x, "y": y, "xerr": x_err, "yerr": y_err})
485
+ points.dropna(inplace=True)
486
+
487
+ return confidence_data, param[0], points, param, subs
488
+
489
+
490
+ def gauss_agv_err(
491
+ concentrations: np.ndarray | list[float],
492
+ errors: np.ndarray | list[float],
493
+ cut_off: float = 0.000001,
494
+ ) -> tuple[dict[str, np.ndarray], dict[str, Any]]:
495
+ """
496
+ Compute an aggregate Gaussian distribution from values and uncertainties.
497
+
498
+ Combines multiple normal distributions into a single kernel density estimate.
499
+ Uses trapezoidal integration (via scipy.integrate.trapezoid) to normalize.
500
+
501
+ Parameters
502
+ ----------
503
+ concentrations : array-like
504
+ Central values for each Gaussian component.
505
+ errors : array-like
506
+ Standard deviations for each Gaussian component.
507
+ cut_off : float, optional
508
+ Probability density threshold for filtering low values.
509
+ Default is 1e-6.
510
+
511
+ Returns
512
+ -------
513
+ tuple
514
+ distribution : dict
515
+ Dictionary containing `x` and `y` arrays for the normalized density.
516
+ statistics : dict
517
+ Summary information including mean, mode, midpoint, and bounds.
518
+ """
519
+ def gaussian(x, sigma, avg):
520
+ return (1 / (sigma * np.sqrt(2 * np.pi))) * np.exp(
521
+ -0.5 * ((x - avg) / sigma) ** 2
522
+ )
523
+
524
+ def CI_bound(xi, data, bound_fraction):
525
+ for n, val in enumerate(np.cumsum(data)):
526
+ if val > bound_fraction:
527
+ return round(xi[n], 2)
528
+
529
+ def find_range(avgs, sigmas):
530
+ max_val = np.max(avgs) + 3 * np.max(sigmas)
531
+ min_val = np.min(avgs) - 3 * np.max(sigmas)
532
+ return min_val, max_val
533
+
534
+ min_val, max_val = find_range(concentrations, errors)
535
+ xi = np.arange(min_val, max_val, 0.01)
536
+ x = np.tile(xi, (len(concentrations), 1))
537
+ unnormed_data = np.sum(gaussian(x.T, errors, concentrations), axis=1)
538
+ # Use np.trapezoid (scipy >= 1.15) instead of deprecated np.trapz
539
+ data = unnormed_data / np.trapezoid(unnormed_data)
540
+
541
+ average = np.dot(xi, data) / np.sum(data)
542
+ most_frequent = xi[np.argmax(data)]
543
+ best_fit = concentrations[0]
544
+
545
+ center_of_mass = CI_bound(xi, data, 0.50)
546
+ one_sigma_bounds = CI_bound(xi, data, 0.16), CI_bound(xi, data, 0.84)
547
+ two_sigma_bounds = CI_bound(xi, data, 0.05), CI_bound(xi, data, 0.95)
548
+ CI_one_sigma = (
549
+ round(center_of_mass - one_sigma_bounds[0], 2),
550
+ round(one_sigma_bounds[1] - center_of_mass, 2),
551
+ )
552
+ CI_two_sigma = (
553
+ round(center_of_mass - two_sigma_bounds[0], 2),
554
+ round(two_sigma_bounds[1] - center_of_mass, 2),
555
+ )
556
+
557
+ return (
558
+ {"x": xi, "y": data},
559
+ {
560
+ "simple_best_fit": best_fit,
561
+ "mean": average,
562
+ "mode": most_frequent,
563
+ "mid_point": center_of_mass,
564
+ "one_sigma_bounds": one_sigma_bounds,
565
+ "two_sigma_bounds": two_sigma_bounds,
566
+ "CI_one_sigma": CI_one_sigma,
567
+ "CI_two_sigma": CI_two_sigma,
568
+ "n": len(concentrations),
569
+ },
570
+ )
571
+
572
+
573
+ def plot_datapoints(
574
+ data: dict[str, np.ndarray],
575
+ bounds: dict[str, Any],
576
+ ax: plt.Axes | None = None,
577
+ sample_name: str | None = None,
578
+ ) -> plt.Axes:
579
+ """
580
+ Plot a probability density curve and annotate summary statistics.
581
+
582
+ Parameters
583
+ ----------
584
+ data : dict
585
+ Dictionary containing `x` and `y` density arrays.
586
+ bounds : dict
587
+ Summary statistics returned by `gauss_agv_err`.
588
+ ax : matplotlib.axes.Axes, optional
589
+ Axis object to draw on. If None, the current axis is used.
590
+ sample_name : str, optional
591
+ Optional label or title text.
592
+
593
+ Returns
594
+ -------
595
+ matplotlib.axes.Axes
596
+ Axis containing the plotted density curve.
597
+ """
598
+ ax = ax or plt.gca()
599
+ x = data["x"]
600
+ y = data["y"]
601
+ ax.plot(x, y, linewidth=3)
602
+ ax.set_xlabel("Concentration ppm")
603
+ ax.set_ylabel("Probability")
604
+
605
+ ax.axvline(
606
+ x=bounds["mean"],
607
+ ymin=0,
608
+ color="b",
609
+ linestyle="dashed",
610
+ linewidth=3,
611
+ label="Mean",
612
+ )
613
+ ax.axvline(
614
+ x=bounds["mid_point"],
615
+ ymin=0,
616
+ color="g",
617
+ linestyle="dashed",
618
+ linewidth=3,
619
+ label="Mid-point & 65% CI",
620
+ )
621
+
622
+ CI_one_sigma = bounds["CI_one_sigma"]
623
+ CI_two_sigma = bounds["CI_two_sigma"]
624
+
625
+ ax.annotate(
626
+ f"""
627
+ Simple Best Fit: {float(bounds['simple_best_fit']):.2f}
628
+ Mean: {float(bounds['mean']):.2f}
629
+ Mode: {float(bounds['mode']):.2f}
630
+ Mid-point: {float(bounds['mid_point']):.2f}
631
+ Confidence Intervals
632
+ 68%: - {CI_one_sigma[0]:.2f} / +{CI_one_sigma[1]:.2f}
633
+ 95%: - {CI_two_sigma[0]:.2f} / +{CI_two_sigma[1]:.2f}
634
+ n: {bounds['n']}
635
+ """,
636
+ xy=(0.02, 0.68),
637
+ xycoords="axes fraction",
638
+ bbox=dict(boxstyle="square", fc="w", alpha=0.85),
639
+ )
640
+ eb = ax.errorbar(
641
+ x=bounds["mid_point"],
642
+ y=np.max(y) / 2,
643
+ xerr=np.array([[CI_one_sigma[0]], [CI_one_sigma[1]]]),
644
+ capsize=10,
645
+ elinewidth=3,
646
+ capthick=3,
647
+ ecolor="g",
648
+ linestyle="dashed",
649
+ )
650
+ eb[-1][0].set_linestyle("dashed")
651
+
652
+ ax.set_ylim(bottom=0)
653
+ ax.legend(loc="upper right", framealpha=0.85)
654
+ return ax
655
+
656
+
657
+ def plot_Calibration_Estimates(
658
+ fit_params: np.ndarray | list[list[float]],
659
+ fit_error: np.ndarray | list[list[float]],
660
+ Title: str = "Calibration Line Fits",
661
+ ) -> plt.Figure:
662
+ """
663
+ Plot calibration slope and intercept estimate distributions.
664
+
665
+ Parameters
666
+ ----------
667
+ fit_params : array-like
668
+ Fit parameters for slope and intercept, shape (n, 2).
669
+ fit_error : array-like
670
+ Fit uncertainties for slope and intercept, shape (n, 2).
671
+ Title : str, optional
672
+ Figure title. Default is "Calibration Line Fits".
673
+
674
+ Returns
675
+ -------
676
+ matplotlib.figure.Figure
677
+ Figure containing the slope and intercept estimate plots.
678
+ """
679
+ fig, (ax1, ax2) = plt.subplots(nrows=1, ncols=2, figsize=(12, 6))
680
+
681
+ Slope_Fit_Params = gauss_agv_err(np.array(fit_params)[:, 0], np.array(fit_error)[:, 0])
682
+ plot_datapoints(Slope_Fit_Params[0], Slope_Fit_Params[1], ax=ax1)
683
+ ax1.set_xlabel("Calibration Slope", fontsize=20)
684
+ ax1.set_ylabel("Probability", fontsize=20)
685
+
686
+ Intercept_Fit_Params = gauss_agv_err(
687
+ np.array(fit_params)[:, 1], np.array(fit_error)[:, 1]
688
+ )
689
+ plot_datapoints(Intercept_Fit_Params[0], Intercept_Fit_Params[1], ax=ax2)
690
+ ax2.set_xlabel("Calibration Y-Intercept ppm", fontsize=20)
691
+ ax2.set_ylabel("Probability", fontsize=20)
692
+
693
+ plt.suptitle(Title, fontsize=20)
694
+ fig.tight_layout()
695
+ return fig
@@ -0,0 +1,290 @@
1
+ Metadata-Version: 2.5
2
+ Name: odr-bootstrap
3
+ Version: 0.1.0
4
+ Summary: Orthogonal Distance Regression with Bootstrap Resampling for SIMS Calibration
5
+ Project-URL: Repository, https://github.com/whtowbin/odr-bootstrap
6
+ Project-URL: Issues, https://github.com/whtowbin/odr-bootstrap/issues
7
+ Project-URL: Documentation, https://odr-bootstrap.readthedocs.io
8
+ Project-URL: Changelog, https://github.com/whtowbin/odr-bootstrap/blob/main/CHANGELOG.md
9
+ Author-email: Henry Towbin <htowbin@caltech.edu>
10
+ License: MIT
11
+ License-File: LICENSE
12
+ Keywords: ODR,SIMS,bootstrap,calibration,uncertainty
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Scientific/Engineering
22
+ Requires-Python: >=3.10
23
+ Requires-Dist: matplotlib>=3.10.1
24
+ Requires-Dist: numpy>=2.2.4
25
+ Requires-Dist: pandas>=2.2.3
26
+ Requires-Dist: scipy>=1.15.2
27
+ Provides-Extra: dev
28
+ Requires-Dist: mypy>=1.10.0; extra == 'dev'
29
+ Requires-Dist: pre-commit>=4.0.0; extra == 'dev'
30
+ Requires-Dist: pytest-cov>=4.0; extra == 'dev'
31
+ Requires-Dist: pytest>=7.0; extra == 'dev'
32
+ Requires-Dist: ruff>=0.8.0; extra == 'dev'
33
+ Provides-Extra: docs
34
+ Requires-Dist: sphinx-autodoc-typehints>=2.0.0; extra == 'docs'
35
+ Requires-Dist: sphinx-rtd-theme>=2.0.0; extra == 'docs'
36
+ Requires-Dist: sphinx>=8.0.0; extra == 'docs'
37
+ Provides-Extra: test
38
+ Requires-Dist: pytest-cov>=4.0; extra == 'test'
39
+ Requires-Dist: pytest>=7.0; extra == 'test'
40
+ Description-Content-Type: text/markdown
41
+
42
+ # ODR Bootstrap
43
+
44
+ [![Tests](https://github.com/whtowbin/odr-bootstrap/actions/workflows/tests.yml/badge.svg)](https://github.com/whtowbin/odr-bootstrap/actions/workflows/tests.yml)
45
+ [![codecov](https://codecov.io/gh/whtowbin/odr-bootstrap/branch/main/graph/badge.svg)](https://codecov.io/gh/whtowbin/odr-bootstrap)
46
+ [![PyPI](https://img.shields.io/pypi/v/odr-bootstrap.svg)](https://pypi.org/project/odr-bootstrap/)
47
+ [![Python 3.12+](https://img.shields.io/badge/python-3.12+-blue.svg)](https://www.python.org/downloads/release/python-3120/)
48
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
49
+
50
+ Orthogonal Distance Regression with Bootstrap Resampling for SIMS Calibration
51
+
52
+ A Python package for robust calibration curve fitting with proper uncertainty quantification in both x and y measurements.
53
+
54
+ ## Overview
55
+
56
+ **What is ODR Bootstrap?**
57
+
58
+ When fitting calibration curves to scientific data, measurement errors exist in both the independent variable (x, e.g., concentration) and dependent variable (y, e.g., ion intensity). Ordinary least squares regression assumes errors only in y, leading to biased fits.
59
+
60
+ **Orthogonal Distance Regression (ODR)** properly accounts for uncertainties in both x and y. **Bootstrap resampling** estimates confidence intervals by repeatedly refitting the model to random subsamples of the calibration data.
61
+
62
+ This package combines these techniques for publication-ready uncertainty quantification in SIMS (Secondary Ion Mass Spectrometry) calibration analysis.
63
+
64
+ ## Features
65
+
66
+ ✅ NumPy-style documentation for all functions
67
+ ✅ 22 comprehensive unit tests (100% pass rate)
68
+ ✅ Runnable example workflow with synthetic data
69
+ ✅ Compatible with scipy 1.15+ (deprecated API updates)
70
+ ✅ Zero-intercept fits with proper parameter handling
71
+ ✅ Publication-ready calibration plots
72
+
73
+ ## Installation
74
+
75
+ ### With UV (recommended)
76
+
77
+ ```bash
78
+ uv pip install odr-bootstrap
79
+ ```
80
+
81
+ ### With pip
82
+
83
+ ```bash
84
+ pip install odr-bootstrap
85
+ ```
86
+
87
+ ### From source
88
+
89
+ ```bash
90
+ git clone https://github.com/whtowbin/odr-bootstrap.git
91
+ cd odr-bootstrap
92
+ uv sync
93
+ # or
94
+ pip install -e .
95
+ ```
96
+
97
+ ## Documentation
98
+
99
+ Full documentation is available at [Read the Docs](https://odr-bootstrap.readthedocs.io).
100
+
101
+ For quick reference, see:
102
+ - [API Reference](https://odr-bootstrap.readthedocs.io/en/latest/api.html)
103
+ - [Tutorial & Examples](https://odr-bootstrap.readthedocs.io/en/latest/tutorial.html)
104
+ - [Examples Directory](./examples)
105
+
106
+ ## Quick Start
107
+ import numpy as np
108
+ import matplotlib.pyplot as plt
109
+ from odr_bootstrap import ODR_Bootstrap, plot_regression
110
+
111
+ # Prepare calibration data
112
+ x_standards = np.array([0.1, 0.5, 1.0, 2.0, 5.0]) # Concentrations
113
+ y_intensity = np.array([45, 200, 350, 700, 1450]) # Ion counts
114
+ x_uncertainty = np.array([0.01, 0.05, 0.1, 0.2, 0.5]) # Measurement errors in x
115
+ y_uncertainty = np.array([5, 20, 35, 60, 120]) # Measurement errors in y
116
+
117
+ # Run ODR bootstrap with 2000 resamples
118
+ confidence_data, best_fit_params, points, all_params, subsamples = ODR_Bootstrap(
119
+ x=x_standards,
120
+ y=y_intensity,
121
+ x_err=x_uncertainty,
122
+ y_err=y_uncertainty,
123
+ resample_draws=2000,
124
+ InterceptFit=True,
125
+ InitialGuess=[250, 10],
126
+ Confidence_Bound=0.95,
127
+ LineMax=6,
128
+ )
129
+
130
+ # Plot the result
131
+ fig, ax = plt.subplots(figsize=(8, 5))
132
+ plot_regression(confidence_data, datapoints=points, ax=ax,
133
+ ecolor='lightblue', line_color='darkblue', linewidth=2)
134
+ ax.set_xlabel('Concentration (ppm)')
135
+ ax.set_ylabel('Ion Intensity (counts)')
136
+ ax.set_title('SIMS Calibration Curve with 95% Bootstrap CI')
137
+ plt.tight_layout()
138
+ plt.savefig('calibration_curve.png', dpi=150)
139
+ plt.show()
140
+
141
+ # Access results
142
+ print(f"Slope: {best_fit_params[0]:.2f}")
143
+ print(f"Intercept: {best_fit_params[1]:.2f}")
144
+ ```
145
+
146
+ ### Plotting Calibration Estimate Distributions
147
+
148
+ ```python
149
+ from odr_bootstrap import plot_Calibration_Estimates
150
+ import numpy as np
151
+
152
+ # Extract bootstrap parameter distributions
153
+ all_slopes = np.array([p[0] for p in all_params])
154
+ all_intercepts = np.array([p[1] for p in all_params])
155
+
156
+ # Create synthetic measurement ensemble (for visualization)
157
+ fit_params = np.array([[all_slopes.mean(), all_intercepts.mean()]] * 5)
158
+ fit_params += np.random.normal(0, [all_slopes.std(), all_intercepts.std()], (5, 2))
159
+ fit_error = np.array([[all_slopes.std(), all_intercepts.std()]] * 5)
160
+
161
+ # Generate plot
162
+ fig = plot_Calibration_Estimates(fit_params, fit_error,
163
+ Title="Calibration Slope & Intercept Distributions")
164
+ plt.savefig('calibration_estimates.png', dpi=150)
165
+ plt.show()
166
+ ```
167
+
168
+ ## Module Functions
169
+
170
+ ### Core Fitting
171
+
172
+ - **`ODR_Linear(x, y, x_err, y_err, intercept=False, InitialGuess=[100, 1])`**
173
+ Single orthogonal distance regression fit with optional y-intercept.
174
+ Returns: (fitted_params, param_uncertainties)
175
+
176
+ - **`ODR_Linear_Test(x, y, x_err, y_err, ...)`**
177
+ ODR fit returning full scipy.odr output object for diagnostics.
178
+
179
+ - **`Bootstrap_fit(x, y, x_err, y_err, resample_draws, ...)`**
180
+ Resample data N times and fit ODR model to each resample.
181
+ Returns: (all_fit_params, resampled_data)
182
+
183
+ ### Confidence Intervals
184
+
185
+ - **`Eval_Conf(Fit_Param, Confidence_Bound=0.95, LineMax=200, ...)`**
186
+ Compute confidence intervals from bootstrap parameter distributions.
187
+ Returns: DataFrame with confidence bounds indexed by x-values.
188
+
189
+ - **`ODR_Bootstrap(...)`**
190
+ Convenience wrapper combining Bootstrap_fit + Eval_Conf in one call.
191
+
192
+ ### Statistics
193
+
194
+ - **`gauss_agv_err(concentrations, errors, ...)`**
195
+ Aggregate multiple normal distributions into single KDE estimate.
196
+ Returns: (distribution_dict, statistics_dict)
197
+
198
+ ### Plotting
199
+
200
+ - **`plot_regression(confidence_df, datapoints=None, ax=None, ...)`**
201
+ Plot best-fit line with shaded confidence band.
202
+
203
+ - **`plot_datapoints(data, bounds, ax=None, ...)`**
204
+ Plot probability density curve with summary statistics.
205
+
206
+ - **`plot_Calibration_Estimates(fit_params, fit_error, Title=...)`**
207
+ Side-by-side slope and intercept distribution plots.
208
+
209
+ ## Testing
210
+
211
+ Run the comprehensive test suite:
212
+
213
+ ```bash
214
+ python -m unittest tests/test_odr_bootstrap.py -v
215
+ ```
216
+
217
+ Or with pytest:
218
+
219
+ ```bash
220
+ pytest tests/ -v
221
+ ```
222
+
223
+ All 22 tests should pass.
224
+
225
+ ## Example Workflow
226
+
227
+ See [examples/example.py](examples/example.py) for a complete working example:
228
+
229
+ ```bash
230
+ cd examples
231
+ python example.py
232
+ ```
233
+
234
+ This generates:
235
+ - `calibration_curve.png` - Fitted line with confidence band
236
+ - `calibration_estimates.png` - Bootstrap parameter distributions
237
+
238
+ ## Recent Updates (April 2025)
239
+
240
+ - ✅ Fixed zero-intercept ODR fits with proper scipy.odr parameter wrapping
241
+ - ✅ Updated deprecated np.trapz API to np.trapezoid (scipy 1.15+ compatible)
242
+ - ✅ Added comprehensive NumPy-style docstrings to all functions
243
+ - ✅ Created 22 unit tests with 100% pass rate
244
+ - ✅ Generated runnable example workflow
245
+
246
+ ## API Reference
247
+
248
+ Full documentation is available in function docstrings:
249
+
250
+ ```python
251
+ from odr_bootstrap import ODR_Bootstrap
252
+ help(ODR_Bootstrap)
253
+ ```
254
+
255
+ ## Requirements
256
+
257
+ - Python >= 3.12
258
+ - numpy >= 2.2.4
259
+ - scipy >= 1.15.2
260
+ - pandas >= 2.2.3
261
+ - matplotlib >= 3.10.1
262
+
263
+ ## License
264
+
265
+ MIT License - See [LICENSE](LICENSE) file for details.
266
+
267
+ ## Citation
268
+
269
+ If you use this package in research, please cite:
270
+
271
+ ```bibtex
272
+ @software{towbin2025odr,
273
+ title={ODR Bootstrap: Orthogonal Distance Regression with Bootstrap Resampling},
274
+ author={Towbin, Henry},
275
+ year={2025},
276
+ url={https://github.com/whtowbin/odr-bootstrap}
277
+ }
278
+ ```
279
+
280
+ ## Contributing
281
+
282
+ Contributions welcome! Please open an issue or submit a pull request.
283
+
284
+ ## Acknowledgments
285
+
286
+ Developed at Caltech for SIMS calibration analysis workflows.
287
+
288
+ ---
289
+
290
+ **Status**: Beta (0.1.0) | **Last Updated**: April 2025
@@ -0,0 +1,6 @@
1
+ odr_bootstrap/__init__.py,sha256=25xLKtD8aCq0bd0kimWiKph-XWJXFDI7DI-0X7DKhts,735
2
+ odr_bootstrap/core.py,sha256=sLOopbAsIOtSEHkaTaCbZzFfzLwwhCvMlFH1FE74lmo,20890
3
+ odr_bootstrap-0.1.0.dist-info/METADATA,sha256=0uWi6sDFvfk6Kd6DPWu4g_22zyZCpRGre1uuRi3HdL4,9389
4
+ odr_bootstrap-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
5
+ odr_bootstrap-0.1.0.dist-info/licenses/LICENSE,sha256=Jzd-Y8zUy8E5yPbw12N8GtnwpqW7keKt5lOZD2iPmis,1069
6
+ odr_bootstrap-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Henry Towbin
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.