ppidest 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ppidest/__init__.py ADDED
@@ -0,0 +1,199 @@
1
+ """Plotting Position-Information Divergence (PPID) framework.
2
+
3
+ Parameter estimation for univariate distributions based on discretizing
4
+ the cumulative distribution function with plotting positions and
5
+ minimizing information divergences against the empirical gaps.
6
+
7
+ Public API
8
+ ----------
9
+ - Distributions: :func:`normal_cdf`, :func:`normal_pdf`, :func:`gev_cdf`,
10
+ :func:`gev_pdf`, :func:`gev_quantile`, :func:`gev_return_level`,
11
+ :func:`weibull_cdf`, :func:`weibull_pdf`.
12
+ - Plotting positions: :func:`plotting_positions`.
13
+ - Divergences: :func:`kl_divergence`, :func:`kl_generalized_divergence`,
14
+ :func:`kl_symmetric_divergence`, :func:`jensen_shannon_divergence`,
15
+ :func:`beta_divergence`, :func:`power_divergence`,
16
+ :func:`renyi_divergence`.
17
+ - Estimators: :class:`ML`, :class:`PPID`
18
+ and the convenience wrapper :func:`find_ppid_min`.
19
+
20
+ The ``calc_*`` names from earlier releases are re-exported here as
21
+ backward-compatible aliases (e.g. :data:`calc_gev_cdf`).
22
+
23
+ Command line
24
+ ------------
25
+ The ``ppidest`` console script fits a distribution to sample values given
26
+ on the command line::
27
+
28
+ ppidest 0.38 0.51 1.44 2.14
29
+ ppidest 0.38 0.51 1.44 2.14 --distribution normal
30
+ ppidest 0.38 0.51 1.44 2.14 --plotting-position Weibull \\
31
+ --information-divergence "Jensen-Shannon"
32
+ """
33
+
34
+ import argparse
35
+
36
+ import numpy as np
37
+
38
+ from .distributions import (
39
+ gev_cdf,
40
+ gev_pdf,
41
+ gev_quantile,
42
+ gev_return_level,
43
+ normal_cdf,
44
+ normal_pdf,
45
+ weibull_cdf,
46
+ weibull_pdf,
47
+ calc_gev_cdf,
48
+ calc_gev_pdf,
49
+ calc_gev_quantile,
50
+ calc_norm_cdf,
51
+ calc_norm_pdf,
52
+ calc_return_level_gev,
53
+ calc_weibull_cdf,
54
+ calc_weibull_pdf,
55
+ )
56
+ from .divergences import (
57
+ beta_divergence,
58
+ jensen_shannon_divergence,
59
+ kl_divergence,
60
+ kl_generalized_divergence,
61
+ kl_symmetric_divergence,
62
+ power_divergence,
63
+ renyi_divergence,
64
+ )
65
+ from .estimators import (
66
+ ML,
67
+ PPID,
68
+ DIVERGENCE_NAMES,
69
+ find_ppid_min,
70
+ )
71
+ from .plotting import (
72
+ PLOTTING_POSITIONS,
73
+ plotting_position_coefficients,
74
+ plotting_positions,
75
+ )
76
+
77
+ __version__ = "0.1.0"
78
+
79
+ __all__ = [
80
+ # distributions
81
+ "normal_cdf", "normal_pdf",
82
+ "gev_cdf", "gev_pdf", "gev_quantile", "gev_return_level",
83
+ "weibull_cdf", "weibull_pdf",
84
+ # plotting positions
85
+ "PLOTTING_POSITIONS", "plotting_position_coefficients",
86
+ "plotting_positions",
87
+ # divergences
88
+ "kl_divergence", "kl_generalized_divergence", "kl_symmetric_divergence",
89
+ "jensen_shannon_divergence", "beta_divergence", "power_divergence",
90
+ "renyi_divergence",
91
+ # estimators
92
+ "ML", "PPID", "find_ppid_min",
93
+ # legacy aliases
94
+ "calc_norm_cdf", "calc_norm_pdf",
95
+ "calc_gev_cdf", "calc_gev_pdf", "calc_gev_quantile",
96
+ "calc_return_level_gev", "calc_weibull_cdf", "calc_weibull_pdf",
97
+ ]
98
+
99
+
100
+ #: Name of the parameter vector entry for each supported distribution.
101
+ _PARAMETER_NAMES = {
102
+ "gev": ("loc", "scale", "shape"),
103
+ "normal": ("loc", "scale"),
104
+ "weibull": ("threshold", "scale", "shape"),
105
+ }
106
+
107
+ #: CDF callable for each supported distribution.
108
+ _DISTRIBUTIONS = {
109
+ "gev": gev_cdf,
110
+ "normal": normal_cdf,
111
+ "weibull": weibull_cdf,
112
+ }
113
+
114
+
115
+ def _initial_parameters(name, xs):
116
+ """Return a sensible parameter guess derived from the sample."""
117
+ loc = float(np.mean(xs))
118
+ scale = float(np.std(xs))
119
+ if name == "normal":
120
+ return [loc, scale]
121
+ if name == "gev":
122
+ return [loc, scale, 0.0]
123
+ # three-parameter Weibull: threshold below the minimum observation,
124
+ # scale on the order of the observed range.
125
+ mn, mx = float(xs.min()), float(xs.max())
126
+ threshold = mn - 0.1 * (mx - mn) - 1e-9
127
+ return [threshold, (mx - mn) + 1e-9, 2.0]
128
+
129
+
130
+ def main(argv=None):
131
+ """Console entry point: fit a distribution to command-line sample data.
132
+
133
+ The parameters are fitted with the :class:`PPID` estimator using the
134
+ plotting position and information divergence given by
135
+ ``--plotting-position`` (default ``Hazen``) and
136
+ ``--information-divergence`` (default ``Kullback-Leibler``).
137
+ """
138
+ parser = argparse.ArgumentParser(
139
+ prog="ppidest",
140
+ description="Fit a univariate distribution by minimizing an "
141
+ "information divergence between plotting-position "
142
+ "empirical gaps and the candidate CDF gaps.")
143
+ parser.add_argument(
144
+ "xs", type=float, nargs="+", metavar="x",
145
+ help="sample values to fit")
146
+ parser.add_argument(
147
+ "-d", "--distribution", choices=sorted(_DISTRIBUTIONS),
148
+ default="gev",
149
+ help="distribution family to fit (default: gev)")
150
+ parser.add_argument(
151
+ "-p", "--plotting-position", choices=sorted(PLOTTING_POSITIONS),
152
+ default="Hazen",
153
+ help="plotting-position scheme (default: Hazen)")
154
+ parser.add_argument(
155
+ "-i", "--information-divergence", choices=sorted(DIVERGENCE_NAMES),
156
+ default="Kullback-Leibler",
157
+ help="information divergence to minimize (default: "
158
+ "Kullback-Leibler)")
159
+ parser.add_argument(
160
+ "--pdiv", type=float, default=0.5, metavar="VALUE",
161
+ help="order/index parameter for the beta, power or Renyi "
162
+ "divergences (default: 0.5)")
163
+ parser.add_argument(
164
+ "--initial", type=float, nargs="+", metavar="PAR",
165
+ help="initial parameters, overriding the data-driven defaults "
166
+ "(e.g. '--initial 0 1 0.25' for GEV)")
167
+ parser.add_argument(
168
+ "--version", action="version", version=f"%(prog)s {__version__}")
169
+ args = parser.parse_args(argv)
170
+
171
+ xs = np.asarray(args.xs)
172
+ if xs.size < 2:
173
+ parser.error("at least two sample values are required")
174
+
175
+ name = args.distribution
176
+ cdf = _DISTRIBUTIONS[name]
177
+ par_names = _PARAMETER_NAMES[name]
178
+ if args.initial is not None:
179
+ if len(args.initial) != len(par_names):
180
+ parser.error(
181
+ f"--initial must provide {len(par_names)} values for "
182
+ f"{name} ({', '.join(par_names)})")
183
+ par_0 = list(args.initial)
184
+ else:
185
+ par_0 = _initial_parameters(name, xs)
186
+
187
+ res = PPID(xs, cdf, plotposition=args.plotting_position,
188
+ divergence=args.information_divergence).find_min_div(
189
+ par_0, pdiv=args.pdiv)
190
+
191
+ header = ("distribution", "plotting position", "information divergence")
192
+ values = (name, args.plotting_position, args.information_divergence)
193
+ for key, value in zip(header, values):
194
+ print(f"{key:26s}: {value}")
195
+ for pname, pvalue in zip(par_names, res.x):
196
+ print(f"{pname:26s}: {pvalue:.6g}")
197
+ print(f"{'minimum divergence':26s}: {res.fun:.6g}")
198
+ print(f"{'converged':26s}: {res.success}")
199
+ return res
@@ -0,0 +1,240 @@
1
+ """Probability distributions used by the estimation methods.
2
+
3
+ Every distribution function in this module follows the same calling
4
+ convention: the first argument is the evaluation point (a scalar or a
5
+ NumPy array) and the second is the parameter vector ``par``. Keeping the
6
+ signature uniform lets estimators treat ``cdf`` and ``pdf`` callables as
7
+ first-class objects.
8
+
9
+ Parameters are supplied in plain tuples or lists so that they can be passed
10
+ straight to an optimizer (e.g. ``scipy.optimize.minimize``).
11
+ """
12
+
13
+ import numpy as np
14
+ from scipy.special import ndtr
15
+
16
+ SQRT_2PI = np.sqrt(2.0 * np.pi)
17
+
18
+
19
+ ################################################################################
20
+ # Normal distribution
21
+ ################################################################################
22
+
23
+
24
+ def normal_cdf(x, par):
25
+ """Cumulative distribution function of the normal distribution.
26
+
27
+ Parameters
28
+ ----------
29
+ x : float or ndarray
30
+ Evaluation point(s).
31
+ par : sequence of float
32
+ Parameters ``(loc, scale)``.
33
+
34
+ Returns
35
+ -------
36
+ float or ndarray
37
+ ``Phi((x - loc) / scale)``.
38
+ """
39
+ return ndtr((x - par[0]) / par[1])
40
+
41
+
42
+ def normal_pdf(x, par):
43
+ """Probability density function of the normal distribution.
44
+
45
+ Parameters
46
+ ----------
47
+ x : float or ndarray
48
+ Evaluation point(s).
49
+ par : sequence of float
50
+ Parameters ``(loc, scale)``.
51
+
52
+ Returns
53
+ -------
54
+ float or ndarray
55
+ Density at each evaluation point.
56
+ """
57
+ z = (x - par[0]) / par[1]
58
+ return np.exp(-0.5 * z ** 2) / (par[1] * SQRT_2PI)
59
+
60
+
61
+ ################################################################################
62
+ # Generalized extreme value (GEV) distribution
63
+ ################################################################################
64
+
65
+
66
+ def gev_cdf(x, par):
67
+ """Cumulative distribution function of the GEV distribution.
68
+
69
+ The GEV distribution function is
70
+
71
+ .. math::
72
+
73
+ F(x) = \\exp\\left[-\\left(1 + \\xi \\frac{x - \\mu}{\\sigma}
74
+ \\right)^{-1/\\xi}\\right]
75
+
76
+ with the Gumbel limit :math:`F(x) = \\exp\\left[-\\exp(-y)\\right]`
77
+ taken when :math:`\\xi \\approx 0`.
78
+
79
+ Parameters
80
+ ----------
81
+ x : float or ndarray
82
+ Evaluation point(s).
83
+ par : sequence of float
84
+ Parameters ``(location, scale, shape)`` :math:`(\\mu, \\sigma, \\xi)`.
85
+ The shape parameter follows the convention in which
86
+ :math:`\\xi > 0` gives the Fréchet (heavy tail) family.
87
+
88
+ Returns
89
+ -------
90
+ float or ndarray
91
+ ``F(x)``.
92
+ """
93
+ y = (x - par[0]) / par[1]
94
+ xi = par[2]
95
+ if abs(xi) < 1e-10:
96
+ return np.exp(-np.exp(-y))
97
+ # During optimization the parameters may leave the support of the
98
+ # distribution; such evaluations legitimately yield NaN.
99
+ with np.errstate(invalid="ignore", over="ignore", divide="ignore"):
100
+ return np.exp(-(1.0 + xi * y) ** (-1 / xi))
101
+
102
+
103
+ def gev_pdf(x, par):
104
+ """Probability density function of the GEV distribution.
105
+
106
+ Parameters
107
+ ----------
108
+ x : float or ndarray
109
+ Evaluation point(s).
110
+ par : sequence of float
111
+ Parameters ``(location, scale, shape)``.
112
+
113
+ Returns
114
+ -------
115
+ float or ndarray
116
+ Density at each evaluation point.
117
+ """
118
+ with np.errstate(invalid="ignore", over="ignore", divide="ignore"):
119
+ if abs(par[2]) < 1e-10:
120
+ t = np.exp(-(x - par[0]) / par[1])
121
+ return 1 / par[1] * t * np.exp(-t)
122
+ t = (1 + par[2] / par[1] * (x - par[0])) ** (-1 / par[2])
123
+ return 1 / par[1] * t ** (par[2] + 1) * np.exp(-t)
124
+
125
+
126
+ def gev_quantile(ff, par):
127
+ """Quantile function (inverse CDF) of the GEV distribution.
128
+
129
+ Returns the value :math:`x` such that :math:`F(x) = ff`.
130
+
131
+ Parameters
132
+ ----------
133
+ ff : float or ndarray
134
+ Cumulative probabilities in :math:`(0, 1)`.
135
+ par : sequence of float
136
+ Parameters ``(location, scale, shape)``.
137
+
138
+ Returns
139
+ -------
140
+ float or ndarray
141
+ Quantile(s).
142
+ """
143
+ if abs(par[2]) < 1e-10:
144
+ return par[0] - par[1] * np.log(-np.log(ff))
145
+ return par[0] - par[1] / par[2] * (1 - (-np.log(ff)) ** (-par[2]))
146
+
147
+
148
+ def gev_return_level(p, par):
149
+ """Return level of the GEV distribution for a given return period.
150
+
151
+ The return level :math:`z_p` associated with the return period
152
+ :math:`p` (given as multiples of the block length) satisfies
153
+ :math:`F(z_p) = 1 - 1 / p`.
154
+
155
+ Parameters
156
+ ----------
157
+ p : float
158
+ Return period, expressed as multiples of the block length.
159
+ par : sequence of float
160
+ Parameters ``(location, scale, shape)``.
161
+
162
+ Returns
163
+ -------
164
+ float
165
+ Return level.
166
+ """
167
+ ff = 1 - 1 / p
168
+ return gev_quantile(ff, par)
169
+
170
+
171
+ ################################################################################
172
+ # Three-parameter Weibull distribution
173
+ ################################################################################
174
+
175
+
176
+ def weibull_cdf(x, par):
177
+ """Cumulative distribution function of the three-parameter Weibull
178
+ distribution with threshold :math:`\\gamma`.
179
+
180
+ .. math::
181
+
182
+ F(x) = 1 - \\exp\\left[-\\left(\\frac{x - \\gamma}{\\lambda}
183
+ \\right)^k\\right]
184
+
185
+ Parameters
186
+ ----------
187
+ x : float or ndarray
188
+ Evaluation point(s).
189
+ par : sequence of float
190
+ Parameters ``(threshold, scale, shape)``
191
+ :math:`(\\gamma, \\lambda, k)`.
192
+
193
+ Returns
194
+ -------
195
+ float or ndarray
196
+ ``F(x)``.
197
+ """
198
+ return 1 - np.exp(-((x - par[0]) / par[1]) ** par[2])
199
+
200
+
201
+ def weibull_pdf(x, par):
202
+ """Probability density function of the three-parameter Weibull
203
+ distribution.
204
+
205
+ Parameters
206
+ ----------
207
+ x : float or ndarray
208
+ Evaluation point(s).
209
+ par : sequence of float
210
+ Parameters ``(threshold, scale, shape)``.
211
+
212
+ Returns
213
+ -------
214
+ float or ndarray
215
+ Density at each evaluation point.
216
+ """
217
+ return par[2] / par[1] * ((x - par[0]) / par[1]) ** (par[2] - 1) \
218
+ * np.exp(-((x - par[0]) / par[1]) ** par[2])
219
+
220
+
221
+ ################################################################################
222
+ # Legacy aliases
223
+ ################################################################################
224
+
225
+ #: Backward-compatible alias for :func:`normal_cdf`.
226
+ calc_norm_cdf = normal_cdf
227
+ #: Backward-compatible alias for :func:`normal_pdf`.
228
+ calc_norm_pdf = normal_pdf
229
+ #: Backward-compatible alias for :func:`gev_cdf`.
230
+ calc_gev_cdf = gev_cdf
231
+ #: Backward-compatible alias for :func:`gev_pdf`.
232
+ calc_gev_pdf = gev_pdf
233
+ #: Backward-compatible alias for :func:`gev_quantile`.
234
+ calc_gev_quantile = gev_quantile
235
+ #: Backward-compatible alias for :func:`gev_return_level`.
236
+ calc_return_level_gev = gev_return_level
237
+ #: Backward-compatible alias for :func:`weibull_cdf`.
238
+ calc_weibull_cdf = weibull_cdf
239
+ #: Backward-compatible alias for :func:`weibull_pdf`.
240
+ calc_weibull_pdf = weibull_pdf
ppidest/divergences.py ADDED
@@ -0,0 +1,185 @@
1
+ """Information divergences between discrete probability vectors.
2
+
3
+ In the PPID framework a candidate parametric distribution is compared
4
+ with the empirical discretization of the data: ``das`` holds the empirical
5
+ gap probabilities derived from plotting positions and ``ddas`` holds the
6
+ corresponding model gaps :math:`F(x_j) - F(x_{j-1})` from the candidate
7
+ CDF. Each function below quantifies the agreement of the two probability
8
+ vectors.
9
+
10
+ All functions accept the two probability vectors as 1-D NumPy arrays of
11
+ equal length and return a non-negative scalar divergence.
12
+ """
13
+
14
+ import numpy as np
15
+
16
+
17
+
18
+ def kl_divergence(das, ddas):
19
+ """Kullback-Leibler divergence ``sum(das * log(das / ddas))``.
20
+
21
+ Parameters
22
+ ----------
23
+ das : ndarray
24
+ Empirical gap probabilities.
25
+ ddas : ndarray
26
+ Model gap probabilities.
27
+
28
+ Returns
29
+ -------
30
+ float
31
+ Kullback-Leibler divergence.
32
+ """
33
+ return np.sum(das * np.log(das / ddas))
34
+
35
+
36
+ def kl_generalized_divergence(das, ddas):
37
+ """Generalized Kullback-Leibler divergence
38
+ ``sum(das * log(das / ddas) - das + ddas)``.
39
+
40
+ Unlike the plain KL divergence this variant is symmetric in its
41
+ treatment of the two vectors (it vanishes when they coincide even
42
+ when the vectors are not normalized).
43
+
44
+ Parameters
45
+ ----------
46
+ das : ndarray
47
+ Empirical gap probabilities.
48
+ ddas : ndarray
49
+ Model gap probabilities.
50
+
51
+ Returns
52
+ -------
53
+ float
54
+ Generalized Kullback-Leibler divergence.
55
+ """
56
+ return np.sum(das * np.log(das / ddas) - das + ddas)
57
+
58
+
59
+ def kl_symmetric_divergence(das, ddas):
60
+ """Symmetric Kullback-Leibler divergence
61
+ ``sum((ddas - das) * log(ddas / das))``.
62
+
63
+ Parameters
64
+ ----------
65
+ das : ndarray
66
+ Empirical gap probabilities.
67
+ ddas : ndarray
68
+ Model gap probabilities.
69
+
70
+ Returns
71
+ -------
72
+ float
73
+ Symmetric Kullback-Leibler divergence.
74
+ """
75
+ return np.sum((ddas - das) * np.log(ddas / das))
76
+
77
+
78
+ def jensen_shannon_divergence(das, ddas):
79
+ """Jensen-Shannon divergence based on the midpoint mixture
80
+ ``mas = 0.5 * (das + ddas)``.
81
+
82
+ Parameters
83
+ ----------
84
+ das : ndarray
85
+ Empirical gap probabilities.
86
+ ddas : ndarray
87
+ Model gap probabilities.
88
+
89
+ Returns
90
+ -------
91
+ float
92
+ Jensen-Shannon divergence.
93
+ """
94
+ mas = 0.5 * (das + ddas)
95
+ return 0.5 * np.sum(
96
+ das * np.log(das / mas) + ddas * np.log(ddas / mas))
97
+
98
+
99
+ def beta_divergence(das, ddas, beta):
100
+ """Beta (Cichocki) divergence of order ``beta`` (with ``beta != 0, 1``).
101
+
102
+ .. math::
103
+
104
+ D_\\beta(d \\,||\\, \\hat d)
105
+ = \\sum_j \\left[
106
+ \\frac{d_j (d_j^{\\beta - 1} - \\hat d_j^{\\beta - 1})}
107
+ {\\beta - 1}
108
+ - \\frac{d_j^{\\beta} - \\hat d_j^{\\beta}}{\\beta}
109
+ \\right]
110
+
111
+ Parameters
112
+ ----------
113
+ das : ndarray
114
+ Empirical gap probabilities.
115
+ ddas : ndarray
116
+ Model gap probabilities.
117
+ beta : float
118
+ Order parameter of the divergence. The special cases
119
+ ``beta -> 1`` (Kullback-Leibler) and ``beta -> 2`` (Euclidean)
120
+ are approached only in the limit.
121
+
122
+ Returns
123
+ -------
124
+ float
125
+ Beta divergence.
126
+ """
127
+ aa = das * (das ** (beta - 1) - ddas ** (beta - 1)) / (beta - 1)
128
+ bb = (das ** beta - ddas ** beta) / beta
129
+ # NaN from out-of-support model gaps is expected while optimizing.
130
+ with np.errstate(invalid="ignore"):
131
+ return aa.sum() - bb.sum()
132
+
133
+
134
+ def power_divergence(das, ddas, lmbd):
135
+ """Power divergence of index ``lmbd``.
136
+
137
+ .. math::
138
+
139
+ D_\\lambda(d \\,||\\, \\hat d)
140
+ = \\frac{1}{\\lambda (\\lambda + 1)}
141
+ \\sum_j d_j \\left[
142
+ \\left(\\frac{d_j}{\\hat d_j}\\right)^{\\lambda} - 1
143
+ \\right]
144
+
145
+ Parameters
146
+ ----------
147
+ das : ndarray
148
+ Empirical gap probabilities.
149
+ ddas : ndarray
150
+ Model gap probabilities.
151
+ lmbd : float
152
+ Index of the divergence (``lmbd != 0, -1``).
153
+
154
+ Returns
155
+ -------
156
+ float
157
+ Power divergence.
158
+ """
159
+ return 1 / (lmbd * (lmbd + 1)) * np.sum(das * ((das / ddas) ** lmbd - 1))
160
+
161
+
162
+ def renyi_divergence(das, ddas, alpha):
163
+ """Rényi divergence of order ``alpha``.
164
+
165
+ .. math::
166
+
167
+ D_\\alpha(d \\,||\\, \\hat d)
168
+ = \\frac{1}{\\alpha - 1}
169
+ \\log \\sum_j d_j^{\\alpha} \\hat d_j^{1 - \\alpha}
170
+
171
+ Parameters
172
+ ----------
173
+ das : ndarray
174
+ Empirical gap probabilities.
175
+ ddas : ndarray
176
+ Model gap probabilities.
177
+ alpha : float
178
+ Order of the divergence (``alpha != 1``).
179
+
180
+ Returns
181
+ -------
182
+ float
183
+ Rényi divergence.
184
+ """
185
+ return 1 / (alpha - 1) * np.log(np.sum(das ** alpha * ddas ** (1 - alpha)))
ppidest/estimators.py ADDED
@@ -0,0 +1,395 @@
1
+ """Parameter estimators based on CDF discretization.
2
+
3
+ This module collects the estimators of the Plotting Position-Information
4
+ Divergence (PPID) framework:
5
+
6
+ - :class:`ML` -- maximum likelihood (uses the PDF directly);
7
+ - :class:`PPID` -- PPID estimators minimizing one of several
8
+ information divergences between the empirical gaps (plotting positions)
9
+ and the model gaps;
10
+
11
+ All estimators expose a ``find_*`` method that wraps
12
+ ``scipy.optimize.minimize`` and mirror its keyword arguments.
13
+ """
14
+
15
+ import numpy as np
16
+ import scipy.optimize
17
+
18
+ from .divergences import (
19
+ beta_divergence,
20
+ jensen_shannon_divergence,
21
+ kl_divergence,
22
+ kl_generalized_divergence,
23
+ kl_symmetric_divergence,
24
+ power_divergence,
25
+ renyi_divergence,
26
+ )
27
+ from .plotting import (
28
+ _plotting_positions_from_order_stats,
29
+ plotting_position_coefficients,
30
+ )
31
+
32
+
33
+ #: Divergences that require an extra scalar parameter (``beta``, ``power``,
34
+ #: ``Renyi``): mapping divergence name to the ``PPID`` objective method.
35
+ _PARAMETRIC_DIVERGENCES = {
36
+ "beta": "beta_div",
37
+ "power": "power_div",
38
+ "Renyi": "renyi_div",
39
+ }
40
+
41
+ #: Divergences used directly on the gap vectors: mapping divergence name
42
+ #: to the ``PPID`` objective method.
43
+ _NONPARAMETRIC_DIVERGENCES = {
44
+ "Kullback-Leibler": "kld",
45
+ "KL_generalized": "kld_generalized",
46
+ "KL_symmetric": "kld_symmetric",
47
+ "Jensen-Shannon": "jensen_shannon_div",
48
+ }
49
+
50
+ #: Supported information divergences for :class:`PPID`.
51
+ DIVERGENCE_NAMES = tuple(
52
+ _NONPARAMETRIC_DIVERGENCES) + tuple(_PARAMETRIC_DIVERGENCES)
53
+
54
+
55
+ ################################################################################
56
+ # Shared helpers
57
+ ################################################################################
58
+
59
+
60
+ def _minimize(fun, x0, method="Nelder-Mead", jac=None, hess=None, hessp=None,
61
+ bounds=None, constraints=(), tol=None, callback=None,
62
+ options=None, args=()):
63
+ """Thin wrapper around ``scipy.optimize.minimize`` sharing the estimator
64
+ keyword interface."""
65
+ x0 = np.asarray(x0, dtype=float).ravel()
66
+ return scipy.optimize.minimize(
67
+ fun, x0, args=args, method=method, jac=jac, hess=hess, hessp=hessp,
68
+ bounds=bounds, constraints=constraints, tol=tol,
69
+ callback=callback, options=options)
70
+
71
+
72
+ def _cdf_gaps(cdf, uniques, par, *args):
73
+ """Return the model gap probabilities on the augmented support
74
+ ``[0, uniques, 1]``. Extra ``args`` are forwarded to ``cdf``."""
75
+ ffs = np.concatenate([np.array([0.0]), cdf(uniques, par, *args),
76
+ np.array([1.0])])
77
+ return np.diff(ffs)
78
+
79
+
80
+ ################################################################################
81
+ # Maximum likelihood
82
+ ################################################################################
83
+
84
+
85
+ class ML:
86
+ """Maximum-likelihood estimator based on a parametric PDF."""
87
+
88
+ def __init__(self, xs, pdf):
89
+ """Initialize the estimator.
90
+
91
+ Parameters
92
+ ----------
93
+ xs : array_like
94
+ Observed values.
95
+ pdf : callable
96
+ ``pdf(x, par)`` returning the density at ``x`` for the
97
+ parameter vector ``par``.
98
+ """
99
+ self.xs = np.sort(np.asarray(xs))
100
+ self.pdf = pdf
101
+
102
+ def nll(self, par):
103
+ """Negative log-likelihood at ``par``.
104
+
105
+ Parameters
106
+ ----------
107
+ par : sequence of float
108
+ Candidate parameters.
109
+
110
+ Returns
111
+ -------
112
+ float
113
+ ``- sum(log(pdf(x_i, par) + 1e-10))``.
114
+ """
115
+ return -np.sum(np.log(self.pdf(self.xs, par) + 1e-10))
116
+
117
+ def loglikelihood(self, par):
118
+ """Elementwise log-density of each observation at ``par``.
119
+
120
+ Parameters
121
+ ----------
122
+ par : sequence of float
123
+ Candidate parameters.
124
+
125
+ Returns
126
+ -------
127
+ ndarray
128
+ Log-density evaluated at every sample point.
129
+ """
130
+ return np.log(self.pdf(self.xs, par))
131
+
132
+ def find_mle(self, par_0, method="Nelder-Mead", jac=None, hess=None,
133
+ hessp=None, bounds=None, constraints=(), tol=None,
134
+ callback=None, options=None):
135
+ """Locate the maximum-likelihood estimate.
136
+
137
+ Parameters
138
+ ----------
139
+ par_0 : sequence of float
140
+ Initial parameter guess.
141
+ method, jac, hess, hessp, bounds, constraints, tol, callback, options
142
+ Forwarded to :func:`scipy.optimize.minimize`.
143
+
144
+ Returns
145
+ -------
146
+ OptimizeResult
147
+ Result of the minimization.
148
+ """
149
+ return _minimize(
150
+ self.nll, par_0, method=method, jac=jac, hess=hess, hessp=hessp,
151
+ bounds=bounds, constraints=constraints, tol=tol,
152
+ callback=callback, options=options)
153
+
154
+
155
+ ################################################################################
156
+ # PPID estimators
157
+ ################################################################################
158
+
159
+
160
+ class PPID:
161
+ """Plotting Position-Information Divergence estimator.
162
+
163
+ The empirical cumulative distribution is discretized by plotting
164
+ positions (see :mod:`ppidest.plotting`) and a parametric CDF is fitted
165
+ by minimizing an information divergence between the empirical gaps and
166
+ the model gaps.
167
+ """
168
+
169
+ #: Parametric divergences (name to objective method).
170
+ _PARAMETRIC_DIVERGENCES = _PARAMETRIC_DIVERGENCES
171
+ #: Non-parametric divergences (name to objective method).
172
+ _NONPARAMETRIC_DIVERGENCES = _NONPARAMETRIC_DIVERGENCES
173
+
174
+ def __init__(self, xs, cdf, plotposition="Hazen",
175
+ divergence="Kullback-Leibler"):
176
+ """Initialize the estimator.
177
+
178
+ Parameters
179
+ ----------
180
+ xs : array_like
181
+ Observed values.
182
+ cdf : callable
183
+ ``cdf(x, par)`` returning the CDF at ``x`` for the parameter
184
+ vector ``par``.
185
+ plotposition : str, optional
186
+ Plotting-position scheme; see :mod:`ppidest.plotting`.
187
+ Defaults to ``"Hazen"``.
188
+ divergence : str, optional
189
+ Information divergence minimized by :meth:`find_min_div`.
190
+ Defaults to ``"Kullback-Leibler"``.
191
+ """
192
+ self.xs = np.sort(np.asarray(xs))
193
+ self.cdf = cdf
194
+ self.uniques, self.idxs, self.counts = np.unique(
195
+ self.xs, return_index=True, return_counts=True)
196
+ self.n_xs = self.xs.size
197
+ self.n_uniques = self.counts.size
198
+ self.plotposition = plotposition
199
+ self.divergence = divergence
200
+
201
+ @property
202
+ def divergence(self):
203
+ """Information divergence minimized by :meth:`find_min_div`.
204
+
205
+ Assigning a new divergence validates the name against the
206
+ supported set.
207
+ """
208
+ return self._divergence
209
+
210
+ @divergence.setter
211
+ def divergence(self, name):
212
+ self._validate_divergence(name)
213
+ self._divergence = name
214
+
215
+ def _validate_divergence(self, name):
216
+ if (name not in self._NONPARAMETRIC_DIVERGENCES
217
+ and name not in self._PARAMETRIC_DIVERGENCES):
218
+ raise ValueError(
219
+ f"Unknown divergence {name!r}. "
220
+ f"Choose from: {sorted(DIVERGENCE_NAMES)}")
221
+
222
+ def set_divergence(self, divergence):
223
+ """Switch the divergence minimized by :meth:`find_min_div`."""
224
+ self.divergence = divergence
225
+
226
+ @property
227
+ def plotposition(self):
228
+ """Plotting-position scheme in use.
229
+
230
+ Assigning a new scheme recomputes :attr:`p_ast_s` and :attr:`das`
231
+ from the stored sample.
232
+ """
233
+ return self._plotposition
234
+
235
+ @plotposition.setter
236
+ def plotposition(self, name):
237
+ plotting_position_coefficients(name)
238
+ self._plotposition = name
239
+ self.p_ast_s, self.das = _plotting_positions_from_order_stats(
240
+ self.uniques, self.idxs, self.counts, self.n_xs, name)
241
+
242
+ @property
243
+ def plottingposition(self):
244
+ """Alias of :attr:`plotposition`."""
245
+ return self.plotposition
246
+
247
+ @plottingposition.setter
248
+ def plottingposition(self, name):
249
+ self.plotposition = name
250
+
251
+ def set_plotting_positions(self, plotposition):
252
+ """Switch the plotting-position scheme, recomputing the empirical
253
+ gaps from the stored sample."""
254
+ self.plotposition = plotposition
255
+
256
+ def get_plotting_positions(self):
257
+ """Return the empirical gaps of the current plotting-position scheme.
258
+
259
+ Returns
260
+ -------
261
+ p_ast_s : ndarray
262
+ Cumulative plotting probabilities (with boundary points).
263
+ d_ast_s : ndarray
264
+ Empirical gap probabilities.
265
+ """
266
+ return self.p_ast_s, self.das
267
+
268
+ def calc_dd_ast_s(self, par, *args):
269
+ """Model gap probabilities for the candidate parameters.
270
+
271
+ Parameters
272
+ ----------
273
+ par : sequence of float
274
+ Candidate parameters.
275
+ args : tuple, optional
276
+ Extra positional arguments forwarded to the CDF callable.
277
+
278
+ Returns
279
+ -------
280
+ ndarray
281
+ Gaps of ``[0, cdf(uniques, par, *args), 1]``.
282
+ """
283
+ return _cdf_gaps(self.cdf, self.uniques, par, *args)
284
+
285
+ def beta_div(self, par, beta, *args):
286
+ """Beta divergence between the empirical and model gaps."""
287
+ return beta_divergence(self.das, self.calc_dd_ast_s(par, *args), beta)
288
+
289
+ def power_div(self, par, lmbd, *args):
290
+ """Power divergence between the empirical and model gaps."""
291
+ return power_divergence(self.das, self.calc_dd_ast_s(par, *args), lmbd)
292
+
293
+ def renyi_div(self, par, alpha, *args):
294
+ """Rényi divergence between the empirical and model gaps."""
295
+ return renyi_divergence(self.das, self.calc_dd_ast_s(par, *args), alpha)
296
+
297
+ def kld(self, par, *args):
298
+ """Kullback-Leibler divergence between the empirical and model gaps."""
299
+ return kl_divergence(self.das, self.calc_dd_ast_s(par, *args))
300
+
301
+ def kld_generalized(self, par, *args):
302
+ """Generalized KL divergence between the empirical and model gaps."""
303
+ return kl_generalized_divergence(self.das, self.calc_dd_ast_s(par, *args))
304
+
305
+ def kld_symmetric(self, par, *args):
306
+ """Symmetric KL divergence between the empirical and model gaps."""
307
+ return kl_symmetric_divergence(self.das, self.calc_dd_ast_s(par, *args))
308
+
309
+ def jensen_shannon_div(self, par, *args):
310
+ """Jensen-Shannon divergence between the empirical and model gaps."""
311
+ return jensen_shannon_divergence(self.das, self.calc_dd_ast_s(par, *args))
312
+
313
+ def find_min_div(self, par, divergence=None, pdiv=0.5, args=(),
314
+ method="Nelder-Mead", jac=None, hess=None, hessp=None,
315
+ bounds=None, constraints=(), tol=None, callback=None,
316
+ options=None):
317
+ """Minimize the chosen information divergence.
318
+
319
+ Parameters
320
+ ----------
321
+ par : sequence of float
322
+ Initial parameter guess.
323
+ divergence : str, optional
324
+ One of ``"Kullback-Leibler"``, ``"KL_generalized"``,
325
+ ``"KL_symmetric"``, ``"Jensen-Shannon"``, ``"beta"``,
326
+ ``"power"``, ``"Renyi"``. Defaults to the estimator's
327
+ ``divergence`` (set at construction or via
328
+ :meth:`set_divergence`).
329
+ pdiv : float, optional
330
+ Order/index parameter used by the parametric divergences
331
+ (``beta``, ``power``, ``Renyi``). Defaults to ``0.5``.
332
+ args : tuple, optional
333
+ Extra positional arguments forwarded to the CDF callable
334
+ (appended after ``pdiv`` for the parametric divergences).
335
+ method, jac, hess, hessp, bounds, constraints, tol, callback, options
336
+ Forwarded to :func:`scipy.optimize.minimize`.
337
+
338
+ Returns
339
+ -------
340
+ OptimizeResult
341
+ Result of the minimization.
342
+
343
+ Raises
344
+ ------
345
+ ValueError
346
+ If ``divergence`` is not supported.
347
+ """
348
+ if divergence is None:
349
+ divergence = self.divergence
350
+ if divergence in self._PARAMETRIC_DIVERGENCES:
351
+ func = getattr(self, self._PARAMETRIC_DIVERGENCES[divergence])
352
+ opt_args = (pdiv,) + tuple(args)
353
+ elif divergence in self._NONPARAMETRIC_DIVERGENCES:
354
+ func = getattr(
355
+ self, self._NONPARAMETRIC_DIVERGENCES[divergence])
356
+ opt_args = tuple(args)
357
+ else:
358
+ raise ValueError(
359
+ f"Unknown divergence {divergence!r}. "
360
+ f"Choose from: {sorted(DIVERGENCE_NAMES)}")
361
+ return _minimize(
362
+ func, par, args=opt_args, method=method, jac=jac, hess=hess,
363
+ hessp=hessp, bounds=bounds, constraints=constraints, tol=tol,
364
+ callback=callback, options=options)
365
+
366
+
367
+ def find_ppid_min(xs, cdf, par, plotposition="Hazen",
368
+ divergence="Kullback-Leibler"):
369
+ """Convenience wrapper around :class:`PPID`.
370
+
371
+ Builds a :class:`PPID` estimator for ``(xs, cdf)``, fits it with the
372
+ requested divergence and returns the raw minimization result.
373
+
374
+ Parameters
375
+ ----------
376
+ xs : array_like
377
+ Observed values.
378
+ cdf : callable
379
+ ``cdf(x, par)`` returning the CDF at ``x`` for the parameter
380
+ vector ``par``.
381
+ par : sequence of float
382
+ Initial parameter guess.
383
+ plotposition : str, optional
384
+ Plotting-position scheme. Defaults to ``"Hazen"``.
385
+ divergence : str, optional
386
+ Information divergence to minimize. Defaults to
387
+ ``"Kullback-Leibler"``.
388
+
389
+ Returns
390
+ -------
391
+ OptimizeResult
392
+ Result of the minimization.
393
+ """
394
+ inst = PPID(xs, cdf, plotposition=plotposition, divergence=divergence)
395
+ return inst.find_min_div(par)
ppidest/plotting.py ADDED
@@ -0,0 +1,106 @@
1
+ """Plotting positions used to discretize the empirical CDF.
2
+
3
+ Plotting positions assign a plotting probability :math:`p_{(i)}` to the
4
+ i-th order statistic. The PPID framework converts these into a discrete
5
+ probability vector (the "empirical gaps") by aggregating the positions
6
+ shared by tied observations and augmenting the support with the boundary
7
+ bin at :math:`0` and :math:`1`.
8
+ """
9
+
10
+ import numpy as np
11
+
12
+ #: Registry of plotting-position coefficients ``(alpha, beta)`` such that
13
+ #: ``p_i = (i - alpha) / (n + 1 - alpha - beta)``.
14
+ PLOTTING_POSITIONS = {
15
+ "Weibull": (0.0, 0.0), # i / (n + 1); unbiased for the uniform CDF
16
+ "median": (0.3, 0.3), # (i - 0.3) / (n + 0.4); estimates the median
17
+ "Gringorten": (0.44, 0.44), # (i - 0.44) / (n + 0.12); optimal for Gumbel
18
+ "Hazen": (0.5, 0.5), # (i - 0.5) / n; piecewise linear approximation
19
+ }
20
+
21
+
22
+ def plotting_position_coefficients(name):
23
+ """Return the ``(alpha, beta)`` coefficients of a named plotting position.
24
+
25
+ Parameters
26
+ ----------
27
+ name : str
28
+ One of ``"Weibull"``, ``"median"``, ``"Gringorten"``, ``"Hazen"``.
29
+
30
+ Returns
31
+ -------
32
+ tuple of float
33
+ ``(alpha, beta)`` coefficients.
34
+
35
+ Raises
36
+ ------
37
+ ValueError
38
+ If ``name`` is not a registered plotting position.
39
+ """
40
+ if name not in PLOTTING_POSITIONS:
41
+ raise ValueError(
42
+ f"Unknown plotting position {name!r}. "
43
+ f"Choose from: {list(PLOTTING_POSITIONS)}"
44
+ )
45
+ return PLOTTING_POSITIONS[name]
46
+
47
+
48
+ def plotting_positions(xs, name="Hazen"):
49
+ """Discretize a sample into empirical gap probabilities.
50
+
51
+ Each gap :math:`d^*_j` is the aggregated plotting probability of a
52
+ unique value, so the cumulative vector :math:`p^*_s` has mass
53
+ ``[0, p1, ..., pk, 1]`` and the gaps sum to one.
54
+
55
+ Parameters
56
+ ----------
57
+ xs : array_like
58
+ Observed values (not necessarily sorted).
59
+ name : str, optional
60
+ Plotting-position scheme. Defaults to ``"Hazen"``.
61
+
62
+ Returns
63
+ -------
64
+ p_ast_s : ndarray of shape (k + 2,)
65
+ Cumulative plotting probabilities including the boundary points
66
+ ``0`` and ``1``.
67
+ d_ast_s : ndarray of shape (k + 1,)
68
+ Empirical gap probabilities, ``diff(p_ast_s)``.
69
+ """
70
+ xs = np.sort(np.asarray(xs))
71
+ uniques, idxs, counts = np.unique(xs, return_index=True, return_counts=True)
72
+ return _plotting_positions_from_order_stats(
73
+ uniques, idxs, counts, xs.size, name)
74
+
75
+
76
+ def _plotting_positions_from_order_stats(uniques, idxs, counts, n_xs, name):
77
+ """Aggregate plotting positions from already-computed order statistics.
78
+
79
+ Parameters
80
+ ----------
81
+ uniques : ndarray
82
+ Sorted unique values.
83
+ idxs : ndarray
84
+ Indices of the first occurrence of each unique value in the
85
+ sorted sample.
86
+ counts : ndarray
87
+ Number of occurrences of each unique value.
88
+ n_xs : int
89
+ Sample size.
90
+ name : str
91
+ Plotting-position scheme.
92
+
93
+ Returns
94
+ -------
95
+ p_ast_s, d_ast_s : ndarray
96
+ Same shapes and semantics as :func:`plotting_positions`.
97
+ """
98
+ alpha, beta = plotting_position_coefficients(name)
99
+ n_e = n_xs + 1 - alpha - beta
100
+ ps = (np.arange(1, n_xs + 1) - alpha) / n_e
101
+ # cum_sum[i:j] == ps[i:j].sum() (prefix sum with a leading zero).
102
+ cum_sum = np.cumsum(np.insert(ps, 0, 0))
103
+ pre_p_ast = (cum_sum[idxs + counts] - cum_sum[idxs]) / counts
104
+ p_ast_s = np.concatenate([np.array([0.0]), pre_p_ast, np.array([1.0])])
105
+ d_ast_s = np.diff(p_ast_s)
106
+ return p_ast_s, d_ast_s
@@ -0,0 +1,195 @@
1
+ Metadata-Version: 2.4
2
+ Name: ppidest
3
+ Version: 0.1.0
4
+ Summary: Plotting Position-Information Divergence framework for parameter estimation of univariate distributions
5
+ Keywords: statistics,parameter-estimation,information-divergence,plotting-positions,extremes
6
+ Author: Takuya Kawanishi
7
+ Author-email: Takuya Kawanishi <takuya@exanalytics.sakura.ne.jp>
8
+ License-Expression: MIT
9
+ Classifier: Development Status :: 4 - Beta
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.13
14
+ Classifier: Topic :: Scientific/Engineering :: Mathematics
15
+ Requires-Dist: numpy>=2.5.3
16
+ Requires-Dist: scipy>=1.18.1
17
+ Requires-Python: >=3.13
18
+ Project-URL: Repository, https://codeberg.org/takuya_kawanishi/ppidest
19
+ Project-URL: Homepage, https://codeberg.org/takuya_kawanishi/ppidest
20
+ Description-Content-Type: text/markdown
21
+
22
+ # ppidest
23
+
24
+ **Plotting Position — Information Divergence** framework for parameter
25
+ estimation of univariate distributions.
26
+
27
+ The empirical cumulative distribution function of a sample is discretized
28
+ using *plotting positions*, and a parametric distribution is fitted by
29
+ minimizing an *information divergence* between the empirical gaps and the
30
+ gaps implied by the candidate CDF.
31
+
32
+ ## Installation
33
+
34
+ Requires Python ≥ 3.13.
35
+
36
+ ```bash
37
+ uv sync # create the environment and install the package
38
+ uv run python -m unittest discover -s tests # run the test suite
39
+ ```
40
+
41
+ Dependencies: `numpy`, `scipy`. The package is also installable from a
42
+ source checkout with `pip install .` or `uv pip install .`, and the
43
+ distributions on PyPI ship a `ppidest` console script (see
44
+ [Command line](#command-line)).
45
+
46
+ ## Quick start
47
+
48
+ Fit a GEV model to a small sample using the default (Kullback-Leibler)
49
+ divergence:
50
+
51
+ ```python
52
+ import numpy as np
53
+ import ppidest
54
+
55
+ xs = np.array([0.38, 0.51, 1.44, 2.14])
56
+
57
+ # Plotting Position – Information Divergence estimator
58
+ res = ppidest.find_ppid_min(xs, ppidest.gev_cdf, [0.0, 1.0, 0.25])
59
+ print(res.x) # fitted (loc, scale, shape) -> [0.5600, 0.3639, 1.0585]
60
+ print(res.fun) # minimized divergence -> 0.17543
61
+
62
+ # Compute a return level from the fitted model
63
+ ppidest.gev_return_level(100.0, res.x)
64
+ ```
65
+
66
+ Comparison across estimation methods (`scipy.stats` is used only to
67
+ generate the data / provide the density here):
68
+
69
+ ```python
70
+ import scipy.stats
71
+
72
+ xs = np.sort(scipy.stats.norm.rvs(loc=1.0, scale=2.0, size=8))
73
+
74
+ # Density/CDF callables must use the (x, par) convention
75
+ def norm_pdf(x, par):
76
+ return scipy.stats.norm.pdf(x, loc=par[0], scale=par[1])
77
+
78
+ # Maximum likelihood from the PDF
79
+ ml = ppidest.ML(xs, norm_pdf).find_mle([1.0, 2.0])
80
+
81
+ # PPID with a different divergence and plot position
82
+ ppid = ppidest.PPID(xs, ppidest.normal_cdf, plotposition="Weibull")
83
+ res = ppid.find_min_div([1.0, 2.0], divergence="Jensen-Shannon")
84
+
85
+ # Forward extra positional arguments to the CDF callable
86
+ def norm_cdf_given_mu(x, par, mu):
87
+ return ppidest.normal_cdf(x, [mu, par[0]])
88
+
89
+ ppid = ppidest.PPID(xs, norm_cdf_given_mu)
90
+ res = ppid.find_min_div([1.0, 2.0], args=(0.0,)) # scale only, mu fixed
91
+ ```
92
+ ```
93
+
94
+ ## Background
95
+
96
+ For a sorted sample `x_(1) <= ... <= x_(n)` a plotting position assigns a
97
+ probability to the i-th order statistic:
98
+
99
+ ```
100
+ p_i = (i - alpha) / (n + 1 - alpha - beta)
101
+ ```
102
+
103
+ `alpha` and `beta` are fixed by the chosen scheme:
104
+
105
+ | Scheme | alpha | beta |
106
+ |--------------|-------|------|
107
+ | Weibull | 0.0 | 0.0 |
108
+ | median | 0.3 | 0.3 |
109
+ | Gringorten | 0.44 | 0.44 |
110
+ | Hazen | 0.5 | 0.5 |
111
+
112
+ Ties are aggregated so each unique value carries the summed probability
113
+ of the order statistics sharing it, and the support is augmented with two
114
+ boundary bins at `0` and `1`. The result is an empirical probability
115
+ vector `d*` (the "empirical gaps"). A candidate distribution with
116
+ parameters `θ` gives model gaps
117
+
118
+ ```
119
+ dd_j(θ) = F(x_j; θ) - F(x_{j-1}; θ), with x_0 = -inf, x_{k+1} = +inf
120
+ ```
121
+
122
+ The estimator minimizes an information divergence `D(d* || dd(θ))` over
123
+ `θ`:
124
+
125
+ - **Kullback-Leibler** —
126
+ `Σ d* log(d*/dd)`
127
+ - **generalized KL** —
128
+ `Σ d* log(d*/dd) - d* + dd`
129
+ - **symmetric KL** —
130
+ `Σ (dd - d*) log(dd/d*)`
131
+ - **Jensen-Shannon** —
132
+ `½ Σ d* log(d*/m) + dd log(dd/m)`, `m = (d* + dd)/2`
133
+ - **beta** (order `β`) —
134
+ `Σ d*(d*^(β-1) - dd^(β-1))/(β-1) - (d*^β - dd^β)/β`
135
+ - **power** (index `λ`) —
136
+ `1/(λ(λ+1)) Σ d*[(d*/dd)^λ - 1]`
137
+ - **Rényi** (order `α`) —
138
+ `1/(α-1) log Σ d*^α dd^(1-α)`
139
+
140
+ ## Package layout
141
+
142
+ ```
143
+ src/ppidest/
144
+ distributions.py distribution cdf/pdf/quantile functions
145
+ (normal, GEV, three-parameter Weibull)
146
+ plotting.py plotting-position schemes and empirical gaps
147
+ divergences.py information divergences between probability vectors
148
+ estimators.py ML, PPID estimators
149
+ __init__.py public API and legacy calc_* aliases
150
+ ```
151
+
152
+ ### Distributions
153
+
154
+ All distribution functions share the signature `f(x, par)` with
155
+ `par = (loc, scale, shape)` (normal and Weibull parameters differ, see
156
+ their docstrings):
157
+
158
+ - `normal_cdf`, `normal_pdf`
159
+ - `gev_cdf`, `gev_pdf`, `gev_quantile`, `gev_return_level`
160
+ (GEV shape `ξ` uses the convention `ξ > 0` → heavy tail)
161
+ - `weibull_cdf`, `weibull_pdf`
162
+
163
+ ### Divergences
164
+
165
+ `ppidest.kl_divergence`, `kl_generalized_divergence`,
166
+ `kl_symmetric_divergence`, `jensen_shannon_divergence`,
167
+ `beta_divergence(das, ddas, beta)`, `power_divergence(das, ddas, lmbd)`,
168
+ `renyi_divergence(das, ddas, alpha)` — all take the empirical gaps
169
+ `das` and model gaps `ddas` as 1-D arrays.
170
+
171
+ ### Estimators
172
+
173
+ | Estimator | Objective | Fit method |
174
+ |-----------|-----------|------------|
175
+ | `ML(xs, pdf)` | negative log-likelihood | `find_mle` |
176
+ | `PPID(xs, cdf)` | chosen information divergence | `find_min_div` |
177
+
178
+ Every `find_*` method mirrors the keyword arguments of
179
+ `scipy.optimize.minimize` (`method`, `jac`, `hess`, `bounds`, ...).
180
+ `PPID.find_min_div` accepts `divergence` and, for the parametric
181
+ divergences, `pdiv`. The parametric divergence arguments are passed as
182
+ singular floats (the pre-1.0 API required a length-one list).
183
+
184
+ The legacy `calc_gev_cdf`, ... names are exported from the package root as
185
+ aliases for backward compatibility, e.g. `ppidest.calc_gev_cdf` is the
186
+ same object as `ppidest.gev_cdf`.
187
+
188
+ ## Notes on optimization
189
+
190
+ Nelder–Mead is the default solver because the objective only needs
191
+ CDF evaluation. During the search the parameters may leave the support
192
+ of the distribution, producing `NaN`; these evaluations are harmless and
193
+ the corresponding warnings are suppressed inside the distribution and
194
+ divergence functions. Careful initial values (e.g. from the method of
195
+ moments) are recommended for the scale and shape parameters.
@@ -0,0 +1,9 @@
1
+ ppidest/__init__.py,sha256=uRXTqJGA6puAoEwFiTFmgto8uL6TbcC3-pSTrXi78PA,6680
2
+ ppidest/distributions.py,sha256=Z130id3V2h42rN8Jlqjrms21Ouealn003gu0bXESlGU,6857
3
+ ppidest/divergences.py,sha256=6Ho8GIkSo5a9dHLQGzhOqKgYcsoTaRobhWOtkBIcPzU,4780
4
+ ppidest/estimators.py,sha256=SLqoChKY70X3NduldcDdrqGvtPEbKxcG1DzSEKW40gE,13702
5
+ ppidest/plotting.py,sha256=Klwr5iAonoMrj7E_ucb6tUsaeeWWCC4jbcTq51h8gVs,3568
6
+ ppidest-0.1.0.dist-info/WHEEL,sha256=_d8F1e7SqtoW6CDj4Gi8lFC26a_7I17R7zPLCKTp4Fg,81
7
+ ppidest-0.1.0.dist-info/entry_points.txt,sha256=oDABWj4ndH6xJvBVI26yx7sZyBvOWi2OPSzdjMhctyM,42
8
+ ppidest-0.1.0.dist-info/METADATA,sha256=ddZiWjhbZzxcpIvBAJl44OjeMzAFBdtwpl2cVzIp1ws,6788
9
+ ppidest-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: uv 0.12.15
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ ppidest = ppidest:main
3
+