ppidest 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ppidest/__init__.py +199 -0
- ppidest/distributions.py +240 -0
- ppidest/divergences.py +185 -0
- ppidest/estimators.py +395 -0
- ppidest/plotting.py +106 -0
- ppidest-0.1.0.dist-info/METADATA +195 -0
- ppidest-0.1.0.dist-info/RECORD +9 -0
- ppidest-0.1.0.dist-info/WHEEL +4 -0
- ppidest-0.1.0.dist-info/entry_points.txt +3 -0
ppidest/__init__.py
ADDED
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""Plotting Position-Information Divergence (PPID) framework.
|
|
2
|
+
|
|
3
|
+
Parameter estimation for univariate distributions based on discretizing
|
|
4
|
+
the cumulative distribution function with plotting positions and
|
|
5
|
+
minimizing information divergences against the empirical gaps.
|
|
6
|
+
|
|
7
|
+
Public API
|
|
8
|
+
----------
|
|
9
|
+
- Distributions: :func:`normal_cdf`, :func:`normal_pdf`, :func:`gev_cdf`,
|
|
10
|
+
:func:`gev_pdf`, :func:`gev_quantile`, :func:`gev_return_level`,
|
|
11
|
+
:func:`weibull_cdf`, :func:`weibull_pdf`.
|
|
12
|
+
- Plotting positions: :func:`plotting_positions`.
|
|
13
|
+
- Divergences: :func:`kl_divergence`, :func:`kl_generalized_divergence`,
|
|
14
|
+
:func:`kl_symmetric_divergence`, :func:`jensen_shannon_divergence`,
|
|
15
|
+
:func:`beta_divergence`, :func:`power_divergence`,
|
|
16
|
+
:func:`renyi_divergence`.
|
|
17
|
+
- Estimators: :class:`ML`, :class:`PPID`
|
|
18
|
+
and the convenience wrapper :func:`find_ppid_min`.
|
|
19
|
+
|
|
20
|
+
The ``calc_*`` names from earlier releases are re-exported here as
|
|
21
|
+
backward-compatible aliases (e.g. :data:`calc_gev_cdf`).
|
|
22
|
+
|
|
23
|
+
Command line
|
|
24
|
+
------------
|
|
25
|
+
The ``ppidest`` console script fits a distribution to sample values given
|
|
26
|
+
on the command line::
|
|
27
|
+
|
|
28
|
+
ppidest 0.38 0.51 1.44 2.14
|
|
29
|
+
ppidest 0.38 0.51 1.44 2.14 --distribution normal
|
|
30
|
+
ppidest 0.38 0.51 1.44 2.14 --plotting-position Weibull \\
|
|
31
|
+
--information-divergence "Jensen-Shannon"
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
import argparse
|
|
35
|
+
|
|
36
|
+
import numpy as np
|
|
37
|
+
|
|
38
|
+
from .distributions import (
|
|
39
|
+
gev_cdf,
|
|
40
|
+
gev_pdf,
|
|
41
|
+
gev_quantile,
|
|
42
|
+
gev_return_level,
|
|
43
|
+
normal_cdf,
|
|
44
|
+
normal_pdf,
|
|
45
|
+
weibull_cdf,
|
|
46
|
+
weibull_pdf,
|
|
47
|
+
calc_gev_cdf,
|
|
48
|
+
calc_gev_pdf,
|
|
49
|
+
calc_gev_quantile,
|
|
50
|
+
calc_norm_cdf,
|
|
51
|
+
calc_norm_pdf,
|
|
52
|
+
calc_return_level_gev,
|
|
53
|
+
calc_weibull_cdf,
|
|
54
|
+
calc_weibull_pdf,
|
|
55
|
+
)
|
|
56
|
+
from .divergences import (
|
|
57
|
+
beta_divergence,
|
|
58
|
+
jensen_shannon_divergence,
|
|
59
|
+
kl_divergence,
|
|
60
|
+
kl_generalized_divergence,
|
|
61
|
+
kl_symmetric_divergence,
|
|
62
|
+
power_divergence,
|
|
63
|
+
renyi_divergence,
|
|
64
|
+
)
|
|
65
|
+
from .estimators import (
|
|
66
|
+
ML,
|
|
67
|
+
PPID,
|
|
68
|
+
DIVERGENCE_NAMES,
|
|
69
|
+
find_ppid_min,
|
|
70
|
+
)
|
|
71
|
+
from .plotting import (
|
|
72
|
+
PLOTTING_POSITIONS,
|
|
73
|
+
plotting_position_coefficients,
|
|
74
|
+
plotting_positions,
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
__version__ = "0.1.0"
|
|
78
|
+
|
|
79
|
+
__all__ = [
|
|
80
|
+
# distributions
|
|
81
|
+
"normal_cdf", "normal_pdf",
|
|
82
|
+
"gev_cdf", "gev_pdf", "gev_quantile", "gev_return_level",
|
|
83
|
+
"weibull_cdf", "weibull_pdf",
|
|
84
|
+
# plotting positions
|
|
85
|
+
"PLOTTING_POSITIONS", "plotting_position_coefficients",
|
|
86
|
+
"plotting_positions",
|
|
87
|
+
# divergences
|
|
88
|
+
"kl_divergence", "kl_generalized_divergence", "kl_symmetric_divergence",
|
|
89
|
+
"jensen_shannon_divergence", "beta_divergence", "power_divergence",
|
|
90
|
+
"renyi_divergence",
|
|
91
|
+
# estimators
|
|
92
|
+
"ML", "PPID", "find_ppid_min",
|
|
93
|
+
# legacy aliases
|
|
94
|
+
"calc_norm_cdf", "calc_norm_pdf",
|
|
95
|
+
"calc_gev_cdf", "calc_gev_pdf", "calc_gev_quantile",
|
|
96
|
+
"calc_return_level_gev", "calc_weibull_cdf", "calc_weibull_pdf",
|
|
97
|
+
]
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
#: Name of the parameter vector entry for each supported distribution.
|
|
101
|
+
_PARAMETER_NAMES = {
|
|
102
|
+
"gev": ("loc", "scale", "shape"),
|
|
103
|
+
"normal": ("loc", "scale"),
|
|
104
|
+
"weibull": ("threshold", "scale", "shape"),
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
#: CDF callable for each supported distribution.
|
|
108
|
+
_DISTRIBUTIONS = {
|
|
109
|
+
"gev": gev_cdf,
|
|
110
|
+
"normal": normal_cdf,
|
|
111
|
+
"weibull": weibull_cdf,
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _initial_parameters(name, xs):
|
|
116
|
+
"""Return a sensible parameter guess derived from the sample."""
|
|
117
|
+
loc = float(np.mean(xs))
|
|
118
|
+
scale = float(np.std(xs))
|
|
119
|
+
if name == "normal":
|
|
120
|
+
return [loc, scale]
|
|
121
|
+
if name == "gev":
|
|
122
|
+
return [loc, scale, 0.0]
|
|
123
|
+
# three-parameter Weibull: threshold below the minimum observation,
|
|
124
|
+
# scale on the order of the observed range.
|
|
125
|
+
mn, mx = float(xs.min()), float(xs.max())
|
|
126
|
+
threshold = mn - 0.1 * (mx - mn) - 1e-9
|
|
127
|
+
return [threshold, (mx - mn) + 1e-9, 2.0]
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def main(argv=None):
|
|
131
|
+
"""Console entry point: fit a distribution to command-line sample data.
|
|
132
|
+
|
|
133
|
+
The parameters are fitted with the :class:`PPID` estimator using the
|
|
134
|
+
plotting position and information divergence given by
|
|
135
|
+
``--plotting-position`` (default ``Hazen``) and
|
|
136
|
+
``--information-divergence`` (default ``Kullback-Leibler``).
|
|
137
|
+
"""
|
|
138
|
+
parser = argparse.ArgumentParser(
|
|
139
|
+
prog="ppidest",
|
|
140
|
+
description="Fit a univariate distribution by minimizing an "
|
|
141
|
+
"information divergence between plotting-position "
|
|
142
|
+
"empirical gaps and the candidate CDF gaps.")
|
|
143
|
+
parser.add_argument(
|
|
144
|
+
"xs", type=float, nargs="+", metavar="x",
|
|
145
|
+
help="sample values to fit")
|
|
146
|
+
parser.add_argument(
|
|
147
|
+
"-d", "--distribution", choices=sorted(_DISTRIBUTIONS),
|
|
148
|
+
default="gev",
|
|
149
|
+
help="distribution family to fit (default: gev)")
|
|
150
|
+
parser.add_argument(
|
|
151
|
+
"-p", "--plotting-position", choices=sorted(PLOTTING_POSITIONS),
|
|
152
|
+
default="Hazen",
|
|
153
|
+
help="plotting-position scheme (default: Hazen)")
|
|
154
|
+
parser.add_argument(
|
|
155
|
+
"-i", "--information-divergence", choices=sorted(DIVERGENCE_NAMES),
|
|
156
|
+
default="Kullback-Leibler",
|
|
157
|
+
help="information divergence to minimize (default: "
|
|
158
|
+
"Kullback-Leibler)")
|
|
159
|
+
parser.add_argument(
|
|
160
|
+
"--pdiv", type=float, default=0.5, metavar="VALUE",
|
|
161
|
+
help="order/index parameter for the beta, power or Renyi "
|
|
162
|
+
"divergences (default: 0.5)")
|
|
163
|
+
parser.add_argument(
|
|
164
|
+
"--initial", type=float, nargs="+", metavar="PAR",
|
|
165
|
+
help="initial parameters, overriding the data-driven defaults "
|
|
166
|
+
"(e.g. '--initial 0 1 0.25' for GEV)")
|
|
167
|
+
parser.add_argument(
|
|
168
|
+
"--version", action="version", version=f"%(prog)s {__version__}")
|
|
169
|
+
args = parser.parse_args(argv)
|
|
170
|
+
|
|
171
|
+
xs = np.asarray(args.xs)
|
|
172
|
+
if xs.size < 2:
|
|
173
|
+
parser.error("at least two sample values are required")
|
|
174
|
+
|
|
175
|
+
name = args.distribution
|
|
176
|
+
cdf = _DISTRIBUTIONS[name]
|
|
177
|
+
par_names = _PARAMETER_NAMES[name]
|
|
178
|
+
if args.initial is not None:
|
|
179
|
+
if len(args.initial) != len(par_names):
|
|
180
|
+
parser.error(
|
|
181
|
+
f"--initial must provide {len(par_names)} values for "
|
|
182
|
+
f"{name} ({', '.join(par_names)})")
|
|
183
|
+
par_0 = list(args.initial)
|
|
184
|
+
else:
|
|
185
|
+
par_0 = _initial_parameters(name, xs)
|
|
186
|
+
|
|
187
|
+
res = PPID(xs, cdf, plotposition=args.plotting_position,
|
|
188
|
+
divergence=args.information_divergence).find_min_div(
|
|
189
|
+
par_0, pdiv=args.pdiv)
|
|
190
|
+
|
|
191
|
+
header = ("distribution", "plotting position", "information divergence")
|
|
192
|
+
values = (name, args.plotting_position, args.information_divergence)
|
|
193
|
+
for key, value in zip(header, values):
|
|
194
|
+
print(f"{key:26s}: {value}")
|
|
195
|
+
for pname, pvalue in zip(par_names, res.x):
|
|
196
|
+
print(f"{pname:26s}: {pvalue:.6g}")
|
|
197
|
+
print(f"{'minimum divergence':26s}: {res.fun:.6g}")
|
|
198
|
+
print(f"{'converged':26s}: {res.success}")
|
|
199
|
+
return res
|
ppidest/distributions.py
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
"""Probability distributions used by the estimation methods.
|
|
2
|
+
|
|
3
|
+
Every distribution function in this module follows the same calling
|
|
4
|
+
convention: the first argument is the evaluation point (a scalar or a
|
|
5
|
+
NumPy array) and the second is the parameter vector ``par``. Keeping the
|
|
6
|
+
signature uniform lets estimators treat ``cdf`` and ``pdf`` callables as
|
|
7
|
+
first-class objects.
|
|
8
|
+
|
|
9
|
+
Parameters are supplied in plain tuples or lists so that they can be passed
|
|
10
|
+
straight to an optimizer (e.g. ``scipy.optimize.minimize``).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
from scipy.special import ndtr
|
|
15
|
+
|
|
16
|
+
SQRT_2PI = np.sqrt(2.0 * np.pi)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
################################################################################
|
|
20
|
+
# Normal distribution
|
|
21
|
+
################################################################################
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def normal_cdf(x, par):
|
|
25
|
+
"""Cumulative distribution function of the normal distribution.
|
|
26
|
+
|
|
27
|
+
Parameters
|
|
28
|
+
----------
|
|
29
|
+
x : float or ndarray
|
|
30
|
+
Evaluation point(s).
|
|
31
|
+
par : sequence of float
|
|
32
|
+
Parameters ``(loc, scale)``.
|
|
33
|
+
|
|
34
|
+
Returns
|
|
35
|
+
-------
|
|
36
|
+
float or ndarray
|
|
37
|
+
``Phi((x - loc) / scale)``.
|
|
38
|
+
"""
|
|
39
|
+
return ndtr((x - par[0]) / par[1])
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def normal_pdf(x, par):
|
|
43
|
+
"""Probability density function of the normal distribution.
|
|
44
|
+
|
|
45
|
+
Parameters
|
|
46
|
+
----------
|
|
47
|
+
x : float or ndarray
|
|
48
|
+
Evaluation point(s).
|
|
49
|
+
par : sequence of float
|
|
50
|
+
Parameters ``(loc, scale)``.
|
|
51
|
+
|
|
52
|
+
Returns
|
|
53
|
+
-------
|
|
54
|
+
float or ndarray
|
|
55
|
+
Density at each evaluation point.
|
|
56
|
+
"""
|
|
57
|
+
z = (x - par[0]) / par[1]
|
|
58
|
+
return np.exp(-0.5 * z ** 2) / (par[1] * SQRT_2PI)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
################################################################################
|
|
62
|
+
# Generalized extreme value (GEV) distribution
|
|
63
|
+
################################################################################
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def gev_cdf(x, par):
|
|
67
|
+
"""Cumulative distribution function of the GEV distribution.
|
|
68
|
+
|
|
69
|
+
The GEV distribution function is
|
|
70
|
+
|
|
71
|
+
.. math::
|
|
72
|
+
|
|
73
|
+
F(x) = \\exp\\left[-\\left(1 + \\xi \\frac{x - \\mu}{\\sigma}
|
|
74
|
+
\\right)^{-1/\\xi}\\right]
|
|
75
|
+
|
|
76
|
+
with the Gumbel limit :math:`F(x) = \\exp\\left[-\\exp(-y)\\right]`
|
|
77
|
+
taken when :math:`\\xi \\approx 0`.
|
|
78
|
+
|
|
79
|
+
Parameters
|
|
80
|
+
----------
|
|
81
|
+
x : float or ndarray
|
|
82
|
+
Evaluation point(s).
|
|
83
|
+
par : sequence of float
|
|
84
|
+
Parameters ``(location, scale, shape)`` :math:`(\\mu, \\sigma, \\xi)`.
|
|
85
|
+
The shape parameter follows the convention in which
|
|
86
|
+
:math:`\\xi > 0` gives the Fréchet (heavy tail) family.
|
|
87
|
+
|
|
88
|
+
Returns
|
|
89
|
+
-------
|
|
90
|
+
float or ndarray
|
|
91
|
+
``F(x)``.
|
|
92
|
+
"""
|
|
93
|
+
y = (x - par[0]) / par[1]
|
|
94
|
+
xi = par[2]
|
|
95
|
+
if abs(xi) < 1e-10:
|
|
96
|
+
return np.exp(-np.exp(-y))
|
|
97
|
+
# During optimization the parameters may leave the support of the
|
|
98
|
+
# distribution; such evaluations legitimately yield NaN.
|
|
99
|
+
with np.errstate(invalid="ignore", over="ignore", divide="ignore"):
|
|
100
|
+
return np.exp(-(1.0 + xi * y) ** (-1 / xi))
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def gev_pdf(x, par):
|
|
104
|
+
"""Probability density function of the GEV distribution.
|
|
105
|
+
|
|
106
|
+
Parameters
|
|
107
|
+
----------
|
|
108
|
+
x : float or ndarray
|
|
109
|
+
Evaluation point(s).
|
|
110
|
+
par : sequence of float
|
|
111
|
+
Parameters ``(location, scale, shape)``.
|
|
112
|
+
|
|
113
|
+
Returns
|
|
114
|
+
-------
|
|
115
|
+
float or ndarray
|
|
116
|
+
Density at each evaluation point.
|
|
117
|
+
"""
|
|
118
|
+
with np.errstate(invalid="ignore", over="ignore", divide="ignore"):
|
|
119
|
+
if abs(par[2]) < 1e-10:
|
|
120
|
+
t = np.exp(-(x - par[0]) / par[1])
|
|
121
|
+
return 1 / par[1] * t * np.exp(-t)
|
|
122
|
+
t = (1 + par[2] / par[1] * (x - par[0])) ** (-1 / par[2])
|
|
123
|
+
return 1 / par[1] * t ** (par[2] + 1) * np.exp(-t)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def gev_quantile(ff, par):
|
|
127
|
+
"""Quantile function (inverse CDF) of the GEV distribution.
|
|
128
|
+
|
|
129
|
+
Returns the value :math:`x` such that :math:`F(x) = ff`.
|
|
130
|
+
|
|
131
|
+
Parameters
|
|
132
|
+
----------
|
|
133
|
+
ff : float or ndarray
|
|
134
|
+
Cumulative probabilities in :math:`(0, 1)`.
|
|
135
|
+
par : sequence of float
|
|
136
|
+
Parameters ``(location, scale, shape)``.
|
|
137
|
+
|
|
138
|
+
Returns
|
|
139
|
+
-------
|
|
140
|
+
float or ndarray
|
|
141
|
+
Quantile(s).
|
|
142
|
+
"""
|
|
143
|
+
if abs(par[2]) < 1e-10:
|
|
144
|
+
return par[0] - par[1] * np.log(-np.log(ff))
|
|
145
|
+
return par[0] - par[1] / par[2] * (1 - (-np.log(ff)) ** (-par[2]))
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def gev_return_level(p, par):
|
|
149
|
+
"""Return level of the GEV distribution for a given return period.
|
|
150
|
+
|
|
151
|
+
The return level :math:`z_p` associated with the return period
|
|
152
|
+
:math:`p` (given as multiples of the block length) satisfies
|
|
153
|
+
:math:`F(z_p) = 1 - 1 / p`.
|
|
154
|
+
|
|
155
|
+
Parameters
|
|
156
|
+
----------
|
|
157
|
+
p : float
|
|
158
|
+
Return period, expressed as multiples of the block length.
|
|
159
|
+
par : sequence of float
|
|
160
|
+
Parameters ``(location, scale, shape)``.
|
|
161
|
+
|
|
162
|
+
Returns
|
|
163
|
+
-------
|
|
164
|
+
float
|
|
165
|
+
Return level.
|
|
166
|
+
"""
|
|
167
|
+
ff = 1 - 1 / p
|
|
168
|
+
return gev_quantile(ff, par)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
################################################################################
|
|
172
|
+
# Three-parameter Weibull distribution
|
|
173
|
+
################################################################################
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def weibull_cdf(x, par):
|
|
177
|
+
"""Cumulative distribution function of the three-parameter Weibull
|
|
178
|
+
distribution with threshold :math:`\\gamma`.
|
|
179
|
+
|
|
180
|
+
.. math::
|
|
181
|
+
|
|
182
|
+
F(x) = 1 - \\exp\\left[-\\left(\\frac{x - \\gamma}{\\lambda}
|
|
183
|
+
\\right)^k\\right]
|
|
184
|
+
|
|
185
|
+
Parameters
|
|
186
|
+
----------
|
|
187
|
+
x : float or ndarray
|
|
188
|
+
Evaluation point(s).
|
|
189
|
+
par : sequence of float
|
|
190
|
+
Parameters ``(threshold, scale, shape)``
|
|
191
|
+
:math:`(\\gamma, \\lambda, k)`.
|
|
192
|
+
|
|
193
|
+
Returns
|
|
194
|
+
-------
|
|
195
|
+
float or ndarray
|
|
196
|
+
``F(x)``.
|
|
197
|
+
"""
|
|
198
|
+
return 1 - np.exp(-((x - par[0]) / par[1]) ** par[2])
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def weibull_pdf(x, par):
|
|
202
|
+
"""Probability density function of the three-parameter Weibull
|
|
203
|
+
distribution.
|
|
204
|
+
|
|
205
|
+
Parameters
|
|
206
|
+
----------
|
|
207
|
+
x : float or ndarray
|
|
208
|
+
Evaluation point(s).
|
|
209
|
+
par : sequence of float
|
|
210
|
+
Parameters ``(threshold, scale, shape)``.
|
|
211
|
+
|
|
212
|
+
Returns
|
|
213
|
+
-------
|
|
214
|
+
float or ndarray
|
|
215
|
+
Density at each evaluation point.
|
|
216
|
+
"""
|
|
217
|
+
return par[2] / par[1] * ((x - par[0]) / par[1]) ** (par[2] - 1) \
|
|
218
|
+
* np.exp(-((x - par[0]) / par[1]) ** par[2])
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
################################################################################
|
|
222
|
+
# Legacy aliases
|
|
223
|
+
################################################################################
|
|
224
|
+
|
|
225
|
+
#: Backward-compatible alias for :func:`normal_cdf`.
|
|
226
|
+
calc_norm_cdf = normal_cdf
|
|
227
|
+
#: Backward-compatible alias for :func:`normal_pdf`.
|
|
228
|
+
calc_norm_pdf = normal_pdf
|
|
229
|
+
#: Backward-compatible alias for :func:`gev_cdf`.
|
|
230
|
+
calc_gev_cdf = gev_cdf
|
|
231
|
+
#: Backward-compatible alias for :func:`gev_pdf`.
|
|
232
|
+
calc_gev_pdf = gev_pdf
|
|
233
|
+
#: Backward-compatible alias for :func:`gev_quantile`.
|
|
234
|
+
calc_gev_quantile = gev_quantile
|
|
235
|
+
#: Backward-compatible alias for :func:`gev_return_level`.
|
|
236
|
+
calc_return_level_gev = gev_return_level
|
|
237
|
+
#: Backward-compatible alias for :func:`weibull_cdf`.
|
|
238
|
+
calc_weibull_cdf = weibull_cdf
|
|
239
|
+
#: Backward-compatible alias for :func:`weibull_pdf`.
|
|
240
|
+
calc_weibull_pdf = weibull_pdf
|
ppidest/divergences.py
ADDED
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""Information divergences between discrete probability vectors.
|
|
2
|
+
|
|
3
|
+
In the PPID framework a candidate parametric distribution is compared
|
|
4
|
+
with the empirical discretization of the data: ``das`` holds the empirical
|
|
5
|
+
gap probabilities derived from plotting positions and ``ddas`` holds the
|
|
6
|
+
corresponding model gaps :math:`F(x_j) - F(x_{j-1})` from the candidate
|
|
7
|
+
CDF. Each function below quantifies the agreement of the two probability
|
|
8
|
+
vectors.
|
|
9
|
+
|
|
10
|
+
All functions accept the two probability vectors as 1-D NumPy arrays of
|
|
11
|
+
equal length and return a non-negative scalar divergence.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def kl_divergence(das, ddas):
|
|
19
|
+
"""Kullback-Leibler divergence ``sum(das * log(das / ddas))``.
|
|
20
|
+
|
|
21
|
+
Parameters
|
|
22
|
+
----------
|
|
23
|
+
das : ndarray
|
|
24
|
+
Empirical gap probabilities.
|
|
25
|
+
ddas : ndarray
|
|
26
|
+
Model gap probabilities.
|
|
27
|
+
|
|
28
|
+
Returns
|
|
29
|
+
-------
|
|
30
|
+
float
|
|
31
|
+
Kullback-Leibler divergence.
|
|
32
|
+
"""
|
|
33
|
+
return np.sum(das * np.log(das / ddas))
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def kl_generalized_divergence(das, ddas):
|
|
37
|
+
"""Generalized Kullback-Leibler divergence
|
|
38
|
+
``sum(das * log(das / ddas) - das + ddas)``.
|
|
39
|
+
|
|
40
|
+
Unlike the plain KL divergence this variant is symmetric in its
|
|
41
|
+
treatment of the two vectors (it vanishes when they coincide even
|
|
42
|
+
when the vectors are not normalized).
|
|
43
|
+
|
|
44
|
+
Parameters
|
|
45
|
+
----------
|
|
46
|
+
das : ndarray
|
|
47
|
+
Empirical gap probabilities.
|
|
48
|
+
ddas : ndarray
|
|
49
|
+
Model gap probabilities.
|
|
50
|
+
|
|
51
|
+
Returns
|
|
52
|
+
-------
|
|
53
|
+
float
|
|
54
|
+
Generalized Kullback-Leibler divergence.
|
|
55
|
+
"""
|
|
56
|
+
return np.sum(das * np.log(das / ddas) - das + ddas)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def kl_symmetric_divergence(das, ddas):
|
|
60
|
+
"""Symmetric Kullback-Leibler divergence
|
|
61
|
+
``sum((ddas - das) * log(ddas / das))``.
|
|
62
|
+
|
|
63
|
+
Parameters
|
|
64
|
+
----------
|
|
65
|
+
das : ndarray
|
|
66
|
+
Empirical gap probabilities.
|
|
67
|
+
ddas : ndarray
|
|
68
|
+
Model gap probabilities.
|
|
69
|
+
|
|
70
|
+
Returns
|
|
71
|
+
-------
|
|
72
|
+
float
|
|
73
|
+
Symmetric Kullback-Leibler divergence.
|
|
74
|
+
"""
|
|
75
|
+
return np.sum((ddas - das) * np.log(ddas / das))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def jensen_shannon_divergence(das, ddas):
|
|
79
|
+
"""Jensen-Shannon divergence based on the midpoint mixture
|
|
80
|
+
``mas = 0.5 * (das + ddas)``.
|
|
81
|
+
|
|
82
|
+
Parameters
|
|
83
|
+
----------
|
|
84
|
+
das : ndarray
|
|
85
|
+
Empirical gap probabilities.
|
|
86
|
+
ddas : ndarray
|
|
87
|
+
Model gap probabilities.
|
|
88
|
+
|
|
89
|
+
Returns
|
|
90
|
+
-------
|
|
91
|
+
float
|
|
92
|
+
Jensen-Shannon divergence.
|
|
93
|
+
"""
|
|
94
|
+
mas = 0.5 * (das + ddas)
|
|
95
|
+
return 0.5 * np.sum(
|
|
96
|
+
das * np.log(das / mas) + ddas * np.log(ddas / mas))
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def beta_divergence(das, ddas, beta):
|
|
100
|
+
"""Beta (Cichocki) divergence of order ``beta`` (with ``beta != 0, 1``).
|
|
101
|
+
|
|
102
|
+
.. math::
|
|
103
|
+
|
|
104
|
+
D_\\beta(d \\,||\\, \\hat d)
|
|
105
|
+
= \\sum_j \\left[
|
|
106
|
+
\\frac{d_j (d_j^{\\beta - 1} - \\hat d_j^{\\beta - 1})}
|
|
107
|
+
{\\beta - 1}
|
|
108
|
+
- \\frac{d_j^{\\beta} - \\hat d_j^{\\beta}}{\\beta}
|
|
109
|
+
\\right]
|
|
110
|
+
|
|
111
|
+
Parameters
|
|
112
|
+
----------
|
|
113
|
+
das : ndarray
|
|
114
|
+
Empirical gap probabilities.
|
|
115
|
+
ddas : ndarray
|
|
116
|
+
Model gap probabilities.
|
|
117
|
+
beta : float
|
|
118
|
+
Order parameter of the divergence. The special cases
|
|
119
|
+
``beta -> 1`` (Kullback-Leibler) and ``beta -> 2`` (Euclidean)
|
|
120
|
+
are approached only in the limit.
|
|
121
|
+
|
|
122
|
+
Returns
|
|
123
|
+
-------
|
|
124
|
+
float
|
|
125
|
+
Beta divergence.
|
|
126
|
+
"""
|
|
127
|
+
aa = das * (das ** (beta - 1) - ddas ** (beta - 1)) / (beta - 1)
|
|
128
|
+
bb = (das ** beta - ddas ** beta) / beta
|
|
129
|
+
# NaN from out-of-support model gaps is expected while optimizing.
|
|
130
|
+
with np.errstate(invalid="ignore"):
|
|
131
|
+
return aa.sum() - bb.sum()
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def power_divergence(das, ddas, lmbd):
|
|
135
|
+
"""Power divergence of index ``lmbd``.
|
|
136
|
+
|
|
137
|
+
.. math::
|
|
138
|
+
|
|
139
|
+
D_\\lambda(d \\,||\\, \\hat d)
|
|
140
|
+
= \\frac{1}{\\lambda (\\lambda + 1)}
|
|
141
|
+
\\sum_j d_j \\left[
|
|
142
|
+
\\left(\\frac{d_j}{\\hat d_j}\\right)^{\\lambda} - 1
|
|
143
|
+
\\right]
|
|
144
|
+
|
|
145
|
+
Parameters
|
|
146
|
+
----------
|
|
147
|
+
das : ndarray
|
|
148
|
+
Empirical gap probabilities.
|
|
149
|
+
ddas : ndarray
|
|
150
|
+
Model gap probabilities.
|
|
151
|
+
lmbd : float
|
|
152
|
+
Index of the divergence (``lmbd != 0, -1``).
|
|
153
|
+
|
|
154
|
+
Returns
|
|
155
|
+
-------
|
|
156
|
+
float
|
|
157
|
+
Power divergence.
|
|
158
|
+
"""
|
|
159
|
+
return 1 / (lmbd * (lmbd + 1)) * np.sum(das * ((das / ddas) ** lmbd - 1))
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def renyi_divergence(das, ddas, alpha):
|
|
163
|
+
"""Rényi divergence of order ``alpha``.
|
|
164
|
+
|
|
165
|
+
.. math::
|
|
166
|
+
|
|
167
|
+
D_\\alpha(d \\,||\\, \\hat d)
|
|
168
|
+
= \\frac{1}{\\alpha - 1}
|
|
169
|
+
\\log \\sum_j d_j^{\\alpha} \\hat d_j^{1 - \\alpha}
|
|
170
|
+
|
|
171
|
+
Parameters
|
|
172
|
+
----------
|
|
173
|
+
das : ndarray
|
|
174
|
+
Empirical gap probabilities.
|
|
175
|
+
ddas : ndarray
|
|
176
|
+
Model gap probabilities.
|
|
177
|
+
alpha : float
|
|
178
|
+
Order of the divergence (``alpha != 1``).
|
|
179
|
+
|
|
180
|
+
Returns
|
|
181
|
+
-------
|
|
182
|
+
float
|
|
183
|
+
Rényi divergence.
|
|
184
|
+
"""
|
|
185
|
+
return 1 / (alpha - 1) * np.log(np.sum(das ** alpha * ddas ** (1 - alpha)))
|
ppidest/estimators.py
ADDED
|
@@ -0,0 +1,395 @@
|
|
|
1
|
+
"""Parameter estimators based on CDF discretization.
|
|
2
|
+
|
|
3
|
+
This module collects the estimators of the Plotting Position-Information
|
|
4
|
+
Divergence (PPID) framework:
|
|
5
|
+
|
|
6
|
+
- :class:`ML` -- maximum likelihood (uses the PDF directly);
|
|
7
|
+
- :class:`PPID` -- PPID estimators minimizing one of several
|
|
8
|
+
information divergences between the empirical gaps (plotting positions)
|
|
9
|
+
and the model gaps;
|
|
10
|
+
|
|
11
|
+
All estimators expose a ``find_*`` method that wraps
|
|
12
|
+
``scipy.optimize.minimize`` and mirror its keyword arguments.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
import scipy.optimize
|
|
17
|
+
|
|
18
|
+
from .divergences import (
|
|
19
|
+
beta_divergence,
|
|
20
|
+
jensen_shannon_divergence,
|
|
21
|
+
kl_divergence,
|
|
22
|
+
kl_generalized_divergence,
|
|
23
|
+
kl_symmetric_divergence,
|
|
24
|
+
power_divergence,
|
|
25
|
+
renyi_divergence,
|
|
26
|
+
)
|
|
27
|
+
from .plotting import (
|
|
28
|
+
_plotting_positions_from_order_stats,
|
|
29
|
+
plotting_position_coefficients,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
#: Divergences that require an extra scalar parameter (``beta``, ``power``,
|
|
34
|
+
#: ``Renyi``): mapping divergence name to the ``PPID`` objective method.
|
|
35
|
+
_PARAMETRIC_DIVERGENCES = {
|
|
36
|
+
"beta": "beta_div",
|
|
37
|
+
"power": "power_div",
|
|
38
|
+
"Renyi": "renyi_div",
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
#: Divergences used directly on the gap vectors: mapping divergence name
|
|
42
|
+
#: to the ``PPID`` objective method.
|
|
43
|
+
_NONPARAMETRIC_DIVERGENCES = {
|
|
44
|
+
"Kullback-Leibler": "kld",
|
|
45
|
+
"KL_generalized": "kld_generalized",
|
|
46
|
+
"KL_symmetric": "kld_symmetric",
|
|
47
|
+
"Jensen-Shannon": "jensen_shannon_div",
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
#: Supported information divergences for :class:`PPID`.
|
|
51
|
+
DIVERGENCE_NAMES = tuple(
|
|
52
|
+
_NONPARAMETRIC_DIVERGENCES) + tuple(_PARAMETRIC_DIVERGENCES)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
################################################################################
|
|
56
|
+
# Shared helpers
|
|
57
|
+
################################################################################
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _minimize(fun, x0, method="Nelder-Mead", jac=None, hess=None, hessp=None,
|
|
61
|
+
bounds=None, constraints=(), tol=None, callback=None,
|
|
62
|
+
options=None, args=()):
|
|
63
|
+
"""Thin wrapper around ``scipy.optimize.minimize`` sharing the estimator
|
|
64
|
+
keyword interface."""
|
|
65
|
+
x0 = np.asarray(x0, dtype=float).ravel()
|
|
66
|
+
return scipy.optimize.minimize(
|
|
67
|
+
fun, x0, args=args, method=method, jac=jac, hess=hess, hessp=hessp,
|
|
68
|
+
bounds=bounds, constraints=constraints, tol=tol,
|
|
69
|
+
callback=callback, options=options)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _cdf_gaps(cdf, uniques, par, *args):
|
|
73
|
+
"""Return the model gap probabilities on the augmented support
|
|
74
|
+
``[0, uniques, 1]``. Extra ``args`` are forwarded to ``cdf``."""
|
|
75
|
+
ffs = np.concatenate([np.array([0.0]), cdf(uniques, par, *args),
|
|
76
|
+
np.array([1.0])])
|
|
77
|
+
return np.diff(ffs)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
################################################################################
|
|
81
|
+
# Maximum likelihood
|
|
82
|
+
################################################################################
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class ML:
|
|
86
|
+
"""Maximum-likelihood estimator based on a parametric PDF."""
|
|
87
|
+
|
|
88
|
+
def __init__(self, xs, pdf):
|
|
89
|
+
"""Initialize the estimator.
|
|
90
|
+
|
|
91
|
+
Parameters
|
|
92
|
+
----------
|
|
93
|
+
xs : array_like
|
|
94
|
+
Observed values.
|
|
95
|
+
pdf : callable
|
|
96
|
+
``pdf(x, par)`` returning the density at ``x`` for the
|
|
97
|
+
parameter vector ``par``.
|
|
98
|
+
"""
|
|
99
|
+
self.xs = np.sort(np.asarray(xs))
|
|
100
|
+
self.pdf = pdf
|
|
101
|
+
|
|
102
|
+
def nll(self, par):
|
|
103
|
+
"""Negative log-likelihood at ``par``.
|
|
104
|
+
|
|
105
|
+
Parameters
|
|
106
|
+
----------
|
|
107
|
+
par : sequence of float
|
|
108
|
+
Candidate parameters.
|
|
109
|
+
|
|
110
|
+
Returns
|
|
111
|
+
-------
|
|
112
|
+
float
|
|
113
|
+
``- sum(log(pdf(x_i, par) + 1e-10))``.
|
|
114
|
+
"""
|
|
115
|
+
return -np.sum(np.log(self.pdf(self.xs, par) + 1e-10))
|
|
116
|
+
|
|
117
|
+
def loglikelihood(self, par):
|
|
118
|
+
"""Elementwise log-density of each observation at ``par``.
|
|
119
|
+
|
|
120
|
+
Parameters
|
|
121
|
+
----------
|
|
122
|
+
par : sequence of float
|
|
123
|
+
Candidate parameters.
|
|
124
|
+
|
|
125
|
+
Returns
|
|
126
|
+
-------
|
|
127
|
+
ndarray
|
|
128
|
+
Log-density evaluated at every sample point.
|
|
129
|
+
"""
|
|
130
|
+
return np.log(self.pdf(self.xs, par))
|
|
131
|
+
|
|
132
|
+
def find_mle(self, par_0, method="Nelder-Mead", jac=None, hess=None,
|
|
133
|
+
hessp=None, bounds=None, constraints=(), tol=None,
|
|
134
|
+
callback=None, options=None):
|
|
135
|
+
"""Locate the maximum-likelihood estimate.
|
|
136
|
+
|
|
137
|
+
Parameters
|
|
138
|
+
----------
|
|
139
|
+
par_0 : sequence of float
|
|
140
|
+
Initial parameter guess.
|
|
141
|
+
method, jac, hess, hessp, bounds, constraints, tol, callback, options
|
|
142
|
+
Forwarded to :func:`scipy.optimize.minimize`.
|
|
143
|
+
|
|
144
|
+
Returns
|
|
145
|
+
-------
|
|
146
|
+
OptimizeResult
|
|
147
|
+
Result of the minimization.
|
|
148
|
+
"""
|
|
149
|
+
return _minimize(
|
|
150
|
+
self.nll, par_0, method=method, jac=jac, hess=hess, hessp=hessp,
|
|
151
|
+
bounds=bounds, constraints=constraints, tol=tol,
|
|
152
|
+
callback=callback, options=options)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
################################################################################
|
|
156
|
+
# PPID estimators
|
|
157
|
+
################################################################################
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
class PPID:
|
|
161
|
+
"""Plotting Position-Information Divergence estimator.
|
|
162
|
+
|
|
163
|
+
The empirical cumulative distribution is discretized by plotting
|
|
164
|
+
positions (see :mod:`ppidest.plotting`) and a parametric CDF is fitted
|
|
165
|
+
by minimizing an information divergence between the empirical gaps and
|
|
166
|
+
the model gaps.
|
|
167
|
+
"""
|
|
168
|
+
|
|
169
|
+
#: Parametric divergences (name to objective method).
|
|
170
|
+
_PARAMETRIC_DIVERGENCES = _PARAMETRIC_DIVERGENCES
|
|
171
|
+
#: Non-parametric divergences (name to objective method).
|
|
172
|
+
_NONPARAMETRIC_DIVERGENCES = _NONPARAMETRIC_DIVERGENCES
|
|
173
|
+
|
|
174
|
+
def __init__(self, xs, cdf, plotposition="Hazen",
|
|
175
|
+
divergence="Kullback-Leibler"):
|
|
176
|
+
"""Initialize the estimator.
|
|
177
|
+
|
|
178
|
+
Parameters
|
|
179
|
+
----------
|
|
180
|
+
xs : array_like
|
|
181
|
+
Observed values.
|
|
182
|
+
cdf : callable
|
|
183
|
+
``cdf(x, par)`` returning the CDF at ``x`` for the parameter
|
|
184
|
+
vector ``par``.
|
|
185
|
+
plotposition : str, optional
|
|
186
|
+
Plotting-position scheme; see :mod:`ppidest.plotting`.
|
|
187
|
+
Defaults to ``"Hazen"``.
|
|
188
|
+
divergence : str, optional
|
|
189
|
+
Information divergence minimized by :meth:`find_min_div`.
|
|
190
|
+
Defaults to ``"Kullback-Leibler"``.
|
|
191
|
+
"""
|
|
192
|
+
self.xs = np.sort(np.asarray(xs))
|
|
193
|
+
self.cdf = cdf
|
|
194
|
+
self.uniques, self.idxs, self.counts = np.unique(
|
|
195
|
+
self.xs, return_index=True, return_counts=True)
|
|
196
|
+
self.n_xs = self.xs.size
|
|
197
|
+
self.n_uniques = self.counts.size
|
|
198
|
+
self.plotposition = plotposition
|
|
199
|
+
self.divergence = divergence
|
|
200
|
+
|
|
201
|
+
@property
|
|
202
|
+
def divergence(self):
|
|
203
|
+
"""Information divergence minimized by :meth:`find_min_div`.
|
|
204
|
+
|
|
205
|
+
Assigning a new divergence validates the name against the
|
|
206
|
+
supported set.
|
|
207
|
+
"""
|
|
208
|
+
return self._divergence
|
|
209
|
+
|
|
210
|
+
@divergence.setter
|
|
211
|
+
def divergence(self, name):
|
|
212
|
+
self._validate_divergence(name)
|
|
213
|
+
self._divergence = name
|
|
214
|
+
|
|
215
|
+
def _validate_divergence(self, name):
|
|
216
|
+
if (name not in self._NONPARAMETRIC_DIVERGENCES
|
|
217
|
+
and name not in self._PARAMETRIC_DIVERGENCES):
|
|
218
|
+
raise ValueError(
|
|
219
|
+
f"Unknown divergence {name!r}. "
|
|
220
|
+
f"Choose from: {sorted(DIVERGENCE_NAMES)}")
|
|
221
|
+
|
|
222
|
+
def set_divergence(self, divergence):
|
|
223
|
+
"""Switch the divergence minimized by :meth:`find_min_div`."""
|
|
224
|
+
self.divergence = divergence
|
|
225
|
+
|
|
226
|
+
@property
|
|
227
|
+
def plotposition(self):
|
|
228
|
+
"""Plotting-position scheme in use.
|
|
229
|
+
|
|
230
|
+
Assigning a new scheme recomputes :attr:`p_ast_s` and :attr:`das`
|
|
231
|
+
from the stored sample.
|
|
232
|
+
"""
|
|
233
|
+
return self._plotposition
|
|
234
|
+
|
|
235
|
+
@plotposition.setter
|
|
236
|
+
def plotposition(self, name):
|
|
237
|
+
plotting_position_coefficients(name)
|
|
238
|
+
self._plotposition = name
|
|
239
|
+
self.p_ast_s, self.das = _plotting_positions_from_order_stats(
|
|
240
|
+
self.uniques, self.idxs, self.counts, self.n_xs, name)
|
|
241
|
+
|
|
242
|
+
@property
|
|
243
|
+
def plottingposition(self):
|
|
244
|
+
"""Alias of :attr:`plotposition`."""
|
|
245
|
+
return self.plotposition
|
|
246
|
+
|
|
247
|
+
@plottingposition.setter
|
|
248
|
+
def plottingposition(self, name):
|
|
249
|
+
self.plotposition = name
|
|
250
|
+
|
|
251
|
+
def set_plotting_positions(self, plotposition):
|
|
252
|
+
"""Switch the plotting-position scheme, recomputing the empirical
|
|
253
|
+
gaps from the stored sample."""
|
|
254
|
+
self.plotposition = plotposition
|
|
255
|
+
|
|
256
|
+
def get_plotting_positions(self):
|
|
257
|
+
"""Return the empirical gaps of the current plotting-position scheme.
|
|
258
|
+
|
|
259
|
+
Returns
|
|
260
|
+
-------
|
|
261
|
+
p_ast_s : ndarray
|
|
262
|
+
Cumulative plotting probabilities (with boundary points).
|
|
263
|
+
d_ast_s : ndarray
|
|
264
|
+
Empirical gap probabilities.
|
|
265
|
+
"""
|
|
266
|
+
return self.p_ast_s, self.das
|
|
267
|
+
|
|
268
|
+
def calc_dd_ast_s(self, par, *args):
|
|
269
|
+
"""Model gap probabilities for the candidate parameters.
|
|
270
|
+
|
|
271
|
+
Parameters
|
|
272
|
+
----------
|
|
273
|
+
par : sequence of float
|
|
274
|
+
Candidate parameters.
|
|
275
|
+
args : tuple, optional
|
|
276
|
+
Extra positional arguments forwarded to the CDF callable.
|
|
277
|
+
|
|
278
|
+
Returns
|
|
279
|
+
-------
|
|
280
|
+
ndarray
|
|
281
|
+
Gaps of ``[0, cdf(uniques, par, *args), 1]``.
|
|
282
|
+
"""
|
|
283
|
+
return _cdf_gaps(self.cdf, self.uniques, par, *args)
|
|
284
|
+
|
|
285
|
+
def beta_div(self, par, beta, *args):
|
|
286
|
+
"""Beta divergence between the empirical and model gaps."""
|
|
287
|
+
return beta_divergence(self.das, self.calc_dd_ast_s(par, *args), beta)
|
|
288
|
+
|
|
289
|
+
def power_div(self, par, lmbd, *args):
|
|
290
|
+
"""Power divergence between the empirical and model gaps."""
|
|
291
|
+
return power_divergence(self.das, self.calc_dd_ast_s(par, *args), lmbd)
|
|
292
|
+
|
|
293
|
+
def renyi_div(self, par, alpha, *args):
|
|
294
|
+
"""Rényi divergence between the empirical and model gaps."""
|
|
295
|
+
return renyi_divergence(self.das, self.calc_dd_ast_s(par, *args), alpha)
|
|
296
|
+
|
|
297
|
+
def kld(self, par, *args):
|
|
298
|
+
"""Kullback-Leibler divergence between the empirical and model gaps."""
|
|
299
|
+
return kl_divergence(self.das, self.calc_dd_ast_s(par, *args))
|
|
300
|
+
|
|
301
|
+
def kld_generalized(self, par, *args):
|
|
302
|
+
"""Generalized KL divergence between the empirical and model gaps."""
|
|
303
|
+
return kl_generalized_divergence(self.das, self.calc_dd_ast_s(par, *args))
|
|
304
|
+
|
|
305
|
+
def kld_symmetric(self, par, *args):
|
|
306
|
+
"""Symmetric KL divergence between the empirical and model gaps."""
|
|
307
|
+
return kl_symmetric_divergence(self.das, self.calc_dd_ast_s(par, *args))
|
|
308
|
+
|
|
309
|
+
def jensen_shannon_div(self, par, *args):
|
|
310
|
+
"""Jensen-Shannon divergence between the empirical and model gaps."""
|
|
311
|
+
return jensen_shannon_divergence(self.das, self.calc_dd_ast_s(par, *args))
|
|
312
|
+
|
|
313
|
+
def find_min_div(self, par, divergence=None, pdiv=0.5, args=(),
|
|
314
|
+
method="Nelder-Mead", jac=None, hess=None, hessp=None,
|
|
315
|
+
bounds=None, constraints=(), tol=None, callback=None,
|
|
316
|
+
options=None):
|
|
317
|
+
"""Minimize the chosen information divergence.
|
|
318
|
+
|
|
319
|
+
Parameters
|
|
320
|
+
----------
|
|
321
|
+
par : sequence of float
|
|
322
|
+
Initial parameter guess.
|
|
323
|
+
divergence : str, optional
|
|
324
|
+
One of ``"Kullback-Leibler"``, ``"KL_generalized"``,
|
|
325
|
+
``"KL_symmetric"``, ``"Jensen-Shannon"``, ``"beta"``,
|
|
326
|
+
``"power"``, ``"Renyi"``. Defaults to the estimator's
|
|
327
|
+
``divergence`` (set at construction or via
|
|
328
|
+
:meth:`set_divergence`).
|
|
329
|
+
pdiv : float, optional
|
|
330
|
+
Order/index parameter used by the parametric divergences
|
|
331
|
+
(``beta``, ``power``, ``Renyi``). Defaults to ``0.5``.
|
|
332
|
+
args : tuple, optional
|
|
333
|
+
Extra positional arguments forwarded to the CDF callable
|
|
334
|
+
(appended after ``pdiv`` for the parametric divergences).
|
|
335
|
+
method, jac, hess, hessp, bounds, constraints, tol, callback, options
|
|
336
|
+
Forwarded to :func:`scipy.optimize.minimize`.
|
|
337
|
+
|
|
338
|
+
Returns
|
|
339
|
+
-------
|
|
340
|
+
OptimizeResult
|
|
341
|
+
Result of the minimization.
|
|
342
|
+
|
|
343
|
+
Raises
|
|
344
|
+
------
|
|
345
|
+
ValueError
|
|
346
|
+
If ``divergence`` is not supported.
|
|
347
|
+
"""
|
|
348
|
+
if divergence is None:
|
|
349
|
+
divergence = self.divergence
|
|
350
|
+
if divergence in self._PARAMETRIC_DIVERGENCES:
|
|
351
|
+
func = getattr(self, self._PARAMETRIC_DIVERGENCES[divergence])
|
|
352
|
+
opt_args = (pdiv,) + tuple(args)
|
|
353
|
+
elif divergence in self._NONPARAMETRIC_DIVERGENCES:
|
|
354
|
+
func = getattr(
|
|
355
|
+
self, self._NONPARAMETRIC_DIVERGENCES[divergence])
|
|
356
|
+
opt_args = tuple(args)
|
|
357
|
+
else:
|
|
358
|
+
raise ValueError(
|
|
359
|
+
f"Unknown divergence {divergence!r}. "
|
|
360
|
+
f"Choose from: {sorted(DIVERGENCE_NAMES)}")
|
|
361
|
+
return _minimize(
|
|
362
|
+
func, par, args=opt_args, method=method, jac=jac, hess=hess,
|
|
363
|
+
hessp=hessp, bounds=bounds, constraints=constraints, tol=tol,
|
|
364
|
+
callback=callback, options=options)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def find_ppid_min(xs, cdf, par, plotposition="Hazen",
|
|
368
|
+
divergence="Kullback-Leibler"):
|
|
369
|
+
"""Convenience wrapper around :class:`PPID`.
|
|
370
|
+
|
|
371
|
+
Builds a :class:`PPID` estimator for ``(xs, cdf)``, fits it with the
|
|
372
|
+
requested divergence and returns the raw minimization result.
|
|
373
|
+
|
|
374
|
+
Parameters
|
|
375
|
+
----------
|
|
376
|
+
xs : array_like
|
|
377
|
+
Observed values.
|
|
378
|
+
cdf : callable
|
|
379
|
+
``cdf(x, par)`` returning the CDF at ``x`` for the parameter
|
|
380
|
+
vector ``par``.
|
|
381
|
+
par : sequence of float
|
|
382
|
+
Initial parameter guess.
|
|
383
|
+
plotposition : str, optional
|
|
384
|
+
Plotting-position scheme. Defaults to ``"Hazen"``.
|
|
385
|
+
divergence : str, optional
|
|
386
|
+
Information divergence to minimize. Defaults to
|
|
387
|
+
``"Kullback-Leibler"``.
|
|
388
|
+
|
|
389
|
+
Returns
|
|
390
|
+
-------
|
|
391
|
+
OptimizeResult
|
|
392
|
+
Result of the minimization.
|
|
393
|
+
"""
|
|
394
|
+
inst = PPID(xs, cdf, plotposition=plotposition, divergence=divergence)
|
|
395
|
+
return inst.find_min_div(par)
|
ppidest/plotting.py
ADDED
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""Plotting positions used to discretize the empirical CDF.
|
|
2
|
+
|
|
3
|
+
Plotting positions assign a plotting probability :math:`p_{(i)}` to the
|
|
4
|
+
i-th order statistic. The PPID framework converts these into a discrete
|
|
5
|
+
probability vector (the "empirical gaps") by aggregating the positions
|
|
6
|
+
shared by tied observations and augmenting the support with the boundary
|
|
7
|
+
bin at :math:`0` and :math:`1`.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
|
|
12
|
+
#: Registry of plotting-position coefficients ``(alpha, beta)`` such that
|
|
13
|
+
#: ``p_i = (i - alpha) / (n + 1 - alpha - beta)``.
|
|
14
|
+
PLOTTING_POSITIONS = {
|
|
15
|
+
"Weibull": (0.0, 0.0), # i / (n + 1); unbiased for the uniform CDF
|
|
16
|
+
"median": (0.3, 0.3), # (i - 0.3) / (n + 0.4); estimates the median
|
|
17
|
+
"Gringorten": (0.44, 0.44), # (i - 0.44) / (n + 0.12); optimal for Gumbel
|
|
18
|
+
"Hazen": (0.5, 0.5), # (i - 0.5) / n; piecewise linear approximation
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def plotting_position_coefficients(name):
|
|
23
|
+
"""Return the ``(alpha, beta)`` coefficients of a named plotting position.
|
|
24
|
+
|
|
25
|
+
Parameters
|
|
26
|
+
----------
|
|
27
|
+
name : str
|
|
28
|
+
One of ``"Weibull"``, ``"median"``, ``"Gringorten"``, ``"Hazen"``.
|
|
29
|
+
|
|
30
|
+
Returns
|
|
31
|
+
-------
|
|
32
|
+
tuple of float
|
|
33
|
+
``(alpha, beta)`` coefficients.
|
|
34
|
+
|
|
35
|
+
Raises
|
|
36
|
+
------
|
|
37
|
+
ValueError
|
|
38
|
+
If ``name`` is not a registered plotting position.
|
|
39
|
+
"""
|
|
40
|
+
if name not in PLOTTING_POSITIONS:
|
|
41
|
+
raise ValueError(
|
|
42
|
+
f"Unknown plotting position {name!r}. "
|
|
43
|
+
f"Choose from: {list(PLOTTING_POSITIONS)}"
|
|
44
|
+
)
|
|
45
|
+
return PLOTTING_POSITIONS[name]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def plotting_positions(xs, name="Hazen"):
|
|
49
|
+
"""Discretize a sample into empirical gap probabilities.
|
|
50
|
+
|
|
51
|
+
Each gap :math:`d^*_j` is the aggregated plotting probability of a
|
|
52
|
+
unique value, so the cumulative vector :math:`p^*_s` has mass
|
|
53
|
+
``[0, p1, ..., pk, 1]`` and the gaps sum to one.
|
|
54
|
+
|
|
55
|
+
Parameters
|
|
56
|
+
----------
|
|
57
|
+
xs : array_like
|
|
58
|
+
Observed values (not necessarily sorted).
|
|
59
|
+
name : str, optional
|
|
60
|
+
Plotting-position scheme. Defaults to ``"Hazen"``.
|
|
61
|
+
|
|
62
|
+
Returns
|
|
63
|
+
-------
|
|
64
|
+
p_ast_s : ndarray of shape (k + 2,)
|
|
65
|
+
Cumulative plotting probabilities including the boundary points
|
|
66
|
+
``0`` and ``1``.
|
|
67
|
+
d_ast_s : ndarray of shape (k + 1,)
|
|
68
|
+
Empirical gap probabilities, ``diff(p_ast_s)``.
|
|
69
|
+
"""
|
|
70
|
+
xs = np.sort(np.asarray(xs))
|
|
71
|
+
uniques, idxs, counts = np.unique(xs, return_index=True, return_counts=True)
|
|
72
|
+
return _plotting_positions_from_order_stats(
|
|
73
|
+
uniques, idxs, counts, xs.size, name)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _plotting_positions_from_order_stats(uniques, idxs, counts, n_xs, name):
|
|
77
|
+
"""Aggregate plotting positions from already-computed order statistics.
|
|
78
|
+
|
|
79
|
+
Parameters
|
|
80
|
+
----------
|
|
81
|
+
uniques : ndarray
|
|
82
|
+
Sorted unique values.
|
|
83
|
+
idxs : ndarray
|
|
84
|
+
Indices of the first occurrence of each unique value in the
|
|
85
|
+
sorted sample.
|
|
86
|
+
counts : ndarray
|
|
87
|
+
Number of occurrences of each unique value.
|
|
88
|
+
n_xs : int
|
|
89
|
+
Sample size.
|
|
90
|
+
name : str
|
|
91
|
+
Plotting-position scheme.
|
|
92
|
+
|
|
93
|
+
Returns
|
|
94
|
+
-------
|
|
95
|
+
p_ast_s, d_ast_s : ndarray
|
|
96
|
+
Same shapes and semantics as :func:`plotting_positions`.
|
|
97
|
+
"""
|
|
98
|
+
alpha, beta = plotting_position_coefficients(name)
|
|
99
|
+
n_e = n_xs + 1 - alpha - beta
|
|
100
|
+
ps = (np.arange(1, n_xs + 1) - alpha) / n_e
|
|
101
|
+
# cum_sum[i:j] == ps[i:j].sum() (prefix sum with a leading zero).
|
|
102
|
+
cum_sum = np.cumsum(np.insert(ps, 0, 0))
|
|
103
|
+
pre_p_ast = (cum_sum[idxs + counts] - cum_sum[idxs]) / counts
|
|
104
|
+
p_ast_s = np.concatenate([np.array([0.0]), pre_p_ast, np.array([1.0])])
|
|
105
|
+
d_ast_s = np.diff(p_ast_s)
|
|
106
|
+
return p_ast_s, d_ast_s
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ppidest
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Plotting Position-Information Divergence framework for parameter estimation of univariate distributions
|
|
5
|
+
Keywords: statistics,parameter-estimation,information-divergence,plotting-positions,extremes
|
|
6
|
+
Author: Takuya Kawanishi
|
|
7
|
+
Author-email: Takuya Kawanishi <takuya@exanalytics.sakura.ne.jp>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
15
|
+
Requires-Dist: numpy>=2.5.3
|
|
16
|
+
Requires-Dist: scipy>=1.18.1
|
|
17
|
+
Requires-Python: >=3.13
|
|
18
|
+
Project-URL: Repository, https://codeberg.org/takuya_kawanishi/ppidest
|
|
19
|
+
Project-URL: Homepage, https://codeberg.org/takuya_kawanishi/ppidest
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
22
|
+
# ppidest
|
|
23
|
+
|
|
24
|
+
**Plotting Position — Information Divergence** framework for parameter
|
|
25
|
+
estimation of univariate distributions.
|
|
26
|
+
|
|
27
|
+
The empirical cumulative distribution function of a sample is discretized
|
|
28
|
+
using *plotting positions*, and a parametric distribution is fitted by
|
|
29
|
+
minimizing an *information divergence* between the empirical gaps and the
|
|
30
|
+
gaps implied by the candidate CDF.
|
|
31
|
+
|
|
32
|
+
## Installation
|
|
33
|
+
|
|
34
|
+
Requires Python ≥ 3.13.
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
uv sync # create the environment and install the package
|
|
38
|
+
uv run python -m unittest discover -s tests # run the test suite
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Dependencies: `numpy`, `scipy`. The package is also installable from a
|
|
42
|
+
source checkout with `pip install .` or `uv pip install .`, and the
|
|
43
|
+
distributions on PyPI ship a `ppidest` console script (see
|
|
44
|
+
[Command line](#command-line)).
|
|
45
|
+
|
|
46
|
+
## Quick start
|
|
47
|
+
|
|
48
|
+
Fit a GEV model to a small sample using the default (Kullback-Leibler)
|
|
49
|
+
divergence:
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import numpy as np
|
|
53
|
+
import ppidest
|
|
54
|
+
|
|
55
|
+
xs = np.array([0.38, 0.51, 1.44, 2.14])
|
|
56
|
+
|
|
57
|
+
# Plotting Position – Information Divergence estimator
|
|
58
|
+
res = ppidest.find_ppid_min(xs, ppidest.gev_cdf, [0.0, 1.0, 0.25])
|
|
59
|
+
print(res.x) # fitted (loc, scale, shape) -> [0.5600, 0.3639, 1.0585]
|
|
60
|
+
print(res.fun) # minimized divergence -> 0.17543
|
|
61
|
+
|
|
62
|
+
# Compute a return level from the fitted model
|
|
63
|
+
ppidest.gev_return_level(100.0, res.x)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Comparison across estimation methods (`scipy.stats` is used only to
|
|
67
|
+
generate the data / provide the density here):
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
import scipy.stats
|
|
71
|
+
|
|
72
|
+
xs = np.sort(scipy.stats.norm.rvs(loc=1.0, scale=2.0, size=8))
|
|
73
|
+
|
|
74
|
+
# Density/CDF callables must use the (x, par) convention
|
|
75
|
+
def norm_pdf(x, par):
|
|
76
|
+
return scipy.stats.norm.pdf(x, loc=par[0], scale=par[1])
|
|
77
|
+
|
|
78
|
+
# Maximum likelihood from the PDF
|
|
79
|
+
ml = ppidest.ML(xs, norm_pdf).find_mle([1.0, 2.0])
|
|
80
|
+
|
|
81
|
+
# PPID with a different divergence and plot position
|
|
82
|
+
ppid = ppidest.PPID(xs, ppidest.normal_cdf, plotposition="Weibull")
|
|
83
|
+
res = ppid.find_min_div([1.0, 2.0], divergence="Jensen-Shannon")
|
|
84
|
+
|
|
85
|
+
# Forward extra positional arguments to the CDF callable
|
|
86
|
+
def norm_cdf_given_mu(x, par, mu):
|
|
87
|
+
return ppidest.normal_cdf(x, [mu, par[0]])
|
|
88
|
+
|
|
89
|
+
ppid = ppidest.PPID(xs, norm_cdf_given_mu)
|
|
90
|
+
res = ppid.find_min_div([1.0, 2.0], args=(0.0,)) # scale only, mu fixed
|
|
91
|
+
```
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## Background
|
|
95
|
+
|
|
96
|
+
For a sorted sample `x_(1) <= ... <= x_(n)` a plotting position assigns a
|
|
97
|
+
probability to the i-th order statistic:
|
|
98
|
+
|
|
99
|
+
```
|
|
100
|
+
p_i = (i - alpha) / (n + 1 - alpha - beta)
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
`alpha` and `beta` are fixed by the chosen scheme:
|
|
104
|
+
|
|
105
|
+
| Scheme | alpha | beta |
|
|
106
|
+
|--------------|-------|------|
|
|
107
|
+
| Weibull | 0.0 | 0.0 |
|
|
108
|
+
| median | 0.3 | 0.3 |
|
|
109
|
+
| Gringorten | 0.44 | 0.44 |
|
|
110
|
+
| Hazen | 0.5 | 0.5 |
|
|
111
|
+
|
|
112
|
+
Ties are aggregated so each unique value carries the summed probability
|
|
113
|
+
of the order statistics sharing it, and the support is augmented with two
|
|
114
|
+
boundary bins at `0` and `1`. The result is an empirical probability
|
|
115
|
+
vector `d*` (the "empirical gaps"). A candidate distribution with
|
|
116
|
+
parameters `θ` gives model gaps
|
|
117
|
+
|
|
118
|
+
```
|
|
119
|
+
dd_j(θ) = F(x_j; θ) - F(x_{j-1}; θ), with x_0 = -inf, x_{k+1} = +inf
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
The estimator minimizes an information divergence `D(d* || dd(θ))` over
|
|
123
|
+
`θ`:
|
|
124
|
+
|
|
125
|
+
- **Kullback-Leibler** —
|
|
126
|
+
`Σ d* log(d*/dd)`
|
|
127
|
+
- **generalized KL** —
|
|
128
|
+
`Σ d* log(d*/dd) - d* + dd`
|
|
129
|
+
- **symmetric KL** —
|
|
130
|
+
`Σ (dd - d*) log(dd/d*)`
|
|
131
|
+
- **Jensen-Shannon** —
|
|
132
|
+
`½ Σ d* log(d*/m) + dd log(dd/m)`, `m = (d* + dd)/2`
|
|
133
|
+
- **beta** (order `β`) —
|
|
134
|
+
`Σ d*(d*^(β-1) - dd^(β-1))/(β-1) - (d*^β - dd^β)/β`
|
|
135
|
+
- **power** (index `λ`) —
|
|
136
|
+
`1/(λ(λ+1)) Σ d*[(d*/dd)^λ - 1]`
|
|
137
|
+
- **Rényi** (order `α`) —
|
|
138
|
+
`1/(α-1) log Σ d*^α dd^(1-α)`
|
|
139
|
+
|
|
140
|
+
## Package layout
|
|
141
|
+
|
|
142
|
+
```
|
|
143
|
+
src/ppidest/
|
|
144
|
+
distributions.py distribution cdf/pdf/quantile functions
|
|
145
|
+
(normal, GEV, three-parameter Weibull)
|
|
146
|
+
plotting.py plotting-position schemes and empirical gaps
|
|
147
|
+
divergences.py information divergences between probability vectors
|
|
148
|
+
estimators.py ML, PPID estimators
|
|
149
|
+
__init__.py public API and legacy calc_* aliases
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
### Distributions
|
|
153
|
+
|
|
154
|
+
All distribution functions share the signature `f(x, par)` with
|
|
155
|
+
`par = (loc, scale, shape)` (normal and Weibull parameters differ, see
|
|
156
|
+
their docstrings):
|
|
157
|
+
|
|
158
|
+
- `normal_cdf`, `normal_pdf`
|
|
159
|
+
- `gev_cdf`, `gev_pdf`, `gev_quantile`, `gev_return_level`
|
|
160
|
+
(GEV shape `ξ` uses the convention `ξ > 0` → heavy tail)
|
|
161
|
+
- `weibull_cdf`, `weibull_pdf`
|
|
162
|
+
|
|
163
|
+
### Divergences
|
|
164
|
+
|
|
165
|
+
`ppidest.kl_divergence`, `kl_generalized_divergence`,
|
|
166
|
+
`kl_symmetric_divergence`, `jensen_shannon_divergence`,
|
|
167
|
+
`beta_divergence(das, ddas, beta)`, `power_divergence(das, ddas, lmbd)`,
|
|
168
|
+
`renyi_divergence(das, ddas, alpha)` — all take the empirical gaps
|
|
169
|
+
`das` and model gaps `ddas` as 1-D arrays.
|
|
170
|
+
|
|
171
|
+
### Estimators
|
|
172
|
+
|
|
173
|
+
| Estimator | Objective | Fit method |
|
|
174
|
+
|-----------|-----------|------------|
|
|
175
|
+
| `ML(xs, pdf)` | negative log-likelihood | `find_mle` |
|
|
176
|
+
| `PPID(xs, cdf)` | chosen information divergence | `find_min_div` |
|
|
177
|
+
|
|
178
|
+
Every `find_*` method mirrors the keyword arguments of
|
|
179
|
+
`scipy.optimize.minimize` (`method`, `jac`, `hess`, `bounds`, ...).
|
|
180
|
+
`PPID.find_min_div` accepts `divergence` and, for the parametric
|
|
181
|
+
divergences, `pdiv`. The parametric divergence arguments are passed as
|
|
182
|
+
singular floats (the pre-1.0 API required a length-one list).
|
|
183
|
+
|
|
184
|
+
The legacy `calc_gev_cdf`, ... names are exported from the package root as
|
|
185
|
+
aliases for backward compatibility, e.g. `ppidest.calc_gev_cdf` is the
|
|
186
|
+
same object as `ppidest.gev_cdf`.
|
|
187
|
+
|
|
188
|
+
## Notes on optimization
|
|
189
|
+
|
|
190
|
+
Nelder–Mead is the default solver because the objective only needs
|
|
191
|
+
CDF evaluation. During the search the parameters may leave the support
|
|
192
|
+
of the distribution, producing `NaN`; these evaluations are harmless and
|
|
193
|
+
the corresponding warnings are suppressed inside the distribution and
|
|
194
|
+
divergence functions. Careful initial values (e.g. from the method of
|
|
195
|
+
moments) are recommended for the scale and shape parameters.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
ppidest/__init__.py,sha256=uRXTqJGA6puAoEwFiTFmgto8uL6TbcC3-pSTrXi78PA,6680
|
|
2
|
+
ppidest/distributions.py,sha256=Z130id3V2h42rN8Jlqjrms21Ouealn003gu0bXESlGU,6857
|
|
3
|
+
ppidest/divergences.py,sha256=6Ho8GIkSo5a9dHLQGzhOqKgYcsoTaRobhWOtkBIcPzU,4780
|
|
4
|
+
ppidest/estimators.py,sha256=SLqoChKY70X3NduldcDdrqGvtPEbKxcG1DzSEKW40gE,13702
|
|
5
|
+
ppidest/plotting.py,sha256=Klwr5iAonoMrj7E_ucb6tUsaeeWWCC4jbcTq51h8gVs,3568
|
|
6
|
+
ppidest-0.1.0.dist-info/WHEEL,sha256=_d8F1e7SqtoW6CDj4Gi8lFC26a_7I17R7zPLCKTp4Fg,81
|
|
7
|
+
ppidest-0.1.0.dist-info/entry_points.txt,sha256=oDABWj4ndH6xJvBVI26yx7sZyBvOWi2OPSzdjMhctyM,42
|
|
8
|
+
ppidest-0.1.0.dist-info/METADATA,sha256=ddZiWjhbZzxcpIvBAJl44OjeMzAFBdtwpl2cVzIp1ws,6788
|
|
9
|
+
ppidest-0.1.0.dist-info/RECORD,,
|