ogeth 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ogeth/__init__.py +11 -0
- ogeth/calibrate.py +108 -0
- ogeth/constants.py +490 -0
- ogeth/income.py +529 -0
- ogeth/input_output.py +103 -0
- ogeth/labor.py +198 -0
- ogeth/macro_params.py +294 -0
- ogeth/ogeth_default_parameters.json +101449 -0
- ogeth/test_est_chi_n.py +46 -0
- ogeth/utils.py +44 -0
- ogeth-0.0.0.dist-info/METADATA +126 -0
- ogeth-0.0.0.dist-info/RECORD +15 -0
- ogeth-0.0.0.dist-info/WHEEL +5 -0
- ogeth-0.0.0.dist-info/licenses/LICENSE +121 -0
- ogeth-0.0.0.dist-info/top_level.txt +1 -0
ogeth/income.py
ADDED
|
@@ -0,0 +1,529 @@
|
|
|
1
|
+
"""
|
|
2
|
+
-----------------------------------------------------------------
|
|
3
|
+
Functions for created the matrix of ability levels, e. This can
|
|
4
|
+
only be used for looking at the 25, 50, 70, 80, 90, 99, and 100th
|
|
5
|
+
percentiles, as it uses fitted polynomials to those percentiles.
|
|
6
|
+
-----------------------------------------------------------------
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
import scipy.optimize as opt
|
|
11
|
+
import scipy.interpolate as si
|
|
12
|
+
from ogcore import parameter_plots as pp
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def arctan_func(xvals, a, b, c):
|
|
16
|
+
r"""
|
|
17
|
+
This function generates predicted ability levels given data (xvals)
|
|
18
|
+
and parameters a, b, and c, from the following arctan function:
|
|
19
|
+
|
|
20
|
+
.. math::
|
|
21
|
+
y = (-a / \pi) * \arctan(b * x + c) + (a / 2)
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
xvals (Numpy array): data inputs to arctan function
|
|
25
|
+
a (scalar): scale parameter for arctan function
|
|
26
|
+
b (scalar): curvature parameter for arctan function
|
|
27
|
+
c (scalar): shift parameter for arctan function
|
|
28
|
+
|
|
29
|
+
Returns:
|
|
30
|
+
yvals (Numpy array): predicted values (output) of arctan
|
|
31
|
+
function
|
|
32
|
+
|
|
33
|
+
"""
|
|
34
|
+
yvals = (-a / np.pi) * np.arctan(b * xvals + c) + (a / 2)
|
|
35
|
+
return yvals
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def arctan_deriv_func(xvals, a, b, c):
|
|
39
|
+
r"""
|
|
40
|
+
This function generates predicted derivatives of arctan function
|
|
41
|
+
given data (xvals) and parameters a, b, and c. The functional form
|
|
42
|
+
of the derivative of the function is the following:
|
|
43
|
+
|
|
44
|
+
.. math::
|
|
45
|
+
y = - (a * b) / (\pi * (1 + (b * xvals + c)^2))
|
|
46
|
+
|
|
47
|
+
Args:
|
|
48
|
+
xvals (Numpy array): data inputs to arctan derivative function
|
|
49
|
+
a (scalar): scale parameter for arctan function
|
|
50
|
+
b (scalar): curvature parameter for arctan function
|
|
51
|
+
c (scalar): shift parameter for arctan function
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
yvals (Numpy array): predicted values (output) of arctan
|
|
55
|
+
derivative function
|
|
56
|
+
|
|
57
|
+
"""
|
|
58
|
+
yvals = -(a * b) / (np.pi * (1 + (b * xvals + c) ** 2))
|
|
59
|
+
return yvals
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def arc_error(abc_vals, params):
|
|
63
|
+
"""
|
|
64
|
+
This function returns a vector of errors in the three criteria on
|
|
65
|
+
which the arctan function is fit to predict extrapolated ability in
|
|
66
|
+
ages 81 to 100.::
|
|
67
|
+
|
|
68
|
+
1) The arctan function value at age 80 must match the estimated
|
|
69
|
+
original function value at age 80.
|
|
70
|
+
2) The arctan function slope at age 80 must match the estimated
|
|
71
|
+
original function slope at age 80.
|
|
72
|
+
3) The level of ability at age 100 must be a given fraction
|
|
73
|
+
(abil_deprec) below the ability level at age 80.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
abc_vals (tuple): contains (a,b,c)
|
|
77
|
+
|
|
78
|
+
* a (scalar): scale parameter for arctan function
|
|
79
|
+
* b (scalar): curvature parameter for arctan function
|
|
80
|
+
* c (scalar): shift parameter for arctan function
|
|
81
|
+
params (tuple): contains (first_point, coef1, coef2, coef3,
|
|
82
|
+
abil_deprec)
|
|
83
|
+
|
|
84
|
+
* first_point (scalar): ability level at age 80, > 0
|
|
85
|
+
* coef1 (scalar): coefficient in log ability equation on
|
|
86
|
+
linear term in age
|
|
87
|
+
* coef2 (scalar): coefficient in log ability equation on
|
|
88
|
+
quadratic term in age
|
|
89
|
+
* coef3 (scalar): coefficient in log ability equation on
|
|
90
|
+
cubic term in age
|
|
91
|
+
* abil_deprec (scalar): ability depreciation rate between
|
|
92
|
+
ages 80 and 100, in (0, 1).
|
|
93
|
+
|
|
94
|
+
Returns:
|
|
95
|
+
error_vec (Numpy array): errors ([error1, error2, error3])
|
|
96
|
+
|
|
97
|
+
* error1 (scalar): error between ability level at age 80
|
|
98
|
+
from original function minus the predicted ability at
|
|
99
|
+
age 80 from the arctan function given a, b, and c
|
|
100
|
+
* error2 (scalar): error between the slope of the original
|
|
101
|
+
function at age 80 minus the slope of the arctan
|
|
102
|
+
function at age 80 given a, b, and c
|
|
103
|
+
* error3 (scalar): error between the ability level at age
|
|
104
|
+
100 predicted by the original model value times
|
|
105
|
+
abil_deprec minus the ability predicted by the arctan
|
|
106
|
+
function at age 100 given a, b, and c
|
|
107
|
+
|
|
108
|
+
"""
|
|
109
|
+
a, b, c = abc_vals
|
|
110
|
+
first_point, coef1, coef2, coef3, abil_deprec = params
|
|
111
|
+
error1 = first_point - arctan_func(80, a, b, c)
|
|
112
|
+
if (3 * coef3 * 80**2 + 2 * coef2 * 80 + coef1) < 0:
|
|
113
|
+
error2 = (
|
|
114
|
+
3 * coef3 * 80**2 + 2 * coef2 * 80 + coef1
|
|
115
|
+
) * first_point - arctan_deriv_func(80, a, b, c)
|
|
116
|
+
else:
|
|
117
|
+
error2 = -0.02 * first_point - arctan_deriv_func(80, a, b, c)
|
|
118
|
+
error3 = abil_deprec * first_point - arctan_func(100, a, b, c)
|
|
119
|
+
error_vec = np.array([error1, error2, error3])
|
|
120
|
+
|
|
121
|
+
return error_vec
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def arctan_fit(first_point, coef1, coef2, coef3, abil_deprec, init_guesses):
|
|
125
|
+
"""
|
|
126
|
+
This function fits an arctan function to the last 20 years of the
|
|
127
|
+
ability levels of a particular ability group to extrapolate
|
|
128
|
+
abilities by trying to match the slope in the 80th year and the
|
|
129
|
+
ability depreciation rate between years 80 and 100.
|
|
130
|
+
|
|
131
|
+
Args:
|
|
132
|
+
first_point (scalar): ability level at age 80, > 0
|
|
133
|
+
coef1 (scalar): coefficient in log ability equation on linear
|
|
134
|
+
term in age
|
|
135
|
+
coef2 (scalar): coefficient in log ability equation on
|
|
136
|
+
quadratic term in age
|
|
137
|
+
coef3 (scalar): coefficient in log ability equation on cubic
|
|
138
|
+
term in age
|
|
139
|
+
abil_deprec (scalar): ability depreciation rate between
|
|
140
|
+
ages 80 and 100, in (0, 1)
|
|
141
|
+
init_guesses (Numpy array): initial guesses
|
|
142
|
+
|
|
143
|
+
Returns:
|
|
144
|
+
abil_last (Numpy array): extrapolated ability levels for ages
|
|
145
|
+
81 to 100, length 20
|
|
146
|
+
|
|
147
|
+
"""
|
|
148
|
+
params = [first_point, coef1, coef2, coef3, abil_deprec]
|
|
149
|
+
solution = opt.root(arc_error, init_guesses, args=params, method="lm")
|
|
150
|
+
[a, b, c] = solution.x
|
|
151
|
+
old_ages = np.linspace(81, 100, 20)
|
|
152
|
+
abil_last = arctan_func(old_ages, a, b, c)
|
|
153
|
+
return abil_last
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def get_e_interp(S, age_wgts, age_wgts_80, abil_wgts, plot_path=None):
|
|
157
|
+
"""
|
|
158
|
+
This function takes a source matrix of lifetime earnings profiles
|
|
159
|
+
(abilities, emat) of size (80, 7), where 80 is the number of ages
|
|
160
|
+
and 7 is the number of ability types in the source matrix, and
|
|
161
|
+
interpolates new values of a new S x J sized matrix of abilities
|
|
162
|
+
using linear interpolation. [NOTE: For this application, cubic
|
|
163
|
+
spline interpolation introduces too much curvature.]
|
|
164
|
+
|
|
165
|
+
This function also includes the two cases in which J = 9 and J = 10
|
|
166
|
+
that include higher lifetime earning percentiles calibrated using
|
|
167
|
+
Piketty and Saez (2003).
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
Args:
|
|
171
|
+
S (int): number of ages to interpolate. This method assumes that
|
|
172
|
+
ages are evenly spaced between the beginning of the 21st
|
|
173
|
+
year and the end of the 100th year, >= 3
|
|
174
|
+
age_wgts (Numpy array): distribution of population in each age
|
|
175
|
+
for the interpolated ages, length S
|
|
176
|
+
age_wgts_80 (Numpy array): percent of population in each
|
|
177
|
+
one-year age from 21 to 100, length 80
|
|
178
|
+
abil_wgts (Numpy array): distribution of population in each
|
|
179
|
+
ability group, length J
|
|
180
|
+
plot_path (str)): if True, creates plots of emat_orig and the new
|
|
181
|
+
interpolated emat_new
|
|
182
|
+
|
|
183
|
+
Returns:
|
|
184
|
+
emat_new_scaled (Numpy array): interpolated ability matrix scaled
|
|
185
|
+
so that population-weighted average is 1, size SxJ
|
|
186
|
+
|
|
187
|
+
"""
|
|
188
|
+
# Get original 80 x 7 ability matrix
|
|
189
|
+
abil_wgts_orig = np.array([0.25, 0.25, 0.2, 0.1, 0.1, 0.09, 0.01])
|
|
190
|
+
emat_orig = get_e_orig(age_wgts_80, abil_wgts_orig, plot_path)
|
|
191
|
+
if (
|
|
192
|
+
S == 80
|
|
193
|
+
and np.array_equal(
|
|
194
|
+
np.squeeze(abil_wgts),
|
|
195
|
+
np.array([0.25, 0.25, 0.2, 0.1, 0.1, 0.09, 0.01]),
|
|
196
|
+
)
|
|
197
|
+
is True
|
|
198
|
+
):
|
|
199
|
+
emat_new_scaled = emat_orig
|
|
200
|
+
elif (
|
|
201
|
+
S == 80
|
|
202
|
+
and np.array_equal(
|
|
203
|
+
np.squeeze(abil_wgts),
|
|
204
|
+
np.array(
|
|
205
|
+
[0.25, 0.25, 0.2, 0.1, 0.1, 0.09, 0.005, 0.004, 0.0009, 0.0001]
|
|
206
|
+
),
|
|
207
|
+
)
|
|
208
|
+
is True
|
|
209
|
+
):
|
|
210
|
+
emat_new = np.zeros((S, len(abil_wgts)))
|
|
211
|
+
emat_new[:, :7] = emat_orig
|
|
212
|
+
# Create profiles for top 0.5%, top 0.1% and top 0.01% using
|
|
213
|
+
# Piketty and Saez estimates
|
|
214
|
+
# (https://eml.berkeley.edu/~saez/pikettyqje.pdf)
|
|
215
|
+
# updated for 2018 to create scaling factor
|
|
216
|
+
# assumption is that profile shape of these top 3 groups are
|
|
217
|
+
# same as the top 1% estimated in tax data, just scaled up by
|
|
218
|
+
# ratio determined from P&S 2018 estimates (Table 0, ex cap gains)
|
|
219
|
+
emat_new[:, 5] = emat_orig[:, -2] * 1.25
|
|
220
|
+
emat_new[:, 6] = emat_orig[:, -1] * 0.458759521 * 2.75
|
|
221
|
+
emat_new[:, 7] = emat_orig[:, -1] * 0.847252448 * 3.5
|
|
222
|
+
emat_new[:, 8] = emat_orig[:, -1] * 2.713698465 * 3.5
|
|
223
|
+
emat_new[:, 9] = emat_orig[:, -1] * 18.74863983 * 4.0
|
|
224
|
+
emat_new_scaled = (
|
|
225
|
+
emat_new
|
|
226
|
+
/ (
|
|
227
|
+
emat_new * age_wgts.reshape(80, 1) * abil_wgts.reshape(1, 10)
|
|
228
|
+
).sum()
|
|
229
|
+
)
|
|
230
|
+
elif (
|
|
231
|
+
S == 80
|
|
232
|
+
and np.array_equal(
|
|
233
|
+
np.squeeze(abil_wgts),
|
|
234
|
+
np.array([0.25, 0.25, 0.2, 0.1, 0.1, 0.09, 0.005, 0.004, 0.001]),
|
|
235
|
+
)
|
|
236
|
+
is True
|
|
237
|
+
):
|
|
238
|
+
emat_new = np.zeros((S, len(abil_wgts)))
|
|
239
|
+
emat_new[:, :7] = emat_orig
|
|
240
|
+
# Create profiles for top 0.5%, top 0.1% using
|
|
241
|
+
# Piketty and Saez estimates
|
|
242
|
+
# (https://eml.berkeley.edu/~saez/pikettyqje.pdf)
|
|
243
|
+
# updated for 2018 to create scaling factor
|
|
244
|
+
# assumption is that profile shape of these top 3 groups are
|
|
245
|
+
# same as the top 1% estimated in tax data, just scaled up by
|
|
246
|
+
# ratio determined from P&S 2018 estimates (Table 0, ex cap gains)
|
|
247
|
+
emat_new[:, 6] = emat_orig[:, -1] * 0.458759521
|
|
248
|
+
emat_new[:, 7] = emat_orig[:, -1] * 0.847252448
|
|
249
|
+
emat_new[:, 8] = emat_orig[:, -1] * 4.317192601
|
|
250
|
+
emat_new_scaled = (
|
|
251
|
+
emat_new
|
|
252
|
+
/ (
|
|
253
|
+
emat_new * age_wgts.reshape(80, 1) * abil_wgts.reshape(1, 9)
|
|
254
|
+
).sum()
|
|
255
|
+
)
|
|
256
|
+
else:
|
|
257
|
+
# generate abil_midp vector
|
|
258
|
+
J = abil_wgts.shape[0]
|
|
259
|
+
abil_midp = np.zeros(J)
|
|
260
|
+
pct_lb = 0.0
|
|
261
|
+
for j in range(J):
|
|
262
|
+
abil_midp[j] = pct_lb + 0.5 * abil_wgts[j]
|
|
263
|
+
pct_lb += abil_wgts[j]
|
|
264
|
+
|
|
265
|
+
# Make sure that values in abil_midp are within interpolating
|
|
266
|
+
# bounds set by the hard coded abil_wgts_orig
|
|
267
|
+
if abil_midp.min() < 0.125 or abil_midp.max() > 0.995:
|
|
268
|
+
err = (
|
|
269
|
+
"One or more entries in abils vector is outside the "
|
|
270
|
+
+ "allowable bounds."
|
|
271
|
+
)
|
|
272
|
+
raise RuntimeError(err)
|
|
273
|
+
|
|
274
|
+
emat_j_midp = np.array(
|
|
275
|
+
[0.125, 0.375, 0.600, 0.750, 0.850, 0.945, 0.995]
|
|
276
|
+
)
|
|
277
|
+
emat_s_midp = np.linspace(20.5, 99.5, 80)
|
|
278
|
+
emat_j_mesh, emat_s_mesh = np.meshgrid(emat_j_midp, emat_s_midp)
|
|
279
|
+
newstep = 80 / S
|
|
280
|
+
new_s_midp = np.linspace(20 + 0.5 * newstep, 100 - 0.5 * newstep, S)
|
|
281
|
+
new_j_mesh, new_s_mesh = np.meshgrid(abil_midp, new_s_midp)
|
|
282
|
+
newcoords = np.hstack(
|
|
283
|
+
(
|
|
284
|
+
emat_s_mesh.reshape((80 * 7, 1)),
|
|
285
|
+
emat_j_mesh.reshape((80 * 7, 1)),
|
|
286
|
+
)
|
|
287
|
+
)
|
|
288
|
+
emat_new = si.griddata(
|
|
289
|
+
newcoords,
|
|
290
|
+
emat_orig.flatten(),
|
|
291
|
+
(new_s_mesh, new_j_mesh),
|
|
292
|
+
method="linear",
|
|
293
|
+
)
|
|
294
|
+
emat_new_scaled = (
|
|
295
|
+
emat_new
|
|
296
|
+
/ (
|
|
297
|
+
emat_new * age_wgts.reshape(S, 1) * abil_wgts.reshape(1, J)
|
|
298
|
+
).sum()
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
if plot_path is not None:
|
|
302
|
+
kwargs = {"path": plot_path, "filesuffix": "_intrp_scaled"}
|
|
303
|
+
pp.plot_income_data(
|
|
304
|
+
new_s_midp,
|
|
305
|
+
abil_midp,
|
|
306
|
+
abil_wgts,
|
|
307
|
+
emat_new_scaled,
|
|
308
|
+
plot_path,
|
|
309
|
+
**kwargs,
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
return emat_new_scaled
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def get_e_orig(age_wgts, abil_wgts, plot_path=None):
|
|
316
|
+
r"""
|
|
317
|
+
This function generates the 80 x 7 matrix of lifetime earnings
|
|
318
|
+
ability profiles, corresponding to annual ages from 21 to 100 and to
|
|
319
|
+
paths based on income percentiles 0-25, 25-50, 50-70, 70-80, 80-90,
|
|
320
|
+
90-99, 99-100. The ergodic population distribution is an input in
|
|
321
|
+
order to rescale the paths so that the weighted average equals 1.
|
|
322
|
+
|
|
323
|
+
The base curves are the ones in OG-USA, which are then adjusted for ETH.
|
|
324
|
+
|
|
325
|
+
The polynomials are of the form
|
|
326
|
+
|
|
327
|
+
.. math::
|
|
328
|
+
\ln(abil) = \alpha + \beta_{1}\text{age} + \beta_{2}\text{age}^2
|
|
329
|
+
+ \beta_{3}\text{age}^3
|
|
330
|
+
|
|
331
|
+
To calibrate for ETH, the USA curves are adjusted in 2 ways (in this order)
|
|
332
|
+
1) Adjustment by income (J): adjust the gaps between the J-income earning curves
|
|
333
|
+
using data from WID.
|
|
334
|
+
2) Adjustment by age (S): adjust the shape/distribution of each J-income earning
|
|
335
|
+
profile curve using data from NTA.
|
|
336
|
+
|
|
337
|
+
The methodology is described here:
|
|
338
|
+
https://github.com/EAPD-DRB/OG-ETH/issues/18#issuecomment-1368580323
|
|
339
|
+
|
|
340
|
+
Args:
|
|
341
|
+
age_wgts (Numpy array): ergodic age distribution, length S
|
|
342
|
+
abil_wgts (Numpy array): population weights in each lifetime
|
|
343
|
+
earnings group, length J
|
|
344
|
+
plot_path (str): Path to save plots to
|
|
345
|
+
|
|
346
|
+
Returns:
|
|
347
|
+
e_orig_scaled (Numpy array): = lifetime ability profiles scaled
|
|
348
|
+
so that population-weighted average is 1, size SxJ
|
|
349
|
+
|
|
350
|
+
"""
|
|
351
|
+
# Return and error if age_wgts is not a vector of size (80,)
|
|
352
|
+
if age_wgts.shape[0] != 80:
|
|
353
|
+
err = "Vector age_wgts does not have 80 elements."
|
|
354
|
+
raise RuntimeError(err)
|
|
355
|
+
# Return and error if abil_wgts is not a vector of size (7,)
|
|
356
|
+
if abil_wgts.shape[0] != 7:
|
|
357
|
+
err = "Vector abil_wgts does not have 7 elements."
|
|
358
|
+
raise RuntimeError(err)
|
|
359
|
+
|
|
360
|
+
# 1) Generate polynomials using USA data and use them to get income profiles for
|
|
361
|
+
# ages 21 to 80.
|
|
362
|
+
one = np.array(
|
|
363
|
+
[
|
|
364
|
+
-0.09720122,
|
|
365
|
+
0.05995294,
|
|
366
|
+
0.17654618,
|
|
367
|
+
0.21168263,
|
|
368
|
+
0.21638731,
|
|
369
|
+
0.04500235,
|
|
370
|
+
0.09229392,
|
|
371
|
+
]
|
|
372
|
+
)
|
|
373
|
+
two = np.array(
|
|
374
|
+
[
|
|
375
|
+
0.00247639,
|
|
376
|
+
-0.00004086,
|
|
377
|
+
-0.00240656,
|
|
378
|
+
-0.00306555,
|
|
379
|
+
-0.00321041,
|
|
380
|
+
0.00094253,
|
|
381
|
+
0.00012902,
|
|
382
|
+
]
|
|
383
|
+
)
|
|
384
|
+
three = np.array(
|
|
385
|
+
[
|
|
386
|
+
-0.00001842,
|
|
387
|
+
-0.00000521,
|
|
388
|
+
0.00001039,
|
|
389
|
+
0.00001438,
|
|
390
|
+
0.00001579,
|
|
391
|
+
-0.00001470,
|
|
392
|
+
-0.00001169,
|
|
393
|
+
]
|
|
394
|
+
)
|
|
395
|
+
const = np.array(
|
|
396
|
+
[
|
|
397
|
+
3.41e00,
|
|
398
|
+
0.69689692,
|
|
399
|
+
-0.78761958,
|
|
400
|
+
-1.11e00,
|
|
401
|
+
-0.93939272,
|
|
402
|
+
1.60e00,
|
|
403
|
+
1.89e00,
|
|
404
|
+
]
|
|
405
|
+
)
|
|
406
|
+
ages_short = np.tile(np.linspace(21, 80, 60).reshape((60, 1)), (1, 7))
|
|
407
|
+
log_abil_paths = (
|
|
408
|
+
const
|
|
409
|
+
+ (one * ages_short)
|
|
410
|
+
+ (two * (ages_short**2))
|
|
411
|
+
+ (three * (ages_short**3))
|
|
412
|
+
)
|
|
413
|
+
|
|
414
|
+
# New estimated coefficients for ETH after adjustment by income (J) and by age (S)
|
|
415
|
+
const = np.array(
|
|
416
|
+
[
|
|
417
|
+
1.10766851280735,
|
|
418
|
+
-1.47205271208099,
|
|
419
|
+
-2.79826519632522,
|
|
420
|
+
-2.84592025503416,
|
|
421
|
+
-2.33264177437992,
|
|
422
|
+
0.820108734133472,
|
|
423
|
+
0.573684959034946,
|
|
424
|
+
]
|
|
425
|
+
)
|
|
426
|
+
one = np.array(
|
|
427
|
+
[
|
|
428
|
+
-0.0577752937758472,
|
|
429
|
+
0.0993788662241527,
|
|
430
|
+
0.215972106224152,
|
|
431
|
+
0.251108556224153,
|
|
432
|
+
0.255813236224153,
|
|
433
|
+
0.0844282762241525,
|
|
434
|
+
0.131719846224152,
|
|
435
|
+
]
|
|
436
|
+
)
|
|
437
|
+
two = np.array(
|
|
438
|
+
[
|
|
439
|
+
0.00313926193376278,
|
|
440
|
+
0.000622011933762788,
|
|
441
|
+
-0.00174368806623721,
|
|
442
|
+
-0.00240267806623721,
|
|
443
|
+
-0.00254753806623722,
|
|
444
|
+
0.00160540193376279,
|
|
445
|
+
0.000791891933762785,
|
|
446
|
+
]
|
|
447
|
+
)
|
|
448
|
+
three = np.array(
|
|
449
|
+
[
|
|
450
|
+
-0.000035350068460927,
|
|
451
|
+
-2.21400684609271e-05,
|
|
452
|
+
-6.54006846092713e-06,
|
|
453
|
+
-2.55006846092713e-06,
|
|
454
|
+
-1.14006846092704e-06,
|
|
455
|
+
-3.16300684609271e-05,
|
|
456
|
+
-2.86200684609271e-05,
|
|
457
|
+
]
|
|
458
|
+
)
|
|
459
|
+
# compute the lifetime income profiles using the new coefficients
|
|
460
|
+
ages_short_adj = np.tile(np.linspace(21, 80, 60).reshape((60, 1)), (1, 7))
|
|
461
|
+
log_abil_paths_adj = (
|
|
462
|
+
const
|
|
463
|
+
+ (one * ages_short_adj)
|
|
464
|
+
+ (two * (ages_short_adj**2))
|
|
465
|
+
+ (three * (ages_short_adj**3))
|
|
466
|
+
)
|
|
467
|
+
abil_paths_adj = np.exp(log_abil_paths_adj)
|
|
468
|
+
|
|
469
|
+
e_orig = np.zeros((80, 7))
|
|
470
|
+
e_orig[:60, :] = abil_paths_adj
|
|
471
|
+
e_orig[60:, :] = 0.0
|
|
472
|
+
|
|
473
|
+
# 2) Forecast (with some art) the path of the final 20 years of
|
|
474
|
+
# ability types. This following variable is what percentage of
|
|
475
|
+
# ability at age 80 ability falls to at age 100. In general, we
|
|
476
|
+
# wanted people to lose half of their ability over a 20-year
|
|
477
|
+
# period. The first entry is 0.47, though, because nothing higher
|
|
478
|
+
# would converge. The second-to-last is 0.7 because this group
|
|
479
|
+
# actually has a slightly higher ability at age 80 than the last
|
|
480
|
+
# group, so this value makes it decrease more so it ends up being
|
|
481
|
+
# monotonic.
|
|
482
|
+
abil_deprec = np.array([0.47, 0.5, 0.5, 0.5, 0.5, 0.7, 0.5])
|
|
483
|
+
# Initial guesses for the arctan. They're pretty sensitive.
|
|
484
|
+
init_guesses = np.array(
|
|
485
|
+
[
|
|
486
|
+
[58, 0.0756438545595, -5.6940142786],
|
|
487
|
+
[27, 0.069, -5],
|
|
488
|
+
[35, 0.06, -5],
|
|
489
|
+
[37, 0.339936555352, -33.5987329144],
|
|
490
|
+
[70.5229181668, 0.0701993896947, -6.37746859905],
|
|
491
|
+
[35, 0.06, -5],
|
|
492
|
+
[35, 0.06, -5],
|
|
493
|
+
]
|
|
494
|
+
)
|
|
495
|
+
for j in range(7):
|
|
496
|
+
e_orig[60:, j] = arctan_fit(
|
|
497
|
+
e_orig[59, j],
|
|
498
|
+
one[j],
|
|
499
|
+
two[j],
|
|
500
|
+
three[j],
|
|
501
|
+
abil_deprec[j],
|
|
502
|
+
init_guesses[j],
|
|
503
|
+
)
|
|
504
|
+
|
|
505
|
+
# 3) Rescale the lifetime earnings path matrix so that the
|
|
506
|
+
# population weighted average equals 1.
|
|
507
|
+
e_orig_scaled = (
|
|
508
|
+
e_orig
|
|
509
|
+
/ (e_orig * age_wgts.reshape(80, 1) * abil_wgts.reshape(1, 7)).sum()
|
|
510
|
+
)
|
|
511
|
+
|
|
512
|
+
if plot_path is not None:
|
|
513
|
+
ages_long = np.linspace(21, 100, 80)
|
|
514
|
+
abil_midp = np.array([12.5, 37.5, 60.0, 75.0, 85.0, 94.5, 99.5])
|
|
515
|
+
# Plot original unscaled 80 x 7 ability matrix
|
|
516
|
+
kwargs = {"path": plot_path, "filesuffix": "_orig_unscaled"}
|
|
517
|
+
pp.plot_income_data(ages_long, abil_midp, abil_wgts, e_orig, **kwargs)
|
|
518
|
+
|
|
519
|
+
# Plot original scaled 80 x 7 ability matrix
|
|
520
|
+
kwargs = {"path": plot_path, "filesuffix": "_orig_scaled"}
|
|
521
|
+
pp.plot_income_data(
|
|
522
|
+
ages_long,
|
|
523
|
+
abil_midp,
|
|
524
|
+
abil_wgts,
|
|
525
|
+
e_orig_scaled,
|
|
526
|
+
**kwargs,
|
|
527
|
+
)
|
|
528
|
+
|
|
529
|
+
return e_orig_scaled
|
ogeth/input_output.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
import numpy as np
|
|
3
|
+
from ogeth.utils import is_connected
|
|
4
|
+
from ogeth.constants import CONS_DICT, PROD_DICT
|
|
5
|
+
|
|
6
|
+
"""
|
|
7
|
+
Read in Social Accounting Matrix (SAM) file
|
|
8
|
+
This is the most recent SAM for 2019 available from the following page as a downloadable zip folder from UNU WIDER:
|
|
9
|
+
https://www.wider.unu.edu/sites/default/files/Publications/Technical-note/tn2023-1-2019-SASAM-for-distribution.zip
|
|
10
|
+
"""
|
|
11
|
+
# Read in SAM file
|
|
12
|
+
storage_options = {"User-Agent": "Mozilla/5.0"}
|
|
13
|
+
SAM_path = "https://raw.githubusercontent.com/EAPD-DRB/SAM-files/main/Data/ETH/tn2023-1-2019-SASAM-for-distribution.xlsx"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def read_SAM():
|
|
17
|
+
if is_connected():
|
|
18
|
+
try:
|
|
19
|
+
SAM = pd.read_excel(
|
|
20
|
+
SAM_path,
|
|
21
|
+
sheet_name="SASAM 2019 61Ind 4Educ", # Can alternatively use sheet_name="SASM 2019 61Ind4Occ"
|
|
22
|
+
skiprows=3,
|
|
23
|
+
index_col=0,
|
|
24
|
+
storage_options=storage_options,
|
|
25
|
+
)
|
|
26
|
+
print("Successfully read SAM from Github repository.")
|
|
27
|
+
except Exception as e:
|
|
28
|
+
print(f"Failed to read from the GitHub repository: {e}")
|
|
29
|
+
SAM = None
|
|
30
|
+
# If both attempts fail, SAM will be None
|
|
31
|
+
if SAM is None:
|
|
32
|
+
print("Failed to read SAM from both sources.")
|
|
33
|
+
else: # pragma: no cover
|
|
34
|
+
SAM = None
|
|
35
|
+
print("No internet connection. SAM cannot be read.")
|
|
36
|
+
return SAM
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def get_alpha_c(sam=None, cons_dict=CONS_DICT):
|
|
40
|
+
"""
|
|
41
|
+
Calibrate the alpha_c vector, showing the shares of household
|
|
42
|
+
expenditures for each consumption category
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
sam (pd.DataFrame): SAM file
|
|
46
|
+
cons_dict (dict): Dictionary of consumption categories
|
|
47
|
+
|
|
48
|
+
Returns:
|
|
49
|
+
alpha_c (dict): Dictionary of shares of household expenditures
|
|
50
|
+
"""
|
|
51
|
+
if sam is None:
|
|
52
|
+
sam = read_SAM()
|
|
53
|
+
alpha_c = {}
|
|
54
|
+
overall_sum = 0
|
|
55
|
+
for key, value in cons_dict.items():
|
|
56
|
+
# note the subtraction of the row to focus on domestic consumption
|
|
57
|
+
category_total = (
|
|
58
|
+
sam.loc[sam.index.isin(value), "total"].sum()
|
|
59
|
+
- sam.loc[sam.index.isin(value), "row"].sum()
|
|
60
|
+
)
|
|
61
|
+
alpha_c[key] = category_total
|
|
62
|
+
overall_sum += category_total
|
|
63
|
+
for key, value in cons_dict.items():
|
|
64
|
+
alpha_c[key] = alpha_c[key] / overall_sum
|
|
65
|
+
|
|
66
|
+
return alpha_c
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def get_io_matrix(sam=None, cons_dict=CONS_DICT, prod_dict=PROD_DICT):
|
|
70
|
+
"""
|
|
71
|
+
Calibrate the io_matrix array. This array relates the share of each
|
|
72
|
+
production category in each consumption category
|
|
73
|
+
|
|
74
|
+
Args:
|
|
75
|
+
sam (pd.DataFrame): SAM file
|
|
76
|
+
cons_dict (dict): Dictionary of consumption categories
|
|
77
|
+
prod_dict (dict): Dictionary of production categories
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
io_df (pd.DataFrame): Dataframe of io_matrix
|
|
81
|
+
"""
|
|
82
|
+
if sam is None:
|
|
83
|
+
sam = read_SAM()
|
|
84
|
+
# Create initial matrix as dataframe of 0's to fill in
|
|
85
|
+
io_dict = {}
|
|
86
|
+
for key in prod_dict.keys():
|
|
87
|
+
io_dict[key] = np.zeros(len(cons_dict.keys()))
|
|
88
|
+
io_df = pd.DataFrame(io_dict, index=cons_dict.keys())
|
|
89
|
+
# Fill in the matrix
|
|
90
|
+
# Note, each cell in the SAM represents a payment from the columns
|
|
91
|
+
# account to the row account
|
|
92
|
+
# (see https://www.un.org/en/development/desa/policy/capacity/presentations/manila/6_sam_mams_philippines.pdf)
|
|
93
|
+
# We are thus going to take the consumption categories from rows and
|
|
94
|
+
# the production categories from columns
|
|
95
|
+
for ck, cv in cons_dict.items():
|
|
96
|
+
for pk, pv in prod_dict.items():
|
|
97
|
+
io_df.loc[io_df.index == ck, pk] = sam.loc[
|
|
98
|
+
sam.index.isin(cv), pv
|
|
99
|
+
].values.sum()
|
|
100
|
+
# change from levels to share (where each row sums to one)
|
|
101
|
+
io_df = io_df.div(io_df.sum(axis=1), axis=0)
|
|
102
|
+
|
|
103
|
+
return io_df
|