ogeth 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ogeth/income.py ADDED
@@ -0,0 +1,529 @@
1
+ """
2
+ -----------------------------------------------------------------
3
+ Functions for created the matrix of ability levels, e. This can
4
+ only be used for looking at the 25, 50, 70, 80, 90, 99, and 100th
5
+ percentiles, as it uses fitted polynomials to those percentiles.
6
+ -----------------------------------------------------------------
7
+ """
8
+
9
+ import numpy as np
10
+ import scipy.optimize as opt
11
+ import scipy.interpolate as si
12
+ from ogcore import parameter_plots as pp
13
+
14
+
15
+ def arctan_func(xvals, a, b, c):
16
+ r"""
17
+ This function generates predicted ability levels given data (xvals)
18
+ and parameters a, b, and c, from the following arctan function:
19
+
20
+ .. math::
21
+ y = (-a / \pi) * \arctan(b * x + c) + (a / 2)
22
+
23
+ Args:
24
+ xvals (Numpy array): data inputs to arctan function
25
+ a (scalar): scale parameter for arctan function
26
+ b (scalar): curvature parameter for arctan function
27
+ c (scalar): shift parameter for arctan function
28
+
29
+ Returns:
30
+ yvals (Numpy array): predicted values (output) of arctan
31
+ function
32
+
33
+ """
34
+ yvals = (-a / np.pi) * np.arctan(b * xvals + c) + (a / 2)
35
+ return yvals
36
+
37
+
38
+ def arctan_deriv_func(xvals, a, b, c):
39
+ r"""
40
+ This function generates predicted derivatives of arctan function
41
+ given data (xvals) and parameters a, b, and c. The functional form
42
+ of the derivative of the function is the following:
43
+
44
+ .. math::
45
+ y = - (a * b) / (\pi * (1 + (b * xvals + c)^2))
46
+
47
+ Args:
48
+ xvals (Numpy array): data inputs to arctan derivative function
49
+ a (scalar): scale parameter for arctan function
50
+ b (scalar): curvature parameter for arctan function
51
+ c (scalar): shift parameter for arctan function
52
+
53
+ Returns:
54
+ yvals (Numpy array): predicted values (output) of arctan
55
+ derivative function
56
+
57
+ """
58
+ yvals = -(a * b) / (np.pi * (1 + (b * xvals + c) ** 2))
59
+ return yvals
60
+
61
+
62
+ def arc_error(abc_vals, params):
63
+ """
64
+ This function returns a vector of errors in the three criteria on
65
+ which the arctan function is fit to predict extrapolated ability in
66
+ ages 81 to 100.::
67
+
68
+ 1) The arctan function value at age 80 must match the estimated
69
+ original function value at age 80.
70
+ 2) The arctan function slope at age 80 must match the estimated
71
+ original function slope at age 80.
72
+ 3) The level of ability at age 100 must be a given fraction
73
+ (abil_deprec) below the ability level at age 80.
74
+
75
+ Args:
76
+ abc_vals (tuple): contains (a,b,c)
77
+
78
+ * a (scalar): scale parameter for arctan function
79
+ * b (scalar): curvature parameter for arctan function
80
+ * c (scalar): shift parameter for arctan function
81
+ params (tuple): contains (first_point, coef1, coef2, coef3,
82
+ abil_deprec)
83
+
84
+ * first_point (scalar): ability level at age 80, > 0
85
+ * coef1 (scalar): coefficient in log ability equation on
86
+ linear term in age
87
+ * coef2 (scalar): coefficient in log ability equation on
88
+ quadratic term in age
89
+ * coef3 (scalar): coefficient in log ability equation on
90
+ cubic term in age
91
+ * abil_deprec (scalar): ability depreciation rate between
92
+ ages 80 and 100, in (0, 1).
93
+
94
+ Returns:
95
+ error_vec (Numpy array): errors ([error1, error2, error3])
96
+
97
+ * error1 (scalar): error between ability level at age 80
98
+ from original function minus the predicted ability at
99
+ age 80 from the arctan function given a, b, and c
100
+ * error2 (scalar): error between the slope of the original
101
+ function at age 80 minus the slope of the arctan
102
+ function at age 80 given a, b, and c
103
+ * error3 (scalar): error between the ability level at age
104
+ 100 predicted by the original model value times
105
+ abil_deprec minus the ability predicted by the arctan
106
+ function at age 100 given a, b, and c
107
+
108
+ """
109
+ a, b, c = abc_vals
110
+ first_point, coef1, coef2, coef3, abil_deprec = params
111
+ error1 = first_point - arctan_func(80, a, b, c)
112
+ if (3 * coef3 * 80**2 + 2 * coef2 * 80 + coef1) < 0:
113
+ error2 = (
114
+ 3 * coef3 * 80**2 + 2 * coef2 * 80 + coef1
115
+ ) * first_point - arctan_deriv_func(80, a, b, c)
116
+ else:
117
+ error2 = -0.02 * first_point - arctan_deriv_func(80, a, b, c)
118
+ error3 = abil_deprec * first_point - arctan_func(100, a, b, c)
119
+ error_vec = np.array([error1, error2, error3])
120
+
121
+ return error_vec
122
+
123
+
124
+ def arctan_fit(first_point, coef1, coef2, coef3, abil_deprec, init_guesses):
125
+ """
126
+ This function fits an arctan function to the last 20 years of the
127
+ ability levels of a particular ability group to extrapolate
128
+ abilities by trying to match the slope in the 80th year and the
129
+ ability depreciation rate between years 80 and 100.
130
+
131
+ Args:
132
+ first_point (scalar): ability level at age 80, > 0
133
+ coef1 (scalar): coefficient in log ability equation on linear
134
+ term in age
135
+ coef2 (scalar): coefficient in log ability equation on
136
+ quadratic term in age
137
+ coef3 (scalar): coefficient in log ability equation on cubic
138
+ term in age
139
+ abil_deprec (scalar): ability depreciation rate between
140
+ ages 80 and 100, in (0, 1)
141
+ init_guesses (Numpy array): initial guesses
142
+
143
+ Returns:
144
+ abil_last (Numpy array): extrapolated ability levels for ages
145
+ 81 to 100, length 20
146
+
147
+ """
148
+ params = [first_point, coef1, coef2, coef3, abil_deprec]
149
+ solution = opt.root(arc_error, init_guesses, args=params, method="lm")
150
+ [a, b, c] = solution.x
151
+ old_ages = np.linspace(81, 100, 20)
152
+ abil_last = arctan_func(old_ages, a, b, c)
153
+ return abil_last
154
+
155
+
156
+ def get_e_interp(S, age_wgts, age_wgts_80, abil_wgts, plot_path=None):
157
+ """
158
+ This function takes a source matrix of lifetime earnings profiles
159
+ (abilities, emat) of size (80, 7), where 80 is the number of ages
160
+ and 7 is the number of ability types in the source matrix, and
161
+ interpolates new values of a new S x J sized matrix of abilities
162
+ using linear interpolation. [NOTE: For this application, cubic
163
+ spline interpolation introduces too much curvature.]
164
+
165
+ This function also includes the two cases in which J = 9 and J = 10
166
+ that include higher lifetime earning percentiles calibrated using
167
+ Piketty and Saez (2003).
168
+
169
+
170
+ Args:
171
+ S (int): number of ages to interpolate. This method assumes that
172
+ ages are evenly spaced between the beginning of the 21st
173
+ year and the end of the 100th year, >= 3
174
+ age_wgts (Numpy array): distribution of population in each age
175
+ for the interpolated ages, length S
176
+ age_wgts_80 (Numpy array): percent of population in each
177
+ one-year age from 21 to 100, length 80
178
+ abil_wgts (Numpy array): distribution of population in each
179
+ ability group, length J
180
+ plot_path (str)): if True, creates plots of emat_orig and the new
181
+ interpolated emat_new
182
+
183
+ Returns:
184
+ emat_new_scaled (Numpy array): interpolated ability matrix scaled
185
+ so that population-weighted average is 1, size SxJ
186
+
187
+ """
188
+ # Get original 80 x 7 ability matrix
189
+ abil_wgts_orig = np.array([0.25, 0.25, 0.2, 0.1, 0.1, 0.09, 0.01])
190
+ emat_orig = get_e_orig(age_wgts_80, abil_wgts_orig, plot_path)
191
+ if (
192
+ S == 80
193
+ and np.array_equal(
194
+ np.squeeze(abil_wgts),
195
+ np.array([0.25, 0.25, 0.2, 0.1, 0.1, 0.09, 0.01]),
196
+ )
197
+ is True
198
+ ):
199
+ emat_new_scaled = emat_orig
200
+ elif (
201
+ S == 80
202
+ and np.array_equal(
203
+ np.squeeze(abil_wgts),
204
+ np.array(
205
+ [0.25, 0.25, 0.2, 0.1, 0.1, 0.09, 0.005, 0.004, 0.0009, 0.0001]
206
+ ),
207
+ )
208
+ is True
209
+ ):
210
+ emat_new = np.zeros((S, len(abil_wgts)))
211
+ emat_new[:, :7] = emat_orig
212
+ # Create profiles for top 0.5%, top 0.1% and top 0.01% using
213
+ # Piketty and Saez estimates
214
+ # (https://eml.berkeley.edu/~saez/pikettyqje.pdf)
215
+ # updated for 2018 to create scaling factor
216
+ # assumption is that profile shape of these top 3 groups are
217
+ # same as the top 1% estimated in tax data, just scaled up by
218
+ # ratio determined from P&S 2018 estimates (Table 0, ex cap gains)
219
+ emat_new[:, 5] = emat_orig[:, -2] * 1.25
220
+ emat_new[:, 6] = emat_orig[:, -1] * 0.458759521 * 2.75
221
+ emat_new[:, 7] = emat_orig[:, -1] * 0.847252448 * 3.5
222
+ emat_new[:, 8] = emat_orig[:, -1] * 2.713698465 * 3.5
223
+ emat_new[:, 9] = emat_orig[:, -1] * 18.74863983 * 4.0
224
+ emat_new_scaled = (
225
+ emat_new
226
+ / (
227
+ emat_new * age_wgts.reshape(80, 1) * abil_wgts.reshape(1, 10)
228
+ ).sum()
229
+ )
230
+ elif (
231
+ S == 80
232
+ and np.array_equal(
233
+ np.squeeze(abil_wgts),
234
+ np.array([0.25, 0.25, 0.2, 0.1, 0.1, 0.09, 0.005, 0.004, 0.001]),
235
+ )
236
+ is True
237
+ ):
238
+ emat_new = np.zeros((S, len(abil_wgts)))
239
+ emat_new[:, :7] = emat_orig
240
+ # Create profiles for top 0.5%, top 0.1% using
241
+ # Piketty and Saez estimates
242
+ # (https://eml.berkeley.edu/~saez/pikettyqje.pdf)
243
+ # updated for 2018 to create scaling factor
244
+ # assumption is that profile shape of these top 3 groups are
245
+ # same as the top 1% estimated in tax data, just scaled up by
246
+ # ratio determined from P&S 2018 estimates (Table 0, ex cap gains)
247
+ emat_new[:, 6] = emat_orig[:, -1] * 0.458759521
248
+ emat_new[:, 7] = emat_orig[:, -1] * 0.847252448
249
+ emat_new[:, 8] = emat_orig[:, -1] * 4.317192601
250
+ emat_new_scaled = (
251
+ emat_new
252
+ / (
253
+ emat_new * age_wgts.reshape(80, 1) * abil_wgts.reshape(1, 9)
254
+ ).sum()
255
+ )
256
+ else:
257
+ # generate abil_midp vector
258
+ J = abil_wgts.shape[0]
259
+ abil_midp = np.zeros(J)
260
+ pct_lb = 0.0
261
+ for j in range(J):
262
+ abil_midp[j] = pct_lb + 0.5 * abil_wgts[j]
263
+ pct_lb += abil_wgts[j]
264
+
265
+ # Make sure that values in abil_midp are within interpolating
266
+ # bounds set by the hard coded abil_wgts_orig
267
+ if abil_midp.min() < 0.125 or abil_midp.max() > 0.995:
268
+ err = (
269
+ "One or more entries in abils vector is outside the "
270
+ + "allowable bounds."
271
+ )
272
+ raise RuntimeError(err)
273
+
274
+ emat_j_midp = np.array(
275
+ [0.125, 0.375, 0.600, 0.750, 0.850, 0.945, 0.995]
276
+ )
277
+ emat_s_midp = np.linspace(20.5, 99.5, 80)
278
+ emat_j_mesh, emat_s_mesh = np.meshgrid(emat_j_midp, emat_s_midp)
279
+ newstep = 80 / S
280
+ new_s_midp = np.linspace(20 + 0.5 * newstep, 100 - 0.5 * newstep, S)
281
+ new_j_mesh, new_s_mesh = np.meshgrid(abil_midp, new_s_midp)
282
+ newcoords = np.hstack(
283
+ (
284
+ emat_s_mesh.reshape((80 * 7, 1)),
285
+ emat_j_mesh.reshape((80 * 7, 1)),
286
+ )
287
+ )
288
+ emat_new = si.griddata(
289
+ newcoords,
290
+ emat_orig.flatten(),
291
+ (new_s_mesh, new_j_mesh),
292
+ method="linear",
293
+ )
294
+ emat_new_scaled = (
295
+ emat_new
296
+ / (
297
+ emat_new * age_wgts.reshape(S, 1) * abil_wgts.reshape(1, J)
298
+ ).sum()
299
+ )
300
+
301
+ if plot_path is not None:
302
+ kwargs = {"path": plot_path, "filesuffix": "_intrp_scaled"}
303
+ pp.plot_income_data(
304
+ new_s_midp,
305
+ abil_midp,
306
+ abil_wgts,
307
+ emat_new_scaled,
308
+ plot_path,
309
+ **kwargs,
310
+ )
311
+
312
+ return emat_new_scaled
313
+
314
+
315
+ def get_e_orig(age_wgts, abil_wgts, plot_path=None):
316
+ r"""
317
+ This function generates the 80 x 7 matrix of lifetime earnings
318
+ ability profiles, corresponding to annual ages from 21 to 100 and to
319
+ paths based on income percentiles 0-25, 25-50, 50-70, 70-80, 80-90,
320
+ 90-99, 99-100. The ergodic population distribution is an input in
321
+ order to rescale the paths so that the weighted average equals 1.
322
+
323
+ The base curves are the ones in OG-USA, which are then adjusted for ETH.
324
+
325
+ The polynomials are of the form
326
+
327
+ .. math::
328
+ \ln(abil) = \alpha + \beta_{1}\text{age} + \beta_{2}\text{age}^2
329
+ + \beta_{3}\text{age}^3
330
+
331
+ To calibrate for ETH, the USA curves are adjusted in 2 ways (in this order)
332
+ 1) Adjustment by income (J): adjust the gaps between the J-income earning curves
333
+ using data from WID.
334
+ 2) Adjustment by age (S): adjust the shape/distribution of each J-income earning
335
+ profile curve using data from NTA.
336
+
337
+ The methodology is described here:
338
+ https://github.com/EAPD-DRB/OG-ETH/issues/18#issuecomment-1368580323
339
+
340
+ Args:
341
+ age_wgts (Numpy array): ergodic age distribution, length S
342
+ abil_wgts (Numpy array): population weights in each lifetime
343
+ earnings group, length J
344
+ plot_path (str): Path to save plots to
345
+
346
+ Returns:
347
+ e_orig_scaled (Numpy array): = lifetime ability profiles scaled
348
+ so that population-weighted average is 1, size SxJ
349
+
350
+ """
351
+ # Return and error if age_wgts is not a vector of size (80,)
352
+ if age_wgts.shape[0] != 80:
353
+ err = "Vector age_wgts does not have 80 elements."
354
+ raise RuntimeError(err)
355
+ # Return and error if abil_wgts is not a vector of size (7,)
356
+ if abil_wgts.shape[0] != 7:
357
+ err = "Vector abil_wgts does not have 7 elements."
358
+ raise RuntimeError(err)
359
+
360
+ # 1) Generate polynomials using USA data and use them to get income profiles for
361
+ # ages 21 to 80.
362
+ one = np.array(
363
+ [
364
+ -0.09720122,
365
+ 0.05995294,
366
+ 0.17654618,
367
+ 0.21168263,
368
+ 0.21638731,
369
+ 0.04500235,
370
+ 0.09229392,
371
+ ]
372
+ )
373
+ two = np.array(
374
+ [
375
+ 0.00247639,
376
+ -0.00004086,
377
+ -0.00240656,
378
+ -0.00306555,
379
+ -0.00321041,
380
+ 0.00094253,
381
+ 0.00012902,
382
+ ]
383
+ )
384
+ three = np.array(
385
+ [
386
+ -0.00001842,
387
+ -0.00000521,
388
+ 0.00001039,
389
+ 0.00001438,
390
+ 0.00001579,
391
+ -0.00001470,
392
+ -0.00001169,
393
+ ]
394
+ )
395
+ const = np.array(
396
+ [
397
+ 3.41e00,
398
+ 0.69689692,
399
+ -0.78761958,
400
+ -1.11e00,
401
+ -0.93939272,
402
+ 1.60e00,
403
+ 1.89e00,
404
+ ]
405
+ )
406
+ ages_short = np.tile(np.linspace(21, 80, 60).reshape((60, 1)), (1, 7))
407
+ log_abil_paths = (
408
+ const
409
+ + (one * ages_short)
410
+ + (two * (ages_short**2))
411
+ + (three * (ages_short**3))
412
+ )
413
+
414
+ # New estimated coefficients for ETH after adjustment by income (J) and by age (S)
415
+ const = np.array(
416
+ [
417
+ 1.10766851280735,
418
+ -1.47205271208099,
419
+ -2.79826519632522,
420
+ -2.84592025503416,
421
+ -2.33264177437992,
422
+ 0.820108734133472,
423
+ 0.573684959034946,
424
+ ]
425
+ )
426
+ one = np.array(
427
+ [
428
+ -0.0577752937758472,
429
+ 0.0993788662241527,
430
+ 0.215972106224152,
431
+ 0.251108556224153,
432
+ 0.255813236224153,
433
+ 0.0844282762241525,
434
+ 0.131719846224152,
435
+ ]
436
+ )
437
+ two = np.array(
438
+ [
439
+ 0.00313926193376278,
440
+ 0.000622011933762788,
441
+ -0.00174368806623721,
442
+ -0.00240267806623721,
443
+ -0.00254753806623722,
444
+ 0.00160540193376279,
445
+ 0.000791891933762785,
446
+ ]
447
+ )
448
+ three = np.array(
449
+ [
450
+ -0.000035350068460927,
451
+ -2.21400684609271e-05,
452
+ -6.54006846092713e-06,
453
+ -2.55006846092713e-06,
454
+ -1.14006846092704e-06,
455
+ -3.16300684609271e-05,
456
+ -2.86200684609271e-05,
457
+ ]
458
+ )
459
+ # compute the lifetime income profiles using the new coefficients
460
+ ages_short_adj = np.tile(np.linspace(21, 80, 60).reshape((60, 1)), (1, 7))
461
+ log_abil_paths_adj = (
462
+ const
463
+ + (one * ages_short_adj)
464
+ + (two * (ages_short_adj**2))
465
+ + (three * (ages_short_adj**3))
466
+ )
467
+ abil_paths_adj = np.exp(log_abil_paths_adj)
468
+
469
+ e_orig = np.zeros((80, 7))
470
+ e_orig[:60, :] = abil_paths_adj
471
+ e_orig[60:, :] = 0.0
472
+
473
+ # 2) Forecast (with some art) the path of the final 20 years of
474
+ # ability types. This following variable is what percentage of
475
+ # ability at age 80 ability falls to at age 100. In general, we
476
+ # wanted people to lose half of their ability over a 20-year
477
+ # period. The first entry is 0.47, though, because nothing higher
478
+ # would converge. The second-to-last is 0.7 because this group
479
+ # actually has a slightly higher ability at age 80 than the last
480
+ # group, so this value makes it decrease more so it ends up being
481
+ # monotonic.
482
+ abil_deprec = np.array([0.47, 0.5, 0.5, 0.5, 0.5, 0.7, 0.5])
483
+ # Initial guesses for the arctan. They're pretty sensitive.
484
+ init_guesses = np.array(
485
+ [
486
+ [58, 0.0756438545595, -5.6940142786],
487
+ [27, 0.069, -5],
488
+ [35, 0.06, -5],
489
+ [37, 0.339936555352, -33.5987329144],
490
+ [70.5229181668, 0.0701993896947, -6.37746859905],
491
+ [35, 0.06, -5],
492
+ [35, 0.06, -5],
493
+ ]
494
+ )
495
+ for j in range(7):
496
+ e_orig[60:, j] = arctan_fit(
497
+ e_orig[59, j],
498
+ one[j],
499
+ two[j],
500
+ three[j],
501
+ abil_deprec[j],
502
+ init_guesses[j],
503
+ )
504
+
505
+ # 3) Rescale the lifetime earnings path matrix so that the
506
+ # population weighted average equals 1.
507
+ e_orig_scaled = (
508
+ e_orig
509
+ / (e_orig * age_wgts.reshape(80, 1) * abil_wgts.reshape(1, 7)).sum()
510
+ )
511
+
512
+ if plot_path is not None:
513
+ ages_long = np.linspace(21, 100, 80)
514
+ abil_midp = np.array([12.5, 37.5, 60.0, 75.0, 85.0, 94.5, 99.5])
515
+ # Plot original unscaled 80 x 7 ability matrix
516
+ kwargs = {"path": plot_path, "filesuffix": "_orig_unscaled"}
517
+ pp.plot_income_data(ages_long, abil_midp, abil_wgts, e_orig, **kwargs)
518
+
519
+ # Plot original scaled 80 x 7 ability matrix
520
+ kwargs = {"path": plot_path, "filesuffix": "_orig_scaled"}
521
+ pp.plot_income_data(
522
+ ages_long,
523
+ abil_midp,
524
+ abil_wgts,
525
+ e_orig_scaled,
526
+ **kwargs,
527
+ )
528
+
529
+ return e_orig_scaled
ogeth/input_output.py ADDED
@@ -0,0 +1,103 @@
1
+ import pandas as pd
2
+ import numpy as np
3
+ from ogeth.utils import is_connected
4
+ from ogeth.constants import CONS_DICT, PROD_DICT
5
+
6
+ """
7
+ Read in Social Accounting Matrix (SAM) file
8
+ This is the most recent SAM for 2019 available from the following page as a downloadable zip folder from UNU WIDER:
9
+ https://www.wider.unu.edu/sites/default/files/Publications/Technical-note/tn2023-1-2019-SASAM-for-distribution.zip
10
+ """
11
+ # Read in SAM file
12
+ storage_options = {"User-Agent": "Mozilla/5.0"}
13
+ SAM_path = "https://raw.githubusercontent.com/EAPD-DRB/SAM-files/main/Data/ETH/tn2023-1-2019-SASAM-for-distribution.xlsx"
14
+
15
+
16
+ def read_SAM():
17
+ if is_connected():
18
+ try:
19
+ SAM = pd.read_excel(
20
+ SAM_path,
21
+ sheet_name="SASAM 2019 61Ind 4Educ", # Can alternatively use sheet_name="SASM 2019 61Ind4Occ"
22
+ skiprows=3,
23
+ index_col=0,
24
+ storage_options=storage_options,
25
+ )
26
+ print("Successfully read SAM from Github repository.")
27
+ except Exception as e:
28
+ print(f"Failed to read from the GitHub repository: {e}")
29
+ SAM = None
30
+ # If both attempts fail, SAM will be None
31
+ if SAM is None:
32
+ print("Failed to read SAM from both sources.")
33
+ else: # pragma: no cover
34
+ SAM = None
35
+ print("No internet connection. SAM cannot be read.")
36
+ return SAM
37
+
38
+
39
+ def get_alpha_c(sam=None, cons_dict=CONS_DICT):
40
+ """
41
+ Calibrate the alpha_c vector, showing the shares of household
42
+ expenditures for each consumption category
43
+
44
+ Args:
45
+ sam (pd.DataFrame): SAM file
46
+ cons_dict (dict): Dictionary of consumption categories
47
+
48
+ Returns:
49
+ alpha_c (dict): Dictionary of shares of household expenditures
50
+ """
51
+ if sam is None:
52
+ sam = read_SAM()
53
+ alpha_c = {}
54
+ overall_sum = 0
55
+ for key, value in cons_dict.items():
56
+ # note the subtraction of the row to focus on domestic consumption
57
+ category_total = (
58
+ sam.loc[sam.index.isin(value), "total"].sum()
59
+ - sam.loc[sam.index.isin(value), "row"].sum()
60
+ )
61
+ alpha_c[key] = category_total
62
+ overall_sum += category_total
63
+ for key, value in cons_dict.items():
64
+ alpha_c[key] = alpha_c[key] / overall_sum
65
+
66
+ return alpha_c
67
+
68
+
69
+ def get_io_matrix(sam=None, cons_dict=CONS_DICT, prod_dict=PROD_DICT):
70
+ """
71
+ Calibrate the io_matrix array. This array relates the share of each
72
+ production category in each consumption category
73
+
74
+ Args:
75
+ sam (pd.DataFrame): SAM file
76
+ cons_dict (dict): Dictionary of consumption categories
77
+ prod_dict (dict): Dictionary of production categories
78
+
79
+ Returns:
80
+ io_df (pd.DataFrame): Dataframe of io_matrix
81
+ """
82
+ if sam is None:
83
+ sam = read_SAM()
84
+ # Create initial matrix as dataframe of 0's to fill in
85
+ io_dict = {}
86
+ for key in prod_dict.keys():
87
+ io_dict[key] = np.zeros(len(cons_dict.keys()))
88
+ io_df = pd.DataFrame(io_dict, index=cons_dict.keys())
89
+ # Fill in the matrix
90
+ # Note, each cell in the SAM represents a payment from the columns
91
+ # account to the row account
92
+ # (see https://www.un.org/en/development/desa/policy/capacity/presentations/manila/6_sam_mams_philippines.pdf)
93
+ # We are thus going to take the consumption categories from rows and
94
+ # the production categories from columns
95
+ for ck, cv in cons_dict.items():
96
+ for pk, pv in prod_dict.items():
97
+ io_df.loc[io_df.index == ck, pk] = sam.loc[
98
+ sam.index.isin(cv), pv
99
+ ].values.sum()
100
+ # change from levels to share (where each row sums to one)
101
+ io_df = io_df.div(io_df.sum(axis=1), axis=0)
102
+
103
+ return io_df