ogeth 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ogeth/labor.py ADDED
@@ -0,0 +1,198 @@
1
+ """
2
+ ------------------------------------------------------------------------
3
+ Computes the average labor participation rate for each age cohort.
4
+ ------------------------------------------------------------------------
5
+ """
6
+
7
+ import os
8
+ import numpy as np
9
+ import pandas as pd
10
+ import scipy.ndimage.filters as filter
11
+ from scipy import interpolate
12
+ import matplotlib
13
+ import matplotlib.pyplot as plt
14
+ from mpl_toolkits.mplot3d import Axes3D
15
+
16
+ CUR_DIR = os.path.abspath(os.path.dirname(__file__))
17
+
18
+
19
+ def get_labor_data(
20
+ year=2023, data_dir=os.path.join(CUR_DIR, "..", "ogeth", "data", "qlfs")
21
+ ):
22
+ """
23
+ Read in "raw" Quarterly Labour Force Survey data to calculate moments.
24
+
25
+ Args:
26
+ year (int): year of data to read in
27
+ data_dir (str): path to directory with QLFS data
28
+
29
+ Returns:
30
+ df (Pandas DataFrame): QLFS data to compute labor supply from
31
+
32
+ """
33
+ # read in data for all quarters
34
+ df_list = []
35
+ for q in range(1, 5):
36
+ file = os.path.join(data_dir, f"qlfs-{year}-q{q}-worker-v1.csv")
37
+ df = pd.read_csv(file, encoding="latin-1", low_memory=False)
38
+ df_list.append(df)
39
+ df = pd.concat(df_list)
40
+
41
+ # rename some variables
42
+ df.rename(
43
+ columns={
44
+ "Q418HRSWRK": "hours",
45
+ # "Q14AGE": "age",
46
+ "age_grp1": "age_group",
47
+ # "Hrswrk": "hours2",
48
+ "Weight": "weight",
49
+ },
50
+ inplace=True,
51
+ )
52
+ # if hours is a string, take only part after space
53
+ df["hours"] = df["hours"].str.split().str[-1]
54
+ df["hours"] = pd.to_numeric(df["hours"], errors="coerce")
55
+ # create weighted mean hours by age
56
+ # replace missing hours with zero
57
+ df["hours"] = df["hours"].fillna(0)
58
+ # drop if hours are missing
59
+ # df = df[~df['hours'].isna()]
60
+
61
+ return df
62
+
63
+
64
+ def compute_labor_moments(df, S=80):
65
+ """
66
+ Compute moments from labor data.
67
+
68
+ Args:
69
+ df (Pandas DataFrame): QLFS data to compute labor supply from
70
+ S (int): number of periods of economic life for model households
71
+
72
+ Returns:
73
+ labor_dist_out (Numpy array): fraction of time spent working
74
+ by age
75
+
76
+ """
77
+
78
+ # Find fraction of total time people work on average by age group
79
+ by_age = pd.DataFrame(
80
+ df.groupby("age_group").apply(
81
+ lambda x: (x["hours"] * x["weight"]).sum() / x["weight"].sum()
82
+ )
83
+ )
84
+ # give column name to hours
85
+ by_age.columns = ["hours"]
86
+ # drop with indices that are in ['00-04', '05-09', '10-14', '14-Oct', '9-May']
87
+ by_age = by_age.drop(["00-04", "05-09", "10-14", "14-Oct", "9-May"])
88
+ # also drop age 15-19 since not in model
89
+ # by_age = by_age.drop('15-19')
90
+ # rename index for 75+ to 75-85 to be able to get midpoint
91
+ by_age = by_age.rename(index={"75+": "75-85"})
92
+ # compute midpoints of age groups
93
+ age_midpoints = (
94
+ pd.Series(by_age.index)
95
+ .str.split("-")
96
+ .apply(lambda x: (int(x[0]) + int(x[1])) / 2)
97
+ )
98
+
99
+ # get fraction of time endowment worked (assume time
100
+ # endowment is 24 hours minus required time to sleep)
101
+ by_age["frac_work"] = by_age["hours"] / ((24 - 8) * 7)
102
+
103
+ # fit a cubic spline to these data points -- only through age 57
104
+ # labor_dist = interpolate.interp1d(
105
+ # age_midpoints[:-5], by_age['frac_work'][:-5], kind='cubic')
106
+ labor_dist = interpolate.interp1d(
107
+ age_midpoints, by_age["frac_work"], kind="cubic"
108
+ )
109
+ # now evaluate the spline at each age
110
+ labor_spline = labor_dist(np.linspace(20, 80, 60))
111
+
112
+ # Data have sufficient obs through age 57 (55-59 age group)
113
+ # Fit a line to the last few years of the average labor
114
+ # participation which extends from ages 57 to 100.
115
+ slope = (labor_spline[-1] - labor_spline[-8]) / (8 + 1)
116
+ # intercept = by_age['frac_work'][-1] - slope*len(by_age['frac_work'])
117
+ # extension = slope * (np.linspace(56, 80, 23)) + intercept
118
+ # to_dot = slope * (np.linspace(45, 56, 11)) + intercept
119
+
120
+ labor_dist_data = np.zeros(80)
121
+ labor_dist_data[:60] = labor_spline
122
+ labor_dist_data[60:] = labor_spline[-1] + slope * range(20)
123
+
124
+ # the above computes moments if the model period is a year
125
+ # the following adjusts those moments in case it is smaller
126
+ labor_dist_out = (
127
+ 1 # filter.uniform_filter(labor_dist_data, size=int(80 / S))[
128
+ )
129
+ # :: int(80 / S)
130
+ # ]
131
+
132
+ return labor_dist_data, age_midpoints, by_age, labor_dist_out
133
+
134
+
135
+ def VCV_moments(qlfs, n=1000, S=80):
136
+ """
137
+ Compute Variance-Covariance matrix for labor moments by
138
+ bootstrapping data.
139
+
140
+ Args:
141
+ cps (Pandas DataFrame): CPS data to compute labor supply from
142
+ S (int): number of periods of economic life for model households
143
+ n (int): number of bootstrap iterations to run
144
+ bin_weights (Numpy array): ability weight, length J
145
+
146
+ Output:
147
+ VCV (Numpy array): = variance-covariance matrix of labor
148
+ moments, size SxS
149
+
150
+ """
151
+ labor_moments_boot = np.zeros((n, S))
152
+ for i in range(n):
153
+ boot = qlfs[np.random.randint(2, size=len(qlfs.index)).astype(bool)]
154
+ _, _, _, labor_moments_boot[i, :] = compute_labor_moments(boot, S)
155
+
156
+ VCV = np.cov(labor_moments_boot.T)
157
+
158
+ return VCV
159
+
160
+
161
+ def labor_data_graphs(
162
+ year=2023,
163
+ data_dir=os.path.join(CUR_DIR, "..", "ogeth", "data", "qlfs"),
164
+ S=80,
165
+ output_dir=None,
166
+ ):
167
+ """
168
+ Plot labor supply data.
169
+
170
+ Args:
171
+ weighted (Numpy array):
172
+ S (int): number of periods of economic life for model households
173
+ J (int): number of lifetime income groups
174
+ output_dir (str): path to save figures to
175
+
176
+ Returns:
177
+ None
178
+
179
+ """
180
+ # get labor data
181
+ interpolated_data, age_midpoints, by_age, _ = compute_labor_moments(
182
+ get_labor_data(year, data_dir), S
183
+ )
184
+ plt.plot(np.linspace(20, 100, 80), interpolated_data)
185
+ # add scatter plot of raw data
186
+ plt.scatter(age_midpoints, by_age["frac_work"], color="red", alpha=0.5)
187
+ plt.xlabel("Age")
188
+ plt.ylabel("Labor supply")
189
+ plt.title("Labor supply by age")
190
+ plt.legend(["Interpolated", "Data"])
191
+ if output_dir:
192
+ plt.savefig(
193
+ os.path.join(output_dir, "labor_dist_data.png"),
194
+ bbox_inches="tight",
195
+ dpi=300,
196
+ )
197
+ else:
198
+ return plt
ogeth/macro_params.py ADDED
@@ -0,0 +1,294 @@
1
+ """
2
+ This module uses data from World Bank WDI, World Bank Quarterly Public
3
+ Sector Debt (QPSD) database, the IMF, and UN ILO to find values for
4
+ parameters for the OG-ETH model that rely on macro data for calibration.
5
+ """
6
+
7
+ # imports
8
+ from pandas_datareader import wb
9
+ import pandas as pd
10
+ import numpy as np
11
+ import requests
12
+ import datetime
13
+ import statsmodels.api as sm
14
+ from io import StringIO
15
+
16
+
17
+ def get_macro_params(
18
+ data_start_date=datetime.datetime(1947, 1, 1),
19
+ data_end_date=datetime.datetime(2024, 12, 31),
20
+ country_iso="ETH",
21
+ update_from_api=False,
22
+ ):
23
+ """
24
+ Compute values of parameters that are derived from macro data
25
+
26
+ Args:
27
+ data_start_date (datetime): start date for data
28
+ data_end_date (datetime): end date for data
29
+ country_iso (str): ISO code for country
30
+
31
+ Returns:
32
+ macro_parameters (dict): dictionary of parameter values
33
+ """
34
+ # initialize a dictionary of parameters
35
+ macro_parameters = {}
36
+ # baseline date formatted for World Bank data
37
+ baseline_YYYYQ = (
38
+ str(data_end_date.year)
39
+ + "Q"
40
+ + str(pd.Timestamp(data_end_date).quarter)
41
+ )
42
+
43
+ """
44
+ Retrieve data from the World Bank World Development Indicators.
45
+ """
46
+ # Dictionaries of variables and their corresponding World Bank codes
47
+ # Annual data
48
+ wb_a_variable_dict = {
49
+ "GDP per capita (constant 2015 US$)": "NY.GDP.PCAP.KD",
50
+ "Real GDP (constant 2015 US$)": "NY.GDP.MKTP.KD",
51
+ "Nominal GDP (current US$)": "NY.GDP.MKTP.CD",
52
+ "General government final consumption expenditure (current US$)": "NE.CON.GOVT.CD",
53
+ }
54
+ # Quarterly data
55
+ wb_q_variable_dict = {
56
+ "Gross PSD USD - domestic creditors": "DP.DOD.DECD.CR.PS.CD",
57
+ "Gross PSD USD - external creditors": "DP.DOD.DECX.CR.PS.CD",
58
+ "Gross PSD Gen Gov - percentage of GDP": "DP.DOD.DECT.CR.GG.Z1",
59
+ }
60
+ if update_from_api:
61
+ try:
62
+ # pull series of interest from the WB using pandas_datareader
63
+ # Annual data
64
+ wb_data_a = wb.download(
65
+ indicator=wb_a_variable_dict.values(),
66
+ country=country_iso,
67
+ start=data_start_date,
68
+ end=data_end_date,
69
+ )
70
+ wb_data_a.rename(
71
+ columns=dict((y, x) for x, y in wb_a_variable_dict.items()),
72
+ inplace=True,
73
+ )
74
+ # Quarterly data
75
+ wb_data_q = wb.download(
76
+ indicator=wb_q_variable_dict.values(),
77
+ country=country_iso,
78
+ start=data_start_date,
79
+ end=data_end_date,
80
+ )
81
+ wb_data_q.rename(
82
+ columns=dict((y, x) for x, y in wb_q_variable_dict.items()),
83
+ inplace=True,
84
+ )
85
+ # Remove the hierarchical index (country and year) of
86
+ # wb_data_q and create a single row index using year
87
+ wb_data_q = wb_data_q.reset_index()
88
+ wb_data_q = wb_data_q.set_index("year")
89
+
90
+ # Function to get the latest valid data if baseline_YYYYQ is missing or NaN
91
+ def get_valid_data(series, baseline_YYYYQ):
92
+ value = series.get(baseline_YYYYQ, None)
93
+
94
+ if pd.isna(value):
95
+ latest_non_nan = series.dropna().last_valid_index()
96
+
97
+ if latest_non_nan is not None:
98
+ print(
99
+ f"Warning: No data for {baseline_YYYYQ}. Using last available quarter: {latest_non_nan}"
100
+ )
101
+ value = series.get(latest_non_nan, None)
102
+ else:
103
+ print(
104
+ "Warning: No historical data available. Skipping update."
105
+ )
106
+ value = None
107
+
108
+ return value
109
+
110
+ # Compute macro parameters from WB data
111
+ macro_parameters["initial_debt_ratio"] = get_valid_data(
112
+ pd.Series(wb_data_q["Gross PSD Gen Gov - percentage of GDP"])
113
+ / 100,
114
+ baseline_YYYYQ,
115
+ )
116
+ print(
117
+ f"initial_debt_ratio updated from World Bank API: {macro_parameters['initial_debt_ratio']}"
118
+ )
119
+
120
+ # Compute initial_foreign_debt_ratio safely
121
+ if (
122
+ "Gross PSD USD - external creditors" in wb_data_q.columns
123
+ and "Gross PSD USD - domestic creditors" in wb_data_q.columns
124
+ ):
125
+
126
+ total_debt = (
127
+ wb_data_q["Gross PSD USD - domestic creditors"]
128
+ + wb_data_q["Gross PSD USD - external creditors"]
129
+ )
130
+
131
+ # Avoid division by zero
132
+ wb_data_q["foreign_debt_ratio"] = wb_data_q[
133
+ "Gross PSD USD - external creditors"
134
+ ] / total_debt.replace(0, np.nan)
135
+
136
+ macro_parameters["initial_foreign_debt_ratio"] = (
137
+ get_valid_data(
138
+ wb_data_q["foreign_debt_ratio"], baseline_YYYYQ
139
+ )
140
+ )
141
+ else:
142
+ print(
143
+ "Warning: Missing debt variables in World Bank data. Skipping update for initial_foreign_debt_ratio."
144
+ )
145
+
146
+ print(
147
+ f"initial_foreign_debt_ratio updated from World Bank API: {macro_parameters['initial_foreign_debt_ratio']}"
148
+ )
149
+
150
+ # Compute zeta_D safely
151
+ macro_parameters["zeta_D"] = [
152
+ macro_parameters["initial_foreign_debt_ratio"]
153
+ ] # Since it's the same formula, we use the same calculated value
154
+
155
+ print(
156
+ f"zeta_D updated from World Bank API: {macro_parameters['zeta_D']}"
157
+ )
158
+
159
+ # Compute annual GDP growth safely
160
+ if "GDP per capita (constant 2015 US$)" in wb_data_a.columns:
161
+ g_y_series = wb_data_a[
162
+ "GDP per capita (constant 2015 US$)"
163
+ ].pct_change(-1)
164
+
165
+ # If all values are NaN, return None
166
+ macro_parameters["g_y_annual"] = (
167
+ g_y_series.mean() if not g_y_series.isna().all() else None
168
+ )
169
+ else:
170
+ print(
171
+ "Warning: Missing GDP per capita data in World Bank data. Skipping update for g_y_annual."
172
+ )
173
+
174
+ print(
175
+ f"g_y_annual updated from World Bank API: {macro_parameters['g_y_annual']}"
176
+ )
177
+ except:
178
+ print("Failed to retrieve data from World Bank")
179
+ print("Will not update the following parameters:")
180
+ print(
181
+ "[initial_debt_ratio, initial_foreign_debt_ratio, zeta_D, g_y]"
182
+ )
183
+ else:
184
+ print("Not updating from World Bank API")
185
+
186
+ """
187
+ Retrieve labour share data from the United Nations ILOSTAT Data API
188
+ (see https://rshiny.ilo.org/dataexplorer9/?lang=en)
189
+ The series code is SDG_1041_NOC_RT_A (capital share)
190
+ Labor share (gamma) = 1 - capital share
191
+ If this fails we will not update gamma in 'default_parameters.json'
192
+ """
193
+ if update_from_api:
194
+ try:
195
+ target = (
196
+ "https://rplumber.ilo.org/data/indicator/"
197
+ + "?id=SDG_1041_NOC_RT_A"
198
+ + "&ref_area="
199
+ + str(country_iso)
200
+ + "&timefrom="
201
+ + str(data_start_date.year)
202
+ + "&timeto="
203
+ + str(data_end_date.year)
204
+ + "&type=both&format=.csv"
205
+ )
206
+ # Add headers
207
+ headers = {
208
+ "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36"
209
+ }
210
+
211
+ print("Attempting to update gamma from ILOSTAT")
212
+ response = requests.get(target, headers=headers)
213
+ if response.status_code != 200:
214
+ print(f"Error: Received status code {response.status_code}")
215
+ else:
216
+ print("Request successful.")
217
+ csv_content = StringIO(response.text)
218
+ df_temp = pd.read_csv(csv_content)
219
+ ilo_data = df_temp[["time", "obs_value"]]
220
+ # find gamma, capital's share of income
221
+ macro_parameters["gamma"] = [
222
+ 1
223
+ - (
224
+ (
225
+ ilo_data.loc[
226
+ ilo_data["time"] == data_end_date.year, "obs_value"
227
+ ].squeeze()
228
+ )
229
+ / 100
230
+ )
231
+ ]
232
+ print(
233
+ f"gamma updated from ILOSTAT API: {macro_parameters['gamma']}"
234
+ )
235
+ except:
236
+ print("Failed to retrieve data from ILOSTAT")
237
+ print("Will not update gamma")
238
+ else:
239
+ print("Not updating from ILOSTAT API")
240
+
241
+ """
242
+ Calibrate parameters from IMF data
243
+ """
244
+
245
+ if update_from_api:
246
+ # alpha_T, non-social security benefits as a fraction of GDP
247
+ # source: https://data.imf.org/?sk=78d0bcc1-7a8f-44eb-8a2c-d4e472b8e64b&hide_uv=1
248
+ # alpha_T = Employment-related social benefits expense - Social security benefits expense
249
+ macro_parameters["alpha_T"] = [0.36 - 0.0] # 2022 = 0.36
250
+
251
+ # alpha_G, gov't consumption expenditures as a fraction of GDP
252
+ # source: https://data.imf.org/?sk=23ca1c1d-e6a5-4f18-bc2e-7e215837f971&hide_uv=1
253
+ # alpha_G = Expense - Interest expense - Social benefits expense
254
+ macro_parameters["alpha_G"] = [0.324 - 0.047 - 0.036] # 2022 = 0.241
255
+
256
+ """"
257
+ Esimate the discount on sovereign yields relative to private debt
258
+ Follow the methodology in Li, Magud, Werner, Witte (2021)
259
+ available at:
260
+ https://www.imf.org/en/Publications/WP/Issues/2021/06/04/The-Long-Run-Impact-of-Sovereign-Yields-on-Corporate-Yields-in-Emerging-Markets-50224
261
+ discussion is here: https://github.com/EAPD-DRB/OG-ETH/issues/22
262
+ Steps:
263
+ 1) Generate modelled corporate yields (corp_yhat) for a range of
264
+ sovereign yields (sov_y) using the estimated equation in col 2 of
265
+ table 8 (and figure 3). 2) Estimate the OLS using sovereign yields
266
+ as the dependent variable
267
+ """
268
+
269
+ # # estimate r_gov_shift and r_gov_scale
270
+ sov_y = np.arange(20, 120) / 10
271
+ corp_yhat = 8.199 - (2.975 * sov_y) + (0.478 * sov_y**2)
272
+ corp_yhat = sm.add_constant(corp_yhat)
273
+ mod = sm.OLS(
274
+ sov_y,
275
+ corp_yhat,
276
+ )
277
+ res = mod.fit()
278
+ # First term is the constant and needs to be divided by 100 to have
279
+ # the correct unit. Second term is the coefficient
280
+ macro_parameters["r_gov_shift"] = [-res.params[0] / 100]
281
+ macro_parameters["r_gov_scale"] = [res.params[1]]
282
+ # Report new values
283
+ print(f"alpha_T updated from IMF data: {macro_parameters['alpha_T']}")
284
+ print(f"alpha_G updated from IMF data: {macro_parameters['alpha_G']}")
285
+ print(
286
+ f"r_gov_shift updated from IMF data: {macro_parameters['r_gov_shift']}"
287
+ )
288
+ print(
289
+ f"r_gov_scale updated from IMF data: {macro_parameters['r_gov_scale']}"
290
+ )
291
+ else:
292
+ print("Not updating alpha_T, alpha_G, r_gov_shift, r_gov_scale")
293
+
294
+ return macro_parameters