diffindiff 2.5.5__tar.gz → 2.5.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {diffindiff-2.5.5 → diffindiff-2.5.7}/PKG-INFO +26 -7
- {diffindiff-2.5.5 → diffindiff-2.5.7}/README.md +6 -5
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/config.py +3 -3
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/didanalysis_helper.py +70 -43
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/didtools.py +133 -78
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff.egg-info/PKG-INFO +26 -7
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff.egg-info/requires.txt +4 -2
- {diffindiff-2.5.5 → diffindiff-2.5.7}/setup.py +7 -3
- {diffindiff-2.5.5 → diffindiff-2.5.7}/MANIFEST.in +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/__init__.py +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/didanalysis.py +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/diddata.py +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/tests/__init__.py +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/tests/data/Corona_Hesse.xlsx +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/tests/data/counties_DE.csv +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/tests/data/curfew_DE.csv +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff/tests/tests_diffindiff.py +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff.egg-info/SOURCES.txt +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff.egg-info/dependency_links.txt +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/diffindiff.egg-info/top_level.txt +0 -0
- {diffindiff-2.5.5 → diffindiff-2.5.7}/setup.cfg +0 -0
|
@@ -1,10 +1,28 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: diffindiff
|
|
3
|
-
Version: 2.5.
|
|
3
|
+
Version: 2.5.7
|
|
4
4
|
Summary: diffindiff: Python library for convenient Difference-in-Differences analyses
|
|
5
5
|
Author: Thomas Wieland
|
|
6
6
|
Author-email: geowieland@googlemail.com
|
|
7
7
|
Description-Content-Type: text/markdown
|
|
8
|
+
Requires-Dist: pandas
|
|
9
|
+
Requires-Dist: numpy
|
|
10
|
+
Requires-Dist: statsmodels>=0.14.5
|
|
11
|
+
Requires-Dist: scipy>=1.17
|
|
12
|
+
Requires-Dist: scikit-learn
|
|
13
|
+
Requires-Dist: openpyxl
|
|
14
|
+
Requires-Dist: matplotlib
|
|
15
|
+
Requires-Dist: patsy
|
|
16
|
+
Provides-Extra: optional
|
|
17
|
+
Requires-Dist: lightgbm; extra == "optional"
|
|
18
|
+
Requires-Dist: xgboost; extra == "optional"
|
|
19
|
+
Dynamic: author
|
|
20
|
+
Dynamic: author-email
|
|
21
|
+
Dynamic: description
|
|
22
|
+
Dynamic: description-content-type
|
|
23
|
+
Dynamic: provides-extra
|
|
24
|
+
Dynamic: requires-dist
|
|
25
|
+
Dynamic: summary
|
|
8
26
|
|
|
9
27
|
# diffindiff: Python library for convenient Difference-in-Differences analyses
|
|
10
28
|
|
|
@@ -29,7 +47,7 @@ A case study that utilizes the diffindiff library is available on [arXiv](https:
|
|
|
29
47
|
|
|
30
48
|
If you use this software, please cite:
|
|
31
49
|
|
|
32
|
-
Wieland, T. (2026). diffindiff: A Python library for convenient difference-in-differences analyses (Version 2.5.
|
|
50
|
+
Wieland, T. (2026). diffindiff: A Python library for convenient difference-in-differences analyses (Version 2.5.7) [Computer software]. Zenodo. https://doi.org/10.5281/zenodo.18656820
|
|
33
51
|
|
|
34
52
|
|
|
35
53
|
## Installation
|
|
@@ -177,11 +195,12 @@ See the /tests directory for usage examples of most of the included functions.
|
|
|
177
195
|
|
|
178
196
|
## AI Usage Statement
|
|
179
197
|
|
|
180
|
-
This software was developed without the use of AI-generated code. The
|
|
198
|
+
This software was developed without the use of AI-generated code. The GitHub Copilot Chat in Microsoft Visual Studio Code using the GPT-5 mini model (by OpenAI) was used solely to assist in drafting and refining docstrings for documentation. The corresponding guidelines and constraints defined by the author are documented in `AGENTS-docstrings.md` in the [public GitHub repository](https://github.com/geowieland/diffindiff_official).
|
|
181
199
|
|
|
182
200
|
|
|
183
|
-
## What's new (v2.5.
|
|
201
|
+
## What's new (v2.5.7)
|
|
184
202
|
|
|
185
203
|
- Bugfixes
|
|
186
|
-
-
|
|
187
|
-
-
|
|
204
|
+
- Catching RecursionError when model formulas include too many variables
|
|
205
|
+
- Checking input data frames for duplicated columns
|
|
206
|
+
- Avoiding treatment group deviation in summary when treatment/control groups are identical
|
|
@@ -21,7 +21,7 @@ A case study that utilizes the diffindiff library is available on [arXiv](https:
|
|
|
21
21
|
|
|
22
22
|
If you use this software, please cite:
|
|
23
23
|
|
|
24
|
-
Wieland, T. (2026). diffindiff: A Python library for convenient difference-in-differences analyses (Version 2.5.
|
|
24
|
+
Wieland, T. (2026). diffindiff: A Python library for convenient difference-in-differences analyses (Version 2.5.7) [Computer software]. Zenodo. https://doi.org/10.5281/zenodo.18656820
|
|
25
25
|
|
|
26
26
|
|
|
27
27
|
## Installation
|
|
@@ -169,11 +169,12 @@ See the /tests directory for usage examples of most of the included functions.
|
|
|
169
169
|
|
|
170
170
|
## AI Usage Statement
|
|
171
171
|
|
|
172
|
-
This software was developed without the use of AI-generated code. The
|
|
172
|
+
This software was developed without the use of AI-generated code. The GitHub Copilot Chat in Microsoft Visual Studio Code using the GPT-5 mini model (by OpenAI) was used solely to assist in drafting and refining docstrings for documentation. The corresponding guidelines and constraints defined by the author are documented in `AGENTS-docstrings.md` in the [public GitHub repository](https://github.com/geowieland/diffindiff_official).
|
|
173
173
|
|
|
174
174
|
|
|
175
|
-
## What's new (v2.5.
|
|
175
|
+
## What's new (v2.5.7)
|
|
176
176
|
|
|
177
177
|
- Bugfixes
|
|
178
|
-
-
|
|
179
|
-
-
|
|
178
|
+
- Catching RecursionError when model formulas include too many variables
|
|
179
|
+
- Checking input data frames for duplicated columns
|
|
180
|
+
- Avoiding treatment group deviation in summary when treatment/control groups are identical
|
|
@@ -4,15 +4,15 @@
|
|
|
4
4
|
# Author: Thomas Wieland
|
|
5
5
|
# ORCID: 0000-0001-5168-9846
|
|
6
6
|
# mail: geowieland@googlemail.com
|
|
7
|
-
# Version: 1.0.
|
|
8
|
-
# Last update: 2026-
|
|
7
|
+
# Version: 1.0.28
|
|
8
|
+
# Last update: 2026-10-10 09:58
|
|
9
9
|
# Copyright (c) 2025-2026 Thomas Wieland
|
|
10
10
|
#-----------------------------------------------------------------------
|
|
11
11
|
|
|
12
12
|
# Basic config:
|
|
13
13
|
|
|
14
14
|
PACKAGE_NAME = "diffindiff"
|
|
15
|
-
PACKAGE_VERSION = "2.5.
|
|
15
|
+
PACKAGE_VERSION = "2.5.7"
|
|
16
16
|
|
|
17
17
|
VERBOSE = False
|
|
18
18
|
|
|
@@ -4,8 +4,8 @@
|
|
|
4
4
|
# Author: Thomas Wieland
|
|
5
5
|
# ORCID: 0000-0001-5168-9846
|
|
6
6
|
# mail: geowieland@googlemail.com
|
|
7
|
-
# Version: 1.2.
|
|
8
|
-
# Last update: 2026-
|
|
7
|
+
# Version: 1.2.5
|
|
8
|
+
# Last update: 2026-10-09 23:04
|
|
9
9
|
# Copyright (c) 2025-2026 Thomas Wieland
|
|
10
10
|
#-----------------------------------------------------------------------
|
|
11
11
|
|
|
@@ -1085,33 +1085,47 @@ def ols_fit(
|
|
|
1085
1085
|
>>> ols_fit(df, 'y ~ x')
|
|
1086
1086
|
"""
|
|
1087
1087
|
|
|
1088
|
+
ols_model = None
|
|
1089
|
+
ols_coef = None
|
|
1090
|
+
ols_coef_se = None
|
|
1091
|
+
ols_coef_t = None
|
|
1092
|
+
ols_coef_p = None
|
|
1093
|
+
ols_coef_ci = None
|
|
1094
|
+
ols_predictions = None
|
|
1095
|
+
|
|
1088
1096
|
if verbose:
|
|
1089
1097
|
print("Estimating model via Ordinary Least Squares", end = " ... ")
|
|
1090
1098
|
|
|
1091
|
-
|
|
1099
|
+
try:
|
|
1092
1100
|
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1101
|
+
if cluster_SE_by is not None:
|
|
1102
|
+
|
|
1103
|
+
ols_model = ols(
|
|
1104
|
+
formula,
|
|
1105
|
+
data=data
|
|
1106
|
+
).fit(
|
|
1107
|
+
cov_type="cluster",
|
|
1108
|
+
cov_kwds={"groups": data[cluster_SE_by]} if cluster_SE_by else None
|
|
1109
|
+
)
|
|
1100
1110
|
|
|
1101
|
-
|
|
1111
|
+
else:
|
|
1112
|
+
|
|
1113
|
+
ols_model = ols(formula, data).fit()
|
|
1114
|
+
|
|
1115
|
+
ols_coef = ols_model.params
|
|
1116
|
+
ols_coef_se = ols_model.bse
|
|
1117
|
+
ols_coef_t = ols_model.tvalues
|
|
1118
|
+
ols_coef_p = ols_model.pvalues
|
|
1119
|
+
ols_coef_ci = ols_model.conf_int(alpha = confint_alpha)
|
|
1102
1120
|
|
|
1103
|
-
|
|
1121
|
+
ols_predictions = ols_model.get_prediction(data).summary_frame(alpha=confint_alpha)
|
|
1104
1122
|
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
ols_predictions = ols_model.get_prediction(data).summary_frame(alpha=confint_alpha)
|
|
1112
|
-
|
|
1113
|
-
if verbose:
|
|
1114
|
-
print("OK")
|
|
1123
|
+
if verbose:
|
|
1124
|
+
print("OK")
|
|
1125
|
+
|
|
1126
|
+
except RecursionError as e:
|
|
1127
|
+
|
|
1128
|
+
raise ValueError(f"OLS estimation has induced a RecursionError: {str(e)}. There are likely too many variables in the model. Try sys.setrecursionlimit() or diff-in-diff analysis with demean = True")
|
|
1115
1129
|
|
|
1116
1130
|
return [
|
|
1117
1131
|
ols_model,
|
|
@@ -1160,26 +1174,32 @@ def ml_fit(
|
|
|
1160
1174
|
|
|
1161
1175
|
if verbose:
|
|
1162
1176
|
print("Estimating model via Maximum Likelihood", end = " ... ")
|
|
1163
|
-
|
|
1164
|
-
y, X = dmatrices(
|
|
1165
|
-
formula,
|
|
1166
|
-
data=data,
|
|
1167
|
-
return_type = "dataframe"
|
|
1168
|
-
)
|
|
1169
1177
|
|
|
1170
|
-
|
|
1171
|
-
y,
|
|
1172
|
-
X,
|
|
1173
|
-
family=family(link=link)
|
|
1174
|
-
).fit()
|
|
1175
|
-
|
|
1176
|
-
mle_coef = mle_model.params
|
|
1177
|
-
mle_coef_se = mle_model.bse
|
|
1178
|
-
mle_coef_z = mle_model.tvalues
|
|
1179
|
-
mle_coef_p = mle_model.pvalues
|
|
1180
|
-
mle_coef_ci = mle_model.conf_int(alpha=confint_alpha)
|
|
1178
|
+
try:
|
|
1181
1179
|
|
|
1182
|
-
|
|
1180
|
+
y, X = dmatrices(
|
|
1181
|
+
formula,
|
|
1182
|
+
data=data,
|
|
1183
|
+
return_type = "dataframe"
|
|
1184
|
+
)
|
|
1185
|
+
|
|
1186
|
+
mle_model = sm.GLM(
|
|
1187
|
+
y,
|
|
1188
|
+
X,
|
|
1189
|
+
family=family(link=link)
|
|
1190
|
+
).fit()
|
|
1191
|
+
|
|
1192
|
+
mle_coef = mle_model.params
|
|
1193
|
+
mle_coef_se = mle_model.bse
|
|
1194
|
+
mle_coef_z = mle_model.tvalues
|
|
1195
|
+
mle_coef_p = mle_model.pvalues
|
|
1196
|
+
mle_coef_ci = mle_model.conf_int(alpha=confint_alpha)
|
|
1197
|
+
|
|
1198
|
+
mle_predictions = mle_model.get_prediction(X).summary_frame(alpha=confint_alpha)
|
|
1199
|
+
|
|
1200
|
+
except RecursionError as e:
|
|
1201
|
+
|
|
1202
|
+
raise ValueError(f"ML estimation has induced a RecursionError: {str(e)}. There are likely too many variables in the model. Try sys.setrecursionlimit() or diff-in-diff analysis with demean = True")
|
|
1183
1203
|
|
|
1184
1204
|
if verbose:
|
|
1185
1205
|
print("OK")
|
|
@@ -1333,7 +1353,7 @@ def extract_model_results(
|
|
|
1333
1353
|
}
|
|
1334
1354
|
|
|
1335
1355
|
if config.REMOVE_DUPLICATES_FROM_RESULTS_DICT:
|
|
1336
|
-
beta_1 = remove_duplicates_from_dict(beta_1)
|
|
1356
|
+
beta_1 = remove_duplicates_from_dict(beta_1, ignore_key=config.OLS_MODEL_RESULTS["coef_name"]["model_results_key"])
|
|
1337
1357
|
|
|
1338
1358
|
model_results[config.EFFECTS_TYPES["beta_1"]["model_results_key"]] = beta_1
|
|
1339
1359
|
|
|
@@ -1800,7 +1820,8 @@ def create_timestamp(function) -> dict:
|
|
|
1800
1820
|
return timestamp_dict
|
|
1801
1821
|
|
|
1802
1822
|
def remove_duplicates_from_dict(
|
|
1803
|
-
any_dict: dict
|
|
1823
|
+
any_dict: dict,
|
|
1824
|
+
ignore_key = None
|
|
1804
1825
|
) -> dict:
|
|
1805
1826
|
|
|
1806
1827
|
"""
|
|
@@ -1822,7 +1843,13 @@ def remove_duplicates_from_dict(
|
|
|
1822
1843
|
|
|
1823
1844
|
for key, value in any_dict.items():
|
|
1824
1845
|
|
|
1825
|
-
|
|
1846
|
+
if ignore_key is not None:
|
|
1847
|
+
identifier = tuple(sorted(
|
|
1848
|
+
(k, v) for k, v in value.items()
|
|
1849
|
+
if k != ignore_key
|
|
1850
|
+
))
|
|
1851
|
+
else:
|
|
1852
|
+
identifier = tuple(sorted(value.items()))
|
|
1826
1853
|
|
|
1827
1854
|
if identifier not in seen:
|
|
1828
1855
|
seen.add(identifier)
|
|
@@ -4,8 +4,8 @@
|
|
|
4
4
|
# Author: Thomas Wieland
|
|
5
5
|
# ORCID: 0000-0001-5168-9846
|
|
6
6
|
# mail: geowieland@googlemail.com
|
|
7
|
-
# Version: 2.2.
|
|
8
|
-
# Last update: 2026-
|
|
7
|
+
# Version: 2.2.7
|
|
8
|
+
# Last update: 2026-10-10 09:57
|
|
9
9
|
# Copyright (c) 2025-2026 Thomas Wieland
|
|
10
10
|
#-----------------------------------------------------------------------
|
|
11
11
|
|
|
@@ -22,11 +22,20 @@ from sklearn.svm import SVR
|
|
|
22
22
|
from sklearn.neighbors import KNeighborsRegressor
|
|
23
23
|
from sklearn.pipeline import Pipeline
|
|
24
24
|
from sklearn.preprocessing import StandardScaler
|
|
25
|
-
from xgboost import XGBRegressor
|
|
26
|
-
from lightgbm import LGBMRegressor
|
|
27
25
|
from sklearn.linear_model import LinearRegression
|
|
28
26
|
from sklearn.model_selection import train_test_split
|
|
29
27
|
from sklearn.neural_network import MLPRegressor
|
|
28
|
+
|
|
29
|
+
try:
|
|
30
|
+
from xgboost import XGBRegressor
|
|
31
|
+
except ImportError:
|
|
32
|
+
XGBRegressor = None
|
|
33
|
+
try:
|
|
34
|
+
from lightgbm import LGBMRegressor
|
|
35
|
+
except ImportError:
|
|
36
|
+
LGBMRegressor = None
|
|
37
|
+
|
|
38
|
+
|
|
30
39
|
import diffindiff.config as config
|
|
31
40
|
|
|
32
41
|
|
|
@@ -37,7 +46,8 @@ def check_columns(
|
|
|
37
46
|
):
|
|
38
47
|
|
|
39
48
|
"""
|
|
40
|
-
Check that the given columns exist in a DataFrame
|
|
49
|
+
Check that the given columns exist in a DataFrame
|
|
50
|
+
and whether there are duplicated column names.
|
|
41
51
|
|
|
42
52
|
Parameters
|
|
43
53
|
----------
|
|
@@ -56,6 +66,8 @@ def check_columns(
|
|
|
56
66
|
------
|
|
57
67
|
KeyError
|
|
58
68
|
If any column from ``columns`` is missing in ``df``.
|
|
69
|
+
KeyError
|
|
70
|
+
If any column from ``columns`` is duplicated.
|
|
59
71
|
|
|
60
72
|
Examples
|
|
61
73
|
--------
|
|
@@ -77,6 +89,17 @@ def check_columns(
|
|
|
77
89
|
if missing_columns:
|
|
78
90
|
raise KeyError(f"Data do not contain column(s): {', '.join(missing_columns)}")
|
|
79
91
|
|
|
92
|
+
if verbose:
|
|
93
|
+
print("Checking whether columns are duplicated in data frame", end = " ... ")
|
|
94
|
+
|
|
95
|
+
cols_duplicated = df.columns[df.columns.duplicated()].tolist()
|
|
96
|
+
|
|
97
|
+
if verbose:
|
|
98
|
+
print("OK")
|
|
99
|
+
|
|
100
|
+
if len(cols_duplicated) > 0:
|
|
101
|
+
raise KeyError(f"Data contain duplicated relevant columns: {', '.join(cols_duplicated)}")
|
|
102
|
+
|
|
80
103
|
def is_numeric(
|
|
81
104
|
df: pd.DataFrame,
|
|
82
105
|
columns: list,
|
|
@@ -826,83 +849,105 @@ def is_parallel(
|
|
|
826
849
|
treatment_col = treatment_col,
|
|
827
850
|
verbose = False
|
|
828
851
|
)
|
|
829
|
-
|
|
830
|
-
if verbose:
|
|
831
|
-
print(f"Testing outcome '{outcome_col}' for parallel time trends", end = " ... ")
|
|
832
|
-
|
|
833
|
-
if pre_post or not modeldata_isnotreatment:
|
|
834
|
-
parallel = "not_tested"
|
|
835
|
-
test_ols_model = None
|
|
836
|
-
|
|
837
|
-
treatment_group = modeldata_isnotreatment[1]
|
|
838
852
|
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
if len(data[(data[unit_col].isin(treatment_group)) & (data[treatment_col] > 0)]) > 0:
|
|
842
|
-
|
|
843
|
-
first_day_of_treatment = min(data[(data[unit_col].isin(treatment_group)) & (data[treatment_col] > 0)][time_col])
|
|
844
|
-
|
|
845
|
-
data_test = data[data[time_col] < first_day_of_treatment].copy()
|
|
846
|
-
data_test[config.TG_COL] = 0
|
|
847
|
-
data_test.loc[data_test[unit_col].isin(treatment_group), config.TG_COL] = 1
|
|
848
|
-
|
|
849
|
-
if config.TIME_COUNTER_COL not in data_test.columns:
|
|
850
|
-
data_test = date_counter(
|
|
851
|
-
df = data_test,
|
|
852
|
-
date_col = time_col,
|
|
853
|
-
new_col = config.TIME_COUNTER_COL,
|
|
854
|
-
verbose = False
|
|
855
|
-
)
|
|
856
|
-
data_test[f"{config.TG_COL}_x_{config.TIME_COL}"] = data_test[config.TG_COL]*data_test[config.TIME_COUNTER_COL]
|
|
853
|
+
parallel = "not_tested"
|
|
854
|
+
test_ols_model = None
|
|
857
855
|
|
|
858
|
-
|
|
859
|
-
coef_TG_x_t_p = test_ols_model.pvalues[f"{config.TG_COL}_x_{config.TIME_COL}"]
|
|
856
|
+
if not pre_post:
|
|
860
857
|
|
|
861
|
-
|
|
862
|
-
parallel = False
|
|
863
|
-
else:
|
|
864
|
-
parallel = True
|
|
858
|
+
no_pre_period = False
|
|
865
859
|
|
|
866
|
-
|
|
867
|
-
parallel = "
|
|
868
|
-
test_ols_model = None
|
|
860
|
+
if verbose:
|
|
861
|
+
print(f"Testing outcome '{outcome_col}' for parallel time trends", end = " ... ")
|
|
869
862
|
|
|
870
|
-
|
|
863
|
+
treatment_group = modeldata_isnotreatment[1]
|
|
871
864
|
|
|
872
|
-
if
|
|
873
|
-
|
|
874
|
-
first_day_of_treatment = min(data[(data[unit_col].isin(treatment_group)) & (data[treatment_col] == 1)][time_col])
|
|
865
|
+
if config.ACCEPT_CONTINUOUS_TREATMENTS:
|
|
875
866
|
|
|
876
|
-
|
|
877
|
-
data_test[config.TG_COL] = 0
|
|
878
|
-
data_test.loc[data_test[unit_col].isin(treatment_group), config.TG_COL] = 1
|
|
867
|
+
if len(data[(data[unit_col].isin(treatment_group)) & (data[treatment_col] > 0)]) > 0:
|
|
879
868
|
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
869
|
+
first_day_of_treatment = min(data[(data[unit_col].isin(treatment_group)) & (data[treatment_col] > 0)][time_col])
|
|
870
|
+
|
|
871
|
+
data_test = data[data[time_col] < first_day_of_treatment].copy()
|
|
872
|
+
data_test[config.TG_COL] = 0
|
|
873
|
+
data_test.loc[data_test[unit_col].isin(treatment_group), config.TG_COL] = 1
|
|
874
|
+
|
|
875
|
+
if config.TIME_COUNTER_COL not in data_test.columns:
|
|
876
|
+
data_test = date_counter(
|
|
877
|
+
df = data_test,
|
|
878
|
+
date_col = time_col,
|
|
879
|
+
new_col = config.TIME_COUNTER_COL,
|
|
880
|
+
verbose = False
|
|
881
|
+
)
|
|
882
|
+
data_test[f"{config.TG_COL}_x_{config.TIME_COL}"] = data_test[config.TG_COL]*data_test[config.TIME_COUNTER_COL]
|
|
883
|
+
|
|
884
|
+
if len(data_test) > 0:
|
|
888
885
|
|
|
889
|
-
|
|
890
|
-
|
|
886
|
+
test_ols_model = ols(f'{outcome_col} ~ {config.TG_COL} + {config.TIME_COUNTER_COL} + {config.TG_COL}_x_{config.TIME_COL}', data = data_test).fit()
|
|
887
|
+
coef_TG_x_t_p = test_ols_model.pvalues[f"{config.TG_COL}_x_{config.TIME_COL}"]
|
|
891
888
|
|
|
892
|
-
|
|
893
|
-
|
|
889
|
+
if coef_TG_x_t_p < alpha:
|
|
890
|
+
parallel = False
|
|
891
|
+
else:
|
|
892
|
+
parallel = True
|
|
893
|
+
|
|
894
|
+
else:
|
|
895
|
+
no_pre_period = True
|
|
896
|
+
|
|
894
897
|
else:
|
|
895
|
-
parallel =
|
|
898
|
+
parallel = "not_tested"
|
|
899
|
+
test_ols_model = None
|
|
896
900
|
|
|
897
901
|
else:
|
|
898
|
-
parallel = "not_tested"
|
|
899
|
-
test_ols_model = None
|
|
900
|
-
|
|
901
|
-
if verbose:
|
|
902
|
-
print("OK")
|
|
903
902
|
|
|
904
|
-
|
|
905
|
-
|
|
903
|
+
if len(data[(data[unit_col].isin(treatment_group)) & (data[treatment_col] == 1)]) > 0:
|
|
904
|
+
|
|
905
|
+
first_day_of_treatment = min(data[(data[unit_col].isin(treatment_group)) & (data[treatment_col] == 1)][time_col])
|
|
906
|
+
|
|
907
|
+
data_test = data[data[time_col] < first_day_of_treatment].copy()
|
|
908
|
+
data_test[config.TG_COL] = 0
|
|
909
|
+
data_test.loc[data_test[unit_col].isin(treatment_group), config.TG_COL] = 1
|
|
910
|
+
|
|
911
|
+
if config.TIME_COUNTER_COL not in data_test.columns:
|
|
912
|
+
data_test = date_counter(
|
|
913
|
+
df = data_test,
|
|
914
|
+
date_col = time_col,
|
|
915
|
+
new_col = config.TIME_COUNTER_COL,
|
|
916
|
+
verbose = False
|
|
917
|
+
)
|
|
918
|
+
data_test[f"{config.TG_COL}_x_{config.TIME_COL}"] = data_test[config.TG_COL]*data_test[config.TIME_COUNTER_COL]
|
|
919
|
+
|
|
920
|
+
if len(data_test) > 0:
|
|
921
|
+
|
|
922
|
+
test_ols_model = ols(f'{outcome_col} ~ {config.TG_COL} + {config.TIME_COUNTER_COL} + {config.TG_COL}_x_{config.TIME_COL}', data = data_test).fit()
|
|
923
|
+
coef_TG_x_t_p = test_ols_model.pvalues[f"{config.TG_COL}_x_{config.TIME_COL}"]
|
|
924
|
+
|
|
925
|
+
if coef_TG_x_t_p < alpha:
|
|
926
|
+
parallel = False
|
|
927
|
+
else:
|
|
928
|
+
parallel = True
|
|
929
|
+
|
|
930
|
+
else:
|
|
931
|
+
no_pre_period = True
|
|
932
|
+
|
|
933
|
+
else:
|
|
934
|
+
parallel = "not_tested"
|
|
935
|
+
test_ols_model = None
|
|
936
|
+
|
|
937
|
+
if verbose:
|
|
938
|
+
print("OK")
|
|
939
|
+
|
|
940
|
+
if not pre_post:
|
|
941
|
+
|
|
942
|
+
if parallel == "not_tested":
|
|
943
|
+
print("WARNING: Data could not be tested for parallel time trends.")
|
|
944
|
+
if no_pre_period:
|
|
945
|
+
print("WARNING: Data could not be tested for parallel time trends because there is no pre-treatment period.")
|
|
946
|
+
|
|
947
|
+
else:
|
|
948
|
+
|
|
949
|
+
if verbose:
|
|
950
|
+
print("NOTE: Data is pre-post data and parallel trends are not tested.")
|
|
906
951
|
|
|
907
952
|
return [
|
|
908
953
|
parallel,
|
|
@@ -1549,18 +1594,28 @@ def model_wrapper(
|
|
|
1549
1594
|
model = SVR(kernel=svr_kernel)
|
|
1550
1595
|
|
|
1551
1596
|
elif model_type == "xgb":
|
|
1552
|
-
|
|
1553
|
-
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1597
|
+
|
|
1598
|
+
if XGBRegressor is not None:
|
|
1599
|
+
model = XGBRegressor(
|
|
1600
|
+
learning_rate = xgb_learning_rate,
|
|
1601
|
+
n_estimators = gb_iterations,
|
|
1602
|
+
random_state = random_state
|
|
1603
|
+
)
|
|
1604
|
+
else:
|
|
1605
|
+
model_estimation_error = True
|
|
1606
|
+
model_estimation_error_text = "XGBRegressor is not available. Please install xgboost to use this model type."
|
|
1557
1607
|
|
|
1558
1608
|
elif model_type == "lgbm":
|
|
1559
|
-
|
|
1560
|
-
|
|
1561
|
-
|
|
1562
|
-
|
|
1563
|
-
|
|
1609
|
+
|
|
1610
|
+
if LGBMRegressor is not None:
|
|
1611
|
+
model = LGBMRegressor(
|
|
1612
|
+
learning_rate = lgbm_learning_rate,
|
|
1613
|
+
n_estimators = gb_iterations,
|
|
1614
|
+
random_state = random_state
|
|
1615
|
+
)
|
|
1616
|
+
else:
|
|
1617
|
+
model_estimation_error = True
|
|
1618
|
+
model_estimation_error_text = "LGBMRegressor is not available. Please install lightgbm to use this model type."
|
|
1564
1619
|
|
|
1565
1620
|
elif model_type == "mlp":
|
|
1566
1621
|
model = Pipeline(
|
|
@@ -1,10 +1,28 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: diffindiff
|
|
3
|
-
Version: 2.5.
|
|
3
|
+
Version: 2.5.7
|
|
4
4
|
Summary: diffindiff: Python library for convenient Difference-in-Differences analyses
|
|
5
5
|
Author: Thomas Wieland
|
|
6
6
|
Author-email: geowieland@googlemail.com
|
|
7
7
|
Description-Content-Type: text/markdown
|
|
8
|
+
Requires-Dist: pandas
|
|
9
|
+
Requires-Dist: numpy
|
|
10
|
+
Requires-Dist: statsmodels>=0.14.5
|
|
11
|
+
Requires-Dist: scipy>=1.17
|
|
12
|
+
Requires-Dist: scikit-learn
|
|
13
|
+
Requires-Dist: openpyxl
|
|
14
|
+
Requires-Dist: matplotlib
|
|
15
|
+
Requires-Dist: patsy
|
|
16
|
+
Provides-Extra: optional
|
|
17
|
+
Requires-Dist: lightgbm; extra == "optional"
|
|
18
|
+
Requires-Dist: xgboost; extra == "optional"
|
|
19
|
+
Dynamic: author
|
|
20
|
+
Dynamic: author-email
|
|
21
|
+
Dynamic: description
|
|
22
|
+
Dynamic: description-content-type
|
|
23
|
+
Dynamic: provides-extra
|
|
24
|
+
Dynamic: requires-dist
|
|
25
|
+
Dynamic: summary
|
|
8
26
|
|
|
9
27
|
# diffindiff: Python library for convenient Difference-in-Differences analyses
|
|
10
28
|
|
|
@@ -29,7 +47,7 @@ A case study that utilizes the diffindiff library is available on [arXiv](https:
|
|
|
29
47
|
|
|
30
48
|
If you use this software, please cite:
|
|
31
49
|
|
|
32
|
-
Wieland, T. (2026). diffindiff: A Python library for convenient difference-in-differences analyses (Version 2.5.
|
|
50
|
+
Wieland, T. (2026). diffindiff: A Python library for convenient difference-in-differences analyses (Version 2.5.7) [Computer software]. Zenodo. https://doi.org/10.5281/zenodo.18656820
|
|
33
51
|
|
|
34
52
|
|
|
35
53
|
## Installation
|
|
@@ -177,11 +195,12 @@ See the /tests directory for usage examples of most of the included functions.
|
|
|
177
195
|
|
|
178
196
|
## AI Usage Statement
|
|
179
197
|
|
|
180
|
-
This software was developed without the use of AI-generated code. The
|
|
198
|
+
This software was developed without the use of AI-generated code. The GitHub Copilot Chat in Microsoft Visual Studio Code using the GPT-5 mini model (by OpenAI) was used solely to assist in drafting and refining docstrings for documentation. The corresponding guidelines and constraints defined by the author are documented in `AGENTS-docstrings.md` in the [public GitHub repository](https://github.com/geowieland/diffindiff_official).
|
|
181
199
|
|
|
182
200
|
|
|
183
|
-
## What's new (v2.5.
|
|
201
|
+
## What's new (v2.5.7)
|
|
184
202
|
|
|
185
203
|
- Bugfixes
|
|
186
|
-
-
|
|
187
|
-
-
|
|
204
|
+
- Catching RecursionError when model formulas include too many variables
|
|
205
|
+
- Checking input data frames for duplicated columns
|
|
206
|
+
- Avoiding treatment group deviation in summary when treatment/control groups are identical
|
|
@@ -7,7 +7,7 @@ def read_README():
|
|
|
7
7
|
|
|
8
8
|
setup(
|
|
9
9
|
name='diffindiff',
|
|
10
|
-
version='2.5.
|
|
10
|
+
version='2.5.7',
|
|
11
11
|
description='diffindiff: Python library for convenient Difference-in-Differences analyses',
|
|
12
12
|
packages=find_packages(include=["diffindiff", "diffindiff.tests"]),
|
|
13
13
|
include_package_data=True,
|
|
@@ -25,11 +25,15 @@ setup(
|
|
|
25
25
|
'statsmodels>=0.14.5',
|
|
26
26
|
'scipy>=1.17',
|
|
27
27
|
'scikit-learn',
|
|
28
|
-
'xgboost',
|
|
29
|
-
'lightgbm',
|
|
30
28
|
'openpyxl',
|
|
31
29
|
'matplotlib',
|
|
32
30
|
'patsy',
|
|
33
31
|
],
|
|
32
|
+
extras_require={
|
|
33
|
+
"optional": [
|
|
34
|
+
'lightgbm',
|
|
35
|
+
'xgboost',
|
|
36
|
+
]
|
|
37
|
+
},
|
|
34
38
|
test_suite='tests',
|
|
35
39
|
)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|