diff-diff 1.3.0__py3-none-any.whl → 1.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diff_diff/__init__.py +1 -1
- diff_diff/results.py +14 -1
- diff_diff/synthetic_did.py +185 -11
- {diff_diff-1.3.0.dist-info → diff_diff-1.3.1.dist-info}/METADATA +2 -1
- {diff_diff-1.3.0.dist-info → diff_diff-1.3.1.dist-info}/RECORD +7 -7
- {diff_diff-1.3.0.dist-info → diff_diff-1.3.1.dist-info}/WHEEL +0 -0
- {diff_diff-1.3.0.dist-info → diff_diff-1.3.1.dist-info}/top_level.txt +0 -0
diff_diff/__init__.py
CHANGED
diff_diff/results.py
CHANGED
|
@@ -507,6 +507,8 @@ class SyntheticDiDResults:
|
|
|
507
507
|
List of pre-treatment period identifiers.
|
|
508
508
|
post_periods : list
|
|
509
509
|
List of post-treatment period identifiers.
|
|
510
|
+
variance_method : str
|
|
511
|
+
Method used for variance estimation: "bootstrap" or "placebo".
|
|
510
512
|
"""
|
|
511
513
|
|
|
512
514
|
att: float
|
|
@@ -522,9 +524,11 @@ class SyntheticDiDResults:
|
|
|
522
524
|
pre_periods: List[Any]
|
|
523
525
|
post_periods: List[Any]
|
|
524
526
|
alpha: float = 0.05
|
|
527
|
+
variance_method: str = field(default="bootstrap")
|
|
525
528
|
lambda_reg: Optional[float] = field(default=None)
|
|
526
529
|
pre_treatment_fit: Optional[float] = field(default=None)
|
|
527
530
|
placebo_effects: Optional[np.ndarray] = field(default=None)
|
|
531
|
+
n_bootstrap: Optional[int] = field(default=None)
|
|
528
532
|
|
|
529
533
|
def __repr__(self) -> str:
|
|
530
534
|
"""Concise string representation."""
|
|
@@ -571,6 +575,11 @@ class SyntheticDiDResults:
|
|
|
571
575
|
if self.pre_treatment_fit is not None:
|
|
572
576
|
lines.append(f"{'Pre-treatment fit (RMSE):':<25} {self.pre_treatment_fit:>10.4f}")
|
|
573
577
|
|
|
578
|
+
# Variance method info
|
|
579
|
+
lines.append(f"{'Variance method:':<25} {self.variance_method:>10}")
|
|
580
|
+
if self.variance_method == "bootstrap" and self.n_bootstrap is not None:
|
|
581
|
+
lines.append(f"{'Bootstrap replications:':<25} {self.n_bootstrap:>10}")
|
|
582
|
+
|
|
574
583
|
lines.extend([
|
|
575
584
|
"",
|
|
576
585
|
"-" * 75,
|
|
@@ -624,7 +633,7 @@ class SyntheticDiDResults:
|
|
|
624
633
|
Dict[str, Any]
|
|
625
634
|
Dictionary containing all estimation results.
|
|
626
635
|
"""
|
|
627
|
-
|
|
636
|
+
result = {
|
|
628
637
|
"att": self.att,
|
|
629
638
|
"se": self.se,
|
|
630
639
|
"t_stat": self.t_stat,
|
|
@@ -636,9 +645,13 @@ class SyntheticDiDResults:
|
|
|
636
645
|
"n_control": self.n_control,
|
|
637
646
|
"n_pre_periods": len(self.pre_periods),
|
|
638
647
|
"n_post_periods": len(self.post_periods),
|
|
648
|
+
"variance_method": self.variance_method,
|
|
639
649
|
"lambda_reg": self.lambda_reg,
|
|
640
650
|
"pre_treatment_fit": self.pre_treatment_fit,
|
|
641
651
|
}
|
|
652
|
+
if self.n_bootstrap is not None:
|
|
653
|
+
result["n_bootstrap"] = self.n_bootstrap
|
|
654
|
+
return result
|
|
642
655
|
|
|
643
656
|
def to_dataframe(self) -> pd.DataFrame:
|
|
644
657
|
"""
|
diff_diff/synthetic_did.py
CHANGED
|
@@ -14,7 +14,6 @@ from diff_diff.results import SyntheticDiDResults
|
|
|
14
14
|
from diff_diff.utils import (
|
|
15
15
|
compute_confidence_interval,
|
|
16
16
|
compute_p_value,
|
|
17
|
-
compute_placebo_effects,
|
|
18
17
|
compute_sdid_estimator,
|
|
19
18
|
compute_synthetic_weights,
|
|
20
19
|
compute_time_weights,
|
|
@@ -46,9 +45,17 @@ class SyntheticDiD(DifferenceInDifferences):
|
|
|
46
45
|
time weights (closer to standard DiD).
|
|
47
46
|
alpha : float, default=0.05
|
|
48
47
|
Significance level for confidence intervals.
|
|
48
|
+
variance_method : str, default="bootstrap"
|
|
49
|
+
Method for variance estimation:
|
|
50
|
+
- "bootstrap": Block bootstrap at unit level (default)
|
|
51
|
+
- "placebo": Placebo-based variance matching R's synthdid::vcov(method="placebo").
|
|
52
|
+
Implements Algorithm 4 from Arkhangelsky et al. (2021): randomly permutes
|
|
53
|
+
control units, designates N₁ as pseudo-treated, renormalizes original
|
|
54
|
+
weights for remaining pseudo-controls, and computes SDID estimate.
|
|
49
55
|
n_bootstrap : int, default=200
|
|
50
|
-
Number of
|
|
51
|
-
|
|
56
|
+
Number of replications for variance estimation. Used for both:
|
|
57
|
+
- Bootstrap: Number of bootstrap samples
|
|
58
|
+
- Placebo: Number of random permutations (matches R's `replications` argument)
|
|
52
59
|
seed : int, optional
|
|
53
60
|
Random seed for reproducibility. If None (default), results
|
|
54
61
|
will vary between runs.
|
|
@@ -125,15 +132,25 @@ class SyntheticDiD(DifferenceInDifferences):
|
|
|
125
132
|
lambda_reg: float = 0.0,
|
|
126
133
|
zeta: float = 1.0,
|
|
127
134
|
alpha: float = 0.05,
|
|
135
|
+
variance_method: str = "bootstrap",
|
|
128
136
|
n_bootstrap: int = 200,
|
|
129
137
|
seed: Optional[int] = None
|
|
130
138
|
):
|
|
131
139
|
super().__init__(robust=True, cluster=None, alpha=alpha)
|
|
132
140
|
self.lambda_reg = lambda_reg
|
|
133
141
|
self.zeta = zeta
|
|
142
|
+
self.variance_method = variance_method
|
|
134
143
|
self.n_bootstrap = n_bootstrap
|
|
135
144
|
self.seed = seed
|
|
136
145
|
|
|
146
|
+
# Validate variance_method
|
|
147
|
+
valid_methods = ("bootstrap", "placebo")
|
|
148
|
+
if variance_method not in valid_methods:
|
|
149
|
+
raise ValueError(
|
|
150
|
+
f"variance_method must be one of {valid_methods}, "
|
|
151
|
+
f"got '{variance_method}'"
|
|
152
|
+
)
|
|
153
|
+
|
|
137
154
|
self._unit_weights = None
|
|
138
155
|
self._time_weights = None
|
|
139
156
|
|
|
@@ -285,23 +302,27 @@ class SyntheticDiD(DifferenceInDifferences):
|
|
|
285
302
|
synthetic_pre = Y_pre_control @ unit_weights
|
|
286
303
|
pre_fit_rmse = np.sqrt(np.mean((Y_pre_treated_mean - synthetic_pre) ** 2))
|
|
287
304
|
|
|
288
|
-
# Compute standard errors
|
|
289
|
-
if self.
|
|
290
|
-
se,
|
|
305
|
+
# Compute standard errors based on variance_method
|
|
306
|
+
if self.variance_method == "bootstrap":
|
|
307
|
+
se, bootstrap_estimates = self._bootstrap_se(
|
|
291
308
|
working_data, outcome, unit, time,
|
|
292
309
|
pre_periods, post_periods, treated_units, control_units
|
|
293
310
|
)
|
|
311
|
+
placebo_effects = bootstrap_estimates
|
|
312
|
+
inference_method = "bootstrap"
|
|
294
313
|
else:
|
|
295
|
-
# Use placebo-based
|
|
296
|
-
placebo_effects =
|
|
314
|
+
# Use placebo-based variance (R's synthdid Algorithm 4)
|
|
315
|
+
se, placebo_effects = self._placebo_variance_se(
|
|
297
316
|
Y_pre_control,
|
|
298
317
|
Y_post_control,
|
|
299
318
|
Y_pre_treated_mean,
|
|
319
|
+
Y_post_treated_mean,
|
|
300
320
|
unit_weights,
|
|
301
321
|
time_weights,
|
|
302
|
-
|
|
322
|
+
n_treated=len(treated_units),
|
|
323
|
+
replications=self.n_bootstrap # Reuse n_bootstrap for replications
|
|
303
324
|
)
|
|
304
|
-
|
|
325
|
+
inference_method = "placebo"
|
|
305
326
|
|
|
306
327
|
# Compute test statistics
|
|
307
328
|
if se > 0:
|
|
@@ -343,9 +364,11 @@ class SyntheticDiD(DifferenceInDifferences):
|
|
|
343
364
|
pre_periods=pre_periods,
|
|
344
365
|
post_periods=post_periods,
|
|
345
366
|
alpha=self.alpha,
|
|
367
|
+
variance_method=inference_method,
|
|
346
368
|
lambda_reg=self.lambda_reg,
|
|
347
369
|
pre_treatment_fit=pre_fit_rmse,
|
|
348
|
-
placebo_effects=placebo_effects if len(placebo_effects) > 0 else None
|
|
370
|
+
placebo_effects=placebo_effects if len(placebo_effects) > 0 else None,
|
|
371
|
+
n_bootstrap=self.n_bootstrap if inference_method == "bootstrap" else None
|
|
349
372
|
)
|
|
350
373
|
|
|
351
374
|
self._unit_weights = unit_weights
|
|
@@ -544,12 +567,163 @@ class SyntheticDiD(DifferenceInDifferences):
|
|
|
544
567
|
|
|
545
568
|
return se, bootstrap_estimates
|
|
546
569
|
|
|
570
|
+
def _placebo_variance_se(
|
|
571
|
+
self,
|
|
572
|
+
Y_pre_control: np.ndarray,
|
|
573
|
+
Y_post_control: np.ndarray,
|
|
574
|
+
Y_pre_treated_mean: np.ndarray,
|
|
575
|
+
Y_post_treated_mean: np.ndarray,
|
|
576
|
+
unit_weights: np.ndarray,
|
|
577
|
+
time_weights: np.ndarray,
|
|
578
|
+
n_treated: int,
|
|
579
|
+
replications: int = 200
|
|
580
|
+
) -> Tuple[float, np.ndarray]:
|
|
581
|
+
"""
|
|
582
|
+
Compute placebo-based variance matching R's synthdid methodology.
|
|
583
|
+
|
|
584
|
+
This implements Algorithm 4 from Arkhangelsky et al. (2021),
|
|
585
|
+
matching R's synthdid::vcov(method = "placebo"):
|
|
586
|
+
|
|
587
|
+
1. Randomly sample N₀ control indices (permutation)
|
|
588
|
+
2. Designate last N₁ as pseudo-treated, first (N₀-N₁) as pseudo-controls
|
|
589
|
+
3. Renormalize original unit weights for pseudo-controls
|
|
590
|
+
4. Compute SDID estimate using renormalized weights
|
|
591
|
+
5. Repeat `replications` times
|
|
592
|
+
6. SE = sqrt((r-1)/r) * sd(estimates)
|
|
593
|
+
|
|
594
|
+
Parameters
|
|
595
|
+
----------
|
|
596
|
+
Y_pre_control : np.ndarray
|
|
597
|
+
Control outcomes in pre-treatment periods, shape (n_pre, n_control).
|
|
598
|
+
Y_post_control : np.ndarray
|
|
599
|
+
Control outcomes in post-treatment periods, shape (n_post, n_control).
|
|
600
|
+
Y_pre_treated_mean : np.ndarray
|
|
601
|
+
Mean treated outcomes in pre-treatment periods, shape (n_pre,).
|
|
602
|
+
Y_post_treated_mean : np.ndarray
|
|
603
|
+
Mean treated outcomes in post-treatment periods, shape (n_post,).
|
|
604
|
+
unit_weights : np.ndarray
|
|
605
|
+
Original unit weights from main estimation, shape (n_control,).
|
|
606
|
+
time_weights : np.ndarray
|
|
607
|
+
Time weights from main estimation, shape (n_pre,).
|
|
608
|
+
n_treated : int
|
|
609
|
+
Number of treated units in the original estimation.
|
|
610
|
+
replications : int, default=200
|
|
611
|
+
Number of placebo replications.
|
|
612
|
+
|
|
613
|
+
Returns
|
|
614
|
+
-------
|
|
615
|
+
tuple
|
|
616
|
+
(se, placebo_effects) where se is the standard error and
|
|
617
|
+
placebo_effects is the array of placebo treatment effects.
|
|
618
|
+
|
|
619
|
+
References
|
|
620
|
+
----------
|
|
621
|
+
Arkhangelsky, D., Athey, S., Hirshberg, D. A., Imbens, G. W., & Wager, S.
|
|
622
|
+
(2021). Synthetic Difference-in-Differences. American Economic Review,
|
|
623
|
+
111(12), 4088-4118. Algorithm 4.
|
|
624
|
+
"""
|
|
625
|
+
rng = np.random.default_rng(self.seed)
|
|
626
|
+
n_pre, n_control = Y_pre_control.shape
|
|
627
|
+
|
|
628
|
+
# Ensure we have enough controls for the split
|
|
629
|
+
n_pseudo_control = n_control - n_treated
|
|
630
|
+
if n_pseudo_control < 1:
|
|
631
|
+
warnings.warn(
|
|
632
|
+
f"Not enough control units ({n_control}) for placebo variance "
|
|
633
|
+
f"estimation with {n_treated} treated units. "
|
|
634
|
+
f"Consider using variance_method='bootstrap'.",
|
|
635
|
+
UserWarning,
|
|
636
|
+
stacklevel=3,
|
|
637
|
+
)
|
|
638
|
+
return 0.0, np.array([])
|
|
639
|
+
|
|
640
|
+
placebo_estimates = []
|
|
641
|
+
|
|
642
|
+
for _ in range(replications):
|
|
643
|
+
try:
|
|
644
|
+
# Random permutation of control indices (Algorithm 4, step 1)
|
|
645
|
+
perm = rng.permutation(n_control)
|
|
646
|
+
|
|
647
|
+
# Split into pseudo-controls and pseudo-treated (step 2)
|
|
648
|
+
pseudo_control_idx = perm[:n_pseudo_control]
|
|
649
|
+
pseudo_treated_idx = perm[n_pseudo_control:]
|
|
650
|
+
|
|
651
|
+
# Renormalize original weights for pseudo-controls (step 3)
|
|
652
|
+
# This keeps the relative importance from the main estimation
|
|
653
|
+
pseudo_weights = unit_weights[pseudo_control_idx]
|
|
654
|
+
weight_sum = pseudo_weights.sum()
|
|
655
|
+
if weight_sum > 0:
|
|
656
|
+
pseudo_weights = pseudo_weights / weight_sum
|
|
657
|
+
else:
|
|
658
|
+
# Fallback to uniform if weights sum to zero
|
|
659
|
+
pseudo_weights = np.ones(n_pseudo_control) / n_pseudo_control
|
|
660
|
+
|
|
661
|
+
# Get pseudo-treated outcomes (mean across pseudo-treated units)
|
|
662
|
+
Y_pre_pseudo_treated = np.mean(
|
|
663
|
+
Y_pre_control[:, pseudo_treated_idx], axis=1
|
|
664
|
+
)
|
|
665
|
+
Y_post_pseudo_treated = np.mean(
|
|
666
|
+
Y_post_control[:, pseudo_treated_idx], axis=1
|
|
667
|
+
)
|
|
668
|
+
|
|
669
|
+
# Get pseudo-control outcomes
|
|
670
|
+
Y_pre_pseudo_control = Y_pre_control[:, pseudo_control_idx]
|
|
671
|
+
Y_post_pseudo_control = Y_post_control[:, pseudo_control_idx]
|
|
672
|
+
|
|
673
|
+
# Compute placebo SDID estimate (step 4)
|
|
674
|
+
tau = compute_sdid_estimator(
|
|
675
|
+
Y_pre_pseudo_control,
|
|
676
|
+
Y_post_pseudo_control,
|
|
677
|
+
Y_pre_pseudo_treated,
|
|
678
|
+
Y_post_pseudo_treated,
|
|
679
|
+
pseudo_weights,
|
|
680
|
+
time_weights
|
|
681
|
+
)
|
|
682
|
+
placebo_estimates.append(tau)
|
|
683
|
+
|
|
684
|
+
except (ValueError, LinAlgError, ZeroDivisionError):
|
|
685
|
+
# Skip failed iterations
|
|
686
|
+
continue
|
|
687
|
+
|
|
688
|
+
placebo_estimates = np.array(placebo_estimates)
|
|
689
|
+
n_successful = len(placebo_estimates)
|
|
690
|
+
|
|
691
|
+
if n_successful < 2:
|
|
692
|
+
warnings.warn(
|
|
693
|
+
f"Only {n_successful} placebo replications completed successfully. "
|
|
694
|
+
f"Standard error cannot be estimated reliably. "
|
|
695
|
+
f"Consider using variance_method='bootstrap' or increasing "
|
|
696
|
+
f"the number of control units.",
|
|
697
|
+
UserWarning,
|
|
698
|
+
stacklevel=3,
|
|
699
|
+
)
|
|
700
|
+
return 0.0, placebo_estimates
|
|
701
|
+
|
|
702
|
+
# Warn if many replications failed
|
|
703
|
+
failure_rate = 1 - (n_successful / replications)
|
|
704
|
+
if failure_rate > 0.05:
|
|
705
|
+
warnings.warn(
|
|
706
|
+
f"Only {n_successful}/{replications} placebo replications succeeded "
|
|
707
|
+
f"({failure_rate:.1%} failure rate). Standard errors may be unreliable.",
|
|
708
|
+
UserWarning,
|
|
709
|
+
stacklevel=3,
|
|
710
|
+
)
|
|
711
|
+
|
|
712
|
+
# Compute SE using R's formula: sqrt((r-1)/r) * sd(estimates)
|
|
713
|
+
# This matches synthdid::vcov.R exactly
|
|
714
|
+
se = np.sqrt((n_successful - 1) / n_successful) * np.std(
|
|
715
|
+
placebo_estimates, ddof=1
|
|
716
|
+
)
|
|
717
|
+
|
|
718
|
+
return se, placebo_estimates
|
|
719
|
+
|
|
547
720
|
def get_params(self) -> Dict[str, Any]:
|
|
548
721
|
"""Get estimator parameters."""
|
|
549
722
|
return {
|
|
550
723
|
"lambda_reg": self.lambda_reg,
|
|
551
724
|
"zeta": self.zeta,
|
|
552
725
|
"alpha": self.alpha,
|
|
726
|
+
"variance_method": self.variance_method,
|
|
553
727
|
"n_bootstrap": self.n_bootstrap,
|
|
554
728
|
"seed": self.seed,
|
|
555
729
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: diff-diff
|
|
3
|
-
Version: 1.3.
|
|
3
|
+
Version: 1.3.1
|
|
4
4
|
Summary: A library for Difference-in-Differences causal inference analysis
|
|
5
5
|
Author: diff-diff contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -116,6 +116,7 @@ Signif. codes: '***' 0.001, '**' 0.01, '*' 0.05, '.' 0.1
|
|
|
116
116
|
- **Pre-trends power analysis**: Roth (2022) minimum detectable violation (MDV) and power curves for pre-trends tests
|
|
117
117
|
- **Power analysis**: MDE, sample size, and power calculations for study design; simulation-based power for any estimator
|
|
118
118
|
- **Data prep utilities**: Helper functions for common data preparation tasks
|
|
119
|
+
- **Validated against R**: Benchmarked against `did`, `synthdid`, and `fixest` packages (see [benchmarks](docs/benchmarks.rst))
|
|
119
120
|
|
|
120
121
|
## Tutorials
|
|
121
122
|
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
diff_diff/__init__.py,sha256=
|
|
1
|
+
diff_diff/__init__.py,sha256=1Izex7iZprjS_UGqIJj6iw9fO17US6ZI1HRINYW_tsA,4361
|
|
2
2
|
diff_diff/bacon.py,sha256=AgQOtGUN-gtMh8q6KsGmOiVF4uCCFG2IexbeKS0mBlQ,36819
|
|
3
3
|
diff_diff/diagnostics.py,sha256=1yOKfauW9RP7alXYlaR8oq5tmol5T24idsfqzyAxBPQ,29572
|
|
4
4
|
diff_diff/estimators.py,sha256=4Jj9xAnF39EWElweFoYUa91YnyELelT8mRDzZKZWj4E,35807
|
|
@@ -6,15 +6,15 @@ diff_diff/honest_did.py,sha256=J4l-JyHHhhBiUT-F9Wls_8kjhxWhEj-N3BOe02pv0AA,46925
|
|
|
6
6
|
diff_diff/power.py,sha256=cpdbWG-lxqAKjeAoM_8Un8gYaDwADIeYdWUIkEBfXZg,42800
|
|
7
7
|
diff_diff/prep.py,sha256=bTPXTWajBzdNVPcLQ_IBYjww63DhgS3Z37V9h5tsIgc,46985
|
|
8
8
|
diff_diff/pretrends.py,sha256=NeYjTxC9s_iIj-3beYupiDAZROYaLjFIzbm-76XhqKk,36538
|
|
9
|
-
diff_diff/results.py,sha256=
|
|
9
|
+
diff_diff/results.py,sha256=ymN7_WVd-XTzJGmuULO8Ryda929uBOJX0FQZ341iJek,22746
|
|
10
10
|
diff_diff/staggered.py,sha256=DPx6VRlF67IryT8cXYq_gfdclQZEDHIYzQakR3OSvWA,69465
|
|
11
11
|
diff_diff/sun_abraham.py,sha256=ux6igLILW7Y1bh3TK33P781ATUWJS4fkwrjnpTARPlU,41685
|
|
12
|
-
diff_diff/synthetic_did.py,sha256=
|
|
12
|
+
diff_diff/synthetic_did.py,sha256=PjVFvCWQeE0X_4YZhw8iKSkyoyS2LEy7-yiwIQQd4bE,26765
|
|
13
13
|
diff_diff/triple_diff.py,sha256=Jb8ld5ME6E7uQNvS2pgNmi0kHY3WikrLWm4TjTe__xI,44679
|
|
14
14
|
diff_diff/twfe.py,sha256=iB-hvmOQd97B5_3hWvoaYGCSfcwJbXfg6towhRYD_Bw,11931
|
|
15
15
|
diff_diff/utils.py,sha256=F0vRZ32k065VsP4YcAcUdaF_dom834QOgYj3ZaGS2pI,44211
|
|
16
16
|
diff_diff/visualization.py,sha256=1X074fUk_fZleAlHdSGYc-ihdShY9FJ6WoKmBFCV6oI,51537
|
|
17
|
-
diff_diff-1.3.
|
|
18
|
-
diff_diff-1.3.
|
|
19
|
-
diff_diff-1.3.
|
|
20
|
-
diff_diff-1.3.
|
|
17
|
+
diff_diff-1.3.1.dist-info/METADATA,sha256=nML3_aj-pgfeQiWsz7mOI0HRAczQc1nPPOSJkqqqpNw,77719
|
|
18
|
+
diff_diff-1.3.1.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
19
|
+
diff_diff-1.3.1.dist-info/top_level.txt,sha256=-7mAFgjEQIA2okDLHlh5pDwBQXnsO1Z85qvtHjZILoQ,10
|
|
20
|
+
diff_diff-1.3.1.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|