hdfe-stream 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hdfe_stream/__init__.py +138 -0
- hdfe_stream/_types.py +18 -0
- hdfe_stream/am_interval.py +119 -0
- hdfe_stream/api.py +184 -0
- hdfe_stream/estimator.py +429 -0
- hdfe_stream/families.py +162 -0
- hdfe_stream/feterms.py +32 -0
- hdfe_stream/formula.py +261 -0
- hdfe_stream/glm.py +615 -0
- hdfe_stream/inference.py +652 -0
- hdfe_stream/inverse.py +470 -0
- hdfe_stream/kernels_base.py +442 -0
- hdfe_stream/kernels_glm.py +54 -0
- hdfe_stream/kernels_graph.py +135 -0
- hdfe_stream/kernels_inverse.py +313 -0
- hdfe_stream/kernels_se.py +98 -0
- hdfe_stream/kernels_slopes.py +300 -0
- hdfe_stream/leaveout.py +1270 -0
- hdfe_stream/leaveout_match.py +208 -0
- hdfe_stream/leaveout_se.py +1025 -0
- hdfe_stream/leaveout_weakid.py +725 -0
- hdfe_stream/passes.py +473 -0
- hdfe_stream/py.typed +0 -0
- hdfe_stream/report.py +40 -0
- hdfe_stream/reporting.py +197 -0
- hdfe_stream/results.py +479 -0
- hdfe_stream/simulate.py +310 -0
- hdfe_stream/solve.py +360 -0
- hdfe_stream/utils.py +201 -0
- hdfe_stream/workspace.py +170 -0
- hdfe_stream-0.1.0.dist-info/METADATA +591 -0
- hdfe_stream-0.1.0.dist-info/RECORD +35 -0
- hdfe_stream-0.1.0.dist-info/WHEEL +5 -0
- hdfe_stream-0.1.0.dist-info/licenses/LICENSE.md +21 -0
- hdfe_stream-0.1.0.dist-info/top_level.txt +1 -0
hdfe_stream/__init__.py
ADDED
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""
|
|
2
|
+
hdfe_stream: out-of-core OLS with several high-dimensional fixed effects, and
|
|
3
|
+
Poisson, logit and probit (`fepois_stream`, `feglm_stream`) by iteratively
|
|
4
|
+
reweighted least squares over the same passes (see glm.py).
|
|
5
|
+
|
|
6
|
+
from hdfe_stream import feols_stream
|
|
7
|
+
fit = feols_stream("log_earn ~ age_squared + age_cubed | worker_id + firm_id + year",
|
|
8
|
+
"data/*.parquet", workdir="hdfe_work", vcov={"CRV1": "worker_id"})
|
|
9
|
+
fit.summary()
|
|
10
|
+
|
|
11
|
+
Formulas follow pyfixest syntax and are parsed with pyfixest's own parser:
|
|
12
|
+
covariates, interactions (`:` and `*`), `I()`, `log()`, `C()`, `i()`,
|
|
13
|
+
several dependent variables (`y1 + y2 ~ ...`), `sw()` / `csw()` stepwise
|
|
14
|
+
syntax, fixed-effect interactions (`firm_id^year`), and IV
|
|
15
|
+
(`y ~ x | fe | endog ~ z`). Every covariate term is compiled to a Polars
|
|
16
|
+
expression, so constructing the design never leaves the streaming engine.
|
|
17
|
+
Weighted least squares via `weights=` (aweights or fweights). The
|
|
18
|
+
lower-level `StreamingHDFE` class takes column names or Polars expressions
|
|
19
|
+
directly.
|
|
20
|
+
|
|
21
|
+
Reporting: `result.to_pyfixest()` returns a pyfixest Feols/Feiv view for
|
|
22
|
+
pf.etable, pf.summary, pf.coefplot and pf.iplot; `hdfe_stream.etable(...)`
|
|
23
|
+
does the conversion for you and accepts ordinary pyfixest models too.
|
|
24
|
+
|
|
25
|
+
Model: y = X b + sum_d gamma_d[level_d(i)] + e for d = 0 ... D-1
|
|
26
|
+
|
|
27
|
+
One fixed-effect dimension (`fe[0]` internally) is *streamed*: it is never
|
|
28
|
+
represented as a vector in memory. It should be the one with the most levels
|
|
29
|
+
whose groups each touch only a few levels of the other dimensions (workers,
|
|
30
|
+
not firms). It is chosen as: the `stream=` option if given; else the
|
|
31
|
+
dimension with varying slopes; else the highest approximate cardinality
|
|
32
|
+
(HyperLogLog counts gathered during pass 0's first scan), ties going to the
|
|
33
|
+
dimension listed first. Every other dimension is represented by level-sized
|
|
34
|
+
vectors in RAM.
|
|
35
|
+
|
|
36
|
+
Varying slopes (FEIS) on the streamed dimension use fixest syntax:
|
|
37
|
+
`| worker_id[t] + firm_id` gives worker effects plus worker-specific slopes on
|
|
38
|
+
t (several: `worker_id[t, t2]`). Projecting out the streamed dimension
|
|
39
|
+
then means removing each worker's own weighted regression on [1, slopes],
|
|
40
|
+
which is still local to the worker.
|
|
41
|
+
|
|
42
|
+
Fewer than two dimensions. With one, it is streamed and there is nothing
|
|
43
|
+
left to solve for: step 2 is skipped and the fit is the within regression.
|
|
44
|
+
With none, there is no dimension to group the rows by, so pass 0 writes them
|
|
45
|
+
unpartitioned and unsorted, passes 1 and 1b and step 2 are skipped, and the
|
|
46
|
+
row passes read plain batches without demeaning; the model keeps an
|
|
47
|
+
"Intercept" coefficient, as in pyfixest.
|
|
48
|
+
|
|
49
|
+
Memory: rows are only ever streamed. The arrays held in RAM are sized by the
|
|
50
|
+
non-streamed dimensions' level counts (times a block of at most `rhs_block`
|
|
51
|
+
variables at a time), plus, for the explicit solver, the sparse reduced
|
|
52
|
+
matrix S, whose size depends on how levels co-occur and not on rows. Polars
|
|
53
|
+
steps stream; the sort and cell group_by run one fe[0] hash bucket at a time.
|
|
54
|
+
|
|
55
|
+
Pipeline
|
|
56
|
+
--------
|
|
57
|
+
Pass 0 (Polars, streaming) evaluate the design as Polars expressions; drop
|
|
58
|
+
rows with missing / non-finite values; factorize
|
|
59
|
+
the non-streamed FE and cluster ids;
|
|
60
|
+
hash-partition rows into fe[0] buckets; sort each
|
|
61
|
+
bucket and assign dense fe[0] codes.
|
|
62
|
+
Pass 1 (Polars, streaming) per bucket: group_by(fe[0], other codes) -> cell
|
|
63
|
+
table with n and sums (plus within-cell
|
|
64
|
+
cross-products when assembly="cells").
|
|
65
|
+
Pass 1b (streamed + numba) keep cells of fe[0] groups with more than one
|
|
66
|
+
cell as memory-mapped arrays; Jacobi diagonal.
|
|
67
|
+
Step 2 (solver, per block) solve S Gamma = D_o' M_0 V, S = D_o' M_0 D_o,
|
|
68
|
+
for all variables V in blocks of `rhs_block`
|
|
69
|
+
columns: "explicit" (sparse S built once, block
|
|
70
|
+
PCG in memory), "stream_cg" (S applied by
|
|
71
|
+
streaming the cells), or "within".
|
|
72
|
+
Step 3 (numba + BLAS) assemble V' M_D V (from cells, or from a row
|
|
73
|
+
pass when there are many variables: numba
|
|
74
|
+
residualizes each chunk in place, BLAS forms its
|
|
75
|
+
cross-products); drop collinear covariates per
|
|
76
|
+
model; beta.
|
|
77
|
+
Step 4 (streamed, numba + per model: stream rows for fe[0] effects,
|
|
78
|
+
BLAS) residuals and RSS (numba), and the bread and the
|
|
79
|
+
meat for HC1 / CRV1 (BLAS, chunk by chunk).
|
|
80
|
+
|
|
81
|
+
Layout
|
|
82
|
+
------
|
|
83
|
+
api.py `feols_stream`, `fepois_stream`, `feglm_stream`
|
|
84
|
+
estimator.py `StreamingHDFE`: options, layout, the driver
|
|
85
|
+
glm.py `StreamingGLM`: IRLS over the same passes
|
|
86
|
+
families.py GLM families: Poisson, logit, probit
|
|
87
|
+
passes.py pass 0/1/1b (mixin)
|
|
88
|
+
solve.py step 2 (mixin)
|
|
89
|
+
inference.py steps 3/4 (mixin)
|
|
90
|
+
kernels_base.py numba kernels
|
|
91
|
+
kernels_slopes.py numba kernels, varying slopes
|
|
92
|
+
kernels_glm.py numba kernels, GLM cell sums and group effects
|
|
93
|
+
formula.py pyfixest formula -> Polars expressions (input side)
|
|
94
|
+
reporting.py pyfixest etable/coefplot views (output side)
|
|
95
|
+
feterms.py 'worker_id[t]' / 'firm_id^year' parsing
|
|
96
|
+
results.py `HDFEResult`, `HDFEMulti`
|
|
97
|
+
workspace.py run directories, `cleanup()`
|
|
98
|
+
report.py logging / printing
|
|
99
|
+
utils.py small shared helpers
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
from .api import feglm_stream, fepois_stream, feols_stream
|
|
103
|
+
from .estimator import StreamingHDFE
|
|
104
|
+
from .glm import StreamingGLM
|
|
105
|
+
from .leaveout_se import ComponentSE
|
|
106
|
+
from .leaveout_weakid import WeakIdDiagnostics
|
|
107
|
+
from .leaveout import leave_one_out_connected, leave_out_kss
|
|
108
|
+
from .reporting import etable
|
|
109
|
+
from .results import HDFEMulti, HDFEResult
|
|
110
|
+
from .utils import iter_group_chunks
|
|
111
|
+
from .workspace import cleanup
|
|
112
|
+
|
|
113
|
+
__all__ = [
|
|
114
|
+
"feols_stream",
|
|
115
|
+
"fepois_stream",
|
|
116
|
+
"feglm_stream",
|
|
117
|
+
"StreamingHDFE",
|
|
118
|
+
"StreamingGLM",
|
|
119
|
+
"HDFEResult",
|
|
120
|
+
"HDFEMulti",
|
|
121
|
+
"etable",
|
|
122
|
+
"leave_out_kss",
|
|
123
|
+
"ComponentSE",
|
|
124
|
+
"WeakIdDiagnostics",
|
|
125
|
+
"leave_one_out_connected",
|
|
126
|
+
"cleanup",
|
|
127
|
+
"iter_group_chunks",
|
|
128
|
+
]
|
|
129
|
+
|
|
130
|
+
def _detect_version():
|
|
131
|
+
try:
|
|
132
|
+
from importlib.metadata import version
|
|
133
|
+
return version("hdfe-stream")
|
|
134
|
+
except Exception: # source tree without an installed dist
|
|
135
|
+
return "0.1.0"
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
__version__ = _detect_version()
|
hdfe_stream/_types.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Type aliases shared by the public signatures."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any, Mapping, Sequence, Union
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
|
|
10
|
+
PathLike = Union[str, Path]
|
|
11
|
+
# a Parquet path or glob, or a Polars LazyFrame
|
|
12
|
+
Source = Union[PathLike, pl.LazyFrame]
|
|
13
|
+
# 'iid', 'hetero', 'HC1', 'CRV1:var', or {'CRV1': 'var'}
|
|
14
|
+
Vcov = Union[str, Mapping[str, str]]
|
|
15
|
+
# column name, (name, expression), or a mapping / sequence of them
|
|
16
|
+
Variables = Union[
|
|
17
|
+
str, tuple, Sequence[Union[str, tuple]], Mapping[str, pl.Expr]]
|
|
18
|
+
Options = Any
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
"""The Andrews-Mikusheva interval for a weakly identified quadratic form.
|
|
2
|
+
|
|
3
|
+
Pure numerics, no data: given the leading eigenvalue lambda_1 and the joint
|
|
4
|
+
estimate (b1-hat, theta1-hat) with its 2x2 covariance, return the q = 1
|
|
5
|
+
confidence interval of Kline, Saggio and Solvsten (2020), Section 6.1:
|
|
6
|
+
|
|
7
|
+
C = [ min, max ] of lambda_1 b^2 + t over the ellipse
|
|
8
|
+
{ (b1-hat - b, theta1-hat - t) Sigma^-1 (.)' <= z_{alpha,kappa}^2 }
|
|
9
|
+
|
|
10
|
+
Projecting a 2-d ellipse through a quadratic map would ordinarily need the
|
|
11
|
+
chi-squared(2) critical value to control size; Andrews and Mikusheva (2016)
|
|
12
|
+
show the curvature of the map lets it come down, toward the chi-squared(1) value
|
|
13
|
+
when the map is nearly linear. KSS's curvature for q = 1 is
|
|
14
|
+
|
|
15
|
+
kappa = 2 |lambda_1| V[b1] / ( V[theta1]^(1/2) (1 - rho^2)^(1/2) )
|
|
16
|
+
|
|
17
|
+
and z_{alpha,kappa} is the (1 - alpha) quantile of
|
|
18
|
+
|
|
19
|
+
rho(kappa) = sqrt(chi_1^2 + (chi_1' + 1/kappa)^2) - 1/kappa
|
|
20
|
+
|
|
21
|
+
for independent chi variates (Appendix C.6.1). Saggio's reference tabulates that
|
|
22
|
+
quantile by simulation; for q = 1 it has an exact one-dimensional integral form,
|
|
23
|
+
used here. With a, b independent half-normals and c = 1/kappa,
|
|
24
|
+
|
|
25
|
+
rho <= z <=> a^2 + (b + c)^2 <= (z + c)^2
|
|
26
|
+
P(rho <= z) = int_0^z 2 phi(b) [2 Phi(sqrt((z - b)(z + b + 2c))) - 1] db
|
|
27
|
+
|
|
28
|
+
written in factored form so nothing cancels as kappa -> 0. The limits are the
|
|
29
|
+
chi-squared(1) quantile at kappa = 0 and chi-squared(2) as kappa -> infinity.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import numpy as np
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def am_cdf(z: float, kappa: float) -> float:
|
|
38
|
+
"""P(rho(kappa) <= z) for q = 1."""
|
|
39
|
+
from scipy import integrate, stats
|
|
40
|
+
|
|
41
|
+
if z <= 0:
|
|
42
|
+
return 0.0
|
|
43
|
+
if kappa <= 0:
|
|
44
|
+
return float(2.0 * stats.norm.cdf(z) - 1.0)
|
|
45
|
+
c = 1.0 / kappa
|
|
46
|
+
|
|
47
|
+
def integrand(b):
|
|
48
|
+
inner = np.sqrt(max((z - b) * (z + b + 2.0 * c), 0.0))
|
|
49
|
+
return 2.0 * stats.norm.pdf(b) * (2.0 * stats.norm.cdf(inner) - 1.0)
|
|
50
|
+
|
|
51
|
+
value, _err = integrate.quad(integrand, 0.0, z, epsabs=1e-13, epsrel=1e-12,
|
|
52
|
+
limit=200)
|
|
53
|
+
return float(value)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def am_critical_value(kappa: float, alpha: float = 0.05) -> float:
|
|
57
|
+
"""z_{alpha, kappa} for q = 1: between sqrt(chi2_1) and sqrt(chi2_2)."""
|
|
58
|
+
from scipy import optimize, stats
|
|
59
|
+
|
|
60
|
+
lo = float(np.sqrt(stats.chi2.ppf(1.0 - alpha, 1)))
|
|
61
|
+
if kappa <= 0:
|
|
62
|
+
return lo
|
|
63
|
+
hi = float(np.sqrt(stats.chi2.ppf(1.0 - alpha, 2)))
|
|
64
|
+
target = 1.0 - alpha
|
|
65
|
+
# the quantile rises with kappa from lo to hi; bracket just outside both
|
|
66
|
+
return float(optimize.brentq(lambda z: am_cdf(z, kappa) - target,
|
|
67
|
+
lo * (1 - 1e-9), hi * (1 + 1e-9),
|
|
68
|
+
xtol=1e-13, rtol=1e-13))
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def curvature(lam1: float, sigma: np.ndarray) -> float:
|
|
72
|
+
"""KSS's q = 1 curvature from lambda_1 and the 2x2 covariance."""
|
|
73
|
+
v_b, v_t, cov = float(sigma[0, 0]), float(sigma[1, 1]), float(sigma[0, 1])
|
|
74
|
+
rho2 = cov * cov / (v_b * v_t)
|
|
75
|
+
return 2.0 * abs(lam1) * v_b / np.sqrt(v_t * (1.0 - rho2))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def am_interval(lam1: float, b1: float, theta1: float, sigma: np.ndarray,
|
|
79
|
+
alpha: float = 0.05, grid: int = 4096) -> tuple[float, float, dict[str, float]]:
|
|
80
|
+
"""(lower, upper, info) for the q = 1 weak-identification interval.
|
|
81
|
+
|
|
82
|
+
`sigma` is the 2x2 covariance of (b1-hat, theta1-hat), which must be
|
|
83
|
+
positive definite. The extremes of lambda_1 b^2 + t over the ellipse lie on
|
|
84
|
+
its boundary -- the map is linear in t with coefficient one, so any interior
|
|
85
|
+
point can be improved by moving t -- so the problem is one-dimensional in
|
|
86
|
+
the boundary angle. The objective there is a trigonometric polynomial of
|
|
87
|
+
degree two, with at most four stationary points, so a dense grid followed by
|
|
88
|
+
a bounded Brent refinement finds both global extremes. KSS's closed form
|
|
89
|
+
(Appendix C.6.2) solves the same first-order condition as a quartic; the
|
|
90
|
+
tests check the two agree.
|
|
91
|
+
"""
|
|
92
|
+
from scipy import optimize
|
|
93
|
+
|
|
94
|
+
sigma = np.asarray(sigma, dtype=np.float64)
|
|
95
|
+
kappa = curvature(lam1, sigma)
|
|
96
|
+
z = am_critical_value(kappa, alpha)
|
|
97
|
+
L = np.linalg.cholesky(sigma)
|
|
98
|
+
|
|
99
|
+
def g(t):
|
|
100
|
+
b = b1 + z * L[0, 0] * np.cos(t)
|
|
101
|
+
th = theta1 + z * (L[1, 0] * np.cos(t) + L[1, 1] * np.sin(t))
|
|
102
|
+
return lam1 * b * b + th
|
|
103
|
+
|
|
104
|
+
ts = np.linspace(0.0, 2.0 * np.pi, grid, endpoint=False)
|
|
105
|
+
values = g(ts)
|
|
106
|
+
step = ts[1] - ts[0]
|
|
107
|
+
|
|
108
|
+
def refine(index, sign):
|
|
109
|
+
t0 = ts[index]
|
|
110
|
+
res = optimize.minimize_scalar(lambda t: sign * g(t),
|
|
111
|
+
bounds=(t0 - step, t0 + step),
|
|
112
|
+
method="bounded",
|
|
113
|
+
options={"xatol": 1e-14})
|
|
114
|
+
return float(g(res.x)) if sign > 0 else float(g(res.x))
|
|
115
|
+
|
|
116
|
+
lower = min(refine(int(np.argmin(values)), 1.0), float(values.min()))
|
|
117
|
+
upper = max(refine(int(np.argmax(values)), -1.0), float(values.max()))
|
|
118
|
+
rho = float(sigma[0, 1] / np.sqrt(sigma[0, 0] * sigma[1, 1]))
|
|
119
|
+
return lower, upper, {"kappa": float(kappa), "z": float(z), "rho": rho}
|
hdfe_stream/api.py
ADDED
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
"""`feols_stream`, `fepois_stream`, `feglm_stream`: fit a pyfixest-style
|
|
2
|
+
formula out of core."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from typing import Any, Sequence
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
|
|
10
|
+
from ._types import PathLike, Source, Vcov
|
|
11
|
+
|
|
12
|
+
from .estimator import StreamingHDFE
|
|
13
|
+
from .feterms import _parse_fe_term
|
|
14
|
+
from .glm import StreamingGLM
|
|
15
|
+
from .report import _log
|
|
16
|
+
from .formula import _plan_formula, _pyfixest_formula_api
|
|
17
|
+
from .results import HDFEMulti, HDFEResult, _vcov_key
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class _FormulaEstimator:
|
|
21
|
+
def __init__(self, fml: str, workdir: PathLike | None, options: dict[str, Any]) -> None:
|
|
22
|
+
self.fml, self.workdir, self.options = fml, workdir, options
|
|
23
|
+
|
|
24
|
+
def fit(self, data: Source, vcov: Vcov | None = None, cluster: Sequence[str] | str = (),
|
|
25
|
+
fe_dof: str = "exact") -> HDFEResult | HDFEMulti:
|
|
26
|
+
return feols_stream(self.fml, data, self.workdir, vcov=vcov, cluster=cluster,
|
|
27
|
+
fe_dof=fe_dof, **self.options)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def feols_stream(fml: str, data: Source, workdir: PathLike | None = None,
|
|
31
|
+
vcov: Vcov | None = None, cluster: Sequence[str] | str = (),
|
|
32
|
+
fe_dof: str = "exact", **options: Any) -> HDFEResult | HDFEMulti:
|
|
33
|
+
"""
|
|
34
|
+
Out-of-core OLS with high-dimensional fixed effects from a pyfixest-style
|
|
35
|
+
formula, e.g. "y ~ x1 + i(year, treat, ref=2010) | worker_id + firm_id^year".
|
|
36
|
+
Any number of fixed effects works, including one ("y ~ x | worker_id",
|
|
37
|
+
a within regression) and none ("y ~ x", OLS with an intercept).
|
|
38
|
+
|
|
39
|
+
data : Parquet path/glob or Polars LazyFrame.
|
|
40
|
+
workdir : directory under which run directories are created (default:
|
|
41
|
+
$HDFE_STREAM_WORKDIR if set, else the system temporary
|
|
42
|
+
directory); see StreamingHDFE for the
|
|
43
|
+
outputs=, save_resid= and keep_intermediates= options.
|
|
44
|
+
vcov : as in pyfixest; default {'CRV1': <first FE>} like pyfixest, or
|
|
45
|
+
iid without fixed effects. {'CRV3': var} (one-way, OLS) needs
|
|
46
|
+
every fixed effect nested within the clusters, or none.
|
|
47
|
+
cluster : extra cluster specs to compute CRV1 for (one-way or 'a+b').
|
|
48
|
+
**options : passed to StreamingHDFE (stream, weights, weights_type,
|
|
49
|
+
solver, precond, keep, verbose, logger, log_level, ...).
|
|
50
|
+
|
|
51
|
+
Varying slopes on the streamed dimension: "y ~ x | worker_id[t] + firm_id".
|
|
52
|
+
The worker FE file then has fe_worker_id (intercept) and fe_worker_id[t] (slope)
|
|
53
|
+
columns; the residual file's fe_worker_id column is the worker's total
|
|
54
|
+
contribution for that row.
|
|
55
|
+
|
|
56
|
+
IV formulas ("y ~ x | fe | endog ~ z") are estimated by 2SLS; each result
|
|
57
|
+
carries its first-stage regressions (`first_stage`) and the Wald F of the
|
|
58
|
+
excluded instruments computed with the same vcov (`f_stat_1st_stage`).
|
|
59
|
+
|
|
60
|
+
All models with the same fixed effects share one set of passes and one
|
|
61
|
+
solve, and are estimated on the rows where *all* of their variables are
|
|
62
|
+
non-missing (pyfixest drops missing values model by model).
|
|
63
|
+
"""
|
|
64
|
+
def estimators(fe, g):
|
|
65
|
+
yield StreamingHDFE(y=list(g["y"].items()), x=list(g["x"].items()), fe=list(fe),
|
|
66
|
+
workdir=workdir, models=g["models"], **options)
|
|
67
|
+
return _fit_formula(fml, data, estimators, vcov, cluster, fe_dof, options)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _fit_formula(fml, data, estimators, vcov, cluster, fe_dof, options):
|
|
71
|
+
"""Plan the formula, fit each estimator that `estimators(fe, group)`
|
|
72
|
+
makes for each set of fixed effects, and return the results in the
|
|
73
|
+
formula's order."""
|
|
74
|
+
lf = data if isinstance(data, pl.LazyFrame) else pl.scan_parquet(data)
|
|
75
|
+
_log(options.get("verbose", True), f"planning {fml!r} (formula expansion, level discovery)",
|
|
76
|
+
options.get("logger"), options.get("log_level"))
|
|
77
|
+
groups = _plan_formula(fml, lf)
|
|
78
|
+
results = []
|
|
79
|
+
try:
|
|
80
|
+
for fe, g in groups.items():
|
|
81
|
+
# pyfixest's default: cluster by the first fixed effect as written,
|
|
82
|
+
# or iid when there is none
|
|
83
|
+
default = {"CRV1": _parse_fe_term(fe[0])[0]} if fe else "iid"
|
|
84
|
+
for est in estimators(fe, g):
|
|
85
|
+
res = est.fit(lf, vcov=vcov if vcov is not None else default, cluster=cluster,
|
|
86
|
+
fe_dof=fe_dof)
|
|
87
|
+
results += list(res) if isinstance(res, HDFEMulti) else [res]
|
|
88
|
+
except BaseException:
|
|
89
|
+
# a later fit failed: don't leave the earlier fits' files behind
|
|
90
|
+
if not options.get("keep_intermediates"):
|
|
91
|
+
for r in results:
|
|
92
|
+
r.cleanup()
|
|
93
|
+
raise
|
|
94
|
+
Formula, _ = _pyfixest_formula_api()
|
|
95
|
+
order = {s.formula: i for i, s in enumerate(Formula.parse(fml))}
|
|
96
|
+
results.sort(key=lambda r: order.get(r.fml, len(order)))
|
|
97
|
+
return results[0] if len(results) == 1 else HDFEMulti(results)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def fepois_stream(fml: str, data: Source, workdir: PathLike | None = None,
|
|
101
|
+
vcov: Vcov | None = None, cluster: Sequence[str] | str = (),
|
|
102
|
+
fe_dof: str = "exact", offset: str | pl.Expr | None = None,
|
|
103
|
+
iwls_tol: float = 1e-8, iwls_maxiter: int = 25,
|
|
104
|
+
separation_check: bool = True, **options: Any) -> HDFEResult | HDFEMulti:
|
|
105
|
+
"""
|
|
106
|
+
Out-of-core Poisson regression (log link) with high-dimensional fixed
|
|
107
|
+
effects, from a pyfixest-style formula, e.g.
|
|
108
|
+
"trade ~ log_dist | exporter^year + importer^year". Estimated by
|
|
109
|
+
iteratively reweighted least squares; each step is one read of the rows
|
|
110
|
+
and one solve of the fixed effects (see `StreamingGLM`).
|
|
111
|
+
|
|
112
|
+
data, workdir, vcov, cluster, fe_dof, **options : as in `feols_stream`,
|
|
113
|
+
except that the vcov may not be CRV3, and IV formulas and
|
|
114
|
+
varying slopes are not available. weights= and weights_type=
|
|
115
|
+
work as in pyfixest's fepois.
|
|
116
|
+
offset : column name added to the linear predictor with a coefficient of
|
|
117
|
+
one (e.g. log exposure).
|
|
118
|
+
iwls_tol, iwls_maxiter : convergence tolerance on the relative change in
|
|
119
|
+
deviance, and the most IRLS steps (pyfixest's defaults).
|
|
120
|
+
separation_check : drop fixed-effect levels whose outcome is zero on
|
|
121
|
+
every row (their effect would be minus infinity), as pyfixest's
|
|
122
|
+
"fe" check, which it runs by default. Covariates that separate
|
|
123
|
+
the outcome (its "ir" check) are not detected.
|
|
124
|
+
|
|
125
|
+
The dependent variable must be nonnegative. Each dependent variable is
|
|
126
|
+
its own set of IRLS fits (the separation check depends on it); models of
|
|
127
|
+
the same outcome share pass 0.
|
|
128
|
+
"""
|
|
129
|
+
return _fit_glm(fml, data, "poisson", workdir, vcov, cluster, fe_dof,
|
|
130
|
+
dict(options, offset=offset, iwls_tol=iwls_tol, iwls_maxiter=iwls_maxiter,
|
|
131
|
+
separation_check=separation_check))
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def feglm_stream(fml: str, data: Source, family: str, workdir: PathLike | None = None,
|
|
135
|
+
vcov: Vcov | None = None, cluster: Sequence[str] | str = (),
|
|
136
|
+
fe_dof: str = "exact", iwls_tol: float = 1e-8, iwls_maxiter: int = 25,
|
|
137
|
+
separation_check: bool = True, **options: Any) -> HDFEResult | HDFEMulti:
|
|
138
|
+
"""
|
|
139
|
+
Out-of-core logit or probit regression with high-dimensional fixed
|
|
140
|
+
effects, from a pyfixest-style formula: `family` is "logit" or "probit".
|
|
141
|
+
Estimated by iteratively reweighted least squares, with step-halving, as
|
|
142
|
+
pyfixest's feglm (see `StreamingGLM`).
|
|
143
|
+
|
|
144
|
+
Arguments as in `fepois_stream`, without the offset. The dependent
|
|
145
|
+
variable must be 0 or 1. The separation check drops fixed-effect levels
|
|
146
|
+
whose outcome is the same on every row (all 0 or all 1), repeating until
|
|
147
|
+
none is left; pyfixest's "fe" check drops only levels whose outcome is
|
|
148
|
+
all 0, once.
|
|
149
|
+
|
|
150
|
+
Weights (weights=, weights_type=) multiply each row's log-likelihood, as
|
|
151
|
+
in fepois; pyfixest's feglm takes none. Frequency weights give exactly
|
|
152
|
+
the fit of the data with each row repeated that many times.
|
|
153
|
+
|
|
154
|
+
No correction is made for the incidental parameter problem: with fixed
|
|
155
|
+
effects estimated from few observations each, the coefficients are
|
|
156
|
+
biased, as in pyfixest and fixest.
|
|
157
|
+
"""
|
|
158
|
+
if str(family).lower() not in ("logit", "probit"):
|
|
159
|
+
raise ValueError(f"family must be 'logit' or 'probit', got {family!r}"
|
|
160
|
+
+ ("; use fepois_stream" if str(family).lower() == "poisson" else "")
|
|
161
|
+
+ ("; for a linear model use feols_stream"
|
|
162
|
+
if str(family).lower() == "gaussian" else ""))
|
|
163
|
+
return _fit_glm(fml, data, str(family).lower(), workdir, vcov, cluster, fe_dof,
|
|
164
|
+
dict(options, iwls_tol=iwls_tol, iwls_maxiter=iwls_maxiter,
|
|
165
|
+
separation_check=separation_check))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _fit_glm(fml, data, family, workdir, vcov, cluster, fe_dof, options):
|
|
169
|
+
key = _vcov_key(vcov)
|
|
170
|
+
if key is not None and key.startswith("CRV3:"):
|
|
171
|
+
raise ValueError("CRV3 standard errors are not available for GLMs; use CRV1")
|
|
172
|
+
|
|
173
|
+
def estimators(fe, g):
|
|
174
|
+
by_y = {}
|
|
175
|
+
for model in g["models"]:
|
|
176
|
+
if model.get("iv"):
|
|
177
|
+
raise ValueError(f"{model['fml']}: IV (2SLS) is not available for GLMs")
|
|
178
|
+
by_y.setdefault(model["y"], []).append(model)
|
|
179
|
+
for yname, models in by_y.items():
|
|
180
|
+
xs = list(dict.fromkeys(x for model in models for x in model["x"]))
|
|
181
|
+
yield StreamingGLM(y=[(yname, g["y"][yname])], x=[(x, g["x"][x]) for x in xs],
|
|
182
|
+
fe=list(fe), family=family, workdir=workdir, models=models,
|
|
183
|
+
**options)
|
|
184
|
+
return _fit_formula(fml, data, estimators, vcov, cluster, fe_dof, options)
|