hdfe-stream 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hdfe_stream-0.1.0/CHANGELOG.md +15 -0
- hdfe_stream-0.1.0/CITATION.cff +21 -0
- hdfe_stream-0.1.0/LICENSE.md +21 -0
- hdfe_stream-0.1.0/MANIFEST.in +8 -0
- hdfe_stream-0.1.0/PKG-INFO +591 -0
- hdfe_stream-0.1.0/README.md +542 -0
- hdfe_stream-0.1.0/docs/figures/akm_disk.dark.svg +607 -0
- hdfe_stream-0.1.0/docs/figures/akm_disk.light.svg +607 -0
- hdfe_stream-0.1.0/docs/figures/akm_memory.dark.svg +679 -0
- hdfe_stream-0.1.0/docs/figures/akm_memory.light.svg +679 -0
- hdfe_stream-0.1.0/docs/figures/akm_time.dark.svg +878 -0
- hdfe_stream-0.1.0/docs/figures/akm_time.light.svg +878 -0
- hdfe_stream-0.1.0/docs/figures/covariates_disk.dark.svg +422 -0
- hdfe_stream-0.1.0/docs/figures/covariates_disk.light.svg +422 -0
- hdfe_stream-0.1.0/docs/figures/covariates_memory.dark.svg +461 -0
- hdfe_stream-0.1.0/docs/figures/covariates_memory.light.svg +461 -0
- hdfe_stream-0.1.0/docs/figures/covariates_time.dark.svg +633 -0
- hdfe_stream-0.1.0/docs/figures/covariates_time.light.svg +633 -0
- hdfe_stream-0.1.0/docs/figures/glm_covariates_disk.dark.svg +777 -0
- hdfe_stream-0.1.0/docs/figures/glm_covariates_disk.light.svg +777 -0
- hdfe_stream-0.1.0/docs/figures/glm_covariates_memory.dark.svg +655 -0
- hdfe_stream-0.1.0/docs/figures/glm_covariates_memory.light.svg +655 -0
- hdfe_stream-0.1.0/docs/figures/glm_covariates_time.dark.svg +888 -0
- hdfe_stream-0.1.0/docs/figures/glm_covariates_time.light.svg +888 -0
- hdfe_stream-0.1.0/docs/figures/glm_disk.dark.svg +958 -0
- hdfe_stream-0.1.0/docs/figures/glm_disk.light.svg +958 -0
- hdfe_stream-0.1.0/docs/figures/glm_memory.dark.svg +916 -0
- hdfe_stream-0.1.0/docs/figures/glm_memory.light.svg +916 -0
- hdfe_stream-0.1.0/docs/figures/glm_time.dark.svg +999 -0
- hdfe_stream-0.1.0/docs/figures/glm_time.light.svg +999 -0
- hdfe_stream-0.1.0/docs/figures/kss_disk.dark.svg +565 -0
- hdfe_stream-0.1.0/docs/figures/kss_disk.light.svg +565 -0
- hdfe_stream-0.1.0/docs/figures/kss_memory.dark.svg +597 -0
- hdfe_stream-0.1.0/docs/figures/kss_memory.light.svg +597 -0
- hdfe_stream-0.1.0/docs/figures/kss_time.dark.svg +811 -0
- hdfe_stream-0.1.0/docs/figures/kss_time.light.svg +811 -0
- hdfe_stream-0.1.0/docs/glm.md +318 -0
- hdfe_stream-0.1.0/docs/kss.md +604 -0
- hdfe_stream-0.1.0/docs/kss_methodological_differences.md +683 -0
- hdfe_stream-0.1.0/hdfe_stream/__init__.py +138 -0
- hdfe_stream-0.1.0/hdfe_stream/_types.py +18 -0
- hdfe_stream-0.1.0/hdfe_stream/am_interval.py +119 -0
- hdfe_stream-0.1.0/hdfe_stream/api.py +184 -0
- hdfe_stream-0.1.0/hdfe_stream/estimator.py +429 -0
- hdfe_stream-0.1.0/hdfe_stream/families.py +162 -0
- hdfe_stream-0.1.0/hdfe_stream/feterms.py +32 -0
- hdfe_stream-0.1.0/hdfe_stream/formula.py +261 -0
- hdfe_stream-0.1.0/hdfe_stream/glm.py +615 -0
- hdfe_stream-0.1.0/hdfe_stream/inference.py +652 -0
- hdfe_stream-0.1.0/hdfe_stream/inverse.py +470 -0
- hdfe_stream-0.1.0/hdfe_stream/kernels_base.py +442 -0
- hdfe_stream-0.1.0/hdfe_stream/kernels_glm.py +54 -0
- hdfe_stream-0.1.0/hdfe_stream/kernels_graph.py +135 -0
- hdfe_stream-0.1.0/hdfe_stream/kernels_inverse.py +313 -0
- hdfe_stream-0.1.0/hdfe_stream/kernels_se.py +98 -0
- hdfe_stream-0.1.0/hdfe_stream/kernels_slopes.py +300 -0
- hdfe_stream-0.1.0/hdfe_stream/leaveout.py +1270 -0
- hdfe_stream-0.1.0/hdfe_stream/leaveout_match.py +208 -0
- hdfe_stream-0.1.0/hdfe_stream/leaveout_se.py +1025 -0
- hdfe_stream-0.1.0/hdfe_stream/leaveout_weakid.py +725 -0
- hdfe_stream-0.1.0/hdfe_stream/passes.py +473 -0
- hdfe_stream-0.1.0/hdfe_stream/py.typed +0 -0
- hdfe_stream-0.1.0/hdfe_stream/report.py +40 -0
- hdfe_stream-0.1.0/hdfe_stream/reporting.py +197 -0
- hdfe_stream-0.1.0/hdfe_stream/results.py +479 -0
- hdfe_stream-0.1.0/hdfe_stream/simulate.py +310 -0
- hdfe_stream-0.1.0/hdfe_stream/solve.py +360 -0
- hdfe_stream-0.1.0/hdfe_stream/utils.py +201 -0
- hdfe_stream-0.1.0/hdfe_stream/workspace.py +170 -0
- hdfe_stream-0.1.0/hdfe_stream.egg-info/PKG-INFO +591 -0
- hdfe_stream-0.1.0/hdfe_stream.egg-info/SOURCES.txt +91 -0
- hdfe_stream-0.1.0/hdfe_stream.egg-info/dependency_links.txt +1 -0
- hdfe_stream-0.1.0/hdfe_stream.egg-info/requires.txt +27 -0
- hdfe_stream-0.1.0/hdfe_stream.egg-info/top_level.txt +1 -0
- hdfe_stream-0.1.0/pyproject.toml +77 -0
- hdfe_stream-0.1.0/setup.cfg +4 -0
- hdfe_stream-0.1.0/tests/conftest.py +197 -0
- hdfe_stream-0.1.0/tests/test_am_interval.py +128 -0
- hdfe_stream-0.1.0/tests/test_cleanup.py +239 -0
- hdfe_stream-0.1.0/tests/test_crv3.py +107 -0
- hdfe_stream-0.1.0/tests/test_feis.py +204 -0
- hdfe_stream-0.1.0/tests/test_fewer_fe.py +216 -0
- hdfe_stream-0.1.0/tests/test_formula.py +225 -0
- hdfe_stream-0.1.0/tests/test_glm.py +339 -0
- hdfe_stream-0.1.0/tests/test_inverse.py +547 -0
- hdfe_stream-0.1.0/tests/test_iv_weights.py +173 -0
- hdfe_stream-0.1.0/tests/test_leaveout.py +2161 -0
- hdfe_stream-0.1.0/tests/test_logging.py +147 -0
- hdfe_stream-0.1.0/tests/test_lowlevel.py +230 -0
- hdfe_stream-0.1.0/tests/test_ols.py +205 -0
- hdfe_stream-0.1.0/tests/test_reporting.py +265 -0
- hdfe_stream-0.1.0/tests/test_simulate.py +168 -0
- hdfe_stream-0.1.0/tests/test_vcov.py +188 -0
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0
|
|
4
|
+
|
|
5
|
+
First release.
|
|
6
|
+
|
|
7
|
+
- `feols_stream`: out-of-core OLS with several high-dimensional fixed effects,
|
|
8
|
+
varying slopes, weights, IV, and CRV1/CRV3 and other variance estimators.
|
|
9
|
+
- `fepois_stream` and `feglm_stream`: Poisson, logit and probit by iteratively
|
|
10
|
+
reweighted least squares over the same passes.
|
|
11
|
+
- AKM variance decomposition and leave-out (KSS) bias correction with
|
|
12
|
+
standard errors and weak-identification diagnostics.
|
|
13
|
+
- Reporting through pyfixest (`to_pyfixest()`, `etable`).
|
|
14
|
+
- `summary_json()` and `summary_dict()`: the model-level information of
|
|
15
|
+
`summary()` as JSON or a dict, for saving to disk.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
message: "If you use this software in your research, please cite it as below."
|
|
3
|
+
type: software
|
|
4
|
+
title: "hdfe-stream"
|
|
5
|
+
abstract: "Out-of-core regression with several high-dimensional fixed effects: OLS, Poisson, logit and probit, with AKM/KSS variance decompositions."
|
|
6
|
+
authors:
|
|
7
|
+
- family-names: Tucker
|
|
8
|
+
given-names: Lee
|
|
9
|
+
email: lee.c.tucker@gmail.com
|
|
10
|
+
version: 0.1.0
|
|
11
|
+
date-released: 2026-09-30
|
|
12
|
+
license: MIT
|
|
13
|
+
repository-code: "https://github.com/leetucker/hdfe-stream"
|
|
14
|
+
url: "https://github.com/leetucker/hdfe-stream"
|
|
15
|
+
keywords:
|
|
16
|
+
- econometrics
|
|
17
|
+
- fixed effects
|
|
18
|
+
- high-dimensional
|
|
19
|
+
- out-of-core
|
|
20
|
+
- AKM
|
|
21
|
+
- KSS
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Lee Tucker
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,591 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hdfe-stream
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Out-of-core regression with several high-dimensional fixed effects: OLS, Poisson, logit and probit, with AKM/KSS variance decompositions
|
|
5
|
+
Author: Lee Tucker
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/leetucker/hdfe-stream
|
|
8
|
+
Project-URL: Source, https://github.com/leetucker/hdfe-stream
|
|
9
|
+
Project-URL: Issues, https://github.com/leetucker/hdfe-stream/issues
|
|
10
|
+
Project-URL: Documentation, https://github.com/leetucker/hdfe-stream/tree/main/docs
|
|
11
|
+
Project-URL: Changelog, https://github.com/leetucker/hdfe-stream/blob/main/CHANGELOG.md
|
|
12
|
+
Keywords: econometrics,fixed effects,high-dimensional,out-of-core,AKM,KSS,pyfixest
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering
|
|
23
|
+
Classifier: Typing :: Typed
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
License-File: LICENSE.md
|
|
27
|
+
Requires-Dist: polars>=1.37
|
|
28
|
+
Requires-Dist: numpy>=1.24
|
|
29
|
+
Requires-Dist: numba>=0.59
|
|
30
|
+
Requires-Dist: pyarrow>=14.0
|
|
31
|
+
Requires-Dist: scipy>=1.11
|
|
32
|
+
Provides-Extra: within
|
|
33
|
+
Requires-Dist: within-py>=0.3.0; extra == "within"
|
|
34
|
+
Provides-Extra: amg
|
|
35
|
+
Requires-Dist: pyamg>=5.1; extra == "amg"
|
|
36
|
+
Provides-Extra: formula
|
|
37
|
+
Requires-Dist: pyfixest<0.61.0,>=0.60.0; extra == "formula"
|
|
38
|
+
Requires-Dist: formulaic>=1.0; extra == "formula"
|
|
39
|
+
Requires-Dist: pandas>=2.0; extra == "formula"
|
|
40
|
+
Provides-Extra: all
|
|
41
|
+
Requires-Dist: hdfe-stream[amg,formula,within]; extra == "all"
|
|
42
|
+
Provides-Extra: benchmark
|
|
43
|
+
Requires-Dist: hdfe-stream[formula]; extra == "benchmark"
|
|
44
|
+
Requires-Dist: matplotlib>=3.7; extra == "benchmark"
|
|
45
|
+
Provides-Extra: test
|
|
46
|
+
Requires-Dist: pytest>=7.0; extra == "test"
|
|
47
|
+
Requires-Dist: hdfe-stream[all]; extra == "test"
|
|
48
|
+
Dynamic: license-file
|
|
49
|
+
|
|
50
|
+
# hdfe-stream
|
|
51
|
+
|
|
52
|
+
Regression with several high-dimensional fixed effects, on data that does not
|
|
53
|
+
fit in memory: linear (`feols_stream`), and Poisson, logit and probit
|
|
54
|
+
(`fepois_stream`, `feglm_stream`).
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
from hdfe_stream import feols_stream
|
|
58
|
+
|
|
59
|
+
fit = feols_stream(
|
|
60
|
+
"log_earn ~ age_squared + age_cubed | worker_id + firm_id + year",
|
|
61
|
+
"data/*.parquet", # a Parquet path or glob, never fully loaded
|
|
62
|
+
workdir="scratch", # where intermediates go: disk, not memory
|
|
63
|
+
vcov={"CRV1": "worker_id"},
|
|
64
|
+
)
|
|
65
|
+
fit.summary()
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Formulas follow [pyfixest](https://github.com/py-econometrics/pyfixest)/fixest
|
|
69
|
+
syntax and are parsed by pyfixest's own parser, so the syntax, the coefficient
|
|
70
|
+
names and the results all match. The difference is where the work happens: rows
|
|
71
|
+
are streamed off disk in batches and never held as a design matrix.
|
|
72
|
+
|
|
73
|
+
## When to use this, and when not to
|
|
74
|
+
|
|
75
|
+
**Use pyfixest.** It is the better tool whenever your data fits in memory:
|
|
76
|
+
mature, broader in scope, more inference options, and no scratch directory to
|
|
77
|
+
think about. If a `pandas` DataFrame of your data fits comfortably in RAM, stop
|
|
78
|
+
reading.
|
|
79
|
+
|
|
80
|
+
**Use this** when it doesn't. The case it was built for is a matched
|
|
81
|
+
employer–employee panel — tens of millions of rows, millions of worker effects,
|
|
82
|
+
an AKM variance decomposition — on a machine or a secure enclave where the data
|
|
83
|
+
is larger than the memory you are allowed. Concretely, reach for it when:
|
|
84
|
+
|
|
85
|
+
- the data does not fit in memory, or fits so tightly that nothing else does;
|
|
86
|
+
- you have one very high-cardinality dimension (workers, patients, students)
|
|
87
|
+
whose groups each touch only a few levels of the others;
|
|
88
|
+
- you want **worker-specific slopes** as well as intercepts (`worker_id[t]`),
|
|
89
|
+
which multiplies the parameter count by the number of slopes;
|
|
90
|
+
- the design is **wide**: hundreds of covariates, such as a categorical
|
|
91
|
+
expanded into fine indicators or many interactions. An in-memory library
|
|
92
|
+
needs rows × covariates in memory before it starts; this needs one batch of
|
|
93
|
+
rows × covariates at a time, with any number of fixed effects, including one
|
|
94
|
+
or none (see [many covariates](#many-covariates));
|
|
95
|
+
- you need a hard, predictable memory ceiling — a shared cluster, or a job that
|
|
96
|
+
must not be the one that gets killed.
|
|
97
|
+
|
|
98
|
+
You are trading memory for disk, and you need scratch space: about five times
|
|
99
|
+
the size of your Parquet input while the fit runs, and only the result files
|
|
100
|
+
once it finishes. See [benchmarks](#benchmarks) for what that buys. Point
|
|
101
|
+
`workdir=` at it, or set the `HDFE_STREAM_WORKDIR` environment variable once
|
|
102
|
+
(for example to a cluster's scratch filesystem) and leave `workdir` out; a
|
|
103
|
+
`workdir` passed to a call takes precedence, and with neither, the system
|
|
104
|
+
temporary directory is used.
|
|
105
|
+
|
|
106
|
+
## How it works, in one paragraph
|
|
107
|
+
|
|
108
|
+
One fixed-effect dimension — normally the largest — is **streamed**: it
|
|
109
|
+
is never represented as a vector in memory. Every other dimension is held as a
|
|
110
|
+
vector sized by its number of *levels*, not rows. Rows are read in batches,
|
|
111
|
+
hash-partitioned into buckets by the streamed dimension, sorted one bucket at a
|
|
112
|
+
time, and reduced to a table of cells (one per combination of fixed-effect
|
|
113
|
+
levels). The reduced normal equations for the non-streamed dimensions are then
|
|
114
|
+
solved by conjugate gradient, and the streamed effects are recovered group by
|
|
115
|
+
group in a final pass that writes residuals and fixed effects straight to
|
|
116
|
+
Parquet. So memory scales with firms × years, not with workers × rows, and that
|
|
117
|
+
asymmetry is the whole design. The covariates travel with the rows: what is
|
|
118
|
+
held at once is one bucket or batch of rows × covariates, never all of them,
|
|
119
|
+
which is why a wide design fits where an in-memory one does not. Full detail is in
|
|
120
|
+
[`hdfe_stream/__init__.py`](https://github.com/leetucker/hdfe-stream/blob/main/hdfe_stream/__init__.py).
|
|
121
|
+
|
|
122
|
+
## Install
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
pip install hdfe-stream[formula] # formula syntax needs pyfixest
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
`polars`, `numpy`, `numba`, `pyarrow` and `scipy` are required. The extras:
|
|
129
|
+
|
|
130
|
+
| extra | what it adds |
|
|
131
|
+
|---|---|
|
|
132
|
+
| `formula` | `feols_stream` formula syntax and pyfixest reporting (pyfixest, formulaic, pandas) |
|
|
133
|
+
| `within` | the `solver="within"` backend |
|
|
134
|
+
| `amg` | algebraic multigrid preconditioning (`precond="amg"`) for hard geometries |
|
|
135
|
+
| `all` | all of the above |
|
|
136
|
+
| `test` | the test suite |
|
|
137
|
+
|
|
138
|
+
Without `formula` you still get `StreamingHDFE`, the lower-level interface that
|
|
139
|
+
takes column names or Polars expressions instead of a formula string.
|
|
140
|
+
|
|
141
|
+
## Features
|
|
142
|
+
|
|
143
|
+
**Fixed effects.** Any number of dimensions, including one (a within
|
|
144
|
+
regression, `y ~ x | worker_id`) and none (OLS with an intercept, `y ~ x`).
|
|
145
|
+
Interactions with `^`
|
|
146
|
+
(`firm_id^year`). Varying slopes on the streamed dimension in fixest syntax
|
|
147
|
+
(`worker_id[t]`, `worker_id[t, t2]`). Connected components of the
|
|
148
|
+
worker–firm graph are computed and reported, and the degrees-of-freedom
|
|
149
|
+
correction counts the fixed-effect parameters that are actually identified
|
|
150
|
+
(`fe_dof="exact"`) or follows pyfixest's convention (`fe_dof="pyfixest"`).
|
|
151
|
+
|
|
152
|
+
**Covariates.** Transformations (`I(age**2)`, `log(x)`), categoricals (`C(x)`,
|
|
153
|
+
string columns), interactions (`:`, `*`), event-study terms
|
|
154
|
+
(`i(year, treat, ref=2009)`). Terms spanned by the fixed effects are dropped
|
|
155
|
+
exactly as pyfixest drops them. Every term compiles to a Polars expression, so
|
|
156
|
+
the design is built while the data streams, one bucket at a time: a categorical
|
|
157
|
+
expanded into hundreds of indicators is carried through the partitioning as the
|
|
158
|
+
one column it comes from.
|
|
159
|
+
|
|
160
|
+
**Standard errors.** `iid`, heteroskedasticity-robust (`hetero`/HC1), and CRV1
|
|
161
|
+
clustered on any column, any `^` interaction, or several dimensions at once
|
|
162
|
+
(multi-way Cameron–Gelbach–Miller, any number of ways). Ask for several at fit
|
|
163
|
+
time with `cluster=` and switch between them afterwards with `with_vcov()` —
|
|
164
|
+
they are all computed in the one residual pass. CRV3, the cluster jackknife
|
|
165
|
+
(`vcov={"CRV3": var}`), for OLS when every fixed effect is nested within the
|
|
166
|
+
clusters or there are none; see [limitations](#limitations).
|
|
167
|
+
|
|
168
|
+
**Estimators.** OLS, weighted least squares (`weights=`, analytic or frequency),
|
|
169
|
+
and 2SLS (`y ~ exog | fe | endog ~ instruments`) with a first-stage F. Poisson
|
|
170
|
+
(`fepois_stream`, with an offset) and logit and probit (`feglm_stream`), with
|
|
171
|
+
weights; see [below](#poisson-logit-and-probit).
|
|
172
|
+
|
|
173
|
+
**Multiple models.** Several outcomes (`y1 + y2 ~ ...`), stepwise covariate sets
|
|
174
|
+
(`sw()`, `csw()`) and stepwise fixed-effect sets. Models sharing a fixed-effect
|
|
175
|
+
set share one pass over the data and one solve.
|
|
176
|
+
|
|
177
|
+
**Leave-out variance components.** The Kline–Saggio–Sølvsten bias correction for
|
|
178
|
+
the AKM decomposition, via `leave_out_kss` — including the leave-one-out
|
|
179
|
+
connected set, Johnson–Lindenstrauss leverages, weights, standard errors with
|
|
180
|
+
95% intervals (`se=True`), KSS's weak-identification diagnostic saying whether
|
|
181
|
+
those intervals are justified, and the interval that stays valid when they are
|
|
182
|
+
not. See [below](#leave-out-variance-components-kss) and
|
|
183
|
+
[docs/kss.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/kss.md).
|
|
184
|
+
|
|
185
|
+
**Output.** Coefficients as a Polars DataFrame (`tidy()`); the model-level
|
|
186
|
+
information of `summary()` (observations, fixed-effect counts, fit statistics,
|
|
187
|
+
solver) as JSON with `summary_json()`, optionally written to a file with
|
|
188
|
+
`summary_json("fit.json")` (`summary_dict()` returns the same as a dict);
|
|
189
|
+
residuals and
|
|
190
|
+
per-row fixed effects as lazy Polars scans, so aggregates like a variance
|
|
191
|
+
decomposition run as a streaming pass; estimated effects per dimension via
|
|
192
|
+
`fixef()`. `to_pyfixest()` converts a result into a pyfixest model for
|
|
193
|
+
`pf.etable`, `pf.summary`, `pf.coefplot` and `pf.iplot`, and `hdfe_stream.etable`
|
|
194
|
+
mixes streaming and in-memory models in one table.
|
|
195
|
+
|
|
196
|
+
**Operations.** Progress, warnings and summaries to a `logging` logger as they
|
|
197
|
+
happen. Each fit works in its own run directory; intermediates are deleted as
|
|
198
|
+
soon as they are no longer needed, everything is removed if the fit fails, and
|
|
199
|
+
`hdfe_stream.cleanup()` sweeps up after killed jobs.
|
|
200
|
+
|
|
201
|
+
## Benchmarks
|
|
202
|
+
|
|
203
|
+
A standard three-way AKM specification,
|
|
204
|
+
`log_earn ~ age_squared + age_cubed | worker_id + firm_id + year`, clustered by
|
|
205
|
+
worker, on the simulated panel that ships with the library. It is compared with
|
|
206
|
+
[pyfixest](https://github.com/py-econometrics/pyfixest), under both of its
|
|
207
|
+
demeaners, and with [xhdfe](https://github.com/reisportela/xhdfe-xfe), on its
|
|
208
|
+
CPU backend. Sizes run from 25,000 to 5,000,000 workers, with firms = workers /
|
|
209
|
+
15 at every size and about 8.5 rows per worker, so the largest panel has 42.5
|
|
210
|
+
million rows. Every configuration agrees with every other on the coefficients
|
|
211
|
+
to within 4e-10, so these are measurements of equally good answers. Reproduce
|
|
212
|
+
with:
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
pip install -e ".[benchmark]" # xhdfe installs separately: see benchmarks/README.md
|
|
216
|
+
python benchmarks/akm_benchmark.py
|
|
217
|
+
python benchmarks/plot_benchmarks.py
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
The numbers behind the figures are in
|
|
221
|
+
[benchmarks/results/akm.csv](https://github.com/leetucker/hdfe-stream/blob/main/benchmarks/results/akm.csv), and
|
|
222
|
+
[benchmarks/README.md](https://github.com/leetucker/hdfe-stream/blob/main/benchmarks/README.md) says how each is measured. In
|
|
223
|
+
brief: wall time includes reading the data, because pyfixest and xhdfe need it in
|
|
224
|
+
memory before they can start and hdfe_stream reads it itself — that difference is
|
|
225
|
+
the comparison, not an artifact. Peak memory is `VmHWM` for the whole process,
|
|
226
|
+
including about 0.5 GB of imports. Each configuration runs in its own process.
|
|
227
|
+
|
|
228
|
+
<picture>
|
|
229
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/akm_time.dark.svg">
|
|
230
|
+
<img alt="AKM regression: wall time against the number of workers, one line per configuration, log-log" src="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/akm_time.light.svg">
|
|
231
|
+
</picture>
|
|
232
|
+
|
|
233
|
+
<picture>
|
|
234
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/akm_memory.dark.svg">
|
|
235
|
+
<img alt="AKM regression: peak memory against the number of workers, one line per configuration, log-log" src="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/akm_memory.light.svg">
|
|
236
|
+
</picture>
|
|
237
|
+
|
|
238
|
+
<picture>
|
|
239
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/akm_disk.dark.svg">
|
|
240
|
+
<img alt="AKM regression: peak disk use of hdfe_stream against the number of workers, log-log" src="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/akm_disk.light.svg">
|
|
241
|
+
</picture>
|
|
242
|
+
|
|
243
|
+
At 5 million workers:
|
|
244
|
+
|
|
245
|
+
| configuration | wall time | peak memory | peak disk |
|
|
246
|
+
|---|---:|---:|---:|
|
|
247
|
+
| pyfixest (MAP, its default) | 888 s | 20.7 GB | — |
|
|
248
|
+
| pyfixest (LSMR) | 150 s | 25.8 GB | — |
|
|
249
|
+
| xhdfe | 126 s | 19.1 GB | — |
|
|
250
|
+
| hdfe_stream (`stream_cg`) | 40 s | 5.4 GB | 3.9 GB |
|
|
251
|
+
| **hdfe_stream (`stream_cg`, low memory)** | **48 s** | **3.7 GB** | 4.0 GB |
|
|
252
|
+
|
|
253
|
+
### Reading the figures
|
|
254
|
+
|
|
255
|
+
**Memory is the point.** At 5 million workers the in-memory libraries peak at
|
|
256
|
+
19–26 GB: pyfixest's LSMR run needed almost all of this machine's 26 GB.
|
|
257
|
+
hdfe_stream peaks at 5.4 GB with its defaults and 3.7 GB with the batch and
|
|
258
|
+
bucket sizes turned down ("low memory": `rows_per_bucket=250_000`,
|
|
259
|
+
`batch_rows=200_000`), for about 20% more time and the same answer. What it
|
|
260
|
+
spends instead is disk: about 90 bytes per row, 4 GB at 42.5 million rows, freed
|
|
261
|
+
when the fit finishes. The in-memory libraries' peak grows in proportion to the
|
|
262
|
+
data. hdfe_stream's low-memory setting grows far more slowly: 0.8 GB at 25,000
|
|
263
|
+
workers, 1.3 GB at a million, 3.7 GB at five million.
|
|
264
|
+
|
|
265
|
+
**It is not slower for it.** From about 400,000 workers up, hdfe_stream is the
|
|
266
|
+
fastest configuration measured. At 5 million workers it takes 40 s against
|
|
267
|
+
xhdfe's 126 s and pyfixest's 150 s (LSMR) or 888 s (MAP). With AKM panel data,
|
|
268
|
+
reducing rows to worker-firm cells before solving more than pays for the disk
|
|
269
|
+
traffic. The simulated panel has about 8.5 rows per worker and one cell per
|
|
270
|
+
row, which suits this approach; a specification where the cell table is no
|
|
271
|
+
smaller than the data, and the fixed effects are less local, may do worse.
|
|
272
|
+
|
|
273
|
+
**Below about 400,000 workers, the in-memory libraries use less memory.**
|
|
274
|
+
Running any hdfe_stream fit costs about 0.8 GB, most of it importing polars,
|
|
275
|
+
numba and scipy and starting Polars' streaming engine and numba's thread pools.
|
|
276
|
+
On small data that floor dominates: at 25,000 workers xhdfe peaks at 0.27 GB and
|
|
277
|
+
pyfixest at about 0.5 GB. This is the concrete version of "use pyfixest (or
|
|
278
|
+
xhdfe) if your data fits".
|
|
279
|
+
|
|
280
|
+
**The default pyfixest demeaner is not the one to compare against.**
|
|
281
|
+
`MapDemeaner` (alternating projections) is about 6x slower than `LsmrDemeaner`
|
|
282
|
+
from 400,000 workers up, since many worker effects are the case alternating
|
|
283
|
+
projections struggles with. If you are comparing, compare against LSMR.
|
|
284
|
+
|
|
285
|
+
**hdfe_stream's `within` solver** uses more memory at scale: 14.7 GB at 5
|
|
286
|
+
million workers, where `stream_cg` and `explicit` stay near 5.4 GB. Prefer
|
|
287
|
+
those when memory is the constraint.
|
|
288
|
+
|
|
289
|
+
### Many covariates
|
|
290
|
+
|
|
291
|
+
The same kind of comparison, holding the data fixed and varying its width: the
|
|
292
|
+
million-worker panel (8.5 million rows, 66,665 firms) with age entered as
|
|
293
|
+
indicators, `log_earn ~ i(age_bin) | worker_id + firm_id`, clustered by worker,
|
|
294
|
+
in bins from five years wide (8 indicators) down to one month (503). hdfe_stream
|
|
295
|
+
runs at its defaults and "sized to the design", with bucket, batch and row-group
|
|
296
|
+
sizes scaled down as the design widens (the rule is below). The numbers are in
|
|
297
|
+
[benchmarks/results/covariates.csv](https://github.com/leetucker/hdfe-stream/blob/main/benchmarks/results/covariates.csv); every
|
|
298
|
+
configuration that finished agrees with every other to within 1e-8.
|
|
299
|
+
|
|
300
|
+
```bash
|
|
301
|
+
python benchmarks/covariates_benchmark.py
|
|
302
|
+
```
|
|
303
|
+
|
|
304
|
+
<picture>
|
|
305
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/covariates_memory.dark.svg">
|
|
306
|
+
<img alt="Many covariates: peak memory against the number of covariates, one line per configuration, log-log" src="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/covariates_memory.light.svg">
|
|
307
|
+
</picture>
|
|
308
|
+
|
|
309
|
+
<picture>
|
|
310
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/covariates_time.dark.svg">
|
|
311
|
+
<img alt="Many covariates: wall time against the number of covariates, one line per configuration, log-log" src="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/covariates_time.light.svg">
|
|
312
|
+
</picture>
|
|
313
|
+
|
|
314
|
+
<picture>
|
|
315
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/covariates_disk.dark.svg">
|
|
316
|
+
<img alt="Many covariates: peak disk use of hdfe_stream against the number of covariates, log-log" src="https://raw.githubusercontent.com/leetucker/hdfe-stream/main/docs/figures/covariates_disk.light.svg">
|
|
317
|
+
</picture>
|
|
318
|
+
|
|
319
|
+
Wall time and peak memory ("—": ran out of this machine's 26 GB):
|
|
320
|
+
|
|
321
|
+
| covariates | pyfixest (MAP) | pyfixest (LSMR) | xhdfe | hdfe_stream | hdfe_stream, sized |
|
|
322
|
+
|---:|---:|---:|---:|---:|---:|
|
|
323
|
+
| 8 | 219 s, 5.7 GB | 17 s, 7.3 GB | 18 s, 9.0 GB | 7 s, 2.8 GB | 7 s, 2.9 GB |
|
|
324
|
+
| 41 | 522 s, 13.1 GB | 56 s, 17.3 GB | 55 s, 18.1 GB | 21 s, 7.0 GB | 20 s, 3.7 GB |
|
|
325
|
+
| 83 | 1,026 s, 24.8 GB | 189 s, 25.4 GB | — | 71 s, 12.5 GB | 56 s, 4.5 GB |
|
|
326
|
+
| 167 | — | — | — | 114 s, 23.6 GB | 112 s, 5.1 GB |
|
|
327
|
+
| 503 | — | — | — | — | **351 s, 8.6 GB** |
|
|
328
|
+
|
|
329
|
+
**The in-memory libraries run out first.** Their peak grows with rows ×
|
|
330
|
+
covariates. At 8.5 million rows pyfixest was within 1 GB of this machine's 26 GB
|
|
331
|
+
at 83 indicators and out of memory at 167; xhdfe was already out of memory at
|
|
332
|
+
83. hdfe_stream sized to the
|
|
333
|
+
design goes from 2.9 GB at 8 indicators to 5.1 GB at 167 and 8.6 GB at 503. Of
|
|
334
|
+
that last figure, about 4.9 GB is the memory-mapped cell table: file pages the
|
|
335
|
+
kernel can drop under pressure. The memory the fit itself allocated peaked at
|
|
336
|
+
4.6 GB (a separate profile, reading the process's anonymous and file-backed
|
|
337
|
+
pages apart).
|
|
338
|
+
|
|
339
|
+
**Size the batches to the design.** At its defaults hdfe_stream holds a whole
|
|
340
|
+
bucket of up to 20 million rows, with every covariate, while it sorts; that is
|
|
341
|
+
fine for a handful of covariates and not for hundreds (23.6 GB at 167, out of
|
|
342
|
+
memory at 503). Every step that holds rows holds all the covariates, so scale
|
|
343
|
+
`rows_per_bucket`, `batch_rows` and `row_group_size` down with the width. The
|
|
344
|
+
benchmark aims at about 1 GB of design per bucket and 250 MB per batch and row
|
|
345
|
+
group, never above the defaults:
|
|
346
|
+
|
|
347
|
+
```python
|
|
348
|
+
per_row = 8 * (k + 4) # k covariates, plus outcome and ids
|
|
349
|
+
batch = min(2_000_000, int(250e6 // per_row))
|
|
350
|
+
fit = feols_stream(fml, data, rows_per_bucket=min(20_000_000, int(1e9 // per_row)),
|
|
351
|
+
batch_rows=batch, row_group_size=min(500_000, batch))
|
|
352
|
+
```
|
|
353
|
+
|
|
354
|
+
**Time grows with the square of the width.** The cross-products cost k × k per
|
|
355
|
+
row; they are formed chunk by chunk as BLAS matrix products, and 503 indicators
|
|
356
|
+
take 6 minutes where 167 take 2. hdfe_stream is the fastest configuration at
|
|
357
|
+
every width measured.
|
|
358
|
+
|
|
359
|
+
**Timings on this machine are noisy where memory is tight.** Each point is one
|
|
360
|
+
run. Repeated runs of the default configuration at 83 indicators, which holds
|
|
361
|
+
12.5 GB, took between 53 and 83 s; the sized configuration varied by about 15%.
|
|
362
|
+
The differences between libraries above are far larger than that.
|
|
363
|
+
|
|
364
|
+
**Disk grows with the width too**, since the working copy of the rows carries
|
|
365
|
+
every covariate: 0.3 GB at 8 indicators, 6 GB at 503.
|
|
366
|
+
|
|
367
|
+
**How the design is built matters.** A formula term such as `i(age_bin)` is
|
|
368
|
+
carried through the partitioning as the one column it is computed from, and the
|
|
369
|
+
indicators are built one bucket at a time. That works for any expression whose
|
|
370
|
+
value on a row depends on that row alone, which formula terms always do. With
|
|
371
|
+
the low-level `StreamingHDFE`, an expression that needs the whole column
|
|
372
|
+
(`pl.col("x") - pl.col("x").mean()`, a window, a shift) is detected and the design
|
|
373
|
+
is built before partitioning instead, with a warning: the estimates are the
|
|
374
|
+
same, the memory is not. Define such a column in the input LazyFrame to avoid
|
|
375
|
+
it. `result.diagnostics["design_evaluated"]` says which happened.
|
|
376
|
+
[examples/wide_designs.py](https://github.com/leetucker/hdfe-stream/blob/main/examples/wide_designs.py) puts this together.
|
|
377
|
+
|
|
378
|
+
### Trading memory for time
|
|
379
|
+
|
|
380
|
+
Same data, same solver, same answer — only how much is in flight at once:
|
|
381
|
+
|
|
382
|
+
```bash
|
|
383
|
+
python benchmarks/akm_benchmark.py --memory-sweep 1000000
|
|
384
|
+
```
|
|
385
|
+
|
|
386
|
+
**8,500,353 rows, 1,000,000 worker effects**
|
|
387
|
+
|
|
388
|
+
| `rows_per_bucket` | `batch_rows` | wall time | peak memory | peak disk |
|
|
389
|
+
|---:|---:|---:|---:|---:|
|
|
390
|
+
| 20,000,000 (default) | 2,000,000 | 7.1 s | 3,861 MB | 779 MB |
|
|
391
|
+
| 1,000,000 | 500,000 | 7.4 s | 1,560 MB | 781 MB |
|
|
392
|
+
| 250,000 | 200,000 | 7.8 s | 1,343 MB | 795 MB |
|
|
393
|
+
| 100,000 | 100,000 | 9.3 s | 1,280 MB | 789 MB |
|
|
394
|
+
|
|
395
|
+
Most of the saving comes from the first step down. The defaults are tuned for a
|
|
396
|
+
machine with room to spare; if memory is the binding constraint, set
|
|
397
|
+
`rows_per_bucket` to something near what you can afford and leave the rest
|
|
398
|
+
alone (with many covariates, scale `batch_rows` and `row_group_size` down too;
|
|
399
|
+
see [many covariates](#many-covariates)). `rhs_block` bounds the other big
|
|
400
|
+
array (levels × variables), and
|
|
401
|
+
`max_s_gb` caps the explicit reduced matrix, above which `solver="auto"` falls
|
|
402
|
+
back to `stream_cg` by itself.
|
|
403
|
+
|
|
404
|
+
Leave-out estimation has a memory knob of its own, `scratch_mb`; see
|
|
405
|
+
[docs/kss.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/kss.md#performance).
|
|
406
|
+
|
|
407
|
+
Measured on an Intel Core Ultra 7 258V, 8 cores, 26 GB RAM, Linux (WSL2);
|
|
408
|
+
Python 3.13.5, polars 1.44.2, numpy 2.5.3, numba 0.67.0, scipy 1.18.1,
|
|
409
|
+
pyfixest 0.60.0, xhdfe 2.28.0 (CPU backend). The exact versions are in
|
|
410
|
+
[benchmarks/results/machine.json](https://github.com/leetucker/hdfe-stream/blob/main/benchmarks/results/machine.json). Numba
|
|
411
|
+
kernels are compiled on first use and cached to disk; the benchmark runs a
|
|
412
|
+
warm-up fit so compilation is not charged to any configuration.
|
|
413
|
+
|
|
414
|
+
## Choosing a solver
|
|
415
|
+
|
|
416
|
+
| `solver` | what it does | when |
|
|
417
|
+
|---|---|---|
|
|
418
|
+
| `"auto"` (default) | `explicit`, falling back to `stream_cg` if the reduced matrix would exceed `max_s_gb` | leave it alone |
|
|
419
|
+
| `"explicit"` | builds the reduced matrix once and solves it in memory | fastest when it fits; its size depends on how levels co-occur, not on rows |
|
|
420
|
+
| `"stream_cg"` | applies the reduced matrix by streaming the cell table each iteration | when the reduced matrix is the thing that does not fit |
|
|
421
|
+
| `"within"` | hands the reduced system to `within-py` | an alternative; no varying-slopes support |
|
|
422
|
+
|
|
423
|
+
The streamed dimension is chosen for you: an explicit `stream=` if given, else
|
|
424
|
+
the dimension carrying varying slopes, else the highest approximate cardinality.
|
|
425
|
+
`result.diagnostics["stream"]` says which and why.
|
|
426
|
+
|
|
427
|
+
## Examples
|
|
428
|
+
|
|
429
|
+
Runnable, on simulated data, no setup — see [examples/](https://github.com/leetucker/hdfe-stream/blob/main/examples/):
|
|
430
|
+
|
|
431
|
+
| | |
|
|
432
|
+
|---|---|
|
|
433
|
+
| [`quickstart.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/quickstart.py) | fit a model and read the results |
|
|
434
|
+
| [`akm_variance.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/akm_variance.py) | the variance decomposition, in a streaming pass |
|
|
435
|
+
| [`formulas.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/formulas.py) | the formula syntax, end to end |
|
|
436
|
+
| [`weights_and_iv.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/weights_and_iv.py) | weights and 2SLS |
|
|
437
|
+
| [`fewer_fixed_effects.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/fewer_fixed_effects.py) | one fixed effect, or none |
|
|
438
|
+
| [`glm.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/glm.py) | Poisson with an offset, logit and probit, separation, and the incidental parameter bias |
|
|
439
|
+
| [`wide_designs.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/wide_designs.py) | hundreds of covariates, and sizing the batches to them |
|
|
440
|
+
| [`varying_slopes.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/varying_slopes.py) | worker-specific trends, and the low-level interface |
|
|
441
|
+
| [`out_of_core.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/out_of_core.py) | memory, disk, solvers, logging |
|
|
442
|
+
| [`reporting.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/reporting.py) | tables and plots via pyfixest |
|
|
443
|
+
| [`leave_out_kss.py`](https://github.com/leetucker/hdfe-stream/blob/main/examples/leave_out_kss.py) | the KSS bias correction, checked against the effects the data was built from |
|
|
444
|
+
|
|
445
|
+
## Simulated data
|
|
446
|
+
|
|
447
|
+
`hdfe_stream.simulate` ships with the package, so you can try the library, or
|
|
448
|
+
check your own code, against data whose answer you know:
|
|
449
|
+
|
|
450
|
+
```python
|
|
451
|
+
from hdfe_stream.simulate import simulate_akm
|
|
452
|
+
|
|
453
|
+
simulate_akm(n_workers=1_000_000).write_parquet("sim.parquet")
|
|
454
|
+
```
|
|
455
|
+
|
|
456
|
+
`simulate_akm` draws worker and firm effects explicitly and positively assorted,
|
|
457
|
+
with enough mobility to connect the firms into one component. `simulate_rich`
|
|
458
|
+
adds categoricals, non-fixed-effect cluster variables, weights, an IV block and
|
|
459
|
+
missing values; `simulate_trends` adds worker-specific time trends. All are
|
|
460
|
+
deterministic given a `seed`.
|
|
461
|
+
|
|
462
|
+
## Poisson, logit and probit
|
|
463
|
+
|
|
464
|
+
```python
|
|
465
|
+
from hdfe_stream import fepois_stream, feglm_stream
|
|
466
|
+
|
|
467
|
+
pois = fepois_stream("visits ~ x | worker_id + firm_id + year", "data/*.parquet",
|
|
468
|
+
offset="log_exposure", workdir="scratch")
|
|
469
|
+
logit = feglm_stream("promoted ~ x | firm_id + year", "data/*.parquet", "logit",
|
|
470
|
+
workdir="scratch")
|
|
471
|
+
```
|
|
472
|
+
|
|
473
|
+
These are fitted by iteratively reweighted least squares, each step a weighted
|
|
474
|
+
version of the linear regression above. Pass 0 sorts the rows once. Each step
|
|
475
|
+
then reads them once, rebuilding every row's linear predictor from the current
|
|
476
|
+
coefficients, and solves the reduced system, starting from the previous step's
|
|
477
|
+
solution.
|
|
478
|
+
|
|
479
|
+
Nothing row-sized is rewritten between steps, so the disk needed is about what
|
|
480
|
+
OLS needs. A fit takes a few times as long as OLS on the same data, since it
|
|
481
|
+
usually needs 6 to 10 steps. Benchmarks against pyfixest, over the AKM
|
|
482
|
+
benchmark's panel sizes and over the number of covariates, are in
|
|
483
|
+
[docs/glm.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/glm.md#benchmarks).
|
|
484
|
+
|
|
485
|
+
Fixed-effect levels whose effect would be infinite are dropped first: all-zero
|
|
486
|
+
outcomes for Poisson, constant ones for logit and probit. Estimates, standard
|
|
487
|
+
errors and deviance match pyfixest to about 1e-12. The few places where they
|
|
488
|
+
deliberately differ are frequency weights, which here mean repeated rows, and
|
|
489
|
+
logit and probit levels whose outcome is all 1, which are dropped. Both are
|
|
490
|
+
described in **[docs/glm.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/glm.md)**, along with the options and results.
|
|
491
|
+
|
|
492
|
+
## Leave-out variance components (KSS)
|
|
493
|
+
|
|
494
|
+
The plug-in AKM decomposition is biased: worker and firm effects are estimated
|
|
495
|
+
with error, which inflates their variances and attenuates their covariance.
|
|
496
|
+
`leave_out_kss` applies the bias correction of
|
|
497
|
+
[Kline, Saggio and Sølvsten (2020)](https://doi.org/10.3982/ECTA16410), streaming
|
|
498
|
+
like the rest of the package:
|
|
499
|
+
|
|
500
|
+
```python
|
|
501
|
+
from hdfe_stream import leave_out_kss
|
|
502
|
+
|
|
503
|
+
lo = leave_out_kss("log_earn ~ age_squared | worker_id + firm_id",
|
|
504
|
+
"data/*.parquet", workdir="scratch", se=True)
|
|
505
|
+
print(lo.summary())
|
|
506
|
+
```
|
|
507
|
+
|
|
508
|
+
It follows Saggio's reference implementation,
|
|
509
|
+
[LeaveOutTwoWay](https://github.com/rsaggio87/LeaveOutTwoWay). It prunes to the
|
|
510
|
+
leave-one-out connected set, leaves out a worker–firm match by default, and
|
|
511
|
+
approximates leverages by random projection. It supports weights, standard
|
|
512
|
+
errors, and KSS's weak-identification diagnostic and q = 1 interval.
|
|
513
|
+
|
|
514
|
+
**[docs/kss.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/kss.md)** covers the options, the standard errors and their
|
|
515
|
+
measured coverage, weak identification, validation, reproducibility, and
|
|
516
|
+
[performance against xhdfe](https://github.com/leetucker/hdfe-stream/blob/main/docs/kss.md#performance): at 5 million workers,
|
|
517
|
+
5.3 GB of memory against xhdfe's 20 GB, with standard errors in 46 minutes
|
|
518
|
+
where xhdfe's did not finish in three hours.
|
|
519
|
+
**[docs/kss_methodological_differences.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/kss_methodological_differences.md)**
|
|
520
|
+
lists every way this implementation differs from LeaveOutTwoWay,
|
|
521
|
+
VarianceComponentsHDFE.jl, xhdfe and pytwoway, with the motivation and measured
|
|
522
|
+
effect of each.
|
|
523
|
+
|
|
524
|
+
## Reproducibility
|
|
525
|
+
|
|
526
|
+
Everything in a fit is deterministic: the same data and the same options give
|
|
527
|
+
the same numbers, with no random component anywhere — to floating-point rounding.
|
|
528
|
+
Repeated fits normally agree bit for bit, but aggregation runs in parallel, and
|
|
529
|
+
once, under heavy CPU load from another process, two identical fits differed in
|
|
530
|
+
the last digit. Treat agreement to about 1e-12 as the guarantee, not bitwise
|
|
531
|
+
identity.
|
|
532
|
+
|
|
533
|
+
The exception is leave-out estimation, which uses random projection and random
|
|
534
|
+
draws. Its results are reproducible from the seed but not deterministic;
|
|
535
|
+
[docs/kss.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/kss.md#reproducibility) says exactly what they depend on and
|
|
536
|
+
what to report in a paper.
|
|
537
|
+
|
|
538
|
+
## Correctness
|
|
539
|
+
|
|
540
|
+
The test suite checks every feature against pyfixest on simulated data —
|
|
541
|
+
coefficients, standard errors under each vcov, residuals, fixed effects, fit
|
|
542
|
+
statistics, the sample kept after dropping missing values, and the exact set of
|
|
543
|
+
coefficient names and dropped collinear terms. Varying slopes have no pyfixest
|
|
544
|
+
equivalent, so they are checked against a brute-force regression with explicit
|
|
545
|
+
worker-by-slope dummies; three-way clustering is checked against an
|
|
546
|
+
inclusion–exclusion sum built from pyfixest's own score matrix. Poisson, logit
|
|
547
|
+
and probit are checked against pyfixest's `fepois` and `feglm`, and their
|
|
548
|
+
frequency weights against pyfixest on the data with each row repeated.
|
|
549
|
+
|
|
550
|
+
```bash
|
|
551
|
+
pip install -e ".[test]"
|
|
552
|
+
pytest
|
|
553
|
+
```
|
|
554
|
+
|
|
555
|
+
## Limitations
|
|
556
|
+
|
|
557
|
+
- **Varying slopes only on the streamed dimension**, and only one dimension may
|
|
558
|
+
carry them.
|
|
559
|
+
- **CRV3 only with nested fixed effects.** The cluster jackknife is computed by
|
|
560
|
+
downdating the fit, which is exact when every fixed effect is nested within
|
|
561
|
+
the clusters (worker effects clustered by worker, firm × year effects by
|
|
562
|
+
firm) or there are none. With fixed effects that are not nested, as in an
|
|
563
|
+
AKM model clustered by worker, the jackknife has to re-estimate the fixed
|
|
564
|
+
effects once per cluster, and the fit refuses; use CRV1, or pyfixest, which
|
|
565
|
+
refits. CRV3 is also one-way and OLS only (pyfixest has no IV CRV3 either).
|
|
566
|
+
No CRV2 and no wild bootstrap.
|
|
567
|
+
- **Poisson, logit and probit:** no CRV3, IV, varying slopes or leave-out
|
|
568
|
+
estimation; no detection of separation by the covariates (pyfixest's `"ir"`
|
|
569
|
+
check), only by the fixed effects; and no correction for the incidental
|
|
570
|
+
parameter bias of logit and probit with fixed effects estimated from few
|
|
571
|
+
observations, as in pyfixest and fixest. During the iterations the streamed
|
|
572
|
+
dimension's effects are held in memory, one float per group per coefficient
|
|
573
|
+
set. See [docs/glm.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/glm.md#not-available).
|
|
574
|
+
- **A converted result cannot recompute its own vcov**, because it holds no
|
|
575
|
+
data. Ask for what you need at fit time via `cluster=`.
|
|
576
|
+
- **Formula support pins a pyfixest range** (`>=0.50,<0.61`), because it uses
|
|
577
|
+
pyfixest's internal formula modules. `StreamingHDFE` has no such dependency.
|
|
578
|
+
- **Needs scratch disk**, roughly the size of your data during the fit. Leave-out
|
|
579
|
+
estimation adds to that: it memory-maps its row-sized accumulators rather than
|
|
580
|
+
holding them, which is what keeps it inside a bounded memory footprint, at
|
|
581
|
+
about 90 bytes per row of scratch while it runs.
|
|
582
|
+
- **Wide designs need the batch sizes scaled down** (see
|
|
583
|
+
[many covariates](#many-covariates)): the defaults hold a whole bucket of rows
|
|
584
|
+
with every covariate. And the cross-products cost k × k per row, so time grows
|
|
585
|
+
with the square of the number of covariates.
|
|
586
|
+
- **Leave-out estimation handles one outcome at a time**, and its standard
|
|
587
|
+
errors omit the split-sample refinement of KSS §4.2 (so they are conservative,
|
|
588
|
+
as their Lemma 5 provides for). Under weak identification it supplies the
|
|
589
|
+
q = 1 interval, not KSS's q > 1 generalization. Leaving out a match, as it
|
|
590
|
+
does by default, var(alpha) has no standard error and the covariance no q = 1
|
|
591
|
+
interval. See [docs/kss.md](https://github.com/leetucker/hdfe-stream/blob/main/docs/kss.md#what-it-does-not-do-yet).
|