hdfe-stream 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,138 @@
1
+ """
2
+ hdfe_stream: out-of-core OLS with several high-dimensional fixed effects, and
3
+ Poisson, logit and probit (`fepois_stream`, `feglm_stream`) by iteratively
4
+ reweighted least squares over the same passes (see glm.py).
5
+
6
+ from hdfe_stream import feols_stream
7
+ fit = feols_stream("log_earn ~ age_squared + age_cubed | worker_id + firm_id + year",
8
+ "data/*.parquet", workdir="hdfe_work", vcov={"CRV1": "worker_id"})
9
+ fit.summary()
10
+
11
+ Formulas follow pyfixest syntax and are parsed with pyfixest's own parser:
12
+ covariates, interactions (`:` and `*`), `I()`, `log()`, `C()`, `i()`,
13
+ several dependent variables (`y1 + y2 ~ ...`), `sw()` / `csw()` stepwise
14
+ syntax, fixed-effect interactions (`firm_id^year`), and IV
15
+ (`y ~ x | fe | endog ~ z`). Every covariate term is compiled to a Polars
16
+ expression, so constructing the design never leaves the streaming engine.
17
+ Weighted least squares via `weights=` (aweights or fweights). The
18
+ lower-level `StreamingHDFE` class takes column names or Polars expressions
19
+ directly.
20
+
21
+ Reporting: `result.to_pyfixest()` returns a pyfixest Feols/Feiv view for
22
+ pf.etable, pf.summary, pf.coefplot and pf.iplot; `hdfe_stream.etable(...)`
23
+ does the conversion for you and accepts ordinary pyfixest models too.
24
+
25
+ Model: y = X b + sum_d gamma_d[level_d(i)] + e for d = 0 ... D-1
26
+
27
+ One fixed-effect dimension (`fe[0]` internally) is *streamed*: it is never
28
+ represented as a vector in memory. It should be the one with the most levels
29
+ whose groups each touch only a few levels of the other dimensions (workers,
30
+ not firms). It is chosen as: the `stream=` option if given; else the
31
+ dimension with varying slopes; else the highest approximate cardinality
32
+ (HyperLogLog counts gathered during pass 0's first scan), ties going to the
33
+ dimension listed first. Every other dimension is represented by level-sized
34
+ vectors in RAM.
35
+
36
+ Varying slopes (FEIS) on the streamed dimension use fixest syntax:
37
+ `| worker_id[t] + firm_id` gives worker effects plus worker-specific slopes on
38
+ t (several: `worker_id[t, t2]`). Projecting out the streamed dimension
39
+ then means removing each worker's own weighted regression on [1, slopes],
40
+ which is still local to the worker.
41
+
42
+ Fewer than two dimensions. With one, it is streamed and there is nothing
43
+ left to solve for: step 2 is skipped and the fit is the within regression.
44
+ With none, there is no dimension to group the rows by, so pass 0 writes them
45
+ unpartitioned and unsorted, passes 1 and 1b and step 2 are skipped, and the
46
+ row passes read plain batches without demeaning; the model keeps an
47
+ "Intercept" coefficient, as in pyfixest.
48
+
49
+ Memory: rows are only ever streamed. The arrays held in RAM are sized by the
50
+ non-streamed dimensions' level counts (times a block of at most `rhs_block`
51
+ variables at a time), plus, for the explicit solver, the sparse reduced
52
+ matrix S, whose size depends on how levels co-occur and not on rows. Polars
53
+ steps stream; the sort and cell group_by run one fe[0] hash bucket at a time.
54
+
55
+ Pipeline
56
+ --------
57
+ Pass 0 (Polars, streaming) evaluate the design as Polars expressions; drop
58
+ rows with missing / non-finite values; factorize
59
+ the non-streamed FE and cluster ids;
60
+ hash-partition rows into fe[0] buckets; sort each
61
+ bucket and assign dense fe[0] codes.
62
+ Pass 1 (Polars, streaming) per bucket: group_by(fe[0], other codes) -> cell
63
+ table with n and sums (plus within-cell
64
+ cross-products when assembly="cells").
65
+ Pass 1b (streamed + numba) keep cells of fe[0] groups with more than one
66
+ cell as memory-mapped arrays; Jacobi diagonal.
67
+ Step 2 (solver, per block) solve S Gamma = D_o' M_0 V, S = D_o' M_0 D_o,
68
+ for all variables V in blocks of `rhs_block`
69
+ columns: "explicit" (sparse S built once, block
70
+ PCG in memory), "stream_cg" (S applied by
71
+ streaming the cells), or "within".
72
+ Step 3 (numba + BLAS) assemble V' M_D V (from cells, or from a row
73
+ pass when there are many variables: numba
74
+ residualizes each chunk in place, BLAS forms its
75
+ cross-products); drop collinear covariates per
76
+ model; beta.
77
+ Step 4 (streamed, numba + per model: stream rows for fe[0] effects,
78
+ BLAS) residuals and RSS (numba), and the bread and the
79
+ meat for HC1 / CRV1 (BLAS, chunk by chunk).
80
+
81
+ Layout
82
+ ------
83
+ api.py `feols_stream`, `fepois_stream`, `feglm_stream`
84
+ estimator.py `StreamingHDFE`: options, layout, the driver
85
+ glm.py `StreamingGLM`: IRLS over the same passes
86
+ families.py GLM families: Poisson, logit, probit
87
+ passes.py pass 0/1/1b (mixin)
88
+ solve.py step 2 (mixin)
89
+ inference.py steps 3/4 (mixin)
90
+ kernels_base.py numba kernels
91
+ kernels_slopes.py numba kernels, varying slopes
92
+ kernels_glm.py numba kernels, GLM cell sums and group effects
93
+ formula.py pyfixest formula -> Polars expressions (input side)
94
+ reporting.py pyfixest etable/coefplot views (output side)
95
+ feterms.py 'worker_id[t]' / 'firm_id^year' parsing
96
+ results.py `HDFEResult`, `HDFEMulti`
97
+ workspace.py run directories, `cleanup()`
98
+ report.py logging / printing
99
+ utils.py small shared helpers
100
+ """
101
+
102
+ from .api import feglm_stream, fepois_stream, feols_stream
103
+ from .estimator import StreamingHDFE
104
+ from .glm import StreamingGLM
105
+ from .leaveout_se import ComponentSE
106
+ from .leaveout_weakid import WeakIdDiagnostics
107
+ from .leaveout import leave_one_out_connected, leave_out_kss
108
+ from .reporting import etable
109
+ from .results import HDFEMulti, HDFEResult
110
+ from .utils import iter_group_chunks
111
+ from .workspace import cleanup
112
+
113
+ __all__ = [
114
+ "feols_stream",
115
+ "fepois_stream",
116
+ "feglm_stream",
117
+ "StreamingHDFE",
118
+ "StreamingGLM",
119
+ "HDFEResult",
120
+ "HDFEMulti",
121
+ "etable",
122
+ "leave_out_kss",
123
+ "ComponentSE",
124
+ "WeakIdDiagnostics",
125
+ "leave_one_out_connected",
126
+ "cleanup",
127
+ "iter_group_chunks",
128
+ ]
129
+
130
+ def _detect_version():
131
+ try:
132
+ from importlib.metadata import version
133
+ return version("hdfe-stream")
134
+ except Exception: # source tree without an installed dist
135
+ return "0.1.0"
136
+
137
+
138
+ __version__ = _detect_version()
hdfe_stream/_types.py ADDED
@@ -0,0 +1,18 @@
1
+ """Type aliases shared by the public signatures."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Any, Mapping, Sequence, Union
7
+
8
+ import polars as pl
9
+
10
+ PathLike = Union[str, Path]
11
+ # a Parquet path or glob, or a Polars LazyFrame
12
+ Source = Union[PathLike, pl.LazyFrame]
13
+ # 'iid', 'hetero', 'HC1', 'CRV1:var', or {'CRV1': 'var'}
14
+ Vcov = Union[str, Mapping[str, str]]
15
+ # column name, (name, expression), or a mapping / sequence of them
16
+ Variables = Union[
17
+ str, tuple, Sequence[Union[str, tuple]], Mapping[str, pl.Expr]]
18
+ Options = Any
@@ -0,0 +1,119 @@
1
+ """The Andrews-Mikusheva interval for a weakly identified quadratic form.
2
+
3
+ Pure numerics, no data: given the leading eigenvalue lambda_1 and the joint
4
+ estimate (b1-hat, theta1-hat) with its 2x2 covariance, return the q = 1
5
+ confidence interval of Kline, Saggio and Solvsten (2020), Section 6.1:
6
+
7
+ C = [ min, max ] of lambda_1 b^2 + t over the ellipse
8
+ { (b1-hat - b, theta1-hat - t) Sigma^-1 (.)' <= z_{alpha,kappa}^2 }
9
+
10
+ Projecting a 2-d ellipse through a quadratic map would ordinarily need the
11
+ chi-squared(2) critical value to control size; Andrews and Mikusheva (2016)
12
+ show the curvature of the map lets it come down, toward the chi-squared(1) value
13
+ when the map is nearly linear. KSS's curvature for q = 1 is
14
+
15
+ kappa = 2 |lambda_1| V[b1] / ( V[theta1]^(1/2) (1 - rho^2)^(1/2) )
16
+
17
+ and z_{alpha,kappa} is the (1 - alpha) quantile of
18
+
19
+ rho(kappa) = sqrt(chi_1^2 + (chi_1' + 1/kappa)^2) - 1/kappa
20
+
21
+ for independent chi variates (Appendix C.6.1). Saggio's reference tabulates that
22
+ quantile by simulation; for q = 1 it has an exact one-dimensional integral form,
23
+ used here. With a, b independent half-normals and c = 1/kappa,
24
+
25
+ rho <= z <=> a^2 + (b + c)^2 <= (z + c)^2
26
+ P(rho <= z) = int_0^z 2 phi(b) [2 Phi(sqrt((z - b)(z + b + 2c))) - 1] db
27
+
28
+ written in factored form so nothing cancels as kappa -> 0. The limits are the
29
+ chi-squared(1) quantile at kappa = 0 and chi-squared(2) as kappa -> infinity.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import numpy as np
35
+
36
+
37
+ def am_cdf(z: float, kappa: float) -> float:
38
+ """P(rho(kappa) <= z) for q = 1."""
39
+ from scipy import integrate, stats
40
+
41
+ if z <= 0:
42
+ return 0.0
43
+ if kappa <= 0:
44
+ return float(2.0 * stats.norm.cdf(z) - 1.0)
45
+ c = 1.0 / kappa
46
+
47
+ def integrand(b):
48
+ inner = np.sqrt(max((z - b) * (z + b + 2.0 * c), 0.0))
49
+ return 2.0 * stats.norm.pdf(b) * (2.0 * stats.norm.cdf(inner) - 1.0)
50
+
51
+ value, _err = integrate.quad(integrand, 0.0, z, epsabs=1e-13, epsrel=1e-12,
52
+ limit=200)
53
+ return float(value)
54
+
55
+
56
+ def am_critical_value(kappa: float, alpha: float = 0.05) -> float:
57
+ """z_{alpha, kappa} for q = 1: between sqrt(chi2_1) and sqrt(chi2_2)."""
58
+ from scipy import optimize, stats
59
+
60
+ lo = float(np.sqrt(stats.chi2.ppf(1.0 - alpha, 1)))
61
+ if kappa <= 0:
62
+ return lo
63
+ hi = float(np.sqrt(stats.chi2.ppf(1.0 - alpha, 2)))
64
+ target = 1.0 - alpha
65
+ # the quantile rises with kappa from lo to hi; bracket just outside both
66
+ return float(optimize.brentq(lambda z: am_cdf(z, kappa) - target,
67
+ lo * (1 - 1e-9), hi * (1 + 1e-9),
68
+ xtol=1e-13, rtol=1e-13))
69
+
70
+
71
+ def curvature(lam1: float, sigma: np.ndarray) -> float:
72
+ """KSS's q = 1 curvature from lambda_1 and the 2x2 covariance."""
73
+ v_b, v_t, cov = float(sigma[0, 0]), float(sigma[1, 1]), float(sigma[0, 1])
74
+ rho2 = cov * cov / (v_b * v_t)
75
+ return 2.0 * abs(lam1) * v_b / np.sqrt(v_t * (1.0 - rho2))
76
+
77
+
78
+ def am_interval(lam1: float, b1: float, theta1: float, sigma: np.ndarray,
79
+ alpha: float = 0.05, grid: int = 4096) -> tuple[float, float, dict[str, float]]:
80
+ """(lower, upper, info) for the q = 1 weak-identification interval.
81
+
82
+ `sigma` is the 2x2 covariance of (b1-hat, theta1-hat), which must be
83
+ positive definite. The extremes of lambda_1 b^2 + t over the ellipse lie on
84
+ its boundary -- the map is linear in t with coefficient one, so any interior
85
+ point can be improved by moving t -- so the problem is one-dimensional in
86
+ the boundary angle. The objective there is a trigonometric polynomial of
87
+ degree two, with at most four stationary points, so a dense grid followed by
88
+ a bounded Brent refinement finds both global extremes. KSS's closed form
89
+ (Appendix C.6.2) solves the same first-order condition as a quartic; the
90
+ tests check the two agree.
91
+ """
92
+ from scipy import optimize
93
+
94
+ sigma = np.asarray(sigma, dtype=np.float64)
95
+ kappa = curvature(lam1, sigma)
96
+ z = am_critical_value(kappa, alpha)
97
+ L = np.linalg.cholesky(sigma)
98
+
99
+ def g(t):
100
+ b = b1 + z * L[0, 0] * np.cos(t)
101
+ th = theta1 + z * (L[1, 0] * np.cos(t) + L[1, 1] * np.sin(t))
102
+ return lam1 * b * b + th
103
+
104
+ ts = np.linspace(0.0, 2.0 * np.pi, grid, endpoint=False)
105
+ values = g(ts)
106
+ step = ts[1] - ts[0]
107
+
108
+ def refine(index, sign):
109
+ t0 = ts[index]
110
+ res = optimize.minimize_scalar(lambda t: sign * g(t),
111
+ bounds=(t0 - step, t0 + step),
112
+ method="bounded",
113
+ options={"xatol": 1e-14})
114
+ return float(g(res.x)) if sign > 0 else float(g(res.x))
115
+
116
+ lower = min(refine(int(np.argmin(values)), 1.0), float(values.min()))
117
+ upper = max(refine(int(np.argmax(values)), -1.0), float(values.max()))
118
+ rho = float(sigma[0, 1] / np.sqrt(sigma[0, 0] * sigma[1, 1]))
119
+ return lower, upper, {"kappa": float(kappa), "z": float(z), "rho": rho}
hdfe_stream/api.py ADDED
@@ -0,0 +1,184 @@
1
+ """`feols_stream`, `fepois_stream`, `feglm_stream`: fit a pyfixest-style
2
+ formula out of core."""
3
+
4
+ from __future__ import annotations
5
+
6
+ from typing import Any, Sequence
7
+
8
+ import polars as pl
9
+
10
+ from ._types import PathLike, Source, Vcov
11
+
12
+ from .estimator import StreamingHDFE
13
+ from .feterms import _parse_fe_term
14
+ from .glm import StreamingGLM
15
+ from .report import _log
16
+ from .formula import _plan_formula, _pyfixest_formula_api
17
+ from .results import HDFEMulti, HDFEResult, _vcov_key
18
+
19
+
20
+ class _FormulaEstimator:
21
+ def __init__(self, fml: str, workdir: PathLike | None, options: dict[str, Any]) -> None:
22
+ self.fml, self.workdir, self.options = fml, workdir, options
23
+
24
+ def fit(self, data: Source, vcov: Vcov | None = None, cluster: Sequence[str] | str = (),
25
+ fe_dof: str = "exact") -> HDFEResult | HDFEMulti:
26
+ return feols_stream(self.fml, data, self.workdir, vcov=vcov, cluster=cluster,
27
+ fe_dof=fe_dof, **self.options)
28
+
29
+
30
+ def feols_stream(fml: str, data: Source, workdir: PathLike | None = None,
31
+ vcov: Vcov | None = None, cluster: Sequence[str] | str = (),
32
+ fe_dof: str = "exact", **options: Any) -> HDFEResult | HDFEMulti:
33
+ """
34
+ Out-of-core OLS with high-dimensional fixed effects from a pyfixest-style
35
+ formula, e.g. "y ~ x1 + i(year, treat, ref=2010) | worker_id + firm_id^year".
36
+ Any number of fixed effects works, including one ("y ~ x | worker_id",
37
+ a within regression) and none ("y ~ x", OLS with an intercept).
38
+
39
+ data : Parquet path/glob or Polars LazyFrame.
40
+ workdir : directory under which run directories are created (default:
41
+ $HDFE_STREAM_WORKDIR if set, else the system temporary
42
+ directory); see StreamingHDFE for the
43
+ outputs=, save_resid= and keep_intermediates= options.
44
+ vcov : as in pyfixest; default {'CRV1': <first FE>} like pyfixest, or
45
+ iid without fixed effects. {'CRV3': var} (one-way, OLS) needs
46
+ every fixed effect nested within the clusters, or none.
47
+ cluster : extra cluster specs to compute CRV1 for (one-way or 'a+b').
48
+ **options : passed to StreamingHDFE (stream, weights, weights_type,
49
+ solver, precond, keep, verbose, logger, log_level, ...).
50
+
51
+ Varying slopes on the streamed dimension: "y ~ x | worker_id[t] + firm_id".
52
+ The worker FE file then has fe_worker_id (intercept) and fe_worker_id[t] (slope)
53
+ columns; the residual file's fe_worker_id column is the worker's total
54
+ contribution for that row.
55
+
56
+ IV formulas ("y ~ x | fe | endog ~ z") are estimated by 2SLS; each result
57
+ carries its first-stage regressions (`first_stage`) and the Wald F of the
58
+ excluded instruments computed with the same vcov (`f_stat_1st_stage`).
59
+
60
+ All models with the same fixed effects share one set of passes and one
61
+ solve, and are estimated on the rows where *all* of their variables are
62
+ non-missing (pyfixest drops missing values model by model).
63
+ """
64
+ def estimators(fe, g):
65
+ yield StreamingHDFE(y=list(g["y"].items()), x=list(g["x"].items()), fe=list(fe),
66
+ workdir=workdir, models=g["models"], **options)
67
+ return _fit_formula(fml, data, estimators, vcov, cluster, fe_dof, options)
68
+
69
+
70
+ def _fit_formula(fml, data, estimators, vcov, cluster, fe_dof, options):
71
+ """Plan the formula, fit each estimator that `estimators(fe, group)`
72
+ makes for each set of fixed effects, and return the results in the
73
+ formula's order."""
74
+ lf = data if isinstance(data, pl.LazyFrame) else pl.scan_parquet(data)
75
+ _log(options.get("verbose", True), f"planning {fml!r} (formula expansion, level discovery)",
76
+ options.get("logger"), options.get("log_level"))
77
+ groups = _plan_formula(fml, lf)
78
+ results = []
79
+ try:
80
+ for fe, g in groups.items():
81
+ # pyfixest's default: cluster by the first fixed effect as written,
82
+ # or iid when there is none
83
+ default = {"CRV1": _parse_fe_term(fe[0])[0]} if fe else "iid"
84
+ for est in estimators(fe, g):
85
+ res = est.fit(lf, vcov=vcov if vcov is not None else default, cluster=cluster,
86
+ fe_dof=fe_dof)
87
+ results += list(res) if isinstance(res, HDFEMulti) else [res]
88
+ except BaseException:
89
+ # a later fit failed: don't leave the earlier fits' files behind
90
+ if not options.get("keep_intermediates"):
91
+ for r in results:
92
+ r.cleanup()
93
+ raise
94
+ Formula, _ = _pyfixest_formula_api()
95
+ order = {s.formula: i for i, s in enumerate(Formula.parse(fml))}
96
+ results.sort(key=lambda r: order.get(r.fml, len(order)))
97
+ return results[0] if len(results) == 1 else HDFEMulti(results)
98
+
99
+
100
+ def fepois_stream(fml: str, data: Source, workdir: PathLike | None = None,
101
+ vcov: Vcov | None = None, cluster: Sequence[str] | str = (),
102
+ fe_dof: str = "exact", offset: str | pl.Expr | None = None,
103
+ iwls_tol: float = 1e-8, iwls_maxiter: int = 25,
104
+ separation_check: bool = True, **options: Any) -> HDFEResult | HDFEMulti:
105
+ """
106
+ Out-of-core Poisson regression (log link) with high-dimensional fixed
107
+ effects, from a pyfixest-style formula, e.g.
108
+ "trade ~ log_dist | exporter^year + importer^year". Estimated by
109
+ iteratively reweighted least squares; each step is one read of the rows
110
+ and one solve of the fixed effects (see `StreamingGLM`).
111
+
112
+ data, workdir, vcov, cluster, fe_dof, **options : as in `feols_stream`,
113
+ except that the vcov may not be CRV3, and IV formulas and
114
+ varying slopes are not available. weights= and weights_type=
115
+ work as in pyfixest's fepois.
116
+ offset : column name added to the linear predictor with a coefficient of
117
+ one (e.g. log exposure).
118
+ iwls_tol, iwls_maxiter : convergence tolerance on the relative change in
119
+ deviance, and the most IRLS steps (pyfixest's defaults).
120
+ separation_check : drop fixed-effect levels whose outcome is zero on
121
+ every row (their effect would be minus infinity), as pyfixest's
122
+ "fe" check, which it runs by default. Covariates that separate
123
+ the outcome (its "ir" check) are not detected.
124
+
125
+ The dependent variable must be nonnegative. Each dependent variable is
126
+ its own set of IRLS fits (the separation check depends on it); models of
127
+ the same outcome share pass 0.
128
+ """
129
+ return _fit_glm(fml, data, "poisson", workdir, vcov, cluster, fe_dof,
130
+ dict(options, offset=offset, iwls_tol=iwls_tol, iwls_maxiter=iwls_maxiter,
131
+ separation_check=separation_check))
132
+
133
+
134
+ def feglm_stream(fml: str, data: Source, family: str, workdir: PathLike | None = None,
135
+ vcov: Vcov | None = None, cluster: Sequence[str] | str = (),
136
+ fe_dof: str = "exact", iwls_tol: float = 1e-8, iwls_maxiter: int = 25,
137
+ separation_check: bool = True, **options: Any) -> HDFEResult | HDFEMulti:
138
+ """
139
+ Out-of-core logit or probit regression with high-dimensional fixed
140
+ effects, from a pyfixest-style formula: `family` is "logit" or "probit".
141
+ Estimated by iteratively reweighted least squares, with step-halving, as
142
+ pyfixest's feglm (see `StreamingGLM`).
143
+
144
+ Arguments as in `fepois_stream`, without the offset. The dependent
145
+ variable must be 0 or 1. The separation check drops fixed-effect levels
146
+ whose outcome is the same on every row (all 0 or all 1), repeating until
147
+ none is left; pyfixest's "fe" check drops only levels whose outcome is
148
+ all 0, once.
149
+
150
+ Weights (weights=, weights_type=) multiply each row's log-likelihood, as
151
+ in fepois; pyfixest's feglm takes none. Frequency weights give exactly
152
+ the fit of the data with each row repeated that many times.
153
+
154
+ No correction is made for the incidental parameter problem: with fixed
155
+ effects estimated from few observations each, the coefficients are
156
+ biased, as in pyfixest and fixest.
157
+ """
158
+ if str(family).lower() not in ("logit", "probit"):
159
+ raise ValueError(f"family must be 'logit' or 'probit', got {family!r}"
160
+ + ("; use fepois_stream" if str(family).lower() == "poisson" else "")
161
+ + ("; for a linear model use feols_stream"
162
+ if str(family).lower() == "gaussian" else ""))
163
+ return _fit_glm(fml, data, str(family).lower(), workdir, vcov, cluster, fe_dof,
164
+ dict(options, iwls_tol=iwls_tol, iwls_maxiter=iwls_maxiter,
165
+ separation_check=separation_check))
166
+
167
+
168
+ def _fit_glm(fml, data, family, workdir, vcov, cluster, fe_dof, options):
169
+ key = _vcov_key(vcov)
170
+ if key is not None and key.startswith("CRV3:"):
171
+ raise ValueError("CRV3 standard errors are not available for GLMs; use CRV1")
172
+
173
+ def estimators(fe, g):
174
+ by_y = {}
175
+ for model in g["models"]:
176
+ if model.get("iv"):
177
+ raise ValueError(f"{model['fml']}: IV (2SLS) is not available for GLMs")
178
+ by_y.setdefault(model["y"], []).append(model)
179
+ for yname, models in by_y.items():
180
+ xs = list(dict.fromkeys(x for model in models for x in model["x"]))
181
+ yield StreamingGLM(y=[(yname, g["y"][yname])], x=[(x, g["x"][x]) for x in xs],
182
+ fe=list(fe), family=family, workdir=workdir, models=models,
183
+ **options)
184
+ return _fit_formula(fml, data, estimators, vcov, cluster, fe_dof, options)