pitbacktest 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,243 @@
1
+ """Public-data adapter: build a Panel from yfinance.
2
+
3
+ Dependency: `pip install yfinance` (not needed by the core, used only by this adapter)
4
+
5
+ Usage
6
+ from pitbacktest.adapters.yfinance import load_panel
7
+ panel = load_panel(["AAPL", "MSFT", "NVDA"], "2018-01-01", "2024-12-31")
8
+
9
+ Warning: the limits of free data
10
+ - **Survivorship bias**: a ticker list given as of today leaves out securities delisted since.
11
+ It is not a point-in-time universe, so performance is biased upward. Use it for exploration only.
12
+ - **Adjusted prices**: auto_adjust=True applies dividends and splits. Without it a split day
13
+ creates a fake return.
14
+ - Firm characteristics (ROA, book-to-market, ...) are not provided. Running without chars raises a warning,
15
+ and it is a legitimate one: "price controls alone cannot filter out someone else's alpha".
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import warnings
21
+
22
+ import numpy as np
23
+ import pandas as pd
24
+
25
+ from ..core.panel import Panel, build_pit_eligible
26
+
27
+
28
+ def load_panel(tickers: list[str], start: str, end: str, *,
29
+ min_dollar_volume: float = 1e6,
30
+ market: str = "US", entry_lag: int = 1,
31
+ with_market_cap: bool = True,
32
+ with_chars: bool = False,
33
+ shares_outstanding: dict[str, float] | None = None) -> Panel:
34
+ """Build a (date x ticker) panel from yfinance.
35
+
36
+ tickers securities to look up. **Passing only securities listed today creates survivorship bias.**
37
+ min_dollar_volume minimum average traded value for eligibility (20 days)
38
+ with_market_cap market cap from get_shares_full (share count over time) x close.
39
+ Switches on the size and turnover controls. One API call per security, so somewhat slow.
40
+ with_chars builds the 5 firm characteristics from quarterly financials (log_size, log_bm, momentum, roa, asset_growth).
41
+ **The key input of the neutralisation test.** Two more API calls per security.
42
+ shares_outstanding manual {ticker: shares}. A fallback when with_market_cap fails.
43
+ """
44
+ try:
45
+ import yfinance as yf
46
+ except ImportError as e: # pragma: no cover
47
+ raise ImportError(
48
+ "yfinance is required: pip install yfinance"
49
+ ) from e
50
+
51
+ raw = yf.download(tickers, start=start, end=end, auto_adjust=True,
52
+ progress=False, group_by="column")
53
+ if raw is None or raw.empty:
54
+ raise ValueError("yfinance returned no data (check the tickers and the period)")
55
+
56
+ def pick(field: str) -> pd.DataFrame | None:
57
+ if isinstance(raw.columns, pd.MultiIndex):
58
+ if field not in raw.columns.get_level_values(0):
59
+ return None
60
+ d = raw[field]
61
+ else:
62
+ if field not in raw.columns:
63
+ return None
64
+ d = raw[[field]].rename(columns={field: tickers[0]})
65
+ return d.sort_index()
66
+
67
+ close = pick("Close")
68
+ if close is None:
69
+ raise ValueError("there is no Close column")
70
+ close.index = pd.to_datetime(close.index).tz_localize(None)
71
+ volume = pick("Volume")
72
+ high, low, open_ = pick("High"), pick("Low"), pick("Open")
73
+ for d in (volume, high, low, open_):
74
+ if d is not None:
75
+ d.index = close.index
76
+
77
+ mkt_cap = None
78
+ if with_market_cap:
79
+ mkt_cap = _market_cap(yf, list(close.columns), close)
80
+ if mkt_cap is None and shares_outstanding:
81
+ so = pd.Series(shares_outstanding).reindex(close.columns)
82
+ if so.notna().any():
83
+ mkt_cap = close.mul(so, axis=1)
84
+ if mkt_cap is None:
85
+ warnings.warn("Market cap could not be built: the size and turnover controls are switched off", stacklevel=2)
86
+
87
+ chars = {}
88
+ if with_chars:
89
+ if mkt_cap is None:
90
+ warnings.warn("There is no market cap, so firm characteristics cannot be built", stacklevel=2)
91
+ else:
92
+ chars = _characteristics(yf, close, mkt_cap)
93
+
94
+ eligible = build_pit_eligible(close, min_adv=min_dollar_volume, volume=volume)
95
+ n_drop = int((~eligible).sum().sum())
96
+ if eligible.values.sum() == 0:
97
+ raise ValueError("eligible is all False: lower min_dollar_volume")
98
+
99
+ warnings.warn(
100
+ f"The yfinance panel is **not a point-in-time universe**. Only tickers listed today are looked up, so "
101
+ f"delisted securities are missing and performance is biased upward. ({n_drop:,} cells excluded)",
102
+ stacklevel=2)
103
+
104
+ ppy = 365 if market.upper() == "CRYPTO" else 252 # a 24/7 market has 365 trading days a year
105
+ n_ended = sum(1 for c in close.columns if close[c].last_valid_index() is not None
106
+ and close[c].last_valid_index() < close.index[-1] - pd.Timedelta(days=30))
107
+ if len(close.columns) >= 30 and n_ended == 0:
108
+ warnings.warn("No security in this panel has a price that stops inside the period: it holds only securities alive today, "
109
+ "so it has survivorship bias (yfinance hardly returns past prices of delisted or acquired securities). "
110
+ "See docs/survivorship.md.", stacklevel=2)
111
+ return Panel(close=close, eligible=eligible, open=open_, high=high, low=low,
112
+ volume=volume, mkt_cap=mkt_cap, chars=chars,
113
+ market=market, entry_lag=entry_lag, periods_per_year=ppy)
114
+
115
+
116
+ def _market_cap(yf, tickers: list[str], close: pd.DataFrame) -> pd.DataFrame | None:
117
+ """Share count over time x close. get_shares_full reflects changes in shares outstanding (buybacks, issuance)."""
118
+ cols = {}
119
+ for t in tickers:
120
+ try:
121
+ sh = yf.Ticker(t).get_shares_full(start=str(close.index[0].date()))
122
+ except Exception: # noqa: BLE001, PERF203
123
+ continue
124
+ if sh is None or not len(sh):
125
+ continue
126
+ s = pd.Series(sh)
127
+ s.index = pd.to_datetime(s.index).tz_localize(None)
128
+ s = s[~s.index.duplicated(keep="last")].sort_index()
129
+ cols[t] = s.reindex(close.index, method="ffill")
130
+ if not cols:
131
+ return None
132
+ shares = pd.DataFrame(cols).reindex(columns=close.columns)
133
+ got = int(shares.notna().any().sum())
134
+ if got < len(tickers):
135
+ warnings.warn(f"Received shares for only {got}/{len(tickers)} securities: "
136
+ f"for the rest the market cap is NaN, so they are left out of the controls", stacklevel=3)
137
+ return close * shares
138
+
139
+
140
+ def _characteristics(yf, close: pd.DataFrame, mkt_cap: pd.DataFrame) -> dict:
141
+ """The five firm characteristics: standard Fama-French style definitions.
142
+
143
+ Warning: yfinance's financial coverage is short: 5 to 7 quarters and 5 years of annual data only.
144
+ A long backtest such as 2018 to 2024 would be empty for most of the period, so quarterly and annual data are combined and
145
+ **a characteristic with too little coverage is dropped altogether** (an empty column ruins the regression).
146
+ For long studies attach a paid financial source (Compustat, Sharadar and the like).
147
+
148
+ Disclosure lag: yfinance gives no filing date, so **period end + 45 business days** is used as an approximation.
149
+ """
150
+ LAG = 45
151
+ MIN_COVER = 0.30 # a characteristic is dropped if its share of valid cells is below this
152
+ ta, be, ni = {}, {}, {}
153
+ for t in close.columns:
154
+ try:
155
+ tk = yf.Ticker(t)
156
+ bs, inc = tk.quarterly_balance_sheet, tk.quarterly_income_stmt
157
+ except Exception: # noqa: BLE001, PERF203
158
+ continue
159
+ if bs is None or bs.empty:
160
+ continue
161
+
162
+ def grab(df, keys):
163
+ if df is None or df.empty:
164
+ return None
165
+ for k in keys:
166
+ hit = [i for i in df.index if k in str(i)]
167
+ if hit:
168
+ s = df.loc[hit[0]].dropna()
169
+ s.index = pd.to_datetime(s.index).tz_localize(None)
170
+ return s.sort_index()
171
+ return None
172
+
173
+ # combine quarterly (last 5 to 7) and annual (last 5 years) to widen the coverage
174
+ try:
175
+ abs_, ainc = tk.balance_sheet, tk.income_stmt
176
+ except Exception: # noqa: BLE001
177
+ abs_ = ainc = None
178
+
179
+ def merge(q_, a_):
180
+ parts = [x for x in (q_, a_) if x is not None and len(x)]
181
+ if not parts:
182
+ return None
183
+ m = pd.concat(parts)
184
+ return m[~m.index.duplicated(keep="first")].sort_index()
185
+
186
+ a = merge(grab(bs, ["Total Assets"]), grab(abs_, ["Total Assets"]))
187
+ e = merge(grab(bs, ["Stockholders Equity", "Common Stock Equity"]),
188
+ grab(abs_, ["Stockholders Equity", "Common Stock Equity"]))
189
+ n = merge(grab(inc, ["Net Income From Continuing Operation Net Minority Interest",
190
+ "Net Income"]),
191
+ grab(ainc, ["Net Income From Continuing Operation Net Minority Interest",
192
+ "Net Income"]))
193
+ for src, dst in ((a, ta), (e, be), (n, ni)):
194
+ if src is not None and len(src):
195
+ dst[t] = src
196
+
197
+ def to_daily(d: dict) -> pd.DataFrame | None:
198
+ if not d:
199
+ return None
200
+ df = pd.DataFrame(d)
201
+ # use from the period end + LAG business days (approximating the disclosure lag)
202
+ df.index = df.index + pd.tseries.offsets.BDay(LAG)
203
+ return df.reindex(close.index.union(df.index)).ffill().reindex(close.index)\
204
+ .reindex(columns=close.columns)
205
+
206
+ TA, BE, NI = to_daily(ta), to_daily(be), to_daily(ni)
207
+ if TA is None and BE is None:
208
+ warnings.warn("Quarterly financials could not be fetched, so the firm characteristics are empty", stacklevel=3)
209
+ return {}
210
+ out = {"log_size": np.log(mkt_cap.clip(lower=1)),
211
+ "momentum": close.pct_change(252) - close.pct_change(21)}
212
+ if BE is not None:
213
+ out["log_bm"] = np.log((BE / mkt_cap.replace(0, np.nan)).clip(lower=1e-6))
214
+ if NI is not None and TA is not None:
215
+ out["roa"] = NI / TA.replace(0, np.nan)
216
+ if TA is not None:
217
+ out["asset_growth"] = TA.pct_change(252)
218
+ # drop characteristics with low coverage: an empty column wipes out the regression sample
219
+ kept, dropped = {}, {}
220
+ for k, v in out.items():
221
+ cover = float(v.notna().values.mean())
222
+ (kept if cover >= MIN_COVER else dropped)[k] = cover
223
+ if cover >= MIN_COVER:
224
+ kept[k] = v
225
+ dropped_names = {k: f"{c*100:.0f}%" for k, c in dropped.items()}
226
+ kept_out = {k: out[k] for k in out if k not in dropped}
227
+ if dropped_names:
228
+ warnings.warn(
229
+ f"Firm characteristics {list(dropped_names)} were dropped because their coverage ({dropped_names}) is low "
230
+ f"(yfinance provides only 5 to 7 quarters and 5 years of annual data). "
231
+ f"For long backtests attach a paid financial source.", stacklevel=3)
232
+ got = {k: f"{float(v.notna().values.mean())*100:.0f}%" for k, v in kept_out.items()}
233
+ warnings.warn(f"Firm characteristic valid cells: {got} | period-end + {LAG} business-day lag applied", stacklevel=3)
234
+ return kept_out
235
+
236
+
237
+ def sp500_tickers() -> list[str]:
238
+ """Wikipedia's current S&P 500 constituents. **A list as of today, so it has survivorship bias.**"""
239
+ try:
240
+ t = pd.read_html("https://en.wikipedia.org/wiki/List_of_S%26P_500_companies")[0]
241
+ return sorted(t["Symbol"].astype(str).str.replace(".", "-", regex=False).tolist())
242
+ except Exception as e: # pragma: no cover # noqa: BLE001
243
+ raise RuntimeError(f"Could not fetch the S&P 500 list: {e}") from e
@@ -0,0 +1,234 @@
1
+ """Reading a result: alpha and beta, information coefficients, and bootstrap intervals for the Sharpe ratio.
2
+
3
+ Three questions a reviewer asks first, each with the defaults that avoid the usual mistakes.
4
+
5
+ alpha_beta Is the return alpha, or exposure to something I could have bought? (OLS with Newey-West errors)
6
+ information_coefficient Does the signal rank next-period returns? (daily rank correlation; delisted names kept)
7
+ sharpe_ci / sharpe_diff_ci How precise is a Sharpe ratio, and is the difference between two of them real?
8
+ (stationary block bootstrap, so autocorrelation in returns is respected)
9
+
10
+ None of these proves a strategy works. They say how much the data can tell apart.
11
+
12
+ References
13
+ Newey, West (1987, 1994): heteroskedasticity and autocorrelation consistent covariance, plug-in lag rule.
14
+ Politis, Romano (1994): the stationary bootstrap.
15
+ Lo (2002): the Sharpe ratio's sampling error is large and depends on autocorrelation.
16
+ """
17
+ from __future__ import annotations
18
+
19
+ import math
20
+ import warnings
21
+ from statistics import NormalDist
22
+
23
+ import numpy as np
24
+ import pandas as pd
25
+
26
+ _N = NormalDist()
27
+
28
+
29
+ # --------------------------------------------------------------------------------------------------- alpha and beta
30
+ def _align(y: pd.Series, X: pd.DataFrame) -> tuple[pd.Series, pd.DataFrame]:
31
+ df = pd.concat([y.rename("__y__"), X], axis=1, join="inner").replace([np.inf, -np.inf], np.nan).dropna()
32
+ return df["__y__"], df.drop(columns="__y__")
33
+
34
+
35
+ def alpha_beta(returns: pd.Series, factors: pd.Series | pd.DataFrame, *, periods_per_year: int = 252,
36
+ lag: int | None = None, min_obs: int = 60) -> dict:
37
+ """Regress `returns` on `factors` (for example the market) with an intercept.
38
+
39
+ Standard errors are Newey-West (Bartlett kernel, no small-sample correction), with the lag from the
40
+ Newey-West (1994) plug-in rule `int(4 * (T / 100) ** (2 / 9))` unless `lag` is given. The t-statistics are
41
+ asymptotic, compared with a normal distribution. **They are still too optimistic in finite samples**: with
42
+ AR(1) errors of 0.5 and 500 observations, a true alpha of zero is rejected at |t| > 1.96 in about 9% of datasets
43
+ instead of 5% (the uncorrected figure is 26%; `tests/test_analytics.py` A4). Read |t| below about 2.5 as no evidence. Dates are aligned on the intersection and rows with a missing
44
+ value are dropped; at least `min_obs` rows (and 3 per coefficient) must remain.
45
+
46
+ `alpha_annual` is the per-period intercept times `periods_per_year` (arithmetic, not compounded).
47
+ `appraisal` is `alpha_annual` over the annualised residual volatility.
48
+ Net returns include costs and funding while a benchmark does not, so a small negative alpha is expected for a
49
+ strategy with no edge.
50
+ """
51
+ y = pd.Series(returns).astype(float)
52
+ F = factors.to_frame(factors.name or "factor") if isinstance(factors, pd.Series) else factors.astype(float)
53
+ if F.shape[1] == 0:
54
+ raise ValueError("no factor columns")
55
+ y, F = _align(y, F)
56
+ T, k = len(y), F.shape[1] + 1
57
+ if T < max(min_obs, 3 * k):
58
+ raise ValueError(f"only {T} usable observations for {k} coefficients (need {max(min_obs, 3 * k)})")
59
+ if (F.std(ddof=0) <= 1e-12 * np.maximum(1.0, F.abs().mean())).any(): # relative: identical floats can have std ~1e-18
60
+ raise ValueError("a factor column is constant")
61
+ X = np.column_stack([np.ones(T), F.to_numpy(float)])
62
+ if np.linalg.matrix_rank(X) < k:
63
+ raise ValueError("factor columns are collinear")
64
+ Y = y.to_numpy(float)
65
+ coef, *_ = np.linalg.lstsq(X, Y, rcond=None)
66
+ e = Y - X @ coef
67
+ L = int(4 * (T / 100) ** (2 / 9)) if lag is None else int(lag)
68
+ L = max(0, min(L, T - 1))
69
+ Xe = X * e[:, None]
70
+ S = Xe.T @ Xe
71
+ for l in range(1, L + 1):
72
+ G = Xe[l:].T @ Xe[:-l]
73
+ S += (1.0 - l / (L + 1.0)) * (G + G.T)
74
+ bread = np.linalg.inv(X.T @ X)
75
+ V = bread @ S @ bread
76
+ se = np.sqrt(np.diag(V))
77
+ with np.errstate(divide="ignore", invalid="ignore"):
78
+ t = coef / se # a perfect fit has zero error: t is inf or nan
79
+ sse = float(e @ e)
80
+ sst = float(((Y - Y.mean()) ** 2).sum())
81
+ resid_vol = float(e.std(ddof=k) * math.sqrt(periods_per_year))
82
+ ann = coef[0] * periods_per_year
83
+
84
+ def p(tv: float) -> float:
85
+ return float(2 * (1 - _N.cdf(abs(tv)))) if np.isfinite(tv) else float("nan")
86
+
87
+ return {"alpha": float(coef[0]), "alpha_annual": float(ann), "alpha_t": float(t[0]), "alpha_p": p(t[0]),
88
+ "betas": {c: {"beta": float(coef[i + 1]), "t": float(t[i + 1])} for i, c in enumerate(F.columns)},
89
+ "r2": float(1 - sse / sst) if sst > 0 else float("nan"), "n": T, "lag": L,
90
+ "resid_vol_annual": resid_vol, "appraisal": float(ann / resid_vol) if resid_vol > 0 else float("nan")}
91
+
92
+
93
+ # ----------------------------------------------------------------------------------------- information coefficient
94
+ def forward_returns(panel, h: int, *, delist_return: float | None = None) -> pd.DataFrame:
95
+ """Compounded return from close(t + lag) to close(t + lag + h), for a signal at the close of t.
96
+
97
+ A name with no price on a day earns 0 that day (it is treated as cash after its last bar), the same rule as
98
+ `backtest_portfolio`; with `delist_return` set, the day after a delisting flagged in `panel.delist_after` earns
99
+ that return instead. Dropping such names (a NaN forward return) would silently remove the losers from the test.
100
+ Windows that run past the end of the sample are NaN."""
101
+ ret = np.nan_to_num(panel.ret1().to_numpy(float), nan=0.0)
102
+ if delist_return is not None and panel.delist_after is not None:
103
+ da = panel.delist_after.reindex(index=panel.dates, columns=panel.tickers).fillna(False).to_numpy(bool)
104
+ nxt = np.zeros(ret.shape, dtype=bool)
105
+ nxt[1:] = da[:-1]
106
+ ret = np.where(nxt, float(delist_return), ret)
107
+ T = ret.shape[0]
108
+ g = np.ones(ret.shape)
109
+ for k in range(1, h + 1):
110
+ s = panel.entry_lag + k
111
+ sh = np.full(ret.shape, np.nan)
112
+ if s < T:
113
+ sh[: T - s] = 1.0 + ret[s:]
114
+ g = g * sh
115
+ return pd.DataFrame(g - 1.0, index=panel.dates, columns=panel.tickers)
116
+
117
+
118
+ def information_coefficient(factor: pd.DataFrame, forward: pd.DataFrame, eligible: pd.DataFrame, *,
119
+ min_obs: int = 20) -> pd.Series:
120
+ """Daily cross-sectional Spearman correlation between the factor and the forward return, among eligible names
121
+ with both values. Days with fewer than `min_obs` names, or no variation, are NaN."""
122
+ f = factor.reindex(index=forward.index, columns=forward.columns)
123
+ el = eligible.reindex(index=forward.index, columns=forward.columns).fillna(False).astype(bool)
124
+ ok = el & f.notna() & forward.notna()
125
+ rf = f.where(ok).rank(axis=1)
126
+ ry = forward.where(ok).rank(axis=1)
127
+ n = ok.sum(axis=1).to_numpy(float)
128
+ a, b = rf.to_numpy(float), ry.to_numpy(float)
129
+ with np.errstate(invalid="ignore", divide="ignore"), warnings.catch_warnings():
130
+ warnings.simplefilter("ignore", RuntimeWarning) # all-NaN rows are expected (days with no signal)
131
+ a = a - np.nanmean(a, axis=1, keepdims=True)
132
+ b = b - np.nanmean(b, axis=1, keepdims=True)
133
+ num = np.nansum(a * b, axis=1)
134
+ den = np.sqrt(np.nansum(a * a, axis=1) * np.nansum(b * b, axis=1))
135
+ ic = np.where((n >= min_obs) & (den > 0), num / den, np.nan)
136
+ return pd.Series(ic, index=forward.index, name="ic")
137
+
138
+
139
+ def ic_report(panel, factor: pd.DataFrame, horizons=(1, 5, 20), *, delist_return: float | None = None,
140
+ min_obs: int = 20) -> dict:
141
+ """IC summary per horizon: mean IC, its standard deviation, ICIR (mean over standard deviation), the share of
142
+ positive days, and a Newey-West t-statistic of the mean with lag = horizon (the forward windows overlap)."""
143
+ from .core.estimators import newey_west_t
144
+ out = {}
145
+ for h in horizons:
146
+ ic = information_coefficient(factor, forward_returns(panel, h, delist_return=delist_return),
147
+ panel.eligible, min_obs=min_obs).dropna()
148
+ if len(ic) < 30:
149
+ out[h] = {"n_days": int(len(ic)), "mean_ic": float("nan"), "std_ic": float("nan"), "icir": float("nan"),
150
+ "hit_rate": float("nan"), "t_nw": float("nan")}
151
+ continue
152
+ mu, _, t, T = newey_west_t(ic.to_numpy(), lag=max(1, h))
153
+ sd = float(ic.std(ddof=1))
154
+ out[h] = {"n_days": int(T), "mean_ic": float(mu), "std_ic": sd, "icir": float(mu / sd) if sd > 0 else float("nan"),
155
+ "hit_rate": float((ic > 0).mean()), "t_nw": float(t)}
156
+ return out
157
+
158
+
159
+ # --------------------------------------------------------------------------------------------- bootstrap intervals
160
+ def _stationary_indices(T: int, n: int, mean_block: float, rng: np.random.Generator) -> np.ndarray:
161
+ """(n, T) resampling indices for the stationary bootstrap: blocks of geometric length (mean `mean_block`),
162
+ wrapping around the end of the sample."""
163
+ p = 1.0 / max(1.0, mean_block)
164
+ start = rng.integers(0, T, size=(n, T))
165
+ new = rng.random((n, T)) < p
166
+ idx = np.empty((n, T), dtype=np.int64)
167
+ idx[:, 0] = start[:, 0]
168
+ for t in range(1, T):
169
+ idx[:, t] = np.where(new[:, t], start[:, t], (idx[:, t - 1] + 1) % T)
170
+ return idx
171
+
172
+
173
+ def _sharpe_rows(R: np.ndarray, ppy: int) -> np.ndarray:
174
+ return R.mean(axis=1) / R.std(axis=1, ddof=1) * math.sqrt(ppy)
175
+
176
+
177
+ def _default_block(T: int) -> float:
178
+ return float(max(1, round(T ** (1 / 3))))
179
+
180
+
181
+ def sharpe_ci(returns: pd.Series, *, periods_per_year: int = 252, n: int = 2000, mean_block: float | None = None,
182
+ level: float = 0.95, seed: int = 0) -> dict:
183
+ """Percentile interval for the annualised Sharpe ratio from a stationary block bootstrap.
184
+
185
+ `mean_block` defaults to `round(T ** (1/3))`, a common rule of thumb; use a larger one if returns are strongly
186
+ autocorrelated. Intervals from short samples tend to be too narrow, so treat a lower bound close to zero with
187
+ caution. Sharpe uses risk-free 0 and the sample standard deviation (ddof=1)."""
188
+ x = pd.Series(returns).dropna().to_numpy(float)
189
+ T = len(x)
190
+ if T < 30:
191
+ raise ValueError(f"{T} observations is too few for a bootstrap")
192
+ b = _default_block(T) if mean_block is None else float(mean_block)
193
+ rng = np.random.default_rng(seed)
194
+ sr = []
195
+ for c in _chunks(n):
196
+ sr.append(_sharpe_rows(x[_stationary_indices(T, c, b, rng)], periods_per_year))
197
+ sr = np.concatenate(sr)
198
+ a = (1 - level) / 2
199
+ return {"sharpe": float(x.mean() / x.std(ddof=1) * math.sqrt(periods_per_year)),
200
+ "lo": float(np.quantile(sr, a)), "hi": float(np.quantile(sr, 1 - a)), "se": float(sr.std(ddof=1)),
201
+ "n_boot": int(n), "mean_block": b, "level": level, "T": T}
202
+
203
+
204
+ def sharpe_diff_ci(a: pd.Series, b: pd.Series, *, periods_per_year: int = 252, n: int = 2000,
205
+ mean_block: float | None = None, level: float = 0.95, seed: int = 0) -> dict:
206
+ """Sharpe(b) - Sharpe(a) with a **paired** stationary bootstrap: the same resampled dates are used for both
207
+ series, so the common movement of the two cancels, which is what makes a small difference detectable.
208
+ `p_boot` is twice the smaller tail of the bootstrap distribution around zero (a two-sided percentile p-value,
209
+ approximate). Series are aligned on their common dates."""
210
+ df = pd.concat([pd.Series(a).rename("a"), pd.Series(b).rename("b")], axis=1, join="inner").dropna()
211
+ T = len(df)
212
+ if T < 30:
213
+ raise ValueError(f"{T} common observations is too few for a bootstrap")
214
+ xa, xb = df["a"].to_numpy(float), df["b"].to_numpy(float)
215
+ bl = _default_block(T) if mean_block is None else float(mean_block)
216
+ rng = np.random.default_rng(seed)
217
+ d = []
218
+ for c in _chunks(n):
219
+ idx = _stationary_indices(T, c, bl, rng)
220
+ d.append(_sharpe_rows(xb[idx], periods_per_year) - _sharpe_rows(xa[idx], periods_per_year))
221
+ d = np.concatenate(d)
222
+ al = (1 - level) / 2
223
+ est = float((xb.mean() / xb.std(ddof=1) - xa.mean() / xa.std(ddof=1)) * math.sqrt(periods_per_year))
224
+ p = min(1.0, 2 * min(float((d <= 0).mean()), float((d >= 0).mean())))
225
+ return {"diff": est, "lo": float(np.quantile(d, al)), "hi": float(np.quantile(d, 1 - al)), "se": float(d.std(ddof=1)),
226
+ "p_boot": p, "n_boot": int(n), "mean_block": bl, "level": level, "T": T}
227
+
228
+
229
+ def _chunks(n: int, size: int = 400):
230
+ left = n
231
+ while left > 0:
232
+ c = min(size, left)
233
+ yield c
234
+ left -= c
File without changes
@@ -0,0 +1,126 @@
1
+ """Controls: 7 price-based and 5 firm characteristics.
2
+
3
+ Why firm characteristics are the default
4
+ Testing with price controls alone makes you **mistake someone else's alpha for your own**. Without controlling for
5
+ profitability and value, the apparent alpha often shrinks a lot or flips sign
6
+ (T6 in tests/test_synthetic.py feeds a control in as the factor and reproduces this disappearance).
7
+ Price controls (mom20/mom120) alone cannot capture standard momentum (12-1 months), profitability or value.
8
+ So chars are **included automatically** when present, and a warning is raised when they are not.
9
+
10
+ Normalisation
11
+ Everything is a per-date percentile rank -> mean 0, variance 1. Robust to outliers and lets a coefficient be read as 'bp per 1 SD'.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import warnings
17
+
18
+ import numpy as np
19
+ import pandas as pd
20
+
21
+ EPS = 1e-12
22
+ PRICE_CONTROLS = ("rev1", "rev5", "mom21", "mom63", "mom252_21", "logsize", "turnover", "vol21")
23
+ CHAR_NAMES = ("log_size", "log_bm", "momentum", "roa", "asset_growth")
24
+
25
+
26
+ def xs_rank(v: pd.DataFrame, eligible: pd.DataFrame) -> pd.DataFrame:
27
+ """Per-date percentile rank [0,1] among the eligible. A cross-sectional operation, so the market-wide component drops out by itself."""
28
+ return v.where(eligible).rank(axis=1, pct=True, na_option="keep")
29
+
30
+
31
+ def xs_norm(v: pd.DataFrame, eligible: pd.DataFrame) -> pd.DataFrame:
32
+ """Rank -> mean 0, variance 1 scale."""
33
+ return ((xs_rank(v, eligible) - 0.5) * np.sqrt(12.0)).astype(np.float32)
34
+
35
+
36
+ def build_controls(panel, *, include_chars: bool = True) -> dict[str, pd.DataFrame]:
37
+ """The bundle of controls. Firm characteristics are included when panel.chars exists."""
38
+ close, el = panel.close, panel.eligible
39
+ ret = close.pct_change()
40
+ out: dict[str, pd.DataFrame] = {
41
+ # rev1 is required: the order flow of mean-reversion traders is a proxy for 'the day's fall'.
42
+ # The 1-day reversal is the strongest short-term effect, so leaving it out makes you rediscover reversal and call it a signal.
43
+ "rev1": ret,
44
+ "rev5": close.pct_change(5),
45
+ "mom21": close.pct_change(21),
46
+ "mom63": close.pct_change(63),
47
+ "vol21": ret.rolling(21, min_periods=21).std(),
48
+ }
49
+ # standard momentum 12-1: mom21/63 alone cannot capture it. Its lookback is one year of bars; a panel shorter than that would give a
50
+ # control that is NaN everywhere, and every date would silently drop out of the regressions, so it is left out (with a warning).
51
+ lookback = int(getattr(panel, "periods_per_year", 252))
52
+ if lookback < len(close) - 21:
53
+ out["mom252_21"] = close.pct_change(lookback) - close.pct_change(21)
54
+ else:
55
+ warnings.warn(f"The 12-1 momentum control needs {lookback} bars of history and the panel has {len(close)}: it is left out of the controls.",
56
+ stacklevel=2)
57
+ if panel.mkt_cap is not None:
58
+ out["logsize"] = np.log(panel.mkt_cap.replace(0, np.nan))
59
+ adv = panel.adv(20)
60
+ if adv is not None:
61
+ out["turnover"] = adv / (panel.mkt_cap + EPS)
62
+
63
+ if include_chars:
64
+ if not panel.chars:
65
+ warnings.warn(
66
+ "Firm characteristics (chars) are missing. With price controls alone the ROA and book-to-market exposures are not captured, so "
67
+ "'someone else's alpha' can be mistaken for the factor's own. "
68
+ "Fill chars through an adapter, or state include_chars=False explicitly.",
69
+ stacklevel=2,
70
+ )
71
+ for k, v in panel.chars.items():
72
+ out[f"char_{k}"] = v
73
+
74
+ return {k: xs_norm(v, el).astype(np.float32) for k, v in out.items() if v is not None}
75
+
76
+
77
+ def neutralize(factor: pd.DataFrame, controls: dict[str, pd.DataFrame],
78
+ eligible: pd.DataFrame, *, min_obs: int = 50) -> pd.DataFrame:
79
+ """Per-date cross-sectional regression residual: orthogonalised against the controls.
80
+
81
+ For continuous factors. Event (0/1) signals use the dummy regression in event.py.
82
+ A day whose residual is below 1e-5 of the factor's size (the controls explain the factor almost completely) stays NaN.
83
+ """
84
+ fv = factor.values.astype(np.float64)
85
+ ev = eligible.values
86
+ cvs = [c.values.astype(np.float64) for c in controls.values()]
87
+ out = np.full(fv.shape, np.nan)
88
+ for i in range(fv.shape[0]):
89
+ ok = ev[i] & np.isfinite(fv[i])
90
+ for c in cvs:
91
+ ok &= np.isfinite(c[i])
92
+ if ok.sum() < min_obs:
93
+ continue
94
+ y = fv[i][ok]
95
+ X = np.column_stack([np.ones(ok.sum())] + [c[i][ok] for c in cvs])
96
+ try:
97
+ Q, _ = np.linalg.qr(X)
98
+ except np.linalg.LinAlgError:
99
+ continue
100
+ res = y - Q @ (Q.T @ y)
101
+ # A factor inside the span of the controls leaves only rounding noise (about 1e-8 of its size in float32).
102
+ # Left in, a later rank-normalisation would blow that noise up to unit scale and the result would depend on the
103
+ # BLAS build. No information is left, so the day stays NaN.
104
+ if not (res.std() > 1e-5 * y.std()):
105
+ continue
106
+ out[i][ok] = res
107
+ return pd.DataFrame(out, index=factor.index, columns=factor.columns)
108
+
109
+
110
+ def build_chars_from_financials(close: pd.DataFrame, mkt_cap: pd.DataFrame,
111
+ book_equity: pd.DataFrame | None = None,
112
+ net_income: pd.DataFrame | None = None,
113
+ total_assets: pd.DataFrame | None = None) -> dict:
114
+ """Five firm characteristics: standard Fama-French style definitions.
115
+
116
+ Quarterly financials are passed already forward-filled to daily (the adapter's job).
117
+ """
118
+ chars = {"log_size": np.log(mkt_cap.clip(lower=1)),
119
+ "momentum": close.pct_change(252) - close.pct_change(21)}
120
+ if book_equity is not None:
121
+ chars["log_bm"] = np.log((book_equity / mkt_cap.replace(0, np.nan)).clip(lower=1e-6))
122
+ if net_income is not None and total_assets is not None:
123
+ chars["roa"] = net_income / total_assets.replace(0, np.nan)
124
+ if total_assets is not None:
125
+ chars["asset_growth"] = total_assets.pct_change(252)
126
+ return chars