pitbacktest 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,80 @@
1
+ """Trading costs and capacity for crypto perpetuals.
2
+
3
+ There is no public order-book history in the archive, so costs here are *estimates*, and they are labelled as such.
4
+ - Fees: a flat taker fee (default 5 bp, the standard retail tier). Change it to your tier.
5
+ - Spread: a flat assumption plus a thin-contract penalty. A daily high-low estimator (Corwin-Schultz) was tried first
6
+ and rejected because it is invalid for crypto volatility (see `liquidity_cost_bp`).
7
+ - Capacity: instead of pretending to model market impact without data, `participation_report` shows how large the
8
+ trades are relative to each contract's turnover for a given account size.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import warnings
13
+
14
+ import numpy as np
15
+ import pandas as pd
16
+
17
+ from ..core.costs import corwin_schultz
18
+
19
+
20
+ def liquidity_cost_bp(panel, *, taker_fee_bp: float = 5.0, half_spread_bp: float = 2.0, window: int = 30,
21
+ thin_usd: float = 1e8, thin_extra_bp: float = 5.0,
22
+ spread_estimator: str = "fixed", tick_size=None) -> pd.DataFrame:
23
+ """(date x ticker) one-way cost in bp = taker fee + half spread (+ a penalty for thin contracts).
24
+
25
+ spread_estimator
26
+ "fixed" a flat half spread (`half_spread_bp`). The default. It is an assumption, not a measurement, so
27
+ use `cost_sensitivity` or a grid of values instead of trusting one number.
28
+ "corwin_schultz" half the Corwin-Schultz estimate from daily highs and lows. **Do not use it for crypto.**
29
+ On Binance perpetuals it gave a median spread of about 1.5% and about 38 bp one way for
30
+ BTC, orders of magnitude above any real quote (the study in `studies/crypto_cross_section`
31
+ documents this in Amendment 1). I did not establish why; daily crypto ranges are dominated
32
+ by jumps and volatility clustering, which the estimator's assumptions do not cover. It is
33
+ kept only to reproduce the preregistered run.
34
+
35
+ "tick" half of one price tick as a share of the price: `0.5 * tick_size / close * 1e4` bp (`tick_size` is a Series by ticker,
36
+ for example `ArchiveStore.symbol_rules()["tick_size"]`; a ticker without one raises). Measured against the order book
37
+ (docs/crypto_spread_check.md): the quoted half spread was one tick for 9 of 12 sampled contracts (0.8 to 1.25 times the
38
+ prediction). The 3 others were 10, 10 and 100 times wider: the exchange has cut their tick since, and `tick_size` is
39
+ today's value, so there it is a **floor**. It is the cost of a small order at the best quote, not of a large one:
40
+ impact is not in it.
41
+
42
+ Contracts whose trailing turnover is below `thin_usd` pay `thin_extra_bp` more, a blunt stand-in for impact.
43
+ Uses only data up to day t."""
44
+ adv = panel.adv(window)
45
+ extra = (adv < thin_usd).astype(float) * thin_extra_bp if adv is not None else 0.0
46
+ if spread_estimator == "corwin_schultz":
47
+ warnings.warn("Corwin-Schultz overstates crypto spreads by orders of magnitude; use only to reproduce old runs.",
48
+ stacklevel=2)
49
+ cs = corwin_schultz(panel.high, panel.low)
50
+ half = (cs.rolling(window, min_periods=10).median() * 0.5 * 1e4).clip(lower=1.0)
51
+ elif spread_estimator == "tick":
52
+ if tick_size is None:
53
+ raise ValueError("spread_estimator='tick' needs tick_size, a Series by ticker")
54
+ tk = pd.Series(tick_size, dtype=float).reindex(panel.tickers)
55
+ if tk.isna().any() or (tk <= 0).any():
56
+ raise ValueError(f"tick_size is missing or not positive for {int((tk.isna() | (tk <= 0)).sum())} tickers "
57
+ f"(first: {list(tk.index[(tk.isna() | (tk <= 0)).to_numpy()][:3])}); a delisted contract has none in the exchange rules")
58
+ half = 0.5 * tk / panel.close * 1e4
59
+ elif spread_estimator == "fixed":
60
+ half = half_spread_bp
61
+ else:
62
+ raise ValueError(spread_estimator)
63
+ return (taker_fee_bp + half + extra).astype(np.float64)
64
+
65
+
66
+ def participation_report(panel, weights: np.ndarray, aum_usd: float, window: int = 30) -> dict:
67
+ """Trade size as a share of trailing daily turnover, for an account of `aum_usd`.
68
+
69
+ weights (date x ticker) target weights, e.g. the `holdings` of a portfolio. Returns percentiles of
70
+ |delta weight| * aum / ADV over all trades, and the account size at which the 95th percentile would reach 1%
71
+ of a contract's daily turnover (a common ceiling for not moving the market)."""
72
+ adv = panel.adv(window).values
73
+ dw = np.abs(np.diff(weights, axis=0))
74
+ part = dw * aum_usd / np.where(adv[1:] > 0, adv[1:], np.nan)
75
+ part = part[(dw > 1e-9) & np.isfinite(part)]
76
+ if part.size == 0:
77
+ return {"aum_usd": aum_usd, "n_trades": 0}
78
+ p50, p95, p99 = (float(np.percentile(part, q)) for q in (50, 95, 99))
79
+ return {"aum_usd": aum_usd, "n_trades": int(part.size), "participation_p50": p50, "participation_p95": p95,
80
+ "participation_p99": p99, "aum_at_1pct_p95": float(aum_usd * 0.01 / p95) if p95 > 0 else np.inf}
@@ -0,0 +1,326 @@
1
+ """Intraday bars for USDT perpetuals from the Binance public archive, as a point-in-time Panel.
2
+
3
+ Read `docs/intraday_design.md` first. What this layer can say: how fast an edge decays with latency, and at what cost it stops
4
+ paying. What it cannot say: whether the signal can be traded. The archive has bars, not a book, so there is no queue position, no
5
+ partial fill and no impact below the bar.
6
+
7
+ Time convention. Rows are indexed by bar **end** time (UTC, naive). A row holds only what was known at that instant, so a signal
8
+ from row t is entered at row t + lag with lag >= 1.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import concurrent.futures as cf
13
+ import io
14
+ import zipfile
15
+ from dataclasses import replace
16
+ from pathlib import Path
17
+
18
+ import numpy as np
19
+ import pandas as pd
20
+
21
+ from ..core.panel import Panel
22
+ from ..portfolio import backtest_portfolio
23
+ from . import binance_archive as ba
24
+
25
+ COLS = ["open", "high", "low", "close", "volume", "quote_volume", "trades"]
26
+ MIN_PER_DAY = 1440
27
+
28
+
29
+ # ------------------------------------------------------------------------------------------------------ parsing
30
+ def parse_minute_zip(blob: bytes) -> pd.DataFrame:
31
+ """One archive file of 1-minute klines -> frame indexed by the bar **end** time (open time + 1 minute).
32
+ Millisecond and microsecond timestamps are told apart per row; header rows and duplicates are dropped."""
33
+ raw = ba._read_klines_zip(blob)
34
+ if raw.empty:
35
+ return pd.DataFrame(columns=COLS)
36
+ ts = raw["open_time"].astype("int64").to_numpy()
37
+ ms = np.where(ts > 10**14, ts // 1000, ts)
38
+ idx = (pd.to_datetime(ms, unit="ms", utc=True).tz_localize(None).astype("datetime64[ns]") + pd.Timedelta(minutes=1)) # ns in every pandas
39
+ df = raw[COLS].apply(pd.to_numeric, errors="coerce")
40
+ df.index = idx
41
+ df = df[~df.index.duplicated(keep="last")].sort_index()
42
+ return df.dropna(subset=["close"])
43
+
44
+
45
+ def _bar_minutes(bar: str) -> int:
46
+ m = int(pd.Timedelta(bar) / pd.Timedelta(minutes=1))
47
+ if m < 1 or MIN_PER_DAY % m:
48
+ raise ValueError(f"bar {bar!r} must be a whole number of minutes that divides a day (1min, 5min, 15min, 1h, ...)")
49
+ return m
50
+
51
+
52
+ def aggregate_bars(df1m: pd.DataFrame, bar: str) -> pd.DataFrame:
53
+ """1-minute frame (end-time index) -> bars labelled by their end. Open is the first minute's, close the last, high and low the
54
+ extremes, volumes summed; `n_min` is how many minutes the bar holds. A last bar that would end after the data ends is dropped."""
55
+ m = _bar_minutes(bar)
56
+ if df1m.empty:
57
+ return pd.DataFrame(columns=COLS + ["n_min"])
58
+ if m == 1:
59
+ out = df1m.copy()
60
+ out["n_min"] = 1
61
+ return out
62
+ opn = df1m.copy()
63
+ opn.index = opn.index - pd.Timedelta(minutes=1) # back to open time so bars align to the clock
64
+ g = opn.resample(f"{m}min", label="left", closed="left", origin="epoch")
65
+ out = pd.DataFrame({"open": g["open"].first(), "high": g["high"].max(), "low": g["low"].min(), "close": g["close"].last(),
66
+ "volume": g["volume"].sum(min_count=1), "quote_volume": g["quote_volume"].sum(min_count=1),
67
+ "trades": g["trades"].sum(min_count=1), "n_min": g["close"].count()})
68
+ out = out[out["n_min"] > 0]
69
+ out.index = out.index + pd.Timedelta(minutes=m) # label by end
70
+ return out[out.index <= df1m.index.max()]
71
+
72
+
73
+ # ------------------------------------------------------------------------------------------------------ download
74
+ def _month_keys(start: pd.Timestamp, end: pd.Timestamp) -> list[pd.Period]:
75
+ return list(pd.period_range(start.to_period("M"), end.to_period("M"), freq="M"))
76
+
77
+
78
+ def _paths(store, sym: str) -> Path:
79
+ d = store.dir / "minutes" / sym
80
+ d.mkdir(parents=True, exist_ok=True)
81
+ return d
82
+
83
+
84
+ def fetch_minutes(store, symbols, start, end, *, workers: int = 6, warmup_days: int = 91, getter=None, today=None) -> dict:
85
+ """Download 1-minute klines for `symbols` from `start - warmup_days` to `end`. Complete months come from the monthly archive,
86
+ the current partial month from the daily archive. Resumable: a month or day already cached, or recorded as missing (404), is
87
+ not requested again. `getter(key) -> bytes | None` can be injected for tests; `today` is the current UTC date."""
88
+ get = getter or (lambda key: ba._get(f"{ba.DL_URL}/{key}"))
89
+ today = pd.Timestamp(today) if today is not None else pd.Timestamp.utcnow().tz_localize(None).normalize()
90
+ lo = pd.Timestamp(start).normalize() - pd.Timedelta(days=warmup_days)
91
+ hi = min(pd.Timestamp(end).normalize(), today - pd.Timedelta(days=1))
92
+ cur = today.to_period("M")
93
+ stats = {"months": 0, "days": 0, "missing": 0, "cached": 0}
94
+
95
+ def one(sym: str) -> dict:
96
+ s = {"months": 0, "days": 0, "missing": 0, "cached": 0}
97
+ d = _paths(store, sym)
98
+ for per in _month_keys(lo, hi):
99
+ if per < cur: # a complete month
100
+ f, none = d / f"{per}.pkl", d / f"{per}.none"
101
+ if f.exists() or none.exists():
102
+ s["cached"] += 1
103
+ continue
104
+ blob = get(f"data/futures/um/monthly/klines/{sym}/1m/{sym}-1m-{per}.zip")
105
+ if blob is None:
106
+ none.write_text("")
107
+ s["missing"] += 1
108
+ continue
109
+ parse_minute_zip(blob).to_pickle(f) # parsed fully first: a bad file raises and caches nothing
110
+ s["months"] += 1
111
+ else: # the partial month, one file per finished day
112
+ for day in pd.date_range(max(lo, per.start_time), min(hi, per.end_time.normalize())):
113
+ f, none = d / f"d{day:%Y-%m-%d}.pkl", d / f"d{day:%Y-%m-%d}.none"
114
+ if f.exists() or none.exists():
115
+ s["cached"] += 1
116
+ continue
117
+ blob = get(f"data/futures/um/daily/klines/{sym}/1m/{sym}-1m-{day:%Y-%m-%d}.zip")
118
+ if blob is None:
119
+ none.write_text("")
120
+ s["missing"] += 1
121
+ continue
122
+ parse_minute_zip(blob).to_pickle(f)
123
+ s["days"] += 1
124
+ return s
125
+
126
+ with cf.ThreadPoolExecutor(workers) as ex:
127
+ for r in ex.map(one, list(symbols)):
128
+ for k in stats:
129
+ stats[k] += r[k]
130
+ return stats
131
+
132
+
133
+ def load_minutes(store, sym: str, start, end) -> pd.DataFrame:
134
+ """Cached 1-minute frame for `sym` between `start` (inclusive) and `end` (inclusive day), end-time index."""
135
+ d = _paths(store, sym)
136
+ parts = [pd.read_pickle(f) for f in sorted(d.glob("*.pkl"))]
137
+ parts = [p for p in parts if len(p)]
138
+ if not parts:
139
+ return pd.DataFrame(columns=COLS)
140
+ df = pd.concat(parts)
141
+ df = df[~df.index.duplicated(keep="last")].sort_index()
142
+ return df[(df.index > pd.Timestamp(start)) & (df.index <= pd.Timestamp(end).normalize() + pd.Timedelta(days=1))]
143
+
144
+
145
+ def parse_funding_zip(blob: bytes) -> pd.DataFrame:
146
+ """One monthly fundingRate archive file -> DataFrame[calc_time (ms or us), rate]."""
147
+ with zipfile.ZipFile(io.BytesIO(blob)) as z:
148
+ dd = pd.read_csv(z.open(z.namelist()[0]), dtype=str)
149
+ if dd.shape[1] < 3:
150
+ return pd.DataFrame(columns=["calc_time", "rate"])
151
+ dd.columns = ["calc_time", "interval", "rate"][: dd.shape[1]] + list(dd.columns[3:])
152
+ return dd[["calc_time", "rate"]]
153
+
154
+
155
+ def funding_series(frames: list[pd.DataFrame]) -> pd.Series:
156
+ """Settlements with timestamps rounded to the nearest minute (the archive stamps them a few milliseconds late)."""
157
+ out = pd.concat(frames) if frames else pd.DataFrame(columns=["calc_time", "rate"])
158
+ t = pd.to_numeric(out["calc_time"], errors="coerce")
159
+ r = pd.to_numeric(out["rate"], errors="coerce")
160
+ ok = t.notna() & r.notna()
161
+ t, r = t[ok].astype("int64").to_numpy(), r[ok].to_numpy(float)
162
+ ms = np.where(t > 10**14, t // 1000, t)
163
+ idx = pd.to_datetime(ms, unit="ms", utc=True).tz_localize(None).astype("datetime64[ns]").round("min")
164
+ return pd.Series(r, index=idx, dtype=float).groupby(level=0).sum().sort_index()
165
+
166
+
167
+ def fetch_funding_events(store, sym: str, *, lister=None, getter=None) -> pd.Series:
168
+ """Funding settlements with their instants, cached. The daily store sums them per day and loses the instant, which an
169
+ intraday simulation needs. `lister(sym) -> keys` and `getter(key) -> bytes | None` can be injected for tests."""
170
+ f = store.dir / "funding_events" / f"{sym}.pkl"
171
+ f.parent.mkdir(parents=True, exist_ok=True)
172
+ if f.exists():
173
+ return pd.read_pickle(f)
174
+ keys = lister(sym) if lister else store._months(sym, "fundingRate")
175
+ get = getter or (lambda key: ba._get(f"{ba.DL_URL}/{key}"))
176
+ frames = [parse_funding_zip(blob) for k in keys if (blob := get(k))]
177
+ s = funding_series(frames)
178
+ s.to_pickle(f)
179
+ return s
180
+
181
+
182
+ # ------------------------------------------------------------------------------------------------------ the panel
183
+ def estimate_memory(n_bars: int, n_symbols: int, n_matrices: int = 20) -> float:
184
+ """Rough peak gigabytes: `backtest_portfolio` peaked at about 14 (bar x contract) float64 matrices when measured on a panel
185
+ with close, eligible and volume (tests/memprobe in the design notes); a panel from `build_intraday_panel` also carries open,
186
+ high, low, funding and delisting flags, so 20 is used. Measure again if you change the engine."""
187
+ return n_bars * n_symbols * 8 * n_matrices / 1e9
188
+
189
+
190
+ def build_intraday_panel(store, symbols, start, end, *, bar: str = "5min", adv_window: int = 30, min_age_days: int = 60,
191
+ min_adv_usd: float = 5e6, top_n: int | None = None, entry_lag: int = 1, max_gb: float = 3.0,
192
+ funding: bool = True, funding_events: dict | None = None) -> Panel:
193
+ """Point-in-time intraday panel. Eligibility for every bar of day D uses only days up to D - 1. Download first with
194
+ `fetch_minutes(store, symbols, start, end)`, which also loads the warm-up days."""
195
+ symbols = list(symbols)
196
+ if entry_lag < 1:
197
+ raise ValueError("entry_lag must be at least 1 bar: a bar's close is only known at its end, so it cannot be traded on")
198
+ m = _bar_minutes(bar)
199
+ start, end = pd.Timestamp(start).normalize(), pd.Timestamp(end).normalize()
200
+ warm = max(adv_window, min_age_days) + 1
201
+ lo = start - pd.Timedelta(days=warm)
202
+ grid_end = end + pd.Timedelta(days=1)
203
+ grid = pd.date_range(start + pd.Timedelta(minutes=m), grid_end, freq=f"{m}min").astype("datetime64[ns]") # pandas 3 defaults to us
204
+ mem = estimate_memory(len(grid), len(symbols))
205
+ if mem > max_gb:
206
+ fit = [b for b in ("1min", "5min", "15min", "30min", "1h", "2h", "4h") if _bar_minutes(b) >= m and
207
+ estimate_memory(len(grid) * m // _bar_minutes(b), len(symbols)) <= max_gb]
208
+ raise ValueError(f"about {mem:.1f} GB needed for {len(grid):,} bars x {len(symbols)} contracts (limit {max_gb}); "
209
+ f"use a longer bar ({fit[0] if fit else 'fewer contracts or a shorter period'}), fewer contracts or a shorter period")
210
+ bars: dict[str, pd.DataFrame] = {}
211
+ for s in symbols:
212
+ d = _paths(store, s)
213
+ missing = [p for p in _month_keys(lo, min(end, pd.Timestamp.utcnow().tz_localize(None).normalize() - pd.Timedelta(days=1)))
214
+ if p < pd.Timestamp.utcnow().tz_localize(None).to_period("M")
215
+ and not (d / f"{p}.pkl").exists() and not (d / f"{p}.none").exists()]
216
+ if missing:
217
+ raise ValueError(f"{s}: months not downloaded {[str(p) for p in missing[:4]]}; run fetch_minutes(store, symbols, start, end) "
218
+ f"first (it loads {warm} warm-up days before the start)")
219
+ raw = load_minutes(store, s, lo, end)
220
+ if raw.empty:
221
+ continue
222
+ b = aggregate_bars(raw, bar)
223
+ bars[s] = b
224
+ if not bars:
225
+ raise ValueError("no data for any symbol")
226
+ cols = sorted(bars)
227
+
228
+ # ---- terminal zero-volume tail (a halted or delisted contract keeps printing flat bars): drop only that tail ------------
229
+ zero_interior = 0
230
+ for s in cols:
231
+ b = bars[s]
232
+ live = np.flatnonzero((b["quote_volume"] > 0).to_numpy())
233
+ if len(live) == 0:
234
+ bars[s] = b.iloc[0:0]
235
+ continue
236
+ zero_interior += int((b["quote_volume"].iloc[: live[-1] + 1] <= 0).sum())
237
+ bars[s] = b.iloc[: live[-1] + 1]
238
+ cols = [s for s in cols if len(bars[s])]
239
+
240
+ # ---- daily liquidity, then eligibility from the PREVIOUS day -----------------------------------------------------
241
+ day_of = lambda ix: (ix - pd.Timedelta(minutes=1)).floor("D")
242
+ dq = pd.DataFrame({s: bars[s]["quote_volume"].groupby(day_of(bars[s].index)).sum() for s in cols})
243
+ dq = dq.reindex(pd.date_range(dq.index.min(), max(dq.index.max(), end), freq="D"))
244
+ has = dq.fillna(0) > 0
245
+ age = has.cumsum()
246
+ adv = dq.where(has).rolling(adv_window, min_periods=adv_window).median()
247
+ ok_day = (age >= min_age_days) & (adv >= min_adv_usd)
248
+ if top_n:
249
+ rank = adv.where(ok_day).rank(axis=1, ascending=False, method="first")
250
+ ok_day = ok_day & (rank <= top_n)
251
+ ok_for_day = ok_day.shift(1, fill_value=False) # day D is judged by days <= D - 1
252
+
253
+ def mat(col: str) -> pd.DataFrame:
254
+ return pd.DataFrame({s: bars[s][col] for s in cols}).reindex(grid)
255
+
256
+ close, high, low, opn = mat("close"), mat("high"), mat("low"), mat("open")
257
+ qv = mat("quote_volume")
258
+ elig = ok_for_day.reindex(day_of(grid)).fillna(False).astype(bool).to_numpy()
259
+ eligible = pd.DataFrame(elig, index=grid, columns=cols) & close.notna()
260
+
261
+ # ---- delisting: last real bar of a contract that ended well before the panel end ---------------------------------------
262
+ last = close.apply(lambda c: c.last_valid_index())
263
+ da = pd.DataFrame(False, index=grid, columns=cols)
264
+ for s in cols:
265
+ if last[s] is not None and last[s] < grid[-1] - pd.Timedelta(days=3):
266
+ da.loc[last[s], s] = True
267
+
268
+ # ---- funding at the settlement instant ----------------------------------------------------------------------------
269
+ fund, n_events = None, 0
270
+ if funding:
271
+ F = np.zeros((len(grid), len(cols)))
272
+ for j, s in enumerate(cols):
273
+ ev = funding_events[s] if funding_events is not None and s in funding_events else fetch_funding_events(store, s)
274
+ ev = ev[(ev.index > grid[0] - pd.Timedelta(minutes=m)) & (ev.index <= grid[-1])]
275
+ if ev.empty:
276
+ continue
277
+ pos = grid.searchsorted(ev.index.to_numpy(), side="left") # first bar whose end is at or after the settlement
278
+ ok = pos < len(grid)
279
+ np.add.at(F, (pos[ok], np.full(int(ok.sum()), j)), ev.to_numpy(float)[ok])
280
+ n_events += int(ok.sum())
281
+ fund = pd.DataFrame(F, index=grid, columns=cols)
282
+ panel = Panel(close=close, eligible=eligible, open=opn, high=high, low=low, volume=qv / close, market="CRYPTO-INTRADAY",
283
+ entry_lag=entry_lag, periods_per_year=int(365 * MIN_PER_DAY / m), funding=fund, delist_after=da)
284
+ n = panel.eligible.sum(axis=1)
285
+ panel.meta = {"bar": bar, "bar_minutes": m, "symbols_used": len(cols), "bars": len(grid), "partial_bars": int(sum((bars[s]["n_min"] < m).sum() for s in cols)),
286
+ "zero_volume_bars_kept": zero_interior, "delisted_in_panel": int(da.values.any(axis=0).sum()),
287
+ "eligible_median": float(n[n > 0].median()) if (n > 0).any() else 0.0, "estimated_gb": round(mem, 2),
288
+ "funding_events": n_events}
289
+ return panel
290
+
291
+
292
+ # ------------------------------------------------------------------------------------------------------ analysis
293
+ def backtest_intraday(panel: Panel, factor: pd.DataFrame, *, one_way_bp: float = 0.0, long_q: float = 0.2, short_q: float | None = 0.2,
294
+ hold: int = 1, **kw):
295
+ """`backtest_portfolio` with the cost given as **one-way** basis points (the engine's scalar is a round trip), no grid and no
296
+ benchmark. `hold` is in bars."""
297
+ if panel.entry_lag < 1:
298
+ raise ValueError("an intraday panel needs entry_lag >= 1: a bar's close is only known when the bar ends, so it cannot be traded on")
299
+ kw.setdefault("benchmark", None)
300
+ kw.setdefault("grid", False)
301
+ return backtest_portfolio(panel, factor, long_q=long_q, short_q=short_q, hold=hold, spread_bp=2.0 * one_way_bp, **kw)
302
+
303
+
304
+ def latency_sweep(panel: Panel, factor: pd.DataFrame, lags=(1, 2, 5, 15), *, one_way_bp: float = 0.0, **kw) -> pd.DataFrame:
305
+ """The same portfolio entered `lag` bars after the signal. A real edge decays smoothly with the delay."""
306
+ m = panel.meta.get("bar_minutes", 1)
307
+ rows = []
308
+ for k in lags:
309
+ r = backtest_intraday(replace(panel, entry_lag=int(k)), factor, one_way_bp=one_way_bp, **kw)
310
+ met = r.metrics
311
+ rows.append({"lag_bars": int(k), "delay_minutes": int(k) * m, "sharpe": met["Sharpe"], "cagr": met["CAGR"],
312
+ "gross_cagr": met["gross_CAGR"], "turnover_per_bar": met["turnover_daily"]})
313
+ return pd.DataFrame(rows)
314
+
315
+
316
+ def breakeven_cost(panel: Panel, factor: pd.DataFrame, **kw) -> dict:
317
+ """One-way cost in bp at which the net mean return is zero. Net return is linear in the cost per unit traded, so two runs
318
+ give it exactly: c* = mean(net at 0 bp) / mean(net at 0 bp - net at 1 bp)."""
319
+ kw.pop("one_way_bp", None)
320
+ r0 = backtest_intraday(panel, factor, one_way_bp=0.0, **kw)
321
+ r1 = backtest_intraday(panel, factor, one_way_bp=1.0, **kw)
322
+ n0, n1 = r0.net_returns, r1.net_returns
323
+ per_bp = float((n0 - n1).mean())
324
+ mean0 = float(n0.mean())
325
+ c = mean0 / per_bp if per_bp > 0 else float("nan")
326
+ return {"breakeven_one_way_bp": c, "net_mean_at_0bp": mean0, "cost_per_bp": per_bp, "gross_sharpe": r0.metrics["Sharpe"]}
@@ -0,0 +1,120 @@
1
+ """Build a point-in-time (PIT) Panel from the Binance USDT-M archive.
2
+
3
+ Bias controls, in order of importance
4
+ 1. Survivorship: the universe is every contract that ever traded, delisted ones included. `survivors_only=True`
5
+ reproduces the usual shortcut (only contracts that still trade today) so the bias can be measured, not guessed.
6
+ 2. Universe look-ahead: a contract is eligible on day t only if, using data up to and including day t, it has
7
+ enough history and enough trailing dollar volume. Nothing from t+1 is used.
8
+ 3. Stale prices: bars with zero volume (a frozen last price after a halt or a delisting) are dropped so they
9
+ cannot create fake zero returns or fake liquidity.
10
+ 4. Funding: the daily funding rate is attached to the panel and charged by `backtest_portfolio`.
11
+ 5. Delisting: the last real bar of a contract that stopped trading is marked in `delist_after`, so the engine can
12
+ apply an explicit delisting return instead of silently assuming the position was closed at the last price.
13
+
14
+ Prices are daily UTC closes. Volumes are converted so that `close * volume` equals the USDT turnover.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import re
19
+
20
+ import numpy as np
21
+ import pandas as pd
22
+
23
+ from ..core.panel import Panel
24
+ from .binance_archive import ArchiveStore
25
+
26
+ # Pegged coins, fiat pairs and index perpetuals are not tradable risk assets for a cross-sectional study.
27
+ DEFAULT_EXCLUDE = re.compile(
28
+ r"^(USDC|BUSD|TUSD|FDUSD|USDP|DAI|USTC|EUR|GBP|AUD|TRY|BRL)USDT$|^(BTCDOM|DEFI|FOOTBALL|BLUEBIRD|ALL)USDT$|DOMUSDT$"
29
+ )
30
+
31
+
32
+ def build_panel(store: ArchiveStore | None = None, *, symbols: list[str] | None = None,
33
+ start: str = "2020-06-01", end: str | None = None,
34
+ adv_window: int = 30, min_adv_usd: float = 5e6, min_age_days: int = 60,
35
+ top_n: int | None = None, survivors_only: bool = False,
36
+ exclude: re.Pattern | None = DEFAULT_EXCLUDE, entry_lag: int = 1,
37
+ live: set[str] | None = None) -> Panel:
38
+ """Return a Panel with PIT eligibility, funding and delisting information.
39
+
40
+ min_adv_usd trailing median USDT turnover needed to be eligible (median, so one spike cannot qualify a coin)
41
+ min_age_days bars a contract must already have, so the first weeks after a listing are excluded
42
+ top_n optionally keep only the top-N by trailing turnover among eligible contracts, each day
43
+ survivors_only keep only contracts still trading today (for measuring survivorship bias)
44
+ """
45
+ store = store or ArchiveStore()
46
+ syms = symbols or store.symbols()
47
+ if exclude is not None:
48
+ syms = [s for s in syms if not exclude.search(s)]
49
+ if survivors_only and live is None:
50
+ live = store.trading_now()
51
+ live = live or set()
52
+ bars: dict[str, pd.DataFrame] = {}
53
+ fund: dict[str, pd.Series] = {}
54
+ for s in syms:
55
+ f = store.dir / "daily" / f"{s}.pkl"
56
+ d = pd.read_pickle(f) if f.exists() else store.fetch_daily(s, False)
57
+ if d is None or d.empty:
58
+ continue
59
+ if survivors_only and s not in live:
60
+ continue
61
+ d = d[d["quote_volume"] > 0] # drop frozen, zero-volume bars (halts, delisting tail)
62
+ d = d[d["close"] > 0]
63
+ if d.empty:
64
+ continue
65
+ bars[s] = d
66
+ ff = store.dir / "funding" / f"{s}.pkl"
67
+ fund[s] = pd.read_pickle(ff) if ff.exists() else store.fetch_funding(s, False)
68
+ if not bars:
69
+ raise ValueError("no data: run pitbacktest.crypto.fetch_all() first")
70
+
71
+ idx = pd.date_range(min(d.index[0] for d in bars.values()), max(d.index[-1] for d in bars.values()), freq="D")
72
+ cols = sorted(bars)
73
+
74
+ def mat(col: str) -> pd.DataFrame:
75
+ return pd.DataFrame({s: bars[s][col] for s in cols}).reindex(idx)
76
+
77
+ close, open_, high, low = mat("close"), mat("open"), mat("high"), mat("low")
78
+ qv = mat("quote_volume")
79
+ funding = pd.DataFrame({s: fund[s] for s in cols if fund.get(s) is not None and len(fund[s])}).reindex(idx)
80
+ funding = funding.reindex(columns=cols)
81
+
82
+ # --- point-in-time eligibility: only information up to and including day t ---
83
+ age = close.notna().cumsum()
84
+ adv = qv.rolling(adv_window, min_periods=adv_window).median()
85
+ ok = close.notna() & (age >= min_age_days) & (adv >= min_adv_usd)
86
+ if top_n:
87
+ rank = adv.where(ok).rank(axis=1, ascending=False, method="first")
88
+ ok = ok & (rank <= top_n)
89
+
90
+ # --- delisting: last real bar of a contract that ended well before the panel end ---
91
+ last = close.apply(lambda c: c.last_valid_index())
92
+ delist_after = pd.DataFrame(False, index=idx, columns=cols)
93
+ for s in cols:
94
+ lv = last[s]
95
+ if lv is not None and lv < idx[-1] - pd.Timedelta(days=3):
96
+ delist_after.loc[lv, s] = True
97
+
98
+ # --- crypto-specific controls (replace the equity ROA / book-to-market controls) ---
99
+ ret = close.pct_change(fill_method=None)
100
+ btc = ret["BTCUSDT"] if "BTCUSDT" in ret.columns else ret.median(axis=1)
101
+ beta = ret.rolling(60, min_periods=40).cov(btc).div(btc.rolling(60, min_periods=40).var(), axis=0)
102
+ chars = {"log_dollar_vol": np.log(qv.rolling(adv_window, min_periods=adv_window).mean().clip(lower=1)),
103
+ "btc_beta_60": beta}
104
+
105
+ vol_base = qv / close # so that close * volume == USDT turnover
106
+ keep = idx >= pd.Timestamp(start)
107
+ if end:
108
+ keep &= idx <= pd.Timestamp(end)
109
+ sl = lambda df: df.loc[keep]
110
+ p = Panel(close=sl(close), eligible=sl(ok), open=sl(open_), high=sl(high), low=sl(low), volume=sl(vol_base),
111
+ chars={k: sl(v) for k, v in chars.items()}, market="CRYPTO", entry_lag=entry_lag, periods_per_year=365,
112
+ funding=sl(funding), delist_after=sl(delist_after))
113
+ n = p.eligible.sum(axis=1)
114
+ p.meta = {
115
+ "symbols_in_archive": len(store.symbols()), "symbols_used": len(cols),
116
+ "survivors_only": survivors_only, "delisted_in_panel": int(delist_after.values.any(axis=0).sum()),
117
+ "eligible_median": float(n[n > 0].median()), "min_adv_usd": min_adv_usd, "min_age_days": min_age_days,
118
+ "funding_coverage": float(funding.reindex(index=p.close.index).notna().values[p.close.notna().values].mean()),
119
+ }
120
+ return p
@@ -0,0 +1,10 @@
1
+ """Equity survivorship tools that work without paid data.
2
+
3
+ None of this removes survivorship bias from a survivors-only price source; it measures how much of the real
4
+ universe is missing and shows, by explicit scenarios, how much that could matter. Real removal needs point-in-time data
5
+ (see `pitbacktest.adapters.long_format` for a strict way to plug it in, and `pitbacktest.adapters.krx` for Korea).
6
+ """
7
+ from .master import load_us_master, universe_coverage
8
+ from .scenarios import inject_delistings, survivorship_scenarios, survivors_only
9
+
10
+ __all__ = ["load_us_master", "universe_coverage", "inject_delistings", "survivorship_scenarios", "survivors_only"]
@@ -0,0 +1,60 @@
1
+ """A free, keyless list of US stock tickers with their listing windows, delisted ones included.
2
+
3
+ Source: Tiingo's published `supported_tickers.zip` (no API key needed). It is a **security master, not a price
4
+ source**: it tells you which tickers existed in which period. It is incomplete before about 2013 and misses some recent
5
+ failures (for example SIVB and FRC were absent when this was written), so coverage numbers built on it are lower
6
+ bounds on how many names a survivors-only panel is missing.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import io
11
+ import urllib.request
12
+ import zipfile
13
+ from pathlib import Path
14
+
15
+ import pandas as pd
16
+
17
+ URL = "https://apimedia.tiingo.com/docs/tiingo/daily/supported_tickers.zip"
18
+ US_EXCHANGES = {"NYSE", "NASDAQ", "NYSE MKT", "NYSE ARCA", "AMEX", "NYSE American"}
19
+
20
+
21
+ def load_us_master(cache_dir: str | Path | None = None, refresh: bool = False) -> pd.DataFrame:
22
+ """DataFrame [ticker, exchange, start, end, alive_today] for US-listed common stocks."""
23
+ d = Path(cache_dir or Path.home() / ".cache" / "quantbt").expanduser()
24
+ d.mkdir(parents=True, exist_ok=True)
25
+ f = d / "tiingo_supported_tickers.zip"
26
+ if refresh or not f.exists():
27
+ req = urllib.request.Request(URL, headers={"User-Agent": "pitbacktest-research/0.2 (public data only)"})
28
+ f.write_bytes(urllib.request.urlopen(req, timeout=60).read())
29
+ with zipfile.ZipFile(f) as z:
30
+ df = pd.read_csv(io.BytesIO(z.read(z.namelist()[0])))
31
+ return _clean(df)
32
+
33
+
34
+ def _clean(df: pd.DataFrame) -> pd.DataFrame:
35
+ df = df[(df["assetType"] == "Stock") & df["exchange"].isin(US_EXCHANGES)].copy()
36
+ df["start"] = pd.to_datetime(df["startDate"], errors="coerce")
37
+ df["end"] = pd.to_datetime(df["endDate"], errors="coerce")
38
+ df = df.dropna(subset=["start", "end"])
39
+ df["alive_today"] = df["end"] >= df["end"].max() - pd.Timedelta(days=7)
40
+ return df[["ticker", "exchange", "start", "end", "alive_today"]].reset_index(drop=True)
41
+
42
+
43
+ def universe_coverage(panel_tickers, master: pd.DataFrame, years=range(2010, 2025)) -> pd.DataFrame:
44
+ """For each year: how many US stocks were listed, how many of them are in your panel, and (the telling number)
45
+ how many of the names that stopped trading that year are in your panel.
46
+
47
+ A ticker only counts as present if its listing window in the master overlaps the year, so a reused ticker is not
48
+ credited to the company that used to have it (the master has one row per ticker and window)."""
49
+ have = set(panel_tickers)
50
+ rows = []
51
+ for y in years:
52
+ a, b = pd.Timestamp(f"{y}-01-01"), pd.Timestamp(f"{y}-12-31")
53
+ alive = master[(master["start"] <= b) & (master["end"] >= a)]
54
+ died = alive[(alive["end"] <= b) & ~alive["alive_today"]]
55
+ rows.append({"year": y, "listed": len(alive), "in_panel": int(alive["ticker"].isin(have).sum()),
56
+ "stopped_trading": len(died), "stopped_in_panel": int(died["ticker"].isin(have).sum())})
57
+ out = pd.DataFrame(rows).set_index("year")
58
+ out["share_listed_in_panel"] = out["in_panel"] / out["listed"].replace(0, float("nan"))
59
+ out["share_stopped_in_panel"] = out["stopped_in_panel"] / out["stopped_trading"].replace(0, float("nan"))
60
+ return out