pitbacktest 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pitbacktest/__init__.py +27 -0
- pitbacktest/adapters/__init__.py +0 -0
- pitbacktest/adapters/krx.py +249 -0
- pitbacktest/adapters/long_format.py +87 -0
- pitbacktest/adapters/tiingo.py +236 -0
- pitbacktest/adapters/yfinance.py +243 -0
- pitbacktest/analytics.py +234 -0
- pitbacktest/core/__init__.py +0 -0
- pitbacktest/core/controls.py +126 -0
- pitbacktest/core/costs.py +144 -0
- pitbacktest/core/estimators.py +216 -0
- pitbacktest/core/gates.py +175 -0
- pitbacktest/core/panel.py +306 -0
- pitbacktest/crypto/__init__.py +12 -0
- pitbacktest/crypto/binance_archive.py +287 -0
- pitbacktest/crypto/costs.py +80 -0
- pitbacktest/crypto/intraday.py +326 -0
- pitbacktest/crypto/panel.py +120 -0
- pitbacktest/equity/__init__.py +10 -0
- pitbacktest/equity/master.py +60 -0
- pitbacktest/equity/scenarios.py +104 -0
- pitbacktest/event.py +206 -0
- pitbacktest/execution.py +310 -0
- pitbacktest/ledger.py +128 -0
- pitbacktest/portfolio.py +373 -0
- pitbacktest/screen.py +150 -0
- pitbacktest/shorting.py +46 -0
- pitbacktest/validation.py +118 -0
- pitbacktest/weights.py +271 -0
- pitbacktest-0.2.0.dist-info/METADATA +395 -0
- pitbacktest-0.2.0.dist-info/RECORD +34 -0
- pitbacktest-0.2.0.dist-info/WHEEL +5 -0
- pitbacktest-0.2.0.dist-info/licenses/LICENSE +21 -0
- pitbacktest-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
"""Panel convention: the input contract of every test.
|
|
2
|
+
|
|
3
|
+
Design principles
|
|
4
|
+
- The core is **plain pandas and numpy** and does not depend on any external data source (adapters do that).
|
|
5
|
+
- Every matrix is (date x ticker) and the index and columns must be identical.
|
|
6
|
+
- `eligible` is the **point-in-time universe**: True only for securities that could actually be traded that day.
|
|
7
|
+
It is the only device against survivorship bias, so it is a required input.
|
|
8
|
+
|
|
9
|
+
Timing convention (the only one used in the whole project)
|
|
10
|
+
A signal is **fixed** at close(t). Entry is at close(t+entry_lag) and exit at close(t+entry_lag+h).
|
|
11
|
+
entry_lag defaults to 1: "see the signal, buy the next day". 0 assumes a fill at the same close (aggressive).
|
|
12
|
+
Breaking this convention is caught by Panel.assert_no_lookahead().
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass, field, replace
|
|
18
|
+
from typing import Iterable
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
|
|
23
|
+
REQUIRED = ("close", "eligible")
|
|
24
|
+
OPTIONAL = ("open", "high", "low", "volume", "mkt_cap", "funding", "delist_after")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class Panel:
|
|
29
|
+
"""A bundle of (date x ticker) matrices.
|
|
30
|
+
|
|
31
|
+
close adjusted closing price: the only source of returns
|
|
32
|
+
eligible point-in-time universe bool: listed, tradable and passing the liquidity condition that day
|
|
33
|
+
The rest is optional. With volume and mkt_cap the liquidity and size controls are switched on.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
close: pd.DataFrame
|
|
37
|
+
eligible: pd.DataFrame
|
|
38
|
+
open: pd.DataFrame | None = None
|
|
39
|
+
high: pd.DataFrame | None = None
|
|
40
|
+
low: pd.DataFrame | None = None
|
|
41
|
+
volume: pd.DataFrame | None = None
|
|
42
|
+
mkt_cap: pd.DataFrame | None = None
|
|
43
|
+
chars: dict[str, pd.DataFrame] = field(default_factory=dict)
|
|
44
|
+
market: str = "KR"
|
|
45
|
+
entry_lag: int = 1
|
|
46
|
+
# Trading days per year used for annualising: 252 for stocks, 365 for 24/7 markets (crypto). A wrong value makes CAGR, Sharpe and volatility wrong.
|
|
47
|
+
periods_per_year: int = 252
|
|
48
|
+
# Optional: futures funding (date x ticker, daily sum, positive means longs pay). Delisting flag (True on the last real bar). Notes for adapters.
|
|
49
|
+
funding: pd.DataFrame | None = None
|
|
50
|
+
delist_after: pd.DataFrame | None = None
|
|
51
|
+
# Optional: (date x ticker) bool, True where a security can be sold short on the execution day (borrowable, and no short-selling ban). None means every
|
|
52
|
+
# security can. A date or security missing from the frame counts as **not** shortable: an unknown is not permission.
|
|
53
|
+
shortable: pd.DataFrame | None = None
|
|
54
|
+
# Optional: (date x ticker) bool, False where a trade that increases (can_buy) or decreases (can_sell) a position cannot be done on that day
|
|
55
|
+
# (halted, or locked at a daily price limit). None means always possible; a date or security missing from the frame counts as not possible.
|
|
56
|
+
# `execution.tradability` builds them from prices and volume.
|
|
57
|
+
can_buy: pd.DataFrame | None = None
|
|
58
|
+
can_sell: pd.DataFrame | None = None
|
|
59
|
+
meta: dict = field(default_factory=dict)
|
|
60
|
+
|
|
61
|
+
# ---------------------------------------------------------------- construction and validation
|
|
62
|
+
def __post_init__(self) -> None:
|
|
63
|
+
self.close = self.close.sort_index()
|
|
64
|
+
if self.close.index.has_duplicates: # before any reindex, which would fail with a less helpful pandas message
|
|
65
|
+
raise ValueError("close.index has duplicate dates")
|
|
66
|
+
self.eligible = self.eligible.reindex(
|
|
67
|
+
index=self.close.index, columns=self.close.columns
|
|
68
|
+
).fillna(False).astype(bool)
|
|
69
|
+
for k in OPTIONAL:
|
|
70
|
+
v = getattr(self, k)
|
|
71
|
+
if v is not None:
|
|
72
|
+
setattr(self, k, v.reindex(index=self.close.index, columns=self.close.columns))
|
|
73
|
+
self.chars = {
|
|
74
|
+
k: v.reindex(index=self.close.index, columns=self.close.columns)
|
|
75
|
+
for k, v in self.chars.items()
|
|
76
|
+
}
|
|
77
|
+
if self.delist_after is not None:
|
|
78
|
+
self.delist_after = self.delist_after.fillna(False).astype(bool)
|
|
79
|
+
for k in ("shortable", "can_buy", "can_sell"):
|
|
80
|
+
v = getattr(self, k)
|
|
81
|
+
if v is not None:
|
|
82
|
+
setattr(self, k, v.reindex(index=self.close.index, columns=self.close.columns).fillna(False).astype(bool))
|
|
83
|
+
self.validate()
|
|
84
|
+
|
|
85
|
+
def validate(self) -> None:
|
|
86
|
+
"""Raise ValueError if the panel breaks its invariants: the date index ascending and without duplicates, at least one eligible
|
|
87
|
+
security, `entry_lag` not negative, and at least 10 eligible securities on half of the dates or more. Runs on construction."""
|
|
88
|
+
if not self.close.index.is_monotonic_increasing:
|
|
89
|
+
raise ValueError("close.index is not ascending")
|
|
90
|
+
if self.close.index.has_duplicates:
|
|
91
|
+
raise ValueError("close.index has duplicate dates")
|
|
92
|
+
if self.eligible.values.sum() == 0:
|
|
93
|
+
raise ValueError("eligible is all False")
|
|
94
|
+
if self.entry_lag < 0:
|
|
95
|
+
raise ValueError("entry_lag must be 0 or more")
|
|
96
|
+
n = self.eligible.sum(axis=1)
|
|
97
|
+
if (n[n > 0] < 10).mean() > 0.5:
|
|
98
|
+
raise ValueError(
|
|
99
|
+
"eligible has fewer than 10 securities on more than half of the dates: cross-sectional tests are impossible"
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
# ---------------------------------------------------------------- derived
|
|
103
|
+
@property
|
|
104
|
+
def dates(self) -> pd.DatetimeIndex:
|
|
105
|
+
return self.close.index
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def tickers(self) -> pd.Index:
|
|
109
|
+
return self.close.columns
|
|
110
|
+
|
|
111
|
+
def ret1(self) -> pd.DataFrame:
|
|
112
|
+
"""Daily return close(t)/close(t-1) - 1."""
|
|
113
|
+
return self.close.pct_change(fill_method=None)
|
|
114
|
+
|
|
115
|
+
def forward(self, h: int, delist_return: float | None = None) -> pd.DataFrame:
|
|
116
|
+
"""Return from close(t + lag) to close(t + lag + h) for a signal at t: a plain price ratio, NaN when either price is missing.
|
|
117
|
+
|
|
118
|
+
A security flagged in `delist_after` is carried at its last price after its last bar (as cash), or at `last price x (1 + delist_return)`
|
|
119
|
+
when `delist_return` is given, so a holding period that runs into a delisting keeps its result instead of dropping out. Without that
|
|
120
|
+
the statistics of `screen` and `backtest_event` would leave out exactly the trades that ended in a delisting. A security that is only
|
|
121
|
+
missing for some days (a halt) still gives NaN when an end price is missing. Compare `analytics.forward_returns`, which compounds daily
|
|
122
|
+
returns and treats a missing price as a return of 0."""
|
|
123
|
+
lag = self.entry_lag
|
|
124
|
+
px = self.close
|
|
125
|
+
if self.delist_after is not None and bool(np.asarray(self.delist_after.values).any()):
|
|
126
|
+
da = self.delist_after.reindex(index=px.index, columns=px.columns).fillna(False).astype(bool).astype(int)
|
|
127
|
+
gone = (da.cumsum() - da) > 0 # strictly after the security's last bar
|
|
128
|
+
carried = px.ffill() * (1.0 + (0.0 if delist_return is None else float(delist_return)))
|
|
129
|
+
px = px.where(~gone, carried)
|
|
130
|
+
entry = px.shift(-lag)
|
|
131
|
+
exit_ = px.shift(-(lag + h))
|
|
132
|
+
return (exit_ / entry - 1.0).astype(np.float32)
|
|
133
|
+
|
|
134
|
+
def fingerprint(self) -> str:
|
|
135
|
+
"""Short hash of the numbers this panel holds (every matrix that is present, including volume and market cap, which
|
|
136
|
+
the impact model and the size controls read), used by the trial ledger to tell whether two runs saw the same data.
|
|
137
|
+
The date index and the ticker names are not hashed: the same values in the same shape give the same fingerprint."""
|
|
138
|
+
import hashlib
|
|
139
|
+
h = hashlib.sha256()
|
|
140
|
+
mats = [(n, getattr(self, n)) for n in ("close", "eligible", "open", "high", "low", "volume", "mkt_cap", "funding", "delist_after", "shortable", "can_buy", "can_sell")]
|
|
141
|
+
mats += [(f"chars.{k}", self.chars[k]) for k in sorted(self.chars)]
|
|
142
|
+
for name, v in mats:
|
|
143
|
+
if v is not None:
|
|
144
|
+
a = np.ascontiguousarray(v.to_numpy(dtype=np.float32))
|
|
145
|
+
h.update(name.encode() + repr(a.shape).encode() + a.tobytes())
|
|
146
|
+
h.update(f"{self.entry_lag}|{self.periods_per_year}|{self.market}".encode())
|
|
147
|
+
return h.hexdigest()[:16]
|
|
148
|
+
|
|
149
|
+
def adv(self, window: int = 20) -> pd.DataFrame:
|
|
150
|
+
"""Average traded value. None if there is no volume."""
|
|
151
|
+
if self.volume is None:
|
|
152
|
+
return None
|
|
153
|
+
return (self.close * self.volume).rolling(window, min_periods=window).mean()
|
|
154
|
+
|
|
155
|
+
# ---------------------------------------------------------------- safeguards
|
|
156
|
+
def truncate(self, last: pd.Timestamp) -> "Panel":
|
|
157
|
+
"""A copy that holds only the rows up to and including `last`: what a user standing on that date would have had."""
|
|
158
|
+
cut = lambda f: None if f is None else f.loc[:last]
|
|
159
|
+
return replace(self, close=self.close.loc[:last], eligible=self.eligible.loc[:last], open=cut(self.open), high=cut(self.high), low=cut(self.low),
|
|
160
|
+
volume=cut(self.volume), mkt_cap=cut(self.mkt_cap), funding=cut(self.funding), delist_after=cut(self.delist_after),
|
|
161
|
+
shortable=cut(self.shortable), can_buy=cut(self.can_buy), can_sell=cut(self.can_sell),
|
|
162
|
+
chars={k: v.loc[:last] for k, v in self.chars.items()})
|
|
163
|
+
|
|
164
|
+
def assert_causal(self, make_signal, n_cuts: int = 6, seed: int = 0, min_history: int = 300, rtol: float = 1e-9) -> dict:
|
|
165
|
+
"""The exact look-ahead check: a signal that is a function of the panel must not change when the future is removed.
|
|
166
|
+
|
|
167
|
+
`make_signal(panel)` returns the (date x ticker) signal. For `n_cuts` random dates t (at least `min_history` rows in, so that rolling windows are full) the signal is
|
|
168
|
+
rebuilt from the panel truncated at t, and its last row must equal the row of the full-data signal at t, name by name (NaN equal to NaN). Any use of a later row
|
|
169
|
+
(a negative shift, a centred window, a mean or standard deviation over the whole history, a rank across time) makes the rows differ.
|
|
170
|
+
Returns {"pass", "cuts", "mismatched_cuts", "max_abs_diff"}. It checks that the function is causal; it does not say the signal is good, and it cannot see look-ahead in
|
|
171
|
+
the *data* itself (a price already revised), only in how the signal is computed. It needs the function, not a frame: that is what makes it exact."""
|
|
172
|
+
full = make_signal(self)
|
|
173
|
+
if not isinstance(full, pd.DataFrame):
|
|
174
|
+
raise ValueError("make_signal must return a (date x ticker) DataFrame")
|
|
175
|
+
n = len(self.dates)
|
|
176
|
+
if n <= min_history + 1:
|
|
177
|
+
raise ValueError(f"the panel has {n} rows; at least {min_history + 2} are needed to cut it")
|
|
178
|
+
rng = np.random.default_rng(seed)
|
|
179
|
+
cuts = sorted(int(i) for i in rng.choice(np.arange(min_history, n - 1), size=min(n_cuts, n - 1 - min_history), replace=False))
|
|
180
|
+
bad, worst = [], 0.0
|
|
181
|
+
for i in cuts:
|
|
182
|
+
t = self.dates[i]
|
|
183
|
+
part = make_signal(self.truncate(t))
|
|
184
|
+
a = part.iloc[-1].reindex(self.tickers).to_numpy(float)
|
|
185
|
+
b = full.loc[t].reindex(self.tickers).to_numpy(float)
|
|
186
|
+
same_nan = np.isnan(a) == np.isnan(b)
|
|
187
|
+
both = ~np.isnan(a) & ~np.isnan(b)
|
|
188
|
+
diff = float(np.max(np.abs(a[both] - b[both]) / (np.abs(b[both]) + 1e-12))) if both.any() else 0.0
|
|
189
|
+
worst = max(worst, diff)
|
|
190
|
+
if (not same_nan.all()) or diff > rtol:
|
|
191
|
+
bad.append(str(t.date()))
|
|
192
|
+
return {"pass": not bad, "cuts": [str(self.dates[i].date()) for i in cuts], "mismatched_cuts": bad, "max_abs_diff": worst}
|
|
193
|
+
|
|
194
|
+
def assert_no_lookahead(self, signal: pd.DataFrame, h: int = 5,
|
|
195
|
+
n_null: int = 8, seed: int = 0) -> dict:
|
|
196
|
+
"""A weak heuristic for look-ahead; prefer `assert_causal`, which is exact. If the result improves when the signal is pushed one day **later**, it is looking at the future.
|
|
197
|
+
|
|
198
|
+
**Known blind spots, measured on real data (see docs/market_validation.md):** a signal that is the future return itself is **not** flagged (delaying it by a day
|
|
199
|
+
makes it worse, not better), and a legitimate signal whose own edge is negative **is** flagged (delaying shrinks the loss, which reads as an improvement).
|
|
200
|
+
Treat a pass as "no evidence", never as proof, and a fail on a signal with a negative edge as possibly false.
|
|
201
|
+
|
|
202
|
+
A normal signal gets worse when delayed (the information decays).
|
|
203
|
+
If it improves after the delay, the signal already contains future information.
|
|
204
|
+
|
|
205
|
+
Warning: judging by a bare `lagged > base` makes a **random factor wrong 50% of the time**:
|
|
206
|
+
both are near zero so the sign flips by chance. So the noise scale (sd) of the spread is measured with random shuffles
|
|
207
|
+
and twice that sd is used as the tolerance.
|
|
208
|
+
"""
|
|
209
|
+
fwd = self.forward(h)
|
|
210
|
+
base = _spread(signal, fwd, self.eligible)
|
|
211
|
+
lagged = _spread(signal.shift(1), fwd, self.eligible)
|
|
212
|
+
rng = np.random.default_rng(seed)
|
|
213
|
+
nulls = []
|
|
214
|
+
for _ in range(n_null):
|
|
215
|
+
sh = signal.values.copy()
|
|
216
|
+
ev = self.eligible.values
|
|
217
|
+
for i in range(sh.shape[0]):
|
|
218
|
+
j = np.where(ev[i])[0]
|
|
219
|
+
if len(j) > 1:
|
|
220
|
+
sh[i, j] = sh[i, rng.permutation(j)]
|
|
221
|
+
v = _spread(pd.DataFrame(sh, index=signal.index, columns=signal.columns),
|
|
222
|
+
fwd, self.eligible)
|
|
223
|
+
if np.isfinite(v):
|
|
224
|
+
nulls.append(v)
|
|
225
|
+
tol = 2 * float(np.std(nulls)) if len(nulls) >= 3 else 0.0
|
|
226
|
+
ok = not (np.isfinite(base) and np.isfinite(lagged) and lagged > base + tol)
|
|
227
|
+
return {"base_bp": base * 1e4, "lagged_bp": lagged * 1e4,
|
|
228
|
+
"tol_bp": tol * 1e4, "pass": bool(ok)}
|
|
229
|
+
|
|
230
|
+
def audit(self) -> dict:
|
|
231
|
+
"""Data integrity audit: run it once before any test.
|
|
232
|
+
|
|
233
|
+
It catches common data defects:
|
|
234
|
+
- silent truncation (data vanishes wholesale after some date)
|
|
235
|
+
- adjusted-price mismatch (more extreme gaps than daily moves: the adj_open defect type)
|
|
236
|
+
- universe jumps (the population triples over time, for example)
|
|
237
|
+
"""
|
|
238
|
+
out = {}
|
|
239
|
+
n = self.eligible.sum(axis=1)
|
|
240
|
+
out["dates"] = len(self.dates)
|
|
241
|
+
out["tickers"] = int(self.eligible.any(axis=0).sum())
|
|
242
|
+
out["eligible_mean"] = float(n.mean())
|
|
243
|
+
yr = n.groupby(self.dates.year).mean()
|
|
244
|
+
out["eligible_by_year"] = {int(k): float(v) for k, v in yr.items()}
|
|
245
|
+
out["universe_growth"] = float(yr.iloc[-1] / yr.iloc[0]) if len(yr) > 1 and yr.iloc[0] else np.nan
|
|
246
|
+
|
|
247
|
+
r = self.ret1().where(self.eligible)
|
|
248
|
+
out["ret_extreme_pct"] = float((r.abs() > 0.30).mean().mean() * 100)
|
|
249
|
+
gap = None
|
|
250
|
+
if self.open is not None:
|
|
251
|
+
gap = (self.open / self.close.shift(1) - 1.0).where(self.eligible)
|
|
252
|
+
out["gap_extreme_pct"] = float((gap.abs() > 0.30).mean().mean() * 100)
|
|
253
|
+
ratio = out["gap_extreme_pct"] / max(out["ret_extreme_pct"], 1e-9)
|
|
254
|
+
out["gap_vs_daily_ratio"] = float(ratio)
|
|
255
|
+
out["adj_price_consistent"] = bool(ratio < 3.0)
|
|
256
|
+
|
|
257
|
+
# silent truncation: security coverage drops sharply after some date
|
|
258
|
+
cov = self.close.notna().sum(axis=1)
|
|
259
|
+
if len(cov) > 60:
|
|
260
|
+
tail = cov.iloc[-20:].mean(); body = cov.iloc[:-20].median()
|
|
261
|
+
out["tail_coverage_ratio"] = float(tail / body) if body else np.nan
|
|
262
|
+
out["truncation_suspected"] = bool(tail < body * 0.5)
|
|
263
|
+
|
|
264
|
+
# suspected survivorship bias: in a real market a steady share of securities disappears every year. If almost no security's price stops before the panel's end,
|
|
265
|
+
# the panel holds only 'securities alive today'. Measured: of 41 well-known delisted or acquired stocks looked up in yfinance,
|
|
266
|
+
# none came back with a correct history (docs/survivorship.md).
|
|
267
|
+
last = self.close.apply(lambda c: c.last_valid_index())
|
|
268
|
+
ended = last.dropna() < (self.dates[-1] - pd.Timedelta(days=30))
|
|
269
|
+
span_years = (self.dates[-1] - self.dates[0]).days / 365.25
|
|
270
|
+
out["ended_before_end_share"] = float(ended.mean()) if len(ended) else np.nan
|
|
271
|
+
out["survivorship_suspected"] = bool(len(ended) >= 30 and span_years >= 3 and ended.mean() < 0.01)
|
|
272
|
+
return out
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _spread(signal: pd.DataFrame, fwd: pd.DataFrame, eligible: pd.DataFrame,
|
|
276
|
+
q: float = 0.10) -> float:
|
|
277
|
+
"""Mean top-q minus bottom-q spread (internal helper for the look-ahead check)."""
|
|
278
|
+
rk = signal.where(eligible).rank(axis=1, pct=True, na_option="keep")
|
|
279
|
+
hi, lo = ((rk >= 1 - q) & eligible).values, ((rk <= q) & eligible).values
|
|
280
|
+
c = fwd.values.astype(np.float64)
|
|
281
|
+
out = []
|
|
282
|
+
for i in range(c.shape[0]):
|
|
283
|
+
a, b = hi[i] & np.isfinite(c[i]), lo[i] & np.isfinite(c[i])
|
|
284
|
+
if a.sum() >= 3 and b.sum() >= 3:
|
|
285
|
+
out.append(c[i][a].mean() - c[i][b].mean())
|
|
286
|
+
return float(np.mean(out)) if out else np.nan
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def build_pit_eligible(close: pd.DataFrame, *, listed: pd.DataFrame | None = None,
|
|
290
|
+
min_adv: float = 0.0, volume: pd.DataFrame | None = None,
|
|
291
|
+
exclude: pd.DataFrame | None = None) -> pd.DataFrame:
|
|
292
|
+
"""Helper that builds a point-in-time universe mask.
|
|
293
|
+
|
|
294
|
+
listed listing flag bool (falls back to whether close exists)
|
|
295
|
+
min_adv minimum average traded value (needs volume)
|
|
296
|
+
exclude mask of securities to exclude, such as administrative issues and trading halts (True = exclude)
|
|
297
|
+
"""
|
|
298
|
+
ok = close.notna() & (close > 0)
|
|
299
|
+
if listed is not None:
|
|
300
|
+
ok &= listed.reindex_like(close).fillna(False).astype(bool)
|
|
301
|
+
if min_adv > 0 and volume is not None:
|
|
302
|
+
adv = (close * volume).rolling(20, min_periods=20).mean()
|
|
303
|
+
ok &= adv >= min_adv
|
|
304
|
+
if exclude is not None:
|
|
305
|
+
ok &= ~exclude.reindex_like(close).fillna(False).astype(bool)
|
|
306
|
+
return ok.fillna(False)
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Crypto perpetual-futures data and panel construction (Binance USDT-M).
|
|
2
|
+
|
|
3
|
+
The point of this subpackage is to remove, as far as public data allows, three biases that crypto backtests
|
|
4
|
+
usually carry: survivorship (delisted contracts are included), universe look-ahead (eligibility uses only data up
|
|
5
|
+
to the signal date) and ignored funding (the daily funding rate is charged to the position).
|
|
6
|
+
"""
|
|
7
|
+
from .binance_archive import ArchiveStore, fetch_all
|
|
8
|
+
from .panel import build_panel
|
|
9
|
+
from .costs import liquidity_cost_bp, participation_report
|
|
10
|
+
from . import intraday
|
|
11
|
+
|
|
12
|
+
__all__ = ["ArchiveStore", "fetch_all", "build_panel", "liquidity_cost_bp", "participation_report", "intraday"]
|
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
"""Binance public data archive (data.binance.vision) for USDT-margined perpetuals.
|
|
2
|
+
|
|
3
|
+
Why the archive and not the live API: the archive keeps **delisted** contracts (LUNA, FTT and hundreds of
|
|
4
|
+
others), so a universe built from it is not limited to today's survivors. The live API is used only to fill the
|
|
5
|
+
last partial month for contracts that still trade.
|
|
6
|
+
|
|
7
|
+
Daily bars are UTC days. Funding is settled every 8 hours (older contracts) or per `funding_interval_hours`;
|
|
8
|
+
`daily_funding` sums every settlement that falls inside the UTC day.
|
|
9
|
+
|
|
10
|
+
Everything is cached under `cache_dir` (default ~/.cache/quantbt/binance_um) so a download happens once.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import concurrent.futures as cf
|
|
15
|
+
import io
|
|
16
|
+
import json
|
|
17
|
+
import re
|
|
18
|
+
import time
|
|
19
|
+
import urllib.error
|
|
20
|
+
import urllib.parse
|
|
21
|
+
import urllib.request
|
|
22
|
+
import zipfile
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
import pandas as pd
|
|
26
|
+
|
|
27
|
+
LIST_URL = "https://s3-ap-northeast-1.amazonaws.com/data.binance.vision"
|
|
28
|
+
DL_URL = "https://data.binance.vision"
|
|
29
|
+
API = "https://fapi.binance.com"
|
|
30
|
+
KL_COLS = ["open_time", "open", "high", "low", "close", "volume", "close_time", "quote_volume", "trades",
|
|
31
|
+
"taker_buy_base", "taker_buy_quote", "ignore"]
|
|
32
|
+
UA = {"User-Agent": "pitbacktest-research/0.2 (public data only)"}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _get(url: str, retries: int = 4, timeout: float = 30) -> bytes | None:
|
|
36
|
+
"""GET with backoff. Returns None on 404 (a missing archive file is normal)."""
|
|
37
|
+
url = urllib.parse.quote(url, safe=":/?&=%") # non-ASCII symbols (e.g. Chinese tickers) need encoding
|
|
38
|
+
delay = 1.0
|
|
39
|
+
for i in range(retries + 1):
|
|
40
|
+
try:
|
|
41
|
+
with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=timeout) as r:
|
|
42
|
+
return r.read()
|
|
43
|
+
except urllib.error.HTTPError as e:
|
|
44
|
+
if e.code == 404:
|
|
45
|
+
return None
|
|
46
|
+
if e.code in (418, 429, 500, 502, 503, 504) and i < retries:
|
|
47
|
+
time.sleep(delay * (3 if e.code in (418, 429) else 1))
|
|
48
|
+
delay *= 2
|
|
49
|
+
continue
|
|
50
|
+
raise
|
|
51
|
+
except (urllib.error.URLError, TimeoutError, ConnectionError):
|
|
52
|
+
if i < retries:
|
|
53
|
+
time.sleep(delay)
|
|
54
|
+
delay *= 2
|
|
55
|
+
continue
|
|
56
|
+
raise
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _list(prefix: str) -> list[str]:
|
|
61
|
+
"""All keys under a prefix (S3 listing, paginated)."""
|
|
62
|
+
keys, marker = [], ""
|
|
63
|
+
while True:
|
|
64
|
+
q = {"prefix": prefix, "max-keys": "1000"}
|
|
65
|
+
if marker:
|
|
66
|
+
q["marker"] = marker
|
|
67
|
+
x = (_get(f"{LIST_URL}?{urllib.parse.urlencode(q)}") or b"").decode("utf-8")
|
|
68
|
+
got = re.findall(r"<Key>([^<]+)</Key>", x)
|
|
69
|
+
keys += got
|
|
70
|
+
if "<IsTruncated>true</IsTruncated>" in x and got:
|
|
71
|
+
marker = got[-1]
|
|
72
|
+
else:
|
|
73
|
+
return keys
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _list_dirs(prefix: str) -> list[str]:
|
|
77
|
+
out, marker = [], ""
|
|
78
|
+
while True:
|
|
79
|
+
q = {"prefix": prefix, "delimiter": "/", "max-keys": "1000"}
|
|
80
|
+
if marker:
|
|
81
|
+
q["marker"] = marker
|
|
82
|
+
x = (_get(f"{LIST_URL}?{urllib.parse.urlencode(q)}") or b"").decode("utf-8")
|
|
83
|
+
got = re.findall(r"<Prefix>([^<]+)</Prefix>", x)
|
|
84
|
+
got = [g for g in got if g != prefix]
|
|
85
|
+
out += got
|
|
86
|
+
nm = re.search(r"<NextMarker>([^<]+)</NextMarker>", x)
|
|
87
|
+
if "<IsTruncated>true</IsTruncated>" in x and nm:
|
|
88
|
+
marker = nm.group(1)
|
|
89
|
+
else:
|
|
90
|
+
return out
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _read_klines_zip(blob: bytes) -> pd.DataFrame:
|
|
94
|
+
with zipfile.ZipFile(io.BytesIO(blob)) as z:
|
|
95
|
+
raw = pd.read_csv(z.open(z.namelist()[0]), header=None, names=KL_COLS, dtype=str)
|
|
96
|
+
t = pd.to_numeric(raw["open_time"], errors="coerce") # some files carry a header row
|
|
97
|
+
raw = raw[t.notna()].copy()
|
|
98
|
+
raw["open_time"] = t[t.notna()].astype("int64")
|
|
99
|
+
return raw
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _frame_from_klines(raw: pd.DataFrame) -> pd.DataFrame:
|
|
103
|
+
if raw.empty:
|
|
104
|
+
return pd.DataFrame()
|
|
105
|
+
ts = raw["open_time"].astype("int64")
|
|
106
|
+
unit = "us" if ts.iloc[0] > 10**14 else "ms" # newer archives use microseconds
|
|
107
|
+
idx = pd.to_datetime(ts, unit=unit, utc=True).dt.tz_localize(None).dt.normalize()
|
|
108
|
+
df = raw[["open", "high", "low", "close", "volume", "quote_volume", "trades"]].apply(pd.to_numeric, errors="coerce")
|
|
109
|
+
df.index = idx
|
|
110
|
+
df = df[~df.index.duplicated(keep="last")].sort_index()
|
|
111
|
+
return df
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class ArchiveStore:
|
|
115
|
+
"""Local cache plus downloader for one market (USDT-M perpetuals)."""
|
|
116
|
+
|
|
117
|
+
def __init__(self, cache_dir: str | Path | None = None):
|
|
118
|
+
self.dir = Path(cache_dir or Path.home() / ".cache" / "quantbt" / "binance_um").expanduser()
|
|
119
|
+
(self.dir / "daily").mkdir(parents=True, exist_ok=True)
|
|
120
|
+
(self.dir / "funding").mkdir(parents=True, exist_ok=True)
|
|
121
|
+
|
|
122
|
+
# ------------------------------------------------------------ symbols
|
|
123
|
+
def symbols(self, refresh: bool = False, quote: str = "USDT") -> list[str]:
|
|
124
|
+
"""Every USDT perpetual that ever existed in the archive (delisted ones included).
|
|
125
|
+
Delivery contracts such as BTCUSDT_250627 are excluded."""
|
|
126
|
+
f = self.dir / "symbols.json"
|
|
127
|
+
if f.exists() and not refresh:
|
|
128
|
+
return json.loads(f.read_text())
|
|
129
|
+
pre = "data/futures/um/monthly/klines/"
|
|
130
|
+
names = [d[len(pre):].strip("/") for d in _list_dirs(pre)]
|
|
131
|
+
syms = sorted(n for n in names if n.endswith(quote) and "_" not in n)
|
|
132
|
+
f.write_text(json.dumps(syms))
|
|
133
|
+
return syms
|
|
134
|
+
|
|
135
|
+
def trading_now(self) -> set[str]:
|
|
136
|
+
"""Symbols the live exchange reports as TRADING (used to know which ones need the live top-up)."""
|
|
137
|
+
raw = _get(f"{API}/fapi/v1/exchangeInfo")
|
|
138
|
+
info = json.loads(raw)
|
|
139
|
+
return {s["symbol"] for s in info["symbols"] if s.get("contractType") == "PERPETUAL" and s.get("status") == "TRADING"}
|
|
140
|
+
|
|
141
|
+
def symbol_rules(self, refresh: bool = False) -> pd.DataFrame:
|
|
142
|
+
"""Trading rules per contract from the live exchange-info endpoint (no key): quantity step, minimum quantity, minimum order value,
|
|
143
|
+
price tick, status and the onboard and delivery dates, indexed by symbol. **Today's values only**: the endpoint has no history, so a rule
|
|
144
|
+
that changed (BTCUSDT's minimum order value is not what it was in 2020) is applied to the past as it is now, and a contract that is no longer
|
|
145
|
+
listed (delisted ones) is absent. The frame carries the day it was fetched in `.attrs["fetched"]`. Cached as `symbol_rules.json`."""
|
|
146
|
+
f = self.dir / "symbol_rules.json"
|
|
147
|
+
if f.exists() and not refresh:
|
|
148
|
+
blob = json.loads(f.read_text())
|
|
149
|
+
else:
|
|
150
|
+
raw = _get(f"{API}/fapi/v1/exchangeInfo")
|
|
151
|
+
if raw is None:
|
|
152
|
+
raise RuntimeError("exchange info is not available")
|
|
153
|
+
rows = {}
|
|
154
|
+
for s in json.loads(raw)["symbols"]:
|
|
155
|
+
flt = {x["filterType"]: x for x in s.get("filters", [])}
|
|
156
|
+
lot, mn, pf = flt.get("LOT_SIZE"), flt.get("MIN_NOTIONAL"), flt.get("PRICE_FILTER")
|
|
157
|
+
if lot is None or mn is None:
|
|
158
|
+
continue
|
|
159
|
+
rows[s["symbol"]] = {"step_size": float(lot["stepSize"]), "min_qty": float(lot["minQty"]), "min_notional": float(mn["notional"]),
|
|
160
|
+
"tick_size": float(pf["tickSize"]) if pf else float("nan"), "status": s.get("status"),
|
|
161
|
+
"contract_type": s.get("contractType"), "onboard_ms": s.get("onboardDate"), "delivery_ms": s.get("deliveryDate")}
|
|
162
|
+
blob = {"fetched": pd.Timestamp.now("UTC").strftime("%Y-%m-%d"), "rules": rows}
|
|
163
|
+
f.write_text(json.dumps(blob))
|
|
164
|
+
df = pd.DataFrame.from_dict(blob["rules"], orient="index")
|
|
165
|
+
df.attrs["fetched"] = blob["fetched"]
|
|
166
|
+
return df
|
|
167
|
+
|
|
168
|
+
# ------------------------------------------------------------ one symbol
|
|
169
|
+
def _months(self, sym: str, kind: str) -> list[str]:
|
|
170
|
+
keys = _list(f"data/futures/um/monthly/{kind}/{sym}/" + ("1d/" if kind == "klines" else ""))
|
|
171
|
+
return sorted(k for k in keys if k.endswith(".zip"))
|
|
172
|
+
|
|
173
|
+
def fetch_daily(self, sym: str, live: bool) -> pd.DataFrame | None:
|
|
174
|
+
"""Daily bars (open, high, low, close, volume, quote_volume, trades) indexed by UTC date: the monthly archive files plus, when `live`
|
|
175
|
+
is True, the live API for the days after the last archived one. The unfinished current day is dropped. Cached as a pickle."""
|
|
176
|
+
f = self.dir / "daily" / f"{sym}.pkl"
|
|
177
|
+
if f.exists():
|
|
178
|
+
return pd.read_pickle(f)
|
|
179
|
+
parts = []
|
|
180
|
+
for k in self._months(sym, "klines"):
|
|
181
|
+
blob = _get(f"{DL_URL}/{k}")
|
|
182
|
+
if blob:
|
|
183
|
+
parts.append(_read_klines_zip(blob))
|
|
184
|
+
raw = pd.concat(parts) if parts else pd.DataFrame(columns=KL_COLS)
|
|
185
|
+
df = _frame_from_klines(raw)
|
|
186
|
+
if live: # top-up from the live API
|
|
187
|
+
start = (df.index[-1] + pd.Timedelta(days=1)) if len(df) else pd.Timestamp("2019-09-01")
|
|
188
|
+
ms = int(start.tz_localize("UTC").timestamp() * 1000)
|
|
189
|
+
blob = _get(f"{API}/fapi/v1/klines?symbol={sym}&interval=1d&startTime={ms}&limit=1500")
|
|
190
|
+
if blob:
|
|
191
|
+
rows = json.loads(blob)
|
|
192
|
+
if rows:
|
|
193
|
+
live_df = _frame_from_klines(pd.DataFrame(rows, columns=KL_COLS).assign(
|
|
194
|
+
open_time=lambda d: d["open_time"].astype("int64")))
|
|
195
|
+
df = pd.concat([df, live_df]).pipe(lambda d: d[~d.index.duplicated(keep="last")]).sort_index()
|
|
196
|
+
if len(df):
|
|
197
|
+
today = pd.Timestamp.utcnow().tz_localize(None).normalize()
|
|
198
|
+
df = df[df.index < today] # never keep an unfinished bar
|
|
199
|
+
df.to_pickle(f)
|
|
200
|
+
return df
|
|
201
|
+
|
|
202
|
+
def fetch_funding(self, sym: str, live: bool) -> pd.Series | None:
|
|
203
|
+
"""Funding rate summed per UTC day (every settlement inside the day), from the monthly fundingRate files plus, when `live` is True,
|
|
204
|
+
the live API. The current day is dropped. Cached as a pickle. For settlement instants see `crypto.intraday.fetch_funding_events`."""
|
|
205
|
+
f = self.dir / "funding" / f"{sym}.pkl"
|
|
206
|
+
if f.exists():
|
|
207
|
+
return pd.read_pickle(f)
|
|
208
|
+
frames = []
|
|
209
|
+
for k in self._months(sym, "fundingRate"):
|
|
210
|
+
blob = _get(f"{DL_URL}/{k}")
|
|
211
|
+
if not blob:
|
|
212
|
+
continue
|
|
213
|
+
with zipfile.ZipFile(io.BytesIO(blob)) as z:
|
|
214
|
+
d = pd.read_csv(z.open(z.namelist()[0]), dtype=str)
|
|
215
|
+
if d.shape[1] >= 3:
|
|
216
|
+
d.columns = ["calc_time", "interval", "rate"][: d.shape[1]] + list(d.columns[3:])
|
|
217
|
+
frames.append(d[["calc_time", "rate"]])
|
|
218
|
+
out = pd.concat(frames) if frames else pd.DataFrame(columns=["calc_time", "rate"])
|
|
219
|
+
out["calc_time"] = pd.to_numeric(out["calc_time"], errors="coerce")
|
|
220
|
+
out["rate"] = pd.to_numeric(out["rate"], errors="coerce")
|
|
221
|
+
out = out.dropna()
|
|
222
|
+
if live:
|
|
223
|
+
last = out["calc_time"].max() if len(out) else 1567296000000
|
|
224
|
+
blob = _get(f"{API}/fapi/v1/fundingRate?symbol={sym}&startTime={int(last) + 1}&limit=1000")
|
|
225
|
+
if blob:
|
|
226
|
+
rows = json.loads(blob)
|
|
227
|
+
if rows:
|
|
228
|
+
add = pd.DataFrame({"calc_time": [r["fundingTime"] for r in rows], "rate": [float(r["fundingRate"]) for r in rows]})
|
|
229
|
+
out = pd.concat([out, add])
|
|
230
|
+
if out.empty:
|
|
231
|
+
s = pd.Series(dtype=float)
|
|
232
|
+
else:
|
|
233
|
+
ts = out["calc_time"].astype("int64")
|
|
234
|
+
unit = "us" if ts.iloc[0] > 10**14 else "ms"
|
|
235
|
+
day = pd.to_datetime(ts, unit=unit, utc=True).dt.tz_localize(None).dt.normalize()
|
|
236
|
+
s = out.set_index(day)["rate"].groupby(level=0).sum().sort_index() # sum of settlements in the UTC day
|
|
237
|
+
today = pd.Timestamp.utcnow().tz_localize(None).normalize()
|
|
238
|
+
s = s[s.index < today]
|
|
239
|
+
s.to_pickle(f)
|
|
240
|
+
return s
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def fetch_all(store: ArchiveStore | None = None, symbols: list[str] | None = None, workers: int = 12,
|
|
244
|
+
progress: bool = True) -> dict:
|
|
245
|
+
"""Download (or load from cache) daily bars and funding for every symbol. Returns counts."""
|
|
246
|
+
store = store or ArchiveStore()
|
|
247
|
+
syms = symbols or store.symbols()
|
|
248
|
+
live = store.trading_now()
|
|
249
|
+
done = {"ok": 0, "empty": 0, "error": 0}
|
|
250
|
+
errs: list[str] = []
|
|
251
|
+
|
|
252
|
+
def one(s: str):
|
|
253
|
+
d = store.fetch_daily(s, s in live)
|
|
254
|
+
f = store.fetch_funding(s, s in live)
|
|
255
|
+
return s, (0 if d is None else len(d)), (0 if f is None else len(f))
|
|
256
|
+
|
|
257
|
+
with cf.ThreadPoolExecutor(workers) as ex:
|
|
258
|
+
futs = {ex.submit(one, s): s for s in syms}
|
|
259
|
+
for n, fu in enumerate(cf.as_completed(futs), 1):
|
|
260
|
+
try:
|
|
261
|
+
_, nd, nf = fu.result()
|
|
262
|
+
done["ok" if nd else "empty"] += 1
|
|
263
|
+
except Exception as e: # one bad symbol must not stop the run
|
|
264
|
+
done["error"] += 1
|
|
265
|
+
errs.append(f"{futs[fu]}: {type(e).__name__}: {e}")
|
|
266
|
+
if progress and n % 25 == 0:
|
|
267
|
+
print(f" {n}/{len(syms)} {done}", flush=True)
|
|
268
|
+
done["errors"] = errs[:20]
|
|
269
|
+
return done
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def execution_rules(rules: pd.DataFrame, tickers, *, missing: str = "raise") -> tuple[pd.Series, pd.Series]:
|
|
273
|
+
"""(lot, min_trade_value) per ticker for `backtest_weights(lot=..., min_trade_value=...)`, from `ArchiveStore.symbol_rules()`.
|
|
274
|
+
|
|
275
|
+
lot is the quantity step (at least the minimum quantity), in units of the contract; min_trade_value the minimum order value in the quote
|
|
276
|
+
currency. `rules` holds today's values only and has no delisted contracts. For a ticker it does not know, `missing="raise"` (default) raises,
|
|
277
|
+
and `missing="nan"` returns NaN for it so that you can leave it out of a small-account run. No substitute rule is invented: a step is in units of
|
|
278
|
+
the contract, which are worth anything from a thousandth of a dollar to tens of thousands, so there is no safe value to fill in."""
|
|
279
|
+
if missing not in ("raise", "nan"):
|
|
280
|
+
raise ValueError(f"missing must be 'raise' or 'nan', not {missing!r}")
|
|
281
|
+
tickers = pd.Index(tickers)
|
|
282
|
+
unknown = tickers[~tickers.isin(rules.index)]
|
|
283
|
+
if len(unknown) and missing == "raise":
|
|
284
|
+
raise ValueError(f"no trading rules for {len(unknown)} tickers (first: {list(unknown[:3])}); they are probably delisted. "
|
|
285
|
+
f"Pass missing='nan' to get NaN for them and leave them out")
|
|
286
|
+
r = rules.reindex(tickers)
|
|
287
|
+
return r[["step_size", "min_qty"]].max(axis=1).rename("lot"), r["min_notional"].rename("min_trade_value")
|