pitbacktest 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pitbacktest/__init__.py +27 -0
- pitbacktest/adapters/__init__.py +0 -0
- pitbacktest/adapters/krx.py +249 -0
- pitbacktest/adapters/long_format.py +87 -0
- pitbacktest/adapters/tiingo.py +236 -0
- pitbacktest/adapters/yfinance.py +243 -0
- pitbacktest/analytics.py +234 -0
- pitbacktest/core/__init__.py +0 -0
- pitbacktest/core/controls.py +126 -0
- pitbacktest/core/costs.py +144 -0
- pitbacktest/core/estimators.py +216 -0
- pitbacktest/core/gates.py +175 -0
- pitbacktest/core/panel.py +306 -0
- pitbacktest/crypto/__init__.py +12 -0
- pitbacktest/crypto/binance_archive.py +287 -0
- pitbacktest/crypto/costs.py +80 -0
- pitbacktest/crypto/intraday.py +326 -0
- pitbacktest/crypto/panel.py +120 -0
- pitbacktest/equity/__init__.py +10 -0
- pitbacktest/equity/master.py +60 -0
- pitbacktest/equity/scenarios.py +104 -0
- pitbacktest/event.py +206 -0
- pitbacktest/execution.py +310 -0
- pitbacktest/ledger.py +128 -0
- pitbacktest/portfolio.py +373 -0
- pitbacktest/screen.py +150 -0
- pitbacktest/shorting.py +46 -0
- pitbacktest/validation.py +118 -0
- pitbacktest/weights.py +271 -0
- pitbacktest-0.2.0.dist-info/METADATA +395 -0
- pitbacktest-0.2.0.dist-info/RECORD +34 -0
- pitbacktest-0.2.0.dist-info/WHEEL +5 -0
- pitbacktest-0.2.0.dist-info/licenses/LICENSE +21 -0
- pitbacktest-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Trading costs and capacity for crypto perpetuals.
|
|
2
|
+
|
|
3
|
+
There is no public order-book history in the archive, so costs here are *estimates*, and they are labelled as such.
|
|
4
|
+
- Fees: a flat taker fee (default 5 bp, the standard retail tier). Change it to your tier.
|
|
5
|
+
- Spread: a flat assumption plus a thin-contract penalty. A daily high-low estimator (Corwin-Schultz) was tried first
|
|
6
|
+
and rejected because it is invalid for crypto volatility (see `liquidity_cost_bp`).
|
|
7
|
+
- Capacity: instead of pretending to model market impact without data, `participation_report` shows how large the
|
|
8
|
+
trades are relative to each contract's turnover for a given account size.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import warnings
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
import pandas as pd
|
|
16
|
+
|
|
17
|
+
from ..core.costs import corwin_schultz
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def liquidity_cost_bp(panel, *, taker_fee_bp: float = 5.0, half_spread_bp: float = 2.0, window: int = 30,
|
|
21
|
+
thin_usd: float = 1e8, thin_extra_bp: float = 5.0,
|
|
22
|
+
spread_estimator: str = "fixed", tick_size=None) -> pd.DataFrame:
|
|
23
|
+
"""(date x ticker) one-way cost in bp = taker fee + half spread (+ a penalty for thin contracts).
|
|
24
|
+
|
|
25
|
+
spread_estimator
|
|
26
|
+
"fixed" a flat half spread (`half_spread_bp`). The default. It is an assumption, not a measurement, so
|
|
27
|
+
use `cost_sensitivity` or a grid of values instead of trusting one number.
|
|
28
|
+
"corwin_schultz" half the Corwin-Schultz estimate from daily highs and lows. **Do not use it for crypto.**
|
|
29
|
+
On Binance perpetuals it gave a median spread of about 1.5% and about 38 bp one way for
|
|
30
|
+
BTC, orders of magnitude above any real quote (the study in `studies/crypto_cross_section`
|
|
31
|
+
documents this in Amendment 1). I did not establish why; daily crypto ranges are dominated
|
|
32
|
+
by jumps and volatility clustering, which the estimator's assumptions do not cover. It is
|
|
33
|
+
kept only to reproduce the preregistered run.
|
|
34
|
+
|
|
35
|
+
"tick" half of one price tick as a share of the price: `0.5 * tick_size / close * 1e4` bp (`tick_size` is a Series by ticker,
|
|
36
|
+
for example `ArchiveStore.symbol_rules()["tick_size"]`; a ticker without one raises). Measured against the order book
|
|
37
|
+
(docs/crypto_spread_check.md): the quoted half spread was one tick for 9 of 12 sampled contracts (0.8 to 1.25 times the
|
|
38
|
+
prediction). The 3 others were 10, 10 and 100 times wider: the exchange has cut their tick since, and `tick_size` is
|
|
39
|
+
today's value, so there it is a **floor**. It is the cost of a small order at the best quote, not of a large one:
|
|
40
|
+
impact is not in it.
|
|
41
|
+
|
|
42
|
+
Contracts whose trailing turnover is below `thin_usd` pay `thin_extra_bp` more, a blunt stand-in for impact.
|
|
43
|
+
Uses only data up to day t."""
|
|
44
|
+
adv = panel.adv(window)
|
|
45
|
+
extra = (adv < thin_usd).astype(float) * thin_extra_bp if adv is not None else 0.0
|
|
46
|
+
if spread_estimator == "corwin_schultz":
|
|
47
|
+
warnings.warn("Corwin-Schultz overstates crypto spreads by orders of magnitude; use only to reproduce old runs.",
|
|
48
|
+
stacklevel=2)
|
|
49
|
+
cs = corwin_schultz(panel.high, panel.low)
|
|
50
|
+
half = (cs.rolling(window, min_periods=10).median() * 0.5 * 1e4).clip(lower=1.0)
|
|
51
|
+
elif spread_estimator == "tick":
|
|
52
|
+
if tick_size is None:
|
|
53
|
+
raise ValueError("spread_estimator='tick' needs tick_size, a Series by ticker")
|
|
54
|
+
tk = pd.Series(tick_size, dtype=float).reindex(panel.tickers)
|
|
55
|
+
if tk.isna().any() or (tk <= 0).any():
|
|
56
|
+
raise ValueError(f"tick_size is missing or not positive for {int((tk.isna() | (tk <= 0)).sum())} tickers "
|
|
57
|
+
f"(first: {list(tk.index[(tk.isna() | (tk <= 0)).to_numpy()][:3])}); a delisted contract has none in the exchange rules")
|
|
58
|
+
half = 0.5 * tk / panel.close * 1e4
|
|
59
|
+
elif spread_estimator == "fixed":
|
|
60
|
+
half = half_spread_bp
|
|
61
|
+
else:
|
|
62
|
+
raise ValueError(spread_estimator)
|
|
63
|
+
return (taker_fee_bp + half + extra).astype(np.float64)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def participation_report(panel, weights: np.ndarray, aum_usd: float, window: int = 30) -> dict:
|
|
67
|
+
"""Trade size as a share of trailing daily turnover, for an account of `aum_usd`.
|
|
68
|
+
|
|
69
|
+
weights (date x ticker) target weights, e.g. the `holdings` of a portfolio. Returns percentiles of
|
|
70
|
+
|delta weight| * aum / ADV over all trades, and the account size at which the 95th percentile would reach 1%
|
|
71
|
+
of a contract's daily turnover (a common ceiling for not moving the market)."""
|
|
72
|
+
adv = panel.adv(window).values
|
|
73
|
+
dw = np.abs(np.diff(weights, axis=0))
|
|
74
|
+
part = dw * aum_usd / np.where(adv[1:] > 0, adv[1:], np.nan)
|
|
75
|
+
part = part[(dw > 1e-9) & np.isfinite(part)]
|
|
76
|
+
if part.size == 0:
|
|
77
|
+
return {"aum_usd": aum_usd, "n_trades": 0}
|
|
78
|
+
p50, p95, p99 = (float(np.percentile(part, q)) for q in (50, 95, 99))
|
|
79
|
+
return {"aum_usd": aum_usd, "n_trades": int(part.size), "participation_p50": p50, "participation_p95": p95,
|
|
80
|
+
"participation_p99": p99, "aum_at_1pct_p95": float(aum_usd * 0.01 / p95) if p95 > 0 else np.inf}
|
|
@@ -0,0 +1,326 @@
|
|
|
1
|
+
"""Intraday bars for USDT perpetuals from the Binance public archive, as a point-in-time Panel.
|
|
2
|
+
|
|
3
|
+
Read `docs/intraday_design.md` first. What this layer can say: how fast an edge decays with latency, and at what cost it stops
|
|
4
|
+
paying. What it cannot say: whether the signal can be traded. The archive has bars, not a book, so there is no queue position, no
|
|
5
|
+
partial fill and no impact below the bar.
|
|
6
|
+
|
|
7
|
+
Time convention. Rows are indexed by bar **end** time (UTC, naive). A row holds only what was known at that instant, so a signal
|
|
8
|
+
from row t is entered at row t + lag with lag >= 1.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import concurrent.futures as cf
|
|
13
|
+
import io
|
|
14
|
+
import zipfile
|
|
15
|
+
from dataclasses import replace
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
import numpy as np
|
|
19
|
+
import pandas as pd
|
|
20
|
+
|
|
21
|
+
from ..core.panel import Panel
|
|
22
|
+
from ..portfolio import backtest_portfolio
|
|
23
|
+
from . import binance_archive as ba
|
|
24
|
+
|
|
25
|
+
COLS = ["open", "high", "low", "close", "volume", "quote_volume", "trades"]
|
|
26
|
+
MIN_PER_DAY = 1440
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# ------------------------------------------------------------------------------------------------------ parsing
|
|
30
|
+
def parse_minute_zip(blob: bytes) -> pd.DataFrame:
|
|
31
|
+
"""One archive file of 1-minute klines -> frame indexed by the bar **end** time (open time + 1 minute).
|
|
32
|
+
Millisecond and microsecond timestamps are told apart per row; header rows and duplicates are dropped."""
|
|
33
|
+
raw = ba._read_klines_zip(blob)
|
|
34
|
+
if raw.empty:
|
|
35
|
+
return pd.DataFrame(columns=COLS)
|
|
36
|
+
ts = raw["open_time"].astype("int64").to_numpy()
|
|
37
|
+
ms = np.where(ts > 10**14, ts // 1000, ts)
|
|
38
|
+
idx = (pd.to_datetime(ms, unit="ms", utc=True).tz_localize(None).astype("datetime64[ns]") + pd.Timedelta(minutes=1)) # ns in every pandas
|
|
39
|
+
df = raw[COLS].apply(pd.to_numeric, errors="coerce")
|
|
40
|
+
df.index = idx
|
|
41
|
+
df = df[~df.index.duplicated(keep="last")].sort_index()
|
|
42
|
+
return df.dropna(subset=["close"])
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _bar_minutes(bar: str) -> int:
|
|
46
|
+
m = int(pd.Timedelta(bar) / pd.Timedelta(minutes=1))
|
|
47
|
+
if m < 1 or MIN_PER_DAY % m:
|
|
48
|
+
raise ValueError(f"bar {bar!r} must be a whole number of minutes that divides a day (1min, 5min, 15min, 1h, ...)")
|
|
49
|
+
return m
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def aggregate_bars(df1m: pd.DataFrame, bar: str) -> pd.DataFrame:
|
|
53
|
+
"""1-minute frame (end-time index) -> bars labelled by their end. Open is the first minute's, close the last, high and low the
|
|
54
|
+
extremes, volumes summed; `n_min` is how many minutes the bar holds. A last bar that would end after the data ends is dropped."""
|
|
55
|
+
m = _bar_minutes(bar)
|
|
56
|
+
if df1m.empty:
|
|
57
|
+
return pd.DataFrame(columns=COLS + ["n_min"])
|
|
58
|
+
if m == 1:
|
|
59
|
+
out = df1m.copy()
|
|
60
|
+
out["n_min"] = 1
|
|
61
|
+
return out
|
|
62
|
+
opn = df1m.copy()
|
|
63
|
+
opn.index = opn.index - pd.Timedelta(minutes=1) # back to open time so bars align to the clock
|
|
64
|
+
g = opn.resample(f"{m}min", label="left", closed="left", origin="epoch")
|
|
65
|
+
out = pd.DataFrame({"open": g["open"].first(), "high": g["high"].max(), "low": g["low"].min(), "close": g["close"].last(),
|
|
66
|
+
"volume": g["volume"].sum(min_count=1), "quote_volume": g["quote_volume"].sum(min_count=1),
|
|
67
|
+
"trades": g["trades"].sum(min_count=1), "n_min": g["close"].count()})
|
|
68
|
+
out = out[out["n_min"] > 0]
|
|
69
|
+
out.index = out.index + pd.Timedelta(minutes=m) # label by end
|
|
70
|
+
return out[out.index <= df1m.index.max()]
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# ------------------------------------------------------------------------------------------------------ download
|
|
74
|
+
def _month_keys(start: pd.Timestamp, end: pd.Timestamp) -> list[pd.Period]:
|
|
75
|
+
return list(pd.period_range(start.to_period("M"), end.to_period("M"), freq="M"))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _paths(store, sym: str) -> Path:
|
|
79
|
+
d = store.dir / "minutes" / sym
|
|
80
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
81
|
+
return d
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def fetch_minutes(store, symbols, start, end, *, workers: int = 6, warmup_days: int = 91, getter=None, today=None) -> dict:
|
|
85
|
+
"""Download 1-minute klines for `symbols` from `start - warmup_days` to `end`. Complete months come from the monthly archive,
|
|
86
|
+
the current partial month from the daily archive. Resumable: a month or day already cached, or recorded as missing (404), is
|
|
87
|
+
not requested again. `getter(key) -> bytes | None` can be injected for tests; `today` is the current UTC date."""
|
|
88
|
+
get = getter or (lambda key: ba._get(f"{ba.DL_URL}/{key}"))
|
|
89
|
+
today = pd.Timestamp(today) if today is not None else pd.Timestamp.utcnow().tz_localize(None).normalize()
|
|
90
|
+
lo = pd.Timestamp(start).normalize() - pd.Timedelta(days=warmup_days)
|
|
91
|
+
hi = min(pd.Timestamp(end).normalize(), today - pd.Timedelta(days=1))
|
|
92
|
+
cur = today.to_period("M")
|
|
93
|
+
stats = {"months": 0, "days": 0, "missing": 0, "cached": 0}
|
|
94
|
+
|
|
95
|
+
def one(sym: str) -> dict:
|
|
96
|
+
s = {"months": 0, "days": 0, "missing": 0, "cached": 0}
|
|
97
|
+
d = _paths(store, sym)
|
|
98
|
+
for per in _month_keys(lo, hi):
|
|
99
|
+
if per < cur: # a complete month
|
|
100
|
+
f, none = d / f"{per}.pkl", d / f"{per}.none"
|
|
101
|
+
if f.exists() or none.exists():
|
|
102
|
+
s["cached"] += 1
|
|
103
|
+
continue
|
|
104
|
+
blob = get(f"data/futures/um/monthly/klines/{sym}/1m/{sym}-1m-{per}.zip")
|
|
105
|
+
if blob is None:
|
|
106
|
+
none.write_text("")
|
|
107
|
+
s["missing"] += 1
|
|
108
|
+
continue
|
|
109
|
+
parse_minute_zip(blob).to_pickle(f) # parsed fully first: a bad file raises and caches nothing
|
|
110
|
+
s["months"] += 1
|
|
111
|
+
else: # the partial month, one file per finished day
|
|
112
|
+
for day in pd.date_range(max(lo, per.start_time), min(hi, per.end_time.normalize())):
|
|
113
|
+
f, none = d / f"d{day:%Y-%m-%d}.pkl", d / f"d{day:%Y-%m-%d}.none"
|
|
114
|
+
if f.exists() or none.exists():
|
|
115
|
+
s["cached"] += 1
|
|
116
|
+
continue
|
|
117
|
+
blob = get(f"data/futures/um/daily/klines/{sym}/1m/{sym}-1m-{day:%Y-%m-%d}.zip")
|
|
118
|
+
if blob is None:
|
|
119
|
+
none.write_text("")
|
|
120
|
+
s["missing"] += 1
|
|
121
|
+
continue
|
|
122
|
+
parse_minute_zip(blob).to_pickle(f)
|
|
123
|
+
s["days"] += 1
|
|
124
|
+
return s
|
|
125
|
+
|
|
126
|
+
with cf.ThreadPoolExecutor(workers) as ex:
|
|
127
|
+
for r in ex.map(one, list(symbols)):
|
|
128
|
+
for k in stats:
|
|
129
|
+
stats[k] += r[k]
|
|
130
|
+
return stats
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def load_minutes(store, sym: str, start, end) -> pd.DataFrame:
|
|
134
|
+
"""Cached 1-minute frame for `sym` between `start` (inclusive) and `end` (inclusive day), end-time index."""
|
|
135
|
+
d = _paths(store, sym)
|
|
136
|
+
parts = [pd.read_pickle(f) for f in sorted(d.glob("*.pkl"))]
|
|
137
|
+
parts = [p for p in parts if len(p)]
|
|
138
|
+
if not parts:
|
|
139
|
+
return pd.DataFrame(columns=COLS)
|
|
140
|
+
df = pd.concat(parts)
|
|
141
|
+
df = df[~df.index.duplicated(keep="last")].sort_index()
|
|
142
|
+
return df[(df.index > pd.Timestamp(start)) & (df.index <= pd.Timestamp(end).normalize() + pd.Timedelta(days=1))]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def parse_funding_zip(blob: bytes) -> pd.DataFrame:
|
|
146
|
+
"""One monthly fundingRate archive file -> DataFrame[calc_time (ms or us), rate]."""
|
|
147
|
+
with zipfile.ZipFile(io.BytesIO(blob)) as z:
|
|
148
|
+
dd = pd.read_csv(z.open(z.namelist()[0]), dtype=str)
|
|
149
|
+
if dd.shape[1] < 3:
|
|
150
|
+
return pd.DataFrame(columns=["calc_time", "rate"])
|
|
151
|
+
dd.columns = ["calc_time", "interval", "rate"][: dd.shape[1]] + list(dd.columns[3:])
|
|
152
|
+
return dd[["calc_time", "rate"]]
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def funding_series(frames: list[pd.DataFrame]) -> pd.Series:
|
|
156
|
+
"""Settlements with timestamps rounded to the nearest minute (the archive stamps them a few milliseconds late)."""
|
|
157
|
+
out = pd.concat(frames) if frames else pd.DataFrame(columns=["calc_time", "rate"])
|
|
158
|
+
t = pd.to_numeric(out["calc_time"], errors="coerce")
|
|
159
|
+
r = pd.to_numeric(out["rate"], errors="coerce")
|
|
160
|
+
ok = t.notna() & r.notna()
|
|
161
|
+
t, r = t[ok].astype("int64").to_numpy(), r[ok].to_numpy(float)
|
|
162
|
+
ms = np.where(t > 10**14, t // 1000, t)
|
|
163
|
+
idx = pd.to_datetime(ms, unit="ms", utc=True).tz_localize(None).astype("datetime64[ns]").round("min")
|
|
164
|
+
return pd.Series(r, index=idx, dtype=float).groupby(level=0).sum().sort_index()
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def fetch_funding_events(store, sym: str, *, lister=None, getter=None) -> pd.Series:
|
|
168
|
+
"""Funding settlements with their instants, cached. The daily store sums them per day and loses the instant, which an
|
|
169
|
+
intraday simulation needs. `lister(sym) -> keys` and `getter(key) -> bytes | None` can be injected for tests."""
|
|
170
|
+
f = store.dir / "funding_events" / f"{sym}.pkl"
|
|
171
|
+
f.parent.mkdir(parents=True, exist_ok=True)
|
|
172
|
+
if f.exists():
|
|
173
|
+
return pd.read_pickle(f)
|
|
174
|
+
keys = lister(sym) if lister else store._months(sym, "fundingRate")
|
|
175
|
+
get = getter or (lambda key: ba._get(f"{ba.DL_URL}/{key}"))
|
|
176
|
+
frames = [parse_funding_zip(blob) for k in keys if (blob := get(k))]
|
|
177
|
+
s = funding_series(frames)
|
|
178
|
+
s.to_pickle(f)
|
|
179
|
+
return s
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
# ------------------------------------------------------------------------------------------------------ the panel
|
|
183
|
+
def estimate_memory(n_bars: int, n_symbols: int, n_matrices: int = 20) -> float:
|
|
184
|
+
"""Rough peak gigabytes: `backtest_portfolio` peaked at about 14 (bar x contract) float64 matrices when measured on a panel
|
|
185
|
+
with close, eligible and volume (tests/memprobe in the design notes); a panel from `build_intraday_panel` also carries open,
|
|
186
|
+
high, low, funding and delisting flags, so 20 is used. Measure again if you change the engine."""
|
|
187
|
+
return n_bars * n_symbols * 8 * n_matrices / 1e9
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def build_intraday_panel(store, symbols, start, end, *, bar: str = "5min", adv_window: int = 30, min_age_days: int = 60,
|
|
191
|
+
min_adv_usd: float = 5e6, top_n: int | None = None, entry_lag: int = 1, max_gb: float = 3.0,
|
|
192
|
+
funding: bool = True, funding_events: dict | None = None) -> Panel:
|
|
193
|
+
"""Point-in-time intraday panel. Eligibility for every bar of day D uses only days up to D - 1. Download first with
|
|
194
|
+
`fetch_minutes(store, symbols, start, end)`, which also loads the warm-up days."""
|
|
195
|
+
symbols = list(symbols)
|
|
196
|
+
if entry_lag < 1:
|
|
197
|
+
raise ValueError("entry_lag must be at least 1 bar: a bar's close is only known at its end, so it cannot be traded on")
|
|
198
|
+
m = _bar_minutes(bar)
|
|
199
|
+
start, end = pd.Timestamp(start).normalize(), pd.Timestamp(end).normalize()
|
|
200
|
+
warm = max(adv_window, min_age_days) + 1
|
|
201
|
+
lo = start - pd.Timedelta(days=warm)
|
|
202
|
+
grid_end = end + pd.Timedelta(days=1)
|
|
203
|
+
grid = pd.date_range(start + pd.Timedelta(minutes=m), grid_end, freq=f"{m}min").astype("datetime64[ns]") # pandas 3 defaults to us
|
|
204
|
+
mem = estimate_memory(len(grid), len(symbols))
|
|
205
|
+
if mem > max_gb:
|
|
206
|
+
fit = [b for b in ("1min", "5min", "15min", "30min", "1h", "2h", "4h") if _bar_minutes(b) >= m and
|
|
207
|
+
estimate_memory(len(grid) * m // _bar_minutes(b), len(symbols)) <= max_gb]
|
|
208
|
+
raise ValueError(f"about {mem:.1f} GB needed for {len(grid):,} bars x {len(symbols)} contracts (limit {max_gb}); "
|
|
209
|
+
f"use a longer bar ({fit[0] if fit else 'fewer contracts or a shorter period'}), fewer contracts or a shorter period")
|
|
210
|
+
bars: dict[str, pd.DataFrame] = {}
|
|
211
|
+
for s in symbols:
|
|
212
|
+
d = _paths(store, s)
|
|
213
|
+
missing = [p for p in _month_keys(lo, min(end, pd.Timestamp.utcnow().tz_localize(None).normalize() - pd.Timedelta(days=1)))
|
|
214
|
+
if p < pd.Timestamp.utcnow().tz_localize(None).to_period("M")
|
|
215
|
+
and not (d / f"{p}.pkl").exists() and not (d / f"{p}.none").exists()]
|
|
216
|
+
if missing:
|
|
217
|
+
raise ValueError(f"{s}: months not downloaded {[str(p) for p in missing[:4]]}; run fetch_minutes(store, symbols, start, end) "
|
|
218
|
+
f"first (it loads {warm} warm-up days before the start)")
|
|
219
|
+
raw = load_minutes(store, s, lo, end)
|
|
220
|
+
if raw.empty:
|
|
221
|
+
continue
|
|
222
|
+
b = aggregate_bars(raw, bar)
|
|
223
|
+
bars[s] = b
|
|
224
|
+
if not bars:
|
|
225
|
+
raise ValueError("no data for any symbol")
|
|
226
|
+
cols = sorted(bars)
|
|
227
|
+
|
|
228
|
+
# ---- terminal zero-volume tail (a halted or delisted contract keeps printing flat bars): drop only that tail ------------
|
|
229
|
+
zero_interior = 0
|
|
230
|
+
for s in cols:
|
|
231
|
+
b = bars[s]
|
|
232
|
+
live = np.flatnonzero((b["quote_volume"] > 0).to_numpy())
|
|
233
|
+
if len(live) == 0:
|
|
234
|
+
bars[s] = b.iloc[0:0]
|
|
235
|
+
continue
|
|
236
|
+
zero_interior += int((b["quote_volume"].iloc[: live[-1] + 1] <= 0).sum())
|
|
237
|
+
bars[s] = b.iloc[: live[-1] + 1]
|
|
238
|
+
cols = [s for s in cols if len(bars[s])]
|
|
239
|
+
|
|
240
|
+
# ---- daily liquidity, then eligibility from the PREVIOUS day -----------------------------------------------------
|
|
241
|
+
day_of = lambda ix: (ix - pd.Timedelta(minutes=1)).floor("D")
|
|
242
|
+
dq = pd.DataFrame({s: bars[s]["quote_volume"].groupby(day_of(bars[s].index)).sum() for s in cols})
|
|
243
|
+
dq = dq.reindex(pd.date_range(dq.index.min(), max(dq.index.max(), end), freq="D"))
|
|
244
|
+
has = dq.fillna(0) > 0
|
|
245
|
+
age = has.cumsum()
|
|
246
|
+
adv = dq.where(has).rolling(adv_window, min_periods=adv_window).median()
|
|
247
|
+
ok_day = (age >= min_age_days) & (adv >= min_adv_usd)
|
|
248
|
+
if top_n:
|
|
249
|
+
rank = adv.where(ok_day).rank(axis=1, ascending=False, method="first")
|
|
250
|
+
ok_day = ok_day & (rank <= top_n)
|
|
251
|
+
ok_for_day = ok_day.shift(1, fill_value=False) # day D is judged by days <= D - 1
|
|
252
|
+
|
|
253
|
+
def mat(col: str) -> pd.DataFrame:
|
|
254
|
+
return pd.DataFrame({s: bars[s][col] for s in cols}).reindex(grid)
|
|
255
|
+
|
|
256
|
+
close, high, low, opn = mat("close"), mat("high"), mat("low"), mat("open")
|
|
257
|
+
qv = mat("quote_volume")
|
|
258
|
+
elig = ok_for_day.reindex(day_of(grid)).fillna(False).astype(bool).to_numpy()
|
|
259
|
+
eligible = pd.DataFrame(elig, index=grid, columns=cols) & close.notna()
|
|
260
|
+
|
|
261
|
+
# ---- delisting: last real bar of a contract that ended well before the panel end ---------------------------------------
|
|
262
|
+
last = close.apply(lambda c: c.last_valid_index())
|
|
263
|
+
da = pd.DataFrame(False, index=grid, columns=cols)
|
|
264
|
+
for s in cols:
|
|
265
|
+
if last[s] is not None and last[s] < grid[-1] - pd.Timedelta(days=3):
|
|
266
|
+
da.loc[last[s], s] = True
|
|
267
|
+
|
|
268
|
+
# ---- funding at the settlement instant ----------------------------------------------------------------------------
|
|
269
|
+
fund, n_events = None, 0
|
|
270
|
+
if funding:
|
|
271
|
+
F = np.zeros((len(grid), len(cols)))
|
|
272
|
+
for j, s in enumerate(cols):
|
|
273
|
+
ev = funding_events[s] if funding_events is not None and s in funding_events else fetch_funding_events(store, s)
|
|
274
|
+
ev = ev[(ev.index > grid[0] - pd.Timedelta(minutes=m)) & (ev.index <= grid[-1])]
|
|
275
|
+
if ev.empty:
|
|
276
|
+
continue
|
|
277
|
+
pos = grid.searchsorted(ev.index.to_numpy(), side="left") # first bar whose end is at or after the settlement
|
|
278
|
+
ok = pos < len(grid)
|
|
279
|
+
np.add.at(F, (pos[ok], np.full(int(ok.sum()), j)), ev.to_numpy(float)[ok])
|
|
280
|
+
n_events += int(ok.sum())
|
|
281
|
+
fund = pd.DataFrame(F, index=grid, columns=cols)
|
|
282
|
+
panel = Panel(close=close, eligible=eligible, open=opn, high=high, low=low, volume=qv / close, market="CRYPTO-INTRADAY",
|
|
283
|
+
entry_lag=entry_lag, periods_per_year=int(365 * MIN_PER_DAY / m), funding=fund, delist_after=da)
|
|
284
|
+
n = panel.eligible.sum(axis=1)
|
|
285
|
+
panel.meta = {"bar": bar, "bar_minutes": m, "symbols_used": len(cols), "bars": len(grid), "partial_bars": int(sum((bars[s]["n_min"] < m).sum() for s in cols)),
|
|
286
|
+
"zero_volume_bars_kept": zero_interior, "delisted_in_panel": int(da.values.any(axis=0).sum()),
|
|
287
|
+
"eligible_median": float(n[n > 0].median()) if (n > 0).any() else 0.0, "estimated_gb": round(mem, 2),
|
|
288
|
+
"funding_events": n_events}
|
|
289
|
+
return panel
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
# ------------------------------------------------------------------------------------------------------ analysis
|
|
293
|
+
def backtest_intraday(panel: Panel, factor: pd.DataFrame, *, one_way_bp: float = 0.0, long_q: float = 0.2, short_q: float | None = 0.2,
|
|
294
|
+
hold: int = 1, **kw):
|
|
295
|
+
"""`backtest_portfolio` with the cost given as **one-way** basis points (the engine's scalar is a round trip), no grid and no
|
|
296
|
+
benchmark. `hold` is in bars."""
|
|
297
|
+
if panel.entry_lag < 1:
|
|
298
|
+
raise ValueError("an intraday panel needs entry_lag >= 1: a bar's close is only known when the bar ends, so it cannot be traded on")
|
|
299
|
+
kw.setdefault("benchmark", None)
|
|
300
|
+
kw.setdefault("grid", False)
|
|
301
|
+
return backtest_portfolio(panel, factor, long_q=long_q, short_q=short_q, hold=hold, spread_bp=2.0 * one_way_bp, **kw)
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def latency_sweep(panel: Panel, factor: pd.DataFrame, lags=(1, 2, 5, 15), *, one_way_bp: float = 0.0, **kw) -> pd.DataFrame:
|
|
305
|
+
"""The same portfolio entered `lag` bars after the signal. A real edge decays smoothly with the delay."""
|
|
306
|
+
m = panel.meta.get("bar_minutes", 1)
|
|
307
|
+
rows = []
|
|
308
|
+
for k in lags:
|
|
309
|
+
r = backtest_intraday(replace(panel, entry_lag=int(k)), factor, one_way_bp=one_way_bp, **kw)
|
|
310
|
+
met = r.metrics
|
|
311
|
+
rows.append({"lag_bars": int(k), "delay_minutes": int(k) * m, "sharpe": met["Sharpe"], "cagr": met["CAGR"],
|
|
312
|
+
"gross_cagr": met["gross_CAGR"], "turnover_per_bar": met["turnover_daily"]})
|
|
313
|
+
return pd.DataFrame(rows)
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def breakeven_cost(panel: Panel, factor: pd.DataFrame, **kw) -> dict:
|
|
317
|
+
"""One-way cost in bp at which the net mean return is zero. Net return is linear in the cost per unit traded, so two runs
|
|
318
|
+
give it exactly: c* = mean(net at 0 bp) / mean(net at 0 bp - net at 1 bp)."""
|
|
319
|
+
kw.pop("one_way_bp", None)
|
|
320
|
+
r0 = backtest_intraday(panel, factor, one_way_bp=0.0, **kw)
|
|
321
|
+
r1 = backtest_intraday(panel, factor, one_way_bp=1.0, **kw)
|
|
322
|
+
n0, n1 = r0.net_returns, r1.net_returns
|
|
323
|
+
per_bp = float((n0 - n1).mean())
|
|
324
|
+
mean0 = float(n0.mean())
|
|
325
|
+
c = mean0 / per_bp if per_bp > 0 else float("nan")
|
|
326
|
+
return {"breakeven_one_way_bp": c, "net_mean_at_0bp": mean0, "cost_per_bp": per_bp, "gross_sharpe": r0.metrics["Sharpe"]}
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""Build a point-in-time (PIT) Panel from the Binance USDT-M archive.
|
|
2
|
+
|
|
3
|
+
Bias controls, in order of importance
|
|
4
|
+
1. Survivorship: the universe is every contract that ever traded, delisted ones included. `survivors_only=True`
|
|
5
|
+
reproduces the usual shortcut (only contracts that still trade today) so the bias can be measured, not guessed.
|
|
6
|
+
2. Universe look-ahead: a contract is eligible on day t only if, using data up to and including day t, it has
|
|
7
|
+
enough history and enough trailing dollar volume. Nothing from t+1 is used.
|
|
8
|
+
3. Stale prices: bars with zero volume (a frozen last price after a halt or a delisting) are dropped so they
|
|
9
|
+
cannot create fake zero returns or fake liquidity.
|
|
10
|
+
4. Funding: the daily funding rate is attached to the panel and charged by `backtest_portfolio`.
|
|
11
|
+
5. Delisting: the last real bar of a contract that stopped trading is marked in `delist_after`, so the engine can
|
|
12
|
+
apply an explicit delisting return instead of silently assuming the position was closed at the last price.
|
|
13
|
+
|
|
14
|
+
Prices are daily UTC closes. Volumes are converted so that `close * volume` equals the USDT turnover.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
|
|
23
|
+
from ..core.panel import Panel
|
|
24
|
+
from .binance_archive import ArchiveStore
|
|
25
|
+
|
|
26
|
+
# Pegged coins, fiat pairs and index perpetuals are not tradable risk assets for a cross-sectional study.
|
|
27
|
+
DEFAULT_EXCLUDE = re.compile(
|
|
28
|
+
r"^(USDC|BUSD|TUSD|FDUSD|USDP|DAI|USTC|EUR|GBP|AUD|TRY|BRL)USDT$|^(BTCDOM|DEFI|FOOTBALL|BLUEBIRD|ALL)USDT$|DOMUSDT$"
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def build_panel(store: ArchiveStore | None = None, *, symbols: list[str] | None = None,
|
|
33
|
+
start: str = "2020-06-01", end: str | None = None,
|
|
34
|
+
adv_window: int = 30, min_adv_usd: float = 5e6, min_age_days: int = 60,
|
|
35
|
+
top_n: int | None = None, survivors_only: bool = False,
|
|
36
|
+
exclude: re.Pattern | None = DEFAULT_EXCLUDE, entry_lag: int = 1,
|
|
37
|
+
live: set[str] | None = None) -> Panel:
|
|
38
|
+
"""Return a Panel with PIT eligibility, funding and delisting information.
|
|
39
|
+
|
|
40
|
+
min_adv_usd trailing median USDT turnover needed to be eligible (median, so one spike cannot qualify a coin)
|
|
41
|
+
min_age_days bars a contract must already have, so the first weeks after a listing are excluded
|
|
42
|
+
top_n optionally keep only the top-N by trailing turnover among eligible contracts, each day
|
|
43
|
+
survivors_only keep only contracts still trading today (for measuring survivorship bias)
|
|
44
|
+
"""
|
|
45
|
+
store = store or ArchiveStore()
|
|
46
|
+
syms = symbols or store.symbols()
|
|
47
|
+
if exclude is not None:
|
|
48
|
+
syms = [s for s in syms if not exclude.search(s)]
|
|
49
|
+
if survivors_only and live is None:
|
|
50
|
+
live = store.trading_now()
|
|
51
|
+
live = live or set()
|
|
52
|
+
bars: dict[str, pd.DataFrame] = {}
|
|
53
|
+
fund: dict[str, pd.Series] = {}
|
|
54
|
+
for s in syms:
|
|
55
|
+
f = store.dir / "daily" / f"{s}.pkl"
|
|
56
|
+
d = pd.read_pickle(f) if f.exists() else store.fetch_daily(s, False)
|
|
57
|
+
if d is None or d.empty:
|
|
58
|
+
continue
|
|
59
|
+
if survivors_only and s not in live:
|
|
60
|
+
continue
|
|
61
|
+
d = d[d["quote_volume"] > 0] # drop frozen, zero-volume bars (halts, delisting tail)
|
|
62
|
+
d = d[d["close"] > 0]
|
|
63
|
+
if d.empty:
|
|
64
|
+
continue
|
|
65
|
+
bars[s] = d
|
|
66
|
+
ff = store.dir / "funding" / f"{s}.pkl"
|
|
67
|
+
fund[s] = pd.read_pickle(ff) if ff.exists() else store.fetch_funding(s, False)
|
|
68
|
+
if not bars:
|
|
69
|
+
raise ValueError("no data: run pitbacktest.crypto.fetch_all() first")
|
|
70
|
+
|
|
71
|
+
idx = pd.date_range(min(d.index[0] for d in bars.values()), max(d.index[-1] for d in bars.values()), freq="D")
|
|
72
|
+
cols = sorted(bars)
|
|
73
|
+
|
|
74
|
+
def mat(col: str) -> pd.DataFrame:
|
|
75
|
+
return pd.DataFrame({s: bars[s][col] for s in cols}).reindex(idx)
|
|
76
|
+
|
|
77
|
+
close, open_, high, low = mat("close"), mat("open"), mat("high"), mat("low")
|
|
78
|
+
qv = mat("quote_volume")
|
|
79
|
+
funding = pd.DataFrame({s: fund[s] for s in cols if fund.get(s) is not None and len(fund[s])}).reindex(idx)
|
|
80
|
+
funding = funding.reindex(columns=cols)
|
|
81
|
+
|
|
82
|
+
# --- point-in-time eligibility: only information up to and including day t ---
|
|
83
|
+
age = close.notna().cumsum()
|
|
84
|
+
adv = qv.rolling(adv_window, min_periods=adv_window).median()
|
|
85
|
+
ok = close.notna() & (age >= min_age_days) & (adv >= min_adv_usd)
|
|
86
|
+
if top_n:
|
|
87
|
+
rank = adv.where(ok).rank(axis=1, ascending=False, method="first")
|
|
88
|
+
ok = ok & (rank <= top_n)
|
|
89
|
+
|
|
90
|
+
# --- delisting: last real bar of a contract that ended well before the panel end ---
|
|
91
|
+
last = close.apply(lambda c: c.last_valid_index())
|
|
92
|
+
delist_after = pd.DataFrame(False, index=idx, columns=cols)
|
|
93
|
+
for s in cols:
|
|
94
|
+
lv = last[s]
|
|
95
|
+
if lv is not None and lv < idx[-1] - pd.Timedelta(days=3):
|
|
96
|
+
delist_after.loc[lv, s] = True
|
|
97
|
+
|
|
98
|
+
# --- crypto-specific controls (replace the equity ROA / book-to-market controls) ---
|
|
99
|
+
ret = close.pct_change(fill_method=None)
|
|
100
|
+
btc = ret["BTCUSDT"] if "BTCUSDT" in ret.columns else ret.median(axis=1)
|
|
101
|
+
beta = ret.rolling(60, min_periods=40).cov(btc).div(btc.rolling(60, min_periods=40).var(), axis=0)
|
|
102
|
+
chars = {"log_dollar_vol": np.log(qv.rolling(adv_window, min_periods=adv_window).mean().clip(lower=1)),
|
|
103
|
+
"btc_beta_60": beta}
|
|
104
|
+
|
|
105
|
+
vol_base = qv / close # so that close * volume == USDT turnover
|
|
106
|
+
keep = idx >= pd.Timestamp(start)
|
|
107
|
+
if end:
|
|
108
|
+
keep &= idx <= pd.Timestamp(end)
|
|
109
|
+
sl = lambda df: df.loc[keep]
|
|
110
|
+
p = Panel(close=sl(close), eligible=sl(ok), open=sl(open_), high=sl(high), low=sl(low), volume=sl(vol_base),
|
|
111
|
+
chars={k: sl(v) for k, v in chars.items()}, market="CRYPTO", entry_lag=entry_lag, periods_per_year=365,
|
|
112
|
+
funding=sl(funding), delist_after=sl(delist_after))
|
|
113
|
+
n = p.eligible.sum(axis=1)
|
|
114
|
+
p.meta = {
|
|
115
|
+
"symbols_in_archive": len(store.symbols()), "symbols_used": len(cols),
|
|
116
|
+
"survivors_only": survivors_only, "delisted_in_panel": int(delist_after.values.any(axis=0).sum()),
|
|
117
|
+
"eligible_median": float(n[n > 0].median()), "min_adv_usd": min_adv_usd, "min_age_days": min_age_days,
|
|
118
|
+
"funding_coverage": float(funding.reindex(index=p.close.index).notna().values[p.close.notna().values].mean()),
|
|
119
|
+
}
|
|
120
|
+
return p
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""Equity survivorship tools that work without paid data.
|
|
2
|
+
|
|
3
|
+
None of this removes survivorship bias from a survivors-only price source; it measures how much of the real
|
|
4
|
+
universe is missing and shows, by explicit scenarios, how much that could matter. Real removal needs point-in-time data
|
|
5
|
+
(see `pitbacktest.adapters.long_format` for a strict way to plug it in, and `pitbacktest.adapters.krx` for Korea).
|
|
6
|
+
"""
|
|
7
|
+
from .master import load_us_master, universe_coverage
|
|
8
|
+
from .scenarios import inject_delistings, survivorship_scenarios, survivors_only
|
|
9
|
+
|
|
10
|
+
__all__ = ["load_us_master", "universe_coverage", "inject_delistings", "survivorship_scenarios", "survivors_only"]
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""A free, keyless list of US stock tickers with their listing windows, delisted ones included.
|
|
2
|
+
|
|
3
|
+
Source: Tiingo's published `supported_tickers.zip` (no API key needed). It is a **security master, not a price
|
|
4
|
+
source**: it tells you which tickers existed in which period. It is incomplete before about 2013 and misses some recent
|
|
5
|
+
failures (for example SIVB and FRC were absent when this was written), so coverage numbers built on it are lower
|
|
6
|
+
bounds on how many names a survivors-only panel is missing.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import io
|
|
11
|
+
import urllib.request
|
|
12
|
+
import zipfile
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
import pandas as pd
|
|
16
|
+
|
|
17
|
+
URL = "https://apimedia.tiingo.com/docs/tiingo/daily/supported_tickers.zip"
|
|
18
|
+
US_EXCHANGES = {"NYSE", "NASDAQ", "NYSE MKT", "NYSE ARCA", "AMEX", "NYSE American"}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def load_us_master(cache_dir: str | Path | None = None, refresh: bool = False) -> pd.DataFrame:
|
|
22
|
+
"""DataFrame [ticker, exchange, start, end, alive_today] for US-listed common stocks."""
|
|
23
|
+
d = Path(cache_dir or Path.home() / ".cache" / "quantbt").expanduser()
|
|
24
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
25
|
+
f = d / "tiingo_supported_tickers.zip"
|
|
26
|
+
if refresh or not f.exists():
|
|
27
|
+
req = urllib.request.Request(URL, headers={"User-Agent": "pitbacktest-research/0.2 (public data only)"})
|
|
28
|
+
f.write_bytes(urllib.request.urlopen(req, timeout=60).read())
|
|
29
|
+
with zipfile.ZipFile(f) as z:
|
|
30
|
+
df = pd.read_csv(io.BytesIO(z.read(z.namelist()[0])))
|
|
31
|
+
return _clean(df)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _clean(df: pd.DataFrame) -> pd.DataFrame:
|
|
35
|
+
df = df[(df["assetType"] == "Stock") & df["exchange"].isin(US_EXCHANGES)].copy()
|
|
36
|
+
df["start"] = pd.to_datetime(df["startDate"], errors="coerce")
|
|
37
|
+
df["end"] = pd.to_datetime(df["endDate"], errors="coerce")
|
|
38
|
+
df = df.dropna(subset=["start", "end"])
|
|
39
|
+
df["alive_today"] = df["end"] >= df["end"].max() - pd.Timedelta(days=7)
|
|
40
|
+
return df[["ticker", "exchange", "start", "end", "alive_today"]].reset_index(drop=True)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def universe_coverage(panel_tickers, master: pd.DataFrame, years=range(2010, 2025)) -> pd.DataFrame:
|
|
44
|
+
"""For each year: how many US stocks were listed, how many of them are in your panel, and (the telling number)
|
|
45
|
+
how many of the names that stopped trading that year are in your panel.
|
|
46
|
+
|
|
47
|
+
A ticker only counts as present if its listing window in the master overlaps the year, so a reused ticker is not
|
|
48
|
+
credited to the company that used to have it (the master has one row per ticker and window)."""
|
|
49
|
+
have = set(panel_tickers)
|
|
50
|
+
rows = []
|
|
51
|
+
for y in years:
|
|
52
|
+
a, b = pd.Timestamp(f"{y}-01-01"), pd.Timestamp(f"{y}-12-31")
|
|
53
|
+
alive = master[(master["start"] <= b) & (master["end"] >= a)]
|
|
54
|
+
died = alive[(alive["end"] <= b) & ~alive["alive_today"]]
|
|
55
|
+
rows.append({"year": y, "listed": len(alive), "in_panel": int(alive["ticker"].isin(have).sum()),
|
|
56
|
+
"stopped_trading": len(died), "stopped_in_panel": int(died["ticker"].isin(have).sum())})
|
|
57
|
+
out = pd.DataFrame(rows).set_index("year")
|
|
58
|
+
out["share_listed_in_panel"] = out["in_panel"] / out["listed"].replace(0, float("nan"))
|
|
59
|
+
out["share_stopped_in_panel"] = out["stopped_in_panel"] / out["stopped_trading"].replace(0, float("nan"))
|
|
60
|
+
return out
|