pitbacktest 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,306 @@
1
+ """Panel convention: the input contract of every test.
2
+
3
+ Design principles
4
+ - The core is **plain pandas and numpy** and does not depend on any external data source (adapters do that).
5
+ - Every matrix is (date x ticker) and the index and columns must be identical.
6
+ - `eligible` is the **point-in-time universe**: True only for securities that could actually be traded that day.
7
+ It is the only device against survivorship bias, so it is a required input.
8
+
9
+ Timing convention (the only one used in the whole project)
10
+ A signal is **fixed** at close(t). Entry is at close(t+entry_lag) and exit at close(t+entry_lag+h).
11
+ entry_lag defaults to 1: "see the signal, buy the next day". 0 assumes a fill at the same close (aggressive).
12
+ Breaking this convention is caught by Panel.assert_no_lookahead().
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass, field, replace
18
+ from typing import Iterable
19
+
20
+ import numpy as np
21
+ import pandas as pd
22
+
23
+ REQUIRED = ("close", "eligible")
24
+ OPTIONAL = ("open", "high", "low", "volume", "mkt_cap", "funding", "delist_after")
25
+
26
+
27
+ @dataclass
28
+ class Panel:
29
+ """A bundle of (date x ticker) matrices.
30
+
31
+ close adjusted closing price: the only source of returns
32
+ eligible point-in-time universe bool: listed, tradable and passing the liquidity condition that day
33
+ The rest is optional. With volume and mkt_cap the liquidity and size controls are switched on.
34
+ """
35
+
36
+ close: pd.DataFrame
37
+ eligible: pd.DataFrame
38
+ open: pd.DataFrame | None = None
39
+ high: pd.DataFrame | None = None
40
+ low: pd.DataFrame | None = None
41
+ volume: pd.DataFrame | None = None
42
+ mkt_cap: pd.DataFrame | None = None
43
+ chars: dict[str, pd.DataFrame] = field(default_factory=dict)
44
+ market: str = "KR"
45
+ entry_lag: int = 1
46
+ # Trading days per year used for annualising: 252 for stocks, 365 for 24/7 markets (crypto). A wrong value makes CAGR, Sharpe and volatility wrong.
47
+ periods_per_year: int = 252
48
+ # Optional: futures funding (date x ticker, daily sum, positive means longs pay). Delisting flag (True on the last real bar). Notes for adapters.
49
+ funding: pd.DataFrame | None = None
50
+ delist_after: pd.DataFrame | None = None
51
+ # Optional: (date x ticker) bool, True where a security can be sold short on the execution day (borrowable, and no short-selling ban). None means every
52
+ # security can. A date or security missing from the frame counts as **not** shortable: an unknown is not permission.
53
+ shortable: pd.DataFrame | None = None
54
+ # Optional: (date x ticker) bool, False where a trade that increases (can_buy) or decreases (can_sell) a position cannot be done on that day
55
+ # (halted, or locked at a daily price limit). None means always possible; a date or security missing from the frame counts as not possible.
56
+ # `execution.tradability` builds them from prices and volume.
57
+ can_buy: pd.DataFrame | None = None
58
+ can_sell: pd.DataFrame | None = None
59
+ meta: dict = field(default_factory=dict)
60
+
61
+ # ---------------------------------------------------------------- construction and validation
62
+ def __post_init__(self) -> None:
63
+ self.close = self.close.sort_index()
64
+ if self.close.index.has_duplicates: # before any reindex, which would fail with a less helpful pandas message
65
+ raise ValueError("close.index has duplicate dates")
66
+ self.eligible = self.eligible.reindex(
67
+ index=self.close.index, columns=self.close.columns
68
+ ).fillna(False).astype(bool)
69
+ for k in OPTIONAL:
70
+ v = getattr(self, k)
71
+ if v is not None:
72
+ setattr(self, k, v.reindex(index=self.close.index, columns=self.close.columns))
73
+ self.chars = {
74
+ k: v.reindex(index=self.close.index, columns=self.close.columns)
75
+ for k, v in self.chars.items()
76
+ }
77
+ if self.delist_after is not None:
78
+ self.delist_after = self.delist_after.fillna(False).astype(bool)
79
+ for k in ("shortable", "can_buy", "can_sell"):
80
+ v = getattr(self, k)
81
+ if v is not None:
82
+ setattr(self, k, v.reindex(index=self.close.index, columns=self.close.columns).fillna(False).astype(bool))
83
+ self.validate()
84
+
85
+ def validate(self) -> None:
86
+ """Raise ValueError if the panel breaks its invariants: the date index ascending and without duplicates, at least one eligible
87
+ security, `entry_lag` not negative, and at least 10 eligible securities on half of the dates or more. Runs on construction."""
88
+ if not self.close.index.is_monotonic_increasing:
89
+ raise ValueError("close.index is not ascending")
90
+ if self.close.index.has_duplicates:
91
+ raise ValueError("close.index has duplicate dates")
92
+ if self.eligible.values.sum() == 0:
93
+ raise ValueError("eligible is all False")
94
+ if self.entry_lag < 0:
95
+ raise ValueError("entry_lag must be 0 or more")
96
+ n = self.eligible.sum(axis=1)
97
+ if (n[n > 0] < 10).mean() > 0.5:
98
+ raise ValueError(
99
+ "eligible has fewer than 10 securities on more than half of the dates: cross-sectional tests are impossible"
100
+ )
101
+
102
+ # ---------------------------------------------------------------- derived
103
+ @property
104
+ def dates(self) -> pd.DatetimeIndex:
105
+ return self.close.index
106
+
107
+ @property
108
+ def tickers(self) -> pd.Index:
109
+ return self.close.columns
110
+
111
+ def ret1(self) -> pd.DataFrame:
112
+ """Daily return close(t)/close(t-1) - 1."""
113
+ return self.close.pct_change(fill_method=None)
114
+
115
+ def forward(self, h: int, delist_return: float | None = None) -> pd.DataFrame:
116
+ """Return from close(t + lag) to close(t + lag + h) for a signal at t: a plain price ratio, NaN when either price is missing.
117
+
118
+ A security flagged in `delist_after` is carried at its last price after its last bar (as cash), or at `last price x (1 + delist_return)`
119
+ when `delist_return` is given, so a holding period that runs into a delisting keeps its result instead of dropping out. Without that
120
+ the statistics of `screen` and `backtest_event` would leave out exactly the trades that ended in a delisting. A security that is only
121
+ missing for some days (a halt) still gives NaN when an end price is missing. Compare `analytics.forward_returns`, which compounds daily
122
+ returns and treats a missing price as a return of 0."""
123
+ lag = self.entry_lag
124
+ px = self.close
125
+ if self.delist_after is not None and bool(np.asarray(self.delist_after.values).any()):
126
+ da = self.delist_after.reindex(index=px.index, columns=px.columns).fillna(False).astype(bool).astype(int)
127
+ gone = (da.cumsum() - da) > 0 # strictly after the security's last bar
128
+ carried = px.ffill() * (1.0 + (0.0 if delist_return is None else float(delist_return)))
129
+ px = px.where(~gone, carried)
130
+ entry = px.shift(-lag)
131
+ exit_ = px.shift(-(lag + h))
132
+ return (exit_ / entry - 1.0).astype(np.float32)
133
+
134
+ def fingerprint(self) -> str:
135
+ """Short hash of the numbers this panel holds (every matrix that is present, including volume and market cap, which
136
+ the impact model and the size controls read), used by the trial ledger to tell whether two runs saw the same data.
137
+ The date index and the ticker names are not hashed: the same values in the same shape give the same fingerprint."""
138
+ import hashlib
139
+ h = hashlib.sha256()
140
+ mats = [(n, getattr(self, n)) for n in ("close", "eligible", "open", "high", "low", "volume", "mkt_cap", "funding", "delist_after", "shortable", "can_buy", "can_sell")]
141
+ mats += [(f"chars.{k}", self.chars[k]) for k in sorted(self.chars)]
142
+ for name, v in mats:
143
+ if v is not None:
144
+ a = np.ascontiguousarray(v.to_numpy(dtype=np.float32))
145
+ h.update(name.encode() + repr(a.shape).encode() + a.tobytes())
146
+ h.update(f"{self.entry_lag}|{self.periods_per_year}|{self.market}".encode())
147
+ return h.hexdigest()[:16]
148
+
149
+ def adv(self, window: int = 20) -> pd.DataFrame:
150
+ """Average traded value. None if there is no volume."""
151
+ if self.volume is None:
152
+ return None
153
+ return (self.close * self.volume).rolling(window, min_periods=window).mean()
154
+
155
+ # ---------------------------------------------------------------- safeguards
156
+ def truncate(self, last: pd.Timestamp) -> "Panel":
157
+ """A copy that holds only the rows up to and including `last`: what a user standing on that date would have had."""
158
+ cut = lambda f: None if f is None else f.loc[:last]
159
+ return replace(self, close=self.close.loc[:last], eligible=self.eligible.loc[:last], open=cut(self.open), high=cut(self.high), low=cut(self.low),
160
+ volume=cut(self.volume), mkt_cap=cut(self.mkt_cap), funding=cut(self.funding), delist_after=cut(self.delist_after),
161
+ shortable=cut(self.shortable), can_buy=cut(self.can_buy), can_sell=cut(self.can_sell),
162
+ chars={k: v.loc[:last] for k, v in self.chars.items()})
163
+
164
+ def assert_causal(self, make_signal, n_cuts: int = 6, seed: int = 0, min_history: int = 300, rtol: float = 1e-9) -> dict:
165
+ """The exact look-ahead check: a signal that is a function of the panel must not change when the future is removed.
166
+
167
+ `make_signal(panel)` returns the (date x ticker) signal. For `n_cuts` random dates t (at least `min_history` rows in, so that rolling windows are full) the signal is
168
+ rebuilt from the panel truncated at t, and its last row must equal the row of the full-data signal at t, name by name (NaN equal to NaN). Any use of a later row
169
+ (a negative shift, a centred window, a mean or standard deviation over the whole history, a rank across time) makes the rows differ.
170
+ Returns {"pass", "cuts", "mismatched_cuts", "max_abs_diff"}. It checks that the function is causal; it does not say the signal is good, and it cannot see look-ahead in
171
+ the *data* itself (a price already revised), only in how the signal is computed. It needs the function, not a frame: that is what makes it exact."""
172
+ full = make_signal(self)
173
+ if not isinstance(full, pd.DataFrame):
174
+ raise ValueError("make_signal must return a (date x ticker) DataFrame")
175
+ n = len(self.dates)
176
+ if n <= min_history + 1:
177
+ raise ValueError(f"the panel has {n} rows; at least {min_history + 2} are needed to cut it")
178
+ rng = np.random.default_rng(seed)
179
+ cuts = sorted(int(i) for i in rng.choice(np.arange(min_history, n - 1), size=min(n_cuts, n - 1 - min_history), replace=False))
180
+ bad, worst = [], 0.0
181
+ for i in cuts:
182
+ t = self.dates[i]
183
+ part = make_signal(self.truncate(t))
184
+ a = part.iloc[-1].reindex(self.tickers).to_numpy(float)
185
+ b = full.loc[t].reindex(self.tickers).to_numpy(float)
186
+ same_nan = np.isnan(a) == np.isnan(b)
187
+ both = ~np.isnan(a) & ~np.isnan(b)
188
+ diff = float(np.max(np.abs(a[both] - b[both]) / (np.abs(b[both]) + 1e-12))) if both.any() else 0.0
189
+ worst = max(worst, diff)
190
+ if (not same_nan.all()) or diff > rtol:
191
+ bad.append(str(t.date()))
192
+ return {"pass": not bad, "cuts": [str(self.dates[i].date()) for i in cuts], "mismatched_cuts": bad, "max_abs_diff": worst}
193
+
194
+ def assert_no_lookahead(self, signal: pd.DataFrame, h: int = 5,
195
+ n_null: int = 8, seed: int = 0) -> dict:
196
+ """A weak heuristic for look-ahead; prefer `assert_causal`, which is exact. If the result improves when the signal is pushed one day **later**, it is looking at the future.
197
+
198
+ **Known blind spots, measured on real data (see docs/market_validation.md):** a signal that is the future return itself is **not** flagged (delaying it by a day
199
+ makes it worse, not better), and a legitimate signal whose own edge is negative **is** flagged (delaying shrinks the loss, which reads as an improvement).
200
+ Treat a pass as "no evidence", never as proof, and a fail on a signal with a negative edge as possibly false.
201
+
202
+ A normal signal gets worse when delayed (the information decays).
203
+ If it improves after the delay, the signal already contains future information.
204
+
205
+ Warning: judging by a bare `lagged > base` makes a **random factor wrong 50% of the time**:
206
+ both are near zero so the sign flips by chance. So the noise scale (sd) of the spread is measured with random shuffles
207
+ and twice that sd is used as the tolerance.
208
+ """
209
+ fwd = self.forward(h)
210
+ base = _spread(signal, fwd, self.eligible)
211
+ lagged = _spread(signal.shift(1), fwd, self.eligible)
212
+ rng = np.random.default_rng(seed)
213
+ nulls = []
214
+ for _ in range(n_null):
215
+ sh = signal.values.copy()
216
+ ev = self.eligible.values
217
+ for i in range(sh.shape[0]):
218
+ j = np.where(ev[i])[0]
219
+ if len(j) > 1:
220
+ sh[i, j] = sh[i, rng.permutation(j)]
221
+ v = _spread(pd.DataFrame(sh, index=signal.index, columns=signal.columns),
222
+ fwd, self.eligible)
223
+ if np.isfinite(v):
224
+ nulls.append(v)
225
+ tol = 2 * float(np.std(nulls)) if len(nulls) >= 3 else 0.0
226
+ ok = not (np.isfinite(base) and np.isfinite(lagged) and lagged > base + tol)
227
+ return {"base_bp": base * 1e4, "lagged_bp": lagged * 1e4,
228
+ "tol_bp": tol * 1e4, "pass": bool(ok)}
229
+
230
+ def audit(self) -> dict:
231
+ """Data integrity audit: run it once before any test.
232
+
233
+ It catches common data defects:
234
+ - silent truncation (data vanishes wholesale after some date)
235
+ - adjusted-price mismatch (more extreme gaps than daily moves: the adj_open defect type)
236
+ - universe jumps (the population triples over time, for example)
237
+ """
238
+ out = {}
239
+ n = self.eligible.sum(axis=1)
240
+ out["dates"] = len(self.dates)
241
+ out["tickers"] = int(self.eligible.any(axis=0).sum())
242
+ out["eligible_mean"] = float(n.mean())
243
+ yr = n.groupby(self.dates.year).mean()
244
+ out["eligible_by_year"] = {int(k): float(v) for k, v in yr.items()}
245
+ out["universe_growth"] = float(yr.iloc[-1] / yr.iloc[0]) if len(yr) > 1 and yr.iloc[0] else np.nan
246
+
247
+ r = self.ret1().where(self.eligible)
248
+ out["ret_extreme_pct"] = float((r.abs() > 0.30).mean().mean() * 100)
249
+ gap = None
250
+ if self.open is not None:
251
+ gap = (self.open / self.close.shift(1) - 1.0).where(self.eligible)
252
+ out["gap_extreme_pct"] = float((gap.abs() > 0.30).mean().mean() * 100)
253
+ ratio = out["gap_extreme_pct"] / max(out["ret_extreme_pct"], 1e-9)
254
+ out["gap_vs_daily_ratio"] = float(ratio)
255
+ out["adj_price_consistent"] = bool(ratio < 3.0)
256
+
257
+ # silent truncation: security coverage drops sharply after some date
258
+ cov = self.close.notna().sum(axis=1)
259
+ if len(cov) > 60:
260
+ tail = cov.iloc[-20:].mean(); body = cov.iloc[:-20].median()
261
+ out["tail_coverage_ratio"] = float(tail / body) if body else np.nan
262
+ out["truncation_suspected"] = bool(tail < body * 0.5)
263
+
264
+ # suspected survivorship bias: in a real market a steady share of securities disappears every year. If almost no security's price stops before the panel's end,
265
+ # the panel holds only 'securities alive today'. Measured: of 41 well-known delisted or acquired stocks looked up in yfinance,
266
+ # none came back with a correct history (docs/survivorship.md).
267
+ last = self.close.apply(lambda c: c.last_valid_index())
268
+ ended = last.dropna() < (self.dates[-1] - pd.Timedelta(days=30))
269
+ span_years = (self.dates[-1] - self.dates[0]).days / 365.25
270
+ out["ended_before_end_share"] = float(ended.mean()) if len(ended) else np.nan
271
+ out["survivorship_suspected"] = bool(len(ended) >= 30 and span_years >= 3 and ended.mean() < 0.01)
272
+ return out
273
+
274
+
275
+ def _spread(signal: pd.DataFrame, fwd: pd.DataFrame, eligible: pd.DataFrame,
276
+ q: float = 0.10) -> float:
277
+ """Mean top-q minus bottom-q spread (internal helper for the look-ahead check)."""
278
+ rk = signal.where(eligible).rank(axis=1, pct=True, na_option="keep")
279
+ hi, lo = ((rk >= 1 - q) & eligible).values, ((rk <= q) & eligible).values
280
+ c = fwd.values.astype(np.float64)
281
+ out = []
282
+ for i in range(c.shape[0]):
283
+ a, b = hi[i] & np.isfinite(c[i]), lo[i] & np.isfinite(c[i])
284
+ if a.sum() >= 3 and b.sum() >= 3:
285
+ out.append(c[i][a].mean() - c[i][b].mean())
286
+ return float(np.mean(out)) if out else np.nan
287
+
288
+
289
+ def build_pit_eligible(close: pd.DataFrame, *, listed: pd.DataFrame | None = None,
290
+ min_adv: float = 0.0, volume: pd.DataFrame | None = None,
291
+ exclude: pd.DataFrame | None = None) -> pd.DataFrame:
292
+ """Helper that builds a point-in-time universe mask.
293
+
294
+ listed listing flag bool (falls back to whether close exists)
295
+ min_adv minimum average traded value (needs volume)
296
+ exclude mask of securities to exclude, such as administrative issues and trading halts (True = exclude)
297
+ """
298
+ ok = close.notna() & (close > 0)
299
+ if listed is not None:
300
+ ok &= listed.reindex_like(close).fillna(False).astype(bool)
301
+ if min_adv > 0 and volume is not None:
302
+ adv = (close * volume).rolling(20, min_periods=20).mean()
303
+ ok &= adv >= min_adv
304
+ if exclude is not None:
305
+ ok &= ~exclude.reindex_like(close).fillna(False).astype(bool)
306
+ return ok.fillna(False)
@@ -0,0 +1,12 @@
1
+ """Crypto perpetual-futures data and panel construction (Binance USDT-M).
2
+
3
+ The point of this subpackage is to remove, as far as public data allows, three biases that crypto backtests
4
+ usually carry: survivorship (delisted contracts are included), universe look-ahead (eligibility uses only data up
5
+ to the signal date) and ignored funding (the daily funding rate is charged to the position).
6
+ """
7
+ from .binance_archive import ArchiveStore, fetch_all
8
+ from .panel import build_panel
9
+ from .costs import liquidity_cost_bp, participation_report
10
+ from . import intraday
11
+
12
+ __all__ = ["ArchiveStore", "fetch_all", "build_panel", "liquidity_cost_bp", "participation_report", "intraday"]
@@ -0,0 +1,287 @@
1
+ """Binance public data archive (data.binance.vision) for USDT-margined perpetuals.
2
+
3
+ Why the archive and not the live API: the archive keeps **delisted** contracts (LUNA, FTT and hundreds of
4
+ others), so a universe built from it is not limited to today's survivors. The live API is used only to fill the
5
+ last partial month for contracts that still trade.
6
+
7
+ Daily bars are UTC days. Funding is settled every 8 hours (older contracts) or per `funding_interval_hours`;
8
+ `daily_funding` sums every settlement that falls inside the UTC day.
9
+
10
+ Everything is cached under `cache_dir` (default ~/.cache/quantbt/binance_um) so a download happens once.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import concurrent.futures as cf
15
+ import io
16
+ import json
17
+ import re
18
+ import time
19
+ import urllib.error
20
+ import urllib.parse
21
+ import urllib.request
22
+ import zipfile
23
+ from pathlib import Path
24
+
25
+ import pandas as pd
26
+
27
+ LIST_URL = "https://s3-ap-northeast-1.amazonaws.com/data.binance.vision"
28
+ DL_URL = "https://data.binance.vision"
29
+ API = "https://fapi.binance.com"
30
+ KL_COLS = ["open_time", "open", "high", "low", "close", "volume", "close_time", "quote_volume", "trades",
31
+ "taker_buy_base", "taker_buy_quote", "ignore"]
32
+ UA = {"User-Agent": "pitbacktest-research/0.2 (public data only)"}
33
+
34
+
35
+ def _get(url: str, retries: int = 4, timeout: float = 30) -> bytes | None:
36
+ """GET with backoff. Returns None on 404 (a missing archive file is normal)."""
37
+ url = urllib.parse.quote(url, safe=":/?&=%") # non-ASCII symbols (e.g. Chinese tickers) need encoding
38
+ delay = 1.0
39
+ for i in range(retries + 1):
40
+ try:
41
+ with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=timeout) as r:
42
+ return r.read()
43
+ except urllib.error.HTTPError as e:
44
+ if e.code == 404:
45
+ return None
46
+ if e.code in (418, 429, 500, 502, 503, 504) and i < retries:
47
+ time.sleep(delay * (3 if e.code in (418, 429) else 1))
48
+ delay *= 2
49
+ continue
50
+ raise
51
+ except (urllib.error.URLError, TimeoutError, ConnectionError):
52
+ if i < retries:
53
+ time.sleep(delay)
54
+ delay *= 2
55
+ continue
56
+ raise
57
+ return None
58
+
59
+
60
+ def _list(prefix: str) -> list[str]:
61
+ """All keys under a prefix (S3 listing, paginated)."""
62
+ keys, marker = [], ""
63
+ while True:
64
+ q = {"prefix": prefix, "max-keys": "1000"}
65
+ if marker:
66
+ q["marker"] = marker
67
+ x = (_get(f"{LIST_URL}?{urllib.parse.urlencode(q)}") or b"").decode("utf-8")
68
+ got = re.findall(r"<Key>([^<]+)</Key>", x)
69
+ keys += got
70
+ if "<IsTruncated>true</IsTruncated>" in x and got:
71
+ marker = got[-1]
72
+ else:
73
+ return keys
74
+
75
+
76
+ def _list_dirs(prefix: str) -> list[str]:
77
+ out, marker = [], ""
78
+ while True:
79
+ q = {"prefix": prefix, "delimiter": "/", "max-keys": "1000"}
80
+ if marker:
81
+ q["marker"] = marker
82
+ x = (_get(f"{LIST_URL}?{urllib.parse.urlencode(q)}") or b"").decode("utf-8")
83
+ got = re.findall(r"<Prefix>([^<]+)</Prefix>", x)
84
+ got = [g for g in got if g != prefix]
85
+ out += got
86
+ nm = re.search(r"<NextMarker>([^<]+)</NextMarker>", x)
87
+ if "<IsTruncated>true</IsTruncated>" in x and nm:
88
+ marker = nm.group(1)
89
+ else:
90
+ return out
91
+
92
+
93
+ def _read_klines_zip(blob: bytes) -> pd.DataFrame:
94
+ with zipfile.ZipFile(io.BytesIO(blob)) as z:
95
+ raw = pd.read_csv(z.open(z.namelist()[0]), header=None, names=KL_COLS, dtype=str)
96
+ t = pd.to_numeric(raw["open_time"], errors="coerce") # some files carry a header row
97
+ raw = raw[t.notna()].copy()
98
+ raw["open_time"] = t[t.notna()].astype("int64")
99
+ return raw
100
+
101
+
102
+ def _frame_from_klines(raw: pd.DataFrame) -> pd.DataFrame:
103
+ if raw.empty:
104
+ return pd.DataFrame()
105
+ ts = raw["open_time"].astype("int64")
106
+ unit = "us" if ts.iloc[0] > 10**14 else "ms" # newer archives use microseconds
107
+ idx = pd.to_datetime(ts, unit=unit, utc=True).dt.tz_localize(None).dt.normalize()
108
+ df = raw[["open", "high", "low", "close", "volume", "quote_volume", "trades"]].apply(pd.to_numeric, errors="coerce")
109
+ df.index = idx
110
+ df = df[~df.index.duplicated(keep="last")].sort_index()
111
+ return df
112
+
113
+
114
+ class ArchiveStore:
115
+ """Local cache plus downloader for one market (USDT-M perpetuals)."""
116
+
117
+ def __init__(self, cache_dir: str | Path | None = None):
118
+ self.dir = Path(cache_dir or Path.home() / ".cache" / "quantbt" / "binance_um").expanduser()
119
+ (self.dir / "daily").mkdir(parents=True, exist_ok=True)
120
+ (self.dir / "funding").mkdir(parents=True, exist_ok=True)
121
+
122
+ # ------------------------------------------------------------ symbols
123
+ def symbols(self, refresh: bool = False, quote: str = "USDT") -> list[str]:
124
+ """Every USDT perpetual that ever existed in the archive (delisted ones included).
125
+ Delivery contracts such as BTCUSDT_250627 are excluded."""
126
+ f = self.dir / "symbols.json"
127
+ if f.exists() and not refresh:
128
+ return json.loads(f.read_text())
129
+ pre = "data/futures/um/monthly/klines/"
130
+ names = [d[len(pre):].strip("/") for d in _list_dirs(pre)]
131
+ syms = sorted(n for n in names if n.endswith(quote) and "_" not in n)
132
+ f.write_text(json.dumps(syms))
133
+ return syms
134
+
135
+ def trading_now(self) -> set[str]:
136
+ """Symbols the live exchange reports as TRADING (used to know which ones need the live top-up)."""
137
+ raw = _get(f"{API}/fapi/v1/exchangeInfo")
138
+ info = json.loads(raw)
139
+ return {s["symbol"] for s in info["symbols"] if s.get("contractType") == "PERPETUAL" and s.get("status") == "TRADING"}
140
+
141
+ def symbol_rules(self, refresh: bool = False) -> pd.DataFrame:
142
+ """Trading rules per contract from the live exchange-info endpoint (no key): quantity step, minimum quantity, minimum order value,
143
+ price tick, status and the onboard and delivery dates, indexed by symbol. **Today's values only**: the endpoint has no history, so a rule
144
+ that changed (BTCUSDT's minimum order value is not what it was in 2020) is applied to the past as it is now, and a contract that is no longer
145
+ listed (delisted ones) is absent. The frame carries the day it was fetched in `.attrs["fetched"]`. Cached as `symbol_rules.json`."""
146
+ f = self.dir / "symbol_rules.json"
147
+ if f.exists() and not refresh:
148
+ blob = json.loads(f.read_text())
149
+ else:
150
+ raw = _get(f"{API}/fapi/v1/exchangeInfo")
151
+ if raw is None:
152
+ raise RuntimeError("exchange info is not available")
153
+ rows = {}
154
+ for s in json.loads(raw)["symbols"]:
155
+ flt = {x["filterType"]: x for x in s.get("filters", [])}
156
+ lot, mn, pf = flt.get("LOT_SIZE"), flt.get("MIN_NOTIONAL"), flt.get("PRICE_FILTER")
157
+ if lot is None or mn is None:
158
+ continue
159
+ rows[s["symbol"]] = {"step_size": float(lot["stepSize"]), "min_qty": float(lot["minQty"]), "min_notional": float(mn["notional"]),
160
+ "tick_size": float(pf["tickSize"]) if pf else float("nan"), "status": s.get("status"),
161
+ "contract_type": s.get("contractType"), "onboard_ms": s.get("onboardDate"), "delivery_ms": s.get("deliveryDate")}
162
+ blob = {"fetched": pd.Timestamp.now("UTC").strftime("%Y-%m-%d"), "rules": rows}
163
+ f.write_text(json.dumps(blob))
164
+ df = pd.DataFrame.from_dict(blob["rules"], orient="index")
165
+ df.attrs["fetched"] = blob["fetched"]
166
+ return df
167
+
168
+ # ------------------------------------------------------------ one symbol
169
+ def _months(self, sym: str, kind: str) -> list[str]:
170
+ keys = _list(f"data/futures/um/monthly/{kind}/{sym}/" + ("1d/" if kind == "klines" else ""))
171
+ return sorted(k for k in keys if k.endswith(".zip"))
172
+
173
+ def fetch_daily(self, sym: str, live: bool) -> pd.DataFrame | None:
174
+ """Daily bars (open, high, low, close, volume, quote_volume, trades) indexed by UTC date: the monthly archive files plus, when `live`
175
+ is True, the live API for the days after the last archived one. The unfinished current day is dropped. Cached as a pickle."""
176
+ f = self.dir / "daily" / f"{sym}.pkl"
177
+ if f.exists():
178
+ return pd.read_pickle(f)
179
+ parts = []
180
+ for k in self._months(sym, "klines"):
181
+ blob = _get(f"{DL_URL}/{k}")
182
+ if blob:
183
+ parts.append(_read_klines_zip(blob))
184
+ raw = pd.concat(parts) if parts else pd.DataFrame(columns=KL_COLS)
185
+ df = _frame_from_klines(raw)
186
+ if live: # top-up from the live API
187
+ start = (df.index[-1] + pd.Timedelta(days=1)) if len(df) else pd.Timestamp("2019-09-01")
188
+ ms = int(start.tz_localize("UTC").timestamp() * 1000)
189
+ blob = _get(f"{API}/fapi/v1/klines?symbol={sym}&interval=1d&startTime={ms}&limit=1500")
190
+ if blob:
191
+ rows = json.loads(blob)
192
+ if rows:
193
+ live_df = _frame_from_klines(pd.DataFrame(rows, columns=KL_COLS).assign(
194
+ open_time=lambda d: d["open_time"].astype("int64")))
195
+ df = pd.concat([df, live_df]).pipe(lambda d: d[~d.index.duplicated(keep="last")]).sort_index()
196
+ if len(df):
197
+ today = pd.Timestamp.utcnow().tz_localize(None).normalize()
198
+ df = df[df.index < today] # never keep an unfinished bar
199
+ df.to_pickle(f)
200
+ return df
201
+
202
+ def fetch_funding(self, sym: str, live: bool) -> pd.Series | None:
203
+ """Funding rate summed per UTC day (every settlement inside the day), from the monthly fundingRate files plus, when `live` is True,
204
+ the live API. The current day is dropped. Cached as a pickle. For settlement instants see `crypto.intraday.fetch_funding_events`."""
205
+ f = self.dir / "funding" / f"{sym}.pkl"
206
+ if f.exists():
207
+ return pd.read_pickle(f)
208
+ frames = []
209
+ for k in self._months(sym, "fundingRate"):
210
+ blob = _get(f"{DL_URL}/{k}")
211
+ if not blob:
212
+ continue
213
+ with zipfile.ZipFile(io.BytesIO(blob)) as z:
214
+ d = pd.read_csv(z.open(z.namelist()[0]), dtype=str)
215
+ if d.shape[1] >= 3:
216
+ d.columns = ["calc_time", "interval", "rate"][: d.shape[1]] + list(d.columns[3:])
217
+ frames.append(d[["calc_time", "rate"]])
218
+ out = pd.concat(frames) if frames else pd.DataFrame(columns=["calc_time", "rate"])
219
+ out["calc_time"] = pd.to_numeric(out["calc_time"], errors="coerce")
220
+ out["rate"] = pd.to_numeric(out["rate"], errors="coerce")
221
+ out = out.dropna()
222
+ if live:
223
+ last = out["calc_time"].max() if len(out) else 1567296000000
224
+ blob = _get(f"{API}/fapi/v1/fundingRate?symbol={sym}&startTime={int(last) + 1}&limit=1000")
225
+ if blob:
226
+ rows = json.loads(blob)
227
+ if rows:
228
+ add = pd.DataFrame({"calc_time": [r["fundingTime"] for r in rows], "rate": [float(r["fundingRate"]) for r in rows]})
229
+ out = pd.concat([out, add])
230
+ if out.empty:
231
+ s = pd.Series(dtype=float)
232
+ else:
233
+ ts = out["calc_time"].astype("int64")
234
+ unit = "us" if ts.iloc[0] > 10**14 else "ms"
235
+ day = pd.to_datetime(ts, unit=unit, utc=True).dt.tz_localize(None).dt.normalize()
236
+ s = out.set_index(day)["rate"].groupby(level=0).sum().sort_index() # sum of settlements in the UTC day
237
+ today = pd.Timestamp.utcnow().tz_localize(None).normalize()
238
+ s = s[s.index < today]
239
+ s.to_pickle(f)
240
+ return s
241
+
242
+
243
+ def fetch_all(store: ArchiveStore | None = None, symbols: list[str] | None = None, workers: int = 12,
244
+ progress: bool = True) -> dict:
245
+ """Download (or load from cache) daily bars and funding for every symbol. Returns counts."""
246
+ store = store or ArchiveStore()
247
+ syms = symbols or store.symbols()
248
+ live = store.trading_now()
249
+ done = {"ok": 0, "empty": 0, "error": 0}
250
+ errs: list[str] = []
251
+
252
+ def one(s: str):
253
+ d = store.fetch_daily(s, s in live)
254
+ f = store.fetch_funding(s, s in live)
255
+ return s, (0 if d is None else len(d)), (0 if f is None else len(f))
256
+
257
+ with cf.ThreadPoolExecutor(workers) as ex:
258
+ futs = {ex.submit(one, s): s for s in syms}
259
+ for n, fu in enumerate(cf.as_completed(futs), 1):
260
+ try:
261
+ _, nd, nf = fu.result()
262
+ done["ok" if nd else "empty"] += 1
263
+ except Exception as e: # one bad symbol must not stop the run
264
+ done["error"] += 1
265
+ errs.append(f"{futs[fu]}: {type(e).__name__}: {e}")
266
+ if progress and n % 25 == 0:
267
+ print(f" {n}/{len(syms)} {done}", flush=True)
268
+ done["errors"] = errs[:20]
269
+ return done
270
+
271
+
272
+ def execution_rules(rules: pd.DataFrame, tickers, *, missing: str = "raise") -> tuple[pd.Series, pd.Series]:
273
+ """(lot, min_trade_value) per ticker for `backtest_weights(lot=..., min_trade_value=...)`, from `ArchiveStore.symbol_rules()`.
274
+
275
+ lot is the quantity step (at least the minimum quantity), in units of the contract; min_trade_value the minimum order value in the quote
276
+ currency. `rules` holds today's values only and has no delisted contracts. For a ticker it does not know, `missing="raise"` (default) raises,
277
+ and `missing="nan"` returns NaN for it so that you can leave it out of a small-account run. No substitute rule is invented: a step is in units of
278
+ the contract, which are worth anything from a thousandth of a dollar to tens of thousands, so there is no safe value to fill in."""
279
+ if missing not in ("raise", "nan"):
280
+ raise ValueError(f"missing must be 'raise' or 'nan', not {missing!r}")
281
+ tickers = pd.Index(tickers)
282
+ unknown = tickers[~tickers.isin(rules.index)]
283
+ if len(unknown) and missing == "raise":
284
+ raise ValueError(f"no trading rules for {len(unknown)} tickers (first: {list(unknown[:3])}); they are probably delisted. "
285
+ f"Pass missing='nan' to get NaN for them and leave them out")
286
+ r = rules.reindex(tickers)
287
+ return r[["step_size", "min_qty"]].max(axis=1).rename("lot"), r["min_notional"].rename("min_trade_value")