pitbacktest 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pitbacktest/__init__.py +27 -0
- pitbacktest/adapters/__init__.py +0 -0
- pitbacktest/adapters/krx.py +249 -0
- pitbacktest/adapters/long_format.py +87 -0
- pitbacktest/adapters/tiingo.py +236 -0
- pitbacktest/adapters/yfinance.py +243 -0
- pitbacktest/analytics.py +234 -0
- pitbacktest/core/__init__.py +0 -0
- pitbacktest/core/controls.py +126 -0
- pitbacktest/core/costs.py +144 -0
- pitbacktest/core/estimators.py +216 -0
- pitbacktest/core/gates.py +175 -0
- pitbacktest/core/panel.py +306 -0
- pitbacktest/crypto/__init__.py +12 -0
- pitbacktest/crypto/binance_archive.py +287 -0
- pitbacktest/crypto/costs.py +80 -0
- pitbacktest/crypto/intraday.py +326 -0
- pitbacktest/crypto/panel.py +120 -0
- pitbacktest/equity/__init__.py +10 -0
- pitbacktest/equity/master.py +60 -0
- pitbacktest/equity/scenarios.py +104 -0
- pitbacktest/event.py +206 -0
- pitbacktest/execution.py +310 -0
- pitbacktest/ledger.py +128 -0
- pitbacktest/portfolio.py +373 -0
- pitbacktest/screen.py +150 -0
- pitbacktest/shorting.py +46 -0
- pitbacktest/validation.py +118 -0
- pitbacktest/weights.py +271 -0
- pitbacktest-0.2.0.dist-info/METADATA +395 -0
- pitbacktest-0.2.0.dist-info/RECORD +34 -0
- pitbacktest-0.2.0.dist-info/WHEEL +5 -0
- pitbacktest-0.2.0.dist-info/licenses/LICENSE +21 -0
- pitbacktest-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Survivorship scenarios: put delistings back into a survivors-only panel and see how the result moves.
|
|
2
|
+
|
|
3
|
+
This is a **what-if**, not a correction. The actual delisted companies are unknown here, so we draw them at random
|
|
4
|
+
(optionally tilted toward small, illiquid or volatile names, which is where delistings concentrate in the literature) and
|
|
5
|
+
give them an explicit delisting return. Run it over a grid of annual rates and delisting returns and report the range, not a
|
|
6
|
+
single "corrected" number.
|
|
7
|
+
|
|
8
|
+
Order-of-magnitude guide from the delisting literature (check it for your period and market): a few percent of listed
|
|
9
|
+
names disappear each year, roughly half of them for performance reasons; the average return on a performance-related
|
|
10
|
+
delisting is about -30% on NYSE/AMEX (Shumway 1997) and about -55% on Nasdaq (Shumway and Warther 1999).
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from dataclasses import replace
|
|
15
|
+
from typing import Callable
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
import pandas as pd
|
|
19
|
+
|
|
20
|
+
from ..core.panel import Panel
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _weights(panel: Panel, year_start: pd.Timestamp, names: pd.Index, hazard: str) -> np.ndarray:
|
|
24
|
+
if hazard == "uniform":
|
|
25
|
+
return np.ones(len(names))
|
|
26
|
+
before = panel.dates < year_start
|
|
27
|
+
if not before.any():
|
|
28
|
+
return np.ones(len(names)) # first year: nothing earlier to tilt on, so uniform
|
|
29
|
+
if hazard == "volatile":
|
|
30
|
+
v = panel.close.pct_change(fill_method=None).loc[before].tail(60).std().reindex(names)
|
|
31
|
+
elif hazard == "illiquid":
|
|
32
|
+
adv = panel.adv(30)
|
|
33
|
+
v = -(adv.loc[before].tail(1).iloc[0].reindex(names)) if adv is not None else pd.Series(0.0, index=names)
|
|
34
|
+
else:
|
|
35
|
+
raise ValueError(hazard)
|
|
36
|
+
r = v.rank(pct=True).fillna(0.5).to_numpy()
|
|
37
|
+
return 0.2 + r ** 2 # tilted, but every name keeps a chance
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def inject_delistings(panel: Panel, *, annual_rate: float = 0.03, hazard: str = "uniform", seed: int = 0,
|
|
41
|
+
min_age_days: int = 60) -> Panel:
|
|
42
|
+
"""Return a copy of `panel` in which a fraction `annual_rate` of the names alive at the start of each calendar year stop
|
|
43
|
+
trading at a random date later that year. The last real bar is flagged in `delist_after` and prices, volume and
|
|
44
|
+
eligibility are removed afterwards."""
|
|
45
|
+
if hazard not in ("uniform", "volatile", "illiquid"):
|
|
46
|
+
raise ValueError(f"hazard must be 'uniform', 'volatile' or 'illiquid', not {hazard!r}")
|
|
47
|
+
if hazard == "illiquid" and panel.volume is None:
|
|
48
|
+
raise ValueError("hazard='illiquid' needs panel.volume")
|
|
49
|
+
if not (0 <= annual_rate <= 1):
|
|
50
|
+
raise ValueError(f"annual_rate must be between 0 and 1, got {annual_rate!r}")
|
|
51
|
+
rng = np.random.default_rng(seed)
|
|
52
|
+
close = panel.close.copy()
|
|
53
|
+
elig = panel.eligible.copy()
|
|
54
|
+
vol = None if panel.volume is None else panel.volume.copy()
|
|
55
|
+
da = (pd.DataFrame(False, index=panel.dates, columns=panel.tickers) if panel.delist_after is None
|
|
56
|
+
else panel.delist_after.copy())
|
|
57
|
+
for y in sorted(set(panel.dates.year)):
|
|
58
|
+
idx = panel.dates[panel.dates.year == y]
|
|
59
|
+
if len(idx) < 100:
|
|
60
|
+
continue # skip partial years
|
|
61
|
+
alive = close.loc[idx[0]].dropna().index
|
|
62
|
+
k = int(round(annual_rate * len(alive)))
|
|
63
|
+
if k == 0:
|
|
64
|
+
continue
|
|
65
|
+
w = _weights(panel, idx[0], alive, hazard)
|
|
66
|
+
pick = rng.choice(len(alive), size=min(k, len(alive)), replace=False, p=w / w.sum())
|
|
67
|
+
for j in pick:
|
|
68
|
+
name = alive[j]
|
|
69
|
+
when = idx[int(rng.integers(min_age_days // 3, len(idx) - 1))]
|
|
70
|
+
close.loc[when + pd.Timedelta(days=1):, name] = np.nan
|
|
71
|
+
elig.loc[when + pd.Timedelta(days=1):, name] = False
|
|
72
|
+
if vol is not None:
|
|
73
|
+
vol.loc[when + pd.Timedelta(days=1):, name] = np.nan
|
|
74
|
+
da.loc[when, name] = True
|
|
75
|
+
return replace(panel, close=close, eligible=elig, volume=vol, delist_after=da)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def survivorship_scenarios(panel: Panel, evaluate: Callable[[Panel, float], float], *,
|
|
79
|
+
rates=(0.01, 0.03, 0.05), delist_returns=(-0.30, -0.55), hazards=("uniform", "volatile"),
|
|
80
|
+
n: int = 10, seed: int = 0) -> pd.DataFrame:
|
|
81
|
+
"""Run `evaluate(panel, delist_return)` (it should return one number, for example the net Sharpe) on the original panel
|
|
82
|
+
and on `n` randomly delisting copies for every combination. Returns the baseline and the spread of outcomes."""
|
|
83
|
+
base = float(evaluate(panel, 0.0))
|
|
84
|
+
rows = []
|
|
85
|
+
for hz in hazards:
|
|
86
|
+
for r in rates:
|
|
87
|
+
for d in delist_returns:
|
|
88
|
+
vals = [float(evaluate(inject_delistings(panel, annual_rate=r, hazard=hz, seed=seed + i), d)) for i in range(n)]
|
|
89
|
+
rows.append({"hazard": hz, "annual_rate": r, "delist_return": d, "baseline": base,
|
|
90
|
+
"mean": float(np.mean(vals)), "p5": float(np.percentile(vals, 5)),
|
|
91
|
+
"p95": float(np.percentile(vals, 95)), "mean_change": float(np.mean(vals) - base)})
|
|
92
|
+
return pd.DataFrame(rows)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def survivors_only(panel: Panel, end_gap_days: int = 5) -> Panel:
|
|
96
|
+
"""The usual shortcut, made explicit: keep only the securities that still have a price on the last day.
|
|
97
|
+
Compare a result on `panel` with the same result on `survivors_only(panel)` to measure survivorship bias directly."""
|
|
98
|
+
last = panel.close.apply(lambda c: c.last_valid_index())
|
|
99
|
+
keep = [c for c in panel.close.columns if last[c] is not None and last[c] >= panel.dates[-1] - pd.Timedelta(days=end_gap_days)]
|
|
100
|
+
sub = lambda d: None if d is None else d[keep]
|
|
101
|
+
return replace(panel, close=panel.close[keep], eligible=panel.eligible[keep], open=sub(panel.open), high=sub(panel.high),
|
|
102
|
+
low=sub(panel.low), volume=sub(panel.volume), mkt_cap=sub(panel.mkt_cap),
|
|
103
|
+
chars={k: v[keep] for k, v in panel.chars.items()}, funding=sub(panel.funding),
|
|
104
|
+
delist_after=sub(panel.delist_after))
|
pitbacktest/event.py
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""C. Event-signal test: does it hold up as individual alerts?
|
|
2
|
+
|
|
3
|
+
Portfolio metrics (mean excess return) assume the mean is realised. A user receives alerts one by one, though, so the **win rate,
|
|
4
|
+
median and payoff ratio** decide what it feels like.
|
|
5
|
+
|
|
6
|
+
What is always decomposed
|
|
7
|
+
win rate = base rate (how the sample was picked) + lift (what the signal adds)
|
|
8
|
+
- A longer holding period raises the win rate and the base rate together. Judge by the lift only.
|
|
9
|
+
- A win rate below the base rate (lift < 0) means the signal does harm.
|
|
10
|
+
mean vs median
|
|
11
|
+
- If the signs differ, a few big winners cover many losses: it works as a portfolio and not as alerts.
|
|
12
|
+
- The typical case is a positive mean with a negative median.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import warnings
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
|
|
23
|
+
from .core.controls import build_controls, xs_norm
|
|
24
|
+
from .core.estimators import fama_macbeth, newey_west_t, paired_diff
|
|
25
|
+
from .core.gates import GateConfig, fire_structure, run_gates
|
|
26
|
+
from .core.panel import Panel
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class EventResult:
|
|
31
|
+
spec: dict
|
|
32
|
+
per_horizon: dict
|
|
33
|
+
gates: dict
|
|
34
|
+
structure: dict
|
|
35
|
+
neutralized: dict | None
|
|
36
|
+
lookahead: dict
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _trade_stats(fire: np.ndarray, cum: np.ndarray, eligible: np.ndarray,
|
|
40
|
+
cost: float) -> dict | None:
|
|
41
|
+
"""Per-trade profit distribution plus the base win rate."""
|
|
42
|
+
tr, bh, bn = [], 0.0, 0
|
|
43
|
+
for i in range(cum.shape[0]):
|
|
44
|
+
s = fire[i] & np.isfinite(cum[i])
|
|
45
|
+
k = int(s.sum())
|
|
46
|
+
if k < 1:
|
|
47
|
+
continue
|
|
48
|
+
tr.append(cum[i][s] - cost)
|
|
49
|
+
pool = eligible[i] & np.isfinite(cum[i])
|
|
50
|
+
if pool.sum() >= 20:
|
|
51
|
+
bh += float((cum[i][pool] - cost > 0).mean()) * k
|
|
52
|
+
bn += k
|
|
53
|
+
if not tr:
|
|
54
|
+
return None
|
|
55
|
+
t = np.concatenate(tr)
|
|
56
|
+
if len(t) < 100:
|
|
57
|
+
return None
|
|
58
|
+
win = float((t > 0).mean())
|
|
59
|
+
base = bh / bn if bn else np.nan
|
|
60
|
+
w, l = t[t > 0], t[t < 0]
|
|
61
|
+
return {
|
|
62
|
+
"n_trades": int(len(t)),
|
|
63
|
+
"win_rate": win * 100,
|
|
64
|
+
"base_rate": base * 100 if np.isfinite(base) else np.nan,
|
|
65
|
+
"lift_pp": (win - base) * 100 if np.isfinite(base) else np.nan,
|
|
66
|
+
"mean_bp": float(t.mean() * 1e4),
|
|
67
|
+
"median_bp": float(np.median(t) * 1e4),
|
|
68
|
+
"avg_win_bp": float(w.mean() * 1e4) if len(w) else np.nan,
|
|
69
|
+
"avg_loss_bp": float(l.mean() * 1e4) if len(l) else np.nan,
|
|
70
|
+
"payoff": float(w.mean() / abs(l.mean())) if len(l) else np.nan,
|
|
71
|
+
"p25_bp": float(np.percentile(t, 25) * 1e4),
|
|
72
|
+
"p75_bp": float(np.percentile(t, 75) * 1e4),
|
|
73
|
+
"skew_warning": bool(t.mean() > 0 > np.median(t)),
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _daily_excess(fire: np.ndarray, cum: np.ndarray, eligible: np.ndarray,
|
|
78
|
+
*, min_fire: int = 1, min_pool: int | None = None) -> np.ndarray:
|
|
79
|
+
"""Daily excess return of the firing group (gate input).
|
|
80
|
+
|
|
81
|
+
A fixed min_pool throws away every date in a small universe (for example a 20-name test).
|
|
82
|
+
The default **adapts** to half the median universe size (at most 30).
|
|
83
|
+
"""
|
|
84
|
+
if min_pool is None:
|
|
85
|
+
med = float(np.median(eligible.sum(axis=1)))
|
|
86
|
+
min_pool = int(max(5, min(30, med * 0.5)))
|
|
87
|
+
out = []
|
|
88
|
+
for i in range(cum.shape[0]):
|
|
89
|
+
s = fire[i] & np.isfinite(cum[i])
|
|
90
|
+
pool = eligible[i] & np.isfinite(cum[i])
|
|
91
|
+
if s.sum() >= min_fire and pool.sum() >= min_pool:
|
|
92
|
+
out.append(cum[i][s].mean() - cum[i][pool].mean())
|
|
93
|
+
return np.array(out)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def backtest_event(panel: Panel, signal: pd.DataFrame, *,
|
|
97
|
+
horizons: tuple[int, ...] = (1, 2, 5, 10, 21),
|
|
98
|
+
cost_bp: float = 20.0,
|
|
99
|
+
neutralize_check: bool = True,
|
|
100
|
+
gate_config: GateConfig | None = None,
|
|
101
|
+
null_threshold: float | None = None,
|
|
102
|
+
delist_return: float | None = None) -> EventResult:
|
|
103
|
+
"""Event-signal test.
|
|
104
|
+
|
|
105
|
+
signal bool or 0/1 matrix (date x ticker). True = fires that day.
|
|
106
|
+
cost_bp round-trip trading cost. If per-security measured costs exist, use the spread panel on the portfolio side.
|
|
107
|
+
delist_return return assumed on the day after a security's last bar when it is flagged in panel.delist_after (None: carried at its last price).
|
|
108
|
+
|
|
109
|
+
With the neutralisation test on (the default) the contribution after controls is re-measured with a **dummy regression**.
|
|
110
|
+
An event signal is not a continuous factor, so it is read through a dummy coefficient and not through residuals.
|
|
111
|
+
"""
|
|
112
|
+
if not horizons or any(isinstance(h, bool) or not isinstance(h, (int, np.integer)) or h < 1 for h in horizons):
|
|
113
|
+
raise ValueError(f"horizons must be whole numbers of periods, at least 1, got {horizons!r}")
|
|
114
|
+
if not (cost_bp >= 0 and np.isfinite(cost_bp)):
|
|
115
|
+
raise ValueError(f"cost_bp must be finite and not negative, got {cost_bp!r}")
|
|
116
|
+
sig = signal.reindex(index=panel.dates, columns=panel.tickers).fillna(False)
|
|
117
|
+
fire = (sig.astype(bool) & panel.eligible).values
|
|
118
|
+
ev = panel.eligible.values
|
|
119
|
+
cost = cost_bp / 1e4
|
|
120
|
+
|
|
121
|
+
lookahead = panel.assert_no_lookahead(sig.astype(float), h=max(horizons))
|
|
122
|
+
cums = {h: panel.forward(h, delist_return).values.astype(np.float64) for h in horizons}
|
|
123
|
+
|
|
124
|
+
per_h = {}
|
|
125
|
+
for h in horizons:
|
|
126
|
+
st = _trade_stats(fire, cums[h], ev, cost)
|
|
127
|
+
if st is None:
|
|
128
|
+
continue
|
|
129
|
+
de = _daily_excess(fire, cums[h], ev)
|
|
130
|
+
mu, _, t, n = newey_west_t(de, lag=max(h, 21))
|
|
131
|
+
per_h[h] = {**st, "excess_bp": float(mu * 1e4), "t": float(t), "n_days": int(n),
|
|
132
|
+
"net_bp": float(mu * 1e4 - cost_bp)}
|
|
133
|
+
|
|
134
|
+
if not per_h:
|
|
135
|
+
raise ValueError(
|
|
136
|
+
"No horizon can be tested. Either the signal fires too rarely (fewer than 100 events) or "
|
|
137
|
+
"too few securities are eligible. Check signal, eligible and horizons.")
|
|
138
|
+
# representative horizon = where the lift is largest
|
|
139
|
+
best_h = max(per_h, key=lambda h: per_h[h].get("lift_pp", -99))
|
|
140
|
+
de = _daily_excess(fire, cums[best_h], ev)
|
|
141
|
+
if len(de) < 100:
|
|
142
|
+
warnings.warn(
|
|
143
|
+
f"Only {len(de)} days of daily excess return, so the gates cannot be trusted. "
|
|
144
|
+
f"_daily_excess counts only days with at least one firing name and a pool of at least max(5, min(30, half the "
|
|
145
|
+
f"median universe)) names, so a short sample or a rarely firing signal leaves few days.", stacklevel=2)
|
|
146
|
+
|
|
147
|
+
from .core.estimators import decile_profile
|
|
148
|
+
dec = decile_profile(sig.astype(float), panel.forward(best_h, delist_return), panel.eligible,
|
|
149
|
+
lag=max(best_h, 21))
|
|
150
|
+
struct = fire_structure(fire, panel.dates, panel.tickers)
|
|
151
|
+
cfg = gate_config or GateConfig(null_threshold=null_threshold)
|
|
152
|
+
gates = run_gates(de, panel.dates, t_stat=per_h[best_h]["t"],
|
|
153
|
+
net_bp=per_h[best_h]["net_bp"], rho=dec["monotonicity_rho"],
|
|
154
|
+
fire=fire, tickers=panel.tickers, cfg=cfg, lag=max(best_h, 21))
|
|
155
|
+
|
|
156
|
+
neu = None
|
|
157
|
+
if neutralize_check:
|
|
158
|
+
neu = _dummy_neutralized(panel, fire, cums, horizons)
|
|
159
|
+
|
|
160
|
+
return EventResult(
|
|
161
|
+
spec={"horizons": list(horizons), "cost_bp": cost_bp, "entry_lag": panel.entry_lag,
|
|
162
|
+
"market": panel.market, "best_horizon": best_h,
|
|
163
|
+
"decile": dec},
|
|
164
|
+
per_horizon=per_h, gates=gates, structure=struct,
|
|
165
|
+
neutralized=neu, lookahead=lookahead)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _dummy_neutralized(panel: Panel, fire: np.ndarray, cums: dict,
|
|
169
|
+
horizons: tuple[int, ...], min_obs: int = 30) -> dict:
|
|
170
|
+
"""Event dummy regression: does firing still predict returns after controlling for characteristics?
|
|
171
|
+
|
|
172
|
+
Sample: every eligible name that day. Dependent: the future return. Independent: [firing dummy] + controls.
|
|
173
|
+
The dummy coefficient is 'what firing adds among names with the same characteristics'.
|
|
174
|
+
"""
|
|
175
|
+
ctrl = build_controls(panel)
|
|
176
|
+
cvs = [c.values.astype(np.float64) for c in ctrl.values()]
|
|
177
|
+
ev = panel.eligible.values
|
|
178
|
+
out = {}
|
|
179
|
+
for h in horizons:
|
|
180
|
+
cum = cums[h]
|
|
181
|
+
b_raw = np.full(cum.shape[0], np.nan)
|
|
182
|
+
b_neu = np.full(cum.shape[0], np.nan)
|
|
183
|
+
for i in range(cum.shape[0]):
|
|
184
|
+
pool = ev[i] & np.isfinite(cum[i])
|
|
185
|
+
for c in cvs:
|
|
186
|
+
pool = pool & np.isfinite(c[i])
|
|
187
|
+
n = int(pool.sum())
|
|
188
|
+
if n < min_obs:
|
|
189
|
+
continue
|
|
190
|
+
d = fire[i][pool].astype(np.float64)
|
|
191
|
+
if d.sum() < 3 or d.sum() > n - 3:
|
|
192
|
+
continue
|
|
193
|
+
y = cum[i][pool]
|
|
194
|
+
try:
|
|
195
|
+
b_raw[i] = np.linalg.lstsq(np.column_stack([np.ones(n), d]), y, rcond=None)[0][1]
|
|
196
|
+
X = np.column_stack([np.ones(n), d] + [c[i][pool] for c in cvs])
|
|
197
|
+
b_neu[i] = np.linalg.lstsq(X, y, rcond=None)[0][1]
|
|
198
|
+
except np.linalg.LinAlgError:
|
|
199
|
+
continue
|
|
200
|
+
m0, _, t0, _ = newey_west_t(b_raw, lag=max(h, 21))
|
|
201
|
+
m1, _, t1, _ = newey_west_t(b_neu, lag=max(h, 21))
|
|
202
|
+
keep = (m1 / m0 * 100) if (np.isfinite(m0) and abs(m0) > 1e-9) else np.nan
|
|
203
|
+
out[h] = {"raw_bp": m0 * 1e4 if np.isfinite(m0) else np.nan, "raw_t": t0,
|
|
204
|
+
"neutral_bp": m1 * 1e4 if np.isfinite(m1) else np.nan, "neutral_t": t1,
|
|
205
|
+
"survival_pct": float(keep) if np.isfinite(keep) else np.nan}
|
|
206
|
+
return out
|
pitbacktest/execution.py
ADDED
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
"""Whether a trade can be done, and at what size, at the price the backtest uses.
|
|
2
|
+
|
|
3
|
+
A daily backtest that trades every position change at the close assumes three things that are often false: that the security
|
|
4
|
+
was open (a halted name cannot be traded), that the price was not locked at a daily limit (a name closed at the upper limit has a
|
|
5
|
+
queue of buyers and no sellers, so you cannot buy it), and that you can hold any fraction of a share. This module states those
|
|
6
|
+
assumptions as data the engines read.
|
|
7
|
+
|
|
8
|
+
can_buy, can_sell `Panel` fields (bool, date x ticker). A trade that increases a position needs `can_buy` on the execution
|
|
9
|
+
day, one that decreases it needs `can_sell`. A blocked trade does not happen and the position stays as it was.
|
|
10
|
+
tradability() builds them from prices and volume: no price or no volume means halted; a daily move at the limit means locked.
|
|
11
|
+
at_prices() a copy of the panel that trades at another price (the open instead of the close).
|
|
12
|
+
realize() the engine's position path under those rules, plus whole-lot sizes and a minimum trade value.
|
|
13
|
+
|
|
14
|
+
Nothing here is specific to one market. Limits and lot sizes differ by market and change by rule, so they are arguments, not a table
|
|
15
|
+
inside the library.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from dataclasses import replace
|
|
20
|
+
|
|
21
|
+
import numpy as np
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
from .core.costs import rate_schedule
|
|
25
|
+
from .core.panel import Panel
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def tradability(panel: Panel, *, limit=None, tol: float = 0.01, halt_on_zero_volume: bool = True,
|
|
29
|
+
max_gap_days: int | None = 60) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
30
|
+
"""(can_buy, can_sell) built from the panel's prices and volume.
|
|
31
|
+
|
|
32
|
+
After a security's flagged delisting (`delist_after`) both trades are allowed, so that a position in it can be closed.
|
|
33
|
+
Halted: no close that day, or (if the panel has `volume` and `halt_on_zero_volume`) no volume. A security with a close but a
|
|
34
|
+
missing volume is also treated as halted, because an unknown is not permission. A panel without `volume` is judged by prices alone.
|
|
35
|
+
limit the daily price limit as a fraction (0.30 for 30%), a list of (effective_date, fraction) when it changed, or None for a
|
|
36
|
+
market without one. A day whose move is at least `limit - tol` up cannot be bought (a locked-up name has no sellers) and a
|
|
37
|
+
day at least `limit - tol` down cannot be sold. The tolerance is there because tick rounding keeps the limit move slightly
|
|
38
|
+
under the limit. This is **conservative**: it also blocks days that reached the limit but traded freely before it.
|
|
39
|
+
tol in fractions, 0 <= tol < limit.
|
|
40
|
+
max_gap_days after this many consecutive days **without any price**, a position is treated as settled and can be closed either way
|
|
41
|
+
(None: never). An exchange settles a delisted contract and a vendor gap or a ticker that came back years later (a contract
|
|
42
|
+
that was absent for years and then listed again, in the Binance cache) is not a position you can hold through. A name with a price
|
|
43
|
+
but no volume (a Korean trading suspension) is not covered: that one really is frozen, for as long as the data says."""
|
|
44
|
+
c = panel.close
|
|
45
|
+
ok = c.notna()
|
|
46
|
+
if halt_on_zero_volume and panel.volume is not None:
|
|
47
|
+
ok = ok & (panel.volume.fillna(0.0) > 0)
|
|
48
|
+
can_buy, can_sell = ok.copy(), ok.copy()
|
|
49
|
+
if panel.delist_after is not None and bool(np.asarray(panel.delist_after.values).any()):
|
|
50
|
+
# After a flagged delisting there is no price for ever. The engines already settle such a position (`delist_return`, or the last
|
|
51
|
+
# price), so closing it is allowed in **both** directions (a long is sold, a short is bought back): refusing the exit would freeze a
|
|
52
|
+
# dead position in the book for the rest of the sample, and a frozen short is exactly what the first real-data run showed.
|
|
53
|
+
da = panel.delist_after.reindex(index=c.index, columns=c.columns).fillna(False).astype(bool)
|
|
54
|
+
gone = (da.astype(int).cumsum() - da.astype(int)) > 0
|
|
55
|
+
can_buy, can_sell = can_buy | gone, can_sell | gone
|
|
56
|
+
if max_gap_days is not None:
|
|
57
|
+
isn = c.isna().to_numpy()
|
|
58
|
+
run = np.zeros(c.shape[1], dtype=int)
|
|
59
|
+
long_gap = np.zeros(c.shape, dtype=bool)
|
|
60
|
+
for t in range(c.shape[0]):
|
|
61
|
+
run = np.where(isn[t], run + 1, 0)
|
|
62
|
+
long_gap[t] = run > max_gap_days
|
|
63
|
+
settled = pd.DataFrame(long_gap, index=c.index, columns=c.columns)
|
|
64
|
+
can_buy, can_sell = can_buy | settled, can_sell | settled
|
|
65
|
+
if limit is not None:
|
|
66
|
+
lim = (rate_schedule(panel.dates, limit, "limit").to_numpy(float) if isinstance(limit, (list, tuple))
|
|
67
|
+
else np.full(len(panel.dates), float(limit)))
|
|
68
|
+
if not (np.all(lim > 0) and np.all(lim < 1)):
|
|
69
|
+
raise ValueError(f"limit must be a fraction between 0 and 1 (0.30 for 30%), got {limit!r}")
|
|
70
|
+
if not (0 <= tol < lim.min()):
|
|
71
|
+
raise ValueError(f"tol must be at least 0 and below the limit, got {tol!r}")
|
|
72
|
+
ret = c.pct_change(fill_method=None).to_numpy(float)
|
|
73
|
+
thr = (lim - tol)[:, None]
|
|
74
|
+
with np.errstate(invalid="ignore"):
|
|
75
|
+
up, down = ret >= thr, ret <= -thr
|
|
76
|
+
can_buy = can_buy & ~pd.DataFrame(up, index=c.index, columns=c.columns)
|
|
77
|
+
can_sell = can_sell & ~pd.DataFrame(down, index=c.index, columns=c.columns)
|
|
78
|
+
return can_buy, can_sell
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def halt_runs(panel: Panel) -> np.ndarray:
|
|
82
|
+
"""(date x ticker) int: how many consecutive days, counting the day itself, the security has had a price but no (or unknown) volume; 0 on a day it trades or has no price.
|
|
83
|
+
A panel without `volume` has none."""
|
|
84
|
+
c = panel.close
|
|
85
|
+
if panel.volume is None:
|
|
86
|
+
return np.zeros(c.shape, dtype=int)
|
|
87
|
+
halted = (c.notna() & ~(panel.volume.fillna(0.0) > 0)).to_numpy()
|
|
88
|
+
run = np.zeros(c.shape, dtype=int)
|
|
89
|
+
for t in range(c.shape[0]):
|
|
90
|
+
run[t] = np.where(halted[t], (run[t - 1] if t else 0) + 1, 0)
|
|
91
|
+
return run
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def freeze_hits(panel: Panel, freeze_days: int | None) -> np.ndarray | None:
|
|
95
|
+
"""(date x ticker) bool indexed by the **signal date**: True where the execution day (signal date + `entry_lag`) is exactly the `freeze_days`-th day of a suspension.
|
|
96
|
+
None when `freeze_days` is None. A suspension marks a position down once and not again however long it lasts. The markdown is booked on the signal-date row, which is
|
|
97
|
+
the `freeze_days`-th suspended day minus `entry_lag`, and it enters the return of that row (close of the execution day to the close of the next), so it is the
|
|
98
|
+
`freeze_days`-th suspended day's close that takes the loss."""
|
|
99
|
+
if freeze_days is None:
|
|
100
|
+
return None
|
|
101
|
+
if isinstance(freeze_days, bool) or not isinstance(freeze_days, (int, np.integer)) or freeze_days < 1:
|
|
102
|
+
raise ValueError(f"freeze_days must be a whole number of days, at least 1, or None, got {freeze_days!r}")
|
|
103
|
+
hit = halt_runs(panel) == freeze_days
|
|
104
|
+
lag = panel.entry_lag
|
|
105
|
+
return np.vstack([hit[lag:], np.zeros((lag, hit.shape[1]), bool)]) if lag else hit
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def check_freeze_return(freeze_return: float) -> float:
|
|
109
|
+
"""`freeze_return` as a float, or ValueError unless it is between -1 and 0 (a markdown is never a gain)."""
|
|
110
|
+
if isinstance(freeze_return, (bool, np.bool_)) or not isinstance(freeze_return, (int, float, np.integer, np.floating)):
|
|
111
|
+
raise ValueError(f"freeze_return must be a number between -1 and 0, got {freeze_return!r}")
|
|
112
|
+
if not (np.isfinite(freeze_return) and -1.0 <= freeze_return <= 0.0):
|
|
113
|
+
raise ValueError(f"freeze_return must be between -1 and 0 (a markdown is never a gain), got {freeze_return!r}")
|
|
114
|
+
return float(freeze_return)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def freeze_episodes(panel: Panel, *, min_days: int = 20, relist_days: int = 20) -> pd.DataFrame:
|
|
118
|
+
"""The stretches of at least `min_days` consecutive days on which a security had a price but no volume (a trading suspension), and how each ended.
|
|
119
|
+
|
|
120
|
+
One row per episode: ticker, start, end (the last suspended day), days, outcome and `ret`, the return from the last price before the suspension
|
|
121
|
+
to the price at the end of the story.
|
|
122
|
+
A day without a price (NaN) ends a suspension, so a halt with a gap of missing prices inside it is cut into two episodes. A suspension that is already running
|
|
123
|
+
on the panel's first day has no price before it: such an episode is listed, but its `ret` is NaN, and its length is truncated.
|
|
124
|
+
outcome "resumed" the first priced day after the suspension, and the data goes on (or the security is not flagged in `delist_after`):
|
|
125
|
+
ret is that day's price over the price before the suspension. That day may itself have no volume if missing prices sit between.
|
|
126
|
+
"resumed_then_delisted" the security is flagged in `delist_after` and its last bar falls within `relist_days` of that day (a liquidation window): ret is its
|
|
127
|
+
last price over the price before. Without the flag the same story is labeled "resumed".
|
|
128
|
+
"ended_in_halt" the security's last bar is a suspended day and it is flagged in `delist_after`: the price never moved, ret is 0 (NaN if there is
|
|
129
|
+
no price before), and the loss, if any, was never in the data
|
|
130
|
+
"ongoing" no priced day follows: either still suspended on the last day of the panel, or the data ends in the suspension without a delisting
|
|
131
|
+
flag; ret is NaN
|
|
132
|
+
The point is to measure how often a position that cannot be sold is lost, instead of assuming it. Counts are small and depend on the market and
|
|
133
|
+
the period; read the distribution of `ret`, not a mean."""
|
|
134
|
+
c = panel.close
|
|
135
|
+
halted = (c.notna() & ~(panel.volume.fillna(0.0) > 0)).to_numpy() if panel.volume is not None else np.zeros(c.shape, bool)
|
|
136
|
+
px = c.to_numpy(float)
|
|
137
|
+
da = panel.delist_after.reindex(index=c.index, columns=c.columns).fillna(False).to_numpy(bool) if panel.delist_after is not None else np.zeros(c.shape, bool)
|
|
138
|
+
T, N = c.shape
|
|
139
|
+
rows = []
|
|
140
|
+
for j in range(N):
|
|
141
|
+
h = halted[:, j]
|
|
142
|
+
if not h.any():
|
|
143
|
+
continue
|
|
144
|
+
t = 0
|
|
145
|
+
valid = np.where(~np.isnan(px[:, j]))[0]
|
|
146
|
+
last_bar = int(valid[-1]) if len(valid) else -1
|
|
147
|
+
flagged = bool(da[:, j].any())
|
|
148
|
+
while t < T:
|
|
149
|
+
if not h[t]:
|
|
150
|
+
t += 1
|
|
151
|
+
continue
|
|
152
|
+
e = t
|
|
153
|
+
while e + 1 < T and h[e + 1]:
|
|
154
|
+
e += 1
|
|
155
|
+
if e - t + 1 >= min_days:
|
|
156
|
+
before = next((px[k, j] for k in range(t - 1, -1, -1) if not np.isnan(px[k, j]) and not h[k]), np.nan)
|
|
157
|
+
if e == T - 1 and not (flagged and last_bar == e):
|
|
158
|
+
outcome, ret = "ongoing", np.nan
|
|
159
|
+
elif flagged and last_bar == e:
|
|
160
|
+
outcome, ret = "ended_in_halt", 0.0 if np.isfinite(before) else np.nan
|
|
161
|
+
else:
|
|
162
|
+
nxt = e + 1
|
|
163
|
+
while nxt < T and np.isnan(px[nxt, j]):
|
|
164
|
+
nxt += 1
|
|
165
|
+
if nxt >= T:
|
|
166
|
+
outcome, ret = "ongoing", np.nan
|
|
167
|
+
elif flagged and last_bar - nxt <= relist_days:
|
|
168
|
+
outcome, ret = "resumed_then_delisted", px[last_bar, j] / before - 1.0 if np.isfinite(before) else np.nan
|
|
169
|
+
else:
|
|
170
|
+
outcome, ret = "resumed", px[nxt, j] / before - 1.0 if np.isfinite(before) else np.nan
|
|
171
|
+
rows.append({"ticker": c.columns[j], "start": c.index[t], "end": c.index[e], "days": e - t + 1, "outcome": outcome, "ret": ret})
|
|
172
|
+
t = e + 1
|
|
173
|
+
return pd.DataFrame(rows, columns=["ticker", "start", "end", "days", "outcome", "ret"])
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def at_prices(panel: Panel, price="open") -> Panel:
|
|
177
|
+
"""A copy of `panel` whose prices are another series, so that a position is entered and marked at it. `price` is the name of a panel
|
|
178
|
+
field (`"open"`, `"high"`, `"low"`) or a (date x ticker) frame. With the open, a signal from the close of day d is entered at the open
|
|
179
|
+
of d+lag and held to the open of d+lag+1. A price that is zero or negative is treated as missing. Only `close` changes: eligibility, volume and the rest were built from the original
|
|
180
|
+
closes, and `screen` and `backtest_event` still compute their controls from whatever `close` now holds."""
|
|
181
|
+
px = getattr(panel, price) if isinstance(price, str) else price
|
|
182
|
+
if px is None:
|
|
183
|
+
raise ValueError(f"the panel has no {price!r}")
|
|
184
|
+
if not isinstance(px, pd.DataFrame):
|
|
185
|
+
raise ValueError("price must be a field name or a (date x ticker) frame")
|
|
186
|
+
px = px.reindex(index=panel.dates, columns=panel.tickers)
|
|
187
|
+
return replace(panel, close=px.where(px > 0)) # a zero or negative price is no price: the engines treat a missing price as a return of 0, a zero would give infinity
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def exec_masks(panel: Panel):
|
|
191
|
+
"""(can_buy, can_sell) as arrays indexed by the **signal date**, i.e. read on the execution day d + entry_lag. None when the panel
|
|
192
|
+
has neither field. A missing frame counts as always possible."""
|
|
193
|
+
if panel.can_buy is None and panel.can_sell is None:
|
|
194
|
+
return None, None
|
|
195
|
+
lag = panel.entry_lag
|
|
196
|
+
|
|
197
|
+
def one(f):
|
|
198
|
+
if f is None:
|
|
199
|
+
return np.ones(panel.close.shape, dtype=bool)
|
|
200
|
+
return f.shift(-lag, fill_value=True).to_numpy(bool)
|
|
201
|
+
return one(panel.can_buy), one(panel.can_sell)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def round_to_lots(weights: np.ndarray, capital: float, price: np.ndarray, lot) -> np.ndarray:
|
|
205
|
+
"""The weights you can actually hold: whole lots at the given prices. Shares = weight x capital / price, rounded to a multiple of
|
|
206
|
+
`lot`. `price` must be the **real** price level (not a back-adjusted series, whose level is arbitrary). A name with no price keeps weight NaN."""
|
|
207
|
+
with np.errstate(divide="ignore", invalid="ignore"):
|
|
208
|
+
shares = weights * capital / price
|
|
209
|
+
lots = np.round(shares / lot) * lot
|
|
210
|
+
return lots * price / capital
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def realize(target: np.ndarray, can_buy: np.ndarray | None = None, can_sell: np.ndarray | None = None, *,
|
|
214
|
+
capital: float | None = None, price: np.ndarray | None = None, lot=1.0, min_trade_value: float = 0.0, cap_gross: bool = False):
|
|
215
|
+
"""The positions that result from asking for `target` row by row (row 0 is a build from cash).
|
|
216
|
+
|
|
217
|
+
Each day, per name: round the target to whole lots (when `capital` is given; `price` is read on the **same row** as the target, so pass the price of the day the
|
|
218
|
+
trade is done: `backtest_weights` shifts it by `entry_lag`); skip the trade if its value is under `min_trade_value` (a number, or one per security);
|
|
219
|
+
skip it if it increases the position where `can_buy` is False, or decreases it where `can_sell` is False. A skipped trade leaves the
|
|
220
|
+
position as it was the day before. A name with no price on a day (NaN) keeps its position.
|
|
221
|
+
The weight is what the engines keep constant, so a day without a trade keeps the previous **weight**, not the previous share count: after the price moves, the
|
|
222
|
+
shares it implies need not be a multiple of the lot (a halted name does not move, so there it is).
|
|
223
|
+
|
|
224
|
+
`cap_gross`: a position that cannot be traded ties up capital. Without this, new targets are added on top of it and the gross exposure can exceed the
|
|
225
|
+
target's. With it, the day's whole target is scaled by a k in [0, 1], the largest one found by bisection for which the resulting gross exposure does not exceed
|
|
226
|
+
the gross of the day's target (after lot rounding, so rounding itself is never cut). A position that cannot be sold stays where it is; one that cannot be
|
|
227
|
+
bought stays too unless the scaled target falls below it, in which case it is sold down, so it can shrink. A scaled target can turn a buy into a sell, which
|
|
228
|
+
changes what is blocked, so k is searched for rather than solved for, and the feasible set being an interval [0, k] is not proven (no counter-example was found
|
|
229
|
+
in 653 random small cases). On a day with nothing blocked or skipped k is exactly 1 and the result is the same as without it, with or without `capital`. If the
|
|
230
|
+
stuck positions alone exceed the target gross, nothing can meet the cap: k is 0 and the day is counted in `infeasible_days`.
|
|
231
|
+
|
|
232
|
+
Returns (positions, stats). `mean_stuck_weight` is the average over days of the weight held where the target said otherwise because a trade was
|
|
233
|
+
blocked; `longest_freeze_days` the longest run of consecutive days a wanted trade in one name was blocked; `mean_free_scale` the average k (1 without `cap_gross`), `infeasible_days` the days on which the stuck positions alone exceeded the target gross."""
|
|
234
|
+
T, N = target.shape
|
|
235
|
+
use_lots = capital is not None
|
|
236
|
+
mtv = np.asarray(min_trade_value, dtype=float) # a number, or one per security
|
|
237
|
+
mtv_any = bool(np.any(mtv > 0))
|
|
238
|
+
out = np.zeros_like(target, dtype=float)
|
|
239
|
+
prev = np.zeros(N)
|
|
240
|
+
asked = blocked_turn = skipped_turn = gap = stuck = k_sum = 0.0
|
|
241
|
+
n_blocked = n_skipped = infeasible = 0
|
|
242
|
+
streak = np.zeros(N, dtype=int)
|
|
243
|
+
longest = 0
|
|
244
|
+
|
|
245
|
+
def decide(raw: np.ndarray, t: int):
|
|
246
|
+
tgt = raw
|
|
247
|
+
if use_lots:
|
|
248
|
+
lt = lot[t] if isinstance(lot, np.ndarray) and lot.ndim == 2 else lot
|
|
249
|
+
tgt = round_to_lots(raw, capital, price[t], lt)
|
|
250
|
+
# A flat target needs no price to carry out (a delisted name is simply closed); a non-flat target with no price cannot be sized, so the
|
|
251
|
+
# position stays.
|
|
252
|
+
tgt = np.where(raw == 0, 0.0, np.where(np.isfinite(tgt), tgt, prev))
|
|
253
|
+
d = tgt - prev
|
|
254
|
+
stay = np.zeros(N, dtype=bool)
|
|
255
|
+
if use_lots and mtv_any:
|
|
256
|
+
stay |= (np.abs(d) * capital < mtv) & (d != 0)
|
|
257
|
+
blk = np.zeros(N, dtype=bool)
|
|
258
|
+
if can_buy is not None:
|
|
259
|
+
blk |= (d > 0) & ~can_buy[t]
|
|
260
|
+
if can_sell is not None:
|
|
261
|
+
blk |= (d < 0) & ~can_sell[t]
|
|
262
|
+
return tgt, d, stay, blk
|
|
263
|
+
|
|
264
|
+
for t in range(T):
|
|
265
|
+
raw0 = target[t]
|
|
266
|
+
k = 1.0
|
|
267
|
+
tgt, d, stay, blk = decide(raw0, t)
|
|
268
|
+
if cap_gross:
|
|
269
|
+
budget = float(np.abs(tgt).sum()) + 1e-12 # what the day would hold with nothing in the way (after lot rounding)
|
|
270
|
+
|
|
271
|
+
def gross_at(kk: float) -> float:
|
|
272
|
+
g_tgt, _, g_stay, g_blk = decide(raw0 * kk, t)
|
|
273
|
+
return float(np.abs(np.where(g_stay | g_blk, prev, g_tgt)).sum())
|
|
274
|
+
|
|
275
|
+
if float(np.abs(np.where(stay | blk, prev, tgt)).sum()) > budget:
|
|
276
|
+
if gross_at(0.0) > budget:
|
|
277
|
+
k = 0.0
|
|
278
|
+
infeasible += 1
|
|
279
|
+
else:
|
|
280
|
+
lo, hi = 0.0, 1.0 # lo is always feasible, hi is not
|
|
281
|
+
while hi - lo > 1e-14:
|
|
282
|
+
mid = 0.5 * (lo + hi)
|
|
283
|
+
lo, hi = (mid, hi) if gross_at(mid) <= budget else (lo, mid)
|
|
284
|
+
k = lo
|
|
285
|
+
raw = raw0 * k
|
|
286
|
+
tgt, d, stay, blk = decide(raw, t)
|
|
287
|
+
else:
|
|
288
|
+
raw = raw0
|
|
289
|
+
else:
|
|
290
|
+
raw = raw0
|
|
291
|
+
k_sum += k
|
|
292
|
+
if use_lots:
|
|
293
|
+
gap += float(np.abs(tgt - raw).sum())
|
|
294
|
+
cur = np.where(stay | blk, prev, tgt)
|
|
295
|
+
held = blk & ~stay
|
|
296
|
+
stuck += float(np.abs(cur - tgt)[held].sum()) # weight sitting where it should not be, today
|
|
297
|
+
streak = np.where(held, streak + 1, 0)
|
|
298
|
+
longest = max(longest, int(streak.max())) if N else longest
|
|
299
|
+
if t > 0:
|
|
300
|
+
asked += float(np.abs(d).sum())
|
|
301
|
+
blocked_turn += float(np.abs(d[blk & ~stay]).sum())
|
|
302
|
+
skipped_turn += float(np.abs(d[stay]).sum())
|
|
303
|
+
n_blocked += int((blk & ~stay).sum())
|
|
304
|
+
n_skipped += int(stay.sum())
|
|
305
|
+
out[t] = cur
|
|
306
|
+
prev = cur
|
|
307
|
+
stats = {"mean_stuck_weight": stuck / T if T else 0.0, "longest_freeze_days": longest, "asked_turnover": asked, "blocked_turnover": blocked_turn, "blocked_trades": n_blocked, "blocked_turnover_share": blocked_turn / asked if asked > 0 else 0.0,
|
|
308
|
+
"min_trade_skipped": n_skipped, "min_trade_skipped_share": skipped_turn / asked if asked > 0 else 0.0,
|
|
309
|
+
"mean_abs_rounding_gap": gap / (T * N) if use_lots else 0.0, "mean_free_scale": k_sum / T if T else 1.0, "infeasible_days": infeasible}
|
|
310
|
+
return out, stats
|