pitbacktest 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,104 @@
1
+ """Survivorship scenarios: put delistings back into a survivors-only panel and see how the result moves.
2
+
3
+ This is a **what-if**, not a correction. The actual delisted companies are unknown here, so we draw them at random
4
+ (optionally tilted toward small, illiquid or volatile names, which is where delistings concentrate in the literature) and
5
+ give them an explicit delisting return. Run it over a grid of annual rates and delisting returns and report the range, not a
6
+ single "corrected" number.
7
+
8
+ Order-of-magnitude guide from the delisting literature (check it for your period and market): a few percent of listed
9
+ names disappear each year, roughly half of them for performance reasons; the average return on a performance-related
10
+ delisting is about -30% on NYSE/AMEX (Shumway 1997) and about -55% on Nasdaq (Shumway and Warther 1999).
11
+ """
12
+ from __future__ import annotations
13
+
14
+ from dataclasses import replace
15
+ from typing import Callable
16
+
17
+ import numpy as np
18
+ import pandas as pd
19
+
20
+ from ..core.panel import Panel
21
+
22
+
23
+ def _weights(panel: Panel, year_start: pd.Timestamp, names: pd.Index, hazard: str) -> np.ndarray:
24
+ if hazard == "uniform":
25
+ return np.ones(len(names))
26
+ before = panel.dates < year_start
27
+ if not before.any():
28
+ return np.ones(len(names)) # first year: nothing earlier to tilt on, so uniform
29
+ if hazard == "volatile":
30
+ v = panel.close.pct_change(fill_method=None).loc[before].tail(60).std().reindex(names)
31
+ elif hazard == "illiquid":
32
+ adv = panel.adv(30)
33
+ v = -(adv.loc[before].tail(1).iloc[0].reindex(names)) if adv is not None else pd.Series(0.0, index=names)
34
+ else:
35
+ raise ValueError(hazard)
36
+ r = v.rank(pct=True).fillna(0.5).to_numpy()
37
+ return 0.2 + r ** 2 # tilted, but every name keeps a chance
38
+
39
+
40
+ def inject_delistings(panel: Panel, *, annual_rate: float = 0.03, hazard: str = "uniform", seed: int = 0,
41
+ min_age_days: int = 60) -> Panel:
42
+ """Return a copy of `panel` in which a fraction `annual_rate` of the names alive at the start of each calendar year stop
43
+ trading at a random date later that year. The last real bar is flagged in `delist_after` and prices, volume and
44
+ eligibility are removed afterwards."""
45
+ if hazard not in ("uniform", "volatile", "illiquid"):
46
+ raise ValueError(f"hazard must be 'uniform', 'volatile' or 'illiquid', not {hazard!r}")
47
+ if hazard == "illiquid" and panel.volume is None:
48
+ raise ValueError("hazard='illiquid' needs panel.volume")
49
+ if not (0 <= annual_rate <= 1):
50
+ raise ValueError(f"annual_rate must be between 0 and 1, got {annual_rate!r}")
51
+ rng = np.random.default_rng(seed)
52
+ close = panel.close.copy()
53
+ elig = panel.eligible.copy()
54
+ vol = None if panel.volume is None else panel.volume.copy()
55
+ da = (pd.DataFrame(False, index=panel.dates, columns=panel.tickers) if panel.delist_after is None
56
+ else panel.delist_after.copy())
57
+ for y in sorted(set(panel.dates.year)):
58
+ idx = panel.dates[panel.dates.year == y]
59
+ if len(idx) < 100:
60
+ continue # skip partial years
61
+ alive = close.loc[idx[0]].dropna().index
62
+ k = int(round(annual_rate * len(alive)))
63
+ if k == 0:
64
+ continue
65
+ w = _weights(panel, idx[0], alive, hazard)
66
+ pick = rng.choice(len(alive), size=min(k, len(alive)), replace=False, p=w / w.sum())
67
+ for j in pick:
68
+ name = alive[j]
69
+ when = idx[int(rng.integers(min_age_days // 3, len(idx) - 1))]
70
+ close.loc[when + pd.Timedelta(days=1):, name] = np.nan
71
+ elig.loc[when + pd.Timedelta(days=1):, name] = False
72
+ if vol is not None:
73
+ vol.loc[when + pd.Timedelta(days=1):, name] = np.nan
74
+ da.loc[when, name] = True
75
+ return replace(panel, close=close, eligible=elig, volume=vol, delist_after=da)
76
+
77
+
78
+ def survivorship_scenarios(panel: Panel, evaluate: Callable[[Panel, float], float], *,
79
+ rates=(0.01, 0.03, 0.05), delist_returns=(-0.30, -0.55), hazards=("uniform", "volatile"),
80
+ n: int = 10, seed: int = 0) -> pd.DataFrame:
81
+ """Run `evaluate(panel, delist_return)` (it should return one number, for example the net Sharpe) on the original panel
82
+ and on `n` randomly delisting copies for every combination. Returns the baseline and the spread of outcomes."""
83
+ base = float(evaluate(panel, 0.0))
84
+ rows = []
85
+ for hz in hazards:
86
+ for r in rates:
87
+ for d in delist_returns:
88
+ vals = [float(evaluate(inject_delistings(panel, annual_rate=r, hazard=hz, seed=seed + i), d)) for i in range(n)]
89
+ rows.append({"hazard": hz, "annual_rate": r, "delist_return": d, "baseline": base,
90
+ "mean": float(np.mean(vals)), "p5": float(np.percentile(vals, 5)),
91
+ "p95": float(np.percentile(vals, 95)), "mean_change": float(np.mean(vals) - base)})
92
+ return pd.DataFrame(rows)
93
+
94
+
95
+ def survivors_only(panel: Panel, end_gap_days: int = 5) -> Panel:
96
+ """The usual shortcut, made explicit: keep only the securities that still have a price on the last day.
97
+ Compare a result on `panel` with the same result on `survivors_only(panel)` to measure survivorship bias directly."""
98
+ last = panel.close.apply(lambda c: c.last_valid_index())
99
+ keep = [c for c in panel.close.columns if last[c] is not None and last[c] >= panel.dates[-1] - pd.Timedelta(days=end_gap_days)]
100
+ sub = lambda d: None if d is None else d[keep]
101
+ return replace(panel, close=panel.close[keep], eligible=panel.eligible[keep], open=sub(panel.open), high=sub(panel.high),
102
+ low=sub(panel.low), volume=sub(panel.volume), mkt_cap=sub(panel.mkt_cap),
103
+ chars={k: v[keep] for k, v in panel.chars.items()}, funding=sub(panel.funding),
104
+ delist_after=sub(panel.delist_after))
pitbacktest/event.py ADDED
@@ -0,0 +1,206 @@
1
+ """C. Event-signal test: does it hold up as individual alerts?
2
+
3
+ Portfolio metrics (mean excess return) assume the mean is realised. A user receives alerts one by one, though, so the **win rate,
4
+ median and payoff ratio** decide what it feels like.
5
+
6
+ What is always decomposed
7
+ win rate = base rate (how the sample was picked) + lift (what the signal adds)
8
+ - A longer holding period raises the win rate and the base rate together. Judge by the lift only.
9
+ - A win rate below the base rate (lift < 0) means the signal does harm.
10
+ mean vs median
11
+ - If the signs differ, a few big winners cover many losses: it works as a portfolio and not as alerts.
12
+ - The typical case is a positive mean with a negative median.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import warnings
18
+ from dataclasses import dataclass
19
+
20
+ import numpy as np
21
+ import pandas as pd
22
+
23
+ from .core.controls import build_controls, xs_norm
24
+ from .core.estimators import fama_macbeth, newey_west_t, paired_diff
25
+ from .core.gates import GateConfig, fire_structure, run_gates
26
+ from .core.panel import Panel
27
+
28
+
29
+ @dataclass
30
+ class EventResult:
31
+ spec: dict
32
+ per_horizon: dict
33
+ gates: dict
34
+ structure: dict
35
+ neutralized: dict | None
36
+ lookahead: dict
37
+
38
+
39
+ def _trade_stats(fire: np.ndarray, cum: np.ndarray, eligible: np.ndarray,
40
+ cost: float) -> dict | None:
41
+ """Per-trade profit distribution plus the base win rate."""
42
+ tr, bh, bn = [], 0.0, 0
43
+ for i in range(cum.shape[0]):
44
+ s = fire[i] & np.isfinite(cum[i])
45
+ k = int(s.sum())
46
+ if k < 1:
47
+ continue
48
+ tr.append(cum[i][s] - cost)
49
+ pool = eligible[i] & np.isfinite(cum[i])
50
+ if pool.sum() >= 20:
51
+ bh += float((cum[i][pool] - cost > 0).mean()) * k
52
+ bn += k
53
+ if not tr:
54
+ return None
55
+ t = np.concatenate(tr)
56
+ if len(t) < 100:
57
+ return None
58
+ win = float((t > 0).mean())
59
+ base = bh / bn if bn else np.nan
60
+ w, l = t[t > 0], t[t < 0]
61
+ return {
62
+ "n_trades": int(len(t)),
63
+ "win_rate": win * 100,
64
+ "base_rate": base * 100 if np.isfinite(base) else np.nan,
65
+ "lift_pp": (win - base) * 100 if np.isfinite(base) else np.nan,
66
+ "mean_bp": float(t.mean() * 1e4),
67
+ "median_bp": float(np.median(t) * 1e4),
68
+ "avg_win_bp": float(w.mean() * 1e4) if len(w) else np.nan,
69
+ "avg_loss_bp": float(l.mean() * 1e4) if len(l) else np.nan,
70
+ "payoff": float(w.mean() / abs(l.mean())) if len(l) else np.nan,
71
+ "p25_bp": float(np.percentile(t, 25) * 1e4),
72
+ "p75_bp": float(np.percentile(t, 75) * 1e4),
73
+ "skew_warning": bool(t.mean() > 0 > np.median(t)),
74
+ }
75
+
76
+
77
+ def _daily_excess(fire: np.ndarray, cum: np.ndarray, eligible: np.ndarray,
78
+ *, min_fire: int = 1, min_pool: int | None = None) -> np.ndarray:
79
+ """Daily excess return of the firing group (gate input).
80
+
81
+ A fixed min_pool throws away every date in a small universe (for example a 20-name test).
82
+ The default **adapts** to half the median universe size (at most 30).
83
+ """
84
+ if min_pool is None:
85
+ med = float(np.median(eligible.sum(axis=1)))
86
+ min_pool = int(max(5, min(30, med * 0.5)))
87
+ out = []
88
+ for i in range(cum.shape[0]):
89
+ s = fire[i] & np.isfinite(cum[i])
90
+ pool = eligible[i] & np.isfinite(cum[i])
91
+ if s.sum() >= min_fire and pool.sum() >= min_pool:
92
+ out.append(cum[i][s].mean() - cum[i][pool].mean())
93
+ return np.array(out)
94
+
95
+
96
+ def backtest_event(panel: Panel, signal: pd.DataFrame, *,
97
+ horizons: tuple[int, ...] = (1, 2, 5, 10, 21),
98
+ cost_bp: float = 20.0,
99
+ neutralize_check: bool = True,
100
+ gate_config: GateConfig | None = None,
101
+ null_threshold: float | None = None,
102
+ delist_return: float | None = None) -> EventResult:
103
+ """Event-signal test.
104
+
105
+ signal bool or 0/1 matrix (date x ticker). True = fires that day.
106
+ cost_bp round-trip trading cost. If per-security measured costs exist, use the spread panel on the portfolio side.
107
+ delist_return return assumed on the day after a security's last bar when it is flagged in panel.delist_after (None: carried at its last price).
108
+
109
+ With the neutralisation test on (the default) the contribution after controls is re-measured with a **dummy regression**.
110
+ An event signal is not a continuous factor, so it is read through a dummy coefficient and not through residuals.
111
+ """
112
+ if not horizons or any(isinstance(h, bool) or not isinstance(h, (int, np.integer)) or h < 1 for h in horizons):
113
+ raise ValueError(f"horizons must be whole numbers of periods, at least 1, got {horizons!r}")
114
+ if not (cost_bp >= 0 and np.isfinite(cost_bp)):
115
+ raise ValueError(f"cost_bp must be finite and not negative, got {cost_bp!r}")
116
+ sig = signal.reindex(index=panel.dates, columns=panel.tickers).fillna(False)
117
+ fire = (sig.astype(bool) & panel.eligible).values
118
+ ev = panel.eligible.values
119
+ cost = cost_bp / 1e4
120
+
121
+ lookahead = panel.assert_no_lookahead(sig.astype(float), h=max(horizons))
122
+ cums = {h: panel.forward(h, delist_return).values.astype(np.float64) for h in horizons}
123
+
124
+ per_h = {}
125
+ for h in horizons:
126
+ st = _trade_stats(fire, cums[h], ev, cost)
127
+ if st is None:
128
+ continue
129
+ de = _daily_excess(fire, cums[h], ev)
130
+ mu, _, t, n = newey_west_t(de, lag=max(h, 21))
131
+ per_h[h] = {**st, "excess_bp": float(mu * 1e4), "t": float(t), "n_days": int(n),
132
+ "net_bp": float(mu * 1e4 - cost_bp)}
133
+
134
+ if not per_h:
135
+ raise ValueError(
136
+ "No horizon can be tested. Either the signal fires too rarely (fewer than 100 events) or "
137
+ "too few securities are eligible. Check signal, eligible and horizons.")
138
+ # representative horizon = where the lift is largest
139
+ best_h = max(per_h, key=lambda h: per_h[h].get("lift_pp", -99))
140
+ de = _daily_excess(fire, cums[best_h], ev)
141
+ if len(de) < 100:
142
+ warnings.warn(
143
+ f"Only {len(de)} days of daily excess return, so the gates cannot be trusted. "
144
+ f"_daily_excess counts only days with at least one firing name and a pool of at least max(5, min(30, half the "
145
+ f"median universe)) names, so a short sample or a rarely firing signal leaves few days.", stacklevel=2)
146
+
147
+ from .core.estimators import decile_profile
148
+ dec = decile_profile(sig.astype(float), panel.forward(best_h, delist_return), panel.eligible,
149
+ lag=max(best_h, 21))
150
+ struct = fire_structure(fire, panel.dates, panel.tickers)
151
+ cfg = gate_config or GateConfig(null_threshold=null_threshold)
152
+ gates = run_gates(de, panel.dates, t_stat=per_h[best_h]["t"],
153
+ net_bp=per_h[best_h]["net_bp"], rho=dec["monotonicity_rho"],
154
+ fire=fire, tickers=panel.tickers, cfg=cfg, lag=max(best_h, 21))
155
+
156
+ neu = None
157
+ if neutralize_check:
158
+ neu = _dummy_neutralized(panel, fire, cums, horizons)
159
+
160
+ return EventResult(
161
+ spec={"horizons": list(horizons), "cost_bp": cost_bp, "entry_lag": panel.entry_lag,
162
+ "market": panel.market, "best_horizon": best_h,
163
+ "decile": dec},
164
+ per_horizon=per_h, gates=gates, structure=struct,
165
+ neutralized=neu, lookahead=lookahead)
166
+
167
+
168
+ def _dummy_neutralized(panel: Panel, fire: np.ndarray, cums: dict,
169
+ horizons: tuple[int, ...], min_obs: int = 30) -> dict:
170
+ """Event dummy regression: does firing still predict returns after controlling for characteristics?
171
+
172
+ Sample: every eligible name that day. Dependent: the future return. Independent: [firing dummy] + controls.
173
+ The dummy coefficient is 'what firing adds among names with the same characteristics'.
174
+ """
175
+ ctrl = build_controls(panel)
176
+ cvs = [c.values.astype(np.float64) for c in ctrl.values()]
177
+ ev = panel.eligible.values
178
+ out = {}
179
+ for h in horizons:
180
+ cum = cums[h]
181
+ b_raw = np.full(cum.shape[0], np.nan)
182
+ b_neu = np.full(cum.shape[0], np.nan)
183
+ for i in range(cum.shape[0]):
184
+ pool = ev[i] & np.isfinite(cum[i])
185
+ for c in cvs:
186
+ pool = pool & np.isfinite(c[i])
187
+ n = int(pool.sum())
188
+ if n < min_obs:
189
+ continue
190
+ d = fire[i][pool].astype(np.float64)
191
+ if d.sum() < 3 or d.sum() > n - 3:
192
+ continue
193
+ y = cum[i][pool]
194
+ try:
195
+ b_raw[i] = np.linalg.lstsq(np.column_stack([np.ones(n), d]), y, rcond=None)[0][1]
196
+ X = np.column_stack([np.ones(n), d] + [c[i][pool] for c in cvs])
197
+ b_neu[i] = np.linalg.lstsq(X, y, rcond=None)[0][1]
198
+ except np.linalg.LinAlgError:
199
+ continue
200
+ m0, _, t0, _ = newey_west_t(b_raw, lag=max(h, 21))
201
+ m1, _, t1, _ = newey_west_t(b_neu, lag=max(h, 21))
202
+ keep = (m1 / m0 * 100) if (np.isfinite(m0) and abs(m0) > 1e-9) else np.nan
203
+ out[h] = {"raw_bp": m0 * 1e4 if np.isfinite(m0) else np.nan, "raw_t": t0,
204
+ "neutral_bp": m1 * 1e4 if np.isfinite(m1) else np.nan, "neutral_t": t1,
205
+ "survival_pct": float(keep) if np.isfinite(keep) else np.nan}
206
+ return out
@@ -0,0 +1,310 @@
1
+ """Whether a trade can be done, and at what size, at the price the backtest uses.
2
+
3
+ A daily backtest that trades every position change at the close assumes three things that are often false: that the security
4
+ was open (a halted name cannot be traded), that the price was not locked at a daily limit (a name closed at the upper limit has a
5
+ queue of buyers and no sellers, so you cannot buy it), and that you can hold any fraction of a share. This module states those
6
+ assumptions as data the engines read.
7
+
8
+ can_buy, can_sell `Panel` fields (bool, date x ticker). A trade that increases a position needs `can_buy` on the execution
9
+ day, one that decreases it needs `can_sell`. A blocked trade does not happen and the position stays as it was.
10
+ tradability() builds them from prices and volume: no price or no volume means halted; a daily move at the limit means locked.
11
+ at_prices() a copy of the panel that trades at another price (the open instead of the close).
12
+ realize() the engine's position path under those rules, plus whole-lot sizes and a minimum trade value.
13
+
14
+ Nothing here is specific to one market. Limits and lot sizes differ by market and change by rule, so they are arguments, not a table
15
+ inside the library.
16
+ """
17
+ from __future__ import annotations
18
+
19
+ from dataclasses import replace
20
+
21
+ import numpy as np
22
+ import pandas as pd
23
+
24
+ from .core.costs import rate_schedule
25
+ from .core.panel import Panel
26
+
27
+
28
+ def tradability(panel: Panel, *, limit=None, tol: float = 0.01, halt_on_zero_volume: bool = True,
29
+ max_gap_days: int | None = 60) -> tuple[pd.DataFrame, pd.DataFrame]:
30
+ """(can_buy, can_sell) built from the panel's prices and volume.
31
+
32
+ After a security's flagged delisting (`delist_after`) both trades are allowed, so that a position in it can be closed.
33
+ Halted: no close that day, or (if the panel has `volume` and `halt_on_zero_volume`) no volume. A security with a close but a
34
+ missing volume is also treated as halted, because an unknown is not permission. A panel without `volume` is judged by prices alone.
35
+ limit the daily price limit as a fraction (0.30 for 30%), a list of (effective_date, fraction) when it changed, or None for a
36
+ market without one. A day whose move is at least `limit - tol` up cannot be bought (a locked-up name has no sellers) and a
37
+ day at least `limit - tol` down cannot be sold. The tolerance is there because tick rounding keeps the limit move slightly
38
+ under the limit. This is **conservative**: it also blocks days that reached the limit but traded freely before it.
39
+ tol in fractions, 0 <= tol < limit.
40
+ max_gap_days after this many consecutive days **without any price**, a position is treated as settled and can be closed either way
41
+ (None: never). An exchange settles a delisted contract and a vendor gap or a ticker that came back years later (a contract
42
+ that was absent for years and then listed again, in the Binance cache) is not a position you can hold through. A name with a price
43
+ but no volume (a Korean trading suspension) is not covered: that one really is frozen, for as long as the data says."""
44
+ c = panel.close
45
+ ok = c.notna()
46
+ if halt_on_zero_volume and panel.volume is not None:
47
+ ok = ok & (panel.volume.fillna(0.0) > 0)
48
+ can_buy, can_sell = ok.copy(), ok.copy()
49
+ if panel.delist_after is not None and bool(np.asarray(panel.delist_after.values).any()):
50
+ # After a flagged delisting there is no price for ever. The engines already settle such a position (`delist_return`, or the last
51
+ # price), so closing it is allowed in **both** directions (a long is sold, a short is bought back): refusing the exit would freeze a
52
+ # dead position in the book for the rest of the sample, and a frozen short is exactly what the first real-data run showed.
53
+ da = panel.delist_after.reindex(index=c.index, columns=c.columns).fillna(False).astype(bool)
54
+ gone = (da.astype(int).cumsum() - da.astype(int)) > 0
55
+ can_buy, can_sell = can_buy | gone, can_sell | gone
56
+ if max_gap_days is not None:
57
+ isn = c.isna().to_numpy()
58
+ run = np.zeros(c.shape[1], dtype=int)
59
+ long_gap = np.zeros(c.shape, dtype=bool)
60
+ for t in range(c.shape[0]):
61
+ run = np.where(isn[t], run + 1, 0)
62
+ long_gap[t] = run > max_gap_days
63
+ settled = pd.DataFrame(long_gap, index=c.index, columns=c.columns)
64
+ can_buy, can_sell = can_buy | settled, can_sell | settled
65
+ if limit is not None:
66
+ lim = (rate_schedule(panel.dates, limit, "limit").to_numpy(float) if isinstance(limit, (list, tuple))
67
+ else np.full(len(panel.dates), float(limit)))
68
+ if not (np.all(lim > 0) and np.all(lim < 1)):
69
+ raise ValueError(f"limit must be a fraction between 0 and 1 (0.30 for 30%), got {limit!r}")
70
+ if not (0 <= tol < lim.min()):
71
+ raise ValueError(f"tol must be at least 0 and below the limit, got {tol!r}")
72
+ ret = c.pct_change(fill_method=None).to_numpy(float)
73
+ thr = (lim - tol)[:, None]
74
+ with np.errstate(invalid="ignore"):
75
+ up, down = ret >= thr, ret <= -thr
76
+ can_buy = can_buy & ~pd.DataFrame(up, index=c.index, columns=c.columns)
77
+ can_sell = can_sell & ~pd.DataFrame(down, index=c.index, columns=c.columns)
78
+ return can_buy, can_sell
79
+
80
+
81
+ def halt_runs(panel: Panel) -> np.ndarray:
82
+ """(date x ticker) int: how many consecutive days, counting the day itself, the security has had a price but no (or unknown) volume; 0 on a day it trades or has no price.
83
+ A panel without `volume` has none."""
84
+ c = panel.close
85
+ if panel.volume is None:
86
+ return np.zeros(c.shape, dtype=int)
87
+ halted = (c.notna() & ~(panel.volume.fillna(0.0) > 0)).to_numpy()
88
+ run = np.zeros(c.shape, dtype=int)
89
+ for t in range(c.shape[0]):
90
+ run[t] = np.where(halted[t], (run[t - 1] if t else 0) + 1, 0)
91
+ return run
92
+
93
+
94
+ def freeze_hits(panel: Panel, freeze_days: int | None) -> np.ndarray | None:
95
+ """(date x ticker) bool indexed by the **signal date**: True where the execution day (signal date + `entry_lag`) is exactly the `freeze_days`-th day of a suspension.
96
+ None when `freeze_days` is None. A suspension marks a position down once and not again however long it lasts. The markdown is booked on the signal-date row, which is
97
+ the `freeze_days`-th suspended day minus `entry_lag`, and it enters the return of that row (close of the execution day to the close of the next), so it is the
98
+ `freeze_days`-th suspended day's close that takes the loss."""
99
+ if freeze_days is None:
100
+ return None
101
+ if isinstance(freeze_days, bool) or not isinstance(freeze_days, (int, np.integer)) or freeze_days < 1:
102
+ raise ValueError(f"freeze_days must be a whole number of days, at least 1, or None, got {freeze_days!r}")
103
+ hit = halt_runs(panel) == freeze_days
104
+ lag = panel.entry_lag
105
+ return np.vstack([hit[lag:], np.zeros((lag, hit.shape[1]), bool)]) if lag else hit
106
+
107
+
108
+ def check_freeze_return(freeze_return: float) -> float:
109
+ """`freeze_return` as a float, or ValueError unless it is between -1 and 0 (a markdown is never a gain)."""
110
+ if isinstance(freeze_return, (bool, np.bool_)) or not isinstance(freeze_return, (int, float, np.integer, np.floating)):
111
+ raise ValueError(f"freeze_return must be a number between -1 and 0, got {freeze_return!r}")
112
+ if not (np.isfinite(freeze_return) and -1.0 <= freeze_return <= 0.0):
113
+ raise ValueError(f"freeze_return must be between -1 and 0 (a markdown is never a gain), got {freeze_return!r}")
114
+ return float(freeze_return)
115
+
116
+
117
+ def freeze_episodes(panel: Panel, *, min_days: int = 20, relist_days: int = 20) -> pd.DataFrame:
118
+ """The stretches of at least `min_days` consecutive days on which a security had a price but no volume (a trading suspension), and how each ended.
119
+
120
+ One row per episode: ticker, start, end (the last suspended day), days, outcome and `ret`, the return from the last price before the suspension
121
+ to the price at the end of the story.
122
+ A day without a price (NaN) ends a suspension, so a halt with a gap of missing prices inside it is cut into two episodes. A suspension that is already running
123
+ on the panel's first day has no price before it: such an episode is listed, but its `ret` is NaN, and its length is truncated.
124
+ outcome "resumed" the first priced day after the suspension, and the data goes on (or the security is not flagged in `delist_after`):
125
+ ret is that day's price over the price before the suspension. That day may itself have no volume if missing prices sit between.
126
+ "resumed_then_delisted" the security is flagged in `delist_after` and its last bar falls within `relist_days` of that day (a liquidation window): ret is its
127
+ last price over the price before. Without the flag the same story is labeled "resumed".
128
+ "ended_in_halt" the security's last bar is a suspended day and it is flagged in `delist_after`: the price never moved, ret is 0 (NaN if there is
129
+ no price before), and the loss, if any, was never in the data
130
+ "ongoing" no priced day follows: either still suspended on the last day of the panel, or the data ends in the suspension without a delisting
131
+ flag; ret is NaN
132
+ The point is to measure how often a position that cannot be sold is lost, instead of assuming it. Counts are small and depend on the market and
133
+ the period; read the distribution of `ret`, not a mean."""
134
+ c = panel.close
135
+ halted = (c.notna() & ~(panel.volume.fillna(0.0) > 0)).to_numpy() if panel.volume is not None else np.zeros(c.shape, bool)
136
+ px = c.to_numpy(float)
137
+ da = panel.delist_after.reindex(index=c.index, columns=c.columns).fillna(False).to_numpy(bool) if panel.delist_after is not None else np.zeros(c.shape, bool)
138
+ T, N = c.shape
139
+ rows = []
140
+ for j in range(N):
141
+ h = halted[:, j]
142
+ if not h.any():
143
+ continue
144
+ t = 0
145
+ valid = np.where(~np.isnan(px[:, j]))[0]
146
+ last_bar = int(valid[-1]) if len(valid) else -1
147
+ flagged = bool(da[:, j].any())
148
+ while t < T:
149
+ if not h[t]:
150
+ t += 1
151
+ continue
152
+ e = t
153
+ while e + 1 < T and h[e + 1]:
154
+ e += 1
155
+ if e - t + 1 >= min_days:
156
+ before = next((px[k, j] for k in range(t - 1, -1, -1) if not np.isnan(px[k, j]) and not h[k]), np.nan)
157
+ if e == T - 1 and not (flagged and last_bar == e):
158
+ outcome, ret = "ongoing", np.nan
159
+ elif flagged and last_bar == e:
160
+ outcome, ret = "ended_in_halt", 0.0 if np.isfinite(before) else np.nan
161
+ else:
162
+ nxt = e + 1
163
+ while nxt < T and np.isnan(px[nxt, j]):
164
+ nxt += 1
165
+ if nxt >= T:
166
+ outcome, ret = "ongoing", np.nan
167
+ elif flagged and last_bar - nxt <= relist_days:
168
+ outcome, ret = "resumed_then_delisted", px[last_bar, j] / before - 1.0 if np.isfinite(before) else np.nan
169
+ else:
170
+ outcome, ret = "resumed", px[nxt, j] / before - 1.0 if np.isfinite(before) else np.nan
171
+ rows.append({"ticker": c.columns[j], "start": c.index[t], "end": c.index[e], "days": e - t + 1, "outcome": outcome, "ret": ret})
172
+ t = e + 1
173
+ return pd.DataFrame(rows, columns=["ticker", "start", "end", "days", "outcome", "ret"])
174
+
175
+
176
+ def at_prices(panel: Panel, price="open") -> Panel:
177
+ """A copy of `panel` whose prices are another series, so that a position is entered and marked at it. `price` is the name of a panel
178
+ field (`"open"`, `"high"`, `"low"`) or a (date x ticker) frame. With the open, a signal from the close of day d is entered at the open
179
+ of d+lag and held to the open of d+lag+1. A price that is zero or negative is treated as missing. Only `close` changes: eligibility, volume and the rest were built from the original
180
+ closes, and `screen` and `backtest_event` still compute their controls from whatever `close` now holds."""
181
+ px = getattr(panel, price) if isinstance(price, str) else price
182
+ if px is None:
183
+ raise ValueError(f"the panel has no {price!r}")
184
+ if not isinstance(px, pd.DataFrame):
185
+ raise ValueError("price must be a field name or a (date x ticker) frame")
186
+ px = px.reindex(index=panel.dates, columns=panel.tickers)
187
+ return replace(panel, close=px.where(px > 0)) # a zero or negative price is no price: the engines treat a missing price as a return of 0, a zero would give infinity
188
+
189
+
190
+ def exec_masks(panel: Panel):
191
+ """(can_buy, can_sell) as arrays indexed by the **signal date**, i.e. read on the execution day d + entry_lag. None when the panel
192
+ has neither field. A missing frame counts as always possible."""
193
+ if panel.can_buy is None and panel.can_sell is None:
194
+ return None, None
195
+ lag = panel.entry_lag
196
+
197
+ def one(f):
198
+ if f is None:
199
+ return np.ones(panel.close.shape, dtype=bool)
200
+ return f.shift(-lag, fill_value=True).to_numpy(bool)
201
+ return one(panel.can_buy), one(panel.can_sell)
202
+
203
+
204
+ def round_to_lots(weights: np.ndarray, capital: float, price: np.ndarray, lot) -> np.ndarray:
205
+ """The weights you can actually hold: whole lots at the given prices. Shares = weight x capital / price, rounded to a multiple of
206
+ `lot`. `price` must be the **real** price level (not a back-adjusted series, whose level is arbitrary). A name with no price keeps weight NaN."""
207
+ with np.errstate(divide="ignore", invalid="ignore"):
208
+ shares = weights * capital / price
209
+ lots = np.round(shares / lot) * lot
210
+ return lots * price / capital
211
+
212
+
213
+ def realize(target: np.ndarray, can_buy: np.ndarray | None = None, can_sell: np.ndarray | None = None, *,
214
+ capital: float | None = None, price: np.ndarray | None = None, lot=1.0, min_trade_value: float = 0.0, cap_gross: bool = False):
215
+ """The positions that result from asking for `target` row by row (row 0 is a build from cash).
216
+
217
+ Each day, per name: round the target to whole lots (when `capital` is given; `price` is read on the **same row** as the target, so pass the price of the day the
218
+ trade is done: `backtest_weights` shifts it by `entry_lag`); skip the trade if its value is under `min_trade_value` (a number, or one per security);
219
+ skip it if it increases the position where `can_buy` is False, or decreases it where `can_sell` is False. A skipped trade leaves the
220
+ position as it was the day before. A name with no price on a day (NaN) keeps its position.
221
+ The weight is what the engines keep constant, so a day without a trade keeps the previous **weight**, not the previous share count: after the price moves, the
222
+ shares it implies need not be a multiple of the lot (a halted name does not move, so there it is).
223
+
224
+ `cap_gross`: a position that cannot be traded ties up capital. Without this, new targets are added on top of it and the gross exposure can exceed the
225
+ target's. With it, the day's whole target is scaled by a k in [0, 1], the largest one found by bisection for which the resulting gross exposure does not exceed
226
+ the gross of the day's target (after lot rounding, so rounding itself is never cut). A position that cannot be sold stays where it is; one that cannot be
227
+ bought stays too unless the scaled target falls below it, in which case it is sold down, so it can shrink. A scaled target can turn a buy into a sell, which
228
+ changes what is blocked, so k is searched for rather than solved for, and the feasible set being an interval [0, k] is not proven (no counter-example was found
229
+ in 653 random small cases). On a day with nothing blocked or skipped k is exactly 1 and the result is the same as without it, with or without `capital`. If the
230
+ stuck positions alone exceed the target gross, nothing can meet the cap: k is 0 and the day is counted in `infeasible_days`.
231
+
232
+ Returns (positions, stats). `mean_stuck_weight` is the average over days of the weight held where the target said otherwise because a trade was
233
+ blocked; `longest_freeze_days` the longest run of consecutive days a wanted trade in one name was blocked; `mean_free_scale` the average k (1 without `cap_gross`), `infeasible_days` the days on which the stuck positions alone exceeded the target gross."""
234
+ T, N = target.shape
235
+ use_lots = capital is not None
236
+ mtv = np.asarray(min_trade_value, dtype=float) # a number, or one per security
237
+ mtv_any = bool(np.any(mtv > 0))
238
+ out = np.zeros_like(target, dtype=float)
239
+ prev = np.zeros(N)
240
+ asked = blocked_turn = skipped_turn = gap = stuck = k_sum = 0.0
241
+ n_blocked = n_skipped = infeasible = 0
242
+ streak = np.zeros(N, dtype=int)
243
+ longest = 0
244
+
245
+ def decide(raw: np.ndarray, t: int):
246
+ tgt = raw
247
+ if use_lots:
248
+ lt = lot[t] if isinstance(lot, np.ndarray) and lot.ndim == 2 else lot
249
+ tgt = round_to_lots(raw, capital, price[t], lt)
250
+ # A flat target needs no price to carry out (a delisted name is simply closed); a non-flat target with no price cannot be sized, so the
251
+ # position stays.
252
+ tgt = np.where(raw == 0, 0.0, np.where(np.isfinite(tgt), tgt, prev))
253
+ d = tgt - prev
254
+ stay = np.zeros(N, dtype=bool)
255
+ if use_lots and mtv_any:
256
+ stay |= (np.abs(d) * capital < mtv) & (d != 0)
257
+ blk = np.zeros(N, dtype=bool)
258
+ if can_buy is not None:
259
+ blk |= (d > 0) & ~can_buy[t]
260
+ if can_sell is not None:
261
+ blk |= (d < 0) & ~can_sell[t]
262
+ return tgt, d, stay, blk
263
+
264
+ for t in range(T):
265
+ raw0 = target[t]
266
+ k = 1.0
267
+ tgt, d, stay, blk = decide(raw0, t)
268
+ if cap_gross:
269
+ budget = float(np.abs(tgt).sum()) + 1e-12 # what the day would hold with nothing in the way (after lot rounding)
270
+
271
+ def gross_at(kk: float) -> float:
272
+ g_tgt, _, g_stay, g_blk = decide(raw0 * kk, t)
273
+ return float(np.abs(np.where(g_stay | g_blk, prev, g_tgt)).sum())
274
+
275
+ if float(np.abs(np.where(stay | blk, prev, tgt)).sum()) > budget:
276
+ if gross_at(0.0) > budget:
277
+ k = 0.0
278
+ infeasible += 1
279
+ else:
280
+ lo, hi = 0.0, 1.0 # lo is always feasible, hi is not
281
+ while hi - lo > 1e-14:
282
+ mid = 0.5 * (lo + hi)
283
+ lo, hi = (mid, hi) if gross_at(mid) <= budget else (lo, mid)
284
+ k = lo
285
+ raw = raw0 * k
286
+ tgt, d, stay, blk = decide(raw, t)
287
+ else:
288
+ raw = raw0
289
+ else:
290
+ raw = raw0
291
+ k_sum += k
292
+ if use_lots:
293
+ gap += float(np.abs(tgt - raw).sum())
294
+ cur = np.where(stay | blk, prev, tgt)
295
+ held = blk & ~stay
296
+ stuck += float(np.abs(cur - tgt)[held].sum()) # weight sitting where it should not be, today
297
+ streak = np.where(held, streak + 1, 0)
298
+ longest = max(longest, int(streak.max())) if N else longest
299
+ if t > 0:
300
+ asked += float(np.abs(d).sum())
301
+ blocked_turn += float(np.abs(d[blk & ~stay]).sum())
302
+ skipped_turn += float(np.abs(d[stay]).sum())
303
+ n_blocked += int((blk & ~stay).sum())
304
+ n_skipped += int(stay.sum())
305
+ out[t] = cur
306
+ prev = cur
307
+ stats = {"mean_stuck_weight": stuck / T if T else 0.0, "longest_freeze_days": longest, "asked_turnover": asked, "blocked_turnover": blocked_turn, "blocked_trades": n_blocked, "blocked_turnover_share": blocked_turn / asked if asked > 0 else 0.0,
308
+ "min_trade_skipped": n_skipped, "min_trade_skipped_share": skipped_turn / asked if asked > 0 else 0.0,
309
+ "mean_abs_rounding_gap": gap / (T * N) if use_lots else 0.0, "mean_free_scale": k_sum / T if T else 1.0, "infeasible_days": infeasible}
310
+ return out, stats