pitbacktest 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,46 @@
1
+ """Which securities can be sold short on which dates.
2
+
3
+ A backtest that shorts a security on a day when short selling is banned, or when there is nothing to borrow, earns money that
4
+ was never available. `Panel.shortable` is the (date x ticker) bool frame the engines read; this module builds it from ban periods.
5
+
6
+ The library does not ship a calendar of bans. Dates, and which securities were exempt, differ by market and change by decision, and
7
+ a wrong table would be worse than none, so you pass the periods you have checked against the regulator's notice.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import numpy as np
12
+ import pandas as pd
13
+
14
+
15
+ def shortable_from_bans(dates: pd.DatetimeIndex, tickers: pd.Index, bans, exempt=None) -> pd.DataFrame:
16
+ """(dates x tickers) bool: True where short selling is allowed.
17
+
18
+ bans list of (start, end) pairs, both inclusive; `end=None` means the ban has not ended. Dates may be strings.
19
+ Periods may overlap. Outside every period everything is shortable.
20
+ exempt securities that may still be sold short during a ban (for example market-maker or index-constituent exceptions):
21
+ None (nobody), a list or set of tickers (exempt in every ban period), or a (dates x tickers) bool frame (True =
22
+ exempt that day; anything not in the frame is not exempt).
23
+
24
+ A partial ban (only some securities may be shorted again) is a ban for the others: pass the securities that may be shorted as
25
+ `exempt`. If you do not know them, do not pretend: treating the whole period as a full ban (no exempt) is the cautious choice (no short is
26
+ assumed possible). It is not a lower bound on the result: when the short leg was losing money, a ban makes the result better."""
27
+ dates = pd.DatetimeIndex(dates)
28
+ banned = pd.Series(False, index=dates)
29
+ for item in bans:
30
+ if len(item) != 2:
31
+ raise ValueError(f"each ban must be a (start, end) pair, got {item!r}")
32
+ start, end = pd.Timestamp(item[0]), (None if item[1] is None else pd.Timestamp(item[1]))
33
+ if end is not None and end < start:
34
+ raise ValueError(f"a ban ends before it starts: {item!r}")
35
+ banned |= (dates >= start) & (True if end is None else dates <= end)
36
+ out = pd.DataFrame(True, index=dates, columns=tickers)
37
+ if not banned.any():
38
+ return out
39
+ if exempt is None:
40
+ allowed = pd.DataFrame(False, index=dates, columns=tickers)
41
+ elif isinstance(exempt, pd.DataFrame):
42
+ allowed = exempt.reindex(index=dates, columns=tickers).fillna(False).astype(bool)
43
+ else:
44
+ allowed = pd.DataFrame(False, index=dates, columns=tickers)
45
+ allowed[[t for t in tickers if t in set(exempt)]] = True
46
+ return pd.DataFrame(np.where(banned.to_numpy()[:, None], allowed.to_numpy(bool), True), index=dates, columns=tickers)
@@ -0,0 +1,118 @@
1
+ """Overfitting checks for a set of tried strategies: deflated Sharpe, PBO (CSCV) and a permutation test.
2
+
3
+ All three answer the same question from different angles: *given that I tried many variants and kept the best one,
4
+ is the result still believable?* None of them proves a strategy works; each can only fail to reject.
5
+
6
+ Inputs are returns matrices of shape (T, N): T periods, N tried variants (parameter sets, factors, ...).
7
+ Everything is per-period unless `periods_per_year` is given.
8
+
9
+ References
10
+ Bailey, Lopez de Prado (2014), The Deflated Sharpe Ratio.
11
+ Bailey, Borwein, Lopez de Prado, Zhu (2015), The Probability of Backtest Overfitting (CSCV).
12
+ Masters, Monte Carlo permutation of the whole search.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import itertools
17
+ import math
18
+ from statistics import NormalDist
19
+ from typing import Callable
20
+
21
+ import numpy as np
22
+
23
+ _N = NormalDist()
24
+ _EULER = 0.5772156649015329
25
+
26
+
27
+ def sharpe(R: np.ndarray, periods_per_year: int = 1) -> np.ndarray:
28
+ """Sharpe ratio of each column (risk-free 0), annualised by sqrt(periods_per_year)."""
29
+ R = np.asarray(R, dtype=np.float64)
30
+ return R.mean(axis=0) / R.std(axis=0, ddof=1) * math.sqrt(periods_per_year)
31
+
32
+
33
+ def deflated_sharpe(R: np.ndarray, *, trials: int | None = None, periods_per_year: int = 1) -> dict:
34
+ """Probability that the best column's true Sharpe is above what the best of `trials` random strategies
35
+ would show by luck. `trials` defaults to the number of columns; pass a larger number if you also tried
36
+ variants you did not keep in R (and say so honestly)."""
37
+ R = np.asarray(R, dtype=np.float64)
38
+ T, n = R.shape
39
+ sr_all = R.mean(axis=0) / R.std(axis=0, ddof=1)
40
+ j = int(np.nanargmax(sr_all))
41
+ x = R[:, j]
42
+ sr = float(sr_all[j])
43
+ z = (x - x.mean()) / x.std(ddof=0)
44
+ skew, kurt = float((z ** 3).mean()), float((z ** 4).mean())
45
+ k = trials or n
46
+ var = float(np.nanvar(sr_all, ddof=1)) if n > 1 else 0.0
47
+ sr0 = math.sqrt(var) * ((1 - _EULER) * _N.inv_cdf(1 - 1 / k) + _EULER * _N.inv_cdf(1 - 1 / (k * math.e))) if k > 1 else 0.0
48
+ denom = math.sqrt(max(1e-12, 1 - skew * sr + (kurt - 1) / 4 * sr * sr))
49
+ dsr = _N.cdf((sr - sr0) * math.sqrt(T - 1) / denom)
50
+ a = math.sqrt(periods_per_year)
51
+ return {"best": j, "sharpe": sr * a, "sharpe_luck_benchmark": sr0 * a, "trials": k, "skew": skew,
52
+ "kurtosis": kurt, "T": T, "dsr": float(dsr)}
53
+
54
+
55
+ def pbo_cscv(R: np.ndarray, *, blocks: int = 16, periods_per_year: int = 1, max_splits: int | None = None,
56
+ seed: int = 0) -> dict:
57
+ """Probability of backtest overfitting by combinatorially symmetric cross-validation.
58
+
59
+ The sample is cut into `blocks` contiguous blocks; every way of choosing half of them as in-sample is tried.
60
+ PBO is the share of splits where the in-sample best column ranks at or below the median out of sample.
61
+ Around 0.5 means the in-sample winner carries no information about out-of-sample rank."""
62
+ R = np.asarray(R, dtype=np.float64)
63
+ T = (len(R) // blocks) * blocks
64
+ R = R[len(R) - T:]
65
+ L = T // blocks
66
+ S = np.stack([R[i * L:(i + 1) * L].sum(0) for i in range(blocks)])
67
+ Q = np.stack([(R[i * L:(i + 1) * L] ** 2).sum(0) for i in range(blocks)])
68
+
69
+ def sr(idx):
70
+ c = L * len(idx)
71
+ s, q = S[list(idx)].sum(0), Q[list(idx)].sum(0)
72
+ return (s / c) / np.sqrt((q - s * s / c) / (c - 1))
73
+
74
+ combos = list(itertools.combinations(range(blocks), blocks // 2))
75
+ if max_splits and len(combos) > max_splits:
76
+ rng = np.random.default_rng(seed)
77
+ combos = [combos[i] for i in rng.choice(len(combos), max_splits, replace=False)]
78
+ logits, is_best, oos_best = [], [], []
79
+ a = math.sqrt(periods_per_year)
80
+ for isb in combos:
81
+ osb = tuple(i for i in range(blocks) if i not in isb)
82
+ x, y = sr(isb), sr(osb)
83
+ j = int(np.nanargmax(x))
84
+ w = (1 + int((y < y[j]).sum())) / (len(y) + 1)
85
+ logits.append(math.log(w / (1 - w)))
86
+ is_best.append(float(x[j] * a))
87
+ oos_best.append(float(y[j] * a))
88
+ logits = np.asarray(logits)
89
+ return {"pbo": float((logits <= 0).mean()), "splits": len(logits), "is_best_sharpe_mean": float(np.mean(is_best)),
90
+ "oos_of_best_sharpe_mean": float(np.mean(oos_best))}
91
+
92
+
93
+ def permutation_test(returns: np.ndarray, search: Callable[[np.ndarray], float], *, n: int = 1000, seed: int = 0,
94
+ block: int = 1) -> dict:
95
+ """Monte Carlo permutation of the **whole search**.
96
+
97
+ `search(returns)` must run your entire selection procedure on a returns (or price-change) array and return
98
+ the best statistic it found (for example the best Sharpe over a parameter grid). The returns are shuffled
99
+ (in blocks of `block` periods, to keep some autocorrelation if you need it), the whole search is repeated, and
100
+ p = (1 + number of shuffles with a best statistic >= the real one) / (n + 1).
101
+ The null is "the timing has no value". Note that shuffling also destroys volatility clustering."""
102
+ r = np.asarray(returns, dtype=np.float64)
103
+ real = float(search(r))
104
+ rng = np.random.default_rng(seed)
105
+ nb = len(r) // block
106
+ cnt = 0
107
+ nulls = []
108
+ for _ in range(n):
109
+ if block == 1:
110
+ rs = r[rng.permutation(len(r))]
111
+ else:
112
+ order = rng.permutation(nb)
113
+ rs = np.concatenate([r[i * block:(i + 1) * block] for i in order] + [r[nb * block:]])
114
+ v = float(search(rs))
115
+ nulls.append(v)
116
+ cnt += v >= real
117
+ return {"real": real, "p": (1 + cnt) / (n + 1), "n": n, "null_mean": float(np.mean(nulls)),
118
+ "null_p95": float(np.percentile(nulls, 95))}
pitbacktest/weights.py ADDED
@@ -0,0 +1,271 @@
1
+ """Simulate portfolio weights you supply, with spread, borrow and market-impact costs, and read off capacity.
2
+
3
+ `backtest_portfolio` turns a factor into quantile portfolios. Institutions separate the steps: a signal, then a portfolio
4
+ construction step (an optimiser with risk and turnover limits), then a simulation. This module is the third step. It does
5
+ not build portfolios and does not ship an optimiser; give it the weights your optimiser produced.
6
+
7
+ Timing (the same as `backtest_portfolio`)
8
+ weights.loc[d] are target weights, as a fraction of capital, chosen with data up to the close of day d. They are entered at
9
+ close(d + lag), earn close(d + lag) -> close(d + lag + 1), and are replaced by the next row. Long is positive, short negative.
10
+ Gross exposure may exceed 1. Rows with no change cost nothing; the first row, a build from cash, is not charged (as in
11
+ `backtest_portfolio`, so the two agree exactly).
12
+
13
+ Costs, all charged on the signal date of the trade, in return per unit of capital
14
+ spread_bp scalar: round-trip spread, each unit traded pays half of it. Panel (date x ticker): one-way cost in bp.
15
+ The same units as `backtest_portfolio`.
16
+ buy_bp, sell_bp one-way cost per unit of weight bought / sold (bp, on top of the spread): a float, a Series by date (a rate that
17
+ changes over time) or a (date x ticker) frame. A weight going down is a sell, so opening a short pays sell_bp. A missing
18
+ value raises. The same as in `backtest_portfolio`.
19
+ borrow_bp annual borrow fee in bp (scalar or panel) on the short market value, charged per period.
20
+ impact square-root law, per unit traded: Y * sigma * sqrt(|trade| * AUM / ADV)
21
+ sigma the trailing daily volatility, ADV the trailing average dollar volume, both known at the signal date.
22
+ Y is of order 1 in the literature (Torre 1997; Almgren et al. 2005; Toth et al. 2011) but it is not known for a
23
+ given market, so run `capacity_curve` over several values instead of trusting one. Capital is taken to equal
24
+ AUM throughout (no compounding of the base).
25
+
26
+ You cannot open or increase a position in a security that is not eligible that day: an optimiser that buys names outside
27
+ the point-in-time universe is using information it should not have. A position already held may stay after the name leaves the
28
+ universe until you reduce it. Long and short exposure are checked separately, so turning a long into a smaller short in a name that is
29
+ not eligible is an increase of its short and raises.
30
+
31
+ Execution (what can actually be traded)
32
+ `panel.can_buy` / `panel.can_sell` (see `pitbacktest.execution`): a trade that increases a position on a day when it cannot be bought, or
33
+ decreases one where it cannot be sold, does not happen and the position stays. They are read on the execution day, `lag` after the signal.
34
+ `capital`, `price`, `lot`, `min_trade_value`: with `capital` given, each day's target is rounded to whole lots at the real prices and trades
35
+ worth less than `min_trade_value` are skipped, so that a small account does not hold 0.3 of a share or trade a few dollars.
36
+ The returned `holdings` are the positions after all of that.
37
+ If `panel.shortable` is given, opening or increasing a short in a security that cannot be sold short on the execution day (signal date +
38
+ `entry_lag`) raises as well.
39
+ """
40
+ from __future__ import annotations
41
+
42
+ from dataclasses import dataclass
43
+
44
+ import numpy as np
45
+ import pandas as pd
46
+
47
+ from .core.costs import apply_side_cost, apply_turnover_cost, side_cost_input
48
+ from .core.panel import Panel
49
+ from .execution import check_freeze_return, exec_masks, freeze_hits, realize
50
+ from .portfolio import (PortfolioResult, _benchmark_returns, _check_benchmark, _check_cost, _forward_arrays, _side_config,
51
+ _side_label, metrics)
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class ImpactModel:
56
+ aum: float # dollars of capital
57
+ y: float = 1.0 # coefficient of the square-root law
58
+ vol_window: int = 20
59
+ adv_window: int = 30
60
+ max_cost_bp: float = 100.0 # cap per unit traded; also what a name without volatility or volume history is charged
61
+
62
+ def __post_init__(self) -> None:
63
+ if not (self.aum >= 0 and np.isfinite(self.aum)):
64
+ raise ValueError(f"aum must be a finite number of dollars, not negative, got {self.aum!r}")
65
+ if not (self.y >= 0 and np.isfinite(self.y)):
66
+ raise ValueError(f"y must be finite and not negative, got {self.y!r}")
67
+ if self.vol_window < 2 or self.adv_window < 1:
68
+ raise ValueError("vol_window must be at least 2 and adv_window at least 1")
69
+ if not (self.max_cost_bp >= 0):
70
+ raise ValueError(f"max_cost_bp must not be negative, got {self.max_cost_bp!r}")
71
+
72
+
73
+ def _impact_cost(panel: Panel, H: np.ndarray, m: ImpactModel) -> tuple[np.ndarray, np.ndarray]:
74
+ """(cost per signal date, participation |trade| * AUM / ADV per date and name)."""
75
+ if panel.volume is None:
76
+ raise ValueError("market impact needs panel.volume (dollar volume = close * volume)")
77
+ sigma = panel.ret1().rolling(m.vol_window, min_periods=m.vol_window // 2).std().to_numpy(float)
78
+ adv = panel.adv(m.adv_window).to_numpy(float)
79
+ trade = np.zeros_like(H)
80
+ trade[1:] = np.abs(np.diff(H, axis=0))
81
+ with np.errstate(divide="ignore", invalid="ignore"):
82
+ part = np.where(adv > 0, trade * m.aum / adv, np.inf)
83
+ unit = m.y * sigma * np.sqrt(part)
84
+ cap = m.max_cost_bp / 1e4
85
+ unit = np.where(np.isfinite(unit), np.minimum(unit, cap), cap) # unknown or untradable: charged the cap
86
+ unit = np.where(trade > 0, unit, 0.0)
87
+ return (trade * unit).sum(axis=1), np.where(trade > 0, part, np.nan)
88
+
89
+
90
+ def backtest_weights(panel: Panel, weights: pd.DataFrame, *, spread_bp=0.0, buy_bp=0.0, sell_bp=0.0, borrow_bp=0.0,
91
+ impact: ImpactModel | None = None, funding: bool = True, delist_return: float | None = None,
92
+ benchmark: str | None = "cap", ledger=None, family: str = "default",
93
+ name: str | None = None, check_universe: bool = True, check_shortable: bool = True,
94
+ capital: float | None = None, price=None, lot=1.0, min_trade_value: float = 0.0,
95
+ freeze_days: int | None = None, freeze_return: float = 0.0, cap_gross: bool = False) -> PortfolioResult:
96
+ """Net returns of the weights you supply. See the module docstring for timing and costs.
97
+
98
+ `check_universe` raises if a long or a short position is opened or increased in a security that is not eligible that day. It looks at
99
+ net weights, so a name that is long from one signal and short from another (the two cancel) can show an increase when the
100
+ short expires; holdings built from overlapping long and short tranches (`PortfolioResult.holdings`) can trip it, and
101
+ then `check_universe=False` is the honest setting. Weights from an optimiser that nets positions are not affected.
102
+ `capital` (money, same currency as `price` and `ImpactModel.aum`), `price` (date x ticker **real** price level on the day of the row, as for the weights:
103
+ a back-adjusted series has an arbitrary level and gives wrong share counts; the engine sizes row t at `price` of day t + `entry_lag`, the day it is traded), `lot` (shares per lot: a number, a Series by ticker or a date x ticker frame) and
104
+ `min_trade_value` (money; a number or a Series by ticker) switch on whole-lot sizes; without `capital` they must be left alone.
105
+ `freeze_days`, `freeze_return`: a long position in a security whose suspension (a price but no volume) reaches `freeze_days` days is marked down once by
106
+ `freeze_return` (between -1 and 0): the loss is taken by the close of the `freeze_days`-th suspended day (booked on the signal row `entry_lag` days earlier); shorts are not credited. A scenario, not a measurement (see `execution.freeze_episodes`).
107
+ `cap_gross`: positions that cannot be traded tie up capital; the free names are scaled down so that the gross exposure stays at the target's (see `execution.realize`).
108
+ `check_shortable` (only with `panel.shortable`) raises if a short is opened or increased where the security cannot be sold short.
109
+
110
+ Costs are charged on the **net** trade per security. `backtest_portfolio` charges its long and short legs as separate sleeves,
111
+ so feeding it its own overlapping-tranche holdings can come out slightly cheaper here (a name long in one tranche and short in
112
+ another is netted); without such overlap the two agree exactly."""
113
+ _check_benchmark(benchmark)
114
+ _check_cost(spread_bp, "spread_bp")
115
+ _check_cost(borrow_bp, "borrow_bp")
116
+ fz = freeze_hits(panel, freeze_days)
117
+ freeze_return = check_freeze_return(freeze_return)
118
+ bb = side_cost_input(buy_bp, panel.dates, panel.tickers, "buy_bp")
119
+ sb = side_cost_input(sell_bp, panel.dates, panel.tickers, "sell_bp")
120
+ W = weights.reindex(index=panel.dates, columns=panel.tickers)
121
+ raw = W.to_numpy(float)
122
+ if np.isinf(raw).any():
123
+ raise ValueError("weights contain inf")
124
+ n_nan = int(np.isnan(raw).sum())
125
+ H = np.nan_to_num(raw, nan=0.0)
126
+ prev = np.vstack([np.zeros((1, H.shape[1])), H[:-1]])
127
+ more_long = np.maximum(H, 0.0) > np.maximum(prev, 0.0) + 1e-12
128
+ more_short = np.maximum(-H, 0.0) > np.maximum(-prev, 0.0) + 1e-12
129
+ bad = (more_long | more_short) & ~panel.eligible.to_numpy(bool) # opening or increasing outside the universe
130
+ if check_universe and bad.any():
131
+ i, j = np.argwhere(bad)[0]
132
+ raise ValueError(f"{int(bad.sum())} positions opened or increased on securities that are not eligible that day "
133
+ f"(first: {panel.tickers[j]} on {panel.dates[i].date()}); the universe is point in time")
134
+ if check_shortable and panel.shortable is not None:
135
+ nos = more_short & ~panel.shortable.shift(-panel.entry_lag, fill_value=True).to_numpy(bool) # read on the execution day, like can_sell
136
+ if nos.any():
137
+ i, j = np.argwhere(nos)[0]
138
+ raise ValueError(f"{int(nos.sum())} short positions opened or increased on securities that cannot be sold short on the execution day "
139
+ f"(first: {panel.tickers[j]} on {panel.dates[i].date()}); see Panel.shortable")
140
+ bo, so = exec_masks(panel)
141
+ ex = None
142
+ if isinstance(min_trade_value, pd.Series):
143
+ mtv = min_trade_value.reindex(panel.tickers).to_numpy(float)
144
+ if np.isnan(mtv).any():
145
+ raise ValueError("min_trade_value must be known for every security when it is given per security")
146
+ else:
147
+ mtv = float(min_trade_value)
148
+ if capital is None and (price is not None or np.any(np.asarray(mtv) > 0) or (np.ndim(lot) or float(lot) != 1.0)):
149
+ raise ValueError("price, lot and min_trade_value need capital: whole-lot sizes depend on how much money there is")
150
+ if capital is not None:
151
+ if not (np.isfinite(capital) and capital > 0):
152
+ raise ValueError(f"capital must be a positive number, got {capital!r}")
153
+ if price is None:
154
+ raise ValueError("capital needs price, the real price level in the same currency (not a back-adjusted series)")
155
+ if not (np.all(np.isfinite(mtv)) and np.all(np.asarray(mtv) >= 0)):
156
+ raise ValueError(f"min_trade_value must be at least 0, got {min_trade_value!r}")
157
+ if not isinstance(price, pd.DataFrame):
158
+ raise ValueError("price must be a (date x ticker) frame")
159
+ # Row t of the weights is traded at close(t + lag), so it is sized at that day's price. The last `lag` rows have no such day: unpriced, kept.
160
+ px = price.reindex(index=panel.dates, columns=panel.tickers).shift(-panel.entry_lag).to_numpy(float)
161
+ if isinstance(lot, pd.DataFrame):
162
+ lot_ = lot.reindex(index=panel.dates, columns=panel.tickers).to_numpy(float)
163
+ elif isinstance(lot, pd.Series):
164
+ lot_ = lot.reindex(panel.tickers).to_numpy(float)
165
+ else:
166
+ lot_ = float(lot)
167
+ if np.isnan(lot_).any() or not (np.asarray(lot_) > 0).all():
168
+ raise ValueError("lot must be positive and known for every security")
169
+ H_in = H # what was asked for; `H` becomes what could be held
170
+ if bo is not None or capital is not None:
171
+ H, ex = realize(H, bo, so, capital=capital, price=px if capital is not None else None,
172
+ lot=lot_ if capital is not None else 1.0, min_trade_value=mtv, cap_gross=cap_gross)
173
+ fwd, fwdf, hit_next = _forward_arrays(panel, funding, delist_return)
174
+ cut = len(panel.dates) - (panel.entry_lag + 1)
175
+ gross = (H * fwd).sum(axis=1)
176
+ mark = (np.maximum(H, 0.0) * fz).sum(axis=1) * freeze_return if fz is not None else np.zeros(len(gross))
177
+ gross = gross + mark
178
+ spread = apply_turnover_cost(H, spread_bp) # the first row, a build from cash, is not charged
179
+ side = apply_side_cost(H, bb, sb)
180
+ short = np.maximum(-H, 0.0)
181
+ b = np.asarray(borrow_bp, float)
182
+ borrow = (short * (b / 1e4 / panel.periods_per_year)).sum(axis=1) if b.ndim == 0 else \
183
+ (short * (np.nan_to_num(b, nan=0.0) / 1e4 / panel.periods_per_year)).sum(axis=1)
184
+ fcost = (H * fwdf).sum(axis=1)
185
+ imp, part = (np.zeros(len(H)), None)
186
+ if impact is not None:
187
+ imp, part = _impact_cost(panel, H, impact)
188
+ net = gross - spread - side - borrow - imp - fcost
189
+
190
+ ppy = panel.periods_per_year
191
+ m = metrics(net[:cut], panel.dates, ppy)
192
+ tr = np.zeros_like(H)
193
+ tr[1:] = np.abs(np.diff(H, axis=0))
194
+ m.update({"turnover_daily": float(0.5 * tr[:cut].sum(axis=1).mean()), "gross_CAGR": metrics(gross[:cut], panel.dates, ppy)["CAGR"],
195
+ "spread_annual_bp": float(spread[:cut].mean() * ppy * 1e4), "borrow_annual_bp": float(borrow[:cut].mean() * ppy * 1e4),
196
+ "side_cost_annual_bp": float(side[:cut].mean() * ppy * 1e4),
197
+ "impact_annual_bp": float(imp[:cut].mean() * ppy * 1e4), "funding_annual_bp": float(fcost[:cut].mean() * ppy * 1e4),
198
+ "avg_gross_exposure": float(np.abs(H[:cut]).sum(axis=1).mean()), "avg_net_exposure": float(H[:cut].sum(axis=1).mean()),
199
+ "delist_events_held": int(((H != 0) & hit_next)[:cut].sum()), "nan_weights_treated_as_zero": n_nan})
200
+ if fz is not None:
201
+ m["freeze_markdown_annual_bp"] = float(mark[:cut].mean() * ppy * 1e4)
202
+ m["freeze_markdown_events"] = int(((H > 0) & fz)[:cut].sum())
203
+ if ex is not None:
204
+ m["blocked_trades"] = int(ex["blocked_trades"])
205
+ m["blocked_turnover_share"] = float(ex["blocked_turnover"] / ex["asked_turnover"]) if ex["asked_turnover"] > 0 else 0.0
206
+ m["mean_stuck_weight"] = float(ex["mean_stuck_weight"])
207
+ m["longest_freeze_days"] = int(ex["longest_freeze_days"])
208
+ if cap_gross:
209
+ m["mean_free_scale"] = float(ex["mean_free_scale"])
210
+ m["cap_infeasible_days"] = int(ex["infeasible_days"])
211
+ if capital is not None:
212
+ m.update({"min_trade_skipped": int(ex["min_trade_skipped"]), "min_trade_skipped_share": float(ex["min_trade_skipped_share"]),
213
+ "mean_abs_rounding_gap": float(ex["mean_abs_rounding_gap"])})
214
+ if part is not None:
215
+ p = part[:cut][np.isfinite(part[:cut])]
216
+ m["participation_p99"] = float(np.percentile(p, 99)) if len(p) else float("nan")
217
+ m["participation_max"] = float(p.max()) if len(p) else float("nan")
218
+ m["trades_over_10pct_adv"] = float((p > 0.10).mean()) if len(p) else float("nan")
219
+ bench = bexc = bret = None
220
+ if benchmark:
221
+ br = _benchmark_returns(panel, fwd, benchmark)
222
+ bench = metrics(br[:cut], panel.dates, ppy)
223
+ bexc = metrics((net - br)[:cut], panel.dates, ppy)
224
+ bret = pd.Series(br[:cut], index=panel.dates[:cut], name="benchmark")
225
+ s = pd.Series(net[:cut], index=panel.dates[:cut])
226
+ yr = s.groupby(s.index.year).apply(lambda g: float((1 + g).prod() - 1))
227
+ spec = {"kind": "weights", "entry_lag": panel.entry_lag, "market": panel.market, "periods_per_year": ppy,
228
+ "spread": "panel" if np.ndim(spread_bp) else f"{spread_bp}bp round trip",
229
+ "buy_bp": _side_label(bb), "sell_bp": _side_label(sb), "borrow_bp": "panel" if b.ndim else float(b),
230
+ "impact": None if impact is None else {"aum": impact.aum, "y": impact.y, "vol_window": impact.vol_window,
231
+ "adv_window": impact.adv_window, "max_cost_bp": impact.max_cost_bp},
232
+ "funding": bool(funding and panel.funding is not None), "delist_return": delist_return,
233
+ "freeze": None if freeze_days is None else {"days": int(freeze_days), "return": freeze_return},
234
+ "execution": None if (bo is None and capital is None) else {"blocked": bo is not None, "capital": capital,
235
+ "min_trade_value": float(mtv) if np.ndim(mtv) == 0 else "by security"}}
236
+ if ledger is not None:
237
+ from .ledger import array_fingerprint
238
+ ledger.record(family, name or "weights", s, {**{k: v for k, v in spec.items() if k not in ("buy_bp", "sell_bp", "execution", "freeze")}, **_side_config(bb, sb), **_exec_config(capital, price, lot, mtv),
239
+ **({"freeze_days": int(freeze_days), "freeze_return": freeze_return} if freeze_days is not None else {}), **({"cap_gross": True} if cap_gross else {}),
240
+ "spread_bp": spread_bp if np.ndim(spread_bp) == 0 else array_fingerprint(spread_bp),
241
+ "weights": array_fingerprint(H_in), "data": panel.fingerprint()})
242
+ return PortfolioResult(spec=spec, metrics=m, benchmark=bench, excess=bexc, yearly={str(k): v for k, v in yr.items()}, grid=None,
243
+ holdings=H, net_returns=s, benchmark_returns=bret)
244
+
245
+
246
+ def _exec_config(capital, price, lot, mtv) -> dict:
247
+ """Ledger configuration of the execution settings; empty when none is used, so older runs keep their fingerprint."""
248
+ if capital is None:
249
+ return {}
250
+ from .ledger import array_fingerprint
251
+ return {"capital": float(capital), "min_trade_value": float(mtv) if np.ndim(mtv) == 0 else array_fingerprint(np.asarray(mtv)),
252
+ "price": array_fingerprint(price.to_numpy(float)),
253
+ "lot": float(lot) if np.ndim(lot) == 0 else array_fingerprint(np.asarray(lot, dtype=float))}
254
+
255
+
256
+ def capacity_curve(panel: Panel, weights: pd.DataFrame, aums, *, y_values=(1.0,), max_cost_bp: float = 100.0,
257
+ **kw) -> pd.DataFrame:
258
+ """Net Sharpe, CAGR and costs for each AUM (and each impact coefficient `y`). Extra keyword arguments go to
259
+ `backtest_weights` (spread, borrow, funding, ...). Read the AUM where the net CAGR or Sharpe falls to the level you can
260
+ accept; do not read a single capacity number, because Y is uncertain by a factor of a few and costs grow with sqrt(AUM).
261
+ `max_cost_bp` caps the impact per unit traded (default 100 bp); where it binds, cost grows more slowly than the law says,
262
+ which flatters very large AUM, so check `participation_p99` and `trades_over_10pct_adv` next to the cost."""
263
+ rows = []
264
+ for y in y_values:
265
+ for a in aums:
266
+ r = backtest_weights(panel, weights, impact=ImpactModel(aum=float(a), y=float(y), max_cost_bp=max_cost_bp), benchmark=None, **kw)
267
+ m = r.metrics
268
+ rows.append({"y": y, "aum": a, "sharpe": m["Sharpe"], "cagr": m["CAGR"], "gross_cagr": m["gross_CAGR"],
269
+ "impact_annual_bp": m["impact_annual_bp"], "spread_annual_bp": m["spread_annual_bp"],
270
+ "participation_p99": m["participation_p99"], "trades_over_10pct_adv": m["trades_over_10pct_adv"]})
271
+ return pd.DataFrame(rows)