pitbacktest 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pitbacktest/__init__.py +27 -0
- pitbacktest/adapters/__init__.py +0 -0
- pitbacktest/adapters/krx.py +249 -0
- pitbacktest/adapters/long_format.py +87 -0
- pitbacktest/adapters/tiingo.py +236 -0
- pitbacktest/adapters/yfinance.py +243 -0
- pitbacktest/analytics.py +234 -0
- pitbacktest/core/__init__.py +0 -0
- pitbacktest/core/controls.py +126 -0
- pitbacktest/core/costs.py +144 -0
- pitbacktest/core/estimators.py +216 -0
- pitbacktest/core/gates.py +175 -0
- pitbacktest/core/panel.py +306 -0
- pitbacktest/crypto/__init__.py +12 -0
- pitbacktest/crypto/binance_archive.py +287 -0
- pitbacktest/crypto/costs.py +80 -0
- pitbacktest/crypto/intraday.py +326 -0
- pitbacktest/crypto/panel.py +120 -0
- pitbacktest/equity/__init__.py +10 -0
- pitbacktest/equity/master.py +60 -0
- pitbacktest/equity/scenarios.py +104 -0
- pitbacktest/event.py +206 -0
- pitbacktest/execution.py +310 -0
- pitbacktest/ledger.py +128 -0
- pitbacktest/portfolio.py +373 -0
- pitbacktest/screen.py +150 -0
- pitbacktest/shorting.py +46 -0
- pitbacktest/validation.py +118 -0
- pitbacktest/weights.py +271 -0
- pitbacktest-0.2.0.dist-info/METADATA +395 -0
- pitbacktest-0.2.0.dist-info/RECORD +34 -0
- pitbacktest-0.2.0.dist-info/WHEEL +5 -0
- pitbacktest-0.2.0.dist-info/licenses/LICENSE +21 -0
- pitbacktest-0.2.0.dist-info/top_level.txt +1 -0
pitbacktest/__init__.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""pitbacktest: a backtest toolkit that measures its own biases (US, Korea, crypto).
|
|
2
|
+
|
|
3
|
+
Three entry points
|
|
4
|
+
screen() A. Factor screening with neutralisation (research stage)
|
|
5
|
+
backtest_event() C. Event signals (alert products)
|
|
6
|
+
backtest_portfolio() B. Portfolio alpha (CAGR, MDD, Sharpe)
|
|
7
|
+
|
|
8
|
+
Design principles
|
|
9
|
+
- The core is plain pandas and numpy; data sources live in `adapters`.
|
|
10
|
+
- There is one timing convention and it is enforced with an assert.
|
|
11
|
+
- Controls for firm characteristics are the default.
|
|
12
|
+
- The statistics that decide a verdict are computed with controls; uncontrolled figures are returned next to them, not in their place.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from .core.panel import Panel, build_pit_eligible
|
|
16
|
+
from .core.gates import GateConfig
|
|
17
|
+
from .event import backtest_event
|
|
18
|
+
from .portfolio import backtest_portfolio, assert_timing
|
|
19
|
+
from .screen import screen
|
|
20
|
+
from . import validation, analytics, execution
|
|
21
|
+
from .ledger import Ledger
|
|
22
|
+
from .weights import backtest_weights, capacity_curve, ImpactModel
|
|
23
|
+
from .shorting import shortable_from_bans
|
|
24
|
+
|
|
25
|
+
__version__ = "0.2.0"
|
|
26
|
+
__all__ = ["shortable_from_bans", "execution", "validation", "analytics", "Ledger", "backtest_weights", "capacity_curve", "ImpactModel", "Panel", "build_pit_eligible", "GateConfig", "screen",
|
|
27
|
+
"backtest_event", "backtest_portfolio", "assert_timing"]
|
|
File without changes
|
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
"""Korean equities from the official KRX OpenAPI, point in time and survivorship-free.
|
|
2
|
+
|
|
3
|
+
How it removes survivorship bias: the API returns **every stock that was listed on a given date**, including the ones
|
|
4
|
+
delisted or merged later (on 2015-01-02, 125 of the 899 KOSPI names are gone by 2024). Fetching day by day therefore
|
|
5
|
+
builds the universe the way it really was.
|
|
6
|
+
|
|
7
|
+
How it handles corporate actions without a price-adjustment table: each row carries the change versus the *reference
|
|
8
|
+
price* (`CMPPREVDD_PRC`), and the reference price already reflects splits and rights issues. So
|
|
9
|
+
`close / (close - change) - 1` is the adjusted return. It matched the published return to within rounding for all 953
|
|
10
|
+
KOSPI stocks on 2024-01-02, and gives -2.08% (not -98%) for Samsung Electronics on its 50:1 split day (2018-05-04).
|
|
11
|
+
The adjusted close used by the panel is the cumulative product of those returns. Dividends are not included (price
|
|
12
|
+
return only).
|
|
13
|
+
|
|
14
|
+
You need a KRX OpenAPI key (free, the services have to be approved on openapi.krx.co.kr). Put it in the environment as
|
|
15
|
+
KRX_OPENAPI_KEY or pass `key=`; `load_key(env_path)` reads only that one variable from a .env file.
|
|
16
|
+
|
|
17
|
+
fetch_days("2013-01-01", "2026-10-06", "~/.cache/quantbt/krx") # resumable, about 2 calls per trading day
|
|
18
|
+
panel = build_krx_panel("~/.cache/quantbt/krx", start="2014-01-01")
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import re
|
|
25
|
+
import time
|
|
26
|
+
import urllib.error
|
|
27
|
+
import urllib.request
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
import numpy as np
|
|
31
|
+
import pandas as pd
|
|
32
|
+
|
|
33
|
+
from ..core.panel import Panel
|
|
34
|
+
|
|
35
|
+
BASE = "https://data-dbg.krx.co.kr/svc/apis"
|
|
36
|
+
ENDPOINTS = {"KOSPI": "sto/stk_bydd_trd", "KOSDAQ": "sto/ksq_bydd_trd"}
|
|
37
|
+
RENAME = {"ISU_CD": "code", "ISU_NM": "name", "MKT_NM": "market", "TDD_CLSPRC": "close", "CMPPREVDD_PRC": "change",
|
|
38
|
+
"TDD_OPNPRC": "open", "TDD_HGPRC": "high", "TDD_LWPRC": "low", "ACC_TRDVOL": "volume", "ACC_TRDVAL": "value",
|
|
39
|
+
"MKTCAP": "mktcap", "LIST_SHRS": "shares"}
|
|
40
|
+
# Not tradable common stock for a cross-sectional study: SPACs, REITs and funds listed as stocks.
|
|
41
|
+
# Korean name patterns that mark SPACs (스팩), REITs (리츠), numbered funds (N호), and infrastructure funds (인프라, 맥쿼리). These must stay Korean:
|
|
42
|
+
# they are matched against the names KRX returns.
|
|
43
|
+
DEFAULT_EXCLUDE = re.compile(r"스팩|리츠|\d+호|인프라|맥쿼리")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class QuotaExceeded(RuntimeError):
|
|
47
|
+
"""The API refused further calls (daily limit). Run again later; the cache keeps what was fetched."""
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def load_key(env_path: str | Path | None = None) -> str:
|
|
51
|
+
"""KRX_OPENAPI_KEY from the environment, else from a .env file. Only that one variable is read."""
|
|
52
|
+
if os.environ.get("KRX_OPENAPI_KEY"):
|
|
53
|
+
return os.environ["KRX_OPENAPI_KEY"]
|
|
54
|
+
if env_path:
|
|
55
|
+
for line in Path(env_path).expanduser().read_text().splitlines():
|
|
56
|
+
k, _, v = line.partition("=")
|
|
57
|
+
if k.strip() == "KRX_OPENAPI_KEY":
|
|
58
|
+
return v.strip().strip('"').strip("'")
|
|
59
|
+
raise KeyError("KRX_OPENAPI_KEY is not set")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _call(path: str, date: str, key: str, retries: int = 4) -> list[dict]:
|
|
63
|
+
req = urllib.request.Request(f"{BASE}/{path}?basDd={date}", headers={"AUTH_KEY": key, "User-Agent": "pitbacktest-research/0.2"})
|
|
64
|
+
delay = 1.0
|
|
65
|
+
for i in range(retries + 1):
|
|
66
|
+
try:
|
|
67
|
+
with urllib.request.urlopen(req, timeout=40) as r:
|
|
68
|
+
return json.loads(r.read()).get("OutBlock_1", [])
|
|
69
|
+
except urllib.error.HTTPError as e:
|
|
70
|
+
body = e.read()[:200].decode("utf-8", "replace")
|
|
71
|
+
if e.code in (429, 403) or "limit" in body.lower() or "초과" in body: # "초과" = "exceeded", the wording of the KRX quota message
|
|
72
|
+
raise QuotaExceeded(f"HTTP {e.code}: {body}") from None
|
|
73
|
+
if e.code >= 500 and i < retries:
|
|
74
|
+
time.sleep(delay); delay *= 2
|
|
75
|
+
continue
|
|
76
|
+
raise
|
|
77
|
+
except (urllib.error.URLError, TimeoutError, ConnectionError, ValueError): # ValueError: an empty or non-JSON body
|
|
78
|
+
if i < retries:
|
|
79
|
+
time.sleep(delay); delay *= 2
|
|
80
|
+
continue
|
|
81
|
+
raise
|
|
82
|
+
return []
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _frame(rows: list[dict]) -> pd.DataFrame:
|
|
86
|
+
df = pd.DataFrame(rows).rename(columns=RENAME)[list(RENAME.values())]
|
|
87
|
+
for c in ("close", "change", "open", "high", "low", "volume", "value", "mktcap", "shares"):
|
|
88
|
+
df[c] = pd.to_numeric(df[c].replace({"-": np.nan, "": np.nan}), errors="coerce")
|
|
89
|
+
return df
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def fetch_days(start, end, store_dir, *, key: str | None = None, markets=("KOSPI", "KOSDAQ"), sleep: float = 0.1,
|
|
93
|
+
max_calls: int | None = None, caller=None, progress: bool = True) -> dict:
|
|
94
|
+
"""Download one file per weekday into `store_dir`. Resumable: finished days (and holidays) are skipped.
|
|
95
|
+
`caller(path, date)` can be injected for tests. Stops cleanly if the API reports a quota limit."""
|
|
96
|
+
store = Path(store_dir).expanduser()
|
|
97
|
+
store.mkdir(parents=True, exist_ok=True)
|
|
98
|
+
key = key or (None if caller else load_key())
|
|
99
|
+
call = caller or (lambda p, d: _call(p, d, key))
|
|
100
|
+
done = {"days": 0, "holidays": 0, "skipped": 0, "calls": 0, "stopped": None}
|
|
101
|
+
for ts in pd.bdate_range(start, end):
|
|
102
|
+
d = ts.strftime("%Y%m%d")
|
|
103
|
+
if (store / f"{d}.pkl").exists() or (store / f"{d}.hol").exists():
|
|
104
|
+
done["skipped"] += 1
|
|
105
|
+
continue
|
|
106
|
+
if max_calls is not None and done["calls"] + len(markets) > max_calls:
|
|
107
|
+
done["stopped"] = "max_calls"
|
|
108
|
+
break
|
|
109
|
+
frames = []
|
|
110
|
+
try:
|
|
111
|
+
for m in markets:
|
|
112
|
+
rows = call(ENDPOINTS[m], d)
|
|
113
|
+
done["calls"] += 1
|
|
114
|
+
if rows:
|
|
115
|
+
frames.append(_frame(rows))
|
|
116
|
+
time.sleep(sleep)
|
|
117
|
+
except QuotaExceeded as e:
|
|
118
|
+
done["stopped"] = f"quota: {e}"
|
|
119
|
+
break
|
|
120
|
+
if frames and len(frames) < len(markets) and caller is None:
|
|
121
|
+
done.setdefault("partial", []).append(d) # one market answered empty: do not save, retry next run
|
|
122
|
+
continue
|
|
123
|
+
if frames:
|
|
124
|
+
pd.concat(frames).to_pickle(store / f"{d}.pkl")
|
|
125
|
+
done["days"] += 1
|
|
126
|
+
else:
|
|
127
|
+
(store / f"{d}.hol").write_text("") # holiday: both markets empty
|
|
128
|
+
done["holidays"] += 1
|
|
129
|
+
if progress and (done["days"] + done["holidays"]) % 100 == 0:
|
|
130
|
+
print(f" {d} {done}", flush=True)
|
|
131
|
+
return done
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def build_krx_panel(store_dir, *, start=None, end=None, min_age_days: int = 60, adv_window: int = 30,
|
|
135
|
+
min_value_krw: float = 1e9, common_only: bool = True, exclude=DEFAULT_EXCLUDE,
|
|
136
|
+
reuse_gap_days: int = 120, entry_lag: int = 1, drop_suspect_above: float | None = None) -> Panel:
|
|
137
|
+
"""Point-in-time Panel of Korean stocks from the cached daily files.
|
|
138
|
+
|
|
139
|
+
eligible on day t uses data up to t only: at least `min_age_days` bars of history, trailing median traded value of
|
|
140
|
+
at least `min_value_krw`, positive volume that day, common shares only (code ends in 0) and no SPAC or REIT names.
|
|
141
|
+
A code that disappears for more than `reuse_gap_days` and comes back is treated as a different security (suffix #2).
|
|
142
|
+
|
|
143
|
+
Known defect of the return rule. The adjusted return is `close / (close - change) - 1`, which is right when the exchange's reference price already
|
|
144
|
+
reflects a split or rights issue. When a security is **suspended and a consolidation or capital reduction happens meanwhile**, the first day of trading
|
|
145
|
+
again carries a `change` against the old, unadjusted close, and the formula returns a move of thousands of percent that never happened (one case in the
|
|
146
|
+
cache: 2,080 won to 625,000 won, +29,948%). Every daily return beyond +-100% is listed in `meta["suspect_returns"]` (ticker, date, return, and whether the
|
|
147
|
+
previous day had zero or missing volume): 38 of 7.8 million in the whole cache (from 2013; the largest is +6,699,900%), 14 of them on such a day, and 29 and 10 since June 2015.
|
|
148
|
+
All are upward: a return cannot go below -100%. `drop_suspect_above=1.0` treats
|
|
149
|
+
such a day as a return of 0 (the price chain is cut there, the true move is unknown); the default None leaves the series as it was, so earlier results
|
|
150
|
+
reproduce. The strategies in this repository do not hold suspended names (eligibility needs volume), so none of them was affected; a position that is
|
|
151
|
+
**stuck** in a suspended name (`Panel.can_buy`/`can_sell`, `freeze_*`) can be."""
|
|
152
|
+
store = Path(store_dir).expanduser()
|
|
153
|
+
files = sorted(store.glob("*.pkl"))
|
|
154
|
+
if start:
|
|
155
|
+
files = [f for f in files if f.stem >= pd.Timestamp(start).strftime("%Y%m%d")]
|
|
156
|
+
if end:
|
|
157
|
+
files = [f for f in files if f.stem <= pd.Timestamp(end).strftime("%Y%m%d")]
|
|
158
|
+
if not files:
|
|
159
|
+
raise ValueError("no cached days: run fetch_days() first")
|
|
160
|
+
parts = []
|
|
161
|
+
for f in files:
|
|
162
|
+
d = pd.read_pickle(f)
|
|
163
|
+
d["date"] = pd.Timestamp(f.stem)
|
|
164
|
+
parts.append(d)
|
|
165
|
+
df = pd.concat(parts, ignore_index=True).sort_values(["code", "date"])
|
|
166
|
+
gap = df.groupby("code")["date"].diff().dt.days
|
|
167
|
+
life = (gap > reuse_gap_days).groupby(df["code"]).cumsum()
|
|
168
|
+
df["sid"] = np.where(life == 0, df["code"], df["code"] + "#" + (life + 1).astype(str))
|
|
169
|
+
names = df.drop_duplicates("sid", keep="last").set_index("sid")["name"].to_dict()
|
|
170
|
+
market_names = sorted(df["market"].dropna().unique())
|
|
171
|
+
df["_mk"] = df["market"].map({m: i for i, m in enumerate(market_names)}).astype("float64")
|
|
172
|
+
|
|
173
|
+
def mat(c: str) -> pd.DataFrame:
|
|
174
|
+
return df.pivot(index="date", columns="sid", values=c).sort_index()
|
|
175
|
+
|
|
176
|
+
close, chg = mat("close"), mat("change")
|
|
177
|
+
ref = close - chg
|
|
178
|
+
ret = (close / ref - 1.0).where(ref > 0)
|
|
179
|
+
first = close.notna() & ~close.notna().cumsum().shift(1, fill_value=0).astype(bool) # a security's first bar
|
|
180
|
+
ret = ret.mask(first) # the listing-day move is not tradable
|
|
181
|
+
vol_prev0 = (mat("volume").fillna(0) <= 0).shift(1, fill_value=False)
|
|
182
|
+
sus = ret.stack()
|
|
183
|
+
sus = sus[sus.abs() > 1.0]
|
|
184
|
+
suspect = pd.DataFrame({"ticker": sus.index.get_level_values(1), "date": sus.index.get_level_values(0), "ret": sus.to_numpy(),
|
|
185
|
+
"after_suspension": [bool(vol_prev0.loc[d, t]) for d, t in sus.index]}).reset_index(drop=True)
|
|
186
|
+
if drop_suspect_above is not None:
|
|
187
|
+
if not (drop_suspect_above > 0):
|
|
188
|
+
raise ValueError(f"drop_suspect_above must be positive, got {drop_suspect_above!r}")
|
|
189
|
+
ret = ret.mask(ret.abs() > drop_suspect_above) # unknown, so the chain treats it as 0
|
|
190
|
+
adj = 100.0 * np.exp(np.log1p(ret.fillna(0.0)).cumsum().where(close.notna()))
|
|
191
|
+
k = adj / close # adjust open/high/low to the same level
|
|
192
|
+
value, vol_raw = mat("value"), mat("volume")
|
|
193
|
+
age = close.notna().cumsum()
|
|
194
|
+
adv = value.rolling(adv_window, min_periods=adv_window).median()
|
|
195
|
+
ok = close.notna() & (vol_raw > 0) & (age >= min_age_days) & (adv >= min_value_krw)
|
|
196
|
+
if common_only:
|
|
197
|
+
ok = ok & pd.Series({c: c.split("#")[0].endswith("0") for c in ok.columns}).reindex(ok.columns).values
|
|
198
|
+
if exclude is not None:
|
|
199
|
+
bad = [s for s in ok.columns if exclude.search(names.get(s, ""))]
|
|
200
|
+
ok[bad] = False
|
|
201
|
+
last = close.apply(lambda c: c.last_valid_index())
|
|
202
|
+
da = pd.DataFrame(False, index=close.index, columns=close.columns)
|
|
203
|
+
for s in close.columns:
|
|
204
|
+
if last[s] is not None and last[s] < close.index[-1] - pd.Timedelta(days=5):
|
|
205
|
+
da.loc[last[s], s] = True
|
|
206
|
+
def intraday(c: str) -> pd.DataFrame:
|
|
207
|
+
"""Open, high or low adjusted to the close's level. The raw files write 0 on a day without trades (volume 0), which is no price: it becomes NaN."""
|
|
208
|
+
m = mat(c)
|
|
209
|
+
return (m * k).where(m > 0)
|
|
210
|
+
|
|
211
|
+
p = Panel(close=adj, eligible=ok, open=intraday("open"), high=intraday("high"), low=intraday("low"),
|
|
212
|
+
volume=value / adj, mkt_cap=mat("mktcap"), market="KR", periods_per_year=245, entry_lag=entry_lag,
|
|
213
|
+
delist_after=da)
|
|
214
|
+
n = ok.sum(axis=1)
|
|
215
|
+
mk = mat("_mk").fillna(-1).astype("int8") # the market of each security on each day (-1: not listed)
|
|
216
|
+
p.meta = {"suspect_returns": suspect, "market_codes": {i: m for i, m in enumerate(market_names)}, "market_by_date": mk,
|
|
217
|
+
"raw_close": close.astype("float32"), # the real price level (won), for whole-share sizes; `close` is back-adjusted
|
|
218
|
+
"securities": int(close.shape[1]), "delisted_in_panel": int(da.values.any(axis=0).sum()),
|
|
219
|
+
"eligible_median": float(n[n > 0].median()), "first_day": str(close.index[0].date()),
|
|
220
|
+
"last_day": str(close.index[-1].date()), "names": names}
|
|
221
|
+
return p
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def sell_tax_panel(panel: Panel, schedule: dict) -> pd.DataFrame:
|
|
225
|
+
"""(date x ticker) sell-side transaction tax in bp, for `backtest_portfolio(sell_bp=...)` and `backtest_weights(sell_bp=...)`.
|
|
226
|
+
|
|
227
|
+
schedule {market name: [(effective_date, bp), ...]}, for example {"KOSPI": [("2025-01-01", 15.0), ...], "KOSDAQ": [...]}. Each
|
|
228
|
+
market's rate holds from its effective date until the next entry. The panel must come from `build_krx_panel`, which
|
|
229
|
+
records the market of every security **on every day** (a stock that moved from KOSDAQ to KOSPI changes rate the day it
|
|
230
|
+
moves). A date before the first entry of a market, or a market in the data that is missing from `schedule`, raises.
|
|
231
|
+
|
|
232
|
+
The library ships no rate table: the rates and their effective dates are law and change, so you pass the ones you have checked
|
|
233
|
+
against the National Tax Service or the statute. Include every levy that the seller pays (the securities transaction tax and, for
|
|
234
|
+
KOSPI, the rural development special tax). Days on which a security is not listed get the highest rate of that day, the cautious
|
|
235
|
+
choice, because a position held through a gap is still charged when it is sold."""
|
|
236
|
+
from ..core.costs import rate_schedule
|
|
237
|
+
mk, codes = panel.meta.get("market_by_date"), panel.meta.get("market_codes")
|
|
238
|
+
if mk is None or codes is None:
|
|
239
|
+
raise ValueError("this panel has no market information: build it with build_krx_panel")
|
|
240
|
+
unknown = sorted(set(codes.values()) - set(schedule))
|
|
241
|
+
if unknown:
|
|
242
|
+
raise ValueError(f"no tax schedule for market(s) {unknown}; give one for every market in the data ({sorted(codes.values())})")
|
|
243
|
+
rates = {m: rate_schedule(panel.dates, schedule[m], f"schedule[{m!r}]").to_numpy(float) for m in codes.values()}
|
|
244
|
+
arr = mk.reindex(index=panel.dates, columns=panel.tickers).to_numpy()
|
|
245
|
+
worst = np.max(np.vstack([rates[m] for m in codes.values()]), axis=0)
|
|
246
|
+
out = np.repeat(worst[:, None], arr.shape[1], axis=1)
|
|
247
|
+
for i, m in codes.items():
|
|
248
|
+
out = np.where(arr == i, rates[m][:, None], out)
|
|
249
|
+
return pd.DataFrame(out, index=panel.dates, columns=panel.tickers)
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Bring your own point-in-time data: a long table (one row per security and day) into a Panel, with strict checks.
|
|
2
|
+
|
|
3
|
+
This is the door for CRSP (WRDS), Sharadar, Norgate, Polygon or any other source that keeps delisted securities. The
|
|
4
|
+
framework cannot create survivorship-free data; it can refuse to accept data that is ambiguous.
|
|
5
|
+
|
|
6
|
+
Required columns (names are configurable): a **permanent security id** (not the ticker), a date and a close.
|
|
7
|
+
Optional: volume, open/high/low, market cap, a point-in-time `in_universe` flag (for example S&P 500 membership on that
|
|
8
|
+
date), a `delist_return` on the last row of a security, and a ticker column (only used to detect ticker reuse).
|
|
9
|
+
|
|
10
|
+
Checks: no duplicate (id, date), positive closes, sorted dates. Delisting is detected as the last bar of a security that
|
|
11
|
+
ended well before the panel end. If a delisting return is given it is compounded into the last close (the CRSP
|
|
12
|
+
convention: the return on the last day is (1 + r_last) * (1 + r_delist) - 1), so the engine sees a real loss.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
import pandas as pd
|
|
18
|
+
|
|
19
|
+
from ..core.panel import Panel
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def panel_from_long(df: pd.DataFrame, *, id_col: str = "id", date_col: str = "date", close_col: str = "close",
|
|
23
|
+
volume_col: str | None = None, open_col: str | None = None, high_col: str | None = None,
|
|
24
|
+
low_col: str | None = None, mktcap_col: str | None = None, in_universe_col: str | None = None,
|
|
25
|
+
delist_return_col: str | None = None, ticker_col: str | None = None,
|
|
26
|
+
min_age_days: int = 60, adv_window: int = 30, min_adv: float | None = None,
|
|
27
|
+
ended_gap_days: int = 30, entry_lag: int = 1, market: str = "US",
|
|
28
|
+
periods_per_year: int = 252) -> Panel:
|
|
29
|
+
"""Build a Panel from a long table, one row per security and day (columns and checks: see the module docstring).
|
|
30
|
+
|
|
31
|
+
Raises KeyError for a missing required column and ValueError for duplicate (id, date) rows or non-positive closes.
|
|
32
|
+
Eligibility is `in_universe_col` when given; otherwise at least `min_age_days` of history and, when `min_adv` and a volume column are
|
|
33
|
+
given, a trailing median traded value (close x volume over `adv_window` rows) of at least `min_adv`. `volume_col` is a quantity.
|
|
34
|
+
A security whose last row is more than `ended_gap_days` before the last date is flagged in `delist_after`, and only for those a
|
|
35
|
+
`delist_return_col` value is compounded into the last close. `panel.meta` records the row and security counts, how many ended early,
|
|
36
|
+
how many delisting returns were applied, and the tickers used by more than one id."""
|
|
37
|
+
need = [id_col, date_col, close_col]
|
|
38
|
+
miss = [c for c in need if c not in df.columns]
|
|
39
|
+
if miss:
|
|
40
|
+
raise KeyError(f"missing columns: {miss}")
|
|
41
|
+
d = df.copy()
|
|
42
|
+
d[date_col] = pd.to_datetime(d[date_col])
|
|
43
|
+
if d.duplicated([id_col, date_col]).any():
|
|
44
|
+
raise ValueError("duplicate (id, date) rows: the id must identify one security on one date")
|
|
45
|
+
if (d[close_col] <= 0).any():
|
|
46
|
+
raise ValueError("non-positive closes: clean them or drop the rows")
|
|
47
|
+
d = d.sort_values([id_col, date_col])
|
|
48
|
+
meta: dict = {"rows": int(len(d)), "securities": int(d[id_col].nunique())}
|
|
49
|
+
|
|
50
|
+
last_row = d.groupby(id_col).tail(1).set_index(id_col)
|
|
51
|
+
end = d[date_col].max()
|
|
52
|
+
ended = last_row[date_col] < end - pd.Timedelta(days=ended_gap_days)
|
|
53
|
+
meta["ended_before_end"] = int(ended.sum())
|
|
54
|
+
meta["ended_share"] = float(ended.mean())
|
|
55
|
+
|
|
56
|
+
n_dl = 0
|
|
57
|
+
if delist_return_col and delist_return_col in d.columns:
|
|
58
|
+
idx = d.groupby(id_col).tail(1).index
|
|
59
|
+
dr = d.loc[idx, delist_return_col]
|
|
60
|
+
hit = idx[dr.notna().values & ended.reindex(d.loc[idx, id_col]).values]
|
|
61
|
+
d.loc[hit, close_col] = d.loc[hit, close_col] * (1.0 + d.loc[hit, delist_return_col])
|
|
62
|
+
n_dl = len(hit)
|
|
63
|
+
meta["delist_returns_applied"] = n_dl
|
|
64
|
+
|
|
65
|
+
def mat(c: str) -> pd.DataFrame:
|
|
66
|
+
return d.pivot(index=date_col, columns=id_col, values=c).sort_index()
|
|
67
|
+
|
|
68
|
+
close = mat(close_col)
|
|
69
|
+
vol = mat(volume_col) if volume_col else None
|
|
70
|
+
age = close.notna().cumsum()
|
|
71
|
+
if in_universe_col:
|
|
72
|
+
ok = mat(in_universe_col).fillna(False).astype(bool) & close.notna()
|
|
73
|
+
else:
|
|
74
|
+
ok = close.notna() & (age >= min_age_days)
|
|
75
|
+
if min_adv is not None and vol is not None:
|
|
76
|
+
ok &= (close * vol).rolling(adv_window, min_periods=adv_window).median() >= min_adv
|
|
77
|
+
da = pd.DataFrame(False, index=close.index, columns=close.columns)
|
|
78
|
+
for s in ended[ended].index:
|
|
79
|
+
da.loc[last_row.loc[s, date_col], s] = True
|
|
80
|
+
if ticker_col and ticker_col in d.columns:
|
|
81
|
+
t = d.groupby(ticker_col)[id_col].nunique()
|
|
82
|
+
meta["reused_tickers"] = sorted(t[t > 1].index.astype(str).tolist()) # one ticker, several securities
|
|
83
|
+
p = Panel(close=close, eligible=ok, open=mat(open_col) if open_col else None, high=mat(high_col) if high_col else None,
|
|
84
|
+
low=mat(low_col) if low_col else None, volume=vol, mkt_cap=mat(mktcap_col) if mktcap_col else None,
|
|
85
|
+
market=market, periods_per_year=periods_per_year, entry_lag=entry_lag, delist_after=da)
|
|
86
|
+
p.meta = meta
|
|
87
|
+
return p
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""Tiingo end-of-day prices (free account) as a point-in-time panel that keeps delisted securities.
|
|
2
|
+
|
|
3
|
+
What it is for
|
|
4
|
+
Tiingo's end-of-day API returns the full history of many US stocks that were later acquired or delisted, which yfinance
|
|
5
|
+
does not. The free tier is small, so this adapter is built for a **random sample** of tickers, not the whole market:
|
|
6
|
+
`draw_order` gives a seeded random order and `fetch_symbols` downloads in that order, resumably, and stops cleanly when
|
|
7
|
+
the account runs out of requests. Any prefix of the order is a random sample.
|
|
8
|
+
|
|
9
|
+
What it does not fix
|
|
10
|
+
A probe (`docs/tiingo_probe.py`) found the history of 23 of 29 takeover or rename cases but of **none** of 12 bankruptcies
|
|
11
|
+
and rescue sales. A panel from this adapter is less survivor-biased than one from yfinance, not unbiased. Treat any
|
|
12
|
+
survivors-versus-this comparison as a lower bound.
|
|
13
|
+
|
|
14
|
+
Choices worth knowing
|
|
15
|
+
- Returns use `adjClose` (split and dividend adjusted, so total return). Eligibility uses the raw `close` and `volume`.
|
|
16
|
+
- A ticker with several listing windows in the ticker list (reuse) becomes several securities, each cut to its own window.
|
|
17
|
+
- A security whose last bar is more than `end_gap_days` before the last calendar date is flagged in `delist_after`.
|
|
18
|
+
- A ticker for which the API has nothing is recorded (`.none`) and counted, never silently replaced.
|
|
19
|
+
- Only the `TIINGO_API_KEY` variable is read.
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import json
|
|
24
|
+
import os
|
|
25
|
+
import re
|
|
26
|
+
import time
|
|
27
|
+
import urllib.error
|
|
28
|
+
import urllib.request
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
import numpy as np
|
|
32
|
+
import pandas as pd
|
|
33
|
+
|
|
34
|
+
from ..core.panel import Panel
|
|
35
|
+
|
|
36
|
+
BASE = "https://api.tiingo.com/tiingo/daily"
|
|
37
|
+
KEEP = ("date", "close", "volume", "adjClose")
|
|
38
|
+
# With `full=True` the open, high, low (adjusted for splits and dividends like adjClose), the dividend paid and the split factor are kept too.
|
|
39
|
+
FULL = KEEP + ("open", "high", "low", "adjOpen", "adjHigh", "adjLow", "divCash", "splitFactor")
|
|
40
|
+
_NOT_PLAIN = re.compile(r"[-.]")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class QuotaExceeded(RuntimeError):
|
|
44
|
+
"""The API refused further requests (rate or monthly limit). Run again later; the cache keeps what was fetched."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def load_key(env_path: str | Path | None = None) -> str:
|
|
48
|
+
"""TIINGO_API_KEY from the environment, else from a .env file. Only that one variable is read."""
|
|
49
|
+
if os.environ.get("TIINGO_API_KEY"):
|
|
50
|
+
return os.environ["TIINGO_API_KEY"]
|
|
51
|
+
if env_path:
|
|
52
|
+
for line in Path(env_path).expanduser().read_text().splitlines():
|
|
53
|
+
k, _, v = line.partition("=")
|
|
54
|
+
if k.strip() == "TIINGO_API_KEY":
|
|
55
|
+
return v.strip().strip('"').strip("'")
|
|
56
|
+
raise KeyError("TIINGO_API_KEY is not set")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# ---------------------------------------------------------------------------------------------------- the sample
|
|
60
|
+
def study_frame(master: pd.DataFrame, since: str = "2013-01-01") -> pd.DataFrame:
|
|
61
|
+
"""Rows of the ticker list that can enter a study starting at `since`: plain tickers (no '-' or '.', and no
|
|
62
|
+
five-letter ticker ending in W, U, R or P: warrants, units, rights, preferreds) whose window ends on or after `since`."""
|
|
63
|
+
t = master["ticker"].astype(str)
|
|
64
|
+
plain = ~t.str.contains(_NOT_PLAIN) & ~((t.str.len() == 5) & t.str[-1].isin(list("WURP")))
|
|
65
|
+
return master[plain & (master["end"] >= pd.Timestamp(since))].reset_index(drop=True)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def draw_order(tickers, seed: int = 0) -> list[str]:
|
|
69
|
+
"""A seeded random permutation of the unique tickers (sorted first, so it does not depend on the input order).
|
|
70
|
+
Download in this order and every prefix is a simple random sample."""
|
|
71
|
+
u = sorted(set(map(str, tickers)))
|
|
72
|
+
rng = np.random.default_rng(seed)
|
|
73
|
+
return [u[i] for i in rng.permutation(len(u))]
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# ---------------------------------------------------------------------------------------------------- downloading
|
|
77
|
+
def _call(ticker: str, start: str, key: str, retries: int = 3):
|
|
78
|
+
"""Bars for one ticker, or None if the API does not know it. Raises QuotaExceeded on HTTP 429."""
|
|
79
|
+
url = f"{BASE}/{ticker.lower()}/prices?startDate={start}&format=json"
|
|
80
|
+
req = urllib.request.Request(url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json",
|
|
81
|
+
"User-Agent": "pitbacktest-research/0.2"})
|
|
82
|
+
delay = 1.0
|
|
83
|
+
for i in range(retries + 1):
|
|
84
|
+
try:
|
|
85
|
+
with urllib.request.urlopen(req, timeout=60) as r:
|
|
86
|
+
data = json.loads(r.read())
|
|
87
|
+
# The monthly limit on distinct symbols (500 on a free account) comes back as HTTP 200 with a message, not as a 429.
|
|
88
|
+
if isinstance(data, dict) and re.search(r"run over|allocation|look ?up|upgrade", str(data.get("detail", "")), re.I):
|
|
89
|
+
raise QuotaExceeded(f"HTTP 200: {data.get('detail')}")
|
|
90
|
+
return data
|
|
91
|
+
except urllib.error.HTTPError as e:
|
|
92
|
+
body = e.read()[:300].decode("utf-8", "replace")
|
|
93
|
+
if e.code == 429:
|
|
94
|
+
raise QuotaExceeded(f"HTTP 429: {body}") from None
|
|
95
|
+
if e.code == 404:
|
|
96
|
+
return None
|
|
97
|
+
if e.code in (401, 403):
|
|
98
|
+
raise PermissionError(f"HTTP {e.code}: {body}") from None
|
|
99
|
+
if e.code >= 500 and i < retries:
|
|
100
|
+
time.sleep(delay); delay *= 2
|
|
101
|
+
continue
|
|
102
|
+
raise
|
|
103
|
+
except (urllib.error.URLError, TimeoutError, ConnectionError, ValueError):
|
|
104
|
+
if i < retries:
|
|
105
|
+
time.sleep(delay); delay *= 2
|
|
106
|
+
continue
|
|
107
|
+
raise
|
|
108
|
+
return None
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _check(ticker: str, rows, fields=KEEP) -> list[dict]:
|
|
112
|
+
if not isinstance(rows, list):
|
|
113
|
+
raise RuntimeError(f"Tiingo {ticker}: unexpected response type {type(rows).__name__}")
|
|
114
|
+
out = []
|
|
115
|
+
for r in rows:
|
|
116
|
+
if not all(k in r for k in fields):
|
|
117
|
+
raise RuntimeError(f"Tiingo {ticker}: bar without {[k for k in fields if k not in r]}; the format changed")
|
|
118
|
+
out.append({k: (str(r[k])[:10] if k == "date" else r[k]) for k in fields})
|
|
119
|
+
return out
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _safe(ticker: str) -> str:
|
|
123
|
+
return re.sub(r"[^A-Za-z0-9_]", "_", ticker.upper())
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def fetch_symbols(tickers, store_dir, *, start: str = "2012-01-01", key: str | None = None, sleep: float = 0.3,
|
|
127
|
+
max_new: int | None = None, caller=None, progress: bool = True, full: bool = False) -> dict:
|
|
128
|
+
"""Download one file per ticker into `store_dir` (`T.json` with the needed fields, or `T.none` if the API has nothing).
|
|
129
|
+
Resumable: tickers already stored are skipped. `max_new` limits requests in this run. Stops cleanly on a quota error.
|
|
130
|
+
`full=True` keeps the open, high, low, dividend and split factor as well (use a different `store_dir` from a store made without it).
|
|
131
|
+
`caller(ticker, start)` can be injected for tests (return None for an unknown ticker)."""
|
|
132
|
+
store = Path(store_dir).expanduser()
|
|
133
|
+
store.mkdir(parents=True, exist_ok=True)
|
|
134
|
+
key = key or (None if caller else load_key())
|
|
135
|
+
call = caller or (lambda t, s: _call(t, s, key))
|
|
136
|
+
done = {"fetched": 0, "none": 0, "skipped": 0, "calls": 0, "stopped": None}
|
|
137
|
+
for t in tickers:
|
|
138
|
+
f = store / f"{_safe(t)}.json"
|
|
139
|
+
n = store / f"{_safe(t)}.none"
|
|
140
|
+
if f.exists() or n.exists():
|
|
141
|
+
done["skipped"] += 1
|
|
142
|
+
continue
|
|
143
|
+
if max_new is not None and done["calls"] >= max_new:
|
|
144
|
+
done["stopped"] = "max_new"
|
|
145
|
+
break
|
|
146
|
+
try:
|
|
147
|
+
rows = call(t, start)
|
|
148
|
+
except QuotaExceeded as e:
|
|
149
|
+
done["stopped"] = f"quota: {e}"
|
|
150
|
+
break
|
|
151
|
+
done["calls"] += 1
|
|
152
|
+
if rows is None or (isinstance(rows, list) and len(rows) == 0):
|
|
153
|
+
n.write_text("")
|
|
154
|
+
done["none"] += 1
|
|
155
|
+
else:
|
|
156
|
+
f.write_text(json.dumps(_check(t, rows, FULL if full else KEEP))) # validated first: a malformed answer is never saved
|
|
157
|
+
done["fetched"] += 1
|
|
158
|
+
if progress and done["calls"] % 25 == 0:
|
|
159
|
+
print(f" {done['calls']} requests, {done['fetched']} with data, {done['none']} none", flush=True)
|
|
160
|
+
time.sleep(sleep)
|
|
161
|
+
return done
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# ---------------------------------------------------------------------------------------------------- the panel
|
|
165
|
+
def build_tiingo_panel(store_dir, master: pd.DataFrame, tickers, *, start: str = "2012-01-01", end: str | None = None,
|
|
166
|
+
min_age_bars: int = 60, adv_window: int = 30, min_dollar_volume: float = 2e6, min_price: float = 1.0,
|
|
167
|
+
min_names_per_date: int = 100, end_gap_days: int = 5, entry_lag: int = 1) -> Panel:
|
|
168
|
+
"""Point-in-time panel from the stored files. Eligibility on day t uses only data up to t.
|
|
169
|
+
|
|
170
|
+
Each listing window of a ticker in `master` (columns ticker, start, end) is one security, cut to that window.
|
|
171
|
+
The trading calendar is the set of dates on which at least `min_names_per_date` securities have a bar.
|
|
172
|
+
`start` is the first date kept: pass the start of the warm-up year and cut the study period afterwards. The age of a
|
|
173
|
+
security counts bars from the start of the stored data (`fetch_symbols(start=...)`), so a warm-up year of 60+ bars
|
|
174
|
+
makes that irrelevant for the study period."""
|
|
175
|
+
store = Path(store_dir).expanduser()
|
|
176
|
+
adj, raw, vol, oh, div, meta = {}, {}, {}, {"adjOpen": {}, "adjHigh": {}, "adjLow": {}}, {}, {"tickers_with_data": 0, "tickers_none": 0, "tickers_missing": 0, "windows_cut": 0}
|
|
177
|
+
for t in map(str, tickers):
|
|
178
|
+
f, n = store / f"{_safe(t)}.json", store / f"{_safe(t)}.none"
|
|
179
|
+
if n.exists():
|
|
180
|
+
meta["tickers_none"] += 1
|
|
181
|
+
continue
|
|
182
|
+
if not f.exists():
|
|
183
|
+
meta["tickers_missing"] += 1
|
|
184
|
+
continue
|
|
185
|
+
bars = pd.DataFrame(json.loads(f.read_text()))
|
|
186
|
+
if bars.empty:
|
|
187
|
+
meta["tickers_none"] += 1
|
|
188
|
+
continue
|
|
189
|
+
meta["tickers_with_data"] += 1
|
|
190
|
+
bars["date"] = pd.to_datetime(bars["date"])
|
|
191
|
+
bars = bars.drop_duplicates("date", keep="last").set_index("date").sort_index()
|
|
192
|
+
wins = master[master["ticker"].astype(str) == t].sort_values("start")
|
|
193
|
+
if wins.empty:
|
|
194
|
+
wins = pd.DataFrame({"start": [bars.index[0]], "end": [bars.index[-1]]})
|
|
195
|
+
multi = len(wins) > 1
|
|
196
|
+
meta["windows_cut"] += int(multi)
|
|
197
|
+
for _, w in wins.iterrows():
|
|
198
|
+
b = bars.loc[(bars.index >= w["start"]) & (bars.index <= w["end"])]
|
|
199
|
+
b = b[(b["adjClose"] > 0) & (b["close"] > 0)]
|
|
200
|
+
if b.empty:
|
|
201
|
+
continue
|
|
202
|
+
sid = f"{t}@{w['start']:%Y%m%d}" if multi else t
|
|
203
|
+
adj[sid], raw[sid], vol[sid] = b["adjClose"], b["close"], b["volume"]
|
|
204
|
+
if "adjOpen" in b.columns: # a store made with full=True
|
|
205
|
+
for k in oh:
|
|
206
|
+
oh[k][sid] = b[k]
|
|
207
|
+
div[sid] = b["divCash"]
|
|
208
|
+
if not adj:
|
|
209
|
+
raise ValueError("no usable securities in the store")
|
|
210
|
+
A, R, V = pd.DataFrame(adj), pd.DataFrame(raw), pd.DataFrame(vol)
|
|
211
|
+
cal = A.index[(A.notna().sum(axis=1) >= min_names_per_date) & (A.index >= pd.Timestamp(start))]
|
|
212
|
+
if end:
|
|
213
|
+
cal = cal[cal <= pd.Timestamp(end)]
|
|
214
|
+
A, R, V = A.reindex(cal), R.reindex(cal), V.reindex(cal)
|
|
215
|
+
dvol = R * V
|
|
216
|
+
# eligibility, all trailing
|
|
217
|
+
age = A.notna().cumsum()
|
|
218
|
+
med = dvol.rolling(adv_window, min_periods=adv_window).median()
|
|
219
|
+
elig = (age >= min_age_bars) & (med >= min_dollar_volume) & (R >= min_price) & (V > 0) & A.notna()
|
|
220
|
+
last = A.apply(lambda c: c.last_valid_index())
|
|
221
|
+
last_date = cal[-1]
|
|
222
|
+
da = pd.DataFrame(False, index=cal, columns=A.columns)
|
|
223
|
+
for c in A.columns:
|
|
224
|
+
if last[c] is not None and last[c] < last_date - pd.Timedelta(days=end_gap_days):
|
|
225
|
+
da.loc[last[c], c] = True
|
|
226
|
+
r = A.pct_change(fill_method=None)
|
|
227
|
+
extra = {}
|
|
228
|
+
if oh["adjOpen"]:
|
|
229
|
+
O, Hh, Ll = (pd.DataFrame(oh[k]).reindex(index=cal, columns=A.columns) for k in ("adjOpen", "adjHigh", "adjLow"))
|
|
230
|
+
extra = {"open": O, "high": Hh, "low": Ll}
|
|
231
|
+
meta["div_cash"] = pd.DataFrame(div).reindex(index=cal, columns=A.columns).astype("float32") # dollars per share paid that day (raw prices)
|
|
232
|
+
meta["suspect_returns_gt_10x"] = int((r.abs() > 10).sum().sum())
|
|
233
|
+
meta["securities"] = int(A.shape[1])
|
|
234
|
+
meta["raw_close"] = R.astype("float32") # the real price level (dollars), for whole-share sizes; `close` is adjusted
|
|
235
|
+
return Panel(close=A, eligible=elig, volume=dvol / A, market="US", entry_lag=entry_lag, periods_per_year=252,
|
|
236
|
+
delist_after=da, meta=meta, **extra)
|