pitbacktest 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,27 @@
1
+ """pitbacktest: a backtest toolkit that measures its own biases (US, Korea, crypto).
2
+
3
+ Three entry points
4
+ screen() A. Factor screening with neutralisation (research stage)
5
+ backtest_event() C. Event signals (alert products)
6
+ backtest_portfolio() B. Portfolio alpha (CAGR, MDD, Sharpe)
7
+
8
+ Design principles
9
+ - The core is plain pandas and numpy; data sources live in `adapters`.
10
+ - There is one timing convention and it is enforced with an assert.
11
+ - Controls for firm characteristics are the default.
12
+ - The statistics that decide a verdict are computed with controls; uncontrolled figures are returned next to them, not in their place.
13
+ """
14
+
15
+ from .core.panel import Panel, build_pit_eligible
16
+ from .core.gates import GateConfig
17
+ from .event import backtest_event
18
+ from .portfolio import backtest_portfolio, assert_timing
19
+ from .screen import screen
20
+ from . import validation, analytics, execution
21
+ from .ledger import Ledger
22
+ from .weights import backtest_weights, capacity_curve, ImpactModel
23
+ from .shorting import shortable_from_bans
24
+
25
+ __version__ = "0.2.0"
26
+ __all__ = ["shortable_from_bans", "execution", "validation", "analytics", "Ledger", "backtest_weights", "capacity_curve", "ImpactModel", "Panel", "build_pit_eligible", "GateConfig", "screen",
27
+ "backtest_event", "backtest_portfolio", "assert_timing"]
File without changes
@@ -0,0 +1,249 @@
1
+ """Korean equities from the official KRX OpenAPI, point in time and survivorship-free.
2
+
3
+ How it removes survivorship bias: the API returns **every stock that was listed on a given date**, including the ones
4
+ delisted or merged later (on 2015-01-02, 125 of the 899 KOSPI names are gone by 2024). Fetching day by day therefore
5
+ builds the universe the way it really was.
6
+
7
+ How it handles corporate actions without a price-adjustment table: each row carries the change versus the *reference
8
+ price* (`CMPPREVDD_PRC`), and the reference price already reflects splits and rights issues. So
9
+ `close / (close - change) - 1` is the adjusted return. It matched the published return to within rounding for all 953
10
+ KOSPI stocks on 2024-01-02, and gives -2.08% (not -98%) for Samsung Electronics on its 50:1 split day (2018-05-04).
11
+ The adjusted close used by the panel is the cumulative product of those returns. Dividends are not included (price
12
+ return only).
13
+
14
+ You need a KRX OpenAPI key (free, the services have to be approved on openapi.krx.co.kr). Put it in the environment as
15
+ KRX_OPENAPI_KEY or pass `key=`; `load_key(env_path)` reads only that one variable from a .env file.
16
+
17
+ fetch_days("2013-01-01", "2026-10-06", "~/.cache/quantbt/krx") # resumable, about 2 calls per trading day
18
+ panel = build_krx_panel("~/.cache/quantbt/krx", start="2014-01-01")
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import json
23
+ import os
24
+ import re
25
+ import time
26
+ import urllib.error
27
+ import urllib.request
28
+ from pathlib import Path
29
+
30
+ import numpy as np
31
+ import pandas as pd
32
+
33
+ from ..core.panel import Panel
34
+
35
+ BASE = "https://data-dbg.krx.co.kr/svc/apis"
36
+ ENDPOINTS = {"KOSPI": "sto/stk_bydd_trd", "KOSDAQ": "sto/ksq_bydd_trd"}
37
+ RENAME = {"ISU_CD": "code", "ISU_NM": "name", "MKT_NM": "market", "TDD_CLSPRC": "close", "CMPPREVDD_PRC": "change",
38
+ "TDD_OPNPRC": "open", "TDD_HGPRC": "high", "TDD_LWPRC": "low", "ACC_TRDVOL": "volume", "ACC_TRDVAL": "value",
39
+ "MKTCAP": "mktcap", "LIST_SHRS": "shares"}
40
+ # Not tradable common stock for a cross-sectional study: SPACs, REITs and funds listed as stocks.
41
+ # Korean name patterns that mark SPACs (스팩), REITs (리츠), numbered funds (N호), and infrastructure funds (인프라, 맥쿼리). These must stay Korean:
42
+ # they are matched against the names KRX returns.
43
+ DEFAULT_EXCLUDE = re.compile(r"스팩|리츠|\d+호|인프라|맥쿼리")
44
+
45
+
46
+ class QuotaExceeded(RuntimeError):
47
+ """The API refused further calls (daily limit). Run again later; the cache keeps what was fetched."""
48
+
49
+
50
+ def load_key(env_path: str | Path | None = None) -> str:
51
+ """KRX_OPENAPI_KEY from the environment, else from a .env file. Only that one variable is read."""
52
+ if os.environ.get("KRX_OPENAPI_KEY"):
53
+ return os.environ["KRX_OPENAPI_KEY"]
54
+ if env_path:
55
+ for line in Path(env_path).expanduser().read_text().splitlines():
56
+ k, _, v = line.partition("=")
57
+ if k.strip() == "KRX_OPENAPI_KEY":
58
+ return v.strip().strip('"').strip("'")
59
+ raise KeyError("KRX_OPENAPI_KEY is not set")
60
+
61
+
62
+ def _call(path: str, date: str, key: str, retries: int = 4) -> list[dict]:
63
+ req = urllib.request.Request(f"{BASE}/{path}?basDd={date}", headers={"AUTH_KEY": key, "User-Agent": "pitbacktest-research/0.2"})
64
+ delay = 1.0
65
+ for i in range(retries + 1):
66
+ try:
67
+ with urllib.request.urlopen(req, timeout=40) as r:
68
+ return json.loads(r.read()).get("OutBlock_1", [])
69
+ except urllib.error.HTTPError as e:
70
+ body = e.read()[:200].decode("utf-8", "replace")
71
+ if e.code in (429, 403) or "limit" in body.lower() or "초과" in body: # "초과" = "exceeded", the wording of the KRX quota message
72
+ raise QuotaExceeded(f"HTTP {e.code}: {body}") from None
73
+ if e.code >= 500 and i < retries:
74
+ time.sleep(delay); delay *= 2
75
+ continue
76
+ raise
77
+ except (urllib.error.URLError, TimeoutError, ConnectionError, ValueError): # ValueError: an empty or non-JSON body
78
+ if i < retries:
79
+ time.sleep(delay); delay *= 2
80
+ continue
81
+ raise
82
+ return []
83
+
84
+
85
+ def _frame(rows: list[dict]) -> pd.DataFrame:
86
+ df = pd.DataFrame(rows).rename(columns=RENAME)[list(RENAME.values())]
87
+ for c in ("close", "change", "open", "high", "low", "volume", "value", "mktcap", "shares"):
88
+ df[c] = pd.to_numeric(df[c].replace({"-": np.nan, "": np.nan}), errors="coerce")
89
+ return df
90
+
91
+
92
+ def fetch_days(start, end, store_dir, *, key: str | None = None, markets=("KOSPI", "KOSDAQ"), sleep: float = 0.1,
93
+ max_calls: int | None = None, caller=None, progress: bool = True) -> dict:
94
+ """Download one file per weekday into `store_dir`. Resumable: finished days (and holidays) are skipped.
95
+ `caller(path, date)` can be injected for tests. Stops cleanly if the API reports a quota limit."""
96
+ store = Path(store_dir).expanduser()
97
+ store.mkdir(parents=True, exist_ok=True)
98
+ key = key or (None if caller else load_key())
99
+ call = caller or (lambda p, d: _call(p, d, key))
100
+ done = {"days": 0, "holidays": 0, "skipped": 0, "calls": 0, "stopped": None}
101
+ for ts in pd.bdate_range(start, end):
102
+ d = ts.strftime("%Y%m%d")
103
+ if (store / f"{d}.pkl").exists() or (store / f"{d}.hol").exists():
104
+ done["skipped"] += 1
105
+ continue
106
+ if max_calls is not None and done["calls"] + len(markets) > max_calls:
107
+ done["stopped"] = "max_calls"
108
+ break
109
+ frames = []
110
+ try:
111
+ for m in markets:
112
+ rows = call(ENDPOINTS[m], d)
113
+ done["calls"] += 1
114
+ if rows:
115
+ frames.append(_frame(rows))
116
+ time.sleep(sleep)
117
+ except QuotaExceeded as e:
118
+ done["stopped"] = f"quota: {e}"
119
+ break
120
+ if frames and len(frames) < len(markets) and caller is None:
121
+ done.setdefault("partial", []).append(d) # one market answered empty: do not save, retry next run
122
+ continue
123
+ if frames:
124
+ pd.concat(frames).to_pickle(store / f"{d}.pkl")
125
+ done["days"] += 1
126
+ else:
127
+ (store / f"{d}.hol").write_text("") # holiday: both markets empty
128
+ done["holidays"] += 1
129
+ if progress and (done["days"] + done["holidays"]) % 100 == 0:
130
+ print(f" {d} {done}", flush=True)
131
+ return done
132
+
133
+
134
+ def build_krx_panel(store_dir, *, start=None, end=None, min_age_days: int = 60, adv_window: int = 30,
135
+ min_value_krw: float = 1e9, common_only: bool = True, exclude=DEFAULT_EXCLUDE,
136
+ reuse_gap_days: int = 120, entry_lag: int = 1, drop_suspect_above: float | None = None) -> Panel:
137
+ """Point-in-time Panel of Korean stocks from the cached daily files.
138
+
139
+ eligible on day t uses data up to t only: at least `min_age_days` bars of history, trailing median traded value of
140
+ at least `min_value_krw`, positive volume that day, common shares only (code ends in 0) and no SPAC or REIT names.
141
+ A code that disappears for more than `reuse_gap_days` and comes back is treated as a different security (suffix #2).
142
+
143
+ Known defect of the return rule. The adjusted return is `close / (close - change) - 1`, which is right when the exchange's reference price already
144
+ reflects a split or rights issue. When a security is **suspended and a consolidation or capital reduction happens meanwhile**, the first day of trading
145
+ again carries a `change` against the old, unadjusted close, and the formula returns a move of thousands of percent that never happened (one case in the
146
+ cache: 2,080 won to 625,000 won, +29,948%). Every daily return beyond +-100% is listed in `meta["suspect_returns"]` (ticker, date, return, and whether the
147
+ previous day had zero or missing volume): 38 of 7.8 million in the whole cache (from 2013; the largest is +6,699,900%), 14 of them on such a day, and 29 and 10 since June 2015.
148
+ All are upward: a return cannot go below -100%. `drop_suspect_above=1.0` treats
149
+ such a day as a return of 0 (the price chain is cut there, the true move is unknown); the default None leaves the series as it was, so earlier results
150
+ reproduce. The strategies in this repository do not hold suspended names (eligibility needs volume), so none of them was affected; a position that is
151
+ **stuck** in a suspended name (`Panel.can_buy`/`can_sell`, `freeze_*`) can be."""
152
+ store = Path(store_dir).expanduser()
153
+ files = sorted(store.glob("*.pkl"))
154
+ if start:
155
+ files = [f for f in files if f.stem >= pd.Timestamp(start).strftime("%Y%m%d")]
156
+ if end:
157
+ files = [f for f in files if f.stem <= pd.Timestamp(end).strftime("%Y%m%d")]
158
+ if not files:
159
+ raise ValueError("no cached days: run fetch_days() first")
160
+ parts = []
161
+ for f in files:
162
+ d = pd.read_pickle(f)
163
+ d["date"] = pd.Timestamp(f.stem)
164
+ parts.append(d)
165
+ df = pd.concat(parts, ignore_index=True).sort_values(["code", "date"])
166
+ gap = df.groupby("code")["date"].diff().dt.days
167
+ life = (gap > reuse_gap_days).groupby(df["code"]).cumsum()
168
+ df["sid"] = np.where(life == 0, df["code"], df["code"] + "#" + (life + 1).astype(str))
169
+ names = df.drop_duplicates("sid", keep="last").set_index("sid")["name"].to_dict()
170
+ market_names = sorted(df["market"].dropna().unique())
171
+ df["_mk"] = df["market"].map({m: i for i, m in enumerate(market_names)}).astype("float64")
172
+
173
+ def mat(c: str) -> pd.DataFrame:
174
+ return df.pivot(index="date", columns="sid", values=c).sort_index()
175
+
176
+ close, chg = mat("close"), mat("change")
177
+ ref = close - chg
178
+ ret = (close / ref - 1.0).where(ref > 0)
179
+ first = close.notna() & ~close.notna().cumsum().shift(1, fill_value=0).astype(bool) # a security's first bar
180
+ ret = ret.mask(first) # the listing-day move is not tradable
181
+ vol_prev0 = (mat("volume").fillna(0) <= 0).shift(1, fill_value=False)
182
+ sus = ret.stack()
183
+ sus = sus[sus.abs() > 1.0]
184
+ suspect = pd.DataFrame({"ticker": sus.index.get_level_values(1), "date": sus.index.get_level_values(0), "ret": sus.to_numpy(),
185
+ "after_suspension": [bool(vol_prev0.loc[d, t]) for d, t in sus.index]}).reset_index(drop=True)
186
+ if drop_suspect_above is not None:
187
+ if not (drop_suspect_above > 0):
188
+ raise ValueError(f"drop_suspect_above must be positive, got {drop_suspect_above!r}")
189
+ ret = ret.mask(ret.abs() > drop_suspect_above) # unknown, so the chain treats it as 0
190
+ adj = 100.0 * np.exp(np.log1p(ret.fillna(0.0)).cumsum().where(close.notna()))
191
+ k = adj / close # adjust open/high/low to the same level
192
+ value, vol_raw = mat("value"), mat("volume")
193
+ age = close.notna().cumsum()
194
+ adv = value.rolling(adv_window, min_periods=adv_window).median()
195
+ ok = close.notna() & (vol_raw > 0) & (age >= min_age_days) & (adv >= min_value_krw)
196
+ if common_only:
197
+ ok = ok & pd.Series({c: c.split("#")[0].endswith("0") for c in ok.columns}).reindex(ok.columns).values
198
+ if exclude is not None:
199
+ bad = [s for s in ok.columns if exclude.search(names.get(s, ""))]
200
+ ok[bad] = False
201
+ last = close.apply(lambda c: c.last_valid_index())
202
+ da = pd.DataFrame(False, index=close.index, columns=close.columns)
203
+ for s in close.columns:
204
+ if last[s] is not None and last[s] < close.index[-1] - pd.Timedelta(days=5):
205
+ da.loc[last[s], s] = True
206
+ def intraday(c: str) -> pd.DataFrame:
207
+ """Open, high or low adjusted to the close's level. The raw files write 0 on a day without trades (volume 0), which is no price: it becomes NaN."""
208
+ m = mat(c)
209
+ return (m * k).where(m > 0)
210
+
211
+ p = Panel(close=adj, eligible=ok, open=intraday("open"), high=intraday("high"), low=intraday("low"),
212
+ volume=value / adj, mkt_cap=mat("mktcap"), market="KR", periods_per_year=245, entry_lag=entry_lag,
213
+ delist_after=da)
214
+ n = ok.sum(axis=1)
215
+ mk = mat("_mk").fillna(-1).astype("int8") # the market of each security on each day (-1: not listed)
216
+ p.meta = {"suspect_returns": suspect, "market_codes": {i: m for i, m in enumerate(market_names)}, "market_by_date": mk,
217
+ "raw_close": close.astype("float32"), # the real price level (won), for whole-share sizes; `close` is back-adjusted
218
+ "securities": int(close.shape[1]), "delisted_in_panel": int(da.values.any(axis=0).sum()),
219
+ "eligible_median": float(n[n > 0].median()), "first_day": str(close.index[0].date()),
220
+ "last_day": str(close.index[-1].date()), "names": names}
221
+ return p
222
+
223
+
224
+ def sell_tax_panel(panel: Panel, schedule: dict) -> pd.DataFrame:
225
+ """(date x ticker) sell-side transaction tax in bp, for `backtest_portfolio(sell_bp=...)` and `backtest_weights(sell_bp=...)`.
226
+
227
+ schedule {market name: [(effective_date, bp), ...]}, for example {"KOSPI": [("2025-01-01", 15.0), ...], "KOSDAQ": [...]}. Each
228
+ market's rate holds from its effective date until the next entry. The panel must come from `build_krx_panel`, which
229
+ records the market of every security **on every day** (a stock that moved from KOSDAQ to KOSPI changes rate the day it
230
+ moves). A date before the first entry of a market, or a market in the data that is missing from `schedule`, raises.
231
+
232
+ The library ships no rate table: the rates and their effective dates are law and change, so you pass the ones you have checked
233
+ against the National Tax Service or the statute. Include every levy that the seller pays (the securities transaction tax and, for
234
+ KOSPI, the rural development special tax). Days on which a security is not listed get the highest rate of that day, the cautious
235
+ choice, because a position held through a gap is still charged when it is sold."""
236
+ from ..core.costs import rate_schedule
237
+ mk, codes = panel.meta.get("market_by_date"), panel.meta.get("market_codes")
238
+ if mk is None or codes is None:
239
+ raise ValueError("this panel has no market information: build it with build_krx_panel")
240
+ unknown = sorted(set(codes.values()) - set(schedule))
241
+ if unknown:
242
+ raise ValueError(f"no tax schedule for market(s) {unknown}; give one for every market in the data ({sorted(codes.values())})")
243
+ rates = {m: rate_schedule(panel.dates, schedule[m], f"schedule[{m!r}]").to_numpy(float) for m in codes.values()}
244
+ arr = mk.reindex(index=panel.dates, columns=panel.tickers).to_numpy()
245
+ worst = np.max(np.vstack([rates[m] for m in codes.values()]), axis=0)
246
+ out = np.repeat(worst[:, None], arr.shape[1], axis=1)
247
+ for i, m in codes.items():
248
+ out = np.where(arr == i, rates[m][:, None], out)
249
+ return pd.DataFrame(out, index=panel.dates, columns=panel.tickers)
@@ -0,0 +1,87 @@
1
+ """Bring your own point-in-time data: a long table (one row per security and day) into a Panel, with strict checks.
2
+
3
+ This is the door for CRSP (WRDS), Sharadar, Norgate, Polygon or any other source that keeps delisted securities. The
4
+ framework cannot create survivorship-free data; it can refuse to accept data that is ambiguous.
5
+
6
+ Required columns (names are configurable): a **permanent security id** (not the ticker), a date and a close.
7
+ Optional: volume, open/high/low, market cap, a point-in-time `in_universe` flag (for example S&P 500 membership on that
8
+ date), a `delist_return` on the last row of a security, and a ticker column (only used to detect ticker reuse).
9
+
10
+ Checks: no duplicate (id, date), positive closes, sorted dates. Delisting is detected as the last bar of a security that
11
+ ended well before the panel end. If a delisting return is given it is compounded into the last close (the CRSP
12
+ convention: the return on the last day is (1 + r_last) * (1 + r_delist) - 1), so the engine sees a real loss.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import numpy as np
17
+ import pandas as pd
18
+
19
+ from ..core.panel import Panel
20
+
21
+
22
+ def panel_from_long(df: pd.DataFrame, *, id_col: str = "id", date_col: str = "date", close_col: str = "close",
23
+ volume_col: str | None = None, open_col: str | None = None, high_col: str | None = None,
24
+ low_col: str | None = None, mktcap_col: str | None = None, in_universe_col: str | None = None,
25
+ delist_return_col: str | None = None, ticker_col: str | None = None,
26
+ min_age_days: int = 60, adv_window: int = 30, min_adv: float | None = None,
27
+ ended_gap_days: int = 30, entry_lag: int = 1, market: str = "US",
28
+ periods_per_year: int = 252) -> Panel:
29
+ """Build a Panel from a long table, one row per security and day (columns and checks: see the module docstring).
30
+
31
+ Raises KeyError for a missing required column and ValueError for duplicate (id, date) rows or non-positive closes.
32
+ Eligibility is `in_universe_col` when given; otherwise at least `min_age_days` of history and, when `min_adv` and a volume column are
33
+ given, a trailing median traded value (close x volume over `adv_window` rows) of at least `min_adv`. `volume_col` is a quantity.
34
+ A security whose last row is more than `ended_gap_days` before the last date is flagged in `delist_after`, and only for those a
35
+ `delist_return_col` value is compounded into the last close. `panel.meta` records the row and security counts, how many ended early,
36
+ how many delisting returns were applied, and the tickers used by more than one id."""
37
+ need = [id_col, date_col, close_col]
38
+ miss = [c for c in need if c not in df.columns]
39
+ if miss:
40
+ raise KeyError(f"missing columns: {miss}")
41
+ d = df.copy()
42
+ d[date_col] = pd.to_datetime(d[date_col])
43
+ if d.duplicated([id_col, date_col]).any():
44
+ raise ValueError("duplicate (id, date) rows: the id must identify one security on one date")
45
+ if (d[close_col] <= 0).any():
46
+ raise ValueError("non-positive closes: clean them or drop the rows")
47
+ d = d.sort_values([id_col, date_col])
48
+ meta: dict = {"rows": int(len(d)), "securities": int(d[id_col].nunique())}
49
+
50
+ last_row = d.groupby(id_col).tail(1).set_index(id_col)
51
+ end = d[date_col].max()
52
+ ended = last_row[date_col] < end - pd.Timedelta(days=ended_gap_days)
53
+ meta["ended_before_end"] = int(ended.sum())
54
+ meta["ended_share"] = float(ended.mean())
55
+
56
+ n_dl = 0
57
+ if delist_return_col and delist_return_col in d.columns:
58
+ idx = d.groupby(id_col).tail(1).index
59
+ dr = d.loc[idx, delist_return_col]
60
+ hit = idx[dr.notna().values & ended.reindex(d.loc[idx, id_col]).values]
61
+ d.loc[hit, close_col] = d.loc[hit, close_col] * (1.0 + d.loc[hit, delist_return_col])
62
+ n_dl = len(hit)
63
+ meta["delist_returns_applied"] = n_dl
64
+
65
+ def mat(c: str) -> pd.DataFrame:
66
+ return d.pivot(index=date_col, columns=id_col, values=c).sort_index()
67
+
68
+ close = mat(close_col)
69
+ vol = mat(volume_col) if volume_col else None
70
+ age = close.notna().cumsum()
71
+ if in_universe_col:
72
+ ok = mat(in_universe_col).fillna(False).astype(bool) & close.notna()
73
+ else:
74
+ ok = close.notna() & (age >= min_age_days)
75
+ if min_adv is not None and vol is not None:
76
+ ok &= (close * vol).rolling(adv_window, min_periods=adv_window).median() >= min_adv
77
+ da = pd.DataFrame(False, index=close.index, columns=close.columns)
78
+ for s in ended[ended].index:
79
+ da.loc[last_row.loc[s, date_col], s] = True
80
+ if ticker_col and ticker_col in d.columns:
81
+ t = d.groupby(ticker_col)[id_col].nunique()
82
+ meta["reused_tickers"] = sorted(t[t > 1].index.astype(str).tolist()) # one ticker, several securities
83
+ p = Panel(close=close, eligible=ok, open=mat(open_col) if open_col else None, high=mat(high_col) if high_col else None,
84
+ low=mat(low_col) if low_col else None, volume=vol, mkt_cap=mat(mktcap_col) if mktcap_col else None,
85
+ market=market, periods_per_year=periods_per_year, entry_lag=entry_lag, delist_after=da)
86
+ p.meta = meta
87
+ return p
@@ -0,0 +1,236 @@
1
+ """Tiingo end-of-day prices (free account) as a point-in-time panel that keeps delisted securities.
2
+
3
+ What it is for
4
+ Tiingo's end-of-day API returns the full history of many US stocks that were later acquired or delisted, which yfinance
5
+ does not. The free tier is small, so this adapter is built for a **random sample** of tickers, not the whole market:
6
+ `draw_order` gives a seeded random order and `fetch_symbols` downloads in that order, resumably, and stops cleanly when
7
+ the account runs out of requests. Any prefix of the order is a random sample.
8
+
9
+ What it does not fix
10
+ A probe (`docs/tiingo_probe.py`) found the history of 23 of 29 takeover or rename cases but of **none** of 12 bankruptcies
11
+ and rescue sales. A panel from this adapter is less survivor-biased than one from yfinance, not unbiased. Treat any
12
+ survivors-versus-this comparison as a lower bound.
13
+
14
+ Choices worth knowing
15
+ - Returns use `adjClose` (split and dividend adjusted, so total return). Eligibility uses the raw `close` and `volume`.
16
+ - A ticker with several listing windows in the ticker list (reuse) becomes several securities, each cut to its own window.
17
+ - A security whose last bar is more than `end_gap_days` before the last calendar date is flagged in `delist_after`.
18
+ - A ticker for which the API has nothing is recorded (`.none`) and counted, never silently replaced.
19
+ - Only the `TIINGO_API_KEY` variable is read.
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import json
24
+ import os
25
+ import re
26
+ import time
27
+ import urllib.error
28
+ import urllib.request
29
+ from pathlib import Path
30
+
31
+ import numpy as np
32
+ import pandas as pd
33
+
34
+ from ..core.panel import Panel
35
+
36
+ BASE = "https://api.tiingo.com/tiingo/daily"
37
+ KEEP = ("date", "close", "volume", "adjClose")
38
+ # With `full=True` the open, high, low (adjusted for splits and dividends like adjClose), the dividend paid and the split factor are kept too.
39
+ FULL = KEEP + ("open", "high", "low", "adjOpen", "adjHigh", "adjLow", "divCash", "splitFactor")
40
+ _NOT_PLAIN = re.compile(r"[-.]")
41
+
42
+
43
+ class QuotaExceeded(RuntimeError):
44
+ """The API refused further requests (rate or monthly limit). Run again later; the cache keeps what was fetched."""
45
+
46
+
47
+ def load_key(env_path: str | Path | None = None) -> str:
48
+ """TIINGO_API_KEY from the environment, else from a .env file. Only that one variable is read."""
49
+ if os.environ.get("TIINGO_API_KEY"):
50
+ return os.environ["TIINGO_API_KEY"]
51
+ if env_path:
52
+ for line in Path(env_path).expanduser().read_text().splitlines():
53
+ k, _, v = line.partition("=")
54
+ if k.strip() == "TIINGO_API_KEY":
55
+ return v.strip().strip('"').strip("'")
56
+ raise KeyError("TIINGO_API_KEY is not set")
57
+
58
+
59
+ # ---------------------------------------------------------------------------------------------------- the sample
60
+ def study_frame(master: pd.DataFrame, since: str = "2013-01-01") -> pd.DataFrame:
61
+ """Rows of the ticker list that can enter a study starting at `since`: plain tickers (no '-' or '.', and no
62
+ five-letter ticker ending in W, U, R or P: warrants, units, rights, preferreds) whose window ends on or after `since`."""
63
+ t = master["ticker"].astype(str)
64
+ plain = ~t.str.contains(_NOT_PLAIN) & ~((t.str.len() == 5) & t.str[-1].isin(list("WURP")))
65
+ return master[plain & (master["end"] >= pd.Timestamp(since))].reset_index(drop=True)
66
+
67
+
68
+ def draw_order(tickers, seed: int = 0) -> list[str]:
69
+ """A seeded random permutation of the unique tickers (sorted first, so it does not depend on the input order).
70
+ Download in this order and every prefix is a simple random sample."""
71
+ u = sorted(set(map(str, tickers)))
72
+ rng = np.random.default_rng(seed)
73
+ return [u[i] for i in rng.permutation(len(u))]
74
+
75
+
76
+ # ---------------------------------------------------------------------------------------------------- downloading
77
+ def _call(ticker: str, start: str, key: str, retries: int = 3):
78
+ """Bars for one ticker, or None if the API does not know it. Raises QuotaExceeded on HTTP 429."""
79
+ url = f"{BASE}/{ticker.lower()}/prices?startDate={start}&format=json"
80
+ req = urllib.request.Request(url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json",
81
+ "User-Agent": "pitbacktest-research/0.2"})
82
+ delay = 1.0
83
+ for i in range(retries + 1):
84
+ try:
85
+ with urllib.request.urlopen(req, timeout=60) as r:
86
+ data = json.loads(r.read())
87
+ # The monthly limit on distinct symbols (500 on a free account) comes back as HTTP 200 with a message, not as a 429.
88
+ if isinstance(data, dict) and re.search(r"run over|allocation|look ?up|upgrade", str(data.get("detail", "")), re.I):
89
+ raise QuotaExceeded(f"HTTP 200: {data.get('detail')}")
90
+ return data
91
+ except urllib.error.HTTPError as e:
92
+ body = e.read()[:300].decode("utf-8", "replace")
93
+ if e.code == 429:
94
+ raise QuotaExceeded(f"HTTP 429: {body}") from None
95
+ if e.code == 404:
96
+ return None
97
+ if e.code in (401, 403):
98
+ raise PermissionError(f"HTTP {e.code}: {body}") from None
99
+ if e.code >= 500 and i < retries:
100
+ time.sleep(delay); delay *= 2
101
+ continue
102
+ raise
103
+ except (urllib.error.URLError, TimeoutError, ConnectionError, ValueError):
104
+ if i < retries:
105
+ time.sleep(delay); delay *= 2
106
+ continue
107
+ raise
108
+ return None
109
+
110
+
111
+ def _check(ticker: str, rows, fields=KEEP) -> list[dict]:
112
+ if not isinstance(rows, list):
113
+ raise RuntimeError(f"Tiingo {ticker}: unexpected response type {type(rows).__name__}")
114
+ out = []
115
+ for r in rows:
116
+ if not all(k in r for k in fields):
117
+ raise RuntimeError(f"Tiingo {ticker}: bar without {[k for k in fields if k not in r]}; the format changed")
118
+ out.append({k: (str(r[k])[:10] if k == "date" else r[k]) for k in fields})
119
+ return out
120
+
121
+
122
+ def _safe(ticker: str) -> str:
123
+ return re.sub(r"[^A-Za-z0-9_]", "_", ticker.upper())
124
+
125
+
126
+ def fetch_symbols(tickers, store_dir, *, start: str = "2012-01-01", key: str | None = None, sleep: float = 0.3,
127
+ max_new: int | None = None, caller=None, progress: bool = True, full: bool = False) -> dict:
128
+ """Download one file per ticker into `store_dir` (`T.json` with the needed fields, or `T.none` if the API has nothing).
129
+ Resumable: tickers already stored are skipped. `max_new` limits requests in this run. Stops cleanly on a quota error.
130
+ `full=True` keeps the open, high, low, dividend and split factor as well (use a different `store_dir` from a store made without it).
131
+ `caller(ticker, start)` can be injected for tests (return None for an unknown ticker)."""
132
+ store = Path(store_dir).expanduser()
133
+ store.mkdir(parents=True, exist_ok=True)
134
+ key = key or (None if caller else load_key())
135
+ call = caller or (lambda t, s: _call(t, s, key))
136
+ done = {"fetched": 0, "none": 0, "skipped": 0, "calls": 0, "stopped": None}
137
+ for t in tickers:
138
+ f = store / f"{_safe(t)}.json"
139
+ n = store / f"{_safe(t)}.none"
140
+ if f.exists() or n.exists():
141
+ done["skipped"] += 1
142
+ continue
143
+ if max_new is not None and done["calls"] >= max_new:
144
+ done["stopped"] = "max_new"
145
+ break
146
+ try:
147
+ rows = call(t, start)
148
+ except QuotaExceeded as e:
149
+ done["stopped"] = f"quota: {e}"
150
+ break
151
+ done["calls"] += 1
152
+ if rows is None or (isinstance(rows, list) and len(rows) == 0):
153
+ n.write_text("")
154
+ done["none"] += 1
155
+ else:
156
+ f.write_text(json.dumps(_check(t, rows, FULL if full else KEEP))) # validated first: a malformed answer is never saved
157
+ done["fetched"] += 1
158
+ if progress and done["calls"] % 25 == 0:
159
+ print(f" {done['calls']} requests, {done['fetched']} with data, {done['none']} none", flush=True)
160
+ time.sleep(sleep)
161
+ return done
162
+
163
+
164
+ # ---------------------------------------------------------------------------------------------------- the panel
165
+ def build_tiingo_panel(store_dir, master: pd.DataFrame, tickers, *, start: str = "2012-01-01", end: str | None = None,
166
+ min_age_bars: int = 60, adv_window: int = 30, min_dollar_volume: float = 2e6, min_price: float = 1.0,
167
+ min_names_per_date: int = 100, end_gap_days: int = 5, entry_lag: int = 1) -> Panel:
168
+ """Point-in-time panel from the stored files. Eligibility on day t uses only data up to t.
169
+
170
+ Each listing window of a ticker in `master` (columns ticker, start, end) is one security, cut to that window.
171
+ The trading calendar is the set of dates on which at least `min_names_per_date` securities have a bar.
172
+ `start` is the first date kept: pass the start of the warm-up year and cut the study period afterwards. The age of a
173
+ security counts bars from the start of the stored data (`fetch_symbols(start=...)`), so a warm-up year of 60+ bars
174
+ makes that irrelevant for the study period."""
175
+ store = Path(store_dir).expanduser()
176
+ adj, raw, vol, oh, div, meta = {}, {}, {}, {"adjOpen": {}, "adjHigh": {}, "adjLow": {}}, {}, {"tickers_with_data": 0, "tickers_none": 0, "tickers_missing": 0, "windows_cut": 0}
177
+ for t in map(str, tickers):
178
+ f, n = store / f"{_safe(t)}.json", store / f"{_safe(t)}.none"
179
+ if n.exists():
180
+ meta["tickers_none"] += 1
181
+ continue
182
+ if not f.exists():
183
+ meta["tickers_missing"] += 1
184
+ continue
185
+ bars = pd.DataFrame(json.loads(f.read_text()))
186
+ if bars.empty:
187
+ meta["tickers_none"] += 1
188
+ continue
189
+ meta["tickers_with_data"] += 1
190
+ bars["date"] = pd.to_datetime(bars["date"])
191
+ bars = bars.drop_duplicates("date", keep="last").set_index("date").sort_index()
192
+ wins = master[master["ticker"].astype(str) == t].sort_values("start")
193
+ if wins.empty:
194
+ wins = pd.DataFrame({"start": [bars.index[0]], "end": [bars.index[-1]]})
195
+ multi = len(wins) > 1
196
+ meta["windows_cut"] += int(multi)
197
+ for _, w in wins.iterrows():
198
+ b = bars.loc[(bars.index >= w["start"]) & (bars.index <= w["end"])]
199
+ b = b[(b["adjClose"] > 0) & (b["close"] > 0)]
200
+ if b.empty:
201
+ continue
202
+ sid = f"{t}@{w['start']:%Y%m%d}" if multi else t
203
+ adj[sid], raw[sid], vol[sid] = b["adjClose"], b["close"], b["volume"]
204
+ if "adjOpen" in b.columns: # a store made with full=True
205
+ for k in oh:
206
+ oh[k][sid] = b[k]
207
+ div[sid] = b["divCash"]
208
+ if not adj:
209
+ raise ValueError("no usable securities in the store")
210
+ A, R, V = pd.DataFrame(adj), pd.DataFrame(raw), pd.DataFrame(vol)
211
+ cal = A.index[(A.notna().sum(axis=1) >= min_names_per_date) & (A.index >= pd.Timestamp(start))]
212
+ if end:
213
+ cal = cal[cal <= pd.Timestamp(end)]
214
+ A, R, V = A.reindex(cal), R.reindex(cal), V.reindex(cal)
215
+ dvol = R * V
216
+ # eligibility, all trailing
217
+ age = A.notna().cumsum()
218
+ med = dvol.rolling(adv_window, min_periods=adv_window).median()
219
+ elig = (age >= min_age_bars) & (med >= min_dollar_volume) & (R >= min_price) & (V > 0) & A.notna()
220
+ last = A.apply(lambda c: c.last_valid_index())
221
+ last_date = cal[-1]
222
+ da = pd.DataFrame(False, index=cal, columns=A.columns)
223
+ for c in A.columns:
224
+ if last[c] is not None and last[c] < last_date - pd.Timedelta(days=end_gap_days):
225
+ da.loc[last[c], c] = True
226
+ r = A.pct_change(fill_method=None)
227
+ extra = {}
228
+ if oh["adjOpen"]:
229
+ O, Hh, Ll = (pd.DataFrame(oh[k]).reindex(index=cal, columns=A.columns) for k in ("adjOpen", "adjHigh", "adjLow"))
230
+ extra = {"open": O, "high": Hh, "low": Ll}
231
+ meta["div_cash"] = pd.DataFrame(div).reindex(index=cal, columns=A.columns).astype("float32") # dollars per share paid that day (raw prices)
232
+ meta["suspect_returns_gt_10x"] = int((r.abs() > 10).sum().sum())
233
+ meta["securities"] = int(A.shape[1])
234
+ meta["raw_close"] = R.astype("float32") # the real price level (dollars), for whole-share sizes; `close` is adjusted
235
+ return Panel(close=A, eligible=elig, volume=dvol / A, market="US", entry_lag=entry_lag, periods_per_year=252,
236
+ delist_after=da, meta=meta, **extra)