firstrate-data 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,432 @@
1
+ import http
2
+ from collections import deque
3
+ from concurrent.futures import Future, ThreadPoolExecutor
4
+ from datetime import date, datetime
5
+ from pathlib import Path
6
+ from typing import Self
7
+
8
+ import requests
9
+ from requests.adapters import HTTPAdapter
10
+ from urllib3.util.retry import Retry
11
+
12
+ from firstrate_data import config
13
+ from firstrate_data.config import DEFAULT_BASE_URL
14
+ from firstrate_data.domain import (
15
+ AssetType,
16
+ TickerListing,
17
+ )
18
+ from firstrate_data.domain.bar_type import BarType
19
+ from firstrate_data.domain.enums import (
20
+ ContinuousFuturesAdjustment,
21
+ ContractFiles,
22
+ DelistedArchive,
23
+ DelistedUpdate,
24
+ EquitiesAdjustment,
25
+ OtherData,
26
+ Period,
27
+ Timeframe,
28
+ Unadjusted,
29
+ )
30
+ from firstrate_data.download import progress
31
+ from firstrate_data.download.bundles import BundleConfig, bundle_requests
32
+ from firstrate_data.download.requests import (
33
+ BarsRequest,
34
+ ContractBarsRequest,
35
+ DelistedBarsRequest,
36
+ LastUpdateRequest,
37
+ OtherDataRequest,
38
+ Request,
39
+ TickerListingRequest,
40
+ )
41
+ from firstrate_data.store.store import Ingested, Store
42
+
43
+ # 1 MiB: large enough that a multi-GB archive isn't paid for one syscall at a
44
+ # time, small enough that a progress bar still moves on a slow line
45
+ CHUNK_SIZE = 1 << 20
46
+
47
+ # a rate limit and the transient 5xx family. A 4xx is the request being wrong,
48
+ # and repeating it won't make it right.
49
+ _RETRYABLE_STATUS = (
50
+ http.HTTPStatus.TOO_MANY_REQUESTS,
51
+ http.HTTPStatus.INTERNAL_SERVER_ERROR,
52
+ http.HTTPStatus.BAD_GATEWAY,
53
+ http.HTTPStatus.SERVICE_UNAVAILABLE,
54
+ http.HTTPStatus.GATEWAY_TIMEOUT,
55
+ )
56
+
57
+ # connecting succeeds fast or not at all. A body, though, arrives at whatever
58
+ # rate the link gives, so the read budget is per-chunk and generous
59
+ _TIMEOUT = (30, 120)
60
+
61
+
62
+ class Client:
63
+ def __init__(
64
+ self,
65
+ user_id: str,
66
+ store: Store,
67
+ base_url: str = DEFAULT_BASE_URL,
68
+ ) -> None:
69
+ self._user_id = user_id
70
+ self._store = store
71
+ self._base_url = base_url.rstrip("/")
72
+
73
+ self._session = _session()
74
+
75
+ @classmethod
76
+ def from_env(cls) -> Self:
77
+ return cls(
78
+ config.firstrate_user_id(),
79
+ Store.from_env(),
80
+ config.base_url(),
81
+ )
82
+
83
+ def download_bundle(
84
+ self,
85
+ bundle_config: BundleConfig,
86
+ prefetch: int = 2,
87
+ *,
88
+ refresh: bool = False,
89
+ ) -> list[Ingested]:
90
+ """Download a bundle, fetching the next archives while one is written.
91
+
92
+ A fetch waits on the vendor and a write on DuckDB, so the two overlap.
93
+ Writes stay on this thread: the store holds one connection, and the
94
+ spool would carry the whole bundle at once if fetches ran unbounded.
95
+
96
+ An archive the store already holds at the vendor's last update is not
97
+ fetched again, and is absent from what this returns. `refresh` fetches
98
+ every archive the bundle names regardless of what is filed.
99
+ """
100
+ ingested: list[Ingested] = []
101
+ fetching: deque[tuple[Request, Future[Path]]] = deque()
102
+ # one last_update call per asset type, and only once something filed
103
+ # turns out to be worth dating
104
+ served: dict[AssetType, date] = {}
105
+
106
+ # `prefetch` worker threads that run beside this one inside this same
107
+ # process, sharing its memory -- so a worker hands back a `Path` with
108
+ # no copying, and the session's connection pool is the one object all
109
+ # of them use. Threads, not processes, because a fetch spends its life
110
+ # blocked on a socket and a blocked thread costs no CPU; processes are
111
+ # what a second CPU's worth of computation would need, and would have
112
+ # to pickle every argument and result across the process boundary.
113
+ # Leaving the `with` waits for every worker still running
114
+ # https://docs.python.org/3/library/concurrent.futures.html#concurrent.futures.ThreadPoolExecutor
115
+ with ThreadPoolExecutor(prefetch, thread_name_prefix="fetch") as fetchers:
116
+ for request in bundle_requests(bundle_config):
117
+ if not refresh and self._is_current(request, served):
118
+ progress.note(f"filed already: {self._spool_name(request)}")
119
+ continue
120
+ # hands `self._fetch(request)` to a worker and returns
121
+ # immediately with a `Future` -- a receipt for an archive that
122
+ # has not arrived. The download runs while this thread carries
123
+ # on to the write below, which is the whole point. Submitting
124
+ # more than there are workers just queues them
125
+ # https://docs.python.org/3/library/concurrent.futures.html#concurrent.futures.Executor.submit
126
+ fetching.append((request, fetchers.submit(self._fetch, request)))
127
+ if len(fetching) > prefetch:
128
+ # take the oldest receipt off the left end and write it,
129
+ # which is what stops the loop from submitting the whole
130
+ # bundle and spooling 64 archives to disk at once
131
+ # https://docs.python.org/3/library/collections.html#collections.deque.popleft
132
+ ingested.append(self._write(*fetching.popleft()))
133
+ while fetching:
134
+ ingested.append(self._write(*fetching.popleft()))
135
+
136
+ return ingested
137
+
138
+ def _is_current(self, request: Request, served: dict[AssetType, date]) -> bool:
139
+ """Whether the store already holds this request's archive, up to date."""
140
+ # a plain full-history archive only. A delisted archive and a contract
141
+ # half name tickers the catalog cannot tell one selector's from
142
+ # another's, and their tickers stopped trading, so no bar of theirs
143
+ # ever reaches the vendor's last update; a metafile leaves no catalog
144
+ # row at all. None of the three can be judged from what is filed
145
+ if type(request) is not BarsRequest:
146
+ return False
147
+
148
+ asset_type = request.bar_type.asset_type
149
+ if asset_type is None:
150
+ return False
151
+
152
+ filed = self.store.last_bar(request.bar_type, request.ticker_range)
153
+ if filed is None:
154
+ return False
155
+
156
+ if asset_type not in served:
157
+ served[asset_type] = _as_date(self.last_update(asset_type))
158
+ return filed.date() >= served[asset_type]
159
+
160
+ def _write(self, request: Request, fetched: Future[Path]) -> Ingested:
161
+ # parks this thread until that worker's archive has landed, then gives
162
+ # back the spooled path. If the worker raised -- a timeout, a 404 --
163
+ # the exception is re-raised here, on this thread, at this line: it is
164
+ # not lost in the worker, and it is not seen until this call
165
+ # https://docs.python.org/3/library/concurrent.futures.html#concurrent.futures.Future.result
166
+ return self.store.write(fetched.result(), request)
167
+
168
+ @property
169
+ def store(self) -> Store:
170
+ return self._store
171
+
172
+ @property
173
+ def spool(self) -> Path:
174
+ return self.store.spool
175
+
176
+ # ------------------------------------------------------------------
177
+ # text endpoints
178
+ # ------------------------------------------------------------------
179
+
180
+ def download_ticker_listing(self, asset_type: AssetType) -> list[TickerListing]:
181
+ listing = TickerListing.from_csv(
182
+ self.read(TickerListingRequest(asset_type)),
183
+ )
184
+ self.store.write_ticker_listing(asset_type, listing)
185
+ return listing
186
+
187
+ # NOTE see docs
188
+ # https://firstratedata.com/about/api-docs?type=stock&userID=xxx#lastupdate
189
+ def last_update(
190
+ self,
191
+ asset_type: AssetType,
192
+ *,
193
+ is_full_update: bool | None = None,
194
+ ) -> date | datetime:
195
+ """The date the vendor last published data for `asset_type`."""
196
+ return self._parse_last_update(
197
+ self.read(LastUpdateRequest(asset_type, is_full_update)),
198
+ )
199
+
200
+ @staticmethod
201
+ def _parse_last_update(body: str) -> date | datetime:
202
+ text = body.strip()
203
+ if not text:
204
+ msg = "last_update answered with an empty body"
205
+ raise ValueError(msg)
206
+
207
+ try:
208
+ # a bare date first: datetime.fromisoformat takes one too, and stamps it
209
+ # with a midnight the vendor said nothing about
210
+ return date.fromisoformat(text)
211
+ except ValueError:
212
+ pass
213
+
214
+ try:
215
+ return datetime.fromisoformat(text)
216
+ except ValueError as unreadable:
217
+ msg = f"last_update answered with {text!r}, which is not a date"
218
+ raise ValueError(
219
+ msg,
220
+ ) from unreadable
221
+
222
+ # ------------------------------------------------------------------
223
+ # Stocks
224
+ # ------------------------------------------------------------------
225
+
226
+ def download_stocks_bars(
227
+ self,
228
+ period: Period,
229
+ timeframe: Timeframe,
230
+ adjustment: EquitiesAdjustment,
231
+ ticker_range: str | None = None,
232
+ ) -> Ingested:
233
+ request = BarsRequest(
234
+ BarType(AssetType.STOCK, timeframe=timeframe, adjustment=adjustment),
235
+ period,
236
+ ticker_range=ticker_range,
237
+ )
238
+ return self._download(request)
239
+
240
+ # Splits / Dividends Requests --------------------------------------
241
+
242
+ def download_splits(self) -> Ingested:
243
+ return self._download(OtherDataRequest(AssetType.STOCK, OtherData.SPLITS))
244
+
245
+ def download_dividends(self) -> Ingested:
246
+ return self._download(OtherDataRequest(AssetType.STOCK, OtherData.DIVIDENDS))
247
+
248
+ # Delisted Ticker Data ---------------------------------------------
249
+
250
+ def download_delisted_bars(
251
+ self,
252
+ selector: DelistedArchive | DelistedUpdate,
253
+ timeframe: Timeframe,
254
+ adjustment: EquitiesAdjustment,
255
+ ) -> Ingested:
256
+ return self._download(
257
+ DelistedBarsRequest(
258
+ BarType(AssetType.STOCK, timeframe=timeframe, adjustment=adjustment),
259
+ selector=selector,
260
+ ),
261
+ )
262
+
263
+ # NOTE no endpoint for this
264
+ # def download_company_profiles(self) -> ...:
265
+ # pass
266
+
267
+ # ------------------------------------------------------------------
268
+ # etf
269
+ # ------------------------------------------------------------------
270
+
271
+ def download_etf_bars(
272
+ self,
273
+ period: Period,
274
+ timeframe: Timeframe,
275
+ adjustment: EquitiesAdjustment,
276
+ ticker_range: str | None = None,
277
+ ) -> Ingested:
278
+ request = BarsRequest(
279
+ BarType(AssetType.ETF, timeframe=timeframe, adjustment=adjustment),
280
+ period,
281
+ ticker_range=ticker_range,
282
+ )
283
+ return self._download(request)
284
+
285
+ # ------------------------------------------------------------------
286
+ # index
287
+ # ------------------------------------------------------------------
288
+
289
+ def download_index_bars(
290
+ self,
291
+ period: Period,
292
+ timeframe: Timeframe,
293
+ ) -> Ingested:
294
+ # UNADJUSTED is the store's word for it, not the vendor's: the request
295
+ # carries no ``adjustment`` on the wire, and the bar type path needs one
296
+ request = BarsRequest(
297
+ BarType(
298
+ AssetType.INDEX,
299
+ timeframe=timeframe,
300
+ adjustment=Unadjusted.UNADJUSTED,
301
+ ),
302
+ period,
303
+ )
304
+ return self._download(request)
305
+
306
+ # ------------------------------------------------------------------
307
+ # futures
308
+ # ------------------------------------------------------------------
309
+
310
+ def download_futures_continuous_bars(
311
+ self,
312
+ period: Period,
313
+ timeframe: Timeframe,
314
+ adjustment: ContinuousFuturesAdjustment,
315
+ ) -> Ingested:
316
+ request = BarsRequest(
317
+ BarType(AssetType.FUTURES, timeframe=timeframe, adjustment=adjustment),
318
+ period,
319
+ )
320
+ return self._download(request)
321
+
322
+ # Individual Contract Data -----------------------------------------
323
+
324
+ def download_futures_contract_bars(
325
+ self,
326
+ contract_files: ContractFiles,
327
+ timeframe: Timeframe,
328
+ ) -> Ingested:
329
+ return self._download(
330
+ ContractBarsRequest(
331
+ BarType(AssetType.FUTURES, timeframe=timeframe),
332
+ contract_files=contract_files,
333
+ ),
334
+ )
335
+
336
+ def download_contract_dates(self) -> Ingested:
337
+ return self._download(
338
+ OtherDataRequest(AssetType.FUTURES, OtherData.CONTRACT_DATES)
339
+ )
340
+
341
+ # ------------------------------------------------------------------
342
+ # fetch, ingest, download
343
+ # ------------------------------------------------------------------
344
+
345
+ def _fetch(
346
+ self,
347
+ request: Request,
348
+ ) -> Path:
349
+ """Fetch a file or archive from FirstRate and return its local path."""
350
+ label = f"{request.endpoint} {'/'.join(request.to_params().values())}"
351
+ destination = self.spool / self._spool_name(request)
352
+
353
+ try:
354
+ with self._get(request, stream=True) as response:
355
+ response.raise_for_status()
356
+ self._stream_body(response, destination, label)
357
+ except (requests.RequestException, OSError):
358
+ # a half-written archive is not an archive, and the spool must not
359
+ # hold anything a later run could mistake for one
360
+ destination.unlink(missing_ok=True)
361
+ raise
362
+
363
+ return destination
364
+
365
+ def read(self, request: Request) -> str:
366
+ with self._get(request) as response:
367
+ response.raise_for_status()
368
+ return response.text
369
+
370
+ def close(self) -> None:
371
+ self._session.close()
372
+
373
+ def _download(self, request: Request) -> Ingested:
374
+ return self.store.write(self._fetch(request), request)
375
+
376
+ # ------------------------------------------------------------------
377
+ # helpers
378
+ # ------------------------------------------------------------------
379
+
380
+ def _get(self, request: Request, *, stream: bool = False) -> requests.Response:
381
+ return self._session.get(
382
+ f"{self._base_url}/{request.endpoint}",
383
+ params={**request.to_params(), "userid": self._user_id},
384
+ timeout=_TIMEOUT,
385
+ stream=stream,
386
+ )
387
+
388
+ def _stream_body(
389
+ self,
390
+ response: requests.Response,
391
+ destination: Path,
392
+ label: str,
393
+ ) -> None:
394
+ """Write responde to the spool dir."""
395
+ declared = response.headers.get("Content-Length")
396
+ with (
397
+ progress.track(label, int(declared) if declared else None, "B") as advance,
398
+ destination.open("wb") as spooled,
399
+ ):
400
+ for chunk in response.iter_content(chunk_size=CHUNK_SIZE):
401
+ spooled.write(chunk)
402
+ advance(len(chunk))
403
+
404
+ @staticmethod
405
+ def _spool_name(request: Request) -> str:
406
+ parts = [request.endpoint, *request.to_params().values()]
407
+ safe = (
408
+ "".join(c if c.isalnum() or c in "-." else "_" for c in p) for p in parts
409
+ )
410
+ return "_".join(safe)
411
+
412
+
413
+ def _as_date(moment: date | datetime) -> date:
414
+ # datetime is a subclass of date, so the narrower test comes first
415
+ # https://docs.python.org/3/library/datetime.html#datetime.datetime
416
+ return moment.date() if isinstance(moment, datetime) else moment
417
+
418
+
419
+ def _session() -> requests.Session:
420
+ session = requests.Session()
421
+ adapter = HTTPAdapter(
422
+ max_retries=Retry(
423
+ total=3,
424
+ backoff_factor=1,
425
+ status_forcelist=_RETRYABLE_STATUS,
426
+ allowed_methods=frozenset({"GET"}),
427
+ respect_retry_after_header=True,
428
+ ),
429
+ )
430
+ session.mount("http://", adapter)
431
+ session.mount("https://", adapter)
432
+ return session
@@ -0,0 +1,45 @@
1
+ from collections.abc import Callable, Generator
2
+ from contextlib import contextmanager
3
+ from typing import Never
4
+
5
+ from tqdm import tqdm
6
+
7
+ # report that n more units of the tracked work have finished
8
+ type Advance = Callable[[int], None]
9
+
10
+
11
+ @contextmanager
12
+ def track(label: str, total: int | None, unit: str) -> Generator[Advance]:
13
+ """Draw one bar for one piece of work during the context.
14
+
15
+ `total` is None when the size is not known up front -- a response that
16
+ declares no Content-Length -- which means a bar without an ETA, not an
17
+ error.
18
+ """
19
+ # Never: update() calls drive this bar. Nothing iterates over it.
20
+ bar: tqdm[Never] = tqdm(
21
+ total=total,
22
+ desc=label,
23
+ unit=unit,
24
+ # bytes read best as MB/GB with a 1024 divisor. Cells stay cells.
25
+ unit_scale=unit == "B",
26
+ unit_divisor=1024,
27
+ dynamic_ncols=True,
28
+ # None means "off when stderr isn't a terminal": a bar redirected to
29
+ # a log file is thousands of redraw frames nobody will read
30
+ disable=None,
31
+ )
32
+
33
+ def advance(units: int) -> None:
34
+ bar.update(units)
35
+
36
+ try:
37
+ yield advance
38
+ finally:
39
+ bar.close()
40
+
41
+
42
+ def note(message: str) -> None:
43
+ """Print a line without breaking a bar that's drawing on the same stream."""
44
+ # https://tqdm.github.io/docs/tqdm/#write
45
+ tqdm.write(message)