themeparks 3.2.0__tar.gz → 3.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {themeparks-3.2.0 → themeparks-3.3.0}/CHANGELOG.md +46 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/PKG-INFO +20 -4
- {themeparks-3.2.0 → themeparks-3.3.0}/README.md +19 -3
- {themeparks-3.2.0 → themeparks-3.3.0}/pyproject.toml +7 -1
- themeparks-3.3.0/tests/fixtures/README.md +37 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/history.py +68 -5
- themeparks-3.3.0/themeparks/backfill.py +884 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/.gitignore +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/LICENSE +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/MIGRATION.md +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/__init__.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_cache.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_client.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/__init__.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/dates.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/destinations.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/entity.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/live.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_errors.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_generated/__init__.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_generated/models.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ratelimit.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_raw.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_transport.py +0 -0
- {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/py.typed +0 -0
|
@@ -1,5 +1,51 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [3.3.0] - 2026-09-28
|
|
4
|
+
|
|
5
|
+
### Added
|
|
6
|
+
|
|
7
|
+
- **`themeparks-backfill`: the archive download as a command.** It was an example
|
|
8
|
+
to copy off GitHub. The first paying customer followed that link and had to
|
|
9
|
+
work out that the library needed installing, then what the arguments were,
|
|
10
|
+
then read a traceback. Now:
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
pip install themeparks
|
|
14
|
+
themeparks-backfill "Disneyland Park"
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
- Takes a park or a **destination**, by name or id. A destination back fills
|
|
18
|
+
every park in it, one file each. `"Walt Disney World Resort"` is the handle
|
|
19
|
+
people actually have; four park uuids is not.
|
|
20
|
+
- `--list [text]` prints destinations with their parks underneath, and **needs
|
|
21
|
+
no key**, so you can find your park before deciding whether to pay.
|
|
22
|
+
- Refuses to guess between two matches. Two parks are named exactly
|
|
23
|
+
"Disneyland Park" (Anaheim and Paris), so the candidate list names the
|
|
24
|
+
destination as well.
|
|
25
|
+
- **Runs without a key**, reading the 7 days anonymous access allows, and says
|
|
26
|
+
what a key would add. It used to refuse to start with a message that
|
|
27
|
+
mentioned anonymous access in the same breath.
|
|
28
|
+
- NDJSON by default, `--format csv` for one wide row per entity per day.
|
|
29
|
+
- Checkpoints against the hourly history budget and exits 75 (`EX_TEMPFAIL`),
|
|
30
|
+
so a cron or timer retries rather than alerting. Re-running continues.
|
|
31
|
+
|
|
32
|
+
`python -m themeparks.backfill` is the same thing. `examples/backfill.py`
|
|
33
|
+
remains as a shim so existing links keep working.
|
|
34
|
+
|
|
35
|
+
### Fixed
|
|
36
|
+
|
|
37
|
+
- **The history window recovery now actually works.** 3.2.0's `examples/backfill.py`
|
|
38
|
+
read `earliestAllowedDate` from the top level of the 403 body; the API nests it
|
|
39
|
+
under `error`. So the recovery shipped doing nothing and a Pro customer still
|
|
40
|
+
got a traceback on their first request. The tests passed because the fixture was
|
|
41
|
+
built from the formatted text in a traceback rather than a real response, so the
|
|
42
|
+
code and the test were wrong together. The fixture is now captured from
|
|
43
|
+
production and a test fails if anyone flattens it.
|
|
44
|
+
|
|
45
|
+
The underlying gap is in the API, not the client: `/history/coverage` reports
|
|
46
|
+
where the archive starts and where your window ends, and nothing about where
|
|
47
|
+
your window begins. Until it does, the 403 is the only place that date exists.
|
|
48
|
+
|
|
3
49
|
## [3.2.0] - 2026-09-26
|
|
4
50
|
|
|
5
51
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: themeparks
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.3.0
|
|
4
4
|
Summary: Official SDK for the ThemeParks.wiki API
|
|
5
5
|
Project-URL: Homepage, https://api.themeparks.wiki
|
|
6
6
|
Project-URL: Source, https://github.com/ThemeParks/ThemeParks_Python
|
|
@@ -350,9 +350,25 @@ except BudgetExhaustedError as exc:
|
|
|
350
350
|
print(f"resume in {exc.retry_after:.0f}s")
|
|
351
351
|
```
|
|
352
352
|
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
353
|
+
### Or skip the code: there is a command
|
|
354
|
+
|
|
355
|
+
Installing the library installs `themeparks-backfill`, which does all of the
|
|
356
|
+
above and stops before the walls:
|
|
357
|
+
|
|
358
|
+
```bash
|
|
359
|
+
themeparks-backfill "Disneyland Park" # a park, by name or id
|
|
360
|
+
themeparks-backfill "Walt Disney World Resort" # a destination: every park in it
|
|
361
|
+
themeparks-backfill --list disney # find an id. Needs no key.
|
|
362
|
+
```
|
|
363
|
+
|
|
364
|
+
It reads how far back your own key may ask and starts there, writes NDJSON or
|
|
365
|
+
`--format csv`, names every row with the park and the entity, records what it
|
|
366
|
+
has done so re-running never duplicates a file, and exits 75 when the hourly
|
|
367
|
+
history budget runs out so a scheduler retries rather than alerts.
|
|
368
|
+
|
|
369
|
+
`python -m themeparks.backfill` is the same thing, which is the one to use if
|
|
370
|
+
`pip install --user` put the script somewhere off your PATH. `themeparks-backfill
|
|
371
|
+
--help` has the rest.
|
|
356
372
|
|
|
357
373
|
## Low-level escape hatch
|
|
358
374
|
|
|
@@ -311,9 +311,25 @@ except BudgetExhaustedError as exc:
|
|
|
311
311
|
print(f"resume in {exc.retry_after:.0f}s")
|
|
312
312
|
```
|
|
313
313
|
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
314
|
+
### Or skip the code: there is a command
|
|
315
|
+
|
|
316
|
+
Installing the library installs `themeparks-backfill`, which does all of the
|
|
317
|
+
above and stops before the walls:
|
|
318
|
+
|
|
319
|
+
```bash
|
|
320
|
+
themeparks-backfill "Disneyland Park" # a park, by name or id
|
|
321
|
+
themeparks-backfill "Walt Disney World Resort" # a destination: every park in it
|
|
322
|
+
themeparks-backfill --list disney # find an id. Needs no key.
|
|
323
|
+
```
|
|
324
|
+
|
|
325
|
+
It reads how far back your own key may ask and starts there, writes NDJSON or
|
|
326
|
+
`--format csv`, names every row with the park and the entity, records what it
|
|
327
|
+
has done so re-running never duplicates a file, and exits 75 when the hourly
|
|
328
|
+
history budget runs out so a scheduler retries rather than alerts.
|
|
329
|
+
|
|
330
|
+
`python -m themeparks.backfill` is the same thing, which is the one to use if
|
|
331
|
+
`pip install --user` put the script somewhere off your PATH. `themeparks-backfill
|
|
332
|
+
--help` has the rest.
|
|
317
333
|
|
|
318
334
|
## Low-level escape hatch
|
|
319
335
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "themeparks"
|
|
7
|
-
version = "3.
|
|
7
|
+
version = "3.3.0"
|
|
8
8
|
description = "Official SDK for the ThemeParks.wiki API"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -30,6 +30,12 @@ dependencies = [
|
|
|
30
30
|
"eval-type-backport>=0.2; python_version < '3.10'",
|
|
31
31
|
]
|
|
32
32
|
|
|
33
|
+
[project.scripts]
|
|
34
|
+
# The archive backfill, as a command rather than a file to copy off GitHub.
|
|
35
|
+
# `pip install themeparks` then `themeparks-backfill "Disneyland Park"` is the
|
|
36
|
+
# whole path from nothing to a file of history.
|
|
37
|
+
themeparks-backfill = "themeparks.backfill:main"
|
|
38
|
+
|
|
33
39
|
[project.urls]
|
|
34
40
|
Homepage = "https://api.themeparks.wiki"
|
|
35
41
|
Source = "https://github.com/ThemeParks/ThemeParks_Python"
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# Fixtures captured from the live API
|
|
2
|
+
|
|
3
|
+
Not hand-written. Each file here was taken from a real response, and the header
|
|
4
|
+
below says which endpoint and when. That matters because of a specific failure
|
|
5
|
+
this project keeps repeating.
|
|
6
|
+
|
|
7
|
+
## Why this directory exists
|
|
8
|
+
|
|
9
|
+
On 2026-09-28 a 403-recovery fix shipped doing nothing. Its tests passed because
|
|
10
|
+
the fixture was built from the *formatted text of a traceback* rather than from a
|
|
11
|
+
response body, so the code and the test were wrong together and agreed with each
|
|
12
|
+
other. The memory note for that pattern is `self-confirming-harness`, and it was
|
|
13
|
+
its fifth occurrence in one day.
|
|
14
|
+
|
|
15
|
+
A fixture the author of the code also invented proves only that the two are
|
|
16
|
+
consistent. A fixture taken from the real thing can disagree.
|
|
17
|
+
|
|
18
|
+
## destinations_slice.json
|
|
19
|
+
|
|
20
|
+
`GET https://api.themeparks.wiki/v1/destinations`, captured 2026-09-28, trimmed
|
|
21
|
+
to 9 destinations and 20 parks and otherwise verbatim — every id, name and slug
|
|
22
|
+
is exactly what the API returned.
|
|
23
|
+
|
|
24
|
+
It is trimmed to keep, deliberately, every case that broke name resolution:
|
|
25
|
+
|
|
26
|
+
| Case | Why it is here |
|
|
27
|
+
|---|---|
|
|
28
|
+
| `Walt Disney World® Resort` | U+00AE. The command's own documented example, `"Walt Disney World Resort"`, did not match it. |
|
|
29
|
+
| `LEGOLAND® Korea` | the same, on a destination whose park shares the name |
|
|
30
|
+
| `Walibi Rhône-Alpes` | U+00F4, so a normaliser must decompose accents |
|
|
31
|
+
| `Knott's Berry Farm` + `Knott’s Soak City` | ASCII `'` and U+2019 **in one destination**. Nobody types the curly one. |
|
|
32
|
+
| `Disneyland Park` ×2 | Anaheim and Paris, identical park names. The reason the candidate list names the destination. |
|
|
33
|
+
| `Hurricane Harbor` + `Hurricane Harbor Chicago` | an exact match with a substring rival: the shape that silently downloaded the wrong park |
|
|
34
|
+
| `Cedar Point` | destination name == park name, with a second park, so "exact destination beats exact park" is observable |
|
|
35
|
+
|
|
36
|
+
**Do not edit these by hand.** Re-capture them. If a name upstream has drifted,
|
|
37
|
+
that is a real change and the test should notice.
|
|
@@ -111,17 +111,54 @@ def _reraise_if_too_long(exc: RateLimitError, max_wait: float) -> None:
|
|
|
111
111
|
raise exc
|
|
112
112
|
|
|
113
113
|
|
|
114
|
-
|
|
115
|
-
"""
|
|
116
|
-
|
|
114
|
+
class EntityRef(NamedTuple):
|
|
115
|
+
"""Who a history row belongs to, AS THE HISTORY RESPONSE REPORTS IT.
|
|
116
|
+
|
|
117
|
+
The name matters and the source of it matters more. A park's current
|
|
118
|
+
`/children` list gives today's name, which is the wrong label for a row
|
|
119
|
+
recorded years ago: rides are renamed, and stamping today's name on old data
|
|
120
|
+
quietly rewrites history. The history envelope carries its own `name` and
|
|
121
|
+
`entityType` per entity, and that is the name to use.
|
|
122
|
+
"""
|
|
123
|
+
|
|
124
|
+
id: str
|
|
125
|
+
name: str
|
|
126
|
+
entity_type: str
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _ref(entity: Any) -> EntityRef:
|
|
130
|
+
kind = getattr(entity, "entityType", None)
|
|
131
|
+
# The generated models use an enum, and str(EntityType.SHOW) is
|
|
132
|
+
# "EntityType.SHOW". `.value` is what the API sends.
|
|
133
|
+
inner = getattr(kind, "value", kind)
|
|
134
|
+
return EntityRef(
|
|
135
|
+
entity.id, getattr(entity, "name", "") or "", "" if inner is None else str(inner)
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _daily_entity_rows(envelope: DailyEnvelope) -> Iterator[tuple[EntityRef, HistoryDailyRow]]:
|
|
140
|
+
"""Yield (entity ref, row), keeping the name the response gave.
|
|
141
|
+
|
|
142
|
+
`_daily_rows` below is the same walk with the ref flattened to its id, kept
|
|
143
|
+
because `days()` has yielded `(id, row)` since 3.0 and that shape is public.
|
|
144
|
+
"""
|
|
117
145
|
entities = getattr(envelope, "entities", None)
|
|
118
146
|
if entities is not None:
|
|
119
147
|
for entity in entities:
|
|
148
|
+
ref = _ref(entity)
|
|
120
149
|
for row in entity.days or []:
|
|
121
|
-
yield (
|
|
150
|
+
yield (ref, row)
|
|
122
151
|
return
|
|
152
|
+
ref = _ref(envelope)
|
|
123
153
|
for row in getattr(envelope, "days", None) or []:
|
|
124
|
-
yield (
|
|
154
|
+
yield (ref, row)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _daily_rows(envelope: DailyEnvelope) -> Iterator[tuple[str, HistoryDailyRow]]:
|
|
158
|
+
"""Yield (entity id, row). A park envelope carries many entities; an entity
|
|
159
|
+
envelope carries its own rows, so both flatten to the same stream."""
|
|
160
|
+
for ref, row in _daily_entity_rows(envelope):
|
|
161
|
+
yield (ref.id, row)
|
|
125
162
|
|
|
126
163
|
|
|
127
164
|
def _raw_rows(envelope: RawEnvelope) -> Iterator[tuple[str, HistoryRow]]:
|
|
@@ -154,6 +191,29 @@ class HistoryApi:
|
|
|
154
191
|
"""
|
|
155
192
|
return _span(self.coverage())
|
|
156
193
|
|
|
194
|
+
def days_with_entities(
|
|
195
|
+
self,
|
|
196
|
+
start: str | _date | None = None,
|
|
197
|
+
end: str | _date | None = None,
|
|
198
|
+
*,
|
|
199
|
+
max_wait: float = DEFAULT_MAX_WAIT_SECONDS,
|
|
200
|
+
) -> Iterator[tuple[EntityRef, HistoryDailyRow]]:
|
|
201
|
+
"""`days()`, but each row arrives with the entity's name and type.
|
|
202
|
+
|
|
203
|
+
Use this when you are writing history to a file. The name comes from the
|
|
204
|
+
history response itself, so it is the label that response gives for those
|
|
205
|
+
rows rather than the park's current `/children` list -- rides get renamed,
|
|
206
|
+
and today's name on a row from three years ago is a quiet rewrite of the
|
|
207
|
+
record.
|
|
208
|
+
|
|
209
|
+
It also saves a request: the name is already in the payload, so nothing
|
|
210
|
+
needs to ask what an id refers to.
|
|
211
|
+
"""
|
|
212
|
+
envelope: DailyEnvelope | None = self._first_daily(start, end, max_wait)
|
|
213
|
+
while envelope is not None:
|
|
214
|
+
yield from _daily_entity_rows(envelope)
|
|
215
|
+
envelope = self._next_daily(envelope, max_wait)
|
|
216
|
+
|
|
157
217
|
def days(
|
|
158
218
|
self,
|
|
159
219
|
start: str | _date | None = None,
|
|
@@ -165,6 +225,9 @@ class HistoryApi:
|
|
|
165
225
|
|
|
166
226
|
Pages automatically. Given a park id this uses the park call, which
|
|
167
227
|
answers every entity in the park in one request.
|
|
228
|
+
|
|
229
|
+
`days_with_entities()` is the same stream with the entity's name and type
|
|
230
|
+
attached; this shape is kept because it is public API from 3.0.
|
|
168
231
|
"""
|
|
169
232
|
envelope: DailyEnvelope | None = self._first_daily(start, end, max_wait)
|
|
170
233
|
while envelope is not None:
|
|
@@ -0,0 +1,884 @@
|
|
|
1
|
+
"""Download a park's whole daily history to a file, and survive the budget.
|
|
2
|
+
|
|
3
|
+
Run it:
|
|
4
|
+
|
|
5
|
+
themeparks-backfill "Disneyland Park"
|
|
6
|
+
python -m themeparks.backfill "Disneyland Park" # same thing
|
|
7
|
+
|
|
8
|
+
WHY THIS IS IN THE PACKAGE RATHER THAN AN EXAMPLE TO COPY. It used to be
|
|
9
|
+
`examples/backfill.py` on GitHub, linked from the docs. The first paying customer
|
|
10
|
+
followed that link and had to work out that the library needed installing, then
|
|
11
|
+
what the arguments were, then read a traceback. Every one of those is a wall
|
|
12
|
+
between someone who has paid for the archive and the archive. A recipe you have
|
|
13
|
+
to reconstruct is not a recipe; this is one command.
|
|
14
|
+
|
|
15
|
+
What it does that is easy to get wrong by hand:
|
|
16
|
+
|
|
17
|
+
1. It asks the PARK, not the rides. Both history endpoints answer every entity in
|
|
18
|
+
a park in one request, so a park-level backfill of a large resort is around a
|
|
19
|
+
hundred times fewer calls than the same data pulled ride by ride.
|
|
20
|
+
|
|
21
|
+
2. It bounds the range at BOTH ends. `span().retrievable_through` is the latest
|
|
22
|
+
day your key may ask for. There is no field for the earliest -- coverage
|
|
23
|
+
reports where the archive starts, which on any plan short of the full archive
|
|
24
|
+
is before your window -- so the first request is refused and the floor is read
|
|
25
|
+
out of that 403. See `_window_floor`.
|
|
26
|
+
|
|
27
|
+
3. It records what it has done, in `<park-id>.backfill-state.json`: the format,
|
|
28
|
+
the range, the furthest day written, and whether it finished. The history
|
|
29
|
+
budget is hourly, so a spent one can be most of an hour from resetting; the
|
|
30
|
+
SDK raises BudgetExhaustedError rather than sleeping through that, and this
|
|
31
|
+
writes the state and exits 75 (EX_TEMPFAIL) so a scheduler retries rather
|
|
32
|
+
than alerts.
|
|
33
|
+
|
|
34
|
+
Re-running is then safe in every direction: an unfinished park continues, a
|
|
35
|
+
FINISHED park is left alone rather than appended to twice, and a file this
|
|
36
|
+
command did not write is never touched without `--overwrite`. It re-reads the
|
|
37
|
+
furthest day on purpose -- a page can end mid-day -- so `(entityId, date)` is
|
|
38
|
+
the natural key if you load blind.
|
|
39
|
+
|
|
40
|
+
4. It takes NAMES as well as ids, and DESTINATIONS as well as parks. A
|
|
41
|
+
customer has "Walt Disney World Resort", not four park uuids, and making
|
|
42
|
+
them look those up first was another wall. A destination back fills every
|
|
43
|
+
park in it, into one file each.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
from __future__ import annotations
|
|
47
|
+
|
|
48
|
+
import argparse
|
|
49
|
+
import contextlib
|
|
50
|
+
import csv
|
|
51
|
+
import json
|
|
52
|
+
import os
|
|
53
|
+
import sys
|
|
54
|
+
import unicodedata
|
|
55
|
+
from datetime import date
|
|
56
|
+
from pathlib import Path
|
|
57
|
+
from typing import Any, NamedTuple, TextIO, Union
|
|
58
|
+
|
|
59
|
+
from themeparks import APIError, BudgetExhaustedError, RateLimitError, ThemeParks
|
|
60
|
+
from themeparks._ergonomic.history import EntityRef
|
|
61
|
+
|
|
62
|
+
USER_AGENT = "themeparks-backfill/1"
|
|
63
|
+
|
|
64
|
+
EX_TEMPFAIL = 75
|
|
65
|
+
|
|
66
|
+
CSV_COLUMNS = [
|
|
67
|
+
# Identity first. A reader opening this in a spreadsheet should know what a
|
|
68
|
+
# row is before they reach the numbers, and a table loaded from several files
|
|
69
|
+
# needs parkId to tell them apart.
|
|
70
|
+
"parkId",
|
|
71
|
+
"parkName",
|
|
72
|
+
"entityId",
|
|
73
|
+
"entityName",
|
|
74
|
+
"entityType",
|
|
75
|
+
"date",
|
|
76
|
+
"firstOperatingAt",
|
|
77
|
+
"lastClosedAt",
|
|
78
|
+
"operatingMinutes",
|
|
79
|
+
"downMinutes",
|
|
80
|
+
"showCount",
|
|
81
|
+
"changes",
|
|
82
|
+
"standbyMin",
|
|
83
|
+
"standbyP50",
|
|
84
|
+
"standbyMean",
|
|
85
|
+
"standbyP90",
|
|
86
|
+
"standbyMax",
|
|
87
|
+
"singleRiderP50",
|
|
88
|
+
"singleRiderMax",
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _csv_row(ref: EntityRef, row: Any, ident: _RowIdentity) -> dict[str, Any]:
|
|
93
|
+
"""Flatten the nested standby/singleRider stats into one wide row."""
|
|
94
|
+
standby = row.standby
|
|
95
|
+
single = row.singleRider
|
|
96
|
+
return {
|
|
97
|
+
"parkId": ident.park.id,
|
|
98
|
+
"parkName": ident.park.name,
|
|
99
|
+
"entityId": ref.id,
|
|
100
|
+
"entityName": ref.name,
|
|
101
|
+
"entityType": ref.entity_type,
|
|
102
|
+
"date": row.date.isoformat(),
|
|
103
|
+
"firstOperatingAt": row.firstOperatingAt.isoformat() if row.firstOperatingAt else "",
|
|
104
|
+
"lastClosedAt": row.lastClosedAt.isoformat() if row.lastClosedAt else "",
|
|
105
|
+
"operatingMinutes": row.operatingMinutes,
|
|
106
|
+
"downMinutes": row.downMinutes,
|
|
107
|
+
"showCount": row.showCount if row.showCount is not None else "",
|
|
108
|
+
"changes": row.changes,
|
|
109
|
+
"standbyMin": standby.min if standby else "",
|
|
110
|
+
"standbyP50": standby.p50 if standby else "",
|
|
111
|
+
"standbyMean": standby.mean if standby else "",
|
|
112
|
+
"standbyP90": standby.p90 if standby else "",
|
|
113
|
+
"standbyMax": standby.max if standby else "",
|
|
114
|
+
"singleRiderP50": single.p50 if single else "",
|
|
115
|
+
"singleRiderMax": single.max if single else "",
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class _Park(NamedTuple):
|
|
120
|
+
"""A park's identity, so rows can name themselves.
|
|
121
|
+
|
|
122
|
+
`backfill_park` used to take only the id, so every row carried a bare
|
|
123
|
+
`entityId` and nothing else. A customer loading two files into one table
|
|
124
|
+
could not tell the parks apart -- the park id existed only in the FILENAME --
|
|
125
|
+
and 60 attraction GUIDs with no labels meant writing the lookup code they
|
|
126
|
+
bought this to avoid.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
id: str
|
|
130
|
+
name: str
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
class _RowIdentity(NamedTuple):
|
|
134
|
+
"""What a row inherits from the run, as opposed to from the response.
|
|
135
|
+
|
|
136
|
+
Only the park. The entity's name and type come from the history response
|
|
137
|
+
itself (`days_with_entities`), because those are per row and change over
|
|
138
|
+
time: a ride renamed in 2024 must not have its 2021 rows relabelled with
|
|
139
|
+
today's name. The park id is here rather than in the filename alone so two
|
|
140
|
+
files can be loaded into one table.
|
|
141
|
+
"""
|
|
142
|
+
|
|
143
|
+
park: _Park
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class Writer:
|
|
147
|
+
"""NDJSON or CSV behind one `write(entity_id, row)`."""
|
|
148
|
+
|
|
149
|
+
def __init__(self, handle: TextIO, fmt: str, write_header: bool, ident: _RowIdentity) -> None:
|
|
150
|
+
self._handle = handle
|
|
151
|
+
self._fmt = fmt
|
|
152
|
+
self._ident = ident
|
|
153
|
+
self._csv = None
|
|
154
|
+
if fmt == "csv":
|
|
155
|
+
self._csv = csv.DictWriter(handle, fieldnames=CSV_COLUMNS)
|
|
156
|
+
if write_header:
|
|
157
|
+
self._csv.writeheader()
|
|
158
|
+
|
|
159
|
+
def write(self, ref: EntityRef, row: Any) -> None:
|
|
160
|
+
if self._csv is not None:
|
|
161
|
+
self._csv.writerow(_csv_row(ref, row, self._ident))
|
|
162
|
+
return
|
|
163
|
+
# Identity keys come FIRST in the object, so a human reading one line of
|
|
164
|
+
# NDJSON sees what it is before the numbers.
|
|
165
|
+
payload = {
|
|
166
|
+
"parkId": self._ident.park.id,
|
|
167
|
+
"parkName": self._ident.park.name,
|
|
168
|
+
"entityId": ref.id,
|
|
169
|
+
"entityName": ref.name,
|
|
170
|
+
"entityType": ref.entity_type,
|
|
171
|
+
**row.model_dump(mode="json"),
|
|
172
|
+
}
|
|
173
|
+
self._handle.write(json.dumps(payload) + "\n")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _use_utf8(*streams: TextIO) -> None:
|
|
177
|
+
"""Stop a legacy Windows code page killing the run on a park name.
|
|
178
|
+
|
|
179
|
+
stdout's error handler is `strict`, and real names carry characters cp437,
|
|
180
|
+
cp850 and cp932 cannot encode -- `Walt Disney World® Resort`,
|
|
181
|
+
`LEGOLAND® Korea`, `Knott’s Soak City`, `Walibi Rhône-Alpes`. So `--list`
|
|
182
|
+
died with UnicodeEncodeError partway through, after writing 200 good lines,
|
|
183
|
+
whenever output was redirected on a non-1252 system. Redirecting is the
|
|
184
|
+
obvious thing to do with 358 lines of parks.
|
|
185
|
+
|
|
186
|
+
`reconfigure` is 3.7+; `backslashreplace` degrades an unencodable character
|
|
187
|
+
to an escape instead of ending the run.
|
|
188
|
+
"""
|
|
189
|
+
for stream in streams:
|
|
190
|
+
reconfigure = getattr(stream, "reconfigure", None)
|
|
191
|
+
if reconfigure is not None:
|
|
192
|
+
with contextlib.suppress(OSError, ValueError):
|
|
193
|
+
reconfigure(encoding="utf-8", errors="backslashreplace")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _window_floor(exc: APIError) -> str | None:
|
|
197
|
+
"""The earliest day this key may ask for, read out of a 403 body.
|
|
198
|
+
|
|
199
|
+
THE BODY IS NESTED, and the first version of this function was not. What the
|
|
200
|
+
API actually sends is:
|
|
201
|
+
|
|
202
|
+
403 {"error": {"type": "HISTORY_WINDOW_EXCEEDED",
|
|
203
|
+
"message": "This key can see history back to 2025-08-25 (400 days).",
|
|
204
|
+
"earliestAllowedDate": "2025-08-25"}}
|
|
205
|
+
|
|
206
|
+
The first version read those keys off the TOP level, because it was written
|
|
207
|
+
from the formatted text in a traceback rather than from a real response. Its
|
|
208
|
+
tests passed -- they built the fixture the same wrong way -- and it shipped
|
|
209
|
+
doing nothing at all. `tests/unit/test_backfill.py` now pins a body captured
|
|
210
|
+
from production, which is the only version of this test that can fail.
|
|
211
|
+
|
|
212
|
+
Both shapes are accepted: the nested one the API sends, and a bare one, so
|
|
213
|
+
this cannot break again if an error envelope is ever flattened.
|
|
214
|
+
|
|
215
|
+
`/history/coverage` does not carry the floor -- it reports where the archive
|
|
216
|
+
starts and where your window ends, and nothing in between -- so the 403 is
|
|
217
|
+
the only place this date exists. Exposing it on the coverage document is
|
|
218
|
+
tracked upstream.
|
|
219
|
+
"""
|
|
220
|
+
body = exc.body
|
|
221
|
+
if not isinstance(body, dict):
|
|
222
|
+
return None
|
|
223
|
+
inner = body.get("error")
|
|
224
|
+
payload = inner if isinstance(inner, dict) else body
|
|
225
|
+
if payload.get("type") != "HISTORY_WINDOW_EXCEEDED":
|
|
226
|
+
return None
|
|
227
|
+
floor = payload.get("earliestAllowedDate")
|
|
228
|
+
return floor if isinstance(floor, str) and floor else None
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
# --------------------------------------------------------------------------
|
|
232
|
+
# Per-park state. One file, and it closes five separate defects.
|
|
233
|
+
# --------------------------------------------------------------------------
|
|
234
|
+
#
|
|
235
|
+
# The old scheme was a bare `<park-id>.checkpoint` holding one date, and "this
|
|
236
|
+
# park is finished" was encoded as THE ABSENCE of that file -- which is
|
|
237
|
+
# indistinguishable from "never started". The data file was always opened in
|
|
238
|
+
# append mode. Between them that produced:
|
|
239
|
+
#
|
|
240
|
+
# - re-running a finished park appended a second complete copy, silently.
|
|
241
|
+
# Measured: 382 rows became 764.
|
|
242
|
+
# - a destination run that hit the hourly budget re-downloaded every COMPLETED
|
|
243
|
+
# park in full on each retry, spending the new budget on work already done,
|
|
244
|
+
# so a later park might never advance while the finished files grew by a
|
|
245
|
+
# copy an hour.
|
|
246
|
+
# - switching --format mid-resume wrote a new file starting at the checkpoint
|
|
247
|
+
# day and silently lost everything before it, exit 0.
|
|
248
|
+
# - a truncated or empty checkpoint sent `?from=&to=` forever, with no way to
|
|
249
|
+
# know a hidden file was the cause.
|
|
250
|
+
# - nothing recorded which range had been written, so nothing could tell.
|
|
251
|
+
#
|
|
252
|
+
# So state is explicit: the format, the range asked for, the high-water day, and
|
|
253
|
+
# whether it finished. Unreadable state is treated as no state rather than
|
|
254
|
+
# crashing -- a corrupt file must not be a permanent wall.
|
|
255
|
+
STATE_SUFFIX = ".backfill-state.json"
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _read_state(path: Path) -> dict[str, Any]:
|
|
259
|
+
try:
|
|
260
|
+
value = json.loads(path.read_text(encoding="utf-8"))
|
|
261
|
+
except (OSError, ValueError):
|
|
262
|
+
return {}
|
|
263
|
+
return value if isinstance(value, dict) else {}
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _write_state(path: Path, **fields: Any) -> None:
|
|
267
|
+
path.write_text(json.dumps(fields, sort_keys=True) + "\n", encoding="utf-8")
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
class _StateFile(NamedTuple):
|
|
271
|
+
"""Where the state lives and the range it describes, fixed for one park."""
|
|
272
|
+
|
|
273
|
+
path: Path
|
|
274
|
+
fmt: str
|
|
275
|
+
start: Day
|
|
276
|
+
end: Day
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _record(sf: _StateFile, last_day: date | None, *, complete: bool) -> None:
|
|
280
|
+
"""Write the state file. `complete` is the fact the old checkpoint could not express.
|
|
281
|
+
|
|
282
|
+
`sf.start` is the ORIGINAL start of the range, not the day a resumed run
|
|
283
|
+
happened to begin at. The two call sites used to disagree about that, so a
|
|
284
|
+
run interrupted twice recorded the second resume point as though it were the
|
|
285
|
+
beginning and lost the real range.
|
|
286
|
+
"""
|
|
287
|
+
_write_state(
|
|
288
|
+
sf.path,
|
|
289
|
+
format=sf.fmt,
|
|
290
|
+
start=str(sf.start),
|
|
291
|
+
end=str(sf.end),
|
|
292
|
+
last_day=last_day.isoformat() if last_day else None,
|
|
293
|
+
complete=complete,
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
Day = Union[str, date, None]
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _is_empty_window(first_day: Day, end: Day) -> bool:
|
|
301
|
+
"""True when the plan's floor sits past the park's last day of data.
|
|
302
|
+
|
|
303
|
+
`end` is the newest day the park has data for; the 403 recovery clamps the
|
|
304
|
+
start UP to the first day this key may read. For a park that stopped
|
|
305
|
+
reporting before the window opens -- a seasonal water park, a closed ride --
|
|
306
|
+
the clamp can push start past end, and the API answers
|
|
307
|
+
`400 INVALID_RANGE: to must not be before from`. That killed a six-park
|
|
308
|
+
destination run three parks in, leaving a 0-byte file and two parks never
|
|
309
|
+
attempted, on every plan.
|
|
310
|
+
"""
|
|
311
|
+
return end is not None and first_day is not None and str(first_day) > str(end)
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _say_empty(end: Day, start: Day | None = None) -> None:
|
|
315
|
+
reach = f", and your plan reaches back to {start}" if start is not None else ""
|
|
316
|
+
print(
|
|
317
|
+
f" nothing in your window: this park's data ends {end}{reach} — skipping",
|
|
318
|
+
file=sys.stderr,
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
class _Plan(NamedTuple):
|
|
323
|
+
"""What a run should do about a park, once its state file has been read."""
|
|
324
|
+
|
|
325
|
+
start: Day
|
|
326
|
+
has_rows: bool
|
|
327
|
+
prior_start: str | None
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _decide(
|
|
331
|
+
out_path: Path, state_path: Path, fmt: str, overwrite: bool, archive_from: Day
|
|
332
|
+
) -> _Plan | int:
|
|
333
|
+
"""A `_Plan` to proceed with, or an exit code meaning "do not".
|
|
334
|
+
|
|
335
|
+
Split out of `backfill_park` because it got long enough for ruff to object,
|
|
336
|
+
and ruff was right: deciding whether to write is a different job from
|
|
337
|
+
writing. Every branch here exists for a defect measured in review -- see the
|
|
338
|
+
STATE block above for the five of them.
|
|
339
|
+
"""
|
|
340
|
+
state = {} if overwrite else _read_state(state_path)
|
|
341
|
+
if overwrite:
|
|
342
|
+
out_path.unlink(missing_ok=True)
|
|
343
|
+
state_path.unlink(missing_ok=True)
|
|
344
|
+
|
|
345
|
+
file_exists = out_path.exists() and out_path.stat().st_size > 0
|
|
346
|
+
same_format = state.get("format") == fmt
|
|
347
|
+
|
|
348
|
+
# Finished already. Say so and stop, rather than appending a second copy.
|
|
349
|
+
if state.get("complete") and same_format and file_exists:
|
|
350
|
+
print(
|
|
351
|
+
f" already complete: {state.get('start')} .. {state.get('end')} "
|
|
352
|
+
f"in {out_path.name} — pass --overwrite to fetch it again",
|
|
353
|
+
file=sys.stderr,
|
|
354
|
+
)
|
|
355
|
+
return 0
|
|
356
|
+
|
|
357
|
+
# A file we have no record of writing. Refusing is the only safe answer:
|
|
358
|
+
# appending doubles it, truncating throws away someone's data.
|
|
359
|
+
if file_exists and not state:
|
|
360
|
+
print(
|
|
361
|
+
f" {out_path.name} already has rows and there is no state file beside it.\n"
|
|
362
|
+
f" --overwrite replace it\n"
|
|
363
|
+
f" or move it aside and run again",
|
|
364
|
+
file=sys.stderr,
|
|
365
|
+
)
|
|
366
|
+
return 1
|
|
367
|
+
|
|
368
|
+
# A format switch cannot resume: the half-written file is the other format.
|
|
369
|
+
if state and not same_format and not state.get("complete"):
|
|
370
|
+
other = state.get("format")
|
|
371
|
+
print(
|
|
372
|
+
f" {other} was interrupted part-way for this park. Finish it in "
|
|
373
|
+
f"{other}, or pass --overwrite to start again in {fmt}",
|
|
374
|
+
file=sys.stderr,
|
|
375
|
+
)
|
|
376
|
+
return 1
|
|
377
|
+
|
|
378
|
+
resuming = bool(state) and same_format and not state.get("complete")
|
|
379
|
+
last_written = state.get("last_day") if resuming else None
|
|
380
|
+
return _Plan(
|
|
381
|
+
start=last_written or archive_from,
|
|
382
|
+
has_rows=file_exists and resuming,
|
|
383
|
+
prior_start=state.get("start") if resuming else None,
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
class _Job(NamedTuple):
|
|
388
|
+
"""Everything streaming one park needs, so the streamer takes two arguments."""
|
|
389
|
+
|
|
390
|
+
history: Any
|
|
391
|
+
out_path: Path
|
|
392
|
+
fmt: str
|
|
393
|
+
end: Day
|
|
394
|
+
has_rows: bool
|
|
395
|
+
ident: _RowIdentity
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
class _Progress:
|
|
399
|
+
"""How far the stream got. MUTABLE, and that is the point.
|
|
400
|
+
|
|
401
|
+
`_stream` used to return this as a tuple, which meant an exception threw the
|
|
402
|
+
numbers away: the budget handler then had nothing to record and read the state
|
|
403
|
+
file back instead, which on a first run does not exist. The resume point was
|
|
404
|
+
lost on exactly the interruption it exists for, and a re-run started over.
|
|
405
|
+
|
|
406
|
+
Owned by the caller, updated in place, so it is readable after a raise.
|
|
407
|
+
"""
|
|
408
|
+
|
|
409
|
+
def __init__(self) -> None:
|
|
410
|
+
self.written = 0
|
|
411
|
+
self.last_day: date | None = None
|
|
412
|
+
self.skipped = False
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def _stream(job: _Job, start: Day, progress: _Progress) -> None:
|
|
416
|
+
"""Write the range to the file, recovering once from a window 403.
|
|
417
|
+
|
|
418
|
+
Separated from `backfill_park` because that function was deciding, printing,
|
|
419
|
+
streaming and recording in one place, and ruff counted the statements before
|
|
420
|
+
a reader had to. This is the streaming.
|
|
421
|
+
"""
|
|
422
|
+
|
|
423
|
+
def write_rows(writer: Writer, first_day: Day) -> None:
|
|
424
|
+
"""Stream one range into the file. Raises whatever the SDK raises."""
|
|
425
|
+
for ref, row in job.history.days_with_entities(first_day, job.end):
|
|
426
|
+
writer.write(ref, row)
|
|
427
|
+
progress.written += 1
|
|
428
|
+
# MAX, not last-seen. `_daily_rows` walks entities and then each
|
|
429
|
+
# entity's days, so the final row belongs to the alphabetically last
|
|
430
|
+
# entity, which may have stopped reporting mid-page. Taking it as the
|
|
431
|
+
# high-water mark could rewind the resume point by up to a whole
|
|
432
|
+
# 31-day page, while the module claimed the overlap was "one day".
|
|
433
|
+
progress.last_day = (
|
|
434
|
+
row.date if progress.last_day is None else max(progress.last_day, row.date)
|
|
435
|
+
)
|
|
436
|
+
if progress.written % 5000 == 0:
|
|
437
|
+
print(f" {progress.written} rows, at {progress.last_day}", file=sys.stderr)
|
|
438
|
+
|
|
439
|
+
with job.out_path.open("a", newline="", encoding="utf-8") as handle:
|
|
440
|
+
# ONE Writer for the whole park, so the header decision is made once. It
|
|
441
|
+
# used to be built inside write_rows with `written == 0` in the predicate,
|
|
442
|
+
# and the recovery below calls that again precisely when written is 0 --
|
|
443
|
+
# so every CSV on every plan short of the full archive got TWO headers,
|
|
444
|
+
# and pandas read the second as data.
|
|
445
|
+
writer = Writer(handle, job.fmt, not job.has_rows, job.ident)
|
|
446
|
+
try:
|
|
447
|
+
write_rows(writer, start)
|
|
448
|
+
except APIError as exc:
|
|
449
|
+
floor = _window_floor(exc)
|
|
450
|
+
# Retry only when nothing was written: a 403 mid-stream is not a plan
|
|
451
|
+
# boundary, and restarting would duplicate rows.
|
|
452
|
+
if floor is None or progress.written:
|
|
453
|
+
raise
|
|
454
|
+
print(
|
|
455
|
+
f" this key reaches back to {floor}, not {start} — starting there",
|
|
456
|
+
file=sys.stderr,
|
|
457
|
+
)
|
|
458
|
+
if _is_empty_window(floor, job.end):
|
|
459
|
+
# Only visible after the clamp, and the file exists by now because
|
|
460
|
+
# opening it created it. Flagged, not returned, so the empty file
|
|
461
|
+
# is removed after the handle closes.
|
|
462
|
+
_say_empty(job.end)
|
|
463
|
+
progress.skipped = True
|
|
464
|
+
else:
|
|
465
|
+
write_rows(writer, floor)
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def backfill_park(
|
|
469
|
+
tp: ThemeParks, park: _Park, out_dir: Path, fmt: str, overwrite: bool = False
|
|
470
|
+
) -> int:
|
|
471
|
+
"""Write one park's daily history. Returns 0, or EX_TEMPFAIL if the budget ran out."""
|
|
472
|
+
park_id = park.id
|
|
473
|
+
history = tp.entity(park_id).history
|
|
474
|
+
|
|
475
|
+
# SPAN IS INSIDE THE BUDGET HANDLING, and it was not.
|
|
476
|
+
#
|
|
477
|
+
# `coverage()` is the one history call the SDK does not wrap in
|
|
478
|
+
# `_reraise_if_too_long`, so a spent hourly budget surfaces here as a plain
|
|
479
|
+
# RateLimitError rather than BudgetExhaustedError. This call used to sit
|
|
480
|
+
# outside the try, so that exception escaped `main` entirely: a nine-frame
|
|
481
|
+
# traceback and exit 1.
|
|
482
|
+
#
|
|
483
|
+
# That is the MOST LIKELY path after any exit 75. The scheduler re-runs while
|
|
484
|
+
# the hourly window is still shut, and this is the first request the resumed
|
|
485
|
+
# run makes -- so the retry alerted instead of retrying, which is the exact
|
|
486
|
+
# opposite of what exit 75 exists for. BudgetExhaustedError subclasses
|
|
487
|
+
# RateLimitError, so one except covers both.
|
|
488
|
+
try:
|
|
489
|
+
span = history.span()
|
|
490
|
+
except RateLimitError as exc:
|
|
491
|
+
wait = getattr(exc, "retry_after", None) or 0
|
|
492
|
+
print(
|
|
493
|
+
f"{park_id}: history budget is spent; rerun the same command in "
|
|
494
|
+
f"{wait / 60:.0f} min to continue",
|
|
495
|
+
file=sys.stderr,
|
|
496
|
+
)
|
|
497
|
+
return EX_TEMPFAIL
|
|
498
|
+
|
|
499
|
+
ext = "csv" if fmt == "csv" else "ndjson"
|
|
500
|
+
out_path = out_dir / f"{park_id}.{ext}"
|
|
501
|
+
state_path = out_dir / f"{park_id}{STATE_SUFFIX}"
|
|
502
|
+
end = span.retrievable_through
|
|
503
|
+
|
|
504
|
+
decided = _decide(out_path, state_path, fmt, overwrite, span.archive_from)
|
|
505
|
+
if isinstance(decided, int):
|
|
506
|
+
return decided
|
|
507
|
+
start, has_rows, prior_start = decided
|
|
508
|
+
resuming = prior_start is not None
|
|
509
|
+
sf = _StateFile(state_path, fmt, prior_start or start, end)
|
|
510
|
+
|
|
511
|
+
print(
|
|
512
|
+
f"{park_id}: {start} .. {end}{' (resumed)' if resuming else ''} -> {out_path}",
|
|
513
|
+
file=sys.stderr,
|
|
514
|
+
)
|
|
515
|
+
|
|
516
|
+
if _is_empty_window(start, end):
|
|
517
|
+
_say_empty(end, start)
|
|
518
|
+
out_path.unlink(missing_ok=True)
|
|
519
|
+
return 0
|
|
520
|
+
|
|
521
|
+
ident = _RowIdentity(park)
|
|
522
|
+
job = _Job(history, out_path, fmt, end, has_rows, ident)
|
|
523
|
+
progress = _Progress()
|
|
524
|
+
try:
|
|
525
|
+
_stream(job, start, progress)
|
|
526
|
+
except BudgetExhaustedError as exc:
|
|
527
|
+
# The budget is hourly, so a spent one can be most of an hour from
|
|
528
|
+
# resetting. Record how far we got and exit 75 rather than sleeping.
|
|
529
|
+
_record(sf, progress.last_day, complete=False)
|
|
530
|
+
wait = exc.retry_after or 0
|
|
531
|
+
print(
|
|
532
|
+
f" budget spent; rerun the same command in {wait / 60:.0f} min to continue",
|
|
533
|
+
file=sys.stderr,
|
|
534
|
+
)
|
|
535
|
+
return EX_TEMPFAIL
|
|
536
|
+
|
|
537
|
+
if progress.skipped and progress.written == 0:
|
|
538
|
+
out_path.unlink(missing_ok=True)
|
|
539
|
+
state_path.unlink(missing_ok=True)
|
|
540
|
+
return 0
|
|
541
|
+
|
|
542
|
+
# Completion is RECORDED, never inferred from a missing file. That is the
|
|
543
|
+
# distinction the old checkpoint could not make.
|
|
544
|
+
_record(sf, progress.last_day, complete=True)
|
|
545
|
+
print(f" done: {progress.written} rows -> {out_path}", file=sys.stderr)
|
|
546
|
+
return 0
|
|
547
|
+
|
|
548
|
+
|
|
549
|
+
# --------------------------------------------------------------------------
|
|
550
|
+
# Finding what to back fill, without knowing any uuid.
|
|
551
|
+
# --------------------------------------------------------------------------
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
def _normalize(value: str) -> str:
|
|
555
|
+
"""Fold a name to something a person could plausibly have typed.
|
|
556
|
+
|
|
557
|
+
THIS EXISTS BECAUSE THE DOCUMENTED EXAMPLE DID NOT WORK. Matching used bare
|
|
558
|
+
`casefold()`, and the live name is `Walt Disney World® Resort`, so
|
|
559
|
+
`themeparks-backfill "Walt Disney World Resort"` -- the command's own epilog
|
|
560
|
+
example -- answered "no park or destination matching". Five live names were
|
|
561
|
+
unreachable that way:
|
|
562
|
+
|
|
563
|
+
Walt Disney World® Resort U+00AE
|
|
564
|
+
LEGOLAND® Korea U+00AE
|
|
565
|
+
Walibi Rhône-Alpes U+00F4
|
|
566
|
+
Knott’s Soak City U+2019, while its sibling in the SAME
|
|
567
|
+
destination is ASCII Knott's Berry Farm
|
|
568
|
+
|
|
569
|
+
NFKD splits an accented letter into letter plus combining mark, the mark is
|
|
570
|
+
dropped, and every non-alphanumeric character goes -- so ®, apostrophes of
|
|
571
|
+
either kind, spaces, hyphens and punctuation stop mattering. A side effect
|
|
572
|
+
worth having: the URL slug form matches too, and slugs are what people copy
|
|
573
|
+
out of an address bar.
|
|
574
|
+
|
|
575
|
+
`_ergonomic/destinations.py` has a lighter version of this, written first.
|
|
576
|
+
That is the one this should have reused.
|
|
577
|
+
"""
|
|
578
|
+
folded = unicodedata.normalize("NFKD", value.casefold())
|
|
579
|
+
return "".join(c for c in folded if c.isalnum() and not unicodedata.combining(c))
|
|
580
|
+
|
|
581
|
+
|
|
582
|
+
def _catalogue(tp: ThemeParks) -> list[tuple[str, str, str, str]]:
|
|
583
|
+
"""Every park as (park id, park name, destination id, destination name).
|
|
584
|
+
|
|
585
|
+
One call to /destinations, which is public and cacheable, so this costs
|
|
586
|
+
nothing worth optimising and works before you have a key at all -- which is
|
|
587
|
+
the point: you can find your park before deciding whether to pay.
|
|
588
|
+
"""
|
|
589
|
+
out: list[tuple[str, str, str, str]] = []
|
|
590
|
+
# `tp.raw` is the generated client; the ergonomic surface has no
|
|
591
|
+
# destinations call and does not need one for this.
|
|
592
|
+
for dest in tp.raw.get_destinations().destinations:
|
|
593
|
+
for park in dest.parks or []:
|
|
594
|
+
out.append((park.id, park.name, dest.id, dest.name))
|
|
595
|
+
return out
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def _by_id(catalogue: list[tuple[str, str, str, str]], wanted: str) -> list[tuple[str, str]] | None:
|
|
599
|
+
"""Parks for an exact id, or None if `wanted` is not an id we know.
|
|
600
|
+
|
|
601
|
+
A destination id is checked first and expands to its parks. Destination and
|
|
602
|
+
park ids never collide, so the order is about being deliberate rather than
|
|
603
|
+
about resolving a conflict.
|
|
604
|
+
"""
|
|
605
|
+
in_destination = [(pid, pname) for pid, pname, did, _ in catalogue if did == wanted]
|
|
606
|
+
if in_destination:
|
|
607
|
+
return in_destination
|
|
608
|
+
for pid, pname, _, _ in catalogue:
|
|
609
|
+
if pid == wanted:
|
|
610
|
+
return [(pid, pname)]
|
|
611
|
+
if _looks_like_id(wanted):
|
|
612
|
+
# An id we do not list: a park with no destination row, an attraction, or
|
|
613
|
+
# a typo. Pass it through and let the API say which, rather than
|
|
614
|
+
# second-guessing it here.
|
|
615
|
+
return [(wanted, wanted)]
|
|
616
|
+
return None
|
|
617
|
+
|
|
618
|
+
|
|
619
|
+
def _by_name(catalogue: list[tuple[str, str, str, str]], wanted: str) -> list[tuple[str, str]]:
|
|
620
|
+
"""Parks for a name, or SystemExit listing the candidates.
|
|
621
|
+
|
|
622
|
+
Exact wins outright, so "Magic Kingdom Park" is not ambiguous merely because
|
|
623
|
+
something else contains it. A destination name expands exactly as its id
|
|
624
|
+
does. More than one match is an error: guessing between two parks would
|
|
625
|
+
quietly download the wrong one and look like it worked, which is the worst
|
|
626
|
+
outcome available here.
|
|
627
|
+
"""
|
|
628
|
+
lowered = _normalize(wanted)
|
|
629
|
+
|
|
630
|
+
def parks_in(did: str) -> list[tuple[str, str]]:
|
|
631
|
+
return [(pid, pname) for pid, pname, d, _ in catalogue if d == did]
|
|
632
|
+
|
|
633
|
+
exact_dest = {did: dname for _, _, did, dname in catalogue if _normalize(dname) == lowered}
|
|
634
|
+
if len(exact_dest) == 1:
|
|
635
|
+
return parks_in(next(iter(exact_dest)))
|
|
636
|
+
|
|
637
|
+
exact_park = [(pid, pname) for pid, pname, _, _ in catalogue if _normalize(pname) == lowered]
|
|
638
|
+
if len(exact_park) == 1:
|
|
639
|
+
return exact_park
|
|
640
|
+
|
|
641
|
+
park_hits = [(pid, pname) for pid, pname, _, _ in catalogue if lowered in _normalize(pname)]
|
|
642
|
+
dest_hits = {did: dname for _, _, did, dname in catalogue if lowered in _normalize(dname)}
|
|
643
|
+
if len(dest_hits) == 1 and not park_hits:
|
|
644
|
+
return parks_in(next(iter(dest_hits)))
|
|
645
|
+
|
|
646
|
+
# THE DESTINATION GOES IN THE LABEL, and it is load-bearing: TWO parks are
|
|
647
|
+
# named exactly "Disneyland Park" -- Anaheim and Paris -- so a list of bare
|
|
648
|
+
# park names offers a choice between two identical lines.
|
|
649
|
+
dest_of = {pid: dname for pid, _, _, dname in catalogue}
|
|
650
|
+
candidates = [(pid, f"{pname} ({dest_of[pid]})") for pid, pname in park_hits] or [
|
|
651
|
+
(did, f"{dname} (destination, {len(parks_in(did))} parks)")
|
|
652
|
+
for did, dname in dest_hits.items()
|
|
653
|
+
]
|
|
654
|
+
if not candidates:
|
|
655
|
+
raise SystemExit(
|
|
656
|
+
f'no park or destination matching "{wanted}".\n'
|
|
657
|
+
f" themeparks-backfill --list everything\n"
|
|
658
|
+
f' themeparks-backfill --list disney the ones matching "disney"'
|
|
659
|
+
)
|
|
660
|
+
if len(candidates) > 1:
|
|
661
|
+
lines = "\n".join(
|
|
662
|
+
f" {cid} {name}" for cid, name in sorted(candidates, key=lambda c: c[1])
|
|
663
|
+
)
|
|
664
|
+
raise SystemExit(
|
|
665
|
+
f'"{wanted}" matches {len(candidates)}. Pass an id, or the destination'
|
|
666
|
+
f" name to get all of its parks:\n{lines}"
|
|
667
|
+
)
|
|
668
|
+
return candidates
|
|
669
|
+
|
|
670
|
+
|
|
671
|
+
def _resolve(catalogue: list[tuple[str, str, str, str]], wanted: str) -> list[tuple[str, str]]:
|
|
672
|
+
"""The parks to back fill, as (id, name), from whichever handle they have.
|
|
673
|
+
|
|
674
|
+
Accepts four things, because a customer has whichever one they found:
|
|
675
|
+
|
|
676
|
+
- a park id -> that park
|
|
677
|
+
- a DESTINATION id -> every park in it
|
|
678
|
+
- a park name -> that park
|
|
679
|
+
- a destination name -> every park in it
|
|
680
|
+
|
|
681
|
+
"Walt Disney World Resort" is the shape people actually have, and making them
|
|
682
|
+
look up four park uuids first was a wall for no reason.
|
|
683
|
+
"""
|
|
684
|
+
by_id = _by_id(catalogue, wanted)
|
|
685
|
+
return by_id if by_id is not None else _by_name(catalogue, wanted)
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
UUID_LENGTH = 36
|
|
689
|
+
UUID_DASHES = 4
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
def _looks_like_id(value: str) -> bool:
|
|
693
|
+
"""A uuid, loosely. Loose on purpose: the API decides what is valid, not us."""
|
|
694
|
+
return len(value) == UUID_LENGTH and value.count("-") == UUID_DASHES
|
|
695
|
+
|
|
696
|
+
|
|
697
|
+
def _print_list(tp: ThemeParks, needle: str | None) -> int:
|
|
698
|
+
"""Parks grouped under their destination, so a destination id is visible too."""
|
|
699
|
+
catalogue = _catalogue(tp)
|
|
700
|
+
if needle:
|
|
701
|
+
lowered = _normalize(needle)
|
|
702
|
+
catalogue = [
|
|
703
|
+
c for c in catalogue if lowered in _normalize(c[1]) or lowered in _normalize(c[3])
|
|
704
|
+
]
|
|
705
|
+
if not catalogue:
|
|
706
|
+
print(f'nothing matching "{needle}"', file=sys.stderr)
|
|
707
|
+
return 1
|
|
708
|
+
|
|
709
|
+
by_dest: dict[tuple[str, str], list[tuple[str, str]]] = {}
|
|
710
|
+
for pid, pname, did, dname in catalogue:
|
|
711
|
+
by_dest.setdefault((did, dname), []).append((pid, pname))
|
|
712
|
+
for (did, dname), parks in sorted(by_dest.items(), key=lambda kv: kv[0][1]):
|
|
713
|
+
# The destination line is indented left of its parks and labelled, so it
|
|
714
|
+
# reads as "pass this to get all of them" rather than as another park.
|
|
715
|
+
print(f"{did} {dname} <- destination: all {len(parks)} parks")
|
|
716
|
+
for pid, pname in sorted(parks, key=lambda p: p[1]):
|
|
717
|
+
print(f" {pid} {pname}")
|
|
718
|
+
return 0
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
EPILOG = """examples:
|
|
722
|
+
themeparks-backfill "Disneyland Park"
|
|
723
|
+
the whole daily history your plan reaches, as NDJSON, into the current
|
|
724
|
+
directory
|
|
725
|
+
|
|
726
|
+
themeparks-backfill "Walt Disney World Resort"
|
|
727
|
+
a DESTINATION: every park in it, one file each
|
|
728
|
+
|
|
729
|
+
themeparks-backfill --list disney
|
|
730
|
+
find an id, or check the spelling. Lists destinations with their parks
|
|
731
|
+
indented underneath. Works without a key.
|
|
732
|
+
|
|
733
|
+
themeparks-backfill "Epcot" --format csv --out ./data
|
|
734
|
+
one wide CSV row per entity per day, into ./data
|
|
735
|
+
|
|
736
|
+
themeparks-backfill <id-a> <id-b> <id-c>
|
|
737
|
+
several parks in one run, sharing one connection and one budget
|
|
738
|
+
|
|
739
|
+
exit codes:
|
|
740
|
+
0 done
|
|
741
|
+
75 the hourly history budget ran out. Progress is checkpointed; run the same
|
|
742
|
+
command again to continue. This is EX_TEMPFAIL, so a cron or systemd timer
|
|
743
|
+
retries instead of alerting.
|
|
744
|
+
|
|
745
|
+
how far back this reaches is your plan: 7 days with no key at all, 30 on a free
|
|
746
|
+
key, 400 on Pro, the whole archive on Business. It runs either way -- it asks the
|
|
747
|
+
API what you may see and starts there, so you never have to work it out, and it
|
|
748
|
+
never asks for a day you are not entitled to twice.
|
|
749
|
+
"""
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
def main(argv: list[str] | None = None) -> int:
|
|
753
|
+
parser = argparse.ArgumentParser(
|
|
754
|
+
prog="themeparks-backfill",
|
|
755
|
+
description="Download a park's daily history to a file."
|
|
756
|
+
" One request per page, checkpointed, resumable.",
|
|
757
|
+
epilog=EPILOG,
|
|
758
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
759
|
+
)
|
|
760
|
+
parser.add_argument(
|
|
761
|
+
"parks",
|
|
762
|
+
nargs="*",
|
|
763
|
+
metavar="PARK",
|
|
764
|
+
help="park or DESTINATION, by name or id. A destination back fills every park in it.",
|
|
765
|
+
)
|
|
766
|
+
parser.add_argument(
|
|
767
|
+
"--list",
|
|
768
|
+
nargs="?",
|
|
769
|
+
const="",
|
|
770
|
+
metavar="TEXT",
|
|
771
|
+
dest="list_parks",
|
|
772
|
+
help="list ids and names, optionally filtered, then exit. Needs no key.",
|
|
773
|
+
)
|
|
774
|
+
parser.add_argument(
|
|
775
|
+
"--api-key",
|
|
776
|
+
default=os.environ.get("THEMEPARKS_API_KEY"),
|
|
777
|
+
help="API key. Defaults to $THEMEPARKS_API_KEY.",
|
|
778
|
+
)
|
|
779
|
+
parser.add_argument(
|
|
780
|
+
"--format",
|
|
781
|
+
choices=["ndjson", "csv"],
|
|
782
|
+
default="ndjson",
|
|
783
|
+
help="output format (default: ndjson)",
|
|
784
|
+
)
|
|
785
|
+
parser.add_argument(
|
|
786
|
+
"--overwrite",
|
|
787
|
+
action="store_true",
|
|
788
|
+
help="replace an existing file instead of refusing. Without it, a park"
|
|
789
|
+
" that finished is not fetched twice and a file this command did not"
|
|
790
|
+
" write is never touched.",
|
|
791
|
+
)
|
|
792
|
+
parser.add_argument(
|
|
793
|
+
"--out",
|
|
794
|
+
type=Path,
|
|
795
|
+
default=Path("."),
|
|
796
|
+
metavar="DIR",
|
|
797
|
+
help="output directory (default: .)",
|
|
798
|
+
)
|
|
799
|
+
args = parser.parse_args(argv)
|
|
800
|
+
|
|
801
|
+
_use_utf8(sys.stdout, sys.stderr)
|
|
802
|
+
|
|
803
|
+
# --list first: it is how you find a park, so it must work before you have a
|
|
804
|
+
# key and before you have decided to pay for anything.
|
|
805
|
+
if args.list_parks is not None:
|
|
806
|
+
with ThemeParks(api_key=args.api_key, user_agent=USER_AGENT) as tp:
|
|
807
|
+
return _print_list(tp, args.list_parks or None)
|
|
808
|
+
|
|
809
|
+
if not args.parks:
|
|
810
|
+
parser.error("which park or destination? try: themeparks-backfill --list disney")
|
|
811
|
+
|
|
812
|
+
# NO KEY IS NOT AN ERROR. It used to be: the command refused to start, with a
|
|
813
|
+
# message that said in the same breath that anonymous access reads 7 days.
|
|
814
|
+
# Telling someone the thing works and then declining to do it is worse than
|
|
815
|
+
# either. Anonymous reads 7 days, so it runs, says so, and says what a key
|
|
816
|
+
# would add -- which is also the honest sales pitch: the person evaluating
|
|
817
|
+
# whether to pay is exactly the person who should be able to run this.
|
|
818
|
+
if not args.api_key:
|
|
819
|
+
print(
|
|
820
|
+
"no API key: reading the 7 days anonymous access allows.\n"
|
|
821
|
+
" a free key reads 30 days, Pro 400, Business the whole archive\n"
|
|
822
|
+
" set THEMEPARKS_API_KEY, or pass --api-key\n"
|
|
823
|
+
" keys: https://www.themeparks.wiki/profile\n",
|
|
824
|
+
file=sys.stderr,
|
|
825
|
+
)
|
|
826
|
+
|
|
827
|
+
args.out.mkdir(parents=True, exist_ok=True)
|
|
828
|
+
|
|
829
|
+
# One client for every park: the connection pool is worth reusing and the
|
|
830
|
+
# budget is per account either way.
|
|
831
|
+
with ThemeParks(api_key=args.api_key, user_agent=USER_AGENT) as tp:
|
|
832
|
+
catalogue = _catalogue(tp)
|
|
833
|
+
dest_of = {pid: dname for pid, _, _, dname in catalogue}
|
|
834
|
+
targets: list[tuple[str, str]] = []
|
|
835
|
+
seen: set[str] = set()
|
|
836
|
+
for wanted in args.parks:
|
|
837
|
+
for pid, pname in _resolve(catalogue, wanted):
|
|
838
|
+
# A destination and one of its parks can both be named on one
|
|
839
|
+
# command line. Back filling the same park twice would double
|
|
840
|
+
# every row in the file.
|
|
841
|
+
if pid not in seen:
|
|
842
|
+
seen.add(pid)
|
|
843
|
+
targets.append((pid, pname))
|
|
844
|
+
|
|
845
|
+
# ALWAYS ECHO WHAT A NAME RESOLVED TO, even for a single park, and tell
|
|
846
|
+
# the caller to use the id next time.
|
|
847
|
+
#
|
|
848
|
+
# Names are for FINDING a park once. Ids are for asking for it. Twelve
|
|
849
|
+
# live parks contain "Hurricane Harbor" and the bare name is an exact
|
|
850
|
+
# match for the St. Louis one, so someone in Chicago could have typed a
|
|
851
|
+
# reasonable thing, got no warning, and loaded another park's history
|
|
852
|
+
# believing it was theirs. Echoing the resolution is what makes that
|
|
853
|
+
# visible; recommending the id is what stops it recurring in a script.
|
|
854
|
+
resolved_by_name = any(not _looks_like_id(w) for w in args.parks)
|
|
855
|
+
if len(targets) > 1:
|
|
856
|
+
print(f"{len(targets)} parks to back fill:", file=sys.stderr)
|
|
857
|
+
for pid, pname in targets:
|
|
858
|
+
where = dest_of.get(pid)
|
|
859
|
+
suffix = f" ({where})" if where and where != pname else ""
|
|
860
|
+
print(f" {pname}{suffix} {pid}", file=sys.stderr)
|
|
861
|
+
elif resolved_by_name:
|
|
862
|
+
pid, pname = targets[0]
|
|
863
|
+
where = dest_of.get(pid)
|
|
864
|
+
suffix = f" ({where})" if where and where != pname else ""
|
|
865
|
+
print(f"resolved to {pname}{suffix} {pid}", file=sys.stderr)
|
|
866
|
+
if resolved_by_name:
|
|
867
|
+
ids = " ".join(pid for pid, _ in targets)
|
|
868
|
+
print(
|
|
869
|
+
f" use the id next time — names are convenient once, ids are exact:\n"
|
|
870
|
+
f" themeparks-backfill {ids}",
|
|
871
|
+
file=sys.stderr,
|
|
872
|
+
)
|
|
873
|
+
|
|
874
|
+
for park_id, pname in targets:
|
|
875
|
+
status = backfill_park(tp, _Park(park_id, pname), args.out, args.format, args.overwrite)
|
|
876
|
+
if status != 0:
|
|
877
|
+
# Stop at the first exhausted budget. Carrying on to the next
|
|
878
|
+
# park only spends the retry-after on 429s.
|
|
879
|
+
return status
|
|
880
|
+
return 0
|
|
881
|
+
|
|
882
|
+
|
|
883
|
+
if __name__ == "__main__":
|
|
884
|
+
raise SystemExit(main())
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|