themeparks 3.2.0__tar.gz → 4.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {themeparks-3.2.0 → themeparks-4.0.0}/CHANGELOG.md +179 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/PKG-INFO +20 -4
- {themeparks-3.2.0 → themeparks-4.0.0}/README.md +19 -3
- {themeparks-3.2.0 → themeparks-4.0.0}/pyproject.toml +7 -1
- themeparks-4.0.0/tests/fixtures/README.md +50 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/__init__.py +8 -1
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/history.py +116 -6
- themeparks-4.0.0/themeparks/_generated/models.py +1337 -0
- themeparks-4.0.0/themeparks/_models_base.py +29 -0
- themeparks-4.0.0/themeparks/backfill.py +1326 -0
- themeparks-3.2.0/themeparks/_generated/models.py +0 -1121
- {themeparks-3.2.0 → themeparks-4.0.0}/.gitignore +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/LICENSE +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/MIGRATION.md +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_cache.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_client.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/__init__.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/dates.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/destinations.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/entity.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/live.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_errors.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_generated/__init__.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ratelimit.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_raw.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_transport.py +0 -0
- {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/py.typed +0 -0
|
@@ -1,5 +1,184 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [4.0.0] - 2026-09-28
|
|
4
|
+
|
|
5
|
+
**3.3.0 was yanked: incomplete CSV export and a resume defect.**
|
|
6
|
+
|
|
7
|
+
A major version because **the CSV header changed**: fifteen columns were added and
|
|
8
|
+
the order is now the schema's, so a reader that takes columns by position gets the
|
|
9
|
+
wrong ones rather than an error. Read by name. The library API is backward
|
|
10
|
+
compatible.
|
|
11
|
+
|
|
12
|
+
Everything here came out of porting `themeparks-backfill` to the JavaScript SDK
|
|
13
|
+
and then diffing the two outputs over the same park, and out of six reviews of the
|
|
14
|
+
result. Two independent implementations reading one API disagree in exactly the
|
|
15
|
+
places one of them is wrong. Magic Kingdom's full archive now comes back
|
|
16
|
+
**byte for byte identical** from both SDKs: 94,223 rows, 41 columns, the only
|
|
17
|
+
differences being today's row, which grows as the day elapses.
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
|
|
21
|
+
- **`themeparks-backfill "magic kingdom"` wrote the wrong park name into every
|
|
22
|
+
row.** A name that matched one park by substring returned the formatted display
|
|
23
|
+
label, so `parkName` read `Magic Kingdom Park (Walt Disney World® Resort)` for
|
|
24
|
+
all ~94,000 rows, and the resolution echo printed the destination twice. Four
|
|
25
|
+
live names reached it.
|
|
26
|
+
|
|
27
|
+
- **The CSV was missing ten of the thirty-six fields the API sends, on every row.**
|
|
28
|
+
`unknownMinutes`, the whole `inParkHours` block (the day's numbers limited to the
|
|
29
|
+
park's published hours -- usually the ones you want, since a ride "down" at 2am
|
|
30
|
+
is not down), `extremeWaits` (how many readings of 480+ minutes are folded into
|
|
31
|
+
the statistics, which is how you spot a feed error), and three of `singleRider`'s
|
|
32
|
+
five percentiles while `standby` carried all five. On a five-year Magic Kingdom
|
|
33
|
+
export, 72,200 of 94,223 rows were missing their in-park statistics. **The column
|
|
34
|
+
list is now derived from the model**, so it cannot drift again.
|
|
35
|
+
|
|
36
|
+
- **Vendored models were stale, and pydantic drops what it does not declare**, so
|
|
37
|
+
those three fields were deleted at parse time for every caller of `days()`, not
|
|
38
|
+
just for the CSV. Models regenerated, and every model now keeps fields the schema
|
|
39
|
+
does not declare (`themeparks._models_base.ApiModel`, `extra="allow"`), so a
|
|
40
|
+
field the API adds tomorrow survives parsing and reaches `model_dump()` and the
|
|
41
|
+
NDJSON output before this SDK knows it exists. It does not reach the CSV, whose
|
|
42
|
+
columns come from the schema.
|
|
43
|
+
|
|
44
|
+
- **A resumed download duplicated a day.** The checkpoint was the newest row
|
|
45
|
+
written; the page it came from covered further, because an entity that stopped
|
|
46
|
+
reporting has no rows for the tail days. A rerun re-fetched a day already in the
|
|
47
|
+
file and appended every row of it again, breaking the `(entityId, date)` key --
|
|
48
|
+
on the exit-75 path, which is the ordinary path for a long back fill. The
|
|
49
|
+
checkpoint is now the day the server's own `next` URL starts on.
|
|
50
|
+
|
|
51
|
+
- **A failure on a resumed run deleted everything already downloaded.** `written
|
|
52
|
+
== 0` means "this process wrote nothing", not "the file is empty". The state file
|
|
53
|
+
survived pointing mid-archive, so the next run appended only the tail and
|
|
54
|
+
recorded `complete: true`. Same for a window that closes under a resumed run --
|
|
55
|
+
a key rotated out of a scheduler's environment, a lapsed subscription -- which
|
|
56
|
+
additionally exited 0, so the scheduler logged success, and became a permanent
|
|
57
|
+
trap.
|
|
58
|
+
|
|
59
|
+
- **Resuming across versions, formats or SDKs corrupted the file.** One state file
|
|
60
|
+
served both formats, so `ndjson` then `csv` then `ndjson` doubled every row in
|
|
61
|
+
the first file; and the state carried nothing about the header, so 3.3.0's
|
|
62
|
+
19-column file resumed under this build appended 41-field rows beneath it. The
|
|
63
|
+
state file is now `<parkId>.<format>.backfill-state.json` and records the SDK,
|
|
64
|
+
its version, a state version and a fingerprint of the exact header. Anything that
|
|
65
|
+
does not match is refused with a message saying why, never resumed.
|
|
66
|
+
|
|
67
|
+
- **A network failure or timeout now exits 75, not 1**, so a scheduler retries
|
|
68
|
+
rather than alerting; anything the API actively rejected still exits 1. The
|
|
69
|
+
JavaScript SDK had these the other way round.
|
|
70
|
+
|
|
71
|
+
- **A carriage return in an entity name was written unquoted on Python 3.9 and
|
|
72
|
+
3.10**, so one row parsed as two with every later column shifted. The `csv`
|
|
73
|
+
module's QUOTE_MINIMAL only quotes characters that appear in the line terminator,
|
|
74
|
+
and this command sets LF; 3.11 changed the module to always quote CR and LF, so
|
|
75
|
+
the defect was invisible on a modern interpreter and live on two supported ones.
|
|
76
|
+
The CSV writer now does its own minimal quoting, which also makes the output
|
|
77
|
+
byte-identical across Python versions rather than only within one.
|
|
78
|
+
|
|
79
|
+
- **UTC timestamps are written `Z`, not `+00:00`**, and CSV line endings are LF.
|
|
80
|
+
Between them these accounted for 39,201 differing lines against the JavaScript
|
|
81
|
+
SDK's output for no difference in meaning.
|
|
82
|
+
|
|
83
|
+
- **One park's failure no longer abandons the rest of a destination.** Every park
|
|
84
|
+
is tried, what failed is named at the end, and the exit code still says something
|
|
85
|
+
went wrong. A spent budget still stops everything, deliberately.
|
|
86
|
+
|
|
87
|
+
- **A failed park no longer leaves a 0-byte file** that reads as "this park has no
|
|
88
|
+
history", including when the budget runs out before the first page.
|
|
89
|
+
|
|
90
|
+
- **A network failure, a full disk or Ctrl-C is a sentence, not a traceback.**
|
|
91
|
+
|
|
92
|
+
- **The user agent named neither version.** It was the literal
|
|
93
|
+
`themeparks-backfill/1`, and it replaced the SDK's own, so a support question had
|
|
94
|
+
no version to work from at either end.
|
|
95
|
+
|
|
96
|
+
- **`--list <text>` reported the wrong total**, printing "all 1 parks" for a
|
|
97
|
+
destination with six -- on the one line whose whole job is that number.
|
|
98
|
+
|
|
99
|
+
- **An ambiguous name listed the wrong candidates**, widening to substrings and
|
|
100
|
+
offering a third park that was not what was typed. It now lists the ids of the
|
|
101
|
+
parks that actually match, sorted by name.
|
|
102
|
+
|
|
103
|
+
- **A collection of nested models would have produced phantom columns** and then an
|
|
104
|
+
`AttributeError` on the first row. Duplicate column names are now impossible at
|
|
105
|
+
import rather than a wrong number under a right-looking header.
|
|
106
|
+
|
|
107
|
+
- **The NDJSON identity columns could be overwritten by the row** once models kept
|
|
108
|
+
undeclared fields.
|
|
109
|
+
|
|
110
|
+
### Added
|
|
111
|
+
|
|
112
|
+
- **The CSV carries a UTF-8 BOM**, so Excel on Windows stops rendering
|
|
113
|
+
`Walt Disney World® Resort` as mojibake.
|
|
114
|
+
- **A cell a spreadsheet would execute is prefixed with an apostrophe** (`=`, `+`,
|
|
115
|
+
`-`, `@`, tab, CR). Numeric cells are left alone, so a negative number stays a
|
|
116
|
+
number.
|
|
117
|
+
- **`on_page` on `days()` and `days_with_entities()`**, called once every row of a
|
|
118
|
+
page has been yielded, with a `HistoryPage` (`start`, `end`, `next_url`). The page
|
|
119
|
+
boundary is the server's own answer to "where do I carry on", and the rows cannot
|
|
120
|
+
tell you.
|
|
121
|
+
- **`--version`.**
|
|
122
|
+
- `EntityRef` and `HistoryPage` are exported from the package.
|
|
123
|
+
- `tests/fixtures/csv_contract.json`, an identical copy of which lives in the
|
|
124
|
+
JavaScript SDK. Both suites assert their column list against it, because this is
|
|
125
|
+
one command with two implementations and a customer using both should get one
|
|
126
|
+
file format.
|
|
127
|
+
|
|
128
|
+
### Changed
|
|
129
|
+
|
|
130
|
+
- The `themeparks-backfill` entry point is `themeparks.backfill:cli`, which adds
|
|
131
|
+
the top-level error handling. `main()` is unchanged for anyone calling it.
|
|
132
|
+
- Model equality and `model_json_schema()` reflect `extra="allow"`: two responses
|
|
133
|
+
differing only in an undeclared field now compare unequal, and dumps may contain
|
|
134
|
+
fields the schema does not list.
|
|
135
|
+
|
|
136
|
+
## [3.3.0] - 2026-09-28
|
|
137
|
+
|
|
138
|
+
### Added
|
|
139
|
+
|
|
140
|
+
- **`themeparks-backfill`: the archive download as a command.** It was an example
|
|
141
|
+
to copy off GitHub. The first paying customer followed that link and had to
|
|
142
|
+
work out that the library needed installing, then what the arguments were,
|
|
143
|
+
then read a traceback. Now:
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
pip install themeparks
|
|
147
|
+
themeparks-backfill "Disneyland Park"
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
- Takes a park or a **destination**, by name or id. A destination back fills
|
|
151
|
+
every park in it, one file each. `"Walt Disney World Resort"` is the handle
|
|
152
|
+
people actually have; four park uuids is not.
|
|
153
|
+
- `--list [text]` prints destinations with their parks underneath, and **needs
|
|
154
|
+
no key**, so you can find your park before deciding whether to pay.
|
|
155
|
+
- Refuses to guess between two matches. Two parks are named exactly
|
|
156
|
+
"Disneyland Park" (Anaheim and Paris), so the candidate list names the
|
|
157
|
+
destination as well.
|
|
158
|
+
- **Runs without a key**, reading the 7 days anonymous access allows, and says
|
|
159
|
+
what a key would add. It used to refuse to start with a message that
|
|
160
|
+
mentioned anonymous access in the same breath.
|
|
161
|
+
- NDJSON by default, `--format csv` for one wide row per entity per day.
|
|
162
|
+
- Checkpoints against the hourly history budget and exits 75 (`EX_TEMPFAIL`),
|
|
163
|
+
so a cron or timer retries rather than alerting. Re-running continues.
|
|
164
|
+
|
|
165
|
+
`python -m themeparks.backfill` is the same thing. `examples/backfill.py`
|
|
166
|
+
remains as a shim so existing links keep working.
|
|
167
|
+
|
|
168
|
+
### Fixed
|
|
169
|
+
|
|
170
|
+
- **The history window recovery now actually works.** 3.2.0's `examples/backfill.py`
|
|
171
|
+
read `earliestAllowedDate` from the top level of the 403 body; the API nests it
|
|
172
|
+
under `error`. So the recovery shipped doing nothing and a Pro customer still
|
|
173
|
+
got a traceback on their first request. The tests passed because the fixture was
|
|
174
|
+
built from the formatted text in a traceback rather than a real response, so the
|
|
175
|
+
code and the test were wrong together. The fixture is now captured from
|
|
176
|
+
production and a test fails if anyone flattens it.
|
|
177
|
+
|
|
178
|
+
The underlying gap is in the API, not the client: `/history/coverage` reports
|
|
179
|
+
where the archive starts and where your window ends, and nothing about where
|
|
180
|
+
your window begins. Until it does, the 403 is the only place that date exists.
|
|
181
|
+
|
|
3
182
|
## [3.2.0] - 2026-09-26
|
|
4
183
|
|
|
5
184
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: themeparks
|
|
3
|
-
Version:
|
|
3
|
+
Version: 4.0.0
|
|
4
4
|
Summary: Official SDK for the ThemeParks.wiki API
|
|
5
5
|
Project-URL: Homepage, https://api.themeparks.wiki
|
|
6
6
|
Project-URL: Source, https://github.com/ThemeParks/ThemeParks_Python
|
|
@@ -350,9 +350,25 @@ except BudgetExhaustedError as exc:
|
|
|
350
350
|
print(f"resume in {exc.retry_after:.0f}s")
|
|
351
351
|
```
|
|
352
352
|
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
353
|
+
### Or skip the code: there is a command
|
|
354
|
+
|
|
355
|
+
Installing the library installs `themeparks-backfill`, which does all of the
|
|
356
|
+
above and stops before the walls:
|
|
357
|
+
|
|
358
|
+
```bash
|
|
359
|
+
themeparks-backfill "Disneyland Park" # a park, by name or id
|
|
360
|
+
themeparks-backfill "Walt Disney World Resort" # a destination: every park in it
|
|
361
|
+
themeparks-backfill --list disney # find an id. Needs no key.
|
|
362
|
+
```
|
|
363
|
+
|
|
364
|
+
It reads how far back your own key may ask and starts there, writes NDJSON or
|
|
365
|
+
`--format csv`, names every row with the park and the entity, records what it
|
|
366
|
+
has done so re-running never duplicates a file, and exits 75 when the hourly
|
|
367
|
+
history budget runs out so a scheduler retries rather than alerts.
|
|
368
|
+
|
|
369
|
+
`python -m themeparks.backfill` is the same thing, which is the one to use if
|
|
370
|
+
`pip install --user` put the script somewhere off your PATH. `themeparks-backfill
|
|
371
|
+
--help` has the rest.
|
|
356
372
|
|
|
357
373
|
## Low-level escape hatch
|
|
358
374
|
|
|
@@ -311,9 +311,25 @@ except BudgetExhaustedError as exc:
|
|
|
311
311
|
print(f"resume in {exc.retry_after:.0f}s")
|
|
312
312
|
```
|
|
313
313
|
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
314
|
+
### Or skip the code: there is a command
|
|
315
|
+
|
|
316
|
+
Installing the library installs `themeparks-backfill`, which does all of the
|
|
317
|
+
above and stops before the walls:
|
|
318
|
+
|
|
319
|
+
```bash
|
|
320
|
+
themeparks-backfill "Disneyland Park" # a park, by name or id
|
|
321
|
+
themeparks-backfill "Walt Disney World Resort" # a destination: every park in it
|
|
322
|
+
themeparks-backfill --list disney # find an id. Needs no key.
|
|
323
|
+
```
|
|
324
|
+
|
|
325
|
+
It reads how far back your own key may ask and starts there, writes NDJSON or
|
|
326
|
+
`--format csv`, names every row with the park and the entity, records what it
|
|
327
|
+
has done so re-running never duplicates a file, and exits 75 when the hourly
|
|
328
|
+
history budget runs out so a scheduler retries rather than alerts.
|
|
329
|
+
|
|
330
|
+
`python -m themeparks.backfill` is the same thing, which is the one to use if
|
|
331
|
+
`pip install --user` put the script somewhere off your PATH. `themeparks-backfill
|
|
332
|
+
--help` has the rest.
|
|
317
333
|
|
|
318
334
|
## Low-level escape hatch
|
|
319
335
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "themeparks"
|
|
7
|
-
version = "
|
|
7
|
+
version = "4.0.0"
|
|
8
8
|
description = "Official SDK for the ThemeParks.wiki API"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -30,6 +30,12 @@ dependencies = [
|
|
|
30
30
|
"eval-type-backport>=0.2; python_version < '3.10'",
|
|
31
31
|
]
|
|
32
32
|
|
|
33
|
+
[project.scripts]
|
|
34
|
+
# The archive backfill, as a command rather than a file to copy off GitHub.
|
|
35
|
+
# `pip install themeparks` then `themeparks-backfill "Disneyland Park"` is the
|
|
36
|
+
# whole path from nothing to a file of history.
|
|
37
|
+
themeparks-backfill = "themeparks.backfill:cli"
|
|
38
|
+
|
|
33
39
|
[project.urls]
|
|
34
40
|
Homepage = "https://api.themeparks.wiki"
|
|
35
41
|
Source = "https://github.com/ThemeParks/ThemeParks_Python"
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# Fixtures captured from the live API
|
|
2
|
+
|
|
3
|
+
Not hand-written. Each file here was taken from a real response, and the header
|
|
4
|
+
below says which endpoint and when. That matters because of a specific failure
|
|
5
|
+
this project keeps repeating.
|
|
6
|
+
|
|
7
|
+
## Why this directory exists
|
|
8
|
+
|
|
9
|
+
On 2026-09-28 a 403-recovery fix shipped doing nothing. Its tests passed because
|
|
10
|
+
the fixture was built from the *formatted text of a traceback* rather than from a
|
|
11
|
+
response body, so the code and the test were wrong together and agreed with each
|
|
12
|
+
other. The memory note for that pattern is `self-confirming-harness`, and it was
|
|
13
|
+
its fifth occurrence in one day.
|
|
14
|
+
|
|
15
|
+
A fixture the author of the code also invented proves only that the two are
|
|
16
|
+
consistent. A fixture taken from the real thing can disagree.
|
|
17
|
+
|
|
18
|
+
## destinations_slice.json
|
|
19
|
+
|
|
20
|
+
`GET https://api.themeparks.wiki/v1/destinations`, captured 2026-09-28, trimmed
|
|
21
|
+
to 9 destinations and 20 parks and otherwise verbatim — every id, name and slug
|
|
22
|
+
is exactly what the API returned.
|
|
23
|
+
|
|
24
|
+
It is trimmed to keep, deliberately, every case that broke name resolution:
|
|
25
|
+
|
|
26
|
+
| Case | Why it is here |
|
|
27
|
+
|---|---|
|
|
28
|
+
| `Walt Disney World® Resort` | U+00AE. The command's own documented example, `"Walt Disney World Resort"`, did not match it. |
|
|
29
|
+
| `LEGOLAND® Korea` | the same, on a destination whose park shares the name |
|
|
30
|
+
| `Walibi Rhône-Alpes` | U+00F4, so a normaliser must decompose accents |
|
|
31
|
+
| `Knott's Berry Farm` + `Knott’s Soak City` | ASCII `'` and U+2019 **in one destination**. Nobody types the curly one. |
|
|
32
|
+
| `Disneyland Park` ×2 | Anaheim and Paris, identical park names. The reason the candidate list names the destination. |
|
|
33
|
+
| `Hurricane Harbor` + `Hurricane Harbor Chicago` | an exact match with a substring rival: the shape that silently downloaded the wrong park |
|
|
34
|
+
| `Cedar Point` | destination name == park name, with a second park, so "exact destination beats exact park" is observable |
|
|
35
|
+
|
|
36
|
+
**Do not edit these by hand.** Re-capture them. If a name upstream has drifted,
|
|
37
|
+
that is a real change and the test should notice.
|
|
38
|
+
|
|
39
|
+
## mk_park_daily_page1.json / mk_park_daily_page2.json
|
|
40
|
+
|
|
41
|
+
Two consecutive pages of one real request, captured 2026-09-28:
|
|
42
|
+
`GET /entity/75ea578a-adc8-4116-a54d-dccb60765ef9/history/daily?from=2026-08-01&to=2026-09-20`
|
|
43
|
+
then its `next` followed verbatim. Trimmed to three entities (an attraction, a
|
|
44
|
+
show, a restaurant); `range`, `next` and every row are the server's.
|
|
45
|
+
|
|
46
|
+
They are the oracle for resumable paging. Page one covers through 2026-08-31 and
|
|
47
|
+
the server says carry on at 2026-09-01, but two of its three entities have no
|
|
48
|
+
rows after 2026-08-30 -- so a checkpoint taken from the newest ROW rewinds and
|
|
49
|
+
re-downloads days already written. The same files are in the JavaScript SDK, so
|
|
50
|
+
both ports are tested against identical bytes.
|
|
@@ -1,7 +1,12 @@
|
|
|
1
1
|
from themeparks._cache import Cache, CacheConfig, InMemoryLRUCache
|
|
2
2
|
from themeparks._client import AsyncThemeParks, ThemeParks
|
|
3
3
|
from themeparks._ergonomic.dates import parse_api_datetime
|
|
4
|
-
from themeparks._ergonomic.history import
|
|
4
|
+
from themeparks._ergonomic.history import (
|
|
5
|
+
BudgetExhaustedError,
|
|
6
|
+
EntityRef,
|
|
7
|
+
HistoryPage,
|
|
8
|
+
HistorySpan,
|
|
9
|
+
)
|
|
5
10
|
from themeparks._ergonomic.live import current_wait_time, iter_queues
|
|
6
11
|
from themeparks._errors import (
|
|
7
12
|
APIError,
|
|
@@ -17,6 +22,8 @@ __all__ = [
|
|
|
17
22
|
"APIError",
|
|
18
23
|
"AsyncThemeParks",
|
|
19
24
|
"BudgetExhaustedError",
|
|
25
|
+
"EntityRef",
|
|
26
|
+
"HistoryPage",
|
|
20
27
|
"HistorySpan",
|
|
21
28
|
"RateLimit",
|
|
22
29
|
"RateLimits",
|
|
@@ -20,7 +20,7 @@ the cheap path here without having to know the expensive one exists.
|
|
|
20
20
|
|
|
21
21
|
from __future__ import annotations
|
|
22
22
|
|
|
23
|
-
from collections.abc import AsyncIterator, Iterator
|
|
23
|
+
from collections.abc import AsyncIterator, Callable, Iterator
|
|
24
24
|
from datetime import date as _date
|
|
25
25
|
from typing import Any, NamedTuple, Union
|
|
26
26
|
|
|
@@ -111,17 +111,87 @@ def _reraise_if_too_long(exc: RateLimitError, max_wait: float) -> None:
|
|
|
111
111
|
raise exc
|
|
112
112
|
|
|
113
113
|
|
|
114
|
-
|
|
115
|
-
"""
|
|
116
|
-
|
|
114
|
+
class HistoryPage(NamedTuple):
|
|
115
|
+
"""One page of daily history, as the server described it.
|
|
116
|
+
|
|
117
|
+
`start` and `end` are the park-local days the page ACTUALLY covered, which is
|
|
118
|
+
not the range you asked for: a park daily call serves at most 31 days, so a
|
|
119
|
+
50-day request comes back as 31 days plus a `next`. `next_url` is the URL of
|
|
120
|
+
the following page, or None on the last one.
|
|
121
|
+
|
|
122
|
+
This exists for resumable downloads. A checkpoint taken from the ROWS is
|
|
123
|
+
wrong in both directions: the newest row's date can be earlier than the page
|
|
124
|
+
covered, because an entity that stopped reporting has no rows for the tail
|
|
125
|
+
days, so resuming there re-fetches days already written and duplicates them;
|
|
126
|
+
and a half-written page is indistinguishable from a finished one. The page
|
|
127
|
+
boundary is the server's own answer to "where do I carry on", so it is the
|
|
128
|
+
only safe checkpoint.
|
|
129
|
+
"""
|
|
130
|
+
|
|
131
|
+
start: str
|
|
132
|
+
end: str
|
|
133
|
+
next_url: str | None
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class EntityRef(NamedTuple):
|
|
137
|
+
"""Who a history row belongs to, AS THE HISTORY RESPONSE REPORTS IT.
|
|
138
|
+
|
|
139
|
+
The name matters and the source of it matters more. A park's current
|
|
140
|
+
`/children` list gives today's name, which is the wrong label for a row
|
|
141
|
+
recorded years ago: rides are renamed, and stamping today's name on old data
|
|
142
|
+
quietly rewrites history. The history envelope carries its own `name` and
|
|
143
|
+
`entityType` per entity, and that is the name to use.
|
|
144
|
+
"""
|
|
145
|
+
|
|
146
|
+
id: str
|
|
147
|
+
name: str
|
|
148
|
+
entity_type: str
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _ref(entity: Any) -> EntityRef:
|
|
152
|
+
kind = getattr(entity, "entityType", None)
|
|
153
|
+
# The generated models use an enum, and str(EntityType.SHOW) is
|
|
154
|
+
# "EntityType.SHOW". `.value` is what the API sends.
|
|
155
|
+
inner = getattr(kind, "value", kind)
|
|
156
|
+
return EntityRef(
|
|
157
|
+
entity.id, getattr(entity, "name", "") or "", "" if inner is None else str(inner)
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _page_of(envelope: DailyEnvelope) -> HistoryPage:
|
|
162
|
+
"""The page an envelope represents, for :class:`HistoryPage`'s callers."""
|
|
163
|
+
rng = getattr(envelope, "range", None)
|
|
164
|
+
nxt = getattr(envelope, "next", None)
|
|
165
|
+
return HistoryPage(
|
|
166
|
+
getattr(rng, "from_", "") or "",
|
|
167
|
+
getattr(rng, "to", "") or "",
|
|
168
|
+
nxt or None,
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _daily_entity_rows(envelope: DailyEnvelope) -> Iterator[tuple[EntityRef, HistoryDailyRow]]:
|
|
173
|
+
"""Yield (entity ref, row), keeping the name the response gave.
|
|
174
|
+
|
|
175
|
+
`_daily_rows` below is the same walk with the ref flattened to its id, kept
|
|
176
|
+
because `days()` has yielded `(id, row)` since 3.0 and that shape is public.
|
|
177
|
+
"""
|
|
117
178
|
entities = getattr(envelope, "entities", None)
|
|
118
179
|
if entities is not None:
|
|
119
180
|
for entity in entities:
|
|
181
|
+
ref = _ref(entity)
|
|
120
182
|
for row in entity.days or []:
|
|
121
|
-
yield (
|
|
183
|
+
yield (ref, row)
|
|
122
184
|
return
|
|
185
|
+
ref = _ref(envelope)
|
|
123
186
|
for row in getattr(envelope, "days", None) or []:
|
|
124
|
-
yield (
|
|
187
|
+
yield (ref, row)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _daily_rows(envelope: DailyEnvelope) -> Iterator[tuple[str, HistoryDailyRow]]:
|
|
191
|
+
"""Yield (entity id, row). A park envelope carries many entities; an entity
|
|
192
|
+
envelope carries its own rows, so both flatten to the same stream."""
|
|
193
|
+
for ref, row in _daily_entity_rows(envelope):
|
|
194
|
+
yield (ref.id, row)
|
|
125
195
|
|
|
126
196
|
|
|
127
197
|
def _raw_rows(envelope: RawEnvelope) -> Iterator[tuple[str, HistoryRow]]:
|
|
@@ -154,21 +224,58 @@ class HistoryApi:
|
|
|
154
224
|
"""
|
|
155
225
|
return _span(self.coverage())
|
|
156
226
|
|
|
227
|
+
def days_with_entities(
|
|
228
|
+
self,
|
|
229
|
+
start: str | _date | None = None,
|
|
230
|
+
end: str | _date | None = None,
|
|
231
|
+
*,
|
|
232
|
+
max_wait: float = DEFAULT_MAX_WAIT_SECONDS,
|
|
233
|
+
on_page: Callable[[HistoryPage], None] | None = None,
|
|
234
|
+
) -> Iterator[tuple[EntityRef, HistoryDailyRow]]:
|
|
235
|
+
"""`days()`, but each row arrives with the entity's name and type.
|
|
236
|
+
|
|
237
|
+
Use this when you are writing history to a file. The name comes from the
|
|
238
|
+
history response itself, so it is the label that response gives for those
|
|
239
|
+
rows rather than the park's current `/children` list -- rides get renamed,
|
|
240
|
+
and today's name on a row from three years ago is a quiet rewrite of the
|
|
241
|
+
record.
|
|
242
|
+
|
|
243
|
+
It also saves a request: the name is already in the payload, so nothing
|
|
244
|
+
needs to ask what an id refers to.
|
|
245
|
+
|
|
246
|
+
`on_page` is called once every row of a page has been yielded, with a
|
|
247
|
+
:class:`HistoryPage`. Checkpoint on that, never on the last row you saw.
|
|
248
|
+
"""
|
|
249
|
+
envelope: DailyEnvelope | None = self._first_daily(start, end, max_wait)
|
|
250
|
+
while envelope is not None:
|
|
251
|
+
yield from _daily_entity_rows(envelope)
|
|
252
|
+
# AFTER the rows, never before: a caller checkpointing on this has to
|
|
253
|
+
# be able to trust that everything the page held is already written.
|
|
254
|
+
if on_page is not None:
|
|
255
|
+
on_page(_page_of(envelope))
|
|
256
|
+
envelope = self._next_daily(envelope, max_wait)
|
|
257
|
+
|
|
157
258
|
def days(
|
|
158
259
|
self,
|
|
159
260
|
start: str | _date | None = None,
|
|
160
261
|
end: str | _date | None = None,
|
|
161
262
|
*,
|
|
162
263
|
max_wait: float = DEFAULT_MAX_WAIT_SECONDS,
|
|
264
|
+
on_page: Callable[[HistoryPage], None] | None = None,
|
|
163
265
|
) -> Iterator[tuple[str, HistoryDailyRow]]:
|
|
164
266
|
"""One summary row per park-local day, as (entity id, row).
|
|
165
267
|
|
|
166
268
|
Pages automatically. Given a park id this uses the park call, which
|
|
167
269
|
answers every entity in the park in one request.
|
|
270
|
+
|
|
271
|
+
`days_with_entities()` is the same stream with the entity's name and type
|
|
272
|
+
attached; this shape is kept because it is public API from 3.0.
|
|
168
273
|
"""
|
|
169
274
|
envelope: DailyEnvelope | None = self._first_daily(start, end, max_wait)
|
|
170
275
|
while envelope is not None:
|
|
171
276
|
yield from _daily_rows(envelope)
|
|
277
|
+
if on_page is not None:
|
|
278
|
+
on_page(_page_of(envelope))
|
|
172
279
|
envelope = self._next_daily(envelope, max_wait)
|
|
173
280
|
|
|
174
281
|
def _first_daily(
|
|
@@ -237,6 +344,7 @@ class AsyncHistoryApi:
|
|
|
237
344
|
end: str | _date | None = None,
|
|
238
345
|
*,
|
|
239
346
|
max_wait: float = DEFAULT_MAX_WAIT_SECONDS,
|
|
347
|
+
on_page: Callable[[HistoryPage], None] | None = None,
|
|
240
348
|
) -> AsyncIterator[tuple[str, HistoryDailyRow]]:
|
|
241
349
|
try:
|
|
242
350
|
envelope: DailyEnvelope | None = await self._raw.get_entity_history_daily(
|
|
@@ -248,6 +356,8 @@ class AsyncHistoryApi:
|
|
|
248
356
|
while envelope is not None:
|
|
249
357
|
for pair in _daily_rows(envelope):
|
|
250
358
|
yield pair
|
|
359
|
+
if on_page is not None:
|
|
360
|
+
on_page(_page_of(envelope))
|
|
251
361
|
nxt = getattr(envelope, "next", None)
|
|
252
362
|
if not nxt:
|
|
253
363
|
return
|