themeparks 3.2.0__tar.gz → 3.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. {themeparks-3.2.0 → themeparks-3.3.0}/CHANGELOG.md +46 -0
  2. {themeparks-3.2.0 → themeparks-3.3.0}/PKG-INFO +20 -4
  3. {themeparks-3.2.0 → themeparks-3.3.0}/README.md +19 -3
  4. {themeparks-3.2.0 → themeparks-3.3.0}/pyproject.toml +7 -1
  5. themeparks-3.3.0/tests/fixtures/README.md +37 -0
  6. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/history.py +68 -5
  7. themeparks-3.3.0/themeparks/backfill.py +884 -0
  8. {themeparks-3.2.0 → themeparks-3.3.0}/.gitignore +0 -0
  9. {themeparks-3.2.0 → themeparks-3.3.0}/LICENSE +0 -0
  10. {themeparks-3.2.0 → themeparks-3.3.0}/MIGRATION.md +0 -0
  11. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/__init__.py +0 -0
  12. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_cache.py +0 -0
  13. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_client.py +0 -0
  14. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/__init__.py +0 -0
  15. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/dates.py +0 -0
  16. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/destinations.py +0 -0
  17. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/entity.py +0 -0
  18. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ergonomic/live.py +0 -0
  19. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_errors.py +0 -0
  20. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_generated/__init__.py +0 -0
  21. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_generated/models.py +0 -0
  22. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_ratelimit.py +0 -0
  23. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_raw.py +0 -0
  24. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/_transport.py +0 -0
  25. {themeparks-3.2.0 → themeparks-3.3.0}/themeparks/py.typed +0 -0
@@ -1,5 +1,51 @@
1
1
  # Changelog
2
2
 
3
+ ## [3.3.0] - 2026-09-28
4
+
5
+ ### Added
6
+
7
+ - **`themeparks-backfill`: the archive download as a command.** It was an example
8
+ to copy off GitHub. The first paying customer followed that link and had to
9
+ work out that the library needed installing, then what the arguments were,
10
+ then read a traceback. Now:
11
+
12
+ ```bash
13
+ pip install themeparks
14
+ themeparks-backfill "Disneyland Park"
15
+ ```
16
+
17
+ - Takes a park or a **destination**, by name or id. A destination back fills
18
+ every park in it, one file each. `"Walt Disney World Resort"` is the handle
19
+ people actually have; four park uuids is not.
20
+ - `--list [text]` prints destinations with their parks underneath, and **needs
21
+ no key**, so you can find your park before deciding whether to pay.
22
+ - Refuses to guess between two matches. Two parks are named exactly
23
+ "Disneyland Park" (Anaheim and Paris), so the candidate list names the
24
+ destination as well.
25
+ - **Runs without a key**, reading the 7 days anonymous access allows, and says
26
+ what a key would add. It used to refuse to start with a message that
27
+ mentioned anonymous access in the same breath.
28
+ - NDJSON by default, `--format csv` for one wide row per entity per day.
29
+ - Checkpoints against the hourly history budget and exits 75 (`EX_TEMPFAIL`),
30
+ so a cron or timer retries rather than alerting. Re-running continues.
31
+
32
+ `python -m themeparks.backfill` is the same thing. `examples/backfill.py`
33
+ remains as a shim so existing links keep working.
34
+
35
+ ### Fixed
36
+
37
+ - **The history window recovery now actually works.** 3.2.0's `examples/backfill.py`
38
+ read `earliestAllowedDate` from the top level of the 403 body; the API nests it
39
+ under `error`. So the recovery shipped doing nothing and a Pro customer still
40
+ got a traceback on their first request. The tests passed because the fixture was
41
+ built from the formatted text in a traceback rather than a real response, so the
42
+ code and the test were wrong together. The fixture is now captured from
43
+ production and a test fails if anyone flattens it.
44
+
45
+ The underlying gap is in the API, not the client: `/history/coverage` reports
46
+ where the archive starts and where your window ends, and nothing about where
47
+ your window begins. Until it does, the 403 is the only place that date exists.
48
+
3
49
  ## [3.2.0] - 2026-09-26
4
50
 
5
51
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: themeparks
3
- Version: 3.2.0
3
+ Version: 3.3.0
4
4
  Summary: Official SDK for the ThemeParks.wiki API
5
5
  Project-URL: Homepage, https://api.themeparks.wiki
6
6
  Project-URL: Source, https://github.com/ThemeParks/ThemeParks_Python
@@ -350,9 +350,25 @@ except BudgetExhaustedError as exc:
350
350
  print(f"resume in {exc.retry_after:.0f}s")
351
351
  ```
352
352
 
353
- A complete backfill script with resume and CSV output is in
354
- [`examples/backfill.py`](examples/backfill.py); it pulls Disneyland Resort's
355
- whole daily archive, 98,452 rows, in one run.
353
+ ### Or skip the code: there is a command
354
+
355
+ Installing the library installs `themeparks-backfill`, which does all of the
356
+ above and stops before the walls:
357
+
358
+ ```bash
359
+ themeparks-backfill "Disneyland Park" # a park, by name or id
360
+ themeparks-backfill "Walt Disney World Resort" # a destination: every park in it
361
+ themeparks-backfill --list disney # find an id. Needs no key.
362
+ ```
363
+
364
+ It reads how far back your own key may ask and starts there, writes NDJSON or
365
+ `--format csv`, names every row with the park and the entity, records what it
366
+ has done so re-running never duplicates a file, and exits 75 when the hourly
367
+ history budget runs out so a scheduler retries rather than alerts.
368
+
369
+ `python -m themeparks.backfill` is the same thing, which is the one to use if
370
+ `pip install --user` put the script somewhere off your PATH. `themeparks-backfill
371
+ --help` has the rest.
356
372
 
357
373
  ## Low-level escape hatch
358
374
 
@@ -311,9 +311,25 @@ except BudgetExhaustedError as exc:
311
311
  print(f"resume in {exc.retry_after:.0f}s")
312
312
  ```
313
313
 
314
- A complete backfill script with resume and CSV output is in
315
- [`examples/backfill.py`](examples/backfill.py); it pulls Disneyland Resort's
316
- whole daily archive, 98,452 rows, in one run.
314
+ ### Or skip the code: there is a command
315
+
316
+ Installing the library installs `themeparks-backfill`, which does all of the
317
+ above and stops before the walls:
318
+
319
+ ```bash
320
+ themeparks-backfill "Disneyland Park" # a park, by name or id
321
+ themeparks-backfill "Walt Disney World Resort" # a destination: every park in it
322
+ themeparks-backfill --list disney # find an id. Needs no key.
323
+ ```
324
+
325
+ It reads how far back your own key may ask and starts there, writes NDJSON or
326
+ `--format csv`, names every row with the park and the entity, records what it
327
+ has done so re-running never duplicates a file, and exits 75 when the hourly
328
+ history budget runs out so a scheduler retries rather than alerts.
329
+
330
+ `python -m themeparks.backfill` is the same thing, which is the one to use if
331
+ `pip install --user` put the script somewhere off your PATH. `themeparks-backfill
332
+ --help` has the rest.
317
333
 
318
334
  ## Low-level escape hatch
319
335
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "themeparks"
7
- version = "3.2.0"
7
+ version = "3.3.0"
8
8
  description = "Official SDK for the ThemeParks.wiki API"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -30,6 +30,12 @@ dependencies = [
30
30
  "eval-type-backport>=0.2; python_version < '3.10'",
31
31
  ]
32
32
 
33
+ [project.scripts]
34
+ # The archive backfill, as a command rather than a file to copy off GitHub.
35
+ # `pip install themeparks` then `themeparks-backfill "Disneyland Park"` is the
36
+ # whole path from nothing to a file of history.
37
+ themeparks-backfill = "themeparks.backfill:main"
38
+
33
39
  [project.urls]
34
40
  Homepage = "https://api.themeparks.wiki"
35
41
  Source = "https://github.com/ThemeParks/ThemeParks_Python"
@@ -0,0 +1,37 @@
1
+ # Fixtures captured from the live API
2
+
3
+ Not hand-written. Each file here was taken from a real response, and the header
4
+ below says which endpoint and when. That matters because of a specific failure
5
+ this project keeps repeating.
6
+
7
+ ## Why this directory exists
8
+
9
+ On 2026-09-28 a 403-recovery fix shipped doing nothing. Its tests passed because
10
+ the fixture was built from the *formatted text of a traceback* rather than from a
11
+ response body, so the code and the test were wrong together and agreed with each
12
+ other. The memory note for that pattern is `self-confirming-harness`, and it was
13
+ its fifth occurrence in one day.
14
+
15
+ A fixture the author of the code also invented proves only that the two are
16
+ consistent. A fixture taken from the real thing can disagree.
17
+
18
+ ## destinations_slice.json
19
+
20
+ `GET https://api.themeparks.wiki/v1/destinations`, captured 2026-09-28, trimmed
21
+ to 9 destinations and 20 parks and otherwise verbatim — every id, name and slug
22
+ is exactly what the API returned.
23
+
24
+ It is trimmed to keep, deliberately, every case that broke name resolution:
25
+
26
+ | Case | Why it is here |
27
+ |---|---|
28
+ | `Walt Disney World® Resort` | U+00AE. The command's own documented example, `"Walt Disney World Resort"`, did not match it. |
29
+ | `LEGOLAND® Korea` | the same, on a destination whose park shares the name |
30
+ | `Walibi Rhône-Alpes` | U+00F4, so a normaliser must decompose accents |
31
+ | `Knott's Berry Farm` + `Knott’s Soak City` | ASCII `'` and U+2019 **in one destination**. Nobody types the curly one. |
32
+ | `Disneyland Park` ×2 | Anaheim and Paris, identical park names. The reason the candidate list names the destination. |
33
+ | `Hurricane Harbor` + `Hurricane Harbor Chicago` | an exact match with a substring rival: the shape that silently downloaded the wrong park |
34
+ | `Cedar Point` | destination name == park name, with a second park, so "exact destination beats exact park" is observable |
35
+
36
+ **Do not edit these by hand.** Re-capture them. If a name upstream has drifted,
37
+ that is a real change and the test should notice.
@@ -111,17 +111,54 @@ def _reraise_if_too_long(exc: RateLimitError, max_wait: float) -> None:
111
111
  raise exc
112
112
 
113
113
 
114
- def _daily_rows(envelope: DailyEnvelope) -> Iterator[tuple[str, HistoryDailyRow]]:
115
- """Yield (entity id, row). A park envelope carries many entities; an entity
116
- envelope carries its own rows, so both flatten to the same stream."""
114
+ class EntityRef(NamedTuple):
115
+ """Who a history row belongs to, AS THE HISTORY RESPONSE REPORTS IT.
116
+
117
+ The name matters and the source of it matters more. A park's current
118
+ `/children` list gives today's name, which is the wrong label for a row
119
+ recorded years ago: rides are renamed, and stamping today's name on old data
120
+ quietly rewrites history. The history envelope carries its own `name` and
121
+ `entityType` per entity, and that is the name to use.
122
+ """
123
+
124
+ id: str
125
+ name: str
126
+ entity_type: str
127
+
128
+
129
+ def _ref(entity: Any) -> EntityRef:
130
+ kind = getattr(entity, "entityType", None)
131
+ # The generated models use an enum, and str(EntityType.SHOW) is
132
+ # "EntityType.SHOW". `.value` is what the API sends.
133
+ inner = getattr(kind, "value", kind)
134
+ return EntityRef(
135
+ entity.id, getattr(entity, "name", "") or "", "" if inner is None else str(inner)
136
+ )
137
+
138
+
139
+ def _daily_entity_rows(envelope: DailyEnvelope) -> Iterator[tuple[EntityRef, HistoryDailyRow]]:
140
+ """Yield (entity ref, row), keeping the name the response gave.
141
+
142
+ `_daily_rows` below is the same walk with the ref flattened to its id, kept
143
+ because `days()` has yielded `(id, row)` since 3.0 and that shape is public.
144
+ """
117
145
  entities = getattr(envelope, "entities", None)
118
146
  if entities is not None:
119
147
  for entity in entities:
148
+ ref = _ref(entity)
120
149
  for row in entity.days or []:
121
- yield (entity.id, row)
150
+ yield (ref, row)
122
151
  return
152
+ ref = _ref(envelope)
123
153
  for row in getattr(envelope, "days", None) or []:
124
- yield (envelope.id, row)
154
+ yield (ref, row)
155
+
156
+
157
+ def _daily_rows(envelope: DailyEnvelope) -> Iterator[tuple[str, HistoryDailyRow]]:
158
+ """Yield (entity id, row). A park envelope carries many entities; an entity
159
+ envelope carries its own rows, so both flatten to the same stream."""
160
+ for ref, row in _daily_entity_rows(envelope):
161
+ yield (ref.id, row)
125
162
 
126
163
 
127
164
  def _raw_rows(envelope: RawEnvelope) -> Iterator[tuple[str, HistoryRow]]:
@@ -154,6 +191,29 @@ class HistoryApi:
154
191
  """
155
192
  return _span(self.coverage())
156
193
 
194
+ def days_with_entities(
195
+ self,
196
+ start: str | _date | None = None,
197
+ end: str | _date | None = None,
198
+ *,
199
+ max_wait: float = DEFAULT_MAX_WAIT_SECONDS,
200
+ ) -> Iterator[tuple[EntityRef, HistoryDailyRow]]:
201
+ """`days()`, but each row arrives with the entity's name and type.
202
+
203
+ Use this when you are writing history to a file. The name comes from the
204
+ history response itself, so it is the label that response gives for those
205
+ rows rather than the park's current `/children` list -- rides get renamed,
206
+ and today's name on a row from three years ago is a quiet rewrite of the
207
+ record.
208
+
209
+ It also saves a request: the name is already in the payload, so nothing
210
+ needs to ask what an id refers to.
211
+ """
212
+ envelope: DailyEnvelope | None = self._first_daily(start, end, max_wait)
213
+ while envelope is not None:
214
+ yield from _daily_entity_rows(envelope)
215
+ envelope = self._next_daily(envelope, max_wait)
216
+
157
217
  def days(
158
218
  self,
159
219
  start: str | _date | None = None,
@@ -165,6 +225,9 @@ class HistoryApi:
165
225
 
166
226
  Pages automatically. Given a park id this uses the park call, which
167
227
  answers every entity in the park in one request.
228
+
229
+ `days_with_entities()` is the same stream with the entity's name and type
230
+ attached; this shape is kept because it is public API from 3.0.
168
231
  """
169
232
  envelope: DailyEnvelope | None = self._first_daily(start, end, max_wait)
170
233
  while envelope is not None:
@@ -0,0 +1,884 @@
1
+ """Download a park's whole daily history to a file, and survive the budget.
2
+
3
+ Run it:
4
+
5
+ themeparks-backfill "Disneyland Park"
6
+ python -m themeparks.backfill "Disneyland Park" # same thing
7
+
8
+ WHY THIS IS IN THE PACKAGE RATHER THAN AN EXAMPLE TO COPY. It used to be
9
+ `examples/backfill.py` on GitHub, linked from the docs. The first paying customer
10
+ followed that link and had to work out that the library needed installing, then
11
+ what the arguments were, then read a traceback. Every one of those is a wall
12
+ between someone who has paid for the archive and the archive. A recipe you have
13
+ to reconstruct is not a recipe; this is one command.
14
+
15
+ What it does that is easy to get wrong by hand:
16
+
17
+ 1. It asks the PARK, not the rides. Both history endpoints answer every entity in
18
+ a park in one request, so a park-level backfill of a large resort is around a
19
+ hundred times fewer calls than the same data pulled ride by ride.
20
+
21
+ 2. It bounds the range at BOTH ends. `span().retrievable_through` is the latest
22
+ day your key may ask for. There is no field for the earliest -- coverage
23
+ reports where the archive starts, which on any plan short of the full archive
24
+ is before your window -- so the first request is refused and the floor is read
25
+ out of that 403. See `_window_floor`.
26
+
27
+ 3. It records what it has done, in `<park-id>.backfill-state.json`: the format,
28
+ the range, the furthest day written, and whether it finished. The history
29
+ budget is hourly, so a spent one can be most of an hour from resetting; the
30
+ SDK raises BudgetExhaustedError rather than sleeping through that, and this
31
+ writes the state and exits 75 (EX_TEMPFAIL) so a scheduler retries rather
32
+ than alerts.
33
+
34
+ Re-running is then safe in every direction: an unfinished park continues, a
35
+ FINISHED park is left alone rather than appended to twice, and a file this
36
+ command did not write is never touched without `--overwrite`. It re-reads the
37
+ furthest day on purpose -- a page can end mid-day -- so `(entityId, date)` is
38
+ the natural key if you load blind.
39
+
40
+ 4. It takes NAMES as well as ids, and DESTINATIONS as well as parks. A
41
+ customer has "Walt Disney World Resort", not four park uuids, and making
42
+ them look those up first was another wall. A destination back fills every
43
+ park in it, into one file each.
44
+ """
45
+
46
+ from __future__ import annotations
47
+
48
+ import argparse
49
+ import contextlib
50
+ import csv
51
+ import json
52
+ import os
53
+ import sys
54
+ import unicodedata
55
+ from datetime import date
56
+ from pathlib import Path
57
+ from typing import Any, NamedTuple, TextIO, Union
58
+
59
+ from themeparks import APIError, BudgetExhaustedError, RateLimitError, ThemeParks
60
+ from themeparks._ergonomic.history import EntityRef
61
+
62
+ USER_AGENT = "themeparks-backfill/1"
63
+
64
+ EX_TEMPFAIL = 75
65
+
66
+ CSV_COLUMNS = [
67
+ # Identity first. A reader opening this in a spreadsheet should know what a
68
+ # row is before they reach the numbers, and a table loaded from several files
69
+ # needs parkId to tell them apart.
70
+ "parkId",
71
+ "parkName",
72
+ "entityId",
73
+ "entityName",
74
+ "entityType",
75
+ "date",
76
+ "firstOperatingAt",
77
+ "lastClosedAt",
78
+ "operatingMinutes",
79
+ "downMinutes",
80
+ "showCount",
81
+ "changes",
82
+ "standbyMin",
83
+ "standbyP50",
84
+ "standbyMean",
85
+ "standbyP90",
86
+ "standbyMax",
87
+ "singleRiderP50",
88
+ "singleRiderMax",
89
+ ]
90
+
91
+
92
+ def _csv_row(ref: EntityRef, row: Any, ident: _RowIdentity) -> dict[str, Any]:
93
+ """Flatten the nested standby/singleRider stats into one wide row."""
94
+ standby = row.standby
95
+ single = row.singleRider
96
+ return {
97
+ "parkId": ident.park.id,
98
+ "parkName": ident.park.name,
99
+ "entityId": ref.id,
100
+ "entityName": ref.name,
101
+ "entityType": ref.entity_type,
102
+ "date": row.date.isoformat(),
103
+ "firstOperatingAt": row.firstOperatingAt.isoformat() if row.firstOperatingAt else "",
104
+ "lastClosedAt": row.lastClosedAt.isoformat() if row.lastClosedAt else "",
105
+ "operatingMinutes": row.operatingMinutes,
106
+ "downMinutes": row.downMinutes,
107
+ "showCount": row.showCount if row.showCount is not None else "",
108
+ "changes": row.changes,
109
+ "standbyMin": standby.min if standby else "",
110
+ "standbyP50": standby.p50 if standby else "",
111
+ "standbyMean": standby.mean if standby else "",
112
+ "standbyP90": standby.p90 if standby else "",
113
+ "standbyMax": standby.max if standby else "",
114
+ "singleRiderP50": single.p50 if single else "",
115
+ "singleRiderMax": single.max if single else "",
116
+ }
117
+
118
+
119
+ class _Park(NamedTuple):
120
+ """A park's identity, so rows can name themselves.
121
+
122
+ `backfill_park` used to take only the id, so every row carried a bare
123
+ `entityId` and nothing else. A customer loading two files into one table
124
+ could not tell the parks apart -- the park id existed only in the FILENAME --
125
+ and 60 attraction GUIDs with no labels meant writing the lookup code they
126
+ bought this to avoid.
127
+ """
128
+
129
+ id: str
130
+ name: str
131
+
132
+
133
+ class _RowIdentity(NamedTuple):
134
+ """What a row inherits from the run, as opposed to from the response.
135
+
136
+ Only the park. The entity's name and type come from the history response
137
+ itself (`days_with_entities`), because those are per row and change over
138
+ time: a ride renamed in 2024 must not have its 2021 rows relabelled with
139
+ today's name. The park id is here rather than in the filename alone so two
140
+ files can be loaded into one table.
141
+ """
142
+
143
+ park: _Park
144
+
145
+
146
+ class Writer:
147
+ """NDJSON or CSV behind one `write(entity_id, row)`."""
148
+
149
+ def __init__(self, handle: TextIO, fmt: str, write_header: bool, ident: _RowIdentity) -> None:
150
+ self._handle = handle
151
+ self._fmt = fmt
152
+ self._ident = ident
153
+ self._csv = None
154
+ if fmt == "csv":
155
+ self._csv = csv.DictWriter(handle, fieldnames=CSV_COLUMNS)
156
+ if write_header:
157
+ self._csv.writeheader()
158
+
159
+ def write(self, ref: EntityRef, row: Any) -> None:
160
+ if self._csv is not None:
161
+ self._csv.writerow(_csv_row(ref, row, self._ident))
162
+ return
163
+ # Identity keys come FIRST in the object, so a human reading one line of
164
+ # NDJSON sees what it is before the numbers.
165
+ payload = {
166
+ "parkId": self._ident.park.id,
167
+ "parkName": self._ident.park.name,
168
+ "entityId": ref.id,
169
+ "entityName": ref.name,
170
+ "entityType": ref.entity_type,
171
+ **row.model_dump(mode="json"),
172
+ }
173
+ self._handle.write(json.dumps(payload) + "\n")
174
+
175
+
176
+ def _use_utf8(*streams: TextIO) -> None:
177
+ """Stop a legacy Windows code page killing the run on a park name.
178
+
179
+ stdout's error handler is `strict`, and real names carry characters cp437,
180
+ cp850 and cp932 cannot encode -- `Walt Disney World® Resort`,
181
+ `LEGOLAND® Korea`, `Knott’s Soak City`, `Walibi Rhône-Alpes`. So `--list`
182
+ died with UnicodeEncodeError partway through, after writing 200 good lines,
183
+ whenever output was redirected on a non-1252 system. Redirecting is the
184
+ obvious thing to do with 358 lines of parks.
185
+
186
+ `reconfigure` is 3.7+; `backslashreplace` degrades an unencodable character
187
+ to an escape instead of ending the run.
188
+ """
189
+ for stream in streams:
190
+ reconfigure = getattr(stream, "reconfigure", None)
191
+ if reconfigure is not None:
192
+ with contextlib.suppress(OSError, ValueError):
193
+ reconfigure(encoding="utf-8", errors="backslashreplace")
194
+
195
+
196
+ def _window_floor(exc: APIError) -> str | None:
197
+ """The earliest day this key may ask for, read out of a 403 body.
198
+
199
+ THE BODY IS NESTED, and the first version of this function was not. What the
200
+ API actually sends is:
201
+
202
+ 403 {"error": {"type": "HISTORY_WINDOW_EXCEEDED",
203
+ "message": "This key can see history back to 2025-08-25 (400 days).",
204
+ "earliestAllowedDate": "2025-08-25"}}
205
+
206
+ The first version read those keys off the TOP level, because it was written
207
+ from the formatted text in a traceback rather than from a real response. Its
208
+ tests passed -- they built the fixture the same wrong way -- and it shipped
209
+ doing nothing at all. `tests/unit/test_backfill.py` now pins a body captured
210
+ from production, which is the only version of this test that can fail.
211
+
212
+ Both shapes are accepted: the nested one the API sends, and a bare one, so
213
+ this cannot break again if an error envelope is ever flattened.
214
+
215
+ `/history/coverage` does not carry the floor -- it reports where the archive
216
+ starts and where your window ends, and nothing in between -- so the 403 is
217
+ the only place this date exists. Exposing it on the coverage document is
218
+ tracked upstream.
219
+ """
220
+ body = exc.body
221
+ if not isinstance(body, dict):
222
+ return None
223
+ inner = body.get("error")
224
+ payload = inner if isinstance(inner, dict) else body
225
+ if payload.get("type") != "HISTORY_WINDOW_EXCEEDED":
226
+ return None
227
+ floor = payload.get("earliestAllowedDate")
228
+ return floor if isinstance(floor, str) and floor else None
229
+
230
+
231
+ # --------------------------------------------------------------------------
232
+ # Per-park state. One file, and it closes five separate defects.
233
+ # --------------------------------------------------------------------------
234
+ #
235
+ # The old scheme was a bare `<park-id>.checkpoint` holding one date, and "this
236
+ # park is finished" was encoded as THE ABSENCE of that file -- which is
237
+ # indistinguishable from "never started". The data file was always opened in
238
+ # append mode. Between them that produced:
239
+ #
240
+ # - re-running a finished park appended a second complete copy, silently.
241
+ # Measured: 382 rows became 764.
242
+ # - a destination run that hit the hourly budget re-downloaded every COMPLETED
243
+ # park in full on each retry, spending the new budget on work already done,
244
+ # so a later park might never advance while the finished files grew by a
245
+ # copy an hour.
246
+ # - switching --format mid-resume wrote a new file starting at the checkpoint
247
+ # day and silently lost everything before it, exit 0.
248
+ # - a truncated or empty checkpoint sent `?from=&to=` forever, with no way to
249
+ # know a hidden file was the cause.
250
+ # - nothing recorded which range had been written, so nothing could tell.
251
+ #
252
+ # So state is explicit: the format, the range asked for, the high-water day, and
253
+ # whether it finished. Unreadable state is treated as no state rather than
254
+ # crashing -- a corrupt file must not be a permanent wall.
255
+ STATE_SUFFIX = ".backfill-state.json"
256
+
257
+
258
+ def _read_state(path: Path) -> dict[str, Any]:
259
+ try:
260
+ value = json.loads(path.read_text(encoding="utf-8"))
261
+ except (OSError, ValueError):
262
+ return {}
263
+ return value if isinstance(value, dict) else {}
264
+
265
+
266
+ def _write_state(path: Path, **fields: Any) -> None:
267
+ path.write_text(json.dumps(fields, sort_keys=True) + "\n", encoding="utf-8")
268
+
269
+
270
+ class _StateFile(NamedTuple):
271
+ """Where the state lives and the range it describes, fixed for one park."""
272
+
273
+ path: Path
274
+ fmt: str
275
+ start: Day
276
+ end: Day
277
+
278
+
279
+ def _record(sf: _StateFile, last_day: date | None, *, complete: bool) -> None:
280
+ """Write the state file. `complete` is the fact the old checkpoint could not express.
281
+
282
+ `sf.start` is the ORIGINAL start of the range, not the day a resumed run
283
+ happened to begin at. The two call sites used to disagree about that, so a
284
+ run interrupted twice recorded the second resume point as though it were the
285
+ beginning and lost the real range.
286
+ """
287
+ _write_state(
288
+ sf.path,
289
+ format=sf.fmt,
290
+ start=str(sf.start),
291
+ end=str(sf.end),
292
+ last_day=last_day.isoformat() if last_day else None,
293
+ complete=complete,
294
+ )
295
+
296
+
297
+ Day = Union[str, date, None]
298
+
299
+
300
+ def _is_empty_window(first_day: Day, end: Day) -> bool:
301
+ """True when the plan's floor sits past the park's last day of data.
302
+
303
+ `end` is the newest day the park has data for; the 403 recovery clamps the
304
+ start UP to the first day this key may read. For a park that stopped
305
+ reporting before the window opens -- a seasonal water park, a closed ride --
306
+ the clamp can push start past end, and the API answers
307
+ `400 INVALID_RANGE: to must not be before from`. That killed a six-park
308
+ destination run three parks in, leaving a 0-byte file and two parks never
309
+ attempted, on every plan.
310
+ """
311
+ return end is not None and first_day is not None and str(first_day) > str(end)
312
+
313
+
314
+ def _say_empty(end: Day, start: Day | None = None) -> None:
315
+ reach = f", and your plan reaches back to {start}" if start is not None else ""
316
+ print(
317
+ f" nothing in your window: this park's data ends {end}{reach} — skipping",
318
+ file=sys.stderr,
319
+ )
320
+
321
+
322
+ class _Plan(NamedTuple):
323
+ """What a run should do about a park, once its state file has been read."""
324
+
325
+ start: Day
326
+ has_rows: bool
327
+ prior_start: str | None
328
+
329
+
330
+ def _decide(
331
+ out_path: Path, state_path: Path, fmt: str, overwrite: bool, archive_from: Day
332
+ ) -> _Plan | int:
333
+ """A `_Plan` to proceed with, or an exit code meaning "do not".
334
+
335
+ Split out of `backfill_park` because it got long enough for ruff to object,
336
+ and ruff was right: deciding whether to write is a different job from
337
+ writing. Every branch here exists for a defect measured in review -- see the
338
+ STATE block above for the five of them.
339
+ """
340
+ state = {} if overwrite else _read_state(state_path)
341
+ if overwrite:
342
+ out_path.unlink(missing_ok=True)
343
+ state_path.unlink(missing_ok=True)
344
+
345
+ file_exists = out_path.exists() and out_path.stat().st_size > 0
346
+ same_format = state.get("format") == fmt
347
+
348
+ # Finished already. Say so and stop, rather than appending a second copy.
349
+ if state.get("complete") and same_format and file_exists:
350
+ print(
351
+ f" already complete: {state.get('start')} .. {state.get('end')} "
352
+ f"in {out_path.name} — pass --overwrite to fetch it again",
353
+ file=sys.stderr,
354
+ )
355
+ return 0
356
+
357
+ # A file we have no record of writing. Refusing is the only safe answer:
358
+ # appending doubles it, truncating throws away someone's data.
359
+ if file_exists and not state:
360
+ print(
361
+ f" {out_path.name} already has rows and there is no state file beside it.\n"
362
+ f" --overwrite replace it\n"
363
+ f" or move it aside and run again",
364
+ file=sys.stderr,
365
+ )
366
+ return 1
367
+
368
+ # A format switch cannot resume: the half-written file is the other format.
369
+ if state and not same_format and not state.get("complete"):
370
+ other = state.get("format")
371
+ print(
372
+ f" {other} was interrupted part-way for this park. Finish it in "
373
+ f"{other}, or pass --overwrite to start again in {fmt}",
374
+ file=sys.stderr,
375
+ )
376
+ return 1
377
+
378
+ resuming = bool(state) and same_format and not state.get("complete")
379
+ last_written = state.get("last_day") if resuming else None
380
+ return _Plan(
381
+ start=last_written or archive_from,
382
+ has_rows=file_exists and resuming,
383
+ prior_start=state.get("start") if resuming else None,
384
+ )
385
+
386
+
387
+ class _Job(NamedTuple):
388
+ """Everything streaming one park needs, so the streamer takes two arguments."""
389
+
390
+ history: Any
391
+ out_path: Path
392
+ fmt: str
393
+ end: Day
394
+ has_rows: bool
395
+ ident: _RowIdentity
396
+
397
+
398
+ class _Progress:
399
+ """How far the stream got. MUTABLE, and that is the point.
400
+
401
+ `_stream` used to return this as a tuple, which meant an exception threw the
402
+ numbers away: the budget handler then had nothing to record and read the state
403
+ file back instead, which on a first run does not exist. The resume point was
404
+ lost on exactly the interruption it exists for, and a re-run started over.
405
+
406
+ Owned by the caller, updated in place, so it is readable after a raise.
407
+ """
408
+
409
+ def __init__(self) -> None:
410
+ self.written = 0
411
+ self.last_day: date | None = None
412
+ self.skipped = False
413
+
414
+
415
+ def _stream(job: _Job, start: Day, progress: _Progress) -> None:
416
+ """Write the range to the file, recovering once from a window 403.
417
+
418
+ Separated from `backfill_park` because that function was deciding, printing,
419
+ streaming and recording in one place, and ruff counted the statements before
420
+ a reader had to. This is the streaming.
421
+ """
422
+
423
+ def write_rows(writer: Writer, first_day: Day) -> None:
424
+ """Stream one range into the file. Raises whatever the SDK raises."""
425
+ for ref, row in job.history.days_with_entities(first_day, job.end):
426
+ writer.write(ref, row)
427
+ progress.written += 1
428
+ # MAX, not last-seen. `_daily_rows` walks entities and then each
429
+ # entity's days, so the final row belongs to the alphabetically last
430
+ # entity, which may have stopped reporting mid-page. Taking it as the
431
+ # high-water mark could rewind the resume point by up to a whole
432
+ # 31-day page, while the module claimed the overlap was "one day".
433
+ progress.last_day = (
434
+ row.date if progress.last_day is None else max(progress.last_day, row.date)
435
+ )
436
+ if progress.written % 5000 == 0:
437
+ print(f" {progress.written} rows, at {progress.last_day}", file=sys.stderr)
438
+
439
+ with job.out_path.open("a", newline="", encoding="utf-8") as handle:
440
+ # ONE Writer for the whole park, so the header decision is made once. It
441
+ # used to be built inside write_rows with `written == 0` in the predicate,
442
+ # and the recovery below calls that again precisely when written is 0 --
443
+ # so every CSV on every plan short of the full archive got TWO headers,
444
+ # and pandas read the second as data.
445
+ writer = Writer(handle, job.fmt, not job.has_rows, job.ident)
446
+ try:
447
+ write_rows(writer, start)
448
+ except APIError as exc:
449
+ floor = _window_floor(exc)
450
+ # Retry only when nothing was written: a 403 mid-stream is not a plan
451
+ # boundary, and restarting would duplicate rows.
452
+ if floor is None or progress.written:
453
+ raise
454
+ print(
455
+ f" this key reaches back to {floor}, not {start} — starting there",
456
+ file=sys.stderr,
457
+ )
458
+ if _is_empty_window(floor, job.end):
459
+ # Only visible after the clamp, and the file exists by now because
460
+ # opening it created it. Flagged, not returned, so the empty file
461
+ # is removed after the handle closes.
462
+ _say_empty(job.end)
463
+ progress.skipped = True
464
+ else:
465
+ write_rows(writer, floor)
466
+
467
+
468
+ def backfill_park(
469
+ tp: ThemeParks, park: _Park, out_dir: Path, fmt: str, overwrite: bool = False
470
+ ) -> int:
471
+ """Write one park's daily history. Returns 0, or EX_TEMPFAIL if the budget ran out."""
472
+ park_id = park.id
473
+ history = tp.entity(park_id).history
474
+
475
+ # SPAN IS INSIDE THE BUDGET HANDLING, and it was not.
476
+ #
477
+ # `coverage()` is the one history call the SDK does not wrap in
478
+ # `_reraise_if_too_long`, so a spent hourly budget surfaces here as a plain
479
+ # RateLimitError rather than BudgetExhaustedError. This call used to sit
480
+ # outside the try, so that exception escaped `main` entirely: a nine-frame
481
+ # traceback and exit 1.
482
+ #
483
+ # That is the MOST LIKELY path after any exit 75. The scheduler re-runs while
484
+ # the hourly window is still shut, and this is the first request the resumed
485
+ # run makes -- so the retry alerted instead of retrying, which is the exact
486
+ # opposite of what exit 75 exists for. BudgetExhaustedError subclasses
487
+ # RateLimitError, so one except covers both.
488
+ try:
489
+ span = history.span()
490
+ except RateLimitError as exc:
491
+ wait = getattr(exc, "retry_after", None) or 0
492
+ print(
493
+ f"{park_id}: history budget is spent; rerun the same command in "
494
+ f"{wait / 60:.0f} min to continue",
495
+ file=sys.stderr,
496
+ )
497
+ return EX_TEMPFAIL
498
+
499
+ ext = "csv" if fmt == "csv" else "ndjson"
500
+ out_path = out_dir / f"{park_id}.{ext}"
501
+ state_path = out_dir / f"{park_id}{STATE_SUFFIX}"
502
+ end = span.retrievable_through
503
+
504
+ decided = _decide(out_path, state_path, fmt, overwrite, span.archive_from)
505
+ if isinstance(decided, int):
506
+ return decided
507
+ start, has_rows, prior_start = decided
508
+ resuming = prior_start is not None
509
+ sf = _StateFile(state_path, fmt, prior_start or start, end)
510
+
511
+ print(
512
+ f"{park_id}: {start} .. {end}{' (resumed)' if resuming else ''} -> {out_path}",
513
+ file=sys.stderr,
514
+ )
515
+
516
+ if _is_empty_window(start, end):
517
+ _say_empty(end, start)
518
+ out_path.unlink(missing_ok=True)
519
+ return 0
520
+
521
+ ident = _RowIdentity(park)
522
+ job = _Job(history, out_path, fmt, end, has_rows, ident)
523
+ progress = _Progress()
524
+ try:
525
+ _stream(job, start, progress)
526
+ except BudgetExhaustedError as exc:
527
+ # The budget is hourly, so a spent one can be most of an hour from
528
+ # resetting. Record how far we got and exit 75 rather than sleeping.
529
+ _record(sf, progress.last_day, complete=False)
530
+ wait = exc.retry_after or 0
531
+ print(
532
+ f" budget spent; rerun the same command in {wait / 60:.0f} min to continue",
533
+ file=sys.stderr,
534
+ )
535
+ return EX_TEMPFAIL
536
+
537
+ if progress.skipped and progress.written == 0:
538
+ out_path.unlink(missing_ok=True)
539
+ state_path.unlink(missing_ok=True)
540
+ return 0
541
+
542
+ # Completion is RECORDED, never inferred from a missing file. That is the
543
+ # distinction the old checkpoint could not make.
544
+ _record(sf, progress.last_day, complete=True)
545
+ print(f" done: {progress.written} rows -> {out_path}", file=sys.stderr)
546
+ return 0
547
+
548
+
549
+ # --------------------------------------------------------------------------
550
+ # Finding what to back fill, without knowing any uuid.
551
+ # --------------------------------------------------------------------------
552
+
553
+
554
+ def _normalize(value: str) -> str:
555
+ """Fold a name to something a person could plausibly have typed.
556
+
557
+ THIS EXISTS BECAUSE THE DOCUMENTED EXAMPLE DID NOT WORK. Matching used bare
558
+ `casefold()`, and the live name is `Walt Disney World® Resort`, so
559
+ `themeparks-backfill "Walt Disney World Resort"` -- the command's own epilog
560
+ example -- answered "no park or destination matching". Five live names were
561
+ unreachable that way:
562
+
563
+ Walt Disney World® Resort U+00AE
564
+ LEGOLAND® Korea U+00AE
565
+ Walibi Rhône-Alpes U+00F4
566
+ Knott’s Soak City U+2019, while its sibling in the SAME
567
+ destination is ASCII Knott's Berry Farm
568
+
569
+ NFKD splits an accented letter into letter plus combining mark, the mark is
570
+ dropped, and every non-alphanumeric character goes -- so ®, apostrophes of
571
+ either kind, spaces, hyphens and punctuation stop mattering. A side effect
572
+ worth having: the URL slug form matches too, and slugs are what people copy
573
+ out of an address bar.
574
+
575
+ `_ergonomic/destinations.py` has a lighter version of this, written first.
576
+ That is the one this should have reused.
577
+ """
578
+ folded = unicodedata.normalize("NFKD", value.casefold())
579
+ return "".join(c for c in folded if c.isalnum() and not unicodedata.combining(c))
580
+
581
+
582
+ def _catalogue(tp: ThemeParks) -> list[tuple[str, str, str, str]]:
583
+ """Every park as (park id, park name, destination id, destination name).
584
+
585
+ One call to /destinations, which is public and cacheable, so this costs
586
+ nothing worth optimising and works before you have a key at all -- which is
587
+ the point: you can find your park before deciding whether to pay.
588
+ """
589
+ out: list[tuple[str, str, str, str]] = []
590
+ # `tp.raw` is the generated client; the ergonomic surface has no
591
+ # destinations call and does not need one for this.
592
+ for dest in tp.raw.get_destinations().destinations:
593
+ for park in dest.parks or []:
594
+ out.append((park.id, park.name, dest.id, dest.name))
595
+ return out
596
+
597
+
598
+ def _by_id(catalogue: list[tuple[str, str, str, str]], wanted: str) -> list[tuple[str, str]] | None:
599
+ """Parks for an exact id, or None if `wanted` is not an id we know.
600
+
601
+ A destination id is checked first and expands to its parks. Destination and
602
+ park ids never collide, so the order is about being deliberate rather than
603
+ about resolving a conflict.
604
+ """
605
+ in_destination = [(pid, pname) for pid, pname, did, _ in catalogue if did == wanted]
606
+ if in_destination:
607
+ return in_destination
608
+ for pid, pname, _, _ in catalogue:
609
+ if pid == wanted:
610
+ return [(pid, pname)]
611
+ if _looks_like_id(wanted):
612
+ # An id we do not list: a park with no destination row, an attraction, or
613
+ # a typo. Pass it through and let the API say which, rather than
614
+ # second-guessing it here.
615
+ return [(wanted, wanted)]
616
+ return None
617
+
618
+
619
+ def _by_name(catalogue: list[tuple[str, str, str, str]], wanted: str) -> list[tuple[str, str]]:
620
+ """Parks for a name, or SystemExit listing the candidates.
621
+
622
+ Exact wins outright, so "Magic Kingdom Park" is not ambiguous merely because
623
+ something else contains it. A destination name expands exactly as its id
624
+ does. More than one match is an error: guessing between two parks would
625
+ quietly download the wrong one and look like it worked, which is the worst
626
+ outcome available here.
627
+ """
628
+ lowered = _normalize(wanted)
629
+
630
+ def parks_in(did: str) -> list[tuple[str, str]]:
631
+ return [(pid, pname) for pid, pname, d, _ in catalogue if d == did]
632
+
633
+ exact_dest = {did: dname for _, _, did, dname in catalogue if _normalize(dname) == lowered}
634
+ if len(exact_dest) == 1:
635
+ return parks_in(next(iter(exact_dest)))
636
+
637
+ exact_park = [(pid, pname) for pid, pname, _, _ in catalogue if _normalize(pname) == lowered]
638
+ if len(exact_park) == 1:
639
+ return exact_park
640
+
641
+ park_hits = [(pid, pname) for pid, pname, _, _ in catalogue if lowered in _normalize(pname)]
642
+ dest_hits = {did: dname for _, _, did, dname in catalogue if lowered in _normalize(dname)}
643
+ if len(dest_hits) == 1 and not park_hits:
644
+ return parks_in(next(iter(dest_hits)))
645
+
646
+ # THE DESTINATION GOES IN THE LABEL, and it is load-bearing: TWO parks are
647
+ # named exactly "Disneyland Park" -- Anaheim and Paris -- so a list of bare
648
+ # park names offers a choice between two identical lines.
649
+ dest_of = {pid: dname for pid, _, _, dname in catalogue}
650
+ candidates = [(pid, f"{pname} ({dest_of[pid]})") for pid, pname in park_hits] or [
651
+ (did, f"{dname} (destination, {len(parks_in(did))} parks)")
652
+ for did, dname in dest_hits.items()
653
+ ]
654
+ if not candidates:
655
+ raise SystemExit(
656
+ f'no park or destination matching "{wanted}".\n'
657
+ f" themeparks-backfill --list everything\n"
658
+ f' themeparks-backfill --list disney the ones matching "disney"'
659
+ )
660
+ if len(candidates) > 1:
661
+ lines = "\n".join(
662
+ f" {cid} {name}" for cid, name in sorted(candidates, key=lambda c: c[1])
663
+ )
664
+ raise SystemExit(
665
+ f'"{wanted}" matches {len(candidates)}. Pass an id, or the destination'
666
+ f" name to get all of its parks:\n{lines}"
667
+ )
668
+ return candidates
669
+
670
+
671
+ def _resolve(catalogue: list[tuple[str, str, str, str]], wanted: str) -> list[tuple[str, str]]:
672
+ """The parks to back fill, as (id, name), from whichever handle they have.
673
+
674
+ Accepts four things, because a customer has whichever one they found:
675
+
676
+ - a park id -> that park
677
+ - a DESTINATION id -> every park in it
678
+ - a park name -> that park
679
+ - a destination name -> every park in it
680
+
681
+ "Walt Disney World Resort" is the shape people actually have, and making them
682
+ look up four park uuids first was a wall for no reason.
683
+ """
684
+ by_id = _by_id(catalogue, wanted)
685
+ return by_id if by_id is not None else _by_name(catalogue, wanted)
686
+
687
+
688
+ UUID_LENGTH = 36
689
+ UUID_DASHES = 4
690
+
691
+
692
+ def _looks_like_id(value: str) -> bool:
693
+ """A uuid, loosely. Loose on purpose: the API decides what is valid, not us."""
694
+ return len(value) == UUID_LENGTH and value.count("-") == UUID_DASHES
695
+
696
+
697
+ def _print_list(tp: ThemeParks, needle: str | None) -> int:
698
+ """Parks grouped under their destination, so a destination id is visible too."""
699
+ catalogue = _catalogue(tp)
700
+ if needle:
701
+ lowered = _normalize(needle)
702
+ catalogue = [
703
+ c for c in catalogue if lowered in _normalize(c[1]) or lowered in _normalize(c[3])
704
+ ]
705
+ if not catalogue:
706
+ print(f'nothing matching "{needle}"', file=sys.stderr)
707
+ return 1
708
+
709
+ by_dest: dict[tuple[str, str], list[tuple[str, str]]] = {}
710
+ for pid, pname, did, dname in catalogue:
711
+ by_dest.setdefault((did, dname), []).append((pid, pname))
712
+ for (did, dname), parks in sorted(by_dest.items(), key=lambda kv: kv[0][1]):
713
+ # The destination line is indented left of its parks and labelled, so it
714
+ # reads as "pass this to get all of them" rather than as another park.
715
+ print(f"{did} {dname} <- destination: all {len(parks)} parks")
716
+ for pid, pname in sorted(parks, key=lambda p: p[1]):
717
+ print(f" {pid} {pname}")
718
+ return 0
719
+
720
+
721
+ EPILOG = """examples:
722
+ themeparks-backfill "Disneyland Park"
723
+ the whole daily history your plan reaches, as NDJSON, into the current
724
+ directory
725
+
726
+ themeparks-backfill "Walt Disney World Resort"
727
+ a DESTINATION: every park in it, one file each
728
+
729
+ themeparks-backfill --list disney
730
+ find an id, or check the spelling. Lists destinations with their parks
731
+ indented underneath. Works without a key.
732
+
733
+ themeparks-backfill "Epcot" --format csv --out ./data
734
+ one wide CSV row per entity per day, into ./data
735
+
736
+ themeparks-backfill <id-a> <id-b> <id-c>
737
+ several parks in one run, sharing one connection and one budget
738
+
739
+ exit codes:
740
+ 0 done
741
+ 75 the hourly history budget ran out. Progress is checkpointed; run the same
742
+ command again to continue. This is EX_TEMPFAIL, so a cron or systemd timer
743
+ retries instead of alerting.
744
+
745
+ how far back this reaches is your plan: 7 days with no key at all, 30 on a free
746
+ key, 400 on Pro, the whole archive on Business. It runs either way -- it asks the
747
+ API what you may see and starts there, so you never have to work it out, and it
748
+ never asks for a day you are not entitled to twice.
749
+ """
750
+
751
+
752
+ def main(argv: list[str] | None = None) -> int:
753
+ parser = argparse.ArgumentParser(
754
+ prog="themeparks-backfill",
755
+ description="Download a park's daily history to a file."
756
+ " One request per page, checkpointed, resumable.",
757
+ epilog=EPILOG,
758
+ formatter_class=argparse.RawDescriptionHelpFormatter,
759
+ )
760
+ parser.add_argument(
761
+ "parks",
762
+ nargs="*",
763
+ metavar="PARK",
764
+ help="park or DESTINATION, by name or id. A destination back fills every park in it.",
765
+ )
766
+ parser.add_argument(
767
+ "--list",
768
+ nargs="?",
769
+ const="",
770
+ metavar="TEXT",
771
+ dest="list_parks",
772
+ help="list ids and names, optionally filtered, then exit. Needs no key.",
773
+ )
774
+ parser.add_argument(
775
+ "--api-key",
776
+ default=os.environ.get("THEMEPARKS_API_KEY"),
777
+ help="API key. Defaults to $THEMEPARKS_API_KEY.",
778
+ )
779
+ parser.add_argument(
780
+ "--format",
781
+ choices=["ndjson", "csv"],
782
+ default="ndjson",
783
+ help="output format (default: ndjson)",
784
+ )
785
+ parser.add_argument(
786
+ "--overwrite",
787
+ action="store_true",
788
+ help="replace an existing file instead of refusing. Without it, a park"
789
+ " that finished is not fetched twice and a file this command did not"
790
+ " write is never touched.",
791
+ )
792
+ parser.add_argument(
793
+ "--out",
794
+ type=Path,
795
+ default=Path("."),
796
+ metavar="DIR",
797
+ help="output directory (default: .)",
798
+ )
799
+ args = parser.parse_args(argv)
800
+
801
+ _use_utf8(sys.stdout, sys.stderr)
802
+
803
+ # --list first: it is how you find a park, so it must work before you have a
804
+ # key and before you have decided to pay for anything.
805
+ if args.list_parks is not None:
806
+ with ThemeParks(api_key=args.api_key, user_agent=USER_AGENT) as tp:
807
+ return _print_list(tp, args.list_parks or None)
808
+
809
+ if not args.parks:
810
+ parser.error("which park or destination? try: themeparks-backfill --list disney")
811
+
812
+ # NO KEY IS NOT AN ERROR. It used to be: the command refused to start, with a
813
+ # message that said in the same breath that anonymous access reads 7 days.
814
+ # Telling someone the thing works and then declining to do it is worse than
815
+ # either. Anonymous reads 7 days, so it runs, says so, and says what a key
816
+ # would add -- which is also the honest sales pitch: the person evaluating
817
+ # whether to pay is exactly the person who should be able to run this.
818
+ if not args.api_key:
819
+ print(
820
+ "no API key: reading the 7 days anonymous access allows.\n"
821
+ " a free key reads 30 days, Pro 400, Business the whole archive\n"
822
+ " set THEMEPARKS_API_KEY, or pass --api-key\n"
823
+ " keys: https://www.themeparks.wiki/profile\n",
824
+ file=sys.stderr,
825
+ )
826
+
827
+ args.out.mkdir(parents=True, exist_ok=True)
828
+
829
+ # One client for every park: the connection pool is worth reusing and the
830
+ # budget is per account either way.
831
+ with ThemeParks(api_key=args.api_key, user_agent=USER_AGENT) as tp:
832
+ catalogue = _catalogue(tp)
833
+ dest_of = {pid: dname for pid, _, _, dname in catalogue}
834
+ targets: list[tuple[str, str]] = []
835
+ seen: set[str] = set()
836
+ for wanted in args.parks:
837
+ for pid, pname in _resolve(catalogue, wanted):
838
+ # A destination and one of its parks can both be named on one
839
+ # command line. Back filling the same park twice would double
840
+ # every row in the file.
841
+ if pid not in seen:
842
+ seen.add(pid)
843
+ targets.append((pid, pname))
844
+
845
+ # ALWAYS ECHO WHAT A NAME RESOLVED TO, even for a single park, and tell
846
+ # the caller to use the id next time.
847
+ #
848
+ # Names are for FINDING a park once. Ids are for asking for it. Twelve
849
+ # live parks contain "Hurricane Harbor" and the bare name is an exact
850
+ # match for the St. Louis one, so someone in Chicago could have typed a
851
+ # reasonable thing, got no warning, and loaded another park's history
852
+ # believing it was theirs. Echoing the resolution is what makes that
853
+ # visible; recommending the id is what stops it recurring in a script.
854
+ resolved_by_name = any(not _looks_like_id(w) for w in args.parks)
855
+ if len(targets) > 1:
856
+ print(f"{len(targets)} parks to back fill:", file=sys.stderr)
857
+ for pid, pname in targets:
858
+ where = dest_of.get(pid)
859
+ suffix = f" ({where})" if where and where != pname else ""
860
+ print(f" {pname}{suffix} {pid}", file=sys.stderr)
861
+ elif resolved_by_name:
862
+ pid, pname = targets[0]
863
+ where = dest_of.get(pid)
864
+ suffix = f" ({where})" if where and where != pname else ""
865
+ print(f"resolved to {pname}{suffix} {pid}", file=sys.stderr)
866
+ if resolved_by_name:
867
+ ids = " ".join(pid for pid, _ in targets)
868
+ print(
869
+ f" use the id next time — names are convenient once, ids are exact:\n"
870
+ f" themeparks-backfill {ids}",
871
+ file=sys.stderr,
872
+ )
873
+
874
+ for park_id, pname in targets:
875
+ status = backfill_park(tp, _Park(park_id, pname), args.out, args.format, args.overwrite)
876
+ if status != 0:
877
+ # Stop at the first exhausted budget. Carrying on to the next
878
+ # park only spends the retry-after on 429s.
879
+ return status
880
+ return 0
881
+
882
+
883
+ if __name__ == "__main__":
884
+ raise SystemExit(main())
File without changes
File without changes
File without changes