themeparks 3.2.0__tar.gz → 4.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {themeparks-3.2.0 → themeparks-4.0.0}/CHANGELOG.md +179 -0
  2. {themeparks-3.2.0 → themeparks-4.0.0}/PKG-INFO +20 -4
  3. {themeparks-3.2.0 → themeparks-4.0.0}/README.md +19 -3
  4. {themeparks-3.2.0 → themeparks-4.0.0}/pyproject.toml +7 -1
  5. themeparks-4.0.0/tests/fixtures/README.md +50 -0
  6. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/__init__.py +8 -1
  7. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/history.py +116 -6
  8. themeparks-4.0.0/themeparks/_generated/models.py +1337 -0
  9. themeparks-4.0.0/themeparks/_models_base.py +29 -0
  10. themeparks-4.0.0/themeparks/backfill.py +1326 -0
  11. themeparks-3.2.0/themeparks/_generated/models.py +0 -1121
  12. {themeparks-3.2.0 → themeparks-4.0.0}/.gitignore +0 -0
  13. {themeparks-3.2.0 → themeparks-4.0.0}/LICENSE +0 -0
  14. {themeparks-3.2.0 → themeparks-4.0.0}/MIGRATION.md +0 -0
  15. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_cache.py +0 -0
  16. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_client.py +0 -0
  17. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/__init__.py +0 -0
  18. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/dates.py +0 -0
  19. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/destinations.py +0 -0
  20. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/entity.py +0 -0
  21. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ergonomic/live.py +0 -0
  22. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_errors.py +0 -0
  23. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_generated/__init__.py +0 -0
  24. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_ratelimit.py +0 -0
  25. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_raw.py +0 -0
  26. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/_transport.py +0 -0
  27. {themeparks-3.2.0 → themeparks-4.0.0}/themeparks/py.typed +0 -0
@@ -1,5 +1,184 @@
1
1
  # Changelog
2
2
 
3
+ ## [4.0.0] - 2026-09-28
4
+
5
+ **3.3.0 was yanked: incomplete CSV export and a resume defect.**
6
+
7
+ A major version because **the CSV header changed**: fifteen columns were added and
8
+ the order is now the schema's, so a reader that takes columns by position gets the
9
+ wrong ones rather than an error. Read by name. The library API is backward
10
+ compatible.
11
+
12
+ Everything here came out of porting `themeparks-backfill` to the JavaScript SDK
13
+ and then diffing the two outputs over the same park, and out of six reviews of the
14
+ result. Two independent implementations reading one API disagree in exactly the
15
+ places one of them is wrong. Magic Kingdom's full archive now comes back
16
+ **byte for byte identical** from both SDKs: 94,223 rows, 41 columns, the only
17
+ differences being today's row, which grows as the day elapses.
18
+
19
+ ### Fixed
20
+
21
+ - **`themeparks-backfill "magic kingdom"` wrote the wrong park name into every
22
+ row.** A name that matched one park by substring returned the formatted display
23
+ label, so `parkName` read `Magic Kingdom Park (Walt Disney World® Resort)` for
24
+ all ~94,000 rows, and the resolution echo printed the destination twice. Four
25
+ live names reached it.
26
+
27
+ - **The CSV was missing ten of the thirty-six fields the API sends, on every row.**
28
+ `unknownMinutes`, the whole `inParkHours` block (the day's numbers limited to the
29
+ park's published hours -- usually the ones you want, since a ride "down" at 2am
30
+ is not down), `extremeWaits` (how many readings of 480+ minutes are folded into
31
+ the statistics, which is how you spot a feed error), and three of `singleRider`'s
32
+ five percentiles while `standby` carried all five. On a five-year Magic Kingdom
33
+ export, 72,200 of 94,223 rows were missing their in-park statistics. **The column
34
+ list is now derived from the model**, so it cannot drift again.
35
+
36
+ - **Vendored models were stale, and pydantic drops what it does not declare**, so
37
+ those three fields were deleted at parse time for every caller of `days()`, not
38
+ just for the CSV. Models regenerated, and every model now keeps fields the schema
39
+ does not declare (`themeparks._models_base.ApiModel`, `extra="allow"`), so a
40
+ field the API adds tomorrow survives parsing and reaches `model_dump()` and the
41
+ NDJSON output before this SDK knows it exists. It does not reach the CSV, whose
42
+ columns come from the schema.
43
+
44
+ - **A resumed download duplicated a day.** The checkpoint was the newest row
45
+ written; the page it came from covered further, because an entity that stopped
46
+ reporting has no rows for the tail days. A rerun re-fetched a day already in the
47
+ file and appended every row of it again, breaking the `(entityId, date)` key --
48
+ on the exit-75 path, which is the ordinary path for a long back fill. The
49
+ checkpoint is now the day the server's own `next` URL starts on.
50
+
51
+ - **A failure on a resumed run deleted everything already downloaded.** `written
52
+ == 0` means "this process wrote nothing", not "the file is empty". The state file
53
+ survived pointing mid-archive, so the next run appended only the tail and
54
+ recorded `complete: true`. Same for a window that closes under a resumed run --
55
+ a key rotated out of a scheduler's environment, a lapsed subscription -- which
56
+ additionally exited 0, so the scheduler logged success, and became a permanent
57
+ trap.
58
+
59
+ - **Resuming across versions, formats or SDKs corrupted the file.** One state file
60
+ served both formats, so `ndjson` then `csv` then `ndjson` doubled every row in
61
+ the first file; and the state carried nothing about the header, so 3.3.0's
62
+ 19-column file resumed under this build appended 41-field rows beneath it. The
63
+ state file is now `<parkId>.<format>.backfill-state.json` and records the SDK,
64
+ its version, a state version and a fingerprint of the exact header. Anything that
65
+ does not match is refused with a message saying why, never resumed.
66
+
67
+ - **A network failure or timeout now exits 75, not 1**, so a scheduler retries
68
+ rather than alerting; anything the API actively rejected still exits 1. The
69
+ JavaScript SDK had these the other way round.
70
+
71
+ - **A carriage return in an entity name was written unquoted on Python 3.9 and
72
+ 3.10**, so one row parsed as two with every later column shifted. The `csv`
73
+ module's QUOTE_MINIMAL only quotes characters that appear in the line terminator,
74
+ and this command sets LF; 3.11 changed the module to always quote CR and LF, so
75
+ the defect was invisible on a modern interpreter and live on two supported ones.
76
+ The CSV writer now does its own minimal quoting, which also makes the output
77
+ byte-identical across Python versions rather than only within one.
78
+
79
+ - **UTC timestamps are written `Z`, not `+00:00`**, and CSV line endings are LF.
80
+ Between them these accounted for 39,201 differing lines against the JavaScript
81
+ SDK's output for no difference in meaning.
82
+
83
+ - **One park's failure no longer abandons the rest of a destination.** Every park
84
+ is tried, what failed is named at the end, and the exit code still says something
85
+ went wrong. A spent budget still stops everything, deliberately.
86
+
87
+ - **A failed park no longer leaves a 0-byte file** that reads as "this park has no
88
+ history", including when the budget runs out before the first page.
89
+
90
+ - **A network failure, a full disk or Ctrl-C is a sentence, not a traceback.**
91
+
92
+ - **The user agent named neither version.** It was the literal
93
+ `themeparks-backfill/1`, and it replaced the SDK's own, so a support question had
94
+ no version to work from at either end.
95
+
96
+ - **`--list <text>` reported the wrong total**, printing "all 1 parks" for a
97
+ destination with six -- on the one line whose whole job is that number.
98
+
99
+ - **An ambiguous name listed the wrong candidates**, widening to substrings and
100
+ offering a third park that was not what was typed. It now lists the ids of the
101
+ parks that actually match, sorted by name.
102
+
103
+ - **A collection of nested models would have produced phantom columns** and then an
104
+ `AttributeError` on the first row. Duplicate column names are now impossible at
105
+ import rather than a wrong number under a right-looking header.
106
+
107
+ - **The NDJSON identity columns could be overwritten by the row** once models kept
108
+ undeclared fields.
109
+
110
+ ### Added
111
+
112
+ - **The CSV carries a UTF-8 BOM**, so Excel on Windows stops rendering
113
+ `Walt Disney World® Resort` as mojibake.
114
+ - **A cell a spreadsheet would execute is prefixed with an apostrophe** (`=`, `+`,
115
+ `-`, `@`, tab, CR). Numeric cells are left alone, so a negative number stays a
116
+ number.
117
+ - **`on_page` on `days()` and `days_with_entities()`**, called once every row of a
118
+ page has been yielded, with a `HistoryPage` (`start`, `end`, `next_url`). The page
119
+ boundary is the server's own answer to "where do I carry on", and the rows cannot
120
+ tell you.
121
+ - **`--version`.**
122
+ - `EntityRef` and `HistoryPage` are exported from the package.
123
+ - `tests/fixtures/csv_contract.json`, an identical copy of which lives in the
124
+ JavaScript SDK. Both suites assert their column list against it, because this is
125
+ one command with two implementations and a customer using both should get one
126
+ file format.
127
+
128
+ ### Changed
129
+
130
+ - The `themeparks-backfill` entry point is `themeparks.backfill:cli`, which adds
131
+ the top-level error handling. `main()` is unchanged for anyone calling it.
132
+ - Model equality and `model_json_schema()` reflect `extra="allow"`: two responses
133
+ differing only in an undeclared field now compare unequal, and dumps may contain
134
+ fields the schema does not list.
135
+
136
+ ## [3.3.0] - 2026-09-28
137
+
138
+ ### Added
139
+
140
+ - **`themeparks-backfill`: the archive download as a command.** It was an example
141
+ to copy off GitHub. The first paying customer followed that link and had to
142
+ work out that the library needed installing, then what the arguments were,
143
+ then read a traceback. Now:
144
+
145
+ ```bash
146
+ pip install themeparks
147
+ themeparks-backfill "Disneyland Park"
148
+ ```
149
+
150
+ - Takes a park or a **destination**, by name or id. A destination back fills
151
+ every park in it, one file each. `"Walt Disney World Resort"` is the handle
152
+ people actually have; four park uuids is not.
153
+ - `--list [text]` prints destinations with their parks underneath, and **needs
154
+ no key**, so you can find your park before deciding whether to pay.
155
+ - Refuses to guess between two matches. Two parks are named exactly
156
+ "Disneyland Park" (Anaheim and Paris), so the candidate list names the
157
+ destination as well.
158
+ - **Runs without a key**, reading the 7 days anonymous access allows, and says
159
+ what a key would add. It used to refuse to start with a message that
160
+ mentioned anonymous access in the same breath.
161
+ - NDJSON by default, `--format csv` for one wide row per entity per day.
162
+ - Checkpoints against the hourly history budget and exits 75 (`EX_TEMPFAIL`),
163
+ so a cron or timer retries rather than alerting. Re-running continues.
164
+
165
+ `python -m themeparks.backfill` is the same thing. `examples/backfill.py`
166
+ remains as a shim so existing links keep working.
167
+
168
+ ### Fixed
169
+
170
+ - **The history window recovery now actually works.** 3.2.0's `examples/backfill.py`
171
+ read `earliestAllowedDate` from the top level of the 403 body; the API nests it
172
+ under `error`. So the recovery shipped doing nothing and a Pro customer still
173
+ got a traceback on their first request. The tests passed because the fixture was
174
+ built from the formatted text in a traceback rather than a real response, so the
175
+ code and the test were wrong together. The fixture is now captured from
176
+ production and a test fails if anyone flattens it.
177
+
178
+ The underlying gap is in the API, not the client: `/history/coverage` reports
179
+ where the archive starts and where your window ends, and nothing about where
180
+ your window begins. Until it does, the 403 is the only place that date exists.
181
+
3
182
  ## [3.2.0] - 2026-09-26
4
183
 
5
184
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: themeparks
3
- Version: 3.2.0
3
+ Version: 4.0.0
4
4
  Summary: Official SDK for the ThemeParks.wiki API
5
5
  Project-URL: Homepage, https://api.themeparks.wiki
6
6
  Project-URL: Source, https://github.com/ThemeParks/ThemeParks_Python
@@ -350,9 +350,25 @@ except BudgetExhaustedError as exc:
350
350
  print(f"resume in {exc.retry_after:.0f}s")
351
351
  ```
352
352
 
353
- A complete backfill script with resume and CSV output is in
354
- [`examples/backfill.py`](examples/backfill.py); it pulls Disneyland Resort's
355
- whole daily archive, 98,452 rows, in one run.
353
+ ### Or skip the code: there is a command
354
+
355
+ Installing the library installs `themeparks-backfill`, which does all of the
356
+ above and stops before the walls:
357
+
358
+ ```bash
359
+ themeparks-backfill "Disneyland Park" # a park, by name or id
360
+ themeparks-backfill "Walt Disney World Resort" # a destination: every park in it
361
+ themeparks-backfill --list disney # find an id. Needs no key.
362
+ ```
363
+
364
+ It reads how far back your own key may ask and starts there, writes NDJSON or
365
+ `--format csv`, names every row with the park and the entity, records what it
366
+ has done so re-running never duplicates a file, and exits 75 when the hourly
367
+ history budget runs out so a scheduler retries rather than alerts.
368
+
369
+ `python -m themeparks.backfill` is the same thing, which is the one to use if
370
+ `pip install --user` put the script somewhere off your PATH. `themeparks-backfill
371
+ --help` has the rest.
356
372
 
357
373
  ## Low-level escape hatch
358
374
 
@@ -311,9 +311,25 @@ except BudgetExhaustedError as exc:
311
311
  print(f"resume in {exc.retry_after:.0f}s")
312
312
  ```
313
313
 
314
- A complete backfill script with resume and CSV output is in
315
- [`examples/backfill.py`](examples/backfill.py); it pulls Disneyland Resort's
316
- whole daily archive, 98,452 rows, in one run.
314
+ ### Or skip the code: there is a command
315
+
316
+ Installing the library installs `themeparks-backfill`, which does all of the
317
+ above and stops before the walls:
318
+
319
+ ```bash
320
+ themeparks-backfill "Disneyland Park" # a park, by name or id
321
+ themeparks-backfill "Walt Disney World Resort" # a destination: every park in it
322
+ themeparks-backfill --list disney # find an id. Needs no key.
323
+ ```
324
+
325
+ It reads how far back your own key may ask and starts there, writes NDJSON or
326
+ `--format csv`, names every row with the park and the entity, records what it
327
+ has done so re-running never duplicates a file, and exits 75 when the hourly
328
+ history budget runs out so a scheduler retries rather than alerts.
329
+
330
+ `python -m themeparks.backfill` is the same thing, which is the one to use if
331
+ `pip install --user` put the script somewhere off your PATH. `themeparks-backfill
332
+ --help` has the rest.
317
333
 
318
334
  ## Low-level escape hatch
319
335
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "themeparks"
7
- version = "3.2.0"
7
+ version = "4.0.0"
8
8
  description = "Official SDK for the ThemeParks.wiki API"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -30,6 +30,12 @@ dependencies = [
30
30
  "eval-type-backport>=0.2; python_version < '3.10'",
31
31
  ]
32
32
 
33
+ [project.scripts]
34
+ # The archive backfill, as a command rather than a file to copy off GitHub.
35
+ # `pip install themeparks` then `themeparks-backfill "Disneyland Park"` is the
36
+ # whole path from nothing to a file of history.
37
+ themeparks-backfill = "themeparks.backfill:cli"
38
+
33
39
  [project.urls]
34
40
  Homepage = "https://api.themeparks.wiki"
35
41
  Source = "https://github.com/ThemeParks/ThemeParks_Python"
@@ -0,0 +1,50 @@
1
+ # Fixtures captured from the live API
2
+
3
+ Not hand-written. Each file here was taken from a real response, and the header
4
+ below says which endpoint and when. That matters because of a specific failure
5
+ this project keeps repeating.
6
+
7
+ ## Why this directory exists
8
+
9
+ On 2026-09-28 a 403-recovery fix shipped doing nothing. Its tests passed because
10
+ the fixture was built from the *formatted text of a traceback* rather than from a
11
+ response body, so the code and the test were wrong together and agreed with each
12
+ other. The memory note for that pattern is `self-confirming-harness`, and it was
13
+ its fifth occurrence in one day.
14
+
15
+ A fixture the author of the code also invented proves only that the two are
16
+ consistent. A fixture taken from the real thing can disagree.
17
+
18
+ ## destinations_slice.json
19
+
20
+ `GET https://api.themeparks.wiki/v1/destinations`, captured 2026-09-28, trimmed
21
+ to 9 destinations and 20 parks and otherwise verbatim — every id, name and slug
22
+ is exactly what the API returned.
23
+
24
+ It is trimmed to keep, deliberately, every case that broke name resolution:
25
+
26
+ | Case | Why it is here |
27
+ |---|---|
28
+ | `Walt Disney World® Resort` | U+00AE. The command's own documented example, `"Walt Disney World Resort"`, did not match it. |
29
+ | `LEGOLAND® Korea` | the same, on a destination whose park shares the name |
30
+ | `Walibi Rhône-Alpes` | U+00F4, so a normaliser must decompose accents |
31
+ | `Knott's Berry Farm` + `Knott’s Soak City` | ASCII `'` and U+2019 **in one destination**. Nobody types the curly one. |
32
+ | `Disneyland Park` ×2 | Anaheim and Paris, identical park names. The reason the candidate list names the destination. |
33
+ | `Hurricane Harbor` + `Hurricane Harbor Chicago` | an exact match with a substring rival: the shape that silently downloaded the wrong park |
34
+ | `Cedar Point` | destination name == park name, with a second park, so "exact destination beats exact park" is observable |
35
+
36
+ **Do not edit these by hand.** Re-capture them. If a name upstream has drifted,
37
+ that is a real change and the test should notice.
38
+
39
+ ## mk_park_daily_page1.json / mk_park_daily_page2.json
40
+
41
+ Two consecutive pages of one real request, captured 2026-09-28:
42
+ `GET /entity/75ea578a-adc8-4116-a54d-dccb60765ef9/history/daily?from=2026-08-01&to=2026-09-20`
43
+ then its `next` followed verbatim. Trimmed to three entities (an attraction, a
44
+ show, a restaurant); `range`, `next` and every row are the server's.
45
+
46
+ They are the oracle for resumable paging. Page one covers through 2026-08-31 and
47
+ the server says carry on at 2026-09-01, but two of its three entities have no
48
+ rows after 2026-08-30 -- so a checkpoint taken from the newest ROW rewinds and
49
+ re-downloads days already written. The same files are in the JavaScript SDK, so
50
+ both ports are tested against identical bytes.
@@ -1,7 +1,12 @@
1
1
  from themeparks._cache import Cache, CacheConfig, InMemoryLRUCache
2
2
  from themeparks._client import AsyncThemeParks, ThemeParks
3
3
  from themeparks._ergonomic.dates import parse_api_datetime
4
- from themeparks._ergonomic.history import BudgetExhaustedError, HistorySpan
4
+ from themeparks._ergonomic.history import (
5
+ BudgetExhaustedError,
6
+ EntityRef,
7
+ HistoryPage,
8
+ HistorySpan,
9
+ )
5
10
  from themeparks._ergonomic.live import current_wait_time, iter_queues
6
11
  from themeparks._errors import (
7
12
  APIError,
@@ -17,6 +22,8 @@ __all__ = [
17
22
  "APIError",
18
23
  "AsyncThemeParks",
19
24
  "BudgetExhaustedError",
25
+ "EntityRef",
26
+ "HistoryPage",
20
27
  "HistorySpan",
21
28
  "RateLimit",
22
29
  "RateLimits",
@@ -20,7 +20,7 @@ the cheap path here without having to know the expensive one exists.
20
20
 
21
21
  from __future__ import annotations
22
22
 
23
- from collections.abc import AsyncIterator, Iterator
23
+ from collections.abc import AsyncIterator, Callable, Iterator
24
24
  from datetime import date as _date
25
25
  from typing import Any, NamedTuple, Union
26
26
 
@@ -111,17 +111,87 @@ def _reraise_if_too_long(exc: RateLimitError, max_wait: float) -> None:
111
111
  raise exc
112
112
 
113
113
 
114
- def _daily_rows(envelope: DailyEnvelope) -> Iterator[tuple[str, HistoryDailyRow]]:
115
- """Yield (entity id, row). A park envelope carries many entities; an entity
116
- envelope carries its own rows, so both flatten to the same stream."""
114
+ class HistoryPage(NamedTuple):
115
+ """One page of daily history, as the server described it.
116
+
117
+ `start` and `end` are the park-local days the page ACTUALLY covered, which is
118
+ not the range you asked for: a park daily call serves at most 31 days, so a
119
+ 50-day request comes back as 31 days plus a `next`. `next_url` is the URL of
120
+ the following page, or None on the last one.
121
+
122
+ This exists for resumable downloads. A checkpoint taken from the ROWS is
123
+ wrong in both directions: the newest row's date can be earlier than the page
124
+ covered, because an entity that stopped reporting has no rows for the tail
125
+ days, so resuming there re-fetches days already written and duplicates them;
126
+ and a half-written page is indistinguishable from a finished one. The page
127
+ boundary is the server's own answer to "where do I carry on", so it is the
128
+ only safe checkpoint.
129
+ """
130
+
131
+ start: str
132
+ end: str
133
+ next_url: str | None
134
+
135
+
136
+ class EntityRef(NamedTuple):
137
+ """Who a history row belongs to, AS THE HISTORY RESPONSE REPORTS IT.
138
+
139
+ The name matters and the source of it matters more. A park's current
140
+ `/children` list gives today's name, which is the wrong label for a row
141
+ recorded years ago: rides are renamed, and stamping today's name on old data
142
+ quietly rewrites history. The history envelope carries its own `name` and
143
+ `entityType` per entity, and that is the name to use.
144
+ """
145
+
146
+ id: str
147
+ name: str
148
+ entity_type: str
149
+
150
+
151
+ def _ref(entity: Any) -> EntityRef:
152
+ kind = getattr(entity, "entityType", None)
153
+ # The generated models use an enum, and str(EntityType.SHOW) is
154
+ # "EntityType.SHOW". `.value` is what the API sends.
155
+ inner = getattr(kind, "value", kind)
156
+ return EntityRef(
157
+ entity.id, getattr(entity, "name", "") or "", "" if inner is None else str(inner)
158
+ )
159
+
160
+
161
+ def _page_of(envelope: DailyEnvelope) -> HistoryPage:
162
+ """The page an envelope represents, for :class:`HistoryPage`'s callers."""
163
+ rng = getattr(envelope, "range", None)
164
+ nxt = getattr(envelope, "next", None)
165
+ return HistoryPage(
166
+ getattr(rng, "from_", "") or "",
167
+ getattr(rng, "to", "") or "",
168
+ nxt or None,
169
+ )
170
+
171
+
172
+ def _daily_entity_rows(envelope: DailyEnvelope) -> Iterator[tuple[EntityRef, HistoryDailyRow]]:
173
+ """Yield (entity ref, row), keeping the name the response gave.
174
+
175
+ `_daily_rows` below is the same walk with the ref flattened to its id, kept
176
+ because `days()` has yielded `(id, row)` since 3.0 and that shape is public.
177
+ """
117
178
  entities = getattr(envelope, "entities", None)
118
179
  if entities is not None:
119
180
  for entity in entities:
181
+ ref = _ref(entity)
120
182
  for row in entity.days or []:
121
- yield (entity.id, row)
183
+ yield (ref, row)
122
184
  return
185
+ ref = _ref(envelope)
123
186
  for row in getattr(envelope, "days", None) or []:
124
- yield (envelope.id, row)
187
+ yield (ref, row)
188
+
189
+
190
+ def _daily_rows(envelope: DailyEnvelope) -> Iterator[tuple[str, HistoryDailyRow]]:
191
+ """Yield (entity id, row). A park envelope carries many entities; an entity
192
+ envelope carries its own rows, so both flatten to the same stream."""
193
+ for ref, row in _daily_entity_rows(envelope):
194
+ yield (ref.id, row)
125
195
 
126
196
 
127
197
  def _raw_rows(envelope: RawEnvelope) -> Iterator[tuple[str, HistoryRow]]:
@@ -154,21 +224,58 @@ class HistoryApi:
154
224
  """
155
225
  return _span(self.coverage())
156
226
 
227
+ def days_with_entities(
228
+ self,
229
+ start: str | _date | None = None,
230
+ end: str | _date | None = None,
231
+ *,
232
+ max_wait: float = DEFAULT_MAX_WAIT_SECONDS,
233
+ on_page: Callable[[HistoryPage], None] | None = None,
234
+ ) -> Iterator[tuple[EntityRef, HistoryDailyRow]]:
235
+ """`days()`, but each row arrives with the entity's name and type.
236
+
237
+ Use this when you are writing history to a file. The name comes from the
238
+ history response itself, so it is the label that response gives for those
239
+ rows rather than the park's current `/children` list -- rides get renamed,
240
+ and today's name on a row from three years ago is a quiet rewrite of the
241
+ record.
242
+
243
+ It also saves a request: the name is already in the payload, so nothing
244
+ needs to ask what an id refers to.
245
+
246
+ `on_page` is called once every row of a page has been yielded, with a
247
+ :class:`HistoryPage`. Checkpoint on that, never on the last row you saw.
248
+ """
249
+ envelope: DailyEnvelope | None = self._first_daily(start, end, max_wait)
250
+ while envelope is not None:
251
+ yield from _daily_entity_rows(envelope)
252
+ # AFTER the rows, never before: a caller checkpointing on this has to
253
+ # be able to trust that everything the page held is already written.
254
+ if on_page is not None:
255
+ on_page(_page_of(envelope))
256
+ envelope = self._next_daily(envelope, max_wait)
257
+
157
258
  def days(
158
259
  self,
159
260
  start: str | _date | None = None,
160
261
  end: str | _date | None = None,
161
262
  *,
162
263
  max_wait: float = DEFAULT_MAX_WAIT_SECONDS,
264
+ on_page: Callable[[HistoryPage], None] | None = None,
163
265
  ) -> Iterator[tuple[str, HistoryDailyRow]]:
164
266
  """One summary row per park-local day, as (entity id, row).
165
267
 
166
268
  Pages automatically. Given a park id this uses the park call, which
167
269
  answers every entity in the park in one request.
270
+
271
+ `days_with_entities()` is the same stream with the entity's name and type
272
+ attached; this shape is kept because it is public API from 3.0.
168
273
  """
169
274
  envelope: DailyEnvelope | None = self._first_daily(start, end, max_wait)
170
275
  while envelope is not None:
171
276
  yield from _daily_rows(envelope)
277
+ if on_page is not None:
278
+ on_page(_page_of(envelope))
172
279
  envelope = self._next_daily(envelope, max_wait)
173
280
 
174
281
  def _first_daily(
@@ -237,6 +344,7 @@ class AsyncHistoryApi:
237
344
  end: str | _date | None = None,
238
345
  *,
239
346
  max_wait: float = DEFAULT_MAX_WAIT_SECONDS,
347
+ on_page: Callable[[HistoryPage], None] | None = None,
240
348
  ) -> AsyncIterator[tuple[str, HistoryDailyRow]]:
241
349
  try:
242
350
  envelope: DailyEnvelope | None = await self._raw.get_entity_history_daily(
@@ -248,6 +356,8 @@ class AsyncHistoryApi:
248
356
  while envelope is not None:
249
357
  for pair in _daily_rows(envelope):
250
358
  yield pair
359
+ if on_page is not None:
360
+ on_page(_page_of(envelope))
251
361
  nxt = getattr(envelope, "next", None)
252
362
  if not nxt:
253
363
  return