From c5dcd21cf569bc4a01d00a900c691ed1026bbaa7 Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Mon, 28 Sep 2026 21:29:06 +0100 Subject: [PATCH 1/9] feat(history): changes() exposes the opening state; span().final_through `history.changes()` yielded the rows of the raw history response and dropped its `opening` object, the state in force at the start of the range. Without it the minutes between midnight and an entity's first change have no known status, so a day rebuilt from raw history disagrees with the daily summary whenever a ride is still running from the night before. `changes()` now returns a `HistoryChanges` iterator: iterating it yields exactly the `(entity id, row)` pairs it always did, and its `opening` attribute is a dict of `HistoryOpening` keyed by entity id, one per entity in the response. The request is still made on first use, and reading `opening` costs no extra request. The async mirror returns `AsyncHistoryChanges`, whose `opening` is readable once the response has arrived (iterate first, or `await .load()`). `HistorySpan.final_through` is the newest day whose daily row will not change again: the earlier of `recorded_to` and `retrievable_through`. A property, so the three-field tuple is unchanged. Tested against a real capture (Space Mountain, 2026-09-26, whose opening is OPERATING): opening plus rows cover the day with no unknown second, and the rebuilt first open and last close match the daily row's. Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/api/history.md | 12 + tests/fixtures/README.md | 14 + .../space_mountain_daily_2026-09-26.json | 47 + .../space_mountain_history_2026-09-26.json | 3356 +++++++++++++++++ tests/unit/test_history.py | 21 +- tests/unit/test_history_opening.py | 210 ++ themeparks/__init__.py | 4 + themeparks/_ergonomic/history.py | 173 +- 8 files changed, 3814 insertions(+), 23 deletions(-) create mode 100644 tests/fixtures/space_mountain_daily_2026-09-26.json create mode 100644 tests/fixtures/space_mountain_history_2026-09-26.json create mode 100644 tests/unit/test_history_opening.py diff --git a/docs/api/history.md b/docs/api/history.md index 8dd9689..acb1f49 100644 --- a/docs/api/history.md +++ b/docs/api/history.md @@ -15,6 +15,18 @@ hundred times fewer calls than the same data fetched ride by ride. options: heading_level: 2 +::: themeparks.HistoryChanges + options: + heading_level: 2 + +::: themeparks.AsyncHistoryChanges + options: + heading_level: 2 + +::: themeparks.HistorySpan + options: + heading_level: 2 + ::: themeparks.BudgetExhaustedError options: heading_level: 2 diff --git a/tests/fixtures/README.md b/tests/fixtures/README.md index 253ebc4..e25a499 100644 --- a/tests/fixtures/README.md +++ b/tests/fixtures/README.md @@ -48,3 +48,17 @@ the server says carry on at 2026-09-01, but two of its three entities have no rows after 2026-08-30 -- so a checkpoint taken from the newest ROW rewinds and re-downloads days already written. The same files are in the JavaScript SDK, so both ports are tested against identical bytes. + +## space_mountain_history_2026-09-26.json / space_mountain_daily_2026-09-26.json + +`GET /entity/b2260923-9315-40fd-9c6b-44dd811dbe64/history?date=2026-09-26` and +`GET /entity/b2260923-9315-40fd-9c6b-44dd811dbe64/history/daily?date=2026-09-26`, +both captured 2026-09-28 without a key, verbatim. + +They are the oracle for `changes().opening`. Space Mountain's opening that day is +`OPERATING`, because the previous night's hours ran past midnight, and its first +row is the close at 00:01:03 local. A day rebuilt from the rows alone has 63 +seconds with no known status; rebuilt from the opening plus the rows it has +none, and its first open and last close are the daily row's `firstOperatingAt` +and `lastClosedAt`. If a re-capture picks a day whose opening is `CLOSED`, the +tests stop being able to tell the two apart, and one of them says so. diff --git a/tests/fixtures/space_mountain_daily_2026-09-26.json b/tests/fixtures/space_mountain_daily_2026-09-26.json new file mode 100644 index 0000000..bc8612d --- /dev/null +++ b/tests/fixtures/space_mountain_daily_2026-09-26.json @@ -0,0 +1,47 @@ +{ + "id": "b2260923-9315-40fd-9c6b-44dd811dbe64", + "name": "Space Mountain", + "entityType": "ATTRACTION", + "parentId": "75ea578a-adc8-4116-a54d-dccb60765ef9", + "destinationId": "e957da41-3552-4cf6-b636-5babc5cbc4e5", + "timezone": "America/New_York", + "range": { + "from": "2026-09-26", + "to": "2026-09-26" + }, + "coverage": { + "firstRecordedAt": "2021-07-03" + }, + "days": [ + { + "date": "2026-09-26", + "firstOperatingAt": "2026-09-26T11:30:53Z", + "lastClosedAt": "2026-09-27T03:01:04Z", + "operatingMinutes": 929, + "downMinutes": 0, + "unknownMinutes": 4, + "standby": { + "min": 5, + "p50": 40, + "mean": 37, + "p90": 55, + "max": 65 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 929, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 40, + "mean": 37, + "p90": 55, + "max": 65 + } + }, + "changes": 187 + } + ], + "next": null +} \ No newline at end of file diff --git a/tests/fixtures/space_mountain_history_2026-09-26.json b/tests/fixtures/space_mountain_history_2026-09-26.json new file mode 100644 index 0000000..f0ee8f7 --- /dev/null +++ b/tests/fixtures/space_mountain_history_2026-09-26.json @@ -0,0 +1,3356 @@ +{ + "id": "b2260923-9315-40fd-9c6b-44dd811dbe64", + "name": "Space Mountain", + "entityType": "ATTRACTION", + "parentId": "75ea578a-adc8-4116-a54d-dccb60765ef9", + "destinationId": "e957da41-3552-4cf6-b636-5babc5cbc4e5", + "timezone": "America/New_York", + "range": { + "from": "2026-09-26", + "to": "2026-09-26" + }, + "coverage": { + "firstRecordedAt": "2021-07-03" + }, + "opening": { + "time": "2026-09-26T04:00:00Z", + "observedAt": "2026-09-26T03:56:53Z", + "observedAtByKind": { + "status": "2026-09-26T03:56:53Z", + "queue.STANDBY": "2026-09-26T03:56:53Z", + "queue.RETURN_TIME": "2026-09-26T03:56:53Z" + }, + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 15 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + "history": [ + { + "time": "2026-09-26T04:01:03Z", + "changed": [ + "status", + "queue.STANDBY.waitTime" + ], + "status": "CLOSED", + "queue": { + "STANDBY": { + "waitTime": null + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T04:04:53Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "CLOSED", + "queue": { + "STANDBY": { + "waitTime": null + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:10:00-04:00", + "returnEnd": "2026-09-26T09:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T11:30:53Z", + "changed": [ + "status", + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 10 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:10:00-04:00", + "returnEnd": "2026-09-26T09:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T11:31:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:10:00-04:00", + "returnEnd": "2026-09-26T09:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T11:34:39Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:00:00-04:00", + "returnEnd": "2026-09-26T09:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T11:55:45Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:05:00-04:00", + "returnEnd": "2026-09-26T10:05:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:05:43Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:55:00-04:00", + "returnEnd": "2026-09-26T09:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:09:16Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:10:00-04:00", + "returnEnd": "2026-09-26T09:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:15:06Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:55:00-04:00", + "returnEnd": "2026-09-26T09:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:21:18Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:40:00-04:00", + "returnEnd": "2026-09-26T09:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:24:32Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:25:00-04:00", + "returnEnd": "2026-09-26T09:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:30:01Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T08:35:00-04:00", + "returnEnd": "2026-09-26T09:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:34:37Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:05:00-04:00", + "returnEnd": "2026-09-26T10:05:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:39:54Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:00:00-04:00", + "returnEnd": "2026-09-26T10:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:46:10Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:15:00-04:00", + "returnEnd": "2026-09-26T10:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:55:30Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 5 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:20:00-04:00", + "returnEnd": "2026-09-26T10:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T12:57:59Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 10 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:20:00-04:00", + "returnEnd": "2026-09-26T10:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:04:51Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 10 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:25:00-04:00", + "returnEnd": "2026-09-26T10:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:11:06Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 10 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:35:00-04:00", + "returnEnd": "2026-09-26T10:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:15:14Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 10 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:30:00-04:00", + "returnEnd": "2026-09-26T10:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:15:56Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 15 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:30:00-04:00", + "returnEnd": "2026-09-26T10:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:18:01Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:30:00-04:00", + "returnEnd": "2026-09-26T10:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:20:26Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:45:00-04:00", + "returnEnd": "2026-09-26T10:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:24:44Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T09:55:00-04:00", + "returnEnd": "2026-09-26T10:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:29:45Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T10:10:00-04:00", + "returnEnd": "2026-09-26T11:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:34:42Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T10:20:00-04:00", + "returnEnd": "2026-09-26T11:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:39:10Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T10:25:00-04:00", + "returnEnd": "2026-09-26T11:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:45:23Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T10:35:00-04:00", + "returnEnd": "2026-09-26T11:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:49:47Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T10:40:00-04:00", + "returnEnd": "2026-09-26T11:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T13:54:23Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T10:45:00-04:00", + "returnEnd": "2026-09-26T11:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:00:06Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T10:55:00-04:00", + "returnEnd": "2026-09-26T11:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:06:00Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:00:00-04:00", + "returnEnd": "2026-09-26T12:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:08:57Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:15:00-04:00", + "returnEnd": "2026-09-26T12:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:15:28Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:20:00-04:00", + "returnEnd": "2026-09-26T12:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:21:30Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:25:00-04:00", + "returnEnd": "2026-09-26T12:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:22:03Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:25:00-04:00", + "returnEnd": "2026-09-26T12:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:24:32Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:20:00-04:00", + "returnEnd": "2026-09-26T12:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:30:41Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:45:00-04:00", + "returnEnd": "2026-09-26T12:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:40:04Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:55:00-04:00", + "returnEnd": "2026-09-26T12:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:43:04Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:55:00-04:00", + "returnEnd": "2026-09-26T12:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:45:01Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:05:00-04:00", + "returnEnd": "2026-09-26T13:05:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:49:27Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:15:00-04:00", + "returnEnd": "2026-09-26T13:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:54:56Z", + "changed": [ + "queue.STANDBY.waitTime", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:20:00-04:00", + "returnEnd": "2026-09-26T13:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T14:59:25Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:10:00-04:00", + "returnEnd": "2026-09-26T13:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:05:08Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:30:00-04:00", + "returnEnd": "2026-09-26T13:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:06:12Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:30:00-04:00", + "returnEnd": "2026-09-26T13:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:11:26Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T11:15:00-04:00", + "returnEnd": "2026-09-26T12:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:14:31Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:00:00-04:00", + "returnEnd": "2026-09-26T13:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:16:56Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:00:00-04:00", + "returnEnd": "2026-09-26T13:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:20:46Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:25:00-04:00", + "returnEnd": "2026-09-26T13:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:20:56Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:25:00-04:00", + "returnEnd": "2026-09-26T13:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:24:25Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:50:00-04:00", + "returnEnd": "2026-09-26T13:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:30:10Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:55:00-04:00", + "returnEnd": "2026-09-26T13:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:36:28Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:50:00-04:00", + "returnEnd": "2026-09-26T13:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:39:46Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T13:20:00-04:00", + "returnEnd": "2026-09-26T14:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:44:59Z", + "changed": [ + "queue.STANDBY.waitTime", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T13:35:00-04:00", + "returnEnd": "2026-09-26T14:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:49:12Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T13:50:00-04:00", + "returnEnd": "2026-09-26T14:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:54:04Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T13:50:00-04:00", + "returnEnd": "2026-09-26T14:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:54:08Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:40:00-04:00", + "returnEnd": "2026-09-26T13:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:58:12Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T12:40:00-04:00", + "returnEnd": "2026-09-26T13:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T15:59:02Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T14:25:00-04:00", + "returnEnd": "2026-09-26T15:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:04:37Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T14:40:00-04:00", + "returnEnd": "2026-09-26T15:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:06:56Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T14:40:00-04:00", + "returnEnd": "2026-09-26T15:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:10:55Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T14:45:00-04:00", + "returnEnd": "2026-09-26T15:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:14:03Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:00:00-04:00", + "returnEnd": "2026-09-26T16:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:17:56Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:00:00-04:00", + "returnEnd": "2026-09-26T16:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:20:16Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T14:45:00-04:00", + "returnEnd": "2026-09-26T15:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:26:36Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:00:00-04:00", + "returnEnd": "2026-09-26T16:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:27:05Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:00:00-04:00", + "returnEnd": "2026-09-26T16:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:29:47Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:20:00-04:00", + "returnEnd": "2026-09-26T16:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:34:56Z", + "changed": [ + "queue.STANDBY.waitTime", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:30:00-04:00", + "returnEnd": "2026-09-26T16:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:36:03Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:30:00-04:00", + "returnEnd": "2026-09-26T16:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:39:09Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:40:00-04:00", + "returnEnd": "2026-09-26T16:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:42:19Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:40:00-04:00", + "returnEnd": "2026-09-26T16:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:44:58Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:50:00-04:00", + "returnEnd": "2026-09-26T16:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:46:56Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:50:00-04:00", + "returnEnd": "2026-09-26T16:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:50:58Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:35:00-04:00", + "returnEnd": "2026-09-26T16:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:52:55Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:35:00-04:00", + "returnEnd": "2026-09-26T16:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:58:15Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T15:35:00-04:00", + "returnEnd": "2026-09-26T16:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T16:59:59Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T14:30:00-04:00", + "returnEnd": "2026-09-26T15:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:04:14Z", + "changed": [ + "queue.STANDBY.waitTime", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T14:40:00-04:00", + "returnEnd": "2026-09-26T15:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:14:41Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T16:35:00-04:00", + "returnEnd": "2026-09-26T17:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:20:52Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T16:55:00-04:00", + "returnEnd": "2026-09-26T17:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:24:04Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T17:00:00-04:00", + "returnEnd": "2026-09-26T18:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:26:24Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T17:00:00-04:00", + "returnEnd": "2026-09-26T18:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:30:05Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T17:10:00-04:00", + "returnEnd": "2026-09-26T18:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:33:11Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T17:20:00-04:00", + "returnEnd": "2026-09-26T18:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:39:36Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T17:25:00-04:00", + "returnEnd": "2026-09-26T18:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:45:34Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T17:40:00-04:00", + "returnEnd": "2026-09-26T18:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:49:07Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T17:35:00-04:00", + "returnEnd": "2026-09-26T18:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:55:03Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T17:55:00-04:00", + "returnEnd": "2026-09-26T18:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T17:58:00Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T14:30:00-04:00", + "returnEnd": "2026-09-26T15:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:04:15Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:20:00-04:00", + "returnEnd": "2026-09-26T19:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:10:37Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:10:00-04:00", + "returnEnd": "2026-09-26T19:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:12:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:10:00-04:00", + "returnEnd": "2026-09-26T19:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:14:56Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:35:00-04:00", + "returnEnd": "2026-09-26T19:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:19:58Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:45:00-04:00", + "returnEnd": "2026-09-26T19:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:22:56Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:45:00-04:00", + "returnEnd": "2026-09-26T19:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:24:54Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:55:00-04:00", + "returnEnd": "2026-09-26T19:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:26:03Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:55:00-04:00", + "returnEnd": "2026-09-26T19:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:29:14Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:05:00-04:00", + "returnEnd": "2026-09-26T20:05:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:32:55Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:05:00-04:00", + "returnEnd": "2026-09-26T20:05:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:35:36Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:15:00-04:00", + "returnEnd": "2026-09-26T20:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:40:16Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:30:00-04:00", + "returnEnd": "2026-09-26T20:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:44:59Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:20:00-04:00", + "returnEnd": "2026-09-26T20:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:49:31Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:45:00-04:00", + "returnEnd": "2026-09-26T20:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T18:54:39Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:00:00-04:00", + "returnEnd": "2026-09-26T21:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:00:38Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:45:00-04:00", + "returnEnd": "2026-09-26T20:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:05:21Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:10:00-04:00", + "returnEnd": "2026-09-26T21:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:14:03Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:10:00-04:00", + "returnEnd": "2026-09-26T21:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:18:09Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:25:00-04:00", + "returnEnd": "2026-09-26T21:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:27:31Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:35:00-04:00", + "returnEnd": "2026-09-26T21:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:34:03Z", + "changed": [ + "queue.STANDBY.waitTime", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:40:00-04:00", + "returnEnd": "2026-09-26T21:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:47:33Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:45:00-04:00", + "returnEnd": "2026-09-26T21:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:53:49Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:50:00-04:00", + "returnEnd": "2026-09-26T21:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:54:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:50:00-04:00", + "returnEnd": "2026-09-26T21:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:56:02Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:50:00-04:00", + "returnEnd": "2026-09-26T21:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T19:58:26Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:45:00-04:00", + "returnEnd": "2026-09-26T21:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:04:20Z", + "changed": [ + "queue.STANDBY.waitTime", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:40:00-04:00", + "returnEnd": "2026-09-26T21:40:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:08:04Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:50:00-04:00", + "returnEnd": "2026-09-26T21:50:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:14:23Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:20:00-04:00", + "returnEnd": "2026-09-26T22:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:18:10Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:25:00-04:00", + "returnEnd": "2026-09-26T22:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:20:05Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:25:00-04:00", + "returnEnd": "2026-09-26T22:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:22:04Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:25:00-04:00", + "returnEnd": "2026-09-26T22:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:23:59Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:35:00-04:00", + "returnEnd": "2026-09-26T22:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:28:22Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:05:00-04:00", + "returnEnd": "2026-09-26T19:05:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:28:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T18:05:00-04:00", + "returnEnd": "2026-09-26T19:05:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:33:18Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:45:00-04:00", + "returnEnd": "2026-09-26T22:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:38:41Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:35:00-04:00", + "returnEnd": "2026-09-26T22:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:42:06Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:35:00-04:00", + "returnEnd": "2026-09-26T22:35:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:43:22Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:55:00-04:00", + "returnEnd": "2026-09-26T22:55:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:49:17Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:45:00-04:00", + "returnEnd": "2026-09-26T22:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:52:45Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:00:00-04:00", + "returnEnd": "2026-09-26T23:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T20:59:12Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:00:00-04:00", + "returnEnd": "2026-09-26T20:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:02:57Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T19:00:00-04:00", + "returnEnd": "2026-09-26T20:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:05:12Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:00:00-04:00", + "returnEnd": "2026-09-26T23:00:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:09:10Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:05:00-04:00", + "returnEnd": "2026-09-26T23:05:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:14:54Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:10:00-04:00", + "returnEnd": "2026-09-26T23:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:16:57Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:10:00-04:00", + "returnEnd": "2026-09-26T23:10:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:18:33Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T21:20:00-04:00", + "returnEnd": "2026-09-26T22:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:24:04Z", + "changed": [ + "queue.STANDBY.waitTime", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:15:00-04:00", + "returnEnd": "2026-09-26T23:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:24:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:15:00-04:00", + "returnEnd": "2026-09-26T23:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:28:59Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:15:00-04:00", + "returnEnd": "2026-09-26T23:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:32:00Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:15:00-04:00", + "returnEnd": "2026-09-26T23:15:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:34:07Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:20:00-04:00", + "returnEnd": "2026-09-26T23:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:43:41Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:25:00-04:00", + "returnEnd": "2026-09-26T23:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:48:32Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:20:00-04:00", + "returnEnd": "2026-09-26T23:20:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:56:16Z", + "changed": [ + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:30:00-04:00", + "returnEnd": "2026-09-26T23:30:00-04:00" + } + } + }, + { + "time": "2026-09-26T21:59:25Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T22:12:06Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T22:24:59Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:45:00-04:00", + "returnEnd": "2026-09-26T21:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T22:25:59Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:45:00-04:00", + "returnEnd": "2026-09-26T21:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T22:26:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:45:00-04:00", + "returnEnd": "2026-09-26T21:45:00-04:00" + } + } + }, + { + "time": "2026-09-26T22:28:06Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T22:33:04Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T22:34:06Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T22:36:06Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T22:43:01Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 65 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T22:53:06Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T22:54:10Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T20:25:00-04:00", + "returnEnd": "2026-09-26T21:25:00-04:00" + } + } + }, + { + "time": "2026-09-26T22:57:46Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 60 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T23:15:00Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 55 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T23:20:06Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T23:30:57Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-26T23:36:09Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T00:02:12Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T00:12:57Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T00:24:05Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T00:44:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 30 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T00:58:02Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T01:06:06Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T01:13:31Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:10:00-04:00", + "returnEnd": "2026-09-26T23:10:00-04:00" + } + } + }, + { + "time": "2026-09-27T01:16:07Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:10:00-04:00", + "returnEnd": "2026-09-26T23:10:00-04:00" + } + } + }, + { + "time": "2026-09-27T01:19:12Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 35 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T01:20:05Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 30 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T01:28:47Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 30 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:05:00-04:00", + "returnEnd": "2026-09-26T23:05:00-04:00" + } + } + }, + { + "time": "2026-09-27T01:32:02Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 25 + }, + "RETURN_TIME": { + "state": "AVAILABLE", + "returnStart": "2026-09-26T22:05:00-04:00", + "returnEnd": "2026-09-26T23:05:00-04:00" + } + } + }, + { + "time": "2026-09-27T01:34:46Z", + "changed": [ + "queue.RETURN_TIME.state", + "queue.RETURN_TIME.returnStart", + "queue.RETURN_TIME.returnEnd" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 25 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T01:42:06Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 20 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T01:58:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 25 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T02:02:01Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 30 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T02:13:01Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 40 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T02:14:05Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T02:14:58Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 50 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T02:31:00Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 45 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T02:36:00Z", + "changed": [ + "queue.STANDBY.waitTime" + ], + "status": "OPERATING", + "queue": { + "STANDBY": { + "waitTime": 30 + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + }, + { + "time": "2026-09-27T03:01:04Z", + "changed": [ + "status", + "queue.STANDBY.waitTime" + ], + "status": "CLOSED", + "queue": { + "STANDBY": { + "waitTime": null + }, + "RETURN_TIME": { + "state": "FINISHED", + "returnStart": null, + "returnEnd": null + } + } + } + ], + "next": null +} \ No newline at end of file diff --git a/tests/unit/test_history.py b/tests/unit/test_history.py index 148701e..e7753da 100644 --- a/tests/unit/test_history.py +++ b/tests/unit/test_history.py @@ -11,7 +11,7 @@ import httpx import pytest -from themeparks import BudgetExhaustedError, RetryConfig, ThemeParks +from themeparks import BudgetExhaustedError, HistorySpan, RetryConfig, ThemeParks from themeparks._errors import RateLimitError @@ -381,6 +381,25 @@ def test_retrievable_through_is_not_the_same_as_recorded_to(self): span = self._span_for(PARK_COVERAGE) assert span.retrievable_through < span.recorded_to + def test_final_through_is_the_earlier_of_filed_and_retrievable(self): + # Days after `recorded_to` are served but can still change, and days + # after `retrievable_through` are not this key's to ask for. The final + # boundary has to respect both, so each fixture keeps them apart in the + # opposite direction. + span = self._span_for(PARK_COVERAGE) + assert span.final_through == span.retrievable_through == _d(2026, 8, 23) + live = span._replace(recorded_to=_d(2026, 9, 26), retrievable_through=_d(2026, 9, 28)) + assert live.final_through == _d(2026, 9, 26) + + def test_final_through_is_unknown_when_nothing_is_filed(self): + assert HistorySpan(None, None, _d(2026, 9, 28)).final_through is None + assert HistorySpan(None, _d(2026, 9, 26), None).final_through is None + + def test_final_through_does_not_change_the_tuple(self): + # A property, not a fourth field: `a, b, c = span` is public usage. + archive_from, recorded_to, retrievable_through = self._span_for(PARK_COVERAGE) + assert len(self._span_for(PARK_COVERAGE)) == 3 + def test_span_feeds_days_directly(self): calls = [] diff --git a/tests/unit/test_history_opening.py b/tests/unit/test_history_opening.py new file mode 100644 index 0000000..8e7a072 --- /dev/null +++ b/tests/unit/test_history_opening.py @@ -0,0 +1,210 @@ +"""`changes()` hands back the `opening` state, so a day can be rebuilt from it. + +Every row of `/history` is the complete live data from its `time` until the next +row. What the rows cannot say is the state BEFORE the first of them: that is the +envelope's `opening` object, and `changes()` used to yield the rows and throw the +envelope away. A caller rebuilding a day then had no status for the minutes +between the start of the range and the first change, which on a night a ride runs +past midnight is real operating time. + +The oracle here is a real capture (see tests/fixtures/README.md): Space Mountain +on 2026-09-26, whose opening is OPERATING because the previous night's extra hours +ran past midnight, and the daily summary the API computed for the same day. +""" + +from __future__ import annotations + +import json +from datetime import datetime, timedelta, timezone +from pathlib import Path + +import httpx +import pytest + +from themeparks import AsyncThemeParks, ThemeParks +from themeparks._ergonomic.history import HistoryChanges +from themeparks._generated.models import HistoryOpening + +FIXTURES = Path(__file__).resolve().parents[1] / "fixtures" +RAW = json.loads((FIXTURES / "space_mountain_history_2026-09-26.json").read_text(encoding="utf-8")) +DAILY = json.loads((FIXTURES / "space_mountain_daily_2026-09-26.json").read_text(encoding="utf-8")) +SPACE_MOUNTAIN = RAW["id"] + + +def _client(payload: dict, seen: list[str] | None = None) -> ThemeParks: + def handler(request: httpx.Request) -> httpx.Response: + if seen is not None: + seen.append(str(request.url)) + return httpx.Response(200, json=payload) + + return ThemeParks(transport=httpx.MockTransport(handler), cache=False) + + +def _async_client(payload: dict) -> AsyncThemeParks: + return AsyncThemeParks( + transport=httpx.MockTransport(lambda request: httpx.Response(200, json=payload)), + cache=False, + ) + + +def _timeline( + opening: HistoryOpening | None, rows: list, day_start: datetime, day_end: datetime +) -> list[tuple[datetime, datetime, str | None]]: + """(from, to, status) segments covering the day, the way a customer rebuilds one. + + Without an opening, the stretch before the first change has no known status, + which is exactly the gap this change closes. + """ + segments = [] + cursor = day_start + status = opening.status if opening is not None else None + for row in rows: + if row.time > cursor: + segments.append((cursor, row.time, status)) + cursor = max(cursor, row.time) + status = row.status + segments.append((cursor, day_end, status)) + return segments + + +def _seconds(segments, wanted: str | None) -> float: + return sum((b - a).total_seconds() for a, b, s in segments if s == wanted) + + +class TestTheCaptureIsWhatThisFileSays: + def test_the_opening_carries_a_run_over_midnight(self) -> None: + # Guard the oracle. If a re-capture picks a day whose opening is CLOSED, + # every test below still passes with and without the fix. + assert RAW["opening"]["status"] == "OPERATING" + assert RAW["history"][0]["status"] == "CLOSED" + assert RAW["history"][0]["time"] > RAW["opening"]["time"] + + +class TestOpeningIsExposed: + def test_the_rows_are_unchanged(self) -> None: + # Backwards compatible: iterating yields exactly what 4.0 yielded. + rows = list(_client(RAW).entity(SPACE_MOUNTAIN).history.changes("2026-09-26")) + assert len(rows) == len(RAW["history"]) + assert all(entity_id == SPACE_MOUNTAIN for entity_id, _ in rows) + assert rows[0][1].status == "CLOSED" + + def test_opening_is_keyed_by_entity_id(self) -> None: + changes = _client(RAW).entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + opening = changes.opening[SPACE_MOUNTAIN] + assert isinstance(opening, HistoryOpening) + assert opening.status == "OPERATING" + assert opening.time == datetime(2026, 9, 26, 4, 0, tzinfo=timezone.utc) + assert opening.queue is not None and opening.queue.STANDBY is not None + assert opening.queue.STANDBY.waitTime == 15 + + def test_opening_before_iterating_costs_one_request_not_two(self) -> None: + seen: list[str] = [] + changes = _client(RAW, seen).entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + assert seen == [], "the request is made when the result is first used, as before" + assert changes.opening[SPACE_MOUNTAIN].status == "OPERATING" + rows = list(changes) + assert len(rows) == len(RAW["history"]) + assert len(seen) == 1 + + def test_it_is_still_an_iterator(self) -> None: + changes = _client(RAW).entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + assert isinstance(changes, HistoryChanges) + assert iter(changes) is changes + first = next(changes) + assert first[1].status == "CLOSED" + + def test_a_park_gives_one_opening_per_entity(self) -> None: + park = { + "id": "park-1", + "name": "Magic Kingdom Park", + "entityType": "PARK", + "timezone": "America/New_York", + "range": {"from": "2026-09-26", "to": "2026-09-26"}, + "entities": [ + { + "id": SPACE_MOUNTAIN, + "name": "Space Mountain", + "entityType": "ATTRACTION", + "coverage": RAW["coverage"], + "opening": RAW["opening"], + "history": RAW["history"][:2], + }, + { + "id": "quiet-ride", + "name": "Nothing Changed Today", + "entityType": "ATTRACTION", + "coverage": {"firstRecordedAt": "2021-07-03"}, + "opening": {"time": "2026-09-26T04:00:00Z", "status": "REFURBISHMENT"}, + "history": [], + }, + ], + "next": None, + } + changes = _client(park).entity("park-1").history.changes("2026-09-26") + assert list(changes.opening) == [SPACE_MOUNTAIN, "quiet-ride"] + # An entity with no rows at all is only knowable through its opening. + assert changes.opening["quiet-ride"].status == "REFURBISHMENT" + assert [entity_id for entity_id, _ in changes] == [SPACE_MOUNTAIN, SPACE_MOUNTAIN] + + def test_an_error_still_surfaces_on_first_use(self) -> None: + tp = _client(RAW) + changes = tp.entity(SPACE_MOUNTAIN).history.changes("2026-09-26", start="2026-09-26") + with pytest.raises(ValueError, match="either date, or from/to"): + _ = changes.opening + + +class TestADayRebuildsFromOpeningPlusChanges: + """What the opening is FOR, checked against the API's own daily row.""" + + DAY_START = datetime(2026, 9, 26, 4, 0, tzinfo=timezone.utc) # 00:00 EDT + DAY_END = DAY_START + timedelta(days=1) + + def _rebuild(self, *, with_opening: bool): + changes = _client(RAW).entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + rows = [row for _, row in changes] + opening = changes.opening[SPACE_MOUNTAIN] if with_opening else None + return _timeline(opening, rows, self.DAY_START, self.DAY_END) + + def test_every_second_of_the_day_has_a_known_status(self) -> None: + segments = self._rebuild(with_opening=True) + assert sum((b - a).total_seconds() for a, b, _ in segments) == 86400 + assert _seconds(segments, None) == 0 + + def test_without_the_opening_the_first_minute_is_unknown(self) -> None: + # The defect, measured: 63 seconds of a ride OPERATING past midnight that + # a rebuild from rows alone cannot place. + assert _seconds(self._rebuild(with_opening=False), None) == 63 + + def test_the_rebuild_agrees_with_the_daily_row(self) -> None: + # An oracle this repo did not write: the API's summary of the same day. + # The first change TO operating and the last close are the daily row's + # firstOperatingAt and lastClosedAt. + segments = self._rebuild(with_opening=True) + daily = DAILY["days"][0] + opened = [a for a, _b, s in segments if s == "OPERATING" and a > self.DAY_START] + closed = [b for _a, b, s in segments if s == "OPERATING" and b < self.DAY_END] + assert opened[0].isoformat().replace("+00:00", "Z") == daily["firstOperatingAt"] + assert closed[-1].isoformat().replace("+00:00", "Z") == daily["lastClosedAt"] + # 63 s carried over midnight, then 11:30:53Z to 03:01:04Z. + assert _seconds(segments, "OPERATING") == 63 + 55811 + + +class TestAsyncOpening: + async def test_opening_after_iterating(self) -> None: + async with _async_client(RAW) as tp: + changes = tp.entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + rows = [pair async for pair in changes] + assert len(rows) == len(RAW["history"]) + assert changes.opening[SPACE_MOUNTAIN].status == "OPERATING" + + async def test_load_fetches_it_up_front(self) -> None: + async with _async_client(RAW) as tp: + changes = await tp.entity(SPACE_MOUNTAIN).history.changes("2026-09-26").load() + assert changes.opening[SPACE_MOUNTAIN].status == "OPERATING" + assert len([pair async for pair in changes]) == len(RAW["history"]) + + async def test_opening_before_the_response_says_how_to_get_it(self) -> None: + async with _async_client(RAW) as tp: + changes = tp.entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + with pytest.raises(RuntimeError, match="load"): + _ = changes.opening diff --git a/themeparks/__init__.py b/themeparks/__init__.py index cb0bd0f..406d430 100644 --- a/themeparks/__init__.py +++ b/themeparks/__init__.py @@ -2,8 +2,10 @@ from themeparks._client import AsyncThemeParks, ThemeParks from themeparks._ergonomic.dates import parse_api_datetime from themeparks._ergonomic.history import ( + AsyncHistoryChanges, BudgetExhaustedError, EntityRef, + HistoryChanges, HistoryPage, HistorySpan, ) @@ -20,9 +22,11 @@ __all__ = [ "APIError", + "AsyncHistoryChanges", "AsyncThemeParks", "BudgetExhaustedError", "EntityRef", + "HistoryChanges", "HistoryPage", "HistorySpan", "RateLimit", diff --git a/themeparks/_ergonomic/history.py b/themeparks/_ergonomic/history.py index 8edecb0..f9dbec8 100644 --- a/themeparks/_ergonomic/history.py +++ b/themeparks/_ergonomic/history.py @@ -20,7 +20,7 @@ from __future__ import annotations -from collections.abc import AsyncIterator, Callable, Iterator +from collections.abc import AsyncIterator, Awaitable, Callable, Iterator from datetime import date as _date from typing import Any, NamedTuple, Union @@ -30,6 +30,7 @@ HistoryDailyEnvelope, HistoryDailyRow, HistoryEnvelope, + HistoryOpening, HistoryParkCoverageDocument, HistoryParkDailyEnvelope, HistoryParkRawEnvelope, @@ -67,6 +68,24 @@ class HistorySpan(NamedTuple): recorded_to: _date | None retrievable_through: _date | None + @property + def final_through(self) -> _date | None: + """The newest day whose daily rows will not change again, or None. + + `retrievable_through` is usually today, and today's row is the day so + far. Recent days can still change after that: a run that crosses + midnight is reported on the day it started, and the archive records + days 2 to 3 behind live data. `recorded_to` is the newest day the + archive holds, so a day on or before it is final. Store those; ask for + anything later again once `final_through` has moved past it. + + The earlier of the two dates, because a key may be entitled to fewer + days than the archive holds. None when either is unknown. + """ + if self.recorded_to is None or self.retrievable_through is None: + return None + return min(self.recorded_to, self.retrievable_through) + def _span(document: CoverageDocument) -> HistorySpan: summary = getattr(document, "summary", None) @@ -194,6 +213,14 @@ def _daily_rows(envelope: DailyEnvelope) -> Iterator[tuple[str, HistoryDailyRow] yield (ref.id, row) +def _openings(envelope: RawEnvelope) -> dict[str, HistoryOpening]: + """Each entity's `opening`, keyed by id, in the order the response lists them.""" + entities = getattr(envelope, "entities", None) + if entities is not None: + return {entity.id: entity.opening for entity in entities} + return {envelope.id: envelope.opening} # type: ignore[union-attr] + + def _raw_rows(envelope: RawEnvelope) -> Iterator[tuple[str, HistoryRow]]: entities = getattr(envelope, "entities", None) if entities is not None: @@ -205,6 +232,94 @@ def _raw_rows(envelope: RawEnvelope) -> Iterator[tuple[str, HistoryRow]]: yield (envelope.id, row) +class HistoryChanges(Iterator[tuple[str, HistoryRow]]): + """What :meth:`HistoryApi.changes` returns: the rows, and the state before them. + + Iterate it exactly as before, for `(entity id, row)` pairs. Each row is the + entity's complete live data from its `time` until the next row's. + + `opening` is the one thing the rows cannot tell you: the state in force at + the START of the range, before the first change. Without it the stretch + between midnight and an entity's first change has no known status -- on a + night a ride runs past midnight that is real operating time, and a day + rebuilt from the rows alone disagrees with the daily summary. It is a dict + of :class:`HistoryOpening` keyed by entity id, with an entry for every + entity in the response, including one that did not change at all that day. + + Nothing is requested until the result is first used, as when this was a + plain generator, and reading `opening` before or after iterating costs the + same single request. + """ + + def __init__(self, fetch: Callable[[], RawEnvelope]) -> None: + self._fetch = fetch + self._envelope: RawEnvelope | None = None + self._rows: Iterator[tuple[str, HistoryRow]] | None = None + + def _loaded(self) -> RawEnvelope: + if self._envelope is None: + self._envelope = self._fetch() + return self._envelope + + @property + def opening(self) -> dict[str, HistoryOpening]: + """The state at the start of the range, per entity id.""" + return _openings(self._loaded()) + + def __iter__(self) -> HistoryChanges: + return self + + def __next__(self) -> tuple[str, HistoryRow]: + if self._rows is None: + self._rows = _raw_rows(self._loaded()) + return next(self._rows) + + +class AsyncHistoryChanges(AsyncIterator[tuple[str, HistoryRow]]): + """What :meth:`AsyncHistoryApi.changes` returns. See :class:`HistoryChanges`. + + Iterate it with `async for`. A property cannot await, so `opening` is + readable once the response has arrived: after the first step of iteration, + or straight away with `changes = await history.changes(day).load()`. + """ + + def __init__(self, fetch: Callable[[], Awaitable[RawEnvelope]]) -> None: + self._fetch = fetch + self._envelope: RawEnvelope | None = None + self._rows: Iterator[tuple[str, HistoryRow]] | None = None + + async def _loaded(self) -> RawEnvelope: + if self._envelope is None: + self._envelope = await self._fetch() + return self._envelope + + async def load(self) -> AsyncHistoryChanges: + """Make the request now, if it has not been made, and return this object.""" + await self._loaded() + return self + + @property + def opening(self) -> dict[str, HistoryOpening]: + """The state at the start of the range, per entity id.""" + if self._envelope is None: + raise RuntimeError( + "the response has not arrived yet: iterate first, or " + "`await changes.load()` before reading `opening`" + ) + return _openings(self._envelope) + + def __aiter__(self) -> AsyncHistoryChanges: + return self + + async def __anext__(self) -> tuple[str, HistoryRow]: + if self._rows is None: + self._rows = _raw_rows(await self._loaded()) + try: + return next(self._rows) + except StopIteration: + raise StopAsyncIteration from None + + class HistoryApi: """History for one entity id, reached as ``tp.entity(id).history``.""" @@ -307,21 +422,28 @@ def changes( start: str | _date | None = None, end: str | _date | None = None, max_wait: float = DEFAULT_MAX_WAIT_SECONDS, - ) -> Iterator[tuple[str, HistoryRow]]: - """Every recorded change, as (entity id, row). + ) -> HistoryChanges: + """Every recorded change, as (entity id, row), plus the state before them. A single day for a park, or up to 31 days for one entity. The caller does not have to know which cap applies: ask for what you want and the API answers or tells you the range is too long. + + Iterate the result for the rows. Its `opening` is each entity's state at + the start of the range, which is what a day has to be rebuilt from -- + see :class:`HistoryChanges`. """ - try: - envelope = self._raw.get_entity_history( - self._id, date=_as_day(date), start=_as_day(start), end=_as_day(end) - ) - except RateLimitError as exc: - _reraise_if_too_long(exc, max_wait) - raise - yield from _raw_rows(envelope) + + def fetch() -> RawEnvelope: + try: + return self._raw.get_entity_history( + self._id, date=_as_day(date), start=_as_day(start), end=_as_day(end) + ) + except RateLimitError as exc: + _reraise_if_too_long(exc, max_wait) + raise + + return HistoryChanges(fetch) class AsyncHistoryApi: @@ -367,20 +489,27 @@ async def days( _reraise_if_too_long(exc, max_wait) raise - async def changes( + def changes( self, date: str | _date | None = None, *, start: str | _date | None = None, end: str | _date | None = None, max_wait: float = DEFAULT_MAX_WAIT_SECONDS, - ) -> AsyncIterator[tuple[str, HistoryRow]]: - try: - envelope = await self._raw.get_entity_history( - self._id, date=_as_day(date), start=_as_day(start), end=_as_day(end) - ) - except RateLimitError as exc: - _reraise_if_too_long(exc, max_wait) - raise - for pair in _raw_rows(envelope): - yield pair + ) -> AsyncHistoryChanges: + """Asynchronous mirror of :meth:`HistoryApi.changes`. + + Not a coroutine, as before: `async for pair in history.changes(day)` + works unchanged. See :class:`AsyncHistoryChanges` for `opening`. + """ + + async def fetch() -> RawEnvelope: + try: + return await self._raw.get_entity_history( + self._id, date=_as_day(date), start=_as_day(start), end=_as_day(end) + ) + except RateLimitError as exc: + _reraise_if_too_long(exc, max_wait) + raise + + return AsyncHistoryChanges(fetch) From 75a72178458d5e28a2692386b8036224a64c1376 Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Mon, 28 Sep 2026 21:29:18 +0100 Subject: [PATCH 2/9] fix(backfill): --since/--until, reruns add new days, only final days written Three defects from a customer-style run of 4.0.1, each reproduced first: - No date range. A key that reaches the whole archive downloaded all of it every time. `--since` and `--until` take a real calendar day as YYYY-MM-DD, both inclusive; since after until is refused before any request. - A finished park was finished forever. A rerun printed "already complete" and exited 0 without fetching a new day, so a nightly cron never updated. A finished file is now carried forward from the day after its last one. - The newest rows were partial. A run ended on retrievableThrough, usually today, whose row is the day so far, and the archive records days 2 to 3 behind live data. A run now ends at span().final_through, says so, and the next run adds the held-back days once final. Every row written is final. State files move to version 2 (`end` is now the newest final day). A version 1 file from this SDK is upgraded rather than refused: rows dated within seven days of its old end are removed and fetched again, since which of them were final was never recorded. Kept rows are left byte for byte; a file whose newest row is older than the cut is not rewritten. Fixed on the way, each with a test and a committed mutant: - a finished state file written to another column layout fell through to a fresh start in append mode, writing the archive a second time under a second header, exit 0. Refused now, as an unfinished one already was. - a rerun interrupted before its first page recorded no resume point, and the next run would have started from the top of the archive and appended it again. The run's own start is recorded. - a state file whose data file was deleted was continued, producing a file that starts part-way through its range. The park is fetched again instead. A `--since` inside the existing file is accepted (a fixed or rolling cron line works); one before the file's first day, or one that would leave a gap, is refused with the way out rather than ignored. Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/mutation/mutants.json | 74 ++- tests/unit/test_backfill.py | 4 +- tests/unit/test_backfill_incremental.py | 707 ++++++++++++++++++++++++ themeparks/backfill.py | 482 ++++++++++++++-- 4 files changed, 1206 insertions(+), 61 deletions(-) create mode 100644 tests/unit/test_backfill_incremental.py diff --git a/tests/mutation/mutants.json b/tests/mutation/mutants.json index f122b4f..5ace545 100644 --- a/tests/mutation/mutants.json +++ b/tests/mutation/mutants.json @@ -29,8 +29,8 @@ { "name": "resume ignores the recorded boundary", "file": "themeparks/backfill.py", - "find": "resume_at = (state.get(\"resumeFrom\") or state.get(\"lastDay\")) if resuming else None", - "replace": "resume_at = state.get(\"lastDay\") if resuming else None", + "find": "resume_at = state.get(\"resumeFrom\") or state.get(\"lastDay\")", + "replace": "resume_at = state.get(\"lastDay\")", "why": "The fallback is for state files with no boundary. Preferring it always reintroduces the duplicate." }, { @@ -151,6 +151,76 @@ "find": " model_config = ConfigDict(extra=\"allow\")", "replace": " model_config = ConfigDict(extra=\"ignore\")", "why": "A stale vendored spec then deletes real data at parse time, which is how unknownMinutes, inParkHours and extremeWaits were lost." + }, + { + "name": "a run ends at retrievable_through again", + "file": "themeparks/backfill.py", + "find": "end: Day = span.final_through", + "replace": "end: Day = span.retrievable_through", + "why": "4.0.1 ended every run on today, so the newest rows in every file were partial days, and nothing ever replaced them." + }, + { + "name": "final_through ignores recorded_to", + "file": "themeparks/_ergonomic/history.py", + "find": "return min(self.recorded_to, self.retrievable_through)", + "replace": "return self.retrievable_through", + "why": "The same defect one layer down: the final boundary becomes today again for every caller of span()." + }, + { + "name": "a finished park is left alone again", + "file": "themeparks/backfill.py", + "find": "if end is None or continue_at > str(end):", + "replace": "if True:", + "why": "4.0.1's behaviour: a rerun of a finished park printed that it was complete and exited 0, so a nightly cron never fetched another day." + }, + { + "name": "a rerun interrupted before its first page forgets where it began", + "file": "themeparks/backfill.py", + "find": " return self.first_day\n", + "replace": " return None\n", + "why": "With no page boundary and no row, the state recorded nothing, and the next run fell back to the top of the archive and appended a second copy of it." + }, + { + "name": "a finished state written to another contract is appended to", + "file": "themeparks/backfill.py", + "find": " if mismatch is not None:\n", + "replace": " if mismatch is not None and not state.get(\"complete\"):\n", + "why": "Only unfinished mismatched states were refused; a finished one fell through to a fresh start in append mode and doubled the file under a second header." + }, + { + "name": "a state file with no data file is continued", + "file": "themeparks/backfill.py", + "find": " if state and not file_exists:\n state = {}\n", + "replace": "", + "why": "Continuing writes a file that starts part-way through its range and records it complete." + }, + { + "name": "a 4.0 file keeps its partial tail", + "file": "themeparks/backfill.py", + "find": " _trim_after(out_path, fmt, keep_through)\n", + "replace": " pass\n", + "why": "The partial days 4.0 wrote stay in the file for good, next to nothing that would ever correct them." + }, + { + "name": "--since is ignored on a first run", + "file": "themeparks/backfill.py", + "find": "start = _later(rng.since, ask.archive_from) if rng.since is not None else ask.archive_from", + "replace": "start = ask.archive_from", + "why": "The customer asked for a year and got the whole archive, which is the defect --since exists to fix." + }, + { + "name": "a --since that would leave a gap is accepted", + "file": "themeparks/backfill.py", + "find": "if since is not None and str(since) > continue_at:", + "replace": "if False:", + "why": "The file would skip days the state file then claims it holds." + }, + { + "name": "changes() loses a park's openings", + "file": "themeparks/_ergonomic/history.py", + "find": "return {entity.id: entity.opening for entity in entities}", + "replace": "return {}", + "why": "The opening is the state before the first change; without it a day cannot be rebuilt from raw history." } ] } diff --git a/tests/unit/test_backfill.py b/tests/unit/test_backfill.py index 733c7af..26d5791 100644 --- a/tests/unit/test_backfill.py +++ b/tests/unit/test_backfill.py @@ -822,7 +822,7 @@ def test_every_park_is_tried_and_the_failures_are_named( ) -> None: attempted: list[str] = [] - def fake_backfill(tp, park, out_dir, fmt, overwrite=False): + def fake_backfill(tp, park, out_dir, fmt, overwrite=False, **_kw): attempted.append(park.id) if park.id == "p2": raise APIError("500 Server Error", status=500, body={}, url="u") @@ -842,7 +842,7 @@ def test_a_spent_budget_stops_the_whole_run(self, tmp_path: Path, monkeypatch) - # where it got to. attempted: list[str] = [] - def fake_backfill(tp, park, out_dir, fmt, overwrite=False): + def fake_backfill(tp, park, out_dir, fmt, overwrite=False, **_kw): attempted.append(park.id) return backfill.EX_TEMPFAIL diff --git a/tests/unit/test_backfill_incremental.py b/tests/unit/test_backfill_incremental.py new file mode 100644 index 0000000..4f82d7f --- /dev/null +++ b/tests/unit/test_backfill_incremental.py @@ -0,0 +1,707 @@ +"""`themeparks-backfill` over time: a date range, reruns that add new days, and +only final days in the file. + +Three defects from one customer-style run of 4.0.1, each reproduced before it +was fixed: + +1. No way to ask for less than everything. A key that reaches the whole archive + downloaded all of it, every time, with no `--since`. +2. A finished park was finished forever. A rerun printed "already complete" and + exited 0 without asking for a single new day, so a nightly cron looked + healthy and never updated. The only way to get yesterday was `--overwrite`, + which downloads the whole archive again. +3. The run ended at `retrievableThrough`, which is today. Today's row is the day + so far, and the archive records days 2 to 3 behind, so the newest rows in the + file were partial -- Magic Kingdom's last day summed to about half the + operating minutes of a full one -- and, because of (2), they were never + corrected. + +The stub below serves a park the way the API does: one row per entity per day, +31 days a page, final values through `recorded_to` and partial values after it. +A partial row carries half the operating minutes, so a test can tell from the +file alone whether a non-final day was ever written. +""" + +from __future__ import annotations + +import csv +import inspect +import io +import json +from collections import Counter +from datetime import date, timedelta +from pathlib import Path + +import pytest + +from themeparks import APIError, BudgetExhaustedError, backfill +from themeparks._client import PACKAGE_VERSION +from themeparks._ergonomic.history import EntityRef, HistoryApi, HistoryPage, HistorySpan +from themeparks._generated.models import HistoryDailyRow +from themeparks.backfill import _Park, _Range + +FINAL_MINUTES = 600 +PARTIAL_MINUTES = 300 +PAGE_DAYS = 31 + + +def _day(value: object) -> date: + return value if isinstance(value, date) else date.fromisoformat(str(value)) + + +def _row(day: date, *, final: bool) -> HistoryDailyRow: + return HistoryDailyRow.model_validate( + { + "date": day.isoformat(), + "firstOperatingAt": f"{day.isoformat()}T13:00:00Z", + "lastClosedAt": f"{day.isoformat()}T23:00:00Z" if final else None, + "operatingMinutes": FINAL_MINUTES if final else PARTIAL_MINUTES, + "downMinutes": 0, + "changes": 3, + } + ) + + +class _Archive: + """A park's daily history, served as the API serves it. + + `through` is retrievableThrough (usually today), `recorded_to` the newest day + the archive holds. Days after `recorded_to` are served with partial values, + which is what today's row and the days still being recorded look like. + """ + + def __init__( + self, + archive_from: str = "2026-06-01", + recorded_to: str = "2026-09-26", + through: str = "2026-09-28", + floor: str | None = None, + entities: tuple[str, ...] = ("ent-a", "ent-b"), + ) -> None: + self.archive_from = archive_from + self.recorded_to = recorded_to + self.through = through + self.floor = floor + self.entities = entities + self.calls: list[tuple[str, str]] = [] + #: Raise BudgetExhaustedError on the Nth page request (1-based), or never. + self.budget_on_page: int | None = None + self._pages_served = 0 + + def advance(self, days: int) -> None: + """Time passes: the archive and today both move on.""" + self.recorded_to = (_day(self.recorded_to) + timedelta(days=days)).isoformat() + self.through = (_day(self.through) + timedelta(days=days)).isoformat() + + def span(self) -> HistorySpan: + return HistorySpan(_day(self.archive_from), _day(self.recorded_to), _day(self.through)) + + def days_with_entities(self, start=None, end=None, *, max_wait=120.0, on_page=None): + self.calls.append((str(start), str(end))) + if self.floor is not None and str(start) < self.floor: + raise backfill_test_403(self.floor) + first = _day(start) + last = min(_day(end), _day(self.through)) + page_start = first + while page_start <= last: + self._pages_served += 1 + if self.budget_on_page == self._pages_served: + raise BudgetExhaustedError("429", status=429, body={}, url="u", retry_after=2700.0) + page_end = min(page_start + timedelta(days=PAGE_DAYS - 1), last) + for entity in self.entities: + ref = EntityRef(entity, f"Ride {entity}", "ATTRACTION") + day = page_start + while day <= page_end: + yield ref, _row(day, final=day <= _day(self.recorded_to)) + day += timedelta(days=1) + following = page_end + timedelta(days=1) + has_next = following <= last + if on_page is not None: + nxt = ( + f"https://api.themeparks.wiki/v1/entity/p/history/daily" + f"?from={following.isoformat()}&to={last.isoformat()}" + if has_next + else None + ) + on_page(HistoryPage(page_start.isoformat(), page_end.isoformat(), nxt)) + page_start = following + + +def backfill_test_403(earliest: str) -> APIError: + return APIError( + f"403 Forbidden: This key can see history back to {earliest}.", + status=403, + body={ + "error": { + "type": "HISTORY_WINDOW_EXCEEDED", + "message": f"This key can see history back to {earliest} (400 days).", + "earliestAllowedDate": earliest, + } + }, + url="https://api.themeparks.wiki/v1/entity/p/history/daily", + ) + + +class _Client: + def __init__(self, archive: _Archive) -> None: + self._archive = archive + + def entity(self, _id: str): + archive = self._archive + + class _Entity: + history = archive + + return _Entity() + + +def _run(archive: _Archive, tmp_path: Path, fmt: str = "ndjson", **kw) -> int: + return backfill.backfill_park(_Client(archive), _Park("p", "Park"), tmp_path, fmt, **kw) + + +def _rows(tmp_path: Path, fmt: str = "ndjson") -> list[dict]: + path = tmp_path / f"p.{fmt}" + if fmt == "csv": + with path.open(encoding="utf-8-sig", newline="") as handle: + return list(csv.DictReader(handle)) + return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()] + + +def _state(tmp_path: Path, fmt: str = "ndjson") -> dict: + return json.loads(backfill.state_path_for(tmp_path, "p", fmt).read_text(encoding="utf-8")) + + +def _keys(rows: list[dict]) -> Counter: + return Counter((r["entityId"], r["date"]) for r in rows) + + +def _assert_every_row_final_and_unique(rows: list[dict]) -> None: + keys = _keys(rows) + dupes = [k for k, n in keys.items() if n > 1] + assert dupes == [], f"(entityId, date) written more than once: {dupes[:5]}" + partial = [r["date"] for r in rows if int(r["operatingMinutes"]) != FINAL_MINUTES] + assert partial == [], f"non-final days reached the file: {sorted(set(partial))}" + + +class TestTheStubIsTheApi: + def test_the_stub_takes_what_the_real_method_takes(self) -> None: + real = inspect.signature(HistoryApi.days_with_entities) + stub = inspect.signature(_Archive.days_with_entities) + assert list(real.parameters) == list(stub.parameters) + + def test_the_stub_serves_partial_days_past_recorded_to(self) -> None: + # Guard the oracle: if the stub never served a partial row, "no partial + # row in the file" would pass whether or not the command stops early. + archive = _Archive() + served = list(archive.days_with_entities("2026-09-25", "2026-09-28")) + minutes = {row.date.isoformat(): row.operatingMinutes for _, row in served} + assert minutes["2026-09-26"] == FINAL_MINUTES + assert minutes["2026-09-27"] == PARTIAL_MINUTES + assert minutes["2026-09-28"] == PARTIAL_MINUTES + + +class TestOnlyFinalDaysAreWritten: + """Defect 3: the newest rows were partial, and then frozen.""" + + def test_the_run_stops_at_the_newest_final_day(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + assert _run(archive, tmp_path) == 0 + assert archive.calls == [("2026-06-01", "2026-09-26")] + rows = _rows(tmp_path) + assert max(r["date"] for r in rows) == "2026-09-26" + _assert_every_row_final_and_unique(rows) + err = capsys.readouterr().err + assert "stopping at 2026-09-26" in err + assert _state(tmp_path)["end"] == "2026-09-26" + + def test_the_next_run_adds_those_days_once_they_are_final(self, tmp_path: Path) -> None: + # The other half: the days held back are not lost, they arrive on the + # next run with their final values, once. + archive = _Archive() + _run(archive, tmp_path) + archive.advance(2) + assert _run(archive, tmp_path) == 0 + assert archive.calls[-1] == ("2026-09-27", "2026-09-28") + rows = _rows(tmp_path) + assert max(r["date"] for r in rows) == "2026-09-28" + _assert_every_row_final_and_unique(rows) + + def test_nothing_final_yet_writes_nothing(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + + def span() -> HistorySpan: + return HistorySpan(None, None, date(2026, 9, 28)) + + archive.span = span # type: ignore[method-assign] + assert _run(archive, tmp_path) == 0 + assert archive.calls == [] + assert not (tmp_path / "p.ndjson").exists() + assert "nothing final" in capsys.readouterr().err + + def test_no_message_when_the_archive_is_already_final(self, tmp_path: Path, capsys) -> None: + # A park that stopped reporting: everything it has is final, and saying + # "stopping early" would be noise. + archive = _Archive(recorded_to="2026-08-31", through="2026-08-31") + _run(archive, tmp_path) + assert "stopping at" not in capsys.readouterr().err + + +class TestARerunAddsNewDays: + """Defect 2: a finished park never fetched another day.""" + + def test_a_rerun_with_new_final_days_fetches_only_those(self, tmp_path: Path) -> None: + archive = _Archive() + _run(archive, tmp_path) + before = _rows(tmp_path) + archive.advance(1) + assert _run(archive, tmp_path) == 0 + assert archive.calls[-1] == ("2026-09-27", "2026-09-27") + after = _rows(tmp_path) + assert after[: len(before)] == before, "the rows already there changed" + assert len(after) == len(before) + 2 # two entities, one new day + _assert_every_row_final_and_unique(after) + + def test_the_state_keeps_the_original_start_and_moves_the_end(self, tmp_path: Path) -> None: + archive = _Archive() + _run(archive, tmp_path) + archive.advance(3) + _run(archive, tmp_path) + state = _state(tmp_path) + assert state["start"] == "2026-06-01" + assert state["end"] == "2026-09-29" + assert state["complete"] is True + + def test_a_rerun_with_nothing_new_asks_for_nothing(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + _run(archive, tmp_path) + capsys.readouterr() + before = (tmp_path / "p.ndjson").read_bytes() + assert _run(archive, tmp_path) == 0 + assert len(archive.calls) == 1, "asked the API again with nothing to ask for" + assert (tmp_path / "p.ndjson").read_bytes() == before + assert "up to date" in capsys.readouterr().err + + def test_a_nightly_cron_for_a_week(self, tmp_path: Path) -> None: + # The use case exit code 75 is designed for, end to end. + archive = _Archive() + for _ in range(7): + assert _run(archive, tmp_path) == 0 + archive.advance(1) + rows = _rows(tmp_path) + assert max(r["date"] for r in rows) == "2026-10-02" + _assert_every_row_final_and_unique(rows) + days = sorted({r["date"] for r in rows}) + expected = (date(2026, 10, 2) - date(2026, 6, 1)).days + 1 + assert len(days) == expected, "a gap or an overlap between runs" + + def test_a_csv_rerun_gains_no_second_header_or_bom(self, tmp_path: Path) -> None: + archive = _Archive() + _run(archive, tmp_path, "csv") + archive.advance(2) + _run(archive, tmp_path, "csv") + raw = (tmp_path / "p.csv").read_bytes() + assert raw.count(b"\xef\xbb\xbf") == 1 + assert raw.decode("utf-8-sig").count("parkId,") == 1 + _assert_every_row_final_and_unique(_rows(tmp_path, "csv")) + + def test_an_interrupted_rerun_continues_from_its_own_start(self, tmp_path: Path) -> None: + # The budget runs out on the rerun's FIRST request, before any page. With + # no page boundary and no row, the only thing saying where this run began + # is the state file; without it the next run started over from the top + # of the archive and appended a second copy of everything. + archive = _Archive() + _run(archive, tmp_path) + rows_before = len(_rows(tmp_path)) + archive.advance(40) + archive.budget_on_page = archive._pages_served + 1 + assert _run(archive, tmp_path) == backfill.EX_TEMPFAIL + assert len(_rows(tmp_path)) == rows_before, "the file changed on a failed run" + assert _state(tmp_path)["resumeFrom"] == "2026-09-27" + + archive.budget_on_page = None + assert _run(archive, tmp_path) == 0 + assert archive.calls[-1][0] == "2026-09-27" + _assert_every_row_final_and_unique(_rows(tmp_path)) + + def test_an_interrupted_rerun_mid_way_resumes_at_the_page_boundary( + self, tmp_path: Path + ) -> None: + archive = _Archive() + _run(archive, tmp_path) + archive.advance(40) # two pages of new days + archive.budget_on_page = archive._pages_served + 2 + assert _run(archive, tmp_path) == backfill.EX_TEMPFAIL + assert _state(tmp_path)["resumeFrom"] == "2026-10-28" + archive.budget_on_page = None + assert _run(archive, tmp_path) == 0 + assert archive.calls[-1][0] == "2026-10-28" + _assert_every_row_final_and_unique(_rows(tmp_path)) + + +class TestSinceAndUntil: + """Defect 1: no way to ask for less than everything.""" + + def test_since_is_where_the_first_request_starts(self, tmp_path: Path) -> None: + archive = _Archive() + assert _run(archive, tmp_path, window=_Range(since="2026-09-01")) == 0 + assert archive.calls == [("2026-09-01", "2026-09-26")] + assert min(r["date"] for r in _rows(tmp_path)) == "2026-09-01" + assert _state(tmp_path)["start"] == "2026-09-01" + + def test_until_is_where_it_ends(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + _run(archive, tmp_path, window=_Range(since="2026-07-01", until="2026-07-31")) + assert archive.calls == [("2026-07-01", "2026-07-31")] + # --until is a choice, not a day held back for being unfinished. + assert "stopping at" not in capsys.readouterr().err + + def test_until_past_the_final_day_still_stops_at_the_final_day(self, tmp_path: Path) -> None: + archive = _Archive() + _run(archive, tmp_path, window=_Range(until="2026-12-31")) + assert archive.calls == [("2026-06-01", "2026-09-26")] + _assert_every_row_final_and_unique(_rows(tmp_path)) + + def test_since_before_the_archive_starts_at_the_archive(self, tmp_path: Path) -> None: + archive = _Archive() + _run(archive, tmp_path, window=_Range(since="2019-01-01")) + assert archive.calls[0][0] == "2026-06-01" + + def test_since_before_the_plan_floor_starts_at_the_floor(self, tmp_path: Path, capsys) -> None: + archive = _Archive(floor="2026-08-01") + assert _run(archive, tmp_path, window=_Range(since="2026-07-01")) == 0 + assert [c[0] for c in archive.calls] == ["2026-07-01", "2026-08-01"] + assert "reaches back to 2026-08-01" in capsys.readouterr().err + + def test_since_after_the_newest_final_day_writes_nothing(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + assert _run(archive, tmp_path, window=_Range(since="2026-09-27")) == 0 + assert archive.calls == [] + assert not (tmp_path / "p.ndjson").exists() + assert "2026-09-26" in capsys.readouterr().err + + def test_the_same_since_on_every_run_continues_the_file(self, tmp_path: Path) -> None: + # A cron line with a fixed --since, run nightly. + archive = _Archive() + _run(archive, tmp_path, window=_Range(since="2026-09-01")) + archive.advance(1) + assert _run(archive, tmp_path, window=_Range(since="2026-09-01")) == 0 + assert archive.calls[-1] == ("2026-09-27", "2026-09-27") + _assert_every_row_final_and_unique(_rows(tmp_path)) + + def test_the_same_since_before_the_archive_is_not_a_change(self, tmp_path: Path) -> None: + # --since 2019 asked for the archive's start; asking again is the same run. + archive = _Archive() + _run(archive, tmp_path, window=_Range(since="2019-01-01")) + archive.advance(1) + assert _run(archive, tmp_path, window=_Range(since="2019-01-01")) == 0 + + def test_a_rolling_since_continues_the_file(self, tmp_path: Path) -> None: + # `--since $(date -d '-30 days' +%F)` in a cron: later every night, and + # always inside the file, so the file just carries on. + archive = _Archive() + _run(archive, tmp_path, window=_Range(since="2026-08-27")) + archive.advance(1) + assert _run(archive, tmp_path, window=_Range(since="2026-08-28")) == 0 + assert archive.calls[-1] == ("2026-09-27", "2026-09-27") + + def test_an_earlier_since_than_the_file_is_refused(self, tmp_path: Path, capsys) -> None: + # Appending older days after newer ones cannot make the file start + # earlier without rewriting it, and pretending otherwise would record a + # range the file does not hold. + archive = _Archive() + _run(archive, tmp_path, window=_Range(since="2026-09-01")) + before = (tmp_path / "p.ndjson").read_bytes() + assert _run(archive, tmp_path, window=_Range(since="2026-08-01")) == 1 + assert (tmp_path / "p.ndjson").read_bytes() == before + err = capsys.readouterr().err + assert "2026-09-01" in err and "--overwrite" in err + + def test_a_since_that_would_leave_a_gap_is_refused(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + _run(archive, tmp_path, window=_Range(until="2026-07-31")) + assert _run(archive, tmp_path, window=_Range(since="2026-09-01")) == 1 + assert "gap" in capsys.readouterr().err + + def test_an_until_before_the_file_is_refused(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + _run(archive, tmp_path, window=_Range(since="2026-09-01")) + assert _run(archive, tmp_path, window=_Range(until="2026-08-15")) == 1 + assert "--overwrite" in capsys.readouterr().err + + def test_until_then_no_until_extends_the_file(self, tmp_path: Path) -> None: + archive = _Archive() + _run(archive, tmp_path, window=_Range(until="2026-07-31")) + assert _run(archive, tmp_path) == 0 + assert archive.calls[-1] == ("2026-08-01", "2026-09-26") + _assert_every_row_final_and_unique(_rows(tmp_path)) + + def test_an_until_already_covered_is_up_to_date(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + _run(archive, tmp_path) + assert _run(archive, tmp_path, window=_Range(until="2026-08-01")) == 0 + assert len(archive.calls) == 1 + assert "up to date" in capsys.readouterr().err + + +class TestTheCommandLine: + def _main(self, monkeypatch, argv: list[str]) -> list[_Range]: + seen: list[_Range] = [] + + def fake_run_all(tp, targets, args): + seen.append(_Range(args.since, args.until)) + return 0 + + monkeypatch.setattr(backfill, "_catalogue", lambda tp: [("p", "Park", "d", "Dest")]) + monkeypatch.setattr(backfill, "_run_all", fake_run_all) + monkeypatch.setattr(backfill, "ThemeParks", lambda **kw: _NullClient()) + assert backfill.main([*argv, "--api-key", "tpw_test"]) == 0 + return seen + + def test_since_and_until_reach_the_run(self, tmp_path: Path, monkeypatch) -> None: + seen = self._main( + monkeypatch, + ["p", "--since", "2025-01-01", "--until", "2025-12-31", "--out", str(tmp_path)], + ) + assert seen == [_Range("2025-01-01", "2025-12-31")] + + def test_neither_is_required(self, tmp_path: Path, monkeypatch) -> None: + assert self._main(monkeypatch, ["p", "--out", str(tmp_path)]) == [_Range(None, None)] + + @pytest.mark.parametrize( + "bad", ["2025-13-01", "2025-02-30", "2025-1-1", "20250101", "yesterday", ""] + ) + def test_a_day_that_is_not_yyyy_mm_dd_is_refused(self, bad: str, capsys) -> None: + with pytest.raises(SystemExit) as caught: + backfill.main(["p", "--since", bad]) + assert caught.value.code == 2 + assert "YYYY-MM-DD" in capsys.readouterr().err + + def test_since_after_until_is_refused(self, capsys) -> None: + with pytest.raises(SystemExit) as caught: + backfill.main(["p", "--since", "2025-06-01", "--until", "2025-05-31"]) + assert caught.value.code == 2 + assert "--since" in capsys.readouterr().err + + def test_help_documents_both_and_the_final_day_rule(self, capsys) -> None: + with pytest.raises(SystemExit): + backfill.main(["--help"]) + out = capsys.readouterr().out + assert "--since" in out and "--until" in out + assert "final" in out + assert "again" in out + + def test_run_all_hands_the_range_to_every_park(self, tmp_path: Path, monkeypatch) -> None: + got: list[object] = [] + + def fake_backfill(tp, park, out_dir, fmt, overwrite=False, **kw): + got.append(kw["window"]) + return 0 + + class Args: + out = tmp_path + format = "ndjson" + overwrite = False + since = "2025-01-01" + until = None + + monkeypatch.setattr(backfill, "backfill_park", fake_backfill) + backfill._run_all(None, [("p1", "One"), ("p2", "Two")], Args()) + assert got == [_Range("2025-01-01", None)] * 2 + + +class _NullClient: + def __enter__(self): + return self + + def __exit__(self, *_exc): + return False + + +# -------------------------------------------------------------------------- +# Files written by 4.0.x: their newest days may be partial. +# -------------------------------------------------------------------------- + + +def _v1_state(tmp_path: Path, fmt: str, **fields: object) -> None: + """A state file with the fields and values 4.0.1 writes for a finished park.""" + state: dict[str, object] = { + "columns": backfill._columns_fingerprint(fmt), + "complete": True, + "end": "2026-09-28", + "format": fmt, + "lastDay": "2026-09-28", + "resumeFrom": None, + "sdk": "py", + "sdkVersion": "4.0.1", + "start": "2026-06-01", + "stateVersion": 1, + } + state.update(fields) + backfill.state_path_for(tmp_path, "p", fmt).write_text( + json.dumps(state, sort_keys=True) + "\n", encoding="utf-8" + ) + + +def _write_as_4_0(tmp_path: Path, fmt: str, archive: _Archive, names: dict[str, str]) -> None: + """The file 4.0.1 left behind: every day through `through`, partial tail included.""" + path = tmp_path / f"p.{fmt}" + ident = backfill._RowIdentity(_Park("p", "Park")) + with path.open("w", encoding="utf-8", newline="") as handle: + writer = backfill.Writer(handle, fmt, True, ident) + for ref, row in archive.days_with_entities(archive.archive_from, archive.through): + writer.write(ref._replace(name=names.get(ref.id, ref.name)), row) + archive.calls.clear() + + +class TestA40FileIsCorrectedNotFrozen: + """The partial rows 4.0 wrote are replaced once, on the first run of this build.""" + + @pytest.mark.parametrize("fmt", ["ndjson", "csv"]) + def test_the_partial_tail_is_replaced_with_final_rows(self, tmp_path: Path, fmt: str) -> None: + archive = _Archive() + _write_as_4_0(tmp_path, fmt, archive, {}) + _v1_state(tmp_path, fmt) + partial_before = [r for r in _rows(tmp_path, fmt) if int(r["operatingMinutes"]) != 600] + assert partial_before, "the 4.0 file this test starts from has no partial rows" + + assert _run(archive, tmp_path, fmt) == 0 + # Seven days back from the old end, so every day that could have been + # partial is fetched again, and nothing earlier is. + assert archive.calls == [("2026-09-22", "2026-09-26")] + rows = _rows(tmp_path, fmt) + assert max(r["date"] for r in rows) == "2026-09-26" + _assert_every_row_final_and_unique(rows) + assert _state(tmp_path, fmt)["stateVersion"] == backfill.STATE_VERSION + + archive.advance(2) + _run(archive, tmp_path, fmt) + _assert_every_row_final_and_unique(_rows(tmp_path, fmt)) + + def test_rows_that_are_kept_are_kept_byte_for_byte(self, tmp_path: Path) -> None: + # The CSV is parsed and written back, so a name that needs quoting has to + # come out exactly as it went in. Compared against a file written only + # up to the cut in the first place. + nasty = {"ent-a": 'Space, "Mountain"\rFastPass', "ent-b": "=cmd|' /C calc'!A0"} + archive = _Archive(archive_from="2026-09-01") + _write_as_4_0(tmp_path, "csv", archive, nasty) + _v1_state(tmp_path, "csv") + backfill._trim_after(tmp_path / "p.csv", "csv", "2026-09-21") + + expected_dir = tmp_path / "expected" + expected_dir.mkdir() + short = _Archive(archive_from="2026-09-01", recorded_to="2026-09-21", through="2026-09-21") + _write_as_4_0(expected_dir, "csv", short, nasty) + assert (tmp_path / "p.csv").read_bytes() == (expected_dir / "p.csv").read_bytes() + + def test_an_ndjson_line_that_does_not_parse_is_kept(self, tmp_path: Path) -> None: + # Not this command's to judge: a line it cannot read is left where it is. + path = tmp_path / "p.ndjson" + path.write_text( + '{"date": "2026-09-01"}\n{"date": "2026-09-27"}\n{"trunc\n', encoding="utf-8" + ) + backfill._trim_after(path, "ndjson", "2026-09-21") + assert path.read_text(encoding="utf-8") == '{"date": "2026-09-01"}\n{"trunc\n' + + def test_a_csv_with_no_date_column_is_copied_whole(self, tmp_path: Path) -> None: + path = tmp_path / "p.csv" + path.write_text("\ufeffa,b\n1,2099-01-01\n", encoding="utf-8") + backfill._trim_after(path, "csv", "2026-09-21") + assert path.read_text(encoding="utf-8") == "\ufeffa,b\n1,2099-01-01\n" + + def test_an_interrupted_4_0_run_on_its_last_pages_is_trimmed_too(self, tmp_path: Path) -> None: + # Interrupted close enough to its end to have written unsettled days: + # those are removed and fetched again, like a finished file's. + archive = _Archive() + _write_as_4_0(tmp_path, "ndjson", archive, {}) + _v1_state(tmp_path, "ndjson", complete=False, lastDay="2026-09-28", resumeFrom=None) + assert _run(archive, tmp_path) == 0 + assert archive.calls == [("2026-09-22", "2026-09-26")] + _assert_every_row_final_and_unique(_rows(tmp_path)) + + def test_an_old_file_whose_newest_day_is_long_final_is_not_rewritten( + self, tmp_path: Path, monkeypatch + ) -> None: + # A park that stopped reporting: its 4.0 state ends months ago and every + # row is final. Reading a large file to find nothing to remove is waste. + archive = _Archive(recorded_to="2026-07-31", through="2026-07-31") + _write_as_4_0(tmp_path, "ndjson", archive, {}) + _v1_state(tmp_path, "ndjson", end="2026-09-28", lastDay="2026-07-31") + trimmed: list[str] = [] + monkeypatch.setattr(backfill, "_trim_after", lambda *a: trimmed.append(a[2])) + assert _run(archive, tmp_path) == 0 + assert trimmed == [] + + def test_an_interrupted_4_0_run_resumes_where_it_stopped(self, tmp_path: Path) -> None: + # Far from the tail, nothing it wrote can be partial: resume as before. + archive = _Archive() + path = tmp_path / "p.ndjson" + path.write_text('{"date": "2026-06-01"}\n', encoding="utf-8") + _v1_state(tmp_path, "ndjson", complete=False, lastDay="2026-06-30", resumeFrom="2026-07-01") + assert _run(archive, tmp_path) == 0 + assert archive.calls == [("2026-07-01", "2026-09-26")] + + def test_a_4_0_state_from_the_other_sdk_is_still_refused(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + (tmp_path / "p.ndjson").write_text('{"date": "2026-06-01"}\n', encoding="utf-8") + _v1_state(tmp_path, "ndjson", sdk="js") + assert _run(archive, tmp_path) == 1 + assert "js SDK" in capsys.readouterr().err + + +class TestAFinishedFileWrittenToAnotherContractIsRefused: + def test_a_complete_state_with_another_column_layout_is_not_appended_to( + self, tmp_path: Path, capsys + ) -> None: + # Found while making reruns incremental: only an UNFINISHED mismatched + # state was refused. A finished one fell through to a fresh start, and + # the fresh start opened the existing file in append mode -- a second + # full copy of the archive under a second header, exit 0. + (tmp_path / "p.csv").write_text("old,header\n1,2\n", encoding="utf-8") + path = backfill.state_path_for(tmp_path, "p", "csv") + path.write_text( + json.dumps( + { + "sdk": "py", + "sdkVersion": PACKAGE_VERSION, + "stateVersion": backfill.STATE_VERSION, + "format": "csv", + "columns": "0000deadbeef0000", + "start": "2026-06-01", + "end": "2026-09-20", + "lastDay": "2026-09-20", + "resumeFrom": None, + "complete": True, + } + ), + encoding="utf-8", + ) + archive = _Archive() + assert _run(archive, tmp_path, "csv") == 1 + assert archive.calls == [] + assert (tmp_path / "p.csv").read_text(encoding="utf-8") == "old,header\n1,2\n" + assert "column layout changed" in capsys.readouterr().err + + +class TestAStateWithNoFileStartsAgain: + def test_a_deleted_data_file_is_downloaded_again_not_resumed_into(self, tmp_path: Path) -> None: + # The state describes a file that is no longer there. Continuing would + # write a file that starts half-way through and record it as complete. + archive = _Archive() + _run(archive, tmp_path) + (tmp_path / "p.ndjson").unlink() + archive.advance(1) + assert _run(archive, tmp_path) == 0 + assert archive.calls[-1][0] == "2026-06-01" + rows = _rows(tmp_path) + assert min(r["date"] for r in rows) == "2026-06-01" + _assert_every_row_final_and_unique(rows) + + +def test_a_csv_line_round_trips_through_the_csv_module() -> None: + # `_trim_after` relies on csv.reader reading back exactly what `_csv_line` + # wrote. Pinned directly, for the characters that need quoting. + cells = {c: "" for c in backfill.CSV_COLUMNS} + cells.update({"parkName": 'a,"b"\r\nc', "date": "2026-09-01"}) + line = backfill._csv_line(cells) + parsed = next(csv.reader(io.StringIO(line, newline=""))) + assert ",".join(backfill._csv_cell(v) for v in parsed) + "\n" == line diff --git a/themeparks/backfill.py b/themeparks/backfill.py index 27921c1..056c16f 100644 --- a/themeparks/backfill.py +++ b/themeparks/backfill.py @@ -32,28 +32,36 @@ than alerts. Re-running is then safe in every direction: an unfinished park continues, a - FINISHED park is left alone rather than appended to twice, and a file this - command did not write is never touched without `--overwrite`. It re-reads the - furthest day on purpose -- a page can end mid-day -- so `(entityId, date)` is - the natural key if you load blind. + FINISHED park is brought up to date from the day after its last one rather + than appended to twice, and a file this command did not write is never + touched without `--overwrite`. `(entityId, date)` is the natural key if you + load blind: a run that died inside its first page resumes on the last day it + wrote, so that one day can appear twice. 4. It takes NAMES as well as ids, and DESTINATIONS as well as parks. A customer has "Walt Disney World Resort", not four park uuids, and making them look those up first was another wall. A destination back fills every park in it, into one file each. + +5. It writes FINAL days only. Today's row is the day so far, and the archive + records days 2 to 3 behind live data, so the newest days the API serves can + still change. The run ends at `span().final_through`, the newest day the + archive holds, and the next run carries on from the day after. Every row in + the file is one that will not change, so a nightly run only ever appends. """ from __future__ import annotations import argparse import contextlib +import csv import hashlib import json import os import re import sys import unicodedata -from datetime import date, datetime +from datetime import date, datetime, timedelta from datetime import date as _date from pathlib import Path from typing import Any, NamedTuple, TextIO, Union, get_args, get_origin @@ -71,7 +79,7 @@ ) from themeparks import TimeoutError as ApiTimeoutError # noqa: A004 - the SDK's, not the builtin's from themeparks._client import PACKAGE_VERSION, _default_user_agent -from themeparks._ergonomic.history import EntityRef, HistoryPage +from themeparks._ergonomic.history import EntityRef, HistoryPage, HistorySpan from themeparks._generated.models import HistoryDailyRow # The command's identity IN FRONT OF the SDK's, not instead of it. It used to be @@ -275,6 +283,17 @@ def _csv_row(ref: EntityRef, row: Any, ident: _RowIdentity) -> dict[str, Any]: } +class _Range(NamedTuple): + """The days asked for with `--since` and `--until`, both inclusive, or None. + + None at either end means "as far as there is": back to where the plan + reaches, forward to the newest final day. + """ + + since: str | None = None + until: str | None = None + + class _Park(NamedTuple): """A park's identity, so rows can name themselves. @@ -426,8 +445,19 @@ def _window_floor(exc: APIError) -> str | None: STATE_SUFFIX = ".backfill-state.json" #: Bumped when the meaning of a field changes. A state file from another version -#: is refused rather than guessed at. -STATE_VERSION = 1 +#: is refused rather than guessed at, with one exception: version 1, below. +#: +#: 2: `end` is the newest FINAL day, and nothing after it is in the file. In +#: version 1 it was `retrievableThrough`, usually today, so the newest rows of a +#: finished file were partial days that no later run replaced. +STATE_VERSION = 2 + +#: How far back from a version-1 file's `end` its rows may be partial. The +#: archive records days 2 to 3 behind live data, so a 4.0 run that ended on its +#: `retrievableThrough` wrote two or three days that were not final. A week +#: covers that with room to spare, and every day re-fetched costs nothing more +#: than the one request its page already needs. +V1_UNSETTLED_DAYS = 7 SDK_NAME = "py" @@ -479,12 +509,12 @@ def _state_mismatch(state: dict[str, Any], fmt: str) -> str | None: rows and resumed by the JavaScript command produced 172 rows with 64 duplicated keys and `complete: true`. """ - if state.get("stateVersion") != STATE_VERSION: - written = state.get("stateVersion") - return f"it was written by a different version of this command (state v{written})" if state.get("sdk") != SDK_NAME: other = state.get("sdk") return f"it was written by the {other} SDK, and resuming across SDKs is not supported" + if state.get("stateVersion") != STATE_VERSION: + written = state.get("stateVersion") + return f"it was written by a different version of this command (state v{written})" if state.get("format") != fmt: return f"it is a {state.get('format')} run" if state.get("columns") != _columns_fingerprint(fmt): @@ -582,11 +612,78 @@ class _Plan(NamedTuple): #: in this module has to consult it: `written == 0` means "this process wrote #: nothing", which on a resumed run is not the same as "the file is empty". resumed: bool = False + #: True when a FINISHED file is being carried forward to new final days. + extending: bool = False + +def _next_day(value: Day) -> str: + """The day after `value`, as YYYY-MM-DD.""" + return (date.fromisoformat(str(value)) + timedelta(days=1)).isoformat() -def _decide( - out_path: Path, state_path: Path, fmt: str, overwrite: bool, archive_from: Day -) -> _Plan | int: + +def _later(a: Day, b: Day) -> Day: + """The later of two days, either of which may be None. ISO days sort as text.""" + if a is None: + return b + if b is None: + return a + return a if str(a) >= str(b) else b + + +def _refuse(out_path: Path, why: str) -> int: + print( + f" {out_path.name}: {why}.\n" + f" --overwrite replace it with the range asked for\n" + f" or pass a different --out and run again", + file=sys.stderr, + ) + return 1 + + +def _range_fits( + out_path: Path, state: dict[str, Any], rng: _Range, archive_from: Day, continue_at: str +) -> int | None: + """None when `--since`/`--until` agree with the file being continued, else an exit code. + + A file here is one contiguous range of days, and a run can only append to it, + so two requests cannot be honoured without rewriting it: a `--since` before + the file starts, and one after the day it would continue from, which would + leave a gap the state file could not describe. Both are refused rather than + quietly ignored, which would hand back a file that is not what was asked for. + + A `--since` inside the file is fine and common: a cron line with a fixed + `--since` runs it every night, and one computed as "30 days ago" moves + forward every night. Before the archive starts is the same as its start. + """ + file_start = state.get("start") + since = _later(rng.since, archive_from) if rng.since is not None else None + if since is not None and file_start is not None and str(since) < str(file_start): + return _refuse( + out_path, + f"it was started from {file_start}, and --since {rng.since} would need days " + f"before that. Appending cannot add them", + ) + if rng.until is not None and file_start is not None and rng.until < str(file_start): + return _refuse(out_path, f"it was started from {file_start}, after --until {rng.until}") + if since is not None and str(since) > continue_at: + return _refuse( + out_path, + f"it continues from {continue_at}, so starting at --since {rng.since} would " + f"leave a gap", + ) + return None + + +class _Ask(NamedTuple): + """The range a run is asked for: where the archive starts, the newest day + to fetch, and `--since`/`--until`.""" + + archive_from: Day + end: Day + rng: _Range + + +def _decide(out_path: Path, state_path: Path, fmt: str, overwrite: bool, ask: _Ask) -> _Plan | int: """A `_Plan` to proceed with, or an exit code meaning "do not". Split out of `backfill_park` because it got long enough for ruff to object, @@ -594,23 +691,31 @@ def _decide( writing. Every branch here exists for a defect measured in review -- see the STATE block above for the five of them. """ + if ask.end is None: + # Nothing is final: a park the archive has not recorded a day of yet. + # Nothing is touched, so an existing file and its state stay as they are. + print( + f" nothing final to fetch into {out_path.name} yet. The archive records " + f"days 2 to 3 behind live data; run again later", + file=sys.stderr, + ) + return 0 + state = {} if overwrite else _read_state(state_path) if overwrite: out_path.unlink(missing_ok=True) state_path.unlink(missing_ok=True) file_exists = out_path.exists() and out_path.stat().st_size > 0 - mismatch = _state_mismatch(state, fmt) if state else None - resumable = bool(state) and mismatch is None - # Finished already. Say so and stop, rather than appending a second copy. - if state.get("complete") and resumable and file_exists: - print( - f" already complete: {state.get('start')} .. {state.get('end')} " - f"in {out_path.name} — pass --overwrite to fetch it again", - file=sys.stderr, - ) - return 0 + # A state file describing a data file that is no longer there. Continuing + # would write a file that starts part-way through its range and then record + # it as complete. There is nothing to continue, so start again. + if state and not file_exists: + state = {} + + if state and _upgradable(state, fmt): + state = _upgrade_v1(state, out_path, state_path, fmt) # A file we have no record of writing. Refusing is the only safe answer: # appending doubles it, truncating throws away someone's data. @@ -623,12 +728,19 @@ def _decide( ) return 1 - # A state file this build cannot resume. Refusing is the only safe answer: the - # file beside it was written to a different contract, and appending to it + # A state file this build cannot continue. Refusing is the only safe answer: + # the file beside it was written to a different contract, and appending to it # produces a file no reader can parse -- or worse, one that parses wrongly. - if state and mismatch is not None and not state.get("complete"): + # + # FINISHED OR NOT. Only an unfinished one used to be refused: a finished one + # fell through to a fresh start, and a fresh start opens the existing file in + # append mode, so the whole archive went in a second time under a second + # header, exit 0. A finished file is continued now, so it is refused too. + mismatch = _state_mismatch(state, fmt) if state else None + if mismatch is not None: + kind = "a finished" if state.get("complete") else "an unfinished" print( - f" there is an unfinished {out_path.name} beside this state file, but " + f" there is {kind} {out_path.name} beside this state file, but " f"{mismatch}.\n" f" --overwrite start this park again from the beginning\n" f" or move both files aside and run again", @@ -636,7 +748,53 @@ def _decide( ) return 1 - resuming = resumable and not state.get("complete") + if not state: + return _first_run(out_path, ask) + return _continue(out_path, state, ask) + + +def _first_run(out_path: Path, ask: _Ask) -> _Plan | int: + """A park with no file yet: from `--since`, or wherever the archive starts.""" + rng = ask.rng + start = _later(rng.since, ask.archive_from) if rng.since is not None else ask.archive_from + if rng.since is not None and str(start) > str(ask.end): + print( + f" nothing to fetch into {out_path.name}: --since {rng.since} is after " + f"{ask.end}, the newest final day", + file=sys.stderr, + ) + return 0 + return _Plan(start=start, has_rows=False, prior_start=None) + + +def _continue(out_path: Path, state: dict[str, Any], ask: _Ask) -> _Plan | int: + """A file this command wrote: carry it forward, or say why not.""" + rng, archive_from, end = ask.rng, ask.archive_from, ask.end + if state.get("complete"): + # FINISHED IS NOT FOREVER. It used to be: a rerun printed "already + # complete" and exited 0 without asking for a single new day, so a + # nightly cron looked healthy and never updated. The file holds every + # final day through `end`, so the next day is where it carries on. + recorded_end = state.get("end") + continue_at = _next_day(recorded_end) if recorded_end else str(state.get("start")) + refused = _range_fits(out_path, state, rng, archive_from, continue_at) + if refused is not None: + return refused + if end is None or continue_at > str(end): + print( + f" up to date: {out_path.name} is complete through {recorded_end}, " + f"and there is no final day after it yet", + file=sys.stderr, + ) + return 0 + return _Plan( + start=continue_at, + has_rows=True, + prior_start=state.get("start"), + resumed=True, + extending=True, + ) + # THE PAGE BOUNDARY, not the newest row. `last_day` is the highest date # written; the page it came from covered further, because an entity that # stopped reporting has no rows for the tail days. Resuming at `last_day` @@ -644,19 +802,120 @@ def _decide( # on the exit-75 path, which is the ordinary path for a long back fill, and # it breaks the (entityId, date) key the file is documented to have. # - # `last_day` stays as the fallback for the two cases with no boundary - # recorded: a state file written by 3.3.0, and a run that died part-way - # through its FIRST page. One duplicated day beats starting from the top and - # appending a second copy of the whole archive. - resume_at = (state.get("resumeFrom") or state.get("lastDay")) if resuming else None + # `last_day` stays as the fallback for the one case with no boundary + # recorded: a run that died part-way through its FIRST page. One duplicated + # day beats starting from the top and appending a second copy of the whole + # archive. + resume_at = state.get("resumeFrom") or state.get("lastDay") + continue_at = str(resume_at or state.get("start") or archive_from) + refused = _range_fits(out_path, state, rng, archive_from, continue_at) + if refused is not None: + return refused return _Plan( - start=resume_at or archive_from, - has_rows=file_exists and resuming, - prior_start=state.get("start") if resuming else None, - resumed=resuming, + start=continue_at, + has_rows=True, + prior_start=state.get("start"), + resumed=True, + ) + + +# -------------------------------------------------------------------------- +# Files written by 4.0.x, whose newest rows may be partial days. +# -------------------------------------------------------------------------- + + +def _upgradable(state: dict[str, Any], fmt: str) -> bool: + """A version-1 state file from this SDK, for this format and column layout.""" + return ( + state.get("stateVersion") == 1 + and state.get("sdk") == SDK_NAME + and state.get("format") == fmt + and state.get("columns") == _columns_fingerprint(fmt) ) +def _upgrade_v1( + state: dict[str, Any], out_path: Path, state_path: Path, fmt: str +) -> dict[str, Any]: + """Make a 4.0 file one this build can continue, replacing its unsettled tail. + + 4.0 ended every run at `retrievableThrough`, usually today, so the last few + days of a finished 4.0 file were written while they were still changing: + Magic Kingdom's last day summed to about half the operating minutes of a + full one. Nothing ever replaced them, because a finished park was never + fetched again. + + Which of those days were final at the time was not recorded, so every row + dated within `V1_UNSETTLED_DAYS` of that run's end is removed and the state + is set to carry on from the day after the cut. The next request then fetches + those days again, final this time. A file whose newest row is already older + than the cut, a park that stopped reporting long ago, is not read at all. + + The new state is written straight away, so a run that fails after this + point does not trim the same file twice. + """ + end = state.get("end") + upgraded = {**state, "stateVersion": STATE_VERSION} + if not end: + return upgraded + keep_through = (date.fromisoformat(str(end)) - timedelta(days=V1_UNSETTLED_DAYS)).isoformat() + last_day = state.get("lastDay") + if last_day is None or str(last_day) > keep_through: + _trim_after(out_path, fmt, keep_through) + upgraded["lastDay"] = keep_through if last_day is not None else None + if state.get("complete"): + upgraded["end"] = keep_through + else: + resume = state.get("resumeFrom") or last_day + if resume is None or str(resume) > _next_day(keep_through): + upgraded["resumeFrom"] = _next_day(keep_through) + _write_state(state_path, **upgraded) + return upgraded + + +def _trim_after(out_path: Path, fmt: str, keep_through: str) -> None: + """Remove every row dated after `keep_through`, leaving the rest byte for byte. + + Streamed into a file beside the original and swapped in with one rename, so + an interruption leaves either the old file or the new one, never half of + each. A line or record that cannot be read is kept: this command does not + get to decide that something it does not understand is worthless. + + The CSV is parsed and written back with this module's own quoting, which is + a function of the text alone, so a kept record comes out as it went in. + """ + scratch = out_path.with_name(out_path.name + ".trimming") + with out_path.open(encoding="utf-8", newline="") as src: + _copy_rows_through(src, scratch, fmt, keep_through) + os.replace(scratch, out_path) + + +def _copy_rows_through(src: TextIO, scratch: Path, fmt: str, keep_through: str) -> None: + """The body of `_trim_after`: copy every row dated on or before `keep_through`.""" + with scratch.open("w", encoding="utf-8", newline="") as dst: + if fmt == "csv": + reader = csv.reader(src) + header = next(reader, None) + if header is not None: + dst.write(",".join(_csv_cell(cell) for cell in header) + "\n") + names = [cell.lstrip("\ufeff") for cell in header] + # No `date` column: not a file this can read, so it is copied whole. + column = names.index("date") if "date" in names else -1 + for record in reader: + if 0 <= column < len(record) and record[column] > keep_through: + continue + dst.write(",".join(_csv_cell(cell) for cell in record) + "\n") + else: + for line in src: + try: + day = json.loads(line).get("date") + except (ValueError, AttributeError): + day = None + if isinstance(day, str) and day > keep_through: + continue + dst.write(line) + + class _Job(NamedTuple): """Everything streaming one park needs, so the streamer takes two arguments.""" @@ -684,6 +943,27 @@ def __init__(self) -> None: self.last_day: date | None = None self.resume_from: str | None = None self.skipped = False + #: The day this run's request started on, which is where a rerun has to + #: carry on if the run fails before its first page is finished. + self.first_day: str | None = None + + def resume_point(self) -> str | None: + """Where a rerun should continue, for the state file. + + The next page's start once a page is done. Before that, with rows + written, None: `lastDay` is the fallback and costs one duplicated day. + Before ANY row, the day this run began on. That last case had nothing + recorded, which was harmless while only a first run could reach it -- a + rerun of a file with no rows starts from the top anyway -- and is not + now that a finished file is carried forward. A rerun interrupted before + its first page would otherwise have fallen back to the archive's start + and appended the whole archive a second time. + """ + if self.resume_from is not None: + return self.resume_from + if self.last_day is not None: + return None + return self.first_day def _stream(job: _Job, start: Day, progress: _Progress) -> None: @@ -705,6 +985,7 @@ def note_page(page: HistoryPage) -> None: def write_rows(writer: Writer, first_day: Day) -> None: """Stream one range into the file. Raises whatever the SDK raises.""" + progress.first_day = None if first_day is None else str(first_day) for ref, row in job.history.days_with_entities(first_day, job.end, on_page=note_page): writer.write(ref, row) progress.written += 1 @@ -780,17 +1061,56 @@ def _nothing_written( remove, so the state is kept and the exit code says the run did not finish. """ if resumed: - _record(sf, progress.last_day, progress.resume_from, complete=False) + _record(sf, progress.last_day, progress.resume_point(), complete=False) return 1 out_path.unlink(missing_ok=True) state_path.unlink(missing_ok=True) return 0 -def backfill_park( - tp: ThemeParks, park: _Park, out_dir: Path, fmt: str, overwrite: bool = False +def _run_end(span: HistorySpan, rng: _Range) -> Day: + """The last day this run asks for: the newest FINAL day, or `--until` if earlier. + + Not `retrievable_through`. That is usually today, and today's row is the day + so far; the archive records days 2 to 3 behind, so the days in between can + still change too. Ending there wrote partial rows -- Magic Kingdom's last day + at about half a full day's operating minutes -- and, since a finished park + was never fetched again, they stayed partial. + """ + end: Day = span.final_through + if rng.until is not None and end is not None and rng.until < str(end): + end = rng.until + return end + + +def _announce(park_id: str, plan: _Plan, ask: _Ask, span: HistorySpan, out_path: Path) -> None: + """Say what this run will fetch, and why it stops where it does.""" + note = " (new days)" if plan.extending else " (resumed)" if plan.resumed else "" + print(f"{park_id}: {plan.start} .. {ask.end}{note} -> {out_path}", file=sys.stderr) + through = span.retrievable_through + if through is not None and str(through) > str(ask.end) and ask.end == span.final_through: + print( + f" stopping at {ask.end}: the days after it are still being recorded and " + f"can change. The next run adds them once they are final", + file=sys.stderr, + ) + + +def backfill_park( # noqa: PLR0913 - window is keyword-only, added without breaking callers + tp: ThemeParks, + park: _Park, + out_dir: Path, + fmt: str, + overwrite: bool = False, + *, + window: _Range | None = None, ) -> int: - """Write one park's daily history. Returns 0, or EX_TEMPFAIL if the budget ran out.""" + """Write one park's daily history. Returns 0, or EX_TEMPFAIL if the budget ran out. + + `window` is `--since`/`--until`. Without it the range is everything the key + may read, through the newest final day. + """ + rng = window or _Range() park_id = park.id history = tp.entity(park_id).history @@ -821,19 +1141,14 @@ def backfill_park( ext = "csv" if fmt == "csv" else "ndjson" out_path = out_dir / f"{park_id}.{ext}" state_path = state_path_for(out_dir, park_id, fmt) - end = span.retrievable_through - - decided = _decide(out_path, state_path, fmt, overwrite, span.archive_from) + ask = _Ask(span.archive_from, _run_end(span, rng), rng) + decided = _decide(out_path, state_path, fmt, overwrite, ask) if isinstance(decided, int): return decided - start, has_rows, prior_start, resumed_run = decided - resuming = prior_start is not None + start, has_rows, prior_start, resumed_run = decided[:4] + end = ask.end sf = _StateFile(state_path, fmt, prior_start or start, end) - - print( - f"{park_id}: {start} .. {end}{' (resumed)' if resuming else ''} -> {out_path}", - file=sys.stderr, - ) + _announce(park_id, decided, ask, span, out_path) if _is_empty_window(start, end): return _window_closed(out_path, end, start, resumed=resumed_run) @@ -846,7 +1161,7 @@ def backfill_park( except BudgetExhaustedError as exc: # The budget is hourly, so a spent one can be most of an hour from # resetting. Record how far we got and exit 75 rather than sleeping. - _record(sf, progress.last_day, progress.resume_from, complete=False) + _record(sf, progress.last_day, progress.resume_point(), complete=False) if progress.written == 0 and not resumed_run: # A budget spent before the first page left a 0-byte file that reads # as "this park has no history". @@ -869,7 +1184,7 @@ def backfill_park( # "this park has no history" -- on a six-park destination the customer # counts six files and never sees which one is empty. if progress.last_day is not None: - _record(sf, progress.last_day, progress.resume_from, complete=False) + _record(sf, progress.last_day, progress.resume_point(), complete=False) # `written` counts rows THIS process wrote, so on a resumed run it is 0 # while the file holds everything the previous runs fetched. Deleting it # there destroyed the archive and left the state file pointing into the @@ -1089,6 +1404,25 @@ def _print_list(catalogue: list[tuple[str, str, str, str]], needle: str | None) return 0 +_ISO_DAY = re.compile(r"^\d{4}-\d{2}-\d{2}$") + + +def _iso_day(value: str) -> str: + """An argparse type: a real calendar day written YYYY-MM-DD, returned as given. + + Strict on purpose. From 3.11 `date.fromisoformat` also takes `20250101` and + week dates, and 3.9 does not, so the same command line would mean something + on one interpreter and fail on another. Only the form the API itself uses is + accepted, everywhere. + """ + try: + if _ISO_DAY.match(value): + return date.fromisoformat(value).isoformat() + except ValueError: + pass + raise argparse.ArgumentTypeError(f"expected a day as YYYY-MM-DD, got {value!r}") + + EPILOG = """examples: export THEMEPARKS_API_KEY=tpw_your_key how far back this reaches is your plan, so without a key you get the 7 @@ -1111,6 +1445,23 @@ def _print_list(catalogue: list[tuple[str, str, str, str]], needle: str | None) themeparks-backfill several parks in one run, sharing one connection and one budget + themeparks-backfill "Epcot" --since 2025-01-01 + from a day of your choosing instead of as far back as your plan reaches. + --until YYYY-MM-DD sets the last day. Both are inclusive. + + themeparks-backfill "Epcot" (again, from cron, every night) + adds the days that became final since the last run, and nothing else. + +only final days are written. Today's row is the day so far, and the archive +records days 2 to 3 behind live data, so the newest days can still change. A run +ends at the newest final day and the next run carries on from the day after, so +the file only ever grows and no row in it changes later. + +--since applies when a file is started. A later run continues that file forward +and accepts the same --since, or a later one. One earlier than the file's first +day, or past the day it continues from, is refused: use --overwrite, or a +different --out. + exit codes: 0 done 75 the hourly history budget ran out. Progress is checkpointed; run the same @@ -1161,8 +1512,20 @@ def main(argv: list[str] | None = None) -> int: "--overwrite", action="store_true", help="replace an existing file instead of refusing. Without it, a park" - " that finished is not fetched twice and a file this command did not" - " write is never touched.", + " that finished is brought up to date rather than fetched twice, and a" + " file this command did not write is never touched.", + ) + parser.add_argument( + "--since", + type=_iso_day, + metavar="YYYY-MM-DD", + help="first day to download, inclusive (default: as far back as your plan reaches)", + ) + parser.add_argument( + "--until", + type=_iso_day, + metavar="YYYY-MM-DD", + help="last day to download, inclusive (default: the newest final day)", ) parser.add_argument( "--version", @@ -1178,6 +1541,8 @@ def main(argv: list[str] | None = None) -> int: help="output directory (default: .)", ) args = parser.parse_args(argv) + if args.since is not None and args.until is not None and args.since > args.until: + parser.error(f"--since {args.since} is after --until {args.until}") _use_utf8(sys.stdout, sys.stderr) @@ -1280,9 +1645,12 @@ def _run_all(tp: ThemeParks, targets: list[tuple[str, str]], args: Any) -> int: code still says something went wrong. """ failed: list[str] = [] + window = _Range(getattr(args, "since", None), getattr(args, "until", None)) for park_id, pname in targets: try: - status = backfill_park(tp, _Park(park_id, pname), args.out, args.format, args.overwrite) + status = backfill_park( + tp, _Park(park_id, pname), args.out, args.format, args.overwrite, window=window + ) except (ThemeParksError, OSError) as exc: print(f"{park_id}: {exc}", file=sys.stderr) failed.append(pname) From 7219731a438fcdc37cbcec1fcb7d5e86824f78cc Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Mon, 28 Sep 2026 21:29:18 +0100 Subject: [PATCH 3/9] docs: waitTime is a float, and the backfill and history changes The README's queue table said `waitTime: int | None`. The API's schema declares it a JSON `number` and the generated models type it `float`, so a raw row dumped to JSON says `45.0`. The table now says `float | None`, explains the `.0` and how to get an int, and a test holds the table to the models' types. README, cookbook and CHANGELOG cover `--since`/`--until`, reruns that bring a file up to date, the final-day boundary, the one-time correction of 4.0 files, and `changes().opening`. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 71 +++++++++++++++++++++++++++++++++ README.md | 56 ++++++++++++++++++++++---- docs/cookbook.md | 37 +++++++++++++++-- tests/unit/test_readme_types.py | 54 +++++++++++++++++++++++++ 4 files changed, 207 insertions(+), 11 deletions(-) create mode 100644 tests/unit/test_readme_types.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 1d36dd4..9475c61 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,76 @@ # Changelog +## [Unreleased] + +### Added + +- **`themeparks-backfill --since YYYY-MM-DD` and `--until YYYY-MM-DD`.** A key + that reaches the whole archive used to download all of it, every time: a + buyer who wanted the last twelve months got five years. Both days are + inclusive and must be real calendar days written `YYYY-MM-DD`; `--since` + after `--until` is refused before anything is requested. `--since` applies when a file is started. A later run accepts the + same `--since` or a later one (a cron line computing "30 days ago" works), and + refuses one earlier than the file's first day, or one that would leave a gap, + instead of quietly handing back a file that is not what was asked for. + +- **`history.changes()` exposes the `opening` state.** The raw history + response carries, per entity, the state in force at the start of the range, + and `changes()` threw it away. Without it the minutes between midnight and an + entity's first change had no known status, so a day rebuilt from raw history + disagreed with the daily summary whenever a ride was still running from the + night before. The result of `changes()` now has `opening`, a dict of + `HistoryOpening` keyed by entity id, covering every entity in the response, + including one that did not change all day. Iterating it yields exactly what it + always did, and reading `opening` costs no extra request. The async client's + result has the same attribute once the response has arrived: iterate first, + or `await changes.load()`. + +- **`HistorySpan.final_through`**: the newest day whose daily row will not + change again, the earlier of `recorded_to` and `retrievable_through`. A + property, so `a, b, c = span` still works. + +### Fixed + +- **A finished park now updates on the next run.** A rerun used to print + `already complete` and exit 0 without fetching a single new day, so a nightly + cron looked healthy and never updated; the only way to get yesterday was + `--overwrite`, which downloaded the whole archive again. A finished file is now + carried forward from the day after its last one, appending only the new days. + An interrupted run still resumes exactly where it stopped, including a rerun + interrupted before its first page, which would otherwise have started again + from the top of the archive. + +- **The newest rows of a backfill were partial days, and stayed that way.** A run + ended on `retrievableThrough`, which is usually today: today's row is the day + so far, and the archive records days 2 to 3 behind live data, so the last few + days of every file were still changing when they were written. Magic Kingdom's + last day summed to about half the operating minutes of a full one. A run now + ends at `final_through`, says so when it holds days back, and the next run adds + them once they are final. Every row in the file is one that will not change. + + **Files written by 4.0.x are corrected once.** Their state file does not say + which of their newest days were final, so the first run of this version + removes the rows from the last seven days before that run's end and fetches + those days again. Every other row is left byte for byte as it was. A 4.0 file + whose newest row is older than that is not rewritten at all. The state file + format moves to version 2 for this; version 1 files from this SDK are upgraded, + not refused. + +- **A finished file written to a different column layout was appended to.** Only + an unfinished one was refused. A finished one fell through to a fresh start, + which opened the existing file in append mode and wrote the whole archive into + it a second time under a second header, exit 0. It is refused now, the same + way. + +- **A state file whose data file had been deleted was continued**, producing a + file that started part-way through its range and was then recorded as + complete. The park is downloaded again from the start instead. + +- **The README said `waitTime` is an `int`.** The API's schema declares it a + JSON `number`, the models type it `float`, and a raw row dumped to JSON says + `45.0`. The README now says `float | None`, explains the `45.0`, and a test + holds its table to the models' types. + ## [4.0.1] - 2026-09-28 ### Fixed diff --git a/README.md b/README.md index 9703e9a..b14dd0d 100644 --- a/README.md +++ b/README.md @@ -147,13 +147,19 @@ Each variant is exposed as an attribute on `entry.queue`. All are | Attribute | Type | Fields | |------------------------|-----------------------|------------------------------------------------------------------------------------------------------| -| `queue.STANDBY` | `StandbyQueue` | `waitTime: int \| None` | -| `queue.SINGLE_RIDER` | `SingleRiderQueue` | `waitTime: int \| None` | -| `queue.PAID_STANDBY` | `PaidStandbyQueue` | `waitTime: int \| None` | +| `queue.STANDBY` | `StandbyQueue` | `waitTime: float \| None` | +| `queue.SINGLE_RIDER` | `SingleRiderQueue` | `waitTime: float \| None` | +| `queue.PAID_STANDBY` | `PaidStandbyQueue` | `waitTime: float \| None` | | `queue.RETURN_TIME` | `ReturnTimeQueue` | `state`, `returnStart`, `returnEnd` | | `queue.PAID_RETURN_TIME` | `PaidReturnTimeQueue` | `state`, `returnStart`, `returnEnd`, `price` | | `queue.BOARDING_GROUP` | `BoardingGroupQueue` | `allocationStatus`, `currentGroupStart`, `currentGroupEnd`, `nextAllocationTime`, `estimatedWait` | +`waitTime` is a `float` because the API's schema declares it a JSON `number`, +not an integer, so a 45-minute wait arrives as `45.0` and a raw row dumped to +JSON says `45.0`. Use `current_wait_time(entry)` or `int(...)` when you want an +`int`, and format with `{wait:.0f}`, not `{wait:d}`, if you print the field +directly. + #### Direct access ```python @@ -201,7 +207,7 @@ with ThemeParks() as tp: live = tp.entity("75ea578a-adc8-4116-a54d-dccb60765ef9").live() for entry in live.liveData or []: for q in iter_queues(entry): - # q is a dict, e.g. {"type": "STANDBY", "waitTime": 35} + # q is a dict, e.g. {"type": "STANDBY", "waitTime": 35.0} # or {"type": "PAID_RETURN_TIME", "state": "AVAILABLE", ...} print(entry.name, q) ``` @@ -274,19 +280,32 @@ with ThemeParks(api_key=KEY) as tp: history = tp.entity(DISNEYLAND).history # What exists, and what your key may read. Same three fields whether the - # id is a park or a single ride. + # id is a park or a single ride. `final_through` is the newest day whose + # row will not change again: store through that, ask for the rest later. span = history.span() print(span.archive_from, span.recorded_to, span.retrievable_through) + print(span.final_through) # One summary row per park-local day, as (entity id, row). for entity_id, row in history.days(span.archive_from, span.retrievable_through): print(row.date, entity_id, row.operatingMinutes, row.standby.p50 if row.standby else None) - # Every recorded change on one day. - for entity_id, row in history.changes("2026-09-20"): + # Every recorded change on one day, and the state before the first of them. + changes = history.changes("2026-09-20") + for entity_id, row in changes: print(row.time, entity_id, row.status) + for entity_id, opening in changes.opening.items(): + print(entity_id, "at the start of the day:", opening.status) ``` +**A day rebuilds from `opening` plus the rows.** Each row is the entity's +complete state from its `time` until the next row. `changes.opening` is the +state in force before the first row, keyed by entity id, so the minutes between +midnight and a ride's first change have a status too: a ride still running from +the night before, say. It covers every entity in the response, including one +that did not change all day. Iterating `changes()` yields exactly what it +always did, and reading `opening` costs no extra request. + **Ask the park, not the rides.** Both history endpoints answer every entity in a park in one request. Pulling the same data ride by ride is around a hundred times more calls for a large resort, against the same budget. Pass a park id @@ -331,6 +350,29 @@ It reads how far back your own key may ask and starts there, writes NDJSON or has done so re-running never duplicates a file, and exits 75 when the hourly history budget runs out so a scheduler retries rather than alerts. +```bash +themeparks-backfill "Epcot" --since 2025-01-01 # not the whole archive +themeparks-backfill "Epcot" --since 2025-01-01 --until 2025-12-31 # one year, both days inclusive +``` + +**Run it again to bring a file up to date.** A finished park is carried +forward from the day after its last one, so the same command in a nightly cron +appends the new days and nothing else. Only **final** days are written: today's +row is the day so far, and the archive records days 2 to 3 behind live data, so +the newest days can still change. A run stops at the newest final day +(`span().final_through`) and says so, and the next run adds the rest. Every row +in the file is one that will not change later. + +`--since` applies when a file is started. Later runs continue that file and +accept the same `--since`, or a later one, such as a cron line computing "30 +days ago". One earlier than the file's first day, or one that would leave a gap, +is refused rather than ignored: pass `--overwrite`, or a different `--out`. + +Files written by 4.0.x ended on today, so their newest rows can be partial. The +first run of this version removes the rows from the last week of such a file and +fetches those days again, final this time. Everything else in the file is left +exactly as it was. + `python -m themeparks.backfill` is the same thing, which is the one to use if `pip install --user` put the script somewhere off your PATH. `themeparks-backfill --help` has the rest. diff --git a/docs/cookbook.md b/docs/cookbook.md index ba95c7d..b8e4080 100644 --- a/docs/cookbook.md +++ b/docs/cookbook.md @@ -283,7 +283,7 @@ with ThemeParks() as tp: live = tp.entity(MAGIC_KINGDOM).live() for entry in live.liveData or []: for q in iter_queues(entry): - # q is e.g. {"type": "STANDBY", "waitTime": 45} + # q is e.g. {"type": "STANDBY", "waitTime": 45.0} # or {"type": "PAID_RETURN_TIME", "state": "AVAILABLE", "price": {...}, ...} print(entry.name, q["type"], q) ``` @@ -306,7 +306,9 @@ last day your key may retrieve, so you ask for days that exist instead of discovering the ends by trial. Bound the backfill by `retrievable_through`, not by `recorded_to`: the archive holds more than a free or Pro key is entitled to read, and asking past the entitlement is how a backfill walks into -a wall of 403s at the end of a long run. +a wall of 403s at the end of a long run. If you write each day once and never +revisit it, end at `span.final_through` instead: the earlier of the two, and +the newest day whose row will not change again. ```python import json @@ -392,8 +394,15 @@ range above is refused on its FIRST request unless you already know your floor. The command reads it out of the 403 and starts again there. **Re-running.** The loop above appends, so running it twice doubles the file. -The command records what it wrote and declines to fetch a finished park again -unless you pass `--overwrite`. +The command records what it wrote, and a second run carries a finished park +forward from the day after its last one instead, so a nightly cron keeps the +file current. `--since` and `--until` pick the days; `--overwrite` starts again. + +**Days that are not final yet.** The loop above ends at `retrievable_through`, +usually today, and today's row is the day so far. Recent days can still change +too, because the archive records days 2 to 3 behind live data. Stop at +`span.final_through` if you store rows once, as the command does, or fetch the +days after it again on your next run. It also takes destinations as well as parks, names every row with the park and the entity as the history response reported them, and has `--list` for finding @@ -413,6 +422,26 @@ with ThemeParks(api_key="YOUR_KEY") as tp: print(row.time, entity_id, row.status, row.queue) ``` +To rebuild what an entity was doing at any moment, you also need the state +before its first change. That is `opening`, keyed by entity id, on the same +result: + +```python +with ThemeParks(api_key="YOUR_KEY") as tp: + changes = tp.entity(DISNEYLAND).history.changes("2026-09-20") + rows = list(changes) + for entity_id, opening in changes.opening.items(): + # In force from opening.time until this entity's first row. + print(entity_id, opening.time, opening.status) +``` + +Each row holds from its `time` until the next row's, so `opening` plus the +rows cover the whole range with no gap. Without it, a ride that was still +running from the night before has no known status until its first change. +`opening.observedAt` says when that state was last seen, which can be long +before the range for a ride whose feed stopped. With the async client, iterate +first or call `await changes.load()` before reading `opening`. + A park answers one day per call. A single entity answers up to 31 days, so pass `start=` and `end=` there instead of `date=`. You do not have to remember which cap applies: ask for the range you want, and the API either diff --git a/tests/unit/test_readme_types.py b/tests/unit/test_readme_types.py new file mode 100644 index 0000000..1b09a8d --- /dev/null +++ b/tests/unit/test_readme_types.py @@ -0,0 +1,54 @@ +"""The README's queue table states field types; they must be the models' types. + +It said `waitTime: int | None` while the generated models, and the spec they +come from, say a JSON `number`, which is a `float` here. A customer who believed +the README wrote `f"{wait:d}"` and got a ValueError, or dumped raw history rows +and found `6.0` where they expected `6`. The spec is the authority; the README +follows it, and this test is what keeps it following. +""" + +from __future__ import annotations + +import re +from pathlib import Path +from typing import Union, get_args, get_type_hints + +from themeparks._generated import models + +README = Path(__file__).resolve().parents[2] / "README.md" + +# | `queue.STANDBY` | `StandbyQueue` | `waitTime: float \\| None` | +ROW = re.compile(r"^\|\s*`queue\.\w+`\s*\|\s*`(\w+)`\s*\|\s*`(\w+): ([^`]+)`\s*\|\s*$") + +_NAMES = {"int": int, "float": float, "str": str, "bool": bool, "None": type(None)} + + +def _declared() -> list[tuple[str, str, str]]: + return [ + (m.group(1), m.group(2), m.group(3).replace("\\|", "|")) + for line in README.read_text(encoding="utf-8").splitlines() + if (m := ROW.match(line)) + ] + + +def _parse(text: str) -> set[type]: + return {_NAMES[part.strip()] for part in text.split("|")} + + +def test_the_table_was_found() -> None: + # A reworded table would make the test below vacuously pass. + assert len(_declared()) == 3 + + +def test_every_stated_type_is_the_model_type() -> None: + for model_name, field, stated in _declared(): + hint = get_type_hints(getattr(models, model_name))[field] + actual = set(get_args(hint)) if get_args(hint) else {hint} + assert _parse(stated) == actual, f"README says {model_name}.{field}: {stated}" + + +def test_the_model_type_is_what_the_spec_says() -> None: + # `number` in the spec. If the generator ever maps it to int, the README + # line above has to change with it, and this says why it failed. + hint = get_type_hints(models.StandbyQueue)["waitTime"] + assert hint == Union[float, None] From 91ea8d139977a7226336a89ee35aa604ba9bbded Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Mon, 28 Sep 2026 21:31:56 +0100 Subject: [PATCH 4/9] test: read the README's field types from pydantic, which 3.9 can evaluate typing.get_type_hints cannot evaluate the models' `float | None` on 3.9; pydantic already resolved it with the backport the package depends on. Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/unit/test_readme_types.py | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/tests/unit/test_readme_types.py b/tests/unit/test_readme_types.py index 1b09a8d..0507204 100644 --- a/tests/unit/test_readme_types.py +++ b/tests/unit/test_readme_types.py @@ -11,7 +11,7 @@ import re from pathlib import Path -from typing import Union, get_args, get_type_hints +from typing import Any, Union, get_args from themeparks._generated import models @@ -31,6 +31,16 @@ def _declared() -> list[tuple[str, str, str]]: ] +def _annotation(model_name: str, field: str) -> Any: + """The field's type as pydantic resolved it. + + Not `typing.get_type_hints`: the models are written `float | None`, which + 3.9 cannot evaluate. Pydantic already resolved it, with the backport the + package depends on for exactly this. + """ + return getattr(models, model_name).model_fields[field].annotation + + def _parse(text: str) -> set[type]: return {_NAMES[part.strip()] for part in text.split("|")} @@ -42,7 +52,7 @@ def test_the_table_was_found() -> None: def test_every_stated_type_is_the_model_type() -> None: for model_name, field, stated in _declared(): - hint = get_type_hints(getattr(models, model_name))[field] + hint = _annotation(model_name, field) actual = set(get_args(hint)) if get_args(hint) else {hint} assert _parse(stated) == actual, f"README says {model_name}.{field}: {stated}" @@ -50,5 +60,5 @@ def test_every_stated_type_is_the_model_type() -> None: def test_the_model_type_is_what_the_spec_says() -> None: # `number` in the spec. If the generator ever maps it to int, the README # line above has to change with it, and this says why it failed. - hint = get_type_hints(models.StandbyQueue)["waitTime"] + hint = _annotation("StandbyQueue", "waitTime") assert hint == Union[float, None] From 6bdc46ffcbb29295574a6e2423b6dee01a6d22d1 Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Mon, 28 Sep 2026 22:09:01 +0100 Subject: [PATCH 5/9] fix(backfill): never continue a file past a gap; re-download a 4.0 file inside the cut A continued file whose next day is older than the key may read (a cron that missed more days than the window, or a lapsed plan) carried on from the key's first day, leaving a gap the state file did not record. It is refused now, exit 1, with the file and state untouched and a message naming --overwrite. A 4.0 file lying wholly inside the seven-day refetch window, as every anonymous 7-day file does, was trimmed to nothing with a state continuing from a day the key could no longer read. It is downloaded again instead. A fixed --since older than the key's window is pinned by a test: the file starts at the key's first day and the same command line succeeds every night. Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/mutation/mutants.json | 18 ++++- tests/unit/test_backfill_incremental.py | 99 +++++++++++++++++++++++++ themeparks/backfill.py | 80 ++++++++++++++++---- 3 files changed, 182 insertions(+), 15 deletions(-) diff --git a/tests/mutation/mutants.json b/tests/mutation/mutants.json index 5ace545..989687f 100644 --- a/tests/mutation/mutants.json +++ b/tests/mutation/mutants.json @@ -197,8 +197,8 @@ { "name": "a 4.0 file keeps its partial tail", "file": "themeparks/backfill.py", - "find": " _trim_after(out_path, fmt, keep_through)\n", - "replace": " pass\n", + "find": "if _trim_after(out_path, fmt, keep_through) == 0:", + "replace": "if False:", "why": "The partial days 4.0 wrote stay in the file for good, next to nothing that would ever correct them." }, { @@ -221,6 +221,20 @@ "find": "return {entity.id: entity.opening for entity in entities}", "replace": "return {}", "why": "The opening is the state before the first change; without it a day cannot be rebuilt from raw history." + }, + { + "name": "a 4.0 file wholly inside the cut is trimmed to nothing", + "file": "themeparks/backfill.py", + "find": "if _trim_after(out_path, fmt, keep_through) == 0:", + "replace": "if _trim_after(out_path, fmt, keep_through) == -1:", + "why": "Every anonymous 4.0 file: an empty file whose state continues from a day the key can no longer read, refused on every later run." + }, + { + "name": "a continued file jumps to the key's first day", + "file": "themeparks/backfill.py", + "find": " if job.resumed:\n", + "replace": " if False:\n", + "why": "A cron that missed more nights than the key's window carries on from the key's first day, leaving days missing from the middle of a file whose state claims them." } ] } diff --git a/tests/unit/test_backfill_incremental.py b/tests/unit/test_backfill_incremental.py index 4f82d7f..4e5e100 100644 --- a/tests/unit/test_backfill_incremental.py +++ b/tests/unit/test_backfill_incremental.py @@ -648,6 +648,105 @@ def test_a_4_0_state_from_the_other_sdk_is_still_refused(self, tmp_path: Path, c assert "js SDK" in capsys.readouterr().err +class TestTheKeysWindowAndAFileBeingContinued: + """A key's window moves forward every day; a file being continued does not.""" + + def test_a_fixed_since_before_the_window_works_every_night(self, tmp_path: Path) -> None: + # `--since 2025-01-01` in a cron, on a key whose window starts later: the + # first run starts at the key's first day, and the same line must keep + # succeeding on every later night. + archive = _Archive(archive_from="2025-06-01", floor="2026-08-01") + rng = _Range(since="2026-01-01") + for _ in range(3): + assert _run(archive, tmp_path, window=rng) == 0 + archive.advance(1) + archive.floor = (_day(archive.floor) + timedelta(days=1)).isoformat() + rows = _rows(tmp_path) + assert min(r["date"] for r in rows) == "2026-08-01" + assert max(r["date"] for r in rows) == "2026-09-28" + _assert_every_row_final_and_unique(rows) + + def test_a_continued_file_never_jumps_to_the_keys_first_day( + self, tmp_path: Path, capsys + ) -> None: + # The cron missed more nights than the key's window holds. Carrying on + # from the key's first day would leave days missing from the middle of + # the file while its state claimed them. Refused, and nothing touched. + archive = _Archive(archive_from="2026-09-01") + _run(archive, tmp_path) + before = (tmp_path / "p.ndjson").read_bytes() + state_before = _state(tmp_path) + archive.advance(20) + archive.floor = "2026-10-05" + capsys.readouterr() + + assert _run(archive, tmp_path) == 1 + assert archive.calls[-1][0] == "2026-09-27", "asked for anything but its own next day" + assert (tmp_path / "p.ndjson").read_bytes() == before + assert _state(tmp_path) == state_before + err = capsys.readouterr().err + assert "reaches back to 2026-10-05" in err + assert "continues from 2026-09-27" in err + assert "gap" in err and "--overwrite" in err + # And it stays refused until someone decides, rather than carrying on. + assert _run(archive, tmp_path) == 1 + + def test_an_interrupted_run_is_not_resumed_past_a_gap_either(self, tmp_path: Path) -> None: + archive = _Archive(archive_from="2026-06-01") + (tmp_path / "p.ndjson").write_text('{"date": "2026-06-01"}\n', encoding="utf-8") + state = backfill.state_path_for(tmp_path, "p", "ndjson") + state.write_text( + json.dumps( + { + "sdk": "py", + "sdkVersion": PACKAGE_VERSION, + "stateVersion": backfill.STATE_VERSION, + "format": "ndjson", + "columns": "", + "start": "2026-06-01", + "end": "2026-09-26", + "lastDay": "2026-06-30", + "resumeFrom": "2026-07-01", + "complete": False, + } + ), + encoding="utf-8", + ) + archive.floor = "2026-08-01" + assert _run(archive, tmp_path) == 1 + assert (tmp_path / "p.ndjson").read_text(encoding="utf-8") == '{"date": "2026-06-01"}\n' + + def test_a_4_0_file_wholly_inside_the_refetch_window_is_downloaded_again( + self, tmp_path: Path + ) -> None: + # Every anonymous 4.0 file: seven days, all within the cut. Trimmed, it + # would be empty with a state continuing from a day the key can no longer + # read. It is started again instead. + archive = _Archive(archive_from="2021-07-03", floor="2026-09-22") + _write_4_0_window(tmp_path, archive, "2026-09-22") + _v1_state(tmp_path, "ndjson", start="2021-07-03") + archive.floor = "2026-09-23" # a day later, the window has moved on + assert _run(archive, tmp_path) == 0 + rows = _rows(tmp_path) + assert min(r["date"] for r in rows) == "2026-09-23" + assert max(r["date"] for r in rows) == "2026-09-26" + _assert_every_row_final_and_unique(rows) + assert _state(tmp_path)["stateVersion"] == backfill.STATE_VERSION + + +def _write_4_0_window(tmp_path: Path, archive: _Archive, first: str) -> None: + """A 4.0 file that holds only `first` .. `through`, as an anonymous run wrote it.""" + path = tmp_path / "p.ndjson" + ident = backfill._RowIdentity(_Park("p", "Park")) + floor, archive.floor = archive.floor, None + with path.open("w", encoding="utf-8", newline="") as handle: + writer = backfill.Writer(handle, "ndjson", True, ident) + for ref, row in archive.days_with_entities(first, archive.through): + writer.write(ref, row) + archive.floor = floor + archive.calls.clear() + + class TestAFinishedFileWrittenToAnotherContractIsRefused: def test_a_complete_state_with_another_column_layout_is_not_appended_to( self, tmp_path: Path, capsys diff --git a/themeparks/backfill.py b/themeparks/backfill.py index 056c16f..436ca4e 100644 --- a/themeparks/backfill.py +++ b/themeparks/backfill.py @@ -715,7 +715,9 @@ def _decide(out_path: Path, state_path: Path, fmt: str, overwrite: bool, ask: _A state = {} if state and _upgradable(state, fmt): - state = _upgrade_v1(state, out_path, state_path, fmt) + upgraded = _upgrade_v1(state, out_path, state_path, fmt) + state = upgraded if upgraded is not None else {} + file_exists = upgraded is not None # A file we have no record of writing. Refusing is the only safe answer: # appending doubles it, truncating throws away someone's data. @@ -836,7 +838,7 @@ def _upgradable(state: dict[str, Any], fmt: str) -> bool: def _upgrade_v1( state: dict[str, Any], out_path: Path, state_path: Path, fmt: str -) -> dict[str, Any]: +) -> dict[str, Any] | None: """Make a 4.0 file one this build can continue, replacing its unsettled tail. 4.0 ended every run at `retrievableThrough`, usually today, so the last few @@ -851,6 +853,11 @@ def _upgrade_v1( those days again, final this time. A file whose newest row is already older than the cut, a park that stopped reporting long ago, is not read at all. + A file that lies WHOLLY inside the cut, as every anonymous 7-day file does, + is downloaded again instead: None, with both files removed. Trimmed, it + would be empty with a state continuing from a day the key may no longer + read, which a continued file is not allowed to skip past. + The new state is written straight away, so a run that fails after this point does not trim the same file twice. """ @@ -861,7 +868,10 @@ def _upgrade_v1( keep_through = (date.fromisoformat(str(end)) - timedelta(days=V1_UNSETTLED_DAYS)).isoformat() last_day = state.get("lastDay") if last_day is None or str(last_day) > keep_through: - _trim_after(out_path, fmt, keep_through) + if _trim_after(out_path, fmt, keep_through) == 0: + out_path.unlink(missing_ok=True) + state_path.unlink(missing_ok=True) + return None upgraded["lastDay"] = keep_through if last_day is not None else None if state.get("complete"): upgraded["end"] = keep_through @@ -873,9 +883,11 @@ def _upgrade_v1( return upgraded -def _trim_after(out_path: Path, fmt: str, keep_through: str) -> None: +def _trim_after(out_path: Path, fmt: str, keep_through: str) -> int: """Remove every row dated after `keep_through`, leaving the rest byte for byte. + Returns how many dated rows were kept. + Streamed into a file beside the original and swapped in with one rename, so an interruption leaves either the old file or the new one, never half of each. A line or record that cannot be read is kept: this command does not @@ -886,12 +898,14 @@ def _trim_after(out_path: Path, fmt: str, keep_through: str) -> None: """ scratch = out_path.with_name(out_path.name + ".trimming") with out_path.open(encoding="utf-8", newline="") as src: - _copy_rows_through(src, scratch, fmt, keep_through) + kept = _copy_rows_through(src, scratch, fmt, keep_through) os.replace(scratch, out_path) + return kept -def _copy_rows_through(src: TextIO, scratch: Path, fmt: str, keep_through: str) -> None: +def _copy_rows_through(src: TextIO, scratch: Path, fmt: str, keep_through: str) -> int: """The body of `_trim_after`: copy every row dated on or before `keep_through`.""" + kept = 0 with scratch.open("w", encoding="utf-8", newline="") as dst: if fmt == "csv": reader = csv.reader(src) @@ -902,8 +916,10 @@ def _copy_rows_through(src: TextIO, scratch: Path, fmt: str, keep_through: str) # No `date` column: not a file this can read, so it is copied whole. column = names.index("date") if "date" in names else -1 for record in reader: - if 0 <= column < len(record) and record[column] > keep_through: - continue + if 0 <= column < len(record): + if record[column] > keep_through: + continue + kept += 1 dst.write(",".join(_csv_cell(cell) for cell in record) + "\n") else: for line in src: @@ -911,9 +927,12 @@ def _copy_rows_through(src: TextIO, scratch: Path, fmt: str, keep_through: str) day = json.loads(line).get("date") except (ValueError, AttributeError): day = None - if isinstance(day, str) and day > keep_through: - continue + if isinstance(day, str): + if day > keep_through: + continue + kept += 1 dst.write(line) + return kept class _Job(NamedTuple): @@ -925,6 +944,8 @@ class _Job(NamedTuple): end: Day has_rows: bool ident: _RowIdentity + #: True when the file is being CONTINUED, so its next day is fixed. + resumed: bool = False class _Progress: @@ -943,6 +964,8 @@ def __init__(self) -> None: self.last_day: date | None = None self.resume_from: str | None = None self.skipped = False + #: The key cannot read the day a continued file carries on from. + self.out_of_reach = False #: The day this run's request started on, which is where a rerun has to #: carry on if the run fails before its first page is finished. self.first_day: str | None = None @@ -1015,6 +1038,25 @@ def write_rows(writer: Writer, first_day: Day) -> None: # boundary, and restarting would duplicate rows. if floor is None or progress.written: raise + # A FILE BEING CONTINUED CANNOT JUMP FORWARD. Starting at the key's + # first day instead of the day the file continues from leaves a gap + # the state file cannot describe, so the file would claim days it + # does not hold. It happens when a cron has not run for longer than + # the key's window, or the key lost its plan. Refused, file untouched. + if job.resumed: + if start is None or floor <= str(start): + raise + print( + f" this key reaches back to {floor}, but {job.out_path.name} " + f"continues from {start}: the days between are out of reach, and " + f"carrying on from {floor} would leave a gap in the file. The rows " + f"already downloaded are left alone.\n" + f" --overwrite start the file again from what this key can read\n" + f" or pass a different --out and run again", + file=sys.stderr, + ) + progress.out_of_reach = True + return print( f" this key reaches back to {floor}, not {start} — starting there", file=sys.stderr, @@ -1059,7 +1101,13 @@ def _nothing_written( On a first run both files go: an empty file reads as "this park has no history". On a RESUMED run an earlier run's rows are real and are not ours to remove, so the state is kept and the exit code says the run did not finish. + + A continued file the key can no longer reach the next day of is refused + outright: nothing written and the state untouched, so the next run meets the + same refusal until someone decides, rather than carrying on with a gap. """ + if progress.out_of_reach: + return 1 if resumed: _record(sf, progress.last_day, progress.resume_point(), complete=False) return 1 @@ -1154,7 +1202,7 @@ def backfill_park( # noqa: PLR0913 - window is keyword-only, added without brea return _window_closed(out_path, end, start, resumed=resumed_run) ident = _RowIdentity(park) - job = _Job(history, out_path, fmt, end, has_rows, ident) + job = _Job(history, out_path, fmt, end, has_rows, ident, resumed=resumed_run) progress = _Progress() try: _stream(job, start, progress) @@ -1194,7 +1242,7 @@ def backfill_park( # noqa: PLR0913 - window is keyword-only, added without brea out_path.unlink(missing_ok=True) raise - if progress.skipped and progress.written == 0: + if progress.out_of_reach or (progress.skipped and progress.written == 0): return _nothing_written(out_path, state_path, sf, progress, resumed=resumed_run) # Completion is RECORDED, never inferred from a missing file. That is the @@ -1460,7 +1508,13 @@ def _iso_day(value: str) -> str: --since applies when a file is started. A later run continues that file forward and accepts the same --since, or a later one. One earlier than the file's first day, or past the day it continues from, is refused: use --overwrite, or a -different --out. +different --out. A --since before your plan's window starts the file at the +first day your key can read, and the same --since keeps working on every later +run. + +a file is never continued past a gap. If the day it continues from is older than +your key can read (a cron that missed more days than your window, or a plan that +lapsed), the run is refused and the file left alone. exit codes: 0 done From bdbf80b3a1362878456f2164b22285d7a55d615a Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Mon, 28 Sep 2026 22:09:01 +0100 Subject: [PATCH 6/9] docs: a continued file is never carried past a gap; old files inside the cut Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 15 ++++++++++++--- README.md | 11 +++++++++-- 2 files changed, 21 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9475c61..e17944d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -52,9 +52,18 @@ which of their newest days were final, so the first run of this version removes the rows from the last seven days before that run's end and fetches those days again. Every other row is left byte for byte as it was. A 4.0 file - whose newest row is older than that is not rewritten at all. The state file - format moves to version 2 for this; version 1 files from this SDK are upgraded, - not refused. + whose newest row is older than that is not rewritten at all, and one that lies + wholly inside those seven days, as every anonymous 7-day file does, is simply + downloaded again. The state file format moves to version 2 for this; version 1 + files from this SDK are upgraded, not refused. + +- **A continued file could skip ahead to the key's first day.** When the day a + file continues from is older than the key may read, because a cron missed more + days than the window or a plan lapsed, the run carried on from the key's first + day and left a gap the state file did not record. It is refused now with exit + 1, the file and its state untouched, and a message naming `--overwrite`. A + fixed `--since` older than the window is not affected: the file starts at the + key's first day, and the same command line keeps working every night. - **A finished file written to a different column layout was appended to.** Only an unfinished one was refused. A finished one fell through to a fresh start, diff --git a/README.md b/README.md index b14dd0d..a0df218 100644 --- a/README.md +++ b/README.md @@ -366,12 +366,19 @@ in the file is one that will not change later. `--since` applies when a file is started. Later runs continue that file and accept the same `--since`, or a later one, such as a cron line computing "30 days ago". One earlier than the file's first day, or one that would leave a gap, -is refused rather than ignored: pass `--overwrite`, or a different `--out`. +is refused rather than ignored: pass `--overwrite`, or a different `--out`. A +fixed `--since` older than your key's window starts the file at the first day +your key can read, and the same command line keeps working every night. + +A file is never continued past a gap. If the day it would continue from is +older than your key can read, because a cron missed more days than your window +or a plan lapsed, the run is refused with exit 1 and the file is left alone. Files written by 4.0.x ended on today, so their newest rows can be partial. The first run of this version removes the rows from the last week of such a file and fetches those days again, final this time. Everything else in the file is left -exactly as it was. +exactly as it was. A 4.0 file that lies wholly inside that week, as every +anonymous 7-day file does, is simply downloaded again. `python -m themeparks.backfill` is the same thing, which is the one to use if `pip install --user` put the script somewhere off your PATH. `themeparks-backfill From c940bd70931415012663c0a2e25a76deffdbd233 Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Tue, 29 Sep 2026 08:57:04 +0100 Subject: [PATCH 7/9] feat(history): close()/aclose() on changes(); export HistoryOpening HistoryChanges.close() and AsyncHistoryChanges.aclose() end iteration early, as they did on the generators these replaced. HistoryOpening is exported so callers can annotate against it, and the docstring explains `degraded` and `observedAt`. final_through no longer promises the day will never change: it is the newest day the archive has recorded, the place to stop if each day is fetched once. The server can re-record a past day after a feed repair. Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/api/history.md | 4 ++++ tests/unit/test_history_opening.py | 30 ++++++++++++++++++++++++++++++ themeparks/__init__.py | 2 ++ themeparks/_ergonomic/history.py | 25 ++++++++++++++++++++++--- 4 files changed, 58 insertions(+), 3 deletions(-) diff --git a/docs/api/history.md b/docs/api/history.md index acb1f49..d6da4f2 100644 --- a/docs/api/history.md +++ b/docs/api/history.md @@ -23,6 +23,10 @@ hundred times fewer calls than the same data fetched ride by ride. options: heading_level: 2 +::: themeparks.HistoryOpening + options: + heading_level: 2 + ::: themeparks.HistorySpan options: heading_level: 2 diff --git a/tests/unit/test_history_opening.py b/tests/unit/test_history_opening.py index 8e7a072..887f6e0 100644 --- a/tests/unit/test_history_opening.py +++ b/tests/unit/test_history_opening.py @@ -21,6 +21,7 @@ import httpx import pytest +import themeparks from themeparks import AsyncThemeParks, ThemeParks from themeparks._ergonomic.history import HistoryChanges from themeparks._generated.models import HistoryOpening @@ -208,3 +209,32 @@ async def test_opening_before_the_response_says_how_to_get_it(self) -> None: changes = tp.entity(SPACE_MOUNTAIN).history.changes("2026-09-26") with pytest.raises(RuntimeError, match="load"): _ = changes.opening + + +class TestClosing: + def test_close_ends_iteration_without_a_request(self) -> None: + seen: list[str] = [] + changes = _client(RAW, seen).entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + changes.close() + assert list(changes) == [] + assert seen == [] + + def test_close_part_way_stops_there(self) -> None: + changes = _client(RAW).entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + next(changes) + changes.close() + assert list(changes) == [] + + async def test_aclose_ends_async_iteration(self) -> None: + async with _async_client(RAW) as tp: + changes = tp.entity(SPACE_MOUNTAIN).history.changes("2026-09-26") + await changes.__anext__() + await changes.aclose() + assert [pair async for pair in changes] == [] + + +def test_history_opening_is_exported() -> None: + # The type of `opening`'s values, importable to annotate against. + assert themeparks.HistoryOpening is HistoryOpening + assert "HistoryOpening" in themeparks.__all__ + assert "degraded" in HistoryOpening.model_fields diff --git a/themeparks/__init__.py b/themeparks/__init__.py index 406d430..68f6124 100644 --- a/themeparks/__init__.py +++ b/themeparks/__init__.py @@ -17,6 +17,7 @@ ThemeParksError, TimeoutError, ) +from themeparks._generated.models import HistoryOpening from themeparks._ratelimit import RateLimit, RateLimits from themeparks._transport import RetryConfig @@ -27,6 +28,7 @@ "BudgetExhaustedError", "EntityRef", "HistoryChanges", + "HistoryOpening", "HistoryPage", "HistorySpan", "RateLimit", diff --git a/themeparks/_ergonomic/history.py b/themeparks/_ergonomic/history.py index f9dbec8..7666d4b 100644 --- a/themeparks/_ergonomic/history.py +++ b/themeparks/_ergonomic/history.py @@ -70,15 +70,20 @@ class HistorySpan(NamedTuple): @property def final_through(self) -> _date | None: - """The newest day whose daily rows will not change again, or None. + """The newest day the archive has recorded that this key may read, or None. `retrievable_through` is usually today, and today's row is the day so far. Recent days can still change after that: a run that crosses midnight is reported on the day it started, and the archive records days 2 to 3 behind live data. `recorded_to` is the newest day the - archive holds, so a day on or before it is final. Store those; ask for + archive holds, so a day on or before it is what the server has + recorded, and is the place to stop if you fetch each day once. Ask for anything later again once `final_through` has moved past it. + Recorded is not immutable: the server can re-record a past day, for + example after repairing a park's feed. Fetch a range again if you need + to pick that up. + The earlier of the two dates, because a key may be entitled to fewer days than the archive holds. None when either is unknown. """ @@ -248,7 +253,13 @@ class HistoryChanges(Iterator[tuple[str, HistoryRow]]): Nothing is requested until the result is first used, as when this was a plain generator, and reading `opening` before or after iterating costs the - same single request. + same single request. `close()` ends it early, as it did a generator. + + An opening can be incomplete: `degraded` is true when the server could not + look far enough back in time for this response, and `degradedReason` says + why. Fields may then be missing from it; ask again a minute later for the + full opening. `observedAt` is when that state was last seen, which can be + long before the range for an entity whose feed stopped. """ def __init__(self, fetch: Callable[[], RawEnvelope]) -> None: @@ -274,6 +285,10 @@ def __next__(self) -> tuple[str, HistoryRow]: self._rows = _raw_rows(self._loaded()) return next(self._rows) + def close(self) -> None: + """Stop iterating. Later `next()` raises StopIteration; no request is made.""" + self._rows = iter(()) + class AsyncHistoryChanges(AsyncIterator[tuple[str, HistoryRow]]): """What :meth:`AsyncHistoryApi.changes` returns. See :class:`HistoryChanges`. @@ -319,6 +334,10 @@ async def __anext__(self) -> tuple[str, HistoryRow]: except StopIteration: raise StopAsyncIteration from None + async def aclose(self) -> None: + """Stop iterating, as `aclose()` did on the async generator this replaced.""" + self._rows = iter(()) + class HistoryApi: """History for one entity id, reached as ``tp.entity(id).history``.""" From bc6d95c00deb31f682456b4cd12b04758e0d3d6e Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Tue, 29 Sep 2026 08:57:04 +0100 Subject: [PATCH 8/9] fix(backfill): checkpoint every page so an interrupted run never duplicates a day An interrupted nightly extension appended the same days twice. The state was written only at the end of a run or on an SDK error, so Ctrl-C, SIGTERM or SIGKILL during an extension left `complete: true` with the old end, and the rerun appended those days again (70 duplicate rows, exit 0). The state is now written before the first request and after every page, with the data file's size at that moment, and the rerun first truncates the file to that size. Anything written after the last checkpoint is discarded and fetched again, so stopping at any point costs at most the page in flight. A file shorter than its checkpoint is refused. State writes are atomic (temp file and rename). Also, each with a test: - `start` is the first day actually written (after the key's window floor) and a new `since` field keeps the start asked for. The same --since keeps working; a different one before the file's first day is refused. - A partial first page resumed from `lastDay`, the newest day ANY entity reached, losing the days of entities behind it. Checkpoints remove the case for new files; an older state with only `lastDay` goes back a whole page. - SIGTERM stops like Ctrl-C and exits 143. - An advisory lock refuses a second run on the same park and --out (POSIX). - A failed trim removes its scratch copy. - An unfinished file with --until before its resume point says so. - The anonymous notice says only final days are written, usually 4 or 5. The committed mutant list gains 14 mutants, including the four a review found surviving; 43/43 are killed. Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/mutation/mutants.json | 120 +++++- tests/unit/test_backfill.py | 36 +- tests/unit/test_backfill_incremental.py | 342 +++++++++++++++- themeparks/backfill.py | 501 ++++++++++++++++++------ 4 files changed, 836 insertions(+), 163 deletions(-) diff --git a/tests/mutation/mutants.json b/tests/mutation/mutants.json index 989687f..e6e36fe 100644 --- a/tests/mutation/mutants.json +++ b/tests/mutation/mutants.json @@ -43,8 +43,8 @@ { "name": "a resumed run failing any OTHER way deletes the accumulated file", "file": "themeparks/backfill.py", - "find": " if progress.written == 0 and not resumed_run:\n out_path.unlink(missing_ok=True)\n raise", - "replace": " if progress.written == 0:\n out_path.unlink(missing_ok=True)\n raise", + "find": " if progress.written == 0 and not resumed_run:\n out_path.unlink(missing_ok=True)\n state_path.unlink(missing_ok=True)\n raise", + "replace": " if progress.written == 0:\n out_path.unlink(missing_ok=True)\n state_path.unlink(missing_ok=True)\n raise", "why": "`written` counts rows THIS process wrote. On a resumed run the file holds everything the previous runs fetched." }, { @@ -173,13 +173,6 @@ "replace": "if True:", "why": "4.0.1's behaviour: a rerun of a finished park printed that it was complete and exited 0, so a nightly cron never fetched another day." }, - { - "name": "a rerun interrupted before its first page forgets where it began", - "file": "themeparks/backfill.py", - "find": " return self.first_day\n", - "replace": " return None\n", - "why": "With no page boundary and no row, the state recorded nothing, and the next run fell back to the top of the archive and appended a second copy of it." - }, { "name": "a finished state written to another contract is appended to", "file": "themeparks/backfill.py", @@ -197,8 +190,8 @@ { "name": "a 4.0 file keeps its partial tail", "file": "themeparks/backfill.py", - "find": "if _trim_after(out_path, fmt, keep_through) == 0:", - "replace": "if False:", + "find": " _trim_after(out_path, fmt, keep_through)\n upgraded", + "replace": " upgraded", "why": "The partial days 4.0 wrote stay in the file for good, next to nothing that would ever correct them." }, { @@ -222,19 +215,110 @@ "replace": "return {}", "why": "The opening is the state before the first change; without it a day cannot be rebuilt from raw history." }, - { - "name": "a 4.0 file wholly inside the cut is trimmed to nothing", - "file": "themeparks/backfill.py", - "find": "if _trim_after(out_path, fmt, keep_through) == 0:", - "replace": "if _trim_after(out_path, fmt, keep_through) == -1:", - "why": "Every anonymous 4.0 file: an empty file whose state continues from a day the key can no longer read, refused on every later run." - }, { "name": "a continued file jumps to the key's first day", "file": "themeparks/backfill.py", "find": " if job.resumed:\n", "replace": " if False:\n", "why": "A cron that missed more nights than the key's window carries on from the key's first day, leaving days missing from the middle of a file whose state claims them." + }, + { + "name": "the NDJSON trim drops the cut day itself", + "file": "themeparks/backfill.py", + "find": "if isinstance(day, str) and day > keep_through:", + "replace": "if isinstance(day, str) and day >= keep_through:", + "why": "The day at the cut is one the state says the file holds; dropping it leaves a hole." + }, + { + "name": "a 403 at exactly the resume day is reported as a gap", + "file": "themeparks/backfill.py", + "find": "if start is None or floor <= str(start):", + "replace": "if start is None or floor < str(start):", + "why": "The key reaches the day the file continues from, so the refusal is an error to surface, not a gap." + }, + { + "name": "the upgraded 4.0 state is not written", + "file": "themeparks/backfill.py", + "find": " _write_state(state_path, **upgraded)\n return upgraded", + "replace": " return upgraded", + "why": "A run that stops right after the upgrade leaves a trimmed file under a version-1 state, trimmed again next time." + }, + { + "name": "the upgrade keeps the old lastDay", + "file": "themeparks/backfill.py", + "find": "upgraded[\"lastDay\"] = keep_through if last_day is not None else None", + "replace": "pass", + "why": "The state would claim a day the trim just removed." + }, + { + "name": "no checkpoint after a page", + "file": "themeparks/backfill.py", + "find": "progress.checkpoint(job.state, _size(handle), progress.resume_from)", + "replace": "pass", + "why": "The defect a review found: an interrupted extension left the old state, and the rerun appended the same days again." + }, + { + "name": "no checkpoint before the first request", + "file": "themeparks/backfill.py", + "find": "progress.checkpoint(job.state, _size(handle), None if start is None else str(start))", + "replace": "pass", + "why": "An extension killed mid-first-page left `complete: true` with the old end, and the rerun appended those rows twice." + }, + { + "name": "the rerun does not cut back to the checkpoint", + "file": "themeparks/backfill.py", + "find": " handle.truncate(size)\n", + "replace": " pass\n", + "why": "Whatever was written after the last checkpoint is appended again after it." + }, + { + "name": "the state file is written in place", + "file": "themeparks/backfill.py", + "find": "scratch.write_text(json.dumps(fields, sort_keys=True) + \"\\n\", encoding=\"utf-8\")\n os.replace(scratch, path)", + "replace": "path.write_text(json.dumps(fields, sort_keys=True) + \"\\n\", encoding=\"utf-8\")", + "why": "A crash or a full disk mid-write leaves half a state file, which reads as none." + }, + { + "name": "a failed trim leaves its scratch copy", + "file": "themeparks/backfill.py", + "find": " except BaseException:\n scratch.unlink(missing_ok=True)\n raise\n return kept", + "replace": " except BaseException:\n raise\n return kept", + "why": "A full disk during a trim leaves a second partial copy of the data file behind." + }, + { + "name": "a legacy state resumes at lastDay again", + "file": "themeparks/backfill.py", + "find": "back = _days_before(last_day, PAGE_DAYS)", + "replace": "back = _days_before(last_day, 1)", + "why": "lastDay is the newest day ANY entity reached; the entities behind it lose the days in between." + }, + { + "name": "the state records the start asked for, not the first day written", + "file": "themeparks/backfill.py", + "find": " progress.file_start = first_day\n", + "replace": " pass\n", + "why": "After the window clamp, a --since earlier than the real first day passed silently." + }, + { + "name": "the same --since before the file's first day is refused", + "file": "themeparks/backfill.py", + "find": " and str(since) != str(asked)\n", + "replace": "", + "why": "A cron line with a fixed --since older than the key's window fails every night after the first." + }, + { + "name": "two runs on one park are not locked out", + "file": "themeparks/backfill.py", + "find": "fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)", + "replace": "pass", + "why": "Two runs read the same state and append the same days." + }, + { + "name": "a 4.0 file emptied by the cut is continued, not started again", + "file": "themeparks/backfill.py", + "find": " if size == 0:\n # Nothing survived the cut", + "replace": " if False:\n # Nothing survived the cut", + "why": "Every anonymous 4.0 file: continued from a day the key can no longer read, and refused on every later run." } ] } diff --git a/tests/unit/test_backfill.py b/tests/unit/test_backfill.py index 26d5791..4cebf12 100644 --- a/tests/unit/test_backfill.py +++ b/tests/unit/test_backfill.py @@ -486,21 +486,22 @@ def test_returns_ex_tempfail_so_a_scheduler_retries(self, tmp_path: Path) -> Non code = backfill.backfill_park(_Client(hist), _Park("p", "P"), tmp_path, "ndjson") assert code == backfill.EX_TEMPFAIL == 75 - def test_records_the_furthest_day_so_a_rerun_continues(self, tmp_path: Path) -> None: + def test_a_budget_spent_mid_first_page_resumes_from_the_last_checkpoint( + self, tmp_path: Path + ) -> None: hist = self._hist_that_runs_out() backfill.backfill_park(_Client(hist), _Park("p", "P"), tmp_path, "ndjson") state_file = backfill.state_path_for(tmp_path, "p", "ndjson") state = json.loads(state_file.read_text(encoding="utf-8")) assert state["complete"] is False - # The MAX day seen, not the last row yielded. The stub's final row is - # 2025-02-14, so recording "last" instead of "max" would rewind the resume - # point by three weeks and re-download them. - assert state["lastDay"] == "2025-03-09" - # The budget died part-way through the FIRST page, so no page boundary was - # ever reported and there is nothing exact to resume from. `last_day` is - # the documented fallback for precisely this: one duplicated day, rather - # than starting from the top and appending a second copy of everything. - assert state["resumeFrom"] is None + # The budget died part-way through the FIRST page, so the last checkpoint + # is the one written before the first request: the run's own start, and + # a file of 0 bytes. The rerun cuts the partial page off and starts it + # again. This used to resume at `lastDay`, the newest day ANY entity had + # reached -- 2025-03-09 here, while ent-2 had only got to 2025-02-14 -- + # which duplicated one day and lost the slower entities' days for good. + assert state["resumeFrom"] == "2025-01-01" + assert state["size"] == 0 def test_a_spent_budget_on_the_coverage_call_also_returns_75(self, tmp_path: Path) -> None: # span() is the FIRST request a resumed run makes, while the hourly window @@ -606,18 +607,19 @@ def test_a_rerun_asks_for_the_page_boundary_not_the_newest_row(self, tmp_path: P backfill.backfill_park(_Client(hist), _Park("p", "P"), tmp_path, "ndjson") assert hist.calls == ["2025-02-01"], "resumed from the newest row, not the boundary" - def test_falls_back_to_last_day_for_a_state_file_without_a_boundary( - self, tmp_path: Path - ) -> None: - # 3.3.0 wrote no resume_from, and a run that dies inside its first page - # never reports one. One duplicated day beats starting from the top and - # appending a second copy of the whole archive. + def test_a_state_file_without_a_boundary_goes_back_a_whole_page(self, tmp_path: Path) -> None: + # 3.3.0 wrote no resume_from, and a 4.0 run that died inside its first + # page never reported one. Only `lastDay` is left. (tmp_path / "p.ndjson").write_text('{"a": 1}\n', encoding="utf-8") # No boundary recorded: a run that died inside its first page. _state_file(tmp_path, lastDay="2025-01-28", resumeFrom=None) hist = _History(archive_from="2025-01-01", through="2026-09-28", floor=None) backfill.backfill_park(_Client(hist), _Park("p", "P"), tmp_path, "ndjson") - assert hist.calls == ["2025-01-28"] + # A page is up to 31 days, so the page that run died in began no earlier + # than 30 days before `lastDay`; resuming there, clamped to the file's own + # first day, loses no entity's days. Resuming AT `lastDay` lost the days + # of every entity that had not reached it. + assert hist.calls == ["2025-01-01"] def test_the_boundary_is_read_from_the_servers_own_url(self) -> None: # No date arithmetic anywhere: the server says where the next page starts diff --git a/tests/unit/test_backfill_incremental.py b/tests/unit/test_backfill_incremental.py index 4e5e100..b2d5804 100644 --- a/tests/unit/test_backfill_incremental.py +++ b/tests/unit/test_backfill_incremental.py @@ -28,6 +28,8 @@ import inspect import io import json +import os +import signal as sig from collections import Counter from datetime import date, timedelta from pathlib import Path @@ -86,6 +88,14 @@ def __init__( self.calls: list[tuple[str, str]] = [] #: Raise BudgetExhaustedError on the Nth page request (1-based), or never. self.budget_on_page: int | None = None + #: Raise `interrupt` after this many rows have been yielded, or never. + #: A BaseException, so nothing in the command can catch it: this is + #: Ctrl-C, SIGTERM, or (with no handler running at all) SIGKILL. + self.interrupt_after_rows: int | None = None + self.interrupt: type[BaseException] = KeyboardInterrupt + #: Raise OSError (a full disk) after this many rows, or never. + self.disk_full_after_rows: int | None = None + self._rows_served = 0 self._pages_served = 0 def advance(self, days: int) -> None: @@ -112,6 +122,11 @@ def days_with_entities(self, start=None, end=None, *, max_wait=120.0, on_page=No ref = EntityRef(entity, f"Ride {entity}", "ATTRACTION") day = page_start while day <= page_end: + if self._rows_served == self.interrupt_after_rows: + raise self.interrupt + if self._rows_served == self.disk_full_after_rows: + raise OSError(28, "No space left on device") + self._rows_served += 1 yield ref, _row(day, final=day <= _day(self.recorded_to)) day += timedelta(days=1) following = page_end + timedelta(days=1) @@ -683,7 +698,13 @@ def test_a_continued_file_never_jumps_to_the_keys_first_day( assert _run(archive, tmp_path) == 1 assert archive.calls[-1][0] == "2026-09-27", "asked for anything but its own next day" assert (tmp_path / "p.ndjson").read_bytes() == before - assert _state(tmp_path) == state_before + # Marked unfinished before the request, continuing from the same day, + # vouching for the same bytes: nothing about the file moved. + state = _state(tmp_path) + assert state["complete"] is False + assert state["resumeFrom"] == "2026-09-27" + assert state["size"] == state_before["size"] + assert state["start"] == state_before["start"] err = capsys.readouterr().err assert "reaches back to 2026-10-05" in err assert "continues from 2026-09-27" in err @@ -804,3 +825,322 @@ def test_a_csv_line_round_trips_through_the_csv_module() -> None: line = backfill._csv_line(cells) parsed = next(csv.reader(io.StringIO(line, newline=""))) assert ",".join(backfill._csv_cell(v) for v in parsed) + "\n" == line + + +# -------------------------------------------------------------------------- +# Interruptions: Ctrl-C, SIGTERM, SIGKILL, a full disk. +# -------------------------------------------------------------------------- + + +class _Kill(BaseException): + """Stands in for SIGKILL: no handler in the command can catch it.""" + + +class TestAnInterruptedRunNeverDuplicatesADay: + """The state was written only at the end or on an SDK error. + + So Ctrl-C, SIGTERM or SIGKILL during a nightly extension left the state + saying `complete: true` with the OLD end, and the rerun appended the same + days again: 70 duplicate rows, exit 0. The state is now checkpointed after + every page with the file's size at that moment, and a rerun first cuts the + file back to that size, so whatever was written after the last checkpoint is + discarded and fetched again. No day is ever appended twice. + """ + + @pytest.mark.parametrize("interrupt", [KeyboardInterrupt, SystemExit, _Kill]) + @pytest.mark.parametrize("after_rows", [0, 1, 31, 45, 62, 70]) + def test_an_extension_interrupted_anywhere_reruns_cleanly( + self, tmp_path: Path, interrupt: type[BaseException], after_rows: int + ) -> None: + archive = _Archive() + _run(archive, tmp_path) + archive.advance(40) # two pages of new days, 80 rows + archive.interrupt = interrupt + archive.interrupt_after_rows = archive._rows_served + after_rows + with pytest.raises(interrupt): + _run(archive, tmp_path) + + archive.interrupt_after_rows = None + assert _run(archive, tmp_path) == 0 + rows = _rows(tmp_path) + _assert_every_row_final_and_unique(rows) + assert max(r["date"] for r in rows) == "2026-11-05" + days = {r["date"] for r in rows} + assert len(days) == (date(2026, 11, 5) - date(2026, 6, 1)).days + 1, "a day went missing" + + @pytest.mark.parametrize("fmt", ["ndjson", "csv"]) + def test_a_first_run_interrupted_mid_page_resumes_without_duplicates( + self, tmp_path: Path, fmt: str + ) -> None: + archive = _Archive() + archive.interrupt_after_rows = 100 # inside the second page + with pytest.raises(KeyboardInterrupt): + _run(archive, tmp_path, fmt) + archive.interrupt_after_rows = None + assert _run(archive, tmp_path, fmt) == 0 + _assert_every_row_final_and_unique(_rows(tmp_path, fmt)) + if fmt == "csv": + raw = (tmp_path / "p.csv").read_bytes() + assert raw.count(b"\xef\xbb\xbf") == 1 + assert raw.decode("utf-8-sig").count("parkId,") == 1 + + def test_a_first_run_interrupted_before_its_first_row_starts_again( + self, tmp_path: Path + ) -> None: + # Nothing was written, so there is nothing to continue: no refusal over + # a file "this command did not write", and no empty file left behind. + archive = _Archive() + archive.interrupt_after_rows = 0 + with pytest.raises(KeyboardInterrupt): + _run(archive, tmp_path) + archive.interrupt_after_rows = None + assert _run(archive, tmp_path) == 0 + _assert_every_row_final_and_unique(_rows(tmp_path)) + + def test_the_state_is_checkpointed_after_every_page(self, tmp_path: Path) -> None: + archive = _Archive() + seen: list[dict] = [] + original = archive.days_with_entities + + def spying(start=None, end=None, *, max_wait=120.0, on_page=None): + def hook(page: HistoryPage) -> None: + on_page(page) + seen.append(_state(tmp_path)) + + yield from original(start, end, max_wait=max_wait, on_page=hook) + + archive.days_with_entities = spying # type: ignore[method-assign] + _run(archive, tmp_path) + assert [s["resumeFrom"] for s in seen[:-1]] == ["2026-07-02", "2026-08-02", "2026-09-02"] + sizes = [s["size"] for s in seen[:-1]] + assert sizes == sorted(sizes) and len(set(sizes)) == len(sizes) + # The last page has no next day to record; completion does that. + assert all(s["complete"] is False for s in seen) + final = _state(tmp_path) + assert final["complete"] is True + assert final["size"] == (tmp_path / "p.ndjson").stat().st_size + + def test_a_file_shorter_than_its_checkpoint_is_refused(self, tmp_path: Path, capsys) -> None: + # Something other than this command cut the file. Appending would leave + # a hole the state claims is filled. + archive = _Archive() + _run(archive, tmp_path) + path = tmp_path / "p.ndjson" + path.write_bytes(path.read_bytes()[:-10]) + archive.advance(1) + assert _run(archive, tmp_path) == 1 + assert "--overwrite" in capsys.readouterr().err + + +class TestAPartialFirstPageLosesNoEntity: + """A full disk mid-page resumed from the newest day of ANY entity. + + `lastDay` is the maximum across entities, and rows arrive entity by entity, + so entity A could be written through day 20 while entity B had only reached + day 3; resuming at day 20 lost B's days 4 to 19 for good. + """ + + def test_a_disk_full_mid_first_page_loses_nothing(self, tmp_path: Path) -> None: + archive = _Archive() + archive.disk_full_after_rows = 40 # ent-a's 31 days, then 9 of ent-b's + with pytest.raises(OSError): + _run(archive, tmp_path) + archive.disk_full_after_rows = None + assert _run(archive, tmp_path) == 0 + rows = _rows(tmp_path) + _assert_every_row_final_and_unique(rows) + for entity in ("ent-a", "ent-b"): + days = {r["date"] for r in rows if r["entityId"] == entity} + assert len(days) == (date(2026, 9, 26) - date(2026, 6, 1)).days + 1, entity + + def test_a_legacy_state_with_only_last_day_resumes_a_whole_page_back( + self, tmp_path: Path + ) -> None: + # A 4.0 file interrupted in its first page recorded only lastDay. The + # page could have started up to 30 days earlier, so that is where the + # rerun goes, after removing every row from there on. + archive = _Archive() + lines = [] + for entity, last in (("ent-a", "2026-07-20"), ("ent-b", "2026-07-03")): + day = date(2026, 6, 1) + while day <= date.fromisoformat(last): + lines.append( + json.dumps( + {"entityId": entity, "date": day.isoformat(), "operatingMinutes": 600} + ) + ) + day += timedelta(days=1) + (tmp_path / "p.ndjson").write_text("\n".join(lines) + "\n", encoding="utf-8") + _v1_state(tmp_path, "ndjson", complete=False, lastDay="2026-07-20", resumeFrom=None) + assert _run(archive, tmp_path) == 0 + assert archive.calls[-1][0] == "2026-06-20" + rows = _rows(tmp_path) + assert len(_keys(rows)) == len(rows) + for entity in ("ent-a", "ent-b"): + days = {r["date"] for r in rows if r["entityId"] == entity} + assert len(days) == (date(2026, 9, 26) - date(2026, 6, 1)).days + 1, entity + + +class TestTheStateRecordsTheFilesRealFirstDay: + def test_after_the_window_clamp_start_is_the_first_day_written(self, tmp_path: Path) -> None: + archive = _Archive(archive_from="2025-06-01", floor="2026-09-01") + _run(archive, tmp_path, window=_Range(since="2026-02-01")) + state = _state(tmp_path) + assert state["start"] == "2026-09-01" + assert state["since"] == "2026-02-01" + assert min(r["date"] for r in _rows(tmp_path)) == "2026-09-01" + + def test_a_different_since_before_the_real_first_day_is_refused( + self, tmp_path: Path, capsys + ) -> None: + archive = _Archive(archive_from="2025-06-01", floor="2026-09-01") + _run(archive, tmp_path, window=_Range(since="2026-02-01")) + archive.advance(1) + # The key's window moved; this --since asks for days the file never had. + assert _run(archive, tmp_path, window=_Range(since="2026-03-01")) == 1 + err = capsys.readouterr().err + assert "2026-09-01" in err and "--overwrite" in err + + def test_the_same_since_keeps_working(self, tmp_path: Path) -> None: + archive = _Archive(archive_from="2025-06-01", floor="2026-09-01") + _run(archive, tmp_path, window=_Range(since="2026-02-01")) + archive.advance(1) + assert _run(archive, tmp_path, window=_Range(since="2026-02-01")) == 0 + + +class TestTheWindowFloorEdge: + def test_a_403_whose_floor_is_the_resume_day_is_an_error_not_a_gap( + self, tmp_path: Path + ) -> None: + # The key reaches exactly the day the file continues from, yet the API + # refused: not a gap to report, an error to surface. + archive = _Archive() + _run(archive, tmp_path) + archive.advance(1) + + def refuse(start=None, end=None, *, max_wait=120.0, on_page=None): + raise backfill_test_403(str(start)) + yield # pragma: no cover + + archive.days_with_entities = refuse # type: ignore[method-assign] + with pytest.raises(APIError): + _run(archive, tmp_path) + + +class TestTheUpgradeIsRecordedBeforeTheRunGoesOn: + def test_a_run_that_stops_after_the_upgrade_leaves_the_upgraded_state( + self, tmp_path: Path + ) -> None: + # The archive has not moved past the cut, so the run is up to date right + # after upgrading and never writes a checkpoint of its own: the state on + # disk is the upgrade's, or nothing. + archive = _Archive(recorded_to="2026-09-21") + _write_as_4_0(tmp_path, "ndjson", archive, {}) + _v1_state(tmp_path, "ndjson") + + assert _run(archive, tmp_path) == 0 + assert archive.calls == [] + state = _state(tmp_path) + assert state["size"] == (tmp_path / "p.ndjson").stat().st_size + assert state["stateVersion"] == backfill.STATE_VERSION + assert state["end"] == "2026-09-21" + assert state["lastDay"] == "2026-09-21" + assert max(r["date"] for r in _rows(tmp_path)) == "2026-09-21" + + +class TestTrimBoundaries: + @pytest.mark.parametrize("fmt", ["ndjson", "csv"]) + def test_the_cut_day_itself_is_kept(self, tmp_path: Path, fmt: str) -> None: + archive = _Archive( + archive_from="2026-09-19", through="2026-09-23", recorded_to="2026-09-23" + ) + _write_as_4_0(tmp_path, fmt, archive, {}) + backfill._trim_after(tmp_path / f"p.{fmt}", fmt, "2026-09-21") + assert sorted({r["date"] for r in _rows(tmp_path, fmt)}) == [ + "2026-09-19", + "2026-09-20", + "2026-09-21", + ] + + def test_the_scratch_file_is_removed_when_the_trim_fails( + self, tmp_path: Path, monkeypatch + ) -> None: + path = tmp_path / "p.ndjson" + path.write_text('{"date": "2026-09-01"}\n', encoding="utf-8") + + def boom(_src, scratch, *_a): + scratch.write_text('{"date": "2026-09-01"', encoding="utf-8") # half a copy + raise OSError(28, "No space left on device") + + monkeypatch.setattr(backfill, "_copy_rows_through", boom) + with pytest.raises(OSError): + backfill._trim_after(path, "ndjson", "2026-09-21") + assert sorted(p.name for p in tmp_path.iterdir()) == ["p.ndjson"] + + +class TestStateWritesAreAtomic: + def test_a_failed_write_leaves_the_old_state(self, tmp_path: Path, monkeypatch) -> None: + path = tmp_path / "s.json" + backfill._write_state(path, a=1) + + def boom(*_a, **_k): + raise OSError(28, "No space left on device") + + monkeypatch.setattr(backfill.os, "replace", boom) + with pytest.raises(OSError): + backfill._write_state(path, a=2) + assert json.loads(path.read_text(encoding="utf-8")) == {"a": 1} + + +class TestTwoRunsOnOneFile: + def test_a_second_run_on_the_same_park_and_out_is_refused(self, tmp_path: Path, capsys) -> None: + pytest.importorskip("fcntl") + archive = _Archive() + with backfill._park_lock(tmp_path, "p", "ndjson") as held: + assert held + assert _run(archive, tmp_path) == 1 + assert "another themeparks-backfill" in capsys.readouterr().err + assert archive.calls == [] + assert _run(archive, tmp_path) == 0 + + +class TestAnUnfinishedFileAndAnEarlyUntil: + def test_until_before_the_resume_point_says_so(self, tmp_path: Path, capsys) -> None: + archive = _Archive() + archive.budget_on_page = 3 + assert _run(archive, tmp_path) == backfill.EX_TEMPFAIL + archive.budget_on_page = None + capsys.readouterr() + assert _run(archive, tmp_path, window=_Range(until="2026-06-15")) == 0 + err = capsys.readouterr().err + assert "continues from 2026-08-02" in err + assert "--until 2026-06-15" in err + assert "plan no longer reaches" not in err + + +class TestSigterm: + def test_sigterm_stops_like_ctrl_c_with_exit_143(self, capsys, monkeypatch) -> None: + if not hasattr(sig, "SIGTERM") or os.name == "nt": + pytest.skip("POSIX signals only") + + def main(argv=None): + os.kill(os.getpid(), sig.SIGTERM) + raise AssertionError("SIGTERM did not interrupt") # pragma: no cover + + before = sig.getsignal(sig.SIGTERM) + monkeypatch.setattr(backfill, "main", main) + assert backfill.cli() == 143 + assert "run the same command again" in capsys.readouterr().err + assert sig.getsignal(sig.SIGTERM) is before, "the handler was left installed" + + def test_the_anonymous_notice_says_how_many_final_days( + self, tmp_path: Path, capsys, monkeypatch + ) -> None: + monkeypatch.delenv("THEMEPARKS_API_KEY", raising=False) + monkeypatch.setattr(backfill, "_catalogue", lambda tp: [("p", "Park", "d", "Dest")]) + monkeypatch.setattr(backfill, "_run_all", lambda tp, targets, args: 0) + monkeypatch.setattr(backfill, "ThemeParks", lambda **kw: _NullClient()) + backfill.main(["p", "--out", str(tmp_path)]) + err = capsys.readouterr().err + assert err.count("4 or 5") == 2 + assert "final" in err diff --git a/themeparks/backfill.py b/themeparks/backfill.py index 436ca4e..ebcd06c 100644 --- a/themeparks/backfill.py +++ b/themeparks/backfill.py @@ -46,8 +46,10 @@ 5. It writes FINAL days only. Today's row is the day so far, and the archive records days 2 to 3 behind live data, so the newest days the API serves can still change. The run ends at `span().final_through`, the newest day the - archive holds, and the next run carries on from the day after. Every row in - the file is one that will not change, so a nightly run only ever appends. + archive holds, and the next run carries on from the day after. Each day is + fetched once, as the archive recorded it, so a nightly run only ever + appends. The archive can re-record a past day during a repair; the file does + not follow, and `--since`/`--until` into another `--out` fetches it again. """ from __future__ import annotations @@ -59,8 +61,10 @@ import json import os import re +import signal import sys import unicodedata +from collections.abc import Iterator from datetime import date, datetime, timedelta from datetime import date as _date from pathlib import Path @@ -496,7 +500,19 @@ def _read_state(path: Path) -> dict[str, Any]: def _write_state(path: Path, **fields: Any) -> None: - path.write_text(json.dumps(fields, sort_keys=True) + "\n", encoding="utf-8") + """Replace the state file in one step. + + Written beside it and renamed over it, so a crash or a full disk mid-write + leaves the previous state rather than half of a new one. A torn state file + reads as no state, and no state beside a file with rows is a refusal. + """ + scratch = path.with_name(path.name + ".writing") + try: + scratch.write_text(json.dumps(fields, sort_keys=True) + "\n", encoding="utf-8") + os.replace(scratch, path) + except BaseException: + scratch.unlink(missing_ok=True) + raise def _state_mismatch(state: dict[str, Any], fmt: str) -> str | None: @@ -527,20 +543,37 @@ class _StateFile(NamedTuple): path: Path fmt: str - start: Day + #: The first day the file was ASKED to start from: `--since`, or where the + #: archive starts. Kept apart from the first day actually written, which the + #: key's window can push later, so the same `--since` keeps working. + since: Day end: Day -def _record( - sf: _StateFile, last_day: date | None, resume_from: str | None, *, complete: bool -) -> None: +class _Checkpoint(NamedTuple): + """Where a run has got to, as the state file records it.""" + + #: The first day actually written to the file. On a first run the key's + #: window can push it later than `since`. + start: Day + last_day: date | str | None + resume_from: str | None + #: Bytes of the data file this state vouches for. A rerun first cuts the + #: file back to exactly this, so nothing written after the checkpoint (a + #: half page when the run was killed) can ever be appended twice. + size: int + complete: bool + + +def _record(sf: _StateFile, cp: _Checkpoint) -> None: """Write the state file. `complete` is the fact the old checkpoint could not express. - `sf.start` is the ORIGINAL start of the range, not the day a resumed run + `cp.start` is the ORIGINAL first day of the file, not the day a resumed run happened to begin at. The two call sites used to disagree about that, so a run interrupted twice recorded the second resume point as though it were the beginning and lost the real range. """ + last = cp.last_day _write_state( sf.path, sdk=SDK_NAME, @@ -550,11 +583,13 @@ def _record( columns=_columns_fingerprint(sf.fmt), # None, never the string "None". `str(None)` put the literal "None" in # the file where the JavaScript SDK writes null, and "None" is truthy. - start=_day_str(sf.start), + start=_day_str(cp.start), + since=_day_str(sf.since), end=_day_str(sf.end), - lastDay=last_day.isoformat() if last_day else None, - resumeFrom=resume_from, - complete=complete, + lastDay=last.isoformat() if isinstance(last, date) else last, + resumeFrom=cp.resume_from, + size=cp.size, + complete=cp.complete, ) @@ -607,6 +642,7 @@ class _Plan(NamedTuple): start: Day has_rows: bool + #: The first day already in the file, or None on a first run. prior_start: str | None #: True when rows from an EARLIER run are already in the file. Every deletion #: in this module has to consult it: `written == 0` means "this process wrote @@ -614,6 +650,10 @@ class _Plan(NamedTuple): resumed: bool = False #: True when a FINISHED file is being carried forward to new final days. extending: bool = False + #: The first day the file was asked to start from. See `_StateFile.since`. + since: Day = None + #: The newest day already in the file, carried into the next checkpoint. + prior_last_day: str | None = None def _next_day(value: Day) -> str: @@ -621,6 +661,10 @@ def _next_day(value: Day) -> str: return (date.fromisoformat(str(value)) + timedelta(days=1)).isoformat() +def _days_before(value: Day, days: int) -> str: + return (date.fromisoformat(str(value)) - timedelta(days=days)).isoformat() + + def _later(a: Day, b: Day) -> Day: """The later of two days, either of which may be None. ISO days sort as text.""" if a is None: @@ -647,17 +691,29 @@ def _range_fits( A file here is one contiguous range of days, and a run can only append to it, so two requests cannot be honoured without rewriting it: a `--since` before - the file starts, and one after the day it would continue from, which would - leave a gap the state file could not describe. Both are refused rather than - quietly ignored, which would hand back a file that is not what was asked for. - - A `--since` inside the file is fine and common: a cron line with a fixed - `--since` runs it every night, and one computed as "30 days ago" moves - forward every night. Before the archive starts is the same as its start. + the file's first day, and one after the day it would continue from, which + would leave a gap the state file could not describe. Both are refused rather + than quietly ignored, which would hand back a file that is not what was asked + for. + + THE SAME `--since` IS ALWAYS ACCEPTED, even when it is before the file's first + day. On a plan short of the full archive, `--since 2025-01-01` starts the + file at the first day the key can read, and a cron line repeating it every + night must keep working. So a `--since` is judged against the first day + WRITTEN, except that the one the file was asked to start from (`since` in the + state) is always fine. A `--since` inside the file is fine and common too: one + computed as "30 days ago" moves forward every night. Before the archive + starts is the same as its start. """ file_start = state.get("start") + asked = state.get("since") or file_start since = _later(rng.since, archive_from) if rng.since is not None else None - if since is not None and file_start is not None and str(since) < str(file_start): + if ( + since is not None + and file_start is not None + and str(since) < str(file_start) + and str(since) != str(asked) + ): return _refuse( out_path, f"it was started from {file_start}, and --since {rng.since} would need days " @@ -715,9 +771,7 @@ def _decide(out_path: Path, state_path: Path, fmt: str, overwrite: bool, ask: _A state = {} if state and _upgradable(state, fmt): - upgraded = _upgrade_v1(state, out_path, state_path, fmt) - state = upgraded if upgraded is not None else {} - file_exists = upgraded is not None + state = _upgrade_v1(state, out_path, state_path, fmt) # A file we have no record of writing. Refusing is the only safe answer: # appending doubles it, truncating throws away someone's data. @@ -750,11 +804,103 @@ def _decide(out_path: Path, state_path: Path, fmt: str, overwrite: bool, ask: _A ) return 1 + if state: + checked = _back_to_checkpoint(out_path, state_path, state) + if isinstance(checked, int): + return checked + state = checked if not state: return _first_run(out_path, ask) return _continue(out_path, state, ask) +def _back_to_checkpoint( + out_path: Path, state_path: Path, state: dict[str, Any] +) -> dict[str, Any] | int: + """Cut the file back to what the state vouches for: the checkpoint's `size`. + + THIS IS WHAT MAKES A RERUN IDEMPOTENT. The state used to be written only at + the end of a run or on an error the SDK raised, so Ctrl-C, SIGTERM or + SIGKILL during a nightly extension left `complete: true` with the old end, + and the rerun appended the same days again: 70 duplicate rows, exit 0. The + state is now written after every page with the file's size at that moment, + so whatever is past that size was written after the last checkpoint -- half a + page, or a torn line -- and is discarded here, then fetched again. + + A file SHORTER than its checkpoint was changed by something else, and + appending would leave a hole the state claims is filled, so it is refused. + + A state with no `size` predates it. Its file is cut back by date instead, to + the day it continues from, which reads the file once and then records a size. + Returns the state to go on with; an empty dict when nothing is left in the + file, which is a first run. + """ + actual = out_path.stat().st_size + size = state.get("size") + if not isinstance(size, int): + keep_through = _legacy_keep_through(state) + fmt = str(state.get("format")) + if keep_through is not None and _trim_after(out_path, fmt, keep_through) == 0: + out_path.unlink(missing_ok=True) + state_path.unlink(missing_ok=True) + return {} + if keep_through is not None and not state.get("complete"): + state = {**state, "resumeFrom": _next_day(keep_through)} + state = {**state, "size": out_path.stat().st_size} + _write_state(state_path, **state) + return state + if actual < size: + return _refuse( + out_path, + f"it is {actual} bytes, shorter than the {size} its state file records, so " + f"something other than this command changed it", + ) + if actual > size: + print( + f" discarding the last {actual - size} bytes of {out_path.name}: written " + f"after the last checkpoint, and fetched again now", + file=sys.stderr, + ) + with out_path.open("r+b") as handle: + handle.truncate(size) + if size == 0: + # Nothing survived the cut: this is a first run, from the range asked. + state_path.unlink(missing_ok=True) + return {} + return state + + +def _legacy_keep_through(state: dict[str, Any]) -> str | None: + """For a state without `size`: the newest day its file can be trusted to hold. + + A finished file holds every day through `end`. An unfinished one holds every + day before `resumeFrom`, the page boundary; rows from that day on were + written after it, by a run that was then killed. Without a boundary there is + only `lastDay`, the newest day ANY entity reached -- and rows arrive entity by + entity, so another entity may have stopped days earlier. The page that run + was on began at most 30 days before `lastDay` (a page is up to 31 days), so + that is where it is safe to go back to. Resuming at `lastDay` itself, as + before, lost the later entities' days for good. + """ + if state.get("complete"): + end = state.get("end") + return str(end) if end else None + resume = state.get("resumeFrom") + if resume: + return _days_before(resume, 1) + last_day = state.get("lastDay") + start = state.get("start") + if last_day: + # No earlier than the file's own first day: nothing before it exists. + back = _days_before(last_day, PAGE_DAYS) + return back if not start or back >= _days_before(start, 1) else _days_before(start, 1) + return _days_before(start, 1) if start else None + + +#: The most park-local days one page of the park daily endpoint covers. +PAGE_DAYS = 31 + + def _first_run(out_path: Path, ask: _Ask) -> _Plan | int: """A park with no file yet: from `--since`, or wherever the archive starts.""" rng = ask.rng @@ -766,17 +912,25 @@ def _first_run(out_path: Path, ask: _Ask) -> _Plan | int: file=sys.stderr, ) return 0 - return _Plan(start=start, has_rows=False, prior_start=None) + return _Plan(start=start, has_rows=False, prior_start=None, since=start) def _continue(out_path: Path, state: dict[str, Any], ask: _Ask) -> _Plan | int: """A file this command wrote: carry it forward, or say why not.""" rng, archive_from, end = ask.rng, ask.archive_from, ask.end + carried = _Plan( + start=None, + has_rows=True, + prior_start=state.get("start"), + resumed=True, + since=state.get("since") or state.get("start"), + prior_last_day=state.get("lastDay"), + ) if state.get("complete"): # FINISHED IS NOT FOREVER. It used to be: a rerun printed "already # complete" and exited 0 without asking for a single new day, so a # nightly cron looked healthy and never updated. The file holds every - # final day through `end`, so the next day is where it carries on. + # day through `end`, so the next day is where it carries on. recorded_end = state.get("end") continue_at = _next_day(recorded_end) if recorded_end else str(state.get("start")) refused = _range_fits(out_path, state, rng, archive_from, continue_at) @@ -789,36 +943,27 @@ def _continue(out_path: Path, state: dict[str, Any], ask: _Ask) -> _Plan | int: file=sys.stderr, ) return 0 - return _Plan( - start=continue_at, - has_rows=True, - prior_start=state.get("start"), - resumed=True, - extending=True, - ) + return carried._replace(start=continue_at, extending=True) - # THE PAGE BOUNDARY, not the newest row. `last_day` is the highest date + # THE PAGE BOUNDARY, not the newest row. `lastDay` is the highest date # written; the page it came from covered further, because an entity that # stopped reporting has no rows for the tail days. Resuming at `last_day` - # re-fetches a day already in the file and appends every row of it again -- - # on the exit-75 path, which is the ordinary path for a long back fill, and - # it breaks the (entityId, date) key the file is documented to have. - # - # `last_day` stays as the fallback for the one case with no boundary - # recorded: a run that died part-way through its FIRST page. One duplicated - # day beats starting from the top and appending a second copy of the whole - # archive. + # re-fetches a day already in the file and appends every row of it again. + # Every state this build writes has a boundary; `_back_to_checkpoint` gives + # one to an older state that did not. resume_at = state.get("resumeFrom") or state.get("lastDay") continue_at = str(resume_at or state.get("start") or archive_from) refused = _range_fits(out_path, state, rng, archive_from, continue_at) if refused is not None: return refused - return _Plan( - start=continue_at, - has_rows=True, - prior_start=state.get("start"), - resumed=True, - ) + if rng.until is not None and continue_at > str(end): + print( + f" nothing to add: {out_path.name} continues from {continue_at}, after " + f"--until {rng.until}", + file=sys.stderr, + ) + return 0 + return carried._replace(start=continue_at) # -------------------------------------------------------------------------- @@ -838,7 +983,7 @@ def _upgradable(state: dict[str, Any], fmt: str) -> bool: def _upgrade_v1( state: dict[str, Any], out_path: Path, state_path: Path, fmt: str -) -> dict[str, Any] | None: +) -> dict[str, Any]: """Make a 4.0 file one this build can continue, replacing its unsettled tail. 4.0 ended every run at `retrievableThrough`, usually today, so the last few @@ -854,9 +999,10 @@ def _upgrade_v1( than the cut, a park that stopped reporting long ago, is not read at all. A file that lies WHOLLY inside the cut, as every anonymous 7-day file does, - is downloaded again instead: None, with both files removed. Trimmed, it - would be empty with a state continuing from a day the key may no longer - read, which a continued file is not allowed to skip past. + is downloaded again instead: the cut leaves it empty, its checkpoint says 0 + bytes, and `_back_to_checkpoint` treats that as a first run. Continued, it + would carry on from a day the key may no longer read, which a continued file + is not allowed to skip past. The new state is written straight away, so a run that fails after this point does not trim the same file twice. @@ -865,13 +1011,11 @@ def _upgrade_v1( upgraded = {**state, "stateVersion": STATE_VERSION} if not end: return upgraded - keep_through = (date.fromisoformat(str(end)) - timedelta(days=V1_UNSETTLED_DAYS)).isoformat() + keep_through = _days_before(end, V1_UNSETTLED_DAYS) last_day = state.get("lastDay") - if last_day is None or str(last_day) > keep_through: - if _trim_after(out_path, fmt, keep_through) == 0: - out_path.unlink(missing_ok=True) - state_path.unlink(missing_ok=True) - return None + trimmed = last_day is None or str(last_day) > keep_through + if trimmed: + _trim_after(out_path, fmt, keep_through) upgraded["lastDay"] = keep_through if last_day is not None else None if state.get("complete"): upgraded["end"] = keep_through @@ -879,6 +1023,9 @@ def _upgrade_v1( resume = state.get("resumeFrom") or last_day if resume is None or str(resume) > _next_day(keep_through): upgraded["resumeFrom"] = _next_day(keep_through) + if trimmed or state.get("complete"): + # Nothing past the cut is left, so the whole file is vouched for. + upgraded["size"] = out_path.stat().st_size _write_state(state_path, **upgraded) return upgraded @@ -886,20 +1033,25 @@ def _upgrade_v1( def _trim_after(out_path: Path, fmt: str, keep_through: str) -> int: """Remove every row dated after `keep_through`, leaving the rest byte for byte. - Returns how many dated rows were kept. + Returns how many rows were kept, counting any it could not read. Streamed into a file beside the original and swapped in with one rename, so an interruption leaves either the old file or the new one, never half of - each. A line or record that cannot be read is kept: this command does not - get to decide that something it does not understand is worthless. + each, and the half-written copy is removed. A line or record that cannot be + read is kept: this command does not get to decide that something it does not + understand is worthless. The CSV is parsed and written back with this module's own quoting, which is a function of the text alone, so a kept record comes out as it went in. """ scratch = out_path.with_name(out_path.name + ".trimming") - with out_path.open(encoding="utf-8", newline="") as src: - kept = _copy_rows_through(src, scratch, fmt, keep_through) - os.replace(scratch, out_path) + try: + with out_path.open(encoding="utf-8", newline="") as src: + kept = _copy_rows_through(src, scratch, fmt, keep_through) + os.replace(scratch, out_path) + except BaseException: + scratch.unlink(missing_ok=True) + raise return kept @@ -916,10 +1068,9 @@ def _copy_rows_through(src: TextIO, scratch: Path, fmt: str, keep_through: str) # No `date` column: not a file this can read, so it is copied whole. column = names.index("date") if "date" in names else -1 for record in reader: - if 0 <= column < len(record): - if record[column] > keep_through: - continue - kept += 1 + if 0 <= column < len(record) and record[column] > keep_through: + continue + kept += 1 dst.write(",".join(_csv_cell(cell) for cell in record) + "\n") else: for line in src: @@ -927,10 +1078,9 @@ def _copy_rows_through(src: TextIO, scratch: Path, fmt: str, keep_through: str) day = json.loads(line).get("date") except (ValueError, AttributeError): day = None - if isinstance(day, str): - if day > keep_through: - continue - kept += 1 + if isinstance(day, str) and day > keep_through: + continue + kept += 1 dst.write(line) return kept @@ -944,6 +1094,8 @@ class _Job(NamedTuple): end: Day has_rows: bool ident: _RowIdentity + #: Where every checkpoint is written. + state: _StateFile #: True when the file is being CONTINUED, so its next day is fixed. resumed: bool = False @@ -959,34 +1111,26 @@ class _Progress: Owned by the caller, updated in place, so it is readable after a raise. """ - def __init__(self) -> None: + def __init__(self, file_start: Day = None, last_day: str | None = None) -> None: self.written = 0 - self.last_day: date | None = None + self.last_day: date | str | None = last_day self.resume_from: str | None = None self.skipped = False #: The key cannot read the day a continued file carries on from. self.out_of_reach = False - #: The day this run's request started on, which is where a rerun has to - #: carry on if the run fails before its first page is finished. - self.first_day: str | None = None - - def resume_point(self) -> str | None: - """Where a rerun should continue, for the state file. - - The next page's start once a page is done. Before that, with rows - written, None: `lastDay` is the fallback and costs one duplicated day. - Before ANY row, the day this run began on. That last case had nothing - recorded, which was harmless while only a first run could reach it -- a - rerun of a file with no rows starts from the top anyway -- and is not - now that a finished file is carried forward. A rerun interrupted before - its first page would otherwise have fallen back to the archive's start - and appended the whole archive a second time. - """ - if self.resume_from is not None: - return self.resume_from - if self.last_day is not None: - return None - return self.first_day + #: The first day in the file: the prior run's on a continued file, the + #: first day this run actually requested successfully on a new one. + self.file_start: Day = file_start + + def checkpoint(self, sf: _StateFile, size: int, resume_from: str | None) -> None: + """Record that the first `size` bytes of the file are good through here.""" + _record(sf, _Checkpoint(self.file_start, self.last_day, resume_from, size, False)) + + +def _size(handle: TextIO) -> int: + """Bytes written so far, after pushing Python's buffer to the OS.""" + handle.flush() + return os.fstat(handle.fileno()).st_size def _stream(job: _Job, start: Day, progress: _Progress) -> None: @@ -995,35 +1139,51 @@ def _stream(job: _Job, start: Day, progress: _Progress) -> None: Separated from `backfill_park` because that function was deciding, printing, streaming and recording in one place, and ruff counted the statements before a reader had to. This is the streaming. + + THE STATE IS WRITTEN BEFORE THE FIRST REQUEST AND AFTER EVERY PAGE, with the + file's size at that moment. Nothing else has to run for it to be right: a + run stopped by Ctrl-C, SIGTERM or SIGKILL leaves a state that names the last + page boundary, and the rerun cuts the file back to it. An extending run + marks the file unfinished before it appends its first row, so no rerun can + mistake a half-extended file for a finished one. """ + handle: TextIO def note_page(page: HistoryPage) -> None: """Checkpoint, called once every row of a page is written. The day the NEXT page starts on, taken from the server's own `next` URL, so a resumed run asks for nothing twice. None on the last page, where - there is nothing left to carry on from. + there is nothing left to carry on from; completion is recorded by the + caller once the file is closed. """ progress.resume_from = _next_page_start(page.next_url) + if progress.resume_from is not None: + progress.checkpoint(job.state, _size(handle), progress.resume_from) def write_rows(writer: Writer, first_day: Day) -> None: """Stream one range into the file. Raises whatever the SDK raises.""" - progress.first_day = None if first_day is None else str(first_day) for ref, row in job.history.days_with_entities(first_day, job.end, on_page=note_page): + if progress.written == 0 and progress.file_start is None: + # The first row of a new file: this range was not refused, so its + # first day is the file's, whatever `--since` asked for. + progress.file_start = first_day writer.write(ref, row) progress.written += 1 # MAX, not last-seen. `_daily_rows` walks entities and then each # entity's days, so the final row belongs to the alphabetically last - # entity, which may have stopped reporting mid-page. Taking it as the - # high-water mark could rewind the resume point by up to a whole - # 31-day page, while the module claimed the overlap was "one day". + # entity, which may have stopped reporting mid-page. + previous = progress.last_day progress.last_day = ( - row.date if progress.last_day is None else max(progress.last_day, row.date) + row.date if previous is None else max(date.fromisoformat(str(previous)), row.date) ) if progress.written % 5000 == 0: print(f" {progress.written} rows, at {progress.last_day}", file=sys.stderr) with job.out_path.open("a", newline="", encoding="utf-8") as handle: + # Before the first request: from here on the file may grow, so the state + # must say where it was good up to. + progress.checkpoint(job.state, _size(handle), None if start is None else str(start)) # ONE Writer for the whole park, so the header decision is made once. It # used to be built inside write_rows with `written == 0` in the predicate, # and the recovery below calls that again precisely when written is 0 -- @@ -1094,7 +1254,7 @@ def _window_closed(out_path: Path, end: Day, start: Day, *, resumed: bool) -> in def _nothing_written( - out_path: Path, state_path: Path, sf: _StateFile, progress: _Progress, *, resumed: bool + out_path: Path, state_path: Path, progress: _Progress, *, resumed: bool ) -> int: """The park had nothing in this key's window. Tidy up, or refuse to. @@ -1106,16 +1266,44 @@ def _nothing_written( outright: nothing written and the state untouched, so the next run meets the same refusal until someone decides, rather than carrying on with a gap. """ - if progress.out_of_reach: - return 1 - if resumed: - _record(sf, progress.last_day, progress.resume_point(), complete=False) + if progress.out_of_reach or resumed: + # The checkpoint written before the first request already says where + # the file continues from. return 1 out_path.unlink(missing_ok=True) state_path.unlink(missing_ok=True) return 0 +@contextlib.contextmanager +def _park_lock(out_dir: Path, park_id: str, fmt: str) -> Iterator[bool]: + """Hold an advisory lock on one park's output, yielding whether it was got. + + Two runs on the same park and `--out`, say a cron overlapping a manual run, + would each read the same state and append the same days. The lock is a + separate file because the state and data files are replaced by rename, and a + lock on a replaced file protects nothing. The operating system releases it + when the process ends, however it ends, so a killed run never leaves a stale + lock behind. Where `fcntl` does not exist (Windows) this does not lock. + """ + try: + import fcntl # noqa: PLC0415 - POSIX only; absent on Windows + except ImportError: # pragma: no cover - exercised on Windows only + yield True + return + path = out_dir / f".{park_id}.{fmt}.backfill-lock" + with path.open("a") as handle: + try: + fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + except OSError: + yield False + return + try: + yield True + finally: + fcntl.flock(handle.fileno(), fcntl.LOCK_UN) + + def _run_end(span: HistorySpan, rng: _Range) -> Day: """The last day this run asks for: the newest FINAL day, or `--until` if earlier. @@ -1158,7 +1346,22 @@ def backfill_park( # noqa: PLR0913 - window is keyword-only, added without brea `window` is `--since`/`--until`. Without it the range is everything the key may read, through the newest final day. """ - rng = window or _Range() + with _park_lock(out_dir, park.id, fmt) as held: + if not held: + print( + f"{park.id}: another themeparks-backfill is writing this park into " + f"{out_dir} right now. Two at once would each append the same days; " + f"wait for it to finish", + file=sys.stderr, + ) + return 1 + return _backfill_park(tp, park, out_dir, fmt, overwrite=overwrite, rng=window or _Range()) + + +def _backfill_park( # noqa: PLR0913 - the arguments of backfill_park, resolved + tp: ThemeParks, park: _Park, out_dir: Path, fmt: str, *, overwrite: bool, rng: _Range +) -> int: + """`backfill_park`, once this process holds the park's lock.""" park_id = park.id history = tp.entity(park_id).history @@ -1195,25 +1398,26 @@ def backfill_park( # noqa: PLR0913 - window is keyword-only, added without brea return decided start, has_rows, prior_start, resumed_run = decided[:4] end = ask.end - sf = _StateFile(state_path, fmt, prior_start or start, end) + sf = _StateFile(state_path, fmt, decided.since, end) _announce(park_id, decided, ask, span, out_path) if _is_empty_window(start, end): return _window_closed(out_path, end, start, resumed=resumed_run) ident = _RowIdentity(park) - job = _Job(history, out_path, fmt, end, has_rows, ident, resumed=resumed_run) - progress = _Progress() + job = _Job(history, out_path, fmt, end, has_rows, ident, sf, resumed=resumed_run) + progress = _Progress(prior_start, decided.prior_last_day) try: _stream(job, start, progress) except BudgetExhaustedError as exc: # The budget is hourly, so a spent one can be most of an hour from - # resetting. Record how far we got and exit 75 rather than sleeping. - _record(sf, progress.last_day, progress.resume_point(), complete=False) + # resetting. Exit 75 rather than sleeping; the last checkpoint already + # says where to carry on. if progress.written == 0 and not resumed_run: # A budget spent before the first page left a 0-byte file that reads # as "this park has no history". out_path.unlink(missing_ok=True) + state_path.unlink(missing_ok=True) wait = exc.retry_after or 0 print( f" budget spent; rerun the same command in {wait / 60:.0f} min to continue", @@ -1225,14 +1429,13 @@ def backfill_park( # noqa: PLR0913 - window is keyword-only, added without brea # turns exit 75 into a traceback and exit 1 -- the precise regression the # budget handler exists to prevent. except (ThemeParksError, OSError): - # Every other failure still records where it got to, or the next run - # starts over and appends a second partial copy. And AN EMPTY FILE IS A - # LIE: opening the file created it before the first request, so a park - # that failed with nothing written left a 0-byte file that reads as - # "this park has no history" -- on a six-park destination the customer - # counts six files and never sees which one is empty. - if progress.last_day is not None: - _record(sf, progress.last_day, progress.resume_point(), complete=False) + # The last checkpoint already says where to carry on, and the rerun cuts + # off anything written after it. AN EMPTY FILE IS A LIE: opening the file + # created it before the first request, so a park that failed with nothing + # written left a 0-byte file that reads as "this park has no history" -- + # on a six-park destination the customer counts six files and never sees + # which one is empty. + # # `written` counts rows THIS process wrote, so on a resumed run it is 0 # while the file holds everything the previous runs fetched. Deleting it # there destroyed the archive and left the state file pointing into the @@ -1240,14 +1443,16 @@ def backfill_park( # noqa: PLR0913 - window is keyword-only, added without brea # `complete: true`. if progress.written == 0 and not resumed_run: out_path.unlink(missing_ok=True) + state_path.unlink(missing_ok=True) raise if progress.out_of_reach or (progress.skipped and progress.written == 0): - return _nothing_written(out_path, state_path, sf, progress, resumed=resumed_run) + return _nothing_written(out_path, state_path, progress, resumed=resumed_run) # Completion is RECORDED, never inferred from a missing file. That is the # distinction the old checkpoint could not make. - _record(sf, progress.last_day, None, complete=True) + size = out_path.stat().st_size + _record(sf, _Checkpoint(progress.file_start, progress.last_day, None, size, True)) print(f" done: {progress.written} rows -> {out_path}", file=sys.stderr) return 0 @@ -1503,7 +1708,12 @@ def _iso_day(value: str) -> str: only final days are written. Today's row is the day so far, and the archive records days 2 to 3 behind live data, so the newest days can still change. A run ends at the newest final day and the next run carries on from the day after, so -the file only ever grows and no row in it changes later. +the file only ever grows. Each day is fetched once, as the archive recorded it; +if the archive later re-records past days (a repaired feed), fetch them again +with --since/--until into a different --out, or start again with --overwrite. + +stopping is safe at any point: the state is saved after every page, and the next +run cuts off anything written after it, so no day is appended twice. --since applies when a file is started. A later run continues that file forward and accepts the same --since, or a later one. One earlier than the file's first @@ -1617,7 +1827,9 @@ def main(argv: list[str] | None = None) -> int: # whether to pay is exactly the person who should be able to run this. if not args.api_key: print( - "no API key: reading the 7 days anonymous access allows.\n" + "no API key: reading the 7 days anonymous access allows. Only final days\n" + " are written, and the newest 2 to 3 are still being recorded, so that\n" + " is usually 4 or 5 days per park.\n" " a free key reads 30 days, Pro 400, Business the whole archive\n" " set THEMEPARKS_API_KEY, or pass --api-key\n" " keys: https://www.themeparks.wiki/profile\n", @@ -1679,7 +1891,8 @@ def main(argv: list[str] | None = None) -> int: # success. They paid for 400 days and got seven, exit 0, no complaint. if not args.api_key: print( - "\nthat was ANONYMOUS ACCESS: the last 7 days only.\n" + "\nthat was ANONYMOUS ACCESS: the final days among the last 7 days only,\n" + " usually 4 or 5 per park.\n" " a free key reads 30 days, Pro 400, Business the whole archive\n" " set THEMEPARKS_API_KEY and run the same command again\n" " keys: https://www.themeparks.wiki/profile", @@ -1733,11 +1946,19 @@ def cli() -> int: these are bugs in it, and a customer who has just paid reads one as the tool being broken. """ + previous = _stop_on_sigterm() try: return main() - except KeyboardInterrupt: - print("\nstopped. Run the same command again to continue.", file=sys.stderr) - return 130 + except KeyboardInterrupt as exc: + # Nothing to save here. The state file was written after the last whole + # page, and the next run cuts off anything written since, so stopping at + # any point costs at most the page in flight. + print( + "\nstopped. Everything through the last whole page is saved; run the " + "same command again to continue.", + file=sys.stderr, + ) + return 143 if isinstance(exc, _Terminated) else 130 except (NetworkError, ApiTimeoutError) as exc: # A connection reset or a timeout IS resumable, so this is EX_TEMPFAIL and a # scheduler retries rather than alerting. The JavaScript SDK said 75 here @@ -1760,6 +1981,32 @@ def cli() -> int: # a few hundred MB, so this is not hypothetical. print(f"cannot write the output: {exc}", file=sys.stderr) return 1 + finally: + if previous is not None: + signal.signal(signal.SIGTERM, previous) + + +class _Terminated(KeyboardInterrupt): + """SIGTERM, raised where the process is, so it unwinds like Ctrl-C. + + The default SIGTERM action ends the process without running any `finally` + or `with` exit: the output file is not closed and nothing says what + happened. A scheduler stopping a run (systemd, a container shutdown) sends + exactly this, so it gets the same orderly stop and message as Ctrl-C, and + exit 143, the conventional code for it. + """ + + +def _stop_on_sigterm() -> Any: + """Turn SIGTERM into `_Terminated` for this process. Returns the old handler.""" + + def handler(_signum: int, _frame: Any) -> None: + raise _Terminated + + try: + return signal.signal(signal.SIGTERM, handler) + except ValueError: # pragma: no cover - not the main thread + return None if __name__ == "__main__": From f7b31b7045a4f27c777318c2a2be3a4b345bdcfe Mon Sep 17 00:00:00 2001 From: Jamie Holding Date: Tue, 29 Sep 2026 08:57:04 +0100 Subject: [PATCH 9/9] docs: interrupted runs, fetching a range again, recorded rather than immutable Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 39 +++++++++++++++++++++++++++++++++++---- README.md | 23 ++++++++++++++++++++--- docs/cookbook.md | 3 ++- 3 files changed, 57 insertions(+), 8 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index e17944d..bcbad35 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,9 +25,21 @@ result has the same attribute once the response has arrived: iterate first, or `await changes.load()`. -- **`HistorySpan.final_through`**: the newest day whose daily row will not - change again, the earlier of `recorded_to` and `retrievable_through`. A - property, so `a, b, c = span` still works. +- **`HistorySpan.final_through`**: the newest day the archive has recorded that + the key may read, the earlier of `recorded_to` and `retrievable_through`: the + place to stop if you fetch each day once. A property, so `a, b, c = span` + still works. + +- **`HistoryOpening`** is exported, and `HistoryChanges.close()` / + `AsyncHistoryChanges.aclose()` end iteration early, as they did on the + generators these replaced. + +- **An interrupted `themeparks-backfill` never appends a day twice.** The state + file is written before the first request and after every page, atomically, + with the size of the data file at that moment; the next run first cuts the + file back to that size. Ctrl-C, SIGTERM (now exit 143) and SIGKILL at any point + cost at most the page in flight. Two runs on the same park and `--out` at once + are refused (POSIX). ### Fixed @@ -46,7 +58,9 @@ days of every file were still changing when they were written. Magic Kingdom's last day summed to about half the operating minutes of a full one. A run now ends at `final_through`, says so when it holds days back, and the next run adds - them once they are final. Every row in the file is one that will not change. + them once they are final. Each day is fetched once, as the archive recorded + it; the README says how to fetch a range again if the archive later + re-records it. **Files written by 4.0.x are corrected once.** Their state file does not say which of their newest days were final, so the first run of this version @@ -57,6 +71,23 @@ downloaded again. The state file format moves to version 2 for this; version 1 files from this SDK are upgraded, not refused. +- **An interrupted nightly extension appended the same days twice.** The state + was written only at the end of a run or on an SDK error, so Ctrl-C, SIGTERM + or SIGKILL during an extension left it saying finished through the old day, + and the rerun appended those days again. See the checkpointing above. + +- **A run that stopped part-way through its first page lost days.** It resumed + from the newest day any entity had reached; rows arrive entity by entity, so + the entities behind it lost the days in between. Checkpoints make this + impossible for new files, and an older state with only that day goes back a + whole page (31 days) instead. + +- **The state recorded the start asked for, not the first day written.** After + the key's window moved it later, a `--since` earlier than the real first day + passed silently. `start` is now the first day written and a new `since` + field keeps the one asked for, so the same `--since` keeps working and a + different one before the file is refused. + - **A continued file could skip ahead to the key's first day.** When the day a file continues from is older than the key may read, because a cron missed more days than the window or a plan lapsed, the run carried on from the key's first diff --git a/README.md b/README.md index a0df218..62d38e9 100644 --- a/README.md +++ b/README.md @@ -281,7 +281,7 @@ with ThemeParks(api_key=KEY) as tp: # What exists, and what your key may read. Same three fields whether the # id is a park or a single ride. `final_through` is the newest day whose - # row will not change again: store through that, ask for the rest later. + # the archive has recorded: store through that, ask for the rest later. span = history.span() print(span.archive_from, span.recorded_to, span.retrievable_through) print(span.final_through) @@ -360,8 +360,25 @@ forward from the day after its last one, so the same command in a nightly cron appends the new days and nothing else. Only **final** days are written: today's row is the day so far, and the archive records days 2 to 3 behind live data, so the newest days can still change. A run stops at the newest final day -(`span().final_through`) and says so, and the next run adds the rest. Every row -in the file is one that will not change later. +(`span().final_through`) and says so, and the next run adds the rest. Each day +is fetched once, as the archive recorded it. + +**Fetching a range again.** The archive can occasionally re-record past days, +for example when a park's feed is repaired. A file never rewrites rows it +already holds, so to pick up a correction, download the affected days into a +separate directory and replace those `(entityId, date)` rows where you load the +data, or start the file again: + +```bash +themeparks-backfill "Epcot" --since 2026-06-01 --until 2026-06-30 --out ./refetch +themeparks-backfill "Epcot" --overwrite # or: the whole file again +``` + +Stopping a run at any point is safe. The state file is written after every page +with the size of the file at that moment, and the next run first cuts off +anything written after it, so no day is ever appended twice. Ctrl-C and SIGTERM +exit 130 and 143; a killed run needs nothing either. Two runs on the same park +and `--out` at once are refused. `--since` applies when a file is started. Later runs continue that file and accept the same `--since`, or a later one, such as a cron line computing "30 diff --git a/docs/cookbook.md b/docs/cookbook.md index b8e4080..8174094 100644 --- a/docs/cookbook.md +++ b/docs/cookbook.md @@ -308,7 +308,8 @@ not by `recorded_to`: the archive holds more than a free or Pro key is entitled to read, and asking past the entitlement is how a backfill walks into a wall of 403s at the end of a long run. If you write each day once and never revisit it, end at `span.final_through` instead: the earlier of the two, and -the newest day whose row will not change again. +the newest day the archive has recorded. The archive can re-record a past day +after a feed repair, so fetch a range again if you need to pick that up. ```python import json