diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 928a1de..7355c2d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -37,3 +37,6 @@ jobs: - run: npm run typecheck - run: npm run build - run: npm test + # Packs the tarball, installs it elsewhere and runs the binary. The unit + # suite cannot see a bin that does not resolve when installed. + - run: npm run test:package diff --git a/.github/workflows/spec-drift.yml b/.github/workflows/spec-drift.yml index fef6331..f9f5ba8 100644 --- a/.github/workflows/spec-drift.yml +++ b/.github/workflows/spec-drift.yml @@ -58,7 +58,15 @@ jobs: uses: peter-evans/create-pull-request@v8 with: token: ${{ steps.app-token.outputs.token }} - add-paths: src/_generated/schema.ts + # BOTH GENERATED FILES. `npm run regenerate` writes the schema and the + # daily column list; committing only the schema meant the next field + # upstream added would be typed and not exported, the CSV would silently + # drop it again, and the Python SDK -- which derives its columns at + # runtime -- would pick it up, so the two headers would diverge with + # nothing red anywhere. + add-paths: | + src/_generated/schema.ts + src/_generated/dailyColumns.ts branch: automated/spec-drift commit-message: 'chore: regenerate schema from upstream spec' title: 'chore: regenerate schema from upstream spec' diff --git a/CHANGELOG.md b/CHANGELOG.md index 10f8714..eb0fc0a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,97 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [8.3.0] - 2026-09-28 + +### Added + +- **`themeparks-backfill`: the archive download as a command.** The Python SDK + shipped this first; this is the same tool, and the two write byte-for-byte + identical CSVs. Magic Kingdom's full five-year archive: 94,223 rows, 41 columns, + identical from both, the only differences being today's row, which grows as the + day elapses. + + ```bash + npm install themeparks + npx themeparks-backfill "magic kingdom" + ``` + + - Takes a park or a **destination**, by name or id, and a name that identifies one + park unambiguously is enough. A destination back fills every park in it, one + file each. An ambiguous name lists the ids that match, sorted by park name. + - `--list [text]` prints destinations with their parks underneath and **needs no + key**, so you can find your park before deciding whether to pay. + - **Runs without a key**, reading the 7 days anonymous access allows, and says + what a key would add. + - NDJSON by default, `--format csv` for one wide row per entity per day. Every row + carries `parkId`, `parkName`, `entityId`, `entityName` and `entityType`, so two + files load into one table and `(entityId, date)` is the natural key. The entity + name is the one the history response gave for those rows, not the park's current + children list: rides get renamed, and today's name on a row from three years ago + rewrites the record. Files are named for the park's id, because names change. + - The CSV carries a **UTF-8 BOM** so Excel on Windows does not mangle `®` and + accents, and a cell a spreadsheet would execute as a formula is prefixed with an + apostrophe. Numeric cells are untouched, so a negative number stays a number. + - **Resumable.** It checkpoints against the hourly history budget and exits 75 + (`EX_TEMPFAIL`), so a cron or systemd timer retries rather than alerting, and + running the same command again continues. The checkpoint is the day the server's + own `next` URL starts on, never the newest row written -- an entity that stopped + reporting has no rows for the tail days of its page, so resuming from a row + re-fetches days already in the file. + - The state file is `..backfill-state.json` and records the SDK, + its version, a state version and a fingerprint of the exact header. Anything + that does not match is refused with a message saying why, never resumed -- + including a state file written by the Python SDK, whose keys differ. + - One park's failure does not abandon the rest of a destination; what did not + finish is named at the end. A network failure or timeout exits 75, anything the + API rejected exits 1, and neither is a traceback. + - An earlier run's rows are never deleted. A failure or a closed window on a + resumed run keeps the file and says the run did not finish. + +- **`onPage` on `days()`**, called once every row of a page has been yielded, with a + `HistoryPage` (`from`, `to`, `next`). The page boundary is the server's own answer + to "where do I carry on", and the rows cannot tell you -- so it is the only safe + checkpoint for a resumable download. `HistoryPage` and `PageOptions` are exported. + +- **`DailyEntry` carries `name` and `entityType`**, taken from the history response + itself. Already in the payload, so nothing has to ask what an id refers to. Both + are required fields, so a hand-built `DailyEntry` in a test double needs them. + +- **`test/fixtures/csv_contract.json`**, an identical copy of which lives in the + Python SDK. Both suites assert their column list against it, because this is one + command with two implementations and a customer using both should get one file + format. Before it existed, this SDK wrote 32 columns and Python wrote 41. + +- **`npm run test:package`**, in CI and `prepublishOnly`: it packs the tarball, + installs it elsewhere and runs the binary. See below for why. + +### Fixed + +- **The vendored OpenAPI schema was stale.** `unknownMinutes`, `inParkHours` (the + day's numbers limited to the park's published hours) and `extremeWaits` (how many + readings of 480+ minutes are folded into the statistics, which is how you spot a + feed error) are on the rows the API returns and were in none of the types. The CSV + column list is now **generated from the spec**, the nightly drift job commits it + alongside the schema, and the generator refuses a duplicate column name or a + missing nested block. + +- **A failed write was reported as success.** Node hands `end`'s callback the + stream's error; the callback took no arguments and resolved regardless, so on + ENOSPC or EDQUOT mid-download the command printed `done: N rows`, recorded + `complete: true` and exited 0 with a truncated file that no rerun would continue. + The stream also had no `'error'` listener until the flush, so an earlier failure + became an unhandled `'error'` event that killed the whole run. + +- **A bare `\r` in an entity name was written unquoted**, so one row parsed as two + with every later column shifted. + +- **`--list` with no value exited 2** although the help advertises `--list [TEXT]`; + `-h` was not accepted; a query that folds to nothing (`東京`) listed all 127 parks + instead of none; an empty `--api-key` or `THEMEPARKS_API_KEY=""` counted as a key; + running with no arguments fetched `/destinations` before saying so, which exited + 75 with no network; `--version` printed a bare number; and the 404 hint for a + mistyped id sat where nothing could reach it. + ## [8.2.0] - 2026-09-26 ### Added diff --git a/README.md b/README.md index 86081a9..e2968ab 100644 --- a/README.md +++ b/README.md @@ -332,7 +332,33 @@ try { `history.changeRows(query)` is the same treatment for `changes`: one flattened stream of `{ entityId, row }` whether you asked a park or a ride. -A complete backfill with resume and CSV output is in +### Or skip the code: there is a command + +Installing the package puts `themeparks-backfill` on your path. It is the same +job as the example below, resumable, and it is what to reach for if what you +want is the file rather than the code: + +```bash +npx themeparks-backfill --list disney # find your park. No key needed. +npx themeparks-backfill "magic kingdom" # NDJSON, into the current directory +npx themeparks-backfill "Walt Disney World Resort" --format csv --out ./data +``` + +A park or a **destination**, by name or by id; a destination writes one file per +park. Every row carries `parkId`, `parkName`, `entityId`, `entityName` and +`entityType`, so two files load into one table and `(entityId, date)` is the +natural key. Files are named for the park's id, because names change. + +How far back it reaches is your plan, and it asks the API rather than making you +work it out. It checkpoints against the hourly history budget and exits 75 +(`EX_TEMPFAIL`) when that runs out, so a cron or systemd timer retries instead of +alerting and the same command continues where it stopped. `--help` has the rest. + +The CSV is byte-for-byte identical to the Python SDK's, which runs the same +command: Magic Kingdom's five-year archive is 94,223 rows and 41 columns from +either. + +A library version of the same loop, if you want to own it, is in [`examples/backfill.mjs`](examples/backfill.mjs). It pulled Disneyland Resort's whole daily archive, 98,452 rows, in one run. diff --git a/package-lock.json b/package-lock.json index f5540fc..30b984c 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,13 +1,16 @@ { "name": "themeparks", - "version": "8.1.0", + "version": "8.3.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "themeparks", - "version": "8.1.0", + "version": "8.3.0", "license": "MIT", + "bin": { + "themeparks-backfill": "dist/backfill-cli.js" + }, "devDependencies": { "@types/node": "^26.4.1", "@typescript-eslint/eslint-plugin": "^8.0.0", @@ -20,7 +23,8 @@ "typedoc": "^0.28.19", "typedoc-plugin-markdown": "^4.11.0", "typescript": "^5.4.0", - "vitest": "^4.1.11" + "vitest": "^4.1.11", + "yaml": "^2.5.0" }, "engines": { "node": ">=20" diff --git a/package.json b/package.json index 524bf7a..29b92e9 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "themeparks", - "version": "8.2.0", + "version": "8.3.0", "description": "Official SDK for the ThemeParks.wiki API", "license": "MIT", "repository": "github:ThemeParks/ThemeParks_JavaScript", @@ -17,6 +17,9 @@ "require": "./dist/index.cjs" } }, + "bin": { + "themeparks-backfill": "./dist/backfill-cli.js" + }, "files": [ "dist", "README.md", @@ -39,7 +42,8 @@ "regenerate": "tsx scripts/regenerate.ts", "docs": "typedoc", "docs:serve": "npx serve docs-site", - "prepublishOnly": "npm run build" + "prepublishOnly": "npm run build && npm run test:package", + "test:package": "tsx scripts/check-package.ts" }, "devDependencies": { "@types/node": "^26.4.1", @@ -53,6 +57,7 @@ "typedoc": "^0.28.19", "typedoc-plugin-markdown": "^4.11.0", "typescript": "^5.4.0", - "vitest": "^4.1.11" + "vitest": "^4.1.11", + "yaml": "^2.5.0" } } diff --git a/scripts/check-package.ts b/scripts/check-package.ts new file mode 100644 index 0000000..7303b6c --- /dev/null +++ b/scripts/check-package.ts @@ -0,0 +1,66 @@ +/** + * Pack the tarball, install it somewhere else, and run the binary. + * + * This exists because of a defect that passed the whole unit suite, worked in the + * repo, and would have shipped: the executable's "am I the entry point" guard + * compared `import.meta.url` against `process.argv[1]`, and npm installs a binary + * as a SYMLINK in `node_modules/.bin`. The paths differ, the guard was false, and + * `npx themeparks-backfill --version` printed nothing and exited 0. + * + * Nothing short of installing it shows that. So: pack, install into a temporary + * directory, run the binary the way a customer does, and require real output. + */ +import { execFileSync } from 'node:child_process'; +import { mkdtempSync, readdirSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; + +const root = process.cwd(); +const dir = mkdtempSync(join(tmpdir(), 'themeparks-package-')); + +function sh(command: string, args: string[], cwd: string): string { + return execFileSync(command, args, { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }); +} + +try { + sh('npm', ['pack', '--pack-destination', dir], root); + const tarball = readdirSync(dir).find((f) => f.endsWith('.tgz')); + if (tarball === undefined) throw new Error('npm pack produced no tarball'); + + writeFileSync(join(dir, 'package.json'), JSON.stringify({ name: 'consumer', private: true })); + sh('npm', ['install', '--no-audit', '--no-fund', join(dir, tarball)], dir); + + const bin = join(dir, 'node_modules', '.bin', 'themeparks-backfill'); + const version = sh(bin, ['--version'], dir).trim(); + const expected = JSON.parse(sh('npm', ['pkg', 'get', 'version'], root) as string) as string; + // Named, matching the Python SDK: a bare number cannot be pasted into a bug report. + if (version !== `themeparks-backfill ${expected}`) { + throw new Error( + `the installed binary printed "${version}", expected "themeparks-backfill ${expected}"`, + ); + } + + const help = sh(bin, ['--help'], dir); + if (!help.includes('themeparks-backfill')) throw new Error('--help printed nothing usable'); + + // A bare `--list` exits 2 if parseArgs treats the option as value-taking, which is + // exactly the class of defect only an installed run shows. Needs no key. + const listed = sh(bin, ['--list', 'epcot'], dir); + if (!listed.includes('EPCOT')) throw new Error(`--list epcot printed nothing usable: ${listed}`); + + // The library import path, which is a different resolution from the binary. + const imported = sh( + process.execPath, + [ + '--input-type=module', + '-e', + "import {ThemeParks} from 'themeparks'; console.log(typeof ThemeParks)", + ], + dir, + ).trim(); + if (imported !== 'function') throw new Error(`importing the package gave ${imported}`); + + console.log(`package ok: binary and import both work from a clean install (${version})`); +} finally { + rmSync(dir, { recursive: true, force: true }); +} diff --git a/scripts/regenerate.ts b/scripts/regenerate.ts index 1073cac..5808433 100644 --- a/scripts/regenerate.ts +++ b/scripts/regenerate.ts @@ -1,9 +1,11 @@ import { writeFile } from 'node:fs/promises'; import { resolve } from 'node:path'; import openapiTS, { astToString } from 'openapi-typescript'; +import { parse } from 'yaml'; const SPEC_URL = 'https://api.themeparks.wiki/docs/v1.yaml'; const OUTPUT = resolve(process.cwd(), 'src/_generated/schema.ts'); +const COLUMNS_OUTPUT = resolve(process.cwd(), 'src/_generated/dailyColumns.ts'); const header = `/* eslint-disable */ /** @@ -12,11 +14,80 @@ const header = `/* eslint-disable */ */ `; +interface SchemaNode { + properties?: Record; + $ref?: string; +} + +/** + * Every scalar on a daily history row, flattened to one column name each, in the + * order the spec declares them. + * + * GENERATED, not typed out. The hand-written list in backfill.ts had drifted + * three ways at once: `unknownMinutes` and the whole `inParkHours` block were on + * every row the API returns and in no column, `extremeWaits` likewise, and + * `singleRider` carried two of its five percentiles while `standby` carried all + * five. Ten of thirty-six fields silently absent from a file people pay for, and + * the two SDKs disagreeing about the header of a file they both claim to write. + * Regenerate and the columns follow. + */ +function dailyColumns(schemas: Record): string[] { + const walk = (name: string, prefix: string): string[] => { + const node = schemas[name]; + const out: string[] = []; + for (const [field, value] of Object.entries(node?.properties ?? {})) { + const head = prefix === '' ? field : `${prefix}${field[0]!.toUpperCase()}${field.slice(1)}`; + const ref = value.$ref?.split('/').pop(); + if (ref !== undefined && schemas[ref]?.properties !== undefined) { + out.push(...walk(ref, head)); + } else { + out.push(head); + } + } + return out; + }; + return walk('HistoryDailyRow', ''); +} + async function main() { const ast = await openapiTS(new URL(SPEC_URL)); const body = astToString(ast); await writeFile(OUTPUT, header + body, 'utf8'); console.log(`Wrote ${OUTPUT}`); + + const spec = parse(await (await fetch(SPEC_URL)).text()) as { + components: { schemas: Record }; + }; + const columns = dailyColumns(spec.components.schemas); + // A FLOOR NEAR THE REAL COUNT. `< 10` caught a total wipe-out and nothing else: + // a spec that expressed one nested block inline or behind allOf would drop seven + // columns and pass, and the header would gain an always-empty `inParkHours`. + if (columns.length < 30) { + throw new Error( + `only ${String(columns.length)} daily columns, expected 36ish: has the spec moved to allOf/inline blocks?`, + ); + } + for (const block of ['standby', 'singleRider', 'extremeWaits', 'inParkHours']) { + if (!columns.some((c) => c.startsWith(block))) { + throw new Error( + `no ${block} columns: the generator only follows $ref, and this block is no longer one`, + ); + } + } + // The rule is not injective. A collision writes one value into two + // slots under a right-looking header, so it fails the build instead. + const dupes = columns.filter((c, i) => columns.indexOf(c) !== i); + if (dupes.length > 0) throw new Error(`duplicate daily column names: ${dupes.join(', ')}`); + // No eslint-disable on this one: it is a plain array, and an unused directive + // is itself a warning. + const columnsHeader = header.replace('/* eslint-disable */\n', ''); + await writeFile( + COLUMNS_OUTPUT, + `${columnsHeader}\n/** Every scalar on a daily history row, flattened, in spec order. */\n` + + `export const DAILY_COLUMNS = [\n${columns.map((c) => ` '${c}',`).join('\n')}\n] as const;\n`, + 'utf8', + ); + console.log(`Wrote ${COLUMNS_OUTPUT} (${columns.length} columns)`); } main().catch((err) => { diff --git a/src/_generated/dailyColumns.ts b/src/_generated/dailyColumns.ts new file mode 100644 index 0000000..80bda3b --- /dev/null +++ b/src/_generated/dailyColumns.ts @@ -0,0 +1,44 @@ +/** + * AUTO-GENERATED by scripts/regenerate.ts from https://api.themeparks.wiki/docs/v1.yaml. + * Do not edit by hand. Run `npm run regenerate` to update. + */ + +/** Every scalar on a daily history row, flattened, in spec order. */ +export const DAILY_COLUMNS = [ + 'date', + 'firstOperatingAt', + 'lastClosedAt', + 'operatingMinutes', + 'downMinutes', + 'unknownMinutes', + 'standbyMin', + 'standbyP50', + 'standbyMean', + 'standbyP90', + 'standbyMax', + 'singleRiderMin', + 'singleRiderP50', + 'singleRiderMean', + 'singleRiderP90', + 'singleRiderMax', + 'extremeWaitsStandby', + 'extremeWaitsSingleRider', + 'showCount', + 'inParkHoursScheduledMinutes', + 'inParkHoursOperatingMinutes', + 'inParkHoursDownMinutes', + 'inParkHoursUnknownMinutes', + 'inParkHoursStandbyMin', + 'inParkHoursStandbyP50', + 'inParkHoursStandbyMean', + 'inParkHoursStandbyP90', + 'inParkHoursStandbyMax', + 'inParkHoursSingleRiderMin', + 'inParkHoursSingleRiderP50', + 'inParkHoursSingleRiderMean', + 'inParkHoursSingleRiderP90', + 'inParkHoursSingleRiderMax', + 'inParkHoursExtremeWaitsStandby', + 'inParkHoursExtremeWaitsSingleRider', + 'changes', +] as const; diff --git a/src/_generated/schema.ts b/src/_generated/schema.ts index 7e47e6e..ba01091 100644 --- a/src/_generated/schema.ts +++ b/src/_generated/schema.ts @@ -12,8 +12,8 @@ export interface paths { cookie?: never; }; /** - * GET /v1/destinations - * @description Every destination we hold, each with its parks as id and name. This is the entry point: take a destination or park id from here and ask GET /v1/entity/{id}/children for what is inside it, /live for what it is doing, or /schedule for when it is open. Soft-deleted parks are excluded. The list is small and changes when a resort opens or closes a park, which is rarely — cached `public, max-age=300, s-maxage=3600`, and there is no reason to ask for it more than once a run. + * List destinations + * @description Returns every destination with its parks (id and name). This is the place to start: take an id from here and call GET /v1/entity/{id}/children for what is inside it, /live for what it is doing, or /schedule for its opening hours. Cache it; it rarely changes. */ get: operations["getAllDestinations"]; put?: never; @@ -32,8 +32,8 @@ export interface paths { cookie?: never; }; /** - * GET /v1/entity/{id} - * @description The entity document: what this thing IS, not what it is doing. Name, entityType, timezone, location, its place in the destination/park hierarchy, and the ride's own attributes. `id` is an entity's UUID, its slug, or — for a destination only — its upstream external id. An id that resolves to nothing is 404, including one malformed enough that the query itself rejects it. Attributes recorded as tags are spread into the document as top-level keys named after the tag — `minimumHeight`, `mayGetWet` and so on — so the set of keys varies by entity and by park; read the ones you need and ignore the rest rather than expecting a fixed shape. For what the entity is doing right now, GET /v1/entity/{id}/live. Cached `public, max-age=300, s-maxage=3600`: this changes when a park changes its data, which is rarely, and polling it faster buys requests rather than freshness. + * Get an entity + * @description Returns what this entity is: its name, entityType, timezone, location, parent and park, and its attributes. `id` can be an entity's UUID, its slug or, for a destination, its `externalId`. An unknown or malformed id returns 404. Attributes appear as extra top-level keys named after them, such as `minimumHeight` and `mayGetWet`, so the set of keys varies; read the ones you need. For what the entity is doing now, call GET /v1/entity/{id}/live. */ get: operations["getEntityById"]; put?: never; @@ -52,8 +52,8 @@ export interface paths { cookie?: never; }; /** - * GET /v1/entity/{id}/children - * @description The entities beneath this one, as a flat array — id, name, entityType, slug, external id, coordinates and parentId each. `id` is an entity's UUID, its slug, or — for a destination only — its upstream external id. An id that resolves to nothing is 404, including one malformed enough that the query itself rejects it. HOW FAR DOWN depends on what you asked about, and this is the part clients get wrong: a DESTINATION returns every entity in the destination and a PARK returns every entity in the park — the whole subtree, not one level — while every other entityType returns its direct children only. Rebuild the hierarchy from `parentId` rather than assuming one level. Soft-deleted entities are never included. Cached `public, max-age=300, s-maxage=3600`. + * List the entities inside an entity + * @description Returns the entities beneath this one as a flat array, each with id, name, entityType, slug, externalId, location and parentId. `id` can be an entity's UUID, its slug or, for a destination, its `externalId`. An unknown or malformed id returns 404. For a DESTINATION or a PARK you get everything inside it, not just one level; for any other entityType you get its direct children. Rebuild the tree from `parentId`. Removed entities are not listed. */ get: operations["getEntityChildren"]; put?: never; @@ -72,8 +72,12 @@ export interface paths { cookie?: never; }; /** - * GET /v1/entity/{id}/history - * @description History of an entity as one row per change, every row the complete live-data object at that instant (same keys as /live) plus `time` and `changed`. Ask for a park-local day (`date=YYYY-MM-DD`), a range of days (`from`/`to`, inclusive), or RFC 3339 instants with an offset (half-open). No parameters means today. At most 31 days per call. `opening` is the state effective at the start of the range; `coverage.firstRecordedAt` is the first day this entity has any history. Anonymous callers see 7 days, a free API key 30; deeper ranges return 403 HISTORY_WINDOW_EXCEEDED with the earliest allowed date. History requests have their own hourly budget, separate from the per-minute REST limit: 60 an hour without a key, 600 with a free key, 1200 on Pro, 3000 on Business, unmetered on Enterprise (published per tier in GET /tiers as limits.historyRequestsPerHour); over it returns 429 HISTORY_RATE_LIMITED with retryAfter. The per-minute REST limit still applies first and answers with the API-wide 429 body. A range that includes TODAY may be up to an hour old for callers without an API key, so a poller can see a response up to an hour behind the live feed; completed days are cached for longer because they cannot change. For the current state of an entity rather than its history, GET /v1/entity/{id}/live is not cached that way. FOR A PARK, this path answers the WHOLE PARK and the 200 is a different schema: `HistoryParkRawEnvelope`, carrying an `entities[]` array — one entry per entity of the park that has history, ascending by name, the park itself included when it has history of its own — each entry holding the same `coverage`, `opening` and `history` block a single entity gets. An entity with no history is absent from that array rather than present and empty. A park range is 1 park-local day, not 31: a day of a park is every recorded change for every entity in it, so a wider range is 400 RANGE_TOO_LONG and the call is never paged. Every other entityType — a DESTINATION included — returns the single-entity envelope described above, so the 200 is a union of two schemas and the entity's `entityType` is what selects between them. Either way the call costs ONE unit of the hourly history budget, whatever the park's size. + * History of an entity + * @description Returns an entity's history as one row per change. Each row is the entity's complete live data at that moment (the same keys as /live) plus `time` and `changed`. Ask for one park-local day (`date=YYYY-MM-DD`), a range of days (`from` and `to`, both inclusive), or RFC 3339 instants with an offset (`to` exclusive). With no parameters you get today. A call covers up to 31 days. The `opening` object is the state at the start of the range: the last reading of every field, however old, with `opening.observedAt` saying when it was seen, so the first row is already complete. `coverage.firstRecordedAt` is the first day we hold anything for this entity. + * + * For a park, the 200 is `HistoryParkRawEnvelope`: an `entities[]` array with the same `coverage`, `opening` object and `history` for each entity that has history, the park included if it has its own. A park call covers 1 park-local day, and a longer range returns 400 RANGE_TOO_LONG. Every other entityType gets the single-entity shape, so the 200 is a union of two schemas selected by `entityType`. A call costs one unit of the hourly history budget, however large the park. + * + * See "Limits" above for the history window and hourly budget. */ get: operations["getHistory"]; put?: never; @@ -92,8 +96,10 @@ export interface paths { cookie?: never; }; /** - * GET /v1/entity/{id}/history/coverage - * @description What history we hold for this entity, broken down per live-data field: the first and last park-local day each field (`status`, `queue.STANDBY`, `showtimes`, and so on) was reported, keyed by the same live-data path GET /v1/entity/{id}/live and GET /v1/entity/{id}/history and .../history/daily use, so a key matches straight across all four. A field the entity never reported is ABSENT from `kinds`. Read `last` as "newest day in the ARCHIVE", not "last day reported": the archive is written two to three days behind live data, so a field being published right now still has a `last` a few days old, and an active field is indistinguishable from a withdrawn one here. To ask whether a field is still live, call GET /v1/entity/{id}/live. For how recent a day you can actually ASK for, read `retrievableThrough`: history calls serve recent readings as well as the archive, so it is normally TODAY for an entity still reporting and therefore sits two to three days AHEAD of `lastRecordedAt` — that gap is expected, not a fault. For an entity that stopped reporting long ago it equals `lastRecordedAt`, so it never promises data that is not there. `firstRecordedAt` and `lastRecordedAt` are the entity-wide ARCHIVE bounds; an entity with nothing recorded returns the document with both null and `kinds: {}`, never a 404 — "we hold nothing for this entity" is a real, actionable answer, and a different claim from "this entity does not exist". No parameters, no paging and no tier window: the document is the same handful of day strings whatever the entity's history depth, so this call is not entitlement-gated. It still shares the hourly history budget with GET /v1/entity/{id}/history and .../history/daily, separate from the per-minute REST limit: 60 an hour without a key, 600 with a free key, 1200 on Pro, 3000 on Business, unmetered on Enterprise (published per tier in GET /tiers as limits.historyRequestsPerHour); over it returns 429 HISTORY_RATE_LIMITED with retryAfter. The per-minute REST limit still applies first and answers with the API-wide 429 body. The response is the same bytes for every caller, so it is publicly cacheable for an hour — every value in it is a whole day, so a cached copy can only differ from a fresh one in the first hour after park-local midnight, when `retrievableThrough` may still name the previous day. + * History held for an entity + * @description Returns the history we hold for this entity, per field: the first and last park-local day we recorded each field (`status`, `queue.STANDBY`, `showtimes` and so on), keyed by the same live-data paths as /live, /history and /history/daily. Fields the entity never reported are left out of `kinds`. Recording runs 2 to 3 days behind live data, so `last` and `lastRecordedAt` usually trail today by that much, even for a field still being published; use /live to see what is current. `retrievableThrough` is the newest day you can ask /history and /history/daily for, usually today. An entity with nothing recorded returns `kinds: {}` with null days, not 404. For a park, the 200 is `HistoryParkCoverageDocument`, a summary across the park's entities. This call takes no parameters, is not paged and has no history window, but it does spend the hourly history budget. + * + * See "Limits" above for the history window and hourly budget. */ get: operations["getHistoryCoverage"]; put?: never; @@ -112,8 +118,24 @@ export interface paths { cookie?: never; }; /** - * GET /v1/entity/{id}/history/daily - * @description One summary row per park-local day: how long the entity was OPERATING and DOWN, when it first opened and last closed, minute-weighted standby and single-rider wait statistics, a show count where the entity publishes showtimes, and how many changes were recorded. Ask for a park-local day (`date=YYYY-MM-DD`), a range of days (`from`/`to`, inclusive), or RFC 3339 instants with an offset; `range` always comes back as park-local days, because a row summarises a whole day. No parameters means today. At most 3660 days per call, and the response is never paged. A day with no data is ABSENT from `days` — there is no row of zeroes, because "we have nothing for this day" is a different claim from "observed, closed all day" — and `standby`, `singleRider` and `showCount` are omitted rather than null when the entity published nothing of that kind. `coverage.firstRecordedAt` is the first day this entity has any history. Anonymous callers see 7 days, a free API key 30; deeper ranges return 403 HISTORY_WINDOW_EXCEEDED with the earliest allowed date. These calls share the hourly history budget with GET /v1/entity/{id}/history, separate from the per-minute REST limit: 60 an hour without a key, 600 with a free key, 1200 on Pro, 3000 on Business, unmetered on Enterprise (published per tier in GET /tiers as limits.historyRequestsPerHour); over it returns 429 HISTORY_RATE_LIMITED with retryAfter. The per-minute REST limit still applies first and answers with the API-wide 429 body. FOR A PARK, this path answers the WHOLE PARK and the 200 is a different schema: `HistoryParkDailyEnvelope`, carrying an `entities[]` array — one entry per entity of the park that has history, ascending by name, the park itself included when it has history of its own — instead of this envelope's `coverage` and `days`. An entity with no history is absent from that array rather than present and empty. A park call is also the one PAGED call in this family: it serves at most 31 park-local days and `next` is an absolute URL for the rest, with your other parameters preserved. Every other entityType — a DESTINATION included — returns the single-entity envelope described above, so the 200 is a union of two schemas and the entity's `entityType` is what selects between them. Either way the call costs ONE unit of the hourly history budget, whatever the park's size. + * Daily summaries of an entity's history + * @description Returns one summary row per park-local day: minutes operating, down and unknown, first opening and last close, wait statistics, show count and change count. Parameters are as for /history; rows cover whole days. A call covers up to 3660 days and is never paged. Days with no data are left out. So are `standby`, `singleRider` and `showCount` when there was nothing of that kind. + * + * A run lasts from a change to OPERATING until the next change to CLOSED or REFURBISHMENT, DOWN periods included. A row is counted like this: + * + * - Outside the park's published hours, an OPERATING status counts as `unknownMinutes`, not `operatingMinutes`, after 4 hours or more unconfirmed. On a day with published hours, only a status change confirms it; a status carried over midnight is unconfirmed. On other days, any change but `showtimes` confirms it, across midnight. + * - A wait counts only while the entity is OPERATING and on the day it was posted. The wait showing when a ride opens counts if it last changed within 24 hours. + * - `extremeWaits` counts readings of 480 minutes or more, usually feed errors. Nothing is left out: those readings stay in the statistics, and `extremeWaits` flags them. + * - `firstOperatingAt` is null when the entity did not change to OPERATING that day; it may still have operated, carried over from the day before. + * - `lastClosedAt` is when the last run that started that day closed, which can be after midnight. The next day's row does not repeat it. + * - `inParkHours` repeats the numbers for the park's published hours, when it published any. + * - A row without `unknownMinutes` was counted under earlier rules. + * + * To rebuild a row from /history, use that day's and the next day's history, the park's schedule for both days, `opening.observedAtByKind` and `opening.observedAt`. + * + * For a park, the 200 is `HistoryParkDailyEnvelope`, with an `entities[]` array in place of `coverage` and `days`. Park calls are paged, 31 park-local days at a time, with `next` for the rest. Every other entityType gets the single-entity shape, so the 200 is a union of two schemas selected by `entityType`. A call costs one unit of the hourly history budget, however large the park. + * + * See "Limits" above for the history window and hourly budget. */ get: operations["getHistoryDaily"]; put?: never; @@ -132,8 +154,8 @@ export interface paths { cookie?: never; }; /** - * GET /v1/entity/{id}/live - * @description Live data for this entity AND everything beneath it, in one call: wait times, ride status, return-time and boarding-group windows, and show times. `id` is an entity's UUID, its slug, or — for a destination only — its upstream external id. An id that resolves to nothing is 404, including one malformed enough that the query itself rejects it. An entity with nothing to report is absent from `liveData` rather than present and empty, so treat a missing entry as no data rather than as a closed ride. `?entityType=ATTRACTION,SHOW` filters the array to those types (comma-separated). Cached `public, max-age=60, s-maxage=60`: the collectors run on a cadence, and a faster poll returns the same body. For what an entity reported in the past rather than now, GET /v1/entity/{id}/history. + * Live data for an entity + * @description Returns live data for this entity and everything beneath it: status, wait times, return-time and boarding-group windows, and show times. `id` can be an entity's UUID, its slug or, for a destination, its `externalId`. An unknown or malformed id returns 404. An entity with nothing to report is left out of `liveData`; treat that as no data. Filter by type with `entityType`. The data is current to about a minute. For past data, call GET /v1/entity/{id}/history. */ get: operations["getEntityLiveData"]; put?: never; @@ -152,8 +174,8 @@ export interface paths { cookie?: never; }; /** - * GET /v1/entity/{id}/schedule - * @description Opening hours for this entity, for the default window: today through the next 30 days, anchored to the entity's own timezone rather than the caller's or UTC. `id` is an entity's UUID, its slug, or — for a destination only — its upstream external id. An id that resolves to nothing is 404, including one malformed enough that the query itself rejects it. Days the park has not published are absent from the array rather than present as closed. For a specific month, including a past one, use GET /v1/entity/{id}/schedule/{year}/{month}. Cached `public, max-age=300, s-maxage=3600`. + * Upcoming opening hours + * @description Returns opening hours from today through the next 30 days, in the entity's own timezone. `id` can be an entity's UUID, its slug or, for a destination, its `externalId`. An unknown or malformed id returns 404. Days the park has not published are left out. For a particular month, past months included, call GET /v1/entity/{id}/schedule/{year}/{month}. */ get: operations["getEntitySchedule"]; put?: never; @@ -172,8 +194,8 @@ export interface paths { cookie?: never; }; /** - * GET /v1/entity/{id}/schedule/{year}/{month} - * @description Opening hours for one calendar month in the entity's own timezone. `month` is TWO digits, 01-12 — `/2026/9` is 400, `/2026/09` is right — and `year` is a four-digit year between 1970 and 2150; anything else is 400 before the entity is even looked up. `id` is an entity's UUID, its slug, or — for a destination only — its upstream external id. An id that resolves to nothing is 404, including one malformed enough that the query itself rejects it. Past months are served from what was recorded at the time and are not backfilled, so a month before this entity was collected comes back empty rather than 404. Days the park has not published are absent rather than present as closed. Cached `public, max-age=300, s-maxage=3600`. + * Opening hours for a month + * @description Returns opening hours for one calendar month, in the entity's own timezone. `month` must be two digits (`/2026/09`, not `/2026/9`) and `year` a number from 1970 to 2150; anything else returns 400. `id` can be an entity's UUID, its slug or, for a destination, its `externalId`. An unknown or malformed id returns 404. Past months show what was published at the time. A month before we started recording this entity comes back empty. Days the park has not published are left out. */ get: operations["getEntityScheduleYearMonth"]; put?: never; @@ -184,10 +206,49 @@ export interface paths { patch?: never; trace?: never; }; + "/v1/me": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Your plan and remaining budgets + * @description Returns your plan and what is left of it: `tier`, how far back your history calls reach (`historyDays`, or `historyAllArchive: true` when you can read everything we hold), the earliest day you may ask for, and two budgets: `rateLimit`, the request limit every call counts against, and `historyRateLimit`, the hourly history budget. Each budget has `unmetered`; when it is false, it also has `limit`, `windowSeconds`, `remaining` and `reset`, the same figures as the RateLimit headers. Use it on an unmetered plan, which gets no such headers, or to check your history budget without spending it. The call counts as one ordinary request and needs an API key. + */ + get: operations["getMe"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; } export type webhooks = Record; export interface components { schemas: { + /** @description 401: this endpoint needs an API key (X-API-Key header) or a session token, and none was sent or the one sent was not accepted. A revoked, mistyped or truncated key answers this too. */ + AuthenticationRequired: { + /** + * @description Always false. + * @enum {boolean} + */ + success: false; + error: { + /** @enum {string} */ + type: "Authentication failed"; + /** @description Says whether no credential was sent or the one sent was not accepted. */ + message: string; + /** + * @description Repeats the HTTP status. + * @enum {integer} + */ + code?: 401; + }; + }; /** * @description State of boarding group availability * @enum {string} @@ -200,7 +261,7 @@ export interface components { name: string; /** @description URL-friendly slug for the destination */ slug?: string | null; - /** @description External entity ID from the source data provider */ + /** @description The park operator's own id for this destination */ externalId?: string | null; /** @description Array of parks within this destination */ parks: components["schemas"]["Park"][]; @@ -239,7 +300,7 @@ export interface components { timezone?: string; children?: components["schemas"]["EntityChild"][]; }; - /** @description A single entity. Beyond the properties listed here, an entity may carry additional tag-derived properties named after the tag's slug, for example `minimumHeight` (integer, centimetres) or `mayGetWet` (boolean). The set is open-ended and driven by data rather than fixed by this contract, so clients should read them defensively rather than assume any particular tag is present. */ + /** @description A single entity. It may also carry attribute keys not listed here, named after the attribute, e.g. `minimumHeight` (integer, centimetres) or `mayGetWet` (boolean). Which ones appear varies by entity, so read the ones you need and do not assume any is present. */ EntityData: { /** @description Unique entity identifier */ id: string; @@ -260,17 +321,17 @@ export interface components { /** @description Entity timezone */ timezone: string; location?: components["schemas"]["EntityLocation"]; - /** @description Identifier used by the source data provider. */ + /** @description The park operator's own id for this entity. */ externalId?: string; /** @description URL-friendly slug. Served for destinations. */ slug?: string | null; } & { [key: string]: unknown; }; - /** @description 400: a path parameter is the wrong shape. Checked before the entity is looked up, so a bad `year` or `month` answers 400 whether or not the id exists. */ + /** @description 400: a path parameter is the wrong shape, such as a one-digit `month` or a `year` outside the accepted range. An unknown id returns 404 first. */ EntityInvalidParameter: { /** - * @description Always false. Branch on this rather than on the status alone. + * @description Always false. * @enum {boolean} */ success: false; @@ -319,10 +380,10 @@ export interface components { /** @description Longitude coordinate of the entity location */ longitude?: number | null; }; - /** @description 404: nothing resolved from `id`. An id is a UUID, a slug, or — for a destination — its upstream external id; an id malformed enough that the lookup itself rejects it answers 404 as well, rather than 400 or 500. The body is the same whether the entity never existed or has been removed, so it cannot be used to tell those apart. */ + /** @description 404: nothing matches `id` (a UUID, a slug, or a destination's externalId; a malformed id is 404 too). The answer is the same whether it never existed or was removed. */ EntityNotFound: { /** - * @description Always false. Branch on this rather than on the status alone. + * @description Always false. * @enum {boolean} */ success: false; @@ -359,11 +420,11 @@ export interface components { HistoryCoverage: { /** * Format: date - * @description First park-local day with recorded history for this entity, or null when nothing has been archived yet. Per-kind detail and gaps: GET /v1/entity/{id}/history/coverage. + * @description First park-local day we hold anything for this entity, or null. Per-field detail: GET /v1/entity/{id}/history/coverage. */ firstRecordedAt: string | null; }; - /** @description What history is actually held for one entity, broken down per live-data field. Two different questions, answered separately: `firstRecordedAt`/`lastRecordedAt` and the per-field spans describe the ARCHIVE, while `retrievableThrough` is the newest day a history call could return for this entity — which is normally today, and normally two to three days AHEAD of `lastRecordedAt`. An entity with nothing recorded is a 200 with kinds: {} and every day null, never a 404: "we hold nothing for this entity" is a real, actionable answer and a different claim from "this entity does not exist". */ + /** @description What history we hold for one entity, per field. An entity with nothing recorded is a 200 with `kinds: {}` and null days, not 404. For a PARK the same call returns HistoryParkCoverageDocument. */ HistoryCoverageDocument: { id: string; name: string; @@ -374,20 +435,20 @@ export interface components { timezone: string; /** * Format: date - * @description First park-local day with any recorded history for this entity, or null when nothing has been archived yet. May legitimately be earlier than any individual kind's first: the two are recorded independently and answer slightly different questions. + * @description First park-local day we hold anything for this entity, or null. */ firstRecordedAt: string | null; /** * Format: date - * @description The newest `last` across `kinds` — the most recent park-local day we hold anything at all for this entity — or null when nothing is recorded. Like the per-field `last`, this tracks the ARCHIVE and lags live data by two to three days, so it sits in the past for an entity reporting normally. + * @description The newest `last` across `kinds`, or null. */ lastRecordedAt: string | null; /** * Format: date - * @description The newest park-local day GET /v1/entity/{id}/history and .../history/daily could return data for this entity — what you can ASK FOR, as opposed to what has been filed. Normally TODAY for an entity still reporting, because those endpoints answer from recent readings as well as the archive, and it therefore sits AHEAD of `lastRecordedAt` by two to three days for a healthy entity. That gap is the whole point of this field: `lastRecordedAt` and every per-field `last` describe the ARCHIVE only, and reading them as capability is what makes coverage look as though it has stopped a couple of days short. It is the LATER of `lastRecordedAt` and the newest day recent readings can still answer for — so an entity that stopped reporting long ago reports its archive day here, NOT today, and the field never promises data that is not there. Recent readings do not go back indefinitely, so a day is reported here only if one of those endpoints can actually return it: an entity whose last reading is old enough falls back to its archive day rather than naming the day that reading was taken. Null only when we hold nothing for this entity in either place. A request for a range up to this day can still be narrowed by your tier's history window, which bounds how far BACK you may ask, never how recent. + * @description The newest day you can ask /history and /history/daily for: usually today for an entity still reporting, or its last recorded day for one that stopped long ago. Null when we hold nothing. Your history window limits how far back you can ask. */ retrievableThrough: string | null; - /** @description Keyed by LIVE-DATA PATH (status, queue.STANDBY, showtimes, ...), not by the internal kind name, so a key matches straight against what GET /v1/entity/{id}/live and GET /v1/entity/{id}/history return. A kind the entity never reported is ABSENT here and absent from every /history row — there is no zero-span entry for it. See `last` for why a span ending in the past does NOT mean the field stopped being reported: the archive lags live data by two to three days, so an actively published field ends in the past too. Every span here is archive-only; `retrievableThrough` is the entity-wide answer to how recent a day you can actually ask for. */ + /** @description Keyed by live-data path (status, queue.STANDBY, showtimes and so on), as /live and /history name them. Fields the entity never reported are left out. */ kinds: { [key: string]: components["schemas"]["HistoryCoverageKindSpan"]; }; @@ -401,11 +462,11 @@ export interface components { first: string; /** * Format: date - * @description Newest park-local day this field appears in the ARCHIVE. This is not the same as the last day the field was reported, and it does NOT mean the field has stopped: the archive is written two to three days behind live data, so a field being published right now still has a `last` a few days in the past. Every active field looks the same as a withdrawn one here. Use this to know how far back the archive goes and how current it is, not to decide whether a field is still live — GET /v1/entity/{id}/live answers that directly. + * @description The newest day we hold this field for. The coverage call says how far it trails today. */ last: string; }; - /** @description A day-by-day summary of one entity's history. GET /v1/entity/{id}/history/daily returns this shape for every entityType EXCEPT PARK; for a PARK the same path returns HistoryParkDailyEnvelope, which carries an entities[] array instead of this envelope's coverage/days block. range.from and range.to ALWAYS echo park-local calendar days (YYYY-MM-DD), even when the call supplied RFC 3339 instants, because this endpoint summarises whole park-local days and an instant echo would claim a precision the rows do not have. TODAY's row is the day so far — its counters cover only elapsed minutes and grow as the day does — and a response to a request without an API key may be up to an hour old, so a poller can see a value that far behind the live feed. If you need the current state of an entity rather than its day so far, GET /v1/entity/{id}/live is not cached that way. */ + /** @description One entity's daily summary. For a PARK the same call returns HistoryParkDailyEnvelope. range is always whole park-local days (YYYY-MM-DD), and today's row is the day so far. */ HistoryDailyEnvelope: { id: string; name: string; @@ -416,12 +477,33 @@ export interface components { timezone: string; range: components["schemas"]["HistoryRange"]; coverage: components["schemas"]["HistoryCoverage"]; - /** @description One row per park-local day, ascending by date. A day with no data is ABSENT: there is no row of zeroes, because "we have nothing for this day" is a different claim from "observed, closed all day". */ + /** @description One row per day, ascending. Days with no data are left out. */ days: components["schemas"]["HistoryDailyRow"][]; - /** @description URL of the next page, or null. Always null for a single entity: a daily call is unpaged. Park calls page; see HistoryParkDailyEnvelope. */ + /** @description Always null for a single entity. */ next: string | null; }; - /** @description One park-local day of an entity's history reduced to the numbers a crowd calendar needs. standby, singleRider and showCount are ABSENT rather than null when there is nothing to report: an absent standby means no numeric standby wait was in force while OPERATING for a whole sampled minute that day — the statistics are sampled at minute resolution, so a wait published only inside a sub-minute window produces no block at all, even though /history records it and the day's operatingMinutes count it. The same applies to singleRider. An absent showCount means the entity published no showtimes. showCount counts distinct performance start times in the local day. */ + /** @description How many wait readings of 480 minutes or more the day's statistics include. Such readings are usually feed errors (values like 999); they are counted here as a flag and stay in every statistic. Counted per reading while OPERATING. Present only when there were some, and then with both counts. */ + HistoryDailyExtremeWaits: { + /** @description Standby readings of 480 minutes or more included in the day's standby statistics. */ + standby: number; + /** @description Single-rider readings of 480 minutes or more included in the day's single-rider statistics. */ + singleRider: number; + }; + /** @description The day's numbers limited to the park's published hours. Present only when the park published hours that day. */ + HistoryDailyInParkHours: { + /** @description Minutes of the day inside the park's published hours: its schedule entries of type OPERATING, EXTRA_HOURS and TICKETED_EVENT. Hours from the previous day's schedule that run past midnight are not included. Today counts elapsed minutes only. */ + scheduledMinutes: number; + /** @description Of scheduledMinutes, the minutes the entity was OPERATING. */ + operatingMinutes: number; + /** @description Of scheduledMinutes, the minutes the entity was DOWN. */ + downMinutes: number; + /** @description Of scheduledMinutes, the minutes with no known status. */ + unknownMinutes: number; + standby?: components["schemas"]["HistoryDailyStats"]; + singleRider?: components["schemas"]["HistoryDailyStats"]; + extremeWaits?: components["schemas"]["HistoryDailyExtremeWaits"]; + }; + /** @description One day of an entity's history. standby, singleRider, extremeWaits, showCount and inParkHours are absent when there is nothing to report; a standby block needs a valid wait showing for at least one whole minute. A row without unknownMinutes was counted under earlier rules, and also lacks extremeWaits and inParkHours. */ HistoryDailyRow: { /** * Format: date @@ -430,39 +512,43 @@ export interface components { date: string; /** * Format: date-time - * @description UTC instant (whole seconds) the entity first became OPERATING on this park-local day, or null if it never did. A day that opened already OPERATING reports the start of the local day. + * @description When the entity changed to OPERATING that day (UTC, whole seconds). null when it did not change to OPERATING that day; it may still have operated, carried over from the day before. */ firstOperatingAt: string | null; /** * Format: date-time - * @description UTC instant (whole seconds) of the last transition out of OPERATING or DOWN into CLOSED or REFURBISHMENT after firstOperatingAt, or null if there was none. For a park closing after local midnight this instant falls on the following UTC day. + * @description When the last run that started this day closed (UTC, whole seconds). A run lasts from a change to OPERATING until the next change to CLOSED or REFURBISHMENT, DOWN periods included. A run still going at midnight closes the next day and is reported on this row. null if it did not close by the end of the next day, or if the close came after 4 hours or more outside published hours with nothing to confirm the run (see unknownMinutes: a change of status on a day with published hours, a change to any field other than showtimes on a day without). */ lastClosedAt: string | null; - /** @description Minutes of this park-local day the entity was OPERATING. Minutes with no observed state count as neither operating nor down, so the two counters need not add up to the length of the day. For TODAY the count covers only the minutes that have already elapsed, so it grows through the day and is final once the day ends: a caller polling today's row sees it rise, which is the day filling in rather than the answer changing. */ + /** @description Minutes OPERATING with a known status. Today counts elapsed minutes only. CLOSED and REFURBISHMENT minutes are not counted, so operatingMinutes, downMinutes and unknownMinutes need not add up to the day. */ operatingMinutes: number; - /** @description Minutes of this park-local day the entity was DOWN. As with operatingMinutes, today's count covers only the elapsed part of the day. */ + /** @description Minutes DOWN. Today counts elapsed minutes only. */ downMinutes: number; + /** @description Minutes with no known status: none was reported, or the entity was OPERATING outside the park's published hours with nothing to confirm it for 4 hours or more. On a day the park published hours, only a change of status confirms it, and a status carried over midnight is unknown outside those hours until it changes. On a day with no published hours, or for an entity outside any park, a change to any field other than showtimes (status, a wait or anything else) confirms it, and for a status carried over midnight the hours count from the last change before midnight. Absent on rows counted under earlier rules. */ + unknownMinutes?: number; standby?: components["schemas"]["HistoryDailyStats"]; singleRider?: components["schemas"]["HistoryDailyStats"]; + extremeWaits?: components["schemas"]["HistoryDailyExtremeWaits"]; /** @description Distinct performance start times whose park-local day is this day. Present only for entities that published showtimes on the day. */ showCount?: number; - /** @description Number of history rows recorded on this day, i.e. instants at which any kind changed. Short flaps are counted as they happened; this is not a cleaned figure. */ + inParkHours?: components["schemas"]["HistoryDailyInParkHours"]; + /** @description How many times any field changed this day. */ changes: number; }; - /** @description Wait statistics for one park-local day. The percentiles and the mean are weighted by the MINUTES the wait was posted rather than by the number of readings, so a wait that stood for three hours counts three hours and a brief flap does not drag the median; they are sampled at minute resolution and the percentiles are nearest-rank, never interpolated. `min` and `max` are TRUE extremes over every value posted, so a spike too short to be sampled still shows there. Only periods where the entity was OPERATING and published a numeric wait count at all. The block is absent when it never did. */ + /** @description Wait statistics for the day. p50, mean and p90 are weighted by how many minutes each wait was showing; min and max are the extremes of every value posted. Only minutes where the entity was OPERATING (not unknown) and the wait was valid count. A wait is valid during the run and on the day it was posted; the wait showing when a ride opens counts from the opening if it last changed within 24 hours; a wait carried over midnight does not count until it changes. Nothing the park reported is excluded: a reading of 480 minutes or more stays in these statistics and is counted in the row's `extremeWaits`. Absent when no minute counted. */ HistoryDailyStats: { - /** @description Lowest wait, in minutes, the entity published while OPERATING that day. A true extreme over every value posted, including one that stood for less than a minute — so unlike the percentiles below it is not minute-weighted. */ + /** @description Lowest wait posted during the counted minutes, including one that showed for under a minute. */ min: number; - /** @description Median wait, nearest-rank over the minute weights (a value actually posted, never interpolated). */ + /** @description Median wait, weighted by minutes; always a value that was actually posted. */ p50: number; /** @description Minute-weighted average wait, rounded to the nearest whole minute. */ mean: number; - /** @description 90th-percentile wait, nearest-rank over the minute weights. */ + /** @description 90th-percentile wait, weighted by minutes. */ p90: number; - /** @description Highest wait, in minutes, the entity published while OPERATING that day. A true extreme, as min is: a spike that lasted forty seconds counts here and is invisible to the percentiles, which is the intended difference between the two halves of this block — extremes answer "what did it ever reach", percentiles answer "what was it usually like". */ + /** @description Highest wait posted during the counted minutes, including a brief spike. It can be a feed error: the row's extremeWaits counts readings of 480 minutes or more. */ max: number; }; - /** @description One entity's history as full-state change rows. GET /v1/entity/{id}/history returns this shape for every entityType EXCEPT PARK; for a PARK the same path returns HistoryParkRawEnvelope, which carries an entities[] array instead of this envelope's coverage/opening/history block. */ + /** @description One entity's history. For a PARK the same call returns HistoryParkRawEnvelope instead. */ HistoryEnvelope: { id: string; name: string; @@ -476,10 +562,10 @@ export interface components { opening: components["schemas"]["HistoryOpening"]; /** @description Ascending by time. */ history: components["schemas"]["HistoryRow"][]; - /** @description URL of the next page, or null. Always null for a single entity (a call covers up to 31 days). Park calls page; see HistoryParkRawEnvelope. */ + /** @description Always null for a single entity. */ next: string | null; }; - /** @description 502: the range needs archived history and that backend is temporarily unavailable. The request is retryable. */ + /** @description 502: history is temporarily unavailable. Retry shortly. */ HistoryErrorBackendUnavailable: { error: { /** @enum {string} */ @@ -513,16 +599,16 @@ export interface components { message: string; }; }; - /** @description 400: the range is longer than the path allows. The cap is NOT the same on every path, and this one error type is returned by all of them: GET /v1/entity/{id}/history allows 31 park-local days for a single entity and 1 for a PARK; GET /v1/entity/{id}/history/daily allows 3660 (ten years) for a single entity, and for a PARK serves 31 days a page and gives you `next` for the rest. The message names the cap that applied. Split the ask into consecutive calls. */ + /** @description 400: the range is too long. The limit is not the same on every path: 31 days on /history (1 for a park), and 3660 on /history/daily (a park pages at 31 instead). Split the range into shorter calls. */ HistoryErrorRangeTooLong: { error: { /** @enum {string} */ type: "RANGE_TOO_LONG"; - /** @description Names the cap that was exceeded and the span that was asked for, e.g. "A history call covers at most 31 park-local days (2026-01-01 to 2026-03-01 is 60). Ask for a shorter range." The number is the cap for the path that answered, not a constant: read it from the message rather than hard-coding 31. */ + /** @description Names the limit that applied and your span, e.g. "A history call covers at most 31 park-local days (2026-01-01 to 2026-03-01 is 60)." */ message: string; }; }; - /** @description 429: the caller's hourly history request budget is spent. The budget is separate from the per-minute REST limit and is published per tier in GET /tiers as limits.historyRequestsPerHour (anonymous.historyRequestsPerHour for keyless calls). */ + /** @description 429: your hourly history budget is spent. It is separate from the per-minute limit; the limits per plan are on the pricing page. */ HistoryErrorRateLimited: { error: { /** @enum {string} */ @@ -537,7 +623,7 @@ export interface components { error: { /** @enum {string} */ type: "HISTORY_WINDOW_EXCEEDED"; - /** @description A fact and a date, naming no plan and selling nothing. e.g. "This key can see history back to 2026-09-08 (7 days).", or "Requests without an API key can see history back to 2026-09-08 (7 days)." when you sent no key. */ + /** @description e.g. "This key can see history back to 2026-09-08 (7 days)." */ message: string; /** * Format: date @@ -546,7 +632,7 @@ export interface components { earliestAllowedDate: string; }; }; - /** @description The full live-data state effective at the start of the range, in the same shape as a row. A key is present only when the entity has that kind. When the value is unknown at that instant (typically an older range, answered from the archive rather than from recent readings) the kind carries its EMPTY live value rather than a null container — an unknown standby is {"waitTime": null}, an unknown showtimes list is [] — and status, which has no empty value, is null. */ + /** @description The `opening` object: the complete state at the start of the range, in the same shape as a row. A field is present only if the entity has it. An unknown value is its empty form (standby `{"waitTime": null}`, showtimes `[]`); an unknown status is null. */ HistoryOpening: { /** * Format: date-time @@ -555,21 +641,29 @@ export interface components { time: string; /** * Format: date-time - * @description The UTC instant (whole seconds) at which this state was actually OBSERVED, as opposed to `time`, which is the start of the range you asked for. The two are different questions and only this one tells you whether to trust the state. - * - * A state carried forward from before the range has an `observedAt` BEFORE `time` — sometimes long before, because the lookup is deliberately unbounded and returns the last reading of each kind at any age. So a day we hold nothing for still reports the newest state we ever saw, which is usually right and occasionally very wrong: a ride whose feed simply stopped mid-operation carries its last wait time forward indefinitely. - * - * Compare it against `range.from` to decide. Equal to or after the range start means the state was seen inside the range. Before it means carried forward, and how far back you tolerate is yours to choose — a reading an hour before the day began is ordinary, one from three weeks earlier is not evidence about this day. Absent means nothing survives to carry: we can say nothing at all about the state at the start of this range. - * - * It is the NEWEST instant any kind in this opening was seen. Kinds can be observed at different moments, so an older kind may be staler than this field suggests; treat it as the most generous reading of the opening's age, not a guarantee about every key. Days that genuinely have data are better read from the rows, which carry their own `time`. + * @description When this state was last seen (UTC, whole seconds): the newest entry in `observedAtByKind`. Earlier than `time` means it was carried in from before the range, possibly from long ago; a ride whose feed stopped keeps its last values. Absent means we hold nothing for this entity from before the range, unless `degraded` is set. */ observedAt?: string; + /** @description When each field of the opening was last seen (UTC, whole seconds), keyed by its path: `status`, `queue.STANDBY`, `showtimes` and so on. A wait that did not change for weeks is dated weeks back, so check a queue's own entry before treating it as current. Usually, but not always, the moment the value last changed. */ + observedAtByKind?: { + [key: string]: string; + }; + /** + * @description Present, and true, when we could not look far enough back for this response, so the opening may be missing a field we hold. Ask again in a minute for the full opening. + * @enum {boolean} + */ + degraded?: true; + /** + * @description Why the lookup was cut short: `timeout`, `error`, or `capacity` when it needed more reading than one request is allowed. + * @enum {string} + */ + degradedReason?: "timeout" | "error" | "capacity"; /** @description Live status at the start of the range; null when unknown. */ status?: string | null; queue?: components["schemas"]["LiveQueue"]; showtimes?: components["schemas"]["LiveShowTime"][] | null; }; - /** @description How far back each entity's record of one field goes, as counts. Calendar years, measured from the day the document was built (`summary.measuredOn`). Counts, not a percentage or an average: a single figure for a park would hide that most of one park's standby entities have two to four years while its oldest reach back to the start of the archive. */ + /** @description How far back each entity's record of this field goes, as counts per band of calendar years, measured from `summary.measuredOn`. */ HistoryParkCoverageDepth: { /** @description Entities whose record of this field goes back four calendar years or more. */ fourYearsPlus: number; @@ -577,17 +671,17 @@ export interface components { twoToFourYears: number; /** @description Entities with at least one and under two calendar years. */ oneToTwoYears: number; - /** @description Entities with under a calendar year, typically something that opened recently rather than a gap. */ + /** @description Entities with under a calendar year, usually something that opened recently. */ underOneYear: number; }; - /** @description What history is held across a whole PARK. GET /v1/entity/{id}/history/coverage returns this shape when the entity is a PARK; every other entityType, a DESTINATION included, returns HistoryCoverageDocument for that entity alone. A park records nothing itself, so without this rollup the honest-looking answer for a park would be "nothing". It reports depth and breadth only: it does not detect missing days and does not judge whether a recorded value was correct. */ + /** @description What history we hold across a whole PARK, returned by /history/coverage when the entity is a PARK. Its names map to the single-entity document: `fields` is `kinds`, and each `from` and `newest` is a `first` and `last`. It reports depth and breadth; it does not find missing days. */ HistoryParkCoverageDocument: { id: string; name: string; entityType: string; parentId: string | null; destinationId: string | null; - /** @description IANA timezone the park-local days are resolved in. The park's children inherit it. */ + /** @description IANA timezone of the park. */ timezone: string; summary: components["schemas"]["HistoryParkCoverageSummary"]; /** @description Keyed by live-data field path, in live-data order. */ @@ -597,7 +691,7 @@ export interface components { /** @description Every entity of the park history is held for. */ entities: components["schemas"]["HistoryParkCoverageEntity"][]; }; - /** @description One entity of the park that history is held for. An entity nothing is held for is ABSENT rather than listed as empty: it is usually a parade, a show or a land, which never had a queue to record, and listing it would read as a gap. */ + /** @description One entity we hold history for. Entities with none (often parades, shows and lands) are absent. */ HistoryParkCoverageEntity: { id: string; name: string; @@ -612,12 +706,12 @@ export interface components { * @description The newest park-local day any field of this entity was recorded. */ newest: string; - /** @description The live-data field paths held for this entity, named exactly as the single-entity coverage document names them, so one client type reads both. */ + /** @description Field paths held, named as in the single-entity coverage document. */ fields: string[]; - /** @description False when the park no longer lists this entity. Its history is still held and still retrievable, and every count in this document includes it; this says where the entity is now, not what the archive has. */ + /** @description false when the park no longer lists this entity. Its history is still available. */ stillListed: boolean; }; - /** @description What one live-data field looks like across the whole park. `entities` counts what is HELD and is never a fraction: entities that have never reported this field are simply absent from the count, because a denominator drawn from entityType would publish parades, shows and lands as missing wait times. */ + /** @description One field across the park. `entities` counts the entities we hold it for; entities that never reported it are not counted. */ HistoryParkCoverageField: { /** @description How many of the park's entities this field is held for. */ entities: number; @@ -628,18 +722,18 @@ export interface components { from: string; /** * Format: date - * @description The newest park-local day any entity in the park reported it. A day in the past is not staleness: when a park stops publishing a field the ending is recorded, so the span genuinely stops there. + * @description Newest day any entity reported it. A past day can mean the park stopped publishing the field. */ newest: string; depth: components["schemas"]["HistoryParkCoverageDepth"]; }; - /** @description The park in four numbers. Every one describes what is held; none is a fraction of a total, and nothing here asserts that anything is missing. */ + /** @description Summary figures for the park, all describing what we hold. */ HistoryParkCoverageSummary: { - /** @description Entities of this park any live-data history is held for - attractions, restaurants, shows and anything else that has ever reported. Larger than the number with wait times: `fields` breaks it down. Counts entities the park no longer lists as well, since their history is still held; those carry `stillListed: false` in `entities`. */ + /** @description Entities we hold any history for, including ones the park no longer lists (`stillListed: false`). */ entitiesWithData: number; /** * Format: date - * @description The earliest park-local day anything in this park was recorded, or null when nothing has been. + * @description Earliest day anything in this park was recorded, or null. */ archiveFrom: string; /** @@ -649,71 +743,71 @@ export interface components { recordedTo: string; /** * Format: date - * @description The newest park-local day a caller can actually retrieve. Runs ahead of `recordedTo` by a day or two: the most recent days are served from live data before they are sealed into the archive. Null when nothing is recorded. + * @description The newest day you can ask for, usually today. Null when nothing is recorded. */ retrievableThrough: string; /** * Format: date - * @description The park-local day these figures were computed. They move as the archive grows, so a reader comparing two copies of this document needs to know which day each was built. + * @description The day these figures were computed; they grow over time. */ measuredOn: string; }; - /** @description A day-by-day summary of a whole PARK: every entity of the park that has history, in one call. GET /v1/entity/{id}/history/daily returns THIS shape when the entity is a PARK (entityType: "PARK") and HistoryDailyEnvelope for every other entityType, so a client should branch on the presence of entities[] or on the entity's type. range.from and range.to are always park-local calendar days, and range.to is THIS PAGE's last day rather than the whole range you asked for: a call serves at most 31 park-local days and next carries the rest. */ + /** @description The daily summary of a whole PARK, returned by /history/daily when the entity's entityType is PARK. A call serves up to 31 park-local days; `range.to` is this page's last day. */ HistoryParkDailyEnvelope: { id: string; name: string; entityType: string; parentId: string | null; destinationId: string | null; - /** @description IANA timezone the park-local days are resolved in. The park is the authority on where its day boundaries fall, so every entity below is summarised in THIS zone. */ + /** @description IANA timezone of the park; every entity is summarised in it. */ timezone: string; range: components["schemas"]["HistoryRange"]; - /** @description One entry per entity of the park that has history, ascending by name. An entity with no history is ABSENT — never an entry of nulls or an empty days[] — because "we hold nothing for this entity" is a different claim from "we hold nothing for these days". The park itself is included when it has history of its own. */ + /** @description One entry per entity with history, by name. Entities with no history are absent. */ entities: components["schemas"]["HistoryParkEntityDaily"][]; - /** @description Absolute URL of the next page, or null on the last one. A park daily call serves at most 31 park-local days and pages by DAY: the next URL repeats your other parameters with from advanced past this page's last day. Entity order never affects paging. */ + /** @description Absolute URL of the next page (your parameters kept, `from` moved on), or null on the last page. Up to 31 park-local days a page. */ next: string | null; }; - /** @description One entity of a park in a park DAILY response: its identity, the first day it has any history, and its day rows. The park itself appears as an entry too when it has history of its own (match it by id against the envelope's id). An entity with no history at all is ABSENT from entities[]. */ + /** @description One entity of the park, with its first recorded day and its rows. The park appears too if it has history of its own. */ HistoryParkEntityDaily: { id: string; name: string; entityType: string; coverage: components["schemas"]["HistoryCoverage"]; - /** @description One row per park-local day, ascending by date, exactly as GET /v1/entity/{id}/history/daily returns for this entity on its own. A day with no data is ABSENT, and an entity with history but nothing in the requested days has an empty array rather than vanishing from entities[] — so the entity list keeps its shape from one page to the next. */ + /** @description The entity's rows for this page, as /history/daily returns them for the entity alone. Empty when it has no data in these days. */ days: components["schemas"]["HistoryDailyRow"][]; }; - /** @description One entity of a park in a park RAW response: the same coverage, opening and history block GET /v1/entity/{id}/history returns for that entity on its own, so one client type reads both. The park itself appears as an entry too when it has history of its own (match it by id against the envelope's id). */ + /** @description One entity of the park, with the same coverage, opening and history as a single-entity call. The park appears too if it has history of its own. */ HistoryParkEntityRaw: { id: string; name: string; entityType: string; coverage: components["schemas"]["HistoryCoverage"]; opening: components["schemas"]["HistoryOpening"]; - /** @description Ascending by time. Empty when this entity recorded no change in the requested range, which is a different claim from having no history at all — an entity with no history is absent from entities[]. */ + /** @description Ascending by time. Empty when nothing changed in the range. */ history: components["schemas"]["HistoryRow"][]; }; - /** @description Full-state change rows for a whole PARK: every entity of the park that has history, in one call, for exactly 1 park-local day. GET /v1/entity/{id}/history returns THIS shape when the entity is a PARK (entityType: "PARK") and HistoryEnvelope for every other entityType, so a client should branch on the presence of entities[] or on the entity's type. A range spanning more than one day is 400 RANGE_TOO_LONG: a day of a park is every recorded change for every entity in it, and the one-day limit is what bounds that. Ask day by day. */ + /** @description History of a whole PARK for 1 park-local day, returned by /history when the entity's entityType is PARK. A longer range is 400 RANGE_TOO_LONG. */ HistoryParkRawEnvelope: { id: string; name: string; entityType: string; parentId: string | null; destinationId: string | null; - /** @description IANA timezone the park-local day is resolved in. The park is the authority on where its day boundaries fall, so every entity below is resolved in THIS zone. */ + /** @description IANA timezone of the park; every entity is resolved in it. */ timezone: string; range: components["schemas"]["HistoryRange"]; - /** @description One entry per entity of the park that has history, ascending by name. An entity with no history is ABSENT — never an entry of nulls — because "we hold nothing for this entity" is a different claim from "this entity recorded no change today". The park itself is included when it has history of its own. */ + /** @description One entry per entity with history, by name. Entities with no history are absent. */ entities: components["schemas"]["HistoryParkEntityRaw"][]; - /** @description Always null: a park history call covers one park-local day, so there is never a next page. */ + /** @description Always null: one day per call. */ next: string | null; }; HistoryRange: { - /** @description The requested start. A park-local day comes back verbatim (YYYY-MM-DD); an instant comes back NORMALISED to UTC whole seconds (2026-09-13T14:00:00Z), so an offset or sub-second precision you sent is not echoed back. On a day-granular endpoint such as /history/daily this is ALWAYS a park-local day, even when you asked with an instant: that endpoint's rows are whole days and cannot be sliced finer, so echoing your instant back would claim a precision the data does not have. */ + /** @description The start you asked for. A day comes back as you sent it; an instant comes back normalised to UTC whole seconds. On /history/daily it is always a day. */ from: string; - /** @description The requested end, in the same form as from, and normalised the same way. Omitted instants default to now; omitted days default to today, park-local. The same day-granular rule as from applies on /history/daily. */ + /** @description The end, in the same form as from. Defaults to today (days) or now (instants). */ to: string; }; - /** @description One row per instant at which any kind changed. Every present kind is carried forward, so a row is the complete live-data object at that instant (same keys, nesting and enum values as GET /v1/entity/{id}/live). */ + /** @description One row per moment any field changed. Every field is carried forward, so each row is the complete live data at that moment (same keys as GET /v1/entity/{id}/live). */ HistoryRow: { /** * Format: date-time @@ -726,6 +820,7 @@ export interface components { queue?: components["schemas"]["LiveQueue"]; showtimes?: components["schemas"]["LiveShowTime"][] | null; }; + /** @description The queues an entity has, each present only when the entity publishes it. STANDBY: the ordinary line. SINGLE_RIDER: a separate line for guests riding alone. RETURN_TIME: a free reservation for a later time window. PAID_RETURN_TIME: the same, paid for. BOARDING_GROUP: a virtual queue that calls groups by number. PAID_STANDBY: a paid line with its own wait. */ LiveQueue: { STANDBY?: { /** @description Current standby wait time in minutes */ @@ -790,7 +885,7 @@ export interface components { endTime?: string | null; }; /** - * @description Current operating status of an entity + * @description An entity's status. OPERATING: open and running. DOWN: stopped for now, for example by a breakdown, when it would otherwise be running. CLOSED: not open. REFURBISHMENT: closed for a longer period of maintenance or rebuilding. * @enum {string} */ LiveStatusType: "OPERATING" | "DOWN" | "CLOSED" | "REFURBISHMENT"; @@ -811,6 +906,18 @@ export interface components { timezone?: string; schedule?: components["schemas"]["ScheduleEntry"][]; }; + /** @description 503: your plan could not be read just now. Try again after the Retry-After header's number of seconds. */ + PlanUnavailable: { + /** @enum {boolean} */ + success: false; + error: { + /** @enum {string} */ + type: "Server error"; + message: string; + /** @enum {integer} */ + code?: 503; + }; + }; PriceData: { /** @description Numerical price amount, in the currency's lowest denomination (e.g. cents). null when the item costs money but the provider does not publish an amount; 0 means genuinely free */ amount: number | null; @@ -819,6 +926,17 @@ export interface components { /** @description Formatted price string */ formatted?: string; }; + /** @description 429 from the request limit every call counts against. See "Limits" at the top of this document. */ + RateLimited: { + /** @enum {string} */ + error: "Too Many Requests"; + /** @description What happened, e.g. "Too Many Requests". */ + message: string; + /** @description Seconds to wait before retrying. Also sent as the Retry-After header. */ + retryAfter: number; + /** @description Present when you were looking entities up one guessed slug at a time: the call to make instead, e.g. "GET /v1/destinations". */ + useInstead?: string; + }; /** * @description State of return time availability * @enum {string} @@ -867,6 +985,35 @@ export interface components { * @enum {string} */ SchedulePriceType: "ADMISSION" | "PACKAGE" | "ATTRACTION"; + /** @description Your plan and what is left of it. `rateLimit` is the request limit every call counts against, this one included. `historyRateLimit` is the hourly budget the history calls count against; reading it here does not spend it. Figures are per account, so every key on one account reports the same. */ + V1Me: { + /** @description Your plan, e.g. "free", "pro", "business", "enterprise". */ + tier: string; + /** @description How many days back your history calls may reach, today included. null when `historyAllArchive` is true. */ + historyDays: number | null; + /** @description true when this key can read everything we hold for an entity, back to the first day we recorded it. `historyDays` and `historyEarliestDate` are then null. */ + historyAllArchive: boolean; + /** + * Format: date + * @description The earliest day (YYYY-MM-DD) a history call may ask for, computed in UTC. Each park counts days in its own time zone, so near midnight this can be off by one. Earlier days answer 403 HISTORY_WINDOW_EXCEEDED. null when `historyAllArchive` is true. + */ + historyEarliestDate: string | null; + rateLimit: components["schemas"]["V1MeBudget"]; + historyRateLimit: components["schemas"]["V1MeBudget"]; + }; + /** @description One budget: the same figures the RateLimit-* and RateLimit-History-* headers carry. */ + V1MeBudget: { + /** @description true when this budget is not metered on your plan. The other fields are then absent, and no RateLimit headers are sent for it. */ + unmetered: boolean; + /** @description Requests allowed per window. */ + limit?: number; + /** @description Window length in seconds. */ + windowSeconds?: number; + /** @description Requests left in the current window; null if it could not be read just now. */ + remaining?: number | null; + /** @description Seconds until the window resets; null if nothing has been counted in this window yet, or if it could not be read. */ + reset?: number | null; + }; }; responses: never; parameters: never; @@ -894,23 +1041,13 @@ export interface operations { "application/json": components["schemas"]["DestinationsResponse"]; }; }; - /** @description Too Many Requests - Rate limit exceeded */ + /** @description Too many requests. Wait `retryAfter` seconds (also the Retry-After header). */ 429: { headers: { [name: string]: unknown; }; content: { - "application/json": { - /** - * @description Error type - * @enum {string} - */ - error: "Rate limit exceeded"; - /** @description Rate limit exceeded message */ - message: string; - /** @description Time in seconds to wait before retrying */ - retryAfter?: number; - }; + "application/json": components["schemas"]["RateLimited"]; }; }; }; @@ -920,6 +1057,7 @@ export interface operations { query?: never; header?: never; path: { + /** @description The entity: its UUID, its slug, or for a destination its `externalId`. */ id: string; }; cookie?: never; @@ -944,23 +1082,13 @@ export interface operations { "application/json": components["schemas"]["EntityNotFound"]; }; }; - /** @description Too Many Requests - Rate limit exceeded */ + /** @description Too many requests. Wait `retryAfter` seconds (also the Retry-After header). */ 429: { headers: { [name: string]: unknown; }; content: { - "application/json": { - /** - * @description Error type - * @enum {string} - */ - error: "Rate limit exceeded"; - /** @description Rate limit exceeded message */ - message: string; - /** @description Time in seconds to wait before retrying */ - retryAfter?: number; - }; + "application/json": components["schemas"]["RateLimited"]; }; }; }; @@ -970,6 +1098,7 @@ export interface operations { query?: never; header?: never; path: { + /** @description The entity: its UUID, its slug, or for a destination its `externalId`. */ id: string; }; cookie?: never; @@ -994,23 +1123,13 @@ export interface operations { "application/json": components["schemas"]["EntityNotFound"]; }; }; - /** @description Too Many Requests - Rate limit exceeded */ + /** @description Too many requests. Wait `retryAfter` seconds (also the Retry-After header). */ 429: { headers: { [name: string]: unknown; }; content: { - "application/json": { - /** - * @description Error type - * @enum {string} - */ - error: "Rate limit exceeded"; - /** @description Rate limit exceeded message */ - message: string; - /** @description Time in seconds to wait before retrying */ - retryAfter?: number; - }; + "application/json": components["schemas"]["RateLimited"]; }; }; }; @@ -1018,12 +1137,16 @@ export interface operations { getHistory: { parameters: { query?: { + /** @description One park-local day, YYYY-MM-DD. Cannot be combined with from/to. */ date?: string; + /** @description Start: a park-local day (YYYY-MM-DD, inclusive) or an RFC 3339 instant with an offset (inclusive). from and to must be the same form. */ from?: string; + /** @description End: a day (inclusive) or an instant (exclusive). Defaults to today, or now. Up to 31 park-local days per call for a single entity. For a park, 1 park-local day, and a longer range returns 400 RANGE_TOO_LONG. */ to?: string; }; header?: never; path: { + /** @description The entity: its UUID, its slug, or for a destination its `externalId`. */ id: string; }; cookie?: never; @@ -1072,10 +1195,10 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HistoryErrorRateLimited"]; + "application/json": components["schemas"]["HistoryErrorRateLimited"] | components["schemas"]["RateLimited"]; }; }; - /** @description Upstream unavailable */ + /** @description Temporarily unavailable */ 502: { headers: { [name: string]: unknown; @@ -1091,6 +1214,7 @@ export interface operations { query?: never; header?: never; path: { + /** @description The entity: its UUID, its slug, or for a destination its `externalId`. */ id: string; }; cookie?: never; @@ -1121,7 +1245,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HistoryErrorRateLimited"]; + "application/json": components["schemas"]["HistoryErrorRateLimited"] | components["schemas"]["RateLimited"]; }; }; }; @@ -1129,12 +1253,16 @@ export interface operations { getHistoryDaily: { parameters: { query?: { + /** @description One park-local day, YYYY-MM-DD. Cannot be combined with from/to. */ date?: string; + /** @description Start: a park-local day (YYYY-MM-DD) or an RFC 3339 instant with an offset; rows are always whole days. from and to must be the same form. */ from?: string; + /** @description End: a day (inclusive) or an instant (exclusive). Defaults to today. Up to 3660 park-local days per call. */ to?: string; }; header?: never; path: { + /** @description The entity: its UUID, its slug, or for a destination its `externalId`. */ id: string; }; cookie?: never; @@ -1183,16 +1311,20 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HistoryErrorRateLimited"]; + "application/json": components["schemas"]["HistoryErrorRateLimited"] | components["schemas"]["RateLimited"]; }; }; }; }; getEntityLiveData: { parameters: { - query?: never; + query?: { + /** @description Only return entities of these types, comma-separated, e.g. ATTRACTION,SHOW. */ + entityType?: string; + }; header?: never; path: { + /** @description The entity: its UUID, its slug, or for a destination its `externalId`. */ id: string; }; cookie?: never; @@ -1217,23 +1349,13 @@ export interface operations { "application/json": components["schemas"]["EntityNotFound"]; }; }; - /** @description Too Many Requests - Rate limit exceeded */ + /** @description Too many requests. Wait `retryAfter` seconds (also the Retry-After header). */ 429: { headers: { [name: string]: unknown; }; content: { - "application/json": { - /** - * @description Error type - * @enum {string} - */ - error: "Rate limit exceeded"; - /** @description Rate limit exceeded message */ - message: string; - /** @description Time in seconds to wait before retrying */ - retryAfter?: number; - }; + "application/json": components["schemas"]["RateLimited"]; }; }; }; @@ -1243,6 +1365,7 @@ export interface operations { query?: never; header?: never; path: { + /** @description The entity: its UUID, its slug, or for a destination its `externalId`. */ id: string; }; cookie?: never; @@ -1267,23 +1390,13 @@ export interface operations { "application/json": components["schemas"]["EntityNotFound"]; }; }; - /** @description Too Many Requests - Rate limit exceeded */ + /** @description Too many requests. Wait `retryAfter` seconds (also the Retry-After header). */ 429: { headers: { [name: string]: unknown; }; content: { - "application/json": { - /** - * @description Error type - * @enum {string} - */ - error: "Rate limit exceeded"; - /** @description Rate limit exceeded message */ - message: string; - /** @description Time in seconds to wait before retrying */ - retryAfter?: number; - }; + "application/json": components["schemas"]["RateLimited"]; }; }; }; @@ -1293,8 +1406,11 @@ export interface operations { query?: never; header?: never; path: { + /** @description The entity: its UUID, its slug, or for a destination its `externalId`. */ id: string; + /** @description A year from 1970 to 2150, e.g. 2026. */ year: string; + /** @description Month as two digits, 01 to 12. */ month: string; }; cookie?: never; @@ -1328,23 +1444,60 @@ export interface operations { "application/json": components["schemas"]["EntityNotFound"]; }; }; - /** @description Too Many Requests - Rate limit exceeded */ + /** @description Too many requests. Wait `retryAfter` seconds (also the Retry-After header). */ + 429: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["RateLimited"]; + }; + }; + }; + }; + getMe: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["V1Me"]; + }; + }; + /** @description Authentication failed */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["AuthenticationRequired"]; + }; + }; + /** @description Too many requests. Wait `retryAfter` seconds (also the Retry-After header). */ 429: { headers: { [name: string]: unknown; }; content: { - "application/json": { - /** - * @description Error type - * @enum {string} - */ - error: "Rate limit exceeded"; - /** @description Rate limit exceeded message */ - message: string; - /** @description Time in seconds to wait before retrying */ - retryAfter?: number; - }; + "application/json": components["schemas"]["RateLimited"]; + }; + }; + /** @description Response */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["PlanUnavailable"]; }; }; }; diff --git a/src/backfill-cli.ts b/src/backfill-cli.ts new file mode 100644 index 0000000..5f6dec6 --- /dev/null +++ b/src/backfill-cli.ts @@ -0,0 +1,39 @@ +#!/usr/bin/env node +/** + * The `themeparks-backfill` executable. Nothing but the call. + * + * This is a separate file because of how the previous version failed. The runner + * lived at the bottom of `backfill.ts` behind + * + * import.meta.url === pathToFileURL(process.argv[1]).href + * + * which is the usual "am I the entry point" test and is WRONG for an installed + * binary: npm puts a symlink in `node_modules/.bin`, so argv[1] is the link's + * path and the module's URL is the real file's. The comparison failed, `main` + * never ran, and `npx themeparks-backfill --version` printed nothing and exited + * 0. It passed every test in the suite and worked in the repo, where the file IS + * invoked by its own path; only installing the tarball and running the binary + * showed it. + * + * A file whose only job is to run has no condition to get wrong. + */ +import { run } from './backfill.js'; + +// Ctrl-C says how to continue, like the Python SDK. 130 is the shell's convention +// for SIGINT, and the state files on disk already say where each park got to. +process.on('SIGINT', () => { + process.stderr.write('\nstopped. Run the same command again to continue.\n'); + process.exit(130); +}); + +run().then( + (code) => { + process.exitCode = code; + }, + (error: unknown) => { + // Anything `run` did not recognise. The stack is deliberate: an unexpected + // failure with no detail is worse than an ugly one. + console.error(error); + process.exitCode = 1; + }, +); diff --git a/src/backfill.ts b/src/backfill.ts new file mode 100644 index 0000000..a2cfc70 --- /dev/null +++ b/src/backfill.ts @@ -0,0 +1,1108 @@ +/** + * Download a park's whole daily history to a file, and survive the budget. + * + * themeparks-backfill "Disneyland Park" + * npx themeparks-backfill --list disney + * + * The Python SDK got this as a command first, and the reason applies here + * identically: someone following a docs link to an example + * script, and had to work out that a library needed installing, then what the + * arguments were, then read a traceback. A recipe you have to reconstruct is not + * a recipe. + * + * What it does that is easy to get wrong by hand: + * + * 1. It asks the PARK, not the rides. The history endpoints answer every entity + * in a park in one request, so a park-level backfill of a large resort is + * around a hundred times fewer calls than the same data pulled ride by ride. + * + * 2. It bounds the range at BOTH ends. `span().retrievableThrough` is the latest + * day your key may ask for; there is no field for the earliest, and + * `archiveFrom` is where the ARCHIVE starts, which on any plan short of the + * full archive is before your window opens. So the first request is refused, + * and the floor is read out of that 403. + * + * 3. It records what it has written, so re-running continues an interrupted run + * and never appends a second copy of a finished one. + * + * 4. It takes names and destinations, not just park ids. A customer has + * "Walt Disney World Resort", not four uuids. + */ +// Several internals are exported for tests. They are not in the package's public +// surface: `bin` points at this file and `src/index.ts` does not re-export it, so +// nothing here is reachable from `import { ... } from 'themeparks'`. The +// alternative is driving every branch through argv, which is how the resolution +// layer ended up with no tests at all in the Python SDK. +import { + createWriteStream, + existsSync, + mkdirSync, + readFileSync, + rmSync, + statSync, + writeFileSync, +} from 'node:fs'; +import { basename, join } from 'node:path'; +import { createHash } from 'node:crypto'; +import { parseArgs } from 'node:util'; + +import { DEFAULT_USER_AGENT, PACKAGE_VERSION, ThemeParks } from './client.js'; +import { DAILY_COLUMNS } from './_generated/dailyColumns.js'; +import type { FetchLike } from './transport.js'; +import { ApiError, NetworkError, RateLimitError, TimeoutError } from './errors.js'; +import { + BudgetExhaustedError, + type DailyEntry, + type HistoryPage, + type HistorySpan, +} from './ergonomic/history.js'; + +/** Exit code that makes a scheduler retry rather than alert. */ +export const EX_TEMPFAIL = 75; +// THE FORMAT IS IN THE FILENAME, not only inside the file. One state file served +// both formats, so ndjson -> csv -> ndjson left the ndjson file with no state +// describing it: not resuming, no rows recorded, opened in append mode. Every row +// duplicated, exit 0. +const STATE_SUFFIX = '.backfill-state.json'; + +/** Bumped when a field changes meaning. A foreign version is refused, not guessed. */ +const STATE_VERSION = 1; +const SDK_NAME = 'js'; + +/** Where this park's state lives, for this format. */ +export function statePathFor(outDir: string, parkId: string, format: string): string { + return join(outDir, `${parkId}.${format}${STATE_SUFFIX}`); +} + +/** + * A short hash of the exact header this build writes. + * + * THE HEADER IS PART OF THE RESUME CONTRACT and nothing recorded it. The Python + * SDK's 3.3.0 wrote 19 columns in a different order and 4.0.0 writes 41; its + * `decide` compared only the format, so 41-field rows were appended under a + * 19-column header. Generating the column list removed the reviewed diff that + * used to make a header change visible; this is what replaces it. + * + * The same string the Python SDK hashes, so the two agree by construction. + */ +export function columnsFingerprint(format: string): string { + if (format !== 'csv') return ''; + return createHash('sha256').update(CSV_COLUMNS.join('\n')).digest('hex').slice(0, 16); +} +const USER_AGENT_PREFIX = 'themeparks-backfill'; + +/** + * Fold a name to something a person could plausibly have typed. + * + * The live name is `Walt Disney World® Resort`, so a plain lowercase comparison + * rejects the string a customer actually types — and that exact string is the + * documented example. NFKD splits an accented letter into letter plus combining + * mark, the mark is dropped, and every non-alphanumeric goes, so ®, apostrophes + * of either kind, spaces and hyphens stop mattering. The URL slug form matches + * for free, and slugs are what people copy out of an address bar. + */ +export function normalize(value: string): string { + return value + .normalize('NFKD') + .toLowerCase() + .replace(/[̀-ͯ]/gu, '') + .replace(/[^a-z0-9]/gu, ''); +} + +/** + * The day a paged history URL starts on, or null if it does not say. + * + * The server hands back a whole `next` URL and the SDK follows it verbatim. All + * that is wanted here is its `from`, to write into the state file as the point a + * later run carries on from -- a bare day, so the state stays readable and a + * resume goes through the same code path as a first run. + */ +function nextPageFrom(next: string): string | null { + try { + return new URL(next).searchParams.get('from'); + } catch { + return null; + } +} + +/** A park's identity, so rows can name themselves. */ +interface Park { + id: string; + name: string; + destination: string; +} + +/** The earliest day this key may ask for, read out of a 403 body. */ +export function windowFloor(error: unknown): string | null { + if (!(error instanceof ApiError)) return null; + const body: unknown = error.body; + if (body == null || typeof body !== 'object') return null; + // THE BODY IS NESTED. The API sends + // {"error": {"type": ..., "message": ..., "earliestAllowedDate": ...}} + // and the Python SDK shipped a version of this reading the top level, because + // it was written from the formatted text in a traceback rather than a real + // response. It did nothing at all. Both shapes are accepted here so the same + // mistake cannot be made from the other direction. + const inner = (body as { error?: unknown }).error; + const payload = (inner != null && typeof inner === 'object' ? inner : body) as { + type?: unknown; + earliestAllowedDate?: unknown; + }; + if (payload.type !== 'HISTORY_WINDOW_EXCEEDED') return null; + const floor = payload.earliestAllowedDate; + return typeof floor === 'string' && floor !== '' ? floor : null; +} + +/** + * True when the plan's floor sits past the park's last day of data. + * + * `end` is the newest day the park has; the 403 recovery clamps the start UP to + * the first day this key may read. For a park that stopped reporting before the + * window opens — a seasonal water park — the clamp pushes start past end and the + * API answers `400 INVALID_RANGE`. In Python that killed a six-park destination + * run three parks in, leaving an empty file and two parks never attempted. + */ +export function isEmptyWindow(from: string | null, to: string | null): boolean { + return to != null && from != null && from > to; +} + +// --------------------------------------------------------------------------- +// Finding what to back fill, without knowing any uuid. +// --------------------------------------------------------------------------- + +interface CatalogueRow { + parkId: string; + parkName: string; + destId: string; + destName: string; +} + +export async function catalogue(tp: ThemeParks): Promise { + const rows: CatalogueRow[] = []; + const { destinations } = await tp.destinations.list(); + for (const dest of destinations) { + for (const park of dest.parks ?? []) { + rows.push({ parkId: park.id, parkName: park.name, destId: dest.id, destName: dest.name }); + } + } + return rows; +} + +/** + * Whether the token after `--list` is its filter rather than the next flag. + * + * `--list disney` filters; `--list` alone lists everything; `--list --out x` lists + * everything into x. A following token that starts with `-` belongs to the next + * option, not to this one. + */ +function isFilterNext(argv: string[], i: number): boolean { + const next = argv[i + 1]; + return next !== undefined && !next.startsWith('-'); +} + +/** A uuid, loosely. Loose on purpose: the API decides what is valid, not us. */ +export function looksLikeId(value: string): boolean { + return value.length === 36 && value.split('-').length === 5; +} + +function byId(rows: CatalogueRow[], wanted: string): Park[] | null { + // A destination id expands to its parks. Checked first, deliberately. + const inDestination = rows.filter((r) => r.destId === wanted); + if (inDestination.length > 0) return inDestination.map(toPark); + const park = rows.find((r) => r.parkId === wanted); + if (park) return [toPark(park)]; + // An id we do not list: a park with no destination row, an attraction, or a + // typo. Pass it through and let the API say which. + if (looksLikeId(wanted)) return [{ id: wanted, name: wanted, destination: '' }]; + return null; +} + +function toPark(r: CatalogueRow): Park { + return { id: r.parkId, name: r.parkName, destination: r.destName }; +} + +/** + * Parks for a name, or a thrown message listing the candidates. + * + * Exact wins outright, so "Magic Kingdom Park" is not ambiguous merely because + * something else contains it. A destination name expands exactly as its id does. + * More than one match is an error that lists them: guessing between two parks + * would quietly download the wrong one and look like it worked. + */ +function byName(rows: CatalogueRow[], wanted: string): Park[] { + const needle = normalize(wanted); + // A QUERY THAT FOLDS TO NOTHING MATCHES NOTHING. `normalize('東京')` is '', and + // ''.includes is true of every string, so a CJK or Cyrillic query listed all 127 + // parks as candidates instead of saying it found none. + if (needle === '') { + throw new Error( + `no park or destination matching "${wanted}".\n` + + ` themeparks-backfill --list everything\n` + + ` themeparks-backfill --list disney the ones matching "disney"`, + ); + } + const parksIn = (destId: string): Park[] => rows.filter((r) => r.destId === destId).map(toPark); + + const exactDest = [ + ...new Set(rows.filter((r) => normalize(r.destName) === needle).map((r) => r.destId)), + ]; + const [onlyDest] = exactDest; + if (exactDest.length === 1 && onlyDest !== undefined) return parksIn(onlyDest); + + const exactPark = rows.filter((r) => normalize(r.parkName) === needle); + const [onlyPark] = exactPark; + if (exactPark.length === 1 && onlyPark !== undefined) return [toPark(onlyPark)]; + + // WHEN THE EXACT NAME IS AMBIGUOUS, the exact matches ARE the candidates. + // "Disneyland Park" is two live parks, Anaheim and Paris; widening to + // substrings adds Hong Kong Disneyland Park, which is not what was typed and + // pads the one list whose job is "which of these did you mean". + const parkHits = + exactPark.length > 1 ? exactPark : rows.filter((r) => normalize(r.parkName).includes(needle)); + const destHits = [ + ...new Set(rows.filter((r) => normalize(r.destName).includes(needle)).map((r) => r.destId)), + ]; + const [onlyDestHit] = destHits; + if (destHits.length === 1 && parkHits.length === 0 && onlyDestHit !== undefined) { + return parksIn(onlyDestHit); + } + + // THE DESTINATION GOES IN THE LABEL, and it is load-bearing: two live parks + // are named exactly "Disneyland Park" — Anaheim and Paris — so a list of bare + // park names offers a choice between two identical lines. + // A UNIQUE SUBSTRING RESOLVES. `themeparks-backfill "magic kingdom"` names exactly + // one park, and refusing a query that is unambiguous is hostile. This threw + // `"magic kingdom" matches 1. Pass an id` while the Python SDK downloaded it -- + // one SDK refusing what the other accepts, for four real names. + const [onlyHit] = parkHits; + if (parkHits.length === 1 && onlyHit !== undefined) return [toPark(onlyHit)]; + + // SORTED BY PARK NAME. Sorting the formatted line sorts by the uuid it starts + // with, so "Hurricane Harbor" -- ten live parks -- came back in an order that + // looks random to the person reading it. The id stays first on the line because + // the id is the part they copy. + const candidates = ( + parkHits.length > 0 + ? parkHits.map((r) => ({ + sort: r.parkName, + line: ` ${r.parkId} ${r.parkName} (${r.destName})`, + })) + : destHits.map((d) => { + const name = rows.find((r) => r.destId === d)?.destName ?? d; + return { + sort: name, + line: ` ${d} ${name} (destination, ${String(parksIn(d).length)} parks)`, + }; + }) + ) + .sort((a, b) => (a.sort < b.sort ? -1 : a.sort > b.sort ? 1 : 0)) + .map((c) => c.line); + + if (candidates.length === 0) { + throw new Error( + `no park or destination matching "${wanted}".\n` + + ` themeparks-backfill --list everything\n` + + ` themeparks-backfill --list disney the ones matching "disney"`, + ); + } + throw new Error( + `"${wanted}" matches ${String(candidates.length)}. Pass one of these ids, or the ` + + `destination name to get all of its parks:\n${candidates.join('\n')}`, + ); +} + +export function resolve(rows: CatalogueRow[], wanted: string): Park[] { + return byId(rows, wanted) ?? byName(rows, wanted); +} + +export function printList(rows: CatalogueRow[], needle: string | undefined): number { + const all = rows; + const shown = needle + ? rows.filter( + (r) => + normalize(r.parkName).includes(normalize(needle)) || + normalize(r.destName).includes(normalize(needle)), + ) + : rows; + if (shown.length === 0) { + process.stderr.write(`nothing matching "${needle ?? ''}"\n`); + return 1; + } + const byDest = new Map(); + for (const r of shown) { + const entry = byDest.get(r.destId) ?? { name: r.destName, parks: [] }; + entry.parks.push(r); + byDest.set(r.destId, entry); + } + for (const [destId, entry] of [...byDest].sort((a, b) => a[1].name.localeCompare(b[1].name))) { + // The TOTAL, counted from the unfiltered catalogue. Counting the filtered + // rows made `--list epcot` report "all 1 parks" for a destination with six, + // on the one line whose whole job is that number. + const total = all.filter((r) => r.destId === destId).length; + const shownNote = total === entry.parks.length ? '' : ` (${String(entry.parks.length)} shown)`; + const parkWord = total === 1 ? 'park' : 'parks'; + process.stdout.write( + `${destId} ${entry.name} <- destination: all ${String(total)} ${parkWord}${shownNote}\n`, + ); + for (const p of entry.parks.sort((a, b) => a.parkName.localeCompare(b.parkName))) { + process.stdout.write(` ${p.parkId} ${p.parkName}\n`); + } + } + return 0; +} + +// --------------------------------------------------------------------------- +// Output. Identity first, so a row says what it is before it says numbers. +// --------------------------------------------------------------------------- + +/** Identity first, so a row says what it is before it says numbers. */ +export const IDENTITY_COLUMNS = [ + 'parkId', + 'parkName', + 'entityId', + 'entityName', + 'entityType', +] as const; + +/** + * The CSV header. The data half is GENERATED from the OpenAPI spec + * (`src/_generated/dailyColumns.ts`), not typed out here. + * + * Hand-written, it had drifted three ways at once: `unknownMinutes` and the whole + * `inParkHours` block were on every row the API returns and in no column, + * `extremeWaits` likewise, and `singleRider` carried two of its five percentiles + * while `standby` carried all five. Ten of thirty-six fields missing from a file + * people pay for -- and the Python SDK's header did not match this one, so the + * same command in two languages wrote two different files. + */ +export const CSV_COLUMNS = [...IDENTITY_COLUMNS, ...DAILY_COLUMNS] as const; + +/** + * Characters that make a spreadsheet execute a cell rather than display it. + * Tab and CR are here because Excel strips them and reads what follows. + */ +const FORMULA_LEADERS = /^[=+\-@\t\r]/u; + +/** + * A number, strictly: no surrounding whitespace, no sign-only, no `\t5`. + * + * The Python SDK uses the same pattern. Testing with `Number(text)` instead would + * call `'\t'` numeric (it is 0) and leave a tab-led cell undefended, and the two + * SDKs would disagree about a cell they both claim to write identically. + */ +const NUMERIC = /^[+-]?(\d+\.?\d*|\.\d+)([eE][+-]?\d+)?$/u; + +/** + * Prefix a cell a spreadsheet would run as a formula. + * + * Every string that reaches a cell here comes from the API -- park and entity + * names -- and there is no path from a command-line argument into one, so this + * needs an upstream park feed to publish such a name. Cheap enough regardless. + * + * NUMERIC CELLS ARE LEFT ALONE, which is why this is not a bare test of the + * leading character: `-5` is a number and must stay one, or every negative value + * in the file becomes text and arithmetic breaks in the tool this protects. + */ +export function defuse(text: string): string { + if (!FORMULA_LEADERS.test(text) || NUMERIC.test(text)) return text; + return `'${text}`; +} + +function csvCell(value: unknown): string { + if (value == null) return ''; + const text = defuse(String(value)); + // \r IS IN HERE NOW. Without it a bare CR went through unquoted, one row parsed + // as two, and every later column shifted -- while Python's csv module quoted it, + // so the two files stopped being identical as well as one being malformed. + return /[",\r\n]/u.test(text) ? `"${text.replace(/"/gu, '""')}"` : text; +} + +/** + * A row flattened to `{column: value}` the same way the columns were generated: + * a nested object contributes `` for each of its own keys. + * + * Derived from the DATA rather than from the column list, so a field the API adds + * shows up here immediately; the test that compares this against a real capture + * is what turns a new field into a failing build instead of a silent loss. + */ +function flatten(row: Record, prefix = ''): Record { + const out: Record = {}; + for (const [key, value] of Object.entries(row)) { + const name = prefix === '' ? key : `${prefix}${key[0]!.toUpperCase()}${key.slice(1)}`; + if (value !== null && typeof value === 'object' && !Array.isArray(value)) { + Object.assign(out, flatten(value as Record, name)); + } else { + out[name] = value; + } + } + return out; +} + +export function csvLine(park: Park, entry: DailyEntry): string { + const cells: Record = { + parkId: park.id, + parkName: park.name, + entityId: entry.entityId, + entityName: entry.name, + entityType: entry.entityType, + ...flatten(entry.row as Record), + }; + return CSV_COLUMNS.map((c) => csvCell(cells[c])).join(','); +} + +export function ndjsonLine(park: Park, entry: DailyEntry): string { + return `${JSON.stringify({ + parkId: park.id, + parkName: park.name, + entityId: entry.entityId, + entityName: entry.name, + entityType: entry.entityType, + ...(entry.row as Record), + })}\n`; +} + +// --------------------------------------------------------------------------- +// Per-park state. One file, and it is what makes re-running safe. +// --------------------------------------------------------------------------- +// +// The Python version shipped with "finished" encoded as THE ABSENCE of a +// checkpoint file, which is indistinguishable from "never started", plus an +// always-append data file. That silently doubled the rows on a second run, made a +// destination retry re-download completed parks in full, and let a +// --format switch truncate the output with a zero exit code. The state is +// explicit here from the start. +interface BackfillState { + sdk: string; + sdkVersion: string; + stateVersion: number; + columns: string; + format: string; + start: string | null; + end: string | null; + /** Newest day written. For the human reading the file, not for resuming. */ + lastDay: string | null; + /** + * The day to carry on from: the `from` of the page after the last one fully + * written, straight from the server's own `next`. Resuming at `lastDay` + * instead re-fetches a day that is already in the file and duplicates every + * row of it, and `(entityId, date)` stops being a key -- on the EX_TEMPFAIL + * path, which is the ordinary path for a long back fill, not an edge case. + */ + resumeFrom: string | null; + complete: boolean; +} + +function readState(path: string): BackfillState | null { + try { + const value: unknown = JSON.parse(readFileSync(path, 'utf8')); + return value != null && typeof value === 'object' ? (value as BackfillState) : null; + } catch { + // Unreadable state is treated as no state. A corrupt file must not be a + // permanent wall the customer cannot see. + return null; + } +} + +/** + * Why this state file cannot be resumed by this build, or null. + * + * The two SDKs wrote the same filename and spelled `format`, `start`, `end` and + * `complete` identically, differing only in `lastDay`/`resumeFrom` versus + * `last_day`/`resume_from` -- the two keys that matter on the interrupted path. + * So the safe paths interoperated and nothing warned, while a Python run + * interrupted at 64 rows and resumed by this command produced 172 rows, 64 of + * them duplicates, marked complete. + */ +function stateMismatch(state: BackfillState, format: string): string | null { + if (state.stateVersion !== STATE_VERSION) { + return `it was written by a different version of this command (state v${String(state.stateVersion)})`; + } + if (state.sdk !== SDK_NAME) { + return `it was written by the ${String(state.sdk)} SDK, and resuming across SDKs is not supported`; + } + if (state.format !== format) return `it is a ${String(state.format)} run`; + if (state.columns !== columnsFingerprint(format)) { + return 'the column layout changed since it was written'; + } + return null; +} + +function writeState(path: string, state: BackfillState): void { + writeFileSync(path, `${JSON.stringify(state, null, 0)}\n`, 'utf8'); +} + +// --------------------------------------------------------------------------- +// One park. +// --------------------------------------------------------------------------- + +interface RunOptions { + outDir: string; + format: 'ndjson' | 'csv'; + overwrite: boolean; +} + +/** A plan to proceed with, or an exit code meaning do not. */ +type Decision = + | { + start: string | null; + hasRows: boolean; + priorStart: string | null; + /** + * True when rows from an EARLIER run are already in the file. Every deletion + * in this module must consult it: `written === 0` means "this process wrote + * nothing", which on a resumed run is not "the file is empty". + */ + resumed: boolean; + } + | number; + +export function decide( + outPath: string, + statePath: string, + opts: RunOptions, + archiveFrom: string | null, +): Decision { + if (opts.overwrite) { + rmSync(outPath, { force: true }); + rmSync(statePath, { force: true }); + } + const state = opts.overwrite ? null : readState(statePath); + const fileHasRows = existsSync(outPath) && statSync(outPath).size > 0; + const mismatch = state === null ? null : stateMismatch(state, opts.format); + const resumable = state !== null && mismatch === null; + + if (state?.complete === true && resumable && fileHasRows) { + process.stderr.write( + ` already complete: ${String(state.start)} .. ${String(state.end)} ` + + `in ${basename(outPath)} — pass --overwrite to fetch it again\n`, + ); + return 0; + } + if (fileHasRows && state === null) { + // Appending would double it; truncating would destroy someone's data. + process.stderr.write( + ` ${basename(outPath)} already has rows and there is no state file beside it.\n` + + ` --overwrite replace it\n` + + ` or move it aside and run again\n`, + ); + return 1; + } + // A state file this build cannot resume. Refusing is the only safe answer: the + // file beside it was written to a different contract, and appending to it + // produces a file no reader can parse, or one that parses wrongly. + if (state !== null && mismatch !== null && state.complete !== true) { + process.stderr.write( + ` there is an unfinished ${basename(outPath)} beside this state file, but ${mismatch}.\n` + + ` --overwrite start this park again from the beginning\n` + + ` or move both files aside and run again\n`, + ); + return 1; + } + const resuming = resumable && state.complete !== true; + // `lastDay` is the fallback for the two cases with no page boundary to use: a + // state file written by an older version, and a run that died part-way through + // its FIRST page. It re-fetches one day, so that day's rows appear twice -- + // bad, and still far better than starting from the top and appending a second + // copy of everything, which is what an unconditional archiveFrom would do. + return { + start: (resuming ? (state.resumeFrom ?? state.lastDay) : null) ?? archiveFrom, + hasRows: fileHasRows && resuming, + priorStart: resuming ? state.start : null, + resumed: resuming, + }; +} + +/** The minimum of a writable stream this module needs, so a test can stand in. */ +interface Closable { + end(cb: (error?: Error | null) => void): unknown; +} + +/** + * Flush and close, rejecting if the stream failed. + * + * THE ERROR ARGUMENT WAS DISCARDED. Node calls `end`'s callback as `cb(err)` on a + * failed stream, and this took no arguments and called `done()` regardless -- so a + * failed stream resolved, the state file recorded `complete: true`, and the command + * printed `done: N rows` and exited 0 with the file truncated. On ENOSPC or EDQUOT + * mid-download that is a short file marked finished, which no rerun would continue. + * Demonstrated: `end(cb)` receives `['EACCES']`. + * + * Extracted so it can be tested directly. Arranging a stream that writes fine and + * then fails on flush is not portable; the callback contract is the thing that was + * wrong, so the callback contract is what this pins. + */ +export function flushAndClose(handle: Closable, pending: () => Error | null): Promise { + return new Promise((done, fail) => { + handle.end((error?: Error | null) => { + const failure = error ?? pending(); + if (failure) fail(failure); + else done(); + }); + }); +} + +export async function backfillPark(tp: ThemeParks, park: Park, opts: RunOptions): Promise { + const history = tp.entity(park.id).history; + + // SPAN IS INSIDE THE BUDGET HANDLING. `coverage()` is the one history call the + // SDK does not wrap, so a spent hourly budget surfaces as a plain + // RateLimitError. In Python this sat outside the handler and escaped as a + // traceback with exit 1 — on the MOST LIKELY path after any exit 75, because + // the retry runs while the window is still shut and this is its first request. + let span: HistorySpan; + try { + span = await history.span(); + } catch (error) { + if (error instanceof RateLimitError) { + const mins = Math.round((error.retryAfterMs ?? 0) / 60_000); + process.stderr.write( + `${park.id}: history budget is spent; rerun the same command` + + `${mins > 0 ? ` in ${String(mins)} min` : ' later'} to continue\n`, + ); + return EX_TEMPFAIL; + } + throw error; + } + + const ext = opts.format === 'csv' ? 'csv' : 'ndjson'; + const outPath = join(opts.outDir, `${park.id}.${ext}`); + const statePath = statePathFor(opts.outDir, park.id, opts.format); + const end = span.retrievableThrough; + + const decided = decide(outPath, statePath, opts, span.archiveFrom); + if (typeof decided === 'number') return decided; + let { start } = decided; + const { hasRows, priorStart, resumed } = decided; + + process.stderr.write( + `${park.id}: ${String(start)} .. ${String(end)}` + + `${priorStart !== null ? ' (resumed)' : ''} -> ${outPath}\n`, + ); + + if (isEmptyWindow(start, end)) { + process.stderr.write( + ` nothing in your window: this park's data ends ${String(end)}, and your ` + + `plan reaches back to ${String(start)} — skipping\n`, + ); + // NEVER DELETE ROWS AN EARLIER RUN DOWNLOADED. `start` is the resume point on a + // rerun, so a key rotated out of a scheduler's environment or a lapsed + // subscription used to wipe the partial archive and exit 0 -- the scheduler + // logged success -- then trap: the file gone, the state surviving, `hasRows` + // false, every later run re-entering this branch and exiting 0 with no data. + if (resumed) { + process.stderr.write( + ` the rows already downloaded are left alone. Your plan no longer reaches ` + + `the day this run would continue from\n`, + ); + return 1; + } + rmSync(outPath, { force: true }); + return 0; + } + + const handle = createWriteStream(outPath, { flags: 'a', encoding: 'utf8' }); + // A LISTENER FROM THE MOMENT THE STREAM EXISTS. `handle.write()` never throws + // synchronously, and the only listener used to be attached inside `finish()`, + // so a filesystem error before that became an unhandled 'error' event: an + // uncaught exception and a stack dump, with the five remaining parks of a + // destination never attempted. Whichever of the two arrives first -- the error + // or the end of the stream -- this records it and the park fails cleanly. + let streamError: Error | null = null; + handle.on('error', (error: Error) => { + streamError = error; + }); + // ONE header decision for the whole park. Python built the writer inside the + // retried closure with `written === 0` in the predicate, and the 403 recovery + // runs precisely when that is true — so every CSV on every plan short of the + // full archive got TWO header rows, and pandas read the second as data. + if (opts.format === 'csv' && !hasRows) { + // A UTF-8 BOM, so Excel on Windows does not read the local code page and render + // `Walt Disney World® Resort` as mojibake. The primary reader of this file is a + // spreadsheet. Written with the header, so a resume never adds a second. + handle.write(`\ufeff${CSV_COLUMNS.join(',')}\n`); + } + + let written = 0; + let lastDay: string | null = null; + let resumeFrom: string | null = null; + let skipped = false; + + const stream = async (from: string | null): Promise => { + // `exactOptionalPropertyTypes` means an explicit undefined is not the same + // as an absent key, so the query is built rather than spread with nulls. + const query: { from?: string; to?: string; onPage: (page: HistoryPage) => void } = { + // The checkpoint. Fires once a page's rows are all written, carrying the + // day the NEXT page starts on, so a resume asks for nothing twice. + onPage: (page) => { + resumeFrom = page.next === null ? null : nextPageFrom(page.next); + }, + }; + if (from !== null) query.from = from; + if (end !== null) query.to = end; + for await (const entry of history.days(query)) { + // Checked every row: the stream reports failures asynchronously, so without + // this the loop keeps "writing" into a broken stream for the rest of the + // archive and only the flush would notice. + if (streamError) throw streamError; + handle.write(opts.format === 'csv' ? `${csvLine(park, entry)}\n` : ndjsonLine(park, entry)); + written += 1; + const day = (entry.row as { date?: string }).date ?? null; + // MAX, not last-seen. Entities arrive name-ordered with independent day + // lists, so the final row can belong to an entity that stopped reporting + // mid-page — taking it would rewind the resume point by up to a full page. + if (day !== null && (lastDay === null || day > lastDay)) lastDay = day; + if (written % 5000 === 0) + process.stderr.write(` ${String(written)} rows, at ${String(lastDay)}\n`); + } + }; + + /** + * Flush and close, rejecting if the stream failed. + * + * THE ERROR ARGUMENT WAS DISCARDED. Node calls `end`'s callback as `cb(err)` on + * a failed stream, and this took no arguments and called `done()` regardless -- + * so `finish()` resolved, `record(true)` ran, and the command printed + * `done: N rows` and returned 0 with the file truncated. On ENOSPC or EDQUOT + * mid-download the customer got a short file marked complete, which no rerun + * would ever continue. Demonstrated: `end(cb)` receives `['EACCES']`. + */ + const finish = (): Promise => flushAndClose(handle, () => streamError); + + const record = (complete: boolean): void => { + writeState(statePath, { + sdk: SDK_NAME, + sdkVersion: PACKAGE_VERSION, + stateVersion: STATE_VERSION, + columns: columnsFingerprint(opts.format), + format: opts.format, + start: priorStart ?? start, + end, + lastDay, + resumeFrom, + complete, + }); + }; + + try { + try { + await stream(start); + } catch (error) { + const floor = windowFloor(error); + // Retry only when nothing was written: a 403 mid-stream is not a plan + // boundary, and restarting would duplicate rows. + if (floor === null || written > 0) throw error; + process.stderr.write( + ` this key reaches back to ${floor}, not ${String(start)} — starting there\n`, + ); + start = floor; + if (isEmptyWindow(floor, end)) { + process.stderr.write( + ` nothing in your window: this park's data ends ${String(end)} — skipping\n`, + ); + skipped = true; + } else { + await stream(floor); + } + } + } catch (error) { + await finish(); + if (error instanceof BudgetExhaustedError) { + record(false); + if (written === 0 && !resumed) rmSync(outPath, { force: true }); + process.stderr.write( + ` budget spent after ${String(written)} rows; rerun the same command to continue\n`, + ); + return EX_TEMPFAIL; + } + // Any other failure still records where it got to, or the next run starts + // over and appends a second partial copy. + if (lastDay !== null) record(false); + if (written === 0 && resumed) { + // An earlier run's rows are real and are not ours to remove. + process.stderr.write(` the rows already downloaded are kept\n`); + throw error; + } + // AN EMPTY FILE IS A LIE. The stream opened the file before the first + // request, so a park that failed with nothing written leaves a 0-byte file + // that looks like "this park has no history" — and on a six-park + // destination the customer counts six files and never sees which one is + // empty. Written rows stay: they are real, and the state file beside them + // says where to carry on. + if (written === 0) rmSync(outPath, { force: true }); + throw error; + } + + await finish(); + if (skipped && written === 0) { + if (resumed) { + // Same rule: keep the file, record where it got to, and say it did not finish. + record(false); + return 1; + } + rmSync(outPath, { force: true }); + rmSync(statePath, { force: true }); + return 0; + } + record(true); + process.stderr.write(` done: ${String(written)} rows -> ${outPath}\n`); + return 0; +} + +const HELP = `themeparks-backfill — download a park's daily history to a file + +usage: + themeparks-backfill [options] PARK... + + PARK is a park or a DESTINATION, by name or id. A destination back fills every + park in it, one file each. + +options: + --list [TEXT] list ids and names, optionally filtered, then exit. No key needed. + --api-key KEY API key. Defaults to $THEMEPARKS_API_KEY. + --format FORMAT ndjson (default) or csv + --out DIR output directory (default: .) + --overwrite replace an existing file instead of refusing + --help this + --version print the package version + +examples: + themeparks-backfill --list disney + find an id, or check a spelling. Destinations with their parks indented + underneath. Works before you have a key. + + themeparks-backfill "Disneyland Park" + the daily history your plan reaches, as NDJSON, into the current directory + + themeparks-backfill "Walt Disney World Resort" + a DESTINATION: every park in it, one file each + + themeparks-backfill 7340550b-c14d-4def-80bb-acdb51d49a66 --format csv --out ./data + +exit codes: + 0 done + 75 the hourly history budget ran out. Progress is recorded; run the same + command again to continue. 75 is EX_TEMPFAIL by convention — systemd needs + RestartForceExitStatus=75 to treat it as retry rather than failure, and + cron mails on output rather than on exit code. + +how far back this reaches is your plan: 7 days with no key at all, 30 on a free +key, more on the paid tiers. It runs either way, asks the API what you may see, +and starts there. + +files are named for the park's id, not its name, because names change. Every row +carries parkId, parkName, entityId, entityName and entityType, so two files load +into one table and \`(entityId, date)\` is the natural key. +`; + +/** + * Run the command. `deps.fetch` is the seam the tests drive it through: the whole + * command, argument parsing to written file, against captured responses. + */ +export async function main( + argv: string[] = process.argv.slice(2), + deps: { fetch?: unknown } = {}, +): Promise { + // `--list` TAKES AN OPTIONAL VALUE and parseArgs has no way to say so: with + // `type: 'string'` a bare `--list` is "argument missing" and exit 2, though the + // help advertises `--list [TEXT]` and the Python SDK lists everything. Rewriting + // it to `--list=` before parsing is the least surprising way to get there. + argv = argv.map((arg, i) => (arg === '--list' && !isFilterNext(argv, i) ? '--list=' : arg)); + // `-h`, because every other command in the world accepts it. + argv = argv.map((arg) => (arg === '-h' ? '--help' : arg)); + let parsed; + try { + parsed = parseArgs({ + args: argv, + allowPositionals: true, + options: { + list: { type: 'string' }, + 'api-key': { type: 'string' }, + format: { type: 'string', default: 'ndjson' }, + out: { type: 'string', default: '.' }, + overwrite: { type: 'boolean', default: false }, + help: { type: 'boolean', default: false }, + version: { type: 'boolean', default: false }, + }, + }); + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n\n${HELP}`); + return 2; + } + const { values, positionals } = parsed; + + if (values.help === true) { + process.stdout.write(HELP); + return 0; + } + if (values.version === true) { + // Named, not a bare number: `8.3.0` alone cannot be pasted into a bug report, + // and the Python SDK prints `themeparks-backfill 3.4.0`. + process.stdout.write(`themeparks-backfill ${PACKAGE_VERSION}\n`); + return 0; + } + if (values.format !== 'ndjson' && values.format !== 'csv') { + process.stderr.write(`--format must be ndjson or csv, not "${String(values.format)}"\n`); + return 2; + } + + // `?? ` only catches null and undefined, so `THEMEPARKS_API_KEY=""` -- a + // clobbered env var in a cron file, which is how this actually happens -- read + // as a key and the anonymous notice never printed. Both SDKs treat empty as + // absent now. + const rawKey = values['api-key'] ?? process.env.THEMEPARKS_API_KEY; + const apiKey = rawKey === '' ? undefined : rawKey; + const listing = Object.hasOwn(values, 'list'); + + // NO KEY IS NOT AN ERROR. Anonymous access reads 7 days, so it runs and says + // what a key would add. Refusing to start while explaining that anonymous + // access exists is worse than either. + if (apiKey == null && !listing) { + process.stderr.write( + 'no API key: reading the 7 days anonymous access allows.\n' + + ' a free key reads 30 days, and the paid tiers reach further\n' + + ' set THEMEPARKS_API_KEY, or pass --api-key\n' + + ' keys: https://www.themeparks.wiki/profile\n\n', + ); + } + + const tp = new ThemeParks({ + ...(apiKey != null ? { apiKey } : {}), + ...(deps.fetch !== undefined ? { fetch: deps.fetch as FetchLike } : {}), + // The command's identity IN FRONT OF the SDK's, not instead of it: the + // server's logs are how a support question gets answered, and "which SDK + // version" is the first thing anyone asks. + userAgent: `${USER_AGENT_PREFIX}/${PACKAGE_VERSION} ${DEFAULT_USER_AGENT}`, + }); + + // CHECKED BEFORE THE REQUEST. This fetched /destinations first, so + // `themeparks-backfill` with no arguments and no network exited 75 -- telling a + // scheduler to retry a command that can never succeed. + if (!listing && positionals.length === 0) { + process.stderr.write(`which park or destination? try: themeparks-backfill --list disney\n`); + return 2; + } + const rows = await catalogue(tp); + if (listing) { + const needle = values.list === '' ? undefined : values.list; + return printList(rows, needle); + } + if (positionals.length === 0) { + process.stderr.write(`which park or destination? try: themeparks-backfill --list disney\n`); + return 2; + } + + let targets: Park[]; + try { + const seen = new Set(); + targets = []; + for (const wanted of positionals) { + for (const park of resolve(rows, wanted)) { + // A destination and one of its parks can both be named on one command + // line; back filling the same park twice would double every row. + if (!seen.has(park.id)) { + seen.add(park.id); + targets.push(park); + } + } + } + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); + return 1; + } + + // ALWAYS ECHO WHAT A NAME RESOLVED TO, with its destination, and recommend the + // id. Twelve live parks contain "Hurricane Harbor" and the bare name is an + // exact match for one of them, so a reasonable query could fetch the wrong + // park and say nothing. Names find a park once; ids are what you script with. + const resolvedByName = positionals.some((p) => !looksLikeId(p)); + if (targets.length > 1) { + process.stderr.write(`${String(targets.length)} parks to back fill:\n`); + for (const p of targets) { + const where = p.destination !== '' && p.destination !== p.name ? ` (${p.destination})` : ''; + process.stderr.write(` ${p.name}${where} ${p.id}\n`); + } + } else if (resolvedByName && targets[0] !== undefined) { + const p = targets[0]; + const where = p.destination !== '' && p.destination !== p.name ? ` (${p.destination})` : ''; + process.stderr.write(`resolved to ${p.name}${where} ${p.id}\n`); + } + if (resolvedByName) { + process.stderr.write( + ` use the id next time — names are convenient once, ids are exact:\n` + + ` themeparks-backfill ${targets.map((p) => p.id).join(' ')}\n`, + ); + } + + mkdirSync(values.out, { recursive: true }); + const opts: RunOptions = { + outDir: values.out, + format: values.format, + overwrite: values.overwrite, + }; + // ONE PARK'S FAILURE IS NOT THE DESTINATION'S. A 500 on Animal Kingdom used + // to throw straight out of here, so the four parks after it were never + // attempted and the customer got a partial download with a traceback and no + // statement of what was missing. Every park is tried, what failed is named at + // the end, and the exit code still says something went wrong. + const failed: string[] = []; + for (const park of targets) { + let status: number; + try { + status = await backfillPark(tp, park, opts); + } catch (error) { + process.stderr.write( + `${park.id}: ${error instanceof Error ? error.message : String(error)}\n`, + ); + // THE HINT BELONGS HERE. It sat in the top-level handler, which this catch + // makes unreachable for every ApiError, so the one message that helps with a + // mistyped id was never printed. An unlisted uuid is passed through to the + // API deliberately, so a 404 here is the likeliest single mistake. + if (error instanceof ApiError && error.status === 404) { + process.stderr.write( + ` that id is not something with history. try: themeparks-backfill --list\n`, + ); + } + failed.push(park.name); + continue; + } + // A spent budget stops everything: the next park would spend the retry-after + // for nothing, and the state files say where each one got to. + if (status === EX_TEMPFAIL) return status; + if (status !== 0) failed.push(park.name); + } + if (failed.length > 0) { + process.stderr.write( + `\n${String(failed.length)} of ${String(targets.length)} did not finish: ${failed.join(', ')}\n` + + ` the rest are written. Run the same command again to retry just these.\n`, + ); + return 1; + } + return 0; +} + +/** + * Run the command and turn a failure into a sentence. + * + * A traceback is a bug report about this tool; an unreachable API, a 404 on a + * mistyped id and a spent budget are none of them bugs in it, and a customer who + * has just paid reads one as the tool being broken. Errors we can name get one + * line; anything else still prints its stack, because an unexpected failure with + * no detail is worse than an ugly one. + */ +export async function run(argv?: string[], deps: { fetch?: unknown } = {}): Promise { + try { + return await main(argv, deps); + } catch (error) { + if (error instanceof ApiError) { + process.stderr.write(`\n${error.message}\n`); + return 1; + } + if (error instanceof NetworkError || error instanceof TimeoutError) { + process.stderr.write( + `\n${error.message}\n the run is resumable: the same command continues it\n`, + ); + return EX_TEMPFAIL; + } + throw error; + } +} diff --git a/src/client.ts b/src/client.ts index 21ef1b0..0da01a3 100644 --- a/src/client.ts +++ b/src/client.ts @@ -11,8 +11,10 @@ import { import type { RateLimits } from './ratelimit'; const DEFAULT_BASE_URL = 'https://api.themeparks.wiki/v1'; -const PACKAGE_VERSION = '8.2.0'; -const DEFAULT_USER_AGENT = `themeparks-sdk-js/${PACKAGE_VERSION}`; +/** Exported so the backfill command can announce the same version rather + * than carrying a second literal that drifts. test/unit/version.test.ts pins it. */ +export const PACKAGE_VERSION = '8.3.0'; +export const DEFAULT_USER_AGENT = `themeparks-sdk-js/${PACKAGE_VERSION}`; export interface ThemeParksOptions { baseUrl?: string; diff --git a/src/ergonomic/history.ts b/src/ergonomic/history.ts index 776012d..608fe4e 100644 --- a/src/ergonomic/history.ts +++ b/src/ergonomic/history.ts @@ -65,9 +65,23 @@ function asBudgetError(error: unknown, maxWaitMs: number): unknown { ); } -/** One daily row, tagged with the entity it belongs to. */ +/** + * One daily row, tagged with the entity it belongs to. + * + * `name` and `entityType` come from the history response itself, which matters + * for more than convenience: they are the labels that response gives for these + * rows. A park's current `/children` list gives TODAY's name, and stamping that + * on a row recorded three years ago rewrites the record, because rides get + * renamed. They are also already in the payload, so nothing needs to ask what + * an id refers to. + * + * Empty strings rather than undefined when the envelope omits them: a writer + * should not have to branch, and "" is what lands in a CSV cell either way. + */ export interface DailyEntry { entityId: string; + name: string; + entityType: string; row: HistoryDailyRow; } @@ -118,14 +132,23 @@ function toSpan(document: EntityHistoryCoverage): HistorySpan { * A park envelope carries many entities; an entity envelope carries its own * rows. Both flatten to the same stream, so a caller writes one loop. */ +function labelOf(source: { name?: string; entityType?: string }): { + name: string; + entityType: string; +} { + return { name: source.name ?? '', entityType: source.entityType ?? '' }; +} + function* dailyEntries(envelope: EntityHistoryDaily): Generator { if ('entities' in envelope) { for (const entity of envelope.entities) { - for (const row of entity.days) yield { entityId: entity.id, row }; + const label = labelOf(entity); + for (const row of entity.days) yield { entityId: entity.id, ...label, row }; } return; } - for (const row of envelope.days) yield { entityId: envelope.id, row }; + const label = labelOf(envelope); + for (const row of envelope.days) yield { entityId: envelope.id, ...label, row }; } function* changeEntries(envelope: EntityHistory): Generator { @@ -143,10 +166,37 @@ export interface BudgetOptions { maxWaitMs?: number; } -export type DaysOptions = HistoryQuery & BudgetOptions; +/** + * One page of daily history, as the server described it. + * + * `from`/`to` are the park-local days this page actually covered, which is not + * the range you asked for: a park call is capped, so a 50-day request comes + * back as 31 days plus a `next`. `next` is the URL of the following page, or + * null on the last one. + * + * This exists for resumable downloads. A checkpoint taken from the ROWS is + * wrong in both directions: the newest row's date can be earlier than the page + * covered, since an entity that stopped reporting has no rows for the tail + * days, so resuming there re-fetches days already written; and there is no way + * to tell a complete page from one interrupted mid-write. The page boundary is + * the server's own answer to "where do I carry on", so it is the only safe + * checkpoint. + */ +export interface HistoryPage { + from: string; + to: string; + next: string | null; +} + +export interface PageOptions { + /** Called after every row of a page has been yielded. See {@link HistoryPage}. */ + onPage?: (page: HistoryPage) => void; +} + +export type DaysOptions = HistoryQuery & BudgetOptions & PageOptions; export type ChangesOptions = HistoryQuery & BudgetOptions; -function toQuery(options: HistoryQuery & BudgetOptions): HistoryQuery { +function toQuery(options: HistoryQuery & BudgetOptions & PageOptions): HistoryQuery { const query: HistoryQuery = {}; if (options.date !== undefined) query.date = options.date; if (options.from !== undefined) query.from = options.from; @@ -203,6 +253,13 @@ export class HistoryApi { for (;;) { yield* dailyEntries(envelope); const next = envelope.next; + // AFTER the rows, never before: a caller checkpointing on this has to be + // able to trust that everything the page held is already written. + options.onPage?.({ + from: envelope.range.from, + to: envelope.range.to, + next: next === '' ? null : next, + }); if (next === null || next === '') return; try { // Followed verbatim: the server has already applied every parameter, diff --git a/src/index.ts b/src/index.ts index fc2cbbc..21d4cf9 100644 --- a/src/index.ts +++ b/src/index.ts @@ -30,7 +30,13 @@ export { type ChangesOptions, type DailyEntry, type DaysOptions, + // `HistoryPage` is what `onPage` hands you, so a caller who wants to name the + // type or store a page needs it. It was declared `export` in its own module and + // never re-exported here, so `import type { HistoryPage } from 'themeparks'` + // failed while the changelog advertised it as the new API. + type HistoryPage, type HistorySpan, + type PageOptions, } from './ergonomic/history'; export { DestinationsApi } from './ergonomic/destinations'; export { diff --git a/test/fixtures/README.md b/test/fixtures/README.md new file mode 100644 index 0000000..2c0cbd5 --- /dev/null +++ b/test/fixtures/README.md @@ -0,0 +1,27 @@ +## history_rate_limited.json + +The hourly history budget's 429, in the shape the API sends it +(`{error: {type, message, retryAfter}}`, with `Retry-After` in seconds alongside). `retryAfter` is longer than the SDK's +120s `maxWaitMs`, which is what turns it into `BudgetExhaustedError` rather +than a retry — the distinction the back fill's exit 75 depends on. + +**This one is hand-written, not captured**, and the SDK reads only the +`Retry-After` header, never the body — so replacing this body with nonsense +leaves the suite green. What it does not prove is that a real hourly-budget 429 +carries that header. Capturing one costs an hour of a metered key's budget; until +then, treat the header contract as unverified against production. + +## mk_park_daily_page1.json / mk_park_daily_page2.json + +Two consecutive pages of one real request, captured 2026-09-28: +`GET /entity/75ea578a-adc8-4116-a54d-dccb60765ef9/history/daily?from=2026-08-01&to=2026-09-20` +then its `next` followed verbatim. Trimmed to three entities (an attraction, a +show, a restaurant) and otherwise untouched: `range`, `next` and every row are +the server's. + +They are the oracle for resumable paging. Page 1 covers through 2026-08-31 and +the server says continue at 2026-09-01, but two of its three entities have no +rows after 2026-08-30 — so a checkpoint taken from the newest ROW rewinds and +re-downloads days already written. On the full 72-entity capture the same page +holds 1,684 rows, of which every one carries `unknownMinutes` and an +`inParkHours` block that the published schema does not mention. diff --git a/test/fixtures/csv_contract.json b/test/fixtures/csv_contract.json new file mode 100644 index 0000000..8c5263b --- /dev/null +++ b/test/fixtures/csv_contract.json @@ -0,0 +1,58 @@ +{ + "_comment": [ + "THE CSV CONTRACT, shared by the Python and JavaScript SDKs.", + "Both repos hold an identical copy of this file and assert their own column", + "list against it, because `themeparks-backfill` is one command with two", + "implementations and a customer using both must get one file format.", + "Before this existed, one SDK wrote 32 columns and the other 41, with", + "inParkScheduledMinutes against inParkHoursScheduledMinutes, and the test that", + "claimed to check it was four spot-checks.", + "Generated from the OpenAPI spec: run `npm run regenerate` in the JavaScript", + "SDK, copy this file to both repos, and expect both suites to fail until they", + "agree." + ], + "fingerprint": "1b6ea478049dc7ca", + "columns": [ + "parkId", + "parkName", + "entityId", + "entityName", + "entityType", + "date", + "firstOperatingAt", + "lastClosedAt", + "operatingMinutes", + "downMinutes", + "unknownMinutes", + "standbyMin", + "standbyP50", + "standbyMean", + "standbyP90", + "standbyMax", + "singleRiderMin", + "singleRiderP50", + "singleRiderMean", + "singleRiderP90", + "singleRiderMax", + "extremeWaitsStandby", + "extremeWaitsSingleRider", + "showCount", + "inParkHoursScheduledMinutes", + "inParkHoursOperatingMinutes", + "inParkHoursDownMinutes", + "inParkHoursUnknownMinutes", + "inParkHoursStandbyMin", + "inParkHoursStandbyP50", + "inParkHoursStandbyMean", + "inParkHoursStandbyP90", + "inParkHoursStandbyMax", + "inParkHoursSingleRiderMin", + "inParkHoursSingleRiderP50", + "inParkHoursSingleRiderMean", + "inParkHoursSingleRiderP90", + "inParkHoursSingleRiderMax", + "inParkHoursExtremeWaitsStandby", + "inParkHoursExtremeWaitsSingleRider", + "changes" + ] +} diff --git a/test/fixtures/history_rate_limited.json b/test/fixtures/history_rate_limited.json new file mode 100644 index 0000000..e5e6587 --- /dev/null +++ b/test/fixtures/history_rate_limited.json @@ -0,0 +1,7 @@ +{ + "error": { + "type": "HISTORY_RATE_LIMITED", + "message": "This key can make 120 history requests an hour.", + "retryAfter": 1847 + } +} diff --git a/test/fixtures/mk_park_daily_page1.json b/test/fixtures/mk_park_daily_page1.json new file mode 100644 index 0000000..3aa3c9b --- /dev/null +++ b/test/fixtures/mk_park_daily_page1.json @@ -0,0 +1,1465 @@ +{ + "id": "75ea578a-adc8-4116-a54d-dccb60765ef9", + "name": "Magic Kingdom Park", + "entityType": "PARK", + "parentId": "e957da41-3552-4cf6-b636-5babc5cbc4e5", + "destinationId": "e957da41-3552-4cf6-b636-5babc5cbc4e5", + "timezone": "America/New_York", + "range": { + "from": "2026-08-01", + "to": "2026-08-31" + }, + "entities": [ + { + "id": "f5aad2d4-a419-4384-bd9a-42f86385c750", + "name": "\"it's a small world\"", + "entityType": "ATTRACTION", + "coverage": { + "firstRecordedAt": "2021-07-03" + }, + "days": [ + { + "date": "2026-08-01", + "firstOperatingAt": "2026-08-01T12:31:02Z", + "lastClosedAt": "2026-08-02T03:00:14Z", + "operatingMinutes": 734, + "downMinutes": 134, + "unknownMinutes": 1, + "standby": { + "min": 5, + "p50": 10, + "mean": 14, + "p90": 30, + "max": 40 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 734, + "downMinutes": 134, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 10, + "mean": 14, + "p90": 30, + "max": 40 + } + }, + "changes": 177 + }, + { + "date": "2026-08-02", + "firstOperatingAt": "2026-08-02T12:31:10Z", + "lastClosedAt": "2026-08-03T03:00:16Z", + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 10, + "mean": 10, + "p90": 20, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 10, + "p90": 20, + "max": 30 + } + }, + "changes": 191 + }, + { + "date": "2026-08-03", + "firstOperatingAt": "2026-08-03T12:31:26Z", + "lastClosedAt": "2026-08-04T03:00:24Z", + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 10, + "mean": 14, + "p90": 25, + "max": 35 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 14, + "p90": 25, + "max": 35 + } + }, + "changes": 191 + }, + { + "date": "2026-08-04", + "firstOperatingAt": "2026-08-04T12:31:30Z", + "lastClosedAt": "2026-08-05T03:00:34Z", + "operatingMinutes": 859, + "downMinutes": 9, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 15, + "mean": 16, + "p90": 30, + "max": 35 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 859, + "downMinutes": 9, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 15, + "mean": 16, + "p90": 30, + "max": 35 + } + }, + "changes": 203 + }, + { + "date": "2026-08-05", + "firstOperatingAt": "2026-08-05T12:30:35Z", + "lastClosedAt": null, + "operatingMinutes": 882, + "downMinutes": 47, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 15, + "mean": 17, + "p90": 30, + "max": 40 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 882, + "downMinutes": 47, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 15, + "mean": 17, + "p90": 30, + "max": 40 + } + }, + "changes": 188 + }, + { + "date": "2026-08-06", + "firstOperatingAt": "2026-08-06T12:31:45Z", + "lastClosedAt": "2026-08-07T03:00:52Z", + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 10, + "mean": 18, + "p90": 45, + "max": 50 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 15, + "mean": 18, + "p90": 45, + "max": 50 + } + }, + "changes": 187 + }, + { + "date": "2026-08-07", + "firstOperatingAt": "2026-08-07T16:01:57Z", + "lastClosedAt": null, + "operatingMinutes": 656, + "downMinutes": 271, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 5, + "mean": 11, + "p90": 25, + "max": 35 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 656, + "downMinutes": 271, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 11, + "p90": 25, + "max": 35 + } + }, + "changes": 69 + }, + { + "date": "2026-08-08", + "firstOperatingAt": "2026-08-08T11:31:03Z", + "lastClosedAt": "2026-08-09T03:00:11Z", + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 15, + "mean": 14, + "p90": 25, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 15, + "mean": 14, + "p90": 25, + "max": 30 + } + }, + "changes": 211 + }, + { + "date": "2026-08-09", + "firstOperatingAt": "2026-08-09T12:31:17Z", + "lastClosedAt": "2026-08-10T03:00:27Z", + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 10, + "p90": 20, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 10, + "p90": 20, + "max": 30 + } + }, + "changes": 202 + }, + { + "date": "2026-08-10", + "firstOperatingAt": "2026-08-10T12:31:24Z", + "lastClosedAt": "2026-08-11T03:00:31Z", + "operatingMinutes": 861, + "downMinutes": 7, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 10, + "p90": 20, + "max": 40 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 861, + "downMinutes": 7, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 10, + "p90": 20, + "max": 40 + } + }, + "changes": 212 + }, + { + "date": "2026-08-11", + "firstOperatingAt": "2026-08-11T11:32:08Z", + "lastClosedAt": null, + "operatingMinutes": 986, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 927, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + } + }, + "changes": 161 + }, + { + "date": "2026-08-12", + "firstOperatingAt": "2026-08-12T12:31:25Z", + "lastClosedAt": null, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 20, + "mean": 17, + "p90": 30, + "max": 40 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 20, + "mean": 17, + "p90": 30, + "max": 40 + } + }, + "changes": 200 + }, + { + "date": "2026-08-13", + "firstOperatingAt": "2026-08-13T12:30:33Z", + "lastClosedAt": "2026-08-14T03:00:41Z", + "operatingMinutes": 929, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 5, + "p50": 10, + "mean": 12, + "p90": 15, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 10, + "mean": 12, + "p90": 15, + "max": 30 + } + }, + "changes": 180 + }, + { + "date": "2026-08-14", + "firstOperatingAt": "2026-08-14T11:30:49Z", + "lastClosedAt": null, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 5, + "p50": 5, + "mean": 8, + "p90": 10, + "max": 20 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 5, + "mean": 8, + "p90": 10, + "max": 20 + } + }, + "changes": 155 + }, + { + "date": "2026-08-15", + "firstOperatingAt": "2026-08-15T11:30:56Z", + "lastClosedAt": "2026-08-16T03:01:01Z", + "operatingMinutes": 737, + "downMinutes": 192, + "unknownMinutes": 3, + "standby": { + "min": 5, + "p50": 15, + "mean": 13, + "p90": 25, + "max": 35 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 737, + "downMinutes": 192, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 15, + "mean": 13, + "p90": 25, + "max": 35 + } + }, + "changes": 164 + }, + { + "date": "2026-08-16", + "firstOperatingAt": "2026-08-16T12:31:12Z", + "lastClosedAt": "2026-08-17T02:00:15Z", + "operatingMinutes": 808, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 5, + "p50": 10, + "mean": 9, + "p90": 15, + "max": 20 + }, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 808, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 10, + "mean": 9, + "p90": 15, + "max": 20 + } + }, + "changes": 179 + }, + { + "date": "2026-08-17", + "firstOperatingAt": "2026-08-17T12:30:28Z", + "lastClosedAt": "2026-08-18T03:00:25Z", + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 10, + "mean": 10, + "p90": 15, + "max": 25 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 10, + "p90": 15, + "max": 25 + } + }, + "changes": 193 + }, + { + "date": "2026-08-18", + "firstOperatingAt": "2026-08-18T11:31:25Z", + "lastClosedAt": null, + "operatingMinutes": 907, + "downMinutes": 50, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 10, + "max": 25 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 878, + "downMinutes": 50, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 15, + "max": 25 + } + }, + "changes": 157 + }, + { + "date": "2026-08-19", + "firstOperatingAt": "2026-08-19T12:31:42Z", + "lastClosedAt": null, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 5, + "p50": 10, + "mean": 13, + "p90": 25, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 10, + "mean": 13, + "p90": 25, + "max": 30 + } + }, + "changes": 196 + }, + { + "date": "2026-08-20", + "firstOperatingAt": "2026-08-20T12:30:52Z", + "lastClosedAt": "2026-08-21T03:00:55Z", + "operatingMinutes": 929, + "downMinutes": 0, + "unknownMinutes": 3, + "standby": { + "min": 0, + "p50": 20, + "mean": 18, + "p90": 35, + "max": 45 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 20, + "mean": 18, + "p90": 35, + "max": 45 + } + }, + "changes": 202 + }, + { + "date": "2026-08-21", + "firstOperatingAt": "2026-08-21T11:30:58Z", + "lastClosedAt": null, + "operatingMinutes": 957, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 929, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + } + }, + "changes": 145 + }, + { + "date": "2026-08-22", + "firstOperatingAt": "2026-08-22T11:31:09Z", + "lastClosedAt": "2026-08-23T03:00:21Z", + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 3, + "standby": { + "min": 5, + "p50": 10, + "mean": 11, + "p90": 20, + "max": 25 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 10, + "mean": 11, + "p90": 20, + "max": 25 + } + }, + "changes": 206 + }, + { + "date": "2026-08-23", + "firstOperatingAt": "2026-08-23T11:30:23Z", + "lastClosedAt": null, + "operatingMinutes": 958, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 7, + "p90": 10, + "max": 20 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 929, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 7, + "p90": 15, + "max": 20 + } + }, + "changes": 159 + }, + { + "date": "2026-08-24", + "firstOperatingAt": "2026-08-24T12:31:33Z", + "lastClosedAt": "2026-08-25T03:27:44Z", + "operatingMinutes": 836, + "downMinutes": 58, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 10, + "mean": 12, + "p90": 25, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 810, + "downMinutes": 58, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 13, + "p90": 25, + "max": 30 + } + }, + "changes": 204 + }, + { + "date": "2026-08-25", + "firstOperatingAt": "2026-08-25T12:30:46Z", + "lastClosedAt": null, + "operatingMinutes": 898, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 10, + "max": 15 + } + }, + "changes": 136 + }, + { + "date": "2026-08-26", + "firstOperatingAt": "2026-08-26T12:32:07Z", + "lastClosedAt": null, + "operatingMinutes": 752, + "downMinutes": 115, + "unknownMinutes": 61, + "standby": { + "min": 0, + "p50": 10, + "mean": 11, + "p90": 20, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 752, + "downMinutes": 115, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 11, + "p90": 20, + "max": 30 + } + }, + "changes": 135 + }, + { + "date": "2026-08-27", + "firstOperatingAt": "2026-08-27T13:01:18Z", + "lastClosedAt": null, + "operatingMinutes": 803, + "downMinutes": 7, + "unknownMinutes": 511, + "standby": { + "min": 0, + "p50": 10, + "mean": 10, + "p90": 25, + "max": 25 + }, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 803, + "downMinutes": 7, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 10, + "p90": 25, + "max": 25 + } + }, + "changes": 177 + }, + { + "date": "2026-08-28", + "firstOperatingAt": "2026-08-28T12:30:32Z", + "lastClosedAt": null, + "operatingMinutes": 898, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 10, + "max": 10 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 10 + } + }, + "changes": 136 + }, + { + "date": "2026-08-29", + "firstOperatingAt": "2026-08-29T12:30:36Z", + "lastClosedAt": "2026-08-30T03:00:43Z", + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 10, + "mean": 10, + "p90": 20, + "max": 25 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 10, + "p90": 20, + "max": 25 + } + }, + "changes": 224 + }, + { + "date": "2026-08-30", + "firstOperatingAt": "2026-08-30T12:30:52Z", + "lastClosedAt": null, + "operatingMinutes": 898, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 5, + "max": 10 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 5, + "max": 10 + } + }, + "changes": 142 + }, + { + "date": "2026-08-31", + "firstOperatingAt": "2026-08-31T12:31:00Z", + "lastClosedAt": "2026-09-01T02:01:13Z", + "operatingMinutes": 809, + "downMinutes": 0, + "unknownMinutes": 3, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 5, + "max": 10 + }, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 809, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 5, + "max": 10 + } + }, + "changes": 196 + } + ] + }, + { + "id": "51392ca4-f824-42d8-8808-8110ec8e0e22", + "name": "Main Street Philharmonic at Main Street, U.S.A.", + "entityType": "SHOW", + "coverage": { + "firstRecordedAt": "2021-07-04" + }, + "days": [ + { + "date": "2026-08-01", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-02", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 2, + "showCount": 0, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-04", + "firstOperatingAt": "2026-08-04T04:05:15Z", + "lastClosedAt": null, + "operatingMinutes": 1110, + "downMinutes": 0, + "unknownMinutes": 324, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-05", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-06", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-07", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-08", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-09", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 4, + "showCount": 0, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-11", + "firstOperatingAt": "2026-08-11T04:07:02Z", + "lastClosedAt": null, + "operatingMinutes": 1170, + "downMinutes": 0, + "unknownMinutes": 262, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-12", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-13", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-14", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-15", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-16", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 6, + "showCount": 0, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-18", + "firstOperatingAt": "2026-08-18T04:05:35Z", + "lastClosedAt": null, + "operatingMinutes": 1170, + "downMinutes": 0, + "unknownMinutes": 264, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-19", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-20", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-21", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-22", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-23", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 4, + "showCount": 0, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-25", + "firstOperatingAt": "2026-08-25T04:01:17Z", + "lastClosedAt": null, + "operatingMinutes": 1110, + "downMinutes": 0, + "unknownMinutes": 328, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-26", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-27", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 630, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-28", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-29", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-08-30", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 4, + "showCount": 0, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + } + ] + }, + { + "id": "55bdcccc-217b-416c-b8b0-4b6a87d16179", + "name": "Cinderella's Royal Table", + "entityType": "RESTAURANT", + "coverage": { + "firstRecordedAt": "2021-07-03" + }, + "days": [ + { + "date": "2026-08-02", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-08-04", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-08-09", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-08-18", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-08-22", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-08-28", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-08-30", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + } + ] + } + ], + "next": "https://api.themeparks.wiki/v1/entity/75ea578a-adc8-4116-a54d-dccb60765ef9/history/daily?from=2026-09-01&to=2026-09-20" +} diff --git a/test/fixtures/mk_park_daily_page2.json b/test/fixtures/mk_park_daily_page2.json new file mode 100644 index 0000000..35c9699 --- /dev/null +++ b/test/fixtures/mk_park_daily_page2.json @@ -0,0 +1,1003 @@ +{ + "id": "75ea578a-adc8-4116-a54d-dccb60765ef9", + "name": "Magic Kingdom Park", + "entityType": "PARK", + "parentId": "e957da41-3552-4cf6-b636-5babc5cbc4e5", + "destinationId": "e957da41-3552-4cf6-b636-5babc5cbc4e5", + "timezone": "America/New_York", + "range": { + "from": "2026-09-01", + "to": "2026-09-20" + }, + "entities": [ + { + "id": "f5aad2d4-a419-4384-bd9a-42f86385c750", + "name": "\"it's a small world\"", + "entityType": "ATTRACTION", + "coverage": { + "firstRecordedAt": "2021-07-03" + }, + "days": [ + { + "date": "2026-09-01", + "firstOperatingAt": "2026-09-01T12:31:10Z", + "lastClosedAt": null, + "operatingMinutes": 874, + "downMinutes": 23, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 10 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 845, + "downMinutes": 23, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 10 + } + }, + "changes": 109 + }, + { + "date": "2026-09-02", + "firstOperatingAt": "2026-09-02T12:31:23Z", + "lastClosedAt": "2026-09-03T02:00:22Z", + "operatingMinutes": 808, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 808, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + } + }, + "changes": 187 + }, + { + "date": "2026-09-03", + "firstOperatingAt": "2026-09-03T12:30:30Z", + "lastClosedAt": "2026-09-04T02:00:32Z", + "operatingMinutes": 801, + "downMinutes": 9, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 7, + "p90": 15, + "max": 25 + }, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 800, + "downMinutes": 9, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 7, + "p90": 15, + "max": 25 + } + }, + "changes": 187 + }, + { + "date": "2026-09-04", + "firstOperatingAt": "2026-09-04T12:30:47Z", + "lastClosedAt": null, + "operatingMinutes": 898, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 7, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 7, + "p90": 10, + "max": 15 + } + }, + "changes": 159 + }, + { + "date": "2026-09-05", + "firstOperatingAt": "2026-09-05T12:30:59Z", + "lastClosedAt": "2026-09-06T03:01:10Z", + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 3, + "standby": { + "min": 5, + "p50": 20, + "mean": 17, + "p90": 30, + "max": 45 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 20, + "mean": 17, + "p90": 30, + "max": 45 + } + }, + "changes": 200 + }, + { + "date": "2026-09-06", + "firstOperatingAt": "2026-09-06T12:31:02Z", + "lastClosedAt": "2026-09-07T02:00:06Z", + "operatingMinutes": 808, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 15, + "mean": 16, + "p90": 25, + "max": 35 + }, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 808, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 15, + "mean": 16, + "p90": 25, + "max": 35 + } + }, + "changes": 186 + }, + { + "date": "2026-09-07", + "firstOperatingAt": "2026-09-07T12:30:15Z", + "lastClosedAt": "2026-09-08T02:00:21Z", + "operatingMinutes": 809, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 5, + "max": 5 + }, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 809, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 5, + "max": 5 + } + }, + "changes": 154 + }, + { + "date": "2026-09-08", + "firstOperatingAt": "2026-09-08T12:43:21Z", + "lastClosedAt": null, + "operatingMinutes": 875, + "downMinutes": 22, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 856, + "downMinutes": 12, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + } + }, + "changes": 133 + }, + { + "date": "2026-09-09", + "firstOperatingAt": "2026-09-09T12:30:47Z", + "lastClosedAt": null, + "operatingMinutes": 929, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 5, + "p50": 10, + "mean": 12, + "p90": 25, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 929, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 10, + "mean": 12, + "p90": 25, + "max": 30 + } + }, + "changes": 186 + }, + { + "date": "2026-09-10", + "firstOperatingAt": "2026-09-10T12:31:03Z", + "lastClosedAt": "2026-09-11T03:01:01Z", + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 3, + "standby": { + "min": 0, + "p50": 5, + "mean": 10, + "p90": 20, + "max": 25 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 10, + "p90": 20, + "max": 25 + } + }, + "changes": 193 + }, + { + "date": "2026-09-11", + "firstOperatingAt": "2026-09-11T11:31:11Z", + "lastClosedAt": null, + "operatingMinutes": 908, + "downMinutes": 50, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 878, + "downMinutes": 50, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + } + }, + "changes": 131 + }, + { + "date": "2026-09-12", + "firstOperatingAt": "2026-09-12T11:31:22Z", + "lastClosedAt": "2026-09-13T03:00:22Z", + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 10, + "mean": 13, + "p90": 25, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 13, + "p90": 25, + "max": 30 + } + }, + "changes": 214 + }, + { + "date": "2026-09-13", + "firstOperatingAt": "2026-09-13T11:31:35Z", + "lastClosedAt": null, + "operatingMinutes": 957, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + } + }, + "changes": 129 + }, + { + "date": "2026-09-14", + "firstOperatingAt": "2026-09-14T12:30:44Z", + "lastClosedAt": "2026-09-15T03:00:43Z", + "operatingMinutes": 864, + "downMinutes": 5, + "unknownMinutes": 2, + "standby": { + "min": 5, + "p50": 10, + "mean": 13, + "p90": 25, + "max": 40 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 864, + "downMinutes": 5, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 10, + "mean": 13, + "p90": 25, + "max": 40 + } + }, + "changes": 176 + }, + { + "date": "2026-09-15", + "firstOperatingAt": "2026-09-15T12:30:50Z", + "lastClosedAt": null, + "operatingMinutes": 897, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 5, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 869, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 5, + "p90": 5, + "max": 15 + } + }, + "changes": 126 + }, + { + "date": "2026-09-16", + "firstOperatingAt": "2026-09-16T12:30:22Z", + "lastClosedAt": "2026-09-17T02:00:19Z", + "operatingMinutes": 809, + "downMinutes": 0, + "unknownMinutes": 3, + "standby": { + "min": 5, + "p50": 10, + "mean": 11, + "p90": 20, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 809, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 5, + "p50": 10, + "mean": 11, + "p90": 20, + "max": 30 + } + }, + "changes": 189 + }, + { + "date": "2026-09-17", + "firstOperatingAt": "2026-09-17T12:31:24Z", + "lastClosedAt": "2026-09-18T03:00:32Z", + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 10, + "mean": 12, + "p90": 20, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 868, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 10, + "mean": 12, + "p90": 20, + "max": 30 + } + }, + "changes": 185 + }, + { + "date": "2026-09-18", + "firstOperatingAt": "2026-09-18T11:32:34Z", + "lastClosedAt": null, + "operatingMinutes": 956, + "downMinutes": 0, + "unknownMinutes": 1, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 927, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 6, + "p90": 10, + "max": 15 + } + }, + "changes": 146 + }, + { + "date": "2026-09-19", + "firstOperatingAt": "2026-09-19T11:31:47Z", + "lastClosedAt": "2026-09-20T03:00:56Z", + "operatingMinutes": 927, + "downMinutes": 1, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 15, + "mean": 16, + "p90": 30, + "max": 40 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 927, + "downMinutes": 1, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 15, + "mean": 16, + "p90": 30, + "max": 40 + } + }, + "changes": 218 + }, + { + "date": "2026-09-20", + "firstOperatingAt": "2026-09-20T11:31:52Z", + "lastClosedAt": null, + "operatingMinutes": 956, + "downMinutes": 0, + "unknownMinutes": 2, + "standby": { + "min": 0, + "p50": 5, + "mean": 9, + "p90": 15, + "max": 30 + }, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 928, + "downMinutes": 0, + "unknownMinutes": 0, + "standby": { + "min": 0, + "p50": 5, + "mean": 8, + "p90": 15, + "max": 30 + } + }, + "changes": 142 + } + ] + }, + { + "id": "51392ca4-f824-42d8-8808-8110ec8e0e22", + "name": "Main Street Philharmonic at Main Street, U.S.A.", + "entityType": "SHOW", + "coverage": { + "firstRecordedAt": "2021-07-04" + }, + "days": [ + { + "date": "2026-09-01", + "firstOperatingAt": "2026-09-01T04:00:50Z", + "lastClosedAt": null, + "operatingMinutes": 1110, + "downMinutes": 0, + "unknownMinutes": 329, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-02", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 630, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-03", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 630, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-04", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-05", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-06", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 3, + "showCount": 0, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-08", + "firstOperatingAt": "2026-09-08T04:02:07Z", + "lastClosedAt": null, + "operatingMinutes": 1110, + "downMinutes": 0, + "unknownMinutes": 327, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-09", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-10", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-11", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-12", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-13", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 1, + "showCount": 0, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-15", + "firstOperatingAt": "2026-09-15T04:02:56Z", + "lastClosedAt": null, + "operatingMinutes": 1110, + "downMinutes": 0, + "unknownMinutes": 327, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-16", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 630, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-17", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-18", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 4, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-19", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "showCount": 5, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + }, + { + "date": "2026-09-20", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 2, + "showCount": 0, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 0, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 1 + } + ] + }, + { + "id": "55bdcccc-217b-416c-b8b0-4b6a87d16179", + "name": "Cinderella's Royal Table", + "entityType": "RESTAURANT", + "coverage": { + "firstRecordedAt": "2021-07-03" + }, + "days": [ + { + "date": "2026-09-01", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-09-05", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-09-11", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-09-14", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 570, + "inParkHours": { + "scheduledMinutes": 870, + "operatingMinutes": 870, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-09-16", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 630, + "inParkHours": { + "scheduledMinutes": 810, + "operatingMinutes": 810, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + }, + { + "date": "2026-09-18", + "firstOperatingAt": null, + "lastClosedAt": null, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 510, + "inParkHours": { + "scheduledMinutes": 930, + "operatingMinutes": 930, + "downMinutes": 0, + "unknownMinutes": 0 + }, + "changes": 0 + } + ] + } + ], + "next": null +} diff --git a/test/unit/backfill.test.ts b/test/unit/backfill.test.ts new file mode 100644 index 0000000..c0c284e --- /dev/null +++ b/test/unit/backfill.test.ts @@ -0,0 +1,1266 @@ +/** + * The packaged back fill command: `themeparks-backfill`. + * + * This is the one piece of the SDK a customer runs rather than imports, usually + * within minutes of paying, so a failure here reads as "I paid and got nothing". + * The Python port of the same command shipped five defects that a passing suite + * did not catch, every one of them because the test agreed with the code instead + * of with the API. So the oracles here are real captures: + * + * history_window_exceeded.json a real 403 body, nested under `error` + * destinations.json 101 real destinations, 127 parks + * mk_park_daily_page1/2.json two real pages of one request, verbatim + * `range` and `next` + * + * Nothing in this file invents an API response shape. + */ + +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import { + readFileSync, + existsSync, + writeFileSync, + mkdtempSync, + readdirSync, + rmSync, + chmodSync, +} from 'node:fs'; +import { readFile } from 'node:fs/promises'; +import { resolve as resolvePath, join } from 'node:path'; +import { tmpdir } from 'node:os'; +import { + CSV_COLUMNS, + columnsFingerprint, + defuse, + flushAndClose, + statePathFor, + run, + csvLine, + decide, + isEmptyWindow, + looksLikeId, + main, + ndjsonLine, + normalize, + printList, + resolve as resolveParks, + windowFloor, + EX_TEMPFAIL, +} from '../../src/backfill'; +import { ApiError } from '../../src/errors'; +import { PACKAGE_VERSION } from '../../src/client'; +import type { DailyEntry } from '../../src/ergonomic/history'; + +const FIXTURES = resolvePath(__dirname, '../fixtures'); + +function fixture(name: string): Record { + return JSON.parse(readFileSync(join(FIXTURES, name), 'utf8')) as Record; +} + +async function loadFixture(name: string): Promise> { + return JSON.parse(await readFile(join(FIXTURES, name), 'utf8')) as Record; +} + +const DESTINATIONS = fixture('destinations.json'); +const WINDOW_403 = fixture('history_window_exceeded.json'); +const RATE_LIMITED = fixture('history_rate_limited.json'); + +const WDW = 'e957da41-3552-4cf6-b636-5babc5cbc4e5'; +const MK = '75ea578a-adc8-4116-a54d-dccb60765ef9'; + +interface Row { + parkId: string; + parkName: string; + destId: string; + destName: string; +} + +/** The catalogue the CLI builds, from the real destinations capture. */ +function catalogueRows(): Row[] { + const rows: Row[] = []; + for (const dest of DESTINATIONS.destinations as { + id: string; + name: string; + parks?: { id: string; name: string }[]; + }[]) { + for (const park of dest.parks ?? []) { + rows.push({ parkId: park.id, parkName: park.name, destId: dest.id, destName: dest.name }); + } + } + return rows; +} + +const ROWS = catalogueRows(); + +// --------------------------------------------------------------------------- + +/** + * Write a state file this build accepts, overriding only what a test cares about. + * + * Built from the code's own vocabulary rather than typed out, so a test cannot + * silently drift from the contract the code enforces. + */ +function stateFile( + dir: string, + over: Record = {}, + park = 'p', + format = 'ndjson', +): string { + const path = statePathFor(dir, park, format); + writeFileSync( + path, + JSON.stringify({ + sdk: 'js', + sdkVersion: PACKAGE_VERSION, + stateVersion: 1, + columns: columnsFingerprint(format), + format, + start: '2025-01-01', + end: '2026-09-28', + lastDay: null, + resumeFrom: null, + complete: false, + ...over, + }), + ); + return path; +} + +describe('the 403 fixture is a real response, not a retyped traceback', () => { + it('nests the window error under `error`', () => { + // THE DEFECT THIS PINS. The Python port read `body['type']` because it was + // written from the formatted text in a traceback, where the envelope has + // already been stripped. It matched nothing, so the recovery never ran, and + // nine tests passed because their fixture had been retyped from the same + // traceback. Assert the shape against the capture, in the open, so the + // reading code cannot be "fixed" back to the flat form. + expect(Object.keys(WINDOW_403)).toEqual(['error']); + const inner = WINDOW_403.error as Record; + expect(inner.type).toBe('HISTORY_WINDOW_EXCEEDED'); + expect(inner.earliestAllowedDate).toMatch(/^\d{4}-\d{2}-\d{2}$/u); + expect(WINDOW_403.type).toBeUndefined(); + }); +}); + +describe('windowFloor', () => { + const asError = (body: unknown) => + new ApiError('403 Forbidden', { status: 403, body, url: 'https://api.themeparks.wiki/v1/x' }); + + it('reads the floor out of the real nested body', () => { + const floor = (WINDOW_403.error as { earliestAllowedDate: string }).earliestAllowedDate; + expect(windowFloor(asError(WINDOW_403))).toBe(floor); + }); + + it('also accepts the flat shape', () => { + // Tolerated deliberately: it costs one line and removes the chance of + // making the same mistake from the other direction. + expect(windowFloor(asError(WINDOW_403.error))).toBe( + (WINDOW_403.error as { earliestAllowedDate: string }).earliestAllowedDate, + ); + }); + + it('ignores a different error that happens to carry a date', () => { + // INVALID_RANGE also carries earliestAllowedDate. Clamping to it would turn + // a bad request into a silently different one. + expect( + windowFloor(asError({ error: { type: 'INVALID_RANGE', earliestAllowedDate: '2026-01-01' } })), + ).toBeNull(); + }); + + it('returns null for anything that is not a window error', () => { + expect(windowFloor(asError({ error: { type: 'HISTORY_WINDOW_EXCEEDED' } }))).toBeNull(); + expect( + windowFloor(asError({ error: { type: 'HISTORY_WINDOW_EXCEEDED', earliestAllowedDate: '' } })), + ).toBeNull(); + expect(windowFloor(asError('403 Forbidden'))).toBeNull(); + expect(windowFloor(asError(null))).toBeNull(); + expect(windowFloor(new Error('boom'))).toBeNull(); + expect(windowFloor(undefined)).toBeNull(); + }); +}); + +describe('normalize', () => { + it('folds the characters that live park names actually contain', () => { + const names = (DESTINATIONS.destinations as { name: string }[]).map((d) => d.name); + // Each of these is in the capture, and each one broke matching. + expect(names).toContain('Walt Disney World® Resort'); + expect(names).toContain('Walibi Rhône-Alpes'); + expect(normalize('Walt Disney World® Resort')).toBe(normalize('Walt Disney World Resort')); + expect(normalize('Walibi Rhône-Alpes')).toBe(normalize('walibi rhone alpes')); + expect(normalize('Knott’s Soak City')).toBe(normalize("Knott's Soak City")); + }); + + it('matches the url slug form, because that is what people paste', () => { + expect(normalize('walt-disney-world-resort')).toBe(normalize('Walt Disney World® Resort')); + }); +}); + +describe('looksLikeId', () => { + it('accepts a uuid and rejects a name', () => { + expect(looksLikeId(WDW)).toBe(true); + expect(looksLikeId('Magic Kingdom Park')).toBe(false); + expect(looksLikeId(`${WDW}x`)).toBe(false); + }); +}); + +describe('resolving what to back fill, against the real catalogue', () => { + it('expands a destination id to every park in it', () => { + const parks = resolveParks(ROWS, WDW); + expect(parks.map((p) => p.id)).toContain(MK); + expect(parks).toHaveLength(6); + }); + + it('expands a destination name the same way', () => { + expect( + resolveParks(ROWS, 'Walt Disney World Resort') + .map((p) => p.id) + .sort(), + ).toEqual( + resolveParks(ROWS, WDW) + .map((p) => p.id) + .sort(), + ); + }); + + it('takes a park id on its own', () => { + const [park] = resolveParks(ROWS, MK); + expect(park?.id).toBe(MK); + expect(park?.name).toBe('Magic Kingdom Park'); + }); + + it('takes an exact park name even when other names contain it', () => { + const parks = resolveParks(ROWS, 'EPCOT'); + expect(parks).toHaveLength(1); + expect(parks[0]?.id).toBe('47f90d2c-e191-4239-a466-5892ef59a88b'); + }); + + it('refuses a name two live parks share, and names the destinations', () => { + // "Disneyland Park" is Anaheim AND Paris in this capture. Picking one would + // download the wrong park and look like it worked. + const shared = ROWS.filter((r) => r.parkName === 'Disneyland Park'); + expect(shared.length).toBe(2); + let message = ''; + try { + resolveParks(ROWS, 'Disneyland Park'); + } catch (error) { + message = (error as Error).message; + } + expect(message).toContain('matches 2'); + expect(message).not.toContain('Hong Kong'); + for (const row of shared) { + expect(message).toContain(row.parkId); + expect(message).toContain(row.destName); + } + }); + + it('passes an unlisted uuid through for the API to judge', () => { + const unknown = '00000000-1111-2222-3333-444444444444'; + expect(resolveParks(ROWS, unknown)).toEqual([{ id: unknown, name: unknown, destination: '' }]); + }); + + it('points at --list when nothing matches', () => { + expect(() => resolveParks(ROWS, 'Dollywoodd')).toThrow(/--list/u); + }); +}); + +describe('printList', () => { + let out: string[]; + beforeEach(() => { + out = []; + vi.spyOn(process.stdout, 'write').mockImplementation((chunk) => { + out.push(String(chunk)); + return true; + }); + vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + }); + afterEach(() => { + vi.restoreAllMocks(); + }); + + it('counts the destination total from the whole catalogue, not the filter', () => { + // The bug: `--list epcot` printed "all 1 parks" for a destination with six, + // on the one line whose entire job is that number. + expect(printList(ROWS, 'EPCOT')).toBe(0); + const text = out.join(''); + expect(text).toContain('all 6 parks (1 shown)'); + }); + + it('matches a destination by its folded name', () => { + expect(printList(ROWS, 'walt disney world resort')).toBe(0); + expect(out.join('')).toContain(MK); + }); + + it('exits 1 when the filter matches nothing', () => { + expect(printList(ROWS, 'zzzz')).toBe(1); + }); +}); + +describe('the CSV carries every field the API sends', () => { + it('has a column for every scalar in a real page of rows', () => { + // THE DRIFT GATE. `unknownMinutes`, `inParkHours` and `extremeWaits` are on + // the rows the API returns and were in none of the published schema this SDK + // had vendored, so the CSV dropped them while NDJSON kept them. The columns + // are generated from the spec now; this fails the build if a real response + // ever carries a field the generated list does not. + const page = fixture('mk_park_daily_page1.json'); + const rows = (page.entities as { days: Record[] }[]).flatMap((e) => e.days); + expect(rows.length).toBeGreaterThan(50); + const flat = (row: Record, prefix = ''): string[] => + Object.entries(row).flatMap(([key, value]) => { + const name = prefix === '' ? key : `${prefix}${key[0]!.toUpperCase()}${key.slice(1)}`; + return value !== null && typeof value === 'object' + ? flat(value as Record, name) + : [name]; + }); + const seen = new Set(rows.flatMap((row) => flat(row))); + const columns = new Set(CSV_COLUMNS); + expect([...seen].filter((name) => !columns.has(name))).toEqual([]); + // And the generated list is a superset, not a coincidence: this park has no + // single-rider queue, so those columns are in the header and in no row here. + expect(columns.has('singleRiderP90')).toBe(true); + expect(seen.has('singleRiderP90')).toBe(false); + }); + + it('keeps every stats block symmetric', () => { + // Python shipped all five percentiles for standby and only p50 and max for + // singleRider. Asymmetric, silent, and for no reason. + const of = (prefix: string) => + CSV_COLUMNS.filter((c) => c.startsWith(prefix) && /(Min|P50|Mean|P90|Max)$/u.test(c)).map( + (c) => c.slice(prefix.length), + ); + expect(of('standby')).toEqual(['Min', 'P50', 'Mean', 'P90', 'Max']); + expect(of('singleRider')).toEqual(of('standby')); + expect(of('inParkHoursStandby')).toEqual(of('standby')); + expect(of('inParkHoursSingleRider')).toEqual(of('standby')); + }); + + it('matches the Python SDK column for column', () => { + // The same command in two languages must write the same file. Before the + // columns were generated, this SDK wrote 32 columns and Python wrote 41, with + // `inParkScheduledMinutes` against `inParkHoursScheduledMinutes` -- a + // customer using both got two incompatible CSVs of the same park. + expect(CSV_COLUMNS.slice(0, 5)).toEqual([ + 'parkId', + 'parkName', + 'entityId', + 'entityName', + 'entityType', + ]); + expect(CSV_COLUMNS).toContain('inParkHoursScheduledMinutes'); + expect(CSV_COLUMNS).toContain('extremeWaitsSingleRider'); + expect(CSV_COLUMNS).not.toContain('inParkScheduledMinutes'); + }); +}); + +describe('row output', () => { + const park = { id: MK, name: 'Magic Kingdom Park', destination: 'Walt Disney World® Resort' }; + const realRow = (): Record => { + const page = fixture('mk_park_daily_page1.json'); + const entity = (page.entities as { days: Record[] }[])[0]; + return entity?.days[0] as Record; + }; + const entry = (over: Partial = {}): DailyEntry => + ({ + entityId: 'd9d12438-d999-4482-894b-8955fdb20ccf', + name: "it's a small world", + entityType: 'ATTRACTION', + row: realRow(), + ...over, + }) as DailyEntry; + + it('puts identity before numbers and fills every column', () => { + const cells = csvLine(park, entry()).split(','); + expect(cells).toHaveLength(CSV_COLUMNS.length); + expect(cells[0]).toBe(MK); + expect(cells[2]).toBe('d9d12438-d999-4482-894b-8955fdb20ccf'); + const at = (name: string) => cells[CSV_COLUMNS.indexOf(name as (typeof CSV_COLUMNS)[number])]; + expect(at('date')).toBe(realRow().date); + expect(at('unknownMinutes')).toBe(String(realRow().unknownMinutes)); + const inPark = realRow().inParkHours as { scheduledMinutes: number }; + expect(at('inParkHoursScheduledMinutes')).toBe(String(inPark.scheduledMinutes)); + }); + + it('quotes a name containing a comma or a quote', () => { + const cells = csvLine({ ...park, name: 'Foo, Bar' }, entry({ name: 'He said "hi"' })); + expect(cells).toContain('"Foo, Bar"'); + expect(cells).toContain('"He said ""hi"""'); + }); + + it('writes the entity type as a plain string', () => { + // Python emitted `EntityType.SHOW` from an enum repr, so filtering on + // 'SHOW' matched nothing and said so in no way at all. + expect(csvLine(park, entry({ entityType: 'SHOW' })).split(',')[4]).toBe('SHOW'); + expect(JSON.parse(ndjsonLine(park, entry({ entityType: 'SHOW' }))).entityType).toBe('SHOW'); + }); + + it('names the park and entity on every ndjson row', () => { + const row = JSON.parse(ndjsonLine(park, entry())) as Record; + expect(row.parkId).toBe(MK); + expect(row.parkName).toBe('Magic Kingdom Park'); + expect(row.entityName).toBe("it's a small world"); + // NDJSON is lossless: the whole row, including what the schema omits. + expect(row.inParkHours).toEqual(realRow().inParkHours); + expect(row.unknownMinutes).toBe(realRow().unknownMinutes); + }); +}); + +describe('isEmptyWindow', () => { + it('is true only when the clamped start passes the last day of data', () => { + // Typhoon Lagoon: closed, so its data ends before a 30-day window opens. + // Asking anyway is a 400 that killed a six-park run three parks in. + expect(isEmptyWindow('2026-09-01', '2026-08-01')).toBe(true); + expect(isEmptyWindow('2026-08-01', '2026-09-01')).toBe(false); + expect(isEmptyWindow('2026-09-01', '2026-09-01')).toBe(false); + expect(isEmptyWindow(null, '2026-09-01')).toBe(false); + expect(isEmptyWindow('2026-09-01', null)).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// State, which is what makes running it twice safe. +// --------------------------------------------------------------------------- + +describe('decide', () => { + let dir: string; + let out: string; + let state: string; + const opts = (over: Partial<{ format: 'csv' | 'ndjson'; overwrite: boolean }> = {}) => ({ + outDir: dir, + format: 'ndjson' as 'csv' | 'ndjson', + overwrite: false, + ...over, + }); + + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'bf-decide-')); + out = join(dir, 'p.ndjson'); + state = statePathFor(dir, 'p', 'ndjson'); + vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + }); + afterEach(() => { + vi.restoreAllMocks(); + rmSync(dir, { recursive: true, force: true }); + }); + + it('starts at the archive floor when there is nothing there', () => { + expect(decide(out, state, opts(), '2021-07-03')).toEqual({ + start: '2021-07-03', + hasRows: false, + priorStart: null, + resumed: false, + }); + }); + + it('does nothing and exits 0 when the park is already complete', () => { + writeFileSync(out, '{"a":1}\n'); + stateFile(dir, { + start: '2021-07-03', + end: '2026-09-23', + lastDay: '2026-09-23', + resumeFrom: null, + complete: true, + }); + expect(decide(out, state, opts(), '2021-07-03')).toBe(0); + expect(readFileSync(out, 'utf8')).toBe('{"a":1}\n'); + }); + + it('refuses a file with rows and no state beside it', () => { + // Appending would double someone's data; truncating would destroy it. + writeFileSync(out, '{"a":1}\n'); + expect(decide(out, state, opts(), '2021-07-03')).toBe(1); + expect(readFileSync(out, 'utf8')).toBe('{"a":1}\n'); + }); + + it('refuses to switch format part-way through a park', () => { + writeFileSync(out, '{"a":1}\n'); + stateFile(dir, { + start: '2021-07-03', + end: '2026-09-23', + lastDay: '2024-01-01', + resumeFrom: '2024-01-02', + complete: false, + }); + expect(decide(join(dir, 'p.csv'), state, opts({ format: 'csv' }), '2021-07-03')).toBe(1); + }); + + it('resumes from the page boundary, not the newest row', () => { + // THE DEFECT. `lastDay` is the newest row written; the page it came from + // covered further, because an entity that stopped reporting has no rows for + // the tail days. Resuming at lastDay re-fetches a day already in the file + // and duplicates every row of it -- on the exit-75 path, which is the + // ordinary path for a long back fill. + writeFileSync(out, '{"a":1}\n'); + stateFile(dir, { + start: '2021-07-03', + end: '2026-09-23', + lastDay: '2026-08-30', + resumeFrom: '2026-09-01', + complete: false, + }); + expect(decide(out, state, opts(), '2021-07-03')).toEqual({ + start: '2026-09-01', + hasRows: true, + priorStart: '2021-07-03', + resumed: true, + }); + }); + + it('falls back to lastDay for a state file written before resumeFrom existed', () => { + // One duplicated day beats starting from the top and appending a second + // copy of the whole archive. + writeFileSync(out, '{"a":1}\n'); + stateFile(dir, { + start: '2021-07-03', + end: '2026-09-23', + lastDay: '2026-08-30', + complete: false, + }); + expect(decide(out, state, opts(), '2021-07-03')).toMatchObject({ + start: '2026-08-30', + hasRows: true, + }); + }); + + it('treats corrupt state as a file it must not touch', () => { + writeFileSync(out, '{"a":1}\n'); + writeFileSync(state, 'not json'); + expect(decide(out, state, opts(), '2021-07-03')).toBe(1); + }); + + it('--overwrite clears both files first', () => { + writeFileSync(out, '{"a":1}\n'); + stateFile(dir, { complete: true }); + expect(decide(out, state, opts({ overwrite: true }), '2021-07-03')).toEqual({ + start: '2021-07-03', + hasRows: false, + priorStart: null, + resumed: false, + }); + expect(existsSync(out)).toBe(false); + expect(existsSync(state)).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// End to end, through main(), against the real pages. +// --------------------------------------------------------------------------- + +describe('a whole run', () => { + let dir: string; + let err: string[]; + + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'bf-run-')); + err = []; + vi.spyOn(process.stderr, 'write').mockImplementation((chunk) => { + err.push(String(chunk)); + return true; + }); + vi.spyOn(process.stdout, 'write').mockImplementation(() => true); + }); + afterEach(() => { + vi.restoreAllMocks(); + rmSync(dir, { recursive: true, force: true }); + }); + + const json = (body: unknown, status = 200, headers: Record = {}) => + new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json', ...headers }, + }); + + /** + * The hourly history budget's refusal, headers and all. The header is what + * makes it a budget rather than a blip: under maxWaitMs the SDK waits it out, + * over it the caller is told to checkpoint, and exit 75 is that distinction. + */ + const budgetSpent = () => + json(RATE_LIMITED, 429, { + 'retry-after': String((RATE_LIMITED.error as { retryAfter: number }).retryAfter), + }); + + /** + * A fetch that answers the four URLs the command uses, from real captures. + * `daily` decides what each successive daily call returns. + */ + async function server(options: { + daily: (url: string, call: number) => Response | Promise; + coverage?: Record; + }) { + const coverage = options.coverage ?? (await loadFixture('mk_history_coverage.json')); + let calls = 0; + const seen: string[] = []; + const fetchFn = vi.fn((input: unknown) => { + const url = String(input); + seen.push(url); + if (url.includes('/destinations')) return Promise.resolve(json(DESTINATIONS)); + if (url.includes('/history/coverage')) return Promise.resolve(json(coverage)); + if (url.includes('/history/daily')) { + calls += 1; + return Promise.resolve(options.daily(url, calls)); + } + return Promise.resolve(json({ error: { type: 'NOT_FOUND' } }, 404)); + }); + return { fetchFn, seen, calls: () => calls }; + } + + it('checkpoints the page boundary the server gave, not the newest row', async () => { + // Page 1 covers 2026-08-01..08-31 and its newest ROW is 08-31, but two of + // its three entities stop at 08-30. The server says carry on at 09-01. + // Page 2 then fails on the budget, so the state file is the checkpoint a + // rerun uses -- and it must say 09-01. + const page1 = await loadFixture('mk_park_daily_page1.json'); + const { fetchFn, calls } = await server({ + daily: (_url, call) => (call === 1 ? json(page1) : budgetSpent()), + }); + const code = await mainWith(fetchFn, [MK, '--out', dir]); + expect(code).toBe(EX_TEMPFAIL); + expect(calls()).toBe(2); + const state = JSON.parse(readFileSync(statePathFor(dir, MK, 'ndjson'), 'utf8')) as Record< + string, + unknown + >; + expect(state.resumeFrom).toBe('2026-09-01'); + expect(state.lastDay).toBe('2026-08-31'); + expect(state.complete).toBe(false); + const written = readFileSync(join(dir, `${MK}.ndjson`), 'utf8') + .trim() + .split('\n'); + expect(written).toHaveLength(64); + }); + + it('a rerun asks for the checkpoint day and appends without duplicating', async () => { + const page1 = await loadFixture('mk_park_daily_page1.json'); + const page2 = await loadFixture('mk_park_daily_page2.json'); + const first = await server({ + daily: (_url, call) => (call === 1 ? json(page1) : budgetSpent()), + }); + expect(await mainWith(first.fetchFn, [MK, '--out', dir])).toBe(EX_TEMPFAIL); + const after1 = readFileSync(join(dir, `${MK}.ndjson`), 'utf8') + .trim() + .split('\n').length; + + const second = await server({ daily: () => json(page2) }); + expect(await mainWith(second.fetchFn, [MK, '--out', dir])).toBe(0); + const dailyUrl = second.seen.find((u) => u.includes('/history/daily')) ?? ''; + expect(new URL(dailyUrl).searchParams.get('from')).toBe('2026-09-01'); + + const lines = readFileSync(join(dir, `${MK}.ndjson`), 'utf8') + .trim() + .split('\n'); + expect(lines.length).toBe(after1 + 44); + const keys = lines.map((l) => { + const row = JSON.parse(l) as { entityId: string; date: string }; + return `${row.entityId}|${row.date}`; + }); + expect(new Set(keys).size).toBe(keys.length); + expect( + ( + JSON.parse(readFileSync(statePathFor(dir, MK, 'ndjson'), 'utf8')) as { + complete: boolean; + } + ).complete, + ).toBe(true); + }); + + it('writes one CSV header even when the 403 recovery restarts the stream', async () => { + // The Python bug: the writer was built inside the retried closure with + // `written === 0` in the predicate, and the recovery runs exactly when that + // is true. Every CSV on every plan short of the full archive got two header + // rows and pandas read the second as data. + const page2 = await loadFixture('mk_park_daily_page2.json'); + const { fetchFn } = await server({ + daily: (_url, call) => (call === 1 ? json(WINDOW_403, 403) : json(page2)), + }); + expect(await mainWith(fetchFn, [MK, '--format', 'csv', '--out', dir])).toBe(0); + const lines = readFileSync(join(dir, `${MK}.csv`), 'utf8') + .trim() + .split('\n'); + expect(lines.filter((l) => l.startsWith('parkId,'))).toHaveLength(1); + expect(lines[0]).toBe(CSV_COLUMNS.join(',')); + expect(lines).toHaveLength(45); + expect(err.join('')).toContain('reaches back to'); + }); + + it('skips a park whose data ends before the window opens, leaving no file', async () => { + const coverage = (await loadFixture('mk_history_coverage.json')) as { + summary: Record; + }; + coverage.summary.archiveFrom = '2019-01-01'; + coverage.summary.retrievableThrough = '2020-03-15'; + const { fetchFn } = await server({ + coverage: coverage as unknown as Record, + daily: () => json(WINDOW_403, 403), + }); + expect(await mainWith(fetchFn, [MK, '--out', dir])).toBe(0); + expect(err.join('')).toContain('nothing in your window'); + expect(readdirSync(dir)).toEqual([]); + }); + + it('a destination back fills every park in it, one file each', async () => { + const page2 = await loadFixture('mk_park_daily_page2.json'); + const { fetchFn } = await server({ daily: () => json(page2) }); + expect(await mainWith(fetchFn, [WDW, '--out', dir])).toBe(0); + const files = readdirSync(dir).filter((f) => f.endsWith('.ndjson')); + expect(files).toHaveLength(6); + }); + + it('carries on to the next park when one fails, and names what did not finish', async () => { + // One park's 500 used to throw straight out of main, so the parks after it + // were never attempted: a customer paying for the archive got a partial + // download and a traceback that did not say which parks were missing. + // 400 rather than 500 because a 500 is retried, and this is about what + // happens when the retries are spent. + const page2 = await loadFixture('mk_park_daily_page2.json'); + const doomed = '1c84a229-8862-4648-9c71-378ddd2c7693'; // Animal Kingdom + const { fetchFn } = await server({ + daily: (url) => + url.includes(doomed) ? json({ error: { type: 'INVALID_RANGE' } }, 400) : json(page2), + }); + const code = await mainWith(fetchFn, [WDW, '--out', dir]); + expect(code).toBe(1); + const files = readdirSync(dir).filter((f) => f.endsWith('.ndjson')); + expect(files).toHaveLength(5); + expect(files).not.toContain(`${doomed}.ndjson`); + const text = err.join(''); + expect(text).toContain('1 of 6 did not finish'); + expect(text).toContain("Disney's Animal Kingdom Theme Park"); + }); + + it('a spent budget on the coverage call is exit 75, not a traceback', async () => { + // The most likely path after any exit 75: the rerun's first request is + // coverage, and the window is still shut. In Python this escaped the + // handler and exited 1, so a scheduler alerted instead of retrying. + const fetchFn = vi.fn((input: unknown) => { + const url = String(input); + if (url.includes('/destinations')) return Promise.resolve(json(DESTINATIONS)); + return Promise.resolve( + new Response(JSON.stringify(RATE_LIMITED), { + status: 429, + headers: { + 'content-type': 'application/json', + 'retry-after': String((RATE_LIMITED.error as { retryAfter: number }).retryAfter), + }, + }), + ); + }); + expect(await mainWith(fetchFn, [MK, '--out', dir])).toBe(EX_TEMPFAIL); + expect(err.join('')).toContain('budget'); + }); + + it('a second run of a finished park changes nothing', async () => { + const page2 = await loadFixture('mk_park_daily_page2.json'); + const first = await server({ daily: () => json(page2) }); + expect(await mainWith(first.fetchFn, [MK, '--out', dir])).toBe(0); + const before = readFileSync(join(dir, `${MK}.ndjson`), 'utf8'); + const second = await server({ daily: () => json(page2) }); + expect(await mainWith(second.fetchFn, [MK, '--out', dir])).toBe(0); + expect(readFileSync(join(dir, `${MK}.ndjson`), 'utf8')).toBe(before); + expect(second.calls()).toBe(0); + expect(err.join('')).toContain('already complete'); + }); + + it('a typo exits 1 with the listing hint and no traceback', async () => { + const { fetchFn } = await server({ daily: () => json({}, 500) }); + expect(await mainWith(fetchFn, ['Dollywoodd', '--out', dir])).toBe(1); + expect(err.join('')).toContain('--list'); + }); +}); + +/** main() with a stubbed transport. */ +async function mainWith(fetchFn: unknown, argv: string[]): Promise { + return main([...argv, '--api-key', 'test-key'], { fetch: fetchFn }); +} + +describe('what the command tells the server it is', () => { + it('names itself and the SDK version, not one or the other', async () => { + // It used to send `themeparks-backfill/1`: a hardcoded 1, and it replaced + // the SDK's own user agent, so a support question about a bad download had + // no version to work from at either end. + const seen: string[] = []; + const fetchFn = vi.fn((_url: unknown, init?: { headers?: Record }) => { + seen.push(init?.headers?.['user-agent'] ?? ''); + return Promise.resolve( + new Response(JSON.stringify(DESTINATIONS), { + headers: { 'content-type': 'application/json' }, + }), + ); + }); + vi.spyOn(process.stdout, 'write').mockImplementation(() => true); + vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + await main(['--list', 'epcot'], { fetch: fetchFn }); + vi.restoreAllMocks(); + expect(seen[0]).toBe( + `themeparks-backfill/${PACKAGE_VERSION} themeparks-sdk-js/${PACKAGE_VERSION}`, + ); + }); + + it('--version names the command as well as the version', async () => { + const out: string[] = []; + vi.spyOn(process.stdout, 'write').mockImplementation((c) => { + out.push(String(c)); + return true; + }); + const code = await main(['--version']); + vi.restoreAllMocks(); + expect(code).toBe(0); + // A bare `8.3.0` cannot be pasted into a bug report, and the Python SDK + // prints `themeparks-backfill 3.4.0`. + expect(out.join('').trim()).toBe(`themeparks-backfill ${PACKAGE_VERSION}`); + }); +}); + +describe('run', () => { + it('turns a named failure into a sentence and an exit code', async () => { + // The executable is `src/backfill-cli.ts`, which does nothing but call this. + // The previous version put this logic behind an `import.meta.url === + // pathToFileURL(process.argv[1])` guard, which is false for an installed + // binary because npm symlinks it into node_modules/.bin -- so the whole + // command silently did nothing. `scripts/check-package.ts` is the gate for + // that; this is the gate for what it does once it runs. + const err: string[] = []; + vi.spyOn(process.stderr, 'write').mockImplementation((c) => { + err.push(String(c)); + return true; + }); + vi.spyOn(process.stdout, 'write').mockImplementation(() => true); + const fetchFn = vi.fn(() => + Promise.resolve( + new Response(JSON.stringify({ error: { type: 'NOT_FOUND' } }), { + status: 404, + headers: { 'content-type': 'application/json' }, + }), + ), + ); + // A uuid that resolves to nothing: byId passes it through for the API to + // judge, which used to reach the customer as a raw stack. + const code = await run(['00000000-1111-2222-3333-444444444444', '--api-key', 'k'], { + fetch: fetchFn, + }); + vi.restoreAllMocks(); + expect(code).toBe(1); + expect(fetchFn).toHaveBeenCalled(); + expect(err.join('')).not.toContain('at Object.'); + }); +}); + +describe('the CSV contract shared with the Python SDK', () => { + it('matches the checked-in contract column for column', () => { + // `themeparks-backfill` is ONE COMMAND WITH TWO IMPLEMENTATIONS. This SDK wrote + // 32 columns while Python wrote 41, and `inParkScheduledMinutes` against + // `inParkHoursScheduledMinutes` -- a customer using both got two incompatible + // CSVs of the same park. The test that claimed to check it compared four + // strings. Both repos hold an identical copy of this fixture and assert + // against it, so a change in one turns red in the other. + const contract = fixture('csv_contract.json') as unknown as { + fingerprint: string; + columns: string[]; + }; + expect([...CSV_COLUMNS]).toEqual(contract.columns); + expect(columnsFingerprint('csv')).toBe(contract.fingerprint); + }); + + it('the fingerprint tracks the real header', () => { + // Guard the guard: a constant would refuse nothing and agree with everything. + expect(columnsFingerprint('csv')).toHaveLength(16); + expect(columnsFingerprint('ndjson')).toBe(''); + }); +}); + +describe('a failed write is never reported as success', () => { + let dir: string; + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'bf-write-')); + vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + vi.spyOn(process.stdout, 'write').mockImplementation(() => true); + }); + afterEach(() => { + vi.restoreAllMocks(); + chmodSync(dir, 0o755); + rmSync(dir, { recursive: true, force: true }); + }); + + it('an unwritable output file fails the park instead of printing done', async () => { + // Node hands `end`'s callback the stream's error as `cb(err)`; `finish()` took + // no arguments and called `done()` regardless, so a failed stream resolved, + // `record(true)` ran, and the command printed `done: N rows` and exited 0 with + // the file truncated. On ENOSPC mid-download that is a short file marked + // complete, which no rerun would ever continue. + const page = await loadFixture('mk_park_daily_page2.json'); + const coverage = await loadFixture('mk_history_coverage.json'); + const fetchFn = vi.fn((input: unknown) => { + const url = String(input); + const body = url.includes('/history/coverage') + ? coverage + : url.includes('/history/daily') + ? page + : DESTINATIONS; + return Promise.resolve( + new Response(JSON.stringify(body), { headers: { 'content-type': 'application/json' } }), + ); + }); + // A read-only directory: the stream cannot be created or written. + chmodSync(dir, 0o500); + const code = await main([MK, '--out', dir, '--api-key', 'k'], { fetch: fetchFn }); + expect(code).not.toBe(0); + // And no state file claiming the park finished. + const state = statePathFor(dir, MK, 'ndjson'); + if (existsSync(state)) { + expect((JSON.parse(readFileSync(state, 'utf8')) as { complete: boolean }).complete).toBe( + false, + ); + } + }); +}); + +describe('an earlier run’s rows are never deleted', () => { + let dir: string; + let err: string[]; + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'bf-keep-')); + err = []; + vi.spyOn(process.stderr, 'write').mockImplementation((c) => { + err.push(String(c)); + return true; + }); + vi.spyOn(process.stdout, 'write').mockImplementation(() => true); + }); + afterEach(() => { + vi.restoreAllMocks(); + rmSync(dir, { recursive: true, force: true }); + }); + + const partial = () => { + writeFileSync(join(dir, `${MK}.ndjson`), '{"row":1}\n{"row":2}\n'); + stateFile(dir, { lastDay: '2026-08-30', resumeFrom: '2026-09-01' }, MK, 'ndjson'); + }; + + const server = (dailyStatus: number, coverage: Record) => + vi.fn((input: unknown) => { + const url = String(input); + if (url.includes('/destinations')) + return Promise.resolve( + new Response(JSON.stringify(DESTINATIONS), { + headers: { 'content-type': 'application/json' }, + }), + ); + if (url.includes('/history/coverage')) + return Promise.resolve( + new Response(JSON.stringify(coverage), { + headers: { 'content-type': 'application/json' }, + }), + ); + return Promise.resolve( + new Response(JSON.stringify({ error: { type: 'INVALID_RANGE' } }), { + status: dailyStatus, + headers: { 'content-type': 'application/json' }, + }), + ); + }); + + it('keeps the file when a resumed run fails before writing a row', async () => { + // `written === 0` means "this process wrote nothing", which on a resumed run is + // not "the file is empty". Deleting it destroyed the archive and left the state + // pointing mid-range, so the next run appended only the tail and recorded + // complete. + partial(); + const coverage = await loadFixture('mk_history_coverage.json'); + const code = await main([MK, '--out', dir, '--api-key', 'k'], { + fetch: server(400, coverage), + }); + expect(code).toBe(1); + expect(existsSync(join(dir, `${MK}.ndjson`))).toBe(true); + expect( + readFileSync(join(dir, `${MK}.ndjson`), 'utf8') + .trim() + .split('\n'), + ).toHaveLength(2); + }); + + it('keeps the file when the window closes under a resumed run, and fails', async () => { + // A key rotated out of a cron's environment, or a lapsed subscription. Deleting + // and exiting 0 made the scheduler log success and left a permanent trap. + partial(); + const coverage = (await loadFixture('mk_history_coverage.json')) as { + summary: Record; + }; + coverage.summary.archiveFrom = '2019-01-01'; + coverage.summary.retrievableThrough = '2020-03-15'; + const code = await main([MK, '--out', dir, '--api-key', 'k'], { + fetch: server(200, coverage as unknown as Record), + }); + expect(code).toBe(1); + expect(existsSync(join(dir, `${MK}.ndjson`))).toBe(true); + expect(err.join('')).toContain('left alone'); + }); +}); + +describe('a state file this build cannot resume', () => { + let dir: string; + let err: string[]; + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'bf-state-')); + err = []; + vi.spyOn(process.stderr, 'write').mockImplementation((c) => { + err.push(String(c)); + return true; + }); + }); + afterEach(() => { + vi.restoreAllMocks(); + rmSync(dir, { recursive: true, force: true }); + }); + + const opts = (format: 'csv' | 'ndjson' = 'ndjson') => ({ outDir: dir, format, overwrite: false }); + + it('refuses a different column layout rather than appending under the old header', () => { + writeFileSync(join(dir, 'p.csv'), 'old,header\n1,2\n'); + stateFile(dir, { columns: '0000deadbeef0000', lastDay: '2026-08-30' }, 'p', 'csv'); + expect( + decide(join(dir, 'p.csv'), statePathFor(dir, 'p', 'csv'), opts('csv'), '2021-07-03'), + ).toBe(1); + expect(err.join('')).toContain('column layout changed'); + }); + + it('refuses a state file from the other SDK', () => { + // Same filename, and format/start/end/complete spelled identically, so the safe + // paths interoperated and nothing warned -- while a Python run interrupted at 64 + // rows and resumed here produced 172 rows, 64 duplicated, marked complete. + writeFileSync(join(dir, 'p.ndjson'), '{"a":1}\n'); + stateFile(dir, { sdk: 'py', lastDay: '2026-08-30' }); + expect( + decide(join(dir, 'p.ndjson'), statePathFor(dir, 'p', 'ndjson'), opts(), '2021-07-03'), + ).toBe(1); + expect(err.join('')).toContain('py SDK'); + }); + + it('refuses a future state version', () => { + writeFileSync(join(dir, 'p.ndjson'), '{"a":1}\n'); + stateFile(dir, { stateVersion: 99, lastDay: '2026-08-30' }); + expect( + decide(join(dir, 'p.ndjson'), statePathFor(dir, 'p', 'ndjson'), opts(), '2021-07-03'), + ).toBe(1); + expect(err.join('')).toContain('different version'); + }); + + it('gives each format its own state file', () => { + expect(statePathFor(dir, 'p', 'csv')).not.toBe(statePathFor(dir, 'p', 'ndjson')); + expect(statePathFor(dir, 'p', 'csv')).toContain('p.csv.backfill-state.json'); + }); +}); + +describe('flushAndClose', () => { + it('rejects with the error Node hands the end callback', async () => { + // The shipped version took no arguments and called done() regardless, so a + // failed stream resolved and the park was recorded complete. + const boom = new Error('EACCES: permission denied'); + await expect(flushAndClose({ end: (cb) => cb(boom) }, () => null)).rejects.toThrow('EACCES'); + }); + + it('rejects with an error the stream reported earlier, even if end succeeds', async () => { + // `write()` never throws synchronously, so the failure can arrive on the + // 'error' event long before the flush. Either route has to fail the park. + const earlier = new Error('ENOSPC: no space left on device'); + await expect(flushAndClose({ end: (cb) => cb(null) }, () => earlier)).rejects.toThrow('ENOSPC'); + }); + + it('resolves when the stream closed cleanly', async () => { + await expect(flushAndClose({ end: (cb) => cb(null) }, () => null)).resolves.toBeUndefined(); + }); +}); + +describe('the spreadsheet is the reader', () => { + let dir: string; + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'bf-sheet-')); + vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + vi.spyOn(process.stdout, 'write').mockImplementation(() => true); + }); + afterEach(() => { + vi.restoreAllMocks(); + rmSync(dir, { recursive: true, force: true }); + }); + + async function writeCsv(entityName?: string): Promise { + const page = (await loadFixture('mk_park_daily_page2.json')) as { + entities: { name: string }[]; + }; + if (entityName !== undefined) page.entities[0]!.name = entityName; + const coverage = await loadFixture('mk_history_coverage.json'); + const fetchFn = vi.fn((input: unknown) => { + const url = String(input); + const body = url.includes('/history/coverage') + ? coverage + : url.includes('/history/daily') + ? page + : DESTINATIONS; + return Promise.resolve( + new Response(JSON.stringify(body), { headers: { 'content-type': 'application/json' } }), + ); + }); + const code = await main([MK, '--format', 'csv', '--out', dir, '--api-key', 'k'], { + fetch: fetchFn, + }); + expect(code).toBe(0); + return readFileSync(join(dir, `${MK}.csv`)); + } + + it('starts the file with a single UTF-8 BOM', async () => { + // Without it Excel on Windows reads the local code page and renders + // `Walt Disney World® Resort` as mojibake. + const raw = await writeCsv(); + expect(raw.subarray(0, 3)).toEqual(Buffer.from([0xef, 0xbb, 0xbf])); + expect( + [...raw].filter((_b, i) => raw.subarray(i, i + 3).equals(Buffer.from([0xef, 0xbb, 0xbf]))), + ).toHaveLength(1); + }); + + it('quotes a carriage return so one row cannot become two', async () => { + // A bare CR went through unquoted: one row parsed as two with every later + // column shifted, and Python quoted it, so the files diverged as well. + const text = (await writeCsv('Space Mountain\rFastPass')).toString('utf8').replace(/^/u, ''); + expect(text).toContain('"Space Mountain\rFastPass"'); + const rows = text + .trim() + .split('\n') + .filter((l) => !l.includes('\r') || l.startsWith('"')); + expect(rows.length).toBeGreaterThan(1); + }); + + it('defuses a name a spreadsheet would execute', async () => { + const text = (await writeCsv("=cmd|' /C calc'!A0")).toString('utf8'); + expect(text).toContain("'=cmd"); + }); +}); + +describe('defuse', () => { + it('prefixes what a spreadsheet would run', () => { + expect(defuse('=1+1')).toBe("'=1+1"); + expect(defuse('@SUM(1)')).toBe("'@SUM(1)"); + expect(defuse('\tx')).toBe("'\tx"); + expect(defuse('\rx')).toBe("'\rx"); + }); + + it('leaves a number a number', () => { + // The reason this is not a bare test of the leading character: prefixing `-5` + // turns every negative value in the file into text. + expect(defuse('-5')).toBe('-5'); + expect(defuse('-5.25')).toBe('-5.25'); + expect(defuse('+1')).toBe('+1'); + expect(defuse('1e-7')).toBe('1e-7'); + expect(defuse('.5')).toBe('.5'); + }); + + it('agrees with the Python SDK on the awkward cases', () => { + // `Number('\t')` is 0, so a naive numeric test leaves a tab-led cell + // undefended here and defended there, for a file both claim to write + // identically. Both use the same strict pattern. + expect(defuse('\t5')).toBe("'\t5"); + expect(defuse('+')).toBe("'+"); + expect(defuse('-')).toBe("'-"); + expect(defuse('Walt Disney World® Resort')).toBe('Walt Disney World® Resort'); + }); +}); + +describe('argument handling', () => { + let out: string[]; + let err: string[]; + beforeEach(() => { + out = []; + err = []; + vi.spyOn(process.stdout, 'write').mockImplementation((c) => { + out.push(String(c)); + return true; + }); + vi.spyOn(process.stderr, 'write').mockImplementation((c) => { + err.push(String(c)); + return true; + }); + }); + afterEach(() => { + vi.restoreAllMocks(); + }); + + const dests = () => + vi.fn(() => + Promise.resolve( + new Response(JSON.stringify(DESTINATIONS), { + headers: { 'content-type': 'application/json' }, + }), + ), + ); + + it('accepts a bare --list, as the help says it does', async () => { + // `type: 'string'` made this "argument missing" and exit 2, while the help + // advertised `--list [TEXT]` and the Python SDK listed everything. + expect(await main(['--list'], { fetch: dests() })).toBe(0); + expect(out.join('').split('\n').length).toBeGreaterThan(100); + }); + + it('still takes a filter, and does not eat the next flag', async () => { + expect(await main(['--list', 'epcot'], { fetch: dests() })).toBe(0); + expect(out.join('')).toContain('all 6 parks (1 shown)'); + out.length = 0; + expect(await main(['--list', '--api-key', 'k'], { fetch: dests() })).toBe(0); + expect(out.join('').split('\n').length).toBeGreaterThan(100); + }); + + it('accepts -h', async () => { + expect(await main(['-h'])).toBe(0); + expect(out.join('')).toContain('themeparks-backfill'); + }); + + it('asks which park before spending a request', async () => { + // This fetched /destinations first, so no arguments and no network exited 75: + // telling a scheduler to retry a command that can never succeed. + const fetchFn = dests(); + expect(await main([], { fetch: fetchFn })).toBe(2); + expect(fetchFn).not.toHaveBeenCalled(); + }); + + it('treats an empty API key as no key', async () => { + // `THEMEPARKS_API_KEY=""` is a clobbered env var in a cron file, which is how + // this happens; it read as a key and the anonymous notice never printed. Not + // via --list, which suppresses the notice deliberately: listing works without + // a key and saying so there would be noise. + const fetchFn = vi.fn((input: unknown) => + Promise.resolve( + String(input).includes('/destinations') + ? new Response(JSON.stringify(DESTINATIONS), { + headers: { 'content-type': 'application/json' }, + }) + : new Response(JSON.stringify({ error: { type: 'NOT_FOUND' } }), { + status: 404, + headers: { 'content-type': 'application/json' }, + }), + ), + ); + const dir = mkdtempSync(join(tmpdir(), 'bf-key-')); + try { + await main([MK, '--out', dir, '--api-key', ''], { fetch: fetchFn }); + expect(err.join('')).toContain('no API key'); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }); + + it('lists an ambiguous name by park name, not by uuid', async () => { + // Sorting the formatted line sorts by the uuid it starts with, so ten + // "Hurricane Harbor" parks came back in an order that looks random. + expect(await main(['Hurricane Harbor', '--api-key', 'k'], { fetch: dests() })).toBe(1); + const listed = err + .join('') + .split('\n') + .filter((l) => /^ {2}[0-9a-f]{8}-/u.test(l)); + expect(listed.length).toBeGreaterThan(5); + const names = listed.map((l) => l.split(' ')[2] ?? ''); + expect(names).toEqual([...names].sort()); + }); + + it('matches nothing for a query that folds to nothing', async () => { + // normalize('東京') is '', and ''.includes is true of every string, so this + // listed all 127 parks as candidates. + expect(await main(['東京', '--api-key', 'k'], { fetch: dests() })).toBe(1); + expect(err.join('')).toContain('no park or destination matching'); + }); +}); diff --git a/test/unit/history-paging.test.ts b/test/unit/history-paging.test.ts index 8d29e51..ac1a7e3 100644 --- a/test/unit/history-paging.test.ts +++ b/test/unit/history-paging.test.ts @@ -292,3 +292,86 @@ describe('the API key', () => { expect(seen[1]!['x-api-key']).toBe('tpw_example'); }); }); + +describe('what a daily row is labelled with', () => { + it('names each row from the history response, not a lookup', async () => { + // The name AS RECORDED. A park's current /children list gives today's name, + // and stamping that on a row from three years ago rewrites the record — + // rides get renamed, and the history is supposed to be what was true then. + // It is also already in the payload, so there is nothing to ask. + const page = await loadFixture('mk_park_daily_page1.json'); + const entities = page.entities as { id: string; name: string; entityType: string }[]; + const rows = await collect( + client(vi.fn(() => Promise.resolve(json({ ...page, next: null })))) + .entity('mk') + .history.days(), + ); + expect(rows.length).toBeGreaterThan(50); + for (const row of rows) { + const source = entities.find((e) => e.id === row.entityId); + expect(source).toBeDefined(); + expect(row.name).toBe(source?.name); + expect(row.entityType).toBe(source?.entityType); + } + // Three different types in this capture, so a single hardcoded value fails. + expect(new Set(rows.map((r) => r.entityType)).size).toBe(3); + }); +}); + +describe('the page hook', () => { + it('reports the range and the next page the server gave', async () => { + const page1 = await loadFixture('mk_park_daily_page1.json'); + const page2 = await loadFixture('mk_park_daily_page2.json'); + let call = 0; + const fetchFn = vi.fn(() => { + call += 1; + return Promise.resolve(json(call === 1 ? page1 : page2)); + }); + const pages: unknown[] = []; + await collect( + client(fetchFn) + .entity('mk') + .history.days({ onPage: (p) => pages.push(p) }), + ); + expect(pages).toEqual([ + { from: '2026-08-01', to: '2026-08-31', next: page1.next }, + { from: '2026-09-01', to: '2026-09-20', next: null }, + ]); + }); + + it('fires only after every row of its page has been yielded', async () => { + // A resumable download checkpoints on this. If it fired first, a consumer + // that died mid-page would have recorded a checkpoint past rows it never + // wrote, and those rows would be missing from the file for good — the one + // failure mode worse than duplicating them. + const page1 = await loadFixture('mk_park_daily_page1.json'); + const page2 = await loadFixture('mk_park_daily_page2.json'); + let call = 0; + const fetchFn = vi.fn(() => { + call += 1; + return Promise.resolve(json(call === 1 ? page1 : page2)); + }); + const order: string[] = []; + for await (const row of client(fetchFn) + .entity('mk') + .history.days({ onPage: (p) => order.push(`page:${p.to}`) })) { + order.push(`row:${(row.row as { date: string }).date}`); + } + const firstPageAt = order.indexOf('page:2026-08-31'); + const lastRowOfPage1 = order.lastIndexOf('row:2026-08-31'); + expect(firstPageAt).toBeGreaterThan(lastRowOfPage1); + // And every row of page one comes before the page-one boundary. + const page1Dates = new Set( + (page1.entities as { days: { date: string }[] }[]).flatMap((e) => e.days.map((d) => d.date)), + ); + for (const [i, item] of order.entries()) { + if ( + item.startsWith('row:') && + page1Dates.has(item.slice(4)) && + !order.slice(0, i).includes('page:2026-08-31') + ) { + expect(i).toBeLessThan(firstPageAt); + } + } + }); +}); diff --git a/tsup.config.ts b/tsup.config.ts index cbce0b9..fce6b78 100644 --- a/tsup.config.ts +++ b/tsup.config.ts @@ -1,7 +1,11 @@ import { defineConfig } from 'tsup'; export default defineConfig({ - entry: ['src/index.ts'], + // Three entries: the library, the back fill module (importable, and what the + // tests drive), and the executable `bin` points at. `format` is global, so each + // one is emitted as both ESM and CJS; the CLI's CJS build is unreferenced but + // harmless, and tsup preserves the shebang on the ESM one that `bin` names. + entry: ['src/index.ts', 'src/backfill.ts', 'src/backfill-cli.ts'], format: ['esm', 'cjs'], dts: true, sourcemap: true,