From 210f6397ce22fab50c78b39ce57bc855c7d80b86 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Sat, 29 Aug 2026 00:00:49 +0200 Subject: [PATCH 1/2] Retire Chronicle target profiles --- .github/chronicle-agents.yml | 16 - .github/pull_request_template.md | 4 +- .github/workflows/ci.yml | 1 - AGENTS.md | 13 +- LANE_C2_REPORT.md | 10 +- LANE_C5_REPORT.md | 4 +- README.md | 56 +- chronicle/__init__.py | 2 +- chronicle/core.py | 15 - chronicle/harness.py | 21 +- docs/adr-chronicle-facts-only.md | 26 +- docs/agent-source-package-harness.md | 19 +- docs/architecture.md | 12 +- docs/chronicle-governance.md | 12 +- docs/pe-uk-source-checklist.md | 20 +- docs/target-construction-harness-plan.md | 40 +- policyengine_chronicle/__init__.py | 10 - policyengine_chronicle/consumer.py | 1005 +------------ .../target_profiles/__init__.py | 19 - .../target_profiles/model.py | 342 ----- .../target_profiles/uk_firms.json | 272 ---- .../target_profiles/uk_local_geography.json | 346 ----- policyengine_chronicle/targets/__init__.py | 4 +- pyproject.toml | 2 +- tests/test_chronicle_consumer.py | 1265 ++--------------- tests/test_chronicle_consumer_contract.py | 65 +- tests/test_chronicle_governance.py | 4 +- tests/test_chronicle_source_package.py | 100 -- tests/test_policyengine_chronicle_imports.py | 24 +- ..._policyengine_chronicle_target_profiles.py | 305 ---- 30 files changed, 271 insertions(+), 3763 deletions(-) delete mode 100644 policyengine_chronicle/target_profiles/__init__.py delete mode 100644 policyengine_chronicle/target_profiles/model.py delete mode 100644 policyengine_chronicle/target_profiles/uk_firms.json delete mode 100644 policyengine_chronicle/target_profiles/uk_local_geography.json delete mode 100644 tests/test_policyengine_chronicle_target_profiles.py diff --git a/.github/chronicle-agents.yml b/.github/chronicle-agents.yml index e2e818ab..43f302ca 100644 --- a/.github/chronicle-agents.yml +++ b/.github/chronicle-agents.yml @@ -19,20 +19,6 @@ approved_agents: required_judges: - ledger-source-fidelity - ledger-boundary - - id: ledger-target-profile-author - purpose: Add source-backed target profiles and model measurement contracts without target values. - allowed_paths: - - policyengine_chronicle/target_profiles/** - - chronicle/targets/** - - tests/test_policyengine_chronicle_target_profiles.py - - tests/test_us_poverty_target_coverage.py - required_deterministic_checks: - - no embedded target values - - sum-only operation - - selector references raw Chronicle facts - required_judges: - - ledger-target-profile - - ledger-boundary - id: ledger-contract-maintainer purpose: Change Chronicle schemas, identity, provenance, or consumer contracts. allowed_paths: @@ -54,8 +40,6 @@ approved_agents: required_judges: ledger-source-fidelity: verdict: PASS if every new fact is directly traceable to publisher bytes/cells and no source values are invented. - ledger-target-profile: - verdict: PASS if profiles contain selectors and measurement contracts only, with no target values or active calibration decisions. ledger-contract: verdict: PASS if schema or identity changes preserve source provenance and do not move Microcosm responsibilities into Chronicle. ledger-boundary: diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index 384c0c18..d5c5300b 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -2,17 +2,15 @@ ## Chronicle Governance -For any source package, target profile, consumer contract, schema, or source-data +For any source package, consumer contract, schema, or source-data boundary change: - Approved Chronicle agent role: - `ledger-source-ingestor` - - `ledger-target-profile-author` - `ledger-contract-maintainer` - Deterministic checks run: - LLM judge verdicts: - `ledger-source-fidelity`: - - `ledger-target-profile`: - `ledger-contract`: - `ledger-boundary`: diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 39c24093..6fd6d616 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -58,7 +58,6 @@ jobs: import policyengine_chronicle import policyengine_chronicle.normalization import policyengine_chronicle.sources - import policyengine_chronicle.target_profiles import policyengine_chronicle.targets from policyengine_chronicle.schema import ( CONSUMER_FACT_SCHEMA_SHA256, diff --git a/AGENTS.md b/AGENTS.md index be14e33e..0886b2cf 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,7 +1,8 @@ # Chronicle Agent Rules Chronicle is a source-backed fact store. It may parse publisher artifacts, normalize -representation, preserve provenance, and declare target-profile contracts. +representation, and preserve provenance. Selection and measurement contracts live +in consumers such as Microcosm. Every fact value must trace to a publisher. The boundary is who asserted the value, not level versus projection: a publisher's own projection (CBO baseline, @@ -16,13 +17,13 @@ Do not put Microcosm work in Chronicle: - no imputation - no support-aware target activation - no solver-ready target construction -- no target values in target profiles +- no target profiles or model-measurement bindings - no PolicyEngine-computed values stored as facts -Resolving profile targets at a period other than a fact's reference period -requires the consumer's explicit `PeriodAlignmentDeclaration`; Chronicle records -the declaration and returns the published level, never the aligned number. +Chronicle records every fact's publisher reference period. Consumers own and +enforce any declaration that aligns those facts to another period; Chronicle +returns the published level and never an aligned number. Only approved Chronicle agent roles in `.github/chronicle-agents.yml` should add or -modify source packages, target profiles, or contract schemas. Source-data PRs +modify source packages or contract schemas. Source-data PRs need deterministic validation plus the listed Chronicle judge reviews before merge. diff --git a/LANE_C2_REPORT.md b/LANE_C2_REPORT.md index 6873346f..78c1e80c 100644 --- a/LANE_C2_REPORT.md +++ b/LANE_C2_REPORT.md @@ -13,8 +13,8 @@ The cached PDF resolves three material ambiguities in the task shorthand: - The Annex columns put most requested External series in publisher periods 2021–2023, not 2022–2024. Chronicle preserves the printed periods. A consumer - that needs another period must supply an explicit `PeriodAlignmentDeclaration`; - Chronicle returns the published level and never an aligned number. + that needs another period owns the contract for aligning it; Chronicle + returns the published level and never an aligned number. - The `poa=2,273k` and `psu=79k` observations are in the SILC column. Their External cells are `NaN`, so they are stored as `survey_aggregate` and are validation-only. @@ -439,8 +439,8 @@ acceptance, and both required Chronicle judge reviews pass. `Add Belgium national accounts and JRC external facts`. - No push and no R2 upload were performed. The two manifests merely declare the required content-addressed R2 keys. -- A consumer using a fact outside its publisher period must supply an explicit - `PeriodAlignmentDeclaration`; Chronicle returns the published value and does - not compute an aligned value. +- A consumer using a fact outside its publisher period owns the alignment + contract; Chronicle returns the published value and does not compute an + aligned value. LANE C2 DONE diff --git a/LANE_C5_REPORT.md b/LANE_C5_REPORT.md index f22c9052..4f89b4b6 100644 --- a/LANE_C5_REPORT.md +++ b/LANE_C5_REPORT.md @@ -355,7 +355,7 @@ a within-source pair ending in 2025 in this lane. Any choice of proxy family, ratio calculation, aging, cross-source mapping, period alignment, target activation, or solver construction belongs to the consumer. When a target period differs from a fact reference period, the -consumer must provide its explicit `PeriodAlignmentDeclaration`; Chronicle -returns the published levels, never the aligned number. +consumer-owned contract must declare the alignment; Chronicle returns the +published levels, never the aligned number. LANE C5 DONE diff --git a/README.md b/README.md index 288f8832..042485ee 100644 --- a/README.md +++ b/README.md @@ -10,15 +10,13 @@ values as structured, queryable facts. Chronicle may normalize structure: parse files, type values, declare units and scales, assign geography and period identifiers, preserve lineage back to -source artifacts, and publish target profiles that identify source-backed facts -and measurement contracts. Chronicle does not reconcile inconsistent sources, -impute missing data, store raw survey microdata, or execute simulator-specific -calibration. +source artifacts, and publish source-backed facts. Chronicle does not own +selection or measurement contracts, reconcile inconsistent sources, impute +missing data, store raw survey microdata, or execute simulator-specific calibration. -Microcosm consumes Chronicle facts and target profiles, selects the subset its -current support universe can target, applies minimal period alignment when -declared, and runs calibration. Thesis can consume the same facts and -measurement contracts as official observations. +Microcosm consumes Chronicle facts, owns the contracts that select and bind +them, applies declared period alignment, and runs calibration. Thesis can +consume the same facts as official observations. ## Purpose @@ -35,11 +33,8 @@ This repository provides: their `survey_instrument`. - **Normalization**: Low-assumption representation changes such as unit/scale conversion and source-published total/share arithmetic. -- **Target profiles**: Source-backed target contracts and model-measurement - bindings that Microcosm, Thesis, and future rule engines can consume. - **Consumer artifacts**: Versioned, reproducible bundles of consumer-contract - fact rows plus profiles (`chronicle build-consumer-artifact`) with a resolution - API that enforces the period contract. + fact rows plus manifest hashes (`chronicle build-consumer-artifact`). - **Jurisdiction loaders**: Source-specific ETL that emits the shared Chronicle schema. @@ -50,27 +45,24 @@ They are source-backed claims with provenance. The load-bearing rule: -> Chronicle may re-express a published value and declare target contracts, but may -> not reconcile, impute, or transform published values in ways that change their -> meaning. +> Chronicle may re-express a published value, but may not select it for a +> consumer or transform it in ways that change its meaning. The store is facts-only, and the line is who asserted the value. Everything a publisher asserted — including the publisher's own projections — is a fact. Everything PolicyEngine computes (aged, uprated, forecast, or reconciled levels) is a downstream build artifact and never enters the store; Microcosm owns aging as a named, versioned model over Chronicle growth-factor facts. A -fact's `period` is the period its value refers to, and resolving a target at -any other period hard-fails without an explicit consumer -`PeriodAlignmentDeclaration` — the guard against silent un-aged calibration -(see [`docs/adr-chronicle-facts-only.md`](docs/adr-chronicle-facts-only.md)). +fact's `period` is the period its value refers to. Consumers must enforce any +contract that aligns it to another period (see +[`docs/adr-chronicle-facts-only.md`](docs/adr-chronicle-facts-only.md)). | Layer | Owns | Examples | |-------|------|----------| | Chronicle Sources | Source artifacts and provenance | URLs, checksums, source files, parsed tables/cells | | Chronicle Facts | Structured source claims | SOI cells, ACS estimates, CPI values, CBO-published projections | | Chronicle Normalization | Representation changes | Unit scales, typed values, geography/date identifiers | -| Chronicle Target Profiles | Source-backed calibration contracts | SOI EITC totals, CBO baselines, source-published growth factors, measurement bindings | -| Microcosm Targets | Build-ready active subset | Support-aware activation, solver inputs, diagnostics | +| Microcosm Target Contracts | Selection, measurement bindings, and active subset | Period alignment, support-aware activation, solver inputs, diagnostics | The storage split is documented in [`docs/storage-architecture.md`](docs/storage-architecture.md): `ledger-raw` @@ -126,8 +118,8 @@ chronicle/ └── docs/ # Architecture and source documentation ``` -New code should prefer `policyengine_chronicle` for source-backed fact and target -profile consumers. Existing in-repo implementation code may continue using +New code should prefer `policyengine_chronicle` for source-backed fact +consumers. Existing in-repo implementation code may continue using legacy implementation modules while the namespace migration is completed. Solver execution and calibrated dataset construction belong in Microcosm. @@ -493,10 +485,8 @@ LEDGER_EXPLORER_DATA_DIRS=/tmp/chronicle-build-a,/tmp/chronicle-build-b npm run ```python from policyengine_chronicle.targets import DataSource, Target, TargetType, query_targets -from policyengine_chronicle.target_profiles import load_target_profile target_rows = query_targets(jurisdiction="us", year=2024) -profile = load_target_profile("us_fiscal") ``` ## Target Input Schema @@ -508,8 +498,8 @@ Target inputs use a three-table schema: - **stratum_constraints**: Rules defining each stratum. - **targets**: Source-published aggregate values linked to strata. -These are inputs to Chronicle target profiles. Microcosm owns the active -support-aware subset and calibrated solver execution. +These are source-backed inputs. Microcosm owns the contracts that select them, +the active support-aware subset, and calibrated solver execution. ## Chronicle Facts And Microcosm Targets @@ -521,8 +511,8 @@ publishes the total/share relationship. Inflation, cross-source reconciliation, and support-aware activation belong in Microcosm unless the source itself publishes the adjusted or projected series. -Target profiles in Chronicle may declare the source-backed rows and measurement -bindings Microcosm is allowed to activate. +Microcosm contracts declare which source-backed rows and measurement bindings a +build may activate. ```python from policyengine_chronicle.facts import SourceFact @@ -558,10 +548,10 @@ normalized_fact = convert_units(fact, 1000, "count") ## Boundaries - **Chronicle** owns government-statistics release artifacts, provenance, source - facts, aggregate facts, target profiles, and measurement contracts. -- **Microcosm** owns support-aware target activation, minimal period alignment, - raw microdata access, simulation interfaces, entity modeling, weights, - diagnostics, and calibration execution. + facts, and aggregate facts. +- **Microcosm** owns selection and measurement contracts, support-aware target + activation, period alignment, raw microdata access, simulation interfaces, + entity modeling, weights, diagnostics, and calibration execution. - **Jurisdiction source packages** such as `ledger-us` and `chronicle-uk` own source-specific parsers and specs that emit shared Chronicle records. - **Jurisdiction simulation packages** own simulation-specific variable diff --git a/chronicle/__init__.py b/chronicle/__init__.py index be6130fb..04d50943 100644 --- a/chronicle/__init__.py +++ b/chronicle/__init__.py @@ -1,7 +1,7 @@ """Chronicle source-data foundation. Chronicle owns government-statistics releases: source artifacts, source-backed -facts, constraints, provenance, and target profiles. Raw microdata storage, +facts, constraints, and provenance. Selection contracts, raw microdata storage, source reconciliation, aging, imputation, target activation, and calibration belong in downstream systems such as Microcosm. """ diff --git a/chronicle/core.py b/chronicle/core.py index 744b2558..c43f7574 100644 --- a/chronicle/core.py +++ b/chronicle/core.py @@ -82,21 +82,6 @@ } ALLOWED_ASSERTIONS = {"observation", "source_projection"} DEFAULT_ASSERTION = "observation" -# How profile resolution treats the assertion axis. observed_only is the safe -# default: projections are invisible and a projection-only family fails loudly. -# prefer_observed applies per series (one geography/entity/dimension tuple): -# a series with any observed fact resolves only from observations, and a -# series with none may fall back to projections — no single series ever -# mixes bases across periods, and a projection-only series is never starved -# by a neighbouring series' observation. allow_source_projection treats both -# equally (for forecast families such as the OBR EFO lines), with the -# observation winning an exact-period tie within a series. -ASSERTION_POLICIES = { - "observed_only", - "prefer_observed", - "allow_source_projection", -} -DEFAULT_ASSERTION_POLICY = "observed_only" ALLOWED_PROVENANCE_CLASSES = { "administrative", "census", diff --git a/chronicle/harness.py b/chronicle/harness.py index d9134b2e..e45f3604 100644 --- a/chronicle/harness.py +++ b/chronicle/harness.py @@ -754,10 +754,8 @@ def main(argv: list[str] | None = None) -> int: consumer_artifact_parser = subparsers.add_parser( "build-consumer-artifact", - help=( - "Build a versioned consumer artifact from consumer-contract facts " - "and Chronicle target profiles" - ), + help="Build a versioned facts-only consumer artifact", + description="Build a versioned facts-only consumer artifact.", ) consumer_artifact_parser.add_argument( "--facts", @@ -765,19 +763,6 @@ def main(argv: list[str] | None = None) -> int: required=True, help="Path to a consumer_facts.jsonl file or a bundle directory", ) - consumer_artifact_parser.add_argument( - "--profile", - action="append", - default=[], - help="Packaged target profile id (may be repeated)", - ) - consumer_artifact_parser.add_argument( - "--profile-path", - action="append", - type=Path, - default=[], - help="Path to a target profile JSON file (may be repeated)", - ) consumer_artifact_parser.add_argument( "--out", type=Path, @@ -1285,8 +1270,6 @@ def main(argv: list[str] | None = None) -> int: artifact_report = build_consumer_artifact( args.out, facts_path=args.facts, - profile_ids=args.profile, - profile_paths=args.profile_path, replace=args.replace, ) print(json.dumps(artifact_report.to_dict(), indent=2, sort_keys=True)) diff --git a/docs/adr-chronicle-facts-only.md b/docs/adr-chronicle-facts-only.md index fbcb919d..39bc1d34 100644 --- a/docs/adr-chronicle-facts-only.md +++ b/docs/adr-chronicle-facts-only.md @@ -20,22 +20,16 @@ the value**, not level versus projection: aging), implemented as named, versioned models that consume growth-factor facts from Chronicle and emit their own lineage. -Instead of projection objects, Chronicle contributes three guarantees: +Instead of projection objects, Chronicle contributes two guarantees: - **Reference-period semantics.** `PeriodDimension` identifies the period a value refers to; `PeriodCoverage` records non-identity provenance (start and end dates, basis, the publisher's period label, accounting basis) for cases like BE-SILC incomes that reference the year before the survey label. -- **Period-contract enforcement.** Resolving a profile target at a period - other than the fact's reference period raises `PeriodContractError` - unless the consumer passes an explicit `PeriodAlignmentDeclaration` - (model id, version, parameters — never values). Chronicle records the - declaration in resolved rows and returns the published level untouched. -- **Basis-aware diagnostics.** Resolved rows carry `basis` (`fact` or - `declared_alignment`), `fact_period`, `requested_period`, and the - declaration, so downstream diagnostics distinguish "missed a published - fact" from "missed an aged level." +- **Facts-only consumer artifacts.** Chronicle publishes schema-validated fact + rows with manifest hashes. Consumers own the selection, measurement, + period-alignment, and model-binding contracts that interpret those rows. ## Why not facts plus projections in one schema @@ -67,9 +61,13 @@ Instead of projection objects, Chronicle contributes three guarantees: values other than `observation` and `source_projection` fail validation with an error explaining that PolicyEngine-computed values are not facts. - Consumer-contract rows always carry `assertion` explicitly, and the - consumer artifact (`chronicle build-consumer-artifact`) embeds profiles, - fact rows, coverage diagnostics, and manifest hashes so Microcosm can - build a target registry without database access or copied values - (issue #61). + consumer artifact (`chronicle build-consumer-artifact`) contains only fact + rows and the manifest hashes needed to verify them. Microcosm packages its + own selection contracts and builds its target registry without Chronicle + profiles (issues #166 and #172). +- The retired `policyengine_ledger.target_profile.v1` and + `policyengine_ledger.resolved_target.v1` schema IDs have no v2 successor in + issue #143. The Chronicle-side Belgian profile plan in issue #70 is + superseded; Belgian contracts also live consumer-side. - Geography vintage translation (microcosm#205) follows the same pattern: a declared consumer-side transform over facts, never an edit to them. diff --git a/docs/agent-source-package-harness.md b/docs/agent-source-package-harness.md index 546c589c..5816ec47 100644 --- a/docs/agent-source-package-harness.md +++ b/docs/agent-source-package-harness.md @@ -680,8 +680,8 @@ budget-allocation years). `provenance_class` says how the publisher measured; national-balance-sheet estimate is `model_output` + `observation`, an NPP projection year is `model_output` + `source_projection`. -Consumer profiles resolve this axis explicitly rather than by convention. -Each profile (or individual target) declares an `assertion_policy`: +Microcosm's consumer-owned contracts resolve this axis explicitly rather than +by convention. A contract can declare an `assertion_policy` such as `observed_only` (the default — projections are invisible and a projection-only family fails loudly with `only_projection_facts`), `prefer_observed` (per series — one geography/entity/dimension tuple: a @@ -692,17 +692,10 @@ neighbouring series' observation), or `allow_source_projection` (both compete under the period policy; an observation and a projection colliding within one series at the chosen period resolve to the observation with an `ambiguous_assertion_at_period` -warning naming the series rather than double-counting it). A target whose `chronicle_selector` names `assertion` -explicitly bypasses the policy — the selector is already maximal intent — -and declaring both on one target is rejected at profile load as a -contradiction. Whenever a -projection is resolved, the report carries a `resolved_from_projection` -warning and the resolved row exposes its `assertion`, so downstream builds -never discover the estimate/projection boundary by accident. A fact that -predates the axis loads as `observation`: both the file loader and the -resolver default a missing `assertion` and reject unknown values, so every -pre-existing package keeps its meaning and a typo'd assertion fails loudly -instead of vanishing from the candidate set. +warning naming the series rather than double-counting it). Those selector and +resolution rules are validated in Microcosm. Chronicle's responsibility is to +emit the typed `assertion` and reject unknown values so consumers never infer +the estimate/projection boundary from convention. ```yaml record_sets: diff --git a/docs/architecture.md b/docs/architecture.md index 2a2048e8..9492c9b3 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -8,7 +8,7 @@ structured, queryable facts. Microcosm consumes Chronicle facts to produce final calibrated simulation inputs. This document describes the source-publication pipeline from government -statistics releases to source-backed facts and target profiles. Chronicle is global at the schema, +statistics releases to source-backed facts. Chronicle is global at the schema, validation, and build-harness layer. Jurisdiction source packages such as `ledger-us` and `chronicle-uk` emit records into that shared contract. @@ -16,8 +16,8 @@ validation, and build-harness layer. Jurisdiction source packages such as | Layer | Owns | Does not own | |-------|------|--------------| -| Chronicle | Source artifacts, provenance, aggregate facts, constraints, target profiles | Raw microdata storage, source reconciliation, aging, imputation, active target selection | -| Microcosm Targets | Source selection, reconciliation, aging, imputation, active target sets | Source artifact storage and provenance | +| Chronicle | Source artifacts, provenance, aggregate facts, constraints | Selection and measurement contracts, raw microdata storage, source reconciliation, aging, imputation, active target selection | +| Microcosm Targets | Selection and measurement contracts, reconciliation, aging, imputation, active target sets | Source artifact storage and provenance | | Microcosm | Entity model, weights, calibration interfaces, calibrated output | Source ETL and source provenance | | Jurisdiction source packages | Source-specific parsers and specs that emit Chronicle records | Forked fact or constraint schemas | | Jurisdiction simulation packages | Model-specific adapters, variable mappings, target recipes | Source facts | @@ -98,11 +98,9 @@ policyengine_chronicle.aggregate_facts (published aggregate facts) | v -policyengine_chronicle.target_profiles - | - v Microcosm Targets - (selected, reconciled, + (consumer-owned contracts; + selected, reconciled, aged active target sets) | v diff --git a/docs/chronicle-governance.md b/docs/chronicle-governance.md index fb4b064f..f60969fc 100644 --- a/docs/chronicle-governance.md +++ b/docs/chronicle-governance.md @@ -28,11 +28,6 @@ Chronicle may: needs disambiguation - normalize representation, such as units, scales, dates, geography IDs, and same-source total/share arithmetic when the publisher defines that relation -- declare target profiles that select source-backed facts and measurement - contracts without target values -- enforce the period contract at resolution: consuming a fact at a period - other than its reference period requires the consumer's explicit - `PeriodAlignmentDeclaration`, which Chronicle records and passes through Chronicle must not: @@ -40,10 +35,10 @@ Chronicle must not: - age facts to a build year - store PolicyEngine-computed values (aged, uprated, forecast, or reconciled levels) as facts or in any other store object -- compute an aligned value during resolution; it returns the published level - and the consumer's declaration only +- compute aligned values; it publishes the source period and value unchanged - impute missing values - store raw survey or administrative microdata +- own selection, measurement, period-alignment, or model-binding contracts - choose a support-aware active target subset - build solver-ready calibration targets - invent derived facts whose source is Chronicle itself @@ -55,7 +50,7 @@ The repository uses `.github/CODEOWNERS` to route all changes through review. Approved agent roles live in `.github/chronicle-agents.yml`. Contributions that -touch source packages, target profiles, or consumer contracts should name the +touch source packages or consumer contracts should name the agent role used and attach the required deterministic checks and judge verdicts. ## Judge Model @@ -65,7 +60,6 @@ reviewers judge the source-data boundary with that evidence. The required judge types are: - `ledger-source-fidelity` -- `ledger-target-profile` - `ledger-contract` - `ledger-boundary` diff --git a/docs/pe-uk-source-checklist.md b/docs/pe-uk-source-checklist.md index 5cc8783f..300d16b7 100644 --- a/docs/pe-uk-source-checklist.md +++ b/docs/pe-uk-source-checklist.md @@ -6,7 +6,8 @@ Tracker for the migration of every calibration-target family in 2026-07-26) onto Chronicle source packages. Scope issues: [#132](https://github.com/PolicyEngine/chronicle/issues/132) (wave 1, production 149-target surface), [#133](https://github.com/PolicyEngine/chronicle/issues/133) -(wave 2, remaining national registry + `uk_national` profile), +(wave 2, remaining national registry; the consumer contract moved in +microcosm#707), [#134](https://github.com/PolicyEngine/chronicle/issues/134) (wave 3, local geography). @@ -22,7 +23,7 @@ Status vocabulary: without Chronicle having computed it. - **no publisher column / no publisher measure** — the artifact is pinned and parsed, but the publisher simply does not print the quantity uk-data uses. - Resolving the affected target is a profile decision, not a sourcing gap. + Resolving the affected target is a Microcosm contract decision, not a sourcing gap. - **blocked** — the publisher source cannot be obtained (lost, never cited, or never published as a table), so the family stays out of Chronicle until a source surfaces. Each blocked row states what was tried. @@ -78,13 +79,13 @@ cannot be obtained. Nothing in uk-data is absent from this table. | SPP Review NI-relief bases ([`obr.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/obr.py)) | `obr/salary_sacrifice_{employee,employer}_ni_relief` (£1.2bn / £2.9bn at 2024) | **blocked — source lost** | — | — | uk-data's pinned asset (`media/67ce0e7c…/2025_SPP_Review.pdf`) returns `status: not found` and has **no Wayback snapshot** — the document is currently unrecoverable, and searches do not surface a re-hosted copy. The ported [`hmrc-salary-sacrifice-relief-2024`](https://github.com/PolicyEngine/chronicle/pull/141) package carries the successor TY2024-25 relief estimates (employee £1.0bn / employer £3.4bn) as an anchor. If anyone locates the SPP Review's new home, it becomes a small `pdf_text_numbers` package. | | Constituency age ([`local_age.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_age.py)) | 8 `age/{band}` × 650 constituencies | **ported** | `ons-pcon24-population-by-age-2024` (10,925 facts, 575 E&W seats), `nrs-pcon24-population-by-age-2024` (5,244, 57 Scottish seats), `nisra-pcon24-population-by-age-2024` (360, 18 NI seats) | wave-3 branch | **Replaced, and re-based to 2024 boundaries.** uk-data's `age.csv` is a 2020 [House of Commons Library](https://commonslibrary.parliament.uk/constituency-statistics-population-by-age/) extract on **2010** constituencies, with constituencies missing from the source filled from country mean age profiles and NI derived from SNPP via `demographics.csv` — all **excluded (computed)**, along with the 2010→2024 mapping matrix and the `×0.9` national consistency scaling. Chronicle pins each nation's own publisher at PCON24: ONS mid-2024 via Nomis `NM_2014_1` (quinary bands), NRS special area tables (single year), NISRA PxStat `MYE01T013` (five year bands). All 650 seats resolve under all eight ten-year `record_set_spec_id` groups. | | LA age ([`local_age.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_age.py)) | 8 `age/{band}` × 360 local authorities | **ported** | `ons-lad-population-by-age-2024` (7,220 facts, 361 authorities) | wave-3 branch | **Same publisher, pinned.** uk-data reads a committed export of Nomis [`pestsyoala`](https://www.nomisweb.co.uk/datasets/pestsyoala); Chronicle pins the API response for `NM_31_1` at mid-2024 — the one UK-wide dataset at this grain, carrying the 11 NI local government districts. 361 authorities against uk-data's 360; delta enumerated with the parity fixture. | -| Constituency / LA income ([`local_income.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_income.py)) | `hmrc/{employment,self_employment}_income/{count,amount}` at both levels | **ported; amounts have no publisher column** | `hmrc-spi-income-by-area-2023-24` (19,816 facts: 650 constituencies, 360 local authorities, 27 counties, 9 regions, 3 countries × 19 measures) | wave-3 branch | **Same publication, from bytes.** Tables 3.15 and 3.14 of the TY2023-24 [collated ODS](https://assets.publishing.service.gov.uk/media/69f1f17cc42061e837e3ac3b/Collated_Tables_3_12_to_3_15a_2324.ods), on PCON24 boundaries. Counts are published in thousands and the total-tax amount in £ millions; both carry `value_scale`. **There is no income-amount column** — the publisher gives count, mean and median per income type and an amount only for total tax — so the profile's two `*_amount` targets have no publisher fact and uk-data's count × mean is **excluded (computed)**. `[Not available]` cells emit no fact; Isles of Scilly is suppressed in every column of 3.14, so it carries none. The 9 regions and 3 countries appear in both worksheets cell for cell and are ported once, from 3.15. | +| Constituency / LA income ([`local_income.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_income.py)) | `hmrc/{employment,self_employment}_income/{count,amount}` at both levels | **ported; amounts have no publisher column** | `hmrc-spi-income-by-area-2023-24` (19,816 facts: 650 constituencies, 360 local authorities, 27 counties, 9 regions, 3 countries × 19 measures) | wave-3 branch | **Same publication, from bytes.** Tables 3.15 and 3.14 of the TY2023-24 [collated ODS](https://assets.publishing.service.gov.uk/media/69f1f17cc42061e837e3ac3b/Collated_Tables_3_12_to_3_15a_2324.ods), on PCON24 boundaries. Counts are published in thousands and the total-tax amount in £ millions; both carry `value_scale`. **There is no income-amount column** — the publisher gives count, mean and median per income type and an amount only for total tax — so the Microcosm contract's two `*_amount` targets have no publisher fact and uk-data's count × mean is **excluded (computed)**. `[Not available]` cells emit no fact; Isles of Scilly is suppressed in every column of 3.14, so it carries none. The 9 regions and 3 countries appear in both worksheets cell for cell and are ported once, from 3.15. | | Constituency / LA UC ([`local_uc.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_uc.py)) | `uc_households` at both levels, `uc_hh_{n}_children` at constituency | **ported (GB)** | `dwp-uc-households-by-constituency-may-2025` (632), `dwp-uc-households-by-local-authority-may-2025` (350), `dwp-uc-households-by-constituency-children-may-2025` (5,056) | wave-3 branch | **Same cube, queried directly.** Stat-Xplore carries a first-class PCON24 field, so the constituency cut is on 2024 boundaries. uk-data's by-children split applies November-2023 country shares to 2025 GB totals — **excluded (computed)**; the publisher tabulates constituency × children directly. **Northern Ireland is a signed exclusion**: DWP's UC statistics are GB, DfC publishes NI separately. Two publisher facts recorded but not ported because they are not geography-keyed: the "Unknown" area row (7,054 households), and the cross-cube marginal difference (571 of 632 constituencies differ between the totals and children cubes, median 4 and at most 15 households — DWP adjusts cells at source). | -| LA ONS income ([`local_la_extras.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_la_extras.py)) | `ons/equiv_net_income_{bhc,ahc}`, `ons/equiv_housing_costs` | **ported at MSOA; housing costs have no publisher measure** | `ons-small-area-income-msoa-fye2023` (29,056 facts: 7,264 MSOAs × 4 measures) | wave-3 branch | **Re-pinned to the current edition.** [FYE2023 estimates](https://www.ons.gov.uk/peoplepopulationandcommunity/personalandhouseholdfinances/incomeandwealth/bulletins/smallareamodelbasedincomeestimates/financialyearending2023) (released 2025-12-10) against uk-data's FYE2020 file. Published at **MSOA only** — no LA-grain table exists — so facts land at `geography_level: msoa` and the profile's three `ons_income` targets move there. `provenance_class: model_output` (small-area model over the FRS; populace's `frs_model_based_target_circularity` adjudication applies). **`ons.equiv_housing_costs` has no publisher fact at all** — there is no housing-costs measure; uk-data forms one as BHC−AHC, **excluded (computed)**, as are its unweighted mean of MSOA means, household-count multiplication and uprating factors. | +| LA ONS income ([`local_la_extras.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_la_extras.py)) | `ons/equiv_net_income_{bhc,ahc}`, `ons/equiv_housing_costs` | **ported at MSOA; housing costs have no publisher measure** | `ons-small-area-income-msoa-fye2023` (29,056 facts: 7,264 MSOAs × 4 measures) | wave-3 branch | **Re-pinned to the current edition.** [FYE2023 estimates](https://www.ons.gov.uk/peoplepopulationandcommunity/personalandhouseholdfinances/incomeandwealth/bulletins/smallareamodelbasedincomeestimates/financialyearending2023) (released 2025-12-10) against uk-data's FYE2020 file. Published at **MSOA only** — no LA-grain table exists — so facts land at `geography_level: msoa` and the Microcosm contract's three `ons_income` targets move there. `provenance_class: model_output` (small-area model over the FRS; populace's `frs_model_based_target_circularity` adjudication applies). **`ons.equiv_housing_costs` has no publisher fact at all** — there is no housing-costs measure; uk-data forms one as BHC−AHC, **excluded (computed)**, as are its unweighted mean of MSOA means, household-count multiplication and uprating factors. | | LA tenure ([`local_la_extras.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_la_extras.py), [`ons_tenure.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/ons_tenure.py)) | `tenure/{owned_outright,owned_mortgage,private_rent,social_rent}` | **ported, UK-wide** | `ons-census2021-ts054-tenure-lad` (2,862), `nrs-census2022-uv404-tenure-council-area` (288), `nisra-census2021-tenure-lgd` (121), plus `ons-subnational-dwellings-by-tenure-2024` (4,425) as a separate dwelling series | wave-3 branch | **Concept-correct and UK-wide.** The targets count households, so all three census legs are ported: [TS054](https://www.nomisweb.co.uk/datasets/c2021ts054) (318 E&W districts), NRS UV404 (32 Scottish councils, via the UK Data Service export), NISRA HH_TENURE (11 NI districts, full eleven-category classification, not the four-category roll-up). Classifications differ by leg, so matching concepts share a publisher-neutral `record_set_spec_id` bucket and unmatched ones (Scotland's Shared Equity) keep their own. uk-data instead applies [SPREE](https://www.ons.gov.uk/peoplepopulationandcommunity/housing/datasets/subnationaldwellingstockbytenureestimates) percentages (England, **dwellings**) to census household counts — that product is **excluded (computed)**; SPREE itself is ported as `entity: dwelling` with its own spec ids so it can never resolve into a household target. **NRS applies cell-level disclosure control**: in 26 of 32 councils the details differ from the published total by −6 to +10 households; values are carried as published, unreconciled. | | LA private rent ([`local_la_extras.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/local_la_extras.py)) | `rent/private_rent` | **ported** | `ons-pipr-rents-by-area-june-2026` (348 facts: 316 E&W districts, 9 regions, countries and UK/GB aggregates, 18 Scottish BRMAs) | wave-3 branch | **No longer at risk — solved by a new lane.** The operational blocker recorded on #133 was the monthly workbook's ~2M-cell used-range parse. This is the first consumer of `xlsx_table_full_rows`: the sheet is preserved as source rows and `selected_rows` restrict cell emission, so the 348 facts cost 13,960 cells. uk-data reads the discontinued PRMS median-rent file and multiplies by 12 — **excluded (computed)**. Northern Ireland publishes `[x]` at June 2026 (its series lags; the Dec-2025 NI level is ported in the wave-2 bulletin package), and City of London and Isles of Scilly are absent from the publisher's table — all enumerated. | | LA council tax ([`la_council_tax.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/targets/sources/la_council_tax.py)) | `voa/council_tax/{code}/{band}`, `ons/council_tax_band_d/{code}`, `housing/council_tax_net/{code}` | **ported (all three publisher inputs)** | `voa-council-tax-stock-by-lad-2025` (2,881), `mhclg-council-tax-levels-england-2026-27` (2,368), `scotgov-band-d-equivalents-2025` (1,419), `scotgov-band-d-council-tax-rates-2026-27` (1,023), `welshgov-council-tax-levels-2026-27` (198) | wave-3 branch | **Band counts** from [CTSOP1.1](https://www.gov.uk/government/statistics/council-tax-stock-of-properties-2025) LAUA rows (E&W; the REGL/NATL rows would collide with wave 2's CTSOP2.0 package, and 299 cells are `..`/`-` so emit no facts). **Band D** per authority from [MHCLG Table 10](https://www.gov.uk/government/statistics/council-tax-levels-set-by-local-authorities-in-england-2026-to-2027) (England, with the taxbase in the same table), [Welsh Government Table 1](https://www.gov.wales/council-tax-levels-april-2026-march-2027-html) and [ScotGov CTAS 2026](https://www.gov.scot/publications/council-tax-datasets/). **Net council tax**: Wales publishes its council tax income directly (WG Table 3, ported); England's is **excluded (computed)** as taxbase × Band D, and both inputs are now ported, so a consumer can re-declare it. Scottish band counts per council area are **not** ported — uk-data lacks them too (VOA is E&W only) and wave 2 took only the Scotland total from CTAXBASE; the per-council rows sit in that already-pinned artifact and are a cheap follow-up beyond parity. gov.wales and gov.scot 403 a plain fetch; both artifacts were retrieved with browser headers. | -| Per-area household counts (populace `uk_local_target_census`) | `households` at both levels | **ported, UK-wide at both levels** | `ons-census2021-ts041-households-pcon24` (575), `ons-census2021-ts041-households-lad` (318), `nrs-census2022-households-ukpc24` (57), `nisra-census2021-households-pcon24` (18), `nisra-census2021-households-lgd` (11) | wave-3 branch | **Published grain, three census legs.** [TS041](https://www.nomisweb.co.uk/datasets/c2021ts041) for England and Wales at both levels, NRS UV404's all-occupied-households row at UK Parliamentary Constituency 2024 for Scotland, NISRA's HOUSEHOLD flexible table for Northern Ireland. 575 + 57 + 18 = 650 constituencies; 318 + 11 here plus the 32 Scottish councils already in `nrs-census2022-uv404-tenure-council-area` = 361 local authorities — a second Scottish council-area package would collide on `fact_key`. The three legs share one `record_set_spec_id` per level, so one profile target spans them despite three census days (E&W 2021-03-21, Scotland 2022-03-20, NI 2021-03-21). Chronicle does not re-sum output areas; populace's sha-pinned OA ladder keeps its own path until consumption switches. The profile declares no `households` target today — added in the profile PR. | +| Per-area household counts (populace `uk_local_target_census`) | `households` at both levels | **ported, UK-wide at both levels** | `ons-census2021-ts041-households-pcon24` (575), `ons-census2021-ts041-households-lad` (318), `nrs-census2022-households-ukpc24` (57), `nisra-census2021-households-pcon24` (18), `nisra-census2021-households-lgd` (11) | wave-3 branch | **Published grain, three census legs.** [TS041](https://www.nomisweb.co.uk/datasets/c2021ts041) for England and Wales at both levels, NRS UV404's all-occupied-households row at UK Parliamentary Constituency 2024 for Scotland, NISRA's HOUSEHOLD flexible table for Northern Ireland. 575 + 57 + 18 = 650 constituencies; 318 + 11 here plus the 32 Scottish councils already in `nrs-census2022-uv404-tenure-council-area` = 361 local authorities — a second Scottish council-area package would collide on `fact_key`. The three legs share one `record_set_spec_id` per level, so one Microcosm contract target spans them despite three census days (E&W 2021-03-21, Scotland 2022-03-20, NI 2021-03-21). Chronicle does not re-sum output areas; Microcosm's sha-pinned OA ladder keeps its own path until consumption switches. | | Constituency ASHE earnings ([`create_employment_incomes.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/datasets/local_areas/constituencies/targets/create_employment_incomes.py)) | `employment_income.csv` band distribution | **excluded (computed)** | — | — | The committed NOMIS ASHE workbook feeds an interpolated band distribution (`interp1d` + `quad` over earnings percentiles) that `loss.py` never reads — it is not a target input. The publisher percentiles could be ported if a target ever needs them. | | Devolved constituency rent anchors ([`devolved_housing.py`](https://github.com/PolicyEngine/policyengine-uk-data/blob/ebf733c/policyengine_uk_data/datasets/local_areas/constituencies/devolved_housing.py)) | Wales/Scotland `private_renter_households`, `annual_private_rent` at constituency | **excluded (computed); base facts ported** | `ons-pipr-rents-by-area-june-2026`, `ons-census2021-ts054-tenure-lad`, `nrs-census2022-uv404-tenure-council-area` | wave-3 branch | **Re-based, not blocked.** uk-data hardcodes country totals (200,700 households / £795pm Wales; 357,706 / £999pm Scotland) with no citation and allocates them across 2010 constituencies by population share — both the values and the allocation are **excluded (computed)**. The publisher facts that let a consumer rebuild them are now ported: PIPR carries an average private rent for Wales (`W92000004`) and Scotland (`S92000003`), and the census tenure legs carry private-rented household counts per local authority. An earlier revision of this checklist called this family blocked; that is no longer true. | @@ -144,11 +145,12 @@ pin (derived from the target id) is the nation while the OBR rows are stamped `K02000001`. Pointing those ids at these packages, or re-pinning them, is a microcosm-side decision recorded in the PR that added these packages. -## Remaining — not yet started +## Consumer contracts -| family | targets | plan | -|---|---|---| -| `uk_national` target profile | — | Own `ledger-target-profile-author`-lane PR once the packages it selects exist (#133). Declares `allow_source_projection` on the OBR forecast lines and the SLC borrower-forecast series (#154 semantics). | +The UK national, local-geography, and firms selection contracts now live in +Microcosm. The national contract moved in microcosm#707; local geography and +firms moved in microcosm#795. Chronicle publishes only the facts those contracts +select. ## Legacy ETL cleanup diff --git a/docs/target-construction-harness-plan.md b/docs/target-construction-harness-plan.md index 4a65a6ed..65514f1f 100644 --- a/docs/target-construction-harness-plan.md +++ b/docs/target-construction-harness-plan.md @@ -689,26 +689,22 @@ Acceptance: - Chronicle also exposes source records for non-targeted Table 1.4 columns/rows. - Source records have concept/statistic/universe metadata and no simulator model variable IDs. -### Phase 3: Chronicle Target Profile Handoff +### Phase 3: Microcosm Target Contract Handoff Tasks: -- Declare target profiles that select source-backed facts and define - model-measurement contracts. -- Store stable selectors, source-record IDs, source periods, units, geography, - value definitions, and profile metadata. -- Add validation that target profiles contain no target values, source - reconciliation, aging, active support decisions, or simulator execution - logic. +- Publish stable source-record IDs, periods, units, geography, and value + definitions in Chronicle facts. +- Declare selectors and model-measurement contracts in Microcosm. +- Validate consumer contracts against a hash-pinned Chronicle facts artifact. Acceptance: -- Downstream systems can resolve every profile row back to Chronicle source records - and source artifacts. -- Profile validation fails closed when a row includes active target values, - model-runtime code, or unsupported operations. -- Microcosm can consume the profile as a contract, but owns all active target - values, aging, model-measure compilation, scoring, and differential tests. +- Microcosm can resolve every contract row back to Chronicle source records and + source artifacts. +- Chronicle remains unaware of model variables and active calibration decisions. +- Microcosm owns target contracts, active values, aging, model-measure + compilation, scoring, and differential tests. ### Phase 4: Expand SOI National Family @@ -728,12 +724,10 @@ Tasks: - Add CI job for source manifest smoke tests. - Add selector/source-record spec validation tests. -- Add target profile validation tests. - Surface Chronicle-owned results in the source observatory: - source coverage - parsed-cell coverage - source-record coverage - - target profile selector coverage - known source coverage gaps Acceptance: @@ -741,8 +735,8 @@ Acceptance: - A developer can see whether a source-package version preserves more official statistics release content, regresses source fidelity, or intentionally omits source rows. -- CI catches accidental source omission, selector breakage, target profile - schema violations, and source coverage regressions. +- CI catches accidental source omission, selector breakage, and source coverage + regressions. Microcosm CI covers its target-contract schema and fact resolution. ### Phase 8: State And Local Expansion @@ -761,7 +755,7 @@ Acceptance: ## Open Design Answers 1. Source-record specs should live in Chronicle as YAML source of truth plus compiled DB rows. -2. Active target specs should start in Microcosm. Split later only when multiple adapters/profiles need independent versioning. +2. Selection contracts and active target specs live in Microcosm. 3. Count targets should be sums of count-valued measures, never a separate count aggregation. 4. PE parity should be required enough to classify differences, not enough to force exact reproduction of PE omissions or legacy shortcuts. 5. Chronicle should preserve every parsed source cell, then layer semantic source records on top. Do not automatically turn every cell into a semantic record. @@ -794,8 +788,6 @@ First-slice acceptance criteria: - Parsed cells preserve raw/displayed/formula/missingness distinctions. - Selector specs cover all PE-used cells and adjacent omitted source columns. - Source records are simulator-neutral and contain concept/statistic/universe metadata. -- Target profiles select source records without storing target values, runtime - code, active support decisions, or model compiler configuration. -- Downstream Microcosm recipes can reference the source records and target - profiles, but their values, model measures, fixtures, and differential checks - remain outside Chronicle. +- Downstream Microcosm contracts select Chronicle source records without moving + model measures, runtime code, active support decisions, or compiler + configuration into Chronicle. diff --git a/policyengine_chronicle/__init__.py b/policyengine_chronicle/__init__.py index 87d2e00b..06399e5c 100644 --- a/policyengine_chronicle/__init__.py +++ b/policyengine_chronicle/__init__.py @@ -28,13 +28,8 @@ ) from policyengine_chronicle.consumer import ( ConsumerArtifact, - PeriodAlignmentDeclaration, - PeriodContractError, - ResolutionReport, - ResolvedTarget, build_consumer_artifact, load_consumer_artifact, - resolve_profile_targets, ) __all__ = [ @@ -48,12 +43,8 @@ "EntityDimension", "GeographyDimension", "Measure", - "PeriodAlignmentDeclaration", - "PeriodContractError", "PeriodCoverage", "PeriodDimension", - "ResolutionReport", - "ResolvedTarget", "SourceProvenance", "SourceRecordLayout", "ValidationIssue", @@ -63,7 +54,6 @@ "build_fact_key", "build_label", "load_consumer_artifact", - "resolve_profile_targets", "validate_fact", "validate_facts", ] diff --git a/policyengine_chronicle/consumer.py b/policyengine_chronicle/consumer.py index 7856b836..8894081f 100644 --- a/policyengine_chronicle/consumer.py +++ b/policyengine_chronicle/consumer.py @@ -1,18 +1,8 @@ -"""Consumer artifacts and period-contract resolution for Chronicle profiles. +"""Build and verify facts-only Chronicle consumer artifacts. -This module is the supported way for downstream builds (Microcosm, Thesis, -future rule engines) to consume Chronicle facts: a versioned on-disk artifact -plus a resolution API that selects profile targets from consumer-contract -fact rows. - -Resolution enforces the period contract. A fact's value refers to the fact's -own reference period; consuming it at any other period silently is the -failure mode that produced PolicyEngine/microcosm#212 (SOI tax-year levels -applied un-aged to a later build year). Resolving at a different period -therefore hard-fails unless the consumer passes an explicit, named -:class:`PeriodAlignmentDeclaration`. Chronicle records the declaration in the -resolved rows; it never computes the aligned value. Aging, uprating, and -reconciliation stay in the consumer. +Chronicle publishes source-backed fact rows and the hashes needed to verify +them. Selection, measurement, period-alignment, and model-binding contracts +belong to consumers such as Microcosm. """ from __future__ import annotations @@ -21,9 +11,8 @@ import json import math import shutil -from collections.abc import Mapping, Sequence -from dataclasses import asdict, dataclass, field -from importlib.resources import files as _resource_files +from collections.abc import Mapping +from dataclasses import asdict, dataclass from pathlib import Path from typing import Any @@ -37,824 +26,44 @@ CONSUMER_FACT_SCHEMA_SHA256, validate_consumer_fact_row, ) -from policyengine_chronicle.target_profiles import ( - TargetProfile, - TargetProfileTarget, - load_target_profile, - target_profile_from_mapping, -) -from policyengine_chronicle.target_profiles.model import ( - FORBIDDEN_RUNTIME_KEYS, - FORBIDDEN_VALUE_KEYS, -) CONSUMER_ARTIFACT_SCHEMA_VERSION = "policyengine_ledger.consumer_artifact.v1" -RESOLVED_TARGET_SCHEMA_VERSION = "policyengine_ledger.resolved_target.v1" -SUPPORTED_BASE_PERIOD_POLICIES = {"latest_not_after_build_base_period"} - -_SELECTOR_KEYS = { - "source_name", - "source_table", - "source_measure_id", - "source_concept", - "concept", - "record_set_id", - "record_set_spec_id", - "groupby_dimension", - "dimensions", - "domain", - "entity", - "assertion", - "provenance_class", -} - - -@dataclass(frozen=True) -class PeriodAlignmentDeclaration: - """A consumer's explicit declaration of how it will align a fact period. - - The declaration names the consumer-side transformation (for example a - versioned growth-factor aging model) that will be applied to the fact - value outside Chronicle. It carries no values: parameters reference factor - series or model configuration, never target amounts. - """ - - model_id: str - model_version: str - parameters: Mapping[str, str | int | float | bool] = field(default_factory=dict) - notes: str | None = None - - def __post_init__(self) -> None: - if not self.model_id or not str(self.model_id).strip(): - raise ValueError("Period alignment declarations need a model_id.") - if not self.model_version or not str(self.model_version).strip(): - raise ValueError("Period alignment declarations need a model_version.") - forbidden = FORBIDDEN_VALUE_KEYS | FORBIDDEN_RUNTIME_KEYS - present = sorted(key for key in forbidden if key in self.parameters) - if present: - raise ValueError( - f"Period alignment parameters must not declare {present}; " - "declarations reference models and factor series, never " - "target values or runtime hooks." - ) - for key, value in self.parameters.items(): - if isinstance(value, list | dict | tuple | set): - raise ValueError( - f"Period alignment parameter {key!r} has a non-scalar value." - ) - - def to_dict(self) -> dict[str, Any]: - """Return a JSON-serializable declaration.""" - payload: dict[str, Any] = { - "model_id": self.model_id, - "model_version": self.model_version, - } - if self.parameters: - payload["parameters"] = dict(sorted(self.parameters.items())) - if self.notes: - payload["notes"] = self.notes - return payload - - -@dataclass(frozen=True) -class PeriodContractViolation: - """One target resolved at a period its facts do not cover.""" - - profile_id: str - target_id: str - fact_period: dict[str, Any] - requested_period: dict[str, Any] - message: str - - def to_dict(self) -> dict[str, Any]: - """Return a JSON-serializable violation.""" - return asdict(self) - - -class PeriodContractError(ValueError): - """Raised when facts would be consumed at the wrong period silently.""" - - def __init__(self, violations: Sequence[PeriodContractViolation]) -> None: - self.violations = tuple(violations) - details = "; ".join( - f"{violation.target_id}: fact period " - f"{violation.fact_period['type']}:{violation.fact_period['value']} " - f"!= requested " - f"{violation.requested_period['type']}:" - f"{violation.requested_period['value']}" - for violation in self.violations - ) - super().__init__( - "Period contract violation: facts cannot be consumed at a period " - "other than their reference period without an explicit " - "PeriodAlignmentDeclaration. Pass alignments={target_id: " - "PeriodAlignmentDeclaration(model_id=..., model_version=...)} " - f"for: {details}" - ) - - -@dataclass(frozen=True) -class ResolutionIssue: - """One non-period resolution problem.""" - - code: str - message: str - profile_id: str - target_id: str | None = None - severity: str = "error" - - def to_dict(self) -> dict[str, Any]: - """Return a JSON-serializable issue.""" - return asdict(self) - - -@dataclass(frozen=True) -class ResolvedTarget: - """One profile target row resolved against a Chronicle fact row. - - ``value`` is always the published fact value at ``fact_period``. When - ``basis`` is ``declared_alignment`` the consumer has declared it will - transform the value to ``requested_period`` with ``alignment``; Chronicle - passes the declaration through untouched. - """ - - profile_id: str - target_id: str - basis: str - value: Any - value_type: str - unit: str | None - assertion: str - provenance_class: str - fact_period: dict[str, Any] - requested_period: dict[str, Any] - aggregate_fact_key: str - semantic_fact_key: str - geography: dict[str, Any] - entity: dict[str, Any] - dimensions: dict[str, Any] - universe_constraints: dict[str, Any] - source: dict[str, Any] - lineage: dict[str, Any] - alignment: dict[str, Any] | None = None - label: str | None = None - survey_instrument: str | None = None - - def to_dict(self) -> dict[str, Any]: - """Return a JSON-serializable resolved row.""" - payload = { - "schema_version": RESOLVED_TARGET_SCHEMA_VERSION, - **asdict(self), - } - if payload.get("alignment") is None: - payload.pop("alignment", None) - if payload.get("label") is None: - payload.pop("label", None) - if payload.get("survey_instrument") is None: - payload.pop("survey_instrument", None) - return payload - - -@dataclass(frozen=True) -class ResolutionReport: - """Resolved targets plus contract diagnostics for one profile.""" - - profile_id: str - requested_period: dict[str, Any] - resolved: tuple[ResolvedTarget, ...] - violations: tuple[PeriodContractViolation, ...] = () - issues: tuple[ResolutionIssue, ...] = () - - @property - def valid(self) -> bool: - """Whether resolution produced no violations or blocking issues.""" - return not self.violations and not any( - issue.severity == "error" for issue in self.issues - ) - - def to_dict(self) -> dict[str, Any]: - """Return a JSON-serializable report.""" - return { - "profile_id": self.profile_id, - "requested_period": self.requested_period, - "valid": self.valid, - "resolved": [row.to_dict() for row in self.resolved], - "violations": [violation.to_dict() for violation in self.violations], - "issues": [issue.to_dict() for issue in self.issues], - } - - -def resolve_profile_targets( - profile: TargetProfile, - rows: Sequence[Mapping[str, Any]], - requested_period: Mapping[str, Any], - *, - alignments: ( - Mapping[str, PeriodAlignmentDeclaration] | PeriodAlignmentDeclaration | None - ) = None, - geography_level: str = "country", - strict: bool = True, -) -> ResolutionReport: - """Resolve profile targets from consumer-contract fact rows. - - ``alignments`` maps ``target_id`` (or ``"*"`` for all targets) to the - consumer's :class:`PeriodAlignmentDeclaration`. ``strict`` raises - :class:`PeriodContractError` on period violations and ``ValueError`` on - blocking coverage issues instead of returning an invalid report. - """ - requested = _normalize_period(requested_period) - rows = _normalize_assertion_rows(rows) - alignment_map = _normalize_alignments(alignments) - if profile.base_period_policy not in SUPPORTED_BASE_PERIOD_POLICIES: - raise ValueError( - f"Unsupported base_period_policy {profile.base_period_policy!r}; " - f"supported: {sorted(SUPPORTED_BASE_PERIOD_POLICIES)}." - ) - - resolved: list[ResolvedTarget] = [] - violations: list[PeriodContractViolation] = [] - issues: list[ResolutionIssue] = [] - - for target in profile.targets_for_geography(geography_level): - candidates, selector_issues = _select_rows( - profile.profile_id, - target, - rows, - geography_level=geography_level, - ) - issues.extend(selector_issues) - if not candidates: - issues.append( - ResolutionIssue( - code="no_matching_facts", - message=( - f"Target {target.target_id!r} matched no consumer fact " - "rows; the profile selector and fact coverage disagree." - ), - profile_id=profile.profile_id, - target_id=target.target_id, - ) - ) - continue - - assertion_policy = target.assertion_policy or profile.default_assertion_policy - if "assertion" in target.chronicle_selector: - # An explicit assertion selector is maximal author intent; the - # policy governs only targets that do not select on assertion. - assertion_policy = "allow_source_projection" - observed = [row for row in candidates if row["assertion"] == "observation"] - if assertion_policy == "observed_only": - if not observed: - issues.append( - ResolutionIssue( - code="only_projection_facts", - message=( - f"Target {target.target_id!r} matched only " - "source_projection facts and its assertion_policy " - "is 'observed_only'; declare 'prefer_observed' or " - "'allow_source_projection' to resolve projections " - "deliberately." - ), - profile_id=profile.profile_id, - target_id=target.target_id, - severity="error", - ) - ) - continue - candidates = observed - elif assertion_policy == "prefer_observed" and observed: - # Per series, not per family: a series with any observed fact - # resolves only from observations, while a projection-only - # series keeps its projections instead of being starved by a - # neighbouring series' observation. - observed_series = {_series_key(row) for row in observed} - candidates = [ - row - for row in candidates - if row["assertion"] == "observation" - or _series_key(row) not in observed_series - ] - - chosen_period, period_issue = _choose_period( - profile.profile_id, - target, - candidates, - requested, - ) - if period_issue is not None: - issues.append(period_issue) - continue - - alignment = alignment_map.get(target.target_id, alignment_map.get("*")) - period_matches = chosen_period == requested - if period_matches: - basis = "fact" - if target.target_id in alignment_map: - issues.append( - ResolutionIssue( - code="unused_alignment", - message=( - f"Target {target.target_id!r} resolves at its fact " - "period; the declared alignment was not needed." - ), - profile_id=profile.profile_id, - target_id=target.target_id, - severity="warning", - ) - ) - alignment = None - elif alignment is None: - violations.append( - PeriodContractViolation( - profile_id=profile.profile_id, - target_id=target.target_id, - fact_period=chosen_period, - requested_period=requested, - message=( - "Fact period differs from the requested period and no " - "period alignment was declared." - ), - ) - ) - continue - else: - basis = "declared_alignment" - - rows_at_period = [ - row for row in candidates if dict(row["period"]) == chosen_period - ] - series_at_period: dict[str, list[Mapping[str, Any]]] = {} - for row in rows_at_period: - series_at_period.setdefault(_series_key(row), []).append(row) - drop_keys: set[str] = set() - for series_rows in series_at_period.values(): - if len({row["assertion"] for row in series_rows}) < 2: - continue - # An observation and a publisher projection collide within one - # series at the chosen period; emitting both would double-count - # it. The realized value wins the tie, loudly. Series that were - # never in a tie — a geography whose only fact is a projection — - # are untouched. - drop_keys.update( - row["aggregate_fact_key"] - for row in series_rows - if row["assertion"] != "observation" - ) - sample = series_rows[0] - geography = sample.get("geography", {}) - dimensions = sample.get("dimensions") or {} - where = f"geography {geography.get('level')}:{geography.get('id')}" - if dimensions: - where += f", dimensions {json.dumps(dimensions, sort_keys=True)}" - issues.append( - ResolutionIssue( - code="ambiguous_assertion_at_period", - message=( - f"Target {target.target_id!r} matched both an " - f"observation and a source_projection for one series " - f"({where}) at " - f"{chosen_period['type']}:{chosen_period['value']}; " - "resolved the observation. Select on assertion or " - "tighten dimensions/record_set_id to address the " - "overlap explicitly." - ), - profile_id=profile.profile_id, - target_id=target.target_id, - severity="warning", - ) - ) - if drop_keys: - rows_at_period = [ - row - for row in rows_at_period - if row["aggregate_fact_key"] not in drop_keys - ] - - # The flag is set inside the row loop deliberately: one - # resolved_from_projection warning per target, however many of its - # rows resolve from projections. - projection_resolved = False - for row in rows_at_period: - if row["assertion"] == "source_projection": - projection_resolved = True - resolved.append( - _resolved_target( - profile.profile_id, - target, - row, - basis=basis, - requested_period=requested, - alignment=alignment, - ) - ) - if projection_resolved: - issues.append( - ResolutionIssue( - code="resolved_from_projection", - message=( - f"Target {target.target_id!r} resolved from a " - "source_projection fact at " - f"{chosen_period['type']}:{chosen_period['value']} " - f"under assertion_policy {assertion_policy!r}." - ), - profile_id=profile.profile_id, - target_id=target.target_id, - severity="warning", - ) - ) - - report = ResolutionReport( - profile_id=profile.profile_id, - requested_period=requested, - resolved=tuple(resolved), - violations=tuple(violations), - issues=tuple(issues), - ) - if strict: - if report.violations: - raise PeriodContractError(report.violations) - blocking = [issue for issue in report.issues if issue.severity == "error"] - if blocking: - raise ValueError( - "Profile resolution failed: " - + "; ".join(issue.message for issue in blocking) - ) - return report - - -def _resolved_target( - profile_id: str, - target: TargetProfileTarget, - row: Mapping[str, Any], - *, - basis: str, - requested_period: dict[str, Any], - alignment: PeriodAlignmentDeclaration | None, -) -> ResolvedTarget: - observed_measure = row.get("observed_measure", {}) - return ResolvedTarget( - profile_id=profile_id, - target_id=target.target_id, - basis=basis, - value=row["value"], - value_type=row["value_type"], - unit=observed_measure.get("unit"), - assertion=row["assertion"], - provenance_class=row["provenance_class"], - fact_period=dict(row["period"]), - requested_period=requested_period, - aggregate_fact_key=row["aggregate_fact_key"], - semantic_fact_key=row["semantic_fact_key"], - geography=dict(row.get("geography", {})), - entity=dict(row.get("entity", {})), - dimensions=dict(row.get("dimensions", {})), - universe_constraints=dict(row.get("universe_constraints", {})), - source=dict(row.get("source", {})), - lineage=dict(row.get("lineage", {})), - alignment=alignment.to_dict() if alignment is not None else None, - label=row.get("label"), - survey_instrument=row.get("survey_instrument"), - ) - - -def _select_rows( - profile_id: str, - target: TargetProfileTarget, - rows: Sequence[Mapping[str, Any]], - *, - geography_level: str, -) -> tuple[list[Mapping[str, Any]], list[ResolutionIssue]]: - issues: list[ResolutionIssue] = [] - selector = dict(target.chronicle_selector) - unknown = sorted(set(selector) - _SELECTOR_KEYS) - if unknown: - issues.append( - ResolutionIssue( - code="unknown_selector_key", - message=( - f"Target {target.target_id!r} selector has unknown keys " - f"{unknown}; supported keys: {sorted(_SELECTOR_KEYS)}." - ), - profile_id=profile_id, - target_id=target.target_id, - ) - ) - return [], issues - geography_matched = [ - row for row in rows if row.get("geography", {}).get("level") == geography_level - ] - non_dimension_selector = { - key: value for key, value in selector.items() if key != "dimensions" - } - matched = [ - row - for row in geography_matched - if all( - _selector_matches(row, key, value) - for key, value in non_dimension_selector.items() - ) - ] - if "dimensions" not in selector: - return matched, issues - - dimension_selector = selector["dimensions"] - if isinstance(dimension_selector, Mapping): - if not dimension_selector: - issues.append( - ResolutionIssue( - code="empty_dimensions_selector", - message=( - f"Target {target.target_id!r} selector has an empty " - "'dimensions' mapping; dimension-value selectors must " - "name at least one dimension." - ), - profile_id=profile_id, - target_id=target.target_id, - ) - ) - return [], issues - if matched: - available = sorted( - {name for row in matched for name in dict(row.get("dimensions", {}))} - ) - unknown_dimensions = sorted(set(dimension_selector) - set(available)) - if unknown_dimensions: - issues.append( - ResolutionIssue( - code="unknown_dimension_selector", - message=( - f"Target {target.target_id!r} selector has unknown " - f"dimensions {unknown_dimensions}; available " - f"dimensions after other selectors: {available}." - ), - profile_id=profile_id, - target_id=target.target_id, - ) - ) - return [], issues - - return [ - row - for row in matched - if _selector_matches(row, "dimensions", dimension_selector) - ], issues - - -def _selector_matches(row: Mapping[str, Any], key: str, value: Any) -> bool: - if key == "dimensions" and isinstance(value, Mapping): - dimensions = dict(row.get("dimensions", {})) - return dimensions == dict(value) - actual = _selector_value(row, key) - if isinstance(actual, list): - # Dimension-identity selectors match order-insensitively on the exact - # set of dimension variable names the row carries. - return isinstance(value, list) and sorted(actual) == sorted(value) - return actual == value - - -def _selector_value(row: Mapping[str, Any], key: str) -> Any: - source = row.get("source", {}) - observed_measure = row.get("observed_measure", {}) - layout = row.get("layout", {}) - if key == "source_name": - return source.get("source_name") - if key == "source_table": - return source.get("source_table") - if key == "source_measure_id": - return observed_measure.get("source_measure_id") - if key == "source_concept": - return observed_measure.get("source_concept") - if key == "concept": - alignment = row.get("concept_alignment", {}) - return alignment.get("canonical_concept") or observed_measure.get( - "source_concept" - ) - if key == "record_set_id": - return layout.get("record_set_id") - if key == "record_set_spec_id": - return layout.get("record_set_spec_id") - if key == "groupby_dimension": - return layout.get("groupby_dimension") - if key == "dimensions": - return sorted(row.get("dimensions", {})) - if key == "domain": - return row.get("universe_constraints", {}).get("domain") - if key == "entity": - return row.get("entity", {}).get("name") - if key == "assertion": - return row.get("assertion") - if key == "provenance_class": - return row.get("provenance_class") - raise KeyError(key) - - -def _choose_period( - profile_id: str, - target: TargetProfileTarget, - candidates: Sequence[Mapping[str, Any]], - requested: dict[str, Any], -) -> tuple[dict[str, Any], None] | tuple[None, ResolutionIssue]: - periods = {(row["period"]["type"], row["period"]["value"]) for row in candidates} - if (requested["type"], requested["value"]) in periods: - return requested, None - - same_type = { - value for period_type, value in periods if period_type == requested["type"] - } - eligible = [value for value in same_type if _not_after(value, requested["value"])] - if eligible: - return {"type": requested["type"], "value": _latest(eligible)}, None - - period_types = sorted({period_type for period_type, _ in periods}) - if len(period_types) == 1: - values = {value for _, value in periods} - return {"type": period_types[0], "value": _latest(values)}, None - - return None, ResolutionIssue( - code="ambiguous_period_type", - message=( - f"Target {target.target_id!r} matched facts across period types " - f"{period_types} with no exact match for " - f"{requested['type']}:{requested['value']}; narrow the selector." - ), - profile_id=profile_id, - target_id=target.target_id, - ) - - -def _not_after(candidate: Any, requested: Any) -> bool: - if isinstance(candidate, int) and isinstance(requested, int): - return candidate <= requested - return str(candidate) <= str(requested) - - -def _latest(values) -> Any: - values = list(values) - if all(isinstance(value, int) for value in values): - return max(values) - return max(values, key=str) - - -def _series_key(row: Mapping[str, Any]) -> str: - """Identity of a co-resolving series, blind to source, period, assertion. - - Groups the rows a target resolves together so the per-series rules (the - prefer_observed filter, the assertion tie-break) never let one series' - observation starve a different series that only has a projection. - Source identity is deliberately excluded so one publisher's estimate and - another table's projection of the same series still collide. The - canonical concept is not carried on the row, so two different concepts - sharing every axis below within one selector match are a selector-hygiene - problem this key does not adjudicate. - """ - observed_measure = row.get("observed_measure", {}) - return json.dumps( - { - "geography": row.get("geography"), - "entity": row.get("entity"), - "aggregation": row.get("aggregation"), - "dimension_set_key": row.get("dimension_set_key"), - "universe_constraint_set_key": row.get("universe_constraint_set_key"), - "unit": observed_measure.get("unit"), - }, - sort_keys=True, - ) - - -def _normalize_assertion_rows( - rows: Sequence[Mapping[str, Any]], -) -> list[dict[str, Any]]: - """Police the assertion axis for rows that bypassed the file loader. - - ``_load_consumer_rows`` already defaults and validates ``assertion``; - rows handed to :func:`resolve_profile_targets` directly get the same - treatment here, so a typo such as ``assertion: projection`` fails loudly - instead of silently vanishing from every policy's candidate set. - """ - normalized: list[dict[str, Any]] = [] - for index, row in enumerate(rows): - mutable = dict(row) - assertion = mutable.setdefault("assertion", DEFAULT_ASSERTION) - if assertion not in ALLOWED_ASSERTIONS: - raise ValueError( - f"Consumer fact row {index} has unsupported assertion " - f"{assertion!r}; allowed: {sorted(ALLOWED_ASSERTIONS)}." - ) - normalized.append(mutable) - return normalized - - -def _normalize_period(period: Mapping[str, Any]) -> dict[str, Any]: - if not isinstance(period, Mapping): - raise ValueError("requested_period must be a mapping with type and value.") - period_type = period.get("type") - value = period.get("value") - if not isinstance(period_type, str) or not period_type: - raise ValueError("requested_period needs a non-empty period type.") - if value is None or (isinstance(value, str) and not value.strip()): - raise ValueError("requested_period needs a period value.") - return {"type": period_type, "value": value} - - -def _normalize_alignments( - alignments: ( - Mapping[str, PeriodAlignmentDeclaration] | PeriodAlignmentDeclaration | None - ), -) -> dict[str, PeriodAlignmentDeclaration]: - if alignments is None: - return {} - if isinstance(alignments, PeriodAlignmentDeclaration): - return {"*": alignments} - normalized: dict[str, PeriodAlignmentDeclaration] = {} - for target_id, declaration in alignments.items(): - if not isinstance(declaration, PeriodAlignmentDeclaration): - raise ValueError( - "alignments values must be PeriodAlignmentDeclaration objects." - ) - normalized[str(target_id)] = declaration - return normalized @dataclass(frozen=True) class ConsumerArtifact: - """A loaded Chronicle consumer artifact. - - ``profile_hash_semantics`` records, per profile id, which manifest hash - semantics the load accepted: ``exact`` for the byte-for-byte file hash, or - ``legacy_profile_hash`` for a pre-fix manifest whose profile hash omitted - the trailing newline. Tampered profile bytes never match either and fail - the load. - """ + """A verified facts-only Chronicle consumer artifact.""" path: Path manifest: dict[str, Any] rows: tuple[dict[str, Any], ...] - profiles: Mapping[str, TargetProfile] - profile_hash_semantics: Mapping[str, str] = field(default_factory=dict) - - def resolve( - self, - profile_id: str, - requested_period: Mapping[str, Any], - *, - alignments: ( - Mapping[str, PeriodAlignmentDeclaration] | PeriodAlignmentDeclaration | None - ) = None, - geography_level: str = "country", - strict: bool = True, - ) -> ResolutionReport: - """Resolve one embedded profile against the artifact's fact rows.""" - try: - profile = self.profiles[profile_id] - except KeyError: - raise KeyError( - f"Artifact has no profile {profile_id!r}; available: " - f"{sorted(self.profiles)}." - ) from None - return resolve_profile_targets( - profile, - self.rows, - requested_period, - alignments=alignments, - geography_level=geography_level, - strict=strict, - ) @dataclass(frozen=True) class ConsumerArtifactBuildReport: - """Build summary for one consumer artifact.""" + """Build summary for one facts-only consumer artifact.""" schema_version: str output_dir: str fact_row_count: int - profile_ids: tuple[str, ...] - coverage: dict[str, Any] def to_dict(self) -> dict[str, Any]: """Return a JSON-serializable report.""" - return { - "schema_version": self.schema_version, - "output_dir": self.output_dir, - "fact_row_count": self.fact_row_count, - "profile_ids": list(self.profile_ids), - "coverage": self.coverage, - } + return asdict(self) def build_consumer_artifact( output_dir: str | Path, *, facts_path: str | Path, - profile_ids: Sequence[str] = (), - profile_paths: Sequence[str | Path] = (), replace: bool = False, ) -> ConsumerArtifactBuildReport: - """Build a versioned consumer artifact from consumer facts and profiles. + """Build a reproducible facts-only artifact from consumer fact rows. ``facts_path`` is a ``consumer_facts.jsonl`` file or a bundle directory - containing one. The artifact is reproducible: no timestamps, canonical - JSON, and manifest hashes for the fact rows and every profile. + containing one. The artifact contains canonical fact rows and a manifest + that pins their schema and content hashes. Target contracts are packaged + by the consumer, not Chronicle. """ output_path = Path(output_dir) if output_path.exists(): @@ -866,29 +75,12 @@ def build_consumer_artifact( output_path.mkdir(parents=True) rows = _load_consumer_rows(_resolve_facts_path(facts_path), validate_schema=True) - profiles = _load_profiles(profile_ids, profile_paths) - facts_out = output_path / "consumer_facts.jsonl" with facts_out.open("w") as file: for row in rows: file.write(json.dumps(row, sort_keys=True)) file.write("\n") - profiles_dir = output_path / "profiles" - profiles_dir.mkdir() - profile_meta: dict[str, Any] = {} - for profile_id, payload in profiles.items(): - profile_json = json.dumps(payload, sort_keys=True, indent=2) - profile_bytes = (profile_json + "\n").encode("utf-8") - (profiles_dir / f"{profile_id}.json").write_bytes(profile_bytes) - profile_meta[profile_id] = { - "sha256": hashlib.sha256(profile_bytes).hexdigest(), - "target_count": len(payload["targets"]), - } - - coverage = _artifact_coverage(rows, profiles) - _write_json(output_path / "coverage.json", coverage) - manifest = { "schema_version": CONSUMER_ARTIFACT_SCHEMA_VERSION, "consumer_fact_schema_versions": sorted( @@ -897,7 +89,6 @@ def build_consumer_artifact( "consumer_fact_schema_sha256": CONSUMER_FACT_SCHEMA_SHA256, "fact_row_count": len(rows), "facts_sha256": _sha256_file(facts_out), - "profiles": profile_meta, } _write_json(output_path / "manifest.json", manifest) @@ -905,20 +96,11 @@ def build_consumer_artifact( schema_version=CONSUMER_ARTIFACT_SCHEMA_VERSION, output_dir=str(output_path), fact_row_count=len(rows), - profile_ids=tuple(sorted(profiles)), - coverage=coverage, ) def load_consumer_artifact(path: str | Path) -> ConsumerArtifact: - """Load a consumer artifact directory and verify its manifest hashes. - - Verification is fail-closed: the manifest's declared consumer-fact schema - (when present) must match the packaged schema, fact rows are re-hashed and - schema-validated, and every profile file is re-hashed against the manifest. - A manifest that predates the profile-hash fix may match through the - explicit ``legacy_profile_hash`` path, recorded on the returned artifact. - """ + """Load a facts-only consumer artifact and verify its manifest hashes.""" artifact_path = Path(path) manifest = json.loads((artifact_path / "manifest.json").read_text()) if manifest.get("schema_version") != CONSUMER_ARTIFACT_SCHEMA_VERSION: @@ -926,6 +108,11 @@ def load_consumer_artifact(path: str | Path) -> ConsumerArtifact: "Unsupported consumer artifact schema_version: " f"{manifest.get('schema_version')!r}." ) + if "profiles" in manifest: + raise ValueError( + "Consumer artifact manifests must not contain profiles; target profiles " + "are consumer-owned contracts and must be loaded by Microcosm." + ) manifest_schema_sha256 = manifest.get("consumer_fact_schema_sha256") if ( manifest_schema_sha256 is not None @@ -936,86 +123,25 @@ def load_consumer_artifact(path: str | Path) -> ConsumerArtifact: f"{manifest_schema_sha256!r}, which does not match the packaged " f"consumer-fact schema {CONSUMER_FACT_SCHEMA_SHA256!r}." ) + facts_file = artifact_path / "consumer_facts.jsonl" actual_sha256 = _sha256_file(facts_file) if actual_sha256 != manifest["facts_sha256"]: raise ValueError( - f"Consumer artifact fact rows do not match the manifest hash: " + "Consumer artifact fact rows do not match the manifest hash: " f"{actual_sha256} != {manifest['facts_sha256']}." ) rows = _load_consumer_rows(facts_file, validate_schema=True) declared_row_count = manifest.get("fact_row_count") if declared_row_count is not None and declared_row_count != len(rows): raise ValueError( - f"Consumer artifact manifest declares fact_row_count " + "Consumer artifact manifest declares fact_row_count " f"{declared_row_count} but the feed carries {len(rows)} rows." ) - manifest_predates_fix = "consumer_fact_schema_sha256" not in manifest - profiles: dict[str, TargetProfile] = {} - profile_hash_semantics: dict[str, str] = {} - for profile_id, profile_meta in manifest.get("profiles", {}).items(): - profile_file = artifact_path / "profiles" / f"{profile_id}.json" - profile_hash_semantics[profile_id] = _verify_profile_hash( - profile_id, - profile_file, - profile_meta, - manifest_predates_fix=manifest_predates_fix, - ) - payload = json.loads(profile_file.read_text()) - profile = target_profile_from_mapping(payload) - declared_targets = ( - profile_meta.get("target_count") - if isinstance(profile_meta, Mapping) - else None - ) - if declared_targets is not None and declared_targets != len(payload["targets"]): - raise ValueError( - f"Consumer artifact manifest declares target_count " - f"{declared_targets} for profile {profile_id!r} but the file " - f"carries {len(payload['targets'])} targets." - ) - profiles[profile_id] = profile return ConsumerArtifact( path=artifact_path, manifest=manifest, rows=tuple(rows), - profiles=profiles, - profile_hash_semantics=profile_hash_semantics, - ) - - -def _verify_profile_hash( - profile_id: str, - profile_file: Path, - profile_meta: Any, - *, - manifest_predates_fix: bool, -) -> str: - """Return the profile hash semantics matched, or raise on any mismatch. - - ``exact`` matches the byte-for-byte file hash written since the fix. - ``legacy_profile_hash`` matches a pre-fix manifest whose hash omitted the - trailing newline; it is accepted only when the manifest predates the fix - (no ``consumer_fact_schema_sha256``). Tampered bytes match neither. - """ - expected = profile_meta.get("sha256") if isinstance(profile_meta, Mapping) else None - if not expected: - raise ValueError( - f"Consumer artifact manifest is missing a sha256 for profile " - f"{profile_id!r}." - ) - file_bytes = profile_file.read_bytes() - if hashlib.sha256(file_bytes).hexdigest() == expected: - return "exact" - if ( - manifest_predates_fix - and file_bytes.endswith(b"\n") - and hashlib.sha256(file_bytes[:-1]).hexdigest() == expected - ): - return "legacy_profile_hash" - raise ValueError( - f"Consumer artifact profile {profile_id!r} does not match the manifest " - f"hash: {hashlib.sha256(file_bytes).hexdigest()} != {expected}." ) @@ -1038,12 +164,7 @@ def _reject_non_finite(value: Any) -> Any: def _assert_finite_numbers(value: Any, *, line_number: int, path: Path) -> None: - """Reject NaN/Infinity even inside nested structures. - - ``json.loads`` accepts ``NaN``/``Infinity`` tokens that are not valid JSON - under the consumer-fact contract; a non-finite value must never enter a - schema-valid, hash-valid artifact. - """ + """Reject non-finite numbers, including those in nested structures.""" if isinstance(value, float) and not math.isfinite(value): raise ValueError( f"Row {line_number} of {path} contains a non-finite number: {value!r}." @@ -1057,13 +178,7 @@ def _assert_finite_numbers(value: Any, *, line_number: int, path: Path) -> None: def _recompute_aggregate_fact_key(row: dict[str, Any]) -> str: - """Recompute the aggregate fact key from the row's own content. - - The producer derives ``aggregate_fact_key`` over the row's component keys - plus its raw aggregation/period/geography/entity/assertion; recomputing it - here and comparing rejects a forged or drifted identity key that schema - validation (which only checks key SYNTAX) and uniqueness cannot catch. - """ + """Recompute the aggregate fact key from the row's content.""" assertion = row.get("assertion") payload = { "source_release_key": row.get("source_release_key"), @@ -1154,70 +269,6 @@ def _validate_consumer_row_provenance( ) -def _load_profiles( - profile_ids: Sequence[str], - profile_paths: Sequence[str | Path], -) -> dict[str, dict[str, Any]]: - if not profile_ids and not profile_paths: - return {} - payloads: dict[str, dict[str, Any]] = {} - for profile_id in profile_ids: - profile = load_target_profile(profile_id) - payload = json.loads( - _resource_files("policyengine_chronicle.target_profiles") - .joinpath(f"{profile_id}.json") - .read_text() - ) - payloads[profile.profile_id] = payload - for profile_path in profile_paths: - payload = json.loads(Path(profile_path).read_text()) - profile = target_profile_from_mapping(payload) - if profile.profile_id in payloads: - raise ValueError(f"Duplicate profile_id {profile.profile_id!r}.") - payloads[profile.profile_id] = payload - return payloads - - -def _artifact_coverage( - rows: Sequence[Mapping[str, Any]], - profiles: Mapping[str, dict[str, Any]], -) -> dict[str, Any]: - coverage: dict[str, Any] = {} - for profile_id, payload in profiles.items(): - profile = target_profile_from_mapping(payload) - targets: dict[str, Any] = {} - for target in profile.targets: - per_level: dict[str, Any] = {} - for level in target.geography_levels: - matched, issues = _select_rows( - profile_id, - target, - rows, - geography_level=level, - ) - if issues: - per_level[level] = { - "matched_row_count": 0, - "issues": [issue.to_dict() for issue in issues], - } - continue - periods = sorted( - { - f"{row['period']['type']}:{row['period']['value']}" - for row in matched - } - ) - assertions = sorted({row["assertion"] for row in matched}) - per_level[level] = { - "matched_row_count": len(matched), - "fact_periods": periods, - "assertions": assertions, - } - targets[target.target_id] = per_level - coverage[profile_id] = targets - return coverage - - def _write_json(path: Path, payload: dict[str, Any]) -> None: path.write_text(json.dumps(payload, sort_keys=True, indent=2) + "\n") @@ -1232,16 +283,8 @@ def _sha256_file(path: Path) -> str: __all__ = [ "CONSUMER_ARTIFACT_SCHEMA_VERSION", - "RESOLVED_TARGET_SCHEMA_VERSION", "ConsumerArtifact", "ConsumerArtifactBuildReport", - "PeriodAlignmentDeclaration", - "PeriodContractError", - "PeriodContractViolation", - "ResolutionIssue", - "ResolutionReport", - "ResolvedTarget", "build_consumer_artifact", "load_consumer_artifact", - "resolve_profile_targets", ] diff --git a/policyengine_chronicle/target_profiles/__init__.py b/policyengine_chronicle/target_profiles/__init__.py deleted file mode 100644 index 59f97983..00000000 --- a/policyengine_chronicle/target_profiles/__init__.py +++ /dev/null @@ -1,19 +0,0 @@ -"""Packaged Chronicle target profiles.""" - -from policyengine_chronicle.target_profiles.model import ( - TARGET_PROFILE_SCHEMA_VERSION, - TargetProfile, - TargetProfileBinding, - TargetProfileTarget, - load_target_profile, - target_profile_from_mapping, -) - -__all__ = [ - "TARGET_PROFILE_SCHEMA_VERSION", - "TargetProfile", - "TargetProfileBinding", - "TargetProfileTarget", - "load_target_profile", - "target_profile_from_mapping", -] diff --git a/policyengine_chronicle/target_profiles/model.py b/policyengine_chronicle/target_profiles/model.py deleted file mode 100644 index 5c56106f..00000000 --- a/policyengine_chronicle/target_profiles/model.py +++ /dev/null @@ -1,342 +0,0 @@ -"""Chronicle-owned target profiles for government statistics source records. - -Target profiles describe which source-backed Chronicle facts a downstream build -may select and the semantic quantity those facts represent. They do not contain -target values, runtime hooks, or solver instructions. Values come from Chronicle -fact rows selected by the profile's selectors. -""" - -from __future__ import annotations - -import json -from collections.abc import Mapping -from dataclasses import dataclass -from importlib.resources import files -from typing import Any - -from chronicle.core import ASSERTION_POLICIES, DEFAULT_ASSERTION_POLICY - -TARGET_PROFILE_SCHEMA_VERSION = "policyengine_ledger.target_profile.v1" -FORBIDDEN_VALUE_KEYS = {"aggregation", "operation", "registry", "target_value", "value"} -FORBIDDEN_RUNTIME_KEYS = { - "callable", - "command", - "execute", - "executable", - "function", - "import", - "imports", - "module", - "python_code", - "runtime_code", - "script", - "solver", -} - - -@dataclass(frozen=True) -class TargetProfileBinding: - """Backend-specific semantic reference for one source quantity.""" - - backend: str - metric_name: str - payload: Mapping[str, Any] - - -@dataclass(frozen=True) -class TargetProfileTarget: - """One profile target family and its source quantity contract.""" - - target_id: str - family: str - geography_levels: tuple[str, ...] - chronicle_selector: Mapping[str, Any] - measurement: Mapping[str, Any] - bindings: Mapping[str, TargetProfileBinding] - tolerance: float | None = None - assertion_policy: str | None = None - - def binding(self, backend: str) -> TargetProfileBinding: - """Return the binding for ``backend`` or raise a useful error.""" - try: - return self.bindings[backend] - except KeyError: - raise KeyError( - f"Target profile row {self.target_id!r} has no {backend!r} binding." - ) from None - - -@dataclass(frozen=True) -class TargetProfile: - """A Chronicle-owned source profile referenced by downstream builders.""" - - profile_id: str - country: str - label: str - base_period_policy: str - default_operation: str - default_assertion_policy: str - targets: tuple[TargetProfileTarget, ...] - - def targets_for_geography( - self, - geography_level: str, - ) -> tuple[TargetProfileTarget, ...]: - """Return profile rows active for a geography level.""" - return tuple( - target - for target in self.targets - if geography_level in target.geography_levels - ) - - -def load_target_profile(profile_id: str) -> TargetProfile: - """Load a packaged Chronicle target profile by ID.""" - if not profile_id or "/" in profile_id or "\\" in profile_id: - raise ValueError(f"Invalid target profile id {profile_id!r}.") - path = files(__package__).joinpath(f"{profile_id}.json") - try: - payload = json.loads(path.read_text()) - except FileNotFoundError as exc: - raise FileNotFoundError(f"No packaged target profile {profile_id!r}.") from exc - return target_profile_from_mapping(payload) - - -def target_profile_from_mapping(raw: Mapping[str, Any]) -> TargetProfile: - """Validate and parse a JSON-like target profile mapping.""" - schema_version = raw.get("schema_version") - if schema_version != TARGET_PROFILE_SCHEMA_VERSION: - raise ValueError( - "target profile schema_version must be " - f"{TARGET_PROFILE_SCHEMA_VERSION!r}, got {schema_version!r}." - ) - _reject_forbidden_value_keys(raw, context="target profile") - profile_id = _required_string(raw, "profile_id") - country = _required_string(raw, "country") - label = _required_string(raw, "label") - defaults = _required_mapping(raw, "defaults") - base_period_policy = _required_string(defaults, "base_period_policy") - default_operation = _required_string(defaults, "operation") - if default_operation != "sum": - raise ValueError( - f"target profile {profile_id!r} must use operation 'sum', " - f"got {default_operation!r}." - ) - default_assertion_policy = defaults.get( - "assertion_policy", DEFAULT_ASSERTION_POLICY - ) - if default_assertion_policy not in ASSERTION_POLICIES: - raise ValueError( - f"target profile {profile_id!r} defaults.assertion_policy must be " - f"one of {sorted(ASSERTION_POLICIES)}, got " - f"{default_assertion_policy!r}." - ) - targets = tuple( - _target_from_mapping(target) - for target in _required_mapping_sequence(raw, "targets") - ) - if not targets: - raise ValueError(f"target profile {profile_id!r} must declare targets.") - duplicate_ids = sorted( - target_id - for target_id in {target.target_id for target in targets} - if sum(target.target_id == target_id for target in targets) > 1 - ) - if duplicate_ids: - raise ValueError( - f"target profile {profile_id!r} has duplicate target_id(s): " - f"{duplicate_ids}." - ) - return TargetProfile( - profile_id=profile_id, - country=country, - label=label, - base_period_policy=base_period_policy, - default_operation=default_operation, - default_assertion_policy=default_assertion_policy, - targets=targets, - ) - - -def _target_from_mapping(raw: Mapping[str, Any]) -> TargetProfileTarget: - _reject_forbidden_value_keys(raw, context="target profile row") - target_id = _required_string(raw, "target_id") - family = _required_string(raw, "family") - geography_levels = tuple(_required_string_sequence(raw, "geography_levels")) - if not geography_levels: - raise ValueError(f"target profile row {target_id!r} needs geography_levels.") - chronicle_selector = _required_mapping(raw, "chronicle_selector") - measurement = _required_mapping(raw, "measurement") - _reject_forbidden_contract_keys( - chronicle_selector, - context=f"target profile row {target_id!r} chronicle_selector", - ) - _reject_forbidden_contract_keys( - measurement, - context=f"target profile row {target_id!r} measurement", - ) - bindings_payload = _required_mapping(raw, "bindings") - bindings = { - backend: _binding_from_mapping( - backend, - payload, - target_id=target_id, - ) - for backend, payload in bindings_payload.items() - } - if not bindings: - raise ValueError(f"target profile row {target_id!r} needs bindings.") - tolerance = raw.get("tolerance") - if tolerance is not None: - if not isinstance(tolerance, int | float) or isinstance(tolerance, bool): - raise ValueError(f"target profile row {target_id!r}: invalid tolerance.") - tolerance = float(tolerance) - assertion_policy = raw.get("assertion_policy") - if assertion_policy is not None and assertion_policy not in ASSERTION_POLICIES: - raise ValueError( - f"target profile row {target_id!r} assertion_policy must be one of " - f"{sorted(ASSERTION_POLICIES)}, got {assertion_policy!r}." - ) - if assertion_policy is not None and "assertion" in chronicle_selector: - raise ValueError( - f"target profile row {target_id!r} declares both assertion_policy " - f"{assertion_policy!r} and a chronicle_selector on 'assertion'; " - "the selector already pins the axis — declare one or the other." - ) - return TargetProfileTarget( - target_id=target_id, - family=family, - geography_levels=geography_levels, - chronicle_selector=chronicle_selector, - measurement=measurement, - bindings=bindings, - tolerance=tolerance, - assertion_policy=assertion_policy, - ) - - -def _binding_from_mapping( - backend: str, - raw: Any, - *, - target_id: str, -) -> TargetProfileBinding: - if not isinstance(backend, str) or not backend: - raise ValueError(f"target profile row {target_id!r}: bad binding backend.") - if not isinstance(raw, Mapping): - raise ValueError( - f"target profile row {target_id!r}: binding {backend!r} must be an object." - ) - _reject_forbidden_value_keys(raw, context=f"{backend} binding") - _reject_forbidden_contract_keys( - raw, - context=f"target profile row {target_id!r} {backend} binding", - ) - metric_name = _required_string(raw, "metric_name") - return TargetProfileBinding( - backend=backend, - metric_name=metric_name, - payload=dict(raw), - ) - - -def _reject_forbidden_value_keys(raw: Mapping[str, Any], *, context: str) -> None: - forbidden = FORBIDDEN_VALUE_KEYS | FORBIDDEN_RUNTIME_KEYS - present = sorted(key for key in forbidden if key in raw) - if present: - raise ValueError( - f"{context} must not declare {present}; Chronicle profiles use " - "implicit Chronicle source selection, sum-only measurement, no " - "runtime execution hooks, and values coming from Chronicle facts." - ) - - -def _reject_forbidden_contract_keys(value: Any, *, context: str) -> None: - """Reject target-value or registry controls nested in contract payloads. - - Filter thresholds such as ``{"operator": ">", "value": 0}`` are valid - measurement predicates, so this recursive guard allows ``value`` only in - recognized filter predicate objects. Other ``value`` keys are rejected so - target amounts cannot hide inside selectors or measurement contracts. - """ - - if isinstance(value, Mapping): - forbidden = FORBIDDEN_VALUE_KEYS - {"value"} - if not _is_filter_predicate(value): - forbidden = forbidden | {"value"} - forbidden = forbidden | FORBIDDEN_RUNTIME_KEYS - present = sorted(key for key in forbidden if key in value) - if present: - raise ValueError( - f"{context} must not declare {present}; Chronicle target profiles " - "use implicit source selection, sum-only measurement, no " - "runtime execution hooks, and values coming from Chronicle facts." - ) - for key, item in value.items(): - _reject_forbidden_contract_keys(item, context=f"{context}.{key}") - elif isinstance(value, list | tuple): - for index, item in enumerate(value): - _reject_forbidden_contract_keys(item, context=f"{context}[{index}]") - - -def _is_filter_predicate(value: Mapping[str, Any]) -> bool: - return ( - "value" in value - and "operator" in value - and ("concept" in value or "variable" in value) - ) - - -def _required_string(raw: Mapping[str, Any], key: str) -> str: - value = raw.get(key) - if not isinstance(value, str) or not value: - raise ValueError(f"target profile field {key!r} must be a non-empty string.") - return value - - -def _required_mapping(raw: Mapping[str, Any], key: str) -> Mapping[str, Any]: - value = raw.get(key) - if not isinstance(value, Mapping): - raise ValueError(f"target profile field {key!r} must be an object.") - return value - - -def _required_mapping_sequence( - raw: Mapping[str, Any], - key: str, -) -> tuple[Mapping[str, Any], ...]: - value = raw.get(key) - if not isinstance(value, list | tuple): - raise ValueError(f"target profile field {key!r} must be a list.") - rows: list[Mapping[str, Any]] = [] - for index, row in enumerate(value): - if not isinstance(row, Mapping): - raise ValueError( - f"target profile field {key!r} row {index} must be an object." - ) - rows.append(row) - return tuple(rows) - - -def _required_string_sequence(raw: Mapping[str, Any], key: str) -> tuple[str, ...]: - value = raw.get(key) - if not isinstance(value, list | tuple): - raise ValueError(f"target profile field {key!r} must be a list.") - strings: list[str] = [] - for index, item in enumerate(value): - if not isinstance(item, str) or not item: - raise ValueError( - f"target profile field {key!r} item {index} must be a non-empty string." - ) - strings.append(item) - return tuple(strings) - - -__all__ = [ - "TARGET_PROFILE_SCHEMA_VERSION", - "TargetProfile", - "TargetProfileBinding", - "TargetProfileTarget", - "load_target_profile", - "target_profile_from_mapping", -] diff --git a/policyengine_chronicle/target_profiles/uk_firms.json b/policyengine_chronicle/target_profiles/uk_firms.json deleted file mode 100644 index 89dc0d0a..00000000 --- a/policyengine_chronicle/target_profiles/uk_firms.json +++ /dev/null @@ -1,272 +0,0 @@ -{ - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "uk_firms", - "country": "uk", - "label": "UK firm calibration", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum" - }, - "targets": [ - { - "target_id": "ons.uk_business.enterprise_count.turnover_bands", - "family": "ons_uk_business", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "ons", - "source_measure_id": "enterprise_count", - "record_set_id": "ons.uk_business.cy2025.enterprise_count.by_turnover_band", - "groupby_dimension": "uk.firm.annual_turnover" - }, - "measurement": { - "entity": "firm", - "concept": "uk.firm.count", - "groupby_dimension": "uk.firm.annual_turnover" - }, - "bindings": { - "microcosm": { - "metric_name": "ons/uk_business/enterprise_count/turnover_bands", - "value_variable": "firm_count", - "from_entity": "firm", - "groupby_variable": "annual_turnover" - }, - "axiom": { - "metric_name": "ons/uk_business/enterprise_count/turnover_bands", - "status": "pending", - "value_rule": "uk.firm.count" - } - } - }, - { - "target_id": "ons.uk_business.enterprise_count.employment_bands", - "family": "ons_uk_business", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "ons", - "source_measure_id": "enterprise_count", - "record_set_id": "ons.uk_business.cy2025.enterprise_count.by_employment_band", - "groupby_dimension": "uk.firm.employees" - }, - "measurement": { - "entity": "firm", - "concept": "uk.firm.count", - "groupby_dimension": "uk.firm.employees" - }, - "bindings": { - "microcosm": { - "metric_name": "ons/uk_business/enterprise_count/employment_bands", - "value_variable": "firm_count", - "from_entity": "firm", - "groupby_variable": "employment" - }, - "axiom": { - "metric_name": "ons/uk_business/enterprise_count/employment_bands", - "status": "pending", - "value_rule": "uk.firm.count" - } - } - }, - { - "target_id": "hmrc.vat.registered_trader_count.turnover_bands", - "family": "hmrc_vat", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "hmrc", - "source_measure_id": "vat_registered_trader_count", - "record_set_id": "hmrc.vat.fy2024_25.registered_trader_count.by_turnover_band", - "groupby_dimension": "uk.firm.annual_turnover" - }, - "measurement": { - "entity": "firm", - "concept": "uk.firm.count", - "groupby_dimension": "uk.firm.annual_turnover", - "filters": [ - {"concept": "uk.firm.vat_registered", "operator": "==", "value": true} - ] - }, - "bindings": { - "microcosm": { - "metric_name": "hmrc/vat/registered_trader_count/turnover_bands", - "value_variable": "firm_count", - "from_entity": "firm", - "groupby_variable": "annual_turnover", - "filters": [ - {"variable": "vat_registered", "operator": "==", "value": true} - ] - }, - "axiom": { - "metric_name": "hmrc/vat/registered_trader_count/turnover_bands", - "status": "pending", - "value_rule": "uk.firm.count", - "filter_rule": "uk:policies/govuk/vat#firm_vat_registered" - } - } - }, - { - "target_id": "hmrc.vat.net_liability.turnover_bands", - "family": "hmrc_vat", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "hmrc", - "source_measure_id": "net_vat_liability", - "record_set_id": "hmrc.vat.fy2024_25.net_liability.by_turnover_band", - "groupby_dimension": "uk.firm.annual_turnover" - }, - "measurement": { - "entity": "firm", - "concept": "uk.tax.vat.net_liability", - "groupby_dimension": "uk.firm.annual_turnover", - "filters": [ - {"concept": "uk.firm.vat_registered", "operator": "==", "value": true} - ] - }, - "bindings": { - "microcosm": { - "metric_name": "hmrc/vat/net_liability/turnover_bands", - "value_variable": "vat_liability", - "from_entity": "firm", - "groupby_variable": "annual_turnover", - "filters": [ - {"variable": "vat_registered", "operator": "==", "value": true} - ] - }, - "axiom": { - "metric_name": "hmrc/vat/net_liability/turnover_bands", - "status": "pending", - "value_rule": "uk:policies/govuk/vat#net_vat_liability", - "filter_rule": "uk:policies/govuk/vat#firm_vat_registered" - } - } - }, - { - "target_id": "ons.uk_business.enterprise_count.sic_turnover_bands", - "family": "ons_uk_business", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "ons", - "source_measure_id": "enterprise_count", - "record_set_id": "ons.uk_business.cy2025.enterprise_count.by_sic_turnover_band", - "dimensions": ["uk.firm.sic_code", "uk.firm.turnover_band"] - }, - "measurement": { - "entity": "firm", - "concept": "uk.firm.count", - "dimensions": ["uk.firm.sic_code", "uk.firm.turnover_band"] - }, - "bindings": { - "microcosm": { - "metric_name": "ons/uk_business/enterprise_count/sic_turnover_bands", - "value_variable": "firm_count", - "from_entity": "firm", - "groupby_variables": ["sic_code", "annual_turnover"] - }, - "axiom": { - "metric_name": "ons/uk_business/enterprise_count/sic_turnover_bands", - "status": "pending", - "value_rule": "uk.firm.count" - } - } - }, - { - "target_id": "ons.uk_business.enterprise_count.sic_employment_bands", - "family": "ons_uk_business", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "ons", - "source_measure_id": "enterprise_count", - "record_set_id": "ons.uk_business.cy2025.enterprise_count.by_sic_employment_band", - "dimensions": ["uk.firm.sic_code", "uk.firm.employment_band"] - }, - "measurement": { - "entity": "firm", - "concept": "uk.firm.count", - "dimensions": ["uk.firm.sic_code", "uk.firm.employment_band"] - }, - "bindings": { - "microcosm": { - "metric_name": "ons/uk_business/enterprise_count/sic_employment_bands", - "value_variable": "firm_count", - "from_entity": "firm", - "groupby_variables": ["sic_code", "employment"] - }, - "axiom": { - "metric_name": "ons/uk_business/enterprise_count/sic_employment_bands", - "status": "pending", - "value_rule": "uk.firm.count" - } - } - }, - { - "target_id": "hmrc.vat.registered_trader_count.sic_sectors", - "family": "hmrc_vat", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "hmrc", - "source_measure_id": "vat_registered_trader_count", - "record_set_id": "hmrc.vat.fy2024_25.registered_trader_count.by_sic", - "groupby_dimension": "uk.firm.sic_code" - }, - "measurement": { - "entity": "firm", - "concept": "uk.firm.count", - "groupby_dimension": "uk.firm.sic_code", - "filters": [ - {"concept": "uk.firm.vat_registered", "operator": "==", "value": true} - ] - }, - "bindings": { - "microcosm": { - "metric_name": "hmrc/vat/registered_trader_count/sic_sectors", - "value_variable": "firm_count", - "from_entity": "firm", - "groupby_variable": "sic_code", - "filters": [ - {"variable": "vat_registered", "operator": "==", "value": true} - ] - }, - "axiom": { - "metric_name": "hmrc/vat/registered_trader_count/sic_sectors", - "status": "pending", - "value_rule": "uk.firm.count", - "filter_rule": "uk:policies/govuk/vat#firm_vat_registered" - } - } - }, - { - "target_id": "hmrc.vat.net_liability.sic_sectors", - "family": "hmrc_vat", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "hmrc", - "source_measure_id": "net_vat_liability", - "record_set_id": "hmrc.vat.fy2024_25.net_liability.by_sic", - "groupby_dimension": "uk.firm.sic_code" - }, - "measurement": { - "entity": "firm", - "concept": "uk.tax.vat.net_liability", - "groupby_dimension": "uk.firm.sic_code", - "filters": [ - {"concept": "uk.firm.vat_registered", "operator": "==", "value": true} - ] - }, - "bindings": { - "microcosm": { - "metric_name": "hmrc/vat/net_liability/sic_sectors", - "value_variable": "vat_liability", - "from_entity": "firm", - "groupby_variable": "sic_code", - "filters": [ - {"variable": "vat_registered", "operator": "==", "value": true} - ] - }, - "axiom": { - "metric_name": "hmrc/vat/net_liability/sic_sectors", - "status": "pending", - "value_rule": "uk:policies/govuk/vat#net_vat_liability", - "filter_rule": "uk:policies/govuk/vat#firm_vat_registered" - } - } - } - ] -} diff --git a/policyengine_chronicle/target_profiles/uk_local_geography.json b/policyengine_chronicle/target_profiles/uk_local_geography.json deleted file mode 100644 index 845eef1f..00000000 --- a/policyengine_chronicle/target_profiles/uk_local_geography.json +++ /dev/null @@ -1,346 +0,0 @@ -{ - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "uk_local_geography", - "country": "uk", - "label": "UK local geography calibration", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum" - }, - "targets": [ - { - "target_id": "hmrc.self_employment_income.amount", - "family": "hmrc", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": { - "source_name": "hmrc", - "source_measure_id": "self_employment_income_amount" - }, - "measurement": { - "entity": "person", - "map_to": "household", - "concept": "uk.income.self_employment.amount", - "filters": [{"concept": "uk.tax.income_tax", "operator": ">", "value": 0}] - }, - "bindings": { - "policyengine": { - "metric_name": "hmrc/self_employment_income/amount", - "value_variable": "self_employment_income", - "from_entity": "person", - "map_to": "household", - "filters": [{"variable": "income_tax", "operator": ">", "value": 0}] - }, - "axiom": { - "metric_name": "hmrc/self_employment_income/amount", - "status": "pending", - "value_rule": "uk.income.self_employment.amount" - } - } - }, - { - "target_id": "hmrc.self_employment_income.count", - "family": "hmrc", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": { - "source_name": "hmrc", - "source_measure_id": "self_employment_income_count" - }, - "measurement": { - "entity": "person", - "map_to": "household", - "concept": "uk.person.count", - "filters": [ - {"concept": "uk.income.self_employment.amount", "operator": "!=", "value": 0}, - {"concept": "uk.tax.income_tax", "operator": ">", "value": 0} - ] - }, - "bindings": { - "policyengine": { - "metric_name": "hmrc/self_employment_income/count", - "value_variable": "person_count", - "from_entity": "person", - "map_to": "household", - "filters": [ - {"variable": "self_employment_income", "operator": "!=", "value": 0}, - {"variable": "income_tax", "operator": ">", "value": 0} - ] - }, - "axiom": { - "metric_name": "hmrc/self_employment_income/count", - "status": "pending", - "value_rule": "uk.person.count" - } - } - }, - { - "target_id": "hmrc.employment_income.amount", - "family": "hmrc", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": { - "source_name": "hmrc", - "source_measure_id": "employment_income_amount" - }, - "measurement": { - "entity": "person", - "map_to": "household", - "concept": "uk.income.employment.amount", - "filters": [{"concept": "uk.tax.income_tax", "operator": ">", "value": 0}] - }, - "bindings": { - "policyengine": { - "metric_name": "hmrc/employment_income/amount", - "value_variable": "employment_income", - "from_entity": "person", - "map_to": "household", - "filters": [{"variable": "income_tax", "operator": ">", "value": 0}] - }, - "axiom": { - "metric_name": "hmrc/employment_income/amount", - "status": "pending", - "value_rule": "uk.income.employment.amount" - } - } - }, - { - "target_id": "hmrc.employment_income.count", - "family": "hmrc", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": { - "source_name": "hmrc", - "source_measure_id": "employment_income_count" - }, - "measurement": { - "entity": "person", - "map_to": "household", - "concept": "uk.person.count", - "filters": [ - {"concept": "uk.income.employment.amount", "operator": "!=", "value": 0}, - {"concept": "uk.tax.income_tax", "operator": ">", "value": 0} - ] - }, - "bindings": { - "policyengine": { - "metric_name": "hmrc/employment_income/count", - "value_variable": "person_count", - "from_entity": "person", - "map_to": "household", - "filters": [ - {"variable": "employment_income", "operator": "!=", "value": 0}, - {"variable": "income_tax", "operator": ">", "value": 0} - ] - }, - "axiom": { - "metric_name": "hmrc/employment_income/count", - "status": "pending", - "value_rule": "uk.person.count" - } - } - }, - { - "target_id": "ons.age.0_10", - "family": "ons_population", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "population"}, - "measurement": { - "entity": "person", - "map_to": "household", - "concept": "uk.person.count", - "filters": [{"concept": "uk.demographics.age", "lower": 0, "upper": 10}] - }, - "bindings": { - "policyengine": { - "metric_name": "age/0_10", - "value_variable": "person_count", - "from_entity": "person", - "map_to": "household", - "filters": [{"variable": "age", "lower": 0, "upper": 10}] - }, - "axiom": {"metric_name": "age/0_10", "status": "pending", "value_rule": "uk.person.count"} - } - }, - { - "target_id": "ons.age.10_20", - "family": "ons_population", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "population"}, - "measurement": {"entity": "person", "map_to": "household", "concept": "uk.person.count", "filters": [{"concept": "uk.demographics.age", "lower": 10, "upper": 20}]}, - "bindings": { - "policyengine": {"metric_name": "age/10_20", "value_variable": "person_count", "from_entity": "person", "map_to": "household", "filters": [{"variable": "age", "lower": 10, "upper": 20}]}, - "axiom": {"metric_name": "age/10_20", "status": "pending", "value_rule": "uk.person.count"} - } - }, - { - "target_id": "ons.age.20_30", - "family": "ons_population", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "population"}, - "measurement": {"entity": "person", "map_to": "household", "concept": "uk.person.count", "filters": [{"concept": "uk.demographics.age", "lower": 20, "upper": 30}]}, - "bindings": { - "policyengine": {"metric_name": "age/20_30", "value_variable": "person_count", "from_entity": "person", "map_to": "household", "filters": [{"variable": "age", "lower": 20, "upper": 30}]}, - "axiom": {"metric_name": "age/20_30", "status": "pending", "value_rule": "uk.person.count"} - } - }, - { - "target_id": "ons.age.30_40", - "family": "ons_population", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "population"}, - "measurement": {"entity": "person", "map_to": "household", "concept": "uk.person.count", "filters": [{"concept": "uk.demographics.age", "lower": 30, "upper": 40}]}, - "bindings": { - "policyengine": {"metric_name": "age/30_40", "value_variable": "person_count", "from_entity": "person", "map_to": "household", "filters": [{"variable": "age", "lower": 30, "upper": 40}]}, - "axiom": {"metric_name": "age/30_40", "status": "pending", "value_rule": "uk.person.count"} - } - }, - { - "target_id": "ons.age.40_50", - "family": "ons_population", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "population"}, - "measurement": {"entity": "person", "map_to": "household", "concept": "uk.person.count", "filters": [{"concept": "uk.demographics.age", "lower": 40, "upper": 50}]}, - "bindings": { - "policyengine": {"metric_name": "age/40_50", "value_variable": "person_count", "from_entity": "person", "map_to": "household", "filters": [{"variable": "age", "lower": 40, "upper": 50}]}, - "axiom": {"metric_name": "age/40_50", "status": "pending", "value_rule": "uk.person.count"} - } - }, - { - "target_id": "ons.age.50_60", - "family": "ons_population", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "population"}, - "measurement": {"entity": "person", "map_to": "household", "concept": "uk.person.count", "filters": [{"concept": "uk.demographics.age", "lower": 50, "upper": 60}]}, - "bindings": { - "policyengine": {"metric_name": "age/50_60", "value_variable": "person_count", "from_entity": "person", "map_to": "household", "filters": [{"variable": "age", "lower": 50, "upper": 60}]}, - "axiom": {"metric_name": "age/50_60", "status": "pending", "value_rule": "uk.person.count"} - } - }, - { - "target_id": "ons.age.60_70", - "family": "ons_population", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "population"}, - "measurement": {"entity": "person", "map_to": "household", "concept": "uk.person.count", "filters": [{"concept": "uk.demographics.age", "lower": 60, "upper": 70}]}, - "bindings": { - "policyengine": {"metric_name": "age/60_70", "value_variable": "person_count", "from_entity": "person", "map_to": "household", "filters": [{"variable": "age", "lower": 60, "upper": 70}]}, - "axiom": {"metric_name": "age/60_70", "status": "pending", "value_rule": "uk.person.count"} - } - }, - { - "target_id": "ons.age.70_80", - "family": "ons_population", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "population"}, - "measurement": {"entity": "person", "map_to": "household", "concept": "uk.person.count", "filters": [{"concept": "uk.demographics.age", "lower": 70, "upper": 80}]}, - "bindings": { - "policyengine": {"metric_name": "age/70_80", "value_variable": "person_count", "from_entity": "person", "map_to": "household", "filters": [{"variable": "age", "lower": 70, "upper": 80}]}, - "axiom": {"metric_name": "age/70_80", "status": "pending", "value_rule": "uk.person.count"} - } - }, - { - "target_id": "dwp.universal_credit.households", - "family": "dwp_universal_credit", - "geography_levels": ["constituency", "local_authority"], - "chronicle_selector": {"source_name": "dwp", "source_measure_id": "universal_credit_households"}, - "measurement": {"entity": "benunit", "map_to": "household", "concept": "uk.benefit_unit.count", "filters": [{"concept": "uk.benefits.universal_credit.amount", "operator": ">", "value": 0}]}, - "bindings": { - "policyengine": {"metric_name": "uc_households", "value_variable": "benunit_count", "from_entity": "benunit", "map_to": "household", "filters": [{"variable": "universal_credit", "operator": ">", "value": 0}]}, - "axiom": {"metric_name": "uc_households", "status": "pending", "value_rule": "uk.benefit_unit.count"} - } - }, - { - "target_id": "dwp.universal_credit.households.0_children", - "family": "dwp_universal_credit", - "geography_levels": ["constituency"], - "chronicle_selector": {"source_name": "dwp", "source_measure_id": "universal_credit_households_0_children"}, - "measurement": {"entity": "household", "concept": "uk.household.count", "filters": [{"concept": "uk.benefits.universal_credit.household_receives", "equals": true}, {"concept": "uk.household.children", "equals": 0}]}, - "bindings": {"policyengine": {"metric_name": "uc_hh_0_children", "value_variable": "household_count"}, "axiom": {"metric_name": "uc_hh_0_children", "status": "pending", "value_rule": "uk.household.count"}} - }, - { - "target_id": "dwp.universal_credit.households.1_child", - "family": "dwp_universal_credit", - "geography_levels": ["constituency"], - "chronicle_selector": {"source_name": "dwp", "source_measure_id": "universal_credit_households_1_child"}, - "measurement": {"entity": "household", "concept": "uk.household.count", "filters": [{"concept": "uk.benefits.universal_credit.household_receives", "equals": true}, {"concept": "uk.household.children", "equals": 1}]}, - "bindings": {"policyengine": {"metric_name": "uc_hh_1_child", "value_variable": "household_count"}, "axiom": {"metric_name": "uc_hh_1_child", "status": "pending", "value_rule": "uk.household.count"}} - }, - { - "target_id": "dwp.universal_credit.households.2_children", - "family": "dwp_universal_credit", - "geography_levels": ["constituency"], - "chronicle_selector": {"source_name": "dwp", "source_measure_id": "universal_credit_households_2_children"}, - "measurement": {"entity": "household", "concept": "uk.household.count", "filters": [{"concept": "uk.benefits.universal_credit.household_receives", "equals": true}, {"concept": "uk.household.children", "equals": 2}]}, - "bindings": {"policyengine": {"metric_name": "uc_hh_2_children", "value_variable": "household_count"}, "axiom": {"metric_name": "uc_hh_2_children", "status": "pending", "value_rule": "uk.household.count"}} - }, - { - "target_id": "dwp.universal_credit.households.3plus_children", - "family": "dwp_universal_credit", - "geography_levels": ["constituency"], - "chronicle_selector": {"source_name": "dwp", "source_measure_id": "universal_credit_households_3plus_children"}, - "measurement": {"entity": "household", "concept": "uk.household.count", "filters": [{"concept": "uk.benefits.universal_credit.household_receives", "equals": true}, {"concept": "uk.household.children", "lower": 3}]}, - "bindings": {"policyengine": {"metric_name": "uc_hh_3plus_children", "value_variable": "household_count"}, "axiom": {"metric_name": "uc_hh_3plus_children", "status": "pending", "value_rule": "uk.household.count"}} - }, - { - "target_id": "ons.equiv_net_income_bhc", - "family": "ons_income", - "geography_levels": ["local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "equivalised_net_income_before_housing_costs"}, - "measurement": {"entity": "household", "concept": "uk.household.equivalised_net_income_bhc"}, - "bindings": {"policyengine": {"metric_name": "ons/equiv_net_income_bhc", "value_variable": "equiv_hbai_household_net_income"}, "axiom": {"metric_name": "ons/equiv_net_income_bhc", "status": "pending", "value_rule": "uk.household.equivalised_net_income_bhc"}} - }, - { - "target_id": "ons.equiv_net_income_ahc", - "family": "ons_income", - "geography_levels": ["local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "equivalised_net_income_after_housing_costs"}, - "measurement": {"entity": "household", "concept": "uk.household.equivalised_net_income_ahc"}, - "bindings": {"policyengine": {"metric_name": "ons/equiv_net_income_ahc", "value_variable": "equiv_hbai_household_net_income_ahc"}, "axiom": {"metric_name": "ons/equiv_net_income_ahc", "status": "pending", "value_rule": "uk.household.equivalised_net_income_ahc"}} - }, - { - "target_id": "ons.equiv_housing_costs", - "family": "ons_income", - "geography_levels": ["local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "equivalised_housing_costs"}, - "measurement": {"entity": "household", "concept": "uk.household.equivalised_housing_costs"}, - "bindings": {"policyengine": {"metric_name": "ons/equiv_housing_costs", "value_expression": "equiv_hbai_household_net_income - equiv_hbai_household_net_income_ahc"}, "axiom": {"metric_name": "ons/equiv_housing_costs", "status": "pending", "value_rule": "uk.household.equivalised_housing_costs"}} - }, - { - "target_id": "ons.tenure.owned_outright", - "family": "ons_housing", - "geography_levels": ["local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "owned_outright_households"}, - "measurement": {"entity": "household", "concept": "uk.household.count", "filters": [{"concept": "uk.household.tenure", "equals": "owned_outright"}]}, - "bindings": {"policyengine": {"metric_name": "tenure/owned_outright", "value_variable": "household_count", "filters": [{"variable": "tenure_type", "equals": "OWNED_OUTRIGHT"}]}, "axiom": {"metric_name": "tenure/owned_outright", "status": "pending", "value_rule": "uk.household.count"}} - }, - { - "target_id": "ons.tenure.owned_mortgage", - "family": "ons_housing", - "geography_levels": ["local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "owned_with_mortgage_households"}, - "measurement": {"entity": "household", "concept": "uk.household.count", "filters": [{"concept": "uk.household.tenure", "equals": "owned_with_mortgage"}]}, - "bindings": {"policyengine": {"metric_name": "tenure/owned_mortgage", "value_variable": "household_count", "filters": [{"variable": "tenure_type", "equals": "OWNED_WITH_MORTGAGE"}]}, "axiom": {"metric_name": "tenure/owned_mortgage", "status": "pending", "value_rule": "uk.household.count"}} - }, - { - "target_id": "ons.tenure.private_rent", - "family": "ons_housing", - "geography_levels": ["local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "private_rent_households"}, - "measurement": {"entity": "household", "concept": "uk.household.count", "filters": [{"concept": "uk.household.tenure", "equals": "private_rent"}]}, - "bindings": {"policyengine": {"metric_name": "tenure/private_rent", "value_variable": "household_count", "filters": [{"variable": "tenure_type", "equals": "RENT_PRIVATELY"}]}, "axiom": {"metric_name": "tenure/private_rent", "status": "pending", "value_rule": "uk.household.count"}} - }, - { - "target_id": "ons.tenure.social_rent", - "family": "ons_housing", - "geography_levels": ["local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "social_rent_households"}, - "measurement": {"entity": "household", "concept": "uk.household.count", "filters": [{"concept": "uk.household.tenure", "in": ["rent_from_council", "rent_from_housing_association"]}]}, - "bindings": {"policyengine": {"metric_name": "tenure/social_rent", "value_variable": "household_count", "filters": [{"variable": "tenure_type", "in": ["RENT_FROM_COUNCIL", "RENT_FROM_HA"]}]}, "axiom": {"metric_name": "tenure/social_rent", "status": "pending", "value_rule": "uk.household.count"}} - }, - { - "target_id": "ons.rent.private_rent", - "family": "ons_housing", - "geography_levels": ["local_authority"], - "chronicle_selector": {"source_name": "ons", "source_measure_id": "private_rent"}, - "measurement": {"entity": "benunit", "map_to": "household", "concept": "uk.housing.private_rent.amount"}, - "bindings": {"policyengine": {"metric_name": "rent/private_rent", "value_variable": "benunit_rent", "from_entity": "benunit", "map_to": "household", "filters": [{"variable": "tenure_type", "equals": "RENT_PRIVATELY"}]}, "axiom": {"metric_name": "rent/private_rent", "status": "pending", "value_rule": "uk.housing.private_rent.amount"}} - } - ] -} diff --git a/policyengine_chronicle/targets/__init__.py b/policyengine_chronicle/targets/__init__.py index e792f497..4055e36a 100644 --- a/policyengine_chronicle/targets/__init__.py +++ b/policyengine_chronicle/targets/__init__.py @@ -1,8 +1,8 @@ """Chronicle target-input helpers. Chronicle owns source-backed facts, target-eligible source inputs, and target -profiles. Consumers such as Microcosm decide which profile rows their support -universe can activate and how to calibrate outside Chronicle. +source coverage. Consumers such as Microcosm own selection contracts, decide +which rows their support universe can activate, and calibrate outside Chronicle. """ __all__ = [ diff --git a/pyproject.toml b/pyproject.toml index f73783d3..2efd874f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "policyengine-chronicle" version = "0.1.0" -description = "PolicyEngine Chronicle source-data foundation: source facts and target profiles" +description = "PolicyEngine Chronicle source-data foundation: source-backed facts" readme = "README.md" license = { text = "MIT" } requires-python = ">=3.11" diff --git a/tests/test_chronicle_consumer.py b/tests/test_chronicle_consumer.py index 9ad19c37..a8c8ae3b 100644 --- a/tests/test_chronicle_consumer.py +++ b/tests/test_chronicle_consumer.py @@ -1,24 +1,14 @@ -"""Tests for consumer artifacts and period-contract resolution. - -The load-bearing behavior: consuming a fact at a period other than its -reference period is a hard error unless the consumer declares a named, -versioned alignment. This is the schema-level guard against the -PolicyEngine/populace#212 failure, where SOI tax-year dollar levels were -calibrated un-aged at a later build year and nothing complained. -""" +"""Tests for facts-only Chronicle consumer artifacts.""" from __future__ import annotations import hashlib import json -from pathlib import Path import pytest -import policyengine_chronicle.target_profiles as target_profiles_pkg from chronicle.consumer_contract import consumer_fact_rows from chronicle.core import ( - AggregateConstraint, AggregateFact, Aggregation, EntityDimension, @@ -28,382 +18,146 @@ SourceProvenance, SourceRecordLayout, ) +from chronicle.harness import main from policyengine_chronicle.consumer import ( - PeriodAlignmentDeclaration, - PeriodContractError, build_consumer_artifact, load_consumer_artifact, - resolve_profile_targets, ) from policyengine_chronicle.schema import CONSUMER_FACT_SCHEMA_SHA256 -from policyengine_chronicle.target_profiles import target_profile_from_mapping SHA = "ab" * 32 -def _fact( - *, - value, - period_value, - period_type="tax_year", - source_name="irs_soi", - measure_id="agi", - concept="irs_soi.adjusted_gross_income", - assertion="observation", - provenance_class="administrative", - survey_instrument=None, - geography_id="0100000US", -): +def _fact(*, value, period_value): return AggregateFact( value=value, - period=PeriodDimension(type=period_type, value=period_value), + period=PeriodDimension(type="tax_year", value=period_value), geography=GeographyDimension( level="country", - id=geography_id, + id="0100000US", vintage="2020_census", ), entity=EntityDimension(name="tax_unit", role="filing_unit"), - measure=Measure(concept=concept, unit="usd"), + measure=Measure(concept="irs_soi.adjusted_gross_income", unit="usd"), aggregation=Aggregation(method="sum"), - provenance_class=provenance_class, - survey_instrument=survey_instrument, + provenance_class="administrative", source=SourceProvenance( - source_name=source_name, + source_name="irs_soi", source_table="Table T", source_file="t.xls", url="https://example.gov/t.xls", - vintage=f"{period_type}_{period_value}", + vintage=f"tax_year_{period_value}", extracted_at="2026-05-01", extraction_method="test", source_sha256=SHA, source_size_bytes=10, raw_r2_bucket="ledger-raw", - raw_r2_key=f"raw/{source_name}/t/{period_value}/{SHA}/t.xls", - raw_r2_uri=f"r2://ledger-raw/raw/{source_name}/t/{period_value}/{SHA}/t.xls", + raw_r2_key=f"raw/irs_soi/t/{period_value}/{SHA}/t.xls", + raw_r2_uri=f"r2://ledger-raw/raw/irs_soi/t/{period_value}/{SHA}/t.xls", ), domain="all_returns", - source_record_id=f"{source_name}.{period_value}.t.all.{measure_id}", - source_cell_keys=(f"ledger.source_cell.v1:{period_value}{measure_id}",), + source_record_id=f"irs_soi.{period_value}.t.all.agi", + source_cell_keys=(f"ledger.source_cell.v1:{period_value}agi",), layout=SourceRecordLayout( - record_set_id=f"{source_name}.{period_value}.t", - record_set_spec_id=f"{source_name}.t.v1", - measure_id=measure_id, + record_set_id=f"irs_soi.{period_value}.t", + record_set_spec_id="irs_soi.t.v1", + measure_id="agi", groupby_dimension="us.agi", ), - assertion=assertion, ) def _rows(): return consumer_fact_rows( - [ - _fact(value=100, period_value=2021), - _fact(value=110, period_value=2022), - _fact( - value=250, - period_value=2027, - period_type="calendar_year", - source_name="cbo", - measure_id="receipts", - concept="cbo.individual_income_tax_receipts", - assertion="source_projection", - ), - ] - ) - - -def _profile(selector=None, target_id="soi.agi.total"): - return target_profile_from_mapping( - { - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "test_profile", - "country": "us", - "label": "Test profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": target_id, - "family": "irs_soi", - "geography_levels": ["country"], - "chronicle_selector": selector - or {"source_name": "irs_soi", "source_measure_id": "agi"}, - "measurement": { - "entity": "tax_unit", - "concept": "us.agi", - }, - "bindings": { - "microcosm": {"metric_name": "irs_soi/agi/total"}, - }, - } - ], - } + [_fact(value=100, period_value=2021), _fact(value=110, period_value=2022)] ) -def test_resolving_at_the_fact_period_is_fact_basis(): - report = resolve_profile_targets( - _profile(), - _rows(), - {"type": "tax_year", "value": 2022}, - ) - assert report.valid - assert len(report.resolved) == 1 - row = report.resolved[0] - assert row.basis == "fact" - assert row.value == 110 - assert row.assertion == "observation" - assert row.fact_period == {"type": "tax_year", "value": 2022} - assert row.requested_period == {"type": "tax_year", "value": 2022} - assert row.alignment is None - - -def test_un_aged_consumption_hard_fails(): - with pytest.raises(PeriodContractError) as excinfo: - resolve_profile_targets( - _profile(), - _rows(), - {"type": "tax_year", "value": 2025}, - ) - message = str(excinfo.value) - assert "PeriodAlignmentDeclaration" in message - (violation,) = excinfo.value.violations - assert violation.fact_period == {"type": "tax_year", "value": 2022} - assert violation.requested_period == {"type": "tax_year", "value": 2025} - - -def test_declared_alignment_resolves_and_is_recorded(): - declaration = PeriodAlignmentDeclaration( - model_id="cbo_growth_factor_aging", - model_version="2026.1", - parameters={"factor_series": "cbo.baseline_2026_01.agi_growth"}, - ) - report = resolve_profile_targets( - _profile(), - _rows(), - {"type": "tax_year", "value": 2025}, - alignments={"soi.agi.total": declaration}, - ) - assert report.valid - (row,) = report.resolved - assert row.basis == "declared_alignment" - # Chronicle returns the published level; the consumer applies the model. - assert row.value == 110 - assert row.fact_period == {"type": "tax_year", "value": 2022} - assert row.requested_period == {"type": "tax_year", "value": 2025} - assert row.alignment["model_id"] == "cbo_growth_factor_aging" - assert row.alignment["model_version"] == "2026.1" - - -def test_latest_not_after_selects_the_newest_covered_period(): - report = resolve_profile_targets( - _profile(), - _rows(), - {"type": "tax_year", "value": 2021}, - ) - (row,) = report.resolved - assert row.basis == "fact" - assert row.value == 100 - - -def test_wildcard_alignment_covers_all_targets(): - declaration = PeriodAlignmentDeclaration( - model_id="cbo_growth_factor_aging", - model_version="2026.1", - ) - report = resolve_profile_targets( - _profile(), - _rows(), - {"type": "tax_year", "value": 2025}, - alignments=declaration, - ) - assert report.valid - assert report.resolved[0].basis == "declared_alignment" - - -def test_source_projections_resolve_at_their_own_period_as_facts(): - profile = _profile( - selector={ - "source_name": "cbo", - "source_measure_id": "receipts", - "assertion": "source_projection", - }, - target_id="cbo.receipts.2027", - ) - report = resolve_profile_targets( - profile, - _rows(), - {"type": "calendar_year", "value": 2027}, - ) - (row,) = report.resolved - assert row.basis == "fact" - assert row.assertion == "source_projection" - assert row.value == 250 - - -def test_alignment_declarations_reject_values_and_runtime_hooks(): - with pytest.raises(ValueError, match="never"): - PeriodAlignmentDeclaration( - model_id="m", - model_version="1", - parameters={"target_value": 130e9}, - ) - with pytest.raises(ValueError, match="model_version"): - PeriodAlignmentDeclaration(model_id="m", model_version=" ") - with pytest.raises(ValueError, match="non-scalar"): - PeriodAlignmentDeclaration( - model_id="m", - model_version="1", - parameters={"factors": [1.02, 1.03]}, - ) - - -def test_unknown_selector_keys_fail_loudly(): - profile = _profile(selector={"source_name": "irs_soi", "spreadsheet": "t.xls"}) - with pytest.raises(ValueError, match="unknown keys"): - resolve_profile_targets( - profile, - _rows(), - {"type": "tax_year", "value": 2022}, - ) - - -def test_no_matching_facts_fails_strict_and_reports_lenient(): - profile = _profile(selector={"source_name": "nonexistent"}) - with pytest.raises(ValueError, match="matched no consumer fact rows"): - resolve_profile_targets( - profile, - _rows(), - {"type": "tax_year", "value": 2022}, - ) - report = resolve_profile_targets( - profile, - _rows(), - {"type": "tax_year", "value": 2022}, - strict=False, - ) - assert not report.valid - assert report.issues[0].code == "no_matching_facts" - - -def _write_artifact_inputs(tmp_path): +def _write_facts(tmp_path): facts_path = tmp_path / "consumer_facts.jsonl" with facts_path.open("w") as file: for row in _rows(): file.write(json.dumps(row, sort_keys=True) + "\n") - profile_path = tmp_path / "test_profile.json" - profile_path.write_text( - json.dumps( - { - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "test_profile", - "country": "us", - "label": "Test profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": "soi.agi.total", - "family": "irs_soi", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "irs_soi", - "source_measure_id": "agi", - "provenance_class": "administrative", - }, - "measurement": {"entity": "tax_unit", "concept": "us.agi"}, - "bindings": { - "microcosm": {"metric_name": "irs_soi/agi/total"}, - }, - } - ], - } - ) - ) - return facts_path, profile_path + return facts_path -def test_artifact_build_load_resolve_round_trip(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) +def _rewrite_manifest_hash(out_dir): + facts_file = out_dir / "consumer_facts.jsonl" + manifest_path = out_dir / "manifest.json" + manifest = json.loads(manifest_path.read_text()) + manifest["facts_sha256"] = hashlib.sha256(facts_file.read_bytes()).hexdigest() + manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") + + +def test_artifact_build_load_round_trip_is_facts_only(tmp_path): + facts_path = _write_facts(tmp_path) out_dir = tmp_path / "artifact" - report = build_consumer_artifact( - out_dir, - facts_path=facts_path, - profile_paths=[profile_path], - ) - assert report.fact_row_count == 3 - assert report.profile_ids == ("test_profile",) + report = build_consumer_artifact(out_dir, facts_path=facts_path) artifact = load_consumer_artifact(out_dir) - assert artifact.manifest["fact_row_count"] == 3 - resolution = artifact.resolve( - "test_profile", - {"type": "tax_year", "value": 2022}, - ) - assert resolution.resolved[0].value == 110 - assert resolution.resolved[0].provenance_class == "administrative" - assert resolution.resolved[0].survey_instrument is None - assert resolution.resolved[0].to_dict()["provenance_class"] == "administrative" - assert "survey_instrument" not in resolution.resolved[0].to_dict() + manifest = json.loads((out_dir / "manifest.json").read_text()) + + assert report.to_dict() == { + "schema_version": "policyengine_ledger.consumer_artifact.v1", + "output_dir": str(out_dir), + "fact_row_count": 2, + } + assert manifest == { + "schema_version": "policyengine_ledger.consumer_artifact.v1", + "consumer_fact_schema_versions": ["ledger.consumer_fact.v1"], + "consumer_fact_schema_sha256": CONSUMER_FACT_SCHEMA_SHA256, + "fact_row_count": 2, + "facts_sha256": hashlib.sha256( + (out_dir / "consumer_facts.jsonl").read_bytes() + ).hexdigest(), + } + assert {path.name for path in out_dir.iterdir()} == { + "consumer_facts.jsonl", + "manifest.json", + } + assert artifact.path == out_dir + assert len(artifact.rows) == 2 - with pytest.raises(PeriodContractError): - artifact.resolve("test_profile", {"type": "tax_year", "value": 2025}) +def test_artifact_is_reproducible(tmp_path): + facts_path = _write_facts(tmp_path) + first = tmp_path / "first" + second = tmp_path / "second" -def test_artifact_coverage_reports_periods_and_assertions(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) - out_dir = tmp_path / "artifact" - report = build_consumer_artifact( - out_dir, - facts_path=facts_path, - profile_paths=[profile_path], - ) - target_coverage = report.coverage["test_profile"]["soi.agi.total"]["country"] - assert target_coverage["matched_row_count"] == 2 - assert target_coverage["fact_periods"] == ["tax_year:2021", "tax_year:2022"] - assert target_coverage["assertions"] == ["observation"] - coverage_on_disk = json.loads((out_dir / "coverage.json").read_text()) - assert coverage_on_disk == report.coverage + build_consumer_artifact(first, facts_path=facts_path) + build_consumer_artifact(second, facts_path=facts_path) + + for name in ("manifest.json", "consumer_facts.jsonl"): + assert (first / name).read_bytes() == (second / name).read_bytes() def test_artifact_load_rejects_tampered_facts(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) + facts_path = _write_facts(tmp_path) out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, - facts_path=facts_path, - profile_paths=[profile_path], - ) + build_consumer_artifact(out_dir, facts_path=facts_path) + facts_file = out_dir / "consumer_facts.jsonl" rows = facts_file.read_text().splitlines() tampered = json.loads(rows[0]) tampered["value"] = 999 rows[0] = json.dumps(tampered, sort_keys=True) facts_file.write_text("\n".join(rows) + "\n") + with pytest.raises(ValueError, match="manifest hash"): load_consumer_artifact(out_dir) -def _apply_provenance_case(row, case): - if case == "missing": - row.pop("provenance_class") - elif case == "unknown": - row["provenance_class"] = "unknown" - elif case == "wrong_type": - row["provenance_class"] = 1 - elif case == "survey_missing_instrument": - row["provenance_class"] = "survey_aggregate" - elif case == "survey_blank_instrument": - row["provenance_class"] = "survey_aggregate" - row["survey_instrument"] = " " - elif case == "misplaced_instrument": - row["survey_instrument"] = "ACS 1-year" - else: # pragma: no cover - test authoring guard - raise AssertionError(case) +def test_artifact_load_rejects_profile_metadata(tmp_path): + facts_path = _write_facts(tmp_path) + out_dir = tmp_path / "artifact" + build_consumer_artifact(out_dir, facts_path=facts_path) + manifest_path = out_dir / "manifest.json" + manifest = json.loads(manifest_path.read_text()) + manifest["profiles"] = {"legacy": {"sha256": "00" * 32, "target_count": 1}} + manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") + + with pytest.raises(ValueError, match="target profiles are consumer-owned"): + load_consumer_artifact(out_dir) @pytest.mark.parametrize( @@ -413,871 +167,102 @@ def _apply_provenance_case(row, case): ("unknown", "provenance_class"), ("wrong_type", "provenance_class"), ("survey_missing_instrument", "survey_instrument"), - ("survey_blank_instrument", "survey_instrument"), ("misplaced_instrument", "survey_instrument"), ], ) def test_artifact_build_rejects_malformed_provenance(tmp_path, case, message): - facts_path, profile_path = _write_artifact_inputs(tmp_path) + facts_path = _write_facts(tmp_path) rows = facts_path.read_text().splitlines() first = json.loads(rows[0]) - _apply_provenance_case(first, case) + if case == "missing": + first.pop("provenance_class") + elif case == "unknown": + first["provenance_class"] = "unknown" + elif case == "wrong_type": + first["provenance_class"] = 1 + elif case == "survey_missing_instrument": + first["provenance_class"] = "survey_aggregate" + elif case == "misplaced_instrument": + first["survey_instrument"] = "ACS 1-year" rows[0] = json.dumps(first, sort_keys=True) facts_path.write_text("\n".join(rows) + "\n") with pytest.raises(ValueError, match=message): - build_consumer_artifact( - tmp_path / "artifact", - facts_path=facts_path, - profile_paths=[profile_path], - ) + build_consumer_artifact(tmp_path / "artifact", facts_path=facts_path) -@pytest.mark.parametrize( - ("case", "message"), - [ - ("missing", "provenance_class"), - ("unknown", "provenance_class"), - ("misplaced_instrument", "survey_instrument"), - ], -) -def test_artifact_load_rejects_malformed_provenance(tmp_path, case, message): - facts_path, profile_path = _write_artifact_inputs(tmp_path) +def test_artifact_load_rejects_row_missing_required_field(tmp_path): + facts_path = _write_facts(tmp_path) out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, - facts_path=facts_path, - profile_paths=[profile_path], - ) + build_consumer_artifact(out_dir, facts_path=facts_path) facts_file = out_dir / "consumer_facts.jsonl" rows = facts_file.read_text().splitlines() first = json.loads(rows[0]) - _apply_provenance_case(first, case) + first.pop("entity") rows[0] = json.dumps(first, sort_keys=True) facts_file.write_text("\n".join(rows) + "\n") - manifest_path = out_dir / "manifest.json" - manifest = json.loads(manifest_path.read_text()) - manifest["facts_sha256"] = hashlib.sha256(facts_file.read_bytes()).hexdigest() - manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") + _rewrite_manifest_hash(out_dir) - with pytest.raises(ValueError, match=message): - load_consumer_artifact(out_dir) - - -def test_artifact_facts_only_round_trips(tmp_path): - facts_path, _ = _write_artifact_inputs(tmp_path) - out_dir = tmp_path / "artifact" - - report = build_consumer_artifact( - out_dir, - facts_path=facts_path, - ) - artifact = load_consumer_artifact(out_dir) - manifest = json.loads((out_dir / "manifest.json").read_text()) - - assert report.fact_row_count == 3 - assert report.profile_ids == () - assert report.coverage == {} - assert manifest["profiles"] == {} - assert (out_dir / "profiles").is_dir() - assert list((out_dir / "profiles").iterdir()) == [] - assert len(artifact.rows) == 3 - assert artifact.profiles == {} - - -def test_artifact_is_reproducible(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) - first = tmp_path / "first" - second = tmp_path / "second" - for out_dir in (first, second): - build_consumer_artifact( - out_dir, - facts_path=facts_path, - profile_paths=[profile_path], - ) - for name in ("manifest.json", "coverage.json", "consumer_facts.jsonl"): - assert (first / name).read_bytes() == (second / name).read_bytes() - - -def _ons_firm_crosstab_fact(*, record_set_id, dimensions, value, cell): - constraints = tuple( - AggregateConstraint(variable=key, operator="==", value=band) - for key, band in sorted(dimensions.items()) - ) - return AggregateFact( - value=value, - provenance_class="administrative", - period=PeriodDimension(type="calendar_year", value=2025), - geography=GeographyDimension(level="country", id="K02000001", vintage="2025"), - entity=EntityDimension(name="firm"), - measure=Measure(concept="uk.firm.count", unit="count"), - aggregation=Aggregation(method="sum"), - source=SourceProvenance( - source_name="ons", - source_table="UK Business Counts 2025", - source_file="ukbusinesscounts2025.xlsx", - url="https://www.ons.gov.uk/ukbusinesscounts2025.xlsx", - vintage="cy2025", - extracted_at="2026-06-01", - extraction_method="test", - source_sha256=SHA, - source_size_bytes=100, - raw_r2_bucket="ledger-raw", - raw_r2_key=f"raw/ons/ukbc/{cell}/{SHA}/x.xlsx", - raw_r2_uri=f"r2://ledger-raw/raw/ons/ukbc/{cell}/{SHA}/x.xlsx", - ), - domain="uk_enterprises", - source_record_id=f"ons.uk_business.cy2025.{cell}", - source_cell_keys=(f"ledger.source_cell.v1:{cell}",), - filters=dict(dimensions), - constraints=constraints, - layout=SourceRecordLayout( - record_set_id=record_set_id, - record_set_spec_id="ons.uk_business.v1", - measure_id="enterprise_count", - ), - ) - - -def _uk_firms_crosstab_rows(): - by_sic_turnover = "ons.uk_business.cy2025.enterprise_count.by_sic_turnover_band" - by_sic_employment = "ons.uk_business.cy2025.enterprise_count.by_sic_employment_band" - return consumer_fact_rows( - [ - _ons_firm_crosstab_fact( - record_set_id=by_sic_turnover, - dimensions={ - "uk.firm.sic_code": "A", - "uk.firm.turnover_band": "0_99k", - }, - value=1200, - cell="sic_turnover.A.0_99k", - ), - _ons_firm_crosstab_fact( - record_set_id=by_sic_turnover, - dimensions={ - "uk.firm.sic_code": "C", - "uk.firm.turnover_band": "100_249k", - }, - value=800, - cell="sic_turnover.C.100_249k", - ), - _ons_firm_crosstab_fact( - record_set_id=by_sic_employment, - dimensions={ - "uk.firm.sic_code": "A", - "uk.firm.employment_band": "0_9", - }, - value=1500, - cell="sic_employment.A.0_9", - ), - _ons_firm_crosstab_fact( - record_set_id=by_sic_employment, - dimensions={ - "uk.firm.sic_code": "C", - "uk.firm.employment_band": "10_49", - }, - value=430, - cell="sic_employment.C.10_49", - ), - ] - ) - - -def _uk_firms_crosstab_profile_payload(): - payload = json.loads( - (Path(target_profiles_pkg.__file__).parent / "uk_firms.json").read_text() - ) - crosstab_ids = { - "ons.uk_business.enterprise_count.sic_turnover_bands", - "ons.uk_business.enterprise_count.sic_employment_bands", - } - payload["targets"] = [ - target for target in payload["targets"] if target["target_id"] in crosstab_ids - ] - return payload - - -def test_uk_firms_cross_tab_targets_resolve_through_dimensions_selector(tmp_path): - facts_path = tmp_path / "consumer_facts.jsonl" - with facts_path.open("w") as file: - for row in _uk_firms_crosstab_rows(): - file.write(json.dumps(row, sort_keys=True) + "\n") - profile_path = tmp_path / "uk_firms.json" - profile_path.write_text(json.dumps(_uk_firms_crosstab_profile_payload())) - - out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, - facts_path=facts_path, - profile_paths=[profile_path], - ) - artifact = load_consumer_artifact(out_dir) - - report = artifact.resolve("uk_firms", {"type": "calendar_year", "value": 2025}) - - assert report.valid - assert artifact.profile_hash_semantics == {"uk_firms": "exact"} - by_target: dict[str, list] = {} - for row in report.resolved: - assert row.basis == "fact" - by_target.setdefault(row.target_id, []).append(row) - - sic_turnover = by_target["ons.uk_business.enterprise_count.sic_turnover_bands"] - sic_employment = by_target["ons.uk_business.enterprise_count.sic_employment_bands"] - assert sorted(row.value for row in sic_turnover) == [800, 1200] - assert sorted(row.value for row in sic_employment) == [430, 1500] - assert all( - sorted(row.dimensions) == ["uk.firm.sic_code", "uk.firm.turnover_band"] - for row in sic_turnover - ) - assert all( - sorted(row.dimensions) == ["uk.firm.employment_band", "uk.firm.sic_code"] - for row in sic_employment - ) - - -def _uk_firms_dimension_value_profile(dimensions_selector): - profile = target_profile_from_mapping( - { - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "uk_firms_dimension_value_smoke", - "country": "uk", - "label": "UK firms dimension value smoke profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": "ons.uk_business.enterprise_count.sic_a_turnover_0_99k", - "family": "uk_firms", - "geography_levels": ["country"], - "chronicle_selector": { - "record_set_id": ( - "ons.uk_business.cy2025.enterprise_count." - "by_sic_turnover_band" - ), - "dimensions": dimensions_selector, - }, - "measurement": {"unit": "count"}, - "bindings": { - "microcosm": { - "metric_name": "ons.uk_business.enterprise_count" - }, - }, - }, - ], - } - ) - return profile - - -def test_dimensions_selector_mapping_matches_dimension_values(): - profile = _uk_firms_dimension_value_profile( - { - "uk.firm.sic_code": "A", - "uk.firm.turnover_band": "0_99k", - } - ) - - report = resolve_profile_targets( - profile, - _uk_firms_crosstab_rows(), - {"type": "calendar_year", "value": 2025}, - ) - - assert report.valid - assert [row.value for row in report.resolved] == [1200] - - -def test_dimensions_selector_mapping_requires_exact_dimension_values(): - profile = _uk_firms_dimension_value_profile({"uk.firm.sic_code": "A"}) - - with pytest.raises(ValueError, match="matched no consumer fact rows"): - resolve_profile_targets( - profile, - _uk_firms_crosstab_rows(), - {"type": "calendar_year", "value": 2025}, - ) - - report = resolve_profile_targets( - profile, - _uk_firms_crosstab_rows(), - {"type": "calendar_year", "value": 2025}, - strict=False, - ) - - assert not report.valid - assert [issue.code for issue in report.issues] == ["no_matching_facts"] - - -def test_dimensions_selector_mapping_rejects_empty_mapping(): - profile = _uk_firms_dimension_value_profile({}) - - with pytest.raises(ValueError, match="empty 'dimensions' mapping"): - resolve_profile_targets( - profile, - _uk_firms_crosstab_rows(), - {"type": "calendar_year", "value": 2025}, - ) - - report = resolve_profile_targets( - profile, - _uk_firms_crosstab_rows(), - {"type": "calendar_year", "value": 2025}, - strict=False, - ) - - assert not report.valid - assert [issue.code for issue in report.issues] == [ - "empty_dimensions_selector", - "no_matching_facts", - ] - - -def test_dimensions_selector_mapping_rejects_unknown_dimension_names(): - profile = _uk_firms_dimension_value_profile( - { - "uk.firm.sic_cod": "A", - "uk.firm.turnover_band": "0_99k", - } - ) - - with pytest.raises(ValueError, match="unknown dimensions"): - resolve_profile_targets( - profile, - _uk_firms_crosstab_rows(), - {"type": "calendar_year", "value": 2025}, - ) - - report = resolve_profile_targets( - profile, - _uk_firms_crosstab_rows(), - {"type": "calendar_year", "value": 2025}, - strict=False, - ) - - assert not report.valid - assert [issue.code for issue in report.issues] == [ - "unknown_dimension_selector", - "no_matching_facts", - ] - - -def _rewrite_facts_file(out_dir, rows): - facts_file = out_dir / "consumer_facts.jsonl" - facts_file.write_text( - "".join(json.dumps(row, sort_keys=True) + "\n" for row in rows) - ) - manifest_path = out_dir / "manifest.json" - manifest = json.loads(manifest_path.read_text()) - manifest["facts_sha256"] = hashlib.sha256(facts_file.read_bytes()).hexdigest() - manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") - - -def test_artifact_load_rejects_row_missing_required_field(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) - out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, facts_path=facts_path, profile_paths=[profile_path] - ) - - rows = _rows() - del rows[0]["observed_measure"]["unit"] - _rewrite_facts_file(out_dir, rows) - - with pytest.raises(ValueError, match="unit"): - load_consumer_artifact(out_dir) - - -def test_artifact_load_rejects_unknown_extra_field(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) - out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, facts_path=facts_path, profile_paths=[profile_path] - ) - - rows = _rows() - rows[0]["unexpected_field"] = "surprise" - _rewrite_facts_file(out_dir, rows) - - with pytest.raises(ValueError, match="unexpected_field"): + with pytest.raises(ValueError, match="schema validation"): load_consumer_artifact(out_dir) def test_artifact_load_rejects_unknown_schema_sha256(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) - out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, facts_path=facts_path, profile_paths=[profile_path] - ) - - manifest_path = out_dir / "manifest.json" - manifest = json.loads(manifest_path.read_text()) - manifest["consumer_fact_schema_sha256"] = "0" * 64 - manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") - - with pytest.raises(ValueError, match="consumer_fact_schema_sha256"): - load_consumer_artifact(out_dir) - - -def test_artifact_load_rejects_tampered_profile(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) - out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, facts_path=facts_path, profile_paths=[profile_path] - ) - - profile_file = out_dir / "profiles" / "test_profile.json" - tampered = profile_file.read_bytes().replace(b"Test profile", b"Xest profile", 1) - assert tampered != profile_file.read_bytes() - profile_file.write_bytes(tampered) - - with pytest.raises(ValueError, match="does not match the manifest"): - load_consumer_artifact(out_dir) - - -def test_artifact_load_accepts_legacy_profile_hash_only_via_explicit_path(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) + facts_path = _write_facts(tmp_path) out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, facts_path=facts_path, profile_paths=[profile_path] - ) - - profile_file = out_dir / "profiles" / "test_profile.json" - profile_bytes = profile_file.read_bytes() - assert profile_bytes.endswith(b"\n") - legacy_hash = hashlib.sha256(profile_bytes[:-1]).hexdigest() - + build_consumer_artifact(out_dir, facts_path=facts_path) manifest_path = out_dir / "manifest.json" manifest = json.loads(manifest_path.read_text()) - # A pre-fix manifest carries no schema sha and hashed the profile without - # its trailing newline. - del manifest["consumer_fact_schema_sha256"] - manifest["profiles"]["test_profile"]["sha256"] = legacy_hash + manifest["consumer_fact_schema_sha256"] = "00" * 32 manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") - artifact = load_consumer_artifact(out_dir) - assert artifact.profile_hash_semantics == {"test_profile": "legacy_profile_hash"} - - # The same legacy hash is rejected once the manifest is post-fix. - manifest["consumer_fact_schema_sha256"] = CONSUMER_FACT_SCHEMA_SHA256 - manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") - with pytest.raises(ValueError, match="does not match the manifest"): + with pytest.raises(ValueError, match="does not match the packaged"): load_consumer_artifact(out_dir) def test_artifact_build_rejects_duplicate_aggregate_fact_key(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) - with facts_path.open("a") as file: - file.write(json.dumps(_rows()[0], sort_keys=True) + "\n") + facts_path = _write_facts(tmp_path) + first = facts_path.read_text().splitlines()[0] + facts_path.write_text(first + "\n" + first + "\n") - with pytest.raises(ValueError, match="aggregate_fact_key"): - build_consumer_artifact( - tmp_path / "artifact", - facts_path=facts_path, - profile_paths=[profile_path], - ) + with pytest.raises(ValueError, match="must be unique"): + build_consumer_artifact(tmp_path / "artifact", facts_path=facts_path) -def _build_and_load_with_row_mutation(tmp_path, mutate): - facts_path, profile_path = _write_artifact_inputs(tmp_path) +def test_load_rejects_a_forged_identity_key(tmp_path): + facts_path = _write_facts(tmp_path) out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, facts_path=facts_path, profile_paths=[profile_path] - ) + build_consumer_artifact(out_dir, facts_path=facts_path) facts_file = out_dir / "consumer_facts.jsonl" - lines = [ln for ln in facts_file.read_text().splitlines() if ln.strip()] - rows = [json.loads(ln) for ln in lines] - mutate(rows) - facts_file.write_text("".join(json.dumps(r, sort_keys=True) + "\n" for r in rows)) - # Re-point the manifest facts hash so the row checks (not the file hash) fire. - import hashlib as _hashlib - - manifest_path = out_dir / "manifest.json" - manifest = json.loads(manifest_path.read_text()) - manifest["facts_sha256"] = _hashlib.sha256(facts_file.read_bytes()).hexdigest() - manifest_path.write_text(json.dumps(manifest)) - return out_dir - - -def test_load_rejects_a_forged_all_zero_identity_key(tmp_path): - # Sol finding 7: schema validates key SYNTAX; a syntactically valid but - # forged identity key that does not hash the row's content must be rejected. - def zero_keys(rows): - zero = "0" * 24 - rows[0]["aggregate_fact_key"] = f"ledger.aggregate_fact.v2:{zero}" - rows[0]["source_release_key"] = f"ledger.source_release.v2:{zero}" - rows[0]["source_series_key"] = f"ledger.source_series.v2:{zero}" - rows[0]["observed_measure_key"] = f"ledger.observed_measure.v2:{zero}" - rows[0]["dimension_set_key"] = f"ledger.dimension_set.v2:{zero}" - rows[0]["universe_constraint_set_key"] = ( - f"ledger.universe_constraint_set.v2:{zero}" - ) - - out_dir = _build_and_load_with_row_mutation(tmp_path, zero_keys) - with pytest.raises(ValueError, match="does not match the row"): - load_consumer_artifact(out_dir) - - -def test_load_rejects_a_non_finite_number(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) - out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, facts_path=facts_path, profile_paths=[profile_path] - ) - facts_file = out_dir / "consumer_facts.jsonl" - lines = [ln for ln in facts_file.read_text().splitlines() if ln.strip()] - # Inject a raw NaN token that json.loads would otherwise accept. - lines[0] = lines[0].replace('"value":110', '"value":NaN', 1) - if "NaN" not in lines[0]: - lines[0] = lines[0][:-1] + ',"rogue":NaN}' - facts_file.write_text("\n".join(lines) + "\n") - import hashlib as _hashlib + rows = facts_file.read_text().splitlines() + first = json.loads(rows[0]) + first["aggregate_fact_key"] = "ledger.aggregate_fact.v2:" + "0" * 24 + rows[0] = json.dumps(first, sort_keys=True) + facts_file.write_text("\n".join(rows) + "\n") + _rewrite_manifest_hash(out_dir) - manifest_path = out_dir / "manifest.json" - manifest = json.loads(manifest_path.read_text()) - manifest["facts_sha256"] = _hashlib.sha256(facts_file.read_bytes()).hexdigest() - manifest_path.write_text(json.dumps(manifest)) - with pytest.raises(ValueError, match="non-finite"): + with pytest.raises(ValueError, match="identity key does not match the row"): load_consumer_artifact(out_dir) def test_load_rejects_a_false_manifest_row_count(tmp_path): - facts_path, profile_path = _write_artifact_inputs(tmp_path) + facts_path = _write_facts(tmp_path) out_dir = tmp_path / "artifact" - build_consumer_artifact( - out_dir, facts_path=facts_path, profile_paths=[profile_path] - ) + build_consumer_artifact(out_dir, facts_path=facts_path) manifest_path = out_dir / "manifest.json" manifest = json.loads(manifest_path.read_text()) manifest["fact_row_count"] = 999 - manifest_path.write_text(json.dumps(manifest)) - with pytest.raises(ValueError, match="fact_row_count"): - load_consumer_artifact(out_dir) - - -def _mixed_assertion_rows(): - """Observed 2021/2022 plus a 2023 projection of the same soi series.""" - return consumer_fact_rows( - [ - _fact(value=100, period_value=2021), - _fact(value=110, period_value=2022), - _fact(value=120, period_value=2023, assertion="source_projection"), - ] - ) - - -def _policy_profile( - *, - defaults_policy=None, - target_policy=None, - selector=None, - target_id="soi.agi.total", -): - mapping = { - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "test_profile", - "country": "us", - "label": "Test profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": target_id, - "family": "irs_soi", - "geography_levels": ["country"], - "chronicle_selector": selector - or {"source_name": "irs_soi", "source_measure_id": "agi"}, - "measurement": {"entity": "tax_unit", "concept": "us.agi"}, - "bindings": { - "microcosm": {"metric_name": "irs_soi/agi/total"}, - }, - } - ], - } - if defaults_policy is not None: - mapping["defaults"]["assertion_policy"] = defaults_policy - if target_policy is not None: - mapping["targets"][0]["assertion_policy"] = target_policy - return target_profile_from_mapping(mapping) - - -def test_observed_only_default_never_resolves_a_projection(): - # The 2023 projection is invisible; the newest observed fact is 2022, and - # taking it at a requested 2023 base period correctly demands an explicit - # alignment instead of resolving anything silently. - report = resolve_profile_targets( - _policy_profile(), - _mixed_assertion_rows(), - {"type": "tax_year", "value": 2023}, - strict=False, - ) - assert not report.resolved - (violation,) = report.violations - assert violation.fact_period == {"type": "tax_year", "value": 2022} - assert not [i for i in report.issues if i.code == "resolved_from_projection"] - - -def test_observed_only_fails_loudly_on_projection_only_families(): - report = resolve_profile_targets( - _policy_profile( - selector={"source_name": "cbo", "source_measure_id": "receipts"}, - target_id="cbo.receipts", - ), - _rows(), - {"type": "calendar_year", "value": 2027}, - strict=False, - ) - assert not report.resolved - (issue,) = [i for i in report.issues if i.code == "only_projection_facts"] - assert issue.severity == "error" - assert not report.valid - - -def test_prefer_observed_takes_the_observation_over_a_newer_projection(): - # Observations win even when a projection sits exactly at the requested - # period; the observed 2022 fact is chosen and the period contract then - # asks for an alignment rather than silently substituting the projection. - report = resolve_profile_targets( - _policy_profile(defaults_policy="prefer_observed"), - _mixed_assertion_rows(), - {"type": "tax_year", "value": 2023}, - strict=False, - ) - assert not report.resolved - (violation,) = report.violations - assert violation.fact_period == {"type": "tax_year", "value": 2022} - - -def test_prefer_observed_falls_back_to_projections_with_a_warning(): - report = resolve_profile_targets( - _policy_profile( - defaults_policy="prefer_observed", - selector={"source_name": "cbo", "source_measure_id": "receipts"}, - target_id="cbo.receipts", - ), - _rows(), - {"type": "calendar_year", "value": 2027}, - ) - assert report.valid - (row,) = report.resolved - assert row.assertion == "source_projection" - assert row.value == 250 - (issue,) = [i for i in report.issues if i.code == "resolved_from_projection"] - assert issue.severity == "warning" - - -def test_allow_source_projection_resolves_the_latest_projection(): - report = resolve_profile_targets( - _policy_profile(target_policy="allow_source_projection"), - _mixed_assertion_rows(), - {"type": "tax_year", "value": 2023}, - ) - assert report.valid - (row,) = report.resolved - assert row.assertion == "source_projection" - assert row.value == 120 - assert [i for i in report.issues if i.code == "resolved_from_projection"] - - -def _tied_assertion_rows(): - """An observation and a projection colliding at the same 2023 period.""" - return consumer_fact_rows( - [ - _fact(value=110, period_value=2022), - _fact(value=130, period_value=2023), - _fact(value=120, period_value=2023, assertion="source_projection"), - ] - ) - - -def test_allow_source_projection_resolves_the_observation_on_a_period_tie(): - # Emitting both rows would double-count the series; the realized value - # wins the tie and the overlap is flagged instead of passing silently. - report = resolve_profile_targets( - _policy_profile(target_policy="allow_source_projection"), - _tied_assertion_rows(), - {"type": "tax_year", "value": 2023}, - ) - assert report.valid - (row,) = report.resolved - assert row.assertion == "observation" - assert row.value == 130 - (issue,) = [i for i in report.issues if i.code == "ambiguous_assertion_at_period"] - assert issue.severity == "warning" - assert not [i for i in report.issues if i.code == "resolved_from_projection"] - - -def test_explicit_assertion_selector_reaches_the_projection_despite_a_tie(): - # Selecting on assertion is maximal intent: the selector filters the tie - # away before resolution, so the projection resolves without ambiguity. - report = resolve_profile_targets( - _policy_profile( - selector={ - "source_name": "irs_soi", - "source_measure_id": "agi", - "assertion": "source_projection", - } - ), - _tied_assertion_rows(), - {"type": "tax_year", "value": 2023}, - ) - assert report.valid - (row,) = report.resolved - assert row.assertion == "source_projection" - assert row.value == 120 - assert [i for i in report.issues if i.code == "resolved_from_projection"] - assert not [i for i in report.issues if i.code == "ambiguous_assertion_at_period"] - - -def test_assertion_tie_break_is_scoped_to_the_series(): - # Geography A carries a genuine tie (observation + projection at the - # chosen period); geography B has only a projection. The tie-break must - # drop A's projection, keep A's observation — and leave B's projection - # alone: B was never in a tie with anything. - rows = consumer_fact_rows( - [ - _fact(value=130, period_value=2023), - _fact(value=120, period_value=2023, assertion="source_projection"), - _fact( - value=99, - period_value=2023, - assertion="source_projection", - geography_id="0100000GB", - ), - ] - ) - report = resolve_profile_targets( - _policy_profile(target_policy="allow_source_projection"), - rows, - {"type": "tax_year", "value": 2023}, - ) - assert report.valid - resolved = {row.geography["id"]: row for row in report.resolved} - assert set(resolved) == {"0100000US", "0100000GB"} - assert resolved["0100000US"].assertion == "observation" - assert resolved["0100000US"].value == 130 - assert resolved["0100000GB"].assertion == "source_projection" - assert resolved["0100000GB"].value == 99 - (ambiguous,) = [ - i for i in report.issues if i.code == "ambiguous_assertion_at_period" - ] - assert "0100000US" in ambiguous.message - assert "0100000GB" not in ambiguous.message - assert [i for i in report.issues if i.code == "resolved_from_projection"] - - -def test_prefer_observed_lets_a_projection_only_series_resolve(): - # The Northern-Ireland shape: one geography observed, another carrying - # only a projection at the same period. Per-series preference resolves - # both instead of starving the projection-only series. - rows = consumer_fact_rows( - [ - _fact(value=130, period_value=2023), - _fact( - value=99, - period_value=2023, - assertion="source_projection", - geography_id="0100000GB", - ), - ] - ) - report = resolve_profile_targets( - _policy_profile(defaults_policy="prefer_observed"), - rows, - {"type": "tax_year", "value": 2023}, - ) - assert report.valid - resolved = {row.geography["id"]: row for row in report.resolved} - assert resolved["0100000US"].assertion == "observation" - assert resolved["0100000GB"].assertion == "source_projection" - assert [i for i in report.issues if i.code == "resolved_from_projection"] - - -def test_prefer_observed_still_never_mixes_bases_within_one_series(): - # Within a single series the family rule survives the per-series change: - # an observed 2022 fact still beats a projection sitting at the - # requested 2023 period, ending in a period-contract violation rather - # than a silent base mix. - report = resolve_profile_targets( - _policy_profile(defaults_policy="prefer_observed"), - _mixed_assertion_rows(), - {"type": "tax_year", "value": 2023}, - strict=False, - ) - assert not report.resolved - (violation,) = report.violations - assert violation.fact_period == {"type": "tax_year", "value": 2022} - - -def test_target_assertion_policy_overrides_the_profile_default(): - report = resolve_profile_targets( - _policy_profile( - defaults_policy="allow_source_projection", - target_policy="observed_only", - ), - _mixed_assertion_rows(), - {"type": "tax_year", "value": 2022}, - ) - assert report.valid - (row,) = report.resolved - assert row.assertion == "observation" - assert row.value == 110 - assert not [i for i in report.issues if i.code == "resolved_from_projection"] - - -def test_invalid_assertion_policy_values_are_rejected(): - with pytest.raises(ValueError, match="assertion_policy"): - _policy_profile(defaults_policy="projections_welcome") - with pytest.raises(ValueError, match="assertion_policy"): - _policy_profile(target_policy="observed") - - -def test_unknown_assertion_values_fail_resolution_loudly(): - # A typo'd assertion must not quietly vanish from every policy's - # candidate set; the resolver polices the enum like the file loader does. - rows = _mixed_assertion_rows() - rows[-1] = dict(rows[-1], assertion="projection") - with pytest.raises(ValueError, match="unsupported assertion"): - resolve_profile_targets( - _policy_profile(), - rows, - {"type": "tax_year", "value": 2022}, - ) + manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") + with pytest.raises(ValueError, match="declares fact_row_count"): + load_consumer_artifact(out_dir) -def test_missing_assertion_defaults_to_observation_at_resolve(): - # Back-compat parity with the file loader: rows that predate the axis - # resolve as observations. - rows = _mixed_assertion_rows() - legacy = dict(rows[1]) - del legacy["assertion"] - rows[1] = legacy - report = resolve_profile_targets( - _policy_profile(), - rows, - {"type": "tax_year", "value": 2022}, - ) - assert report.valid - (row,) = report.resolved - assert row.assertion == "observation" - assert row.value == 110 +def test_build_consumer_artifact_help_has_no_profile_options(capsys): + with pytest.raises(SystemExit) as excinfo: + main(["build-consumer-artifact", "--help"]) -def test_assertion_selector_and_target_policy_together_are_rejected(): - # The selector already pins the axis; a per-target policy alongside it is - # a contradiction the author should hear about at load, not have - # silently resolved. - with pytest.raises(ValueError, match="declares both assertion_policy"): - _policy_profile( - target_policy="observed_only", - selector={ - "source_name": "irs_soi", - "source_measure_id": "agi", - "assertion": "source_projection", - }, - ) + assert excinfo.value.code == 0 + help_text = capsys.readouterr().out + assert "--profile" not in help_text + assert "facts-only" in help_text diff --git a/tests/test_chronicle_consumer_contract.py b/tests/test_chronicle_consumer_contract.py index 31c717d7..8dd13e46 100644 --- a/tests/test_chronicle_consumer_contract.py +++ b/tests/test_chronicle_consumer_contract.py @@ -34,12 +34,9 @@ from chronicle.jurisdictions.us.soi import build_soi_table_1_1_facts from chronicle.store import save_facts_jsonl from policyengine_chronicle.consumer import ( - PeriodContractError, build_consumer_artifact, load_consumer_artifact, - resolve_profile_targets, ) -from policyengine_chronicle.target_profiles import target_profile_from_mapping CONSUMER_FACT_SCHEMA_PATH = ( Path(__file__).parents[1] / "docs" / "schemas" / "consumer_fact.v1.schema.json" @@ -507,67 +504,15 @@ def test_academic_year_rows_round_trip_through_consumer_artifact(tmp_path): {"type": "academic_year", "value": 2024}, ] - profile_mapping = { - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "academic_year_round_trip", - "country": "uk", - "label": "Academic-year period round trip", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": "slc.maintenance_loan_recipients", - "family": "slc", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": rows[0]["source"]["source_name"], - "source_measure_id": rows[0]["observed_measure"][ - "source_measure_id" - ], - }, - "measurement": { - "entity": "person", - "concept": "uk.education.maintenance_loan_recipients", - }, - "bindings": { - "policyengine": { - "metric_name": "slc/maintenance_loan/recipients", - } - }, - } - ], - } - target_profile_from_mapping(profile_mapping) - profile_path = tmp_path / "academic_year_round_trip.json" - profile_path.write_text(json.dumps(profile_mapping, indent=2) + "\n") - artifact_dir = tmp_path / "artifact" - build_consumer_artifact( - artifact_dir, - facts_path=facts_path, - profile_paths=[profile_path], - ) + build_consumer_artifact(artifact_dir, facts_path=facts_path) artifact = load_consumer_artifact(artifact_dir) - report = resolve_profile_targets( - artifact.profiles["academic_year_round_trip"], - artifact.rows, + assert [row["period"] for row in artifact.rows] == [ + {"type": "academic_year", "value": 2023}, {"type": "academic_year", "value": 2024}, - ) - assert report.valid - (resolved,) = report.resolved - assert resolved.basis == "fact" - assert resolved.value == 100.0 - assert resolved.fact_period == {"type": "academic_year", "value": 2024} - - with pytest.raises(PeriodContractError): - resolve_profile_targets( - artifact.profiles["academic_year_round_trip"], - artifact.rows, - {"type": "fiscal_year", "value": 2024}, - ) + ] + assert [row["value"] for row in artifact.rows] == [90.0, 100.0] def test_consumer_fact_row_marks_decimal_values_as_decimal_strings(): diff --git a/tests/test_chronicle_governance.py b/tests/test_chronicle_governance.py index 3f2f6b88..f914237c 100644 --- a/tests/test_chronicle_governance.py +++ b/tests/test_chronicle_governance.py @@ -27,17 +27,17 @@ def test_chronicle_governance_files_define_required_review_surface(): role_ids = {agent["id"] for agent in agents["approved_agents"]} assert { "ledger-source-ingestor", - "ledger-target-profile-author", "ledger-contract-maintainer", } <= role_ids + assert "ledger-target-profile-author" not in role_ids required_judges = set(agents["required_judges"]) assert { "ledger-source-fidelity", - "ledger-target-profile", "ledger-contract", "ledger-boundary", } <= required_judges + assert "ledger-target-profile" not in required_judges for agent in agents["approved_agents"]: assert set(agent["required_judges"]) <= required_judges assert agent["allowed_paths"] diff --git a/tests/test_chronicle_source_package.py b/tests/test_chronicle_source_package.py index bdd9cd0d..7e095e50 100644 --- a/tests/test_chronicle_source_package.py +++ b/tests/test_chronicle_source_package.py @@ -31,8 +31,6 @@ from chronicle.sources.cells import build_source_cell_key, validate_source_cells from chronicle.sources.rows import validate_source_rows from chronicle.suite import build_source_suite -from policyengine_chronicle.consumer import resolve_profile_targets -from policyengine_chronicle.target_profiles import target_profile_from_mapping REPO_ROOT = Path(__file__).resolve().parents[1] ALLOWED_PROVENANCE_CLASSES = { @@ -291,48 +289,6 @@ def test_hmrc_cgt_size_of_gain_package_builds_microcosm_visible_band_facts(): assert row["dimensions"] == {"cgt_gain_band": "gain_12300_to_24999"} assert row["observed_measure"]["source_concept"] == ("hmrc.cgt_gains_individuals") - profile = target_profile_from_mapping( - { - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "uk_national_cgt_size_smoke", - "country": "uk", - "label": "UK national CGT size smoke profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": "hmrc.cgt.gains.gain_12300_to_24999", - "family": "capital_gains_tax", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "hmrc", - "record_set_spec_id": ( - "hmrc.cgt_size_of_gain_2025.table2_1a." - "individuals.by_gain_band.ty2023.v1" - ), - "source_concept": "hmrc.cgt_gains_individuals", - "dimensions": {"cgt_gain_band": "gain_12300_to_24999"}, - }, - "measurement": {"unit": "gbp"}, - "bindings": { - "microcosm": {"metric_name": "hmrc.cgt.gains_by_gain_band"} - }, - } - ], - } - ) - resolution = resolve_profile_targets( - profile, - consumer_rows, - {"type": "tax_year", "value": 2023}, - ) - - assert resolution.valid - assert len(resolution.resolved) == 1 - assert resolution.resolved[0].value == 1_418_000_000 - def test_every_source_package_record_set_declares_provenance_class(): missing: list[str] = [] @@ -3864,7 +3820,6 @@ def test_ons_households_by_type_country_package_builds_scotland_microcosm_target package = load_source_package("ons-households-by-type-country-2025") cells = package.build_source_cells(2025) facts = package.build_facts(2025, cells=cells) - consumer_rows = consumer_fact_rows(facts) assert validate_source_cells(cells).valid assert validate_facts(facts).valid @@ -3885,61 +3840,6 @@ def test_ons_households_by_type_country_package_builds_scotland_microcosm_target ) assert fact.filters == {"household_type": "couple_3_plus_children_households"} - profile = target_profile_from_mapping( - { - "schema_version": "policyengine_ledger.target_profile.v1", - "profile_id": "uk_national_scotland_households_smoke", - "country": "uk", - "label": "UK national Scotland household-composition smoke profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": ( - "ons.scotland.households.couple_3_plus_children_households" - ), - "family": "household_composition", - "geography_levels": ["country"], - "chronicle_selector": { - "source_name": "ons", - "record_set_spec_id": ( - "ons.households_by_type_country_2025." - "scotland.table7.cy2025.v1" - ), - "source_concept": ( - "ons.households_by_type_regions_countries_table7" - ), - "dimensions": { - "household_type": ("couple_3_plus_children_households") - }, - }, - "measurement": {"unit": "count"}, - "bindings": { - "microcosm": { - "metric_name": ( - "ons.scotland.households.couple_3_plus_children" - ) - } - }, - } - ], - } - ) - resolution = resolve_profile_targets( - profile, - consumer_rows, - {"type": "calendar_year", "value": 2025}, - ) - - assert resolution.valid - assert len(resolution.resolved) == 1 - assert resolution.resolved[0].value == 57_000 - assert resolution.resolved[0].dimensions == { - "household_type": "couple_3_plus_children_households" - } - def test_render_string_templates_integer_year_with_filing_year(): """Integer years must still template both ``{year}`` and ``{filing_year}``.""" diff --git a/tests/test_policyengine_chronicle_imports.py b/tests/test_policyengine_chronicle_imports.py index 47a983df..f186b123 100644 --- a/tests/test_policyengine_chronicle_imports.py +++ b/tests/test_policyengine_chronicle_imports.py @@ -1,3 +1,7 @@ +import importlib + +import pytest + from policyengine_chronicle import ( AggregateFact, Aggregation, @@ -12,10 +16,11 @@ import policyengine_chronicle.normalization as chronicle_normalization from policyengine_chronicle.cli import main as chronicle_main from policyengine_chronicle.targets.us_poverty import hard_target_package_aliases -import pytest -def test__given_chronicle_import_path__then_it_reexports_chronicle_fact_schema() -> None: +def test__given_chronicle_import_path__then_it_reexports_chronicle_fact_schema() -> ( + None +): # Given fact = AggregateFact( value=1, @@ -43,7 +48,9 @@ def test__given_chronicle_import_path__then_it_reexports_chronicle_fact_schema() assert key.startswith("ledger.fact.v1:") -def test__given_chronicle_facts_import_path__then_it_reexports_chronicle_facts() -> None: +def test__given_chronicle_facts_import_path__then_it_reexports_chronicle_facts() -> ( + None +): # When from chronicle.facts import AggregateFact as ChronicleCoreAggregateFact from policyengine_chronicle.facts import AggregateFact as ChronicleAggregateFact @@ -52,7 +59,9 @@ def test__given_chronicle_facts_import_path__then_it_reexports_chronicle_facts() assert ChronicleAggregateFact is ChronicleCoreAggregateFact -def test__given_chronicle_target_import_path__then_it_reexports_target_contracts() -> None: +def test__given_chronicle_target_import_path__then_it_reexports_target_contracts() -> ( + None +): # When aliases = hard_target_package_aliases() @@ -64,12 +73,15 @@ def test__given_chronicle_target_import_path__then_it_reexports_target_contracts def test__given_public_chronicle_namespaces__then_core_helpers_are_importable() -> None: from policyengine_chronicle.normalization import convert_units from policyengine_chronicle.sources import SourceFile, query_sources - from policyengine_chronicle.target_profiles import load_target_profile assert SourceFile is not None assert query_sources is not None assert convert_units is not None - assert load_target_profile is not None + + +def test__given_retired_profile_namespace__then_it_is_not_importable() -> None: + with pytest.raises(ModuleNotFoundError): + importlib.import_module("policyengine_chronicle.target_profiles") def test__given_public_chronicle_normalization__then_target_construction_is_hidden() -> ( diff --git a/tests/test_policyengine_chronicle_target_profiles.py b/tests/test_policyengine_chronicle_target_profiles.py deleted file mode 100644 index b50d98f2..00000000 --- a/tests/test_policyengine_chronicle_target_profiles.py +++ /dev/null @@ -1,305 +0,0 @@ -from __future__ import annotations - -from pathlib import Path - -import pytest - -import policyengine_chronicle.target_profiles as target_profiles_pkg -from policyengine_chronicle.consumer import _select_rows -from policyengine_chronicle.target_profiles import ( - TARGET_PROFILE_SCHEMA_VERSION, - load_target_profile, - target_profile_from_mapping, -) - -_PROFILE_DIR = Path(target_profiles_pkg.__file__).parent - - -def _packaged_profile_ids() -> list[str]: - return sorted(path.stem for path in _PROFILE_DIR.glob("*.json")) - - -def test__given_uk_local_profile__then_it_declares_measurement_contracts() -> None: - # When - profile = load_target_profile("uk_local_geography") - - # Then - assert profile.country == "uk" - assert profile.default_operation == "sum" - assert profile.base_period_policy == "latest_not_after_build_base_period" - - constituency_metrics = [ - target.binding("policyengine").metric_name - for target in profile.targets_for_geography("constituency") - ] - assert constituency_metrics[:4] == [ - "hmrc/self_employment_income/amount", - "hmrc/self_employment_income/count", - "hmrc/employment_income/amount", - "hmrc/employment_income/count", - ] - assert "uc_hh_3plus_children" in constituency_metrics - assert "rent/private_rent" not in constituency_metrics - - local_authority_metrics = [ - target.binding("policyengine").metric_name - for target in profile.targets_for_geography("local_authority") - ] - assert "uc_households" in local_authority_metrics - assert "ons/equiv_net_income_bhc" in local_authority_metrics - assert "rent/private_rent" in local_authority_metrics - assert "uc_hh_0_children" not in local_authority_metrics - - -def test__given_count_like_profile_rows__then_they_are_still_sum_measurements() -> None: - # When - profile = load_target_profile("uk_local_geography") - employment_count = next( - target - for target in profile.targets - if target.target_id == "hmrc.employment_income.count" - ) - - # Then - assert profile.default_operation == "sum" - assert employment_count.measurement["concept"] == "uk.person.count" - assert employment_count.binding("policyengine").payload["value_variable"] == ( - "person_count" - ) - - -def test__given_uk_firms_profile__then_it_declares_chronicle_only_firm_targets() -> None: - # When - profile = load_target_profile("uk_firms") - - # Then - assert profile.country == "uk" - assert profile.default_operation == "sum" - assert profile.base_period_policy == "latest_not_after_build_base_period" - assert [ - target.target_id for target in profile.targets_for_geography("country") - ] == [ - "ons.uk_business.enterprise_count.turnover_bands", - "ons.uk_business.enterprise_count.employment_bands", - "hmrc.vat.registered_trader_count.turnover_bands", - "hmrc.vat.net_liability.turnover_bands", - "ons.uk_business.enterprise_count.sic_turnover_bands", - "ons.uk_business.enterprise_count.sic_employment_bands", - "hmrc.vat.registered_trader_count.sic_sectors", - "hmrc.vat.net_liability.sic_sectors", - ] - - targets_by_id = {target.target_id: target for target in profile.targets} - turnover_count = targets_by_id["ons.uk_business.enterprise_count.turnover_bands"] - assert turnover_count.measurement["entity"] == "firm" - assert turnover_count.chronicle_selector == { - "source_name": "ons", - "source_measure_id": "enterprise_count", - "record_set_id": "ons.uk_business.cy2025.enterprise_count.by_turnover_band", - "groupby_dimension": "uk.firm.annual_turnover", - } - assert turnover_count.binding("microcosm").metric_name == ( - "ons/uk_business/enterprise_count/turnover_bands" - ) - - registered_count = targets_by_id["hmrc.vat.registered_trader_count.turnover_bands"] - assert registered_count.binding("axiom").payload["filter_rule"] == ( - "uk:policies/govuk/vat#firm_vat_registered" - ) - - sic_turnover = targets_by_id["ons.uk_business.enterprise_count.sic_turnover_bands"] - assert sic_turnover.chronicle_selector == { - "source_name": "ons", - "source_measure_id": "enterprise_count", - "record_set_id": ( - "ons.uk_business.cy2025.enterprise_count.by_sic_turnover_band" - ), - "dimensions": ["uk.firm.sic_code", "uk.firm.turnover_band"], - } - assert sic_turnover.binding("microcosm").payload["groupby_variables"] == [ - "sic_code", - "annual_turnover", - ] - - sic_population = targets_by_id["hmrc.vat.registered_trader_count.sic_sectors"] - assert sic_population.chronicle_selector["record_set_id"] == ( - "hmrc.vat.fy2024_25.registered_trader_count.by_sic" - ) - assert sic_population.binding("axiom").payload["filter_rule"] == ( - "uk:policies/govuk/vat#firm_vat_registered" - ) - - vat_liability = targets_by_id["hmrc.vat.net_liability.turnover_bands"] - assert vat_liability.measurement["concept"] == "uk.tax.vat.net_liability" - assert vat_liability.binding("axiom").payload["value_rule"] == ( - "uk:policies/govuk/vat#net_vat_liability" - ) - assert vat_liability.binding("axiom").payload["filter_rule"] == ( - "uk:policies/govuk/vat#firm_vat_registered" - ) - - sic_vat_liability = targets_by_id["hmrc.vat.net_liability.sic_sectors"] - assert sic_vat_liability.measurement["groupby_dimension"] == "uk.firm.sic_code" - assert sic_vat_liability.binding("axiom").payload["value_rule"] == ( - "uk:policies/govuk/vat#net_vat_liability" - ) - - -@pytest.mark.parametrize("forbidden", ["registry", "aggregation", "target_value"]) -def test__given_forbidden_profile_option__then_profile_is_rejected( - forbidden: str, -) -> None: - # Given - payload = { - "schema_version": TARGET_PROFILE_SCHEMA_VERSION, - "profile_id": "bad", - "country": "uk", - "label": "Bad profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": "bad.target", - "family": "bad", - "geography_levels": ["country"], - "chronicle_selector": {"source_name": "bad"}, - "measurement": {"entity": "household", "concept": "bad"}, - "bindings": { - "policyengine": { - "metric_name": "bad", - forbidden: "not allowed", - } - }, - } - ], - } - - # When / Then - with pytest.raises(ValueError, match=forbidden): - target_profile_from_mapping(payload) - - -@pytest.mark.parametrize( - "forbidden", - ["runtime_code", "python_code", "solver", "execute", "module", "command"], -) -def test__given_runtime_binding_option__then_profile_is_rejected( - forbidden: str, -) -> None: - # Given - payload = _minimal_profile_payload() - payload["targets"][0]["bindings"]["policyengine"][forbidden] = "not allowed" - - # When / Then - with pytest.raises(ValueError, match=forbidden): - target_profile_from_mapping(payload) - - -@pytest.mark.parametrize( - ("container", "forbidden"), - [ - ("chronicle_selector", "value"), - ("chronicle_selector", "target_value"), - ("measurement", "value"), - ("measurement", "aggregation"), - ("measurement", "registry"), - ], -) -def test__given_nested_forbidden_profile_option__then_profile_is_rejected( - container: str, - forbidden: str, -) -> None: - # Given - payload = _minimal_profile_payload() - payload["targets"][0][container][forbidden] = "not allowed" - - # When / Then - with pytest.raises(ValueError, match=forbidden): - target_profile_from_mapping(payload) - - -def test__given_filter_threshold_values__then_profile_is_allowed() -> None: - # Given - payload = _minimal_profile_payload() - payload["targets"][0]["measurement"]["filters"] = [ - {"concept": "uk.tax.income_tax", "operator": ">", "value": 0} - ] - - # When - profile = target_profile_from_mapping(payload) - - # Then - assert profile.targets[0].measurement["filters"][0]["value"] == 0 - - -def test__given_non_sum_default_operation__then_profile_is_rejected() -> None: - # Given - payload = { - "schema_version": TARGET_PROFILE_SCHEMA_VERSION, - "profile_id": "bad", - "country": "uk", - "label": "Bad profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "count", - }, - "targets": [], - } - - # When / Then - with pytest.raises(ValueError, match="operation 'sum'"): - target_profile_from_mapping(payload) - - -def test_every_packaged_profile_selector_uses_supported_keys() -> None: - # Given the packaged target profiles - profile_ids = _packaged_profile_ids() - assert profile_ids - - # Then every chronicle_selector resolves against the supported vocabulary - for profile_id in profile_ids: - profile = load_target_profile(profile_id) - for target in profile.targets: - for level in target.geography_levels: - _, issues = _select_rows( - profile.profile_id, - target, - [], - geography_level=level, - ) - unknown = [ - issue for issue in issues if issue.code == "unknown_selector_key" - ] - assert not unknown, ( - f"{profile_id}/{target.target_id} ships an unsupported " - f"selector: {[issue.message for issue in unknown]}" - ) - - -def _minimal_profile_payload() -> dict[str, object]: - return { - "schema_version": TARGET_PROFILE_SCHEMA_VERSION, - "profile_id": "test_profile", - "country": "uk", - "label": "Test profile", - "defaults": { - "base_period_policy": "latest_not_after_build_base_period", - "operation": "sum", - }, - "targets": [ - { - "target_id": "test.target", - "family": "test", - "geography_levels": ["country"], - "chronicle_selector": {"source_name": "test"}, - "measurement": {"entity": "household", "concept": "test"}, - "bindings": { - "policyengine": { - "metric_name": "test", - } - }, - } - ], - } From 581f9691580a5ad2d44d26ab20a564c5d673cbf4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Sat, 29 Aug 2026 20:43:07 +0200 Subject: [PATCH 2/2] Fix issues from review: version consumer artifact --- README.md | 5 +++++ docs/adr-chronicle-facts-only.md | 5 ++++- policyengine_chronicle/consumer.py | 2 +- tests/test_chronicle_consumer.py | 20 ++++++++++++++++++-- 4 files changed, 28 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 042485ee..576919cd 100644 --- a/README.md +++ b/README.md @@ -296,6 +296,11 @@ uv run chronicle build-bundle --suite uk --out /tmp/chronicle-uk --replace uv run chronicle build-consumer-artifact --facts /tmp/chronicle-uk --out /tmp/chronicle-uk-artifact --replace ``` +The command writes a `policyengine_ledger.consumer_artifact.v2` artifact containing +only `consumer_facts.jsonl` and `manifest.json`. Version 2 is incompatible with the +retired v1 profile-bearing contract: loaders reject v1 manifests so downstreams must +adopt the facts-only surface explicitly. + `--year` is inert for `--suite uk` because the UK packages are year-pinned. The US off-year bundle behavior is unchanged and out of scope here. diff --git a/docs/adr-chronicle-facts-only.md b/docs/adr-chronicle-facts-only.md index 39bc1d34..8bfa88dd 100644 --- a/docs/adr-chronicle-facts-only.md +++ b/docs/adr-chronicle-facts-only.md @@ -29,7 +29,10 @@ Instead of projection objects, Chronicle contributes two guarantees: label. - **Facts-only consumer artifacts.** Chronicle publishes schema-validated fact rows with manifest hashes. Consumers own the selection, measurement, - period-alignment, and model-binding contracts that interpret those rows. + period-alignment, and model-binding contracts that interpret those rows. The + facts-only artifact is `policyengine_ledger.consumer_artifact.v2`; the version + bump makes the removal of v1's embedded profiles and resolution surface an + explicit incompatible transition. ## Why not facts plus projections in one schema diff --git a/policyengine_chronicle/consumer.py b/policyengine_chronicle/consumer.py index 8894081f..f1d808e5 100644 --- a/policyengine_chronicle/consumer.py +++ b/policyengine_chronicle/consumer.py @@ -27,7 +27,7 @@ validate_consumer_fact_row, ) -CONSUMER_ARTIFACT_SCHEMA_VERSION = "policyengine_ledger.consumer_artifact.v1" +CONSUMER_ARTIFACT_SCHEMA_VERSION = "policyengine_ledger.consumer_artifact.v2" @dataclass(frozen=True) diff --git a/tests/test_chronicle_consumer.py b/tests/test_chronicle_consumer.py index a8c8ae3b..a7334710 100644 --- a/tests/test_chronicle_consumer.py +++ b/tests/test_chronicle_consumer.py @@ -98,12 +98,12 @@ def test_artifact_build_load_round_trip_is_facts_only(tmp_path): manifest = json.loads((out_dir / "manifest.json").read_text()) assert report.to_dict() == { - "schema_version": "policyengine_ledger.consumer_artifact.v1", + "schema_version": "policyengine_ledger.consumer_artifact.v2", "output_dir": str(out_dir), "fact_row_count": 2, } assert manifest == { - "schema_version": "policyengine_ledger.consumer_artifact.v1", + "schema_version": "policyengine_ledger.consumer_artifact.v2", "consumer_fact_schema_versions": ["ledger.consumer_fact.v1"], "consumer_fact_schema_sha256": CONSUMER_FACT_SCHEMA_SHA256, "fact_row_count": 2, @@ -160,6 +160,22 @@ def test_artifact_load_rejects_profile_metadata(tmp_path): load_consumer_artifact(out_dir) +def test_artifact_load_rejects_legacy_v1_schema(tmp_path): + facts_path = _write_facts(tmp_path) + out_dir = tmp_path / "artifact" + build_consumer_artifact(out_dir, facts_path=facts_path) + manifest_path = out_dir / "manifest.json" + manifest = json.loads(manifest_path.read_text()) + manifest["schema_version"] = "policyengine_ledger.consumer_artifact.v1" + manifest_path.write_text(json.dumps(manifest, sort_keys=True, indent=2) + "\n") + + with pytest.raises( + ValueError, + match="Unsupported consumer artifact schema_version", + ): + load_consumer_artifact(out_dir) + + @pytest.mark.parametrize( ("case", "message"), [